gitoriaLog in with ident

mpackdb

All repositories: gitoria

ReadmeCodePull requestsReleasesTicketsSettings
Commitb4db6391b4db6391initial commitcaramboleyob4db6391/src/IndexManager.js

16.4 KB

  1. import { writeFile, readFile, open, rename } from 'fs/promises';
  2. import { createWriteStream } from 'fs';
  3. import { resolve } from 'path';
  4. import { deserialize } from './mpack.js';
  5. import { PrimaryKeyType, IndexType } from './MPackDB.js';
  6. const BLOCK_SIZE = 4096; // 4KB
  7. /**
  8. * Manages indexes for the database (in-memory deltas + on-disk persistence)
  9. */
  10. export class IndexManager {
  11. _dbDir;
  12. _dbName;
  13. _indexes;
  14. _indexTypes;
  15. _primaryKeyType;
  16. _persistIntervalId = null;
  17. _indexPaths = {};
  18. _deltaIndexes = {};
  19. _tombstones = {};
  20. _persistThreshold = 1000;
  21. _totalDeltaCount = 0;
  22. /**
  23. * Creates a new IndexManager
  24. * @param {string} dbDir - Database directory path
  25. * @param {string} dbName - Database name
  26. * @param {string[]} indexes - Array of field names to index
  27. * @param {Object} indexTypes - Map of field names to IndexType
  28. * @param {number} primaryKeyType - Primary key type
  29. */
  30. constructor(dbDir, dbName, indexes, indexTypes, primaryKeyType) {
  31. this._dbDir = dbDir;
  32. this._dbName = dbName;
  33. this._indexes = indexes;
  34. this._indexTypes = indexTypes;
  35. this._primaryKeyType = primaryKeyType;
  36. }
  37. /**
  38. * Initializes the index manager (rebuilds indexes from data file)
  39. * @param {string} dataPath - Path to the data file
  40. * @returns {Promise<void>}
  41. */
  42. async init(dataPath) {
  43. for (const field of this._indexes) {
  44. this._deltaIndexes[field] = [];
  45. this._tombstones[field] = new Map();
  46. const indexPath = resolve(this._dbDir, `${this._dbName}.${field}.txt`);
  47. this._indexPaths[field] = indexPath;
  48. try {
  49. const fileHandle = await open(indexPath, 'r');
  50. const stats = await fileHandle.stat();
  51. await fileHandle.close();
  52. if (stats.size === 0) {
  53. await this._rebuildIndex(dataPath, field);
  54. }
  55. } catch (e) {
  56. if (e.code === 'ENOENT') {
  57. await this._rebuildIndex(dataPath, field);
  58. } else {
  59. throw e;
  60. }
  61. }
  62. }
  63. }
  64. /**
  65. * Rebuilds an index from the data file
  66. * @param {string} dataPath - Path to the data file
  67. * @param {string} field - Field name to index
  68. * @returns {Promise<void>}
  69. * @private
  70. */
  71. async _rebuildIndex(dataPath, field) {
  72. const tempIndexPath = `${this._indexPaths[field]}.${Date.now()}.tmp`;
  73. const writeStream = createWriteStream(tempIndexPath, { flags: 'w', encoding: 'utf-8' });
  74. const docIterator = this._getDocsAndLocationsFromDataFile(dataPath);
  75. const entries = [];
  76. for await (const { doc, loc } of docIterator) {
  77. if (doc && doc[field] !== undefined) {
  78. const key = doc[field];
  79. entries.push({ key, loc });
  80. }
  81. }
  82. // Sort entries before writing to the index file
  83. entries.sort((a, b) => this._compareKeys(a.key, b.key));
  84. for (const entry of entries) {
  85. writeStream.write(`${entry.key},${entry.loc[0]},${entry.loc[1]}\n`);
  86. }
  87. await new Promise(resolve => writeStream.end(resolve));
  88. await rename(tempIndexPath, this._indexPaths[field]);
  89. }
  90. /**
  91. * Generator that yields documents and their locations from the data file
  92. * @param {string} dataPath - Path to the data file
  93. * @yields {{doc: Object, loc: Array}} Document and location tuple
  94. * @private
  95. */
  96. async* _getDocsAndLocationsFromDataFile(dataPath) {
  97. let fileHandle;
  98. try {
  99. fileHandle = await open(dataPath, 'r');
  100. const stats = await fileHandle.stat();
  101. let offset = 0;
  102. while (offset < stats.size) {
  103. // Read size header
  104. const sizeBuffer = Buffer.alloc(4);
  105. await fileHandle.read(sizeBuffer, 0, 4, offset);
  106. const size = sizeBuffer.readInt32LE(0);
  107. if (size <= 0 || offset + size > stats.size) {
  108. break; // Corrupted or incomplete record
  109. }
  110. const docBuffer = Buffer.alloc(size);
  111. await fileHandle.read(docBuffer, 0, size, offset);
  112. const doc = deserialize(docBuffer);
  113. yield { doc, loc: [offset, size] };
  114. offset += size;
  115. }
  116. } catch (e) {
  117. if (e.code !== 'ENOENT') {
  118. throw e;
  119. }
  120. // if file does not exist, do nothing
  121. } finally {
  122. await fileHandle?.close();
  123. }
  124. }
  125. /**
  126. * Compares two keys for sorting
  127. * @param {any} keyA - First key
  128. * @param {any} keyB - Second key
  129. * @returns {number} -1 if keyA < keyB, 0 if equal, 1 if keyA > keyB
  130. * @private
  131. */
  132. _compareKeys(keyA, keyB) {
  133. if (typeof keyA === 'number' && typeof keyB === 'number') {
  134. return keyA - keyB;
  135. }
  136. return String(keyA).localeCompare(String(keyB));
  137. }
  138. /**
  139. * Starts automatic periodic persistence of indexes
  140. * @param {number} interval - Interval in milliseconds between auto-persists
  141. * @param {number} threshold - Number of operations before triggering auto-persist
  142. */
  143. startAutoPersist(interval, threshold) {
  144. this._persistThreshold = threshold;
  145. // Start periodic persist
  146. if (interval > 0) {
  147. this._persistIntervalId = setInterval(async () => {
  148. if (this._totalDeltaCount > 0) {
  149. await this.persist();
  150. }
  151. }, interval);
  152. // Don't keep process alive just for this timer
  153. this._persistIntervalId.unref();
  154. }
  155. }
  156. /**
  157. * Closes the index manager (persists indexes and clears interval)
  158. * @returns {Promise<void>}
  159. */
  160. async close() {
  161. if (this._persistIntervalId) {
  162. clearInterval(this._persistIntervalId);
  163. this._persistIntervalId = null;
  164. }
  165. await this.persist();
  166. }
  167. /**
  168. * Persists all in-memory index deltas to disk
  169. * @returns {Promise<void>}
  170. */
  171. async persist() {
  172. for (const field of this._indexes) {
  173. let onDiskEntries = [];
  174. const indexPath = this._indexPaths[field];
  175. try {
  176. const indexContent = await readFile(indexPath, 'utf-8');
  177. onDiskEntries = indexContent.trim().split('\n').filter(Boolean).map(line => {
  178. const [key, offset, length] = line.split(',');
  179. return { key: this._parseKey(key, field), loc: [parseInt(offset), parseInt(length)] };
  180. });
  181. } catch (e) {
  182. if (e.code !== 'ENOENT') throw e;
  183. }
  184. const fieldTombstones = this._tombstones[field];
  185. const validDiskEntries = onDiskEntries.filter(e => !fieldTombstones.has(e.loc[0]));
  186. const deltaEntries = this._deltaIndexes[field] || [];
  187. const finalIndexData = [...validDiskEntries, ...deltaEntries];
  188. finalIndexData.sort((a, b) => this._compareKeys(a.key, b.key));
  189. if (finalIndexData.length > 0) {
  190. const indexContent = finalIndexData.map(e => `${e.key},${e.loc[0]},${e.loc[1]}`).join('\n') + '\n';
  191. await writeFile(indexPath, indexContent, 'utf-8');
  192. } else {
  193. await writeFile(indexPath, '', 'utf-8');
  194. }
  195. // Clear the in-memory changes now that they are persisted
  196. this._deltaIndexes[field] = [];
  197. this._tombstones[field].clear();
  198. }
  199. this._totalDeltaCount = 0;
  200. }
  201. /**
  202. * Parses a key based on the field's index type
  203. * @param {any} key - The key to parse
  204. * @param {string} field - Field name
  205. * @returns {any} Parsed key (number or string)
  206. * @private
  207. */
  208. _parseKey(key, field) {
  209. const indexType = this._indexTypes[field];
  210. // Use index type if specified, otherwise fall back to primary key type logic
  211. if (indexType === IndexType.NUMERIC) {
  212. const num = parseInt(key, 10);
  213. if (!isNaN(num)) return num;
  214. } else if (indexType === IndexType.LEXICAL) {
  215. return key;
  216. }
  217. // Fallback to primary key type for fields without explicit index type
  218. if (this._primaryKeyType === PrimaryKeyType.NUMBER) {
  219. const num = parseInt(key, 10);
  220. if (!isNaN(num)) return num;
  221. }
  222. return key;
  223. }
  224. /**
  225. * Performs a binary search on the blocks of an index file on disk.
  226. * This is a direct port of Joshua Bloch's famously correct binary search
  227. * algorithm, adapted for file blocks.
  228. *
  229. * @param {string} field The index field to search.
  230. * @param {string|number} key The key to search for.
  231. * @returns {Promise<number>} The starting offset in the file for the linear scan.
  232. */
  233. async _binarySearchOnDisk(field, key) {
  234. try {
  235. const fileHandle = await open(this._indexPaths[field], 'r');
  236. const stats = await fileHandle.stat();
  237. const totalBlocks = Math.ceil(stats.size / BLOCK_SIZE);
  238. let low = 0;
  239. let high = totalBlocks - 1;
  240. const buffer = Buffer.alloc(1024);
  241. while (low <= high) {
  242. const mid = Math.floor((low + high) / 2);
  243. const offset = mid * BLOCK_SIZE;
  244. const { bytesRead } = await fileHandle.read(buffer, 0, 1024, offset);
  245. if (bytesRead === 0) break;
  246. const blockContent = buffer.subarray(0, bytesRead).toString('utf-8');
  247. const lines = blockContent.split('\n');
  248. const firstKeyInBlock = this._parseKey(lines[0].split(',')[0], field);
  249. const lastKeyInBlock = this._parseKey(lines[lines.length - 2]?.split(',')[0] || firstKeyInBlock, field);
  250. const cmpLast = this._compareKeys(lastKeyInBlock, key);
  251. if (cmpLast < 0) {
  252. // The entire block is smaller than the key
  253. low = mid + 1;
  254. } else {
  255. // The key might be in this block or a previous one
  256. high = mid - 1;
  257. }
  258. }
  259. await fileHandle.close();
  260. // low is the insertion point, start scan from the previous block to be safe.
  261. const finalOffset = Math.max(0, low - 1) * BLOCK_SIZE;
  262. return finalOffset;
  263. } catch (e) {
  264. if (e.code === 'ENOENT') {
  265. return -1;
  266. }
  267. throw e;
  268. }
  269. }
  270. /**
  271. * Finds the last entry for a given field and key (checks both delta and disk)
  272. * @param {string} field - Field name
  273. * @param {any} key - Key value to search for
  274. * @returns {Promise<Object|null>} Entry object with key and loc, or null if not found
  275. */
  276. async findLastEntry(field, key) {
  277. const parsedKey = this._parseKey(key, field);
  278. const deltaIndex = this._deltaIndexes[field] || [];
  279. for (let i = deltaIndex.length - 1; i >= 0; i--) {
  280. const entry = deltaIndex[i];
  281. if (this._compareKeys(entry.key, parsedKey) === 0) {
  282. return entry;
  283. }
  284. }
  285. const onDiskEntry = await this._findLastEntryOnDisk(field, parsedKey);
  286. if (!onDiskEntry) return null;
  287. const isTombstoned = this._tombstones[field].has(onDiskEntry.loc[0]);
  288. return isTombstoned ? null : onDiskEntry;
  289. }
  290. /**
  291. * Finds the last entry for a key on disk (linear scan from binary search position)
  292. * @param {string} field - Field name
  293. * @param {any} key - Key to search for
  294. * @returns {Promise<Object|null>} Entry object or null
  295. * @private
  296. */
  297. async _findLastEntryOnDisk(field, key) {
  298. const startOffset = await this._binarySearchOnDisk(field, key);
  299. if (startOffset === -1) return null;
  300. let lastMatch = null;
  301. const fileHandle = await open(this._indexPaths[field], 'r');
  302. const buffer = Buffer.alloc(BLOCK_SIZE);
  303. let currentOffset = startOffset;
  304. try {
  305. const stats = await fileHandle.stat();
  306. while (currentOffset < stats.size) {
  307. const { bytesRead } = await fileHandle.read(buffer, 0, BLOCK_SIZE, currentOffset);
  308. if (bytesRead === 0) break;
  309. const blockContent = buffer.slice(0, bytesRead).toString('utf-8');
  310. const lines = blockContent.split('\n');
  311. let found = false;
  312. for (const line of lines) {
  313. if (!line) continue;
  314. const [keyStr, offset, length] = line.split(',');
  315. const lineKey = this._parseKey(keyStr, field);
  316. if (this._compareKeys(lineKey, key) === 0) {
  317. found = true;
  318. lastMatch = { key: lineKey, loc: [parseInt(offset), parseInt(length)] };
  319. }
  320. }
  321. if (found) break; // We can stop once we find the block with the key
  322. currentOffset += BLOCK_SIZE;
  323. }
  324. } finally {
  325. await fileHandle.close();
  326. }
  327. return lastMatch;
  328. }
  329. /**
  330. * Adds a document to the indexes
  331. * @param {Object} doc - The document to index
  332. * @param {Array} loc - [offset, length] location in data file
  333. */
  334. insert(doc, loc) {
  335. for (const field of this._indexes) {
  336. if (doc[field] !== undefined) {
  337. this._deltaIndexes[field].push({ key: doc[field], loc });
  338. this._totalDeltaCount++;
  339. }
  340. }
  341. // Auto-persist if threshold reached
  342. if (this._totalDeltaCount >= this._persistThreshold) {
  343. // Persist async without blocking
  344. this.persist().catch(err => console.error('Index persist error:', err));
  345. }
  346. }
  347. /**
  348. * Removes a document from the indexes
  349. * @param {Object} doc - The document to remove
  350. * @param {string} primaryKeyField - Primary key field name
  351. * @returns {Promise<void>}
  352. */
  353. async remove(doc, primaryKeyField) {
  354. const pkValue = doc[primaryKeyField];
  355. const deltaPkIndex = this._deltaIndexes[primaryKeyField] || [];
  356. const indexInDelta = deltaPkIndex.findIndex(e => this._compareKeys(e.key, pkValue) === 0);
  357. if (indexInDelta !== -1) {
  358. const offsetToRemove = deltaPkIndex[indexInDelta].loc[0];
  359. for (const field of this._indexes) {
  360. this._deltaIndexes[field] = (this._deltaIndexes[field] || []).filter(e => e.loc[0] !== offsetToRemove);
  361. }
  362. } else {
  363. const onDiskEntry = await this._findLastEntryOnDisk(primaryKeyField, pkValue);
  364. if (onDiskEntry) {
  365. for (const field of this._indexes) {
  366. if (doc[field] !== undefined) {
  367. this._tombstones[field].set(onDiskEntry.loc[0], true);
  368. }
  369. }
  370. }
  371. }
  372. }
  373. /**
  374. * Gets all document locations from the first index (used for full scans)
  375. * @returns {Promise<Array[]>} Array of all [offset, length] locations
  376. */
  377. async getAllLocations() {
  378. const field = this._indexes[0];
  379. const indexPath = this._indexPaths[field];
  380. let onDiskEntries = [];
  381. try {
  382. const indexContent = await readFile(indexPath, 'utf-8');
  383. onDiskEntries = indexContent.trim().split('\n').filter(Boolean).map(line => {
  384. const [, offset, length] = line.split(',');
  385. return { loc: [parseInt(offset), parseInt(length)] };
  386. });
  387. } catch (e) {
  388. if (e.code !== 'ENOENT') throw e;
  389. }
  390. const fieldTombstones = this._tombstones[field] || new Map();
  391. const validDiskEntries = onDiskEntries.filter(e => !fieldTombstones.has(e.loc[0]));
  392. const deltaEntries = this._deltaIndexes[field] || [];
  393. const finalEntries = new Map();
  394. for (const entry of validDiskEntries) {
  395. finalEntries.set(`${entry.loc[0]}:${entry.loc[1]}`, entry.loc);
  396. }
  397. for (const entry of deltaEntries) {
  398. finalEntries.set(`${entry.loc[0]}:${entry.loc[1]}`, entry.loc);
  399. }
  400. return Array.from(finalEntries.values());
  401. }
  402. /**
  403. * Gets document locations for a given field and key
  404. * @param {string} field - Field name
  405. * @param {any} key - Key value to search for
  406. * @returns {Promise<Array[]>} Array of [offset, length] locations
  407. */
  408. async get(field, key) {
  409. const parsedKey = this._parseKey(key, field);
  410. const onDiskResults = await this._getOnDisk(field, parsedKey);
  411. const deltaResults = (this._deltaIndexes[field] || []).filter(e => this._compareKeys(e.key, parsedKey) === 0);
  412. const tombstonedOffsets = this._tombstones[field] || new Map();
  413. const finalEntries = new Map();
  414. for (const entry of onDiskResults) {
  415. if (!tombstonedOffsets.has(entry.loc[0])) {
  416. finalEntries.set(`${entry.loc[0]}:${entry.loc[1]}`, entry);
  417. }
  418. }
  419. for (const entry of deltaResults) {
  420. finalEntries.set(`${entry.loc[0]}:${entry.loc[1]}`, entry);
  421. }
  422. return Array.from(finalEntries.values());
  423. }
  424. /**
  425. * Gets all entries for a key from disk
  426. * @param {string} field - Field name
  427. * @param {any} key - Key to search for
  428. * @returns {Promise<Array>} Array of entry objects
  429. * @private
  430. */
  431. async _getOnDisk(field, key) {
  432. const startOffset = await this._binarySearchOnDisk(field, key);
  433. if (startOffset === -1) return [];
  434. const results = [];
  435. const fileHandle = await open(this._indexPaths[field], 'r');
  436. const buffer = Buffer.alloc(BLOCK_SIZE);
  437. let currentOffset = startOffset;
  438. try {
  439. const stats = await fileHandle.stat();
  440. while (currentOffset < stats.size) {
  441. const { bytesRead } = await fileHandle.read(buffer, 0, BLOCK_SIZE, currentOffset);
  442. if (bytesRead === 0) break;
  443. const blockContent = buffer.subarray(0, bytesRead).toString('utf-8');
  444. const lines = blockContent.split('\n');
  445. let firstKeyInBlock;
  446. let foundMatchInBlock = false;
  447. for (const line of lines) {
  448. if (!line) continue;
  449. const [keyStr, locOffset, locLength] = line.split(',');
  450. if (firstKeyInBlock === undefined) {
  451. firstKeyInBlock = this._parseKey(keyStr, field);
  452. }
  453. const lineKey = this._parseKey(keyStr, field);
  454. if (this._compareKeys(lineKey, key) === 0) {
  455. foundMatchInBlock = true;
  456. results.push({ key: lineKey, loc: [parseInt(locOffset), parseInt(locLength)] });
  457. }
  458. }
  459. if (results.length > 0 && !foundMatchInBlock) {
  460. break;
  461. }
  462. if (foundMatchInBlock === false && firstKeyInBlock && this._compareKeys(firstKeyInBlock, key) > 0) {
  463. break;
  464. }
  465. currentOffset += BLOCK_SIZE;
  466. }
  467. } finally {
  468. await fileHandle.close();
  469. }
  470. return results;
  471. }
  472. async* _getDocsByLocation(dataPath, locations) {
  473. const fileHandle = await open(dataPath, 'r');
  474. try {
  475. for (const loc of locations) {
  476. if (loc && loc.length === 2 && loc[1] > 0) {
  477. const [offset, length] = loc;
  478. const buffer = Buffer.alloc(length);
  479. await fileHandle.read(buffer, 0, length, offset);
  480. yield deserialize(buffer);
  481. }
  482. }
  483. } finally {
  484. await fileHandle.close();
  485. }
  486. }
  487. }
  488. export default IndexManager;

Branches

Latest commits

  • b4db6391initial commitcaramboleyo