gitoriaLog in with ident

mpackdb

All repositories: gitoria

ReadmeCodePull requestsReleasesTicketsSettings
Commitb8ffc1a0b8ffc1a0release 1.0.5caramboleyob8ffc1a0/src/IndexManager.js

19.7 KB

  1. import { writeFile, readFile, open, rename } from 'fs/promises';
  2. import { createWriteStream } from 'fs';
  3. import { resolve } from 'path';
  4. import { deserialize } from './mpack.js';
  5. import { PrimaryKeyType, IndexType } from './MPackDB.js';
  6. const BLOCK_SIZE = 4096; // 4KB
  7. /**
  8. * Manages indexes for the database (in-memory deltas + on-disk persistence)
  9. */
  10. export class IndexManager {
  11. _dbDir;
  12. _dbName;
  13. _indexes;
  14. _indexTypes;
  15. _primaryKeyType;
  16. _persistIntervalId = null;
  17. _indexPaths = {};
  18. _deltaIndexes = {};
  19. _tombstones = {};
  20. _persistThreshold = 1000;
  21. _totalDeltaCount = 0;
  22. /**
  23. * Creates a new IndexManager
  24. * @param {string} dbDir - Database directory path
  25. * @param {string} dbName - Database name
  26. * @param {string[]} indexes - Array of field names to index
  27. * @param {Object} indexTypes - Map of field names to IndexType
  28. * @param {number} primaryKeyType - Primary key type
  29. */
  30. constructor(dbDir, dbName, indexes, indexTypes, primaryKeyType) {
  31. this._dbDir = dbDir;
  32. this._dbName = dbName;
  33. this._indexes = indexes;
  34. this._indexTypes = indexTypes;
  35. this._primaryKeyType = primaryKeyType;
  36. }
  37. /**
  38. * Initializes the index manager (rebuilds indexes from data file)
  39. * @param {string} dataPath - Path to the data file
  40. * @returns {Promise<void>}
  41. */
  42. async init(dataPath, { forceRebuild = false } = {}) {
  43. for (const field of this._indexes) {
  44. this._deltaIndexes[field] = [];
  45. this._tombstones[field] = new Map();
  46. const indexPath = resolve(this._dbDir, `${this._dbName}.${field}.txt`);
  47. this._indexPaths[field] = indexPath;
  48. if (forceRebuild) {
  49. await this._rebuildIndex(dataPath, field);
  50. continue;
  51. }
  52. try {
  53. const fileHandle = await open(indexPath, 'r');
  54. const stats = await fileHandle.stat();
  55. await fileHandle.close();
  56. if (stats.size === 0) {
  57. await this._rebuildIndex(dataPath, field);
  58. }
  59. } catch (e) {
  60. if (e.code === 'ENOENT') {
  61. await this._rebuildIndex(dataPath, field);
  62. } else {
  63. throw e;
  64. }
  65. }
  66. }
  67. }
  68. /**
  69. * Rebuilds an index from the data file
  70. * @param {string} dataPath - Path to the data file
  71. * @param {string} field - Field name to index
  72. * @returns {Promise<void>}
  73. * @private
  74. */
  75. async _rebuildIndex(dataPath, field) {
  76. const tempIndexPath = `${this._indexPaths[field]}.${Date.now()}.tmp`;
  77. const writeStream = createWriteStream(tempIndexPath, { flags: 'w', encoding: 'utf-8' });
  78. const docIterator = this._getDocsAndLocationsFromDataFile(dataPath);
  79. const entries = [];
  80. for await (const { doc, loc } of docIterator) {
  81. if (doc && doc[field] !== undefined) {
  82. const key = doc[field];
  83. entries.push({ key, loc });
  84. }
  85. }
  86. // Sort entries before writing to the index file
  87. entries.sort((a, b) => this._compareKeys(a.key, b.key));
  88. for (const entry of entries) {
  89. writeStream.write(`${entry.key},${entry.loc[0]},${entry.loc[1]}\n`);
  90. }
  91. await new Promise(resolve => writeStream.end(resolve));
  92. await rename(tempIndexPath, this._indexPaths[field]);
  93. }
  94. /**
  95. * Generator that yields documents and their locations from the data file
  96. * @param {string} dataPath - Path to the data file
  97. * @yields {{doc: Object, loc: Array}} Document and location tuple
  98. * @private
  99. */
  100. async* _getDocsAndLocationsFromDataFile(dataPath) {
  101. let fileHandle;
  102. try {
  103. fileHandle = await open(dataPath, 'r');
  104. const stats = await fileHandle.stat();
  105. let offset = 0;
  106. while (offset < stats.size) {
  107. // Read size header
  108. const sizeBuffer = Buffer.alloc(4);
  109. await fileHandle.read(sizeBuffer, 0, 4, offset);
  110. const size = sizeBuffer.readInt32LE(0);
  111. if (size <= 0 || offset + size > stats.size) {
  112. break; // Corrupted or incomplete record
  113. }
  114. const docBuffer = Buffer.alloc(size);
  115. await fileHandle.read(docBuffer, 0, size, offset);
  116. const doc = deserialize(docBuffer.subarray(4));
  117. yield { doc, loc: [offset, size] };
  118. offset += size;
  119. }
  120. } catch (e) {
  121. if (e.code !== 'ENOENT') {
  122. throw e;
  123. }
  124. // if file does not exist, do nothing
  125. } finally {
  126. await fileHandle?.close();
  127. }
  128. }
  129. /**
  130. * Compares two keys for sorting
  131. * @param {any} keyA - First key
  132. * @param {any} keyB - Second key
  133. * @returns {number} -1 if keyA < keyB, 0 if equal, 1 if keyA > keyB
  134. * @private
  135. */
  136. _compareKeys(keyA, keyB) {
  137. if (typeof keyA === 'number' && typeof keyB === 'number') {
  138. return keyA - keyB;
  139. }
  140. return String(keyA).localeCompare(String(keyB));
  141. }
  142. /**
  143. * Starts automatic periodic persistence of indexes
  144. * @param {number} interval - Interval in milliseconds between auto-persists
  145. * @param {number} threshold - Number of operations before triggering auto-persist
  146. */
  147. startAutoPersist(interval, threshold) {
  148. this._persistThreshold = threshold;
  149. // Start periodic persist
  150. if (interval > 0) {
  151. this._persistIntervalId = setInterval(async () => {
  152. if (this._totalDeltaCount > 0) {
  153. await this.persist();
  154. }
  155. }, interval);
  156. // Don't keep process alive just for this timer
  157. this._persistIntervalId.unref();
  158. }
  159. }
  160. /**
  161. * Closes the index manager (persists indexes and clears interval)
  162. * @returns {Promise<void>}
  163. */
  164. async close() {
  165. if (this._persistIntervalId) {
  166. clearInterval(this._persistIntervalId);
  167. this._persistIntervalId = null;
  168. }
  169. await this.persist();
  170. }
  171. /**
  172. * Persists all in-memory index deltas to disk
  173. * @returns {Promise<void>}
  174. */
  175. async persist() {
  176. for (const field of this._indexes) {
  177. let onDiskEntries = [];
  178. const indexPath = this._indexPaths[field];
  179. try {
  180. const indexContent = await readFile(indexPath, 'utf-8');
  181. onDiskEntries = indexContent.trim().split('\n').filter(Boolean).map(line => {
  182. const [key, offset, length] = line.split(',');
  183. return { key: this._parseKey(key, field), loc: [parseInt(offset), parseInt(length)] };
  184. });
  185. } catch (e) {
  186. if (e.code !== 'ENOENT') throw e;
  187. }
  188. const fieldTombstones = this._tombstones[field];
  189. const validDiskEntries = onDiskEntries.filter(e => !fieldTombstones.has(e.loc[0]));
  190. const deltaEntries = this._deltaIndexes[field] || [];
  191. const finalIndexData = [...validDiskEntries, ...deltaEntries];
  192. finalIndexData.sort((a, b) => this._compareKeys(a.key, b.key));
  193. if (finalIndexData.length > 0) {
  194. const indexContent = finalIndexData.map(e => `${e.key},${e.loc[0]},${e.loc[1]}`).join('\n') + '\n';
  195. await writeFile(indexPath, indexContent, 'utf-8');
  196. } else {
  197. await writeFile(indexPath, '', 'utf-8');
  198. }
  199. // Clear the in-memory changes now that they are persisted
  200. this._deltaIndexes[field] = [];
  201. this._tombstones[field].clear();
  202. }
  203. this._totalDeltaCount = 0;
  204. }
  205. /**
  206. * Parses a key based on the field's index type
  207. * @param {any} key - The key to parse
  208. * @param {string} field - Field name
  209. * @returns {any} Parsed key (number or string)
  210. * @private
  211. */
  212. _parseKey(key, field) {
  213. const indexType = this._indexTypes[field];
  214. // Use index type if specified, otherwise fall back to primary key type logic
  215. if (indexType === IndexType.NUMERIC) {
  216. const num = parseInt(key, 10);
  217. if (!isNaN(num)) return num;
  218. } else if (indexType === IndexType.LEXICAL) {
  219. return key;
  220. }
  221. // Fallback to primary key type for fields without explicit index type
  222. if (this._primaryKeyType === PrimaryKeyType.NUMBER) {
  223. const num = parseInt(key, 10);
  224. if (!isNaN(num)) return num;
  225. }
  226. return key;
  227. }
  228. /**
  229. * Performs a binary search on the blocks of an index file on disk.
  230. * This is a direct port of Joshua Bloch's famously correct binary search
  231. * algorithm, adapted for file blocks.
  232. *
  233. * @param {string} field The index field to search.
  234. * @param {string|number} key The key to search for.
  235. * @returns {Promise<number>} The starting offset in the file for the linear scan.
  236. */
  237. async _binarySearchOnDisk(field, key) {
  238. try {
  239. const fileHandle = await open(this._indexPaths[field], 'r');
  240. const stats = await fileHandle.stat();
  241. const totalBlocks = Math.ceil(stats.size / BLOCK_SIZE);
  242. let low = 0;
  243. let high = totalBlocks - 1;
  244. const buffer = Buffer.alloc(1024);
  245. while (low <= high) {
  246. const mid = Math.floor((low + high) / 2);
  247. const offset = mid * BLOCK_SIZE;
  248. const { bytesRead } = await fileHandle.read(buffer, 0, 1024, offset);
  249. if (bytesRead === 0) break;
  250. const blockContent = buffer.subarray(0, bytesRead).toString('utf-8');
  251. const lines = blockContent.split('\n');
  252. const firstKeyInBlock = this._parseKey(lines[0].split(',')[0], field);
  253. const lastKeyInBlock = this._parseKey(lines[lines.length - 2]?.split(',')[0] || firstKeyInBlock, field);
  254. const cmpLast = this._compareKeys(lastKeyInBlock, key);
  255. if (cmpLast < 0) {
  256. // The entire block is smaller than the key
  257. low = mid + 1;
  258. } else {
  259. // The key might be in this block or a previous one
  260. high = mid - 1;
  261. }
  262. }
  263. await fileHandle.close();
  264. // low is the insertion point, start scan from the previous block to be safe.
  265. const finalOffset = Math.max(0, low - 1) * BLOCK_SIZE;
  266. return finalOffset;
  267. } catch (e) {
  268. if (e.code === 'ENOENT') {
  269. return -1;
  270. }
  271. throw e;
  272. }
  273. }
  274. /**
  275. * Finds the last entry for a given field and key (checks both delta and disk)
  276. * @param {string} field - Field name
  277. * @param {any} key - Key value to search for
  278. * @returns {Promise<Object|null>} Entry object with key and loc, or null if not found
  279. */
  280. async findLastEntry(field, key) {
  281. const parsedKey = this._parseKey(key, field);
  282. const deltaIndex = this._deltaIndexes[field] || [];
  283. for (let i = deltaIndex.length - 1; i >= 0; i--) {
  284. const entry = deltaIndex[i];
  285. if (this._compareKeys(entry.key, parsedKey) === 0) {
  286. return entry;
  287. }
  288. }
  289. const onDiskEntry = await this._findLastEntryOnDisk(field, parsedKey);
  290. if (!onDiskEntry) return null;
  291. const isTombstoned = this._tombstones[field].has(onDiskEntry.loc[0]);
  292. return isTombstoned ? null : onDiskEntry;
  293. }
  294. /**
  295. * Finds the last entry for a key on disk (linear scan from binary search position)
  296. * @param {string} field - Field name
  297. * @param {any} key - Key to search for
  298. * @returns {Promise<Object|null>} Entry object or null
  299. * @private
  300. */
  301. async _findLastEntryOnDisk(field, key) {
  302. const startOffset = await this._binarySearchOnDisk(field, key);
  303. if (startOffset === -1) return null;
  304. let lastMatch = null;
  305. const fileHandle = await open(this._indexPaths[field], 'r');
  306. const buffer = Buffer.alloc(BLOCK_SIZE);
  307. let currentOffset = startOffset;
  308. try {
  309. const stats = await fileHandle.stat();
  310. while (currentOffset < stats.size) {
  311. const { bytesRead } = await fileHandle.read(buffer, 0, BLOCK_SIZE, currentOffset);
  312. if (bytesRead === 0) break;
  313. const blockContent = buffer.slice(0, bytesRead).toString('utf-8');
  314. const lines = blockContent.split('\n');
  315. let found = false;
  316. for (const line of lines) {
  317. if (!line) continue;
  318. const [keyStr, offset, length] = line.split(',');
  319. const lineKey = this._parseKey(keyStr, field);
  320. if (this._compareKeys(lineKey, key) === 0) {
  321. found = true;
  322. lastMatch = { key: lineKey, loc: [parseInt(offset), parseInt(length)] };
  323. }
  324. }
  325. if (found) break; // We can stop once we find the block with the key
  326. currentOffset += BLOCK_SIZE;
  327. }
  328. } finally {
  329. await fileHandle.close();
  330. }
  331. return lastMatch;
  332. }
  333. /**
  334. * Adds a document to the indexes
  335. * @param {Object} doc - The document to index
  336. * @param {Array} loc - [offset, length] location in data file
  337. */
  338. insert(doc, loc) {
  339. for (const field of this._indexes) {
  340. if (doc[field] !== undefined) {
  341. this._deltaIndexes[field].push({ key: doc[field], loc });
  342. this._totalDeltaCount++;
  343. }
  344. }
  345. // Auto-persist if threshold reached
  346. if (this._totalDeltaCount >= this._persistThreshold) {
  347. // Persist async without blocking
  348. this.persist().catch(err => console.error('Index persist error:', err));
  349. }
  350. }
  351. /**
  352. * Removes a document from the indexes
  353. * @param {Object} doc - The document to remove
  354. * @param {string} primaryKeyField - Primary key field name
  355. * @returns {Promise<void>}
  356. */
  357. async remove(doc, primaryKeyField) {
  358. const pkValue = doc[primaryKeyField];
  359. const deltaPkIndex = this._deltaIndexes[primaryKeyField] || [];
  360. const indexInDelta = deltaPkIndex.findIndex(e => this._compareKeys(e.key, pkValue) === 0);
  361. if (indexInDelta !== -1) {
  362. const offsetToRemove = deltaPkIndex[indexInDelta].loc[0];
  363. for (const field of this._indexes) {
  364. this._deltaIndexes[field] = (this._deltaIndexes[field] || []).filter(e => e.loc[0] !== offsetToRemove);
  365. }
  366. } else {
  367. const onDiskEntry = await this._findLastEntryOnDisk(primaryKeyField, pkValue);
  368. if (onDiskEntry) {
  369. for (const field of this._indexes) {
  370. if (doc[field] !== undefined) {
  371. this._tombstones[field].set(onDiskEntry.loc[0], true);
  372. }
  373. }
  374. }
  375. }
  376. }
  377. /**
  378. * Gets all document locations from the first index (used for full scans)
  379. * @returns {Promise<Array[]>} Array of all [offset, length] locations
  380. */
  381. async getAllLocations() {
  382. const field = this._indexes[0];
  383. const indexPath = this._indexPaths[field];
  384. let onDiskEntries = [];
  385. try {
  386. const indexContent = await readFile(indexPath, 'utf-8');
  387. onDiskEntries = indexContent.trim().split('\n').filter(Boolean).map(line => {
  388. const [, offset, length] = line.split(',');
  389. return { loc: [parseInt(offset), parseInt(length)] };
  390. });
  391. } catch (e) {
  392. if (e.code !== 'ENOENT') throw e;
  393. }
  394. const fieldTombstones = this._tombstones[field] || new Map();
  395. const validDiskEntries = onDiskEntries.filter(e => !fieldTombstones.has(e.loc[0]));
  396. const deltaEntries = this._deltaIndexes[field] || [];
  397. const finalEntries = new Map();
  398. for (const entry of validDiskEntries) {
  399. finalEntries.set(`${entry.loc[0]}:${entry.loc[1]}`, entry.loc);
  400. }
  401. for (const entry of deltaEntries) {
  402. finalEntries.set(`${entry.loc[0]}:${entry.loc[1]}`, entry.loc);
  403. }
  404. return Array.from(finalEntries.values());
  405. }
  406. /**
  407. * Gets document locations for a given field and key
  408. * @param {string} field - Field name
  409. * @param {any} key - Key value to search for
  410. * @returns {Promise<Array[]>} Array of [offset, length] locations
  411. */
  412. async get(field, key) {
  413. const parsedKey = this._parseKey(key, field);
  414. const onDiskResults = await this._getOnDisk(field, parsedKey);
  415. const deltaResults = (this._deltaIndexes[field] || []).filter(e => this._compareKeys(e.key, parsedKey) === 0);
  416. const tombstonedOffsets = this._tombstones[field] || new Map();
  417. const finalEntries = new Map();
  418. for (const entry of onDiskResults) {
  419. if (!tombstonedOffsets.has(entry.loc[0])) {
  420. finalEntries.set(`${entry.loc[0]}:${entry.loc[1]}`, entry);
  421. }
  422. }
  423. for (const entry of deltaResults) {
  424. finalEntries.set(`${entry.loc[0]}:${entry.loc[1]}`, entry);
  425. }
  426. return Array.from(finalEntries.values());
  427. }
  428. /**
  429. * Gets all entries for a key from disk
  430. * @param {string} field - Field name
  431. * @param {any} key - Key to search for
  432. * @returns {Promise<Array>} Array of entry objects
  433. * @private
  434. */
  435. async _getOnDisk(field, key) {
  436. const startOffset = await this._binarySearchOnDisk(field, key);
  437. if (startOffset === -1) return [];
  438. const results = [];
  439. const fileHandle = await open(this._indexPaths[field], 'r');
  440. const buffer = Buffer.alloc(BLOCK_SIZE);
  441. let currentOffset = startOffset;
  442. try {
  443. const stats = await fileHandle.stat();
  444. while (currentOffset < stats.size) {
  445. const { bytesRead } = await fileHandle.read(buffer, 0, BLOCK_SIZE, currentOffset);
  446. if (bytesRead === 0) break;
  447. const blockContent = buffer.subarray(0, bytesRead).toString('utf-8');
  448. const lines = blockContent.split('\n');
  449. let firstKeyInBlock;
  450. let foundMatchInBlock = false;
  451. for (const line of lines) {
  452. if (!line) continue;
  453. const [keyStr, locOffset, locLength] = line.split(',');
  454. if (firstKeyInBlock === undefined) {
  455. firstKeyInBlock = this._parseKey(keyStr, field);
  456. }
  457. const lineKey = this._parseKey(keyStr, field);
  458. if (this._compareKeys(lineKey, key) === 0) {
  459. foundMatchInBlock = true;
  460. results.push({ key: lineKey, loc: [parseInt(locOffset), parseInt(locLength)] });
  461. }
  462. }
  463. if (results.length > 0 && !foundMatchInBlock) {
  464. break;
  465. }
  466. if (foundMatchInBlock === false && firstKeyInBlock && this._compareKeys(firstKeyInBlock, key) > 0) {
  467. break;
  468. }
  469. currentOffset += BLOCK_SIZE;
  470. }
  471. } finally {
  472. await fileHandle.close();
  473. }
  474. return results;
  475. }
  476. /**
  477. * Yields index entries in sorted order by streaming blocks from disk.
  478. * Merges with in-memory deltas, excludes tombstones.
  479. * @param {string} field - Field name
  480. * @param {Object} [options]
  481. * @param {any} [options.from] - Start key (inclusive), uses binary search to skip ahead
  482. * @param {number} [options.limit] - Max entries to yield
  483. * @yields {{key: any, loc: [number, number]}}
  484. */
  485. async *entries(field, { from, limit } = {}) {
  486. const tombstones = this._tombstones[field] || new Map();
  487. // Collect and sort deltas (small, always in memory)
  488. const deltas = (this._deltaIndexes[field] || [])
  489. .filter(e => !tombstones.has(e.loc[0]))
  490. .slice()
  491. .sort((a, b) => this._compareKeys(a.key, b.key));
  492. let deltaIdx = 0;
  493. // Skip deltas before `from`
  494. if (from !== undefined) {
  495. while (deltaIdx < deltas.length && this._compareKeys(deltas[deltaIdx].key, from) < 0) {
  496. deltaIdx++;
  497. }
  498. }
  499. // Stream disk entries block by block
  500. let fileHandle;
  501. let fileSize = 0;
  502. let startOffset = 0;
  503. try {
  504. fileHandle = await open(this._indexPaths[field], 'r');
  505. fileSize = (await fileHandle.stat()).size;
  506. if (from !== undefined) {
  507. startOffset = await this._binarySearchOnDisk(field, from);
  508. if (startOffset === -1) startOffset = 0;
  509. }
  510. } catch (e) {
  511. if (e.code !== 'ENOENT') throw e;
  512. }
  513. const buffer = Buffer.alloc(BLOCK_SIZE);
  514. let currentOffset = startOffset;
  515. let count = 0;
  516. const seen = new Set();
  517. // Generator for disk entries
  518. const self = this;
  519. async function* diskEntries() {
  520. if (!fileHandle) return;
  521. try {
  522. while (currentOffset < fileSize) {
  523. const { bytesRead } = await fileHandle.read(buffer, 0, BLOCK_SIZE, currentOffset);
  524. if (bytesRead === 0) break;
  525. const lines = buffer.subarray(0, bytesRead).toString('utf-8').split('\n');
  526. for (const line of lines) {
  527. if (!line) continue;
  528. const [keyStr, locOffset, locLength] = line.split(',');
  529. const key = self._parseKey(keyStr, field);
  530. const loc = [parseInt(locOffset), parseInt(locLength)];
  531. if (tombstones.has(loc[0])) continue;
  532. if (from !== undefined && self._compareKeys(key, from) < 0) continue;
  533. yield { key, loc };
  534. }
  535. currentOffset += BLOCK_SIZE;
  536. }
  537. } finally {
  538. await fileHandle.close();
  539. }
  540. }
  541. // Merge disk + deltas in sorted order
  542. const diskIter = diskEntries();
  543. let diskNext = await diskIter.next();
  544. while (true) {
  545. const hasDisk = !diskNext.done;
  546. const hasDelta = deltaIdx < deltas.length;
  547. if (!hasDisk && !hasDelta) break;
  548. let entry;
  549. if (!hasDisk) {
  550. entry = deltas[deltaIdx++];
  551. } else if (!hasDelta) {
  552. entry = diskNext.value;
  553. diskNext = await diskIter.next();
  554. } else if (this._compareKeys(diskNext.value.key, deltas[deltaIdx].key) <= 0) {
  555. entry = diskNext.value;
  556. diskNext = await diskIter.next();
  557. } else {
  558. entry = deltas[deltaIdx++];
  559. }
  560. const locKey = `${entry.loc[0]}:${entry.loc[1]}`;
  561. if (seen.has(locKey)) continue;
  562. seen.add(locKey);
  563. yield entry;
  564. count++;
  565. if (limit !== undefined && count >= limit) {
  566. // Clean up disk iterator
  567. if (!diskNext.done) await diskIter.return();
  568. return;
  569. }
  570. }
  571. }
  572. async* _getDocsByLocation(dataPath, locations) {
  573. const fileHandle = await open(dataPath, 'r');
  574. try {
  575. for (const loc of locations) {
  576. if (loc && loc.length === 2 && loc[1] > 0) {
  577. const [offset, length] = loc;
  578. const buffer = Buffer.alloc(length);
  579. await fileHandle.read(buffer, 0, length, offset);
  580. yield deserialize(buffer.subarray(4));
  581. }
  582. }
  583. } finally {
  584. await fileHandle.close();
  585. }
  586. }
  587. }
  588. export default IndexManager;

Branches

Latest commits

  • b8ffc1a0release 1.0.5caramboleyo
  • d47876a1reimplemented lost features like indexed find and more testscaramboleyo
  • 7f08da9afixed insert ignoring model definitioncaramboleyo
  • 705774a9added flush before findcaramboleyo
  • b4db6391initial commitcaramboleyo