gitoriaLog in with ident

mpackdb

All repositories: gitoria

ReadmeCodePull requestsReleasesTicketsSettings
Commitd01dda02d01dda02add index hints, intersection, boundingBox; remove findByIndexcaramboleyod01dda02/src/MPackDB.js

27.0 KB

  1. import { dirname, basename, extname, resolve } from 'path';
  2. import { createReadStream, createWriteStream } from 'fs';
  3. import { mkdir, open, stat, readFile, writeFile, rename, unlink } from 'fs/promises';
  4. import { serialize, deserialize, uuid, PrimaryKeyType, IndexType } from './mpack.js';
  5. import { Cursor } from './Cursor.js';
  6. import { IndexManager } from './IndexManager.js';
  7. export { PrimaryKeyType, IndexType };
  8. /**
  9. * MPackDB - A fast, local, append-only JSON database with MessagePack serialization
  10. *
  11. * Features:
  12. * - Append-only writes for high performance
  13. * - MessagePack binary serialization
  14. * - Optional indexes (numeric and lexical)
  15. * - File-based locking for concurrent access
  16. * - Auto-compaction on startup
  17. * - Auto-persistence of indexes
  18. */
  19. export class MPackDB {
  20. _dbFile = null;
  21. _primaryKeyType = null;
  22. _primaryKey = null;
  23. _classToUse = null;
  24. _indexes = [];
  25. _indexTypes = {};
  26. _uniqueIndexes = new Set();
  27. _initPromise = null;
  28. _initState = 0;
  29. _dataPath = null;
  30. _dataStream = null;
  31. _lockPath = null;
  32. _lockDepth = 0;
  33. _indexManager = null;
  34. _meta = {
  35. nextId: 0,
  36. deleted: [],
  37. };
  38. _debug = false;
  39. _indexPersistInterval = 60000; // 60 seconds default
  40. _indexPersistThreshold = 1000; // 1000 entries default
  41. _processExitHandler = null;
  42. /**
  43. * Create a new MPackDB instance
  44. *
  45. * @param {string} dbFile - Path to the database file (without extension)
  46. * @param {Object} options - Configuration options
  47. * @param {string} [options.primaryKey] - Primary key field name. Prefix with * for numeric (e.g., '*id'), @ for UUID (e.g., '@uuid'), or no prefix for string
  48. * @param {PrimaryKeyType} [options.primaryKeyType] - Explicit primary key type (overrides prefix)
  49. * @param {string[]} [options.indexes] - Array of field names to index. Use * prefix for numeric, @ for UUID
  50. * @param {boolean} [options.debug=false] - Enable debug logging
  51. * @param {number} [options.indexPersistInterval=60000] - Milliseconds between automatic index persistence (0 to disable)
  52. * @param {number} [options.indexPersistThreshold=1000] - Number of changes before auto-persisting indexes
  53. * @param {boolean} [options.compact=true] - Run compaction on init (set false for read-only / secondary instances)
  54. *
  55. * @example
  56. * const db = new MPackDB('data/users', {
  57. * primaryKey: '*id', // Numeric auto-increment
  58. * indexes: ['email', '*age'], // Index email (lexical) and age (numeric)
  59. * indexPersistThreshold: 100
  60. * });
  61. */
  62. _compact = true;
  63. constructor(dbFile, { primaryKey, primaryKeyType, indexes, debug, indexPersistInterval, indexPersistThreshold, compact } = {}) {
  64. this._dbFile = dbFile || this._dbFile;
  65. this._debug = debug || this._debug;
  66. if (compact !== undefined) this._compact = compact;
  67. if (indexPersistInterval !== undefined) this._indexPersistInterval = indexPersistInterval;
  68. if (indexPersistThreshold !== undefined) this._indexPersistThreshold = indexPersistThreshold;
  69. // Parse primary key with optional prefix
  70. if (primaryKey) {
  71. if (primaryKey.startsWith('*')) {
  72. // *id = numeric primary key
  73. this._primaryKey = primaryKey.slice(1);
  74. this._primaryKeyType = PrimaryKeyType.NUMBER;
  75. } else if (primaryKey.startsWith('@')) {
  76. // @id = UUID primary key
  77. this._primaryKey = primaryKey.slice(1);
  78. this._primaryKeyType = PrimaryKeyType.UUID;
  79. } else {
  80. // id = string primary key (lexical)
  81. this._primaryKey = primaryKey;
  82. this._primaryKeyType = PrimaryKeyType.STRING;
  83. }
  84. // Allow explicit override
  85. if (primaryKeyType !== undefined) {
  86. this._primaryKeyType = primaryKeyType;
  87. }
  88. }
  89. // Primary key is always unique
  90. if (this._primaryKey) {
  91. this._uniqueIndexes.add(this._primaryKey);
  92. }
  93. const allIndexes = [...new Set([
  94. ...(this._primaryKey ? [this._primaryKey] : []),
  95. ...(indexes || []),
  96. ...(this._indexes || []),
  97. ])];
  98. this._indexes = [];
  99. for (let index of allIndexes) {
  100. let cleanIndex = index;
  101. // !field = unique index (can combine with * and @: !*field, !@field)
  102. let isUnique = false;
  103. if (cleanIndex.startsWith('!')) {
  104. isUnique = true;
  105. cleanIndex = cleanIndex.slice(1);
  106. }
  107. if (cleanIndex.startsWith('*')) {
  108. // *field = numeric index
  109. cleanIndex = cleanIndex.slice(1);
  110. this._indexTypes[cleanIndex] = IndexType.NUMERIC;
  111. } else if (cleanIndex.startsWith('@')) {
  112. // @field = UUID index (lexical)
  113. cleanIndex = cleanIndex.slice(1);
  114. this._indexTypes[cleanIndex] = IndexType.LEXICAL;
  115. } else if (cleanIndex === this._primaryKey && this._primaryKeyType === PrimaryKeyType.NUMBER) {
  116. // Primary key is numeric
  117. this._indexTypes[cleanIndex] = IndexType.NUMERIC;
  118. } else {
  119. // Default to lexical
  120. this._indexTypes[cleanIndex] = IndexType.LEXICAL;
  121. }
  122. if (isUnique) {
  123. this._uniqueIndexes.add(cleanIndex);
  124. }
  125. this._indexes.push(cleanIndex);
  126. }
  127. }
  128. /**
  129. * Initialize the database (called automatically by other methods)
  130. * Performs compaction and loads metadata
  131. *
  132. * @returns {Promise<MPackDB>} The database instance
  133. */
  134. async init() {
  135. if (this._initState === 2) return this;
  136. if (this._initState === 1) {
  137. return this._initPromise;
  138. }
  139. this._initState = 1;
  140. return this._initPromise = new Promise(async success => {
  141. if (!this._dbFile) {
  142. throw new Error('No database file specified.');
  143. }
  144. const dbDir = dirname(this._dbFile);
  145. const baseName = basename(this._dbFile, extname(this._dbFile));
  146. await mkdir(dbDir, { recursive: true });
  147. this._dataPath = resolve(dbDir, `${baseName}.mpack`);
  148. this._metaPath = resolve(dbDir, `${baseName}.meta.json`);
  149. this._lockPath = resolve(dbDir, `${baseName}.lock`);
  150. try {
  151. this._meta = JSON.parse(await readFile(this._metaPath));
  152. } catch (e) {
  153. this._meta = {
  154. nextId: 0,
  155. deleted: [],
  156. };
  157. }
  158. this._initState = 2; // compact causes init to run again thats why we set it to 2 here already
  159. if (this._compact) await this.compact();
  160. this._dataStream = createWriteStream(this._dataPath, { flags: 'a' }); // needs to be set after compact as it replaces the file with a tmp file
  161. // Initialize IndexManager if indexes are specified
  162. if (this._indexes.length > 0) {
  163. const dbDir = dirname(this._dbFile);
  164. const baseName = basename(this._dbFile, extname(this._dbFile));
  165. this._indexManager = new IndexManager(dbDir, baseName, this._indexes, this._indexTypes, this._primaryKeyType);
  166. await this._indexManager.init(this._dataPath);
  167. // Start auto-persist with configured interval and threshold
  168. this._indexManager.startAutoPersist(this._indexPersistInterval, this._indexPersistThreshold);
  169. // Register signal handlers for Ctrl+C and kill signals
  170. this._processExitHandler = async () => {
  171. await this.close();
  172. process.exit(0);
  173. };
  174. process.once('SIGINT', this._processExitHandler);
  175. process.once('SIGTERM', this._processExitHandler);
  176. }
  177. this._initPromise = null;
  178. success(this);
  179. });
  180. }
  181. /**
  182. * Insert a new record into the database
  183. *
  184. * @param {Object} record - The record to insert
  185. * @param {Object} [options]
  186. * @param {boolean} [options.skipPrimaryKey=false] - Skip auto-generating primary key
  187. * @returns {Promise<Object>} The inserted record with primary key
  188. *
  189. * @example
  190. * await db.insert({ name: 'Alice', age: 30 });
  191. * // Returns: { id: 0, name: 'Alice', age: 30 }
  192. */
  193. async insert(record, { skipPrimaryKey = false } = {}) {
  194. await this.init();
  195. await this._acquireLock('insert', record);
  196. try {
  197. const recToInsert = this._classToUse
  198. ? Object.assign(new this._classToUse(), record)
  199. : { ...record };
  200. if (this._primaryKey && !skipPrimaryKey && !recToInsert[this._primaryKey]) {
  201. if (this._primaryKeyType === PrimaryKeyType.NUMBER) {
  202. recToInsert[this._primaryKey] = this._meta.nextId++;
  203. } else if (this._primaryKeyType === PrimaryKeyType.UUID) {
  204. recToInsert[this._primaryKey] = uuid();
  205. }
  206. }
  207. // Check unique index constraints (skip auto-generated primary keys — guaranteed unique)
  208. if (this._indexManager && this._uniqueIndexes.size > 0) {
  209. const autoGenPK = this._primaryKey && !skipPrimaryKey && !record[this._primaryKey]
  210. && (this._primaryKeyType === PrimaryKeyType.NUMBER || this._primaryKeyType === PrimaryKeyType.UUID);
  211. const deletedSet = new Set(this._meta.deleted);
  212. for (const field of this._uniqueIndexes) {
  213. if (autoGenPK && field === this._primaryKey) continue;
  214. const value = recToInsert[field];
  215. if (value === undefined) continue;
  216. const existing = await this._indexManager.get(field, value);
  217. const live = existing.filter(e => !deletedSet.has(e.loc[0]));
  218. if (live.length > 0) {
  219. const err = new Error(`Duplicate key: ${field}=${value}`);
  220. err.code = 'DUPLICATE_KEY';
  221. err.field = field;
  222. err.value = value;
  223. throw err;
  224. }
  225. }
  226. }
  227. const packedBuffer = serialize(recToInsert);
  228. const fileStat = await stat(this._dataPath).catch(() => ({ size: 0 }));
  229. const offset = fileStat.size;
  230. await new Promise(resolve => this._dataStream.write(packedBuffer, resolve));
  231. const loc = [offset, packedBuffer.length];
  232. // indexes
  233. if (this._indexManager) {
  234. this._indexManager.insert(recToInsert, loc);
  235. }
  236. // Persist meta if we incremented nextId
  237. if (this._primaryKey && this._primaryKeyType === PrimaryKeyType.NUMBER && !skipPrimaryKey && !record[this._primaryKey]) {
  238. await this.persistMeta();
  239. }
  240. return this._primaryKey ? recToInsert[this._primaryKey] : recToInsert;
  241. } finally {
  242. await this._releaseLock();
  243. }
  244. }
  245. /**
  246. * Update records matching a query
  247. *
  248. * @param {string|Function} mixed - Primary key value or query function
  249. * @param {Object|Function} dataOrCallback - Data to update or callback function
  250. * @param {Object} [options]
  251. * @param {boolean} [options.upsert=false] - Insert if no records match
  252. * @param {Object|Array} [options.index] - Index hint(s) to narrow disk reads
  253. * @returns {Promise<Object[]>} Array of updated records
  254. *
  255. * @example
  256. * // Update by primary key
  257. * await db.update(0, { age: 31 });
  258. *
  259. * // Update with query function
  260. * await db.update(r => r.age > 30, { status: 'senior' });
  261. *
  262. * // Update with callback
  263. * await db.update(r => r.age > 30, r => ({ ...r, age: r.age + 1 }));
  264. */
  265. async update(mixed, dataOrCallback, { upsert = false, index } = {}) {
  266. let insertedRecords = await this.delete(mixed, {
  267. index,
  268. callback: async record => {
  269. return this.insert(typeof dataOrCallback === 'function'
  270. ? await dataOrCallback(record)
  271. : dataOrCallback,
  272. { skipPrimaryKey: true }
  273. );
  274. }
  275. });
  276. if (upsert && insertedRecords.length === 0) {
  277. insertedRecords = await this.insert(typeof dataOrCallback === 'function'
  278. ? await dataOrCallback({})
  279. : dataOrCallback);
  280. }
  281. return insertedRecords;
  282. }
  283. /**
  284. * Update records or insert if not found
  285. *
  286. * @param {string|Function} mixed - Primary key value or query function
  287. * @param {Object|Function} dataOrCallback - Data to update/insert or callback function
  288. * @param {Object} [options]
  289. * @param {Object|Array} [options.index] - Index hint(s) to narrow disk reads
  290. * @returns {Promise<Object[]>} Array of updated/inserted records
  291. */
  292. async upsert(mixed, dataOrCallback, { index } = {}) {
  293. return this.update(mixed, dataOrCallback, { upsert: true, index });
  294. }
  295. /**
  296. * Delete records matching a query
  297. *
  298. * @param {string|Function} mixed - Primary key value or query function
  299. * @param {Object} [options]
  300. * @param {Object|Array} [options.index] - Index hint(s) to narrow disk reads
  301. * @param {Function} [options.callback] - Internal callback per deleted record (used by update)
  302. * @returns {Promise<Object[]>} Array of deleted records
  303. *
  304. * @example
  305. * // Delete by primary key
  306. * await db.delete(0);
  307. *
  308. * // Delete with query function
  309. * await db.delete(r => r.age < 18);
  310. *
  311. * // Delete with index hint
  312. * await db.delete(r => r.status === 'inactive', { index: { field: 'status', value: 'inactive' } });
  313. */
  314. async delete(mixed, { index, callback = async record => record } = {}) {
  315. await this.init();
  316. await this._acquireLock('delete', mixed);
  317. try {
  318. const promises = [];
  319. for await (const [record, offset] of this.find(mixed, { mode: 'mixed', index })) {
  320. this._meta.deleted.push(offset);
  321. if (this._indexManager) {
  322. await this._indexManager.remove(record, this._primaryKey);
  323. }
  324. promises.push(callback(record));
  325. }
  326. await this.persistMeta();
  327. return Promise.all(promises);
  328. } finally {
  329. await this._releaseLock();
  330. }
  331. }
  332. /**
  333. * Finds records in the database
  334. * @param {undefined|string|number|function} [mixed] - Query: undefined/null for all, primary key value for PK lookup, function for filter
  335. * @param {Object} [options] - Query options
  336. * @param {Object|Array} [options.index] - Index hint(s) to narrow disk reads before filtering
  337. * @returns {Cursor} A cursor for iterating over results
  338. * @example
  339. * // All records
  340. * for await (const user of db.find()) { ... }
  341. *
  342. * // Primary key lookup
  343. * const [user] = await db.find(68);
  344. *
  345. * // Filter with index range — only reads records in x 100+
  346. * for await (const doc of db.find(r => r.x <= 200, { index: { field: 'x', from: 100 } })) { ... }
  347. *
  348. * // Index intersection — intersects offsets first, then streams matches
  349. * for await (const doc of db.find(r => r.x <= 200, {
  350. * index: [
  351. * { field: 'x', from: 100 },
  352. * { field: 'status', value: 'active' }
  353. * ]
  354. * })) { ... }
  355. */
  356. find(mixed, options = {}) {
  357. if (typeof mixed === 'undefined' || mixed === null || typeof mixed === 'function') {
  358. return new Cursor(this, mixed, options);
  359. } else {
  360. if (!this._primaryKey) {
  361. throw new Error('No primary key specified.');
  362. }
  363. // Use indexed lookup when available
  364. if (this._indexManager) {
  365. return new Cursor(this, null, {
  366. ...options,
  367. _indexLookup: { field: this._primaryKey, value: mixed }
  368. });
  369. }
  370. return new Cursor(this, record => {
  371. return record[this._primaryKey] === mixed;
  372. }, options);
  373. }
  374. }
  375. /**
  376. * Find records within a bounding box defined by 4 corners.
  377. * Requires numeric indexes on x and y fields.
  378. * Corners can be in any order — min/max are extracted automatically.
  379. *
  380. * @param {Array<{x: number, y: number}>} corners - 4 corner coordinates
  381. * @param {function} [filter] - Optional additional filter function
  382. * @returns {Cursor}
  383. * @example
  384. * const results = await db.boundingBox([
  385. * { x: -5, y: 5 }, { x: 5, y: 5 },
  386. * { x: 5, y: -5 }, { x: -5, y: -5 }
  387. * ]);
  388. */
  389. boundingBox(corners, filter) {
  390. const xs = corners.map(c => c.x);
  391. const ys = corners.map(c => c.y);
  392. const minX = Math.min(...xs), maxX = Math.max(...xs);
  393. const minY = Math.min(...ys), maxY = Math.max(...ys);
  394. return this.find(filter || null, {
  395. index: [
  396. { field: 'x', from: minX, to: maxX },
  397. { field: 'y', from: minY, to: maxY }
  398. ]
  399. });
  400. }
  401. /**
  402. * Execute a callback while holding the database lock.
  403. * The lock is re-entrant: find/insert/delete/update called inside
  404. * the callback reuse the same lock instead of deadlocking.
  405. *
  406. * Use this for compound operations that must be atomic, e.g.
  407. * find-then-insert (login pattern).
  408. *
  409. * @param {Function} callback - Async function to execute under lock
  410. * @returns {Promise<any>} The return value of the callback
  411. *
  412. * @example
  413. * const user = await db.withLock(async () => {
  414. * const [existing] = await db.find(u => u.email === email);
  415. * if (existing) return existing;
  416. * return db.insert({ email });
  417. * });
  418. */
  419. async withLock(callback) {
  420. await this.init();
  421. await this._acquireLock('withLock');
  422. try {
  423. return await callback();
  424. } finally {
  425. await this._releaseLock();
  426. }
  427. }
  428. /**
  429. * Generator that yields records from the database file
  430. * @param {function|null} [queryFn=null] - Optional filter function
  431. * @param {Object} [options] - Generator options
  432. * @param {string} [options.mode='record'] - Mode: 'record', 'raw', 'offset', or 'mixed'
  433. * @yields {Object|Buffer|Array} Records, buffers, or [record, offset, size] tuples depending on mode
  434. */
  435. async *recordGenerator(queryFn = null, { mode = 'record', _indexLookup, index } = {}) {
  436. await this.init();
  437. await this.refresh();
  438. // Indexed primary key lookup — O(log n) instead of full scan
  439. if (_indexLookup && this._indexManager) {
  440. yield* this._indexedLookup(_indexLookup.field, _indexLookup.value, mode);
  441. return;
  442. }
  443. // Index-based find: collect offsets from index(es), then stream only those records
  444. if (index && this._indexManager) {
  445. yield* this._indexedStream(index, queryFn, mode);
  446. return;
  447. }
  448. // Flush pending writes so reads see all inserted data
  449. if (this._dataStream && this._dataStream.writableLength > 0) {
  450. await new Promise(resolve => this._dataStream.once('drain', resolve));
  451. }
  452. // Check if the data file exists before attempting to read it
  453. try {
  454. await stat(this._dataPath);
  455. } catch (error) {
  456. // File doesn't exist - return empty generator (no records)
  457. return;
  458. }
  459. // Convert deleted array to Set for O(1) lookup instead of O(n)
  460. const deletedSet = new Set(this._meta.deleted);
  461. const readStream = createReadStream(this._dataPath);
  462. let chunks = [];
  463. let totalLength = 0;
  464. let processedBytes = 0;
  465. for await (const chunk of readStream) {
  466. chunks.push(chunk);
  467. totalLength += chunk.length;
  468. while (true) {
  469. // Exit 1: Not enough data to even read the 4-byte size header.
  470. if (totalLength < 4) {
  471. break;
  472. }
  473. // Safely read the header, even if it's split across chunks
  474. let headerBuffer;
  475. if (chunks[0].length >= 4) {
  476. headerBuffer = chunks[0];
  477. } else {
  478. // The header is fragmented, so we must concat just enough to read it.
  479. headerBuffer = Buffer.concat(chunks, 4);
  480. }
  481. const recSize = headerBuffer.readInt32LE(0);
  482. // Validate the record size to prevent infinite loops
  483. // A record must be at least as large as its header (4 bytes).
  484. // A size of 0 or less is invalid and indicates corruption.
  485. if (recSize <= 4) {
  486. throw new Error(`Invalid record size read from stream: ${recSize}`);
  487. }
  488. // Exit 2: We have the size, but not the full record yet.
  489. if (totalLength < recSize) {
  490. break;
  491. }
  492. // skip deleted records (only after we have the full record)
  493. if (deletedSet.has(processedBytes)) {
  494. [processedBytes, totalLength] = removeChunk(recSize, processedBytes, totalLength, chunks);
  495. continue; // Goes back to while (true)
  496. }
  497. switch (mode) {
  498. case 'raw':
  499. yield Buffer.concat(chunks, recSize).subarray(0, recSize);
  500. break;
  501. case 'mixed':
  502. case 'record':
  503. const recBuffer = Buffer.concat(chunks, recSize).subarray(0, recSize);
  504. // Skip the 4-byte size header
  505. const data = deserialize(recBuffer.subarray(4));
  506. const rec = this._classToUse
  507. ? Object.assign(new this._classToUse(), data)
  508. : data;
  509. if (queryFn) {
  510. if (queryFn(rec)) {
  511. yield mode === 'mixed' ? [rec, processedBytes, recSize] : rec;
  512. }
  513. } else {
  514. yield mode === 'mixed' ? [rec, processedBytes, recSize] : rec;
  515. }
  516. break;
  517. case 'offset':
  518. // if someone uses a queryFn with offset, we have to deserialize it first to perform the queryFn
  519. if (queryFn) {
  520. const recBuffer = Buffer.concat(chunks, recSize).subarray(0, recSize);
  521. const rec = deserialize(recBuffer.subarray(4));
  522. if (queryFn(rec)) {
  523. yield [processedBytes, recSize];
  524. }
  525. } else {
  526. yield [processedBytes, recSize];
  527. }
  528. break;
  529. default:
  530. throw new Error(`Invalid mode: ${mode}`);
  531. }
  532. [processedBytes, totalLength] = removeChunk(recSize, processedBytes, totalLength, chunks);
  533. }
  534. }
  535. }
  536. /**
  537. * Indexed lookup — reads records directly by offset from the index.
  538. * Uses binary search on the index file for O(log n) lookups.
  539. * @private
  540. */
  541. async *_indexedLookup(field, value, mode = 'record') {
  542. const entries = await this._indexManager.get(field, value);
  543. if (entries.length === 0) return;
  544. const deletedSet = new Set(this._meta.deleted);
  545. const fileHandle = await open(this._dataPath, 'r');
  546. try {
  547. for (const entry of entries) {
  548. const [offset, length] = entry.loc;
  549. if (deletedSet.has(offset)) continue;
  550. const buffer = Buffer.alloc(length);
  551. await fileHandle.read(buffer, 0, length, offset);
  552. switch (mode) {
  553. case 'raw':
  554. yield buffer;
  555. break;
  556. case 'mixed': {
  557. const data = deserialize(buffer.subarray(4));
  558. const rec = this._classToUse
  559. ? Object.assign(new this._classToUse(), data)
  560. : data;
  561. yield [rec, offset, length];
  562. break;
  563. }
  564. case 'record':
  565. default: {
  566. const data = deserialize(buffer.subarray(4));
  567. const rec = this._classToUse
  568. ? Object.assign(new this._classToUse(), data)
  569. : data;
  570. yield rec;
  571. break;
  572. }
  573. }
  574. }
  575. } finally {
  576. await fileHandle.close();
  577. }
  578. }
  579. /**
  580. * Collect offsets from one or more indexes, optionally intersect, then stream records.
  581. * @param {Object|Array} index - Single index hint or array of hints
  582. * @param {function|null} queryFn - Optional filter function
  583. * @param {string} mode - Output mode: 'record', 'raw', or 'mixed'
  584. * @private
  585. */
  586. async *_indexedStream(index, queryFn, mode = 'record') {
  587. const hints = Array.isArray(index) ? index : [index];
  588. const deletedSet = new Set(this._meta.deleted);
  589. // Collect offset sets from each index hint
  590. const offsetSets = [];
  591. for (const hint of hints) {
  592. const offsets = new Map(); // offset → [offset, length]
  593. if (hint.value !== undefined) {
  594. // Exact match via get()
  595. const entries = await this._indexManager.get(hint.field, hint.value);
  596. for (const e of entries) {
  597. if (!deletedSet.has(e.loc[0])) offsets.set(e.loc[0], e.loc);
  598. }
  599. } else {
  600. // Range scan via entries()
  601. for await (const e of this._indexManager.entries(hint.field, { from: hint.from, to: hint.to, direction: hint.direction })) {
  602. if (!deletedSet.has(e.loc[0])) offsets.set(e.loc[0], e.loc);
  603. }
  604. }
  605. offsetSets.push(offsets);
  606. }
  607. // Intersect: keep only offsets present in ALL sets
  608. let locations;
  609. if (offsetSets.length === 1) {
  610. locations = Array.from(offsetSets[0].values());
  611. } else {
  612. // Start with smallest set for efficiency
  613. offsetSets.sort((a, b) => a.size - b.size);
  614. const [smallest, ...rest] = offsetSets;
  615. locations = [];
  616. for (const [offset, loc] of smallest) {
  617. if (rest.every(s => s.has(offset))) locations.push(loc);
  618. }
  619. }
  620. // Stream records from the intersected locations
  621. const fileHandle = await open(this._dataPath, 'r');
  622. try {
  623. for (const [offset, length] of locations) {
  624. const buffer = Buffer.alloc(length);
  625. await fileHandle.read(buffer, 0, length, offset);
  626. if (mode === 'raw') {
  627. yield buffer;
  628. } else {
  629. const data = deserialize(buffer.subarray(4));
  630. const rec = this._classToUse
  631. ? Object.assign(new this._classToUse(), data)
  632. : data;
  633. if (queryFn && !queryFn(rec)) continue;
  634. if (mode === 'mixed') {
  635. yield [rec, offset, length];
  636. } else {
  637. yield rec;
  638. }
  639. }
  640. }
  641. } finally {
  642. await fileHandle.close();
  643. }
  644. }
  645. /**
  646. * Re-read meta.json from disk so this instance sees changes made by other processes.
  647. * Called automatically before every read operation.
  648. */
  649. async refresh() {
  650. try {
  651. this._meta = JSON.parse(await readFile(this._metaPath));
  652. } catch (e) {
  653. // file missing or corrupt — keep current meta
  654. }
  655. }
  656. async persistMeta() {
  657. const metaToSave = { ...this._meta };
  658. // Only save nextId if we have a numeric primary key
  659. if (this._primaryKeyType !== PrimaryKeyType.NUMBER) {
  660. delete metaToSave.nextId;
  661. }
  662. await writeFile(this._metaPath, JSON.stringify(metaToSave));
  663. }
  664. /**
  665. * Compact the database by removing deleted records
  666. * This rewrites the data file without tombstones
  667. *
  668. * @returns {Promise<void>}
  669. */
  670. async compact() {
  671. await this._acquireLock('compact');
  672. try {
  673. const writeStream = createWriteStream(this._dataPath + '.tmp', { flags: 'w' });
  674. for await (const binary of this.find(null, { mode: 'raw' })) {
  675. writeStream.write(binary);
  676. }
  677. await new Promise((resolve, reject) => {
  678. writeStream.end((err) => err ? reject(err) : resolve());
  679. });
  680. // rename is atomic, so we can just rename the file and it will replace the old one
  681. await rename(this._dataPath + '.tmp', this._dataPath);
  682. this._meta.deleted = [];
  683. await this.persistMeta();
  684. // Rebuild indexes — offsets changed after compaction
  685. if (this._indexManager) {
  686. await this._indexManager.init(this._dataPath, { forceRebuild: true });
  687. }
  688. } finally {
  689. await this._releaseLock();
  690. }
  691. }
  692. async _acquireLock(operation, record, retries = 0) {
  693. if (this._lockDepth > 0) {
  694. this._lockDepth++;
  695. return;
  696. }
  697. try {
  698. await writeFile(this._lockPath, String(process.pid), { flag: 'wx' });
  699. this._lockDepth = 1;
  700. } catch (e) {
  701. if (e.code === 'EEXIST') {
  702. this.debug(`Waiting for lock file ${operation} retry #${retries}`, record);
  703. await new Promise(resolve => setTimeout(resolve, 100));
  704. return this._acquireLock(operation, record, retries + 1);
  705. }
  706. throw e;
  707. }
  708. }
  709. async _releaseLock() {
  710. this._lockDepth--;
  711. if (this._lockDepth <= 0) {
  712. this._lockDepth = 0;
  713. await unlink(this._lockPath).catch(() => { });
  714. }
  715. }
  716. /**
  717. * Retrieves a document by its file offset and length
  718. * @param {Array} location - [offset, length] tuple
  719. * @returns {Promise<Object|null>} The deserialized document or null
  720. * @private
  721. * @deprecated Currently unused - may be removed in future versions
  722. */
  723. async _getDocByLocation([offset, length]) {
  724. if (!offset || length === 0) return null;
  725. const fileHandle = await open(this._dataPath, 'r');
  726. try {
  727. const buffer = Buffer.alloc(length);
  728. await fileHandle.read(buffer, 0, length, offset);
  729. return deserialize(buffer);
  730. } finally {
  731. await fileHandle.close();
  732. }
  733. }
  734. /**
  735. * Close the database and persist all pending changes
  736. * Should be called before process exit
  737. *
  738. * @returns {Promise<void>}
  739. */
  740. async close() {
  741. // Persist indexes before closing
  742. if (this._indexManager) {
  743. await this._indexManager.close();
  744. }
  745. // Close data stream
  746. if (this._dataStream) {
  747. await new Promise((resolve, reject) => {
  748. this._dataStream.end((err) => err ? reject(err) : resolve());
  749. });
  750. }
  751. // Remove signal handlers
  752. if (this._processExitHandler) {
  753. process.off('SIGINT', this._processExitHandler);
  754. process.off('SIGTERM', this._processExitHandler);
  755. this._processExitHandler = null;
  756. }
  757. }
  758. debug(...args) {
  759. if (this._debug) {
  760. console.log(...args);
  761. }
  762. }
  763. }
  764. /**
  765. * Helper function to remove a processed chunk from the buffer
  766. * @param {number} recSize - Size of the record to remove
  767. * @param {number} processedBytes - Current offset in the file
  768. * @param {number} totalLength - Total length of buffered data
  769. * @param {Buffer[]} chunks - Array of buffer chunks
  770. * @returns {[number, number]} Updated [processedBytes, totalLength]
  771. */
  772. function removeChunk(recSize, processedBytes, totalLength, chunks) {
  773. processedBytes += recSize;
  774. totalLength -= recSize;
  775. let bytesToRemove = recSize;
  776. while (bytesToRemove > 0 && chunks.length > 0) {
  777. const currentChunk = chunks[0];
  778. if (bytesToRemove >= currentChunk.length) {
  779. bytesToRemove -= currentChunk.length;
  780. chunks.shift();
  781. } else {
  782. chunks[0] = currentChunk.subarray(bytesToRemove);
  783. bytesToRemove = 0;
  784. }
  785. }
  786. return [processedBytes, totalLength];
  787. }
  788. export default MPackDB;

Branches

Latest commits

  • d01dda02add index hints, intersection, boundingBox; remove findByIndexcaramboleyo
  • b8ffc1a0release 1.0.5caramboleyo
  • d47876a1reimplemented lost features like indexed find and more testscaramboleyo
  • 7f08da9afixed insert ignoring model definitioncaramboleyo
  • 705774a9added flush before findcaramboleyo
  • b4db6391initial commitcaramboleyo