mpackdb
All repositories: gitoria
35.5 KB
import { dirname, basename, extname, resolve } from 'node:path';import { createReadStream, createWriteStream } from 'node:fs';import { mkdir, open, stat, readFile, writeFile, rename, unlink } from 'node:fs/promises';import { AsyncLocalStorage } from 'node:async_hooks';import { serialize, deserialize, uuid, PrimaryKeyType, IndexType } from './mpack.js';import { Cursor } from './Cursor.js';import { IndexManager } from './IndexManager.js';// Tracks which async call chain currently owns a db's lock, so nested operations// (e.g. find inside delete, insert inside withLock) re-enter, while unrelated// concurrent operations queue up instead of walking into the critical section.const lockContext = new AsyncLocalStorage();export { PrimaryKeyType, IndexType };/*** MPackDB - A fast, local, append-only JSON database with MessagePack serialization** Features:* - Append-only writes for high performance* - MessagePack binary serialization* - Optional indexes (numeric and lexical)* - File-based locking for concurrent access* - Auto-compaction on startup* - Auto-persistence of indexes*/export class MPackDB {_dbFile = null;_primaryKeyType = null;_primaryKey = null;_classToUse = null;_indexes = [];_indexTypes = {};_uniqueIndexes = new Set();_initPromise = null;_initState = 0;_dataPath = null;_dataStream = null;_dataIno = null;_lastWrite = null;_lockPath = null;_mutexTail = Promise.resolve();_staleLockTimeout = 30000;_schema = null;_indexManager = null;_meta = {nextId: 0,deleted: [],};_debug = false;_indexPersistInterval = 60000; // 60 seconds default_indexPersistThreshold = 1000; // 1000 entries default_processExitHandler = null;/*** Create a new MPackDB instance** @param {string} dbFile - Path to the database file (without extension)* @param {Object} options - Configuration options* @param {string} [options.primaryKey] - Primary key field name. Prefix with * for numeric (e.g., '*id'), @ for UUID (e.g., '@uuid'), or no prefix for string* @param {PrimaryKeyType} [options.primaryKeyType] - Explicit primary key type (overrides prefix)* @param {string[]} [options.indexes] - Array of field names to index. Use * prefix for numeric, @ for UUID* @param {boolean} [options.debug=false] - Enable debug logging* @param {number} [options.indexPersistInterval=60000] - Milliseconds between automatic index persistence (0 to disable)* @param {number} [options.indexPersistThreshold=1000] - Number of changes before auto-persisting indexes* @param {boolean} [options.compact=true] - Run compaction on init (set false whenever another process may have the same files open)* @param {number} [options.staleLockTimeout=30000] - Take over lock files older than this many ms (crashed holder); 0 disables** @example* const db = new MPackDB('data/users', {* primaryKey: '*id', // Numeric auto-increment* indexes: ['email', '*age'], // Index email (lexical) and age (numeric)* indexPersistThreshold: 100* });*/_compact = true;_hasPrimaryKeyValue(record) {if (!this._primaryKey || !record || !Object.prototype.hasOwnProperty.call(record, this._primaryKey)) {return false;}const value = record[this._primaryKey];return value !== undefined && value !== null;}constructor(dbFile, { primaryKey, primaryKeyType, indexes, debug, indexPersistInterval, indexPersistThreshold, compact, staleLockTimeout } = {}) {this._dbFile = dbFile || this._dbFile;this._debug = debug || this._debug;if (compact !== undefined) this._compact = compact;if (indexPersistInterval !== undefined) this._indexPersistInterval = indexPersistInterval;if (indexPersistThreshold !== undefined) this._indexPersistThreshold = indexPersistThreshold;if (staleLockTimeout !== undefined) this._staleLockTimeout = staleLockTimeout;// Parse primary key with optional prefixif (primaryKey) {if (primaryKey.startsWith('*')) {// *id = numeric primary keythis._primaryKey = primaryKey.slice(1);this._primaryKeyType = PrimaryKeyType.NUMBER;} else if (primaryKey.startsWith('@')) {// @id = UUID primary keythis._primaryKey = primaryKey.slice(1);this._primaryKeyType = PrimaryKeyType.UUID;} else {// id = string primary key (lexical)this._primaryKey = primaryKey;this._primaryKeyType = PrimaryKeyType.STRING;}// Allow explicit overrideif (primaryKeyType !== undefined) {this._primaryKeyType = primaryKeyType;}}// Primary key is always uniqueif (this._primaryKey) {this._uniqueIndexes.add(this._primaryKey);}const allIndexes = [...new Set([...(this._primaryKey ? [this._primaryKey] : []),...(indexes || []),...(this._indexes || []),])];this._indexes = [];for (let index of allIndexes) {let cleanIndex = index;// !field = unique index (can combine with * and @: !*field, !@field)let isUnique = false;if (cleanIndex.startsWith('!')) {isUnique = true;cleanIndex = cleanIndex.slice(1);}if (cleanIndex.startsWith('*')) {// *field = numeric indexcleanIndex = cleanIndex.slice(1);this._indexTypes[cleanIndex] = IndexType.NUMERIC;} else if (cleanIndex.startsWith('@')) {// @field = UUID index (lexical)cleanIndex = cleanIndex.slice(1);this._indexTypes[cleanIndex] = IndexType.LEXICAL;} else if (cleanIndex === this._primaryKey && this._primaryKeyType === PrimaryKeyType.NUMBER) {// Primary key is numericthis._indexTypes[cleanIndex] = IndexType.NUMERIC;} else {// Default to lexicalthis._indexTypes[cleanIndex] = IndexType.LEXICAL;}if (isUnique) {this._uniqueIndexes.add(cleanIndex);}this._indexes.push(cleanIndex);}// Schema descriptor persisted into meta.json so other tools (e.g. mpackdb-admin)// can open this database with the correct primary key and indexes.if (this._primaryKey || this._indexes.length > 0) {const pkPrefix = this._primaryKeyType === PrimaryKeyType.NUMBER ? '*': this._primaryKeyType === PrimaryKeyType.UUID ? '@' : '';this._schema = {primaryKey: this._primaryKey ? pkPrefix + this._primaryKey : undefined,indexes: this._indexes.filter(field => field !== this._primaryKey).map(field =>(this._uniqueIndexes.has(field) ? '!' : '')+ (this._indexTypes[field] === IndexType.NUMERIC ? '*' : '')+ field),};}}/*** Initialize the database (called automatically by other methods)* Performs compaction and loads metadata** @returns {Promise<MPackDB>} The database instance*/async init() {if (this._initState === 2) return this;if (this._initState === 1) {return this._initPromise;}this._initState = 1;return this._initPromise = new Promise(async success => {if (!this._dbFile) {throw new Error('No database file specified.');}const dbDir = dirname(this._dbFile);const baseName = basename(this._dbFile, extname(this._dbFile));await mkdir(dbDir, { recursive: true });this._dataPath = resolve(dbDir, `${baseName}.mpack`);this._metaPath = resolve(dbDir, `${baseName}.meta.json`);this._lockPath = resolve(dbDir, `${baseName}.lock`);try {this._meta = JSON.parse(await readFile(this._metaPath));} catch (e) {this._meta = {nextId: 0,deleted: [],};}const didCompact = this._compact;if (didCompact) await this.compact({ duringInit: true });this._dataStream = createWriteStream(this._dataPath, { flags: 'a' }); // needs to be set after compact as it replaces the file with a tmp filethis._dataIno = (await stat(this._dataPath).catch(() => null))?.ino ?? null;// Persist schema into meta.json so other tools (mpackdb-admin) can// open this db with the right primary key / indexes. The compare// runs under the lock: refresh() has adopted any newer on-disk meta// by then, so a concurrent process' meta is never clobbered.if (this._schema) {await this._withFileLock('init-schema', async () => {if (JSON.stringify(this._meta.schema) !== JSON.stringify(this._schema)) {await this.persistMeta();}});}// Initialize IndexManager if indexes are specifiedif (this._indexes.length > 0) {const dbDir = dirname(this._dbFile);const baseName = basename(this._dbFile, extname(this._dbFile));this._indexManager = new IndexManager(dbDir, baseName, this._indexes, this._indexTypes, this._primaryKeyType, {// Index persistence rewrites shared files — must hold the db lock.// lockContext.exit: a threshold-triggered persist inside insert()// must NOT inherit insert's lock ownership (it outlives it), so it// queues for its own turn instead.runExclusive: fn => lockContext.exit(() => this._withFileLock('index-persist', fn)),});// Index files are shared with other processes — (re)build them under the lockawait this._withFileLock('index-init', () =>this._indexManager.init(this._dataPath, { forceRebuild: didCompact }));// Start auto-persist with configured interval and thresholdthis._indexManager.startAutoPersist(this._indexPersistInterval, this._indexPersistThreshold);// Register signal handlers for Ctrl+C and kill signalsthis._processExitHandler = async () => {await this.close();process.exit(0);};process.once('SIGINT', this._processExitHandler);process.once('SIGTERM', this._processExitHandler);}this._initState = 2;this._initPromise = null;success(this);});}/*** Insert a new record into the database** @param {Object} record - The record to insert* @param {Object} [options]* @param {boolean} [options.skipPrimaryKey=false] - Skip auto-generating primary key* @returns {Promise<string|number|Object>} The primary-key value when configured, otherwise the inserted record** @example* const id = await db.insert({ name: 'Alice', age: 30 });* // Returns: 0 (the generated primary-key value)*/async insert(record, { skipPrimaryKey = false } = {}) {await this.init();return this._withFileLock('insert', async () => {const recToInsert = this._classToUse? Object.assign(new this._classToUse(), record): { ...record };const hasPrimaryKeyValue = this._hasPrimaryKeyValue(recToInsert);if (this._primaryKey && !skipPrimaryKey && !hasPrimaryKeyValue) {if (this._primaryKeyType === PrimaryKeyType.NUMBER) {recToInsert[this._primaryKey] = this._meta.nextId++;} else if (this._primaryKeyType === PrimaryKeyType.UUID) {recToInsert[this._primaryKey] = uuid();}}// Check unique index constraints (skip auto-generated primary keys — guaranteed unique)if (this._indexManager && this._uniqueIndexes.size > 0) {const autoGenPK = this._primaryKey && !skipPrimaryKey && !hasPrimaryKeyValue&& (this._primaryKeyType === PrimaryKeyType.NUMBER || this._primaryKeyType === PrimaryKeyType.UUID);const deletedSet = new Set(this._meta.deleted);for (const field of this._uniqueIndexes) {if (autoGenPK && field === this._primaryKey) continue;const value = recToInsert[field];if (value === undefined) continue;const existing = await this._indexManager.get(field, value);const live = existing.filter(e => !deletedSet.has(e.loc[0]));if (live.length > 0) {const err = new Error(`Duplicate key: ${field}=${value}`);err.code = 'DUPLICATE_KEY';err.field = field;err.value = value;throw err;}}}const packedBuffer = serialize(recToInsert);const fileStat = await stat(this._dataPath).catch(() => ({ size: 0 }));const offset = fileStat.size;const writePromise = new Promise(resolve => this._dataStream.write(packedBuffer, resolve));this._lastWrite = writePromise;await writePromise;const loc = [offset, packedBuffer.length];// indexesif (this._indexManager) {this._indexManager.insert(recToInsert, loc);}// Persist meta if we incremented nextIdif (this._primaryKey && this._primaryKeyType === PrimaryKeyType.NUMBER && !skipPrimaryKey && !hasPrimaryKeyValue) {await this.persistMeta();}return this._primaryKey ? recToInsert[this._primaryKey] : recToInsert;});}/*** Update records matching a query.* The callback receives the old record and must return the new record.** @param {string|number|Function} mixed - Primary key value or query function* @param {Function} callback - Receives old record, must return new record* @param {Object} [options]* @param {boolean} [options.upsert=false] - Insert if no records match* @param {Object|Array} [options.index] - Index hint(s) to narrow disk reads* @returns {Promise<Object[]>} Array of updated records** @example* await db.update(0, record => {* record.age = 32;* delete record.address;* return record;* });*/async update(mixed, callback, { upsert = false, index } = {}) {if (typeof callback !== 'function') {throw new Error('update requires a callback function as second argument');}let insertedRecords = await this.delete(mixed, {index,callback: async record => {return this.insert(await callback(record), { skipPrimaryKey: true });}});if (upsert && insertedRecords.length === 0) {insertedRecords = await this.insert(await callback({}));}return insertedRecords;}/*** Update records or insert if not found** @param {string|number|Function} mixed - Primary key value or query function* @param {Function} callback - Receives old record (or {} if inserting), must return new record* @param {Object} [options]* @param {Object|Array} [options.index] - Index hint(s) to narrow disk reads* @returns {Promise<Object[]>} Array of updated/inserted records*/async upsert(mixed, callback, { index } = {}) {return this.update(mixed, callback, { upsert: true, index });}/*** Delete records matching a query** @param {string|Function} mixed - Primary key value or query function* @param {Object} [options]* @param {Object|Array} [options.index] - Index hint(s) to narrow disk reads* @param {Function} [options.callback] - Internal callback per deleted record (used by update)* @returns {Promise<Object[]>} Array of deleted records** @example* // Delete by primary key* await db.delete(0);** // Delete with query function* await db.delete(r => r.age < 18);** // Delete with index hint* await db.delete(r => r.status === 'inactive', { index: { field: 'status', value: 'inactive' } });*/async delete(mixed, { index, callback = async record => record } = {}) {await this.init();return this._withFileLock('delete', async () => {const promises = [];for await (const [record, offset] of this.find(mixed, { mode: 'mixed', index })) {this._meta.deleted.push(offset);if (this._indexManager) {await this._indexManager.remove(record, this._primaryKey);}promises.push(callback(record));}await this.persistMeta();return Promise.all(promises);});}/*** Finds records in the database* @param {undefined|string|number|function} [mixed] - Query: undefined/null for all, primary key value for PK lookup, function for filter* @param {Object} [options] - Query options* @param {Object|Array} [options.index] - Index hint(s) to narrow disk reads before filtering* @returns {Cursor} A cursor for iterating over results* @example* // All records* for await (const user of db.find()) { ... }** // Primary key lookup* const [user] = await db.find(68);** // Filter with index range — only reads records in x 100+* for await (const doc of db.find(r => r.x <= 200, { index: { field: 'x', from: 100 } })) { ... }** // Index intersection — intersects offsets first, then streams matches* for await (const doc of db.find(r => r.x <= 200, {* index: [* { field: 'x', from: 100 },* { field: 'status', value: 'active' }* ]* })) { ... }*/find(mixed, options = {}) {if (typeof mixed === 'undefined' || mixed === null || typeof mixed === 'function') {return new Cursor(this, mixed, options);} else {if (!this._primaryKey) {throw new Error('No primary key specified.');}// Use indexed lookup when availableif (this._indexManager) {return new Cursor(this, null, {...options,_indexLookup: { field: this._primaryKey, value: mixed }});}return new Cursor(this, record => {return record[this._primaryKey] === mixed;}, options);}}/*** Find records within a bounding box defined by 4 corners.* Requires numeric indexes on x and y fields.* Corners can be in any order — min/max are extracted automatically.** @param {Array<{x: number, y: number}>} corners - 4 corner coordinates* @param {function} [filter] - Optional additional filter function* @returns {Cursor}* @example* const results = await db.boundingBox([* { x: -5, y: 5 }, { x: 5, y: 5 },* { x: 5, y: -5 }, { x: -5, y: -5 }* ]);*/boundingBox(corners, filter) {const xs = corners.map(c => c.x);const ys = corners.map(c => c.y);const minX = Math.min(...xs), maxX = Math.max(...xs);const minY = Math.min(...ys), maxY = Math.max(...ys);return this.find(filter || null, {index: [{ field: 'x', from: minX, to: maxX },{ field: 'y', from: minY, to: maxY }]});}/*** Execute a callback while holding the database lock.* The lock is re-entrant: find/insert/delete/update called inside* the callback reuse the same lock instead of deadlocking.** Use this for compound operations that must be atomic, e.g.* find-then-insert (login pattern).** @param {Function} callback - Async function to execute under lock* @returns {Promise<any>} The return value of the callback** @example* const user = await db.withLock(async () => {* const [existing] = await db.find(u => u.email === email);* if (existing) return existing;* return db.insert({ email });* });*/async withLock(callback) {await this.init();return this._withFileLock('withLock', callback);}/*** Generator that yields records from the database file* @param {function|null} [queryFn=null] - Optional filter function* @param {Object} [options] - Generator options* @param {string} [options.mode='record'] - Mode: 'record', 'raw', 'offset', or 'mixed'* @yields {Object|Buffer|Array} Records, buffers, or [record, offset, size] tuples depending on mode*/async *recordGenerator(queryFn = null, { mode = 'record', _indexLookup, index } = {}) {await this.init();await this.refresh();// Indexed primary key lookup — O(log n) instead of full scanif (_indexLookup && this._indexManager) {yield* this._indexedLookup(_indexLookup.field, _indexLookup.value, mode);return;}// Index-based find: collect offsets from index(es), then stream only those recordsif (index && this._indexManager) {yield* this._indexedStream(index, queryFn, mode);return;}// Flush pending writes so reads see all inserted data.// (Waiting for 'drain' here could hang forever — 'drain' only fires// after a write() returned false, which small writes never do.)if (this._lastWrite) {await this._lastWrite;}// Check if the data file exists before attempting to read ittry {await stat(this._dataPath);} catch (error) {// File doesn't exist - return empty generator (no records)return;}// Convert deleted array to Set for O(1) lookup instead of O(n)const deletedSet = new Set(this._meta.deleted);const readStream = createReadStream(this._dataPath);let chunks = [];let totalLength = 0;let processedBytes = 0;for await (const chunk of readStream) {chunks.push(chunk);totalLength += chunk.length;while (true) {// Exit 1: Not enough data to even read the 4-byte size header.if (totalLength < 4) {break;}// Safely read the header, even if it's split across chunkslet headerBuffer;if (chunks[0].length >= 4) {headerBuffer = chunks[0];} else {// The header is fragmented, so we must concat just enough to read it.headerBuffer = Buffer.concat(chunks, 4);}const recSize = headerBuffer.readInt32LE(0);// Validate the record size to prevent infinite loops// A record must be at least as large as its header (4 bytes).// A size of 0 or less is invalid and indicates corruption.if (recSize <= 4) {throw new Error(`Invalid record size read from stream: ${recSize}`);}// Exit 2: We have the size, but not the full record yet.if (totalLength < recSize) {break;}// skip deleted records (only after we have the full record)if (deletedSet.has(processedBytes)) {[processedBytes, totalLength] = removeChunk(recSize, processedBytes, totalLength, chunks);continue; // Goes back to while (true)}switch (mode) {case 'raw':yield Buffer.concat(chunks, recSize).subarray(0, recSize);break;case 'mixed':case 'record':const recBuffer = Buffer.concat(chunks, recSize).subarray(0, recSize);// Skip the 4-byte size headerconst data = deserialize(recBuffer.subarray(4));const rec = this._classToUse? Object.assign(new this._classToUse(), data): data;if (queryFn) {if (queryFn(rec)) {yield mode === 'mixed' ? [rec, processedBytes, recSize] : rec;}} else {yield mode === 'mixed' ? [rec, processedBytes, recSize] : rec;}break;case 'offset':// if someone uses a queryFn with offset, we have to deserialize it first to perform the queryFnif (queryFn) {const recBuffer = Buffer.concat(chunks, recSize).subarray(0, recSize);const rec = deserialize(recBuffer.subarray(4));if (queryFn(rec)) {yield [processedBytes, recSize];}} else {yield [processedBytes, recSize];}break;default:throw new Error(`Invalid mode: ${mode}`);}[processedBytes, totalLength] = removeChunk(recSize, processedBytes, totalLength, chunks);}}}/*** Indexed lookup — reads records directly by offset from the index.* Uses binary search on the index file for O(log n) lookups.* @private*/async *_indexedLookup(field, value, mode = 'record') {const entries = await this._indexManager.get(field, value);if (entries.length === 0) return;const deletedSet = new Set(this._meta.deleted);const fileHandle = await open(this._dataPath, 'r');try {for (const entry of entries) {const [offset, length] = entry.loc;if (deletedSet.has(offset)) continue;const buffer = Buffer.alloc(length);await fileHandle.read(buffer, 0, length, offset);switch (mode) {case 'raw':yield buffer;break;case 'mixed': {const data = deserialize(buffer.subarray(4));const rec = this._classToUse? Object.assign(new this._classToUse(), data): data;yield [rec, offset, length];break;}case 'record':default: {const data = deserialize(buffer.subarray(4));const rec = this._classToUse? Object.assign(new this._classToUse(), data): data;yield rec;break;}}}} finally {await fileHandle.close();}}/*** Collect offsets from one or more indexes, optionally intersect, then stream records.* @param {Object|Array} index - Single index hint or array of hints* @param {function|null} queryFn - Optional filter function* @param {string} mode - Output mode: 'record', 'raw', or 'mixed'* @private*/async *_indexedStream(index, queryFn, mode = 'record') {const hints = Array.isArray(index) ? index : [index];const deletedSet = new Set(this._meta.deleted);// Collect offset sets from each index hintconst offsetSets = [];for (const hint of hints) {const offsets = new Map(); // offset → [offset, length]if (hint.value !== undefined) {// Exact match via get()const entries = await this._indexManager.get(hint.field, hint.value);for (const e of entries) {if (!deletedSet.has(e.loc[0])) offsets.set(e.loc[0], e.loc);}} else {// Range scan via entries()for await (const e of this._indexManager.entries(hint.field, { from: hint.from, to: hint.to, direction: hint.direction })) {if (!deletedSet.has(e.loc[0])) offsets.set(e.loc[0], e.loc);}}offsetSets.push(offsets);}// Intersect: keep only offsets present in ALL setslet locations;if (offsetSets.length === 1) {locations = Array.from(offsetSets[0].values());} else {// Start with smallest set for efficiencyoffsetSets.sort((a, b) => a.size - b.size);const [smallest, ...rest] = offsetSets;locations = [];for (const [offset, loc] of smallest) {if (rest.every(s => s.has(offset))) locations.push(loc);}}// Stream records from the intersected locationsconst fileHandle = await open(this._dataPath, 'r');try {for (const [offset, length] of locations) {const buffer = Buffer.alloc(length);await fileHandle.read(buffer, 0, length, offset);if (mode === 'raw') {yield buffer;} else {const data = deserialize(buffer.subarray(4));const rec = this._classToUse? Object.assign(new this._classToUse(), data): data;if (queryFn && !queryFn(rec)) continue;if (mode === 'mixed') {yield [rec, offset, length];} else {yield rec;}}}} finally {await fileHandle.close();}}/*** Pick up changes made by OTHER processes. Called automatically before every* read operation and after acquiring the write lock.** The in-memory meta is authoritative for this process — it may hold* un-persisted mutations of an in-flight write. It is only replaced when the* on-disk copy carries a NEWER version, i.e. another process persisted under* the file lock. (Unconditionally reloading here was the root cause of the* lost-tombstone/lost-insert corruption under concurrent reads + writes.)** Also detects external data file changes: appended records are indexed via* a tail scan, a replaced file (external compaction) triggers a full reopen.*/async refresh() {try {const diskMeta = JSON.parse(await readFile(this._metaPath));if ((diskMeta.version || 0) > (this._meta.version || 0)) {this._meta = { nextId: 0, deleted: [], ...diskMeta };}} catch (e) {// file missing or torn write in progress — keep current meta}const dataStat = await stat(this._dataPath).catch(() => null);if (dataStat) {if (this._dataIno !== null && dataStat.ino !== this._dataIno) {await this._reopenAfterExternalReplace(dataStat);} else if (this._indexManager) {await this._indexManager.catchUp(this._dataPath, dataStat.size);}}}/*** The data file was replaced by another process (external compaction):* all offsets changed and our write stream points at the orphaned inode.* Reopen the stream, force-reload meta, and reset the index state.* @private*/async _reopenAfterExternalReplace(dataStat) {this.debug(`Data file was replaced externally (compaction), reopening ${this._dataPath}`);if (this._dataStream) {await new Promise((resolve, reject) => {this._dataStream.end((err) => err ? reject(err) : resolve());});this._dataStream = createWriteStream(this._dataPath, { flags: 'a' });}this._dataIno = dataStat.ino;try {this._meta = { nextId: 0, deleted: [], ...JSON.parse(await readFile(this._metaPath)) };} catch (e) { }if (this._indexManager) {await this._indexManager.resetAfterReplace(this._dataPath);}}async persistMeta() {// Monotonic version: refresh() in other processes reloads only when it// sees a version newer than its in-memory onethis._meta.version = (this._meta.version || 0) + 1;const metaToSave = { ...this._meta };// Only save nextId if we have a numeric primary keyif (this._primaryKeyType !== PrimaryKeyType.NUMBER) {delete metaToSave.nextId;}if (this._schema) {this._meta.schema = this._schema;metaToSave.schema = this._schema;}// Atomic write (tmp + rename) so concurrent readers never parse a torn fileconst tmpPath = `${this._metaPath}.${process.pid}.tmp`;await writeFile(tmpPath, JSON.stringify(metaToSave));await rename(tmpPath, this._metaPath);}/*** Compact the database by removing deleted records* This rewrites the data file without tombstones** @returns {Promise<void>}*/async compact({ duringInit = false } = {}) {return this._withFileLock('compact', async () => {const writeStream = createWriteStream(this._dataPath + '.tmp', { flags: 'w' });const records = duringInit? this._rawRecordsForCompaction(): this.find(null, { mode: 'raw' });for await (const binary of records) {writeStream.write(binary);}await new Promise((resolve, reject) => {writeStream.end((err) => err ? reject(err) : resolve());});// The old write stream points at the inode that rename() is about to// orphan — writes to it would be silently lost. Close it first.if (this._dataStream) {await new Promise((resolve, reject) => {this._dataStream.end((err) => err ? reject(err) : resolve());});}// rename is atomic, so we can just rename the file and it will replace the old oneawait rename(this._dataPath + '.tmp', this._dataPath);if (this._dataStream) {this._dataStream = createWriteStream(this._dataPath, { flags: 'a' });}this._dataIno = (await stat(this._dataPath).catch(() => null))?.ino ?? null;this._meta.deleted = [];await this.persistMeta();// Rebuild indexes — offsets changed after compactionif (!duringInit && this._indexManager) {await this._indexManager.init(this._dataPath, { forceRebuild: true });}});}async *_rawRecordsForCompaction() {const deletedSet = new Set(this._meta.deleted);try {await stat(this._dataPath);} catch (error) {return;}const readStream = createReadStream(this._dataPath);let chunks = [];let totalLength = 0;let processedBytes = 0;for await (const chunk of readStream) {chunks.push(chunk);totalLength += chunk.length;while (true) {if (totalLength < 4) break;const headerBuffer = chunks[0].length >= 4? chunks[0]: Buffer.concat(chunks, 4);const recSize = headerBuffer.readInt32LE(0);if (recSize <= 4) {throw new Error(`Invalid record size read from stream: ${recSize}`);}if (totalLength < recSize) break;const record = Buffer.concat(chunks, recSize).subarray(0, recSize);if (!deletedSet.has(processedBytes)) {yield record;}[processedBytes, totalLength] = removeChunk(recSize, processedBytes, totalLength, chunks);}}}/*** Run a function exclusively: serialized against other operations in this* process (FIFO queue) and against other processes (lock file).** Re-entrant only for operations called INSIDE the callback's async call* chain (find inside delete, insert inside withLock, ...). Operations* started concurrently from outside wait their turn.** On acquiring the file lock, the on-disk meta is refreshed so writes in* other processes (fresh nextId, tombstones) are visible before mutating.* @private*/async _withFileLock(operation, fn) {if (lockContext.getStore()?.owner === this) {// Causal re-entry: our call chain already holds this db's lockreturn fn();}// In-process FIFO queue — avoids the 100ms lock file polling between// concurrent operations of the same instanceconst prev = this._mutexTail;let releaseQueue;this._mutexTail = new Promise(r => releaseQueue = r);await prev;try {await this._acquireFileLock(operation);try {// Sync with other processes' writes before entering the critical// section — also during init (paths are set before any locked op)if (this._metaPath) await this.refresh();return await lockContext.run({ owner: this }, fn);} finally {await unlink(this._lockPath).catch(() => { });}} finally {releaseQueue();}}async _acquireFileLock(operation, retries = 0) {while (true) {try {await writeFile(this._lockPath, String(process.pid), { flag: 'wx' });return;} catch (e) {if (e.code !== 'EEXIST') throw e;// Stale lock recovery: a crashed holder never unlinks its lock file.// If the lock file is older than staleLockTimeout, take it over.if (this._staleLockTimeout > 0) {const lockStat = await stat(this._lockPath).catch(() => null);if (lockStat && Date.now() - lockStat.mtimeMs > this._staleLockTimeout) {const stalePath = `${this._lockPath}.stale-${process.pid}`;try {await rename(this._lockPath, stalePath);const staleStat = await stat(stalePath);if (Date.now() - staleStat.mtimeMs > this._staleLockTimeout) {console.warn(`[mpackdb] removed stale lock ${this._lockPath} (held > ${this._staleLockTimeout}ms, holder presumed dead)`);await unlink(stalePath).catch(() => { });} else {// raced a fresh lock — put it backawait rename(stalePath, this._lockPath).catch(() => { });}} catch (e2) {// another waiter beat us to the takeover}continue;}}this.debug(`Waiting for lock file ${operation} retry #${retries}`);await new Promise(resolve => setTimeout(resolve, 25));retries++;}}}/*** Retrieves a document by its file offset and length* @param {Array} location - [offset, length] tuple* @returns {Promise<Object|null>} The deserialized document or null* @private* @deprecated Currently unused - may be removed in future versions*/async _getDocByLocation([offset, length]) {if (!offset || length === 0) return null;const fileHandle = await open(this._dataPath, 'r');try {const buffer = Buffer.alloc(length);await fileHandle.read(buffer, 0, length, offset);return deserialize(buffer);} finally {await fileHandle.close();}}/*** Close the database and persist all pending changes* Should be called before process exit** @returns {Promise<void>}*/async close() {// Persist indexes before closingif (this._indexManager) {await this._indexManager.close();}// Close data streamif (this._dataStream) {await new Promise((resolve, reject) => {this._dataStream.end((err) => err ? reject(err) : resolve());});}// Remove signal handlersif (this._processExitHandler) {process.off('SIGINT', this._processExitHandler);process.off('SIGTERM', this._processExitHandler);this._processExitHandler = null;}}debug(...args) {if (this._debug) {console.log(...args);}}}/*** Helper function to remove a processed chunk from the buffer* @param {number} recSize - Size of the record to remove* @param {number} processedBytes - Current offset in the file* @param {number} totalLength - Total length of buffered data* @param {Buffer[]} chunks - Array of buffer chunks* @returns {[number, number]} Updated [processedBytes, totalLength]*/function removeChunk(recSize, processedBytes, totalLength, chunks) {processedBytes += recSize;totalLength -= recSize;let bytesToRemove = recSize;while (bytesToRemove > 0 && chunks.length > 0) {const currentChunk = chunks[0];if (bytesToRemove >= currentChunk.length) {bytesToRemove -= currentChunk.length;chunks.shift();} else {chunks[0] = currentChunk.subarray(bytesToRemove);bytesToRemove = 0;}}return [processedBytes, totalLength];}export default MPackDB;
Branches
- mastermain branch
Latest commits
- c4cdb9b6node: import prefixes (Deno compat) + pre-existing index-state WIPcaramboleyo
- 0afb8f4bupdate now must be a callbackcaramboleyo
- cde73eb4release 1.0.6caramboleyo
- d01dda02add index hints, intersection, boundingBox; remove findByIndexcaramboleyo
- b8ffc1a0release 1.0.5caramboleyo
- d47876a1reimplemented lost features like indexed find and more testscaramboleyo
- 7f08da9afixed insert ignoring model definitioncaramboleyo
- 705774a9added flush before findcaramboleyo
- b4db6391initial commitcaramboleyo