mpackdb
All repositories: gitoria
22.9 KB
import { writeFile, readFile, open, rename } from 'fs/promises';import { createWriteStream } from 'fs';import { resolve } from 'path';import { deserialize } from './mpack.js';import { PrimaryKeyType, IndexType } from './MPackDB.js';const BLOCK_SIZE = 4096; // 4KB/*** Manages indexes for the database (in-memory deltas + on-disk persistence)*/export class IndexManager {_dbDir;_dbName;_indexes;_indexTypes;_primaryKeyType;_persistIntervalId = null;_indexPaths = {};_deltaIndexes = {};_tombstones = {};_persistThreshold = 1000;_totalDeltaCount = 0;/*** Creates a new IndexManager* @param {string} dbDir - Database directory path* @param {string} dbName - Database name* @param {string[]} indexes - Array of field names to index* @param {Object} indexTypes - Map of field names to IndexType* @param {number} primaryKeyType - Primary key type*/constructor(dbDir, dbName, indexes, indexTypes, primaryKeyType) {this._dbDir = dbDir;this._dbName = dbName;this._indexes = indexes;this._indexTypes = indexTypes;this._primaryKeyType = primaryKeyType;}/*** Initializes the index manager (rebuilds indexes from data file)* @param {string} dataPath - Path to the data file* @returns {Promise<void>}*/async init(dataPath, { forceRebuild = false } = {}) {for (const field of this._indexes) {this._deltaIndexes[field] = [];this._tombstones[field] = new Map();const indexPath = resolve(this._dbDir, `${this._dbName}.${field}.txt`);this._indexPaths[field] = indexPath;if (forceRebuild) {await this._rebuildIndex(dataPath, field);continue;}try {const fileHandle = await open(indexPath, 'r');const stats = await fileHandle.stat();await fileHandle.close();if (stats.size === 0) {await this._rebuildIndex(dataPath, field);}} catch (e) {if (e.code === 'ENOENT') {await this._rebuildIndex(dataPath, field);} else {throw e;}}}}/*** Rebuilds an index from the data file* @param {string} dataPath - Path to the data file* @param {string} field - Field name to index* @returns {Promise<void>}* @private*/async _rebuildIndex(dataPath, field) {const tempIndexPath = `${this._indexPaths[field]}.${Date.now()}.tmp`;const writeStream = createWriteStream(tempIndexPath, { flags: 'w', encoding: 'utf-8' });const docIterator = this._getDocsAndLocationsFromDataFile(dataPath);const entries = [];for await (const { doc, loc } of docIterator) {if (doc && doc[field] !== undefined) {const key = doc[field];entries.push({ key, loc });}}// Sort entries before writing to the index fileentries.sort((a, b) => this._compareKeys(a.key, b.key));for (const entry of entries) {writeStream.write(`${entry.key},${entry.loc[0]},${entry.loc[1]}\n`);}await new Promise(resolve => writeStream.end(resolve));await rename(tempIndexPath, this._indexPaths[field]);}/*** Generator that yields documents and their locations from the data file* @param {string} dataPath - Path to the data file* @yields {{doc: Object, loc: Array}} Document and location tuple* @private*/async* _getDocsAndLocationsFromDataFile(dataPath) {let fileHandle;try {fileHandle = await open(dataPath, 'r');const stats = await fileHandle.stat();let offset = 0;while (offset < stats.size) {// Read size headerconst sizeBuffer = Buffer.alloc(4);await fileHandle.read(sizeBuffer, 0, 4, offset);const size = sizeBuffer.readInt32LE(0);if (size <= 0 || offset + size > stats.size) {break; // Corrupted or incomplete record}const docBuffer = Buffer.alloc(size);await fileHandle.read(docBuffer, 0, size, offset);const doc = deserialize(docBuffer.subarray(4));yield { doc, loc: [offset, size] };offset += size;}} catch (e) {if (e.code !== 'ENOENT') {throw e;}// if file does not exist, do nothing} finally {await fileHandle?.close();}}/*** Compares two keys for sorting* @param {any} keyA - First key* @param {any} keyB - Second key* @returns {number} -1 if keyA < keyB, 0 if equal, 1 if keyA > keyB* @private*/_compareKeys(keyA, keyB) {if (typeof keyA === 'number' && typeof keyB === 'number') {return keyA - keyB;}return String(keyA).localeCompare(String(keyB));}/*** Starts automatic periodic persistence of indexes* @param {number} interval - Interval in milliseconds between auto-persists* @param {number} threshold - Number of operations before triggering auto-persist*/startAutoPersist(interval, threshold) {this._persistThreshold = threshold;// Start periodic persistif (interval > 0) {this._persistIntervalId = setInterval(async () => {if (this._totalDeltaCount > 0) {await this.persist();}}, interval);// Don't keep process alive just for this timerthis._persistIntervalId.unref();}}/*** Closes the index manager (persists indexes and clears interval)* @returns {Promise<void>}*/async close() {if (this._persistIntervalId) {clearInterval(this._persistIntervalId);this._persistIntervalId = null;}await this.persist();}/*** Persists all in-memory index deltas to disk* @returns {Promise<void>}*/async persist() {for (const field of this._indexes) {let onDiskEntries = [];const indexPath = this._indexPaths[field];try {const indexContent = await readFile(indexPath, 'utf-8');onDiskEntries = indexContent.trim().split('\n').filter(Boolean).map(line => {const [key, offset, length] = line.split(',');return { key: this._parseKey(key, field), loc: [parseInt(offset), parseInt(length)] };});} catch (e) {if (e.code !== 'ENOENT') throw e;}const fieldTombstones = this._tombstones[field];const validDiskEntries = onDiskEntries.filter(e => !fieldTombstones.has(e.loc[0]));const deltaEntries = this._deltaIndexes[field] || [];const finalIndexData = [...validDiskEntries, ...deltaEntries];finalIndexData.sort((a, b) => this._compareKeys(a.key, b.key));if (finalIndexData.length > 0) {const indexContent = finalIndexData.map(e => `${e.key},${e.loc[0]},${e.loc[1]}`).join('\n') + '\n';await writeFile(indexPath, indexContent, 'utf-8');} else {await writeFile(indexPath, '', 'utf-8');}// Clear the in-memory changes now that they are persistedthis._deltaIndexes[field] = [];this._tombstones[field].clear();}this._totalDeltaCount = 0;}/*** Parses a key based on the field's index type* @param {any} key - The key to parse* @param {string} field - Field name* @returns {any} Parsed key (number or string)* @private*/_parseKey(key, field) {const indexType = this._indexTypes[field];// Use index type if specified, otherwise fall back to primary key type logicif (indexType === IndexType.NUMERIC) {const num = parseInt(key, 10);if (!isNaN(num)) return num;} else if (indexType === IndexType.LEXICAL) {return key;}// Fallback to primary key type for fields without explicit index typeif (this._primaryKeyType === PrimaryKeyType.NUMBER) {const num = parseInt(key, 10);if (!isNaN(num)) return num;}return key;}/*** Performs a binary search on the blocks of an index file on disk.* This is a direct port of Joshua Bloch's famously correct binary search* algorithm, adapted for file blocks.** @param {string} field The index field to search.* @param {string|number} key The key to search for.* @returns {Promise<number>} The starting offset in the file for the linear scan.*/async _binarySearchOnDisk(field, key) {try {const fileHandle = await open(this._indexPaths[field], 'r');const stats = await fileHandle.stat();const totalBlocks = Math.ceil(stats.size / BLOCK_SIZE);let low = 0;let high = totalBlocks - 1;const buffer = Buffer.alloc(1024);while (low <= high) {const mid = Math.floor((low + high) / 2);const offset = mid * BLOCK_SIZE;const { bytesRead } = await fileHandle.read(buffer, 0, 1024, offset);if (bytesRead === 0) break;const blockContent = buffer.subarray(0, bytesRead).toString('utf-8');const lines = blockContent.split('\n');const firstKeyInBlock = this._parseKey(lines[0].split(',')[0], field);const lastKeyInBlock = this._parseKey(lines[lines.length - 2]?.split(',')[0] || firstKeyInBlock, field);const cmpLast = this._compareKeys(lastKeyInBlock, key);if (cmpLast < 0) {// The entire block is smaller than the keylow = mid + 1;} else {// The key might be in this block or a previous onehigh = mid - 1;}}await fileHandle.close();// low is the insertion point, start scan from the previous block to be safe.const finalOffset = Math.max(0, low - 1) * BLOCK_SIZE;return finalOffset;} catch (e) {if (e.code === 'ENOENT') {return -1;}throw e;}}/*** Finds the last entry for a given field and key (checks both delta and disk)* @param {string} field - Field name* @param {any} key - Key value to search for* @returns {Promise<Object|null>} Entry object with key and loc, or null if not found*/async findLastEntry(field, key) {const parsedKey = this._parseKey(key, field);const deltaIndex = this._deltaIndexes[field] || [];for (let i = deltaIndex.length - 1; i >= 0; i--) {const entry = deltaIndex[i];if (this._compareKeys(entry.key, parsedKey) === 0) {return entry;}}const onDiskEntry = await this._findLastEntryOnDisk(field, parsedKey);if (!onDiskEntry) return null;const isTombstoned = this._tombstones[field].has(onDiskEntry.loc[0]);return isTombstoned ? null : onDiskEntry;}/*** Finds the last entry for a key on disk (linear scan from binary search position)* @param {string} field - Field name* @param {any} key - Key to search for* @returns {Promise<Object|null>} Entry object or null* @private*/async _findLastEntryOnDisk(field, key) {const startOffset = await this._binarySearchOnDisk(field, key);if (startOffset === -1) return null;let lastMatch = null;const fileHandle = await open(this._indexPaths[field], 'r');const buffer = Buffer.alloc(BLOCK_SIZE);let currentOffset = startOffset;try {const stats = await fileHandle.stat();while (currentOffset < stats.size) {const { bytesRead } = await fileHandle.read(buffer, 0, BLOCK_SIZE, currentOffset);if (bytesRead === 0) break;const blockContent = buffer.slice(0, bytesRead).toString('utf-8');const lines = blockContent.split('\n');let found = false;for (const line of lines) {if (!line) continue;const [keyStr, offset, length] = line.split(',');const lineKey = this._parseKey(keyStr, field);if (this._compareKeys(lineKey, key) === 0) {found = true;lastMatch = { key: lineKey, loc: [parseInt(offset), parseInt(length)] };}}if (found) break; // We can stop once we find the block with the keycurrentOffset += BLOCK_SIZE;}} finally {await fileHandle.close();}return lastMatch;}/*** Adds a document to the indexes* @param {Object} doc - The document to index* @param {Array} loc - [offset, length] location in data file*/insert(doc, loc) {for (const field of this._indexes) {if (doc[field] !== undefined) {this._deltaIndexes[field].push({ key: doc[field], loc });this._totalDeltaCount++;}}// Auto-persist if threshold reachedif (this._totalDeltaCount >= this._persistThreshold) {// Persist async without blockingthis.persist().catch(err => console.error('Index persist error:', err));}}/*** Removes a document from the indexes* @param {Object} doc - The document to remove* @param {string} primaryKeyField - Primary key field name* @returns {Promise<void>}*/async remove(doc, primaryKeyField) {const pkValue = doc[primaryKeyField];const deltaPkIndex = this._deltaIndexes[primaryKeyField] || [];const indexInDelta = deltaPkIndex.findIndex(e => this._compareKeys(e.key, pkValue) === 0);if (indexInDelta !== -1) {const offsetToRemove = deltaPkIndex[indexInDelta].loc[0];for (const field of this._indexes) {this._deltaIndexes[field] = (this._deltaIndexes[field] || []).filter(e => e.loc[0] !== offsetToRemove);}} else {const onDiskEntry = await this._findLastEntryOnDisk(primaryKeyField, pkValue);if (onDiskEntry) {for (const field of this._indexes) {if (doc[field] !== undefined) {this._tombstones[field].set(onDiskEntry.loc[0], true);}}}}}/*** Gets all document locations from the first index (used for full scans)* @returns {Promise<Array[]>} Array of all [offset, length] locations*/async getAllLocations() {const field = this._indexes[0];const indexPath = this._indexPaths[field];let onDiskEntries = [];try {const indexContent = await readFile(indexPath, 'utf-8');onDiskEntries = indexContent.trim().split('\n').filter(Boolean).map(line => {const [, offset, length] = line.split(',');return { loc: [parseInt(offset), parseInt(length)] };});} catch (e) {if (e.code !== 'ENOENT') throw e;}const fieldTombstones = this._tombstones[field] || new Map();const validDiskEntries = onDiskEntries.filter(e => !fieldTombstones.has(e.loc[0]));const deltaEntries = this._deltaIndexes[field] || [];const finalEntries = new Map();for (const entry of validDiskEntries) {finalEntries.set(`${entry.loc[0]}:${entry.loc[1]}`, entry.loc);}for (const entry of deltaEntries) {finalEntries.set(`${entry.loc[0]}:${entry.loc[1]}`, entry.loc);}return Array.from(finalEntries.values());}/*** Gets document locations for a given field and key* @param {string} field - Field name* @param {any} key - Key value to search for* @returns {Promise<Array[]>} Array of [offset, length] locations*/async get(field, key) {const parsedKey = this._parseKey(key, field);const onDiskResults = await this._getOnDisk(field, parsedKey);const deltaResults = (this._deltaIndexes[field] || []).filter(e => this._compareKeys(e.key, parsedKey) === 0);const tombstonedOffsets = this._tombstones[field] || new Map();const finalEntries = new Map();for (const entry of onDiskResults) {if (!tombstonedOffsets.has(entry.loc[0])) {finalEntries.set(`${entry.loc[0]}:${entry.loc[1]}`, entry);}}for (const entry of deltaResults) {finalEntries.set(`${entry.loc[0]}:${entry.loc[1]}`, entry);}return Array.from(finalEntries.values());}/*** Gets all entries for a key from disk* @param {string} field - Field name* @param {any} key - Key to search for* @returns {Promise<Array>} Array of entry objects* @private*/async _getOnDisk(field, key) {const startOffset = await this._binarySearchOnDisk(field, key);if (startOffset === -1) return [];const results = [];const fileHandle = await open(this._indexPaths[field], 'r');const buffer = Buffer.alloc(BLOCK_SIZE);let currentOffset = startOffset;try {const stats = await fileHandle.stat();while (currentOffset < stats.size) {const { bytesRead } = await fileHandle.read(buffer, 0, BLOCK_SIZE, currentOffset);if (bytesRead === 0) break;const blockContent = buffer.subarray(0, bytesRead).toString('utf-8');const lines = blockContent.split('\n');let firstKeyInBlock;let foundMatchInBlock = false;for (const line of lines) {if (!line) continue;const [keyStr, locOffset, locLength] = line.split(',');if (firstKeyInBlock === undefined) {firstKeyInBlock = this._parseKey(keyStr, field);}const lineKey = this._parseKey(keyStr, field);if (this._compareKeys(lineKey, key) === 0) {foundMatchInBlock = true;results.push({ key: lineKey, loc: [parseInt(locOffset), parseInt(locLength)] });}}if (results.length > 0 && !foundMatchInBlock) {break;}if (foundMatchInBlock === false && firstKeyInBlock && this._compareKeys(firstKeyInBlock, key) > 0) {break;}currentOffset += BLOCK_SIZE;}} finally {await fileHandle.close();}return results;}/*** Yields index entries in sorted order by streaming blocks from disk.* Merges with in-memory deltas, excludes tombstones.* @param {string} field - Field name* @param {Object} [options]* @param {any} [options.from] - Start key (inclusive), uses binary search to skip ahead* @param {any} [options.to] - End key (inclusive), stops scan when exceeded* @param {'asc'|'desc'} [options.direction='asc'] - Scan direction* @yields {{key: any, loc: [number, number]}}*/async *entries(field, { from, to, direction = 'asc' } = {}) {const desc = direction === 'desc';const tombstones = this._tombstones[field] || new Map();// Collect and sort deltas (small, always in memory)const deltas = (this._deltaIndexes[field] || []).filter(e => !tombstones.has(e.loc[0])).slice().sort((a, b) => this._compareKeys(a.key, b.key));let deltaIdx;if (desc) {deltaIdx = deltas.length - 1;if (from !== undefined) {while (deltaIdx >= 0 && this._compareKeys(deltas[deltaIdx].key, from) > 0) {deltaIdx--;}}} else {deltaIdx = 0;if (from !== undefined) {while (deltaIdx < deltas.length && this._compareKeys(deltas[deltaIdx].key, from) < 0) {deltaIdx++;}}}// Helper to check if a key is within the `to` boundconst pastTo = (key) => {if (to === undefined) return false;return desc ? this._compareKeys(key, to) < 0 : this._compareKeys(key, to) > 0;};// Stream disk entries block by blocklet fileHandle;let fileSize = 0;let startOffset = 0;try {fileHandle = await open(this._indexPaths[field], 'r');fileSize = (await fileHandle.stat()).size;if (desc && from === undefined) {// Start from the last blockstartOffset = Math.max(0, Math.floor((fileSize - 1) / BLOCK_SIZE) * BLOCK_SIZE);} else if (from !== undefined) {startOffset = await this._binarySearchOnDisk(field, from);if (startOffset === -1) startOffset = desc ? Math.max(0, Math.floor((fileSize - 1) / BLOCK_SIZE) * BLOCK_SIZE) : 0;}} catch (e) {if (e.code !== 'ENOENT') throw e;}const buffer = Buffer.alloc(BLOCK_SIZE);const seen = new Set();// Generator for disk entries (ascending)const self = this;async function* diskEntriesAsc() {if (!fileHandle) return;let currentOffset = startOffset;try {while (currentOffset < fileSize) {const { bytesRead } = await fileHandle.read(buffer, 0, BLOCK_SIZE, currentOffset);if (bytesRead === 0) break;const lines = buffer.subarray(0, bytesRead).toString('utf-8').split('\n');for (const line of lines) {if (!line) continue;const [keyStr, locOffset, locLength] = line.split(',');const key = self._parseKey(keyStr, field);if (to !== undefined && self._compareKeys(key, to) > 0) return;const loc = [parseInt(locOffset), parseInt(locLength)];if (tombstones.has(loc[0])) continue;if (from !== undefined && self._compareKeys(key, from) < 0) continue;yield { key, loc };}currentOffset += BLOCK_SIZE;}} finally {await fileHandle.close();}}// Generator for disk entries (descending)async function* diskEntriesDesc() {if (!fileHandle) return;let currentOffset = startOffset;try {// Collect all entries from startOffset forward (to handle `from` key block),// then walk backwards block by blockconst forwardEntries = [];// First read from startOffset forward to collect entries <= fromwhile (currentOffset < fileSize) {const { bytesRead } = await fileHandle.read(buffer, 0, BLOCK_SIZE, currentOffset);if (bytesRead === 0) break;const lines = buffer.subarray(0, bytesRead).toString('utf-8').split('\n');for (const line of lines) {if (!line) continue;const [keyStr, locOffset, locLength] = line.split(',');const key = self._parseKey(keyStr, field);if (from !== undefined && self._compareKeys(key, from) > 0) continue;if (to !== undefined && self._compareKeys(key, to) < 0) continue;const loc = [parseInt(locOffset), parseInt(locLength)];if (tombstones.has(loc[0])) continue;forwardEntries.push({ key, loc });}currentOffset += BLOCK_SIZE;}// Yield collected entries in reversefor (let i = forwardEntries.length - 1; i >= 0; i--) {yield forwardEntries[i];}// Now walk backwards from startOffsetcurrentOffset = startOffset - BLOCK_SIZE;while (currentOffset >= 0) {const { bytesRead } = await fileHandle.read(buffer, 0, BLOCK_SIZE, currentOffset);if (bytesRead === 0) break;const lines = buffer.subarray(0, bytesRead).toString('utf-8').split('\n');const blockEntries = [];for (const line of lines) {if (!line) continue;const [keyStr, locOffset, locLength] = line.split(',');const key = self._parseKey(keyStr, field);if (to !== undefined && self._compareKeys(key, to) < 0) return;const loc = [parseInt(locOffset), parseInt(locLength)];if (tombstones.has(loc[0])) continue;blockEntries.push({ key, loc });}for (let i = blockEntries.length - 1; i >= 0; i--) {yield blockEntries[i];}currentOffset -= BLOCK_SIZE;}} finally {await fileHandle.close();}}const diskIter = desc ? diskEntriesDesc() : diskEntriesAsc();let diskNext = await diskIter.next();// Merge disk + deltas in sorted orderwhile (true) {const hasDisk = !diskNext.done;const hasDelta = desc ? deltaIdx >= 0 : deltaIdx < deltas.length;if (!hasDisk && !hasDelta) break;// Skip deltas past `to`if (hasDelta && pastTo(deltas[deltaIdx].key)) {if (desc) deltaIdx = -1; else deltaIdx = deltas.length;continue;}let entry;if (!hasDisk) {entry = deltas[desc ? deltaIdx-- : deltaIdx++];} else if (!hasDelta) {entry = diskNext.value;diskNext = await diskIter.next();} else {const cmp = this._compareKeys(diskNext.value.key, deltas[deltaIdx].key);const pickDisk = desc ? cmp >= 0 : cmp <= 0;if (pickDisk) {entry = diskNext.value;diskNext = await diskIter.next();} else {entry = deltas[desc ? deltaIdx-- : deltaIdx++];}}if (pastTo(entry.key)) {if (!diskNext.done) await diskIter.return();return;}const locKey = `${entry.loc[0]}:${entry.loc[1]}`;if (seen.has(locKey)) continue;seen.add(locKey);yield entry;}}async* _getDocsByLocation(dataPath, locations) {const fileHandle = await open(dataPath, 'r');try {for (const loc of locations) {if (loc && loc.length === 2 && loc[1] > 0) {const [offset, length] = loc;const buffer = Buffer.alloc(length);await fileHandle.read(buffer, 0, length, offset);yield deserialize(buffer.subarray(4));}}} finally {await fileHandle.close();}}}export default IndexManager;
Branches
- mastermain branch
Latest commits
- d01dda02add index hints, intersection, boundingBox; remove findByIndexcaramboleyo
- b8ffc1a0release 1.0.5caramboleyo
- d47876a1reimplemented lost features like indexed find and more testscaramboleyo
- 7f08da9afixed insert ignoring model definitioncaramboleyo
- 705774a9added flush before findcaramboleyo
- b4db6391initial commitcaramboleyo