770 lines
31 KiB
JavaScript
770 lines
31 KiB
JavaScript
|
|
/**
|
||
|
|
* Metadata Index System
|
||
|
|
* Maintains inverted indexes for fast metadata filtering
|
||
|
|
* Automatically updates indexes when data changes
|
||
|
|
*/
|
||
|
|
import { MetadataIndexCache } from './metadataIndexCache.js';
|
||
|
|
import { prodLog } from './logger.js';
|
||
|
|
/**
|
||
|
|
* Manages metadata indexes for fast filtering
|
||
|
|
* Maintains inverted indexes: field+value -> list of IDs
|
||
|
|
*/
|
||
|
|
export class MetadataIndexManager {
|
||
|
|
constructor(storage, config = {}) {
|
||
|
|
this.indexCache = new Map();
|
||
|
|
this.dirtyEntries = new Set();
|
||
|
|
this.isRebuilding = false;
|
||
|
|
this.fieldIndexes = new Map();
|
||
|
|
this.dirtyFields = new Set();
|
||
|
|
this.lastFlushTime = Date.now();
|
||
|
|
this.autoFlushThreshold = 10; // Start with 10 for more frequent non-blocking flushes
|
||
|
|
this.storage = storage;
|
||
|
|
this.config = {
|
||
|
|
maxIndexSize: config.maxIndexSize ?? 10000,
|
||
|
|
rebuildThreshold: config.rebuildThreshold ?? 0.1,
|
||
|
|
autoOptimize: config.autoOptimize ?? true,
|
||
|
|
indexedFields: config.indexedFields ?? [],
|
||
|
|
excludeFields: config.excludeFields ?? ['id', 'createdAt', 'updatedAt', 'embedding', 'vector', 'embeddings', 'vectors']
|
||
|
|
};
|
||
|
|
// Initialize metadata cache with similar config to search cache
|
||
|
|
this.metadataCache = new MetadataIndexCache({
|
||
|
|
maxAge: 5 * 60 * 1000, // 5 minutes
|
||
|
|
maxSize: 500, // 500 entries (field indexes + value chunks)
|
||
|
|
enabled: true
|
||
|
|
});
|
||
|
|
}
|
||
|
|
/**
|
||
|
|
* Get index key for field and value
|
||
|
|
*/
|
||
|
|
getIndexKey(field, value) {
|
||
|
|
const normalizedValue = this.normalizeValue(value);
|
||
|
|
return `${field}:${normalizedValue}`;
|
||
|
|
}
|
||
|
|
/**
|
||
|
|
* Generate field index filename for filter discovery
|
||
|
|
*/
|
||
|
|
getFieldIndexFilename(field) {
|
||
|
|
return `field_${field}`;
|
||
|
|
}
|
||
|
|
/**
|
||
|
|
* Generate value chunk filename for scalable storage
|
||
|
|
*/
|
||
|
|
getValueChunkFilename(field, value, chunkIndex = 0) {
|
||
|
|
const normalizedValue = this.normalizeValue(value);
|
||
|
|
const safeValue = this.makeSafeFilename(normalizedValue);
|
||
|
|
return `${field}_${safeValue}_chunk${chunkIndex}`;
|
||
|
|
}
|
||
|
|
/**
|
||
|
|
* Make a value safe for use in filenames
|
||
|
|
*/
|
||
|
|
makeSafeFilename(value) {
|
||
|
|
// Replace unsafe characters and limit length
|
||
|
|
return value
|
||
|
|
.replace(/[^a-zA-Z0-9-_]/g, '_')
|
||
|
|
.substring(0, 50)
|
||
|
|
.toLowerCase();
|
||
|
|
}
|
||
|
|
/**
|
||
|
|
* Normalize value for consistent indexing
|
||
|
|
*/
|
||
|
|
normalizeValue(value) {
|
||
|
|
if (value === null || value === undefined)
|
||
|
|
return '__NULL__';
|
||
|
|
if (typeof value === 'boolean')
|
||
|
|
return value ? '__TRUE__' : '__FALSE__';
|
||
|
|
if (typeof value === 'number')
|
||
|
|
return value.toString();
|
||
|
|
if (Array.isArray(value)) {
|
||
|
|
const joined = value.map(v => this.normalizeValue(v)).join(',');
|
||
|
|
// Hash very long array values to avoid filesystem limits
|
||
|
|
if (joined.length > 100) {
|
||
|
|
return this.hashValue(joined);
|
||
|
|
}
|
||
|
|
return joined;
|
||
|
|
}
|
||
|
|
const stringValue = String(value).toLowerCase().trim();
|
||
|
|
// Hash very long string values to avoid filesystem limits
|
||
|
|
if (stringValue.length > 100) {
|
||
|
|
return this.hashValue(stringValue);
|
||
|
|
}
|
||
|
|
return stringValue;
|
||
|
|
}
|
||
|
|
/**
|
||
|
|
* Create a short hash for long values to avoid filesystem filename limits
|
||
|
|
*/
|
||
|
|
hashValue(value) {
|
||
|
|
// Simple hash function to create shorter keys
|
||
|
|
let hash = 0;
|
||
|
|
for (let i = 0; i < value.length; i++) {
|
||
|
|
const char = value.charCodeAt(i);
|
||
|
|
hash = ((hash << 5) - hash) + char;
|
||
|
|
hash = hash & hash; // Convert to 32-bit integer
|
||
|
|
}
|
||
|
|
return `__HASH_${Math.abs(hash).toString(36)}`;
|
||
|
|
}
|
||
|
|
/**
|
||
|
|
* Check if field should be indexed
|
||
|
|
*/
|
||
|
|
shouldIndexField(field) {
|
||
|
|
if (this.config.excludeFields.includes(field))
|
||
|
|
return false;
|
||
|
|
if (this.config.indexedFields.length > 0) {
|
||
|
|
return this.config.indexedFields.includes(field);
|
||
|
|
}
|
||
|
|
return true;
|
||
|
|
}
|
||
|
|
/**
|
||
|
|
* Extract indexable field-value pairs from metadata
|
||
|
|
*/
|
||
|
|
extractIndexableFields(metadata) {
|
||
|
|
const fields = [];
|
||
|
|
const extract = (obj, prefix = '') => {
|
||
|
|
for (const [key, value] of Object.entries(obj)) {
|
||
|
|
const fullKey = prefix ? `${prefix}.${key}` : key;
|
||
|
|
if (!this.shouldIndexField(fullKey))
|
||
|
|
continue;
|
||
|
|
if (value && typeof value === 'object' && !Array.isArray(value)) {
|
||
|
|
// Recurse into nested objects
|
||
|
|
extract(value, fullKey);
|
||
|
|
}
|
||
|
|
else {
|
||
|
|
// Index this field
|
||
|
|
fields.push({ field: fullKey, value });
|
||
|
|
// If it's an array, also index each element
|
||
|
|
if (Array.isArray(value)) {
|
||
|
|
for (const item of value) {
|
||
|
|
fields.push({ field: fullKey, value: item });
|
||
|
|
}
|
||
|
|
}
|
||
|
|
}
|
||
|
|
}
|
||
|
|
};
|
||
|
|
if (metadata && typeof metadata === 'object') {
|
||
|
|
extract(metadata);
|
||
|
|
}
|
||
|
|
return fields;
|
||
|
|
}
|
||
|
|
/**
|
||
|
|
* Add item to metadata indexes
|
||
|
|
*/
|
||
|
|
async addToIndex(id, metadata, skipFlush = false) {
|
||
|
|
const fields = this.extractIndexableFields(metadata);
|
||
|
|
for (let i = 0; i < fields.length; i++) {
|
||
|
|
const { field, value } = fields[i];
|
||
|
|
const key = this.getIndexKey(field, value);
|
||
|
|
// Get or create index entry
|
||
|
|
let entry = this.indexCache.get(key);
|
||
|
|
if (!entry) {
|
||
|
|
const loadedEntry = await this.loadIndexEntry(key);
|
||
|
|
entry = loadedEntry ?? {
|
||
|
|
field,
|
||
|
|
value: this.normalizeValue(value),
|
||
|
|
ids: new Set(),
|
||
|
|
lastUpdated: Date.now()
|
||
|
|
};
|
||
|
|
this.indexCache.set(key, entry);
|
||
|
|
}
|
||
|
|
// Add ID to entry
|
||
|
|
entry.ids.add(id);
|
||
|
|
entry.lastUpdated = Date.now();
|
||
|
|
this.dirtyEntries.add(key);
|
||
|
|
// Update field index
|
||
|
|
await this.updateFieldIndex(field, value, 1);
|
||
|
|
// Yield to event loop every 5 fields to prevent blocking
|
||
|
|
if (i % 5 === 4) {
|
||
|
|
await this.yieldToEventLoop();
|
||
|
|
}
|
||
|
|
}
|
||
|
|
// Adaptive auto-flush based on usage patterns
|
||
|
|
if (!skipFlush) {
|
||
|
|
const timeSinceLastFlush = Date.now() - this.lastFlushTime;
|
||
|
|
const shouldAutoFlush = this.dirtyEntries.size >= this.autoFlushThreshold || // Size threshold
|
||
|
|
(this.dirtyEntries.size > 10 && timeSinceLastFlush > 5000); // Time threshold (5 seconds)
|
||
|
|
if (shouldAutoFlush) {
|
||
|
|
const startTime = Date.now();
|
||
|
|
await this.flush();
|
||
|
|
const flushTime = Date.now() - startTime;
|
||
|
|
// Adapt threshold based on flush performance
|
||
|
|
if (flushTime < 50) {
|
||
|
|
// Fast flush, can handle more entries
|
||
|
|
this.autoFlushThreshold = Math.min(200, this.autoFlushThreshold * 1.2);
|
||
|
|
}
|
||
|
|
else if (flushTime > 200) {
|
||
|
|
// Slow flush, reduce batch size
|
||
|
|
this.autoFlushThreshold = Math.max(20, this.autoFlushThreshold * 0.8);
|
||
|
|
}
|
||
|
|
// Yield to event loop after flush to prevent blocking
|
||
|
|
await this.yieldToEventLoop();
|
||
|
|
}
|
||
|
|
}
|
||
|
|
// Invalidate cache for these fields
|
||
|
|
for (const { field } of fields) {
|
||
|
|
this.metadataCache.invalidatePattern(`field_values_${field}`);
|
||
|
|
}
|
||
|
|
}
|
||
|
|
/**
|
||
|
|
* Update field index with value count
|
||
|
|
*/
|
||
|
|
async updateFieldIndex(field, value, delta) {
|
||
|
|
let fieldIndex = this.fieldIndexes.get(field);
|
||
|
|
if (!fieldIndex) {
|
||
|
|
// Load from storage if not in memory
|
||
|
|
fieldIndex = await this.loadFieldIndex(field) ?? {
|
||
|
|
values: {},
|
||
|
|
lastUpdated: Date.now()
|
||
|
|
};
|
||
|
|
this.fieldIndexes.set(field, fieldIndex);
|
||
|
|
}
|
||
|
|
const normalizedValue = this.normalizeValue(value);
|
||
|
|
fieldIndex.values[normalizedValue] = (fieldIndex.values[normalizedValue] || 0) + delta;
|
||
|
|
// Remove if count drops to 0
|
||
|
|
if (fieldIndex.values[normalizedValue] <= 0) {
|
||
|
|
delete fieldIndex.values[normalizedValue];
|
||
|
|
}
|
||
|
|
fieldIndex.lastUpdated = Date.now();
|
||
|
|
this.dirtyFields.add(field);
|
||
|
|
}
|
||
|
|
/**
|
||
|
|
* Remove item from metadata indexes
|
||
|
|
*/
|
||
|
|
async removeFromIndex(id, metadata) {
|
||
|
|
if (metadata) {
|
||
|
|
// Remove from specific field indexes
|
||
|
|
const fields = this.extractIndexableFields(metadata);
|
||
|
|
for (const { field, value } of fields) {
|
||
|
|
const key = this.getIndexKey(field, value);
|
||
|
|
let entry = this.indexCache.get(key);
|
||
|
|
if (!entry) {
|
||
|
|
const loadedEntry = await this.loadIndexEntry(key);
|
||
|
|
entry = loadedEntry ?? undefined;
|
||
|
|
}
|
||
|
|
if (entry) {
|
||
|
|
entry.ids.delete(id);
|
||
|
|
entry.lastUpdated = Date.now();
|
||
|
|
this.dirtyEntries.add(key);
|
||
|
|
// Update field index
|
||
|
|
await this.updateFieldIndex(field, value, -1);
|
||
|
|
// If no IDs left, mark for cleanup
|
||
|
|
if (entry.ids.size === 0) {
|
||
|
|
this.indexCache.delete(key);
|
||
|
|
await this.deleteIndexEntry(key);
|
||
|
|
}
|
||
|
|
}
|
||
|
|
// Invalidate cache
|
||
|
|
this.metadataCache.invalidatePattern(`field_values_${field}`);
|
||
|
|
}
|
||
|
|
}
|
||
|
|
else {
|
||
|
|
// Remove from all indexes (slower, requires scanning)
|
||
|
|
for (const [key, entry] of this.indexCache.entries()) {
|
||
|
|
if (entry.ids.has(id)) {
|
||
|
|
entry.ids.delete(id);
|
||
|
|
entry.lastUpdated = Date.now();
|
||
|
|
this.dirtyEntries.add(key);
|
||
|
|
if (entry.ids.size === 0) {
|
||
|
|
this.indexCache.delete(key);
|
||
|
|
await this.deleteIndexEntry(key);
|
||
|
|
}
|
||
|
|
}
|
||
|
|
}
|
||
|
|
}
|
||
|
|
}
|
||
|
|
/**
|
||
|
|
* Get IDs for a specific field-value combination with caching
|
||
|
|
*/
|
||
|
|
async getIds(field, value) {
|
||
|
|
const key = this.getIndexKey(field, value);
|
||
|
|
// Check metadata cache first
|
||
|
|
const cacheKey = `ids_${key}`;
|
||
|
|
const cachedIds = this.metadataCache.get(cacheKey);
|
||
|
|
if (cachedIds) {
|
||
|
|
return cachedIds;
|
||
|
|
}
|
||
|
|
// Try in-memory cache
|
||
|
|
let entry = this.indexCache.get(key);
|
||
|
|
// Load from storage if not cached
|
||
|
|
if (!entry) {
|
||
|
|
const loadedEntry = await this.loadIndexEntry(key);
|
||
|
|
if (loadedEntry) {
|
||
|
|
entry = loadedEntry;
|
||
|
|
this.indexCache.set(key, entry);
|
||
|
|
}
|
||
|
|
}
|
||
|
|
const ids = entry ? Array.from(entry.ids) : [];
|
||
|
|
// Cache the result
|
||
|
|
this.metadataCache.set(cacheKey, ids);
|
||
|
|
return ids;
|
||
|
|
}
|
||
|
|
/**
|
||
|
|
* Get all available values for a field (for filter discovery)
|
||
|
|
*/
|
||
|
|
async getFilterValues(field) {
|
||
|
|
// Check cache first
|
||
|
|
const cacheKey = `field_values_${field}`;
|
||
|
|
const cachedValues = this.metadataCache.get(cacheKey);
|
||
|
|
if (cachedValues) {
|
||
|
|
return cachedValues;
|
||
|
|
}
|
||
|
|
// Check in-memory field indexes first
|
||
|
|
let fieldIndex = this.fieldIndexes.get(field);
|
||
|
|
// If not in memory, load from storage
|
||
|
|
if (!fieldIndex) {
|
||
|
|
const loaded = await this.loadFieldIndex(field);
|
||
|
|
if (loaded) {
|
||
|
|
fieldIndex = loaded;
|
||
|
|
this.fieldIndexes.set(field, loaded);
|
||
|
|
}
|
||
|
|
}
|
||
|
|
if (!fieldIndex) {
|
||
|
|
return [];
|
||
|
|
}
|
||
|
|
const values = Object.keys(fieldIndex.values);
|
||
|
|
// Cache the result
|
||
|
|
this.metadataCache.set(cacheKey, values);
|
||
|
|
return values;
|
||
|
|
}
|
||
|
|
/**
|
||
|
|
* Get all indexed fields (for filter discovery)
|
||
|
|
*/
|
||
|
|
async getFilterFields() {
|
||
|
|
// Check cache first
|
||
|
|
const cacheKey = 'all_filter_fields';
|
||
|
|
const cachedFields = this.metadataCache.get(cacheKey);
|
||
|
|
if (cachedFields) {
|
||
|
|
return cachedFields;
|
||
|
|
}
|
||
|
|
// Get fields from in-memory indexes and storage
|
||
|
|
const fields = new Set(this.fieldIndexes.keys());
|
||
|
|
// Also scan storage for persisted field indexes (in case not loaded)
|
||
|
|
// This would require a new storage method to list field indexes
|
||
|
|
// For now, just use in-memory fields
|
||
|
|
const fieldsArray = Array.from(fields);
|
||
|
|
// Cache the result
|
||
|
|
this.metadataCache.set(cacheKey, fieldsArray);
|
||
|
|
return fieldsArray;
|
||
|
|
}
|
||
|
|
/**
|
||
|
|
* Convert MongoDB-style filter to simple field-value criteria for indexing
|
||
|
|
*/
|
||
|
|
convertFilterToCriteria(filter) {
|
||
|
|
const criteria = [];
|
||
|
|
if (!filter || typeof filter !== 'object') {
|
||
|
|
return criteria;
|
||
|
|
}
|
||
|
|
for (const [key, value] of Object.entries(filter)) {
|
||
|
|
// Skip logical operators for now - handle them separately
|
||
|
|
if (key.startsWith('$'))
|
||
|
|
continue;
|
||
|
|
if (value && typeof value === 'object' && !Array.isArray(value)) {
|
||
|
|
// Handle MongoDB operators
|
||
|
|
for (const [op, operand] of Object.entries(value)) {
|
||
|
|
switch (op) {
|
||
|
|
case '$in':
|
||
|
|
if (Array.isArray(operand)) {
|
||
|
|
criteria.push({ field: key, values: operand });
|
||
|
|
}
|
||
|
|
break;
|
||
|
|
case '$eq':
|
||
|
|
criteria.push({ field: key, values: [operand] });
|
||
|
|
break;
|
||
|
|
case '$includes':
|
||
|
|
// For $includes, the operand is the value we're looking for in an array field
|
||
|
|
criteria.push({ field: key, values: [operand] });
|
||
|
|
break;
|
||
|
|
// For other operators, we can't use index efficiently, skip for now
|
||
|
|
default:
|
||
|
|
break;
|
||
|
|
}
|
||
|
|
}
|
||
|
|
}
|
||
|
|
else {
|
||
|
|
// Direct value or array
|
||
|
|
const values = Array.isArray(value) ? value : [value];
|
||
|
|
criteria.push({ field: key, values });
|
||
|
|
}
|
||
|
|
}
|
||
|
|
return criteria;
|
||
|
|
}
|
||
|
|
/**
|
||
|
|
* Get IDs matching MongoDB-style metadata filter using indexes where possible
|
||
|
|
*/
|
||
|
|
async getIdsForFilter(filter) {
|
||
|
|
if (!filter || Object.keys(filter).length === 0) {
|
||
|
|
return [];
|
||
|
|
}
|
||
|
|
// Handle logical operators
|
||
|
|
if (filter.$and && Array.isArray(filter.$and)) {
|
||
|
|
// For $and, we need intersection of all sub-filters
|
||
|
|
const allIds = [];
|
||
|
|
for (const subFilter of filter.$and) {
|
||
|
|
const subIds = await this.getIdsForFilter(subFilter);
|
||
|
|
allIds.push(subIds);
|
||
|
|
}
|
||
|
|
if (allIds.length === 0)
|
||
|
|
return [];
|
||
|
|
if (allIds.length === 1)
|
||
|
|
return allIds[0];
|
||
|
|
// Intersection of all sets
|
||
|
|
return allIds.reduce((intersection, currentSet) => intersection.filter(id => currentSet.includes(id)));
|
||
|
|
}
|
||
|
|
if (filter.$or && Array.isArray(filter.$or)) {
|
||
|
|
// For $or, we need union of all sub-filters
|
||
|
|
const unionIds = new Set();
|
||
|
|
for (const subFilter of filter.$or) {
|
||
|
|
const subIds = await this.getIdsForFilter(subFilter);
|
||
|
|
subIds.forEach(id => unionIds.add(id));
|
||
|
|
}
|
||
|
|
return Array.from(unionIds);
|
||
|
|
}
|
||
|
|
// Handle regular field filters
|
||
|
|
const criteria = this.convertFilterToCriteria(filter);
|
||
|
|
const idSets = [];
|
||
|
|
for (const { field, values } of criteria) {
|
||
|
|
const unionIds = new Set();
|
||
|
|
for (const value of values) {
|
||
|
|
const ids = await this.getIds(field, value);
|
||
|
|
ids.forEach(id => unionIds.add(id));
|
||
|
|
}
|
||
|
|
idSets.push(Array.from(unionIds));
|
||
|
|
}
|
||
|
|
if (idSets.length === 0)
|
||
|
|
return [];
|
||
|
|
if (idSets.length === 1)
|
||
|
|
return idSets[0];
|
||
|
|
// Intersection of all field criteria (implicit $and)
|
||
|
|
return idSets.reduce((intersection, currentSet) => intersection.filter(id => currentSet.includes(id)));
|
||
|
|
}
|
||
|
|
/**
|
||
|
|
* Get IDs matching multiple criteria (intersection) - LEGACY METHOD
|
||
|
|
* @deprecated Use getIdsForFilter instead
|
||
|
|
*/
|
||
|
|
async getIdsForCriteria(criteria) {
|
||
|
|
return this.getIdsForFilter(criteria);
|
||
|
|
}
|
||
|
|
/**
|
||
|
|
* Flush dirty entries to storage (non-blocking version)
|
||
|
|
*/
|
||
|
|
async flush() {
|
||
|
|
if (this.dirtyEntries.size === 0 && this.dirtyFields.size === 0) {
|
||
|
|
return; // Nothing to flush
|
||
|
|
}
|
||
|
|
// Process in smaller batches to avoid blocking
|
||
|
|
const BATCH_SIZE = 20;
|
||
|
|
const allPromises = [];
|
||
|
|
// Flush value entries in batches
|
||
|
|
const dirtyEntriesArray = Array.from(this.dirtyEntries);
|
||
|
|
for (let i = 0; i < dirtyEntriesArray.length; i += BATCH_SIZE) {
|
||
|
|
const batch = dirtyEntriesArray.slice(i, i + BATCH_SIZE);
|
||
|
|
const batchPromises = batch.map(key => {
|
||
|
|
const entry = this.indexCache.get(key);
|
||
|
|
return entry ? this.saveIndexEntry(key, entry) : Promise.resolve();
|
||
|
|
});
|
||
|
|
allPromises.push(...batchPromises);
|
||
|
|
// Yield to event loop between batches
|
||
|
|
if (i + BATCH_SIZE < dirtyEntriesArray.length) {
|
||
|
|
await this.yieldToEventLoop();
|
||
|
|
}
|
||
|
|
}
|
||
|
|
// Flush field indexes in batches
|
||
|
|
const dirtyFieldsArray = Array.from(this.dirtyFields);
|
||
|
|
for (let i = 0; i < dirtyFieldsArray.length; i += BATCH_SIZE) {
|
||
|
|
const batch = dirtyFieldsArray.slice(i, i + BATCH_SIZE);
|
||
|
|
const batchPromises = batch.map(field => {
|
||
|
|
const fieldIndex = this.fieldIndexes.get(field);
|
||
|
|
return fieldIndex ? this.saveFieldIndex(field, fieldIndex) : Promise.resolve();
|
||
|
|
});
|
||
|
|
allPromises.push(...batchPromises);
|
||
|
|
// Yield to event loop between batches
|
||
|
|
if (i + BATCH_SIZE < dirtyFieldsArray.length) {
|
||
|
|
await this.yieldToEventLoop();
|
||
|
|
}
|
||
|
|
}
|
||
|
|
// Wait for all operations to complete
|
||
|
|
await Promise.all(allPromises);
|
||
|
|
this.dirtyEntries.clear();
|
||
|
|
this.dirtyFields.clear();
|
||
|
|
this.lastFlushTime = Date.now();
|
||
|
|
}
|
||
|
|
/**
|
||
|
|
* Yield control back to the Node.js event loop
|
||
|
|
* Prevents blocking during long-running operations
|
||
|
|
*/
|
||
|
|
async yieldToEventLoop() {
|
||
|
|
return new Promise(resolve => setImmediate(resolve));
|
||
|
|
}
|
||
|
|
/**
|
||
|
|
* Load field index from storage
|
||
|
|
*/
|
||
|
|
async loadFieldIndex(field) {
|
||
|
|
try {
|
||
|
|
const filename = this.getFieldIndexFilename(field);
|
||
|
|
const cacheKey = `field_index_${filename}`;
|
||
|
|
// Check cache first
|
||
|
|
const cached = this.metadataCache.get(cacheKey);
|
||
|
|
if (cached) {
|
||
|
|
return cached;
|
||
|
|
}
|
||
|
|
// Load from storage
|
||
|
|
const indexId = `__metadata_field_index__${filename}`;
|
||
|
|
const data = await this.storage.getMetadata(indexId);
|
||
|
|
if (data) {
|
||
|
|
const fieldIndex = {
|
||
|
|
values: data.values || {},
|
||
|
|
lastUpdated: data.lastUpdated || Date.now()
|
||
|
|
};
|
||
|
|
// Cache it
|
||
|
|
this.metadataCache.set(cacheKey, fieldIndex);
|
||
|
|
return fieldIndex;
|
||
|
|
}
|
||
|
|
}
|
||
|
|
catch (error) {
|
||
|
|
// Field index doesn't exist yet
|
||
|
|
}
|
||
|
|
return null;
|
||
|
|
}
|
||
|
|
/**
|
||
|
|
* Save field index to storage
|
||
|
|
*/
|
||
|
|
async saveFieldIndex(field, fieldIndex) {
|
||
|
|
const filename = this.getFieldIndexFilename(field);
|
||
|
|
const indexId = `__metadata_field_index__${filename}`;
|
||
|
|
await this.storage.saveMetadata(indexId, {
|
||
|
|
values: fieldIndex.values,
|
||
|
|
lastUpdated: fieldIndex.lastUpdated
|
||
|
|
});
|
||
|
|
// Invalidate cache
|
||
|
|
this.metadataCache.invalidatePattern(`field_index_${filename}`);
|
||
|
|
}
|
||
|
|
/**
|
||
|
|
* Get index statistics
|
||
|
|
*/
|
||
|
|
async getStats() {
|
||
|
|
const fields = new Set();
|
||
|
|
let totalEntries = 0;
|
||
|
|
let totalIds = 0;
|
||
|
|
for (const entry of this.indexCache.values()) {
|
||
|
|
fields.add(entry.field);
|
||
|
|
totalEntries++;
|
||
|
|
totalIds += entry.ids.size;
|
||
|
|
}
|
||
|
|
return {
|
||
|
|
totalEntries,
|
||
|
|
totalIds,
|
||
|
|
fieldsIndexed: Array.from(fields),
|
||
|
|
lastRebuild: 0, // TODO: track rebuild timestamp
|
||
|
|
indexSize: totalEntries * 100 // rough estimate
|
||
|
|
};
|
||
|
|
}
|
||
|
|
/**
|
||
|
|
* Rebuild entire index from scratch using pagination
|
||
|
|
* Non-blocking version that yields control back to event loop
|
||
|
|
*/
|
||
|
|
async rebuild() {
|
||
|
|
if (this.isRebuilding)
|
||
|
|
return;
|
||
|
|
this.isRebuilding = true;
|
||
|
|
try {
|
||
|
|
prodLog.info('🔄 Starting non-blocking metadata index rebuild with batch processing to prevent socket exhaustion...');
|
||
|
|
prodLog.info(`📊 Storage adapter: ${this.storage.constructor.name}`);
|
||
|
|
prodLog.info(`🔧 Batch processing available: ${!!this.storage.getMetadataBatch}`);
|
||
|
|
// Clear existing indexes
|
||
|
|
this.indexCache.clear();
|
||
|
|
this.dirtyEntries.clear();
|
||
|
|
this.fieldIndexes.clear();
|
||
|
|
this.dirtyFields.clear();
|
||
|
|
// Rebuild noun metadata indexes using pagination
|
||
|
|
let nounOffset = 0;
|
||
|
|
const nounLimit = 25; // Even smaller batches during initialization to prevent socket exhaustion
|
||
|
|
let hasMoreNouns = true;
|
||
|
|
let totalNounsProcessed = 0;
|
||
|
|
while (hasMoreNouns) {
|
||
|
|
const result = await this.storage.getNouns({
|
||
|
|
pagination: { offset: nounOffset, limit: nounLimit }
|
||
|
|
});
|
||
|
|
// CRITICAL FIX: Use batch metadata reading to prevent socket exhaustion
|
||
|
|
const nounIds = result.items.map(noun => noun.id);
|
||
|
|
let metadataBatch;
|
||
|
|
if (this.storage.getMetadataBatch) {
|
||
|
|
// Use batch reading if available (prevents socket exhaustion)
|
||
|
|
prodLog.info(`📦 Processing metadata batch ${Math.floor(totalNounsProcessed / nounLimit) + 1} (${nounIds.length} items)...`);
|
||
|
|
metadataBatch = await this.storage.getMetadataBatch(nounIds);
|
||
|
|
const successRate = ((metadataBatch.size / nounIds.length) * 100).toFixed(1);
|
||
|
|
prodLog.info(`✅ Batch loaded ${metadataBatch.size}/${nounIds.length} metadata objects (${successRate}% success)`);
|
||
|
|
}
|
||
|
|
else {
|
||
|
|
// Fallback to individual calls with strict concurrency control
|
||
|
|
prodLog.warn(`⚠️ FALLBACK: Storage adapter missing getMetadataBatch - using individual calls with concurrency limit`);
|
||
|
|
metadataBatch = new Map();
|
||
|
|
const CONCURRENCY_LIMIT = 3; // Very conservative limit
|
||
|
|
for (let i = 0; i < nounIds.length; i += CONCURRENCY_LIMIT) {
|
||
|
|
const batch = nounIds.slice(i, i + CONCURRENCY_LIMIT);
|
||
|
|
const batchPromises = batch.map(async (id) => {
|
||
|
|
try {
|
||
|
|
const metadata = await this.storage.getMetadata(id);
|
||
|
|
return { id, metadata };
|
||
|
|
}
|
||
|
|
catch (error) {
|
||
|
|
prodLog.debug(`Failed to read metadata for ${id}:`, error);
|
||
|
|
return { id, metadata: null };
|
||
|
|
}
|
||
|
|
});
|
||
|
|
const batchResults = await Promise.all(batchPromises);
|
||
|
|
for (const { id, metadata } of batchResults) {
|
||
|
|
if (metadata) {
|
||
|
|
metadataBatch.set(id, metadata);
|
||
|
|
}
|
||
|
|
}
|
||
|
|
// Yield between batches to prevent socket exhaustion
|
||
|
|
await this.yieldToEventLoop();
|
||
|
|
}
|
||
|
|
}
|
||
|
|
// Process the metadata batch
|
||
|
|
for (const noun of result.items) {
|
||
|
|
const metadata = metadataBatch.get(noun.id);
|
||
|
|
if (metadata) {
|
||
|
|
// Skip flush during rebuild for performance
|
||
|
|
await this.addToIndex(noun.id, metadata, true);
|
||
|
|
}
|
||
|
|
}
|
||
|
|
// Yield after processing the entire batch
|
||
|
|
await this.yieldToEventLoop();
|
||
|
|
totalNounsProcessed += result.items.length;
|
||
|
|
hasMoreNouns = result.hasMore;
|
||
|
|
nounOffset += nounLimit;
|
||
|
|
// Progress logging and event loop yield after each batch
|
||
|
|
if (totalNounsProcessed % 100 === 0 || !hasMoreNouns) {
|
||
|
|
prodLog.debug(`📊 Indexed ${totalNounsProcessed} nouns...`);
|
||
|
|
}
|
||
|
|
await this.yieldToEventLoop();
|
||
|
|
}
|
||
|
|
// Rebuild verb metadata indexes using pagination
|
||
|
|
let verbOffset = 0;
|
||
|
|
const verbLimit = 25; // Even smaller batches during initialization to prevent socket exhaustion
|
||
|
|
let hasMoreVerbs = true;
|
||
|
|
let totalVerbsProcessed = 0;
|
||
|
|
while (hasMoreVerbs) {
|
||
|
|
const result = await this.storage.getVerbs({
|
||
|
|
pagination: { offset: verbOffset, limit: verbLimit }
|
||
|
|
});
|
||
|
|
// CRITICAL FIX: Use batch verb metadata reading to prevent socket exhaustion
|
||
|
|
const verbIds = result.items.map(verb => verb.id);
|
||
|
|
let verbMetadataBatch;
|
||
|
|
if (this.storage.getVerbMetadataBatch) {
|
||
|
|
// Use batch reading if available (prevents socket exhaustion)
|
||
|
|
verbMetadataBatch = await this.storage.getVerbMetadataBatch(verbIds);
|
||
|
|
prodLog.debug(`📦 Batch loaded ${verbMetadataBatch.size}/${verbIds.length} verb metadata objects`);
|
||
|
|
}
|
||
|
|
else {
|
||
|
|
// Fallback to individual calls with strict concurrency control
|
||
|
|
verbMetadataBatch = new Map();
|
||
|
|
const CONCURRENCY_LIMIT = 3; // Very conservative limit to prevent socket exhaustion
|
||
|
|
for (let i = 0; i < verbIds.length; i += CONCURRENCY_LIMIT) {
|
||
|
|
const batch = verbIds.slice(i, i + CONCURRENCY_LIMIT);
|
||
|
|
const batchPromises = batch.map(async (id) => {
|
||
|
|
try {
|
||
|
|
const metadata = await this.storage.getVerbMetadata(id);
|
||
|
|
return { id, metadata };
|
||
|
|
}
|
||
|
|
catch (error) {
|
||
|
|
prodLog.debug(`Failed to read verb metadata for ${id}:`, error);
|
||
|
|
return { id, metadata: null };
|
||
|
|
}
|
||
|
|
});
|
||
|
|
const batchResults = await Promise.all(batchPromises);
|
||
|
|
for (const { id, metadata } of batchResults) {
|
||
|
|
if (metadata) {
|
||
|
|
verbMetadataBatch.set(id, metadata);
|
||
|
|
}
|
||
|
|
}
|
||
|
|
// Yield between batches to prevent socket exhaustion
|
||
|
|
await this.yieldToEventLoop();
|
||
|
|
}
|
||
|
|
}
|
||
|
|
// Process the verb metadata batch
|
||
|
|
for (const verb of result.items) {
|
||
|
|
const metadata = verbMetadataBatch.get(verb.id);
|
||
|
|
if (metadata) {
|
||
|
|
// Skip flush during rebuild for performance
|
||
|
|
await this.addToIndex(verb.id, metadata, true);
|
||
|
|
}
|
||
|
|
}
|
||
|
|
// Yield after processing the entire batch
|
||
|
|
await this.yieldToEventLoop();
|
||
|
|
totalVerbsProcessed += result.items.length;
|
||
|
|
hasMoreVerbs = result.hasMore;
|
||
|
|
verbOffset += verbLimit;
|
||
|
|
// Progress logging and event loop yield after each batch
|
||
|
|
if (totalVerbsProcessed % 100 === 0 || !hasMoreVerbs) {
|
||
|
|
prodLog.debug(`🔗 Indexed ${totalVerbsProcessed} verbs...`);
|
||
|
|
}
|
||
|
|
await this.yieldToEventLoop();
|
||
|
|
}
|
||
|
|
// Flush to storage with final yield
|
||
|
|
prodLog.debug('💾 Flushing metadata index to storage...');
|
||
|
|
await this.flush();
|
||
|
|
await this.yieldToEventLoop();
|
||
|
|
prodLog.info(`✅ Metadata index rebuild completed! Processed ${totalNounsProcessed} nouns and ${totalVerbsProcessed} verbs`);
|
||
|
|
prodLog.info(`🎯 Initial indexing may show minor socket timeouts - this is expected and doesn't affect data processing`);
|
||
|
|
}
|
||
|
|
finally {
|
||
|
|
this.isRebuilding = false;
|
||
|
|
}
|
||
|
|
}
|
||
|
|
/**
|
||
|
|
* Load index entry from storage using safe filenames
|
||
|
|
*/
|
||
|
|
async loadIndexEntry(key) {
|
||
|
|
try {
|
||
|
|
// Extract field and value from key
|
||
|
|
const [field, value] = key.split(':', 2);
|
||
|
|
const filename = this.getValueChunkFilename(field, value);
|
||
|
|
// Load from metadata indexes directory with safe filename
|
||
|
|
const indexId = `__metadata_index__${filename}`;
|
||
|
|
const data = await this.storage.getMetadata(indexId);
|
||
|
|
if (data) {
|
||
|
|
return {
|
||
|
|
field: data.field,
|
||
|
|
value: data.value,
|
||
|
|
ids: new Set(data.ids || []),
|
||
|
|
lastUpdated: data.lastUpdated || Date.now()
|
||
|
|
};
|
||
|
|
}
|
||
|
|
}
|
||
|
|
catch (error) {
|
||
|
|
// Index entry doesn't exist yet
|
||
|
|
}
|
||
|
|
return null;
|
||
|
|
}
|
||
|
|
/**
|
||
|
|
* Save index entry to storage using safe filenames
|
||
|
|
*/
|
||
|
|
async saveIndexEntry(key, entry) {
|
||
|
|
const data = {
|
||
|
|
field: entry.field,
|
||
|
|
value: entry.value,
|
||
|
|
ids: Array.from(entry.ids),
|
||
|
|
lastUpdated: entry.lastUpdated
|
||
|
|
};
|
||
|
|
// Extract field and value from key for safe filename generation
|
||
|
|
const [field, value] = key.split(':', 2);
|
||
|
|
const filename = this.getValueChunkFilename(field, value);
|
||
|
|
// Store metadata indexes with safe filename
|
||
|
|
const indexId = `__metadata_index__${filename}`;
|
||
|
|
await this.storage.saveMetadata(indexId, data);
|
||
|
|
}
|
||
|
|
/**
|
||
|
|
* Delete index entry from storage using safe filenames
|
||
|
|
*/
|
||
|
|
async deleteIndexEntry(key) {
|
||
|
|
try {
|
||
|
|
const [field, value] = key.split(':', 2);
|
||
|
|
const filename = this.getValueChunkFilename(field, value);
|
||
|
|
const indexId = `__metadata_index__${filename}`;
|
||
|
|
await this.storage.saveMetadata(indexId, null);
|
||
|
|
}
|
||
|
|
catch (error) {
|
||
|
|
// Entry might not exist
|
||
|
|
}
|
||
|
|
}
|
||
|
|
}
|
||
|
|
//# sourceMappingURL=metadataIndex.js.map
|