fix: update() field asymmetry causing index corruption

CRITICAL: Fixed metadata index corruption on update() operations where
removalMetadata only contained custom metadata + type, while entityForIndexing
contained ALL indexed fields. This caused 7 fields to accumulate on every
update, eventually making queries return 0 results.

- Fix removalMetadata to include all indexed fields (src/brainy.ts)
- Add validateIndexConsistency() and getIndexStats() public APIs
- Add auto-corruption detection and repair on startup
- Add getOrAssignSync() for EntityIdMapper persistence
- Add comprehensive regression tests
This commit is contained in:
David Snelling 2026-01-26 12:12:11 -08:00
parent 478c6e6342
commit a94219e720
5 changed files with 470 additions and 4 deletions

View file

@ -1147,11 +1147,29 @@ export class Brainy<T = any> implements BrainyInterface<T> {
}
// Operation 5-6: Update metadata index (remove old, add new)
// v6.2.1: Fix - Include type in removal metadata so noun index is properly updated
// existing.metadata only contains custom fields, not 'type' which maps to 'noun'
// v7.5.0 FIX: Include ALL indexed fields in removalMetadata (not just type)
// Previously, only metadata + type was removed, but entityForIndexing includes:
// confidence, weight, createdAt, updatedAt, service, data, createdBy
// This asymmetry caused 7 fields to accumulate on EVERY update, eventually
// making queries return 0 results (77x overcounting at scale).
//
// DEBUG: Log what we're removing and adding
// console.log('[UPDATE DEBUG] existing.metadata:', JSON.stringify(existing.metadata))
// console.log('[UPDATE DEBUG] entityForIndexing keys:', Object.keys(entityForIndexing))
//
// v7.5.0 FIX: removalMetadata must MATCH entityForIndexing structure
// entityForIndexing has: { type, confidence, ..., metadata: {...} }
// So removalMetadata must also have: { type, confidence, ..., metadata: {...} }
const removalMetadata = {
...existing.metadata,
type: existing.type // Include type so it maps to 'noun' during removal
type: existing.type,
confidence: existing.confidence,
weight: existing.weight,
createdAt: existing.createdAt,
updatedAt: existing.updatedAt, // CRITICAL: removes old timestamp
service: existing.service,
data: existing.data,
createdBy: existing.createdBy,
metadata: existing.metadata // CRITICAL: keep as nested 'metadata' property!
}
tx.addOperation(
new RemoveFromMetadataIndexOperation(this.metadataIndex, params.id, removalMetadata)
@ -4694,6 +4712,52 @@ export class Brainy<T = any> implements BrainyInterface<T> {
}
}
/**
* v7.5.0: Validate metadata index consistency and detect corruption
*
* Returns health status and recommendations for repair. Corruption typically
* manifests as high avg entries/entity (expected ~30, corrupted can be 100+)
* caused by the update() field asymmetry bug (fixed in v7.5.0).
*
* @returns Promise resolving to validation results
*
* @example
* const validation = await brain.validateIndexConsistency()
* if (!validation.healthy) {
* console.log(validation.recommendation)
* // Run brain.rebuildIndex() to repair
* }
*/
async validateIndexConsistency(): Promise<{
healthy: boolean
avgEntriesPerEntity: number
entityCount: number
indexEntryCount: number
recommendation: string | null
}> {
await this.ensureInitialized()
return this.metadataIndex.validateConsistency()
}
/**
* v7.5.0: Get metadata index statistics
*
* Returns detailed statistics about the metadata index including
* total entries, IDs indexed, and fields indexed.
*
* @returns Promise resolving to index statistics
*/
async getIndexStats(): Promise<{
totalEntries: number
totalIds: number
fieldsIndexed: string[]
lastRebuild: number
indexSize: number
}> {
await this.ensureInitialized()
return this.metadataIndex.getStats()
}
/**
* Get graph neighbors of an entity
*

View file

@ -86,6 +86,26 @@ export class EntityIdMapper {
return newId
}
/**
* v7.5.0: Get integer ID for UUID with immediate persistence guarantee
* Unlike getOrAssign(), this method flushes to storage immediately after assigning
* a new ID. This prevents UUIDint mapping divergence if the process crashes
* before a normal flush() occurs.
*
* Use this for critical operations where data integrity is paramount.
* Normal operations can use getOrAssign() with batched flushing for better performance.
*/
async getOrAssignSync(uuid: string): Promise<number> {
const id = this.getOrAssign(uuid)
// If a new ID was assigned, immediately persist to storage
if (this.dirty) {
await this.flush()
}
return id
}
/**
* Get UUID for integer ID
*/

View file

@ -223,6 +223,40 @@ export class MetadataIndexManager {
// Phase 1b: Sync loaded counts to fixed-size arrays
// Now correctly happens AFTER lazyLoadCounts() finishes
this.syncTypeCountsToFixed()
// v7.5.0: Detect index corruption and auto-rebuild if necessary
// The update() field asymmetry bug caused indexes to accumulate stale entries
// This check runs on startup to detect and repair corrupted indexes automatically
await this.detectAndRepairCorruption()
}
/**
* v7.5.0: Detect index corruption and automatically repair via rebuild
* This catches the update() field asymmetry bug that causes 7 fields to accumulate per update
* Corruption threshold: 100 avg entries/entity (expected ~30)
*/
private async detectAndRepairCorruption(): Promise<void> {
const validation = await this.validateConsistency()
if (!validation.healthy) {
prodLog.warn(`⚠️ Index corruption detected (${validation.avgEntriesPerEntity.toFixed(1)} avg entries/entity)`)
prodLog.warn('🔄 Auto-rebuilding index to repair...')
// Clear and rebuild
await this.clearAllIndexData()
await this.rebuild()
// Re-validate after rebuild
const postRebuild = await this.validateConsistency()
if (postRebuild.healthy) {
prodLog.info(`✅ Index rebuilt successfully (${postRebuild.avgEntriesPerEntity.toFixed(1)} avg entries/entity)`)
} else {
prodLog.error(
`❌ Index still appears corrupted after rebuild (${postRebuild.avgEntriesPerEntity.toFixed(1)} avg entries/entity). ` +
`This may indicate a different issue.`
)
}
}
}
/**
@ -2608,6 +2642,71 @@ export class MetadataIndexManager {
}
}
/**
* v7.5.0: Validate index consistency and detect corruption
* Returns health status and recommendations for repair
*
* Corruption typically manifests as high avg entries/entity (expected ~30, corrupted can be 100+)
* caused by the update() field asymmetry bug (fixed in v7.5.0)
*/
async validateConsistency(): Promise<{
healthy: boolean
avgEntriesPerEntity: number
entityCount: number
indexEntryCount: number
recommendation: string | null
}> {
const entityCount = this.idMapper.size
// If no entities, index is trivially healthy
if (entityCount === 0) {
return {
healthy: true,
avgEntriesPerEntity: 0,
entityCount: 0,
indexEntryCount: 0,
recommendation: null
}
}
// Count total index entries across all fields
let indexEntryCount = 0
for (const field of this.fieldIndexes.keys()) {
const sparseIndex = await this.loadSparseIndex(field)
if (sparseIndex) {
for (const chunkId of sparseIndex.getAllChunkIds()) {
const chunk = await this.chunkManager.loadChunk(field, chunkId)
if (chunk) {
for (const ids of chunk.entries.values()) {
indexEntryCount += ids.size
}
}
}
}
}
const avgEntriesPerEntity = indexEntryCount / entityCount
// Threshold: 100 entries/entity is clearly corrupted (expected ~30)
// This catches the update() asymmetry bug which causes 7 fields to accumulate per update
const CORRUPTION_THRESHOLD = 100
const healthy = avgEntriesPerEntity <= CORRUPTION_THRESHOLD
let recommendation: string | null = null
if (!healthy) {
recommendation = `Index corruption detected (${avgEntriesPerEntity.toFixed(1)} avg entries/entity, expected ~30). ` +
`Run brain.index.clearAllIndexData() followed by brain.index.rebuild() to repair.`
}
return {
healthy,
avgEntriesPerEntity,
entityCount,
indexEntryCount,
recommendation
}
}
/**
* Rebuild entire index from scratch using pagination
* Non-blocking version that yields control back to event loop