fix: exclude __words__ keyword index from corruption detection and getStats()

The __words__ keyword index stores 50-5000 entries per entity (one per
word), which inflated avg entries/entity well above the corruption
threshold of 100. This caused:

1. validateConsistency() to falsely detect corruption on every startup,
   triggering unnecessary clearAllIndexData() + rebuild() cycles
2. getStats() to log false "Metadata index may be corrupted" warnings
   and report inflated totalEntries/totalIds stats

Both methods now skip __words__ when counting, so stats and health
checks reflect metadata fields only (noun, type, createdAt, etc.).
Keyword search is unaffected since the __words__ field index itself
is not modified.
This commit is contained in:
David Snelling 2026-01-27 15:38:21 -08:00
parent 32dbdcec61
commit 364360d447
128 changed files with 5637 additions and 5682 deletions

View file

@ -42,17 +42,17 @@ export interface SmartCSVOptions extends FormatHandlerOptions {
csvDelimiter?: string
csvHeaders?: boolean
/** Progress callback (v3.39.0: Enhanced with performance metrics) */
/** Progress callback (Enhanced with performance metrics) */
onProgress?: (stats: {
processed: number
total: number
entities: number
relationships: number
/** Rows per second (v3.39.0) */
/** Rows per second */
throughput?: number
/** Estimated time remaining in ms (v3.39.0) */
/** Estimated time remaining in ms */
eta?: number
/** Current phase (v3.39.0) */
/** Current phase */
phase?: string
}) => void
}
@ -169,7 +169,7 @@ export class SmartCSVImporter {
}
// Parse CSV using existing handler
// v4.5.0: Pass progress hooks to handler for file parsing progress
// Pass progress hooks to handler for file parsing progress
const processedData = await this.csvHandler.process(buffer, {
...options,
csvDelimiter: opts.csvDelimiter,
@ -217,7 +217,7 @@ export class SmartCSVImporter {
// Detect column names
const columns = this.detectColumns(rows[0], opts)
// Process each row with BATCHED PARALLEL PROCESSING (v3.39.0)
// Process each row with BATCHED PARALLEL PROCESSING
const extractedRows: ExtractedRow[] = []
const entityMap = new Map<string, string>()
const stats = {
@ -375,7 +375,7 @@ export class SmartCSVImporter {
total: rows.length,
entities: extractedRows.reduce((sum, row) => sum + 1 + row.relatedEntities.length, 0),
relationships: extractedRows.reduce((sum, row) => sum + row.relationships.length, 0),
// Additional performance metrics (v3.39.0)
// Additional performance metrics
throughput: Math.round(rowsPerSecond * 10) / 10,
eta: Math.round(estimatedTimeRemaining),
phase: 'extracting'