fix: resolve import hang and index rebuild data loss bugs
Critical Bug Fixes (v3.43.2): Bug #1 - Import Infinite Loop: - Fix placeholder entity infinite loop in ImportCoordinator - Use exact matching instead of fuzzy .includes() for entity names - Search entities array (not rows) for existing placeholders - Add duplicate relationship prevention in brain.relate() Bug #2 - Index Rebuild File Discovery: - Fix fileSystemStorage to scan sharded subdirectories - Update getAllNodes() to use getAllShardedFiles() - Update getAllEdges() to use getAllShardedFiles() - Update getNodesByNounType() to use getAllShardedFiles() - Fix getStorageStatus() to use O(1) persisted counts Additional Improvements: - Add brain.flush() API for explicit index persistence - Make GraphAdjacencyIndex.flush() public - Add auto-flush at end of import pipeline - Update duplicate relationship test to expect deduplication Files Modified: - src/storage/adapters/fileSystemStorage.ts - src/import/ImportCoordinator.ts - src/brainy.ts - src/graph/graphAdjacencyIndex.ts - tests/unit/brainy/relate.test.ts
This commit is contained in:
parent
165def11a9
commit
dcf20ffa1d
6 changed files with 184 additions and 83 deletions
|
|
@ -737,6 +737,21 @@ export class Brainy<T = any> implements BrainyInterface<T> {
|
|||
throw new Error(`Target entity ${params.to} not found`)
|
||||
}
|
||||
|
||||
// CRITICAL FIX (v3.43.2): Check for duplicate relationships
|
||||
// This prevents infinite loops where same relationship is created repeatedly
|
||||
// Bug #1 showed incrementing verb counts (7→8→9...) indicating duplicates
|
||||
const existingVerbs = await this.storage.getVerbsBySource(params.from)
|
||||
const duplicate = existingVerbs.find(v =>
|
||||
v.targetId === params.to &&
|
||||
v.type === params.type
|
||||
)
|
||||
|
||||
if (duplicate) {
|
||||
// Relationship already exists - return existing ID instead of creating duplicate
|
||||
console.log(`[DEBUG] Skipping duplicate relationship: ${params.from} → ${params.to} (${params.type})`)
|
||||
return duplicate.id
|
||||
}
|
||||
|
||||
// Generate ID
|
||||
const id = uuidv4()
|
||||
|
||||
|
|
@ -1957,6 +1972,57 @@ export class Brainy<T = any> implements BrainyInterface<T> {
|
|||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Flush all indexes and caches to persistent storage
|
||||
* CRITICAL FIX (v3.43.2): Ensures data survives server restarts
|
||||
*
|
||||
* Flushes all 4 core indexes:
|
||||
* 1. Storage counts (entity/verb counts by type)
|
||||
* 2. Metadata index (field indexes + EntityIdMapper)
|
||||
* 3. Graph adjacency index (relationship cache)
|
||||
* 4. HNSW vector index (no flush needed - saves directly)
|
||||
*
|
||||
* @example
|
||||
* // Flush after bulk operations
|
||||
* await brain.import('./data.xlsx')
|
||||
* await brain.flush()
|
||||
*
|
||||
* // Flush before shutdown
|
||||
* process.on('SIGTERM', async () => {
|
||||
* await brain.flush()
|
||||
* process.exit(0)
|
||||
* })
|
||||
*/
|
||||
async flush(): Promise<void> {
|
||||
await this.ensureInitialized()
|
||||
|
||||
console.log('🔄 Flushing Brainy indexes and caches to disk...')
|
||||
|
||||
const startTime = Date.now()
|
||||
|
||||
// Flush all components in parallel for performance
|
||||
await Promise.all([
|
||||
// 1. Flush storage adapter counts (entity/verb counts by type)
|
||||
(async () => {
|
||||
if (this.storage && typeof (this.storage as any).flushCounts === 'function') {
|
||||
await (this.storage as any).flushCounts()
|
||||
}
|
||||
})(),
|
||||
|
||||
// 2. Flush metadata index (field indexes + EntityIdMapper)
|
||||
this.metadataIndex.flush(),
|
||||
|
||||
// 3. Flush graph adjacency index (relationship cache)
|
||||
// Note: Graph structure is already persisted via storage.saveVerb() calls
|
||||
// This just flushes the in-memory cache for performance
|
||||
this.graphIndex.flush()
|
||||
])
|
||||
|
||||
const elapsed = Date.now() - startTime
|
||||
|
||||
console.log(`✅ All indexes flushed to disk in ${elapsed}ms`)
|
||||
}
|
||||
|
||||
/**
|
||||
* Efficient Pagination API - Production-scale pagination using index-first approach
|
||||
* Automatically optimizes based on query type and applies pagination at the index level
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue