/** * @module graph/graphAdjacencyIndex * @description GraphAdjacencyIndex — billion-scale graph traversal engine. * * Adjacency lives in two verb-id LSM trees (sourceId → verbIds, * targetId → verbIds) filtered through an in-memory live-verb tombstone set, * with full verb objects loaded on demand through the unified cache. The * verb set is the single source of truth: neighbor reads derive from live * verbs, so removals are visible to every read path immediately. ID-only * in-memory tracking keeps the resident footprint to the verb-id set (~8 * bytes per relationship) regardless of relationship payload size. * * **8.0 u64 boundary:** this class implements the BigInt * {@link GraphIndexProvider} contract — entity ints in, entity/verb ints out. * Internally everything stays string-keyed (UUID-keyed LSM trees) and u32: * entity-int params resolve to UUIDs via the shared entity-id mapper * (`getUuid(Number(big))` — lossless under the `EntityIdSpaceExceeded` u32 * guard), and returns convert back with `BigInt(getOrAssign(uuid))`. Verb ints * come from a small in-process interning map (see * {@link GraphAdjacencyIndex.verbIntsToIds}); the interning is derived state, * never persisted — `rebuild()` / the verb-id-set recovery path re-derive it * from storage, which the JS index already does on cold start. */ import { GraphVerb, StorageAdapter } from '../coreTypes.js' import { UnifiedCache, getGlobalCache } from '../utils/unifiedCache.js' import { prodLog } from '../utils/logger.js' import { LSMTree } from './lsm/LSMTree.js' import type { GraphIndexProvider } from '../plugin.js' export interface GraphIndexConfig { maxIndexSize?: number // Default: 100000 rebuildThreshold?: number // Default: 0.1 autoOptimize?: boolean // Default: true flushInterval?: number // Default: 30000ms } /** * @description The minimal UUID ↔ int resolver surface the JS graph index * needs at its BigInt boundary. Satisfied by `EntityIdMapper` (and by whatever * `MetadataIndexManager.getIdMapper()` / a native metadata index returns) — * declared structurally here so the graph layer doesn't import the metadata * layer. */ export interface GraphEntityIdResolver { /** Resolve a UUID to its int, assigning a new one if absent (write path). */ getOrAssign(uuid: string): number /** Resolve a UUID to its int without assigning (read path). */ getInt(uuid: string): number | undefined /** Reverse-resolve an int to its UUID (`undefined` = unknown/deleted). */ getUuid(intId: number): string | undefined } export interface GraphIndexStats { totalRelationships: number sourceNodes: number targetNodes: number memoryUsage: number // in bytes lastRebuild: number rebuildTime: number // in ms } /** * GraphAdjacencyIndex - Billion-scale adjacency list with LSM-tree storage * * Disk-resident verb-id adjacency (LSM trees with bloom filter optimization) * plus an in-memory live-verb tombstone set; neighbor reads derive from live * verbs via cache-assisted batch loads — O(node degree) per lookup, correct * under verb removal. */ export class GraphAdjacencyIndex implements GraphIndexProvider { // LSM-tree storage for verb ID lookups — the single adjacency source of // truth. Neighbor reads derive from these (live-verb filtered via // verbIdSet) so removeVerb tombstones are honored by EVERY read path; // a separate entity→entity edge tree cannot be tombstone-filtered (it // carries no verb ids) and previously served stale neighbors forever. private lsmTreeVerbsBySource: LSMTree // sourceId -> verbIds private lsmTreeVerbsByTarget: LSMTree // targetId -> verbIds // ID-only tracking for billion-scale memory optimization // Previous: Map stored full objects (128GB @ 1B verbs) // Now: Set stores only IDs (~100KB @ 1B verbs) = 1,280,000x reduction private verbIdSet = new Set() // Verb-id interning for the BigInt boundary (8.0 u64 contract). // Process-lifetime derived state: assigned on addVerb / rebuild / // verb-id-set recovery, NEVER persisted. removeVerb keeps the entry so a // verb int stays stable (and resolvable) for the index's lifetime. private verbIdToInt = new Map() private verbIntToId: string[] = [] // Shared UUID ↔ int resolver for entity ints at the BigInt boundary. // Threaded in by the coordinator (brainy.ts) from the metadata index's // idMapper — see setEntityIdMapper(). private entityIdMapper?: GraphEntityIdResolver // Infrastructure integration private storage: StorageAdapter private unifiedCache: UnifiedCache private config: Required // Performance optimization private isRebuilding = false private flushTimer?: NodeJS.Timeout private rebuildStartTime = 0 private totalRelationshipsIndexed = 0 // Production-scale relationship counting by type private relationshipCountsByType = new Map() // Initialization flag private initialized = false /** * Check if index is initialized and ready for use */ get isInitialized(): boolean { return this.initialized } constructor( storage: StorageAdapter, config: GraphIndexConfig = {}, entityIdMapper?: GraphEntityIdResolver ) { this.storage = storage this.entityIdMapper = entityIdMapper this.config = { maxIndexSize: config.maxIndexSize ?? 100000, rebuildThreshold: config.rebuildThreshold ?? 0.1, autoOptimize: config.autoOptimize ?? true, flushInterval: config.flushInterval ?? 30000 } // Create LSM-trees for verb ID lookups (billion-scale optimization) this.lsmTreeVerbsBySource = new LSMTree(storage, { memTableThreshold: 100000, storagePrefix: 'graph-lsm-verbs-source', enableCompaction: true }) this.lsmTreeVerbsByTarget = new LSMTree(storage, { memTableThreshold: 100000, storagePrefix: 'graph-lsm-verbs-target', enableCompaction: true }) // Use SAME UnifiedCache as MetadataIndexManager for coordinated memory management this.unifiedCache = getGlobalCache() prodLog.info('GraphAdjacencyIndex initialized with LSM-tree storage (2 LSM-trees total)') } /** * @description Thread in the shared UUID ↔ int resolver used for entity-int * conversion at the BigInt boundary. The coordinator (`brainy.ts`) calls this * with `metadataIndex.getIdMapper()` right after index construction (init, * fork, and checkout paths) so reads/writes share one int universe with the * metadata index. Idempotent; safe to call again after a branch switch. * @param mapper - The shared entity-id resolver. * @returns Nothing. */ setEntityIdMapper(mapper: GraphEntityIdResolver): void { this.entityIdMapper = mapper } /** * Resolve the entity-id mapper or fail loudly. The BigInt read methods are * meaningless without a shared int universe — a missing mapper is a wiring * bug, not a recoverable condition. */ private requireEntityIdMapper(): GraphEntityIdResolver { if (!this.entityIdMapper) { throw new Error( 'GraphAdjacencyIndex: entityIdMapper not wired. The coordinator must ' + 'call setEntityIdMapper(metadataIndex.getIdMapper()) (or pass it to ' + 'the constructor) before BigInt-boundary reads.' ) } return this.entityIdMapper } /** * Intern a verb-id string, assigning the next sequential u32 on first sight. * Append-only for the index's lifetime — removeVerb keeps the entry so verb * ints handed to callers stay resolvable. */ private internVerbId(verbId: string): number { const existing = this.verbIdToInt.get(verbId) if (existing !== undefined) return existing const next = this.verbIntToId.length this.verbIdToInt.set(verbId, next) this.verbIntToId.push(verbId) return next } /** * Initialize the graph index (lazy initialization) * Added defensive auto-rebuild check for verbIdSet consistency */ private async ensureInitialized(): Promise { if (this.initialized) { return } await this.lsmTreeVerbsBySource.init() await this.lsmTreeVerbsByTarget.init() // Defensive check - if LSM-trees have data but verbIdSet is empty, // the index was created without proper rebuild (shouldn't happen with singleton // pattern but protects against edge cases and future refactoring) const lsmTreeSize = this.lsmTreeVerbsBySource.size() if (lsmTreeSize > 0 && this.verbIdSet.size === 0) { prodLog.warn( `GraphAdjacencyIndex: LSM-trees have ${lsmTreeSize} relationships but verbIdSet is empty. ` + `Triggering auto-rebuild to restore consistency.` ) // Note: We don't await rebuild() here to avoid infinite loop // (rebuild calls ensureInitialized). Instead, we'll populate verbIdSet // by loading all verb IDs from storage. await this.populateVerbIdSetFromStorage() } // Start auto-flush timer after initialization this.startAutoFlush() this.initialized = true } /** * Populate verbIdSet from storage without full rebuild * Lighter weight than full rebuild - only loads verb IDs, not all verb data * @private */ private async populateVerbIdSetFromStorage(): Promise { prodLog.info('GraphAdjacencyIndex: Populating verbIdSet from storage...') const startTime = Date.now() // Use pagination to load all verb IDs let hasMore = true let cursor: string | undefined = undefined let count = 0 while (hasMore) { const result = await this.storage.getVerbs({ pagination: { limit: 10000, cursor } }) for (const verb of result.items) { this.verbIdSet.add(verb.id) // Re-derive the verb-int interning (process-lifetime state, never // persisted — this recovery path is one of the two cold-start sources, // alongside rebuild()). this.internVerbId(verb.id) // Also update counts const verbType = verb.verb || 'unknown' this.relationshipCountsByType.set( verbType, (this.relationshipCountsByType.get(verbType) || 0) + 1 ) count++ } hasMore = result.hasMore cursor = result.nextCursor } const elapsed = Date.now() - startTime prodLog.info(`GraphAdjacencyIndex: Populated verbIdSet with ${count} verb IDs in ${elapsed}ms`) } /** * @description Core API — neighbor lookup (BigInt boundary). Neighbors * derive from the node's live verbs (tombstone-filtered verb-id LSM trees * + unified-cache batch loads), so `removeVerb()` is honored immediately; * cost is O(node degree) with cache-assisted verb resolution. Pagination * support for high-degree nodes. The entity int is resolved to a UUID via * the shared mapper (unknown int → empty result) and neighbor UUIDs * convert back via `BigInt(getOrAssign(uuid))`. * * @param id - Entity int to get neighbors for (from the shared idMapper). * @param options - Optional direction ('both' default) + limit/offset. * @returns Neighbor entity ints (paginated if limit/offset specified). * * @example * // Get all neighbors of an entity int * const all = await graphIndex.getNeighbors(42n) * * @example * // Get first 50 outgoing neighbors * const page1 = await graphIndex.getNeighbors(42n, { direction: 'out', limit: 50 }) */ async getNeighbors( id: bigint, options?: { direction?: 'in' | 'out' | 'both' limit?: number offset?: number } ): Promise { await this.ensureInitialized() const mapper = this.requireEntityIdMapper() const uuid = mapper.getUuid(Number(id)) if (uuid === undefined) { // Unknown/deleted entity int — no edges by definition. return [] } const neighborUuids = await this.getNeighborUuids(uuid, options) return neighborUuids.map(neighborUuid => BigInt(mapper.getOrAssign(neighborUuid))) } /** * String-keyed neighbor lookup — the internal implementation behind * {@link getNeighbors}. Kept UUID-based because the LSM trees are keyed by * UUID; only the public contract speaks BigInt. * * Neighbors are derived from the node's **live** verbs (the verb-id LSM * trees filtered through the `verbIdSet` tombstones, then batch-loaded via * the unified cache) rather than from a separate entity→entity edge tree. * That keeps `removeVerb()` visible to traversal: an entity-level edge * entry carries no verb id, so it could never be tombstone-filtered, and * a removed relationship would keep its endpoints "connected" forever. */ private async getNeighborUuids( id: string, options?: { direction?: 'in' | 'out' | 'both' limit?: number offset?: number } ): Promise { const startTime = performance.now() const direction = options?.direction || 'both' const neighbors = new Set() if (direction !== 'in') { for (const verb of (await this.liveVerbsForNode(this.lsmTreeVerbsBySource, id)).values()) { neighbors.add(verb.targetId) } } if (direction !== 'out') { for (const verb of (await this.liveVerbsForNode(this.lsmTreeVerbsByTarget, id)).values()) { neighbors.add(verb.sourceId) } } // Convert to array for pagination let result = Array.from(neighbors) // Apply pagination if requested if (options?.limit !== undefined || options?.offset !== undefined) { const offset = options?.offset || 0 const limit = options?.limit !== undefined ? options.limit : result.length result = result.slice(offset, offset + limit) } const elapsed = performance.now() - startTime // Performance assertion - should be sub-5ms with LSM-tree if (elapsed > 5.0) { prodLog.warn(`GraphAdjacencyIndex: Slow neighbor lookup for ${id}: ${elapsed.toFixed(2)}ms`) } return result } /** * @description The live (non-tombstoned) verb objects adjacent to `nodeId` * in one verb-id LSM tree: read the node's verb-id list, drop ids deleted * by `removeVerb()` (the `verbIdSet` tombstone filter — same rule as * {@link verbIdsToPaginatedInts}), and batch-load the survivors through the * unified cache. Shared by {@link getNeighborUuids} for both directions. * @param tree - `lsmTreeVerbsBySource` (out-edges) or `lsmTreeVerbsByTarget` (in-edges). * @param nodeId - The node's UUID (LSM trees are UUID-keyed). * @returns The node's live verbs in that direction, keyed by verb id. */ private async liveVerbsForNode(tree: LSMTree, nodeId: string): Promise> { const verbIds = (await tree.get(nodeId)) || [] const liveIds = [...new Set(verbIds)].filter(verbId => this.verbIdSet.has(verbId)) if (liveIds.length === 0) return new Map() return this.getVerbsBatchCached(liveIds) } /** * @description Verb ints for all edges originating at `sourceInt` (BigInt * boundary). O(log n) LSM-tree lookup with bloom filter optimization; * filters out deleted verb IDs (tombstone deletion workaround); pagination * support for entities with many relationships. Unknown entity int → empty * result. Resolve returned ints back to verb-id strings with * {@link verbIntsToIds}. * * @param sourceInt - Source entity int (from the shared idMapper). * @param options - Optional limit/offset pagination. * @returns Verb ints originating from this source (excluding deleted). * * @example * const verbInts = await graphIndex.getVerbIdsBySource(42n, { limit: 50 }) * const verbIds = await graphIndex.verbIntsToIds(verbInts) */ async getVerbIdsBySource( sourceInt: bigint, options?: { limit?: number offset?: number } ): Promise { await this.ensureInitialized() const mapper = this.requireEntityIdMapper() const sourceId = mapper.getUuid(Number(sourceInt)) if (sourceId === undefined) return [] const startTime = performance.now() const verbIds = await this.lsmTreeVerbsBySource.get(sourceId) const elapsed = performance.now() - startTime // Performance assertion - should be sub-5ms with LSM-tree if (elapsed > 5.0) { prodLog.warn(`GraphAdjacencyIndex: Slow getVerbIdsBySource for ${sourceId}: ${elapsed.toFixed(2)}ms`) } return this.verbIdsToPaginatedInts(verbIds || [], options) } /** * @description Verb ints for all edges pointing at `targetInt` (BigInt * boundary). O(log n) LSM-tree lookup with bloom filter optimization; * filters out deleted verb IDs (tombstone deletion workaround); pagination * support for popular target entities. Unknown entity int → empty result. * Resolve returned ints back to verb-id strings with {@link verbIntsToIds}. * * @param targetInt - Target entity int (from the shared idMapper). * @param options - Optional limit/offset pagination. * @returns Verb ints pointing to this target (excluding deleted). * * @example * const verbInts = await graphIndex.getVerbIdsByTarget(42n, { limit: 50 }) * const verbIds = await graphIndex.verbIntsToIds(verbInts) */ async getVerbIdsByTarget( targetInt: bigint, options?: { limit?: number offset?: number } ): Promise { await this.ensureInitialized() const mapper = this.requireEntityIdMapper() const targetId = mapper.getUuid(Number(targetInt)) if (targetId === undefined) return [] const startTime = performance.now() const verbIds = await this.lsmTreeVerbsByTarget.get(targetId) const elapsed = performance.now() - startTime // Performance assertion - should be sub-5ms with LSM-tree if (elapsed > 5.0) { prodLog.warn(`GraphAdjacencyIndex: Slow getVerbIdsByTarget for ${targetId}: ${elapsed.toFixed(2)}ms`) } return this.verbIdsToPaginatedInts(verbIds || [], options) } /** * Shared tail for the verb-int read methods: drop tombstoned ids (LSM trees * retain all ids; verbIdSet tracks deletions), apply pagination, and intern * the survivors to verb ints. Interning on the read path is safe — ids are * assigned deterministically within the process lifetime and never persisted. */ private verbIdsToPaginatedInts( allIds: string[], options?: { limit?: number; offset?: number } ): bigint[] { let result = allIds.filter(id => this.verbIdSet.has(id)) // Apply pagination if requested if (options?.limit !== undefined || options?.offset !== undefined) { const offset = options?.offset || 0 const limit = options?.limit !== undefined ? options.limit : result.length result = result.slice(offset, offset + limit) } return result.map(id => BigInt(this.internVerbId(id))) } /** * @description Batch reverse resolver: verb ints → verb-id strings (the * REQUIRED half of the 8.0 contract Brainy's warm cache feeds from). Reads * the in-process interning map populated by `addVerb`, `rebuild()`, and the * verb-id-set recovery path — the JS index derives the interning from * storage on cold start, so no sidecar persistence exists or is needed. * @param verbInts - Verb ints as returned by the verb-int read methods. * @returns One entry per input, order-preserving; `null` for unknown ints. */ async verbIntsToIds(verbInts: bigint[]): Promise<(string | null)[]> { return verbInts.map(verbInt => this.verbIntToId[Number(verbInt)] ?? null) } /** * Get verb from cache or storage - Billion-scale memory optimization * Uses UnifiedCache with LRU eviction instead of storing all verbs in memory * * @param verbId Verb ID to retrieve * @returns GraphVerb or null if not found */ async getVerbCached(verbId: string): Promise { const cacheKey = `graph:verb:${verbId}` // Try to get from cache, load if not present const verb = await this.unifiedCache.get(cacheKey, async () => { // Load from storage (fallback if not in cache) const loadedVerb = await this.storage.getVerb(verbId) // Cache the loaded verb with metadata if (loadedVerb) { this.unifiedCache.set(cacheKey, loadedVerb, 'other', 128, 50) // 128 bytes estimated size, 50ms rebuild cost } return loadedVerb }) return verb } /** * Batch get multiple verbs with caching * * **Performance**: Eliminates N+1 pattern for verb loading * - Current: N × getVerbCached() = N × 50ms on GCS = 250ms for 5 verbs * - Batched: 1 × getVerbsBatchCached() = 1 × 50ms on GCS = 50ms (**5x faster**) * * **Use cases:** * - relate() duplicate checking (check multiple existing relationships) * - Loading relationship chains * - Pre-loading verbs for analysis * * **Cache behavior:** * - Checks UnifiedCache first (fast path) * - Batch-loads uncached verbs from storage * - Caches loaded verbs for future access * * @param verbIds Array of verb IDs to fetch * @returns Map of verbId → GraphVerb (only successful reads included) * */ async getVerbsBatchCached(verbIds: string[]): Promise> { const results = new Map() const uncached: string[] = [] // Phase 1: Check cache for each verb for (const verbId of verbIds) { const cacheKey = `graph:verb:${verbId}` const cached = this.unifiedCache.getSync(cacheKey) if (cached) { results.set(verbId, cached) } else { uncached.push(verbId) } } // Phase 2: Batch-load uncached verbs from storage if (uncached.length > 0 && this.storage.getVerbsBatch) { const loadedVerbs = await this.storage.getVerbsBatch(uncached) for (const [verbId, verb] of loadedVerbs.entries()) { const cacheKey = `graph:verb:${verbId}` // Cache the loaded verb with metadata // Note: HNSWVerbWithMetadata is compatible with GraphVerb (both interfaces) this.unifiedCache.set(cacheKey, verb as any, 'other', 128, 50) // 128 bytes estimated size, 50ms rebuild cost results.set(verbId, verb as any) } } return results } /** * Get total relationship count - O(1) operation */ size(): number { // Use LSM-tree size for accurate count (one entry per indexed verb) return this.lsmTreeVerbsBySource.size() } /** * Get relationship count by type - O(1) operation using existing tracking */ getRelationshipCountByType(type: string): number { return this.relationshipCountsByType.get(type) || 0 } /** * Get total relationship count - O(1) operation */ getTotalRelationshipCount(): number { return this.verbIdSet.size } /** * Get all relationship types and their counts - O(1) operation */ getAllRelationshipCounts(): Map { return new Map(this.relationshipCountsByType) } /** * Get relationship statistics with enhanced counting information */ getRelationshipStats(): { totalRelationships: number relationshipsByType: Record uniqueSourceNodes: number uniqueTargetNodes: number totalNodes: number } { const totalRelationships = this.lsmTreeVerbsBySource.size() const relationshipsByType = Object.fromEntries(this.relationshipCountsByType) // Note: Exact unique node counts would require full LSM-tree scan // Using verbIdSet (ID-only tracking) for memory efficiency const uniqueSourceNodes = this.verbIdSet.size const uniqueTargetNodes = this.verbIdSet.size const totalNodes = this.verbIdSet.size return { totalRelationships, relationshipsByType, uniqueSourceNodes, uniqueTargetNodes, totalNodes } } /** * @description Add a relationship to the index (BigInt boundary). The * coordinator resolves both endpoint ints via `idMapper.getOrAssign` and * mirrors them onto `verb.sourceInt`/`verb.targetInt` before calling. The * JS index keys its LSM trees by the verb's endpoint UUIDs, so the int * params carry no extra information here — they exist for contract parity * with native providers whose trees are int-keyed. * @param verb - The verb to index (endpoint UUIDs are authoritative). * @param sourceInt - The source entity's interned int (contract parity). * @param targetInt - The target entity's interned int (contract parity). * @returns The interned verb int for `verb.id` (stable for the index lifetime). */ async addVerb(verb: GraphVerb, sourceInt: bigint, targetInt: bigint): Promise { await this.ensureInitialized() return BigInt(await this.indexVerb(verb)) } /** * String-keyed indexing core shared by {@link addVerb} and {@link rebuild}. * Returns the interned verb int. */ private async indexVerb(verb: GraphVerb): Promise { const startTime = performance.now() // Track verb ID (memory-efficient: IDs only, full objects loaded on-demand via UnifiedCache) this.verbIdSet.add(verb.id) const verbInt = this.internVerbId(verb.id) // Seed the unified cache with the authoritative verb object: neighbor // reads ({@link liveVerbsForNode}) resolve verbs through the cache with a // storage fallback, so an indexed verb is immediately traversable — and // freshly written verbs are the likeliest next reads. this.unifiedCache.set(`graph:verb:${verb.id}`, verb, 'other', 128, 50) // Add to the verb-id adjacency LSM-trees (the single adjacency source of // truth — neighbor and verb-id reads both derive from these). await this.lsmTreeVerbsBySource.add(verb.sourceId, verb.id) await this.lsmTreeVerbsByTarget.add(verb.targetId, verb.id) // Update type-specific counts atomically const verbType = verb.type || 'unknown' this.relationshipCountsByType.set( verbType, (this.relationshipCountsByType.get(verbType) || 0) + 1 ) const elapsed = performance.now() - startTime this.totalRelationshipsIndexed++ // Performance assertion if (elapsed > 10.0) { prodLog.warn(`GraphAdjacencyIndex: Slow addVerb for ${verb.id}: ${elapsed.toFixed(2)}ms`) } return verbInt } /** * @description Remove a relationship from the index by its id string. * Deletion is tombstone-based: the verb id leaves `verbIdSet`, and every * read path (verb-id reads via {@link verbIdsToPaginatedInts}, neighbor * reads via {@link liveVerbsForNode}) filters the append-only LSM trees * through that set — so the verb disappears from traversal immediately * while the trees stay immutable. The verb's interned int is intentionally * retained so previously returned verb ints stay resolvable via * {@link verbIntsToIds}. * @param verbId - The verb's UUID string. * @returns Resolves once the verb no longer appears in reads. */ async removeVerb(verbId: string): Promise { await this.ensureInitialized() // Load verb from cache/storage to get type info const verb = await this.getVerbCached(verbId) if (!verb) return const startTime = performance.now() // Remove from verb ID set this.verbIdSet.delete(verbId) // Update type-specific counts atomically const verbType = verb.type || 'unknown' const currentCount = this.relationshipCountsByType.get(verbType) || 0 if (currentCount > 1) { this.relationshipCountsByType.set(verbType, currentCount - 1) } else { this.relationshipCountsByType.delete(verbType) } const elapsed = performance.now() - startTime // Performance assertion if (elapsed > 5.0) { prodLog.warn(`GraphAdjacencyIndex: Slow removeVerb for ${verbId}: ${elapsed.toFixed(2)}ms`) } } /** * Rebuild entire index from storage * Critical for cold starts and data consistency */ async rebuild(): Promise { await this.ensureInitialized() if (this.isRebuilding) { prodLog.warn('GraphAdjacencyIndex: Rebuild already in progress') return } this.isRebuilding = true this.rebuildStartTime = Date.now() try { prodLog.info('GraphAdjacencyIndex: Starting rebuild with LSM-tree...') // Clear current index this.verbIdSet.clear() this.totalRelationshipsIndexed = 0 // CRITICAL FIX - Clear relationship counts to prevent accumulation this.relationshipCountsByType.clear() // Re-derive verb-int interning from scratch — it's process-lifetime // derived state (never persisted), so a rebuild starts a fresh // generation of verb ints alongside the fresh verbIdSet. this.verbIdToInt.clear() this.verbIntToId = [] // Note: LSM-trees will be recreated from storage via their own initialization // Verb data will be loaded on-demand via UnifiedCache // Brainy 8.0: storage is always local (filesystem or memory — the // cloud adapters were removed). Load all verbs at once. const storageType = this.storage?.constructor.name || '' let totalVerbs = 0 prodLog.info(`GraphAdjacencyIndex: Load all verbs at once (${storageType})`) const result = await this.storage.getVerbs({ pagination: { limit: 10000000 } // Effectively unlimited for local storage }) for (const verb of result.items) { const graphVerb: GraphVerb = { id: verb.id, sourceId: verb.sourceId, targetId: verb.targetId, vector: verb.vector, verb: verb.verb, createdAt: { seconds: Math.floor(verb.createdAt / 1000), nanoseconds: (verb.createdAt % 1000) * 1000000 }, updatedAt: { seconds: Math.floor(verb.updatedAt / 1000), nanoseconds: (verb.updatedAt % 1000) * 1000000 }, createdBy: verb.createdBy || { augmentation: 'unknown', version: '0.0.0' }, service: verb.service, data: verb.data, embedding: verb.vector, confidence: verb.confidence, weight: verb.weight } await this.indexVerb(graphVerb) totalVerbs++ } prodLog.info( `GraphAdjacencyIndex: Loaded ${totalVerbs.toLocaleString()} verbs (${storageType})` ) const rebuildTime = Date.now() - this.rebuildStartTime const memoryUsage = this.calculateMemoryUsage() prodLog.info(`GraphAdjacencyIndex: Rebuild complete in ${rebuildTime}ms`) prodLog.info(` - Total relationships: ${totalVerbs}`) prodLog.info(` - Memory usage: ${(memoryUsage / 1024 / 1024).toFixed(1)}MB`) prodLog.info(` - LSM-tree stats:`, this.lsmTreeVerbsBySource.getStats()) } finally { this.isRebuilding = false } } /** * Calculate current memory usage (LSM-tree mostly on disk) */ private calculateMemoryUsage(): number { let bytes = 0 // LSM-tree memory (MemTable + bloom filters + zone maps) const sourceStats = this.lsmTreeVerbsBySource.getStats() const targetStats = this.lsmTreeVerbsByTarget.getStats() bytes += sourceStats.memTableMemory bytes += targetStats.memTableMemory // Verb ID set (memory-efficient: IDs only, ~8 bytes per ID pointer) // Previous verbIndex Map stored full objects (128 bytes each = 128GB @ 1B verbs) // Now: verbIdSet stores only IDs (~8 bytes each = ~100KB @ 1B verbs) = 1,280,000x reduction bytes += this.verbIdSet.size * 8 // Note: Bloom filters and zone maps are in LSM-tree MemTable memory // Full verb objects loaded on-demand via UnifiedCache with LRU eviction return bytes } /** * Get comprehensive statistics */ getStats(): GraphIndexStats { const sourceStats = this.lsmTreeVerbsBySource.getStats() const targetStats = this.lsmTreeVerbsByTarget.getStats() return { totalRelationships: this.size(), sourceNodes: sourceStats.sstableCount, targetNodes: targetStats.sstableCount, memoryUsage: this.calculateMemoryUsage(), lastRebuild: this.rebuildStartTime, rebuildTime: this.isRebuilding ? Date.now() - this.rebuildStartTime : 0 } } /** * Start auto-flush timer */ private startAutoFlush(): void { this.flushTimer = setInterval(async () => { await this.flush() }, this.config.flushInterval) } /** * Flush LSM-tree MemTables to disk * CRITICAL FIX: Now public so it can be called from brain.flush() */ async flush(): Promise { if (!this.initialized) { return } const startTime = Date.now() // Flush both LSM-trees in parallel (MemTables → SSTables on disk) await Promise.all([ this.lsmTreeVerbsBySource.flush().then(() => { prodLog.debug(`GraphAdjacencyIndex: Flushed verbs-by-source tree`) }), this.lsmTreeVerbsByTarget.flush().then(() => { prodLog.debug(`GraphAdjacencyIndex: Flushed verbs-by-target tree`) }), ]) const elapsed = Date.now() - startTime prodLog.debug(`GraphAdjacencyIndex: Flush completed in ${elapsed}ms`) } /** * Clean shutdown */ async close(): Promise { if (this.flushTimer) { clearInterval(this.flushTimer) this.flushTimer = undefined } // Close both LSM-trees (will flush MemTables to SSTables) if (this.initialized) { await Promise.all([ this.lsmTreeVerbsBySource.close(), this.lsmTreeVerbsByTarget.close(), ]) } prodLog.info('GraphAdjacencyIndex: Shutdown complete') } /** * Check if index is healthy */ isHealthy(): boolean { if (!this.initialized) { return false } return ( !this.isRebuilding && this.lsmTreeVerbsBySource.isHealthy() && this.lsmTreeVerbsByTarget.isHealthy() ) } }