brainy/src/graph/graphAdjacencyIndex.ts
David Snelling a77b064bd7 fix: exception-safe aggregation backfill + generation-verified adoption + loud open-path guards
- Aggregation backfill walks build into a STAGING map and swap in atomically
  on completion. A mid-walk failure drops the staging map — the previous live
  state keeps serving, the aggregate stays flagged pending, and the storage
  error surfaces to the failing query. Previously the walk wiped live state
  before a scan that could throw, never cleared the pending flag on failure,
  and re-ran a full walk on every subsequent query: a silent wipe/walk/throw
  loop at the caller's retry rate.
- Failed walks are latched: retries within a 30s cooldown rethrow the recorded
  error instantly instead of re-walking, so a tight caller-side retry loop
  costs one loud error per query, never a full store walk per query.
- Persisted aggregation state is stamped with the store's committed generation
  at flush; reopen adoption requires stamp equality. Stale state (unclean
  shutdown) or over-counting state (a log truncation on a copied store pulled
  the watermark back) triggers exactly one loud rescan, never a silent adopt.
- The backfill/adoption path narrates: adoption decisions, walk start/finish
  with counts and duration, and failures all log by default.
- getNouns/getVerbs refuse a supplied-but-undecodable pagination cursor with a
  loud error instead of silently restarting the walk at offset 0 (which
  re-served page 1 forever to any while(hasMore) caller).
- The graph cold-load verb walk aborts loudly on a missing or non-advancing
  cursor with hasMore=true.
- A versioned index provider whose generation is AHEAD of the committed
  watermark (torn copy / crash-recovery truncation) is now named loudly at
  open, alongside the existing behind-direction message.
2026-07-17 16:00:11 -07:00

977 lines
36 KiB
TypeScript
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

/**
* @module graph/graphAdjacencyIndex
* @description GraphAdjacencyIndex — billion-scale graph traversal engine.
*
* Adjacency lives in two verb-id LSM trees (sourceId → verbIds,
* targetId → verbIds) filtered through an in-memory live-verb tombstone set,
* with full verb objects loaded on demand through the unified cache. The
* verb set is the single source of truth: neighbor reads derive from live
* verbs, so removals are visible to every read path immediately. ID-only
* in-memory tracking keeps the resident footprint to the verb-id set (~8
* bytes per relationship) regardless of relationship payload size.
*
* **8.0 u64 boundary:** this class implements the BigInt
* {@link GraphIndexProvider} contract — entity ints in, entity/verb ints out.
* Internally everything stays string-keyed (UUID-keyed LSM trees) and u32:
* entity-int params resolve to UUIDs via the shared entity-id mapper
* (`getUuid(Number(big))` — lossless under the `EntityIdSpaceExceeded` u32
* guard), and returns convert back with `BigInt(getOrAssign(uuid))`. Verb ints
* come from a small in-process interning map (see
* {@link GraphAdjacencyIndex.verbIntsToIds}); the interning is derived state,
* never persisted — `rebuild()` / the verb-id-set recovery path re-derive it
* from storage, which the JS index already does on cold start.
*/
import { GraphVerb, StorageAdapter } from '../coreTypes.js'
import { UnifiedCache, getGlobalCache } from '../utils/unifiedCache.js'
import { prodLog } from '../utils/logger.js'
import { LSMTree } from './lsm/LSMTree.js'
import type { GraphIndexProvider } from '../plugin.js'
export interface GraphIndexConfig {
maxIndexSize?: number // Default: 100000
rebuildThreshold?: number // Default: 0.1
autoOptimize?: boolean // Default: true
flushInterval?: number // Default: 30000ms
}
/**
* @description The minimal UUID ↔ int resolver surface the JS graph index
* needs at its BigInt boundary. Satisfied by `EntityIdMapper` (and by whatever
* `MetadataIndexManager.getIdMapper()` / a native metadata index returns) —
* declared structurally here so the graph layer doesn't import the metadata
* layer.
*/
export interface GraphEntityIdResolver {
/** Resolve a UUID to its int, assigning a new one if absent (write path). */
getOrAssign(uuid: string): number
/** Resolve a UUID to its int without assigning (read path). */
getInt(uuid: string): number | undefined
/** Reverse-resolve an int to its UUID (`undefined` = unknown/deleted). */
getUuid(intId: number): string | undefined
}
export interface GraphIndexStats {
totalRelationships: number
sourceNodes: number
targetNodes: number
memoryUsage: number // in bytes
lastRebuild: number
rebuildTime: number // in ms
}
/**
* GraphAdjacencyIndex - Billion-scale adjacency list with LSM-tree storage
*
* Disk-resident verb-id adjacency (LSM trees with bloom filter optimization)
* plus an in-memory live-verb tombstone set; neighbor reads derive from live
* verbs via cache-assisted batch loads — O(node degree) per lookup, correct
* under verb removal.
*/
export class GraphAdjacencyIndex implements GraphIndexProvider {
// LSM-tree storage for verb ID lookups — the single adjacency source of
// truth. Neighbor reads derive from these (live-verb filtered via
// verbIdSet) so removeVerb tombstones are honored by EVERY read path;
// a separate entity→entity edge tree cannot be tombstone-filtered (it
// carries no verb ids) and previously served stale neighbors forever.
private lsmTreeVerbsBySource: LSMTree // sourceId -> verbIds
private lsmTreeVerbsByTarget: LSMTree // targetId -> verbIds
// ID-only membership tracking: a Set<string> of verb ids rather than a
// Map<string, GraphVerb> of full objects, so per-verb resident memory is one id
// string instead of a whole relationship record. (Unmeasured order-of-magnitude
// figures dropped — the durable property is "ids, not objects".)
private verbIdSet = new Set<string>()
// Verb-id interning for the BigInt boundary (8.0 u64 contract).
// Process-lifetime derived state: assigned on addVerb / rebuild /
// verb-id-set recovery, NEVER persisted. removeVerb keeps the entry so a
// verb int stays stable (and resolvable) for the index's lifetime.
private verbIdToInt = new Map<string, number>()
private verbIntToId: string[] = []
// Shared UUID ↔ int resolver for entity ints at the BigInt boundary.
// Threaded in by the coordinator (brainy.ts) from the metadata index's
// idMapper — see setEntityIdMapper().
private entityIdMapper?: GraphEntityIdResolver
// Infrastructure integration
private storage: StorageAdapter
private unifiedCache: UnifiedCache
private config: Required<GraphIndexConfig>
// Performance optimization
private isRebuilding = false
private flushTimer?: NodeJS.Timeout
private rebuildStartTime = 0
private totalRelationshipsIndexed = 0
// Production-scale relationship counting by type
private relationshipCountsByType = new Map<string, number>()
// Initialization flag
private initialized = false
/**
* Check if index is initialized and ready for use
*/
get isInitialized(): boolean {
return this.initialized
}
constructor(
storage: StorageAdapter,
config: GraphIndexConfig = {},
entityIdMapper?: GraphEntityIdResolver
) {
this.storage = storage
this.entityIdMapper = entityIdMapper
this.config = {
maxIndexSize: config.maxIndexSize ?? 100000,
rebuildThreshold: config.rebuildThreshold ?? 0.1,
autoOptimize: config.autoOptimize ?? true,
flushInterval: config.flushInterval ?? 30000
}
// Create LSM-trees for verb ID lookups (billion-scale optimization)
this.lsmTreeVerbsBySource = new LSMTree(storage, {
memTableThreshold: 100000,
storagePrefix: 'graph-lsm-verbs-source',
enableCompaction: true
})
this.lsmTreeVerbsByTarget = new LSMTree(storage, {
memTableThreshold: 100000,
storagePrefix: 'graph-lsm-verbs-target',
enableCompaction: true
})
// Use SAME UnifiedCache as MetadataIndexManager for coordinated memory management
this.unifiedCache = getGlobalCache()
prodLog.info('GraphAdjacencyIndex initialized with LSM-tree storage (2 LSM-trees total)')
}
/**
* @description Thread in the shared UUID ↔ int resolver used for entity-int
* conversion at the BigInt boundary. The coordinator (`brainy.ts`) calls this
* with `metadataIndex.getIdMapper()` right after index construction (init,
* fork, and checkout paths) so reads/writes share one int universe with the
* metadata index. Idempotent; safe to call again after a branch switch.
* @param mapper - The shared entity-id resolver.
* @returns Nothing.
*/
setEntityIdMapper(mapper: GraphEntityIdResolver): void {
this.entityIdMapper = mapper
}
/**
* Resolve the entity-id mapper or fail loudly. The BigInt read methods are
* meaningless without a shared int universe — a missing mapper is a wiring
* bug, not a recoverable condition.
*/
private requireEntityIdMapper(): GraphEntityIdResolver {
if (!this.entityIdMapper) {
throw new Error(
'GraphAdjacencyIndex: entityIdMapper not wired. The coordinator must ' +
'call setEntityIdMapper(metadataIndex.getIdMapper()) (or pass it to ' +
'the constructor) before BigInt-boundary reads.'
)
}
return this.entityIdMapper
}
/**
* Intern a verb-id string, assigning the next sequential u32 on first sight.
* Append-only for the index's lifetime — removeVerb keeps the entry so verb
* ints handed to callers stay resolvable.
*/
private internVerbId(verbId: string): number {
const existing = this.verbIdToInt.get(verbId)
if (existing !== undefined) return existing
const next = this.verbIntToId.length
this.verbIdToInt.set(verbId, next)
this.verbIntToId.push(verbId)
return next
}
/**
* Eager cold-load of the persisted adjacency (readiness contract).
*
* Loads the LSM manifests + SSTables so `size()` reports the durable edge
* count at the rebuild gate — a warm reopen must load the persisted index,
* never re-derive it from a full canonical verb scan (the every-boot O(E)
* cost this method exists to eliminate). Idempotent; the lazy read paths
* call the same `ensureInitialized()` on demand.
*
* NOTE: the JS index deliberately does NOT expose `isReady()`. That signal
* (see `GraphIndexProvider.isReady`) asserts "traversals are trustworthy",
* which this side cannot honestly promise without comparing against the
* canonical store — the query-time known-edge probe
* (`verifyGraphAdjacencyLive` Strategy 2) remains the JS trust check.
*/
async init(): Promise<void> {
await this.ensureInitialized()
}
/**
* Initialize the graph index (lazy initialization)
* Added defensive auto-rebuild check for verbIdSet consistency
*/
private async ensureInitialized(): Promise<void> {
if (this.initialized) {
return
}
await this.lsmTreeVerbsBySource.init()
await this.lsmTreeVerbsByTarget.init()
// Defensive check - if LSM-trees have data but verbIdSet is empty,
// the index was created without proper rebuild (shouldn't happen with singleton
// pattern but protects against edge cases and future refactoring)
const lsmTreeSize = this.lsmTreeVerbsBySource.size()
if (lsmTreeSize > 0 && this.verbIdSet.size === 0) {
prodLog.warn(
`GraphAdjacencyIndex: LSM-trees have ${lsmTreeSize} relationships but verbIdSet is empty. ` +
`Triggering auto-rebuild to restore consistency.`
)
// Note: We don't await rebuild() here to avoid infinite loop
// (rebuild calls ensureInitialized). Instead, we'll populate verbIdSet
// by loading all verb IDs from storage.
await this.populateVerbIdSetFromStorage()
}
// Start auto-flush timer after initialization
this.startAutoFlush()
this.initialized = true
}
/**
* Populate verbIdSet from storage without full rebuild
* Lighter weight than full rebuild - only loads verb IDs, not all verb data
* @private
*/
private async populateVerbIdSetFromStorage(): Promise<void> {
prodLog.info('GraphAdjacencyIndex: Populating verbIdSet from storage...')
const startTime = Date.now()
// Use pagination to load all verb IDs
let hasMore = true
let cursor: string | undefined = undefined
let count = 0
while (hasMore) {
const result = await this.storage.getVerbs({
pagination: { limit: 10000, cursor }
})
for (const verb of result.items) {
this.verbIdSet.add(verb.id)
// Re-derive the verb-int interning (process-lifetime state, never
// persisted — this recovery path is one of the two cold-start sources,
// alongside rebuild()).
this.internVerbId(verb.id)
// Also update counts
const verbType = verb.verb || 'unknown'
this.relationshipCountsByType.set(
verbType,
(this.relationshipCountsByType.get(verbType) || 0) + 1
)
count++
}
hasMore = result.hasMore
if (hasMore && (!result.nextCursor || result.nextCursor === cursor)) {
// A stalled cursor with hasMore=true would re-read the same page
// forever — a silent full-CPU loop at cold open. Abort loudly; a
// graph read failing beats a process that spins without a log line.
throw new Error(
`GraphAdjacencyIndex: verb walk stalled after ${count} verbs — storage returned ` +
`hasMore=true with ${result.nextCursor ? 'a non-advancing' : 'no'} cursor. ` +
`Aborting the cold-load; run brain.repairIndex() if this persists.`
)
}
cursor = result.nextCursor
}
const elapsed = Date.now() - startTime
prodLog.info(`GraphAdjacencyIndex: Populated verbIdSet with ${count} verb IDs in ${elapsed}ms`)
}
/**
* @description Core API — neighbor lookup (BigInt boundary). Neighbors
* derive from the node's live verbs (tombstone-filtered verb-id LSM trees
* + unified-cache batch loads), so `removeVerb()` is honored immediately;
* cost is O(node degree) with cache-assisted verb resolution. Pagination
* support for high-degree nodes. The entity int is resolved to a UUID via
* the shared mapper (unknown int → empty result) and neighbor UUIDs
* convert back via `BigInt(getOrAssign(uuid))`.
*
* @param id - Entity int to get neighbors for (from the shared idMapper).
* @param options - Optional direction ('both' default) + limit/offset.
* @returns Neighbor entity ints (paginated if limit/offset specified).
*
* @example
* // Get all neighbors of an entity int
* const all = await graphIndex.getNeighbors(42n)
*
* @example
* // Get first 50 outgoing neighbors
* const page1 = await graphIndex.getNeighbors(42n, { direction: 'out', limit: 50 })
*/
async getNeighbors(
id: bigint,
options?: {
direction?: 'in' | 'out' | 'both'
limit?: number
offset?: number
}
): Promise<bigint[]> {
await this.ensureInitialized()
const mapper = this.requireEntityIdMapper()
const uuid = mapper.getUuid(Number(id))
if (uuid === undefined) {
// Unknown/deleted entity int — no edges by definition.
return []
}
const neighborUuids = await this.getNeighborUuids(uuid, options)
return neighborUuids.map(neighborUuid => BigInt(mapper.getOrAssign(neighborUuid)))
}
/**
* String-keyed neighbor lookup — the internal implementation behind
* {@link getNeighbors}. Kept UUID-based because the LSM trees are keyed by
* UUID; only the public contract speaks BigInt.
*
* Neighbors are derived from the node's **live** verbs (the verb-id LSM
* trees filtered through the `verbIdSet` tombstones, then batch-loaded via
* the unified cache) rather than from a separate entity→entity edge tree.
* That keeps `removeVerb()` visible to traversal: an entity-level edge
* entry carries no verb id, so it could never be tombstone-filtered, and
* a removed relationship would keep its endpoints "connected" forever.
*/
private async getNeighborUuids(
id: string,
options?: {
direction?: 'in' | 'out' | 'both'
limit?: number
offset?: number
}
): Promise<string[]> {
const startTime = performance.now()
const direction = options?.direction || 'both'
const neighbors = new Set<string>()
if (direction !== 'in') {
for (const verb of (await this.liveVerbsForNode(this.lsmTreeVerbsBySource, id)).values()) {
neighbors.add(verb.targetId)
}
}
if (direction !== 'out') {
for (const verb of (await this.liveVerbsForNode(this.lsmTreeVerbsByTarget, id)).values()) {
neighbors.add(verb.sourceId)
}
}
// Convert to array for pagination
let result = Array.from(neighbors)
// Apply pagination if requested
if (options?.limit !== undefined || options?.offset !== undefined) {
const offset = options?.offset || 0
const limit = options?.limit !== undefined ? options.limit : result.length
result = result.slice(offset, offset + limit)
}
const elapsed = performance.now() - startTime
// Performance assertion - should be sub-5ms with LSM-tree
if (elapsed > 5.0) {
prodLog.warn(`GraphAdjacencyIndex: Slow neighbor lookup for ${id}: ${elapsed.toFixed(2)}ms`)
}
return result
}
/**
* @description The live (non-tombstoned) verb objects adjacent to `nodeId`
* in one verb-id LSM tree: read the node's verb-id list, drop ids deleted
* by `removeVerb()` (the `verbIdSet` tombstone filter — same rule as
* {@link verbIdsToPaginatedInts}), and batch-load the survivors through the
* unified cache. Shared by {@link getNeighborUuids} for both directions.
* @param tree - `lsmTreeVerbsBySource` (out-edges) or `lsmTreeVerbsByTarget` (in-edges).
* @param nodeId - The node's UUID (LSM trees are UUID-keyed).
* @returns The node's live verbs in that direction, keyed by verb id.
*/
private async liveVerbsForNode(tree: LSMTree, nodeId: string): Promise<Map<string, GraphVerb>> {
const verbIds = (await tree.get(nodeId)) || []
const liveIds = [...new Set(verbIds)].filter(verbId => this.verbIdSet.has(verbId))
if (liveIds.length === 0) return new Map()
return this.getVerbsBatchCached(liveIds)
}
/**
* @description Verb ints for all edges originating at `sourceInt` (BigInt
* boundary). O(log n) LSM-tree lookup with bloom filter optimization;
* filters out deleted verb IDs (tombstone deletion workaround); pagination
* support for entities with many relationships. Unknown entity int → empty
* result. Resolve returned ints back to verb-id strings with
* {@link verbIntsToIds}.
*
* @param sourceInt - Source entity int (from the shared idMapper).
* @param options - Optional limit/offset pagination.
* @returns Verb ints originating from this source (excluding deleted).
*
* @example
* const verbInts = await graphIndex.getVerbIdsBySource(42n, { limit: 50 })
* const verbIds = await graphIndex.verbIntsToIds(verbInts)
*/
async getVerbIdsBySource(
sourceInt: bigint,
options?: {
limit?: number
offset?: number
}
): Promise<bigint[]> {
await this.ensureInitialized()
const mapper = this.requireEntityIdMapper()
const sourceId = mapper.getUuid(Number(sourceInt))
if (sourceId === undefined) return []
const startTime = performance.now()
const verbIds = await this.lsmTreeVerbsBySource.get(sourceId)
const elapsed = performance.now() - startTime
// Performance assertion - should be sub-5ms with LSM-tree
if (elapsed > 5.0) {
prodLog.warn(`GraphAdjacencyIndex: Slow getVerbIdsBySource for ${sourceId}: ${elapsed.toFixed(2)}ms`)
}
return this.verbIdsToPaginatedInts(verbIds || [], options)
}
/**
* @description Verb ints for all edges pointing at `targetInt` (BigInt
* boundary). O(log n) LSM-tree lookup with bloom filter optimization;
* filters out deleted verb IDs (tombstone deletion workaround); pagination
* support for popular target entities. Unknown entity int → empty result.
* Resolve returned ints back to verb-id strings with {@link verbIntsToIds}.
*
* @param targetInt - Target entity int (from the shared idMapper).
* @param options - Optional limit/offset pagination.
* @returns Verb ints pointing to this target (excluding deleted).
*
* @example
* const verbInts = await graphIndex.getVerbIdsByTarget(42n, { limit: 50 })
* const verbIds = await graphIndex.verbIntsToIds(verbInts)
*/
async getVerbIdsByTarget(
targetInt: bigint,
options?: {
limit?: number
offset?: number
}
): Promise<bigint[]> {
await this.ensureInitialized()
const mapper = this.requireEntityIdMapper()
const targetId = mapper.getUuid(Number(targetInt))
if (targetId === undefined) return []
const startTime = performance.now()
const verbIds = await this.lsmTreeVerbsByTarget.get(targetId)
const elapsed = performance.now() - startTime
// Performance assertion - should be sub-5ms with LSM-tree
if (elapsed > 5.0) {
prodLog.warn(`GraphAdjacencyIndex: Slow getVerbIdsByTarget for ${targetId}: ${elapsed.toFixed(2)}ms`)
}
return this.verbIdsToPaginatedInts(verbIds || [], options)
}
/**
* Shared tail for the verb-int read methods: drop tombstoned ids (LSM trees
* retain all ids; verbIdSet tracks deletions), apply pagination, and intern
* the survivors to verb ints. Interning on the read path is safe — ids are
* assigned deterministically within the process lifetime and never persisted.
*/
private verbIdsToPaginatedInts(
allIds: string[],
options?: { limit?: number; offset?: number }
): bigint[] {
let result = allIds.filter(id => this.verbIdSet.has(id))
// Apply pagination if requested
if (options?.limit !== undefined || options?.offset !== undefined) {
const offset = options?.offset || 0
const limit = options?.limit !== undefined ? options.limit : result.length
result = result.slice(offset, offset + limit)
}
return result.map(id => BigInt(this.internVerbId(id)))
}
/**
* @description Batch reverse resolver: verb ints → verb-id strings (the
* REQUIRED half of the 8.0 contract Brainy's warm cache feeds from). Reads
* the in-process interning map populated by `addVerb`, `rebuild()`, and the
* verb-id-set recovery path — the JS index derives the interning from
* storage on cold start, so no sidecar persistence exists or is needed.
* @param verbInts - Verb ints as returned by the verb-int read methods.
* @returns One entry per input, order-preserving; `null` for unknown ints.
*/
async verbIntsToIds(verbInts: bigint[]): Promise<(string | null)[]> {
return verbInts.map(verbInt => this.verbIntToId[Number(verbInt)] ?? null)
}
/**
* Get verb from cache or storage - Billion-scale memory optimization
* Uses UnifiedCache with LRU eviction instead of storing all verbs in memory
*
* @param verbId Verb ID to retrieve
* @returns GraphVerb or null if not found
*/
async getVerbCached(verbId: string): Promise<GraphVerb | null> {
const cacheKey = `graph:verb:${verbId}`
// Try to get from cache, load if not present
const verb = await this.unifiedCache.get(cacheKey, async () => {
// Load from storage (fallback if not in cache)
const loadedVerb = await this.storage.getVerb(verbId)
// Cache the loaded verb with metadata
if (loadedVerb) {
this.unifiedCache.set(cacheKey, loadedVerb, 'other', 128, 50) // 128 bytes estimated size, 50ms rebuild cost
}
return loadedVerb
})
return verb
}
/**
* Batch get multiple verbs with caching
*
* **Performance**: Eliminates N+1 pattern for verb loading
* - Current: N × getVerbCached() = N × 50ms on GCS = 250ms for 5 verbs
* - Batched: 1 × getVerbsBatchCached() = 1 × 50ms on GCS = 50ms (**5x faster**)
*
* **Use cases:**
* - relate() duplicate checking (check multiple existing relationships)
* - Loading relationship chains
* - Pre-loading verbs for analysis
*
* **Cache behavior:**
* - Checks UnifiedCache first (fast path)
* - Batch-loads uncached verbs from storage
* - Caches loaded verbs for future access
*
* @param verbIds Array of verb IDs to fetch
* @returns Map of verbId → GraphVerb (only successful reads included)
*
*/
async getVerbsBatchCached(verbIds: string[]): Promise<Map<string, GraphVerb>> {
const results = new Map<string, GraphVerb>()
const uncached: string[] = []
// Phase 1: Check cache for each verb
for (const verbId of verbIds) {
const cacheKey = `graph:verb:${verbId}`
const cached = this.unifiedCache.getSync(cacheKey)
if (cached) {
results.set(verbId, cached)
} else {
uncached.push(verbId)
}
}
// Phase 2: Batch-load uncached verbs from storage
if (uncached.length > 0 && this.storage.getVerbsBatch) {
const loadedVerbs = await this.storage.getVerbsBatch(uncached)
for (const [verbId, verb] of loadedVerbs.entries()) {
const cacheKey = `graph:verb:${verbId}`
// Cache the loaded verb with metadata
// Note: HNSWVerbWithMetadata is structurally assignable to GraphVerb
this.unifiedCache.set(cacheKey, verb, 'other', 128, 50) // 128 bytes estimated size, 50ms rebuild cost
results.set(verbId, verb)
}
}
return results
}
/**
* Get total relationship count - O(1) operation
*/
size(): number {
// Use LSM-tree size for accurate count (one entry per indexed verb)
return this.lsmTreeVerbsBySource.size()
}
/**
* Get relationship count by type - O(1) operation using existing tracking
*/
getRelationshipCountByType(type: string): number {
return this.relationshipCountsByType.get(type) || 0
}
/**
* Get total relationship count - O(1) operation
*/
getTotalRelationshipCount(): number {
return this.verbIdSet.size
}
/**
* Get all relationship types and their counts - O(1) operation
*/
getAllRelationshipCounts(): Map<string, number> {
return new Map(this.relationshipCountsByType)
}
/**
* Get relationship statistics with enhanced counting information
*/
getRelationshipStats(): {
totalRelationships: number
relationshipsByType: Record<string, number>
uniqueSourceNodes: number
uniqueTargetNodes: number
totalNodes: number
} {
const totalRelationships = this.lsmTreeVerbsBySource.size()
const relationshipsByType = Object.fromEntries(this.relationshipCountsByType)
// Note: Exact unique node counts would require full LSM-tree scan
// Using verbIdSet (ID-only tracking) for memory efficiency
const uniqueSourceNodes = this.verbIdSet.size
const uniqueTargetNodes = this.verbIdSet.size
const totalNodes = this.verbIdSet.size
return {
totalRelationships,
relationshipsByType,
uniqueSourceNodes,
uniqueTargetNodes,
totalNodes
}
}
/**
* @description Add a relationship to the index (BigInt boundary). The
* coordinator resolves both endpoint ints via `idMapper.getOrAssign` and
* mirrors them onto `verb.sourceInt`/`verb.targetInt` before calling. The
* JS index keys its LSM trees by the verb's endpoint UUIDs, so the int
* params carry no extra information here — they exist for contract parity
* with native providers whose trees are int-keyed.
* @param verb - The verb to index (endpoint UUIDs are authoritative).
* @param sourceInt - The source entity's interned int (contract parity).
* @param targetInt - The target entity's interned int (contract parity).
* @param generation - The commit generation (contract parity). The JS index
* keeps a single live adjacency view, not a per-generation edge chain, so
* it ignores this: `db.asOf(g)` graph hops on the open-core path see edges
* as-of-now (the one documented graph time-travel limitation — native
* providers thread this into a versioned endpoint store for correctness).
* @returns The interned verb int for `verb.id` (stable for the index lifetime).
*/
async addVerb(
verb: GraphVerb,
sourceInt: bigint,
targetInt: bigint,
generation: bigint
): Promise<bigint> {
void generation // Contract parity — no per-generation chain in the JS index.
await this.ensureInitialized()
return BigInt(await this.indexVerb(verb))
}
/**
* String-keyed indexing core shared by {@link addVerb} and {@link rebuild}.
* Returns the interned verb int.
*/
private async indexVerb(verb: GraphVerb): Promise<number> {
const startTime = performance.now()
// Track verb ID (memory-efficient: IDs only, full objects loaded on-demand via UnifiedCache)
this.verbIdSet.add(verb.id)
const verbInt = this.internVerbId(verb.id)
// Seed the unified cache with the authoritative verb object: neighbor
// reads ({@link liveVerbsForNode}) resolve verbs through the cache with a
// storage fallback, so an indexed verb is immediately traversable — and
// freshly written verbs are the likeliest next reads.
this.unifiedCache.set(`graph:verb:${verb.id}`, verb, 'other', 128, 50)
// Add to the verb-id adjacency LSM-trees (the single adjacency source of
// truth — neighbor and verb-id reads both derive from these).
await this.lsmTreeVerbsBySource.add(verb.sourceId, verb.id)
await this.lsmTreeVerbsByTarget.add(verb.targetId, verb.id)
// Update type-specific counts atomically
const verbType = verb.type || 'unknown'
this.relationshipCountsByType.set(
verbType,
(this.relationshipCountsByType.get(verbType) || 0) + 1
)
const elapsed = performance.now() - startTime
this.totalRelationshipsIndexed++
// Performance assertion
if (elapsed > 10.0) {
prodLog.warn(`GraphAdjacencyIndex: Slow addVerb for ${verb.id}: ${elapsed.toFixed(2)}ms`)
}
return verbInt
}
/**
* @description Remove a relationship from the index by its id string.
* Deletion is tombstone-based: the verb id leaves `verbIdSet`, and every
* read path (verb-id reads via {@link verbIdsToPaginatedInts}, neighbor
* reads via {@link liveVerbsForNode}) filters the append-only LSM trees
* through that set — so the verb disappears from traversal immediately
* while the trees stay immutable. The verb's interned int is intentionally
* retained so previously returned verb ints stay resolvable via
* {@link verbIntsToIds}.
* @param verbId - The verb's UUID string.
* @param generation - The commit generation (contract parity). The JS index
* tombstones immediately rather than chaining the removal per generation,
* so it ignores this (see {@link addVerb} for the time-travel rationale).
* @returns Resolves once the verb no longer appears in reads.
*/
async removeVerb(verbId: string, generation: bigint): Promise<void> {
void generation // Contract parity — no per-generation chain in the JS index.
await this.ensureInitialized()
// Load verb from cache/storage to get type info
const verb = await this.getVerbCached(verbId)
if (!verb) return
const startTime = performance.now()
// Remove from verb ID set
this.verbIdSet.delete(verbId)
// Update type-specific counts atomically
const verbType = verb.type || 'unknown'
const currentCount = this.relationshipCountsByType.get(verbType) || 0
if (currentCount > 1) {
this.relationshipCountsByType.set(verbType, currentCount - 1)
} else {
this.relationshipCountsByType.delete(verbType)
}
const elapsed = performance.now() - startTime
// Performance assertion
if (elapsed > 5.0) {
prodLog.warn(`GraphAdjacencyIndex: Slow removeVerb for ${verbId}: ${elapsed.toFixed(2)}ms`)
}
}
/**
* Rebuild entire index from storage
* Critical for cold starts and data consistency
*/
async rebuild(): Promise<void> {
await this.ensureInitialized()
if (this.isRebuilding) {
prodLog.warn('GraphAdjacencyIndex: Rebuild already in progress')
return
}
this.isRebuilding = true
this.rebuildStartTime = Date.now()
try {
prodLog.info('GraphAdjacencyIndex: Starting rebuild with LSM-tree...')
// Clear current index
this.verbIdSet.clear()
this.totalRelationshipsIndexed = 0
// CRITICAL FIX - Clear relationship counts to prevent accumulation
this.relationshipCountsByType.clear()
// Re-derive verb-int interning from scratch — it's process-lifetime
// derived state (never persisted), so a rebuild starts a fresh
// generation of verb ints alongside the fresh verbIdSet.
this.verbIdToInt.clear()
this.verbIntToId = []
// Note: LSM-trees will be recreated from storage via their own initialization
// Verb data will be loaded on-demand via UnifiedCache
// Brainy 8.0: storage is always local (filesystem or memory — the
// cloud adapters were removed). Load all verbs at once.
const storageType = this.storage?.constructor.name || ''
let totalVerbs = 0
prodLog.info(`GraphAdjacencyIndex: Load all verbs at once (${storageType})`)
const result = await this.storage.getVerbs({
pagination: { limit: 10000000 } // Effectively unlimited for local storage
})
for (const verb of result.items) {
const graphVerb: GraphVerb = {
id: verb.id,
sourceId: verb.sourceId,
targetId: verb.targetId,
vector: verb.vector,
verb: verb.verb,
createdAt: { seconds: Math.floor(verb.createdAt / 1000), nanoseconds: (verb.createdAt % 1000) * 1000000 },
updatedAt: { seconds: Math.floor(verb.updatedAt / 1000), nanoseconds: (verb.updatedAt % 1000) * 1000000 },
createdBy: verb.createdBy || { augmentation: 'unknown', version: '0.0.0' },
service: verb.service,
data: verb.data,
embedding: verb.vector,
confidence: verb.confidence,
weight: verb.weight
}
await this.indexVerb(graphVerb)
totalVerbs++
}
prodLog.info(
`GraphAdjacencyIndex: Loaded ${totalVerbs.toLocaleString()} verbs (${storageType})`
)
const rebuildTime = Date.now() - this.rebuildStartTime
const memoryUsage = this.calculateMemoryUsage()
prodLog.info(`GraphAdjacencyIndex: Rebuild complete in ${rebuildTime}ms`)
prodLog.info(` - Total relationships: ${totalVerbs}`)
prodLog.info(` - Memory usage: ${(memoryUsage / 1024 / 1024).toFixed(1)}MB`)
prodLog.info(` - LSM-tree stats:`, this.lsmTreeVerbsBySource.getStats())
} finally {
this.isRebuilding = false
}
}
/**
* Calculate current memory usage (LSM-tree mostly on disk)
*/
private calculateMemoryUsage(): number {
let bytes = 0
// LSM-tree memory (MemTable + bloom filters + zone maps)
const sourceStats = this.lsmTreeVerbsBySource.getStats()
const targetStats = this.lsmTreeVerbsByTarget.getStats()
bytes += sourceStats.memTableMemory
bytes += targetStats.memTableMemory
// Verb ID set (memory-efficient: IDs only, ~8 bytes per ID pointer)
// Previous verbIndex Map stored full objects (128 bytes each = 128GB @ 1B verbs)
// Now: verbIdSet stores only IDs (~8 bytes each = ~100KB @ 1B verbs) = 1,280,000x reduction
bytes += this.verbIdSet.size * 8
// Note: Bloom filters and zone maps are in LSM-tree MemTable memory
// Full verb objects loaded on-demand via UnifiedCache with LRU eviction
return bytes
}
/**
* Get comprehensive statistics
*/
getStats(): GraphIndexStats {
const sourceStats = this.lsmTreeVerbsBySource.getStats()
const targetStats = this.lsmTreeVerbsByTarget.getStats()
return {
totalRelationships: this.size(),
sourceNodes: sourceStats.sstableCount,
targetNodes: targetStats.sstableCount,
memoryUsage: this.calculateMemoryUsage(),
lastRebuild: this.rebuildStartTime,
rebuildTime: this.isRebuilding ? Date.now() - this.rebuildStartTime : 0
}
}
/**
* Start auto-flush timer
*/
private startAutoFlush(): void {
this.flushTimer = setInterval(async () => {
await this.flush()
}, this.config.flushInterval)
// Background maintenance must never keep the host process alive —
// close()/flush() handle durability; the interval is best-effort.
if (typeof this.flushTimer.unref === 'function') {
this.flushTimer.unref()
}
}
/**
* Flush LSM-tree MemTables to disk
* CRITICAL FIX: Now public so it can be called from brain.flush()
*/
async flush(): Promise<void> {
if (!this.initialized) {
return
}
const startTime = Date.now()
// Flush both LSM-trees in parallel (MemTables → SSTables on disk)
await Promise.all([
this.lsmTreeVerbsBySource.flush().then(() => {
prodLog.debug(`GraphAdjacencyIndex: Flushed verbs-by-source tree`)
}),
this.lsmTreeVerbsByTarget.flush().then(() => {
prodLog.debug(`GraphAdjacencyIndex: Flushed verbs-by-target tree`)
}),
])
const elapsed = Date.now() - startTime
prodLog.debug(`GraphAdjacencyIndex: Flush completed in ${elapsed}ms`)
}
/**
* Clean shutdown
*/
async close(): Promise<void> {
if (this.flushTimer) {
clearInterval(this.flushTimer)
this.flushTimer = undefined
}
// Close both LSM-trees (will flush MemTables to SSTables)
if (this.initialized) {
await Promise.all([
this.lsmTreeVerbsBySource.close(),
this.lsmTreeVerbsByTarget.close(),
])
}
prodLog.info('GraphAdjacencyIndex: Shutdown complete')
}
/**
* Check if index is healthy
*/
isHealthy(): boolean {
if (!this.initialized) {
return false
}
return (
!this.isRebuilding &&
this.lsmTreeVerbsBySource.isHealthy() &&
this.lsmTreeVerbsByTarget.isHealthy()
)
}
}