- find() search-mode: collapsed the two overlapping options to one canonical `searchMode: SearchMode` (the redundant `mode` alias was silently ignored on the primary find() path while honored on the historical path — a footgun) and removed the unwired `explain?` FindParams field (never read by find()). - Documentation accuracy on the public type surface: entity ids documented as UUID v7 (auto) / v5 (natural key), not v4; dropped the stale "Brainy 3.0" banners and "backward compatibility" hedging on the fresh-8.0 Result type; documented the via/type alias; removed the dead GraphConstraints.bidirectional; fixed the no-op `EmbeddedGraphVerb = Omit<GraphVerb,'source'>` to omit the real sourceId; corrected the GraphIndexProvider.isInitialized JSDoc; documented removeMany's per-chunk generation granularity; JSDoc'd the public VFS surface. - Honest perf comments in source: removed unmeasured billion-scale figures from the LSMTree / graphAdjacencyIndex headers (the JS fallback is in-memory after open; native is the scale path). - Robustness: getVerbMetadata propagates read errors symmetrically with getNounMetadata; the find() egress integrity-guard keeps a row when the JS matcher doesn't implement an operator the provider already matched on; hardened the boundary-no-native CI guard to catch side-effect imports + re-exports. - Tests: de-theatricalized a relateMany test to assert the real contract; fixed a stale Model-B header. Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
1117 lines
49 KiB
TypeScript
1117 lines
49 KiB
TypeScript
/**
|
|
* Brainy Plugin System
|
|
*
|
|
* Simple plugin architecture for two use cases:
|
|
* 1. Native acceleration (@soulcraft/cortex)
|
|
* 2. Custom storage adapters (e.g., Redis, DynamoDB, custom backends)
|
|
*
|
|
* Plugins are loaded from an explicit `plugins: [...]` config list or
|
|
* registered manually via `brain.use()` — there is no implicit detection.
|
|
*/
|
|
|
|
import type {
|
|
StorageAdapter,
|
|
Vector,
|
|
VectorDocument,
|
|
GraphVerb,
|
|
} from './coreTypes.js'
|
|
import type { NounType, VerbType } from './types/graphTypes.js'
|
|
import type { MetadataIndexStats } from './utils/metadataIndex.js'
|
|
import type { GraphIndexStats } from './graph/graphAdjacencyIndex.js'
|
|
|
|
// Re-export the provider contracts that already live closer to their
|
|
// implementations so a plugin author (Cortex) can import the *entire*
|
|
// provider surface from one stable entrypoint: `@soulcraft/brainy/plugin`.
|
|
export type { ColumnStoreProvider } from './indexes/columnStore/types.js'
|
|
export type {
|
|
AggregationProvider,
|
|
AggregateDefinition,
|
|
AggregateGroupState,
|
|
AggregateResult,
|
|
AggregateQueryParams,
|
|
GroupByDimension,
|
|
} from './types/brainy.types.js'
|
|
|
|
/**
|
|
* Plugin interface — all brainy plugins must implement this.
|
|
*/
|
|
export interface BrainyPlugin {
|
|
/** Unique plugin name (typically the npm package name) */
|
|
name: string
|
|
|
|
/**
|
|
* Optional semver range of `@soulcraft/brainy` this plugin supports
|
|
* (e.g. `'>=8.0.0 <9.0.0'` or `'^8.0.0'`). When set and the running brainy is
|
|
* OUTSIDE the range, `init()` THROWS rather than silently falling back to the
|
|
* default JS engine. This is the version-coupling guard for the native
|
|
* accelerator (`@soulcraft/cor`): brainy 8.x and cor 3.x are a matched pair,
|
|
* and a mismatch must fail loud, not degrade invisibly. Supported range
|
|
* grammar: space-separated AND of `>=`, `>`, `<=`, `<`, `=` comparators, plus
|
|
* `^X.Y.Z` (same-major floor). Prerelease tags on the running version are
|
|
* tolerated (`8.0.0-rc1` satisfies `>=8.0.0`).
|
|
*/
|
|
brainyRange?: string
|
|
|
|
/**
|
|
* Called by brainy during init() to activate the plugin.
|
|
* Return `true` if activation succeeded. Returning `false` is a graceful
|
|
* decline (brainy logs a loud warning and uses the default engine for this
|
|
* plugin's providers). THROWING aborts init — a registered plugin is always
|
|
* explicitly requested (via `config.plugins` or `brain.use()`; brainy does no
|
|
* auto-detection), so a hard failure is fatal, never a silent skip.
|
|
*/
|
|
activate(context: BrainyPluginContext): Promise<boolean>
|
|
|
|
/**
|
|
* Called when brainy.close() is invoked. Optional cleanup.
|
|
*/
|
|
deactivate?(): Promise<void>
|
|
}
|
|
|
|
/**
|
|
* Context passed to plugins during activation.
|
|
*/
|
|
export interface BrainyPluginContext {
|
|
/**
|
|
* Register a provider for a named subsystem.
|
|
*
|
|
* Well-known provider keys (used by cortex):
|
|
* - 'metadataIndex' — MetadataIndexManager replacement
|
|
* - 'graphIndex' — GraphAdjacencyIndex replacement
|
|
* - 'entityIdMapper' — EntityIdMapper replacement
|
|
* - 'cache' — UnifiedCache replacement
|
|
* - 'vector' — JsHnswVectorIndex replacement (vector index engine)
|
|
* - 'roaring' — RoaringBitmap32 replacement
|
|
* - 'embeddings' — Embedding engine replacement (single text)
|
|
* - 'embedBatch' — Batch embedding engine (texts[] → vectors[])
|
|
* - 'distance' — Distance function overrides
|
|
* - 'msgpack' — Msgpack encode/decode
|
|
* - 'aggregation' — AggregationIndex replacement (incremental aggregates)
|
|
*
|
|
* Storage adapter keys:
|
|
* - 'storage:<name>' — Custom storage adapter factory
|
|
*/
|
|
registerProvider(key: string, implementation: unknown): void
|
|
|
|
/** Brainy version for compatibility checks */
|
|
readonly version: string
|
|
}
|
|
|
|
// ===========================================================================
|
|
// Provider contracts — the EXACT surface Brainy calls on each registered
|
|
// provider.
|
|
//
|
|
// These interfaces are the type-level half of the provider-parity guarantee.
|
|
// They capture only what Brainy actually invokes, so a native accelerator
|
|
// (Cortex) that declares `implements MetadataIndexProvider` gets a compile
|
|
// error the moment a method Brainy depends on is dropped or its signature
|
|
// drifts. Brainy's own baseline classes (`MetadataIndexManager`,
|
|
// `GraphAdjacencyIndex`, `JsHnswVectorIndex`, `EntityIdMapper`, `UnifiedCache`)
|
|
// implement them too, so the contract can never silently diverge from the
|
|
// thing Brainy ships.
|
|
//
|
|
// Keep these in lockstep with the call sites in `brainy.ts` and the index
|
|
// classes. When Brainy starts calling a new member, add it here — every
|
|
// implementation then fails to compile until it provides the member.
|
|
// ===========================================================================
|
|
|
|
/**
|
|
* The `'metadataIndex'` provider — a drop-in for `MetadataIndexManager`.
|
|
* Brainy calls this surface via `this.metadataIndex.*` (see `brainy.ts`) and
|
|
* the transactional add/remove operations.
|
|
*/
|
|
export interface MetadataIndexProvider {
|
|
init(): Promise<void>
|
|
flush(): Promise<void>
|
|
rebuild(): Promise<void>
|
|
|
|
addToIndex(id: string, entityOrMetadata: any, skipFlush?: boolean, deferWrites?: boolean): Promise<void>
|
|
removeFromIndex(id: string, metadata?: any): Promise<void>
|
|
|
|
getIds(field: string, value: any): Promise<string[]>
|
|
getIdsForFilter(filter: any): Promise<string[]>
|
|
/**
|
|
* @description OPTIONAL: resolve a `where` filter to its matching id universe as
|
|
* an {@link OpaqueIdSet} (a serialized roaring `Buffer`) WITHOUT materializing
|
|
* the ids in TypeScript — the producer half of the predicate-pushdown
|
|
* (`CTX-PUSHDOWN-ALLOWEDIDS`) and graph query→expand fusion. Brainy forwards the
|
|
* returned set opaquely to `VectorIndexProvider.search({ allowedIds })` and
|
|
* `GraphAccelerationProvider.traverse(seeds)`; it NEVER inspects it. A native
|
|
* (cor) metadata index returns its roaring filter result directly (zero
|
|
* crossing); the JS index does not implement this — Brainy falls back to
|
|
* {@link MetadataIndexProvider.getIdsForFilter} + a `ReadonlySet<string>` when
|
|
* absent. Present ⟺ the native stack is the active provider, so the matching
|
|
* native vector/graph engines can decode the same envelope.
|
|
* @param filter - The same filter shape accepted by `getIdsForFilter`.
|
|
* @returns The matching id universe as an opaque set.
|
|
*/
|
|
getIdSetForFilter?(filter: any): Promise<OpaqueIdSet>
|
|
getIdsForTextQuery(query: string): Promise<Array<{ id: string; matchCount: number }>>
|
|
getSortedIdsForFilter(filter: any, orderBy: string, order?: 'asc' | 'desc', topK?: number): Promise<string[]>
|
|
getFilterValues(field: string): Promise<string[]>
|
|
getFilterFields(): Promise<string[]>
|
|
getFieldValueForEntity(entityId: string, field: string): Promise<any>
|
|
getFieldsForType(nounType: NounType): Promise<Array<{ field: string; affinity: number; occurrences: number; totalEntities: number }>>
|
|
getFieldStatistics(): Promise<Map<string, unknown>>
|
|
getFieldsWithCardinality(): Promise<Array<{ field: string; cardinality: number; distribution: string }>>
|
|
getOptimalQueryPlan(filters: Record<string, any>): Promise<unknown>
|
|
|
|
/** Report which index path a `where` clause on `field` will hit (drives `brain.explain()`). */
|
|
explainField(field: string): Promise<{ path: 'column-store' | 'sparse-chunked' | 'none'; notes?: string }>
|
|
|
|
getCountForCriteria(field: string, value: any): Promise<number>
|
|
getEntityCountByType(type: string): number
|
|
getEntityCountByTypeEnum(type: NounType): number
|
|
getTotalEntityCount(): number
|
|
getAllEntityCounts(): Map<string, number>
|
|
getTopNounTypes(n: number): NounType[]
|
|
getTopVerbTypes(n: number): VerbType[]
|
|
getAllNounTypeCounts(): Map<NounType, number>
|
|
getAllVerbTypeCounts(): Map<VerbType, number>
|
|
getAllVFSEntityCounts(): Promise<Map<string, number>>
|
|
|
|
detectAndRepairCorruption(): Promise<void>
|
|
validateConsistency(): Promise<{
|
|
healthy: boolean
|
|
avgEntriesPerEntity: number
|
|
entityCount: number
|
|
indexEntryCount: number
|
|
recommendation: string | null
|
|
}>
|
|
|
|
tokenize(text: string): string[]
|
|
extractTextContent(data: any): string
|
|
|
|
getStats(): Promise<MetadataIndexStats>
|
|
|
|
/**
|
|
* The shared UUID ↔ int mapper — the single source of truth for entity-int
|
|
* resolution at the provider boundary. The coordinator (`brainy.ts`) resolves
|
|
* UUID → int exactly once before every graph-index call (`getOrAssign` on
|
|
* writes, `getInt` on reads — `undefined` means "never mapped", i.e. the
|
|
* entity has no relations) and converts provider-returned ints back with
|
|
* `getUuid`. Ints are u32 today (the `EntityIdSpaceExceeded` guard enforces
|
|
* the ceiling on the JS path), so `Number(bigint)` narrowing is lossless.
|
|
*/
|
|
getIdMapper(): {
|
|
getOrAssign(uuid: string): number
|
|
getInt(uuid: string): number | undefined
|
|
getUuid(intId: number): string | undefined
|
|
}
|
|
|
|
/** The column store the coordinator delegates `where`/`orderBy` to. */
|
|
readonly columnStore: import('./indexes/columnStore/types.js').ColumnStoreProvider
|
|
}
|
|
|
|
/**
|
|
* The `'graphIndex'` provider — a drop-in for `GraphAdjacencyIndex`.
|
|
* Brainy calls this surface via `this.graphIndex.*` (some optional-chained).
|
|
*
|
|
* **8.0 u64 contract — BigInt at the boundary.** Reads take entity ints
|
|
* (from the metadata index's idMapper) and return entity/verb ints as
|
|
* `bigint[]`. The coordinator owns ALL UUID ↔ int conversion: it resolves
|
|
* UUIDs to ints once at the `brainy.ts` boundary (`getOrAssign` on writes,
|
|
* `getInt` on reads, returning empty results for unmapped UUIDs without
|
|
* calling the provider) and maps returned ints back (`getUuid` for entities,
|
|
* {@link GraphIndexProvider.verbIntsToIds} for verbs). Implementations may
|
|
* stay u32 internally — `Number(bigint)` narrowing is lossless under the
|
|
* shipped `EntityIdSpaceExceeded` u32 guard — but must speak the bigint
|
|
* contract at this surface.
|
|
*/
|
|
export interface GraphIndexProvider {
|
|
/**
|
|
* `false` until the provider has loaded its persisted state. A provider should
|
|
* self-load on first read; this flag lets it report readiness for diagnostics
|
|
* and tooling. (Brainy's graph reads route through the provider's own methods,
|
|
* which are expected to be self-loading — it is the {@link GraphAccelerationProvider}
|
|
* fast paths that gate on `isInitialized`.)
|
|
*/
|
|
readonly isInitialized: boolean
|
|
|
|
/**
|
|
* @description Entity ints reachable from `id` (1 hop), deduped.
|
|
* @param id - The entity's interned int (from the shared idMapper).
|
|
* @param options - Direction (`'both'` default) and limit/offset pagination.
|
|
* @returns Neighbor entity ints. Empty when the entity has no edges.
|
|
*/
|
|
getNeighbors(
|
|
id: bigint,
|
|
options?: { direction?: 'in' | 'out' | 'both'; limit?: number; offset?: number }
|
|
): Promise<bigint[]>
|
|
/**
|
|
* @description Verb ints for all edges originating at `sourceInt`.
|
|
* @param sourceInt - The source entity's interned int.
|
|
* @param options - Optional limit/offset pagination.
|
|
* @returns Verb ints, resolvable via {@link GraphIndexProvider.verbIntsToIds}.
|
|
*/
|
|
getVerbIdsBySource(sourceInt: bigint, options?: { limit?: number; offset?: number }): Promise<bigint[]>
|
|
/**
|
|
* @description Verb ints for all edges pointing at `targetInt`.
|
|
* @param targetInt - The target entity's interned int.
|
|
* @param options - Optional limit/offset pagination.
|
|
* @returns Verb ints, resolvable via {@link GraphIndexProvider.verbIntsToIds}.
|
|
*/
|
|
getVerbIdsByTarget(targetInt: bigint, options?: { limit?: number; offset?: number }): Promise<bigint[]>
|
|
/**
|
|
* @description Batch reverse resolver: verb ints → verb-id strings. REQUIRED —
|
|
* the provider owns the durable verb-int interning (Brainy keeps only a
|
|
* bounded in-memory warm cache fed by `addVerb` returns and this resolver;
|
|
* pure optimization, no durability role).
|
|
* @param verbInts - Verb ints as returned by the read methods.
|
|
* @returns One entry per input, order-preserving; `null` for unknown ints.
|
|
*/
|
|
verbIntsToIds(verbInts: bigint[]): Promise<(string | null)[]>
|
|
getVerbsBatchCached(verbIds: string[]): Promise<Map<string, GraphVerb>>
|
|
|
|
/**
|
|
* @description Index one verb. The coordinator resolves both endpoint ints
|
|
* via `idMapper.getOrAssign` and mirrors them onto `verb.sourceInt` /
|
|
* `verb.targetInt` before the call.
|
|
* @param verb - The verb to index (endpoint UUIDs + derived `sourceInt`/`targetInt`).
|
|
* @param sourceInt - The source entity's interned int.
|
|
* @param targetInt - The target entity's interned int.
|
|
* @param generation - Brainy's commit generation for this write — the same
|
|
* watermark the storage layer stamps onto the record. A provider with a
|
|
* per-generation edge chain records the edge's existence at this generation
|
|
* so `db.asOf(g)` graph hops resolve historically correct endpoints; the
|
|
* JS baseline has no such chain and ignores it (graph time-travel is a
|
|
* native-provider capability — see the consistency-model doc).
|
|
* @returns The interned verb int for `verb.id` (feeds Brainy's warm cache).
|
|
*/
|
|
addVerb(verb: GraphVerb, sourceInt: bigint, targetInt: bigint, generation: bigint): Promise<bigint>
|
|
/**
|
|
* @description Remove one verb from the index by its id string. The verb's
|
|
* interned int stays reserved (ints are never recycled within a generation).
|
|
* @param verbId - The verb's UUID string.
|
|
* @param generation - Brainy's commit generation for this removal. A provider
|
|
* with a per-generation edge chain tombstones the edge at this generation
|
|
* (so it remains visible to `db.asOf(g)` for `g` before the removal); the
|
|
* JS baseline removes immediately and ignores it.
|
|
* @returns Resolves once the verb no longer appears in reads.
|
|
*/
|
|
removeVerb(verbId: string, generation: bigint): Promise<void>
|
|
|
|
rebuild(): Promise<void>
|
|
flush(): Promise<void>
|
|
close(): Promise<void>
|
|
size(): number
|
|
|
|
getStats(): GraphIndexStats
|
|
getRelationshipStats(): {
|
|
totalRelationships: number
|
|
relationshipsByType: Record<string, number>
|
|
uniqueSourceNodes: number
|
|
uniqueTargetNodes: number
|
|
totalNodes: number
|
|
}
|
|
getRelationshipCountByType(type: string): number
|
|
getTotalRelationshipCount(): number
|
|
getAllRelationshipCounts(): Map<string, number>
|
|
}
|
|
|
|
// ============= Graph Acceleration (optional native engine) =============
|
|
|
|
/**
|
|
* Opaque serialized id-set — a cor-internal roaring payload (a Node `Buffer` at
|
|
* runtime when the native engine is present), passed straight from a `find()`
|
|
* universe into `traverse` / `VectorIndexProvider.search` with NO id
|
|
* materialization in TypeScript (the O(1)-crossing query→expand win). Brainy
|
|
* NEVER inspects it; the provider version-tags the envelope and throws on a
|
|
* format mismatch (it owns integrity + the id-space-width guarantee). Typed as
|
|
* `Uint8Array` (which a Node `Buffer` satisfies) so the universal build stays
|
|
* browser-safe.
|
|
*/
|
|
export type OpaqueIdSet = Uint8Array
|
|
|
|
/** Direction of a graph traversal relative to each frontier node. */
|
|
export type GraphTraversalDirection = 'in' | 'out' | 'both'
|
|
|
|
/**
|
|
* Columnar subgraph — the shared wire format for every native graph read
|
|
* (`traverse`, `edgesForNode`, cursor chunks). Parallel typed arrays, never an
|
|
* array-of-objects, so the NAPI boundary transfers them near-zero-copy and
|
|
* Brainy maps u64↔UUID **lazily** (only for the rows a caller actually renders,
|
|
* via `EntityIdMapperProvider.entityIntsToUuids` + `GraphIndexProvider.verbIntsToIds`).
|
|
* The three `edge*` arrays are parallel (index `i` is one edge); `nodes` /
|
|
* `nodeDepth` are parallel.
|
|
*/
|
|
export interface Subgraph {
|
|
/** Discovered entity ints; resolve via `entityIntsToUuids`. */
|
|
nodes: BigInt64Array
|
|
/** Hop distance of each node from the nearest seed (parallel to `nodes`). Present iff `includeDepth`. */
|
|
nodeDepth?: Uint8Array
|
|
/** Edge source entity ints. */
|
|
edgeSources: BigInt64Array
|
|
/** Edge target entity ints. */
|
|
edgeTargets: BigInt64Array
|
|
/** Edge verb ints; resolve via `verbIntsToIds`. */
|
|
edgeVerbInts: BigInt64Array
|
|
/** Edge verb-type indices (stable TypeIdx; resolve via `TypeUtils.getVerbFromIndex`). */
|
|
edgeTypes: Uint16Array
|
|
/**
|
|
* `true` when a `maxNodes` / `maxEdges` cap truncated the result. The returned
|
|
* rows are the deterministic BFS-order prefix; page the remainder with a
|
|
* `graphCursor` (traverse stays stateless/bounded — no continuation token).
|
|
*/
|
|
truncated?: boolean
|
|
/** When `truncated`, how many nodes/edges were cut (for "+N more" affordances). */
|
|
truncatedNodeCount?: number
|
|
truncatedEdgeCount?: number
|
|
}
|
|
|
|
/** Knobs shared by the bounded traversals. */
|
|
export interface TraverseOptions {
|
|
/** Max hop distance from any seed. */
|
|
depth: number
|
|
/** Edge direction to follow (default `'both'`). */
|
|
direction?: GraphTraversalDirection
|
|
/** Restrict traversed edges to these verb-type indices (TypeIdx). */
|
|
verbTypes?: number[]
|
|
/** Restrict traversed edges to these subtypes. */
|
|
subtypes?: string[]
|
|
/**
|
|
* Visibility tiers to EXCLUDE from the frontier (default: `['internal','system']`,
|
|
* matching `related()`). Pass `[]` for an all-tiers (admin) view.
|
|
*/
|
|
excludeVisibility?: string[]
|
|
/** Cap on returned nodes (deterministic BFS-order prefix; sets `Subgraph.truncated`). */
|
|
maxNodes?: number
|
|
/** Cap on returned edges. */
|
|
maxEdges?: number
|
|
/** Include the edge arrays (`false` = nodes-only reachability). Default `true`. */
|
|
includeEdges?: boolean
|
|
/** Populate `Subgraph.nodeDepth`. Default `false`. */
|
|
includeDepth?: boolean
|
|
}
|
|
|
|
/** Options for the both-direction single-node edge read. */
|
|
export interface EdgesForNodeOptions {
|
|
/** Edge direction (default `'both'`). */
|
|
direction?: GraphTraversalDirection
|
|
verbTypes?: number[]
|
|
subtypes?: string[]
|
|
excludeVisibility?: string[]
|
|
limit?: number
|
|
}
|
|
|
|
/** Opaque server-side cursor handle (TTL-bounded; pinned to a generation at open). */
|
|
export type GraphCursorHandle = string
|
|
|
|
/** Options for opening a streaming whole-graph (or seeded) cursor. */
|
|
export interface GraphCursorOptions {
|
|
direction?: GraphTraversalDirection
|
|
excludeVisibility?: string[]
|
|
/** `'light'` = ids/types only, no metadata-join columns (viz default); `'full'` = with joins. */
|
|
projection?: 'light' | 'full'
|
|
/** Seed set to bound the walk; omitted = the whole graph. */
|
|
seeds?: bigint[] | OpaqueIdSet
|
|
/**
|
|
* Resume token from a prior chunk — used to continue after a `SnapshotExpired`
|
|
* (the pinned generation's TTL lapsed): reopen with this to resume the walk.
|
|
*/
|
|
cursor?: string
|
|
}
|
|
|
|
/** One streamed chunk of a graph cursor walk. */
|
|
export interface GraphCursorChunk {
|
|
/** The chunk's nodes + edges, columnar. */
|
|
subgraph: Subgraph
|
|
/** Opaque resume token (pass to `graphCursorOpen({ cursor })` after `SnapshotExpired`). */
|
|
cursor?: string
|
|
/** `true` once the walk is exhausted; `graphCursorNext` should not be called again. */
|
|
done: boolean
|
|
}
|
|
|
|
/** Per-node scores returned by `rank` / `mostConnected`; `scores[i]` belongs to `nodeInts[i]` (descending). */
|
|
export interface GraphScores {
|
|
nodeInts: BigInt64Array
|
|
scores: Float64Array
|
|
}
|
|
|
|
/** Community/cluster labelling; `communityIds[i]` is the group of `nodeInts[i]`. */
|
|
export interface GraphCommunities {
|
|
nodeInts: BigInt64Array
|
|
communityIds: Uint32Array
|
|
communityCount: number
|
|
}
|
|
|
|
/** A path between two nodes — the node sequence + the edges between them (or `null` if unreachable). */
|
|
export interface GraphPath {
|
|
nodeInts: BigInt64Array
|
|
edgeVerbInts: BigInt64Array
|
|
/** Cost of the path: number of hops, or summed edge weight when `by: 'weight'`. */
|
|
cost: number
|
|
}
|
|
|
|
/**
|
|
* Options for `rank` (importance / influence ranking). INTENT-level only — the
|
|
* ranking ALGORITHM is the provider's choice (the JS fallback uses PageRank; a
|
|
* native provider may use personalized PageRank, eigenvector centrality, etc.),
|
|
* so no algorithm-tuning knobs are exposed here.
|
|
*/
|
|
export interface RankOptions {
|
|
/** Return only the top-K nodes by score (default: all). */
|
|
topK?: number
|
|
/** Visibility tiers to EXCLUDE from the ranked set (default `['internal','system']`). */
|
|
excludeVisibility?: string[]
|
|
}
|
|
|
|
/** Options for `communities` (grouping related nodes). The grouping algorithm is the provider's choice. */
|
|
export interface CommunitiesOptions {
|
|
/** Treat the graph as directed when grouping (default `false` — direction-agnostic). */
|
|
directed?: boolean
|
|
excludeVisibility?: string[]
|
|
}
|
|
|
|
/** Options for `path` (best route between two nodes). */
|
|
export interface PathOptions {
|
|
direction?: GraphTraversalDirection
|
|
/** Optimize for fewest `'hops'` (default) or least summed edge `'weight'`. */
|
|
by?: 'hops' | 'weight'
|
|
/** Restrict the route to these verb-type indices (TypeIdx). */
|
|
verbTypes?: number[]
|
|
excludeVisibility?: string[]
|
|
/** Abandon the search past this many hops. */
|
|
maxDepth?: number
|
|
}
|
|
|
|
/** Options for `sample` (a representative neighborhood sample for dense-graph viz). */
|
|
export interface SampleOptions {
|
|
/** Max hop distance from any seed. */
|
|
depth: number
|
|
direction?: GraphTraversalDirection
|
|
/** Max neighbors kept per node (random, seeded for reproducibility). */
|
|
fanout: number
|
|
/** RNG seed so a given (seeds, opts, seed) yields a stable sample. */
|
|
seed?: number
|
|
excludeVisibility?: string[]
|
|
maxNodes?: number
|
|
}
|
|
|
|
/** Options for `mostConnected` (the most-connected nodes). */
|
|
export interface MostConnectedOptions {
|
|
/** Return the top-K most-connected nodes. */
|
|
topK: number
|
|
direction?: GraphTraversalDirection
|
|
excludeVisibility?: string[]
|
|
}
|
|
|
|
/**
|
|
* The `'graphAcceleration'` provider — an OPTIONAL native graph engine (the
|
|
* cor 3.0 acceleration layer). Brainy feature-detects it on the registered
|
|
* providers: when present, the public `brain.graph.*` surface and the
|
|
* `related({ node })` / `neighbors({ depth })` / `find({ connected })` paths
|
|
* route here; when absent, Brainy serves the same operations from its pure-TS
|
|
* adjacency fallback (correct, small-graph-scale). It is SEPARATE from the
|
|
* required {@link GraphIndexProvider} (which stays the bigint-adjacency
|
|
* contract) so "optional native acceleration" is explicit at the seam.
|
|
*
|
|
* **Generation-aware (8.0 time-travel).** Every read takes an optional trailing
|
|
* `generation?: bigint`; omitted = the current generation ("now"). With a
|
|
* generation, the provider resolves the graph **as of** that generation
|
|
* (`db.asOf(g).graph.*`), reading its generation-filtered adjacency rather than
|
|
* the now-only fast path.
|
|
*
|
|
* **Columnar + opaque-id-set.** Reads return the columnar {@link Subgraph};
|
|
* `traverse` / sample accept seeds as `bigint[]` OR an {@link OpaqueIdSet}
|
|
* (a `find()` universe forwarded with no id materialization — the query→expand
|
|
* fusion). Visibility tiers are excluded at the index by default.
|
|
*/
|
|
export interface GraphAccelerationProvider {
|
|
/** `false` until the native engine has loaded; Brainy probes before routing. */
|
|
readonly isInitialized: boolean
|
|
|
|
/**
|
|
* @description Bounded multi-hop expansion from `seeds`, returning the reachable
|
|
* subgraph. One call replaces Brainy's per-hop BFS round-trips.
|
|
* @param seeds - Start nodes as entity ints, OR an {@link OpaqueIdSet} (a
|
|
* `find()` universe — the frontier is intersected in id-space, zero crossing).
|
|
* @param options - Depth, direction, edge/visibility filters, and caps.
|
|
* @param generation - Optional as-of generation (omitted = now).
|
|
* @returns The reachable {@link Subgraph} (BFS-order; `truncated` set if capped).
|
|
*/
|
|
traverse(seeds: bigint[] | OpaqueIdSet, options: TraverseOptions, generation?: bigint): Promise<Subgraph>
|
|
|
|
/**
|
|
* @description All edges incident to one node, both directions, as structure +
|
|
* cor-indexed fields (Brainy hydrates full edge metadata lazily — the provider
|
|
* cannot recover verb-id strings from its interned form).
|
|
* @param nodeInt - The node's interned entity int.
|
|
* @param options - Direction, edge/visibility filters, limit.
|
|
* @param generation - Optional as-of generation.
|
|
* @returns A {@link Subgraph} of the node's incident edges.
|
|
*/
|
|
edgesForNode(nodeInt: bigint, options: EdgesForNodeOptions, generation?: bigint): Promise<Subgraph>
|
|
|
|
/**
|
|
* @description Open a snapshot-consistent streaming walk of the whole graph (or a
|
|
* seeded region) for O(N) viz loads. Pins a generation at open; the walk reads
|
|
* as-of that pin (no dup/skip under concurrent writes). Handles are TTL-bounded —
|
|
* `graphCursorNext` after expiry rejects with `SnapshotExpired`; reopen with the
|
|
* last chunk's `cursor` to resume. Call `graphCursorClose` on normal completion.
|
|
* @param options - Direction, visibility, projection mode, optional seeds / resume cursor.
|
|
* @param generation - Optional explicit as-of generation to pin (else pins current).
|
|
* @returns A handle for `graphCursorNext` / `graphCursorClose`.
|
|
*/
|
|
graphCursorOpen(options: GraphCursorOptions, generation?: bigint): Promise<GraphCursorHandle>
|
|
/**
|
|
* @description Pull the next chunk from an open cursor.
|
|
* @param handle - From `graphCursorOpen`.
|
|
* @param chunkSize - Target number of nodes/edges in this chunk.
|
|
* @returns The next {@link GraphCursorChunk}; `done: true` when exhausted.
|
|
*/
|
|
graphCursorNext(handle: GraphCursorHandle, chunkSize: number): Promise<GraphCursorChunk>
|
|
/**
|
|
* @description Release an open cursor's pinned generation and server-side state.
|
|
* @param handle - From `graphCursorOpen`.
|
|
*/
|
|
graphCursorClose(handle: GraphCursorHandle): Promise<void>
|
|
|
|
/**
|
|
* @description Rank nodes by influence/importance over the VISIBLE graph (hidden
|
|
* tiers excluded by default, so the ranking reflects the public view). The
|
|
* ranking ALGORITHM is the provider's choice — the JS fallback uses PageRank; a
|
|
* native provider may use personalized PageRank, eigenvector centrality, etc. The
|
|
* intent ("which nodes matter most") is the contract, not the algorithm.
|
|
* @param options - Top-K + visibility (intent-level; no algorithm tuning).
|
|
* @param generation - Optional as-of generation.
|
|
* @returns Per-node scores, descending.
|
|
*/
|
|
rank(options: RankOptions, generation?: bigint): Promise<GraphScores>
|
|
/**
|
|
* @description Group related nodes into communities over the VISIBLE graph. The
|
|
* grouping ALGORITHM is the provider's choice — the JS fallback uses connected
|
|
* components; a native provider may use community detection (e.g. Louvain/Leiden).
|
|
* @param options - Directed-grouping toggle + visibility.
|
|
* @param generation - Optional as-of generation.
|
|
* @returns Per-node community labels + community count.
|
|
*/
|
|
communities(options: CommunitiesOptions, generation?: bigint): Promise<GraphCommunities>
|
|
/**
|
|
* @description Find the best path between two nodes — fewest hops by default, or
|
|
* least summed edge weight with `by: 'weight'`. The pathfinding ALGORITHM is the
|
|
* provider's choice (JS fallback: BFS / Dijkstra).
|
|
* @param fromInt - Start node int.
|
|
* @param toInt - End node int.
|
|
* @param options - Direction, cost basis (`by`), verb-type filter, max depth, visibility.
|
|
* @param generation - Optional as-of generation.
|
|
* @returns The path, or `null` if `toInt` is unreachable from `fromInt`.
|
|
*/
|
|
path(fromInt: bigint, toInt: bigint, options: PathOptions, generation?: bigint): Promise<GraphPath | null>
|
|
/**
|
|
* @description A bounded, representative SAMPLE of the neighborhood around the seeds
|
|
* (capped fan-out per node) — for dense-graph viz. Seeded for reproducibility.
|
|
* @param seeds - Start nodes (ints or an {@link OpaqueIdSet}).
|
|
* @param options - Depth, fan-out, RNG seed, visibility, node cap.
|
|
* @param generation - Optional as-of generation.
|
|
* @returns The sampled {@link Subgraph}.
|
|
*/
|
|
sample(seeds: bigint[] | OpaqueIdSet, options: SampleOptions, generation?: bigint): Promise<Subgraph>
|
|
/**
|
|
* @description The top-K most-connected nodes over the VISIBLE graph. The
|
|
* connectivity metric is the provider's choice (JS fallback: edge-count degree).
|
|
* @param options - Top-K, direction, visibility.
|
|
* @param generation - Optional as-of generation.
|
|
* @returns Per-node connectivity scores, descending.
|
|
*/
|
|
mostConnected(options: MostConnectedOptions, generation?: bigint): Promise<GraphScores>
|
|
}
|
|
|
|
/**
|
|
* Optional capability interface for index providers that maintain versioned
|
|
* (generation-aware) internal state — the provider-side half of Brainy 8.0's
|
|
* generational MVCC contract. Implemented by native providers whose storage
|
|
* engines keep immutable versions (e.g. LSM snapshots); Brainy's own JS
|
|
* indexes do not implement it (they are rebuilt from the storage records,
|
|
* which carry the versioning).
|
|
*
|
|
* **Detection.** Brainy feature-detects this interface on every registered
|
|
* index provider (`'vector'`, `'metadataIndex'`, `'graphIndex'`): when a
|
|
* provider exposes `pin`/`release` functions, Brainy calls them in lockstep
|
|
* with `Db` lifecycle — `pin(g)` when a `Db` value pins generation `g`
|
|
* (`brain.now()`, `brain.transact()`, `brain.asOf()`), `release(g)` when that
|
|
* `Db` is released (explicitly or via the GC backstop). Pins are refcounted
|
|
* on the Brainy side; a provider may receive multiple `pin(g)` calls for the
|
|
* same generation and will receive exactly one matching `release(g)` per pin.
|
|
*
|
|
* **Consistency model (locked cross-team design).** Index providers are
|
|
* *post-commit appliers*: the storage-record commit (atomic manifest rename)
|
|
* is the source of truth, and provider index state is derived, applied after
|
|
* the commit point. On open, a provider compares its own persisted
|
|
* `generation()` against the store's committed generation and replays the
|
|
* gap from the storage records (or requests a rebuild) — there are no
|
|
* provider rollback hooks, because an uncommitted transaction is repaired at
|
|
* the storage layer before any index is opened. The explicit pin/release
|
|
* lifetime OVERRIDES any time-based snapshot retention the provider has
|
|
* (e.g. an LSM snapshot TTL): a pinned generation must stay readable until
|
|
* released, regardless of age.
|
|
*
|
|
* **Speculative reads.** `db.with(txData)` overlays are Brainy-side only —
|
|
* providers are never asked to read uncommitted/speculative state; they
|
|
* always serve committed generations.
|
|
*/
|
|
export interface VersionedIndexProvider {
|
|
/**
|
|
* @description The newest generation this provider's persisted index state
|
|
* reflects. Brainy compares it against the storage layer's committed
|
|
* generation on open to detect a replay gap (index behind storage after a
|
|
* crash between commit and index apply).
|
|
* @returns The provider's current generation as a `bigint` (u64 at the
|
|
* native boundary; Brainy's generation counter is a safe integer today).
|
|
*/
|
|
generation(): bigint
|
|
|
|
/**
|
|
* @description True ⇒ the provider can serve consistent reads at
|
|
* `generation` (segments retained). Combined with pin: pinning a VISIBLE
|
|
* generation guarantees it stays servable until release. Pinning an
|
|
* invisible generation is permitted (refcount-only) — Brainy serves that
|
|
* generation from canonical storage instead.
|
|
*
|
|
* Brainy consults this at pin time (the read-routing rule: a `Db` at
|
|
* generation `g` uses provider-accelerated reads when
|
|
* `isGenerationVisible(g)` was true at pin time; otherwise canonical
|
|
* generation records).
|
|
* @param generation - The generation a `Db` value is about to pin.
|
|
*/
|
|
isGenerationVisible(generation: bigint): boolean
|
|
|
|
/**
|
|
* @description Pin a generation: the provider must keep index state for
|
|
* `generation` readable until the matching {@link VersionedIndexProvider.release}
|
|
* call, overriding any time-based snapshot retention. Called once per
|
|
* Brainy-side pin (refcounted upstream — expect balanced pin/release pairs).
|
|
* @param generation - The generation a live `Db` value just pinned.
|
|
*/
|
|
pin(generation: bigint): void
|
|
|
|
/**
|
|
* @description Release one pin on `generation`. After the last release the
|
|
* provider may reclaim resources for that generation at its discretion.
|
|
* @param generation - The generation being released (matches a prior
|
|
* {@link VersionedIndexProvider.pin} call).
|
|
*/
|
|
release(generation: bigint): void
|
|
}
|
|
|
|
/**
|
|
* @description Feature-detection guard for {@link VersionedIndexProvider}.
|
|
* Brainy applies it to every registered index provider (vector, metadata,
|
|
* graph) when a `Db` value pins or releases a generation: providers exposing
|
|
* all four capability methods receive lockstep `pin`/`release` calls;
|
|
* everything else (including Brainy's own JS indexes) is skipped.
|
|
*
|
|
* @param candidate - A registered index provider instance.
|
|
* @returns Whether the candidate implements the versioned capability.
|
|
*/
|
|
export function isVersionedIndexProvider(
|
|
candidate: unknown
|
|
): candidate is VersionedIndexProvider {
|
|
if (candidate === null || typeof candidate !== 'object') {
|
|
return false
|
|
}
|
|
const c = candidate as Record<string, unknown>
|
|
return (
|
|
typeof c.generation === 'function' &&
|
|
typeof c.isGenerationVisible === 'function' &&
|
|
typeof c.pin === 'function' &&
|
|
typeof c.release === 'function'
|
|
)
|
|
}
|
|
|
|
/**
|
|
* @description Feature-detect an optional {@link GraphAccelerationProvider} (the
|
|
* native graph engine). Brainy probes the registered `'graphAcceleration'`
|
|
* provider with this: present ⇒ `brain.graph.*` routes to native `traverse` /
|
|
* cursor / analytics; absent ⇒ Brainy serves the same operations from its
|
|
* pure-TS adjacency fallback. Checks the two anchor methods (`traverse` +
|
|
* `graphCursorOpen`) — a partial implementation that lacks them is treated as
|
|
* absent (fall back rather than crash).
|
|
* @param candidate - The value registered under `'graphAcceleration'`, or undefined.
|
|
* @returns Whether `candidate` is a usable {@link GraphAccelerationProvider}.
|
|
*/
|
|
export function isGraphAccelerationProvider(
|
|
candidate: unknown
|
|
): candidate is GraphAccelerationProvider {
|
|
if (candidate === null || typeof candidate !== 'object') {
|
|
return false
|
|
}
|
|
const c = candidate as Record<string, unknown>
|
|
return typeof c.traverse === 'function' && typeof c.graphCursorOpen === 'function'
|
|
}
|
|
|
|
/**
|
|
* Columnar at-`generation` candidate vectors for the 8.0 #35 FILTERED exact-rerank.
|
|
* Brainy resolves the per-generation vector before-images for the at-gen filtered
|
|
* universe (from its canonical generation records) and ships them flat, so a native
|
|
* provider reranks against the historically-correct vectors with ZERO per-vector FFI
|
|
* crossing — without the provider duplicating the per-gen retention Brainy already does.
|
|
*
|
|
* - **Row-major:** row `i` is `vectors[i*dim .. (i+1)*dim]` and belongs to `ids[i]`.
|
|
* INVARIANT: `vectors.length === ids.length * dim` (the provider asserts + refuses on mismatch).
|
|
* - **`ids` IS the candidate set:** entity ints (interned via the shared mapper) for the
|
|
* at-gen metadata∩graph filtered universe. The provider reranks EXACTLY these
|
|
* `(id, vector)` pairs and returns top-k by distance; any `allowedIds` passed alongside
|
|
* is redundant on this path (the provider may AND it as belt-and-suspenders).
|
|
* - **`dim`** must equal the provider's configured index dimension (asserted; mismatch → refuse).
|
|
*
|
|
* Supplied only when there is a BOUNDED filtered universe; the unfiltered-deep at-gen
|
|
* case stays the provider's own retained-segment ANN ({@link VersionedIndexProvider.isGenerationVisible}
|
|
* reports false there, so Brainy falls back to its materialization overlay).
|
|
*/
|
|
export interface AtGenerationVectors {
|
|
/** Entity ints (shared-mapper interned), one per candidate; the candidate set. */
|
|
ids: BigInt64Array
|
|
/** Flat row-major vectors: `vectors[i*dim .. (i+1)*dim]` is `ids[i]`'s at-gen vector. */
|
|
vectors: Float32Array
|
|
/** Vector dimension; MUST equal the provider's configured index dim. */
|
|
dim: number
|
|
}
|
|
|
|
/**
|
|
* The object returned by the `'vector'` provider factory — Brainy's vector
|
|
* index contract. Implementations include Brainy's own JS HNSW index and any
|
|
* native acceleration provider (e.g. cortex's Adaptive DiskANN).
|
|
*
|
|
* Brainy calls this surface via `this.index.*` plus the transactional add/remove
|
|
* operations. `enableCOW`, `getItem`, and `setPersistMode` are intentionally
|
|
* absent: Brainy guards each with feature-detection (`typeof x === 'function'`),
|
|
* so they are optional and not part of the required contract (Brainy's own JS
|
|
* HNSW index omits `setPersistMode`, for instance).
|
|
*
|
|
* **Provider key:** registered under `'vector'` — the only key Brainy
|
|
* consults for the vector index. The pre-8.0 `'hnsw'` and `'diskann'` keys
|
|
* are retired and never looked up.
|
|
*/
|
|
export interface VectorIndexProvider {
|
|
addItem(item: VectorDocument): Promise<string>
|
|
removeItem(id: string): Promise<boolean>
|
|
search(
|
|
queryVector: Vector,
|
|
k?: number,
|
|
filter?: (id: string) => Promise<boolean>,
|
|
options?: {
|
|
rerank?: { multiplier: number }
|
|
candidateIds?: string[]
|
|
/**
|
|
* Predicate-pushdown universe — restrict the result to this id-set, applied
|
|
* INSIDE the beam walk (walk traverses all nodes, collects only allowed), which
|
|
* recovers the filtered recall that post-filtering loses. A native provider with
|
|
* cor as the metadata index gets the `find()` universe as an {@link OpaqueIdSet}
|
|
* (zero id materialization); the pure-JS path gets a `ReadonlySet<string>`. Absent
|
|
* = no restriction. Optional — a provider that doesn't pushdown ignores it and
|
|
* falls back to `filter`.
|
|
*/
|
|
allowedIds?: OpaqueIdSet | ReadonlySet<string>
|
|
/**
|
|
* As-of generation for historical (time-travel) reads — the vector-side
|
|
* mirror of the {@link GraphAccelerationProvider} trailing `generation?`.
|
|
* Omitted = "now" (the live index). When set, a {@link VersionedIndexProvider}
|
|
* serves the kNN / exact-rerank over its retained at-`generation` segments
|
|
* instead of the current vectors, so `db.asOf(g)` semantic queries need no
|
|
* O(n@G) JS-HNSW rebuild. Brainy only passes it when the provider advertised
|
|
* `isGenerationVisible(generation)` at pin time; a provider that does not
|
|
* honor it MUST refuse/fall back (never score current vectors as if at-gen) —
|
|
* Brainy then serves the historical leg from its own materialization. Brainy's
|
|
* `generation` is the same u64 counter handed to the graph index on writes.
|
|
*/
|
|
generation?: bigint
|
|
/**
|
|
* 8.0 #35 part-3: the at-`generation` candidate vectors for the FILTERED
|
|
* exact-rerank — Brainy supplies the historically-correct vectors so the
|
|
* provider need not retain them. Present only alongside `generation` on the
|
|
* filtered at-gen path; the provider reranks `atGenerationVectors.ids` by
|
|
* distance and returns top-k. See {@link AtGenerationVectors}. The built-in
|
|
* JS index ignores it (it serves "now" only).
|
|
*/
|
|
atGenerationVectors?: AtGenerationVectors
|
|
}
|
|
): Promise<Array<[string, number]>>
|
|
size(): number
|
|
clear(): void
|
|
rebuild(options?: any): Promise<void>
|
|
flush(): Promise<number>
|
|
getPersistMode(): 'immediate' | 'deferred'
|
|
}
|
|
|
|
|
|
/**
|
|
* The `'entityIdMapper'` provider — a drop-in for `EntityIdMapper`. Injected
|
|
* into the TypeScript `MetadataIndexManager` when a native metadata index is
|
|
* not also registered; that coordinator calls this full surface (incl.
|
|
* `getAllIntIds`, the all-ids universe for negation / `exists:false` filters).
|
|
*/
|
|
export interface EntityIdMapperProvider {
|
|
init(): Promise<void>
|
|
/**
|
|
* @description OPTIONAL: reload the uuid↔int mapping from the CURRENT storage,
|
|
* discarding in-memory state — called by `brain.restore()` AFTER the storage
|
|
* has been replaced from a snapshot and BEFORE the graph index rebuilds, so
|
|
* the graph resolves each verb endpoint through a mapper that reflects the
|
|
* snapshot's assignments (without it, native adjacency edges — keyed on these
|
|
* ints — resolve to stale/missing ints and are silently dropped). A native
|
|
* mapper reloads from its restored binary KV; the JS fallback re-reads its
|
|
* persisted metadata. Optional so a mapper that already reloads via `init()`
|
|
* stays compatible — `restore()` falls back to `init()` when this is absent.
|
|
*/
|
|
rebuild?(): Promise<void>
|
|
getOrAssign(uuid: string): number
|
|
getUuid(intId: number): string | undefined
|
|
getInt(uuid: string): number | undefined
|
|
remove(uuid: string): boolean
|
|
flush(): Promise<void>
|
|
clear(): Promise<void>
|
|
getAllIntIds(): number[]
|
|
intsIterableToUuids(ints: Iterable<number>): string[]
|
|
/**
|
|
* @description Batch reverse-resolve u64 entity ints → UUID strings — the
|
|
* bigint counterpart of {@link EntityIdMapperProvider.intsIterableToUuids},
|
|
* for the native graph engine ({@link GraphAccelerationProvider}, whose
|
|
* `Subgraph.nodes` is a `BigInt64Array`). Brainy resolves lazily — only the
|
|
* rows a caller renders — but viz/export render many at once, so one batch
|
|
* crossing beats N. Order-preserving (one entry per input). A native mapper
|
|
* resolves every int from its binary int↔uuid store; the JS fallback resolves
|
|
* assigned ints and yields `''` for a never-assigned int (which does not occur
|
|
* for graph-engine results, since those only reference assigned ints).
|
|
* @param nodeInts - Entity ints as returned in a {@link Subgraph}.
|
|
* @returns One UUID per input, in order.
|
|
*/
|
|
entityIntsToUuids(nodeInts: BigInt64Array): string[]
|
|
readonly size: number
|
|
}
|
|
|
|
/**
|
|
* The `'cache'` provider — a drop-in for `UnifiedCache`. Brainy installs it as
|
|
* the global cache (`setGlobalCache`) and calls this surface via
|
|
* `getGlobalCache()`.
|
|
*/
|
|
export interface CacheProvider {
|
|
getSync(key: string): any | undefined
|
|
set(key: string, data: any, type: 'vectors' | 'metadata' | 'embedding' | 'other', size: number, rebuildCost?: number): void
|
|
delete(key: string): boolean
|
|
deleteByPrefix(prefix: string): number
|
|
clear(type?: 'vectors' | 'metadata' | 'embedding' | 'other'): void
|
|
}
|
|
|
|
// The `'embeddings'` / `'embedBatch'` providers are function-shaped and already
|
|
// typed by the existing `EmbeddingFunction` (see `coreTypes.ts`), which Brainy
|
|
// uses at the `getProvider('embeddings')` call site. No separate interface is
|
|
// added here to avoid a duplicate, unwired contract.
|
|
|
|
/**
|
|
* The `'graph:compression'` provider — pure-function encode/decode for HNSW
|
|
* connection lists as compact delta-varint byte sequences (cortex's
|
|
* `encodeConnections` / `decodeConnections`).
|
|
*
|
|
* Brainy's `JsHnswVectorIndex` consumes this via a `ConnectionsCodec` that translates
|
|
* UUIDs to stable int slots via the `EntityIdMapper`, encodes, and persists
|
|
* the compressed bytes through the binary-blob primitive. On load, the blob
|
|
* is fetched + decoded back into UUID sets — `setConnectionsCodec()` on
|
|
* `JsHnswVectorIndex` is the injection point. Read path is dual-format: when no blob
|
|
* exists for a node, the connections fall back to the legacy JSON-array path
|
|
* embedded in `saveVectorIndexData`, so pre-2.4.0 indexes keep loading unchanged
|
|
* and convergence to the compressed form happens lazily on next save.
|
|
*
|
|
* Activated only when the storage adapter exposes the binary-blob primitive
|
|
* AND the metadata index resolves a stable idMapper. Cloud adapters that
|
|
* lack a real local-path resolution still benefit, since the blob primitive
|
|
* itself works across every adapter as of brainy 7.25.0.
|
|
*/
|
|
export interface GraphCompressionProvider {
|
|
/** Encode a list of u32 ints to compact delta-varint bytes. Sorts internally. */
|
|
encode(ids: number[]): Buffer
|
|
/** Decode delta-varint bytes back to a u32 list. */
|
|
decode(data: Buffer): number[]
|
|
}
|
|
|
|
/**
|
|
* Storage adapter factory — plugins register these to provide
|
|
* new storage backends that users reference by name.
|
|
*
|
|
* Example: A Redis plugin registers 'storage:redis', then users
|
|
* can use `new Brainy({ storage: 'redis', redis: { host: '...' } })`
|
|
*/
|
|
export interface StorageAdapterFactory {
|
|
create(config: Record<string, unknown>): StorageAdapter | Promise<StorageAdapter>
|
|
name: string
|
|
}
|
|
|
|
/**
|
|
* Plugin registry — manages plugin lifecycle and provider resolution.
|
|
*/
|
|
/** Parse `X.Y.Z[-prerelease][+build]` → `[major, minor, patch]`, prerelease/build stripped. */
|
|
function parseSemverCore(version: string): [number, number, number] | null {
|
|
const core = version.trim().split('+')[0].split('-')[0]
|
|
const parts = core.split('.')
|
|
if (parts.length < 1) return null
|
|
const nums = [0, 1, 2].map((i) => {
|
|
const n = parseInt(parts[i] ?? '0', 10)
|
|
return Number.isFinite(n) ? n : NaN
|
|
})
|
|
if (nums.some((n) => Number.isNaN(n))) return null
|
|
return nums as [number, number, number]
|
|
}
|
|
|
|
/** Compare two semver cores. <0 if a<b, 0 if equal, >0 if a>b. */
|
|
function compareSemverCore(a: [number, number, number], b: [number, number, number]): number {
|
|
for (let i = 0; i < 3; i++) {
|
|
if (a[i] !== b[i]) return a[i] - b[i]
|
|
}
|
|
return 0
|
|
}
|
|
|
|
/**
|
|
* Minimal, dependency-free semver-range check for plugin↔brainy version coupling
|
|
* (see {@link BrainyPlugin.brainyRange}). Brainy is a public MIT library, so we
|
|
* avoid a full `semver` dependency; the coupling we enforce is major-version
|
|
* lockstep with the native accelerator. Supports a space-separated AND of
|
|
* `>=`, `>`, `<=`, `<`, `=` comparators plus `^X.Y.Z` (same-major, >= floor).
|
|
* Prerelease tags on the running `version` are tolerated (compared by core), so
|
|
* an RC build (`8.0.0-rc1`) satisfies `>=8.0.0`.
|
|
*
|
|
* @returns true if `version` satisfies `range`; on an unparseable range, returns
|
|
* true (fail-open on a malformed declaration rather than block a valid brain).
|
|
*/
|
|
export function pluginRangeSatisfies(version: string, range: string): boolean {
|
|
const v = parseSemverCore(version)
|
|
if (!v) return false
|
|
const comparators = range.trim().split(/\s+/).filter(Boolean)
|
|
if (comparators.length === 0) return true
|
|
for (const comp of comparators) {
|
|
const m = comp.match(/^(\^|>=|<=|>|<|=)?\s*(\d+(?:\.\d+){0,2}(?:[-+].*)?)$/)
|
|
if (!m) return true // unparseable comparator → fail-open
|
|
const op = m[1] || '='
|
|
const target = parseSemverCore(m[2])
|
|
if (!target) return true
|
|
const cmp = compareSemverCore(v, target)
|
|
let ok: boolean
|
|
switch (op) {
|
|
case '>=': ok = cmp >= 0; break
|
|
case '>': ok = cmp > 0; break
|
|
case '<=': ok = cmp <= 0; break
|
|
case '<': ok = cmp < 0; break
|
|
case '^': ok = v[0] === target[0] && cmp >= 0; break // same major, >= floor
|
|
default: ok = cmp === 0 // '='
|
|
}
|
|
if (!ok) return false
|
|
}
|
|
return true
|
|
}
|
|
|
|
export class PluginRegistry {
|
|
private plugins: Map<string, BrainyPlugin> = new Map()
|
|
private providers: Map<string, unknown> = new Map()
|
|
private activated: Set<string> = new Set()
|
|
|
|
/**
|
|
* Register a plugin manually.
|
|
*/
|
|
register(plugin: BrainyPlugin): void {
|
|
this.plugins.set(plugin.name, plugin)
|
|
}
|
|
|
|
|
|
/**
|
|
* Activate all registered plugins.
|
|
*/
|
|
async activateAll(context: BrainyPluginContext): Promise<string[]> {
|
|
const activated: string[] = []
|
|
|
|
for (const [name, plugin] of this.plugins) {
|
|
if (this.activated.has(name)) continue
|
|
|
|
// Version-coupling guard. A registered plugin is ALWAYS explicitly
|
|
// requested (config.plugins or brain.use() — brainy does no
|
|
// auto-detection), so a version mismatch must FAIL LOUD, never silently
|
|
// degrade to the default JS engine. (This was the #1 cross-repo drift:
|
|
// an old @soulcraft/cor running invisibly against brainy 8.x.)
|
|
if (plugin.brainyRange && !pluginRangeSatisfies(context.version, plugin.brainyRange)) {
|
|
throw new Error(
|
|
`[brainy] Plugin "${name}" supports brainy ${plugin.brainyRange}, but this is brainy ` +
|
|
`${context.version}. The engines must be version-matched (brainy 8.x ↔ @soulcraft/cor 3.x); ` +
|
|
`install a compatible ${name} (or brainy). brainy will NOT silently run the default engine ` +
|
|
`in place of a mismatched accelerator.`
|
|
)
|
|
}
|
|
|
|
let success: boolean
|
|
try {
|
|
success = await plugin.activate(context)
|
|
} catch (error) {
|
|
// Hard activation failure on an explicitly-requested plugin is fatal,
|
|
// not a swallow-and-fall-back-to-JS.
|
|
throw new Error(
|
|
`[brainy] Plugin "${name}" failed to activate: ` +
|
|
`${error instanceof Error ? error.message : String(error)}`
|
|
)
|
|
}
|
|
|
|
if (success) {
|
|
this.activated.add(name)
|
|
activated.push(name)
|
|
} else {
|
|
// Documented graceful decline (activate() → false). Surface it loudly so
|
|
// a silent degrade to the default engine never goes unnoticed.
|
|
console.warn(
|
|
`[brainy] Plugin "${name}" declined activation (activate() returned false); ` +
|
|
`the default engine is in use for its providers.`
|
|
)
|
|
}
|
|
}
|
|
|
|
return activated
|
|
}
|
|
|
|
/**
|
|
* Deactivate all plugins (called during close()).
|
|
*/
|
|
async deactivateAll(): Promise<void> {
|
|
for (const [name, plugin] of this.plugins) {
|
|
if (!this.activated.has(name)) continue
|
|
try {
|
|
await plugin.deactivate?.()
|
|
this.activated.delete(name)
|
|
} catch {
|
|
// Non-fatal
|
|
}
|
|
}
|
|
}
|
|
|
|
/**
|
|
* Get a registered provider by key.
|
|
*/
|
|
getProvider<T = unknown>(key: string): T | undefined {
|
|
return this.providers.get(key) as T | undefined
|
|
}
|
|
|
|
/**
|
|
* Check if a provider is registered.
|
|
*/
|
|
hasProvider(key: string): boolean {
|
|
return this.providers.has(key)
|
|
}
|
|
|
|
/**
|
|
* Register a provider (called by plugins via BrainyPluginContext).
|
|
*/
|
|
registerProvider(key: string, implementation: unknown): void {
|
|
this.providers.set(key, implementation)
|
|
}
|
|
|
|
/**
|
|
* Get a storage adapter factory by name.
|
|
*/
|
|
getStorageFactory(name: string): StorageAdapterFactory | undefined {
|
|
return this.providers.get(`storage:${name}`) as StorageAdapterFactory | undefined
|
|
}
|
|
|
|
/** Get active plugin names */
|
|
getActivePlugins(): string[] {
|
|
return [...this.activated]
|
|
}
|
|
|
|
/** Check if any plugins are active */
|
|
hasActivePlugins(): boolean {
|
|
return this.activated.size > 0
|
|
}
|
|
}
|