Merge branch 'release/8.10.0'

This commit is contained in:
David Snelling 2026-07-23 10:50:39 -07:00
commit d918c060f4
21 changed files with 1283 additions and 346 deletions

View file

@ -63,7 +63,9 @@ import type {
PathOptions,
MetadataIndexProvider,
OpaqueIdSet,
AtGenerationVectors
AtGenerationVectors,
VectorIndexProvider,
GraphIndexProvider
} from './plugin.js'
import type {
BrainyPlugin,
@ -88,12 +90,12 @@ import { findCallerLocation } from './utils/callerLocation.js'
import {
SaveNounMetadataOperation,
SaveNounOperation,
AddToHNSWOperation,
AddToVectorIndexOperation,
AddToMetadataIndexOperation,
SaveVerbMetadataOperation,
SaveVerbOperation,
AddToGraphIndexOperation,
RemoveFromHNSWOperation,
RemoveFromVectorIndexOperation,
RemoveFromMetadataIndexOperation,
RemoveFromGraphIndexOperation,
UpdateNounMetadataOperation,
@ -266,6 +268,7 @@ type ResolvedBrainyConfig = Required<
| 'retention'
| 'eagerEmbeddings'
| 'migrationWaitTimeoutMs'
| 'transactionBudgetFloorMs'
>
> &
Pick<
@ -278,6 +281,7 @@ type ResolvedBrainyConfig = Required<
| 'retention'
| 'eagerEmbeddings'
| 'migrationWaitTimeoutMs'
| 'transactionBudgetFloorMs'
>
/**
@ -388,6 +392,38 @@ class InsertPreconditionExistsSignal extends Error {
*/
export type IndexFamily = 'vector' | 'metadata' | 'graph'
/**
* @description Honest per-surface outcome for {@link Brainy.warm}. Literal
* meanings never conflate the first two:
* - `'warmed'` the provider's own `warm?()` hook ran (vector/graph), or the
* surface's full-hydration seam loaded EVERY shard/field/segment from
* storage (metadata; graph's fallback path). The surface is genuinely at
* steady-state cost for the next operation.
* - `'probed'` no `warm?()` hook was available, so a best-effort read
* (e.g. one `search()` call) faulted in *some* backing storage as a side
* effect. Real work happened, but it is NOT the same guarantee as
* `'warmed'` never reported as `'warmed'`.
* - `'unavailable'` nothing ran: no hook, no hydration seam, and (for the
* vector probe fallback) nothing to probe (an empty index or unknown
* vector dimension). The surface is unchanged by this `warm()` call.
*/
export type WarmOutcome = 'warmed' | 'probed' | 'unavailable'
/**
* @description Result of {@link Brainy.warm}: one {@link WarmOutcome} +
* elapsed time per index surface, plus the total wall-clock time for the
* whole call. `durationMs` is measured around exactly the work described by
* that surface's `outcome` (e.g. the vector entry's `durationMs` times the
* provider `warm()` call OR the probe `search()` call whichever ran).
*/
export interface WarmReport {
vector: { outcome: WarmOutcome; durationMs: number }
metadata: { outcome: WarmOutcome; durationMs: number }
graph: { outcome: WarmOutcome; durationMs: number }
/** Total wall-clock time for the whole `warm()` call (all three surfaces). */
totalDurationMs: number
}
/**
* How long a failed aggregation-backfill walk suppresses fresh walk attempts.
* Within the window, queries rethrow the recorded failure instantly (loud,
@ -1382,6 +1418,17 @@ export class Brainy<T = any> implements BrainyInterface<T> {
})
}
// Eager index warm (operator opt-in, `warmOnOpen: true`). Runs AFTER
// every step above — index construction, crash recovery, migrations,
// VFS bootstrap — never in place of any of it, so `warm()` always
// operates on a fully-initialized brain. Blocking BY DESIGN: the
// operator traded a longer startup for a first-request that runs at
// steady-state cost instead of paying demand-load latency on the
// critical path. See the `warmOnOpen` JSDoc in brainy.types.ts.
if (this.config.warmOnOpen) {
await this.warm()
}
// Resolve ready Promise - consumers awaiting brain.ready will now proceed
if (this._readyResolve) {
this._readyResolve()
@ -1780,7 +1827,9 @@ export class Brainy<T = any> implements BrainyInterface<T> {
await this.generationStore.runWithoutGeneration(() =>
this.transactionManager.executeTransaction(run, {
timeout: transactTimeoutBudget(
(touched.nouns?.length ?? 0) + (touched.verbs?.length ?? 0)
(touched.nouns?.length ?? 0) + (touched.verbs?.length ?? 0),
undefined,
this.config.transactionBudgetFloorMs
)
})
)
@ -1797,7 +1846,9 @@ export class Brainy<T = any> implements BrainyInterface<T> {
execute: () =>
this.transactionManager.executeTransaction(run, {
timeout: transactTimeoutBudget(
(touched.nouns?.length ?? 0) + (touched.verbs?.length ?? 0)
(touched.nouns?.length ?? 0) + (touched.verbs?.length ?? 0),
undefined,
this.config.transactionBudgetFloorMs
)
})
})
@ -2109,7 +2160,7 @@ export class Brainy<T = any> implements BrainyInterface<T> {
// Operation 3: Add to HNSW index (after entity saved)
tx.addOperation(
new AddToHNSWOperation(this.index, id, vector)
new AddToVectorIndexOperation(this.index, id, vector)
)
// Operation 4: Add to metadata index
@ -3098,10 +3149,10 @@ export class Brainy<T = any> implements BrainyInterface<T> {
// Operation 3-4: Update HNSW index (remove and re-add if reindexing needed)
if (needsReindexing) {
tx.addOperation(
new RemoveFromHNSWOperation(this.index, params.id, existing.vector)
new RemoveFromVectorIndexOperation(this.index, params.id, existing.vector)
)
tx.addOperation(
new AddToHNSWOperation(this.index, params.id, vector)
new AddToVectorIndexOperation(this.index, params.id, vector)
)
}
@ -3217,7 +3268,7 @@ export class Brainy<T = any> implements BrainyInterface<T> {
// Operation 1: Remove from vector index
if (noun) {
tx.addOperation(
new RemoveFromHNSWOperation(this.index, id, noun.vector)
new RemoveFromVectorIndexOperation(this.index, id, noun.vector)
)
}
@ -7025,7 +7076,7 @@ export class Brainy<T = any> implements BrainyInterface<T> {
// Add delete operations to transaction
if (noun) {
tx.addOperation(
new RemoveFromHNSWOperation(this.index, id, noun.vector)
new RemoveFromVectorIndexOperation(this.index, id, noun.vector)
)
}
@ -7888,7 +7939,13 @@ export class Brainy<T = any> implements BrainyInterface<T> {
// Budget scales with the batch (or the caller's explicit
// timeoutMs): a flat 30s cap silently limited honest bulk work
// to ~15 ops on network disks (~2s/op measured in the field).
{ timeout: transactTimeoutBudget(plan.operations.length, options?.timeoutMs) }
{
timeout: transactTimeoutBudget(
plan.operations.length,
options?.timeoutMs,
this.config.transactionBudgetFloorMs
)
}
)
}
}))
@ -9311,7 +9368,7 @@ export class Brainy<T = any> implements BrainyInterface<T> {
plan.operations.push(
new SaveNounMetadataOperation(this.storage, id, storageMetadata, isNew),
new SaveNounOperation(this.storage, { id, vector, connections: new Map(), level: 0 }, isNew),
new AddToHNSWOperation(this.index, id, vector),
new AddToVectorIndexOperation(this.index, id, vector),
new AddToMetadataIndexOperation(this.metadataIndex, id, entityForIndexing)
)
plan.touchedNouns.push(id)
@ -9479,8 +9536,8 @@ export class Brainy<T = any> implements BrainyInterface<T> {
)
if (needsReindexing) {
plan.operations.push(
new RemoveFromHNSWOperation(this.index, params.id, existing.vector),
new AddToHNSWOperation(this.index, params.id, vector)
new RemoveFromVectorIndexOperation(this.index, params.id, existing.vector),
new AddToVectorIndexOperation(this.index, params.id, vector)
)
}
plan.operations.push(
@ -9565,7 +9622,7 @@ export class Brainy<T = any> implements BrainyInterface<T> {
}
if (noun) {
plan.operations.push(new RemoveFromHNSWOperation(this.index, id, noun.vector))
plan.operations.push(new RemoveFromVectorIndexOperation(this.index, id, noun.vector))
}
if (metadata) {
plan.operations.push(new RemoveFromMetadataIndexOperation(this.metadataIndex, id, metadata))
@ -14280,6 +14337,118 @@ export class Brainy<T = any> implements BrainyInterface<T> {
return this.embedder(textToEmbed)
}
/**
* Eagerly load/fault-in the vector index, metadata index, and graph
* adjacency so the FIRST real operation after this call runs at
* steady-state cost no page-cache-miss / demand-load latency on the
* critical path. Complements {@link warmupEmbeddings} (which warms the
* embedding engine, not the storage indexes).
*
* Sequence per surface:
* - **Vector**: calls the provider's own `warm?()` when the active
* `'vector'` provider implements it (`'warmed'`). Otherwise runs a
* best-effort probe one `search()` with a deterministic unit vector of
* the index's known dimension, `k = min(100, size())` which faults in
* *some* backing storage as a side effect but is reported honestly as
* `'probed'`, never `'warmed'`. An empty index or unknown dimension has
* nothing to probe (`'unavailable'`).
* - **Metadata**: full hydration every persisted field's sparse index is
* loaded from storage (`MetadataIndexManager.hydrateAll()`), not just the
* heuristic common-fields subset `init()` warms.
* - **Graph**: calls the provider's own `warm?()` when the active graph
* provider implements it; otherwise re-runs its existing eager cold-load
* `init()` seam (idempotent the JS adjacency index's `init()` already
* loads the full LSM manifests + SSTables, so re-invoking it here is a
* genuine full hydration, not a stat call).
*
* A single surface's `'unavailable'`/probe outcome never fails the whole
* call `warm()` is a best-effort readiness step, not a correctness gate;
* queries still demand-load normally regardless of what `warm()` achieved.
*
* @returns A {@link WarmReport}: honest per-surface outcome + timing.
* @example
* ```typescript
* const brain = new Brainy({ storage: { path: '/data' } })
* await brain.init()
* const report = await brain.warm() // blocking, explicit control over timing
* console.log(report.vector.outcome, report.vector.durationMs)
*
* // Or opt into the same work automatically during init():
* const eager = new Brainy({ storage: { path: '/data' }, warmOnOpen: true })
* await eager.init() // warm() already ran before this resolves
* ```
*/
async warm(): Promise<WarmReport> {
await this.ensureInitialized({ needs: ['vector', 'metadata', 'graph'] })
const totalStart = Date.now()
// --- Vector ---------------------------------------------------------
const vectorStart = Date.now()
let vectorOutcome: WarmOutcome
const vectorProvider = this.index as VectorIndexProvider & { warm?: () => Promise<void> }
if (typeof vectorProvider.warm === 'function') {
await vectorProvider.warm()
vectorOutcome = 'warmed'
} else {
const size = this.index.size()
const dimension = this.dimensions
if (size > 0 && dimension && dimension > 0) {
// A deterministic unit vector (L2 norm 1) — same probe every call,
// no reliance on stored data shape.
const probeVector = new Array(dimension).fill(1 / Math.sqrt(dimension))
await this.index.search(probeVector, Math.min(100, size))
vectorOutcome = 'probed'
} else {
// Nothing to probe: an empty index, or the vector dimension is not
// yet known (no entity has ever been added).
vectorOutcome = 'unavailable'
}
}
const vectorDurationMs = Date.now() - vectorStart
// --- Metadata --------------------------------------------------------
const metadataStart = Date.now()
let metadataOutcome: WarmOutcome
const metadataWithHydrate = this.metadataIndex as unknown as { hydrateAll?: () => Promise<void> }
if (typeof metadataWithHydrate.hydrateAll === 'function') {
await metadataWithHydrate.hydrateAll()
metadataOutcome = 'warmed'
} else {
// No hydration seam on this metadata provider — nothing to run.
metadataOutcome = 'unavailable'
}
const metadataDurationMs = Date.now() - metadataStart
// --- Graph -------------------------------------------------------------
const graphStart = Date.now()
let graphOutcome: WarmOutcome
const graphProvider = this.graphIndex as GraphIndexProvider & {
warm?: () => Promise<void>
init?: () => Promise<void>
}
if (typeof graphProvider.warm === 'function') {
await graphProvider.warm()
graphOutcome = 'warmed'
} else if (typeof graphProvider.init === 'function') {
// Existing eager cold-load seam (readiness contract): idempotent, and
// for the JS adjacency index it's already a genuine full load of the
// LSM manifests + SSTables — a real hydration, not a stat call.
await graphProvider.init()
graphOutcome = 'warmed'
} else {
graphOutcome = 'unavailable'
}
const graphDurationMs = Date.now() - graphStart
return {
vector: { outcome: vectorOutcome, durationMs: vectorDurationMs },
metadata: { outcome: metadataOutcome, durationMs: metadataDurationMs },
graph: { outcome: graphOutcome, durationMs: graphDurationMs },
totalDurationMs: Date.now() - totalStart
}
}
/**
* Explicitly warm up the embedding engine
*
@ -14798,6 +14967,9 @@ export class Brainy<T = any> implements BrainyInterface<T> {
// Migration LOCK wait budget — left undefined when omitted so
// awaitMigrationLock() applies its 30 s default at the read site.
migrationWaitTimeoutMs: config?.migrationWaitTimeoutMs ?? undefined,
// Apply-phase transaction budget floor — left undefined when omitted so
// transactTimeoutBudget() applies its 30 s default at the read site.
transactionBudgetFloorMs: config?.transactionBudgetFloorMs ?? undefined,
// Pre-upgrade backup — default-on; opt out with `migrationBackup: false`.
migrationBackup: config?.migrationBackup ?? true,
// Vector index configuration (8.0) — algorithm-neutral surface with
@ -14813,6 +14985,9 @@ export class Brainy<T = any> implements BrainyInterface<T> {
// is the active one on a non-reader writer outside unit tests). Explicit
// true/false always wins. See the eagerEmbeddings JSDoc in brainy.types.ts.
eagerEmbeddings: config?.eagerEmbeddings ?? undefined,
// Eager index warm at init() — default off (lazy demand-load), the
// pre-8.9 behavior. See the warmOnOpen JSDoc in brainy.types.ts.
warmOnOpen: config?.warmOnOpen ?? false,
// Plugin configuration - undefined = auto-detect
plugins: config?.plugins ?? undefined,
// Integration Hub - undefined/false = disabled

View file

@ -54,6 +54,9 @@ export class HnswFlushError extends Error {
* acceleration provider).
*/
export class JsHnswVectorIndex implements VectorIndexProvider {
/** Self-identifies as the built-in JS fallback engine — see {@link VectorIndexProvider.name}. */
readonly name = 'hnsw-js'
private nouns: Map<string, HNSWNoun> = new Map()
/**
* Reverse adjacency: `target id → (level → set of node ids that link TO target)`.

View file

@ -28,6 +28,9 @@ export type { FileVersion } from './vfs/types.js'
// Export diagnostics result type
export type { DiagnosticsResult } from './brainy.js'
// brain.warm() — eager index/storage readiness report (per-surface honest
// outcome + timing). See the WarmReport JSDoc in brainy.ts.
export type { WarmReport, WarmOutcome } from './brainy.js'
export type {
GraphAuditReport,
GraphAuditDiscrepancy

View file

@ -381,6 +381,20 @@ export interface GraphIndexProvider {
*/
init?(): Promise<void>
/**
* @description OPTIONAL. Eagerly load/fault-in backing storage (e.g. mmap
* pretouch) so first operations run at steady-state cost. Optional;
* absence means the provider demand-loads. Distinct from `init?()`: `init`
* is called once automatically during brain startup as part of the
* readiness contract (cold-load + rebuild gating); `warm` is an explicit,
* separate readiness step a caller opts into via `brain.warm()` (or
* `warmOnOpen`) specifically to pre-pay demand-load cost that `init` left
* lazy. A provider that already loads everything eagerly in `init?()` may
* implement `warm` as a no-op or omit it `brain.warm()` falls back to a
* best-effort read-through probe when absent.
*/
warm?(): Promise<void>
/**
* @description OPTIONAL. A native provider returns true from the moment its
* `init()` detects a large epoch-drift until its background
@ -947,6 +961,26 @@ export interface AtGenerationVectors {
* are retired and never looked up.
*/
export interface VectorIndexProvider {
/**
* @description REQUIRED self-reported implementation identity, rendered as
* `[vector-index:<name>]` in prose log lines and stamped into the
* transaction op-name strings that surface in journals/timings (e.g.
* `AddToVectorIndex(hnsw-js)`) see
* `src/transaction/operations/IndexOperations.ts`. A provider must name
* itself TRUTHFULLY (its own algorithm/engine, e.g. its plugin/package
* name) and must never inherit a default guessing a backend name would
* resurrect the exact bug this field exists to prevent: an operator
* hunting an index that isn't the one actually running. The built-in JS
* index always sets this to `'hnsw-js'`; a native provider picks its own
* string. At the TypeScript level this field is required; a runtime
* instance from an older provider compiled against the previous optional
* `providerId` contract is tolerated (never crashes, never silently
* mislabeled) see `resolveVectorProviderId` in
* `src/transaction/operations/IndexOperations.ts`, which stamps
* `'unknown-provider'` and emits one loud warning for that case.
*/
readonly name: string
addItem(item: VectorDocument): Promise<string>
removeItem(id: string): Promise<boolean>
search(
@ -1009,6 +1043,20 @@ export interface VectorIndexProvider {
*/
init?(): Promise<void>
/**
* @description OPTIONAL. Eagerly load/fault-in backing storage (e.g. mmap
* pretouch) so first operations run at steady-state cost. Optional;
* absence means the provider demand-loads. Distinct from `init?()`: `init`
* runs automatically once during brain startup (the cold-load + rebuild
* readiness contract above); `warm` is a separate, explicit step a caller
* opts into via `brain.warm()` (or `warmOnOpen`) to pre-pay demand-load
* cost `init` left lazy e.g. touching every mmap page rather than just
* opening the file. A provider that already loads everything eagerly in
* `init?()` may implement `warm` as a no-op or omit it `brain.warm()`
* falls back to a best-effort probe `search()` when absent.
*/
warm?(): Promise<void>
/**
* @description OPTIONAL honest durability signal (readiness contract,
* mirrors {@link GraphIndexProvider.isReady}). `true` the persisted

View file

@ -37,22 +37,47 @@ const DEFAULT_OPTIONS: Required<TransactionOptions> = {
maxRollbackRetries: 3
}
/**
* The floor term of {@link transactTimeoutBudget}'s scaling formula when the
* caller supplies neither an explicit override nor `config.transactionBudgetFloorMs`.
*/
const DEFAULT_BUDGET_FLOOR_MS = 30_000
/**
* The apply-phase budget for a batch of `opCount` operations.
*
* An explicit override wins untouched. Otherwise the budget SCALES with the
* batch: `max(30 000 ms, opCount × 2 000 ms)`. The per-op term is calibrated
* from field data bulk imports on network-attached disks measure ~2 s per
* operation (each op pays canonical writes + fsync + index maintenance) so
* a flat 30 s budget silently capped honest work at ~15 operations while
* looking generous for small batches. Scaling keeps small transacts
* fast-failing and gives bulk ones a budget proportional to the work they
* actually asked for; a trip still rolls back atomically and throws a
* retryable, fully-labeled TransactionTimeoutError.
* An explicit `override` wins untouched (a caller-specified deadline for this
* one batch). Otherwise the budget SCALES with the batch:
* `max(floorMs, opCount × 2 000)`, where `floorMs` defaults to `30 000` and
* is overridable via `BrainyConfig.transactionBudgetFloorMs` for the whole
* brain. The per-op term is calibrated from field data bulk imports on
* network-attached disks measure ~2 s per operation (each op pays canonical
* writes + fsync + index maintenance) so a flat 30 s budget silently capped
* honest work at ~15 operations while looking generous for small batches. A
* 16-op batch, for example, scales to `max(30 000, 16 × 2 000) = 32 000`.
* Scaling keeps small transacts fast-failing and gives bulk ones a budget
* proportional to the work they actually asked for.
*
* This is a **mid-batch** budget only: it governs whether the transaction's
* NEXT operation may start (see {@link Transaction.execute}), never whether
* already-completed work is rolled back after the fact. A trip mid-batch
* still rolls back every operation applied so far, atomically, and throws a
* retryable, fully-labeled TransactionTimeoutError that zero-loss guarantee
* doesn't change; only the point at which the clock stops mattering does (at
* the last operation, not one check later).
*
* @param opCount - Number of operations in the batch.
* @param override - A full override for this call; wins over everything else.
* @param floorMs - The scaling floor for this call (e.g. `config.transactionBudgetFloorMs`).
* Ignored when `override` is set. Defaults to `30 000` when omitted.
*/
export function transactTimeoutBudget(opCount: number, override?: number): number {
export function transactTimeoutBudget(
opCount: number,
override?: number,
floorMs?: number
): number {
if (override !== undefined) return override
return Math.max(30_000, opCount * 2_000)
return Math.max(floorMs ?? DEFAULT_BUDGET_FLOOR_MS, opCount * 2_000)
}
/**
@ -98,7 +123,20 @@ export class Transaction implements TransactionContext {
}
/**
* Execute all operations atomically
* Execute all operations atomically.
*
* Budget semantics: the configured budget gates starting the next
* operation; completed work is never rolled back for elapsed time. Elapsed
* time is checked ONLY before starting operation `i` for `i ≥ 1` never
* before operation 0 (an operation always gets to start) and never again
* after the final operation completes. Concretely: a single-op transaction
* can never time out post-hoc it either runs (and its result stands, no
* matter how long it took) or a `TransactionTimeoutError` is thrown before
* it starts, which cannot happen since there is no operation before it. A
* multi-op transaction whose op `i` overruns the budget stops op `i+1` from
* starting, rolls back operations `0..i` (reverse order), and throws
* `TransactionTimeoutError` that zero-loss rollback guarantee for
* mid-batch trips is unchanged; only the post-completion check is gone.
*/
async execute(): Promise<void> {
if (this.state !== 'pending') {
@ -128,10 +166,14 @@ export class Transaction implements TransactionContext {
// already-applied writes as torn, generation-less state in canonical
// storage.
for (let i = 0; i < this.operations.length; i++) {
// Budget check BEFORE starting the next operation. A trip here throws
// into the catch below and rolls back like any other failure — it must
// never bypass rollback.
if (Date.now() - this.startTime > this.options.timeout) {
// Budget gates STARTING the next operation; completed work is never
// rolled back for elapsed time. Skipped for i === 0 (operation 0
// always gets to start — nothing has run yet to have overrun) and
// never re-checked after the final operation completes (there is no
// "next operation" left to gate). A trip here throws into the catch
// below and rolls back like any other failure — it must never bypass
// rollback.
if (i > 0 && Date.now() - this.startTime > this.options.timeout) {
throw new TransactionTimeoutError(this.options.timeout, i, {
elapsedMs: Date.now() - this.startTime,
totalOperations: this.operations.length,

View file

@ -2,34 +2,73 @@
* Index Operations with Rollback Support
*
* Provides transactional operations for all indexes:
* - JsHnswVectorIndex (unified vector index)
* - VectorIndexProvider (the JS HNSW fallback, or a native acceleration provider)
* - MetadataIndexManager (roaring bitmap filtering)
* - GraphAdjacencyIndex (LSM-tree graph storage)
*
* Each operation can be executed and rolled back atomically.
*/
import type { JsHnswVectorIndex } from '../../hnsw/hnswIndex.js'
import type { VectorIndexProvider, GraphIndexProvider } from '../../plugin.js'
import type { MetadataIndexManager } from '../../utils/metadataIndex.js'
import type { GraphIndexProvider } from '../../plugin.js'
import type { GraphVerb } from '../../coreTypes.js'
import type { Operation, RollbackAction } from '../types.js'
/**
* Add to HNSW index with rollback support
* Backend identity stamped into an operation's emitted `name` string (e.g.
* `AddToVectorIndex(hnsw-js)`), resolved from the provider's own
* {@link VectorIndexProvider.name} self-report.
*
* These operation classes are backend-neutral (the vector index they wrap may
* be Brainy's own JS HNSW fallback OR a native acceleration provider), but
* their names surface directly in consumer-visible transaction journals and
* timings. `name` is REQUIRED at the TypeScript level but a native provider
* instance compiled against the previous (pre-required) contract can still
* reach this function at runtime without it. That legacy case is tolerated,
* never crashed on and never silently mislabeled: it resolves to
* `'unknown-provider'` and emits ONE loud warning naming the missing contract
* field, so the fix (implement `name`) is discoverable rather than a silent
* fossil label in every journal line thereafter.
*/
const warnedMissingName = new WeakSet<VectorIndexProvider>()
function resolveVectorProviderId(index: VectorIndexProvider): string {
const name = (index as { name?: unknown }).name
if (typeof name === 'string') return name
if (!warnedMissingName.has(index)) {
warnedMissingName.add(index)
console.warn(
'[vector-index] provider is missing the required `name` field (VectorIndexProvider.name, ' +
'required since 8.10.0) — stamping "unknown-provider" in transaction op names until the ' +
'provider declares its own identity.'
)
}
return 'unknown-provider'
}
/**
* Add to the vector index with rollback support.
*
* Backend-neutral: `index` is whatever the `'vector'` provider factory
* returns Brainy's own JS HNSW fallback, or a native acceleration provider
* (e.g. DiskANN). The emitted `name` stamps the active backend (see
* {@link resolveVectorProviderId}) so operators reading a transaction journal
* or timing trace see which engine actually ran, never a fossil name from
* whichever engine happened to be active when this op class was written.
*
* Rollback strategy:
* - Remove item from index
*/
export class AddToHNSWOperation implements Operation {
readonly name = 'AddToHNSW'
export class AddToVectorIndexOperation implements Operation {
readonly name: string
constructor(
private readonly index: JsHnswVectorIndex,
private readonly index: VectorIndexProvider,
private readonly id: string,
private readonly vector: number[]
) {}
) {
this.name = `AddToVectorIndex(${resolveVectorProviderId(index)})`
}
async execute(): Promise<RollbackAction> {
// Check if item already exists (for rollback decision)
@ -59,11 +98,12 @@ export class AddToHNSWOperation implements Operation {
* pre-existence as "existed" made every rollback skip removeItem, leaving
* phantom entries in the index after a failed transaction. The safe default
* is to remove what this operation added update flows pair this op with a
* RemoveFromHNSWOperation whose own rollback restores the prior vector, so
* reverse-order rollback reconstructs the original state either way.
* RemoveFromVectorIndexOperation whose own rollback restores the prior
* vector, so reverse-order rollback reconstructs the original state either
* way.
*/
private async itemExists(id: string): Promise<boolean> {
const index = this.index as JsHnswVectorIndex & {
const index = this.index as VectorIndexProvider & {
getItem?: (id: string) => Promise<unknown>
}
if (typeof index.getItem !== 'function') return false
@ -77,21 +117,27 @@ export class AddToHNSWOperation implements Operation {
}
/**
* Remove from HNSW index with rollback support
* Remove from the vector index with rollback support.
*
* Backend-neutral: see {@link AddToVectorIndexOperation} `index` may be the
* JS HNSW fallback or a native acceleration provider; the emitted `name`
* stamps the active backend.
*
* Rollback strategy:
* - Re-add item to index with original vector
*
* Note: Requires storing the vector for rollback
*/
export class RemoveFromHNSWOperation implements Operation {
readonly name = 'RemoveFromHNSW'
export class RemoveFromVectorIndexOperation implements Operation {
readonly name: string
constructor(
private readonly index: JsHnswVectorIndex,
private readonly index: VectorIndexProvider,
private readonly id: string,
private readonly vector: number[] // Required for rollback
) {}
) {
this.name = `RemoveFromVectorIndex(${resolveVectorProviderId(index)})`
}
async execute(): Promise<RollbackAction> {
// Remove from index
@ -285,22 +331,24 @@ export class RemoveFromGraphIndexOperation implements Operation {
}
/**
* Batch operation: Add multiple items to HNSW index
* Batch operation: Add multiple items to the vector index (backend-neutral
* see {@link AddToVectorIndexOperation}).
*
* Useful for bulk imports with transaction support.
* Rolls back all items if any fail.
*/
export class BatchAddToHNSWOperation implements Operation {
readonly name = 'BatchAddToHNSW'
export class BatchAddToVectorIndexOperation implements Operation {
readonly name: string
private operations: AddToHNSWOperation[]
private operations: AddToVectorIndexOperation[]
constructor(
index: JsHnswVectorIndex,
index: VectorIndexProvider,
items: Array<{ id: string; vector: number[] }>
) {
this.name = `BatchAddToVectorIndex(${resolveVectorProviderId(index)})`
this.operations = items.map(
item => new AddToHNSWOperation(index, item.id, item.vector)
item => new AddToVectorIndexOperation(index, item.id, item.vector)
)
}

View file

@ -21,12 +21,12 @@ export {
// Index Operations
export {
AddToHNSWOperation,
RemoveFromHNSWOperation,
AddToVectorIndexOperation,
RemoveFromVectorIndexOperation,
AddToMetadataIndexOperation,
RemoveFromMetadataIndexOperation,
AddToGraphIndexOperation,
RemoveFromGraphIndexOperation,
BatchAddToHNSWOperation,
BatchAddToVectorIndexOperation,
BatchAddToMetadataIndexOperation
} from './IndexOperations.js'

View file

@ -1651,6 +1651,25 @@ export interface BrainyConfig {
*/
disableAutoRebuild?: boolean
/**
* The floor (ms) of the apply-phase transaction budget's scaling formula:
* `max(transactionBudgetFloorMs, opCount × 2 000)`. The budget governs
* whether a `transact()` / single-op write's NEXT internal operation may
* **start** never whether already-completed work gets rolled back after
* the fact (a single-op write can never time out post-hoc: it either runs
* or it commits). A trip mid-batch still rolls back every applied operation
* atomically and throws a retryable `TransactionTimeoutError`; only the
* floor of the formula is configurable here.
*
* Raise this when a cold store's first writes after a restart legitimately
* take longer than 30s per operation (e.g. page-cache-cold canonical writes
* on a large brain) so a bulk `transact()` batch gets a proportionally
* larger runway instead of tripping mid-batch. Lower it to fail faster on a
* latency-sensitive write path. Default: `30000` (30 s) unchanged from
* pre-8.9 behavior.
*/
transactionBudgetFloorMs?: number
/**
* How long (ms) an operation **waits** on the coordinated 7.x 8.0 migration
* before throwing a retryable `MigrationInProgressError`.
@ -1806,6 +1825,23 @@ export interface BrainyConfig {
*/
eagerEmbeddings?: boolean
/**
* When `true`, `init()` **awaits `warm()`** (see {@link Brainy.warm}) before
* it resolves blocking startup until the vector index, metadata index,
* and graph adjacency have all faulted their backing storage in. This is a
* deliberate operator opt-in trade: startup takes longer, but the FIRST
* request after a cold restart runs at steady-state cost instead of paying
* page-cache-miss / demand-load latency on the critical path.
*
* Runs AFTER `init()`'s own sequence completes (index construction, crash
* recovery, migrations, VFS bootstrap) never in place of any of it so
* `warm()` always operates on a fully-initialized brain.
*
* Default: `false` (lazy the pre-8.9 behavior: indexes demand-load on
* first access).
*/
warmOnOpen?: boolean
// Plugin configuration
// Controls which plugins are loaded during init().
// - undefined (default): guarded auto-detection of the first-party

View file

@ -423,6 +423,46 @@ export class MetadataIndexManager implements MetadataIndexProvider {
prodLog.debug('✅ Type-aware cache warming completed')
}
/**
* Full hydration the {@link Brainy.warm} readiness seam for the metadata
* index. Unlike {@link warmCache} / {@link warmCacheForTopTypes} (which
* warm only a heuristic subset: common fields plus the top-N types' top
* fields), this loads EVERY field's sparse index the field registry knows
* about a real read through {@link loadSparseIndex} into the unified
* cache for each field, not a stat/existence check. Idempotent: an
* already-cached field's `loadSparseIndex` call is a cheap cache hit.
*
* Re-reads the field registry first when `fieldIndexes` is empty (a warm()
* call issued before `init()` populated it would otherwise hydrate
* nothing), then loads every discovered field in parallel.
*/
async hydrateAll(): Promise<void> {
if (this.fieldIndexes.size === 0) {
await this.loadFieldRegistry()
}
const fields = Array.from(this.fieldIndexes.keys())
if (fields.length === 0) {
prodLog.debug('[MetadataIndex] hydrateAll: no persisted fields to hydrate')
return
}
prodLog.debug(`[MetadataIndex] hydrateAll: loading ${fields.length} field(s) — ${fields.join(', ')}`)
await Promise.all(
fields.map(async field => {
try {
await this.loadSparseIndex(field)
} catch (error) {
// A single field's load failure doesn't abort the rest of the
// hydration — warm() is a best-effort readiness step, never a
// correctness gate (queries still demand-load on miss).
prodLog.debug(`[MetadataIndex] hydrateAll: field '${field}' failed to load:`, error)
}
})
)
}
/**
* Acquire an in-memory lock for coordinating concurrent metadata index writes
* Uses in-memory locks since MetadataIndexManager doesn't have direct file system access