Three pieces addressing the cold-restart-write incident where a production deployment's first writes after every restart (33-35s each on a cold page cache) blew the op-count-scaled transact budget mid-batch: every write is itself a multi-op transaction, so one cold operation consumed the whole budget, the gate before the next operation tripped, and the write rolled back atomically - refused, retried, and refused again until the page cache warmed passively. - The budget's start-gating contract is now explicit and pinned: it gates STARTING the next operation, never rolling back completed work for elapsed time (the shipped schedule since 8.7.0, now stated in contract JSDoc, guarded by code for operation 0, and enforced by regression tests). The 30s floor is configurable via transactionBudgetFloorMs for stores whose cold operations legitimately run long. - New brain.warm() eagerly loads the vector index, metadata index, and graph adjacency so first operations after a cold restart run at steady-state cost. Returns a WarmReport with an honest per-surface outcome (warmed / probed / unavailable) - never reports a probe as a warm. warmOnOpen: true runs it during init(). New optional provider hook warm() on the vector and graph plugin contracts. - Vector-index transaction op classes renamed from the backend-specific AddToHNSWOperation / RemoveFromHNSWOperation to backend-neutral AddToVectorIndexOperation / RemoveFromVectorIndexOperation, stamping the active backend into the emitted op-name string (AddToVectorIndex(js-hnsw) vs a native provider's own identity) so journals never misdirect an operator toward an index that isn't running.
309 lines
11 KiB
TypeScript
309 lines
11 KiB
TypeScript
/**
|
|
* @module tests/unit/brainy/warm
|
|
* @description Coverage for `brain.warm()` / `warmOnOpen` / the provider
|
|
* `warm?()` contract (the cold-restart readiness fix): first operations after
|
|
* a cold restart should run at steady-state cost instead of paying
|
|
* demand-load latency on the critical path.
|
|
*
|
|
* Uses a real, in-process fake plugin provider implementing the actual
|
|
* `VectorIndexProvider` contract from `src/plugin.ts` — the real seam a
|
|
* native provider (e.g. a disk-native accelerator) plugs into. The real
|
|
* built-in JS metadata index and graph adjacency index run against real
|
|
* (filesystem or in-memory) storage with pre-existing data, so their
|
|
* hydration paths are exercised for real, not mocked.
|
|
*/
|
|
import { describe, it, expect, afterEach } from 'vitest'
|
|
import * as fs from 'node:fs'
|
|
import * as os from 'node:os'
|
|
import * as path from 'node:path'
|
|
import { Brainy } from '../../../src/brainy.js'
|
|
import { NounType, VerbType } from '../../../src/types/graphTypes.js'
|
|
import type { VectorIndexProvider } from '../../../src/plugin.js'
|
|
import type { VectorDocument, Vector } from '../../../src/coreTypes.js'
|
|
import { MetadataIndexManager } from '../../../src/utils/metadataIndex.js'
|
|
import { GraphAdjacencyIndex } from '../../../src/graph/graphAdjacencyIndex.js'
|
|
|
|
const tmpDirs: string[] = []
|
|
function mkTmp(): string {
|
|
const d = fs.mkdtempSync(path.join(os.tmpdir(), 'brainy-warm-'))
|
|
tmpDirs.push(d)
|
|
return d
|
|
}
|
|
afterEach(() => {
|
|
for (const d of tmpDirs.splice(0)) fs.rmSync(d, { recursive: true, force: true })
|
|
})
|
|
|
|
// Brainy's ValidationConfig fixes vectors at exactly 384 dimensions
|
|
// (src/utils/paramValidation.ts) — match it so `add()` doesn't reject test data.
|
|
const DIM = 384
|
|
const V = (seed = 1): number[] => Array.from({ length: DIM }, (_, i) => Math.sin(seed + i))
|
|
|
|
/**
|
|
* A real (not mocked) VectorIndexProvider implementation, backed by a plain
|
|
* Map, that optionally implements `warm()` — the exact seam
|
|
* `AddToVectorIndexOperation` / `brain.warm()` call through.
|
|
*/
|
|
class FakeVectorProvider implements VectorIndexProvider {
|
|
readonly items = new Map<string, Vector>()
|
|
warmCalls = 0
|
|
searchCalls: Array<{ k?: number }> = []
|
|
// Only present on the instance when the constructor is told to — mirrors a
|
|
// real provider that may or may not implement the optional hook. Assigned
|
|
// in the constructor BODY (not as a field initializer): under native ES
|
|
// class fields, field initializers run before constructor-body statements
|
|
// — including the parameter-property assignment — so referencing a
|
|
// parameter property from a field initializer would see it as still
|
|
// `undefined`.
|
|
warm?: () => Promise<void>
|
|
|
|
constructor(hasWarm: boolean) {
|
|
if (hasWarm) {
|
|
this.warm = async (): Promise<void> => {
|
|
this.warmCalls++
|
|
}
|
|
}
|
|
}
|
|
|
|
async addItem(item: VectorDocument): Promise<string> {
|
|
this.items.set(item.id, item.vector)
|
|
return item.id
|
|
}
|
|
async removeItem(id: string): Promise<boolean> {
|
|
return this.items.delete(id)
|
|
}
|
|
async search(_queryVector: Vector, k?: number): Promise<Array<[string, number]>> {
|
|
this.searchCalls.push({ k })
|
|
return [...this.items.keys()].slice(0, k ?? 10).map((id) => [id, 0])
|
|
}
|
|
size(): number {
|
|
return this.items.size
|
|
}
|
|
clear(): void {
|
|
this.items.clear()
|
|
}
|
|
async rebuild(): Promise<void> {}
|
|
async flush(): Promise<number> {
|
|
return 0
|
|
}
|
|
getPersistMode(): 'immediate' | 'deferred' {
|
|
return 'deferred'
|
|
}
|
|
}
|
|
|
|
/** Registers `provider` under the `'vector'` plugin key, before `init()`. */
|
|
function useFakeVectorProvider(brain: Brainy<any>, provider: FakeVectorProvider): void {
|
|
brain.use({
|
|
name: 'fake-vector-provider',
|
|
activate: async (ctx: any) => {
|
|
ctx.registerProvider('vector', () => provider)
|
|
return true
|
|
}
|
|
})
|
|
}
|
|
|
|
describe('brain.warm()', () => {
|
|
it('(a) calls the vector provider\'s warm() when present and reports "warmed"', async () => {
|
|
const brain = new Brainy({
|
|
requireSubtype: false,
|
|
storage: { type: 'memory' },
|
|
silent: true
|
|
})
|
|
const provider = new FakeVectorProvider(true)
|
|
useFakeVectorProvider(brain, provider)
|
|
await brain.init()
|
|
await brain.add({ data: 'a', type: NounType.Thing, vector: V(1) })
|
|
|
|
const report = await brain.warm()
|
|
|
|
expect(provider.warmCalls).toBe(1)
|
|
expect(provider.searchCalls.length).toBe(0) // warm() ran — no probe fallback
|
|
expect(report.vector.outcome).toBe('warmed')
|
|
await brain.close()
|
|
})
|
|
|
|
it('(b) falls back to a probe search() when warm() is absent and reports "probed", never "warmed"', async () => {
|
|
const brain = new Brainy({
|
|
requireSubtype: false,
|
|
storage: { type: 'memory' },
|
|
silent: true
|
|
})
|
|
const provider = new FakeVectorProvider(false)
|
|
useFakeVectorProvider(brain, provider)
|
|
await brain.init()
|
|
await brain.add({ data: 'a', type: NounType.Thing, vector: V(1) })
|
|
await brain.add({ data: 'b', type: NounType.Thing, vector: V(2) })
|
|
|
|
const report = await brain.warm()
|
|
|
|
expect(provider.warmCalls).toBe(0) // no warm() on this provider
|
|
expect(provider.searchCalls.length).toBe(1) // the probe ran
|
|
expect(provider.searchCalls[0].k).toBe(Math.min(100, provider.size())) // k = min(100, size)
|
|
expect(report.vector.outcome).toBe('probed')
|
|
expect(report.vector.outcome).not.toBe('warmed') // never conflated
|
|
await brain.close()
|
|
})
|
|
|
|
it('vector: reports "unavailable" when there is nothing to probe (empty index, unknown dimension)', async () => {
|
|
const brain = new Brainy({
|
|
requireSubtype: false,
|
|
storage: { type: 'memory' },
|
|
silent: true
|
|
})
|
|
// A disk-native provider MAY honestly report 0 resident entries while
|
|
// durable data exists on disk (the same posture documented on
|
|
// `VectorIndexProvider.isReady` — "an mmap/disk-native index may
|
|
// legitimately report 0 resident entries"). warm()'s probe fallback
|
|
// reads `size()`, so this is the real trigger for "nothing to probe":
|
|
// never a k=0 search, an honest skip.
|
|
const provider = new FakeVectorProvider(false)
|
|
provider.size = () => 0
|
|
useFakeVectorProvider(brain, provider)
|
|
await brain.init() // the VFS root bootstrap write sets `dimensions`, but size() still reports 0
|
|
|
|
const report = await brain.warm()
|
|
|
|
expect(provider.searchCalls.length).toBe(0) // nothing probed
|
|
expect(report.vector.outcome).toBe('unavailable')
|
|
await brain.close()
|
|
})
|
|
|
|
it('(c) metadata + graph hydration paths actually execute against filesystem storage with pre-existing data', async () => {
|
|
const dir = mkTmp()
|
|
|
|
// Build a brain with real data (including a field NOT in the metadata
|
|
// index's own common-fields warm subset — 'wave' — so hydrateAll()'s
|
|
// FULL hydration is distinguishable from init()'s partial warmCache()),
|
|
// and real graph edges, then close it (persisting everything).
|
|
const seed = new Brainy({
|
|
requireSubtype: false,
|
|
storage: { type: 'filesystem', path: dir },
|
|
silent: true
|
|
})
|
|
await seed.init()
|
|
const ids: string[] = []
|
|
for (let i = 0; i < 6; i++) {
|
|
ids.push(
|
|
await seed.add({
|
|
data: `entity ${i}`,
|
|
type: NounType.Thing,
|
|
metadata: { wave: i % 3 },
|
|
vector: V(i + 1)
|
|
})
|
|
)
|
|
}
|
|
for (let i = 0; i + 1 < ids.length; i++) {
|
|
await seed.relate({ from: ids[i], to: ids[i + 1], type: VerbType.RelatedTo })
|
|
}
|
|
await seed.close()
|
|
|
|
// Cold-reopen a FRESH instance and spy on the real hydration seams before
|
|
// calling warm(), so we assert they actually ran (not just that the
|
|
// report claims they did).
|
|
const metaHydrateCalls: number[] = []
|
|
const loadedFields: string[] = []
|
|
const graphInitCalls: number[] = []
|
|
|
|
const origHydrateAll = MetadataIndexManager.prototype.hydrateAll
|
|
const origLoadSparseIndex = (MetadataIndexManager.prototype as any).loadSparseIndex
|
|
const origGraphInit = GraphAdjacencyIndex.prototype.init
|
|
|
|
MetadataIndexManager.prototype.hydrateAll = async function (...args: any[]) {
|
|
metaHydrateCalls.push(1)
|
|
return origHydrateAll.apply(this, args as any)
|
|
}
|
|
;(MetadataIndexManager.prototype as any).loadSparseIndex = async function (
|
|
field: string,
|
|
...args: any[]
|
|
) {
|
|
loadedFields.push(field)
|
|
return origLoadSparseIndex.apply(this, [field, ...args] as any)
|
|
}
|
|
GraphAdjacencyIndex.prototype.init = async function (...args: any[]) {
|
|
graphInitCalls.push(1)
|
|
return origGraphInit.apply(this, args as any)
|
|
}
|
|
|
|
try {
|
|
const brain = new Brainy({
|
|
requireSubtype: false,
|
|
storage: { type: 'filesystem', path: dir },
|
|
silent: true
|
|
})
|
|
await brain.init()
|
|
|
|
const report = await brain.warm()
|
|
|
|
expect(metaHydrateCalls.length).toBe(1) // hydrateAll() actually ran
|
|
expect(loadedFields).toContain('wave') // a NON-common field was loaded — full hydration, not the heuristic subset
|
|
expect(report.metadata.outcome).toBe('warmed')
|
|
|
|
expect(graphInitCalls.length).toBeGreaterThanOrEqual(1) // graph's full-load seam ran (once at brain init, once via warm())
|
|
expect(report.graph.outcome).toBe('warmed')
|
|
|
|
// Correctness survives: the hydrated data still answers queries.
|
|
const byWhere = await brain.find({ type: NounType.Thing, where: { wave: 1 } })
|
|
expect(byWhere.length).toBe(2) // waves 1,4 of 0..5
|
|
|
|
await brain.close()
|
|
} finally {
|
|
MetadataIndexManager.prototype.hydrateAll = origHydrateAll
|
|
;(MetadataIndexManager.prototype as any).loadSparseIndex = origLoadSparseIndex
|
|
GraphAdjacencyIndex.prototype.init = origGraphInit
|
|
}
|
|
})
|
|
|
|
it('(d) warmOnOpen: true runs warm() during init() — observable via the fake provider', async () => {
|
|
const brain = new Brainy({
|
|
requireSubtype: false,
|
|
storage: { type: 'memory' },
|
|
warmOnOpen: true,
|
|
silent: true
|
|
})
|
|
const provider = new FakeVectorProvider(true)
|
|
useFakeVectorProvider(brain, provider)
|
|
|
|
// No explicit brain.warm() call — warmOnOpen must have run it as part of init().
|
|
await brain.init()
|
|
|
|
expect(provider.warmCalls).toBe(1)
|
|
await brain.close()
|
|
})
|
|
|
|
it('warmOnOpen defaults to false — init() does NOT run warm() unless opted in', async () => {
|
|
const brain = new Brainy({
|
|
requireSubtype: false,
|
|
storage: { type: 'memory' },
|
|
silent: true
|
|
})
|
|
const provider = new FakeVectorProvider(true)
|
|
useFakeVectorProvider(brain, provider)
|
|
|
|
await brain.init()
|
|
|
|
expect(provider.warmCalls).toBe(0)
|
|
await brain.close()
|
|
})
|
|
|
|
it('(e) WarmReport shape: outcome literal + durationMs per surface, plus totalDurationMs', async () => {
|
|
const brain = new Brainy({
|
|
requireSubtype: false,
|
|
storage: { type: 'memory' },
|
|
silent: true
|
|
})
|
|
const provider = new FakeVectorProvider(true)
|
|
useFakeVectorProvider(brain, provider)
|
|
await brain.init()
|
|
await brain.add({ data: 'a', type: NounType.Thing, vector: V(1) })
|
|
|
|
const report = await brain.warm()
|
|
|
|
for (const surface of ['vector', 'metadata', 'graph'] as const) {
|
|
expect(['warmed', 'probed', 'unavailable']).toContain(report[surface].outcome)
|
|
expect(typeof report[surface].durationMs).toBe('number')
|
|
expect(report[surface].durationMs).toBeGreaterThanOrEqual(0)
|
|
}
|
|
expect(typeof report.totalDurationMs).toBe('number')
|
|
expect(report.totalDurationMs).toBeGreaterThanOrEqual(0)
|
|
await brain.close()
|
|
})
|
|
})
|