A canonical row persisted with vector: [] (a system row, a deferred embed not yet landed, or any other legitimately-unvectored record) is a normal, enumerable row -- but rebuild()'s storage walk had no guard against it. storage.getVectorIndexData() derives its answer from the row's own record, so it returns non-null for any existing noun whether or not that noun was ever actually indexed -- rebuild() admitted such rows into the live graph with a length-0 vector. A vector-less node could become the entry point (or occupy any graph position); the next real insert then ran a distance calculation against it and blew up with a dimension mismatch. Fix at two layers in src/hnsw/hnswIndex.ts: - rebuild() now skips any row whose vector.length === 0 before it ever becomes a graph node (one summary count line, never per-row spam), and restores the pinned dimension from the first real vector it loads -- previously the pin stayed null across a restart, since addItem/updateItem are the only sites that set it and rebuild() never goes through either. - addItem/updateItem now refuse a length-0 vector with a typed EmptyVectorIndexError instead of ever pinning dimension to 0 or storing a vector-less node, so no future fill/rebuild/load path can poison the index silently. getVectorSafe's lazy-load "not found" check also missed that an empty array is truthy -- tightened to catch it. IndexOperations.ts's ReplaceInVectorIndexOperation rollback paths now skip re-adding an oldVector of length 0 (never a legal index member) instead of attempting an illegal empty re-insert on rollback. biography.test.ts's final ledger-exactness assertion assumed every noun the lane creates is vectored, including the VFS root counted in vfsBaselineNouns -- but the root is deliberately persisted unvectored. Corrected the expected formula to exclude it. Adds tests/integration/index-skips-unvectored.test.ts pinning: rebuild() indexes only vectored rows with the dimension pinned correctly; clear() then real adds never trip a dimension mismatch; addItem/updateItem refuse a length-0 vector; and a crash/repair cycle stays dimension-consistent.
418 lines
18 KiB
TypeScript
418 lines
18 KiB
TypeScript
/**
|
|
* @module tests/lifecycle/biography
|
|
* @description THE LIFECYCLE LANE — see `tests/lifecycle/README.md` for what
|
|
* this proves and how to run it. One scenario, "the working store": a single
|
|
* brain driven through founding, a working day, a clean restart, a crash, a
|
|
* repair, and a second life, verified chapter by chapter against an
|
|
* independent shadow-model referee (`biographyHarness.ts`).
|
|
*
|
|
* Split into two `it` blocks so a currently-failing later chapter (see the
|
|
* second block's header comment — a live engine finding, not a defect in
|
|
* this lane) never hides the earlier chapters' passing coverage. The two
|
|
* blocks share one brain's directory and one shadow model, run in the SAME
|
|
* fixed order the single scenario always has (`describe.sequential` below
|
|
* exists to say so explicitly, though vitest's own default is sequential
|
|
* within a file) — this is a split for REPORTING clarity, not a reordering
|
|
* or conditional skip of any chapter.
|
|
*/
|
|
import { describe, it, expect } from 'vitest'
|
|
import * as fs from 'node:fs'
|
|
import { NounType, VerbType } from '../../src/types/graphTypes.js'
|
|
import type { Brainy } from '../../src/brainy.js'
|
|
import type { AddParams, RelateParams, UpdateParams, UpdateRelationParams } from '../../src/index.js'
|
|
import { abandonAsCrashed, makeTempDir, openBrain, uid } from '../helpers/durabilityKillMatrix.js'
|
|
import {
|
|
createModel,
|
|
getCanonicalCountsFor,
|
|
modelAdd,
|
|
modelDelete,
|
|
modelRelate,
|
|
modelUpdate,
|
|
modelUpdateRelation,
|
|
recordVfsFileWrite,
|
|
snapshotVfsBaseline,
|
|
verifyChapter,
|
|
type HubCheck,
|
|
type ShadowModel
|
|
} from './biographyHarness.js'
|
|
|
|
const STATUSES = ['active', 'pending', 'closed', 'archived'] as const
|
|
|
|
/** Cycle a status value to the next one in the fixed rotation — used so
|
|
* Ch2's 40 updates provably MOVE entities across find() buckets rather than
|
|
* risking a no-op reassignment of the same value. */
|
|
function nextStatus(current: unknown): (typeof STATUSES)[number] {
|
|
const currentStr = typeof current === 'string' ? current : STATUSES[0]
|
|
const idx = STATUSES.indexOf(currentStr as (typeof STATUSES)[number])
|
|
return STATUSES[(idx < 0 ? 0 : idx + 1) % STATUSES.length]
|
|
}
|
|
|
|
// ---------------------------------------------------------------------------
|
|
// Shared biography state — set up by the first `it`, consumed by the second.
|
|
// The two blocks are one continuous story told in two named pieces; nothing
|
|
// here resets or diverges between them.
|
|
// ---------------------------------------------------------------------------
|
|
let dir: string
|
|
let model: ShadowModel
|
|
let brain: Brainy
|
|
let hubs: HubCheck[]
|
|
let employees: string[]
|
|
let customers: string[]
|
|
let invoices: string[]
|
|
let tasks: string[]
|
|
let projects: string[]
|
|
let nonHub: string[]
|
|
|
|
// ---- Wrappers: every call to the real brain updates the shadow model in
|
|
// the same statement, so the two can never drift apart by construction.
|
|
// Defined once, closing over the `let` bindings above so both `it` blocks
|
|
// (and any future reopen inside them) operate on the current brain/model.
|
|
async function doAdd(label: string, params: Omit<AddParams, 'id'>): Promise<string> {
|
|
const id = uid(label)
|
|
await brain.add({ ...params, id })
|
|
modelAdd(model, id, {
|
|
type: params.type,
|
|
subtype: params.subtype,
|
|
metadata: params.metadata ?? {},
|
|
visibility: params.visibility
|
|
})
|
|
return id
|
|
}
|
|
|
|
async function doUpdate(id: string, patch: Omit<UpdateParams, 'id'>): Promise<void> {
|
|
await brain.update({ ...patch, id })
|
|
modelUpdate(model, id, { metadata: patch.metadata, merge: patch.merge, visibility: patch.visibility })
|
|
}
|
|
|
|
async function doRemove(id: string): Promise<void> {
|
|
await brain.remove(id)
|
|
modelDelete(model, id)
|
|
}
|
|
|
|
async function doRelate(params: RelateParams): Promise<string> {
|
|
const id = await brain.relate(params)
|
|
modelRelate(model, id, {
|
|
from: params.from,
|
|
to: params.to,
|
|
type: params.type,
|
|
subtype: params.subtype,
|
|
metadata: params.metadata
|
|
})
|
|
return id
|
|
}
|
|
|
|
async function doUpdateRelation(id: string, patch: Omit<UpdateRelationParams, 'id'>): Promise<void> {
|
|
await brain.updateRelation({ ...patch, id })
|
|
modelUpdateRelation(model, id, { metadata: patch.metadata, merge: patch.merge })
|
|
}
|
|
|
|
async function doVfsWrite(path: string, content: string): Promise<void> {
|
|
await brain.vfs.writeFile(path, content)
|
|
recordVfsFileWrite(model)
|
|
}
|
|
|
|
describe.sequential('lifecycle — the working store', () => {
|
|
it(
|
|
'Ch1 FOUNDING -> Ch2 A WORKING DAY -> Ch3 CLEAN RESTART: every read serves truth',
|
|
async () => {
|
|
process.env.BRAINY_DETERMINISTIC_EMBEDDINGS = 'true'
|
|
dir = makeTempDir()
|
|
model = createModel()
|
|
|
|
// logAuthority: 'adopt' from the first open, mirrored across every
|
|
// reopen — see write-flow-production-shape.test.ts, which the later
|
|
// crash chapter's at-ack law is pinned against.
|
|
brain = await openBrain(dir, { logAuthority: 'adopt' })
|
|
|
|
// =================================================================
|
|
// CHAPTER 1 — FOUNDING
|
|
// =================================================================
|
|
// Baseline MUST be snapshotted before any biography act — it is the
|
|
// VFS root's own system-tier footprint, measured, never hardcoded.
|
|
await snapshotVfsBaseline(brain, model)
|
|
|
|
employees = []
|
|
for (let i = 0; i < 20; i++) {
|
|
employees.push(
|
|
await doAdd(`emp-${i}`, {
|
|
data: `employee record ${i}`,
|
|
type: NounType.Person,
|
|
subtype: 'employee',
|
|
metadata: { status: STATUSES[i % STATUSES.length], department: ['engineering', 'sales', 'support'][i % 3] }
|
|
})
|
|
)
|
|
}
|
|
customers = []
|
|
for (let i = 0; i < 20; i++) {
|
|
customers.push(
|
|
await doAdd(`cust-${i}`, {
|
|
data: `customer record ${i}`,
|
|
type: NounType.Person,
|
|
subtype: 'customer',
|
|
metadata: { status: STATUSES[i % STATUSES.length], tier: i % 2 === 0 ? 'gold' : 'standard' }
|
|
})
|
|
)
|
|
}
|
|
invoices = []
|
|
for (let i = 0; i < 30; i++) {
|
|
invoices.push(
|
|
await doAdd(`inv-${i}`, {
|
|
data: `invoice record ${i}`,
|
|
type: NounType.Document,
|
|
subtype: 'invoice',
|
|
metadata: { status: STATUSES[i % STATUSES.length], amount: 100 + i * 17 }
|
|
})
|
|
)
|
|
}
|
|
tasks = []
|
|
for (let i = 0; i < 25; i++) {
|
|
tasks.push(
|
|
await doAdd(`task-${i}`, {
|
|
data: `task record ${i}`,
|
|
type: NounType.Task,
|
|
subtype: 'milestone',
|
|
metadata: { status: STATUSES[i % STATUSES.length], priority: (i % 5) + 1 }
|
|
})
|
|
)
|
|
}
|
|
projects = []
|
|
for (let i = 0; i < 25; i++) {
|
|
projects.push(
|
|
await doAdd(`proj-${i}`, {
|
|
data: `project record ${i}`,
|
|
type: NounType.Project,
|
|
metadata: { status: STATUSES[i % STATUSES.length], budget: 1000 * (i + 1) }
|
|
})
|
|
)
|
|
}
|
|
expect(employees.length + customers.length + invoices.length + tasks.length + projects.length).toBe(120)
|
|
|
|
// Five hubs (proj-0..proj-4) fan out to tasks (Contains) and employees
|
|
// (WorksWith); a residual band of invoice->customer RelatedTo edges is
|
|
// unrelated to any hub. Hubs are never touched again for the rest of
|
|
// the biography, so they stay valid adjacency samples in every chapter.
|
|
const hubIds = projects.slice(0, 5)
|
|
for (let h = 0; h < 5; h++) {
|
|
for (let k = 0; k < 15; k++) {
|
|
const taskIdx = (h * 5 + k) % tasks.length
|
|
await doRelate({ from: hubIds[h], to: tasks[taskIdx], type: VerbType.Contains, subtype: 'delivers' })
|
|
}
|
|
for (let k = 0; k < 10; k++) {
|
|
const empIdx = (h * 4 + k) % employees.length
|
|
await doRelate({ from: hubIds[h], to: employees[empIdx], type: VerbType.WorksWith })
|
|
}
|
|
}
|
|
for (let j = 0; j < 25; j++) {
|
|
await doRelate({ from: invoices[j], to: customers[j % customers.length], type: VerbType.RelatedTo, subtype: 'billed-to' })
|
|
}
|
|
expect(model.relations.size).toBe(150)
|
|
|
|
// A handful of VFS files.
|
|
for (let i = 0; i < 5; i++) {
|
|
await doVfsWrite(`/report-${i}.txt`, `founding report ${i}`)
|
|
}
|
|
|
|
await brain.flush()
|
|
|
|
hubs = hubIds.map((id) => ({ id, typeFilters: [VerbType.Contains, VerbType.WorksWith] }))
|
|
await verifyChapter(brain, model, 'Ch1 FOUNDING', { hubs, bucketField: 'status' })
|
|
|
|
// =================================================================
|
|
// CHAPTER 2 — A WORKING DAY
|
|
// =================================================================
|
|
// Non-hub pool for every mutation below.
|
|
nonHub = [...employees, ...customers, ...invoices, ...tasks, ...projects.slice(5)]
|
|
|
|
// 40 updates that provably MOVE entities across find() status buckets.
|
|
const updateTargets = nonHub.slice(0, 40)
|
|
for (const id of updateTargets) {
|
|
const current = model.entities.get(id)!.metadata.status
|
|
await doUpdate(id, { metadata: { status: nextStatus(current) } })
|
|
}
|
|
|
|
// 10 visibility flips (public -> internal).
|
|
const visibilityTargets = nonHub.slice(40, 50)
|
|
for (const id of visibilityTargets) {
|
|
await doUpdate(id, { visibility: 'internal' })
|
|
}
|
|
|
|
// 15 deletes — some hub members (their edges cascade away), 3 of them
|
|
// earmarked for Ch6's resurrection.
|
|
const resurrectIds = [tasks[0], tasks[1], employees[0]]
|
|
const otherDeletes = [
|
|
tasks[2], tasks[3], tasks[4], tasks[5], tasks[6],
|
|
employees[1], employees[2], employees[3],
|
|
customers[0], customers[1], customers[2], customers[3]
|
|
]
|
|
const ch2DeleteTargets = [...resurrectIds, ...otherDeletes]
|
|
expect(ch2DeleteTargets.length).toBe(15)
|
|
for (const id of ch2DeleteTargets) {
|
|
await doRemove(id)
|
|
}
|
|
|
|
// 20 new adds.
|
|
const ch2NewTypes = [NounType.Person, NounType.Document, NounType.Task]
|
|
for (let i = 0; i < 20; i++) {
|
|
await doAdd(`ch2-new-${i}`, {
|
|
data: `working-day addition ${i}`,
|
|
type: ch2NewTypes[i % ch2NewTypes.length],
|
|
subtype: 'ad-hoc',
|
|
metadata: { status: STATUSES[i % STATUSES.length] }
|
|
})
|
|
}
|
|
|
|
// 10 updateRelation metadata patches — read AFTER the deletes above,
|
|
// so only relations the cascade left alive are ever targeted.
|
|
const survivingRelationIds = [...model.relations.keys()].slice(0, 10)
|
|
expect(survivingRelationIds.length).toBe(10)
|
|
for (const relId of survivingRelationIds) {
|
|
await doUpdateRelation(relId, { metadata: { reviewed: true } })
|
|
}
|
|
|
|
await brain.flush()
|
|
await verifyChapter(brain, model, 'Ch2 A WORKING DAY', { hubs, bucketField: 'status' })
|
|
|
|
// =================================================================
|
|
// CHAPTER 3 — CLEAN RESTART
|
|
// =================================================================
|
|
await brain.close()
|
|
brain = await openBrain(dir, { logAuthority: 'adopt' })
|
|
await verifyChapter(brain, model, 'Ch3 CLEAN RESTART', { hubs, bucketField: 'status' })
|
|
|
|
// Leave the brain closed and the directory intact for the next `it`
|
|
// (the biography continues there) — do NOT remove `dir` here.
|
|
await brain.close()
|
|
},
|
|
300000
|
|
)
|
|
|
|
it(
|
|
'Ch4 CRASH -> Ch5 REPAIR -> Ch6 SECOND LIFE: continues the Ch3 store',
|
|
async () => {
|
|
try {
|
|
brain = await openBrain(dir, { logAuthority: 'adopt' })
|
|
|
|
// ===============================================================
|
|
// CHAPTER 4 — CRASH
|
|
// ===============================================================
|
|
const ch4Types = [NounType.Person, NounType.Document, NounType.Task, NounType.Project]
|
|
for (let i = 0; i < 10; i++) {
|
|
await doAdd(`ch4-new-${i}`, {
|
|
data: `crash-window addition ${i}`,
|
|
type: ch4Types[i % ch4Types.length],
|
|
metadata: { status: STATUSES[i % STATUSES.length] }
|
|
})
|
|
}
|
|
const ch4UpdateTargets = nonHub.slice(50, 55) // invoices[10..14] — untouched so far
|
|
for (const id of ch4UpdateTargets) {
|
|
await doUpdate(id, { metadata: { status: 'active' } })
|
|
}
|
|
// NO flush — abandon exactly the way process death would (the
|
|
// at-ack law: every write already awaited above must survive).
|
|
await abandonAsCrashed(brain)
|
|
brain = await openBrain(dir, { logAuthority: 'adopt' })
|
|
await verifyChapter(brain, model, 'Ch4 CRASH', { hubs, bucketField: 'status' })
|
|
|
|
// ===============================================================
|
|
// CHAPTER 5 — REPAIR
|
|
// ===============================================================
|
|
const report = await brain.repairIndex()
|
|
for (const family of report.families) {
|
|
const accounted =
|
|
family.checked === true || (family.checked === false && typeof family.skipped === 'string' && family.skipped.length > 0)
|
|
expect(
|
|
accounted,
|
|
`[Ch5 REPAIR] family '${family.family}' must be checked or explicitly skipped with a reason; got ${JSON.stringify(family)}`
|
|
).toBe(true)
|
|
}
|
|
// A healthy store: repair must change nothing the model doesn't
|
|
// already expect — verifyChapter against the UNCHANGED model proves it.
|
|
await verifyChapter(brain, model, 'Ch5 REPAIR', { hubs, bucketField: 'status' })
|
|
|
|
// ===============================================================
|
|
// CHAPTER 6 — SECOND LIFE
|
|
// ===============================================================
|
|
const ch6Types = [NounType.Person, NounType.Document, NounType.Task, NounType.Project]
|
|
for (let i = 0; i < 10; i++) {
|
|
await doAdd(`ch6-new-${i}`, {
|
|
data: `second-life addition ${i}`,
|
|
type: ch6Types[i % ch6Types.length],
|
|
metadata: { status: STATUSES[i % STATUSES.length] }
|
|
})
|
|
}
|
|
const ch6UpdateTargets = nonHub.slice(55, 65) // invoices[15..24] — untouched so far
|
|
expect(ch6UpdateTargets.every((id) => model.entities.get(id)!.alive)).toBe(true)
|
|
for (const id of ch6UpdateTargets) {
|
|
await doUpdate(id, { metadata: { status: 'closed' } })
|
|
}
|
|
const ch6DeleteTargets = nonHub
|
|
.slice(65, 90) // invoices[25..29] + tasks[0..19] (some already dead — filtered below)
|
|
.filter((id) => model.entities.get(id)!.alive)
|
|
.slice(0, 7)
|
|
expect(ch6DeleteTargets.length).toBe(7)
|
|
for (const id of ch6DeleteTargets) {
|
|
await doRemove(id)
|
|
}
|
|
|
|
// Resurrection: the SAME three ids Ch2 deleted, reinserted with
|
|
// BRAND-NEW metadata — the model expects the new metadata only.
|
|
await doAdd('task-0', { data: 'resurrected task 0', type: NounType.Task, subtype: 'milestone', metadata: { status: 'active', resurrected: true } })
|
|
await doAdd('task-1', { data: 'resurrected task 1', type: NounType.Task, subtype: 'milestone', metadata: { status: 'pending', resurrected: true } })
|
|
await doAdd('emp-0', { data: 'resurrected employee 0', type: NounType.Person, subtype: 'employee', metadata: { status: 'active', resurrected: true } })
|
|
expect(tasks[0]).toBe(uid('task-0')) // same id as Ch1/Ch2 — the resurrection-adjacent shape
|
|
|
|
await brain.close()
|
|
brain = await openBrain(dir, { logAuthority: 'adopt' })
|
|
await verifyChapter(brain, model, 'Ch6 SECOND LIFE', { hubs, bucketField: 'status' })
|
|
|
|
// Final, standalone getCanonicalCounts() exactness check (beyond
|
|
// verifyChapter's own (f) leg) — the whole ledger, in one shot.
|
|
const finalCounts = await getCanonicalCountsFor(brain)
|
|
const aliveEntities = [...model.entities.values()].filter((e) => e.alive)
|
|
const alivePublicEntities = aliveEntities.filter((e) => (e.visibility ?? 'public') === 'public')
|
|
const aliveVerbs = model.relations.size
|
|
expect(finalCounts, 'final getCanonicalCounts() exactness — Ch6 SECOND LIFE').toEqual({
|
|
nouns: {
|
|
counted: alivePublicEntities.length + model.vfsFileNouns,
|
|
all: aliveEntities.length + model.vfsFileNouns + model.vfsBaselineNouns
|
|
},
|
|
verbs: {
|
|
counted: aliveVerbs + model.vfsContainsVerbs,
|
|
all: aliveVerbs + model.vfsContainsVerbs + model.vfsBaselineVerbs
|
|
},
|
|
// Every noun this biography ever adds carries an explicit/computed
|
|
// vector (the harness never defers an embed), so the vectored-noun
|
|
// scalar tracks nouns.all exactly EXCEPT for the VFS root counted
|
|
// in `vfsBaselineNouns`: the root is deliberately persisted with
|
|
// `vector: []` (the sanctioned "unvectored" shape — see
|
|
// VirtualFileSystem.doInitializeRoot()'s zero-norm-avoidance
|
|
// comment) so it never pays the WASM engine's cold-compile cost and
|
|
// never crosses an engine boundary as a false attractor. It is the
|
|
// ONE hidden-tier record `vfsBaselineNouns` represents (see
|
|
// biographyHarness's module header), so it is excluded here even
|
|
// though it counts toward `nouns.all`.
|
|
vectors: {
|
|
all: aliveEntities.length + model.vfsFileNouns
|
|
},
|
|
suspect: false
|
|
})
|
|
} finally {
|
|
await brain.close().catch(() => {})
|
|
// Best-effort, retried: a still-draining background persistence
|
|
// write (e.g. count/index write-through) can race a single rmSync
|
|
// and leave a partial directory behind — retry a couple of times
|
|
// rather than let this temp dir leak.
|
|
for (let attempt = 0; attempt < 3; attempt++) {
|
|
try {
|
|
fs.rmSync(dir, { recursive: true, force: true })
|
|
if (!fs.existsSync(dir)) break
|
|
} catch {
|
|
// ignore and retry
|
|
}
|
|
await new Promise((resolve) => setTimeout(resolve, 100))
|
|
}
|
|
}
|
|
},
|
|
300000
|
|
)
|
|
})
|