open-brainy/tests/lifecycle/biography.test.ts
David Snelling 9730835bdf
All checks were successful
CI / Node 24 (push) Successful in 12m22s
CI / Node 22 (push) Successful in 12m32s
CI / Integration + conformance (Node 22) (push) Successful in 19m42s
CI / Bun (latest) (push) Successful in 12m15s
feat(vector): the vectored-noun scalar joins the count ledger; the open gate closes the vector leg
The coverage denominator the health-by-accounting ratification named for
the vector family — never built until now, and its absence was measured as
the exact outage class it existed to prevent: a migrated store with
canonical vectors and no derived index opened with the vector leg EMPTY,
served [] from vector search with no error, and the report-driven read gate
had nothing to refuse on (the provider's coverage invariant was honestly
unledgered — the denominator was ours to supply).

- getCanonicalCounts() gains vectors: { all } — the count of canonical
  nouns holding a REAL vector. Incremented where a vector lands (the
  isNew-gated metadata seam for explicit vectors — the same discipline that
  keeps HNSW neighbor-link re-saves from inflating counts; a narrow
  noteVectorLanded hook for the deferred-embed landing, gated on the
  worker's own pre-embed read). Decremented on a proven delete of a
  vectored noun; a vector-uncertain delete marks the ledger suspect rather
  than guessing (no new reads on the delete path). Recounted by the
  sanctioned recount; legacy counts.json derives it once (a deferred noun's
  vector file exists with an empty vector, so presence requires one
  content read at derivation — never on the hot path).
- The open gate's vector leg: when a health-reporting provider claims
  serving while the index holds zero nodes and the ledger proves vectored
  canonical rows exist, open BUILDS (narrated) — routed through the
  provider's idempotent fillFromCanonical() when exposed (the joint door;
  a partial shortfall stays repair()'s operator business), the JS rebuild
  otherwise — or fails typed pre-serve. Scoped exactly: bare isReady()
  providers, migrating providers, and white-box size stubs open as before.

Pinned end-to-end from the partner gate's probe shape (store with vectored
canonical rows, no derived index, reopen → search serves N, never []),
red-proved against the pre-fix path; the inverse (zero vectored rows) opens
without building and serves [] honestly.
2026-08-25 15:31:19 -07:00

410 lines
17 KiB
TypeScript

/**
* @module tests/lifecycle/biography
* @description THE LIFECYCLE LANE — see `tests/lifecycle/README.md` for what
* this proves and how to run it. One scenario, "the working store": a single
* brain driven through founding, a working day, a clean restart, a crash, a
* repair, and a second life, verified chapter by chapter against an
* independent shadow-model referee (`biographyHarness.ts`).
*
* Split into two `it` blocks so a currently-failing later chapter (see the
* second block's header comment — a live engine finding, not a defect in
* this lane) never hides the earlier chapters' passing coverage. The two
* blocks share one brain's directory and one shadow model, run in the SAME
* fixed order the single scenario always has (`describe.sequential` below
* exists to say so explicitly, though vitest's own default is sequential
* within a file) — this is a split for REPORTING clarity, not a reordering
* or conditional skip of any chapter.
*/
import { describe, it, expect } from 'vitest'
import * as fs from 'node:fs'
import { NounType, VerbType } from '../../src/types/graphTypes.js'
import type { Brainy } from '../../src/brainy.js'
import type { AddParams, RelateParams, UpdateParams, UpdateRelationParams } from '../../src/index.js'
import { abandonAsCrashed, makeTempDir, openBrain, uid } from '../helpers/durabilityKillMatrix.js'
import {
createModel,
getCanonicalCountsFor,
modelAdd,
modelDelete,
modelRelate,
modelUpdate,
modelUpdateRelation,
recordVfsFileWrite,
snapshotVfsBaseline,
verifyChapter,
type HubCheck,
type ShadowModel
} from './biographyHarness.js'
const STATUSES = ['active', 'pending', 'closed', 'archived'] as const
/** Cycle a status value to the next one in the fixed rotation — used so
* Ch2's 40 updates provably MOVE entities across find() buckets rather than
* risking a no-op reassignment of the same value. */
function nextStatus(current: unknown): (typeof STATUSES)[number] {
const currentStr = typeof current === 'string' ? current : STATUSES[0]
const idx = STATUSES.indexOf(currentStr as (typeof STATUSES)[number])
return STATUSES[(idx < 0 ? 0 : idx + 1) % STATUSES.length]
}
// ---------------------------------------------------------------------------
// Shared biography state — set up by the first `it`, consumed by the second.
// The two blocks are one continuous story told in two named pieces; nothing
// here resets or diverges between them.
// ---------------------------------------------------------------------------
let dir: string
let model: ShadowModel
let brain: Brainy
let hubs: HubCheck[]
let employees: string[]
let customers: string[]
let invoices: string[]
let tasks: string[]
let projects: string[]
let nonHub: string[]
// ---- Wrappers: every call to the real brain updates the shadow model in
// the same statement, so the two can never drift apart by construction.
// Defined once, closing over the `let` bindings above so both `it` blocks
// (and any future reopen inside them) operate on the current brain/model.
async function doAdd(label: string, params: Omit<AddParams, 'id'>): Promise<string> {
const id = uid(label)
await brain.add({ ...params, id })
modelAdd(model, id, {
type: params.type,
subtype: params.subtype,
metadata: params.metadata ?? {},
visibility: params.visibility
})
return id
}
async function doUpdate(id: string, patch: Omit<UpdateParams, 'id'>): Promise<void> {
await brain.update({ ...patch, id })
modelUpdate(model, id, { metadata: patch.metadata, merge: patch.merge, visibility: patch.visibility })
}
async function doRemove(id: string): Promise<void> {
await brain.remove(id)
modelDelete(model, id)
}
async function doRelate(params: RelateParams): Promise<string> {
const id = await brain.relate(params)
modelRelate(model, id, {
from: params.from,
to: params.to,
type: params.type,
subtype: params.subtype,
metadata: params.metadata
})
return id
}
async function doUpdateRelation(id: string, patch: Omit<UpdateRelationParams, 'id'>): Promise<void> {
await brain.updateRelation({ ...patch, id })
modelUpdateRelation(model, id, { metadata: patch.metadata, merge: patch.merge })
}
async function doVfsWrite(path: string, content: string): Promise<void> {
await brain.vfs.writeFile(path, content)
recordVfsFileWrite(model)
}
describe.sequential('lifecycle — the working store', () => {
it(
'Ch1 FOUNDING -> Ch2 A WORKING DAY -> Ch3 CLEAN RESTART: every read serves truth',
async () => {
process.env.BRAINY_DETERMINISTIC_EMBEDDINGS = 'true'
dir = makeTempDir()
model = createModel()
// logAuthority: 'adopt' from the first open, mirrored across every
// reopen — see write-flow-production-shape.test.ts, which the later
// crash chapter's at-ack law is pinned against.
brain = await openBrain(dir, { logAuthority: 'adopt' })
// =================================================================
// CHAPTER 1 — FOUNDING
// =================================================================
// Baseline MUST be snapshotted before any biography act — it is the
// VFS root's own system-tier footprint, measured, never hardcoded.
await snapshotVfsBaseline(brain, model)
employees = []
for (let i = 0; i < 20; i++) {
employees.push(
await doAdd(`emp-${i}`, {
data: `employee record ${i}`,
type: NounType.Person,
subtype: 'employee',
metadata: { status: STATUSES[i % STATUSES.length], department: ['engineering', 'sales', 'support'][i % 3] }
})
)
}
customers = []
for (let i = 0; i < 20; i++) {
customers.push(
await doAdd(`cust-${i}`, {
data: `customer record ${i}`,
type: NounType.Person,
subtype: 'customer',
metadata: { status: STATUSES[i % STATUSES.length], tier: i % 2 === 0 ? 'gold' : 'standard' }
})
)
}
invoices = []
for (let i = 0; i < 30; i++) {
invoices.push(
await doAdd(`inv-${i}`, {
data: `invoice record ${i}`,
type: NounType.Document,
subtype: 'invoice',
metadata: { status: STATUSES[i % STATUSES.length], amount: 100 + i * 17 }
})
)
}
tasks = []
for (let i = 0; i < 25; i++) {
tasks.push(
await doAdd(`task-${i}`, {
data: `task record ${i}`,
type: NounType.Task,
subtype: 'milestone',
metadata: { status: STATUSES[i % STATUSES.length], priority: (i % 5) + 1 }
})
)
}
projects = []
for (let i = 0; i < 25; i++) {
projects.push(
await doAdd(`proj-${i}`, {
data: `project record ${i}`,
type: NounType.Project,
metadata: { status: STATUSES[i % STATUSES.length], budget: 1000 * (i + 1) }
})
)
}
expect(employees.length + customers.length + invoices.length + tasks.length + projects.length).toBe(120)
// Five hubs (proj-0..proj-4) fan out to tasks (Contains) and employees
// (WorksWith); a residual band of invoice->customer RelatedTo edges is
// unrelated to any hub. Hubs are never touched again for the rest of
// the biography, so they stay valid adjacency samples in every chapter.
const hubIds = projects.slice(0, 5)
for (let h = 0; h < 5; h++) {
for (let k = 0; k < 15; k++) {
const taskIdx = (h * 5 + k) % tasks.length
await doRelate({ from: hubIds[h], to: tasks[taskIdx], type: VerbType.Contains, subtype: 'delivers' })
}
for (let k = 0; k < 10; k++) {
const empIdx = (h * 4 + k) % employees.length
await doRelate({ from: hubIds[h], to: employees[empIdx], type: VerbType.WorksWith })
}
}
for (let j = 0; j < 25; j++) {
await doRelate({ from: invoices[j], to: customers[j % customers.length], type: VerbType.RelatedTo, subtype: 'billed-to' })
}
expect(model.relations.size).toBe(150)
// A handful of VFS files.
for (let i = 0; i < 5; i++) {
await doVfsWrite(`/report-${i}.txt`, `founding report ${i}`)
}
await brain.flush()
hubs = hubIds.map((id) => ({ id, typeFilters: [VerbType.Contains, VerbType.WorksWith] }))
await verifyChapter(brain, model, 'Ch1 FOUNDING', { hubs, bucketField: 'status' })
// =================================================================
// CHAPTER 2 — A WORKING DAY
// =================================================================
// Non-hub pool for every mutation below.
nonHub = [...employees, ...customers, ...invoices, ...tasks, ...projects.slice(5)]
// 40 updates that provably MOVE entities across find() status buckets.
const updateTargets = nonHub.slice(0, 40)
for (const id of updateTargets) {
const current = model.entities.get(id)!.metadata.status
await doUpdate(id, { metadata: { status: nextStatus(current) } })
}
// 10 visibility flips (public -> internal).
const visibilityTargets = nonHub.slice(40, 50)
for (const id of visibilityTargets) {
await doUpdate(id, { visibility: 'internal' })
}
// 15 deletes — some hub members (their edges cascade away), 3 of them
// earmarked for Ch6's resurrection.
const resurrectIds = [tasks[0], tasks[1], employees[0]]
const otherDeletes = [
tasks[2], tasks[3], tasks[4], tasks[5], tasks[6],
employees[1], employees[2], employees[3],
customers[0], customers[1], customers[2], customers[3]
]
const ch2DeleteTargets = [...resurrectIds, ...otherDeletes]
expect(ch2DeleteTargets.length).toBe(15)
for (const id of ch2DeleteTargets) {
await doRemove(id)
}
// 20 new adds.
const ch2NewTypes = [NounType.Person, NounType.Document, NounType.Task]
for (let i = 0; i < 20; i++) {
await doAdd(`ch2-new-${i}`, {
data: `working-day addition ${i}`,
type: ch2NewTypes[i % ch2NewTypes.length],
subtype: 'ad-hoc',
metadata: { status: STATUSES[i % STATUSES.length] }
})
}
// 10 updateRelation metadata patches — read AFTER the deletes above,
// so only relations the cascade left alive are ever targeted.
const survivingRelationIds = [...model.relations.keys()].slice(0, 10)
expect(survivingRelationIds.length).toBe(10)
for (const relId of survivingRelationIds) {
await doUpdateRelation(relId, { metadata: { reviewed: true } })
}
await brain.flush()
await verifyChapter(brain, model, 'Ch2 A WORKING DAY', { hubs, bucketField: 'status' })
// =================================================================
// CHAPTER 3 — CLEAN RESTART
// =================================================================
await brain.close()
brain = await openBrain(dir, { logAuthority: 'adopt' })
await verifyChapter(brain, model, 'Ch3 CLEAN RESTART', { hubs, bucketField: 'status' })
// Leave the brain closed and the directory intact for the next `it`
// (the biography continues there) — do NOT remove `dir` here.
await brain.close()
},
300000
)
it(
'Ch4 CRASH -> Ch5 REPAIR -> Ch6 SECOND LIFE: continues the Ch3 store',
async () => {
try {
brain = await openBrain(dir, { logAuthority: 'adopt' })
// ===============================================================
// CHAPTER 4 — CRASH
// ===============================================================
const ch4Types = [NounType.Person, NounType.Document, NounType.Task, NounType.Project]
for (let i = 0; i < 10; i++) {
await doAdd(`ch4-new-${i}`, {
data: `crash-window addition ${i}`,
type: ch4Types[i % ch4Types.length],
metadata: { status: STATUSES[i % STATUSES.length] }
})
}
const ch4UpdateTargets = nonHub.slice(50, 55) // invoices[10..14] — untouched so far
for (const id of ch4UpdateTargets) {
await doUpdate(id, { metadata: { status: 'active' } })
}
// NO flush — abandon exactly the way process death would (the
// at-ack law: every write already awaited above must survive).
await abandonAsCrashed(brain)
brain = await openBrain(dir, { logAuthority: 'adopt' })
await verifyChapter(brain, model, 'Ch4 CRASH', { hubs, bucketField: 'status' })
// ===============================================================
// CHAPTER 5 — REPAIR
// ===============================================================
const report = await brain.repairIndex()
for (const family of report.families) {
const accounted =
family.checked === true || (family.checked === false && typeof family.skipped === 'string' && family.skipped.length > 0)
expect(
accounted,
`[Ch5 REPAIR] family '${family.family}' must be checked or explicitly skipped with a reason; got ${JSON.stringify(family)}`
).toBe(true)
}
// A healthy store: repair must change nothing the model doesn't
// already expect — verifyChapter against the UNCHANGED model proves it.
await verifyChapter(brain, model, 'Ch5 REPAIR', { hubs, bucketField: 'status' })
// ===============================================================
// CHAPTER 6 — SECOND LIFE
// ===============================================================
const ch6Types = [NounType.Person, NounType.Document, NounType.Task, NounType.Project]
for (let i = 0; i < 10; i++) {
await doAdd(`ch6-new-${i}`, {
data: `second-life addition ${i}`,
type: ch6Types[i % ch6Types.length],
metadata: { status: STATUSES[i % STATUSES.length] }
})
}
const ch6UpdateTargets = nonHub.slice(55, 65) // invoices[15..24] — untouched so far
expect(ch6UpdateTargets.every((id) => model.entities.get(id)!.alive)).toBe(true)
for (const id of ch6UpdateTargets) {
await doUpdate(id, { metadata: { status: 'closed' } })
}
const ch6DeleteTargets = nonHub
.slice(65, 90) // invoices[25..29] + tasks[0..19] (some already dead — filtered below)
.filter((id) => model.entities.get(id)!.alive)
.slice(0, 7)
expect(ch6DeleteTargets.length).toBe(7)
for (const id of ch6DeleteTargets) {
await doRemove(id)
}
// Resurrection: the SAME three ids Ch2 deleted, reinserted with
// BRAND-NEW metadata — the model expects the new metadata only.
await doAdd('task-0', { data: 'resurrected task 0', type: NounType.Task, subtype: 'milestone', metadata: { status: 'active', resurrected: true } })
await doAdd('task-1', { data: 'resurrected task 1', type: NounType.Task, subtype: 'milestone', metadata: { status: 'pending', resurrected: true } })
await doAdd('emp-0', { data: 'resurrected employee 0', type: NounType.Person, subtype: 'employee', metadata: { status: 'active', resurrected: true } })
expect(tasks[0]).toBe(uid('task-0')) // same id as Ch1/Ch2 — the resurrection-adjacent shape
await brain.close()
brain = await openBrain(dir, { logAuthority: 'adopt' })
await verifyChapter(brain, model, 'Ch6 SECOND LIFE', { hubs, bucketField: 'status' })
// Final, standalone getCanonicalCounts() exactness check (beyond
// verifyChapter's own (f) leg) — the whole ledger, in one shot.
const finalCounts = await getCanonicalCountsFor(brain)
const aliveEntities = [...model.entities.values()].filter((e) => e.alive)
const alivePublicEntities = aliveEntities.filter((e) => (e.visibility ?? 'public') === 'public')
const aliveVerbs = model.relations.size
expect(finalCounts, 'final getCanonicalCounts() exactness — Ch6 SECOND LIFE').toEqual({
nouns: {
counted: alivePublicEntities.length + model.vfsFileNouns,
all: aliveEntities.length + model.vfsFileNouns + model.vfsBaselineNouns
},
verbs: {
counted: aliveVerbs + model.vfsContainsVerbs,
all: aliveVerbs + model.vfsContainsVerbs + model.vfsBaselineVerbs
},
// Every noun this biography ever adds carries an explicit/computed
// vector (the harness never defers an embed), so the vectored-noun
// scalar tracks nouns.all exactly.
vectors: {
all: aliveEntities.length + model.vfsFileNouns + model.vfsBaselineNouns
},
suspect: false
})
} finally {
await brain.close().catch(() => {})
// Best-effort, retried: a still-draining background persistence
// write (e.g. count/index write-through) can race a single rmSync
// and leave a partial directory behind — retry a couple of times
// rather than let this temp dir leak.
for (let attempt = 0; attempt < 3; attempt++) {
try {
fs.rmSync(dir, { recursive: true, force: true })
if (!fs.existsSync(dir)) break
} catch {
// ignore and retry
}
await new Promise((resolve) => setTimeout(resolve, 100))
}
}
},
300000
)
})