open-brainy/tests/unit/vector-cold-read-guard.test.ts

108 lines
4.9 KiB
TypeScript
Raw Normal View History

2026-07-13 13:08:52 -07:00
/**
* @module tests/unit/vector-cold-read-guard
* @description Pattern-A / Finding 1: a pure semantic find({ query }) has no
* filter, so verifyMetadataLive never fires nothing guarded the vector index.
* A cold native vector index that loaded its COUNT but not its serving structure
feat(health): the gate reads the named report — reads refuse loudly, never rebuild; open serves before it returns; the ceremony door The read gate stops consulting the unnamed isReady() boolean: every provider may expose healthReport() (sync, O(1), composed from exact ledgers — HealthReport with a monotonic generation, per-invariant source ledger|deep|unledgered, missing {count, sample}), and one readiness authority (assessProviderHealth) derives the verdict. Unledgered families are UNKNOWN — never healthy, never broken; a report that throws is a loud not-ready, never a shrug. Reads at the four index choke points refuse with the typed NotReady errors, narrated once per (provider, generation) — a read NEVER starts a store walk: - the first-read lazy build retires (open builds instead, regardless of size — the ≥10k deferral and the "lazy loading on first query" branch go; disableAutoRebuild is re-meant honestly in its docs); - the verify*Live read-path rebuild triggers retire (refuse-or-serve); - the read-time consistency probe that could launch a dark rebuild from an ordinary find() retires; - repairIndex({ rebuild: ['metadata'|'graph'|'vector'] | 'all' }) is the one explicit door: rebuilds the named leg unconditionally and reports rebuilt per family; bare repairIndex() stays report-driven. test(lifecycle): the biography lane — a store's whole life, refereed tests/lifecycle/: an independent shadow model referees every read after every chapter (founding, a working day, clean restart, crash, repair, second life). Chapters 1-3 green. Chapters 4-6 assert the true contract and are marked .fails as a release-blocking finding (the kill-matrix convention): after a crash + adopt reopen the metadata index computes its 'catchup' watermark verdict and nothing consumes it — find() serves the pre-crash index while canonical and counts recover. The catchup wiring is the cure; a passing .fails will force the marker's removal. The lane runs in the integration gate (config + coverage guard).
2026-08-24 12:45:51 -07:00
* returned a silent []. verifyVectorLive() closes that: the health-report/isReady()
* authority first, else a known-vector self-match probe.
*
* RE-POINTED to the health-gate law: the guard NEVER rebuilds and NEVER walks
* the store from a read a read-path rebuild is exactly the dark-rebuild
* failure mode the law retires (open() alone owns building). A not-serving
* signal (from either strategy) THROWS VectorIndexNotReadyError immediately,
* with no rebuild attempt in between never a silent empty result.
2026-07-13 13:08:52 -07:00
*/
import { describe, it, expect, beforeEach } from 'vitest'
import { Brainy, NounType, VectorIndexNotReadyError } from '../../src/index.js'
const V = (): number[] => Array.from({ length: 384 }, (_, i) => Math.sin(i * 0.1) + 0.001)
describe('Vector cold-read guard (verifyVectorLive) — silent-[] on cold semantic find', () => {
let brain: any
beforeEach(async () => {
process.env.BRAINY_DETERMINISTIC_EMBEDDINGS = 'true'
brain = new Brainy({ requireSubtype: false, storage: { type: 'memory' }, dimensions: 384 })
await brain.init()
await brain.add({ vector: V(), type: NounType.Concept, metadata: { status: 'active' } })
await brain.add({ vector: V(), type: NounType.Concept, metadata: { status: 'archived' } })
await brain.flush()
})
it('warm brain: semantic find is correct and the guard does not rebuild', async () => {
const vi = brain.index
let rebuilds = 0
const origRebuild = vi.rebuild.bind(vi)
vi.rebuild = async (...a: any[]) => { rebuilds++; return origRebuild(...a) }
await brain.find({ query: 'anything', searchMode: 'semantic', limit: 100 })
expect(rebuilds).toBe(0)
expect(brain._vectorVerified).toBe(true)
vi.rebuild = origRebuild
})
feat(health): the gate reads the named report — reads refuse loudly, never rebuild; open serves before it returns; the ceremony door The read gate stops consulting the unnamed isReady() boolean: every provider may expose healthReport() (sync, O(1), composed from exact ledgers — HealthReport with a monotonic generation, per-invariant source ledger|deep|unledgered, missing {count, sample}), and one readiness authority (assessProviderHealth) derives the verdict. Unledgered families are UNKNOWN — never healthy, never broken; a report that throws is a loud not-ready, never a shrug. Reads at the four index choke points refuse with the typed NotReady errors, narrated once per (provider, generation) — a read NEVER starts a store walk: - the first-read lazy build retires (open builds instead, regardless of size — the ≥10k deferral and the "lazy loading on first query" branch go; disableAutoRebuild is re-meant honestly in its docs); - the verify*Live read-path rebuild triggers retire (refuse-or-serve); - the read-time consistency probe that could launch a dark rebuild from an ordinary find() retires; - repairIndex({ rebuild: ['metadata'|'graph'|'vector'] | 'all' }) is the one explicit door: rebuilds the named leg unconditionally and reports rebuilt per family; bare repairIndex() stays report-driven. test(lifecycle): the biography lane — a store's whole life, refereed tests/lifecycle/: an independent shadow model referees every read after every chapter (founding, a working day, clean restart, crash, repair, second life). Chapters 1-3 green. Chapters 4-6 assert the true contract and are marked .fails as a release-blocking finding (the kill-matrix convention): after a crash + adopt reopen the metadata index computes its 'catchup' watermark verdict and nothing consumes it — find() serves the pre-crash index while canonical and counts recover. The catchup wiring is the cure; a passing .fails will force the marker's removal. The lane runs in the integration gate (config + coverage guard).
2026-08-24 12:45:51 -07:00
it('cold index (no isReady()): verifyVectorLive REFUSES immediately — throws VectorIndexNotReadyError, NEVER rebuilds', async () => {
2026-07-13 13:08:52 -07:00
const vi = brain.index
const origSearch = vi.search.bind(vi)
feat(health): the gate reads the named report — reads refuse loudly, never rebuild; open serves before it returns; the ceremony door The read gate stops consulting the unnamed isReady() boolean: every provider may expose healthReport() (sync, O(1), composed from exact ledgers — HealthReport with a monotonic generation, per-invariant source ledger|deep|unledgered, missing {count, sample}), and one readiness authority (assessProviderHealth) derives the verdict. Unledgered families are UNKNOWN — never healthy, never broken; a report that throws is a loud not-ready, never a shrug. Reads at the four index choke points refuse with the typed NotReady errors, narrated once per (provider, generation) — a read NEVER starts a store walk: - the first-read lazy build retires (open builds instead, regardless of size — the ≥10k deferral and the "lazy loading on first query" branch go; disableAutoRebuild is re-meant honestly in its docs); - the verify*Live read-path rebuild triggers retire (refuse-or-serve); - the read-time consistency probe that could launch a dark rebuild from an ordinary find() retires; - repairIndex({ rebuild: ['metadata'|'graph'|'vector'] | 'all' }) is the one explicit door: rebuilds the named leg unconditionally and reports rebuilt per family; bare repairIndex() stays report-driven. test(lifecycle): the biography lane — a store's whole life, refereed tests/lifecycle/: an independent shadow model referees every read after every chapter (founding, a working day, clean restart, crash, repair, second life). Chapters 1-3 green. Chapters 4-6 assert the true contract and are marked .fails as a release-blocking finding (the kill-matrix convention): after a crash + adopt reopen the metadata index computes its 'catchup' watermark verdict and nothing consumes it — find() serves the pre-crash index while canonical and counts recover. The catchup wiring is the cure; a passing .fails will force the marker's removal. The lane runs in the integration gate (config + coverage guard).
2026-08-24 12:45:51 -07:00
let rebuilds = 0
2026-07-13 13:08:52 -07:00
const origRebuild = vi.rebuild.bind(vi)
brain._vectorVerified = false
feat(health): the gate reads the named report — reads refuse loudly, never rebuild; open serves before it returns; the ceremony door The read gate stops consulting the unnamed isReady() boolean: every provider may expose healthReport() (sync, O(1), composed from exact ledgers — HealthReport with a monotonic generation, per-invariant source ledger|deep|unledgered, missing {count, sample}), and one readiness authority (assessProviderHealth) derives the verdict. Unledgered families are UNKNOWN — never healthy, never broken; a report that throws is a loud not-ready, never a shrug. Reads at the four index choke points refuse with the typed NotReady errors, narrated once per (provider, generation) — a read NEVER starts a store walk: - the first-read lazy build retires (open builds instead, regardless of size — the ≥10k deferral and the "lazy loading on first query" branch go; disableAutoRebuild is re-meant honestly in its docs); - the verify*Live read-path rebuild triggers retire (refuse-or-serve); - the read-time consistency probe that could launch a dark rebuild from an ordinary find() retires; - repairIndex({ rebuild: ['metadata'|'graph'|'vector'] | 'all' }) is the one explicit door: rebuilds the named leg unconditionally and reports rebuilt per family; bare repairIndex() stays report-driven. test(lifecycle): the biography lane — a store's whole life, refereed tests/lifecycle/: an independent shadow model referees every read after every chapter (founding, a working day, clean restart, crash, repair, second life). Chapters 1-3 green. Chapters 4-6 assert the true contract and are marked .fails as a release-blocking finding (the kill-matrix convention): after a crash + adopt reopen the metadata index computes its 'catchup' watermark verdict and nothing consumes it — find() serves the pre-crash index while canonical and counts recover. The catchup wiring is the cure; a passing .fails will force the marker's removal. The lane runs in the integration gate (config + coverage guard).
2026-08-24 12:45:51 -07:00
// size()>0 (count present) but search never returns a hit for the known vector.
vi.search = async () => []
vi.rebuild = async (...a: any[]) => { rebuilds++; return origRebuild(...a) }
2026-07-13 13:08:52 -07:00
try {
await expect(
brain.find({ query: 'x', searchMode: 'semantic', limit: 100 })
).rejects.toBeInstanceOf(VectorIndexNotReadyError)
feat(health): the gate reads the named report — reads refuse loudly, never rebuild; open serves before it returns; the ceremony door The read gate stops consulting the unnamed isReady() boolean: every provider may expose healthReport() (sync, O(1), composed from exact ledgers — HealthReport with a monotonic generation, per-invariant source ledger|deep|unledgered, missing {count, sample}), and one readiness authority (assessProviderHealth) derives the verdict. Unledgered families are UNKNOWN — never healthy, never broken; a report that throws is a loud not-ready, never a shrug. Reads at the four index choke points refuse with the typed NotReady errors, narrated once per (provider, generation) — a read NEVER starts a store walk: - the first-read lazy build retires (open builds instead, regardless of size — the ≥10k deferral and the "lazy loading on first query" branch go; disableAutoRebuild is re-meant honestly in its docs); - the verify*Live read-path rebuild triggers retire (refuse-or-serve); - the read-time consistency probe that could launch a dark rebuild from an ordinary find() retires; - repairIndex({ rebuild: ['metadata'|'graph'|'vector'] | 'all' }) is the one explicit door: rebuilds the named leg unconditionally and reports rebuilt per family; bare repairIndex() stays report-driven. test(lifecycle): the biography lane — a store's whole life, refereed tests/lifecycle/: an independent shadow model referees every read after every chapter (founding, a working day, clean restart, crash, repair, second life). Chapters 1-3 green. Chapters 4-6 assert the true contract and are marked .fails as a release-blocking finding (the kill-matrix convention): after a crash + adopt reopen the metadata index computes its 'catchup' watermark verdict and nothing consumes it — find() serves the pre-crash index while canonical and counts recover. The catchup wiring is the cure; a passing .fails will force the marker's removal. The lane runs in the integration gate (config + coverage guard).
2026-08-24 12:45:51 -07:00
expect(rebuilds).toBe(0) // the guard never rebuilds from a read — it refuses loudly instead
2026-07-13 13:08:52 -07:00
} finally {
vi.search = origSearch; vi.rebuild = origRebuild
}
})
feat(health): the gate reads the named report — reads refuse loudly, never rebuild; open serves before it returns; the ceremony door The read gate stops consulting the unnamed isReady() boolean: every provider may expose healthReport() (sync, O(1), composed from exact ledgers — HealthReport with a monotonic generation, per-invariant source ledger|deep|unledgered, missing {count, sample}), and one readiness authority (assessProviderHealth) derives the verdict. Unledgered families are UNKNOWN — never healthy, never broken; a report that throws is a loud not-ready, never a shrug. Reads at the four index choke points refuse with the typed NotReady errors, narrated once per (provider, generation) — a read NEVER starts a store walk: - the first-read lazy build retires (open builds instead, regardless of size — the ≥10k deferral and the "lazy loading on first query" branch go; disableAutoRebuild is re-meant honestly in its docs); - the verify*Live read-path rebuild triggers retire (refuse-or-serve); - the read-time consistency probe that could launch a dark rebuild from an ordinary find() retires; - repairIndex({ rebuild: ['metadata'|'graph'|'vector'] | 'all' }) is the one explicit door: rebuilds the named leg unconditionally and reports rebuilt per family; bare repairIndex() stays report-driven. test(lifecycle): the biography lane — a store's whole life, refereed tests/lifecycle/: an independent shadow model referees every read after every chapter (founding, a working day, clean restart, crash, repair, second life). Chapters 1-3 green. Chapters 4-6 assert the true contract and are marked .fails as a release-blocking finding (the kill-matrix convention): after a crash + adopt reopen the metadata index computes its 'catchup' watermark verdict and nothing consumes it — find() serves the pre-crash index while canonical and counts recover. The catchup wiring is the cure; a passing .fails will force the marker's removal. The lane runs in the integration gate (config + coverage guard).
2026-08-24 12:45:51 -07:00
it('native provider reporting isReady()===false THROWS immediately — never rebuilds', async () => {
2026-07-13 13:08:52 -07:00
const vi = brain.index
feat(health): the gate reads the named report — reads refuse loudly, never rebuild; open serves before it returns; the ceremony door The read gate stops consulting the unnamed isReady() boolean: every provider may expose healthReport() (sync, O(1), composed from exact ledgers — HealthReport with a monotonic generation, per-invariant source ledger|deep|unledgered, missing {count, sample}), and one readiness authority (assessProviderHealth) derives the verdict. Unledgered families are UNKNOWN — never healthy, never broken; a report that throws is a loud not-ready, never a shrug. Reads at the four index choke points refuse with the typed NotReady errors, narrated once per (provider, generation) — a read NEVER starts a store walk: - the first-read lazy build retires (open builds instead, regardless of size — the ≥10k deferral and the "lazy loading on first query" branch go; disableAutoRebuild is re-meant honestly in its docs); - the verify*Live read-path rebuild triggers retire (refuse-or-serve); - the read-time consistency probe that could launch a dark rebuild from an ordinary find() retires; - repairIndex({ rebuild: ['metadata'|'graph'|'vector'] | 'all' }) is the one explicit door: rebuilds the named leg unconditionally and reports rebuilt per family; bare repairIndex() stays report-driven. test(lifecycle): the biography lane — a store's whole life, refereed tests/lifecycle/: an independent shadow model referees every read after every chapter (founding, a working day, clean restart, crash, repair, second life). Chapters 1-3 green. Chapters 4-6 assert the true contract and are marked .fails as a release-blocking finding (the kill-matrix convention): after a crash + adopt reopen the metadata index computes its 'catchup' watermark verdict and nothing consumes it — find() serves the pre-crash index while canonical and counts recover. The catchup wiring is the cure; a passing .fails will force the marker's removal. The lane runs in the integration gate (config + coverage guard).
2026-08-24 12:45:51 -07:00
let rebuilds = 0
2026-07-13 13:08:52 -07:00
const origRebuild = vi.rebuild.bind(vi)
brain._vectorVerified = false
feat(health): the gate reads the named report — reads refuse loudly, never rebuild; open serves before it returns; the ceremony door The read gate stops consulting the unnamed isReady() boolean: every provider may expose healthReport() (sync, O(1), composed from exact ledgers — HealthReport with a monotonic generation, per-invariant source ledger|deep|unledgered, missing {count, sample}), and one readiness authority (assessProviderHealth) derives the verdict. Unledgered families are UNKNOWN — never healthy, never broken; a report that throws is a loud not-ready, never a shrug. Reads at the four index choke points refuse with the typed NotReady errors, narrated once per (provider, generation) — a read NEVER starts a store walk: - the first-read lazy build retires (open builds instead, regardless of size — the ≥10k deferral and the "lazy loading on first query" branch go; disableAutoRebuild is re-meant honestly in its docs); - the verify*Live read-path rebuild triggers retire (refuse-or-serve); - the read-time consistency probe that could launch a dark rebuild from an ordinary find() retires; - repairIndex({ rebuild: ['metadata'|'graph'|'vector'] | 'all' }) is the one explicit door: rebuilds the named leg unconditionally and reports rebuilt per family; bare repairIndex() stays report-driven. test(lifecycle): the biography lane — a store's whole life, refereed tests/lifecycle/: an independent shadow model referees every read after every chapter (founding, a working day, clean restart, crash, repair, second life). Chapters 1-3 green. Chapters 4-6 assert the true contract and are marked .fails as a release-blocking finding (the kill-matrix convention): after a crash + adopt reopen the metadata index computes its 'catchup' watermark verdict and nothing consumes it — find() serves the pre-crash index while canonical and counts recover. The catchup wiring is the cure; a passing .fails will force the marker's removal. The lane runs in the integration gate (config + coverage guard).
2026-08-24 12:45:51 -07:00
vi.isReady = () => false
vi.rebuild = async (...a: any[]) => { rebuilds++; return origRebuild(...a) }
2026-07-13 13:08:52 -07:00
try {
feat(health): the gate reads the named report — reads refuse loudly, never rebuild; open serves before it returns; the ceremony door The read gate stops consulting the unnamed isReady() boolean: every provider may expose healthReport() (sync, O(1), composed from exact ledgers — HealthReport with a monotonic generation, per-invariant source ledger|deep|unledgered, missing {count, sample}), and one readiness authority (assessProviderHealth) derives the verdict. Unledgered families are UNKNOWN — never healthy, never broken; a report that throws is a loud not-ready, never a shrug. Reads at the four index choke points refuse with the typed NotReady errors, narrated once per (provider, generation) — a read NEVER starts a store walk: - the first-read lazy build retires (open builds instead, regardless of size — the ≥10k deferral and the "lazy loading on first query" branch go; disableAutoRebuild is re-meant honestly in its docs); - the verify*Live read-path rebuild triggers retire (refuse-or-serve); - the read-time consistency probe that could launch a dark rebuild from an ordinary find() retires; - repairIndex({ rebuild: ['metadata'|'graph'|'vector'] | 'all' }) is the one explicit door: rebuilds the named leg unconditionally and reports rebuilt per family; bare repairIndex() stays report-driven. test(lifecycle): the biography lane — a store's whole life, refereed tests/lifecycle/: an independent shadow model referees every read after every chapter (founding, a working day, clean restart, crash, repair, second life). Chapters 1-3 green. Chapters 4-6 assert the true contract and are marked .fails as a release-blocking finding (the kill-matrix convention): after a crash + adopt reopen the metadata index computes its 'catchup' watermark verdict and nothing consumes it — find() serves the pre-crash index while canonical and counts recover. The catchup wiring is the cure; a passing .fails will force the marker's removal. The lane runs in the integration gate (config + coverage guard).
2026-08-24 12:45:51 -07:00
await expect(
brain.find({ query: 'x', searchMode: 'semantic', limit: 100 })
).rejects.toBeInstanceOf(VectorIndexNotReadyError)
expect(rebuilds).toBe(0) // a not-ready report throws immediately — it is never a rebuild trigger
2026-07-13 13:08:52 -07:00
} finally {
delete vi.isReady; vi.rebuild = origRebuild
}
})
it('a text-only query does not trigger the vector guard', async () => {
brain._vectorVerified = false
await brain.find({ query: 'active', searchMode: 'text', limit: 5 })
expect(brain._vectorVerified).toBe(false) // executeVectorSearch never called
})
// Regression: the probe must check "returns ANY hit", not an exact self-match —
// HNSW is approximate and get() may re-hydrate the vector, so a healthy
// many-entity index would false-positive under an exact-self check, wrongly
// rebuild, and throw VectorIndexNotReadyError on working data.
it('a healthy many-entity brain with distinct vectors serves semantic find, never throws/rebuilds', async () => {
const many = new Brainy({ requireSubtype: false, storage: { type: 'memory' }, dimensions: 384 })
await many.init()
for (let i = 0; i < 25; i++) {
const v = Array.from({ length: 384 }, (_, j) => Math.sin((i * 7 + j) * 0.13) + 0.001)
await many.add({ vector: v, type: NounType.Concept, metadata: { n: i } })
}
await many.flush()
let rebuilds = 0
const vi = (many as any).index
const origRebuild = vi.rebuild.bind(vi)
vi.rebuild = async (...a: any[]) => { rebuilds++; return origRebuild(...a) }
const res = await many.find({ query: 'x', searchMode: 'semantic', limit: 10 })
expect(res.length).toBeGreaterThan(0)
expect(rebuilds).toBe(0)
expect((many as any)._vectorVerified).toBe(true)
vi.rebuild = origRebuild
await many.close()
})
})