test: lens-consistency regression — combined vs subtype-only vs canonical ground truth
Ports the fresh-brain probe that closed the type+subtype lens-drop investigation into the permanent suite: a 7-pair corpus seeded through the real write API (never restored bytes), every lens checked id-for-id against an unfiltered canonical scan, warm AND after a cold reopen, plus the type+subtype update-flip leg. Guards the under-inclusion class the egress integrity guard structurally cannot catch.
This commit is contained in:
parent
af8c1795bd
commit
4fb41f9a7c
1 changed files with 138 additions and 0 deletions
138
tests/integration/lens-consistency.test.ts
Normal file
138
tests/integration/lens-consistency.test.ts
Normal file
|
|
@ -0,0 +1,138 @@
|
|||
/**
|
||||
* @module tests/integration/lens-consistency
|
||||
* @description The three metadata "lenses" over one corpus must agree with
|
||||
* canonical ground truth id-for-id, warm AND after a cold reopen:
|
||||
* - combined: find({ type: T, where: { subtype: S } })
|
||||
* - subtype-only: find({ where: { subtype: S } })
|
||||
* - type-only: find({ type: T })
|
||||
* Ported from the fresh-brain probe that closed the type+subtype lens-drop
|
||||
* investigation (a restored pre-8.2.2 torn capture had entities visible to the
|
||||
* subtype-only lens but dropped by the combined lens — "0 of 2 migrated, all
|
||||
* gates green"). The corpus is seeded through the REAL write API — never
|
||||
* restored bytes — which is what made the original datapoint decisive. The
|
||||
* invariants: every lens matches an unfiltered canonical scan exactly (no
|
||||
* missing ids, no extras) and combined ⊆ subtype-only always holds.
|
||||
*/
|
||||
import { describe, it, expect, beforeAll, afterAll } from 'vitest'
|
||||
import * as fs from 'node:fs'
|
||||
import * as os from 'node:os'
|
||||
import * as path from 'node:path'
|
||||
import { Brainy } from '../../src/index.js'
|
||||
|
||||
/** The corpus: 7 (type, subtype) pairs, uneven counts, incl. the incident's 2-of-a-pair shape. */
|
||||
const CORPUS: Array<{ type: string; subtype: string; count: number }> = [
|
||||
{ type: 'proposition', subtype: 'decision', count: 2 }, // the incident shape: "0 of 2"
|
||||
{ type: 'concept', subtype: 'decision', count: 3 },
|
||||
{ type: 'task', subtype: 'decision', count: 2 },
|
||||
{ type: 'concept', subtype: 'action', count: 4 },
|
||||
{ type: 'message', subtype: 'note', count: 5 },
|
||||
{ type: 'message', subtype: 'ship', count: 3 },
|
||||
{ type: 'document', subtype: 'guide', count: 4 }
|
||||
]
|
||||
|
||||
/** Canonical ground truth: unfiltered enumeration, post-filtered IN THE TEST. */
|
||||
async function groundTruth(
|
||||
brain: any,
|
||||
match: { type?: string; subtype?: string }
|
||||
): Promise<Set<string>> {
|
||||
const ids = new Set<string>()
|
||||
let cursor: string | undefined
|
||||
for (;;) {
|
||||
const page = await brain.storage.getNounsWithPagination({ limit: 500, cursor })
|
||||
for (const noun of page.items) {
|
||||
// Hydrated shape: `type`/`subtype` are TOP-LEVEL; `metadata` holds only
|
||||
// custom user fields (vfsType is one — the VFS plumbing marker).
|
||||
const n = noun as any
|
||||
if (n.metadata?.vfsType) continue // VFS plumbing is not corpus
|
||||
if (!n.type || !n.subtype) continue
|
||||
if (match.type && n.type !== match.type) continue
|
||||
if (match.subtype && n.subtype !== match.subtype) continue
|
||||
ids.add(n.id)
|
||||
}
|
||||
if (!page.hasMore) break
|
||||
cursor = page.nextCursor
|
||||
}
|
||||
return ids
|
||||
}
|
||||
|
||||
const idSet = (results: Array<{ id: string }>): Set<string> => new Set(results.map((r) => r.id))
|
||||
|
||||
/** Every lens vs ground truth, id-for-id, for every pair in the corpus. */
|
||||
async function assertAllLenses(brain: any): Promise<void> {
|
||||
const types = [...new Set(CORPUS.map((c) => c.type))]
|
||||
const subtypes = [...new Set(CORPUS.map((c) => c.subtype))]
|
||||
|
||||
for (const { type, subtype } of CORPUS) {
|
||||
const combined = idSet(await brain.find({ type, where: { subtype }, limit: 1000 }))
|
||||
const subtypeOnly = idSet(await brain.find({ where: { subtype }, limit: 1000 }))
|
||||
const truthPair = await groundTruth(brain, { type, subtype })
|
||||
const truthSubtype = await groundTruth(brain, { subtype })
|
||||
|
||||
expect([...combined].sort()).toEqual([...truthPair].sort()) // no drops, no extras
|
||||
expect([...subtypeOnly].sort()).toEqual([...truthSubtype].sort())
|
||||
for (const id of combined) expect(subtypeOnly.has(id)).toBe(true) // combined ⊆ subtype-only
|
||||
}
|
||||
|
||||
for (const type of types) {
|
||||
const typeOnly = idSet(await brain.find({ type, limit: 1000 }))
|
||||
const truthType = await groundTruth(brain, { type })
|
||||
expect([...typeOnly].sort()).toEqual([...truthType].sort())
|
||||
}
|
||||
|
||||
// Count cross-check against the corpus definition itself.
|
||||
for (const subtype of subtypes) {
|
||||
const expected = CORPUS.filter((c) => c.subtype === subtype).reduce((s, c) => s + c.count, 0)
|
||||
const got = (await brain.find({ where: { subtype }, limit: 1000 })).length
|
||||
expect(got).toBe(expected)
|
||||
}
|
||||
}
|
||||
|
||||
describe('lens consistency — combined vs subtype-only vs canonical ground truth', () => {
|
||||
let dir: string
|
||||
let brain: any
|
||||
|
||||
beforeAll(async () => {
|
||||
process.env.BRAINY_DETERMINISTIC_EMBEDDINGS = 'true'
|
||||
dir = fs.mkdtempSync(path.join(os.tmpdir(), 'brainy-lens-'))
|
||||
brain = new Brainy({ requireSubtype: false, storage: { type: 'filesystem', path: dir }, silent: true, dimensions: 384 })
|
||||
await brain.init()
|
||||
// Seed through the REAL write API — never restored bytes.
|
||||
let i = 0
|
||||
for (const { type, subtype, count } of CORPUS) {
|
||||
for (let k = 0; k < count; k++) {
|
||||
await brain.add({ data: `${type} ${subtype} ${i++}`, type, subtype, metadata: { k } })
|
||||
}
|
||||
}
|
||||
await brain.flush()
|
||||
})
|
||||
afterAll(async () => {
|
||||
await brain.close?.().catch(() => {})
|
||||
fs.rmSync(dir, { recursive: true, force: true })
|
||||
})
|
||||
|
||||
it('WARM: all lenses agree with ground truth id-for-id', async () => {
|
||||
await assertAllLenses(brain)
|
||||
})
|
||||
|
||||
it('COLD REOPEN: all lenses still agree after close + reopen from disk', async () => {
|
||||
await brain.close()
|
||||
brain = new Brainy({ requireSubtype: false, storage: { type: 'filesystem', path: dir }, silent: true, dimensions: 384 })
|
||||
await brain.init()
|
||||
await assertAllLenses(brain)
|
||||
})
|
||||
|
||||
it('after an update() flips type AND subtype, every lens tracks the move exactly', async () => {
|
||||
// The historical cross-bucket-staleness path: change (concept, action) -> (task, review).
|
||||
const victims = await brain.find({ type: 'concept', where: { subtype: 'action' }, limit: 1 })
|
||||
expect(victims.length).toBe(1)
|
||||
const id = victims[0].id
|
||||
await brain.update({ id, type: 'task', subtype: 'review' })
|
||||
|
||||
const oldCombined = idSet(await brain.find({ type: 'concept', where: { subtype: 'action' }, limit: 1000 }))
|
||||
expect(oldCombined.has(id)).toBe(false) // unposted from the old buckets
|
||||
const newCombined = idSet(await brain.find({ type: 'task', where: { subtype: 'review' }, limit: 1000 }))
|
||||
expect(newCombined.has(id)).toBe(true) // posted to the new buckets
|
||||
const subtypeOnly = idSet(await brain.find({ where: { subtype: 'review' }, limit: 1000 }))
|
||||
expect(subtypeOnly.has(id)).toBe(true)
|
||||
})
|
||||
})
|
||||
Loading…
Add table
Add a link
Reference in a new issue