Write/index-spine hardening, first batch of Pass 1. Each fix restores an invariant the surrounding code already intended; every one has a fail-before/pass-after test. - Pattern C, finding 5 (baseStorage): delete now decrements the user-facing scalar total symmetrically — deleteNounMetadata was decrementing only the per-type bucket, deleteVerbMetadata neither the bucket nor the scalar, so getNounCount()/getVerbCount() inflated permanently (the stale scalar wins pagination via Math.max and is persisted). Invariant now holds: scalar total === Σ per-type across add/update/delete and reopen. - Pattern A, finding 3 (graph/lsm/LSMTree): a partial SSTable-load failure no longer publishes the manifest's full relationship count as healthy. Any per-SSTable load failure throws after the batch, which resets to honest-empty and lets the existing size()===0 self-heal rebuild run — size()/isHealthy() can no longer lie about a partial load. - Pattern B, finding 6 (hnsw/hnswIndex): deferred flush() no longer clears dirty nodes whose connections failed to persist — failed nodes stay in the retry set, and flush() throws HnswFlushError instead of returning a lying node count. The immediate-mode first-noun saveHNSWSystem is un-swallowed, so addItem() rejects rather than returning an id for a rootless index. - Pattern B, finding 11 (part — storage reads): new shared isAbsentError() helper (utils/errorClassification, ENOENT-only absence) applied to loadBinaryBlob and readObjectFromPath — a real IO fault (EIO/EACCES/EMFILE) now propagates loudly instead of masquerading as "absent", which had driven needless rebuilds / empty reads (loadBinaryBlob feeds the native provider). Regression: 78 green across the 3 new suites + db-mvcc, generationStore, temporal-vfs, rollback-trapdoor, restore-nondestructive. Full gate runs before the Pass-1 release (David-gated). Remaining Pass 1: finding 11 getNoun/getVerb legs, finding 4 (ColumnStore), finding 8 (pending-flush), finding 10 (degraded), finding 7 (clear). Pattern A guards (1,2,9) as a follow-up release.
157 lines
6.2 KiB
TypeScript
157 lines
6.2 KiB
TypeScript
/**
|
|
* HNSW deferred-flush durability tests (Finding 6 — spine-plan Part B,
|
|
* "blind catch" audit).
|
|
*
|
|
* `JsHnswVectorIndex.flush()` used to swallow a per-node
|
|
* `persistNodeConnections` failure (and a system-record `saveHNSWSystem`
|
|
* failure) behind `console.error`, then unconditionally clear the dirty set
|
|
* and report the pre-flush node count as "success". A transient write fault
|
|
* therefore silently dropped that node's connections from durable storage
|
|
* forever, with `flush()` having lied about it. These tests pin the fix:
|
|
* a node (or the system record) that fails to persist stays in the
|
|
* dirty/retry set, and `flush()` throws {@link HnswFlushError} instead of
|
|
* returning a count. The immediate-mode first-noun `saveHNSWSystem` swallow
|
|
* (while `addItem()` still returned the id) is covered too.
|
|
*/
|
|
|
|
import { describe, it, expect, vi } from 'vitest'
|
|
import { v4 as uuidv4 } from 'uuid'
|
|
import { JsHnswVectorIndex, HnswFlushError } from '../../../src/hnsw/hnswIndex.js'
|
|
import { euclideanDistance } from '../../../src/utils/index.js'
|
|
import { MemoryStorage } from '../../../src/storage/adapters/memoryStorage.js'
|
|
|
|
// Helper: generate a random vector of given dimension (mirrors lazy-vectors.test.ts).
|
|
function randomVector(dim: number): number[] {
|
|
return Array.from({ length: dim }, () => Math.random() * 2 - 1)
|
|
}
|
|
|
|
describe('HNSW deferred-flush durability (Finding 6)', () => {
|
|
const dim = 8
|
|
|
|
it('a clean flush persists every dirty node and clears the dirty set', async () => {
|
|
const storage = new MemoryStorage()
|
|
const index = new JsHnswVectorIndex(
|
|
{ M: 4, efConstruction: 50, efSearch: 20 },
|
|
euclideanDistance,
|
|
{ useParallelization: false, storage, persistMode: 'deferred' }
|
|
)
|
|
|
|
for (let i = 0; i < 5; i++) {
|
|
await index.addItem({ id: uuidv4(), vector: randomVector(dim) })
|
|
}
|
|
|
|
// Sanity: inserting several items in deferred mode does dirty something.
|
|
expect((index as any).dirtyNodes.size).toBeGreaterThan(0)
|
|
|
|
const flushed = await index.flush()
|
|
expect(typeof flushed).toBe('number')
|
|
expect((index as any).dirtyNodes.size).toBe(0)
|
|
expect((index as any).dirtySystem).toBe(false)
|
|
})
|
|
|
|
it('retains a node whose connections failed to persist and surfaces HnswFlushError instead of reporting success', async () => {
|
|
const storage = new MemoryStorage()
|
|
const index = new JsHnswVectorIndex(
|
|
{ M: 4, efConstruction: 50, efSearch: 20 },
|
|
euclideanDistance,
|
|
{ useParallelization: false, storage, persistMode: 'deferred' }
|
|
)
|
|
|
|
const ids: string[] = []
|
|
for (let i = 0; i < 5; i++) {
|
|
const id = uuidv4()
|
|
ids.push(id)
|
|
await index.addItem({ id, vector: randomVector(dim) })
|
|
}
|
|
|
|
// The very first node is guaranteed to become a neighbor of the second
|
|
// insert (it is the only existing node in the graph at that point), so it
|
|
// is deterministically present in the dirty set before any fault.
|
|
const failId = ids[0]
|
|
const dirtyBefore = (index as any).dirtyNodes as Set<string>
|
|
expect(dirtyBefore.has(failId)).toBe(true)
|
|
|
|
const originalSave = storage.saveVectorIndexData.bind(storage)
|
|
const spy = vi
|
|
.spyOn(storage, 'saveVectorIndexData')
|
|
.mockImplementation(async (nounId, hnswData) => {
|
|
if (nounId === failId) {
|
|
throw Object.assign(new Error('simulated write fault'), { code: 'EIO' })
|
|
}
|
|
return originalSave(nounId, hnswData)
|
|
})
|
|
|
|
await expect(index.flush()).rejects.toBeInstanceOf(HnswFlushError)
|
|
|
|
const dirtyAfter = (index as any).dirtyNodes as Set<string>
|
|
// The failed node stays dirty for the next retry ...
|
|
expect(dirtyAfter.has(failId)).toBe(true)
|
|
// ... and every node that DID persist successfully leaves the dirty set —
|
|
// the failure of one node must not re-dirty (or fail to clear) the rest.
|
|
expect(dirtyAfter.size).toBe(1)
|
|
|
|
// Recovery: once the fault clears, the retained node persists and the
|
|
// flush reports success again (the intended retry path).
|
|
spy.mockRestore()
|
|
const flushed = await index.flush()
|
|
expect(typeof flushed).toBe('number')
|
|
expect((index as any).dirtyNodes.size).toBe(0)
|
|
})
|
|
|
|
it('surfaces a system-record persist failure as HnswFlushError and keeps dirtySystem set for retry', async () => {
|
|
const storage = new MemoryStorage()
|
|
const index = new JsHnswVectorIndex(
|
|
{ M: 4, efConstruction: 50, efSearch: 20 },
|
|
euclideanDistance,
|
|
{ useParallelization: false, storage, persistMode: 'deferred' }
|
|
)
|
|
|
|
// First noun in deferred mode marks dirtySystem (entryPoint/maxLevel).
|
|
await index.addItem({ id: uuidv4(), vector: randomVector(dim) })
|
|
expect((index as any).dirtySystem).toBe(true)
|
|
|
|
const spy = vi
|
|
.spyOn(storage, 'saveHNSWSystem')
|
|
.mockRejectedValue(
|
|
Object.assign(new Error('simulated system write fault'), { code: 'EIO' })
|
|
)
|
|
|
|
let caught: unknown
|
|
try {
|
|
await index.flush()
|
|
} catch (error) {
|
|
caught = error
|
|
}
|
|
|
|
expect(caught).toBeInstanceOf(HnswFlushError)
|
|
expect((caught as HnswFlushError).systemFailed).toBe(true)
|
|
// The system record must stay dirty — a lost entry point/maxLevel update
|
|
// must be retried, not silently dropped.
|
|
expect((index as any).dirtySystem).toBe(true)
|
|
|
|
spy.mockRestore()
|
|
const flushed = await index.flush()
|
|
expect(typeof flushed).toBe('number')
|
|
expect((index as any).dirtySystem).toBe(false)
|
|
})
|
|
|
|
it('surfaces the immediate-mode first-noun system persist failure via a rejecting addItem()', async () => {
|
|
const storage = new MemoryStorage()
|
|
const index = new JsHnswVectorIndex(
|
|
{ M: 4, efConstruction: 50, efSearch: 20 },
|
|
euclideanDistance,
|
|
{ useParallelization: false, storage } // default persistMode: 'immediate'
|
|
)
|
|
|
|
vi.spyOn(storage, 'saveHNSWSystem').mockRejectedValue(
|
|
Object.assign(new Error('simulated system write fault'), { code: 'EIO' })
|
|
)
|
|
|
|
// Previously this swallowed the error (console.error) and addItem()
|
|
// still resolved with the id — stranding a brand-new index whose root
|
|
// (entry point) was never actually persisted. Now it must reject.
|
|
await expect(
|
|
index.addItem({ id: uuidv4(), vector: randomVector(dim) })
|
|
).rejects.toThrow('simulated system write fault')
|
|
})
|
|
})
|