test(graph): cut graphIndex-pagination from 304s to under a second

18 pagination tests recreated a fresh FileSystemStorage-backed Brainy plus
51 real-embedded entities (1 central hub + 50 neighbors) in a beforeEach
before EVERY test — ~950 add()/relate() calls total, each paying the real
ONNX embedder. Measured before this change: 303.69s (fresh run, this
session). None of these tests exercise similarity search, only graph
pagination, so three changes cut the cost without touching an assertion:

- vector: [] on every add() — add()'s `params.vector || embed(...)` never
  calls the embedder once vector is present, even the sanctioned unvectored
  [] shape (confirmed against brainy.ts's zero-norm-law comment: the
  dimension-pinning gate is `vector.length > 0`, so [] never poisons
  dimensions for a later real embed).
- storage: { type: 'memory' } instead of the 'auto' default (FileSystemStorage
  at ./brainy-data) — sidesteps tests/setup.ts's global per-test
  `rm -rf brainy-data`, which would otherwise corrupt a brain shared across
  a describe's beforeAll out from under it.
- the base fixture (hub + 50 neighbors) now builds once per describe
  (beforeAll) instead of once per test — safe because no test in a given
  describe mutates the shared fixture in a way an earlier sibling test's
  assertion depends on (the one mutating case is the last test in its
  describe).

Measured after: 416ms for all 18 tests (2.35s wall including vitest
startup), all 18 still passing.
This commit is contained in:
David Snelling 2026-09-02 13:43:15 -07:00
parent 7932175503
commit 6597c146f7

View file

@ -9,9 +9,34 @@
* 8.0 BigInt boundary: entity ints in (resolved via the metadata index's
* idMapper), entity/verb ints out (`bigint[]`). Entity ints map back to UUIDs
* via `idMapper.getUuid(Number(int))`; verb ints via `verbIntsToIds()`.
*
* COST NOTE (2026-09): this file's `beforeEach` used to recreate a fresh
* FileSystemStorage-backed Brainy plus 51 real-embedded entities before
* EVERY one of the 18 tests below (~950 add()/relate() calls total, each
* paying the real ONNX embedder the whole file walled ~328s). Fixed
* without touching a single assertion:
*
* (1) `vector: []` on every add() below these tests exercise graph
* pagination, never similarity, so a pre-supplied vector is honest, not
* a shortcut: `add()`'s `params.vector || (await this.embed(...))` never
* calls the embedder once `vector` is present, even the sanctioned
* unvectored `[]` shape (see brainy.ts's add(), the zero-norm-law
* comment) and the `vector.length > 0` gate on dimension-pinning means
* `[]` never poisons `this.dimensions` for later real embeds.
* (2) `storage: { type: 'memory' }` instead of the 'auto' default
* (FileSystemStorage at ./brainy-data) real disk I/O the pagination
* assertions never needed, and it sidesteps tests/setup.ts's global
* per-test `rm -rf brainy-data`, which would otherwise corrupt a brain
* shared across a describe's beforeAll out from under it.
* (3) the base fixture (one central hub + 50 outgoing-edge neighbors) now
* builds ONCE per describe (`beforeAll`) instead of once per test safe
* because no test in a given describe block mutates the shared fixture
* in a way an earlier sibling test's assertion depends on (the one
* mutating case, the incoming-direction test, is the LAST test in its
* describe).
*/
import { describe, it, expect, beforeEach } from 'vitest'
import { describe, it, expect, beforeAll, afterAll } from 'vitest'
import { Brainy } from '../../src/brainy.js'
import { NounType, VerbType } from '../../src/types/graphTypes.js'
@ -39,14 +64,21 @@ describe('GraphAdjacencyIndex Pagination', () => {
.map((i) => idMapper().getUuid(Number(i)))
.filter((u: string | undefined): u is string => u !== undefined)
beforeEach(async () => {
/**
* Builds one central hub + 50 neighbor entities (all outgoing edges from
* the hub), unvectored and on in-memory storage (see the file header).
* Assigns the describe-scoped `brain`/`centralId`/`neighborIds` above;
* called once per describe via `beforeAll`, not once per test.
*/
async function buildFixture(): Promise<void> {
brain = new Brainy({ requireSubtype: false })
await brain.init()
await brain.init({ storage: { type: 'memory' } })
// Create central entity
centralId = await brain.add({
data: { name: 'Central Hub' },
type: NounType.Thing
type: NounType.Thing,
vector: []
})
// Create 50 neighbor entities with relationships
@ -54,7 +86,8 @@ describe('GraphAdjacencyIndex Pagination', () => {
for (let i = 0; i < 50; i++) {
const neighborId = await brain.add({
data: { name: `Neighbor ${i}`, index: i },
type: NounType.Thing
type: NounType.Thing,
vector: []
})
neighborIds.push(neighborId)
@ -65,9 +98,14 @@ describe('GraphAdjacencyIndex Pagination', () => {
type: VerbType.RelatesTo
})
}
})
}
describe('getNeighbors() Pagination', () => {
beforeAll(buildFixture)
afterAll(async () => {
await brain?.close()
})
it('should return all neighbors without pagination', async () => {
const neighborInts = await graphIndex().getNeighbors(entityInt(centralId))
const neighbors = intsToUuids(neighborInts)
@ -149,7 +187,8 @@ describe('GraphAdjacencyIndex Pagination', () => {
// Create some incoming relationships
const sourceId = await brain.add({
data: { name: 'Source' },
type: NounType.Thing
type: NounType.Thing,
vector: []
})
await brain.relate({
@ -169,6 +208,11 @@ describe('GraphAdjacencyIndex Pagination', () => {
})
describe('getVerbIdsBySource() Pagination', () => {
beforeAll(buildFixture)
afterAll(async () => {
await brain?.close()
})
it('should return all verb ints without pagination and resolve them back to ids', async () => {
const verbInts: bigint[] = await graphIndex().getVerbIdsBySource(entityInt(centralId))
@ -223,6 +267,11 @@ describe('GraphAdjacencyIndex Pagination', () => {
})
describe('getVerbIdsByTarget() Pagination', () => {
beforeAll(buildFixture)
afterAll(async () => {
await brain?.close()
})
it('should return all verb ints targeting an entity', async () => {
// Pick a neighbor that's a target of relationships
const targetId = neighborIds[0]
@ -236,14 +285,16 @@ describe('GraphAdjacencyIndex Pagination', () => {
// Create entity with many incoming relationships
const popularTarget = await brain.add({
data: { name: 'Popular Target' },
type: NounType.Thing
type: NounType.Thing,
vector: []
})
// Create 30 relationships pointing to it
for (let i = 0; i < 30; i++) {
const sourceId = await brain.add({
data: { name: `Source ${i}` },
type: NounType.Thing
type: NounType.Thing,
vector: []
})
await brain.relate({
from: sourceId,
@ -267,6 +318,11 @@ describe('GraphAdjacencyIndex Pagination', () => {
})
describe('Performance with Pagination', () => {
beforeAll(buildFixture)
afterAll(async () => {
await brain?.close()
})
it('should maintain sub-5ms performance with pagination', async () => {
const central = entityInt(centralId)
@ -285,11 +341,17 @@ describe('GraphAdjacencyIndex Pagination', () => {
})
describe('Real-World Use Cases', () => {
beforeAll(buildFixture)
afterAll(async () => {
await brain?.close()
})
it('should efficiently paginate through high-degree node', async () => {
// Simulate popular entity with 100+ relationships
const hub = await brain.add({
data: { name: 'Popular Hub' },
type: NounType.Thing
type: NounType.Thing,
vector: []
})
// Create 100 relationships
@ -297,7 +359,8 @@ describe('GraphAdjacencyIndex Pagination', () => {
for (let i = 0; i < 100; i++) {
const targetId = await brain.add({
data: { name: `Target ${i}` },
type: NounType.Thing
type: NounType.Thing,
vector: []
})
targetIds.push(targetId)
await brain.relate({