fix: extraction, multi-hop traversal, and aggregate result shape (BR-ADV-FEATURES-BUN)
Three advanced-API correctness fixes, all reproducible on Node (not Bun-specific):
- Entity/concept extraction returned []. SmartExtractor combined agreeing signals
with a weighted sum compared against an absolute 0.60 gate, so a confident
low-weight signal lost selection to a mediocre high-weight one that then failed
the gate, dropping the whole result. Select and gate on a normalized weighted
average instead. The public `confidence` option now controls the threshold (was
a dead hardcoded 0.60), and the embedding-signal timeout is raised 100ms -> 2000ms
so the neural signal is not silently dropped on slower runtimes.
- Multi-hop find({ connected }) returned only the 1-hop neighbour. executeGraphSearch
ignored depth/via; it now delegates to the depth-aware neighbors() BFS.
- find({ aggregate }) hid groupKey/metrics/count under .metadata, so callers
expecting AggregateResult saw empty rows. Expose those fields at the top level.
Adds real-embedding regression tests in tests/integration/advanced-apis-regression.test.ts.
This commit is contained in:
parent
07754d135a
commit
0a9d1d9a17
8 changed files with 297 additions and 74 deletions
131
tests/integration/advanced-apis-regression.test.ts
Normal file
131
tests/integration/advanced-apis-regression.test.ts
Normal file
|
|
@ -0,0 +1,131 @@
|
|||
/**
|
||||
* Regression tests for the three advanced-API defects reported as BR-ADV-FEATURES-BUN:
|
||||
*
|
||||
* 1. Entity/concept extraction returned `[]` for entity-rich text. Root cause: the
|
||||
* SmartExtractor ensemble combined agreeing signals with a weighted *sum* compared against
|
||||
* an absolute gate, so a confident-but-low-weight signal was discarded whenever a
|
||||
* higher-weight signal voted a different (lower-confidence) type. Fixed by selecting and
|
||||
* gating on a normalized weighted *average*.
|
||||
* 2. `find({ aggregate })` rows did not expose the documented AggregateResult fields
|
||||
* (`groupKey`/`metrics`/`count`) at the top level, so callers saw "empty" rows.
|
||||
* 3. `find({ connected: { depth } })` ignored `depth`/`via` and returned only the 1-hop
|
||||
* neighbour. Fixed by delegating to the depth-aware `neighbors()` BFS.
|
||||
*
|
||||
* These run with REAL embeddings (no mocks) — the original failures were invisible under the
|
||||
* zero-vector mock embeddings used by the unit suite.
|
||||
*/
|
||||
|
||||
import { describe, it, expect, beforeAll, afterAll } from 'vitest'
|
||||
import { Brainy } from '../../src/brainy.js'
|
||||
import { NounType, VerbType } from '../../src/types/graphTypes.js'
|
||||
|
||||
describe('BR-ADV-FEATURES-BUN regression', () => {
|
||||
let brain: Brainy
|
||||
|
||||
beforeAll(async () => {
|
||||
brain = new Brainy({ storage: { type: 'memory' } })
|
||||
await brain.init()
|
||||
// Warm the embedding model so the first real extraction isn't paying cold-start latency.
|
||||
await brain.embed('warmup')
|
||||
})
|
||||
|
||||
afterAll(async () => {
|
||||
await brain.close()
|
||||
})
|
||||
|
||||
describe('Bug 1 — entity extraction is not empty', () => {
|
||||
const text = 'Sarah Chen founded Acme Corp in New York'
|
||||
|
||||
it('extractEntities returns the entity spans (was [])', async () => {
|
||||
const entities = await brain.extractEntities(text, { confidence: 0.1 })
|
||||
expect(entities.length).toBeGreaterThan(0)
|
||||
const texts = entities.map(e => e.text)
|
||||
// The candidate spans must be surfaced (type accuracy on dense sentences is a separate
|
||||
// pre-existing concern; this guards the reported "returns nothing" regression).
|
||||
expect(texts).toContain('Sarah Chen')
|
||||
expect(texts).toContain('Acme Corp')
|
||||
})
|
||||
|
||||
it('a confident single-signal candidate classifies correctly in isolation', async () => {
|
||||
// "Acme Corp" carries a strong Organization indicator ("Corp") with no competing context.
|
||||
const entities = await brain.extractEntities('Acme Corp', { confidence: 0.6 })
|
||||
expect(entities.length).toBeGreaterThan(0)
|
||||
expect(entities.some(e => e.type === NounType.Organization)).toBe(true)
|
||||
})
|
||||
|
||||
it('the confidence option actually controls the gate (no longer dead)', async () => {
|
||||
const loose = await brain.extractEntities(text, { confidence: 0.1 })
|
||||
const strict = await brain.extractEntities(text, { confidence: 0.99 })
|
||||
expect(loose.length).toBeGreaterThan(0)
|
||||
// Previously confidence had no effect because SmartExtractor applied a hardcoded 0.60
|
||||
// floor; a near-1.0 threshold must now admit no fewer results than a loose one, and here
|
||||
// strictly fewer (nothing scores ≥ 0.99).
|
||||
expect(strict.length).toBeLessThanOrEqual(loose.length)
|
||||
})
|
||||
})
|
||||
|
||||
describe('Bug 3 — multi-hop traversal honors depth', () => {
|
||||
let a: string
|
||||
let b: string
|
||||
let c: string
|
||||
|
||||
beforeAll(async () => {
|
||||
a = await brain.add({ data: 'graph node A', type: NounType.Concept, metadata: { name: 'A' } })
|
||||
b = await brain.add({ data: 'graph node B', type: NounType.Concept, metadata: { name: 'B' } })
|
||||
c = await brain.add({ data: 'graph node C', type: NounType.Concept, metadata: { name: 'C' } })
|
||||
await brain.relate({ from: a, to: b, type: VerbType.RelatedTo })
|
||||
await brain.relate({ from: b, to: c, type: VerbType.RelatedTo })
|
||||
})
|
||||
|
||||
const names = (rows: Array<{ metadata?: any }>) =>
|
||||
rows.map(r => r.metadata?.name).filter(Boolean).sort()
|
||||
|
||||
it('depth 1 returns only the immediate neighbour', async () => {
|
||||
const rows = await brain.find({ connected: { from: a, depth: 1, direction: 'out' } })
|
||||
expect(names(rows)).toEqual(['B'])
|
||||
})
|
||||
|
||||
it('depth 2 reaches the second hop (was 1-hop only)', async () => {
|
||||
const rows = await brain.find({ connected: { from: a, depth: 2, direction: 'out' } })
|
||||
expect(names(rows)).toEqual(['B', 'C'])
|
||||
})
|
||||
|
||||
it('depth 3 on a 2-edge chain still returns B and C', async () => {
|
||||
const rows = await brain.find({ connected: { from: a, depth: 3, direction: 'out' } })
|
||||
expect(names(rows)).toEqual(['B', 'C'])
|
||||
})
|
||||
})
|
||||
|
||||
describe('Bug 2 — find({ aggregate }) exposes AggregateResult fields', () => {
|
||||
beforeAll(async () => {
|
||||
brain.defineAggregate({
|
||||
name: 'spend_by_cat',
|
||||
source: { type: NounType.Event },
|
||||
groupBy: ['category'],
|
||||
metrics: { total: { op: 'sum', field: 'amount' }, count: { op: 'count' } }
|
||||
})
|
||||
await brain.add({ data: 'coffee', type: NounType.Event, metadata: { category: 'food', amount: 5 } })
|
||||
await brain.add({ data: 'lunch', type: NounType.Event, metadata: { category: 'food', amount: 12 } })
|
||||
await brain.add({ data: 'bus', type: NounType.Event, metadata: { category: 'transit', amount: 3 } })
|
||||
await brain.flush()
|
||||
await new Promise(r => setTimeout(r, 500)) // let materialization debounce settle
|
||||
})
|
||||
|
||||
it('rows carry top-level groupKey / metrics / count (was hidden under metadata)', async () => {
|
||||
const rows: any[] = await brain.find({ aggregate: 'spend_by_cat' })
|
||||
expect(rows.length).toBe(2)
|
||||
|
||||
const food = rows.find(r => r.groupKey?.category === 'food')
|
||||
const transit = rows.find(r => r.groupKey?.category === 'transit')
|
||||
|
||||
expect(food).toBeDefined()
|
||||
expect(food.groupKey).toEqual({ category: 'food' })
|
||||
expect(food.metrics.total).toBe(17)
|
||||
expect(food.count).toBe(2)
|
||||
|
||||
expect(transit).toBeDefined()
|
||||
expect(transit.metrics.total).toBe(3)
|
||||
expect(transit.count).toBe(1)
|
||||
})
|
||||
})
|
||||
})
|
||||
Loading…
Add table
Add a link
Reference in a new issue