69 lines
2.7 KiB
TypeScript
69 lines
2.7 KiB
TypeScript
|
|
/**
|
||
|
|
* @module tests/unit/brainy/similar-threshold
|
||
|
|
* @description Pins `brain.similar({ threshold })` — the min-similarity filter.
|
||
|
|
*
|
||
|
|
* Before 8.0 `similar()` accepted a `threshold` but silently DROPPED it (it was
|
||
|
|
* never forwarded to the query, so callers got unfiltered results). 8.0 applies
|
||
|
|
* it as a post-filter on `result.score` — the canonical way to impose a minimum
|
||
|
|
* score on plain semantic results (top-level vector search does not honor a
|
||
|
|
* `threshold`; see the `find({ near })` guidance). These tests use explicit
|
||
|
|
* vectors so the embedder is never invoked.
|
||
|
|
*/
|
||
|
|
import { describe, it, expect, afterEach } from 'vitest'
|
||
|
|
import { Brainy } from '../../../src/brainy.js'
|
||
|
|
import { NounType } from '../../../src/types/graphTypes.js'
|
||
|
|
|
||
|
|
/** Deterministic 384-dim vectors — no embedder, distinct per seed. */
|
||
|
|
function vec(seed: number): number[] {
|
||
|
|
return Array.from({ length: 384 }, (_, i) => ((seed * 31 + i * 13) % 100) / 100)
|
||
|
|
}
|
||
|
|
|
||
|
|
describe('brain.similar() — threshold post-filter (8.0)', () => {
|
||
|
|
const brains: Brainy[] = []
|
||
|
|
afterEach(async () => {
|
||
|
|
for (const b of brains.splice(0)) await b.close().catch(() => {})
|
||
|
|
})
|
||
|
|
|
||
|
|
it('honors the min-similarity threshold (was silently dropped before 8.0)', async () => {
|
||
|
|
const brain = new Brainy({ requireSubtype: false, storage: { type: 'memory' } })
|
||
|
|
brains.push(brain)
|
||
|
|
await brain.init()
|
||
|
|
|
||
|
|
for (let i = 0; i < 12; i++) {
|
||
|
|
await brain.add({ data: `e${i}`, type: NounType.Thing, vector: vec(i) })
|
||
|
|
}
|
||
|
|
|
||
|
|
const target = vec(0)
|
||
|
|
const all = await brain.similar({ to: target, limit: 100 })
|
||
|
|
expect(all.length).toBe(12)
|
||
|
|
|
||
|
|
const scores = all.map((r) => r.score)
|
||
|
|
const min = Math.min(...scores)
|
||
|
|
const max = Math.max(...scores)
|
||
|
|
// The corpus has a real score spread (one vector is identical to the target).
|
||
|
|
expect(max).toBeGreaterThan(min)
|
||
|
|
|
||
|
|
const threshold = (min + max) / 2
|
||
|
|
const filtered = await brain.similar({ to: target, limit: 100, threshold })
|
||
|
|
|
||
|
|
// THE invariant the fix guarantees: every result meets the threshold.
|
||
|
|
expect(filtered.every((r) => r.score >= threshold)).toBe(true)
|
||
|
|
// The threshold is actually applied — weaker matches dropped, strong kept.
|
||
|
|
expect(filtered.length).toBeGreaterThan(0)
|
||
|
|
expect(filtered.length).toBeLessThan(all.length)
|
||
|
|
})
|
||
|
|
|
||
|
|
it('returns the full set when no threshold is given (unchanged behavior)', async () => {
|
||
|
|
const brain = new Brainy({ requireSubtype: false, storage: { type: 'memory' } })
|
||
|
|
brains.push(brain)
|
||
|
|
await brain.init()
|
||
|
|
|
||
|
|
for (let i = 0; i < 6; i++) {
|
||
|
|
await brain.add({ data: `n${i}`, type: NounType.Thing, vector: vec(i + 50) })
|
||
|
|
}
|
||
|
|
|
||
|
|
const all = await brain.similar({ to: vec(50), limit: 100 })
|
||
|
|
expect(all.length).toBe(6)
|
||
|
|
})
|
||
|
|
})
|