Each file opened one or more Brainy instances (beforeEach, or a small per-test helper like migration-gate-family-scoped's module-level seed()) and never closed them. migration-gate-family-scoped.test.ts now tracks every brain seed() hands back in a describe-scoped array drained by afterEach, since the helper itself lives outside the describe block.
248 lines
9.8 KiB
TypeScript
248 lines
9.8 KiB
TypeScript
/**
|
|
* @module tests/unit/utils/metadataIndex-array-bound
|
|
* @description THE INDEXABLE-ARRAY BOUND — a law with a name and a refusal,
|
|
* not a `continue`.
|
|
*
|
|
* THE DEFECT. An array-valued metadata field indexes one posting per element,
|
|
* so the index has always carried a ceiling. It was 10, and it was applied by a
|
|
* bare `continue` deep inside field extraction:
|
|
*
|
|
* if (Array.isArray(value) && value.length > 10) continue
|
|
*
|
|
* A row whose `tags` array held ELEVEN entries therefore had that field skipped
|
|
* entirely — no posting, no error, no warning. The row then failed to match
|
|
* every filtered search on `tags`, including `{ tags: 'a-tag-it-really-has' }`,
|
|
* and the caller had no way to tell that from "no row matches". Eleven tags is
|
|
* not an exotic shape; the eleventh tag made the row invisible.
|
|
*
|
|
* THE LAW. Arrays of scalars index up to {@link MAX_INDEXED_ARRAY_LENGTH},
|
|
* hardcoded (the zero-config law: no knob), which clears every legitimate
|
|
* multi-value field — tags, authors, keyword lists — and stays below the
|
|
* narrowest embedding this engine meets (384 dimensions). Above it the WRITE
|
|
* IS REFUSED by name — `MetadataArrayTooLargeError`, carrying the field, the
|
|
* length and the bound — at `add`, `update`, `relate` and `updateRelation`
|
|
* alike. Nothing is skipped in silence.
|
|
*
|
|
* THE ONE PLACE THE BOUND STILL SKIPS is a row already on disk, written by an
|
|
* older engine under the old rule and read back by a rebuild, a catch-up fold
|
|
* or a remove. Refusing there would make an existing store un-rebuildable — so
|
|
* the row is admitted and the skipped field is NARRATED. Both sides are pinned.
|
|
*/
|
|
import { describe, it, expect, beforeEach, vi, afterEach } from 'vitest'
|
|
import { Brainy } from '../../../src/brainy'
|
|
import { NounType, VerbType } from '../../../src/types/graphTypes'
|
|
import { MetadataArrayTooLargeError, MAX_INDEXED_ARRAY_LENGTH } from '../../../src/errors/brainyError'
|
|
import { resolveEntityId } from '../../../src/utils/idNormalization'
|
|
import { prodLog } from '../../../src/utils/logger'
|
|
|
|
/** `n` distinct scalar tags. */
|
|
function tags(n: number, prefix = 't'): string[] {
|
|
return Array.from({ length: n }, (_, i) => `${prefix}${i}`)
|
|
}
|
|
|
|
describe('the indexable-array bound', () => {
|
|
let brain: Brainy<any>
|
|
|
|
beforeEach(async () => {
|
|
brain = new Brainy({ requireSubtype: false, storage: { type: 'memory' } })
|
|
await brain.init()
|
|
})
|
|
|
|
afterEach(async () => {
|
|
await brain.close()
|
|
})
|
|
|
|
describe('BELOW the bound: the array indexes, every element of it', () => {
|
|
it('the eleven-element array that used to vanish is searchable', async () => {
|
|
// ELEVEN — one over the old silent limit, the whole shape of the defect.
|
|
await brain.add({
|
|
id: 'eleven',
|
|
data: 'a row with eleven tags',
|
|
type: NounType.Document,
|
|
metadata: { tags: tags(11) },
|
|
vector: []
|
|
})
|
|
|
|
// Every element is a posting, including the eleventh.
|
|
for (const tag of tags(11)) {
|
|
const hits = await brain.find({ where: { tags: tag }, limit: 10 } as any)
|
|
expect(hits.map((r: any) => r.id)).toContain(resolveEntityId('eleven'))
|
|
}
|
|
})
|
|
|
|
it('indexes right up to the bound — every element of it', async () => {
|
|
await brain.add({
|
|
id: 'at-bound',
|
|
data: 'a row at the bound',
|
|
type: NounType.Document,
|
|
metadata: { tags: tags(MAX_INDEXED_ARRAY_LENGTH) },
|
|
vector: []
|
|
})
|
|
|
|
// The first, the last, and one in the middle — all derived from the
|
|
// bound, so the case follows the constant wherever it moves.
|
|
for (const tag of ['t0', `t${MAX_INDEXED_ARRAY_LENGTH - 1}`, `t${Math.floor(MAX_INDEXED_ARRAY_LENGTH / 2)}`]) {
|
|
const hits = await brain.find({ where: { tags: tag }, limit: 10 } as any)
|
|
expect(hits.map((r: any) => r.id)).toContain(resolveEntityId('at-bound'))
|
|
}
|
|
})
|
|
|
|
it('a nested bag\'s array indexes under its dotted address', async () => {
|
|
await brain.add({
|
|
id: 'nested',
|
|
data: 'a row with a nested tag list',
|
|
type: NounType.Document,
|
|
metadata: { facets: { labels: tags(20, 'l') } },
|
|
vector: []
|
|
})
|
|
const hits = await brain.find({ where: { 'facets.labels': 'l19' }, limit: 10 } as any)
|
|
expect(hits.map((r: any) => r.id)).toContain(resolveEntityId('nested'))
|
|
})
|
|
})
|
|
|
|
describe('ABOVE the bound: the write is refused, by name', () => {
|
|
const OVER = MAX_INDEXED_ARRAY_LENGTH + 1
|
|
|
|
it('add() throws a typed error naming the field, the length and the bound', async () => {
|
|
const err = await brain
|
|
.add({
|
|
id: 'too-many',
|
|
data: 'a row with too many tags',
|
|
type: NounType.Document,
|
|
metadata: { tags: tags(OVER) },
|
|
vector: []
|
|
} as any)
|
|
.catch((e: any) => e)
|
|
|
|
expect(err).toBeInstanceOf(MetadataArrayTooLargeError)
|
|
expect(err.field).toBe('tags')
|
|
expect(err.length).toBe(OVER)
|
|
expect(err.limit).toBe(MAX_INDEXED_ARRAY_LENGTH)
|
|
expect(err.type).toBe('VALIDATION')
|
|
// The message carries all three, and names the cures.
|
|
expect(err.message).toContain('tags')
|
|
expect(err.message).toContain(String(OVER))
|
|
expect(err.message).toContain(String(MAX_INDEXED_ARRAY_LENGTH))
|
|
expect(err.message).toContain('vector')
|
|
})
|
|
|
|
it('the refused row is not written at all — no half-indexed ghost', async () => {
|
|
await expect(
|
|
brain.add({
|
|
id: 'refused',
|
|
data: 'refused',
|
|
type: NounType.Document,
|
|
metadata: { tags: tags(OVER) },
|
|
vector: []
|
|
} as any)
|
|
).rejects.toBeInstanceOf(MetadataArrayTooLargeError)
|
|
|
|
expect(await brain.get('refused')).toBeNull()
|
|
const hits = await brain.find({ where: { tags: 't0' }, limit: 10 } as any)
|
|
expect(hits.map((r: any) => r.id)).not.toContain(resolveEntityId('refused'))
|
|
})
|
|
|
|
it('a 384-float embedding parked in the metadata bag is refused, not swallowed', async () => {
|
|
const err = await brain
|
|
.add({
|
|
id: 'bag-vector',
|
|
data: 'an embedding in the wrong place',
|
|
type: NounType.Document,
|
|
metadata: { embedding: Array.from({ length: 384 }, (_, i) => i / 384) },
|
|
vector: []
|
|
} as any)
|
|
.catch((e: any) => e)
|
|
|
|
expect(err).toBeInstanceOf(MetadataArrayTooLargeError)
|
|
expect(err.field).toBe('embedding')
|
|
expect(err.length).toBe(384)
|
|
})
|
|
|
|
it('update() refuses it too', async () => {
|
|
await brain.add({
|
|
id: 'grow',
|
|
data: 'starts small',
|
|
type: NounType.Document,
|
|
metadata: { tags: tags(3) },
|
|
vector: []
|
|
})
|
|
await expect(
|
|
brain.update({ id: 'grow', metadata: { tags: tags(OVER) } } as any)
|
|
).rejects.toBeInstanceOf(MetadataArrayTooLargeError)
|
|
|
|
// And the row keeps the values it had.
|
|
const hits = await brain.find({ where: { tags: 't1' }, limit: 10 } as any)
|
|
expect(hits.map((r: any) => r.id)).toContain(resolveEntityId('grow'))
|
|
})
|
|
|
|
it('relate() refuses it on a verb\'s metadata', async () => {
|
|
await brain.add({ id: 'a', data: 'a', type: NounType.Thing, vector: [] })
|
|
await brain.add({ id: 'b', data: 'b', type: NounType.Thing, vector: [] })
|
|
await expect(
|
|
brain.relate({
|
|
from: 'a',
|
|
to: 'b',
|
|
type: VerbType.RelatedTo,
|
|
metadata: { tags: tags(OVER) }
|
|
} as any)
|
|
).rejects.toBeInstanceOf(MetadataArrayTooLargeError)
|
|
})
|
|
|
|
it('a nested oversize array is refused under its dotted address', async () => {
|
|
const err = await brain
|
|
.add({
|
|
id: 'nested-over',
|
|
data: 'nested and too long',
|
|
type: NounType.Document,
|
|
metadata: { facets: { labels: tags(OVER, 'l') } },
|
|
vector: []
|
|
} as any)
|
|
.catch((e: any) => e)
|
|
expect(err).toBeInstanceOf(MetadataArrayTooLargeError)
|
|
expect(err.field).toBe('facets.labels')
|
|
})
|
|
})
|
|
|
|
describe('a row already on disk is admitted, and the skip is NARRATED', () => {
|
|
afterEach(() => {
|
|
vi.restoreAllMocks()
|
|
})
|
|
|
|
it('extraction over an old oversize row warns by field, length and bound', async () => {
|
|
const warn = vi.spyOn(prodLog, 'warn').mockImplementation(() => {})
|
|
const index = (brain as any).metadataIndex
|
|
|
|
// The shape an older engine persisted: the write door never saw it, so
|
|
// this reaches extraction directly — exactly as a rebuild or a remove
|
|
// reading the row back would.
|
|
const fields = index.extractIndexableFields({
|
|
metadata: { tags: tags(MAX_INDEXED_ARRAY_LENGTH + 5), keep: 'me' }
|
|
})
|
|
|
|
// The oversize field contributes nothing...
|
|
expect(fields.filter((f: any) => f.field === 'tags')).toHaveLength(0)
|
|
// ...the rest of the row indexes normally — the row is not rejected...
|
|
expect(fields.some((f: any) => f.field === 'keep' && f.value === 'me')).toBe(true)
|
|
// ...and the skip is said out loud, with everything needed to act on it.
|
|
expect(warn).toHaveBeenCalled()
|
|
const said = warn.mock.calls.map((c: any[]) => String(c[0])).join('\n')
|
|
expect(said).toContain('tags')
|
|
expect(said).toContain(String(MAX_INDEXED_ARRAY_LENGTH + 5))
|
|
expect(said).toContain(String(MAX_INDEXED_ARRAY_LENGTH))
|
|
expect(said).toContain('NOT indexed')
|
|
})
|
|
|
|
it('an at-bound row on disk is indexed in full and says nothing', async () => {
|
|
const warn = vi.spyOn(prodLog, 'warn').mockImplementation(() => {})
|
|
const index = (brain as any).metadataIndex
|
|
|
|
const fields = index.extractIndexableFields({
|
|
metadata: { tags: tags(MAX_INDEXED_ARRAY_LENGTH) }
|
|
})
|
|
expect(fields.filter((f: any) => f.field === 'tags')).toHaveLength(MAX_INDEXED_ARRAY_LENGTH)
|
|
|
|
const said = warn.mock.calls.map((c: any[]) => String(c[0])).join('\n')
|
|
expect(said).not.toContain('indexing bound')
|
|
})
|
|
})
|
|
})
|