/** * @module tests/unit/utils/metadataIndex-array-bound * @description THE INDEXABLE-ARRAY BOUND — a law with a name and a refusal, * not a `continue`. * * THE DEFECT. An array-valued metadata field indexes one posting per element, * so the index has always carried a ceiling. It was 10, and it was applied by a * bare `continue` deep inside field extraction: * * if (Array.isArray(value) && value.length > 10) continue * * A row whose `tags` array held ELEVEN entries therefore had that field skipped * entirely — no posting, no error, no warning. The row then failed to match * every filtered search on `tags`, including `{ tags: 'a-tag-it-really-has' }`, * and the caller had no way to tell that from "no row matches". Eleven tags is * not an exotic shape; the eleventh tag made the row invisible. * * THE LAW. Arrays of scalars index up to {@link MAX_INDEXED_ARRAY_LENGTH}, * hardcoded (the zero-config law: no knob), which clears every legitimate * multi-value field — tags, authors, keyword lists — and stays below the * narrowest embedding this engine meets (384 dimensions). Above it the WRITE * IS REFUSED by name — `MetadataArrayTooLargeError`, carrying the field, the * length and the bound — at `add`, `update`, `relate` and `updateRelation` * alike. Nothing is skipped in silence. * * THE ONE PLACE THE BOUND STILL SKIPS is a row already on disk, written by an * older engine under the old rule and read back by a rebuild, a catch-up fold * or a remove. Refusing there would make an existing store un-rebuildable — so * the row is admitted and the skipped field is NARRATED. Both sides are pinned. */ import { describe, it, expect, beforeEach, vi, afterEach } from 'vitest' import { Brainy } from '../../../src/brainy' import { NounType, VerbType } from '../../../src/types/graphTypes' import { MetadataArrayTooLargeError, MAX_INDEXED_ARRAY_LENGTH } from '../../../src/errors/brainyError' import { resolveEntityId } from '../../../src/utils/idNormalization' import { prodLog } from '../../../src/utils/logger' /** `n` distinct scalar tags. */ function tags(n: number, prefix = 't'): string[] { return Array.from({ length: n }, (_, i) => `${prefix}${i}`) } describe('the indexable-array bound', () => { let brain: Brainy beforeEach(async () => { brain = new Brainy({ requireSubtype: false, storage: { type: 'memory' } }) await brain.init() }) afterEach(async () => { await brain.close() }) describe('BELOW the bound: the array indexes, every element of it', () => { it('the eleven-element array that used to vanish is searchable', async () => { // ELEVEN — one over the old silent limit, the whole shape of the defect. await brain.add({ id: 'eleven', data: 'a row with eleven tags', type: NounType.Document, metadata: { tags: tags(11) }, vector: [] }) // Every element is a posting, including the eleventh. for (const tag of tags(11)) { const hits = await brain.find({ where: { tags: tag }, limit: 10 } as any) expect(hits.map((r: any) => r.id)).toContain(resolveEntityId('eleven')) } }) it('indexes right up to the bound — every element of it', async () => { await brain.add({ id: 'at-bound', data: 'a row at the bound', type: NounType.Document, metadata: { tags: tags(MAX_INDEXED_ARRAY_LENGTH) }, vector: [] }) // The first, the last, and one in the middle — all derived from the // bound, so the case follows the constant wherever it moves. for (const tag of ['t0', `t${MAX_INDEXED_ARRAY_LENGTH - 1}`, `t${Math.floor(MAX_INDEXED_ARRAY_LENGTH / 2)}`]) { const hits = await brain.find({ where: { tags: tag }, limit: 10 } as any) expect(hits.map((r: any) => r.id)).toContain(resolveEntityId('at-bound')) } }) it('a nested bag\'s array indexes under its dotted address', async () => { await brain.add({ id: 'nested', data: 'a row with a nested tag list', type: NounType.Document, metadata: { facets: { labels: tags(20, 'l') } }, vector: [] }) const hits = await brain.find({ where: { 'facets.labels': 'l19' }, limit: 10 } as any) expect(hits.map((r: any) => r.id)).toContain(resolveEntityId('nested')) }) }) describe('ABOVE the bound: the write is refused, by name', () => { const OVER = MAX_INDEXED_ARRAY_LENGTH + 1 it('add() throws a typed error naming the field, the length and the bound', async () => { const err = await brain .add({ id: 'too-many', data: 'a row with too many tags', type: NounType.Document, metadata: { tags: tags(OVER) }, vector: [] } as any) .catch((e: any) => e) expect(err).toBeInstanceOf(MetadataArrayTooLargeError) expect(err.field).toBe('tags') expect(err.length).toBe(OVER) expect(err.limit).toBe(MAX_INDEXED_ARRAY_LENGTH) expect(err.type).toBe('VALIDATION') // The message carries all three, and names the cures. expect(err.message).toContain('tags') expect(err.message).toContain(String(OVER)) expect(err.message).toContain(String(MAX_INDEXED_ARRAY_LENGTH)) expect(err.message).toContain('vector') }) it('the refused row is not written at all — no half-indexed ghost', async () => { await expect( brain.add({ id: 'refused', data: 'refused', type: NounType.Document, metadata: { tags: tags(OVER) }, vector: [] } as any) ).rejects.toBeInstanceOf(MetadataArrayTooLargeError) expect(await brain.get('refused')).toBeNull() const hits = await brain.find({ where: { tags: 't0' }, limit: 10 } as any) expect(hits.map((r: any) => r.id)).not.toContain(resolveEntityId('refused')) }) it('a 384-float embedding parked in the metadata bag is refused, not swallowed', async () => { const err = await brain .add({ id: 'bag-vector', data: 'an embedding in the wrong place', type: NounType.Document, metadata: { embedding: Array.from({ length: 384 }, (_, i) => i / 384) }, vector: [] } as any) .catch((e: any) => e) expect(err).toBeInstanceOf(MetadataArrayTooLargeError) expect(err.field).toBe('embedding') expect(err.length).toBe(384) }) it('update() refuses it too', async () => { await brain.add({ id: 'grow', data: 'starts small', type: NounType.Document, metadata: { tags: tags(3) }, vector: [] }) await expect( brain.update({ id: 'grow', metadata: { tags: tags(OVER) } } as any) ).rejects.toBeInstanceOf(MetadataArrayTooLargeError) // And the row keeps the values it had. const hits = await brain.find({ where: { tags: 't1' }, limit: 10 } as any) expect(hits.map((r: any) => r.id)).toContain(resolveEntityId('grow')) }) it('relate() refuses it on a verb\'s metadata', async () => { await brain.add({ id: 'a', data: 'a', type: NounType.Thing, vector: [] }) await brain.add({ id: 'b', data: 'b', type: NounType.Thing, vector: [] }) await expect( brain.relate({ from: 'a', to: 'b', type: VerbType.RelatedTo, metadata: { tags: tags(OVER) } } as any) ).rejects.toBeInstanceOf(MetadataArrayTooLargeError) }) it('a nested oversize array is refused under its dotted address', async () => { const err = await brain .add({ id: 'nested-over', data: 'nested and too long', type: NounType.Document, metadata: { facets: { labels: tags(OVER, 'l') } }, vector: [] } as any) .catch((e: any) => e) expect(err).toBeInstanceOf(MetadataArrayTooLargeError) expect(err.field).toBe('facets.labels') }) }) describe('a row already on disk is admitted, and the skip is NARRATED', () => { afterEach(() => { vi.restoreAllMocks() }) it('extraction over an old oversize row warns by field, length and bound', async () => { const warn = vi.spyOn(prodLog, 'warn').mockImplementation(() => {}) const index = (brain as any).metadataIndex // The shape an older engine persisted: the write door never saw it, so // this reaches extraction directly — exactly as a rebuild or a remove // reading the row back would. const fields = index.extractIndexableFields({ metadata: { tags: tags(MAX_INDEXED_ARRAY_LENGTH + 5), keep: 'me' } }) // The oversize field contributes nothing... expect(fields.filter((f: any) => f.field === 'tags')).toHaveLength(0) // ...the rest of the row indexes normally — the row is not rejected... expect(fields.some((f: any) => f.field === 'keep' && f.value === 'me')).toBe(true) // ...and the skip is said out loud, with everything needed to act on it. expect(warn).toHaveBeenCalled() const said = warn.mock.calls.map((c: any[]) => String(c[0])).join('\n') expect(said).toContain('tags') expect(said).toContain(String(MAX_INDEXED_ARRAY_LENGTH + 5)) expect(said).toContain(String(MAX_INDEXED_ARRAY_LENGTH)) expect(said).toContain('NOT indexed') }) it('an at-bound row on disk is indexed in full and says nothing', async () => { const warn = vi.spyOn(prodLog, 'warn').mockImplementation(() => {}) const index = (brain as any).metadataIndex const fields = index.extractIndexableFields({ metadata: { tags: tags(MAX_INDEXED_ARRAY_LENGTH) } }) expect(fields.filter((f: any) => f.field === 'tags')).toHaveLength(MAX_INDEXED_ARRAY_LENGTH) const said = warn.mock.calls.map((c: any[]) => String(c[0])).join('\n') expect(said).not.toContain('indexing bound') }) }) })