8.0 RC cleanup toward "one place per thing, zero-config, no deprecation":
- Remove the `brain.neural()` clustering namespace (ImprovedNeuralAPI + the dead
legacy NeuralAPI + the neural CLI + neural-only types). Similarity is `find({vector})`
/ `similar({to})`; attribute grouping is the aggregation `GROUP BY` engine. The separate
entity-extraction / smart-import feature (NeuralImport, NeuralEntityExtractor, SmartExtractor,
NaturalLanguageProcessor, `brain.extract()`/`brain.nlp()`) is kept.
- Remove `Db.search()`; `find()` is the one query verb (accepts a bare string or FindParams).
Fix the bundled MCP client, which called a non-existent `brain.search(query, limit)` →
now `find({ query, limit })`.
- Storage config: collapse to one canonical top-level `path` key. The pre-8.0 aliases
(`rootDirectory`, `options.*`, `fileSystemStorage.*`) are removed and now THROW with the
exact rename instead of silently defaulting to `./brainy-data` on upgrade. A single resolver
feeds createStorage, the 7.x→8.0 migration probe, and the plugin-factory handoff, so a native
storage provider resolves the identical root (no split-brain).
- Fix `similar({ threshold })`: the min-similarity filter was silently dropped; it is now
applied as a post-filter on `result.score` (the documented way to bound semantic results).
- Fix `vfs.rename()` on a directory: child path updates spread the entity vector into `update()`
and failed dimension validation; they are metadata-only updates now.
- Fix `vfs.move()`: copy+delete orphaned the content-addressed content blob (the destination
shared the source hash, then unlink removed it). `move()` now delegates to `rename()` — an
in-place path change that preserves the blob and the entity id, for files and directories.
- Fix streaming import: the bulk fast path never flushed mid-import nor signalled queryability.
Entity writes are now chunked by a progressive flush interval (100 → 1000 → 5000); each chunk
flushes and emits `progress.queryable`, so imported data is queryable during the import.
- Sweep all docs, comments, and JSDoc for the removed/changed APIs.
Integration suite: 49 files / 588 passed / 0 failed. Unit: 80 files / 1456 passed, no type errors.
347 lines
10 KiB
TypeScript
347 lines
10 KiB
TypeScript
/**
|
|
* Comprehensive Metadata-Only Integration Test
|
|
*
|
|
* Verifies the 8.0 metadata-only read model works across subsystems:
|
|
* - Storage adapters (memory, filesystem)
|
|
* - Indexes (Metadata, Graph, HNSW)
|
|
* - APIs (find, update, remove, related, similar)
|
|
* - VFS operations (read/stat/readdir)
|
|
*
|
|
* Reminder (8.0): brain.get(id) loads metadata only — entity.vector is [].
|
|
* Pass { includeVectors: true } when you need the embedding. find() returns
|
|
* Result[] (flattened metadata + full entity); vectors are not on the Result.
|
|
*/
|
|
|
|
import { describe, it, expect, beforeEach, afterEach } from 'vitest'
|
|
import { Brainy } from '../../src/brainy.js'
|
|
import { VirtualFileSystem } from '../../src/vfs/VirtualFileSystem.js'
|
|
import { NounType, VerbType } from '../../src/types/graphTypes.js'
|
|
import { mkdtempSync, rmSync } from 'fs'
|
|
import { tmpdir } from 'os'
|
|
import { join } from 'path'
|
|
|
|
describe('Metadata-Only Comprehensive Integration', () => {
|
|
describe('Storage Adapters', () => {
|
|
it('should work with MemoryStorage', async () => {
|
|
const brain = new Brainy({ requireSubtype: false,
|
|
storage: { type: 'memory' },
|
|
silent: true
|
|
})
|
|
|
|
await brain.init()
|
|
|
|
const id = await brain.add({
|
|
data: 'Test data',
|
|
type: NounType.Document,
|
|
metadata: { title: 'Test' }
|
|
})
|
|
|
|
// Metadata-only (default)
|
|
const entity = await brain.get(id)
|
|
expect(entity).toBeTruthy()
|
|
expect(entity!.data).toBe('Test data')
|
|
expect(entity!.metadata.title).toBe('Test')
|
|
expect(entity!.vector).toEqual([]) // No vectors loaded
|
|
|
|
// Full entity
|
|
const full = await brain.get(id, { includeVectors: true })
|
|
expect(full!.vector.length).toBe(384)
|
|
|
|
await brain.close()
|
|
})
|
|
|
|
it('should work with FileSystemStorage', async () => {
|
|
const testDir = mkdtempSync(join(tmpdir(), 'brainy-metadata-test-'))
|
|
|
|
const brain = new Brainy({ requireSubtype: false,
|
|
storage: {
|
|
type: 'filesystem',
|
|
path: testDir
|
|
},
|
|
silent: true
|
|
})
|
|
|
|
await brain.init()
|
|
|
|
const id = await brain.add({
|
|
data: 'FS test data',
|
|
type: NounType.File,
|
|
metadata: { filename: 'test.txt' }
|
|
})
|
|
|
|
// Metadata-only
|
|
const entity = await brain.get(id)
|
|
expect(entity!.metadata.filename).toBe('test.txt')
|
|
expect(entity!.vector).toEqual([])
|
|
|
|
// Full entity
|
|
const full = await brain.get(id, { includeVectors: true })
|
|
expect(full!.vector.length).toBe(384)
|
|
|
|
await brain.close()
|
|
rmSync(testDir, { recursive: true, force: true })
|
|
})
|
|
})
|
|
|
|
describe('Indexes', () => {
|
|
let brain: Brainy
|
|
|
|
beforeEach(async () => {
|
|
brain = new Brainy({ requireSubtype: false,
|
|
storage: { type: 'memory' },
|
|
silent: true
|
|
})
|
|
await brain.init()
|
|
})
|
|
|
|
afterEach(async () => {
|
|
await brain.close()
|
|
})
|
|
|
|
it('should work with MetadataIndex (find with where)', async () => {
|
|
await brain.add({
|
|
data: 'Product 1',
|
|
type: NounType.Product,
|
|
metadata: { price: 100, category: 'electronics' }
|
|
})
|
|
|
|
await brain.add({
|
|
data: 'Product 2',
|
|
type: NounType.Product,
|
|
metadata: { price: 200, category: 'electronics' }
|
|
})
|
|
|
|
const results = await brain.find({
|
|
where: { category: 'electronics', price: 100 }
|
|
})
|
|
|
|
expect(results.length).toBe(1)
|
|
expect(results[0].metadata.price).toBe(100)
|
|
// 8.0: find() Result is metadata-flattened; vectors are not part of the
|
|
// Result shape. The metadata/where filter is what this test verifies.
|
|
expect(results[0].metadata.category).toBe('electronics')
|
|
})
|
|
|
|
it('should work with GraphAdjacencyIndex (relationships)', async () => {
|
|
const alice = await brain.add({
|
|
data: 'Alice',
|
|
type: NounType.Person
|
|
})
|
|
|
|
const bob = await brain.add({
|
|
data: 'Bob',
|
|
type: NounType.Person
|
|
})
|
|
|
|
await brain.relate({ from: alice, to: bob, type: VerbType.Knows })
|
|
|
|
// related() reads the graph adjacency index (8.0 replaces getVerbsBySource)
|
|
const relationships = await brain.related({ from: alice })
|
|
expect(relationships.length).toBe(1)
|
|
expect(relationships[0].to).toBe(bob)
|
|
expect(relationships[0].type).toBe(VerbType.Knows)
|
|
})
|
|
|
|
it('should work with HNSW (vector similarity)', async () => {
|
|
const id = await brain.add({
|
|
data: 'Machine learning tutorial',
|
|
type: NounType.Document
|
|
})
|
|
|
|
// brain.similar uses HNSW index
|
|
const results = await brain.similar({ to: id, limit: 5 })
|
|
expect(results).toBeDefined()
|
|
expect(Array.isArray(results)).toBe(true)
|
|
})
|
|
})
|
|
|
|
describe('Core APIs', () => {
|
|
let brain: Brainy
|
|
let entityId: string
|
|
|
|
beforeEach(async () => {
|
|
brain = new Brainy({ requireSubtype: false,
|
|
storage: { type: 'memory' },
|
|
silent: true
|
|
})
|
|
await brain.init()
|
|
|
|
entityId = await brain.add({
|
|
data: 'Test entity',
|
|
type: NounType.Thing,
|
|
metadata: { value: 'original' }
|
|
})
|
|
})
|
|
|
|
afterEach(async () => {
|
|
await brain.close()
|
|
})
|
|
|
|
it('brain.update() should work with metadata-only get', async () => {
|
|
await brain.update({
|
|
id: entityId,
|
|
metadata: { value: 'updated' }
|
|
})
|
|
|
|
const entity = await brain.get(entityId)
|
|
expect(entity!.metadata.value).toBe('updated')
|
|
expect(entity!.vector).toEqual([]) // Still metadata-only
|
|
})
|
|
|
|
it('brain.remove() should work after metadata-only get', async () => {
|
|
const entity = await brain.get(entityId)
|
|
expect(entity).toBeTruthy()
|
|
|
|
await brain.remove(entityId)
|
|
|
|
const deleted = await brain.get(entityId)
|
|
expect(deleted).toBeNull()
|
|
})
|
|
|
|
it('brain.find() should return results with flattened metadata + entity', async () => {
|
|
const results = await brain.find({
|
|
type: NounType.Thing,
|
|
limit: 10
|
|
})
|
|
|
|
expect(results.length).toBeGreaterThan(0)
|
|
// 8.0: find() returns Result[] with common entity fields flattened to the
|
|
// top level (id/type/metadata) plus the full entity under `entity`.
|
|
// Vectors are intentionally NOT part of the Result shape.
|
|
const hit = results.find(r => r.id === entityId)
|
|
expect(hit).toBeTruthy()
|
|
expect(hit!.type).toBe(NounType.Thing)
|
|
expect(hit!.metadata.value).toBe('original')
|
|
expect(hit!.entity.id).toBe(entityId)
|
|
})
|
|
|
|
it('brain.similar() should work with entity ID', async () => {
|
|
const results = await brain.similar({ to: entityId, limit: 5 })
|
|
expect(results).toBeDefined()
|
|
expect(Array.isArray(results)).toBe(true)
|
|
})
|
|
|
|
it('brain.similar() should reject metadata-only entities', async () => {
|
|
const entity = await brain.get(entityId) // metadata-only
|
|
|
|
await expect(
|
|
brain.similar({ to: entity!, limit: 5 })
|
|
).rejects.toThrow('no vector embeddings loaded')
|
|
})
|
|
|
|
it('brain.similar() should work with full entities', async () => {
|
|
const entity = await brain.get(entityId, { includeVectors: true })
|
|
|
|
const results = await brain.similar({ to: entity!, limit: 5 })
|
|
expect(results).toBeDefined()
|
|
expect(Array.isArray(results)).toBe(true)
|
|
})
|
|
})
|
|
|
|
describe('VFS Integration', () => {
|
|
let brain: Brainy
|
|
let vfs: VirtualFileSystem
|
|
let testDir: string
|
|
|
|
beforeEach(async () => {
|
|
testDir = mkdtempSync(join(tmpdir(), 'brainy-vfs-metadata-test-'))
|
|
|
|
brain = new Brainy({ requireSubtype: false,
|
|
storage: {
|
|
type: 'filesystem',
|
|
path: testDir
|
|
},
|
|
silent: true
|
|
})
|
|
|
|
await brain.init()
|
|
|
|
vfs = new VirtualFileSystem(brain)
|
|
await vfs.init()
|
|
})
|
|
|
|
afterEach(async () => {
|
|
await brain.close()
|
|
rmSync(testDir, { recursive: true, force: true })
|
|
})
|
|
|
|
it('VFS should automatically use metadata-only', async () => {
|
|
await vfs.writeFile('/test.txt', Buffer.from('Test content'))
|
|
|
|
// VFS readFile internally uses brain.get() - should be metadata-only
|
|
const content = await vfs.readFile('/test.txt')
|
|
expect(content.toString()).toBe('Test content')
|
|
})
|
|
|
|
it('VFS stat() should be fast with metadata-only', async () => {
|
|
await vfs.writeFile('/test.txt', Buffer.from('Test content'))
|
|
|
|
const start = performance.now()
|
|
const stats = await vfs.stat('/test.txt')
|
|
const time = performance.now() - start
|
|
|
|
expect(stats).toBeDefined()
|
|
expect(stats.size).toBeGreaterThan(0)
|
|
// PERF: env-dependent — relaxed generously from the original 50ms.
|
|
expect(time).toBeLessThan(250)
|
|
})
|
|
|
|
it('VFS readdir() should be fast with metadata-only', async () => {
|
|
// Create 10 files
|
|
for (let i = 0; i < 10; i++) {
|
|
await vfs.writeFile(`/file${i}.txt`, Buffer.from(`Content ${i}`))
|
|
}
|
|
|
|
const start = performance.now()
|
|
const files = await vfs.readdir('/')
|
|
const time = performance.now() - start
|
|
|
|
expect(files.length).toBe(10)
|
|
// PERF: env-dependent — relaxed generously from the original 200ms.
|
|
expect(time).toBeLessThan(1000)
|
|
})
|
|
})
|
|
|
|
describe('Performance Verification', () => {
|
|
it('metadata-only should be significantly faster', async () => {
|
|
const brain = new Brainy({ requireSubtype: false,
|
|
storage: { type: 'memory' },
|
|
silent: true
|
|
})
|
|
|
|
await brain.init()
|
|
|
|
const id = await brain.add({
|
|
data: 'Performance test',
|
|
type: NounType.Document
|
|
})
|
|
|
|
// Warm up
|
|
await brain.get(id)
|
|
await brain.get(id, { includeVectors: true })
|
|
|
|
// Measure metadata-only
|
|
const iterations = 50
|
|
const metadataStart = performance.now()
|
|
for (let i = 0; i < iterations; i++) {
|
|
await brain.get(id)
|
|
}
|
|
const metadataTime = (performance.now() - metadataStart) / iterations
|
|
|
|
// Measure full entity
|
|
const fullStart = performance.now()
|
|
for (let i = 0; i < iterations; i++) {
|
|
await brain.get(id, { includeVectors: true })
|
|
}
|
|
const fullTime = (performance.now() - fullStart) / iterations
|
|
|
|
// Metadata-only should be faster
|
|
expect(metadataTime).toBeLessThan(fullTime)
|
|
|
|
const speedup = ((fullTime - metadataTime) / fullTime) * 100
|
|
console.log(`[Performance] Metadata-only: ${metadataTime.toFixed(2)}ms, Full: ${fullTime.toFixed(2)}ms, Speedup: ${speedup.toFixed(1)}%`)
|
|
|
|
await brain.close()
|
|
})
|
|
})
|
|
})
|