brainy/tests/integration/storage-batch-operations.test.ts

549 lines
19 KiB
TypeScript
Raw Normal View History

/**
* Storage-Level Batch Operations Test Suite v5.12.0
*
* Comprehensive testing of new storage-level batch APIs:
* - storage.getNounMetadataBatch() - Batch metadata reads
* - storage.readBatchWithInheritance() - COW-aware batch reads
* - storage.getVerbsBySourceBatch() - Batch relationship queries
* - brain.batchGet() - High-level batch entity retrieval
* - PathResolver.getChildren() - VFS batch operations
*
* Coverage:
* Type-aware storage compatibility
* Sharding preservation
* COW (Copy-on-Write) integration
* fork() and branch isolation
* Performance improvements (N+1 batched)
* Cloud adapter native batch APIs
*/
import { describe, it, expect, beforeEach, afterEach } from 'vitest'
import { Brainy } from '../../src/brainy'
import { NounType, VerbType } from '../../src/coreTypes'
import { performance } from 'perf_hooks'
describe('Storage-Level Batch Operations v5.12.0', () => {
let brain: Brainy
beforeEach(async () => {
feat(8.0)!: flip requireSubtype default to true (BRAINY-8.0-SUBTYPE-CONTRACT § C-1) Brainy 8.0 makes subtype required by default on every public write path (`add`, `addMany`, `update`, `relate`, `relateMany`, `updateRelation`, import). Per the locked C-1 contract, every entity and relation gets a non-empty subtype string by the time the storage layer sees it. OPT-OUT REMAINS FULLY SUPPORTED The runtime flag is still consumer-controlled. Three opt-out paths cover migration / legacy fixtures / typed escape: - `new Brainy({ requireSubtype: false })` — last-resort: turn off the contract entirely. Recommended only for migration windows or test fixtures that legitimately can't supply a subtype. - `new Brainy({ requireSubtype: { except: [NounType.Thing, ...] } })` — per-type allowlist: strict everywhere except the listed types. - `brain.requireSubtype(type, options)` — per-type registration with optional vocabulary. Composes with the brain-wide flag. Default is now `true`. Opt-out is explicit and documented; nothing silently degrades. TEST SWEEP Bulk-applied `requireSubtype: false` to every `new Brainy({...})` call site across 120 test files. Three sed patterns covered the shapes: - `new Brainy({` → `new Brainy({ requireSubtype: false,` - `new Brainy<T>({` → `new Brainy<T>({ requireSubtype: false,` - `new Brainy()` → `new Brainy({ requireSubtype: false })` tests/helpers/test-factory.ts → createTestConfig() defaults `requireSubtype: false` so test files using the helper inherit the opt-out without per-site edits. The test sites that DO exercise subtype semantics (the subtype-and-facets suite, the strict-mode-self-test suite, the verb- subtype-and-enforcement suite, etc.) already pass real subtypes — they were the 7.30.x acceptance tests for this contract. Those tests continue to pass unchanged. CHANGES src/brainy.ts - normalizeConfig() — `requireSubtype` default `false` → `true`. Comment refreshed to document the three opt-out paths. tests/* (120 files) - Bulk-edited brain construction sites. No functional test changes; the opt-out preserves the test author's original intent. tests/helpers/test-factory.ts - createTestConfig() base config gains `requireSubtype: false`. NO-OP for consumers who were already passing subtype on every write. For consumers who weren't, the upgrade path is one of the three opt-out forms above. Migration recipe documented in 8.0 release notes (next commit). VERIFICATION - npx tsc --noEmit: clean - npm test: 1408 / 1409 (same pre-existing race-condition outstanding; no other regressions from the flip)
2026-06-09 14:58:25 -07:00
brain = new Brainy({ requireSubtype: false,
storage: { type: 'memory' },
enableCOW: true
})
await brain.init()
})
afterEach(async () => {
await brain.close()
})
describe('brain.batchGet() - High-Level Batch API', () => {
it('should batch fetch multiple entities (metadata-only)', async () => {
// Add test entities
const id1 = await brain.add({
type: 'document',
data: 'Entity 1',
metadata: { category: 'A' }
})
const id2 = await brain.add({
type: 'thing',
data: 'Entity 2',
metadata: { category: 'B' }
})
const id3 = await brain.add({
type: 'person',
data: 'Entity 3',
metadata: { category: 'C' }
})
// Batch fetch (metadata-only by default)
const results = await brain.batchGet([id1, id2, id3])
expect(results.size).toBe(3)
expect(results.get(id1)?.data).toBe('Entity 1')
expect(results.get(id2)?.data).toBe('Entity 2')
expect(results.get(id3)?.data).toBe('Entity 3')
// Vectors should NOT be included by default (empty array or undefined)
const vector = results.get(id1)?.vector
expect(vector === undefined || (Array.isArray(vector) && vector.length === 0)).toBe(true)
})
it('should handle missing entities gracefully', async () => {
const id1 = await brain.add({ type: 'document', data: 'Exists' })
const fakeId = '12345678-1234-1234-1234-123456789abc'
const anotherFake = '87654321-4321-4321-4321-abcdef123456'
const results = await brain.batchGet([id1, fakeId, anotherFake])
expect(results.size).toBe(1)
expect(results.get(id1)?.data).toBe('Exists')
expect(results.has(fakeId)).toBe(false)
})
it('should support includeVectors option (fallback)', async () => {
const id1 = await brain.add({
type: 'document',
data: 'With vector',
metadata: { test: true }
})
// With vectors (currently falls back to individual gets)
const results = await brain.batchGet([id1], { includeVectors: true })
expect(results.size).toBe(1)
const entity = results.get(id1)
expect(entity?.data).toBe('With vector')
expect(entity?.vector).toBeDefined()
expect(entity?.vector?.length).toBeGreaterThan(0)
})
it('should be faster than individual gets for large batches', async () => {
// Create 100 entities
const ids: string[] = []
for (let i = 0; i < 100; i++) {
const id = await brain.add({
type: 'document',
data: `Entity ${i}`,
metadata: { index: i }
})
ids.push(id)
}
// Measure individual gets
const startIndividual = performance.now()
for (const id of ids.slice(0, 20)) {
await brain.get(id)
}
const individualTime = performance.now() - startIndividual
// Measure batch get
const startBatch = performance.now()
await brain.batchGet(ids.slice(20, 40))
const batchTime = performance.now() - startBatch
// Batch should be faster (or at least comparable)
console.log(`Individual: ${individualTime.toFixed(2)}ms, Batch: ${batchTime.toFixed(2)}ms`)
expect(batchTime).toBeLessThan(individualTime * 2) // Allow some overhead
})
})
describe('storage.getNounMetadataBatch() - Storage Layer', () => {
it('should batch fetch noun metadata with type caching', async () => {
// Add entities of different types
const id1 = await brain.add({ type: 'document', data: 'Doc' })
const id2 = await brain.add({ type: 'thing', data: 'Thing' })
const id3 = await brain.add({ type: 'person', data: 'Person' })
// Access storage directly
const storage = brain.storage as any
const results = await storage.getNounMetadataBatch([id1, id2, id3])
expect(results.size).toBe(3)
expect(results.get(id1)?.noun).toBe('document')
expect(results.get(id2)?.noun).toBe('thing')
expect(results.get(id3)?.noun).toBe('person')
// Type cache should be populated
expect(storage.nounTypeCache.has(id1)).toBe(true)
expect(storage.nounTypeCache.get(id1)).toBe('document')
})
it('should handle uncached IDs by trying multiple types', async () => {
// Add entity
const id = await brain.add({ type: 'document', data: 'Test' })
// Clear type cache to simulate uncached scenario
const storage = brain.storage as any
storage.nounTypeCache.delete(id)
// Batch fetch should still work (tries all types)
const results = await storage.getNounMetadataBatch([id])
expect(results.size).toBe(1)
expect(results.get(id)?.noun).toBe('document')
// Cache should be repopulated (or may still be empty if metadata doesn't populate it)
// This is acceptable as long as the data is retrieved correctly
const cachedType = storage.nounTypeCache.get(id)
if (cachedType !== undefined) {
expect(cachedType).toBe('document')
}
})
it('should preserve sharding in all paths', async () => {
// Add entity
const id = await brain.add({ type: 'document', data: 'Sharded' })
// Check that path includes shard
const storage = brain.storage as any
const results = await storage.getNounMetadataBatch([id])
expect(results.size).toBe(1)
// Verify shard is in the path used (check internal call)
// Path should be: entities/nouns/document/metadata/{SHARD}/{ID}.json
const shard = storage.getShardIdFromUuid?.(id) || id.substring(0, 2)
expect(shard).toBeDefined()
})
it('should handle large batches efficiently', async () => {
// Create 500 entities
const ids: string[] = []
for (let i = 0; i < 500; i++) {
const id = await brain.add({
type: 'document',
data: `Batch ${i}`,
metadata: { batch: true }
})
ids.push(id)
}
const startTime = performance.now()
const storage = brain.storage as any
const results = await storage.getNounMetadataBatch(ids)
const duration = performance.now() - startTime
expect(results.size).toBe(500)
console.log(`Batched 500 metadata reads in ${duration.toFixed(2)}ms`)
// Should complete in reasonable time
expect(duration).toBeLessThan(5000) // < 5 seconds
})
})
describe('COW Integration - readBatchWithInheritance()', () => {
it('should resolve branch paths before reading', async () => {
// Add entity on main
const id = await brain.add({ type: 'document', data: 'Main branch' })
// Create fork
const fork = await brain.fork('test-branch')
// Add entity on fork
const forkId = await fork.add({ type: 'document', data: 'Fork branch' })
// Batch get on fork should see fork entity
const forkResults = await fork.batchGet([forkId, id])
expect(forkResults.size).toBe(2)
expect(forkResults.get(forkId)?.data).toBe('Fork branch')
expect(forkResults.get(id)?.data).toBe('Main branch') // Inherited
// Batch get on main should NOT see fork entity
const mainResults = await brain.batchGet([forkId, id])
expect(mainResults.size).toBe(1)
expect(mainResults.has(forkId)).toBe(false) // Not on main
expect(mainResults.get(id)?.data).toBe('Main branch')
})
it('should respect write cache for dirty entities', async () => {
// Add entity
const id = await brain.add({ type: 'document', data: 'Original' })
// Update (may be in write cache before flush)
await brain.update({ id, data: 'Updated' })
// Batch get should see updated version
const results = await brain.batchGet([id])
expect(results.get(id)?.data).toBe('Updated')
})
it('should inherit from parent commits for missing entities', async () => {
// Add entities on main
const id1 = await brain.add({ type: 'document', data: 'Main 1' })
const id2 = await brain.add({ type: 'document', data: 'Main 2' })
// Commit
await brain.commit('Initial entities')
// Create fork
const fork = await brain.fork('child-branch')
// Add new entity only on fork
const forkId = await fork.add({ type: 'document', data: 'Fork only' })
// Batch get on fork should inherit main entities
const results = await fork.batchGet([id1, id2, forkId])
expect(results.size).toBe(3)
expect(results.get(id1)?.data).toBe('Main 1') // Inherited
expect(results.get(id2)?.data).toBe('Main 2') // Inherited
expect(results.get(forkId)?.data).toBe('Fork only') // Fork's own
})
})
describe('getVerbsBySourceBatch() - Batch Relationship Queries', () => {
it('should batch fetch relationships by source IDs', async () => {
// Create entities
const source1 = await brain.add({ type: 'person', data: 'Alice' })
const source2 = await brain.add({ type: 'person', data: 'Bob' })
const target1 = await brain.add({ type: 'document', data: 'Doc1' })
const target2 = await brain.add({ type: 'document', data: 'Doc2' })
// Create relationships
await brain.relate({ from: source1, to: target1, type: 'creates' })
await brain.relate({ from: source1, to: target2, type: 'creates' })
await brain.relate({ from: source2, to: target1, type: 'uses' })
// Batch query
const storage = brain.storage as any
const results = await storage.getVerbsBySourceBatch([source1, source2])
expect(results.size).toBe(2)
const source1Verbs = results.get(source1) || []
const source2Verbs = results.get(source2) || []
expect(source1Verbs.length).toBe(2) // 2 relationships
expect(source2Verbs.length).toBe(1) // 1 relationship
// Check verb types
expect(source1Verbs.every((v: any) => v.verb === 'creates')).toBe(true)
expect(source2Verbs[0].verb).toBe('uses')
})
it('should filter by verb type', async () => {
const source = await brain.add({ type: 'person', data: 'User' })
const target1 = await brain.add({ type: 'document', data: 'Doc1' })
const target2 = await brain.add({ type: 'document', data: 'Doc2' })
// Multiple relationship types
await brain.relate({ from: source, to: target1, type: 'creates' })
await brain.relate({ from: source, to: target2, type: 'uses' })
const storage = brain.storage as any
// Filter by 'creates' type
const createsResults = await storage.getVerbsBySourceBatch(
[source],
'creates'
)
const createsVerbs = createsResults.get(source) || []
expect(createsVerbs.length).toBe(1)
expect(createsVerbs[0].verb).toBe('creates')
})
it('should handle sources with no relationships', async () => {
const source1 = await brain.add({ type: 'person', data: 'Isolated' })
const source2 = await brain.add({ type: 'person', data: 'Connected' })
const target = await brain.add({ type: 'document', data: 'Doc' })
await brain.relate({ from: source2, to: target, type: 'relatedTo' })
const storage = brain.storage as any
const results = await storage.getVerbsBySourceBatch([source1, source2])
expect(results.get(source1) || []).toHaveLength(0) // No relationships
expect(results.get(source2) || []).toHaveLength(1) // Has relationship
})
})
describe('VFS Integration - PathResolver.getChildren()', () => {
it('should use batchGet() for directory children', async () => {
if (!brain.vfs) {
await brain.vfs.init()
}
// Create directory with files
await brain.vfs!.mkdir('/batch-test')
await brain.vfs!.writeFile('/batch-test/file1.txt', 'Content 1')
await brain.vfs!.writeFile('/batch-test/file2.txt', 'Content 2')
await brain.vfs!.writeFile('/batch-test/file3.txt', 'Content 3')
// getChildren() should use batchGet() internally
const startTime = performance.now()
const tree = await brain.vfs!.getTreeStructure('/batch-test')
const duration = performance.now() - startTime
expect(tree.children).toHaveLength(3)
console.log(`VFS getTreeStructure with batch: ${duration.toFixed(2)}ms`)
// Verify all children loaded
const filenames = tree.children!.map(c => c.name).sort()
expect(filenames).toEqual(['file1.txt', 'file2.txt', 'file3.txt'])
})
it('should handle nested directories with parallel traversal', async () => {
if (!brain.vfs) {
await brain.vfs.init()
}
// Create nested structure
await brain.vfs!.mkdir('/root')
await brain.vfs!.mkdir('/root/dir1')
await brain.vfs!.mkdir('/root/dir2')
await brain.vfs!.writeFile('/root/dir1/a.txt', 'A')
await brain.vfs!.writeFile('/root/dir1/b.txt', 'B')
await brain.vfs!.writeFile('/root/dir2/c.txt', 'C')
// Should use breadth-first parallel traversal
const tree = await brain.vfs!.getTreeStructure('/root', { recursive: true })
expect(tree.children).toHaveLength(2) // 2 subdirectories
const dir1 = tree.children!.find(c => c.name === 'dir1')
const dir2 = tree.children!.find(c => c.name === 'dir2')
expect(dir1?.children).toHaveLength(2) // 2 files in dir1
expect(dir2?.children).toHaveLength(1) // 1 file in dir2
})
})
describe('Performance: N+1 Query Elimination', () => {
it('should eliminate N+1 pattern for directory with 12 files', async () => {
if (!brain.vfs) {
await brain.vfs.init()
}
// Create directory with 12 files (original bug scenario)
await brain.vfs!.mkdir('/performance-test')
for (let i = 1; i <= 12; i++) {
await brain.vfs!.writeFile(`/performance-test/file${i}.txt`, `Content ${i}`)
}
// Measure with batching
const startBatch = performance.now()
const treeBatch = await brain.vfs!.getTreeStructure('/performance-test')
const batchTime = performance.now() - startBatch
expect(treeBatch.children).toHaveLength(12)
console.log(`12 files with batching: ${batchTime.toFixed(2)}ms`)
// Before v5.12.0: ~12.7s (22 sequential calls × 580ms)
// After v5.12.0: <1s (2-3 batched calls)
expect(batchTime).toBeLessThan(2000) // Should be < 2 seconds
})
it('should scale to 100 entities efficiently', async () => {
// Create 100 entities
const ids: string[] = []
for (let i = 0; i < 100; i++) {
const id = await brain.add({
type: 'document',
data: `Entity ${i}`,
metadata: { index: i }
})
ids.push(id)
}
// Batch get all 100
const startTime = performance.now()
const results = await brain.batchGet(ids)
const duration = performance.now() - startTime
expect(results.size).toBe(100)
console.log(`100 entities batch: ${duration.toFixed(2)}ms (${(100 / duration * 1000).toFixed(0)} entities/sec)`)
// Should achieve high throughput
const throughput = 100 / duration * 1000
expect(throughput).toBeGreaterThan(50) // > 50 entities/sec
})
})
describe('Error Handling', () => {
it('should handle partial batch failures gracefully', async () => {
const id1 = await brain.add({ type: 'document', data: 'Exists' })
const fakeIds = [
'11111111-1111-1111-1111-111111111111',
'22222222-2222-2222-2222-222222222222',
'33333333-3333-3333-3333-333333333333'
]
// Mix of valid and invalid IDs
const results = await brain.batchGet([id1, ...fakeIds])
// Should return only valid entities
expect(results.size).toBe(1)
expect(results.get(id1)).toBeDefined()
// Invalid IDs should be silently skipped
fakeIds.forEach(fakeId => {
expect(results.has(fakeId)).toBe(false)
})
})
it('should handle empty batch gracefully', async () => {
const results = await brain.batchGet([])
expect(results.size).toBe(0)
})
it('should handle duplicate IDs in batch', async () => {
const id = await brain.add({ type: 'document', data: 'Duplicate test' })
// Same ID multiple times
const results = await brain.batchGet([id, id, id])
// Should return single entry
expect(results.size).toBe(1)
expect(results.get(id)?.data).toBe('Duplicate test')
})
})
describe('Type-Aware Storage Verification', () => {
it('should use correct type-first paths for all types', async () => {
// Create entities of each major type
const types: NounType[] = [
NounType.Document,
NounType.Thing,
NounType.Person,
NounType.File,
NounType.Event
]
const ids: string[] = []
for (const type of types) {
const id = await brain.add({
type: type as any,
data: `Type ${type}`,
metadata: { testType: type }
})
ids.push(id)
}
// Batch fetch
const results = await brain.batchGet(ids)
expect(results.size).toBe(types.length)
// Verify each entity has correct type
for (const [id, entity] of results) {
expect(entity.type).toBeDefined()
expect(types.includes(entity.type as NounType)).toBe(true)
}
})
})
describe('Sharding Verification', () => {
it('should maintain shard distribution in batch operations', async () => {
// Create entities with known shard distribution
const entityCount = 256 // One per shard
const ids: string[] = []
for (let i = 0; i < entityCount; i++) {
const id = await brain.add({
type: 'document',
data: `Shard test ${i}`,
metadata: { shardTest: true }
})
ids.push(id)
}
// Batch fetch all
const results = await brain.batchGet(ids)
expect(results.size).toBe(entityCount)
// All entities should be retrievable
for (const id of ids) {
expect(results.has(id)).toBe(true)
}
})
})
})