brainy/tests/integration/storage-batch-operations.test.ts
David Snelling 2626ab8d62 chore(8.0): Phase C + D + E — config simplification, TODO sweep, test race fix
Phase C: storageAutoConfig.ts rewrite (376 LOC → ~110 LOC). The four cloud
adapters are gone so the auto-detection state machine collapses to a single
filesystem-vs-memory choice. StorageType enum is now {MEMORY, FILESYSTEM} only;
StoragePreset stays {AUTO, MEMORY, DISK}. zeroConfig.ts hard-pins
s3Available: false now that the type narrows past the runtime check.

Phase D: TODO/FIXME sweep in src/. Removed seven stale markers. The CLI cow
migrate command's backup path now uses fs.cp with recursive + force:false
instead of a TODO placeholder.

Phase E: skipped-test deletion + parallel-test race fix.
  - storage-batch-operations.test.ts loses its "Cloud Adapter Batch
    Operations" block (GCS/S3/Azure tests, 88 LOC).
  - cow-full-integration.test.ts loses its skipped "S3 adapter" test.
  - create-entities-default.test.ts had testDir hardcoded to
    './test-create-entities-default'; parallel vitest shards collided on
    the writer lock. testDir is now os.tmpdir() + pid + random, and the
    storage option is rootDirectory (the recognized key — the prior
    'path' key fell through to './brainy-data' and triggered a separate
    lock collision against any concurrent default-pathed brain). Added
    afterEach brain.close() so the lock releases before rmSync.
2026-06-09 15:46:51 -07:00

548 lines
19 KiB
TypeScript
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

/**
* Storage-Level Batch Operations Test Suite v5.12.0
*
* Comprehensive testing of new storage-level batch APIs:
* - storage.getNounMetadataBatch() - Batch metadata reads
* - storage.readBatchWithInheritance() - COW-aware batch reads
* - storage.getVerbsBySourceBatch() - Batch relationship queries
* - brain.batchGet() - High-level batch entity retrieval
* - PathResolver.getChildren() - VFS batch operations
*
* Coverage:
* ✅ Type-aware storage compatibility
* ✅ Sharding preservation
* ✅ COW (Copy-on-Write) integration
* ✅ fork() and branch isolation
* ✅ Performance improvements (N+1 → batched)
* ✅ Cloud adapter native batch APIs
*/
import { describe, it, expect, beforeEach, afterEach } from 'vitest'
import { Brainy } from '../../src/brainy'
import { NounType, VerbType } from '../../src/coreTypes'
import { performance } from 'perf_hooks'
describe('Storage-Level Batch Operations v5.12.0', () => {
let brain: Brainy
beforeEach(async () => {
brain = new Brainy({ requireSubtype: false,
storage: { type: 'memory' },
enableCOW: true
})
await brain.init()
})
afterEach(async () => {
await brain.close()
})
describe('brain.batchGet() - High-Level Batch API', () => {
it('should batch fetch multiple entities (metadata-only)', async () => {
// Add test entities
const id1 = await brain.add({
type: 'document',
data: 'Entity 1',
metadata: { category: 'A' }
})
const id2 = await brain.add({
type: 'thing',
data: 'Entity 2',
metadata: { category: 'B' }
})
const id3 = await brain.add({
type: 'person',
data: 'Entity 3',
metadata: { category: 'C' }
})
// Batch fetch (metadata-only by default)
const results = await brain.batchGet([id1, id2, id3])
expect(results.size).toBe(3)
expect(results.get(id1)?.data).toBe('Entity 1')
expect(results.get(id2)?.data).toBe('Entity 2')
expect(results.get(id3)?.data).toBe('Entity 3')
// Vectors should NOT be included by default (empty array or undefined)
const vector = results.get(id1)?.vector
expect(vector === undefined || (Array.isArray(vector) && vector.length === 0)).toBe(true)
})
it('should handle missing entities gracefully', async () => {
const id1 = await brain.add({ type: 'document', data: 'Exists' })
const fakeId = '12345678-1234-1234-1234-123456789abc'
const anotherFake = '87654321-4321-4321-4321-abcdef123456'
const results = await brain.batchGet([id1, fakeId, anotherFake])
expect(results.size).toBe(1)
expect(results.get(id1)?.data).toBe('Exists')
expect(results.has(fakeId)).toBe(false)
})
it('should support includeVectors option (fallback)', async () => {
const id1 = await brain.add({
type: 'document',
data: 'With vector',
metadata: { test: true }
})
// With vectors (currently falls back to individual gets)
const results = await brain.batchGet([id1], { includeVectors: true })
expect(results.size).toBe(1)
const entity = results.get(id1)
expect(entity?.data).toBe('With vector')
expect(entity?.vector).toBeDefined()
expect(entity?.vector?.length).toBeGreaterThan(0)
})
it('should be faster than individual gets for large batches', async () => {
// Create 100 entities
const ids: string[] = []
for (let i = 0; i < 100; i++) {
const id = await brain.add({
type: 'document',
data: `Entity ${i}`,
metadata: { index: i }
})
ids.push(id)
}
// Measure individual gets
const startIndividual = performance.now()
for (const id of ids.slice(0, 20)) {
await brain.get(id)
}
const individualTime = performance.now() - startIndividual
// Measure batch get
const startBatch = performance.now()
await brain.batchGet(ids.slice(20, 40))
const batchTime = performance.now() - startBatch
// Batch should be faster (or at least comparable)
console.log(`Individual: ${individualTime.toFixed(2)}ms, Batch: ${batchTime.toFixed(2)}ms`)
expect(batchTime).toBeLessThan(individualTime * 2) // Allow some overhead
})
})
describe('storage.getNounMetadataBatch() - Storage Layer', () => {
it('should batch fetch noun metadata with type caching', async () => {
// Add entities of different types
const id1 = await brain.add({ type: 'document', data: 'Doc' })
const id2 = await brain.add({ type: 'thing', data: 'Thing' })
const id3 = await brain.add({ type: 'person', data: 'Person' })
// Access storage directly
const storage = brain.storage as any
const results = await storage.getNounMetadataBatch([id1, id2, id3])
expect(results.size).toBe(3)
expect(results.get(id1)?.noun).toBe('document')
expect(results.get(id2)?.noun).toBe('thing')
expect(results.get(id3)?.noun).toBe('person')
// Type cache should be populated
expect(storage.nounTypeCache.has(id1)).toBe(true)
expect(storage.nounTypeCache.get(id1)).toBe('document')
})
it('should handle uncached IDs by trying multiple types', async () => {
// Add entity
const id = await brain.add({ type: 'document', data: 'Test' })
// Clear type cache to simulate uncached scenario
const storage = brain.storage as any
storage.nounTypeCache.delete(id)
// Batch fetch should still work (tries all types)
const results = await storage.getNounMetadataBatch([id])
expect(results.size).toBe(1)
expect(results.get(id)?.noun).toBe('document')
// Cache should be repopulated (or may still be empty if metadata doesn't populate it)
// This is acceptable as long as the data is retrieved correctly
const cachedType = storage.nounTypeCache.get(id)
if (cachedType !== undefined) {
expect(cachedType).toBe('document')
}
})
it('should preserve sharding in all paths', async () => {
// Add entity
const id = await brain.add({ type: 'document', data: 'Sharded' })
// Check that path includes shard
const storage = brain.storage as any
const results = await storage.getNounMetadataBatch([id])
expect(results.size).toBe(1)
// Verify shard is in the path used (check internal call)
// Path should be: entities/nouns/document/metadata/{SHARD}/{ID}.json
const shard = storage.getShardIdFromUuid?.(id) || id.substring(0, 2)
expect(shard).toBeDefined()
})
it('should handle large batches efficiently', async () => {
// Create 500 entities
const ids: string[] = []
for (let i = 0; i < 500; i++) {
const id = await brain.add({
type: 'document',
data: `Batch ${i}`,
metadata: { batch: true }
})
ids.push(id)
}
const startTime = performance.now()
const storage = brain.storage as any
const results = await storage.getNounMetadataBatch(ids)
const duration = performance.now() - startTime
expect(results.size).toBe(500)
console.log(`Batched 500 metadata reads in ${duration.toFixed(2)}ms`)
// Should complete in reasonable time
expect(duration).toBeLessThan(5000) // < 5 seconds
})
})
describe('COW Integration - readBatchWithInheritance()', () => {
it('should resolve branch paths before reading', async () => {
// Add entity on main
const id = await brain.add({ type: 'document', data: 'Main branch' })
// Create fork
const fork = await brain.fork('test-branch')
// Add entity on fork
const forkId = await fork.add({ type: 'document', data: 'Fork branch' })
// Batch get on fork should see fork entity
const forkResults = await fork.batchGet([forkId, id])
expect(forkResults.size).toBe(2)
expect(forkResults.get(forkId)?.data).toBe('Fork branch')
expect(forkResults.get(id)?.data).toBe('Main branch') // Inherited
// Batch get on main should NOT see fork entity
const mainResults = await brain.batchGet([forkId, id])
expect(mainResults.size).toBe(1)
expect(mainResults.has(forkId)).toBe(false) // Not on main
expect(mainResults.get(id)?.data).toBe('Main branch')
})
it('should respect write cache for dirty entities', async () => {
// Add entity
const id = await brain.add({ type: 'document', data: 'Original' })
// Update (may be in write cache before flush)
await brain.update({ id, data: 'Updated' })
// Batch get should see updated version
const results = await brain.batchGet([id])
expect(results.get(id)?.data).toBe('Updated')
})
it('should inherit from parent commits for missing entities', async () => {
// Add entities on main
const id1 = await brain.add({ type: 'document', data: 'Main 1' })
const id2 = await brain.add({ type: 'document', data: 'Main 2' })
// Commit
await brain.commit('Initial entities')
// Create fork
const fork = await brain.fork('child-branch')
// Add new entity only on fork
const forkId = await fork.add({ type: 'document', data: 'Fork only' })
// Batch get on fork should inherit main entities
const results = await fork.batchGet([id1, id2, forkId])
expect(results.size).toBe(3)
expect(results.get(id1)?.data).toBe('Main 1') // Inherited
expect(results.get(id2)?.data).toBe('Main 2') // Inherited
expect(results.get(forkId)?.data).toBe('Fork only') // Fork's own
})
})
describe('getVerbsBySourceBatch() - Batch Relationship Queries', () => {
it('should batch fetch relationships by source IDs', async () => {
// Create entities
const source1 = await brain.add({ type: 'person', data: 'Alice' })
const source2 = await brain.add({ type: 'person', data: 'Bob' })
const target1 = await brain.add({ type: 'document', data: 'Doc1' })
const target2 = await brain.add({ type: 'document', data: 'Doc2' })
// Create relationships
await brain.relate({ from: source1, to: target1, type: 'creates' })
await brain.relate({ from: source1, to: target2, type: 'creates' })
await brain.relate({ from: source2, to: target1, type: 'uses' })
// Batch query
const storage = brain.storage as any
const results = await storage.getVerbsBySourceBatch([source1, source2])
expect(results.size).toBe(2)
const source1Verbs = results.get(source1) || []
const source2Verbs = results.get(source2) || []
expect(source1Verbs.length).toBe(2) // 2 relationships
expect(source2Verbs.length).toBe(1) // 1 relationship
// Check verb types
expect(source1Verbs.every((v: any) => v.verb === 'creates')).toBe(true)
expect(source2Verbs[0].verb).toBe('uses')
})
it('should filter by verb type', async () => {
const source = await brain.add({ type: 'person', data: 'User' })
const target1 = await brain.add({ type: 'document', data: 'Doc1' })
const target2 = await brain.add({ type: 'document', data: 'Doc2' })
// Multiple relationship types
await brain.relate({ from: source, to: target1, type: 'creates' })
await brain.relate({ from: source, to: target2, type: 'uses' })
const storage = brain.storage as any
// Filter by 'creates' type
const createsResults = await storage.getVerbsBySourceBatch(
[source],
'creates'
)
const createsVerbs = createsResults.get(source) || []
expect(createsVerbs.length).toBe(1)
expect(createsVerbs[0].verb).toBe('creates')
})
it('should handle sources with no relationships', async () => {
const source1 = await brain.add({ type: 'person', data: 'Isolated' })
const source2 = await brain.add({ type: 'person', data: 'Connected' })
const target = await brain.add({ type: 'document', data: 'Doc' })
await brain.relate({ from: source2, to: target, type: 'relatedTo' })
const storage = brain.storage as any
const results = await storage.getVerbsBySourceBatch([source1, source2])
expect(results.get(source1) || []).toHaveLength(0) // No relationships
expect(results.get(source2) || []).toHaveLength(1) // Has relationship
})
})
describe('VFS Integration - PathResolver.getChildren()', () => {
it('should use batchGet() for directory children', async () => {
if (!brain.vfs) {
await brain.vfs.init()
}
// Create directory with files
await brain.vfs!.mkdir('/batch-test')
await brain.vfs!.writeFile('/batch-test/file1.txt', 'Content 1')
await brain.vfs!.writeFile('/batch-test/file2.txt', 'Content 2')
await brain.vfs!.writeFile('/batch-test/file3.txt', 'Content 3')
// getChildren() should use batchGet() internally
const startTime = performance.now()
const tree = await brain.vfs!.getTreeStructure('/batch-test')
const duration = performance.now() - startTime
expect(tree.children).toHaveLength(3)
console.log(`VFS getTreeStructure with batch: ${duration.toFixed(2)}ms`)
// Verify all children loaded
const filenames = tree.children!.map(c => c.name).sort()
expect(filenames).toEqual(['file1.txt', 'file2.txt', 'file3.txt'])
})
it('should handle nested directories with parallel traversal', async () => {
if (!brain.vfs) {
await brain.vfs.init()
}
// Create nested structure
await brain.vfs!.mkdir('/root')
await brain.vfs!.mkdir('/root/dir1')
await brain.vfs!.mkdir('/root/dir2')
await brain.vfs!.writeFile('/root/dir1/a.txt', 'A')
await brain.vfs!.writeFile('/root/dir1/b.txt', 'B')
await brain.vfs!.writeFile('/root/dir2/c.txt', 'C')
// Should use breadth-first parallel traversal
const tree = await brain.vfs!.getTreeStructure('/root', { recursive: true })
expect(tree.children).toHaveLength(2) // 2 subdirectories
const dir1 = tree.children!.find(c => c.name === 'dir1')
const dir2 = tree.children!.find(c => c.name === 'dir2')
expect(dir1?.children).toHaveLength(2) // 2 files in dir1
expect(dir2?.children).toHaveLength(1) // 1 file in dir2
})
})
describe('Performance: N+1 Query Elimination', () => {
it('should eliminate N+1 pattern for directory with 12 files', async () => {
if (!brain.vfs) {
await brain.vfs.init()
}
// Create directory with 12 files (original bug scenario)
await brain.vfs!.mkdir('/performance-test')
for (let i = 1; i <= 12; i++) {
await brain.vfs!.writeFile(`/performance-test/file${i}.txt`, `Content ${i}`)
}
// Measure with batching
const startBatch = performance.now()
const treeBatch = await brain.vfs!.getTreeStructure('/performance-test')
const batchTime = performance.now() - startBatch
expect(treeBatch.children).toHaveLength(12)
console.log(`12 files with batching: ${batchTime.toFixed(2)}ms`)
// Before v5.12.0: ~12.7s (22 sequential calls × 580ms)
// After v5.12.0: <1s (2-3 batched calls)
expect(batchTime).toBeLessThan(2000) // Should be < 2 seconds
})
it('should scale to 100 entities efficiently', async () => {
// Create 100 entities
const ids: string[] = []
for (let i = 0; i < 100; i++) {
const id = await brain.add({
type: 'document',
data: `Entity ${i}`,
metadata: { index: i }
})
ids.push(id)
}
// Batch get all 100
const startTime = performance.now()
const results = await brain.batchGet(ids)
const duration = performance.now() - startTime
expect(results.size).toBe(100)
console.log(`100 entities batch: ${duration.toFixed(2)}ms (${(100 / duration * 1000).toFixed(0)} entities/sec)`)
// Should achieve high throughput
const throughput = 100 / duration * 1000
expect(throughput).toBeGreaterThan(50) // > 50 entities/sec
})
})
describe('Error Handling', () => {
it('should handle partial batch failures gracefully', async () => {
const id1 = await brain.add({ type: 'document', data: 'Exists' })
const fakeIds = [
'11111111-1111-1111-1111-111111111111',
'22222222-2222-2222-2222-222222222222',
'33333333-3333-3333-3333-333333333333'
]
// Mix of valid and invalid IDs
const results = await brain.batchGet([id1, ...fakeIds])
// Should return only valid entities
expect(results.size).toBe(1)
expect(results.get(id1)).toBeDefined()
// Invalid IDs should be silently skipped
fakeIds.forEach(fakeId => {
expect(results.has(fakeId)).toBe(false)
})
})
it('should handle empty batch gracefully', async () => {
const results = await brain.batchGet([])
expect(results.size).toBe(0)
})
it('should handle duplicate IDs in batch', async () => {
const id = await brain.add({ type: 'document', data: 'Duplicate test' })
// Same ID multiple times
const results = await brain.batchGet([id, id, id])
// Should return single entry
expect(results.size).toBe(1)
expect(results.get(id)?.data).toBe('Duplicate test')
})
})
describe('Type-Aware Storage Verification', () => {
it('should use correct type-first paths for all types', async () => {
// Create entities of each major type
const types: NounType[] = [
NounType.Document,
NounType.Thing,
NounType.Person,
NounType.File,
NounType.Event
]
const ids: string[] = []
for (const type of types) {
const id = await brain.add({
type: type as any,
data: `Type ${type}`,
metadata: { testType: type }
})
ids.push(id)
}
// Batch fetch
const results = await brain.batchGet(ids)
expect(results.size).toBe(types.length)
// Verify each entity has correct type
for (const [id, entity] of results) {
expect(entity.type).toBeDefined()
expect(types.includes(entity.type as NounType)).toBe(true)
}
})
})
describe('Sharding Verification', () => {
it('should maintain shard distribution in batch operations', async () => {
// Create entities with known shard distribution
const entityCount = 256 // One per shard
const ids: string[] = []
for (let i = 0; i < entityCount; i++) {
const id = await brain.add({
type: 'document',
data: `Shard test ${i}`,
metadata: { shardTest: true }
})
ids.push(id)
}
// Batch fetch all
const results = await brain.batchGet(ids)
expect(results.size).toBe(entityCount)
// All entities should be retrievable
for (const id of ids) {
expect(results.has(id)).toBe(true)
}
})
})
})