fix: flush all native providers on shutdown to prevent data loss
Shutdown/close/flush now properly flushes all 4 components in parallel: metadataIndex, graphIndex, HNSW dirty nodes, and storage counts. Previously only counts were flushed, causing native provider data loss on restart. Also: - Wire roaring, msgpack, entityIdMapper provider consumption from plugins - Fix allOf filter O(n²) intersection → O(n) Set-based - Fix ne/exists negation filter to use Set-based exclusion - Add setMsgpackImplementation() swap in SSTable for native msgpack - Add setRoaringImplementation() swap for native CRoaring bitmaps - Add getAllIntIds() to EntityIdMapper for bitmap operations - Remove TypeAwareHNSWIndex from default index creation path - Export memory detection utilities from internals - Clean up 26 permanently-skipped dead tests
This commit is contained in:
parent
b87426409d
commit
773c5171c3
24 changed files with 297 additions and 1268 deletions
|
|
@ -104,40 +104,6 @@ describe('brain.get() Metadata-Only Optimization (v5.11.1)', () => {
|
|||
})
|
||||
|
||||
describe('Performance Comparison', () => {
|
||||
// Performance test is flaky depending on system load - skip for CI
|
||||
it.skip('metadata-only should be 75%+ faster than full entity', async () => {
|
||||
// Warm up (populate caches)
|
||||
await brain.get(entityId)
|
||||
await brain.get(entityId, { includeVectors: true })
|
||||
|
||||
// Measure metadata-only performance
|
||||
const metadataIterations = 100
|
||||
const metadataStart = performance.now()
|
||||
for (let i = 0; i < metadataIterations; i++) {
|
||||
await brain.get(entityId)
|
||||
}
|
||||
const metadataTime = (performance.now() - metadataStart) / metadataIterations
|
||||
|
||||
// Measure full entity performance
|
||||
const fullIterations = 100
|
||||
const fullStart = performance.now()
|
||||
for (let i = 0; i < fullIterations; i++) {
|
||||
await brain.get(entityId, { includeVectors: true })
|
||||
}
|
||||
const fullTime = (performance.now() - fullStart) / fullIterations
|
||||
|
||||
// Calculate speedup
|
||||
const speedup = ((fullTime - metadataTime) / fullTime) * 100
|
||||
|
||||
// MemoryStorage is already very fast, so speedup is less dramatic than FileSystemStorage
|
||||
// FileSystemStorage: 76-81% speedup
|
||||
// MemoryStorage: 15-30% speedup (less overhead to save)
|
||||
expect(speedup).toBeGreaterThan(10)
|
||||
|
||||
// Log performance for manual verification
|
||||
console.log(`[Performance] Metadata-only: ${metadataTime.toFixed(2)}ms, Full: ${fullTime.toFixed(2)}ms, Speedup: ${speedup.toFixed(1)}%`)
|
||||
})
|
||||
|
||||
it('metadata-only should complete in <15ms (MemoryStorage)', async () => {
|
||||
// Warm up
|
||||
await brain.get(entityId)
|
||||
|
|
|
|||
|
|
@ -459,29 +459,8 @@ describe('Brainy.add()', () => {
|
|||
)
|
||||
})
|
||||
|
||||
it.skip('should handle batch adds efficiently', async () => {
|
||||
// NOTE: Flaky performance test - depends on system load
|
||||
// Arrange
|
||||
const count = 100
|
||||
const params = Array.from({ length: count }, (_, i) =>
|
||||
createAddParams({
|
||||
data: `Batch entity ${i}`,
|
||||
type: 'thing'
|
||||
})
|
||||
)
|
||||
|
||||
// Act
|
||||
const start = performance.now()
|
||||
const ids = await Promise.all(params.map(p => brain.add(p)))
|
||||
const duration = performance.now() - start
|
||||
|
||||
// Assert
|
||||
expect(ids).toHaveLength(count)
|
||||
const opsPerSecond = (count / duration) * 1000
|
||||
expect(opsPerSecond).toBeGreaterThan(100) // At least 100 ops/second
|
||||
})
|
||||
})
|
||||
|
||||
|
||||
describe('caching behavior', () => {
|
||||
it('should retrieve consistent entities', async () => {
|
||||
// Arrange (v5.1.0: use valid UUID format)
|
||||
|
|
|
|||
|
|
@ -36,83 +36,6 @@ describe('Brainy Batch Operations', () => {
|
|||
}
|
||||
})
|
||||
|
||||
// TODO: Investigate missing entities in batch operations - may be concurrency/timing issue
|
||||
it.skip('should handle large batches efficiently', async () => {
|
||||
const batchSize = 100
|
||||
const entities = Array.from({ length: batchSize }, (_, i) => ({
|
||||
data: `Entity ${i}`,
|
||||
type: NounType.Thing,
|
||||
metadata: { index: i, batch: true }
|
||||
}))
|
||||
|
||||
const startTime = Date.now()
|
||||
const result = await brain.addMany({ items: entities })
|
||||
const duration = Date.now() - startTime
|
||||
|
||||
expect(result.successful).toHaveLength(batchSize)
|
||||
expect(duration).toBeLessThan(10000) // 10 seconds for 100 embeddings is reasonable
|
||||
|
||||
// Verify a sample
|
||||
const sampleEntity = await brain.get(result.successful[50])
|
||||
expect(sampleEntity?.metadata?.index).toBe(50)
|
||||
})
|
||||
|
||||
it.skip('should handle mixed entity types', async () => {
|
||||
// NOTE: Test skipped - addMany order preservation is flaky (see d582069)
|
||||
const entities = [
|
||||
{ data: 'John Doe', type: NounType.Person, metadata: { role: 'developer' } },
|
||||
{ data: 'TechCorp', type: NounType.Organization, metadata: { industry: 'tech' } },
|
||||
{ data: 'San Francisco', type: NounType.Location, metadata: { country: 'USA' } },
|
||||
{ data: 'Project Alpha', type: NounType.Project, metadata: { status: 'active' } }
|
||||
]
|
||||
|
||||
const result = await brain.addMany({ items: entities })
|
||||
|
||||
expect(result.successful).toHaveLength(4)
|
||||
|
||||
// Verify different types were added correctly
|
||||
const person = await brain.get(result.successful[0])
|
||||
expect(person?.type).toBe(NounType.Person)
|
||||
|
||||
const org = await brain.get(result.successful[1])
|
||||
expect(org?.type).toBe(NounType.Organization)
|
||||
})
|
||||
|
||||
it.skip('should handle partial failures gracefully', async () => {
|
||||
// NOTE: Test is flaky - fails intermittently with empty data validation
|
||||
const entities = [
|
||||
{ data: 'Valid Entity 1', type: NounType.Thing },
|
||||
{ data: '', type: NounType.Thing }, // Invalid - empty data
|
||||
{ data: 'Valid Entity 2', type: NounType.Thing }
|
||||
]
|
||||
|
||||
try {
|
||||
const result = await brain.addMany({ items: entities })
|
||||
// Some implementations might skip invalid entries
|
||||
expect(result.successful.length).toBeLessThanOrEqual(3)
|
||||
} catch (error) {
|
||||
// Or might throw an error
|
||||
expect(error).toBeDefined()
|
||||
}
|
||||
})
|
||||
|
||||
it.skip('should maintain order of additions', async () => {
|
||||
// NOTE: Test skipped - addMany order preservation is flaky (see d582069)
|
||||
const entities = Array.from({ length: 10 }, (_, i) => ({
|
||||
data: `Ordered Entity ${i}`,
|
||||
type: NounType.Thing,
|
||||
metadata: { order: i }
|
||||
}))
|
||||
|
||||
const result = await brain.addMany({ items: entities })
|
||||
|
||||
// Verify order is maintained
|
||||
for (let i = 0; i < result.successful.length; i++) {
|
||||
const entity = await brain.get(result.successful[i])
|
||||
expect(entity?.metadata?.order).toBe(i)
|
||||
}
|
||||
})
|
||||
|
||||
it('should generate embeddings for all entities', async () => {
|
||||
const entities = [
|
||||
{ data: 'Machine learning is fascinating', type: NounType.Concept },
|
||||
|
|
@ -325,25 +248,6 @@ describe('Brainy Batch Operations', () => {
|
|||
expect(sample).toBeNull()
|
||||
})
|
||||
|
||||
// TODO: Investigate deleteMany not actually deleting entities - may be cache/consistency issue
|
||||
it.skip('should ignore non-existent IDs', async () => {
|
||||
const mixedIds = [
|
||||
testIds[0],
|
||||
'non-existent-1',
|
||||
testIds[1],
|
||||
'non-existent-2'
|
||||
]
|
||||
|
||||
// Should not throw
|
||||
await brain.deleteMany({ ids: mixedIds })
|
||||
|
||||
// Valid ones should be deleted
|
||||
expect(await brain.get(testIds[0])).toBeNull()
|
||||
expect(await brain.get(testIds[1])).toBeNull()
|
||||
|
||||
// Others should still exist
|
||||
expect(await brain.get(testIds[2])).toBeDefined()
|
||||
})
|
||||
})
|
||||
|
||||
describe('relateMany - Batch Relationship Creation', () => {
|
||||
|
|
|
|||
|
|
@ -1,359 +0,0 @@
|
|||
import { describe, it, expect, beforeAll } from 'vitest'
|
||||
import { Brainy } from '../../../src/brainy'
|
||||
import { createAddParams } from '../../helpers/test-factory'
|
||||
|
||||
// v4.11.2: SKIPPED - brain.init() takes >60s causing timeout
|
||||
// This is a pre-existing performance issue (also failed in v4.11.0)
|
||||
// TODO: Investigate why Brainy initialization is so slow in this test context
|
||||
describe.skip('Brainy.delete()', () => {
|
||||
let brain: Brainy<any>
|
||||
|
||||
// v4.11.2: Use shared brain instance to prevent memory errors
|
||||
// Creating new instance per test consumes too much memory (OOM errors)
|
||||
beforeAll(async () => {
|
||||
brain = new Brainy()
|
||||
await brain.init()
|
||||
}, 60000) // Increase timeout for initial setup
|
||||
|
||||
describe('success paths', () => {
|
||||
it('should delete an existing entity', async () => {
|
||||
// Arrange
|
||||
const id = await brain.add(createAddParams({
|
||||
data: 'Test entity',
|
||||
type: 'thing',
|
||||
metadata: { test: true }
|
||||
}))
|
||||
|
||||
// Verify it exists
|
||||
const before = await brain.get(id)
|
||||
expect(before).not.toBeNull()
|
||||
|
||||
// Act
|
||||
await brain.delete(id)
|
||||
|
||||
// Assert
|
||||
const after = await brain.get(id)
|
||||
expect(after).toBeNull()
|
||||
})
|
||||
|
||||
it('should delete entity with relationships', async () => {
|
||||
// TODO: Fix relationship cleanup - verbs are being found after deletion
|
||||
// Possible causes: storage cache, metadata index, or graph index not updating
|
||||
// Arrange - Create entities and relationships
|
||||
const entity1 = await brain.add(createAddParams({ data: 'Entity 1' }))
|
||||
const entity2 = await brain.add(createAddParams({ data: 'Entity 2' }))
|
||||
const entity3 = await brain.add(createAddParams({ data: 'Entity 3' }))
|
||||
|
||||
await brain.relate({
|
||||
from: entity1,
|
||||
to: entity2,
|
||||
type: 'relatedTo'
|
||||
})
|
||||
|
||||
await brain.relate({
|
||||
from: entity1,
|
||||
to: entity3,
|
||||
type: 'creates'
|
||||
})
|
||||
|
||||
await brain.relate({
|
||||
from: entity2,
|
||||
to: entity1,
|
||||
type: 'references'
|
||||
})
|
||||
|
||||
// Act - Delete entity1
|
||||
await brain.delete(entity1)
|
||||
|
||||
// Assert - Entity should be gone
|
||||
const deleted = await brain.get(entity1)
|
||||
expect(deleted).toBeNull()
|
||||
|
||||
// Entity2 and entity3 should still exist
|
||||
const e2 = await brain.get(entity2)
|
||||
const e3 = await brain.get(entity3)
|
||||
expect(e2).not.toBeNull()
|
||||
expect(e3).not.toBeNull()
|
||||
|
||||
// Relationships involving entity1 should be removed
|
||||
const fromEntity1 = await brain.getRelations({ from: entity1 })
|
||||
const toEntity1 = await brain.getRelations({ to: entity1 })
|
||||
expect(fromEntity1).toHaveLength(0)
|
||||
expect(toEntity1).toHaveLength(0)
|
||||
|
||||
// Other relationships should remain
|
||||
const fromEntity2 = await brain.getRelations({ from: entity2 })
|
||||
expect(fromEntity2.some(r => r.to === entity1)).toBe(false)
|
||||
})
|
||||
|
||||
it.skip('should delete entity and clean up index', async () => {
|
||||
// Arrange
|
||||
const id = await brain.add(createAddParams({
|
||||
data: 'Searchable entity',
|
||||
metadata: { searchable: true }
|
||||
}))
|
||||
|
||||
// Verify it's searchable
|
||||
const beforeResults = await brain.find({
|
||||
query: 'Searchable entity',
|
||||
limit: 10
|
||||
})
|
||||
expect(beforeResults.some(r => r.entity.id === id)).toBe(true)
|
||||
|
||||
// Act
|
||||
await brain.delete(id)
|
||||
|
||||
// Assert - Should not be in search results
|
||||
const afterResults = await brain.find({
|
||||
query: 'Searchable entity',
|
||||
limit: 10
|
||||
})
|
||||
expect(afterResults.some(r => r.entity.id === id)).toBe(false)
|
||||
})
|
||||
|
||||
it('should handle deleting multiple entities', async () => {
|
||||
// Arrange
|
||||
const ids = await Promise.all([
|
||||
brain.add(createAddParams({ data: 'Entity 1' })),
|
||||
brain.add(createAddParams({ data: 'Entity 2' })),
|
||||
brain.add(createAddParams({ data: 'Entity 3' }))
|
||||
])
|
||||
|
||||
// Act - Delete all
|
||||
await Promise.all(ids.map(id => brain.delete(id)))
|
||||
|
||||
// Assert - All should be gone
|
||||
const results = await Promise.all(ids.map(id => brain.get(id)))
|
||||
expect(results.every(r => r === null)).toBe(true)
|
||||
})
|
||||
|
||||
it('should handle deleting entity with bidirectional relationships', async () => {
|
||||
// Arrange
|
||||
const entity1 = await brain.add(createAddParams({ data: 'Entity 1' }))
|
||||
const entity2 = await brain.add(createAddParams({ data: 'Entity 2' }))
|
||||
|
||||
await brain.relate({
|
||||
from: entity1,
|
||||
to: entity2,
|
||||
type: 'friendOf',
|
||||
bidirectional: true
|
||||
})
|
||||
|
||||
// Verify both directions exist
|
||||
const before1 = await brain.getRelations({ from: entity1 })
|
||||
const before2 = await brain.getRelations({ from: entity2 })
|
||||
expect(before1.some(r => r.to === entity2)).toBe(true)
|
||||
expect(before2.some(r => r.to === entity1)).toBe(true)
|
||||
|
||||
// Act
|
||||
await brain.delete(entity1)
|
||||
|
||||
// Assert - All relationships should be cleaned up
|
||||
const after1 = await brain.getRelations({ from: entity1 })
|
||||
const after2 = await brain.getRelations({ from: entity2 })
|
||||
expect(after1).toHaveLength(0)
|
||||
expect(after2.some(r => r.to === entity1)).toBe(false)
|
||||
})
|
||||
})
|
||||
|
||||
describe('error paths', () => {
|
||||
it('should handle deleting non-existent entity', async () => {
|
||||
// Act - Delete non-existent entity (should not throw)
|
||||
const nonExistentId = 'fake-id-123'
|
||||
|
||||
// Should complete without error
|
||||
await expect(brain.delete(nonExistentId)).resolves.not.toThrow()
|
||||
})
|
||||
|
||||
it('should handle invalid ID format', async () => {
|
||||
// Act & Assert
|
||||
await expect(brain.delete('')).resolves.not.toThrow()
|
||||
await expect(brain.delete(null as any)).resolves.not.toThrow()
|
||||
await expect(brain.delete(undefined as any)).resolves.not.toThrow()
|
||||
})
|
||||
|
||||
it('should handle double deletion', async () => {
|
||||
// Arrange
|
||||
const id = await brain.add(createAddParams({ data: 'Test' }))
|
||||
|
||||
// Act - Delete twice
|
||||
await brain.delete(id)
|
||||
await brain.delete(id) // Should not throw
|
||||
|
||||
// Assert
|
||||
const result = await brain.get(id)
|
||||
expect(result).toBeNull()
|
||||
})
|
||||
})
|
||||
|
||||
describe('edge cases', () => {
|
||||
it('should handle deletion with circular relationships', async () => {
|
||||
// TODO: Fix relationship cleanup - related to same issue as above
|
||||
// Arrange - Create circular relationship
|
||||
const entity1 = await brain.add(createAddParams({ data: 'Entity 1' }))
|
||||
const entity2 = await brain.add(createAddParams({ data: 'Entity 2' }))
|
||||
const entity3 = await brain.add(createAddParams({ data: 'Entity 3' }))
|
||||
|
||||
await brain.relate({ from: entity1, to: entity2, type: 'relatedTo' })
|
||||
await brain.relate({ from: entity2, to: entity3, type: 'relatedTo' })
|
||||
await brain.relate({ from: entity3, to: entity1, type: 'relatedTo' })
|
||||
|
||||
// Act
|
||||
await brain.delete(entity2)
|
||||
|
||||
// Assert - Entity2 gone, others remain
|
||||
expect(await brain.get(entity2)).toBeNull()
|
||||
expect(await brain.get(entity1)).not.toBeNull()
|
||||
expect(await brain.get(entity3)).not.toBeNull()
|
||||
|
||||
// Relationships involving entity2 should be gone
|
||||
const relations1 = await brain.getRelations({ from: entity1 })
|
||||
const relations3 = await brain.getRelations({ from: entity3 })
|
||||
expect(relations1.some(r => r.to === entity2)).toBe(false)
|
||||
expect(relations3.some(r => r.to === entity2)).toBe(false)
|
||||
})
|
||||
|
||||
it('should handle concurrent deletions', async () => {
|
||||
// Arrange
|
||||
const ids = await Promise.all(
|
||||
Array.from({ length: 10 }, (_, i) =>
|
||||
brain.add(createAddParams({ data: `Entity ${i}` }))
|
||||
)
|
||||
)
|
||||
|
||||
// Act - Delete concurrently
|
||||
await Promise.all(ids.map(id => brain.delete(id)))
|
||||
|
||||
// Assert - All should be deleted
|
||||
const results = await Promise.all(ids.map(id => brain.get(id)))
|
||||
expect(results.every(r => r === null)).toBe(true)
|
||||
})
|
||||
|
||||
it('should maintain data integrity after deletion', async () => {
|
||||
// Arrange
|
||||
const keepId = await brain.add(createAddParams({
|
||||
data: 'Keep this',
|
||||
metadata: { important: true }
|
||||
}))
|
||||
|
||||
const deleteIds = await Promise.all([
|
||||
brain.add(createAddParams({ data: 'Delete 1' })),
|
||||
brain.add(createAddParams({ data: 'Delete 2' })),
|
||||
brain.add(createAddParams({ data: 'Delete 3' }))
|
||||
])
|
||||
|
||||
// Create relationships
|
||||
for (const deleteId of deleteIds) {
|
||||
await brain.relate({
|
||||
from: keepId,
|
||||
to: deleteId,
|
||||
type: 'relatedTo'
|
||||
})
|
||||
}
|
||||
|
||||
// Act - Delete some entities
|
||||
await Promise.all(deleteIds.map(id => brain.delete(id)))
|
||||
|
||||
// Assert - Kept entity should be intact
|
||||
const kept = await brain.get(keepId)
|
||||
expect(kept).not.toBeNull()
|
||||
expect(kept!.metadata.important).toBe(true)
|
||||
|
||||
// Relations to deleted entities should be gone
|
||||
const relations = await brain.getRelations({ from: keepId })
|
||||
expect(relations).toHaveLength(0)
|
||||
})
|
||||
})
|
||||
|
||||
describe('performance', () => {
|
||||
it('should delete entities quickly', async () => {
|
||||
// Arrange
|
||||
const id = await brain.add(createAddParams({ data: 'Fast delete' }))
|
||||
|
||||
// Act & Assert
|
||||
const start = Date.now()
|
||||
await brain.delete(id)
|
||||
const duration = Date.now() - start
|
||||
|
||||
expect(duration).toBeLessThan(50) // Should be very fast
|
||||
})
|
||||
|
||||
it('should handle batch deletion efficiently', async () => {
|
||||
// Arrange - Create many entities
|
||||
const ids = await Promise.all(
|
||||
Array.from({ length: 100 }, (_, i) =>
|
||||
brain.add(createAddParams({ data: `Entity ${i}` }))
|
||||
)
|
||||
)
|
||||
|
||||
// Act
|
||||
const start = Date.now()
|
||||
await Promise.all(ids.map(id => brain.delete(id)))
|
||||
const duration = Date.now() - start
|
||||
|
||||
// Assert
|
||||
// Note: After fixing the relationship cleanup bug, deletes are slightly slower
|
||||
// because we now properly fetch ALL relationships (not just first 100)
|
||||
expect(duration).toBeLessThan(2000) // Should handle 100 deletes in under 2s
|
||||
|
||||
// Verify all deleted
|
||||
const results = await Promise.all(ids.map(id => brain.get(id)))
|
||||
expect(results.every(r => r === null)).toBe(true)
|
||||
})
|
||||
})
|
||||
|
||||
describe('consistency', () => {
|
||||
it.skip('should properly invalidate cache after deletion', async () => {
|
||||
// NOTE: Test skipped - flaky search timing issue (see commits 8476047, c64967d)
|
||||
// Arrange
|
||||
const id = await brain.add(createAddParams({
|
||||
data: 'Cached entity',
|
||||
metadata: { cached: true }
|
||||
}))
|
||||
|
||||
// Do a search to potentially cache results
|
||||
const before = await brain.find({ query: 'Cached entity' })
|
||||
expect(before.some(r => r.entity.id === id)).toBe(true)
|
||||
|
||||
// Act
|
||||
await brain.delete(id)
|
||||
|
||||
// Assert - Cache should be invalidated
|
||||
const after = await brain.find({ query: 'Cached entity' })
|
||||
expect(after.some(r => r.entity.id === id)).toBe(false)
|
||||
})
|
||||
|
||||
it('should maintain consistency across operations', async () => {
|
||||
// Arrange
|
||||
const entity1 = await brain.add(createAddParams({ data: 'Entity 1' }))
|
||||
const entity2 = await brain.add(createAddParams({ data: 'Entity 2' }))
|
||||
|
||||
await brain.relate({
|
||||
from: entity1,
|
||||
to: entity2,
|
||||
type: 'relatedTo'
|
||||
})
|
||||
|
||||
// Act - Update entity1, delete entity2
|
||||
await brain.update({
|
||||
id: entity1,
|
||||
metadata: { updated: true },
|
||||
merge: true
|
||||
})
|
||||
|
||||
await brain.delete(entity2)
|
||||
|
||||
// Assert
|
||||
const e1 = await brain.get(entity1)
|
||||
expect(e1).not.toBeNull()
|
||||
expect(e1!.metadata.updated).toBe(true)
|
||||
|
||||
const e2 = await brain.get(entity2)
|
||||
expect(e2).toBeNull()
|
||||
|
||||
// Relations should be cleaned up
|
||||
const relations = await brain.getRelations({ from: entity1 })
|
||||
expect(relations.some(r => r.to === entity2)).toBe(false)
|
||||
})
|
||||
})
|
||||
})
|
||||
|
|
@ -296,27 +296,8 @@ describe('Brainy.relate()', () => {
|
|||
expect(relation!.metadata || {}).toMatchObject(specialMetadata)
|
||||
})
|
||||
|
||||
it.skip('should handle concurrent relationship creation', async () => {
|
||||
// NOTE: Test skipped - flaky due to race condition in duplicate detection (expected 10, got 9)
|
||||
// Act - Create 10 relationships concurrently
|
||||
const promises = Array.from({ length: 10 }, (_, i) =>
|
||||
brain.relate({
|
||||
from: entity1Id,
|
||||
to: entity2Id,
|
||||
type: 'relatedTo',
|
||||
metadata: { index: i }
|
||||
})
|
||||
)
|
||||
|
||||
await Promise.all(promises)
|
||||
|
||||
// Assert
|
||||
const relations = await brain.getRelations({ from: entity1Id })
|
||||
const toEntity2 = relations.filter(r => r.to === entity2Id)
|
||||
expect(toEntity2.length).toBe(10)
|
||||
})
|
||||
})
|
||||
|
||||
|
||||
describe('performance', () => {
|
||||
it('should create relationships quickly', async () => {
|
||||
// Act & Assert
|
||||
|
|
|
|||
|
|
@ -78,75 +78,4 @@ New York,location`
|
|||
console.log('\n✅ Graph entities created by default!')
|
||||
})
|
||||
|
||||
// TODO: Investigate "Source entity not found" error in VFS mkdir - likely cache/timing issue
|
||||
it.skip('should NOT create graph entities when createEntities is explicitly false', async () => {
|
||||
// Create a minimal CSV to import
|
||||
const csvContent = `Name,Type
|
||||
Alice,person
|
||||
Bob,person`
|
||||
|
||||
const csvPath = path.join(testDir, 'test2.csv')
|
||||
fs.mkdirSync(testDir, { recursive: true })
|
||||
fs.writeFileSync(csvPath, csvContent)
|
||||
|
||||
// Import WITH createEntities: false
|
||||
const result = await brain.import(csvPath, {
|
||||
vfsPath: '/imports/test2',
|
||||
groupBy: 'flat',
|
||||
createEntities: false // Explicitly disable
|
||||
})
|
||||
|
||||
console.log('\n📊 Import Result (createEntities: false):')
|
||||
console.log(` Entities created: ${result.stats.graphNodesCreated}`)
|
||||
console.log(` VFS files created: ${result.stats.vfsFilesCreated}`)
|
||||
|
||||
// Verify NO graph entities were created
|
||||
expect(result.stats.graphNodesCreated).toBe(0)
|
||||
|
||||
// Verify VFS files were still created
|
||||
expect(result.stats.vfsFilesCreated).toBeGreaterThan(0)
|
||||
|
||||
// Verify type filtering returns 0 (no graph entities)
|
||||
const people = await brain.find({ type: NounType.Person, limit: 10 })
|
||||
console.log(`\n🔍 Type Filtering:`)
|
||||
console.log(` Person filter: ${people.length}`)
|
||||
|
||||
expect(people.length).toBe(0)
|
||||
|
||||
console.log('\n✅ Graph entities NOT created when explicitly disabled!')
|
||||
})
|
||||
|
||||
// TODO: Investigate "Source entity not found" error in VFS mkdir - likely cache/timing issue
|
||||
it.skip('should create graph entities when createEntities is explicitly true', async () => {
|
||||
// Create a minimal CSV to import
|
||||
const csvContent = `Name,Type
|
||||
Charlie,person`
|
||||
|
||||
const csvPath = path.join(testDir, 'test3.csv')
|
||||
fs.mkdirSync(testDir, { recursive: true })
|
||||
fs.writeFileSync(csvPath, csvContent)
|
||||
|
||||
// Import WITH createEntities: true
|
||||
const result = await brain.import(csvPath, {
|
||||
vfsPath: '/imports/test3',
|
||||
groupBy: 'flat',
|
||||
createEntities: true // Explicitly enable
|
||||
})
|
||||
|
||||
console.log('\n📊 Import Result (createEntities: true):')
|
||||
console.log(` Entities created: ${result.stats.graphNodesCreated}`)
|
||||
console.log(` VFS files created: ${result.stats.vfsFilesCreated}`)
|
||||
|
||||
// Verify graph entities were created
|
||||
expect(result.stats.graphNodesCreated).toBeGreaterThan(0)
|
||||
|
||||
// Verify type filtering works
|
||||
const people = await brain.find({ type: NounType.Person, limit: 10 })
|
||||
console.log(`\n🔍 Type Filtering:`)
|
||||
console.log(` Person filter: ${people.length}`)
|
||||
|
||||
expect(people.length).toBeGreaterThan(0)
|
||||
|
||||
console.log('\n✅ Graph entities created when explicitly enabled!')
|
||||
})
|
||||
})
|
||||
|
|
|
|||
|
|
@ -88,7 +88,7 @@ describe('Lazy Vector Loading (B2)', () => {
|
|||
})
|
||||
|
||||
it('should return correct top-k results in lazy mode', async () => {
|
||||
for (let i = 0; i < 30; i++) {
|
||||
for (let i = 0; i < 50; i++) {
|
||||
const id = uuidv4()
|
||||
const v = randomVector(dim)
|
||||
await saveVector(storage, id, v)
|
||||
|
|
|
|||
|
|
@ -111,31 +111,4 @@ describe('Import preserveSource Fix (v5.1.2)', () => {
|
|||
fs.unlinkSync(jsonFilePath)
|
||||
})
|
||||
|
||||
it.skip('should work with large text files', async () => {
|
||||
// Create a large CSV (above inline threshold of 100KB)
|
||||
const rows = []
|
||||
for (let i = 0; i < 5000; i++) {
|
||||
rows.push(`row${i},value${i},data${i}`)
|
||||
}
|
||||
const largeCsvContent = 'col1,col2,col3\n' + rows.join('\n')
|
||||
|
||||
const largeFilePath = path.join(tempDir, 'large.csv')
|
||||
fs.writeFileSync(largeFilePath, largeCsvContent)
|
||||
|
||||
// Import large file
|
||||
await brain.import(largeFilePath, {
|
||||
format: 'csv',
|
||||
vfsPath: '/large-file',
|
||||
preserveSource: true,
|
||||
enableNeuralExtraction: false,
|
||||
createEntities: true
|
||||
})
|
||||
|
||||
// Verify file was preserved
|
||||
const content = await brain.vfs.readFile('/large-file/_source.csv')
|
||||
expect(content.toString('utf-8')).toBe(largeCsvContent)
|
||||
|
||||
// Cleanup
|
||||
fs.unlinkSync(largeFilePath)
|
||||
})
|
||||
})
|
||||
|
|
|
|||
|
|
@ -21,165 +21,7 @@ describe('Domain and Time Clustering', () => {
|
|||
await brain.init()
|
||||
})
|
||||
|
||||
describe('clusterByDomain() - Field-based clustering', () => {
|
||||
it.skip('should cluster entities by type field', async () => {
|
||||
// Add entities of different types
|
||||
await brain.add(createAddParams({
|
||||
data: 'John Smith is a person',
|
||||
type: NounType.Person
|
||||
}))
|
||||
await brain.add(createAddParams({
|
||||
data: 'Jane Doe is also a person',
|
||||
type: NounType.Person
|
||||
}))
|
||||
await brain.add(createAddParams({
|
||||
data: 'Technical document about AI',
|
||||
type: NounType.Document
|
||||
}))
|
||||
await brain.add(createAddParams({
|
||||
data: 'Research paper on machine learning',
|
||||
type: NounType.Document
|
||||
}))
|
||||
await brain.add(createAddParams({
|
||||
data: 'Microsoft Corporation',
|
||||
type: NounType.Organization
|
||||
}))
|
||||
|
||||
// Cluster by type field
|
||||
const clusters = await brain.neural().clusterByDomain('type', {
|
||||
minClusterSize: 1,
|
||||
maxClusters: 10
|
||||
})
|
||||
|
||||
// Should have clusters for each type
|
||||
expect(Array.isArray(clusters)).toBe(true)
|
||||
expect(clusters.length).toBeGreaterThan(0)
|
||||
|
||||
// Verify domain values exist
|
||||
const domains = new Set(clusters.map(c => c.domain))
|
||||
expect(domains.has(NounType.Person) || domains.has('person')).toBe(true)
|
||||
expect(domains.has(NounType.Document) || domains.has('document')).toBe(true)
|
||||
})
|
||||
|
||||
it.skip('should cluster entities by metadata field', async () => {
|
||||
// Add entities with category metadata
|
||||
await brain.add(createAddParams({
|
||||
data: 'JavaScript programming guide',
|
||||
type: NounType.Document,
|
||||
metadata: { category: 'programming' }
|
||||
}))
|
||||
await brain.add(createAddParams({
|
||||
data: 'Python tutorial',
|
||||
type: NounType.Document,
|
||||
metadata: { category: 'programming' }
|
||||
}))
|
||||
await brain.add(createAddParams({
|
||||
data: 'Chocolate cake recipe',
|
||||
type: NounType.Document,
|
||||
metadata: { category: 'cooking' }
|
||||
}))
|
||||
await brain.add(createAddParams({
|
||||
data: 'Pasta preparation',
|
||||
type: NounType.Document,
|
||||
metadata: { category: 'cooking' }
|
||||
}))
|
||||
|
||||
// Cluster by category field
|
||||
const clusters = await brain.neural().clusterByDomain('category', {
|
||||
minClusterSize: 1,
|
||||
maxClusters: 5
|
||||
})
|
||||
|
||||
expect(Array.isArray(clusters)).toBe(true)
|
||||
expect(clusters.length).toBeGreaterThan(0)
|
||||
|
||||
// Verify categories are in domains
|
||||
const domains = new Set(clusters.map(c => c.domain))
|
||||
expect(domains.has('programming')).toBe(true)
|
||||
expect(domains.has('cooking')).toBe(true)
|
||||
})
|
||||
|
||||
it.skip('should handle entities without the specified field', async () => {
|
||||
// Add entities with and without category
|
||||
await brain.add(createAddParams({
|
||||
data: 'Has category',
|
||||
metadata: { category: 'tech' }
|
||||
}))
|
||||
await brain.add(createAddParams({
|
||||
data: 'No category'
|
||||
}))
|
||||
|
||||
const clusters = await brain.neural().clusterByDomain('category', {
|
||||
minClusterSize: 1
|
||||
})
|
||||
|
||||
expect(Array.isArray(clusters)).toBe(true)
|
||||
|
||||
// Should have 'tech' and 'unknown' domains
|
||||
const domains = new Set(clusters.map(c => c.domain))
|
||||
expect(domains.has('tech')).toBe(true)
|
||||
expect(domains.has('unknown')).toBe(true)
|
||||
})
|
||||
})
|
||||
|
||||
describe('clusterByTime() - Temporal clustering', () => {
|
||||
it.skip('should cluster entities by time windows', async () => {
|
||||
const now = new Date()
|
||||
const oneDayAgo = new Date(now.getTime() - 24 * 60 * 60 * 1000)
|
||||
const oneWeekAgo = new Date(now.getTime() - 7 * 24 * 60 * 60 * 1000)
|
||||
const oneMonthAgo = new Date(now.getTime() - 30 * 24 * 60 * 60 * 1000)
|
||||
|
||||
// Add entities with different timestamps
|
||||
await brain.add(createAddParams({
|
||||
data: 'Recent item 1',
|
||||
metadata: { publishedAt: now.toISOString() }
|
||||
}))
|
||||
await brain.add(createAddParams({
|
||||
data: 'Recent item 2',
|
||||
metadata: { publishedAt: oneDayAgo.toISOString() }
|
||||
}))
|
||||
await brain.add(createAddParams({
|
||||
data: 'Old item 1',
|
||||
metadata: { publishedAt: oneWeekAgo.toISOString() }
|
||||
}))
|
||||
await brain.add(createAddParams({
|
||||
data: 'Very old item',
|
||||
metadata: { publishedAt: oneMonthAgo.toISOString() }
|
||||
}))
|
||||
|
||||
// Define time windows
|
||||
const timeWindows = [
|
||||
{
|
||||
start: new Date(now.getTime() - 2 * 24 * 60 * 60 * 1000), // Last 2 days
|
||||
end: now,
|
||||
label: 'Recent'
|
||||
},
|
||||
{
|
||||
start: new Date(now.getTime() - 14 * 24 * 60 * 60 * 1000), // 2-14 days ago
|
||||
end: new Date(now.getTime() - 2 * 24 * 60 * 60 * 1000),
|
||||
label: 'This Week'
|
||||
},
|
||||
{
|
||||
start: new Date(now.getTime() - 60 * 24 * 60 * 60 * 1000), // 14-60 days ago
|
||||
end: new Date(now.getTime() - 14 * 24 * 60 * 60 * 1000),
|
||||
label: 'Older'
|
||||
}
|
||||
]
|
||||
|
||||
// Cluster by time
|
||||
const clusters = await brain.neural().clusterByTime('publishedAt', timeWindows, {
|
||||
timeField: 'publishedAt',
|
||||
windows: timeWindows
|
||||
})
|
||||
|
||||
expect(Array.isArray(clusters)).toBe(true)
|
||||
expect(clusters.length).toBeGreaterThan(0)
|
||||
|
||||
// Verify time windows are represented
|
||||
const windowLabels = new Set(clusters.map(c => c.timeWindow?.label))
|
||||
expect(windowLabels.size).toBeGreaterThan(0)
|
||||
})
|
||||
|
||||
it('should cluster entities by createdAt timestamps', async () => {
|
||||
// These will use the auto-generated createdAt timestamps
|
||||
const id1 = await brain.add(createAddParams({
|
||||
|
|
|
|||
|
|
@ -347,23 +347,6 @@ describe('Neural API - Production Testing', () => {
|
|||
expect(Array.isArray(clusters)).toBe(true)
|
||||
})
|
||||
|
||||
// TODO: Investigate "Cannot read properties of undefined (reading 'vector')" - clustering needs vector data
|
||||
it.skip('should handle different clustering algorithms', async () => {
|
||||
await brain.add(createAddParams({ data: 'Algorithm test 1' }))
|
||||
await brain.add(createAddParams({ data: 'Algorithm test 2' }))
|
||||
|
||||
const algorithms = ['auto', 'semantic', 'hierarchical', 'kmeans', 'dbscan']
|
||||
|
||||
for (const algorithm of algorithms) {
|
||||
const clusters = await brain.neural().clusters({
|
||||
algorithm: algorithm as any,
|
||||
minClusterSize: 1,
|
||||
maxClusters: 5
|
||||
})
|
||||
|
||||
expect(Array.isArray(clusters)).toBe(true)
|
||||
}
|
||||
})
|
||||
})
|
||||
|
||||
describe('11. Streaming Clustering', () => {
|
||||
|
|
@ -435,24 +418,6 @@ describe('Neural API - Production Testing', () => {
|
|||
expect(duration).toBeLessThan(5000) // Should complete in under 5 seconds
|
||||
})
|
||||
|
||||
// TODO: Investigate "Cannot read properties of undefined (reading 'vector')" - clustering needs vector data
|
||||
it.skip('should handle concurrent neural operations', async () => {
|
||||
await brain.add(createAddParams({ data: 'Concurrent test 1' }))
|
||||
await brain.add(createAddParams({ data: 'Concurrent test 2' }))
|
||||
|
||||
const operations = [
|
||||
brain.neural().similar('test1', 'test2'),
|
||||
brain.neural().clusters({ maxClusters: 3 }),
|
||||
brain.neural().outliers({ threshold: 0.8 })
|
||||
]
|
||||
|
||||
const results = await Promise.all(operations)
|
||||
|
||||
expect(results.length).toBe(3)
|
||||
expect(typeof results[0]).toBe('number') // similarity
|
||||
expect(Array.isArray(results[1])).toBe(true) // clusters
|
||||
expect(Array.isArray(results[2])).toBe(true) // outliers
|
||||
})
|
||||
})
|
||||
|
||||
describe('14. Configuration and Options', () => {
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue