fix: flush all native providers on shutdown to prevent data loss
Shutdown/close/flush now properly flushes all 4 components in parallel: metadataIndex, graphIndex, HNSW dirty nodes, and storage counts. Previously only counts were flushed, causing native provider data loss on restart. Also: - Wire roaring, msgpack, entityIdMapper provider consumption from plugins - Fix allOf filter O(n²) intersection → O(n) Set-based - Fix ne/exists negation filter to use Set-based exclusion - Add setMsgpackImplementation() swap in SSTable for native msgpack - Add setRoaringImplementation() swap for native CRoaring bitmaps - Add getAllIntIds() to EntityIdMapper for bitmap operations - Remove TypeAwareHNSWIndex from default index creation path - Export memory detection utilities from internals - Clean up 26 permanently-skipped dead tests
This commit is contained in:
parent
b87426409d
commit
773c5171c3
24 changed files with 297 additions and 1268 deletions
|
|
@ -21,165 +21,7 @@ describe('Domain and Time Clustering', () => {
|
|||
await brain.init()
|
||||
})
|
||||
|
||||
describe('clusterByDomain() - Field-based clustering', () => {
|
||||
it.skip('should cluster entities by type field', async () => {
|
||||
// Add entities of different types
|
||||
await brain.add(createAddParams({
|
||||
data: 'John Smith is a person',
|
||||
type: NounType.Person
|
||||
}))
|
||||
await brain.add(createAddParams({
|
||||
data: 'Jane Doe is also a person',
|
||||
type: NounType.Person
|
||||
}))
|
||||
await brain.add(createAddParams({
|
||||
data: 'Technical document about AI',
|
||||
type: NounType.Document
|
||||
}))
|
||||
await brain.add(createAddParams({
|
||||
data: 'Research paper on machine learning',
|
||||
type: NounType.Document
|
||||
}))
|
||||
await brain.add(createAddParams({
|
||||
data: 'Microsoft Corporation',
|
||||
type: NounType.Organization
|
||||
}))
|
||||
|
||||
// Cluster by type field
|
||||
const clusters = await brain.neural().clusterByDomain('type', {
|
||||
minClusterSize: 1,
|
||||
maxClusters: 10
|
||||
})
|
||||
|
||||
// Should have clusters for each type
|
||||
expect(Array.isArray(clusters)).toBe(true)
|
||||
expect(clusters.length).toBeGreaterThan(0)
|
||||
|
||||
// Verify domain values exist
|
||||
const domains = new Set(clusters.map(c => c.domain))
|
||||
expect(domains.has(NounType.Person) || domains.has('person')).toBe(true)
|
||||
expect(domains.has(NounType.Document) || domains.has('document')).toBe(true)
|
||||
})
|
||||
|
||||
it.skip('should cluster entities by metadata field', async () => {
|
||||
// Add entities with category metadata
|
||||
await brain.add(createAddParams({
|
||||
data: 'JavaScript programming guide',
|
||||
type: NounType.Document,
|
||||
metadata: { category: 'programming' }
|
||||
}))
|
||||
await brain.add(createAddParams({
|
||||
data: 'Python tutorial',
|
||||
type: NounType.Document,
|
||||
metadata: { category: 'programming' }
|
||||
}))
|
||||
await brain.add(createAddParams({
|
||||
data: 'Chocolate cake recipe',
|
||||
type: NounType.Document,
|
||||
metadata: { category: 'cooking' }
|
||||
}))
|
||||
await brain.add(createAddParams({
|
||||
data: 'Pasta preparation',
|
||||
type: NounType.Document,
|
||||
metadata: { category: 'cooking' }
|
||||
}))
|
||||
|
||||
// Cluster by category field
|
||||
const clusters = await brain.neural().clusterByDomain('category', {
|
||||
minClusterSize: 1,
|
||||
maxClusters: 5
|
||||
})
|
||||
|
||||
expect(Array.isArray(clusters)).toBe(true)
|
||||
expect(clusters.length).toBeGreaterThan(0)
|
||||
|
||||
// Verify categories are in domains
|
||||
const domains = new Set(clusters.map(c => c.domain))
|
||||
expect(domains.has('programming')).toBe(true)
|
||||
expect(domains.has('cooking')).toBe(true)
|
||||
})
|
||||
|
||||
it.skip('should handle entities without the specified field', async () => {
|
||||
// Add entities with and without category
|
||||
await brain.add(createAddParams({
|
||||
data: 'Has category',
|
||||
metadata: { category: 'tech' }
|
||||
}))
|
||||
await brain.add(createAddParams({
|
||||
data: 'No category'
|
||||
}))
|
||||
|
||||
const clusters = await brain.neural().clusterByDomain('category', {
|
||||
minClusterSize: 1
|
||||
})
|
||||
|
||||
expect(Array.isArray(clusters)).toBe(true)
|
||||
|
||||
// Should have 'tech' and 'unknown' domains
|
||||
const domains = new Set(clusters.map(c => c.domain))
|
||||
expect(domains.has('tech')).toBe(true)
|
||||
expect(domains.has('unknown')).toBe(true)
|
||||
})
|
||||
})
|
||||
|
||||
describe('clusterByTime() - Temporal clustering', () => {
|
||||
it.skip('should cluster entities by time windows', async () => {
|
||||
const now = new Date()
|
||||
const oneDayAgo = new Date(now.getTime() - 24 * 60 * 60 * 1000)
|
||||
const oneWeekAgo = new Date(now.getTime() - 7 * 24 * 60 * 60 * 1000)
|
||||
const oneMonthAgo = new Date(now.getTime() - 30 * 24 * 60 * 60 * 1000)
|
||||
|
||||
// Add entities with different timestamps
|
||||
await brain.add(createAddParams({
|
||||
data: 'Recent item 1',
|
||||
metadata: { publishedAt: now.toISOString() }
|
||||
}))
|
||||
await brain.add(createAddParams({
|
||||
data: 'Recent item 2',
|
||||
metadata: { publishedAt: oneDayAgo.toISOString() }
|
||||
}))
|
||||
await brain.add(createAddParams({
|
||||
data: 'Old item 1',
|
||||
metadata: { publishedAt: oneWeekAgo.toISOString() }
|
||||
}))
|
||||
await brain.add(createAddParams({
|
||||
data: 'Very old item',
|
||||
metadata: { publishedAt: oneMonthAgo.toISOString() }
|
||||
}))
|
||||
|
||||
// Define time windows
|
||||
const timeWindows = [
|
||||
{
|
||||
start: new Date(now.getTime() - 2 * 24 * 60 * 60 * 1000), // Last 2 days
|
||||
end: now,
|
||||
label: 'Recent'
|
||||
},
|
||||
{
|
||||
start: new Date(now.getTime() - 14 * 24 * 60 * 60 * 1000), // 2-14 days ago
|
||||
end: new Date(now.getTime() - 2 * 24 * 60 * 60 * 1000),
|
||||
label: 'This Week'
|
||||
},
|
||||
{
|
||||
start: new Date(now.getTime() - 60 * 24 * 60 * 60 * 1000), // 14-60 days ago
|
||||
end: new Date(now.getTime() - 14 * 24 * 60 * 60 * 1000),
|
||||
label: 'Older'
|
||||
}
|
||||
]
|
||||
|
||||
// Cluster by time
|
||||
const clusters = await brain.neural().clusterByTime('publishedAt', timeWindows, {
|
||||
timeField: 'publishedAt',
|
||||
windows: timeWindows
|
||||
})
|
||||
|
||||
expect(Array.isArray(clusters)).toBe(true)
|
||||
expect(clusters.length).toBeGreaterThan(0)
|
||||
|
||||
// Verify time windows are represented
|
||||
const windowLabels = new Set(clusters.map(c => c.timeWindow?.label))
|
||||
expect(windowLabels.size).toBeGreaterThan(0)
|
||||
})
|
||||
|
||||
it('should cluster entities by createdAt timestamps', async () => {
|
||||
// These will use the auto-generated createdAt timestamps
|
||||
const id1 = await brain.add(createAddParams({
|
||||
|
|
|
|||
|
|
@ -347,23 +347,6 @@ describe('Neural API - Production Testing', () => {
|
|||
expect(Array.isArray(clusters)).toBe(true)
|
||||
})
|
||||
|
||||
// TODO: Investigate "Cannot read properties of undefined (reading 'vector')" - clustering needs vector data
|
||||
it.skip('should handle different clustering algorithms', async () => {
|
||||
await brain.add(createAddParams({ data: 'Algorithm test 1' }))
|
||||
await brain.add(createAddParams({ data: 'Algorithm test 2' }))
|
||||
|
||||
const algorithms = ['auto', 'semantic', 'hierarchical', 'kmeans', 'dbscan']
|
||||
|
||||
for (const algorithm of algorithms) {
|
||||
const clusters = await brain.neural().clusters({
|
||||
algorithm: algorithm as any,
|
||||
minClusterSize: 1,
|
||||
maxClusters: 5
|
||||
})
|
||||
|
||||
expect(Array.isArray(clusters)).toBe(true)
|
||||
}
|
||||
})
|
||||
})
|
||||
|
||||
describe('11. Streaming Clustering', () => {
|
||||
|
|
@ -435,24 +418,6 @@ describe('Neural API - Production Testing', () => {
|
|||
expect(duration).toBeLessThan(5000) // Should complete in under 5 seconds
|
||||
})
|
||||
|
||||
// TODO: Investigate "Cannot read properties of undefined (reading 'vector')" - clustering needs vector data
|
||||
it.skip('should handle concurrent neural operations', async () => {
|
||||
await brain.add(createAddParams({ data: 'Concurrent test 1' }))
|
||||
await brain.add(createAddParams({ data: 'Concurrent test 2' }))
|
||||
|
||||
const operations = [
|
||||
brain.neural().similar('test1', 'test2'),
|
||||
brain.neural().clusters({ maxClusters: 3 }),
|
||||
brain.neural().outliers({ threshold: 0.8 })
|
||||
]
|
||||
|
||||
const results = await Promise.all(operations)
|
||||
|
||||
expect(results.length).toBe(3)
|
||||
expect(typeof results[0]).toBe('number') // similarity
|
||||
expect(Array.isArray(results[1])).toBe(true) // clusters
|
||||
expect(Array.isArray(results[2])).toBe(true) // outliers
|
||||
})
|
||||
})
|
||||
|
||||
describe('14. Configuration and Options', () => {
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue