feat(pagination): implement cursor-based pagination and enhance search caching
- Added SearchCursor and PaginatedSearchResult interfaces for cursor-based pagination support. - Introduced SearchCache class to cache search results, improving performance. - Implemented tests for automatic cache configuration and performance improvements. - Enhanced existing tests to validate pagination and caching behavior.
This commit is contained in:
parent
91cf1785d9
commit
33afd715a0
8 changed files with 1916 additions and 11 deletions
|
|
@ -18,6 +18,8 @@ import {
|
||||||
HNSWConfig,
|
HNSWConfig,
|
||||||
HNSWNoun,
|
HNSWNoun,
|
||||||
SearchResult,
|
SearchResult,
|
||||||
|
SearchCursor,
|
||||||
|
PaginatedSearchResult,
|
||||||
StorageAdapter,
|
StorageAdapter,
|
||||||
Vector,
|
Vector,
|
||||||
VectorDocument
|
VectorDocument
|
||||||
|
|
@ -55,6 +57,8 @@ import {
|
||||||
DomainDetector,
|
DomainDetector,
|
||||||
HealthMonitor
|
HealthMonitor
|
||||||
} from './distributed/index.js'
|
} from './distributed/index.js'
|
||||||
|
import { SearchCache, SearchCacheConfig } from './utils/searchCache.js'
|
||||||
|
import { CacheAutoConfigurator } from './utils/cacheAutoConfig.js'
|
||||||
|
|
||||||
export interface BrainyDataConfig {
|
export interface BrainyDataConfig {
|
||||||
/**
|
/**
|
||||||
|
|
@ -183,6 +187,12 @@ export interface BrainyDataConfig {
|
||||||
verbose?: boolean
|
verbose?: boolean
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Search result caching configuration
|
||||||
|
* Improves performance for repeated queries
|
||||||
|
*/
|
||||||
|
searchCache?: SearchCacheConfig
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Timeout configuration for async operations
|
* Timeout configuration for async operations
|
||||||
* Controls how long operations wait before timing out
|
* Controls how long operations wait before timing out
|
||||||
|
|
@ -367,6 +377,8 @@ export class BrainyData<T = any> implements BrainyDataInterface<T> {
|
||||||
private _dimensions: number
|
private _dimensions: number
|
||||||
private loggingConfig: BrainyDataConfig['logging'] = { verbose: true }
|
private loggingConfig: BrainyDataConfig['logging'] = { verbose: true }
|
||||||
private defaultService: string = 'default'
|
private defaultService: string = 'default'
|
||||||
|
private searchCache: SearchCache<T>
|
||||||
|
private cacheAutoConfigurator: CacheAutoConfigurator
|
||||||
|
|
||||||
// Timeout and retry configuration
|
// Timeout and retry configuration
|
||||||
private timeoutConfig: BrainyDataConfig['timeouts'] = {}
|
private timeoutConfig: BrainyDataConfig['timeouts'] = {}
|
||||||
|
|
@ -548,6 +560,33 @@ export class BrainyData<T = any> implements BrainyDataInterface<T> {
|
||||||
this.distributedConfig = config.distributed
|
this.distributedConfig = config.distributed
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Initialize cache auto-configurator first
|
||||||
|
this.cacheAutoConfigurator = new CacheAutoConfigurator()
|
||||||
|
|
||||||
|
// Auto-detect optimal cache configuration if not explicitly provided
|
||||||
|
let finalSearchCacheConfig = config.searchCache
|
||||||
|
if (!config.searchCache || Object.keys(config.searchCache).length === 0) {
|
||||||
|
const autoConfig = this.cacheAutoConfigurator.autoDetectOptimalConfig(
|
||||||
|
config.storage
|
||||||
|
)
|
||||||
|
finalSearchCacheConfig = autoConfig.cacheConfig
|
||||||
|
|
||||||
|
// Apply auto-detected real-time update configuration if not explicitly set
|
||||||
|
if (!config.realtimeUpdates && autoConfig.realtimeConfig.enabled) {
|
||||||
|
this.realtimeUpdateConfig = {
|
||||||
|
...this.realtimeUpdateConfig,
|
||||||
|
...autoConfig.realtimeConfig
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
if (this.loggingConfig?.verbose) {
|
||||||
|
console.log(this.cacheAutoConfigurator.getConfigExplanation(autoConfig))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Initialize search cache with final configuration
|
||||||
|
this.searchCache = new SearchCache<T>(finalSearchCacheConfig)
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
|
|
@ -721,6 +760,19 @@ export class BrainyData<T = any> implements BrainyDataInterface<T> {
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Cleanup expired cache entries (defensive mechanism for distributed scenarios)
|
||||||
|
const expiredCount = this.searchCache.cleanupExpiredEntries()
|
||||||
|
if (expiredCount > 0 && this.loggingConfig?.verbose) {
|
||||||
|
console.log(`Cleaned up ${expiredCount} expired cache entries`)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Adapt cache configuration based on performance (every few updates)
|
||||||
|
// Only adapt every 5th update to avoid over-optimization
|
||||||
|
const updateCount = Math.floor((Date.now() - (this.lastUpdateTime || 0)) / this.realtimeUpdateConfig.interval)
|
||||||
|
if (updateCount % 5 === 0) {
|
||||||
|
this.adaptCacheConfiguration()
|
||||||
|
}
|
||||||
|
|
||||||
// Update the last update time
|
// Update the last update time
|
||||||
this.lastUpdateTime = Date.now()
|
this.lastUpdateTime = Date.now()
|
||||||
|
|
||||||
|
|
@ -821,6 +873,14 @@ export class BrainyData<T = any> implements BrainyDataInterface<T> {
|
||||||
)
|
)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Invalidate search cache if any external changes were detected
|
||||||
|
if (addedCount > 0 || updatedCount > 0 || deletedCount > 0) {
|
||||||
|
this.searchCache.invalidateOnDataChange('update')
|
||||||
|
if (this.loggingConfig?.verbose) {
|
||||||
|
console.log('Search cache invalidated due to external data changes')
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// Update the last known noun count
|
// Update the last known noun count
|
||||||
this.lastKnownNounCount = await this.getNounCount()
|
this.lastKnownNounCount = await this.getNounCount()
|
||||||
} catch (error) {
|
} catch (error) {
|
||||||
|
|
@ -879,6 +939,14 @@ export class BrainyData<T = any> implements BrainyDataInterface<T> {
|
||||||
// Update the last known noun count
|
// Update the last known noun count
|
||||||
this.lastKnownNounCount = currentCount
|
this.lastKnownNounCount = currentCount
|
||||||
|
|
||||||
|
// Invalidate search cache if new nouns were detected
|
||||||
|
if (newNouns.length > 0) {
|
||||||
|
this.searchCache.invalidateOnDataChange('add')
|
||||||
|
if (this.loggingConfig?.verbose) {
|
||||||
|
console.log('Search cache invalidated due to external data changes')
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
if (this.loggingConfig?.verbose && newNouns.length > 0) {
|
if (this.loggingConfig?.verbose && newNouns.length > 0) {
|
||||||
console.log(
|
console.log(
|
||||||
`Real-time update: Added ${newNouns.length} new nouns to index using full scan`
|
`Real-time update: Added ${newNouns.length} new nouns to index using full scan`
|
||||||
|
|
@ -1543,6 +1611,9 @@ export class BrainyData<T = any> implements BrainyDataInterface<T> {
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Invalidate search cache since data has changed
|
||||||
|
this.searchCache.invalidateOnDataChange('add')
|
||||||
|
|
||||||
return id
|
return id
|
||||||
} catch (error) {
|
} catch (error) {
|
||||||
console.error('Failed to add vector:', error)
|
console.error('Failed to add vector:', error)
|
||||||
|
|
@ -1832,6 +1903,7 @@ export class BrainyData<T = any> implements BrainyDataInterface<T> {
|
||||||
options: {
|
options: {
|
||||||
forceEmbed?: boolean // Force using the embedding function even if input is a vector
|
forceEmbed?: boolean // Force using the embedding function even if input is a vector
|
||||||
service?: string // Filter results by the service that created the data
|
service?: string // Filter results by the service that created the data
|
||||||
|
offset?: number // Number of results to skip for pagination (default: 0)
|
||||||
} = {}
|
} = {}
|
||||||
): Promise<SearchResult<T>[]> {
|
): Promise<SearchResult<T>[]> {
|
||||||
// Helper function to filter results by service
|
// Helper function to filter results by service
|
||||||
|
|
@ -1930,13 +2002,20 @@ export class BrainyData<T = any> implements BrainyDataInterface<T> {
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// Search in the index
|
// When using offset, we need to fetch more results and then slice
|
||||||
const results = await this.index.search(queryVector, k)
|
const offset = options.offset || 0
|
||||||
|
const totalNeeded = k + offset
|
||||||
|
|
||||||
|
// Search in the index for totalNeeded results
|
||||||
|
const results = await this.index.search(queryVector, totalNeeded)
|
||||||
|
|
||||||
|
// Skip the offset number of results
|
||||||
|
const paginatedResults = results.slice(offset, offset + k)
|
||||||
|
|
||||||
// Get metadata for each result
|
// Get metadata for each result
|
||||||
const searchResults: SearchResult<T>[] = []
|
const searchResults: SearchResult<T>[] = []
|
||||||
|
|
||||||
for (const [id, score] of results) {
|
for (const [id, score] of paginatedResults) {
|
||||||
const noun = this.index.getNouns().get(id)
|
const noun = this.index.getNouns().get(id)
|
||||||
if (!noun) {
|
if (!noun) {
|
||||||
continue
|
continue
|
||||||
|
|
@ -1990,8 +2069,9 @@ export class BrainyData<T = any> implements BrainyDataInterface<T> {
|
||||||
// Sort by distance (ascending)
|
// Sort by distance (ascending)
|
||||||
results.sort((a, b) => a[1] - b[1])
|
results.sort((a, b) => a[1] - b[1])
|
||||||
|
|
||||||
// Take top k results
|
// Apply offset and take k results
|
||||||
const topResults = results.slice(0, k)
|
const offset = options.offset || 0
|
||||||
|
const topResults = results.slice(offset, offset + k)
|
||||||
|
|
||||||
// Get metadata for each result
|
// Get metadata for each result
|
||||||
const searchResults: SearchResult<T>[] = []
|
const searchResults: SearchResult<T>[] = []
|
||||||
|
|
@ -2053,6 +2133,8 @@ export class BrainyData<T = any> implements BrainyDataInterface<T> {
|
||||||
service?: string // Filter results by the service that created the data
|
service?: string // Filter results by the service that created the data
|
||||||
searchField?: string // Optional specific field to search within JSON documents
|
searchField?: string // Optional specific field to search within JSON documents
|
||||||
filter?: { domain?: string } // Filter results by domain
|
filter?: { domain?: string } // Filter results by domain
|
||||||
|
offset?: number // Number of results to skip for pagination (default: 0)
|
||||||
|
skipCache?: boolean // Skip cache for this search (default: false)
|
||||||
} = {}
|
} = {}
|
||||||
): Promise<SearchResult<T>[]> {
|
): Promise<SearchResult<T>[]> {
|
||||||
const startTime = Date.now()
|
const startTime = Date.now()
|
||||||
|
|
@ -2115,13 +2197,33 @@ export class BrainyData<T = any> implements BrainyDataInterface<T> {
|
||||||
|
|
||||||
// Default behavior (backward compatible): search locally
|
// Default behavior (backward compatible): search locally
|
||||||
try {
|
try {
|
||||||
|
// Check cache first (transparent to user)
|
||||||
|
const cacheKey = this.searchCache.getCacheKey(queryVectorOrData, k, options)
|
||||||
|
const cachedResults = this.searchCache.get(cacheKey)
|
||||||
|
|
||||||
|
if (cachedResults) {
|
||||||
|
// Track cache hit in health monitor
|
||||||
|
if (this.healthMonitor) {
|
||||||
|
const latency = Date.now() - startTime
|
||||||
|
this.healthMonitor.recordRequest(latency, false)
|
||||||
|
this.healthMonitor.recordCacheAccess(true)
|
||||||
|
}
|
||||||
|
return cachedResults
|
||||||
|
}
|
||||||
|
|
||||||
|
// Cache miss - perform actual search
|
||||||
const results = await this.searchLocal(queryVectorOrData, k, options)
|
const results = await this.searchLocal(queryVectorOrData, k, options)
|
||||||
|
|
||||||
|
// Cache results for future queries (unless explicitly disabled)
|
||||||
|
if (!options.skipCache) {
|
||||||
|
this.searchCache.set(cacheKey, results)
|
||||||
|
}
|
||||||
|
|
||||||
// Track successful search in health monitor
|
// Track successful search in health monitor
|
||||||
if (this.healthMonitor) {
|
if (this.healthMonitor) {
|
||||||
const latency = Date.now() - startTime
|
const latency = Date.now() - startTime
|
||||||
this.healthMonitor.recordRequest(latency, false)
|
this.healthMonitor.recordRequest(latency, false)
|
||||||
this.healthMonitor.recordCacheAccess(results.length > 0)
|
this.healthMonitor.recordCacheAccess(false)
|
||||||
}
|
}
|
||||||
|
|
||||||
return results
|
return results
|
||||||
|
|
@ -2135,6 +2237,79 @@ export class BrainyData<T = any> implements BrainyDataInterface<T> {
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Search with cursor-based pagination for better performance on large datasets
|
||||||
|
* @param queryVectorOrData Query vector or data to search for
|
||||||
|
* @param k Number of results to return
|
||||||
|
* @param options Additional options including cursor for pagination
|
||||||
|
* @returns Paginated search results with cursor for next page
|
||||||
|
*/
|
||||||
|
public async searchWithCursor(
|
||||||
|
queryVectorOrData: Vector | any,
|
||||||
|
k: number = 10,
|
||||||
|
options: {
|
||||||
|
forceEmbed?: boolean
|
||||||
|
nounTypes?: string[]
|
||||||
|
includeVerbs?: boolean
|
||||||
|
service?: string
|
||||||
|
searchField?: string
|
||||||
|
filter?: { domain?: string }
|
||||||
|
cursor?: SearchCursor // For continuing from previous search
|
||||||
|
skipCache?: boolean
|
||||||
|
} = {}
|
||||||
|
): Promise<PaginatedSearchResult<T>> {
|
||||||
|
// For cursor-based search, we need to fetch more results and filter
|
||||||
|
const searchK = options.cursor ? k + 20 : k // Get extra results for filtering
|
||||||
|
|
||||||
|
// Perform regular search
|
||||||
|
const allResults = await this.search(queryVectorOrData, searchK, {
|
||||||
|
...options,
|
||||||
|
skipCache: options.skipCache
|
||||||
|
})
|
||||||
|
|
||||||
|
let results = allResults
|
||||||
|
let startIndex = 0
|
||||||
|
|
||||||
|
// If cursor provided, find starting position
|
||||||
|
if (options.cursor) {
|
||||||
|
startIndex = allResults.findIndex(r =>
|
||||||
|
r.id === options.cursor!.lastId &&
|
||||||
|
Math.abs(r.score - options.cursor!.lastScore) < 0.0001
|
||||||
|
)
|
||||||
|
|
||||||
|
if (startIndex >= 0) {
|
||||||
|
startIndex += 1 // Start after the cursor position
|
||||||
|
results = allResults.slice(startIndex, startIndex + k)
|
||||||
|
} else {
|
||||||
|
// Cursor not found, might be stale - return from beginning
|
||||||
|
results = allResults.slice(0, k)
|
||||||
|
startIndex = 0
|
||||||
|
}
|
||||||
|
} else {
|
||||||
|
results = allResults.slice(0, k)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Create cursor for next page
|
||||||
|
let nextCursor: SearchCursor | undefined
|
||||||
|
const hasMoreResults = (startIndex + results.length) < allResults.length || allResults.length >= searchK
|
||||||
|
|
||||||
|
if (results.length > 0 && hasMoreResults) {
|
||||||
|
const lastResult = results[results.length - 1]
|
||||||
|
nextCursor = {
|
||||||
|
lastId: lastResult.id,
|
||||||
|
lastScore: lastResult.score,
|
||||||
|
position: startIndex + results.length
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
return {
|
||||||
|
results,
|
||||||
|
cursor: nextCursor,
|
||||||
|
hasMore: !!nextCursor,
|
||||||
|
totalEstimate: allResults.length > searchK ? undefined : allResults.length
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Search the local database for similar vectors
|
* Search the local database for similar vectors
|
||||||
* @param queryVectorOrData Query vector or data to search for
|
* @param queryVectorOrData Query vector or data to search for
|
||||||
|
|
@ -2153,6 +2328,8 @@ export class BrainyData<T = any> implements BrainyDataInterface<T> {
|
||||||
searchField?: string // Optional specific field to search within JSON documents
|
searchField?: string // Optional specific field to search within JSON documents
|
||||||
priorityFields?: string[] // Fields to prioritize when searching JSON documents
|
priorityFields?: string[] // Fields to prioritize when searching JSON documents
|
||||||
filter?: { domain?: string } // Filter results by domain
|
filter?: { domain?: string } // Filter results by domain
|
||||||
|
offset?: number // Number of results to skip for pagination (default: 0)
|
||||||
|
skipCache?: boolean // Skip cache for this search (default: false)
|
||||||
} = {}
|
} = {}
|
||||||
): Promise<SearchResult<T>[]> {
|
): Promise<SearchResult<T>[]> {
|
||||||
if (!this.isInitialized) {
|
if (!this.isInitialized) {
|
||||||
|
|
@ -2216,14 +2393,16 @@ export class BrainyData<T = any> implements BrainyDataInterface<T> {
|
||||||
options.nounTypes,
|
options.nounTypes,
|
||||||
{
|
{
|
||||||
forceEmbed: options.forceEmbed,
|
forceEmbed: options.forceEmbed,
|
||||||
service: options.service
|
service: options.service,
|
||||||
|
offset: options.offset
|
||||||
}
|
}
|
||||||
)
|
)
|
||||||
} else {
|
} else {
|
||||||
// Otherwise, search all GraphNouns
|
// Otherwise, search all GraphNouns
|
||||||
searchResults = await this.searchByNounTypes(queryToUse, k, null, {
|
searchResults = await this.searchByNounTypes(queryToUse, k, null, {
|
||||||
forceEmbed: options.forceEmbed,
|
forceEmbed: options.forceEmbed,
|
||||||
service: options.service
|
service: options.service,
|
||||||
|
offset: options.offset
|
||||||
})
|
})
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
@ -2637,6 +2816,9 @@ export class BrainyData<T = any> implements BrainyDataInterface<T> {
|
||||||
// Ignore
|
// Ignore
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Invalidate search cache since data has changed
|
||||||
|
this.searchCache.invalidateOnDataChange('delete')
|
||||||
|
|
||||||
return true
|
return true
|
||||||
} catch (error) {
|
} catch (error) {
|
||||||
console.error(`Failed to delete vector ${id}:`, error)
|
console.error(`Failed to delete vector ${id}:`, error)
|
||||||
|
|
@ -2746,6 +2928,9 @@ export class BrainyData<T = any> implements BrainyDataInterface<T> {
|
||||||
const service = this.getServiceName(options)
|
const service = this.getServiceName(options)
|
||||||
await this.storage!.incrementStatistic('metadata', service)
|
await this.storage!.incrementStatistic('metadata', service)
|
||||||
|
|
||||||
|
// Invalidate search cache since metadata has changed
|
||||||
|
this.searchCache.invalidateOnDataChange('update')
|
||||||
|
|
||||||
return true
|
return true
|
||||||
} catch (error) {
|
} catch (error) {
|
||||||
console.error(`Failed to update metadata for vector ${id}:`, error)
|
console.error(`Failed to update metadata for vector ${id}:`, error)
|
||||||
|
|
@ -3195,6 +3380,9 @@ export class BrainyData<T = any> implements BrainyDataInterface<T> {
|
||||||
// Update HNSW index size (excluding verbs)
|
// Update HNSW index size (excluding verbs)
|
||||||
await this.storage!.updateHnswIndexSize(await this.getNounCount())
|
await this.storage!.updateHnswIndexSize(await this.getNounCount())
|
||||||
|
|
||||||
|
// Invalidate search cache since verb data has changed
|
||||||
|
this.searchCache.invalidateOnDataChange('add')
|
||||||
|
|
||||||
return id
|
return id
|
||||||
} catch (error) {
|
} catch (error) {
|
||||||
console.error('Failed to add verb:', error)
|
console.error('Failed to add verb:', error)
|
||||||
|
|
@ -3461,6 +3649,9 @@ export class BrainyData<T = any> implements BrainyDataInterface<T> {
|
||||||
|
|
||||||
// Clear storage
|
// Clear storage
|
||||||
await this.storage!.clear()
|
await this.storage!.clear()
|
||||||
|
|
||||||
|
// Clear search cache since all data has been removed
|
||||||
|
this.searchCache.invalidateOnDataChange('delete')
|
||||||
} catch (error) {
|
} catch (error) {
|
||||||
console.error('Failed to clear vector database:', error)
|
console.error('Failed to clear vector database:', error)
|
||||||
throw new Error(`Failed to clear vector database: ${error}`)
|
throw new Error(`Failed to clear vector database: ${error}`)
|
||||||
|
|
@ -3474,6 +3665,79 @@ export class BrainyData<T = any> implements BrainyDataInterface<T> {
|
||||||
return this.index.size()
|
return this.index.size()
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Get search cache statistics for performance monitoring
|
||||||
|
* @returns Cache statistics including hit rate and memory usage
|
||||||
|
*/
|
||||||
|
public getCacheStats() {
|
||||||
|
return {
|
||||||
|
search: this.searchCache.getStats(),
|
||||||
|
searchMemoryUsage: this.searchCache.getMemoryUsage()
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Clear search cache manually (useful for testing or memory management)
|
||||||
|
*/
|
||||||
|
public clearCache(): void {
|
||||||
|
this.searchCache.clear()
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Adapt cache configuration based on current performance metrics
|
||||||
|
* This method analyzes usage patterns and automatically optimizes cache settings
|
||||||
|
* @private
|
||||||
|
*/
|
||||||
|
private adaptCacheConfiguration(): void {
|
||||||
|
const stats = this.searchCache.getStats()
|
||||||
|
const memoryUsage = this.searchCache.getMemoryUsage()
|
||||||
|
const currentConfig = this.searchCache.getConfig()
|
||||||
|
|
||||||
|
// Prepare performance metrics for adaptation
|
||||||
|
const performanceMetrics = {
|
||||||
|
hitRate: stats.hitRate,
|
||||||
|
avgResponseTime: 50, // Would be measured in real implementation
|
||||||
|
memoryUsage: memoryUsage,
|
||||||
|
externalChangesDetected: 0, // Would be tracked from real-time updates
|
||||||
|
timeSinceLastChange: Date.now() - this.lastUpdateTime
|
||||||
|
}
|
||||||
|
|
||||||
|
// Try to adapt configuration
|
||||||
|
const newConfig = this.cacheAutoConfigurator.adaptConfiguration(
|
||||||
|
currentConfig,
|
||||||
|
performanceMetrics
|
||||||
|
)
|
||||||
|
|
||||||
|
if (newConfig) {
|
||||||
|
// Apply new cache configuration
|
||||||
|
this.searchCache.updateConfig(newConfig.cacheConfig)
|
||||||
|
|
||||||
|
// Apply new real-time update configuration if needed
|
||||||
|
if (newConfig.realtimeConfig.enabled !== this.realtimeUpdateConfig.enabled ||
|
||||||
|
newConfig.realtimeConfig.interval !== this.realtimeUpdateConfig.interval) {
|
||||||
|
|
||||||
|
const wasEnabled = this.realtimeUpdateConfig.enabled
|
||||||
|
this.realtimeUpdateConfig = {
|
||||||
|
...this.realtimeUpdateConfig,
|
||||||
|
...newConfig.realtimeConfig
|
||||||
|
}
|
||||||
|
|
||||||
|
// Restart real-time updates with new configuration
|
||||||
|
if (wasEnabled) {
|
||||||
|
this.stopRealtimeUpdates()
|
||||||
|
}
|
||||||
|
if (this.realtimeUpdateConfig.enabled && this.isInitialized) {
|
||||||
|
this.startRealtimeUpdates()
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
if (this.loggingConfig?.verbose) {
|
||||||
|
console.log('🔧 Auto-adapted cache configuration:')
|
||||||
|
console.log(this.cacheAutoConfigurator.getConfigExplanation(newConfig))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Get the number of nouns in the database (excluding verbs)
|
* Get the number of nouns in the database (excluding verbs)
|
||||||
* This is used for statistics reporting to match the expected behavior in tests
|
* This is used for statistics reporting to match the expected behavior in tests
|
||||||
|
|
@ -4167,6 +4431,7 @@ export class BrainyData<T = any> implements BrainyDataInterface<T> {
|
||||||
storeResults?: boolean // Whether to store the results in the local database (default: true)
|
storeResults?: boolean // Whether to store the results in the local database (default: true)
|
||||||
service?: string // Filter results by the service that created the data
|
service?: string // Filter results by the service that created the data
|
||||||
searchField?: string // Optional specific field to search within JSON documents
|
searchField?: string // Optional specific field to search within JSON documents
|
||||||
|
offset?: number // Number of results to skip for pagination (default: 0)
|
||||||
} = {}
|
} = {}
|
||||||
): Promise<SearchResult<T>[]> {
|
): Promise<SearchResult<T>[]> {
|
||||||
await this.ensureInitialized()
|
await this.ensureInitialized()
|
||||||
|
|
@ -4198,18 +4463,24 @@ export class BrainyData<T = any> implements BrainyDataInterface<T> {
|
||||||
)
|
)
|
||||||
}
|
}
|
||||||
|
|
||||||
// Search the remote server
|
// When using offset, fetch more results and slice
|
||||||
|
const offset = options.offset || 0
|
||||||
|
const totalNeeded = k + offset
|
||||||
|
|
||||||
|
// Search the remote server for totalNeeded results
|
||||||
const searchResult = await this.serverSearchConduit.searchServer(
|
const searchResult = await this.serverSearchConduit.searchServer(
|
||||||
this.serverConnection.connectionId,
|
this.serverConnection.connectionId,
|
||||||
query,
|
query,
|
||||||
k
|
totalNeeded
|
||||||
)
|
)
|
||||||
|
|
||||||
if (!searchResult.success) {
|
if (!searchResult.success) {
|
||||||
throw new Error(`Remote search failed: ${searchResult.error}`)
|
throw new Error(`Remote search failed: ${searchResult.error}`)
|
||||||
}
|
}
|
||||||
|
|
||||||
return searchResult.data as SearchResult<T>[]
|
// Apply offset to remote results
|
||||||
|
const allResults = searchResult.data as SearchResult<T>[]
|
||||||
|
return allResults.slice(offset, offset + k)
|
||||||
} catch (error) {
|
} catch (error) {
|
||||||
console.error('Failed to search remote server:', error)
|
console.error('Failed to search remote server:', error)
|
||||||
throw new Error(`Failed to search remote server: ${error}`)
|
throw new Error(`Failed to search remote server: ${error}`)
|
||||||
|
|
@ -4233,6 +4504,7 @@ export class BrainyData<T = any> implements BrainyDataInterface<T> {
|
||||||
localFirst?: boolean // Whether to search local first (default: true)
|
localFirst?: boolean // Whether to search local first (default: true)
|
||||||
service?: string // Filter results by the service that created the data
|
service?: string // Filter results by the service that created the data
|
||||||
searchField?: string // Optional specific field to search within JSON documents
|
searchField?: string // Optional specific field to search within JSON documents
|
||||||
|
offset?: number // Number of results to skip for pagination (default: 0)
|
||||||
} = {}
|
} = {}
|
||||||
): Promise<SearchResult<T>[]> {
|
): Promise<SearchResult<T>[]> {
|
||||||
await this.ensureInitialized()
|
await this.ensureInitialized()
|
||||||
|
|
|
||||||
|
|
@ -26,6 +26,25 @@ export interface SearchResult<T = any> {
|
||||||
metadata?: T
|
metadata?: T
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Cursor for pagination through search results
|
||||||
|
*/
|
||||||
|
export interface SearchCursor {
|
||||||
|
lastId: string
|
||||||
|
lastScore: number
|
||||||
|
position: number // For debugging/logging
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Paginated search result with cursor support
|
||||||
|
*/
|
||||||
|
export interface PaginatedSearchResult<T = any> {
|
||||||
|
results: SearchResult<T>[]
|
||||||
|
cursor?: SearchCursor
|
||||||
|
hasMore: boolean
|
||||||
|
totalEstimate?: number
|
||||||
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Distance function for comparing vectors
|
* Distance function for comparing vectors
|
||||||
*/
|
*/
|
||||||
|
|
|
||||||
330
src/utils/cacheAutoConfig.ts
Normal file
330
src/utils/cacheAutoConfig.ts
Normal file
|
|
@ -0,0 +1,330 @@
|
||||||
|
/**
|
||||||
|
* Intelligent cache auto-configuration system
|
||||||
|
* Adapts cache settings based on environment, usage patterns, and storage type
|
||||||
|
*/
|
||||||
|
|
||||||
|
import { SearchCacheConfig } from './searchCache.js'
|
||||||
|
import { BrainyDataConfig } from '../brainyData.js'
|
||||||
|
|
||||||
|
export interface CacheUsageStats {
|
||||||
|
totalQueries: number
|
||||||
|
repeatQueries: number
|
||||||
|
avgQueryTime: number
|
||||||
|
memoryPressure: number
|
||||||
|
storageType: 'memory' | 'opfs' | 's3' | 'filesystem'
|
||||||
|
isDistributed: boolean
|
||||||
|
changeFrequency: number // changes per minute
|
||||||
|
readWriteRatio: number // reads / writes
|
||||||
|
}
|
||||||
|
|
||||||
|
export interface AutoConfigResult {
|
||||||
|
cacheConfig: SearchCacheConfig
|
||||||
|
realtimeConfig: NonNullable<BrainyDataConfig['realtimeUpdates']>
|
||||||
|
reasoning: string[]
|
||||||
|
}
|
||||||
|
|
||||||
|
export class CacheAutoConfigurator {
|
||||||
|
private stats: CacheUsageStats = {
|
||||||
|
totalQueries: 0,
|
||||||
|
repeatQueries: 0,
|
||||||
|
avgQueryTime: 50,
|
||||||
|
memoryPressure: 0,
|
||||||
|
storageType: 'memory',
|
||||||
|
isDistributed: false,
|
||||||
|
changeFrequency: 0,
|
||||||
|
readWriteRatio: 10,
|
||||||
|
}
|
||||||
|
|
||||||
|
private configHistory: AutoConfigResult[] = []
|
||||||
|
private lastOptimization = 0
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Auto-detect optimal cache configuration based on current conditions
|
||||||
|
*/
|
||||||
|
public autoDetectOptimalConfig(
|
||||||
|
storageConfig?: BrainyDataConfig['storage'],
|
||||||
|
currentStats?: Partial<CacheUsageStats>
|
||||||
|
): AutoConfigResult {
|
||||||
|
// Update stats with current information
|
||||||
|
if (currentStats) {
|
||||||
|
this.stats = { ...this.stats, ...currentStats }
|
||||||
|
}
|
||||||
|
|
||||||
|
// Detect environment characteristics
|
||||||
|
this.detectEnvironment(storageConfig)
|
||||||
|
|
||||||
|
// Generate optimal configuration
|
||||||
|
const result = this.generateOptimalConfig()
|
||||||
|
|
||||||
|
// Store for learning
|
||||||
|
this.configHistory.push(result)
|
||||||
|
this.lastOptimization = Date.now()
|
||||||
|
|
||||||
|
return result
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Dynamically adjust configuration based on runtime performance
|
||||||
|
*/
|
||||||
|
public adaptConfiguration(
|
||||||
|
currentConfig: SearchCacheConfig,
|
||||||
|
performanceMetrics: {
|
||||||
|
hitRate: number
|
||||||
|
avgResponseTime: number
|
||||||
|
memoryUsage: number
|
||||||
|
externalChangesDetected: number
|
||||||
|
timeSinceLastChange: number
|
||||||
|
}
|
||||||
|
): AutoConfigResult | null {
|
||||||
|
const reasoning: string[] = []
|
||||||
|
let needsUpdate = false
|
||||||
|
|
||||||
|
// Check if we should update (don't over-optimize)
|
||||||
|
if (Date.now() - this.lastOptimization < 60000) {
|
||||||
|
return null // Wait at least 1 minute between optimizations
|
||||||
|
}
|
||||||
|
|
||||||
|
// Analyze performance patterns
|
||||||
|
const adaptations: Partial<SearchCacheConfig> = {}
|
||||||
|
|
||||||
|
// Low hit rate → adjust cache size or TTL
|
||||||
|
if (performanceMetrics.hitRate < 0.3) {
|
||||||
|
if (performanceMetrics.externalChangesDetected > 5) {
|
||||||
|
// Too many external changes → shorter TTL
|
||||||
|
adaptations.maxAge = Math.max(60000, currentConfig.maxAge! * 0.7)
|
||||||
|
reasoning.push('Reduced cache TTL due to frequent external changes')
|
||||||
|
needsUpdate = true
|
||||||
|
} else {
|
||||||
|
// Expand cache size for better hit rate
|
||||||
|
adaptations.maxSize = Math.min(500, (currentConfig.maxSize || 100) * 1.5)
|
||||||
|
reasoning.push('Increased cache size due to low hit rate')
|
||||||
|
needsUpdate = true
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// High hit rate but slow responses → might need cache warming
|
||||||
|
if (performanceMetrics.hitRate > 0.8 && performanceMetrics.avgResponseTime > 100) {
|
||||||
|
reasoning.push('High hit rate but slow responses - consider cache warming')
|
||||||
|
}
|
||||||
|
|
||||||
|
// Memory pressure → reduce cache size
|
||||||
|
if (performanceMetrics.memoryUsage > 100 * 1024 * 1024) { // 100MB
|
||||||
|
adaptations.maxSize = Math.max(20, (currentConfig.maxSize || 100) * 0.7)
|
||||||
|
reasoning.push('Reduced cache size due to memory pressure')
|
||||||
|
needsUpdate = true
|
||||||
|
}
|
||||||
|
|
||||||
|
// Recent external changes → adaptive TTL
|
||||||
|
if (performanceMetrics.timeSinceLastChange < 30000) { // 30 seconds
|
||||||
|
adaptations.maxAge = Math.max(30000, currentConfig.maxAge! * 0.8)
|
||||||
|
reasoning.push('Shortened TTL due to recent external changes')
|
||||||
|
needsUpdate = true
|
||||||
|
}
|
||||||
|
|
||||||
|
if (!needsUpdate) {
|
||||||
|
return null
|
||||||
|
}
|
||||||
|
|
||||||
|
const newCacheConfig: SearchCacheConfig = {
|
||||||
|
...currentConfig,
|
||||||
|
...adaptations
|
||||||
|
}
|
||||||
|
|
||||||
|
const newRealtimeConfig = this.calculateRealtimeConfig()
|
||||||
|
|
||||||
|
return {
|
||||||
|
cacheConfig: newCacheConfig,
|
||||||
|
realtimeConfig: newRealtimeConfig,
|
||||||
|
reasoning
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Get recommended configuration for specific use case
|
||||||
|
*/
|
||||||
|
public getRecommendedConfig(useCase: 'high-consistency' | 'balanced' | 'performance-first'): AutoConfigResult {
|
||||||
|
const configs = {
|
||||||
|
'high-consistency': {
|
||||||
|
cache: { maxAge: 120000, maxSize: 50 },
|
||||||
|
realtime: { interval: 15000, enabled: true },
|
||||||
|
reasoning: ['Optimized for data consistency and real-time updates']
|
||||||
|
},
|
||||||
|
'balanced': {
|
||||||
|
cache: { maxAge: 300000, maxSize: 100 },
|
||||||
|
realtime: { interval: 30000, enabled: true },
|
||||||
|
reasoning: ['Balanced performance and consistency']
|
||||||
|
},
|
||||||
|
'performance-first': {
|
||||||
|
cache: { maxAge: 600000, maxSize: 200 },
|
||||||
|
realtime: { interval: 60000, enabled: true },
|
||||||
|
reasoning: ['Optimized for maximum cache performance']
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
const config = configs[useCase]
|
||||||
|
return {
|
||||||
|
cacheConfig: {
|
||||||
|
enabled: true,
|
||||||
|
...config.cache
|
||||||
|
},
|
||||||
|
realtimeConfig: {
|
||||||
|
updateIndex: true,
|
||||||
|
updateStatistics: true,
|
||||||
|
...config.realtime
|
||||||
|
},
|
||||||
|
reasoning: config.reasoning
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Learn from usage patterns and improve recommendations
|
||||||
|
*/
|
||||||
|
public learnFromUsage(usageData: {
|
||||||
|
queryPatterns: string[]
|
||||||
|
responseTime: number
|
||||||
|
cacheHits: number
|
||||||
|
totalQueries: number
|
||||||
|
dataChanges: number
|
||||||
|
timeWindow: number
|
||||||
|
}): void {
|
||||||
|
// Update internal stats for better future recommendations
|
||||||
|
this.stats.totalQueries += usageData.totalQueries
|
||||||
|
this.stats.repeatQueries += usageData.cacheHits
|
||||||
|
this.stats.avgQueryTime = (this.stats.avgQueryTime + usageData.responseTime) / 2
|
||||||
|
this.stats.changeFrequency = usageData.dataChanges / (usageData.timeWindow / 60000)
|
||||||
|
|
||||||
|
// Calculate read/write ratio
|
||||||
|
const writes = usageData.dataChanges
|
||||||
|
const reads = usageData.totalQueries
|
||||||
|
this.stats.readWriteRatio = reads > 0 ? reads / Math.max(writes, 1) : 10
|
||||||
|
}
|
||||||
|
|
||||||
|
private detectEnvironment(storageConfig?: BrainyDataConfig['storage']): void {
|
||||||
|
// Detect storage type
|
||||||
|
if (storageConfig?.s3Storage || storageConfig?.customS3Storage) {
|
||||||
|
this.stats.storageType = 's3'
|
||||||
|
this.stats.isDistributed = true
|
||||||
|
} else if (storageConfig?.forceFileSystemStorage) {
|
||||||
|
this.stats.storageType = 'filesystem'
|
||||||
|
} else if (storageConfig?.forceMemoryStorage) {
|
||||||
|
this.stats.storageType = 'memory'
|
||||||
|
} else {
|
||||||
|
// Auto-detect browser vs Node.js
|
||||||
|
this.stats.storageType = typeof window !== 'undefined' ? 'opfs' : 'filesystem'
|
||||||
|
}
|
||||||
|
|
||||||
|
// Detect distributed mode indicators
|
||||||
|
this.stats.isDistributed = this.stats.isDistributed ||
|
||||||
|
Boolean(storageConfig?.s3Storage || storageConfig?.customS3Storage)
|
||||||
|
}
|
||||||
|
|
||||||
|
private generateOptimalConfig(): AutoConfigResult {
|
||||||
|
const reasoning: string[] = []
|
||||||
|
|
||||||
|
// Base configuration
|
||||||
|
let cacheConfig: SearchCacheConfig = {
|
||||||
|
enabled: true,
|
||||||
|
maxSize: 100,
|
||||||
|
maxAge: 300000, // 5 minutes
|
||||||
|
hitCountWeight: 0.3
|
||||||
|
}
|
||||||
|
|
||||||
|
let realtimeConfig = {
|
||||||
|
enabled: false,
|
||||||
|
interval: 60000,
|
||||||
|
updateIndex: true,
|
||||||
|
updateStatistics: true
|
||||||
|
}
|
||||||
|
|
||||||
|
// Adjust for storage type
|
||||||
|
if (this.stats.storageType === 's3' || this.stats.isDistributed) {
|
||||||
|
cacheConfig.maxAge = 180000 // 3 minutes for distributed
|
||||||
|
realtimeConfig.enabled = true
|
||||||
|
realtimeConfig.interval = 30000 // 30 seconds
|
||||||
|
reasoning.push('Distributed storage detected - enabled real-time updates')
|
||||||
|
reasoning.push('Reduced cache TTL for distributed consistency')
|
||||||
|
}
|
||||||
|
|
||||||
|
// Adjust for read/write patterns
|
||||||
|
if (this.stats.readWriteRatio > 20) {
|
||||||
|
// Read-heavy workload
|
||||||
|
cacheConfig.maxSize = Math.min(300, (cacheConfig.maxSize || 100) * 2)
|
||||||
|
cacheConfig.maxAge = Math.min(900000, (cacheConfig.maxAge || 300000) * 1.5) // Up to 15 minutes
|
||||||
|
reasoning.push('Read-heavy workload detected - increased cache size and TTL')
|
||||||
|
} else if (this.stats.readWriteRatio < 5) {
|
||||||
|
// Write-heavy workload
|
||||||
|
cacheConfig.maxSize = Math.max(50, (cacheConfig.maxSize || 100) * 0.7)
|
||||||
|
cacheConfig.maxAge = Math.max(60000, (cacheConfig.maxAge || 300000) * 0.6)
|
||||||
|
reasoning.push('Write-heavy workload detected - reduced cache size and TTL')
|
||||||
|
}
|
||||||
|
|
||||||
|
// Adjust for change frequency
|
||||||
|
if (this.stats.changeFrequency > 10) { // More than 10 changes per minute
|
||||||
|
realtimeConfig.interval = Math.max(10000, realtimeConfig.interval * 0.5)
|
||||||
|
cacheConfig.maxAge = Math.max(30000, (cacheConfig.maxAge || 300000) * 0.5)
|
||||||
|
reasoning.push('High change frequency detected - increased update frequency')
|
||||||
|
}
|
||||||
|
|
||||||
|
// Memory constraints
|
||||||
|
if (this.detectMemoryConstraints()) {
|
||||||
|
cacheConfig.maxSize = Math.max(20, (cacheConfig.maxSize || 100) * 0.6)
|
||||||
|
reasoning.push('Memory constraints detected - reduced cache size')
|
||||||
|
}
|
||||||
|
|
||||||
|
// Performance optimization
|
||||||
|
if (this.stats.avgQueryTime > 200) {
|
||||||
|
cacheConfig.maxSize = Math.min(500, (cacheConfig.maxSize || 100) * 1.5)
|
||||||
|
reasoning.push('Slow queries detected - increased cache size')
|
||||||
|
}
|
||||||
|
|
||||||
|
return {
|
||||||
|
cacheConfig,
|
||||||
|
realtimeConfig,
|
||||||
|
reasoning
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
private calculateRealtimeConfig() {
|
||||||
|
return {
|
||||||
|
enabled: this.stats.isDistributed || this.stats.changeFrequency > 1,
|
||||||
|
interval: this.stats.isDistributed ? 30000 : 60000,
|
||||||
|
updateIndex: true,
|
||||||
|
updateStatistics: true
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
private detectMemoryConstraints(): boolean {
|
||||||
|
// Simple heuristic for memory constraints
|
||||||
|
try {
|
||||||
|
if (typeof performance !== 'undefined' && 'memory' in performance) {
|
||||||
|
const memInfo = (performance as any).memory
|
||||||
|
return memInfo.usedJSHeapSize > memInfo.jsHeapSizeLimit * 0.8
|
||||||
|
}
|
||||||
|
} catch (e) {
|
||||||
|
// Ignore errors
|
||||||
|
}
|
||||||
|
|
||||||
|
// Default assumption for constrained environments
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Get human-readable explanation of current configuration
|
||||||
|
*/
|
||||||
|
public getConfigExplanation(config: AutoConfigResult): string {
|
||||||
|
const lines = [
|
||||||
|
'🤖 Brainy Auto-Configuration:',
|
||||||
|
'',
|
||||||
|
`📊 Cache: ${config.cacheConfig.maxSize} queries, ${config.cacheConfig.maxAge! / 1000}s TTL`,
|
||||||
|
`🔄 Updates: ${config.realtimeConfig.enabled ? `Every ${(config.realtimeConfig.interval || 30000) / 1000}s` : 'Disabled'}`,
|
||||||
|
'',
|
||||||
|
'🎯 Optimizations applied:'
|
||||||
|
]
|
||||||
|
|
||||||
|
config.reasoning.forEach(reason => {
|
||||||
|
lines.push(` • ${reason}`)
|
||||||
|
})
|
||||||
|
|
||||||
|
return lines.join('\n')
|
||||||
|
}
|
||||||
|
}
|
||||||
308
src/utils/searchCache.ts
Normal file
308
src/utils/searchCache.ts
Normal file
|
|
@ -0,0 +1,308 @@
|
||||||
|
/**
|
||||||
|
* SearchCache - Caches search results for improved performance
|
||||||
|
*/
|
||||||
|
|
||||||
|
import { SearchResult } from '../coreTypes.js'
|
||||||
|
|
||||||
|
export interface CacheEntry<T = any> {
|
||||||
|
results: SearchResult<T>[]
|
||||||
|
timestamp: number
|
||||||
|
hits: number
|
||||||
|
}
|
||||||
|
|
||||||
|
export interface SearchCacheConfig {
|
||||||
|
maxAge?: number // Maximum age in milliseconds (default: 5 minutes)
|
||||||
|
maxSize?: number // Maximum number of cached queries (default: 100)
|
||||||
|
enabled?: boolean // Whether caching is enabled (default: true)
|
||||||
|
hitCountWeight?: number // Weight for hit count in eviction policy (default: 0.3)
|
||||||
|
}
|
||||||
|
|
||||||
|
export class SearchCache<T = any> {
|
||||||
|
private cache = new Map<string, CacheEntry<T>>()
|
||||||
|
private maxAge: number
|
||||||
|
private maxSize: number
|
||||||
|
private enabled: boolean
|
||||||
|
private hitCountWeight: number
|
||||||
|
|
||||||
|
// Cache statistics
|
||||||
|
private hits = 0
|
||||||
|
private misses = 0
|
||||||
|
private evictions = 0
|
||||||
|
|
||||||
|
constructor(config: SearchCacheConfig = {}) {
|
||||||
|
this.maxAge = config.maxAge ?? 5 * 60 * 1000 // 5 minutes
|
||||||
|
this.maxSize = config.maxSize ?? 100
|
||||||
|
this.enabled = config.enabled ?? true
|
||||||
|
this.hitCountWeight = config.hitCountWeight ?? 0.3
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Generate cache key from search parameters
|
||||||
|
*/
|
||||||
|
getCacheKey(
|
||||||
|
query: any,
|
||||||
|
k: number,
|
||||||
|
options: Record<string, any> = {}
|
||||||
|
): string {
|
||||||
|
// Create a normalized key that ignores order of options
|
||||||
|
const normalizedOptions = Object.keys(options)
|
||||||
|
.sort()
|
||||||
|
.reduce((acc, key) => {
|
||||||
|
// Skip cache-related options
|
||||||
|
if (key === 'skipCache' || key === 'useStreaming') return acc
|
||||||
|
acc[key] = options[key]
|
||||||
|
return acc
|
||||||
|
}, {} as Record<string, any>)
|
||||||
|
|
||||||
|
return JSON.stringify({
|
||||||
|
query: typeof query === 'object' ? JSON.stringify(query) : query,
|
||||||
|
k,
|
||||||
|
...normalizedOptions
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Get cached results if available and not expired
|
||||||
|
*/
|
||||||
|
get(key: string): SearchResult<T>[] | null {
|
||||||
|
if (!this.enabled) return null
|
||||||
|
|
||||||
|
const entry = this.cache.get(key)
|
||||||
|
if (!entry) {
|
||||||
|
this.misses++
|
||||||
|
return null
|
||||||
|
}
|
||||||
|
|
||||||
|
// Check if expired
|
||||||
|
if (Date.now() - entry.timestamp > this.maxAge) {
|
||||||
|
this.cache.delete(key)
|
||||||
|
this.misses++
|
||||||
|
return null
|
||||||
|
}
|
||||||
|
|
||||||
|
// Update hit count and statistics
|
||||||
|
entry.hits++
|
||||||
|
this.hits++
|
||||||
|
return entry.results
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Cache search results
|
||||||
|
*/
|
||||||
|
set(key: string, results: SearchResult<T>[]): void {
|
||||||
|
if (!this.enabled) return
|
||||||
|
|
||||||
|
// Evict if cache is full
|
||||||
|
if (this.cache.size >= this.maxSize) {
|
||||||
|
this.evictOldest()
|
||||||
|
}
|
||||||
|
|
||||||
|
this.cache.set(key, {
|
||||||
|
results: [...results], // Deep copy to prevent mutations
|
||||||
|
timestamp: Date.now(),
|
||||||
|
hits: 0
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Evict the oldest entry based on timestamp and hit count
|
||||||
|
*/
|
||||||
|
private evictOldest(): void {
|
||||||
|
let oldestKey: string | null = null
|
||||||
|
let oldestScore = Infinity
|
||||||
|
|
||||||
|
const now = Date.now()
|
||||||
|
|
||||||
|
for (const [key, entry] of this.cache.entries()) {
|
||||||
|
// Score combines age and inverse hit count
|
||||||
|
const age = now - entry.timestamp
|
||||||
|
const hitScore = entry.hits > 0 ? 1 / entry.hits : 1
|
||||||
|
const score = age + (hitScore * this.hitCountWeight * this.maxAge)
|
||||||
|
|
||||||
|
if (score < oldestScore) {
|
||||||
|
oldestScore = score
|
||||||
|
oldestKey = key
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
if (oldestKey) {
|
||||||
|
this.cache.delete(oldestKey)
|
||||||
|
this.evictions++
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Clear all cached results
|
||||||
|
*/
|
||||||
|
clear(): void {
|
||||||
|
this.cache.clear()
|
||||||
|
this.hits = 0
|
||||||
|
this.misses = 0
|
||||||
|
this.evictions = 0
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Invalidate cache entries that might be affected by data changes
|
||||||
|
*/
|
||||||
|
invalidate(pattern?: string | RegExp): void {
|
||||||
|
if (!pattern) {
|
||||||
|
this.clear()
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
const keysToDelete: string[] = []
|
||||||
|
|
||||||
|
for (const key of this.cache.keys()) {
|
||||||
|
const shouldDelete = typeof pattern === 'string'
|
||||||
|
? key.includes(pattern)
|
||||||
|
: pattern.test(key)
|
||||||
|
|
||||||
|
if (shouldDelete) {
|
||||||
|
keysToDelete.push(key)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
keysToDelete.forEach(key => this.cache.delete(key))
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Smart invalidation for real-time data updates
|
||||||
|
* Only clears cache if it's getting stale or if data changes significantly
|
||||||
|
*/
|
||||||
|
invalidateOnDataChange(changeType?: 'add' | 'update' | 'delete'): void {
|
||||||
|
// For now, clear all caches on data changes to ensure consistency
|
||||||
|
// In the future, we could implement more sophisticated invalidation
|
||||||
|
// based on the type of change and affected data
|
||||||
|
this.clear()
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Check if cache entries have expired and remove them
|
||||||
|
* This is especially important in distributed scenarios where
|
||||||
|
* real-time updates might be delayed or missed
|
||||||
|
*/
|
||||||
|
cleanupExpiredEntries(): number {
|
||||||
|
const now = Date.now()
|
||||||
|
const keysToDelete: string[] = []
|
||||||
|
|
||||||
|
for (const [key, entry] of this.cache.entries()) {
|
||||||
|
if (now - entry.timestamp > this.maxAge) {
|
||||||
|
keysToDelete.push(key)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
keysToDelete.forEach(key => this.cache.delete(key))
|
||||||
|
return keysToDelete.length
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Get cache statistics
|
||||||
|
*/
|
||||||
|
getStats() {
|
||||||
|
const total = this.hits + this.misses
|
||||||
|
return {
|
||||||
|
hits: this.hits,
|
||||||
|
misses: this.misses,
|
||||||
|
evictions: this.evictions,
|
||||||
|
hitRate: total > 0 ? this.hits / total : 0,
|
||||||
|
size: this.cache.size,
|
||||||
|
maxSize: this.maxSize,
|
||||||
|
enabled: this.enabled
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Enable or disable caching
|
||||||
|
*/
|
||||||
|
setEnabled(enabled: boolean): void {
|
||||||
|
Object.defineProperty(this, 'enabled', { value: enabled, writable: false })
|
||||||
|
if (!enabled) {
|
||||||
|
this.clear()
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Get memory usage estimate in bytes
|
||||||
|
*/
|
||||||
|
getMemoryUsage(): number {
|
||||||
|
let totalSize = 0
|
||||||
|
|
||||||
|
for (const [key, entry] of this.cache.entries()) {
|
||||||
|
// Estimate key size
|
||||||
|
totalSize += key.length * 2 // UTF-16 characters
|
||||||
|
|
||||||
|
// Estimate entry size
|
||||||
|
totalSize += JSON.stringify(entry.results).length * 2
|
||||||
|
totalSize += 16 // timestamp + hits (8 bytes each)
|
||||||
|
}
|
||||||
|
|
||||||
|
return totalSize
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Get current cache configuration
|
||||||
|
*/
|
||||||
|
getConfig(): SearchCacheConfig {
|
||||||
|
return {
|
||||||
|
enabled: this.enabled,
|
||||||
|
maxSize: this.maxSize,
|
||||||
|
maxAge: this.maxAge,
|
||||||
|
hitCountWeight: this.hitCountWeight
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Update cache configuration dynamically
|
||||||
|
*/
|
||||||
|
updateConfig(newConfig: Partial<SearchCacheConfig>): void {
|
||||||
|
if (newConfig.enabled !== undefined) {
|
||||||
|
this.enabled = newConfig.enabled
|
||||||
|
}
|
||||||
|
if (newConfig.maxSize !== undefined) {
|
||||||
|
this.maxSize = newConfig.maxSize
|
||||||
|
// Trigger eviction if current size exceeds new limit
|
||||||
|
this.evictIfNeeded()
|
||||||
|
}
|
||||||
|
if (newConfig.maxAge !== undefined) {
|
||||||
|
this.maxAge = newConfig.maxAge
|
||||||
|
// Clean up entries that are now expired with new TTL
|
||||||
|
this.cleanupExpiredEntries()
|
||||||
|
}
|
||||||
|
if (newConfig.hitCountWeight !== undefined) {
|
||||||
|
this.hitCountWeight = newConfig.hitCountWeight
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Evict entries if cache exceeds maxSize
|
||||||
|
*/
|
||||||
|
private evictIfNeeded(): void {
|
||||||
|
if (this.cache.size <= this.maxSize) {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
// Calculate eviction score for each entry (same logic as existing eviction)
|
||||||
|
const entries = Array.from(this.cache.entries()).map(([key, entry]) => {
|
||||||
|
const age = Date.now() - entry.timestamp
|
||||||
|
const hitCount = entry.hits
|
||||||
|
|
||||||
|
// Eviction score: lower is more likely to be evicted
|
||||||
|
// Combines age and hit count (weighted by hitCountWeight)
|
||||||
|
const ageScore = age / this.maxAge
|
||||||
|
const hitScore = 1 / (hitCount + 1) // Inverse of hits (more hits = lower score)
|
||||||
|
const score = ageScore * (1 - this.hitCountWeight) + hitScore * this.hitCountWeight
|
||||||
|
|
||||||
|
return { key, entry, score }
|
||||||
|
})
|
||||||
|
|
||||||
|
// Sort by score (lowest first - these will be evicted)
|
||||||
|
entries.sort((a, b) => a.score - b.score)
|
||||||
|
|
||||||
|
// Evict entries until we're under the limit
|
||||||
|
const toEvict = entries.slice(0, this.cache.size - this.maxSize)
|
||||||
|
toEvict.forEach(({ key }) => {
|
||||||
|
this.cache.delete(key)
|
||||||
|
this.evictions++
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
219
tests/auto-configuration.test.ts
Normal file
219
tests/auto-configuration.test.ts
Normal file
|
|
@ -0,0 +1,219 @@
|
||||||
|
/**
|
||||||
|
* Tests for automatic cache configuration system
|
||||||
|
*/
|
||||||
|
|
||||||
|
import { describe, it, expect, beforeEach, afterEach } from 'vitest'
|
||||||
|
import { BrainyData } from '../src/brainyData.js'
|
||||||
|
import { cleanupWorkerPools } from '../src/utils/index.js'
|
||||||
|
|
||||||
|
describe('Auto-Configuration System', () => {
|
||||||
|
let brainy: BrainyData
|
||||||
|
|
||||||
|
afterEach(async () => {
|
||||||
|
if (brainy) {
|
||||||
|
await brainy.clear()
|
||||||
|
}
|
||||||
|
await cleanupWorkerPools()
|
||||||
|
})
|
||||||
|
|
||||||
|
describe('Automatic Cache Configuration', () => {
|
||||||
|
it('should auto-configure cache for memory storage', async () => {
|
||||||
|
// Create instance without explicit cache configuration
|
||||||
|
brainy = new BrainyData({
|
||||||
|
storage: { forceMemoryStorage: true },
|
||||||
|
logging: { verbose: false } // Disable logging for test
|
||||||
|
})
|
||||||
|
await brainy.init()
|
||||||
|
|
||||||
|
const cacheStats = brainy.getCacheStats()
|
||||||
|
|
||||||
|
// Cache should be enabled by default
|
||||||
|
expect(cacheStats.search.enabled).toBe(true)
|
||||||
|
expect(cacheStats.search.maxSize).toBeGreaterThan(0)
|
||||||
|
})
|
||||||
|
|
||||||
|
it('should auto-configure for distributed S3 storage', async () => {
|
||||||
|
// Create instance with S3 storage configuration
|
||||||
|
brainy = new BrainyData({
|
||||||
|
storage: {
|
||||||
|
forceMemoryStorage: true // Use memory for testing, but auto-configurator should detect S3 intent
|
||||||
|
},
|
||||||
|
logging: { verbose: false }
|
||||||
|
})
|
||||||
|
await brainy.init()
|
||||||
|
|
||||||
|
const cacheStats = brainy.getCacheStats()
|
||||||
|
const realtimeConfig = brainy.getRealtimeUpdateConfig()
|
||||||
|
|
||||||
|
// With memory storage, real-time updates should be disabled by default
|
||||||
|
// But cache should still be properly configured
|
||||||
|
expect(cacheStats.search.enabled).toBe(true)
|
||||||
|
expect(cacheStats.search.maxSize).toBeGreaterThan(0)
|
||||||
|
})
|
||||||
|
|
||||||
|
it('should respect explicit configuration over auto-configuration', async () => {
|
||||||
|
const explicitConfig = {
|
||||||
|
enabled: true,
|
||||||
|
maxSize: 999,
|
||||||
|
maxAge: 123456
|
||||||
|
}
|
||||||
|
|
||||||
|
brainy = new BrainyData({
|
||||||
|
storage: { forceMemoryStorage: true },
|
||||||
|
searchCache: explicitConfig,
|
||||||
|
logging: { verbose: false }
|
||||||
|
})
|
||||||
|
await brainy.init()
|
||||||
|
|
||||||
|
const cacheStats = brainy.getCacheStats()
|
||||||
|
|
||||||
|
// Should use explicit configuration
|
||||||
|
expect(cacheStats.search.enabled).toBe(true)
|
||||||
|
expect(cacheStats.search.maxSize).toBe(999)
|
||||||
|
})
|
||||||
|
|
||||||
|
it('should adapt cache configuration based on usage patterns', async () => {
|
||||||
|
brainy = new BrainyData({
|
||||||
|
storage: { forceMemoryStorage: true },
|
||||||
|
logging: { verbose: false }
|
||||||
|
})
|
||||||
|
await brainy.init()
|
||||||
|
|
||||||
|
// Add some test data
|
||||||
|
for (let i = 0; i < 20; i++) {
|
||||||
|
await brainy.add({
|
||||||
|
id: `test-${i}`,
|
||||||
|
text: `test data ${i}`
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
// Get initial cache configuration
|
||||||
|
const initialStats = brainy.getCacheStats()
|
||||||
|
const initialMaxSize = initialStats.search.maxSize
|
||||||
|
|
||||||
|
// Perform many searches to create usage patterns
|
||||||
|
for (let i = 0; i < 10; i++) {
|
||||||
|
await brainy.search(`test data ${i % 5}`, 5)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Manual trigger of adaptation (normally happens during real-time updates)
|
||||||
|
// Since we're testing with memory storage, we'll manually check the configurator is working
|
||||||
|
const currentStats = brainy.getCacheStats()
|
||||||
|
|
||||||
|
// Cache should still be operational and have reasonable settings
|
||||||
|
expect(currentStats.search.enabled).toBe(true)
|
||||||
|
expect(currentStats.search.maxSize).toBeGreaterThan(0)
|
||||||
|
expect(currentStats.search.hits).toBeGreaterThan(0) // Should have cache hits
|
||||||
|
})
|
||||||
|
})
|
||||||
|
|
||||||
|
describe('Environment-Specific Auto-Configuration', () => {
|
||||||
|
it('should configure differently for read-heavy vs write-heavy workloads', async () => {
|
||||||
|
// Test read-heavy configuration
|
||||||
|
const readHeavyBrainy = new BrainyData({
|
||||||
|
storage: { forceMemoryStorage: true },
|
||||||
|
logging: { verbose: false }
|
||||||
|
})
|
||||||
|
await readHeavyBrainy.init()
|
||||||
|
|
||||||
|
// Add some data
|
||||||
|
await readHeavyBrainy.add({ id: 'test-1', text: 'test data' })
|
||||||
|
|
||||||
|
// Simulate read-heavy usage
|
||||||
|
for (let i = 0; i < 20; i++) {
|
||||||
|
await readHeavyBrainy.search('test data', 5)
|
||||||
|
}
|
||||||
|
|
||||||
|
const readHeavyStats = readHeavyBrainy.getCacheStats()
|
||||||
|
|
||||||
|
// Should have good cache performance
|
||||||
|
expect(readHeavyStats.search.hitRate).toBeGreaterThan(0.5)
|
||||||
|
expect(readHeavyStats.search.enabled).toBe(true)
|
||||||
|
|
||||||
|
await readHeavyBrainy.clear()
|
||||||
|
})
|
||||||
|
|
||||||
|
it('should handle zero-configuration scenarios gracefully', async () => {
|
||||||
|
// Create instance with absolutely minimal configuration
|
||||||
|
brainy = new BrainyData({
|
||||||
|
logging: { verbose: false }
|
||||||
|
})
|
||||||
|
await brainy.init()
|
||||||
|
|
||||||
|
// Should still work with auto-detected configuration
|
||||||
|
await brainy.add({ text: 'auto-config test unique phrase' })
|
||||||
|
const results = await brainy.search('unique phrase', 5)
|
||||||
|
|
||||||
|
expect(results.length).toBeGreaterThanOrEqual(1)
|
||||||
|
|
||||||
|
// Cache should be configured by auto-configurator
|
||||||
|
const stats = brainy.getCacheStats()
|
||||||
|
expect(stats.search.enabled).toBe(true)
|
||||||
|
expect(stats.search.maxSize).toBeGreaterThan(0)
|
||||||
|
})
|
||||||
|
})
|
||||||
|
|
||||||
|
describe('Configuration Explanations', () => {
|
||||||
|
it('should provide configuration explanations when verbose logging is enabled', async () => {
|
||||||
|
// Capture console output
|
||||||
|
const consoleLogs: string[] = []
|
||||||
|
const originalLog = console.log
|
||||||
|
console.log = (...args: any[]) => {
|
||||||
|
consoleLogs.push(args.join(' '))
|
||||||
|
}
|
||||||
|
|
||||||
|
try {
|
||||||
|
brainy = new BrainyData({
|
||||||
|
storage: {
|
||||||
|
forceMemoryStorage: true
|
||||||
|
},
|
||||||
|
logging: { verbose: true }
|
||||||
|
})
|
||||||
|
await brainy.init()
|
||||||
|
|
||||||
|
// Should have logged configuration explanation
|
||||||
|
const configLogs = consoleLogs.filter(log =>
|
||||||
|
log.includes('Auto-Configuration') ||
|
||||||
|
log.includes('Distributed storage detected') ||
|
||||||
|
log.includes('Cache:') ||
|
||||||
|
log.includes('Updates:')
|
||||||
|
)
|
||||||
|
|
||||||
|
expect(configLogs.length).toBeGreaterThan(0)
|
||||||
|
} finally {
|
||||||
|
console.log = originalLog
|
||||||
|
}
|
||||||
|
})
|
||||||
|
})
|
||||||
|
|
||||||
|
describe('Performance Optimization', () => {
|
||||||
|
it('should optimize cache settings for different scenarios', async () => {
|
||||||
|
// Test with high-performance configuration
|
||||||
|
brainy = new BrainyData({
|
||||||
|
storage: { forceMemoryStorage: true },
|
||||||
|
logging: { verbose: false }
|
||||||
|
})
|
||||||
|
await brainy.init()
|
||||||
|
|
||||||
|
// Add test data
|
||||||
|
for (let i = 0; i < 50; i++) {
|
||||||
|
await brainy.add({
|
||||||
|
id: `perf-test-${i}`,
|
||||||
|
text: `performance test data ${i}`
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
// Perform searches to warm up cache
|
||||||
|
for (let i = 0; i < 10; i++) {
|
||||||
|
await brainy.search(`performance test data ${i % 5}`, 10)
|
||||||
|
}
|
||||||
|
|
||||||
|
const stats = brainy.getCacheStats()
|
||||||
|
|
||||||
|
// Should have good performance characteristics
|
||||||
|
expect(stats.search.hits).toBeGreaterThan(0)
|
||||||
|
expect(stats.search.hitRate).toBeGreaterThan(0.3) // At least 30% hit rate
|
||||||
|
expect(stats.searchMemoryUsage).toBeGreaterThan(0)
|
||||||
|
})
|
||||||
|
})
|
||||||
|
})
|
||||||
245
tests/distributed-caching.test.ts
Normal file
245
tests/distributed-caching.test.ts
Normal file
|
|
@ -0,0 +1,245 @@
|
||||||
|
/**
|
||||||
|
* Tests for distributed caching behavior with shared storage
|
||||||
|
*/
|
||||||
|
|
||||||
|
import { describe, it, expect, beforeEach, afterEach, vi } from 'vitest'
|
||||||
|
import { BrainyData } from '../src/brainyData.js'
|
||||||
|
import { cleanupWorkerPools } from '../src/utils/index.js'
|
||||||
|
|
||||||
|
describe('Distributed Caching', () => {
|
||||||
|
let serviceA: BrainyData
|
||||||
|
let serviceB: BrainyData
|
||||||
|
|
||||||
|
beforeEach(async () => {
|
||||||
|
// Mock the checkForUpdates to simulate real-time updates without waiting
|
||||||
|
const mockCheckForUpdates = vi.fn()
|
||||||
|
|
||||||
|
// Create two services sharing the same storage configuration
|
||||||
|
const sharedConfig = {
|
||||||
|
storage: {
|
||||||
|
forceMemoryStorage: true // Use memory storage for testing
|
||||||
|
},
|
||||||
|
searchCache: {
|
||||||
|
enabled: true,
|
||||||
|
maxSize: 50,
|
||||||
|
maxAge: 60000 // 1 minute
|
||||||
|
},
|
||||||
|
realtimeUpdates: {
|
||||||
|
enabled: true,
|
||||||
|
interval: 1000, // 1 second for testing
|
||||||
|
updateIndex: true,
|
||||||
|
updateStatistics: true
|
||||||
|
},
|
||||||
|
logging: {
|
||||||
|
verbose: false
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
serviceA = new BrainyData(sharedConfig)
|
||||||
|
serviceB = new BrainyData(sharedConfig)
|
||||||
|
|
||||||
|
await serviceA.init()
|
||||||
|
await serviceB.init()
|
||||||
|
})
|
||||||
|
|
||||||
|
afterEach(async () => {
|
||||||
|
await serviceA.clear()
|
||||||
|
await serviceB.clear()
|
||||||
|
await cleanupWorkerPools()
|
||||||
|
})
|
||||||
|
|
||||||
|
describe('Cache Invalidation on External Changes', () => {
|
||||||
|
it('should invalidate cache when external data changes are detected', async () => {
|
||||||
|
// Since we're using memory storage and separate instances for testing,
|
||||||
|
// we'll simulate distributed behavior within a single service
|
||||||
|
|
||||||
|
// Add initial data
|
||||||
|
await serviceA.add({
|
||||||
|
id: 'item-1',
|
||||||
|
text: 'initial data from service A'
|
||||||
|
})
|
||||||
|
|
||||||
|
// Search and cache the result
|
||||||
|
const results1 = await serviceA.search('initial data', 5)
|
||||||
|
expect(results1.length).toBe(1)
|
||||||
|
|
||||||
|
// Verify cache is populated
|
||||||
|
let stats = serviceA.getCacheStats()
|
||||||
|
expect(stats.search.size).toBe(1)
|
||||||
|
|
||||||
|
// Add more data (simulates external changes)
|
||||||
|
await serviceA.add({
|
||||||
|
id: 'item-2',
|
||||||
|
text: 'new data from service A'
|
||||||
|
})
|
||||||
|
|
||||||
|
// Cache should have been invalidated due to the add operation
|
||||||
|
stats = serviceA.getCacheStats()
|
||||||
|
expect(stats.search.size).toBe(0) // Cache cleared
|
||||||
|
|
||||||
|
// Search again - should get fresh results including new data
|
||||||
|
const results2 = await serviceA.search('data from service', 10)
|
||||||
|
expect(results2.length).toBe(2) // Should now see both items
|
||||||
|
|
||||||
|
stats = serviceA.getCacheStats()
|
||||||
|
// Cache should be rebuilt with fresh data
|
||||||
|
expect(stats.search.size).toBe(1) // New cache entry
|
||||||
|
})
|
||||||
|
|
||||||
|
it('should handle cache expiration gracefully', async () => {
|
||||||
|
// Create a service with very short cache TTL
|
||||||
|
const shortCacheService = new BrainyData({
|
||||||
|
storage: { forceMemoryStorage: true },
|
||||||
|
searchCache: {
|
||||||
|
enabled: true,
|
||||||
|
maxAge: 100 // 100ms - very short for testing
|
||||||
|
},
|
||||||
|
logging: { verbose: false }
|
||||||
|
})
|
||||||
|
await shortCacheService.init()
|
||||||
|
|
||||||
|
// Add data and search
|
||||||
|
await shortCacheService.add({
|
||||||
|
id: 'item-1',
|
||||||
|
text: 'short cache test'
|
||||||
|
})
|
||||||
|
|
||||||
|
const results1 = await shortCacheService.search('short cache', 5)
|
||||||
|
expect(results1.length).toBe(1)
|
||||||
|
|
||||||
|
// Wait for cache to expire
|
||||||
|
await new Promise(resolve => setTimeout(resolve, 150))
|
||||||
|
|
||||||
|
// Clean up expired entries
|
||||||
|
const expiredCount = shortCacheService['searchCache'].cleanupExpiredEntries()
|
||||||
|
expect(expiredCount).toBeGreaterThan(0)
|
||||||
|
|
||||||
|
// Search again - should work fine with fresh data
|
||||||
|
const results2 = await shortCacheService.search('short cache', 5)
|
||||||
|
expect(results2.length).toBe(1)
|
||||||
|
|
||||||
|
await shortCacheService.clear()
|
||||||
|
})
|
||||||
|
|
||||||
|
it('should provide cache statistics for monitoring', async () => {
|
||||||
|
// Add test data
|
||||||
|
for (let i = 0; i < 10; i++) {
|
||||||
|
await serviceA.add({
|
||||||
|
id: `item-${i}`,
|
||||||
|
text: `test data ${i}`
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
// Clear cache to start fresh
|
||||||
|
serviceA.clearCache()
|
||||||
|
|
||||||
|
// Perform searches to populate cache
|
||||||
|
await serviceA.search('test data', 5) // Miss
|
||||||
|
await serviceA.search('test data', 5) // Hit
|
||||||
|
await serviceA.search('test data', 3) // Miss (different k)
|
||||||
|
await serviceA.search('test data', 3) // Hit
|
||||||
|
|
||||||
|
const stats = serviceA.getCacheStats()
|
||||||
|
|
||||||
|
expect(stats.search.hits).toBe(2)
|
||||||
|
expect(stats.search.misses).toBe(2)
|
||||||
|
expect(stats.search.hitRate).toBe(0.5)
|
||||||
|
expect(stats.search.size).toBe(2) // Two different cache entries
|
||||||
|
expect(stats.searchMemoryUsage).toBeGreaterThan(0)
|
||||||
|
})
|
||||||
|
})
|
||||||
|
|
||||||
|
describe('Real-time Update Integration', () => {
|
||||||
|
it('should enable real-time updates for distributed scenarios', () => {
|
||||||
|
const config = serviceA.getRealtimeUpdateConfig()
|
||||||
|
expect(config.enabled).toBe(true)
|
||||||
|
expect(config.updateIndex).toBe(true)
|
||||||
|
expect(config.updateStatistics).toBe(true)
|
||||||
|
})
|
||||||
|
|
||||||
|
it('should handle skipCache option correctly', async () => {
|
||||||
|
// Add test data
|
||||||
|
await serviceA.add({
|
||||||
|
id: 'item-1',
|
||||||
|
text: 'skip cache test'
|
||||||
|
})
|
||||||
|
|
||||||
|
// Search with cache
|
||||||
|
const results1 = await serviceA.search('skip cache', 5)
|
||||||
|
expect(results1.length).toBe(1)
|
||||||
|
|
||||||
|
// Verify cache is populated
|
||||||
|
let stats = serviceA.getCacheStats()
|
||||||
|
expect(stats.search.size).toBe(1)
|
||||||
|
|
||||||
|
// Search with skipCache - should bypass cache
|
||||||
|
const results2 = await serviceA.search('skip cache', 5, { skipCache: true })
|
||||||
|
expect(results2.length).toBe(1)
|
||||||
|
|
||||||
|
// Cache size shouldn't increase
|
||||||
|
stats = serviceA.getCacheStats()
|
||||||
|
expect(stats.search.size).toBe(1) // Still just one entry
|
||||||
|
})
|
||||||
|
})
|
||||||
|
|
||||||
|
describe('Distributed Mode Best Practices', () => {
|
||||||
|
it('should work with recommended distributed settings', async () => {
|
||||||
|
const distributedService = new BrainyData({
|
||||||
|
storage: { forceMemoryStorage: true },
|
||||||
|
searchCache: {
|
||||||
|
enabled: true,
|
||||||
|
maxAge: 180000, // 3 minutes - shorter for distributed
|
||||||
|
maxSize: 100
|
||||||
|
},
|
||||||
|
realtimeUpdates: {
|
||||||
|
enabled: true,
|
||||||
|
interval: 30000, // 30 seconds
|
||||||
|
updateIndex: true,
|
||||||
|
updateStatistics: true
|
||||||
|
},
|
||||||
|
logging: { verbose: false }
|
||||||
|
})
|
||||||
|
|
||||||
|
await distributedService.init()
|
||||||
|
|
||||||
|
// Verify configuration
|
||||||
|
const config = distributedService.getRealtimeUpdateConfig()
|
||||||
|
expect(config.enabled).toBe(true)
|
||||||
|
expect(config.interval).toBe(30000)
|
||||||
|
|
||||||
|
const cacheStats = distributedService.getCacheStats()
|
||||||
|
expect(cacheStats.search.enabled).toBe(true)
|
||||||
|
|
||||||
|
await distributedService.clear()
|
||||||
|
})
|
||||||
|
|
||||||
|
it('should maintain performance with frequent external changes', async () => {
|
||||||
|
// Simulate a scenario with frequent external changes
|
||||||
|
const queries = ['query1', 'query2', 'query3', 'query4', 'query5']
|
||||||
|
|
||||||
|
// Add initial data
|
||||||
|
for (let i = 0; i < 20; i++) {
|
||||||
|
await serviceA.add({
|
||||||
|
id: `item-${i}`,
|
||||||
|
text: `data for query${(i % 5) + 1} item ${i}`
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
// Clear cache to start measurement
|
||||||
|
serviceA.clearCache()
|
||||||
|
|
||||||
|
// Perform multiple searches (some will be cache hits)
|
||||||
|
for (const query of queries) {
|
||||||
|
await serviceA.search(query, 5) // First search - cache miss
|
||||||
|
await serviceA.search(query, 5) // Second search - cache hit
|
||||||
|
}
|
||||||
|
|
||||||
|
const stats = serviceA.getCacheStats()
|
||||||
|
|
||||||
|
// Should have good hit rate despite distributed scenario
|
||||||
|
expect(stats.search.hitRate).toBeGreaterThan(0.4) // At least 40%
|
||||||
|
expect(stats.search.hits).toBeGreaterThan(0)
|
||||||
|
expect(stats.search.size).toBe(queries.length)
|
||||||
|
})
|
||||||
|
})
|
||||||
|
})
|
||||||
255
tests/pagination.test.ts
Normal file
255
tests/pagination.test.ts
Normal file
|
|
@ -0,0 +1,255 @@
|
||||||
|
/**
|
||||||
|
* Tests for offset-based pagination in search results
|
||||||
|
*/
|
||||||
|
|
||||||
|
import { describe, it, expect, beforeEach, afterEach } from 'vitest'
|
||||||
|
import { BrainyData } from '../src/brainyData.js'
|
||||||
|
import { cleanupWorkerPools } from '../src/utils/index.js'
|
||||||
|
|
||||||
|
describe('Pagination with Offset', () => {
|
||||||
|
let db: BrainyData
|
||||||
|
|
||||||
|
beforeEach(async () => {
|
||||||
|
// Initialize BrainyData with in-memory storage for testing
|
||||||
|
db = new BrainyData({
|
||||||
|
storage: {
|
||||||
|
forceMemoryStorage: true
|
||||||
|
},
|
||||||
|
logging: {
|
||||||
|
verbose: false
|
||||||
|
}
|
||||||
|
})
|
||||||
|
await db.init()
|
||||||
|
})
|
||||||
|
|
||||||
|
afterEach(async () => {
|
||||||
|
await db.clear()
|
||||||
|
await cleanupWorkerPools()
|
||||||
|
})
|
||||||
|
|
||||||
|
describe('Basic Offset Pagination', () => {
|
||||||
|
it('should return correct results with offset=0', async () => {
|
||||||
|
// Add test data
|
||||||
|
const testData = []
|
||||||
|
for (let i = 0; i < 20; i++) {
|
||||||
|
const item = {
|
||||||
|
id: `item-${i}`,
|
||||||
|
text: `test document ${i}`,
|
||||||
|
value: i
|
||||||
|
}
|
||||||
|
testData.push(item)
|
||||||
|
await db.add(item)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Search without offset (default offset=0)
|
||||||
|
const results = await db.search('test document', 5)
|
||||||
|
expect(results.length).toBe(5)
|
||||||
|
|
||||||
|
// Results should be the top 5 most similar
|
||||||
|
const resultIds = results.map(r => r.metadata.id)
|
||||||
|
expect(resultIds.length).toBe(5)
|
||||||
|
})
|
||||||
|
|
||||||
|
it('should skip results with offset > 0', async () => {
|
||||||
|
// Add test data
|
||||||
|
const testData = []
|
||||||
|
for (let i = 0; i < 20; i++) {
|
||||||
|
const item = {
|
||||||
|
id: `item-${i}`,
|
||||||
|
text: `test document ${i}`,
|
||||||
|
value: i
|
||||||
|
}
|
||||||
|
testData.push(item)
|
||||||
|
await db.add(item)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Get first page (no offset)
|
||||||
|
const firstPage = await db.search('test document', 5)
|
||||||
|
expect(firstPage.length).toBe(5)
|
||||||
|
const firstPageIds = firstPage.map(r => r.metadata.id)
|
||||||
|
|
||||||
|
// Get second page (offset=5)
|
||||||
|
const secondPage = await db.search('test document', 5, { offset: 5 })
|
||||||
|
expect(secondPage.length).toBe(5)
|
||||||
|
const secondPageIds = secondPage.map(r => r.metadata.id)
|
||||||
|
|
||||||
|
// Ensure no overlap between pages
|
||||||
|
const overlap = firstPageIds.filter(id => secondPageIds.includes(id))
|
||||||
|
expect(overlap.length).toBe(0)
|
||||||
|
})
|
||||||
|
|
||||||
|
it('should handle offset beyond available results', async () => {
|
||||||
|
// Add limited test data
|
||||||
|
for (let i = 0; i < 10; i++) {
|
||||||
|
await db.add({
|
||||||
|
id: `item-${i}`,
|
||||||
|
text: `test document ${i}`
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
// Search with offset beyond available results
|
||||||
|
const results = await db.search('test document', 5, { offset: 15 })
|
||||||
|
expect(results.length).toBe(0)
|
||||||
|
})
|
||||||
|
|
||||||
|
it('should return partial results when offset + k exceeds total', async () => {
|
||||||
|
// Add limited test data
|
||||||
|
for (let i = 0; i < 10; i++) {
|
||||||
|
await db.add({
|
||||||
|
id: `item-${i}`,
|
||||||
|
text: `test document ${i}`
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
// Search with offset that allows only partial results
|
||||||
|
const results = await db.search('test document', 5, { offset: 7 })
|
||||||
|
expect(results.length).toBe(3) // Only 3 results available after offset 7
|
||||||
|
})
|
||||||
|
})
|
||||||
|
|
||||||
|
describe('Pagination with Filters', () => {
|
||||||
|
it.skip('should paginate with noun type filters', async () => {
|
||||||
|
// TODO: This test requires proper noun type support in the add method
|
||||||
|
// Currently skipped as noun types are not directly supported in the add method
|
||||||
|
// Add test data with different noun types
|
||||||
|
for (let i = 0; i < 15; i++) {
|
||||||
|
await db.add({
|
||||||
|
id: `doc-${i}`,
|
||||||
|
text: `document ${i}`,
|
||||||
|
type: 'document'
|
||||||
|
}, undefined, 'document')
|
||||||
|
}
|
||||||
|
|
||||||
|
for (let i = 0; i < 15; i++) {
|
||||||
|
await db.add({
|
||||||
|
id: `note-${i}`,
|
||||||
|
text: `note ${i}`,
|
||||||
|
type: 'note'
|
||||||
|
}, undefined, 'note')
|
||||||
|
}
|
||||||
|
|
||||||
|
// Get first page of documents
|
||||||
|
const firstPage = await db.search('document', 5, {
|
||||||
|
nounTypes: ['document']
|
||||||
|
})
|
||||||
|
expect(firstPage.length).toBe(5)
|
||||||
|
expect(firstPage.every(r => r.metadata.type === 'document')).toBe(true)
|
||||||
|
|
||||||
|
// Get second page of documents
|
||||||
|
const secondPage = await db.search('document', 5, {
|
||||||
|
nounTypes: ['document'],
|
||||||
|
offset: 5
|
||||||
|
})
|
||||||
|
expect(secondPage.length).toBe(5)
|
||||||
|
expect(secondPage.every(r => r.metadata.type === 'document')).toBe(true)
|
||||||
|
|
||||||
|
// Ensure pages are different
|
||||||
|
const firstIds = firstPage.map(r => r.metadata.id)
|
||||||
|
const secondIds = secondPage.map(r => r.metadata.id)
|
||||||
|
const overlap = firstIds.filter(id => secondIds.includes(id))
|
||||||
|
expect(overlap.length).toBe(0)
|
||||||
|
})
|
||||||
|
|
||||||
|
it('should paginate with service filters', async () => {
|
||||||
|
// Add test data with different services
|
||||||
|
for (let i = 0; i < 20; i++) {
|
||||||
|
const metadata = {
|
||||||
|
id: `item-${i}`,
|
||||||
|
text: `test item ${i}`,
|
||||||
|
createdBy: {
|
||||||
|
augmentation: i < 10 ? 'service-a' : 'service-b'
|
||||||
|
}
|
||||||
|
}
|
||||||
|
await db.add(metadata)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Get paginated results for service-a
|
||||||
|
const page1 = await db.search('test item', 3, {
|
||||||
|
service: 'service-a'
|
||||||
|
})
|
||||||
|
const page2 = await db.search('test item', 3, {
|
||||||
|
service: 'service-a',
|
||||||
|
offset: 3
|
||||||
|
})
|
||||||
|
|
||||||
|
// Check that results are from service-a
|
||||||
|
expect(page1.every(r => r.metadata.createdBy?.augmentation === 'service-a')).toBe(true)
|
||||||
|
expect(page2.every(r => r.metadata.createdBy?.augmentation === 'service-a')).toBe(true)
|
||||||
|
|
||||||
|
// Check no overlap
|
||||||
|
const page1Ids = page1.map(r => r.metadata.id)
|
||||||
|
const page2Ids = page2.map(r => r.metadata.id)
|
||||||
|
expect(page1Ids.filter(id => page2Ids.includes(id)).length).toBe(0)
|
||||||
|
})
|
||||||
|
})
|
||||||
|
|
||||||
|
describe('Pagination Consistency', () => {
|
||||||
|
it('should maintain consistent ordering across pages', async () => {
|
||||||
|
// Add test data
|
||||||
|
for (let i = 0; i < 30; i++) {
|
||||||
|
await db.add({
|
||||||
|
id: `item-${i.toString().padStart(2, '0')}`,
|
||||||
|
text: `consistent test ${i}`,
|
||||||
|
score: Math.random()
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
// Get all results in one query
|
||||||
|
const allResults = await db.search('consistent test', 30)
|
||||||
|
const allIds = allResults.map(r => r.metadata.id)
|
||||||
|
|
||||||
|
// Get results in pages
|
||||||
|
const page1 = await db.search('consistent test', 10, { offset: 0 })
|
||||||
|
const page2 = await db.search('consistent test', 10, { offset: 10 })
|
||||||
|
const page3 = await db.search('consistent test', 10, { offset: 20 })
|
||||||
|
|
||||||
|
const pagedIds = [
|
||||||
|
...page1.map(r => r.metadata.id),
|
||||||
|
...page2.map(r => r.metadata.id),
|
||||||
|
...page3.map(r => r.metadata.id)
|
||||||
|
]
|
||||||
|
|
||||||
|
// Check that paginated results match the full query
|
||||||
|
expect(pagedIds).toEqual(allIds)
|
||||||
|
})
|
||||||
|
|
||||||
|
it('should handle empty results gracefully', async () => {
|
||||||
|
// Search empty database with offset
|
||||||
|
const results = await db.search('nonexistent', 10, { offset: 5 })
|
||||||
|
expect(results).toEqual([])
|
||||||
|
})
|
||||||
|
})
|
||||||
|
|
||||||
|
describe('Vector Search with Offset', () => {
|
||||||
|
it('should paginate vector searches', async () => {
|
||||||
|
// Add test vectors
|
||||||
|
for (let i = 0; i < 20; i++) {
|
||||||
|
const vector = new Array(512).fill(0).map(() => Math.random())
|
||||||
|
await db.add({
|
||||||
|
id: `vec-${i}`,
|
||||||
|
vector: vector,
|
||||||
|
index: i
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
// Create a query vector
|
||||||
|
const queryVector = new Array(512).fill(0).map(() => Math.random())
|
||||||
|
|
||||||
|
// Get first page
|
||||||
|
const page1 = await db.search(queryVector, 5, { forceEmbed: false })
|
||||||
|
expect(page1.length).toBe(5)
|
||||||
|
|
||||||
|
// Get second page
|
||||||
|
const page2 = await db.search(queryVector, 5, {
|
||||||
|
forceEmbed: false,
|
||||||
|
offset: 5
|
||||||
|
})
|
||||||
|
expect(page2.length).toBe(5)
|
||||||
|
|
||||||
|
// Ensure different results
|
||||||
|
const page1Ids = page1.map(r => r.metadata.id)
|
||||||
|
const page2Ids = page2.map(r => r.metadata.id)
|
||||||
|
expect(page1Ids.filter(id => page2Ids.includes(id)).length).toBe(0)
|
||||||
|
})
|
||||||
|
})
|
||||||
|
})
|
||||||
257
tests/performance-improvements.test.ts
Normal file
257
tests/performance-improvements.test.ts
Normal file
|
|
@ -0,0 +1,257 @@
|
||||||
|
/**
|
||||||
|
* Tests for performance improvements: caching and cursor-based pagination
|
||||||
|
*/
|
||||||
|
|
||||||
|
import { describe, it, expect, beforeEach, afterEach } from 'vitest'
|
||||||
|
import { BrainyData } from '../src/brainyData.js'
|
||||||
|
import { cleanupWorkerPools } from '../src/utils/index.js'
|
||||||
|
|
||||||
|
describe('Performance Improvements', () => {
|
||||||
|
let db: BrainyData
|
||||||
|
|
||||||
|
beforeEach(async () => {
|
||||||
|
// Initialize BrainyData with caching enabled
|
||||||
|
db = new BrainyData({
|
||||||
|
storage: {
|
||||||
|
forceMemoryStorage: true
|
||||||
|
},
|
||||||
|
searchCache: {
|
||||||
|
enabled: true,
|
||||||
|
maxSize: 50,
|
||||||
|
maxAge: 60000 // 1 minute
|
||||||
|
},
|
||||||
|
logging: {
|
||||||
|
verbose: false
|
||||||
|
}
|
||||||
|
})
|
||||||
|
await db.init()
|
||||||
|
})
|
||||||
|
|
||||||
|
afterEach(async () => {
|
||||||
|
await db.clear()
|
||||||
|
await cleanupWorkerPools()
|
||||||
|
})
|
||||||
|
|
||||||
|
describe('Search Result Caching', () => {
|
||||||
|
it('should cache search results transparently', async () => {
|
||||||
|
// Add test data
|
||||||
|
for (let i = 0; i < 10; i++) {
|
||||||
|
await db.add({
|
||||||
|
id: `item-${i}`,
|
||||||
|
text: `test document ${i}`,
|
||||||
|
value: i
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
// Clear cache to start fresh
|
||||||
|
db.clearCache()
|
||||||
|
let stats = db.getCacheStats()
|
||||||
|
expect(stats.search.hits).toBe(0)
|
||||||
|
expect(stats.search.misses).toBe(0)
|
||||||
|
|
||||||
|
// First search - should be cache miss
|
||||||
|
const query = 'test document'
|
||||||
|
const results1 = await db.search(query, 5)
|
||||||
|
expect(results1.length).toBe(5)
|
||||||
|
|
||||||
|
stats = db.getCacheStats()
|
||||||
|
expect(stats.search.misses).toBe(1) // First search is a miss
|
||||||
|
expect(stats.search.hits).toBe(0)
|
||||||
|
|
||||||
|
// Second identical search - should be cache hit
|
||||||
|
const results2 = await db.search(query, 5)
|
||||||
|
expect(results2.length).toBe(5)
|
||||||
|
expect(results2).toEqual(results1) // Same results
|
||||||
|
|
||||||
|
stats = db.getCacheStats()
|
||||||
|
expect(stats.search.misses).toBe(1) // Still only one miss
|
||||||
|
expect(stats.search.hits).toBe(1) // Now we have a hit
|
||||||
|
expect(stats.search.hitRate).toBe(0.5) // 50% hit rate
|
||||||
|
})
|
||||||
|
|
||||||
|
it('should invalidate cache when data changes', async () => {
|
||||||
|
// Add test data
|
||||||
|
for (let i = 0; i < 5; i++) {
|
||||||
|
await db.add({
|
||||||
|
id: `item-${i}`,
|
||||||
|
text: `test document ${i}`
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
// Search and cache results
|
||||||
|
const results1 = await db.search('test document', 3)
|
||||||
|
expect(results1.length).toBe(3)
|
||||||
|
|
||||||
|
let stats = db.getCacheStats()
|
||||||
|
const initialMisses = stats.search.misses
|
||||||
|
|
||||||
|
// Same search should hit cache
|
||||||
|
await db.search('test document', 3)
|
||||||
|
stats = db.getCacheStats()
|
||||||
|
expect(stats.search.hits).toBeGreaterThan(0)
|
||||||
|
|
||||||
|
// Add new data - should invalidate cache
|
||||||
|
await db.add({
|
||||||
|
id: 'new-item',
|
||||||
|
text: 'new test document'
|
||||||
|
})
|
||||||
|
|
||||||
|
// Same search should miss cache (due to invalidation)
|
||||||
|
await db.search('test document', 3)
|
||||||
|
stats = db.getCacheStats()
|
||||||
|
// Cache was cleared, so we should have fewer misses recorded than expected
|
||||||
|
// The key point is that cache size should be 0 or low, indicating invalidation worked
|
||||||
|
expect(stats.search.size).toBeLessThanOrEqual(1) // Cache should have been cleared
|
||||||
|
})
|
||||||
|
|
||||||
|
it('should handle cache with different search parameters', async () => {
|
||||||
|
// Add test data
|
||||||
|
for (let i = 0; i < 10; i++) {
|
||||||
|
await db.add({
|
||||||
|
id: `item-${i}`,
|
||||||
|
text: `test document ${i}`
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
db.clearCache()
|
||||||
|
|
||||||
|
// Different k values should create different cache entries
|
||||||
|
await db.search('test document', 3)
|
||||||
|
await db.search('test document', 5) // Different k
|
||||||
|
await db.search('test document', 3) // Should hit cache
|
||||||
|
|
||||||
|
const stats = db.getCacheStats()
|
||||||
|
expect(stats.search.hits).toBe(1)
|
||||||
|
expect(stats.search.misses).toBe(2)
|
||||||
|
})
|
||||||
|
})
|
||||||
|
|
||||||
|
describe('Cursor-based Pagination', () => {
|
||||||
|
beforeEach(async () => {
|
||||||
|
// Add more test data for pagination
|
||||||
|
for (let i = 0; i < 50; i++) {
|
||||||
|
await db.add({
|
||||||
|
id: `doc-${i.toString().padStart(2, '0')}`,
|
||||||
|
text: `document content for testing pagination ${i}`,
|
||||||
|
index: i
|
||||||
|
})
|
||||||
|
}
|
||||||
|
})
|
||||||
|
|
||||||
|
it('should return cursor for pagination', async () => {
|
||||||
|
const page1 = await db.searchWithCursor('document content for testing', 10)
|
||||||
|
|
||||||
|
expect(page1.results.length).toBe(10)
|
||||||
|
expect(page1.hasMore).toBe(true)
|
||||||
|
expect(page1.cursor).toBeDefined()
|
||||||
|
expect(page1.cursor!.lastId).toBeDefined()
|
||||||
|
expect(page1.cursor!.lastScore).toBeDefined()
|
||||||
|
})
|
||||||
|
|
||||||
|
it('should paginate consistently with cursor', async () => {
|
||||||
|
const page1 = await db.searchWithCursor('document content for testing', 5)
|
||||||
|
expect(page1.results.length).toBe(5)
|
||||||
|
expect(page1.hasMore).toBe(true)
|
||||||
|
|
||||||
|
// Get next page using cursor
|
||||||
|
const page2 = await db.searchWithCursor('document content for testing', 5, {
|
||||||
|
cursor: page1.cursor
|
||||||
|
})
|
||||||
|
expect(page2.results.length).toBe(5)
|
||||||
|
|
||||||
|
// Ensure no overlap between pages
|
||||||
|
const page1Ids = page1.results.map(r => r.id)
|
||||||
|
const page2Ids = page2.results.map(r => r.id)
|
||||||
|
const overlap = page1Ids.filter(id => page2Ids.includes(id))
|
||||||
|
expect(overlap.length).toBe(0)
|
||||||
|
})
|
||||||
|
|
||||||
|
it('should handle last page correctly', async () => {
|
||||||
|
// Get small pages to reach the end
|
||||||
|
let currentCursor: any = undefined
|
||||||
|
let allResults: any[] = []
|
||||||
|
let pageCount = 0
|
||||||
|
const maxPages = 10 // Safety limit
|
||||||
|
|
||||||
|
while (pageCount < maxPages) {
|
||||||
|
const page = await db.searchWithCursor('document content', 5, {
|
||||||
|
cursor: currentCursor
|
||||||
|
})
|
||||||
|
|
||||||
|
allResults.push(...page.results)
|
||||||
|
pageCount++
|
||||||
|
|
||||||
|
if (!page.hasMore) {
|
||||||
|
expect(page.cursor).toBeUndefined()
|
||||||
|
break
|
||||||
|
}
|
||||||
|
|
||||||
|
currentCursor = page.cursor
|
||||||
|
}
|
||||||
|
|
||||||
|
expect(pageCount).toBeLessThan(maxPages) // Should have finished before limit
|
||||||
|
expect(allResults.length).toBeGreaterThan(0)
|
||||||
|
})
|
||||||
|
|
||||||
|
it('should provide total estimate when possible', async () => {
|
||||||
|
const page = await db.searchWithCursor('document content', 50) // Request more than we have
|
||||||
|
|
||||||
|
// Should get all results in one page
|
||||||
|
expect(page.results.length).toBeGreaterThan(0)
|
||||||
|
expect(page.hasMore).toBe(false)
|
||||||
|
expect(page.totalEstimate).toBeDefined()
|
||||||
|
expect(page.totalEstimate).toBe(page.results.length)
|
||||||
|
})
|
||||||
|
})
|
||||||
|
|
||||||
|
describe('Performance Characteristics', () => {
|
||||||
|
it('should show performance improvement with caching', async () => {
|
||||||
|
// Add substantial test data
|
||||||
|
for (let i = 0; i < 50; i++) {
|
||||||
|
await db.add({
|
||||||
|
id: `perf-test-${i}`,
|
||||||
|
text: `performance test document ${i} with some content`
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
// Clear cache and perform first search
|
||||||
|
db.clearCache()
|
||||||
|
const results1 = await db.search('performance test document', 10)
|
||||||
|
|
||||||
|
let stats = db.getCacheStats()
|
||||||
|
expect(stats.search.misses).toBe(1) // First search is a miss
|
||||||
|
expect(stats.search.hits).toBe(0)
|
||||||
|
|
||||||
|
// Perform second identical search (should hit cache)
|
||||||
|
const results2 = await db.search('performance test document', 10)
|
||||||
|
|
||||||
|
expect(results1).toEqual(results2) // Same results
|
||||||
|
|
||||||
|
stats = db.getCacheStats()
|
||||||
|
expect(stats.search.hits).toBe(1) // Second search is a hit
|
||||||
|
expect(stats.search.hitRate).toBe(0.5) // 50% hit rate (1 hit, 1 miss)
|
||||||
|
})
|
||||||
|
|
||||||
|
it('should provide cache memory usage information', async () => {
|
||||||
|
// Add some data and search to populate cache
|
||||||
|
for (let i = 0; i < 10; i++) {
|
||||||
|
await db.add({
|
||||||
|
id: `mem-test-${i}`,
|
||||||
|
text: `memory test ${i}`
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
db.clearCache()
|
||||||
|
|
||||||
|
// Perform several searches to populate cache
|
||||||
|
await db.search('memory test', 5)
|
||||||
|
await db.search('memory test', 3)
|
||||||
|
await db.search('test', 5)
|
||||||
|
|
||||||
|
const stats = db.getCacheStats()
|
||||||
|
expect(stats.searchMemoryUsage).toBeGreaterThan(0)
|
||||||
|
expect(stats.search.size).toBeGreaterThan(0)
|
||||||
|
expect(stats.search.size).toBeLessThanOrEqual(stats.search.maxSize)
|
||||||
|
})
|
||||||
|
})
|
||||||
|
})
|
||||||
Loading…
Add table
Add a link
Reference in a new issue