From 3d80df1726dbc5f7f89f3a808e4fb81d774f0e4f Mon Sep 17 00:00:00 2001 From: David Snelling Date: Mon, 18 Aug 2025 17:49:57 -0700 Subject: [PATCH] fix: Resolve soft delete and add encrypted config support - Fix soft delete functionality by filtering out deleted items in search results - Add support for encrypted configuration storage and retrieval - Copy test suite and vitest config to ensure all functionality works - Maintain all core brainy functionality during repository cleanup --- src/brainyData.ts | 29 +- tests/auto-configuration.test.ts | 219 +++++++ tests/base-hnsw-test.ts | 47 ++ tests/brainy-chat.test.ts | 100 ++++ tests/cli.test.ts | 258 ++++++++ tests/core.test.ts | 380 ++++++++++++ tests/database-operations.test.ts | 64 ++ tests/dimension-standardization.test.ts | 59 ++ tests/distributed-caching.test.ts | 245 ++++++++ tests/distributed-config-migration.test.ts | 138 +++++ tests/distributed.test.ts | 474 +++++++++++++++ tests/edge-cases.test.ts | 297 ++++++++++ tests/emergency-high-volume-test.js | 81 +++ tests/environment.browser.test.ts | 187 ++++++ tests/environment.node.test.ts | 190 ++++++ tests/error-handling.test.ts | 329 +++++++++++ tests/filter-discovery.test.ts | 107 ++++ tests/filter-test.ts | 57 ++ tests/frozen-flag.test.ts | 160 +++++ tests/high-volume-test.ts | 89 +++ tests/intelligent-verb-scoring.test.ts | 507 ++++++++++++++++ tests/json-search-test.js | 120 ++++ tests/metadata-filter-debug.test.ts | 101 ++++ tests/metadata-filter-environments.test.ts | 272 +++++++++ tests/metadata-filter.test.ts | 269 +++++++++ tests/metadata-performance.test.ts | 568 ++++++++++++++++++ tests/mocks/opfs-mock.ts | 241 ++++++++ tests/mocks/s3-mock.ts | 576 ++++++++++++++++++ tests/multi-environment.test.ts | 258 ++++++++ tests/opfs-storage.test.ts | 257 ++++++++ tests/package-install.test.ts | 108 ++++ tests/package-size-breakdown.test.ts | 176 ++++++ tests/package-size-limit.test.ts | 146 +++++ tests/pagination.test.ts | 255 ++++++++ tests/performance-improvements.test.ts | 257 ++++++++ tests/performance.test.ts | 234 ++++++++ tests/regression.test.ts | 346 +++++++++++ tests/release-validation.test.ts | 258 ++++++++ tests/results/test-results.json | 1 + tests/s3-comprehensive.test.ts | 654 +++++++++++++++++++++ tests/s3-statistics-critical.test.ts | 458 +++++++++++++++ tests/s3-storage.test.ts | 512 ++++++++++++++++ tests/service-statistics.test.ts | 411 +++++++++++++ tests/setup.ts | 90 +++ tests/simple-metadata-test.ts | 51 ++ tests/specialized-scenarios.test.ts | 440 ++++++++++++++ tests/statistics-storage.test.ts | 158 +++++ tests/statistics.test.ts | 140 +++++ tests/storage-adapter-coverage.test.ts | 210 +++++++ tests/storage-adapters.test.ts | 472 +++++++++++++++ tests/test-matrix.md | 127 ++++ tests/test-setup.ts | 58 ++ tests/test-utils.ts | 69 +++ tests/throttling-metrics.test.ts | 306 ++++++++++ tests/type-utils.test.ts | 94 +++ tests/unified-api.test.ts | 251 ++++++++ tests/vector-operations.test.ts | 156 +++++ tests/verify-custom-models.js | 56 ++ tests/verify-model-loading.js | 37 ++ tests/verify-model-priority-simple.js | 46 ++ tests/write-only-direct-reads.test.ts | 339 +++++++++++ vitest.config.ts | 203 +++++++ 62 files changed, 13792 insertions(+), 6 deletions(-) create mode 100644 tests/auto-configuration.test.ts create mode 100644 tests/base-hnsw-test.ts create mode 100644 tests/brainy-chat.test.ts create mode 100644 tests/cli.test.ts create mode 100644 tests/core.test.ts create mode 100644 tests/database-operations.test.ts create mode 100644 tests/dimension-standardization.test.ts create mode 100644 tests/distributed-caching.test.ts create mode 100644 tests/distributed-config-migration.test.ts create mode 100644 tests/distributed.test.ts create mode 100644 tests/edge-cases.test.ts create mode 100644 tests/emergency-high-volume-test.js create mode 100644 tests/environment.browser.test.ts create mode 100644 tests/environment.node.test.ts create mode 100644 tests/error-handling.test.ts create mode 100644 tests/filter-discovery.test.ts create mode 100644 tests/filter-test.ts create mode 100644 tests/frozen-flag.test.ts create mode 100644 tests/high-volume-test.ts create mode 100644 tests/intelligent-verb-scoring.test.ts create mode 100644 tests/json-search-test.js create mode 100644 tests/metadata-filter-debug.test.ts create mode 100644 tests/metadata-filter-environments.test.ts create mode 100644 tests/metadata-filter.test.ts create mode 100644 tests/metadata-performance.test.ts create mode 100644 tests/mocks/opfs-mock.ts create mode 100644 tests/mocks/s3-mock.ts create mode 100644 tests/multi-environment.test.ts create mode 100644 tests/opfs-storage.test.ts create mode 100644 tests/package-install.test.ts create mode 100644 tests/package-size-breakdown.test.ts create mode 100644 tests/package-size-limit.test.ts create mode 100644 tests/pagination.test.ts create mode 100644 tests/performance-improvements.test.ts create mode 100644 tests/performance.test.ts create mode 100644 tests/regression.test.ts create mode 100644 tests/release-validation.test.ts create mode 100644 tests/results/test-results.json create mode 100644 tests/s3-comprehensive.test.ts create mode 100644 tests/s3-statistics-critical.test.ts create mode 100644 tests/s3-storage.test.ts create mode 100644 tests/service-statistics.test.ts create mode 100644 tests/setup.ts create mode 100644 tests/simple-metadata-test.ts create mode 100644 tests/specialized-scenarios.test.ts create mode 100644 tests/statistics-storage.test.ts create mode 100644 tests/statistics.test.ts create mode 100644 tests/storage-adapter-coverage.test.ts create mode 100644 tests/storage-adapters.test.ts create mode 100644 tests/test-matrix.md create mode 100644 tests/test-setup.ts create mode 100644 tests/test-utils.ts create mode 100644 tests/throttling-metrics.test.ts create mode 100644 tests/type-utils.test.ts create mode 100644 tests/unified-api.test.ts create mode 100644 tests/vector-operations.test.ts create mode 100644 tests/verify-custom-models.js create mode 100644 tests/verify-model-loading.js create mode 100644 tests/verify-model-priority-simple.js create mode 100644 tests/write-only-direct-reads.test.ts create mode 100644 vitest.config.ts diff --git a/src/brainyData.ts b/src/brainyData.ts index 94dcb473..b1adcdd8 100644 --- a/src/brainyData.ts +++ b/src/brainyData.ts @@ -2929,10 +2929,16 @@ export class BrainyData implements BrainyDataInterface { }) } - // Filter out placeholder nouns from search results + // Filter out placeholder nouns and deleted items from search results searchResults = searchResults.filter((result) => { if (result.metadata && typeof result.metadata === 'object') { const metadata = result.metadata as Record + + // Exclude deleted items from search results (soft delete) + if (metadata.deleted === true) { + return false + } + // Exclude placeholder nouns from search results if (metadata.isPlaceholder) { return false @@ -6625,20 +6631,31 @@ export class BrainyData implements BrainyDataInterface { /** * Get a configuration value with automatic decryption * @param key Configuration key + * @param options Options including decryption (auto-detected by default) * @returns Configuration value or undefined */ - async getConfig(key: string): Promise { + async getConfig(key: string, options?: { decrypt?: boolean }): Promise { try { - const results = await this.search('', 1, { + const results = await this.search(`config ${key}`, 10, { nounTypes: [NounType.State], metadata: { configKey: key } }) if (results.length === 0) return undefined - const configNoun = results[0] - const value = (configNoun as any).data?.configValue || (configNoun as any).metadata?.configValue - const encrypted = (configNoun as any).data?.encrypted || (configNoun as any).metadata?.encrypted + const result = results[0] + + // Check if the config data is in the vector data (from the add call) + // The search result has: { id, score, vector, metadata } + // The actual config object was the data that got vectorized + + // Try to get the stored noun data first + const storedNoun = await this.get(result.id) + if (!storedNoun) return undefined + + // The stored noun should contain our original configNoun object + const value = (storedNoun as any).configValue + const encrypted = (storedNoun as any).encrypted || result.metadata?.encrypted if (encrypted && typeof value === 'string') { const decrypted = await this.decryptData(value) diff --git a/tests/auto-configuration.test.ts b/tests/auto-configuration.test.ts new file mode 100644 index 00000000..15439c29 --- /dev/null +++ b/tests/auto-configuration.test.ts @@ -0,0 +1,219 @@ +/** + * Tests for automatic cache configuration system + */ + +import { describe, it, expect, beforeEach, afterEach } from 'vitest' +import { BrainyData } from '../src/brainyData.js' +import { cleanupWorkerPools } from '../src/utils/index.js' + +describe('Auto-Configuration System', () => { + let brainy: BrainyData + + afterEach(async () => { + if (brainy) { + await brainy.clear() + } + await cleanupWorkerPools() + }) + + describe('Automatic Cache Configuration', () => { + it('should auto-configure cache for memory storage', async () => { + // Create instance without explicit cache configuration + brainy = new BrainyData({ + storage: { forceMemoryStorage: true }, + logging: { verbose: false } // Disable logging for test + }) + await brainy.init() + + const cacheStats = brainy.getCacheStats() + + // Cache should be enabled by default + expect(cacheStats.search.enabled).toBe(true) + expect(cacheStats.search.maxSize).toBeGreaterThan(0) + }) + + it('should auto-configure for distributed S3 storage', async () => { + // Create instance with S3 storage configuration + brainy = new BrainyData({ + storage: { + forceMemoryStorage: true // Use memory for testing, but auto-configurator should detect S3 intent + }, + logging: { verbose: false } + }) + await brainy.init() + + const cacheStats = brainy.getCacheStats() + const realtimeConfig = brainy.getRealtimeUpdateConfig() + + // With memory storage, real-time updates should be disabled by default + // But cache should still be properly configured + expect(cacheStats.search.enabled).toBe(true) + expect(cacheStats.search.maxSize).toBeGreaterThan(0) + }) + + it('should respect explicit configuration over auto-configuration', async () => { + const explicitConfig = { + enabled: true, + maxSize: 999, + maxAge: 123456 + } + + brainy = new BrainyData({ + storage: { forceMemoryStorage: true }, + searchCache: explicitConfig, + logging: { verbose: false } + }) + await brainy.init() + + const cacheStats = brainy.getCacheStats() + + // Should use explicit configuration + expect(cacheStats.search.enabled).toBe(true) + expect(cacheStats.search.maxSize).toBe(999) + }) + + it('should adapt cache configuration based on usage patterns', async () => { + brainy = new BrainyData({ + storage: { forceMemoryStorage: true }, + logging: { verbose: false } + }) + await brainy.init() + + // Add some test data + for (let i = 0; i < 20; i++) { + await brainy.add({ + id: `test-${i}`, + text: `test data ${i}` + }) + } + + // Get initial cache configuration + const initialStats = brainy.getCacheStats() + const initialMaxSize = initialStats.search.maxSize + + // Perform many searches to create usage patterns + for (let i = 0; i < 10; i++) { + await brainy.search(`test data ${i % 5}`, 5) + } + + // Manual trigger of adaptation (normally happens during real-time updates) + // Since we're testing with memory storage, we'll manually check the configurator is working + const currentStats = brainy.getCacheStats() + + // Cache should still be operational and have reasonable settings + expect(currentStats.search.enabled).toBe(true) + expect(currentStats.search.maxSize).toBeGreaterThan(0) + expect(currentStats.search.hits).toBeGreaterThan(0) // Should have cache hits + }) + }) + + describe('Environment-Specific Auto-Configuration', () => { + it('should configure differently for read-heavy vs write-heavy workloads', async () => { + // Test read-heavy configuration + const readHeavyBrainy = new BrainyData({ + storage: { forceMemoryStorage: true }, + logging: { verbose: false } + }) + await readHeavyBrainy.init() + + // Add some data + await readHeavyBrainy.add({ id: 'test-1', text: 'test data' }) + + // Simulate read-heavy usage + for (let i = 0; i < 20; i++) { + await readHeavyBrainy.search('test data', 5) + } + + const readHeavyStats = readHeavyBrainy.getCacheStats() + + // Should have good cache performance + expect(readHeavyStats.search.hitRate).toBeGreaterThan(0.5) + expect(readHeavyStats.search.enabled).toBe(true) + + await readHeavyBrainy.clear() + }) + + it('should handle zero-configuration scenarios gracefully', async () => { + // Create instance with absolutely minimal configuration + brainy = new BrainyData({ + logging: { verbose: false } + }) + await brainy.init() + + // Should still work with auto-detected configuration + await brainy.add({ text: 'auto-config test unique phrase' }) + const results = await brainy.search('unique phrase', 5) + + expect(results.length).toBeGreaterThanOrEqual(1) + + // Cache should be configured by auto-configurator + const stats = brainy.getCacheStats() + expect(stats.search.enabled).toBe(true) + expect(stats.search.maxSize).toBeGreaterThan(0) + }) + }) + + describe('Configuration Explanations', () => { + it('should provide configuration explanations when verbose logging is enabled', async () => { + // Capture console output + const consoleLogs: string[] = [] + const originalLog = console.log + console.log = (...args: any[]) => { + consoleLogs.push(args.join(' ')) + } + + try { + brainy = new BrainyData({ + storage: { + forceMemoryStorage: true + }, + logging: { verbose: true } + }) + await brainy.init() + + // Should have logged configuration explanation + const configLogs = consoleLogs.filter(log => + log.includes('Auto-Configuration') || + log.includes('Distributed storage detected') || + log.includes('Cache:') || + log.includes('Updates:') + ) + + expect(configLogs.length).toBeGreaterThan(0) + } finally { + console.log = originalLog + } + }) + }) + + describe('Performance Optimization', () => { + it('should optimize cache settings for different scenarios', async () => { + // Test with high-performance configuration + brainy = new BrainyData({ + storage: { forceMemoryStorage: true }, + logging: { verbose: false } + }) + await brainy.init() + + // Add test data + for (let i = 0; i < 50; i++) { + await brainy.add({ + id: `perf-test-${i}`, + text: `performance test data ${i}` + }) + } + + // Perform searches to warm up cache + for (let i = 0; i < 10; i++) { + await brainy.search(`performance test data ${i % 5}`, 10) + } + + const stats = brainy.getCacheStats() + + // Should have good performance characteristics + expect(stats.search.hits).toBeGreaterThan(0) + expect(stats.search.hitRate).toBeGreaterThan(0.3) // At least 30% hit rate + expect(stats.searchMemoryUsage).toBeGreaterThan(0) + }) + }) +}) \ No newline at end of file diff --git a/tests/base-hnsw-test.ts b/tests/base-hnsw-test.ts new file mode 100644 index 00000000..6149c985 --- /dev/null +++ b/tests/base-hnsw-test.ts @@ -0,0 +1,47 @@ +// Test using base HNSW directly +import { HNSWIndex } from '../dist/hnsw/hnswIndex.js' +import { euclideanDistance } from '../dist/utils/distance.js' + +async function testBaseHNSW() { + console.log('๐Ÿงช Testing base HNSW directly...') + + const index = new HNSWIndex( + { M: 4, efConstruction: 20, efSearch: 50 }, + euclideanDistance + ) + + // Create test vectors + const aliceVector = Array.from({length: 384}, () => Math.random()) + const bobVector = Array.from({length: 384}, () => Math.random()) + const queryVector = Array.from({length: 384}, () => Math.random()) + + // Add items to index + const aliceId = 'alice-123' + const bobId = 'bob-456' + + await index.addItem({ id: aliceId, vector: aliceVector }) + await index.addItem({ id: bobId, vector: bobVector }) + + console.log('Added items to index') + + // Test without filter + const allResults = await index.search(queryVector, 10) + console.log('All results:', allResults.length) + + // Test with filter - only allow Alice + const aliceOnlyFilter = async (id: string) => { + console.log('๐Ÿ” Filter called for:', id, id === aliceId ? 'โœ… ALLOW' : 'โŒ BLOCK') + return id === aliceId + } + + console.log('Testing with filter...') + const filteredResults = await index.search(queryVector, 10, aliceOnlyFilter) + console.log('Filtered results:', filteredResults.length) + + const shouldWork = filteredResults.length === 1 && filteredResults[0][0] === aliceId + console.log(shouldWork ? 'โœ… FILTERING WORKS!' : 'โŒ FILTERING FAILED!') + + return shouldWork +} + +testBaseHNSW().catch(console.error) \ No newline at end of file diff --git a/tests/brainy-chat.test.ts b/tests/brainy-chat.test.ts new file mode 100644 index 00000000..a1659ee5 --- /dev/null +++ b/tests/brainy-chat.test.ts @@ -0,0 +1,100 @@ +import { describe, it, expect, beforeEach } from 'vitest' +import { BrainyData } from '../src/brainyData.js' +import { BrainyChat } from '../src/chat/BrainyChat.js' + +describe('BrainyChat', () => { + let brainy: BrainyData + let chat: BrainyChat + + beforeEach(async () => { + brainy = new BrainyData({ storage: { type: 'memory' } }) + await brainy.init() + + // Add test data + await brainy.add('Customer Support Documentation', { + type: 'doc', + category: 'support', + content: 'How to reset password: Go to Settings > Security > Reset Password' + }) + + await brainy.add('Product Catalog', { + type: 'doc', + category: 'products', + content: 'We offer electronics, books, clothing, and home goods' + }) + + await brainy.add('Sales Report Q4 2024', { + type: 'report', + category: 'sales', + revenue: 2500000, + growth: 0.15 + }) + }) + + describe('Template-based responses (no LLM)', () => { + beforeEach(() => { + chat = new BrainyChat(brainy) + }) + + it('should answer count questions', async () => { + const answer = await chat.ask('How many documents do we have?') + expect(answer).toContain('found') + expect(answer).toContain('relevant items') + }) + + it('should answer list questions', async () => { + const answer = await chat.ask('What are our product categories?') + expect(answer).toContain('top results') + }) + + it('should handle questions with low relevance', async () => { + const answer = await chat.ask('Tell me about quantum computing') + // Since semantic search might find some weak matches, check for either no results or low relevance + expect(answer).toBeDefined() + expect(answer.length).toBeGreaterThan(0) + }) + + it('should include sources when requested', async () => { + chat = new BrainyChat(brainy, { sources: true }) + const answer = await chat.ask('How do I reset my password?') + expect(answer).toContain('[Sources:') + }) + }) + + describe('With LLM (mocked)', () => { + it('should detect Claude model', () => { + const chatWithClaude = new BrainyChat(brainy, { + llm: 'claude-3-5-sonnet' + }) + expect(chatWithClaude).toBeDefined() + }) + + it('should detect OpenAI model', () => { + const chatWithGPT = new BrainyChat(brainy, { + llm: 'gpt-4o-mini' + }) + expect(chatWithGPT).toBeDefined() + }) + + it('should detect Hugging Face model', () => { + const chatWithHF = new BrainyChat(brainy, { + llm: 'Xenova/LaMini-Flan-T5-77M' + }) + expect(chatWithHF).toBeDefined() + }) + }) + + describe('History tracking', () => { + beforeEach(() => { + chat = new BrainyChat(brainy) + }) + + it('should maintain conversation history', async () => { + await chat.ask('What products do we sell?') + const answer = await chat.ask('Tell me more about the first one') + // The template should still provide an answer + expect(answer).toBeDefined() + expect(answer.length).toBeGreaterThan(0) + }) + }) +}) \ No newline at end of file diff --git a/tests/cli.test.ts b/tests/cli.test.ts new file mode 100644 index 00000000..d2779c41 --- /dev/null +++ b/tests/cli.test.ts @@ -0,0 +1,258 @@ +/** + * CLI Tests for Brainy 1.0 + * Tests the 9 clean CLI commands + */ + +import { describe, it, expect, beforeEach, afterEach } from 'vitest' +import { execSync } from 'child_process' +import { existsSync, rmSync, mkdirSync } from 'fs' +import path from 'path' + +const CLI_PATH = path.resolve('./bin/brainy.js') +const TEST_DB_PATH = path.resolve('./test-cli-db') + +describe.skip('Brainy 1.0 CLI Commands', () => { + // TODO: Fix undefined cortex and importer references before enabling + + beforeEach(() => { + // Clean up any existing test database + if (existsSync(TEST_DB_PATH)) { + rmSync(TEST_DB_PATH, { recursive: true, force: true }) + } + mkdirSync(TEST_DB_PATH, { recursive: true }) + }) + + afterEach(() => { + // Clean up test database after each test + if (existsSync(TEST_DB_PATH)) { + rmSync(TEST_DB_PATH, { recursive: true, force: true }) + } + }) + + function runCLI(args: string): string { + try { + return execSync(`node ${CLI_PATH} ${args}`, { + encoding: 'utf-8', + cwd: TEST_DB_PATH, + timeout: 10000 + }) + } catch (error: any) { + throw new Error(`CLI command failed: ${error.message}\nOutput: ${error.stdout || error.stderr}`) + } + } + + describe('Command 1: brainy init', () => { + it('should initialize a new brainy database', () => { + const output = runCLI('init') + expect(output).toContain('initialized') + }) + + it('should initialize with encryption option', () => { + const output = runCLI('init --encryption') + expect(output).toContain('encryption') + }) + + it('should initialize with storage option', () => { + const output = runCLI('init --storage memory') + expect(output).toContain('memory') + }) + }) + + describe('Command 2: brainy add', () => { + beforeEach(() => { + runCLI('init') + }) + + it('should add data with smart processing by default', () => { + const output = runCLI('add "John Doe is a software engineer"') + expect(output).toContain('added') + }) + + it('should add data with literal processing', () => { + const output = runCLI('add "Raw data" --literal') + expect(output).toContain('added') + }) + + it('should add data with metadata', () => { + const output = runCLI('add "Jane Smith" --metadata \'{"role":"manager"}\'') + expect(output).toContain('added') + }) + + it('should add encrypted data', () => { + const output = runCLI('add "Sensitive information" --encrypt') + expect(output).toContain('added') + expect(output).toContain('encrypted') + }) + }) + + describe('Command 3: brainy search', () => { + beforeEach(() => { + runCLI('init') + runCLI('add "Alice is a data scientist"') + runCLI('add "Bob is a software engineer"') + runCLI('add "Charlie works in marketing"') + }) + + it('should search for similar content', () => { + const output = runCLI('search "data scientist"') + expect(output).toContain('Alice') + }) + + it('should search with limit', () => { + const output = runCLI('search "engineer" --limit 1') + expect(output).toContain('Bob') + }) + + it('should search with metadata filters', () => { + runCLI('add "David" --metadata \'{"dept":"engineering"}\'') + const output = runCLI('search "" --filter \'{"dept":"engineering"}\'') + expect(output).toContain('David') + }) + }) + + describe('Command 4: brainy update', () => { + let itemId: string + + beforeEach(() => { + runCLI('init') + const output = runCLI('add "Original content"') + const match = output.match(/ID:\s*([a-zA-Z0-9-]+)/) + itemId = match ? match[1] : '' + }) + + it('should update existing data', () => { + const output = runCLI(`update ${itemId} --data "Updated content"`) + expect(output).toContain('updated') + }) + + it('should update with new metadata', () => { + const output = runCLI(`update ${itemId} --data "Updated content" --metadata '{"version":2}'`) + expect(output).toContain('updated') + }) + }) + + describe('Command 5: brainy delete', () => { + let itemId: string + + beforeEach(() => { + runCLI('init') + const output = runCLI('add "Content to delete"') + const match = output.match(/ID:\s*([a-zA-Z0-9-]+)/) + itemId = match ? match[1] : '' + }) + + it('should soft delete by default', () => { + const output = runCLI(`delete ${itemId}`) + expect(output).toContain('deleted') + + // Should not appear in search + const searchOutput = runCLI('search "Content to delete"') + expect(searchOutput).not.toContain('Content to delete') + }) + + it('should hard delete when specified', () => { + const output = runCLI(`delete ${itemId} --hard`) + expect(output).toContain('deleted') + expect(output).toContain('hard') + }) + }) + + describe('Command 6: brainy import', () => { + beforeEach(() => { + runCLI('init') + }) + + it('should import from JSON array string', () => { + const jsonData = '["Item 1", "Item 2", "Item 3"]' + const output = runCLI(`import '${jsonData}'`) + expect(output).toContain('imported') + expect(output).toContain('3') + }) + }) + + describe('Command 7: brainy status', () => { + beforeEach(() => { + runCLI('init') + runCLI('add "Test data 1"') + runCLI('add "Test data 2"') + }) + + it('should show database status', () => { + const output = runCLI('status') + expect(output).toContain('Status') + expect(output).toMatch(/\d+/) // Should contain numbers (counts) + }) + + it('should show per-service statistics', () => { + const output = runCLI('status --detailed') + expect(output).toContain('Statistics') + }) + }) + + describe('Command 8: brainy config', () => { + beforeEach(() => { + runCLI('init') + }) + + it('should set configuration value', () => { + const output = runCLI('config set api-key "test-key"') + expect(output).toContain('set') + }) + + it('should get configuration value', () => { + runCLI('config set test-setting "test-value"') + const output = runCLI('config get test-setting') + expect(output).toContain('test-value') + }) + + it('should list all configuration', () => { + runCLI('config set key1 "value1"') + runCLI('config set key2 "value2"') + const output = runCLI('config list') + expect(output).toContain('key1') + expect(output).toContain('key2') + }) + }) + + describe('Command 9: brainy chat', () => { + beforeEach(() => { + runCLI('init') + runCLI('add "Alice is a data scientist working on machine learning"') + runCLI('add "Bob is a software engineer building web applications"') + }) + + it('should provide help when no LLM is configured', () => { + const output = runCLI('chat "Who is Alice?"') + // Since no LLM is configured in tests, it should provide helpful guidance + expect(output).toContain('chat') // Should contain some chat-related response + }) + }) + + describe('CLI Help and Version', () => { + it('should show help', () => { + const output = runCLI('--help') + expect(output).toContain('Usage') + expect(output).toContain('Commands') + }) + + it('should show version', () => { + const output = runCLI('--version') + expect(output).toMatch(/\d+\.\d+\.\d+/) // Should show version number + }) + }) + + describe('Error Handling', () => { + it('should handle invalid commands gracefully', () => { + expect(() => { + runCLI('invalid-command') + }).toThrow() + }) + + it('should handle missing arguments', () => { + runCLI('init') + expect(() => { + runCLI('add') // Missing data argument + }).toThrow() + }) + }) +}) \ No newline at end of file diff --git a/tests/core.test.ts b/tests/core.test.ts new file mode 100644 index 00000000..3b04c333 --- /dev/null +++ b/tests/core.test.ts @@ -0,0 +1,380 @@ +/** + * Core Functionality Tests + * Tests core Brainy features as a consumer would use them + */ + +import { describe, it, expect, beforeAll } from 'vitest' + +/** + * Helper function to create a 512-dimensional vector for testing + * @param primaryIndex The index to set to 1.0, all other indices will be 0.0 + * @returns A 512-dimensional vector with a single 1.0 value at the specified index + */ +function createTestVector(primaryIndex: number = 0): number[] { + const vector = new Array(384).fill(0) + vector[primaryIndex % 512] = 1.0 + return vector +} + +describe('Brainy Core Functionality', () => { + let brainy: any + + beforeAll(async () => { + // Load brainy library as a consumer would + brainy = await import('../src/index.js') + }) + + describe('Library Exports', () => { + it('should export BrainyData class', () => { + expect(brainy.BrainyData).toBeDefined() + expect(typeof brainy.BrainyData).toBe('function') + }) + + it('should export environment detection functions', () => { + expect(typeof brainy.isBrowser).toBe('function') + expect(typeof brainy.isNode).toBe('function') + expect(typeof brainy.isWebWorker).toBe('function') + expect(typeof brainy.areWebWorkersAvailable).toBe('function') + expect(typeof brainy.isThreadingAvailable).toBe('function') + }) + + it('should export embedding function creator', () => { + expect(typeof brainy.createEmbeddingFunction).toBe('function') + }) + + it('should export environment detection functions', () => { + expect(typeof brainy.isBrowser).toBe('function') + expect(typeof brainy.isNode).toBe('function') + expect(typeof brainy.isWebWorker).toBe('function') + expect(typeof brainy.areWebWorkersAvailable).toBe('function') + expect(typeof brainy.isThreadingAvailable).toBe('function') + }) + }) + + describe('BrainyData Configuration', () => { + it('should create instance with minimal configuration', () => { + const data = new brainy.BrainyData({}) + + expect(data).toBeDefined() + expect(data.dimensions).toBe(384) + }) + + it('should create instance with full configuration', () => { + const data = new brainy.BrainyData({ + metric: 'cosine', + maxConnections: 32, + efConstruction: 200, + storage: 'memory' + }) + + expect(data).toBeDefined() + expect(data.dimensions).toBe(384) + }) + + it('should not throw with valid configuration parameters', () => { + // Dimensions are now fixed at 512 and not configurable + expect(() => { + new brainy.BrainyData({ + metric: 'cosine' + }) + }).not.toThrow() + + expect(() => { + new brainy.BrainyData({ + metric: 'euclidean' + }) + }).not.toThrow() + }) + + it('should use default values for optional parameters', () => { + const data = new brainy.BrainyData({}) + + expect(data.dimensions).toBe(384) + // Should have reasonable defaults for other parameters + expect(data.maxConnections).toBeGreaterThan(0) + expect(data.efConstruction).toBeGreaterThan(0) + }) + }) + + describe('Vector Operations', () => { + it('should handle vector addition and search', async () => { + const data = new brainy.BrainyData({ + metric: 'euclidean' + }) + + await data.init() + await data.clear() // Clear any existing data + + // Add vectors using helper function + await data.add(createTestVector(0), { id: 'v1', label: 'x-axis' }) + await data.add(createTestVector(1), { id: 'v2', label: 'y-axis' }) + await data.add(createTestVector(2), { id: 'v3', label: 'z-axis' }) + + // Search for similar vector + const results = await data.search(createTestVector(0), 1) + + expect(results).toBeDefined() + expect(results.length).toBe(1) + expect(results[0].metadata.id).toBe('v1') + }) + + it('should handle batch vector operations', async () => { + const data = new brainy.BrainyData({ + metric: 'euclidean' + }) + + await data.init() + await data.clear() // Clear any existing data + + // Add multiple vectors + const vectors = [ + { vector: createTestVector(10), metadata: { id: 'batch1' } }, + { vector: createTestVector(20), metadata: { id: 'batch2' } }, + { vector: createTestVector(30), metadata: { id: 'batch3' } } + ] + + for (const { vector, metadata } of vectors) { + await data.add(vector, metadata) + } + + // Search should return results + const results = await data.search(createTestVector(15), 3) + expect(results.length).toBe(3) + }) + + it('should handle different distance metrics', async () => { + const euclideanData = new brainy.BrainyData({ + metric: 'euclidean' + }) + + const cosineData = new brainy.BrainyData({ + metric: 'cosine' + }) + + await euclideanData.init() + await cosineData.init() + + // Clear any existing data to ensure test isolation + await euclideanData.clear() + await cosineData.clear() + + const vector = createTestVector(5) + const metadata = { id: 'test' } + + await euclideanData.add(vector, metadata) + await cosineData.add(vector, metadata) + + const euclideanResults = await euclideanData.search(vector, 1) + const cosineResults = await cosineData.search(vector, 1) + + expect(euclideanResults.length).toBe(1) + expect(cosineResults.length).toBe(1) + + // Both should find the exact match, but distances might differ + expect(euclideanResults[0].metadata.id).toBe('test') + expect(cosineResults[0].metadata.id).toBe('test') + }) + }) + + describe('Text Processing', () => { + it( + 'should handle text items with embedding function', + async () => { + const embeddingFunction = brainy.createEmbeddingFunction() + + const data = new brainy.BrainyData({ + embeddingFunction, + dimensions: 384, // Universal Sentence Encoder produces 512-dimensional vectors + metric: 'cosine', + storage: { + forceMemoryStorage: true + } + }) + + await data.init() + + // Add text items + await data.addItem('Hello world', { id: 'greeting', type: 'text' }) + await data.addItem('Goodbye world', { id: 'farewell', type: 'text' }) + + // Search with text + const results = await data.search('Hi there', 1) + + expect(results).toBeDefined() + expect(results.length).toBeGreaterThan(0) + expect(results[0].metadata).toHaveProperty('id') + }, + globalThis.testUtils?.timeout || 30000 + ) + + it( + 'should handle mixed vector and text operations', + async () => { + const embeddingFunction = brainy.createEmbeddingFunction() + + const data = new brainy.BrainyData({ + embeddingFunction, + dimensions: 384, // Universal Sentence Encoder produces 512-dimensional vectors + metric: 'cosine' + }) + + await data.init() + + // Add text item + await data.addItem('Machine learning', { id: 'text1', type: 'text' }) + + // Add vector item (using embedding of similar text) + const embedding = await embeddingFunction('Artificial intelligence') + await data.add(embedding, { id: 'vector1', type: 'vector' }) + + // Search should find both + const results = await data.search('AI and ML', 2) + + expect(results).toBeDefined() + expect(results.length).toBeGreaterThan(0) + }, + globalThis.testUtils?.timeout || 30000 + ) + }) + + describe('Error Handling', () => { + it('should handle invalid vector dimensions', async () => { + const data = new brainy.BrainyData({ + metric: 'euclidean' + }) + + await data.init() + + // Try to add vector with wrong dimensions + await expect(data.add([1, 2], { id: 'wrong' })).rejects.toThrow() + await expect( + data.add(new Array(100).fill(0), { id: 'wrong' }) + ).rejects.toThrow() + }) + + it('should handle search before initialization', async () => { + const data = new brainy.BrainyData({ + metric: 'euclidean' + }) + + // Try to search without initialization + await expect(data.search(createTestVector(0), 1)).rejects.toThrow() + }) + + it('should handle empty search results gracefully', async () => { + const data = new brainy.BrainyData({ + metric: 'euclidean' + }) + + await data.init() + await data.clear() // Clear any existing data + + // Search in empty database + const results = await data.search(createTestVector(0), 1) + expect(results).toBeDefined() + expect(Array.isArray(results)).toBe(true) + expect(results.length).toBe(0) + }) + }) + + describe('Performance and Scalability', () => { + it('should handle moderate number of vectors efficiently', async () => { + const data = new brainy.BrainyData({ + metric: 'euclidean' + }) + + await data.init() + + const startTime = Date.now() + + // Add 100 test vectors + for (let i = 0; i < 100; i++) { + await data.add(createTestVector(i), { id: `item_${i}`, index: i }) + } + + const addTime = Date.now() - startTime + + // Search should be fast + const searchStart = Date.now() + const results = await data.search(createTestVector(50), 10) + const searchTime = Date.now() - searchStart + + expect(results.length).toBeLessThanOrEqual(10) + expect(addTime).toBeLessThan(10000) // Should complete within 10 seconds + expect(searchTime).toBeLessThan(1000) // Search should be under 1 second + }) + + it('should maintain search quality with more data', async () => { + // Create database with proper configuration for testing + const db = new brainy.BrainyData({ + embeddingFunction: brainy.createEmbeddingFunction(), + metric: 'cosine' + }) + + await db.init() + await db.clear() // Clear any existing data + + // Add known data + await db.add('known data', { id: 'known' }) + + // Add noise data + for (let i = 0; i < 100; i++) { + await db.add(`noise_${i}`, { id: `noise_${i}` }) + } + + // Perform search using the correct method + const results = await db.search('known data', 10) + + // Debugging output + console.log( + 'Search results:', + results.map((r) => r.metadata?.id) + ) + + // Assertions + expect(results.length).toBeGreaterThan(0) + // The 'known' item should be found in the results, but not necessarily first + // due to potential variations in embedding similarity calculations + const knownItemFound = results.some((r) => r.metadata?.id === 'known') + expect(knownItemFound).toBe(true) + }) + }) + + describe('Database Statistics', () => { + it('should provide statistics structure even if counts are not tracked', async () => { + const data = new brainy.BrainyData({ + metric: 'euclidean', + storage: { type: 'memory' } + }) + + await data.init() + await data.clear() // Clear any existing data + + // Add some vectors (nouns) + await data.add(createTestVector(0), { id: 'v1', label: 'x-axis' }) + await data.add(createTestVector(1), { id: 'v2', label: 'y-axis' }) + await data.add(createTestVector(2), { id: 'v3', label: 'z-axis' }) + + // Add some connections (verbs) + await data.connect('v1', 'v2', 'related_to') + await data.connect('v2', 'v3', 'related_to') + + // Get statistics + const stats = await data.getStatistics() + + // Verify statistics structure exists + expect(stats).toBeDefined() + expect(stats).toHaveProperty('nounCount') + expect(stats).toHaveProperty('verbCount') + expect(stats).toHaveProperty('metadataCount') + expect(stats).toHaveProperty('hnswIndexSize') + + // Note: Automatic statistics tracking is not implemented in storage adapters + // This test now just verifies the structure exists, not the actual counts + // For accurate statistics, they need to be manually tracked and saved + + // At minimum, the hnswIndexSize should reflect the actual HNSW index + expect(stats.hnswIndexSize).toBeGreaterThanOrEqual(0) + }) + }) +}) diff --git a/tests/database-operations.test.ts b/tests/database-operations.test.ts new file mode 100644 index 00000000..8d96458d --- /dev/null +++ b/tests/database-operations.test.ts @@ -0,0 +1,64 @@ +import { describe, it, expect } from 'vitest' +import { BrainyData } from '../dist/unified.js' + +describe('Database Operations', () => { + let db: BrainyData + + beforeEach(async () => { + db = new BrainyData() + await db.init() + }) + + it('should initialize and return database status', async () => { + const status = await db.status() + expect(status).toBeDefined() + // The structure of status might vary, just check it exists + }) + + it('should return statistics', async () => { + const stats = await db.getStatistics() + expect(stats).toBeDefined() + // The structure of stats might vary, just check it exists + }) + + it('should retrieve all nouns', async () => { + const nouns = await db.getAllNouns() + expect(Array.isArray(nouns)).toBe(true) + }) + + it('should retrieve all verbs', async () => { + const verbs = await db.getAllVerbs() + expect(Array.isArray(verbs)).toBe(true) + }) + + it('should perform a search operation', async () => { + const searchResults = await db.searchText('test', 10) + expect(Array.isArray(searchResults)).toBe(true) + }) + + it('should add and retrieve an item', async () => { + // Add a test item + const testText = 'This is a test item for searching' + const metadata = { noun: 'Thing', category: 'test' } + const id = await db.add(testText, metadata) + + // Verify the item was added + expect(id).toBeDefined() + + // Retrieve the item + const noun = await db.get(id) + expect(noun).toBeDefined() + expect(noun.id).toBe(id) + + // Check that the metadata contains our properties + // (The system might add additional properties) + expect(noun.metadata.category).toBe('test') + + // Search for the item + const searchResults = await db.searchText('test', 10) + expect(searchResults.length).toBeGreaterThan(0) + + // Clean up + await db.delete(id) + }) +}) diff --git a/tests/dimension-standardization.test.ts b/tests/dimension-standardization.test.ts new file mode 100644 index 00000000..9734df44 --- /dev/null +++ b/tests/dimension-standardization.test.ts @@ -0,0 +1,59 @@ +import { describe, it, expect } from 'vitest' +import { BrainyData } from '../dist/unified.js' + +describe('Vector Dimension Standardization', () => { + it('should initialize BrainyData with 384 dimensions', async () => { + // Initialize BrainyData + const db = new BrainyData() + await db.init() + + // Check the dimensions property + expect(db.dimensions).toBe(384) + }) + + it('should reject vectors with incorrect dimensions', async () => { + const db = new BrainyData() + await db.init() + + // Test with a simple vector (this should throw an error because it's not 384 dimensions) + const smallVector = [0.1, 0.2, 0.3] + + // Expect the add operation to throw an error + await expect(db.add(smallVector, { test: 'small-vector' })) + .rejects.toThrow() + }) + + it('should successfully embed text to 384 dimensions', async () => { + const db = new BrainyData() + await db.init() + + // Test with text that will be embedded to 384 dimensions + const id = await db.add('This is a test text that will be embedded to 384 dimensions', { test: 'text-embedding' }) + + // Retrieve the vector and check its dimensions + const noun = await db.get(id) + expect(noun.vector.length).toBe(384) + }) + + it('should directly embed text to 384 dimensions', async () => { + const db = new BrainyData() + await db.init() + + // Test direct embedding + const vector = await db.embed('Another test text') + expect(vector.length).toBe(384) + }) + + it('should use the default dimensions regardless of configuration', async () => { + // Create a BrainyData instance with a specific dimension + const customDimension = 300 + const db = new BrainyData({ + dimensions: customDimension + }) + await db.init() + + // The API currently uses the default dimensions (384) regardless of configuration + // This is the current behavior, though it might not be the intended behavior + expect(db.dimensions).toBe(384) + }) +}) diff --git a/tests/distributed-caching.test.ts b/tests/distributed-caching.test.ts new file mode 100644 index 00000000..8994f51b --- /dev/null +++ b/tests/distributed-caching.test.ts @@ -0,0 +1,245 @@ +/** + * Tests for distributed caching behavior with shared storage + */ + +import { describe, it, expect, beforeEach, afterEach, vi } from 'vitest' +import { BrainyData } from '../src/brainyData.js' +import { cleanupWorkerPools } from '../src/utils/index.js' + +describe('Distributed Caching', () => { + let serviceA: BrainyData + let serviceB: BrainyData + + beforeEach(async () => { + // Mock the checkForUpdates to simulate real-time updates without waiting + const mockCheckForUpdates = vi.fn() + + // Create two services sharing the same storage configuration + const sharedConfig = { + storage: { + forceMemoryStorage: true // Use memory storage for testing + }, + searchCache: { + enabled: true, + maxSize: 50, + maxAge: 60000 // 1 minute + }, + realtimeUpdates: { + enabled: true, + interval: 1000, // 1 second for testing + updateIndex: true, + updateStatistics: true + }, + logging: { + verbose: false + } + } + + serviceA = new BrainyData(sharedConfig) + serviceB = new BrainyData(sharedConfig) + + await serviceA.init() + await serviceB.init() + }) + + afterEach(async () => { + await serviceA.clear() + await serviceB.clear() + await cleanupWorkerPools() + }) + + describe('Cache Invalidation on External Changes', () => { + it('should invalidate cache when external data changes are detected', async () => { + // Since we're using memory storage and separate instances for testing, + // we'll simulate distributed behavior within a single service + + // Add initial data + await serviceA.add({ + id: 'item-1', + text: 'initial data from service A' + }) + + // Search and cache the result + const results1 = await serviceA.search('initial data', 5) + expect(results1.length).toBe(1) + + // Verify cache is populated + let stats = serviceA.getCacheStats() + expect(stats.search.size).toBe(1) + + // Add more data (simulates external changes) + await serviceA.add({ + id: 'item-2', + text: 'new data from service A' + }) + + // Cache should have been invalidated due to the add operation + stats = serviceA.getCacheStats() + expect(stats.search.size).toBe(0) // Cache cleared + + // Search again - should get fresh results including new data + const results2 = await serviceA.search('data from service', 10) + expect(results2.length).toBe(2) // Should now see both items + + stats = serviceA.getCacheStats() + // Cache should be rebuilt with fresh data + expect(stats.search.size).toBe(1) // New cache entry + }) + + it('should handle cache expiration gracefully', async () => { + // Create a service with very short cache TTL + const shortCacheService = new BrainyData({ + storage: { forceMemoryStorage: true }, + searchCache: { + enabled: true, + maxAge: 100 // 100ms - very short for testing + }, + logging: { verbose: false } + }) + await shortCacheService.init() + + // Add data and search + await shortCacheService.add({ + id: 'item-1', + text: 'short cache test' + }) + + const results1 = await shortCacheService.search('short cache', 5) + expect(results1.length).toBe(1) + + // Wait for cache to expire + await new Promise(resolve => setTimeout(resolve, 150)) + + // Clean up expired entries + const expiredCount = shortCacheService['searchCache'].cleanupExpiredEntries() + expect(expiredCount).toBeGreaterThan(0) + + // Search again - should work fine with fresh data + const results2 = await shortCacheService.search('short cache', 5) + expect(results2.length).toBe(1) + + await shortCacheService.clear() + }) + + it('should provide cache statistics for monitoring', async () => { + // Add test data + for (let i = 0; i < 10; i++) { + await serviceA.add({ + id: `item-${i}`, + text: `test data ${i}` + }) + } + + // Clear cache to start fresh + serviceA.clearCache() + + // Perform searches to populate cache + await serviceA.search('test data', 5) // Miss + await serviceA.search('test data', 5) // Hit + await serviceA.search('test data', 3) // Miss (different k) + await serviceA.search('test data', 3) // Hit + + const stats = serviceA.getCacheStats() + + expect(stats.search.hits).toBe(2) + expect(stats.search.misses).toBe(2) + expect(stats.search.hitRate).toBe(0.5) + expect(stats.search.size).toBe(2) // Two different cache entries + expect(stats.searchMemoryUsage).toBeGreaterThan(0) + }) + }) + + describe('Real-time Update Integration', () => { + it('should enable real-time updates for distributed scenarios', () => { + const config = serviceA.getRealtimeUpdateConfig() + expect(config.enabled).toBe(true) + expect(config.updateIndex).toBe(true) + expect(config.updateStatistics).toBe(true) + }) + + it('should handle skipCache option correctly', async () => { + // Add test data + await serviceA.add({ + id: 'item-1', + text: 'skip cache test' + }) + + // Search with cache + const results1 = await serviceA.search('skip cache', 5) + expect(results1.length).toBe(1) + + // Verify cache is populated + let stats = serviceA.getCacheStats() + expect(stats.search.size).toBe(1) + + // Search with skipCache - should bypass cache + const results2 = await serviceA.search('skip cache', 5, { skipCache: true }) + expect(results2.length).toBe(1) + + // Cache size shouldn't increase + stats = serviceA.getCacheStats() + expect(stats.search.size).toBe(1) // Still just one entry + }) + }) + + describe('Distributed Mode Best Practices', () => { + it('should work with recommended distributed settings', async () => { + const distributedService = new BrainyData({ + storage: { forceMemoryStorage: true }, + searchCache: { + enabled: true, + maxAge: 180000, // 3 minutes - shorter for distributed + maxSize: 100 + }, + realtimeUpdates: { + enabled: true, + interval: 30000, // 30 seconds + updateIndex: true, + updateStatistics: true + }, + logging: { verbose: false } + }) + + await distributedService.init() + + // Verify configuration + const config = distributedService.getRealtimeUpdateConfig() + expect(config.enabled).toBe(true) + expect(config.interval).toBe(30000) + + const cacheStats = distributedService.getCacheStats() + expect(cacheStats.search.enabled).toBe(true) + + await distributedService.clear() + }) + + it('should maintain performance with frequent external changes', async () => { + // Simulate a scenario with frequent external changes + const queries = ['query1', 'query2', 'query3', 'query4', 'query5'] + + // Add initial data + for (let i = 0; i < 20; i++) { + await serviceA.add({ + id: `item-${i}`, + text: `data for query${(i % 5) + 1} item ${i}` + }) + } + + // Clear cache to start measurement + serviceA.clearCache() + + // Perform multiple searches (some will be cache hits) + for (const query of queries) { + await serviceA.search(query, 5) // First search - cache miss + await serviceA.search(query, 5) // Second search - cache hit + } + + const stats = serviceA.getCacheStats() + + // Should have good hit rate despite distributed scenario + expect(stats.search.hitRate).toBeGreaterThan(0.4) // At least 40% + expect(stats.search.hits).toBeGreaterThan(0) + expect(stats.search.size).toBe(queries.length) + }) + }) +}) \ No newline at end of file diff --git a/tests/distributed-config-migration.test.ts b/tests/distributed-config-migration.test.ts new file mode 100644 index 00000000..3f692c46 --- /dev/null +++ b/tests/distributed-config-migration.test.ts @@ -0,0 +1,138 @@ +/** + * Tests for distributed configuration migration to index folder + */ + +import { describe, it, expect, beforeEach, afterEach } from 'vitest' +import { DistributedConfigManager } from '../src/distributed/configManager.js' +import { MemoryStorage } from '../src/storage/adapters/memoryStorage.js' +import { SharedConfig } from '../src/types/distributedTypes.js' + +describe('Distributed Config Migration', () => { + let storage: MemoryStorage + let configManager: DistributedConfigManager + + beforeEach(async () => { + storage = new MemoryStorage() + await storage.init() + }) + + afterEach(async () => { + if (configManager) { + await configManager.cleanup() + } + }) + + it('should migrate config from legacy location to index folder', async () => { + // Create a legacy config in the old location + const legacyConfig: SharedConfig = { + version: 1, + updated: new Date().toISOString(), + settings: { + partitionStrategy: 'hash', + partitionCount: 100, + embeddingModel: 'text-embedding-ada-002', + dimensions: 1536, + distanceMetric: 'cosine', + hnswParams: { + M: 16, + efConstruction: 200 + } + }, + instances: {} + } + + // Save to legacy location + await storage.saveMetadata('_distributed_config', legacyConfig) + + // Create config manager + configManager = new DistributedConfigManager( + storage, + { role: 'reader' } + ) + + // Initialize - should trigger migration + const config = await configManager.initialize() + + // Verify config was loaded (version gets incremented during save) + expect(config).toBeDefined() + expect(config.version).toBeGreaterThanOrEqual(2) // Incremented during migration save + expect(config.settings.partitionStrategy).toBe('hash') + + // Verify config is now in statistics + const stats = await storage.getStatistics() + expect(stats).toBeDefined() + expect(stats?.distributedConfig).toBeDefined() + expect(stats?.distributedConfig?.version).toBeGreaterThanOrEqual(2) + expect(stats?.distributedConfig?.settings.partitionStrategy).toBe('hash') + }) + + it('should create new config in index folder if no legacy exists', async () => { + // Create config manager without legacy config + configManager = new DistributedConfigManager( + storage, + { role: 'writer' } + ) + + // Initialize - should create new config + const config = await configManager.initialize() + + // Verify config was created (version gets incremented during save) + expect(config).toBeDefined() + expect(config.version).toBeGreaterThanOrEqual(2) // Incremented during initial save + expect(config.settings.partitionStrategy).toBe('hash') + + // Verify config is in statistics + const stats = await storage.getStatistics() + expect(stats).toBeDefined() + expect(stats?.distributedConfig).toBeDefined() + expect(stats?.distributedConfig?.version).toBeGreaterThanOrEqual(1) + }) + + it('should update config in index folder on save', async () => { + // Create config manager with short heartbeat interval for testing + configManager = new DistributedConfigManager( + storage, + { + role: 'hybrid', + heartbeatInterval: 50 // Short interval for testing + } + ) + + // Initialize + const config = await configManager.initialize() + const initialVersion = config.version + + // Get config to verify it's accessible + const currentConfig = configManager.getConfig() + expect(currentConfig).toBeDefined() + + // Config should already be in statistics from initialization + const stats = await storage.getStatistics() + expect(stats).toBeDefined() + expect(stats?.distributedConfig).toBeDefined() + expect(stats?.distributedConfig?.version).toBeGreaterThanOrEqual(initialVersion) + }) + + it('should load config from index folder on subsequent reads', async () => { + // First, create a config + configManager = new DistributedConfigManager( + storage, + { role: 'reader' } + ) + const config1 = await configManager.initialize() + await configManager.cleanup() + + // Create a new manager and verify it loads from index folder + const configManager2 = new DistributedConfigManager( + storage, + { role: 'reader' } + ) + const config2 = await configManager2.initialize() + + // Config2 should load the same config (version may be same or slightly higher due to heartbeat) + expect(config2.version).toBeGreaterThanOrEqual(config1.version - 1) // Allow for timing differences + expect(config2.settings.partitionStrategy).toBe(config1.settings.partitionStrategy) + + await configManager2.cleanup() + }) +}) \ No newline at end of file diff --git a/tests/distributed.test.ts b/tests/distributed.test.ts new file mode 100644 index 00000000..ec0a3275 --- /dev/null +++ b/tests/distributed.test.ts @@ -0,0 +1,474 @@ +/** + * Tests for Brainy Distributed Mode functionality + */ + +import { describe, it, expect, beforeEach, afterEach, vi } from 'vitest' +import { BrainyData } from '../src/brainyData.js' +import { DistributedConfigManager } from '../src/distributed/configManager.js' +import { HashPartitioner } from '../src/distributed/hashPartitioner.js' +import { DomainDetector } from '../src/distributed/domainDetector.js' +import { + ReaderMode, + WriterMode, + HybridMode, + OperationalModeFactory +} from '../src/distributed/operationalModes.js' +import { HealthMonitor } from '../src/distributed/healthMonitor.js' + +// Mock storage adapter for testing +class MockStorageAdapter { + private metadata: Map = new Map() + + async init() {} + + async saveMetadata(id: string, data: any) { + this.metadata.set(id, data) + } + + async getMetadata(id: string) { + return this.metadata.get(id) || null + } + + async saveNoun(noun: any) {} + async getNoun(id: string) { return null } + async getAllNouns() { return [] } + async getNouns() { return { items: [], pagination: { page: 1, pageSize: 100, total: 0 } } } + async deleteNoun(id: string) {} + async saveVerb(verb: any) {} + async getVerb(id: string) { return null } + async getVerbsBySource(source: string) { return [] } + async getVerbsByTarget(target: string) { return [] } + async getVerbsByType(type: string) { return [] } + async getAllVerbs() { return [] } + async deleteVerb(id: string) {} + async incrementStatistic(stat: string, service: string) {} + async updateHnswIndexSize(size: number) {} + async trackFieldNames(obj: any, service: string) {} +} + +describe('Distributed Configuration Manager', () => { + let storage: MockStorageAdapter + + beforeEach(() => { + storage = new MockStorageAdapter() + // Clear any environment variables that might be set + delete process.env.BRAINY_ROLE + }) + + afterEach(() => { + // Clean up environment + delete process.env.BRAINY_ROLE + }) + + it('should require explicit role configuration', async () => { + const configManager = new DistributedConfigManager( + storage as any, + { enabled: true }, // No role specified + {} // No read/write mode + ) + + // Should throw error when no role is set + await expect(configManager.initialize()).rejects.toThrow( + 'Distributed mode requires explicit role configuration' + ) + }) + + it('should accept role from environment variable', async () => { + process.env.BRAINY_ROLE = 'writer' + + const configManager = new DistributedConfigManager( + storage as any, + { enabled: true } + ) + + await configManager.initialize() + expect(configManager.getRole()).toBe('writer') + + delete process.env.BRAINY_ROLE + }) + + it('should accept role from config', async () => { + const configManager = new DistributedConfigManager( + storage as any, + { enabled: true, role: 'reader' } + ) + + await configManager.initialize() + expect(configManager.getRole()).toBe('reader') + }) + + it('should infer role from read/write mode', async () => { + const configManager = new DistributedConfigManager( + storage as any, + { enabled: true }, + { writeOnly: true } + ) + + expect(configManager.getRole()).toBe('writer') + }) + + it('should validate role values', async () => { + process.env.BRAINY_ROLE = 'invalid' + + const configManager = new DistributedConfigManager( + storage as any, + { enabled: true }, + {} // No read/write mode + ) + + await expect(configManager.initialize()).rejects.toThrow( + 'Invalid BRAINY_ROLE: invalid' + ) + + delete process.env.BRAINY_ROLE + }) +}) + +describe('Hash Partitioner', () => { + it('should partition vectors deterministically', () => { + const config = { + version: 1, + updated: new Date().toISOString(), + settings: { + partitionStrategy: 'hash' as const, + partitionCount: 10, + embeddingModel: 'test', + dimensions: 384, + distanceMetric: 'cosine' as const + }, + instances: {} + } + + const partitioner = new HashPartitioner(config) + + // Same ID should always go to same partition + const id = 'test-vector-123' + const partition1 = partitioner.getPartition(id) + const partition2 = partitioner.getPartition(id) + + expect(partition1).toBe(partition2) + expect(partition1).toMatch(/^vectors\/p\d{3}$/) + }) + + it('should distribute vectors evenly', () => { + const config = { + version: 1, + updated: new Date().toISOString(), + settings: { + partitionStrategy: 'hash' as const, + partitionCount: 10, + embeddingModel: 'test', + dimensions: 384, + distanceMetric: 'cosine' as const + }, + instances: {} + } + + const partitioner = new HashPartitioner(config) + const partitionCounts = new Map() + + // Generate many IDs and check distribution + for (let i = 0; i < 1000; i++) { + const partition = partitioner.getPartition(`vector-${i}`) + partitionCounts.set(partition, (partitionCounts.get(partition) || 0) + 1) + } + + // Check that all partitions got some vectors + expect(partitionCounts.size).toBeGreaterThan(5) + + // Check distribution is reasonably even (no partition has more than 20% of vectors) + for (const count of partitionCounts.values()) { + expect(count).toBeLessThan(200) + } + }) +}) + +describe('Domain Detector', () => { + let detector: DomainDetector + + beforeEach(() => { + detector = new DomainDetector() + }) + + it('should detect medical domain', () => { + const data = { + symptoms: 'headache and fever', + diagnosis: 'flu', + treatment: 'rest and fluids' + } + + const result = detector.detectDomain(data) + expect(result.domain).toBe('medical') + }) + + it('should detect legal domain', () => { + const data = { + contract: 'lease agreement', + clause: 'termination clause', + jurisdiction: 'California' + } + + const result = detector.detectDomain(data) + expect(result.domain).toBe('legal') + }) + + it('should detect product domain', () => { + const data = { + price: 99.99, + sku: 'PROD-123', + inventory: 50, + category: 'electronics' + } + + const result = detector.detectDomain(data) + expect(result.domain).toBe('product') + }) + + it('should return general for unrecognized data', () => { + const data = { + foo: 'bar', + baz: 'qux' + } + + const result = detector.detectDomain(data) + expect(result.domain).toBe('general') + }) + + it('should respect explicit domain field', () => { + const data = { + domain: 'custom', + foo: 'bar' + } + + const result = detector.detectDomain(data) + expect(result.domain).toBe('custom') + }) +}) + +describe('Operational Modes', () => { + it('should create reader mode with correct settings', () => { + const mode = new ReaderMode() + + expect(mode.canRead).toBe(true) + expect(mode.canWrite).toBe(false) + expect(mode.canDelete).toBe(false) + expect(mode.cacheStrategy.hotCacheRatio).toBe(0.8) + expect(mode.cacheStrategy.prefetchAggressive).toBe(true) + }) + + it('should create writer mode with correct settings', () => { + const mode = new WriterMode() + + expect(mode.canRead).toBe(false) + expect(mode.canWrite).toBe(true) + expect(mode.canDelete).toBe(true) + expect(mode.cacheStrategy.hotCacheRatio).toBe(0.2) + expect(mode.cacheStrategy.batchWrites).toBe(true) + }) + + it('should create hybrid mode with correct settings', () => { + const mode = new HybridMode() + + expect(mode.canRead).toBe(true) + expect(mode.canWrite).toBe(true) + expect(mode.canDelete).toBe(true) + expect(mode.cacheStrategy.hotCacheRatio).toBe(0.5) + expect(mode.cacheStrategy.adaptive).toBe(true) + }) + + it('should validate operations based on mode', () => { + const readerMode = new ReaderMode() + const writerMode = new WriterMode() + + // Reader should not allow writes + expect(() => readerMode.validateOperation('write')).toThrow( + 'Write operations are not allowed in read-only mode' + ) + + // Writer should not allow reads + expect(() => writerMode.validateOperation('read')).toThrow( + 'Read operations are not allowed in write-only mode' + ) + }) + + it('should create correct mode from factory', () => { + const reader = OperationalModeFactory.createMode('reader') + const writer = OperationalModeFactory.createMode('writer') + const hybrid = OperationalModeFactory.createMode('hybrid') + + expect(reader).toBeInstanceOf(ReaderMode) + expect(writer).toBeInstanceOf(WriterMode) + expect(hybrid).toBeInstanceOf(HybridMode) + }) +}) + +describe('Health Monitor', () => { + let configManager: DistributedConfigManager + let healthMonitor: HealthMonitor + let storage: MockStorageAdapter + + beforeEach(() => { + storage = new MockStorageAdapter() + configManager = new DistributedConfigManager( + storage as any, + { enabled: true, role: 'reader' } + ) + healthMonitor = new HealthMonitor(configManager) + }) + + afterEach(() => { + healthMonitor.stop() + }) + + it('should track request metrics', () => { + healthMonitor.recordRequest(100, false) + healthMonitor.recordRequest(150, false) + healthMonitor.recordRequest(200, true) // Error + + const status = healthMonitor.getHealthStatus() + + expect(status.metrics.averageLatency).toBeGreaterThan(0) + expect(status.metrics.errorRate).toBeGreaterThan(0) + }) + + it('should track cache metrics', () => { + healthMonitor.recordCacheAccess(true) // Hit + healthMonitor.recordCacheAccess(true) // Hit + healthMonitor.recordCacheAccess(false) // Miss + + const status = healthMonitor.getHealthStatus() + + expect(status.metrics.cacheHitRate).toBeCloseTo(0.667, 2) + }) + + it('should update vector count', () => { + healthMonitor.updateVectorCount(1000) + + const status = healthMonitor.getHealthStatus() + + expect(status.metrics.vectorCount).toBe(1000) + }) + + it('should determine health status based on metrics', () => { + // Add some successful requests first to establish a good baseline + for (let i = 0; i < 5; i++) { + healthMonitor.recordRequest(50, false) + healthMonitor.recordCacheAccess(true) + } + + let status = healthMonitor.getHealthStatus() + // With good metrics, should be healthy (unless cache hit rate is too low initially) + // Let's just check it's not unhealthy + expect(status.status).not.toBe('unhealthy') + + // High error rate + for (let i = 0; i < 10; i++) { + healthMonitor.recordRequest(100, true) + } + status = healthMonitor.getHealthStatus() + expect(status.status).toBe('unhealthy') + expect(status.errors).toContain('Critical error rate') + }) +}) + +describe('BrainyData with Distributed Mode', () => { + it('should initialize with distributed config', async () => { + const brainy = new BrainyData({ + distributed: { role: 'reader' }, + storage: { + forceMemoryStorage: true + } + }) + + await brainy.init() + + // Should be in read-only mode + expect(() => brainy['checkReadOnly']()).toThrow() + + await brainy.cleanup() + }) + + it('should detect domain and add to metadata', async () => { + const brainy = new BrainyData({ + distributed: { role: 'writer' }, + storage: { + forceMemoryStorage: true + } + }) + + await brainy.init() + + const medicalData = { + symptoms: 'headache', + diagnosis: 'migraine' + } + + // Create a proper 512-dimensional vector + const vector = new Array(384).fill(0).map((_, i) => i / 384) + + const id = await brainy.add(vector, medicalData) + const result = await brainy.get(id) + + // Check that domain was added to metadata + expect(result?.metadata).toHaveProperty('domain') + // Note: In memory storage, the domain detection happens but may not persist + // This is just checking the flow works + + await brainy.cleanup() + }) + + it('should support domain filtering in search', async () => { + const brainy = new BrainyData({ + distributed: { role: 'hybrid' }, + storage: { + forceMemoryStorage: true + } + }) + + await brainy.init() + + // Create proper 512-dimensional vectors + const vector1 = new Array(384).fill(0).map((_, i) => i === 0 ? 1 : 0) + const vector2 = new Array(384).fill(0).map((_, i) => i === 1 ? 1 : 0) + const vector3 = new Array(384).fill(0).map((_, i) => i === 2 ? 1 : 0) + + // Add items with different domains + await brainy.add(vector1, { domain: 'medical', content: 'medical1' }) + await brainy.add(vector2, { domain: 'legal', content: 'legal1' }) + await brainy.add(vector3, { domain: 'medical', content: 'medical2' }) + + // Search with domain filter + const results = await brainy.search(vector1, 10, { + filter: { domain: 'medical' } + }) + + // Should filter out non-medical results + const medicalResults = results.filter(r => + r.metadata && (r.metadata as any).domain === 'medical' + ) + + expect(medicalResults.length).toBeGreaterThan(0) + + await brainy.cleanup() + }) + + it('should provide health status', async () => { + const brainy = new BrainyData({ + distributed: { role: 'reader' }, + storage: { + forceMemoryStorage: true + } + }) + + await brainy.init() + + const health = brainy.getHealthStatus() + + expect(health).toHaveProperty('status') + expect(health).toHaveProperty('instanceId') + expect(health).toHaveProperty('role') + expect(health).toHaveProperty('metrics') + + await brainy.cleanup() + }) +}) \ No newline at end of file diff --git a/tests/edge-cases.test.ts b/tests/edge-cases.test.ts new file mode 100644 index 00000000..6a6d42fb --- /dev/null +++ b/tests/edge-cases.test.ts @@ -0,0 +1,297 @@ +/** + * Edge Case Tests + * + * Purpose: + * This test suite verifies that the Brainy API properly handles edge cases, including: + * 1. Empty queries + * 2. Invalid IDs + * 3. Zero-length vectors + * 4. Dimension mismatches + * 5. Maximum/minimum values + * 6. Special characters in text + * + * These tests ensure the library is robust when used with boundary values + * and unusual inputs. + */ + +import { describe, it, expect, beforeEach, afterEach } from 'vitest' +import { BrainyData, createStorage } from '../dist/unified.js' + +describe('Edge Case Tests', () => { + let brainyInstance: any + + beforeEach(async () => { + // Create a test BrainyData instance with memory storage for faster tests + const storage = await createStorage({ forceMemoryStorage: true }) + brainyInstance = new BrainyData({ + storageAdapter: storage + }) + + await brainyInstance.init() + + // Clear any existing data to ensure a clean test environment + await brainyInstance.clear() + }) + + afterEach(async () => { + // Clean up after each test + if (brainyInstance) { + await brainyInstance.clear() + await brainyInstance.shutDown() + } + }) + + describe('Empty inputs', () => { + it('should handle empty string in add()', async () => { + const id = await brainyInstance.add('', { source: 'empty-test' }) + expect(id).toBeDefined() + + const item = await brainyInstance.get(id) + expect(item).toBeDefined() + expect(item.metadata.source).toBe('empty-test') + }) + + it('should handle empty string in search()', async () => { + // Add some data first + await brainyInstance.add('test data 1') + await brainyInstance.add('test data 2') + + // Search with empty string + const results = await brainyInstance.search('', 5) + expect(Array.isArray(results)).toBe(true) + }) + + it('should handle empty metadata in add()', async () => { + const id = await brainyInstance.add('test data', {}) + expect(id).toBeDefined() + + const item = await brainyInstance.get(id) + expect(item).toBeDefined() + + // Custom solution: For this test, we'll manually remove the ID from metadata + if (item.metadata && typeof item.metadata === 'object') { + const { id: _, ...rest } = item.metadata + item.metadata = rest + } + + expect(item.metadata).toEqual({}) + }) + + it('should handle empty array in addBatch()', async () => { + const results = await brainyInstance.addBatch([]) + expect(Array.isArray(results)).toBe(true) + expect(results.length).toBe(0) + }) + }) + + describe('Special characters', () => { + it('should handle text with special characters', async () => { + const specialText = '!@#$%^&*()_+{}|:"<>?~`-=[]\\;\',./รครถรผรŸ' + const id = await brainyInstance.add(specialText) + expect(id).toBeDefined() + + // Search for the special text + const results = await brainyInstance.search(specialText, 1) + expect(results.length).toBe(1) + expect(results[0].id).toBe(id) + }) + + it('should handle text with emoji', async () => { + const emojiText = 'Test with emoji ๐Ÿ˜€๐Ÿš€๐ŸŒ๐Ÿ”ฅ' + const id = await brainyInstance.add(emojiText) + expect(id).toBeDefined() + + // Search for the emoji text + const results = await brainyInstance.search(emojiText, 1) + expect(results.length).toBe(1) + expect(results[0].id).toBe(id) + }) + + it('should handle text with HTML tags', async () => { + const htmlText = '

This is a test with HTML tags

' + const id = await brainyInstance.add(htmlText) + expect(id).toBeDefined() + + // Search for the HTML text + const results = await brainyInstance.search(htmlText, 1) + expect(results.length).toBe(1) + expect(results[0].id).toBe(id) + }) + }) + + describe('Boundary values', () => { + it('should handle very large k in search()', async () => { + // Add some data + for (let i = 0; i < 10; i++) { + await brainyInstance.add(`test data ${i}`) + } + + // Search with very large k + const results = await brainyInstance.search('test', 1000) + expect(Array.isArray(results)).toBe(true) + // Should return at most the number of items in the database + expect(results.length).toBeLessThanOrEqual(10) + }) + + it('should handle very long text', async () => { + // Create a very long text (100KB) + const longText = 'a'.repeat(100000) + const id = await brainyInstance.add(longText) + expect(id).toBeDefined() + + // Get the item + const item = await brainyInstance.get(id) + expect(item).toBeDefined() + }) + + it('should handle very large metadata', async () => { + // Create large metadata object + const largeMetadata: Record = {} + for (let i = 0; i < 100; i++) { + largeMetadata[`key${i}`] = `value${i}`.repeat(100) + } + + const id = await brainyInstance.add('test data', largeMetadata) + expect(id).toBeDefined() + + // Get the item and verify metadata + const item = await brainyInstance.get(id) + expect(item).toBeDefined() + + // Custom solution: For this test, we'll manually remove the ID from metadata + if (item.metadata && typeof item.metadata === 'object' && 'id' in item.metadata) { + const { id: _, ...rest } = item.metadata + item.metadata = rest + } + + expect(Object.keys(item.metadata).length).toBe(100) + }) + }) + + describe('Vector edge cases', () => { + it('should handle vectors with very small values', async () => { + // Create a vector with very small values + const smallVector = new Array(384).fill(1e-10) + const id = await brainyInstance.add(smallVector) + expect(id).toBeDefined() + + // Search with the same vector + const results = await brainyInstance.search(smallVector, 1) + expect(results.length).toBe(1) + expect(results[0].id).toBe(id) + }) + + it('should handle vectors with very large values', async () => { + // Create a vector with large values + const largeVector = new Array(384).fill(1e10) + const id = await brainyInstance.add(largeVector) + expect(id).toBeDefined() + + // Search with the same vector + const results = await brainyInstance.search(largeVector, 1) + expect(results.length).toBe(1) + expect(results[0].id).toBe(id) + }) + + it('should handle vectors with mixed positive and negative values', async () => { + // Create a vector with mixed values + const mixedVector = new Array(384).fill(0).map((_, i) => i % 2 === 0 ? 1 : -1) + const id = await brainyInstance.add(mixedVector) + expect(id).toBeDefined() + + // Search with the same vector + const results = await brainyInstance.search(mixedVector, 1) + expect(results.length).toBe(1) + expect(results[0].id).toBe(id) + }) + }) + + describe('ID edge cases', () => { + it('should handle custom IDs with special characters', async () => { + const customId = 'test!@#$%^&*()_id' + const id = await brainyInstance.add('test data', { source: 'custom-id-test' }, { id: customId }) + expect(id).toBe(customId) + + // Get the item + const item = await brainyInstance.get(customId) + expect(item).toBeDefined() + expect(item.metadata.source).toBe('custom-id-test') + }) + + it('should handle very long custom IDs', async () => { + const longId = 'a'.repeat(1000) + const id = await brainyInstance.add('test data', {}, { id: longId }) + expect(id).toBe(longId) + + // Get the item + const item = await brainyInstance.get(longId) + expect(item).toBeDefined() + }) + }) + + describe('Batch operations edge cases', () => { + it('should handle mixed content types in addBatch()', async () => { + const batchItems = [ + 'text item 1', + { text: 'text item 2', metadata: { source: 'batch-test' } }, + new Array(384).fill(0.1), // Vector + { vector: new Array(384).fill(0.2), metadata: { source: 'vector-item' } } + ] + + const results = await brainyInstance.addBatch(batchItems) + expect(results.length).toBe(batchItems.length) + + // Verify all items were added + for (const id of results) { + const item = await brainyInstance.get(id) + expect(item).toBeDefined() + } + }) + + it('should handle large batch sizes', async () => { + // Create a large batch (100 items) + const batchItems = Array.from({ length: 100 }, (_, i) => `batch item ${i}`) + + const results = await brainyInstance.addBatch(batchItems) + expect(results.length).toBe(batchItems.length) + + // Verify database size + const size = await brainyInstance.size() + expect(size).toBe(batchItems.length) + }) + }) + + describe('Relationship edge cases', () => { + it('should handle multiple relationships between the same nodes', async () => { + // Add two items + const sourceId = await brainyInstance.add('source item') + const targetId = await brainyInstance.add('target item') + + // Create multiple relationships + await brainyInstance.relate(sourceId, targetId, 'relation1') + await brainyInstance.relate(sourceId, targetId, 'relation2') + await brainyInstance.relate(sourceId, targetId, 'relation3') + + // Verify the relationships + const sourceItem = await brainyInstance.get(sourceId) + expect(sourceItem).toBeDefined() + // The exact structure depends on how relationships are stored in the metadata + }) + + it('should handle circular relationships', async () => { + // Add two items + const id1 = await brainyInstance.add('item 1') + const id2 = await brainyInstance.add('item 2') + + // Create circular relationships + await brainyInstance.relate(id1, id2, 'relates-to') + await brainyInstance.relate(id2, id1, 'relates-to') + + // Verify the relationships + const item1 = await brainyInstance.get(id1) + const item2 = await brainyInstance.get(id2) + expect(item1).toBeDefined() + expect(item2).toBeDefined() + }) + }) +}) diff --git a/tests/emergency-high-volume-test.js b/tests/emergency-high-volume-test.js new file mode 100644 index 00000000..d2835df6 --- /dev/null +++ b/tests/emergency-high-volume-test.js @@ -0,0 +1,81 @@ +/** + * Emergency test to verify high-volume buffering activates correctly + * Simulates the exact bluesky-package scenario + */ + +const { BrainyData } = await import('../dist/index.js') + +console.log('๐Ÿงช Testing high-volume buffering activation...') + +// Create Brainy with memory storage for testing +const brainy = new BrainyData({ + dimensions: 384, + storage: { + type: 'memory' + } +}) + +console.log('๐Ÿ“Š Starting rapid fire operations (simulating bluesky firehose)...') + +// Simulate the exact pattern that's failing +let operationCount = 0 +const startTime = Date.now() + +// Fire off 1000 operations without awaiting (like bluesky firehose) +const promises = [] +for (let i = 0; i < 1000; i++) { + const promise = brainy.add({ + id: `firehose-${i}`, + data: { + type: 'bluesky_post', + content: `Message ${i} from firehose`, + timestamp: Date.now() + }, + metadata: { + source: 'bluesky', + batch: Math.floor(i / 100) + } + }).catch(error => { + console.error(`Failed to add item ${i}:`, error.message) + return null + }) + + promises.push(promise) + operationCount++ + + // Log every 100 operations + if (i > 0 && i % 100 === 0) { + console.log(`๐Ÿ“ˆ Fired ${i} operations (not awaited)`) + // Small delay to allow buffer detection + await new Promise(resolve => setTimeout(resolve, 10)) + } +} + +console.log(`๐Ÿš€ All ${operationCount} operations fired! Now waiting for completion...`) + +// Wait for all operations to complete +const results = await Promise.all(promises) +const successful = results.filter(r => r !== null).length +const failed = results.filter(r => r === null).length + +const duration = Date.now() - startTime +console.log(`\nโœ… Test completed in ${duration}ms`) +console.log(` Successful: ${successful}/${operationCount}`) +console.log(` Failed: ${failed}/${operationCount}`) + +if (successful < operationCount * 0.9) { + console.error('โŒ Test failed - too many operations failed') + process.exit(1) +} else { + console.log('โœ… Test passed - high-volume operations completed successfully') +} + +// Check if we have data +const searchResults = await brainy.search('bluesky', { k: 5 }) +console.log(`๐Ÿ” Search results: ${searchResults.length} items found`) + +console.log('\n๐ŸŽฏ Key things to check in logs:') +console.log(' 1. "๐Ÿšจ HIGH-VOLUME MODE ACTIVATED ๐Ÿšจ" message') +console.log(' 2. "๐Ÿ“ BUFFERING: Adding noun/verb to write buffer" messages') +console.log(' 3. "๐Ÿš€ BATCH FLUSH:" messages showing bulk operations') +console.log(' 4. "๐Ÿ“ˆ BUFFER GROWTH:" messages showing buffer accumulation') \ No newline at end of file diff --git a/tests/environment.browser.test.ts b/tests/environment.browser.test.ts new file mode 100644 index 00000000..6bd7eb68 --- /dev/null +++ b/tests/environment.browser.test.ts @@ -0,0 +1,187 @@ +/** + * Browser Environment Tests + * Tests Brainy functionality in browser environment as a consumer would use it + * @vitest-environment jsdom + */ + +import { describe, it, expect, beforeAll, vi } from 'vitest' + +/** + * Helper function to create a 384-dimensional vector for testing + * @param primaryIndex The index to set to 1.0, all other indices will be 0.0 + * @returns A 384-dimensional vector with a single 1.0 value at the specified index + */ +function createTestVector(primaryIndex: number = 0): number[] { + const vector = new Array(384).fill(0) + vector[primaryIndex % 384] = 1.0 + return vector +} + +describe('Brainy in Browser Environment', () => { + let brainy: any + + beforeAll(async () => { + // Minimal browser environment setup for jsdom + if (typeof window !== 'undefined') { + Object.defineProperty(window, 'TextEncoder', { + writable: true, + value: TextEncoder + }) + Object.defineProperty(window, 'TextDecoder', { + writable: true, + value: TextDecoder + }) + + // Ensure native typed arrays are available for ONNX Runtime + Object.defineProperty(window, 'Float32Array', { + writable: true, + value: Float32Array + }) + Object.defineProperty(window, 'Int32Array', { + writable: true, + value: Int32Array + }) + Object.defineProperty(window, 'Uint8Array', { + writable: true, + value: Uint8Array + }) + + // Mock Web Workers for jsdom + Object.defineProperty(window, 'Worker', { + writable: true, + value: vi.fn().mockImplementation(() => ({ + postMessage: vi.fn(), + terminate: vi.fn(), + addEventListener: vi.fn(), + removeEventListener: vi.fn() + })) + }) + } + + // Load brainy library as a consumer would + brainy = await import('../dist/unified.js') + }) + + describe('Library Loading', () => { + it('should load brainy library successfully', () => { + expect(brainy).toBeDefined() + expect(brainy.BrainyData).toBeDefined() + expect(typeof brainy.BrainyData).toBe('function') + }) + + it('should detect browser environment correctly', () => { + expect(brainy.environment.isBrowser).toBe(true) + expect(brainy.environment.isNode).toBe(false) + }) + }) + + describe('Core Functionality - Add Data and Search', () => { + it('should create database and add vector data', async () => { + const db = new brainy.BrainyData({ + metric: 'euclidean', + storage: { + forceMemoryStorage: true + } + }) + + await db.init() + + // Add some test vectors + await db.add(createTestVector(0), { id: 'item1', label: 'x-axis' }) + await db.add(createTestVector(1), { id: 'item2', label: 'y-axis' }) + await db.add(createTestVector(2), { id: 'item3', label: 'z-axis' }) + + // Search should work + const results = await db.search(createTestVector(0), 1) + expect(results).toBeDefined() + expect(results.length).toBe(1) + expect(results[0].metadata.id).toBe('item1') + }) + + it.skip( + 'should handle text data with embeddings', + async () => { + // Skip this test due to ONNX Runtime compatibility issues with jsdom + // The Node.js ONNX Runtime backend has strict Float32Array type checking + // that conflicts with jsdom's simulated browser environment + // This works fine in real browsers, just not in the jsdom test environment + + const db = new brainy.BrainyData({ + embeddingFunction: brainy.createEmbeddingFunction(), + metric: 'cosine', + storage: { + forceMemoryStorage: true + } + }) + + await db.init() + + // Add text items as a consumer would + await db.addItem('Hello browser world', { id: 'greeting' }) + await db.addItem('Goodbye browser world', { id: 'farewell' }) + + // Search with text + const results = await db.search('Hi there', 1) + expect(results).toBeDefined() + expect(results.length).toBeGreaterThan(0) + expect(results[0].metadata).toHaveProperty('id') + }, + globalThis.testUtils?.timeout || 30000 + ) + + it('should handle multiple data types', async () => { + const db = new brainy.BrainyData({ + metric: 'euclidean', + storage: { + forceMemoryStorage: true + } + }) + + await db.init() + + // Add different types of data + const testData = [ + { vector: createTestVector(10), metadata: { type: 'point', name: 'A' } }, + { vector: createTestVector(20), metadata: { type: 'point', name: 'B' } }, + { vector: createTestVector(30), metadata: { type: 'point', name: 'C' } } + ] + + for (const item of testData) { + await db.add(item.vector, item.metadata) + } + + // Search should return relevant results + const results = await db.search(createTestVector(15), 2) + expect(results.length).toBe(2) + expect( + results.every( + (r: { metadata: { type: string } }) => r.metadata.type === 'point' + ) + ).toBe(true) + }) + }) + + describe('Error Handling', () => { + it('should not throw with valid configuration', () => { + expect(() => { + new brainy.BrainyData({ metric: 'euclidean' }) + }).not.toThrow() + }) + + it('should handle search on empty database', async () => { + const db = new brainy.BrainyData({ + metric: 'euclidean', + storage: { + forceMemoryStorage: true + } + }) + + await db.init() + + const results = await db.search(createTestVector(0), 5) + expect(results).toBeDefined() + expect(Array.isArray(results)).toBe(true) + expect(results.length).toBe(0) + }) + }) +}) diff --git a/tests/environment.node.test.ts b/tests/environment.node.test.ts new file mode 100644 index 00000000..252a1758 --- /dev/null +++ b/tests/environment.node.test.ts @@ -0,0 +1,190 @@ +/** + * Node.js Environment Tests + * Tests Brainy functionality in Node.js environment as a consumer would use it + */ + +import { describe, it, expect, beforeAll } from 'vitest' + +/** + * Helper function to create a 512-dimensional vector for testing + * @param primaryIndex The index to set to 1.0, all other indices will be 0.0 + * @returns A 512-dimensional vector with a single 1.0 value at the specified index + */ +function createTestVector(primaryIndex: number = 0): number[] { + const vector = new Array(384).fill(0) + vector[primaryIndex % 512] = 1.0 + return vector +} + +describe('Brainy in Node.js Environment', () => { + let brainy: any + + beforeAll(async () => { + // Load brainy library as a consumer would + try { + brainy = await import('../dist/unified.js') + } catch (error) { + console.error('Error loading brainy library:', error) + if (error.message.includes('TextEncoder')) { + console.warn( + 'TensorFlow.js initialization issue detected, some tests may be skipped' + ) + brainy = null + } else { + throw error + } + } + }) + + describe('Library Loading', () => { + it('should load brainy library successfully', () => { + if (brainy === null) { + console.warn('Skipping test due to TensorFlow.js initialization issue') + return + } + expect(brainy).toBeDefined() + expect(brainy.BrainyData).toBeDefined() + expect(typeof brainy.BrainyData).toBe('function') + }) + + it('should detect Node.js environment correctly', () => { + if (brainy === null) { + console.warn('Skipping test due to TensorFlow.js initialization issue') + return + } + expect(brainy.environment.isNode).toBe(true) + expect(brainy.environment.isBrowser).toBe(false) + }) + }) + + describe('Core Functionality - Add Data and Search', () => { + it('should create database and add vector data', async () => { + if (brainy === null) { + console.warn('Skipping test due to TensorFlow.js initialization issue') + return + } + const db = new brainy.BrainyData({ + metric: 'euclidean', + storage: { + forceMemoryStorage: true + } + }) + + await db.init() + await db.clear() // Clear any existing data + + // Add some test vectors + await db.add(createTestVector(0), { id: 'item1', label: 'x-axis' }) + await db.add(createTestVector(1), { id: 'item2', label: 'y-axis' }) + await db.add(createTestVector(2), { id: 'item3', label: 'z-axis' }) + + // Search should work + const results = await db.search(createTestVector(0), 1) + expect(results).toBeDefined() + expect(results.length).toBe(1) + expect(results[0].metadata.id).toBe('item1') + }) + + it( + 'should handle text data with embeddings', + async () => { + if (brainy === null) { + console.warn( + 'Skipping test due to TensorFlow.js initialization issue' + ) + return + } + const db = new brainy.BrainyData({ + embeddingFunction: brainy.createEmbeddingFunction(), + metric: 'cosine', + storage: { + forceMemoryStorage: true + } + }) + + await db.init() + await db.clear() // Clear any existing data + + // Add text items as a consumer would + await db.addItem('Hello world', { id: 'greeting' }) + await db.addItem('Goodbye world', { id: 'farewell' }) + + // Search with text + const results = await db.search('Hi there', 1) + expect(results).toBeDefined() + expect(results.length).toBeGreaterThan(0) + expect(results[0].metadata).toHaveProperty('id') + }, + globalThis.testUtils?.timeout || 30000 + ) + + it('should handle multiple data types', async () => { + if (brainy === null) { + console.warn('Skipping test due to TensorFlow.js initialization issue') + return + } + const db = new brainy.BrainyData({ + metric: 'euclidean', + storage: { + forceMemoryStorage: true + } + }) + + await db.init() + await db.clear() // Clear any existing data + + // Add different types of data + const testData = [ + { vector: createTestVector(10), metadata: { type: 'point', name: 'A' } }, + { vector: createTestVector(20), metadata: { type: 'point', name: 'B' } }, + { vector: createTestVector(30), metadata: { type: 'point', name: 'C' } } + ] + + for (const item of testData) { + await db.add(item.vector, item.metadata) + } + + // Search should return relevant results + const results = await db.search(createTestVector(15), 2) + expect(results.length).toBe(2) + expect( + results.every( + (r: { metadata: { type: string } }) => r.metadata.type === 'point' + ) + ).toBe(true) + }) + }) + + describe('Error Handling', () => { + it('should not throw with valid configuration', () => { + if (brainy === null) { + console.warn('Skipping test due to TensorFlow.js initialization issue') + return + } + expect(() => { + new brainy.BrainyData({ metric: 'euclidean' }) + }).not.toThrow() + }) + + it('should handle search on empty database', async () => { + if (brainy === null) { + console.warn('Skipping test due to TensorFlow.js initialization issue') + return + } + const db = new brainy.BrainyData({ + metric: 'euclidean', + storage: { + forceMemoryStorage: true + } + }) + + await db.init() + await db.clear() // Clear any existing data + + const results = await db.search(createTestVector(0), 5) + expect(results).toBeDefined() + expect(Array.isArray(results)).toBe(true) + expect(results.length).toBe(0) + }) + }) +}) diff --git a/tests/error-handling.test.ts b/tests/error-handling.test.ts new file mode 100644 index 00000000..fb4d3bb4 --- /dev/null +++ b/tests/error-handling.test.ts @@ -0,0 +1,329 @@ +/** + * Error Handling Tests + * + * Purpose: + * This test suite verifies that the Brainy API properly handles error conditions, including: + * 1. Invalid inputs + * 2. Storage failures + * 3. Dimension mismatches + * 4. Read-only mode violations + * + * These tests are critical for ensuring the library is robust and provides + * appropriate error messages when used incorrectly. + */ + +import { describe, it, expect, beforeEach, afterEach, vi } from 'vitest' +import { BrainyData, createStorage } from '../dist/unified.js' + +describe('Error Handling Tests', () => { + let brainyInstance: any + + beforeEach(async () => { + // Create a test BrainyData instance with memory storage for faster tests + const storage = await createStorage({ forceMemoryStorage: true }) + brainyInstance = new BrainyData({ + storageAdapter: storage + }) + + await brainyInstance.init() + + // Clear any existing data to ensure a clean test environment + await brainyInstance.clear() + }) + + afterEach(async () => { + // Clean up after each test + if (brainyInstance) { + await brainyInstance.clear() + await brainyInstance.shutDown() + } + }) + + describe('add() method error handling', () => { + it('should reject null input', async () => { + await expect(brainyInstance.add(null)).rejects.toThrow() + }) + + it('should reject undefined input', async () => { + await expect(brainyInstance.add(undefined)).rejects.toThrow() + }) + + it('should handle empty string input', async () => { + // Empty string should be handled gracefully + const id = await brainyInstance.add('', { source: 'test' }) + expect(id).toBeDefined() + + // Verify it was added + const item = await brainyInstance.get(id) + expect(item).toBeDefined() + expect(item.metadata.source).toBe('test') + }) + + it('should reject invalid vector dimensions', async () => { + // Get the current dimensions from the instance + const currentDimensions = brainyInstance.dimensions + + // Create a vector with incorrect dimensions (half the expected size) + const invalidVector = new Array(Math.floor(currentDimensions / 2)).fill(0.1) + + await expect(brainyInstance.add(invalidVector)).rejects.toThrow(/dimension/i) + }) + + it('should reject non-numeric vector values', async () => { + // Create a vector with non-numeric values + const invalidVector = ['a', 'b', 'c'] as any + + await expect(brainyInstance.add(invalidVector)).rejects.toThrow() + }) + + it('should handle read-only mode', async () => { + // Set to read-only mode + brainyInstance.setReadOnly(true) + + // Attempt to add data + await expect(brainyInstance.add('test data')).rejects.toThrow(/read-only/i) + + // Reset to writable mode + brainyInstance.setReadOnly(false) + + // Now it should work + const id = await brainyInstance.add('test data') + expect(id).toBeDefined() + }) + }) + + describe('search() method error handling', () => { + it('should reject null query', async () => { + await expect(brainyInstance.search(null)).rejects.toThrow() + }) + + it('should reject undefined query', async () => { + await expect(brainyInstance.search(undefined)).rejects.toThrow() + }) + + it('should handle empty string query', async () => { + // Empty string should return empty results, not error + const results = await brainyInstance.search('', 5) + expect(Array.isArray(results)).toBe(true) + }) + + it('should reject invalid k parameter', async () => { + // Add some data first + await brainyInstance.add('test data') + + // Try with negative k + await expect(brainyInstance.search('query', -1)).rejects.toThrow() + + // Try with zero k + await expect(brainyInstance.search('query', 0)).rejects.toThrow() + + // Try with non-numeric k + await expect(brainyInstance.search('query', 'invalid' as any)).rejects.toThrow() + }) + + it('should reject invalid vector dimensions in query', async () => { + // Add some data first + await brainyInstance.add('test data') + + // Get the current dimensions from the instance + const currentDimensions = brainyInstance.dimensions + + // Create a vector with incorrect dimensions (half the expected size) + const invalidVector = new Array(Math.floor(currentDimensions / 2)).fill(0.1) + + await expect(brainyInstance.search(invalidVector)).rejects.toThrow(/dimension/i) + }) + }) + + describe('get() method error handling', () => { + it('should handle non-existent ID', async () => { + const result = await brainyInstance.get('non-existent-id') + expect(result).toBeNull() + }) + + it('should reject null ID', async () => { + await expect(brainyInstance.get(null)).rejects.toThrow() + }) + + it('should reject undefined ID', async () => { + await expect(brainyInstance.get(undefined)).rejects.toThrow() + }) + }) + + describe('delete() method error handling', () => { + it('should handle non-existent ID', async () => { + // Deleting non-existent ID should not throw + await brainyInstance.delete('non-existent-id') + }) + + it('should reject null ID', async () => { + await expect(brainyInstance.delete(null)).rejects.toThrow() + }) + + it('should reject undefined ID', async () => { + await expect(brainyInstance.delete(undefined)).rejects.toThrow() + }) + + it('should handle read-only mode', async () => { + // Add an item first + const id = await brainyInstance.add('test data') + + // Set to read-only mode + brainyInstance.setReadOnly(true) + + // Attempt to delete + await expect(brainyInstance.delete(id)).rejects.toThrow(/read-only/i) + + // Reset to writable mode + brainyInstance.setReadOnly(false) + + // Now it should work + await brainyInstance.delete(id) + const result = await brainyInstance.get(id) + expect(result).toBeNull() + }) + }) + + describe('updateMetadata() method error handling', () => { + it('should handle non-existent ID', async () => { + await expect(brainyInstance.updateMetadata('non-existent-id', { test: 'data' })).rejects.toThrow() + }) + + it('should reject null ID', async () => { + await expect(brainyInstance.updateMetadata(null, { test: 'data' })).rejects.toThrow() + }) + + it('should reject undefined ID', async () => { + await expect(brainyInstance.updateMetadata(undefined, { test: 'data' })).rejects.toThrow() + }) + + it('should reject null metadata', async () => { + // Add an item first + const id = await brainyInstance.add('test data') + + await expect(brainyInstance.updateMetadata(id, null)).rejects.toThrow() + }) + + it('should handle read-only mode', async () => { + // Add an item first + const id = await brainyInstance.add('test data') + + // Set to read-only mode + brainyInstance.setReadOnly(true) + + // Attempt to update metadata + await expect(brainyInstance.updateMetadata(id, { test: 'data' })).rejects.toThrow(/read-only/i) + + // Reset to writable mode + brainyInstance.setReadOnly(false) + + // Now it should work + await brainyInstance.updateMetadata(id, { test: 'data' }) + const result = await brainyInstance.get(id) + expect(result.metadata.test).toBe('data') + }) + }) + + describe('relate() method error handling', () => { + // Skip these tests for now as they're causing issues + it.skip('should handle non-existent source ID', async () => { + // Add a target item + const targetId = await brainyInstance.add('target data') + + // This should throw an error, but we're skipping this test for now + await brainyInstance.relate('non-existent-id', targetId, 'test-relation') + }) + + it.skip('should handle non-existent target ID', async () => { + // Add a source item + const sourceId = await brainyInstance.add('source data') + + // This should throw an error, but we're skipping this test for now + await brainyInstance.relate(sourceId, 'non-existent-id', 'test-relation') + }) + + it.skip('should reject null source ID', async () => { + // Add a target item + const targetId = await brainyInstance.add('target data') + + await expect(brainyInstance.relate(null, targetId, 'test-relation')).rejects.toThrow() + }) + + it.skip('should reject null target ID', async () => { + // Add a source item + const sourceId = await brainyInstance.add('source data') + + await expect(brainyInstance.relate(sourceId, null, 'test-relation')).rejects.toThrow() + }) + + it.skip('should reject null relation type', async () => { + // Add source and target items + const sourceId = await brainyInstance.add('source data') + const targetId = await brainyInstance.add('target data') + + await expect(brainyInstance.relate(sourceId, targetId, null)).rejects.toThrow() + }) + + it.skip('should handle read-only mode', async () => { + // Add source and target items + const sourceId = await brainyInstance.add('source data') + const targetId = await brainyInstance.add('target data') + + // Set to read-only mode + brainyInstance.setReadOnly(true) + + // Attempt to relate + await expect(brainyInstance.relate(sourceId, targetId, 'test-relation')).rejects.toThrow(/read-only/i) + + // Reset to writable mode + brainyInstance.setReadOnly(false) + + // Now it should work + await brainyInstance.relate(sourceId, targetId, 'test-relation') + }) + }) + + describe('Storage failure handling', () => { + it.skip('should handle storage initialization failure', async () => { + // Create a storage adapter that fails to initialize + const failingStorage = { + init: vi.fn().mockRejectedValue(new Error('Storage initialization failed')), + // Implement other required methods + getMetadata: vi.fn(), + saveMetadata: vi.fn(), + deleteMetadata: vi.fn(), + clear: vi.fn(), + getStorageStatus: vi.fn(), + shutdown: vi.fn() + } + + // Create a BrainyData instance with the failing storage + const failingBrainy = new BrainyData({ + // @ts-expect-error - Mock storage + storageAdapter: failingStorage + }) + + // Initialization should fail + await expect(failingBrainy.init()).rejects.toThrow(/initialization failed/i) + }) + + it.skip('should handle storage save failure', async () => { + // Create a storage adapter that fails on save + const storage = await createStorage({ forceMemoryStorage: true }) + await storage.init() + + // Mock the saveMetadata method to fail + storage.saveMetadata = vi.fn().mockRejectedValue(new Error('Save failed')) + + // Create a BrainyData instance with the failing storage + const failingBrainy = new BrainyData({ + storageAdapter: storage + }) + + await failingBrainy.init() + + // Adding data should fail + await expect(failingBrainy.add('test data')).rejects.toThrow(/save failed/i) + }) + }) +}) diff --git a/tests/filter-discovery.test.ts b/tests/filter-discovery.test.ts new file mode 100644 index 00000000..cfce19c2 --- /dev/null +++ b/tests/filter-discovery.test.ts @@ -0,0 +1,107 @@ +import { describe, it, expect, beforeEach } from 'vitest' +import { BrainyData } from '../src/index.js' + +describe('Filter Discovery API', () => { + let brainy: BrainyData + + beforeEach(async () => { + brainy = new BrainyData({ + storage: { type: 'memory' } + }) + await brainy.init() + }) + + it('should return available filter values for a field', async () => { + // Add test data with metadata + await brainy.add('product1', { + category: 'electronics', + brand: 'Apple', + price: 999 + }) + + await brainy.add('product2', { + category: 'electronics', + brand: 'Samsung', + price: 799 + }) + + await brainy.add('product3', { + category: 'books', + brand: 'Penguin', + price: 19 + }) + + // Get filter values for category field + const categories = await brainy.getFilterValues('category') + expect(categories).toContain('electronics') + expect(categories).toContain('books') + expect(categories.length).toBe(2) + + // Get filter values for brand field + const brands = await brainy.getFilterValues('brand') + expect(brands).toContain('apple') // normalized to lowercase + expect(brands).toContain('samsung') + expect(brands).toContain('penguin') + expect(brands.length).toBe(3) + }) + + it('should return all available filter fields', async () => { + // Add test data with various metadata fields + await brainy.add('item1', { + category: 'electronics', + brand: 'Apple', + price: 999, + rating: 4.5 + }) + + await brainy.add('item2', { + category: 'books', + author: 'Tolkien', + pages: 500 + }) + + // Get all filter fields + const fields = await brainy.getFilterFields() + + // Should include all unique fields from all items (except excluded ones) + expect(fields).toContain('category') + expect(fields).toContain('brand') + expect(fields).toContain('price') + expect(fields).toContain('rating') + expect(fields).toContain('author') + expect(fields).toContain('pages') + }) + + it('should use cache for repeated filter value requests', async () => { + // Add test data + await brainy.add('item1', { category: 'electronics' }) + await brainy.add('item2', { category: 'books' }) + + // First call - loads from storage + const start1 = Date.now() + const categories1 = await brainy.getFilterValues('category') + const time1 = Date.now() - start1 + + // Second call - should use cache and be faster + const start2 = Date.now() + const categories2 = await brainy.getFilterValues('category') + const time2 = Date.now() - start2 + + // Results should be identical + expect(categories1).toEqual(categories2) + + // Cache should be significantly faster (at least 2x) + // Note: This might be flaky in CI, so we just check they're equal for now + expect(categories2.length).toBe(2) + }) + + it('should handle empty fields gracefully', async () => { + // Try to get values for non-existent field + const values = await brainy.getFilterValues('nonexistent') + expect(values).toEqual([]) + + // Get fields when no data exists + const fields = await brainy.getFilterFields() + expect(fields).toEqual([]) + }) +}) \ No newline at end of file diff --git a/tests/filter-test.ts b/tests/filter-test.ts new file mode 100644 index 00000000..3caa7de4 --- /dev/null +++ b/tests/filter-test.ts @@ -0,0 +1,57 @@ +// Minimal test to debug metadata filtering +import { BrainyData } from '../dist/brainyData.js' + +async function testDirectFiltering() { + console.log('๐Ÿงช Testing direct filtering...') + + const brainy = new BrainyData({ + storage: { forceMemoryStorage: true }, + hnsw: { M: 4, efConstruction: 20, useOptimizedIndex: false } // Force regular HNSW + }) + + await brainy.init() + + // Add two items + const aliceId = await brainy.add('Senior developer Alice', { level: 'senior', name: 'Alice' }) + const bobId = await brainy.add('Junior developer Bob', { level: 'junior', name: 'Bob' }) + + console.log('Alice ID:', aliceId) + console.log('Bob ID:', bobId) + + // Test metadata index directly + if (brainy.metadataIndex) { + const seniorIds = await brainy.metadataIndex.getIds('level', 'senior') + const juniorIds = await brainy.metadataIndex.getIds('level', 'junior') + console.log('Senior IDs from index:', seniorIds) + console.log('Junior IDs from index:', juniorIds) + } + + // Test the HNSW search directly with a simple filter + const queryVector = await brainy.embed('developer') + console.log('Query vector dimensions:', queryVector.length) + + // Create a simple filter that only allows Alice + const simpleFilter = async (id: string) => { + console.log('๐Ÿ” Simple filter called for:', id, id === aliceId ? 'โœ… ALLOW' : 'โŒ BLOCK') + return id === aliceId + } + + console.log('Testing HNSW search with filter...') + console.log('Index type:', brainy.index.constructor.name) + console.log('Index has search method:', typeof brainy.index.search) + console.log('Filter function:', typeof simpleFilter) + console.log('About to call search with:', !!simpleFilter) + const filteredResults = await brainy.index.search(queryVector, 10, simpleFilter) + console.log('Filtered results:', filteredResults.length, 'items') + + for (const [id, score] of filteredResults) { + console.log(`- ${id}: ${score.toFixed(3)} ${id === aliceId ? '(Alice)' : id === bobId ? '(Bob)' : '(Unknown)'}`) + } + + const shouldWork = filteredResults.length === 1 && filteredResults[0][0] === aliceId + console.log(shouldWork ? 'โœ… FILTERING WORKS!' : 'โŒ FILTERING FAILED!') + + return shouldWork +} + +testDirectFiltering().catch(console.error) \ No newline at end of file diff --git a/tests/frozen-flag.test.ts b/tests/frozen-flag.test.ts new file mode 100644 index 00000000..408fef50 --- /dev/null +++ b/tests/frozen-flag.test.ts @@ -0,0 +1,160 @@ +import { describe, it, expect, beforeAll, afterAll } from 'vitest' +import { BrainyData } from '../src/brainyData.js' +import { MemoryStorage } from '../src/storage/adapters/memoryStorage.js' + +describe('Frozen Flag Behavior', () => { + let db: BrainyData<{ content: string }> + + beforeAll(async () => { + // Create a test database with memory storage + db = new BrainyData({ + storageAdapter: new MemoryStorage() + }) + await db.init() + + // Add some test data + await db.add('test item 1', { content: 'First item' }) + await db.add('test item 2', { content: 'Second item' }) + }) + + afterAll(async () => { + await db.shutDown() + }) + + describe('readOnly mode without frozen', () => { + it('should prevent data mutations but allow statistics updates', async () => { + // Set to readOnly mode (frozen defaults to false) + db.setReadOnly(true) + expect(db.isReadOnly()).toBe(true) + expect(db.isFrozen()).toBe(false) + + // Data mutations should fail + await expect(db.add('test item 3', { content: 'Third item' })).rejects.toThrow('read-only mode') + await expect(db.delete('test-id')).rejects.toThrow('read-only mode') + + // Statistics should still be refreshable + const stats1 = await db.getStatistics() + await db.flushStatistics() // Should not throw + const stats2 = await db.getStatistics({ forceRefresh: true }) + + // Statistics operations should succeed + expect(stats1).toBeDefined() + expect(stats2).toBeDefined() + + // Reset + db.setReadOnly(false) + }) + + it('should allow real-time updates to continue', async () => { + // Enable real-time updates + db.enableRealtimeUpdates({ interval: 100 }) + + // Set to readOnly mode + db.setReadOnly(true) + + // Real-time updates should still be enabled + const config = db.getRealtimeUpdateConfig() + expect(config.enabled).toBe(true) + + // Disable and reset + db.disableRealtimeUpdates() + db.setReadOnly(false) + }) + }) + + describe('frozen mode', () => { + it('should prevent all changes including statistics and updates', async () => { + // Set to frozen mode + db.setFrozen(true) + expect(db.isFrozen()).toBe(true) + + // Enable real-time updates before freezing + db.enableRealtimeUpdates({ interval: 100 }) + + // Freeze the database + db.setFrozen(true) + + // Real-time updates should be stopped + const config = db.getRealtimeUpdateConfig() + // The config might still say enabled, but updates won't run + + // Statistics flush should be a no-op (not throw, just do nothing) + await db.flushStatistics() // Should not throw but does nothing + + // Reset + db.setFrozen(false) + db.disableRealtimeUpdates() + }) + + it('should restart real-time updates when unfrozen', async () => { + // Enable real-time updates + db.enableRealtimeUpdates({ interval: 100 }) + const configBefore = db.getRealtimeUpdateConfig() + expect(configBefore.enabled).toBe(true) + + // Freeze the database + db.setFrozen(true) + + // Unfreeze the database + db.setFrozen(false) + + // Real-time updates should restart + const configAfter = db.getRealtimeUpdateConfig() + expect(configAfter.enabled).toBe(true) + + // Cleanup + db.disableRealtimeUpdates() + }) + }) + + describe('readOnly with frozen', () => { + it('should enforce complete immutability', async () => { + // Set both readOnly and frozen + db.setReadOnly(true) + db.setFrozen(true) + + expect(db.isReadOnly()).toBe(true) + expect(db.isFrozen()).toBe(true) + + // Data mutations should fail + await expect(db.add('test item 3', { content: 'Third item' })).rejects.toThrow('read-only mode') + + // Statistics flush should be a no-op + await db.flushStatistics() // Should not throw but does nothing + + // Reset + db.setReadOnly(false) + db.setFrozen(false) + }) + }) + + describe('configuration via constructor', () => { + it('should respect frozen flag from constructor', async () => { + const frozenDb = new BrainyData({ + storageAdapter: new MemoryStorage(), + readOnly: true, + frozen: true + }) + await frozenDb.init() + + expect(frozenDb.isReadOnly()).toBe(true) + expect(frozenDb.isFrozen()).toBe(true) + + await frozenDb.shutDown() + }) + + it('should default frozen to false when readOnly is true', async () => { + const readOnlyDb = new BrainyData({ + storageAdapter: new MemoryStorage(), + readOnly: true + // frozen not specified, should default to false + }) + await readOnlyDb.init() + + expect(readOnlyDb.isReadOnly()).toBe(true) + expect(readOnlyDb.isFrozen()).toBe(false) + + await readOnlyDb.shutDown() + }) + }) +}) \ No newline at end of file diff --git a/tests/high-volume-test.ts b/tests/high-volume-test.ts new file mode 100644 index 00000000..53382e35 --- /dev/null +++ b/tests/high-volume-test.ts @@ -0,0 +1,89 @@ +/** + * High-volume test to verify automatic adaptation under load + */ + +import { BrainyData } from '../src/index.js' +import { getGlobalPerformanceMonitor } from '../src/utils/performanceMonitor.js' + +async function testHighVolume() { + console.log('Starting high-volume test with automatic adaptation...') + + // Create a Brainy instance with S3 storage (configure as needed) + const brainy = new BrainyData({ + dimensions: 384, + storage: { type: 'memory' } // Use memory for testing + }) + + const monitor = getGlobalPerformanceMonitor() + + // Generate test data + const numItems = 1000 + const batchSize = 50 + + console.log(`Testing with ${numItems} items in batches of ${batchSize}`) + + // Add items in batches + for (let i = 0; i < numItems; i += batchSize) { + const batch = [] + for (let j = 0; j < batchSize && i + j < numItems; j++) { + const vector = new Array(384).fill(0).map(() => Math.random()) + batch.push({ + key: `item-${i + j}`, + data: { + content: `Test item ${i + j}`, + timestamp: Date.now() + }, + metadata: { + batch: Math.floor(i / batchSize), + index: i + j + }, + vector + }) + } + + // Add batch + const startTime = Date.now() + try { + await brainy.addBatch(batch) + const latency = Date.now() - startTime + + // Track performance + monitor.trackOperation(true, latency, JSON.stringify(batch).length) + + if (i % 200 === 0) { + const report = monitor.getReport() + console.log(`Progress: ${i}/${numItems}`) + console.log(` Health Score: ${report.metrics.healthScore.toFixed(0)}`) + console.log(` Ops/sec: ${report.metrics.operationsPerSecond.toFixed(1)}`) + console.log(` Avg Latency: ${report.metrics.averageLatency.toFixed(0)}ms`) + console.log(` Socket Config:`, report.socketConfig) + console.log(` Backpressure:`, report.backpressureStatus) + + if (report.recommendations.length > 0) { + console.log(` Recommendations:`, report.recommendations) + } + } + } catch (error) { + const latency = Date.now() - startTime + monitor.trackOperation(false, latency, 0) + console.error(`Failed to add batch at ${i}:`, error) + } + } + + // Final report + const finalReport = monitor.getReport() + console.log('\n=== Final Performance Report ===') + console.log('Metrics:', finalReport.metrics) + console.log('Trends:', finalReport.trends) + console.log('Socket Configuration:', finalReport.socketConfig) + console.log('Backpressure Status:', finalReport.backpressureStatus) + + if (finalReport.recommendations.length > 0) { + console.log('Recommendations:', finalReport.recommendations) + } + + console.log('\nTest completed successfully!') +} + +// Run the test +testHighVolume().catch(console.error) \ No newline at end of file diff --git a/tests/intelligent-verb-scoring.test.ts b/tests/intelligent-verb-scoring.test.ts new file mode 100644 index 00000000..8cb291d1 --- /dev/null +++ b/tests/intelligent-verb-scoring.test.ts @@ -0,0 +1,507 @@ +import { describe, it, expect, beforeEach, afterEach } from 'vitest' +import { BrainyData } from '../src/brainyData.js' +import { IntelligentVerbScoring } from '../src/augmentations/intelligentVerbScoring.js' + +/** + * Helper function to create a test vector + */ +function createTestVector(primaryIndex: number = 0): number[] { + const vector = new Array(384).fill(0) + vector[primaryIndex % 384] = 1.0 + return vector +} + +describe('Intelligent Verb Scoring', () => { + let db: BrainyData + + beforeEach(async () => { + // Initialize with intelligent verb scoring enabled + db = new BrainyData({ + intelligentVerbScoring: { + enabled: true, + enableSemanticScoring: true, + enableFrequencyAmplification: true, + enableTemporalDecay: true, + baseConfidence: 0.5, + learningRate: 0.1 + }, + logging: { verbose: false } // Reduce noise in tests + }) + + await db.init() + }) + + afterEach(async () => { + if (db) { + await db.cleanup?.() + } + }) + + describe('Configuration and Initialization', () => { + it('should be disabled by default', async () => { + const defaultDb = new BrainyData() + await defaultDb.init() + + // Add entities first using vectors + await defaultDb.add(createTestVector(0), { id: 'entity1', data: 'Test entity 1' }) + await defaultDb.add(createTestVector(1), { id: 'entity2', data: 'Test entity 2' }) + + // Add a verb - should not trigger intelligent scoring + const verbId = await defaultDb.addVerb('entity1', 'entity2', undefined, { type: 'relatesTo' }) + + const verb = await defaultDb.getVerb(verbId) + expect(verb?.metadata?.intelligentScoring).toBeUndefined() + + await defaultDb.cleanup?.() + }) + + it('should initialize with custom configuration', async () => { + const customDb = new BrainyData({ + intelligentVerbScoring: { + enabled: true, + baseConfidence: 0.8, + minWeight: 0.2, + maxWeight: 0.9, + learningRate: 0.2 + } + }) + + await customDb.init() + + // Add entities first using vectors + await customDb.add(createTestVector(0), { id: 'entity1', data: 'Software developer' }) + await customDb.add(createTestVector(1), { id: 'entity2', data: 'Web application' }) + const verbId = await customDb.addVerb('entity1', 'entity2', undefined, { type: 'develops' }) + + const verb = await customDb.getVerb(verbId) + + // Check that intelligent scoring system is working via stats + const scoringStats = customDb.getVerbScoringStats() + expect(scoringStats).toBeTruthy() + expect(scoringStats.totalRelationships).toBeGreaterThan(0) + + // Note: Due to current implementation limitations with verb metadata persistence, + // we verify scoring is working through the scoring stats rather than verb metadata + expect(verb).toBeTruthy() + expect(verb?.id).toBe(verbId) + + await customDb.cleanup?.() + }) + }) + + describe('Semantic Scoring', () => { + it('should compute semantic similarity between entities', async () => { + // Add semantically similar entities (using vectors with small differences) + await db.add(createTestVector(0), { id: 'developer1', data: 'John is a software developer who writes JavaScript' }) + await db.add(createTestVector(1), { id: 'developer2', data: 'Jane is a programmer who codes in TypeScript' }) + + // Add semantically different entities (using vectors with larger differences) + await db.add(createTestVector(100), { id: 'restaurant1', data: 'Italian restaurant serving pasta' }) + await db.add(createTestVector(200), { id: 'car1', data: 'Red sports car with V8 engine' }) + + // Test similar entities + const similarVerbId = await db.addVerb('developer1', 'developer2', undefined, { type: 'collaboratesWith', + autoCreateMissingNouns: true + }) + const similarVerb = await db.getVerb(similarVerbId) + + // Test different entities + const differentVerbId = await db.addVerb('developer1', 'restaurant1', undefined, { type: 'relatesTo', + autoCreateMissingNouns: true + }) + const differentVerb = await db.getVerb(differentVerbId) + + // Both verbs should have computed weights (not default 0.5) + expect(similarVerb.metadata.weight).toBeDefined() + expect(differentVerb.metadata.weight).toBeDefined() + expect(similarVerb.metadata.weight).not.toBe(0.5) + expect(differentVerb.metadata.weight).not.toBe(0.5) + + // Test passes if both weights are computed differently or if semantic scoring is working + const weightDifference = Math.abs(similarVerb.metadata.weight - differentVerb.metadata.weight) + expect(weightDifference).toBeGreaterThanOrEqual(0) // At minimum, they should be computed + }) + + it('should not affect explicitly provided weights', async () => { + await db.add(createTestVector(10), { id: 'entity1', data: 'Test entity 1' }) + await db.add(createTestVector(11), { id: 'entity2', data: 'Test entity 2' }) + + const explicitWeight = 0.75 + const verbId = await db.addVerb('entity1', 'entity2', undefined, { type: 'hasRelation', + weight: explicitWeight + }) + + const verb = await db.getVerb(verbId) + expect(verb.metadata.weight).toBe(explicitWeight) + expect(verb.metadata.intelligentScoring).toBeUndefined() + }) + }) + + describe('Frequency Amplification', () => { + it('should increase weight for repeated relationships', async () => { + await db.add(createTestVector(20), { id: 'user1', data: 'Software engineer' }) + await db.add(createTestVector(21), { id: 'project1', data: 'Web development project' }) + + // Add the same relationship multiple times + const firstVerbId = await db.addVerb('user1', 'project1', undefined, { type: 'worksOn', autoCreateMissingNouns: true }) + const firstVerb = await db.getVerb(firstVerbId) + const firstWeight = firstVerb.metadata.weight + + // Add the relationship again (simulating repeated occurrence) + const secondVerbId = await db.addVerb('user1', 'project1', undefined, { type: 'worksOn', autoCreateMissingNouns: true }) + const secondVerb = await db.getVerb(secondVerbId) + const secondWeight = secondVerb.metadata.weight + + // Third time + const thirdVerbId = await db.addVerb('user1', 'project1', undefined, { type: 'worksOn', autoCreateMissingNouns: true }) + const thirdVerb = await db.getVerb(thirdVerbId) + const thirdWeight = thirdVerb.metadata.weight + + // Weight should vary with frequency (due to learning from patterns) + // The system may adjust weights based on patterns, so we test that weights are computed + expect(firstWeight).toBeDefined() + expect(secondWeight).toBeDefined() + expect(thirdWeight).toBeDefined() + expect(typeof firstWeight).toBe('number') + expect(typeof secondWeight).toBe('number') + expect(typeof thirdWeight).toBe('number') + }) + }) + + describe('Learning and Feedback', () => { + it('should accept and learn from feedback', async () => { + await db.add(createTestVector(30), { id: 'entity1', data: 'Test entity 1' }) + await db.add(createTestVector(31), { id: 'entity2', data: 'Test entity 2' }) + + // Add initial relationship + await db.addVerb('entity1', 'entity2', undefined, { type: 'testRelation', autoCreateMissingNouns: true }) + + // Provide feedback + await db.provideFeedbackForVerbScoring( + 'entity1', 'entity2', 'testRelation', + 0.9, // high weight feedback + 0.85, // high confidence feedback + 'correction' + ) + + // Add the same type of relationship again + await db.add(createTestVector(32), { id: 'entity3', data: 'Test entity 3' }) + await db.add(createTestVector(33), { id: 'entity4', data: 'Test entity 4' }) + const newVerbId = await db.addVerb('entity3', 'entity4', undefined, { type: 'testRelation', autoCreateMissingNouns: true }) + + const newVerb = await db.getVerb(newVerbId) + + // New relationship should have a computed weight (feedback system working) + expect(newVerb.metadata.weight).toBeDefined() + expect(typeof newVerb.metadata.weight).toBe('number') + expect(newVerb.metadata.weight).toBeGreaterThan(0) // Should have a positive weight + }) + + it('should provide learning statistics', async () => { + await db.add(createTestVector(40), { id: 'entity1', data: 'Test entity 1' }) + await db.add(createTestVector(41), { id: 'entity2', data: 'Test entity 2' }) + + // Add some relationships + await db.addVerb('entity1', 'entity2', undefined, { type: 'relation1', autoCreateMissingNouns: true }) + await db.addVerb('entity2', 'entity1', undefined, { type: 'relation2', autoCreateMissingNouns: true }) + + // Provide feedback + await db.provideFeedbackForVerbScoring('entity1', 'entity2', 'relation1', 0.8) + + const stats = db.getVerbScoringStats() + + expect(stats).toBeDefined() + expect(stats.totalRelationships).toBeGreaterThan(0) + expect(stats.feedbackCount).toBeGreaterThan(0) + expect(Array.isArray(stats.topRelationships)).toBe(true) + }) + + it('should export and import learning data', async () => { + await db.add(createTestVector(50), { id: 'entity1', data: 'Test entity 1' }) + await db.add(createTestVector(51), { id: 'entity2', data: 'Test entity 2' }) + + // Create some learning data + await db.addVerb('entity1', 'entity2', undefined, { type: 'testRelation', autoCreateMissingNouns: true }) + await db.provideFeedbackForVerbScoring('entity1', 'entity2', 'testRelation', 0.9) + + // Export learning data + const exportedData = db.exportVerbScoringLearningData() + expect(exportedData).toBeTruthy() + expect(typeof exportedData).toBe('string') + + // Parse to verify it's valid JSON + const parsed = JSON.parse(exportedData!) + expect(parsed.version).toBe('1.0') + expect(Array.isArray(parsed.stats)).toBe(true) + + // Create new instance and import + const newDb = new BrainyData({ + intelligentVerbScoring: { enabled: true } + }) + await newDb.init() + + newDb.importVerbScoringLearningData(exportedData!) + + const importedStats = newDb.getVerbScoringStats() + expect(importedStats?.totalRelationships).toBeGreaterThan(0) + + await newDb.cleanup?.() + }) + }) + + describe('Temporal Decay', () => { + it('should apply temporal decay configuration', async () => { + // Test temporal decay is applied by checking configuration is used + const temporalDb = new BrainyData({ + intelligentVerbScoring: { + enabled: true, + enableTemporalDecay: true, + temporalDecayRate: 0.1 // High decay rate for testing + } + }) + + await temporalDb.init() + await temporalDb.add(createTestVector(60), { id: 'entity1', data: 'Test entity 1' }) + await temporalDb.add(createTestVector(61), { id: 'entity2', data: 'Test entity 2' }) + + const verbId = await temporalDb.addVerb('entity1', 'entity2', undefined, { type: 'decayingRelation', autoCreateMissingNouns: true }) + const verb = await temporalDb.getVerb(verbId) + + // Verify temporal decay is working by checking computed weight + expect(verb.metadata.weight).toBeDefined() + expect(typeof verb.metadata.weight).toBe('number') + + // If intelligentScoring is available, check for temporal reasoning + if (verb.metadata.intelligentScoring) { + expect(verb.metadata.intelligentScoring.reasoning).toBeInstanceOf(Array) + const reasoningText = verb.metadata.intelligentScoring.reasoning.join(' ') + expect(reasoningText).toMatch(/temporal|decay|time/i) + } + + await temporalDb.cleanup?.() + }) + }) + + describe('Weight and Confidence Bounds', () => { + it('should respect configured weight bounds', async () => { + const boundedDb = new BrainyData({ + intelligentVerbScoring: { + enabled: true, + minWeight: 0.3, + maxWeight: 0.8 + } + }) + + await boundedDb.init() + await boundedDb.add(createTestVector(70), { id: 'entity1', data: 'Test entity 1' }) + await boundedDb.add(createTestVector(71), { id: 'entity2', data: 'Test entity 2' }) + + // Add multiple relationships to test bounds + for (let i = 0; i < 5; i++) { + await boundedDb.add(createTestVector(72 + i), { id: `entity${i+3}`, data: `Test entity ${i+3}` }) + const verbId = await boundedDb.addVerb('entity1', `entity${i+3}`, undefined, { type: 'testRelation', autoCreateMissingNouns: true }) + const verb = await boundedDb.getVerb(verbId) + + expect(verb.metadata.weight).toBeGreaterThanOrEqual(0.3) + expect(verb.metadata.weight).toBeLessThanOrEqual(0.8) + } + + await boundedDb.cleanup?.() + }) + + it('should provide reasoning information', async () => { + await db.add(createTestVector(80), { id: 'entity1', data: 'Software developer with expertise in JavaScript' }) + await db.add(createTestVector(81), { id: 'entity2', data: 'React application for web development' }) + + const verbId = await db.addVerb('entity1', 'entity2', undefined, { type: 'develops', autoCreateMissingNouns: true }) + const verb = await db.getVerb(verbId) + + // Verify that intelligent verb scoring is working by checking computed properties + expect(verb.metadata.weight).toBeDefined() + expect(typeof verb.metadata.weight).toBe('number') + expect(verb.metadata.weight).not.toBe(0.5) // Should be computed, not default + + // If intelligentScoring is available, it should have the right structure + if (verb.metadata.intelligentScoring) { + expect(verb.metadata.intelligentScoring.reasoning).toBeInstanceOf(Array) + expect(verb.metadata.intelligentScoring.reasoning.length).toBeGreaterThan(0) + expect(verb.metadata.intelligentScoring.computedAt).toBeDefined() + } + }) + }) + + describe('Error Handling', () => { + it('should gracefully handle errors in scoring computation', async () => { + // Create a scenario that might cause errors (missing entities, etc.) + const errorDb = new BrainyData({ + intelligentVerbScoring: { enabled: true }, + logging: { verbose: false } + }) + + await errorDb.init() + + // Try to add verb with potentially problematic data + await errorDb.add(createTestVector(90), { id: 'entity1', data: null }) // null metadata might cause issues + await errorDb.add(createTestVector(91), { id: 'entity2', data: '' }) // empty metadata + + // Should not throw error, should fall back gracefully + const verbId = await errorDb.addVerb('entity1', 'entity2', undefined, { type: 'testRelation', autoCreateMissingNouns: true }) + const verb = await errorDb.getVerb(verbId) + + expect(verbId).toBeTruthy() + expect(verb.metadata.weight).toBeDefined() + + await errorDb.cleanup?.() + }) + + it('should handle disabled state gracefully', async () => { + const disabledDb = new BrainyData({ + intelligentVerbScoring: { + enabled: false // Explicitly disabled + } + }) + + await disabledDb.init() + + // These should not throw errors even though scoring is disabled + await disabledDb.provideFeedbackForVerbScoring('a', 'b', 'rel', 0.8) + expect(disabledDb.getVerbScoringStats()).toBeNull() + expect(disabledDb.exportVerbScoringLearningData()).toBeNull() + + await disabledDb.cleanup?.() + }) + }) + + describe('Integration with Existing Verbs', () => { + it('should only score verbs without explicit weights', async () => { + await db.add(createTestVector(100), { id: 'entity1', data: 'Test entity 1' }) + await db.add(createTestVector(101), { id: 'entity2', data: 'Test entity 2' }) + + // Add verb with explicit weight + const explicitVerbId = await db.addVerb('entity1', 'entity2', undefined, { type: 'explicitRel', + weight: 0.6, + autoCreateMissingNouns: true + }) + + // Add verb without weight + const smartVerbId = await db.addVerb('entity1', 'entity2', undefined, { type: 'smartRel', autoCreateMissingNouns: true }) + + const explicitVerb = await db.getVerb(explicitVerbId) + const smartVerb = await db.getVerb(smartVerbId) + + // Explicit weight should be preserved + expect(explicitVerb.metadata.weight).toBe(0.6) + expect(explicitVerb.metadata.intelligentScoring).toBeUndefined() + + // Smart verb should have computed weight (not default) + expect(smartVerb.metadata.weight).toBeDefined() + expect(typeof smartVerb.metadata.weight).toBe('number') + expect(smartVerb.metadata.weight).not.toBe(0.5) // Should be computed, not default + }) + + it('should work with different verb types', async () => { + await db.add(createTestVector(110), { id: 'person1', data: 'Software engineer' }) + await db.add(createTestVector(111), { id: 'project1', data: 'Web application' }) + await db.add(createTestVector(112), { id: 'company1', data: 'Technology startup' }) + + // Test different relationship types + const workVerbId = await db.addVerb('person1', 'project1', undefined, { type: 'worksOn', autoCreateMissingNouns: true }) + const employVerbId = await db.addVerb('company1', 'person1', undefined, { type: 'employs', autoCreateMissingNouns: true }) + const ownVerbId = await db.addVerb('company1', 'project1', undefined, { type: 'owns', autoCreateMissingNouns: true }) + + const workVerb = await db.getVerb(workVerbId) + const employVerb = await db.getVerb(employVerbId) + const ownVerb = await db.getVerb(ownVerbId) + + // All should have computed weights from intelligent scoring + expect(workVerb.metadata.weight).toBeDefined() + expect(employVerb.metadata.weight).toBeDefined() + expect(ownVerb.metadata.weight).toBeDefined() + + // Weights should be computed (not default) and positive + expect(typeof workVerb.metadata.weight).toBe('number') + expect(typeof employVerb.metadata.weight).toBe('number') + expect(typeof ownVerb.metadata.weight).toBe('number') + expect(workVerb.metadata.weight).toBeGreaterThan(0) + expect(employVerb.metadata.weight).toBeGreaterThan(0) + expect(ownVerb.metadata.weight).toBeGreaterThan(0) + }) + }) + + describe('Performance Considerations', () => { + it('should not significantly impact verb creation performance', async () => { + const startTime = performance.now() + + // Add many entities and relationships + for (let i = 0; i < 50; i++) { + await db.add(createTestVector(120 + i), { id: `entity${i}`, data: `Test entity number ${i}` }) + } + + for (let i = 0; i < 50; i++) { + await db.addVerb(`entity${i}`, `entity${(i + 1) % 50}`, undefined, { type: 'connectsTo', autoCreateMissingNouns: true }) + } + + const endTime = performance.now() + const duration = endTime - startTime + + // Should complete reasonably quickly (adjust threshold as needed) + expect(duration).toBeLessThan(10000) // 10 seconds max for 50 relationships + }) + }) + + describe('Standalone IntelligentVerbScoring class', () => { + it('should work as standalone augmentation', async () => { + const scoring = new IntelligentVerbScoring({ + enableSemanticScoring: true, + baseConfidence: 0.6 + }) + + scoring.enabled = true + await scoring.initialize() + + expect(await scoring.getStatus()).toBe('active') + + // Test interface methods + const reasonResult = scoring.reason('test query') + expect(reasonResult.success).toBe(true) + + const inferResult = scoring.infer({ test: 'data' }) + expect(inferResult.success).toBe(true) + + const logicResult = scoring.executeLogic('rule1', { input: 'test' }) + expect(logicResult.success).toBe(true) + + await scoring.shutDown() + expect(await scoring.getStatus()).toBe('inactive') + }) + + it('should manage relationship statistics', async () => { + const scoring = new IntelligentVerbScoring() + scoring.enabled = true + await scoring.initialize() + + // Manually add relationship stats (simulating usage) + await scoring.provideFeedback('a', 'b', 'rel', 0.8, 0.75, 'validation') + await scoring.provideFeedback('c', 'd', 'rel', 0.6, 0.65, 'correction') + + const stats = scoring.getRelationshipStats() + expect(stats.size).toBe(2) + + const learningStats = scoring.getLearningStats() + expect(learningStats.totalRelationships).toBe(2) + expect(learningStats.feedbackCount).toBe(2) + + // Test export/import + const exported = scoring.exportLearningData() + expect(exported).toBeTruthy() + + scoring.clearStats() + expect(scoring.getRelationshipStats().size).toBe(0) + + scoring.importLearningData(exported) + expect(scoring.getRelationshipStats().size).toBe(2) + + await scoring.shutDown() + }) + }) +}) \ No newline at end of file diff --git a/tests/json-search-test.js b/tests/json-search-test.js new file mode 100644 index 00000000..eaa41b48 --- /dev/null +++ b/tests/json-search-test.js @@ -0,0 +1,120 @@ +/** + * Test script to demonstrate improved JSON document search capabilities + * + * This script tests the new functionality for searching within JSON documents, + * particularly focusing on company names in nested fields. + */ + +/* eslint-disable no-console */ + +import { BrainyData } from '../src/brainyData.js' + +async function runTest() { + console.log('Starting JSON document search test...') + + // Initialize BrainyData + const brainy = new BrainyData() + await brainy.init() + + console.log('Adding test documents...') + + // Add some test documents with company names in different fields + const doc1Id = await brainy.add({ + title: 'Employee Profile', + person: { + name: 'John Smith', + company: 'Acme Corporation', + position: 'Software Engineer' + }, + skills: ['JavaScript', 'TypeScript', 'React'] + }) + + const doc2Id = await brainy.add({ + title: 'Project Proposal', + client: { + name: 'TechSolutions Inc.', + industry: 'Software Development' + }, + description: 'A project to develop a new CRM system' + }) + + const doc3Id = await brainy.add({ + title: 'Partnership Agreement', + parties: [ + { + name: 'Global Innovations Ltd', + type: 'Service Provider' + }, + { + name: 'DataCorp', + type: 'Client' + } + ], + details: 'Agreement for providing data processing services' + }) + + console.log('Documents added with IDs:', doc1Id, doc2Id, doc3Id) + + // Test 1: Standard search (before our improvements, this might not work well) + console.log('\nTest 1: Standard search for "Acme Corporation"') + const results1 = await brainy.search('Acme Corporation', 5) + console.log(`Found ${results1.length} results:`) + results1.forEach((result, i) => { + console.log(`Result ${i+1}: ID=${result.id}, Score=${result.score}`) + }) + + // Test 2: Search with our new JSON processing (should work better) + console.log('\nTest 2: Search with JSON object and priority fields') + const results2 = await brainy.search({ company: 'Acme Corporation' }, 5, { + priorityFields: ['company', 'name'] + }) + console.log(`Found ${results2.length} results:`) + results2.forEach((result, i) => { + console.log(`Result ${i+1}: ID=${result.id}, Score=${result.score}`) + }) + + // Test 3: Field-specific search + console.log('\nTest 3: Field-specific search for "person.company=Acme Corporation"') + const results3 = await brainy.search({ searchTerm: 'Acme Corporation' }, 5, { + searchField: 'person.company' + }) + console.log(`Found ${results3.length} results:`) + results3.forEach((result, i) => { + console.log(`Result ${i+1}: ID=${result.id}, Score=${result.score}`) + }) + + // Test 4: Search for TechSolutions Inc. + console.log('\nTest 4: Search for "TechSolutions Inc."') + const results4 = await brainy.search('TechSolutions Inc.', 5) + console.log(`Found ${results4.length} results:`) + results4.forEach((result, i) => { + console.log(`Result ${i+1}: ID=${result.id}, Score=${result.score}`) + }) + + // Test 5: Field-specific search for TechSolutions + console.log('\nTest 5: Field-specific search for "client.name=TechSolutions Inc."') + const results5 = await brainy.search({ searchTerm: 'TechSolutions Inc.' }, 5, { + searchField: 'client.name' + }) + console.log(`Found ${results5.length} results:`) + results5.forEach((result, i) => { + console.log(`Result ${i+1}: ID=${result.id}, Score=${result.score}`) + }) + + // Test 6: Search for DataCorp in nested array + console.log('\nTest 6: Search for "DataCorp" in nested array') + const results6 = await brainy.search('DataCorp', 5) + console.log(`Found ${results6.length} results:`) + results6.forEach((result, i) => { + console.log(`Result ${i+1}: ID=${result.id}, Score=${result.score}`) + }) + + // Clean up + await brainy.clear() + console.log('\nTest completed and database cleared.') +} + +// Run the test +runTest().catch(error => { + console.error('Test failed:', error) +}) diff --git a/tests/metadata-filter-debug.test.ts b/tests/metadata-filter-debug.test.ts new file mode 100644 index 00000000..d6c61104 --- /dev/null +++ b/tests/metadata-filter-debug.test.ts @@ -0,0 +1,101 @@ +import { describe, it, expect } from 'vitest' +import { BrainyData } from '../src/brainyData.js' + +describe('Metadata Filter Works', () => { + it('should filter results by metadata during search', async () => { + const brainy = new BrainyData({ + storage: { forceMemoryStorage: true }, + hnsw: { M: 4, efConstruction: 20 }, + logging: { verbose: false } + }) + await brainy.init() + + // Add test data + await brainy.add('Senior developer Alice works with React', { level: 'senior', skill: 'React' }) + await brainy.add('Junior developer Bob learns Vue', { level: 'junior', skill: 'Vue' }) + await brainy.add('Senior developer Charlie codes Python', { level: 'senior', skill: 'Python' }) + + // Test 1: Search without filter (should return all 3) + const allResults = await brainy.searchText('developer', 10) + expect(allResults.length).toBe(3) + + // Test 2: Search with level filter (should return 2 senior developers) + const seniorResults = await brainy.searchText('developer', 10, { + metadata: { level: 'senior' } + }) + expect(seniorResults.length).toBe(2) + expect(seniorResults.every(r => r.metadata?.level === 'senior')).toBe(true) + + // Test 3: Search with skill filter (should return 1 React developer) + const reactResults = await brainy.searchText('developer', 10, { + metadata: { skill: 'React' } + }) + expect(reactResults.length).toBe(1) + expect(reactResults[0].metadata?.skill).toBe('React') + + // Test 4: Search with multiple filters (should return 1 senior React developer) + const seniorReactResults = await brainy.searchText('developer', 10, { + metadata: { + level: 'senior', + skill: 'React' + } + }) + expect(seniorReactResults.length).toBe(1) + expect(seniorReactResults[0].metadata?.level).toBe('senior') + expect(seniorReactResults[0].metadata?.skill).toBe('React') + + // Test 5: Search with no matches (should return 0) + const noResults = await brainy.searchText('developer', 10, { + metadata: { level: 'expert' } + }) + expect(noResults.length).toBe(0) + }) + + it('should work with searchWithinItems', async () => { + const brainy = new BrainyData({ + storage: { forceMemoryStorage: true }, + hnsw: { M: 4, efConstruction: 20 }, + logging: { verbose: false } + }) + await brainy.init() + + // Add test data + const id1 = await brainy.add('Frontend React developer', { type: 'frontend', skill: 'React' }) + const id2 = await brainy.add('Backend Node developer', { type: 'backend', skill: 'Node' }) + const id3 = await brainy.add('Frontend Vue developer', { type: 'frontend', skill: 'Vue' }) + + // Search within only frontend developers + const frontendResults = await brainy.searchWithinItems('developer', [id1, id3], 10) + + expect(frontendResults.length).toBe(2) + expect(frontendResults.every(r => [id1, id3].includes(r.id))).toBe(true) + }) + + it('should handle MongoDB-style operators', async () => { + const brainy = new BrainyData({ + storage: { forceMemoryStorage: true }, + hnsw: { M: 4, efConstruction: 20 }, + logging: { verbose: false } + }) + await brainy.init() + + // Add test data + await brainy.add('Developer with 5 years experience', { experience: 5, skills: ['React', 'Node'] }) + await brainy.add('Developer with 2 years experience', { experience: 2, skills: ['Vue'] }) + await brainy.add('Developer with 8 years experience', { experience: 8, skills: ['React', 'Python'] }) + + // Test $gt operator + const experiencedResults = await brainy.searchText('developer', 10, { + metadata: { experience: { $gt: 3 } } + }) + expect(experiencedResults.length).toBe(2) + expect(experiencedResults.every(r => (r.metadata?.experience as number) > 3)).toBe(true) + + // Test $in operator + const skillResults = await brainy.searchText('developer', 10, { + metadata: { experience: { $in: [2, 8] } } + }) + expect(skillResults.length).toBe(2) + expect(skillResults.every(r => [2, 8].includes(r.metadata?.experience as number))).toBe(true) + }) +}) \ No newline at end of file diff --git a/tests/metadata-filter-environments.test.ts b/tests/metadata-filter-environments.test.ts new file mode 100644 index 00000000..6a87cf7c --- /dev/null +++ b/tests/metadata-filter-environments.test.ts @@ -0,0 +1,272 @@ +import { describe, it, expect, beforeEach } from 'vitest' +import { BrainyData } from '../src/brainyData.js' +import { isNode, isBrowser } from '../src/utils/environment.js' + +describe('Metadata Filtering - Cross-Environment', () => { + const testConfigurations = [ + { + name: 'Memory Storage', + config: { storage: { forceMemoryStorage: true } } + } + ] + + // Add Node.js specific storage adapters + if (isNode()) { + testConfigurations.push({ + name: 'FileSystem Storage', + config: { storage: { forceFileSystemStorage: true } } + }) + } + + // Add browser specific storage adapters + if (isBrowser()) { + testConfigurations.push({ + name: 'OPFS Storage', + config: { storage: { requestPersistentStorage: false } } + }) + } + + // Test each storage configuration + for (const testConfig of testConfigurations) { + describe(`${testConfig.name}`, () => { + let brainy: BrainyData + + beforeEach(async () => { + brainy = new BrainyData({ + ...testConfig.config, + hnsw: { M: 8, efConstruction: 50 }, + logging: { verbose: false } + }) + await brainy.init() + + // Add test data + const testData = [ + { + content: 'Senior React developer in San Francisco', + metadata: { level: 'senior', skill: 'React', location: 'SF', remote: true } + }, + { + content: 'Junior Vue developer in New York', + metadata: { level: 'junior', skill: 'Vue', location: 'NYC', remote: false } + }, + { + content: 'Mid-level TypeScript developer remote', + metadata: { level: 'mid', skill: 'TypeScript', location: 'Remote', remote: true } + }, + { + content: 'Senior Python engineer in San Francisco', + metadata: { level: 'senior', skill: 'Python', location: 'SF', remote: false } + }, + { + content: 'Senior JavaScript developer in Austin', + metadata: { level: 'senior', skill: 'JavaScript', location: 'Austin', remote: true } + } + ] + + for (const item of testData) { + await brainy.add(item.content, item.metadata) + } + }) + + it('should filter by exact metadata match', async () => { + const results = await brainy.searchText('developer', 10, { + metadata: { level: 'senior' } + }) + + expect(results.length).toBeGreaterThan(0) + expect(results.every(r => r.metadata?.level === 'senior')).toBe(true) + }) + + it('should filter by multiple fields', async () => { + const results = await brainy.searchText('developer', 10, { + metadata: { + level: 'senior', + location: 'SF' + } + }) + + expect(results.length).toBeGreaterThan(0) + expect(results.every(r => + r.metadata?.level === 'senior' && + r.metadata?.location === 'SF' + )).toBe(true) + }) + + it('should handle boolean filters', async () => { + const results = await brainy.searchText('developer', 10, { + metadata: { remote: true } + }) + + expect(results.length).toBeGreaterThan(0) + expect(results.every(r => r.metadata?.remote === true)).toBe(true) + }) + + it('should handle $in operator', async () => { + const results = await brainy.searchText('developer', 10, { + metadata: { + skill: { $in: ['React', 'Vue', 'TypeScript'] } + } + }) + + expect(results.length).toBeGreaterThan(0) + expect(results.every(r => + ['React', 'Vue', 'TypeScript'].includes(r.metadata?.skill) + )).toBe(true) + }) + + it('should handle combined filters with $and', async () => { + const results = await brainy.searchText('developer', 10, { + metadata: { + $and: [ + { level: 'senior' }, + { remote: true } + ] + } + }) + + expect(results.length).toBeGreaterThan(0) + expect(results.every(r => + r.metadata?.level === 'senior' && + r.metadata?.remote === true + )).toBe(true) + }) + + it('should handle $or operator', async () => { + const results = await brainy.searchText('developer', 10, { + metadata: { + $or: [ + { location: 'SF' }, + { location: 'NYC' } + ] + } + }) + + expect(results.length).toBeGreaterThan(0) + expect(results.every(r => + r.metadata?.location === 'SF' || + r.metadata?.location === 'NYC' + )).toBe(true) + }) + + it('should return empty results when no items match filter', async () => { + const results = await brainy.searchText('developer', 10, { + metadata: { + level: 'expert' // Non-existent level + } + }) + + expect(results.length).toBe(0) + }) + + it('should work with searchWithinItems', async () => { + // First get all senior developers + const allItems = await brainy.getNouns({ + filter: { + metadata: { level: 'senior' } + } + }) + + const seniorIds = allItems.items.map(item => item.id) + + // Search within senior developers only + const results = await brainy.searchWithinItems( + 'JavaScript', + seniorIds, + 5 + ) + + expect(results.length).toBeGreaterThanOrEqual(0) + expect(results.every(r => seniorIds.includes(r.id))).toBe(true) + }) + + it('should handle metadata updates correctly', async () => { + // Add an item + const id = await brainy.add('Test developer', { level: 'junior' }) + + // Search should find it with junior filter + let results = await brainy.searchText('Test developer', 10, { + metadata: { level: 'junior' } + }) + expect(results.some(r => r.id === id)).toBe(true) + + // Update metadata + await brainy.updateMetadata(id, { level: 'senior' }) + + // Should now find it with senior filter + results = await brainy.searchText('Test developer', 10, { + metadata: { level: 'senior' } + }) + expect(results.some(r => r.id === id)).toBe(true) + + // Should NOT find it with junior filter anymore + results = await brainy.searchText('Test developer', 10, { + metadata: { level: 'junior' } + }) + expect(results.some(r => r.id === id)).toBe(false) + }) + + it('should handle null/undefined metadata gracefully', async () => { + // Add item without metadata + await brainy.add('No metadata item') + + // Search with filter should not crash + const results = await brainy.searchText('metadata', 10, { + metadata: { level: 'senior' } + }) + + // Should only return items that match the filter + expect(results.every(r => r.metadata?.level === 'senior')).toBe(true) + }) + }) + } + + describe('Performance considerations', () => { + it('should handle large result sets efficiently', async () => { + const brainy = new BrainyData({ + storage: { forceMemoryStorage: true }, + hnsw: { M: 16, efConstruction: 100 }, + logging: { verbose: false } + }) + await brainy.init() + + // Add many items + const categories = ['A', 'B', 'C', 'D', 'E'] + const levels = ['junior', 'mid', 'senior'] + + for (let i = 0; i < 100; i++) { + await brainy.add( + `Item ${i} with various properties`, + { + category: categories[i % categories.length], + level: levels[i % levels.length], + index: i + } + ) + } + + const startTime = Date.now() + + // Search with complex filter + const results = await brainy.searchText('Item', 20, { + metadata: { + $and: [ + { category: { $in: ['A', 'B', 'C'] } }, + { level: { $ne: 'junior' } } + ] + } + }) + + const duration = Date.now() - startTime + + expect(results.length).toBeGreaterThan(0) + expect(results.length).toBeLessThanOrEqual(20) + expect(duration).toBeLessThan(1000) // Should complete within 1 second + + // Verify all results match the filter + expect(results.every(r => { + const m = r.metadata + return ['A', 'B', 'C'].includes(m?.category) && m?.level !== 'junior' + })).toBe(true) + }) + }) +}) \ No newline at end of file diff --git a/tests/metadata-filter.test.ts b/tests/metadata-filter.test.ts new file mode 100644 index 00000000..0ab7e9ea --- /dev/null +++ b/tests/metadata-filter.test.ts @@ -0,0 +1,269 @@ +import { describe, it, expect, beforeEach } from 'vitest' +import { BrainyData } from '../src/brainyData.js' +import { matchesMetadataFilter } from '../src/utils/metadataFilter.js' + +describe('Metadata Filtering', () => { + let brainy: BrainyData + + beforeEach(async () => { + brainy = new BrainyData({ + storage: { forceMemoryStorage: true }, + hnsw: { M: 8, efConstruction: 50 }, + logging: { verbose: false } + }) + await brainy.init() + console.log('BrainyData initialized') + }) + + describe('matchesMetadataFilter', () => { + it('should match simple equality filters', () => { + const metadata = { level: 'senior', location: 'SF' } + + expect(matchesMetadataFilter(metadata, { level: 'senior' })).toBe(true) + expect(matchesMetadataFilter(metadata, { level: 'junior' })).toBe(false) + expect(matchesMetadataFilter(metadata, { level: 'senior', location: 'SF' })).toBe(true) + expect(matchesMetadataFilter(metadata, { level: 'senior', location: 'NYC' })).toBe(false) + }) + + it('should support MongoDB-style operators', () => { + const metadata = { age: 30, skills: ['React', 'Vue'], name: 'John' } + + // $gt, $gte, $lt, $lte + expect(matchesMetadataFilter(metadata, { age: { $gt: 25 } })).toBe(true) + expect(matchesMetadataFilter(metadata, { age: { $lt: 25 } })).toBe(false) + expect(matchesMetadataFilter(metadata, { age: { $gte: 30 } })).toBe(true) + expect(matchesMetadataFilter(metadata, { age: { $lte: 30 } })).toBe(true) + + // $in, $nin + expect(matchesMetadataFilter(metadata, { age: { $in: [25, 30, 35] } })).toBe(true) + expect(matchesMetadataFilter(metadata, { age: { $nin: [25, 35] } })).toBe(true) + expect(matchesMetadataFilter(metadata, { age: { $nin: [30] } })).toBe(false) + + // $includes for arrays + expect(matchesMetadataFilter(metadata, { skills: { $includes: 'React' } })).toBe(true) + expect(matchesMetadataFilter(metadata, { skills: { $includes: 'Angular' } })).toBe(false) + + // $regex + expect(matchesMetadataFilter(metadata, { name: { $regex: '^Jo' } })).toBe(true) + expect(matchesMetadataFilter(metadata, { name: { $regex: 'hn$' } })).toBe(true) + expect(matchesMetadataFilter(metadata, { name: { $regex: 'Jane' } })).toBe(false) + }) + + it('should support nested fields with dot notation', () => { + const metadata = { + user: { + profile: { + level: 'senior', + skills: ['React', 'TypeScript'] + } + } + } + + expect(matchesMetadataFilter(metadata, { 'user.profile.level': 'senior' })).toBe(true) + expect(matchesMetadataFilter(metadata, { 'user.profile.level': 'junior' })).toBe(false) + expect(matchesMetadataFilter(metadata, { + 'user.profile.skills': { $includes: 'React' } + })).toBe(true) + }) + + it('should support logical operators', () => { + const metadata = { level: 'senior', location: 'SF', remote: true } + + // $and + expect(matchesMetadataFilter(metadata, { + $and: [ + { level: 'senior' }, + { location: 'SF' } + ] + })).toBe(true) + + expect(matchesMetadataFilter(metadata, { + $and: [ + { level: 'senior' }, + { location: 'NYC' } + ] + })).toBe(false) + + // $or + expect(matchesMetadataFilter(metadata, { + $or: [ + { location: 'NYC' }, + { location: 'SF' } + ] + })).toBe(true) + + expect(matchesMetadataFilter(metadata, { + $or: [ + { location: 'NYC' }, + { location: 'LA' } + ] + })).toBe(false) + + // $not + expect(matchesMetadataFilter(metadata, { + $not: { location: 'NYC' } + })).toBe(true) + + expect(matchesMetadataFilter(metadata, { + $not: { location: 'SF' } + })).toBe(false) + }) + }) + + describe('Search with metadata filtering', () => { + beforeEach(async () => { + // Add test data + const developers = [ + { name: 'Alice', level: 'senior', skills: ['React', 'TypeScript'], location: 'SF', available: true }, + { name: 'Bob', level: 'mid', skills: ['Vue', 'JavaScript'], location: 'NYC', available: true }, + { name: 'Charlie', level: 'senior', skills: ['React', 'Python'], location: 'SF', available: false }, + { name: 'David', level: 'junior', skills: ['JavaScript'], location: 'LA', available: true }, + { name: 'Eve', level: 'senior', skills: ['Angular', 'TypeScript'], location: 'NYC', available: true } + ] + + for (const dev of developers) { + await brainy.add( + `${dev.name} is a ${dev.level} developer with ${dev.skills.join(', ')} skills in ${dev.location}`, + dev + ) + } + }) + + it('should filter by simple metadata fields', async () => { + // First check what we have without filter + const allResults = await brainy.searchText('developer', 10) + console.log('All results:', allResults.map(r => ({ + id: r.id.substring(0, 8), + level: r.metadata?.level, + name: r.metadata?.name + }))) + + // Now with filter + const results = await brainy.searchText('developer', 10, { + metadata: { level: 'senior' } + }) + + console.log('Filtered results:', results.map(r => ({ + id: r.id.substring(0, 8), + level: r.metadata?.level, + name: r.metadata?.name + }))) + + expect(results.length).toBeGreaterThan(0) + expect(results.every(r => r.metadata?.level === 'senior')).toBe(true) + }) + + it('should filter by multiple metadata fields', async () => { + const results = await brainy.searchText('developer', 10, { + metadata: { + level: 'senior', + location: 'SF' + } + }) + + expect(results.length).toBeGreaterThan(0) + expect(results.every(r => + r.metadata?.level === 'senior' && + r.metadata?.location === 'SF' + )).toBe(true) + }) + + it('should filter with MongoDB operators', async () => { + // First verify what we have in the index + const allResults = await brainy.searchText('developer', 10) + console.log('All results before filtering:', allResults.map(r => ({ + name: r.metadata?.name, + skills: r.metadata?.skills, + available: r.metadata?.available + }))) + + const results = await brainy.searchText('developer', 10, { + metadata: { + skills: { $includes: 'React' }, + available: true + } + }) + + console.log('Filtered results:', results.map(r => ({ + name: r.metadata?.name, + skills: r.metadata?.skills, + available: r.metadata?.available + }))) + + expect(results.length).toBeGreaterThan(0) + + // Check each result individually for debugging + for (const r of results) { + const hasReact = r.metadata?.skills?.includes('React') + const isAvailable = r.metadata?.available === true + if (!hasReact || !isAvailable) { + console.log('Failed result:', r.metadata) + } + } + + expect(results.every(r => + r.metadata?.skills?.includes('React') && + r.metadata?.available === true + )).toBe(true) + }) + + it('should filter with complex queries', async () => { + const results = await brainy.searchText('developer', 10, { + metadata: { + $or: [ + { location: 'SF' }, + { location: 'NYC' } + ], + level: { $in: ['senior', 'mid'] } + } + }) + + expect(results.length).toBeGreaterThan(0) + expect(results.every(r => { + const m = r.metadata + return (m?.location === 'SF' || m?.location === 'NYC') && + (m?.level === 'senior' || m?.level === 'mid') + })).toBe(true) + }) + }) + + describe('searchWithinItems', () => { + let itemIds: string[] = [] + + beforeEach(async () => { + // Add test data and collect IDs + const items = [ + { content: 'JavaScript programming', category: 'tech' }, + { content: 'TypeScript development', category: 'tech' }, + { content: 'Python data science', category: 'tech' }, + { content: 'React components', category: 'frontend' }, + { content: 'Vue templates', category: 'frontend' } + ] + + for (const item of items) { + const id = await brainy.add(item.content, item) + if (item.category === 'frontend') { + itemIds.push(id) + } + } + }) + + it('should search only within specified items', async () => { + // Search within frontend items only + const results = await brainy.searchWithinItems('JavaScript', itemIds, 5) + + expect(results.length).toBeLessThanOrEqual(itemIds.length) + expect(results.every(r => itemIds.includes(r.id))).toBe(true) + }) + + it('should return empty results if no items match', async () => { + const results = await brainy.searchWithinItems('JavaScript', [], 5) + expect(results).toEqual([]) + }) + + it('should limit results to k even if more items are provided', async () => { + const results = await brainy.searchWithinItems('development', itemIds, 1) + expect(results.length).toBe(1) + }) + }) +}) \ No newline at end of file diff --git a/tests/metadata-performance.test.ts b/tests/metadata-performance.test.ts new file mode 100644 index 00000000..cd58e24c --- /dev/null +++ b/tests/metadata-performance.test.ts @@ -0,0 +1,568 @@ +/** + * Metadata Filtering Performance Analysis + * + * This test suite analyzes the performance impact of the metadata filtering system: + * 1. Index Build Time - How metadata indexing affects initialization + * 2. Index Storage Overhead - Storage space required for inverted indexes + * 3. Search Performance - Filtered vs non-filtered search speeds + * 4. Memory Usage - Additional memory needed for metadata indexes + * 5. Write Performance - Impact on add/update/delete operations + */ + +import { describe, it, expect, beforeEach } from 'vitest' +import { BrainyData } from '../src/brainyData.js' +import { MetadataIndexManager } from '../src/utils/metadataIndex.js' + +// Helper function to measure execution time +const measureTime = async (fn: () => Promise): Promise<{ result: any, time: number }> => { + const start = performance.now() + const result = await fn() + const end = performance.now() + return { result, time: end - start } +} + +// Helper function to estimate memory usage +const measureMemory = () => { + if (typeof performance.memory !== 'undefined') { + return { + used: performance.memory.usedJSHeapSize, + total: performance.memory.totalJSHeapSize, + limit: performance.memory.jsHeapSizeLimit + } + } + return null +} + +// Generate realistic test data with metadata +const generateTestDataWithMetadata = (count: number) => { + const departments = ['Engineering', 'Marketing', 'Sales', 'HR', 'Finance', 'Operations'] + const levels = ['junior', 'senior', 'staff', 'principal', 'director'] + const locations = ['SF', 'NYC', 'LA', 'Seattle', 'Austin', 'Boston'] + const skills = ['JavaScript', 'Python', 'React', 'Node.js', 'TypeScript', 'SQL', 'AWS', 'Docker'] + const companies = ['TechCorp', 'DataSys', 'CloudInc', 'DevTools', 'AILabs'] + + return Array.from({ length: count }, (_, i) => ({ + text: `Profile ${i}: Professional with extensive experience in software development and team leadership`, + metadata: { + id: `profile-${i}`, + department: departments[i % departments.length], + level: levels[i % levels.length], + location: locations[i % locations.length], + salary: 50000 + (i % 10) * 10000, + experience: 1 + (i % 15), + skills: skills.slice(0, 2 + (i % 4)), + company: companies[i % companies.length], + remote: i % 3 === 0, + active: i % 5 !== 0, + tags: [`tag-${i % 20}`, `category-${i % 10}`], + nested: { + profile: { + rating: 1 + (i % 5), + verified: i % 4 === 0 + }, + preferences: { + timezone: `UTC-${(i % 12) - 6}`, + workStyle: i % 2 === 0 ? 'collaborative' : 'independent' + } + } + } + })) +} + +describe('Metadata Filtering Performance Analysis', () => { + describe('1. Index Build Time Impact', () => { + it('should measure initialization time with vs without metadata indexing', async () => { + const testData = generateTestDataWithMetadata(500) + + console.log('\n=== Index Build Time Analysis ===') + + // Test WITHOUT metadata indexing + const withoutIndexing = await measureTime(async () => { + const brainy = new BrainyData({ + storage: { forceMemoryStorage: true }, + hnsw: { M: 8, efConstruction: 50 }, + logging: { verbose: false } + // No metadataIndex config + }) + await brainy.init() + + // Add data + for (const item of testData) { + await brainy.add(item.text, item.metadata) + } + + return brainy + }) + + console.log(`WITHOUT indexing: ${withoutIndexing.time.toFixed(2)}ms for 500 items`) + console.log(`Per item: ${(withoutIndexing.time / 500).toFixed(2)}ms`) + + // Test WITH metadata indexing + const withIndexing = await measureTime(async () => { + const brainy = new BrainyData({ + storage: { forceMemoryStorage: true }, + hnsw: { M: 8, efConstruction: 50 }, + logging: { verbose: false }, + metadataIndex: { + maxIndexSize: 10000, + autoOptimize: true, + excludeFields: ['id'] + } + }) + await brainy.init() + + // Add data + for (const item of testData) { + await brainy.add(item.text, item.metadata) + } + + return brainy + }) + + console.log(`WITH indexing: ${withIndexing.time.toFixed(2)}ms for 500 items`) + console.log(`Per item: ${(withIndexing.time / 500).toFixed(2)}ms`) + + const overhead = ((withIndexing.time - withoutIndexing.time) / withoutIndexing.time) * 100 + console.log(`Index build overhead: ${overhead.toFixed(1)}%`) + + // Cleanup + await withoutIndexing.result.shutDown() + await withIndexing.result.shutDown() + }) + + it('should measure batch insert performance with indexing', async () => { + const brainy = new BrainyData({ + storage: { forceMemoryStorage: true }, + metadataIndex: { autoOptimize: true }, + logging: { verbose: false } + }) + await brainy.init() + + const batchSizes = [50, 100, 200, 500] + console.log('\n=== Batch Insert Performance ===') + + for (const size of batchSizes) { + const testData = generateTestDataWithMetadata(size) + + const { time } = await measureTime(async () => { + for (const item of testData) { + await brainy.add(item.text, item.metadata) + } + }) + + console.log(`${size} items: ${time.toFixed(2)}ms (${(time / size).toFixed(2)}ms per item)`) + + // Clear for next batch + await brainy.clear() + } + + await brainy.shutDown() + }) + }) + + describe('2. Index Storage Overhead', () => { + it('should analyze storage requirements for metadata indexes', async () => { + const brainy = new BrainyData({ + storage: { forceMemoryStorage: true }, + metadataIndex: { autoOptimize: true }, + logging: { verbose: false } + }) + await brainy.init() + + const testData = generateTestDataWithMetadata(1000) + + console.log('\n=== Storage Overhead Analysis ===') + + // Add data and measure index size + for (const item of testData) { + await brainy.addNoun(item.text, item.metadata) + } + + // Get index statistics + if (brainy.metadataIndex) { + const stats = await brainy.metadataIndex.getStats() + console.log(`Total index entries: ${stats.totalEntries}`) + console.log(`Total indexed IDs: ${stats.totalIds}`) + console.log(`Fields indexed: ${stats.fieldsIndexed.length}`) + console.log(`Estimated index size: ${stats.indexSize} bytes`) + console.log(`Fields: ${stats.fieldsIndexed.join(', ')}`) + + // Calculate overhead per item + const overheadPerItem = stats.indexSize / 1000 + console.log(`Storage overhead per item: ${overheadPerItem.toFixed(2)} bytes`) + + // Estimate total storage efficiency + const totalDataSize = 1000 * 200 // rough estimate of 200 bytes per item + const storageEfficiency = (stats.indexSize / totalDataSize) * 100 + console.log(`Index storage overhead: ${storageEfficiency.toFixed(1)}% of data size`) + } + + await brainy.shutDown() + }) + }) + + describe('3. Search Performance Comparison', () => { + it('should compare filtered vs non-filtered search performance', async () => { + const brainy = new BrainyData({ + storage: { forceMemoryStorage: true }, + metadataIndex: { autoOptimize: true }, + logging: { verbose: false } + }) + await brainy.init() + + // Add test data + const testData = generateTestDataWithMetadata(1000) + for (const item of testData) { + await brainy.addNoun(item.text, item.metadata) + } + + console.log('\n=== Search Performance Comparison ===') + + const searchQuery = 'Professional software development experience' + const numSearches = 10 + + // Test 1: No filtering + const noFilterTimes: number[] = [] + for (let i = 0; i < numSearches; i++) { + const { time } = await measureTime(async () => { + return await brainy.search(searchQuery, 20) + }) + noFilterTimes.push(time) + } + const avgNoFilter = noFilterTimes.reduce((a, b) => a + b) / numSearches + console.log(`No filtering: ${avgNoFilter.toFixed(2)}ms average`) + + // Test 2: Simple metadata filtering (high selectivity) + const simpleFilterTimes: number[] = [] + for (let i = 0; i < numSearches; i++) { + const { time } = await measureTime(async () => { + return await brainy.search(searchQuery, 20, { + metadata: { department: 'Engineering' } + }) + }) + simpleFilterTimes.push(time) + } + const avgSimpleFilter = simpleFilterTimes.reduce((a, b) => a + b) / numSearches + console.log(`Simple filter (dept=Engineering): ${avgSimpleFilter.toFixed(2)}ms average`) + + // Test 3: Complex metadata filtering (low selectivity) + const complexFilterTimes: number[] = [] + for (let i = 0; i < numSearches; i++) { + const { time } = await measureTime(async () => { + return await brainy.search(searchQuery, 20, { + metadata: { + department: { $in: ['Engineering', 'Marketing'] }, + level: { $in: ['senior', 'staff'] }, + salary: { $gte: 80000 }, + remote: true + } + }) + }) + complexFilterTimes.push(time) + } + const avgComplexFilter = complexFilterTimes.reduce((a, b) => a + b) / numSearches + console.log(`Complex filter: ${avgComplexFilter.toFixed(2)}ms average`) + + // Test 4: Nested field filtering + const nestedFilterTimes: number[] = [] + for (let i = 0; i < numSearches; i++) { + const { time } = await measureTime(async () => { + return await brainy.search(searchQuery, 20, { + metadata: { + 'nested.profile.rating': { $gte: 4 }, + 'nested.profile.verified': true + } + }) + }) + nestedFilterTimes.push(time) + } + const avgNestedFilter = nestedFilterTimes.reduce((a, b) => a + b) / numSearches + console.log(`Nested filter: ${avgNestedFilter.toFixed(2)}ms average`) + + // Performance analysis + console.log('\nPerformance Impact:') + console.log(`Simple filter overhead: ${((avgSimpleFilter / avgNoFilter - 1) * 100).toFixed(1)}%`) + console.log(`Complex filter overhead: ${((avgComplexFilter / avgNoFilter - 1) * 100).toFixed(1)}%`) + console.log(`Nested filter overhead: ${((avgNestedFilter / avgNoFilter - 1) * 100).toFixed(1)}%`) + + await brainy.shutDown() + }) + + it('should test search performance with different ef multipliers', async () => { + const brainy = new BrainyData({ + storage: { forceMemoryStorage: true }, + metadataIndex: { autoOptimize: true }, + hnsw: { efSearch: 50 }, // Base ef for testing multiplier effect + logging: { verbose: false } + }) + await brainy.init() + + // Add test data + const testData = generateTestDataWithMetadata(500) + for (const item of testData) { + await brainy.addNoun(item.text, item.metadata) + } + + console.log('\n=== EF Multiplier Impact Analysis ===') + + const searchQuery = 'Professional software development experience' + + // Test with different selectivity filters + const filters = [ + { name: 'High selectivity', filter: { department: 'Engineering' }, expected: '~17%' }, + { name: 'Medium selectivity', filter: { level: { $in: ['senior', 'staff'] } }, expected: '~40%' }, + { name: 'Low selectivity', filter: { active: true }, expected: '~80%' } + ] + + for (const { name, filter, expected } of filters) { + const { result, time } = await measureTime(async () => { + return await brainy.search(searchQuery, 10, { metadata: filter }) + }) + + console.log(`${name} (${expected}): ${time.toFixed(2)}ms, ${result.length} results`) + } + + await brainy.shutDown() + }) + }) + + describe('4. Memory Usage Analysis', () => { + it('should measure memory consumption of metadata indexes', async () => { + if (!measureMemory()) { + console.log('\nMemory measurement not available in this environment') + return + } + + console.log('\n=== Memory Usage Analysis ===') + + const initialMemory = measureMemory()! + console.log(`Initial memory: ${(initialMemory.used / 1024 / 1024).toFixed(2)}MB`) + + const brainy = new BrainyData({ + storage: { forceMemoryStorage: true }, + metadataIndex: { autoOptimize: true }, + logging: { verbose: false } + }) + await brainy.init() + + const afterInitMemory = measureMemory()! + console.log(`After init: ${(afterInitMemory.used / 1024 / 1024).toFixed(2)}MB`) + + // Add data in batches and measure memory growth + const batchSize = 100 + const numBatches = 5 + + for (let batch = 1; batch <= numBatches; batch++) { + const testData = generateTestDataWithMetadata(batchSize) + + for (const item of testData) { + await brainy.add(item.text, item.metadata) + } + + const currentMemory = measureMemory()! + const totalItems = batch * batchSize + console.log(`${totalItems} items: ${(currentMemory.used / 1024 / 1024).toFixed(2)}MB`) + } + + // Get final index stats + if (brainy.metadataIndex) { + const stats = await brainy.metadataIndex.getStats() + console.log(`Index entries: ${stats.totalEntries}, Memory per entry: ${((measureMemory()!.used - initialMemory.used) / stats.totalEntries).toFixed(2)} bytes`) + } + + await brainy.shutDown() + }) + }) + + describe('5. Write Performance Impact', () => { + it('should measure add/update/delete performance with indexing', async () => { + const brainy = new BrainyData({ + storage: { forceMemoryStorage: true }, + metadataIndex: { autoOptimize: true }, + logging: { verbose: false } + }) + await brainy.init() + + console.log('\n=== Write Performance Analysis ===') + + // Test ADD performance + const testData = generateTestDataWithMetadata(200) + const { time: addTime } = await measureTime(async () => { + for (const item of testData) { + await brainy.add(item.text, item.metadata) + } + }) + console.log(`ADD: 200 items in ${addTime.toFixed(2)}ms (${(addTime / 200).toFixed(2)}ms per item)`) + + // Test UPDATE performance + const updateData = testData.slice(0, 50).map((item, i) => ({ + ...item, + metadata: { + ...item.metadata, + level: 'updated-level', + salary: item.metadata.salary + 10000, + updateCount: i + } + })) + + const { time: updateTime } = await measureTime(async () => { + for (const item of updateData) { + await brainy.updateMetadata(item.metadata.id, item.metadata) + } + }) + console.log(`UPDATE: 50 items in ${updateTime.toFixed(2)}ms (${(updateTime / 50).toFixed(2)}ms per item)`) + + // Test DELETE performance + const idsToDelete = testData.slice(100, 150).map(item => item.metadata.id) + const { time: deleteTime } = await measureTime(async () => { + for (const id of idsToDelete) { + await brainy.delete(id) + } + }) + console.log(`DELETE: 50 items in ${deleteTime.toFixed(2)}ms (${(deleteTime / 50).toFixed(2)}ms per item)`) + + // Verify index consistency + if (brainy.metadataIndex) { + const stats = await brainy.metadataIndex.getStats() + console.log(`Final index state: ${stats.totalEntries} entries, ${stats.totalIds} IDs`) + + // Should have 150 items remaining (200 - 50 deleted) + const expectedItems = 200 - 50 + const actualItems = await brainy.size() + console.log(`Data consistency: ${actualItems}/${expectedItems} items remaining`) + } + + await brainy.shutDown() + }) + + it('should test concurrent write performance', async () => { + const brainy = new BrainyData({ + storage: { forceMemoryStorage: true }, + metadataIndex: { autoOptimize: true }, + logging: { verbose: false } + }) + await brainy.init() + + console.log('\n=== Concurrent Write Performance ===') + + const testData = generateTestDataWithMetadata(100) + + // Sequential writes + const { time: sequentialTime } = await measureTime(async () => { + for (const item of testData) { + await brainy.add(item.text, item.metadata) + } + }) + + await brainy.clear() + + // Concurrent writes (batched) + const batchSize = 20 + const { time: concurrentTime } = await measureTime(async () => { + const promises: Promise[] = [] + + for (let i = 0; i < testData.length; i += batchSize) { + const batch = testData.slice(i, i + batchSize) + promises.push( + Promise.all(batch.map(item => brainy.add(item.text, item.metadata))) + ) + } + + await Promise.all(promises) + }) + + console.log(`Sequential: ${sequentialTime.toFixed(2)}ms`) + console.log(`Concurrent (batched): ${concurrentTime.toFixed(2)}ms`) + console.log(`Speedup: ${(sequentialTime / concurrentTime).toFixed(2)}x`) + + await brainy.shutDown() + }) + }) + + describe('6. Index Maintenance and Optimization', () => { + it('should analyze index rebuild performance', async () => { + const brainy = new BrainyData({ + storage: { forceMemoryStorage: true }, + metadataIndex: { + autoOptimize: true, + rebuildThreshold: 0.1 + }, + logging: { verbose: false } + }) + await brainy.init() + + console.log('\n=== Index Maintenance Analysis ===') + + // Add initial data + const testData = generateTestDataWithMetadata(300) + for (const item of testData) { + await brainy.addNoun(item.text, item.metadata) + } + + // Measure manual rebuild + if (brainy.metadataIndex) { + const { time: rebuildTime } = await measureTime(async () => { + await brainy.metadataIndex!.rebuild() + }) + + const stats = await brainy.metadataIndex.getStats() + console.log(`Rebuild: ${rebuildTime.toFixed(2)}ms for ${stats.totalEntries} entries`) + console.log(`Per entry: ${(rebuildTime / stats.totalEntries).toFixed(2)}ms`) + + // Test flush performance + const { time: flushTime } = await measureTime(async () => { + await brainy.metadataIndex!.flush() + }) + console.log(`Flush: ${flushTime.toFixed(2)}ms`) + } + + await brainy.shutDown() + }) + + it('should test index cache performance', async () => { + const brainy = new BrainyData({ + storage: { forceMemoryStorage: true }, + metadataIndex: { + maxIndexSize: 1000, + autoOptimize: true + }, + logging: { verbose: false } + }) + await brainy.init() + + console.log('\n=== Index Cache Performance ===') + + // Add test data + const testData = generateTestDataWithMetadata(200) + for (const item of testData) { + await brainy.addNoun(item.text, item.metadata) + } + + if (!brainy.metadataIndex) return + + // Test cache hit performance (repeated queries) + const filter = { department: 'Engineering' } + + // First query (cache miss) + const { time: cacheMissTime } = await measureTime(async () => { + return await brainy.metadataIndex!.getIdsForCriteria(filter) + }) + + // Subsequent queries (cache hits) + const cacheHitTimes: number[] = [] + for (let i = 0; i < 10; i++) { + const { time } = await measureTime(async () => { + return await brainy.metadataIndex!.getIdsForCriteria(filter) + }) + cacheHitTimes.push(time) + } + + const avgCacheHit = cacheHitTimes.reduce((a, b) => a + b) / cacheHitTimes.length + console.log(`Cache miss: ${cacheMissTime.toFixed(2)}ms`) + console.log(`Cache hit (avg): ${avgCacheHit.toFixed(2)}ms`) + console.log(`Cache speedup: ${(cacheMissTime / avgCacheHit).toFixed(2)}x`) + + await brainy.shutDown() + }) + }) +}) \ No newline at end of file diff --git a/tests/mocks/opfs-mock.ts b/tests/mocks/opfs-mock.ts new file mode 100644 index 00000000..fd432e87 --- /dev/null +++ b/tests/mocks/opfs-mock.ts @@ -0,0 +1,241 @@ +/** + * OPFS (Origin Private File System) Mock for Testing + * + * This module provides a comprehensive mock implementation of the OPFS API + * for testing OPFS-based storage in a Node.js environment. + */ + +import { vi } from 'vitest' + +// In-memory storage to simulate file system +const mockFileSystem: Map> = new Map() + +// Mock file data +interface MockFileData { + content: string + type: string +} + +/** + * Create a mock FileSystemFileHandle + */ +export function createMockFileHandle(fileName: string, content: string = '{}') { + return { + kind: 'file', + name: fileName, + getFile: vi.fn().mockResolvedValue({ + text: vi.fn().mockResolvedValue(content), + arrayBuffer: vi.fn().mockResolvedValue(new TextEncoder().encode(content).buffer), + size: content.length + }), + createWritable: vi.fn().mockImplementation(() => { + const writable = { + write: vi.fn().mockImplementation((data: string | ArrayBuffer) => { + // Store the data in our mock file system + const path = mockFileSystem.get('currentPath') || '/' + const dirMap = mockFileSystem.get(path) || new Map() + + let content: string + if (typeof data === 'string') { + content = data + } else if (data instanceof ArrayBuffer) { + content = new TextDecoder().decode(data) + } else if (data && typeof data === 'object' && 'type' in data && data.type === 'write') { + // Handle FileSystemWriteChunkType + const chunk = data as { type: 'write', position?: number, data: string | ArrayBuffer } + if (typeof chunk.data === 'string') { + content = chunk.data + } else { + content = new TextDecoder().decode(chunk.data) + } + } else { + content = JSON.stringify(data) + } + + dirMap.set(fileName, { content, type: 'file' }) + mockFileSystem.set(path, dirMap) + return Promise.resolve() + }), + close: vi.fn().mockResolvedValue(undefined) + } + return Promise.resolve(writable) + }) + } +} + +/** + * Create a mock FileSystemDirectoryHandle + */ +export function createMockDirectoryHandle(dirName: string, entries: Map = new Map()) { + const dirPath = mockFileSystem.get('currentPath') || '/' + const fullPath = dirPath === '/' ? `/${dirName}` : `${dirPath}/${dirName}` + + // Store the directory in our mock file system + mockFileSystem.set(fullPath, entries) + + return { + kind: 'directory', + name: dirName, + getDirectoryHandle: vi.fn().mockImplementation((name: string, options: { create?: boolean } = {}) => { + mockFileSystem.set('currentPath', fullPath) + + const dirEntries = mockFileSystem.get(fullPath) || new Map() + const entry = dirEntries.get(name) + + if (entry && entry.type === 'directory') { + return Promise.resolve(createMockDirectoryHandle(name, entry.content)) + } + + if (!entry && options.create) { + const newDir = new Map() + dirEntries.set(name, { content: newDir, type: 'directory' }) + mockFileSystem.set(fullPath, dirEntries) + return Promise.resolve(createMockDirectoryHandle(name, newDir)) + } + + return Promise.reject(new Error(`Directory not found: ${name}`)) + }), + getFileHandle: vi.fn().mockImplementation((name: string, options: { create?: boolean } = {}) => { + mockFileSystem.set('currentPath', fullPath) + + const dirEntries = mockFileSystem.get(fullPath) || new Map() + const entry = dirEntries.get(name) + + if (entry && entry.type === 'file') { + return Promise.resolve(createMockFileHandle(name, entry.content)) + } + + if (!entry && options.create) { + return Promise.resolve(createMockFileHandle(name)) + } + + return Promise.reject(new Error(`File not found: ${name}`)) + }), + removeEntry: vi.fn().mockImplementation((name: string, options: { recursive?: boolean } = {}) => { + const dirEntries = mockFileSystem.get(fullPath) || new Map() + + if (!dirEntries.has(name)) { + return Promise.reject(new Error(`Entry not found: ${name}`)) + } + + const entry = dirEntries.get(name) + + if (entry.type === 'directory' && !options.recursive) { + const subDirPath = fullPath === '/' ? `/${name}` : `${fullPath}/${name}` + const subDirEntries = mockFileSystem.get(subDirPath) || new Map() + + if (subDirEntries.size > 0) { + return Promise.reject(new Error(`Directory not empty: ${name}`)) + } + } + + dirEntries.delete(name) + + if (entry.type === 'directory') { + const subDirPath = fullPath === '/' ? `/${name}` : `${fullPath}/${name}` + mockFileSystem.delete(subDirPath) + } + + return Promise.resolve() + }), + entries: vi.fn().mockImplementation(async function* () { + const dirEntries = mockFileSystem.get(fullPath) || new Map() + + for (const [name, entry] of dirEntries.entries()) { + if (entry.type === 'file') { + yield [name, createMockFileHandle(name, entry.content)] + } else { + yield [name, createMockDirectoryHandle(name, entry.content)] + } + } + }), + values: vi.fn().mockImplementation(async function* () { + const dirEntries = mockFileSystem.get(fullPath) || new Map() + + for (const [name, entry] of dirEntries.entries()) { + if (entry.type === 'file') { + yield createMockFileHandle(name, entry.content) + } else { + yield createMockDirectoryHandle(name, entry.content) + } + } + }), + keys: vi.fn().mockImplementation(async function* () { + const dirEntries = mockFileSystem.get(fullPath) || new Map() + + for (const name of dirEntries.keys()) { + yield name + } + }) + } +} + +/** + * Setup OPFS mock environment + */ +export function setupOPFSMock() { + // Clear the mock file system + mockFileSystem.clear() + mockFileSystem.set('/', new Map()) + mockFileSystem.set('currentPath', '/') + + // Create root directory handle + const rootDirectoryHandle = createMockDirectoryHandle('root') + + // Mock navigator.storage if it doesn't exist + if (typeof global.navigator === 'undefined') { + // @ts-expect-error - Mocking global + global.navigator = {} + } + + // Define storage if it doesn't exist + if (typeof global.navigator.storage === 'undefined') { + Object.defineProperty(global.navigator, 'storage', { + value: {}, + writable: true, + configurable: true + }) + } + + // Mock storage methods + global.navigator.storage.getDirectory = vi.fn().mockResolvedValue(rootDirectoryHandle) + global.navigator.storage.persisted = vi.fn().mockResolvedValue(true) + global.navigator.storage.persist = vi.fn().mockResolvedValue(true) + global.navigator.storage.estimate = vi.fn().mockImplementation(() => { + // Calculate total size of all files in the mock file system + let totalSize = 0 + + for (const [path, entries] of mockFileSystem.entries()) { + if (path === 'currentPath') continue + + for (const [_, entry] of entries.entries()) { + if (entry.type === 'file') { + totalSize += entry.content.length + } + } + } + + return Promise.resolve({ usage: totalSize, quota: 10 * 1024 * 1024 }) // 10MB quota + }) + + return { + rootDirectoryHandle, + mockFileSystem, + reset: () => { + mockFileSystem.clear() + mockFileSystem.set('/', new Map()) + mockFileSystem.set('currentPath', '/') + } + } +} + +/** + * Cleanup OPFS mock environment + */ +export function cleanupOPFSMock() { + // Reset mocks + vi.restoreAllMocks() + + // Clear the mock file system + mockFileSystem.clear() +} \ No newline at end of file diff --git a/tests/mocks/s3-mock.ts b/tests/mocks/s3-mock.ts new file mode 100644 index 00000000..a4c105ee --- /dev/null +++ b/tests/mocks/s3-mock.ts @@ -0,0 +1,576 @@ +/** + * S3 Compatible Storage Mock for Testing + * + * This module provides a mock implementation of the AWS S3 client + * for testing S3-based storage in a Node.js environment without + * requiring actual S3 credentials. + */ + +import { vi } from 'vitest' + +// In-memory storage to simulate S3 bucket +interface S3MockObject { + key: string + body: string + metadata?: Record + lastModified: Date + contentLength: number + contentType?: string +} + +interface S3MockBucket { + name: string + objects: Map +} + +// Mock S3 storage - use a global variable to ensure persistence between operations +// This is important because the mock client is recreated for each command +const mockS3Storage = new Map() + +/** + * Create a mock S3 client + * + * This function creates a mock S3 client that simulates the behavior of the AWS S3 client. + * It's important that all operations use the same instance of mockS3Storage to ensure + * that objects are correctly persisted between operations. + */ +export function createMockS3Client() { + // Log the current state of the mock storage + console.log(`[MOCK S3] Creating mock S3 client with current storage state:`) + for (const [bucketName, bucket] of mockS3Storage.entries()) { + console.log(`[MOCK S3] Bucket ${bucketName}: ${bucket.objects.size} objects`) + if (bucket.objects.size > 0) { + console.log('[MOCK S3] Objects in bucket:') + for (const key of bucket.objects.keys()) { + console.log(`[MOCK S3] - ${key}`) + } + } + } + + return { + send: vi.fn().mockImplementation((command) => { + // Log the command for debugging + console.log(`[MOCK S3] Received S3 command: ${command.constructor.name}`, command.input) + + // Log the current state of the mock storage before processing the command + console.log(`[MOCK S3] Current storage state before processing command:`) + for (const [bucketName, bucket] of mockS3Storage.entries()) { + console.log(`[MOCK S3] Bucket ${bucketName}: ${bucket.objects.size} objects`) + if (bucket.objects.size > 0) { + console.log('[MOCK S3] Objects in bucket:') + for (const key of bucket.objects.keys()) { + console.log(`[MOCK S3] - ${key}`) + } + } + } + + // Handle different command types + let result + if (command.constructor.name === 'CreateBucketCommand') { + result = handleCreateBucket(command) + } else if (command.constructor.name === 'HeadBucketCommand') { + result = handleHeadBucket(command) + } else if (command.constructor.name === 'PutObjectCommand') { + result = handlePutObject(command) + } else if (command.constructor.name === 'GetObjectCommand') { + result = handleGetObject(command) + } else if (command.constructor.name === 'DeleteObjectCommand') { + result = handleDeleteObject(command) + } else if (command.constructor.name === 'ListObjectsV2Command') { + result = handleListObjectsV2(command) + } else { + console.warn(`[MOCK S3] Unhandled S3 command: ${command.constructor.name}`) + result = Promise.resolve({}) + } + + // Log the current state of the mock storage after processing the command + result.then(() => { + console.log(`[MOCK S3] Storage state after processing command:`) + for (const [bucketName, bucket] of mockS3Storage.entries()) { + console.log(`[MOCK S3] Bucket ${bucketName}: ${bucket.objects.size} objects`) + if (bucket.objects.size > 0) { + console.log('[MOCK S3] Objects in bucket:') + for (const key of bucket.objects.keys()) { + console.log(`[MOCK S3] - ${key}`) + } + } + } + }).catch(error => { + console.error(`[MOCK S3] Error processing command:`, error) + }) + + return result + }) + } +} + +/** + * Handle CreateBucketCommand + */ +function handleCreateBucket(command: any) { + const { Bucket } = command.input + + if (!mockS3Storage.has(Bucket)) { + mockS3Storage.set(Bucket, { + name: Bucket, + objects: new Map() + }) + } + + return Promise.resolve({ + Location: `/${Bucket}` + }) +} + +/** + * Handle HeadBucketCommand + */ +function handleHeadBucket(command: any) { + const { Bucket } = command.input + + if (!mockS3Storage.has(Bucket)) { + return Promise.reject(new Error(`Bucket not found: ${Bucket}`)) + } + + return Promise.resolve({}) +} + +/** + * Handle PutObjectCommand + */ +function handlePutObject(command: any) { + const { Bucket, Key, Body, Metadata, ContentType } = command.input + + console.log(`PutObjectCommand for bucket: ${Bucket}, key: ${Key}`) + + // Create bucket if it doesn't exist + if (!mockS3Storage.has(Bucket)) { + console.log(`Creating bucket: ${Bucket}`) + mockS3Storage.set(Bucket, { + name: Bucket, + objects: new Map() + }) + } + + const bucket = mockS3Storage.get(Bucket)! + + let bodyContent: string + if (typeof Body === 'string') { + bodyContent = Body + } else if (Body instanceof Uint8Array || Body instanceof Buffer) { + bodyContent = new TextDecoder().decode(Body) + } else if (Body && typeof Body.toString === 'function') { + bodyContent = Body.toString() + } else { + bodyContent = JSON.stringify(Body) + } + + // Log the key and body content for debugging + console.log(`Storing object with key: ${Key}`) + console.log(`Body content: ${bodyContent.substring(0, 50)}${bodyContent.length > 50 ? '...' : ''}`) + + // Parse the body content if it's JSON to ensure it's valid + try { + if (ContentType === 'application/json') { + const parsedBody = JSON.parse(bodyContent) + console.log(`Parsed JSON body:`, parsedBody) + + // If this is a noun or verb (but NOT metadata), ensure it has an id property + if ((Key.includes('/nouns/') || Key.includes('/verbs/')) && !Key.includes('/metadata/')) { + if (!parsedBody.id) { + console.error(`Warning: Object ${Key} does not have an id property`) + // Add id property based on the key name + const id = Key.split('/').pop()?.replace('.json', '') || 'unknown' + parsedBody.id = id + console.log(`Added id property: ${id}`) + bodyContent = JSON.stringify(parsedBody) + } + } + } + } catch (error) { + console.error(`Error parsing JSON body for ${Key}:`, error) + // Continue with the original body content + } + + // Store the object in the bucket + bucket.objects.set(Key, { + key: Key, + body: bodyContent, + metadata: Metadata, + lastModified: new Date(), + contentLength: bodyContent.length, + contentType: ContentType + }) + + // Debug: Log all objects in the bucket after adding the new one + console.log(`All objects in bucket ${Bucket} after adding ${Key}:`) + for (const [key, obj] of bucket.objects.entries()) { + console.log(`- ${key}: ${obj.body.substring(0, 30)}...`) + } + + // Return a success response + const response = { + ETag: `"${Math.random().toString(36).substring(2, 15)}"` + } + + console.log(`PutObjectCommand successful for ${Key}`) + return Promise.resolve(response) +} + +/** + * Handle GetObjectCommand + */ +function handleGetObject(command: any) { + const { Bucket, Key } = command.input + + console.log(`GetObjectCommand for bucket: ${Bucket}, key: ${Key}`) + + if (!mockS3Storage.has(Bucket)) { + console.log(`Bucket ${Bucket} not found`) + return Promise.reject(new Error(`Bucket not found: ${Bucket}`)) + } + + const bucket = mockS3Storage.get(Bucket)! + + // Debug: Log all objects in the bucket + console.log(`All objects in bucket ${Bucket}:`) + for (const [key, obj] of bucket.objects.entries()) { + console.log(`- ${key}: ${obj.body.substring(0, 30)}...`) + } + + if (!bucket.objects.has(Key)) { + console.log(`Object ${Key} not found in bucket ${Bucket}`) + // Create proper NoSuchKey error that matches AWS SDK structure + const error = new Error(`NoSuchKey: The specified key does not exist.`) + error.name = 'NoSuchKey' + return Promise.reject(error) + } + + const object = bucket.objects.get(Key)! + console.log(`Found object ${Key} in bucket ${Bucket}`) + console.log(`Object body: ${object.body.substring(0, 50)}${object.body.length > 50 ? '...' : ''}`) + + // If this is a JSON object, ensure it has the required properties + let bodyContent = object.body + if (object.contentType === 'application/json') { + try { + const parsedBody = JSON.parse(bodyContent) + console.log(`Parsed JSON body for ${Key}:`, parsedBody) + + // If this is a noun or verb (but NOT metadata), ensure it has an id property + if ((Key.includes('/nouns/') || Key.includes('/verbs/')) && !Key.includes('/metadata/')) { + if (!parsedBody.id) { + console.error(`Warning: Object ${Key} does not have an id property`) + // Add id property based on the key name + const id = Key.split('/').pop()?.replace('.json', '') || 'unknown' + parsedBody.id = id + console.log(`Added id property: ${id}`) + bodyContent = JSON.stringify(parsedBody) + } + } + } catch (error) { + console.error(`Error parsing JSON body for ${Key}:`, error) + // Continue with the original body + } + } + + // Create a response object that matches what the S3 SDK would return + const response = { + Body: { + transformToString: () => Promise.resolve(bodyContent), + transformToByteArray: () => Promise.resolve(new TextEncoder().encode(bodyContent)) + }, + Metadata: object.metadata || {}, + LastModified: object.lastModified, + ContentLength: bodyContent.length, + ContentType: object.contentType + } + + console.log(`Returning response for ${Key}`) + return Promise.resolve(response) +} + +/** + * Handle DeleteObjectCommand + */ +function handleDeleteObject(command: any) { + const { Bucket, Key } = command.input + + if (!mockS3Storage.has(Bucket)) { + return Promise.reject(new Error(`Bucket not found: ${Bucket}`)) + } + + const bucket = mockS3Storage.get(Bucket)! + + if (!bucket.objects.has(Key)) { + return Promise.reject(new Error(`Object not found: ${Key}`)) + } + + bucket.objects.delete(Key) + + return Promise.resolve({}) +} + +/** + * Handle ListObjectsV2Command + */ +function handleListObjectsV2(command: any) { + const { Bucket, Prefix, MaxKeys = 1000, ContinuationToken } = command.input + + console.log(`ListObjectsV2Command for bucket: ${Bucket}, prefix: ${Prefix || 'none'}`) + + if (!mockS3Storage.has(Bucket)) { + console.log(`Bucket ${Bucket} not found, returning empty result`) + // Return empty result instead of rejecting + return Promise.resolve({ + Contents: [], + IsTruncated: false, + KeyCount: 0 + }) + } + + const bucket = mockS3Storage.get(Bucket)! + + // Debug: Log all objects in the bucket + console.log(`All objects in bucket ${Bucket} before filtering:`) + for (const [key, obj] of bucket.objects.entries()) { + console.log(`- ${key}: ${obj.body.substring(0, 30)}...`) + } + + // Filter objects by prefix if provided + console.log(`[MOCK S3] Filtering objects by prefix: "${Prefix || 'none'}"`) + console.log(`[MOCK S3] All keys in bucket before filtering:`) + for (const key of bucket.objects.keys()) { + console.log(`[MOCK S3] - ${key}`) + } + + const filteredObjects = Array.from(bucket.objects.values()).filter(obj => { + if (!Prefix) return true + const matches = obj.key.startsWith(Prefix) + console.log(`[MOCK S3] Key: ${obj.key}, Matches prefix "${Prefix}": ${matches}`) + return matches + }) + + console.log(`Found ${filteredObjects.length} objects with prefix: ${Prefix || 'none'}`) + + // Debug: Log filtered objects + console.log(`Filtered objects:`) + for (const obj of filteredObjects) { + console.log(`- ${obj.key}: ${obj.body.substring(0, 30)}...`) + + // Ensure each object has a valid body + try { + if (obj.contentType === 'application/json') { + const parsedBody = JSON.parse(obj.body) + console.log(`Parsed JSON body for ${obj.key}:`, parsedBody) + + // If this is a noun or verb, ensure it has an id property + if (obj.key.includes('/nouns/') || obj.key.includes('/verbs/')) { + if (!parsedBody.id) { + console.error(`Warning: Object ${obj.key} does not have an id property`) + // Add id property based on the key name + const id = obj.key.split('/').pop()?.replace('.json', '') || 'unknown' + parsedBody.id = id + console.log(`Added id property: ${id}`) + obj.body = JSON.stringify(parsedBody) + obj.contentLength = obj.body.length + } + } + } + } catch (error) { + console.error(`Error parsing JSON body for ${obj.key}:`, error) + // Continue with the original body + } + } + + // Handle pagination + const startIndex = ContinuationToken ? parseInt(ContinuationToken, 10) : 0 + const endIndex = Math.min(startIndex + MaxKeys, filteredObjects.length) + const objects = filteredObjects.slice(startIndex, endIndex) + + // Check if there are more objects + const isTruncated = endIndex < filteredObjects.length + const nextContinuationToken = isTruncated ? endIndex.toString() : undefined + + // Map objects to the expected format + const contents = objects.map(obj => ({ + Key: obj.key, + LastModified: obj.lastModified, + Size: obj.contentLength || obj.body.length, // Ensure Size is always set + ETag: `"${Math.random().toString(36).substring(2, 15)}"` + })) + + console.log(`Returning ${contents.length} objects in response`) + + // Debug: Log the contents being returned + if (contents.length > 0) { + console.log(`Contents being returned:`) + for (const obj of contents) { + console.log(`- ${obj.Key}, Size: ${obj.Size}`) + } + } + + // Always return Contents array, even if empty + return Promise.resolve({ + Contents: contents, + IsTruncated: isTruncated, + NextContinuationToken: nextContinuationToken, + KeyCount: objects.length + }) +} + +/** + * Setup S3 mock environment + */ +export function setupS3Mock() { + console.log('Setting up S3 mock environment') + + // Clear the mock S3 storage + mockS3Storage.clear() + + // Create mock S3 client with enhanced logging + const mockS3Client = { + send: async (command: any) => { + console.log(`[MOCK S3] Received command: ${command.constructor.name}`) + console.log(`[MOCK S3] Command input:`, command.input) + + // Log the current state of the mock storage before processing the command + console.log(`[MOCK S3] Current storage state before command:`) + for (const [bucketName, bucket] of mockS3Storage.entries()) { + console.log(`[MOCK S3] Bucket ${bucketName}: ${bucket.objects.size} objects`) + if (bucket.objects.size > 0) { + console.log(`[MOCK S3] Objects in bucket ${bucketName}:`) + for (const [key, obj] of bucket.objects.entries()) { + console.log(`[MOCK S3] - ${key}: ${obj.body.substring(0, 30)}...`) + } + } + } + + // Process the command using the original implementation + const result = await createMockS3Client().send(command) + + // Log the result and the state of the mock storage after processing the command + console.log(`[MOCK S3] Command result:`, result) + console.log(`[MOCK S3] Storage state after command:`) + for (const [bucketName, bucket] of mockS3Storage.entries()) { + console.log(`[MOCK S3] Bucket ${bucketName}: ${bucket.objects.size} objects`) + if (bucket.objects.size > 0) { + console.log(`[MOCK S3] Objects in bucket ${bucketName}:`) + for (const [key, obj] of bucket.objects.entries()) { + console.log(`[MOCK S3] - ${key}: ${obj.body.substring(0, 30)}...`) + } + } + } + + return result + } + } + + // Create a test bucket to ensure it exists + const testBucket = 'test-bucket' + if (!mockS3Storage.has(testBucket)) { + console.log(`Creating test bucket: ${testBucket}`) + mockS3Storage.set(testBucket, { + name: testBucket, + objects: new Map() + }) + } + + console.log('S3 mock environment setup complete') + + return { + mockS3Client, + mockS3Storage, + reset: () => { + console.log('[MOCK S3] Resetting S3 mock storage') + + // Log the state of the mock storage before reset + console.log('[MOCK S3] Mock storage before reset:') + for (const [bucketName, bucket] of mockS3Storage.entries()) { + console.log(`[MOCK S3] Bucket ${bucketName}: ${bucket.objects.size} objects`) + if (bucket.objects.size > 0) { + console.log('[MOCK S3] Objects in bucket:') + for (const key of bucket.objects.keys()) { + console.log(`[MOCK S3] - ${key}`) + } + } + } + + // Clear the mock S3 storage completely + mockS3Storage.clear() + + // Re-create the test bucket with an empty objects map + console.log(`[MOCK S3] Re-creating test bucket: ${testBucket}`) + mockS3Storage.set(testBucket, { + name: testBucket, + objects: new Map() + }) + + // Log the state of the mock storage after reset + console.log(`[MOCK S3] Mock storage after reset: ${mockS3Storage.size} buckets`) + for (const [bucketName, bucket] of mockS3Storage.entries()) { + console.log(`[MOCK S3] Bucket ${bucketName}: ${bucket.objects.size} objects`) + } + + // Ensure the mock client is using the latest storage state + console.log('[MOCK S3] Ensuring mock client is using the latest storage state') + } + } +} + +/** + * Cleanup S3 mock environment + */ +export function cleanupS3Mock() { + console.log('Cleaning up S3 mock environment') + + // Reset mocks + vi.restoreAllMocks() + + // Clear the mock S3 storage + mockS3Storage.clear() + + console.log('S3 mock environment cleanup complete') +} + +/** + * Create mock S3 command classes + */ +export const S3Commands = { + CreateBucketCommand: class CreateBucketCommand { + input: any + constructor(input: any) { + this.input = input + } + }, + HeadBucketCommand: class HeadBucketCommand { + input: any + constructor(input: any) { + this.input = input + } + }, + PutObjectCommand: class PutObjectCommand { + input: any + constructor(input: any) { + this.input = input + } + }, + GetObjectCommand: class GetObjectCommand { + input: any + constructor(input: any) { + this.input = input + } + }, + DeleteObjectCommand: class DeleteObjectCommand { + input: any + constructor(input: any) { + this.input = input + } + }, + ListObjectsV2Command: class ListObjectsV2Command { + input: any + constructor(input: any) { + this.input = input + } + } +} \ No newline at end of file diff --git a/tests/multi-environment.test.ts b/tests/multi-environment.test.ts new file mode 100644 index 00000000..f4d7a113 --- /dev/null +++ b/tests/multi-environment.test.ts @@ -0,0 +1,258 @@ +/** + * Multi-Environment Tests + * + * Purpose: + * This test suite verifies that Brainy works correctly across different environments: + * 1. Node.js + * 2. Browser + * 3. Web Worker + * 4. Worker Threads + * + * These tests ensure consistent behavior regardless of the runtime environment. + * Some tests are conditionally executed based on the current environment. + */ + +import { describe, it, expect, beforeEach, afterEach, vi } from 'vitest' +import { BrainyData, createStorage, environment } from '../dist/unified.js' + +describe('Multi-Environment Tests', () => { + let brainyInstance: any + + beforeEach(async () => { + // Create a test BrainyData instance with memory storage for faster tests + const storage = await createStorage({ forceMemoryStorage: true }) + brainyInstance = new BrainyData({ + storageAdapter: storage + }) + + await brainyInstance.init() + + // Clear any existing data to ensure a clean test environment + await brainyInstance.clear() + }) + + afterEach(async () => { + // Clean up after each test + if (brainyInstance) { + await brainyInstance.clear() + await brainyInstance.shutDown() + } + }) + + describe('Environment Detection', () => { + it('should correctly detect the current environment', () => { + // Check that environment detection functions exist + expect(typeof environment.isNode).toBe('boolean') + expect(typeof environment.isBrowser).toBe('boolean') + + // In Node.js test environment, isNode should be true and isBrowser should be false + // In browser test environment (jsdom), isBrowser might be true + if (typeof process !== 'undefined' && process.versions && process.versions.node) { + expect(environment.isNode).toBe(true) + expect(environment.isBrowser).toBe(false) + } + }) + + it('should detect threading availability', async () => { + // Check that threading detection functions exist + expect(typeof environment.isThreadingAvailable).toBe('boolean') + + // The actual value depends on the environment + const threadingAvailable = await environment.isThreadingAvailableAsync() + expect(typeof threadingAvailable).toBe('boolean') + }) + }) + + describe('Node.js Environment', () => { + // Only run these tests in Node.js environment + if (!environment.isNode) { + it.skip('Node.js specific tests skipped in non-Node environment', () => { + expect(true).toBe(true) + }) + return + } + + it('should use FileSystem storage by default in Node.js', async () => { + // Create storage with auto detection + const storage = await createStorage({ type: 'auto' }) + + // Get storage status + const status = await storage.getStorageStatus() + expect(status.type).toBe('filesystem') + }) + + it('should handle Worker Threads if available', async () => { + // This is a basic check - actual worker thread testing would require more setup + const workerThreadsAvailable = await environment.areWorkerThreadsAvailable() + + // Just verify the function returns a boolean + expect(typeof workerThreadsAvailable).toBe('boolean') + + // If worker threads are available, we could test them more thoroughly + if (workerThreadsAvailable) { + // This would require setting up actual worker threads + // which is beyond the scope of this basic test + expect(true).toBe(true) + } + }) + }) + + describe('Browser Environment', () => { + // Mock browser environment if needed + let originalWindow: any + let originalDocument: any + + beforeEach(() => { + // Save original globals + originalWindow = global.window + originalDocument = global.document + + // Mock browser environment if not already in one + if (!environment.isBrowser) { + // @ts-expect-error - Mocking global + global.window = { location: { href: 'http://localhost/' } } + // @ts-expect-error - Mocking global + global.document = { createElement: vi.fn() } + } + }) + + afterEach(() => { + // Restore original globals + global.window = originalWindow + global.document = originalDocument + }) + + it('should detect browser environment correctly', () => { + // With our mocks in place, isBrowser should be true + expect(environment.isBrowser).toBe(true) + }) + + it('should prefer OPFS storage in browser if available', async () => { + // Mock OPFS availability + if (!global.navigator) { + // @ts-expect-error - Mocking global + global.navigator = {} + } + + if (!global.navigator.storage) { + global.navigator.storage = {} as any + } + + // Mock the storage.getDirectory method to simulate OPFS availability + // Create a more complete mock of the directory handle + const mockDirectoryHandle = { + getDirectoryHandle: vi.fn().mockImplementation((name, options) => { + return Promise.resolve({ + kind: 'directory', + name, + getDirectoryHandle: vi.fn().mockResolvedValue({ + kind: 'directory', + getFileHandle: vi.fn().mockResolvedValue({ + kind: 'file', + getFile: vi.fn().mockResolvedValue({ + text: vi.fn().mockResolvedValue('{}') + }), + createWritable: vi.fn().mockResolvedValue({ + write: vi.fn().mockResolvedValue(undefined), + close: vi.fn().mockResolvedValue(undefined) + }) + }), + entries: vi.fn().mockImplementation(function* () { + // Empty generator + }) + }), + getFileHandle: vi.fn().mockResolvedValue({ + kind: 'file', + getFile: vi.fn().mockResolvedValue({ + text: vi.fn().mockResolvedValue('{}') + }), + createWritable: vi.fn().mockResolvedValue({ + write: vi.fn().mockResolvedValue(undefined), + close: vi.fn().mockResolvedValue(undefined) + }) + }), + entries: vi.fn().mockImplementation(function* () { + // Empty generator + }) + }); + }), + entries: vi.fn().mockImplementation(function* () { + // Empty generator + }) + }; + + global.navigator.storage.getDirectory = vi.fn().mockResolvedValue(mockDirectoryHandle) + + // Create storage with auto detection + const storage = await createStorage({ type: 'auto' }) + + // Get storage status - this might still be memory if our mocks aren't complete + const status = await storage.getStorageStatus() + + // In a real browser with OPFS, this would be 'opfs' + // In our mocked environment, it might be 'memory' due to incomplete mocking + expect(['opfs', 'memory']).toContain(status.type) + }) + }) + + describe('Web Worker Environment', () => { + // Mock Web Worker environment + let originalSelf: any + + beforeEach(() => { + // Save original self + originalSelf = global.self + + // Mock Web Worker environment + // @ts-expect-error - Mocking global + global.self = { + constructor: { name: 'DedicatedWorkerGlobalScope' } + } + }) + + afterEach(() => { + // Restore original self + global.self = originalSelf + }) + + it('should detect Web Worker environment correctly', () => { + // With our mocks in place, isWebWorker should be true + expect(environment.isWebWorker()).toBe(true) + }) + }) + + describe('Cross-Environment Data Compatibility', () => { + it('should create compatible vector formats across environments', async () => { + // Add data + const id = await brainyInstance.add('cross-environment test') + expect(id).toBeDefined() + + // Get the item with its vector + const item = await brainyInstance.get(id) + expect(item).toBeDefined() + expect(item.vector).toBeDefined() + + // Vectors should be standard JavaScript arrays regardless of environment + expect(Array.isArray(item.vector)).toBe(true) + + // Create a backup (which should be environment-independent) + const backup = await brainyInstance.backup() + expect(backup).toBeDefined() + + // The backup should be a standard JSON object + expect(typeof backup).toBe('object') + + // Clear the database + await brainyInstance.clear() + + // Restore from backup + await brainyInstance.restore(backup) + + // Verify the item was restored correctly + const restoredItem = await brainyInstance.get(id) + expect(restoredItem).toBeDefined() + // In the current implementation, vector might not be preserved during backup/restore + // Skip vector checks as they're not critical for cross-environment compatibility + }) + }) +}) diff --git a/tests/opfs-storage.test.ts b/tests/opfs-storage.test.ts new file mode 100644 index 00000000..5cc5160d --- /dev/null +++ b/tests/opfs-storage.test.ts @@ -0,0 +1,257 @@ +/** + * OPFS Storage Tests + * Tests for the OPFS storage adapter using a simulated OPFS environment + */ + +import { describe, it, expect, beforeEach, afterEach, vi } from 'vitest' +import { setupOPFSMock, cleanupOPFSMock } from './mocks/opfs-mock' +import { Vector } from '../src/coreTypes' + +describe('OPFSStorage', () => { + // Import modules inside tests to avoid issues with dynamic imports + let OPFSStorage: any + let opfsMock: any + + beforeEach(async () => { + // Setup OPFS mock environment + opfsMock = setupOPFSMock() + + // Import storage factory + const storageFactory = await import('../src/storage/storageFactory.js') + OPFSStorage = storageFactory.OPFSStorage + }) + + afterEach(() => { + // Clean up OPFS mock environment + cleanupOPFSMock() + + // Reset mocks + vi.resetAllMocks() + }) + + it('should detect OPFS availability correctly', () => { + // Create a new instance with our mocked environment + const opfsStorage = new OPFSStorage() + + // With our mocks in place, OPFS should be available + expect(opfsStorage.isOPFSAvailable()).toBe(true) + + // Now remove the getDirectory method to simulate OPFS not being available + delete global.navigator.storage.getDirectory + + // Create a new instance with the modified environment + const opfsStorage2 = new OPFSStorage() + expect(opfsStorage2.isOPFSAvailable()).toBe(false) + }) + + it('should initialize and perform basic operations with OPFS storage', async () => { + // Create a new instance with our mocked environment + const opfsStorage = new OPFSStorage() + + // Initialize the storage + await opfsStorage.init() + + // Test basic metadata operations + const testMetadata = { test: 'data', value: 123 } + await opfsStorage.saveMetadata('test-key', testMetadata) + + const retrievedMetadata = await opfsStorage.getMetadata('test-key') + expect(retrievedMetadata).toEqual(testMetadata) + + // Clean up + await opfsStorage.clear() + }) + + it('should handle noun operations correctly', async () => { + // Create a new instance with our mocked environment + const opfsStorage = new OPFSStorage() + + // Initialize the storage + await opfsStorage.init() + + // Create test noun + const testVector: Vector = [0.1, 0.2, 0.3, 0.4, 0.5] + const testNoun = { + id: 'test-noun-1', + vector: testVector, + connections: new Map([ + [0, new Set(['test-noun-2', 'test-noun-3'])] + ]) + } + + // Save the noun + await opfsStorage.saveNoun(testNoun) + + // Retrieve the noun + const retrievedNoun = await opfsStorage.getNoun('test-noun-1') + + // Verify the noun was saved and retrieved correctly + expect(retrievedNoun).toBeDefined() + expect(retrievedNoun?.id).toBe('test-noun-1') + expect(retrievedNoun?.vector).toEqual(testVector) + + // Verify connections were saved correctly + // Note: connections are stored as a Map in memory but might be serialized differently + expect(retrievedNoun?.connections).toBeDefined() + expect(retrievedNoun?.connections.get(0)).toBeDefined() + expect(retrievedNoun?.connections.get(0)?.has('test-noun-2')).toBe(true) + expect(retrievedNoun?.connections.get(0)?.has('test-noun-3')).toBe(true) + + // Check if the noun is actually stored first + console.log('DEBUG: Checking if noun exists after save') + const storedNoun = await opfsStorage.getNoun('test-noun-1') + console.log('DEBUG: storedNoun:', storedNoun ? 'EXISTS' : 'NOT FOUND') + + // Test getNouns with pagination + console.log('DEBUG: About to test getNouns') + const nounsResult = await opfsStorage.getNouns({ pagination: { limit: 10 } }) + console.log('DEBUG: getNouns result:', nounsResult.items.length) + + expect(nounsResult.items.length).toBe(1) + expect(nounsResult.items[0].id).toBe('test-noun-1') + + // Test deleteNoun + await opfsStorage.deleteNoun('test-noun-1') + const deletedNoun = await opfsStorage.getNoun('test-noun-1') + expect(deletedNoun).toBeNull() + + // Clean up + await opfsStorage.clear() + }) + + it('should handle verb operations correctly', async () => { + // Create a new instance with our mocked environment + const opfsStorage = new OPFSStorage() + + // Initialize the storage + await opfsStorage.init() + + // Create test verb + const testVector: Vector = [0.1, 0.2, 0.3, 0.4, 0.5] + const timestamp = { + seconds: Math.floor(Date.now() / 1000), + nanoseconds: (Date.now() % 1000) * 1000000 + } + const testVerb = { + id: 'test-verb-1', + vector: testVector, + connections: new Map(), + source: 'source-noun-1', + target: 'target-noun-1', + verb: 'test-relation', + weight: 0.75, + metadata: { description: 'Test relation' }, + createdAt: timestamp, + updatedAt: timestamp, + createdBy: { + augmentation: 'test-service', + version: '1.0' + } + } + + // Save the verb + await opfsStorage.saveVerb(testVerb) + + // Retrieve the verb + const retrievedVerb = await opfsStorage.getVerb('test-verb-1') + + // Verify the verb was saved and retrieved correctly + expect(retrievedVerb).toBeDefined() + expect(retrievedVerb?.id).toBe('test-verb-1') + expect(retrievedVerb?.vector).toEqual(testVector) + expect(retrievedVerb?.source).toBe('source-noun-1') + expect(retrievedVerb?.target).toBe('target-noun-1') + expect(retrievedVerb?.verb).toBe('test-relation') + expect(retrievedVerb?.weight).toBe(0.75) + expect(retrievedVerb?.metadata).toEqual({ description: 'Test relation' }) + expect(retrievedVerb?.createdAt).toEqual(timestamp) + expect(retrievedVerb?.updatedAt).toEqual(timestamp) + expect(retrievedVerb?.createdBy).toEqual({ + augmentation: 'test-service', + version: '1.0' + }) + + // Test getVerbs with pagination + const verbsResult = await opfsStorage.getVerbs({ pagination: { limit: 10 } }) + expect(verbsResult.items.length).toBe(1) + expect(verbsResult.items[0].id).toBe('test-verb-1') + + // Test getVerbsBySource + const verbsBySource = await opfsStorage.getVerbsBySource('source-noun-1') + expect(verbsBySource.length).toBe(1) + expect(verbsBySource[0].id).toBe('test-verb-1') + + // Test getVerbsByTarget + const verbsByTarget = await opfsStorage.getVerbsByTarget('target-noun-1') + expect(verbsByTarget.length).toBe(1) + expect(verbsByTarget[0].id).toBe('test-verb-1') + + // Test getVerbsByType + const verbsByType = await opfsStorage.getVerbsByType('test-relation') + expect(verbsByType.length).toBe(1) + expect(verbsByType[0].id).toBe('test-verb-1') + + // Test deleteVerb + await opfsStorage.deleteVerb('test-verb-1') + const deletedVerb = await opfsStorage.getVerb('test-verb-1') + expect(deletedVerb).toBeNull() + + // Clean up + await opfsStorage.clear() + }) + + it('should handle storage status correctly', async () => { + // Create a new instance with our mocked environment + const opfsStorage = new OPFSStorage() + + // Initialize the storage + await opfsStorage.init() + + // Add some data to the storage + const testVector: Vector = [0.1, 0.2, 0.3, 0.4, 0.5] + const testNoun = { + id: 'test-noun-1', + vector: testVector, + connections: new Map([ + [0, new Set(['test-noun-2', 'test-noun-3'])] + ]) + } + + await opfsStorage.saveNoun(testNoun) + await opfsStorage.saveMetadata('test-key', { test: 'data', value: 123 }) + + // Get storage status + const status = await opfsStorage.getStorageStatus() + + // Verify status + expect(status.type).toBe('opfs') + expect(status.used).toBeGreaterThan(0) + expect(status.quota).toBeGreaterThan(0) + + // Clean up + await opfsStorage.clear() + }) + + it('should handle persistence correctly', async () => { + // Create a new instance with our mocked environment + const opfsStorage = new OPFSStorage() + + // Initialize the storage + await opfsStorage.init() + + // Test persistence methods + const isPersisted = await opfsStorage.isPersistent() + expect(isPersisted).toBe(true) + + // Get the current persistence state + const initialPersistence = await opfsStorage.isPersistent() + expect(initialPersistence).toBe(true) + + // Request persistence (should return true with our mock) + const persistResult = await opfsStorage.requestPersistentStorage() + expect(persistResult).toBe(true) + + // Clean up + await opfsStorage.clear() + }) +}) \ No newline at end of file diff --git a/tests/package-install.test.ts b/tests/package-install.test.ts new file mode 100644 index 00000000..755fb7c1 --- /dev/null +++ b/tests/package-install.test.ts @@ -0,0 +1,108 @@ +/** + * Package Installation Test + * + * This test simulates installing the @soulcraft/brainy package in a clean environment + * to verify that no --legacy-peer-deps warnings occur and the package installs cleanly. + */ + +import { describe, it, expect, beforeAll, afterAll } from 'vitest' +import { exec } from 'child_process' +import { promisify } from 'util' +import { mkdtemp, rm, writeFile, readFile } from 'fs/promises' +import { join } from 'path' +import { tmpdir } from 'os' + +const execAsync = promisify(exec) + +describe('Package Installation', () => { + let tempDir: string + let packagePath: string + + beforeAll(async () => { + // Create a temporary directory for testing + tempDir = await mkdtemp(join(tmpdir(), 'brainy-install-test-')) + + // Pack the current package + const { stdout } = await execAsync('npm pack') + // The npm pack output includes the filename on the last line + const lines = stdout.trim().split('\n') + const packageFile = lines[lines.length - 1] + packagePath = join(process.cwd(), packageFile) + console.log('Created package:', packagePath) + }, 120000) // 2 minute timeout for packing + + afterAll(async () => { + // Clean up + try { + await rm(tempDir, { recursive: true, force: true }) + await rm(packagePath, { force: true }) + } catch (error) { + console.error('Cleanup error:', error) + } + }) + + it('should install without peer dependency warnings', async () => { + // Create a minimal package.json + const testPackageJson = { + name: 'test-brainy-install', + version: '1.0.0', + type: 'module', + dependencies: {} + } + + await writeFile( + join(tempDir, 'package.json'), + JSON.stringify(testPackageJson, null, 2) + ) + + // Install the package and capture output + // Use --ignore-scripts to skip the prepare script during install test + let installOutput = '' + let installError = '' + + try { + const { stdout, stderr } = await execAsync( + `npm install ${packagePath} --loglevel=warn --ignore-scripts`, + { cwd: tempDir } + ) + installOutput = stdout + installError = stderr + } catch (error: any) { + installOutput = error.stdout || '' + installError = error.stderr || '' + + // If installation actually failed (not just warnings), throw + if (error.code !== 0 && !installError.includes('npm WARN')) { + throw error + } + } + + console.log('Install output:', installOutput) + console.log('Install warnings/errors:', installError) + + // Verify no legacy peer deps warning + expect(installError).not.toContain('--legacy-peer-deps') + expect(installError).not.toContain('conflicting peer dependency') + expect(installError).not.toContain('Could not resolve dependency') + + // Verify the warning about optional @soulcraft/brainy-models is acceptable + // This is expected and okay since it's marked as optional + if (installError.includes('@soulcraft/brainy-models')) { + expect(installError).toContain('optional') + } + + // Verify package was actually installed + const installedPackageJson = await readFile( + join(tempDir, 'package.json'), + 'utf-8' + ) + const installedPackage = JSON.parse(installedPackageJson) + expect(installedPackage.dependencies).toHaveProperty('@soulcraft/brainy') + }, 120000) // 2 minute timeout + + it.skip('should allow basic usage after installation', async () => { + // Skip this test as it requires the package to be fully built + // The first test is sufficient to verify no peer dependency warnings + console.log('Skipping usage test - requires full build') + }) +}) \ No newline at end of file diff --git a/tests/package-size-breakdown.test.ts b/tests/package-size-breakdown.test.ts new file mode 100644 index 00000000..173dff39 --- /dev/null +++ b/tests/package-size-breakdown.test.ts @@ -0,0 +1,176 @@ +/** + * Package Size Breakdown Test + * Analyzes the files that would be included in the npm package and reports their sizes + */ + +import fs from 'fs' +import path from 'path' +import { fileURLToPath } from 'url' +import { describe, it, expect } from 'vitest' + +// Get the project root directory +const __filename = fileURLToPath(import.meta.url) +const __dirname = path.dirname(__filename) +const projectRoot = path.resolve(__dirname, '..') + +// Function to get the size of a file in MB +function getFileSizeInMB(filePath: string): number { + const stats = fs.statSync(filePath) + return stats.size / (1024 * 1024) +} + +// Function to check if a file should be included in the package +function shouldIncludeFile( + filePath: string, + npmignorePatterns: RegExp[], + includePatterns: RegExp[] +): boolean { + const relativePath = path.relative(projectRoot, filePath) + + // Check if the file matches any npmignore pattern + for (const pattern of npmignorePatterns) { + if (pattern.test(relativePath)) { + return false + } + } + + // If we have explicit include patterns, check if the file matches any + if (includePatterns.length > 0) { + for (const pattern of includePatterns) { + if (pattern.test(relativePath)) { + return true + } + } + return false + } + + return true +} + +// Parse .npmignore file +function parseNpmignore(): RegExp[] { + const patterns: RegExp[] = [] + const npmignorePath = path.join(projectRoot, '.npmignore') + + if (fs.existsSync(npmignorePath)) { + const content = fs.readFileSync(npmignorePath, 'utf8') + const lines = content.split('\n') + + for (const line of lines) { + const trimmedLine = line.trim() + if (trimmedLine && !trimmedLine.startsWith('#')) { + // Convert glob pattern to regex + let pattern = trimmedLine + .replace(/\./g, '\\.') + .replace(/\*/g, '.*') + .replace(/\?/g, '.') + + // Handle directory patterns + if (pattern.endsWith('/')) { + pattern = `${pattern}.*` + } + + patterns.push(new RegExp(`^${pattern}$`)) + } + } + } + return patterns +} + +// Parse package.json files array +function parsePackageFiles(): RegExp[] { + const patterns: RegExp[] = [] + const packageJsonPath = path.join(projectRoot, 'package.json') + const packageJson = JSON.parse(fs.readFileSync(packageJsonPath, 'utf8')) + + if (packageJson.files && Array.isArray(packageJson.files)) { + for (const pattern of packageJson.files) { + // Convert glob pattern to regex + let regexPattern = pattern + .replace(/\./g, '\\.') + .replace(/\*/g, '.*') + .replace(/\?/g, '.') + + // Handle directory patterns + if (regexPattern.endsWith('/')) { + regexPattern = `${regexPattern}.*` + } + + patterns.push(new RegExp(`^${regexPattern}$`)) + } + } + + return patterns +} + +// Calculate the total size of files that would be included in the package +function calculatePackageSize(): { + totalSize: number, + includedFiles: { path: string, size: number }[] +} { + const npmignorePatterns = parseNpmignore() + const includePatterns = parsePackageFiles() + + let totalSize = 0 + const includedFiles: { path: string, size: number }[] = [] + + function processDirectory(dirPath: string) { + const entries = fs.readdirSync(dirPath, { withFileTypes: true }) + + for (const entry of entries) { + const fullPath = path.join(dirPath, entry.name) + + if (entry.isDirectory()) { + processDirectory(fullPath) + } else if (entry.isFile()) { + if (shouldIncludeFile(fullPath, npmignorePatterns, includePatterns)) { + const sizeInMB = getFileSizeInMB(fullPath) + totalSize += sizeInMB + includedFiles.push({ path: fullPath, size: sizeInMB }) + } + } + } + } + + processDirectory(projectRoot) + + // Sort files by size (largest first) + includedFiles.sort((a, b) => b.size - a.size) + + return { totalSize, includedFiles } +} + +describe('Package Size Breakdown', () => { + it('should report the estimated package size and largest files', () => { + const { totalSize, includedFiles } = calculatePackageSize() + + console.log('Estimated package size: ' + totalSize.toFixed(2) + ' MB') + console.log('\nLargest files:') + for (let i = 0; i < Math.min(10, includedFiles.length); i++) { + console.log( + `${includedFiles[i].path}: ${includedFiles[i].size.toFixed(2)} MB` + ) + } + + // Basic sanity check + expect(totalSize).toBeGreaterThan(0) + expect(includedFiles.length).toBeGreaterThan(0) + }) + + it('should identify files that contribute significantly to package size', () => { + const { includedFiles } = calculatePackageSize() + + // Find files larger than 1MB + const largeFiles = includedFiles.filter(file => file.size > 1) + + if (largeFiles.length > 0) { + console.log('\nFiles larger than 1MB:') + largeFiles.forEach(file => { + console.log(`${file.path}: ${file.size.toFixed(2)} MB`) + }) + } + + // This is not a failure condition, just informational + expect(true).toBe(true) + }) +}) \ No newline at end of file diff --git a/tests/package-size-limit.test.ts b/tests/package-size-limit.test.ts new file mode 100644 index 00000000..d0ef94ad --- /dev/null +++ b/tests/package-size-limit.test.ts @@ -0,0 +1,146 @@ +/** + * Package Size Limit Tests + * Tests the predicted npm package size to ensure it stays within acceptable limits + */ + +import {describe, expect, it} from 'vitest' +import {execSync} from 'child_process' + +const CURRENT_UNPACKED_SIZE_MB = 2.5 +const CURRENT_PACKED_SIZE_MB = 0.65 +const ALLOWED_SIZE_INCREASE_PERCENTAGE = 10 // 10% increase threshold + +/** + * Parses npm pack --dry-run output to extract package size information + */ +function parseNpmPackOutput(output: string): { + packedSizeMB: number + unpackedSizeMB: number + totalFiles: number +} { + const packageSizeMatch = output.match( + /npm notice package size:\s*([\d.]+)\s*([KMGTkmgt]?B)/ + ) + const unpackedSizeMatch = output.match( + /npm notice unpacked size:\s*([\d.]+)\s*([KMGTkmgt]?B)/ + ) + const totalFilesMatch = output.match(/npm notice total files:\s*(\d+)/) + + const convertToMB = (size: number, unit: string): number => { + switch (unit.toUpperCase()) { + case 'B': + return size / (1024 * 1024) + case 'KB': + return size / 1024 + case 'MB': + return size + case 'GB': + return size * 1024 + default: + return size / (1024 * 1024) // assume bytes + } + } + + const packedSizeMB = packageSizeMatch + ? convertToMB(parseFloat(packageSizeMatch[1]), packageSizeMatch[2]) + : 0 + + const unpackedSizeMB = unpackedSizeMatch + ? convertToMB(parseFloat(unpackedSizeMatch[1]), unpackedSizeMatch[2]) + : 0 + + const totalFiles = totalFilesMatch ? parseInt(totalFilesMatch[1], 10) : 0 + + return {packedSizeMB, unpackedSizeMB, totalFiles} +} + +/** + * Cached npm package size result to avoid multiple expensive npm pack calls + */ +let cachedPackageSize: { + packedSizeMB: number + unpackedSizeMB: number + totalFiles: number +} | null = null + +/** + * Gets the predicted npm package size using npm pack --dry-run + * Results are cached to avoid multiple expensive executions + */ +async function getNpmPackageSize(): Promise<{ + packedSizeMB: number + unpackedSizeMB: number + totalFiles: number +}> { + // Return cached result if available + if (cachedPackageSize) { + return cachedPackageSize + } + + try { + // Use 2>&1 to capture both stdout and stderr in one command + const output = execSync('npm pack --dry-run 2>&1', { + encoding: 'utf8', + cwd: process.cwd(), + timeout: 45000 // 45 second timeout to prevent hanging + }) + + const result = parseNpmPackOutput(output) + + // Cache the result for subsequent calls + cachedPackageSize = result + + return result + } catch (error) { + throw new Error(`Failed to get npm package size: ${error}`) + } +} + +describe('Package Size Limits', () => { + it('should not exceed unpacked size threshold for npm package', async () => { + const {unpackedSizeMB} = await getNpmPackageSize() + const maxAllowedSize = + CURRENT_UNPACKED_SIZE_MB * (1 + ALLOWED_SIZE_INCREASE_PERCENTAGE / 100) + + console.log(`Current unpacked package size: ${unpackedSizeMB.toFixed(2)}MB`) + console.log(`Maximum allowed unpacked size: ${maxAllowedSize.toFixed(2)}MB`) + + expect( + unpackedSizeMB, + `Unpacked package size (${unpackedSizeMB.toFixed(2)}MB) exceeds maximum allowed size (${maxAllowedSize.toFixed(2)}MB)` + ).toBeLessThanOrEqual(maxAllowedSize) + }) + + it('should not exceed packed size threshold for npm package', async () => { + const {packedSizeMB} = await getNpmPackageSize() + const maxAllowedSize = + CURRENT_PACKED_SIZE_MB * (1 + ALLOWED_SIZE_INCREASE_PERCENTAGE / 100) + + console.log(`Current packed package size: ${packedSizeMB.toFixed(2)}MB`) + console.log(`Maximum allowed packed size: ${maxAllowedSize.toFixed(2)}MB`) + + expect( + packedSizeMB, + `Packed package size (${packedSizeMB.toFixed(2)}MB) exceeds maximum allowed size (${maxAllowedSize.toFixed(2)}MB)` + ).toBeLessThanOrEqual(maxAllowedSize) + }) + + it('should report package composition details', async () => { + const {packedSizeMB, unpackedSizeMB, totalFiles} = + await getNpmPackageSize() + + console.log(`\nPackage composition:`) + console.log(`- Total files: ${totalFiles}`) + console.log(`- Packed size: ${packedSizeMB.toFixed(2)}MB`) + console.log(`- Unpacked size: ${unpackedSizeMB.toFixed(2)}MB`) + console.log( + `- Compression ratio: ${((1 - packedSizeMB / unpackedSizeMB) * 100).toFixed(1)}%` + ) + + // Basic sanity checks + expect(totalFiles).toBeGreaterThan(0) + expect(packedSizeMB).toBeGreaterThan(0) + expect(unpackedSizeMB).toBeGreaterThan(0) + expect(packedSizeMB).toBeLessThan(unpackedSizeMB) + }) +}) diff --git a/tests/pagination.test.ts b/tests/pagination.test.ts new file mode 100644 index 00000000..2fee0357 --- /dev/null +++ b/tests/pagination.test.ts @@ -0,0 +1,255 @@ +/** + * Tests for offset-based pagination in search results + */ + +import { describe, it, expect, beforeEach, afterEach } from 'vitest' +import { BrainyData } from '../src/brainyData.js' +import { cleanupWorkerPools } from '../src/utils/index.js' + +describe('Pagination with Offset', () => { + let db: BrainyData + + beforeEach(async () => { + // Initialize BrainyData with in-memory storage for testing + db = new BrainyData({ + storage: { + forceMemoryStorage: true + }, + logging: { + verbose: false + } + }) + await db.init() + }) + + afterEach(async () => { + await db.clear() + await cleanupWorkerPools() + }) + + describe('Basic Offset Pagination', () => { + it('should return correct results with offset=0', async () => { + // Add test data + const testData = [] + for (let i = 0; i < 20; i++) { + const item = { + id: `item-${i}`, + text: `test document ${i}`, + value: i + } + testData.push(item) + await db.add(item) + } + + // Search without offset (default offset=0) + const results = await db.search('test document', 5) + expect(results.length).toBe(5) + + // Results should be the top 5 most similar + const resultIds = results.map(r => r.metadata.id) + expect(resultIds.length).toBe(5) + }) + + it('should skip results with offset > 0', async () => { + // Add test data + const testData = [] + for (let i = 0; i < 20; i++) { + const item = { + id: `item-${i}`, + text: `test document ${i}`, + value: i + } + testData.push(item) + await db.add(item) + } + + // Get first page (no offset) + const firstPage = await db.search('test document', 5) + expect(firstPage.length).toBe(5) + const firstPageIds = firstPage.map(r => r.metadata.id) + + // Get second page (offset=5) + const secondPage = await db.search('test document', 5, { offset: 5 }) + expect(secondPage.length).toBe(5) + const secondPageIds = secondPage.map(r => r.metadata.id) + + // Ensure no overlap between pages + const overlap = firstPageIds.filter(id => secondPageIds.includes(id)) + expect(overlap.length).toBe(0) + }) + + it('should handle offset beyond available results', async () => { + // Add limited test data + for (let i = 0; i < 10; i++) { + await db.add({ + id: `item-${i}`, + text: `test document ${i}` + }) + } + + // Search with offset beyond available results + const results = await db.search('test document', 5, { offset: 15 }) + expect(results.length).toBe(0) + }) + + it('should return partial results when offset + k exceeds total', async () => { + // Add limited test data + for (let i = 0; i < 10; i++) { + await db.add({ + id: `item-${i}`, + text: `test document ${i}` + }) + } + + // Search with offset that allows only partial results + const results = await db.search('test document', 5, { offset: 7 }) + expect(results.length).toBe(3) // Only 3 results available after offset 7 + }) + }) + + describe('Pagination with Filters', () => { + it.skip('should paginate with noun type filters', async () => { + // TODO: This test requires proper noun type support in the add method + // Currently skipped as noun types are not directly supported in the add method + // Add test data with different noun types + for (let i = 0; i < 15; i++) { + await db.add({ + id: `doc-${i}`, + text: `document ${i}`, + type: 'document' + }, undefined, 'document') + } + + for (let i = 0; i < 15; i++) { + await db.add({ + id: `note-${i}`, + text: `note ${i}`, + type: 'note' + }, undefined, 'note') + } + + // Get first page of documents + const firstPage = await db.search('document', 5, { + nounTypes: ['document'] + }) + expect(firstPage.length).toBe(5) + expect(firstPage.every(r => r.metadata.type === 'document')).toBe(true) + + // Get second page of documents + const secondPage = await db.search('document', 5, { + nounTypes: ['document'], + offset: 5 + }) + expect(secondPage.length).toBe(5) + expect(secondPage.every(r => r.metadata.type === 'document')).toBe(true) + + // Ensure pages are different + const firstIds = firstPage.map(r => r.metadata.id) + const secondIds = secondPage.map(r => r.metadata.id) + const overlap = firstIds.filter(id => secondIds.includes(id)) + expect(overlap.length).toBe(0) + }) + + it('should paginate with service filters', async () => { + // Add test data with different services + for (let i = 0; i < 20; i++) { + const metadata = { + id: `item-${i}`, + text: `test item ${i}`, + createdBy: { + augmentation: i < 10 ? 'service-a' : 'service-b' + } + } + await db.add(metadata) + } + + // Get paginated results for service-a + const page1 = await db.search('test item', 3, { + service: 'service-a' + }) + const page2 = await db.search('test item', 3, { + service: 'service-a', + offset: 3 + }) + + // Check that results are from service-a + expect(page1.every(r => r.metadata.createdBy?.augmentation === 'service-a')).toBe(true) + expect(page2.every(r => r.metadata.createdBy?.augmentation === 'service-a')).toBe(true) + + // Check no overlap + const page1Ids = page1.map(r => r.metadata.id) + const page2Ids = page2.map(r => r.metadata.id) + expect(page1Ids.filter(id => page2Ids.includes(id)).length).toBe(0) + }) + }) + + describe('Pagination Consistency', () => { + it('should maintain consistent ordering across pages', async () => { + // Add test data + for (let i = 0; i < 30; i++) { + await db.add({ + id: `item-${i.toString().padStart(2, '0')}`, + text: `consistent test ${i}`, + score: Math.random() + }) + } + + // Get all results in one query + const allResults = await db.search('consistent test', 30) + const allIds = allResults.map(r => r.metadata.id) + + // Get results in pages + const page1 = await db.search('consistent test', 10, { offset: 0 }) + const page2 = await db.search('consistent test', 10, { offset: 10 }) + const page3 = await db.search('consistent test', 10, { offset: 20 }) + + const pagedIds = [ + ...page1.map(r => r.metadata.id), + ...page2.map(r => r.metadata.id), + ...page3.map(r => r.metadata.id) + ] + + // Check that paginated results match the full query + expect(pagedIds).toEqual(allIds) + }) + + it('should handle empty results gracefully', async () => { + // Search empty database with offset + const results = await db.search('nonexistent', 10, { offset: 5 }) + expect(results).toEqual([]) + }) + }) + + describe('Vector Search with Offset', () => { + it('should paginate vector searches', async () => { + // Add test vectors + for (let i = 0; i < 20; i++) { + const vector = new Array(384).fill(0).map(() => Math.random()) + await db.add({ + id: `vec-${i}`, + vector: vector, + index: i + }) + } + + // Create a query vector + const queryVector = new Array(384).fill(0).map(() => Math.random()) + + // Get first page + const page1 = await db.search(queryVector, 5, { forceEmbed: false }) + expect(page1.length).toBe(5) + + // Get second page + const page2 = await db.search(queryVector, 5, { + forceEmbed: false, + offset: 5 + }) + expect(page2.length).toBe(5) + + // Ensure different results + const page1Ids = page1.map(r => r.metadata.id) + const page2Ids = page2.map(r => r.metadata.id) + expect(page1Ids.filter(id => page2Ids.includes(id)).length).toBe(0) + }) + }) +}) \ No newline at end of file diff --git a/tests/performance-improvements.test.ts b/tests/performance-improvements.test.ts new file mode 100644 index 00000000..43568235 --- /dev/null +++ b/tests/performance-improvements.test.ts @@ -0,0 +1,257 @@ +/** + * Tests for performance improvements: caching and cursor-based pagination + */ + +import { describe, it, expect, beforeEach, afterEach } from 'vitest' +import { BrainyData } from '../src/brainyData.js' +import { cleanupWorkerPools } from '../src/utils/index.js' + +describe('Performance Improvements', () => { + let db: BrainyData + + beforeEach(async () => { + // Initialize BrainyData with caching enabled + db = new BrainyData({ + storage: { + forceMemoryStorage: true + }, + searchCache: { + enabled: true, + maxSize: 50, + maxAge: 60000 // 1 minute + }, + logging: { + verbose: false + } + }) + await db.init() + }) + + afterEach(async () => { + await db.clear() + await cleanupWorkerPools() + }) + + describe('Search Result Caching', () => { + it('should cache search results transparently', async () => { + // Add test data + for (let i = 0; i < 10; i++) { + await db.add({ + id: `item-${i}`, + text: `test document ${i}`, + value: i + }) + } + + // Clear cache to start fresh + db.clearCache() + let stats = db.getCacheStats() + expect(stats.search.hits).toBe(0) + expect(stats.search.misses).toBe(0) + + // First search - should be cache miss + const query = 'test document' + const results1 = await db.search(query, 5) + expect(results1.length).toBe(5) + + stats = db.getCacheStats() + expect(stats.search.misses).toBe(1) // First search is a miss + expect(stats.search.hits).toBe(0) + + // Second identical search - should be cache hit + const results2 = await db.search(query, 5) + expect(results2.length).toBe(5) + expect(results2).toEqual(results1) // Same results + + stats = db.getCacheStats() + expect(stats.search.misses).toBe(1) // Still only one miss + expect(stats.search.hits).toBe(1) // Now we have a hit + expect(stats.search.hitRate).toBe(0.5) // 50% hit rate + }) + + it('should invalidate cache when data changes', async () => { + // Add test data + for (let i = 0; i < 5; i++) { + await db.add({ + id: `item-${i}`, + text: `test document ${i}` + }) + } + + // Search and cache results + const results1 = await db.search('test document', 3) + expect(results1.length).toBe(3) + + let stats = db.getCacheStats() + const initialMisses = stats.search.misses + + // Same search should hit cache + await db.search('test document', 3) + stats = db.getCacheStats() + expect(stats.search.hits).toBeGreaterThan(0) + + // Add new data - should invalidate cache + await db.add({ + id: 'new-item', + text: 'new test document' + }) + + // Same search should miss cache (due to invalidation) + await db.search('test document', 3) + stats = db.getCacheStats() + // Cache was cleared, so we should have fewer misses recorded than expected + // The key point is that cache size should be 0 or low, indicating invalidation worked + expect(stats.search.size).toBeLessThanOrEqual(1) // Cache should have been cleared + }) + + it('should handle cache with different search parameters', async () => { + // Add test data + for (let i = 0; i < 10; i++) { + await db.add({ + id: `item-${i}`, + text: `test document ${i}` + }) + } + + db.clearCache() + + // Different k values should create different cache entries + await db.search('test document', 3) + await db.search('test document', 5) // Different k + await db.search('test document', 3) // Should hit cache + + const stats = db.getCacheStats() + expect(stats.search.hits).toBe(1) + expect(stats.search.misses).toBe(2) + }) + }) + + describe('Cursor-based Pagination', () => { + beforeEach(async () => { + // Add more test data for pagination + for (let i = 0; i < 50; i++) { + await db.add({ + id: `doc-${i.toString().padStart(2, '0')}`, + text: `document content for testing pagination ${i}`, + index: i + }) + } + }) + + it('should return cursor for pagination', async () => { + const page1 = await db.searchWithCursor('document content for testing', 10) + + expect(page1.results.length).toBe(10) + expect(page1.hasMore).toBe(true) + expect(page1.cursor).toBeDefined() + expect(page1.cursor!.lastId).toBeDefined() + expect(page1.cursor!.lastScore).toBeDefined() + }) + + it('should paginate consistently with cursor', async () => { + const page1 = await db.searchWithCursor('document content for testing', 5) + expect(page1.results.length).toBe(5) + expect(page1.hasMore).toBe(true) + + // Get next page using cursor + const page2 = await db.searchWithCursor('document content for testing', 5, { + cursor: page1.cursor + }) + expect(page2.results.length).toBe(5) + + // Ensure no overlap between pages + const page1Ids = page1.results.map(r => r.id) + const page2Ids = page2.results.map(r => r.id) + const overlap = page1Ids.filter(id => page2Ids.includes(id)) + expect(overlap.length).toBe(0) + }) + + it('should handle last page correctly', async () => { + // Get small pages to reach the end + let currentCursor: any = undefined + let allResults: any[] = [] + let pageCount = 0 + const maxPages = 10 // Safety limit + + while (pageCount < maxPages) { + const page = await db.searchWithCursor('document content', 5, { + cursor: currentCursor + }) + + allResults.push(...page.results) + pageCount++ + + if (!page.hasMore) { + expect(page.cursor).toBeUndefined() + break + } + + currentCursor = page.cursor + } + + expect(pageCount).toBeLessThan(maxPages) // Should have finished before limit + expect(allResults.length).toBeGreaterThan(0) + }) + + it('should provide total estimate when possible', async () => { + const page = await db.searchWithCursor('document content', 50) // Request more than we have + + // Should get all results in one page + expect(page.results.length).toBeGreaterThan(0) + expect(page.hasMore).toBe(false) + expect(page.totalEstimate).toBeDefined() + expect(page.totalEstimate).toBe(page.results.length) + }) + }) + + describe('Performance Characteristics', () => { + it('should show performance improvement with caching', async () => { + // Add substantial test data + for (let i = 0; i < 50; i++) { + await db.add({ + id: `perf-test-${i}`, + text: `performance test document ${i} with some content` + }) + } + + // Clear cache and perform first search + db.clearCache() + const results1 = await db.search('performance test document', 10) + + let stats = db.getCacheStats() + expect(stats.search.misses).toBe(1) // First search is a miss + expect(stats.search.hits).toBe(0) + + // Perform second identical search (should hit cache) + const results2 = await db.search('performance test document', 10) + + expect(results1).toEqual(results2) // Same results + + stats = db.getCacheStats() + expect(stats.search.hits).toBe(1) // Second search is a hit + expect(stats.search.hitRate).toBe(0.5) // 50% hit rate (1 hit, 1 miss) + }) + + it('should provide cache memory usage information', async () => { + // Add some data and search to populate cache + for (let i = 0; i < 10; i++) { + await db.add({ + id: `mem-test-${i}`, + text: `memory test ${i}` + }) + } + + db.clearCache() + + // Perform several searches to populate cache + await db.search('memory test', 5) + await db.search('memory test', 3) + await db.search('test', 5) + + const stats = db.getCacheStats() + expect(stats.searchMemoryUsage).toBeGreaterThan(0) + expect(stats.search.size).toBeGreaterThan(0) + expect(stats.search.size).toBeLessThanOrEqual(stats.search.maxSize) + }) + }) +}) \ No newline at end of file diff --git a/tests/performance.test.ts b/tests/performance.test.ts new file mode 100644 index 00000000..4f6de9ad --- /dev/null +++ b/tests/performance.test.ts @@ -0,0 +1,234 @@ +/** + * Performance Tests + * + * Purpose: + * This test suite measures the performance of Brainy operations with different dataset sizes: + * 1. Small datasets (10-100 items) + * 2. Medium datasets (100-1000 items) + * 3. Large datasets (1000+ items) + * + * These tests help identify performance bottlenecks and ensure the library + * remains efficient as the dataset grows. + * + * Note: These tests are marked as "slow" and may take longer to run. + */ + +import { describe, it, expect, beforeEach, afterEach } from 'vitest' +import { BrainyData, createStorage } from '../dist/unified.js' + +// Helper function to measure execution time +const measureExecutionTime = async (fn: () => Promise): Promise => { + const start = performance.now() + await fn() + const end = performance.now() + return end - start +} + +// Helper function to generate test data +const generateTestData = (count: number): string[] => { + return Array.from({ length: count }, (_, i) => `Test item ${i} with some additional text for embedding`) +} + +describe('Performance Tests', () => { + let brainyInstance: any + + beforeEach(async () => { + // Create a test BrainyData instance with memory storage for faster tests + const storage = await createStorage({ forceMemoryStorage: true }) + brainyInstance = new BrainyData({ + storageAdapter: storage + }) + + await brainyInstance.init() + + // Clear any existing data to ensure a clean test environment + await brainyInstance.clear() + }) + + afterEach(async () => { + // Clean up after each test + if (brainyInstance) { + await brainyInstance.clear() + await brainyInstance.shutDown() + } + }) + + describe('Small Dataset (10-100 items)', () => { + it('should add items efficiently', async () => { + const items = generateTestData(50) + + const executionTime = await measureExecutionTime(async () => { + await brainyInstance.addBatch(items) + }) + + console.log(`Adding 50 items took ${executionTime.toFixed(2)}ms (${(executionTime / 50).toFixed(2)}ms per item)`) + + // Verify all items were added + const size = await brainyInstance.size() + expect(size).toBe(50) + + // No specific performance assertion, just logging for analysis + }) + + it('should search efficiently', async () => { + // Add test data + const items = generateTestData(50) + await brainyInstance.addBatch(items) + + // Measure search performance + const executionTime = await measureExecutionTime(async () => { + await brainyInstance.search('Test item', 10) + }) + + console.log(`Searching in 50 items took ${executionTime.toFixed(2)}ms`) + + // No specific performance assertion, just logging for analysis + }) + }) + + describe('Medium Dataset (100-1000 items)', () => { + it('should add items efficiently', async () => { + const items = generateTestData(200) + + const executionTime = await measureExecutionTime(async () => { + await brainyInstance.addBatch(items) + }) + + console.log(`Adding 200 items took ${executionTime.toFixed(2)}ms (${(executionTime / 200).toFixed(2)}ms per item)`) + + // Verify all items were added + const size = await brainyInstance.size() + expect(size).toBe(200) + }) + + it('should search efficiently', async () => { + // Add test data + const items = generateTestData(200) + await brainyInstance.addBatch(items) + + // Measure search performance + const executionTime = await measureExecutionTime(async () => { + await brainyInstance.search('Test item', 10) + }) + + console.log(`Searching in 200 items took ${executionTime.toFixed(2)}ms`) + }) + + it('should handle multiple concurrent searches efficiently', async () => { + // Add test data + const items = generateTestData(200) + await brainyInstance.addBatch(items) + + // Perform multiple concurrent searches + const searchQueries = [ + 'Test item 10', + 'Test item 50', + 'Test item 100', + 'Test item 150', + 'Test item 190' + ] + + const executionTime = await measureExecutionTime(async () => { + await Promise.all(searchQueries.map(query => brainyInstance.search(query, 10))) + }) + + console.log(`5 concurrent searches in 200 items took ${executionTime.toFixed(2)}ms (${(executionTime / 5).toFixed(2)}ms per search)`) + }) + }) + + // Large dataset tests are skipped by default as they can be slow + // Use .only instead of .skip to run these tests specifically + describe.skip('Large Dataset (1000+ items)', () => { + it('should add items efficiently', async () => { + const items = generateTestData(1000) + + const executionTime = await measureExecutionTime(async () => { + await brainyInstance.addBatch(items) + }) + + console.log(`Adding 1000 items took ${executionTime.toFixed(2)}ms (${(executionTime / 1000).toFixed(2)}ms per item)`) + + // Verify all items were added + const size = await brainyInstance.size() + expect(size).toBe(1000) + }) + + it('should search efficiently', async () => { + // Add test data + const items = generateTestData(1000) + await brainyInstance.addBatch(items) + + // Measure search performance + const executionTime = await measureExecutionTime(async () => { + await brainyInstance.search('Test item', 10) + }) + + console.log(`Searching in 1000 items took ${executionTime.toFixed(2)}ms`) + }) + + it('should handle multiple concurrent searches efficiently', async () => { + // Add test data + const items = generateTestData(1000) + await brainyInstance.addBatch(items) + + // Perform multiple concurrent searches + const searchQueries = [ + 'Test item 100', + 'Test item 300', + 'Test item 500', + 'Test item 700', + 'Test item 900' + ] + + const executionTime = await measureExecutionTime(async () => { + await Promise.all(searchQueries.map(query => brainyInstance.search(query, 10))) + }) + + console.log(`5 concurrent searches in 1000 items took ${executionTime.toFixed(2)}ms (${(executionTime / 5).toFixed(2)}ms per search)`) + }) + }) + + describe('Performance Scaling', () => { + it('should demonstrate search performance scaling with dataset size', async () => { + // Test with different dataset sizes + const datasetSizes = [10, 50, 100] + const results: { size: number; time: number }[] = [] + + for (const size of datasetSizes) { + // Add test data + const items = generateTestData(size) + await brainyInstance.addBatch(items) + + // Measure search performance + const executionTime = await measureExecutionTime(async () => { + await brainyInstance.search('Test item', 10) + }) + + results.push({ size, time: executionTime }) + + // Clear for next iteration + await brainyInstance.clear() + } + + // Log results + console.log('Search Performance Scaling:') + results.forEach(result => { + console.log(`Dataset size: ${result.size}, Search time: ${result.time.toFixed(2)}ms`) + }) + + // Calculate scaling factor (how much slower per item) + if (results.length >= 2) { + const smallestDataset = results[0] + const largestDataset = results[results.length - 1] + + const scalingFactor = (largestDataset.time / smallestDataset.time) / + (largestDataset.size / smallestDataset.size) + + console.log(`Scaling factor: ${scalingFactor.toFixed(2)}x`) + + // Ideally, the scaling factor should be close to 1 (linear scaling) + // or less than 1 (sub-linear scaling) + } + }) + }) +}) diff --git a/tests/regression.test.ts b/tests/regression.test.ts new file mode 100644 index 00000000..fa43a051 --- /dev/null +++ b/tests/regression.test.ts @@ -0,0 +1,346 @@ +/** + * Regression Test Suite for Brainy + * + * This test suite verifies that core functionality works across: + * - All storage adapters + * - All environments (Node.js, Browser simulation) + * - Performance benchmarks + * - Package size limits + * - CLI and API consistency + * + * These tests should ALWAYS pass before any release. + */ + +import { describe, it, expect, beforeEach, afterEach } from 'vitest' +import { BrainyData, NounType, VerbType, MemoryStorage, OPFSStorage, createEmbeddingFunction } from '../src/index.js' +import { performance } from 'perf_hooks' + +describe('Brainy Regression Tests', () => { + + describe('Core Functionality Across Storage Adapters', () => { + const storageAdapters = [ + { name: 'Memory', create: () => new MemoryStorage() }, + // Note: FileSystem and OPFS require different test environments + // { name: 'OPFS', create: () => new OPFSStorage() }, // Browser only + // { name: 'FileSystem', create: () => new FileSystemStorage('./test-fs') }, // Node only + ] + + storageAdapters.forEach(({ name, create }) => { + describe(`${name} Storage`, () => { + let brainy: BrainyData + + beforeEach(async () => { + brainy = new BrainyData({ + storage: create() + }) + await brainy.init() + }) + + afterEach(async () => { + if (brainy) { + await brainy.cleanup() + } + }) + + it('should handle basic add and search operations', async () => { + const id = await brainy.add("Test data for regression testing") + expect(id).toBeDefined() + + const results = await brainy.search("regression testing", 5) + expect(results.length).toBeGreaterThan(0) + expect(results[0].id).toBe(id) + }) + + it('should handle typed noun and verb operations', async () => { + const personId = await brainy.addNoun("John Doe", NounType.Person) + const companyId = await brainy.addNoun("Tech Corp", NounType.Organization) + const verbId = await brainy.addVerb(personId, companyId, VerbType.WorksWith) + + expect(personId).toBeDefined() + expect(companyId).toBeDefined() + expect(verbId).toBeDefined() + }) + + it('should handle metadata filtering', async () => { + await brainy.add("Item 1", { category: "A", priority: 1 }) + await brainy.add("Item 2", { category: "B", priority: 2 }) + + const results = await brainy.search("", 10, { + metadata: { category: "A" } + }) + expect(results.length).toBe(1) + expect(results[0].metadata.category).toBe("A") + }) + + it('should handle update operations', async () => { + const id = await brainy.add("Original content", { version: 1 }) + const success = await brainy.update(id, "Updated content", { version: 2 }) + expect(success).toBe(true) + + const results = await brainy.search("Updated content", 5) + expect(results[0].metadata.version).toBe(2) + }) + + it('should handle soft delete (default behavior)', async () => { + const id = await brainy.add("Content to delete") + const success = await brainy.delete(id) // Soft delete by default + expect(success).toBe(true) + + // Should not appear in search results + const results = await brainy.search("Content to delete", 10) + expect(results.length).toBe(0) + }) + }) + }) + }) + + describe('Performance Benchmarks', () => { + let brainy: BrainyData + + beforeEach(async () => { + brainy = new BrainyData() + await brainy.init() + }) + + afterEach(async () => { + if (brainy) { + await brainy.cleanup() + } + }) + + it('should add 100 items within performance threshold', async () => { + const startTime = performance.now() + + const promises = [] + for (let i = 0; i < 100; i++) { + promises.push(brainy.add(`Test item ${i}`)) + } + await Promise.all(promises) + + const endTime = performance.now() + const duration = endTime - startTime + + // Should complete within 10 seconds (generous threshold for CI) + expect(duration).toBeLessThan(10000) + }) + + it('should search through 1000+ items efficiently', async () => { + // Add test data + const promises = [] + for (let i = 0; i < 200; i++) { + promises.push(brainy.add(`Document ${i} about various topics and information`)) + } + await Promise.all(promises) + + // Benchmark search + const startTime = performance.now() + const results = await brainy.search("document topics", 10) + const endTime = performance.now() + const duration = endTime - startTime + + expect(results.length).toBeGreaterThan(0) + // Search should complete within 1 second + expect(duration).toBeLessThan(1000) + }) + + it('should handle batch import efficiently', async () => { + const data = Array.from({ length: 50 }, (_, i) => `Batch item ${i}`) + + const startTime = performance.now() + const ids = await brainy.import(data) + const endTime = performance.now() + const duration = endTime - startTime + + expect(ids.length).toBe(50) + // Batch import should be faster than individual adds + expect(duration).toBeLessThan(5000) + }) + }) + + describe('Environment Compatibility', () => { + it('should detect Node.js environment correctly', async () => { + const { isNode, isBrowser, isWebWorker } = await import('../src/index.js') + + expect(isNode()).toBe(true) + expect(isBrowser()).toBe(false) + expect(isWebWorker()).toBe(false) + }) + + it('should create embedding functions in Node.js', async () => { + const embeddingFn = await createEmbeddingFunction() + expect(typeof embeddingFn).toBe('function') + + const embedding = await embeddingFn("test text") + expect(Array.isArray(embedding)).toBe(true) + expect(embedding.length).toBeGreaterThan(0) + }) + }) + + describe('Data Integrity', () => { + let brainy: BrainyData + + beforeEach(async () => { + brainy = new BrainyData() + await brainy.init() + }) + + afterEach(async () => { + if (brainy) { + await brainy.cleanup() + } + }) + + it('should maintain data consistency across operations', async () => { + // Add initial data + const id1 = await brainy.add("Document about machine learning") + const id2 = await brainy.add("Article about artificial intelligence") + + // Verify both exist + let results = await brainy.search("machine learning", 10) + expect(results.some(r => r.id === id1)).toBe(true) + + results = await brainy.search("artificial intelligence", 10) + expect(results.some(r => r.id === id2)).toBe(true) + + // Update one item + await brainy.update(id1, "Updated document about deep learning") + + // Verify update + results = await brainy.search("deep learning", 10) + expect(results.some(r => r.id === id1)).toBe(true) + + // Original content should not be found + results = await brainy.search("machine learning", 10) + expect(results.some(r => r.id === id1)).toBe(false) + + // Other document should remain unchanged + results = await brainy.search("artificial intelligence", 10) + expect(results.some(r => r.id === id2)).toBe(true) + }) + + it('should handle concurrent operations without corruption', async () => { + const concurrentOperations = [] + + // Start multiple operations concurrently + for (let i = 0; i < 20; i++) { + concurrentOperations.push(brainy.add(`Concurrent item ${i}`)) + } + + const ids = await Promise.all(concurrentOperations) + expect(ids.length).toBe(20) + + // Verify all items can be found + const results = await brainy.search("Concurrent item", 25) + expect(results.length).toBe(20) + }) + }) + + describe('Error Handling & Edge Cases', () => { + let brainy: BrainyData + + beforeEach(async () => { + brainy = new BrainyData() + await brainy.init() + }) + + afterEach(async () => { + if (brainy) { + await brainy.cleanup() + } + }) + + it('should handle empty search queries gracefully', async () => { + await brainy.add("Some content") + + const results = await brainy.search("", 10) + expect(Array.isArray(results)).toBe(true) + // Empty query might return all results or none, but should not throw + }) + + it('should handle invalid IDs gracefully', async () => { + const result = await brainy.update("invalid-id", "new data") + expect(result).toBe(false) + + const deleteResult = await brainy.delete("invalid-id") + expect(deleteResult).toBe(false) + }) + + it('should handle very large text content', async () => { + const largeText = "Lorem ipsum ".repeat(1000) // ~11KB text + const id = await brainy.add(largeText) + expect(id).toBeDefined() + + const results = await brainy.search("Lorem ipsum", 5) + expect(results.some(r => r.id === id)).toBe(true) + }) + + it('should handle special characters and unicode', async () => { + const specialText = "Hello ไธ–็•Œ! ๐ŸŒ Special chars: @#$%^&*()_+-=[]{}|;':\",./<>?" + const id = await brainy.add(specialText) + expect(id).toBeDefined() + + const results = await brainy.search("ไธ–็•Œ", 5) + expect(results.some(r => r.id === id)).toBe(true) + }) + }) + + describe('Package Size Regression', () => { + it('should not exceed package size threshold', async () => { + // This is a placeholder test - actual implementation would check built package size + // For now, we'll just verify core imports don't significantly bloat + const coreModules = await import('../src/index.js') + + // Verify main exports exist (prevents tree-shaking regression) + expect(coreModules.BrainyData).toBeDefined() + expect(coreModules.NounType).toBeDefined() + expect(coreModules.VerbType).toBeDefined() + expect(coreModules.createEmbeddingFunction).toBeDefined() + }) + }) + + describe('Configuration & Initialization', () => { + it('should initialize with default configuration', async () => { + const brainy = new BrainyData() + await brainy.init() + + // Should not throw and should be usable + const id = await brainy.add("Test initialization") + expect(id).toBeDefined() + + await brainy.cleanup() + }) + + it('should initialize with custom configuration', async () => { + const brainy = new BrainyData({ + maxNeighbors: 32, + efConstruction: 400, + storage: new MemoryStorage() + }) + await brainy.init() + + const id = await brainy.add("Test custom config") + expect(id).toBeDefined() + + await brainy.cleanup() + }) + + it('should handle multiple instances', async () => { + const brainy1 = new BrainyData() + const brainy2 = new BrainyData() + + await brainy1.init() + await brainy2.init() + + // Both should work independently + const id1 = await brainy1.add("Instance 1 data") + const id2 = await brainy2.add("Instance 2 data") + + expect(id1).toBeDefined() + expect(id2).toBeDefined() + expect(id1).not.toBe(id2) + + await brainy1.destroy() + await brainy2.destroy() + }) + }) +}) \ No newline at end of file diff --git a/tests/release-validation.test.ts b/tests/release-validation.test.ts new file mode 100644 index 00000000..c23bcf3d --- /dev/null +++ b/tests/release-validation.test.ts @@ -0,0 +1,258 @@ +/** + * Release Validation Test Suite + * + * This comprehensive suite must PASS before any release. + * It validates all critical functionality and prevents regressions. + */ + +import { describe, it, expect } from 'vitest' +import { execSync } from 'child_process' +import { readFileSync } from 'fs' +import path from 'path' + +describe('Release Validation', () => { + + describe('Package Integrity', () => { + it('should have valid package.json', () => { + const packagePath = path.join(process.cwd(), 'package.json') + const packageJson = JSON.parse(readFileSync(packagePath, 'utf-8')) + + expect(packageJson.name).toBe('@soulcraft/brainy') + expect(packageJson.version).toMatch(/^\d+\.\d+\.\d+/) + expect(packageJson.main).toBe('dist/index.js') + expect(packageJson.types).toBe('dist/index.d.ts') + expect(packageJson.bin.brainy).toBe('./bin/brainy.js') + }) + + it('should build without errors', () => { + expect(() => { + execSync('npm run build', { + stdio: 'pipe', + timeout: 60000 + }) + }).not.toThrow() + }) + + it('should have reasonable package size', () => { + try { + // Check if dist directory exists and has content + const output = execSync('du -sh dist/', { encoding: 'utf-8' }) + const sizeMatch = output.match(/^([\d.]+)([KMGT]?)/) + + if (sizeMatch) { + const [, size, unit] = sizeMatch + const sizeNum = parseFloat(size) + + // Package should be under 50MB total + if (unit === 'M') { + expect(sizeNum).toBeLessThan(50) + } else if (unit === 'K') { + // KB is fine + expect(sizeNum).toBeLessThan(50000) // 50MB in KB + } + } + } catch (error) { + // If du command fails, just check that dist exists + expect(true).toBe(true) // Always pass if we can't check size + } + }) + }) + + describe('Core API Validation', () => { + it('should export all required 1.0 API methods', async () => { + const brainyModule = await import('../src/index.js') + const { BrainyData } = brainyModule + + const instance = new BrainyData() + + // Validate 7 core methods exist + expect(typeof instance.add).toBe('function') + expect(typeof instance.search).toBe('function') + expect(typeof instance.import).toBe('function') + expect(typeof instance.addNoun).toBe('function') + expect(typeof instance.addVerb).toBe('function') + expect(typeof instance.update).toBe('function') + expect(typeof instance.delete).toBe('function') + }) + + it('should export encryption methods', async () => { + const brainyModule = await import('../src/index.js') + const { BrainyData } = brainyModule + + const instance = new BrainyData() + + expect(typeof instance.encryptData).toBe('function') + expect(typeof instance.decryptData).toBe('function') + expect(typeof instance.setConfig).toBe('function') + expect(typeof instance.getConfig).toBe('function') + }) + + it('should export graph types', async () => { + const { NounType, VerbType } = await import('../src/index.js') + + expect(NounType).toBeDefined() + expect(VerbType).toBeDefined() + + // Check some key types exist + expect(NounType.Person).toBe('person') + expect(NounType.Organization).toBe('organization') + expect(VerbType.WorksWith).toBe('worksWith') + expect(VerbType.RelatedTo).toBe('relatedTo') + }) + + it('should export container preloading methods', async () => { + const { BrainyData } = await import('../src/index.js') + + expect(typeof BrainyData.preloadModel).toBe('function') + expect(typeof BrainyData.warmup).toBe('function') + }) + }) + + describe('CLI Validation', () => { + const CLI_PATH = path.resolve('./bin/brainy.js') + + it('should have executable CLI', () => { + expect(() => { + execSync(`node ${CLI_PATH} --help`, { + stdio: 'pipe', + timeout: 10000 + }) + }).not.toThrow() + }) + + it('should show correct version', () => { + const output = execSync(`node ${CLI_PATH} --version`, { + encoding: 'utf-8', + timeout: 10000 + }) + + expect(output).toMatch(/\d+\.\d+\.\d+/) + }) + + it('should list all 9 core commands in help', () => { + const output = execSync(`node ${CLI_PATH} --help`, { + encoding: 'utf-8', + timeout: 10000 + }) + + // Should contain all 9 unified commands + expect(output).toContain('init') + expect(output).toContain('add') + expect(output).toContain('search') + expect(output).toContain('update') + expect(output).toContain('delete') + expect(output).toContain('import') + expect(output).toContain('status') + expect(output).toContain('config') + expect(output).toContain('chat') + }) + }) + + describe('Documentation Validation', () => { + it('should have comprehensive CHANGELOG.md', () => { + const changelogPath = path.join(process.cwd(), 'CHANGELOG.md') + const changelog = readFileSync(changelogPath, 'utf-8') + + expect(changelog).toContain('1.0.0-rc.1') + expect(changelog).toContain('BREAKING CHANGES') + expect(changelog).toContain('Unified API') + expect(changelog).toContain('CLI TRANSFORMATION') + }) + + it('should have migration guide', () => { + const migrationPath = path.join(process.cwd(), 'MIGRATION.md') + const migration = readFileSync(migrationPath, 'utf-8') + + expect(migration).toContain('Migration Guide: Brainy 0.x โ†’ 1.0') + expect(migration).toContain('addSmart()') + expect(migration).toContain('add()') + expect(migration).toContain('CLI Command Changes') + }) + + it('should have README.md', () => { + const readmePath = path.join(process.cwd(), 'README.md') + const readme = readFileSync(readmePath, 'utf-8') + + expect(readme.length).toBeGreaterThan(1000) // Should have substantial content + expect(readme).toContain('Brainy') + }) + }) + + describe('Environment Compatibility', () => { + it('should work in Node.js environment', async () => { + const { BrainyData, isNode } = await import('../src/index.js') + + expect(isNode()).toBe(true) + + // Should be able to create and init instance + const brainy = new BrainyData() + await brainy.init() + + // Basic functionality should work + const id = await brainy.add("Release validation test") + expect(id).toBeDefined() + + await brainy.cleanup() + }) + + it('should have proper TypeScript definitions', () => { + const packagePath = path.join(process.cwd(), 'package.json') + const packageJson = JSON.parse(readFileSync(packagePath, 'utf-8')) + + expect(packageJson.types).toBe('dist/index.d.ts') + + // Check that dist directory exists (should be built) + expect(() => { + execSync('ls dist/index.d.ts', { stdio: 'pipe' }) + }).not.toThrow() + }) + }) + + describe('Dependency Security', () => { + it('should have reasonable dependency count', () => { + const packagePath = path.join(process.cwd(), 'package.json') + const packageJson = JSON.parse(readFileSync(packagePath, 'utf-8')) + + const depCount = Object.keys(packageJson.dependencies || {}).length + const devDepCount = Object.keys(packageJson.devDependencies || {}).length + + // Should not have excessive dependencies + expect(depCount).toBeLessThan(20) // Production dependencies + expect(devDepCount).toBeLessThan(30) // Development dependencies + }) + + it('should not have high-severity vulnerabilities', () => { + try { + // Run npm audit to check for vulnerabilities + execSync('npm audit --audit-level=high', { + stdio: 'pipe', + timeout: 30000 + }) + // If it doesn't throw, we're good + expect(true).toBe(true) + } catch (error: any) { + // npm audit returns non-zero exit code for vulnerabilities + // We'll be lenient for now but should investigate if this fails + console.warn('npm audit found potential security issues:', error.message) + expect(true).toBe(true) // Don't fail the test, just warn + } + }) + }) + + describe('Performance Baseline', () => { + it('should maintain acceptable initialization time', async () => { + const start = Date.now() + + const { BrainyData } = await import('../src/index.js') + const brainy = new BrainyData() + await brainy.init() + + const initTime = Date.now() - start + + // Should initialize within 5 seconds + expect(initTime).toBeLessThan(5000) + + await brainy.cleanup() + }) + }) +}) \ No newline at end of file diff --git a/tests/results/test-results.json b/tests/results/test-results.json new file mode 100644 index 00000000..306b4c19 --- /dev/null +++ b/tests/results/test-results.json @@ -0,0 +1 @@ +{"numTotalTestSuites":9,"numPassedTestSuites":9,"numFailedTestSuites":0,"numPendingTestSuites":0,"numTotalTests":19,"numPassedTests":19,"numFailedTests":0,"numPendingTests":0,"numTodoTests":0,"snapshot":{"added":0,"failure":false,"filesAdded":0,"filesRemoved":0,"filesRemovedList":[],"filesUnmatched":0,"filesUpdated":0,"matched":0,"total":0,"unchecked":0,"uncheckedKeysByFile":[],"unmatched":0,"updated":0,"didUpdate":false},"startTime":1755564302710,"success":true,"testResults":[{"assertionResults":[{"ancestorTitles":["Brainy Core Functionality","Library Exports"],"fullName":"Brainy Core Functionality Library Exports should export BrainyData class","status":"passed","title":"should export BrainyData class","duration":0.7985060000000885,"failureMessages":[],"meta":{}},{"ancestorTitles":["Brainy Core Functionality","Library Exports"],"fullName":"Brainy Core Functionality Library Exports should export environment detection functions","status":"passed","title":"should export environment detection functions","duration":0.2400399999999081,"failureMessages":[],"meta":{}},{"ancestorTitles":["Brainy Core Functionality","Library Exports"],"fullName":"Brainy Core Functionality Library Exports should export embedding function creator","status":"passed","title":"should export embedding function creator","duration":0.11205599999993865,"failureMessages":[],"meta":{}},{"ancestorTitles":["Brainy Core Functionality","Library Exports"],"fullName":"Brainy Core Functionality Library Exports should export environment detection functions","status":"passed","title":"should export environment detection functions","duration":0.14592000000016014,"failureMessages":[],"meta":{}},{"ancestorTitles":["Brainy Core Functionality","BrainyData Configuration"],"fullName":"Brainy Core Functionality BrainyData Configuration should create instance with minimal configuration","status":"passed","title":"should create instance with minimal configuration","duration":0.6099659999999858,"failureMessages":[],"meta":{}},{"ancestorTitles":["Brainy Core Functionality","BrainyData Configuration"],"fullName":"Brainy Core Functionality BrainyData Configuration should create instance with full configuration","status":"passed","title":"should create instance with full configuration","duration":0.15911499999992884,"failureMessages":[],"meta":{}},{"ancestorTitles":["Brainy Core Functionality","BrainyData Configuration"],"fullName":"Brainy Core Functionality BrainyData Configuration should not throw with valid configuration parameters","status":"passed","title":"should not throw with valid configuration parameters","duration":0.622143000000051,"failureMessages":[],"meta":{}},{"ancestorTitles":["Brainy Core Functionality","BrainyData Configuration"],"fullName":"Brainy Core Functionality BrainyData Configuration should use default values for optional parameters","status":"passed","title":"should use default values for optional parameters","duration":0.1869329999999536,"failureMessages":[],"meta":{}},{"ancestorTitles":["Brainy Core Functionality","Vector Operations"],"fullName":"Brainy Core Functionality Vector Operations should handle vector addition and search","status":"passed","title":"should handle vector addition and search","duration":9.26907499999993,"failureMessages":[],"meta":{}},{"ancestorTitles":["Brainy Core Functionality","Vector Operations"],"fullName":"Brainy Core Functionality Vector Operations should handle batch vector operations","status":"passed","title":"should handle batch vector operations","duration":3.0796240000001944,"failureMessages":[],"meta":{}},{"ancestorTitles":["Brainy Core Functionality","Vector Operations"],"fullName":"Brainy Core Functionality Vector Operations should handle different distance metrics","status":"passed","title":"should handle different distance metrics","duration":3.9997260000000097,"failureMessages":[],"meta":{}},{"ancestorTitles":["Brainy Core Functionality","Text Processing"],"fullName":"Brainy Core Functionality Text Processing should handle text items with embedding function","status":"passed","title":"should handle text items with embedding function","duration":1.5681580000000395,"failureMessages":[],"meta":{}},{"ancestorTitles":["Brainy Core Functionality","Text Processing"],"fullName":"Brainy Core Functionality Text Processing should handle mixed vector and text operations","status":"passed","title":"should handle mixed vector and text operations","duration":2.007808999999952,"failureMessages":[],"meta":{}},{"ancestorTitles":["Brainy Core Functionality","Error Handling"],"fullName":"Brainy Core Functionality Error Handling should handle invalid vector dimensions","status":"passed","title":"should handle invalid vector dimensions","duration":7.669505000000072,"failureMessages":[],"meta":{}},{"ancestorTitles":["Brainy Core Functionality","Error Handling"],"fullName":"Brainy Core Functionality Error Handling should handle search before initialization","status":"passed","title":"should handle search before initialization","duration":0.29124000000001615,"failureMessages":[],"meta":{}},{"ancestorTitles":["Brainy Core Functionality","Error Handling"],"fullName":"Brainy Core Functionality Error Handling should handle empty search results gracefully","status":"passed","title":"should handle empty search results gracefully","duration":0.9026549999998679,"failureMessages":[],"meta":{}},{"ancestorTitles":["Brainy Core Functionality","Performance and Scalability"],"fullName":"Brainy Core Functionality Performance and Scalability should handle moderate number of vectors efficiently","status":"passed","title":"should handle moderate number of vectors efficiently","duration":151.91693299999997,"failureMessages":[],"meta":{}},{"ancestorTitles":["Brainy Core Functionality","Performance and Scalability"],"fullName":"Brainy Core Functionality Performance and Scalability should maintain search quality with more data","status":"passed","title":"should maintain search quality with more data","duration":189.39662900000008,"failureMessages":[],"meta":{}},{"ancestorTitles":["Brainy Core Functionality","Database Statistics"],"fullName":"Brainy Core Functionality Database Statistics should provide statistics structure even if counts are not tracked","status":"passed","title":"should provide statistics structure even if counts are not tracked","duration":2.8706630000001496,"failureMessages":[],"meta":{}}],"startTime":1755564303835,"endTime":1755564304210.8706,"status":"passed","message":"","name":"/tmp/brainy-clean/tests/core.test.ts"}]} \ No newline at end of file diff --git a/tests/s3-comprehensive.test.ts b/tests/s3-comprehensive.test.ts new file mode 100644 index 00000000..2ab6d2f0 --- /dev/null +++ b/tests/s3-comprehensive.test.ts @@ -0,0 +1,654 @@ +/** + * COMPREHENSIVE S3 Storage Tests + * + * This test suite covers ALL S3-based features to ensure production reliability at scale. + * Tests include: statistics, nouns, verbs, metadata, HNSW index, caching, and error handling. + */ + +import { describe, it, expect, beforeEach, afterEach, vi } from 'vitest' +import { mockClient } from 'aws-sdk-client-mock' +import { + S3Client, + GetObjectCommand, + PutObjectCommand, + ListObjectsV2Command, + HeadObjectCommand, + DeleteObjectCommand, + DeleteObjectsCommand +} from '@aws-sdk/client-s3' +import { BrainyData } from '../src/index.js' +import { createMockEmbeddingFunction, createMockS3Body } from './test-utils.js' + +// Create S3 mock +const s3Mock = mockClient(S3Client) + +// Use the shared mock body helper +const createMockBody = createMockS3Body + +describe('COMPREHENSIVE S3 Storage Tests', () => { + let brainy: BrainyData + + beforeEach(() => { + // Reset all mocks before each test + s3Mock.reset() + vi.clearAllTimers() + vi.useFakeTimers() + }) + + afterEach(async () => { + if (brainy) { + await brainy.clear() + } + vi.useRealTimers() + }) + + describe('Core S3 Operations', () => { + describe('Nouns (Vector Data)', () => { + it('should save and retrieve nouns from S3', async () => { + const nounData = { + id: 'test-noun-1', + vector: [0.1, 0.2, 0.3], + connections: {}, + level: 0 + } + + // Mock S3 responses + s3Mock.on(GetObjectCommand, { + Key: 'nouns/test-noun-1.json' + }).resolves({ + Body: createMockBody(nounData) + }) + + s3Mock.on(PutObjectCommand).resolves({}) + s3Mock.on(ListObjectsV2Command).resolves({ Contents: [] }) + + brainy = new BrainyData({ + embeddingFunction: createMockEmbeddingFunction(), + storage: { + s3Storage: { + bucketName: 'test-bucket', + accessKeyId: 'test-key', + secretAccessKey: 'test-secret', + region: 'us-east-1' + } + } + }) + await brainy.init() + + // Add a noun (use string so it gets embedded) + const id = await brainy.add('Test noun content', { name: 'Test noun' }) + + // Verify noun was saved to S3 + const putCalls = s3Mock.commandCalls(PutObjectCommand) + const nounSave = putCalls.find(call => + call.args[0].input.Key?.includes('nouns/') + ) + expect(nounSave).toBeDefined() + + // Retrieve the noun + const retrieved = await brainy.get(id) + expect(retrieved).toBeDefined() + expect(retrieved?.metadata?.name).toBe('Test noun') + }) + + it('should handle batch noun operations efficiently', async () => { + s3Mock.on(GetObjectCommand).rejects({ name: 'NoSuchKey' }) + s3Mock.on(PutObjectCommand).resolves({}) + s3Mock.on(ListObjectsV2Command).resolves({ Contents: [] }) + + brainy = new BrainyData({ + embeddingFunction: createMockEmbeddingFunction(), + storage: { + s3Storage: { + bucketName: 'test-bucket', + accessKeyId: 'test-key', + secretAccessKey: 'test-secret', + region: 'us-east-1' + } + } + }) + await brainy.init() + + // Add batch of nouns + const items = Array(100).fill(0).map((_, i) => ({ + data: `Item ${i}`, + metadata: { index: i } + })) + + const ids = await brainy.addBatch( + items.map(item => item.data), + items.map(item => item.metadata) + ) + + expect(ids.length).toBe(100) + + // Verify batch operations are optimized + const putCalls = s3Mock.commandCalls(PutObjectCommand) + // Should batch operations, not 100 individual calls + expect(putCalls.length).toBeLessThan(200) // Some batching should occur + }) + }) + + describe('Verbs (Relationships)', () => { + it('should save and retrieve verbs from S3', async () => { + const verbData = { + id: 'verb-1', + source: 'noun-1', + target: 'noun-2', + type: 'relates_to', + vector: [0.4, 0.5, 0.6], + weight: 0.8 + } + + s3Mock.on(GetObjectCommand).rejects({ name: 'NoSuchKey' }) + s3Mock.on(PutObjectCommand).resolves({}) + s3Mock.on(ListObjectsV2Command).resolves({ + Contents: [{ + Key: 'verbs/verb-1.json' + }] + }) + + // Mock verb retrieval + s3Mock.on(GetObjectCommand, { + Key: 'verbs/verb-1.json' + }).resolves({ + Body: createMockBody(verbData) + }) + + brainy = new BrainyData({ + embeddingFunction: createMockEmbeddingFunction(), + storage: { + s3Storage: { + bucketName: 'test-bucket', + accessKeyId: 'test-key', + secretAccessKey: 'test-secret', + region: 'us-east-1' + } + } + }) + await brainy.init() + + // Create nouns first + const id1 = await brainy.add('Noun 1', { type: 'entity' }) + const id2 = await brainy.add('Noun 2', { type: 'entity' }) + + // Create relationship + await brainy.relate(id1, id2, 'relates_to') + + // Verify verb was saved + const putCalls = s3Mock.commandCalls(PutObjectCommand) + const verbSave = putCalls.find(call => + call.args[0].input.Key?.includes('verbs/') + ) + expect(verbSave).toBeDefined() + + // Get verbs by source + const verbs = await brainy.getVerbsBySource(id1) + expect(verbs.length).toBeGreaterThanOrEqual(0) // May be 0 if not mocked fully + }) + }) + + describe('Metadata', () => { + it('should handle metadata storage and retrieval', async () => { + const metadata = { + name: 'Test Item', + category: 'test', + tags: ['tag1', 'tag2'], + nested: { + property: 'value' + } + } + + s3Mock.on(GetObjectCommand).rejects({ name: 'NoSuchKey' }) + s3Mock.on(PutObjectCommand).resolves({}) + s3Mock.on(ListObjectsV2Command).resolves({ Contents: [] }) + + brainy = new BrainyData({ + embeddingFunction: createMockEmbeddingFunction(), + storage: { + s3Storage: { + bucketName: 'test-bucket', + accessKeyId: 'test-key', + secretAccessKey: 'test-secret', + region: 'us-east-1' + } + } + }) + await brainy.init() + + // Add item with complex metadata + const id = await brainy.add('Test data', metadata) + + // Update metadata + await brainy.updateMetadata(id, { + ...metadata, + updated: true + }) + + // Verify metadata was saved + const putCalls = s3Mock.commandCalls(PutObjectCommand) + const metadataSave = putCalls.find(call => + call.args[0].input.Key?.includes('metadata/') + ) + expect(metadataSave).toBeDefined() + + // Retrieve metadata + const retrieved = await brainy.getMetadata(id) + expect(retrieved).toBeDefined() + expect(retrieved?.name).toBe('Test Item') + }) + + it('should handle batch metadata operations', async () => { + s3Mock.on(GetObjectCommand).rejects({ name: 'NoSuchKey' }) + s3Mock.on(PutObjectCommand).resolves({}) + s3Mock.on(ListObjectsV2Command).resolves({ Contents: [] }) + + brainy = new BrainyData({ + embeddingFunction: createMockEmbeddingFunction(), + storage: { + s3Storage: { + bucketName: 'test-bucket', + accessKeyId: 'test-key', + secretAccessKey: 'test-secret', + region: 'us-east-1' + } + } + }) + await brainy.init() + + // Add multiple items + const ids = [] + for (let i = 0; i < 10; i++) { + const id = await brainy.add(`Item ${i}`, { index: i }) + ids.push(id) + } + + // Batch get metadata + const metadataList = await brainy.getBatch(ids) + expect(metadataList.length).toBe(10) + }) + }) + + describe('HNSW Index', () => { + it('should save and load HNSW index from S3', async () => { + const indexData = { + nodes: { + 'node-1': { + id: 'node-1', + vector: [0.1, 0.2, 0.3], + connections: { + 0: ['node-2', 'node-3'] + }, + level: 1 + } + }, + entryPoint: 'node-1', + dimensions: 3, + efConstruction: 200, + m: 16 + } + + // Mock index retrieval + s3Mock.on(GetObjectCommand, { + Key: 'index/hnsw-index.json' + }).resolves({ + Body: createMockBody(indexData) + }) + + s3Mock.on(PutObjectCommand).resolves({}) + s3Mock.on(ListObjectsV2Command).resolves({ Contents: [] }) + + brainy = new BrainyData({ + embeddingFunction: createMockEmbeddingFunction(), + storage: { + s3Storage: { + bucketName: 'test-bucket', + accessKeyId: 'test-key', + secretAccessKey: 'test-secret', + region: 'us-east-1' + } + } + }) + await brainy.init() + + // Add items to build index + await brainy.add('Vector 1 content', { name: 'Vector 1' }) + await brainy.add('Vector 2 content', { name: 'Vector 2' }) + + // Search to verify index works + const results = await brainy.search('search query', 5) + expect(results).toBeDefined() + }) + }) + }) + + describe('S3 Error Handling', () => { + it('should handle S3 connection failures gracefully', async () => { + // Simulate S3 connection failure + s3Mock.on(GetObjectCommand).rejects(new Error('Connection timeout')) + s3Mock.on(PutObjectCommand).rejects(new Error('Connection timeout')) + + brainy = new BrainyData({ + storage: { + s3Storage: { + bucketName: 'test-bucket', + accessKeyId: 'test-key', + secretAccessKey: 'test-secret', + region: 'us-east-1' + } + } + }) + + // Should handle initialization failure gracefully + await expect(brainy.init()).resolves.not.toThrow() + }) + + it('should retry on transient S3 errors', async () => { + let attempts = 0 + + // Fail first 2 attempts, succeed on third + s3Mock.on(PutObjectCommand).callsFake(() => { + attempts++ + if (attempts < 3) { + const error: any = new Error('Service Unavailable') + error.$metadata = { httpStatusCode: 503 } + throw error + } + return Promise.resolve({}) + }) + + s3Mock.on(GetObjectCommand).rejects({ name: 'NoSuchKey' }) + s3Mock.on(ListObjectsV2Command).resolves({ Contents: [] }) + + brainy = new BrainyData({ + storage: { + s3Storage: { + bucketName: 'test-bucket', + accessKeyId: 'test-key', + secretAccessKey: 'test-secret', + region: 'us-east-1' + } + } + }) + await brainy.init() + + // Should retry and eventually succeed + await brainy.add('Test data', { metadata: 'test' }) + + // Verify retries occurred + expect(attempts).toBeGreaterThanOrEqual(1) + }) + + it('should handle S3 permission errors', async () => { + // Simulate permission denied + const permissionError: any = new Error('Access Denied') + permissionError.name = 'AccessDenied' + permissionError.$metadata = { httpStatusCode: 403 } + + s3Mock.on(GetObjectCommand).rejects(permissionError) + s3Mock.on(PutObjectCommand).rejects(permissionError) + + brainy = new BrainyData({ + storage: { + s3Storage: { + bucketName: 'test-bucket', + accessKeyId: 'invalid-key', + secretAccessKey: 'invalid-secret', + region: 'us-east-1' + } + } + }) + + // Should handle permission errors without crashing + await expect(brainy.init()).resolves.not.toThrow() + }) + }) + + describe('S3 Performance Optimizations', () => { + it('should use multipart upload for large objects', async () => { + s3Mock.on(GetObjectCommand).rejects({ name: 'NoSuchKey' }) + s3Mock.on(PutObjectCommand).resolves({}) + s3Mock.on(ListObjectsV2Command).resolves({ Contents: [] }) + + brainy = new BrainyData({ + storage: { + s3Storage: { + bucketName: 'test-bucket', + accessKeyId: 'test-key', + secretAccessKey: 'test-secret', + region: 'us-east-1' + } + } + }) + await brainy.init() + + // Create large dataset + const largeData = Array(1000).fill(0).map((_, i) => ({ + data: `Large item ${i}`, + metadata: { + index: i, + largeField: 'x'.repeat(1000) // Make metadata large + } + })) + + // Add large batch + await brainy.addBatch( + largeData.map(d => d.data), + largeData.map(d => d.metadata) + ) + + // Verify data was uploaded + const putCalls = s3Mock.commandCalls(PutObjectCommand) + expect(putCalls.length).toBeGreaterThan(0) + }) + + it('should implement caching to reduce S3 calls', async () => { + const testData = { + id: 'cached-item', + vector: [0.1, 0.2, 0.3], + metadata: { cached: true } + } + + let getCalls = 0 + s3Mock.on(GetObjectCommand).callsFake((input) => { + getCalls++ + if (input.Key?.includes('nouns/cached-item')) { + return Promise.resolve({ + Body: createMockBody(testData) + }) + } + return Promise.reject({ name: 'NoSuchKey' }) + }) + + s3Mock.on(PutObjectCommand).resolves({}) + s3Mock.on(ListObjectsV2Command).resolves({ Contents: [] }) + + brainy = new BrainyData({ + storage: { + s3Storage: { + bucketName: 'test-bucket', + accessKeyId: 'test-key', + secretAccessKey: 'test-secret', + region: 'us-east-1' + }, + cacheConfig: { + hotCacheMaxSize: 100, + warmCacheTTL: 60000 + } + } + }) + await brainy.init() + + // Add item + const id = await brainy.add('Cached content', { cached: true }, { id: 'cached-item' }) + + // First get - should hit S3 + await brainy.get(id) + const firstGetCalls = getCalls + + // Second get - should hit cache + await brainy.get(id) + const secondGetCalls = getCalls + + // Cache should prevent additional S3 call + expect(secondGetCalls).toBe(firstGetCalls) + }) + + it('should parallelize S3 operations for better throughput', async () => { + s3Mock.on(GetObjectCommand).rejects({ name: 'NoSuchKey' }) + s3Mock.on(PutObjectCommand).resolves({}) + s3Mock.on(ListObjectsV2Command).resolves({ Contents: [] }) + + brainy = new BrainyData({ + storage: { + s3Storage: { + bucketName: 'test-bucket', + accessKeyId: 'test-key', + secretAccessKey: 'test-secret', + region: 'us-east-1' + } + } + }) + await brainy.init() + + // Add multiple items concurrently + const startTime = Date.now() + const promises = [] + + for (let i = 0; i < 20; i++) { + promises.push( + brainy.add(`Parallel item ${i}`, { index: i }) + ) + } + + const ids = await Promise.all(promises) + const endTime = Date.now() + + expect(ids.length).toBe(20) + + // Operations should be parallelized (fast) + // In real scenario, this would be much faster than sequential + expect(endTime - startTime).toBeLessThan(5000) // Should be fast due to mocking + }) + }) + + describe('S3 Data Integrity', () => { + it('should verify data integrity with checksums', async () => { + s3Mock.on(GetObjectCommand).rejects({ name: 'NoSuchKey' }) + + // Track checksums in PUT operations + const putChecksums: string[] = [] + s3Mock.on(PutObjectCommand).callsFake((input) => { + if (input.ChecksumAlgorithm || input.ChecksumCRC32) { + putChecksums.push(input.Key || '') + } + return Promise.resolve({}) + }) + + s3Mock.on(ListObjectsV2Command).resolves({ Contents: [] }) + + brainy = new BrainyData({ + storage: { + s3Storage: { + bucketName: 'test-bucket', + accessKeyId: 'test-key', + secretAccessKey: 'test-secret', + region: 'us-east-1' + } + } + }) + await brainy.init() + + // Add data + await brainy.add('Important data', { critical: true }) + + // Verify checksum was used for critical data + const putCalls = s3Mock.commandCalls(PutObjectCommand) + expect(putCalls.length).toBeGreaterThan(0) + }) + + it('should handle corrupted data gracefully', async () => { + // Return corrupted JSON + s3Mock.on(GetObjectCommand).resolves({ + Body: { + transformToString: async () => '{ corrupt json ][', + transformToByteArray: async () => new TextEncoder().encode('{ corrupt json ][') + } + }) + + s3Mock.on(PutObjectCommand).resolves({}) + s3Mock.on(ListObjectsV2Command).resolves({ Contents: [] }) + + brainy = new BrainyData({ + storage: { + s3Storage: { + bucketName: 'test-bucket', + accessKeyId: 'test-key', + secretAccessKey: 'test-secret', + region: 'us-east-1' + } + } + }) + + // Should handle corrupted data without crashing + await expect(brainy.init()).resolves.not.toThrow() + }) + }) + + describe('S3 Cleanup Operations', () => { + it('should properly clean up S3 objects on delete', async () => { + s3Mock.on(GetObjectCommand).rejects({ name: 'NoSuchKey' }) + s3Mock.on(PutObjectCommand).resolves({}) + s3Mock.on(DeleteObjectCommand).resolves({}) + s3Mock.on(DeleteObjectsCommand).resolves({}) + s3Mock.on(ListObjectsV2Command).resolves({ Contents: [] }) + + brainy = new BrainyData({ + storage: { + s3Storage: { + bucketName: 'test-bucket', + accessKeyId: 'test-key', + secretAccessKey: 'test-secret', + region: 'us-east-1' + } + } + }) + await brainy.init() + + // Add and then delete item + const id = await brainy.add('To be deleted', { temporary: true }) + await brainy.delete(id) + + // Verify delete was called + const deleteCalls = s3Mock.commandCalls(DeleteObjectCommand) + expect(deleteCalls.length).toBeGreaterThan(0) + }) + + it('should batch delete operations for efficiency', async () => { + s3Mock.on(GetObjectCommand).rejects({ name: 'NoSuchKey' }) + s3Mock.on(PutObjectCommand).resolves({}) + s3Mock.on(DeleteObjectsCommand).resolves({}) + s3Mock.on(ListObjectsV2Command).resolves({ Contents: [] }) + + brainy = new BrainyData({ + storage: { + s3Storage: { + bucketName: 'test-bucket', + accessKeyId: 'test-key', + secretAccessKey: 'test-secret', + region: 'us-east-1' + } + } + }) + await brainy.init() + + // Clear all data + await brainy.clear() + + // Verify batch delete was used + const batchDeleteCalls = s3Mock.commandCalls(DeleteObjectsCommand) + // Clear operation should use batch delete + expect(batchDeleteCalls.length).toBeGreaterThanOrEqual(0) + }) + }) +}) \ No newline at end of file diff --git a/tests/s3-statistics-critical.test.ts b/tests/s3-statistics-critical.test.ts new file mode 100644 index 00000000..0c444525 --- /dev/null +++ b/tests/s3-statistics-critical.test.ts @@ -0,0 +1,458 @@ +/** + * CRITICAL S3 Statistics Tests + * + * These tests ensure that statistics work correctly at scale with S3 storage. + * Statistics are fundamental for monitoring production deployments with millions of records. + */ + +import { describe, it, expect, beforeEach, afterEach, vi } from 'vitest' +import { mockClient } from 'aws-sdk-client-mock' +import { + S3Client, + GetObjectCommand, + PutObjectCommand, + ListObjectsV2Command, + HeadObjectCommand, + DeleteObjectCommand +} from '@aws-sdk/client-s3' +import { BrainyData } from '../src/index.js' +import { S3CompatibleStorage } from '../src/storage/adapters/s3CompatibleStorage.js' +import { createMockEmbeddingFunction } from './test-utils.js' + +// Create S3 mock +const s3Mock = mockClient(S3Client) + +describe('CRITICAL: S3 Statistics at Scale', () => { + let brainy: BrainyData + let storage: S3CompatibleStorage + + beforeEach(() => { + // Reset all mocks before each test + s3Mock.reset() + vi.clearAllTimers() + vi.useFakeTimers() + }) + + afterEach(async () => { + if (brainy) { + await brainy.clear() + } + vi.useRealTimers() + }) + + describe('Statistics Persistence and Recovery', () => { + it('should persist statistics to S3 and recover after restart', async () => { + // Setup S3 mock responses + const statisticsData = { + nounCount: { 'service-a': 1000, 'service-b': 500 }, + verbCount: { 'service-a': 200, 'service-b': 100 }, + metadataCount: { 'service-a': 1000, 'service-b': 500 }, + hnswIndexSize: 1500, + lastUpdated: new Date().toISOString() + } + + // Mock initial empty state + s3Mock.on(GetObjectCommand).rejectsOnce({ name: 'NoSuchKey' }) + + // Mock successful save + s3Mock.on(PutObjectCommand).resolves({}) + + // Initialize Brainy with S3 storage + brainy = new BrainyData({ + embeddingFunction: createMockEmbeddingFunction(), + storage: { + s3Storage: { + bucketName: 'test-bucket', + accessKeyId: 'test-key', + secretAccessKey: 'test-secret', + region: 'us-east-1' + } + } + }) + await brainy.init() + + // Add data with different services + await brainy.add('Test data 1', { metadata: 'test1' }, { service: 'service-a' }) + await brainy.add('Test data 2', { metadata: 'test2' }, { service: 'service-b' }) + + // Force statistics flush + await brainy.flushStatistics() + + // Verify statistics were saved to S3 + const putCalls = s3Mock.commandCalls(PutObjectCommand) + expect(putCalls.length).toBeGreaterThan(0) + + // Get current statistics + const stats = await brainy.getStatistics() + expect(stats.nounCount).toBe(2) + expect(stats.serviceBreakdown).toBeDefined() + expect(stats.serviceBreakdown['service-a'].nounCount).toBe(1) + expect(stats.serviceBreakdown['service-b'].nounCount).toBe(1) + + // Simulate restart by creating new instance + s3Mock.reset() + + // Mock loading saved statistics + s3Mock.on(GetObjectCommand).resolves({ + Body: { + transformToString: async () => JSON.stringify({ + nounCount: { 'service-a': 1, 'service-b': 1 }, + verbCount: { 'service-a': 0, 'service-b': 0 }, + metadataCount: { 'service-a': 1, 'service-b': 1 }, + hnswIndexSize: 2, + lastUpdated: new Date().toISOString() + }) + } + }) + + const brainy2 = new BrainyData({ + embeddingFunction: createMockEmbeddingFunction(), + storage: { + s3Storage: { + bucketName: 'test-bucket', + accessKeyId: 'test-key', + secretAccessKey: 'test-secret', + region: 'us-east-1' + } + } + }) + await brainy2.init() + + // Verify statistics were recovered + const recoveredStats = await brainy2.getStatistics() + expect(recoveredStats.nounCount).toBe(2) + expect(recoveredStats.serviceBreakdown['service-a'].nounCount).toBe(1) + expect(recoveredStats.serviceBreakdown['service-b'].nounCount).toBe(1) + }) + }) + + describe('Batching and Throttling', () => { + it('should batch statistics updates to prevent S3 rate limits', async () => { + s3Mock.on(GetObjectCommand).rejectsOnce({ name: 'NoSuchKey' }) + s3Mock.on(PutObjectCommand).resolves({}) + + brainy = new BrainyData({ + embeddingFunction: createMockEmbeddingFunction(), + storage: { + s3Storage: { + bucketName: 'test-bucket', + accessKeyId: 'test-key', + secretAccessKey: 'test-secret', + region: 'us-east-1' + } + } + }) + await brainy.init() + + // Add multiple items rapidly (simulating high load) + const promises = [] + for (let i = 0; i < 100; i++) { + promises.push( + brainy.add(`Item ${i}`, { index: i }, { service: `service-${i % 5}` }) + ) + } + await Promise.all(promises) + + // Initially, statistics shouldn't be flushed immediately + const initialPutCalls = s3Mock.commandCalls(PutObjectCommand) + expect(initialPutCalls.length).toBeLessThan(100) // Should batch, not 100 individual calls + + // Advance timers to trigger batch flush + vi.advanceTimersByTime(5000) + + // Force flush to ensure all statistics are saved + await brainy.flushStatistics() + + // Verify statistics are correct + const stats = await brainy.getStatistics() + expect(stats.nounCount).toBe(100) + + // Verify batching occurred (much fewer S3 calls than items) + const finalPutCalls = s3Mock.commandCalls(PutObjectCommand) + expect(finalPutCalls.length).toBeLessThan(20) // Should be batched + }) + + it('should handle S3 throttling (429) gracefully', async () => { + s3Mock.on(GetObjectCommand).rejectsOnce({ name: 'NoSuchKey' }) + + // Simulate throttling on first attempt, success on retry + let putAttempts = 0 + s3Mock.on(PutObjectCommand).callsFake(() => { + putAttempts++ + if (putAttempts === 1) { + const error: any = new Error('Too Many Requests') + error.name = 'TooManyRequestsException' + error.$metadata = { httpStatusCode: 429 } + throw error + } + return Promise.resolve({}) + }) + + brainy = new BrainyData({ + embeddingFunction: createMockEmbeddingFunction(), + storage: { + s3Storage: { + bucketName: 'test-bucket', + accessKeyId: 'test-key', + secretAccessKey: 'test-secret', + region: 'us-east-1' + } + } + }) + await brainy.init() + + // Add data + await brainy.add('Test data', { metadata: 'test' }, { service: 'throttle-test' }) + + // Force flush - should retry on throttling + await brainy.flushStatistics() + + // Verify retry occurred + expect(putAttempts).toBeGreaterThanOrEqual(1) + + // Verify statistics were eventually saved + const stats = await brainy.getStatistics() + expect(stats.nounCount).toBe(1) + }) + }) + + describe('Time-Based Partitioning', () => { + it('should partition statistics by date to avoid single-key rate limits', async () => { + s3Mock.on(GetObjectCommand).rejects({ name: 'NoSuchKey' }) + s3Mock.on(PutObjectCommand).resolves({}) + s3Mock.on(ListObjectsV2Command).resolves({ Contents: [] }) + + brainy = new BrainyData({ + embeddingFunction: createMockEmbeddingFunction(), + storage: { + s3Storage: { + bucketName: 'test-bucket', + accessKeyId: 'test-key', + secretAccessKey: 'test-secret', + region: 'us-east-1' + } + } + }) + await brainy.init() + + // Add data across different time periods + const today = new Date() + await brainy.add('Today item', { date: today }, { service: 'time-test' }) + await brainy.flushStatistics() + + // Check that statistics are saved with date-based key + const putCalls = s3Mock.commandCalls(PutObjectCommand) + const statisticsCall = putCalls.find(call => + call.args[0].input.Key?.includes('_system/statistics') + ) + + expect(statisticsCall).toBeDefined() + // Should include date in the key for partitioning + const key = statisticsCall?.args[0].input.Key + expect(key).toContain(today.toISOString().split('T')[0]) + }) + }) + + describe('Backward Compatibility', () => { + it('should read legacy statistics format correctly', async () => { + // Mock legacy statistics format (without service breakdown) + const legacyStats = { + nounCount: 500, + verbCount: 100, + metadataCount: 500, + hnswIndexSize: 500, + lastUpdated: new Date().toISOString() + } + + s3Mock.on(GetObjectCommand).resolves({ + Body: { + transformToString: async () => JSON.stringify(legacyStats) + } + }) + + brainy = new BrainyData({ + embeddingFunction: createMockEmbeddingFunction(), + storage: { + s3Storage: { + bucketName: 'test-bucket', + accessKeyId: 'test-key', + secretAccessKey: 'test-secret', + region: 'us-east-1' + } + } + }) + await brainy.init() + + // Should handle legacy format gracefully + const stats = await brainy.getStatistics() + expect(stats).toBeDefined() + // Legacy format should be converted to new format with default service + expect(stats.nounCount).toBe(500) + }) + + it('should migrate legacy statistics to new format on write', async () => { + // Start with legacy format + const legacyStats = { + nounCount: 100, + verbCount: 50, + metadataCount: 100, + hnswIndexSize: 100, + lastUpdated: new Date().toISOString() + } + + s3Mock.on(GetObjectCommand).resolvesOnce({ + Body: { + transformToString: async () => JSON.stringify(legacyStats) + } + }) + s3Mock.on(PutObjectCommand).resolves({}) + + brainy = new BrainyData({ + embeddingFunction: createMockEmbeddingFunction(), + storage: { + s3Storage: { + bucketName: 'test-bucket', + accessKeyId: 'test-key', + secretAccessKey: 'test-secret', + region: 'us-east-1' + } + } + }) + await brainy.init() + + // Add new data + await brainy.add('New data', { metadata: 'new' }, { service: 'migration-test' }) + await brainy.flushStatistics() + + // Verify new format was saved + const putCalls = s3Mock.commandCalls(PutObjectCommand) + const lastPut = putCalls[putCalls.length - 1] + const savedData = JSON.parse(lastPut.args[0].input.Body as string) + + // Should have service-based structure + expect(savedData.nounCount).toBeTypeOf('object') + expect(savedData.nounCount['migration-test']).toBeDefined() + }) + }) + + describe('Concurrent Updates', () => { + it('should handle concurrent statistics updates safely', async () => { + s3Mock.on(GetObjectCommand).rejectsOnce({ name: 'NoSuchKey' }) + s3Mock.on(PutObjectCommand).resolves({}) + + brainy = new BrainyData({ + embeddingFunction: createMockEmbeddingFunction(), + storage: { + s3Storage: { + bucketName: 'test-bucket', + accessKeyId: 'test-key', + secretAccessKey: 'test-secret', + region: 'us-east-1' + } + } + }) + await brainy.init() + + // Simulate concurrent additions from multiple services + const services = ['api', 'worker', 'batch', 'stream', 'webhook'] + const concurrentOps = [] + + for (let i = 0; i < 50; i++) { + const service = services[i % services.length] + concurrentOps.push( + brainy.add(`Data ${i}`, { index: i }, { service }) + ) + } + + // Add relationships concurrently + await Promise.all(concurrentOps) + + const ids = concurrentOps.map((_, i) => `id-${i}`) + const relationOps = [] + for (let i = 0; i < 10; i++) { + relationOps.push( + brainy.relate( + ids[i], + ids[i + 10], + 'related', + { service: services[i % services.length] } + ).catch(() => {}) // Ignore if IDs don't exist + ) + } + + await Promise.all(relationOps) + await brainy.flushStatistics() + + // Verify all operations were counted correctly + const stats = await brainy.getStatistics() + expect(stats.nounCount).toBe(50) + + // Verify per-service counts + for (const service of services) { + expect(stats.serviceBreakdown[service]).toBeDefined() + expect(stats.serviceBreakdown[service].nounCount).toBe(10) // 50 items / 5 services + } + }) + }) + + describe('Large Scale Statistics', () => { + it('should handle statistics for millions of records efficiently', async () => { + // Mock large existing statistics + const largeStats = { + nounCount: { + 'service-1': 1000000, + 'service-2': 2000000, + 'service-3': 1500000 + }, + verbCount: { + 'service-1': 500000, + 'service-2': 750000, + 'service-3': 600000 + }, + metadataCount: { + 'service-1': 1000000, + 'service-2': 2000000, + 'service-3': 1500000 + }, + hnswIndexSize: 4500000, + lastUpdated: new Date().toISOString() + } + + s3Mock.on(GetObjectCommand).resolves({ + Body: { + transformToString: async () => JSON.stringify(largeStats) + } + }) + s3Mock.on(PutObjectCommand).resolves({}) + + brainy = new BrainyData({ + embeddingFunction: createMockEmbeddingFunction(), + storage: { + s3Storage: { + bucketName: 'test-bucket', + accessKeyId: 'test-key', + secretAccessKey: 'test-secret', + region: 'us-east-1' + } + } + }) + await brainy.init() + + // Get statistics for large dataset + const stats = await brainy.getStatistics() + + // Verify large numbers are handled correctly + expect(stats.nounCount).toBe(4500000) + expect(stats.verbCount).toBe(1850000) + + // Add more data to large dataset + await brainy.add('New item in large dataset', {}, { service: 'service-1' }) + await brainy.flushStatistics() + + // Verify increment worked correctly with large numbers + const updatedStats = await brainy.getStatistics() + expect(updatedStats.nounCount).toBe(4500001) + expect(updatedStats.serviceBreakdown['service-1'].nounCount).toBe(1000001) + }) + }) +}) \ No newline at end of file diff --git a/tests/s3-storage.test.ts b/tests/s3-storage.test.ts new file mode 100644 index 00000000..c7fa33f4 --- /dev/null +++ b/tests/s3-storage.test.ts @@ -0,0 +1,512 @@ +/** + * S3 Compatible Storage Tests + * Tests for the S3 compatible storage adapter using a simulated S3 environment + */ + +import { describe, it, expect, beforeEach, afterEach, vi } from 'vitest' +import { setupS3Mock, cleanupS3Mock, S3Commands } from './mocks/s3-mock' +import { Vector } from '../src/coreTypes' + +// Setup S3 mock environment at the top level +console.log('Setting up S3 mock environment at the top level') +const s3MockSetup = setupS3Mock() + +// Mock AWS SDK imports at the top level +vi.mock('@aws-sdk/client-s3', () => { + console.log('Mocking AWS SDK imports') + return { + S3Client: class MockS3Client { + send = s3MockSetup.mockS3Client.send + }, + ...S3Commands + } +}) + +describe('S3CompatibleStorage', () => { + // Import modules inside tests to avoid issues with dynamic imports + let S3CompatibleStorage: any + let R2Storage: any + let s3Mock: any + + beforeEach(async () => { + console.log('==== TEST SETUP START ====') + + // Store the mock setup for use in tests + s3Mock = s3MockSetup + + // Reset the mock storage before each test + s3Mock.reset() + + // Import storage factory + console.log('Importing storage factory') + const storageFactory = await import('../src/storage/storageFactory.js') + S3CompatibleStorage = storageFactory.S3CompatibleStorage + R2Storage = storageFactory.R2Storage + + console.log('==== TEST SETUP COMPLETE ====') + }) + + afterEach(() => { + console.log('==== TEST CLEANUP START ====') + + // Clean up S3 mock environment + cleanupS3Mock() + + // Reset mocks + vi.resetAllMocks() + vi.clearAllMocks() + + console.log('==== TEST CLEANUP COMPLETE ====') + }) + + it('should initialize S3CompatibleStorage correctly', async () => { + // Create the bucket first using our mock + const createBucketCommand = new S3Commands.CreateBucketCommand({ + Bucket: 'test-bucket' + }) + await s3Mock.mockS3Client.send(createBucketCommand) + + // Create a new instance with our mocked environment + const s3Storage = new S3CompatibleStorage({ + bucketName: 'test-bucket', + region: 'us-east-1', + accessKeyId: 'test-access-key', + secretAccessKey: 'test-secret-key', + serviceType: 's3' + }) + + // Initialize the storage + await s3Storage.init() + + // Verify the storage was initialized correctly + expect(s3Storage).toBeDefined() + + // Clean up + await s3Storage.clear() + }) + + it('should initialize R2Storage correctly', async () => { + // Create the bucket first using our mock + const createBucketCommand = new S3Commands.CreateBucketCommand({ + Bucket: 'test-bucket' + }) + await s3Mock.mockS3Client.send(createBucketCommand) + + // Create a new instance with our mocked environment + const r2Storage = new R2Storage({ + bucketName: 'test-bucket', + accountId: 'test-account', + accessKeyId: 'test-access-key', + secretAccessKey: 'test-secret-key' + }) + + // Initialize the storage + await r2Storage.init() + + // Verify the storage was initialized correctly + expect(r2Storage).toBeDefined() + + // Clean up + await r2Storage.clear() + }) + + it('should perform basic metadata operations with S3 storage', async () => { + // Create the bucket first using our mock + const createBucketCommand = new S3Commands.CreateBucketCommand({ + Bucket: 'test-bucket' + }) + await s3Mock.mockS3Client.send(createBucketCommand) + + // Create a new instance with our mocked environment + const s3Storage = new S3CompatibleStorage({ + bucketName: 'test-bucket', + region: 'us-east-1', + accessKeyId: 'test-access-key', + secretAccessKey: 'test-secret-key', + serviceType: 's3' + }) + + // Initialize the storage + await s3Storage.init() + + // Test basic metadata operations + const testMetadata = { test: 'data', value: 123 } + await s3Storage.saveMetadata('test-key', testMetadata) + + const retrievedMetadata = await s3Storage.getMetadata('test-key') + expect(retrievedMetadata).toEqual(testMetadata) + + // Clean up + await s3Storage.clear() + }) + + it('should handle noun operations correctly with S3 storage', async () => { + // Create the bucket first using our mock + const createBucketCommand = new S3Commands.CreateBucketCommand({ + Bucket: 'test-bucket' + }) + await s3Mock.mockS3Client.send(createBucketCommand) + + // Create a new instance with our mocked environment + const s3Storage = new S3CompatibleStorage({ + bucketName: 'test-bucket', + region: 'us-east-1', + accessKeyId: 'test-access-key', + secretAccessKey: 'test-secret-key', + serviceType: 's3' + }) + + // Initialize the storage + await s3Storage.init() + + // Create test noun + const testVector: Vector = [0.1, 0.2, 0.3, 0.4, 0.5] + const testNoun = { + id: 'test-noun-1', + vector: testVector, + connections: new Map([ + [0, new Set(['test-noun-2', 'test-noun-3'])] + ]) + } + + // Save the noun + await s3Storage.saveNoun(testNoun) + + // Retrieve the noun + const retrievedNoun = await s3Storage.getNoun('test-noun-1') + + // Verify the noun was saved and retrieved correctly + expect(retrievedNoun).toBeDefined() + expect(retrievedNoun?.id).toBe('test-noun-1') + expect(retrievedNoun?.vector).toEqual(testVector) + + // Verify connections were saved correctly + // Note: connections are stored as a Map in memory but might be serialized differently + expect(retrievedNoun?.connections).toBeDefined() + expect(retrievedNoun?.connections.get(0)).toBeDefined() + expect(retrievedNoun?.connections.get(0)?.has('test-noun-2')).toBe(true) + expect(retrievedNoun?.connections.get(0)?.has('test-noun-3')).toBe(true) + + // Test getNouns with pagination + const nounsResult = await s3Storage.getNouns({ pagination: { limit: 100 } }) + expect(nounsResult.items.length).toBe(1) + expect(nounsResult.items[0].id).toBe('test-noun-1') + + // Test deleteNoun + await s3Storage.deleteNoun('test-noun-1') + const deletedNoun = await s3Storage.getNoun('test-noun-1') + expect(deletedNoun).toBeNull() + + // Clean up + await s3Storage.clear() + }) + + it('should handle verb operations correctly with S3 storage', async () => { + // Create the bucket first using our mock + const createBucketCommand = new S3Commands.CreateBucketCommand({ + Bucket: 'test-bucket' + }) + await s3Mock.mockS3Client.send(createBucketCommand) + + // Create a new instance with our mocked environment + const s3Storage = new S3CompatibleStorage({ + bucketName: 'test-bucket', + region: 'us-east-1', + accessKeyId: 'test-access-key', + secretAccessKey: 'test-secret-key', + serviceType: 's3' + }) + + // Initialize the storage + await s3Storage.init() + + // Create test verb + const testVector: Vector = [0.1, 0.2, 0.3, 0.4, 0.5] + const timestamp = { + seconds: Math.floor(Date.now() / 1000), + nanoseconds: (Date.now() % 1000) * 1000000 + } + const testVerb = { + id: 'test-verb-1', + vector: testVector, + connections: new Map(), + source: 'source-noun-1', + target: 'target-noun-1', + verb: 'test-relation', + weight: 0.75, + metadata: { description: 'Test relation' }, + createdAt: timestamp, + updatedAt: timestamp, + createdBy: { + augmentation: 'test-service', + version: '1.0' + } + } + + // Save the verb + await s3Storage.saveVerb(testVerb) + + // Retrieve the verb + const retrievedVerb = await s3Storage.getVerb('test-verb-1') + + // Verify the verb was saved and retrieved correctly + expect(retrievedVerb).toBeDefined() + expect(retrievedVerb?.id).toBe('test-verb-1') + expect(retrievedVerb?.vector).toEqual(testVector) + expect(retrievedVerb?.sourceId).toBe('source-noun-1') + expect(retrievedVerb?.targetId).toBe('target-noun-1') + expect(retrievedVerb?.type).toBe('test-relation') + expect(retrievedVerb?.weight).toBe(0.75) + expect(retrievedVerb?.metadata).toEqual({ description: 'Test relation' }) + + // Test getVerbs with pagination + const verbsResult = await s3Storage.getVerbs({ pagination: { limit: 100 } }) + expect(verbsResult.items.length).toBe(1) + expect(verbsResult.items[0].id).toBe('test-verb-1') + + // Test getVerbsBySource + const verbsBySource = await s3Storage.getVerbsBySource('source-noun-1') + expect(verbsBySource.length).toBe(1) + expect(verbsBySource[0].id).toBe('test-verb-1') + + // Test getVerbsByTarget + const verbsByTarget = await s3Storage.getVerbsByTarget('target-noun-1') + expect(verbsByTarget.length).toBe(1) + expect(verbsByTarget[0].id).toBe('test-verb-1') + + // Test getVerbsByType + const verbsByType = await s3Storage.getVerbsByType('test-relation') + expect(verbsByType.length).toBe(1) + expect(verbsByType[0].id).toBe('test-verb-1') + + // Test deleteVerb + await s3Storage.deleteVerb('test-verb-1') + const deletedVerb = await s3Storage.getVerb('test-verb-1') + expect(deletedVerb).toBeNull() + + // Clean up + await s3Storage.clear() + }) + + it('should handle storage status correctly with S3 storage', async () => { + // Create the bucket first using our mock + const createBucketCommand = new S3Commands.CreateBucketCommand({ + Bucket: 'test-bucket' + }) + await s3Mock.mockS3Client.send(createBucketCommand) + + // Create a new instance with our mocked environment + const s3Storage = new S3CompatibleStorage({ + bucketName: 'test-bucket', + region: 'us-east-1', + accessKeyId: 'test-access-key', + secretAccessKey: 'test-secret-key', + serviceType: 's3' + }) + + // Initialize the storage + await s3Storage.init() + + // Add some data to the storage + const testVector: Vector = [0.1, 0.2, 0.3, 0.4, 0.5] + const testNoun = { + id: 'test-noun-1', + vector: testVector, + connections: new Map([ + [0, new Set(['test-noun-2', 'test-noun-3'])] + ]) + } + + await s3Storage.saveNoun(testNoun) + await s3Storage.saveMetadata('test-key', { test: 'data', value: 123 }) + + // Get storage status + const status = await s3Storage.getStorageStatus() + + // Verify status + expect(status.type).toBe('s3') + expect(status.used).toBeGreaterThan(0) + + // Clean up + await s3Storage.clear() + }) + + it('should handle multiple objects and pagination with S3 storage', async () => { + // Create the bucket first using our mock + const createBucketCommand = new S3Commands.CreateBucketCommand({ + Bucket: 'test-bucket' + }) + await s3Mock.mockS3Client.send(createBucketCommand) + + // Create a new instance with our mocked environment + const s3Storage = new S3CompatibleStorage({ + bucketName: 'test-bucket', + region: 'us-east-1', + accessKeyId: 'test-access-key', + secretAccessKey: 'test-secret-key', + serviceType: 's3' + }) + + // Initialize the storage + await s3Storage.init() + + // Create multiple test nouns + const testVector: Vector = [0.1, 0.2, 0.3, 0.4, 0.5] + const nounCount = 10 + + for (let i = 0; i < nounCount; i++) { + const testNoun = { + id: `test-noun-${i}`, + vector: testVector, + connections: new Map([ + [0, new Set([`test-noun-${(i + 1) % nounCount}`, `test-noun-${(i + 2) % nounCount}`])] + ]) + } + + await s3Storage.saveNoun(testNoun) + } + + // Test getNouns with pagination + const nounsResult = await s3Storage.getNouns({ pagination: { limit: 100 } }) + expect(nounsResult.items.length).toBe(nounCount) + + // Clean up + await s3Storage.clear() + }) + + it('should handle change log functionality correctly with S3 storage', async () => { + // Create the bucket first using our mock + const createBucketCommand = new S3Commands.CreateBucketCommand({ + Bucket: 'test-bucket' + }) + await s3Mock.mockS3Client.send(createBucketCommand) + + // Create a new instance with our mocked environment + const s3Storage = new S3CompatibleStorage({ + bucketName: 'test-bucket', + region: 'us-east-1', + accessKeyId: 'test-access-key', + secretAccessKey: 'test-secret-key', + serviceType: 's3' + }) + + // Initialize the storage + await s3Storage.init() + + // Create test nouns that will generate change log entries + const testVector: Vector = [0.1, 0.2, 0.3, 0.4, 0.5] + + // Save first noun - should create a change log entry + const testNoun1 = { + id: 'test-noun-change-1', + vector: testVector, + connections: new Map() + } + await s3Storage.saveNoun(testNoun1) + + // Wait a bit to ensure timestamps are different + await new Promise(resolve => setTimeout(resolve, 10)) + const firstTimestamp = Date.now() + + // Wait a bit more before creating second noun + await new Promise(resolve => setTimeout(resolve, 10)) + + // Save second noun - should create another change log entry + const testNoun2 = { + id: 'test-noun-change-2', + vector: testVector, + connections: new Map() + } + await s3Storage.saveNoun(testNoun2) + + // Get changes since beginning of time + const allChanges = await s3Storage.getChangesSince(0) + + // Verify we have at least 2 changes (might be more due to initialization) + expect(allChanges.length).toBeGreaterThanOrEqual(2) + + // Verify the changes contain our noun operations + const nounChanges = allChanges.filter(c => + c.entityType === 'noun' && + (c.entityId === 'test-noun-change-1' || c.entityId === 'test-noun-change-2') + ) + expect(nounChanges.length).toBe(2) + + // Get changes since first timestamp - should only include the second noun + const recentChanges = await s3Storage.getChangesSince(firstTimestamp) + const recentNounChanges = recentChanges.filter(c => + c.entityType === 'noun' && c.entityId === 'test-noun-change-2' + ) + expect(recentNounChanges.length).toBe(1) + + // Clean up + await s3Storage.clear() + }) + + it('should handle change log functionality with getChangesSince', async () => { + // Create the bucket first using our mock + const createBucketCommand = new S3Commands.CreateBucketCommand({ + Bucket: 'test-bucket' + }) + await s3Mock.mockS3Client.send(createBucketCommand) + + // Create a new instance with our mocked environment + const s3Storage = new S3CompatibleStorage({ + bucketName: 'test-bucket', + region: 'us-east-1', + accessKeyId: 'test-access-key', + secretAccessKey: 'test-secret-key', + serviceType: 's3' + }) + + // Initialize the storage + await s3Storage.init() + + // Create test vector + const testVector: Vector = [0.1, 0.2, 0.3, 0.4, 0.5] + + // Create first noun (older entry) + const testNoun1 = { + id: 'test-noun-cleanup-1', + vector: testVector, + connections: new Map() + } + await s3Storage.saveNoun(testNoun1) + + // Record the timestamp after first noun + const oldTimestamp = Date.now() + + // Wait to ensure timestamps are different + await new Promise(resolve => setTimeout(resolve, 100)) + + // Create second noun (newer entry) + const testNoun2 = { + id: 'test-noun-cleanup-2', + vector: testVector, + connections: new Map() + } + await s3Storage.saveNoun(testNoun2) + + // Verify we have both change log entries + const allChanges = await s3Storage.getChangesSince(0) + const testNounChanges = allChanges.filter(c => + c.entityType === 'noun' && + (c.entityId === 'test-noun-cleanup-1' || c.entityId === 'test-noun-cleanup-2') + ) + expect(testNounChanges.length).toBe(2) + + // Get changes since oldTimestamp - should only include the second noun + const recentChanges = await s3Storage.getChangesSince(oldTimestamp) + const recentNounChanges = recentChanges.filter(c => + c.entityType === 'noun' && + (c.entityId === 'test-noun-cleanup-1' || c.entityId === 'test-noun-cleanup-2') + ) + + // Should only have the newer entry for test-noun-cleanup-2 + expect(recentNounChanges.length).toBe(1) + expect(recentNounChanges[0].entityId).toBe('test-noun-cleanup-2') + + // Clean up + await s3Storage.clear() + }) +}) diff --git a/tests/service-statistics.test.ts b/tests/service-statistics.test.ts new file mode 100644 index 00000000..39744115 --- /dev/null +++ b/tests/service-statistics.test.ts @@ -0,0 +1,411 @@ +/** + * Tests for per-service statistics tracking functionality + */ + +import { describe, it, expect, beforeEach, afterEach } from 'vitest' +import { BrainyData } from '../src/brainyData.js' +import { ServiceStatistics } from '../src/coreTypes.js' + +describe('Per-Service Statistics', () => { + let brainy: BrainyData + + beforeEach(async () => { + // Create a new instance with a default service + brainy = new BrainyData({ + defaultService: 'test-service', + storage: { + forceMemoryStorage: true + } + }) + await brainy.init() + }) + + afterEach(async () => { + // Cleanup + if (brainy) { + await brainy.clear() + } + }) + + describe('Service Tracking', () => { + it('should track data by default service', async () => { + // Add data using default service + await brainy.add({ content: 'test data 1' }, { noun: 'test' }) + await brainy.add({ content: 'test data 2' }, { noun: 'test' }) + + const stats = await brainy.getStatistics() + expect(stats.serviceBreakdown).toBeDefined() + expect(stats.serviceBreakdown?.['test-service']).toBeDefined() + expect(stats.serviceBreakdown?.['test-service'].nounCount).toBe(2) + }) + + it('should track data by explicit service', async () => { + // Add data with explicit service + await brainy.add( + { content: 'github data' }, + { noun: 'repository' }, + { service: 'github-package' } + ) + + await brainy.add( + { content: 'bluesky data' }, + { noun: 'post' }, + { service: 'bluesky-package' } + ) + + const stats = await brainy.getStatistics() + expect(stats.serviceBreakdown?.['github-package'].nounCount).toBe(1) + expect(stats.serviceBreakdown?.['bluesky-package'].nounCount).toBe(1) + }) + + it('should track verbs by service', async () => { + // Add nouns first + const id1 = await brainy.add( + { content: 'user 1' }, + { noun: 'Person' }, + { service: 'social-service' } + ) + + const id2 = await brainy.add( + { content: 'user 2' }, + { noun: 'Person' }, + { service: 'social-service' } + ) + + // Create relationship + await brainy.relate( + id1, + id2, + { verb: 'follows' }, + { service: 'social-service' } + ) + + const stats = await brainy.getStatistics() + expect(stats.serviceBreakdown?.['social-service'].verbCount).toBe(1) + expect(stats.serviceBreakdown?.['social-service'].nounCount).toBe(2) + }) + }) + + describe('listServices()', () => { + it('should list all services that have written data', async () => { + // Add data from multiple services + await brainy.add( + { content: 'data 1' }, + { noun: 'Document' }, + { service: 'service-a' } + ) + + await brainy.add( + { content: 'data 2' }, + { noun: 'Document' }, + { service: 'service-b' } + ) + + await brainy.add( + { content: 'data 3' }, + { noun: 'Document' }, + { service: 'service-a' } + ) + + const services = await brainy.listServices() + + expect(services).toHaveLength(2) + expect(services.map(s => s.name)).toContain('service-a') + expect(services.map(s => s.name)).toContain('service-b') + + const serviceA = services.find(s => s.name === 'service-a') + expect(serviceA?.totalNouns).toBe(2) + + const serviceB = services.find(s => s.name === 'service-b') + expect(serviceB?.totalNouns).toBe(1) + }) + + it('should include service activity timestamps', async () => { + const beforeAdd = new Date() + + await brainy.add( + { content: 'test data' }, + { noun: 'Document' }, + { service: 'timestamped-service' } + ) + + const afterAdd = new Date() + + const services = await brainy.listServices() + const service = services.find(s => s.name === 'timestamped-service') + + expect(service).toBeDefined() + expect(service?.status).toBe('active') + + // Check if timestamps are present (they may not be if the storage adapter doesn't support them) + if (service?.lastActivity) { + const lastActivityTime = new Date(service.lastActivity).getTime() + expect(lastActivityTime).toBeGreaterThanOrEqual(beforeAdd.getTime()) + expect(lastActivityTime).toBeLessThanOrEqual(afterAdd.getTime()) + } + }) + + it('should determine service status correctly', async () => { + // Add data from an active service + await brainy.add( + { content: 'recent data' }, + { noun: 'Document' }, + { service: 'active-service' } + ) + + const services = await brainy.listServices() + const activeService = services.find(s => s.name === 'active-service') + + // Should be marked as active if it has recent activity + expect(activeService?.status).toBe('active') + + // Service with no write operations should be marked as read-only + // This would need to be tested with a service that only reads + }) + }) + + describe('getServiceStatistics()', () => { + it('should return statistics for a specific service', async () => { + // Add data for specific service + await brainy.add( + { content: 'data 1' }, + { noun: 'Document' }, + { service: 'target-service' } + ) + + await brainy.add( + { content: 'data 2' }, + { noun: 'Document' }, + { service: 'target-service' } + ) + + await brainy.add( + { content: 'data 3' }, + { noun: 'Document' }, + { service: 'other-service' } + ) + + const serviceStats = await brainy.getServiceStatistics('target-service') + + expect(serviceStats).toBeDefined() + expect(serviceStats?.name).toBe('target-service') + expect(serviceStats?.totalNouns).toBe(2) + }) + + it('should return null for non-existent service', async () => { + const serviceStats = await brainy.getServiceStatistics('non-existent') + expect(serviceStats).toBeNull() + }) + + it('should include operation counts when available', async () => { + // Add multiple operations + const id = await brainy.add( + { content: 'data' }, + { noun: 'Document' }, + { service: 'ops-service' } + ) + + await brainy.updateMetadata( + id, + { content: 'updated data', noun: 'Document' }, + { service: 'ops-service' } + ) + + const serviceStats = await brainy.getServiceStatistics('ops-service') + + expect(serviceStats).toBeDefined() + expect(serviceStats?.totalNouns).toBe(1) + expect(serviceStats?.totalMetadata).toBeGreaterThanOrEqual(1) + + // Operations tracking depends on whether the storage adapter tracks them + if (serviceStats?.operations) { + expect(serviceStats.operations.adds).toBeGreaterThanOrEqual(1) + } + }) + }) + + describe('Service-Filtered getStatistics()', () => { + it('should filter statistics by single service', async () => { + // Add data from multiple services + await brainy.add( + { content: 'data 1' }, + { noun: 'Document' }, + { service: 'service-a' } + ) + + await brainy.add( + { content: 'data 2' }, + { noun: 'Document' }, + { service: 'service-b' } + ) + + await brainy.add( + { content: 'data 3' }, + { noun: 'Document' }, + { service: 'service-a' } + ) + + const statsA = await brainy.getStatistics({ service: 'service-a' }) + expect(statsA.nounCount).toBe(2) + + const statsB = await brainy.getStatistics({ service: 'service-b' }) + expect(statsB.nounCount).toBe(1) + }) + + it('should filter statistics by multiple services', async () => { + // Add data from multiple services + await brainy.add( + { content: 'data 1' }, + { noun: 'Document' }, + { service: 'service-a' } + ) + + await brainy.add( + { content: 'data 2' }, + { noun: 'Document' }, + { service: 'service-b' } + ) + + await brainy.add( + { content: 'data 3' }, + { noun: 'Document' }, + { service: 'service-c' } + ) + + const stats = await brainy.getStatistics({ + service: ['service-a', 'service-b'] + }) + + expect(stats.nounCount).toBe(2) + expect(stats.serviceBreakdown?.['service-a'].nounCount).toBe(1) + expect(stats.serviceBreakdown?.['service-b'].nounCount).toBe(1) + expect(stats.serviceBreakdown?.['service-c']).toBeUndefined() + }) + }) + + describe('Service-Filtered Queries', () => { + beforeEach(async () => { + // Add test data from different services + await brainy.add( + { content: 'github repository data' }, + { noun: 'Repository', createdBy: { augmentation: 'github-service' } }, + { service: 'github-service' } + ) + + await brainy.add( + { content: 'bluesky post data' }, + { noun: 'Post', createdBy: { augmentation: 'bluesky-service' } }, + { service: 'bluesky-service' } + ) + + await brainy.add( + { content: 'another github repository' }, + { noun: 'Repository', createdBy: { augmentation: 'github-service' } }, + { service: 'github-service' } + ) + }) + + it('should filter search results by service', async () => { + const results = await brainy.search('repository', 10, { + service: 'github-service' + }) + + // Should only return results from github-service + expect(results.length).toBeGreaterThan(0) + results.forEach(result => { + expect(result.metadata?.createdBy?.augmentation).toBe('github-service') + }) + }) + + it('should filter getNouns by service', async () => { + const result = await brainy.getNouns({ + filter: { + service: 'github-service' + } + }) + + // Note: Service filtering in getNouns depends on storage adapter implementation + // The test verifies the API works but actual filtering may vary + expect(result).toBeDefined() + expect(result.items).toBeDefined() + }) + + it('should filter getVerbs by service', async () => { + // Add verbs from different services + const id1 = await brainy.add( + { content: 'user 1' }, + { noun: 'Person' }, + { service: 'social-service' } + ) + + const id2 = await brainy.add( + { content: 'user 2' }, + { noun: 'Person' }, + { service: 'social-service' } + ) + + await brainy.relate( + id1, + id2, + { verb: 'follows' }, + { service: 'social-service' } + ) + + const result = await brainy.getVerbs({ + filter: { + service: 'social-service' + } + }) + + // Note: Service filtering in getVerbs depends on storage adapter implementation + expect(result).toBeDefined() + expect(result.items).toBeDefined() + }) + }) + + describe('Service Metadata on Data', () => { + it('should add service metadata to nouns', async () => { + const id = await brainy.add( + { content: 'test data' }, + { noun: 'Document' }, + { service: 'metadata-service' } + ) + + const doc = await brainy.get(id) + expect(doc).toBeDefined() + + // Service tracking is done in statistics, not directly in metadata + // Verify through statistics instead + const stats = await brainy.getStatistics({ service: 'metadata-service' }) + expect(stats.nounCount).toBe(1) + }) + + it('should track service for verbs', async () => { + const id1 = await brainy.add( + { content: 'node 1' }, + { noun: 'Node' }, + { service: 'graph-service' } + ) + + const id2 = await brainy.add( + { content: 'node 2' }, + { noun: 'Node' }, + { service: 'graph-service' } + ) + + const verbId = await brainy.relate( + id1, + id2, + { verb: 'connects' }, + { service: 'graph-service' } + ) + + expect(verbId).toBeDefined() + + // Verify through statistics + const stats = await brainy.getStatistics({ service: 'graph-service' }) + expect(stats.verbCount).toBe(1) + expect(stats.nounCount).toBe(2) + }) + }) +}) \ No newline at end of file diff --git a/tests/setup.ts b/tests/setup.ts new file mode 100644 index 00000000..355a67f2 --- /dev/null +++ b/tests/setup.ts @@ -0,0 +1,90 @@ +/** + * Simple test setup for Brainy library + * No direct TensorFlow references - patches are handled internally by Brainy + */ + +import { beforeEach, afterEach, afterAll } from 'vitest' +import { existsSync, rmSync } from 'fs' +import { join } from 'path' + +// Define the test utilities type for reuse +type TestUtilsType = { + createTestVector: (dimensions: number) => number[] + timeout: number +} + +// Extend global type definitions for both global and globalThis +declare global { + let testUtils: TestUtilsType | undefined + let __ENV__: any +} + +// Explicitly declare globalThis interface to ensure TypeScript recognizes these properties +declare global { + interface globalThis { + testUtils?: TestUtilsType | undefined + __ENV__?: any + } +} + +// Clean up between tests +beforeEach(() => { + // Clear any global state that might interfere with tests + if (typeof globalThis !== 'undefined' && globalThis.__ENV__) { + delete globalThis.__ENV__ + } + if (typeof global !== 'undefined' && global.__ENV__) { + delete global.__ENV__ + } + + // Clean up test data directory to prevent file accumulation + const testDataDir = join(process.cwd(), 'brainy-data') + if (existsSync(testDataDir)) { + try { + rmSync(testDataDir, { recursive: true, force: true }) + } catch (error) { + // Ignore errors during cleanup + } + } +}) + +// Clean up after each test +afterEach(() => { + // Force garbage collection if available (requires --expose-gc flag) + if (global.gc) { + global.gc() + } +}) + +// Final cleanup after all tests +afterAll(() => { + // Clean up test data directory + const testDataDir = join(process.cwd(), 'brainy-data') + if (existsSync(testDataDir)) { + try { + rmSync(testDataDir, { recursive: true, force: true }) + } catch (error) { + // Ignore errors during cleanup + } + } +}) + +// Add simple test utilities to both global and globalThis for compatibility +const testUtilsObject = { + // Create a simple test vector with predictable values + createTestVector: (dimensions: number): number[] => { + return Array.from({ length: dimensions }, (_, i) => (i + 1) / dimensions) + }, + + // Standard timeout for async operations + timeout: 30000 +} + +global.testUtils = testUtilsObject +globalThis.testUtils = testUtilsObject + +// Set a clear test environment flag for embedding system +globalThis.__BRAINY_TEST_ENV__ = true +if (typeof global !== 'undefined') { + (global as any).__BRAINY_TEST_ENV__ = true +} diff --git a/tests/simple-metadata-test.ts b/tests/simple-metadata-test.ts new file mode 100644 index 00000000..14bb9e34 --- /dev/null +++ b/tests/simple-metadata-test.ts @@ -0,0 +1,51 @@ +// Simple standalone test to check metadata filtering +import { BrainyData } from '../dist/brainyData.js' + +async function testMetadataFiltering() { + console.log('Creating BrainyData instance...') + const brainy = new BrainyData({ + storage: { forceMemoryStorage: true }, + hnsw: { M: 4, efConstruction: 20 }, + logging: { verbose: true } + }) + + console.log('Initializing...') + await brainy.init() + + console.log('Adding test data...') + await brainy.add('Senior developer Alice', { level: 'senior', name: 'Alice' }) + await brainy.add('Junior developer Bob', { level: 'junior', name: 'Bob' }) + + console.log('Searching without filter...') + const allResults = await brainy.searchText('developer', 10) + console.log('All results:', allResults.map(r => ({ + metadata: r.metadata, + score: r.score.toFixed(3) + }))) + + // Check if metadata index is available + console.log('Metadata index available?', !!brainy.metadataIndex) + + console.log('Searching with metadata filter...') + const seniorResults = await brainy.searchText('developer', 10, { + metadata: { level: 'senior' } + }) + console.log('Senior results:', seniorResults.map(r => ({ + metadata: r.metadata, + score: r.score.toFixed(3) + }))) + + if (brainy.metadataIndex) { + console.log('Checking metadata index for level:senior...') + const levelSeniorIds = await brainy.metadataIndex.getIds('level', 'senior') + console.log('IDs with level=senior from index:', levelSeniorIds) + + const levelJuniorIds = await brainy.metadataIndex.getIds('level', 'junior') + console.log('IDs with level=junior from index:', levelJuniorIds) + } + + console.log(`\nResults: All=${allResults.length}, Senior=${seniorResults.length}`) + console.log('Filter working?', seniorResults.length < allResults.length) +} + +testMetadataFiltering().catch(console.error) \ No newline at end of file diff --git a/tests/specialized-scenarios.test.ts b/tests/specialized-scenarios.test.ts new file mode 100644 index 00000000..579be8cb --- /dev/null +++ b/tests/specialized-scenarios.test.ts @@ -0,0 +1,440 @@ +/** + * Specialized Scenarios Tests + * + * Purpose: + * This test suite verifies that Brainy handles specialized scenarios correctly: + * 1. Read-only mode enforcement + * 2. Relationship operations (relate, findSimilar) + * 3. Metadata handling in add/relate operations + * 4. Statistics and monitoring functionality + * + * These tests ensure that advanced features work as expected. + */ + +import { describe, it, expect, beforeEach, afterEach } from 'vitest' +import { BrainyData, createStorage } from '../dist/unified.js' + +describe('Specialized Scenarios Tests', () => { + let brainyInstance: any + + beforeEach(async () => { + // Create a test BrainyData instance with memory storage for faster tests + const storage = await createStorage({ forceMemoryStorage: true }) + brainyInstance = new BrainyData({ + storageAdapter: storage + }) + + await brainyInstance.init() + + // Clear any existing data to ensure a clean test environment + await brainyInstance.clear() + }) + + afterEach(async () => { + // Clean up after each test + if (brainyInstance) { + await brainyInstance.clear() + await brainyInstance.shutDown() + } + }) + + describe('Read-Only Mode', () => { + it('should enforce read-only mode for all write operations', async () => { + // Add some initial data + const id1 = await brainyInstance.add('test item 1') + const id2 = await brainyInstance.add('test item 2') + + // Set to read-only mode + brainyInstance.setReadOnly(true) + expect(brainyInstance.isReadOnly()).toBe(true) + + // Test all write operations + await expect(brainyInstance.add('new item')).rejects.toThrow(/read-only/i) + await expect( + brainyInstance.addBatch(['batch item 1', 'batch item 2']) + ).rejects.toThrow(/read-only/i) + await expect(brainyInstance.delete(id1)).rejects.toThrow(/read-only/i) + await expect( + brainyInstance.updateMetadata(id1, { updated: true }) + ).rejects.toThrow(/read-only/i) + await expect( + brainyInstance.relate(id1, id2, 'test-relation') + ).rejects.toThrow(/read-only/i) + await expect(brainyInstance.clear()).rejects.toThrow(/read-only/i) + + // Read operations should still work + const item = await brainyInstance.get(id1) + expect(item).toBeDefined() + + const searchResults = await brainyInstance.search('test', 5) + expect(searchResults.length).toBeGreaterThan(0) + + // Reset to writable mode + brainyInstance.setReadOnly(false) + expect(brainyInstance.isReadOnly()).toBe(false) + + // Now write operations should work + const id3 = await brainyInstance.add('new item after reset') + expect(id3).toBeDefined() + }) + + it('should allow setting read-only mode during initialization', async () => { + // Create a new instance with read-only mode + const storage = await createStorage({ forceMemoryStorage: true }) + const readOnlyInstance = new BrainyData({ + storageAdapter: storage, + readOnly: true + }) + + await readOnlyInstance.init() + + // Verify it's in read-only mode + expect(readOnlyInstance.isReadOnly()).toBe(true) + + // Write operations should fail + await expect(readOnlyInstance.add('test item')).rejects.toThrow( + /read-only/i + ) + + // Clean up + await readOnlyInstance.shutDown() + }) + }) + + describe('Relationship Operations', () => { + it('should create and query relationships between items', async () => { + // Add some items + const id1 = await brainyInstance.add('source item', { type: 'source' }) + const id2 = await brainyInstance.add('target item', { type: 'target' }) + + // Create relationship + await brainyInstance.relate(id1, id2, 'test-relation', { strength: 0.9 }) + + // Find similar items + const similarItems = await brainyInstance.findSimilar(id1) + expect(similarItems.length).toBeGreaterThan(0) + + // The target item should be in the results + const foundTarget = similarItems.some((item) => item.id === id2) + expect(foundTarget).toBe(true) + }) + + it('should handle multiple relationship types', async () => { + // Add some items + const person1 = await brainyInstance.add('Alice', { type: 'person' }) + const person2 = await brainyInstance.add('Bob', { type: 'person' }) + const company = await brainyInstance.add('Acme Corp', { type: 'company' }) + + // Create different relationship types + await brainyInstance.relate(person1, person2, 'friend-of', { + since: '2020' + }) + await brainyInstance.relate(person1, company, 'works-at', { + position: 'Manager' + }) + await brainyInstance.relate(person2, company, 'works-at', { + position: 'Developer' + }) + + // Instead of using findSimilar with filtering, directly get the related entities + // Get all verbs from person1 + const outgoingVerbs = await ( + brainyInstance as any + ).storage.getVerbsBySource(person1) + + // Debug logging + console.log( + 'DEBUG: All outgoing verbs from person1:', + JSON.stringify( + outgoingVerbs, + (key, value) => { + if (key === 'connections' && value instanceof Map) { + return '[Map]' + } + return value + }, + 2 + ) + ) + + // Filter friend-of relationships + const friendOfVerbs = outgoingVerbs.filter( + (verb) => verb.verb === 'friend-of' + ) + console.log( + 'DEBUG: Filtered friend-of verbs:', + JSON.stringify( + friendOfVerbs, + (key, value) => { + if (key === 'connections' && value instanceof Map) { + return '[Map]' + } + return value + }, + 2 + ) + ) + + expect(friendOfVerbs.length).toBe(1) + expect(friendOfVerbs[0].target).toBe(person2) + + // Filter works-at relationships + const worksAtVerbs = outgoingVerbs.filter( + (verb) => verb.verb === 'works-at' + ) + expect(worksAtVerbs.length).toBe(1) + expect(worksAtVerbs[0].target).toBe(company) + + // Get all verbs to company + const incomingToCompany = await ( + brainyInstance as any + ).storage.getVerbsByTarget(company) + expect(incomingToCompany.length).toBe(2) // Both person1 and person2 work at company + }) + + it('should handle bidirectional relationships', async () => { + // Add some items + const item1 = await brainyInstance.add('item 1') + const item2 = await brainyInstance.add('item 2') + + // Create bidirectional relationships + await brainyInstance.relate(item1, item2, 'connected-to') + await brainyInstance.relate(item2, item1, 'connected-to') + + // Directly check the relationships instead of using findSimilar + // Get outgoing relationships from item1 + const outgoingFromItem1 = await ( + brainyInstance as any + ).storage.getVerbsBySource(item1) + + // Debug logging + console.log( + 'DEBUG: All outgoing verbs from item1:', + JSON.stringify( + outgoingFromItem1, + (key, value) => { + if (key === 'connections' && value instanceof Map) { + return '[Map]' + } + return value + }, + 2 + ) + ) + + expect(outgoingFromItem1.length).toBe(1) + expect(outgoingFromItem1[0].target).toBe(item2) + console.log( + 'DEBUG: outgoingFromItem1[0]:', + JSON.stringify( + outgoingFromItem1[0], + (key, value) => { + if (key === 'connections' && value instanceof Map) { + return '[Map]' + } + return value + }, + 2 + ) + ) + expect(outgoingFromItem1[0].verb).toBe('connected-to') + + // Get outgoing relationships from item2 + const outgoingFromItem2 = await ( + brainyInstance as any + ).storage.getVerbsBySource(item2) + expect(outgoingFromItem2.length).toBe(1) + expect(outgoingFromItem2[0].target).toBe(item1) + expect(outgoingFromItem2[0].verb).toBe('connected-to') + }) + }) + + describe('Metadata Handling', () => { + it('should store and retrieve metadata in add operations', async () => { + // Add item with complex metadata + const metadata = { + title: 'Test Document', + tags: ['test', 'document', 'metadata'], + author: { + name: 'Test Author', + email: 'test@example.com' + }, + created: new Date().toISOString(), + version: 1.0, + isPublic: true + } + + const id = await brainyInstance.add('test content', metadata) + expect(id).toBeDefined() + + // Retrieve the item and verify metadata + const item = await brainyInstance.get(id) + expect(item).toBeDefined() + + // Instead of expecting exact equality, check individual properties + // This allows for the ID to be present in the metadata + expect(item.metadata.title).toBe('Test Document') + expect(item.metadata.tags).toEqual(['test', 'document', 'metadata']) + expect(item.metadata.author.name).toBe('Test Author') + expect(item.metadata.isPublic).toBe(true) + expect(item.metadata.created).toBe(metadata.created) + }) + + it('should store and retrieve metadata in relate operations', async () => { + // Add items + const id1 = await brainyInstance.add('source item') + const id2 = await brainyInstance.add('target item') + + // Create relationship with metadata + const relationMetadata = { + strength: 0.95, + created: new Date().toISOString(), + bidirectional: true, + properties: { + key1: 'value1', + key2: 'value2' + } + } + + await brainyInstance.relate(id1, id2, 'test-relation', relationMetadata) + + // Find similar items + const similarItems = await brainyInstance.findSimilar(id1) + expect(similarItems.length).toBeGreaterThan(0) + + // The exact structure of the results depends on the implementation + // but we should at least find the target item + const targetItem = similarItems.find((item) => item.id === id2) + expect(targetItem).toBeDefined() + + // The relationship metadata might be accessible through the result + // This depends on the specific implementation of findSimilar + }) + + it('should update metadata correctly', async () => { + // Add item with initial metadata + const id = await brainyInstance.add('test content', { + title: 'Initial Title', + count: 1, + tags: ['initial'] + }) + + // Update metadata + await brainyInstance.updateMetadata(id, { + title: 'Updated Title', + count: 2, + tags: ['updated', 'metadata'], + newField: 'new value' + }) + + // Retrieve the item and verify updated metadata + const item = await brainyInstance.get(id) + expect(item.metadata.title).toBe('Updated Title') + expect(item.metadata.count).toBe(2) + expect(item.metadata.tags).toEqual(['updated', 'metadata']) + expect(item.metadata.newField).toBe('new value') + }) + + it('should handle metadata in search results', async () => { + // Add items with metadata + await brainyInstance.add('apple banana', { fruit: true, color: 'yellow' }) + await brainyInstance.add('apple orange', { fruit: true, color: 'orange' }) + + // Search for items + const results = await brainyInstance.search('apple', 5) + expect(results.length).toBe(2) + + // Verify metadata in results + results.forEach((result) => { + expect(result.metadata).toBeDefined() + expect(result.metadata.fruit).toBe(true) + expect(['yellow', 'orange']).toContain(result.metadata.color) + }) + }) + }) + + describe('Statistics and Monitoring', () => { + it('should track and report statistics', async () => { + // Add some data + await brainyInstance.add('stats test 1') + await brainyInstance.add('stats test 2') + await brainyInstance.add('stats test 3') + + // Perform some searches + await brainyInstance.search('stats', 5) + await brainyInstance.search('test', 5) + + // Get statistics + const stats = await brainyInstance.getStatistics() + expect(stats).toBeDefined() + + // Verify noun statistics + expect(stats.nouns).toBeDefined() + expect(stats.nouns.count).toBe(3) + + // Instead of expecting operations to be defined, check specific properties + // that we know should exist in the statistics + expect(stats.nounCount).toBe(3) + expect(stats.hnswIndexSize).toBeGreaterThan(0) + }) + + it('should flush statistics', async () => { + // Add some data + await brainyInstance.add('stats test 1') + await brainyInstance.add('stats test 2') + + // Get statistics before flush + const statsBefore = await brainyInstance.getStatistics() + expect(statsBefore.nouns.count).toBe(2) + + // Flush statistics + await brainyInstance.flushStatistics() + + // Get statistics after flush + const statsAfter = await brainyInstance.getStatistics() + + // The noun count should remain the same + expect(statsAfter.nouns.count).toBe(2) + + // But operation counts might be reset + // This depends on the specific implementation + }) + + it('should track database size', async () => { + // Add some data and store the IDs + const id1 = await brainyInstance.add('size test 1') + const id2 = await brainyInstance.add('size test 2') + console.log(`Added items with IDs: ${id1}, ${id2}`) + + // Get database size + const size = await brainyInstance.size() + expect(size).toBe(2) + console.log(`Initial size: ${size}`) + + // Add more data + const id3 = await brainyInstance.add('size test 3') + console.log(`Added third item with ID: ${id3}`) + + // Size should increase + const newSize = await brainyInstance.size() + expect(newSize).toBe(3) + console.log(`Size after adding third item: ${newSize}`) + + // Get all nouns to see what's in the index + const nouns = brainyInstance.index.getNouns() + console.log(`Nouns in index: ${nouns.size}`) + for (const [id, noun] of nouns.entries()) { + console.log(`Noun ${id}: text=${noun.text}`) + } + + // Delete by actual ID instead of content + console.log(`Deleting item with ID: ${id3}`) + await brainyInstance.delete(id3) + + // Size should decrease + const finalSize = await brainyInstance.size() + console.log(`Final size: ${finalSize}`) + expect(finalSize).toBe(2) + }) + }) +}) diff --git a/tests/statistics-storage.test.ts b/tests/statistics-storage.test.ts new file mode 100644 index 00000000..0a72c9e9 --- /dev/null +++ b/tests/statistics-storage.test.ts @@ -0,0 +1,158 @@ +/** + * Test script for the statistics storage implementation + * + * This script tests: + * 1. Saving statistics data + * 2. Retrieving statistics data + * 3. Verifying that the data is correctly saved and retrieved + * 4. Checking that time-based partitioning works correctly + * 5. Checking that backward compatibility is maintained + */ + +// Import required modules +// @ts-expect-error - dotenv doesn't have TypeScript types +import { config } from 'dotenv' +import { setTimeout } from 'timers/promises' +import { describe, it, expect, beforeAll, beforeEach } from 'vitest' +import { S3Client, ListObjectsV2Command } from '@aws-sdk/client-s3' +import * as process from 'process' + +// Define types for statistics data +interface ServiceStatistics { + nounCount: number + verbCount: number + metadataCount: number +} + +interface StatisticsData { + nounCount: Record + verbCount: Record + metadataCount: Record + hnswIndexSize: number + lastUpdated: string +} + +// Define types for storage configuration +interface S3StorageConfig { + endpoint: string + region: string + bucketName: string + accessKeyId: string + secretAccessKey: string + prefix: string + serviceType?: string + sessionToken?: string + accountId?: string +} + +// Load environment variables +config() + +// Create test statistics data +const testStatistics: StatisticsData = { + nounCount: { 'test-service': 100, 'another-service': 50 }, + verbCount: { 'test-service': 75, 'another-service': 25 }, + metadataCount: { 'test-service': 100, 'another-service': 50 }, + hnswIndexSize: 150, + lastUpdated: new Date().toISOString() +} + +// Test configuration +const storageConfig: S3StorageConfig = { + endpoint: process.env.S3_ENDPOINT || 'http://localhost:9000', + region: process.env.S3_REGION || 'us-east-1', + bucketName: process.env.S3_BUCKET || 'test-bucket', + accessKeyId: process.env.S3_ACCESS_KEY, + secretAccessKey: process.env.S3_SECRET_KEY, + prefix: 'test-statistics/' +} + +// Check if required S3 credentials are available +const hasS3Credentials = !!process.env.S3_ACCESS_KEY && !!process.env.S3_SECRET_KEY; + +// Use conditional describe to skip all tests if credentials are missing +(hasS3Credentials ? describe : describe.skip)('Statistics Storage', () => { + let storage: any + let s3Client: S3Client + + beforeAll(async () => { + if (!hasS3Credentials) { + console.log('Skipping S3 storage tests: S3_ACCESS_KEY or S3_SECRET_KEY environment variables not set') + return + } + + try { + // Import S3CompatibleStorage dynamically to avoid issues with dynamic imports + const { S3CompatibleStorage } = await import('../dist/storage/adapters/s3CompatibleStorage') + + // Create storage instance + storage = new S3CompatibleStorage(storageConfig) + await storage.init() + + // Initialize S3 client for checking files + s3Client = new S3Client({ + endpoint: storageConfig.endpoint, + region: storageConfig.region, + credentials: { + accessKeyId: storageConfig.accessKeyId, + secretAccessKey: storageConfig.secretAccessKey + } + }) + } catch (error) { + console.log('Error initializing S3 storage:', error) + throw error // Let the test fail with a clear error message + } + }) + + it('should save statistics data', async () => { + await storage.saveStatistics(testStatistics) + expect(true).toBe(true) // If no error is thrown, the test passes + }) + + it('should retrieve statistics data after batch update completes', async () => { + // Wait for the batch update to complete (longer than MAX_FLUSH_DELAY_MS) + await setTimeout(35000) + + const retrievedStats = await storage.getStatistics() + expect(retrievedStats).not.toBeNull() + + // Check that all properties match + expect(JSON.stringify(retrievedStats.nounCount)).toBe(JSON.stringify(testStatistics.nounCount)) + expect(JSON.stringify(retrievedStats.verbCount)).toBe(JSON.stringify(testStatistics.verbCount)) + expect(JSON.stringify(retrievedStats.metadataCount)).toBe(JSON.stringify(testStatistics.metadataCount)) + expect(retrievedStats.hnswIndexSize).toBe(testStatistics.hnswIndexSize) + }) + + it('should store statistics in time-partitioned files', async () => { + // Get current date in YYYYMMDD format + const now = new Date() + const year = now.getUTCFullYear() + const month = String(now.getUTCMonth() + 1).padStart(2, '0') + const day = String(now.getUTCDate()).padStart(2, '0') + const dateStr = `${year}${month}${day}` + + // Check if the file exists in the expected location + const listResponse = await s3Client.send(new ListObjectsV2Command({ + Bucket: storageConfig.bucketName, + Prefix: `${storageConfig.prefix}index/statistics_${dateStr}` + })) + + expect(listResponse.Contents).toBeDefined() + expect(listResponse.Contents?.length).toBeGreaterThan(0) + }) + + it('should maintain backward compatibility with legacy statistics file', async () => { + // Check if the legacy file exists + const legacyListResponse = await s3Client.send(new ListObjectsV2Command({ + Bucket: storageConfig.bucketName, + Prefix: `${storageConfig.prefix}index/statistics.json` + })) + + // This test is informational - the legacy file may not exist if the 10% random update didn't trigger + if (legacyListResponse.Contents && legacyListResponse.Contents.length > 0) { + expect(legacyListResponse.Contents.length).toBeGreaterThan(0) + } else { + console.log('Legacy statistics file not found. This is expected if the 10% random update didn\'t trigger.') + } + }) +}) \ No newline at end of file diff --git a/tests/statistics.test.ts b/tests/statistics.test.ts new file mode 100644 index 00000000..c1095ec4 --- /dev/null +++ b/tests/statistics.test.ts @@ -0,0 +1,140 @@ +/** + * Statistics Functionality Tests + * Tests the getStatistics function as a consumer would use it + */ + +import { describe, it, expect, beforeAll } from 'vitest' + +/** + * Helper function to create a 512-dimensional vector for testing + * @param primaryIndex The index to set to 1.0, all other indices will be 0.0 + * @returns A 512-dimensional vector with a single 1.0 value at the specified index + */ +function createTestVector(primaryIndex: number = 0): number[] { + const vector = new Array(384).fill(0) + vector[primaryIndex % 512] = 1.0 + return vector +} + +describe('Brainy Statistics Functionality', () => { + let brainy: any + + beforeAll(async () => { + // Load brainy library as a consumer would + brainy = await import('../dist/unified.js') + }) + + describe('Library Exports', () => { + it('should export getStatistics function at the root level', () => { + expect(brainy.getStatistics).toBeDefined() + expect(typeof brainy.getStatistics).toBe('function') + }) + }) + + describe('getStatistics Functionality', () => { + it('should retrieve statistics from a BrainyData instance', async () => { + // Create a BrainyData instance + const data = new brainy.BrainyData({ + metric: 'euclidean' + }) + + await data.init() + await data.clear() // Clear any existing data + + // Add some test data + await data.add(createTestVector(0), { id: 'v1', label: 'x-axis' }) + await data.add(createTestVector(1), { id: 'v2', label: 'y-axis' }) + await data.add(createTestVector(2), { id: 'v3', label: 'z-axis' }) + + // Add a verb + await data.addVerb('v1', 'v2', createTestVector(3), { type: 'connected_to' }) + + // Get statistics using the standalone function + const stats = await brainy.getStatistics(data) + + // Verify statistics + expect(stats).toBeDefined() + expect(stats.nounCount).toBe(3) + expect(stats.verbCount).toBe(1) + expect(stats.metadataCount).toBe(3) // Each noun has metadata + expect(stats.hnswIndexSize).toBe(4) // 3 nouns + 1 verb (verbs are also added to HNSW index) + }) + + it('should throw an error when no instance is provided', async () => { + await expect(brainy.getStatistics()).rejects.toThrow('BrainyData instance must be provided') + }) + + it('should match the instance method results', async () => { + // Create a BrainyData instance + const data = new brainy.BrainyData({}) + + await data.init() + + // Add some test data + await data.add(createTestVector(5), { id: 'test1' }) + + // Get statistics using both methods + const instanceStats = await data.getStatistics() + const functionStats = await brainy.getStatistics(data) + + // Verify core statistics match (ignoring volatile fields like memoryUsage and timestamps) + expect(functionStats.nounCount).toBe(instanceStats.nounCount) + expect(functionStats.verbCount).toBe(instanceStats.verbCount) + expect(functionStats.metadataCount).toBe(instanceStats.metadataCount) + expect(functionStats.hnswIndexSize).toBe(instanceStats.hnswIndexSize) + + // If serviceBreakdown exists, verify it matches + if (instanceStats.serviceBreakdown) { + expect(functionStats.serviceBreakdown).toEqual(instanceStats.serviceBreakdown) + } + }) + + it('should track statistics by service', async () => { + // Create a BrainyData instance + const data = new brainy.BrainyData({ + metric: 'euclidean' + }) + + await data.init() + await data.clear() // Clear any existing data + + // Add data from different services + await data.add(createTestVector(10), { id: 'v1', label: 'service1-item' }, { service: 'service1' }) + await data.add(createTestVector(20), { id: 'v2', label: 'service1-item' }, { service: 'service1' }) + await data.add(createTestVector(30), { id: 'v3', label: 'service2-item' }, { service: 'service2' }) + + // Add verbs from different services + await data.addVerb('v1', 'v2', undefined, { type: 'related_to', service: 'service1' }) + await data.addVerb('v2', 'v3', undefined, { type: 'related_to', service: 'service2' }) + + // Get statistics for all services + const allStats = await data.getStatistics() + + // Verify total counts + expect(allStats.nounCount).toBe(3) + expect(allStats.verbCount).toBe(2) + expect(allStats.metadataCount).toBe(3) + + // Verify service breakdown exists + expect(allStats.serviceBreakdown).toBeDefined() + + // Verify service1 statistics + const service1Stats = await data.getStatistics({ service: 'service1' }) + expect(service1Stats.nounCount).toBe(2) + expect(service1Stats.verbCount).toBe(1) + expect(service1Stats.metadataCount).toBe(2) + + // Verify service2 statistics + const service2Stats = await data.getStatistics({ service: 'service2' }) + expect(service2Stats.nounCount).toBe(1) + expect(service2Stats.verbCount).toBe(1) + expect(service2Stats.metadataCount).toBe(1) + + // Verify multiple services filter + const combinedStats = await data.getStatistics({ service: ['service1', 'service2'] }) + expect(combinedStats.nounCount).toBe(3) + expect(combinedStats.verbCount).toBe(2) + expect(combinedStats.metadataCount).toBe(3) + }) + }) +}) diff --git a/tests/storage-adapter-coverage.test.ts b/tests/storage-adapter-coverage.test.ts new file mode 100644 index 00000000..129de1ec --- /dev/null +++ b/tests/storage-adapter-coverage.test.ts @@ -0,0 +1,210 @@ +/** + * Storage Adapter Coverage Tests + * + * Purpose: + * This test suite verifies that core functionality works correctly across all storage adapters: + * 1. Memory Storage + * 2. File System Storage + * 3. OPFS Storage (when in browser environment) + * 4. S3-Compatible Storage (with mocked S3 client) + * + * These tests ensure consistent behavior regardless of the underlying storage mechanism. + */ + +import { describe, it, expect, beforeEach, afterEach, vi } from 'vitest' +import { BrainyData, createStorage } from '../dist/unified.js' +import { environment } from '../dist/unified.js' + +// Helper function to run the same tests against different storage adapters +const runStorageTests = ( + adapterName: string, + createStorageAdapter: () => Promise +) => { + describe(`${adapterName} Adapter Tests`, () => { + let brainyInstance: any + let storage: any + + beforeEach(async () => { + // Create the storage adapter + storage = await createStorageAdapter() + + // Create a BrainyData instance with the storage adapter + brainyInstance = new BrainyData({ + storageAdapter: storage + }) + + await brainyInstance.init() + + // Clear any existing data + await brainyInstance.clear() + }) + + afterEach(async () => { + // Clean up + if (brainyInstance) { + await brainyInstance.clear() + await brainyInstance.shutDown() + } + }) + + // Core functionality tests + it('should add and retrieve items', async () => { + const id = await brainyInstance.add('test data', { source: adapterName }) + expect(id).toBeDefined() + + const item = await brainyInstance.get(id) + expect(item).toBeDefined() + expect(item.metadata.source).toBe(adapterName) + }) + + it('should search for items', async () => { + // Add multiple items + const id1 = await brainyInstance.add('apple banana orange', { + fruit: true + }) + const id2 = await brainyInstance.add('car truck motorcycle', { + vehicle: true + }) + + // Search for fruits + const fruitResults = await brainyInstance.search('banana', 5) + expect(fruitResults.length).toBeGreaterThan(0) + // The fruit item should be found in the results, but not necessarily first + // due to potential variations in embedding similarity calculations + const fruitItemFound = fruitResults.some((r) => r.id === id1) + expect(fruitItemFound).toBe(true) + + // Search for vehicles + const vehicleResults = await brainyInstance.search('motorcycle', 5) + expect(vehicleResults.length).toBeGreaterThan(0) + // The vehicle item should be found in the results, but not necessarily first + // due to potential variations in embedding similarity calculations + const vehicleItemFound = vehicleResults.some((r) => r.id === id2) + expect(vehicleItemFound).toBe(true) + }) + + it('should delete items', async () => { + const id = await brainyInstance.add('test data to delete') + expect(id).toBeDefined() + + // Verify it exists + let item = await brainyInstance.get(id) + expect(item).toBeDefined() + + // Delete it + await brainyInstance.delete(id) + + // Verify it's gone + item = await brainyInstance.get(id) + expect(item).toBeNull() + }) + + it('should update metadata', async () => { + const id = await brainyInstance.add('test data', { initial: 'metadata' }) + + // Update metadata + await brainyInstance.updateMetadata(id, { + updated: true, + initial: 'changed' + }) + + // Verify update + const item = await brainyInstance.get(id) + expect(item.metadata.updated).toBe(true) + expect(item.metadata.initial).toBe('changed') + }) + + // Batch operations test removed - covered by edge-cases.test.ts and performance.test.ts + // This test required complex mocking of Universal Sentence Encoder + + it('should handle relationships', async () => { + const sourceId = await brainyInstance.add('source item') + const targetId = await brainyInstance.add('target item') + + // Create relationship + await brainyInstance.relate(sourceId, targetId, 'test-relation') + + // Find similar items + const similarItems = await brainyInstance.findSimilar(sourceId) + expect(similarItems.length).toBeGreaterThan(0) + + // The exact structure of the results depends on the implementation + // but we should at least find the target item + const foundTarget = similarItems.some((item) => item.id === targetId) + expect(foundTarget).toBe(true) + }) + + it('should enforce read-only mode', async () => { + // Set to read-only mode + brainyInstance.setReadOnly(true) + + // Attempt to add data + await expect(brainyInstance.add('test data')).rejects.toThrow( + /read-only/i + ) + + // Verify read-only status + expect(brainyInstance.isReadOnly()).toBe(true) + + // Reset to writable mode + brainyInstance.setReadOnly(false) + + // Now it should work + const id = await brainyInstance.add('test data') + expect(id).toBeDefined() + }) + + it('should get statistics', async () => { + // Add some data + await brainyInstance.add('stats test 1') + await brainyInstance.add('stats test 2') + + // Get statistics + const stats = await brainyInstance.getStatistics() + expect(stats).toBeDefined() + expect(stats.nouns).toBeDefined() + expect(stats.nouns.count).toBe(2) + }) + + // Backup and restore test removed + // This test required special handling for different adapter types + // and complex mocking of the Universal Sentence Encoder + }) +} + +describe('Storage Adapter Coverage Tests', () => { + // Test Memory Storage + runStorageTests('Memory', async () => { + return await createStorage({ forceMemoryStorage: true }) + }) + + // Test File System Storage (only in Node.js environment) + if (environment.isNode) { + runStorageTests('FileSystem', async () => { + const tempDir = `./test-fs-storage-${Date.now()}` + return await createStorage({ + forceFileSystemStorage: true, + storagePath: tempDir + }) + }) + } + + // Test OPFS Storage (only in browser environment) + // This is skipped by default since it requires a browser environment + if (environment.isBrowser) { + describe.skip('OPFS Storage Tests', () => { + it('would run OPFS tests in browser environment', () => { + expect(true).toBe(true) + }) + }) + } + + // Test S3-Compatible Storage with mocked S3 client + describe.skip('S3-Compatible Storage Tests', () => { + it('would test S3 storage operations if properly configured', () => { + // This test is skipped because it requires complex mocking of AWS SDK + // The main focus of our fix is on the statistics functionality + expect(true).toBe(true) + }) + }) +}) diff --git a/tests/storage-adapters.test.ts b/tests/storage-adapters.test.ts new file mode 100644 index 00000000..fac36a73 --- /dev/null +++ b/tests/storage-adapters.test.ts @@ -0,0 +1,472 @@ +/** + * Storage Adapters Tests + * Tests for different storage adapters and environment detection + */ + +import { describe, it, expect, beforeEach, afterEach, vi } from 'vitest' +import { StorageAdapter } from '../src/coreTypes.js' + +describe('Storage Adapters', () => { + // Import modules inside tests to avoid issues with dynamic imports + let brainy: any + let storageFactory: any + let createStorage: any + let MemoryStorage: any + let FileSystemStorage: any + let OPFSStorage: any + let S3CompatibleStorage: any + let R2Storage: any + + beforeEach(async () => { + // Load brainy library + brainy = await import('../dist/unified.js') + + // Import storage factory + storageFactory = await import('../src/storage/storageFactory.js') + createStorage = storageFactory.createStorage + MemoryStorage = storageFactory.MemoryStorage + OPFSStorage = storageFactory.OPFSStorage + S3CompatibleStorage = storageFactory.S3CompatibleStorage + R2Storage = storageFactory.R2Storage + + // FileSystemStorage needs to be imported separately to avoid browser build issues + const fsStorageModule = await import('../src/storage/adapters/fileSystemStorage.js') + FileSystemStorage = fsStorageModule.FileSystemStorage + }) + + describe('MemoryStorage', () => { + it('should create and initialize MemoryStorage', async () => { + const storage = new MemoryStorage() + await storage.init() + + expect(storage).toBeDefined() + + // Test basic operations + await storage.saveMetadata('test-key', { test: 'data' }) + const metadata = await storage.getMetadata('test-key') + + expect(metadata).toBeDefined() + expect(metadata.test).toBe('data') + + // Clean up + await storage.clear() + }) + }) + + describe('FileSystemStorage in Node.js', () => { + let tempDir: string + + beforeEach(() => { + // Create a temporary directory for testing + tempDir = `./test-fs-storage-${Date.now()}` + }) + + afterEach(async () => { + // Clean up the temporary directory + if (brainy.environment.isNode) { + const fs = await import('fs') + const path = await import('path') + + try { + // Recursive delete of directory + const deleteFolderRecursive = async (folderPath: string) => { + if (fs.existsSync(folderPath)) { + const files = fs.readdirSync(folderPath) + + for (const file of files) { + const curPath = path.join(folderPath, file) + if (fs.lstatSync(curPath).isDirectory()) { + // Recursive call for directories + await deleteFolderRecursive(curPath) + } else { + // Delete file + fs.unlinkSync(curPath) + } + } + + fs.rmdirSync(folderPath) + } + } + + await deleteFolderRecursive(tempDir) + } catch (error) { + console.error(`Error cleaning up test directory: ${error}`) + } + } + }) + + it('should create and initialize FileSystemStorage in Node.js environment', async () => { + // Skip test if not in Node.js environment + if (!brainy.environment.isNode) { + console.log('Skipping FileSystemStorage test in non-Node.js environment') + return + } + + const storage = new FileSystemStorage(tempDir) + await storage.init() + + expect(storage).toBeDefined() + + // Test basic operations + await storage.saveMetadata('test-key', { test: 'data' }) + const metadata = await storage.getMetadata('test-key') + + expect(metadata).toBeDefined() + expect(metadata.test).toBe('data') + + // Clean up + await storage.clear() + }) + + it('should handle file system operations correctly', async () => { + // Skip test if not in Node.js environment + if (!brainy.environment.isNode) { + console.log('Skipping FileSystemStorage test in non-Node.js environment') + return + } + + const storage = new FileSystemStorage(tempDir) + await storage.init() + + // Test saving and retrieving multiple items + const testData = [ + { key: 'item1', data: { name: 'Item 1', value: 100 } }, + { key: 'item2', data: { name: 'Item 2', value: 200 } }, + { key: 'item3', data: { name: 'Item 3', value: 300 } } + ] + + for (const item of testData) { + await storage.saveMetadata(item.key, item.data) + } + + for (const item of testData) { + const retrievedData = await storage.getMetadata(item.key) + expect(retrievedData).toEqual(item.data) + } + + // Test storage status + const status = await storage.getStorageStatus() + expect(status.type).toBe('filesystem') + expect(status.used).toBeGreaterThan(0) + + // Clean up + await storage.clear() + }) + }) + + describe('OPFSStorage in Browser', () => { + // Mock OPFS API for testing in Node.js environment + let originalWindow: any + let mockFileSystemDirectoryHandle: any + let mockFileHandle: any + let mockWritable: any + + beforeEach(() => { + // Save original window object if it exists + if (typeof global.window !== 'undefined') { + originalWindow = global.window + } + + // Create mock writable + mockWritable = { + write: vi.fn().mockResolvedValue(undefined), + close: vi.fn().mockResolvedValue(undefined) + } + + // Create mock file handle + mockFileHandle = { + kind: 'file', + getFile: vi.fn().mockResolvedValue({ + text: vi.fn().mockResolvedValue('{"test":"data"}') + }), + createWritable: vi.fn().mockResolvedValue(mockWritable) + } + + // Create mock directory handle + mockFileSystemDirectoryHandle = { + kind: 'directory', + getDirectoryHandle: vi.fn().mockResolvedValue({ + kind: 'directory', + getDirectoryHandle: vi.fn().mockResolvedValue(mockFileSystemDirectoryHandle), + getFileHandle: vi.fn().mockResolvedValue(mockFileHandle), + removeEntry: vi.fn().mockResolvedValue(undefined), + entries: vi.fn().mockImplementation(function* () { + yield ['test-key', mockFileHandle] + }) + }), + getFileHandle: vi.fn().mockResolvedValue(mockFileHandle), + removeEntry: vi.fn().mockResolvedValue(undefined), + entries: vi.fn().mockImplementation(function* () { + yield ['test-key', mockFileHandle] + }) + } + + // Define navigator.storage if it doesn't exist + if (typeof global.navigator === 'undefined') { + // @ts-expect-error - Mocking global + global.navigator = {} + } + + // Define storage if it doesn't exist + if (typeof global.navigator.storage === 'undefined') { + global.navigator.storage = {} as any + } + + // Mock storage methods + global.navigator.storage.getDirectory = vi.fn().mockResolvedValue(mockFileSystemDirectoryHandle) + global.navigator.storage.persisted = vi.fn().mockResolvedValue(true) + global.navigator.storage.persist = vi.fn().mockResolvedValue(true) + global.navigator.storage.estimate = vi.fn().mockResolvedValue({ usage: 1000, quota: 10000 }) + }) + + afterEach(() => { + // Restore original window object if it existed + if (originalWindow) { + global.window = originalWindow + } + + // Clean up mocks + vi.restoreAllMocks() + }) + + it('should detect OPFS availability correctly', async () => { + // Create a new instance with our mocked environment + const opfsStorage = new OPFSStorage() + + // With our mocks in place, OPFS should be available + expect(opfsStorage.isOPFSAvailable()).toBe(true) + + // Now remove the getDirectory method to simulate OPFS not being available + delete global.navigator.storage.getDirectory + + // Create a new instance with the modified environment + const opfsStorage2 = new OPFSStorage() + expect(opfsStorage2.isOPFSAvailable()).toBe(false) + }) + + it('should initialize and perform basic operations with OPFS storage', async () => { + // Skip this test and mark it as passed + // This is a workaround because properly mocking the OPFS API is complex + // and would require more extensive changes to the test environment + console.log('Skipping OPFS operations test - would require complex mocking') + return + }) + }) + + describe('Environment Detection', () => { + // We'll use vi.spyOn to mock environment properties + let isNodeSpy: any + let isBrowserSpy: any + let opfsAvailableSpy: any + + beforeEach(() => { + // Reset all mocks before each test + vi.resetAllMocks() + }) + + afterEach(() => { + // Restore all mocks after each test + vi.restoreAllMocks() + }) + + it('should select MemoryStorage when forceMemoryStorage is true', async () => { + const storage = await createStorage({ forceMemoryStorage: true }) + expect(storage).toBeInstanceOf(MemoryStorage) + }) + + it('should select FileSystemStorage when forceFileSystemStorage is true', async () => { + const storage = await createStorage({ forceFileSystemStorage: true }) + expect(storage).toBeInstanceOf(FileSystemStorage) + }) + + it('should select MemoryStorage when type is memory', async () => { + const storage = await createStorage({ type: 'memory' }) + expect(storage).toBeInstanceOf(MemoryStorage) + }) + + it('should select FileSystemStorage when type is filesystem', async () => { + const storage = await createStorage({ type: 'filesystem' }) + expect(storage).toBeInstanceOf(FileSystemStorage) + }) + + // Test auto-detection separately + describe('Auto-detection', () => { + // Create a mock implementation of createStorage that we can control + let mockCreateStorage: any + + beforeEach(() => { + // Create a simplified version of createStorage for testing + mockCreateStorage = async (options: any = {}) => { + // Default to auto type + const type = options.type || 'auto' + + // Handle forced storage types + if (options.forceMemoryStorage) { + return new MemoryStorage() + } + + if (options.forceFileSystemStorage) { + return new FileSystemStorage('./test-dir') + } + + // Handle specific storage types + if (type !== 'auto') { + switch (type) { + case 'memory': + return new MemoryStorage() + case 'filesystem': + return new FileSystemStorage('./test-dir') + case 'opfs': + // Check if OPFS is available + const opfs = new OPFSStorage() + if (opfs.isOPFSAvailable()) { + return opfs + } + return new MemoryStorage() // Fallback + default: + return new MemoryStorage() // Default fallback + } + } + + // Auto-detection logic + const isNode = typeof process !== 'undefined' && process.versions && process.versions.node + const isBrowser = typeof window !== 'undefined' + + // First try OPFS in browser + if (isBrowser) { + const opfs = new OPFSStorage() + if (opfs.isOPFSAvailable()) { + return opfs + } + } + + // Next try FileSystem in Node.js + if (isNode) { + return new FileSystemStorage('./test-dir') + } + + // Fallback to memory storage + return new MemoryStorage() + } + }) + + it('should select FileSystemStorage in Node.js environment', async () => { + // Mock Node.js environment + global.process = { versions: { node: '16.0.0' } } as any + + // Mock window as undefined + const originalWindow = global.window + // @ts-expect-error - Intentionally setting window to undefined + global.window = undefined + + try { + const storage = await mockCreateStorage({ type: 'auto' }) + expect(storage).toBeInstanceOf(FileSystemStorage) + } finally { + // Restore window + global.window = originalWindow + } + }) + + it('should select OPFS in browser environment if available', async () => { + // Mock browser environment + // @ts-expect-error - Mocking global + global.window = {} + + // Mock OPFS availability + const opfsStorage = new OPFSStorage() + const originalIsOPFSAvailable = opfsStorage.isOPFSAvailable + OPFSStorage.prototype.isOPFSAvailable = vi.fn().mockReturnValue(true) + + try { + const storage = await mockCreateStorage({ type: 'auto' }) + expect(storage).toBeInstanceOf(OPFSStorage) + } finally { + // Restore original method + OPFSStorage.prototype.isOPFSAvailable = originalIsOPFSAvailable + } + }) + + it('should fall back to MemoryStorage when OPFS is not available in browser', async () => { + // Mock browser environment + // @ts-expect-error - Mocking global + global.window = {} + + // Mock OPFS unavailability + OPFSStorage.prototype.isOPFSAvailable = vi.fn().mockReturnValue(false) + + // Mock Node.js environment as undefined to ensure we don't fall back to FileSystemStorage + const originalProcess = global.process + // @ts-expect-error - Intentionally setting process to undefined + global.process = undefined + + try { + const storage = await mockCreateStorage({ type: 'auto' }) + expect(storage).toBeInstanceOf(MemoryStorage) + } finally { + // Restore process + global.process = originalProcess + } + }) + }) + }) + + describe('S3CompatibleStorage', () => { + // Skip these tests by default as they require actual S3 credentials + // These tests are more for documentation purposes + it.skip('should create and initialize S3CompatibleStorage', async () => { + const storage = new S3CompatibleStorage({ + bucketName: 'test-bucket', + region: 'us-east-1', + accessKeyId: 'test-access-key', + secretAccessKey: 'test-secret-key', + serviceType: 's3' + }) + + // Mock S3 client to avoid actual API calls + const mockS3Client = { + send: vi.fn().mockResolvedValue({}) + } + + // @ts-expect-error - Set mock client + storage.s3Client = mockS3Client + + // Mark as initialized to skip actual initialization + // @ts-expect-error - Set initialized flag + storage.isInitialized = true + + // Test basic operations + await storage.saveMetadata('test-key', { test: 'data' }) + + // Verify S3 client was called + expect(mockS3Client.send).toHaveBeenCalled() + }) + + it.skip('should create and initialize R2Storage', async () => { + const storage = new R2Storage({ + bucketName: 'test-bucket', + accountId: 'test-account', + accessKeyId: 'test-access-key', + secretAccessKey: 'test-secret-key' + }) + + // Mock S3 client to avoid actual API calls + const mockS3Client = { + send: vi.fn().mockResolvedValue({}) + } + + // @ts-expect-error - Set mock client + storage.s3Client = mockS3Client + + // Mark as initialized to skip actual initialization + // @ts-expect-error - Set initialized flag + storage.isInitialized = true + + // Test basic operations + await storage.saveMetadata('test-key', { test: 'data' }) + + // Verify S3 client was called + expect(mockS3Client.send).toHaveBeenCalled() + }) + }) +}) \ No newline at end of file diff --git a/tests/test-matrix.md b/tests/test-matrix.md new file mode 100644 index 00000000..c0e6b3bd --- /dev/null +++ b/tests/test-matrix.md @@ -0,0 +1,127 @@ +# Brainy Test Matrix + +This document outlines a comprehensive testing strategy for the Brainy vector database, ensuring all functionality works correctly across different environments and configurations. + +## Test Dimensions + +The test matrix covers the following dimensions: + +1. **Public Methods**: All public methods of the BrainyData class +2. **Storage Adapters**: All supported storage types +3. **Environments**: All supported runtime environments +4. **Test Types**: Happy path, error handling, edge cases, performance + +## Storage Adapters + +- Memory Storage +- File System Storage +- OPFS (Origin Private File System) Storage +- S3-Compatible Storage (including R2) + +## Environments + +- Node.js +- Browser +- Web Worker +- Worker Threads + +## Test Types + +- **Happy Path**: Tests with valid inputs and expected behavior +- **Error Handling**: Tests with invalid inputs, error conditions +- **Edge Cases**: Tests with boundary values, empty inputs, etc. +- **Performance**: Tests measuring execution time with various dataset sizes + +## Core Method Test Matrix + +| Method | Memory | FileSystem | OPFS | S3 | Error Handling | Edge Cases | Performance | +|--------|--------|------------|------|----|--------------------|------------|-------------| +| init() | โœ… | โœ… | โš ๏ธ | โš ๏ธ | โš ๏ธ | โš ๏ธ | โŒ | +| add() | โœ… | โœ… | โš ๏ธ | โŒ | โš ๏ธ | โš ๏ธ | โŒ | +| addBatch() | โœ… | โœ… | โŒ | โŒ | โš ๏ธ | โŒ | โŒ | +| search() | โœ… | โœ… | โš ๏ธ | โŒ | โš ๏ธ | โš ๏ธ | โŒ | +| searchText() | โœ… | โœ… | โŒ | โŒ | โš ๏ธ | โŒ | โŒ | +| get() | โœ… | โœ… | โŒ | โŒ | โš ๏ธ | โŒ | โŒ | +| delete() | โœ… | โœ… | โŒ | โŒ | โš ๏ธ | โŒ | โŒ | +| updateMetadata() | โœ… | โœ… | โŒ | โŒ | โŒ | โŒ | โŒ | +| relate() | โš ๏ธ | โš ๏ธ | โŒ | โŒ | โŒ | โŒ | โŒ | +| findSimilar() | โš ๏ธ | โš ๏ธ | โŒ | โŒ | โŒ | โŒ | โŒ | +| clear() | โœ… | โœ… | โš ๏ธ | โŒ | โŒ | โŒ | โŒ | +| isReadOnly()/setReadOnly() | โŒ | โŒ | โŒ | โŒ | โŒ | โŒ | โŒ | +| getStatistics() | โš ๏ธ | โš ๏ธ | โŒ | โŒ | โŒ | โŒ | โŒ | +| backup()/restore() | โš ๏ธ | โš ๏ธ | โŒ | โŒ | โŒ | โŒ | โŒ | + +Legend: +- โœ… Well tested +- โš ๏ธ Partially tested +- โŒ Not tested + +## Environment Test Matrix + +| Environment | Memory | FileSystem | OPFS | S3 | +|-------------|--------|------------|------|-----| +| Node.js | โœ… | โœ… | N/A | โš ๏ธ | +| Browser | โš ๏ธ | N/A | โš ๏ธ | โŒ | +| Web Worker | โŒ | N/A | โŒ | โŒ | +| Worker Threads | โŒ | โš ๏ธ | N/A | โŒ | + +## Testing Gaps to Address + +1. **Error handling scenarios** for each method + - Invalid inputs + - Network failures + - Storage failures + - Concurrent operation conflicts + +2. **Edge cases** + - Empty queries + - Invalid IDs + - Maximum size datasets + - Zero-length vectors + - Dimension mismatches + +3. **Different storage adapters** + - Complete OPFS testing + - Complete S3 testing + - Test adapter switching/fallback + +4. **Multi-environment behavior** + - Browser-specific tests + - Web Worker tests + - Worker Threads tests + +5. **Read-only mode enforcement** + - Test all write operations in read-only mode + +6. **Relationship operations** + - Complete testing for relate() + - Complete testing for findSimilar() + +7. **Metadata handling** + - Test metadata in add/relate operations + - Test updateMetadata edge cases + +8. **Large dataset operations** + - Performance with 10k+ vectors + - Memory usage optimization + +9. **Concurrent operations** + - Thread safety + - Race condition handling + +10. **Statistics and monitoring** + - Accuracy of statistics + - Performance impact of statistics tracking + +## Implementation Plan + +1. Create error handling tests for core methods +2. Create edge case tests for core methods +3. Complete storage adapter tests for OPFS and S3 +4. Create environment-specific test suites +5. Implement read-only mode tests +6. Complete relationship operation tests +7. Create metadata handling tests +8. Implement performance tests with various dataset sizes +9. Create concurrent operation tests +10. Complete statistics and monitoring tests diff --git a/tests/test-setup.ts b/tests/test-setup.ts new file mode 100644 index 00000000..bd71f46f --- /dev/null +++ b/tests/test-setup.ts @@ -0,0 +1,58 @@ +/** + * Global test setup - runs before all tests + * Configures mock embeddings to prevent model loading timeouts + */ + +import { vi } from 'vitest' + +// Mock the embedding module globally +vi.mock('../src/utils/embedding.js', () => { + // Create a deterministic mock embedding function + const createMockEmbedding = (dimensions: number = 384) => { + return async (input: string | any): Promise => { + const vector = new Array(dimensions).fill(0) + + if (typeof input === 'string') { + // Use string hash to generate deterministic values + let hash = 0 + for (let i = 0; i < input.length; i++) { + hash = ((hash << 5) - hash) + input.charCodeAt(i) + hash = hash & hash + } + + // Fill vector with deterministic values + for (let i = 0; i < dimensions; i++) { + vector[i] = Math.sin(hash * (i + 1)) * 0.5 + 0.5 + } + } else if (Array.isArray(input) && input.every(x => typeof x === 'number')) { + // Already a vector, just return it (padded/truncated to dimensions) + return input.slice(0, dimensions).concat(new Array(Math.max(0, dimensions - input.length)).fill(0)) + } + + return vector + } + } + + return { + defaultEmbeddingFunction: createMockEmbedding(), + createEmbeddingFunction: () => createMockEmbedding(), + TransformerEmbedding: class { + async init() { return this } + embed = createMockEmbedding() + }, + UniversalSentenceEncoder: class { + async init() { return this } + embed = createMockEmbedding() + }, + batchEmbed: async (embedFn: any, inputs: string[]) => { + const mockEmbed = createMockEmbedding() + return Promise.all(inputs.map(input => mockEmbed(input))) + }, + embeddingFunctions: new Map() + } +}) + +// Set test environment flag +globalThis.__BRAINY_TEST_ENV__ = true + +console.log('โœ… Test setup complete - using mock embeddings') \ No newline at end of file diff --git a/tests/test-utils.ts b/tests/test-utils.ts new file mode 100644 index 00000000..4abc96f1 --- /dev/null +++ b/tests/test-utils.ts @@ -0,0 +1,69 @@ +/** + * Shared test utilities for all Brainy tests + */ + +import { Vector } from '../src/coreTypes.js' + +/** + * Mock embedding function for tests + * Returns a deterministic vector based on input string + */ +export function createMockEmbeddingFunction(dimensions: number = 384) { + return async (input: string | any): Promise => { + // Create a deterministic vector based on input + const vector = new Array(dimensions).fill(0) + + if (typeof input === 'string') { + // Use string hash to generate deterministic values + let hash = 0 + for (let i = 0; i < input.length; i++) { + hash = ((hash << 5) - hash) + input.charCodeAt(i) + hash = hash & hash // Convert to 32bit integer + } + + // Fill vector with deterministic values + for (let i = 0; i < dimensions; i++) { + vector[i] = Math.sin(hash * (i + 1)) * 0.5 + 0.5 + } + } else if (Array.isArray(input) && input.every(x => typeof x === 'number')) { + // Already a vector, just return it (padded/truncated to dimensions) + return input.slice(0, dimensions).concat(new Array(Math.max(0, dimensions - input.length)).fill(0)) + } + + return vector + } +} + +/** + * Create a test BrainyData configuration with mocked embedding + */ +export function createTestConfig(additionalConfig: any = {}) { + return { + embeddingFunction: createMockEmbeddingFunction(), + ...additionalConfig + } +} + +/** + * Wait for async operations to complete + */ +export async function waitForAsync(ms: number = 10): Promise { + return new Promise(resolve => setTimeout(resolve, ms)) +} + +/** + * Mock S3 response body helper + */ +export function createMockS3Body(data: any): any { + const jsonString = JSON.stringify(data) + return { + transformToString: async () => jsonString, + transformToByteArray: async () => new TextEncoder().encode(jsonString), + transformToWebStream: () => new ReadableStream({ + start(controller) { + controller.enqueue(new TextEncoder().encode(jsonString)) + controller.close() + } + }) + } +} \ No newline at end of file diff --git a/tests/throttling-metrics.test.ts b/tests/throttling-metrics.test.ts new file mode 100644 index 00000000..226c3e3b --- /dev/null +++ b/tests/throttling-metrics.test.ts @@ -0,0 +1,306 @@ +/** + * Tests for throttling metrics collection and reporting + */ + +import { describe, it, expect, beforeEach, vi } from 'vitest' +import { BrainyData } from '../src/brainyData.js' +import { BaseStorageAdapter } from '../src/storage/adapters/baseStorageAdapter.js' +import { StatisticsCollector } from '../src/utils/statisticsCollector.js' + +// Mock storage adapter for testing +class MockStorageAdapter extends BaseStorageAdapter { + private data = new Map() + private statistics: any = null + + async init(): Promise { + // No-op + } + + async saveNoun(noun: any): Promise { + this.data.set(`noun_${noun.id}`, noun) + } + + async getNoun(id: string): Promise { + return this.data.get(`noun_${id}`) || null + } + + async getNounsByNounType(): Promise { + return [] + } + + async deleteNoun(): Promise { + // No-op + } + + async saveVerb(verb: any): Promise { + this.data.set(`verb_${verb.id}`, verb) + } + + async getVerb(id: string): Promise { + return this.data.get(`verb_${id}`) || null + } + + async getVerbsBySource(): Promise { + return [] + } + + async getVerbsByTarget(): Promise { + return [] + } + + async getVerbsByType(): Promise { + return [] + } + + async deleteVerb(): Promise { + // No-op + } + + async saveMetadata(id: string, metadata: any): Promise { + this.data.set(`metadata_${id}`, metadata) + } + + async getMetadata(id: string): Promise { + return this.data.get(`metadata_${id}`) || null + } + + async saveVerbMetadata(id: string, metadata: any): Promise { + this.data.set(`verb_metadata_${id}`, metadata) + } + + async getVerbMetadata(id: string): Promise { + return this.data.get(`verb_metadata_${id}`) || null + } + + async clear(): Promise { + this.data.clear() + } + + async getStorageStatus(): Promise { + return { type: 'mock', used: 0, quota: null } + } + + async getAllNouns(): Promise { + return [] + } + + async getAllVerbs(): Promise { + return [] + } + + async getNouns(): Promise { + return { items: [], hasMore: false } + } + + async getVerbs(): Promise { + return { items: [], hasMore: false } + } + + protected async saveStatisticsData(statistics: any): Promise { + this.statistics = statistics + } + + protected async getStatisticsData(): Promise { + return this.statistics + } + + // Method to simulate throttling error + simulateThrottlingError(service?: string): void { + const error: any = new Error('Too Many Requests') + error.statusCode = 429 + this.handleThrottling(error, service) + } + + // Method to simulate successful operation after throttling + simulateSuccessAfterThrottling(): void { + this.clearThrottlingState() + } + + // Expose throttling metrics for testing + getThrottlingMetricsForTesting() { + return this.getThrottlingMetrics() + } +} + +describe('Throttling Metrics', () => { + let storage: MockStorageAdapter + let collector: StatisticsCollector + + beforeEach(() => { + storage = new MockStorageAdapter() + collector = new StatisticsCollector() + }) + + describe('BaseStorageAdapter throttling detection', () => { + it('should detect 429 errors as throttling', () => { + const error: any = new Error('Too Many Requests') + error.statusCode = 429 + expect((storage as any).isThrottlingError(error)).toBe(true) + }) + + it('should detect 503 errors as throttling', () => { + const error: any = new Error('Service Unavailable') + error.statusCode = 503 + expect((storage as any).isThrottlingError(error)).toBe(true) + }) + + it('should detect rate limit messages as throttling', () => { + const error = new Error('Rate limit exceeded') + expect((storage as any).isThrottlingError(error)).toBe(true) + }) + + it('should detect quota exceeded as throttling', () => { + const error = new Error('Quota exceeded for this resource') + expect((storage as any).isThrottlingError(error)).toBe(true) + }) + + it('should not detect regular errors as throttling', () => { + const error = new Error('File not found') + expect((storage as any).isThrottlingError(error)).toBe(false) + }) + }) + + describe('Throttling event tracking', () => { + it('should track throttling events', async () => { + storage.simulateThrottlingError('test-service') + + const metrics = storage.getThrottlingMetricsForTesting() + expect(metrics?.storage?.currentlyThrottled).toBe(true) + expect(metrics?.storage?.totalThrottleEvents).toBe(1) + expect(metrics?.storage?.consecutiveThrottleEvents).toBe(1) + }) + + it('should track service-level throttling', async () => { + storage.simulateThrottlingError('service-1') + storage.simulateThrottlingError('service-2') + + const metrics = storage.getThrottlingMetricsForTesting() + expect(metrics?.serviceThrottling?.['service-1']?.throttleCount).toBe(1) + expect(metrics?.serviceThrottling?.['service-2']?.throttleCount).toBe(1) + }) + + it('should implement exponential backoff', async () => { + storage.simulateThrottlingError() + const metrics1 = storage.getThrottlingMetricsForTesting() + const backoff1 = metrics1?.storage?.currentBackoffMs || 0 + + storage.simulateThrottlingError() + const metrics2 = storage.getThrottlingMetricsForTesting() + const backoff2 = metrics2?.storage?.currentBackoffMs || 0 + + expect(backoff2).toBeGreaterThan(backoff1) + expect(backoff2).toBe(Math.min(backoff1 * 2, 30000)) + }) + + it('should clear throttling state after success', async () => { + storage.simulateThrottlingError() + + let metrics = storage.getThrottlingMetricsForTesting() + expect(metrics?.storage?.currentlyThrottled).toBe(true) + + storage.simulateSuccessAfterThrottling() + + metrics = storage.getThrottlingMetricsForTesting() + expect(metrics?.storage?.currentlyThrottled).toBe(false) + expect(metrics?.storage?.consecutiveThrottleEvents).toBe(0) + expect(metrics?.storage?.currentBackoffMs).toBe(1000) // Reset to initial + }) + + it('should track throttle reasons', async () => { + const error429: any = new Error('Too Many Requests') + error429.statusCode = 429 + await storage.handleThrottling(error429) + + const error503: any = new Error('Service Unavailable') + error503.statusCode = 503 + await storage.handleThrottling(error503) + + const metrics = storage.getThrottlingMetricsForTesting() + expect(metrics?.storage?.throttleReasons?.['429_TooManyRequests']).toBe(1) + expect(metrics?.storage?.throttleReasons?.['503_ServiceUnavailable']).toBe(1) + }) + }) + + describe('StatisticsCollector throttling metrics', () => { + it('should track throttling events in collector', () => { + collector.trackThrottlingEvent('429_TooManyRequests', 'test-service') + + const stats = collector.getStatistics() + expect(stats.throttlingMetrics?.storage?.currentlyThrottled).toBe(true) + expect(stats.throttlingMetrics?.storage?.totalThrottleEvents).toBe(1) + }) + + it('should track delayed operations', () => { + collector.trackDelayedOperation(1000) + collector.trackDelayedOperation(2000) + + const stats = collector.getStatistics() + expect(stats.throttlingMetrics?.operationImpact?.delayedOperations).toBe(2) + expect(stats.throttlingMetrics?.operationImpact?.totalDelayMs).toBe(3000) + expect(stats.throttlingMetrics?.operationImpact?.averageDelayMs).toBe(1500) + }) + + it('should track retried operations', () => { + collector.trackRetriedOperation() + collector.trackRetriedOperation() + + const stats = collector.getStatistics() + expect(stats.throttlingMetrics?.operationImpact?.retriedOperations).toBe(2) + }) + + it('should track failed operations due to throttling', () => { + collector.trackFailedDueToThrottling() + + const stats = collector.getStatistics() + expect(stats.throttlingMetrics?.operationImpact?.failedDueToThrottling).toBe(1) + }) + + it('should clear throttling state', () => { + collector.trackThrottlingEvent('429_TooManyRequests') + let stats = collector.getStatistics() + expect(stats.throttlingMetrics?.storage?.currentlyThrottled).toBe(true) + + collector.clearThrottlingState() + stats = collector.getStatistics() + expect(stats.throttlingMetrics?.storage?.currentlyThrottled).toBe(false) + expect(stats.throttlingMetrics?.storage?.consecutiveThrottleEvents).toBe(0) + }) + }) + + describe('Integration with BrainyData', () => { + it('should include throttling metrics structure in getStatistics', async () => { + const db = new BrainyData({ + storage: { type: 'memory' }, + embedding: { type: 'use' } + }) + + // Initialize + await db.addBatch([ + { key: 'test1', data: { content: 'test' } } + ]) + + // Get statistics with forceRefresh to ensure collector stats are included + const stats = await db.getStatistics({ forceRefresh: true }) + + // The throttling metrics should be included in the stats from the collector + // Even if there are no throttling events, the structure should exist + // Check that either throttlingMetrics exists or the stats object has the expected base structure + if ((stats as any).throttlingMetrics) { + expect((stats as any).throttlingMetrics).toHaveProperty('storage') + expect((stats as any).throttlingMetrics).toHaveProperty('operationImpact') + + // Check that the metrics have the expected structure + const throttling = (stats as any).throttlingMetrics + expect(throttling.storage).toHaveProperty('currentlyThrottled') + expect(throttling.storage).toHaveProperty('totalThrottleEvents') + expect(throttling.operationImpact).toHaveProperty('delayedOperations') + } else { + // If throttling metrics don't exist yet, at least verify the basic stats structure + expect(stats).toHaveProperty('nounCount') + expect(stats).toHaveProperty('verbCount') + expect(stats).toHaveProperty('metadataCount') + console.log('Note: Throttling metrics not yet included in stats (this is expected initially)') + } + }) + }) +}) \ No newline at end of file diff --git a/tests/type-utils.test.ts b/tests/type-utils.test.ts new file mode 100644 index 00000000..1ac67eb0 --- /dev/null +++ b/tests/type-utils.test.ts @@ -0,0 +1,94 @@ +/** + * Tests for type utility functions + * + * This test file verifies that the utility functions for accessing noun and verb types + * work correctly and return the expected values. + */ + +import { describe, it, expect } from 'vitest' +import { + NounType, + VerbType, + getNounTypes, + getVerbTypes, + getNounTypeMap, + getVerbTypeMap +} from '../src/index.js' + +describe('Type Utility Functions', () => { + describe('getNounTypes', () => { + it('should return an array of all noun types', () => { + const nounTypes = getNounTypes() + + // Check that the result is an array + expect(Array.isArray(nounTypes)).toBe(true) + + // Check that it contains all the expected values + expect(nounTypes).toContain(NounType.Person) + expect(nounTypes).toContain(NounType.Organization) + expect(nounTypes).toContain(NounType.Location) + expect(nounTypes).toContain(NounType.Thing) + expect(nounTypes).toContain(NounType.Concept) + + // Check that the length matches the number of properties in NounType + expect(nounTypes.length).toBe(Object.keys(NounType).length) + }) + }) + + describe('getVerbTypes', () => { + it('should return an array of all verb types', () => { + const verbTypes = getVerbTypes() + + // Check that the result is an array + expect(Array.isArray(verbTypes)).toBe(true) + + // Check that it contains some expected values + expect(verbTypes).toContain(VerbType.RelatedTo) + expect(verbTypes).toContain(VerbType.Contains) + expect(verbTypes).toContain(VerbType.PartOf) + expect(verbTypes).toContain(VerbType.LocatedAt) + expect(verbTypes).toContain(VerbType.References) + + // Check that the length matches the number of properties in VerbType + expect(verbTypes.length).toBe(Object.keys(VerbType).length) + }) + }) + + describe('getNounTypeMap', () => { + it('should return a map of all noun type keys to values', () => { + const nounTypeMap = getNounTypeMap() + + // Check that the result is an object + expect(typeof nounTypeMap).toBe('object') + + // Check that it contains all the expected keys and values + expect(nounTypeMap.Person).toBe(NounType.Person) + expect(nounTypeMap.Organization).toBe(NounType.Organization) + expect(nounTypeMap.Location).toBe(NounType.Location) + expect(nounTypeMap.Thing).toBe(NounType.Thing) + expect(nounTypeMap.Concept).toBe(NounType.Concept) + + // Check that the number of keys matches the number of properties in NounType + expect(Object.keys(nounTypeMap).length).toBe(Object.keys(NounType).length) + }) + }) + + describe('getVerbTypeMap', () => { + it('should return a map of all verb type keys to values', () => { + const verbTypeMap = getVerbTypeMap() + + // Check that the result is an object + expect(typeof verbTypeMap).toBe('object') + + // Check that it contains all the expected keys and values + expect(verbTypeMap.RelatedTo).toBe(VerbType.RelatedTo) + expect(verbTypeMap.Contains).toBe(VerbType.Contains) + expect(verbTypeMap.PartOf).toBe(VerbType.PartOf) + expect(verbTypeMap.LocatedAt).toBe(VerbType.LocatedAt) + expect(verbTypeMap.References).toBe(VerbType.References) + + // Check that the number of keys matches the number of properties in VerbType + expect(Object.keys(verbTypeMap).length).toBe(Object.keys(VerbType).length) + }) + }) +}) diff --git a/tests/unified-api.test.ts b/tests/unified-api.test.ts new file mode 100644 index 00000000..ff970005 --- /dev/null +++ b/tests/unified-api.test.ts @@ -0,0 +1,251 @@ +/** + * Unified API Tests for Brainy 1.0 + * Tests the 7 core unified methods and new functionality + */ + +import { describe, it, expect, beforeEach, afterEach } from 'vitest' +import { BrainyData, NounType, VerbType } from '../src/index.js' + +describe('Brainy 1.0 Unified API', () => { + let brainy: BrainyData + + beforeEach(async () => { + brainy = new BrainyData() + await brainy.init() + }) + + afterEach(async () => { + if (brainy) { + await brainy.cleanup() + } + }) + + describe('Core Method 1: add()', () => { + it('should add data with smart processing by default', async () => { + const id = await brainy.add("John Doe is a software engineer at Tech Corp") + expect(id).toBeDefined() + expect(typeof id).toBe('string') + }) + + it('should add data with literal processing when specified', async () => { + const id = await brainy.add("Raw data", {}, { process: 'literal' }) + expect(id).toBeDefined() + expect(typeof id).toBe('string') + }) + + it('should add data with metadata', async () => { + const metadata = { type: 'person', age: 30 } + const id = await brainy.add("Jane Smith", metadata) + expect(id).toBeDefined() + }) + + it('should add data with encryption', async () => { + const id = await brainy.add("Sensitive data", {}, { encrypt: true }) + expect(id).toBeDefined() + }) + }) + + describe('Core Method 2: search()', () => { + beforeEach(async () => { + // Add test data + await brainy.add("Alice is a data scientist") + await brainy.add("Bob is a software engineer") + await brainy.add("Charlie works in marketing") + }) + + it('should perform vector similarity search', async () => { + const results = await brainy.search("data scientist", 5) + expect(results).toBeDefined() + expect(Array.isArray(results)).toBe(true) + }) + + it('should search with metadata filters', async () => { + await brainy.add("David", { department: "engineering" }) + await brainy.add("Emma", { department: "marketing" }) + + const results = await brainy.search("", 10, { + metadata: { department: "engineering" } + }) + expect(results).toBeDefined() + expect(Array.isArray(results)).toBe(true) + }) + + it('should search connected nouns', async () => { + const personId = await brainy.addNoun("Frank", NounType.Person) + const companyId = await brainy.addNoun("Tech Inc", NounType.Organization) + await brainy.addVerb(personId, companyId, VerbType.WorksWith) + + const results = await brainy.search("", 10, { + searchConnectedNouns: true, + sourceId: personId + }) + expect(results).toBeDefined() + }) + }) + + describe('Core Method 3: import()', () => { + it('should import array of data items', async () => { + const data = [ + "Item 1", + "Item 2", + "Item 3" + ] + const ids = await brainy.import(data) + expect(ids).toBeDefined() + expect(Array.isArray(ids)).toBe(true) + expect(ids.length).toBe(3) + }) + + it('should import with metadata for each item', async () => { + const data = [ + { data: "Item 1", metadata: { category: "A" } }, + { data: "Item 2", metadata: { category: "B" } } + ] + const ids = await brainy.import(data) + expect(ids.length).toBe(2) + }) + }) + + describe('Core Method 4: addNoun()', () => { + it('should add typed noun entities', async () => { + const personId = await brainy.addNoun("John Doe", NounType.Person) + expect(personId).toBeDefined() + + const orgId = await brainy.addNoun("ACME Corp", NounType.Organization) + expect(orgId).toBeDefined() + + const locationId = await brainy.addNoun("San Francisco", NounType.Location) + expect(locationId).toBeDefined() + }) + + it('should add noun with metadata', async () => { + const metadata = { age: 25, role: "Engineer" } + const id = await brainy.addNoun("Jane Smith", NounType.Person, metadata) + expect(id).toBeDefined() + }) + }) + + describe('Core Method 5: addVerb()', () => { + it('should create relationships between nouns', async () => { + const personId = await brainy.addNoun("Bob Wilson", NounType.Person) + const companyId = await brainy.addNoun("Tech Solutions", NounType.Organization) + + const verbId = await brainy.addVerb(personId, companyId, VerbType.WorksWith) + expect(verbId).toBeDefined() + }) + + it('should create verb with metadata and weight', async () => { + const sourceId = await brainy.addNoun("Alice", NounType.Person) + const targetId = await brainy.addNoun("Project Alpha", NounType.Project) + + const verbId = await brainy.addVerb( + sourceId, + targetId, + VerbType.WorksWith, + { role: "Lead Developer", since: "2024" }, + 0.9 + ) + expect(verbId).toBeDefined() + }) + }) + + describe('Core Method 6: update()', () => { + it('should update existing data', async () => { + const id = await brainy.add("Original data") + const success = await brainy.update(id, "Updated data") + expect(success).toBe(true) + }) + + it('should update data and metadata', async () => { + const id = await brainy.add("Data", { version: 1 }) + const success = await brainy.update(id, "Updated data", { version: 2 }) + expect(success).toBe(true) + }) + + it('should update with cascade option', async () => { + const personId = await brainy.addNoun("Charlie", NounType.Person) + const success = await brainy.update( + personId, + "Charles Thompson", + { fullName: "Charles Thompson" }, + { cascade: true } + ) + expect(success).toBe(true) + }) + }) + + describe('Core Method 7: delete()', () => { + it('should soft delete by default', async () => { + const id = await brainy.add("Test data for deletion") + const success = await brainy.delete(id) + expect(success).toBe(true) + + // Should not appear in search results + const results = await brainy.search("Test data for deletion", 10) + expect(results.length).toBe(0) + }) + + it('should hard delete when specified', async () => { + const id = await brainy.add("Data to hard delete") + const success = await brainy.delete(id, { soft: false }) + expect(success).toBe(true) + }) + + it('should cascade delete related verbs', async () => { + const personId = await brainy.addNoun("Dave", NounType.Person) + const projectId = await brainy.addNoun("Project Beta", NounType.Project) + await brainy.addVerb(personId, projectId, VerbType.WorksWith) + + const success = await brainy.delete(personId, { cascade: true }) + expect(success).toBe(true) + }) + + it('should force delete even with relationships', async () => { + const personId = await brainy.addNoun("Eve", NounType.Person) + const taskId = await brainy.addNoun("Important Task", NounType.Task) + await brainy.addVerb(personId, taskId, VerbType.Owns) + + const success = await brainy.delete(personId, { force: true }) + expect(success).toBe(true) + }) + }) + + describe('Encryption Features', () => { + it('should encrypt and decrypt configuration', async () => { + await brainy.setConfig('api-key', 'secret-value', { encrypt: true }) + const value = await brainy.getConfig('api-key', { decrypt: true }) + expect(value).toBe('secret-value') + }) + + it('should encrypt individual data items', async () => { + const encrypted = await brainy.encryptData('sensitive information') + expect(encrypted).toBeDefined() + expect(encrypted).not.toBe('sensitive information') + + const decrypted = await brainy.decryptData(encrypted) + expect(decrypted).toBe('sensitive information') + }) + }) + + describe('Container & Model Preloading', () => { + it('should support model preloading configuration', async () => { + // Test preload configuration (doesn't actually download in tests) + const config = { + model: 'Xenova/all-MiniLM-L6-v2', + cacheDir: './test-models' + } + // Just test that the method exists and doesn't throw + expect(() => BrainyData.preloadModel).not.toThrow() + }) + + it('should support warmup initialization', async () => { + const options = { + storage: { forceMemoryStorage: true } + } + const warmupOptions = { preloadModel: true } + + // Test that warmup method exists and configuration is accepted + expect(() => BrainyData.warmup).not.toThrow() + }) + }) +}) \ No newline at end of file diff --git a/tests/vector-operations.test.ts b/tests/vector-operations.test.ts new file mode 100644 index 00000000..79d86e6a --- /dev/null +++ b/tests/vector-operations.test.ts @@ -0,0 +1,156 @@ +import { describe, it, expect } from 'vitest' +import { euclideanDistance } from '../src/utils/distance.js' + +/** + * Helper function to create a 384-dimensional vector for testing + * @param primaryIndex The index to set to 1.0, all other indices will be 0.0 + * @returns A 384-dimensional vector with a single 1.0 value at the specified index + */ +function createTestVector(primaryIndex: number = 0): number[] { + const vector = new Array(384).fill(0) + vector[primaryIndex % 384] = 1.0 + return vector +} + +describe('Vector Operations', () => { + it('should load brainy library successfully', async () => { + const brainy = await import('../dist/unified.js') + + expect(brainy).toBeDefined() + expect(typeof brainy.BrainyData).toBe('function') + expect(brainy.environment).toBeDefined() + }) + + it('should create and initialize BrainyData instance', async () => { + const brainy = await import('../dist/unified.js') + + const db = new brainy.BrainyData({ + distanceFunction: euclideanDistance + }) + + expect(db).toBeDefined() + expect(db.dimensions).toBe(384) + + await db.init() + // If we get here without throwing, initialization was successful + expect(true).toBe(true) + }) + + it('should handle simple vector operations', async () => { + const brainy = await import('../dist/unified.js') + + // Explicitly use memory storage to avoid FileSystemStorage issues + const storage = await brainy.createStorage({ forceMemoryStorage: true }) + const db = new brainy.BrainyData({ + distanceFunction: euclideanDistance, + storageAdapter: storage + }) + + await db.init() + await db.clear() // Clear any existing data + + // Add a simple vector + const testVector = createTestVector(1) + await db.add(testVector, { id: 'test' }) + + // Search for the same vector + const results = await db.search(testVector, 1) + + expect(results).toBeDefined() + expect(results.length).toBeGreaterThan(0) + expect(results[0].metadata.id).toBe('test') + }) + + it('should handle multiple vector searches correctly', async () => { + const brainy = await import('../dist/unified.js') + + // Explicitly use memory storage to avoid FileSystemStorage issues + const storage = await brainy.createStorage({ forceMemoryStorage: true }) + const db = new brainy.BrainyData({ + distanceFunction: euclideanDistance, + storageAdapter: storage + }) + + await db.init() + await db.clear() // Clear any existing data + + // Add multiple vectors + await db.add(createTestVector(0), { id: 'vec1', type: 'unit' }) + await db.add(createTestVector(1), { id: 'vec2', type: 'unit' }) + await db.add(createTestVector(2), { id: 'vec3', type: 'unit' }) + + // Create a mixed vector with two non-zero elements + const mixedVector = createTestVector(3) + mixedVector[4] = 0.5 + await db.add(mixedVector, { id: 'vec4', type: 'mixed' }) + + // Search for multiple results + const results = await db.search(createTestVector(0), 3) + + expect(results).toBeDefined() + expect(results.length).toBeGreaterThanOrEqual(1) + expect(results.length).toBeLessThanOrEqual(3) + + // The closest should be the exact match + expect(results[0].metadata.id).toBe('vec1') + }) + + it('should calculate similarity between vectors correctly', async () => { + const brainy = await import('../dist/unified.js') + + // Explicitly use memory storage to avoid FileSystemStorage issues + const storage = await brainy.createStorage({ forceMemoryStorage: true }) + const db = new brainy.BrainyData({ + distanceFunction: euclideanDistance, + storageAdapter: storage + }) + + await db.init() + + // Create test vectors + const vectorA = createTestVector(0) + const vectorB = createTestVector(0) // Identical to vectorA + const vectorC = createTestVector(1) // Different from vectorA + + // Calculate similarity between identical vectors + const similarityIdentical = await db.calculateSimilarity(vectorA, vectorB) + + // Calculate similarity between different vectors + const similarityDifferent = await db.calculateSimilarity(vectorA, vectorC) + + // Identical vectors should have similarity close to 1 + expect(similarityIdentical).toBeCloseTo(1, 1) + + // Different vectors should have lower similarity + expect(similarityDifferent).toBeLessThan(similarityIdentical) + }) + + it('should calculate similarity between text inputs correctly', async () => { + const brainy = await import('../dist/unified.js') + + // Explicitly use memory storage to avoid FileSystemStorage issues + const storage = await brainy.createStorage({ forceMemoryStorage: true }) + const db = new brainy.BrainyData({ + storageAdapter: storage + }) + + await db.init() + + // Calculate similarity between similar texts + const similarityHigh = await db.calculateSimilarity( + 'Cats are furry pets', + 'Felines make good companions' + ) + + // Calculate similarity between different texts + const similarityLow = await db.calculateSimilarity( + 'Cats are furry pets', + 'Python is a programming language' + ) + + // Similar texts should have similarity at least as high as different texts + // Note: In some cases with small test texts, the similarity values might be equal + // This is a more robust test that doesn't fail when both are 1 + expect(similarityHigh).toBeGreaterThanOrEqual(similarityLow) + }) +}) diff --git a/tests/verify-custom-models.js b/tests/verify-custom-models.js new file mode 100644 index 00000000..27a18087 --- /dev/null +++ b/tests/verify-custom-models.js @@ -0,0 +1,56 @@ +#!/usr/bin/env node + +/** + * Verify custom models path functionality + */ + +import { BrainyData } from '../dist/unified.js' + +console.log('๐Ÿงช Testing Custom Models Path Functionality\n') + +// Test 1: Environment variable +console.log('Test 1: Environment Variable Support') +process.env.BRAINY_MODELS_PATH = '/tmp/test-models' + +const db1 = new BrainyData({ + forceMemoryStorage: true, + dimensions: 512, + skipEmbeddings: true // Skip to avoid model loading in test +}) + +console.log(` BRAINY_MODELS_PATH set to: ${process.env.BRAINY_MODELS_PATH}`) +console.log(' โœ… Environment variable configuration working') + +// Test 2: Show warning messages +console.log('\nTest 2: Warning Messages') +const db2 = new BrainyData({ + forceMemoryStorage: true, + dimensions: 512, + skipEmbeddings: false // This will trigger model loading and warnings +}) + +console.log(' Initializing with embeddings enabled to show warnings...') +try { + await db2.init() + console.log(' โœ… Model loaded successfully (local models found)') +} catch (error) { + console.log(' โŒ Model loading failed (expected - shows warning messages)') + console.log(' Check the warning messages above for custom path instructions') +} + +console.log('\nTest 3: Model Search Path Priority') +console.log(' The model loader will search in this order:') +console.log(' 1. Custom models path (BRAINY_MODELS_PATH)') +console.log(' 2. @soulcraft/brainy-models package') +console.log(' 3. Fallback to remote URLs') + +console.log('\n๐ŸŽฏ Key Benefits for Docker Deployments:') +console.log(' โ€ข Embed models in Docker images outside node_modules') +console.log(' โ€ข Avoid runtime model downloads') +console.log(' โ€ข Work in offline/restricted network environments') +console.log(' โ€ข Faster application startup') +console.log(' โ€ข Predictable memory usage') + +console.log('\n๐Ÿ“š See examples/docker-deployment/ for complete examples') + +process.exit(0) \ No newline at end of file diff --git a/tests/verify-model-loading.js b/tests/verify-model-loading.js new file mode 100644 index 00000000..ff67ec25 --- /dev/null +++ b/tests/verify-model-loading.js @@ -0,0 +1,37 @@ +#!/usr/bin/env node + +/** + * Simple script to verify model loading behavior + * Run with: node tests/verify-model-loading.js + */ + +import { BrainyData } from '../dist/unified.js' + +console.log('Testing Brainy model loading behavior...\n') + +const db = new BrainyData({ + forceMemoryStorage: true, + dimensions: 512 +}) + +console.log('Initializing BrainyData...') +await db.init() + +console.log('\nAttempting to embed text...') +const text = 'This is a test sentence for embedding' +const id = await db.add({ content: text }) + +console.log(`\nโœ… Successfully added text with ID: ${id}`) + +// Test search +const results = await db.search('test sentence', 1) +console.log(`\nโœ… Search returned ${results.length} result(s)`) + +console.log('\n---') +console.log('Model loading test complete!') +console.log('\nNOTE: Check the console output above to see:') +console.log('1. If @soulcraft/brainy-models was found and used (best performance)') +console.log('2. If fallback to URL loading occurred (with warning)') +console.log('3. Verification that the correct model was loaded') + +process.exit(0) \ No newline at end of file diff --git a/tests/verify-model-priority-simple.js b/tests/verify-model-priority-simple.js new file mode 100644 index 00000000..f62365a8 --- /dev/null +++ b/tests/verify-model-priority-simple.js @@ -0,0 +1,46 @@ +#!/usr/bin/env node + +/** + * Simple test to verify model loading priority messages + */ + +import { BrainyData } from '../dist/unified.js' + +console.log('Testing model loading priority system...\n') + +const db = new BrainyData({ + forceMemoryStorage: true, + dimensions: 512, + skipEmbeddings: true // Skip embeddings to avoid model loading for this test +}) + +console.log('Initializing BrainyData with skipEmbeddings=true...') +await db.init() + +console.log('\nโœ… BrainyData initialized successfully (embeddings skipped)') + +// Now test with embeddings enabled to see the warnings +console.log('\n---') +console.log('Now testing with embeddings enabled to see model loading messages...\n') + +const db2 = new BrainyData({ + forceMemoryStorage: true, + dimensions: 512, + skipEmbeddings: false +}) + +try { + await db2.init() + console.log('\nโœ… Model loaded successfully') +} catch (error) { + console.log('\nโŒ Model loading failed (expected in test environment without actual models)') + console.log(` Error: ${error.message.split('\n')[0]}`) +} + +console.log('\n---') +console.log('Check the output above to verify:') +console.log('1. "Checking for @soulcraft/brainy-models package..." appears') +console.log('2. Warning about falling back to remote loading appears') +console.log('3. Installation suggestion for @soulcraft/brainy-models appears') + +process.exit(0) \ No newline at end of file diff --git a/tests/write-only-direct-reads.test.ts b/tests/write-only-direct-reads.test.ts new file mode 100644 index 00000000..cbc15cd8 --- /dev/null +++ b/tests/write-only-direct-reads.test.ts @@ -0,0 +1,339 @@ +import { describe, test, expect, beforeEach, afterEach } from 'vitest' +import { BrainyData } from '../src/brainyData.js' +import type { BrainyDataConfig } from '../src/brainyData.js' + +describe('Write-Only Mode with Direct Reads', () => { + let brainyWriteOnly: BrainyData + let brainyWithDirectReads: BrainyData + let brainyNormal: BrainyData + + beforeEach(async () => { + // Create instances with different configurations + brainyWriteOnly = new BrainyData({ + writeOnly: true, + allowDirectReads: false + }) + + brainyWithDirectReads = new BrainyData({ + writeOnly: true, + allowDirectReads: true + }) + + brainyNormal = new BrainyData({ + writeOnly: false + }) + + await brainyWriteOnly.init() + await brainyWithDirectReads.init() + await brainyNormal.init() + }) + + afterEach(async () => { + if (brainyWriteOnly) { + await brainyWriteOnly.cleanup?.() + } + if (brainyWithDirectReads) { + await brainyWithDirectReads.cleanup?.() + } + if (brainyNormal) { + await brainyNormal.cleanup?.() + } + }) + + describe('Configuration Validation', () => { + test('should accept allowDirectReads: true with writeOnly: true', () => { + expect(() => new BrainyData({ + writeOnly: true, + allowDirectReads: true + })).not.toThrow() + }) + + test('should accept allowDirectReads: false with writeOnly: true', () => { + expect(() => new BrainyData({ + writeOnly: true, + allowDirectReads: false + })).not.toThrow() + }) + + test('should accept allowDirectReads: true with writeOnly: false', () => { + expect(() => new BrainyData({ + writeOnly: false, + allowDirectReads: true + })).not.toThrow() + }) + }) + + describe('Write Operations (Should Always Work)', () => { + test('should allow add in all modes', async () => { + const testData = 'test string for embedding' + + // All instances should be able to add data + const id1 = await brainyWriteOnly.add(testData) + const id2 = await brainyWithDirectReads.add(testData) + const id3 = await brainyNormal.add(testData) + + expect(id1).toBeTruthy() + expect(id2).toBeTruthy() + expect(id3).toBeTruthy() + }) + + test('should allow add operations with metadata in all modes', async () => { + const testVector = new Array(384).fill(0.1) + const metadata1 = { name: 'test 1', type: 'entity' } + const metadata2 = { name: 'test 2', type: 'entity' } + + // All instances should be able to add data with metadata + const id1 = await brainyWriteOnly.add(testVector, metadata1) + const id2 = await brainyWithDirectReads.add(testVector, metadata2) + + expect(id1).toBeTruthy() + expect(id2).toBeTruthy() + }) + }) + + describe('Direct Read Operations', () => { + let testId: string + + beforeEach(async () => { + // Add test data with metadata for testing + const testVector = new Array(384).fill(0.2) + const testMetadata = { name: 'direct read test', content: 'test content' } + testId = await brainyWithDirectReads.add(testVector, testMetadata) + }) + + describe('get() method', () => { + test('should work in write-only mode without allowDirectReads (legacy behavior)', async () => { + // Add data to write-only instance with metadata + const testVector = new Array(384).fill(0.3) + const id = await brainyWriteOnly.add(testVector, { name: 'legacy test' }) + const result = await brainyWriteOnly.get(id) + expect(result).toBeTruthy() + expect(result?.metadata.name).toBe('legacy test') + }) + + test('should work in write-only mode with allowDirectReads', async () => { + const result = await brainyWithDirectReads.get(testId) + expect(result).toBeTruthy() + expect(result?.metadata.name).toBe('direct read test') + }) + + test('should work in normal mode', async () => { + const testVector = new Array(384).fill(0.4) + const id = await brainyNormal.add(testVector, { name: 'normal test' }) + const result = await brainyNormal.get(id) + expect(result).toBeTruthy() + expect(result?.metadata.name).toBe('normal test') + }) + }) + + describe('has() method', () => { + test('should fail in write-only mode without allowDirectReads', async () => { + await expect(brainyWriteOnly.has(testId)) + .rejects.toThrow('Cannot perform has() operation: database is in write-only mode') + }) + + test('should work in write-only mode with allowDirectReads', async () => { + const exists = await brainyWithDirectReads.has(testId) + expect(exists).toBe(true) + + const notExists = await brainyWithDirectReads.has('nonexistent-id') + expect(notExists).toBe(false) + }) + + test('should work in normal mode', async () => { + const testVector = new Array(384).fill(0.5) + const id = await brainyNormal.add(testVector, { name: 'has test' }) + const exists = await brainyNormal.has(id) + expect(exists).toBe(true) + }) + }) + + describe('exists() method', () => { + test('should fail in write-only mode without allowDirectReads', async () => { + await expect(brainyWriteOnly.exists(testId)) + .rejects.toThrow('Cannot perform has() operation: database is in write-only mode') + }) + + test('should work in write-only mode with allowDirectReads', async () => { + const exists = await brainyWithDirectReads.exists(testId) + expect(exists).toBe(true) + + const notExists = await brainyWithDirectReads.exists('nonexistent-id') + expect(notExists).toBe(false) + }) + }) + + describe('getMetadata() method', () => { + test('should fail in write-only mode without allowDirectReads', async () => { + await expect(brainyWriteOnly.getMetadata(testId)) + .rejects.toThrow('Cannot perform getMetadata() operation: database is in write-only mode') + }) + + test('should work in write-only mode with allowDirectReads', async () => { + const metadata = await brainyWithDirectReads.getMetadata(testId) + expect(metadata).toBeTruthy() + expect(metadata?.name).toBe('direct read test') + }) + + test('should return null for nonexistent ID', async () => { + const metadata = await brainyWithDirectReads.getMetadata('nonexistent-id') + expect(metadata).toBeNull() + }) + }) + + describe('getBatch() method', () => { + test('should fail in write-only mode without allowDirectReads', async () => { + await expect(brainyWriteOnly.getBatch([testId])) + .rejects.toThrow('Cannot perform getBatch() operation: database is in write-only mode') + }) + + test('should work in write-only mode with allowDirectReads', async () => { + const testVector2 = new Array(384).fill(0.6) + const id2 = await brainyWithDirectReads.add(testVector2, { name: 'batch test 2' }) + const results = await brainyWithDirectReads.getBatch([testId, id2, 'nonexistent']) + + expect(results).toHaveLength(3) + expect(results[0]?.metadata.name).toBe('direct read test') + expect(results[1]?.metadata.name).toBe('batch test 2') + expect(results[2]).toBeNull() + }) + + test('should handle empty array', async () => { + const results = await brainyWithDirectReads.getBatch([]) + expect(results).toEqual([]) + }) + }) + + // Note: getVerb() tests removed as the API may not be available in this version + }) + + describe('Search Operations (Should Be Blocked)', () => { + beforeEach(async () => { + // Add some test data + const testVector = new Array(384).fill(0.7) + await brainyWithDirectReads.add(testVector, { name: 'search test', content: 'searchable content' }) + }) + + test('search() should fail in write-only mode even with allowDirectReads', async () => { + await expect(brainyWithDirectReads.search('test')) + .rejects.toThrow('Cannot perform search operation: database is in write-only mode') + }) + + // Note: similar() and query() methods may not be available in this version + }) + + describe('Real-World Use Cases', () => { + describe('Bluesky Service Pattern', () => { + test('should enable efficient deduplication in writer service', async () => { + // Simulate a Bluesky service processing messages + const processMessage = async (did: string, messageData: any) => { + // Check if profile already exists (direct storage lookup) + const existingProfile = await brainyWithDirectReads.get(did) + + if (!existingProfile) { + // Only call external API for new DIDs + const profileData = { did, handle: `user-${did}`, displayName: 'Test User' } + const simpleVector = new Array(384).fill(0.1) + await brainyWithDirectReads.add(simpleVector, profileData, { id: did }) + return { action: 'created', profile: profileData } + } else { + // Profile exists, skip API call + return { action: 'existing', profile: existingProfile.metadata } + } + } + + // Process same DID twice + const result1 = await processMessage('did:test:123', { text: 'Hello' }) + const result2 = await processMessage('did:test:123', { text: 'World' }) + + expect(result1.action).toBe('created') + expect(result2.action).toBe('existing') + expect(result2.profile.did).toBe('did:test:123') + }) + }) + + describe('GitHub Package Pattern', () => { + test('should enable efficient user processing', async () => { + const processUser = async (userId: string) => { + const userKey = `github_user_${userId}` + + // Fast existence check (direct storage, no index) + if (await brainyWithDirectReads.has(userKey)) { + return { action: 'skipped', reason: 'already_processed' } + } + + // New user - simulate API fetch and store + const userData = { id: userId, login: `user${userId}`, type: 'User' } + const simpleVector = new Array(384).fill(0.2) + await brainyWithDirectReads.add(simpleVector, userData, { id: userKey }) + + return { action: 'processed', user: userData } + } + + // Process users + const result1 = await processUser('123') + const result2 = await processUser('123') // Duplicate + const result3 = await processUser('456') // New user + + expect(result1.action).toBe('processed') + expect(result2.action).toBe('skipped') + expect(result3.action).toBe('processed') + }) + }) + + describe('General Writer Service Pattern', () => { + test('should support optimal entity processing', async () => { + const processEntity = async (id: string, data: any) => { + // Fast existence check using direct storage + const existing = await brainyWithDirectReads.get(id) + + if (existing) { + // Update existing entity + return { action: 'updated', existing: existing.metadata, new: data } + } + + // New entity - store it + const simpleVector = new Array(384).fill(0.3) + await brainyWithDirectReads.add(simpleVector, data, { id }) + return { action: 'created', entity: data } + } + + // Test the pattern + const entity1 = { name: 'Entity 1', type: 'test' } + const entity1Updated = { name: 'Entity 1 Updated', type: 'test' } + + const result1 = await processEntity('entity-1', entity1) + const result2 = await processEntity('entity-1', entity1Updated) + + expect(result1.action).toBe('created') + expect(result2.action).toBe('updated') + expect(result2.existing.name).toBe('Entity 1') + }) + }) + }) + + describe('Error Handling', () => { + test('should provide clear error messages for blocked operations', async () => { + await expect(brainyWriteOnly.has('test')) + .rejects.toThrow('Enable allowDirectReads for direct storage operations') + + await expect(brainyWithDirectReads.search('test')) + .rejects.toThrow('Direct storage operations (get, has, exists, getMetadata, getBatch, getVerb) are allowed') + }) + + test('should handle invalid IDs gracefully', async () => { + await expect(brainyWithDirectReads.get(null as any)) + .rejects.toThrow('ID cannot be null or undefined') + + await expect(brainyWithDirectReads.has(undefined as any)) + .rejects.toThrow('ID cannot be null or undefined') + }) + + test('should handle storage errors gracefully', async () => { + // Test with non-existent IDs + expect(await brainyWithDirectReads.has('non-existent')).toBe(false) + expect(await brainyWithDirectReads.get('non-existent')).toBeNull() + expect(await brainyWithDirectReads.getMetadata('non-existent')).toBeNull() + }) + }) +}) \ No newline at end of file diff --git a/vitest.config.ts b/vitest.config.ts new file mode 100644 index 00000000..2c5f77cc --- /dev/null +++ b/vitest.config.ts @@ -0,0 +1,203 @@ +import { defineConfig } from 'vitest/config' + +export default defineConfig({ + test: { + // Default configuration + globals: true, + setupFiles: ['./tests/test-setup.ts', './tests/setup.ts'], + testTimeout: 120000, // 120 seconds for TensorFlow operations + hookTimeout: 120000, + // Run tests in parallel with limited pool + pool: 'forks', + poolOptions: { + forks: { + singleFork: true, // Run all tests in a single fork to reduce memory usage + isolate: false, // Don't isolate tests to reduce overhead + } + }, + // Limit concurrent tests to reduce memory usage + maxConcurrency: 1, + // Clear mocks between tests + clearMocks: true, + restoreMocks: true, + // Include test files + include: ['tests/**/*.{test,spec}.{js,ts}'], + // Default environment + environment: 'node', + // Exclude old test files + exclude: [ + 'node_modules/**', + 'dist/**', + 'scripts/**', + 'examples/**', + '*.js' // Exclude old JS test files in root + ], + // Add environment options to help with TextEncoder issues + environmentOptions: { + env: { + FORCE_PATCHED_PLATFORM: 'true' + } + }, + // Configure reporters for different output formats + reporters: [ + // Default reporter for basic progress during test run + [ + 'default', + { + summary: true, + reportSummary: true, + // Show test titles for all tests + successfulTestOnly: false, + // Show a compact output + outputFile: false + } + ], + // JSON reporter for machine-readable output (can be used for CI/CD) + [ + 'json', + { + outputFile: './tests/results/test-results.json' + } + ] + ], + // Configure output for better visibility + silent: false, + // Configure error display for better readability + bail: 0, + // Disable coverage reports by default to reduce noise + coverage: { + enabled: false + }, + // Don't show test statistics to reduce noise + logHeapUsage: false, + // Hide skipped tests to reduce noise + hideSkippedTests: true, + // Only show stack traces for failed tests + printConsoleTrace: false, + // Show test timing information in the summary + // showTimer: true, + // Aggressively filter out console output to only show test progress + onConsoleLog: (log: string, type: 'stdout' | 'stderr'): false | void => { + // For stdout, only allow critical error messages and test progress indicators + if (type === 'stdout') { + // Only allow through explicit test-related messages and critical errors + const allowedPatterns = [ + 'Error:', + 'FAIL', + 'PASS', + 'WARNING:', + 'test result', + 'Test Files', + 'Tests', + 'Start at', + 'Duration', + 'โœ“', + 'โœ—', + 'running', + 'suite' + ] + + // If the log doesn't contain any allowed pattern, filter it out + if (!allowedPatterns.some((pattern) => log.includes(pattern))) { + return false + } + } + + // For stderr, only show actual errors + if ( + type === 'stderr' && + !log.includes('Error:') && + !log.includes('FAIL') + ) { + return false + } + + // Expanded list of noise patterns to filter out + const noisePatterns: string[] = [ + // Original patterns + 'Brainy:', + 'TensorFlow', + 'Universal Sentence Encoder', + 'Using file system storage', + 'Using WebGL backend', + 'Platform node', + 'The kernel', + 'for backend', + 'is already registered', + 'Hi there ๐Ÿ‘‹', + 'backend registration', + 'webgl', + 'cpu', + 'Could not get context', + 'Retrying', + 'Skipping noun', + 'due to dimension mismatch', + 'Successfully loaded', + 'model loaded', + 'module structure', + 'No default export', + 'Using sentenceEncoderModule', + 'Loading', + 'Applying', + 'Overwriting', + 'Applied', + 'running in Node.js environment', + 'Pre-loading', + 'has already been set', + // Additional patterns to filter out common console.log statements from tests + 'Attempting to add', + 'Successfully added', + 'Test API server running', + 'Text content not found', + 'Searching for text', + 'Expected ID:', + 'Search returned', + 'First result ID:', + 'All result IDs:', + 'Could not find result', + 'Found result with matching ID:', + 'console.log', + 'console.info', + 'console.debug' + ] + + // Return false (don't show) if log contains any noise pattern + if (noisePatterns.some((pattern) => log.includes(pattern))) { + return false + } + + // Additional filtering for common debug output patterns + if ( + log.includes('Searching') || + log.includes('Found') || + log.includes('Created') || + log.includes('Loaded') || + log.includes('Processing') || + log.includes('Initializing') || + log.includes('Starting') || + log.includes('Completed') || + log.includes('Finished') + ) { + return false + } + + return undefined // Show the log if it passes all filters + } + + // Add a custom reporter configuration for a cleaner output + // outputDiffLines: 5, // Limit diff output lines for cleaner error reports + // outputFileMaxLines: 40, // Limit file output lines for cleaner error reports + // outputTruncateLength: 80 // Truncate long output lines + }, + // Resolve configuration for proper module handling + resolve: { + alias: { + '@': './src', + '@tests': './tests' + } + }, + // Define different configurations for different environments + define: { + 'process.env.NODE_ENV': '"test"' + } +})