From c8eb813a1526ec77366e75f75e353ada222c6ffe Mon Sep 17 00:00:00 2001 From: David Snelling Date: Thu, 11 Dec 2025 13:26:07 -0800 Subject: [PATCH] fix(vfs): prevent race condition in bulkWrite by ordering operations MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - Process mkdir operations sequentially first (sorted by path depth) - Then process write/delete/update operations in parallel batches - Prevents duplicate directory entities when mkdir and write for related paths are in the same batch - Add comprehensive tests for bulkWrite race condition scenarios - Update API documentation for accuracy 🤖 Generated with [Claude Code](https://claude.com/claude-code) Co-Authored-By: Claude Opus 4.5 --- docs/README.md | 27 +- docs/api/COMPREHENSIVE_API_OVERVIEW.md | 742 +++++++++++++++---------- docs/api/README.md | 6 +- docs/guides/neural-api.md | 51 +- docs/vfs/VFS_CORE.md | 73 ++- src/vfs/VirtualFileSystem.ts | 72 ++- tests/vfs/vfs-bulkwrite-race.test.ts | 275 +++++++++ 7 files changed, 906 insertions(+), 340 deletions(-) create mode 100644 tests/vfs/vfs-bulkwrite-race.test.ts diff --git a/docs/README.md b/docs/README.md index ab413d43..bed4e26f 100644 --- a/docs/README.md +++ b/docs/README.md @@ -1,4 +1,4 @@ -# Brainy Documentation (v4.0.0) +# Brainy Documentation (v6.5.0) Welcome to the comprehensive documentation for Brainy, the multi-dimensional AI database with Triple Intelligence Engine. @@ -77,28 +77,31 @@ await brain.find("recent articles about AI with high ratings") ## Quick Example ```typescript -import { Brainy } from 'brainy' +import { Brainy, NounType, VerbType } from '@soulcraft/brainy' // Initialize const brain = new Brainy() await brain.init() // Add entities (nouns) -const articleId = await brain.add("Revolutionary AI Breakthrough", { - type: "article", - category: "technology", - rating: 4.8 +const articleId = await brain.add({ + data: "Revolutionary AI Breakthrough", + type: NounType.Document, + metadata: { category: "technology", rating: 4.8 } }) -const authorId = await brain.add("Dr. Sarah Chen", { - type: "person", - role: "researcher" +const authorId = await brain.add({ + data: "Dr. Sarah Chen", + type: NounType.Person, + metadata: { role: "researcher" } }) // Create relationships (verbs) -await brain.relate(authorId, articleId, "authored", { - date: "2024-01-15", - contribution: "primary" +await brain.relate({ + from: authorId, + to: articleId, + type: VerbType.CreatedBy, + metadata: { date: "2024-01-15", contribution: "primary" } }) // Query naturally diff --git a/docs/api/COMPREHENSIVE_API_OVERVIEW.md b/docs/api/COMPREHENSIVE_API_OVERVIEW.md index ae21822a..534759c2 100644 --- a/docs/api/COMPREHENSIVE_API_OVERVIEW.md +++ b/docs/api/COMPREHENSIVE_API_OVERVIEW.md @@ -1,349 +1,503 @@ -# 🧠 **Brainy Complete Public API Overview** +# Brainy Complete Public API Reference -> **Ultra-comprehensive analysis of Brainy's entire API surface for intuitive, consistent developer experience** +> **Accurate API documentation for Brainy v6.5.0+** -## 🎯 **API Consistency Analysis** +## Initialization -### **✅ EXCELLENT Consistency Patterns** - -#### **1. Constructor & Initialization** ```typescript -// Clean, consistent initialization -const brain = new Brainy(config?) -await brain.init() // Always required +import { Brainy, NounType, VerbType } from '@soulcraft/brainy' -// Storage auto-detection works seamlessly -const brain = new Brainy({ storage: { forceMemoryStorage: true } }) -const brain = new Brainy({ storage: { path: './my-data' } }) -``` - -#### **2. Data Operations (CRUD)** -```typescript -// ✅ CONSISTENT: Always (data, metadata) pattern with nounType in metadata -await brain.add(content, { nounType: NounType.Person, role: 'Engineer' }) -await brain.add(content, { nounType: NounType.Document, title: 'API Guide' }) - -// ✅ CONSISTENT: Always (source, target, type, metadata) pattern -await brain.relate(sourceId, targetId, VerbType.RelatedTo, { strength: 0.8 }) -await brain.relate(sourceId, targetId, VerbType.Contains, { confidence: 0.9 }) - -// ✅ CONSISTENT: Batch versions take arrays -await brain.addNouns([...]) // Array of noun objects -await brain.addVerbs([...]) // Array of verb objects -``` - -#### **3. Query Operations** -```typescript -// ✅ CONSISTENT: Always (query, options) pattern -await brain.search('artificial intelligence', { limit: 10, threshold: 0.7 }) -await brain.find('recent documents about AI', { limit: 5 }) // Triple Intelligence - -// ✅ CONSISTENT: Get methods with filters -await brain.getNouns(filter?) // Optional filtering -await brain.getVerbs(filter?) // Optional filtering -await brain.getNoun(id) // Single item by ID -await brain.getVerb(id) // Single item by ID -``` - -#### **4. Main Class Shortcuts (Simple & Common)** -```typescript -// ✅ CONSISTENT: Simple shortcuts for most common operations -await brain.similar(a, b) // Returns simple number -await brain.clusters() // Returns simple array -await brain.related(id, limit?) // Returns simple array -``` - -### **🎨 EXCELLENT API Namespacing** - -#### **Main Data Operations** (Direct on `brain`) -```typescript -// Core CRUD - most common operations -brain.add(content, metadata) -brain.addNouns(items[]) // Batch operation - unchanged -brain.relate(source, target, type, metadata?) -brain.addVerbs(items[]) // Batch operation - unchanged - -brain.search(query, options?) -brain.find(naturalLanguageQuery, options?) // Triple Intelligence -brain.get(id) -brain.getNouns(filter?) -brain.getVerbs(filter?) - -brain.delete(id) -brain.deleteNouns(ids[]) -brain.deleteVerbs(ids[]) -brain.clear() - -// Simple shortcuts for common AI operations -brain.similar(a, b) // Simple similarity -brain.clusters() // Simple clustering -brain.related(id, limit?) // Simple neighbors -``` - -#### **Neural AI Namespace** (`brain.neural.*`) -```typescript -// Advanced AI & Machine Learning operations -brain.neural.similar(a, b, options?) // Full similarity with options -brain.neural.clusters(items?, options?) // Advanced clustering -brain.neural.neighbors(id, options?) // K-nearest neighbors -brain.neural.hierarchy(id, options?) // Semantic hierarchy -brain.neural.outliers(options?) // Anomaly detection -brain.neural.visualize(options?) // Visualization data - -// Advanced clustering methods -brain.neural.clusterByDomain(field, options?) // Domain-aware clustering -brain.neural.clusterByTime(field, windows, options?) // Temporal clustering -brain.neural.clusterStream(options?) // Streaming clustering -brain.neural.updateClusters(items, options?) // Incremental clustering - -// Utility & monitoring -brain.neural.getPerformanceMetrics(operation?) // Performance stats -brain.neural.clearCaches() // Cache management -brain.neural.getCacheStats() // Cache statistics -``` - -#### **Triple Intelligence Namespace** (`brain.triple.*`) -```typescript -// Advanced natural language & complex queries -brain.triple.find(query, options?) // Natural language search -brain.triple.analyze(text, options?) // Text analysis -brain.triple.understand(query, options?) // Query understanding -``` - -#### **Augmentation System** (`brain.augmentations.*`) -```typescript -// Plugin/extension system -brain.augmentations.add(augmentation) -brain.augmentations.remove(name) -brain.augmentations.get(name) -brain.augmentations.list() -brain.augmentations.execute(operation, params) -``` - -#### **Storage & System** (`brain.storage.*`) -```typescript -// Storage management -brain.storage.backup(path?) -brain.storage.restore(path?) -brain.storage.getStatistics() -brain.storage.optimize() -brain.storage.vacuum() -``` - -### **🚀 API Flow & Developer Experience** - -#### **1. Beginner Flow (Simple & Intuitive)** -```typescript -// Dead simple - just works +// Zero-config (just works) const brain = new Brainy() await brain.init() -await brain.add('My first document', { nounType: NounType.Document }) -const results = await brain.search('document') -const similar = await brain.similar('text1', 'text2') -const groups = await brain.clusters() -``` - -#### **2. Intermediate Flow (More Control)** -```typescript -// Add configuration and options +// With configuration const brain = new Brainy({ - storage: { path: './my-brainy-db' }, - neural: { cacheSize: 5000 } + storage: { type: 'memory' }, // or { path: './my-data' } for filesystem + embeddingModel: 'Q8', // Q4, Q8, F16, F32 + silent: true // Suppress logs }) await brain.init() +``` -// Use options for better control -const results = await brain.search('AI research', { - limit: 20, - threshold: 0.8, - filters: { type: 'Document', year: 2024 } +## Core CRUD Operations + +### `brain.add(params)` - Add Entity + +```typescript +// Add with data (auto-embedded) +const id = await brain.add({ + data: 'Machine learning is a subset of AI', + type: NounType.Document, + metadata: { author: 'Alice', tags: ['ml', 'ai'] } }) -// Use neural namespace for advanced features -const clusters = await brain.neural.clusters({ +// Add with pre-computed vector +const id = await brain.add({ + data: 'Content here', + vector: [0.1, 0.2, ...], // 384 dimensions + type: NounType.Concept +}) +``` + +**Parameters:** +- `data` (required): Content to embed (string, object, or any serializable data) +- `type?`: NounType enum value +- `metadata?`: Custom key-value pairs +- `vector?`: Pre-computed embedding vector +- `id?`: Custom ID (auto-generated if not provided) + +### `brain.get(id, options?)` - Get Entity + +```typescript +const entity = await brain.get('entity-id') + +// Include vector embeddings (not loaded by default for performance) +const entity = await brain.get('entity-id', { includeVectors: true }) +``` + +### `brain.update(params)` - Update Entity + +```typescript +await brain.update({ + id: 'entity-id', + data: 'Updated content', // Re-embeds if changed + metadata: { reviewed: true } // Merges with existing +}) +``` + +### `brain.delete(id)` - Delete Entity + +```typescript +await brain.delete('entity-id') +``` + +### `brain.clear()` - Clear All Data + +```typescript +await brain.clear() +``` + +## Relationships + +### `brain.relate(params)` - Create Relationship + +```typescript +const relationId = await brain.relate({ + from: 'source-entity-id', + to: 'target-entity-id', + type: VerbType.RelatedTo, + metadata: { strength: 0.9 } +}) +``` + +**Parameters:** +- `from` (required): Source entity ID +- `to` (required): Target entity ID +- `type` (required): VerbType enum value +- `metadata?`: Custom relationship metadata + +### `brain.getRelations(params)` - Query Relationships + +```typescript +// Get all relationships from an entity +const relations = await brain.getRelations({ from: 'entity-id' }) + +// Get relationships to an entity +const relations = await brain.getRelations({ to: 'entity-id' }) + +// Filter by type +const relations = await brain.getRelations({ + from: 'entity-id', + type: VerbType.Contains +}) +``` + +### `brain.unrelate(id)` - Delete Relationship + +```typescript +await brain.unrelate('relationship-id') +``` + +## Search & Query + +### `brain.find(query)` - Semantic Search + +The primary search method with Triple Intelligence (semantic + graph + metadata). + +```typescript +// Simple text search +const results = await brain.find('machine learning algorithms') + +// With options +const results = await brain.find({ + query: 'machine learning', + limit: 10, + threshold: 0.7, + type: NounType.Document, + where: { author: 'Alice' }, + excludeVFS: true // Exclude VFS files from results +}) + +// Natural language query (Triple Intelligence) +const results = await brain.find( + 'Show me documents about AI written by Alice in 2024' +) +``` + +**Parameters:** +- `query`: Search text (required) +- `limit?`: Max results (default: 10) +- `threshold?`: Minimum similarity (0-1) +- `type?`: Filter by NounType +- `where?`: Metadata filters +- `excludeVFS?`: Exclude VFS entities + +### `brain.similar(params)` - Find Similar Entities + +```typescript +// Find similar to entity ID +const similar = await brain.similar({ + to: 'entity-id', + limit: 5 +}) + +// Find similar to vector +const similar = await brain.similar({ + to: [0.1, 0.2, ...], // Vector + limit: 10, + type: NounType.Document +}) +``` + +## Batch Operations + +### `brain.addMany(params)` - Batch Add + +```typescript +const result = await brain.addMany({ + items: [ + { data: 'First item', type: NounType.Document }, + { data: 'Second item', type: NounType.Concept }, + { data: 'Third item', metadata: { priority: 'high' } } + ], + continueOnError: true, // Don't stop on failures + onProgress: (completed, total) => { + console.log(`${completed}/${total} complete`) + } +}) + +console.log(`Added: ${result.successful.length}, Failed: ${result.failed.length}`) +``` + +### `brain.deleteMany(params)` - Batch Delete + +```typescript +// Delete by IDs +const result = await brain.deleteMany({ + ids: ['id1', 'id2', 'id3'], + continueOnError: true +}) + +// Delete by type +const result = await brain.deleteMany({ + type: NounType.TempData +}) + +// Delete by metadata filter +const result = await brain.deleteMany({ + where: { status: 'archived' } +}) +``` + +### `brain.relateMany(params)` - Batch Relate + +```typescript +const ids = await brain.relateMany({ + items: [ + { from: 'a', to: 'b', type: VerbType.RelatedTo }, + { from: 'b', to: 'c', type: VerbType.Contains }, + { from: 'c', to: 'd', type: VerbType.References } + ], + continueOnError: true +}) +``` + +### `brain.updateMany(params)` - Batch Update + +```typescript +const result = await brain.updateMany({ + items: [ + { id: 'id1', metadata: { reviewed: true } }, + { id: 'id2', data: 'Updated content' } + ], + continueOnError: true +}) +``` + +## Neural API + +Access advanced AI/ML features via `brain.neural()`: + +```typescript +const neural = brain.neural() + +// Similarity calculation +const similarity = await neural.similar('text1', 'text2') +const detailed = await neural.similar('text1', 'text2', { detailed: true }) + +// Clustering +const clusters = await neural.clusters() +const clusters = await neural.clusters({ algorithm: 'hierarchical', maxClusters: 10 }) -``` -#### **3. Advanced Flow (Full Power)** -```typescript -// Complex natural language queries -const insights = await brain.find(` - Show me documents about machine learning from 2024 - that are connected to research papers with high citations -`) +// K-nearest neighbors +const neighbors = await neural.neighbors('entity-id', { k: 5 }) -// Advanced temporal analysis -const trends = await brain.neural.clusterByTime('publishedAt', [ - { start: new Date('2024-01-01'), end: new Date('2024-06-30'), label: 'H1 2024' }, - { start: new Date('2024-07-01'), end: new Date('2024-12-31'), label: 'H2 2024' } +// Semantic hierarchy +const hierarchy = await neural.hierarchy('entity-id') + +// Outlier/anomaly detection +const outliers = await neural.outliers({ threshold: 2.0 }) + +// Domain-aware clustering +const domainClusters = await neural.clusterByDomain('category') + +// Temporal clustering +const temporalClusters = await neural.clusterByTime('createdAt', [ + { start: new Date('2024-01-01'), end: new Date('2024-06-30'), label: 'H1' }, + { start: new Date('2024-07-01'), end: new Date('2024-12-31'), label: 'H2' } ]) -// Real-time streaming clustering -for await (const batch of brain.neural.clusterStream({ batchSize: 50 })) { - console.log(`Processed ${batch.progress.percentage}% - Found ${batch.clusters.length} clusters`) +// Streaming clusters (for large datasets) +for await (const batch of neural.clusterStream({ batchSize: 100 })) { + console.log(`Progress: ${batch.progress.percentage}%`) } ``` -## 📊 **Parameter Consistency Analysis** +## Virtual File System (VFS) -### **✅ Excellent Consistency** +Access via `brain.vfs`: -#### **1. Data-First Pattern** ```typescript -// Always: (data, config/metadata, optional_params) -brain.add(content, metadata) // nounType now in metadata -brain.relate(source, target, VerbType.RelatedTo, metadata?) -brain.search(query, options?) -brain.similar(a, b, options?) +const vfs = brain.vfs +await vfs.init() + +// File operations +await vfs.writeFile('/docs/readme.md', '# Hello') +const content = await vfs.readFile('/docs/readme.md') +await vfs.unlink('/docs/readme.md') + +// Directory operations +await vfs.mkdir('/project/src', { recursive: true }) +const files = await vfs.readdir('/project') +await vfs.rmdir('/project', { recursive: true }) + +// Bulk operations (v6.5.0+) +const result = await vfs.bulkWrite([ + { type: 'mkdir', path: '/data' }, + { type: 'write', path: '/data/config.json', data: '{}' }, + { type: 'write', path: '/data/users.json', data: '[]' } +]) +// Note: mkdir operations run first (sequentially), then other ops in parallel + +// Semantic search +const results = await vfs.search('authentication code', { path: '/src' }) + +// File metadata +const stats = await vfs.stat('/file.txt') +await vfs.setMetadata('/file.txt', { author: 'Alice' }) ``` -#### **2. Options Objects** +## Counts (O(1) Performance) + ```typescript -// Consistent options pattern across all methods -{ - limit?: number - threshold?: number - filters?: Record - algorithm?: string - includeMetadata?: boolean +// Entity counts +const total = brain.counts.entities() +const byType = await brain.counts.byType(NounType.Document) +const nonVFS = await brain.counts.byType({ excludeVFS: true }) + +// Relationship counts +const relations = brain.counts.relationships() +const byVerb = await brain.counts.byVerbType(VerbType.Contains) +``` + +## Versioning & Branching + +```typescript +// Create branch +await brain.fork('feature-branch') + +// List branches +const branches = await brain.listBranches() + +// Switch branch +await brain.checkout('feature-branch') + +// Get current branch +const current = await brain.getCurrentBranch() + +// Commit changes +await brain.commit({ message: 'Added new features' }) + +// View history +const history = await brain.getHistory({ limit: 10 }) + +// Time travel (read-only snapshot) +const snapshot = await brain.asOf('commit-id') +``` + +## Streaming API + +```typescript +// Stream all entities +for await (const entity of brain.streaming.entities()) { + console.log(entity.id) +} + +// Stream with filters +for await (const entity of brain.streaming.entities({ + type: NounType.Document, + where: { status: 'active' } +})) { + // Process each entity +} + +// Stream relationships +for await (const relation of brain.streaming.relations({ + from: 'entity-id' +})) { + console.log(relation) } ``` -#### **3. Array Methods** +## Pagination API + ```typescript -// Pluralized versions always take arrays -brain.addNouns([{ vectorOrData: '...', nounType: NounType.Content }]) -brain.addVerbs([{ source: '...', target: '...', type: VerbType.RelatedTo }]) -brain.deleteNouns(['id1', 'id2']) -brain.deleteVerbs(['id1', 'id2']) +// Paginated queries +const page1 = await brain.pagination.find({ + query: 'machine learning', + page: 1, + pageSize: 20 +}) + +console.log(`Page ${page1.page} of ${page1.totalPages}`) +console.log(`Total results: ${page1.total}`) + +// Get next page +const page2 = await brain.pagination.find({ + query: 'machine learning', + page: 2, + pageSize: 20 +}) ``` -### **Return Type Consistency** +## Augmentations -#### **1. Simple Returns (Shortcuts)** ```typescript -brain.similar(a, b) → Promise // Always simple number -brain.clusters() → Promise // Always simple array -brain.related(id) → Promise // Always simple array +// List active augmentations +const augmentations = brain.augmentations.list() + +// Get specific augmentation +const cache = brain.augmentations.get('cache') + +// Augmentations are auto-loaded based on config +// Common augmentations: cache, display, metrics, intelligent-import ``` -#### **2. Rich Returns (Neural Namespace)** +## Utilities + ```typescript -brain.neural.similar(a, b, { detailed: true }) → Promise -brain.neural.neighbors(id, options) → Promise -brain.neural.clusters(options) → Promise +// Manual embedding +const vector = await brain.embed('text to embed') + +// Flush pending writes +await brain.flush() + +// Get statistics +const stats = await brain.getStats() +const statsNoVFS = await brain.getStats({ excludeVFS: true }) + +// Close (cleanup) +await brain.close() ``` -#### **3. Consistent Error Handling** +## Type Enums + +```typescript +import { NounType, VerbType } from '@soulcraft/brainy' + +// NounType - Entity types +NounType.Document +NounType.Person +NounType.Concept +NounType.Event +NounType.Location +NounType.Organization +NounType.Product +NounType.Content +NounType.Collection +// ... and more + +// VerbType - Relationship types +VerbType.RelatedTo +VerbType.Contains +VerbType.References +VerbType.DependsOn +VerbType.Precedes +VerbType.Follows +VerbType.CreatedBy +VerbType.ModifiedBy +// ... and more +``` + +## Error Handling + ```typescript -// All methods throw descriptive errors with context try { - await brain.neural.similar('invalid', 'data') + await brain.get('nonexistent-id') } catch (error) { - // error.code: 'SIMILARITY_ERROR' - // error.context: { inputA: '...', inputB: '...' } + if (error.message.includes('not found')) { + // Handle missing entity + } +} + +// VFS errors use POSIX codes +try { + await vfs.readFile('/nonexistent') +} catch (error) { + if (error.code === 'ENOENT') { + // File not found + } } ``` -## 🎯 **Key Strengths of Current API** +## Configuration Reference -### **✅ 1. Progressive Disclosure** -- **Simple**: `brain.similar()` → just returns a number -- **Advanced**: `brain.neural.similar()` → full options & detailed results - -### **✅ 2. Intuitive Namespacing** -- **Core data**: Direct on `brain` (addNoun, search, delete) -- **AI features**: `brain.neural.*` (clustering, similarity, analysis) -- **System**: `brain.storage.*`, `brain.augmentations.*` - -### **✅ 3. Consistent Patterns** -- **Always** `(data, options?)` parameter order -- **Always** async/Promise-based -- **Always** descriptive error messages with context - -### **✅ 4. Type Safety** ```typescript -// Excellent TypeScript support -import { Brainy, NounType, VerbType } from '@soulcraft/brainy' - -const brain = new Brainy() -await brain.add('content', { nounType: NounType.Document, title: 'My Doc' }) -// ^^^^^^^^^^^^^^^^ // IDE autocomplete! -``` - -### **✅ 5. Flexible Configuration** -```typescript -// Zero-config (just works) -const brain = new Brainy() - -// Full control when needed const brain = new Brainy({ - storage: { - adapter: 'file', - path: './my-data', - encryption: true + // Storage + storage: { + type: 'memory' | 'filesystem', + path: './data', // For filesystem + forceMemoryStorage: false // Force memory even if path exists }, - neural: { - cacheSize: 10000, - defaultAlgorithm: 'hierarchical' - }, - logging: { verbose: true } + + // Embeddings + embeddingModel: 'Q8', // Q4, Q8, F16, F32 + dimensions: 384, // Auto-detected + + // Performance + silent: false, // Suppress console output + verbose: false, // Extra logging + + // Augmentations (auto-enabled by default) + augmentations: { + cache: true, + display: true, + metrics: true + } }) ``` - -## 🔍 **Minor Improvement Opportunities** - -### **1. Documentation Consistency** -```typescript -// ✅ GREAT: Clear, descriptive JSDoc -/** - * Add semantic relationship between two items - * @param source - Source item ID - * @param target - Target item ID - * @param type - Relationship type (VerbType enum) - * @param metadata - Optional relationship metadata - */ -brain.relate(source, target, type, metadata?) -``` - -### **2. Error Context Enhancement** -```typescript -// Current: Good error messages -// Improvement: Add suggested fixes -throw new SimilarityError('Failed to calculate similarity', { - inputA: 'invalid-id', - inputB: 'valid-id', - suggestion: 'Check that both IDs exist in the database' -}) -``` - -## 🎖️ **Overall API Grade: A+ (Excellent)** - -### **Strengths:** -- **🎯 Intuitive**: Natural method names, clear hierarchy -- **🔄 Consistent**: Same patterns everywhere -- **📈 Progressive**: Simple → advanced as needed -- **🛡️ Type-safe**: Full TypeScript support -- **📚 Well-documented**: Clear examples & guides -- **🚀 Performant**: Smart caching, batching, streaming - -### **Neural API Fits Perfectly:** -- **✅ Namespace consistency**: `brain.neural.*` is clear and logical -- **✅ Parameter consistency**: Follows same `(data, options?)` pattern -- **✅ Return consistency**: Rich objects when needed, simple types for shortcuts -- **✅ Progressive disclosure**: `brain.similar()` → `brain.neural.similar()` -- **✅ Advanced features**: Domain/temporal clustering, streaming, analysis - -### **Developer Experience Score: 🌟🌟🌟🌟🌟 (5/5 stars)** - -The API surface is **exceptionally well designed** with: -- **Beginner-friendly** shortcuts that "just work" -- **Advanced features** available when needed -- **Consistent patterns** across all methods -- **Logical namespacing** that guides developers naturally -- **Rich ecosystem** with augmentations, Triple Intelligence, and neural features - -**The neural namespace integrates seamlessly and enhances rather than complicates the overall API experience.** \ No newline at end of file diff --git a/docs/api/README.md b/docs/api/README.md index f74fdb0f..625ea5e6 100644 --- a/docs/api/README.md +++ b/docs/api/README.md @@ -1,9 +1,9 @@ -# 🧠 Brainy v5.0+ API Reference +# 🧠 Brainy v6.5.0 API Reference -> **Complete API documentation for Brainy v5.0+** +> **Complete API documentation for Brainy v6.5.0** > Zero Configuration • Triple Intelligence • Git-Style Branching • Entity Versioning -**Updated:** 2025-12-09 for v6.3.2 +**Updated:** 2025-12-11 for v6.5.0 **All APIs verified against actual code** --- diff --git a/docs/guides/neural-api.md b/docs/guides/neural-api.md index a703162b..c4b9568d 100644 --- a/docs/guides/neural-api.md +++ b/docs/guides/neural-api.md @@ -4,7 +4,7 @@ ## Overview -The Neural API provides advanced AI-powered features for understanding relationships and patterns in your data. Access it through `brain.neural` after initializing Brainy. +The Neural API provides advanced AI-powered features for understanding relationships and patterns in your data. Access it through `brain.neural()` (method call) after initializing Brainy. ## Quick Start @@ -14,8 +14,8 @@ import { Brainy } from '@soulcraft/brainy' const brain = new Brainy() await brain.init() -// Access Neural API -const neural = brain.neural +// Access Neural API (note: neural() is a method call) +const neural = brain.neural() // Find similar items const similarity = await neural.similar('text1', 'text2') @@ -235,49 +235,64 @@ const sameTopicArticles = currentTopic.members ### Customer Feedback Analysis ```javascript +import { NounType } from '@soulcraft/brainy' + // Add feedback with metadata const feedbackIds = [] for (const feedback of customerFeedback) { - const id = await brain.add(feedback.text, { - rating: feedback.rating, - date: feedback.date, - product: feedback.product + const id = await brain.add({ + data: feedback.text, + type: NounType.Document, + metadata: { + rating: feedback.rating, + date: feedback.date, + product: feedback.product + } }) feedbackIds.push(id) } // Cluster to find themes +const neural = brain.neural() const themes = await neural.clusters(feedbackIds) // Analyze each theme for (const theme of themes) { - const items = await brain.getNouns(theme.members) - - const avgRating = items.reduce((sum, item) => - sum + item.metadata.rating, 0) / items.length - + // Get items using Promise.all with brain.get() + const items = await Promise.all( + theme.members.map(id => brain.get(id)) + ) + + const avgRating = items.reduce((sum, item) => + sum + (item?.metadata?.rating || 0), 0) / items.length + console.log(`Theme with ${theme.members.length} items`) console.log(`Average rating: ${avgRating}`) - + // Find representative feedback for this theme const centroidId = theme.members[0] // Closest to center - const example = await brain.getNoun(centroidId) - console.log(`Example: "${example.data}"`) + const example = await brain.get(centroidId) + console.log(`Example: "${example?.data}"`) } ``` ### Knowledge Base Organization ```javascript -// Analyze existing knowledge base -const allDocs = await brain.getNouns({ type: 'document' }) +import { NounType } from '@soulcraft/brainy' + +// Analyze existing knowledge base - use find() to get documents +const allDocs = await brain.find({ type: NounType.Document, limit: 1000 }) + +// Access neural API +const neural = brain.neural() // Find duplicate or highly similar content const duplicates = [] for (let i = 0; i < allDocs.length; i++) { for (let j = i + 1; j < allDocs.length; j++) { const similarity = await neural.similar( - allDocs[i].id, + allDocs[i].id, allDocs[j].id ) if (similarity > 0.95) { diff --git a/docs/vfs/VFS_CORE.md b/docs/vfs/VFS_CORE.md index def208a2..5634bef2 100644 --- a/docs/vfs/VFS_CORE.md +++ b/docs/vfs/VFS_CORE.md @@ -13,7 +13,7 @@ The main VFS class (`src/vfs/VirtualFileSystem.ts`) provides all filesystem oper ```javascript const brain = new Brainy({ storage: { type: 'memory' } }) await brain.init() -const vfs = brain.vfs() +const vfs = brain.vfs await vfs.init() ``` @@ -366,31 +366,90 @@ VFS scales to millions of files: All VFS methods are available immediately after initialization: ```javascript -const vfs = brain.vfs() +const vfs = brain.vfs await vfs.init() // Core file operations await vfs.writeFile() // Write files await vfs.readFile() // Read files +await vfs.appendFile() // Append to files await vfs.mkdir() // Create directories await vfs.readdir() // List directory contents +await vfs.rmdir() // Remove directories await vfs.stat() // Get file metadata +await vfs.exists() // Check if path exists +await vfs.unlink() // Delete files + +// Path operations +await vfs.copy() // Copy files/directories +await vfs.move() // Move files/directories +await vfs.rename() // Rename files/directories +await vfs.chmod() // Change permissions +await vfs.chown() // Change ownership + +// Symlinks +await vfs.symlink() // Create symbolic link +await vfs.readlink() // Read symlink target +await vfs.realpath() // Resolve symlink to real path // Semantic features await vfs.search() // Semantic search await vfs.findSimilar() // Find similar files await vfs.addRelationship()// Add relationships -// Todo management +// Todo & metadata management await vfs.addTodo() // Add todos await vfs.getTodos() // Get todos +await vfs.setTodos() // Set all todos await vfs.setMetadata() // Set metadata +await vfs.getMetadata() // Get metadata -// Export -await vfs.exportToJSON() // Export to JSON -await vfs.bulkWrite() // Bulk operations +// Tree operations +await vfs.getTreeStructure() // Get tree structure +await vfs.getDirectChildren() // Get direct children + +// Bulk operations +await vfs.bulkWrite() // Bulk write operations + +// Import +await vfs.importFile() // Import file from filesystem +await vfs.importDirectory()// Import directory from filesystem ``` +## Bulk Write Operations + +**v6.5.0**: `bulkWrite` efficiently processes multiple VFS operations with automatic ordering to prevent race conditions. + +```javascript +const result = await vfs.bulkWrite([ + { type: 'mkdir', path: '/project/src' }, + { type: 'mkdir', path: '/project/tests' }, + { type: 'write', path: '/project/src/index.ts', data: '// main file' }, + { type: 'write', path: '/project/package.json', data: '{}' }, + { type: 'delete', path: '/old-file.txt' }, + { type: 'update', path: '/config.json', options: { metadata: { updated: true } } } +]) + +console.log(`Successful: ${result.successful}, Failed: ${result.failed.length}`) +``` + +**Supported operation types:** +- `mkdir` - Create directory (with optional `options: { recursive: true }`) +- `write` - Write file content +- `delete` - Delete file +- `update` - Update file metadata only + +**Operation ordering (v6.5.0):** +1. `mkdir` operations run **first**, sequentially, sorted by path depth (shallowest first) +2. Other operations (`write`, `delete`, `update`) run **after** in parallel batches of 10 + +This ordering prevents race conditions where file writes might fail because parent directories haven't been created yet. + +**Error handling:** +- Operations continue on individual failures +- Failed operations are recorded in `result.failed` with error details +- Use `recursive: true` on mkdir to make them idempotent + ## Complete Example ```javascript @@ -404,7 +463,7 @@ async function vfsExample() { }) await brain.init() - const vfs = brain.vfs() + const vfs = brain.vfs await vfs.init() // Create project structure diff --git a/src/vfs/VirtualFileSystem.ts b/src/vfs/VirtualFileSystem.ts index 7e444535..84c77843 100644 --- a/src/vfs/VirtualFileSystem.ts +++ b/src/vfs/VirtualFileSystem.ts @@ -2338,8 +2338,48 @@ export class VirtualFileSystem implements IVirtualFileSystem { })) } + /** + * Sort bulk operations to prevent race conditions + * + * Strategy: + * 1. mkdir operations first, sorted by path depth (shallowest first) + * 2. Other operations (write, delete, update) after, in original order + * + * This ensures parent directories exist before files are written, + * preventing duplicate entity creation from concurrent mkdir calls. + */ + private sortBulkOperations(operations: Array<{ + type: 'write' | 'delete' | 'mkdir' | 'update' + path: string + data?: Buffer | string + options?: any + }>): Array { + const mkdirOps: typeof operations = [] + const otherOps: typeof operations = [] + + for (const op of operations) { + if (op.type === 'mkdir') { + mkdirOps.push(op) + } else { + otherOps.push(op) + } + } + + // Sort mkdir by path depth (shallowest first) + mkdirOps.sort((a, b) => { + const depthA = (a.path.match(/\//g) || []).length + const depthB = (b.path.match(/\//g) || []).length + return depthA !== depthB ? depthA - depthB : a.path.localeCompare(b.path) + }) + + return [...mkdirOps, ...otherOps] + } + /** * Bulk write operations for performance + * + * v6.5.0: Prevents race condition by processing mkdir operations + * sequentially before parallel batch processing of other operations. */ async bulkWrite(operations: Array<{ type: 'write' | 'delete' | 'mkdir' | 'update' @@ -2357,10 +2397,33 @@ export class VirtualFileSystem implements IVirtualFileSystem { failed: [] as Array<{ operation: any, error: string }> } - // Process operations in batches for better performance + // Sort operations: mkdirs first (by depth), then others + const sortedOps = this.sortBulkOperations(operations) + + // Separate mkdir operations for sequential processing + const mkdirOps = sortedOps.filter(op => op.type === 'mkdir') + const otherOps = sortedOps.filter(op => op.type !== 'mkdir') + + // Phase 1: Process mkdir operations SEQUENTIALLY + // This prevents the race condition where parallel mkdir calls + // create duplicate directory entities due to mutex timing window + for (const op of mkdirOps) { + try { + await this.mkdir(op.path, op.options) + result.successful++ + } catch (error: any) { + result.failed.push({ + operation: op, + error: error.message || 'Unknown error' + }) + } + } + + // Phase 2: Process other operations in parallel batches + // These can safely run in parallel since parent directories now exist const batchSize = 10 - for (let i = 0; i < operations.length; i += batchSize) { - const batch = operations.slice(i, i + batchSize) + for (let i = 0; i < otherOps.length; i += batchSize) { + const batch = otherOps.slice(i, i + batchSize) // Process batch in parallel const promises = batch.map(async (op) => { @@ -2372,9 +2435,6 @@ export class VirtualFileSystem implements IVirtualFileSystem { case 'delete': await this.unlink(op.path) break - case 'mkdir': - await this.mkdir(op.path, op.options) - break case 'update': { // Update only metadata without changing content const entityId = await this.pathResolver.resolve(op.path) diff --git a/tests/vfs/vfs-bulkwrite-race.test.ts b/tests/vfs/vfs-bulkwrite-race.test.ts new file mode 100644 index 00000000..19d883f3 --- /dev/null +++ b/tests/vfs/vfs-bulkwrite-race.test.ts @@ -0,0 +1,275 @@ +/** + * VFS bulkWrite Race Condition Fix Tests (v6.5.0) + * + * Tests for the race condition where parallel mkdir and write operations + * could create duplicate directory entities, causing files to become invisible. + * + * Bug: When mkdir and write for related paths are in the same parallel batch, + * the mkdir mutex race window can create duplicate entities. Files created + * with the "wrong" parent entity become invisible in tree traversal. + * + * Fix: Sort operations so mkdirs run first (sequentially, by depth), then + * other operations in parallel batches. + */ + +import { describe, it, expect, beforeEach } from 'vitest' +import { Brainy } from '../../src/brainy.js' +import { VirtualFileSystem } from '../../src/vfs/VirtualFileSystem.js' + +describe('VFS bulkWrite Race Condition Fix', () => { + let brain: Brainy + let vfs: VirtualFileSystem + + beforeEach(async () => { + brain = new Brainy({ + storage: { type: 'memory' }, + embeddingModel: 'Q8' + }) + await brain.init() + vfs = brain.vfs + await vfs.init() + }) + + describe('operation ordering', () => { + it('should create directories before files when mixed in same batch', async () => { + // This is the exact scenario that triggered the race condition: + // mkdir and write for related paths in the same batch + const result = await vfs.bulkWrite([ + { type: 'write', path: '/data/config.json', data: '{}' }, + { type: 'mkdir', path: '/data' }, + { type: 'write', path: '/data/users.json', data: '[]' }, + { type: 'mkdir', path: '/logs' }, + { type: 'write', path: '/logs/app.log', data: 'log entry' } + ]) + + expect(result.successful).toBe(5) + expect(result.failed.length).toBe(0) + + // Verify all files are visible + const dataFiles = await vfs.readdir('/data') + expect(dataFiles).toContain('config.json') + expect(dataFiles).toContain('users.json') + + const logFiles = await vfs.readdir('/logs') + expect(logFiles).toContain('app.log') + }) + + it('should handle nested directory creation in correct order', async () => { + const result = await vfs.bulkWrite([ + { type: 'write', path: '/a/b/c/file.txt', data: 'content' }, + { type: 'mkdir', path: '/a/b/c' }, // deepest + { type: 'mkdir', path: '/a' }, // shallowest + { type: 'mkdir', path: '/a/b' }, // middle + ]) + + expect(result.successful).toBe(4) + + // Verify tree structure is correct + const rootDirs = await vfs.readdir('/') + expect(rootDirs).toContain('a') + + const aContent = await vfs.readdir('/a') + expect(aContent).toContain('b') + + const bContent = await vfs.readdir('/a/b') + expect(bContent).toContain('c') + + const cContent = await vfs.readdir('/a/b/c') + expect(cContent).toContain('file.txt') + }) + + it('should not create duplicate directory entities under concurrent load', async () => { + // Simulate the exact race condition scenario with many operations + const operations: Array<{ + type: 'write' | 'mkdir' + path: string + data?: string + options?: { recursive?: boolean } + }> = [] + + // Mix mkdir and write operations that would trigger race condition + // Using recursive: true makes mkdir idempotent (no error if exists) + for (let i = 0; i < 20; i++) { + operations.push({ type: 'mkdir', path: `/concurrent-test-${i % 5}`, options: { recursive: true } }) + operations.push({ + type: 'write', + path: `/concurrent-test-${i % 5}/file${i}.txt`, + data: `content ${i}` + }) + } + + const result = await vfs.bulkWrite(operations) + + // All operations should succeed (mkdirs are idempotent with recursive: true) + expect(result.failed.length).toBe(0) + + // Verify no duplicate directories + const rootChildren = await vfs.getDirectChildren('/') + const dirNames = rootChildren + .filter(c => c.metadata.vfsType === 'directory') + .map(c => c.metadata.name) + + // Each directory name should appear exactly once + for (let i = 0; i < 5; i++) { + const count = dirNames.filter(n => n === `concurrent-test-${i}`).length + expect(count).toBe(1) + } + + // Verify all files are visible in their directories + for (let i = 0; i < 5; i++) { + const files = await vfs.readdir(`/concurrent-test-${i}`) + expect(files.length).toBe(4) // 4 files per directory (indices 0,5,10,15 for dir 0, etc.) + } + }) + + it('should handle the Workshop template creation scenario', async () => { + // Exact scenario from bug report: template creation with mixed ops + const operations = [ + { type: 'mkdir' as const, path: '/project/src' }, + { type: 'mkdir' as const, path: '/project/src/components' }, + { type: 'write' as const, path: '/project/src/index.ts', data: '// index' }, + { type: 'write' as const, path: '/project/src/components/App.tsx', data: '// app' }, + { type: 'mkdir' as const, path: '/project/public' }, + { type: 'write' as const, path: '/project/public/index.html', data: '' }, + { type: 'write' as const, path: '/project/package.json', data: '{}' }, + { type: 'write' as const, path: '/project/README.md', data: '# Project' }, + { type: 'mkdir' as const, path: '/project' }, // Parent after children - should work + { type: 'write' as const, path: '/project/tsconfig.json', data: '{}' }, + ] + + const result = await vfs.bulkWrite(operations) + + expect(result.successful).toBe(10) + expect(result.failed.length).toBe(0) + + // Verify tree structure via getTreeStructure (this was failing before fix) + const tree = await vfs.getTreeStructure('/', { maxDepth: 4 }) + + // Find project directory + const projectDir = tree.children?.find(c => c.name === 'project') + expect(projectDir).toBeDefined() + expect(projectDir?.type).toBe('directory') + + // Verify all files are visible + const projectFiles = await vfs.readdir('/project') + expect(projectFiles).toContain('src') + expect(projectFiles).toContain('public') + expect(projectFiles).toContain('package.json') + expect(projectFiles).toContain('README.md') + expect(projectFiles).toContain('tsconfig.json') + + const srcFiles = await vfs.readdir('/project/src') + expect(srcFiles).toContain('index.ts') + expect(srcFiles).toContain('components') + + const componentFiles = await vfs.readdir('/project/src/components') + expect(componentFiles).toContain('App.tsx') + }) + }) + + describe('edge cases', () => { + it('should handle empty operations array', async () => { + const result = await vfs.bulkWrite([]) + expect(result.successful).toBe(0) + expect(result.failed.length).toBe(0) + }) + + it('should handle only mkdir operations', async () => { + const result = await vfs.bulkWrite([ + { type: 'mkdir', path: '/only-dirs/a' }, + { type: 'mkdir', path: '/only-dirs/b' }, + { type: 'mkdir', path: '/only-dirs' } + ]) + + expect(result.successful).toBe(3) + expect(await vfs.exists('/only-dirs/a')).toBe(true) + expect(await vfs.exists('/only-dirs/b')).toBe(true) + }) + + it('should handle only write operations', async () => { + // Pre-create directory + await vfs.mkdir('/files-only', { recursive: true }) + + const result = await vfs.bulkWrite([ + { type: 'write', path: '/files-only/a.txt', data: 'a' }, + { type: 'write', path: '/files-only/b.txt', data: 'b' }, + { type: 'write', path: '/files-only/c.txt', data: 'c' } + ]) + + expect(result.successful).toBe(3) + }) + + it('should handle mkdir for already existing directories gracefully', async () => { + // Pre-create directory + await vfs.mkdir('/existing', { recursive: true }) + + const result = await vfs.bulkWrite([ + { type: 'mkdir', path: '/existing', options: { recursive: true } }, + { type: 'write', path: '/existing/new-file.txt', data: 'content' } + ]) + + // mkdir should succeed (no-op for existing dir with recursive: true) + expect(result.successful).toBe(2) + expect(await vfs.exists('/existing/new-file.txt')).toBe(true) + }) + + it('should handle delete operations after writes', async () => { + // First create some files + await vfs.mkdir('/temp', { recursive: true }) + await vfs.writeFile('/temp/to-delete.txt', 'delete me') + await vfs.writeFile('/temp/to-keep.txt', 'keep me') + + const result = await vfs.bulkWrite([ + { type: 'write', path: '/temp/new-file.txt', data: 'new' }, + { type: 'delete', path: '/temp/to-delete.txt' } + ]) + + expect(result.successful).toBe(2) + expect(await vfs.exists('/temp/new-file.txt')).toBe(true) + expect(await vfs.exists('/temp/to-delete.txt')).toBe(false) + expect(await vfs.exists('/temp/to-keep.txt')).toBe(true) + }) + + it('should handle update operations', async () => { + await vfs.writeFile('/doc.txt', 'original content') + + const result = await vfs.bulkWrite([ + { type: 'update', path: '/doc.txt', options: { metadata: { custom: 'value' } } } + ]) + + expect(result.successful).toBe(1) + + const entity = await vfs.getEntity('/doc.txt') + expect(entity.metadata.custom).toBe('value') + }) + }) + + describe('error handling', () => { + it('should continue processing after mkdir failure', async () => { + // Create a file where we'll try to mkdir + await vfs.writeFile('/not-a-dir', 'content') + + const result = await vfs.bulkWrite([ + { type: 'mkdir', path: '/not-a-dir' }, // Will fail - file exists + { type: 'mkdir', path: '/good-dir' }, + { type: 'write', path: '/good-dir/file.txt', data: 'content' } + ]) + + expect(result.successful).toBe(2) // good-dir and file.txt + expect(result.failed.length).toBe(1) // not-a-dir + expect(await vfs.exists('/good-dir/file.txt')).toBe(true) + }) + + it('should continue processing after write failure', async () => { + const result = await vfs.bulkWrite([ + { type: 'mkdir', path: '/test-dir' }, + { type: 'write', path: '/test-dir/good.txt', data: 'content' }, + { type: 'delete', path: '/nonexistent/file.txt' } // Will fail + ]) + + expect(result.successful).toBe(2) // mkdir and write + expect(result.failed.length).toBe(1) // delete + expect(await vfs.exists('/test-dir/good.txt')).toBe(true) + }) + }) +})