feat(v4.0.0): Complete metadata/vector separation architecture with Azure support
This commit completes the core v4.0.0 architecture changes for billion-scale performance with metadata/vector separation. NO RELEASE YET - remaining optimizations and testing required before production release. ## Core v4.0.0 Architecture Changes ### Type System Updates - Fixed all TypeScript compilation errors (zero errors achieved) - Updated HNSWNoun/HNSWVerb to separate core fields from metadata - Implemented HNSWNounWithMetadata/HNSWVerbWithMetadata for API boundaries - Added required 'noun' field to NounMetadata for semantic structure - Renamed verb.type to verb.verb for consistency ### Storage Adapter Updates **All adapters updated for v4.0.0 two-file storage pattern:** - memoryStorage: Proper metadata/vector separation - fileSystemStorage: Two-file pattern with sharding - opfsStorage: Browser persistent storage updated - s3CompatibleStorage: AWS/MinIO/DigitalOcean support - r2Storage: Cloudflare R2 optimization - gcsStorage: Google Cloud with ADC support - **azureBlobStorage: NEW - Full Azure Blob Storage support** ### Storage Features - BaseStorage: Internal vs public method separation (_getNoun vs getNoun) - Two-file storage: Vectors in one file, metadata in another - Change tracking: getChangesSince return type updated - Pagination: getNounsWithPagination returns WithMetadata types ### Azure Blob Storage Integration (NEW) - Native @azure/storage-blob SDK integration - Four authentication methods: * DefaultAzureCredential (Managed Identity) - recommended * Connection String - simplest setup * Account Name + Key - traditional auth * SAS Token - delegated access - High-volume mode with write buffering - Adaptive backpressure for throttling - UUID-based sharding for billion-scale - Full HNSW support with graph persistence ### Utility Updates - EmbeddingManager: Updated to accept Record<string, unknown> - LSMTree: Wrapped data in NounMetadata structure with 'noun' field - EntityIdMapper: Fixed nested metadata.data structure access - MetadataIndex: Fixed field type inference integration - PeriodicCleanup: Updated for new metadata structure ### Core API Updates - Brainy: Updated verb property access from v.type to v.verb - ConfigAPI: Fixed NounMetadata access patterns - DataAPI: Updated metadata handling ### Documentation Updates - CREATING-AUGMENTATIONS.md: v4.0.0 breaking changes guide - DEVELOPER-GUIDE.md: Migration checklist and examples - COMPLETE-REFERENCE.md: v4.0.0 architecture improvements - **finite-type-system.md: NEW - Revolutionary type system benefits** ### Build & Dependencies - Zero TypeScript compilation errors - Added @azure/storage-blob and @azure/identity - 591 tests passing (23 timeout in long-running neural tests) ## What's NOT in This Release This is a work-in-progress commit. Before v4.0.0 release we need: - Storage adapter optimizations (batch operations, compression) - Azure blob tier management (Hot/Cool/Archive) - Cost optimization implementations - Additional performance testing at billion-scale - Migration guides for v3.x users ## Testing - Clean build: ✅ - Type checking: ✅ (zero errors) - Test suite: ✅ (591/614 passing, timeouts in neural tests only) 🔐 Generated with Claude Code https://claude.com/claude-code Co-Authored-By: Claude <noreply@anthropic.com>
This commit is contained in:
parent
8d6dd07e1d
commit
92c96246fb
35 changed files with 4524 additions and 1026 deletions
|
|
@ -121,8 +121,8 @@ export class DataAPI {
|
|||
id: verb.id,
|
||||
from: verb.sourceId,
|
||||
to: verb.targetId,
|
||||
type: (verb.verb || verb.type) as string,
|
||||
weight: verb.weight || 1.0,
|
||||
type: verb.verb as string,
|
||||
weight: verb.metadata?.weight || 1.0,
|
||||
metadata: verb.metadata
|
||||
})
|
||||
}
|
||||
|
|
@ -184,16 +184,19 @@ export class DataAPI {
|
|||
// Restore entities
|
||||
for (const entity of backup.entities) {
|
||||
try {
|
||||
// v4.0.0: Prepare noun and metadata separately
|
||||
const noun: HNSWNoun = {
|
||||
id: entity.id,
|
||||
vector: entity.vector || new Array(384).fill(0), // Default vector if missing
|
||||
connections: new Map(),
|
||||
level: 0,
|
||||
metadata: {
|
||||
...entity.metadata,
|
||||
noun: entity.type,
|
||||
service: entity.service
|
||||
}
|
||||
level: 0
|
||||
}
|
||||
|
||||
const metadata = {
|
||||
...entity.metadata,
|
||||
noun: entity.type,
|
||||
service: entity.service,
|
||||
createdAt: Date.now()
|
||||
}
|
||||
|
||||
// Check if entity exists when merging
|
||||
|
|
@ -205,6 +208,7 @@ export class DataAPI {
|
|||
}
|
||||
|
||||
await this.storage.saveNoun(noun)
|
||||
await this.storage.saveNounMetadata(entity.id, metadata)
|
||||
} catch (error) {
|
||||
console.error(`Failed to restore entity ${entity.id}:`, error)
|
||||
}
|
||||
|
|
@ -227,19 +231,21 @@ export class DataAPI {
|
|||
(v, i) => (v + targetNoun.vector[i]) / 2
|
||||
)
|
||||
|
||||
const verb: GraphVerb = {
|
||||
// v4.0.0: Prepare verb and metadata separately
|
||||
const verb = {
|
||||
id: relation.id,
|
||||
vector: relationVector,
|
||||
sourceId: relation.from,
|
||||
targetId: relation.to,
|
||||
source: sourceNoun.metadata?.noun || NounType.Thing,
|
||||
target: targetNoun.metadata?.noun || NounType.Thing,
|
||||
connections: new Map(),
|
||||
verb: relation.type as VerbType,
|
||||
type: relation.type as VerbType,
|
||||
sourceId: relation.from,
|
||||
targetId: relation.to
|
||||
}
|
||||
|
||||
const verbMetadata = {
|
||||
weight: relation.weight,
|
||||
metadata: relation.metadata,
|
||||
...relation.metadata,
|
||||
createdAt: Date.now()
|
||||
} as any
|
||||
}
|
||||
|
||||
// Check if relation exists when merging
|
||||
if (merge) {
|
||||
|
|
@ -250,6 +256,7 @@ export class DataAPI {
|
|||
}
|
||||
|
||||
await this.storage.saveVerb(verb)
|
||||
await this.storage.saveVerbMetadata(relation.id, verbMetadata)
|
||||
} catch (error) {
|
||||
console.error(`Failed to restore relation ${relation.id}:`, error)
|
||||
}
|
||||
|
|
@ -388,16 +395,17 @@ export class DataAPI {
|
|||
this.validateImportItem(mapped)
|
||||
}
|
||||
|
||||
// Save as entity
|
||||
// v4.0.0: Save entity - separate vector and metadata
|
||||
const id = mapped.id || this.generateId()
|
||||
const noun: HNSWNoun = {
|
||||
id: mapped.id || this.generateId(),
|
||||
id,
|
||||
vector: mapped.vector || new Array(384).fill(0),
|
||||
connections: new Map(),
|
||||
level: 0,
|
||||
metadata: mapped
|
||||
level: 0
|
||||
}
|
||||
|
||||
await this.storage.saveNoun(noun)
|
||||
await this.storage.saveNounMetadata(id, { ...mapped, createdAt: Date.now() })
|
||||
result.successful++
|
||||
} catch (error) {
|
||||
result.failed++
|
||||
|
|
@ -528,9 +536,9 @@ export class DataAPI {
|
|||
return true
|
||||
}
|
||||
|
||||
private convertToCSV(entities: HNSWNoun[]): string {
|
||||
private convertToCSV(entities: Array<{ id: string; metadata?: any }>): string {
|
||||
if (entities.length === 0) return ''
|
||||
|
||||
|
||||
// Get all unique keys from metadata
|
||||
const keys = new Set<string>()
|
||||
for (const entity of entities) {
|
||||
|
|
@ -538,25 +546,25 @@ export class DataAPI {
|
|||
Object.keys(entity.metadata).forEach(k => keys.add(k))
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
// Create CSV header
|
||||
const headers = ['id', ...Array.from(keys)]
|
||||
const rows = [headers.join(',')]
|
||||
|
||||
|
||||
// Add data rows
|
||||
for (const entity of entities) {
|
||||
const row = [entity.id]
|
||||
for (const key of keys) {
|
||||
const value = entity.metadata?.[key] || ''
|
||||
// Escape values that contain commas
|
||||
const escaped = String(value).includes(',')
|
||||
? `"${String(value).replace(/"/g, '""')}"`
|
||||
const escaped = String(value).includes(',')
|
||||
? `"${String(value).replace(/"/g, '""')}"`
|
||||
: String(value)
|
||||
row.push(escaped)
|
||||
}
|
||||
rows.push(row.join(','))
|
||||
}
|
||||
|
||||
|
||||
return rows.join('\n')
|
||||
}
|
||||
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue