2025-11-05 17:01:44 -08:00
|
|
|
/**
|
|
|
|
|
* Historical Storage Adapter
|
|
|
|
|
*
|
|
|
|
|
* Provides lazy-loading read-only access to a historical commit state.
|
|
|
|
|
* Uses LRU cache to bound memory usage and prevent eager-loading of entire history.
|
|
|
|
|
*
|
|
|
|
|
* Architecture:
|
|
|
|
|
* - Extends BaseStorage to inherit all storage infrastructure
|
|
|
|
|
* - Wraps an underlying storage adapter to access commit state
|
|
|
|
|
* - Implements lazy-loading with LRU cache (bounded memory)
|
|
|
|
|
* - All writes throw read-only errors
|
|
|
|
|
* - All reads load from historical commit state on-demand
|
|
|
|
|
*
|
|
|
|
|
* Usage:
|
|
|
|
|
* const historical = new HistoricalStorageAdapter({
|
|
|
|
|
* underlyingStorage: brain.storage as BaseStorage,
|
|
|
|
|
* commitId: 'abc123...',
|
|
|
|
|
* cacheSize: 10000 // LRU cache size
|
|
|
|
|
* })
|
|
|
|
|
* await historical.init()
|
|
|
|
|
*
|
|
|
|
|
* Performance:
|
|
|
|
|
* - O(1) cache lookups for frequently accessed entities
|
|
|
|
|
* - Bounded memory: max cacheSize entities in memory
|
|
|
|
|
* - Lazy loading: only loads entities when accessed
|
|
|
|
|
* - No eager-loading of entire commit state
|
|
|
|
|
*
|
2026-01-27 15:38:21 -08:00
|
|
|
* Production-ready, billion-scale historical queries
|
2025-11-05 17:01:44 -08:00
|
|
|
*/
|
|
|
|
|
|
|
|
|
|
import { BaseStorage } from '../baseStorage.js'
|
|
|
|
|
import { CommitLog } from '../cow/CommitLog.js'
|
|
|
|
|
import { BlobStorage } from '../cow/BlobStorage.js'
|
|
|
|
|
import {
|
|
|
|
|
HNSWNoun,
|
|
|
|
|
HNSWVerb,
|
|
|
|
|
NounMetadata,
|
|
|
|
|
VerbMetadata,
|
|
|
|
|
StatisticsData
|
|
|
|
|
} from '../../coreTypes.js'
|
|
|
|
|
|
|
|
|
|
/**
|
|
|
|
|
* Simple LRU Cache implementation
|
|
|
|
|
* Bounds memory usage by evicting least-recently-used items
|
|
|
|
|
*/
|
|
|
|
|
class LRUCache<T> {
|
|
|
|
|
private cache = new Map<string, T>()
|
|
|
|
|
private accessOrder: string[] = []
|
|
|
|
|
private maxSize: number
|
|
|
|
|
|
|
|
|
|
constructor(maxSize: number = 10000) {
|
|
|
|
|
this.maxSize = maxSize
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
get(key: string): T | undefined {
|
|
|
|
|
const value = this.cache.get(key)
|
|
|
|
|
if (value !== undefined) {
|
|
|
|
|
// Move to end (most recently used)
|
|
|
|
|
this.accessOrder = this.accessOrder.filter(k => k !== key)
|
|
|
|
|
this.accessOrder.push(key)
|
|
|
|
|
}
|
|
|
|
|
return value
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
set(key: string, value: T): void {
|
|
|
|
|
// Remove if already exists
|
|
|
|
|
if (this.cache.has(key)) {
|
|
|
|
|
this.accessOrder = this.accessOrder.filter(k => k !== key)
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// Add to cache
|
|
|
|
|
this.cache.set(key, value)
|
|
|
|
|
this.accessOrder.push(key)
|
|
|
|
|
|
|
|
|
|
// Evict oldest if over capacity
|
|
|
|
|
if (this.cache.size > this.maxSize) {
|
|
|
|
|
const oldest = this.accessOrder.shift()
|
|
|
|
|
if (oldest) {
|
|
|
|
|
this.cache.delete(oldest)
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
has(key: string): boolean {
|
|
|
|
|
return this.cache.has(key)
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
clear(): void {
|
|
|
|
|
this.cache.clear()
|
|
|
|
|
this.accessOrder = []
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
get size(): number {
|
|
|
|
|
return this.cache.size
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
export interface HistoricalStorageAdapterOptions {
|
|
|
|
|
/** Underlying storage to access commit state from */
|
|
|
|
|
underlyingStorage: BaseStorage
|
|
|
|
|
|
|
|
|
|
/** Commit ID to load historical state from */
|
|
|
|
|
commitId: string
|
|
|
|
|
|
|
|
|
|
/** Max number of entities to cache (default: 10000) */
|
|
|
|
|
cacheSize?: number
|
|
|
|
|
|
|
|
|
|
/** Branch containing the commit (default: 'main') */
|
|
|
|
|
branch?: string
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
/**
|
|
|
|
|
* Historical Storage Adapter
|
|
|
|
|
*
|
|
|
|
|
* Lazy-loading, read-only storage adapter for historical commit state.
|
|
|
|
|
* Implements billion-scale time-travel queries with bounded memory.
|
|
|
|
|
*/
|
|
|
|
|
export class HistoricalStorageAdapter extends BaseStorage {
|
|
|
|
|
private underlyingStorage: BaseStorage
|
|
|
|
|
private commitId: string
|
|
|
|
|
private branch: string
|
|
|
|
|
private cacheSize: number
|
|
|
|
|
|
|
|
|
|
// LRU caches for lazy-loaded entities
|
|
|
|
|
private cache: LRUCache<any>
|
|
|
|
|
|
|
|
|
|
// Historical commit state (loaded lazily) - must match BaseStorage visibility
|
|
|
|
|
public commitLog?: CommitLog
|
|
|
|
|
public blobStorage?: BlobStorage
|
|
|
|
|
|
|
|
|
|
constructor(options: HistoricalStorageAdapterOptions) {
|
|
|
|
|
super()
|
|
|
|
|
this.underlyingStorage = options.underlyingStorage
|
|
|
|
|
this.commitId = options.commitId
|
|
|
|
|
this.branch = options.branch || 'main'
|
|
|
|
|
this.cacheSize = options.cacheSize || 10000
|
|
|
|
|
|
|
|
|
|
this.cache = new LRUCache(this.cacheSize)
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
/**
|
|
|
|
|
* Initialize historical storage adapter
|
|
|
|
|
* Loads commit metadata but NOT entity data (lazy loading)
|
|
|
|
|
*/
|
|
|
|
|
public async init(): Promise<void> {
|
|
|
|
|
// Get COW components from underlying storage
|
2026-01-27 15:38:21 -08:00
|
|
|
// Fixed property names - use public properties without underscore prefix
|
2025-12-02 11:22:04 -08:00
|
|
|
this.commitLog = this.underlyingStorage.commitLog
|
|
|
|
|
this.blobStorage = this.underlyingStorage.blobStorage
|
2025-11-05 17:01:44 -08:00
|
|
|
|
2025-12-02 11:22:04 -08:00
|
|
|
if (!this.commitLog || !this.blobStorage) {
|
2025-11-05 17:01:44 -08:00
|
|
|
throw new Error(
|
|
|
|
|
'Historical storage requires underlying storage to have COW enabled. ' +
|
|
|
|
|
'Call brain.init() first to initialize COW.'
|
|
|
|
|
)
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// Verify commit exists
|
|
|
|
|
const commit = await this.commitLog.getCommit(this.commitId)
|
|
|
|
|
if (!commit) {
|
|
|
|
|
throw new Error(`Commit not found: ${this.commitId}`)
|
|
|
|
|
}
|
|
|
|
|
|
2026-01-27 15:38:21 -08:00
|
|
|
// Initialize GraphAdjacencyIndex and type statistics
|
feat: ID-first storage architecture + remove memory-unsafe APIs (v6.0.0)
BREAKING CHANGES:
**ID-First Storage Paths**
- Direct O(1) entity access without type lookups
- Before: entities/nouns/{TYPE}/metadata/{SHARD}/{ID}.json
- After: entities/nouns/{SHARD}/{ID}/metadata.json
- Migration handled automatically on first init()
**Removed Memory-Unsafe APIs**
- Removed brain.merge() - loaded all entities into memory
- Removed brain.diff() - loaded all entities into memory
- Removed brain.data().backup() - loaded all entities into memory
- Removed brain.data().restore() - depended on backup()
- Removed CLI commands: backup, restore, cow merge
**Migration Paths**
- merge() → Use checkout() or manually copy entities with pagination
- diff() → Use asOf() with manual paginated comparison
- backup() → Use fork() for instant COW snapshots
- restore() → Use checkout() to switch to snapshot branch
Core Improvements:
- ✅ All 8 storage adapters properly call super.init()
- ✅ GraphAdjacencyIndex integration in BaseStorage.init()
- ✅ Fixed ID-first path bugs (vector.json → vectors.json)
- ✅ Fixed MemoryStorage.initializeCounts() for ID-first paths
- ✅ New VFS APIs: du(), access(), find()
- ✅ Comprehensive documentation with migration guides
Storage Adapters Fixed:
- MemoryStorage, FileSystemStorage, AzureBlobStorage
- GCSStorage, R2Storage, S3CompatibleStorage
- OPFSStorage, HistoricalStorageAdapter
Files Changed: 28 files, +1,075/-1,933 lines (net -858)
🤖 Generated with [Claude Code](https://claude.com/claude-code)
Co-Authored-By: Claude <noreply@anthropic.com>
2025-11-19 16:46:11 -08:00
|
|
|
await super.init()
|
2025-11-05 17:01:44 -08:00
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// ============= Abstract Method Implementations =============
|
|
|
|
|
|
|
|
|
|
/**
|
|
|
|
|
* Read object from historical commit state
|
|
|
|
|
* Uses LRU cache to avoid repeated blob reads
|
|
|
|
|
*/
|
|
|
|
|
protected async readObjectFromPath(path: string): Promise<any | null> {
|
|
|
|
|
// Check cache first
|
|
|
|
|
if (this.cache.has(path)) {
|
|
|
|
|
return this.cache.get(path) || null
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
try {
|
|
|
|
|
// Import COW classes
|
|
|
|
|
const { CommitObject } = await import('../cow/CommitObject.js')
|
|
|
|
|
const { TreeObject } = await import('../cow/TreeObject.js')
|
|
|
|
|
const { isNullHash } = await import('../cow/constants.js')
|
|
|
|
|
|
|
|
|
|
// Read commit
|
|
|
|
|
const commit = await CommitObject.read(this.blobStorage!, this.commitId)
|
|
|
|
|
if (isNullHash(commit.tree)) {
|
|
|
|
|
return null
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// Read tree
|
|
|
|
|
const tree = await TreeObject.read(this.blobStorage!, commit.tree)
|
|
|
|
|
|
|
|
|
|
// Walk tree to find matching path
|
|
|
|
|
for await (const entry of TreeObject.walk(this.blobStorage!, tree)) {
|
|
|
|
|
if (entry.type === 'blob' && entry.name === path) {
|
|
|
|
|
// Read blob data
|
|
|
|
|
const blobData = await this.blobStorage!.read(entry.hash)
|
|
|
|
|
const data = JSON.parse(blobData.toString())
|
|
|
|
|
|
|
|
|
|
// Cache the result
|
|
|
|
|
this.cache.set(path, data)
|
|
|
|
|
|
|
|
|
|
return data
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
return null
|
|
|
|
|
} catch (error) {
|
|
|
|
|
// Path doesn't exist in historical state
|
|
|
|
|
return null
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
/**
|
|
|
|
|
* List objects under path in historical commit state
|
|
|
|
|
*/
|
|
|
|
|
protected async listObjectsUnderPath(prefix: string): Promise<string[]> {
|
|
|
|
|
try {
|
|
|
|
|
// Import COW classes
|
|
|
|
|
const { CommitObject } = await import('../cow/CommitObject.js')
|
|
|
|
|
const { TreeObject } = await import('../cow/TreeObject.js')
|
|
|
|
|
const { isNullHash } = await import('../cow/constants.js')
|
|
|
|
|
|
|
|
|
|
// Read commit
|
|
|
|
|
const commit = await CommitObject.read(this.blobStorage!, this.commitId)
|
|
|
|
|
if (isNullHash(commit.tree)) {
|
|
|
|
|
return []
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// Read tree
|
|
|
|
|
const tree = await TreeObject.read(this.blobStorage!, commit.tree)
|
|
|
|
|
|
|
|
|
|
// Walk tree to find all paths matching prefix
|
|
|
|
|
const paths: string[] = []
|
|
|
|
|
for await (const entry of TreeObject.walk(this.blobStorage!, tree)) {
|
|
|
|
|
if (entry.name.startsWith(prefix)) {
|
|
|
|
|
paths.push(entry.name)
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
return paths
|
|
|
|
|
} catch (error) {
|
|
|
|
|
return []
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
/**
|
|
|
|
|
* WRITE BLOCKED: Historical storage is read-only
|
|
|
|
|
*/
|
|
|
|
|
protected async writeObjectToPath(path: string, data: any): Promise<void> {
|
|
|
|
|
throw new Error(
|
|
|
|
|
`Historical storage is read-only. Cannot write to path: ${path}`
|
|
|
|
|
)
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
/**
|
|
|
|
|
* DELETE BLOCKED: Historical storage is read-only
|
|
|
|
|
*/
|
|
|
|
|
protected async deleteObjectFromPath(path: string): Promise<void> {
|
|
|
|
|
throw new Error(
|
|
|
|
|
`Historical storage is read-only. Cannot delete path: ${path}`
|
|
|
|
|
)
|
|
|
|
|
}
|
|
|
|
|
|
feat(storage): add raw binary-blob primitive to every storage adapter
Introduce a first-class binary-blob storage primitive on the StorageAdapter
contract and implement it across all storage backends. This stores opaque byte
payloads verbatim instead of base64-in-JSON, eliminating the ~33% inflation and
full-materialization cost of the JSON envelope. It unblocks zero-copy,
mmap-able column-store segments and batch vector I/O at billion scale.
New methods (declared abstract on BaseStorageAdapter, the class that implements
StorageAdapter, and added to the StorageAdapter interface):
saveBinaryBlob(key, data) raw write, atomic on real filesystems
loadBinaryBlob(key) exact bytes, or null if absent
deleteBinaryBlob(key) idempotent (missing is ignored)
getBinaryBlobPath(key) real local fs path where one exists, else null
Shared key -> location convention across every adapter: the key's
"/"-separated segments nest under a `_blobs/` prefix and are suffixed with
`.bin`, e.g. "graph-lsm/source/sstable-123" ->
"<root>/_blobs/graph-lsm/source/sstable-123.bin". Blobs are not branch-scoped
(COW): they are immutable producer-managed segments.
Per-adapter behavior:
- FileSystemStorage: writes under <rootDir>/_blobs via tmp+rename; returns the
real on-disk path so native code can mmap it directly. Path convention matches
the existing MmapFileSystemStorage subclass byte-for-byte.
- S3CompatibleStorage / R2Storage / GcsStorage / AzureBlobStorage: put/get/delete
raw octet-stream objects; getBinaryBlobPath returns null (remote stores have no
local path).
- MemoryStorage: defensive-copied Map<string, Buffer>; null path; cleared on
clear().
- OPFSStorage: stores raw bytes in the OPFS tree; null path.
- HistoricalStorageAdapter: read-only — save/delete throw; load resolves the
blob from the historical commit tree; null path.
Tests: tests/unit/storage/binaryBlob.test.ts exercises save/load round-trip
(byte-identical, incl. non-UTF8 bytes), overwrite, delete-then-load, load-missing,
and getBinaryBlobPath behavior for all eight adapters. Cloud adapters run against
in-memory client fakes that drive the real adapter code; OPFS runs against an
in-memory FileSystem Access API mock; the historical adapter commits a blob into
a real COW tree. 59 new tests; full unit suite (1398 tests) green.
2026-05-27 11:49:49 -07:00
|
|
|
// ===========================================================================
|
|
|
|
|
// Raw binary-blob primitive (read-only)
|
|
|
|
|
// ===========================================================================
|
|
|
|
|
|
|
|
|
|
/**
|
|
|
|
|
* WRITE BLOCKED: Historical storage is read-only.
|
|
|
|
|
*
|
|
|
|
|
* @param key - The blob key (unused; included for the error message).
|
|
|
|
|
* @throws Always — historical state is immutable.
|
|
|
|
|
*/
|
|
|
|
|
public async saveBinaryBlob(key: string, _data: Buffer): Promise<void> {
|
|
|
|
|
throw new Error(
|
|
|
|
|
`Historical storage is read-only. Cannot save binary blob: ${key}`
|
|
|
|
|
)
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
/**
|
|
|
|
|
* Load the raw bytes of a binary blob from the historical commit state by
|
|
|
|
|
* walking the commit tree for the blob entry at `_blobs/<key>.bin` and reading
|
|
|
|
|
* its content-addressed bytes verbatim (no JSON decode). Returns `null` if the
|
|
|
|
|
* blob was not present in this commit.
|
|
|
|
|
*
|
|
|
|
|
* @param key - The blob key (same convention as the live adapters).
|
|
|
|
|
* @returns The blob bytes as stored at this commit, or `null` if absent.
|
|
|
|
|
*/
|
|
|
|
|
public async loadBinaryBlob(key: string): Promise<Buffer | null> {
|
|
|
|
|
try {
|
|
|
|
|
const { CommitObject } = await import('../cow/CommitObject.js')
|
|
|
|
|
const { TreeObject } = await import('../cow/TreeObject.js')
|
|
|
|
|
const { isNullHash } = await import('../cow/constants.js')
|
|
|
|
|
|
|
|
|
|
const commit = await CommitObject.read(this.blobStorage!, this.commitId)
|
|
|
|
|
if (isNullHash(commit.tree)) {
|
|
|
|
|
return null
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
const tree = await TreeObject.read(this.blobStorage!, commit.tree)
|
|
|
|
|
const blobName = `_blobs/${key}.bin`
|
|
|
|
|
|
|
|
|
|
for await (const entry of TreeObject.walk(this.blobStorage!, tree)) {
|
|
|
|
|
if (entry.type === 'blob' && entry.name === blobName) {
|
|
|
|
|
return await this.blobStorage!.read(entry.hash)
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
return null
|
|
|
|
|
} catch (error) {
|
|
|
|
|
// Blob not present in historical state
|
|
|
|
|
return null
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
/**
|
|
|
|
|
* DELETE BLOCKED: Historical storage is read-only.
|
|
|
|
|
*
|
|
|
|
|
* @param key - The blob key (unused; included for the error message).
|
|
|
|
|
* @throws Always — historical state is immutable.
|
|
|
|
|
*/
|
|
|
|
|
public async deleteBinaryBlob(key: string): Promise<void> {
|
|
|
|
|
throw new Error(
|
|
|
|
|
`Historical storage is read-only. Cannot delete binary blob: ${key}`
|
|
|
|
|
)
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
/**
|
|
|
|
|
* Historical state lives in the content-addressed commit store, not on the
|
|
|
|
|
* local filesystem, so there is no mmap-able path. Always returns `null`.
|
|
|
|
|
*
|
|
|
|
|
* @param _key - The blob key (unused).
|
|
|
|
|
* @returns Always `null`.
|
|
|
|
|
*/
|
|
|
|
|
public getBinaryBlobPath(_key: string): string | null {
|
|
|
|
|
return null
|
|
|
|
|
}
|
|
|
|
|
|
2025-11-05 17:01:44 -08:00
|
|
|
/**
|
|
|
|
|
* Get storage statistics from historical commit
|
|
|
|
|
*/
|
|
|
|
|
protected async getStatisticsData(): Promise<StatisticsData | null> {
|
|
|
|
|
return await this.readObjectFromPath('_system/statistics.json')
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
/**
|
|
|
|
|
* WRITE BLOCKED: Cannot save statistics to historical storage
|
|
|
|
|
*/
|
|
|
|
|
protected async saveStatisticsData(data: StatisticsData): Promise<void> {
|
|
|
|
|
throw new Error('Historical storage is read-only. Cannot save statistics.')
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
/**
|
|
|
|
|
* Clear cache (does not affect historical data)
|
|
|
|
|
*/
|
|
|
|
|
public async clear(): Promise<void> {
|
|
|
|
|
this.cache.clear()
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
/**
|
|
|
|
|
* Get storage status
|
|
|
|
|
*/
|
|
|
|
|
public async getStorageStatus(): Promise<{
|
|
|
|
|
type: string
|
|
|
|
|
used: number
|
|
|
|
|
quota: number | null
|
|
|
|
|
details?: Record<string, any>
|
|
|
|
|
}> {
|
|
|
|
|
return {
|
|
|
|
|
type: 'historical',
|
|
|
|
|
used: this.cache.size,
|
|
|
|
|
quota: this.cacheSize,
|
|
|
|
|
details: {
|
|
|
|
|
commitId: this.commitId,
|
|
|
|
|
branch: this.branch,
|
|
|
|
|
cached: this.cache.size,
|
|
|
|
|
maxCache: this.cacheSize,
|
|
|
|
|
readOnly: true
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
2025-11-17 10:44:35 -08:00
|
|
|
/**
|
|
|
|
|
* Check if COW has been explicitly disabled via clear()
|
2026-01-27 15:38:21 -08:00
|
|
|
* No-op for HistoricalStorageAdapter (read-only, doesn't manage COW)
|
2025-11-17 10:44:35 -08:00
|
|
|
* @returns Always false (read-only adapter doesn't manage COW state)
|
|
|
|
|
* @protected
|
|
|
|
|
*/
|
|
|
|
|
/**
|
2026-01-27 15:38:21 -08:00
|
|
|
* Removed checkClearMarker() and createClearMarker() methods
|
feat: COW always-on architecture + cloud storage clear() fix (v5.11.0)
Major architectural improvements and critical bug fixes:
## COW Always-On Architecture
- Removed cowEnabled flag from BaseStorage (COW cannot be disabled)
- Eliminated marker file system (checkClearMarker, createClearMarker)
- Simplified all code paths to assume COW is always enabled
- COW automatically re-initializes after clear() operations
## Critical Bug Fix: Cloud Storage clear()
- Fixed GCS clear() using correct paths (branches/ instead of entities/nouns/)
- Fixed S3Compatible clear() path structure
- Fixed R2 clear() implementation
- Fixed Azure, FileSystem, OPFS, Memory clear() COW flag handling
- clear() now deletes: branches/, _cow/, _system/
- Result: Cloud buckets can now be fully cleared (previously impossible)
## Container Memory Detection
- Auto-detect Docker/K8s/Cloud Run memory limits (cgroup v1/v2)
- Smart memory allocation (75% graph data, 25% query operations)
- Environment variable support (CLOUD_RUN_MEMORY, MEMORY_LIMIT)
- Production-grade containerized deployment support
## CommitLog streamHistory Feature
- Added streamable commit history with pagination
- Efficient memory usage for large commit histories
- Support for branch filtering and time ranges
## Comprehensive Storage Documentation
- Complete v5.11.0 file structure reference
- Detailed path construction algorithms
- 8 common storage scenarios with examples
- Type-first storage, sharding, COW architecture explained
- Public docs: docs/architecture/data-storage-architecture.md (1063 lines)
## Files Modified (14 files)
- All 8 storage adapters (GCS, S3, R2, Azure, FS, OPFS, Memory, Historical)
- BaseStorage core architecture
- CommitLog with streaming
- Brainy memory configuration
- Parameter validation with container detection
- Storage architecture documentation
## Breaking Changes
NONE - COW was already enabled by default. This removes the ability to disable it.
## Migration
No action required. Upgrade and clear() will work correctly on cloud storage.
## Impact
- Users can now clear cloud storage buckets completely
- No more corrupted buckets after clear() operations
- Container deployments automatically optimize memory allocation
- COW is mandatory and always enabled (safer, simpler)
v5.11.0 - Production ready
2025-11-18 13:44:02 -08:00
|
|
|
* COW is now always enabled - marker files are no longer used
|
2025-11-17 10:44:35 -08:00
|
|
|
*/
|
|
|
|
|
|
2025-11-05 17:01:44 -08:00
|
|
|
// ============= Override Write Methods (Read-Only) =============
|
|
|
|
|
|
|
|
|
|
/**
|
|
|
|
|
* WRITE BLOCKED: Historical storage is read-only
|
|
|
|
|
*/
|
|
|
|
|
public async saveNoun(noun: HNSWNoun): Promise<void> {
|
|
|
|
|
throw new Error('Historical storage is read-only. Cannot save noun.')
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
/**
|
|
|
|
|
* WRITE BLOCKED: Historical storage is read-only
|
|
|
|
|
*/
|
|
|
|
|
public async saveNounMetadata(id: string, metadata: NounMetadata): Promise<void> {
|
|
|
|
|
throw new Error('Historical storage is read-only. Cannot save noun metadata.')
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
/**
|
|
|
|
|
* WRITE BLOCKED: Historical storage is read-only
|
|
|
|
|
*/
|
|
|
|
|
public async deleteNoun(id: string): Promise<void> {
|
|
|
|
|
throw new Error('Historical storage is read-only. Cannot delete noun.')
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
/**
|
|
|
|
|
* WRITE BLOCKED: Historical storage is read-only
|
|
|
|
|
*/
|
|
|
|
|
public async deleteNounMetadata(id: string): Promise<void> {
|
|
|
|
|
throw new Error('Historical storage is read-only. Cannot delete noun metadata.')
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
/**
|
|
|
|
|
* WRITE BLOCKED: Historical storage is read-only
|
|
|
|
|
*/
|
|
|
|
|
public async saveVerb(verb: HNSWVerb): Promise<void> {
|
|
|
|
|
throw new Error('Historical storage is read-only. Cannot save verb.')
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
/**
|
|
|
|
|
* WRITE BLOCKED: Historical storage is read-only
|
|
|
|
|
*/
|
|
|
|
|
public async saveVerbMetadata(id: string, metadata: VerbMetadata): Promise<void> {
|
|
|
|
|
throw new Error('Historical storage is read-only. Cannot save verb metadata.')
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
/**
|
|
|
|
|
* WRITE BLOCKED: Historical storage is read-only
|
|
|
|
|
*/
|
|
|
|
|
public async deleteVerb(id: string): Promise<void> {
|
|
|
|
|
throw new Error('Historical storage is read-only. Cannot delete verb.')
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
/**
|
|
|
|
|
* WRITE BLOCKED: Historical storage is read-only
|
|
|
|
|
*/
|
|
|
|
|
public async deleteVerbMetadata(id: string): Promise<void> {
|
|
|
|
|
throw new Error('Historical storage is read-only. Cannot delete verb metadata.')
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
/**
|
|
|
|
|
* WRITE BLOCKED: Historical storage is read-only
|
|
|
|
|
*/
|
|
|
|
|
public async saveMetadata(id: string, metadata: any): Promise<void> {
|
|
|
|
|
throw new Error('Historical storage is read-only. Cannot save metadata.')
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
/**
|
|
|
|
|
* WRITE BLOCKED: Historical storage is read-only
|
|
|
|
|
*/
|
|
|
|
|
public async saveHNSWData(nounId: string, hnswData: {
|
|
|
|
|
level: number
|
|
|
|
|
connections: Record<string, string[]>
|
|
|
|
|
}): Promise<void> {
|
|
|
|
|
throw new Error('Historical storage is read-only. Cannot save HNSW data.')
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
/**
|
|
|
|
|
* WRITE BLOCKED: Historical storage is read-only
|
|
|
|
|
*/
|
|
|
|
|
public async saveHNSWSystem(systemData: {
|
|
|
|
|
entryPointId: string | null
|
|
|
|
|
maxLevel: number
|
|
|
|
|
}): Promise<void> {
|
|
|
|
|
throw new Error('Historical storage is read-only. Cannot save HNSW system data.')
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
// ============= Additional Abstract Methods =============
|
|
|
|
|
|
|
|
|
|
/**
|
|
|
|
|
* Get noun vector from historical state
|
|
|
|
|
*/
|
|
|
|
|
public async getNounVector(id: string): Promise<number[] | null> {
|
|
|
|
|
const noun = await this.getNoun(id)
|
|
|
|
|
return noun?.vector || null
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
/**
|
|
|
|
|
* Get HNSW data from historical state
|
|
|
|
|
*/
|
|
|
|
|
public async getHNSWData(nounId: string): Promise<{
|
|
|
|
|
level: number
|
|
|
|
|
connections: Record<string, string[]>
|
|
|
|
|
} | null> {
|
|
|
|
|
const path = `_system/hnsw/nodes/${nounId}.json`
|
|
|
|
|
return await this.readObjectFromPath(path)
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
/**
|
|
|
|
|
* Get HNSW system data from historical state
|
|
|
|
|
*/
|
|
|
|
|
public async getHNSWSystem(): Promise<{
|
|
|
|
|
entryPointId: string | null
|
|
|
|
|
maxLevel: number
|
|
|
|
|
} | null> {
|
|
|
|
|
return await this.readObjectFromPath('_system/hnsw/system.json')
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
/**
|
|
|
|
|
* Initialize counts (no-op for historical storage)
|
|
|
|
|
* Counts are loaded from historical state metadata
|
|
|
|
|
*/
|
|
|
|
|
protected async initializeCounts(): Promise<void> {
|
|
|
|
|
// No-op: Historical storage doesn't need to initialize counts
|
|
|
|
|
// They're read from commit state metadata
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
/**
|
|
|
|
|
* WRITE BLOCKED: Cannot persist counts to historical storage
|
|
|
|
|
*/
|
|
|
|
|
protected async persistCounts(): Promise<void> {
|
|
|
|
|
// No-op: Historical storage is read-only
|
|
|
|
|
}
|
|
|
|
|
}
|