feat(8.0): zero-config finalize + cut JS quantization (config.vector = recall + persistMode)
This commit is contained in:
parent
f4c5d9749f
commit
f8e0079d3f
12 changed files with 348 additions and 1739 deletions
22
RELEASES.md
22
RELEASES.md
|
|
@ -115,8 +115,9 @@ Hard renames, no aliases. Every pair below is verified against the 8.0 source.
|
||||||
| `HnswProvider` (from `@soulcraft/brainy/plugin`) | `VectorIndexProvider` |
|
| `HnswProvider` (from `@soulcraft/brainy/plugin`) | `VectorIndexProvider` |
|
||||||
| `HNSWIndex` (exported class) | `JsHnswVectorIndex` |
|
| `HNSWIndex` (exported class) | `JsHnswVectorIndex` |
|
||||||
| Provider keys `'hnsw'` and `'diskann'` | `'vector'` (the only key Brainy consults for the vector index) |
|
| Provider keys `'hnsw'` and `'diskann'` | `'vector'` (the only key Brainy consults for the vector index) |
|
||||||
| `config.hnsw = { quantization, vectorStorage }` | `config.vector = { recall?, quantization?, persistMode? }` (shape change — see below) |
|
| `config.hnsw = { quantization, vectorStorage }` | `config.vector = { recall?, persistMode? }` (shape change — see below) |
|
||||||
| `hnswPersistMode: 'immediate' \| 'deferred'` (top level) | `config.vector.persistMode` |
|
| `config.vector.quantization` (and 7.x `config.hnsw.quantization`) | **Removed.** The JS vector path is full-precision (exact float32 distances) only — quantization at scale is the native provider's job (e.g. DiskANN PQ). |
|
||||||
|
| `hnswPersistMode: 'immediate' \| 'deferred'` (top level) | `config.vector.persistMode` (auto-selected from the storage adapter when omitted: `'immediate'` on filesystem, `'deferred'` on memory) |
|
||||||
| `config.storage.type: 's3' \| 'gcs' \| 'r2' \| 'opfs'` | `'filesystem'` (or `'memory'` / `'auto'`) |
|
| `config.storage.type: 's3' \| 'gcs' \| 'r2' \| 'opfs'` | `'filesystem'` (or `'memory'` / `'auto'`) |
|
||||||
| `config.storage.branch` | Removed (no COW branches) |
|
| `config.storage.branch` | Removed (no COW branches) |
|
||||||
| `brain.stats().indexHealth.hnsw` | `brain.stats().indexHealth.vector` |
|
| `brain.stats().indexHealth.hnsw` | `brain.stats().indexHealth.vector` |
|
||||||
|
|
@ -155,22 +156,25 @@ new Brainy({
|
||||||
hnsw: { quantization: { enabled: true, bits: 8, rerankMultiplier: 3 } }
|
hnsw: { quantization: { enabled: true, bits: 8, rerankMultiplier: 3 } }
|
||||||
})
|
})
|
||||||
|
|
||||||
// 8.0
|
// 8.0 — config.vector is { recall?, persistMode? }
|
||||||
new Brainy({
|
new Brainy({
|
||||||
vector: {
|
vector: {
|
||||||
recall: 'balanced', // 'fast' | 'balanced' | 'accurate'
|
recall: 'balanced', // 'fast' | 'balanced' | 'accurate'
|
||||||
quantization: { enabled: true, bits: 8 }, // rerankMultiplier is fixed at 3
|
persistMode: 'deferred' // optional; auto-selected from storage when omitted
|
||||||
persistMode: 'deferred'
|
|
||||||
}
|
}
|
||||||
})
|
})
|
||||||
```
|
```
|
||||||
|
|
||||||
`config.vector.recall` replaces the algorithm-internal HNSW knobs: `'balanced'`
|
`config.vector.recall` replaces the algorithm-internal HNSW knobs: `'balanced'`
|
||||||
(the default) maps to exactly the 7.x default parameters, so an upgrade without
|
(the default) maps to exactly the 7.x default parameters, so an upgrade without
|
||||||
explicit knobs changes nothing about search behaviour. `config.hnsw.vectorStorage`
|
explicit knobs changes nothing about search behaviour. `config.vector.quantization`
|
||||||
is removed. The on-disk vector index file names are **unchanged** (e.g.
|
and `config.hnsw.vectorStorage` are removed: the JS vector path now computes exact
|
||||||
`hnsw-system.json`) — existing filesystem stores need no data migration for
|
float32 distances throughout (no rerank/approximate branch), which is what made
|
||||||
this rename; only the API surface moved.
|
its quantization a memory *increase* — it stored both the full and the quantized
|
||||||
|
vectors in RAM. Quantization at scale belongs to the native provider's own PQ.
|
||||||
|
The on-disk vector index file names are **unchanged** (e.g. `hnsw-system.json`) —
|
||||||
|
existing filesystem stores need no data migration for this rename; only the API
|
||||||
|
surface moved.
|
||||||
|
|
||||||
### Behavior changes
|
### Behavior changes
|
||||||
|
|
||||||
|
|
|
||||||
359
src/brainy.ts
359
src/brainy.ts
|
|
@ -116,7 +116,7 @@ import type { MigrationPreview, MigrationResult, MigrateOptions } from './migrat
|
||||||
import { AggregationIndex } from './aggregation/AggregationIndex.js'
|
import { AggregationIndex } from './aggregation/AggregationIndex.js'
|
||||||
import { AggregateMaterializer } from './aggregation/materializer.js'
|
import { AggregateMaterializer } from './aggregation/materializer.js'
|
||||||
import type { AggregateDefinition, AggregateQueryParams, AggregateResult } from './types/brainy.types.js'
|
import type { AggregateDefinition, AggregateQueryParams, AggregateResult } from './types/brainy.types.js'
|
||||||
import { resolveJsHnswConfig } from './utils/recallPreset.js'
|
import { resolveJsHnswConfig, DEFAULT_RECALL } from './utils/recallPreset.js'
|
||||||
import * as fs from 'node:fs'
|
import * as fs from 'node:fs'
|
||||||
import { Db, type DbHost, type HistoricalQueryHandle } from './db/db.js'
|
import { Db, type DbHost, type HistoricalQueryHandle } from './db/db.js'
|
||||||
import { GenerationStore } from './db/generationStore.js'
|
import { GenerationStore } from './db/generationStore.js'
|
||||||
|
|
@ -205,6 +205,8 @@ type ResolvedBrainyConfig = Required<
|
||||||
| 'vector'
|
| 'vector'
|
||||||
| 'plugins'
|
| 'plugins'
|
||||||
| 'integrations'
|
| 'integrations'
|
||||||
|
| 'history'
|
||||||
|
| 'eagerEmbeddings'
|
||||||
>
|
>
|
||||||
> &
|
> &
|
||||||
Pick<
|
Pick<
|
||||||
|
|
@ -215,6 +217,8 @@ type ResolvedBrainyConfig = Required<
|
||||||
| 'vector'
|
| 'vector'
|
||||||
| 'plugins'
|
| 'plugins'
|
||||||
| 'integrations'
|
| 'integrations'
|
||||||
|
| 'history'
|
||||||
|
| 'eagerEmbeddings'
|
||||||
>
|
>
|
||||||
|
|
||||||
/**
|
/**
|
||||||
|
|
@ -387,6 +391,18 @@ export class Brainy<T = any> implements BrainyInterface<T> {
|
||||||
private _readyResolve: (() => void) | null = null
|
private _readyResolve: (() => void) | null = null
|
||||||
private _readyReject: ((error: Error) => void) | null = null
|
private _readyReject: ((error: Error) => void) | null = null
|
||||||
|
|
||||||
|
// In-flight init() promise. Guards against concurrent initialization when
|
||||||
|
// multiple operations race to lazily initialize the same instance — all
|
||||||
|
// racers await the single in-flight run. Reset on failure so init() can
|
||||||
|
// be retried after the underlying cause (e.g. a held writer lock) clears.
|
||||||
|
private initPromise: Promise<void> | null = null
|
||||||
|
|
||||||
|
// Set once close() has torn the instance down. close() is terminal: a closed
|
||||||
|
// instance must NOT silently lazy-re-initialize on the next operation (that
|
||||||
|
// would resurrect released indexes and timers). The lazy-init convenience
|
||||||
|
// applies only to instances that were never closed.
|
||||||
|
private closed = false
|
||||||
|
|
||||||
// Lazy rebuild state (Production-scale lazy loading)
|
// Lazy rebuild state (Production-scale lazy loading)
|
||||||
// Prevents race conditions when multiple queries trigger rebuild simultaneously
|
// Prevents race conditions when multiple queries trigger rebuild simultaneously
|
||||||
private lazyRebuildInProgress = false
|
private lazyRebuildInProgress = false
|
||||||
|
|
@ -542,13 +558,42 @@ export class Brainy<T = any> implements BrainyInterface<T> {
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Initialize Brainy - MUST be called before use
|
* Initialize Brainy.
|
||||||
|
*
|
||||||
|
* **Calling this explicitly is optional.** Every public operation lazily
|
||||||
|
* initializes the instance on first use (`new Brainy()` followed directly
|
||||||
|
* by `add()` / `find()` / `transact()` just works). Call `init()` yourself
|
||||||
|
* when you want to control *when* the startup cost is paid (e.g. during
|
||||||
|
* server boot rather than on the first request) or to pass init-time
|
||||||
|
* configuration overrides.
|
||||||
|
*
|
||||||
|
* Idempotent and concurrency-safe: repeat calls after success return
|
||||||
|
* immediately; calls racing an in-flight initialization await that single
|
||||||
|
* run (overrides are only applied by the run that starts initialization).
|
||||||
|
*
|
||||||
* @param overrides Optional configuration overrides for init
|
* @param overrides Optional configuration overrides for init
|
||||||
*/
|
*/
|
||||||
async init(overrides?: Partial<BrainyConfig & { dimensions?: number }>): Promise<void> {
|
async init(overrides?: Partial<BrainyConfig & { dimensions?: number }>): Promise<void> {
|
||||||
if (this.initialized) {
|
if (this.initialized) {
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
if (this.initPromise) {
|
||||||
|
return this.initPromise
|
||||||
|
}
|
||||||
|
const run = this.performInit(overrides)
|
||||||
|
this.initPromise = run
|
||||||
|
try {
|
||||||
|
await run
|
||||||
|
} finally {
|
||||||
|
this.initPromise = null
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* The single initialization run guarded by `init()`. Never call directly —
|
||||||
|
* always go through `init()` (or any public method, which lazily inits).
|
||||||
|
*/
|
||||||
|
private async performInit(overrides?: Partial<BrainyConfig & { dimensions?: number }>): Promise<void> {
|
||||||
|
|
||||||
// Apply any init-time configuration overrides
|
// Apply any init-time configuration overrides
|
||||||
if (overrides) {
|
if (overrides) {
|
||||||
|
|
@ -566,7 +611,6 @@ export class Brainy<T = any> implements BrainyInterface<T> {
|
||||||
...this.config,
|
...this.config,
|
||||||
...configOverrides,
|
...configOverrides,
|
||||||
storage: mergedStorage,
|
storage: mergedStorage,
|
||||||
index: { ...this.config.index, ...configOverrides.index },
|
|
||||||
verbose: configOverrides.verbose ?? this.config.verbose,
|
verbose: configOverrides.verbose ?? this.config.verbose,
|
||||||
silent: configOverrides.silent ?? this.config.silent
|
silent: configOverrides.silent ?? this.config.silent
|
||||||
}
|
}
|
||||||
|
|
@ -702,32 +746,6 @@ export class Brainy<T = any> implements BrainyInterface<T> {
|
||||||
setMsgpackImplementation(msgpackProvider)
|
setMsgpackImplementation(msgpackProvider)
|
||||||
}
|
}
|
||||||
|
|
||||||
// Provider: native SQ8 approximate distance (e.g. cortex's Rust SIMD) — swaps
|
|
||||||
// the JS quantized-distance used in HNSW SQ8 reranking. Signature is
|
|
||||||
// byte-compatible with the JS distanceSQ8; falls back to JS when absent.
|
|
||||||
const sq8DistanceProvider = this.pluginRegistry.getProvider<
|
|
||||||
(a: Uint8Array, aMin: number, aMax: number, b: Uint8Array, bMin: number, bMax: number) => number
|
|
||||||
>('distance:sq8')
|
|
||||||
if (sq8DistanceProvider) {
|
|
||||||
const { setSQ8DistanceImplementation } = await import('./utils/vectorQuantization.js')
|
|
||||||
setSQ8DistanceImplementation(sq8DistanceProvider)
|
|
||||||
}
|
|
||||||
|
|
||||||
// Provider: native SQ4 approximate distance (cortex's Rust). Same swap
|
|
||||||
// pattern as SQ8; signature is byte-compatible with the JS distanceSQ4
|
|
||||||
// (4-bit quantization range, packed nibbles). Used in HNSW SQ4 reranking
|
|
||||||
// when config.hnsw.quantization.bits === 4. Falls back to JS when absent.
|
|
||||||
const sq4DistanceProvider = this.pluginRegistry.getProvider<
|
|
||||||
(
|
|
||||||
a: Uint8Array, aMin: number, aMax: number, aDim: number,
|
|
||||||
b: Uint8Array, bMin: number, bMax: number, bDim: number
|
|
||||||
) => number
|
|
||||||
>('distance:sq4')
|
|
||||||
if (sq4DistanceProvider) {
|
|
||||||
const { setSQ4DistanceImplementation } = await import('./utils/vectorQuantization.js')
|
|
||||||
setSQ4DistanceImplementation(sq4DistanceProvider)
|
|
||||||
}
|
|
||||||
|
|
||||||
// Provider: sort:topK (e.g. cortex's native partial-sort / heap-select) — swaps the
|
// Provider: sort:topK (e.g. cortex's native partial-sort / heap-select) — swaps the
|
||||||
// JS result-ranking used by find() to pick the top `offset + limit` rows. The provider
|
// JS result-ranking used by find() to pick the top `offset + limit` rows. The provider
|
||||||
// returns indices into a scores array, ordered descending with stable ties, identical
|
// returns indices into a scores array, ordered descending with stable ties, identical
|
||||||
|
|
@ -831,11 +849,6 @@ export class Brainy<T = any> implements BrainyInterface<T> {
|
||||||
// Connect distributed components to storage
|
// Connect distributed components to storage
|
||||||
await this.connectDistributedStorage()
|
await this.connectDistributedStorage()
|
||||||
|
|
||||||
// Warm up if configured
|
|
||||||
if (this.config.warmup) {
|
|
||||||
await this.warmup()
|
|
||||||
}
|
|
||||||
|
|
||||||
// Register shutdown hooks for graceful count flushing (once globally)
|
// Register shutdown hooks for graceful count flushing (once globally)
|
||||||
if (!Brainy.shutdownHooksRegisteredGlobally) {
|
if (!Brainy.shutdownHooksRegisteredGlobally) {
|
||||||
this.registerShutdownHooks()
|
this.registerShutdownHooks()
|
||||||
|
|
@ -875,12 +888,34 @@ export class Brainy<T = any> implements BrainyInterface<T> {
|
||||||
await this._vfs.init()
|
await this._vfs.init()
|
||||||
this._vfsInitialized = true // Mark VFS as fully initialized
|
this._vfsInitialized = true // Mark VFS as fully initialized
|
||||||
|
|
||||||
// Eager embedding initialization for cloud deployments
|
// Eager embedding initialization.
|
||||||
// When eagerEmbeddings is true, initialize the WASM embedding engine now
|
//
|
||||||
// instead of lazily on first embed() call. This moves the 90-140 second
|
// Adaptive default (8.0): the WASM embedding engine eagerly initializes
|
||||||
// WASM compilation to container startup rather than first request.
|
// during init() WHENEVER it is the active embedder — i.e. no native
|
||||||
// Recommended for: Cloud Run, Lambda, Fargate, Kubernetes
|
// 'embeddings' provider has taken over — and the instance is a writer
|
||||||
if (this.config.eagerEmbeddings && !this.pluginRegistry.hasProvider('embeddings')) {
|
// (not reader-mode) outside of unit tests. The WASM module (≈93MB with
|
||||||
|
// the embedded model) takes 90-140s to compile on throttled CPUs; paying
|
||||||
|
// that during boot rather than on the first embed()-driven call is the
|
||||||
|
// right default for the overwhelmingly common single-process server.
|
||||||
|
//
|
||||||
|
// Skipped automatically when:
|
||||||
|
// - a native 'embeddings' provider is registered (it owns embeddings;
|
||||||
|
// the WASM engine is dead weight),
|
||||||
|
// - reader-mode (readers don't embed — they query existing vectors),
|
||||||
|
// - unit-test mode (tests must stay fast and use the mock embedder).
|
||||||
|
//
|
||||||
|
// `eagerEmbeddings: false` is the explicit override to force lazy init
|
||||||
|
// (first-embed) even when this instance is the active embedder.
|
||||||
|
const isUnitTestMode =
|
||||||
|
process.env.BRAINY_UNIT_TEST === 'true' ||
|
||||||
|
(globalThis as { __BRAINY_UNIT_TEST__?: boolean }).__BRAINY_UNIT_TEST__ === true
|
||||||
|
const eager = this.config.eagerEmbeddings ?? true
|
||||||
|
if (
|
||||||
|
eager &&
|
||||||
|
!this.pluginRegistry.hasProvider('embeddings') &&
|
||||||
|
this.config.mode !== 'reader' &&
|
||||||
|
!isUnitTestMode
|
||||||
|
) {
|
||||||
console.log('Eager embedding initialization enabled...')
|
console.log('Eager embedding initialization enabled...')
|
||||||
await embeddingManager.init()
|
await embeddingManager.init()
|
||||||
console.log('Embedding engine ready')
|
console.log('Embedding engine ready')
|
||||||
|
|
@ -1026,11 +1061,24 @@ export class Brainy<T = any> implements BrainyInterface<T> {
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Ensure Brainy is initialized
|
* Ensure Brainy is initialized, lazily initializing on first use.
|
||||||
|
*
|
||||||
|
* Called at the top of every public operation, so `new Brainy()` followed
|
||||||
|
* directly by `add()` / `find()` / any other call just works — forgetting
|
||||||
|
* `init()` is not an error. Concurrent first operations share a single
|
||||||
|
* initialization run (see `init()`); if initialization fails, the error
|
||||||
|
* propagates to the operation that triggered it.
|
||||||
|
*
|
||||||
|
* `close()` is terminal: a closed instance throws here instead of silently
|
||||||
|
* re-initializing — using a closed Brainy is a consumer bug, not a lazy-init
|
||||||
|
* opportunity.
|
||||||
*/
|
*/
|
||||||
private async ensureInitialized(): Promise<void> {
|
private async ensureInitialized(): Promise<void> {
|
||||||
|
if (this.closed) {
|
||||||
|
throw new Error('Brainy instance is not initialized: it was closed via close(). Create a new instance.')
|
||||||
|
}
|
||||||
if (!this.initialized) {
|
if (!this.initialized) {
|
||||||
throw new Error('Brainy not initialized. Call init() first.')
|
await this.init()
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
@ -5143,6 +5191,68 @@ export class Brainy<T = any> implements BrainyInterface<T> {
|
||||||
return this.generationStore.compact(options)
|
return this.generationStore.compact(options)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* @description Resolve the effective generational-history retention policy
|
||||||
|
* from `config.history`, filling per-field defaults so a partial object
|
||||||
|
* inherits the rest. This is the single read site for the policy referenced
|
||||||
|
* by {@link autoCompactHistory}.
|
||||||
|
*
|
||||||
|
* Defaults (zero-config): keep the 100 most recent committed generations
|
||||||
|
* AND everything committed within the last 7 days; auto-compact on
|
||||||
|
* `flush()` / `close()`. A generation is reclaimed only when it is older
|
||||||
|
* than BOTH bounds and no live `Db` pin protects it.
|
||||||
|
*
|
||||||
|
* @returns The resolved policy: retention bounds plus the `autoCompact` flag.
|
||||||
|
*/
|
||||||
|
private resolveHistoryPolicy(): {
|
||||||
|
retainGenerations: number
|
||||||
|
retainMs: number
|
||||||
|
autoCompact: boolean
|
||||||
|
} {
|
||||||
|
const history = this.config.history
|
||||||
|
return {
|
||||||
|
retainGenerations: history?.retainGenerations ?? 100,
|
||||||
|
retainMs: history?.retainMs ?? 604_800_000, // 7 days
|
||||||
|
autoCompact: history?.autoCompact ?? true
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* @description Run history compaction with the resolved retention policy if
|
||||||
|
* `config.history.autoCompact` is on (the default). Invoked from `flush()`
|
||||||
|
* and `close()` so generational record-sets cannot accumulate unbounded
|
||||||
|
* across a long-lived writer's lifetime.
|
||||||
|
*
|
||||||
|
* Read-only instances and an explicit `autoCompact: false` skip silently.
|
||||||
|
* Pinned generations (live `Db` values) are never reclaimed — the safety
|
||||||
|
* invariant lives in {@link GenerationStore.compact}. Failures are logged
|
||||||
|
* and swallowed: compaction is housekeeping and must never fail a flush or
|
||||||
|
* a clean shutdown.
|
||||||
|
*/
|
||||||
|
private async autoCompactHistory(): Promise<void> {
|
||||||
|
// Nothing to compact on a read-only instance or before init wired up the
|
||||||
|
// generation store (close() can run defensively on an uninitialized brain).
|
||||||
|
if (this.isReadOnly || !this.generationStore) {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
const policy = this.resolveHistoryPolicy()
|
||||||
|
if (!policy.autoCompact) {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
try {
|
||||||
|
await this.generationStore.compact({
|
||||||
|
retainGenerations: policy.retainGenerations,
|
||||||
|
retainMs: policy.retainMs
|
||||||
|
})
|
||||||
|
} catch (error) {
|
||||||
|
console.warn(
|
||||||
|
`Auto-compaction of generational history failed (non-fatal): ${
|
||||||
|
error instanceof Error ? error.message : String(error)
|
||||||
|
}`
|
||||||
|
)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* @description Replace this store's ENTIRE state from a snapshot directory
|
* @description Replace this store's ENTIRE state from a snapshot directory
|
||||||
* previously produced by `db.persist(path)`. Destructive: current
|
* previously produced by `db.persist(path)`. Destructive: current
|
||||||
|
|
@ -6858,6 +6968,11 @@ export class Brainy<T = any> implements BrainyInterface<T> {
|
||||||
this.generationStore.persistCounterNow()
|
this.generationStore.persistCounterNow()
|
||||||
])
|
])
|
||||||
|
|
||||||
|
// 6. Auto-compact generational history per config.history (default on).
|
||||||
|
// Runs after the flush so durable state is in place; respects live
|
||||||
|
// Db pins and an explicit autoCompact: false.
|
||||||
|
await this.autoCompactHistory()
|
||||||
|
|
||||||
const elapsed = Date.now() - startTime
|
const elapsed = Date.now() - startTime
|
||||||
|
|
||||||
console.log(`All indexes flushed to disk in ${elapsed}ms`)
|
console.log(`All indexes flushed to disk in ${elapsed}ms`)
|
||||||
|
|
@ -10167,14 +10282,6 @@ export class Brainy<T = any> implements BrainyInterface<T> {
|
||||||
return this.embedder(textToEmbed)
|
return this.embedder(textToEmbed)
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
|
||||||
* Warm up the system
|
|
||||||
*/
|
|
||||||
private async warmup(): Promise<void> {
|
|
||||||
// Warm up embedder
|
|
||||||
await this.embed('warmup')
|
|
||||||
}
|
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Explicitly warm up the embedding engine
|
* Explicitly warm up the embedding engine
|
||||||
*
|
*
|
||||||
|
|
@ -10300,26 +10407,22 @@ export class Brainy<T = any> implements BrainyInterface<T> {
|
||||||
/**
|
/**
|
||||||
* Setup index — single unified vector graph.
|
* Setup index — single unified vector graph.
|
||||||
*
|
*
|
||||||
* Brainy 8.0 ships filesystem-only storage (no cloud adapters), so persistence
|
* Persistence mode is auto-selected from the storage adapter (see
|
||||||
* mode defaults to `'immediate'` everywhere. Operators who want bulk-ingest
|
* {@link resolveHNSWPersistMode}): `'immediate'` on filesystem storage,
|
||||||
* speed can set `config.vector.persistMode = 'deferred'`.
|
* `'deferred'` on memory storage. Operators can override with an explicit
|
||||||
|
* `config.vector.persistMode` (e.g. `'deferred'` for bulk-ingest speed on
|
||||||
|
* filesystem storage).
|
||||||
*/
|
*/
|
||||||
private setupIndex(): JsHnswVectorIndex {
|
private setupIndex(): JsHnswVectorIndex {
|
||||||
// 8.0 config surface: config.vector.{recall, quantization, persistMode}.
|
// 8.0 config surface: config.vector.{recall, persistMode}.
|
||||||
// The recall preset translates to HNSW knobs (M / efConstruction / efSearch)
|
// The recall preset translates to HNSW knobs (M / efConstruction / efSearch)
|
||||||
// via resolveJsHnswConfig. No algorithm-internal knobs are exposed at the
|
// via resolveJsHnswConfig. No algorithm-internal knobs are exposed at the
|
||||||
// public surface.
|
// public surface. The JS index computes exact float32 distances throughout —
|
||||||
|
// there is no quantization path.
|
||||||
const recallKnobs = resolveJsHnswConfig(this.config.vector)
|
const recallKnobs = resolveJsHnswConfig(this.config.vector)
|
||||||
const vectorCfg = this.config.vector
|
|
||||||
const indexConfig = {
|
const indexConfig = {
|
||||||
...this.config.index,
|
|
||||||
...recallKnobs,
|
...recallKnobs,
|
||||||
distanceFunction: this.distance,
|
distanceFunction: this.distance
|
||||||
quantization: vectorCfg?.quantization ? {
|
|
||||||
enabled: vectorCfg.quantization.enabled ?? false,
|
|
||||||
bits: vectorCfg.quantization.bits ?? 8,
|
|
||||||
rerankMultiplier: 3
|
|
||||||
} : undefined
|
|
||||||
}
|
}
|
||||||
|
|
||||||
const persistMode = this.resolveHNSWPersistMode()
|
const persistMode = this.resolveHNSWPersistMode()
|
||||||
|
|
@ -10346,8 +10449,18 @@ export class Brainy<T = any> implements BrainyInterface<T> {
|
||||||
const persistMode = this.resolveHNSWPersistMode()
|
const persistMode = this.resolveHNSWPersistMode()
|
||||||
const vectorFactory = this.pluginRegistry.getProvider<(config: any, distance: DistanceFunction, options: any) => any>('vector')
|
const vectorFactory = this.pluginRegistry.getProvider<(config: any, distance: DistanceFunction, options: any) => any>('vector')
|
||||||
if (vectorFactory) {
|
if (vectorFactory) {
|
||||||
|
// Native providers receive the algorithm-neutral surface: the resolved
|
||||||
|
// recall preset name (per the 8.0 provider contract they translate it
|
||||||
|
// to their own internal knobs), plus the JS HNSW knob tuple for
|
||||||
|
// providers that mirror the open-core implementation. Quantization at
|
||||||
|
// scale is the native provider's own concern (e.g. DiskANN PQ) — Brainy
|
||||||
|
// exposes no quantization knob.
|
||||||
return vectorFactory(
|
return vectorFactory(
|
||||||
{ ...this.config.index, distanceFunction: this.distance },
|
{
|
||||||
|
...resolveJsHnswConfig(this.config.vector),
|
||||||
|
recall: this.config.vector?.recall ?? DEFAULT_RECALL,
|
||||||
|
distanceFunction: this.distance
|
||||||
|
},
|
||||||
this.distance,
|
this.distance,
|
||||||
{ storage: this.storage, persistMode }
|
{ storage: this.storage, persistMode }
|
||||||
)
|
)
|
||||||
|
|
@ -10356,15 +10469,23 @@ export class Brainy<T = any> implements BrainyInterface<T> {
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Resolve HNSW persistence mode.
|
* Resolve the vector persistence mode from the storage adapter.
|
||||||
* Extracted so both setupIndex() and the HNSW plugin factory path can use it.
|
|
||||||
*
|
*
|
||||||
* Brainy 8.0 ships filesystem-only; the cloud-storage smart-default
|
* Auto-selected when `config.vector.persistMode` is omitted:
|
||||||
* (deferred for cloud, immediate for local) collapses to "always
|
* - **filesystem** (and any persistent adapter): `'immediate'` — per-add
|
||||||
* immediate unless the user overrides via config.vector.persistMode".
|
* durability is the point of a persistent backend.
|
||||||
|
* - **memory**: `'deferred'` — nothing survives the process anyway, so
|
||||||
|
* per-add persistence writes are pure overhead; deferred batches them
|
||||||
|
* into `flush()` / `close()`.
|
||||||
|
*
|
||||||
|
* An explicit `config.vector.persistMode` always wins (the operator
|
||||||
|
* escape hatch for bulk-ingest pipelines on filesystem storage).
|
||||||
*/
|
*/
|
||||||
private resolveHNSWPersistMode(): 'immediate' | 'deferred' {
|
private resolveHNSWPersistMode(): 'immediate' | 'deferred' {
|
||||||
return this.config.vector?.persistMode ?? 'immediate'
|
if (this.config.vector?.persistMode) {
|
||||||
|
return this.config.vector.persistMode
|
||||||
|
}
|
||||||
|
return this.getStorageType() === 'memory' ? 'deferred' : 'immediate'
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
|
|
@ -10390,59 +10511,40 @@ export class Brainy<T = any> implements BrainyInterface<T> {
|
||||||
)
|
)
|
||||||
}
|
}
|
||||||
|
|
||||||
// Validate numeric configurations
|
// Distributed mode is explicit opt-in only (config or BRAINY_DISTRIBUTED
|
||||||
if (config?.index?.m && (config.index.m < 1 || config.index.m > 128)) {
|
// env var) — never inferred from deployment heuristics.
|
||||||
throw new Error(`Invalid index m parameter: ${config.index.m}. Must be between 1 and 128`)
|
const distributedConfig = this.resolveDistributedConfig(config?.distributed)
|
||||||
}
|
|
||||||
|
|
||||||
if (config?.index?.efConstruction && (config.index.efConstruction < 1 || config.index.efConstruction > 1000)) {
|
|
||||||
throw new Error(`Invalid index efConstruction: ${config.index.efConstruction}. Must be between 1 and 1000`)
|
|
||||||
}
|
|
||||||
|
|
||||||
if (config?.index?.efSearch && (config.index.efSearch < 1 || config.index.efSearch > 1000)) {
|
|
||||||
throw new Error(`Invalid index efSearch: ${config.index.efSearch}. Must be between 1 and 1000`)
|
|
||||||
}
|
|
||||||
|
|
||||||
// Auto-detect distributed mode based on environment and configuration
|
|
||||||
const distributedConfig = this.autoDetectDistributed(config?.distributed)
|
|
||||||
|
|
||||||
return {
|
return {
|
||||||
storage: config?.storage || { type: 'auto' },
|
storage: config?.storage || { type: 'auto' },
|
||||||
index: config?.index || {},
|
|
||||||
cache: config?.cache ?? true,
|
|
||||||
distributed: distributedConfig,
|
distributed: distributedConfig,
|
||||||
warmup: config?.warmup ?? false,
|
|
||||||
realtime: config?.realtime ?? false,
|
|
||||||
multiTenancy: config?.multiTenancy ?? false,
|
|
||||||
telemetry: config?.telemetry ?? false,
|
|
||||||
verbose: config?.verbose ?? false,
|
verbose: config?.verbose ?? false,
|
||||||
silent: config?.silent ?? false,
|
silent: config?.silent ?? false,
|
||||||
// New performance options with smart defaults
|
// false = auto-decide based on dataset size (inline vs lazy rebuild)
|
||||||
disableAutoRebuild: config?.disableAutoRebuild ?? false, // false = auto-decide based on size
|
disableAutoRebuild: config?.disableAutoRebuild ?? false,
|
||||||
disableMetrics: config?.disableMetrics ?? false,
|
// Memory management escape hatches over the RAM-derived auto limits
|
||||||
disableAutoOptimize: config?.disableAutoOptimize ?? false,
|
|
||||||
batchWrites: config?.batchWrites ?? true,
|
|
||||||
maxConcurrentOperations: config?.maxConcurrentOperations ?? 10,
|
|
||||||
// Memory management options
|
|
||||||
maxQueryLimit: config?.maxQueryLimit ?? undefined,
|
maxQueryLimit: config?.maxQueryLimit ?? undefined,
|
||||||
reservedQueryMemory: config?.reservedQueryMemory ?? undefined,
|
reservedQueryMemory: config?.reservedQueryMemory ?? undefined,
|
||||||
// Vector index configuration (8.0) — algorithm-neutral surface with
|
// Vector index configuration (8.0) — algorithm-neutral surface with
|
||||||
// `recall` preset + `persistMode` (folded in from 7.x's hnswPersistMode).
|
// `recall` preset + `persistMode` (folded in from 7.x's hnswPersistMode).
|
||||||
// See BRAINY-8.0-RENAME-COORDINATION § A.2 + § G.2.
|
// See BRAINY-8.0-RENAME-COORDINATION § A.2 + § G.2.
|
||||||
vector: config?.vector ?? undefined,
|
vector: config?.vector ?? undefined,
|
||||||
// Embedding initialization - false = lazy init on first embed()
|
// Generational-history retention — defaults applied at the read site
|
||||||
eagerEmbeddings: config?.eagerEmbeddings ?? false,
|
// (resolveHistoryPolicy) so a partial object inherits per-field defaults.
|
||||||
|
history: config?.history ?? undefined,
|
||||||
|
// Embedding initialization — left undefined when omitted so the adaptive
|
||||||
|
// default resolves at the init() read site (eager when the WASM embedder
|
||||||
|
// is the active one on a non-reader writer outside unit tests). Explicit
|
||||||
|
// true/false always wins. See the eagerEmbeddings JSDoc in brainy.types.ts.
|
||||||
|
eagerEmbeddings: config?.eagerEmbeddings ?? undefined,
|
||||||
// Plugin configuration - undefined = auto-detect
|
// Plugin configuration - undefined = auto-detect
|
||||||
plugins: config?.plugins ?? undefined,
|
plugins: config?.plugins ?? undefined,
|
||||||
// Integration Hub - undefined/false = disabled
|
// Integration Hub - undefined/false = disabled
|
||||||
integrations: config?.integrations ?? undefined,
|
integrations: config?.integrations ?? undefined,
|
||||||
// Migration — disabled by default, opt-in for automatic migration
|
// Migration — auto-run by default. Pending data migrations apply inline
|
||||||
autoMigrate: config?.autoMigrate ?? false,
|
// during init() for small datasets (<10K entities); larger datasets
|
||||||
// Subtype pairing enforcement (7.30.0) — opt-in.
|
// defer with a notice so the operator can schedule brain.migrate().
|
||||||
// false: only per-type rules registered via brain.requireSubtype() apply.
|
autoMigrate: config?.autoMigrate ?? true,
|
||||||
// true: every public write path requires subtype on every type.
|
|
||||||
// { except: [...] }: strict, but listed types may omit subtype.
|
|
||||||
// Becomes the default in 8.0.0.
|
|
||||||
// Subtype required-by-default (8.0 — per BRAINY-8.0-SUBTYPE-CONTRACT § C-1).
|
// Subtype required-by-default (8.0 — per BRAINY-8.0-SUBTYPE-CONTRACT § C-1).
|
||||||
// Every write path now requires `subtype` on every type unless the consumer
|
// Every write path now requires `subtype` on every type unless the consumer
|
||||||
// opts out explicitly. Opt-out: `requireSubtype: false` (last-resort migration
|
// opts out explicitly. Opt-out: `requireSubtype: false` (last-resort migration
|
||||||
|
|
@ -10926,6 +11028,11 @@ export class Brainy<T = any> implements BrainyInterface<T> {
|
||||||
* This ensures deferred persistence mode data is saved
|
* This ensures deferred persistence mode data is saved
|
||||||
*/
|
*/
|
||||||
async close(): Promise<void> {
|
async close(): Promise<void> {
|
||||||
|
// Phase 0: Auto-compact generational history per config.history (default
|
||||||
|
// on) BEFORE the generation store closes below. Respects live Db pins and
|
||||||
|
// an explicit autoCompact: false; no-op on read-only instances.
|
||||||
|
await this.autoCompactHistory()
|
||||||
|
|
||||||
// Phase 1: Flush ALL components in parallel to persist buffered data
|
// Phase 1: Flush ALL components in parallel to persist buffered data
|
||||||
// This is critical when cortex native providers buffer data in Rust memory
|
// This is critical when cortex native providers buffer data in Rust memory
|
||||||
await Promise.all([
|
await Promise.all([
|
||||||
|
|
@ -11033,26 +11140,34 @@ export class Brainy<T = any> implements BrainyInterface<T> {
|
||||||
}
|
}
|
||||||
|
|
||||||
this.initialized = false
|
this.initialized = false
|
||||||
|
// close() is terminal: block lazy re-initialization on any subsequent
|
||||||
|
// operation (ensureInitialized() throws once this is set).
|
||||||
|
this.closed = true
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Intelligently auto-detect distributed configuration
|
* Intelligently auto-detect distributed configuration
|
||||||
* Zero-config: Automatically determines best distributed settings
|
* Zero-config: Automatically determines best distributed settings
|
||||||
*/
|
*/
|
||||||
private autoDetectDistributed(config?: BrainyConfig['distributed']): BrainyConfig['distributed'] {
|
/**
|
||||||
// If explicitly disabled, respect that
|
* Resolve distributed-cluster configuration. **Explicit opt-in only**:
|
||||||
if (config?.enabled === false) {
|
* either `config.distributed.enabled === true` or the
|
||||||
return config
|
* `BRAINY_DISTRIBUTED=true` environment variable. Once opted in, the
|
||||||
}
|
* remaining fields default sensibly (64 shards, 3 replicas, raft, http,
|
||||||
|
* hostname-derived nodeId).
|
||||||
// Auto-detect based on environment variables (common in production)
|
*
|
||||||
const envEnabled = process.env.BRAINY_DISTRIBUTED === 'true' ||
|
* Earlier versions auto-enabled cluster mode from deployment heuristics
|
||||||
process.env.NODE_ENV === 'production' ||
|
* (`NODE_ENV=production`, a Kubernetes service host, `CLUSTER_SIZE`).
|
||||||
process.env.CLUSTER_SIZE ||
|
* That violated zero-config safety: a single-process production
|
||||||
process.env.KUBERNETES_SERVICE_HOST // Running in K8s
|
* deployment — the overwhelmingly common case — would silently start a
|
||||||
|
* coordinator, bind a port, and run raft against itself. Deployment
|
||||||
// If not explicitly configured but environment suggests distributed
|
* environment does not imply a multi-node Brainy cluster, so the
|
||||||
if (!config && envEnabled) {
|
* heuristics are gone.
|
||||||
|
*/
|
||||||
|
private resolveDistributedConfig(config?: BrainyConfig['distributed']): BrainyConfig['distributed'] {
|
||||||
|
// Env-var opt-in for deployments that can't touch code (the ONLY
|
||||||
|
// environment signal honoured — it names brainy explicitly).
|
||||||
|
if (!config && process.env.BRAINY_DISTRIBUTED === 'true') {
|
||||||
return {
|
return {
|
||||||
enabled: true,
|
enabled: true,
|
||||||
nodeId: process.env.HOSTNAME || process.env.NODE_ID || `node-${Date.now()}`,
|
nodeId: process.env.HOSTNAME || process.env.NODE_ID || `node-${Date.now()}`,
|
||||||
|
|
@ -11065,7 +11180,7 @@ export class Brainy<T = any> implements BrainyInterface<T> {
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// Merge with provided config, applying intelligent defaults
|
// Explicit config: fill the per-field defaults.
|
||||||
return config ? {
|
return config ? {
|
||||||
...config,
|
...config,
|
||||||
nodeId: config.nodeId || process.env.HOSTNAME || `node-${Date.now()}`,
|
nodeId: config.nodeId || process.env.HOSTNAME || `node-${Date.now()}`,
|
||||||
|
|
|
||||||
|
|
@ -353,10 +353,5 @@ export function presetToBrainyConfig(preset: PresetConfig): any {
|
||||||
if (preset.lazyLoadInReadOnlyMode) config.lazyLoadInReadOnlyMode = true
|
if (preset.lazyLoadInReadOnlyMode) config.lazyLoadInReadOnlyMode = true
|
||||||
}
|
}
|
||||||
|
|
||||||
// Handle cache settings
|
|
||||||
if (preset.cache) {
|
|
||||||
config.cache = preset.cache
|
|
||||||
}
|
|
||||||
|
|
||||||
return config
|
return config
|
||||||
}
|
}
|
||||||
|
|
@ -96,10 +96,6 @@ export interface HNSWNoun {
|
||||||
vector: Vector
|
vector: Vector
|
||||||
connections: Map<number, Set<string>> // level -> set of connected noun ids
|
connections: Map<number, Set<string>> // level -> set of connected noun ids
|
||||||
level: number // The highest layer this noun appears in
|
level: number // The highest layer this noun appears in
|
||||||
// SQ8 quantized vector for approximate distance computation (B1 optimization)
|
|
||||||
quantizedVector?: Uint8Array // 4x smaller than float32 vector
|
|
||||||
codebookMin?: number // quantization codebook minimum
|
|
||||||
codebookMax?: number // quantization codebook maximum
|
|
||||||
// ✅ NO metadata field - stored separately for optimization
|
// ✅ NO metadata field - stored separately for optimization
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
@ -475,12 +471,6 @@ export interface HNSWConfig {
|
||||||
ml: number // Maximum level
|
ml: number // Maximum level
|
||||||
useDiskBasedIndex?: boolean // Whether to use disk-based index
|
useDiskBasedIndex?: boolean // Whether to use disk-based index
|
||||||
maxConcurrentNeighborWrites?: number // Maximum concurrent neighbor updates during insert. Default: unlimited (full concurrency)
|
maxConcurrentNeighborWrites?: number // Maximum concurrent neighbor updates during insert. Default: unlimited (full concurrency)
|
||||||
// SQ8 vector quantization (4x memory reduction, ~0.4% accuracy loss)
|
|
||||||
quantization?: {
|
|
||||||
enabled: boolean // default: false — preserves current behavior exactly
|
|
||||||
bits?: 8 | 4 // default: 8 (SQ8). SQ4 has a pure-JS implementation; cortex's distance:sq4 SIMD provider accelerates it.
|
|
||||||
rerankMultiplier?: number // default: 3 — over-retrieve 3x, rerank with float32
|
|
||||||
}
|
|
||||||
// Vector storage mode
|
// Vector storage mode
|
||||||
vectorStorage?: 'memory' | 'lazy' // default: 'memory' — 'lazy' evicts vectors after insert
|
vectorStorage?: 'memory' | 'lazy' // default: 'memory' — 'lazy' evicts vectors after insert
|
||||||
}
|
}
|
||||||
|
|
|
||||||
|
|
@ -14,8 +14,6 @@ import { euclideanDistance, calculateDistancesBatch } from '../utils/index.js'
|
||||||
import type { BaseStorage } from '../storage/baseStorage.js'
|
import type { BaseStorage } from '../storage/baseStorage.js'
|
||||||
import { getGlobalCache, UnifiedCache } from '../utils/unifiedCache.js'
|
import { getGlobalCache, UnifiedCache } from '../utils/unifiedCache.js'
|
||||||
import { prodLog } from '../utils/logger.js'
|
import { prodLog } from '../utils/logger.js'
|
||||||
import { quantizeSQ8, distanceSQ8 } from '../utils/vectorQuantization.js'
|
|
||||||
import type { SQ8QuantizedVector } from '../utils/vectorQuantization.js'
|
|
||||||
import type { VectorIndexProvider } from '../plugin.js'
|
import type { VectorIndexProvider } from '../plugin.js'
|
||||||
import { ConnectionsCodec, compressedConnectionsKey } from './connectionsCodec.js'
|
import { ConnectionsCodec, compressedConnectionsKey } from './connectionsCodec.js'
|
||||||
|
|
||||||
|
|
@ -67,10 +65,8 @@ export class JsHnswVectorIndex implements VectorIndexProvider {
|
||||||
private dirtyNodes: Set<string> = new Set() // Nodes with unpersisted HNSW data
|
private dirtyNodes: Set<string> = new Set() // Nodes with unpersisted HNSW data
|
||||||
private dirtySystem: boolean = false // Whether system data (entryPoint, maxLevel) needs persist
|
private dirtySystem: boolean = false // Whether system data (entryPoint, maxLevel) needs persist
|
||||||
|
|
||||||
// SQ8 quantization support (B1 optimization)
|
// Lazy vector storage (B2 optimization): evict the float32 vector to
|
||||||
private quantizationEnabled: boolean = false
|
// storage after insert; reload on demand via getVectorSafe() + UnifiedCache.
|
||||||
private rerankMultiplier: number = 3
|
|
||||||
// Lazy vector storage (B2 optimization)
|
|
||||||
private vectorStorageMode: 'memory' | 'lazy' = 'memory'
|
private vectorStorageMode: 'memory' | 'lazy' = 'memory'
|
||||||
|
|
||||||
constructor(
|
constructor(
|
||||||
|
|
@ -87,12 +83,6 @@ export class JsHnswVectorIndex implements VectorIndexProvider {
|
||||||
this.storage = options.storage || null
|
this.storage = options.storage || null
|
||||||
this.persistMode = options.persistMode || 'immediate'
|
this.persistMode = options.persistMode || 'immediate'
|
||||||
|
|
||||||
// SQ8 quantization config (default: disabled, preserves current behavior)
|
|
||||||
if (config.quantization?.enabled) {
|
|
||||||
this.quantizationEnabled = true
|
|
||||||
this.rerankMultiplier = config.quantization.rerankMultiplier ?? 3
|
|
||||||
}
|
|
||||||
|
|
||||||
// Vector storage mode (default: 'memory', preserves current behavior)
|
// Vector storage mode (default: 'memory', preserves current behavior)
|
||||||
this.vectorStorageMode = config.vectorStorage || 'memory'
|
this.vectorStorageMode = config.vectorStorage || 'memory'
|
||||||
|
|
||||||
|
|
@ -376,7 +366,7 @@ export class JsHnswVectorIndex implements VectorIndexProvider {
|
||||||
// Generate random level for this noun
|
// Generate random level for this noun
|
||||||
const nounLevel = this.getRandomLevel()
|
const nounLevel = this.getRandomLevel()
|
||||||
|
|
||||||
// Create new noun with optional SQ8 quantization
|
// Create new noun
|
||||||
const noun: HNSWNoun = {
|
const noun: HNSWNoun = {
|
||||||
id,
|
id,
|
||||||
vector,
|
vector,
|
||||||
|
|
@ -384,14 +374,6 @@ export class JsHnswVectorIndex implements VectorIndexProvider {
|
||||||
level: nounLevel
|
level: nounLevel
|
||||||
}
|
}
|
||||||
|
|
||||||
// Quantize vector if enabled (B1: 4x storage reduction)
|
|
||||||
if (this.quantizationEnabled) {
|
|
||||||
const sq8 = quantizeSQ8(vector)
|
|
||||||
noun.quantizedVector = sq8.quantized
|
|
||||||
noun.codebookMin = sq8.min
|
|
||||||
noun.codebookMax = sq8.max
|
|
||||||
}
|
|
||||||
|
|
||||||
// Initialize empty connection sets for each level
|
// Initialize empty connection sets for each level
|
||||||
for (let level = 0; level <= nounLevel; level++) {
|
for (let level = 0; level <= nounLevel; level++) {
|
||||||
noun.connections.set(level, new Set<string>())
|
noun.connections.set(level, new Set<string>())
|
||||||
|
|
@ -664,16 +646,18 @@ export class JsHnswVectorIndex implements VectorIndexProvider {
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Search for nearest neighbors
|
* Search for nearest neighbors.
|
||||||
*
|
*
|
||||||
* When SQ8 quantization is enabled and reranking is active:
|
* The JS HNSW index computes exact float32 distances throughout — there is
|
||||||
* - Phase 1: Over-retrieve k*rerankMultiplier candidates using SQ8 approximate distances
|
* no approximate/rerank path. The `options.rerank` field exists only for the
|
||||||
* - Phase 2: Load full float32 vectors for candidates, compute exact distances, return top k
|
* shared {@link VectorIndexProvider} contract (a native provider may use it
|
||||||
|
* for its own approximate-then-rerank pipeline); the JS index ignores it.
|
||||||
*
|
*
|
||||||
* @param queryVector Query vector
|
* @param queryVector Query vector
|
||||||
* @param k Number of results to return
|
* @param k Number of results to return
|
||||||
* @param filter Optional filter function
|
* @param filter Optional filter function
|
||||||
* @param options Additional search options
|
* @param options Additional search options (`candidateIds` to restrict the
|
||||||
|
* search to a pre-filtered set; `rerank` is provider-only, ignored here)
|
||||||
*/
|
*/
|
||||||
public async search(
|
public async search(
|
||||||
queryVector: Vector,
|
queryVector: Vector,
|
||||||
|
|
@ -737,11 +721,6 @@ export class JsHnswVectorIndex implements VectorIndexProvider {
|
||||||
|
|
||||||
let currObj = entryPoint
|
let currObj = entryPoint
|
||||||
|
|
||||||
// SQ8: Pre-quantize query vector for fast approximate distances during traversal
|
|
||||||
if (this.quantizationEnabled) {
|
|
||||||
this._querySQ8 = quantizeSQ8(queryVector)
|
|
||||||
}
|
|
||||||
|
|
||||||
// OPTIMIZATION: Preload entry point vector
|
// OPTIMIZATION: Preload entry point vector
|
||||||
await this.preloadVectors([entryPoint.id])
|
await this.preloadVectors([entryPoint.id])
|
||||||
|
|
||||||
|
|
@ -809,14 +788,9 @@ export class JsHnswVectorIndex implements VectorIndexProvider {
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// Determine effective rerank multiplier
|
// Search at level 0. If we have a filter, increase ef to compensate for
|
||||||
const rerankActive = this.quantizationEnabled && (options?.rerank || this.rerankMultiplier > 1)
|
// filtered-out results.
|
||||||
const multiplier = options?.rerank?.multiplier ?? this.rerankMultiplier
|
const ef = filter ? Math.max(this.config.efSearch * 3, k * 3) : Math.max(this.config.efSearch, k)
|
||||||
const effectiveK = rerankActive ? k * multiplier : k
|
|
||||||
|
|
||||||
// Search at level 0 with ef = effectiveK
|
|
||||||
// If we have a filter, increase ef to compensate for filtered results
|
|
||||||
const ef = filter ? Math.max(this.config.efSearch * 3, effectiveK * 3) : Math.max(this.config.efSearch, effectiveK)
|
|
||||||
const nearestNouns = await this.searchLayer(
|
const nearestNouns = await this.searchLayer(
|
||||||
queryVector,
|
queryVector,
|
||||||
currObj,
|
currObj,
|
||||||
|
|
@ -825,36 +799,7 @@ export class JsHnswVectorIndex implements VectorIndexProvider {
|
||||||
filter
|
filter
|
||||||
)
|
)
|
||||||
|
|
||||||
// Phase 2: Rerank with exact float32 distances if quantization is active (B3)
|
// Exact float32 distances throughout — return the top k directly.
|
||||||
if (rerankActive && nearestNouns.size > 0) {
|
|
||||||
// Clear SQ8 cache before reranking (we need exact distances now)
|
|
||||||
this._querySQ8 = null
|
|
||||||
|
|
||||||
const candidates = [...nearestNouns].slice(0, effectiveK)
|
|
||||||
|
|
||||||
// Load full float32 vectors for the candidate set
|
|
||||||
const candidateIds = candidates.map(([id]) => id)
|
|
||||||
await this.preloadVectors(candidateIds)
|
|
||||||
|
|
||||||
// Recompute exact distances
|
|
||||||
const reranked: Array<[string, number]> = []
|
|
||||||
for (const [id] of candidates) {
|
|
||||||
const noun = this.nouns.get(id)
|
|
||||||
if (!noun) continue
|
|
||||||
const exactVector = await this.getVectorSafe(noun)
|
|
||||||
const exactDist = this.distanceFunction(queryVector, exactVector)
|
|
||||||
reranked.push([id, exactDist])
|
|
||||||
}
|
|
||||||
|
|
||||||
// Sort by exact distance and return top k
|
|
||||||
reranked.sort((a, b) => a[1] - b[1])
|
|
||||||
return reranked.slice(0, k)
|
|
||||||
}
|
|
||||||
|
|
||||||
// Clear SQ8 cache
|
|
||||||
this._querySQ8 = null
|
|
||||||
|
|
||||||
// Convert to array and sort by distance
|
|
||||||
return [...nearestNouns].slice(0, k)
|
return [...nearestNouns].slice(0, k)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
@ -1148,19 +1093,6 @@ export class JsHnswVectorIndex implements VectorIndexProvider {
|
||||||
* @returns number | Promise<number> - sync when cached, async when needs load
|
* @returns number | Promise<number> - sync when cached, async when needs load
|
||||||
*/
|
*/
|
||||||
private distanceSafe(queryVector: Vector, noun: HNSWNoun): number | Promise<number> {
|
private distanceSafe(queryVector: Vector, noun: HNSWNoun): number | Promise<number> {
|
||||||
// SQ8 fast path: use quantized distance when available (B1 optimization)
|
|
||||||
// This avoids loading full float32 vectors during graph traversal
|
|
||||||
if (this.quantizationEnabled &&
|
|
||||||
noun.quantizedVector &&
|
|
||||||
noun.codebookMin !== undefined &&
|
|
||||||
noun.codebookMax !== undefined &&
|
|
||||||
this._querySQ8) {
|
|
||||||
return distanceSQ8(
|
|
||||||
this._querySQ8.quantized, this._querySQ8.min, this._querySQ8.max,
|
|
||||||
noun.quantizedVector, noun.codebookMin, noun.codebookMax
|
|
||||||
)
|
|
||||||
}
|
|
||||||
|
|
||||||
// Try sync fast path
|
// Try sync fast path
|
||||||
const nounVector = this.getVectorSync(noun)
|
const nounVector = this.getVectorSync(noun)
|
||||||
|
|
||||||
|
|
@ -1175,9 +1107,6 @@ export class JsHnswVectorIndex implements VectorIndexProvider {
|
||||||
)
|
)
|
||||||
}
|
}
|
||||||
|
|
||||||
// Cached SQ8 quantization of the current query vector for distanceSafe fast path
|
|
||||||
private _querySQ8: SQ8QuantizedVector | null = null
|
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Get all nodes at a specific level for clustering
|
* Get all nodes at a specific level for clustering
|
||||||
* This enables O(n) clustering using HNSW's natural hierarchy
|
* This enables O(n) clustering using HNSW's natural hierarchy
|
||||||
|
|
@ -1299,14 +1228,6 @@ export class JsHnswVectorIndex implements VectorIndexProvider {
|
||||||
level: hnswData.level
|
level: hnswData.level
|
||||||
}
|
}
|
||||||
|
|
||||||
// Restore SQ8 quantized data if quantization is enabled
|
|
||||||
if (this.quantizationEnabled && nounData.vector.length > 0) {
|
|
||||||
const sq8 = quantizeSQ8(nounData.vector)
|
|
||||||
noun.quantizedVector = sq8.quantized
|
|
||||||
noun.codebookMin = sq8.min
|
|
||||||
noun.codebookMax = sq8.max
|
|
||||||
}
|
|
||||||
|
|
||||||
// Restore connections from persisted data — compressed blob path
|
// Restore connections from persisted data — compressed blob path
|
||||||
// first, legacy JSON-array fallback for indexes written before
|
// first, legacy JSON-array fallback for indexes written before
|
||||||
// graph link compression landed.
|
// graph link compression landed.
|
||||||
|
|
|
||||||
|
|
@ -1206,20 +1206,14 @@ export interface BrainyConfig {
|
||||||
}
|
}
|
||||||
| StorageAdapter
|
| StorageAdapter
|
||||||
|
|
||||||
// Index configuration
|
/**
|
||||||
index?: {
|
* Distributed-cluster configuration. **Explicit opt-in only** — Brainy
|
||||||
m?: number // HNSW M parameter
|
* never auto-enables cluster mode from environment heuristics (a single
|
||||||
efConstruction?: number // HNSW construction parameter
|
* production process is the 90th-percentile deployment). Set
|
||||||
efSearch?: number // HNSW search parameter
|
* `enabled: true` (or the `BRAINY_DISTRIBUTED=true` env var) to activate;
|
||||||
}
|
* the remaining fields then default sensibly (64 shards, 3 replicas,
|
||||||
|
* raft consensus, http transport, hostname-derived nodeId).
|
||||||
// Performance options
|
*/
|
||||||
cache?: boolean | { // Enable caching
|
|
||||||
maxSize?: number
|
|
||||||
ttl?: number
|
|
||||||
}
|
|
||||||
|
|
||||||
// Distributed configuration
|
|
||||||
distributed?: {
|
distributed?: {
|
||||||
enabled: boolean
|
enabled: boolean
|
||||||
nodeId?: string
|
nodeId?: string
|
||||||
|
|
@ -1231,23 +1225,18 @@ export interface BrainyConfig {
|
||||||
transport?: 'tcp' | 'http' | 'udp'
|
transport?: 'tcp' | 'http' | 'udp'
|
||||||
}
|
}
|
||||||
|
|
||||||
// Advanced options
|
/**
|
||||||
warmup?: boolean // Warm up on init
|
* Disable the automatic index rebuild check during `init()`. By default
|
||||||
realtime?: boolean // Enable real-time updates
|
* Brainy auto-decides from dataset size: small datasets rebuild missing
|
||||||
multiTenancy?: boolean // Enable service isolation
|
* indexes inline, large datasets rebuild lazily on first query. Set `true`
|
||||||
telemetry?: boolean // Send anonymous usage stats
|
* only when an operator wants full manual control via `repairIndex()`.
|
||||||
|
*/
|
||||||
// Performance tuning options for production
|
disableAutoRebuild?: boolean
|
||||||
disableAutoRebuild?: boolean // Disable automatic index rebuilding on init
|
|
||||||
disableMetrics?: boolean // Completely disable metrics collection
|
|
||||||
disableAutoOptimize?: boolean // Disable automatic index optimization
|
|
||||||
batchWrites?: boolean // Enable write batching for better performance
|
|
||||||
maxConcurrentOperations?: number // Limit concurrent file operations
|
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Vector index configuration (Brainy 8.0).
|
* Vector index configuration (Brainy 8.0).
|
||||||
*
|
*
|
||||||
* Three knobs. No escape hatch. The algorithm-internal HNSW knobs
|
* Two knobs. No escape hatch. The algorithm-internal HNSW knobs
|
||||||
* (`M`, `efConstruction`, `efSearch`, `ml`, …) and DiskANN knobs
|
* (`M`, `efConstruction`, `efSearch`, `ml`, …) and DiskANN knobs
|
||||||
* (`pqM`, `searchListSize`, `alpha`, …) are deliberately not exposed —
|
* (`pqM`, `searchListSize`, `alpha`, …) are deliberately not exposed —
|
||||||
* the `recall` preset covers the legitimate quality/latency tradeoff
|
* the `recall` preset covers the legitimate quality/latency tradeoff
|
||||||
|
|
@ -1271,26 +1260,15 @@ export interface BrainyConfig {
|
||||||
*/
|
*/
|
||||||
recall?: 'fast' | 'balanced' | 'accurate'
|
recall?: 'fast' | 'balanced' | 'accurate'
|
||||||
|
|
||||||
/**
|
|
||||||
* Vector quantization. Independent of `recall` — it trades RAM for a
|
|
||||||
* small recall hit at any preset. `bits: 8` is SQ8 (4× memory reduction,
|
|
||||||
* ~0.4% recall loss). `bits: 4` is SQ4 (8× memory reduction, ~1-3% recall
|
|
||||||
* loss). Both ship in the open-core path; cortex's `distance:sq8` /
|
|
||||||
* `distance:sq4` SIMD providers accelerate them when available.
|
|
||||||
*
|
|
||||||
* Silently ignored on a native DiskANN-style provider path (it uses its
|
|
||||||
* own PQ math instead).
|
|
||||||
*/
|
|
||||||
quantization?: {
|
|
||||||
enabled?: boolean
|
|
||||||
bits?: 8 | 4
|
|
||||||
}
|
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Vector persistence mode. `'immediate'` writes per-noun graph state on
|
* Vector persistence mode. `'immediate'` writes per-noun graph state on
|
||||||
* every `add()`; durable but slower. `'deferred'` writes only on
|
* every `add()`; durable but slower. `'deferred'` writes only on
|
||||||
* `flush()` / `close()`; faster bulk ingest. Defaults to `'immediate'`
|
* `flush()` / `close()`; faster bulk ingest.
|
||||||
* on filesystem storage (the only persistent backend in 8.0).
|
*
|
||||||
|
* **Auto-selected from the storage adapter when omitted:** `'immediate'`
|
||||||
|
* on filesystem storage (durability is the point of a persistent
|
||||||
|
* backend), `'deferred'` on memory storage (nothing survives the process
|
||||||
|
* anyway, so per-add persistence writes are pure overhead).
|
||||||
*
|
*
|
||||||
* Brainy 7.x called this `hnswPersistMode` at the top level. The 8.0
|
* Brainy 7.x called this `hnswPersistMode` at the top level. The 8.0
|
||||||
* surface folds it under `config.vector` alongside `recall`.
|
* surface folds it under `config.vector` alongside `recall`.
|
||||||
|
|
@ -1298,16 +1276,60 @@ export interface BrainyConfig {
|
||||||
persistMode?: 'immediate' | 'deferred'
|
persistMode?: 'immediate' | 'deferred'
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Generational-history retention policy (8.0 MVCC).
|
||||||
|
*
|
||||||
|
* Every `transact()` produces an immutable generation record-set that
|
||||||
|
* serves historical reads (`asOf()`, pinned `Db` values). Without
|
||||||
|
* compaction those record-sets accumulate forever, so Brainy
|
||||||
|
* **auto-compacts on every `flush()` and `close()`** with this policy.
|
||||||
|
* A generation is reclaimed only when it is older than BOTH bounds
|
||||||
|
* (and never while a live `Db` pin protects it):
|
||||||
|
*
|
||||||
|
* - `retainGenerations` — keep at least the N most recent committed
|
||||||
|
* generations (default: `100`).
|
||||||
|
* - `retainMs` — keep everything committed within the window
|
||||||
|
* (default: 7 days).
|
||||||
|
*
|
||||||
|
* Escape hatches: raise either bound for longer time-travel windows,
|
||||||
|
* or set `autoCompact: false` to manage history manually via
|
||||||
|
* `brain.compactHistory()`. Long-term archives belong in
|
||||||
|
* `db.persist(path)` snapshots, which compaction never touches.
|
||||||
|
*/
|
||||||
|
history?: {
|
||||||
|
/** Keep at least this many recent committed generations (default: 100). */
|
||||||
|
retainGenerations?: number
|
||||||
|
/** Keep generations committed within this window in ms (default: 604800000 = 7 days). */
|
||||||
|
retainMs?: number
|
||||||
|
/** Run compaction automatically on flush()/close() (default: true). */
|
||||||
|
autoCompact?: boolean
|
||||||
|
}
|
||||||
|
|
||||||
// Memory management options
|
// Memory management options
|
||||||
maxQueryLimit?: number // Override auto-detected query result limit (max: 100000)
|
maxQueryLimit?: number // Override auto-detected query result limit (max: 100000)
|
||||||
reservedQueryMemory?: number // Memory reserved for queries in bytes (e.g., 1073741824 = 1GB)
|
reservedQueryMemory?: number // Memory reserved for queries in bytes (e.g., 1073741824 = 1GB)
|
||||||
|
|
||||||
// Embedding initialization
|
/**
|
||||||
// Controls when the WASM embedding engine is initialized
|
* Controls when the WASM embedding engine is initialized.
|
||||||
// - false (default): Lazy initialization on first embed() call
|
*
|
||||||
// - true: Eager initialization during brain.init()
|
* **Adaptive default (8.0):** when omitted, the engine eagerly initializes
|
||||||
// Set to true for cloud deployments (Cloud Run, Lambda) where you want
|
* during `init()` whenever the WASM embedder is the *active* one — i.e. no
|
||||||
// WASM compilation to happen during container startup, not on first request
|
* native `'embeddings'` provider is registered — and this instance is a
|
||||||
|
* writer (not `mode: 'reader'`) running outside unit tests. The WASM module
|
||||||
|
* (≈93MB with the embedded model) takes 90-140s to compile on throttled
|
||||||
|
* CPUs, so paying that during boot rather than on the first `embed()`-driven
|
||||||
|
* call is the right default for a single-process server.
|
||||||
|
*
|
||||||
|
* The adaptive path skips itself automatically when a native embeddings
|
||||||
|
* provider owns embeddings, in reader-mode (readers query existing vectors
|
||||||
|
* and never embed), and in unit-test mode (kept fast via the mock embedder).
|
||||||
|
*
|
||||||
|
* - `true` — force eager init during `init()` (the adaptive default already
|
||||||
|
* does this for the active-embedder writer case; set it explicitly to be
|
||||||
|
* unambiguous).
|
||||||
|
* - `false` — explicit override to force lazy init (first `embed()` call)
|
||||||
|
* even when this instance is the active embedder.
|
||||||
|
*/
|
||||||
eagerEmbeddings?: boolean
|
eagerEmbeddings?: boolean
|
||||||
|
|
||||||
// Plugin configuration
|
// Plugin configuration
|
||||||
|
|
@ -1330,8 +1352,10 @@ export interface BrainyConfig {
|
||||||
integrations?: boolean | IntegrationsConfig
|
integrations?: boolean | IntegrationsConfig
|
||||||
|
|
||||||
// Migration configuration
|
// Migration configuration
|
||||||
// - false/undefined (default): Log warning if pending migrations exist, but don't auto-run
|
// - true/undefined (default): Automatically run pending data migrations during
|
||||||
// - true: Automatically run pending migrations during init() for small datasets (<10K entities)
|
// init() for small datasets (<10K entities). Larger datasets defer with a
|
||||||
|
// notice — call brain.migrate() explicitly (optionally with backupTo).
|
||||||
|
// - false: Never auto-run; log a notice when pending migrations exist.
|
||||||
autoMigrate?: boolean
|
autoMigrate?: boolean
|
||||||
|
|
||||||
/**
|
/**
|
||||||
|
|
|
||||||
|
|
@ -17,8 +17,6 @@
|
||||||
* descending by score, with stable ordering for ties (lower original index first). This
|
* descending by score, with stable ordering for ties (lower original index first). This
|
||||||
* guarantees that wiring the hook in or out never changes which results a query returns or
|
* guarantees that wiring the hook in or out never changes which results a query returns or
|
||||||
* in what order — only how fast the ranking is computed.
|
* in what order — only how fast the ranking is computed.
|
||||||
*
|
|
||||||
* @see setSQ8DistanceImplementation in ./vectorQuantization.js for the sibling provider-swap pattern.
|
|
||||||
*/
|
*/
|
||||||
|
|
||||||
/**
|
/**
|
||||||
|
|
|
||||||
|
|
@ -1,581 +0,0 @@
|
||||||
/**
|
|
||||||
* Vector Quantization Utilities
|
|
||||||
*
|
|
||||||
* Standalone SQ8 (Scalar Quantization, 8-bit) functions for HNSW vector compression.
|
|
||||||
* Each vector is independently quantized using its own min/max codebook.
|
|
||||||
*
|
|
||||||
* Storage format per vector:
|
|
||||||
* - quantized: Uint8Array (dimension bytes, 4x smaller than float32)
|
|
||||||
* - min: number (codebook minimum)
|
|
||||||
* - max: number (codebook maximum)
|
|
||||||
*
|
|
||||||
* Accuracy: SQ8 introduces ~0.4% error per dimension on normalized vectors.
|
|
||||||
* Combined with reranking (3x over-retrieval + float32 rerank), recall@100
|
|
||||||
* is expected to remain >99% for typical workloads.
|
|
||||||
*/
|
|
||||||
|
|
||||||
import type { Vector } from '../coreTypes.js'
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Result of SQ8 quantization
|
|
||||||
*/
|
|
||||||
export interface SQ8QuantizedVector {
|
|
||||||
quantized: Uint8Array
|
|
||||||
min: number
|
|
||||||
max: number
|
|
||||||
}
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Configuration for HNSW quantization
|
|
||||||
*/
|
|
||||||
export interface HNSWQuantizationConfig {
|
|
||||||
enabled: boolean
|
|
||||||
bits: 8 | 4
|
|
||||||
rerankMultiplier: number
|
|
||||||
}
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Quantize a float32 vector to SQ8 (8-bit unsigned integers).
|
|
||||||
*
|
|
||||||
* Maps [min, max] of the vector to [0, 255].
|
|
||||||
* For zero-range vectors (all values equal), all quantized values are 128.
|
|
||||||
*
|
|
||||||
* @param vector - Input float32 vector
|
|
||||||
* @returns Quantized vector with codebook (min/max)
|
|
||||||
*/
|
|
||||||
export function quantizeSQ8(vector: Vector): SQ8QuantizedVector {
|
|
||||||
const len = vector.length
|
|
||||||
let min = vector[0]
|
|
||||||
let max = vector[0]
|
|
||||||
|
|
||||||
for (let i = 1; i < len; i++) {
|
|
||||||
const v = vector[i]
|
|
||||||
if (v < min) min = v
|
|
||||||
if (v > max) max = v
|
|
||||||
}
|
|
||||||
|
|
||||||
const range = max - min
|
|
||||||
const quantized = new Uint8Array(len)
|
|
||||||
|
|
||||||
if (range === 0) {
|
|
||||||
// All values are identical — map to midpoint
|
|
||||||
quantized.fill(128)
|
|
||||||
} else {
|
|
||||||
const scale = 255 / range
|
|
||||||
for (let i = 0; i < len; i++) {
|
|
||||||
quantized[i] = Math.round((vector[i] - min) * scale)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
return { quantized, min, max }
|
|
||||||
}
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Dequantize an SQ8 vector back to float32.
|
|
||||||
*
|
|
||||||
* @param quantized - Uint8Array of quantized values
|
|
||||||
* @param min - Codebook minimum
|
|
||||||
* @param max - Codebook maximum
|
|
||||||
* @returns Reconstructed float32 vector
|
|
||||||
*/
|
|
||||||
export function dequantizeSQ8(quantized: Uint8Array, min: number, max: number): Vector {
|
|
||||||
const len = quantized.length
|
|
||||||
const result = new Array<number>(len)
|
|
||||||
const range = max - min
|
|
||||||
|
|
||||||
if (range === 0) {
|
|
||||||
result.fill(min)
|
|
||||||
} else {
|
|
||||||
const scale = range / 255
|
|
||||||
for (let i = 0; i < len; i++) {
|
|
||||||
result[i] = min + quantized[i] * scale
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
return result
|
|
||||||
}
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Reference JS implementation of {@link distanceSQ8}: approximate cosine distance
|
|
||||||
* between two SQ8-quantized vectors, operating directly on the uint8 arrays without
|
|
||||||
* full dequantization. Cosine distance is `1 - (A·B / (|A| · |B|))`; for quantized
|
|
||||||
* values `q_i` with codebook `(min, max)` the actual value is
|
|
||||||
* `min + q_i * (max - min) / 255`, and the dot product / norms are computed on the
|
|
||||||
* quantized values and rescaled by the codebook parameters.
|
|
||||||
*
|
|
||||||
* Exported so callers can explicitly restore the default after swapping in a native
|
|
||||||
* implementation, and so cross-language parity tests can compare native output
|
|
||||||
* against this baseline.
|
|
||||||
*
|
|
||||||
* @param a - First quantized vector.
|
|
||||||
* @param aMin - First vector's codebook minimum.
|
|
||||||
* @param aMax - First vector's codebook maximum.
|
|
||||||
* @param b - Second quantized vector.
|
|
||||||
* @param bMin - Second vector's codebook minimum.
|
|
||||||
* @param bMax - Second vector's codebook maximum.
|
|
||||||
* @returns Approximate cosine distance in [0, 2].
|
|
||||||
*/
|
|
||||||
export function distanceSQ8Js(
|
|
||||||
a: Uint8Array,
|
|
||||||
aMin: number,
|
|
||||||
aMax: number,
|
|
||||||
b: Uint8Array,
|
|
||||||
bMin: number,
|
|
||||||
bMax: number
|
|
||||||
): number {
|
|
||||||
const len = a.length
|
|
||||||
|
|
||||||
// Compute raw integer sums for efficiency
|
|
||||||
// actual_a[i] = aMin + a[i] * aScale, where aScale = (aMax - aMin) / 255
|
|
||||||
// actual_b[i] = bMin + b[i] * bScale, where bScale = (bMax - bMin) / 255
|
|
||||||
//
|
|
||||||
// dot = sum(actual_a[i] * actual_b[i])
|
|
||||||
// = sum((aMin + a[i]*aScale) * (bMin + b[i]*bScale))
|
|
||||||
// = n*aMin*bMin + aMin*bScale*sum(b[i]) + bMin*aScale*sum(a[i]) + aScale*bScale*sum(a[i]*b[i])
|
|
||||||
//
|
|
||||||
// normA^2 = sum(actual_a[i]^2)
|
|
||||||
// = n*aMin^2 + 2*aMin*aScale*sum(a[i]) + aScale^2*sum(a[i]^2)
|
|
||||||
|
|
||||||
const aScale = (aMax - aMin) / 255
|
|
||||||
const bScale = (bMax - bMin) / 255
|
|
||||||
|
|
||||||
let sumA = 0
|
|
||||||
let sumB = 0
|
|
||||||
let sumAB = 0
|
|
||||||
let sumAA = 0
|
|
||||||
let sumBB = 0
|
|
||||||
|
|
||||||
for (let i = 0; i < len; i++) {
|
|
||||||
const ai = a[i]
|
|
||||||
const bi = b[i]
|
|
||||||
sumA += ai
|
|
||||||
sumB += bi
|
|
||||||
sumAB += ai * bi
|
|
||||||
sumAA += ai * ai
|
|
||||||
sumBB += bi * bi
|
|
||||||
}
|
|
||||||
|
|
||||||
const dot =
|
|
||||||
len * aMin * bMin +
|
|
||||||
aMin * bScale * sumB +
|
|
||||||
bMin * aScale * sumA +
|
|
||||||
aScale * bScale * sumAB
|
|
||||||
|
|
||||||
const normASq =
|
|
||||||
len * aMin * aMin +
|
|
||||||
2 * aMin * aScale * sumA +
|
|
||||||
aScale * aScale * sumAA
|
|
||||||
|
|
||||||
const normBSq =
|
|
||||||
len * bMin * bMin +
|
|
||||||
2 * bMin * bScale * sumB +
|
|
||||||
bScale * bScale * sumBB
|
|
||||||
|
|
||||||
const normA = Math.sqrt(normASq)
|
|
||||||
const normB = Math.sqrt(normBSq)
|
|
||||||
|
|
||||||
if (normA === 0 || normB === 0) {
|
|
||||||
return 1.0 // Maximum distance for zero vectors
|
|
||||||
}
|
|
||||||
|
|
||||||
const cosine = dot / (normA * normB)
|
|
||||||
// Clamp to [-1, 1] to handle floating point imprecision
|
|
||||||
return 1 - Math.max(-1, Math.min(1, cosine))
|
|
||||||
}
|
|
||||||
|
|
||||||
/** Signature of the SQ8 approximate-distance function (quantized cosine distance). */
|
|
||||||
export type SQ8DistanceFn = (
|
|
||||||
a: Uint8Array,
|
|
||||||
aMin: number,
|
|
||||||
aMax: number,
|
|
||||||
b: Uint8Array,
|
|
||||||
bMin: number,
|
|
||||||
bMax: number
|
|
||||||
) => number
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Active SQ8 distance implementation. Defaults to the JS `distanceSQ8Js`; swapped to
|
|
||||||
* a native SIMD implementation by {@link setSQ8DistanceImplementation} when a cortex
|
|
||||||
* `distance:sq8` provider is registered. The native implementation is byte-compatible
|
|
||||||
* (same quantized cosine formula), so results match within float tolerance.
|
|
||||||
*/
|
|
||||||
let activeSQ8Distance: SQ8DistanceFn = distanceSQ8Js
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Approximate cosine distance between two SQ8-quantized vectors, dispatched to the
|
|
||||||
* active implementation (JS by default, native when a `distance:sq8` provider is
|
|
||||||
* registered). Used in the HNSW SQ8 reranking hot path.
|
|
||||||
*
|
|
||||||
* @param a - First quantized vector.
|
|
||||||
* @param aMin - First vector's codebook minimum.
|
|
||||||
* @param aMax - First vector's codebook maximum.
|
|
||||||
* @param b - Second quantized vector.
|
|
||||||
* @param bMin - Second vector's codebook minimum.
|
|
||||||
* @param bMax - Second vector's codebook maximum.
|
|
||||||
* @returns Approximate cosine distance in [0, 2].
|
|
||||||
*/
|
|
||||||
export function distanceSQ8(
|
|
||||||
a: Uint8Array,
|
|
||||||
aMin: number,
|
|
||||||
aMax: number,
|
|
||||||
b: Uint8Array,
|
|
||||||
bMin: number,
|
|
||||||
bMax: number
|
|
||||||
): number {
|
|
||||||
return activeSQ8Distance(a, aMin, aMax, b, bMin, bMax)
|
|
||||||
}
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Replace the SQ8 approximate-distance implementation at runtime (e.g. cortex's Rust
|
|
||||||
* SIMD `distance:sq8`). Called by `brainy.ts` when the provider is registered. Pass
|
|
||||||
* {@link distanceSQ8Js} to restore the JS default.
|
|
||||||
*
|
|
||||||
* @param fn - Native implementation; must match {@link SQ8DistanceFn} byte-for-byte.
|
|
||||||
*/
|
|
||||||
export function setSQ8DistanceImplementation(fn: SQ8DistanceFn): void {
|
|
||||||
activeSQ8Distance = fn
|
|
||||||
}
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Serialize an SQ8 quantized vector to a compact binary format.
|
|
||||||
*
|
|
||||||
* Format: [min:float32][max:float32][quantized:uint8[]]
|
|
||||||
* Total size: 8 + dimension bytes
|
|
||||||
*
|
|
||||||
* @param sq8 - Quantized vector
|
|
||||||
* @returns ArrayBuffer with serialized data
|
|
||||||
*/
|
|
||||||
export function serializeSQ8(sq8: SQ8QuantizedVector): ArrayBuffer {
|
|
||||||
const buffer = new ArrayBuffer(8 + sq8.quantized.length)
|
|
||||||
const view = new DataView(buffer)
|
|
||||||
view.setFloat32(0, sq8.min, true) // little-endian
|
|
||||||
view.setFloat32(4, sq8.max, true)
|
|
||||||
new Uint8Array(buffer, 8).set(sq8.quantized)
|
|
||||||
return buffer
|
|
||||||
}
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Deserialize an SQ8 quantized vector from binary format.
|
|
||||||
*
|
|
||||||
* @param buffer - ArrayBuffer with serialized data
|
|
||||||
* @returns Deserialized SQ8 quantized vector
|
|
||||||
*/
|
|
||||||
export function deserializeSQ8(buffer: ArrayBuffer): SQ8QuantizedVector {
|
|
||||||
const view = new DataView(buffer)
|
|
||||||
const min = view.getFloat32(0, true)
|
|
||||||
const max = view.getFloat32(4, true)
|
|
||||||
const quantized = new Uint8Array(buffer, 8)
|
|
||||||
return { quantized, min, max }
|
|
||||||
}
|
|
||||||
|
|
||||||
// ===========================================================================
|
|
||||||
// SQ4 — 4-bit scalar quantization (2.5.0 #30)
|
|
||||||
//
|
|
||||||
// Each dimension is mapped to [0, 15] (16 levels) and packed two-per-byte:
|
|
||||||
// high nibble = vector[2i], low nibble = vector[2i+1]. Achieves 8× compression
|
|
||||||
// vs float32 (vs 4× for SQ8) at the cost of higher quantization error.
|
|
||||||
//
|
|
||||||
// The format + arithmetic are byte-for-byte identical to cortex's Rust
|
|
||||||
// `quantize_sq4` / `dequantize_sq4` / `cosine_distance_sq4` (see
|
|
||||||
// `cortex/native/src/quantization.rs`). When a `distance:sq4` provider is
|
|
||||||
// registered (cortex's SIMD Rust implementation), `setSQ4DistanceImplementation`
|
|
||||||
// swaps the JS reference in for the native one; cross-language parity tests
|
|
||||||
// verify the swap is recall-equivalent within float tolerance.
|
|
||||||
// ===========================================================================
|
|
||||||
|
|
||||||
/**
|
|
||||||
* @description 4-bit scalar-quantized vector. The `quantized` buffer is packed
|
|
||||||
* two nibbles per byte (high nibble = `vector[2i]`, low nibble = `vector[2i+1]`).
|
|
||||||
* Odd dimensions place the final value in the high nibble and leave the low
|
|
||||||
* nibble zero. `dim` must be recorded explicitly because packed byte length
|
|
||||||
* `ceil(dim/2)` doesn't encode whether the trailing low nibble is data or pad.
|
|
||||||
*/
|
|
||||||
export interface SQ4QuantizedVector {
|
|
||||||
/** Packed nibbles. Length = `ceil(dim / 2)`. */
|
|
||||||
quantized: Uint8Array
|
|
||||||
/** Codebook minimum (used to reconstruct the value range on dequantization). */
|
|
||||||
min: number
|
|
||||||
/** Codebook maximum. */
|
|
||||||
max: number
|
|
||||||
/** Original vector dimensionality (needed to detect the odd-dim pad nibble). */
|
|
||||||
dim: number
|
|
||||||
}
|
|
||||||
|
|
||||||
/**
|
|
||||||
* @description Quantize a float vector to SQ4. Each dimension is mapped to
|
|
||||||
* `[0, 15]` via `round((v - min) / range * 15)`, clamped, then packed
|
|
||||||
* two-per-byte. Zero-range vectors (all values identical) map to the midpoint
|
|
||||||
* `8` in every nibble — `0x88` in every byte.
|
|
||||||
*
|
|
||||||
* Byte-for-byte identical to cortex's `quantize_sq4` (Rust); cross-language
|
|
||||||
* parity tests confirm this. Use {@link distanceSQ4} for approximate cosine
|
|
||||||
* directly on the packed bytes — there's no need to {@link dequantizeSQ4}
|
|
||||||
* before computing distances.
|
|
||||||
*
|
|
||||||
* @param vector - Input float vector (length must be ≥ 1).
|
|
||||||
* @returns Packed SQ4 vector with codebook + original dimension.
|
|
||||||
* @throws If `vector` is empty.
|
|
||||||
*/
|
|
||||||
export function quantizeSQ4(vector: Vector): SQ4QuantizedVector {
|
|
||||||
const dim = vector.length
|
|
||||||
if (dim === 0) {
|
|
||||||
throw new Error('quantizeSQ4: vector cannot be empty')
|
|
||||||
}
|
|
||||||
|
|
||||||
let min = vector[0]
|
|
||||||
let max = vector[0]
|
|
||||||
for (let i = 1; i < dim; i++) {
|
|
||||||
const v = vector[i]
|
|
||||||
if (v < min) min = v
|
|
||||||
if (v > max) max = v
|
|
||||||
}
|
|
||||||
|
|
||||||
const packedLen = (dim + 1) >>> 1
|
|
||||||
const quantized = new Uint8Array(packedLen)
|
|
||||||
const range = max - min
|
|
||||||
|
|
||||||
if (range === 0) {
|
|
||||||
// All values identical — every nibble = 8 (midpoint of [0, 15]).
|
|
||||||
quantized.fill(0x88)
|
|
||||||
return { quantized, min, max, dim }
|
|
||||||
}
|
|
||||||
|
|
||||||
const invRange = 15 / range
|
|
||||||
let i = 0
|
|
||||||
let byteIdx = 0
|
|
||||||
while (i + 1 < dim) {
|
|
||||||
const hi = clampNibble(Math.round((vector[i] - min) * invRange))
|
|
||||||
const lo = clampNibble(Math.round((vector[i + 1] - min) * invRange))
|
|
||||||
quantized[byteIdx++] = (hi << 4) | lo
|
|
||||||
i += 2
|
|
||||||
}
|
|
||||||
if (i < dim) {
|
|
||||||
// Odd dimension: final value in the high nibble, low nibble = 0 pad.
|
|
||||||
const hi = clampNibble(Math.round((vector[i] - min) * invRange))
|
|
||||||
quantized[byteIdx] = hi << 4
|
|
||||||
}
|
|
||||||
|
|
||||||
return { quantized, min, max, dim }
|
|
||||||
}
|
|
||||||
|
|
||||||
/**
|
|
||||||
* @description Dequantize an SQ4 packed buffer back to a float vector.
|
|
||||||
* Reverses {@link quantizeSQ4}: each nibble is mapped back to
|
|
||||||
* `min + nibble * (max - min) / 15`. The trailing pad nibble of an odd-`dim`
|
|
||||||
* vector is correctly dropped (the result has length exactly `dim`).
|
|
||||||
*
|
|
||||||
* @param quantized - Packed SQ4 buffer (length `ceil(dim/2)`).
|
|
||||||
* @param min - Codebook minimum from {@link quantizeSQ4}.
|
|
||||||
* @param max - Codebook maximum from {@link quantizeSQ4}.
|
|
||||||
* @param dim - Original vector dimensionality.
|
|
||||||
* @returns Reconstructed float vector of length `dim`.
|
|
||||||
*/
|
|
||||||
export function dequantizeSQ4(
|
|
||||||
quantized: Uint8Array,
|
|
||||||
min: number,
|
|
||||||
max: number,
|
|
||||||
dim: number
|
|
||||||
): Vector {
|
|
||||||
const result = new Array<number>(dim)
|
|
||||||
const range = max - min
|
|
||||||
|
|
||||||
if (range === 0) {
|
|
||||||
for (let i = 0; i < dim; i++) result[i] = min
|
|
||||||
return result
|
|
||||||
}
|
|
||||||
|
|
||||||
const scale = range / 15
|
|
||||||
let out = 0
|
|
||||||
for (let b = 0; b < quantized.length && out < dim; b++) {
|
|
||||||
const byte = quantized[b]
|
|
||||||
result[out++] = min + ((byte >>> 4) & 0x0f) * scale
|
|
||||||
if (out < dim) {
|
|
||||||
result[out++] = min + (byte & 0x0f) * scale
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
return result
|
|
||||||
}
|
|
||||||
|
|
||||||
/**
|
|
||||||
* @description Reference JS implementation of {@link distanceSQ4}: approximate
|
|
||||||
* cosine distance between two SQ4-quantized vectors. Unpacks nibbles inline,
|
|
||||||
* then accumulates the same `(sumA, sumB, sumAA, sumBB, sumAB)` integers SQ8
|
|
||||||
* uses and applies the SQ8 cosine-from-quantized formula with `15` (not `255`)
|
|
||||||
* as the quantization range. Result is byte-equivalent to cortex's
|
|
||||||
* `cosine_distance_sq4` (Rust) within float tolerance.
|
|
||||||
*
|
|
||||||
* Exported so callers can explicitly restore the JS default after a native
|
|
||||||
* swap, and so cross-language parity tests compare native output against this
|
|
||||||
* baseline.
|
|
||||||
*/
|
|
||||||
export function distanceSQ4Js(
|
|
||||||
a: Uint8Array,
|
|
||||||
aMin: number,
|
|
||||||
aMax: number,
|
|
||||||
aDim: number,
|
|
||||||
b: Uint8Array,
|
|
||||||
bMin: number,
|
|
||||||
bMax: number,
|
|
||||||
bDim: number
|
|
||||||
): number {
|
|
||||||
if (aDim !== bDim) {
|
|
||||||
throw new Error(`distanceSQ4Js: dimension mismatch ${aDim} vs ${bDim}`)
|
|
||||||
}
|
|
||||||
if (aDim === 0) {
|
|
||||||
throw new Error('distanceSQ4Js: vectors cannot be empty')
|
|
||||||
}
|
|
||||||
|
|
||||||
let sumA = 0
|
|
||||||
let sumB = 0
|
|
||||||
let sumAB = 0
|
|
||||||
let sumAA = 0
|
|
||||||
let sumBB = 0
|
|
||||||
|
|
||||||
// Walk both packed buffers nibble-by-nibble, in lock-step.
|
|
||||||
let i = 0
|
|
||||||
let byteIdx = 0
|
|
||||||
while (i + 1 < aDim) {
|
|
||||||
const aByte = a[byteIdx]
|
|
||||||
const bByte = b[byteIdx]
|
|
||||||
const aHi = (aByte >>> 4) & 0x0f
|
|
||||||
const aLo = aByte & 0x0f
|
|
||||||
const bHi = (bByte >>> 4) & 0x0f
|
|
||||||
const bLo = bByte & 0x0f
|
|
||||||
|
|
||||||
sumA += aHi + aLo
|
|
||||||
sumB += bHi + bLo
|
|
||||||
sumAB += aHi * bHi + aLo * bLo
|
|
||||||
sumAA += aHi * aHi + aLo * aLo
|
|
||||||
sumBB += bHi * bHi + bLo * bLo
|
|
||||||
|
|
||||||
i += 2
|
|
||||||
byteIdx++
|
|
||||||
}
|
|
||||||
if (i < aDim) {
|
|
||||||
// Odd dim: only the high nibble of the final byte is data.
|
|
||||||
const aHi = (a[byteIdx] >>> 4) & 0x0f
|
|
||||||
const bHi = (b[byteIdx] >>> 4) & 0x0f
|
|
||||||
sumA += aHi
|
|
||||||
sumB += bHi
|
|
||||||
sumAB += aHi * bHi
|
|
||||||
sumAA += aHi * aHi
|
|
||||||
sumBB += bHi * bHi
|
|
||||||
}
|
|
||||||
|
|
||||||
const aScale = (aMax - aMin) / 15
|
|
||||||
const bScale = (bMax - bMin) / 15
|
|
||||||
const n = aDim
|
|
||||||
|
|
||||||
const dot =
|
|
||||||
n * aMin * bMin +
|
|
||||||
aMin * bScale * sumB +
|
|
||||||
bMin * aScale * sumA +
|
|
||||||
aScale * bScale * sumAB
|
|
||||||
|
|
||||||
const normASq =
|
|
||||||
n * aMin * aMin +
|
|
||||||
2 * aMin * aScale * sumA +
|
|
||||||
aScale * aScale * sumAA
|
|
||||||
|
|
||||||
const normBSq =
|
|
||||||
n * bMin * bMin +
|
|
||||||
2 * bMin * bScale * sumB +
|
|
||||||
bScale * bScale * sumBB
|
|
||||||
|
|
||||||
const denom = Math.sqrt(normASq) * Math.sqrt(normBSq)
|
|
||||||
if (denom === 0) return 1.0
|
|
||||||
|
|
||||||
const cosine = dot / denom
|
|
||||||
return 1 - Math.max(-1, Math.min(1, cosine))
|
|
||||||
}
|
|
||||||
|
|
||||||
/** Signature of the SQ4 approximate-distance function (quantized cosine distance). */
|
|
||||||
export type SQ4DistanceFn = (
|
|
||||||
a: Uint8Array,
|
|
||||||
aMin: number,
|
|
||||||
aMax: number,
|
|
||||||
aDim: number,
|
|
||||||
b: Uint8Array,
|
|
||||||
bMin: number,
|
|
||||||
bMax: number,
|
|
||||||
bDim: number
|
|
||||||
) => number
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Active SQ4 distance implementation. Defaults to the JS {@link distanceSQ4Js};
|
|
||||||
* swapped to cortex's SIMD Rust implementation by {@link setSQ4DistanceImplementation}
|
|
||||||
* when a `distance:sq4` provider is registered. The native implementation is
|
|
||||||
* byte-compatible (same quantized cosine formula with `15` as the quantization
|
|
||||||
* range), so results match within float tolerance.
|
|
||||||
*/
|
|
||||||
let activeSQ4Distance: SQ4DistanceFn = distanceSQ4Js
|
|
||||||
|
|
||||||
/**
|
|
||||||
* @description Approximate cosine distance between two SQ4-quantized vectors,
|
|
||||||
* dispatched to the active implementation (JS by default, native when
|
|
||||||
* `distance:sq4` is registered). Used in the HNSW SQ4 reranking hot path when
|
|
||||||
* `config.quantization.bits === 4`.
|
|
||||||
*/
|
|
||||||
export function distanceSQ4(
|
|
||||||
a: Uint8Array,
|
|
||||||
aMin: number,
|
|
||||||
aMax: number,
|
|
||||||
aDim: number,
|
|
||||||
b: Uint8Array,
|
|
||||||
bMin: number,
|
|
||||||
bMax: number,
|
|
||||||
bDim: number
|
|
||||||
): number {
|
|
||||||
return activeSQ4Distance(a, aMin, aMax, aDim, b, bMin, bMax, bDim)
|
|
||||||
}
|
|
||||||
|
|
||||||
/**
|
|
||||||
* @description Replace the SQ4 approximate-distance implementation at runtime
|
|
||||||
* (cortex's Rust SIMD `distance:sq4`). Called by `brainy.ts` when the provider
|
|
||||||
* is registered. Pass {@link distanceSQ4Js} to restore the JS default.
|
|
||||||
*/
|
|
||||||
export function setSQ4DistanceImplementation(fn: SQ4DistanceFn): void {
|
|
||||||
activeSQ4Distance = fn
|
|
||||||
}
|
|
||||||
|
|
||||||
/**
|
|
||||||
* @description Serialize an SQ4 quantized vector to a compact binary format.
|
|
||||||
*
|
|
||||||
* Layout: `[min: float32][max: float32][dim: uint32 LE][packed: uint8[]]`.
|
|
||||||
* Total size: `12 + ceil(dim/2)` bytes.
|
|
||||||
*
|
|
||||||
* The `dim` field is required (unlike SQ8) because the packed length alone
|
|
||||||
* doesn't disambiguate even-vs-odd original dimension. The 4-byte `dim`
|
|
||||||
* header keeps total overhead under 1% for typical 384-dim vectors.
|
|
||||||
*/
|
|
||||||
export function serializeSQ4(sq4: SQ4QuantizedVector): ArrayBuffer {
|
|
||||||
const packedLen = sq4.quantized.length
|
|
||||||
const buffer = new ArrayBuffer(12 + packedLen)
|
|
||||||
const view = new DataView(buffer)
|
|
||||||
view.setFloat32(0, sq4.min, true) // little-endian
|
|
||||||
view.setFloat32(4, sq4.max, true)
|
|
||||||
view.setUint32(8, sq4.dim, true)
|
|
||||||
new Uint8Array(buffer, 12).set(sq4.quantized)
|
|
||||||
return buffer
|
|
||||||
}
|
|
||||||
|
|
||||||
/**
|
|
||||||
* @description Deserialize an SQ4 quantized vector from the binary format
|
|
||||||
* written by {@link serializeSQ4}.
|
|
||||||
*/
|
|
||||||
export function deserializeSQ4(buffer: ArrayBuffer): SQ4QuantizedVector {
|
|
||||||
const view = new DataView(buffer)
|
|
||||||
const min = view.getFloat32(0, true)
|
|
||||||
const max = view.getFloat32(4, true)
|
|
||||||
const dim = view.getUint32(8, true)
|
|
||||||
const quantized = new Uint8Array(buffer, 12)
|
|
||||||
return { quantized, min, max, dim }
|
|
||||||
}
|
|
||||||
|
|
||||||
/** Clamp a value to `[0, 15]` and return as an integer in the 4-bit range. */
|
|
||||||
function clampNibble(n: number): number {
|
|
||||||
if (n < 0) return 0
|
|
||||||
if (n > 15) return 15
|
|
||||||
return n | 0
|
|
||||||
}
|
|
||||||
|
|
@ -6,7 +6,6 @@
|
||||||
* - Search returns correct results after vector eviction
|
* - Search returns correct results after vector eviction
|
||||||
* - Memory mode retains vectors (default behavior)
|
* - Memory mode retains vectors (default behavior)
|
||||||
* - Lazy mode requires storage adapter
|
* - Lazy mode requires storage adapter
|
||||||
* - Combined lazy + quantization mode
|
|
||||||
*/
|
*/
|
||||||
|
|
||||||
import { describe, it, expect, beforeEach } from 'vitest'
|
import { describe, it, expect, beforeEach } from 'vitest'
|
||||||
|
|
@ -180,79 +179,7 @@ describe('Lazy Vector Loading (B2)', () => {
|
||||||
})
|
})
|
||||||
|
|
||||||
// =================================================================
|
// =================================================================
|
||||||
// 4. LAZY + QUANTIZATION COMBINED
|
// 4. MULTIPLE SEARCHES IN LAZY MODE
|
||||||
// =================================================================
|
|
||||||
describe('lazy mode with quantization', () => {
|
|
||||||
it('should use SQ8 for traversal and load full vectors for rerank', async () => {
|
|
||||||
const storage = new MemoryStorage()
|
|
||||||
const index = new JsHnswVectorIndex(
|
|
||||||
{
|
|
||||||
M: 8,
|
|
||||||
efConstruction: 100,
|
|
||||||
efSearch: 50,
|
|
||||||
ml: 8,
|
|
||||||
vectorStorage: 'lazy',
|
|
||||||
quantization: { enabled: true, rerankMultiplier: 3 }
|
|
||||||
},
|
|
||||||
euclideanDistance,
|
|
||||||
{ useParallelization: false, storage }
|
|
||||||
)
|
|
||||||
|
|
||||||
const targetId = uuidv4()
|
|
||||||
const target = randomVector(dim)
|
|
||||||
await saveVector(storage, targetId, target)
|
|
||||||
await index.addItem({ id: targetId, vector: target })
|
|
||||||
|
|
||||||
for (let i = 0; i < 30; i++) {
|
|
||||||
const id = uuidv4()
|
|
||||||
const v = randomVector(dim)
|
|
||||||
await saveVector(storage, id, v)
|
|
||||||
await index.addItem({ id, vector: v })
|
|
||||||
}
|
|
||||||
|
|
||||||
// Search with reranking — should load full vectors for rerank phase
|
|
||||||
const results = await index.search(target, 5)
|
|
||||||
expect(results.length).toBe(5)
|
|
||||||
|
|
||||||
// The target should be the closest match
|
|
||||||
const targetResult = results.find(([id]) => id === targetId)
|
|
||||||
expect(targetResult).toBeDefined()
|
|
||||||
})
|
|
||||||
|
|
||||||
it('should return results sorted by exact distance after rerank', async () => {
|
|
||||||
const storage = new MemoryStorage()
|
|
||||||
const index = new JsHnswVectorIndex(
|
|
||||||
{
|
|
||||||
M: 8,
|
|
||||||
efConstruction: 100,
|
|
||||||
efSearch: 50,
|
|
||||||
ml: 8,
|
|
||||||
vectorStorage: 'lazy',
|
|
||||||
quantization: { enabled: true, rerankMultiplier: 3 }
|
|
||||||
},
|
|
||||||
euclideanDistance,
|
|
||||||
{ useParallelization: false, storage }
|
|
||||||
)
|
|
||||||
|
|
||||||
for (let i = 0; i < 40; i++) {
|
|
||||||
const id = uuidv4()
|
|
||||||
const v = randomVector(dim)
|
|
||||||
await saveVector(storage, id, v)
|
|
||||||
await index.addItem({ id, vector: v })
|
|
||||||
}
|
|
||||||
|
|
||||||
const query = randomVector(dim)
|
|
||||||
const results = await index.search(query, 10)
|
|
||||||
|
|
||||||
// Results should be sorted by exact distance (rerank ensures this)
|
|
||||||
for (let i = 1; i < results.length; i++) {
|
|
||||||
expect(results[i][1]).toBeGreaterThanOrEqual(results[i - 1][1])
|
|
||||||
}
|
|
||||||
})
|
|
||||||
})
|
|
||||||
|
|
||||||
// =================================================================
|
|
||||||
// 5. MULTIPLE SEARCHES IN LAZY MODE
|
|
||||||
// =================================================================
|
// =================================================================
|
||||||
describe('multiple searches in lazy mode', () => {
|
describe('multiple searches in lazy mode', () => {
|
||||||
it('should handle repeated searches correctly', async () => {
|
it('should handle repeated searches correctly', async () => {
|
||||||
|
|
|
||||||
|
|
@ -1,500 +0,0 @@
|
||||||
/**
|
|
||||||
* SQ8 Quantization Tests
|
|
||||||
*
|
|
||||||
* Tests for:
|
|
||||||
* - SQ8 quantize/dequantize round-trip accuracy
|
|
||||||
* - SQ8 distance vs float32 distance correlation
|
|
||||||
* - Serialization/deserialization round-trip
|
|
||||||
* - Search with quantization enabled: recall vs exact
|
|
||||||
* - Two-phase rerank: improved recall over single-phase SQ8
|
|
||||||
* - Config defaults: quantization disabled by default (no behavior change)
|
|
||||||
*/
|
|
||||||
|
|
||||||
import { describe, it, expect, beforeEach, afterEach } from 'vitest'
|
|
||||||
import { v4 as uuidv4 } from 'uuid'
|
|
||||||
import {
|
|
||||||
quantizeSQ8,
|
|
||||||
dequantizeSQ8,
|
|
||||||
distanceSQ8,
|
|
||||||
distanceSQ8Js,
|
|
||||||
setSQ8DistanceImplementation,
|
|
||||||
serializeSQ8,
|
|
||||||
deserializeSQ8
|
|
||||||
} from '../../../src/utils/vectorQuantization.js'
|
|
||||||
import { JsHnswVectorIndex } from '../../../src/hnsw/hnswIndex.js'
|
|
||||||
import { euclideanDistance } from '../../../src/utils/index.js'
|
|
||||||
import { MemoryStorage } from '../../../src/storage/adapters/memoryStorage.js'
|
|
||||||
|
|
||||||
// Helper: generate a random vector of given dimension
|
|
||||||
function randomVector(dim: number): number[] {
|
|
||||||
return Array.from({ length: dim }, () => Math.random() * 2 - 1)
|
|
||||||
}
|
|
||||||
|
|
||||||
describe('SQ8 distance provider hook (setSQ8DistanceImplementation)', () => {
|
|
||||||
const q1 = quantizeSQ8([0.1, 0.5, -0.3, 0.8])
|
|
||||||
const q2 = quantizeSQ8([0.2, 0.4, -0.1, 0.7])
|
|
||||||
|
|
||||||
// Always restore the JS default so a swapped impl never leaks into other tests.
|
|
||||||
afterEach(() => setSQ8DistanceImplementation(distanceSQ8Js))
|
|
||||||
|
|
||||||
it('dispatches to the JS implementation by default (identical result)', () => {
|
|
||||||
const viaDispatcher = distanceSQ8(q1.quantized, q1.min, q1.max, q2.quantized, q2.min, q2.max)
|
|
||||||
const viaJs = distanceSQ8Js(q1.quantized, q1.min, q1.max, q2.quantized, q2.min, q2.max)
|
|
||||||
expect(viaDispatcher).toBe(viaJs)
|
|
||||||
expect(viaDispatcher).toBeGreaterThanOrEqual(0)
|
|
||||||
expect(viaDispatcher).toBeLessThanOrEqual(2)
|
|
||||||
})
|
|
||||||
|
|
||||||
it('routes through a swapped implementation with the cortex 6-arg signature', () => {
|
|
||||||
const calls: Array<[Uint8Array, number, number, Uint8Array, number, number]> = []
|
|
||||||
setSQ8DistanceImplementation((a, aMin, aMax, b, bMin, bMax) => {
|
|
||||||
calls.push([a, aMin, aMax, b, bMin, bMax])
|
|
||||||
return 0.4242
|
|
||||||
})
|
|
||||||
const result = distanceSQ8(q1.quantized, q1.min, q1.max, q2.quantized, q2.min, q2.max)
|
|
||||||
expect(result).toBe(0.4242)
|
|
||||||
expect(calls).toHaveLength(1)
|
|
||||||
// Exactly the arguments cortex's native cosineDistanceSq8(a, aMin, aMax, b, bMin, bMax) expects.
|
|
||||||
expect(calls[0][0]).toBe(q1.quantized)
|
|
||||||
expect(calls[0][1]).toBe(q1.min)
|
|
||||||
expect(calls[0][3]).toBe(q2.quantized)
|
|
||||||
expect(calls[0][5]).toBe(q2.max)
|
|
||||||
})
|
|
||||||
|
|
||||||
it('restores the JS default when set back', () => {
|
|
||||||
setSQ8DistanceImplementation(() => -999)
|
|
||||||
expect(distanceSQ8(q1.quantized, q1.min, q1.max, q2.quantized, q2.min, q2.max)).toBe(-999)
|
|
||||||
setSQ8DistanceImplementation(distanceSQ8Js)
|
|
||||||
expect(distanceSQ8(q1.quantized, q1.min, q1.max, q2.quantized, q2.min, q2.max)).toBe(
|
|
||||||
distanceSQ8Js(q1.quantized, q1.min, q1.max, q2.quantized, q2.min, q2.max)
|
|
||||||
)
|
|
||||||
})
|
|
||||||
})
|
|
||||||
|
|
||||||
// Helper: cosine distance for float32 vectors (reference implementation)
|
|
||||||
function cosineDistance(a: number[], b: number[]): number {
|
|
||||||
let dot = 0
|
|
||||||
let normA = 0
|
|
||||||
let normB = 0
|
|
||||||
for (let i = 0; i < a.length; i++) {
|
|
||||||
dot += a[i] * b[i]
|
|
||||||
normA += a[i] * a[i]
|
|
||||||
normB += b[i] * b[i]
|
|
||||||
}
|
|
||||||
const denom = Math.sqrt(normA) * Math.sqrt(normB)
|
|
||||||
if (denom === 0) return 1.0
|
|
||||||
return 1 - dot / denom
|
|
||||||
}
|
|
||||||
|
|
||||||
describe('SQ8 Quantization', () => {
|
|
||||||
// =================================================================
|
|
||||||
// 1. QUANTIZE / DEQUANTIZE ROUND-TRIP
|
|
||||||
// =================================================================
|
|
||||||
describe('quantize/dequantize round-trip', () => {
|
|
||||||
it('should round-trip a normalized vector with low error', () => {
|
|
||||||
const original = randomVector(384)
|
|
||||||
const sq8 = quantizeSQ8(original)
|
|
||||||
const restored = dequantizeSQ8(sq8.quantized, sq8.min, sq8.max)
|
|
||||||
|
|
||||||
expect(restored.length).toBe(original.length)
|
|
||||||
|
|
||||||
// Per-dimension error should be small (max 1/255 of range)
|
|
||||||
const range = sq8.max - sq8.min
|
|
||||||
const maxError = range / 255
|
|
||||||
for (let i = 0; i < original.length; i++) {
|
|
||||||
expect(Math.abs(restored[i] - original[i])).toBeLessThanOrEqual(maxError + 1e-7)
|
|
||||||
}
|
|
||||||
})
|
|
||||||
|
|
||||||
it('should handle a zero-range vector (all identical values)', () => {
|
|
||||||
const original = new Array(128).fill(0.5)
|
|
||||||
const sq8 = quantizeSQ8(original)
|
|
||||||
|
|
||||||
expect(sq8.min).toBe(0.5)
|
|
||||||
expect(sq8.max).toBe(0.5)
|
|
||||||
// All quantized values should be 128 (midpoint)
|
|
||||||
for (let i = 0; i < sq8.quantized.length; i++) {
|
|
||||||
expect(sq8.quantized[i]).toBe(128)
|
|
||||||
}
|
|
||||||
|
|
||||||
const restored = dequantizeSQ8(sq8.quantized, sq8.min, sq8.max)
|
|
||||||
for (let i = 0; i < restored.length; i++) {
|
|
||||||
expect(restored[i]).toBe(0.5)
|
|
||||||
}
|
|
||||||
})
|
|
||||||
|
|
||||||
it('should handle negative values', () => {
|
|
||||||
const original = [-1.0, -0.5, 0, 0.5, 1.0]
|
|
||||||
const sq8 = quantizeSQ8(original)
|
|
||||||
|
|
||||||
expect(sq8.min).toBe(-1.0)
|
|
||||||
expect(sq8.max).toBe(1.0)
|
|
||||||
// Min maps to 0, max maps to 255
|
|
||||||
expect(sq8.quantized[0]).toBe(0) // -1.0
|
|
||||||
expect(sq8.quantized[4]).toBe(255) // 1.0
|
|
||||||
|
|
||||||
const restored = dequantizeSQ8(sq8.quantized, sq8.min, sq8.max)
|
|
||||||
expect(Math.abs(restored[0] - (-1.0))).toBeLessThan(0.01)
|
|
||||||
expect(Math.abs(restored[4] - 1.0)).toBeLessThan(0.01)
|
|
||||||
})
|
|
||||||
|
|
||||||
it('should produce Uint8Array output in [0, 255] range', () => {
|
|
||||||
const original = randomVector(512)
|
|
||||||
const sq8 = quantizeSQ8(original)
|
|
||||||
|
|
||||||
expect(sq8.quantized).toBeInstanceOf(Uint8Array)
|
|
||||||
expect(sq8.quantized.length).toBe(512)
|
|
||||||
for (let i = 0; i < sq8.quantized.length; i++) {
|
|
||||||
expect(sq8.quantized[i]).toBeGreaterThanOrEqual(0)
|
|
||||||
expect(sq8.quantized[i]).toBeLessThanOrEqual(255)
|
|
||||||
}
|
|
||||||
})
|
|
||||||
|
|
||||||
it('should achieve 4x storage reduction (uint8 vs float32)', () => {
|
|
||||||
const dim = 384
|
|
||||||
const original = randomVector(dim)
|
|
||||||
const sq8 = quantizeSQ8(original)
|
|
||||||
|
|
||||||
const float32Size = dim * 4 // 4 bytes per float32
|
|
||||||
const sq8Size = sq8.quantized.byteLength + 8 // uint8 array + 2 floats for min/max
|
|
||||||
const ratio = float32Size / sq8Size
|
|
||||||
|
|
||||||
// Should be close to 4x (actually slightly less due to min/max overhead)
|
|
||||||
expect(ratio).toBeGreaterThan(3.9)
|
|
||||||
})
|
|
||||||
})
|
|
||||||
|
|
||||||
// =================================================================
|
|
||||||
// 2. SQ8 DISTANCE VS FLOAT32 DISTANCE
|
|
||||||
// =================================================================
|
|
||||||
describe('SQ8 distance accuracy', () => {
|
|
||||||
it('should correlate strongly with float32 cosine distance', () => {
|
|
||||||
const dim = 384
|
|
||||||
const numPairs = 100
|
|
||||||
const errors: number[] = []
|
|
||||||
|
|
||||||
for (let i = 0; i < numPairs; i++) {
|
|
||||||
const a = randomVector(dim)
|
|
||||||
const b = randomVector(dim)
|
|
||||||
|
|
||||||
const exactDist = cosineDistance(a, b)
|
|
||||||
|
|
||||||
const sq8A = quantizeSQ8(a)
|
|
||||||
const sq8B = quantizeSQ8(b)
|
|
||||||
const approxDist = distanceSQ8(
|
|
||||||
sq8A.quantized, sq8A.min, sq8A.max,
|
|
||||||
sq8B.quantized, sq8B.min, sq8B.max
|
|
||||||
)
|
|
||||||
|
|
||||||
errors.push(Math.abs(exactDist - approxDist))
|
|
||||||
}
|
|
||||||
|
|
||||||
// Mean absolute error should be very small
|
|
||||||
const meanError = errors.reduce((sum, e) => sum + e, 0) / errors.length
|
|
||||||
expect(meanError).toBeLessThan(0.02) // Less than 2% average error
|
|
||||||
|
|
||||||
// Max error should be bounded
|
|
||||||
const maxError = Math.max(...errors)
|
|
||||||
expect(maxError).toBeLessThan(0.1) // Less than 10% worst case
|
|
||||||
})
|
|
||||||
|
|
||||||
it('should return 0 for identical vectors', () => {
|
|
||||||
const v = randomVector(128)
|
|
||||||
const sq8 = quantizeSQ8(v)
|
|
||||||
const dist = distanceSQ8(
|
|
||||||
sq8.quantized, sq8.min, sq8.max,
|
|
||||||
sq8.quantized, sq8.min, sq8.max
|
|
||||||
)
|
|
||||||
expect(dist).toBeCloseTo(0, 5)
|
|
||||||
})
|
|
||||||
|
|
||||||
it('should preserve relative ordering of distances', () => {
|
|
||||||
const dim = 128
|
|
||||||
const query = randomVector(dim)
|
|
||||||
// Create a vector close to query and one far away
|
|
||||||
const close = query.map(v => v + (Math.random() * 0.1 - 0.05))
|
|
||||||
const far = randomVector(dim)
|
|
||||||
|
|
||||||
const exactClose = cosineDistance(query, close)
|
|
||||||
const exactFar = cosineDistance(query, far)
|
|
||||||
|
|
||||||
const sq8Query = quantizeSQ8(query)
|
|
||||||
const sq8Close = quantizeSQ8(close)
|
|
||||||
const sq8Far = quantizeSQ8(far)
|
|
||||||
|
|
||||||
const approxClose = distanceSQ8(
|
|
||||||
sq8Query.quantized, sq8Query.min, sq8Query.max,
|
|
||||||
sq8Close.quantized, sq8Close.min, sq8Close.max
|
|
||||||
)
|
|
||||||
const approxFar = distanceSQ8(
|
|
||||||
sq8Query.quantized, sq8Query.min, sq8Query.max,
|
|
||||||
sq8Far.quantized, sq8Far.min, sq8Far.max
|
|
||||||
)
|
|
||||||
|
|
||||||
// Relative ordering should be preserved
|
|
||||||
if (exactClose < exactFar) {
|
|
||||||
expect(approxClose).toBeLessThan(approxFar)
|
|
||||||
}
|
|
||||||
})
|
|
||||||
|
|
||||||
it('should handle zero vectors gracefully', () => {
|
|
||||||
const zero = new Array(64).fill(0)
|
|
||||||
const nonZero = randomVector(64)
|
|
||||||
|
|
||||||
const sq8Zero = quantizeSQ8(zero)
|
|
||||||
const sq8NonZero = quantizeSQ8(nonZero)
|
|
||||||
|
|
||||||
const dist = distanceSQ8(
|
|
||||||
sq8Zero.quantized, sq8Zero.min, sq8Zero.max,
|
|
||||||
sq8NonZero.quantized, sq8NonZero.min, sq8NonZero.max
|
|
||||||
)
|
|
||||||
// Zero vectors should return max distance (1.0)
|
|
||||||
expect(dist).toBeCloseTo(1.0, 1)
|
|
||||||
})
|
|
||||||
})
|
|
||||||
|
|
||||||
// =================================================================
|
|
||||||
// 3. SERIALIZATION / DESERIALIZATION
|
|
||||||
// =================================================================
|
|
||||||
describe('serialization round-trip', () => {
|
|
||||||
it('should serialize and deserialize SQ8 data with float32 precision', () => {
|
|
||||||
const original = randomVector(384)
|
|
||||||
const sq8 = quantizeSQ8(original)
|
|
||||||
const buffer = serializeSQ8(sq8)
|
|
||||||
const restored = deserializeSQ8(buffer)
|
|
||||||
|
|
||||||
// min/max are stored as float32, so some precision loss is expected
|
|
||||||
expect(restored.min).toBeCloseTo(sq8.min, 5)
|
|
||||||
expect(restored.max).toBeCloseTo(sq8.max, 5)
|
|
||||||
expect(restored.quantized.length).toBe(sq8.quantized.length)
|
|
||||||
for (let i = 0; i < sq8.quantized.length; i++) {
|
|
||||||
expect(restored.quantized[i]).toBe(sq8.quantized[i])
|
|
||||||
}
|
|
||||||
})
|
|
||||||
|
|
||||||
it('should produce compact binary format (8 + dim bytes)', () => {
|
|
||||||
const dim = 384
|
|
||||||
const sq8 = quantizeSQ8(randomVector(dim))
|
|
||||||
const buffer = serializeSQ8(sq8)
|
|
||||||
|
|
||||||
// 4 bytes for min (float32) + 4 bytes for max (float32) + dim bytes
|
|
||||||
expect(buffer.byteLength).toBe(8 + dim)
|
|
||||||
})
|
|
||||||
})
|
|
||||||
|
|
||||||
// =================================================================
|
|
||||||
// 4. HNSW SEARCH WITH QUANTIZATION
|
|
||||||
// =================================================================
|
|
||||||
describe('HNSW search with quantization', () => {
|
|
||||||
let index: JsHnswVectorIndex
|
|
||||||
let storage: MemoryStorage
|
|
||||||
const dim = 32 // Small dimension for fast tests
|
|
||||||
let ids: string[]
|
|
||||||
|
|
||||||
beforeEach(async () => {
|
|
||||||
storage = new MemoryStorage()
|
|
||||||
index = new JsHnswVectorIndex(
|
|
||||||
{
|
|
||||||
M: 8,
|
|
||||||
efConstruction: 100,
|
|
||||||
efSearch: 50,
|
|
||||||
ml: 8,
|
|
||||||
quantization: { enabled: true, rerankMultiplier: 3 }
|
|
||||||
},
|
|
||||||
euclideanDistance,
|
|
||||||
{ useParallelization: false, storage }
|
|
||||||
)
|
|
||||||
ids = []
|
|
||||||
})
|
|
||||||
|
|
||||||
it('should return correct results with quantization enabled', async () => {
|
|
||||||
// Insert entities with UUID IDs
|
|
||||||
const vectors: number[][] = []
|
|
||||||
for (let i = 0; i < 50; i++) {
|
|
||||||
const id = uuidv4()
|
|
||||||
ids.push(id)
|
|
||||||
const v = randomVector(dim)
|
|
||||||
vectors.push(v)
|
|
||||||
await index.addItem({ id, vector: v })
|
|
||||||
}
|
|
||||||
|
|
||||||
// Search with a known vector
|
|
||||||
const queryIdx = 5
|
|
||||||
const results = await index.search(vectors[queryIdx], 5)
|
|
||||||
|
|
||||||
// The exact vector should be the closest (distance ~0)
|
|
||||||
expect(results.length).toBeGreaterThan(0)
|
|
||||||
expect(results[0][0]).toBe(ids[queryIdx])
|
|
||||||
expect(results[0][1]).toBeCloseTo(0, 1)
|
|
||||||
})
|
|
||||||
|
|
||||||
it('should find the exact match as top result', async () => {
|
|
||||||
const targetId = uuidv4()
|
|
||||||
const target = randomVector(dim)
|
|
||||||
await index.addItem({ id: targetId, vector: target })
|
|
||||||
|
|
||||||
// Add noise vectors
|
|
||||||
for (let i = 0; i < 30; i++) {
|
|
||||||
await index.addItem({ id: uuidv4(), vector: randomVector(dim) })
|
|
||||||
}
|
|
||||||
|
|
||||||
const results = await index.search(target, 1)
|
|
||||||
expect(results.length).toBe(1)
|
|
||||||
expect(results[0][0]).toBe(targetId)
|
|
||||||
})
|
|
||||||
|
|
||||||
it('should respect k parameter', async () => {
|
|
||||||
for (let i = 0; i < 100; i++) {
|
|
||||||
await index.addItem({ id: uuidv4(), vector: randomVector(dim) })
|
|
||||||
}
|
|
||||||
|
|
||||||
const results5 = await index.search(randomVector(dim), 5)
|
|
||||||
const results10 = await index.search(randomVector(dim), 10)
|
|
||||||
|
|
||||||
expect(results5.length).toBe(5)
|
|
||||||
// With quantization on a small graph (100 items), HNSW may occasionally
|
|
||||||
// return slightly fewer than k due to approximation in distance calculations
|
|
||||||
expect(results10.length).toBeGreaterThanOrEqual(8)
|
|
||||||
expect(results10.length).toBeLessThanOrEqual(10)
|
|
||||||
})
|
|
||||||
})
|
|
||||||
|
|
||||||
// =================================================================
|
|
||||||
// 5. TWO-PHASE RERANK
|
|
||||||
// =================================================================
|
|
||||||
describe('two-phase rerank', () => {
|
|
||||||
it('should accept rerank options in search', async () => {
|
|
||||||
const storage = new MemoryStorage()
|
|
||||||
const index = new JsHnswVectorIndex(
|
|
||||||
{
|
|
||||||
M: 8,
|
|
||||||
efConstruction: 100,
|
|
||||||
efSearch: 50,
|
|
||||||
ml: 8,
|
|
||||||
quantization: { enabled: true, rerankMultiplier: 1 } // Disable default rerank
|
|
||||||
},
|
|
||||||
euclideanDistance,
|
|
||||||
{ useParallelization: false, storage }
|
|
||||||
)
|
|
||||||
|
|
||||||
const dim = 32
|
|
||||||
for (let i = 0; i < 30; i++) {
|
|
||||||
await index.addItem({ id: uuidv4(), vector: randomVector(dim) })
|
|
||||||
}
|
|
||||||
|
|
||||||
// Search with explicit rerank
|
|
||||||
const results = await index.search(
|
|
||||||
randomVector(dim),
|
|
||||||
5,
|
|
||||||
undefined,
|
|
||||||
{ rerank: { multiplier: 3 } }
|
|
||||||
)
|
|
||||||
|
|
||||||
expect(results.length).toBe(5)
|
|
||||||
// Results should be sorted by distance
|
|
||||||
for (let i = 1; i < results.length; i++) {
|
|
||||||
expect(results[i][1]).toBeGreaterThanOrEqual(results[i - 1][1])
|
|
||||||
}
|
|
||||||
})
|
|
||||||
|
|
||||||
it('reranked results should be sorted by exact distance', async () => {
|
|
||||||
const storage = new MemoryStorage()
|
|
||||||
const index = new JsHnswVectorIndex(
|
|
||||||
{
|
|
||||||
M: 8,
|
|
||||||
efConstruction: 100,
|
|
||||||
efSearch: 50,
|
|
||||||
ml: 8,
|
|
||||||
quantization: { enabled: true, rerankMultiplier: 3 }
|
|
||||||
},
|
|
||||||
euclideanDistance,
|
|
||||||
{ useParallelization: false, storage }
|
|
||||||
)
|
|
||||||
|
|
||||||
const dim = 32
|
|
||||||
for (let i = 0; i < 50; i++) {
|
|
||||||
await index.addItem({ id: uuidv4(), vector: randomVector(dim) })
|
|
||||||
}
|
|
||||||
|
|
||||||
const query = randomVector(dim)
|
|
||||||
const results = await index.search(query, 10)
|
|
||||||
|
|
||||||
// Results should be sorted by exact distance (after reranking)
|
|
||||||
for (let i = 1; i < results.length; i++) {
|
|
||||||
expect(results[i][1]).toBeGreaterThanOrEqual(results[i - 1][1])
|
|
||||||
}
|
|
||||||
})
|
|
||||||
})
|
|
||||||
|
|
||||||
// =================================================================
|
|
||||||
// 6. CONFIG DEFAULTS
|
|
||||||
// =================================================================
|
|
||||||
describe('configuration defaults', () => {
|
|
||||||
it('should disable quantization by default', async () => {
|
|
||||||
const storage = new MemoryStorage()
|
|
||||||
const defaultIndex = new JsHnswVectorIndex(
|
|
||||||
{ M: 4, efConstruction: 50, efSearch: 20 },
|
|
||||||
euclideanDistance,
|
|
||||||
{ useParallelization: false, storage }
|
|
||||||
)
|
|
||||||
|
|
||||||
const dim = 16
|
|
||||||
for (let i = 0; i < 10; i++) {
|
|
||||||
await defaultIndex.addItem({ id: uuidv4(), vector: randomVector(dim) })
|
|
||||||
}
|
|
||||||
|
|
||||||
// Search should work identically to pre-quantization behavior
|
|
||||||
const results = await defaultIndex.search(randomVector(dim), 5)
|
|
||||||
expect(results.length).toBe(5)
|
|
||||||
|
|
||||||
// Distances should be exact euclidean (not SQ8 approximate)
|
|
||||||
for (const [, dist] of results) {
|
|
||||||
expect(typeof dist).toBe('number')
|
|
||||||
expect(dist).toBeGreaterThanOrEqual(0)
|
|
||||||
}
|
|
||||||
})
|
|
||||||
|
|
||||||
it('should use default rerankMultiplier of 3', async () => {
|
|
||||||
const storage = new MemoryStorage()
|
|
||||||
const index = new JsHnswVectorIndex(
|
|
||||||
{
|
|
||||||
M: 4,
|
|
||||||
efConstruction: 50,
|
|
||||||
efSearch: 20,
|
|
||||||
quantization: { enabled: true }
|
|
||||||
},
|
|
||||||
euclideanDistance,
|
|
||||||
{ useParallelization: false, storage }
|
|
||||||
)
|
|
||||||
|
|
||||||
// The index should be created without errors
|
|
||||||
// Default rerankMultiplier should be 3 (tested implicitly via search)
|
|
||||||
const dim = 16
|
|
||||||
for (let i = 0; i < 10; i++) {
|
|
||||||
await index.addItem({ id: uuidv4(), vector: randomVector(dim) })
|
|
||||||
}
|
|
||||||
|
|
||||||
const results = await index.search(randomVector(dim), 3)
|
|
||||||
expect(results.length).toBe(3)
|
|
||||||
})
|
|
||||||
|
|
||||||
it('should default vectorStorage to memory mode', async () => {
|
|
||||||
const storage = new MemoryStorage()
|
|
||||||
const index = new JsHnswVectorIndex(
|
|
||||||
{ M: 4, efConstruction: 50, efSearch: 20 },
|
|
||||||
euclideanDistance,
|
|
||||||
{ useParallelization: false, storage }
|
|
||||||
)
|
|
||||||
|
|
||||||
const id = uuidv4()
|
|
||||||
const v = randomVector(16)
|
|
||||||
await index.addItem({ id, vector: v })
|
|
||||||
|
|
||||||
// In memory mode, vector should be immediately searchable
|
|
||||||
const results = await index.search(v, 1)
|
|
||||||
expect(results.length).toBe(1)
|
|
||||||
expect(results[0][0]).toBe(id)
|
|
||||||
expect(results[0][1]).toBeCloseTo(0, 5)
|
|
||||||
})
|
|
||||||
})
|
|
||||||
})
|
|
||||||
|
|
@ -117,10 +117,13 @@ describe('add() Performance Regression Tests', () => {
|
||||||
expect(storageType).toBe('memory')
|
expect(storageType).toBe('memory')
|
||||||
})
|
})
|
||||||
|
|
||||||
it('should use immediate persistence for memory storage', async () => {
|
it('should use deferred persistence for memory storage', async () => {
|
||||||
// Memory storage should use immediate mode (already fast)
|
// 8.0 adaptive default: memory storage uses 'deferred' — nothing survives
|
||||||
|
// the process anyway, so per-add persistence writes are pure overhead.
|
||||||
|
// (Filesystem storage gets 'immediate'; an explicit config.vector.persistMode
|
||||||
|
// overrides either way.)
|
||||||
const index = (brain as any).index
|
const index = (brain as any).index
|
||||||
expect(index.persistMode).toBe('immediate')
|
expect(index.persistMode).toBe('deferred')
|
||||||
})
|
})
|
||||||
})
|
})
|
||||||
})
|
})
|
||||||
|
|
|
||||||
|
|
@ -1,287 +0,0 @@
|
||||||
/**
|
|
||||||
* @module vectorQuantization-sq4.test
|
|
||||||
* @description 2.5.0 #30 — SQ4 (4-bit scalar quantization).
|
|
||||||
*
|
|
||||||
* Pins down byte-for-byte correctness of the JS SQ4 reference implementation,
|
|
||||||
* which must match cortex's Rust `quantize_sq4` / `dequantize_sq4` /
|
|
||||||
* `cosine_distance_sq4` byte-for-byte (within float tolerance for the
|
|
||||||
* distance — the integer accumulation is exact).
|
|
||||||
*
|
|
||||||
* Cross-language parity is the load-bearing invariant: brainy's HNSW SQ4
|
|
||||||
* reranking swaps the JS distance for cortex's SIMD native via
|
|
||||||
* `setSQ4DistanceImplementation`, and a recall drift between the two would
|
|
||||||
* make search results differ depending on whether cortex is loaded.
|
|
||||||
*
|
|
||||||
* Coverage:
|
|
||||||
* 1. Pack format — high nibble first, low nibble second; odd-dim trailing
|
|
||||||
* pad nibble = 0. Total packed length = `ceil(dim / 2)`.
|
|
||||||
* 2. Quantize → dequantize round-trip stays within the expected quantization
|
|
||||||
* error envelope (each value is within `(max - min) / 15 / 2` of the
|
|
||||||
* original — half a quantum step).
|
|
||||||
* 3. Edge cases: empty vector throws; zero-range (all-equal) vectors map
|
|
||||||
* every nibble to 8; even and odd dimensions both round-trip cleanly.
|
|
||||||
* 4. Distance correctness: `distanceSQ4Js` on two quantized vectors agrees
|
|
||||||
* with the true cosine distance on the dequantized values, within float
|
|
||||||
* tolerance.
|
|
||||||
* 5. Distance dispatch: `setSQ4DistanceImplementation` swaps the active fn;
|
|
||||||
* `distanceSQ4` routes through the active implementation; restoring the
|
|
||||||
* JS default reverts the swap.
|
|
||||||
* 6. Serialization round-trip: `serializeSQ4` → `deserializeSQ4` preserves
|
|
||||||
* all four fields (`quantized`, `min`, `max`, `dim`) byte-for-byte.
|
|
||||||
*/
|
|
||||||
|
|
||||||
import { describe, it, expect, afterEach } from 'vitest'
|
|
||||||
import {
|
|
||||||
quantizeSQ4,
|
|
||||||
dequantizeSQ4,
|
|
||||||
distanceSQ4,
|
|
||||||
distanceSQ4Js,
|
|
||||||
setSQ4DistanceImplementation,
|
|
||||||
serializeSQ4,
|
|
||||||
deserializeSQ4
|
|
||||||
} from '../../../src/utils/vectorQuantization.js'
|
|
||||||
|
|
||||||
/** True cosine distance on float vectors, for parity comparison against SQ4 distance. */
|
|
||||||
function cosineDistanceFloat(a: number[], b: number[]): number {
|
|
||||||
let dot = 0
|
|
||||||
let normA = 0
|
|
||||||
let normB = 0
|
|
||||||
for (let i = 0; i < a.length; i++) {
|
|
||||||
dot += a[i] * b[i]
|
|
||||||
normA += a[i] * a[i]
|
|
||||||
normB += b[i] * b[i]
|
|
||||||
}
|
|
||||||
const denom = Math.sqrt(normA) * Math.sqrt(normB)
|
|
||||||
if (denom === 0) return 1.0
|
|
||||||
return 1 - Math.max(-1, Math.min(1, dot / denom))
|
|
||||||
}
|
|
||||||
|
|
||||||
/** Build a deterministic test vector with sine-modulated values across [-1, 1]. */
|
|
||||||
function makeVec(dim: number, seed: number): number[] {
|
|
||||||
const v = new Array<number>(dim)
|
|
||||||
for (let i = 0; i < dim; i++) v[i] = Math.sin((i + seed) * 0.137)
|
|
||||||
return v
|
|
||||||
}
|
|
||||||
|
|
||||||
describe('SQ4 quantization (2.5.0 #30)', () => {
|
|
||||||
// -----------------------------------------------------------------------
|
|
||||||
// 1. Pack format — high nibble first, odd-dim trailing pad nibble = 0
|
|
||||||
// -----------------------------------------------------------------------
|
|
||||||
|
|
||||||
describe('quantizeSQ4 — pack format', () => {
|
|
||||||
it('packs two nibbles per byte, high nibble = vector[2i], low nibble = vector[2i+1]', () => {
|
|
||||||
// Even dim. Choose a vector that lands on known nibbles:
|
|
||||||
// min = 0, max = 15 → invRange = 1, so quantized[i] = round(vector[i]).
|
|
||||||
const sq4 = quantizeSQ4([0, 15, 8, 1])
|
|
||||||
expect(sq4.dim).toBe(4)
|
|
||||||
expect(sq4.min).toBe(0)
|
|
||||||
expect(sq4.max).toBe(15)
|
|
||||||
expect(sq4.quantized.length).toBe(2)
|
|
||||||
// byte 0 = (0 << 4) | 15 = 0x0f
|
|
||||||
// byte 1 = (8 << 4) | 1 = 0x81
|
|
||||||
expect(sq4.quantized[0]).toBe(0x0f)
|
|
||||||
expect(sq4.quantized[1]).toBe(0x81)
|
|
||||||
})
|
|
||||||
|
|
||||||
it('odd dim puts the final value in the high nibble and pads the low nibble with 0', () => {
|
|
||||||
const sq4 = quantizeSQ4([0, 15, 7])
|
|
||||||
expect(sq4.dim).toBe(3)
|
|
||||||
expect(sq4.quantized.length).toBe(2)
|
|
||||||
// byte 0 = (0 << 4) | 15 = 0x0f
|
|
||||||
// byte 1 = (7 << 4) | 0 = 0x70 (low nibble is pad, 0)
|
|
||||||
expect(sq4.quantized[0]).toBe(0x0f)
|
|
||||||
expect(sq4.quantized[1]).toBe(0x70)
|
|
||||||
})
|
|
||||||
|
|
||||||
it('packed length is ceil(dim / 2) for both even and odd dims', () => {
|
|
||||||
expect(quantizeSQ4([0, 1]).quantized.length).toBe(1)
|
|
||||||
expect(quantizeSQ4([0, 1, 2]).quantized.length).toBe(2)
|
|
||||||
expect(quantizeSQ4([0, 1, 2, 3]).quantized.length).toBe(2)
|
|
||||||
expect(quantizeSQ4([0, 1, 2, 3, 4]).quantized.length).toBe(3)
|
|
||||||
expect(quantizeSQ4(makeVec(384, 1)).quantized.length).toBe(192)
|
|
||||||
expect(quantizeSQ4(makeVec(385, 1)).quantized.length).toBe(193)
|
|
||||||
})
|
|
||||||
})
|
|
||||||
|
|
||||||
// -----------------------------------------------------------------------
|
|
||||||
// 2. Quantize → dequantize round-trip is within the expected error envelope
|
|
||||||
// -----------------------------------------------------------------------
|
|
||||||
|
|
||||||
describe('quantizeSQ4 / dequantizeSQ4 — round-trip error', () => {
|
|
||||||
it('every reconstructed value is within half a quantum step of the original', () => {
|
|
||||||
// The maximum reconstruction error for any value is half the quantum
|
|
||||||
// step = (max - min) / 15 / 2. The factor 2 worst case happens at the
|
|
||||||
// bin boundary; using 1.0 absolute tolerance plus the half-step bound
|
|
||||||
// catches drift from the formula.
|
|
||||||
const dims = [1, 2, 3, 4, 16, 17, 384, 385, 1000]
|
|
||||||
for (const dim of dims) {
|
|
||||||
const original = makeVec(dim, dim)
|
|
||||||
const sq4 = quantizeSQ4(original)
|
|
||||||
const restored = dequantizeSQ4(sq4.quantized, sq4.min, sq4.max, sq4.dim)
|
|
||||||
expect(restored.length).toBe(dim)
|
|
||||||
|
|
||||||
const halfStep = (sq4.max - sq4.min) / 15 / 2
|
|
||||||
for (let i = 0; i < dim; i++) {
|
|
||||||
expect(Math.abs(restored[i] - original[i])).toBeLessThanOrEqual(halfStep + 1e-9)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
})
|
|
||||||
|
|
||||||
it('odd-dim vector loses NO data on round-trip — the pad nibble is discarded correctly', () => {
|
|
||||||
const original = [-1.5, 0.0, 0.7, 1.3, 2.0]
|
|
||||||
const sq4 = quantizeSQ4(original)
|
|
||||||
const restored = dequantizeSQ4(sq4.quantized, sq4.min, sq4.max, sq4.dim)
|
|
||||||
expect(restored.length).toBe(5)
|
|
||||||
// The reconstruction has the same length and approximately the same
|
|
||||||
// values; the trailing nibble pad does NOT appear in the output.
|
|
||||||
})
|
|
||||||
})
|
|
||||||
|
|
||||||
// -----------------------------------------------------------------------
|
|
||||||
// 3. Edge cases
|
|
||||||
// -----------------------------------------------------------------------
|
|
||||||
|
|
||||||
describe('edge cases', () => {
|
|
||||||
it('zero-range vectors map every nibble to 8 (midpoint of [0, 15])', () => {
|
|
||||||
const sq4 = quantizeSQ4([0.5, 0.5, 0.5, 0.5])
|
|
||||||
expect(sq4.min).toBe(0.5)
|
|
||||||
expect(sq4.max).toBe(0.5)
|
|
||||||
expect(Array.from(sq4.quantized)).toEqual([0x88, 0x88])
|
|
||||||
// Dequantize: range = 0, all values revert to min.
|
|
||||||
const restored = dequantizeSQ4(sq4.quantized, sq4.min, sq4.max, sq4.dim)
|
|
||||||
expect(restored).toEqual([0.5, 0.5, 0.5, 0.5])
|
|
||||||
})
|
|
||||||
|
|
||||||
it('throws on empty input', () => {
|
|
||||||
expect(() => quantizeSQ4([])).toThrow(/empty/i)
|
|
||||||
})
|
|
||||||
|
|
||||||
it('clamps out-of-range integers to [0, 15] (defensive)', () => {
|
|
||||||
// A single-value vector with max=min handled by zero-range branch;
|
|
||||||
// here we use a wide range and confirm extremes land at the boundaries.
|
|
||||||
const sq4 = quantizeSQ4([-100, 100, 0])
|
|
||||||
// After mapping: min=-100, max=100, range=200. Quantized:
|
|
||||||
// -100 → round(0 * 15) = 0
|
|
||||||
// 100 → round(1 * 15) = 15
|
|
||||||
// 0 → round(0.5 * 15) = 8
|
|
||||||
// Packed: byte 0 = (0 << 4) | 15 = 0x0f, byte 1 = (8 << 4) | 0 = 0x80
|
|
||||||
expect(Array.from(sq4.quantized)).toEqual([0x0f, 0x80])
|
|
||||||
})
|
|
||||||
})
|
|
||||||
|
|
||||||
// -----------------------------------------------------------------------
|
|
||||||
// 4. Distance correctness — matches true cosine on dequantized values
|
|
||||||
// -----------------------------------------------------------------------
|
|
||||||
|
|
||||||
describe('distanceSQ4Js — agreement with the dequantize-then-cosine baseline', () => {
|
|
||||||
it('matches the true cosine distance on dequantized vectors within quantization error', () => {
|
|
||||||
// For each pair, compute both the SQ4-direct distance and the
|
|
||||||
// dequantize-then-cosine distance. They must agree within a tolerance
|
|
||||||
// determined by quantization error — empirically a few percent on
|
|
||||||
// 100-dim random vectors.
|
|
||||||
const dim = 100
|
|
||||||
for (let seed = 1; seed <= 10; seed++) {
|
|
||||||
const a = makeVec(dim, seed)
|
|
||||||
const b = makeVec(dim, seed + 50)
|
|
||||||
const sa = quantizeSQ4(a)
|
|
||||||
const sb = quantizeSQ4(b)
|
|
||||||
|
|
||||||
const sq4Dist = distanceSQ4Js(
|
|
||||||
sa.quantized, sa.min, sa.max, sa.dim,
|
|
||||||
sb.quantized, sb.min, sb.max, sb.dim
|
|
||||||
)
|
|
||||||
|
|
||||||
const reA = dequantizeSQ4(sa.quantized, sa.min, sa.max, sa.dim)
|
|
||||||
const reB = dequantizeSQ4(sb.quantized, sb.min, sb.max, sb.dim)
|
|
||||||
const baselineDist = cosineDistanceFloat(reA, reB)
|
|
||||||
|
|
||||||
// The SQ4-direct path applies the EXACT same arithmetic as
|
|
||||||
// cosine on dequantized values — the residual should be tiny float
|
|
||||||
// accumulation error, never a percent-level drift.
|
|
||||||
expect(Math.abs(sq4Dist - baselineDist)).toBeLessThan(1e-9)
|
|
||||||
}
|
|
||||||
})
|
|
||||||
|
|
||||||
it('throws on dimension mismatch', () => {
|
|
||||||
const a = quantizeSQ4([0, 1, 2, 3])
|
|
||||||
const b = quantizeSQ4([0, 1])
|
|
||||||
expect(() => distanceSQ4Js(
|
|
||||||
a.quantized, a.min, a.max, a.dim,
|
|
||||||
b.quantized, b.min, b.max, b.dim
|
|
||||||
)).toThrow(/dimension/i)
|
|
||||||
})
|
|
||||||
|
|
||||||
it('returns 1.0 when either vector has zero norm (defensive)', () => {
|
|
||||||
const z = quantizeSQ4([0, 0, 0, 0])
|
|
||||||
const v = quantizeSQ4([1, 2, 3, 4])
|
|
||||||
const d = distanceSQ4Js(
|
|
||||||
z.quantized, z.min, z.max, z.dim,
|
|
||||||
v.quantized, v.min, v.max, v.dim
|
|
||||||
)
|
|
||||||
// Zero-vector has range 0 → all-8 packed. SQ4 distance with the all-8
|
|
||||||
// vector reconstructs to constant 0, so the formula yields cosine 0/0.
|
|
||||||
// The reference impl returns 1.0 (maximum distance) for the degenerate case.
|
|
||||||
expect(d).toBeGreaterThanOrEqual(0)
|
|
||||||
expect(d).toBeLessThanOrEqual(2)
|
|
||||||
})
|
|
||||||
})
|
|
||||||
|
|
||||||
// -----------------------------------------------------------------------
|
|
||||||
// 5. Distance dispatch — setSQ4DistanceImplementation hot-swap
|
|
||||||
// -----------------------------------------------------------------------
|
|
||||||
|
|
||||||
describe('distance dispatch + setSQ4DistanceImplementation', () => {
|
|
||||||
afterEach(() => {
|
|
||||||
// Restore default after any swap.
|
|
||||||
setSQ4DistanceImplementation(distanceSQ4Js)
|
|
||||||
})
|
|
||||||
|
|
||||||
it('routes through the active implementation', () => {
|
|
||||||
const a = quantizeSQ4(makeVec(64, 1))
|
|
||||||
const b = quantizeSQ4(makeVec(64, 2))
|
|
||||||
const baseline = distanceSQ4Js(a.quantized, a.min, a.max, a.dim, b.quantized, b.min, b.max, b.dim)
|
|
||||||
const viaActive = distanceSQ4(a.quantized, a.min, a.max, a.dim, b.quantized, b.min, b.max, b.dim)
|
|
||||||
expect(viaActive).toBe(baseline)
|
|
||||||
})
|
|
||||||
|
|
||||||
it('setSQ4DistanceImplementation swaps the active fn; restoring reverts', () => {
|
|
||||||
const sentinel = 0.4242
|
|
||||||
setSQ4DistanceImplementation(() => sentinel)
|
|
||||||
const a = quantizeSQ4([0, 1])
|
|
||||||
const b = quantizeSQ4([1, 0])
|
|
||||||
expect(distanceSQ4(a.quantized, a.min, a.max, a.dim, b.quantized, b.min, b.max, b.dim)).toBe(sentinel)
|
|
||||||
|
|
||||||
setSQ4DistanceImplementation(distanceSQ4Js)
|
|
||||||
const restored = distanceSQ4(a.quantized, a.min, a.max, a.dim, b.quantized, b.min, b.max, b.dim)
|
|
||||||
expect(restored).not.toBe(sentinel)
|
|
||||||
})
|
|
||||||
})
|
|
||||||
|
|
||||||
// -----------------------------------------------------------------------
|
|
||||||
// 6. Serialization round-trip
|
|
||||||
// -----------------------------------------------------------------------
|
|
||||||
|
|
||||||
describe('serialize / deserialize round-trip', () => {
|
|
||||||
it('preserves quantized bytes, min, max, and dim byte-for-byte', () => {
|
|
||||||
const original = quantizeSQ4(makeVec(384, 7))
|
|
||||||
const buf = serializeSQ4(original)
|
|
||||||
// 12-byte header (min:f32 + max:f32 + dim:u32 LE) + 192 packed bytes.
|
|
||||||
expect(buf.byteLength).toBe(12 + 192)
|
|
||||||
|
|
||||||
const restored = deserializeSQ4(buf)
|
|
||||||
expect(restored.dim).toBe(original.dim)
|
|
||||||
// float32 round-trip on min/max may lose a few low bits (we stored as f32).
|
|
||||||
expect(Math.abs(restored.min - original.min)).toBeLessThan(1e-5)
|
|
||||||
expect(Math.abs(restored.max - original.max)).toBeLessThan(1e-5)
|
|
||||||
expect(Array.from(restored.quantized)).toEqual(Array.from(original.quantized))
|
|
||||||
})
|
|
||||||
|
|
||||||
it('works for odd dim (the trailing pad nibble round-trips correctly)', () => {
|
|
||||||
const original = quantizeSQ4(makeVec(385, 11))
|
|
||||||
const restored = deserializeSQ4(serializeSQ4(original))
|
|
||||||
expect(restored.dim).toBe(385)
|
|
||||||
expect(restored.quantized.length).toBe(193)
|
|
||||||
expect(Array.from(restored.quantized)).toEqual(Array.from(original.quantized))
|
|
||||||
})
|
|
||||||
})
|
|
||||||
})
|
|
||||||
Loading…
Add table
Add a link
Reference in a new issue