feat(8.0): zero-config finalize + cut JS quantization (config.vector = recall + persistMode)
This commit is contained in:
parent
f4c5d9749f
commit
f8e0079d3f
12 changed files with 348 additions and 1739 deletions
|
|
@ -14,8 +14,6 @@ import { euclideanDistance, calculateDistancesBatch } from '../utils/index.js'
|
|||
import type { BaseStorage } from '../storage/baseStorage.js'
|
||||
import { getGlobalCache, UnifiedCache } from '../utils/unifiedCache.js'
|
||||
import { prodLog } from '../utils/logger.js'
|
||||
import { quantizeSQ8, distanceSQ8 } from '../utils/vectorQuantization.js'
|
||||
import type { SQ8QuantizedVector } from '../utils/vectorQuantization.js'
|
||||
import type { VectorIndexProvider } from '../plugin.js'
|
||||
import { ConnectionsCodec, compressedConnectionsKey } from './connectionsCodec.js'
|
||||
|
||||
|
|
@ -67,10 +65,8 @@ export class JsHnswVectorIndex implements VectorIndexProvider {
|
|||
private dirtyNodes: Set<string> = new Set() // Nodes with unpersisted HNSW data
|
||||
private dirtySystem: boolean = false // Whether system data (entryPoint, maxLevel) needs persist
|
||||
|
||||
// SQ8 quantization support (B1 optimization)
|
||||
private quantizationEnabled: boolean = false
|
||||
private rerankMultiplier: number = 3
|
||||
// Lazy vector storage (B2 optimization)
|
||||
// Lazy vector storage (B2 optimization): evict the float32 vector to
|
||||
// storage after insert; reload on demand via getVectorSafe() + UnifiedCache.
|
||||
private vectorStorageMode: 'memory' | 'lazy' = 'memory'
|
||||
|
||||
constructor(
|
||||
|
|
@ -87,12 +83,6 @@ export class JsHnswVectorIndex implements VectorIndexProvider {
|
|||
this.storage = options.storage || null
|
||||
this.persistMode = options.persistMode || 'immediate'
|
||||
|
||||
// SQ8 quantization config (default: disabled, preserves current behavior)
|
||||
if (config.quantization?.enabled) {
|
||||
this.quantizationEnabled = true
|
||||
this.rerankMultiplier = config.quantization.rerankMultiplier ?? 3
|
||||
}
|
||||
|
||||
// Vector storage mode (default: 'memory', preserves current behavior)
|
||||
this.vectorStorageMode = config.vectorStorage || 'memory'
|
||||
|
||||
|
|
@ -376,7 +366,7 @@ export class JsHnswVectorIndex implements VectorIndexProvider {
|
|||
// Generate random level for this noun
|
||||
const nounLevel = this.getRandomLevel()
|
||||
|
||||
// Create new noun with optional SQ8 quantization
|
||||
// Create new noun
|
||||
const noun: HNSWNoun = {
|
||||
id,
|
||||
vector,
|
||||
|
|
@ -384,14 +374,6 @@ export class JsHnswVectorIndex implements VectorIndexProvider {
|
|||
level: nounLevel
|
||||
}
|
||||
|
||||
// Quantize vector if enabled (B1: 4x storage reduction)
|
||||
if (this.quantizationEnabled) {
|
||||
const sq8 = quantizeSQ8(vector)
|
||||
noun.quantizedVector = sq8.quantized
|
||||
noun.codebookMin = sq8.min
|
||||
noun.codebookMax = sq8.max
|
||||
}
|
||||
|
||||
// Initialize empty connection sets for each level
|
||||
for (let level = 0; level <= nounLevel; level++) {
|
||||
noun.connections.set(level, new Set<string>())
|
||||
|
|
@ -664,16 +646,18 @@ export class JsHnswVectorIndex implements VectorIndexProvider {
|
|||
}
|
||||
|
||||
/**
|
||||
* Search for nearest neighbors
|
||||
* Search for nearest neighbors.
|
||||
*
|
||||
* When SQ8 quantization is enabled and reranking is active:
|
||||
* - Phase 1: Over-retrieve k*rerankMultiplier candidates using SQ8 approximate distances
|
||||
* - Phase 2: Load full float32 vectors for candidates, compute exact distances, return top k
|
||||
* The JS HNSW index computes exact float32 distances throughout — there is
|
||||
* no approximate/rerank path. The `options.rerank` field exists only for the
|
||||
* shared {@link VectorIndexProvider} contract (a native provider may use it
|
||||
* for its own approximate-then-rerank pipeline); the JS index ignores it.
|
||||
*
|
||||
* @param queryVector Query vector
|
||||
* @param k Number of results to return
|
||||
* @param filter Optional filter function
|
||||
* @param options Additional search options
|
||||
* @param options Additional search options (`candidateIds` to restrict the
|
||||
* search to a pre-filtered set; `rerank` is provider-only, ignored here)
|
||||
*/
|
||||
public async search(
|
||||
queryVector: Vector,
|
||||
|
|
@ -737,11 +721,6 @@ export class JsHnswVectorIndex implements VectorIndexProvider {
|
|||
|
||||
let currObj = entryPoint
|
||||
|
||||
// SQ8: Pre-quantize query vector for fast approximate distances during traversal
|
||||
if (this.quantizationEnabled) {
|
||||
this._querySQ8 = quantizeSQ8(queryVector)
|
||||
}
|
||||
|
||||
// OPTIMIZATION: Preload entry point vector
|
||||
await this.preloadVectors([entryPoint.id])
|
||||
|
||||
|
|
@ -809,14 +788,9 @@ export class JsHnswVectorIndex implements VectorIndexProvider {
|
|||
}
|
||||
}
|
||||
|
||||
// Determine effective rerank multiplier
|
||||
const rerankActive = this.quantizationEnabled && (options?.rerank || this.rerankMultiplier > 1)
|
||||
const multiplier = options?.rerank?.multiplier ?? this.rerankMultiplier
|
||||
const effectiveK = rerankActive ? k * multiplier : k
|
||||
|
||||
// Search at level 0 with ef = effectiveK
|
||||
// If we have a filter, increase ef to compensate for filtered results
|
||||
const ef = filter ? Math.max(this.config.efSearch * 3, effectiveK * 3) : Math.max(this.config.efSearch, effectiveK)
|
||||
// Search at level 0. If we have a filter, increase ef to compensate for
|
||||
// filtered-out results.
|
||||
const ef = filter ? Math.max(this.config.efSearch * 3, k * 3) : Math.max(this.config.efSearch, k)
|
||||
const nearestNouns = await this.searchLayer(
|
||||
queryVector,
|
||||
currObj,
|
||||
|
|
@ -825,36 +799,7 @@ export class JsHnswVectorIndex implements VectorIndexProvider {
|
|||
filter
|
||||
)
|
||||
|
||||
// Phase 2: Rerank with exact float32 distances if quantization is active (B3)
|
||||
if (rerankActive && nearestNouns.size > 0) {
|
||||
// Clear SQ8 cache before reranking (we need exact distances now)
|
||||
this._querySQ8 = null
|
||||
|
||||
const candidates = [...nearestNouns].slice(0, effectiveK)
|
||||
|
||||
// Load full float32 vectors for the candidate set
|
||||
const candidateIds = candidates.map(([id]) => id)
|
||||
await this.preloadVectors(candidateIds)
|
||||
|
||||
// Recompute exact distances
|
||||
const reranked: Array<[string, number]> = []
|
||||
for (const [id] of candidates) {
|
||||
const noun = this.nouns.get(id)
|
||||
if (!noun) continue
|
||||
const exactVector = await this.getVectorSafe(noun)
|
||||
const exactDist = this.distanceFunction(queryVector, exactVector)
|
||||
reranked.push([id, exactDist])
|
||||
}
|
||||
|
||||
// Sort by exact distance and return top k
|
||||
reranked.sort((a, b) => a[1] - b[1])
|
||||
return reranked.slice(0, k)
|
||||
}
|
||||
|
||||
// Clear SQ8 cache
|
||||
this._querySQ8 = null
|
||||
|
||||
// Convert to array and sort by distance
|
||||
// Exact float32 distances throughout — return the top k directly.
|
||||
return [...nearestNouns].slice(0, k)
|
||||
}
|
||||
|
||||
|
|
@ -1148,19 +1093,6 @@ export class JsHnswVectorIndex implements VectorIndexProvider {
|
|||
* @returns number | Promise<number> - sync when cached, async when needs load
|
||||
*/
|
||||
private distanceSafe(queryVector: Vector, noun: HNSWNoun): number | Promise<number> {
|
||||
// SQ8 fast path: use quantized distance when available (B1 optimization)
|
||||
// This avoids loading full float32 vectors during graph traversal
|
||||
if (this.quantizationEnabled &&
|
||||
noun.quantizedVector &&
|
||||
noun.codebookMin !== undefined &&
|
||||
noun.codebookMax !== undefined &&
|
||||
this._querySQ8) {
|
||||
return distanceSQ8(
|
||||
this._querySQ8.quantized, this._querySQ8.min, this._querySQ8.max,
|
||||
noun.quantizedVector, noun.codebookMin, noun.codebookMax
|
||||
)
|
||||
}
|
||||
|
||||
// Try sync fast path
|
||||
const nounVector = this.getVectorSync(noun)
|
||||
|
||||
|
|
@ -1175,9 +1107,6 @@ export class JsHnswVectorIndex implements VectorIndexProvider {
|
|||
)
|
||||
}
|
||||
|
||||
// Cached SQ8 quantization of the current query vector for distanceSafe fast path
|
||||
private _querySQ8: SQ8QuantizedVector | null = null
|
||||
|
||||
/**
|
||||
* Get all nodes at a specific level for clustering
|
||||
* This enables O(n) clustering using HNSW's natural hierarchy
|
||||
|
|
@ -1299,14 +1228,6 @@ export class JsHnswVectorIndex implements VectorIndexProvider {
|
|||
level: hnswData.level
|
||||
}
|
||||
|
||||
// Restore SQ8 quantized data if quantization is enabled
|
||||
if (this.quantizationEnabled && nounData.vector.length > 0) {
|
||||
const sq8 = quantizeSQ8(nounData.vector)
|
||||
noun.quantizedVector = sq8.quantized
|
||||
noun.codebookMin = sq8.min
|
||||
noun.codebookMax = sq8.max
|
||||
}
|
||||
|
||||
// Restore connections from persisted data — compressed blob path
|
||||
// first, legacy JSON-array fallback for indexes written before
|
||||
// graph link compression landed.
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue