The __words__ keyword index stores 50-5000 entries per entity (one per word), which inflated avg entries/entity well above the corruption threshold of 100. This caused: 1. validateConsistency() to falsely detect corruption on every startup, triggering unnecessary clearAllIndexData() + rebuild() cycles 2. getStats() to log false "Metadata index may be corrupted" warnings and report inflated totalEntries/totalIds stats Both methods now skip __words__ when counting, so stats and health checks reflect metadata fields only (noun, type, createdAt, etc.). Keyword search is unaffected since the __words__ field index itself is not modified.
31 lines
1.1 KiB
TypeScript
31 lines
1.1 KiB
TypeScript
/**
|
|
* Brainy Setup - Minimal Polyfills
|
|
*
|
|
* ARCHITECTURE:
|
|
* Brainy uses Candle WASM (Rust-based) for embeddings.
|
|
* No transformers.js or ONNX Runtime dependency, no hacks required.
|
|
*
|
|
* This file provides minimal polyfills for cross-environment compatibility:
|
|
* - TextEncoder/TextDecoder for older environments
|
|
*
|
|
* BENEFITS:
|
|
* - Clean codebase with no workarounds
|
|
* - Works everywhere: Node.js, Bun, Bun --compile, browsers, Deno
|
|
* - No platform-specific binaries
|
|
* - Model bundled in package (no runtime downloads)
|
|
*/
|
|
|
|
// ============================================================================
|
|
// TextEncoder/TextDecoder Polyfills
|
|
// ============================================================================
|
|
const globalObj = globalThis ?? global ?? self
|
|
|
|
if (globalObj) {
|
|
if (!globalObj.TextEncoder) globalObj.TextEncoder = TextEncoder
|
|
if (!globalObj.TextDecoder) globalObj.TextDecoder = TextDecoder
|
|
;(globalObj as any).__TextEncoder__ = TextEncoder
|
|
;(globalObj as any).__TextDecoder__ = TextDecoder
|
|
}
|
|
|
|
import { applyTensorFlowPatch } from './utils/textEncoding.js'
|
|
applyTensorFlowPatch()
|