open-brainy/src/utils/paramValidation.ts
David Snelling 0de7665930
Some checks failed
CI / Node 22 (push) Successful in 12m22s
CI / Node 24 (push) Successful in 12m15s
CI / Integration + conformance (Node 22) (push) Failing after 14m49s
CI / Bun (latest) (push) Successful in 12m28s
fix(vectors): a zero-norm vector is not a vector, canonical side included, plus the sanctioned unvector door
The engine-pair seam law: a zero-norm vector never crosses an engine
boundary. The index belt already refused to insert one, but the canonical
write and the vectored-noun ledger still counted it, so a near-empty store
whose only vectored row was zero-norm read "1 canonical vectored vs 0
indexed" and threw a not-ready error at open, and a store's own legacy
zero-norm VFS root could trip the same gate before its VFS-init-time cure
ever ran.

- add()/update() (single and transact()) now normalize an explicit
  real all-zero vector to the unvectored [] shape before the dimension
  pin, the ledger flag, and the index ops ever see it (loud, one warn per
  write, canonical write still succeeds).
- The legacy counts.json derivation walk (scanVectoredNounCount) excludes
  a persisted zero-norm row, matching the live ledger's definition.
- A legacy zero-norm VFS root now migrates at open, before the vector-leg
  gate evaluates, via one O(1) fixed-path read (torn-tolerant — skips
  rather than aborting init on a torn root, letting the recovery walk
  heal it) — independent of whether a VirtualFileSystem is ever
  constructed this session.
- update({ id, vector: [] }) (and the same op inside transact()) is now
  the sanctioned, idempotent unvector door: index removal, exactly-once
  ledger decrement, no re-embed, and it clears a pending deferred-embed
  marker rather than leaving it to re-vectorize the row later. The
  combination with deferEmbedding is a typed refusal.
- JsHnswVectorIndex.rebuild() now skips a zero-norm/empty persisted
  vector when repopulating from canonical (the same belt the live
  add/replace paths already had), and health()'s index-parity check now
  compares HNSW size against the vectored-noun ledger rather than the
  raw metadata-entry count, since a store's VFS root is permanently
  unvectored by design.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-08-27 13:53:09 -07:00

782 lines
No EOL
29 KiB
TypeScript
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

/**
* Zero-Config Parameter Validation
*
* Self-configuring validation that adapts to system capabilities
* Only enforces universal truths, learns everything else
*/
import { FindParams, AddParams, UpdateParams, RelateParams, UpdateRelationParams } from '../types/brainy.types.js'
import { NounType, VerbType } from '../types/graphTypes.js'
import { prodLog } from './logger.js'
import { findCallerLocation } from './callerLocation.js'
// 8.0 is Node/Bun/Deno-only (no browser path), so `node:os` / `node:fs` are
// always present — static ESM imports instead of a top-level `await import()`.
// The TLA form poisoned the module graph with a top-level await, which `bun
// build --compile` refuses; the static form also drops the browser/edge
// fallback branches that no supported runtime can reach.
import * as os from 'node:os'
import * as fs from 'node:fs'
import { parseFieldAddress, UnsupportedFindOptionError } from '../db/fieldAddressing.js'
const getSystemMemory = (): number => {
if (os) {
return os.totalmem()
}
return 4 * 1024 * 1024 * 1024
}
/**
* Best-effort "memory we can actually use right now", in bytes.
*
* Prefers `/proc/meminfo` **MemAvailable**, which the kernel computes as the
* memory obtainable for a new workload *including reclaimable page cache*. This
* matters on hosts that memory-map large index files: the cache holding those
* mmap'd pages is instantly reclaimable, yet `os.freemem()` reports only
* **MemFree** (the unused sliver), which collapses to near-zero on a warm box.
* Driving the auto query-cap off MemFree made the cap crater to its floor on
* perfectly healthy machines (BRAINY-QUERYCAP-MISREAD); MemAvailable fixes that.
*
* Falls back to `os.freemem()` where `/proc/meminfo` is absent (non-Linux), then
* to a conservative 2 GB when no OS module is available at all (browser/edge).
*
* @returns Usable memory in bytes.
*/
const getAvailableMemory = (): number => {
if (fs) {
try {
const meminfo = fs.readFileSync('/proc/meminfo', 'utf8') as string
const match = meminfo.match(/^MemAvailable:\s+(\d+)\s*kB/m)
if (match) {
const bytes = parseInt(match[1], 10) * 1024
if (bytes > 0) {
return bytes
}
}
} catch {
// /proc/meminfo unreadable (non-Linux, sandboxed) — fall through.
}
}
if (os) {
return os.freemem()
}
return 2 * 1024 * 1024 * 1024
}
/**
* Detect container memory limit (Docker/Kubernetes/Cloud Run)
*
* Production-grade detection for containerized environments.
* Supports:
* - cgroup v1 (legacy Docker/K8s)
* - cgroup v2 (modern systems)
* - Environment variables (Cloud Run, GCP, AWS, Azure)
*
* @returns Container memory limit in bytes, or null if not containerized
*/
const getContainerMemoryLimit = (): number | null => {
// Not in Node.js environment
if (!fs) {
return null
}
try {
// 1. Check environment variables first (fastest, most reliable for Cloud Run)
// Google Cloud Run
if (process.env.CLOUD_RUN_MEMORY) {
// Format: "512Mi", "1Gi", "2Gi", "4Gi"
const match = process.env.CLOUD_RUN_MEMORY.match(/^(\d+)(Mi|Gi)$/)
if (match) {
const value = parseInt(match[1])
const unit = match[2]
return unit === 'Gi' ? value * 1024 * 1024 * 1024 : value * 1024 * 1024
}
}
// Generic MEMORY_LIMIT env var (bytes)
if (process.env.MEMORY_LIMIT) {
const limit = parseInt(process.env.MEMORY_LIMIT)
if (!isNaN(limit) && limit > 0) {
return limit
}
}
// 2. Check cgroup v2 (modern Docker/K8s)
try {
const cgroupV2Path = '/sys/fs/cgroup/memory.max'
const cgroupV2Content = fs.readFileSync(cgroupV2Path, 'utf8').trim()
// "max" means no limit, otherwise it's bytes
if (cgroupV2Content !== 'max') {
const limit = parseInt(cgroupV2Content)
if (!isNaN(limit) && limit > 0) {
return limit
}
}
} catch (e) {
// cgroup v2 not available, try v1
}
// 3. Check cgroup v1 (legacy Docker/K8s)
try {
const cgroupV1Path = '/sys/fs/cgroup/memory/memory.limit_in_bytes'
const cgroupV1Content = fs.readFileSync(cgroupV1Path, 'utf8').trim()
const limit = parseInt(cgroupV1Content)
// Very large values (> 1 PB) indicate no limit
const ONE_PETABYTE = 1024 * 1024 * 1024 * 1024 * 1024
if (!isNaN(limit) && limit > 0 && limit < ONE_PETABYTE) {
return limit
}
} catch (e) {
// cgroup v1 not available
}
// Not containerized or no limit set
return null
} catch (e) {
// Error reading cgroup files
return null
}
}
/**
* Memory budget per query result, in KB. Used by the auto-configured
* `maxLimit` formula across all three memory-derived priorities (reserved /
* container / free).
*
* Calibration history:
* - Pre-7.30.2: `100` (assumed 100 KB per result). Way too conservative —
* actual entity footprint in Brainy is 7-10 KB (384-dim float32 vector ≈
* 1.5 KB + standard fields + metadata). On a 900 MB free-memory box this
* capped `limit` at 9_000, breaking common safety-cap call patterns like
* `find({ type, where, limit: 10_000 })` that typically return 10-500
* entities. Reported by a production consumer whose 10K safety-cap
* queries started failing after the cap landed.
* - 7.30.2+: `25` (assumes 25 KB per result). Generous over typical (7-10 KB),
* comfortably under the worst case (~20 KB with large metadata blobs).
* Same 900 MB box now gives ~36_000 — typical 10_000 limits pass silently,
* the warning tier surfaces around 36-72 K (encouraging pagination), and
* the throw tier still fires before OOM territory (72 K+).
*/
const MAX_LIMIT_KB_PER_RESULT = 25
/**
* Floor for *auto-detected* query caps (the container- and free-memory tiers).
*
* Memory probing is a heuristic; a transiently low reading must never silently
* throttle legitimate queries to a near-useless ceiling. 10 000 is a generous
* working limit that still sits far below any OOM danger (10 000 × 25 KB ≈
* 250 MB worst case). It is a floor, not an override: consumer-supplied
* `maxQueryLimit` / `reservedQueryMemory` bypass it entirely (Priorities 12),
* because an explicit caller knows their box better than the probe does.
*/
const MIN_AUTO_QUERY_LIMIT = 10000
/**
* One-time-per-call-site warning dedup. Keyed on the caller location returned
* by `findCallerLocation()` plus the exceeding limit value so the warning fires
* once per offending source line (not once per query). Survives the lifetime of
* the process — that's intentional: the warning is a teaching signal, not a
* recurring nag. Used by the two-tier limit enforcement in `validateFindParams`.
*/
const seenLimitWarnings = new Set<string>()
/**
* Reset the limit-warning dedup. For tests that need a clean slate; production
* code should never call this.
*/
export function resetLimitWarningCache(): void {
seenLimitWarnings.clear()
}
/**
* Configuration options for ValidationConfig
*/
export interface ValidationConfigOptions {
/**
* Explicit maximum query limit override
* Bypasses all auto-detection
*/
maxQueryLimit?: number
/**
* Memory reserved for query operations (in bytes)
* Bypasses auto-detection but still applies safety limits
*/
reservedQueryMemory?: number
}
/**
* Auto-configured limits based on system resources.
* Derived from memory (explicit overrides > reserved memory > container limit >
* free memory). Query timing is recorded for diagnostics only — it never
* changes the cap (see `recordQuery`).
*/
export class ValidationConfig {
private static instance: ValidationConfig | null = null
// Dynamic limits based on system
public maxLimit: number
public maxQueryLength: number
public maxVectorDimensions: number
// Tracking for diagnostics
public limitBasis: 'override' | 'reservedMemory' | 'containerMemory' | 'freeMemory'
public detectedContainerLimit: number | null
// Performance observations
private avgQueryTime: number = 0
private queryCount: number = 0
private constructor(options?: ValidationConfigOptions) {
// Vector dimensions (standard for all-MiniLM-L6-v2)
this.maxVectorDimensions = 384
// Detect container memory limit
this.detectedContainerLimit = getContainerMemoryLimit()
// Priority 1: Explicit override (highest priority)
if (options?.maxQueryLimit !== undefined) {
this.maxLimit = Math.min(options.maxQueryLimit, 100000) // Still cap at 100k for safety
this.limitBasis = 'override'
// Scale query length with limit
this.maxQueryLength = Math.min(50000, this.maxLimit * 5)
return
}
// Priority 2: Reserved memory specified
if (options?.reservedQueryMemory !== undefined) {
this.maxLimit = Math.min(
100000,
Math.floor(options.reservedQueryMemory / (1024 * 1024 * MAX_LIMIT_KB_PER_RESULT)) * 1000
)
this.limitBasis = 'reservedMemory'
this.maxQueryLength = Math.min(
50000,
Math.floor(options.reservedQueryMemory / (1024 * 1024 * 10)) * 1000
)
return
}
// Priority 3: Container detected (smart containerized behavior)
if (this.detectedContainerLimit) {
// In containers, assume 75% used by graph data (EXPECTED)
// Reserve 25% for query operations
const queryMemory = this.detectedContainerLimit * 0.25
this.maxLimit = Math.max(
MIN_AUTO_QUERY_LIMIT,
Math.min(
100000,
Math.floor(queryMemory / (1024 * 1024 * MAX_LIMIT_KB_PER_RESULT)) * 1000
)
)
this.limitBasis = 'containerMemory'
this.maxQueryLength = Math.min(
50000,
Math.floor(queryMemory / (1024 * 1024 * 10)) * 1000
)
return
}
// Priority 4: Free memory (fallback, current behavior)
const availableMemory = getAvailableMemory()
this.maxLimit = Math.max(
MIN_AUTO_QUERY_LIMIT,
Math.min(
100000,
Math.floor(availableMemory / (1024 * 1024 * MAX_LIMIT_KB_PER_RESULT)) * 1000
)
)
this.limitBasis = 'freeMemory'
this.maxQueryLength = Math.min(
50000,
Math.floor(availableMemory / (1024 * 1024 * 10)) * 1000
)
}
static getInstance(options?: ValidationConfigOptions): ValidationConfig {
if (!ValidationConfig.instance) {
ValidationConfig.instance = new ValidationConfig(options)
}
return ValidationConfig.instance
}
/**
* Reset singleton (for testing or reconfiguration)
*/
static reset(): void {
ValidationConfig.instance = null
}
/**
* Reconfigure with new options
*/
static reconfigure(options: ValidationConfigOptions): ValidationConfig {
ValidationConfig.instance = new ValidationConfig(options)
return ValidationConfig.instance
}
/**
* Record query timing for diagnostics. Telemetry ONLY — never mutates the cap.
*
* `maxLimit` is a MEMORY-protection bound; query duration says nothing about
* memory-per-result, so duration must never drive it. An earlier version
* "learned" here: while the lifetime-average query time exceeded 1s it shrank
* `maxLimit` by 20% per recorded query down to a floor of 1000 — below the
* documented `MIN_AUTO_QUERY_LIMIT` (10 000) and with no recovery once slow
* samples poisoned the cumulative average. On a production host a burst of
* slow aggregate queries silently strangled every consumer's `find()` to
* 1000 while the error message blamed "available free memory" — exactly the
* silent throttling this module's own contract forbids. The cap now comes
* from its construction-time basis (or explicit overrides) alone.
*/
recordQuery(duration: number, resultCount: number) {
void resultCount
this.queryCount++
this.avgQueryTime = (this.avgQueryTime * (this.queryCount - 1) + duration) / this.queryCount
}
}
/**
* Two-tier limit enforcement for `find({ limit })`. Compares the requested
* limit against the auto-configured (or consumer-overridden) `maxLimit` and
* picks one of three outcomes:
*
* 1. **Pass** — `limit <= maxLimit`. The configured cap was set with this
* consumer's use case in mind; no signal, no friction.
*
* 2. **Warn** — `maxLimit < limit <= 2 * maxLimit`. The consumer is asking
* for more than the cap, but not enough to be in OOM territory. Log a
* one-time warning per call site (dedup keyed on stack location + limit
* value) explaining the cap, the three escape valves, and the docs link.
* The query proceeds. Pre-7.30.2 code that relied on the cap silently
* allowing typical safety-cap limits (`10_000`) keeps working; the
* warning teaches the migration.
*
* 3. **Throw** — `limit > 2 * maxLimit`. Real OOM danger territory. Throw
* with the same message format the warning uses so the consumer gets
* consistent guidance whether they hit the soft or hard threshold.
*
* The 2× soft margin is chosen to absorb existing safety-cap patterns
* (`limit: 10_000` against a `maxLimit: 9_000` box, common case at 7.30 GA)
* without disabling OOM protection. Real OOM territory on a JS in-memory
* brain is hundreds of thousands of results, not 10× the safety cap.
*
* Both warning and throw paths use the same formatter so the rendered message
* is identical — consumers see the same recipe regardless of which threshold
* they crossed.
*
* @param limit - The requested `limit` value (already validated non-negative)
* @param config - The active ValidationConfig (carries `maxLimit` + basis)
* @throws When `limit > 2 * config.maxLimit`
*
* @since 7.30.2 (replaced the unconditional throw added in 7.30.0)
*/
function enforceLimitCap(limit: number, config: ValidationConfig): void {
if (limit <= config.maxLimit) return
const message = formatLimitMessage(limit, config)
const hardCeiling = config.maxLimit * 2
if (limit > hardCeiling) {
// OOM danger zone — throw to prevent runaway memory allocation
throw new Error(message)
}
// Soft margin: warn once per call site + limit value, then let the query proceed.
// Survives the lifetime of the process; the goal is to teach the migration recipe,
// not to spam the log every query.
const caller = findCallerLocation(['validateFindParams', 'enforceLimitCap', 'formatLimitMessage']) ?? '<unknown caller>'
const dedupKey = `${caller}|${limit}`
if (!seenLimitWarnings.has(dedupKey)) {
seenLimitWarnings.add(dedupKey)
prodLog.warn('[Brainy] ' + message)
}
}
/**
* Render the user-facing message used by both the warning tier and the throw
* tier of `enforceLimitCap`. Same body shape as the 7.30.1 enforcement-error
* messages: state the problem, name the recipe, point at the call site, link
* the docs.
*/
function formatLimitMessage(limit: number, config: ValidationConfig): string {
const cap = config.maxLimit
const basis = config.limitBasis
const basisLabel: Record<string, string> = {
override: 'consumer-supplied `maxQueryLimit`',
reservedMemory: 'consumer-supplied `reservedQueryMemory`',
containerMemory: 'detected container memory limit',
freeMemory: 'available free memory'
}
const basisText = basisLabel[basis] ?? 'available memory'
const caller = findCallerLocation(['enforceLimitCap', 'formatLimitMessage'])
const lines = [
`find({ limit: ${limit} }) exceeds the auto-configured query limit of ${cap} (basis: ${basisText}). Choose one:`,
` • Increase the cap: new Brainy({ maxQueryLimit: ${Math.min(limit, 100000)} })`,
` • Reserve more memory: new Brainy({ reservedQueryMemory: ${limit * MAX_LIMIT_KB_PER_RESULT * 1024} })`,
' • Paginate: split the query with { limit, offset } pages'
]
if (caller) {
lines.push(` at ${caller}`)
}
lines.push('Docs: https://soulcraft.com/docs/guides/find-limits')
return lines.join('\n')
}
/**
* Universal validations - things that are always invalid
* These are mathematical/logical truths, not configuration
*/
export function validateFindParams(params: FindParams): void {
const config = ValidationConfig.getInstance()
// Universal truth: negative pagination never makes sense
if (params.limit !== undefined) {
if (params.limit < 0) {
throw new Error('limit must be non-negative')
}
enforceLimitCap(params.limit, config)
}
if (params.offset !== undefined && params.offset < 0) {
throw new Error('offset must be non-negative')
}
// Universal truth: probability/similarity must be 0-1
if (params.near?.threshold !== undefined) {
const t = params.near.threshold
if (t < 0 || t > 1) {
throw new Error('threshold must be between 0 and 1')
}
}
// Universal truth: can't specify both query and vector (they're alternatives)
if (params.query !== undefined && params.vector !== undefined) {
throw new Error('cannot specify both query and vector - they are mutually exclusive')
}
// ACCEPTED-AND-IGNORED DIED AS A CLASS (sealed 2026-08-03): options the
// engine does not implement REFUSE with a typed error instead of silently
// doing nothing — a production consumer discovered a no-op by measurement
// once; never again.
if (params.cursor !== undefined) {
throw new UnsupportedFindOptionError('cursor')
}
if ((params as Record<string, unknown>).includeRelations !== undefined) {
throw new UnsupportedFindOptionError('includeRelations')
}
if ((params as Record<string, unknown>).writeOnly !== undefined) {
throw new UnsupportedFindOptionError('writeOnly')
}
// THE ONE ADDRESSING LAW: the orderBy address must PARSE (bare/metadata. =
// user field, system.<field> = the ruled map, anything else refuses typed
// with the valid map in the message) and order must be a real direction.
if (params.orderBy !== undefined) {
if (typeof params.orderBy !== 'string') {
throw new Error('orderBy must be a string field address')
}
parseFieldAddress(params.orderBy, 'entity') // throws InvalidFieldAddressError on a bad address
}
if (params.order !== undefined && params.order !== 'asc' && params.order !== 'desc') {
throw new Error(`order must be 'asc' or 'desc', got '${String(params.order)}'`)
}
// Auto-limit query length based on memory
if (params.query && params.query.length > config.maxQueryLength) {
throw new Error(`query exceeds auto-configured maximum length of ${config.maxQueryLength} characters`)
}
// Validate vector dimensions if provided
if (params.vector && params.vector.length !== config.maxVectorDimensions) {
throw new Error(`vector must have exactly ${config.maxVectorDimensions} dimensions`)
}
// Validate enum types if specified
if (params.type) {
const types = Array.isArray(params.type) ? params.type : [params.type]
for (const type of types) {
if (!Object.values(NounType).includes(type)) {
throw new Error(`invalid NounType: ${type}`)
}
}
}
}
/**
* Validate add parameters
*/
/**
* The namespace cannot be forged: a USER metadata key literally spelled
* 'system.<anything>' would collide with the engine's explicit address
* namespace at read time — refuse it at the write door, loudly, with the
* fix in the message (sealed 2026-08-03).
*/
function rejectForgedSystemKeys(metadata: Record<string, unknown> | undefined, site: string): void {
if (!metadata) return
for (const key of Object.keys(metadata)) {
if (key.startsWith('system.')) {
throw new Error(
`${site}: metadata key '${key}' is not allowed — the 'system.' prefix is the ` +
`engine's explicit address namespace and cannot be used as a user field name. ` +
`Rename the field (e.g. '${key.slice('system.'.length)}').`
)
}
}
}
export function validateAddParams(params: AddParams): void {
rejectForgedSystemKeys(params.metadata as Record<string, unknown> | undefined, 'add()')
// 'data' is ABSENT only when null/undefined — an empty string ('') is real
// content (a legitimate empty file's first write) and must not be treated
// as missing. Falsy-but-present values (0, false, '') all count as present;
// only the true "nothing was given" case is absent.
const hasData = params.data !== undefined && params.data !== null
// MT5 deferred embedding: an explicit vector has nothing to defer, and a
// deferral without data has nothing to embed — both are caller bugs that
// must refuse with the fix, never be silently reinterpreted.
if ((params as AddParams & { deferEmbedding?: boolean }).deferEmbedding === true) {
if (params.vector) {
throw new Error(
`add(): deferEmbedding cannot be combined with an explicit 'vector' — ` +
`the vector is already computed; drop one of the two.`
)
}
if (!hasData) {
throw new Error(
`add(): deferEmbedding requires 'data' (the content the background worker will embed).`
)
}
}
// Universal truth: must have data or vector
if (!hasData && !params.vector) {
throw new Error(
`Invalid add() parameters: Missing required field 'data'\n` +
`\nReceived: ${JSON.stringify({
type: params.type,
hasMetadata: !!params.metadata,
hasId: !!params.id
}, null, 2)}\n` +
`\nExpected one of:\n` +
` { data: 'text to store', type?: 'note', metadata?: {...} }\n` +
` { vector: [0.1, 0.2, ...], type?: 'embedding', metadata?: {...} }\n` +
`\nExamples:\n` +
` await brain.add({ data: 'Machine learning is AI', type: 'concept' })\n` +
` await brain.add({ data: { title: 'Doc', content: '...' }, type: 'document' })`
)
}
// Validate noun type
if (!Object.values(NounType).includes(params.type)) {
throw new Error(
`Invalid NounType: '${params.type}'\n` +
`\nValid types: ${Object.values(NounType).join(', ')}\n` +
`\nExample: await brain.add({ data: 'text', type: NounType.Document })`
)
}
// Validate vector dimensions if provided. A length-0 vector is the
// "unvectored" shape — an explicit `vector: []` (e.g. the VFS root's
// permanently-vectorless creation, see
// VirtualFileSystem.doInitializeRoot()'s zero-norm fix) carries no
// dimension information, exactly like an absent vector or a deferred
// embed's internal stub, so it is exempt from the dimension check rather
// than refused as a "0-dimensional vector".
if (params.vector && params.vector.length > 0) {
const config = ValidationConfig.getInstance()
if (params.vector.length !== config.maxVectorDimensions) {
throw new Error(`vector must have exactly ${config.maxVectorDimensions} dimensions`)
}
}
}
/**
* Validate update parameters
*/
export function validateUpdateParams(params: UpdateParams): void {
rejectForgedSystemKeys(params.metadata as Record<string, unknown> | undefined, 'update()')
// Same absent-vs-empty distinction as validateAddParams: '' is a real new
// value (e.g. truncating a file to empty content via overwrite), only
// null/undefined means "no new data was given".
const hasData = params.data !== undefined && params.data !== null
if ((params as UpdateParams & { deferEmbedding?: boolean }).deferEmbedding === true) {
if (params.vector && params.vector.length === 0) {
// The nonsensical combination Leg D of the zero-norm/unvector-door law
// refuses: `vector: []` is the SANCTIONED UNVECTOR DOOR — an explicit
// instruction to remove the vector NOW, never "please embed" — so it
// cannot be paired with a request to defer an embed.
throw new Error(
`update(): 'vector: []' (the unvector door) cannot be combined with ` +
`'deferEmbedding: true' — an unvector is an explicit instruction to remove ` +
`the vector now, not a request to defer an embed. Drop one of the two.`
)
}
if (params.vector) {
throw new Error(
`update(): deferEmbedding cannot be combined with an explicit 'vector' — ` +
`the vector is already computed; drop one of the two.`
)
}
if (!hasData) {
throw new Error(
`update(): deferEmbedding requires new 'data' — without a data change there is nothing to re-embed.`
)
}
}
// Universal truth: must have an ID
if (!params.id) {
throw new Error('id is required for update')
}
// Universal truth: must update something
if (
!hasData &&
!params.metadata &&
!params.type &&
!params.vector &&
params.subtype === undefined &&
params.visibility === undefined &&
params.confidence === undefined &&
params.weight === undefined
) {
throw new Error('must specify at least one field to update')
}
// Validate type if changing
if (params.type && !Object.values(NounType).includes(params.type)) {
throw new Error(`invalid NounType: ${params.type}`)
}
// Validate vector dimensions if provided. A length-0 vector is the
// SANCTIONED UNVECTOR DOOR (see brainy.ts update()'s matching comment): an
// explicit `vector: []` — or a real all-zero vector, normalized to `[]`
// upstream by the zero-norm law — carries no dimension information,
// exactly like validateAddParams's identical exemption, so it is exempt
// from the dimension check rather than refused as a "0-dimensional
// vector". (The `deferEmbedding` combination above already refuses
// `vector: []` paired with `deferEmbedding: true` — an empty array is
// truthy, so that guard fires unconditionally on any explicit `vector`.)
if (params.vector && params.vector.length > 0) {
const config = ValidationConfig.getInstance()
if (params.vector.length !== config.maxVectorDimensions) {
throw new Error(`vector must have exactly ${config.maxVectorDimensions} dimensions`)
}
}
}
/**
* Validate relate parameters
*/
export function validateRelateParams(params: RelateParams): void {
rejectForgedSystemKeys(params.metadata as Record<string, unknown> | undefined, 'relate()')
// 8.0 verb-id contract (L.7): verb ids are UUIDs, generated by brainy.
// RelateParams has no `id` field — an untyped caller passing one would
// previously have it silently ignored (a generated UUID was used instead).
// Teach instead of surprise: explain the contract and how ids actually work.
const suppliedId = (params as RelateParams & { id?: unknown }).id
if (suppliedId !== undefined) {
throw new Error(
`relate() does not accept a custom id (got: ${JSON.stringify(suppliedId)}). ` +
'Verb ids are UUIDs by contract in 8.0 — brainy generates one for every ' +
'relationship and returns it from relate(). Graph-index providers key ' +
'verb-int interning on the raw UUID bytes, so custom verb ids are not ' +
'supported. Remove the id field and use the returned id instead.'
)
}
// Universal truths
if (!params.from) {
throw new Error('from entity ID is required')
}
if (!params.to) {
throw new Error('to entity ID is required')
}
// Allow self-referential relationships - they're valid in graph systems
// (e.g., a person can be related to themselves, a file can reference itself, etc.)
// Validate verb type - default to RelatedTo if not specified
if (params.type === undefined) {
params.type = VerbType.RelatedTo
} else if (!Object.values(VerbType).includes(params.type)) {
throw new Error(`invalid VerbType: ${params.type}`)
}
// Universal truth: weight must be 0-1
if (params.weight !== undefined) {
if (params.weight < 0 || params.weight > 1) {
throw new Error('weight must be between 0 and 1')
}
}
}
/**
* Validate UpdateRelationParams. Mirror of validateUpdateParams for verbs —
* requires id + at least one field to change; bounds-checks weight/confidence;
* accepts type/subtype/weight/confidence/data/metadata changes.
*/
export function validateUpdateRelationParams(params: UpdateRelationParams): void {
rejectForgedSystemKeys(params.metadata as Record<string, unknown> | undefined, 'updateRelation()')
if (!params.id) {
throw new Error('id is required for updateRelation')
}
if (
!params.data &&
!params.metadata &&
!params.type &&
params.subtype === undefined &&
params.weight === undefined &&
params.confidence === undefined
) {
throw new Error('updateRelation: must specify at least one field to update')
}
if (params.type !== undefined && !Object.values(VerbType).includes(params.type)) {
throw new Error(`invalid VerbType: ${params.type}`)
}
if (params.weight !== undefined && (params.weight < 0 || params.weight > 1)) {
throw new Error('weight must be between 0 and 1')
}
if (params.confidence !== undefined && (params.confidence < 0 || params.confidence > 1)) {
throw new Error('confidence must be between 0 and 1')
}
}
/**
* Get current validation configuration
* Useful for debugging and monitoring
*/
export function getValidationConfig() {
const config = ValidationConfig.getInstance()
return {
maxLimit: config.maxLimit,
maxQueryLength: config.maxQueryLength,
maxVectorDimensions: config.maxVectorDimensions,
systemMemory: getSystemMemory(),
availableMemory: getAvailableMemory()
}
}
/**
* Record query performance for auto-tuning
*/
export function recordQueryPerformance(duration: number, resultCount: number) {
ValidationConfig.getInstance().recordQuery(duration, resultCount)
}