Closes the "dishonest readiness proxy" anti-pattern (Pattern A): size()>0 /
isInitialized were treated as "this index serves queries", but a cold native
index can load its COUNT before its SERVING structure, so a query returned a
silent [] indistinguishable from "no such data". A shared assessIndexReadiness()
now reads only the provider's honest isReady() signal (never size()), applied at
every site:
- Vector: a one-shot verifyVectorLive() guard on the semantic/proximity search
path (a pure semantic find({query}) has no filter, so nothing guarded it). It
prefers isReady(), else a known-vector self-match probe; self-heals via rebuild
or throws the new VectorIndexNotReadyError instead of a silent [].
- Graph: getVerbsBySource/ByTarget skip the fast path when the provider reports
not-ready (falling to the canonical shard scan), plus a one-shot probe that
self-heals a no-isReady provider whose adjacency did not cold-load.
- getIndexStatus(): folds in per-index honest `ready` (making `populated`
honest) + rebuildFailed/rebuildError/degradedIds, so a readiness probe never
200s a brain that is still warming up or degraded.
Unblocked by the native providers now reporting serving-truth (graph via
SSTable-residency readiness, vector via durableBaseLoadFailed). New export:
VectorIndexNotReadyError. getIndexStatus gains additive fields. No breaking API.
13 new tests; existing readiness guards green.
351 lines
12 KiB
TypeScript
351 lines
12 KiB
TypeScript
/**
|
||
* Custom error types for Brainy operations
|
||
* Provides better error classification and handling
|
||
*/
|
||
|
||
export type BrainyErrorType =
|
||
| 'TIMEOUT'
|
||
| 'NETWORK'
|
||
| 'STORAGE'
|
||
| 'NOT_FOUND'
|
||
| 'RETRY_EXHAUSTED'
|
||
| 'VALIDATION'
|
||
| 'INVALID_QUERY'
|
||
| 'FIELD_NOT_INDEXED'
|
||
| 'GRAPH_INDEX_NOT_READY'
|
||
| 'METADATA_INDEX_NOT_READY'
|
||
| 'VECTOR_INDEX_NOT_READY'
|
||
| 'MIGRATION_IN_PROGRESS'
|
||
|
||
/**
|
||
* Custom error class for Brainy operations
|
||
* Provides error type classification and retry information
|
||
*/
|
||
export class BrainyError extends Error {
|
||
public readonly type: BrainyErrorType
|
||
public readonly retryable: boolean
|
||
public readonly originalError?: Error
|
||
public readonly attemptNumber?: number
|
||
public readonly maxRetries?: number
|
||
|
||
constructor(
|
||
message: string,
|
||
type: BrainyErrorType,
|
||
retryable: boolean = false,
|
||
originalError?: Error,
|
||
attemptNumber?: number,
|
||
maxRetries?: number
|
||
) {
|
||
super(message)
|
||
this.name = 'BrainyError'
|
||
this.type = type
|
||
this.retryable = retryable
|
||
this.originalError = originalError
|
||
this.attemptNumber = attemptNumber
|
||
this.maxRetries = maxRetries
|
||
|
||
// Maintain proper stack trace for where our error was thrown (only available on V8)
|
||
if (Error.captureStackTrace) {
|
||
Error.captureStackTrace(this, BrainyError)
|
||
}
|
||
}
|
||
|
||
/**
|
||
* Create a timeout error
|
||
*/
|
||
static timeout(operation: string, timeoutMs: number, originalError?: Error): BrainyError {
|
||
return new BrainyError(
|
||
`Operation '${operation}' timed out after ${timeoutMs}ms`,
|
||
'TIMEOUT',
|
||
true,
|
||
originalError
|
||
)
|
||
}
|
||
|
||
/**
|
||
* Create a network error
|
||
*/
|
||
static network(message: string, originalError?: Error): BrainyError {
|
||
return new BrainyError(
|
||
`Network error: ${message}`,
|
||
'NETWORK',
|
||
true,
|
||
originalError
|
||
)
|
||
}
|
||
|
||
/**
|
||
* Create a storage error
|
||
*/
|
||
static storage(message: string, originalError?: Error): BrainyError {
|
||
return new BrainyError(
|
||
`Storage error: ${message}`,
|
||
'STORAGE',
|
||
true,
|
||
originalError
|
||
)
|
||
}
|
||
|
||
/**
|
||
* Create a not found error
|
||
*/
|
||
static notFound(resource: string): BrainyError {
|
||
return new BrainyError(
|
||
`Resource not found: ${resource}`,
|
||
'NOT_FOUND',
|
||
false
|
||
)
|
||
}
|
||
|
||
/**
|
||
* Create a retry exhausted error
|
||
*/
|
||
static retryExhausted(operation: string, maxRetries: number, lastError?: Error): BrainyError {
|
||
return new BrainyError(
|
||
`Operation '${operation}' failed after ${maxRetries} retry attempts`,
|
||
'RETRY_EXHAUSTED',
|
||
false,
|
||
lastError,
|
||
maxRetries,
|
||
maxRetries
|
||
)
|
||
}
|
||
|
||
/**
|
||
* Create a "field is not indexed" error. Thrown by metadata-index reads
|
||
* when a `where` clause names a field that has neither a column-store
|
||
* entry nor a sparse-index entry. Callers in `find()` evaluation catch
|
||
* this, translate the offending clause to an empty result, and log so
|
||
* the silent-empty behavior is replaced with a loud one. Use
|
||
* `brain.explain({ where: {...} })` to discover this before running.
|
||
*/
|
||
static fieldNotIndexed(field: string): BrainyError {
|
||
return new BrainyError(
|
||
`Field "${field}" is not indexed. find()/where will not match any entities. ` +
|
||
`Likely causes: (1) the writer registered the field in memory but has not flushed; ` +
|
||
`(2) the field name is mistyped; (3) no entity has ever held this field. ` +
|
||
`Run brain.explain({ where: { ${field}: ... } }) for the diagnostic.`,
|
||
'FIELD_NOT_INDEXED',
|
||
false
|
||
)
|
||
}
|
||
|
||
/**
|
||
* Create a validation error
|
||
*/
|
||
static validation(parameter: string, constraint: string, value?: any): BrainyError {
|
||
return new BrainyError(
|
||
`Invalid ${parameter}: ${constraint}`,
|
||
'VALIDATION',
|
||
false
|
||
)
|
||
}
|
||
|
||
/**
|
||
* Check if an error is retryable
|
||
*/
|
||
static isRetryable(error: Error): boolean {
|
||
if (error instanceof BrainyError) {
|
||
return error.retryable
|
||
}
|
||
|
||
// Check for common retryable error patterns
|
||
const message = error.message.toLowerCase()
|
||
const name = error.name.toLowerCase()
|
||
|
||
// Network-related errors that are typically retryable
|
||
if (
|
||
message.includes('timeout') ||
|
||
message.includes('network') ||
|
||
message.includes('connection') ||
|
||
message.includes('econnreset') ||
|
||
message.includes('enotfound') ||
|
||
message.includes('etimedout') ||
|
||
name.includes('timeout')
|
||
) {
|
||
return true
|
||
}
|
||
|
||
// AWS SDK specific retryable errors
|
||
if (
|
||
message.includes('throttling') ||
|
||
message.includes('rate limit') ||
|
||
message.includes('service unavailable') ||
|
||
message.includes('internal server error') ||
|
||
message.includes('bad gateway') ||
|
||
message.includes('gateway timeout')
|
||
) {
|
||
return true
|
||
}
|
||
|
||
return false
|
||
}
|
||
|
||
/**
|
||
* Convert a generic error to a BrainyError with appropriate classification
|
||
*/
|
||
static fromError(error: Error, operation?: string): BrainyError {
|
||
if (error instanceof BrainyError) {
|
||
return error
|
||
}
|
||
|
||
const message = error.message.toLowerCase()
|
||
const name = error.name.toLowerCase()
|
||
|
||
// Classify the error based on common patterns
|
||
if (message.includes('timeout') || name.includes('timeout')) {
|
||
return BrainyError.timeout(operation || 'unknown', 0, error)
|
||
}
|
||
|
||
if (
|
||
message.includes('network') ||
|
||
message.includes('connection') ||
|
||
message.includes('econnreset') ||
|
||
message.includes('enotfound') ||
|
||
message.includes('etimedout')
|
||
) {
|
||
return BrainyError.network(error.message, error)
|
||
}
|
||
|
||
if (
|
||
message.includes('nosuchkey') ||
|
||
message.includes('not found') ||
|
||
message.includes('does not exist')
|
||
) {
|
||
return BrainyError.notFound(operation || 'resource')
|
||
}
|
||
|
||
if (
|
||
message.includes('invalid') ||
|
||
message.includes('validation') ||
|
||
message.includes('cannot be null') ||
|
||
message.includes('must be')
|
||
) {
|
||
return new BrainyError(error.message, 'VALIDATION', false, error)
|
||
}
|
||
|
||
// Default to storage error for unclassified errors
|
||
return BrainyError.storage(error.message, error)
|
||
}
|
||
}
|
||
|
||
/**
|
||
* Thrown when the graph adjacency index reports that relationships exist (its
|
||
* persisted manifest/count loaded, or its readiness signal says otherwise) but
|
||
* the source→target adjacency itself did NOT load — so graph traversals
|
||
* (`find({ connected })`, `neighbors()`, `related()`) would otherwise return an
|
||
* EMPTY array indistinguishable from "no edges".
|
||
*
|
||
* On 8.0 brainy detects this on the first graph read via the provider's honest
|
||
* sync `isReady()` signal (true ONLY when the edges are loaded; see
|
||
* {@link import('../plugin.js').GraphIndexProvider.isReady}); for older providers
|
||
* that do not expose it, it falls back to a known-edge-sample probe (one persisted
|
||
* verb + one neighbor lookup). Either way it attempts a rebuild from storage and
|
||
* raises this LOUD, catchable error only if even that cannot make the adjacency
|
||
* ready — replacing silent data-invisibility with a clear failure.
|
||
*
|
||
* Observed with a native graph provider whose cold-open adjacency load is
|
||
* swallowed on certain storage adapters; the fix is upstream in the provider,
|
||
* but Brainy refuses to serve `[]` as if it were truth.
|
||
*/
|
||
export class GraphIndexNotReadyError extends BrainyError {
|
||
constructor(message: string, originalError?: Error) {
|
||
super(message, 'GRAPH_INDEX_NOT_READY', false, originalError)
|
||
this.name = 'GraphIndexNotReadyError'
|
||
if (Error.captureStackTrace) {
|
||
Error.captureStackTrace(this, GraphIndexNotReadyError)
|
||
}
|
||
}
|
||
}
|
||
|
||
/**
|
||
* Thrown when the metadata field index reports data but cannot serve a KNOWN
|
||
* persisted field value even after a rebuild — i.e. the `where` / filter
|
||
* postings did not load on a cold open and could not be restored. The
|
||
* field-index counterpart of {@link GraphIndexNotReadyError}: it replaces the
|
||
* silent-empty failure mode (a cold `find({ where })` returning `[]`
|
||
* indistinguishable from "no such data") with a loud, catchable error, so a
|
||
* consumer never renders "nothing found" over data that is simply not-yet-warm.
|
||
*
|
||
* Detected once per brain by a known-value serving probe on the first filtered
|
||
* `find()`; brainy self-heals (rebuilds the index from the canonical records)
|
||
* first and only raises this if the rebuild still cannot serve the known value.
|
||
*/
|
||
export class MetadataIndexNotReadyError extends BrainyError {
|
||
constructor(message: string, originalError?: Error) {
|
||
super(message, 'METADATA_INDEX_NOT_READY', false, originalError)
|
||
this.name = 'MetadataIndexNotReadyError'
|
||
if (Error.captureStackTrace) {
|
||
Error.captureStackTrace(this, MetadataIndexNotReadyError)
|
||
}
|
||
}
|
||
}
|
||
|
||
/**
|
||
* Thrown when the vector index reports vectors (`size() > 0` or, on a native
|
||
* provider, `isReady() === false`) but cannot return a KNOWN persisted vector
|
||
* even after a rebuild — i.e. the semantic serving structure did not load on a
|
||
* cold open and could not be restored. The vector-search counterpart of
|
||
* {@link GraphIndexNotReadyError} / {@link MetadataIndexNotReadyError}: it
|
||
* replaces the silent-empty failure mode (a cold `find({ query })` returning
|
||
* `[]` indistinguishable from "no similar data") with a loud, catchable error,
|
||
* so a consumer never renders "nothing found" over data that is simply
|
||
* not-yet-warm.
|
||
*
|
||
* Detected once per brain by a known-vector serving probe on the first
|
||
* semantic / proximity `find()`; brainy self-heals (rebuilds the index from the
|
||
* canonical records) first and only raises this if the rebuild still cannot
|
||
* serve the known vector.
|
||
*/
|
||
export class VectorIndexNotReadyError extends BrainyError {
|
||
constructor(message: string, originalError?: Error) {
|
||
super(message, 'VECTOR_INDEX_NOT_READY', false, originalError)
|
||
this.name = 'VectorIndexNotReadyError'
|
||
if (Error.captureStackTrace) {
|
||
Error.captureStackTrace(this, VectorIndexNotReadyError)
|
||
}
|
||
}
|
||
}
|
||
|
||
/**
|
||
* Thrown when a data-plane read or write is issued against a brain that is
|
||
* running its one-time, automatic 7.x → 8.0 on-disk upgrade — the coordinated
|
||
* migration LOCK. While a native provider rebuilds all derived indexes from the
|
||
* canonical records, brainy blocks reads and writes so no operation touches a
|
||
* half-built index; the caller waits for the correct answer rather than getting
|
||
* a partial one. This error is raised ONLY when the wait exceeds the configured
|
||
* window ({@link BrainyConfig.migrationWaitTimeoutMs}, default 30 s) — never
|
||
* instead of a partial/incorrect result.
|
||
*
|
||
* It is `retryable`: the upgrade continues in the background. Retry shortly,
|
||
* watch `brain.getIndexStatus().migration` for progress, or for a very large
|
||
* brain run the offline migrator. Data is safe; nothing is lost. Consumers can
|
||
* catch this (e.g. request middleware) and answer HTTP 503 + `Retry-After`.
|
||
*
|
||
* @example
|
||
* try {
|
||
* await brain.find({ query })
|
||
* } catch (e) {
|
||
* if (e instanceof MigrationInProgressError) {
|
||
* res.set('Retry-After', '5').status(503).json({ upgrading: true, percent: e.percent })
|
||
* return
|
||
* }
|
||
* throw e
|
||
* }
|
||
*/
|
||
export class MigrationInProgressError extends BrainyError {
|
||
/** Milliseconds the operation waited on the migration lock before timing out. */
|
||
public readonly elapsedMs: number
|
||
/** Latest observed migration progress (0–100), when the provider reports it. */
|
||
public readonly percent?: number
|
||
|
||
constructor(message: string, elapsedMs: number, percent?: number, originalError?: Error) {
|
||
super(message, 'MIGRATION_IN_PROGRESS', true, originalError)
|
||
this.name = 'MigrationInProgressError'
|
||
this.elapsedMs = elapsedMs
|
||
this.percent = percent
|
||
if (Error.captureStackTrace) {
|
||
Error.captureStackTrace(this, MigrationInProgressError)
|
||
}
|
||
}
|
||
}
|