Replaces rc.8's no-freeze deference with an automatic, observable, coordinated
migration LOCK (David's reversal: "unknown/halfway states are more dangerous
than blocking"). While a native provider runs its one-time 7.x→8.0
rebuild-from-canonical (`isMigrating()===true`), brainy blocks/queues data-plane
reads AND writes so no operation ever touches a half-built index — closing the
three seam-map gaps (lost mid-rebuild writes, degraded reads, read-time rebuild).
Mechanism (one choke point):
- awaitMigrationLock() at ensureInitialized() covers all ~40 data-plane methods;
a non-migrating brain pays one boolean check. Polls isMigrating() at 250ms;
after migrationWaitTimeoutMs (default 30s) throws a retryable, exported
MigrationInProgressError. The timeout bounds the CALLER'S WAIT, not the
migration — the rebuild is unbounded and never interrupted.
- init() awaits the lock before VFS bootstrap + before serving ("not ready until
upgraded"); a rebuild past the budget surfaces MigrationInProgressError at init
(raise the budget, or run the offline migrator).
Observability (readiness-probe correct — never gated):
- getIndexStatus() gains `migrating` + `migration` (MigrationProgress); a probe
maps migrating→HTTP 503+Retry-After, not 500. health()/checkHealth() lock-exempt
(health() reports a `warn` migration check). stampBrainFormat()/close() are
ungated so cor can clear the lock — no deadlock.
- Optional provider migrationStatus() is relayed verbatim for a live %.
verifyGraphAdjacencyLive() honors the lock (no self-rebuild mid-migration); the
stamp is still withheld while any provider migrates (cor stamps after verify).
Adversarially verified: fixed a critical init/VFS-bootstrap deadlock, a mixed-
provider busy-spin (dropped the event-driven signal → pure poll), a not-yet-
initialized getIndexStatus crash, and once-per-window log/clock resets. 10 lock
tests (incl. the init-during-migration regression). Gates: typecheck 0, build 0,
test:unit 1753/1753.
299 lines
10 KiB
TypeScript
299 lines
10 KiB
TypeScript
/**
|
||
* Custom error types for Brainy operations
|
||
* Provides better error classification and handling
|
||
*/
|
||
|
||
export type BrainyErrorType =
|
||
| 'TIMEOUT'
|
||
| 'NETWORK'
|
||
| 'STORAGE'
|
||
| 'NOT_FOUND'
|
||
| 'RETRY_EXHAUSTED'
|
||
| 'VALIDATION'
|
||
| 'FIELD_NOT_INDEXED'
|
||
| 'GRAPH_INDEX_NOT_READY'
|
||
| 'MIGRATION_IN_PROGRESS'
|
||
|
||
/**
|
||
* Custom error class for Brainy operations
|
||
* Provides error type classification and retry information
|
||
*/
|
||
export class BrainyError extends Error {
|
||
public readonly type: BrainyErrorType
|
||
public readonly retryable: boolean
|
||
public readonly originalError?: Error
|
||
public readonly attemptNumber?: number
|
||
public readonly maxRetries?: number
|
||
|
||
constructor(
|
||
message: string,
|
||
type: BrainyErrorType,
|
||
retryable: boolean = false,
|
||
originalError?: Error,
|
||
attemptNumber?: number,
|
||
maxRetries?: number
|
||
) {
|
||
super(message)
|
||
this.name = 'BrainyError'
|
||
this.type = type
|
||
this.retryable = retryable
|
||
this.originalError = originalError
|
||
this.attemptNumber = attemptNumber
|
||
this.maxRetries = maxRetries
|
||
|
||
// Maintain proper stack trace for where our error was thrown (only available on V8)
|
||
if (Error.captureStackTrace) {
|
||
Error.captureStackTrace(this, BrainyError)
|
||
}
|
||
}
|
||
|
||
/**
|
||
* Create a timeout error
|
||
*/
|
||
static timeout(operation: string, timeoutMs: number, originalError?: Error): BrainyError {
|
||
return new BrainyError(
|
||
`Operation '${operation}' timed out after ${timeoutMs}ms`,
|
||
'TIMEOUT',
|
||
true,
|
||
originalError
|
||
)
|
||
}
|
||
|
||
/**
|
||
* Create a network error
|
||
*/
|
||
static network(message: string, originalError?: Error): BrainyError {
|
||
return new BrainyError(
|
||
`Network error: ${message}`,
|
||
'NETWORK',
|
||
true,
|
||
originalError
|
||
)
|
||
}
|
||
|
||
/**
|
||
* Create a storage error
|
||
*/
|
||
static storage(message: string, originalError?: Error): BrainyError {
|
||
return new BrainyError(
|
||
`Storage error: ${message}`,
|
||
'STORAGE',
|
||
true,
|
||
originalError
|
||
)
|
||
}
|
||
|
||
/**
|
||
* Create a not found error
|
||
*/
|
||
static notFound(resource: string): BrainyError {
|
||
return new BrainyError(
|
||
`Resource not found: ${resource}`,
|
||
'NOT_FOUND',
|
||
false
|
||
)
|
||
}
|
||
|
||
/**
|
||
* Create a retry exhausted error
|
||
*/
|
||
static retryExhausted(operation: string, maxRetries: number, lastError?: Error): BrainyError {
|
||
return new BrainyError(
|
||
`Operation '${operation}' failed after ${maxRetries} retry attempts`,
|
||
'RETRY_EXHAUSTED',
|
||
false,
|
||
lastError,
|
||
maxRetries,
|
||
maxRetries
|
||
)
|
||
}
|
||
|
||
/**
|
||
* Create a "field is not indexed" error. Thrown by metadata-index reads
|
||
* when a `where` clause names a field that has neither a column-store
|
||
* entry nor a sparse-index entry. Callers in `find()` evaluation catch
|
||
* this, translate the offending clause to an empty result, and log so
|
||
* the silent-empty behavior is replaced with a loud one. Use
|
||
* `brain.explain({ where: {...} })` to discover this before running.
|
||
*/
|
||
static fieldNotIndexed(field: string): BrainyError {
|
||
return new BrainyError(
|
||
`Field "${field}" is not indexed. find()/where will not match any entities. ` +
|
||
`Likely causes: (1) the writer registered the field in memory but has not flushed; ` +
|
||
`(2) the field name is mistyped; (3) no entity has ever held this field. ` +
|
||
`Run brain.explain({ where: { ${field}: ... } }) for the diagnostic.`,
|
||
'FIELD_NOT_INDEXED',
|
||
false
|
||
)
|
||
}
|
||
|
||
/**
|
||
* Create a validation error
|
||
*/
|
||
static validation(parameter: string, constraint: string, value?: any): BrainyError {
|
||
return new BrainyError(
|
||
`Invalid ${parameter}: ${constraint}`,
|
||
'VALIDATION',
|
||
false
|
||
)
|
||
}
|
||
|
||
/**
|
||
* Check if an error is retryable
|
||
*/
|
||
static isRetryable(error: Error): boolean {
|
||
if (error instanceof BrainyError) {
|
||
return error.retryable
|
||
}
|
||
|
||
// Check for common retryable error patterns
|
||
const message = error.message.toLowerCase()
|
||
const name = error.name.toLowerCase()
|
||
|
||
// Network-related errors that are typically retryable
|
||
if (
|
||
message.includes('timeout') ||
|
||
message.includes('network') ||
|
||
message.includes('connection') ||
|
||
message.includes('econnreset') ||
|
||
message.includes('enotfound') ||
|
||
message.includes('etimedout') ||
|
||
name.includes('timeout')
|
||
) {
|
||
return true
|
||
}
|
||
|
||
// AWS SDK specific retryable errors
|
||
if (
|
||
message.includes('throttling') ||
|
||
message.includes('rate limit') ||
|
||
message.includes('service unavailable') ||
|
||
message.includes('internal server error') ||
|
||
message.includes('bad gateway') ||
|
||
message.includes('gateway timeout')
|
||
) {
|
||
return true
|
||
}
|
||
|
||
return false
|
||
}
|
||
|
||
/**
|
||
* Convert a generic error to a BrainyError with appropriate classification
|
||
*/
|
||
static fromError(error: Error, operation?: string): BrainyError {
|
||
if (error instanceof BrainyError) {
|
||
return error
|
||
}
|
||
|
||
const message = error.message.toLowerCase()
|
||
const name = error.name.toLowerCase()
|
||
|
||
// Classify the error based on common patterns
|
||
if (message.includes('timeout') || name.includes('timeout')) {
|
||
return BrainyError.timeout(operation || 'unknown', 0, error)
|
||
}
|
||
|
||
if (
|
||
message.includes('network') ||
|
||
message.includes('connection') ||
|
||
message.includes('econnreset') ||
|
||
message.includes('enotfound') ||
|
||
message.includes('etimedout')
|
||
) {
|
||
return BrainyError.network(error.message, error)
|
||
}
|
||
|
||
if (
|
||
message.includes('nosuchkey') ||
|
||
message.includes('not found') ||
|
||
message.includes('does not exist')
|
||
) {
|
||
return BrainyError.notFound(operation || 'resource')
|
||
}
|
||
|
||
if (
|
||
message.includes('invalid') ||
|
||
message.includes('validation') ||
|
||
message.includes('cannot be null') ||
|
||
message.includes('must be')
|
||
) {
|
||
return new BrainyError(error.message, 'VALIDATION', false, error)
|
||
}
|
||
|
||
// Default to storage error for unclassified errors
|
||
return BrainyError.storage(error.message, error)
|
||
}
|
||
}
|
||
|
||
/**
|
||
* Thrown when the graph adjacency index reports that relationships exist (its
|
||
* persisted manifest/count loaded, or its readiness signal says otherwise) but
|
||
* the source→target adjacency itself did NOT load — so graph traversals
|
||
* (`find({ connected })`, `neighbors()`, `related()`) would otherwise return an
|
||
* EMPTY array indistinguishable from "no edges".
|
||
*
|
||
* On 8.0 brainy detects this on the first graph read via the provider's honest
|
||
* sync `isReady()` signal (true ONLY when the edges are loaded; see
|
||
* {@link import('../plugin.js').GraphIndexProvider.isReady}); for older providers
|
||
* that do not expose it, it falls back to a known-edge-sample probe (one persisted
|
||
* verb + one neighbor lookup). Either way it attempts a rebuild from storage and
|
||
* raises this LOUD, catchable error only if even that cannot make the adjacency
|
||
* ready — replacing silent data-invisibility with a clear failure.
|
||
*
|
||
* Observed with a native graph provider whose cold-open adjacency load is
|
||
* swallowed on certain storage adapters; the fix is upstream in the provider,
|
||
* but Brainy refuses to serve `[]` as if it were truth.
|
||
*/
|
||
export class GraphIndexNotReadyError extends BrainyError {
|
||
constructor(message: string, originalError?: Error) {
|
||
super(message, 'GRAPH_INDEX_NOT_READY', false, originalError)
|
||
this.name = 'GraphIndexNotReadyError'
|
||
if (Error.captureStackTrace) {
|
||
Error.captureStackTrace(this, GraphIndexNotReadyError)
|
||
}
|
||
}
|
||
}
|
||
|
||
/**
|
||
* Thrown when a data-plane read or write is issued against a brain that is
|
||
* running its one-time, automatic 7.x → 8.0 on-disk upgrade — the coordinated
|
||
* migration LOCK. While a native provider rebuilds all derived indexes from the
|
||
* canonical records, brainy blocks reads and writes so no operation touches a
|
||
* half-built index; the caller waits for the correct answer rather than getting
|
||
* a partial one. This error is raised ONLY when the wait exceeds the configured
|
||
* window ({@link BrainyConfig.migrationWaitTimeoutMs}, default 30 s) — never
|
||
* instead of a partial/incorrect result.
|
||
*
|
||
* It is `retryable`: the upgrade continues in the background. Retry shortly,
|
||
* watch `brain.getIndexStatus().migration` for progress, or for a very large
|
||
* brain run the offline migrator. Data is safe; nothing is lost. Consumers can
|
||
* catch this (e.g. request middleware) and answer HTTP 503 + `Retry-After`.
|
||
*
|
||
* @example
|
||
* try {
|
||
* await brain.find({ query })
|
||
* } catch (e) {
|
||
* if (e instanceof MigrationInProgressError) {
|
||
* res.set('Retry-After', '5').status(503).json({ upgrading: true, percent: e.percent })
|
||
* return
|
||
* }
|
||
* throw e
|
||
* }
|
||
*/
|
||
export class MigrationInProgressError extends BrainyError {
|
||
/** Milliseconds the operation waited on the migration lock before timing out. */
|
||
public readonly elapsedMs: number
|
||
/** Latest observed migration progress (0–100), when the provider reports it. */
|
||
public readonly percent?: number
|
||
|
||
constructor(message: string, elapsedMs: number, percent?: number, originalError?: Error) {
|
||
super(message, 'MIGRATION_IN_PROGRESS', true, originalError)
|
||
this.name = 'MigrationInProgressError'
|
||
this.elapsedMs = elapsedMs
|
||
this.percent = percent
|
||
if (Error.captureStackTrace) {
|
||
Error.captureStackTrace(this, MigrationInProgressError)
|
||
}
|
||
}
|
||
}
|