/** * 🧠 Brainy 3.0 Type Definitions * * Beautiful, consistent, type-safe interfaces for the future of neural databases */ import { StorageAdapter, Vector } from '../coreTypes.js' import type { EntityVisibility } from '../coreTypes.js' import { NounType, VerbType } from './graphTypes.js' import type { EntityMetadataInput, EntityMetadataPatch, RelationMetadataInput, RelationMetadataPatch } from './reservedFields.js' // ============= Core Types ============= /** * Entity (Noun) β€” the fundamental data unit in Brainy * * **Data vs Metadata:** * - `data`: Content used for vector embeddings. Searchable via **semantic similarity** * (HNSW index) and hybrid text+semantic search. NOT queryable via `where` filters. * - `metadata`: Structured fields indexed by MetadataIndex. Queryable via `where` * filters in `find()`. Standard system fields (noun, createdAt, etc.) are stored * alongside user metadata but extracted to top-level Entity fields on read. */ export interface Entity { /** Unique identifier (UUID v4) */ id: string /** Embedding vector (384-dim by default). Empty array when loaded without includeVectors. */ vector: Vector /** Entity type classification (NounType enum) */ type: NounType /** * Per-product sub-classification within the NounType (e.g. a `Person` entity might * have `subtype: 'employee'` or `'customer'`; a `Document` might have `subtype: 'invoice'`). * Flat string, no hierarchy. Top-level standard field β€” indexed on the fast path and * rolled into per-NounType statistics. */ subtype?: string /** * Visibility tier (reserved top-level field). Absent === `'public'` (counted and * returned everywhere). `'internal'` hides the entity from default `find()` / counts / * `stats()` (opt back in with `find({ includeInternal: true })`); `'system'` is Brainy * plumbing, hidden unless `find({ includeSystem: true })`. See {@link EntityVisibility}. */ visibility?: EntityVisibility /** Opaque content β€” used for embeddings and semantic search. Not indexed by MetadataIndex. */ data?: any /** User-defined structured fields β€” indexed and queryable via `where` filters. */ metadata?: T /** Multi-tenancy service identifier */ service?: string /** Creation timestamp (ms since epoch) */ createdAt: number /** Last update timestamp (ms since epoch) */ updatedAt?: number /** Source that created this entity (e.g., augmentation info) */ createdBy?: string /** Type classification confidence (0-1) */ confidence?: number /** Entity importance/salience (0-1) */ weight?: number /** * Monotonic revision counter, auto-bumped on every successful `update()`. * Starts at `1` on `add()`. Pre-7.31.0 entities without `_rev` are read as `1`. * Pass back as `update({ id, ..., ifRev: _rev })` for optimistic-concurrency CAS β€” * throws `RevisionConflictError` if the persisted rev moved since read. */ _rev?: number } /** * Relation (Verb) β€” a typed edge connecting two entities * * **Data vs Metadata (on relationships):** * - `data`: Opaque content stored on the relationship. If provided during relate(), * overrides the auto-computed vector (default: average of source+target vectors). * - `metadata`: Structured queryable fields on the edge (e.g., role, startDate). */ export interface Relation { /** Unique identifier (UUID v4) */ id: string /** Source entity ID */ from: string /** Target entity ID */ to: string /** Relationship type classification (VerbType enum) */ type: VerbType /** * Per-product sub-classification within the VerbType (e.g. a `ReportsTo` * relationship might have `subtype: 'direct'` vs `'dotted-line'`; a `RelatedTo` * edge might carry `'spouse'` / `'sibling'` / `'colleague'`). Flat string, no * hierarchy. Top-level standard field β€” indexed on the fast path and rolled into * per-VerbType statistics so queries like `related({ verb, subtype })` hit * the column-store directly. */ subtype?: string /** * Visibility tier (reserved top-level field), the verb mirror of `Entity.visibility`. * Absent === `'public'` (counted and returned everywhere). `'internal'` hides the edge * from default `related()` / counts / `stats()` (opt back in with * `related({ includeInternal: true })`); `'system'` is Brainy plumbing. See * {@link EntityVisibility}. */ visibility?: EntityVisibility /** Connection strength (0-1, default: 1.0) */ weight?: number /** Opaque content for the relationship (overrides auto-computed vector if provided) */ data?: any /** User-defined structured fields on the edge */ metadata?: T /** Multi-tenancy service identifier */ service?: string /** Creation timestamp (ms since epoch) */ createdAt: number /** Last update timestamp (ms since epoch) */ updatedAt?: number /** Relationship certainty (0-1) */ confidence?: number /** Evidence for why this relationship was detected */ evidence?: RelationEvidence } /** * Evidence for why a relationship was detected */ export interface RelationEvidence { sourceText?: string // Text that indicated this relationship position?: { // Position in source text start: number end: number } method: 'neural' | 'pattern' | 'structural' | 'explicit' // How it was detected reasoning?: string // Human-readable explanation } /** * Search result with similarity score * * Flattens commonly-used entity fields to top level for convenience, * while preserving full entity in 'entity' field for backward compatibility. */ export interface Result { // Search metadata id: string score: number // Convenience: Common entity fields flattened to top level type?: NounType // Entity type (from entity.type) subtype?: string // Per-product sub-classification (from entity.subtype) visibility?: EntityVisibility // Visibility tier (from entity.visibility); absent === 'public' metadata?: T // Entity metadata (from entity.metadata) data?: any // Entity data (from entity.data) confidence?: number // Type classification confidence (from entity.confidence) weight?: number // Entity importance (from entity.weight) _rev?: number // Monotonic revision counter (from entity._rev) β€” pass back as update({ ifRev }) for CAS // Full entity (preserved for backward compatibility) entity: Entity // Score transparency explanation?: ScoreExplanation // Match visibility - shows what matched in hybrid search textMatches?: string[] // Query words found in entity (e.g., ["david", "warrior"]) textScore?: number // Normalized text match score (0-1) semanticScore?: number // Semantic similarity score (0-1) matchSource?: 'text' | 'semantic' | 'both' // Where this result came from rrfScore?: number // Raw RRF fusion score (for advanced ranking analysis) // Aggregation: present only on rows returned by find({ aggregate }). These mirror the // documented AggregateResult shape so callers can read group/metric data at the top level // instead of digging into `metadata`. (The same values are also flattened into `metadata` // for backward compatibility.) groupKey?: Record // Group-by key values for this aggregate row metrics?: Record // Computed metric values (sum/avg/min/max/count) count?: number // Total entity count in this group } /** * Score explanation for transparency */ export interface ScoreExplanation { vectorScore?: number metadataScore?: number graphScore?: number boosts?: Record penalties?: Record } // ============= Operation Parameters ============= /** * Parameters for adding entities * * **Data vs Metadata:** * - `data` is embedded into a vector and searchable via semantic similarity (HNSW). * It is NOT indexed by MetadataIndex and NOT queryable via `where` filters. * - `metadata` is indexed by MetadataIndex and queryable via `where` filters in `find()`. */ /** * Declaration-merging registry for per-`(type, subtype)` metadata shapes * (Brainy 8.0, per BRAINY-8.0-SUBTYPE-CONTRACT Β§ C-3). * * Consumers extend this interface in their own code to tell TypeScript what * shape `metadata` takes for a given `(NounType, subtype)` pair. Brainy * doesn't ship any entries β€” every entry is a consumer concern. * * @example * ```ts * declare module '@soulcraft/brainy' { * interface SubtypeRegistry { * // For NounType.Person, subtype 'employee': * 'person:employee': { employeeId: string; department: string } * // For NounType.Document, subtype 'invoice': * 'document:invoice': { invoiceNumber: string; amount: number } * } * } * ``` * * Keys are `${NounType}:${subtype}`. Values describe the metadata shape. * Brainy reads this registry (when present) to give typed `metadata` * autocomplete + type-checking on `add()` / `update()` / `find()`. Consumers * with no registry entries see no type-level change β€” the metadata bag * stays `T = any`. */ export interface SubtypeRegistry { // Intentionally empty. Consumers extend via declaration merging. } /** * A single back-fill rule for `brain.fillSubtypes()`. * * - A **literal string** assigns that subtype to every matching entry that * lacks one (`'general'` β€” a blanket default). * - A **function** receives the full entry and returns the subtype to assign, * or `undefined` to leave the entry untouched (it is counted as `skipped` * so a later run with a stricter rule can pick it up). Functions can derive * the subtype from existing fields, e.g. `(e) => e.metadata?.kind`. * * @typeParam E - The entry shape the rule sees: `Entity` for NounType keys, * `Relation` for VerbType keys. */ export type FillSubtypeRule = string | ((entry: E) => string | undefined) /** * Rule map for `brain.fillSubtypes()` β€” the 8.0 subtype migration helper. * * Keys are `NounType` values (entity rules) and/or `VerbType` values * (relationship rules); the two vocabularies don't overlap, so a single map * covers both sides. Each value is a {@link FillSubtypeRule}: a literal * subtype string or a function deriving one from the entry. * * @example * ```ts * await brain.fillSubtypes({ * [NounType.Person]: (e) => e.metadata?.kind ?? 'unspecified', * [NounType.Thing]: 'general', // literal default * [VerbType.RelatedTo]: 'unspecified' // relationship rule * }) * ``` */ export type FillSubtypeRules = { [K in NounType]?: FillSubtypeRule> } & { [K in VerbType]?: FillSubtypeRule> } /** * Summary returned by `brain.fillSubtypes()`. * * After a run, `skipped` is exactly the remaining migration debt β€” re-running * `brain.audit()` reports the same entries. Entries that already carry a * subtype count toward `scanned` only (they are not debt, so they are neither * `filled` nor `skipped`). */ export interface FillSubtypesResult { /** Total entries examined (entities + relationships). */ scanned: number /** Entries that received a subtype during this run. */ filled: number /** * Entries still missing a subtype after the run β€” either their type has no * rule in the map, or their rule function returned `undefined`/empty. */ skipped: number /** Per-entry write failures (`fillSubtypes` continues past individual errors). */ errors: Array<{ id: string; error: string }> /** Fill counts grouped by NounType/VerbType key (only types with fills appear). */ byType: Record } export interface AddParams { /** Content to embed and store. Strings are auto-embedded; objects are JSON-stringified for embedding. */ data: any | Vector /** Entity type classification (required) */ type: NounType /** * Per-product sub-classification within the NounType (e.g. a `Person` entity might have * `subtype: 'employee'` or `'customer'`; a `Document` might have `subtype: 'invoice'`). * Flat string, no hierarchy β€” consumers choose the vocabulary. Indexed and rolled up into * per-NounType statistics for fast `find({ type, subtype })` filtering and `groupBy:['subtype']` * aggregation on the standard-field fast path. */ subtype?: string /** * Visibility tier (reserved top-level field). Omit (or pass `'public'`) for normal * data that is counted and returned everywhere. Pass `'internal'` for app-internal * data that should be hidden from default `find()` / counts / `stats()` but stay * retrievable via `find({ includeInternal: true })`. The `'system'` tier is reserved * for Brainy's own plumbing and is intentionally not accepted here. Stored only when * not `'public'`. */ visibility?: 'public' | 'internal' /** * Structured queryable fields β€” indexed by MetadataIndex, used in `where` filters. * * Reserved entity fields (`RESERVED_ENTITY_FIELDS` β€” `noun`, `subtype`, `visibility`, * `createdAt`, `updatedAt`, `confidence`, `weight`, `service`, `data`, `createdBy`, * `_rev`) may NOT appear here β€” they have dedicated top-level params and the type makes * a literal reserved key a compile error. Untyped (JavaScript) callers that pass one * anyway are normalized at write time: user-settable fields remap to their top-level * param (top-level wins when both are supplied), system-managed fields are dropped with * a one-shot warning. */ metadata?: EntityMetadataInput /** Custom entity ID (auto-generated UUID v4 if not provided) */ id?: string /** Pre-computed embedding vector (skips auto-embedding when provided) */ vector?: Vector /** Multi-tenancy service identifier */ service?: string /** Type classification confidence (0-1) */ confidence?: number /** Entity importance/salience (0-1) */ weight?: number /** Track which augmentation created this entity */ createdBy?: { augmentation: string; version: string } /** * Conditional insert. When `true` AND a custom `id` is supplied AND an entity with * that `id` already exists, `add()` returns the existing `id` without writing β€” no * throw, no overwrite. Used for idempotent bootstrap and create-or-noop patterns. * Ignored when `id` is omitted (a freshly generated UUID can never collide). */ ifAbsent?: boolean /** * Create-or-update in a single call. When `true` AND a custom `id` is supplied AND * an entity with that `id` already exists, `add()` applies the supplied fields as an * UPDATE instead of the default destructive overwrite: metadata is MERGED (not * replaced), `data` re-embeds only when it changed, `_rev` is bumped, and the * original `createdAt` is PRESERVED. When no entity with that `id` exists, `add()` * inserts normally (a fresh `_rev` of 1). Ignored β€” behaves as a plain insert β€” when * `id` is omitted, since a freshly generated UUID can never collide. * * Mutually exclusive with {@link AddParams.ifAbsent}: `ifAbsent` SKIPS an existing * entity, `upsert` MERGES into it β€” supplying both throws. * * @example * // First call inserts; a later call with the same id merges + bumps _rev. * await brain.add({ id: 'customer-42', type: 'customer', data: 'Acme', upsert: true }) * await brain.add({ id: 'customer-42', type: 'customer', metadata: { tier: 'gold' }, upsert: true }) */ upsert?: boolean } /** * Parameters for updating entities */ export interface UpdateParams { id: string // Entity to update data?: any // New content to re-embed type?: NounType // Change type subtype?: string // Change subtype (set to '' or null-equivalent via dedicated unset is future work) /** * Change the visibility tier. Omit to preserve the existing value. Toggling between * `'public'` and `'internal'` moves the entity in/out of the default-visible counts and * `find()` results. The `'system'` tier is Brainy-internal (e.g. re-asserted on the VFS * root during repair) β€” consumer code should only ever use `'public'` / `'internal'`. */ visibility?: EntityVisibility /** * Metadata fields to merge (or replace when `merge: false`). Reserved entity * fields (`RESERVED_ENTITY_FIELDS`) may NOT appear here β€” `confidence` / * `weight` / `subtype` / `visibility` have dedicated params on this call, and the rest * are system-managed. A literal reserved key is a compile error; untyped callers * are normalized at write time (remap user-settable, drop system-managed * with a one-shot warning). */ metadata?: EntityMetadataPatch merge?: boolean // Merge or replace metadata (default: true) vector?: Vector // New pre-computed vector confidence?: number // Update type classification confidence weight?: number // Update entity importance/salience /** * Optimistic concurrency check. When provided, the update fails with * `RevisionConflictError` if the persisted entity's `_rev` does not equal `ifRev`. * `_rev` is auto-bumped on every successful update β€” read it from the entity returned * by `get()` / `find()` / `search()`. Omit to skip the check (unconditional update). */ ifRev?: number } /** * Parameters for creating relationships * * **Data vs Metadata (on relationships):** * - `data`: Opaque content for the edge. If provided, overrides the auto-computed * vector (default: average of source+target entity vectors). * - `metadata`: Structured queryable fields on the edge. */ export interface RelateParams { /** Source entity ID (required β€” must exist) */ from: string /** Target entity ID (required β€” must exist) */ to: string /** Relationship type classification (required) */ type: VerbType /** * Per-product sub-classification within the VerbType (e.g. a `ReportsTo` * relationship might have `subtype: 'direct'` or `'dotted-line'`; a `RelatedTo` * edge might carry `'spouse'` or `'colleague'`). Flat string, no hierarchy. * Indexed and rolled up into per-VerbType statistics for fast filtering * (`related({ verb, subtype })`) and aggregation (`groupBy:['subtype']`). */ subtype?: string /** * Visibility tier (reserved top-level field), the verb mirror of `AddParams.visibility`. * Omit (or pass `'public'`) for normal edges that are counted and returned everywhere. * Pass `'internal'` for app-internal edges hidden from default `related()` / counts / * `stats()` but retrievable via `related({ includeInternal: true })`. The `'system'` * tier is reserved for Brainy's own plumbing and is not accepted here. Stored only when * not `'public'`. */ visibility?: 'public' | 'internal' /** Connection strength (0-1, default: 1.0) */ weight?: number /** Content for the relationship (optional β€” overrides auto-computed vector) */ data?: any /** * Structured queryable fields on the edge. Reserved relationship fields * (`RESERVED_RELATION_FIELDS` β€” `verb`, `subtype`, `visibility`, `createdAt`, * `updatedAt`, `confidence`, `weight`, `service`, `data`, `createdBy`, `_rev`) may NOT * appear here β€” they have dedicated params. A literal reserved key is a * compile error; untyped callers are normalized at write time. */ metadata?: RelationMetadataInput /** Create reverse edge too (default: false) */ bidirectional?: boolean /** Multi-tenancy service identifier */ service?: string /** Relationship certainty (0-1) */ confidence?: number /** Evidence for why this relationship exists */ evidence?: RelationEvidence } /** * Parameters for updating relationships */ export interface UpdateRelationParams { id: string // Relation to update type?: VerbType // Change verb type subtype?: string // Change sub-classification (omit to preserve existing) /** * Change the visibility tier. Omit to preserve the existing value. The verb mirror of * `UpdateParams.visibility`; `'system'` is Brainy-internal β€” consumer code should only * use `'public'` / `'internal'`. */ visibility?: EntityVisibility weight?: number // New weight confidence?: number // New confidence (0-1) data?: any // New content /** * Metadata fields to merge (or replace when `merge: false`). Reserved * relationship fields (`RESERVED_RELATION_FIELDS`) may NOT appear here β€” * a literal reserved key is a compile error; untyped callers are * normalized at write time. */ metadata?: RelationMetadataPatch merge?: boolean // Merge or replace metadata } // ============= Query Parameters ============= /** * Unified find parameters β€” Triple Intelligence search * * Combines three search dimensions in one query: * - **Vector:** `query` or `vector` for semantic/hybrid similarity search (searches `data`) * - **Metadata:** `where` for structured field filters (queries `metadata` via MetadataIndex) * - **Graph:** `connected` for relationship traversal (via GraphAdjacencyIndex) * * See also: [Query Operators](../../docs/QUERY_OPERATORS.md) for all `where` operators. */ export interface FindParams { // Vector Intelligence /** Natural language or semantic search query (embedded and matched via HNSW + text index) */ query?: string /** Direct vector search (pre-computed embedding) */ vector?: Vector // Metadata Intelligence /** Filter by entity type(s). Alias for `where.noun`. */ type?: NounType | NounType[] /** * Filter by per-product subtype (top-level standard field β€” uses the fast path, not the * metadata fallback). Pass a single string for equality, or an array for set membership * (e.g. `subtype: ['employee', 'contractor']`). For operator-form predicates (e.g. * `{ exists: true }`, `{ missing: true }`) use `where: { subtype: { …operators… } }`. */ subtype?: string | string[] /** Metadata filters using BFO operators (e.g., `{ year: { greaterThan: 2020 } }`) */ where?: Partial // Visibility /** * Also return `visibility: 'internal'` entities. By default `find()` returns ONLY * public entities (those with no `visibility` field, or `visibility: 'public'`); * app-internal entities are hidden. Set `true` to include them. Applied as a hard * candidate filter, so `limit` / `offset` stay correct. Does NOT include `'system'` * entities β€” use `includeSystem` for those. */ includeInternal?: boolean /** * Also return `visibility: 'system'` entities (Brainy's own plumbing, e.g. the VFS * root). Hidden by default and even when `includeInternal` is set. Set `true` only * when you specifically need to see system entities. Applied as a hard candidate * filter, so `limit` / `offset` stay correct. */ includeSystem?: boolean // Graph Intelligence connected?: GraphConstraints // Proximity search near?: { id: string // Find near this entity threshold?: number // Min similarity (0-1) } // Control options limit?: number // Max results (default: 10) offset?: number // Skip N results cursor?: string // Cursor-based pagination // Sorting orderBy?: string // Field to sort by (e.g., 'createdAt', 'title', 'metadata.priority') order?: 'asc' | 'desc' // Sort direction: 'asc' (default) or 'desc' // Advanced options mode?: SearchMode // Search strategy explain?: boolean // Return scoring explanation includeRelations?: boolean // Include entity relationships excludeVFS?: boolean // Exclude VFS entities from results (default: false - VFS included) service?: string // Multi-tenancy filter /** * Return each result's stored embedding under `entity.vector`. Defaults to `false`, * in which case `entity.vector` is an empty array (the perf default β€” vectors are * large and most reads don't need them). Set `true` to get the persisted vector * back from `find()` without a recompute, e.g. to feed downstream similarity math. * Mirrors {@link GetOptions.includeVectors} on `get()`. */ includeVectors?: boolean // Hybrid search options searchMode?: 'auto' | 'text' | 'semantic' | 'vector' | 'hybrid' // Search strategy: auto (default), text-only, semantic/vector-only, or explicit hybrid hybridAlpha?: number // Weight between text (0.0) and semantic (1.0) search, default: auto-detected by query length // Triple Intelligence Fusion fusion?: { strategy?: 'adaptive' | 'weighted' | 'progressive' weights?: { vector?: number graph?: number field?: number } } // Performance options writeOnly?: boolean // Skip validation for high-speed ingestion // Aggregation /** Query a named aggregate definition. String shorthand or full query params. */ aggregate?: string | AggregateQueryParams } /** * Graph constraints for search */ export interface GraphConstraints { to?: string // Connected to this entity from?: string // Connected from this entity via?: VerbType | VerbType[] // Via these relationship types type?: VerbType | VerbType[] // Alias for via /** * Filter traversal edges by VerbType subtype. Single string for equality, * array for set membership. Composes with `via` β€” `{ via: 'manages', subtype: * 'direct' }` traverses only direct-management edges, not dotted-line. Edges * without a subtype value are excluded when this filter is set. */ subtype?: string | string[] depth?: number // Max traversal depth (default: 1) direction?: 'in' | 'out' | 'both' // Direction of traversal (default: 'both') bidirectional?: boolean // Consider both directions } /** * Search modes */ export type SearchMode = | 'auto' // Automatically choose best mode (default - enables hybrid text+semantic) | 'vector' // Pure vector search (semantic only) | 'semantic' // Alias for vector search | 'text' // Pure text/keyword search | 'metadata' // Pure metadata filtering | 'graph' // Pure graph traversal | 'hybrid' // Combine all intelligences (explicit hybrid mode) /** * Parameters for similarity search */ export interface SimilarParams { to: string | Entity | Vector // Find similar to this limit?: number // Max results (default: 10) threshold?: number // Min similarity score type?: NounType | NounType[] // Restrict to types where?: Partial // Additional filters service?: string // Multi-tenancy excludeVFS?: boolean // Exclude VFS entities (default: false - VFS included) } /** * Parameters for `brain.related()` / `db.related()` * * All parameters are optional. When called without parameters, returns all relationships * with pagination (default limit: 100). * * @example * ```typescript * // Get all relationships (default limit: 100) * const all = await brain.related() * * // Get relationships from a specific entity (string shorthand) * const fromEntity = await brain.related(entityId) * * // Equivalent to: * const fromEntity2 = await brain.related({ from: entityId }) * * // Get relationships to a specific entity * const toEntity = await brain.related({ to: entityId }) * * // Filter by relationship type * const friends = await brain.related({ type: VerbType.FriendOf }) * * // Pagination * const page2 = await brain.related({ offset: 100, limit: 50 }) * * // Combined filters * const filtered = await brain.related({ * from: entityId, * type: VerbType.WorksWith, * limit: 20 * }) * ``` * * Fixed bug where calling without parameters returned empty array * Added string ID shorthand syntax */ export interface RelatedParams { /** * Filter by source entity ID * * Returns all relationships originating from this entity. */ from?: string /** * Filter by target entity ID * * Returns all relationships pointing to this entity. */ to?: string /** * Filter by INCIDENT entity ID β€” every relationship touching this entity in * EITHER direction (where it is the source OR the target), deduped. The * one-call "all edges on a noun" query: equivalent to merging * `related({ from: id })` and `related({ to: id })` but in a single * O(degree) call. Mutually exclusive with `from` / `to` (pass `node` for * both directions, or `from` / `to` for one). Composes with `type` / * `subtype` / visibility filters. * * @example * // Everything connected to this person, in or out. * const edges = await brain.related({ node: personId }) */ node?: string /** * Filter by relationship type(s) * * Can be a single VerbType or array of VerbTypes. */ type?: VerbType | VerbType[] /** * Filter by VerbType subtype * * Top-level standard field β€” uses the fast path, not the metadata fallback. * Pass a single string for equality (e.g. `subtype: 'direct'`), or an array * for set membership (e.g. `subtype: ['direct', 'dotted-line']`). */ subtype?: string | string[] /** * Also return `visibility: 'internal'` relationships. * * By default `related()` returns ONLY public relationships (those with no * `visibility` field, or `visibility: 'public'`); app-internal edges are hidden. * Set `true` to include them. Applied as a hard candidate filter so `limit` / * `offset` stay correct. Does NOT include `'system'` edges β€” use `includeSystem`. */ includeInternal?: boolean /** * Also return `visibility: 'system'` relationships (Brainy's own plumbing). * * Hidden by default and even when `includeInternal` is set. Set `true` only when * you specifically need system edges. Applied as a hard candidate filter so * `limit` / `offset` stay correct. */ includeSystem?: boolean /** * Maximum number of results to return * * @default 100 */ limit?: number /** * Number of results to skip (offset-based pagination) * * @default 0 */ offset?: number /** * Cursor for cursor-based pagination * * More efficient than offset for large result sets. */ cursor?: string /** * Filter by service (multi-tenancy) * * Only return relationships belonging to this service. */ service?: string } // ============= Graph views (brain.graph.*) ============= /** * A node in a {@link GraphView} β€” a lightweight reference (id + classification + * hop depth), NOT a full {@link Entity}. Graph reads are structure-first; resolve * a node to its full entity with `get(id)` when you need its `data` / metadata. */ export interface GraphNode { /** Entity id. */ id: string /** Entity type (present when the view hydrates node classification). */ type?: NounType /** Per-product subtype (present when hydrated). */ subtype?: string /** Hop distance from the nearest seed (0 = a seed); present on subgraph results. */ depth?: number } /** * A hydrated view of a region of the graph β€” the discovered nodes plus the edges * among them, in id form (the public face of the provider's columnar int form). */ export interface GraphView { /** The nodes in this region (seeds + everything reached). */ nodes: GraphNode[] /** The edges among `nodes`, as full {@link Relation} objects. */ edges: Relation[] /** `true` when a `maxNodes` / `maxEdges` cap truncated the result. */ truncated?: boolean } /** * Options for {@link GraphApi.subgraph}. */ export interface SubgraphOptions { /** Max hop distance from any seed (default `1`). */ depth?: number /** Edge direction to follow (default `'both'`). */ direction?: 'in' | 'out' | 'both' /** Restrict traversed edges to these relationship type(s). */ type?: VerbType | VerbType[] /** Restrict traversed edges to these subtype(s). */ subtype?: string | string[] /** Also traverse through `internal`-visibility edges/nodes (hidden by default). */ includeInternal?: boolean /** Also traverse through `system`-visibility edges/nodes (hidden by default). */ includeSystem?: boolean /** Cap on returned nodes (sets `GraphView.truncated`). */ maxNodes?: number /** Cap on returned edges. */ maxEdges?: number /** * Hydrate each node's `type` / `subtype` (one batch read). Default `true`; * set `false` to skip it when you only need graph structure (ids + edges). */ hydrateNodes?: boolean } /** * Options for {@link GraphApi.export}. */ export interface GraphExportOptions { /** Nodes/edges per streamed chunk (default `1000`). */ chunkSize?: number /** Also include `internal`-visibility nodes/edges (hidden by default). */ includeInternal?: boolean /** Also include `system`-visibility nodes/edges (hidden by default). */ includeSystem?: boolean /** Stream the nodes (default `true`; set `false` to stream only edges). */ includeNodes?: boolean /** Stream the edges (default `true`; set `false` to stream only nodes). */ includeEdges?: boolean } /** * Options for {@link GraphApi.communities}. */ export interface GraphCommunitiesOptions { /** Treat edges as directed when grouping (default `false` β€” direction-agnostic). */ directed?: boolean /** Also group through `internal`-visibility nodes/edges (hidden by default). */ includeInternal?: boolean /** Also group through `system`-visibility nodes/edges (hidden by default). */ includeSystem?: boolean } /** * Result of {@link GraphApi.communities} β€” the graph partitioned into connected * groups of entity ids. */ export interface GraphCommunitiesResult { /** Each group is the member entity ids; largest group first. */ groups: string[][] /** Number of distinct groups (`= groups.length`). */ count: number } /** * Options for {@link GraphApi.rank} (importance ranking). INTENT-level only β€” the * ranking ALGORITHM is the provider's choice (the TS fallback uses PageRank), so * no algorithm-tuning knobs are exposed. */ export interface GraphRankOptions { /** Return only the top-K nodes by score (default: all, descending). */ topK?: number /** Also rank through `internal`-visibility nodes/edges (hidden by default). */ includeInternal?: boolean /** Also rank through `system`-visibility nodes/edges (hidden by default). */ includeSystem?: boolean } /** One ranked entity from {@link GraphApi.rank}, descending by `score`. */ export interface GraphRankEntry { /** The entity id. */ id: string /** The importance score (relative; higher = more central). */ score: number } /** * Options for {@link GraphApi.path} (best route between two entities). */ export interface GraphPathOptions { /** Edge direction to follow (default `'both'`). */ direction?: 'in' | 'out' | 'both' /** Optimize for fewest `'hops'` (default) or least summed edge `'weight'`. */ by?: 'hops' | 'weight' /** Restrict the route to these relationship type(s). */ type?: VerbType | VerbType[] /** Also route through `internal`-visibility nodes/edges (hidden by default). */ includeInternal?: boolean /** Also route through `system`-visibility nodes/edges (hidden by default). */ includeSystem?: boolean /** Abandon the search past this many hops. */ maxDepth?: number } /** * Result of {@link GraphApi.path} β€” the route from `from` to `to`, or `null` when * unreachable. */ export interface GraphPathResult { /** The entity ids along the route, `from` … `to` inclusive. */ nodes: string[] /** The relationship (verb) ids traversed, one fewer than `nodes`. */ relationships: string[] /** Route cost: number of hops, or summed edge weight when `by: 'weight'`. */ cost: number } /** * The `brain.graph` namespace β€” graph-shaped reads over the knowledge graph. * Routes to a native {@link import('../plugin.js').GraphAccelerationProvider} * when one is registered, otherwise serves the same results from Brainy's * pure-TS adjacency (correct at small/medium scale). */ export interface GraphApi { /** * @description Extract the subgraph reachable from one or more seed entities β€” * the bounded multi-hop neighborhood, as nodes + edges. The one-call answer to * "show me everything around this node, N hops out". * @param seeds - A seed entity id, or an array of them. * @param options - Depth, direction, edge/visibility filters, and caps. * @returns The hydrated {@link GraphView}. * @example * const view = await brain.graph.subgraph(personId, { depth: 2 }) * // view.nodes = the 2-hop neighborhood; view.edges = the relations among them */ subgraph(seeds: string | string[], options?: SubgraphOptions): Promise> /** * @description Stream the WHOLE graph in one O(N+E) pass as a sequence of * {@link GraphView} chunks β€” the right primitive for visualizing all data * (vs. paging per node). Node chunks come first, then edge chunks; assemble * them on the consumer side. Backed by cursor pagination, so a full walk is * O(N+E) at any chunk size (no re-scan). * @param options - Chunk size, visibility, and node/edge inclusion toggles. * @returns An async-iterable of graph chunks. * @example * for await (const chunk of brain.graph.export()) { * addNodes(chunk.nodes); addEdges(chunk.edges) * } */ export(options?: GraphExportOptions): AsyncIterable> /** * @description Rank entities by graph importance (influence / centrality) β€” the * one-call answer to "which nodes matter most". The TS fallback uses PageRank; * a native provider may use personalized PageRank / eigenvector centrality. * @param options - `topK`, visibility opt-ins. * @returns Entities with scores, DESCENDING (most important first). * @example * const top = await brain.graph.rank({ topK: 10 }) * top[0] // { id, score } β€” the most central entity */ rank(options?: GraphRankOptions): Promise /** * @description Partition the graph into connected communities (clusters of * related entities) β€” "which things group together". * @param options - `directed`, visibility opt-ins. * @returns The groups (member ids) + their count. * @example * const { groups, count } = await brain.graph.communities() */ communities(options?: GraphCommunitiesOptions): Promise /** * @description Find the best route between two entities β€” fewest hops (default) * or least summed edge weight (`by: 'weight'`). The pathfinding ALGORITHM is * the provider's choice (TS fallback: BFS for hops, Dijkstra for weight). * @param from - Start entity id (natural keys are resolved). * @param to - End entity id. * @param options - Direction, `by`, type filter, `maxDepth`, visibility opt-ins. * @returns The route (`nodes` + `relationships` + `cost`), or `null` if unreachable. * @example * const route = await brain.graph.path(aliceId, bobId) * route?.nodes // [aliceId, …, bobId] */ path(from: string, to: string, options?: GraphPathOptions): Promise } // ============= Batch Operations ============= /** * Batch add parameters */ export interface AddManyParams { items: AddParams[] // Items to add parallel?: boolean // Process in parallel (default: true) chunkSize?: number // Batch size (default: 100) onProgress?: (done: number, total: number) => void continueOnError?: boolean // Continue if some fail /** * Conditional insert applied to every item. Equivalent to setting `ifAbsent: true` * on each item individually. Item-level `ifAbsent` overrides the batch flag. */ ifAbsent?: boolean /** * Create-or-update applied to every item. Equivalent to setting `upsert: true` on * each item individually (see {@link AddParams.upsert}): an item whose custom `id` * already exists is MERGED as an update rather than overwritten. Item-level `upsert` * overrides the batch flag. Mutually exclusive with {@link AddManyParams.ifAbsent} * at the item level β€” an item resolving to both throws. */ upsert?: boolean } /** * Batch update parameters */ export interface UpdateManyParams { items: UpdateParams[] // Items to update parallel?: boolean chunkSize?: number onProgress?: (done: number, total: number) => void continueOnError?: boolean } /** * Batch remove parameters */ export interface RemoveManyParams { ids?: string[] // Specific IDs to remove type?: NounType // Remove all of type where?: any // Remove by metadata limit?: number // Max to remove (safety) /** * Entities removed per transaction chunk. Each chunk is one atomic transaction β€” * a chunk commits or rolls back together, and failures are isolated to their chunk. * Defaults to the storage adapter's `maxBatchSize` (memory: 1000, S3/R2: 100, * GCS: 50), so zero-config removals match the adapter's characteristics. Override * only to tune throughput vs. transaction granularity. */ chunkSize?: number onProgress?: (done: number, total: number) => void continueOnError?: boolean // Continue processing if a removal fails } /** * Batch relate parameters */ export interface RelateManyParams { items: RelateParams[] // Relations to create parallel?: boolean chunkSize?: number onProgress?: (done: number, total: number) => void continueOnError?: boolean } /** * Batch result */ export interface BatchResult { successful: T[] // Successfully processed items failed: Array<{ // Failed items with errors item: any error: string }> total: number // Total attempted duration: number // Time taken in ms } // ============= Import Progress ============= /** * Import stage enumeration */ export type ImportStage = | 'detecting' // Detecting file format | 'reading' // Reading file from disk/network | 'parsing' // Parsing file structure (CSV rows, PDF pages, Excel sheets) | 'extracting' // Extracting entities using AI | 'indexing' // Creating graph nodes and relationships | 'completing' // Final cleanup and stats /** * Overall import status */ export type ImportStatus = | 'starting' // Initializing import | 'processing' // Actively importing | 'completing' // Finalizing | 'done' // Complete /** * Comprehensive import progress information * * Provides multi-dimensional progress tracking: * - Bytes processed (always deterministic) * - Entities extracted and indexed * - Stage-specific progress * - Time estimates * - Performance metrics * */ export interface ImportProgress { // Overall Progress overall_progress: number // 0-100 weighted estimate across all stages overall_status: ImportStatus // High-level status // Current Stage stage: ImportStage // What's happening now stage_progress: number // 0-100 within current stage (0 if unknown) stage_message: string // Human-readable: "Extracting entities from PDF..." // Bytes (Always Available - most deterministic metric) bytes_processed: number // Bytes read/processed so far total_bytes: number // Total file size (0 if streaming/unknown) bytes_percentage: number // Convenience: bytes_processed / total_bytes * 100 bytes_per_second?: number // Processing rate (relevant during parsing) // Entities (Available when extraction starts) entities_extracted: number // Entities found during AI extraction entities_indexed: number // Entities added to Brainy graph entities_per_second?: number // Entities/sec (relevant during extraction/indexing) estimated_total_entities?: number // Estimated final count estimation_confidence?: number // 0-1 confidence in estimation // Timing elapsed_ms: number // Time since import started estimated_remaining_ms?: number // Estimated time remaining estimated_total_ms?: number // Estimated total time // Context (helps users understand what's happening) current_item?: string // "Processing page 5 of 23" current_file?: string // "Sheet: Q2 Sales Data" file_number?: number // 3 (when importing multiple files) total_files?: number // 10 // Performance Metrics (for debugging/optimization) metrics?: { parsing_rate_mbps?: number // MB/s during parsing extraction_rate_entities_per_sec?: number // Entities/s during extraction indexing_rate_entities_per_sec?: number // Entities/s during indexing memory_usage_mb?: number // Current memory usage peak_memory_mb?: number // Peak memory usage } // Backwards Compatibility (for legacy code) current: number // Alias for entities_indexed total: number // Alias for estimated_total_entities or 0 } /** * Import progress callback - backwards compatible * * Supports both legacy (current, total) and new (ImportProgress object) signatures */ export type ImportProgressCallback = | ((progress: ImportProgress) => void) | ((current: number, total: number) => void) /** * Stage weight configuration for overall progress calculation * * These weights reflect the typical time distribution across stages. * Extraction is typically the slowest stage (60% of time). */ export interface StageWeights { detecting: number // Default: 0.01 (1%) reading: number // Default: 0.05 (5%) parsing: number // Default: 0.10 (10%) extracting: number // Default: 0.60 (60% - slowest!) indexing: number // Default: 0.20 (20%) completing: number // Default: 0.04 (4%) } /** * Import result statistics */ export interface ImportStats { graphNodesCreated: number // Entities added to graph graphEdgesCreated: number // Relationships created vfsFilesCreated: number // VFS files created duration: number // Total time in ms bytesProcessed: number // Total bytes read averageRate: number // Average entities/sec peakMemoryMB?: number // Peak memory usage } /** * Import operation result */ export interface ImportResult { success: boolean stats: ImportStats errors?: Array<{ stage: ImportStage message: string error?: any }> } // ============= Advanced Operations ============= /** * Options for brain.get() entity retrieval * * **Performance Optimization**: * By default, brain.get() loads ONLY metadata (not vectors), resulting in: * - **76-81% faster** reads (10ms vs 43ms for metadata-only) * - **95% less bandwidth** (300 bytes vs 6KB per entity) * - **87% less memory** (optimal for VFS and large-scale operations) * * **When to use includeVectors**: * - Computing similarity on a specific entity (not search): `brain.similar({ to: entity.vector })` * - Manual vector operations: `cosineSimilarity(entity.vector, otherVector)` * - Inspecting embeddings for debugging * * **When NOT to use includeVectors** (metadata-only is sufficient): * - VFS operations (readFile, stat, readdir) - 100% of cases * - Existence checks: `if (await brain.get(id))` * - Metadata inspection: `entity.metadata`, `entity.data`, `entity.type` * - Relationship traversal: `brain.related({ from: id })` * - Search operations: `brain.find()` generates embeddings automatically * * @example * ```typescript * // βœ… FAST (default): Metadata-only - 10ms, 300 bytes * const entity = await brain.get(id) * console.log(entity.data, entity.metadata) // βœ… Available * console.log(entity.vector) // Empty Float32Array (stub) * * // βœ… FULL: Load vectors when needed - 43ms, 6KB * const fullEntity = await brain.get(id, { includeVectors: true }) * const similarity = cosineSimilarity(fullEntity.vector, otherVector) * * // βœ… VFS automatically uses fast path (no change needed) * await vfs.readFile('/file.txt') // 53ms β†’ 10ms (81% faster) * ``` * */ export interface GetOptions { /** * Include 384-dimensional vector embeddings in the response * * **Default: false** (metadata-only for 76-81% speedup) * * Set to `true` when you need to: * - Compute similarity on this specific entity's vector * - Perform manual vector operations * - Inspect embeddings for debugging * * **Note**: Search operations (`brain.find()`) generate vectors automatically, * so you don't need this flag for search. Only for direct vector operations * on a retrieved entity. * * @default false */ includeVectors?: boolean } /** * Graph traversal parameters */ export interface TraverseParams { from: string | string[] // Starting node(s) direction?: 'out' | 'in' | 'both' // Traversal direction types?: VerbType[] // Edge types to follow depth?: number // Max depth (default: 2) strategy?: 'bfs' | 'dfs' // Breadth or depth first filter?: (entity: Entity, depth: number, path: string[]) => boolean limit?: number // Max nodes to visit } // ============= Aggregation Engine Types ============= /** * Supported aggregation operations */ export type AggregationOp = | 'sum' | 'count' | 'avg' | 'min' | 'max' | 'stddev' | 'variance' /** Exact percentile (requires `p` in `[0,1]` on the metric def) β€” value-multiset, delete-safe. */ | 'percentile' /** Exact count of distinct values β€” value-multiset, delete-safe. */ | 'distinctCount' /** * Time window granularity for GROUP BY time dimensions */ export type TimeWindowGranularity = | 'hour' | 'day' | 'week' | 'month' | 'quarter' | 'year' | { seconds: number } /** * A GROUP BY dimension β€” one of: * - a plain metadata field name (string), * - a time-windowed field (`{ field, window }`), or * - an **unnest** field (`{ field, unnest: true }`): the field holds an array, and the entity * contributes once to a group per distinct element (e.g. `tags: string[]` β†’ tag frequency). * An entity whose unnest field is missing or an empty array contributes to no group. */ export type GroupByDimension = | string | { field: string; window: TimeWindowGranularity } | { field: string; unnest: true } /** * Source filter for which entities feed into an aggregate */ export interface AggregateSource { /** Filter by entity type(s) */ type?: NounType | NounType[] /** Metadata filter (same syntax as find({ where })) */ where?: Record /** Multi-tenancy service filter */ service?: string } /** * Full aggregate definition β€” registered via brain.defineAggregate() */ export interface AggregateDefinition { /** Unique name for this aggregate (used in queries) */ name: string /** Which entities contribute to this aggregate */ source: AggregateSource /** Dimensions to group by */ groupBy: GroupByDimension[] /** Named metrics to compute */ metrics: Record /** Control materialization of results as NounType.Measurement entities */ materialize?: boolean | { debounceMs?: number; trackSources?: boolean } } /** * Single metric definition within an aggregate */ export interface AggregateMetricDef { /** Aggregation operation */ op: AggregationOp /** Metadata field to aggregate (required for all ops except count) */ field?: string /** Percentile fraction in `[0, 1]` β€” required when `op === 'percentile'`, ignored otherwise. */ p?: number } /** * Internal running state for a single metric. * Tracks enough to compute all operations incrementally. */ export interface MetricState { sum: number count: number min: number max: number /** Running M2 for Welford's online variance (sum of squared differences from mean) */ m2?: number /** * Value multiset (String(value) β†’ occurrence count) for exact percentile + distinctCount. * Maintained only for `percentile`/`distinctCount` metrics. JSON-serializable; mirrors * Cortex's `value_counts` so JS and native agree bit-for-bit. */ valueCounts?: Record } /** * Internal state for one aggregate group (one combination of group key values) */ export interface AggregateGroupState { /** The group key values (e.g., { category: 'food', period: '2024-01' }) */ groupKey: Record /** Running metric states keyed by metric name */ metrics: Record /** Entity ID of the materialized Measurement entity (if materialized) */ materializedEntityId?: string /** Timestamp of last update */ lastUpdated: number } /** * Query parameters for reading aggregate results */ export interface AggregateQueryParams { /** Name of the aggregate to query */ name: string /** Filter aggregate groups by their key values */ where?: Record /** * Filter groups by their computed METRIC values (SQL HAVING). Same BFO operators as * `where`, but applied to the derived metric results plus `count`, e.g. * `{ revenue: { greaterThan: 1000 } }`. Evaluated per group (O(groups), independent of * entity count), before sort/pagination. */ having?: Record /** Sort by metric name or group key field */ orderBy?: string /** Sort direction */ order?: 'asc' | 'desc' /** Max results */ limit?: number /** Skip N results */ offset?: number } /** * A single aggregate result row */ export interface AggregateResult { /** Group key values for this row */ groupKey: Record /** Computed metric values (derived from MetricState based on op) */ metrics: Record /** Total entity count in this group */ count: number /** Entity ID of materialized Measurement (if materialized) */ entityId?: string } /** * Provider interface for Cortex-accelerated aggregation. * When registered as 'aggregation' provider, Brainy delegates to this. */ export interface AggregationProvider { /** Register an aggregate definition (caches compiled definition for hot path) */ defineAggregate?(def: AggregateDefinition): void /** Remove a registered aggregate definition */ removeAggregate?(name: string): void /** Incrementally update aggregation state when an entity changes */ incrementalUpdate( name: string, def: AggregateDefinition, entity: Record, op: 'add' | 'update' | 'delete', prev?: Record ): AggregateGroupState[] /** Compute the group key for a given entity */ computeGroupKey( entity: Record, groupBy: GroupByDimension[] ): Record /** Rebuild an entire aggregate from scratch */ rebuildAggregate( def: AggregateDefinition, entities: Array> ): Map /** Query aggregate state with filtering/sorting/pagination */ queryAggregate( state: Map, params: AggregateQueryParams ): AggregateResult[] /** Restore previously serialized state (called during init) */ restoreState?(data: string): void /** Serialize internal state for persistence (called during flush) */ serializeState?(): string } // ============= Configuration ============= /** * Integration Hub configuration * * Enables external tool integrations: Excel, Power BI, Google Sheets, etc. * Works in all environments with zero external dependencies. * * @example * ```typescript * // Enable all integrations with defaults * new Brainy({ integrations: true }) * * // Custom configuration * new Brainy({ * integrations: { * basePath: '/api/v1', * enable: ['odata', 'sheets'] * } * }) * ``` */ export interface IntegrationsConfig { /** Base path for all integration endpoints (default: '') */ basePath?: string /** Which integrations to enable (default: all) */ enable?: ('odata' | 'sheets' | 'sse' | 'webhooks')[] | 'all' /** Per-integration config overrides */ config?: { odata?: { basePath?: string } sheets?: { basePath?: string } sse?: { basePath?: string; heartbeatInterval?: number } webhooks?: { maxRetries?: number } } } /** * Operator-facing summary returned by `brain.stats()`. Designed to fit in a * terminal screen so an incident responder can read it at a glance: counts, * mode, lock owner, indexed fields, and index health flags. */ export interface BrainyStats { /** Whether this instance can mutate. `reader` is set by `openReadOnly()` and `asOf()`. */ mode: 'writer' | 'reader' /** Total entity (noun) count across all types. */ entityCount: number /** Breakdown by NounType name (`person`, `event`, ...). Zero-count types omitted. */ entitiesByType: Record /** Total relationship (verb) count across all types. */ relationCount: number /** Breakdown by VerbType name. Zero-count types omitted. */ relationsByType: Record /** Indexed metadata field names known to the metadata index. */ fieldRegistry: string[] /** Per-index health flags. `true` = index has entries OR no entities exist yet. */ indexHealth: { /** Vector index health (8.0 β€” open-core JS HNSW path or a native acceleration provider). */ vector: boolean metadata: boolean graph: boolean } /** Storage backend info β€” `backend` is the adapter class name. */ storage: { backend: string rootDir?: string } /** * Writer lock metadata if one is currently held on this directory. Present * for both writer instances (their own lock) and readers inspecting a * directory with a live writer. */ writerLock?: { pid: number hostname: string startedAt: string lastHeartbeat: string version: string rootDir?: string } /** The Brainy library version this instance was built against. */ version: string } /** * Brainy configuration */ export interface BrainyConfig { /** * Storage configuration: either a config object resolved through the * storage factory, or a pre-constructed adapter instance (used directly β€” * e.g. `storage: new MemoryStorage()`; Brainy's historical-query * materializer hands a pre-populated instance through this path). */ storage?: | { /** * Storage backend. Optional β€” a top-level `path` implies `'filesystem'`, * so `storage: { path: '/data' }` works without it. `'auto'` (the * default) picks filesystem on Node, memory in a browser. */ type?: 'auto' | 'memory' | 'filesystem' /** * **Canonical** directory for filesystem storage. The rest of the API * already speaks `path` (`persist(path)`, `Brainy.load(path)`, * `asOf(path)`, `restore(path)`). Specifying it implies * `type: 'filesystem'`. Passed through to plugin-provided storage * factories (e.g. native mmap providers) so they resolve the same root. * @example * new Brainy({ storage: { path: '/var/lib/app/data' } }) */ path?: string /** * @deprecated REMOVED in 8.0 β€” passing `rootDirectory` THROWS. Use the * canonical {@link path}. */ rootDirectory?: string /** * @deprecated REMOVED in 8.0 β€” a nested `options.path` / `options.rootDirectory` * THROWS. Use the top-level {@link path}. */ options?: any /** * @deprecated REMOVED in 8.0 β€” `fileSystemStorage.path` / `.rootDirectory` * THROWS. Use the top-level {@link path}. */ fileSystemStorage?: { path?: string; rootDirectory?: string; [key: string]: any } } | StorageAdapter /** * Disable the automatic index rebuild check during `init()`. By default * Brainy auto-decides from dataset size: small datasets rebuild missing * indexes inline, large datasets rebuild lazily on first query. Set `true` * only when an operator wants full manual control via `repairIndex()`. */ disableAutoRebuild?: boolean /** * Vector index configuration (Brainy 8.0). * * Two knobs. No escape hatch. The algorithm-internal HNSW knobs * (`M`, `efConstruction`, `efSearch`, `ml`, …) and DiskANN knobs * (`pqM`, `searchListSize`, `alpha`, …) are deliberately not exposed β€” * the `recall` preset covers the legitimate quality/latency tradeoff * range, and the open-core defaults match Brainy 7.x's defaults * byte-for-byte so a `'balanced'`-default upgrade is a no-op. * * **Closed-form contract** locked in handoff thread * BRAINY-8.0-RENAME-COORDINATION Β§ A.2 + Β§ G.2 (cortex-confirmed 2026-06-09). */ vector?: { /** * Recall preset. * - `'fast'` β€” minimum-latency search, accepts lower recall * - `'balanced'` (default) β€” Brainy 7.x defaults, the right pick for almost everyone * - `'accurate'` β€” maximum recall, accepts higher latency * * Means the same thing whether the open-core JS HNSW path is in play * or a native vector provider has taken over the `'vector'` provider * key β€” Brainy translates to HNSW knobs; native providers translate * to their own (e.g. DiskANN's `defaultLSearch` / `defaultPaddingFactor`). */ recall?: 'fast' | 'balanced' | 'accurate' /** * Vector persistence mode. `'immediate'` writes per-noun graph state on * every `add()`; durable but slower. `'deferred'` writes only on * `flush()` / `close()`; faster bulk ingest. * * **Auto-selected from the storage adapter when omitted:** `'immediate'` * on filesystem storage (durability is the point of a persistent * backend), `'deferred'` on memory storage (nothing survives the process * anyway, so per-add persistence writes are pure overhead). * * Brainy 7.x called this `hnswPersistMode` at the top level. The 8.0 * surface folds it under `config.vector` alongside `recall`. */ persistMode?: 'immediate' | 'deferred' } /** * Generational-history retention policy (8.0 MVCC β€” the `retention` knob). * * Under Model-B EVERY write (`transact()` AND single-op `add`/`update`/ * `remove`/`relate`) produces an immutable generation record-set serving * historical reads (`asOf()`, pinned `Db` values). Without compaction those * accumulate, so Brainy **auto-compacts on every `flush()` and `close()`**. * Live `Db` pins are ALWAYS exempt from reclamation, in every mode. * * Modes: * - **unset β†’ ADAPTIVE (default):** the retention horizon tracks * disk/RAM pressure. A machine-level byte budget governs how much history * is kept β€” driven by `budgetBytes` when a coordinator (e.g. cor's * `ResourceManager`, fair-shared across co-located instances) sets it, else * a local `os.freemem`/filesystem-free probe. Oldest-unpinned history is * reclaimed when the budget is exceeded. Zero-config. * - **`'all'` β†’ unbounded:** never reclaim history (index compaction for * query speed still runs β€” the decouple). The legacy keep-everything * behavior, now explicit opt-in. * - **`{ maxGenerations?, maxAge?, maxBytes? }` β†’ explicit CAPS:** reclaim * the oldest unpinned generations while ANY supplied cap is exceeded * (predictable ops). `maxAge` in ms; `maxBytes` total history bytes. * * `autoCompact: false` disables the automatic flush/close compaction (manage * manually via `brain.compactHistory()`). `budgetBytes` is the settable * adaptive byte budget a coordinator drives (also via `brain.setRetentionBudget()`). * Long-term archives belong in `db.persist(path)` snapshots, which compaction * never touches. */ retention?: | 'all' | 'adaptive' | { /** Keep at most this many recent committed generations (cap). */ maxGenerations?: number /** Keep generations committed within this window in ms (cap). */ maxAge?: number /** Keep total generational-history bytes at or below this (cap). */ maxBytes?: number /** Adaptive byte budget for this brain, driven by a coordinator (e.g. cor). */ budgetBytes?: number /** Run compaction automatically on flush()/close() (default: true). */ autoCompact?: boolean } // Memory management options maxQueryLimit?: number // Override auto-detected query result limit (max: 100000) reservedQueryMemory?: number // Memory reserved for queries in bytes (e.g., 1073741824 = 1GB) /** * Controls when the WASM embedding engine is initialized. * * **Adaptive default (8.0):** when omitted, the engine eagerly initializes * during `init()` whenever the WASM embedder is the *active* one β€” i.e. no * native `'embeddings'` provider is registered β€” and this instance is a * writer (not `mode: 'reader'`) running outside unit tests. The WASM module * (β‰ˆ93MB with the embedded model) takes 90-140s to compile on throttled * CPUs, so paying that during boot rather than on the first `embed()`-driven * call is the right default for a single-process server. * * The adaptive path skips itself automatically when a native embeddings * provider owns embeddings, in reader-mode (readers query existing vectors * and never embed), and in unit-test mode (kept fast via the mock embedder). * * - `true` β€” force eager init during `init()` (the adaptive default already * does this for the active-embedder writer case; set it explicitly to be * unambiguous). * - `false` β€” explicit override to force lazy init (first `embed()` call) * even when this instance is the active embedder. */ eagerEmbeddings?: boolean // Plugin configuration // Controls which plugins are loaded during init() // - undefined (default): Auto-detect installed plugins (@soulcraft/cor, etc.) // - false: No plugins β€” skip auto-detection entirely // - []: No plugins β€” skip auto-detection entirely // - ['@soulcraft/cor']: Load only specified plugins, no auto-detection plugins?: string[] | false // Logging configuration verbose?: boolean // Enable verbose logging silent?: boolean // Suppress all logging output // Integration Hub // Enable external tool integrations: Excel, Power BI, Google Sheets, etc. // - true: Enable all integrations with default paths // - false/undefined: Disable integrations (default) // - IntegrationsConfig: Custom configuration integrations?: boolean | IntegrationsConfig // Migration configuration // - true/undefined (default): Automatically run pending data migrations during // init() for small datasets (<10K entities). Larger datasets defer with a // notice β€” call brain.migrate() explicitly (optionally with backupTo). // - false: Never auto-run; log a notice when pending migrations exist. autoMigrate?: boolean /** * Brain-wide subtype enforcement mode. * * **Default-on since 8.0.0** (`undefined` resolves to `true`; in 7.30.x this * was an opt-in flag defaulting to `false`). * * - `true` / `undefined` (default): every `add()` / `addMany()` / `update()` / * `relate()` / `relateMany()` / `updateRelation()` rejects writes where the * entity's NounType (or relationship's VerbType) has no non-empty `subtype` * value. Brainy's own VFS infrastructure writes (`metadata.isVFSEntity` / * `metadata.isVFS`) bypass the check. * - `{ except: [NounType.Thing, NounType.Custom] }`: same as `true`, but the * listed types are allowed through without a subtype. Use for genuine * catch-all types where no subtype makes sense. * - `false`: disable the brain-wide check entirely. Last-resort escape hatch * for opening pre-8.0 data β€” run `brain.audit()` to find the gaps, back-fill * with `brain.fillSubtypes(rules)`, then remove the opt-out so the default * enforcement protects new writes. * * Per-type registrations always compose with the brain-wide flag β€” a type * registered with `requireSubtype(type, { required: true })` is always * enforced regardless of this flag. */ requireSubtype?: boolean | { except: Array } /** * Process role for multi-process safety on filesystem storage. * * - `'writer'` (default): acquires an exclusive lock on the storage directory at * `init()` and refuses to open if another live writer holds the lock. Required * for any instance that calls `add`/`update`/`remove`/`relate` or any other * mutation. Released on `close()` and on process exit/SIGINT/SIGTERM. * - `'reader'`: opens without acquiring the writer lock. Coexists with a live * writer and with other readers. All mutation methods throw * `Cannot mutate a read-only Brainy instance`. Use `Brainy.openReadOnly()` * as a convenience factory. * * See `docs/concepts/multi-process.md` for the full coordination model. */ mode?: 'writer' | 'reader' /** * Bypass the writer-lock check at init even when another writer holds the lock. * Use only when you know the existing lock is stale and stale-detection * (PID liveness + heartbeat) cannot prove it. Logs a warning regardless. */ force?: boolean /** * How write paths react when an untyped (JavaScript) caller smuggles a * Brainy-reserved field (`RESERVED_ENTITY_FIELDS` / `RESERVED_RELATION_FIELDS` * β€” `confidence`, `weight`, `subtype`, `visibility`, `service`, `createdBy`, * `noun`/`verb`, `data`, `createdAt`, `updatedAt`, `_rev`) **inside the * `metadata` bag** of `add()` / `update()` / `relate()` / `updateRelation()` * (and their `transact()` / `with()` mirrors). TypeScript callers can't write * these shapes at all β€” the compile-time guard on the metadata param types * (`NoReservedEntityKeys` / `NoReservedRelationKeys`) rejects a literal * reserved key β€” so this policy only governs untyped callers that slip one * past the compiler. * * - `'throw'` (**default, 8.0**): a reserved key in the bag throws a clear * `Error` naming the offending key(s) and the correct write path. No silent * remap, no data loss, no surprise. This is the 8.0 "no silent failures" * contract. * - `'warn'`: legacy remapping with a loud, one-shot (per key, per process) * warning for EVERY reserved key found β€” user-mutable fields are remapped to * their dedicated top-level param (top-level wins when both are supplied), * system-managed fields are dropped. Use while migrating untyped call sites. * - `'remap'`: the pre-8.0 silent remapping, no warning. Last-resort * compatibility hatch for code that intentionally relies on the bag path. * * @default 'throw' */ reservedFieldPolicy?: 'throw' | 'warn' | 'remap' } // ============= Neural API Types ============= /** * Neural similarity parameters */ export interface NeuralSimilarityParams { between?: [any, any] // Compare two items items?: any[] // Compare multiple items explain?: boolean // Return detailed breakdown } /** * Neural clustering parameters */ export interface NeuralClusterParams { items?: string[] | Entity[] // Items to cluster (or all) algorithm?: 'hierarchical' | 'kmeans' | 'dbscan' | 'spectral' params?: { k?: number // Number of clusters (kmeans) threshold?: number // Distance threshold (hierarchical) epsilon?: number // DBSCAN epsilon minPoints?: number // DBSCAN min points } visualize?: boolean // Return visualization data } /** * Neural anomaly detection parameters */ export interface NeuralAnomalyParams { threshold?: number // Standard deviations (default: 2.5) type?: NounType // Check specific type method?: 'isolation' | 'lof' | 'statistical' | 'autoencoder' returnScores?: boolean // Return anomaly scores } // ============= Content Extraction Types ============= /** * Detected content type for smart text extraction * */ export type ContentType = 'plaintext' | 'richtext-json' | 'html' | 'markdown' /** * Content category for extracted segments * * Universal categories that work across documents, code, and UI: * - 'title': Names the subject β€” headings, identifiers, labels, JSON keys * - 'annotation': Human explanation β€” comments, docstrings, captions, alt text * - 'content': Body substance β€” paragraphs, list items, flowing text * - 'value': Data literals β€” strings, numbers, form values, error messages * - 'code': Unparsed code blocks (custom parsers decompose into above) * - 'structural': Boilerplate β€” keywords, operators, punctuation, formatting * * Built-in extractors produce: 'title', 'content', 'code'. * All 6 categories are available for custom parsers. */ export type ContentCategory = 'title' | 'annotation' | 'content' | 'value' | 'code' | 'structural' /** * A segment of extracted text with its content category * */ export interface ExtractedSegment { /** The extracted text content */ text: string /** What role this text plays: 'title' | 'annotation' | 'content' | 'value' | 'code' | 'structural' */ contentCategory: ContentCategory } // ============= Semantic Highlighting Types ============= /** * Parameters for hybrid highlighting * * Zero-config highlighting that returns both text (exact) and semantic (concept) matches. * Perfect for UI highlighting at different levels. * * Added contentType hint and contentExtractor callback for structured text * (rich-text JSON, HTML, Markdown). Auto-detects format when not specified. * * @example * ```typescript * // Plain text * const highlights = await brain.highlight({ * query: "david the warrior", * text: "David Smith is a brave fighter who battles dragons" * }) * * // Rich-text JSON (auto-detected) * const highlights = await brain.highlight({ * query: "david the warrior", * text: JSON.stringify(tiptapDocument) * }) * * // Custom extractor (for proprietary formats) * const highlights = await brain.highlight({ * query: "function", * text: sourceCode, * contentExtractor: (text) => treeSitterParse(text) * }) * ``` */ export interface HighlightParams { /** The search query to match against */ query: string /** The text to highlight (e.g., entity.data) */ text: string /** Granularity of highlighting: 'word' (default), 'phrase', or 'sentence' */ granularity?: 'word' | 'phrase' | 'sentence' /** Minimum semantic similarity score for semantic matches (default: 0.5) */ threshold?: number /** * Optional content type hint to skip auto-detection. * When omitted, the content type is detected from the text content. */ contentType?: ContentType /** * Optional custom content extractor function. * When provided, bypasses built-in detection and extraction entirely. * Use this to plug in custom parsers (tree-sitter, Monaco, proprietary formats). */ contentExtractor?: (text: string) => ExtractedSegment[] } /** * A highlight showing which text matched the query * * matchType tells the UI how to style the highlight: * - 'text': Exact word match (strongest signal, highest confidence) * - 'semantic': Conceptually similar match (may need softer highlight) */ export interface Highlight { /** The text that matched */ text: string /** Match score (0-1). For text matches, always 1.0. For semantic, varies by similarity. */ score: number /** Position in original text [start, end] */ position: [number, number] /** Match type: 'text' (exact word match) or 'semantic' (concept match) */ matchType: 'text' | 'semantic' /** * Content category of the source segment: 'title' | 'annotation' | 'content' | 'value' | 'code' | 'structural'. * Present when the input text was structured (JSON, HTML, Markdown). */ contentCategory?: ContentCategory } // ============= Export all types ============= export * from './graphTypes.js' // Re-export NounType, VerbType, etc.