diff --git a/CHANGELOG.md b/CHANGELOG.md
index fd0de54e..b6dd91fd 100644
--- a/CHANGELOG.md
+++ b/CHANGELOG.md
@@ -2,6 +2,15 @@
All notable changes to this project will be documented in this file. See [standard-version](https://github.com/conventional-changelog/standard-version) for commit guidelines.
+### [8.10.0](https://github.com/soulcraftlabs/brainy/compare/v8.9.0...v8.10.0) (2026-07-23)
+
+- docs: adoption storefront — contributing guide, security policy, README support + cor section (9a99a7b)
+- fix(release): push the public mirror explicitly and verify the tag lands at the right commit before publishing (6ba94c8)
+- docs: project guide version line points at npm instead of a hardcoded stale number (3a1efc9)
+- feat: vector provider identity is a required name field (hnsw-js), rendered [vector-index:] (3be4ba9)
+- feat: warm contract (warm/warmOnOpen/provider warm hook), configurable transact budget floor, backend-neutral vector index op names (55b867c)
+
+
### [8.9.0](https://github.com/soulcraftlabs/brainy/compare/v8.8.2...v8.9.0) (2026-07-19)
- docs: measured performance envelopes v1 (per-op p50/p95 at 1k and 10k, pure-JS floor) (5cabd78)
diff --git a/CLAUDE.md b/CLAUDE.md
index 568c10db..c7336a18 100644
--- a/CLAUDE.md
+++ b/CLAUDE.md
@@ -12,7 +12,7 @@ Handoff file: `/home/dpsifr/.strategy/PLATFORM-HANDOFF.md`
**Brainy's current open actions:** None. MIT open-source — no platform-specific actions.
-**Current version:** `@soulcraft/brainy@7.31.5` (latest published; 8.0.0 release candidate on `feat/8.0-u64-ids`)
+**Current version:** run `npm view @soulcraft/brainy version` (never trust a hardcoded number here — this line went stale for months); consumer-facing changes tracked in `RELEASES.md`
---
diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md
index ab8a0246..ef9c4a51 100644
--- a/CONTRIBUTING.md
+++ b/CONTRIBUTING.md
@@ -1,298 +1,70 @@
# Contributing to Brainy
-Thank you for your interest in contributing to Brainy! This document provides guidelines and instructions for contributing to the project.
+Brainy is MIT-licensed and genuinely open to outside contributions. This page
+is the honest, current path — please don't rely on older instructions you
+may find elsewhere in the repo's history.
-## Code of Conduct
+## Where the project lives
-By participating in this project, you agree to abide by our Code of Conduct:
-- Be respectful and inclusive
-- Welcome newcomers and help them get started
-- Focus on constructive criticism
-- Respect differing viewpoints and experiences
+The source of truth is a self-hosted forge: **source.soulcraft.com/soulcraft/brainy**.
+It's anonymously readable and cloneable — no account needed to browse, clone,
+or build.
-## How to Contribute
+**github.com/soulcraftlabs/brainy** is a public read-only mirror. It's a fine
+place to read code or star the project, but issues and pull requests opened
+there won't be picked up — please use one of the paths below instead.
-### Reporting Issues
+## How to contribute
-Before creating an issue, please check existing issues to avoid duplicates.
+**Found a bug, or have an idea?** Email **brainy@soulcraft.com**. No account,
+no ceremony — you'll get a receipt, and it goes to a human.
-When creating an issue, include:
-- Clear, descriptive title
-- Detailed description of the problem
-- Steps to reproduce
-- Expected vs actual behavior
-- System information (OS, Node version, Brainy version)
-- Code examples if applicable
+**Want to send a patch?** Two ways, both first-class:
-### Suggesting Features
+- **Email a patch.** Run `git format-patch` against your change and email the
+ output to **brainy@soulcraft.com**. This is a genuinely supported path, not
+ a fallback — plenty of good contributions arrive this way.
+- **Open a pull request on the forge.** Request an account at
+ **source.soulcraft.com** (registration is request-with-approval, so allow
+ a little lag), clone, push a branch, and open a PR there. Maintainers
+ review and land it.
-Feature requests are welcome! Please provide:
-- Clear use case
-- Proposed API/interface
-- Examples of how it would work
-- Any potential challenges or considerations
+Either way, for anything beyond a small fix, opening an issue first (email is
+fine) to talk through the approach saves everyone rework.
-### Pull Requests
+## Development setup
-#### Before Starting
-
-1. Check existing issues and PRs
-2. Open an issue to discuss significant changes
-3. Fork the repository
-4. Create a feature branch from `main`
-
-#### Development Setup
-
-**Quick Setup (Recommended):**
```bash
-# Clone your fork
-git clone https://github.com/your-username/brainy.git
+git clone https://source.soulcraft.com/soulcraft/brainy.git
cd brainy
-
-# Run setup script (installs all dependencies including Rust)
-./scripts/setup-dev.sh
-```
-
-**Manual Setup:**
-```bash
-# Clone your fork
-git clone https://github.com/your-username/brainy.git
-cd brainy
-
-# Install system dependencies (Ubuntu/Debian)
-sudo apt-get install -y build-essential pkg-config libssl-dev
-
-# Install Rust (for WASM embedding engine)
-curl --proto '=https' --tlsv1.2 -sSf https://sh.rustup.rs | sh
-source ~/.cargo/env
-rustup target add wasm32-unknown-unknown
-cargo install wasm-pack
-
-# Install Node.js dependencies
npm install
-
-# Build Candle WASM embedding engine
-npm run build:candle
-
-# Build TypeScript
npm run build
-
-# Run tests
npm test
```
-#### Making Changes
+Tests run on [Vitest](https://vitest.dev/). `npm test` runs the unit suite;
+see `package.json` for `test:integration`, `test:coverage`, and friends.
-1. **Follow the code style**
- - TypeScript for all source code
- - Clear variable and function names
- - Comments for complex logic
- - JSDoc for public APIs
+## Standards
-2. **Write tests**
- - Add tests for new features
- - Update tests for changes
- - Ensure all tests pass
+- **Strict TypeScript.** No `any` escape hatches to dodge the type checker.
+- **Tests exercise real behavior.** No mocking away the thing you're supposed
+ to be testing.
+- **No stubs, no TODO-code.** If something can't be finished, say so and
+ leave it out — don't merge a placeholder.
+- **JSDoc on every exported function, class, and type.**
+- **[Conventional Commits](https://www.conventionalcommits.org/).** `feat:`,
+ `fix:`, `docs:`, `perf:`, `refactor:`, `test:`, `chore:`. Never
+ `BREAKING CHANGE` in a commit message — major version bumps are a separate,
+ deliberate decision.
+- **Performance claims are measured or labeled projected.** If a PR or its
+ description states a number, cite the benchmark that produced it (see
+ [docs/performance-envelopes.md](docs/performance-envelopes.md) for the
+ pattern). Don't state an estimate as if it were measured.
-3. **Update documentation**
- - Update README if needed
- - Add/update API documentation
- - Include examples
+## License
-#### Commit Guidelines
+Brainy is [MIT licensed](LICENSE). Contributions are accepted under the same
+license — there's no CLA to sign.
-Follow conventional commits format:
-
-```
-type(scope): description
-
-[optional body]
-
-[optional footer]
-```
-
-Types:
-- `feat`: New feature
-- `fix`: Bug fix
-- `docs`: Documentation changes
-- `style`: Code style changes
-- `refactor`: Code refactoring
-- `perf`: Performance improvements
-- `test`: Test changes
-- `chore`: Build/tooling changes
-
-Examples:
-```bash
-feat(triple): add graph traversal depth limit
-fix(storage): handle concurrent write conflicts
-docs(api): update search method documentation
-```
-
-#### Submitting PR
-
-1. Push to your fork
-2. Create PR against `main` branch
-3. Fill out PR template
-4. Ensure CI checks pass
-5. Wait for review
-
-### Testing
-
-#### Running Tests
-
-```bash
-# Run all tests
-npm test
-
-# Run specific test file
-npm test tests/core.test.ts
-
-# Run with coverage
-npm run test:coverage
-
-# Watch mode
-npm run test:watch
-```
-
-#### Writing Tests
-
-```typescript
-import { describe, it, expect } from 'vitest'
-import { Brainy } from '../src'
-
-describe('Feature Name', () => {
- it('should do something specific', async () => {
- const brain = new Brainy()
- await brain.init()
-
- // Test implementation
- const result = await brain.search("test")
-
- expect(result).toBeDefined()
- expect(result.length).toBeGreaterThan(0)
- })
-})
-```
-
-## Architecture Guidelines
-
-### Adding New Features
-
-1. **Check existing functionality**
- - Review `ARCHITECTURE.md`
- - Check if similar features exist
- - Consider if it should be an augmentation
-
-2. **Design considerations**
- - Maintain backward compatibility
- - Consider performance impact
- - Think about all storage adapters
- - Plan for extensibility
-
-3. **Implementation checklist**
- - [ ] Core functionality
- - [ ] Tests (unit and integration)
- - [ ] Documentation
- - [ ] TypeScript types
- - [ ] Examples
- - [ ] Performance benchmarks (if applicable)
-
-### Creating Augmentations
-
-Augmentations extend Brainy's functionality:
-
-```typescript
-import { BrainyAugmentation } from '../types'
-
-export class MyAugmentation extends BrainyAugmentation {
- name = 'MyAugmentation'
-
- async onInit(brain: Brainy): Promise {
- // Initialize augmentation
- }
-
- async onAdd(item: any, brain: Brainy): Promise {
- // Process before adding
- return item
- }
-
- async onSearch(query: any, results: any[], brain: Brainy): Promise {
- // Process search results
- return results
- }
-}
-```
-
-### Performance Considerations
-
-- Use batch operations where possible
-- Implement caching strategically
-- Consider memory usage
-- Profile performance impacts
-- Add benchmarks for critical paths
-
-## Documentation
-
-### API Documentation
-
-Use JSDoc for all public APIs:
-
-```typescript
-/**
- * Searches for similar items using vector similarity
- * @param query - Search query (text or vector)
- * @param options - Search options
- * @returns Array of search results with scores
- * @example
- * ```typescript
- * const results = await brain.search("machine learning", { limit: 10 })
- * ```
- */
-async search(query: string | Vector, options?: SearchOptions): Promise {
- // Implementation
-}
-```
-
-### Examples
-
-Add examples for new features:
-
-```typescript
-// examples/feature-name.ts
-import { Brainy } from 'brainy'
-
-async function exampleUsage() {
- const brain = new Brainy()
- await brain.init()
-
- // Show feature usage
- // Include comments explaining what's happening
- // Handle errors appropriately
-}
-
-exampleUsage().catch(console.error)
-```
-
-## Release Process
-
-1. **Version bump**: Follow semantic versioning
-2. **Update CHANGELOG**: Document all changes
-3. **Run tests**: Ensure all tests pass
-4. **Build**: Generate distribution files
-5. **Tag**: Create git tag for version
-6. **Publish**: Release to npm
-
-## Getting Help
-
-- **Discord**: Join our community
-- **Issues**: Ask questions on GitHub
-- **Discussions**: Share ideas and get feedback
-
-## Recognition
-
-Contributors will be recognized in:
-- CHANGELOG.md for their contributions
-- README.md contributors section
-- GitHub contributors page
-
-Thank you for contributing to Brainy! 🧠
\ No newline at end of file
+Thank you for considering a contribution.
diff --git a/README.md b/README.md
index d1342f52..2fc42060 100644
--- a/README.md
+++ b/README.md
@@ -23,8 +23,9 @@
Quick start ·
One query ·
Features ·
- Scale with Cor ·
- Docs
+ Scale with Cor ·
+ Docs ·
+ Support
---
@@ -172,9 +173,11 @@ await brain.vfs.search('React components with hooks') // semantic file
**[Multi-process model](docs/concepts/multi-process.md)** · **[Inspection guide](docs/guides/inspection.md)**
-## From laptop to hundreds of millions
+## When you outgrow Brainy
-Brainy's TypeScript engines take you a long way. When you outgrow them, add the native engine — **the API doesn't change**:
+Brainy's pure-TypeScript engines carry real workloads a long way on their own — see the measured, per-operation numbers (not marketing figures) in **[docs/performance-envelopes.md](docs/performance-envelopes.md)** for what to expect, unaccelerated, on plain filesystem storage.
+
+When a deployment needs native-scale vector/graph performance — memory-mapped indexes that don't need your dataset in RAM, billion-scale ambitions — add the native engine. **The API doesn't change:**
```bash
npm install @soulcraft/cor
@@ -187,13 +190,14 @@ await brain.init() // @soulcraft/cor detected — same code, native engines un
Installing the package is the opt-in: if `@soulcraft/cor` is present, it loads and announces itself in the init log; if it's present but broken, `init()` **throws** — an installed accelerator never silently vanishes behind the JS engines. Opt out with `plugins: []`, or pin exactly what loads with `plugins: ['@soulcraft/cor']`. [`@soulcraft/cor`](https://www.npmjs.com/package/@soulcraft/cor) (Brainy 8.x ↔ Cor 3.x, version-matched) registers Rust implementations behind every provider seam: SIMD distance kernels, memory-mapped storage, a disk-native vector index that doesn't need your dataset in RAM, durable LSM field/graph indexes that serve cold opens instantly, and native aggregation. Recall@10 measured **0.99 / 0.96 / 0.96 at 1M / 10M / 100M vectors** in Cor's release gate.
-Open core, commercial accelerator: Brainy is MIT and complete on its own; Cor is licensed and funds both.
+Open core, commercial accelerator: Brainy is MIT and complete on its own — Cor is more headroom for when you need it, not capability held back to sell you later. Licensing and support: **cor@soulcraft.com**.
## Performance
+- Per-operation p50/p95 at 1k and 10k entities, pure-JS floor, measured and re-run every release that touches a measured path: **[docs/performance-envelopes.md](docs/performance-envelopes.md)**.
- JS distance kernels: **~6× faster cosine, ~1.4× euclidean** than 7.x (measured: [`tests/benchmarks/distance-microbench.mjs`](tests/benchmarks/distance-microbench.mjs), 384-dim, median of 41).
- Whole-graph reads are single **O(N + E)** cursor walks — a consumer-measured 19k-edge export dropped from ~27 s of per-node calls to one scan.
-- Full numbers and capacity planning: **[docs/PERFORMANCE.md](docs/PERFORMANCE.md)** · **[docs/SCALING.md](docs/SCALING.md)**
+- Capacity planning and architecture: **[docs/PERFORMANCE.md](docs/PERFORMANCE.md)** · **[docs/SCALING.md](docs/SCALING.md)**
## Use cases
@@ -212,6 +216,10 @@ Open core, commercial accelerator: Brainy is MIT and complete on its own; Cor is
**Bun ≥ 1.1** (recommended) or **Node.js ≥ 22**. Brainy 8.x is server-only; the 7.x line remains on npm for browser use.
-## Contributing & license
+## Support & community
-Contributions welcome — see **[CONTRIBUTING.md](CONTRIBUTING.md)**. MIT © Brainy Contributors.
+- **Bugs and ideas** → **brainy@soulcraft.com** — no account needed, you'll get a receipt.
+- **Security reports** → **security@soulcraft.com** — see **[SECURITY.md](SECURITY.md)**.
+- **Contributing** → see **[CONTRIBUTING.md](CONTRIBUTING.md)**.
+
+MIT © Brainy Contributors.
diff --git a/RELEASES.md b/RELEASES.md
index 7799c6f6..89bf38e7 100644
--- a/RELEASES.md
+++ b/RELEASES.md
@@ -31,6 +31,72 @@ is sometimes cited as a 7.x removal — those methods never existed on 7.x; the
---
+## Unreleased (the warm contract: cold-restart writes stop paying demand-load latency)
+
+From a production deployment's cold-restart incident: the FIRST writes after every
+restart on a large brain measured 33–35s each (page cache cold) against the transact
+apply budget — Brainy's own op-count-scaled budget, `max(30s, opCount × 2s)`, e.g.
+32,000ms for a 16-op batch. There is no external deadline in this story, and no
+post-completion veto either: every write is itself a multi-operation transaction (a
+single `add()` applies several operations — canonical writes, the vector-index insert,
+the metadata-index update), so ONE cold operation that legitimately runs ~33s consumes
+the whole budget, the gate before the NEXT operation trips, and the write rolls back
+atomically (zero loss, by design) — refused, retried, refused again, until the page
+cache warms passively (~30 minutes). The cure is not weaker atomicity; it is the two
+new knobs below — a budget floor sized for cold stores, and a warm contract that pays
+demand-load cost OFF the transaction path.
+
+- **The budget's start-gating contract is now explicit, documented, and pinned by
+ regression tests.** The budget gates STARTING the next operation — completed work is
+ never rolled back for elapsed time — and a transaction's first operation now
+ unconditionally starts by code, not merely because elapsed time happens to be ~0 when
+ it is checked. This has been the shipped schedule since 8.7.0 (no behavior change for
+ existing integrations); it is now stated in `Transaction.execute()`'s contract JSDoc
+ and enforced by tests so it cannot silently regress. Mid-batch atomicity is unchanged:
+ a trip before operation `i+1` still rolls back `0..i` and throws a retryable
+ `TransactionTimeoutError`.
+- **The budget's 30s floor is now configurable**: `new Brainy({ transactionBudgetFloorMs })`
+ raises (or lowers) the floor of `max(transactionBudgetFloorMs, opCount × 2000)` for every
+ internal transact batch. Useful for a store whose cold writes legitimately run past 30s
+ per operation, so a bulk batch gets a proportionally larger runway instead of tripping
+ mid-batch on cold-cache latency.
+- **New: `brain.warm()`** — eagerly loads/faults-in the vector index, metadata index, and
+ graph adjacency so the first real operation after a cold restart runs at steady-state
+ cost instead of paying demand-load latency on the critical path. Returns a `WarmReport`
+ with one honest outcome per surface — never conflate the first two:
+ - `'warmed'` — the surface's own provider `warm()` hook ran, or a full-hydration seam
+ loaded every shard/field/segment from storage. Steady-state cost is paid.
+ - `'probed'` — no `warm()` hook was available, so a best-effort read (one `search()` call
+ for the vector index) faulted in *some* backing storage as a side effect — real work,
+ but never reported as `'warmed'`.
+ - `'unavailable'` — nothing ran (no hook, no hydration seam, or nothing to probe).
+- **New config: `warmOnOpen: true`** makes `init()` await `brain.warm()` before it resolves
+ — a deliberate blocking trade-off: startup takes longer, the first request doesn't.
+ Default `false` (unchanged lazy behavior).
+- **New optional provider hook: `warm?(): Promise`** on the vector and graph
+ acceleration provider contracts (`src/plugin.ts`) — a native provider can implement it to
+ eagerly pretouch its own backing storage (e.g. mmap pretouch); absence means brainy falls
+ back to the probe/hydration behavior above.
+- **Transaction op-name strings changed in journals/timings**: the vector-index
+ transaction operations were renamed from `AddToHNSW`/`RemoveFromHNSW` to backend-neutral
+ `AddToVectorIndex(...)`/`RemoveFromVectorIndex(...)` — the old names hard-coded an
+ algorithm that may not be the one actually running (a non-HNSW native vector provider
+ emitting `"RemoveFromHNSW"` has sent an operator hunting an index that doesn't exist).
+ The parenthesized suffix names the ACTIVE backend: `hnsw-js` for the built-in engine, or
+ the native provider's own identity when it self-identifies. **If you parse these op-name
+ strings** (log processors, journal tooling), update your matcher from
+ `AddToHNSW`/`RemoveFromHNSW` to `AddToVectorIndex(`/`RemoveFromVectorIndex(`.
+- **Provider identity is now a REQUIRED `name` field** on the vector provider contract
+ (`VectorIndexProvider.name`, `src/plugin.ts`) — every implementation self-reports its own
+ identity truthfully (its algorithm/engine), never inheriting a default. It renders as the
+ op-name suffix above and, wherever the vector index identifies itself in prose log lines,
+ as the tag `[vector-index:]`. **Native provider adoption is a one-line change**:
+ declare `readonly name = ''`. A provider instance that still lacks
+ `name` at runtime (an older native build compiled against the previous, optional field) is
+ never crashed on and never silently mislabeled: it stamps `unknown-provider` and emits one
+ loud warning naming the missing field, so the gap is discoverable instead of a permanent
+ fossil label in every journal line.
+
## v8.9.0 — 2026-07-19 (flush is durability-only: history maintenance moves to close())
The write path stops paying maintenance costs — the last structural piece of the
diff --git a/SECURITY.md b/SECURITY.md
new file mode 100644
index 00000000..1f3c4732
--- /dev/null
+++ b/SECURITY.md
@@ -0,0 +1,36 @@
+# Security Policy
+
+## Reporting a vulnerability
+
+Email **security@soulcraft.com**. That's the one door for security reports
+across the company, and it works the same way for Brainy: every report is
+read by a human, you'll get a private receipt, and we'll work with you on
+coordinated disclosure — please don't open a public issue for anything
+that isn't already public.
+
+Include what you'd want if you were on the other end: affected version,
+how to reproduce, and what you think the impact is. If you have a patch or
+a suggested fix, send it along — it's welcome but not required.
+
+There is no bounty program today. We're saying that plainly so you know
+what to expect going in.
+
+## Response time
+
+We respond as fast as truth allows. That means: no fixed SLA, no promise of
+a reply within a specific number of hours — but a real report from a real
+person gets read promptly and taken seriously. If you haven't heard anything
+in a reasonable stretch, a follow-up email is completely fine.
+
+## Supported versions
+
+The latest `8.x` minor release line receives security fixes. If you're
+running an older major version, please upgrade before reporting — we can't
+commit to backporting fixes to unsupported lines.
+
+## Scope
+
+This policy covers the `@soulcraft/brainy` package itself — the code in
+this repository. If you're evaluating a deployment that also uses
+`@soulcraft/cor`, report issues in that package the same way, to the same
+address; we'll route internally.
diff --git a/package-lock.json b/package-lock.json
index fb9262e9..37aeb81d 100644
--- a/package-lock.json
+++ b/package-lock.json
@@ -1,12 +1,12 @@
{
"name": "@soulcraft/brainy",
- "version": "8.9.0",
+ "version": "8.10.0",
"lockfileVersion": 3,
"requires": true,
"packages": {
"": {
"name": "@soulcraft/brainy",
- "version": "8.9.0",
+ "version": "8.10.0",
"license": "MIT",
"dependencies": {
"@msgpack/msgpack": "^3.1.2",
diff --git a/package.json b/package.json
index 7366ce98..e4bc8144 100644
--- a/package.json
+++ b/package.json
@@ -1,6 +1,6 @@
{
"name": "@soulcraft/brainy",
- "version": "8.9.0",
+ "version": "8.10.0",
"description": "Universal Knowledge Protocol™ - World's first Triple Intelligence database unifying vector, graph, and document search in one API. Stage 3 CANONICAL: 42 nouns × 127 verbs covering 96-97% of all human knowledge.",
"main": "dist/index.js",
"module": "dist/index.js",
diff --git a/scripts/release.sh b/scripts/release.sh
index 7d860564..42f5b345 100755
--- a/scripts/release.sh
+++ b/scripts/release.sh
@@ -175,10 +175,26 @@ echo -e "${BLUE}7️⃣ Creating git tag v${NEW_VERSION}...${NC}"
git tag -a "v${NEW_VERSION}" -m "Release v${NEW_VERSION}"
echo -e "${GREEN}✅ Tag created${NC}\n"
-# Step 9: Push to GitHub
-echo -e "${BLUE}8️⃣ Pushing to GitHub...${NC}"
+# Step 9: Push to origin (source of truth) and the public GitHub mirror
+echo -e "${BLUE}8️⃣ Pushing to origin...${NC}"
git push --follow-tags origin "$CURRENT_BRANCH"
-echo -e "${GREEN}✅ Pushed to GitHub${NC}\n"
+echo -e "${GREEN}✅ Pushed to origin${NC}\n"
+
+# The public GitHub repo is a mirror of origin with an unknown sync cadence.
+# `gh release create` below targets GitHub directly: if the new tag hasn't
+# reached GitHub yet, gh would CREATE it — pointed at GitHub's default-branch
+# head, i.e. the wrong commit. Push branch+tag to GitHub explicitly, then
+# verify the tag resolves there to the same commit before any release is cut.
+GITHUB_URL="https://github.com/soulcraftlabs/brainy.git"
+echo -e "${BLUE}8️⃣½ Pushing to the public GitHub mirror...${NC}"
+git push --follow-tags "$GITHUB_URL" "$CURRENT_BRANCH"
+LOCAL_TAG_SHA="$(git rev-parse "v${NEW_VERSION}^{}")"
+GITHUB_TAG_SHA="$(git ls-remote --tags "$GITHUB_URL" "v${NEW_VERSION}^{}" | cut -f1)"
+if [ "$LOCAL_TAG_SHA" != "$GITHUB_TAG_SHA" ]; then
+ echo -e "${RED}❌ Tag v${NEW_VERSION} on GitHub (${GITHUB_TAG_SHA:-absent}) does not match local (${LOCAL_TAG_SHA}) — aborting before npm publish. Fix the mirror, then re-run.${NC}"
+ exit 1
+fi
+echo -e "${GREEN}✅ GitHub mirror has the tag at the right commit${NC}\n"
# Step 10: Publish to npm
echo -e "${BLUE}9️⃣ Publishing to npm (dist-tag: ${NPM_TAG})...${NC}"
diff --git a/src/brainy.ts b/src/brainy.ts
index 9ab3acee..fd9e1ffa 100644
--- a/src/brainy.ts
+++ b/src/brainy.ts
@@ -63,7 +63,9 @@ import type {
PathOptions,
MetadataIndexProvider,
OpaqueIdSet,
- AtGenerationVectors
+ AtGenerationVectors,
+ VectorIndexProvider,
+ GraphIndexProvider
} from './plugin.js'
import type {
BrainyPlugin,
@@ -88,12 +90,12 @@ import { findCallerLocation } from './utils/callerLocation.js'
import {
SaveNounMetadataOperation,
SaveNounOperation,
- AddToHNSWOperation,
+ AddToVectorIndexOperation,
AddToMetadataIndexOperation,
SaveVerbMetadataOperation,
SaveVerbOperation,
AddToGraphIndexOperation,
- RemoveFromHNSWOperation,
+ RemoveFromVectorIndexOperation,
RemoveFromMetadataIndexOperation,
RemoveFromGraphIndexOperation,
UpdateNounMetadataOperation,
@@ -266,6 +268,7 @@ type ResolvedBrainyConfig = Required<
| 'retention'
| 'eagerEmbeddings'
| 'migrationWaitTimeoutMs'
+ | 'transactionBudgetFloorMs'
>
> &
Pick<
@@ -278,6 +281,7 @@ type ResolvedBrainyConfig = Required<
| 'retention'
| 'eagerEmbeddings'
| 'migrationWaitTimeoutMs'
+ | 'transactionBudgetFloorMs'
>
/**
@@ -388,6 +392,38 @@ class InsertPreconditionExistsSignal extends Error {
*/
export type IndexFamily = 'vector' | 'metadata' | 'graph'
+/**
+ * @description Honest per-surface outcome for {@link Brainy.warm}. Literal
+ * meanings — never conflate the first two:
+ * - `'warmed'` — the provider's own `warm?()` hook ran (vector/graph), or the
+ * surface's full-hydration seam loaded EVERY shard/field/segment from
+ * storage (metadata; graph's fallback path). The surface is genuinely at
+ * steady-state cost for the next operation.
+ * - `'probed'` — no `warm?()` hook was available, so a best-effort read
+ * (e.g. one `search()` call) faulted in *some* backing storage as a side
+ * effect. Real work happened, but it is NOT the same guarantee as
+ * `'warmed'` — never reported as `'warmed'`.
+ * - `'unavailable'` — nothing ran: no hook, no hydration seam, and (for the
+ * vector probe fallback) nothing to probe (an empty index or unknown
+ * vector dimension). The surface is unchanged by this `warm()` call.
+ */
+export type WarmOutcome = 'warmed' | 'probed' | 'unavailable'
+
+/**
+ * @description Result of {@link Brainy.warm}: one {@link WarmOutcome} +
+ * elapsed time per index surface, plus the total wall-clock time for the
+ * whole call. `durationMs` is measured around exactly the work described by
+ * that surface's `outcome` (e.g. the vector entry's `durationMs` times the
+ * provider `warm()` call OR the probe `search()` call — whichever ran).
+ */
+export interface WarmReport {
+ vector: { outcome: WarmOutcome; durationMs: number }
+ metadata: { outcome: WarmOutcome; durationMs: number }
+ graph: { outcome: WarmOutcome; durationMs: number }
+ /** Total wall-clock time for the whole `warm()` call (all three surfaces). */
+ totalDurationMs: number
+}
+
/**
* How long a failed aggregation-backfill walk suppresses fresh walk attempts.
* Within the window, queries rethrow the recorded failure instantly (loud,
@@ -1382,6 +1418,17 @@ export class Brainy implements BrainyInterface {
})
}
+ // Eager index warm (operator opt-in, `warmOnOpen: true`). Runs AFTER
+ // every step above — index construction, crash recovery, migrations,
+ // VFS bootstrap — never in place of any of it, so `warm()` always
+ // operates on a fully-initialized brain. Blocking BY DESIGN: the
+ // operator traded a longer startup for a first-request that runs at
+ // steady-state cost instead of paying demand-load latency on the
+ // critical path. See the `warmOnOpen` JSDoc in brainy.types.ts.
+ if (this.config.warmOnOpen) {
+ await this.warm()
+ }
+
// Resolve ready Promise - consumers awaiting brain.ready will now proceed
if (this._readyResolve) {
this._readyResolve()
@@ -1780,7 +1827,9 @@ export class Brainy implements BrainyInterface {
await this.generationStore.runWithoutGeneration(() =>
this.transactionManager.executeTransaction(run, {
timeout: transactTimeoutBudget(
- (touched.nouns?.length ?? 0) + (touched.verbs?.length ?? 0)
+ (touched.nouns?.length ?? 0) + (touched.verbs?.length ?? 0),
+ undefined,
+ this.config.transactionBudgetFloorMs
)
})
)
@@ -1797,7 +1846,9 @@ export class Brainy implements BrainyInterface {
execute: () =>
this.transactionManager.executeTransaction(run, {
timeout: transactTimeoutBudget(
- (touched.nouns?.length ?? 0) + (touched.verbs?.length ?? 0)
+ (touched.nouns?.length ?? 0) + (touched.verbs?.length ?? 0),
+ undefined,
+ this.config.transactionBudgetFloorMs
)
})
})
@@ -2109,7 +2160,7 @@ export class Brainy implements BrainyInterface {
// Operation 3: Add to HNSW index (after entity saved)
tx.addOperation(
- new AddToHNSWOperation(this.index, id, vector)
+ new AddToVectorIndexOperation(this.index, id, vector)
)
// Operation 4: Add to metadata index
@@ -3098,10 +3149,10 @@ export class Brainy implements BrainyInterface {
// Operation 3-4: Update HNSW index (remove and re-add if reindexing needed)
if (needsReindexing) {
tx.addOperation(
- new RemoveFromHNSWOperation(this.index, params.id, existing.vector)
+ new RemoveFromVectorIndexOperation(this.index, params.id, existing.vector)
)
tx.addOperation(
- new AddToHNSWOperation(this.index, params.id, vector)
+ new AddToVectorIndexOperation(this.index, params.id, vector)
)
}
@@ -3217,7 +3268,7 @@ export class Brainy implements BrainyInterface {
// Operation 1: Remove from vector index
if (noun) {
tx.addOperation(
- new RemoveFromHNSWOperation(this.index, id, noun.vector)
+ new RemoveFromVectorIndexOperation(this.index, id, noun.vector)
)
}
@@ -7025,7 +7076,7 @@ export class Brainy implements BrainyInterface {
// Add delete operations to transaction
if (noun) {
tx.addOperation(
- new RemoveFromHNSWOperation(this.index, id, noun.vector)
+ new RemoveFromVectorIndexOperation(this.index, id, noun.vector)
)
}
@@ -7888,7 +7939,13 @@ export class Brainy implements BrainyInterface {
// Budget scales with the batch (or the caller's explicit
// timeoutMs): a flat 30s cap silently limited honest bulk work
// to ~15 ops on network disks (~2s/op measured in the field).
- { timeout: transactTimeoutBudget(plan.operations.length, options?.timeoutMs) }
+ {
+ timeout: transactTimeoutBudget(
+ plan.operations.length,
+ options?.timeoutMs,
+ this.config.transactionBudgetFloorMs
+ )
+ }
)
}
}))
@@ -9311,7 +9368,7 @@ export class Brainy implements BrainyInterface {
plan.operations.push(
new SaveNounMetadataOperation(this.storage, id, storageMetadata, isNew),
new SaveNounOperation(this.storage, { id, vector, connections: new Map(), level: 0 }, isNew),
- new AddToHNSWOperation(this.index, id, vector),
+ new AddToVectorIndexOperation(this.index, id, vector),
new AddToMetadataIndexOperation(this.metadataIndex, id, entityForIndexing)
)
plan.touchedNouns.push(id)
@@ -9479,8 +9536,8 @@ export class Brainy implements BrainyInterface {
)
if (needsReindexing) {
plan.operations.push(
- new RemoveFromHNSWOperation(this.index, params.id, existing.vector),
- new AddToHNSWOperation(this.index, params.id, vector)
+ new RemoveFromVectorIndexOperation(this.index, params.id, existing.vector),
+ new AddToVectorIndexOperation(this.index, params.id, vector)
)
}
plan.operations.push(
@@ -9565,7 +9622,7 @@ export class Brainy implements BrainyInterface {
}
if (noun) {
- plan.operations.push(new RemoveFromHNSWOperation(this.index, id, noun.vector))
+ plan.operations.push(new RemoveFromVectorIndexOperation(this.index, id, noun.vector))
}
if (metadata) {
plan.operations.push(new RemoveFromMetadataIndexOperation(this.metadataIndex, id, metadata))
@@ -14280,6 +14337,118 @@ export class Brainy implements BrainyInterface {
return this.embedder(textToEmbed)
}
+ /**
+ * Eagerly load/fault-in the vector index, metadata index, and graph
+ * adjacency so the FIRST real operation after this call runs at
+ * steady-state cost — no page-cache-miss / demand-load latency on the
+ * critical path. Complements {@link warmupEmbeddings} (which warms the
+ * embedding engine, not the storage indexes).
+ *
+ * Sequence per surface:
+ * - **Vector**: calls the provider's own `warm?()` when the active
+ * `'vector'` provider implements it (`'warmed'`). Otherwise runs a
+ * best-effort probe — one `search()` with a deterministic unit vector of
+ * the index's known dimension, `k = min(100, size())` — which faults in
+ * *some* backing storage as a side effect but is reported honestly as
+ * `'probed'`, never `'warmed'`. An empty index or unknown dimension has
+ * nothing to probe (`'unavailable'`).
+ * - **Metadata**: full hydration — every persisted field's sparse index is
+ * loaded from storage (`MetadataIndexManager.hydrateAll()`), not just the
+ * heuristic common-fields subset `init()` warms.
+ * - **Graph**: calls the provider's own `warm?()` when the active graph
+ * provider implements it; otherwise re-runs its existing eager cold-load
+ * `init()` seam (idempotent — the JS adjacency index's `init()` already
+ * loads the full LSM manifests + SSTables, so re-invoking it here is a
+ * genuine full hydration, not a stat call).
+ *
+ * A single surface's `'unavailable'`/probe outcome never fails the whole
+ * call — `warm()` is a best-effort readiness step, not a correctness gate;
+ * queries still demand-load normally regardless of what `warm()` achieved.
+ *
+ * @returns A {@link WarmReport}: honest per-surface outcome + timing.
+ * @example
+ * ```typescript
+ * const brain = new Brainy({ storage: { path: '/data' } })
+ * await brain.init()
+ * const report = await brain.warm() // blocking, explicit control over timing
+ * console.log(report.vector.outcome, report.vector.durationMs)
+ *
+ * // Or opt into the same work automatically during init():
+ * const eager = new Brainy({ storage: { path: '/data' }, warmOnOpen: true })
+ * await eager.init() // warm() already ran before this resolves
+ * ```
+ */
+ async warm(): Promise {
+ await this.ensureInitialized({ needs: ['vector', 'metadata', 'graph'] })
+
+ const totalStart = Date.now()
+
+ // --- Vector ---------------------------------------------------------
+ const vectorStart = Date.now()
+ let vectorOutcome: WarmOutcome
+ const vectorProvider = this.index as VectorIndexProvider & { warm?: () => Promise }
+ if (typeof vectorProvider.warm === 'function') {
+ await vectorProvider.warm()
+ vectorOutcome = 'warmed'
+ } else {
+ const size = this.index.size()
+ const dimension = this.dimensions
+ if (size > 0 && dimension && dimension > 0) {
+ // A deterministic unit vector (L2 norm 1) — same probe every call,
+ // no reliance on stored data shape.
+ const probeVector = new Array(dimension).fill(1 / Math.sqrt(dimension))
+ await this.index.search(probeVector, Math.min(100, size))
+ vectorOutcome = 'probed'
+ } else {
+ // Nothing to probe: an empty index, or the vector dimension is not
+ // yet known (no entity has ever been added).
+ vectorOutcome = 'unavailable'
+ }
+ }
+ const vectorDurationMs = Date.now() - vectorStart
+
+ // --- Metadata --------------------------------------------------------
+ const metadataStart = Date.now()
+ let metadataOutcome: WarmOutcome
+ const metadataWithHydrate = this.metadataIndex as unknown as { hydrateAll?: () => Promise }
+ if (typeof metadataWithHydrate.hydrateAll === 'function') {
+ await metadataWithHydrate.hydrateAll()
+ metadataOutcome = 'warmed'
+ } else {
+ // No hydration seam on this metadata provider — nothing to run.
+ metadataOutcome = 'unavailable'
+ }
+ const metadataDurationMs = Date.now() - metadataStart
+
+ // --- Graph -------------------------------------------------------------
+ const graphStart = Date.now()
+ let graphOutcome: WarmOutcome
+ const graphProvider = this.graphIndex as GraphIndexProvider & {
+ warm?: () => Promise
+ init?: () => Promise
+ }
+ if (typeof graphProvider.warm === 'function') {
+ await graphProvider.warm()
+ graphOutcome = 'warmed'
+ } else if (typeof graphProvider.init === 'function') {
+ // Existing eager cold-load seam (readiness contract): idempotent, and
+ // for the JS adjacency index it's already a genuine full load of the
+ // LSM manifests + SSTables — a real hydration, not a stat call.
+ await graphProvider.init()
+ graphOutcome = 'warmed'
+ } else {
+ graphOutcome = 'unavailable'
+ }
+ const graphDurationMs = Date.now() - graphStart
+
+ return {
+ vector: { outcome: vectorOutcome, durationMs: vectorDurationMs },
+ metadata: { outcome: metadataOutcome, durationMs: metadataDurationMs },
+ graph: { outcome: graphOutcome, durationMs: graphDurationMs },
+ totalDurationMs: Date.now() - totalStart
+ }
+ }
+
/**
* Explicitly warm up the embedding engine
*
@@ -14798,6 +14967,9 @@ export class Brainy implements BrainyInterface {
// Migration LOCK wait budget — left undefined when omitted so
// awaitMigrationLock() applies its 30 s default at the read site.
migrationWaitTimeoutMs: config?.migrationWaitTimeoutMs ?? undefined,
+ // Apply-phase transaction budget floor — left undefined when omitted so
+ // transactTimeoutBudget() applies its 30 s default at the read site.
+ transactionBudgetFloorMs: config?.transactionBudgetFloorMs ?? undefined,
// Pre-upgrade backup — default-on; opt out with `migrationBackup: false`.
migrationBackup: config?.migrationBackup ?? true,
// Vector index configuration (8.0) — algorithm-neutral surface with
@@ -14813,6 +14985,9 @@ export class Brainy implements BrainyInterface {
// is the active one on a non-reader writer outside unit tests). Explicit
// true/false always wins. See the eagerEmbeddings JSDoc in brainy.types.ts.
eagerEmbeddings: config?.eagerEmbeddings ?? undefined,
+ // Eager index warm at init() — default off (lazy demand-load), the
+ // pre-8.9 behavior. See the warmOnOpen JSDoc in brainy.types.ts.
+ warmOnOpen: config?.warmOnOpen ?? false,
// Plugin configuration - undefined = auto-detect
plugins: config?.plugins ?? undefined,
// Integration Hub - undefined/false = disabled
diff --git a/src/hnsw/hnswIndex.ts b/src/hnsw/hnswIndex.ts
index 605d2a0b..eb2acd71 100644
--- a/src/hnsw/hnswIndex.ts
+++ b/src/hnsw/hnswIndex.ts
@@ -54,6 +54,9 @@ export class HnswFlushError extends Error {
* acceleration provider).
*/
export class JsHnswVectorIndex implements VectorIndexProvider {
+ /** Self-identifies as the built-in JS fallback engine — see {@link VectorIndexProvider.name}. */
+ readonly name = 'hnsw-js'
+
private nouns: Map = new Map()
/**
* Reverse adjacency: `target id → (level → set of node ids that link TO target)`.
diff --git a/src/index.ts b/src/index.ts
index b978b9fd..bfab39da 100644
--- a/src/index.ts
+++ b/src/index.ts
@@ -28,6 +28,9 @@ export type { FileVersion } from './vfs/types.js'
// Export diagnostics result type
export type { DiagnosticsResult } from './brainy.js'
+// brain.warm() — eager index/storage readiness report (per-surface honest
+// outcome + timing). See the WarmReport JSDoc in brainy.ts.
+export type { WarmReport, WarmOutcome } from './brainy.js'
export type {
GraphAuditReport,
GraphAuditDiscrepancy
diff --git a/src/plugin.ts b/src/plugin.ts
index df7c7224..02639955 100644
--- a/src/plugin.ts
+++ b/src/plugin.ts
@@ -381,6 +381,20 @@ export interface GraphIndexProvider {
*/
init?(): Promise
+ /**
+ * @description OPTIONAL. Eagerly load/fault-in backing storage (e.g. mmap
+ * pretouch) so first operations run at steady-state cost. Optional;
+ * absence means the provider demand-loads. Distinct from `init?()`: `init`
+ * is called once automatically during brain startup as part of the
+ * readiness contract (cold-load + rebuild gating); `warm` is an explicit,
+ * separate readiness step a caller opts into via `brain.warm()` (or
+ * `warmOnOpen`) specifically to pre-pay demand-load cost that `init` left
+ * lazy. A provider that already loads everything eagerly in `init?()` may
+ * implement `warm` as a no-op or omit it — `brain.warm()` falls back to a
+ * best-effort read-through probe when absent.
+ */
+ warm?(): Promise
+
/**
* @description OPTIONAL. A native provider returns true from the moment its
* `init()` detects a large epoch-drift until its background
@@ -947,6 +961,26 @@ export interface AtGenerationVectors {
* are retired and never looked up.
*/
export interface VectorIndexProvider {
+ /**
+ * @description REQUIRED self-reported implementation identity, rendered as
+ * `[vector-index:]` in prose log lines and stamped into the
+ * transaction op-name strings that surface in journals/timings (e.g.
+ * `AddToVectorIndex(hnsw-js)`) — see
+ * `src/transaction/operations/IndexOperations.ts`. A provider must name
+ * itself TRUTHFULLY (its own algorithm/engine, e.g. its plugin/package
+ * name) and must never inherit a default — guessing a backend name would
+ * resurrect the exact bug this field exists to prevent: an operator
+ * hunting an index that isn't the one actually running. The built-in JS
+ * index always sets this to `'hnsw-js'`; a native provider picks its own
+ * string. At the TypeScript level this field is required; a runtime
+ * instance from an older provider compiled against the previous optional
+ * `providerId` contract is tolerated (never crashes, never silently
+ * mislabeled) — see `resolveVectorProviderId` in
+ * `src/transaction/operations/IndexOperations.ts`, which stamps
+ * `'unknown-provider'` and emits one loud warning for that case.
+ */
+ readonly name: string
+
addItem(item: VectorDocument): Promise
removeItem(id: string): Promise
search(
@@ -1009,6 +1043,20 @@ export interface VectorIndexProvider {
*/
init?(): Promise
+ /**
+ * @description OPTIONAL. Eagerly load/fault-in backing storage (e.g. mmap
+ * pretouch) so first operations run at steady-state cost. Optional;
+ * absence means the provider demand-loads. Distinct from `init?()`: `init`
+ * runs automatically once during brain startup (the cold-load + rebuild
+ * readiness contract above); `warm` is a separate, explicit step a caller
+ * opts into via `brain.warm()` (or `warmOnOpen`) to pre-pay demand-load
+ * cost `init` left lazy — e.g. touching every mmap page rather than just
+ * opening the file. A provider that already loads everything eagerly in
+ * `init?()` may implement `warm` as a no-op or omit it — `brain.warm()`
+ * falls back to a best-effort probe `search()` when absent.
+ */
+ warm?(): Promise
+
/**
* @description OPTIONAL honest durability signal (readiness contract,
* mirrors {@link GraphIndexProvider.isReady}). `true` ⇔ the persisted
diff --git a/src/transaction/Transaction.ts b/src/transaction/Transaction.ts
index 092a2a25..79b53006 100644
--- a/src/transaction/Transaction.ts
+++ b/src/transaction/Transaction.ts
@@ -37,22 +37,47 @@ const DEFAULT_OPTIONS: Required = {
maxRollbackRetries: 3
}
+/**
+ * The floor term of {@link transactTimeoutBudget}'s scaling formula when the
+ * caller supplies neither an explicit override nor `config.transactionBudgetFloorMs`.
+ */
+const DEFAULT_BUDGET_FLOOR_MS = 30_000
+
/**
* The apply-phase budget for a batch of `opCount` operations.
*
- * An explicit override wins untouched. Otherwise the budget SCALES with the
- * batch: `max(30 000 ms, opCount × 2 000 ms)`. The per-op term is calibrated
- * from field data — bulk imports on network-attached disks measure ~2 s per
- * operation (each op pays canonical writes + fsync + index maintenance) — so
- * a flat 30 s budget silently capped honest work at ~15 operations while
- * looking generous for small batches. Scaling keeps small transacts
- * fast-failing and gives bulk ones a budget proportional to the work they
- * actually asked for; a trip still rolls back atomically and throws a
- * retryable, fully-labeled TransactionTimeoutError.
+ * An explicit `override` wins untouched (a caller-specified deadline for this
+ * one batch). Otherwise the budget SCALES with the batch:
+ * `max(floorMs, opCount × 2 000)`, where `floorMs` defaults to `30 000` and
+ * is overridable via `BrainyConfig.transactionBudgetFloorMs` for the whole
+ * brain. The per-op term is calibrated from field data — bulk imports on
+ * network-attached disks measure ~2 s per operation (each op pays canonical
+ * writes + fsync + index maintenance) — so a flat 30 s budget silently capped
+ * honest work at ~15 operations while looking generous for small batches. A
+ * 16-op batch, for example, scales to `max(30 000, 16 × 2 000) = 32 000`.
+ * Scaling keeps small transacts fast-failing and gives bulk ones a budget
+ * proportional to the work they actually asked for.
+ *
+ * This is a **mid-batch** budget only: it governs whether the transaction's
+ * NEXT operation may start (see {@link Transaction.execute}), never whether
+ * already-completed work is rolled back after the fact. A trip mid-batch
+ * still rolls back every operation applied so far, atomically, and throws a
+ * retryable, fully-labeled TransactionTimeoutError — that zero-loss guarantee
+ * doesn't change; only the point at which the clock stops mattering does (at
+ * the last operation, not one check later).
+ *
+ * @param opCount - Number of operations in the batch.
+ * @param override - A full override for this call; wins over everything else.
+ * @param floorMs - The scaling floor for this call (e.g. `config.transactionBudgetFloorMs`).
+ * Ignored when `override` is set. Defaults to `30 000` when omitted.
*/
-export function transactTimeoutBudget(opCount: number, override?: number): number {
+export function transactTimeoutBudget(
+ opCount: number,
+ override?: number,
+ floorMs?: number
+): number {
if (override !== undefined) return override
- return Math.max(30_000, opCount * 2_000)
+ return Math.max(floorMs ?? DEFAULT_BUDGET_FLOOR_MS, opCount * 2_000)
}
/**
@@ -98,7 +123,20 @@ export class Transaction implements TransactionContext {
}
/**
- * Execute all operations atomically
+ * Execute all operations atomically.
+ *
+ * Budget semantics: the configured budget gates starting the next
+ * operation; completed work is never rolled back for elapsed time. Elapsed
+ * time is checked ONLY before starting operation `i` for `i ≥ 1` — never
+ * before operation 0 (an operation always gets to start) and never again
+ * after the final operation completes. Concretely: a single-op transaction
+ * can never time out post-hoc — it either runs (and its result stands, no
+ * matter how long it took) or a `TransactionTimeoutError` is thrown before
+ * it starts, which cannot happen since there is no operation before it. A
+ * multi-op transaction whose op `i` overruns the budget stops op `i+1` from
+ * starting, rolls back operations `0..i` (reverse order), and throws
+ * `TransactionTimeoutError` — that zero-loss rollback guarantee for
+ * mid-batch trips is unchanged; only the post-completion check is gone.
*/
async execute(): Promise {
if (this.state !== 'pending') {
@@ -128,10 +166,14 @@ export class Transaction implements TransactionContext {
// already-applied writes as torn, generation-less state in canonical
// storage.
for (let i = 0; i < this.operations.length; i++) {
- // Budget check BEFORE starting the next operation. A trip here throws
- // into the catch below and rolls back like any other failure — it must
- // never bypass rollback.
- if (Date.now() - this.startTime > this.options.timeout) {
+ // Budget gates STARTING the next operation; completed work is never
+ // rolled back for elapsed time. Skipped for i === 0 (operation 0
+ // always gets to start — nothing has run yet to have overrun) and
+ // never re-checked after the final operation completes (there is no
+ // "next operation" left to gate). A trip here throws into the catch
+ // below and rolls back like any other failure — it must never bypass
+ // rollback.
+ if (i > 0 && Date.now() - this.startTime > this.options.timeout) {
throw new TransactionTimeoutError(this.options.timeout, i, {
elapsedMs: Date.now() - this.startTime,
totalOperations: this.operations.length,
diff --git a/src/transaction/operations/IndexOperations.ts b/src/transaction/operations/IndexOperations.ts
index 97c39270..d130bb3f 100644
--- a/src/transaction/operations/IndexOperations.ts
+++ b/src/transaction/operations/IndexOperations.ts
@@ -2,34 +2,73 @@
* Index Operations with Rollback Support
*
* Provides transactional operations for all indexes:
- * - JsHnswVectorIndex (unified vector index)
+ * - VectorIndexProvider (the JS HNSW fallback, or a native acceleration provider)
* - MetadataIndexManager (roaring bitmap filtering)
* - GraphAdjacencyIndex (LSM-tree graph storage)
*
* Each operation can be executed and rolled back atomically.
*/
-import type { JsHnswVectorIndex } from '../../hnsw/hnswIndex.js'
+import type { VectorIndexProvider, GraphIndexProvider } from '../../plugin.js'
import type { MetadataIndexManager } from '../../utils/metadataIndex.js'
-import type { GraphIndexProvider } from '../../plugin.js'
import type { GraphVerb } from '../../coreTypes.js'
import type { Operation, RollbackAction } from '../types.js'
/**
- * Add to HNSW index with rollback support
+ * Backend identity stamped into an operation's emitted `name` string (e.g.
+ * `AddToVectorIndex(hnsw-js)`), resolved from the provider's own
+ * {@link VectorIndexProvider.name} self-report.
+ *
+ * These operation classes are backend-neutral (the vector index they wrap may
+ * be Brainy's own JS HNSW fallback OR a native acceleration provider), but
+ * their names surface directly in consumer-visible transaction journals and
+ * timings. `name` is REQUIRED at the TypeScript level — but a native provider
+ * instance compiled against the previous (pre-required) contract can still
+ * reach this function at runtime without it. That legacy case is tolerated,
+ * never crashed on and never silently mislabeled: it resolves to
+ * `'unknown-provider'` and emits ONE loud warning naming the missing contract
+ * field, so the fix (implement `name`) is discoverable rather than a silent
+ * fossil label in every journal line thereafter.
+ */
+const warnedMissingName = new WeakSet()
+
+function resolveVectorProviderId(index: VectorIndexProvider): string {
+ const name = (index as { name?: unknown }).name
+ if (typeof name === 'string') return name
+ if (!warnedMissingName.has(index)) {
+ warnedMissingName.add(index)
+ console.warn(
+ '[vector-index] provider is missing the required `name` field (VectorIndexProvider.name, ' +
+ 'required since 8.10.0) — stamping "unknown-provider" in transaction op names until the ' +
+ 'provider declares its own identity.'
+ )
+ }
+ return 'unknown-provider'
+}
+
+/**
+ * Add to the vector index with rollback support.
+ *
+ * Backend-neutral: `index` is whatever the `'vector'` provider factory
+ * returns — Brainy's own JS HNSW fallback, or a native acceleration provider
+ * (e.g. DiskANN). The emitted `name` stamps the active backend (see
+ * {@link resolveVectorProviderId}) so operators reading a transaction journal
+ * or timing trace see which engine actually ran, never a fossil name from
+ * whichever engine happened to be active when this op class was written.
*
* Rollback strategy:
* - Remove item from index
-
*/
-export class AddToHNSWOperation implements Operation {
- readonly name = 'AddToHNSW'
+export class AddToVectorIndexOperation implements Operation {
+ readonly name: string
constructor(
- private readonly index: JsHnswVectorIndex,
+ private readonly index: VectorIndexProvider,
private readonly id: string,
private readonly vector: number[]
- ) {}
+ ) {
+ this.name = `AddToVectorIndex(${resolveVectorProviderId(index)})`
+ }
async execute(): Promise {
// Check if item already exists (for rollback decision)
@@ -59,11 +98,12 @@ export class AddToHNSWOperation implements Operation {
* pre-existence as "existed" made every rollback skip removeItem, leaving
* phantom entries in the index after a failed transaction. The safe default
* is to remove what this operation added — update flows pair this op with a
- * RemoveFromHNSWOperation whose own rollback restores the prior vector, so
- * reverse-order rollback reconstructs the original state either way.
+ * RemoveFromVectorIndexOperation whose own rollback restores the prior
+ * vector, so reverse-order rollback reconstructs the original state either
+ * way.
*/
private async itemExists(id: string): Promise {
- const index = this.index as JsHnswVectorIndex & {
+ const index = this.index as VectorIndexProvider & {
getItem?: (id: string) => Promise
}
if (typeof index.getItem !== 'function') return false
@@ -77,21 +117,27 @@ export class AddToHNSWOperation implements Operation {
}
/**
- * Remove from HNSW index with rollback support
+ * Remove from the vector index with rollback support.
+ *
+ * Backend-neutral: see {@link AddToVectorIndexOperation} — `index` may be the
+ * JS HNSW fallback or a native acceleration provider; the emitted `name`
+ * stamps the active backend.
*
* Rollback strategy:
* - Re-add item to index with original vector
*
* Note: Requires storing the vector for rollback
*/
-export class RemoveFromHNSWOperation implements Operation {
- readonly name = 'RemoveFromHNSW'
+export class RemoveFromVectorIndexOperation implements Operation {
+ readonly name: string
constructor(
- private readonly index: JsHnswVectorIndex,
+ private readonly index: VectorIndexProvider,
private readonly id: string,
private readonly vector: number[] // Required for rollback
- ) {}
+ ) {
+ this.name = `RemoveFromVectorIndex(${resolveVectorProviderId(index)})`
+ }
async execute(): Promise {
// Remove from index
@@ -285,22 +331,24 @@ export class RemoveFromGraphIndexOperation implements Operation {
}
/**
- * Batch operation: Add multiple items to HNSW index
+ * Batch operation: Add multiple items to the vector index (backend-neutral —
+ * see {@link AddToVectorIndexOperation}).
*
* Useful for bulk imports with transaction support.
* Rolls back all items if any fail.
*/
-export class BatchAddToHNSWOperation implements Operation {
- readonly name = 'BatchAddToHNSW'
+export class BatchAddToVectorIndexOperation implements Operation {
+ readonly name: string
- private operations: AddToHNSWOperation[]
+ private operations: AddToVectorIndexOperation[]
constructor(
- index: JsHnswVectorIndex,
+ index: VectorIndexProvider,
items: Array<{ id: string; vector: number[] }>
) {
+ this.name = `BatchAddToVectorIndex(${resolveVectorProviderId(index)})`
this.operations = items.map(
- item => new AddToHNSWOperation(index, item.id, item.vector)
+ item => new AddToVectorIndexOperation(index, item.id, item.vector)
)
}
diff --git a/src/transaction/operations/index.ts b/src/transaction/operations/index.ts
index 422f6dad..c5548e70 100644
--- a/src/transaction/operations/index.ts
+++ b/src/transaction/operations/index.ts
@@ -21,12 +21,12 @@ export {
// Index Operations
export {
- AddToHNSWOperation,
- RemoveFromHNSWOperation,
+ AddToVectorIndexOperation,
+ RemoveFromVectorIndexOperation,
AddToMetadataIndexOperation,
RemoveFromMetadataIndexOperation,
AddToGraphIndexOperation,
RemoveFromGraphIndexOperation,
- BatchAddToHNSWOperation,
+ BatchAddToVectorIndexOperation,
BatchAddToMetadataIndexOperation
} from './IndexOperations.js'
diff --git a/src/types/brainy.types.ts b/src/types/brainy.types.ts
index 1ec4c4c3..8bede8d3 100644
--- a/src/types/brainy.types.ts
+++ b/src/types/brainy.types.ts
@@ -1651,6 +1651,25 @@ export interface BrainyConfig {
*/
disableAutoRebuild?: boolean
+ /**
+ * The floor (ms) of the apply-phase transaction budget's scaling formula:
+ * `max(transactionBudgetFloorMs, opCount × 2 000)`. The budget governs
+ * whether a `transact()` / single-op write's NEXT internal operation may
+ * **start** — never whether already-completed work gets rolled back after
+ * the fact (a single-op write can never time out post-hoc: it either runs
+ * or it commits). A trip mid-batch still rolls back every applied operation
+ * atomically and throws a retryable `TransactionTimeoutError`; only the
+ * floor of the formula is configurable here.
+ *
+ * Raise this when a cold store's first writes after a restart legitimately
+ * take longer than 30s per operation (e.g. page-cache-cold canonical writes
+ * on a large brain) so a bulk `transact()` batch gets a proportionally
+ * larger runway instead of tripping mid-batch. Lower it to fail faster on a
+ * latency-sensitive write path. Default: `30000` (30 s) — unchanged from
+ * pre-8.9 behavior.
+ */
+ transactionBudgetFloorMs?: number
+
/**
* How long (ms) an operation **waits** on the coordinated 7.x → 8.0 migration
* before throwing a retryable `MigrationInProgressError`.
@@ -1806,6 +1825,23 @@ export interface BrainyConfig {
*/
eagerEmbeddings?: boolean
+ /**
+ * When `true`, `init()` **awaits `warm()`** (see {@link Brainy.warm}) before
+ * it resolves — blocking startup until the vector index, metadata index,
+ * and graph adjacency have all faulted their backing storage in. This is a
+ * deliberate operator opt-in trade: startup takes longer, but the FIRST
+ * request after a cold restart runs at steady-state cost instead of paying
+ * page-cache-miss / demand-load latency on the critical path.
+ *
+ * Runs AFTER `init()`'s own sequence completes (index construction, crash
+ * recovery, migrations, VFS bootstrap) — never in place of any of it — so
+ * `warm()` always operates on a fully-initialized brain.
+ *
+ * Default: `false` (lazy — the pre-8.9 behavior: indexes demand-load on
+ * first access).
+ */
+ warmOnOpen?: boolean
+
// Plugin configuration
// Controls which plugins are loaded during init().
// - undefined (default): guarded auto-detection of the first-party
diff --git a/src/utils/metadataIndex.ts b/src/utils/metadataIndex.ts
index c772bce4..fdb17c22 100644
--- a/src/utils/metadataIndex.ts
+++ b/src/utils/metadataIndex.ts
@@ -423,6 +423,46 @@ export class MetadataIndexManager implements MetadataIndexProvider {
prodLog.debug('✅ Type-aware cache warming completed')
}
+ /**
+ * Full hydration — the {@link Brainy.warm} readiness seam for the metadata
+ * index. Unlike {@link warmCache} / {@link warmCacheForTopTypes} (which
+ * warm only a heuristic subset: common fields plus the top-N types' top
+ * fields), this loads EVERY field's sparse index the field registry knows
+ * about — a real read through {@link loadSparseIndex} into the unified
+ * cache for each field, not a stat/existence check. Idempotent: an
+ * already-cached field's `loadSparseIndex` call is a cheap cache hit.
+ *
+ * Re-reads the field registry first when `fieldIndexes` is empty (a warm()
+ * call issued before `init()` populated it would otherwise hydrate
+ * nothing), then loads every discovered field in parallel.
+ */
+ async hydrateAll(): Promise {
+ if (this.fieldIndexes.size === 0) {
+ await this.loadFieldRegistry()
+ }
+
+ const fields = Array.from(this.fieldIndexes.keys())
+ if (fields.length === 0) {
+ prodLog.debug('[MetadataIndex] hydrateAll: no persisted fields to hydrate')
+ return
+ }
+
+ prodLog.debug(`[MetadataIndex] hydrateAll: loading ${fields.length} field(s) — ${fields.join(', ')}`)
+
+ await Promise.all(
+ fields.map(async field => {
+ try {
+ await this.loadSparseIndex(field)
+ } catch (error) {
+ // A single field's load failure doesn't abort the rest of the
+ // hydration — warm() is a best-effort readiness step, never a
+ // correctness gate (queries still demand-load on miss).
+ prodLog.debug(`[MetadataIndex] hydrateAll: field '${field}' failed to load:`, error)
+ }
+ })
+ )
+ }
+
/**
* Acquire an in-memory lock for coordinating concurrent metadata index writes
* Uses in-memory locks since MetadataIndexManager doesn't have direct file system access
diff --git a/tests/unit/brainy/warm.test.ts b/tests/unit/brainy/warm.test.ts
new file mode 100644
index 00000000..ce213696
--- /dev/null
+++ b/tests/unit/brainy/warm.test.ts
@@ -0,0 +1,310 @@
+/**
+ * @module tests/unit/brainy/warm
+ * @description Coverage for `brain.warm()` / `warmOnOpen` / the provider
+ * `warm?()` contract (the cold-restart readiness fix): first operations after
+ * a cold restart should run at steady-state cost instead of paying
+ * demand-load latency on the critical path.
+ *
+ * Uses a real, in-process fake plugin provider implementing the actual
+ * `VectorIndexProvider` contract from `src/plugin.ts` — the real seam a
+ * native provider (e.g. a disk-native accelerator) plugs into. The real
+ * built-in JS metadata index and graph adjacency index run against real
+ * (filesystem or in-memory) storage with pre-existing data, so their
+ * hydration paths are exercised for real, not mocked.
+ */
+import { describe, it, expect, afterEach } from 'vitest'
+import * as fs from 'node:fs'
+import * as os from 'node:os'
+import * as path from 'node:path'
+import { Brainy } from '../../../src/brainy.js'
+import { NounType, VerbType } from '../../../src/types/graphTypes.js'
+import type { VectorIndexProvider } from '../../../src/plugin.js'
+import type { VectorDocument, Vector } from '../../../src/coreTypes.js'
+import { MetadataIndexManager } from '../../../src/utils/metadataIndex.js'
+import { GraphAdjacencyIndex } from '../../../src/graph/graphAdjacencyIndex.js'
+
+const tmpDirs: string[] = []
+function mkTmp(): string {
+ const d = fs.mkdtempSync(path.join(os.tmpdir(), 'brainy-warm-'))
+ tmpDirs.push(d)
+ return d
+}
+afterEach(() => {
+ for (const d of tmpDirs.splice(0)) fs.rmSync(d, { recursive: true, force: true })
+})
+
+// Brainy's ValidationConfig fixes vectors at exactly 384 dimensions
+// (src/utils/paramValidation.ts) — match it so `add()` doesn't reject test data.
+const DIM = 384
+const V = (seed = 1): number[] => Array.from({ length: DIM }, (_, i) => Math.sin(seed + i))
+
+/**
+ * A real (not mocked) VectorIndexProvider implementation, backed by a plain
+ * Map, that optionally implements `warm()` — the exact seam
+ * `AddToVectorIndexOperation` / `brain.warm()` call through.
+ */
+class FakeVectorProvider implements VectorIndexProvider {
+ readonly name = 'fake-vector-provider'
+ readonly items = new Map()
+ warmCalls = 0
+ searchCalls: Array<{ k?: number }> = []
+ // Only present on the instance when the constructor is told to — mirrors a
+ // real provider that may or may not implement the optional hook. Assigned
+ // in the constructor BODY (not as a field initializer): under native ES
+ // class fields, field initializers run before constructor-body statements
+ // — including the parameter-property assignment — so referencing a
+ // parameter property from a field initializer would see it as still
+ // `undefined`.
+ warm?: () => Promise
+
+ constructor(hasWarm: boolean) {
+ if (hasWarm) {
+ this.warm = async (): Promise => {
+ this.warmCalls++
+ }
+ }
+ }
+
+ async addItem(item: VectorDocument): Promise {
+ this.items.set(item.id, item.vector)
+ return item.id
+ }
+ async removeItem(id: string): Promise {
+ return this.items.delete(id)
+ }
+ async search(_queryVector: Vector, k?: number): Promise> {
+ this.searchCalls.push({ k })
+ return [...this.items.keys()].slice(0, k ?? 10).map((id) => [id, 0])
+ }
+ size(): number {
+ return this.items.size
+ }
+ clear(): void {
+ this.items.clear()
+ }
+ async rebuild(): Promise {}
+ async flush(): Promise {
+ return 0
+ }
+ getPersistMode(): 'immediate' | 'deferred' {
+ return 'deferred'
+ }
+}
+
+/** Registers `provider` under the `'vector'` plugin key, before `init()`. */
+function useFakeVectorProvider(brain: Brainy, provider: FakeVectorProvider): void {
+ brain.use({
+ name: 'fake-vector-provider',
+ activate: async (ctx: any) => {
+ ctx.registerProvider('vector', () => provider)
+ return true
+ }
+ })
+}
+
+describe('brain.warm()', () => {
+ it('(a) calls the vector provider\'s warm() when present and reports "warmed"', async () => {
+ const brain = new Brainy({
+ requireSubtype: false,
+ storage: { type: 'memory' },
+ silent: true
+ })
+ const provider = new FakeVectorProvider(true)
+ useFakeVectorProvider(brain, provider)
+ await brain.init()
+ await brain.add({ data: 'a', type: NounType.Thing, vector: V(1) })
+
+ const report = await brain.warm()
+
+ expect(provider.warmCalls).toBe(1)
+ expect(provider.searchCalls.length).toBe(0) // warm() ran — no probe fallback
+ expect(report.vector.outcome).toBe('warmed')
+ await brain.close()
+ })
+
+ it('(b) falls back to a probe search() when warm() is absent and reports "probed", never "warmed"', async () => {
+ const brain = new Brainy({
+ requireSubtype: false,
+ storage: { type: 'memory' },
+ silent: true
+ })
+ const provider = new FakeVectorProvider(false)
+ useFakeVectorProvider(brain, provider)
+ await brain.init()
+ await brain.add({ data: 'a', type: NounType.Thing, vector: V(1) })
+ await brain.add({ data: 'b', type: NounType.Thing, vector: V(2) })
+
+ const report = await brain.warm()
+
+ expect(provider.warmCalls).toBe(0) // no warm() on this provider
+ expect(provider.searchCalls.length).toBe(1) // the probe ran
+ expect(provider.searchCalls[0].k).toBe(Math.min(100, provider.size())) // k = min(100, size)
+ expect(report.vector.outcome).toBe('probed')
+ expect(report.vector.outcome).not.toBe('warmed') // never conflated
+ await brain.close()
+ })
+
+ it('vector: reports "unavailable" when there is nothing to probe (empty index, unknown dimension)', async () => {
+ const brain = new Brainy({
+ requireSubtype: false,
+ storage: { type: 'memory' },
+ silent: true
+ })
+ // A disk-native provider MAY honestly report 0 resident entries while
+ // durable data exists on disk (the same posture documented on
+ // `VectorIndexProvider.isReady` — "an mmap/disk-native index may
+ // legitimately report 0 resident entries"). warm()'s probe fallback
+ // reads `size()`, so this is the real trigger for "nothing to probe":
+ // never a k=0 search, an honest skip.
+ const provider = new FakeVectorProvider(false)
+ provider.size = () => 0
+ useFakeVectorProvider(brain, provider)
+ await brain.init() // the VFS root bootstrap write sets `dimensions`, but size() still reports 0
+
+ const report = await brain.warm()
+
+ expect(provider.searchCalls.length).toBe(0) // nothing probed
+ expect(report.vector.outcome).toBe('unavailable')
+ await brain.close()
+ })
+
+ it('(c) metadata + graph hydration paths actually execute against filesystem storage with pre-existing data', async () => {
+ const dir = mkTmp()
+
+ // Build a brain with real data (including a field NOT in the metadata
+ // index's own common-fields warm subset — 'wave' — so hydrateAll()'s
+ // FULL hydration is distinguishable from init()'s partial warmCache()),
+ // and real graph edges, then close it (persisting everything).
+ const seed = new Brainy({
+ requireSubtype: false,
+ storage: { type: 'filesystem', path: dir },
+ silent: true
+ })
+ await seed.init()
+ const ids: string[] = []
+ for (let i = 0; i < 6; i++) {
+ ids.push(
+ await seed.add({
+ data: `entity ${i}`,
+ type: NounType.Thing,
+ metadata: { wave: i % 3 },
+ vector: V(i + 1)
+ })
+ )
+ }
+ for (let i = 0; i + 1 < ids.length; i++) {
+ await seed.relate({ from: ids[i], to: ids[i + 1], type: VerbType.RelatedTo })
+ }
+ await seed.close()
+
+ // Cold-reopen a FRESH instance and spy on the real hydration seams before
+ // calling warm(), so we assert they actually ran (not just that the
+ // report claims they did).
+ const metaHydrateCalls: number[] = []
+ const loadedFields: string[] = []
+ const graphInitCalls: number[] = []
+
+ const origHydrateAll = MetadataIndexManager.prototype.hydrateAll
+ const origLoadSparseIndex = (MetadataIndexManager.prototype as any).loadSparseIndex
+ const origGraphInit = GraphAdjacencyIndex.prototype.init
+
+ MetadataIndexManager.prototype.hydrateAll = async function (...args: any[]) {
+ metaHydrateCalls.push(1)
+ return origHydrateAll.apply(this, args as any)
+ }
+ ;(MetadataIndexManager.prototype as any).loadSparseIndex = async function (
+ field: string,
+ ...args: any[]
+ ) {
+ loadedFields.push(field)
+ return origLoadSparseIndex.apply(this, [field, ...args] as any)
+ }
+ GraphAdjacencyIndex.prototype.init = async function (...args: any[]) {
+ graphInitCalls.push(1)
+ return origGraphInit.apply(this, args as any)
+ }
+
+ try {
+ const brain = new Brainy({
+ requireSubtype: false,
+ storage: { type: 'filesystem', path: dir },
+ silent: true
+ })
+ await brain.init()
+
+ const report = await brain.warm()
+
+ expect(metaHydrateCalls.length).toBe(1) // hydrateAll() actually ran
+ expect(loadedFields).toContain('wave') // a NON-common field was loaded — full hydration, not the heuristic subset
+ expect(report.metadata.outcome).toBe('warmed')
+
+ expect(graphInitCalls.length).toBeGreaterThanOrEqual(1) // graph's full-load seam ran (once at brain init, once via warm())
+ expect(report.graph.outcome).toBe('warmed')
+
+ // Correctness survives: the hydrated data still answers queries.
+ const byWhere = await brain.find({ type: NounType.Thing, where: { wave: 1 } })
+ expect(byWhere.length).toBe(2) // waves 1,4 of 0..5
+
+ await brain.close()
+ } finally {
+ MetadataIndexManager.prototype.hydrateAll = origHydrateAll
+ ;(MetadataIndexManager.prototype as any).loadSparseIndex = origLoadSparseIndex
+ GraphAdjacencyIndex.prototype.init = origGraphInit
+ }
+ })
+
+ it('(d) warmOnOpen: true runs warm() during init() — observable via the fake provider', async () => {
+ const brain = new Brainy({
+ requireSubtype: false,
+ storage: { type: 'memory' },
+ warmOnOpen: true,
+ silent: true
+ })
+ const provider = new FakeVectorProvider(true)
+ useFakeVectorProvider(brain, provider)
+
+ // No explicit brain.warm() call — warmOnOpen must have run it as part of init().
+ await brain.init()
+
+ expect(provider.warmCalls).toBe(1)
+ await brain.close()
+ })
+
+ it('warmOnOpen defaults to false — init() does NOT run warm() unless opted in', async () => {
+ const brain = new Brainy({
+ requireSubtype: false,
+ storage: { type: 'memory' },
+ silent: true
+ })
+ const provider = new FakeVectorProvider(true)
+ useFakeVectorProvider(brain, provider)
+
+ await brain.init()
+
+ expect(provider.warmCalls).toBe(0)
+ await brain.close()
+ })
+
+ it('(e) WarmReport shape: outcome literal + durationMs per surface, plus totalDurationMs', async () => {
+ const brain = new Brainy({
+ requireSubtype: false,
+ storage: { type: 'memory' },
+ silent: true
+ })
+ const provider = new FakeVectorProvider(true)
+ useFakeVectorProvider(brain, provider)
+ await brain.init()
+ await brain.add({ data: 'a', type: NounType.Thing, vector: V(1) })
+
+ const report = await brain.warm()
+
+ for (const surface of ['vector', 'metadata', 'graph'] as const) {
+ expect(['warmed', 'probed', 'unavailable']).toContain(report[surface].outcome)
+ expect(typeof report[surface].durationMs).toBe('number')
+ expect(report[surface].durationMs).toBeGreaterThanOrEqual(0)
+ }
+ expect(typeof report.totalDurationMs).toBe('number')
+ expect(report.totalDurationMs).toBeGreaterThanOrEqual(0)
+ await brain.close()
+ })
+})
diff --git a/tests/unit/transaction/budget-commit-completed-work.test.ts b/tests/unit/transaction/budget-commit-completed-work.test.ts
new file mode 100644
index 00000000..a32489de
--- /dev/null
+++ b/tests/unit/transaction/budget-commit-completed-work.test.ts
@@ -0,0 +1,149 @@
+/**
+ * @module tests/unit/transaction/budget-commit-completed-work
+ * @description Regression coverage for the "commit-completed-work" transaction
+ * budget semantics: the budget gates STARTING the next operation; it never
+ * converts already-completed work into a rollback. Concretely:
+ *
+ * - A single-op transaction can never time out post-hoc — its one operation
+ * either runs (and the transaction commits, however long it took) or it
+ * never gets to start (which cannot happen: there is no operation before it
+ * to have overrun the budget).
+ * - A multi-op transaction whose op `i` overruns the budget stops op `i+1`
+ * from starting: everything applied so far rolls back atomically and a
+ * `TransactionTimeoutError` is thrown — the mid-batch zero-loss guarantee is
+ * unchanged.
+ * - `transactTimeoutBudget`'s scaling floor (`max(floorMs, opCount * 2000)`)
+ * is overridable per call, the seam `BrainyConfig.transactionBudgetFloorMs`
+ * feeds at the brainy.ts read sites.
+ *
+ * Uses a real class implementing the `Operation` interface (not an anonymous
+ * object literal, not a mock of Transaction internals) so the exercised path
+ * is exactly what a real caller's operation looks like.
+ */
+import { describe, it, expect } from 'vitest'
+import { Transaction, transactTimeoutBudget } from '../../../src/transaction/Transaction.js'
+import type { Operation, RollbackAction } from '../../../src/transaction/types.js'
+import { TransactionTimeoutError } from '../../../src/transaction/errors.js'
+
+const sleep = (ms: number) => new Promise((resolve) => setTimeout(resolve, ms))
+
+/**
+ * A real `Operation` implementation that sleeps for `delayMs` before applying
+ * a write to `log`, and returns a rollback action that removes it again. Used
+ * to deterministically make one operation "slow" relative to a transaction's
+ * budget without touching any Transaction internals.
+ */
+class SlowOperation implements Operation {
+ readonly name: string
+ executed = false
+ rolledBack = false
+
+ constructor(
+ private readonly log: string[],
+ label: string,
+ private readonly delayMs: number
+ ) {
+ this.name = label
+ }
+
+ async execute(): Promise {
+ this.executed = true
+ if (this.delayMs > 0) await sleep(this.delayMs)
+ this.log.push(this.name)
+ return async () => {
+ this.rolledBack = true
+ const idx = this.log.indexOf(this.name)
+ if (idx >= 0) this.log.splice(idx, 1)
+ }
+ }
+}
+
+describe('Transaction budget — commit-completed-work semantics', () => {
+ it('(a) single-op transaction whose op overruns the budget COMMITS — no rollback, no throw', async () => {
+ const log: string[] = []
+ const op = new SlowOperation(log, 'slow-single-op', 40)
+
+ // Budget (5ms) is far smaller than the op's 40ms — under the OLD
+ // (post-completion-check) semantics this would have thrown and rolled
+ // back after the op finished. Under the new semantics there is no
+ // operation after it to gate, so it commits.
+ const tx = new Transaction({ timeout: 5 })
+ tx.addOperation(op)
+
+ await expect(tx.execute()).resolves.toBeUndefined()
+
+ expect(tx.getState()).toBe('committed')
+ expect(op.executed).toBe(true)
+ expect(op.rolledBack).toBe(false)
+ expect(log).toEqual(['slow-single-op'])
+ })
+
+ it('(b) two-op transaction: op0 overruns the budget → op1 never starts, op0 rolls back, throws TransactionTimeoutError', async () => {
+ const log: string[] = []
+ const op0 = new SlowOperation(log, 'op0-overruns', 40)
+ const op1 = new SlowOperation(log, 'op1-never-starts', 0)
+
+ const tx = new Transaction({ timeout: 5 })
+ tx.addOperation(op0)
+ tx.addOperation(op1)
+
+ const error = await tx.execute().catch((e) => e)
+
+ expect(error).toBeInstanceOf(TransactionTimeoutError)
+ expect(op0.executed).toBe(true)
+ expect(op0.rolledBack).toBe(true)
+ expect(op1.executed).toBe(false)
+ expect(op1.rolledBack).toBe(false)
+ expect(log).toEqual([]) // op0's write was undone; op1 never wrote
+ expect(tx.getState()).toBe('rolled_back')
+ })
+
+ it('a single-op transaction never checks the budget before its only operation starts', async () => {
+ // Budget of 0ms: under the old "check before every op including 0" code
+ // this could theoretically trip before op 0 even started (if any
+ // measurable time elapsed between startTime capture and the check). The
+ // documented contract is stronger: operation 0 ALWAYS gets to start.
+ const log: string[] = []
+ const op = new SlowOperation(log, 'only-op', 5)
+
+ const tx = new Transaction({ timeout: 0 })
+ tx.addOperation(op)
+
+ await expect(tx.execute()).resolves.toBeUndefined()
+ expect(tx.getState()).toBe('committed')
+ expect(op.executed).toBe(true)
+ })
+
+ it('(c) transactTimeoutBudget: an explicit floorMs overrides the 30s default floor', () => {
+ // Below the per-op scaling term, a raised floor wins.
+ expect(transactTimeoutBudget(1, undefined, 5_000)).toBe(5_000) // 1 op × 2000 = 2000 < 5000 floor
+ expect(transactTimeoutBudget(1)).toBe(30_000) // unchanged default when floorMs omitted
+
+ // Above the floor, opCount * 2000 still wins over a smaller floor.
+ expect(transactTimeoutBudget(10, undefined, 5_000)).toBe(20_000) // 10 × 2000 = 20000 > 5000 floor
+
+ // A full `override` always wins, regardless of floorMs.
+ expect(transactTimeoutBudget(10, 999, 5_000)).toBe(999)
+ })
+
+ it('(c) a configured floor changes real Transaction behavior end-to-end via TransactionManager-style options', async () => {
+ // Simulates how brainy.ts wires `this.config.transactionBudgetFloorMs`
+ // into `transactTimeoutBudget()`: opCount 0 isolates the floor term
+ // (0 × 2000 = 0), so a lowered floor makes a normally-generous budget
+ // fail-fast for a real 2-op Transaction.
+ const lowFloorBudget = transactTimeoutBudget(0, undefined, 10) // 10ms floor
+ expect(lowFloorBudget).toBe(10)
+
+ const log: string[] = []
+ const op0 = new SlowOperation(log, 'op0', 30)
+ const op1 = new SlowOperation(log, 'op1-gated-by-lowered-floor', 0)
+
+ const tx = new Transaction({ timeout: lowFloorBudget })
+ tx.addOperation(op0)
+ tx.addOperation(op1)
+
+ const error = await tx.execute().catch((e) => e)
+ expect(error).toBeInstanceOf(TransactionTimeoutError)
+ expect(op1.executed).toBe(false)
+ })
+})
diff --git a/tests/unit/transaction/vectorIndexOperations-rename.test.ts b/tests/unit/transaction/vectorIndexOperations-rename.test.ts
new file mode 100644
index 00000000..bdfe58a3
--- /dev/null
+++ b/tests/unit/transaction/vectorIndexOperations-rename.test.ts
@@ -0,0 +1,176 @@
+/**
+ * @module tests/unit/transaction/vectorIndexOperations-rename
+ * @description Coverage for the backend-neutral vector-index transaction
+ * operations (formerly `AddToHNSWOperation` / `RemoveFromHNSWOperation`).
+ * These op-name strings surface directly in consumer-visible transaction
+ * journals and timings — a fossil "HNSW" name misdirects an operator running
+ * a native (non-HNSW) vector provider. Verifies:
+ * 1. rollback wiring stayed byte-identical under the rename,
+ * 2. the emitted `name` stamps the active backend — `'hnsw-js'` for
+ * Brainy's own JS index, a provider's own required `name` when it
+ * self-identifies, and the tolerant-loud `'unknown-provider'` literal
+ * (plus one console.warn) when a runtime instance lacks `name`
+ * altogether (an older native provider compiled against the previous,
+ * optional `providerId` contract) — never a silently-wrong guess.
+ */
+import { describe, it, expect, vi, afterEach } from 'vitest'
+import {
+ AddToVectorIndexOperation,
+ RemoveFromVectorIndexOperation,
+ BatchAddToVectorIndexOperation
+} from '../../../src/transaction/operations/IndexOperations.js'
+import type { VectorIndexProvider } from '../../../src/plugin.js'
+import type { VectorDocument, Vector } from '../../../src/coreTypes.js'
+import { JsHnswVectorIndex } from '../../../src/hnsw/hnswIndex.js'
+
+/**
+ * A minimal, real (not mocked) VectorIndexProvider implementation backed by
+ * a plain Map — exercises the exact contract the operations call through,
+ * without any Brainy internals.
+ */
+class FakeVectorProvider implements VectorIndexProvider {
+ readonly items = new Map()
+ constructor(readonly name: string) {}
+
+ async addItem(item: VectorDocument): Promise {
+ this.items.set(item.id, item.vector)
+ return item.id
+ }
+ async removeItem(id: string): Promise {
+ return this.items.delete(id)
+ }
+ async getItem(id: string): Promise {
+ const vector = this.items.get(id)
+ return vector ? { id, vector } : undefined
+ }
+ async search(): Promise> {
+ return []
+ }
+ size(): number {
+ return this.items.size
+ }
+ clear(): void {
+ this.items.clear()
+ }
+ async rebuild(): Promise {}
+ async flush(): Promise {
+ return 0
+ }
+ getPersistMode(): 'immediate' | 'deferred' {
+ return 'immediate'
+ }
+}
+
+describe('Vector index transaction operations — backend-neutral rename', () => {
+ describe('rollback wiring (byte-identical behavior under the rename)', () => {
+ it('AddToVectorIndexOperation rollback removes a newly-added item', async () => {
+ const provider = new FakeVectorProvider('fake-provider')
+ const op = new AddToVectorIndexOperation(provider, 'id-1', [1, 2, 3])
+
+ const rollback = await op.execute()
+ expect(provider.items.has('id-1')).toBe(true)
+
+ await rollback()
+ expect(provider.items.has('id-1')).toBe(false)
+ })
+
+ it('AddToVectorIndexOperation rollback is a no-op when the item pre-existed (update semantics)', async () => {
+ const provider = new FakeVectorProvider('fake-provider')
+ await provider.addItem({ id: 'id-1', vector: [9, 9, 9] })
+
+ const op = new AddToVectorIndexOperation(provider, 'id-1', [1, 2, 3])
+ const rollback = await op.execute()
+ expect(provider.items.get('id-1')).toEqual([1, 2, 3])
+
+ await rollback()
+ // Pre-existing item is NOT removed by rollback — it stays (the update
+ // itself is not undone by this op; that's RemoveFromVectorIndexOperation's job).
+ expect(provider.items.has('id-1')).toBe(true)
+ })
+
+ it('RemoveFromVectorIndexOperation rollback re-adds the item with its original vector', async () => {
+ const provider = new FakeVectorProvider('fake-provider')
+ await provider.addItem({ id: 'id-1', vector: [4, 5, 6] })
+
+ const op = new RemoveFromVectorIndexOperation(provider, 'id-1', [4, 5, 6])
+ const rollback = await op.execute()
+ expect(provider.items.has('id-1')).toBe(false)
+
+ await rollback()
+ expect(provider.items.get('id-1')).toEqual([4, 5, 6])
+ })
+
+ it('BatchAddToVectorIndexOperation rolls back every item in reverse order on undo', async () => {
+ const provider = new FakeVectorProvider('fake-provider')
+ const op = new BatchAddToVectorIndexOperation(provider, [
+ { id: 'a', vector: [1] },
+ { id: 'b', vector: [2] },
+ { id: 'c', vector: [3] }
+ ])
+
+ const rollback = await op.execute()
+ expect([...provider.items.keys()].sort()).toEqual(['a', 'b', 'c'])
+
+ await rollback()
+ expect(provider.items.size).toBe(0)
+ })
+ })
+
+ describe('backend stamping in the emitted op name', () => {
+ afterEach(() => {
+ vi.restoreAllMocks()
+ })
+
+ it('stamps the real JS HNSW index as hnsw-js (self-identifying name)', () => {
+ const index = new JsHnswVectorIndex()
+ const addOp = new AddToVectorIndexOperation(index, 'id-1', [1, 2, 3])
+ const removeOp = new RemoveFromVectorIndexOperation(index, 'id-1', [1, 2, 3])
+
+ expect(addOp.name).toBe('AddToVectorIndex(hnsw-js)')
+ expect(removeOp.name).toBe('RemoveFromVectorIndex(hnsw-js)')
+ })
+
+ it('stamps a self-identifying provider\'s own name, never "HNSW"', () => {
+ const provider = new FakeVectorProvider('acme-diskann')
+ const addOp = new AddToVectorIndexOperation(provider, 'id-1', [1, 2, 3])
+ const removeOp = new RemoveFromVectorIndexOperation(provider, 'id-1', [1, 2, 3])
+ const batchOp = new BatchAddToVectorIndexOperation(provider, [{ id: 'a', vector: [1] }])
+
+ expect(addOp.name).toBe('AddToVectorIndex(acme-diskann)')
+ expect(removeOp.name).toBe('RemoveFromVectorIndex(acme-diskann)')
+ expect(batchOp.name).toBe('BatchAddToVectorIndex(acme-diskann)')
+ expect(addOp.name).not.toContain('HNSW')
+ expect(removeOp.name).not.toContain('HNSW')
+ })
+
+ it('tolerant-loud: stamps "unknown-provider" and warns exactly once when a runtime provider lacks the required `name` (an older native provider compiled against the previous optional `providerId` contract)', () => {
+ const warnSpy = vi.spyOn(console, 'warn').mockImplementation(() => {})
+ // Simulate a native provider compiled before `name` was required — the
+ // TypeScript contract requires `name`, but `as unknown as` mirrors what
+ // actually reaches this code at runtime from an un-rebuilt native addon.
+ const legacyProvider = {
+ addItem: async (item: VectorDocument) => item.id,
+ removeItem: async () => true,
+ search: async () => [],
+ size: () => 0,
+ clear: () => {},
+ rebuild: async () => {},
+ flush: async () => 0,
+ getPersistMode: () => 'immediate' as const
+ } as unknown as VectorIndexProvider
+
+ const addOp = new AddToVectorIndexOperation(legacyProvider, 'id-1', [1, 2, 3])
+ const removeOp = new RemoveFromVectorIndexOperation(legacyProvider, 'id-1', [1, 2, 3])
+
+ expect(addOp.name).toBe('AddToVectorIndex(unknown-provider)')
+ expect(removeOp.name).toBe('RemoveFromVectorIndex(unknown-provider)')
+ // Never silently falls back to the JS engine's name for a provider it
+ // knows nothing about — that's the exact fossil-naming bug being fixed.
+ expect(addOp.name).not.toContain('hnsw-js')
+ // Loud, not silent — but exactly once per provider instance, not once
+ // per op stamped against it.
+ expect(warnSpy).toHaveBeenCalledTimes(1)
+ expect(warnSpy.mock.calls[0][0]).toContain('name')
+ })
+ })
+})