Compare commits

..

No commits in common. "main" and "v8.8.2" have entirely different histories.
main ... v8.8.2

280 changed files with 4237 additions and 41247 deletions

View file

@ -2,7 +2,7 @@
## What Is Brainy
@soulcraftlabs/brainy (v7.17.0) is a Universal Knowledge Protocol -- a Triple Intelligence database combining vector search, graph traversal, and metadata filtering in a single library. Published to npm as a public MIT-licensed package.
@soulcraft/brainy (v7.17.0) is a Universal Knowledge Protocol -- a Triple Intelligence database combining vector search, graph traversal, and metadata filtering in a single library. Published to npm as a public MIT-licensed package.
## Core Architecture

View file

@ -1,62 +0,0 @@
name: CI
# Branch pushes only — a release TAG deliberately does not re-run CI: the
# tagged commit's CI already ran on its branch push, and the runner is
# sequential, so tag-triggered matrix jobs (~22 min) would queue AHEAD of the
# tag's publish-source run and starve every release (observed on 8.10.3 and
# 9.0.0: the publish sat behind the tag's own redundant CI).
on:
push:
branches: ['**']
pull_request:
jobs:
node:
name: Node ${{ matrix.node-version }}
runs-on: ubuntu-latest
strategy:
fail-fast: false
matrix:
node-version: ['22', '24']
steps:
- uses: actions/checkout@v4
- uses: actions/setup-node@v4
with:
node-version: ${{ matrix.node-version }}
cache: npm
- run: npm ci
- run: npm run test:unit
# The correctness plant's full gate: integration + conformance run here on
# dedicated iron, on every push, so a release never depends on any other
# machine being up. Verdicts live in this run's log (never inferred).
integration:
name: Integration + conformance (Node 22)
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v4
- uses: actions/setup-node@v4
with:
node-version: '22'
cache: npm
- run: npm ci
- run: npm run test:ci-integration
- run: npx vitest run tests/conformance
bun:
name: Bun (latest)
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v4
- uses: actions/setup-node@v4
with:
node-version: '22'
cache: npm
- uses: oven-sh/setup-bun@v2
with:
bun-version: latest
- run: npm ci
# test:bun imports the built dist/, so build first.
- run: npm run build
# Bun as a runtime is the supported Bun story (`bun add` / `bun run`).
- run: npm run test:bun

View file

@ -1,79 +0,0 @@
name: Publish (The Source)
# Datacenter-side publish to The Source (source.soulcraft.com — our
# self-hosted Forgejo; never call it "the forge", Forge is a different
# product), moved off the laptop: an 87MB tarball PUT over the laptop's WAN
# times out; The Source's own runner does it in seconds.
# scripts/release.sh tags + pushes, then polls this workflow's result (npm
# view against The Source's registry) before it ever touches the npmjs leg —
# see the "delegation contract" in scripts/release.sh's home-publish step.
on:
push:
tags:
- 'v*'
jobs:
publish:
name: Publish to The Source registry
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v4
- uses: actions/setup-node@v4
with:
node-version: '22'
cache: npm
- run: npm ci
- run: npm run build
- name: Publish + readback-verify on The Source registry
env:
# The stored repo-settings secret keeps its historical name.
FORGE_NPM_TOKEN: ${{ secrets.FORGE_NPM_TOKEN }}
run: |
set -eo pipefail
SOURCE_NPM_REG="https://source.soulcraft.com/api/packages/soulcraftlabs/npm/"
VERSION="$(node -p "require('./package.json').version")"
# The dist-tag follows the version: a prerelease (any hyphen —
# 10.4.0-rc.1) publishes under 'rc' and must NEVER move 'latest' —
# every consumer resolving 'latest' from this registry would otherwise
# be handed a release candidate. Same rule scripts/release.sh applies
# to the storefront leg.
NPM_TAG="latest"
case "$VERSION" in
*-*) NPM_TAG="rc" ;;
esac
echo "Publishing @soulcraftlabs/brainy@${VERSION} to The Source registry (dist-tag: ${NPM_TAG})..."
TMPRC="$(mktemp)"
chmod 600 "$TMPRC"
{
echo "@soulcraftlabs:registry=${SOURCE_NPM_REG}"
echo "//source.soulcraft.com/api/packages/soulcraftlabs/npm/:_authToken=${FORGE_NPM_TOKEN}"
} > "$TMPRC"
# The release script bumps package.json's version before it tags, so
# this tag's checkout already carries the version being published —
# nothing here re-derives it from the tag name.
PUBLISH_OK=true
if ! npm publish --tag "$NPM_TAG" --userconfig "$TMPRC"; then
PUBLISH_OK=false
fi
# Readback verify is the source of truth, run regardless of the publish
# exit code: a benign duplicate publish (a prior run, or a mirror, already
# landed this exact version) reports failure even though the registry
# already holds the right content.
LANDED_VERSION="$(npm view "@soulcraftlabs/brainy@${VERSION}" version --userconfig "$TMPRC" 2>/dev/null || echo "")"
rm -f "$TMPRC"
if [ "$LANDED_VERSION" != "$VERSION" ]; then
echo "::error::Readback verify FAILED — The Source registry reports version '${LANDED_VERSION:-<none>}', expected '${VERSION}'. This is a genuine publish failure, not a benign duplicate."
exit 1
fi
if [ "$PUBLISH_OK" = true ]; then
echo "Published and verified @soulcraftlabs/brainy@${VERSION} on The Source registry."
else
echo "::warning::npm publish reported failure, but readback confirms @soulcraftlabs/brainy@${VERSION} is already live on The Source (a prior run or mirror landed it) — treating this run as successful, since the registry content is correct. Any OTHER failure mode would have failed the readback check above instead."
fi

40
.github/workflows/ci.yml vendored Normal file
View file

@ -0,0 +1,40 @@
name: CI
on:
push:
pull_request:
jobs:
node:
name: Node ${{ matrix.node-version }}
runs-on: ubuntu-latest
strategy:
fail-fast: false
matrix:
node-version: ['22', '24']
steps:
- uses: actions/checkout@v4
- uses: actions/setup-node@v4
with:
node-version: ${{ matrix.node-version }}
cache: npm
- run: npm ci
- run: npm run test:unit
bun:
name: Bun (latest)
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v4
- uses: actions/setup-node@v4
with:
node-version: '22'
cache: npm
- uses: oven-sh/setup-bun@v2
with:
bun-version: latest
- run: npm ci
# test:bun imports the built dist/, so build first.
- run: npm run build
# Bun as a runtime is the supported Bun story (`bun add` / `bun run`).
- run: npm run test:bun

View file

@ -2,256 +2,6 @@
All notable changes to this project will be documented in this file. See [standard-version](https://github.com/conventional-changelog/standard-version) for commit guidelines.
### [10.4.4](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v10.4.3...v10.4.4) (2026-08-28)
- fix(vfs): the old-root sweep narrates only when it has something to say (d49148e1)
- fix(tests): the health-gate pin follows the verdict, and the VFS suite uses its own store (42e2da25)
- Merge branch 'next/open-lazy-open-and-counts' (5ebd3b40)
- docs: the contract manifest stands alone; public docs describe this engine only (a8c724a2)
- docs(releases): 10.4.4 consumer notes — correctness and observability, with the performance line stated exactly (61a46927)
- docs: measurements in public history carry numbers, not provenance (02c61636)
- feat(open): name the two steps that hold the vfs-bootstrap phase (2cf38010)
- fix(storage): a dead flush watch falls back to the 500ms poll, not the 30s sweep (5c22f950)
- fix(storage): the flush watcher cannot arm twice in its async window (16d2e1a9)
- perf(idle): the flush-request watch is event-driven; the heartbeat is observability (fb1da1c5)
- perf(open): answer "are there any entities?" with one directory read (417ddb51)
- perf(generations): discover generations by directory name, not by walking the log (9dd39921)
- fix(flush): clear() and repairIndex() set the dirty witness themselves (e4c27fbc)
- feat(open): the open names the STEP that cost the time, not just the phase (5a091cca)
- perf(vfs): the old-root sweep runs once per store, not once per open (4a67aa0f)
- chore: keep the generated neural stamps at main's values (c1f09723)
- feat(contract): declare contract 1, serve three operators, refuse four by name (48802ba3)
- fix(open): a provider rebuilding itself is a third state, not a CRITICAL (50676c02)
- feat(open): open never waits for a provider that is rebuilding itself (131daa08)
- perf(flush): an idle brain does no work — no periodic flush without a write (f5a6cb3f)
- feat(repair): repairIndex narrates every phase and its receipt carries the walls (3fffd9c6)
- fix(storage): a suspect count ledger heals itself, and counts.json is written atomically (f4e2d34b)
- feat(open): the open narrates itself, on a channel production cannot clamp (afe08a1f)
- fix(storage): a clean close is recorded, and the writer lock is always given up (e652162c)
- docs: repository links point at soulcraftlabs/open-brainy — the soulcraft/brainy path becomes the native engine's repo tonight (38c3397b)
### [10.4.3](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v10.4.2...v10.4.3) (2026-08-27)
- Merge branch 'next/open-brainy-rename' (a58372f0)
- chore: rename to @soulcraftlabs/brainy for Open Brainy on The Source (a99b1e83)
- docs(releases): 10.4.3 — Open Brainy's first release under the new name, same engine as 10.4.2; The Source is the one registry (9f248b24)
### [10.4.2](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v10.4.2-rc.1...v10.4.2) (2026-08-27)
- docs(releases): 10.4.1 and 10.4.2 consumer notes; 10.4.2 is the last MIT release under this name, Open Brainy continues at @soulcraftlabs/brainy (a082e0ef)
### [10.4.2-rc.1](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v10.4.1...v10.4.2-rc.1) (2026-08-27)
- Merge branch 'next/zero-norm-unvector-door' (9b84ef5b)
- fix(vectors): a zero-norm vector is not a vector, canonical side included, plus the sanctioned unvector door (0de76659)
- fix(hnsw): skip unvectored rows on rebuild; refuse empty vectors in the index (8fc553b1)
- fix(storage): derive the canonical count ledger from identity records, stamp the derivation rule, and mark legacy-derived ledgers suspect at load (fd6b4ce4)
- Merge branch 'next/enumeration-identity-rekey' (204d74c1)
- fix(storage): enumeration re-keys on the identity record, not the vector leg (f8d8ce16)
- fix(init): rethrow plugin activation failures with the original error as cause so the originating frame survives to the caller (2496e09a)
- Merge branch 'next/vfs-root-zero-norm' (4c7b0fab)
- fix(vfs): the VFS root never persists a zero-norm vector (c6cc0de9)
- build: derive generated-file stamps from git commit time, not wall clock (8a5c1245)
- Merge remote-tracking branch 'origin/release/10.4.1' (aad9e2ee)
- docs(concepts): the serving law — a failure is graded by whether an answer could be wrong, never by the cost of the fix; reads refuse per family (2914e0eb)
- chore(release): 10.4.1-rc.1 (7870dc40)
### [10.4.1](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v10.4.0...v10.4.1) (2026-08-26)
- fix(reads): the read gate is per-family; a write carrying unchanged data never re-embeds (c039411e)
- docs(guide): the docs pipeline publishes through the ingest API — the separate deploy step is retired (21e506e8)
### [10.4.0](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v10.4.0-rc.4...v10.4.0) (2026-08-26)
- docs(releases): the 10.4.0 entry catches up to the late trains — repair routing, the vector ledger and open-gate leg, the loud config guard, the JSON-safe crossing (834149ed)
### [10.4.0-rc.4](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v10.4.0-rc.3...v10.4.0-rc.4) (2026-08-25)
- feat(vector): the vectored-noun scalar joins the count ledger; the open gate closes the vector leg (9730835b)
### [10.4.0-rc.3](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v10.4.0-rc.2...v10.4.0-rc.3) (2026-08-25)
- fix(update-seam): the metadata crossing never carries BigInt endpoint ints (f4780c8e)
- Merge branch 'worktree-agent-ad3aff0dffd17a6eb' (f14da34b)
- fix(add): empty string is real data, not a missing field (258e9042)
- feat(vfs): implement readdir's recursive option — typed since 7.30, never read (fc516da6)
- feat(open-path): init never gates on the embedding model; open goes concurrent; slow opens narrate (96624f40)
### [10.4.0-rc.2](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v10.4.0-rc.1...v10.4.0-rc.2) (2026-08-25)
- test(readiness): the report helper's clock freezes — two independently-built reports compared across a millisecond tick made the plant lane red (39b916a3)
- feat(repair): a heal:'repair' verdict routes to the provider's own incremental repair() (553e0d97)
- fix(storage): an unknown nested storage config can never silently land on the shared default root (ddd5e719)
- docs(release): the 10.4.0 entry, the index-health concept doc, and the API surfaces — written from the tree, not the plan (8cced871)
- fix(plugins): the silent-degrade doors close — a broken accelerator install can never read as absent (b9ba50fb)
- feat(recovery): the catchup verdict is consumed; verb rows go live; the metadata rebuild goes online (18f172e0)
- feat(health): the gate reads the named report — reads refuse loudly, never rebuild; open serves before it returns; the ceremony door (f8f64780)
### [10.4.0-rc.1](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v10.3.1...v10.4.0-rc.1) (2026-08-24)
- ci(publish): the home dist-tag follows the version — a prerelease publishes under 'rc' and never moves 'latest' (a1376e4a)
- chore(release): --source-only — a home-only prerelease mode (The Source, never the storefront) (dcbad176)
- test(fold-checkpoint): the ARM-AT-FLIP pin arms its crash instead of racing the pending-flush timer (4176439b)
- fix(health): one contract for a throwing probe — heal is none, serving is not withheld; repair report gains missing/rebuilt/reason (116550eb)
- feat(storage): the canonical count ledger — ALL-visibility scalars, unclamped totals, suspect-on-unprovable-delete (7c8c8be3)
- fix(delete): the null-metadata skip closes — index legs run id-keyed or narrate, never silently strand postings (607e9f54)
- feat(repair): repairIndex returns the per-family receipt and narrates its summary (8d45f964)
- fix(reads): the readiness gate guards every index read surface — serving empty from a not-ready provider is unrepresentable (40e7119b)
- ci(gate): the machine-health preflight and the truncation verdict guard (1e046aa1)
### [10.3.1](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v10.3.0...v10.3.1) (2026-08-18)
- docs(releases): the 10.3.1 consumer entry — the fold that behaves (900cc895)
- fix(recovery): the fold streams and narrates; the checkpoint chain arms at the flip (ed7d1db9)
### [10.3.0](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v10.2.0...v10.3.0) (2026-08-18)
- docs(releases): the 10.3.0 consumer entry — the trust-and-provenance release (97d75649)
- fix(locks): the fence keys ownership on pid+hostname — a same-process re-open never fences its predecessor (0991cf28)
- test(budgets): iron-honest wall-clock budgets — 3x the worst honest-iron measurement (314e0e6c)
- fix(locks): live writers are never auto-evicted; evicted writers are fenced at every commit barrier (292e7c04)
- feat(log): system commits carry their origin; the attested per-id reconcile door (9ac9e706)
### [10.2.0](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v10.1.0...v10.2.0) (2026-08-17)
- docs(releases): the 10.2.0 consumer entry — adoption completes in one call (97538e1f)
- ci: the correctness plant runs integration + conformance on every push — a release never waits on a second machine (b17fdc8e)
- fix(adoption): the baseline backfill runs to completion — one call adopts a pre-log baseline of any size (a5a18838)
### [10.1.0](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v10.0.0...v10.1.0) (2026-08-13)
- docs(releases): the 10.1.0 consumer entry — bounded recovery, restore founding, the two write-path cures (7d3c8696)
- fix(restore): a restore is an unclean event — the swap runs quiesced and the snapshot's durability stamps never survive it (9ca80667)
- feat(recovery): the fold-checkpoint bound — crash folds (checkpoint, head], never the whole log twice (ff43de1a)
- fix(log): pad-frame construction is total; the at-ack sync-failure compensation splits by phase — a production adoption's two write-path defects, cured at their roots (cbe34d11)
- feat(query): the sparse-store cut — where on a never-carried field serves operator truth, never a refusal (7b67db4d)
### [10.0.0](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v9.0.0...v10.0.0) (2026-08-12)
- fix(adoption): the baseline backfill cures hydration-law drift — existing brains reach the crash-safe default with zero operator steps (25f0dd96)
- fix(adoption): the reserved-root mint exemption — int 0 is legitimate for exactly one id (2abe8b38)
- fix(recovery): walks are healers — the typed/tolerant boundary redrawn where block-layer fault injection proved it belonged (0e3facf4)
- feat(log): log authority is the fleet default — adopt-at-open, oracle-gated; plus the power-cut throw-site cures and the loud torn-record contract (214c98b4)
- fix(durability): three block-layer power-loss findings from the first fault-injection box run — all cured, matrix 15/15 (67c606be)
- docs: RELEASES.md frames the release as 10.0.0 — honest major (log format v2 forward-only); comment wording cleanup (d1698fa5)
- fix(persistence): the idle flush trigger debounces under load — deferred to the floor, never dropped, never a flush-per-gap amplifier (a50726e6)
- feat(reprojection): the one doors-open machinery — budget-capped, yielding, foreground-preempted, atomic-swap; poison records quarantine typed (d1651f98)
- feat(embedding): deferred-embed markers become log records — the sidecar recovery path is deleted (b47787bb)
- feat(conformance): the golden-log fold oracle — encoder bytes and fold semantics pinned by content hash (c95bea88)
- feat(engine): the wiring wave — stamps ride every flush, provider generations, waitForIndexed, adopt-backfill, match-all serves (b53e6e89)
- feat(index): watermark stamps on every TS projection — adopt/catchup/rescan verdicts at load, stamp-after-data (b35d87a7)
- feat(log): v2 is the LIVE write format — envelope records with minted ints, genesis, sector seals; v1 readable forever (26c60251)
- docs: RELEASES.md — the unreleased write-path and lifecycle entry (consumer-facing draft; version set at cut) (73eb88d4)
- feat(temporal): as-of semantic recall joins the release contract — past vectors byte-exact, pinned (f7ca0d26)
- fix(log): acked writes survive power loss; rejected writes never silently commit — the kill-matrix goes 11/11 with zero .fails debt (13022c51)
- feat(plugin): every provider write surface carries the real committed generation (2d532684)
- feat(log): fact-log format v2 codec — record envelope, type registry, genesis, sector seals; fault-injection shim (34841074)
- feat(log): the guarded log-authority core — group-commit durable-at-ack, the per-brain switch, the verification oracle (65953097)
- docs: Path Registry rows DP6/DP8/MT5 flip to contracted+pinned — the deferred-embedding and atomic-update train landed with cited tests (9fda6d95)
- feat(embedding): MT5 — deferred embedding with durable markers; write acks never wait on a neural net (287384cf)
- fix(index): the flicker window dies — atomic in-place vector update; lazy open honors every provider's not-ready report; the Path Registry twin table (ebe06cdf)
- feat(persistence): the engine owns its flush cadence — callers never call flush() in hot paths again (3236a01b)
- fix(aggregation): the lifecycle cluster — flush stamps, behind-stamp catches up incrementally, the native rebuild finally gets invoked, deletes are never silently skipped (1dc861d2)
- perf(sort): ordered reads never do per-row storage round-trips — the 199-317s production scan class dies structurally (607b6b56)
- chore: the home registry is The Source, never 'the forge' — sweep the misnomer out of the release rail, workflows, and release notes (Forge is a different product; the stored CI secret keeps its historical name) (09352c2b)
- ci: tags stop triggering the CI matrix (redundant re-run of already-tested commits starved every release's publish run on the sequential runner) + release.sh forge poll window 20→50 min (c6c6ea6b)
- test: version-coupling pins go major-agnostic — the 8.x literals broke at the 9.0.0 bump while the coupling law itself behaved correctly (8a6807e8)
### [9.0.0](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v8.11.0...v9.0.0) (2026-08-04)
- docs: 9.0 namespace-migration guide — the simple story + the mechanical sweep checklist, published for humans and tooling alike (61ab9db2)
- fix(release): storefront leg republishes CI's exact forge artifact — byte-identity by construction, verified by cross-registry shasum before the ceremony reports success (d89df2ed)
- docs: v9.0.0 release notes — the field-addressing law migration ledger; retitle the shipped 8.11.0 canonical-enumeration entry (header went stale at its cut) (55a7512c)
- feat(namespace): merge the field-addressing law train — no special names, system.* scalars, nested-bag storage, epoch-3 index keys (19b477ae)
- feat(namespace): NO SPECIAL NAMES + storage fidelity — the ruled completion of the field-addressing law (24bf6cdb)
- feat(namespace): write-door forgery refusal (user metadata keys may never start 'system.') + refusal messages name both spellings in every branch (the non-colliding case marks system.<f> honestly as NOT valid) — cross-engine message pin alignment (48a6130a)
- feat(namespace): conformance green 19/19 — data-aware did-you-mean on unindexed bare addresses, ordering contract on the column top-K path (never drop, nulls last, ties by id), shape-complete addressed reads (entity views AND raw storage shapes, shadow-proof both scopes), per-key source matching for dotted addresses; refusal classes unified under UnresolvableFieldError (8e962dab)
- feat(namespace): aggregation reads under the law + epoch 3 (the key-split rebuild) + THE ARMING COMMIT — the capability constant, the law module, and the typed refusals export from the package root; both engines' conformance suites light on this signal (7492b6cb)
- feat(namespace): egress guard + validation speak the law — whereMatcher's resolver reads system.* from the record and bare names from the metadata bag only (the bare-system switch is dead); validateFindParams refuses cursor/includeRelations/writeOnly typed (accepted-and-ignored dies as a class), validates order, and parses every orderBy address (c2fb28a2)
- fix(namespace): noun-record updates preserve legacy inline HNSW adjacency — the placeholder-adjacency write stamped out pre-codec records' stored connections (crash-window unreachability); codec-era records were never at risk (empty field is the blob marker); pin covers the legacy shape (4679c894)
- feat(namespace): find's own filter builders speak the frozen keys — params.type/subtype/service become system.* index keys at every construction site (three pipelines + the canonical buildMetadataFilter); the where.type→noun alias is dead (bare 'type' belongs to the user now) (7a28a946)
- feat(namespace): the index speaks the frozen keys — record-frame scalars index under literal 'system.<field>' (legacy 'noun' spelling folds into system.type; plumbing never indexed from a record frame), user fields stay bare in every shape; filter + sorted paths route every address through parseFieldAddress; storage fallbacks read the addressed side of the record (11c724bc)
- docs(namespace): the d.ts JSDoc wave — the sealed field-addressing law on the full find + aggregation surface, present-tense, with the refusal semantics and migration note inline (comment-only; verified zero code lines changed) (fcb24ab6)
- test(namespace): unit pins for the pure law — the ruled maps verbatim (incl. the relation mirror, unpinnable via public API), plumbing refusals both kinds, did-you-mean text (5502abcd)
- fix(namespace): the JS sorted fallback honors the ruled ordering contract — nulls last in BOTH directions (was nulls-first on desc) + deterministic id-ascending tie-break (56deb2e8)
- test(namespace)+docs: the cross-engine conformance suite (self-arming — skips until the resolver exports land) + the public field-addressing docs page; sidebar order deconflicted to 7 (d8d0b55f)
- feat(namespace): the one field-addressing law as a single source of truth — parseFieldAddress + the ruled ten-scalar system maps + plumbing invisibility + refusal builders (module only; query surfaces wire in next) (8f9a9989)
- docs: port the 8.10.3 backport-release changelog entry to main (f6b14d21)
- docs: port the 8.10.2 backport-release changelog entry to main — release branches carry the version bump, main carries the durable record (0b059ac5)
- fix: user metadata named 'level' is a real field everywhere — the engine-internal node layer no longer shadows it in sort/filter/aggregation, and the indexing views stop stamping a phantom 0 into its column; index epoch 2 rebuilds existing brains at first open (1a09be06)
- fix: metadata-only update() never rewrites the noun record — the unconditional whole-vector save turned per-entity stat touches into full rewrites+fsync, amplifying read-heavy sweeps into disk saturation on a production deployment (cb717be2)
- fix(release): double the forge-publish poll budget — the runner executes jobs sequentially and the publish run queues behind the ci matrix (64049631)
- Merge branch 'release/8.11.0' (1865f60a)
- Merge branch 'release/8.10.1' (fc9f0d72)
- chore: the forge is the address — retire the archived mirror from every live surface (415e824a)
- Merge remote-tracking branch 'origin/main' (069a8894)
- Merge branch 'release/8.10.0' (d918c060)
- ci: run the pipeline on the forge (9a5a9ccc)
- feat: two-tier history reads + the repacker + generationDigest — D1+D3 wired end-to-end (1201e255)
- feat: generation-segment store — the D1+D3 packed-tier file format (d8acb377)
- feat: scanFacts liveness contract — first batch or loud failure within a documented bound (f8e6da2b)
### [8.11.0](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v8.10.1...v8.11.0) (2026-07-27)
- docs: the last two archived-host links point home (91ef1c8b)
- feat: includeHidden — export carries every visibility tier for migration-grade canon completeness (63c1eeb9)
- feat(release): the forge publish leg moves to CI on the tag push; the laptop verifies by readback and keeps the abort-before-storefront guard (3e4a17dc)
- feat: canonical enumeration mode for export — storage-walked, canon-complete, with an index-drift report (4d196af4)
- ci: run the pipeline on the forge (999d0ebb)
### [8.10.3](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v8.10.2...v8.10.3) (2026-08-03)
- docs: dedupe the 8.10.2 release-notes entry the cherry doubled onto the branch (8c956608)
- fix: user metadata named 'level' is a real field everywhere — the engine-internal node layer no longer shadows it in sort/filter/aggregation, and the indexing views stop stamping a phantom 0 into its column; index epoch 2 rebuilds existing brains at first open (958a0859)
### [8.10.2](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v8.10.1...v8.10.2) (2026-07-29)
- docs: 8.10.2 consumer release notes — update() write granularity, PathResolver idle-log fix, graph-lsm key recognition (a0123b5b)
- fix: metadata-only update() never rewrites the noun record — the unconditional whole-vector save turned per-entity stat touches into full rewrites+fsync, amplifying read-heavy sweeps into disk saturation on a production deployment (5b65eb82)
### [8.10.1](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v8.10.0...v8.10.1) (2026-07-24)
- refactor: remove the orphaned transaction-result type left behind by the dead-path removal (edf123a5)
- fix: warm() metadata surface routes through the active provider (warm hook added to the metadata contract); add maintenanceDebt() observability surface (5b2cbf74)
- fix: transaction timeouts are a typed no-hot-retry contract; engine-side non-retry pinned; dead transaction path removed (003e2a74)
- chore: the forge is the address — retire the archived mirror from every live surface (22702b81)
### [8.10.0](https://github.com/soulcraftlabs/brainy/compare/v8.9.0...v8.10.0) (2026-07-23)
- docs: adoption storefront — contributing guide, security policy, README support + cor section (9a99a7b)
- fix(release): push the public mirror explicitly and verify the tag lands at the right commit before publishing (6ba94c8)
- docs: project guide version line points at npm instead of a hardcoded stale number (3a1efc9)
- feat: vector provider identity is a required name field (hnsw-js), rendered [vector-index:<name>] (3be4ba9)
- feat: warm contract (warm/warmOnOpen/provider warm hook), configurable transact budget floor, backend-neutral vector index op names (55b867c)
### [8.9.0](https://github.com/soulcraftlabs/brainy/compare/v8.8.2...v8.9.0) (2026-07-19)
- docs: measured performance envelopes v1 (per-op p50/p95 at 1k and 10k, pure-JS floor) (5cabd78)
- fix: release drains in-flight writer-lock heartbeat — no phantom lock after unlink (70e4bc8)
- feat: flush() never compacts — history maintenance moves to close() with bounded passes (300d9f2)
### [8.8.2](https://github.com/soulcraftlabs/brainy/compare/v8.8.1...v8.8.2) (2026-07-19)
- fix: one field-resolution law across aggregation hooks, source.where, removeMany, and find() spellings (945d92d)

View file

@ -12,13 +12,13 @@ Handoff file: `/home/dpsifr/.strategy/PLATFORM-HANDOFF.md`
**Brainy's current open actions:** None. MIT open-source — no platform-specific actions.
**Current version:** run `npm view @soulcraftlabs/brainy version --registry https://source.soulcraft.com/api/packages/soulcraftlabs/npm/` (never trust a hardcoded number here — this line went stale for months); consumer-facing changes tracked in `RELEASES.md`
**Current version:** `@soulcraft/brainy@7.31.5` (latest published; 8.0.0 release candidate on `feat/8.0-u64-ids`)
---
## Project Overview
Brainy is a Universal Knowledge Protocol -- a Triple Intelligence database that combines vector similarity search, graph traversal, and metadata filtering into a single TypeScript library. Published as `@soulcraftlabs/brainy` on The Source (source.soulcraft.com registry) under the MIT license.
Brainy is a Universal Knowledge Protocol -- a Triple Intelligence database that combines vector similarity search, graph traversal, and metadata filtering into a single TypeScript library. Published as `@soulcraft/brainy` on npm under the MIT license.
## Getting Started
@ -91,7 +91,7 @@ test: add/update tests (patch version bump)
## Docs Pipeline — soulcraft.com/docs
Docs in `docs/**/*.md` are published with the npm package (included in `files`) and go live on soulcraft.com/docs via the docs ingest API: the release script's `scripts/push-docs.js` step POSTs every public doc to `https://soulcraft.com/api/docs/ingest` (auth: `DOCS_INGEST_SECRET` in the environment). No separate deploy step is involved (the old deploy-to-publish flow was retired in a platform change, 2026-08). Frontmatter controls what appears publicly.
Docs in `docs/**/*.md` are published with the npm package (included in `files`) and synced to soulcraft.com/docs on every portal deploy. Frontmatter controls what appears publicly.
### Docs check triggers
@ -161,9 +161,9 @@ npm run release:major # Breaking changes (rare, manual decision)
The script: verifies clean git state, builds, tests, bumps version, updates CHANGELOG.md, commits, tags, pushes, publishes to npm, and creates a GitHub release.
After a successful release, remind the user:
> "Published. Docs are live on soulcraft.com/docs (pushed via the ingest API during the release) — spot-check a changed page with curl."
> "Published. Deploy portal to pick up the new docs → go to the portal project and deploy."
There is no separate deploy step anymore. If the docs push failed (the script warns loudly), re-run `node scripts/push-docs.js` with `DOCS_INGEST_SECRET` set.
Do NOT deploy portal from here. Portal is always deployed separately from within the portal project.
## Closed-Source Product Names — HARD RULE

View file

@ -1,77 +1,298 @@
# Contributing to Brainy
Brainy is MIT-licensed and genuinely open to outside contributions. This page
is the honest, current path — please don't rely on older instructions you
may find elsewhere in the repo's history.
Thank you for your interest in contributing to Brainy! This document provides guidelines and instructions for contributing to the project.
## Where the project lives
## Code of Conduct
The source of truth is a self-hosted forge: **source.soulcraft.com/soulcraftlabs/open-brainy**.
It's anonymously readable and cloneable — no account needed to browse, clone,
or build.
By participating in this project, you agree to abide by our Code of Conduct:
- Be respectful and inclusive
- Welcome newcomers and help them get started
- Focus on constructive criticism
- Respect differing viewpoints and experiences
## How to contribute
## How to Contribute
**Found a bug, or have an idea?** Email **brainy@soulcraft.com**. No account,
no ceremony — you'll get a receipt, and it goes to a human.
### Reporting Issues
**Want to send a patch?** Two ways, both first-class:
Before creating an issue, please check existing issues to avoid duplicates.
- **Email a patch.** Run `git format-patch` against your change and email the
output to **brainy@soulcraft.com**. This is a genuinely supported path, not
a fallback — plenty of good contributions arrive this way.
- **Open a pull request on the forge.** Request an account at
**source.soulcraft.com** (registration is request-with-approval, so allow
a little lag), clone, push a branch, and open a PR there. Maintainers
review and land it.
When creating an issue, include:
- Clear, descriptive title
- Detailed description of the problem
- Steps to reproduce
- Expected vs actual behavior
- System information (OS, Node version, Brainy version)
- Code examples if applicable
Either way, for anything beyond a small fix, opening an issue first (email is
fine) to talk through the approach saves everyone rework.
### Suggesting Features
## Development setup
Feature requests are welcome! Please provide:
- Clear use case
- Proposed API/interface
- Examples of how it would work
- Any potential challenges or considerations
### Pull Requests
#### Before Starting
1. Check existing issues and PRs
2. Open an issue to discuss significant changes
3. Fork the repository
4. Create a feature branch from `main`
#### Development Setup
**Quick Setup (Recommended):**
```bash
git clone https://source.soulcraft.com/soulcraftlabs/open-brainy.git
# Clone your fork
git clone https://github.com/your-username/brainy.git
cd brainy
# Run setup script (installs all dependencies including Rust)
./scripts/setup-dev.sh
```
**Manual Setup:**
```bash
# Clone your fork
git clone https://github.com/your-username/brainy.git
cd brainy
# Install system dependencies (Ubuntu/Debian)
sudo apt-get install -y build-essential pkg-config libssl-dev
# Install Rust (for WASM embedding engine)
curl --proto '=https' --tlsv1.2 -sSf https://sh.rustup.rs | sh
source ~/.cargo/env
rustup target add wasm32-unknown-unknown
cargo install wasm-pack
# Install Node.js dependencies
npm install
# Build Candle WASM embedding engine
npm run build:candle
# Build TypeScript
npm run build
# Run tests
npm test
```
Tests run on [Vitest](https://vitest.dev/). `npm test` runs the unit suite;
see `package.json` for `test:integration`, `test:coverage`, and friends.
#### Making Changes
## Standards
1. **Follow the code style**
- TypeScript for all source code
- Clear variable and function names
- Comments for complex logic
- JSDoc for public APIs
- **Strict TypeScript.** No `any` escape hatches to dodge the type checker.
- **Tests exercise real behavior.** No mocking away the thing you're supposed
to be testing.
- **No stubs, no TODO-code.** If something can't be finished, say so and
leave it out — don't merge a placeholder.
- **JSDoc on every exported function, class, and type.**
- **[Conventional Commits](https://www.conventionalcommits.org/).** `feat:`,
`fix:`, `docs:`, `perf:`, `refactor:`, `test:`, `chore:`. Never
`BREAKING CHANGE` in a commit message — major version bumps are a separate,
deliberate decision.
- **Performance claims are measured or labeled projected.** If a PR or its
description states a number, cite the benchmark that produced it (see
[docs/performance-envelopes.md](docs/performance-envelopes.md) for the
pattern). Don't state an estimate as if it were measured.
- **Measurements carry numbers, not provenance.** Public commit messages and
docs give the SHAPE a number was taken at and never where it was taken: no
hostnames, no store or deployment identities, no operational anecdotes about
someone's running system. "A 14,056-noun / 72,679-verb production-shaped
store, measured solo under an exclusive lock" tells a reader everything the
number depends on; the machine it ran on and whose data it was tell them
nothing except where somebody's infrastructure lives.
- **Documents that answer or reference a confidential specification never enter
this repository, even summarized.** The public docs describe THIS engine and
the published contract, and nothing else — a summary of a private document is
still that document's contents.
2. **Write tests**
- Add tests for new features
- Update tests for changes
- Ensure all tests pass
## License
3. **Update documentation**
- Update README if needed
- Add/update API documentation
- Include examples
Brainy is [MIT licensed](LICENSE). Contributions are accepted under the same
license — there's no CLA to sign.
#### Commit Guidelines
Thank you for considering a contribution.
Follow conventional commits format:
```
type(scope): description
[optional body]
[optional footer]
```
Types:
- `feat`: New feature
- `fix`: Bug fix
- `docs`: Documentation changes
- `style`: Code style changes
- `refactor`: Code refactoring
- `perf`: Performance improvements
- `test`: Test changes
- `chore`: Build/tooling changes
Examples:
```bash
feat(triple): add graph traversal depth limit
fix(storage): handle concurrent write conflicts
docs(api): update search method documentation
```
#### Submitting PR
1. Push to your fork
2. Create PR against `main` branch
3. Fill out PR template
4. Ensure CI checks pass
5. Wait for review
### Testing
#### Running Tests
```bash
# Run all tests
npm test
# Run specific test file
npm test tests/core.test.ts
# Run with coverage
npm run test:coverage
# Watch mode
npm run test:watch
```
#### Writing Tests
```typescript
import { describe, it, expect } from 'vitest'
import { Brainy } from '../src'
describe('Feature Name', () => {
it('should do something specific', async () => {
const brain = new Brainy()
await brain.init()
// Test implementation
const result = await brain.search("test")
expect(result).toBeDefined()
expect(result.length).toBeGreaterThan(0)
})
})
```
## Architecture Guidelines
### Adding New Features
1. **Check existing functionality**
- Review `ARCHITECTURE.md`
- Check if similar features exist
- Consider if it should be an augmentation
2. **Design considerations**
- Maintain backward compatibility
- Consider performance impact
- Think about all storage adapters
- Plan for extensibility
3. **Implementation checklist**
- [ ] Core functionality
- [ ] Tests (unit and integration)
- [ ] Documentation
- [ ] TypeScript types
- [ ] Examples
- [ ] Performance benchmarks (if applicable)
### Creating Augmentations
Augmentations extend Brainy's functionality:
```typescript
import { BrainyAugmentation } from '../types'
export class MyAugmentation extends BrainyAugmentation {
name = 'MyAugmentation'
async onInit(brain: Brainy): Promise<void> {
// Initialize augmentation
}
async onAdd(item: any, brain: Brainy): Promise<any> {
// Process before adding
return item
}
async onSearch(query: any, results: any[], brain: Brainy): Promise<any[]> {
// Process search results
return results
}
}
```
### Performance Considerations
- Use batch operations where possible
- Implement caching strategically
- Consider memory usage
- Profile performance impacts
- Add benchmarks for critical paths
## Documentation
### API Documentation
Use JSDoc for all public APIs:
```typescript
/**
* Searches for similar items using vector similarity
* @param query - Search query (text or vector)
* @param options - Search options
* @returns Array of search results with scores
* @example
* ```typescript
* const results = await brain.search("machine learning", { limit: 10 })
* ```
*/
async search(query: string | Vector, options?: SearchOptions): Promise<SearchResult[]> {
// Implementation
}
```
### Examples
Add examples for new features:
```typescript
// examples/feature-name.ts
import { Brainy } from 'brainy'
async function exampleUsage() {
const brain = new Brainy()
await brain.init()
// Show feature usage
// Include comments explaining what's happening
// Handle errors appropriately
}
exampleUsage().catch(console.error)
```
## Release Process
1. **Version bump**: Follow semantic versioning
2. **Update CHANGELOG**: Document all changes
3. **Run tests**: Ensure all tests pass
4. **Build**: Generate distribution files
5. **Tag**: Create git tag for version
6. **Publish**: Release to npm
## Getting Help
- **Discord**: Join our community
- **Issues**: Ask questions on GitHub
- **Discussions**: Share ideas and get feedback
## Recognition
Contributors will be recognized in:
- CHANGELOG.md for their contributions
- README.md contributors section
- GitHub contributors page
Thank you for contributing to Brainy! 🧠

View file

@ -1,5 +1,5 @@
<p align="center">
<img src="https://source.soulcraft.com/soulcraftlabs/open-brainy/raw/branch/main/brainy.png" alt="Brainy" width="180">
<img src="https://raw.githubusercontent.com/soulcraftlabs/brainy/main/brainy.png" alt="Brainy" width="180">
</p>
<h1 align="center">Brainy</h1>
@ -11,9 +11,9 @@
</p>
<p align="center">
<a href="https://source.soulcraft.com/soulcraftlabs/-/packages/npm/brainy"><img src="https://img.shields.io/badge/package-The%20Source-2c3e50.svg" alt="Package on The Source"></a>
<a href="https://source.soulcraft.com/soulcraftlabs/open-brainy"><img src="https://img.shields.io/badge/repo-open--brainy-2c3e50.svg" alt="Repository"></a>
<a href="https://source.soulcraft.com/soulcraftlabs/open-brainy/actions"><img src="https://source.soulcraft.com/soulcraftlabs/open-brainy/actions/workflows/ci.yml/badge.svg?branch=main" alt="CI"></a>
<a href="https://www.npmjs.com/package/@soulcraft/brainy"><img src="https://img.shields.io/npm/v/@soulcraft/brainy.svg" alt="npm version"></a>
<a href="https://www.npmjs.com/package/@soulcraft/brainy"><img src="https://img.shields.io/npm/dm/@soulcraft/brainy.svg" alt="npm downloads"></a>
<a href="https://github.com/soulcraftlabs/brainy/actions/workflows/ci.yml"><img src="https://github.com/soulcraftlabs/brainy/actions/workflows/ci.yml/badge.svg" alt="CI"></a>
<a href="https://soulcraft.com/docs"><img src="https://img.shields.io/badge/docs-soulcraft.com-blue.svg" alt="Documentation"></a>
<a href="LICENSE"><img src="https://img.shields.io/badge/license-MIT-blue.svg" alt="MIT License"></a>
<a href="https://www.typescriptlang.org/"><img src="https://img.shields.io/badge/%3C%2F%3E-TypeScript-%230074c1.svg" alt="TypeScript"></a>
@ -23,15 +23,12 @@
<a href="#quick-start">Quick start</a> ·
<a href="#one-query-three-engines">One query</a> ·
<a href="#feature-tour">Features</a> ·
<a href="#when-you-outgrow-brainy">Scale with Cor</a> ·
<a href="#documentation">Docs</a> ·
<a href="#support--community">Support</a>
<a href="#from-laptop-to-hundreds-of-millions">Scale with Cor</a> ·
<a href="#documentation">Docs</a>
</p>
---
**Open Brainy** is the MIT engine — the open API, client library, types, and protocol; an openly specified canonical on-disk format; and this TypeScript reference engine, scoped as a single-node engine for stores up to roughly one million rows. `@soulcraft/brainy` 10.4.2 was the last release under the old package name — the name passes to the native engine, **Brainy**, at 11.0.0: the same API over the same open format at production scale, and it requires a license.
Built because we were tired of stitching a vector store to a graph database to a document store — and spending weeks on plumbing before writing a line of business logic. Brainy indexes every fact **three ways at once** and lets one call query them together:
| You write | Brainy indexes it as | You query it with |
@ -47,14 +44,12 @@ It runs **inside your process** — no server, no Docker, nothing to operate —
## Quick start
```bash
bun add @soulcraftlabs/brainy # Bun ≥ 1.1 — recommended
npm install @soulcraftlabs/brainy # Node.js ≥ 22
bun add @soulcraft/brainy # Bun ≥ 1.1 — recommended
npm install @soulcraft/brainy # Node.js ≥ 22
```
> **Registry**: add `@soulcraftlabs:registry=https://source.soulcraft.com/api/packages/soulcraftlabs/npm/` to your `.npmrc` (anonymous read).
```javascript
import { Brainy, NounType, VerbType } from '@soulcraftlabs/brainy'
import { Brainy, NounType, VerbType } from '@soulcraft/brainy'
const brain = new Brainy() // in-memory; one line swaps to disk
await brain.init()
@ -177,11 +172,9 @@ await brain.vfs.search('React components with hooks') // semantic file
**[Multi-process model](docs/concepts/multi-process.md)** · **[Inspection guide](docs/guides/inspection.md)**
## When you outgrow Brainy
## From laptop to hundreds of millions
Brainy's pure-TypeScript engines carry real workloads a long way on their own — see the measured, per-operation numbers (not marketing figures) in **[docs/performance-envelopes.md](docs/performance-envelopes.md)** for what to expect, unaccelerated, on plain filesystem storage.
When a deployment needs native-scale vector/graph performance — memory-mapped indexes that don't need your dataset in RAM, billion-scale ambitions — add the native engine. **The API doesn't change:**
Brainy's TypeScript engines take you a long way. When you outgrow them, add the native engine — **the API doesn't change**:
```bash
npm install @soulcraft/cor
@ -194,14 +187,13 @@ await brain.init() // @soulcraft/cor detected — same code, native engines un
Installing the package is the opt-in: if `@soulcraft/cor` is present, it loads and announces itself in the init log; if it's present but broken, `init()` **throws** — an installed accelerator never silently vanishes behind the JS engines. Opt out with `plugins: []`, or pin exactly what loads with `plugins: ['@soulcraft/cor']`. [`@soulcraft/cor`](https://www.npmjs.com/package/@soulcraft/cor) (Brainy 8.x ↔ Cor 3.x, version-matched) registers Rust implementations behind every provider seam: SIMD distance kernels, memory-mapped storage, a disk-native vector index that doesn't need your dataset in RAM, durable LSM field/graph indexes that serve cold opens instantly, and native aggregation. Recall@10 measured **0.99 / 0.96 / 0.96 at 1M / 10M / 100M vectors** in Cor's release gate.
Open core, commercial accelerator: Brainy is MIT and complete on its own — Cor is more headroom for when you need it, not capability held back to sell you later. Licensing and support: **cor@soulcraft.com**.
Open core, commercial accelerator: Brainy is MIT and complete on its own; Cor is licensed and funds both.
## Performance
- Per-operation p50/p95 at 1k and 10k entities, pure-JS floor, measured and re-run every release that touches a measured path: **[docs/performance-envelopes.md](docs/performance-envelopes.md)**.
- JS distance kernels: **~6× faster cosine, ~1.4× euclidean** than 7.x (measured: [`tests/benchmarks/distance-microbench.mjs`](tests/benchmarks/distance-microbench.mjs), 384-dim, median of 41).
- Whole-graph reads are single **O(N + E)** cursor walks — a consumer-measured 19k-edge export dropped from ~27 s of per-node calls to one scan.
- Capacity planning and architecture: **[docs/PERFORMANCE.md](docs/PERFORMANCE.md)** · **[docs/SCALING.md](docs/SCALING.md)**
- Full numbers and capacity planning: **[docs/PERFORMANCE.md](docs/PERFORMANCE.md)** · **[docs/SCALING.md](docs/SCALING.md)**
## Use cases
@ -220,10 +212,6 @@ Open core, commercial accelerator: Brainy is MIT and complete on its own — Cor
**Bun ≥ 1.1** (recommended) or **Node.js ≥ 22**. Brainy 8.x is server-only; the 7.x line remains on npm for browser use.
## Support & community
## Contributing & license
- **Bugs and ideas****brainy@soulcraft.com** — no account needed, you'll get a receipt.
- **Security reports****security@soulcraft.com** — see **[SECURITY.md](SECURITY.md)**.
- **Contributing** → see **[CONTRIBUTING.md](CONTRIBUTING.md)**.
MIT © Brainy Contributors.
Contributions welcome — see **[CONTRIBUTING.md](CONTRIBUTING.md)**. MIT © Brainy Contributors.

View file

@ -1,898 +1,15 @@
# @soulcraft/brainy — Release Notes for Consumers
This file is the **quick reference for downstream sessions** tracking Brainy changes.
Full auto-generated changelog: `CHANGELOG.md` · Releases: https://source.soulcraft.com/soulcraftlabs/open-brainy/releases
Full auto-generated changelog: `CHANGELOG.md` · Releases: https://github.com/soulcraftlabs/brainy/releases
**How to use:** Brainy is the underlying data engine for downstream applications. Read this when:
- Upgrading `@soulcraft/brainy` in your application
- Debugging data, query, or storage behaviour
- A new Brainy feature is available that you want to adopt
## Removed APIs — 7.x → 8.x (the complete ledger)
Every public API removed at the 8.0 major, with its sanctioned replacement. If your code
still calls a left-column name on 8.x it throws (or the config key is rejected) — the
replacement is always a one-line change. (Standing contract from 8.9.0 forward: removals
happen only at majors, after ≥1 minor of loud runtime deprecation naming the replacement.)
| Removed (7.x) | Replacement (8.x) |
|---|---|
| `brain.search(query, k)` | `find({ query })` — semantic; `find({ query, searchMode })` for hybrid |
| `brain.getRelations({...})` | `related(id, opts)` for adjacency; `find({ connected: {...} })` for scoped traversal |
| `brain.neural()` clustering | `find({ vector })` + aggregation `GROUP BY` |
| `Db.search()` | `db.find({ vector })` |
| Pre-8.0 storage path aliases (`directory`, `basePath`, …) | one `storage.path` key (old aliases throw) |
| Reserved keys inside `metadata` bags (silently remapped in 7.x) | top-level params (`subtype`, `visibility`, `confidence`, `weight`, …) — reserved-in-bag throws |
| 7.x COW branches layout (`branches/main/`) | generational MVCC (`asOf()`, `now()`, `db.persist(path)`) — on-disk migration is automatic at first 8.x open |
The fork/snapshot family (`brain.snapshot()`, `createSnapshot()`, `restoreSnapshot()`)
is sometimes cited as a 7.x removal — those methods never existed on 7.x; the 8.0 Db API
(`asOf`/`persist`/`restore({confirm})`) is their first real implementation.
---
## v10.4.4 — 2026-08-28
**A correctness and observability release.** The headline is not speed: it is that a
restart now tells you the truth about itself, a store stops lying about how much it
holds, and the engine stops doing work nobody asked for. There is a performance
improvement and it is modest; it is stated exactly below rather than rounded up.
### The dark restart — fixed at the root
A service could stop cleanly, exit 0, having awaited `close()` on every store it held,
and its next boot would announce `Overwriting stale writer lock … appears dead` for
every one of them. Nothing had crashed. Two deployments hit this; the same defect also
made those boots pay a crash-recovery fold they did not owe.
The cause was not the lock. `close()` released it correctly — when it got there. A
failure part-way through close skipped both the release AND the clean-shutdown marker,
and "the recorded pid is gone" reads identically for an orderly restart and a crash.
- `close()` is now two parts and the second is unconditional: the flush-request watcher,
the **writer lock**, the VFS timers and the terminal `closed` flag are released whether
the durable steps succeeded or not. The original failure is narrated with what it costs
the next open, then rethrown.
- Releasing the lock writes a **clean-close record** naming the lock generation it gave
up. The next open reads that record instead of guessing: recorded → nothing to recover;
absent → it says so, and names the recovery it is about to run. This also ends two
long-standing false alarms — a recycled pid locking a store out of its own reopen, and
`Re-acquiring writer lock … this is a bug` after a perfectly clean close.
- The signal path stopped failing in a batch. One store's failing flush used to strand
every remaining store's lock and markers — at exit code 0. Now: per-store isolation, the
generation store's close (the marker) is part of shutdown, the lock goes in a `finally`,
and the handler no longer calls `process.exit()` when the host application has its own
signal handler, a race that truncated the host's own shutdown mid-flight.
### The count ledger stops lying, and `counts.json` is written atomically
The all-tier scalars are the denominator a coverage check subtracts against. A ledger
derived under the old rule — one entity per id DIRECTORY — counted ghost and scar
containers as rows, and was only FLAGGED suspect: it went on serving wrong numbers for
the life of the store. Two copies of one archive could disagree, and a downstream index
heal reported remaining work that did not exist.
- Such a ledger now derives itself honestly **in the background** after the open, counting
identity records, and persists the correction stamped. Nothing waits for it, because no
read is served from a denominator.
- A derivation that raced a write refuses to stamp its number: one retry on a quiet store,
then the ledger stays SUSPECT and names `repairIndex()` as the door that recounts under
a barrier.
- `counts.json` is written temp+rename. A truncating write left a window in which a
concurrent reader saw the file EMPTY — and an unparseable ledger sends the next open
down the full-rescan path, so the cheapest file in the store was buying the most
expensive recovery.
### An open and a repair narrate themselves — on a channel a log level cannot silence
A store could open for three minutes and print nothing at all. The phase timings existed;
they were written to a channel that every production-looking environment clamps away.
- Narration moved to an always-visible channel. An open now heartbeats the phase it is in,
names each phase as it ends with what it was paying for, and names the expensive STEP
inside a phase. `repairIndex()` does the same and its receipt carries a per-family
`durationMs` — a repair that ran for half an hour with no output could only be watched
through `top`.
- A brain nobody has written to now does nothing: a flush over a clean store is a no-op
and says nothing, the graph index's auto-flush asks before it acts, and the
cross-process flush-request watch is **event-driven** (`fs.watch`) instead of polling a
directory every 500 ms per store forever, with a slow safety sweep behind it and a
narrated fall back to polling where a filesystem cannot be watched.
- A provider that is REBUILDING ITSELF is no longer confused with a broken one. `init()`
does not wait for it, every other family serves, and that family's doors refuse **by
name, carrying the provider's own progress**, saying plainly that they open by
themselves and no action is needed. Health narration dedupes by content, so an unchanged
verdict is silent however a provider's generation counter moves.
### For operators — one behaviour change
**Four `where` operators that previously returned an empty page now raise
`INVALID_QUERY`:** `startsWith`, `endsWith`, `matches` and `length`. An equality/range
posting index cannot evaluate a substring, a pattern or an array length without reading
every row, and it now refuses by name instead of answering with an empty result that
looks like an answer.
**Three that previously returned an empty page are now SERVED:** `hasAll`, `noneOf` and
`excludes`. All 25 accepted operator tokens now agree between this engine and its
accelerated counterpart.
### Performance — stated exactly
Measured on a 14,056-noun / 72,679-verb production-shaped store, both builds solo under
an exclusive lock:
- **Warm reopen after a clean close: 85.7 s → 77.0 s (10.2%).** The whole of that gain is
one fix — generation discovery reads directory NAMES instead of recursively walking the
entire generation log (9.2 s, and it scales with history rather than row count). The
VFS phase is **unchanged**.
- **Cold open: 31.4 s** (518.1 s → 486.7 s), of which the count-ledger derivation moving
off the critical path accounts for storage-init dropping 5,941 ms → 25 ms.
- **A dominant ~38 s remains, diagnosed and NOT fixed.** It is not the VFS — the VFS's own
init is under 2 s of that phase. It is the log-authority adoption and/or the
pending-embed log recovery, both now instrumented so the next measurement names the
culprit outright.
Continuing work, named so nobody has to rediscover it: that ~38 s term; making the
generation store's committed-range set lazy; the hydration path that substitutes
`Date.now()` for an unreadable stored timestamp (inventing data); and a VFS path-prefix
filter built with a `$startsWith` spelling no operator set accepts, so
`searchFiles({ path })` throws today.
---
## v10.4.3 — 2026-08-27 (Open Brainy's first release)
**`@soulcraftlabs/brainy` 10.4.3 is the same engine as `@soulcraft/brainy` 10.4.2, byte for
byte — only the name, the registry, and the pointers changed.** Install:
```bash
npm install @soulcraftlabs/brainy
```
with the registry line in your `.npmrc` (anonymous read):
```
@soulcraftlabs:registry=https://source.soulcraft.com/api/packages/soulcraftlabs/npm/
```
- **The Source is the one registry.** Open Brainy publishes to source.soulcraft.com only; the
npmjs republish step is retired from the release rail. Existing npmjs versions of
`@soulcraft/brainy` stay as they are and receive no new versions.
- **The repository moved** to `soulcraftlabs/open-brainy` on The Source; the old path redirects.
- **No engine change.** Everything in the 10.4.2 notes applies unchanged; adoption is one
install-line change (`@soulcraft/brainy``@soulcraftlabs/brainy`), which downstream
applications make together with their native-engine bump.
## v10.4.2 — 2026-08-27 (a zero-norm vector is not a vector)
**This is the last release of the MIT engine under the `@soulcraft/brainy` name.**
The MIT package continues as **Open Brainy**`@soulcraftlabs/brainy`: the open API,
client library, types and protocol, an openly specified canonical format, and the TypeScript
reference engine, scoped honestly as a single-node engine for stores up to roughly one
million rows. The `@soulcraft/brainy` name passes to the native engine, **Brainy**, at a
major version bump; that engine implements the same API over the same open format at
production scale, requires a license, and refuses loudly without one. Nothing changes
for existing installs until that major ships; the move is announced with it.
Six fixes, one law: a vector with no magnitude carries no information, so it must
never reach a vector index — in any engine — and the canonical store must say so.
- **The permanently-unvectored row.** `add({ ..., vector: [] })` (and the same item
shape in `addMany` / `transact`) is now the sanctioned "no vector" row: persisted
with an empty vector leg, never embedded, never indexed, counted as unvectored in
the canonical ledger. Metadata-only rows — telemetry tallies, counters, plumbing —
no longer need a placeholder vector and never enter the vector leg. `vector: []`
together with `deferEmbedding: true` is refused with a typed error (a supplied
vector has nothing to defer). Previously `vector: []` threw a dimension error.
- **The unvector door.** `update({ id, vector: [] })` (and its `transact()` twin) is
the sanctioned way to strip a vector from an existing row: canonical vector → `[]`,
removal from the vector index, the vectored ledger decremented exactly once — and
idempotent, so a resumed cleanup pass may simply re-issue. It never re-embeds, and
it clears a pending deferred-embed marker durably so the background worker cannot
re-vector the row later. Note that a rebuild never sheds vectors (it re-derives the
index from canonical rows); shedding historical vectors needs this door.
- **Zero-norm vectors are normalized at the write.** An explicit all-zero vector on
any write path is persisted as unvectored (`[]`) with one warning naming the row;
the vector-index operations keep their own refusal as a second line. The engine's
own VFS root, which used to persist a deliberate all-zero placeholder (harmless
under cosine distance, a false attractor under a downstream engine's
squared-euclidean serving — a production incident this week), is now created
unvectored, and an existing store's legacy root is migrated on open by a single
fixed-path read before the health gate runs — never a walk.
- **Enumeration keys on the identity record.** `getNouns()` / `getVerbs()` and the
cursor walks behind them enumerate by the metadata record, the same key the
canonical ledger counts by — previously the walk keyed on the vector file, so a
row holding metadata but no vector was counted yet never yielded (a permanent
"missing" phantom in coverage math), while an orphaned vector-only directory
could be yielded as a phantom id. The recovery fold also never deletes an existing
vector when it replays a metadata-only after-image (preserve-if-absent). One
documented gap remains: a verb's endpoints live only in its vector leg, so a
metadata-only verb is counted and loudly skipped, never fabricated — the fix is a
canonical-format change and lands with the open format.
- **The ledger's one-time derivation counts identity records.** Stores upgraded from
pre-ledger versions derived their ALL-visibility scalars once by counting id
directories, which included ghost and scar containers left by an old partial-delete
defect — an inflated denominator whose coverage row could never reach exact. The
derivation now counts only directories holding a metadata record, `counts.json`
carries a derivation-rule stamp, and a ledger derived under the old rule is marked
`suspect` at open (one O(1) field read, one warning) so the online `repairIndex()`
path clears it with a real recount.
- **The vector index refuses what it cannot hold.** `rebuild()` skips unvectored and
zero-norm rows (one summary line), re-pins the vector dimension from the first real
vector after a restart (previously a restart left the pin unset, so a wrong-length
insert became the new pin instead of being rejected), and `addItem` / `updateItem`
throw a typed `EmptyVectorIndexError` on a length-0 vector instead of ever storing
a vector-less node.
- **Smaller:** a failing plugin activation now rethrows with the original error as
`cause` (the originating file and line survive to the caller's log); build
generators stamp from the repository history of their inputs instead of wall clock,
so two builds of the same tree are byte-identical.
Adoption: one restart, paired with its native-engine release. The first open of an
existing store runs the legacy-root migration (one narrated line) and, on stores that
upgraded from pre-ledger versions, marks the ledger suspect until the next sanctioned
recount — no rebuild in either case.
## v10.4.1 — 2026-08-26 (reads refuse per family; an unchanged write never re-embeds)
Two production defects from the same week, fixed together as a patch to 10.4.0.
- **The read gate is per family.** A read now refuses only when the index family it
actually consults is unhealthy: a metadata filter is served while the vector leg is
rebuilding; a semantic query is refused only by the vector family; a graph
traversal only by the graph family. Previously any unhealthy family refused every
read on the brain — under a long vector rebuild, a production deployment's
metadata-only reads were refused for the duration, and the retries became a write
pump of their own.
- **Unchanged data never re-embeds.** `update()` compares the incoming `data`
structurally with the stored record; an update carrying identical data (a common
shape for periodic upserts) no longer embeds again and no longer churns the vector
leg. Previously every such update re-embedded and re-inserted, which under load
saturated the vector index with near-identical vectors.
Adoption: one restart, paired with its native-engine release.
## v10.4.0 — 2026-08-25 (the health report has a name)
Three related cures, one root cause: an index deciding whether it could be trusted
by sampling itself instead of by exact accounting. This release replaces every
sampled self-probe with ledger-derived truth, and a read against an unhealthy index
now refuses loudly instead of guessing.
- **The canonical count ledger.** Storage now tracks two scalars per family
(nouns/verbs) on the write path: the user-facing `counted` total — unchanged,
still what `getNounCount()` / `getVerbCount()` return — and a new ALL-visibility
`all` total covering every tier, the real denominator a derived index's own
coverage math needs. The unfiltered storage-level `totalCount` returned by
`getNouns()` / `getVerbs()` is now this unclamped ALL scalar; previously it could
only ever move up (`Math.max(scalar, scanned)`), so an inflated counter could
never self-correct. A delete that cannot prove the record it removed actually
existed (no canonical read, no prior image available) no longer decrements on
faith — it marks the ledger `suspect` (narrated once per session) instead of
silently drifting, and the next `repairIndex()` clears the flag with a real
recount.
- **One contract for a throwing health probe.** A provider's `validateInvariants()`
is documented to never throw — but if one does anyway (a bug, a transient fault),
it is now read the same way everywhere: `heal: 'none'`, the error named in the
report, never synthesized into a rebuild trigger and never swallowed into "looks
fine." A flaky check can no longer buy itself a rebuild. `repairIndex()`'s
per-family receipt also gains `missing` (an exact count plus a capped id sample),
`rebuilt` (a full rebuild ran, vs. an incremental heal), and `reason`.
- **The named health report; reads refuse instead of rebuilding.** Any index
provider may now expose a synchronous, O(1) `healthReport()` — composed from the
provider's own exact ledgers, never a sample — and this is the one signal
Brainy's read gate trusts. The first-query lazy-build path is gone: `brain.init()`
now runs every needed rebuild to completion before it returns, always, regardless
of dataset size. A read that lands on a provider whose health report says it
isn't serving throws a typed error instead of triggering a rebuild mid-query —
`GraphIndexNotReadyError`, `MetadataIndexNotReadyError`, or
`VectorIndexNotReadyError` (all exported from `@soulcraft/brainy`), naming the
reasons. `repairIndex({ rebuild: ['metadata' | 'graph' | 'vector'] | 'all' })` is
the new explicit operator door: it rebuilds the named family unconditionally, no
health check consulted — reach for it when you have independent reason to
distrust a family regardless of what it self-reports. Bare `repairIndex()` is
unchanged in spirit: report-driven, heals only what its own checks say needs it.
- New concept doc: [Index Health](docs/concepts/index-health.md) walks the whole
story from a consumer's side — degraded-but-serving vs. not-ready, what
`repairIndex()` checks and heals per family, what `suspect` counts mean.
**Nothing to change to adopt this.** No API removed, no signature narrowed —
`repairIndex()` gains an optional options bag and its return value gains fields,
both additive. The honest notes: if your code ever relied on a `find()` against a
cold/not-yet-built index quietly triggering a rebuild and returning results a beat
later, that behavior is gone — it now throws one of the three typed
`*NotReadyError` classes instead (catch them if you need to distinguish "not ready
yet" from "no results"). And `disableAutoRebuild: true` no longer defers index
construction to the first query — a needed rebuild always runs at `open()` now;
the flag has no effect on timing. Full manual control still lives in
`repairIndex({ rebuild: [...] })`.
- **Crash-reopen catchup.** After an unclean shutdown, the metadata index now
folds the exact fact window it missed — `find()` serves every acked write on
reopen, closing the gap where canonical reads and counts recovered a
crash-window write but the index kept serving its pre-crash state until the
next full rebuild. Related root-cause fixed alongside: `close()` never
stamped the index watermarks (only `flush()` did), so a close without a
prior flush caused a needless full rescan verdict on the next open.
- **Relation rows are live in the metadata index.** Previously verb rows
entered the metadata index only during a rebuild — so a rebuilt store's
relation postings went stale from the first `relate()` after it. Relations
are now posted and retracted on the live write path (relate / unrelate /
updateRelation / remove's cascade, and their `transact()` forms), in the
same commit as the graph leg.
- **The metadata rebuild is online.** `rebuild()` for the metadata family no
longer clears and rebuilds in place (reads went empty for the duration): it
builds a complete replacement beside the serving index, mirrors concurrent
writes to both, swaps atomically, and persists once after the swap. Reads
never observe a partial index. `repairIndex({ rebuild: ['metadata'] })` uses
it automatically.
- **Incremental heal is routed.** A provider invariant that asks for the
incremental heal (`heal: 'repair'`) now routes to the provider's own
`repair()` when it exposes one — re-posting exactly what its ledger names,
never a store-sized rebuild — and the post-heal re-read of the report decides
success; a repair that doesn't converge is recorded with the escalation named.
- **The vector family joins the count ledger.** `getCanonicalCounts()` gains
`vectors: { all }` — the count of canonical entities holding a real vector
(deferred-embed entities count when their vector lands). And the open gate
closes the vector leg: a store whose canonical rows hold vectors but whose
derived vector index is empty now builds at `open()` (or refuses with the
typed error) instead of silently serving empty vector-search results.
- **An unknown storage config shape fails loudly.** A nested `config` object
carrying a path-shaped key (a shape that was never supported) used to fall
through silently to the default shared directory — every instance writing one
store while callers believed each had its own. It now throws, naming the
canonical `path` key.
- **Relation index rows are JSON-safe.** Internal endpoint identifiers can no
longer ride the metadata-index crossing (a native provider serializes it);
they stay on the graph operations where they belong.
- **A broken accelerator install can never read as "not installed."** The
auto-detection free pass now requires the resolution error to name the
accelerator package itself, exactly — a missing platform-binary sibling
package, an inner file path, or a dependency failure is a broken install and
`init()` throws loudly. And a plugin that declines activation is narrated on
the always-on log channel, so `silent: true` can no longer hide a fallback
to the default engines.
---
## v10.3.1 — 2026-08-18 (the fold that behaves)
Three recovery cures from one production first-boot incident (a brain's first
process restart after a live storage-authority flip looked hung and was
restarted three times mid-recovery). **Adopt this version before flipping
brains with existing history** — it is the intended adoption target for
fleets moving to the crash-safe authority.
- **Recovery streams.** The boot-time log fold now consumes the generation
log one segment-batch at a time — memory stays bounded at one segment for
any log size. Previously it materialized every fact into one array, which
on a ~7k-fact log produced multi-GB allocation pressure and a process that
looked wedged while it worked.
- **Recovery narrates.** The fold announces itself before the work begins
("recovery fold beginning — do not restart, the fold is finite") and prints
progress every thousand facts. A visible fold gets to finish; a silent one
gets killed by a well-meaning operator, and each kill makes the next boot
pay the whole fold again.
- **Bounded recovery from the flip itself.** Adopting the log authority now
founds the recovery checkpoint at the moment of the flip (one paged
canonical sync, bounded memory, then the stamp) — so even the FIRST unclean
shutdown after a flip replays only the log's tail. Previously the bound
could only establish itself at a completed crash recovery, which is exactly
the recovery the incident kept interrupting.
---
## v10.3.0 — 2026-08-18 (the trust-and-provenance release)
Four consumer-driven cures. Pairs with the same native accelerator line
(>=4.1.0); adopt alongside the accelerator's 4.2.0 for its paired fixes.
- **Writer-lock fencing.** A live writer is never auto-evicted (staleness now
requires the holding process to be dead — a >60s stall is a slow writer, not
a dead one); the lock claim is atomic (no empty-file window a racer can
misread as torn); and every flush commit and transact barrier verifies lock
ownership first, so a forced-out or lock-deleted writer fails typed
(`BRAINY_WRITER_FENCED`) instead of writing on unaware — the split-brain
class a shared dev store hit is dead at all three roots. The documented
same-process re-open ("warn and take over") stays benign: ownership is
per-process. Consumers that raised stop-timeouts as mitigation can retire
them.
- **Transaction-log provenance.** `TxLogEntry` gains an optional `origin`
field — absent means a user write (existing consumers unchanged);
engine-originated commits stamp themselves (`system:embed-landing`,
`system:adoption-backfill`, `system:reconcile`), and the same stamp rides
the commit fact's meta. Activity feeds filter on fact instead of guessing;
a reported "double tick" (the deferred vector landing indistinguishable from
a user save) is cured without collapsing genuine rapid saves.
- **The attested reconcile door.** `reconcileLogDivergence(id, {attest})`
resolves the one adoption-refusing divergence class
(`log-live-canonical-absent`) with a human's word: `'deleted'` mints the
tombstone the log always lacked; `'restore'` folds the log's only copy back
into canonical; wrong-class calls refuse typed with nothing written. Loud,
narrated, single-row.
- **Iron-honest test budgets.** The wall-clock micro-budgets are recalibrated
as order-of-magnitude guards (3x the worst measurement across three machine
classes) so honest hardware differences can never again read as failures;
real performance enforcement lives in the dedicated perf lanes.
---
## v10.2.0 — 2026-08-17 (adoption completes in one call)
One fix, headline-sized for large stores. Pairs with the same native accelerator
version as 10.1.0 — no accelerator bump needed.
- **The adoption backfill runs to completion.** Adopting the crash-safe storage
authority first re-commits every row the log never saw (a one-time baseline
backfill). That backfill had a fixed ceiling of 800 rows per
`adoptLogAuthority()` call — sized for small drift, not for a large pre-existing
store — so a store with a 12,700-row baseline advanced 800 rows per call and
stayed on the prior authority across restarts (a production deployment's
report). Now one call adopts a baseline of any size: the backfill sees the
entire curable set at once, cures all of it, and loops only until green — the
no-progress guard is the sole stop. Pace rides the write path (~100 rows/s
measured end to end, versus ~1.7 rows/s under the old page-per-scan shape),
and progress is narrated so an operator watching a live service sees motion.
Stores that already adopted are unaffected; stores still on the prior authority
flip in a single call on their next open or on an explicit
`adoptLogAuthority()`.
- Verification report unchanged on the wire (still lists at most 200 mismatches;
counts remain complete) — only the adoption path reads the full set.
---
## v10.1.0 — 2026-08-13 (the bounded-recovery and write-path-cure release)
The theme: **crash recovery is bounded, restores are durably founded, and two
production-reported write-path defects are cured at their roots.** Ships together
with the matching native accelerator version; adopt as a pair.
- **Bounded crash recovery (the fold-checkpoint bound).** Recovery after an unclean
shutdown now replays only the log segment above a durably-stamped checkpoint
instead of the whole log. The checkpoint advances only after a canonical-sync
barrier makes every touched record durable (deletes included), so the bound can
lag but can never overstate durability. Existing stores converge automatically at
their first recovery — zero operator steps; recovery cost stops scaling with
store age.
- **Restores are unclean events, by construction.** `restore()` now runs its swap
fully quiesced (no background flush can race the directory replacement — a
consumer-reported `ENOTEMPTY` crash class is dead), and a snapshot's durability
stamps never survive the restore: the reopen folds the restored log, re-syncs
what it re-applied, and stamps fresh. Restored state is durably founded at
restore time instead of inheriting assertions about bytes the disk never synced.
- **Write-path cures from a production report.** (1) Log pad-frame construction is
total — a size-class boundary hole could previously kill a sync with "pad frame
not constructible". (2) The at-ack sync-failure compensation now splits by phase:
the generation counter can never re-mint a number the log may already carry, so
the non-monotonic append refusal loop reported by a downstream deployment cannot
recur. Both pinned with the reporter's exact shapes.
- **Operator-truthful sparse queries.** `where` on a field no store row has ever
carried now serves the honest answer (`eq`/`in`/range → empty; `ne`/`exists:false`
→ all rows; `exists:true` → empty) with a throttled did-you-mean warning, instead
of refusing. `orderBy` on unknown fields and ambiguous spellings keep their typed
refusals.
- **Cross-package error identity.** `UnresolvableFieldError` thrown across package
boundaries is re-normalized so `instanceof` checks in consuming applications
match regardless of duplicated dependency trees.
- Release tooling: publishes now push the tag before the branch (the publish
workflow can no longer queue behind a redundant CI run) and verify registry
byte-identity with a propagation-tolerant raw-registry probe.
---
## v10.0.0 — 2026-08-10 (the write-path and lifecycle release)
The theme: **writes ack fast and honestly, startup adopts instead of rebuilding, and
every query path serves, announces, or refuses — never silently degrades.** Ships as
one release together with the matching native accelerator version.
**The storage-authority posture (the release's headline):** a NEW brain's default
is **durable-at-ack log authority** — the generation log is the source of truth,
every write acknowledgment is covered by a group-committed fsync, and crash
recovery is a replay of the log (an acked write survives power loss, proven by
fault-injection tests). An EXISTING brain adopts at its first open under 10.0.0,
gated by a verification oracle: the log is replayed and diffed against stored
truth record-by-record; curable gaps are backfilled; the brain flips only on a
green verdict and a brain that cannot verify stays on the previous posture and
says so loudly. The explicit opt-out is `logAuthority: 'defer'` in the config
(no automatic adoption; flip later with `adoptLogAuthority()`).
**Why a major:** the generation log gains write format v2 — new segments carry typed,
versioned records with integrity seals. A 9.x build refuses a v2 segment with a clear
version-naming error (never a misread), which means **a brain written by 10.x cannot
be opened by 9.x**. Existing v1 history stays readable forever; upgrading requires no
migration and no data touch — the format moves forward only as you write.
### New capabilities
- **`deferEmbedding: true`** on `add()`/`update()`: the write acks at durability; the
embedding runs on a crash-safe background worker and the vector swaps in atomically.
The row is id/metadata-findable immediately; semantic recall converges when the embed
lands. Barriers and gauges: `awaitPendingEmbeds()`, `waitForIndexed('semantic')`,
`getIndexStatus().pendingEmbeds`. VFS file writes adopt this end to end — file-write
ack no longer waits on a neural net (measured ~50× faster serial writes on a
production-shaped corpus).
- **`waitForIndexed(path?, { generation?, timeoutMs? })`** — the one honest read
barrier for write-then-recall flows. Typed timeout error naming what was still
pending; never a silent partial wait.
- **Engine-owned persistence cadence** (`persistence.policy: 'auto'`, now the default):
the engine flushes on write-count/interval/idle triggers in the background,
single-flight. **Delete `flush()` calls from hot paths**`flush()` remains as an
awaitable durability barrier. A hung flush can never block a write ack.
- **Time-travel recall contract**: `asOf(G).find()` serves vectors exactly as they
stood at G — a later update never leaks into an earlier pin; deleted rows mask;
beyond-head pins refuse typed.
- **Log-authority storage (opt-in, per brain)**: `verifyLogAuthority()` audits the
generation log against stored truth record-by-record and names every divergence;
`adoptLogAuthority()` flips a brain to log-authoritative storage only on a green
audit (self-healing curable divergences first), enabling durable-at-ack writes:
concurrent writers share one fsync and an acked write survives power loss, by
construction (crash-recovery replay is pinned by fault-injection tests).
### Behaviour changes
- **`find({ where: {} })` now serves match-all** (previously returned an empty result
silently — warm and cold). Same fix applies to count, streaming, and graph-scoped
seeding paths.
- **`removeMany({ where: {} })` now refuses with a typed error** — a match-all bulk
delete must be explicit, never inherited from an empty filter object.
- **Aggregations always answer**: state persists at every `flush()` (not only close),
an unclean exit reconciles incrementally instead of rescanning the store, and
deletes without a before-image flag a loud rescan instead of silently skipping.
- **Vector updates are atomic in place** — a row is never transiently absent from
search during an update (the "flicker" class is gone); type-only re-index of an
unchanged vector is a no-op.
### Format note
- The generation log gains **format v2** (typed, versioned records with integrity
seals). v1 segments remain readable forever; new segments write v2. Older brainy
builds refuse v2 segments with a clear version-naming error rather than misreading
them. Records reserve encryption fields for a future release — zero behaviour today.
## v8.11.0 — 2026-07-27 (canonical enumeration mode for export — storage-walked, canon-complete)
From a fleet data-migration program's requirement for whole-brain exports that are
provably canon-complete: `export()`'s default enumeration for a whole-brain/predicate
selector is a generation-correct paginated `find()` walk — a projection query riding
the metadata index as an acceleration structure. Production has documented both of the
index's failure classes: a lost/stale posting can silently OMIT a canonical record from
an export, and a stale posting can silently INCLUDE a phantom row. Neither is visible
to the caller today.
- **New: `export(selector, { enumeration: 'canonical' })`** (default remains `'index'`
unchanged behavior on this release). Canonical mode walks every live noun/verb
directly off the storage adapter's canonical shard layout (`storage.getNouns()` /
`getVerbs()` — the same primitive `repairIndex()`'s recount and every index-heal
walk use) instead of the metadata/graph indexes, then applies the selector as a
plain predicate over the walked records. This guarantees canon-completeness — index
corruption cannot hide a live record from the export — at the cost of an O(N) walk
regardless of selector selectivity. Relations are also walked canonically in this
mode, for every selector, not just the whole-brain case. Requires the LIVE current
generation: called on a historical `asOf()` view or a speculative `with()` overlay it
throws `CanonicalEnumerationUnavailableError` rather than silently mixing generations
or missing an overlay's own entities — `enumeration: 'index'` (the default) is
unaffected and still composes with `asOf()`/`with()` as before.
- **New: `export(selector, { enumeration: 'canonical', reportIndexDrift: true })`**
also runs the index-based enumeration and diffs it against canonical ground truth,
attaching `PortableGraph.drift: { canonicalOnly: string[], indexOnly: string[] }`
(canon-present ids the index missed; index-visible ids canon-absent — phantoms).
Migration-audit evidence, not a repair: nonzero drift is reported loudly
(`console.warn` with the counts) and nothing is auto-healed — run `brain.repairIndex()`
to reconcile the metadata index once drift is confirmed.
- **New: `export(selector, { includeHidden: true })`** (default: false — unchanged
behavior). Without it, a whole-brain/predicate export could never carry a
`visibility:'internal'` or `'system'` row, in EITHER `enumeration` mode — a real gap
for a bulk-migration fold auditing per-visibility-tier, where a hidden tier is real
user data, not noise to drop. `includeHidden` admits both tiers into candidacy in
both modes (and implies `includeSystem`; `includeSystem` alone keeps its narrower,
pre-existing meaning). **Migration-grade exports set `includeHidden: true`** — a
complete-canon export must carry every visibility tier; consumer-facing exports
leave it off.
- **Ops note (consumer-invisible): the release pipeline's home-registry publish (The
Source, source.soulcraft.com) now runs on CI**, triggered by the release tag, instead of PUTting the tarball from the laptop
over WAN — no change to what gets published or how a consumer installs it.
## v9.0.0 — 2026-08-04 (the field-addressing law: your names and system.*, nothing in between)
**Major.** One law now governs every field name, on every surface:
> **Data is either in main space — where you can use ANY name — or it is in
> `system.*`.**
Read `docs/concepts/field-addressing.md` (published on the docs site) for the
full contract; this entry is the migration ledger.
### Breaking — query surfaces (`where` / `orderBy` / `groupBy` / aggregation)
- **A bare field name ALWAYS addresses your metadata.** `orderBy: 'createdAt'`
no longer silently means the engine timestamp — it now refuses with a typed
`UnresolvableFieldError` naming both candidates unless you actually have a
user field of that name. Engine scalars are addressed explicitly:
`system.id`, `system.type`, `system.subtype`, `system.createdAt`,
`system.updatedAt`, `system.confidence`, `system.weight`,
`system.visibility`, `system.service`, `system.createdBy` (relations mirror
with `system.verb`/`system.sourceId`/`system.targetId`).
**Sweep list:** `where: { subtype: … }``where: { 'system.subtype': … }` ·
`orderBy: 'createdAt'``'system.createdAt'` · `groupBy: ['noun']`
`['system.type']` · any bare `visibility`/`service`/`confidence` filter that
meant the engine value → its `system.*` spelling. Every missed site fails
LOUDLY with the correction in the error message — nothing silently changes
meaning without telling you.
- **Unimplemented `find()` options refuse** (`cursor`, `includeRelations`,
`writeOnly``UnsupportedFindOptionError`); `order` is validated;
accepted-and-ignored is dead as a class.
- **The ordering contract is pinned cross-engine:** missing/null `orderBy`
values sort LAST in both directions, ties break by id ascending, and rows
are never dropped from an ordered read.
### Breaking — write surfaces
- **There are no reserved metadata names anymore.** `metadata: { confidence,
type, id, level, data, content, … }` are ordinary user fields — stored
verbatim, indexed, filterable, sortable, aggregatable, faithful across
restarts, index rebuilds, and `asOf()` time travel. The 8.x
reserved-key-in-bag throw is GONE; code that relied on it (or on the
`'warn'`/`'remap'` lift) must set engine scalars via their dedicated params
(`confidence`, `weight`, `subtype`, `visibility`, …) — the bag never touches
them now.
- **`reservedFieldPolicy` is removed.** Passing it throws at construction with
the migration note. `RESERVED_ENTITY_FIELDS`/`RESERVED_RELATION_FIELDS`
remain exported but now describe the stored record's engine half, not a ban
list; the `NoReservedEntityKeys`/`NoReservedRelationKeys` types are no-op
(deprecated).
- **The one refused spelling:** a metadata key literally starting `system.`
(namespace forgery) — typed error on `add`/`update`/`relate`/`updateRelation`.
- **Name-based index exclusions are gone.** Fields named `content`, `data`,
`id`, `vector`, … in your bag now INDEX like everything else (they were
silently un-indexed before — `where` on them returned `[]` with no error).
Value-shape rules stay, uniform across all names: arrays >10 never become
posting scalars; long values index hashed.
- **Migration transforms receive one normalized view** (engine fields
top-level, your bag nested under `metadata`) regardless of how old the
stored record is, and must return the same shape — a stray non-engine
top-level key refuses with the fix in the message.
### Storage format (automatic, no action)
- New/updated records persist as **nested-bag records** (engine fields
top-level, your bag verbatim under `metadata`, sealed by a format stamp) —
the shape that makes collider names lossless. Old flat records stay
readable forever; nothing rewrites your data in place.
- **Index epoch 3:** derived-index keys split the namespaces (bare user keys ·
literal `system.<field>` keys; the legacy `noun` column is gone). Every
brain rebuilds its derived indexes from canonical once, at first open —
observable via `getIndexStatus()`, no manual step. Pair this release with
the same-day native-accelerator release (its peer floor rises to `>=9`).
- Raw-record consumers (fact-log scanners, export tooling): read bags through
the exported shape-aware splitters (`splitNounMetadataRecord` /
`splitVerbMetadataRecord`) — they handle both record eras.
### Fixed in the same train
- Default visibility exclusion was a silent no-op under the new addressing on
pre-release builds (internal/system-tier rows could leak into default
reads) — now pinned by conformance tests at every lifecycle boundary.
- Per-type count surfaces (`getStats()`, count-by-type) read the new type
column, with a legacy fallback for pre-rebuild reads.
- Aggregation `source.where` evaluated dotted keys as nested paths — dotted
addresses now match per-key, and the internal per-type counts aggregate
rebuilds itself onto the new keys automatically.
### Conformance
Both engines ship a shared self-arming conformance suite (the law cases, the
ordering contract, and the reopen-collider fidelity case: every collider name
written as user data, verified verbatim through live reads, reopen, a forced
epoch rebuild, and time travel). Capability signal:
`FIELD_ADDRESSING_CAPABILITY = 'field-addressing/v1'` plus the typed error
classes, exported from the package root.
## v8.10.3 — 2026-08-03, 8.10-line backport (natural field names stop colliding with engine internals)
From a production report: sorting by a user metadata field named `level` silently
returned insertion order — the engine's internal HNSW node layer (also called
`level`) shadowed the user's field in every by-name read, and the indexing path
stamped a hardcoded `0` into the same index column (multi-valued poison). `level`
is a perfectly natural field name (game characters, priorities, floors); the
engine was wrong, not the caller.
- **`level` is user data now, everywhere.** Engine plumbing no longer resolves by
name, never shadows metadata, and never enters the indexed views. `orderBy:
'level'`, `where: { level: 9 }`, `groupBy: ['level']` all read YOUR field.
Regression pins: `tests/integration/level-field-shadow.test.ts` (the reporting
consumer's exact repro rows).
- **Index epoch 2.** The derived posting set changed, so every existing brain
rebuilds its metadata index from canonical at first open — poisoned columns
heal automatically; no manual step. First open after upgrade pays one rebuild
(observable via `getIndexStatus()`); pair this release with the same-day
native-accelerator release, which makes `level` indexable on the native path.
- **`transact()` metadata-only updates stop rewriting the vector record** — the
v8.10.2 write-granularity law now covers the batch/plan path too (it was
fixed for `update()` but the transact plan builder still staged the
unconditional save). If you batch stat touches through `transact()`, this is
your write-amplification fix.
- (The "coming next" note this entry carried shipped as v9.0.0 — the
field-addressing law above.)
---
## v8.10.2 — 2026-07-29 (metadata-only updates stop rewriting the vector record)
From a production incident on a large deployment: a read-heavy sweep that bumped
per-entity stats (metadata-only `update()` calls) saturated the disk — 5.8GB written
in 40 minutes — because every `update()` unconditionally re-persisted the WHOLE noun
record, unchanged vector included, fsynced.
- **`update()` write granularity fixed at the core.** A metadata-only update (no new
`data`, `vector`, or `type`) now writes the metadata leg and index deltas ONLY —
the vector-bearing noun record is never rewritten. Vector-side writes and HNSW
reindexing still happen exactly when the vector side actually changed. Regression
pins: `tests/integration/update-write-granularity.test.ts`.
- **Consumer guidance:** per-entity stat touches are now cheap, but batch them anyway
(one `transact()` instead of N `update()` calls) — granularity fixes the cost per
touch; batching fixes the count.
- Idle VFS `PathResolver` no longer logs `NaN% hit rate` once a minute (stats log
only on new traffic, at debug level).
- Native graph providers' `graph-lsm-*` storage keys are recognized as system
resources — the per-boot `Unknown key format` warning for them is gone.
Pairs with the native accelerator's same-day patch release; adopt as one bump.
---
## v8.10.1 — 2026-07-24 (the no-hot-retry contract + warm()'s metadata surface under native providers)
From a production incident: a native-provider op ground 38-40s inside a transaction,
blew the ~32s apply-phase budget, was rolled back (zero loss, by design), and a
downstream pipeline hot-retried the identical operation into a 6-minute, 100%-CPU
storm. Investigation confirmed Brainy itself never auto-retries a timed-out
transaction — the storm was entirely the consumer's own retry loop, driven by a
"retryable" doc-prose claim with no machine-readable contract to branch on. This
release closes that contract gap and, separately, fixes a real `warm()` reporting gap
surfaced by the same investigation.
- **`TransactionTimeoutError` is now a machine-readable no-hot-retry contract.** Two
new typed, always-`true` fields replace prose-only guidance:
- `retryable: true` — the operation MAY succeed on a later attempt, once the
underlying slowness resolves or the budget is deliberately raised
(`transactionBudgetFloorMs`, or a batch's own `timeoutMs` override).
- `hotRetryUnsafe: true` — an immediate, identical retry re-pays the FULL cost of
the work that just timed out (it does not resume partway) and can cascade into
exactly the CPU storm above. **Never loop on this error.** The documented pattern
is a latch, not a retry loop:
```
on TransactionTimeoutError:
record { at: Date.now(), error }
rethrow loudly to your own caller
hold a cooldown window before any re-attempt
clear the latch only on a subsequent success
```
- `context` (unchanged, now fully documented) carries the backoff inputs:
`timeoutMs`, `operationIndex`, `elapsedMs`, `totalOperations`, `operationName`.
- Every "retryable" doc-prose site referencing this error (`transact()`'s
`timeoutMs` option, `transactionBudgetFloorMs`, `Transaction.execute()`) now
points at these fields instead of bare prose.
- Regression-pinned: the engine never internally re-drives a timed-out operation
(verified via an execution counter through both the single-op write path and
`add()`'s upsert-race retry loop), so this has always been true — it is now
provable and typed.
- **Dead code removed**: `TransactionManager.executeTransactionWithResult()` had zero
callers in this codebase and is deleted.
- **`brain.warm()`'s metadata surface now routes through the ACTIVE provider.** A
production deployment's warm report showed `metadata: 'unavailable'` under a native
metadata provider — the previous logic only duck-typed the built-in JS manager's
`hydrateAll()` method, which a native provider has no reason to implement. The
metadata provider contract (`MetadataIndexProvider`, `src/plugin.ts`) gains an
optional `warm?(): Promise<void>` hook, mirroring the existing vector and graph
provider hooks. `brain.warm()` now checks the active provider's own `warm()` FIRST,
falls back to the JS manager's `hydrateAll()` when absent, and only reports
`'unavailable'` when neither exists — never `init()` as a stand-in, since a native
provider's `init()` may be a cheap verify rather than a real warm. A native
provider lights this surface up the same way `@soulcraft/cor` already lights the
vector and graph surfaces: implement `warm()` on its metadata provider.
- **New: `brain.maintenanceDebt()`** — the observability seam so an operator sees a
provider's outstanding background maintenance work (pending bytes/items, last pass
outcome, whether it's converging) BEFORE it grinds into the kind of budget-busting
op this release's timeout contract exists for, instead of discovering it as a CPU
storm. It is a pure passthrough: brainy applies no thresholds, no polling, and no
estimation — it calls each active provider's own optional `maintenanceDebt?()` hook
(vector, metadata, graph — the same three contracts `warm?()` lives on) and reports
the payload verbatim, or `'unavailable'` when a surface's provider doesn't track
debt. Useful as a pre-warm/post-warm check or a boot gate. `@soulcraft/cor` does not
yet implement the hook as of this release — expect it on cor's next release; until
then all three surfaces honestly report `'unavailable'`.
## Unreleased (the warm contract: cold-restart writes stop paying demand-load latency)
From a production deployment's cold-restart incident: the FIRST writes after every
restart on a large brain measured 3335s each (page cache cold) against the transact
apply budget — Brainy's own op-count-scaled budget, `max(30s, opCount × 2s)`, e.g.
32,000ms for a 16-op batch. There is no external deadline in this story, and no
post-completion veto either: every write is itself a multi-operation transaction (a
single `add()` applies several operations — canonical writes, the vector-index insert,
the metadata-index update), so ONE cold operation that legitimately runs ~33s consumes
the whole budget, the gate before the NEXT operation trips, and the write rolls back
atomically (zero loss, by design) — refused, retried, refused again, until the page
cache warms passively (~30 minutes). The cure is not weaker atomicity; it is the two
new knobs below — a budget floor sized for cold stores, and a warm contract that pays
demand-load cost OFF the transaction path.
- **The budget's start-gating contract is now explicit, documented, and pinned by
regression tests.** The budget gates STARTING the next operation — completed work is
never rolled back for elapsed time — and a transaction's first operation now
unconditionally starts by code, not merely because elapsed time happens to be ~0 when
it is checked. This has been the shipped schedule since 8.7.0 (no behavior change for
existing integrations); it is now stated in `Transaction.execute()`'s contract JSDoc
and enforced by tests so it cannot silently regress. Mid-batch atomicity is unchanged:
a trip before operation `i+1` still rolls back `0..i` and throws a retryable
`TransactionTimeoutError`.
- **The budget's 30s floor is now configurable**: `new Brainy({ transactionBudgetFloorMs })`
raises (or lowers) the floor of `max(transactionBudgetFloorMs, opCount × 2000)` for every
internal transact batch. Useful for a store whose cold writes legitimately run past 30s
per operation, so a bulk batch gets a proportionally larger runway instead of tripping
mid-batch on cold-cache latency.
- **New: `brain.warm()`** — eagerly loads/faults-in the vector index, metadata index, and
graph adjacency so the first real operation after a cold restart runs at steady-state
cost instead of paying demand-load latency on the critical path. Returns a `WarmReport`
with one honest outcome per surface — never conflate the first two:
- `'warmed'` — the surface's own provider `warm()` hook ran, or a full-hydration seam
loaded every shard/field/segment from storage. Steady-state cost is paid.
- `'probed'` — no `warm()` hook was available, so a best-effort read (one `search()` call
for the vector index) faulted in *some* backing storage as a side effect — real work,
but never reported as `'warmed'`.
- `'unavailable'` — nothing ran (no hook, no hydration seam, or nothing to probe).
- **New config: `warmOnOpen: true`** makes `init()` await `brain.warm()` before it resolves
— a deliberate blocking trade-off: startup takes longer, the first request doesn't.
Default `false` (unchanged lazy behavior).
- **New optional provider hook: `warm?(): Promise<void>`** on the vector and graph
acceleration provider contracts (`src/plugin.ts`) — a native provider can implement it to
eagerly pretouch its own backing storage (e.g. mmap pretouch); absence means brainy falls
back to the probe/hydration behavior above.
- **Transaction op-name strings changed in journals/timings**: the vector-index
transaction operations were renamed from `AddToHNSW`/`RemoveFromHNSW` to backend-neutral
`AddToVectorIndex(...)`/`RemoveFromVectorIndex(...)` — the old names hard-coded an
algorithm that may not be the one actually running (a non-HNSW native vector provider
emitting `"RemoveFromHNSW"` has sent an operator hunting an index that doesn't exist).
The parenthesized suffix names the ACTIVE backend: `hnsw-js` for the built-in engine, or
the native provider's own identity when it self-identifies. **If you parse these op-name
strings** (log processors, journal tooling), update your matcher from
`AddToHNSW`/`RemoveFromHNSW` to `AddToVectorIndex(`/`RemoveFromVectorIndex(`.
- **Provider identity is now a REQUIRED `name` field** on the vector provider contract
(`VectorIndexProvider.name`, `src/plugin.ts`) — every implementation self-reports its own
identity truthfully (its algorithm/engine), never inheriting a default. It renders as the
op-name suffix above and, wherever the vector index identifies itself in prose log lines,
as the tag `[vector-index:<name>]`. **Native provider adoption is a one-line change**:
declare `readonly name = '<your-provider-name>'`. A provider instance that still lacks
`name` at runtime (an older native build compiled against the previous, optional field) is
never crashed on and never silently mislabeled: it stamps `unknown-provider` and emits one
loud warning naming the missing field, so the gap is discoverable instead of a permanent
fossil label in every journal line.
## v8.9.0 — 2026-07-19 (flush is durability-only: history maintenance moves to close())
The write path stops paying maintenance costs — the last structural piece of the
flush-storm class (a production deployment measured single writes blocked 25191s behind
history reclaim running inline on flush under memory pressure):
- **`flush()` never compacts history.** It persists the current window's deltas and
nothing else — its cost no longer depends on history backlog or retention mode, in any
configuration. **`close()` is the auto-compaction site** (time-bounded per pass, ~5s;
an early stop is a consistent prefix and the next pass resumes).
- **`compactHistory()` gains `timeBudgetMs`** — bound your own maintenance windows; the
same resumable-prefix guarantee applies.
- **The documented trade**: a long-lived writer that never closes accumulates history
until its next explicit `compactHistory()`. Predictable writes, explicit maintenance.
If you run bounded retention on an always-on service, schedule a periodic
`compactHistory({ ...caps, timeBudgetMs })` in your maintenance window.
- **New public doc: `docs/performance-envelopes.md`** — measured per-op envelopes
(p50/p95 at stated scales, hardware, and backend, with the measuring script cited).
Refresh rule going forward: any release touching a measured path re-runs that op's
benchmark and updates the envelope in the same release.
- **New in this file: the Removed APIs 7.x→8.x table** (top of this document) — every
removal with its sanctioned replacement, one place, per the engine-currency contract.
Standing from here: removals only at majors, after ≥1 minor of loud runtime deprecation.
## v8.8.2 — 2026-07-19 (one field-resolution law: reserved-field aggregates stop drifting)
Four fixes from a consumer conformance audit, all rooted in the same disease — two field-resolution

View file

@ -1,36 +0,0 @@
# Security Policy
## Reporting a vulnerability
Email **security@soulcraft.com**. That's the one door for security reports
across the company, and it works the same way for Brainy: every report is
read by a human, you'll get a private receipt, and we'll work with you on
coordinated disclosure — please don't open a public issue for anything
that isn't already public.
Include what you'd want if you were on the other end: affected version,
how to reproduce, and what you think the impact is. If you have a patch or
a suggested fix, send it along — it's welcome but not required.
There is no bounty program today. We're saying that plainly so you know
what to expect going in.
## Response time
We respond as fast as truth allows. That means: no fixed SLA, no promise of
a reply within a specific number of hours — but a real report from a real
person gets read promptly and taken seriously. If you haven't heard anything
in a reasonable stretch, a follow-up email is completely fine.
## Supported versions
The latest `8.x` minor release line receives security fixes. If you're
running an older major version, please upgrade before reporting — we can't
commit to backporting fixes to unsupported lines.
## Scope
This policy covers the `@soulcraftlabs/brainy` package itself — the code in
this repository. If you're evaluating a deployment that also uses
`@soulcraft/cor`, report issues in that package the same way, to the same
address; we'll route internally.

View file

@ -3,7 +3,7 @@
/**
* Modern TypeScript CLI Runner
*
* This is the entry point after npm install @soulcraftlabs/brainy
* This is the entry point after npm install @soulcraft/brainy
* It runs the compiled TypeScript CLI code
*/

View file

@ -3,7 +3,7 @@
"configVersion": 0,
"workspaces": {
"": {
"name": "@soulcraftlabs/brainy",
"name": "@soulcraft/brainy",
"dependencies": {
"@aws-sdk/client-s3": "^3.540.0",
"@azure/identity": "^4.0.0",

View file

@ -25,13 +25,13 @@
### Prerequisites
```bash
npm install @soulcraftlabs/brainy
npm install @soulcraft/brainy
```
### Your First Neural Database
```typescript
import { Brainy, NounType } from '@soulcraftlabs/brainy'
import { Brainy, NounType } from '@soulcraft/brainy'
// Step 1: Create and initialize Brainy
const brain = new Brainy({
@ -143,7 +143,7 @@ Once you're comfortable with basic operations, move to **Level 2** to learn abou
### Building a Knowledge Graph
```typescript
import { Brainy, NounType, VerbType } from '@soulcraftlabs/brainy'
import { Brainy, NounType, VerbType } from '@soulcraft/brainy'
const brain = new Brainy({ storage: { type: 'memory' } })
await brain.init()
@ -314,7 +314,7 @@ Ready for AI-powered search and clustering? Move to **Level 3**.
### Triple Intelligence in Action
```typescript
import { Brainy, NounType, VerbType } from '@soulcraftlabs/brainy'
import { Brainy, NounType, VerbType } from '@soulcraft/brainy'
const brain = new Brainy({ storage: { type: 'memory' } })
await brain.init()
@ -529,7 +529,7 @@ Want to treat files as intelligent entities? Learn the **Virtual Filesystem** in
### Files as Intelligent Entities
```typescript
import { Brainy, NounType, VerbType } from '@soulcraftlabs/brainy'
import { Brainy, NounType, VerbType } from '@soulcraft/brainy'
const brain = new Brainy({ storage: { type: 'memory' } })
await brain.init()
@ -832,7 +832,7 @@ Ready for production deployment? Level 5 covers **planet-scale architecture**.
### Production-Ready Deployment
```typescript
import { Brainy, NounType } from '@soulcraftlabs/brainy'
import { Brainy, NounType } from '@soulcraft/brainy'
// 1. PRODUCTION STORAGE - Filesystem with off-site snapshots
console.log('Initializing production storage...\n')

View file

@ -1217,7 +1217,7 @@ where: {
await brain.find({ type: 'Document' })
// ✅ Correct: Use NounType enum
import { NounType } from '@soulcraftlabs/brainy'
import { NounType } from '@soulcraft/brainy'
await brain.find({ type: NounType.Document })
// ❌ Error: Operator not recognized

View file

@ -153,13 +153,13 @@ brainy-data/
### Step 1: Update Brainy Package
```bash
npm install @soulcraftlabs/brainy@latest
npm install @soulcraft/brainy@latest
```
**Check your version:**
```bash
npm list @soulcraftlabs/brainy
# Should show: @soulcraftlabs/brainy@4.0.0
npm list @soulcraft/brainy
# Should show: @soulcraft/brainy@4.0.0
```
### Step 2: No Code Changes Required! ✅
@ -374,7 +374,7 @@ If you encounter issues, you can rollback:
```bash
# Reinstall v3
npm install @soulcraftlabs/brainy@^3.50.0
npm install @soulcraft/brainy@^3.50.0
# Restart application
```
@ -389,7 +389,7 @@ rm -rf ./data
cp -r ./data-backup ./data
# Reinstall v3
npm install @soulcraftlabs/brainy@^3.50.0
npm install @soulcraft/brainy@^3.50.0
```
## Common Migration Scenarios
@ -539,7 +539,7 @@ console.log('Storage type:', status.type)
**Migration Checklist:**
- ✅ Backup data
- ✅ Update npm package (`npm install @soulcraftlabs/brainy@latest`)
- ✅ Update npm package (`npm install @soulcraft/brainy@latest`)
- ✅ Restart application (automatic migration)
- ✅ Verify data integrity
- ✅ Enable lifecycle policies

View file

@ -323,24 +323,58 @@ Only the graph adjacency index carries a committed scale assertion:
- ✅ **Single-Node by Design**: One process owns one `path`; scale out at the service layer
- ✅ **Zero Stubs**: Every line of code is production-ready
## Index Build at Open (10.4+)
## Lazy Loading Performance
As of 10.4, `brain.init()` runs every needed index rebuild to completion before
it returns — always, regardless of dataset size. There is no lazy,
first-query rebuild path: a brain either finishes opening healthy, or `init()`
fails loudly. `disableAutoRebuild` no longer defers index construction to a
first query; it has no effect on *when* a rebuild runs. Manual control over
rebuilds is `repairIndex({ rebuild: [...] })`. See
[Index Health](concepts/index-health.md) for the full read-gate contract
(providers self-report readiness via `healthReport()`; a read against a
not-serving provider throws a typed `*NotReadyError` rather than rebuilding
mid-query).
Brainy supports two initialization modes for optimal performance across different use cases:
<!-- The pre-10.4 "Mode 2: Lazy Loading on First Query" section previously
documented here (disableAutoRebuild deferring index construction to the
first find() call) described a real, now-retired code path. Removed
rather than left to mislead; the concept doc above is the current
contract. -->
### Mode 1: Auto-Rebuild (Default)
```javascript
const brain = new Brainy()
await brain.init() // Rebuilds indexes during init (~500ms-3s for 10K entities)
```
**Performance:**
- Init time: 500ms-3s (depends on dataset size)
- First query: Instant (indexes already loaded)
- Use case: Traditional applications, long-running servers
### Mode 2: Lazy Loading
```javascript
const brain = new Brainy({ disableAutoRebuild: true })
await brain.init() // Returns instantly (0-10ms)
const results = await brain.find({ limit: 10 }) // First query triggers rebuild (~50-200ms)
const more = await brain.find({ limit: 100 }) // Subsequent queries instant (0ms check)
```
**Performance:**
- Init time: 0-10ms (instant)
- First query: 50-200ms (includes index rebuild for 1K-10K entities)
- Subsequent queries: 0ms check (instant)
- Concurrent queries: Wait for same rebuild (mutex prevents duplicates)
**Concurrency Safety:**
```javascript
// 100 concurrent queries immediately after init
await brain.init()
const promises = Array.from({ length: 100 }, () =>
brain.find({ limit: 10 })
)
const results = await Promise.all(promises)
// ✅ Only 1 rebuild triggered (mutex)
// ✅ All 100 queries return correct results
// ✅ Total time: ~60ms (not 6000ms!)
```
**Use Cases for Lazy Loading:**
- **Serverless/Edge**: Minimize cold start time (0-10ms init)
- **Development**: Faster restarts during development
- **Large datasets**: Defer index loading until needed
- **Read-heavy workloads**: Writes don't wait for index rebuild
## Zero Configuration Required
@ -350,6 +384,10 @@ Brainy is designed to be **smart enough to tune itself dynamically**. No configu
// That's it. Brainy handles everything.
const brain = new Brainy()
await brain.init()
// Or with lazy loading for serverless
const brain = new Brainy({ disableAutoRebuild: true })
await brain.init() // Instant (0-10ms)
```
### Automatic Self-Tuning
@ -357,6 +395,7 @@ await brain.init()
- **Metadata Index**: Auto-builds sorted indices for range queries on first use
- **Graph Index**: Auto-flushes every 30 seconds
- **Default Tuning**: Research-based vector index defaults
- **Lazy Loading**: Indices built only when needed
- **Cache Management**: LRU caches with TTL
### Intelligent Defaults

View file

@ -10,7 +10,7 @@ next:
- guides/storage-adapters
---
# Plugin System
# Plugin Development Guide
Brainy has a plugin system that allows third-party packages to replace internal subsystems with custom implementations. This is how `@soulcraft/cor` provides optional native acceleration, and it's the same system available to any developer.
@ -46,7 +46,7 @@ If no plugin provides a given key, brainy uses its built-in JavaScript implement
### 1. Implement the `BrainyPlugin` interface
```typescript
import type { BrainyPlugin, BrainyPluginContext } from '@soulcraftlabs/brainy/plugin'
import type { BrainyPlugin, BrainyPluginContext } from '@soulcraft/brainy/plugin'
const myPlugin: BrainyPlugin = {
name: 'my-brainy-plugin', // Must be unique (typically your npm package name)
@ -90,7 +90,7 @@ await brain.init()
**Programmatic registration:** For plugins not installed as npm packages, use `brain.use()`:
```typescript
import { Brainy } from '@soulcraftlabs/brainy'
import { Brainy } from '@soulcraft/brainy'
import myPlugin from './my-plugin.js'
const brain = new Brainy()
@ -200,30 +200,15 @@ members so a warm reopen never pays a redundant rebuild-from-canonical:
- **`init?(): Promise<void>`** — eager cold-load. Brainy awaits it once during
`brain.init()`, after the metadata provider's `init()` (the id-mapper hydrates first)
and **before the rebuild gate**.
- **`healthReport?(): HealthReport`** — the PREFERRED signal (10.4+). A named,
synchronous, O(1) verdict derived from the provider's own exact ledgers — never a
sample, never I/O, must never throw for a well-formed provider. Brainy's read gate
(`assessProviderHealth()`) reads this INSTEAD of `isReady()` / size heuristics when
present: `serving: false` refuses the read with a typed `*NotReadyError` rather than
triggering a rebuild — a read never starts a store walk. `healthy` marks every
*verified* invariant holding; a family named in `unledgered` counts as neither
healthy nor broken. See `HealthReport` / `LedgerInvariantResult` /
`InvariantSource` in `src/plugin.ts`, and
[Index Health](concepts/index-health.md) for the consumer-facing story.
- **`isReady?(): boolean`** — honest durability signal, the fallback when
`healthReport()` is absent. `true` ⇔ the persisted index is loaded (or cheaply
demand-loadable) and consistent with what was last persisted. When exposed, the
gate defers to this signal **instead of** the `size() === 0` / `totalEntries === 0`
heuristics — a disk-native index may report 0 resident entries while fully durable.
Never return `true` if the durable state failed to load: the signal is honest in
both directions, and a not-ready provider gets its rebuild even when `size() > 0`.
- **`isReady?(): boolean`** — honest durability signal. `true` ⇔ the persisted index is
loaded (or cheaply demand-loadable) and consistent with what was last persisted. When
exposed, the rebuild gate defers to this signal **instead of** the `size() === 0` /
`totalEntries === 0` heuristics — a disk-native index may report 0 resident entries
while fully durable. Never return `true` if the durable state failed to load: the
signal is honest in both directions, and a not-ready provider gets its rebuild even
when `size() > 0`.
- **`isMigrating?(): boolean`** — while `true`, the provider owns its index (background
migration); brainy skips its rebuild entirely.
- **`validateInvariants?(): Promise<ProviderInvariantReport>`** — the async DEEP
diagnostic (full scans allowed), distinct from the bounded, sync `healthReport()`.
Must never throw — a failure is `healthy: false` data, not an exception; a provider
that throws anyway is read as a loud, unverified failure (never as "healthy") by
every caller, never silently retried into a rebuild.
Providers that implement none of these keep the size/count heuristics — correct for
engines whose `rebuild()` *is* their load path (like brainy's built-in JS vector index).
@ -272,10 +257,10 @@ When provided by an optional native acceleration plugin (such as `@soulcraft/cor
#### `cache`
**Type:** `UnifiedCache`
Replaces the global `UnifiedCache` singleton used for VFS path resolution, semantic caching, and vector index caching. Must implement the `UnifiedCache` interface (available from `@soulcraftlabs/brainy/internals`).
Replaces the global `UnifiedCache` singleton used for VFS path resolution, semantic caching, and vector index caching. Must implement the `UnifiedCache` interface (available from `@soulcraft/brainy/internals`).
```typescript
import type { UnifiedCache } from '@soulcraftlabs/brainy/internals'
import type { UnifiedCache } from '@soulcraft/brainy/internals'
context.registerProvider('cache', myNativeCache)
```
@ -325,8 +310,8 @@ Plugins can register custom storage backends that users reference by name.
### Implementing a Storage Adapter
```typescript
import type { StorageAdapterFactory } from '@soulcraftlabs/brainy/plugin'
import type { StorageAdapter } from '@soulcraftlabs/brainy'
import type { StorageAdapterFactory } from '@soulcraft/brainy/plugin'
import type { StorageAdapter } from '@soulcraft/brainy'
class MyStorageAdapter implements StorageAdapter {
async init(): Promise<void> { /* ... */ }
@ -360,9 +345,9 @@ Brainy provides three entry points for plugin developers:
| Import Path | Contents | Stability |
|-------------|----------|-----------|
| `@soulcraftlabs/brainy` | Public API, types, StorageAdapter | Stable (semver) |
| `@soulcraftlabs/brainy/plugin` | BrainyPlugin, BrainyPluginContext, StorageAdapterFactory | Stable (semver) |
| `@soulcraftlabs/brainy/internals` | UnifiedCache, EntityIdMapper, logger utilities | Internal (may change between minor versions) |
| `@soulcraft/brainy` | Public API, types, StorageAdapter | Stable (semver) |
| `@soulcraft/brainy/plugin` | BrainyPlugin, BrainyPluginContext, StorageAdapterFactory | Stable (semver) |
| `@soulcraft/brainy/internals` | UnifiedCache, EntityIdMapper, logger utilities | Internal (may change between minor versions) |
## Diagnostics
@ -440,7 +425,7 @@ A minimal but useful plugin that provides SIMD-accelerated distance calculations
```typescript
// simd-distance-plugin/src/plugin.ts
import type { BrainyPlugin, BrainyPluginContext } from '@soulcraftlabs/brainy/plugin'
import type { BrainyPlugin, BrainyPluginContext } from '@soulcraft/brainy/plugin'
// Hypothetical native module
import { simdCosineDistance } from './native.js'
@ -470,7 +455,7 @@ export default simdDistancePlugin
"main": "./dist/plugin.js",
"types": "./dist/plugin.d.ts",
"peerDependencies": {
"@soulcraftlabs/brainy": ">=7.0.0"
"@soulcraft/brainy": ">=7.0.0"
}
}
```
@ -478,7 +463,7 @@ export default simdDistancePlugin
Usage:
```typescript
import { Brainy } from '@soulcraftlabs/brainy'
import { Brainy } from '@soulcraft/brainy'
const brain = new Brainy({ plugins: ['brainy-simd-distance'] })
await brain.init()

View file

@ -54,7 +54,7 @@ After 40 API calls:
```typescript
// server.ts
import { Brainy } from '@soulcraftlabs/brainy'
import { Brainy } from '@soulcraft/brainy'
// SINGLETON INSTANCE
let brainInstance: Brainy | null = null
@ -174,7 +174,7 @@ process.on('SIGTERM', async () => {
```typescript
// server.ts - Clean Bun implementation
import { Brainy } from '@soulcraftlabs/brainy'
import { Brainy } from '@soulcraft/brainy'
let brain: Brainy | null = null

View file

@ -5,7 +5,7 @@
## Quick Start
```typescript
import { Brainy, NounType, VerbType } from '@soulcraftlabs/brainy'
import { Brainy, NounType, VerbType } from '@soulcraft/brainy'
const brain = new Brainy()
await brain.init()

View file

@ -99,7 +99,7 @@ Examples:
```bash
# 1. Deprecate wrong version on npm
npm deprecate @soulcraftlabs/brainy@X.X.X "Incorrect version - use Y.Y.Y"
npm deprecate @soulcraft/brainy@X.X.X "Incorrect version - use Y.Y.Y"
# 2. Fix version in package.json
# 3. Republish correct version

View file

@ -13,7 +13,7 @@
### In-Memory
```typescript
import Brainy from '@soulcraftlabs/brainy'
import Brainy from '@soulcraft/brainy'
const brain = new Brainy({ storage: { type: 'memory' } })
```
@ -43,7 +43,7 @@ The native vector provider (via the optional `@soulcraft/cor` package) extends t
Numbers below are **measured** by `tests/benchmarks/find-composition-scale.js` (a single
Node 22 process, in-memory storage, 384-dim vectors, `balanced` recall). They are the
open-core (pure-TypeScript) path — what you get from `@soulcraftlabs/brainy` with no native
open-core (pure-TypeScript) path — what you get from `@soulcraft/brainy` with no native
provider installed. Run it yourself: `node --max-old-space-size=8192 tests/benchmarks/find-composition-scale.js 100000`.
`find()` query latency, p50 / p95 (200 queries each):

File diff suppressed because it is too large Load diff

View file

@ -24,7 +24,7 @@ next:
## Quick Start
```typescript
import { Brainy, NounType, VerbType } from '@soulcraftlabs/brainy'
import { Brainy, NounType, VerbType } from '@soulcraft/brainy'
const brain = new Brainy() // Zero config!
await brain.init() // VFS auto-initialized!
@ -1010,7 +1010,7 @@ await db.release() // unpin + free cached materialization
### Db API errors
All exported from `@soulcraftlabs/brainy`:
All exported from `@soulcraft/brainy`:
| Error | Thrown by | Meaning |
|---|---|---|
@ -1451,34 +1451,6 @@ const count = await brain.getVerbCount()
---
### The canonical count ledger (`StorageAdapter.getCanonicalCounts()`)
An OPTIONAL method on the `StorageAdapter` interface (implemented by both
built-in adapters), not a method on `Brainy` itself — relevant if you're
writing a custom storage adapter or composing a provider's own
`healthReport()`. O(1), no I/O. Per family (`nouns`/`verbs`):
```typescript
interface CanonicalCounts {
nouns: { counted: number; all: number }
verbs: { counted: number; all: number }
suspect: boolean
}
```
- `counted` mirrors `getNounCount()` / `getVerbCount()` (public + internal tiers).
- `all` is the ALL-visibility scalar — every tier, including system/internal
records — the denominator a derived index's own coverage math is measured
against.
- `suspect` is `true` when an unprovable delete has left `all` unverified since
the last recount; `brain.repairIndex()` clears it with a real canonical walk.
Adapters without the ledger omit the method; treat absence as "no
denominator," never as zero. See
**[Index Health](../concepts/index-health.md)** for the full story.
---
### Subtype & facet APIs
Full guide: **[Subtypes & Facets](../guides/subtypes-and-facets.md)**.
@ -1859,104 +1831,6 @@ const semanticOnly = await brain.getStats({ excludeVFS: true })
---
### `repairIndex(options?)``Promise<RepairReport>`
The ceremony door for index repair. Bare `repairIndex()` is report-driven: it
prunes orphaned containers, recomputes count rollups, reconciles VFS
containment, and rebuilds only a derived-index family whose own health check
asks for it. Pass `options.rebuild` to force one or more families to rebuild
UNCONDITIONALLY — no health check is consulted — when an operator has
independent reason to reconcile a family regardless of what it self-reports.
```typescript
// Report-driven: only heals what actually needs it
const report = await brain.repairIndex()
console.log(report.healedTotal, report.families)
// Explicit: force the graph adjacency to rebuild from canonical, unconditionally
await brain.repairIndex({ rebuild: ['graph'] })
// Explicit: force all three derived indexes to rebuild
await brain.repairIndex({ rebuild: 'all' })
```
**`RepairReport`:**
- `families: RepairFamilyReport[]` — one row per family checked
- `healedTotal: number` — items healed across every family
- `durationMs: number`
**`RepairFamilyReport`** (one row):
- `family: string` — e.g. `'orphaned-containers'`, `'count-rollups'`,
`'vfs-containment'`, `'metadata-corruption'`, `'provider:metadata'`,
`'provider:graph'`, `'provider:vector'`
- `checked: boolean` — was this family actually examined (`false` ⇒ see `skipped`)
- `healed: number` — items re-posted/corrected in place (the incremental heal count)
- `missing?: { count: number; sample: string[] }` — exact count plus a capped id
sample when the check can name what diverged (never the full list)
- `rebuilt?: boolean` — a full generational rebuild ran (vs. an incremental heal)
- `detail?: string` / `reason?: string` — narration
- `skipped?: string` — why the family wasn't checked
Full walkthrough — what each family checks, degraded-but-serving vs. not-ready,
and what `suspect` counts mean — in
**[Index Health](../concepts/index-health.md)**.
---
### Index readiness: typed errors, `healthReport()`, `disableAutoRebuild`
Every derived-index provider (vector, graph, metadata) may expose a named,
synchronous, O(1) `healthReport()` composed from its own exact ledgers — the
signal Brainy's read gate trusts over sampling or size heuristics. `init()`
brings every provider to serving before it returns; there is no first-query
lazy-rebuild path. A read that reaches a provider whose health report says it
isn't serving throws instead of rebuilding mid-query:
| Error | Thrown by | Meaning |
|---|---|---|
| `GraphIndexNotReadyError` | `find({ connected })`, `neighbors()`, `related()` | Graph adjacency isn't serving |
| `MetadataIndexNotReadyError` | `find({ where })` | Metadata/field index isn't serving |
| `VectorIndexNotReadyError` | `find({ query })`, `similar()` | Vector index isn't serving |
All three are exported from `@soulcraftlabs/brainy`. Catch them to distinguish
"index not ready" from a genuine empty result:
```typescript
import { MetadataIndexNotReadyError } from '@soulcraftlabs/brainy'
try {
const rows = await brain.find({ where: { status: 'active' } })
} catch (err) {
if (err instanceof MetadataIndexNotReadyError) {
// reconcile: await brain.repairIndex(), then retry
} else {
throw err
}
}
```
**`disableAutoRebuild`** no longer defers index construction to the first
query. A needed rebuild always runs at `open()`, regardless of this flag or
dataset size; the flag has no effect on *when* a rebuild runs. Full manual
control lives in `repairIndex({ rebuild: [...] })`, above.
### `validateIndexConsistency()``Promise<...>`
The deep, async diagnostic counterpart to `healthReport()` — safe to run on a
live brain, but does more work (a provider's `validateInvariants()` may run a
full scan, not just read a ledger). Aggregates the JS metadata index's own
consistency check with every derived-index provider's invariant report.
```typescript
const validation = await brain.validateIndexConsistency()
if (!validation.healthy) {
console.log(validation.recommendation) // what to run, e.g. repairIndex()
console.log(validation.providers) // each provider's own invariant report, when exposed
}
```
---
## Lifecycle
### Initialization
@ -2208,7 +2082,7 @@ For the full taxonomy with all 169 types and their descriptions, see:
- **📖 Documentation:** [Full Documentation](../)
- **🐛 Issues:** [GitHub Issues](https://github.com/soulcraftlabs/brainy/issues)
- **💬 Discussions:** [GitHub Discussions](https://github.com/soulcraftlabs/brainy/discussions)
- **📦 NPM:** [@soulcraftlabs/brainy](https://www.npmjs.com/package/@soulcraftlabs/brainy)
- **📦 NPM:** [@soulcraft/brainy](https://www.npmjs.com/package/@soulcraft/brainy)
- **⭐ GitHub:** [Star us](https://github.com/soulcraftlabs/brainy)
---

View file

@ -268,7 +268,7 @@ locks/_flush_responses/ # writer answers with <uuid>.ack
| **Counts/statistics** | Per-type and per-subtype maps | `_system/{type,subtype,verb-subtype}-statistics.json.gz`, `counts.json` | Recomputable by scanning entities (`brainy inspect repair`) |
A pluggable index provider (the 8.0 plugin contract in
`@soulcraftlabs/brainy/plugin`) may replace any of the JS implementations; the
`@soulcraft/brainy/plugin`) may replace any of the JS implementations; the
persisted formats above are contract-bound so JS and native implementations
can interleave on the same directory.

View file

@ -126,7 +126,7 @@ class TypeAwareMetadataIndex {
**The Design**: Specify types clearly in your API calls:
```typescript
import { Brainy, NounType, VerbType } from '@soulcraftlabs/brainy'
import { Brainy, NounType, VerbType } from '@soulcraft/brainy'
// Add entity with explicit type
await brain.add({
@ -231,7 +231,7 @@ class OrgEnrichmentAugmentation {
**Brainy's Approach**: Extract **typed** concepts:
```typescript
import { NaturalLanguageProcessor } from '@soulcraftlabs/brainy'
import { NaturalLanguageProcessor } from '@soulcraft/brainy'
const nlp = new NaturalLanguageProcessor()
const concepts = await nlp.extractConcepts("Alice works at Google in San Francisco")
@ -382,7 +382,7 @@ import {
getVerbTypes,
BrainyTypes,
suggestType
} from '@soulcraftlabs/brainy'
} from '@soulcraft/brainy'
// Get all available noun types
const nounTypes = getNounTypes()

View file

@ -723,14 +723,6 @@ async stats(): Promise<Statistics> {
### 5. Index Rebuilding (Lazy Loading Support)
> **Stale as of 10.4 — "Mode 2: Lazy Loading on First Query" below is
> RETIRED.** `disableAutoRebuild` no longer defers index construction to a
> first query; `brain.init()` now runs every needed rebuild to completion
> before it returns, unconditionally, and a read against a not-serving
> provider throws a typed `*NotReadyError` instead of rebuilding mid-query.
> See `docs/concepts/index-health.md` for the current contract. Left below
> as historical background on the rebuild mechanics.
**Two modes of index loading:**
#### Mode 1: Auto-Rebuild on init() (default)

View file

@ -1,15 +1,5 @@
# Initialization and Rebuild Processes
> **Stale as of 10.4 — "Mode 2: Lazy Loading on First Query" below is RETIRED.**
> `disableAutoRebuild` no longer defers index construction to a first query;
> `brain.init()` now runs every needed rebuild to completion before it
> returns, unconditionally. A read against a not-serving provider throws a
> typed `*NotReadyError` instead of rebuilding mid-query. See
> `docs/concepts/index-health.md` for the current contract; this document's
> line-number references to `src/brainy.ts` also predate the file's current
> size and are unreliable. Left as historical background on the rebuild
> mechanics, not as a current API description.
This document explains how Brainy's four indexes (MetadataIndex, vector index, GraphAdjacencyIndex, DeletedItemsIndex) initialize and rebuild from persisted storage.
## Core Principle: All Indexes Are Disk-Based

View file

@ -127,7 +127,7 @@ For reference, a clean migration path:
`isMultiProcessSafe` type-guard. Keep `hasStorageMethod` for
build/install artifact protection.
5. Document the new contract in `concepts/storage-adapters.md`.
6. Major-version-bump the `@soulcraftlabs/brainy` peerDep range expected by
6. Major-version-bump the `@soulcraft/brainy` peerDep range expected by
plugins.
Estimated work: ~half a day of code, ~2 hours of doc/example updates,

View file

@ -20,7 +20,7 @@ next:
Every example on this page is written against the real Brainy 8.0 API. The setup is always the same:
```typescript
import { Brainy, NounType, VerbType } from '@soulcraftlabs/brainy'
import { Brainy, NounType, VerbType } from '@soulcraft/brainy'
const brain = new Brainy()
await brain.init()
@ -40,7 +40,7 @@ Brainy's **Noun-Verb Taxonomy** achieves broad coverage of human knowledge throu
- **Multi-hop Graph Traversals = Relationship Complexity**
- **Result: Model data across virtually any industry**
Every piece of information can be represented as entities (nouns) connected by relationships (verbs) carrying properties (metadata). The standardized type system from `@soulcraftlabs/brainy` (`NounType`, `VerbType`) gives those nouns and verbs a stable, shared name.
Every piece of information can be represented as entities (nouns) connected by relationships (verbs) carrying properties (metadata). The standardized type system from `@soulcraft/brainy` (`NounType`, `VerbType`) gives those nouns and verbs a stable, shared name.
## The Power of Standardization: Universal Interoperability

View file

@ -35,7 +35,7 @@ constructor and `init()`.
## Instant Start
```typescript
import { Brainy } from '@soulcraftlabs/brainy'
import { Brainy } from '@soulcraft/brainy'
// That's it. No config needed.
const brain = new Brainy()

View file

@ -1,233 +0,0 @@
---
title: Field addressing: your fields and system fields
slug: concepts/field-addressing
public: true
category: concepts
template: concept
order: 7
description: The one rule for every query-surface field name — a bare name always means your metadata, system.<field> reaches the ten engine scalars explicitly, and anything else refuses by name.
next:
- guides/namespace-migration
- concepts/consistency-model
---
# Field addressing: your fields and system fields
Every query surface in Brainy — `find()`'s `where`, `orderBy`, aggregation
`groupBy`, and aggregation `source.where` — resolves field names by one rule,
with no exceptions:
> **A bare field name always means your metadata. `system.<field>` reaches an
> engine scalar, and only when you spell it explicitly.**
```typescript
await brain.find({ orderBy: 'level' }) // reads entity.metadata.level — YOUR field
await brain.find({ orderBy: 'system.createdAt' }) // reads the engine's createdAt scalar
await brain.find({ orderBy: 'metadata.level' }) // identical to bare 'level' — explicit scope
```
There is no priority list, no "try the system field, fall back to metadata"
behavior, and no name that resolves differently depending on what else
happens to exist on your entities. A field called `level`, `score`,
`createdAt`, or `type` in your own `metadata` is read as *your* field, every
time, by its bare name.
## Why this rule exists
An internal report from a production deployment found that a user metadata
field literally named `level` was being silently shadowed by the engine's
own internal index layer field of the same name — every sort by `level`
returned insertion order, with no error raised. This rule makes that class of
bug structurally impossible: bare names belong to you, unconditionally, and
anything that isn't yours has to be spelled out.
## The system scalars
`system.<field>` addresses exactly ten scalars on an entity — no more, no
fewer:
| System field | What it is |
|---|---|
| `system.id` | The entity's id |
| `system.type` | The entity's `NounType` |
| `system.subtype` | The per-app sub-classification passed to `add()` |
| `system.createdAt` | When the entity was created |
| `system.updatedAt` | When the entity was last written |
| `system.confidence` | The `confidence` param (01) |
| `system.weight` | The `weight` param |
| `system.visibility` | `'public'` / `'internal'` (see the visibility tiers in [Consistency Model](./consistency-model.md)) |
| `system.service` | The multi-tenancy `service` tag |
| `system.createdBy` | Who/what created the entity |
Relationships mirror the same eight shared scalars (`subtype`, `createdAt`,
`updatedAt`, `confidence`, `weight`, `visibility`, `service`, `createdBy`)
plus three of their own:
| System field (relationship) | What it is |
|---|---|
| `system.verb` | The relationship's `VerbType` |
| `system.sourceId` | The id of the entity the relationship starts from |
| `system.targetId` | The id of the entity the relationship points to |
Anything not on these two lists is not a system scalar — `system.<name>` for
any other name refuses (see "Refusal semantics" below), even if that name
sounds like it should be engine-owned.
## Invisible plumbing — never addressable, in either spelling
Five names are pure engine internals. They are not reachable as a bare name,
and not reachable as `system.<name>` either — they simply have no place on
the query surface:
- **`vector`** — the stored embedding. It participates in similarity search
(`query`, `near`, vector `find()`), never in `where`/`orderBy`/`groupBy`.
- **`connections`** — graph adjacency. Reached through `connected` and
`brain.related()`, not through field addressing.
- **`level`** — the internal index layer number used by the nearest-neighbor
graph. It is pure index plumbing with no query-surface meaning at all —
which is exactly why a user field of the same name must never be shadowed
by it. `level` as a bare name is always yours; there is no engine-owned
spelling of it to compete with.
- **`data`** — your entity's content payload, not a scalar. It can be a
string, a number, or an arbitrary object, so sorting or filtering it as a
single comparable value would lie about its actual shape. Content is
reached through the content/text-search APIs (`query`, `searchMode:
'text'`), not through `where`/`orderBy`.
- **`_rev`** — the per-entity revision counter used for optimistic
concurrency (`ifRev`). It is a CAS token, not a queryable dimension.
`system.level`, `system.vector`, and `system.data` all refuse for the same
reason: they are not in the ten-scalar system map, full stop.
## `metadata.<field>` — the explicit spelling of "mine"
Prefix any field with `metadata.` to say the same thing a bare name already
says, spelled out. The two are interchangeable everywhere a field name is
accepted, including `orderBy`:
```typescript
await brain.find({ where: { 'customer.tier': 'gold' } })
await brain.find({ where: { 'metadata.customer.tier': 'gold' } }) // identical
await brain.find({ orderBy: 'metadata.score', order: 'desc' }) // identical to orderBy: 'score'
```
Reach for the explicit spelling when it reads more clearly next to a
`system.` field in the same query — for example, sorting by your own `score`
while filtering on `system.confidence`.
## No special names — the write side
The same law governs writes:
> **Data is either in main space, where developers can use anything, or it
> is in `system.*`.**
There are **no reserved metadata names**. A field called `confidence`,
`type`, `id`, `data`, `content`, or anything else inside your `metadata` bag
is an ordinary user field: it is stored verbatim, indexed, filterable,
sortable, aggregatable, and it survives restarts, index rebuilds, and
time-travel (`asOf`) reads exactly as written — even when an engine scalar
shares its spelling. The engine's values are written only through their
dedicated params (`confidence`, `weight`, `subtype`, `visibility`, …) and
read at `system.<field>`; your bag can never touch them and they can never
shadow your bag.
```typescript
const id = await brain.add({
data: 'Ada Lovelace',
type: NounType.Person,
confidence: 0.9, // the ENGINE scalar
metadata: { confidence: 'self-rated' } // YOUR field, same spelling — both live
})
await brain.find({ where: { confidence: 'self-rated' } }) // finds it (yours)
await brain.find({ where: { 'system.confidence': 0.9 } }) // finds it (engine's)
```
The one spelling a write refuses is a metadata key that literally starts
with `system.` — the explicit address namespace cannot be forged as a user
field name. That refusal is typed and names the fix.
Value **shape** rules still apply uniformly to every name (they are not name
carve-outs): arrays longer than 10 elements are not turned into posting-list
scalars, and very long values are indexed by hash.
## Refusal semantics
A name that resolves to neither your metadata nor a system scalar is a typed
refusal, not a silent empty result and not a guess. Refusals name **both**
candidates, so the fix is always in the error text:
```typescript
await brain.find({ orderBy: 'createdAt' })
// UnresolvableFieldError: no metadata field 'createdAt' — did you mean
// system.createdAt or metadata.createdAt?
```
`UnresolvableFieldError` is exported from the package root:
```typescript
import { UnresolvableFieldError } from '@soulcraftlabs/brainy'
try {
await brain.find({ orderBy: 'createdAt' })
} catch (err) {
if (err instanceof UnresolvableFieldError) {
// err.message names both candidates — usually enough to fix the call site.
}
}
```
A handful of `find()` options are not implemented yet: `cursor`,
`includeRelations`, and `writeOnly`. Rather than accepting them and quietly
ignoring the option, `find()` refuses with `UnsupportedFindOptionError`
also exported from the package root — so a call site can never believe an
unimplemented option took effect when it didn't.
## The ordering contract
`orderBy` behaves identically regardless of which engine (the pure-TypeScript
path or a native accelerator) is serving the query:
- An entity missing the `orderBy` field, or holding `null` on it, sorts
**LAST — in both `asc` and `desc`**. It is never treated as "smaller than
everything" in one direction and "larger than everything" in the other; it
is simply last, either way.
- Rows are **never dropped** from an ordered read because they lack the
field — a missing value changes position, never presence.
- Ties on the `orderBy` field break by **id ascending**, regardless of the
primary sort direction.
```typescript
// employees: [{ score: 9 }, { score: 5 }, { /* no score field */ }]
await brain.find({ orderBy: 'score', order: 'desc' }) // [9, 5, missing] — missing is last
await brain.find({ orderBy: 'score', order: 'asc' }) // [5, 9, missing] — missing is STILL last
```
## Migrating existing call sites
If you have call sites written before this rule shipped that rely on a bare
system name — `orderBy: 'createdAt'`, `where: { confidence: { greaterThan:
0.8 } }`, and similar — they now refuse instead of silently resolving to the
engine field. The fix is always in the error: swap the bare name for
`system.<field>` (or `metadata.<field>` if you actually meant your own field
of that name, and it happens to share a name with a system scalar):
```typescript
// Before: bare 'createdAt' silently meant the engine's timestamp.
await brain.find({ orderBy: 'createdAt' })
// After: say which one you meant.
await brain.find({ orderBy: 'system.createdAt' }) // the engine timestamp
await brain.find({ orderBy: 'metadata.createdAt' }) // your own field named createdAt, if you have one
```
There is no silent migration path by design — every ambiguous call site
surfaces as a refusal naming its own fix, once, the first time it runs
against the new rule.
## Where to go next
- [Consistency Model](./consistency-model.md) — visibility tiers, revision
counters, and the rest of the read/write contract this page's
read-time addressing rule.

View file

@ -1,217 +0,0 @@
---
title: Index Health
slug: concepts/index-health
public: true
category: concepts
template: concept
order: 8
description: How Brainy knows whether a derived index can be trusted — exact accounting instead of sampling, the named health report, degraded-but-serving vs. not-ready, and what repairIndex() checks, heals, and rebuilds.
next:
- concepts/generation-fact-log
- guides/inspection
---
# Index Health
Brainy keeps one **canonical** copy of every entity and relationship, and three
**derived** indexes built from it — vector, metadata, and graph — so `find()` can
answer semantically, by filter, and by traversal without re-deriving the answer from
scratch on every query. A derived index is a cache with a serving structure: it can
be present but stale, present but only partially loaded, or fully out of sync with
canonical after a crash. This page is about how Brainy decides whether to trust one,
what it does when it can't, and how you reconcile the two.
## Exact accounting instead of sampling
Older health checks worked by inference: does `size()` return something greater
than zero, does a spot-check on one known item come back correct. Both are proxies.
A cold index can report a nonzero count while its actual serving structure never
loaded, and a spot-check only proves the one item it happened to ask about.
Every derived-index provider may now expose a named, synchronous, O(1)
`healthReport()` — composed from the provider's own **exact ledgers** (real counters
it already maintains on the write path), never a sample or a walk. This is the one
signal Brainy's read gate consults. A provider that doesn't yet expose one falls
back to an honest `isReady()` boolean, and finally to a size heuristic for engines
with neither — but wherever a `healthReport()` exists, it wins.
Underneath, storage itself keeps an analogous **canonical count ledger**: a
`counted` scalar (the user-facing total — what `getNounCount()` / `getVerbCount()`
return) and an `all` scalar (every tier, including internal records a derived
index's own coverage math needs to compare against). This is the real denominator
a provider's `healthReport()` measures itself by, rather than a total that can only
ever ratchet upward. See [What `suspect` counts mean](#what-suspect-counts-mean)
below for the one case that ledger can't stay exact through on its own.
## The named report
A `HealthReport` carries, per provider (`'vector'` / `'graph'` / `'metadata'`):
- **`healthy`** — `true` iff every *verified* invariant holds. An invariant whose
family has no ledger yet is `unledgered`, never counted either way — unknown,
not passing.
- **`serving`** — can this provider answer a query right now. A failing invariant
graded `heal: 'repair'` or `heal: 'none'` still leaves `serving: true` — this is
**degraded-but-serving**: something is off (say, a stale rollup on an
`employee` record's relationship count) but reads keep working. Only a failure
graded `heal: 'rebuild'` flips `serving` to `false`**not-ready** — because the
provider itself is telling you its serving structure cannot answer correctly.
- **`invariants`** — each checked condition, with its provenance
(`source: 'ledger'` — an exact count; `'deep'` — a full scan, diagnostic-only;
`'unledgered'` — not yet tracked) and, for a failing one, an exact `missing`
count plus a capped sample of the affected ids — a verdict, never a dump.
- **`generation`** — bumps on every ledger mutation and rebuild, so a caller can
cache a verdict per generation instead of re-deriving it.
The distinction that matters day to day: `healthy: false` can be entirely benign —
a maintenance window, a divergence `repairIndex()` will clean up on its own
schedule. `serving: false` is not benign. It means this provider is refusing to
answer, on its own word, right now.
**How a failure gets its grade — the serving law.** A provider grades `heal` by
one question only: *could an answer be wrong?* — never *how expensive is the
fix?* A missing-postings shortfall, however large, is `heal: 'repair'` (re-post
exactly what the ledger names, reads serving throughout); it can never withhold
serving just because healing it takes work. `serving` is withheld only by a
small, named set of rebuild-graded conditions — the index not initialized, its
durable state absent, a manifest naming files that are not resident, a replay
that did not complete cleanly — the states in which an answer could genuinely be
wrong. And a read is only ever refused by the family it actually consults: a
metadata filter is answered by the metadata index alone, vector search by the
vector index, traversal by the graph index — one family's refusal never blocks
another family's reads.
## Reads refuse — they never rebuild
A query that reaches a not-serving provider does not trigger a rebuild from inside
the read. Brainy retired that path deliberately: a rebuild kicked off by an ordinary
`find({ where: { status: 'active' } })` call is a dark, unpredictable cost hiding
behind a request that looks like a cheap read. Instead, the read throws a typed,
catchable error naming the reason:
| Error | Thrown when | Meaning |
|---|---|---|
| `GraphIndexNotReadyError` | `find({ connected })`, `neighbors()`, `related()` | The graph adjacency index isn't serving — traversal would otherwise return `[]` indistinguishable from "no relationships" |
| `MetadataIndexNotReadyError` | `find({ where })` | The metadata/field index isn't serving — a filtered read would otherwise return `[]` indistinguishable from "no matches" |
| `VectorIndexNotReadyError` | `find({ query })`, `similar()` | The vector index isn't serving — a semantic search would otherwise return `[]` indistinguishable from "nothing similar" |
All three are exported from `@soulcraftlabs/brainy`. Catch them where your application
needs to distinguish "this index isn't ready yet" from "there's genuinely nothing
here" — a health dashboard, a retry policy, an operator alert. The fix is always
the same: reconcile the index, either by reopening the brain (which brings every
provider to serving before `init()` returns — see the next section) or by calling
`repairIndex()` explicitly.
```typescript
try {
const active = await brain.find({ where: { status: 'active' } })
} catch (err) {
if (err instanceof MetadataIndexNotReadyError) {
// not a "no results" — the index itself refused; alert or retry after repair
} else {
throw err
}
}
```
### Rebuilds happen at open, not on first query
`brain.init()` runs every needed rebuild to completion **before it returns**,
unconditionally, regardless of dataset size. There is no lazy, first-query
rebuild path anymore — a brain either finishes opening healthy, or it fails
open loudly. `disableAutoRebuild: true` no longer defers index construction to
the first query: it has no effect on *when* a needed rebuild runs. Full manual
control over rebuilds is `repairIndex({ rebuild: [...] })` (below), not this flag.
## `repairIndex()` — checking and healing
Bare `repairIndex()` is **report-driven**: it only heals what its own checks say
actually needs it, and it always returns a full per-family receipt.
```typescript
const report = await brain.repairIndex()
report.healedTotal // total items healed across every family
report.durationMs
report.families // one row per family checked
```
Each `RepairFamilyReport` row names what happened:
- **`checked`** — was this family actually examined (`false` means skipped —
see `skipped` for why).
- **`healed`** — items re-posted or corrected in place.
- **`missing`** — when the check can name what diverged: an exact `count` plus a
capped `sample` of ids.
- **`rebuilt`** — a full generational rebuild ran (as opposed to an incremental
heal).
- **`detail`** / **`reason`** / **`skipped`** — the receipt's narration; a row is
always either checked or explains why it wasn't. Nothing is silent.
On every call, bare `repairIndex()`:
1. Prunes orphaned canonical containers left by a partial delete.
2. Recomputes the count rollups from one canonical walk (unconditional — this is
also what clears a `suspect` ledger; see below).
3. Reconciles VFS containment edges, if the VFS is initialized.
4. Runs the metadata index's own corruption detection pass.
5. Consults each of the three derived-index providers' own health check and
rebuilds only a family whose failing invariant actually asks for it
(`heal: 'rebuild'`) — never a provider that reports `healthy` or a lesser
grade.
### The explicit rebuild door
`options.rebuild` skips the health check and rebuilds one or more families
**unconditionally** — the operator override for when you have independent reason
to distrust a family regardless of what it self-reports (a suspicious deploy, a
storage-layer incident, a support ticket that doesn't match what the health report
says):
```typescript
// Force the graph adjacency to rebuild from canonical, no invariant consulted
await brain.repairIndex({ rebuild: ['graph'] })
// Force all three derived indexes
await brain.repairIndex({ rebuild: 'all' })
```
A family named this way is recorded with `rebuilt: true` and
`reason: 'explicit rebuild requested'`, and is skipped by the normal
health-driven pass in the same call — it was already rebuilt unconditionally.
Reach for the explicit door when you need certainty regardless of self-report;
reach for bare `repairIndex()` for routine maintenance and after any incident
where you're not sure which family (if any) needs it.
## What `suspect` counts mean
Storage's canonical count ledger increments the ALL-visibility total on every new
record and decrements it on every *proven* delete — one where the record was read,
or the caller supplied its prior image. A delete that cannot prove what it removed
existed doesn't guess: it flags the ledger `suspect` (an operator-visible
`console.warn`, narrated once per session, not once per delete) rather than risk
decrementing a total that was never incremented for that record in the first
place. This is intentionally rare — it's a defensive fallback for callers on an
unusual removal path, not a per-delete cost.
`suspect` is not directly exposed on any `Brainy` method today — it lives on the
`StorageAdapter`'s optional `getCanonicalCounts()`, primarily consulted by
`repairIndex()`'s recount step and by custom storage adapters composing their own
`healthReport()`. What matters for an application: a `suspect` ledger is not
incorrect, just *unverified since the last recount* — and `repairIndex()`'s
unconditional count-rollup step (step 2, above) recomputes the ALL scalars from a
real canonical walk on every call, clearing the flag with proof either way.
## Practical guidance
- **On a normal restart**, do nothing — `init()` brings every provider to
serving before it returns, or fails loudly.
- **On a `*NotReadyError`** from a live read, reconcile with `repairIndex()`
(report-driven is almost always sufficient) and retry.
- **After an incident** where you distrust a specific family regardless of what
it reports healthy — a storage-layer fault, a suspicious restore — use the
explicit door: `repairIndex({ rebuild: ['metadata' | 'graph' | 'vector'] })`.
- **To audit before trusting a report**, `brain.auditGraph()` walks every stored
relationship and proves (or disproves) that reads return canonical truth,
independent of what any provider self-reports — see
[Inspecting a Live Brainy](../guides/inspection.md).

View file

@ -61,7 +61,7 @@ The only required override is the capability flag. Returning `true` from
to call `acquireWriterLock()` at init.
```typescript
import { FileSystemStorage } from '@soulcraftlabs/brainy'
import { FileSystemStorage } from '@soulcraft/brainy'
export class MmapFileSystemStorage extends FileSystemStorage {
public supportsMultiProcessLocking(): boolean {
@ -79,7 +79,7 @@ If your storage is **not filesystem-backed** (a custom
network backend), extend `BaseStorage` directly:
```typescript
import { BaseStorage } from '@soulcraftlabs/brainy'
import { BaseStorage } from '@soulcraft/brainy'
export class MyCloudStorage extends BaseStorage {
// BaseStorage's default no-op implementations of the multi-process
@ -101,7 +101,7 @@ The defensive check at every new-storage-method call site (`brainy.ts`,
`hasStorageMethod(name)`) does **not** exist to handle "plugin bundles a
stale BaseStorage." Plugins ship a dist that preserves the dynamic ESM
import (verify in your plugin's `dist/`: `import { FileSystemStorage } from
'@soulcraftlabs/brainy'` is not rewritten to a vendored copy). The prototype
'@soulcraft/brainy'` is not rewritten to a vendored copy). The prototype
chain at runtime resolves to whatever Brainy version your consumer has
installed.
@ -109,8 +109,8 @@ installed.
the prototype chain at the consumer-app level:
- **Stale `node_modules`** — a lingering install from before the consumer
upgraded Brainy. The package.json says `@soulcraftlabs/brainy@7.22.0` but
`node_modules/@soulcraftlabs/brainy` is still 7.20.x.
upgraded Brainy. The package.json says `@soulcraft/brainy@7.22.0` but
`node_modules/@soulcraft/brainy` is still 7.20.x.
- **Lockfile drift**`bun.lockb` / `package-lock.json` pins a brainy
version older than the package.json range, and `bun install` honors the
lockfile.
@ -131,7 +131,7 @@ and the warning names the adapter class plus a remediation hint:
methods on its prototype chain. Writer locking and the flush-request RPC are
disabled for this directory. Likely fix: clean install (`rm -rf node_modules
bun.lockb && bun install`) or rebuild your container image to refresh
`@soulcraftlabs/brainy` to ≥7.21. See docs/concepts/storage-adapters.md.
`@soulcraft/brainy` to ≥7.21. See docs/concepts/storage-adapters.md.
```
## Authoring a new storage adapter — minimum checklist
@ -168,7 +168,7 @@ bun.lockb && bun install`) or rebuild your container image to refresh
install time — fix install, not your plugin.
6. **Pin your peer dep generously.** `"peerDependencies": {
"@soulcraftlabs/brainy": "^7.21.0" }` accepts any compatible 7.x. Don't pin
"@soulcraft/brainy": "^7.21.0" }` accepts any compatible 7.x. Don't pin
to an exact patch unless you're tracking a known regression.
## Future direction
@ -185,5 +185,5 @@ follow-up; consumers don't need to anticipate the change.
heartbeat semantics, what the lock protects.
- [`guides/inspection`](../guides/inspection.md) — `brainy inspect` and the
read-only mode.
- `node_modules/@soulcraftlabs/brainy/dist/storage/baseStorage.d.ts` — the
- `node_modules/@soulcraft/brainy/dist/storage/baseStorage.d.ts` — the
authoritative type signatures for every method this page references.

View file

@ -22,7 +22,7 @@ they share a single scan.
## Quick Start
```typescript
import { Brainy, NounType } from '@soulcraftlabs/brainy'
import { Brainy, NounType } from '@soulcraft/brainy'
const brain = new Brainy()
await brain.init()

View file

@ -8,7 +8,7 @@ Brainy is **framework-friendly** - designed to drop into the server side of any
Brainy embeds an HNSW vector index, a graph engine, and a filesystem-backed persistence layer. These belong on the server:
- **Zero configuration**: Just `import { Brainy } from '@soulcraftlabs/brainy'`
- **Zero configuration**: Just `import { Brainy } from '@soulcraft/brainy'`
- **Auto storage detection**: `new Brainy()` auto-selects filesystem persistence on Node
- **Cleaner code**: No browser polyfills, no conditional client/server imports
- **Better DX**: One instance shared across your server routes
@ -18,13 +18,13 @@ Brainy embeds an HNSW vector index, a graph engine, and a filesystem-backed pers
### Install Brainy
```bash
npm install @soulcraftlabs/brainy
npm install @soulcraft/brainy
```
### Basic Integration
```javascript
import { Brainy } from '@soulcraftlabs/brainy'
import { Brainy } from '@soulcraft/brainy'
// Run on the server (API route, server component, backend service)
// new Brainy() auto-detects filesystem persistence on Node
@ -105,7 +105,7 @@ On the server, create one Brainy instance and reuse it across requests. This mod
```javascript
// lib/brain.server.js
import { Brainy } from '@soulcraftlabs/brainy'
import { Brainy } from '@soulcraft/brainy'
let brainPromise
@ -163,7 +163,7 @@ On the server, create one Brainy instance and reuse it across requests:
```javascript
// server/brain.js (server-only module)
import { Brainy } from '@soulcraftlabs/brainy'
import { Brainy } from '@soulcraft/brainy'
let brainPromise
@ -248,7 +248,7 @@ The matching backend endpoint uses Brainy directly (Node/Bun):
```typescript
// server: api/search
import { Brainy } from '@soulcraftlabs/brainy'
import { Brainy } from '@soulcraft/brainy'
const brain = new Brainy() // auto-detects filesystem persistence on Node
await brain.init()
@ -266,7 +266,7 @@ In Next.js, Brainy lives in server code only: API routes, server components, or
```javascript
// lib/brain.server.js (imported only by server code)
import { Brainy } from '@soulcraftlabs/brainy'
import { Brainy } from '@soulcraft/brainy'
let brainPromise
@ -318,7 +318,7 @@ Brainy runs in a server-only module (`*.server.js`); the component fetches resul
```javascript
// src/lib/server/brain.js (server-only — note the .server suffix)
import { Brainy } from '@soulcraftlabs/brainy'
import { Brainy } from '@soulcraft/brainy'
let brainPromise
@ -432,7 +432,7 @@ import { defineConfig } from 'vite'
export default defineConfig({
ssr: {
external: ['@soulcraftlabs/brainy']
external: ['@soulcraft/brainy']
}
})
```
@ -440,7 +440,7 @@ export default defineConfig({
```javascript
// rollup.config.js (server bundle)
export default {
external: ['@soulcraftlabs/brainy', 'node:fs', 'node:path', 'node:crypto']
external: ['@soulcraft/brainy', 'node:fs', 'node:path', 'node:crypto']
}
```
@ -466,7 +466,7 @@ export async function load({ url }) {
```javascript
// For build-time usage (runs in Node during the build)
import { Brainy } from '@soulcraftlabs/brainy'
import { Brainy } from '@soulcraft/brainy'
export async function generateStaticProps() {
const brain = new Brainy({
@ -513,7 +513,7 @@ export async function generateStaticProps() {
### Issue: Large client bundle size
**Cause**: A client module is pulling in Brainy.
**Solution**: Move the `import { Brainy } from '@soulcraftlabs/brainy'` into a server-only module so it never reaches the browser bundle.
**Solution**: Move the `import { Brainy } from '@soulcraft/brainy'` into a server-only module so it never reaches the browser bundle.
### Issue: SSR hydration mismatch
**Solution**: Run the search on the server (loader / server action / API route) and pass the results down as props, so server and client render the same markup.

View file

@ -9,7 +9,7 @@ Brainy's import is **ONE magical method** that understands EVERYTHING:
## The Ultimate Simplicity
```javascript
import { Brainy } from '@soulcraftlabs/brainy'
import { Brainy } from '@soulcraft/brainy'
const brain = new Brainy()
await brain.init()

View file

@ -13,7 +13,7 @@ Brainy provides real-time progress tracking for **all 7 supported file formats**
### Basic Progress Tracking
```typescript
import { Brainy } from '@soulcraftlabs/brainy'
import { Brainy } from '@soulcraft/brainy'
import * as fs from 'fs'
const brain = await Brainy.create()

View file

@ -7,7 +7,7 @@
## Basic Import
```typescript
import { Brainy } from '@soulcraftlabs/brainy'
import { Brainy } from '@soulcraft/brainy'
const brain = new Brainy()
await brain.init()
@ -187,7 +187,7 @@ await brain.import(file, {
## Complete Example
```typescript
import { Brainy } from '@soulcraftlabs/brainy'
import { Brainy } from '@soulcraft/brainy'
import * as fs from 'fs'
async function importCatalog() {

View file

@ -108,7 +108,7 @@ check fails — useful for piping into monitoring or CI.
## Programmatic inspection
```typescript
import { Brainy } from '@soulcraftlabs/brainy'
import { Brainy } from '@soulcraft/brainy'
const reader = await Brainy.openReadOnly({
storage: { type: 'filesystem', path: '/data/brain' }

View file

@ -21,21 +21,21 @@ next:
## Install
```bash
npm install @soulcraftlabs/brainy
npm install @soulcraft/brainy
```
Or with your preferred package manager:
```bash
bun add @soulcraftlabs/brainy
yarn add @soulcraftlabs/brainy
pnpm add @soulcraftlabs/brainy
bun add @soulcraft/brainy
yarn add @soulcraft/brainy
pnpm add @soulcraft/brainy
```
## Verify
```typescript
import { Brainy } from '@soulcraftlabs/brainy'
import { Brainy } from '@soulcraft/brainy'
const brain = new Brainy()
await brain.init()
@ -52,7 +52,7 @@ npm install @soulcraft/cor
```
```typescript
import { Brainy } from '@soulcraftlabs/brainy'
import { Brainy } from '@soulcraft/brainy'
const brain = new Brainy({ plugins: ['@soulcraft/cor'] })
await brain.init() // native providers registered during init
@ -71,7 +71,7 @@ remains available on npm if you need it.
Brainy ships with full TypeScript types. No `@types/` package needed:
```typescript
import { Brainy, NounType, VerbType } from '@soulcraftlabs/brainy'
import { Brainy, NounType, VerbType } from '@soulcraft/brainy'
const brain = new Brainy()
await brain.init()

View file

@ -66,7 +66,7 @@ const results = await brain.search("query")
**New diagnostics for capacity planning and performance tuning.**
```typescript
import { Brainy } from '@soulcraftlabs/brainy'
import { Brainy } from '@soulcraft/brainy'
const brain = new Brainy()
await brain.init()
@ -112,7 +112,7 @@ Recommendations: ${stats.recommendations.join(', ')}
### Step 1: Update Package
```bash
npm install @soulcraftlabs/brainy@latest
npm install @soulcraft/brainy@latest
```
### Step 2: Restart Your Application
@ -134,7 +134,7 @@ npm run start
### Check Adaptive Sizing is Working
```typescript
import { Brainy } from '@soulcraftlabs/brainy'
import { Brainy } from '@soulcraft/brainy'
const brain = new Brainy()
await brain.init()
@ -218,7 +218,7 @@ For debugging or compatibility testing:
If you need to rollback to v3.35.0:
```bash
npm install @soulcraftlabs/brainy@3.35.0
npm install @soulcraft/brainy@3.35.0
```
**Note:** We don't anticipate any issues, but rollback is straightforward if needed.
@ -367,7 +367,7 @@ if (stats.fairness.fairnessViolation) {
## Next Steps
1. ✅ **Upgrade:** `npm install @soulcraftlabs/brainy@latest`
1. ✅ **Upgrade:** `npm install @soulcraft/brainy@latest`
2. 📊 **Monitor:** Use `getCacheStats()` to verify performance improvements
3. 🎯 **Tune:** Adjust based on recommendations (if needed)
4. 📖 **Read:** [Operations Guide](../operations/capacity-planning.md) for capacity planning

View file

@ -37,7 +37,7 @@ This single WASM file contains everything needed for sentence embeddings.
```bash
# Bun as a runtime — supported and recommended
bun add @soulcraftlabs/brainy
bun add @soulcraft/brainy
bun run server.ts
```

View file

@ -1,99 +0,0 @@
---
title: Migrating to 9.0 — your fields and system fields
slug: guides/namespace-migration
public: true
category: guides
template: guide
order: 1
description: The simple story of the 9.0 field-addressing change and the mechanical checklist for updating your call sites — every miss fails loudly with the fix in the error.
next:
- concepts/field-addressing
---
# Migrating to 9.0 — your fields and system fields
The one-sentence version: **your data's field names are now completely
yours, the engine's own fields all live behind one `system.` prefix, and
nothing in between can silently go wrong anymore.**
## What changed, simply
**1. Any field name just works.** Before 9.0 the engine quietly owned
certain names. A field called `level` could be shadowed by the engine's
internal index layer of the same name (sorts silently returned insertion
order); names like `confidence` or `subtype` were rejected inside
`metadata`; names like `content` or `id` were silently never indexed, so
filtering on them returned nothing. All of that is gone. Any name —
`level`, `confidence`, `type`, `id`, `content`, anything — is stored
exactly as written and works with every feature: filtering, sorting,
grouping, aggregation, search, and time-travel reads.
**2. The engine's fields moved behind `system.`.** The engine still keeps
its own per-record bookkeeping — creation time, type, confidence, and so
on. Those are reached one way only now: spelled out, e.g.
`system.createdAt`, `system.type`. They are just as queryable and sortable
as before. `orderBy: 'createdAt'` means *your* field named `createdAt`;
`orderBy: 'system.createdAt'` means the engine's timestamp. No guessing,
no priority rules.
**3. Storage keeps the two physically separate.** New records store your
metadata in its own nested compartment, so a user field named
`confidence` and the engine's confidence live side by side, both intact,
through restarts, index rebuilds, and `asOf()` history. Old records stay
readable forever; nothing rewrites your data.
**4. Mistakes are loud.** An ambiguous or unknown field name is a typed
error naming the fix. Unimplemented options refuse instead of being
ignored. The only forbidden name in your metadata is one literally
starting with `system.`.
## The mechanical checklist
Every missed site fails **loudly** with the correction in the error
message — nothing silently changes meaning. Sweep these patterns:
| Before (8.x) | After (9.0) |
|---|---|
| `orderBy: 'createdAt'` (meaning the engine timestamp) | `orderBy: 'system.createdAt'` |
| `where: { subtype: 'invoice' }` (the engine subtype) | `where: { 'system.subtype': 'invoice' }` |
| `where: { confidence: { greaterThan: 0.8 } }` (the engine scalar) | `where: { 'system.confidence': { greaterThan: 0.8 } }` |
| `groupBy: ['noun']` or `groupBy: ['type']` | `groupBy: ['system.type']` |
| `where: { visibility: 'internal' }` / `{ service: … }` (engine values) | `'system.visibility'` / `'system.service'` |
| `metadata: { confidence: 0.9 }` expecting a throw or a lift to the engine scalar | it is YOUR field now — set the engine scalar via the `confidence` param |
| `new Brainy({ reservedFieldPolicy: … })` | remove the option (it throws with this note) |
| `find({ cursor })` / `includeRelations` / `writeOnly` | refuse with `UnsupportedFindOptionError` — they were silently ignored before |
If a bare name in a query was genuinely *your* field all along (`orderBy:
'score'`, `where: { status: 'active' }`), **change nothing** — bare names
mean your fields, always.
## What happens at first open
Each existing database rebuilds its derived indexes once, automatically,
at the first open on 9.0 (index epoch 3 — the index keys split the two
namespaces). One-time cost, observable via `getIndexStatus()`; no manual
step, and your stored data is not modified.
## For tooling and raw-record readers
If you read raw stored records (fact-log scanners, export tooling), use
the exported shape-aware splitters — they handle both record eras:
```typescript
import { splitNounMetadataRecord } from '@soulcraftlabs/brainy'
const { reserved, custom } = splitNounMetadataRecord(rawRecord)
// reserved = engine fields · custom = the user's bag, ANY names
```
Feature detection (never version-sniff):
```typescript
import * as brainy from '@soulcraftlabs/brainy'
const lawActive = 'FIELD_ADDRESSING_CAPABILITY' in brainy // 'field-addressing/v1'
```
## Where to go next
- [Field addressing](../concepts/field-addressing.md) — the full contract:
the ten system scalars, the relation mirror, refusal semantics, and the
cross-engine ordering guarantees.

View file

@ -9,7 +9,7 @@ Complete guide to integrating Brainy with Next.js applications, covering App Rou
```bash
npx create-next-app@latest my-brainy-app
cd my-brainy-app
npm install @soulcraftlabs/brainy
npm install @soulcraft/brainy
```
### Basic Setup
@ -18,7 +18,7 @@ npm install @soulcraftlabs/brainy
// app/components/BrainyProvider.jsx
'use client'
import { createContext, useContext, useEffect, useState } from 'react'
import { Brainy } from '@soulcraftlabs/brainy'
import { Brainy } from '@soulcraft/brainy'
const BrainyContext = createContext()
@ -271,7 +271,7 @@ export default function SearchPage() {
```javascript
// app/api/search/route.js (App Router)
import { Brainy } from '@soulcraftlabs/brainy'
import { Brainy } from '@soulcraft/brainy'
let brain = null
@ -332,7 +332,7 @@ export async function GET() {
```javascript
// pages/api/search.js (Pages Router)
import { Brainy } from '@soulcraftlabs/brainy'
import { Brainy } from '@soulcraft/brainy'
let brain = null
@ -374,7 +374,7 @@ export default async function handler(req, res) {
```javascript
// app/api/data/route.js
import { Brainy } from '@soulcraftlabs/brainy'
import { Brainy } from '@soulcraft/brainy'
let brain = null
@ -418,7 +418,7 @@ export async function POST(request) {
```jsx
// app/actions/brainy.js
'use server'
import { Brainy } from '@soulcraftlabs/brainy'
import { Brainy } from '@soulcraft/brainy'
let brain = null
@ -630,7 +630,7 @@ CMD ["npm", "start"]
/** @type {import('next').NextConfig} */
const nextConfig = {
experimental: {
serverComponentsExternalPackages: ['@soulcraftlabs/brainy']
serverComponentsExternalPackages: ['@soulcraft/brainy']
},
webpack: (config, { isServer }) => {
if (!isServer) {
@ -797,7 +797,7 @@ export function rateLimit(req, limit = 100, window = 60000) {
// app/contexts/BrainyContext.jsx
'use client'
import { createContext, useContext, useReducer, useEffect } from 'react'
import { Brainy } from '@soulcraftlabs/brainy'
import { Brainy } from '@soulcraft/brainy'
const BrainyContext = createContext()
@ -873,7 +873,7 @@ import { BrainyProvider } from '../app/components/BrainyProvider'
import { Search } from '../app/components/Search'
// Mock Brainy
jest.mock('@soulcraftlabs/brainy', () => ({
jest.mock('@soulcraft/brainy', () => ({
Brainy: jest.fn().mockImplementation(() => ({
init: jest.fn().mockResolvedValue(undefined),
find: jest.fn().mockResolvedValue([

View file

@ -32,7 +32,7 @@ Brainy 7.31.0 adds a per-entity revision counter so multiple writers can coordin
Every distributed-job scheduler eventually wants this exact loop:
```ts
import { Brainy, RevisionConflictError } from '@soulcraftlabs/brainy'
import { Brainy, RevisionConflictError } from '@soulcraft/brainy'
const LOCK_ID = '...uuid for this job slot...'
@ -137,7 +137,7 @@ await brain.addIfMissing({ // ← not a real API
It's race-prone as a plain read-then-write: two concurrent imports both see "not found," both insert, you get duplicates. Without a unique-index primitive (which Brainy doesn't have today), close the race with whole-store CAS — read at a pinned generation, then commit only if nothing moved:
```ts
import { GenerationConflictError } from '@soulcraftlabs/brainy'
import { GenerationConflictError } from '@soulcraft/brainy'
async function addIfMissingByEmail(email: string, data: string) {
for (let attempt = 0; attempt < 5; attempt++) {

View file

@ -18,13 +18,13 @@ Get Brainy running in under a minute.
## 1. Install
```bash
npm install @soulcraftlabs/brainy
npm install @soulcraft/brainy
```
## 2. Initialize
```typescript
import { Brainy, NounType, VerbType } from '@soulcraftlabs/brainy'
import { Brainy, NounType, VerbType } from '@soulcraft/brainy'
const brain = new Brainy()
await brain.init()
@ -67,7 +67,7 @@ await brain.relate({
## 5. Query with Triple Intelligence
```typescript
import type { Result } from '@soulcraftlabs/brainy'
import type { Result } from '@soulcraft/brainy'
// All three search paradigms in one call
const results: Result[] = await brain.find({

View file

@ -344,12 +344,8 @@ For per-entity write coordination (rather than whole-store history), the
## Keeping history bounded
Under Model-B every write is a generation, so history can grow quickly —
Brainy auto-compacts at `close()` (time-bounded per pass) under the
**`retention`** knob (configured on the constructor). Since 8.9.0, `flush()`
never compacts: flushing is durability work and costs only what the current
window's writes cost, regardless of history backlog. A long-lived writer that
never closes keeps its history until its next explicit `compactHistory()`
schedule one in your maintenance window if you run bounded retention:
Brainy auto-compacts on every `flush()`/`close()` under the **`retention`**
knob (configured on the constructor):
```typescript
// Zero-config: ADAPTIVE — keep as much history as free disk/RAM allows,
@ -363,13 +359,10 @@ new Brainy({ retention: 'all' })
new Brainy({ retention: { maxGenerations: 1000, maxAge: 7 * 86_400_000, maxBytes: 512 * 1024 ** 2 } })
```
Reclaim manually at any time (the same caps, plus an optional per-pass time
budget for maintenance windows — an early stop is a consistent prefix and the
next pass resumes):
Reclaim manually at any time (the same caps):
```typescript
await brain.compactHistory({ maxGenerations: 100, maxAge: 7 * 24 * 60 * 60 * 1000 })
await brain.compactHistory({ maxBytes: 512 * 1024 ** 2, timeBudgetMs: 10_000 })
```
Compaction never breaks a pinned read — record-sets are reclaimed only when

View file

@ -11,7 +11,7 @@
### One Interface for Everything
```typescript
import { Brainy } from '@soulcraftlabs/brainy'
import { Brainy } from '@soulcraft/brainy'
const brain = await Brainy.create()
@ -78,7 +78,7 @@ interface ImportProgress {
```typescript
import { useState } from 'react'
import { Brainy } from '@soulcraftlabs/brainy'
import { Brainy } from '@soulcraft/brainy'
function UniversalImportProgress({ file }: { file: File }) {
const [progress, setProgress] = useState({
@ -177,7 +177,7 @@ function UniversalImportProgress({ file }: { file: File }) {
```typescript
import ora from 'ora'
import { Brainy } from '@soulcraftlabs/brainy'
import { Brainy } from '@soulcraft/brainy'
async function importWithProgress(filePath: string) {
const spinner = ora('Starting import...').start()

View file

@ -28,7 +28,7 @@ on-disk layout (memory's "disk" is a JS Map).
## Quick start
```ts
import { Brainy } from '@soulcraftlabs/brainy'
import { Brainy } from '@soulcraft/brainy'
// Filesystem (recommended for any persistent workload):
const brain = new Brainy({
@ -134,7 +134,7 @@ config; the `type` is optional.
If you want to skip the factory:
```ts
import { FileSystemStorage, MemoryStorage } from '@soulcraftlabs/brainy'
import { FileSystemStorage, MemoryStorage } from '@soulcraft/brainy'
const fsStorage = new FileSystemStorage('./brainy-data')
const memStorage = new MemoryStorage()

View file

@ -34,7 +34,7 @@ Three layers solve this:
### Write
```typescript
import { Brainy, NounType } from '@soulcraftlabs/brainy'
import { Brainy, NounType } from '@soulcraft/brainy'
const brain = new Brainy()
await brain.init()
@ -240,7 +240,7 @@ await brain.migrateField({
A realistic adoption sequence for a brain that started without these primitives:
```typescript
import { Brainy, NounType } from '@soulcraftlabs/brainy'
import { Brainy, NounType } from '@soulcraft/brainy'
const brain = new Brainy({ storage: { type: 'filesystem', path: './brain-data' } })
await brain.init()

View file

@ -25,7 +25,7 @@ content — and how 8.0 recovers it for you.
## TL;DR
- **Just upgrade to `@soulcraftlabs/brainy@8.0.12` (or later) and open the store.**
- **Just upgrade to `@soulcraft/brainy@8.0.12` (or later) and open the store.**
If a previous upgrade left VFS content stranded, 8.0.12 **heals it on open**,
with no operator action.
- Want to force or script it? Call **`await brain.vfs.adoptOrphanedBlobs()`**.
@ -90,7 +90,7 @@ So the operator action for a stranded store is simply: **upgrade to 8.0.12 and
open it.**
```ts
import { Brainy } from '@soulcraftlabs/brainy'
import { Brainy } from '@soulcraft/brainy'
// Opening the store is all that is required — recovery runs during init().
const brain = new Brainy({ storage: { type: 'filesystem', path: '/data/my-store' } })
@ -182,5 +182,5 @@ and opening each store is sufficient.
The recovery is copy-only, so no rollback of the recovery itself is ever needed.
If you need to roll back the **whole** 7→8 upgrade, restore the directory from
your pre-upgrade backup (retained automatically while recovery is incomplete, or
your own snapshot) and pin `@soulcraftlabs/brainy@7.x`. 8.0 does not keep the old
your own snapshot) and pin `@soulcraft/brainy@7.x`. 8.0 does not keep the old
branch layout in place, so a directory-level restore is the rollback path.

View file

@ -12,7 +12,7 @@ Complete guide to integrating Brainy with Vue.js applications, covering Vue 3, N
npm create vue@latest my-brainy-app
cd my-brainy-app
npm install
npm install @soulcraftlabs/brainy
npm install @soulcraft/brainy
```
### Basic Setup
@ -574,7 +574,7 @@ Nuxt's server engine (Nitro) is the natural home for Brainy: it runs on Node/Bun
```javascript
// server/utils/brain.js (server-only — Nitro never bundles this into the client)
import { Brainy } from '@soulcraftlabs/brainy'
import { Brainy } from '@soulcraft/brainy'
let brainPromise
@ -1201,7 +1201,7 @@ import vue from '@vitejs/plugin-vue'
export default defineConfig({
plugins: [vue()],
ssr: {
external: ['@soulcraftlabs/brainy']
external: ['@soulcraft/brainy']
}
})
```

View file

@ -24,7 +24,7 @@ Brainy's neural extraction system uses a **4-signal ensemble architecture** to c
### Method 1: Brain Instance (Recommended)
```typescript
import { Brainy, NounType } from '@soulcraftlabs/brainy'
import { Brainy, NounType } from '@soulcraft/brainy'
const brain = new Brainy()
await brain.init()
@ -62,9 +62,9 @@ const people = await brain.extractEntities('...', {
import {
SmartExtractor,
SmartRelationshipExtractor
} from '@soulcraftlabs/brainy'
} from '@soulcraft/brainy'
// Or use subpath imports:
import { SmartExtractor } from '@soulcraftlabs/brainy/neural/SmartExtractor'
import { SmartExtractor } from '@soulcraft/brainy/neural/SmartExtractor'
const brain = new Brainy()
await brain.init()
@ -176,7 +176,7 @@ const withVectors = await brain.extractEntities(text, {
**Direct entity type classifier.** Use when you have pre-detected candidates or need custom configuration.
```typescript
import { SmartExtractor, FormatContext } from '@soulcraftlabs/brainy'
import { SmartExtractor, FormatContext } from '@soulcraft/brainy'
const extractor = new SmartExtractor(brain, {
minConfidence: 0.7, // Threshold
@ -229,7 +229,7 @@ interface ExtractionResult {
**Relationship type classifier.** Determines verb/relationship types between entities.
```typescript
import { SmartRelationshipExtractor } from '@soulcraftlabs/brainy'
import { SmartRelationshipExtractor } from '@soulcraft/brainy'
const relExtractor = new SmartRelationshipExtractor(brain, {
minConfidence: 0.6,
@ -286,7 +286,7 @@ const rel = await relExtractor.infer(
**Full extraction orchestrator.** Handles candidate detection, classification, and deduplication.
```typescript
import { NeuralEntityExtractor } from '@soulcraftlabs/brainy'
import { NeuralEntityExtractor } from '@soulcraft/brainy'
const extractor = new NeuralEntityExtractor(brain)
@ -607,7 +607,7 @@ const locations = entities.filter(e => e.type === NounType.Location)
### Example 2: Excel Data Classification
```typescript
import { SmartExtractor } from '@soulcraftlabs/brainy'
import { SmartExtractor } from '@soulcraft/brainy'
const extractor = new SmartExtractor(brain)
@ -629,7 +629,7 @@ for (let i = 0; i < cells.length; i++) {
### Example 3: Relationship Extraction
```typescript
import { SmartRelationshipExtractor } from '@soulcraftlabs/brainy'
import { SmartRelationshipExtractor } from '@soulcraft/brainy'
const relExtractor = new SmartRelationshipExtractor(brain)

View file

@ -1,86 +0,0 @@
# The Path Registry — brainy's twin table
The brainy half of the cross-engine Path Registry (the native accelerator
maintains the master list; IDs are shared and stable — `LC3`, `DP7`, … are
citable in commits, board rounds, release notes, and pins). Every row owes
five things: **service class** (INDEX-SERVED | BOUNDED-FALLBACK, announced |
TYPED REFUSAL), **latency budget** at 1k/10k/100k/1M (design bar: billions),
**lifecycle behavior**, **failure narration**, and a **test pin**. A path not
in this registry does not ship; an unregistered path is a red gate in the
scan audit.
**The availability bar governing every row: user-visible downtime is
seconds, at restart only.** Migration, heal, compaction, embedding, and
retention run behind the doors — yielding, budget-capped, narrated. No path
may hold the doors while it does housekeeping.
Status legend: ✅ contracted + pinned (test cited) · 🟡 partial (what holds
and what's missing, stated) · 🔴 owed (named, never silent).
## LC — Lifecycle
| ID | Brainy row | Status |
|----|-----------|--------|
| LC1 | Same-version reopen adopts everything: brain-format epoch match → zero rebuilds; aggregation state adopts by stamp; persisted indexes load. | ✅ `tests/unit/brainy/brain-format-handshake` + `migration-deference` (no-drift reopen never rebuilds) |
| LC2 | New empty brain: doors immediate. | ✅ exercised by every suite's setup |
| LC3 | Upgrade, same epoch: as LC1 — new code on unchanged formats owes nothing at open. | ✅ same pins as LC1 (epoch equality is the gate) |
| LC4 | Upgrade with epoch migration: TODAY brainy's epoch rebuild runs at open before doors. | 🔴 **owed — the sev's lockout row.** The doors-open-serving-old-structures design (yielding installments + atomic swap) lands measured-and-gated behind the service-class pair, per the lifecycle-sprint choreography. Acceptance case: the 9,184-row hours-lockout. |
| LC5 | Crash recovery: bounded, resumable, narrated. Aggregation leg ✅ (behind-stamp → incremental catch-up off the fact log + time-travel reconciliation, capped at 5,000 affected before an ANNOUNCED rescan). Vector/metadata legs ride epoch machinery (rebuild-from-canonical, narrated). | 🟡 aggregation pinned (`tests/integration/aggregation-lifecycle-catchup`); the rebuild legs are narrated but not yet installment-yielding (couples to LC4) |
| LC6 | Shutdown under load: close() drains the background flush flight, tears down cadence timers, runs ONE time-bounded compaction pass (~5s budget, resumable). | 🟡 pinned for flush/compaction (8.9.0 suites); SIGTERM drain budget not yet declared |
| LC7 | Rollback/downgrade: an N1 build opening an N brain. | 🔴 owed — no declared read-compat window or typed refusal today (epoch mismatch triggers a rebuild, not a refusal; v2 nested-bag records read as a phantom user field on pre-law builds). Needs the declared-window contract. |
| LC8 | Relocatable brain directory: no absolute paths in artifacts; persist()/load() round-trips. | 🟡 persist/load pinned; byte-for-byte relocation depot cases are the pair gate's (shared corpora) |
| LC9 | Double-open: second writer gets a typed lock refusal (PID-liveness + heartbeat stale detection; `force` escape hatch logs loudly). | ✅ writer-lock suites (8.7.1) |
## DP — Data plane
| ID | Brainy row | Status |
|----|-----------|--------|
| DP1 | `get()` by id: direct storage read + hydrate. INDEX-SERVED (id-mapped). Milliseconds at every scale. | ✅ exercised everywhere; budget rides the pair speed table |
| DP2 | `find({query})`: embed + vector search. The embed dominates (native side owns the budget); JS HNSW serves the search leg. | 🟡 300ms-class p95 is the pair speed-table row; brainy-alone budget declared there |
| DP3 | Filtered/sorted list: column top-K when the field is columnized (INDEX-SERVED, zero canonical reads on the sorted page — value pairs come from ONE batched metadata-record pass); no-column fallback is BOUNDED-ANNOUNCED (one batch pass, announces once per field past 500 rows); unknown field → TYPED REFUSAL naming both candidate spellings. | ✅ `tests/unit/utils/metadataIndex-sort-callshape` (zero per-row reads, batch-only — latency-blind) + `metadataIndex-nested-orderby` (dotted keys serve-or-refuse) + `tests/integration/orderby-sort-bug` |
| DP4 | Aggregation/stats: ALWAYS answers. Write-time incremental; behind-stamp reconciles incrementally; genuine rebuilds go through the native parallel door or the paged JS walk; nothing ever latches off; before-image-less deletes flag a LOUD rescan, never a silent skip. | ✅ `tests/integration/aggregation-lifecycle-catchup` + `tests/unit/aggregation/aggregation-provider-rebuild` |
| DP5 | Graph traversal: `related()` paged via adjacency; whole-graph analytics carry declared cost. | 🟡 paged reads pinned; analytics cost-class declaration owed (rides VENUE-GRAPH-TRUST audit tool) |
| DP6 | Single write: ack at the canonical commit; visibility committed at ack (the atomic vector update kills the remove→add dark window); maintenance NEVER holds the ack (background flush cadence — THE ACK LAW pins: a hung flush cannot block a write, a hung EMBEDDER cannot block a write). | ✅ `tests/unit/brainy/persistence-policy` + `tests/unit/hnsw/update-item-atomic` + `tests/integration/deferred-embedding` |
| DP7 | Bulk ingest: sustained rate holds flat — per-write maintenance taxes must not grow with brain size (A4 removed caller-flush convoys; deferred embedding removes the per-write embed tax where opted). | 🟡 the decay-curve row is a pair speed-table RED GATE; brainy-alone sustained-rate run rides the same corpora |
| DP8 | Read under write pressure: no flicker window — a row that exists is never invisible to recall, even transiently (same-vector re-index is a no-op; changed-vector swaps in place, node never leaves the index; deferred updates serve the OLD vector until the atomic swap — stale-beats-absent). | ✅ brainy leg pinned (`tests/unit/hnsw/update-item-atomic` 9/9 + `deferred-embedding` stale-beats-absent); the symmetry property suite + runtime sentinels remain the B4 program |
| — | **As-of semantic recall** (time-travel vector search): `asOf(G).find()` serves the vectors AS THEY STOOD at G — byte-exact past vectors, tombstone masking, the deferred-embed cell honest on the vector leg, TYPED refusal beyond the head. Brainy-alone leg = ephemeral at-generation materialization (documented O(n log n at G) build, bounded); the at-scale leg rides the accelerated provider's as-of index. | ✅ `tests/integration/asof-semantic-recall` 4/4 (registry ID pending the master table's mint) |
| — | **The lazy-open gate honors EVERY provider's not-ready report** (a not-ready metadata provider can no longer latch the silent-empty state under `disableAutoRebuild`). | ✅ `tests/unit/brainy/lazy-notready-honor` |
## MT — Maintenance (never in the door path)
| ID | Brainy row | Status |
|----|-----------|--------|
| MT1 | Flush/checkpoint: ENGINE-OWNED cadence (write-count/interval/idle triggers, single-flight, background, loud on failure; callers never flush in hot paths; `flush()` stays as an awaitable barrier). | ✅ `tests/unit/brainy/persistence-policy` |
| MT2 | Compaction: never on flush (durability-only law, 8.9.0); close-time pass time-budgeted + resumable; explicit `compactHistory({timeBudgetMs})`. | ✅ 8.9.0 suites |
| MT3 | Index upkeep (mapper folds, delta promotion): native-side machinery; brainy's JS legs are small and synchronous-cheap. | 🟡 declared; yield audit rides the pair |
| MT4 | Heal/rebuild walks (`repairIndex`, backfill walks): paged; failure latches with cooldown; NOT yet yield-to-foreground installments. | 🔴 owed — the priority-isolation clause (couples to LC4; same choreography) |
| MT5 | Deferred embedding worker: ack at durability, durable pending markers (written BEFORE the commit — orphan-safe), crash-recovered at open via a bounded prefix listing, single-flight, 60s hang guard, `awaitPendingEmbeds()` barrier + `pendingEmbeds` gauge. VFS write paths adopt it end-to-end. | ✅ `tests/integration/deferred-embedding` 5/5 |
| MT6 | Retention/archival walks: retention `'all'` does nothing by design; bounded-retention reclaim is close-time/explicit only. | 🟡 8.9.0 behavior pinned; archival profile is the co-frozen D1+D3 unit |
## FM — Failure modes
| ID | Brainy row | Status |
|----|-----------|--------|
| FM1 | Disk full / IO error mid-op: transaction rollback + typed error; failed rollback → StoreInconsistentError quarantines writes until repairIndex(). | 🟡 rollback paths pinned; explicit disk-full depot case owed |
| FM2 | Memory pressure: query limits + reserved-memory config; unified cache eviction. | 🟡 declared budgets; cascade pin owed |
| FM3 | Torn/corrupt file on open: malformed brain-format marker → safe rebuild (never trusting a bad epoch); corrupt records surface loudly. | 🟡 marker pin ✅ (`brain-format-handshake`); broader quarantine is native-side |
| FM4 | Native module unavailable: plugin load failure is LOUD (version-coupling law throws on range mismatch — never silently version-drifted); JS engine serves with its own declared budgets, named as the active backend in op names. | ✅ `tests/unit/plugin-version-coupling` + op-name stamping |
## FL — Fleet
| ID | Brainy row | Status |
|----|-----------|--------|
| FL1 | Cold open on demand: LC1's adopt-everything open; warm() available for eager paths. | 🟡 open cost pinned at LC1; millisecond budget rides the speed table |
| FL2FL4 | Boot storm / upgrade wave / isolation: fleet-layer policies over LC1/LC4 — engine leg = budgeted opens + LC4's behind-doors migration. | 🔴 owed with LC4 |
| FL5 | Brain as product object: create instant (LC2) · erase = `clear()` explicit + complete · export = portable-graph, canon-complete mode available. | ✅ clear-persistence + portable-graph + canonical-enumeration suites |
## Status summary
Contracted + pinned this train: **DP3, DP4, DP6, DP8(brainy leg), MT1,
MT5, LC5(aggregation), the lazy-open not-ready gate, LC1/LC3/LC9, FM4,
FL5** — each with the cited test. Owed, in production-risk order, all
coupled to the priority-isolation program the lifecycle sev opened: **LC4
(doors-open migration), MT4 (yielding heals), LC7 (downgrade contract),
LC6 (SIGTERM budget), FL2FL4, FM1/FM2 depot cases, B4 symmetry suite +
sentinels.** Rows move from owed to contracted only with a cited test —
none lands by prose.

View file

@ -1,83 +0,0 @@
---
title: Performance Envelopes
slug: guides/performance-envelopes
public: true
category: guides
template: guide
order: 40
description: Measured per-operation latency envelopes at stated scales — what to expect, on what hardware, and exactly how each number was produced.
next:
- guides/find-limits
---
# Performance Envelopes
Every number on this page is **measured, never projected** — produced by the script
cited at the bottom, against the built package (the artifact you install), on the stated
hardware. Each entry says what was measured, at what scale, on which storage backend.
When a release touches a measured path, that operation is re-measured and this page
updates in the same release.
Two scopes to keep straight:
- **These envelopes are the pure-JS engine** (no native accelerator registered) on
filesystem storage. This is the floor every deployment gets from `npm install` alone.
- **Accelerated deployments** (the optional native provider) publish their own numbers —
this page never claims them.
## Read operations
Reads are where the architecture pays off: after the write path has done its indexing
work, queries answer from purpose-built indexes without scanning.
| Operation | 1,000 entities | 10,000 entities | Notes |
|---|---|---|---|
| `get(id)` (warm) | p50 < 0.1ms | p50 < 0.1ms | served from cache/metadata index |
| `find` (metadata: indexed equality + range, limit 100) | p50 1.0ms · p95 1.8ms | p50 7.0ms · p95 8.9ms | column-store bitmap paths |
| `related(id)` (per-node adjacency) | p50 < 0.1ms · p95 0.2ms | p50 < 0.1ms | LSM adjacency index O(degree), scale-independent |
| `find` (semantic: embed + HNSW, 1k docs) | p50 178ms · p95 393ms | — | dominated by WASM query embedding (measured on a machine under concurrent load — treat the p95 as an upper bound); the vector search itself is single-digit ms |
## Write operations
Under Model-B **every write is its own durable generation** — a single-op `add` pays
serialization, before-image staging, and fsync before it acks. That durability is priced
into the write path visibly, by design:
| Operation | 1,000 entities | 10,000 entities | Notes |
|---|---|---|---|
| `add` (single-op) | p50 167ms · p95 171ms | p50 165ms · p95 172ms | full durable generation per write — flat across scale |
| `addMany` (bulk) | ~163ms/entity | ~187ms/entity | **currently per-item commits** — see the honest note below |
| `relateMany` | ~0.8ms/edge | ~0.9ms/edge | edges batch efficiently today |
| `flush` (steady-state, 1 pending write) | p50 8ms · p95 10ms | p50 45ms · p95 52ms | durability-only since 8.9.0 — cost no longer depends on history backlog or retention mode |
**The honest note on bulk writes:** `addMany` today commits each item as its own
generation (the same durability as single-op `add`, serialized by the single-writer
lock), so bulk-load cost is N × single-op cost. Batched chunk commits (one generation
and one fsync window per chunk, as `removeMany` already does) are designed into the
unified-commit work on the current roadmap. Until that ships, size bulk imports
accordingly — 10k entities is minutes, not seconds, on filesystem storage.
## Open / close
| Operation | 1,000 entities | 10,000 entities | Notes |
|---|---|---|---|
| `open` (empty store) | ~560ms | ~190ms | includes embedder initialization |
| `open` (warm, populated, clean shutdown) | 763ms | 4.9s | pure-JS vector index load dominates and grows with entity count; the native accelerator exists precisely to remove this |
| `close` | bounded | bounded | auto-compaction pass is time-bounded (~5s max) since 8.9.0 |
A store that was NOT cleanly closed pays index rebuilds on top of the warm-open
number (tens of seconds at 10k) — clean shutdown is worth engineering for.
## How these were produced
- **Hardware**: Intel Core i9-14900HX (32 threads), 62GB RAM, NVMe, Linux, Node v22.
- **Backend**: `storage: { type: 'filesystem' }`, pure JS (no native providers).
- **Embeddings**: deterministic stub for non-semantic ops (isolates engine cost);
the real WASM embedder for the semantic row (that's what you'll run).
- **Method**: p50/p95 over 50200 samples per op against the built `dist/`;
the measuring script ships in the repo history and re-runs per release.
Numbers on different hardware will differ; the *shape* (sub-2ms indexed reads,
~160ms embedding-bound semantic queries, durability-priced writes) is the envelope
you should hold your deployment against. If your measurements diverge from these
shapes by an order of magnitude, something is wrong — file it.

View file

@ -204,8 +204,8 @@ await brain.add({ data: { name: 'Entity' }, type: NounType.Thing })
### Basic Add Operation
```typescript
import { Brainy } from '@soulcraftlabs/brainy'
import { NounType } from '@soulcraftlabs/brainy/types'
import { Brainy } from '@soulcraft/brainy'
import { NounType } from '@soulcraft/brainy/types'
const brain = new Brainy()
await brain.init()
@ -428,7 +428,7 @@ await brain.relate({ ... }) // a crash here leaves the entity unlinked
```typescript
import { describe, it, expect } from 'vitest'
import { Brainy } from '@soulcraftlabs/brainy'
import { Brainy } from '@soulcraft/brainy'
describe('Transaction Tests', () => {
it('should rollback on failure', async () => {

View file

@ -23,7 +23,7 @@ The Universal Display Augmentation is a powerful AI-powered system that automati
### Basic Usage
```typescript
import { Brainy } from '@soulcraftlabs/brainy'
import { Brainy } from '@soulcraft/brainy'
const brainy = new Brainy()
await brainy.init()

View file

@ -71,9 +71,9 @@ Let's build a projection that organizes files by priority (high, medium, low):
### Step 1: Create the Strategy Class
```typescript
import { BaseProjectionStrategy } from '@soulcraftlabs/brainy/vfs/semantic'
import { Brainy } from '@soulcraftlabs/brainy'
import { VirtualFileSystem, VFSEntity } from '@soulcraftlabs/brainy/vfs'
import { BaseProjectionStrategy } from '@soulcraft/brainy/vfs/semantic'
import { Brainy } from '@soulcraft/brainy'
import { VirtualFileSystem, VFSEntity } from '@soulcraft/brainy/vfs'
export class PriorityProjection extends BaseProjectionStrategy {
readonly name = 'priority'
@ -141,7 +141,7 @@ export class PriorityProjection extends BaseProjectionStrategy {
### Step 2: Register the Strategy
```typescript
import { Brainy } from '@soulcraftlabs/brainy'
import { Brainy } from '@soulcraft/brainy'
import { PriorityProjection } from './PriorityProjection'
const brain = new Brainy()
@ -537,7 +537,7 @@ Use the projection's resolve cache:
```typescript
import { describe, it, expect, beforeAll } from 'vitest'
import { Brainy } from '@soulcraftlabs/brainy'
import { Brainy } from '@soulcraft/brainy'
import { PriorityProjection } from './PriorityProjection'
describe('PriorityProjection', () => {
@ -714,7 +714,7 @@ async resolve(brain, vfs, value: string) {
3. Use appropriate limits: Don't fetch more than needed
### Type errors
1. Import correct types: `import { Brainy, VirtualFileSystem } from '@soulcraftlabs/brainy'`
1. Import correct types: `import { Brainy, VirtualFileSystem } from '@soulcraft/brainy'`
2. Use `as VFSEntity` when mapping results
3. Check BaseProjectionStrategy import

View file

@ -14,11 +14,11 @@ A file explorer that:
## ⚡ Step 1: Basic Setup (1 minute)
```bash
npm install @soulcraftlabs/brainy
npm install @soulcraft/brainy
```
```typescript
import { Brainy } from '@soulcraftlabs/brainy'
import { Brainy } from '@soulcraft/brainy'
// ✅ CORRECT: Use filesystem storage for production
const brain = new Brainy({
@ -115,7 +115,7 @@ Here's a complete React component using the correct patterns:
```tsx
import React, { useState, useEffect } from 'react'
import { Brainy } from '@soulcraftlabs/brainy'
import { Brainy } from '@soulcraft/brainy'
export function FileExplorer() {
const [brain, setBrain] = useState(null)
@ -288,8 +288,8 @@ Your file explorer is now working! Here's what to explore next:
### "Module not found" errors
```bash
# Make sure you're using the right import
npm ls @soulcraftlabs/brainy # Check version
npm install @soulcraftlabs/brainy@latest # Update if needed
npm ls @soulcraft/brainy # Check version
npm install @soulcraft/brainy@latest # Update if needed
```
### "VFS not initialized" errors

View file

@ -24,7 +24,7 @@ Brainy VFS is a revolutionary virtual filesystem that runs on top of Brainy's ne
## Quick Start
```javascript
import { VirtualFileSystem } from '@soulcraftlabs/brainy/vfs'
import { VirtualFileSystem } from '@soulcraft/brainy/vfs'
// Initialize the VFS
const vfs = new VirtualFileSystem({
@ -381,7 +381,7 @@ Brainy VFS fully leverages Brainy's revolutionary Triple Intelligence system:
## Installation
```bash
npm install @soulcraftlabs/brainy
npm install @soulcraft/brainy
```
## Requirements

View file

@ -135,7 +135,7 @@ Mount VFS as a native filesystem on Linux/Mac/Windows.
```typescript
// Planned (research phase)
import { mountVFS } from '@soulcraftlabs/brainy/vfs/fuse'
import { mountVFS } from '@soulcraft/brainy/vfs/fuse'
await mountVFS(vfs, {
mountPoint: '/mnt/brainy',
@ -160,7 +160,7 @@ These features would benefit from community contributions. If you're interested
### Express.js Static Middleware
```typescript
// Wanted: Community contribution
import { createStaticMiddleware } from '@soulcraftlabs/brainy/vfs/express'
import { createStaticMiddleware } from '@soulcraft/brainy/vfs/express'
app.use('/files', createStaticMiddleware(vfs, {
index: ['index.html', 'index.md'],
@ -172,7 +172,7 @@ app.use('/files', createStaticMiddleware(vfs, {
### VSCode Extension
```typescript
// Wanted: Community contribution
import { VFSProvider } from '@soulcraftlabs/brainy/vfs/vscode'
import { VFSProvider } from '@soulcraft/brainy/vfs/vscode'
const provider = new VFSProvider(vfs)
vscode.workspace.registerFileSystemProvider('brainy', provider)

View file

@ -327,7 +327,7 @@ console.log(id1 === id2 && id2 === id3) // true
Create your own semantic dimensions:
```typescript
import { BaseProjectionStrategy } from '@soulcraftlabs/brainy/vfs/semantic'
import { BaseProjectionStrategy } from '@soulcraft/brainy/vfs/semantic'
class PriorityProjection extends BaseProjectionStrategy {
readonly name = 'priority'

View file

@ -7,7 +7,7 @@ Brainy's Virtual Filesystem (VFS) provides a POSIX-like filesystem interface tha
## Quick Start
```typescript
import { Brainy } from '@soulcraftlabs/brainy'
import { Brainy } from '@soulcraft/brainy'
// Initialize Brainy
const brain = new Brainy({
@ -598,7 +598,7 @@ const user = await store.findById('users', 'user123')
VFS uses standard POSIX-style errors:
```typescript
import { VFSError, VFSErrorCode } from '@soulcraftlabs/brainy'
import { VFSError, VFSErrorCode } from '@soulcraft/brainy'
try {
await vfs.readFile('/nonexistent.txt')

View file

@ -280,7 +280,7 @@ GitBridge provides Git import/export capabilities:
#### GitBridge Usage
```javascript
// Import and instantiate GitBridge
import { GitBridge } from '@soulcraftlabs/brainy'
import { GitBridge } from '@soulcraft/brainy'
const gitBridge = new GitBridge(vfs, brain)
// Export VFS to Git repository structure
@ -452,7 +452,7 @@ This ordering prevents race conditions where file writes might fail because pare
## Complete Example
```javascript
import { Brainy } from '@soulcraftlabs/brainy'
import { Brainy } from '@soulcraft/brainy'
async function vfsExample() {
// Initialize

View file

@ -196,5 +196,5 @@ await brain.relate({
Always import and use the type enums:
```javascript
import { NounType, VerbType } from '@soulcraftlabs/brainy'
import { NounType, VerbType } from '@soulcraft/brainy'
```

View file

@ -5,7 +5,7 @@
The Brainy VFS is automatically initialized during `brain.init()`. No separate initialization needed!
```javascript
import { Brainy } from '@soulcraftlabs/brainy'
import { Brainy } from '@soulcraft/brainy'
// Create and initialize Brainy
const brain = new Brainy({
@ -71,7 +71,7 @@ VFS stores files as entities and relationships in the same graph as everything e
## Complete Example
```javascript
import { Brainy } from '@soulcraftlabs/brainy'
import { Brainy } from '@soulcraft/brainy'
async function useVFS() {
// Initialize Brainy
@ -100,7 +100,7 @@ useVFS().catch(console.error)
## TypeScript Usage
```typescript
import { Brainy, VirtualFileSystem } from '@soulcraftlabs/brainy'
import { Brainy, VirtualFileSystem } from '@soulcraft/brainy'
class FileManager {
private brain: Brainy

View file

@ -37,7 +37,7 @@ Brainy VFS provides safe, tree-aware methods that prevent these issues:
### Method 1: Use `getDirectChildren()` (Recommended)
```typescript
import { Brainy, VirtualFileSystem } from '@soulcraftlabs/brainy'
import { Brainy, VirtualFileSystem } from '@soulcraft/brainy'
const brain = new Brainy()
await brain.init()
@ -97,7 +97,7 @@ Here's a complete example using React:
```tsx
import React, { useState, useEffect } from 'react'
import { VirtualFileSystem } from '@soulcraftlabs/brainy'
import { VirtualFileSystem } from '@soulcraft/brainy'
interface FileNode {
name: string
@ -177,7 +177,7 @@ function TreeView({ node, onToggle, expanded }) {
If you must build trees manually from flat lists, use the `VFSTreeUtils`:
```typescript
import { VFSTreeUtils } from '@soulcraftlabs/brainy/vfs'
import { VFSTreeUtils } from '@soulcraft/brainy/vfs'
// Get all entities somehow
const allEntities = await vfs.getDescendants('/root')

View file

@ -7,7 +7,7 @@
* the Bluesky firehose with Brainy's distributed architecture
*/
import { Brainy } from '@soulcraftlabs/brainy'
import { Brainy } from '@soulcraft/brainy'
import { WebSocket } from 'ws'
// =====================================================

View file

@ -14,7 +14,7 @@
* ts-node examples/monitor-cache-performance.ts
*/
import { Brainy, NounType } from '@soulcraftlabs/brainy'
import { Brainy, NounType } from '@soulcraft/brainy'
// ANSI color codes for pretty output
const colors = {

View file

@ -5,7 +5,7 @@ Connect Brainy to spreadsheets, BI tools, and external systems with zero configu
## Quick Start
```typescript
import { Brainy } from '@soulcraftlabs/brainy'
import { Brainy } from '@soulcraft/brainy'
const brain = new Brainy({ integrations: true })
await brain.init()
@ -178,7 +178,7 @@ Webhooks include `X-Brainy-Signature` header with HMAC-SHA256 signature.
### Minimal (in-memory):
```typescript
import { Brainy } from '@soulcraftlabs/brainy'
import { Brainy } from '@soulcraft/brainy'
const brain = new Brainy({ integrations: true })
await brain.init()
@ -194,7 +194,7 @@ console.log(brain.hub.getInstructions())
```typescript
import express from 'express'
import { Brainy } from '@soulcraftlabs/brainy'
import { Brainy } from '@soulcraft/brainy'
const app = express()
const brain = new Brainy({
@ -232,7 +232,7 @@ app.listen(3000, () => {
```typescript
import { Hono } from 'hono'
import { Brainy } from '@soulcraftlabs/brainy'
import { Brainy } from '@soulcraft/brainy'
const app = new Hono()

View file

@ -99,7 +99,7 @@ Add the `BRAINY_URL` script property in Apps Script settings.
The simplest way to enable all integrations:
```javascript
import { Brainy } from '@soulcraftlabs/brainy'
import { Brainy } from '@soulcraft/brainy'
const brain = new Brainy({ integrations: true })
await brain.init()
@ -112,7 +112,7 @@ With Express:
```javascript
import express from 'express'
import { Brainy } from '@soulcraftlabs/brainy'
import { Brainy } from '@soulcraft/brainy'
const app = express()
const brain = new Brainy({ integrations: true })

8
package-lock.json generated
View file

@ -1,12 +1,12 @@
{
"name": "@soulcraftlabs/brainy",
"version": "10.4.4",
"name": "@soulcraft/brainy",
"version": "8.8.2",
"lockfileVersion": 3,
"requires": true,
"packages": {
"": {
"name": "@soulcraftlabs/brainy",
"version": "10.4.4",
"name": "@soulcraft/brainy",
"version": "8.8.2",
"license": "MIT",
"dependencies": {
"@msgpack/msgpack": "^3.1.2",

View file

@ -1,7 +1,6 @@
{
"name": "@soulcraftlabs/brainy",
"version": "10.4.4",
"brainyContract": 1,
"name": "@soulcraft/brainy",
"version": "8.8.2",
"description": "Universal Knowledge Protocol™ - World's first Triple Intelligence database unifying vector, graph, and document search in one API. Stage 3 CANONICAL: 42 nouns × 127 verbs covering 96-97% of all human knowledge.",
"main": "dist/index.js",
"module": "dist/index.js",
@ -127,16 +126,15 @@
"license": "MIT",
"private": false,
"publishConfig": {
"access": "public",
"registry": "https://source.soulcraft.com/api/packages/soulcraftlabs/npm/"
"access": "public"
},
"homepage": "https://source.soulcraft.com/soulcraftlabs/open-brainy",
"homepage": "https://github.com/soulcraftlabs/brainy",
"bugs": {
"url": "https://source.soulcraft.com/soulcraftlabs/open-brainy/issues"
"url": "https://github.com/soulcraftlabs/brainy/issues"
},
"repository": {
"type": "git",
"url": "git+https://source.soulcraft.com/soulcraftlabs/open-brainy.git"
"url": "git+https://github.com/soulcraftlabs/brainy.git"
},
"files": [
"dist/**/*.js",

View file

@ -1,76 +0,0 @@
{
"product": "brainy",
"entries": [
{
"version": "11.0.5",
"date": "2026-09-02",
"headline": "Graph-first finds in production, and opens that stop rescanning history",
"items": [
"find({ connected, where }) now walks the neighbours first and filters only those rows through a native door — correct at every page and O(neighbours), never the whole store.",
"related() with a list of verb types returns every requested kind (a fast path had silently kept only the first).",
"Deferred-embedding recovery resumes from a low-water mark instead of rescanning the whole generation log at every open — measured at two minutes on a large brain, now milliseconds."
],
"url": null,
"thumb": null
},
{
"version": "11.0.4",
"date": "2026-09-01",
"headline": "Closes in milliseconds, index rebuilds without the disk-sync storm",
"items": [
"close() no longer pays deferred compaction or waits out an in-flight rebuild — measured 8 ms against the 4-minute closes it replaces; deferred work resumes at the next open, in the background.",
"The metadata index's rebuild syncs to disk per shard instead of per row, and the durability point moved to the publish step — the same guarantee, a fraction of the disk traffic.",
"A new native filter door evaluates queries over exactly the candidate rows a graph walk found, never the whole store."
],
"url": null,
"thumb": null
},
{
"version": "11.0.3",
"date": "2026-09-01",
"headline": "The embedding upgrade ceremony runs on every brain",
"items": [
"A brain opened through the standard plugin now carries its embedding-model identity, so the full-precision upgrade ceremony can run on it.",
"A one-fix release; nothing else changed."
],
"url": null,
"thumb": null
},
{
"version": "11.0.2",
"date": "2026-08-31",
"headline": "One embedding quality everywhere, 34× faster imports",
"items": [
"Every runtime embeds with the same full-precision model — search quality no longer depends on where you run.",
"Bulk embedding measured 3.14.2× faster, and an online re-embed ceremony upgrades existing stores without downtime.",
"The engine's change feed is documented, with the SSE/WebSocket fan-out pattern for realtime surfaces."
],
"url": null,
"thumb": null
},
{
"version": "11.0.1",
"date": "2026-08-31",
"headline": "Deletes inside transactions are safe",
"items": [
"Deleting relations inside a transact() no longer corrupts index bookkeeping.",
"A store that deletes its last relation keeps serving instead of refusing."
],
"url": null,
"thumb": null
},
{
"version": "11.0.0",
"date": "2026-08-28",
"headline": "One install, one engine — Brainy",
"items": [
"The former two-package pair is one package: the native engine under the familiar API. One import is the whole install.",
"A missing native build refuses loudly with its cures named; nothing falls back silently.",
"Stores open in place — no migration."
],
"url": null,
"thumb": null
}
],
"history": "The version line continues from the 4.3.x native-engine releases; their record lives in the product repository's CHANGELOG.md."
}

View file

@ -1,110 +0,0 @@
{
"product": "open-brainy",
"entries": [
{
"version": "10.4.9",
"date": "2026-09-02",
"headline": "Graph-first finds, honest verb arrays, and opens that stop rescanning history",
"items": [
"find({ connected, where }) now walks the neighbours first and filters only those rows — correct at every page, and O(neighbours) instead of O(store).",
"related() with a list of verb types (or sources, or targets) returns every requested kind — four fast paths silently kept only the first.",
"Deferred-embedding recovery resumes from a low-water mark instead of rescanning the whole generation log at every open — measured at two minutes on a large brain, now milliseconds."
],
"url": "https://source.soulcraft.com/soulcraftlabs/open-brainy/releases/tag/v10.4.9",
"thumb": null
},
{
"version": "10.4.7",
"date": "2026-09-01",
"headline": "Count ledgers can no longer race themselves",
"items": [
"Concurrent count flushes coalesce into one writer with a trailing pass — parallel flushes can no longer corrupt a store's count ledger.",
"Atomic writes carry a per-process sequence, so two processes' temp files can never collide."
],
"url": "https://source.soulcraft.com/soulcraftlabs/open-brainy/releases/tag/v10.4.7",
"thumb": null
},
{
"version": "10.4.6",
"date": "2026-08-31",
"headline": "Transactions cross the index seam safely",
"items": [
"Deleting relations inside a transact() no longer fails against the metadata index — operations take a JSON-safe view at the moment they execute.",
"Fixes a class of transaction failures on stores with integer-mapped relation endpoints."
],
"url": "https://source.soulcraft.com/soulcraftlabs/open-brainy/releases/tag/v10.4.6",
"thumb": null
},
{
"version": "10.4.5",
"date": "2026-08-31",
"headline": "Recovery tells the truth, docs live at home",
"items": [
"A torn generation-log tail is a terminal verdict with a named cure — never an endless wait at open.",
"A sealed segment declares only the generations it actually holds.",
"The engine's documentation now publishes from its own repository."
],
"url": "https://source.soulcraft.com/soulcraftlabs/open-brainy/releases/tag/v10.4.5",
"thumb": null
},
{
"version": "10.4.4",
"date": "2026-08-28",
"headline": "Faster opens, quieter idle",
"items": [
"Opening a store discovers generations from directory names instead of walking the log, and answers \"any entities?\" with one directory read.",
"The flush-request watch is event-driven; idle stores stop paying a polling heartbeat.",
"A slow open now names the exact step it is in, so operators see what is being paid and why."
],
"url": "https://source.soulcraft.com/soulcraftlabs/open-brainy/releases/tag/v10.4.4",
"thumb": null
},
{
"version": "10.4.3",
"date": "2026-08-27",
"headline": "Open Brainy, under its own name",
"items": [
"The same engine as 10.4.2, now published as @soulcraftlabs/brainy — the MIT reference engine, on The Source.",
"No code changes; your imports change once and everything else stays put."
],
"url": "https://source.soulcraft.com/soulcraftlabs/open-brainy/releases/tag/v10.4.3",
"thumb": null
},
{
"version": "10.4.2",
"date": "2026-08-27",
"headline": "Vectors that lie are refused, counts that drift are caught",
"items": [
"A zero-norm vector is not a vector: the index refuses them, rebuilds skip them, and a sanctioned unvector door removes them cleanly.",
"The canonical count ledger derives from identity records and marks legacy-derived ledgers suspect at load.",
"Plugin activation failures keep their original error as cause, so the real frame reaches your logs."
],
"url": "https://source.soulcraft.com/soulcraftlabs/open-brainy/releases/tag/v10.4.2",
"thumb": null
},
{
"version": "10.4.1",
"date": "2026-08-26",
"headline": "Writes that change nothing cost nothing",
"items": [
"The read gate is per index family, and a write carrying unchanged data never re-embeds.",
"The vectored-row count joins the ledger, so vector coverage is a number you can read, not a guess."
],
"url": "https://source.soulcraft.com/soulcraftlabs/open-brainy/releases/tag/v10.4.1",
"thumb": null
},
{
"version": "10.4.0",
"date": "2026-08-26",
"headline": "Repair routing, the vector ledger, and honest empties",
"items": [
"Repairs route to the index that owns the damage, and the open gate closes the vector leg until coverage is proven.",
"An empty string is real data, not a missing field.",
"The metadata crossing never carries raw integer relation endpoints — a whole class of serialization faults closed."
],
"url": "https://source.soulcraft.com/soulcraftlabs/open-brainy/releases/tag/v10.4.0",
"thumb": null
}
],
"history": "Earlier releases are recorded in CHANGELOG.md in this repository."
}

View file

@ -10,7 +10,6 @@ import { TransformerEmbedding } from '../src/utils/embedding.js'
import * as fs from 'fs/promises'
import * as path from 'path'
import { fileURLToPath } from 'url'
import { resolveDeterministicStamp } from './lib/deterministicStamp.js'
const __dirname = path.dirname(fileURLToPath(import.meta.url))
@ -98,22 +97,13 @@ async function buildEmbeddedPatterns() {
// Convert to base64 for embedding in TypeScript
const uint8 = new Uint8Array(buffer)
const base64 = Buffer.from(uint8).toString('base64')
// Deterministic stamp: derived from the git commit time of this
// generator's inputs, never from wall-clock time — two builds of the
// same source tree must produce byte-identical output.
const outputPath = path.join(__dirname, '..', 'src', 'neural', 'embeddedPatterns.ts')
const generatedStamp = resolveDeterministicStamp(
[path.join(__dirname, 'buildEmbeddedPatterns.ts'), libraryPath],
outputPath
)
// Generate TypeScript file with everything embedded
const tsContent = `/**
* 🧠 BRAINY EMBEDDED PATTERNS
*
* AUTO-GENERATED - DO NOT EDIT
* Generated: ${generatedStamp}
* Generated: ${new Date().toISOString()}
* Patterns: ${libraryData.patterns.length}
* Coverage: 94-98% of all queries
*
@ -207,6 +197,7 @@ prodLog.info(\`🧠 Brainy Pattern Library loaded: \${EMBEDDED_PATTERNS.length}
`
// Write the TypeScript file
const outputPath = path.join(__dirname, '..', 'src', 'neural', 'embeddedPatterns.ts')
await fs.writeFile(outputPath, tsContent)
// Report statistics

View file

@ -11,7 +11,6 @@ import * as fs from 'fs/promises'
import * as path from 'path'
import { fileURLToPath } from 'url'
import { NounType, VerbType } from '../src/types/graphTypes.js'
import { resolveDeterministicStamp } from './lib/deterministicStamp.js'
const __dirname = path.dirname(fileURLToPath(import.meta.url))
@ -374,24 +373,12 @@ async function buildTypeEmbeddings() {
const uint8 = new Uint8Array(buffer)
const base64 = Buffer.from(uint8).toString('base64')
// Deterministic stamp: derived from the git commit time of this
// generator's inputs, never from wall-clock time — two builds of the
// same source tree must produce byte-identical output.
const outputPath = path.join(__dirname, '..', 'src', 'neural', 'embeddedTypeEmbeddings.ts')
const generatedStamp = resolveDeterministicStamp(
[
path.join(__dirname, 'buildTypeEmbeddings.ts'),
path.join(__dirname, '..', 'src', 'types', 'graphTypes.ts')
],
outputPath
)
// Generate TypeScript file
const tsContent = `/**
* 🧠 BRAINY EMBEDDED TYPE EMBEDDINGS
*
* AUTO-GENERATED - DO NOT EDIT
* Generated: ${generatedStamp}
* Generated: ${new Date().toISOString()}
* Noun Types: ${nounTypes.length}
* Verb Types: ${verbTypes.length}
*
@ -408,7 +395,7 @@ export const TYPE_METADATA = {
verbTypes: ${verbTypes.length},
totalTypes: ${totalTypes},
embeddingDimensions: ${embeddingDim},
generatedAt: "${generatedStamp}",
generatedAt: "${new Date().toISOString()}",
sizeBytes: {
embeddings: ${buffer.byteLength},
base64: ${base64.length}
@ -507,6 +494,7 @@ prodLog.info(\`🧠 Brainy Type Embeddings loaded: \${TYPE_METADATA.nounTypes} n
`
// Write the TypeScript file
const outputPath = path.join(__dirname, '..', 'src', 'neural', 'embeddedTypeEmbeddings.ts')
await fs.writeFile(outputPath, tsContent)
// Report statistics

View file

@ -1,128 +0,0 @@
#!/usr/bin/env node
/**
* Emit this build's API-contract manifest to docs/api-contract.json.
*
* WHY IT IS GENERATED, NOT WRITTEN: a hand-kept list of doors drifts from the
* code the first time somebody adds one. This reads the surface the build
* actually exposes the prototype's own methods and accessors, the exported
* error classes, the `where` operator sets, the field-addressing vocabulary,
* the health verdicts so a diff between two engines' manifests is a diff
* between two engines, never between two authors.
*
* Requirement marking (required / optional per door) is NOT derivable from the
* surface it is a commitment, recorded with the contract's owner rather than
* here. This manifest carries the surface; the promise lives with the contract.
*
* Usage: node scripts/emit-contract-manifest.mjs [--check]
* --check exits non-zero when the committed manifest is stale.
*/
import { writeFileSync, readFileSync, existsSync } from 'node:fs'
import { join, dirname } from 'node:path'
import { fileURLToPath } from 'node:url'
const ROOT = join(dirname(fileURLToPath(import.meta.url)), '..')
const OUT = join(ROOT, 'docs', 'api-contract.json')
const { Brainy } = await import(join(ROOT, 'dist', 'brainy.js'))
const errorsModule = await import(join(ROOT, 'dist', 'errors', 'brainyError.js'))
const versionModule = await import(join(ROOT, 'dist', 'utils', 'version.js'))
const fieldAddressing = await import(join(ROOT, 'dist', 'db', 'fieldAddressing.js'))
/** Every own method and accessor on the class's prototype, minus the private ones. */
function surfaceOf(ctor) {
const doors = []
for (const name of Object.getOwnPropertyNames(ctor.prototype)) {
if (name === 'constructor' || name.startsWith('_')) continue
const descriptor = Object.getOwnPropertyDescriptor(ctor.prototype, name)
if (!descriptor) continue
if (typeof descriptor.value === 'function') {
doors.push({ name, kind: 'method', arity: descriptor.value.length })
} else if (descriptor.get) {
doors.push({ name, kind: 'accessor' })
}
}
return doors.sort((a, b) => a.name.localeCompare(b.name))
}
const errors = Object.entries(errorsModule)
.filter(([name, value]) => typeof value === 'function' && /Error$/.test(name))
.map(([name]) => name)
.sort()
// The operator sets, read from the engine's own refusal message so the
// manifest can never disagree with the validator.
const filterSource = readFileSync(join(ROOT, 'src', 'utils', 'metadataFilter.ts'), 'utf-8')
const acceptedMatch = filterSource.match(/const VALUE_OPERATORS = new Set<string>\(\[([\s\S]*?)\]\)/)
if (!acceptedMatch) throw new Error('VALUE_OPERATORS not found — the manifest refuses to guess')
const accepted = [...acceptedMatch[1].matchAll(/'([^']+)'/g)].map((m) => m[1]).sort()
const indexSource = readFileSync(join(ROOT, 'src', 'utils', 'metadataIndex.ts'), 'utf-8')
const refusedByIndex = ['endsWith', 'length', 'matches', 'startsWith'].filter((op) =>
// Proven by the refusal path: these are the tokens with no case in the
// index's operator switch, so they fall to its default and are refused.
!new RegExp(`case '${op}':`).test(indexSource)
)
const servedOnIndex = accepted.filter((op) => !refusedByIndex.includes(op))
const manifest = {
contractVersion: versionModule.contractVersion(),
engine: '@soulcraftlabs/brainy',
compatibility: {
minor:
'additive — a new optional door, a new served operator, a new error class; every existing implementation still conforms',
major:
'breaking — a door removed, an answer narrowed, an ordering law changed, an optional door promoted to required, or an operator moved from served to refused'
},
doors: surfaceOf(Brainy),
errors,
operators: {
accepted,
servedOnIndexPath: servedOnIndex,
refusedByIndexPath: refusedByIndex,
combinators: ['allOf', 'anyOf', 'not']
},
fieldAddressing: {
systemKeyPrefix: 'system.',
systemEntityScalars: [...(fieldAddressing.SYSTEM_ENTITY_SCALARS ?? [])].sort(),
systemRelationScalars: [...(fieldAddressing.SYSTEM_RELATION_SCALARS ?? [])].sort(),
plumbingFields: [...(fieldAddressing.PLUMBING_FIELDS ?? [])].sort()
},
health: {
verdicts: ['pass', 'warn', 'fail'],
healKinds: ['none', 'repair', 'rebuild'],
servingWithholdingInvariants: [
'index-initialized',
'durable-state-present',
'manifest-residency',
'replay-clean',
'strand-latch'
]
}
}
const rendered = `${JSON.stringify(manifest, null, 2)}\n`
if (process.argv.includes('--check')) {
if (!existsSync(OUT)) {
console.error(`docs/api-contract.json is missing — run: node scripts/emit-contract-manifest.mjs`)
process.exit(1)
}
if (readFileSync(OUT, 'utf-8') !== rendered) {
console.error(
`docs/api-contract.json is STALE — the public surface changed. Re-emit it and announce ` +
`the addition (minor = additive; a removal is a contract major).`
)
process.exit(1)
}
console.log(`docs/api-contract.json is current (${manifest.doors.length} doors, contract ${manifest.contractVersion}).`)
process.exit(0)
}
writeFileSync(OUT, rendered)
console.log(
`Wrote docs/api-contract.json — contract ${manifest.contractVersion}, ` +
`${manifest.doors.length} doors, ${manifest.errors.length} error classes, ` +
`${manifest.operators.accepted.length} operators ` +
`(${manifest.operators.refusedByIndexPath.length} refused by the index path).`
)

View file

@ -1,85 +0,0 @@
# Gate Guards
Two standalone scripts that stand between a test/build gate and a false
verdict: one refuses to let the gate start on a noisy machine, the other
refuses to let a truncated or crashed vitest run be read as green.
## Why these exist
Both guards exist because of the 2026-08-13 lost-day ledger: a gate ran on
a machine under load, and separately a vitest worker pool died mid-suite
while still printing a plausible-looking summary line, and in both cases
the bad result was trusted and acted on for the better part of a day before
anyone noticed. Neither failure mode announces itself — a loaded machine
still finishes and reports numbers, and a truncated test run still prints a
`Test Files` / `Tests` line — so both guards check the evidence explicitly
rather than trusting that a gate finishing means the gate was valid.
## gate-preflight.sh
Run before any gate lane starts. Exits 1 the moment the machine isn't
gate-clean, with one `FATAL:` line per violation naming the exact offender
(the pid and command, the path, the measured value). Prints one `OK:` line
per check that passes. `WARNING:` lines mark checks that were skipped, not
failures.
Checks:
| # | Check | Default threshold | Override |
|---|-------|--------------------|----------|
| a | 1-minute load average | `nproc / 2` | `GATE_MAX_LOAD` |
| b | any non-allowlisted process over 50% of one core | 50% | `GATE_ALLOW_REGEX` (extra pattern matched against the process's args) |
| c | cpu0 scaling governor must be `performance` | — | none (warns and skips if the sysfs path is absent) |
| d | free space on `/` and `/tmp` | 10G each | `GATE_SKIP_DISK_CHECK=1` to skip entirely |
The allowlist for check (b) is always: this script's own process tree
(its ancestors and its direct child processes), `sshd`, `systemd`, and
kernel threads (recognizable by args wrapped in brackets, e.g.
`[kworker/0:1]`). `GATE_ALLOW_REGEX` extends it — it does not replace it.
## vitest-verdict-check.sh
Run after every vitest lane, against that lane's captured log. Fails
loudly, quoting the exact line or string that tripped it, when the log's
own summary can't be trusted:
- no `Test Files` (or, in `--count-tests` mode, `Tests`) summary line is
present at all
- the parenthesized total in that line doesn't match what was expected
- fewer files/tests are accounted for (passed + failed + skipped) than the
total claims — a truncated run
- the log contains `Unhandled Error` or `Timeout calling` anywhere — a dead
worker pool, regardless of what the summary line claims
```
vitest-verdict-check.sh <log-file> <expected-file-count>
vitest-verdict-check.sh --count-tests <log-file> <minimum-test-count>
```
The first form checks `Test Files` for an exact match. The second checks
`Tests` for a minimum (a floor, not an exact count, since the total number
of individual tests moves more often than the number of test files).
## Wiring into a CI lane
```sh
# Before any lane that will report a verdict:
scripts/gate/gate-preflight.sh || exit 1
# Run the suite, capturing its output:
npx vitest run tests/unit 2>&1 | tee /tmp/unit.log
# After every vitest lane, check the log against the actual file count:
EXPECTED_FILES=$(ls tests/unit/**/*.test.ts | wc -l)
scripts/gate/vitest-verdict-check.sh /tmp/unit.log "$EXPECTED_FILES" || exit 1
```
## Exit-code contract
| Script | Exit 0 | Exit 1 |
|--------|--------|--------|
| `gate-preflight.sh` | machine is gate-clean | one or more `FATAL:` violations printed |
| `vitest-verdict-check.sh` | log's summary is trustworthy and matches | usage error, missing/unreadable log, or one or more `FATAL:` violations printed |
Non-zero from either script means: do not trust the gate that was about to
run, or the result of the one that just ran.

View file

@ -1,206 +0,0 @@
#!/bin/bash
set -euo pipefail
# Brainy Gate Preflight
# Refuses to let a test/build gate run on a machine that isn't clean enough
# to trust the numbers it produces. See scripts/gate/README.md for why (the
# 2026-08-13 lost-day ledger).
#
# Checks: 1-minute load average, any non-allowlisted process pinning a core,
# the cpu0 scaling governor, and free space on / and /tmp.
#
# Exit 0 and print one OK line per passing check when the machine is clean.
# Exit 1 and print one FATAL line per violation, naming the offender, when
# it is not.
#
# Known trap: a helper function whose last executed statement is a `while`
# (or any command whose own exit status happens to be nonzero) hands that
# status back as the function's return value. Called as a plain statement,
# that silently kills this script under `set -e`. Every helper below ends
# on an explicit `return 0` as its own statement, never on a loop or test.
#
# The same failure mode hides in plainer-looking lines too: `var=$(cmd)` is
# a bare assignment, so `set -e` DOES treat a nonzero `cmd` (or, under
# `pipefail`, a nonzero stage anywhere in `cmd`'s pipeline) as a failure of
# that statement and kills the script right there — even mid-loop, even
# when the "failure" is routine (a process that exited before a second
# lookup, a path that doesn't exist). Every such assignment below is paired
# with an explicit `|| var=""` fallback so a routine miss degrades to an
# empty value instead of an exit.
VIOLATIONS=0
ANCESTOR_PIDS=""
fatal() {
echo "FATAL: $1"
VIOLATIONS=$((VIOLATIONS + 1))
}
ok() {
echo "OK: $1"
}
# Walks this process's parent chain up to pid 1, then takes one snapshot of
# its direct children (the ps/read pipeline in check_processes), and
# records both in ANCESTOR_PIDS — so the process-scan below can recognize
# its own tree (the shell/terminal/session that launched it, plus its own
# helper commands) instead of flagging it. Children are captured once, up
# front, rather than re-queried per row later, so a helper command that has
# already exited by the time it's looked up can't be mistaken for a miss.
build_ancestor_pids() {
local pid="$$"
local ppid child
ANCESTOR_PIDS=" $pid "
while [ "$pid" != "1" ]; do
ppid=$(ps -o ppid= -p "$pid" 2>/dev/null | tr -d ' ') || ppid=""
if [ -z "$ppid" ]; then
break
fi
ANCESTOR_PIDS="${ANCESTOR_PIDS}${ppid} "
pid="$ppid"
done
while IFS= read -r child; do
[ -z "$child" ] && continue
ANCESTOR_PIDS="${ANCESTOR_PIDS}${child} "
done < <(ps --ppid "$$" -o pid= 2>/dev/null || true)
return 0
}
# (a) 1-minute load average vs. threshold (default: nproc / 2).
check_load() {
local max_load="${GATE_MAX_LOAD:-}"
if [ -z "$max_load" ]; then
max_load=$(( $(nproc) / 2 ))
if [ "$max_load" -lt 1 ]; then
max_load=1
fi
fi
local load_1m
load_1m=$(cut -d' ' -f1 /proc/loadavg)
if awk -v l="$load_1m" -v m="$max_load" 'BEGIN { exit !(l > m) }'; then
fatal "1-minute load average ${load_1m} exceeds threshold ${max_load} (GATE_MAX_LOAD=${max_load})"
else
ok "1-minute load average ${load_1m} is within threshold ${max_load}"
fi
return 0
}
# (b) any process outside the allowlist pinning more than half a core.
# Parsed with `read` into named fields, not an awk/cut chain — a fixed-column
# awk/cut split on `ps` output duplicated fields the first time this was
# tried, because process args vary in word count. `read` with a fixed list
# of variables dumps everything left over into the last one (args), which
# handles that correctly.
check_processes() {
local max_pcpu=50
local extra_regex="${GATE_ALLOW_REGEX:-}"
local violation_found=0
local line pcpu pid args pcpu_int
while IFS= read -r line; do
[ -z "$line" ] && continue
read -r pcpu pid args <<< "$line"
# Kernel threads report their comm in brackets, e.g. "[kworker/0:1]".
case "$args" in
\[*\]) continue ;;
esac
# This script's own tree: its ancestors (shell, terminal, session) and
# its direct children, both captured once by build_ancestor_pids.
case " $ANCESTOR_PIDS " in
*" $pid "*) continue ;;
esac
case "$args" in
*sshd*|*systemd*) continue ;;
esac
if [ -n "$extra_regex" ] && [[ "$args" =~ $extra_regex ]]; then
continue
fi
pcpu_int="${pcpu%.*}"
if [ -z "$pcpu_int" ]; then
pcpu_int=0
fi
if [ "$pcpu_int" -gt "$max_pcpu" ]; then
fatal "pid ${pid} ('${args}') is using ${pcpu}% of one core"
violation_found=1
fi
done < <(ps -eo pcpu,pid,args --sort=-pcpu | tail -n +2)
if [ "$violation_found" -eq 0 ]; then
ok "no process outside the allowlist exceeds ${max_pcpu}% of one core"
fi
return 0
}
# (c) cpu0 scaling governor must be "performance". Skipped with a warning
# (not a violation) when the sysfs path doesn't exist on this machine.
check_governor() {
local gov_path="/sys/devices/system/cpu/cpu0/cpufreq/scaling_governor"
if [ ! -r "$gov_path" ]; then
echo "WARNING: ${gov_path} not present; skipping governor check"
return 0
fi
local governor
governor=$(cat "$gov_path" 2>/dev/null) || governor=""
if [ "$governor" != "performance" ]; then
fatal "cpu0 governor is '${governor}', not 'performance'"
else
ok "cpu0 governor is 'performance'"
fi
return 0
}
# (d) free-space floors on / and /tmp (default 10G each). Skip entirely via
# GATE_SKIP_DISK_CHECK=1.
check_disk() {
if [ "${GATE_SKIP_DISK_CHECK:-0}" = "1" ]; then
echo "WARNING: disk free-space check skipped (GATE_SKIP_DISK_CHECK=1)"
return 0
fi
local floor_gb=10
local floor_bytes=$((floor_gb * 1024 * 1024 * 1024))
local path avail_bytes avail_gb
for path in / /tmp; do
avail_bytes=$(df --output=avail -B1 "$path" 2>/dev/null | tail -n 1 | tr -d ' ') || avail_bytes=""
if [ -z "$avail_bytes" ]; then
echo "WARNING: could not determine free space on ${path}; skipping"
continue
fi
if [ "$avail_bytes" -lt "$floor_bytes" ]; then
avail_gb=$((avail_bytes / 1024 / 1024 / 1024))
fatal "${path} has only ${avail_gb}G free, below the ${floor_gb}G floor"
else
ok "${path} has enough free space (floor ${floor_gb}G)"
fi
done
return 0
}
echo "Brainy gate preflight"
echo "----------------------"
build_ancestor_pids
check_load
check_processes
check_governor
check_disk
echo "----------------------"
if [ "$VIOLATIONS" -gt 0 ]; then
echo "FATAL: gate preflight failed with ${VIOLATIONS} violation(s) — machine is not gate-clean"
exit 1
fi
echo "gate preflight passed — machine is gate-clean"
exit 0

View file

@ -1,158 +0,0 @@
#!/bin/bash
set -euo pipefail
# Brainy Vitest Verdict Check
# Confirms a vitest run's own summary line is trustworthy before anything
# downstream treats a green run as green. See scripts/gate/README.md for why
# (the 2026-08-13 lost-day ledger).
#
# Usage:
# vitest-verdict-check.sh <log-file> <expected-file-count>
# vitest-verdict-check.sh --count-tests <log-file> <minimum-test-count>
#
# The first form checks the "Test Files" summary line's total against an
# exact expected count. The second checks the "Tests" summary line's total
# against a minimum. Both also fail on any sign the worker pool died
# mid-run, whether or not a summary line still made it into the log.
#
# Exit 0 and print one OK line per passing check when the log is clean.
# Exit 1 and print one FATAL line per violation, quoting the exact line or
# string that tripped it, when it is not.
#
# Known trap (shared with gate-preflight.sh): every helper below ends on an
# explicit `return 0` as its own statement, never on a loop or test, so a
# helper's last command can never hand its own exit status back as the
# function's under `set -e`. The same applies to `var=$(cmd)` assignments
# mid-helper: a bare assignment IS checked by `set -e`, so a `grep` that
# legitimately finds nothing (exit 1) would otherwise kill the script
# instead of just leaving the variable empty — every such assignment below
# is paired with an explicit `|| true` inside the substitution.
usage() {
echo "Usage: $0 <log-file> <expected-file-count>"
echo " $0 --count-tests <log-file> <minimum-test-count>"
exit 1
}
MODE="files"
if [ "${1:-}" = "--count-tests" ]; then
MODE="tests"
shift
fi
LOG_FILE="${1:-}"
THRESHOLD="${2:-}"
if [ -z "$LOG_FILE" ] || [ -z "$THRESHOLD" ]; then
usage
fi
if [ ! -f "$LOG_FILE" ]; then
echo "FATAL: log file '${LOG_FILE}' does not exist"
exit 1
fi
if ! [[ "$THRESHOLD" =~ ^[0-9]+$ ]]; then
echo "FATAL: threshold '${THRESHOLD}' is not a non-negative integer"
exit 1
fi
VIOLATIONS=0
fatal() {
echo "FATAL: $1"
VIOLATIONS=$((VIOLATIONS + 1))
}
ok() {
echo "OK: $1"
}
# Vitest colorizes its summary with ANSI escapes; strip them before parsing
# anything, or the color codes end up embedded in the fields we grep for.
CLEAN_LOG="$(sed 's/\x1b\[[0-9;]*m//g' "$LOG_FILE")"
# Worker-pool death: if either string appears, the run's own summary line —
# even if present and even if its numbers look fine — cannot be trusted,
# because the process died mid-suite and vitest's own accounting is what
# died with it.
check_worker_death() {
if echo "$CLEAN_LOG" | grep -q "Unhandled Error"; then
fatal "log contains 'Unhandled Error' — worker pool died mid-run"
fi
if echo "$CLEAN_LOG" | grep -q "Timeout calling"; then
fatal "log contains 'Timeout calling' — worker pool died mid-run"
fi
return 0
}
# Shared shape between the "Test Files" and "Tests" summary lines:
# <label> <n> passed | <n> failed | <n> skipped (<total>)
# `compare` is "eq" (total must equal threshold) or "min" (total must be at
# least threshold).
check_summary_line() {
local label="$1"
local threshold="$2"
local compare="$3"
local summary_line total accounted n
summary_line=$(echo "$CLEAN_LOG" | grep -E "^[[:space:]]*${label}[[:space:]]+" | tail -n 1 || true)
if [ -z "$summary_line" ]; then
fatal "no '${label}' summary line found in ${LOG_FILE}"
return 0
fi
total=$(echo "$summary_line" | grep -oE '\([0-9]+\)' | tr -d '()' | tail -n 1 || true)
if [ -z "$total" ]; then
fatal "'${label}' summary line has no parenthesized total: \"${summary_line}\""
return 0
fi
if [ "$compare" = "eq" ]; then
if [ "$total" -ne "$threshold" ]; then
fatal "'${label}' total is ${total}, expected ${threshold}: \"${summary_line}\""
else
ok "'${label}' total matches expected ${threshold}"
fi
else
if [ "$total" -lt "$threshold" ]; then
fatal "'${label}' total is ${total}, below minimum ${threshold}: \"${summary_line}\""
else
ok "'${label}' total ${total} meets minimum ${threshold}"
fi
fi
accounted=0
for n in $(echo "$summary_line" | grep -oE '[0-9]+ (passed|failed|skipped)' | grep -oE '^[0-9]+'); do
accounted=$((accounted + n))
done
if [ "$accounted" -lt "$total" ]; then
fatal "'${label}' line accounts for only ${accounted} of ${total} — truncated run: \"${summary_line}\""
else
ok "'${label}' line accounts for all ${total}"
fi
return 0
}
echo "Brainy vitest verdict check: ${LOG_FILE}"
echo "----------------------"
check_worker_death
if [ "$MODE" = "files" ]; then
check_summary_line "Test Files" "$THRESHOLD" "eq"
else
check_summary_line "Tests" "$THRESHOLD" "min"
fi
echo "----------------------"
if [ "$VIOLATIONS" -gt 0 ]; then
echo "FATAL: vitest verdict check failed with ${VIOLATIONS} violation(s) for ${LOG_FILE}"
exit 1
fi
echo "vitest verdict check passed for ${LOG_FILE}"
exit 0

View file

@ -1,118 +0,0 @@
/**
* Deterministic generation-stamp resolution for Brainy's build-time code
* generators.
*
* Two builds of the same source tree must produce byte-identical output.
* A wall-clock stamp (`new Date()`) breaks that guarantee, so every
* generator that writes a "Generated:" header or a `generatedAt` field
* into its output must resolve the stamp through this module instead.
*
* Resolution order:
* 1. The newest git commit timestamp among the generator's input files
* (the generator script itself always counts as an input).
* 2. If git metadata is unavailable (for example, building from a
* published npm tarball with no `.git` directory), the stamp already
* recorded in the previously generated output file.
* 3. If neither is available, the fixed epoch string
* `1970-01-01T00:00:00.000Z`.
*
* Every fallback logs a line to stderr deterministic degradation is
* loud, never a silent divergence.
*/
import { execFileSync } from 'child_process'
import * as fs from 'fs'
const EPOCH_STAMP = '1970-01-01T00:00:00.000Z'
const STAMP_PATTERN = /\*\s*Generated:\s*(\S+)/
/**
* Resolve the deterministic stamp for a generator run.
*
* @param inputPaths Absolute paths to every file whose content determines
* the generator's output, including the generator script itself.
* @param previousOutputPath Absolute path to the previously generated
* file, used for the existing-stamp fallback when git is unavailable.
* @returns An ISO-8601 timestamp string that is deterministic for a given
* source tree.
*/
export function resolveDeterministicStamp(
inputPaths: string[],
previousOutputPath: string
): string {
const gitStamp = newestGitCommitTimestamp(inputPaths)
if (gitStamp) {
return gitStamp
}
const existingStamp = readExistingStamp(previousOutputPath)
if (existingStamp) {
process.stderr.write(
`[deterministic-stamp] no git commit history found for generator inputs; ` +
`reusing existing stamp from ${previousOutputPath}: ${existingStamp}\n`
)
return existingStamp
}
process.stderr.write(
`[deterministic-stamp] no git commit history and no previous output at ` +
`${previousOutputPath}; falling back to fixed epoch stamp ${EPOCH_STAMP}\n`
)
return EPOCH_STAMP
}
/**
* Find the newest git commit timestamp among the given input paths.
* Returns null if git is unavailable, the tree is not a git repository,
* or none of the inputs have any commit history yet.
*/
function newestGitCommitTimestamp(inputPaths: string[]): string | null {
let newest: string | null = null
for (const inputPath of inputPaths) {
if (!fs.existsSync(inputPath)) {
continue
}
let out: string
try {
out = execFileSync(
'git',
['log', '-1', '--format=%cI', '--', inputPath],
{ stdio: ['ignore', 'pipe', 'ignore'] }
)
.toString()
.trim()
} catch {
// git missing, not a repository, or no permissions — handled by the
// caller's fallback chain.
continue
}
if (!out) {
// Path exists but has no commit history yet (e.g. newly created,
// uncommitted file).
continue
}
if (!newest || new Date(out).getTime() > new Date(newest).getTime()) {
newest = out
}
}
return newest
}
/**
* Parse the `* Generated: <ISO timestamp>` header out of a previously
* generated file, if one exists.
*/
function readExistingStamp(outputPath: string): string | null {
if (!fs.existsSync(outputPath)) {
return null
}
const content = fs.readFileSync(outputPath, 'utf-8')
const match = content.match(STAMP_PATTERN)
return match ? match[1] : null
}

View file

@ -15,12 +15,6 @@ NC='\033[0m' # No Color
RELEASE_TYPE="${1:-patch}" # patch, minor, or major
SKIP_TESTS=false
DRY_RUN=false
# --source-only is now a no-op: The Source is the one registry, so every
# release already ships Source-only — tag, CI's publish to The Source, the
# release page, and the docs push, with no separate storefront leg to skip.
# The flag is still accepted (for backward-compatible invocations) and just
# prints a notice; it no longer changes behavior.
SOURCE_ONLY=false
for arg in "$@"; do
case $arg in
@ -30,9 +24,6 @@ for arg in "$@"; do
--dry-run)
DRY_RUN=true
;;
--source-only)
SOURCE_ONLY=true
;;
esac
done
@ -109,7 +100,7 @@ else
;;
*)
echo -e "${RED}❌ Invalid release type: ${RELEASE_TYPE}${NC}"
echo "Usage: ./scripts/release.sh [patch|minor|major|<explicit-version>] [--dry-run] [--source-only (no-op; The Source is the one registry)]"
echo "Usage: ./scripts/release.sh [patch|minor|major|<explicit-version>] [--dry-run]"
exit 1
;;
esac
@ -128,9 +119,6 @@ echo -e "${BLUE}New version: ${NEW_VERSION}${NC}"
if [ "$PRERELEASE" = true ]; then
echo -e "${YELLOW}⚠️ Prerelease → npm dist-tag '${NPM_TAG}', GitHub prerelease${NC}"
fi
if [ "$SOURCE_ONLY" = true ]; then
echo -e "${YELLOW}⚠️ The Source is the one registry; --source-only is implied${NC}"
fi
echo ""
if [ "$DRY_RUN" = true ]; then
@ -154,7 +142,7 @@ else
fi
# Create new changelog entry
CHANGELOG_ENTRY="### [${NEW_VERSION}](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v${CURRENT_VERSION}...v${NEW_VERSION}) ($(date +%Y-%m-%d))
CHANGELOG_ENTRY="### [${NEW_VERSION}](https://github.com/soulcraftlabs/brainy/compare/v${CURRENT_VERSION}...v${NEW_VERSION}) ($(date +%Y-%m-%d))
${COMMITS}
"
@ -187,76 +175,42 @@ echo -e "${BLUE}7⃣ Creating git tag v${NEW_VERSION}...${NC}"
git tag -a "v${NEW_VERSION}" -m "Release v${NEW_VERSION}"
echo -e "${GREEN}✅ Tag created${NC}\n"
# Step 9: Push to origin — The Source is the one home (ruled 2026-07-23; the
# old public GitHub repo is archived history, no longer part of any release).
# TAG FIRST, branch second — deliberately two pushes: the runner is
# sequential, and a combined push can queue the release commit's ci.yml run
# AHEAD of the tag's publish-source run (observed on 10.0.0: the publish sat
# ~37 minutes behind a redundant CI run of the very commit the local gates
# had just proven). Pushing the tag alone queues the publish immediately;
# the branch push (and its ci.yml run) follows behind it, harmlessly.
echo -e "${BLUE}8⃣ Pushing to origin (tag first — the publish must never queue behind CI)...${NC}"
git push origin "v${NEW_VERSION}"
git push origin "$CURRENT_BRANCH"
echo -e "${GREEN}✅ Pushed to origin${NC}\n"
# Step 9: Push to GitHub
echo -e "${BLUE}8⃣ Pushing to GitHub...${NC}"
git push --follow-tags origin "$CURRENT_BRANCH"
echo -e "${GREEN}✅ Pushed to GitHub${NC}\n"
# Step 10: The home publish (The Source, source.soulcraft.com) is CI's job
# now, not the laptop's — a tag push (just above) triggers
# .forgejo/workflows/publish-source.yml, which builds and publishes on The
# Source's own runner (datacenter-side: seconds, not the laptop's WAN timing
# out on an 87MB tarball PUT). The laptop holds no home-registry publish
# credential anymore; it only waits for CI's result before continuing on to
# the release page and the docs push.
SOURCE_NPM_REG="https://source.soulcraft.com/api/packages/soulcraftlabs/npm/"
SOURCE_POLL_INTERVAL_S=15
SOURCE_POLL_MAX_ATTEMPTS=200 # 200 × 15s = 50 minutes — the runner is sequential and a busy day's ci.yml
# backlog has twice exceeded the old 20-minute window (8.10.3, 9.0.0);
# ci.yml no longer runs on tag pushes, but same-day branch pushes still queue ahead
echo -e "${BLUE}9⃣ Waiting for CI to publish v${NEW_VERSION} to The Source registry (home)...${NC}"
SOURCE_LANDED=false
for ((attempt = 1; attempt <= SOURCE_POLL_MAX_ATTEMPTS; attempt++)); do
LANDED_VERSION=$(npm view "@soulcraftlabs/brainy@${NEW_VERSION}" version "--@soulcraftlabs:registry=${SOURCE_NPM_REG}" 2>/dev/null || echo "")
if [ "$LANDED_VERSION" = "$NEW_VERSION" ]; then
SOURCE_LANDED=true
break
fi
echo -e "${YELLOW} … not yet on The Source (attempt ${attempt}/${SOURCE_POLL_MAX_ATTEMPTS}); retrying in ${SOURCE_POLL_INTERVAL_S}s${NC}"
sleep "$SOURCE_POLL_INTERVAL_S"
done
# Step 10: Publish to npm
echo -e "${BLUE}9⃣ Publishing to npm (dist-tag: ${NPM_TAG})...${NC}"
npm publish --tag "$NPM_TAG"
# Brainy is the only PUBLIC @soulcraft package — verify visibility after every publish.
npm access get status @soulcraft/brainy || true
echo -e "${GREEN}✅ Published to npm${NC}\n"
if [ "$SOURCE_LANDED" = true ]; then
echo -e "${GREEN}✅ CI published v${NEW_VERSION} to The Source${NC}\n"
# Step 11: Create GitHub release
echo -e "${BLUE}🔟 Creating GitHub release...${NC}"
if [ "$PRERELEASE" = true ]; then
gh release create "v${NEW_VERSION}" --generate-notes --prerelease
else
echo -e "${RED}❌ CI's home publish did not land — check the workflow run on The Source; the pair must not diverge.${NC}"
echo -e "${RED} v${NEW_VERSION} was tagged and pushed, but @soulcraftlabs/brainy@${NEW_VERSION} never became visible on the${NC}"
echo -e "${RED} Source registry after ${SOURCE_POLL_MAX_ATTEMPTS} attempts, ${SOURCE_POLL_INTERVAL_S}s apart. Aborting.${NC}"
exit 1
gh release create "v${NEW_VERSION}" --generate-notes
fi
echo -e "${GREEN}✅ GitHub release created${NC}\n"
# Step 11: Release object on The Source (presentational — the tag, CHANGELOG,
# and RELEASES.md are the record; this just gives The Source's UI a release page).
echo -e "${BLUE}🔟 Creating release page on The Source...${NC}"
if [ -n "${FORGEJO_RELEASE_TOKEN:-}" ]; then
if curl -sf -X POST "https://source.soulcraft.com/api/v1/repos/soulcraftlabs/open-brainy/releases" \
-H "Authorization: token ${FORGEJO_RELEASE_TOKEN}" -H "Content-Type: application/json" \
-d "{\"tag_name\":\"v${NEW_VERSION}\",\"name\":\"v${NEW_VERSION}\",\"prerelease\":${PRERELEASE}}" >/dev/null; then
echo -e "${GREEN}✅ Release page created on The Source${NC}\n"
else
echo -e "${RED}⚠️ Release-page API call failed — tag + CHANGELOG remain the record; create the page via The Source's UI if wanted${NC}\n"
fi
# Step 12: Push public docs to the soulcraft.com docs ingest door
# (VENUE-DOCS-RELEASE-PUSH). Skips with a loud warning when
# DOCS_INGEST_SECRET is unset; fails loudly (without undoing the publish —
# that already happened) when a push errors, so the docs site never
# silently trails npm.
echo -e "${BLUE}1⃣2⃣ Pushing public docs to soulcraft.com/docs...${NC}"
if node scripts/push-docs.js; then
echo -e "${GREEN}✅ Docs push step done${NC}\n"
else
echo -e "${RED}⚠️ FORGEJO_RELEASE_TOKEN unset — no release page created; tag + CHANGELOG remain the record${NC}\n"
echo -e "${RED}❌ Docs push FAILED — soulcraft.com/docs trails npm until re-run or interim sync${NC}\n"
fi
# Step 12 RETIRED (2026-08-31, CORTEX-SITE-BRAINY-RENAME round 12, David-ruled):
# soulcraft.com/docs carries the paid product's documentation only. This
# engine's documentation home is THIS repository — README and docs/ — and the
# site serves 301s for the slugs this rail used to push. The push script stays
# in the tree for history; the rail no longer calls it.
echo -e "${BLUE}Docs step: this engine documents itself in its own repo (site push retired 2026-08-31)${NC}"
echo -e "${GREEN}━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━${NC}"
echo -e "${GREEN}🎉 Release ${NEW_VERSION} complete!${NC}"
echo -e "${GREEN}━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━${NC}"
echo ""
echo -e "🏠 The Source: ${BLUE}https://source.soulcraft.com/soulcraftlabs/open-brainy/releases/tag/v${NEW_VERSION}${NC}"
echo -e "📦 npm: ${BLUE}https://www.npmjs.com/package/@soulcraft/brainy/v/${NEW_VERSION}${NC}"
echo -e "🐙 GitHub: ${BLUE}https://github.com/soulcraftlabs/brainy/releases/tag/v${NEW_VERSION}${NC}"

View file

@ -14,22 +14,7 @@
*/
import type { StorageAdapter, HNSWNounWithMetadata } from '../coreTypes.js'
import { parseFieldAddress, readEntityFieldAddress } from '../db/fieldAddressing.js'
import type { HNSWNounWithMetadata as AddressedEntity } from '../coreTypes.js'
/**
* Read a user-supplied field name under the one addressing law (sealed
* 2026-08-03): bare / `metadata.` = the user's metadata field, `system.<x>` =
* the ruled engine scalar, malformed = typed refusal. The aggregation engine
* NEVER resolves names any other way the pre-law resolver made bare
* `subtype`/`confidence` read engine scalars, silently shadowing user fields.
*/
function readAddressed(e: unknown, name: string): unknown {
return readEntityFieldAddress(
e as AddressedEntity,
parseFieldAddress(name, 'entity')
)
}
import { resolveEntityField } from '../coreTypes.js'
import type {
AggregateDefinition,
AggregateGroupState,
@ -110,15 +95,11 @@ function matchesSource(entity: Record<string, unknown>, source: AggregateDefinit
// live in the custom bag, so those filters could never match anything.
if (source.where && Object.keys(source.where).length > 0) {
const e = entity as unknown as HNSWNounWithMetadata
for (const [key, condition] of Object.entries(source.where)) {
// Evaluate ONE field at a time under a neutral key: the address may be
// dotted ('system.subtype'), and the filter evaluator would otherwise
// walk dots as a nested path instead of treating the key as an address.
const value = readAddressed(e, key)
if (!matchesMetadataFilter({ v: value }, { v: condition } as Record<string, unknown>)) {
return false
}
const resolved: Record<string, unknown> = {}
for (const key of Object.keys(source.where)) {
resolved[key] = resolveEntityField(e, key)
}
if (!matchesMetadataFilter(resolved, source.where)) return false
}
return true
@ -148,11 +129,11 @@ function computeGroupKeys(
for (const dim of groupBy) {
if (typeof dim === 'string') {
const val = readAddressed(e, dim)
const val = resolveEntityField(e, dim)
const v = val !== undefined && val !== null ? String(val) : '__null__'
for (const k of keys) k[dim] = v
} else if ('unnest' in dim) {
const val = readAddressed(e, dim.field)
const val = resolveEntityField(e, dim.field)
const raw = Array.isArray(val) ? val : val !== undefined && val !== null ? [val] : []
// Distinct elements: an entity with duplicate tags counts once per distinct tag.
const elems = Array.from(new Set(raw.map(x => String(x))))
@ -164,7 +145,7 @@ function computeGroupKeys(
keys = next
} else {
// Time-windowed field
const val = readAddressed(e, dim.field)
const val = resolveEntityField(e, dim.field)
const v = typeof val === 'number' ? bucketTimestamp(val, dim.window) : '__null__'
for (const k of keys) k[dim.field] = v
}
@ -193,7 +174,7 @@ function computeGroupKey(
* in metadata are both handled in one place.
*/
function getNumericField(entity: Record<string, unknown>, field: string): number | undefined {
const val = readAddressed(entity as unknown as HNSWNounWithMetadata, field)
const val = resolveEntityField(entity as unknown as HNSWNounWithMetadata, field)
if (typeof val === 'number' && !isNaN(val)) return val
if (typeof val === 'string') {
const num = parseFloat(val)
@ -371,15 +352,6 @@ export class AggregationIndex {
*/
private pendingAdopt = new Set<string>()
/**
* Aggregates adopted with a BEHIND stamp: name the exact generation
* window `(from, to]` whose writes the adopted state has not seen. The
* owner (Brainy) drains this via {@link getPendingCatchUps} +
* {@link reconcileEntity} + {@link finishCatchUp} BEFORE serving queries
* cost bounded by the window's affected entities, never store size.
*/
private pendingCatchUp = new Map<string, { from: number; to: number }>()
/**
* In-flight rescan targets. While a name has a staging map, ALL
* contributions (the walk's and concurrent write hooks') land there instead
@ -446,47 +418,25 @@ export class AggregationIndex {
}
/**
* The adoption verdict for persisted state, against the store's committed
* watermark (SELF-ENGINE-LIFECYCLE-SPRINT ask (b) behind-stamp is no
* longer a whole-store rescan):
*
* - `'adopt'` stamp equals the watermark (clean), or the store has no
* watermark capability (hash-only adoption, the pre-stamp behavior).
* - `'catchup'` stamp is BEHIND the watermark (an unclean exit after
* later writes, or a long-lived writer whose last flush predates recent
* writes). The state is exact AS OF its stamp, so it is adopted and the
* missing window `(stamp, committed]` is reconciled INCREMENTALLY per
* affected entity via time-travel reads bounded by writes since the
* last flush, never by store size. The owner drains
* {@link getPendingCatchUps} before serving queries.
* - `'rescan'` no stamp (pre-stamp state on a stamped store) or stamp
* AHEAD of the watermark (e.g. a fact-log truncation on a copied store
* pulled the watermark back): the state over-counts unverifiably; one
* exact rescan, said out loud.
* May this persisted state be ADOPTED? When the store exposes its committed
* watermark, the state's `sourceGeneration` must EQUAL it: behind means
* later writes are missing from the state (unclean shutdown); ahead means
* it counts writes that no longer exist (e.g. a fact-log truncation on a
* copied store pulled the watermark back). Either way: one exact rescan,
* said out loud never a silent adopt. Stores without the capability (and
* pre-stamp state on them) fall back to hash-only adoption.
*/
private stateAdoptionVerdict(
name: string,
stateData: unknown
): 'adopt' | 'catchup' | 'rescan' {
private stateGenerationAdoptable(name: string, stateData: unknown): boolean {
const committed = this.storage.committedGeneration?.() ?? null
if (committed === null) return 'adopt'
if (committed === null) return true
const raw = (stateData as Record<string, unknown>).sourceGeneration
const stamped = typeof raw === 'number' ? raw : null
if (stamped === committed) return 'adopt'
if (stamped !== null && stamped < committed) {
this.pendingCatchUp.set(name, { from: stamped, to: committed })
prodLog.info(
`[Aggregation] '${name}': persisted state is at generation ${stamped}, store is at ` +
`${committed} — adopting and reconciling the ${committed - stamped}-generation window ` +
`incrementally (no store rescan)`
)
return 'catchup'
}
if (stamped === committed) return true
prodLog.warn(
`[Aggregation] '${name}': persisted state is at generation ${stamped ?? 'unstamped'} ` +
`but the store's committed generation is ${committed} — rescanning instead of adopting`
)
return 'rescan'
return false
}
private async loadPersisted(): Promise<void> {
@ -507,21 +457,20 @@ export class AggregationIndex {
const appHash = this.definitionHashes.get(def.name) || ''
if (appHash === savedHash && this.pendingAdopt.has(def.name)) {
const stateData = await this.storage.getMetadata(`${STATE_KEY_PREFIX}${def.name}__`)
const verdict =
stateData && stateData.groups
? this.stateAdoptionVerdict(def.name, stateData)
: 'rescan'
if (verdict !== 'rescan') {
if (
stateData &&
stateData.groups &&
this.stateGenerationAdoptable(def.name, stateData)
) {
const groupMap = new Map<string, AggregateGroupState>()
for (const group of stateData!.groups as AggregateGroupState[]) {
for (const group of stateData.groups as AggregateGroupState[]) {
groupMap.set(serializeGroupKey(group.groupKey), group)
}
this.states.set(def.name, groupMap)
this.pendingAdopt.delete(def.name)
this.needsBackfill.delete(def.name)
prodLog.info(
`[Aggregation] '${def.name}': adopted persisted state (${groupMap.size} groups) — ` +
(verdict === 'catchup' ? 'incremental catch-up pending' : 'no rescan')
`[Aggregation] '${def.name}': adopted persisted state (${groupMap.size} groups) — no rescan`
)
}
// No/invalid persisted state: stays in pendingAdopt and resolves
@ -536,23 +485,22 @@ export class AggregationIndex {
const currentHash = hashDefinition(def)
const stateData = await this.storage.getMetadata(`${STATE_KEY_PREFIX}${def.name}__`)
const restoreVerdict =
stateData && stateData.groups && savedHash === currentHash
? this.stateAdoptionVerdict(def.name, stateData)
: 'rescan'
if (restoreVerdict !== 'rescan') {
// Definition unchanged — load state (exact as of its stamp; a
// 'catchup' verdict reconciles the missing window incrementally).
if (
stateData &&
stateData.groups &&
savedHash === currentHash &&
this.stateGenerationAdoptable(def.name, stateData)
) {
// Definition unchanged — load state
const groupMap = new Map<string, AggregateGroupState>()
for (const group of stateData!.groups as AggregateGroupState[]) {
for (const group of stateData.groups as AggregateGroupState[]) {
const serialized = serializeGroupKey(group.groupKey)
groupMap.set(serialized, group)
}
this.states.set(def.name, groupMap)
this.needsBackfill.delete(def.name)
prodLog.info(
`[Aggregation] '${def.name}': restored definition + adopted persisted state (${groupMap.size} groups)` +
(restoreVerdict === 'catchup' ? ' — incremental catch-up pending' : '')
`[Aggregation] '${def.name}': restored definition + adopted persisted state (${groupMap.size} groups)`
)
} else {
// Definition changed or no saved state — start fresh and backfill from
@ -570,35 +518,15 @@ export class AggregationIndex {
}
}
// Restore native provider state from persistence — GATED by the same
// adoption verdict as caller-side state (the unconditional adopt was an
// asymmetry: a stale native blob restored over a moved store silently
// over/under-counted). 'adopt' restores; 'catchup' restores too (the
// incremental reconciliation drives the provider through
// incrementalUpdate over the exact missing window); 'rescan' SKIPS the
// blob — the flagged rebuild repopulates the provider from source.
// Legacy unstamped envelopes verdict as rescan, loudly, never silently.
// Restore native provider state from persistence
if (this.nativeProvider?.restoreState) {
const nativeState = await this.storage.getMetadata('__aggregation_native_state__')
const blob =
nativeState && typeof nativeState === 'string'
? nativeState
: nativeState && typeof nativeState === 'object' && nativeState.data
? (nativeState.data as string)
: null
if (blob !== null) {
const verdict = this.stateAdoptionVerdict(
'__native__',
nativeState && typeof nativeState === 'object' ? (nativeState as Record<string, unknown>) : {}
)
if (verdict === 'adopt' || verdict === 'catchup') {
this.nativeProvider.restoreState(blob)
} else {
prodLog.warn(
`[Aggregation] native provider state not adopted (verdict: ${verdict}) — ` +
`the flagged rescan repopulates the provider from source`
)
}
if (nativeState && typeof nativeState === 'string') {
this.nativeProvider.restoreState(nativeState)
} else if (nativeState && typeof nativeState === 'object' && nativeState.data) {
// flush() persists `{ data: serializeState() }`, so `data` is the
// provider's serialized state string.
this.nativeProvider.restoreState(nativeState.data as string)
}
}
}
@ -634,17 +562,12 @@ export class AggregationIndex {
}
}
// Persist native provider state — stamped. noteSourceGeneration lets the
// provider bake the committed watermark into its OWN envelope before
// serializing (so a native-side reopen can verify honesty without our
// wrapper); the wrapper carries the same stamp for OUR adoption verdict.
// Persist native provider state
if (this.nativeProvider?.serializeState) {
const nativeGen = this.storage.committedGeneration?.() ?? null
if (nativeGen !== null) this.nativeProvider.noteSourceGeneration?.(nativeGen)
const nativeState = this.nativeProvider.serializeState()
await this.storage.saveMetadata(
'__aggregation_native_state__',
nativeGen === null ? { data: nativeState } : { data: nativeState, sourceGeneration: nativeGen }
{ data: nativeState }
)
}
@ -805,119 +728,6 @@ export class AggregationIndex {
this.dirty.add(name)
}
// ============= Incremental Catch-Up (behind-stamp adoption) =============
/** The aggregates adopted behind the watermark, with their exact missing windows. */
getPendingCatchUps(): Array<{ name: string; from: number; to: number }> {
return Array.from(this.pendingCatchUp, ([name, w]) => ({ name, ...w }))
}
/**
* Reconcile ONE entity's contribution across a catch-up window using the
* same exact delta algebra the write-time hooks use: remove the
* contribution the adopted state counted (the entity AS OF the stamp),
* add the contribution it should count (AS OF the window's end). `null`
* on either side means the entity did not exist then. Composes exactly
* with live hooks because every application is a precise old/new pair
* order between catch-up and post-window writes cannot drift the totals.
*/
reconcileEntity(
name: string,
id: string,
before: Record<string, unknown> | null,
after: Record<string, unknown> | null
): void {
const def = this.definitions.get(name)
if (!def) return
if (before && after) {
if (isAggregateEntity(after)) return
const oldMatches = matchesSource(before, def.source)
const newMatches = matchesSource(after, def.source)
if (this.nativeProvider && (oldMatches || newMatches)) {
this.applyNativeResults(
name,
this.nativeProvider.incrementalUpdate(name, def, after, 'update', before)
)
return
}
if (oldMatches) this.removeContribution(name, def, before)
if (newMatches) this.addContribution(name, def, after)
return
}
if (after) {
if (isAggregateEntity(after) || !matchesSource(after, def.source)) return
if (this.nativeProvider) {
this.applyNativeResults(name, this.nativeProvider.incrementalUpdate(name, def, after, 'add'))
} else {
this.addContribution(name, def, after)
}
return
}
if (before) {
if (isAggregateEntity(before) || !matchesSource(before, def.source)) return
if (this.nativeProvider) {
this.applyNativeResults(name, this.nativeProvider.incrementalUpdate(name, def, before, 'delete'))
} else {
this.removeContribution(name, def, before)
}
}
}
/** Whether the native provider offers the parallel whole-rebuild path. */
hasProviderRebuild(): boolean {
return typeof this.nativeProvider?.rebuildAggregate === 'function'
}
/** The catch-up window for `name` is fully reconciled; state is current. */
finishCatchUp(name: string): void {
this.pendingCatchUp.delete(name)
this.dirty.add(name)
}
/**
* A catch-up could not complete (window unreadable, affected set over the
* bound, ): demote to an exact rescan, loudly never serve un-reconciled.
*/
demoteCatchUpToBackfill(name: string, reason: string): void {
this.pendingCatchUp.delete(name)
this.needsBackfill.add(name)
prodLog.warn(`[Aggregation] '${name}': catch-up demoted to full rescan — ${reason}`)
}
/**
* Rebuild an aggregate through the native provider's parallel path
* (SELF-ENGINE-LIFECYCLE-SPRINT ask (c) `rebuildAggregate` existed on
* the provider contract but was never invoked; the JS walk fed
* per-entity FFI calls instead). Returns false when no provider rebuild
* exists the caller streams the JS walk as before.
*/
rebuildWithProvider(name: string, entities: Array<Record<string, unknown>>): boolean {
const def = this.definitions.get(name)
if (!def || !this.nativeProvider?.rebuildAggregate) return false
const rebuilt = this.nativeProvider.rebuildAggregate(
def,
entities.filter(e => !isAggregateEntity(e) && matchesSource(e, def.source))
)
this.states.set(name, rebuilt)
this.backfillStaging.delete(name)
this.needsBackfill.delete(name)
this.dirty.add(name)
return true
}
/**
* A write-path hook could not see the entity it needed (e.g. a delete
* whose before-image was unavailable): flag EVERY defined aggregate for
* an exact rescan, loudly the counts must never silently drift
* (SELF-ENGINE-LIFECYCLE-SPRINT ask (d): the gated hook used to SKIP).
*/
flagAllForRescan(reason: string): void {
for (const name of this.definitions.keys()) this.needsBackfill.add(name)
prodLog.warn(
`[Aggregation] all ${this.definitions.size} aggregate(s) flagged for rescan — ${reason}`
)
}
// ============= Write-Time Hooks =============
/**
@ -1180,7 +990,7 @@ export class AggregationIndex {
// distinctCount tracks distinct values of ANY type (strings, numbers, booleans),
// keyed by their string form — NOT numeric-coerced, since its primary use is
// categorical (distinct categories / users / tags), not numeric columns.
const raw = readAddressed(entity as unknown as HNSWNounWithMetadata, metricDef.field!)
const raw = resolveEntityField(entity as unknown as HNSWNounWithMetadata, metricDef.field!)
if (raw !== undefined && raw !== null) {
if (!state.valueCounts) state.valueCounts = {}
const key = String(raw)
@ -1224,7 +1034,7 @@ export class AggregationIndex {
state.count = Math.max(0, state.count - 1)
state.sum = Math.max(0, state.sum - 1)
} else if (metricDef.op === 'distinctCount') {
const raw = readAddressed(entity as unknown as HNSWNounWithMetadata, metricDef.field!)
const raw = resolveEntityField(entity as unknown as HNSWNounWithMetadata, metricDef.field!)
if (raw !== undefined && raw !== null && state.valueCounts) {
const key = String(raw)
const c = state.valueCounts[key]

File diff suppressed because it is too large Load diff

View file

@ -284,12 +284,7 @@ export const STANDARD_ENTITY_FIELDS: ReadonlySet<string> = new Set([
'id',
'vector',
'connections',
// 'level' is deliberately ABSENT: it is HNSW plumbing, not an entity field.
// Listing it here made every by-name read of a user metadata field called
// `level` resolve to the engine's internal node layer instead — a silent
// shadow that broke sort/filter/aggregation on a perfectly natural field
// name (VENUE-BRAINY-ORDERBY-NOOP). Engine plumbing is invisible to the
// query surface; a bare `level` reads `entity.metadata.level`.
'level',
'type',
'subtype',
'visibility',
@ -792,30 +787,6 @@ export interface DerivedFamilyDeclaration {
rebuildable?: boolean
}
/**
* @description The canonical count ledger a storage adapter maintains on its
* write path: per family, the user-facing `counted` scalar and the
* ALL-visibility `all` scalar (every tier the coverage-ledger denominator).
* See {@link StorageAdapter.getCanonicalCounts}.
*/
export interface CanonicalCounts {
nouns: { counted: number; all: number }
verbs: { counted: number; all: number }
/**
* The count of canonical nouns holding a REAL (non-empty) vector the
* coverage denominator a vector index's node-count ledger is measured
* against (`nodeCount === vectors.all` is the whole-store coverage
* verdict for the vector leg, the vector-side mirror of `nouns.all` for
* metadata/graph). A deferred-embed noun (`add({ deferEmbedding: true })`)
* counts only once its vector actually LANDS its canonical record exists
* (counted in `nouns.all`) with an empty vector until then, so it is
* deliberately NOT counted here in the interim.
*/
vectors: { all: number }
/** An unprovable delete has left the `all` scalars unverified since the last recount. */
suspect: boolean
}
export interface StorageAdapter {
init(): Promise<void>
@ -830,68 +801,14 @@ export interface StorageAdapter {
* Save noun metadata separately
* @param id Noun ID
* @param metadata Noun metadata
* @param hasVector - OPTIONAL vectored-noun ledger hint: `true` when this
* write is a FRESH insert (`isNew`) whose vector is a real, non-empty
* array the caller already knows this for free (the insert's own
* `vector` local), so the increment rides the SAME isNew gate that
* already protects `totalNounCountAll` from double-counting on HNSW
* neighbor-link re-saves (`saveNoun_internal` re-runs on every link
* change; this metadata seam does not). Absent/`false` no ledger
* action. A deferred-embed insert passes `false` (its vector lands
* later see {@link StorageAdapter.noteVectorLanded}).
*/
saveNounMetadata(id: string, metadata: NounMetadata, hasVector?: boolean): Promise<void>
saveNounMetadata(id: string, metadata: NounMetadata): Promise<void>
/**
* Delete noun metadata
* @param id Noun ID
* @param priorRecord - OPTIONAL already-known metadata (the caller's
* pre-delete read) see {@link StorageAdapter.deleteNoun}.
* @param hadVector - OPTIONAL vectored-noun ledger hint: `true`/`false`
* when the caller already knows (read as a side effect of ITS OWN delete
* flow e.g. `remove()`'s pre-read for the vector-index removal never
* a read added FOR this ledger), `undefined` when genuinely unknown. A
* known `true` decrements the vectored-noun ledger; a known `false` is a
* no-op (it was never counted); `undefined` marks the ledger SUSPECT
* rather than guessing the delete path must never add a canonical read
* to answer this question.
*/
deleteNounMetadata(id: string, priorRecord?: NounMetadata | null, hadVector?: boolean): Promise<void>
/**
* OPTIONAL narrow ledger hook: record that a canonical noun's vector just
* LANDED for the first time. Exists ONLY for the deferred-embedding
* lifecycle the landing commit (`system:embed-landing`) carries a vector
* write with no accompanying metadata operation, so the normal
* `saveNounMetadata(..., hasVector)` seam never fires for it. Callers MUST
* call this only when the noun held NO real vector before this write (the
* deferred-embed worker already holds that fact for free, from its own
* pre-embed read never an added read). A backend without vectored-noun
* tracking is a no-op via this method's absence (feature-detected).
* @param id - The noun whose vector just landed.
*/
noteVectorLanded?(id: string): Promise<void>
/**
* OPTIONAL narrow ledger hook, the mirror of {@link noteVectorLanded}:
* record that a canonical noun's vector was just REMOVED rewritten from
* a real (non-empty) vector to the "unvectored" empty-array shape. Exists
* for the ONE sanctioned reverse migration this engine supports: the VFS
* root's zero-norm fix (see `VirtualFileSystem.doInitializeRoot()` and
* `Brainy.unvectorNounForRootMigration()`), which rewrites a pre-fix
* store's all-zero placeholder root vector to `[]` and must decrement
* `vectors.all` through this hook so the coverage ledger never drifts.
* NOT a general-purpose "I removed a vector" callback ordinary
* application data has no sanctioned path from vectored back to
* unvectored (`update()` refuses an empty vector as a dimension
* mismatch by design). Callers MUST call this only when the noun held a
* REAL vector immediately before this write (the caller already holds
* that fact for free, from its own pre-write read never an added read).
* A backend without vectored-noun tracking is a no-op via this method's
* absence (feature-detected).
* @param id - The noun whose vector was just removed.
*/
noteVectorUnlanded?(id: string): Promise<void>
deleteNounMetadata(id: string): Promise<void>
/**
* Get noun with metadata combined
@ -940,11 +857,8 @@ export interface StorageAdapter {
* REQUIRE re-reading the record being removed: when the internal read
* returns `null` (replace race, or a ghost left by an earlier version) the
* decrement falls back to this record instead of being silently skipped.
* @param hadVector OPTIONAL vectored-noun ledger hint see
* {@link StorageAdapter.deleteNounMetadata}'s `hadVector` param, which
* this forwards to unchanged.
*/
deleteNoun(id: string, priorMetadata?: NounMetadata | null, hadVector?: boolean): Promise<void>
deleteNoun(id: string, priorMetadata?: NounMetadata | null): Promise<void>
/**
* Save verb - Pure HNSW verb with core fields only
@ -1374,19 +1288,6 @@ export interface StorageAdapter {
*/
getVerbCount(): Promise<number>
/**
* The canonical count ledger O(1), no I/O. `counted` mirrors
* `getNounCount()` / `getVerbCount()` (public + internal tiers); `all` is
* the ALL-visibility scalar every unfiltered storage walk is measured
* against the denominator a derived-index provider's coverage ledger
* subtracts from. `suspect` is `true` when an unprovable delete has left
* `all` unverified since the last sanctioned recount (`repairIndex()`).
* Optional: adapters without the ledger omit it; a consumer treats absence
* as "no denominator", never as zero.
* @returns Both scalars per family plus the suspect flag.
*/
getCanonicalCounts?(): Promise<CanonicalCounts>
/**
* OPTIONAL create a pre-upgrade backup of the whole store and return its
* location, or `null` when there is nothing to back up (empty store). On the

View file

@ -59,10 +59,14 @@ import type {
import type { StorageAdapter } from '../coreTypes.js'
import { exportGraph } from './portableGraph.js'
import type { ExportSelector, ExportOptions, PortableGraph } from './portableGraph.js'
import {
splitNounMetadataRecord,
splitVerbMetadataRecord
} from '../types/reservedFields.js'
import { v4 as uuidv4 } from '../universal/uuid.js'
import { coerceNewEntityId, resolveEntityId, ORIGINAL_ID_KEY } from '../utils/idNormalization.js'
import { EntityNotFoundError } from '../errors/notFound.js'
import { SpeculativeOverlayError, CanonicalEnumerationUnavailableError } from './errors.js'
import { SpeculativeOverlayError } from './errors.js'
import type { GenerationStore } from './generationStore.js'
import type { ChangedIds, TransactReceipt, TxOperation } from './types.js'
import { entityMatchesFind, resolveEntityField, UnsupportedWhereOperatorError } from './whereMatcher.js'
@ -516,37 +520,14 @@ export class Db<T = any> {
* (no generation history) distinct from `persist()` (native whole-brain snapshot
* that preserves history). Restore with `brain.import(backup)`.
*
* `options.enumeration: 'canonical'` (default: `'index'`) walks the storage
* adapter's canonical noun/verb layout directly instead of the metadata/graph
* indexes, guaranteeing canon-completeness against index corruption see
* {@link ExportOptions.enumeration}. It requires the LIVE, current-generation
* view: called on a historical `asOf()` pin or a speculative `with()` overlay it
* throws {@link CanonicalEnumerationUnavailableError} rather than silently mixing
* generations or missing the overlay's own entities.
*
* `options.includeHidden: true` admits BOTH hidden visibility tiers
* (`'internal'` and `'system'`) into a whole-brain/predicate export, in EITHER
* `enumeration` mode see {@link ExportOptions.includeHidden}. Migration-grade
* exports set this; consumer-facing exports leave it off (default: false).
*
* @param selector - WHAT to export (omit for the whole brain). See {@link ExportSelector}.
* @param options - HOW to export (vectors / VFS bytes / edge policy / enumeration mode). See {@link ExportOptions}.
* @param options - HOW to export (vectors / VFS bytes / edge policy). See {@link ExportOptions}.
* @returns A versioned, portable `PortableGraph` document.
* @throws {@link CanonicalEnumerationUnavailableError} if `enumeration:'canonical'` is
* requested on a historical or speculative-overlay view.
* @example
* const backup = await brain.now().export({ collection: id }, { includeVectors: true })
* @example
* // Canon-complete audit export, with an index-drift report attached.
* const audit = await brain.now().export({}, { enumeration: 'canonical', reportIndexDrift: true })
* if (audit.drift) console.log(audit.drift.canonicalOnly, audit.drift.indexOnly)
*/
async export(selector: ExportSelector = {}, options: ExportOptions = {}): Promise<PortableGraph> {
this.assertUsable('export')
if (options.enumeration === 'canonical') {
if (this.overlay) throw new CanonicalEnumerationUnavailableError(this.gen, 'overlay')
if (this.isHistorical()) throw new CanonicalEnumerationUnavailableError(this.gen, 'historical')
}
return exportGraph(this, this.host.storage, selector, options)
}
@ -701,15 +682,23 @@ export class Db<T = any> {
for (const op of ops) {
switch (op.op) {
case 'add': {
// Field-addressing law: the metadata bag is the user's, VERBATIM —
// no reserved-name lift, no drops. Engine scalars come ONLY from
// their dedicated op fields; a bag field named `confidence` is an
// ordinary user field, exactly as on the committed write path.
const custom = { ...(op.metadata as Record<string, unknown> | undefined) }
const confidence = op.confidence
const weight = op.weight
const subtype = op.subtype
const service = op.service
// Reserved-field normalization — mirror of the brain.transact()
// write path: user-settable fields lift to their dedicated field
// (top-level wins), system-managed fields drop, and the entity's
// metadata bag carries ONLY custom fields. Speculative views skip
// the one-shot warnings — committing the same ops through
// `brain.transact()` warns on the real write path.
const { reserved, custom } = splitNounMetadataRecord(
op.metadata as Record<string, unknown> | undefined
)
const confidence =
op.confidence ?? (typeof reserved.confidence === 'number' ? reserved.confidence : undefined)
const weight =
op.weight ?? (typeof reserved.weight === 'number' ? reserved.weight : undefined)
const subtype =
op.subtype ?? (typeof reserved.subtype === 'string' ? reserved.subtype : undefined)
const service =
op.service ?? (typeof reserved.service === 'string' ? reserved.service : undefined)
// Id normalization (8.0) — mirror of the committed transact() add
// path: a natural key coerces to a STABLE UUID (v5), preserving the
@ -747,12 +736,16 @@ export class Db<T = any> {
`with(): entity ${updateId} not found at generation ${this.gen}`
)
}
// Field-addressing law — mirror of the add case: the patch bag is
// the user's verbatim; engine scalars only from dedicated op fields.
const custom = { ...(op.metadata as Record<string, unknown> | undefined) }
const confidence = op.confidence
const weight = op.weight
const subtype = op.subtype
// Same reserved-field normalization as the committed update path.
const { reserved, custom } = splitNounMetadataRecord(
op.metadata as Record<string, unknown> | undefined
)
const confidence =
op.confidence ?? (typeof reserved.confidence === 'number' ? reserved.confidence : undefined)
const weight =
op.weight ?? (typeof reserved.weight === 'number' ? reserved.weight : undefined)
const subtype =
op.subtype ?? (typeof reserved.subtype === 'string' ? reserved.subtype : undefined)
const mergedMetadata =
op.merge !== false
? ({ ...(base.metadata as object), ...custom } as T)
@ -814,14 +807,19 @@ export class Db<T = any> {
}
if (duplicate) break
// Field-addressing law — relationship mirror of the add case: the
// edge bag is the user's verbatim; engine scalars only from
// dedicated op fields.
const custom = { ...(op.metadata as Record<string, unknown> | undefined) }
const confidence = op.confidence
const weight = op.weight
const subtype = op.subtype
const service = op.service
// Reserved-field normalization — relationship mirror of the add
// op above (and of the committed relate() path).
const { reserved, custom } = splitVerbMetadataRecord(
op.metadata as Record<string, unknown> | undefined
)
const confidence =
op.confidence ?? (typeof reserved.confidence === 'number' ? reserved.confidence : undefined)
const weight =
op.weight ?? (typeof reserved.weight === 'number' ? reserved.weight : undefined)
const subtype =
op.subtype ?? (typeof reserved.subtype === 'string' ? reserved.subtype : undefined)
const service =
op.service ?? (typeof reserved.service === 'string' ? reserved.service : undefined)
const id = uuidv4()
overlay.verbs.set(id, {

View file

@ -23,12 +23,8 @@
* serve the full query surface via at-generation index materialization.
* - {@link GenerationCompactedError} `asOf()` asked for a generation whose
* immutable records were reclaimed by `compactHistory()`.
* - {@link CanonicalEnumerationUnavailableError} `export()`'s
* `enumeration:'canonical'` mode was called on a historical `asOf()` view or a
* speculative `with()` overlay; the canonical storage walk only ever answers
* "what is live right now."
*
* All are exported from the package root (`@soulcraftlabs/brainy`).
* All three are exported from the package root (`@soulcraft/brainy`).
*/
/**
@ -164,64 +160,6 @@ export class GenerationCompactedError extends Error {
}
}
/**
* @description Thrown by `db.export(selector, { enumeration: 'canonical' })` when
* the `Db` it is called on is not the live, current-generation view: a historical
* `brain.asOf(g)` pin, or a speculative `db.with()` overlay.
*
* Canonical enumeration mode walks the storage adapter's canonical shard layout
* directly (`storage.getNouns()`/`getVerbs()`) instead of the metadata/graph
* indexes but that walk has no generation parameter, it can only ever answer
* "what is live right now." Serving it against a historical pin would silently
* mix generations (today's canonical records under yesterday's selector), and
* against a speculative overlay it would silently miss the overlay's own
* in-memory entities (which never touched storage). Both are exactly the kind of
* silently-wrong result canonical mode exists to prevent elsewhere so this
* boundary throws instead.
*
* `enumeration: 'index'` (the default) is unaffected: it composes with
* `asOf()`/`with()` exactly as before, via the generation-correct `find()` walk.
*
* @example
* const past = await brain.asOf(g1)
* try {
* await past.export({}, { enumeration: 'canonical' })
* } catch (err) {
* if (err instanceof CanonicalEnumerationUnavailableError) {
* // Time-travel export: use the default index-based enumeration instead.
* await past.export({}, { enumeration: 'index' })
* }
* }
*/
export class CanonicalEnumerationUnavailableError extends Error {
/** The view's pinned generation. */
public readonly generation: number
/** Why canonical mode cannot serve this view. */
public readonly reason: 'historical' | 'overlay'
/**
* @param generation - The view's pinned generation.
* @param reason - `'historical'` (a past `asOf()` pin) or `'overlay'` (a speculative `with()`).
*/
constructor(generation: number, reason: 'historical' | 'overlay') {
const what =
reason === 'historical'
? `a historical view pinned at generation ${generation}`
: `a speculative with() overlay (base generation ${generation})`
super(
`export()'s enumeration:'canonical' requires the live, current-generation view — ` +
`it was called on ${what}. The canonical storage walk has no generation parameter, ` +
`so it can only answer "what is live right now"; serving it here would silently ` +
`mix generations (historical) or miss the overlay's own in-memory entities ` +
`(overlay). Use enumeration:'index' (the default) for a time-travel or what-if ` +
`export, or pin brain.now() for a live canonical export.`
)
this.name = 'CanonicalEnumerationUnavailableError'
this.generation = generation
this.reason = reason
}
}
/** One entity/relationship left in an unreconciled state by a failed rollback. */
export interface UnreconciledRecord {
/** The entity or relationship id. */

File diff suppressed because it is too large Load diff

File diff suppressed because it is too large Load diff

View file

@ -12,11 +12,9 @@
* the verified surface is a small set of rollup invariants (entity/
* relationship counts) plus `sourceGeneration`.
*
* `sourceGeneration` is the COMMITTED generation of the source-of-truth log
* this projection reflects never the allocated counter, which names a
* generation that may never commit (see {@link StampVerdict.torn}) so
* open-time coherence becomes a COMPARISON (stamp vs committed head), not a
* walk:
* `sourceGeneration` is the generation of the source-of-truth log this
* projection reflects open-time coherence becomes a COMPARISON (stamp vs
* log head), not a walk:
*
* - equal + invariants hold coherent, serve.
* - behind the projection missed the tail (crash between commit and stamp);
@ -26,9 +24,6 @@
* - invariants FAIL at equal generation genuine incoherence: loud, and the
* repair ritual (`repairIndex()`, whose recount rebuilds the rollups from a
* canonical walk) heals it.
* - AHEAD a torn generation-log tail: the stamp's fsync outlived the log
* tail's. TERMINAL, never a wait the generation the stamp names does not
* exist to arrive.
*
* Stamps are JSON on purpose every incident gets debugged by reading a
* stamp in a terminal.
@ -75,12 +70,6 @@ export type StampVerdict =
| { state: 'coherent' }
| { state: 'absent' } // legacy store — first stamp writes at the next flush
| { state: 'behind'; stampSource: number; head: number }
/**
* TORN GENERATION-LOG TAIL: the stamp witnesses a source generation the
* store's committed watermark can no longer show. TERMINAL there is no
* generation to wait for, so the open demotes (or refuses) and never spins.
*/
| { state: 'torn'; stampSource: number; head: number }
| { state: 'incoherent'; failures: string[] }
| { state: 'unverifiable'; reason: string } // a FAULT reading the stamp — never conflated with absence
@ -129,15 +118,12 @@ export function verifyFamilyStamp(
): StampVerdict {
if (stamp === null) return { state: 'absent' }
if (stamp.sourceGeneration > head) {
// A stamp AHEAD of committed truth witnesses a generation the store can no
// longer show: the stamp's fsync survived a crash that the log tail did
// not. This is the TORN GENERATION-LOG TAIL — its own class, never folded
// in with `incoherent` (a count that drifted at a generation both sides
// agree on), because the two have opposite cures: incoherence is recounted,
// a tear is DEMOTED. It is also terminal by construction — there is no
// generation the open can wait for, because the one the stamp names is
// gone.
return { state: 'torn', stampSource: stamp.sourceGeneration, head }
// A stamp AHEAD of the log claims state that never committed — the
// projection was stamped against truth that a crash rolled back.
return {
state: 'incoherent',
failures: [`sourceGeneration ${stamp.sourceGeneration} is ahead of the log head ${head}`]
}
}
if (stamp.sourceGeneration < head) {
return { state: 'behind', stampSource: stamp.sourceGeneration, head }

View file

@ -1,164 +0,0 @@
/**
* @module db/faultInjectionStorage
* @description Deterministic fault injection at the fact log's raw-byte
* storage surface the test harness half of the durability protocol. Wraps
* any adapter exposing the {@link FactLogStorage} primitives (the exact
* surface the fact log appends and syncs through) and injects the three
* crash shapes durability tests must prove against:
*
* - **torn write** ({@link FaultInjectionStorage.tearWriteAtByte}): the next
* append persists only its first N bytes, then reports success the shape
* of power loss after a partially-flushed page. The caller-side "crash" is
* simulated by abandoning in-memory state and reopening from storage.
* - **dropped sync** ({@link FaultInjectionStorage.dropNextSync}): the next
* sync becomes a silent no-op an fsync the device acknowledged into a
* volatile cache and lost.
* - **failed append** ({@link FaultInjectionStorage.failNextAppend}): the next
* append throws {@link FaultInjectedError} without writing a byte EIO or
* a full disk, surfaced to the writer.
*
* Every injected fault is journaled on {@link FaultInjectionStorage.injectedFaults}
* so tests can assert not just the outcome but that the fault actually fired.
* Knobs are one-shot (they disarm on firing) and re-arming overwrites the
* pending shot. All other operations pass through untouched.
*/
import type { FactLogStorage } from './factLog.js'
/** The error a {@link FaultInjectionStorage.failNextAppend} shot throws. */
export class FaultInjectedError extends Error {
/** The operation the fault fired on. */
public readonly operation: 'append'
/** The storage path the operation targeted. */
public readonly path: string
constructor(operation: 'append', path: string) {
super(`fault injection: ${operation} to ${path} failed by test design`)
this.name = 'FaultInjectedError'
this.operation = operation
this.path = path
}
}
/** One journaled fault event — proof the injected fault actually fired. */
export interface InjectedFault {
kind: 'torn-write' | 'dropped-sync' | 'failed-append'
/** The target path (torn-write / failed-append). */
path?: string
/** The paths a dropped sync was asked to make durable. */
paths?: string[]
/** Bytes the caller asked to append (torn-write). */
requestedBytes?: number
/** Bytes actually persisted (torn-write). */
writtenBytes?: number
}
/**
* A {@link FactLogStorage} wrapper that injects deterministic storage faults.
* Construct it around any conforming adapter and hand it wherever a
* FactLogStorage is accepted unarmed, it is a transparent passthrough.
*/
export class FaultInjectionStorage implements FactLogStorage {
private readonly inner: FactLogStorage
/** Pending torn-write byte count, or null when unarmed. */
private tearAtByte: number | null = null
/** Pending dropped-sync shot. */
private dropSyncArmed = false
/** Pending failed-append shot. */
private failAppendArmed = false
/** Journal of every fault that fired, in firing order. */
public readonly injectedFaults: InjectedFault[] = []
constructor(inner: FactLogStorage) {
this.inner = inner
}
/**
* Arm a torn write: the NEXT {@link appendRawBytes} persists only the first
* `n` bytes of its buffer (all of it when `n` exceeds the buffer) and then
* reports success. One-shot.
*/
tearWriteAtByte(n: number): void {
if (!Number.isInteger(n) || n < 0) {
throw new Error(`fault injection: tearWriteAtByte needs a non-negative integer; got ${n}`)
}
this.tearAtByte = n
}
/** Arm a dropped sync: the NEXT {@link syncRawObjects} silently does nothing. One-shot. */
dropNextSync(): void {
this.dropSyncArmed = true
}
/**
* Arm a failed append: the NEXT {@link appendRawBytes} throws
* {@link FaultInjectedError} without writing. One-shot; wins over a
* simultaneously-armed torn write (nothing is written at all).
*/
failNextAppend(): void {
this.failAppendArmed = true
}
/** Append bytes — the injection point for torn writes and failed appends. */
async appendRawBytes(path: string, bytes: Uint8Array): Promise<void> {
if (this.failAppendArmed) {
this.failAppendArmed = false
this.injectedFaults.push({ kind: 'failed-append', path })
throw new FaultInjectedError('append', path)
}
if (this.tearAtByte !== null) {
const writtenBytes = Math.min(this.tearAtByte, bytes.length)
this.tearAtByte = null
this.injectedFaults.push({
kind: 'torn-write',
path,
requestedBytes: bytes.length,
writtenBytes
})
if (writtenBytes > 0) {
await this.inner.appendRawBytes(path, bytes.subarray(0, writtenBytes))
}
return
}
return this.inner.appendRawBytes(path, bytes)
}
/** Make paths durable — the injection point for dropped syncs. */
async syncRawObjects(paths: string[]): Promise<void> {
if (this.dropSyncArmed) {
this.dropSyncArmed = false
this.injectedFaults.push({ kind: 'dropped-sync', paths: [...paths] })
return
}
return this.inner.syncRawObjects(paths)
}
/** Passthrough. */
async readRawBytes(path: string): Promise<Uint8Array | null> {
return this.inner.readRawBytes(path)
}
/** Passthrough. */
async writeRawBytes(path: string, bytes: Uint8Array): Promise<void> {
return this.inner.writeRawBytes(path, bytes)
}
/** Passthrough. */
async rawByteSize(path: string): Promise<number | null> {
return this.inner.rawByteSize(path)
}
/** Passthrough. */
async readRawObject(path: string): Promise<any | null> {
return this.inner.readRawObject(path)
}
/** Passthrough. */
async writeRawObject(path: string, data: any): Promise<void> {
return this.inner.writeRawObject(path, data)
}
/** Passthrough. */
async deleteRawObject(path: string): Promise<void> {
return this.inner.deleteRawObject(path)
}
}

View file

@ -1,345 +0,0 @@
/**
* @module db/fieldAddressing
* @description The one field-addressing law for every query surface (find()'s
* `where` / `orderBy` / `groupBy`, aggregation `source.where`), ruled
* 2026-08-03 after a production incident in which a user metadata field
* named `level` was silently shadowed by the engine's internal HNSW node
* layer (VENUE-BRAINY-ORDERBY-NOOP thread id kept verbatim as the audit
* key; it names no product):
*
* 1. A BARE field name addresses the user's metadata field. Always.
* No priority resolution, no fallback chain `orderBy: 'level'`
* reads `entity.metadata.level`, full stop.
* 2. `system.<field>` addresses an engine scalar, reachable ONLY with the
* explicit prefix. The entity map is exactly ten scalars; the relation
* map mirrors it with `verb`/`sourceId`/`targetId` as the structural
* members.
* 3. Engine plumbing (`vector`, `connections`, `level`, `data`, `_rev`) is
* INVISIBLE to the query surface in either spelling `system.level`
* refuses; bare `level` is the user's field.
* 4. `metadata.<field>` is the explicit spelling of the bare form
* identical semantics on every path.
* 5. Anything unresolvable refuses with a TYPED error naming both
* candidate spellings an accepted name either works or refuses;
* there is no third state.
*
* This module is the SINGLE source of truth for the law: parsing, the maps,
* and the refusal builders live here so the JS engine, the provider seams,
* and the cross-engine conformance suite can never drift on the contract.
*/
import type { HNSWNounWithMetadata, HNSWVerbWithMetadata } from '../coreTypes.js'
/**
* @description The entity-side `system.*` map EXACTLY the ten engine
* scalars David ruled queryable (2026-08-03). Adding a name here is a
* cross-engine contract change: the native accelerator's conformance suite
* pins this list verbatim, so any edit must ship as a paired release.
*/
export const SYSTEM_ENTITY_SCALARS: ReadonlySet<string> = new Set([
'id',
'type',
'subtype',
'createdAt',
'updatedAt',
'confidence',
'weight',
'visibility',
'service',
'createdBy'
])
/**
* @description The relation-side `system.*` map the verb mirror of
* {@link SYSTEM_ENTITY_SCALARS}: `verb`, `sourceId`, `targetId` are the
* structural members beside the eight shared scalars. Same one law, same
* pairing rule for edits.
*/
export const SYSTEM_RELATION_SCALARS: ReadonlySet<string> = new Set([
'verb',
'sourceId',
'targetId',
'subtype',
'createdAt',
'updatedAt',
'confidence',
'weight',
'visibility',
'service',
'createdBy'
])
/**
* @description Engine plumbing never addressable from the query surface in
* ANY spelling. `level` is the HNSW node layer (the incident field: listing
* it as resolvable shadowed real user data); `data` is the payload container,
* not a scalar content is reached through the content/text-search APIs,
* and addressing it as a sortable field would lie about its shape.
*/
export const PLUMBING_FIELDS: ReadonlySet<string> = new Set([
'vector',
'connections',
'level',
'data',
'_rev'
])
/** @description Which record kind a field address is being resolved against. */
export type FieldAddressKind = 'entity' | 'relation'
/**
* @description A parsed, law-valid field address. `scope` says which side of
* the record the name lives on; `field` is the unprefixed name to read.
*/
export interface FieldAddress {
/** 'metadata' = the user's field (bare or `metadata.`-prefixed); 'system' = an engine scalar. */
scope: 'metadata' | 'system'
/** The field name with any scope prefix removed. */
field: string
/** The exact spelling the caller used — preserved for error text and telemetry. */
raw: string
}
/**
* Parse a query-surface field name under the one law. Pure and data-blind:
* this validates the ADDRESS (spelling + map membership), not whether any
* row actually carries the field data-aware refusals (the did-you-mean
* for a bare system-scalar name no row carries) belong to the query layer,
* which calls {@link buildUnresolvableMessage} with index knowledge.
*
* @param raw - The field name as the caller wrote it (`level`,
* `metadata.level`, `system.createdAt`, )
* @param kind - Entity or relation resolution (selects the system map)
* @returns The parsed {@link FieldAddress}
* @throws {InvalidFieldAddressError} for a `system.*` name outside the ruled
* map (including every plumbing field) or a malformed spelling the error
* text enumerates the valid system scalars so the fix is in the message.
*
* @example
* parseFieldAddress('level', 'entity') // { scope: 'metadata', field: 'level' }
* parseFieldAddress('metadata.level', 'entity') // { scope: 'metadata', field: 'level' }
* parseFieldAddress('system.createdAt', 'entity') // { scope: 'system', field: 'createdAt' }
* parseFieldAddress('system.level', 'entity') // throws — plumbing is invisible
*/
export function parseFieldAddress(
raw: string,
kind: FieldAddressKind
): FieldAddress {
const systemMap =
kind === 'entity' ? SYSTEM_ENTITY_SCALARS : SYSTEM_RELATION_SCALARS
if (raw.startsWith('system.')) {
const field = raw.slice('system.'.length)
if (!systemMap.has(field)) {
throw new InvalidFieldAddressError(raw, kind, systemMap)
}
return { scope: 'system', field, raw }
}
if (raw.startsWith('metadata.')) {
const field = raw.slice('metadata.'.length)
if (field.length === 0) {
throw new InvalidFieldAddressError(raw, kind, systemMap)
}
return { scope: 'metadata', field, raw }
}
if (raw.length === 0) {
throw new InvalidFieldAddressError(raw, kind, systemMap)
}
// Bare name = the user's metadata field. Always. Even when the same name
// exists in the system map — `confidence` as a bare name is the user's
// metadata field named confidence; the engine scalar is system.confidence.
return { scope: 'metadata', field: raw, raw }
}
/**
* Read the addressed value off an entity. The ONLY sanctioned way a query
* surface turns a {@link FieldAddress} into a value direct property reads
* against records re-create the shadow class this module exists to kill.
*
* @returns The value, or `undefined` when the record does not carry it
* (missing values sort LAST in both directions per the ordering contract
* they are never grounds for dropping a row).
*/
export function readEntityFieldAddress(
entity: HNSWNounWithMetadata,
address: FieldAddress
): unknown {
const rec = entity as unknown as Record<string, unknown>
const bag =
rec.metadata && typeof rec.metadata === 'object'
? (rec.metadata as Record<string, unknown>)
: null
if (address.scope === 'system') {
// System scalars live at the record's top level, NEVER in the user's
// bag — a user field named `confidence` must be unreachable from
// system.confidence (and vice versa). Entity views carry the scalars
// top-level directly; record-derived views spell the type `noun`.
const top = rec[address.field]
if (top !== undefined) return top
if (address.field === 'type') return rec.noun
return undefined
}
// User scope: the bag IS the user's namespace, authoritative — EVERY name
// reads from it, engine spellings included (`bag.confidence` is the user's
// confidence field under the field-addressing law).
if (bag) return bag[address.field]
// No bag at all: a LEGACY flat record (pre-nested-bag storage). Its keys
// matching system/plumbing names are the ENGINE's — the pre-law write door
// refused user colliders — so a bare system name reads as ABSENT rather
// than resurrecting the shadow this module exists to kill. Same for the
// legacy 'noun' spelling.
if (
SYSTEM_ENTITY_SCALARS.has(address.field) ||
PLUMBING_FIELDS.has(address.field) ||
address.field === 'noun'
) {
return undefined
}
return rec[address.field]
}
/**
* Relation twin of {@link readEntityFieldAddress}. The stored flat record
* keys the relation type under `verb`; public Relation shapes may carry it
* as `type` both spellings of the record are read, the ADDRESS is always
* `system.verb`.
*/
export function readRelationFieldAddress(
verb: HNSWVerbWithMetadata,
address: FieldAddress
): unknown {
if (address.scope === 'system') {
const rec = verb as unknown as Record<string, unknown>
if (address.field === 'verb') return rec.verb ?? rec.type
return rec[address.field]
}
return verb.metadata?.[address.field]
}
/**
* Build the ruled did-you-mean refusal text for a bare name that resolved to
* metadata but is UNKNOWN to the index the data-aware half of the law,
* called by the query layer once it has consulted the known-field set:
*
* "no metadata field 'createdAt' did you mean system.createdAt or
* metadata.createdAt?"
*
* When the bare name is NOT a system scalar the system candidate is omitted
* (there is only one thing the caller could have meant; the refusal exists
* because refusing beats silently sorting nothing).
*/
export function buildUnresolvableMessage(
raw: string,
kind: FieldAddressKind
): string {
const systemMap =
kind === 'entity' ? SYSTEM_ENTITY_SCALARS : SYSTEM_RELATION_SCALARS
if (systemMap.has(raw)) {
return (
`no metadata field '${raw}' — did you mean system.${raw} or metadata.${raw}? ` +
`(bare names always address your metadata; engine fields need the system. prefix)`
)
}
return (
`no metadata field '${raw}' on this store — nothing carries it, so an ordered or ` +
`filtered read against it cannot mean anything. Spell it metadata.${raw} once the ` +
`field exists, or check the field name (system.${raw} is NOT valid — '${raw}' is ` +
`not one of the engine's system scalars).`
)
}
/**
* @description Refusal for a syntactically valid address that resolves to
* NOTHING a bare name no user field carries. Carries the did-you-mean
* (both candidate spellings when the name collides with a system scalar) so
* the fix ships inside the error. Thrown by the query layer with index
* knowledge, never by the pure parser.
*/
/**
* Cross-package identity normalizer (the seam belt): the native accelerator
* throws ITS OWN UnresolvableFieldError class, which fails `instanceof`
* against this package's export consumers were forced to match by name.
* Every provider-boundary catch routes suspected field-refusals through
* here: a foreign refusal (matched by name, duck fields tolerated) is
* rethrown as THIS package's class, so exactly one identity ever reaches
* consumers. Anything else returns null (caller rethrows the original).
*/
export function asBrainyFieldRefusal(err: unknown): UnresolvableFieldError | null {
if (err instanceof UnresolvableFieldError) return err
const e = err as { name?: string; message?: string; raw?: string; kind?: string } | null
if (e && e.name === 'UnresolvableFieldError') {
return new UnresolvableFieldError(
e.raw ?? 'unknown-field',
(e.kind as FieldAddressKind) ?? 'entity',
e.message
)
}
return null
}
export class UnresolvableFieldError extends Error {
public readonly raw: string
public readonly kind: FieldAddressKind
constructor(raw: string, kind: FieldAddressKind, messageOverride?: string) {
super(messageOverride ?? buildUnresolvableMessage(raw, kind))
this.name = 'UnresolvableFieldError'
this.raw = raw
this.kind = kind
}
}
/**
* @description Refusal for a malformed or out-of-map field ADDRESS
* `system.<anything-not-in-the-map>` (including all plumbing), an empty
* name, or a bare `metadata.` prefix. The message carries the full valid
* system map so the fix never needs a docs lookup.
*/
export class InvalidFieldAddressError extends UnresolvableFieldError {
constructor(raw: string, kind: FieldAddressKind, systemMap: ReadonlySet<string>) {
const valid = [...systemMap].map((f) => `system.${f}`).join(', ')
super(
raw,
kind,
`'${raw}' is not an addressable ${kind} field. Bare names address your own ` +
`metadata fields; engine fields are exactly: ${valid}. Engine plumbing ` +
`(vector, connections, level, data, _rev) is not part of the query surface.`
)
this.name = 'InvalidFieldAddressError'
}
}
/**
* @description Refusal for a find() option that is accepted by the type
* surface but NOT implemented an accepted option must work or refuse;
* accepted-and-ignored died as a class (sealed 2026-08-03). Names the
* option and the honest state so nobody discovers a no-op by measurement.
*/
export class UnsupportedFindOptionError extends Error {
public readonly option: string
constructor(option: string) {
super(
`find() option '${option}' is not implemented — it used to be silently ` +
`ignored, which read as working. Remove it from the call (or track the ` +
`feature request); it will be honored or refused, never swallowed.`
)
this.name = 'UnsupportedFindOptionError'
this.option = option
}
}
/**
* @description The capability signal both engines' conformance suites arm on
* (never a version guess): its presence at the package root means the one
* field-addressing law is LIVE on every query surface bare = user metadata,
* `system.*` = the ruled scalars, plumbing invisible, refusals typed.
*/
export const FIELD_ADDRESSING_CAPABILITY = 'field-addressing/v1'

View file

@ -1,570 +0,0 @@
/**
* @module db/generationSegments
* @description The generation-segment store Stage-2 D1+D3+repacking's file
* format (co-frozen 2026-07-19; design: the d1-d3-repacking spec).
*
* Packs CONSECUTIVE cold generations' record-sets (before-images + delta)
* into append-once segment files with derived sidecar indexes, so history
* scales in SEGMENTS (tens) instead of FILES-PER-GENERATION (hundreds of
* thousands), and cold-open reads ONE manifest instead of listing the
* backlog. Layout under `_generations/segments/`:
*
* - `seg-<firstGen, zero-padded 20>.bgs` magic "BGS1", then one frame per
* generation: `u32 payloadLen | u32 crc32c | msgpack payload`. Payload is
* POSITIONAL: `[generation, timestamp, delta, records[], flags]` with
* records `[kindByte, id, record]`. `flags` reserves encoding evolution
* (bit 0 = compressed payload v1 always 0; a future writer upgrade,
* never a format break). Sealed segments are IMMUTABLE the fact log's
* own law, generalized.
* - `seg-<firstGen>.idx` DERIVED sidecar (msgpack): per-generation frame
* offsets (point reads = one ranged read, never a listing) + per-id
* generation postings (per-id chain rebuilds read only what they need).
* Corrupt/missing rebuilt from its segment in one sequential read,
* loudly.
* - `manifest.json` the segment catalogue + `compactedBelow` (D3's
* horizon marker). Cold-open reads THIS; the packed backlog is never
* listed.
*
* D3 semantics carried here: bounded-retention reclaim drops WHOLE segments
* at boundaries (O(1) per segment, no rewrite); under the archival profile
* (`retention: 'all'`) nothing here is ever dropped folding is the only
* transform (re-representation, never deletion).
*/
import { encode as msgpackEncode, decode as msgpackDecode } from '@msgpack/msgpack'
import { crc32c } from '../utils/crc32c.js'
import type { FactLogStorage } from './factLog.js'
import { prodLog } from '../utils/logger.js'
/** Directory for segment files + manifest, under the generations prefix. */
export const SEGMENTS_PREFIX = '_generations/segments'
/** Target sealed-segment size (co-freeze proposal; tunable on evidence). */
export const SEGMENT_TARGET_BYTES = 64 * 1024 * 1024
const MAGIC = new TextEncoder().encode('BGS1')
const FRAME_PREFIX_BYTES = 8 // u32 payloadLen + u32 crc32c
const MANIFEST_PATH = `${SEGMENTS_PREFIX}/manifest.json`
/** One generation's fold input — exactly what the live tier holds for it. */
export interface FoldGeneration {
generation: number
timestamp: number
/** The tx.json delta object, carried verbatim. */
delta: unknown
/** The before-image record-set (empty for record-less generations). */
records: Array<{ kind: 'noun' | 'verb'; id: string; record: unknown }>
}
/** Manifest entry for one sealed segment. */
export interface SegmentMeta {
file: string
firstGeneration: number
lastGeneration: number
frames: number
bytes: number
/** crc32c of the full segment byte stream — the digest chain's link. */
checksum: number
}
interface SegmentManifest {
version: 1
compactedBelow: number
segments: SegmentMeta[]
}
interface SidecarIndex {
version: 1
/** [generation, frameOffset, frameLen] ascending by generation. */
generations: Array<[number, number, number]>
/** `${kindByte}:${id}` → ascending generations holding a record for it. */
ids: Record<string, number[]>
}
const segmentFileName = (firstGeneration: number): string =>
`seg-${String(firstGeneration).padStart(20, '0')}.bgs`
const sidecarFileName = (firstGeneration: number): string =>
`seg-${String(firstGeneration).padStart(20, '0')}.idx`
/**
* The generation-segment store. Owns the packed tier ONLY the live
* per-generation tier and the routing between tiers belong to
* `GenerationStore`. All mutating entry points here are called under the
* generation store's commit mutex.
*/
export class GenerationSegmentStore {
private readonly storage: FactLogStorage
private manifest: SegmentManifest = { version: 1, compactedBelow: 0, segments: [] }
/** Sidecar cache — segments are immutable, so entries never invalidate. */
private readonly sidecars = new Map<string, SidecarIndex>()
constructor(storage: FactLogStorage) {
this.storage = storage
}
/** Load the manifest (ONE read — never a directory listing). */
async open(): Promise<void> {
const raw = (await this.storage.readRawObject(MANIFEST_PATH)) as SegmentManifest | null
if (raw) {
if (raw.version !== 1) {
throw new Error(
`[GenerationSegments] manifest version ${String(raw.version)} is newer than this ` +
`engine understands — refusing to serve partial history. Upgrade the engine.`
)
}
this.manifest = raw
}
}
/** The packed tier's catalogue (ascending, immutable snapshot). */
segments(): readonly SegmentMeta[] {
return this.manifest.segments
}
/** D3's horizon marker: generations below this were reclaimed (bounded profiles only). */
compactedBelow(): number {
return this.manifest.compactedBelow
}
/** The covering sealed segment for `gen`, or null if it lives outside the packed tier. */
private coveringSegment(gen: number): SegmentMeta | null {
// Manifest is ascending and ranges never overlap — binary search.
const segs = this.manifest.segments
let lo = 0
let hi = segs.length - 1
while (lo <= hi) {
const mid = (lo + hi) >> 1
const s = segs[mid]
if (gen < s.firstGeneration) hi = mid - 1
else if (gen > s.lastGeneration) lo = mid + 1
else return s
}
return null
}
/** True when `gen` is packed (readable from this tier). */
hasGeneration(gen: number): boolean {
return this.coveringSegment(gen) !== null
}
/**
* @description True when `meta` declares more generations than it holds
* frames a segment sealed by a writer that folded across a hole. The
* manifest records `frames` at fold time, so this is an O(1) comparison
* against the declared span and needs no I/O.
*/
private isSparse(meta: SegmentMeta): boolean {
return meta.lastGeneration - meta.firstGeneration + 1 !== meta.frames
}
/**
* @description The generations this tier ACTUALLY holds, as coalesced
* ascending intervals not what the segments declare.
*
* Dense segments (every one a current writer produces) contribute their
* declared range with no I/O. A SPARSE segment one sealed before the
* density law was enforced, whose declared range spans generations it has
* no frame for has its real generation list read from its sidecar and
* contributed instead, with the discrepancy narrated once.
*
* This is what keeps a store that already carries the damage from wedging.
* `open()` seeds `committedRanges` from these intervals, so a hole is never
* re-admitted as a committed generation, and the auto-compaction pass that
* used to fail on every run with "packed history is damaged" simply never
* asks for the missing frame.
*
* @returns Ascending, non-overlapping `[first, last]` intervals.
*/
async actualRanges(): Promise<Array<[number, number]>> {
const out: Array<[number, number]> = []
for (const meta of this.manifest.segments) {
if (!this.isSparse(meta)) {
out.push([meta.firstGeneration, meta.lastGeneration])
continue
}
const missing = meta.lastGeneration - meta.firstGeneration + 1 - meta.frames
prodLog.warn(
`[GenerationSegments] sealed segment ${meta.file} declares generations ` +
`${meta.firstGeneration}..${meta.lastGeneration} but holds only ${meta.frames} ` +
`frame(s) — ${missing} generation(s) in that span were never folded into it. ` +
`Serving the frames it actually holds; the declared span is not treated as ` +
`committed history. (Written by a pre-density-law writer that folded across a ` +
`gap; the segment itself is intact and no record is lost.)`
)
const idx = await this.sidecarFor(meta)
for (const [gen] of idx.generations) {
const last = out[out.length - 1]
if (last !== undefined && gen === last[1] + 1) last[1] = gen
else out.push([gen, gen])
}
}
return out
}
/**
* Fold consecutive generations into ONE new sealed segment + sidecar and
* append it to the manifest atomically. Caller guarantees: `gens` is
* ascending, contiguous with the packed tier (first = last packed + 1 when
* segments exist), and already durable in the live tier. Crash between the
* segment write and the caller's live-tier delete leaves a DUPLICATE
* representation resolved live-tier-wins by the reader; never a gap.
*/
async fold(gens: FoldGeneration[]): Promise<SegmentMeta> {
if (gens.length === 0) {
throw new Error('[GenerationSegments] fold() requires at least one generation')
}
for (let i = 1; i < gens.length; i++) {
if (gens[i].generation <= gens[i - 1].generation) {
throw new Error('[GenerationSegments] fold() input must be strictly ascending')
}
}
// THE DENSITY LAW, MADE MECHANICAL.
//
// A sealed segment declares a CONTIGUOUS range [firstGeneration,
// lastGeneration] and every reader treats that range as containment:
// `coveringSegment` is an interval test, `hasGeneration` returns true for
// anything inside it, and `open()` seeds committedRanges from it. So a
// segment folded from a SPARSE input silently claims generations it does
// not hold, and the first read of one of those holes throws
// "inside sealed segment ... but has no frame — packed history is damaged".
//
// That is exactly how the damage was produced. `repackHistory` skipped
// generations mid-batch — ones absent from committedRanges, ones still in
// the pending buffer, ones whose tx.json would not read — and handed the
// survivors here, where the range was computed from the first and last of
// them. Worse, the mis-declared range was then merged back into
// committedRanges at the next open, which is what turned a quiet hole into
// a repeating auto-compaction failure on every subsequent run.
//
// Callers now split at discontinuities; this refusal is what keeps any
// future caller from reintroducing the class. A refusal here loses
// nothing — the generations stay in the live tier, readable, and the next
// pass folds them correctly.
for (let i = 1; i < gens.length; i++) {
if (gens[i].generation !== gens[i - 1].generation + 1) {
throw new Error(
`[GenerationSegments] fold() input is not contiguous: ${gens[i - 1].generation}` +
`${gens[i].generation} skips ${gens[i].generation - gens[i - 1].generation - 1} ` +
`generation(s). A sealed segment declares a dense range, so folding a sparse ` +
`batch would claim generations it does not hold. Split the batch at the gap.`
)
}
}
const last = this.manifest.segments[this.manifest.segments.length - 1]
if (last && gens[0].generation <= last.lastGeneration) {
throw new Error(
`[GenerationSegments] fold() overlaps the packed tier: ${gens[0].generation}` +
`sealed ${last.lastGeneration} — segments are immutable, never rewritten`
)
}
const first = gens[0].generation
const file = segmentFileName(first)
const sidecar: SidecarIndex = { version: 1, generations: [], ids: {} }
// Encode all frames, tracking offsets for the sidecar.
const parts: Uint8Array[] = [MAGIC]
let offset = MAGIC.length
for (const g of gens) {
const payload = msgpackEncode([
g.generation,
g.timestamp,
g.delta,
g.records.map((r) => [r.kind === 'noun' ? 0 : 1, r.id, r.record]),
0 // flags: v1 = uncompressed
])
const frame = new Uint8Array(FRAME_PREFIX_BYTES + payload.length)
const view = new DataView(frame.buffer)
view.setUint32(0, payload.length, true)
view.setUint32(4, crc32c(payload), true)
frame.set(payload, FRAME_PREFIX_BYTES)
sidecar.generations.push([g.generation, offset, frame.length])
for (const r of g.records) {
const key = `${r.kind === 'noun' ? 0 : 1}:${r.id}`
;(sidecar.ids[key] ??= []).push(g.generation)
}
parts.push(frame)
offset += frame.length
}
const total = parts.reduce((n, p) => n + p.length, 0)
const bytes = new Uint8Array(total)
let at = 0
for (const p of parts) {
bytes.set(p, at)
at += p.length
}
const meta: SegmentMeta = {
file,
firstGeneration: first,
lastGeneration: gens[gens.length - 1].generation,
frames: gens.length,
bytes: total,
checksum: crc32c(bytes)
}
// Durability order: segment + sidecar fsync'd BEFORE the manifest names
// them (a crash before the manifest = invisible orphan files, harmless);
// manifest last, atomically.
const segPath = `${SEGMENTS_PREFIX}/${file}`
const idxPath = `${SEGMENTS_PREFIX}/${sidecarFileName(first)}`
await this.storage.writeRawBytes(segPath, bytes)
await this.storage.writeRawBytes(idxPath, msgpackEncode(sidecar))
await this.storage.syncRawObjects([segPath, idxPath])
const next: SegmentManifest = {
...this.manifest,
segments: [...this.manifest.segments, meta]
}
await this.storage.writeRawObject(MANIFEST_PATH, next)
await this.storage.syncRawObjects([MANIFEST_PATH])
this.manifest = next
this.sidecars.set(file, sidecar)
return meta
}
/** Load (or rebuild, loudly) a segment's sidecar. */
private async sidecarFor(meta: SegmentMeta): Promise<SidecarIndex> {
const cached = this.sidecars.get(meta.file)
if (cached) return cached
const idxPath = `${SEGMENTS_PREFIX}/${sidecarFileName(meta.firstGeneration)}`
const raw = await this.storage.readRawBytes(idxPath)
if (raw) {
try {
const idx = msgpackDecode(raw) as SidecarIndex
if (idx.version === 1) {
this.sidecars.set(meta.file, idx)
return idx
}
} catch {
// fall through to rebuild
}
}
// Sidecars are DERIVED: rebuild from the segment, loudly — never serve
// wrong offsets silently.
prodLog.warn(
`[GenerationSegments] sidecar for ${meta.file} missing or unreadable — rebuilding from the segment`
)
const rebuilt = await this.rebuildSidecar(meta)
await this.storage.writeRawBytes(idxPath, msgpackEncode(rebuilt))
this.sidecars.set(meta.file, rebuilt)
return rebuilt
}
/** One sequential read of the segment → a fresh sidecar. Verifies every frame CRC. */
private async rebuildSidecar(meta: SegmentMeta): Promise<SidecarIndex> {
const frames = await this.readAllFrames(meta)
const idx: SidecarIndex = { version: 1, generations: [], ids: {} }
for (const f of frames) {
idx.generations.push([f.generation, f.offset, f.frameLen])
for (const r of f.records) {
const key = `${r.kind === 'noun' ? 0 : 1}:${r.id}`
;(idx.ids[key] ??= []).push(f.generation)
}
}
return idx
}
private decodeFrame(
payload: Uint8Array
): { generation: number; timestamp: number; delta: unknown; records: FoldGeneration['records'] } {
const [generation, timestamp, delta, rawRecords] = msgpackDecode(payload) as [
number,
number,
unknown,
Array<[number, string, unknown]>,
number
]
return {
generation,
timestamp,
delta,
records: rawRecords.map(([kindByte, id, record]) => ({
kind: kindByte === 0 ? ('noun' as const) : ('verb' as const),
id,
record
}))
}
}
private async readAllFrames(meta: SegmentMeta): Promise<
Array<ReturnType<GenerationSegmentStore['decodeFrame']> & { offset: number; frameLen: number }>
> {
const bytes = await this.storage.readRawBytes(`${SEGMENTS_PREFIX}/${meta.file}`)
if (!bytes) {
throw new Error(
`[GenerationSegments] sealed segment ${meta.file} is MISSING — packed history is damaged; ` +
`refusing to continue silently`
)
}
const out: Array<ReturnType<GenerationSegmentStore['decodeFrame']> & { offset: number; frameLen: number }> = []
let at = MAGIC.length
const view = new DataView(bytes.buffer, bytes.byteOffset, bytes.byteLength)
while (at + FRAME_PREFIX_BYTES <= bytes.length) {
const payloadLen = view.getUint32(at, true)
const crc = view.getUint32(at + 4, true)
const payload = bytes.subarray(at + FRAME_PREFIX_BYTES, at + FRAME_PREFIX_BYTES + payloadLen)
if (payload.length !== payloadLen || crc32c(payload) !== crc) {
throw new Error(
`[GenerationSegments] frame CRC mismatch in ${meta.file} at offset ${at}` +
`packed history is damaged; refusing to serve it`
)
}
out.push({ ...this.decodeFrame(payload), offset: at, frameLen: FRAME_PREFIX_BYTES + payloadLen })
at += FRAME_PREFIX_BYTES + payloadLen
}
return out
}
/** Read one packed generation's frame via its sidecar offset (one ranged read). */
private async readFrame(
gen: number
): Promise<ReturnType<GenerationSegmentStore['decodeFrame']> | null> {
const meta = this.coveringSegment(gen)
if (!meta) return null
const idx = await this.sidecarFor(meta)
// generations ascending → binary search.
const gens = idx.generations
let lo = 0
let hi = gens.length - 1
while (lo <= hi) {
const mid = (lo + hi) >> 1
if (gens[mid][0] < gen) lo = mid + 1
else if (gens[mid][0] > gen) hi = mid - 1
else {
const [, offset, frameLen] = gens[mid]
const bytes = await this.storage.readRawBytes(`${SEGMENTS_PREFIX}/${meta.file}`)
if (!bytes) {
throw new Error(`[GenerationSegments] sealed segment ${meta.file} is MISSING`)
}
const frame = bytes.subarray(offset, offset + frameLen)
const view = new DataView(frame.buffer, frame.byteOffset, frame.byteLength)
const payloadLen = view.getUint32(0, true)
const crc = view.getUint32(4, true)
const payload = frame.subarray(FRAME_PREFIX_BYTES, FRAME_PREFIX_BYTES + payloadLen)
if (payload.length !== payloadLen || crc32c(payload) !== crc) {
throw new Error(
`[GenerationSegments] frame CRC mismatch for generation ${gen} in ${meta.file}` +
`packed history is damaged; refusing to serve it`
)
}
return this.decodeFrame(payload)
}
}
// Inside the covering range but with no frame. Two very different causes,
// and conflating them is what made this class wedge every maintenance pass
// on the affected stores.
//
// (1) A SPARSE SEGMENT — the manifest's own `frames` count is smaller than
// the span it declares. That segment was sealed by a writer that
// folded across a hole (the class this file's density law now bars).
// The segment is INTACT and nothing is lost; it simply never held this
// generation. Answering "not packed" is the honest answer, and it lets
// the caller's two-tier read decide what a genuinely absent generation
// means, instead of every compaction pass dying on a repeating throw.
// `actualRanges()` keeps such holes out of committedRanges at open, so
// in a healed store nobody asks this question in the first place.
//
// (2) A DENSE SEGMENT missing a frame it says it has — the manifest and
// the sidecar disagree about a segment that claims to be complete.
// That IS damage, and it stays loud.
if (this.isSparse(meta)) {
prodLog.warn(
`[GenerationSegments] generation ${gen} falls inside sealed segment ${meta.file}'s ` +
`declared range ${meta.firstGeneration}..${meta.lastGeneration}, but that segment ` +
`holds ${meta.frames} frame(s) for a ${meta.lastGeneration - meta.firstGeneration + 1}` +
`-generation span — it was sealed across a gap and never held this generation. ` +
`Reporting it as unpacked rather than as damage; no record is lost.`
)
return null
}
throw new Error(
`[GenerationSegments] generation ${gen} is inside sealed segment ${meta.file}'s declared ` +
`range but has no frame, and that segment declares a complete ${meta.frames}-frame ` +
`span — the manifest and the sidecar disagree; packed history is damaged`
)
}
/** The packed tier's delta for `gen` (null = not packed). */
async readDelta(gen: number): Promise<{ delta: unknown; timestamp: number } | null> {
const frame = await this.readFrame(gen)
return frame ? { delta: frame.delta, timestamp: frame.timestamp } : null
}
/** The packed tier's full record-set for `gen` (null = not packed). */
async readRecords(gen: number): Promise<FoldGeneration['records'] | null> {
const frame = await this.readFrame(gen)
return frame ? frame.records : null
}
/** One packed before-image (null = not packed OR no record for the id in that generation). */
async readRecord(gen: number, kind: 'noun' | 'verb', id: string): Promise<unknown | null> {
const frame = await this.readFrame(gen)
if (!frame) return null
const hit = frame.records.find((r) => r.kind === kind && r.id === id)
return hit ? hit.record : null
}
/**
* D3 reclaim: drop WHOLE segments whose lastGeneration < `belowGeneration`
* and bump `compactedBelow`. Partial segments are never dropped the
* boundary waits. NEVER called under the archival profile (the caller
* enforces retention semantics; this method only executes boundary drops).
*/
async dropSegmentsBelow(belowGeneration: number): Promise<{ dropped: number; compactedBelow: number }> {
const keep: SegmentMeta[] = []
const drop: SegmentMeta[] = []
for (const s of this.manifest.segments) {
;(s.lastGeneration < belowGeneration ? drop : keep).push(s)
}
if (drop.length === 0) {
return { dropped: 0, compactedBelow: this.manifest.compactedBelow }
}
const compactedBelow = Math.max(
this.manifest.compactedBelow,
drop[drop.length - 1].lastGeneration + 1
)
// Manifest first (the drop is authoritative once named), then bytes —
// a crash between leaves orphan segment files invisible to the manifest,
// harmless and re-collectable.
const next: SegmentManifest = { ...this.manifest, compactedBelow, segments: keep }
await this.storage.writeRawObject(MANIFEST_PATH, next)
await this.storage.syncRawObjects([MANIFEST_PATH])
this.manifest = next
for (const s of drop) {
await this.storage.deleteRawObject(`${SEGMENTS_PREFIX}/${s.file}`)
await this.storage.deleteRawObject(`${SEGMENTS_PREFIX}/${sidecarFileName(s.firstGeneration)}`)
this.sidecars.delete(s.file)
}
return { dropped: drop.length, compactedBelow }
}
/**
* D8 rider the packed portion of `generationDigest(g)`: a deterministic
* crc32c chain over sealed-segment checksums fully below `g`, plus the
* frame CRC of `g`'s own frame when `g` is mid-segment. O(segments), not
* O(generations); identical history identical digest on any machine.
* The live-tier portion is composed by the caller.
*/
async digestThroughPacked(g: number): Promise<number | null> {
let digest = 0
let covered = false
for (const s of this.manifest.segments) {
if (s.lastGeneration <= g) {
digest = crc32c(new TextEncoder().encode(`${digest}:${s.checksum}`))
if (s.lastGeneration === g) covered = true
} else if (s.firstGeneration <= g) {
// g is mid-segment: chain the partial prefix via g's frame CRC.
const frame = await this.readFrame(g)
if (frame === null) return null
const idx = await this.sidecarFor(s)
const upTo = idx.generations.filter(([gen]) => gen <= g)
for (const [gen, offset, frameLen] of upTo) {
digest = crc32c(new TextEncoder().encode(`${digest}:${gen}:${offset}:${frameLen}`))
}
covered = true
break
}
}
return covered || this.manifest.segments.length > 0 ? digest : null
}
}

File diff suppressed because it is too large Load diff

View file

@ -1,341 +0,0 @@
/**
* @module db/logAuthority
* @description The per-brain LOG-AUTHORITY SWITCH and its verification
* oracle the guarded adoption path for log-canonical storage.
*
* Two storage authorities exist during the adoption window:
* - `'tree'` (the default, today's behavior): the canonical record tree is
* authoritative; the generation log is a complete dual-written journal.
* - `'log'`: the generation log is authoritative for this brain; single-op
* write acks await a covering log fsync (durable-at-ack), and derived
* state treats the log as ground truth.
*
* THE SWITCH IS PER BRAIN, STORED, CHECKED AT OPEN ONLY, and ONE-DIRECTIONAL
* unless explicitly reverted by an operator. A brain flips ONLY when its
* verification oracle is green: a full replay-and-diff of the log against
* the still-authoritative tree (the read-only witness). The oracle failing
* NAMES every divergence a brain with pre-log history (records the log
* never saw) reports them as `pre-log-record` mismatches and needs a
* baseline backfill before it can ever flip.
*
* Nothing in this module mutates data: the oracle is read-only; the flip
* writes ONE artifact. Reverting = rewriting the artifact to 'tree' (the
* tree remained authoritative-quality throughout the window by dual-write).
*/
import type { FactScanHandle } from './factLog.js'
import { prodLog } from '../utils/logger.js'
import { createHash } from 'crypto'
/** Storage-root-relative path of the authority switch artifact. */
export const LOG_AUTHORITY_PATH = '_system/log-authority.json'
/** The persisted shape of the authority switch. */
export interface LogAuthorityRecord {
/** Which store is authoritative for this brain. */
authority: 'tree' | 'log'
/** When the flip happened (ms epoch). Absent while authority = 'tree'. */
flippedAt?: number
/** The oracle verdict that justified the flip (summary, not the full report). */
oracle?: {
verifiedAt: number
generationsScanned: number
nounsChecked: number
verbsChecked: number
}
/**
* Recorded when an OPEN-TIME adoption attempt (the 10.0.0 fleet default)
* was refused the oracle could not go green. Keeps subsequent opens
* cheap; an operator re-runs adoptLogAuthority() after resolving it.
*/
adoptRefusal?: { at: number; reason: string }
}
/** The narrow storage surface this module needs. */
export interface LogAuthorityStorage {
readRawObject(path: string): Promise<unknown | null>
writeRawObject(path: string, data: unknown): Promise<void>
syncRawObjects(paths: string[]): Promise<void>
getNouns(opts: {
pagination: { limit: number; offset?: number; cursor?: string }
}): Promise<{ items: unknown[]; hasMore?: boolean; nextCursor?: string }>
getNounMetadata(id: string): Promise<unknown | null>
}
/** One divergence found by the oracle. */
export interface OracleMismatch {
id: string
kind: 'noun' | 'verb'
reason:
| 'pre-log-record' // canonical row the log never saw — needs baseline backfill
| 'state-differs' // latest log after-image ≠ canonical bytes
| 'log-live-canonical-absent' // log says live, canonical has no record
| 'log-tombstone-canonical-present' // log says deleted, canonical still has it
}
/** The oracle's full report. */
export interface OracleReport {
verdict: 'green' | 'red'
generationsScanned: number
nounsChecked: number
verbsChecked: number
matched: number
mismatches: OracleMismatch[]
/** Mismatch listing is capped; the counts above are always complete. */
mismatchListTruncated: boolean
}
const MISMATCH_LIST_CAP = 200
/** Read the stored authority (absent artifact = 'tree', the safe default). */
export async function readLogAuthority(
storage: Pick<LogAuthorityStorage, 'readRawObject'>
): Promise<LogAuthorityRecord> {
const raw = (await storage
.readRawObject(LOG_AUTHORITY_PATH)
.catch(() => null)) as LogAuthorityRecord | null
if (raw && (raw.authority === 'log' || raw.authority === 'tree')) return raw
return { authority: 'tree' }
}
/**
* Normalize a canonical noun record to its ENTITY TRUTH before diffing:
* the canonical vector-file wrapper denormalizes derived index residue
* (`connections` HNSW graph edges; `level` the node's random skip-list
* level) that the generation log deliberately does NOT carry (projections
* own their own rebuild paths). Digesting the residue would report false
* `state-differs` on ~any brain whose HNSW assigned a nonzero level. Both
* sides of every oracle comparison pass through this normalizer.
*/
export function nounEntityTruth(record: {
metadata: unknown
vector: unknown
}): { metadata: unknown; vector: unknown } {
const v = record.vector
if (v && typeof v === 'object' && !Array.isArray(v)) {
const { connections: _c, level: _l, ...entity } = v as Record<string, unknown>
return { metadata: record.metadata, vector: entity }
}
return { metadata: record.metadata, vector: v }
}
/**
* Stable content hash of a stored record for diffing key-sorted JSON so
* property order can never fake a divergence.
*/
export function recordDigest(record: unknown): string {
const stable = (v: unknown): unknown => {
if (Array.isArray(v)) return v.map(stable)
if (v && typeof v === 'object') {
const out: Record<string, unknown> = {}
for (const k of Object.keys(v as Record<string, unknown>).sort()) {
out[k] = stable((v as Record<string, unknown>)[k])
}
return out
}
return v
}
return createHash('sha256').update(JSON.stringify(stable(record))).digest('hex')
}
/**
* THE VERIFICATION ORACLE: replay the fact log's noun records and diff the
* final state per id against the canonical tree (the witness). Read-only;
* bounded memory (id {tombstoned, digest} digests, never bodies).
*
* Verdict law: 'green' iff EVERY canonical row's latest state is exactly
* reproduced by the log AND the log claims nothing canonical denies. A
* brain older than its log reports its unlogged rows as `pre-log-record`
* mismatches the named cure is a baseline backfill, never a silent pass.
*/
export async function runLogCompletenessOracle(args: {
storage: LogAuthorityStorage
scanFacts: () => FactScanHandle | null
/** Digest the canonical record the same way the log's after-image is digested. */
canonicalNounDigest: (id: string) => Promise<string | null>
/** Digest a log after-image record's payload. */
factRecordDigest: (record: unknown) => string
/**
* Verb legs (optional until every owner wires them): the canonical verb
* digest + the paged verb enumeration. When ABSENT, the oracle counts NO
* verbs and says so via verbsChecked = 0 an honest partial verdict,
* never a silent full-pass claim.
*/
canonicalVerbDigest?: (id: string) => Promise<string | null>
getVerbs?: (opts: {
pagination: { limit: number; offset?: number; cursor?: string }
}) => Promise<{ items: unknown[]; hasMore?: boolean; nextCursor?: string }>
/**
* Cap on the LISTED mismatches (counts are always complete). Defaults to
* the wire-friendly {@link MISMATCH_LIST_CAP}; the adoption backfill passes
* `Infinity` so ONE scan yields the ENTIRE curable set a production
* brain with a 12.7k-row pre-log baseline once advanced only 800 rows per
* adoption call because each pass could see (and cure) at most 200.
*/
mismatchListCap?: number
}): Promise<OracleReport> {
const listCap = args.mismatchListCap ?? MISMATCH_LIST_CAP
const report: OracleReport = {
verdict: 'red',
generationsScanned: 0,
nounsChecked: 0,
verbsChecked: 0,
matched: 0,
mismatches: [],
mismatchListTruncated: false
}
const addMismatch = (m: OracleMismatch): void => {
if (report.mismatches.length < listCap) report.mismatches.push(m)
else report.mismatchListTruncated = true
}
// Pass 1: fold the log — latest state per noun id (digest or tombstone).
const scan = args.scanFacts()
if (!scan) {
// No fact log on this store: nothing can be verified — red, loudly.
prodLog.warn('[logAuthority] oracle: this store has no fact log — cannot verify, verdict red')
return report
}
const logState = new Map<string, { tombstoned: boolean; digest: string | null }>()
const verbLogState = new Map<string, { tombstoned: boolean; digest: string | null }>()
for await (const batch of scan.batches()) {
for (const fact of batch.facts) {
report.generationsScanned++
for (const op of fact.ops) {
const state =
op.record === null
? { tombstoned: true, digest: null }
: { tombstoned: false, digest: args.factRecordDigest(op.record) }
if (op.kind === 'noun') logState.set(op.id, state)
else verbLogState.set(op.id, state)
}
}
}
// Pass 2: walk canonical (paged) and diff.
const seenCanonical = new Set<string>()
const PAGE = 500
let offset = 0
let cursor: string | undefined
for (;;) {
const page = await args.storage.getNouns({
pagination: cursor ? { limit: PAGE, cursor } : { limit: PAGE, offset }
})
for (const item of page.items) {
const id = (item as { id: string }).id
seenCanonical.add(id)
report.nounsChecked++
const inLog = logState.get(id)
if (!inLog) {
addMismatch({ id, kind: 'noun', reason: 'pre-log-record' })
continue
}
if (inLog.tombstoned) {
addMismatch({ id, kind: 'noun', reason: 'log-tombstone-canonical-present' })
continue
}
const canonicalDigest = await args.canonicalNounDigest(id)
if (canonicalDigest === null) {
addMismatch({ id, kind: 'noun', reason: 'pre-log-record' })
continue
}
if (canonicalDigest === inLog.digest) report.matched++
else addMismatch({ id, kind: 'noun', reason: 'state-differs' })
}
if (!page.hasMore || page.items.length === 0) break
if (page.nextCursor) cursor = page.nextCursor
else offset += page.items.length
}
// Pass 3: log-live ids canonical never showed us.
for (const [id, state] of logState) {
if (!state.tombstoned && !seenCanonical.has(id)) {
addMismatch({ id, kind: 'noun', reason: 'log-live-canonical-absent' })
}
}
// Verb passes — only when the owner wired the verb legs; otherwise the
// report says verbsChecked: 0, an honest partial scope, never a claim.
if (args.canonicalVerbDigest && args.getVerbs) {
const seenVerbs = new Set<string>()
let vOffset = 0
let vCursor: string | undefined
for (;;) {
const page = await args.getVerbs({
pagination: vCursor ? { limit: PAGE, cursor: vCursor } : { limit: PAGE, offset: vOffset }
})
for (const item of page.items) {
const id = (item as { id: string }).id
seenVerbs.add(id)
report.verbsChecked++
const inLog = verbLogState.get(id)
if (!inLog) {
addMismatch({ id, kind: 'verb', reason: 'pre-log-record' })
continue
}
if (inLog.tombstoned) {
addMismatch({ id, kind: 'verb', reason: 'log-tombstone-canonical-present' })
continue
}
const canonical = await args.canonicalVerbDigest(id)
if (canonical === null) {
addMismatch({ id, kind: 'verb', reason: 'pre-log-record' })
continue
}
if (canonical === inLog.digest) report.matched++
else addMismatch({ id, kind: 'verb', reason: 'state-differs' })
}
if (!page.hasMore || page.items.length === 0) break
if (page.nextCursor) vCursor = page.nextCursor
else vOffset += page.items.length
}
for (const [id, state] of verbLogState) {
if (!state.tombstoned && !seenVerbs.has(id)) {
addMismatch({ id, kind: 'verb', reason: 'log-live-canonical-absent' })
}
}
}
const totalMismatches =
report.mismatches.length + (report.mismatchListTruncated ? 1 : 0)
report.verdict = totalMismatches === 0 ? 'green' : 'red'
return report
}
/**
* Flip this brain's authority to the log REFUSES unless the supplied
* oracle report is green (the caller runs the oracle; the flip records its
* summary). Writes + fsyncs the switch artifact; the mode takes full effect
* at the NEXT open (checked-at-open-only law), except durable-at-ack which
* the owner may enable immediately.
*/
export async function flipToLogAuthority(
storage: Pick<LogAuthorityStorage, 'writeRawObject' | 'syncRawObjects'>,
oracle: OracleReport
): Promise<LogAuthorityRecord> {
if (oracle.verdict !== 'green') {
throw new Error(
`log-authority flip refused: the verification oracle is RED ` +
`(${oracle.mismatches.length}${oracle.mismatchListTruncated ? '+' : ''} mismatches; ` +
`first: ${oracle.mismatches[0] ? `${oracle.mismatches[0].reason} on ${oracle.mismatches[0].id}` : 'n/a'}). ` +
`A brain flips only on green — fix the divergences (pre-log records need a baseline backfill) and re-run.`
)
}
const record: LogAuthorityRecord = {
authority: 'log',
flippedAt: Date.now(),
oracle: {
verifiedAt: Date.now(),
generationsScanned: oracle.generationsScanned,
nounsChecked: oracle.nounsChecked,
verbsChecked: oracle.verbsChecked
}
}
await storage.writeRawObject(LOG_AUTHORITY_PATH, record)
await storage.syncRawObjects([LOG_AUTHORITY_PATH])
prodLog.info(
`[logAuthority] this brain's storage authority is now the generation log ` +
`(oracle green over ${oracle.nounsChecked} nouns / ${oracle.generationsScanned} generations)`
)
return record
}

View file

@ -27,7 +27,7 @@
import { Entity, Relation, Result } from '../types/brainy.types.js'
import { NounType, VerbType } from '../types/graphTypes.js'
import { StorageAdapter, HNSWVerbWithMetadata } from '../coreTypes.js'
import { StorageAdapter } from '../coreTypes.js'
import { getBrainyVersion } from '../utils/version.js'
import { TxOperation } from './types.js'
@ -94,85 +94,8 @@ export interface ExportOptions {
includeContent?: boolean
/** Include `visibility:'system'` entities (e.g. the VFS root) (default: false). */
includeSystem?: boolean
/**
* Admit BOTH hidden visibility tiers `'internal'` AND `'system'` into the
* whole-brain/predicate candidate set, in EITHER `enumeration` mode (default:
* false today's behavior is byte-identical). `includeSystem` alone only ever
* reached `'system'` for a structural selector's per-entity gate; whole-brain/
* predicate enumeration never forwarded it into the candidate walk at all, so a
* hidden-tier row could never survive a whole-brain export regardless of any
* flag the gap this option closes. `includeHidden: true` IMPLIES
* `includeSystem: true` (both tiers are admitted together; there is no
* "system but not internal" combination via this flag) `includeSystem` on
* its own keeps its narrower, pre-existing meaning for back-compat.
*
* Migration-grade exports set `includeHidden: true` a complete-canon export
* must carry every visibility tier; consumer-facing exports leave it off.
*/
includeHidden?: boolean
/** Which edges to include (default: `'induced'`). */
edges?: 'induced' | 'incident' | 'none'
/**
* How the whole-brain / predicate selector (no `ids`/`collection`/`connected`/
* `vfsPath`) resolves its candidate id set:
*
* - `'index'` (DEFAULT unchanged behavior) the generation-correct
* paginated `find()` walk. Fast (O(matches), not O(N)) but rides the
* metadata index as an acceleration structure: a lost/stale index posting
* can silently OMIT a canonical record, and a stale posting pointing at a
* record that no longer matches can silently produce a phantom (dropped
* later by the same predicate re-check `'canonical'` mode also runs, so
* phantoms never reach `entities` but they ARE lost silently unless
* {@link reportIndexDrift} is set).
* - `'canonical'` walks every live noun/verb directly off the storage
* adapter's canonical shard layout (`storage.getNouns()`/`getVerbs()`
* the same primitive `repairIndex()`'s recount and every index-heal walk
* use), then applies the selector as a plain predicate over the walked
* records. This GUARANTEES canon-completeness the metadata/graph
* indexes are never consulted, so their corruption cannot hide a live
* record at the cost of an O(N) walk regardless of selector
* selectivity (unlike the index path's O(matches)). Structural selectors
* (`ids`/`collection`/`connected`/`vfsPath`) resolve their node set
* exactly as in `'index'` mode either way (they never rode the metadata
* index); `'canonical'` additionally walks relations canonically for
* EVERY selector, since a lost adjacency-index posting can hide a
* relation regardless of how the node set was produced. Requires a
* storage adapter and the CURRENT generation throws on a historical
* `asOf()` view or a speculative `with()` overlay (see
* {@link CanonicalEnumerationUnavailableError}), because the canonical
* walk has no notion of "as of a past generation."
*/
enumeration?: 'index' | 'canonical'
/**
* Only meaningful with `enumeration:'canonical'` (ignored otherwise): ALSO
* run the `'index'` enumeration in parallel and diff it against the
* canonical ground truth, attaching the result as {@link PortableGraph.drift}.
* Never auto-heals anything this is migration-audit evidence, reported
* loudly (`console.warn` with the counts) whenever either list is non-empty,
* never silently. Default: false.
*/
reportIndexDrift?: boolean
}
/**
* @description Index-vs-canonical drift for one `export({ enumeration: 'canonical',
* reportIndexDrift: true })` call. Only populated for the whole-brain / predicate
* selector (structural selectors never consulted the metadata index for their node
* set, so there is nothing to diff both lists are empty for those).
*/
export interface ExportIndexDrift {
/**
* Ids the canonical storage walk confirmed (live, selector-matching) that the
* index-based `find()` enumeration did NOT return canonical records the
* metadata index has lost track of.
*/
canonicalOnly: string[]
/**
* Ids the index-based `find()` enumeration returned for this selector that
* canonical ground truth (the storage walk + the same predicate check) does
* NOT support phantom index rows (stale or cross-bucket postings).
*/
indexOnly: string[]
}
/** Controls how a `PortableGraph` is applied on `import()`. */
@ -228,8 +151,6 @@ export interface PortableGraph {
relations: PortableGraphRelation[]
blobs?: Record<string, string>
danglingIds?: string[]
/** Present only when `export()` was called with `reportIndexDrift: true`. */
drift?: ExportIndexDrift
stats: { entityCount: number; relationCount: number; blobCount: number; vectorDimensions?: number }
}
@ -350,19 +271,11 @@ export function validatePortableGraph(data: unknown): PortableGraphValidation {
/**
* @description Serialize part or all of a graph (read through `reader` at its pinned
* generation) into a portable `PortableGraph` document.
*
* `enumeration:'canonical'` (see {@link ExportOptions.enumeration}) requires a
* storage adapter and the current generation: it throws
* {@link CanonicalEnumerationUnavailableError} if `storage` is absent, and the
* caller (`Db.export()`) throws the same error before this runs if the view is
* historical or a speculative overlay the canonical storage walk has no
* generation parameter, so it can only ever answer "as of right now."
*
* @param reader - Generation-correct read surface (`Db` or `Brainy`).
* @param storage - Storage adapter (VFS blob bytes when `includeContent`; the
* canonical noun/verb walk when `enumeration:'canonical'`).
* @param storage - Storage adapter (used only for VFS blob bytes when `includeContent`).
* @param selector - WHAT to export (omit for the whole brain).
* @param options - HOW to export (vectors / file bytes / edge policy / enumeration mode).
* @param options - HOW to export (vectors / file bytes / edge policy).
* @param dimensions - Embedding dimensionality for the manifest.
*/
export async function exportGraph<T = any>(
reader: PortableGraphReader<T>,
@ -374,81 +287,28 @@ export async function exportGraph<T = any>(
includeVectors = false,
includeContent = false,
includeSystem = false,
includeHidden = false,
edges = 'induced',
enumeration = 'index',
reportIndexDrift = false
edges = 'induced'
} = options
if (enumeration === 'canonical' && !storage) {
throw new Error(
`export(): enumeration:'canonical' requires a storage adapter, but none was supplied ` +
`to this reader. Use enumeration:'index' (the default), or export through a Db/Brainy ` +
`that carries its storage adapter.`
)
}
const wantDrift = enumeration === 'canonical' && reportIndexDrift
// includeHidden IMPLIES includeSystem (see ExportOptions.includeHidden's JSDoc) — every
// system-tier gate below reads THIS combined value, never the raw option, so
// `includeHidden` alone is always sufficient to see system-tier rows too.
const effectiveIncludeSystem = includeSystem || includeHidden
// 1. Resolve the node-id set (+ the index's raw candidate set, only when diffing it).
// Both `enumerateAllCanonical` and `enumerateAllIndexed` receive the SAME
// `effectiveIncludeSystem`/`includeHidden` pair below, so a drift diff can never
// contain tier-policy noise — only genuine index-vs-canonical disagreement.
const { idSet, indexCandidateIds } = await resolveSelector(
reader,
storage,
selector,
effectiveIncludeSystem,
enumeration,
wantDrift,
includeHidden
)
// 1. Resolve the node-id set.
const idSet = await resolveSelector(reader, selector, includeSystem)
// 2. Read canonical entities (reserved fields top-level), applying any predicate filter.
// Identical for both enumeration modes: 'canonical' only changes WHICH ids reach this
// loop, never how a candidate is verified — so the two modes can disagree on candidacy,
// never on what counts as a match.
const usePredicate = hasPredicate(selector)
const entityMap = new Map<string, Entity<T>>()
const entities: PortableGraphEntity[] = []
for (const id of idSet) {
const e = await reader.get(id, { includeVectors })
if (!e) continue
if (!effectiveIncludeSystem && (e as any).visibility === 'system') continue
if (!includeSystem && (e as any).visibility === 'system') continue
if (usePredicate && !matchesPredicate(e, selector)) continue
entityMap.set(id, e)
entities.push(toPortableGraphEntity(e, includeVectors))
}
const keptIds = new Set(entityMap.keys())
// 2b. Finalize the drift report now that ground truth (keptIds) is known.
let drift: ExportIndexDrift | undefined
if (wantDrift) {
const indexIds = indexCandidateIds ?? new Set<string>()
const canonicalOnly = [...keptIds].filter((id) => !indexIds.has(id))
const indexOnly = [...indexIds].filter((id) => !keptIds.has(id))
drift = { canonicalOnly, indexOnly }
if (canonicalOnly.length > 0 || indexOnly.length > 0) {
console.warn(
`[Brainy] export() index drift: ${canonicalOnly.length} canonical-only id(s) ` +
`(canon-present, the index-based enumeration missed them) and ${indexOnly.length} ` +
`index-only id(s) (index-visible, canon-absent — phantom rows). ` +
`See the returned PortableGraph's 'drift' field for the exact ids. Nothing was ` +
`auto-healed — run brain.repairIndex() to reconcile the metadata index.`
)
}
}
// 3. Edges per policy. Canonical mode ALSO walks verbs canonically for every
// selector (not just whole-brain) — a lost adjacency-index posting can hide a
// relation regardless of how the node set was produced.
const { relations, danglingIds } =
enumeration === 'canonical'
? await collectEdgesCanonical(storage!, keptIds, edges, effectiveIncludeSystem, includeHidden)
: await collectEdges(reader, keptIds, edges, includeHidden)
// 3. Edges per policy.
const { relations, danglingIds } = await collectEdges(reader, keptIds, edges)
// 4. VFS blob bytes (only when requested).
let blobs: Record<string, string> | undefined
@ -470,7 +330,6 @@ export async function exportGraph<T = any>(
relations,
...(blobs && blobCount > 0 ? { blobs } : {}),
...(danglingIds && danglingIds.length > 0 ? { danglingIds } : {}),
...(drift ? { drift } : {}),
stats: {
entityCount: entities.length,
relationCount: relations.length,
@ -618,36 +477,12 @@ function hasPredicate(s: ExportSelector): boolean {
)
}
/**
* @param reader - Generation-correct read surface.
* @param storage - Storage adapter (only touched when `enumeration:'canonical'`
* resolves the whole-brain/predicate branch).
* @param s - The export selector.
* @param includeSystem - The ALREADY-COMBINED `includeSystem || includeHidden` value
* (see `exportGraph`'s `effectiveIncludeSystem`) — whether `visibility:'system'`
* entities are wanted.
* @param enumeration - `'index'` (default) or `'canonical'` see {@link ExportOptions.enumeration}.
* Only affects the whole-brain/predicate branch (the `else` below): structural
* selectors (`ids`/`collection`/`connected`/`vfsPath`) never rode the metadata
* index for their node set, so they resolve identically either way.
* @param wantIndexCandidates - When true (only meaningful with `enumeration:'canonical'`
* on the whole-brain/predicate branch), ALSO run the index-based walk and return
* its raw candidate set as `indexCandidateIds`, for {@link ExportIndexDrift}.
* @param includeHidden - Whether `visibility:'internal'` entities are ALSO wanted
* (see {@link ExportOptions.includeHidden}). Threaded to BOTH enumeration
* functions identically so a drift diff never contains tier-policy noise.
*/
async function resolveSelector<T>(
reader: PortableGraphReader<T>,
storage: StorageAdapter | undefined,
s: ExportSelector,
includeSystem: boolean,
enumeration: 'index' | 'canonical',
wantIndexCandidates: boolean,
includeHidden: boolean
): Promise<{ idSet: Set<string>; indexCandidateIds?: Set<string> }> {
includeSystem: boolean
): Promise<Set<string>> {
let idSet: Set<string>
let indexCandidateIds: Set<string> | undefined
if (s.ids && s.ids.length) {
idSet = new Set(s.ids)
} else if (s.collection ?? s.memberOf) {
@ -656,45 +491,20 @@ async function resolveSelector<T>(
idSet = await resolveConnected(reader, s.connected)
} else if (s.vfsPath) {
idSet = await resolveVfsPath(reader, s.vfsPath, s.recursive ?? true, s.depth)
} else if (enumeration === 'canonical') {
idSet = await enumerateAllCanonical(storage!, includeSystem, includeHidden)
if (wantIndexCandidates) indexCandidateIds = await enumerateAllIndexed(reader, s, includeHidden)
} else {
idSet = await enumerateAllIndexed(reader, s, includeHidden)
idSet = await enumerateAll(reader, s)
}
if (!includeSystem) idSet.delete(VFS_ROOT_ID)
return { idSet, indexCandidateIds }
return idSet
}
/**
* @description Whole-brain / predicate enumeration via generation-correct
* paginated `find()`. The metadata index is an acceleration structure over
* this candidate set see {@link enumerateAllCanonical} for the storage-level
* counterpart that never consults it.
*
* @param includeHidden - When true, forwards `includeInternal: true` AND
* `includeSystem: true` into the SAME `find()` call `find()` supports both
* flags simultaneously (confirmed via `FindParams.includeInternal`/`includeSystem`
* and `Brainy`'s `resolveHiddenIds`/`excludedVisibilityTiers`), so ONE pass
* reaches both hidden tiers; no per-tier union pass is needed. When false
* (default), neither flag is forwarded the pre-existing behavior, preserved
* byte-identically for back-compat (`ExportOptions.includeSystem` alone never
* reached this far; see {@link ExportOptions.includeHidden}'s JSDoc).
*/
async function enumerateAllIndexed<T>(
reader: PortableGraphReader<T>,
s: ExportSelector,
includeHidden = false
): Promise<Set<string>> {
/** Whole-brain / predicate enumeration via generation-correct paginated `find()`. */
async function enumerateAll<T>(reader: PortableGraphReader<T>, s: ExportSelector): Promise<Set<string>> {
const params: any = {}
if (s.type !== undefined) params.type = s.type
if (s.subtype !== undefined) params.subtype = s.subtype
if (s.where !== undefined) params.where = s.where
if (s.service !== undefined) params.service = s.service
if (includeHidden) {
params.includeInternal = true
params.includeSystem = true
}
const ids = new Set<string>()
let offset = 0
// eslint-disable-next-line no-constant-condition
@ -707,52 +517,6 @@ async function enumerateAllIndexed<T>(
return ids
}
/**
* @description Canonical (storage-level) counterpart of {@link enumerateAllIndexed}:
* walks every live noun directly off the storage adapter's canonical shard layout
* (`storage.getNouns()` the same primitive `repairIndex()`'s recount and every
* index-heal walk use) instead of going through the metadata index. Guarantees
* canon-completeness a lost or stale metadata-index posting cannot cause a
* canonical record to be silently missing from the returned set at the cost of
* an O(N) walk regardless of selector selectivity (unlike the index path's
* O(matches)). Returns the RAW candidate id set; `exportGraph`'s caller applies
* `matchesPredicate` per-entity via `reader.get()` afterward, exactly as the index
* path does, so both paths share one predicate-evaluation code path and can only
* disagree on candidacy, never on what a match means.
*
* Mirrors `find()`'s hidden-tier policy given the SAME `includeSystem`/`includeHidden`
* pair (see {@link enumerateAllIndexed}) so the two enumeration modes produce
* identical id sets when the index is healthy, at ANY tier-visibility setting.
*
* @param includeSystem - The ALREADY-COMBINED `includeSystem || includeHidden` value.
* @param includeHidden - Whether `'internal'`-tier nouns are ALSO admitted.
*/
async function enumerateAllCanonical(
storage: StorageAdapter,
includeSystem = false,
includeHidden = false
): Promise<Set<string>> {
const ids = new Set<string>()
let offset = 0
let cursor: string | undefined
// eslint-disable-next-line no-constant-condition
while (true) {
const page = await storage.getNouns({ pagination: { limit: ENUM_PAGE, offset, cursor } })
for (const item of page.items) {
if (item.visibility === 'internal' && !includeHidden) continue
if (item.visibility === 'system' && !includeSystem) continue
ids.add(item.id)
}
if (!page.hasMore || page.items.length === 0) break
if (page.nextCursor !== undefined) {
cursor = page.nextCursor
} else {
offset += ENUM_PAGE
}
}
return ids
}
async function resolveCollectionSubtree<T>(
reader: PortableGraphReader<T>,
rootId: string,
@ -914,27 +678,19 @@ function toPortableGraphRelation<T>(r: Relation<T>): PortableGraphRelation {
return br
}
/**
* @param includeHidden - When true, forwards `includeInternal`/`includeSystem` into
* every `related()` call so hidden-tier relations reach candidacy too mirrors
* {@link enumerateAllIndexed}'s `includeHidden` handling, and preserves back-compat
* when false/omitted (the pre-existing, unconditional hidden-tier exclusion).
*/
async function collectEdges<T>(
reader: PortableGraphReader<T>,
idSet: Set<string>,
edges: 'induced' | 'incident' | 'none',
includeHidden = false
edges: 'induced' | 'incident' | 'none'
): Promise<{ relations: PortableGraphRelation[]; danglingIds?: string[] }> {
if (edges === 'none') return { relations: [] }
const tierOptIn = includeHidden ? { includeInternal: true, includeSystem: true } : {}
const relations: PortableGraphRelation[] = []
const dangling = new Set<string>()
const seen = new Set<string>()
for (const id of idSet) {
const rels = await reader.related({ from: id, limit: RELATION_FETCH_LIMIT, ...tierOptIn })
const rels = await reader.related({ from: id, limit: RELATION_FETCH_LIMIT })
for (const r of rels) {
if (seen.has(r.id)) continue
const toIn = idSet.has(r.to)
@ -947,7 +703,7 @@ async function collectEdges<T>(
if (edges === 'incident') {
for (const id of idSet) {
const rels = await reader.related({ to: id, limit: RELATION_FETCH_LIMIT, ...tierOptIn })
const rels = await reader.related({ to: id, limit: RELATION_FETCH_LIMIT })
for (const r of rels) {
if (seen.has(r.id)) continue
if (!idSet.has(r.from)) {
@ -962,81 +718,6 @@ async function collectEdges<T>(
return dangling.size > 0 ? { relations, danglingIds: Array.from(dangling) } : { relations }
}
/** Converts a canonical verb record (as returned by `storage.getVerbs()`) into the wire shape. */
function hnswVerbToPortableGraphRelation(v: HNSWVerbWithMetadata): PortableGraphRelation {
const br: PortableGraphRelation = { id: v.id, from: v.sourceId, to: v.targetId, type: v.verb as string }
if (v.subtype !== undefined) br.subtype = v.subtype
if (v.visibility !== undefined && v.visibility !== 'public') br.visibility = v.visibility
if (v.weight !== undefined) br.weight = v.weight
if (v.confidence !== undefined) br.confidence = v.confidence
if (v.metadata && Object.keys(v.metadata as any).length) br.metadata = v.metadata
return br
}
/**
* @description Canonical (storage-level) counterpart of {@link collectEdges}:
* walks every live verb directly off `storage.getVerbs()` the same primitive
* `repairIndex()`'s recount and every index-heal walk use instead of the graph
* adjacency index (`reader.related()`), so a lost/stale adjacency posting cannot
* cause a canonical relationship to be silently dropped from the export. Used for
* EVERY selector in `enumeration:'canonical'` mode, not just the whole-brain
* branch: relations can be blinded by adjacency-index corruption regardless of
* how `idSet` (the kept node ids) was produced.
*
* Mirrors `related()`'s default hidden-tier policy (hides `'internal'` and
* `'system'` unless `includeHidden`/`includeSystem` say otherwise) so the two
* enumeration modes produce identical relation sets when the index is healthy.
*
* @param includeSystem - Whether `'system'`-tier verbs are admitted (the
* caller passes the ALREADY-combined `includeSystem || includeHidden` value
* see `exportGraph`'s `effectiveIncludeSystem`).
* @param includeHidden - Whether `'internal'`-tier verbs are ALSO admitted.
*/
async function collectEdgesCanonical(
storage: StorageAdapter,
idSet: Set<string>,
edges: 'induced' | 'incident' | 'none',
includeSystem = false,
includeHidden = false
): Promise<{ relations: PortableGraphRelation[]; danglingIds?: string[] }> {
if (edges === 'none') return { relations: [] }
const relations: PortableGraphRelation[] = []
const dangling = new Set<string>()
const seen = new Set<string>()
let offset = 0
let cursor: string | undefined
// eslint-disable-next-line no-constant-condition
while (true) {
const page = await storage.getVerbs({ pagination: { limit: ENUM_PAGE, offset, cursor } })
for (const v of page.items) {
if (seen.has(v.id)) continue
if (v.visibility === 'internal' && !includeHidden) continue
if (v.visibility === 'system' && !includeSystem) continue
const fromIn = idSet.has(v.sourceId)
const toIn = idSet.has(v.targetId)
if (edges === 'induced') {
if (!fromIn || !toIn) continue
} else if (!fromIn && !toIn) {
continue // 'incident': neither endpoint kept — irrelevant to this export
}
if (fromIn && !toIn) dangling.add(v.targetId)
if (toIn && !fromIn) dangling.add(v.sourceId)
seen.add(v.id)
relations.push(hnswVerbToPortableGraphRelation(v))
}
if (!page.hasMore || page.items.length === 0) break
if (page.nextCursor !== undefined) {
cursor = page.nextCursor
} else {
offset += ENUM_PAGE
}
}
return dangling.size > 0 ? { relations, danglingIds: Array.from(dangling) } : { relations }
}
async function collectBlobs<T>(
storage: StorageAdapter | undefined,
entityMap: Map<string, Entity<T>>

View file

@ -121,13 +121,9 @@ export interface TransactOptions {
* with the batch: `max(30 000, opCount × 2 000)` production imports on
* network-attached disks measure ~2 s per operation, so a flat 30 s budget
* silently capped honest bulk work at ~15 operations. A tripped budget
* rolls the whole batch back and throws a `TransactionTimeoutError` naming
* the operation it stopped at, the batch size, and the elapsed/budget
* times. That error is retryable-with-latch, never hot-retry: its
* `retryable` field says a later attempt may succeed, its
* `hotRetryUnsafe` field says an immediate identical retry re-pays the
* full cost that just timed out callers must latch and back off, never
* loop.
* rolls the whole batch back and throws a retryable
* `TransactionTimeoutError` naming the operation it stopped at, the batch
* size, and the elapsed/budget times.
*/
timeoutMs?: number
}
@ -180,15 +176,6 @@ export interface CompactHistoryOptions {
* of each surviving generation's serialized record set (`GenerationDelta.bytes`).
*/
maxBytes?: number
/**
* Stop reclaiming after this many milliseconds even if caps are still
* exceeded (8.9.0). Compaction is maintenance a bounded pass keeps
* `close()` (and any explicit maintenance window) from stalling on a large
* backlog; the next pass resumes where this one stopped (reclamation is
* oldest-first, so an early stop is always a consistent prefix). Unset =
* run to completion.
*/
timeBudgetMs?: number
}
/**
@ -412,17 +399,6 @@ export interface TxLogEntry {
timestamp: number
/** Transaction metadata, when supplied to `transact()`. */
meta?: Record<string, unknown>
/**
* WHO committed. Absent = a user write (every pre-existing consumer's
* reading stays exact). Engine-originated commits stamp themselves
* `'system:embed-landing'` (the deferred vector landing),
* `'system:adoption-backfill'` (baseline re-commits), `'system:reconcile'`
* (the attested per-id divergence door) so activity feeds can filter on
* fact instead of collapsing near-in-time entries (a consumer refused that
* heuristic as a quiet loss, correctly; this field is the honest cure).
* The same stamp rides the commit fact's meta, so log and tx-log agree.
*/
origin?: string
}
// ============================================================================
@ -450,21 +426,6 @@ export interface GenerationStorage {
deleteRawObject(path: string): Promise<void>
/** List raw object paths under a prefix (normalized, `.gz`-stripped). */
listRawObjects(prefix: string): Promise<string[]>
/**
* OPTIONAL: the IMMEDIATE child directory names under a prefix one level,
* no recursion, no file paths.
*
* Why it exists: discovering which generations are on disk needs only the
* top-level directory NAMES under `_generations/`, but the only door for it
* was `listRawObjects`, which recurses the whole tree and returns every file
* in every generation. On a store with a long history that is a full walk of
* the entire generation log, paid on EVERY open, to learn a set of integers
* the directory names already spell out.
*
* An adapter without this door keeps working the caller falls back to the
* recursive listing.
*/
listRawPrefixes?(prefix: string): Promise<string[]>
/** Remove every object under a prefix (and the directory itself on disk). */
removeRawPrefix(prefix: string): Promise<void>
/** Durability barrier: fsync the given object paths (no-op in memory). */
@ -488,28 +449,6 @@ export interface GenerationStorage {
/** @see beginWriteBarrier — fsync every canonical write since begin. */
flushWriteBarrier?(): Promise<void>
/**
* OPTIONAL fold-checkpoint durability barrier: make the listed entities'
* CANONICAL live objects durable fsync each present metadata/vector file
* AND the parent directory entry of each absent one (so a delete is as
* durable as a write). The generation store may only advance the fold
* checkpoint (`_system/fold-checkpoint.json`) after this resolves; the
* checkpoint bounds crash recovery's log fold to `(checkpoint, head]`.
* Adapters whose writes are durable per-call may leave this undefined
* the store then treats canonical durability as immediate.
*/
syncEntityCanonical?(nouns: string[], verbs: string[]): Promise<void>
/**
* OPTIONAL writer fence: throw `BRAINY_WRITER_FENCED` when this instance
* no longer owns the store's writer lock (an operator force-takeover or a
* removed lock file). Called at every flush commit and transact barrier
* one small read per commit window so an evicted writer fails loudly on
* its next commit instead of split-braining the store. Adapters without a
* cross-process lock model omit it.
*/
assertWriterFenceHeld?(): Promise<void>
/** Read an entity's raw stored metadata+vector objects. */
readNounRaw(id: string): Promise<{ metadata: any | null; vector: any | null }>
/** Restore an entity's raw stored objects (`null` part ⇒ delete that file). */

View file

@ -61,44 +61,41 @@ export class UnsupportedWhereOperatorError extends Error {
* @returns The field's value, or `undefined` when absent.
*/
export function resolveEntityField(entity: Entity, field: string): unknown {
// THE ONE ADDRESSING LAW (sealed 2026-08-03): `system.<field>` reads the
// entity scalar; bare and `metadata.`-prefixed names read the user's
// metadata bag (dotted paths traverse INSIDE the bag). The old bare-name
// switch over system fields is dead — bare `createdAt` is the user's own
// field now; the engine scalar is `system.createdAt`. Plumbing (vector,
// connections, level, data, _rev) is invisible: no spelling reaches it.
if (field.startsWith('system.')) {
switch (field.slice('system.'.length)) {
case 'type':
return entity.type
case 'subtype':
return entity.subtype
case 'id':
return entity.id
case 'createdAt':
return entity.createdAt
case 'updatedAt':
return entity.updatedAt
case 'service':
return entity.service
case 'createdBy':
return entity.createdBy
case 'confidence':
return entity.confidence
case 'weight':
return entity.weight
case 'visibility':
return (entity as unknown as Record<string, unknown>).visibility
}
// Out-of-map system spelling: parse refuses these upstream with a typed
// error; reaching here (internal callers only) reads as absent.
return undefined
switch (field) {
case 'noun':
case 'type':
return entity.type
case 'subtype':
return entity.subtype
case 'id':
return entity.id
case 'createdAt':
return entity.createdAt
case 'updatedAt':
return entity.updatedAt
case 'service':
return entity.service
case 'createdBy':
return entity.createdBy
case 'confidence':
return entity.confidence
case 'weight':
return entity.weight
case '_rev':
return entity._rev
case 'data':
return entity.data
}
const path = field.startsWith('metadata.') ? field.slice('metadata.'.length) : field
const bag = (entity.metadata ?? {}) as Record<string, unknown>
if (!path.includes('.')) return bag[path]
return resolvePath(bag, path)
if (field.includes('.')) {
// Dotted path: resolve against the whole entity first (`metadata.x`),
// then against the metadata bag (`address.city` on nested metadata).
const fromEntity = resolvePath(entity as unknown as Record<string, unknown>, field)
if (fromEntity !== undefined) return fromEntity
return resolvePath((entity.metadata ?? {}) as Record<string, unknown>, field)
}
return ((entity.metadata ?? {}) as Record<string, unknown>)[field]
}
/** Walk a dotted path through nested plain objects. */

View file

@ -128,7 +128,7 @@ async function loadBunAssets(): Promise<ModelAssets> {
}
// Strategy 2: node_modules path relative to CWD (for installed packages)
const nmPath = './node_modules/@soulcraftlabs/brainy/assets/models/all-MiniLM-L6-v2'
const nmPath = './node_modules/@soulcraft/brainy/assets/models/all-MiniLM-L6-v2'
pathsToTry.push([
`${nmPath}/model.safetensors`,
`${nmPath}/tokenizer.json`,
@ -168,9 +168,9 @@ async function loadBunAssets(): Promise<ModelAssets> {
// If all strategies fail, provide helpful error message
throw new Error(
'Could not load model assets. For bun --compile, ensure model files are accessible:\n' +
' Option 1: Keep node_modules/@soulcraftlabs/brainy/assets/ alongside your binary\n' +
' Option 1: Keep node_modules/@soulcraft/brainy/assets/ alongside your binary\n' +
' Option 2: Copy assets/ folder to your working directory\n' +
' Option 3: Use --asset flag: bun build --compile --asset="./node_modules/@soulcraftlabs/brainy/assets/**/*"'
' Option 3: Use --asset flag: bun build --compile --asset="./node_modules/@soulcraft/brainy/assets/**/*"'
)
}
@ -190,7 +190,7 @@ async function loadNodeAssets(): Promise<ModelAssets> {
if (!fs.existsSync(assetsDir)) {
throw new Error(
`Model assets not found: ${assetsDir}\n` +
`Ensure @soulcraftlabs/brainy is installed correctly.`
`Ensure @soulcraft/brainy is installed correctly.`
)
}

Some files were not shown because too many files have changed in this diff Show more