diff --git a/.claude/skills/architecture.md b/.claude/skills/architecture.md
index de046b17..4de3c287 100644
--- a/.claude/skills/architecture.md
+++ b/.claude/skills/architecture.md
@@ -2,7 +2,7 @@
## What Is Brainy
-@soulcraft/brainy (v7.17.0) is a Universal Knowledge Protocol -- a Triple Intelligence database combining vector search, graph traversal, and metadata filtering in a single library. Published to npm as a public MIT-licensed package.
+@soulcraftlabs/brainy (v7.17.0) is a Universal Knowledge Protocol -- a Triple Intelligence database combining vector search, graph traversal, and metadata filtering in a single library. Published to npm as a public MIT-licensed package.
## Core Architecture
diff --git a/.forgejo/workflows/ci.yml b/.forgejo/workflows/ci.yml
index 5e93cd96..da5887f6 100644
--- a/.forgejo/workflows/ci.yml
+++ b/.forgejo/workflows/ci.yml
@@ -5,6 +5,10 @@ name: CI
# sequential, so tag-triggered matrix jobs (~22 min) would queue AHEAD of the
# tag's publish-source run and starve every release (observed on 8.10.3 and
# 9.0.0: the publish sat behind the tag's own redundant CI).
+concurrency:
+ group: ci-${{ github.ref }}
+ cancel-in-progress: true
+
on:
push:
branches: ['**']
diff --git a/.forgejo/workflows/delta-gate.yml b/.forgejo/workflows/delta-gate.yml
new file mode 100644
index 00000000..c320594e
--- /dev/null
+++ b/.forgejo/workflows/delta-gate.yml
@@ -0,0 +1,148 @@
+name: Delta Gate
+
+# On-demand candidate-vs-control gate on the capped functional CI lane
+# (label: gate-functional). That lane is Bun-only host-mode — there is no
+# Node.js runtime available to it, so this workflow deliberately avoids every
+# JS-based action (checkout/setup-node/setup-bun/upload-artifact all require
+# one) and does everything with plain git + bun in shell steps instead.
+#
+# Verdict lines a caller should grep for in the run log:
+# COLLECTED patch= control= — collection-truncation guard inputs
+# NEW-RED-COUNT: — failures on candidate absent from control
+# DELTA-GATE: CLEAN | NEW REDS | INVALID | STOPPED-BY-REGISTRY-TRIPWIRE
+#
+# The lane's own housekeeping stops the runner and drops a marker file when
+# host pressure (I/O, registry latency, disk budget) trips — never ours to
+# interpret as a red or a green. The final step checks for that marker before
+# it says anything about pass/fail.
+
+on:
+ workflow_dispatch:
+ inputs:
+ candidate:
+ description: 'Candidate ref (branch or sha) to gate'
+ required: true
+ type: string
+ control:
+ description: 'Control sha to diff against'
+ required: true
+ type: string
+ # workflow_dispatch needs Actions-unit write on the dispatching credential;
+ # push does not (it runs from the pushed ref's own tree), so a plain push
+ # to a release or CI branch is the fallback trigger while that grant is
+ # outstanding — see the ref-resolution step below for what it gates against.
+ push:
+ branches: ['rel/**', 'ci/**']
+
+concurrency:
+ group: delta-gate
+ cancel-in-progress: false
+
+jobs:
+ delta-gate:
+ name: Delta gate — candidate vs control
+ runs-on: gate-functional
+ timeout-minutes: 120
+ steps:
+ - name: Resolve candidate/control refs
+ id: refs
+ run: |
+ candidate="${{ github.event.inputs.candidate }}"
+ control="${{ github.event.inputs.control }}"
+ # workflow_dispatch supplies both explicitly; a push event carries
+ # neither — fall back to the pushed commit as candidate and the
+ # last released, known-good tip (10.4.9) as control, so a plain
+ # push still produces a meaningful gate instead of an empty ref.
+ if [ -z "$candidate" ]; then candidate="${{ github.sha }}"; fi
+ if [ -z "$control" ]; then control="eec90bdd"; fi
+ echo "candidate=$candidate" >> "$GITHUB_OUTPUT"
+ echo "control=$control" >> "$GITHUB_OUTPUT"
+ echo "Resolved (trigger=${{ github.event_name }}): candidate=$candidate control=$control"
+
+ - name: Clean any residue from a prior run
+ run: rm -rf "ob-cand-${{ github.run_id }}" "ob-ctrl-${{ github.run_id }}" "/tmp/ob-${{ github.run_id }}-"*
+
+ - name: Clone + test — candidate
+ id: patch
+ run: |
+ set -o pipefail
+ git clone --quiet "https://source.soulcraft.com/soulcraftlabs/open-brainy.git" "ob-cand-${{ github.run_id }}"
+ cd "ob-cand-${{ github.run_id }}"
+ git checkout --quiet "${{ steps.refs.outputs.candidate }}"
+ git log --oneline -1
+ bun install
+ rc=0
+ bun x vitest run > "/tmp/ob-${{ github.run_id }}-patch.log" 2>&1 || rc=$?
+ echo "PATCH-RC:$rc"
+ grep -aE "Tests .*(passed|failed)" "/tmp/ob-${{ github.run_id }}-patch.log" | tail -1
+ grep -aE "^ FAIL |^\s+×" "/tmp/ob-${{ github.run_id }}-patch.log" | sed -E "s/ [0-9]+ms$//" | sed -E "s/^\s+//" | sort -u > "/tmp/ob-${{ github.run_id }}-patch.fail"
+ echo "PATCH-FAILING:$(wc -l < "/tmp/ob-${{ github.run_id }}-patch.fail")"
+
+ - name: Clone + test — control
+ id: control
+ run: |
+ set -o pipefail
+ git clone --quiet "https://source.soulcraft.com/soulcraftlabs/open-brainy.git" "ob-ctrl-${{ github.run_id }}"
+ cd "ob-ctrl-${{ github.run_id }}"
+ git checkout --quiet "${{ steps.refs.outputs.control }}"
+ git log --oneline -1
+ bun install
+ rc=0
+ bun x vitest run > "/tmp/ob-${{ github.run_id }}-control.log" 2>&1 || rc=$?
+ echo "CONTROL-RC:$rc"
+ grep -aE "Tests .*(passed|failed)" "/tmp/ob-${{ github.run_id }}-control.log" | tail -1
+ grep -aE "^ FAIL |^\s+×" "/tmp/ob-${{ github.run_id }}-control.log" | sed -E "s/ [0-9]+ms$//" | sed -E "s/^\s+//" | sort -u > "/tmp/ob-${{ github.run_id }}-control.fail"
+ echo "CONTROL-FAILING:$(wc -l < "/tmp/ob-${{ github.run_id }}-control.fail")"
+
+ - name: Delta gate verdict
+ if: always()
+ run: |
+ set -o pipefail
+
+ # The lane's own tripwire wins over anything we would otherwise say:
+ # a bare failure/timeout above with this marker present is host
+ # pressure, never a real red and never a real green.
+ if [ -f /srv/gate-lane/TRIPWIRE-STOPPED ]; then
+ echo "DELTA-GATE: STOPPED-BY-REGISTRY-TRIPWIRE"
+ head -1 /srv/gate-lane/TRIPWIRE-STOPPED
+ exit 3
+ fi
+
+ patch_log="/tmp/ob-${{ github.run_id }}-patch.log"
+ control_log="/tmp/ob-${{ github.run_id }}-control.log"
+ patch_fail="/tmp/ob-${{ github.run_id }}-patch.fail"
+ control_fail="/tmp/ob-${{ github.run_id }}-control.fail"
+
+ if [ ! -s "$patch_log" ] || [ ! -s "$control_log" ]; then
+ echo "DELTA-GATE: INVALID — a leg produced no log (see the two steps above for the real cause)"
+ exit 2
+ fi
+
+ pt=$(grep -aoE "\(([0-9]+)\)$" "$patch_log" | tail -1 | tr -d "()")
+ ct=$(grep -aoE "\(([0-9]+)\)$" "$control_log" | tail -1 | tr -d "()")
+ echo "COLLECTED patch=${pt:-0} control=${ct:-0}"
+ if [ "${pt:-0}" -lt 3000 ] || [ "${ct:-0}" -lt 3000 ]; then
+ echo "DELTA-GATE: INVALID — truncated collection"
+ exit 2
+ fi
+
+ echo "=== NEW REDS ==="
+ comm -23 "$patch_fail" "$control_fail"
+ new=$(comm -23 "$patch_fail" "$control_fail" | wc -l)
+ echo "NEW-RED-COUNT:$new"
+
+ echo "=== full candidate fail list ==="
+ cat "$patch_fail"
+ echo "=== full control fail list ==="
+ cat "$control_fail"
+
+ if [ "$new" -eq 0 ]; then
+ echo "DELTA-GATE: CLEAN"
+ else
+ echo "DELTA-GATE: NEW REDS"
+ exit 1
+ fi
+
+ - name: Clean up (mind the lane's disk budget)
+ if: always()
+ run: rm -rf "ob-cand-${{ github.run_id }}" "ob-ctrl-${{ github.run_id }}" "/tmp/ob-${{ github.run_id }}-"*
diff --git a/.forgejo/workflows/publish-source.yml b/.forgejo/workflows/publish-source.yml
index 8220bac9..6bd42b2a 100644
--- a/.forgejo/workflows/publish-source.yml
+++ b/.forgejo/workflows/publish-source.yml
@@ -12,6 +12,11 @@ on:
push:
tags:
- 'v*'
+ workflow_dispatch:
+ inputs:
+ ref_reason:
+ description: 'why this manual run (e.g. tag event dropped)'
+ required: false
jobs:
publish:
@@ -32,22 +37,31 @@ jobs:
run: |
set -eo pipefail
- SOURCE_NPM_REG="https://source.soulcraft.com/api/packages/soulcraft/npm/"
+ SOURCE_NPM_REG="https://source.soulcraft.com/api/packages/soulcraftlabs/npm/"
VERSION="$(node -p "require('./package.json').version")"
- echo "Publishing @soulcraft/brainy@${VERSION} to The Source registry..."
+ # The dist-tag follows the version: a prerelease (any hyphen —
+ # 10.4.0-rc.1) publishes under 'rc' and must NEVER move 'latest' —
+ # every consumer resolving 'latest' from this registry would otherwise
+ # be handed a release candidate. Same rule scripts/release.sh applies
+ # to the storefront leg.
+ NPM_TAG="latest"
+ case "$VERSION" in
+ *-*) NPM_TAG="rc" ;;
+ esac
+ echo "Publishing @soulcraftlabs/brainy@${VERSION} to The Source registry (dist-tag: ${NPM_TAG})..."
TMPRC="$(mktemp)"
chmod 600 "$TMPRC"
{
- echo "@soulcraft:registry=${SOURCE_NPM_REG}"
- echo "//source.soulcraft.com/api/packages/soulcraft/npm/:_authToken=${FORGE_NPM_TOKEN}"
+ echo "@soulcraftlabs:registry=${SOURCE_NPM_REG}"
+ echo "//source.soulcraft.com/api/packages/soulcraftlabs/npm/:_authToken=${FORGE_NPM_TOKEN}"
} > "$TMPRC"
# The release script bumps package.json's version before it tags, so
# this tag's checkout already carries the version being published —
# nothing here re-derives it from the tag name.
PUBLISH_OK=true
- if ! npm publish --tag latest --userconfig "$TMPRC"; then
+ if ! npm publish --tag "$NPM_TAG" --userconfig "$TMPRC"; then
PUBLISH_OK=false
fi
@@ -55,7 +69,7 @@ jobs:
# exit code: a benign duplicate publish (a prior run, or a mirror, already
# landed this exact version) reports failure even though the registry
# already holds the right content.
- LANDED_VERSION="$(npm view "@soulcraft/brainy@${VERSION}" version --userconfig "$TMPRC" 2>/dev/null || echo "")"
+ LANDED_VERSION="$(npm view "@soulcraftlabs/brainy@${VERSION}" version --userconfig "$TMPRC" 2>/dev/null || echo "")"
rm -f "$TMPRC"
if [ "$LANDED_VERSION" != "$VERSION" ]; then
@@ -64,7 +78,7 @@ jobs:
fi
if [ "$PUBLISH_OK" = true ]; then
- echo "Published and verified @soulcraft/brainy@${VERSION} on The Source registry."
+ echo "Published and verified @soulcraftlabs/brainy@${VERSION} on The Source registry."
else
- echo "::warning::npm publish reported failure, but readback confirms @soulcraft/brainy@${VERSION} is already live on The Source (a prior run or mirror landed it) — treating this run as successful, since the registry content is correct. Any OTHER failure mode would have failed the readback check above instead."
+ echo "::warning::npm publish reported failure, but readback confirms @soulcraftlabs/brainy@${VERSION} is already live on The Source (a prior run or mirror landed it) — treating this run as successful, since the registry content is correct. Any OTHER failure mode would have failed the readback check above instead."
fi
diff --git a/CHANGELOG.md b/CHANGELOG.md
index f99584e5..fc577c1d 100644
--- a/CHANGELOG.md
+++ b/CHANGELOG.md
@@ -2,13 +2,187 @@
All notable changes to this project will be documented in this file. See [standard-version](https://github.com/conventional-changelog/standard-version) for commit guidelines.
-### [10.3.1](https://source.soulcraft.com/soulcraft/brainy/compare/v10.3.0...v10.3.1) (2026-08-18)
+
+### [10.4.12](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v10.4.11...v10.4.12) (2026-09-03)
+
+- Mixed-kind fields index exactly, arrays to 256, a drained loop is not a shutdown, and finds project from the column store
+- fix(index): a metadata field holds every value kind it was written with — one posting column per (field, kind); an equality filter reads the query value's own kind, a range routes by its bounds; nothing is refused and nothing is silently dropped; an index written by the old shape opens unchanged (a128f0ed)
+- fix(metadata): metadata arrays index up to 256 elements; a longer array refuses at write time by name (MetadataArrayTooLargeError) — a vector parked in metadata now throws; move it to `vector` (e435da78)
+- fix(shutdown): beforeExit runs a non-closing flush only — a script that never calls close() exits with the writer lock on disk and no clean-shutdown marker, and the next open evicts the stale lock and folds the log, bounded; SIGTERM and SIGINT are unchanged (6baa4d7f)
+- feat(find): field projection — find({fields}) and get({fields}) resolve scalars from the column store on every leg, including vector-leg finds; absent fields stay absent (ad0f493f)
+- fix(find): orderBy is the order on every find path, not only the metadata-only one (5e720d17)
+- fix(metadata): the legacy sparse range path orders values, or refuses by name — never ranks by hash (a7eb7f52)
+- fix(close): a read-only brain writes nothing under `_system/` (f27a7776)
+- fix(contract): the flush gate's internals are private, not doors (72c8ee6a)
+- test(hygiene): the triple-intelligence correctness cases sit in the gate; the idle and connected-find pins name the brain they measure (28083981)
+- ci(release): the rail writes its own wall entry into the shared releases repo — never hand-written again (adcb883e)
+
+### [10.4.11](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v10.4.9...v10.4.11) (2026-09-02)
+
+- ci: superseded pushes cancel their own runs (concurrency per ref) (6053f6d4)
+- test(batch): the batch-size-limit tests add unvectored items — they test batching, not embedding (a1423c6d)
+- fix(flush): the gate settles its waiter from the machine, never from a chain (dea3ec20)
+- test(batch): the batch-vs-individual timing assertion runs in the perf lane, not the correctness gate (ebb3a4bf)
+- test(gate): the coverage guard counts the perf lane's config as a gate (2c5e3474)
+- chore(contract): emit the 10.4.11 manifest (4142f368)
+- fix(close): a read-only brain writes no clean-shutdown evidence — the marker is the writer's word about itself (367ca721)
+- fix(generation-store): commitTransaction refuses while single-ops are pending — the order invariant is enforced, not assumed (a79db434)
+- test(shutdown): pin one owner per brain — real processes, real signals (da951990)
+- fix(shutdown): one owner per brain — the signal handler defers to close(), and flush is single-flight (ec644bde)
+- fix(vfs): a path-scoped search is a served range over the path, not a refused prefix match (65493ba2)
+- ci(test): perf and scale benchmarks leave the correctness gate (dee46b35)
+- test(open): pin the pending-embed checkpoint — stuck id, crash matrix, torn fallback (1fb51093)
+- perf(open): the pending-embed fold is bounded by a checkpoint of the SET, not an empty-only mark (15d4f65d)
+- perf(open): a sealed segment the manifest proves is below the bound is never read (bc70c43d)
+- fix(find): a page the metadata block already cut is not cut again (905c267c)
+- fix(find): the hybrid legs rank inside the filter, and only the page is read (b1c70544)
+- ci(delta-gate): add a push fallback trigger alongside workflow_dispatch (67ae0046)
+- ci: add the delta-gate workflow for the capped functional lane (9922631d)
+- docs(plugin): the planner door's hiddenIds contract is the answer, not the mechanism (2633e8d5)
+- feat(engine): a protected factory for the generation store — a subclass may substitute one that keeps the contract (f763317a)
+- fix(find): near() searches around the anchor's own vector, and refuses by name without one (a8c5fbf9)
+- Merge remote-tracking branches 'origin/fix/planner-provider-door' and 'origin/fix/containment-batching' into rel/10.4.10-candidate (34f1886f)
+- feat(plugin): an optional planFindPage door — an index that can plan a find answers it in one call (4d5f823f)
+- perf(vfs): repairContainment's reconcile is one paged edge walk, not one graph call per file (3e60aded)
+
+
+### [10.4.9](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v10.4.6...v10.4.9) (2026-09-02)
+
+- Merge branch 'fix/pending-embed-low-water' into rel/10.4.9-candidate (2648f56d)
+- fix(open): pending-embed recovery keeps the crash-recovery contract — foreground, bounded by the mark (8a2ebacf)
+- Merge branches 'fix/connected-find-order', 'fix/pending-embed-low-water' and 'fix/related-verb-array' into rel/10.4.9-candidate (d5147ed6)
+- fix(graph): the verb fast paths honour every requested type, source, and target (6a89adc4)
+- perf(open): pending-embed recovery is bounded by a low-water mark and runs behind the doors (88e79729)
+- fix(find): connected finds are graph-first — neighbours, then the filter over those ids, then the page (077cbc0b)
+- fix(storage): counts persistence is single-flight, coalesced, and never races its own temp file (5e3b343a)
+
+
+### [10.4.6](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v10.4.5...v10.4.6) (2026-08-31)
+
+- fix(transact): metadata-index ops take their JSON-safe view at the crossing, not at construction (73500e7d)
+
+
+### [10.4.5](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v10.4.4...v10.4.5) (2026-08-31)
+
+- build(release): the docs-push step retires — this engine documents itself in its own repository (d6bcb14f)
+- fix(generations): a sealed segment may only declare the generations it holds (a963a744)
+- fix(recovery): a torn generation-log tail is a terminal verdict, never a wait (c9930871)
+
+
+### [10.4.4](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v10.4.3...v10.4.4) (2026-08-28)
+
+- fix(vfs): the old-root sweep narrates only when it has something to say (d49148e1)
+- fix(tests): the health-gate pin follows the verdict, and the VFS suite uses its own store (42e2da25)
+- Merge branch 'next/open-lazy-open-and-counts' (5ebd3b40)
+- docs: the contract manifest stands alone; public docs describe this engine only (a8c724a2)
+- docs(releases): 10.4.4 consumer notes — correctness and observability, with the performance line stated exactly (61a46927)
+- docs: measurements in public history carry numbers, not provenance (02c61636)
+- feat(open): name the two steps that hold the vfs-bootstrap phase (2cf38010)
+- fix(storage): a dead flush watch falls back to the 500ms poll, not the 30s sweep (5c22f950)
+- fix(storage): the flush watcher cannot arm twice in its async window (16d2e1a9)
+- perf(idle): the flush-request watch is event-driven; the heartbeat is observability (fb1da1c5)
+- perf(open): answer "are there any entities?" with one directory read (417ddb51)
+- perf(generations): discover generations by directory name, not by walking the log (9dd39921)
+- fix(flush): clear() and repairIndex() set the dirty witness themselves (e4c27fbc)
+- feat(open): the open names the STEP that cost the time, not just the phase (5a091cca)
+- perf(vfs): the old-root sweep runs once per store, not once per open (4a67aa0f)
+- chore: keep the generated neural stamps at main's values (c1f09723)
+- feat(contract): declare contract 1, serve three operators, refuse four by name (48802ba3)
+- fix(open): a provider rebuilding itself is a third state, not a CRITICAL (50676c02)
+- feat(open): open never waits for a provider that is rebuilding itself (131daa08)
+- perf(flush): an idle brain does no work — no periodic flush without a write (f5a6cb3f)
+- feat(repair): repairIndex narrates every phase and its receipt carries the walls (3fffd9c6)
+- fix(storage): a suspect count ledger heals itself, and counts.json is written atomically (f4e2d34b)
+- feat(open): the open narrates itself, on a channel production cannot clamp (afe08a1f)
+- fix(storage): a clean close is recorded, and the writer lock is always given up (e652162c)
+- docs: repository links point at soulcraftlabs/open-brainy — the soulcraft/brainy path becomes the native engine's repo tonight (38c3397b)
+
+
+### [10.4.3](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v10.4.2...v10.4.3) (2026-08-27)
+
+- Merge branch 'next/open-brainy-rename' (a58372f0)
+- chore: rename to @soulcraftlabs/brainy for Open Brainy on The Source (a99b1e83)
+- docs(releases): 10.4.3 — Open Brainy's first release under the new name, same engine as 10.4.2; The Source is the one registry (9f248b24)
+
+
+### [10.4.2](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v10.4.2-rc.1...v10.4.2) (2026-08-27)
+
+- docs(releases): 10.4.1 and 10.4.2 consumer notes; 10.4.2 is the last MIT release under this name, Open Brainy continues at @soulcraftlabs/brainy (a082e0ef)
+
+
+### [10.4.2-rc.1](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v10.4.1...v10.4.2-rc.1) (2026-08-27)
+
+- Merge branch 'next/zero-norm-unvector-door' (9b84ef5b)
+- fix(vectors): a zero-norm vector is not a vector, canonical side included, plus the sanctioned unvector door (0de76659)
+- fix(hnsw): skip unvectored rows on rebuild; refuse empty vectors in the index (8fc553b1)
+- fix(storage): derive the canonical count ledger from identity records, stamp the derivation rule, and mark legacy-derived ledgers suspect at load (fd6b4ce4)
+- Merge branch 'next/enumeration-identity-rekey' (204d74c1)
+- fix(storage): enumeration re-keys on the identity record, not the vector leg (f8d8ce16)
+- fix(init): rethrow plugin activation failures with the original error as cause so the originating frame survives to the caller (2496e09a)
+- Merge branch 'next/vfs-root-zero-norm' (4c7b0fab)
+- fix(vfs): the VFS root never persists a zero-norm vector (c6cc0de9)
+- build: derive generated-file stamps from git commit time, not wall clock (8a5c1245)
+- Merge remote-tracking branch 'origin/release/10.4.1' (aad9e2ee)
+- docs(concepts): the serving law — a failure is graded by whether an answer could be wrong, never by the cost of the fix; reads refuse per family (2914e0eb)
+- chore(release): 10.4.1-rc.1 (7870dc40)
+
+
+### [10.4.1](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v10.4.0...v10.4.1) (2026-08-26)
+
+- fix(reads): the read gate is per-family; a write carrying unchanged data never re-embeds (c039411e)
+- docs(guide): the docs pipeline publishes through the ingest API — the separate deploy step is retired (21e506e8)
+
+
+### [10.4.0](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v10.4.0-rc.4...v10.4.0) (2026-08-26)
+
+- docs(releases): the 10.4.0 entry catches up to the late trains — repair routing, the vector ledger and open-gate leg, the loud config guard, the JSON-safe crossing (834149ed)
+
+
+### [10.4.0-rc.4](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v10.4.0-rc.3...v10.4.0-rc.4) (2026-08-25)
+
+- feat(vector): the vectored-noun scalar joins the count ledger; the open gate closes the vector leg (9730835b)
+
+
+### [10.4.0-rc.3](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v10.4.0-rc.2...v10.4.0-rc.3) (2026-08-25)
+
+- fix(update-seam): the metadata crossing never carries BigInt endpoint ints (f4780c8e)
+- Merge branch 'worktree-agent-ad3aff0dffd17a6eb' (f14da34b)
+- fix(add): empty string is real data, not a missing field (258e9042)
+- feat(vfs): implement readdir's recursive option — typed since 7.30, never read (fc516da6)
+- feat(open-path): init never gates on the embedding model; open goes concurrent; slow opens narrate (96624f40)
+
+
+### [10.4.0-rc.2](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v10.4.0-rc.1...v10.4.0-rc.2) (2026-08-25)
+
+- test(readiness): the report helper's clock freezes — two independently-built reports compared across a millisecond tick made the plant lane red (39b916a3)
+- feat(repair): a heal:'repair' verdict routes to the provider's own incremental repair() (553e0d97)
+- fix(storage): an unknown nested storage config can never silently land on the shared default root (ddd5e719)
+- docs(release): the 10.4.0 entry, the index-health concept doc, and the API surfaces — written from the tree, not the plan (8cced871)
+- fix(plugins): the silent-degrade doors close — a broken accelerator install can never read as absent (b9ba50fb)
+- feat(recovery): the catchup verdict is consumed; verb rows go live; the metadata rebuild goes online (18f172e0)
+- feat(health): the gate reads the named report — reads refuse loudly, never rebuild; open serves before it returns; the ceremony door (f8f64780)
+
+
+### [10.4.0-rc.1](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v10.3.1...v10.4.0-rc.1) (2026-08-24)
+
+- ci(publish): the home dist-tag follows the version — a prerelease publishes under 'rc' and never moves 'latest' (a1376e4a)
+- chore(release): --source-only — a home-only prerelease mode (The Source, never the storefront) (dcbad176)
+- test(fold-checkpoint): the ARM-AT-FLIP pin arms its crash instead of racing the pending-flush timer (4176439b)
+- fix(health): one contract for a throwing probe — heal is none, serving is not withheld; repair report gains missing/rebuilt/reason (116550eb)
+- feat(storage): the canonical count ledger — ALL-visibility scalars, unclamped totals, suspect-on-unprovable-delete (7c8c8be3)
+- fix(delete): the null-metadata skip closes — index legs run id-keyed or narrate, never silently strand postings (607e9f54)
+- feat(repair): repairIndex returns the per-family receipt and narrates its summary (8d45f964)
+- fix(reads): the readiness gate guards every index read surface — serving empty from a not-ready provider is unrepresentable (40e7119b)
+- ci(gate): the machine-health preflight and the truncation verdict guard (1e046aa1)
+
+
+### [10.3.1](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v10.3.0...v10.3.1) (2026-08-18)
- docs(releases): the 10.3.1 consumer entry — the fold that behaves (900cc895)
- fix(recovery): the fold streams and narrates; the checkpoint chain arms at the flip (ed7d1db9)
-### [10.3.0](https://source.soulcraft.com/soulcraft/brainy/compare/v10.2.0...v10.3.0) (2026-08-18)
+### [10.3.0](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v10.2.0...v10.3.0) (2026-08-18)
- docs(releases): the 10.3.0 consumer entry — the trust-and-provenance release (97d75649)
- fix(locks): the fence keys ownership on pid+hostname — a same-process re-open never fences its predecessor (0991cf28)
@@ -17,14 +191,14 @@ All notable changes to this project will be documented in this file. See [standa
- feat(log): system commits carry their origin; the attested per-id reconcile door (9ac9e706)
-### [10.2.0](https://source.soulcraft.com/soulcraft/brainy/compare/v10.1.0...v10.2.0) (2026-08-17)
+### [10.2.0](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v10.1.0...v10.2.0) (2026-08-17)
- docs(releases): the 10.2.0 consumer entry — adoption completes in one call (97538e1f)
- ci: the correctness plant runs integration + conformance on every push — a release never waits on a second machine (b17fdc8e)
- fix(adoption): the baseline backfill runs to completion — one call adopts a pre-log baseline of any size (a5a18838)
-### [10.1.0](https://source.soulcraft.com/soulcraft/brainy/compare/v10.0.0...v10.1.0) (2026-08-13)
+### [10.1.0](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v10.0.0...v10.1.0) (2026-08-13)
- docs(releases): the 10.1.0 consumer entry — bounded recovery, restore founding, the two write-path cures (7d3c8696)
- fix(restore): a restore is an unclean event — the swap runs quiesced and the snapshot's durability stamps never survive it (9ca80667)
@@ -33,7 +207,7 @@ All notable changes to this project will be documented in this file. See [standa
- feat(query): the sparse-store cut — where on a never-carried field serves operator truth, never a refusal (7b67db4d)
-### [10.0.0](https://source.soulcraft.com/soulcraft/brainy/compare/v9.0.0...v10.0.0) (2026-08-12)
+### [10.0.0](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v9.0.0...v10.0.0) (2026-08-12)
- fix(adoption): the baseline backfill cures hydration-law drift — existing brains reach the crash-safe default with zero operator steps (25f0dd96)
- fix(adoption): the reserved-root mint exemption — int 0 is legitimate for exactly one id (2abe8b38)
@@ -65,7 +239,7 @@ All notable changes to this project will be documented in this file. See [standa
- test: version-coupling pins go major-agnostic — the 8.x literals broke at the 9.0.0 bump while the coupling law itself behaved correctly (8a6807e8)
-### [9.0.0](https://source.soulcraft.com/soulcraft/brainy/compare/v8.11.0...v9.0.0) (2026-08-04)
+### [9.0.0](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v8.11.0...v9.0.0) (2026-08-04)
- docs: 9.0 namespace-migration guide — the simple story + the mechanical sweep checklist, published for humans and tooling alike (61ab9db2)
- fix(release): storefront leg republishes CI's exact forge artifact — byte-identity by construction, verified by cross-registry shasum before the ceremony reports success (d89df2ed)
@@ -100,7 +274,7 @@ All notable changes to this project will be documented in this file. See [standa
- feat: scanFacts liveness contract — first batch or loud failure within a documented bound (f8e6da2b)
-### [8.11.0](https://source.soulcraft.com/soulcraft/brainy/compare/v8.10.1...v8.11.0) (2026-07-27)
+### [8.11.0](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v8.10.1...v8.11.0) (2026-07-27)
- docs: the last two archived-host links point home (91ef1c8b)
- feat: includeHidden — export carries every visibility tier for migration-grade canon completeness (63c1eeb9)
@@ -109,19 +283,19 @@ All notable changes to this project will be documented in this file. See [standa
- ci: run the pipeline on the forge (999d0ebb)
-### [8.10.3](https://source.soulcraft.com/soulcraft/brainy/compare/v8.10.2...v8.10.3) (2026-08-03)
+### [8.10.3](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v8.10.2...v8.10.3) (2026-08-03)
- docs: dedupe the 8.10.2 release-notes entry the cherry doubled onto the branch (8c956608)
- fix: user metadata named 'level' is a real field everywhere — the engine-internal node layer no longer shadows it in sort/filter/aggregation, and the indexing views stop stamping a phantom 0 into its column; index epoch 2 rebuilds existing brains at first open (958a0859)
-### [8.10.2](https://source.soulcraft.com/soulcraft/brainy/compare/v8.10.1...v8.10.2) (2026-07-29)
+### [8.10.2](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v8.10.1...v8.10.2) (2026-07-29)
- docs: 8.10.2 consumer release notes — update() write granularity, PathResolver idle-log fix, graph-lsm key recognition (a0123b5b)
- fix: metadata-only update() never rewrites the noun record — the unconditional whole-vector save turned per-entity stat touches into full rewrites+fsync, amplifying read-heavy sweeps into disk saturation on a production deployment (5b65eb82)
-### [8.10.1](https://source.soulcraft.com/soulcraft/brainy/compare/v8.10.0...v8.10.1) (2026-07-24)
+### [8.10.1](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v8.10.0...v8.10.1) (2026-07-24)
- refactor: remove the orphaned transaction-result type left behind by the dead-path removal (edf123a5)
- fix: warm() metadata surface routes through the active provider (warm hook added to the metadata contract); add maintenanceDebt() observability surface (5b2cbf74)
diff --git a/CLAUDE.md b/CLAUDE.md
index c7336a18..56df0b72 100644
--- a/CLAUDE.md
+++ b/CLAUDE.md
@@ -12,13 +12,13 @@ Handoff file: `/home/dpsifr/.strategy/PLATFORM-HANDOFF.md`
**Brainy's current open actions:** None. MIT open-source — no platform-specific actions.
-**Current version:** run `npm view @soulcraft/brainy version` (never trust a hardcoded number here — this line went stale for months); consumer-facing changes tracked in `RELEASES.md`
+**Current version:** run `npm view @soulcraftlabs/brainy version --registry https://source.soulcraft.com/api/packages/soulcraftlabs/npm/` (never trust a hardcoded number here — this line went stale for months); consumer-facing changes tracked in `RELEASES.md`
---
## Project Overview
-Brainy is a Universal Knowledge Protocol -- a Triple Intelligence database that combines vector similarity search, graph traversal, and metadata filtering into a single TypeScript library. Published as `@soulcraft/brainy` on npm under the MIT license.
+Brainy is a Universal Knowledge Protocol -- a Triple Intelligence database that combines vector similarity search, graph traversal, and metadata filtering into a single TypeScript library. Published as `@soulcraftlabs/brainy` on The Source (source.soulcraft.com registry) under the MIT license.
## Getting Started
@@ -91,7 +91,7 @@ test: add/update tests (patch version bump)
## Docs Pipeline — soulcraft.com/docs
-Docs in `docs/**/*.md` are published with the npm package (included in `files`) and synced to soulcraft.com/docs on every portal deploy. Frontmatter controls what appears publicly.
+Docs in `docs/**/*.md` are published with the npm package (included in `files`) and go live on soulcraft.com/docs via the docs ingest API: the release script's `scripts/push-docs.js` step POSTs every public doc to `https://soulcraft.com/api/docs/ingest` (auth: `DOCS_INGEST_SECRET` in the environment). No separate deploy step is involved (the old deploy-to-publish flow was retired in a platform change, 2026-08). Frontmatter controls what appears publicly.
### Docs check triggers
@@ -161,9 +161,9 @@ npm run release:major # Breaking changes (rare, manual decision)
The script: verifies clean git state, builds, tests, bumps version, updates CHANGELOG.md, commits, tags, pushes, publishes to npm, and creates a GitHub release.
After a successful release, remind the user:
-> "Published. Deploy portal to pick up the new docs → go to the portal project and deploy."
+> "Published. Docs are live on soulcraft.com/docs (pushed via the ingest API during the release) — spot-check a changed page with curl."
-Do NOT deploy portal from here. Portal is always deployed separately from within the portal project.
+There is no separate deploy step anymore. If the docs push failed (the script warns loudly), re-run `node scripts/push-docs.js` with `DOCS_INGEST_SECRET` set.
## Closed-Source Product Names — HARD RULE
diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md
index d277091d..c58520b7 100644
--- a/CONTRIBUTING.md
+++ b/CONTRIBUTING.md
@@ -6,7 +6,7 @@ may find elsewhere in the repo's history.
## Where the project lives
-The source of truth is a self-hosted forge: **source.soulcraft.com/soulcraft/brainy**.
+The source of truth is a self-hosted forge: **source.soulcraft.com/soulcraftlabs/open-brainy**.
It's anonymously readable and cloneable — no account needed to browse, clone,
or build.
@@ -31,7 +31,7 @@ fine) to talk through the approach saves everyone rework.
## Development setup
```bash
-git clone https://source.soulcraft.com/soulcraft/brainy.git
+git clone https://source.soulcraft.com/soulcraftlabs/open-brainy.git
cd brainy
npm install
npm run build
@@ -41,6 +41,20 @@ npm test
Tests run on [Vitest](https://vitest.dev/). `npm test` runs the unit suite;
see `package.json` for `test:integration`, `test:coverage`, and friends.
+## Test gate
+
+The release gate is a bare `vitest run` (no `--config` flag) — the same
+command the delta gate and CI's checks invoke. It carries the full
+correctness suite and nothing else: wall-clock/scale benchmarks
+(`tests/performance/**`, `tests/critical-performance-benchmark.test.ts`,
+`tests/api/performance-benchmarks.test.ts`) and the two tests whose outcome
+depends on the host machine or network rather than the code
+(`tests/package-size-limit.test.ts` shells out to the `npm` CLI;
+`tests/model-loading.test.ts` makes a real network call to download a model)
+are excluded from it, because a timing threshold or a flaky network call has
+no business failing a correctness check. That whole family runs on demand,
+in its own exclusive slot, via `npm run test:perf`.
+
## Standards
- **Strict TypeScript.** No `any` escape hatches to dodge the type checker.
@@ -57,6 +71,17 @@ see `package.json` for `test:integration`, `test:coverage`, and friends.
description states a number, cite the benchmark that produced it (see
[docs/performance-envelopes.md](docs/performance-envelopes.md) for the
pattern). Don't state an estimate as if it were measured.
+- **Measurements carry numbers, not provenance.** Public commit messages and
+ docs give the SHAPE a number was taken at and never where it was taken: no
+ hostnames, no store or deployment identities, no operational anecdotes about
+ someone's running system. "A 14,056-noun / 72,679-verb production-shaped
+ store, measured solo under an exclusive lock" tells a reader everything the
+ number depends on; the machine it ran on and whose data it was tell them
+ nothing except where somebody's infrastructure lives.
+- **Documents that answer or reference a confidential specification never enter
+ this repository, even summarized.** The public docs describe THIS engine and
+ the published contract, and nothing else — a summary of a private document is
+ still that document's contents.
## License
diff --git a/README.md b/README.md
index ca558340..762c9ec3 100644
--- a/README.md
+++ b/README.md
@@ -1,5 +1,5 @@
-
+
Brainy
@@ -11,9 +11,9 @@
-
-
-
+
+
+
@@ -30,6 +30,8 @@
---
+**Open Brainy** is the MIT engine — the open API, client library, types, and protocol; an openly specified canonical on-disk format; and this TypeScript reference engine, scoped as a single-node engine for stores up to roughly one million rows. `@soulcraft/brainy` 10.4.2 was the last release under the old package name — the name passes to the native engine, **Brainy**, at 11.0.0: the same API over the same open format at production scale, and it requires a license.
+
Built because we were tired of stitching a vector store to a graph database to a document store — and spending weeks on plumbing before writing a line of business logic. Brainy indexes every fact **three ways at once** and lets one call query them together:
| You write | Brainy indexes it as | You query it with |
@@ -45,12 +47,14 @@ It runs **inside your process** — no server, no Docker, nothing to operate —
## Quick start
```bash
-bun add @soulcraft/brainy # Bun ≥ 1.1 — recommended
-npm install @soulcraft/brainy # Node.js ≥ 22
+bun add @soulcraftlabs/brainy # Bun ≥ 1.1 — recommended
+npm install @soulcraftlabs/brainy # Node.js ≥ 22
```
+> **Registry**: add `@soulcraftlabs:registry=https://source.soulcraft.com/api/packages/soulcraftlabs/npm/` to your `.npmrc` (anonymous read).
+
```javascript
-import { Brainy, NounType, VerbType } from '@soulcraft/brainy'
+import { Brainy, NounType, VerbType } from '@soulcraftlabs/brainy'
const brain = new Brainy() // in-memory; one line swaps to disk
await brain.init()
diff --git a/RELEASES.md b/RELEASES.md
index cc0272c3..c875cb26 100644
--- a/RELEASES.md
+++ b/RELEASES.md
@@ -1,7 +1,14 @@
# @soulcraft/brainy — Release Notes for Consumers
+Machine-readable release notes are published at
+https://source.soulcraft.com/soulcraftlabs/releases/raw/branch/main/open-brainy.json
+(this engine) and
+https://source.soulcraft.com/soulcraftlabs/releases/raw/branch/main/brainy.json
+(the product engine) — read by HQ's `/hq/releases` door, and the source of
+truth ahead of this file.
+
This file is the **quick reference for downstream sessions** tracking Brainy changes.
-Full auto-generated changelog: `CHANGELOG.md` · Releases: https://source.soulcraft.com/soulcraft/brainy/releases
+Full auto-generated changelog: `CHANGELOG.md` · Releases: https://source.soulcraft.com/soulcraftlabs/open-brainy/releases
**How to use:** Brainy is the underlying data engine for downstream applications. Read this when:
- Upgrading `@soulcraft/brainy` in your application
@@ -31,6 +38,330 @@ is sometimes cited as a 7.x removal — those methods never existed on 7.x; the
---
+## v10.4.4 — 2026-08-28
+
+**A correctness and observability release.** The headline is not speed: it is that a
+restart now tells you the truth about itself, a store stops lying about how much it
+holds, and the engine stops doing work nobody asked for. There is a performance
+improvement and it is modest; it is stated exactly below rather than rounded up.
+
+### The dark restart — fixed at the root
+
+A service could stop cleanly, exit 0, having awaited `close()` on every store it held,
+and its next boot would announce `Overwriting stale writer lock … appears dead` for
+every one of them. Nothing had crashed. Two deployments hit this; the same defect also
+made those boots pay a crash-recovery fold they did not owe.
+
+The cause was not the lock. `close()` released it correctly — when it got there. A
+failure part-way through close skipped both the release AND the clean-shutdown marker,
+and "the recorded pid is gone" reads identically for an orderly restart and a crash.
+
+- `close()` is now two parts and the second is unconditional: the flush-request watcher,
+ the **writer lock**, the VFS timers and the terminal `closed` flag are released whether
+ the durable steps succeeded or not. The original failure is narrated with what it costs
+ the next open, then rethrown.
+- Releasing the lock writes a **clean-close record** naming the lock generation it gave
+ up. The next open reads that record instead of guessing: recorded → nothing to recover;
+ absent → it says so, and names the recovery it is about to run. This also ends two
+ long-standing false alarms — a recycled pid locking a store out of its own reopen, and
+ `Re-acquiring writer lock … this is a bug` after a perfectly clean close.
+- The signal path stopped failing in a batch. One store's failing flush used to strand
+ every remaining store's lock and markers — at exit code 0. Now: per-store isolation, the
+ generation store's close (the marker) is part of shutdown, the lock goes in a `finally`,
+ and the handler no longer calls `process.exit()` when the host application has its own
+ signal handler, a race that truncated the host's own shutdown mid-flight.
+
+### The count ledger stops lying, and `counts.json` is written atomically
+
+The all-tier scalars are the denominator a coverage check subtracts against. A ledger
+derived under the old rule — one entity per id DIRECTORY — counted ghost and scar
+containers as rows, and was only FLAGGED suspect: it went on serving wrong numbers for
+the life of the store. Two copies of one archive could disagree, and a downstream index
+heal reported remaining work that did not exist.
+
+- Such a ledger now derives itself honestly **in the background** after the open, counting
+ identity records, and persists the correction stamped. Nothing waits for it, because no
+ read is served from a denominator.
+- A derivation that raced a write refuses to stamp its number: one retry on a quiet store,
+ then the ledger stays SUSPECT and names `repairIndex()` as the door that recounts under
+ a barrier.
+- `counts.json` is written temp+rename. A truncating write left a window in which a
+ concurrent reader saw the file EMPTY — and an unparseable ledger sends the next open
+ down the full-rescan path, so the cheapest file in the store was buying the most
+ expensive recovery.
+
+### An open and a repair narrate themselves — on a channel a log level cannot silence
+
+A store could open for three minutes and print nothing at all. The phase timings existed;
+they were written to a channel that every production-looking environment clamps away.
+
+- Narration moved to an always-visible channel. An open now heartbeats the phase it is in,
+ names each phase as it ends with what it was paying for, and names the expensive STEP
+ inside a phase. `repairIndex()` does the same and its receipt carries a per-family
+ `durationMs` — a repair that ran for half an hour with no output could only be watched
+ through `top`.
+- A brain nobody has written to now does nothing: a flush over a clean store is a no-op
+ and says nothing, the graph index's auto-flush asks before it acts, and the
+ cross-process flush-request watch is **event-driven** (`fs.watch`) instead of polling a
+ directory every 500 ms per store forever, with a slow safety sweep behind it and a
+ narrated fall back to polling where a filesystem cannot be watched.
+- A provider that is REBUILDING ITSELF is no longer confused with a broken one. `init()`
+ does not wait for it, every other family serves, and that family's doors refuse **by
+ name, carrying the provider's own progress**, saying plainly that they open by
+ themselves and no action is needed. Health narration dedupes by content, so an unchanged
+ verdict is silent however a provider's generation counter moves.
+
+### For operators — one behaviour change
+
+**Four `where` operators that previously returned an empty page now raise
+`INVALID_QUERY`:** `startsWith`, `endsWith`, `matches` and `length`. An equality/range
+posting index cannot evaluate a substring, a pattern or an array length without reading
+every row, and it now refuses by name instead of answering with an empty result that
+looks like an answer.
+
+**Three that previously returned an empty page are now SERVED:** `hasAll`, `noneOf` and
+`excludes`. All 25 accepted operator tokens now agree between this engine and its
+accelerated counterpart.
+
+### Performance — stated exactly
+
+Measured on a 14,056-noun / 72,679-verb production-shaped store, both builds solo under
+an exclusive lock:
+
+- **Warm reopen after a clean close: 85.7 s → 77.0 s (−10.2%).** The whole of that gain is
+ one fix — generation discovery reads directory NAMES instead of recursively walking the
+ entire generation log (−9.2 s, and it scales with history rather than row count). The
+ VFS phase is **unchanged**.
+- **Cold open: −31.4 s** (518.1 s → 486.7 s), of which the count-ledger derivation moving
+ off the critical path accounts for storage-init dropping 5,941 ms → 25 ms.
+- **A dominant ~38 s remains, diagnosed and NOT fixed.** It is not the VFS — the VFS's own
+ init is under 2 s of that phase. It is the log-authority adoption and/or the
+ pending-embed log recovery, both now instrumented so the next measurement names the
+ culprit outright.
+
+Continuing work, named so nobody has to rediscover it: that ~38 s term; making the
+generation store's committed-range set lazy; the hydration path that substitutes
+`Date.now()` for an unreadable stored timestamp (inventing data); and a VFS path-prefix
+filter built with a `$startsWith` spelling no operator set accepts, so
+`searchFiles({ path })` throws today.
+
+---
+
+## v10.4.3 — 2026-08-27 (Open Brainy's first release)
+
+**`@soulcraftlabs/brainy` 10.4.3 is the same engine as `@soulcraft/brainy` 10.4.2, byte for
+byte — only the name, the registry, and the pointers changed.** Install:
+
+```bash
+npm install @soulcraftlabs/brainy
+```
+
+with the registry line in your `.npmrc` (anonymous read):
+
+```
+@soulcraftlabs:registry=https://source.soulcraft.com/api/packages/soulcraftlabs/npm/
+```
+
+- **The Source is the one registry.** Open Brainy publishes to source.soulcraft.com only; the
+ npmjs republish step is retired from the release rail. Existing npmjs versions of
+ `@soulcraft/brainy` stay as they are and receive no new versions.
+- **The repository moved** to `soulcraftlabs/open-brainy` on The Source; the old path redirects.
+- **No engine change.** Everything in the 10.4.2 notes applies unchanged; adoption is one
+ install-line change (`@soulcraft/brainy` → `@soulcraftlabs/brainy`), which downstream
+ applications make together with their native-engine bump.
+
+## v10.4.2 — 2026-08-27 (a zero-norm vector is not a vector)
+
+**This is the last release of the MIT engine under the `@soulcraft/brainy` name.**
+The MIT package continues as **Open Brainy** — `@soulcraftlabs/brainy`: the open API,
+client library, types and protocol, an openly specified canonical format, and the TypeScript
+reference engine, scoped honestly as a single-node engine for stores up to roughly one
+million rows. The `@soulcraft/brainy` name passes to the native engine, **Brainy**, at a
+major version bump; that engine implements the same API over the same open format at
+production scale, requires a license, and refuses loudly without one. Nothing changes
+for existing installs until that major ships; the move is announced with it.
+
+Six fixes, one law: a vector with no magnitude carries no information, so it must
+never reach a vector index — in any engine — and the canonical store must say so.
+
+- **The permanently-unvectored row.** `add({ ..., vector: [] })` (and the same item
+ shape in `addMany` / `transact`) is now the sanctioned "no vector" row: persisted
+ with an empty vector leg, never embedded, never indexed, counted as unvectored in
+ the canonical ledger. Metadata-only rows — telemetry tallies, counters, plumbing —
+ no longer need a placeholder vector and never enter the vector leg. `vector: []`
+ together with `deferEmbedding: true` is refused with a typed error (a supplied
+ vector has nothing to defer). Previously `vector: []` threw a dimension error.
+- **The unvector door.** `update({ id, vector: [] })` (and its `transact()` twin) is
+ the sanctioned way to strip a vector from an existing row: canonical vector → `[]`,
+ removal from the vector index, the vectored ledger decremented exactly once — and
+ idempotent, so a resumed cleanup pass may simply re-issue. It never re-embeds, and
+ it clears a pending deferred-embed marker durably so the background worker cannot
+ re-vector the row later. Note that a rebuild never sheds vectors (it re-derives the
+ index from canonical rows); shedding historical vectors needs this door.
+- **Zero-norm vectors are normalized at the write.** An explicit all-zero vector on
+ any write path is persisted as unvectored (`[]`) with one warning naming the row;
+ the vector-index operations keep their own refusal as a second line. The engine's
+ own VFS root, which used to persist a deliberate all-zero placeholder (harmless
+ under cosine distance, a false attractor under a downstream engine's
+ squared-euclidean serving — a production incident this week), is now created
+ unvectored, and an existing store's legacy root is migrated on open by a single
+ fixed-path read before the health gate runs — never a walk.
+- **Enumeration keys on the identity record.** `getNouns()` / `getVerbs()` and the
+ cursor walks behind them enumerate by the metadata record, the same key the
+ canonical ledger counts by — previously the walk keyed on the vector file, so a
+ row holding metadata but no vector was counted yet never yielded (a permanent
+ "missing" phantom in coverage math), while an orphaned vector-only directory
+ could be yielded as a phantom id. The recovery fold also never deletes an existing
+ vector when it replays a metadata-only after-image (preserve-if-absent). One
+ documented gap remains: a verb's endpoints live only in its vector leg, so a
+ metadata-only verb is counted and loudly skipped, never fabricated — the fix is a
+ canonical-format change and lands with the open format.
+- **The ledger's one-time derivation counts identity records.** Stores upgraded from
+ pre-ledger versions derived their ALL-visibility scalars once by counting id
+ directories, which included ghost and scar containers left by an old partial-delete
+ defect — an inflated denominator whose coverage row could never reach exact. The
+ derivation now counts only directories holding a metadata record, `counts.json`
+ carries a derivation-rule stamp, and a ledger derived under the old rule is marked
+ `suspect` at open (one O(1) field read, one warning) so the online `repairIndex()`
+ path clears it with a real recount.
+- **The vector index refuses what it cannot hold.** `rebuild()` skips unvectored and
+ zero-norm rows (one summary line), re-pins the vector dimension from the first real
+ vector after a restart (previously a restart left the pin unset, so a wrong-length
+ insert became the new pin instead of being rejected), and `addItem` / `updateItem`
+ throw a typed `EmptyVectorIndexError` on a length-0 vector instead of ever storing
+ a vector-less node.
+- **Smaller:** a failing plugin activation now rethrows with the original error as
+ `cause` (the originating file and line survive to the caller's log); build
+ generators stamp from the repository history of their inputs instead of wall clock,
+ so two builds of the same tree are byte-identical.
+
+Adoption: one restart, paired with its native-engine release. The first open of an
+existing store runs the legacy-root migration (one narrated line) and, on stores that
+upgraded from pre-ledger versions, marks the ledger suspect until the next sanctioned
+recount — no rebuild in either case.
+
+## v10.4.1 — 2026-08-26 (reads refuse per family; an unchanged write never re-embeds)
+
+Two production defects from the same week, fixed together as a patch to 10.4.0.
+
+- **The read gate is per family.** A read now refuses only when the index family it
+ actually consults is unhealthy: a metadata filter is served while the vector leg is
+ rebuilding; a semantic query is refused only by the vector family; a graph
+ traversal only by the graph family. Previously any unhealthy family refused every
+ read on the brain — under a long vector rebuild, a production deployment's
+ metadata-only reads were refused for the duration, and the retries became a write
+ pump of their own.
+- **Unchanged data never re-embeds.** `update()` compares the incoming `data`
+ structurally with the stored record; an update carrying identical data (a common
+ shape for periodic upserts) no longer embeds again and no longer churns the vector
+ leg. Previously every such update re-embedded and re-inserted, which under load
+ saturated the vector index with near-identical vectors.
+
+Adoption: one restart, paired with its native-engine release.
+
+## v10.4.0 — 2026-08-25 (the health report has a name)
+
+Three related cures, one root cause: an index deciding whether it could be trusted
+by sampling itself instead of by exact accounting. This release replaces every
+sampled self-probe with ledger-derived truth, and a read against an unhealthy index
+now refuses loudly instead of guessing.
+
+- **The canonical count ledger.** Storage now tracks two scalars per family
+ (nouns/verbs) on the write path: the user-facing `counted` total — unchanged,
+ still what `getNounCount()` / `getVerbCount()` return — and a new ALL-visibility
+ `all` total covering every tier, the real denominator a derived index's own
+ coverage math needs. The unfiltered storage-level `totalCount` returned by
+ `getNouns()` / `getVerbs()` is now this unclamped ALL scalar; previously it could
+ only ever move up (`Math.max(scalar, scanned)`), so an inflated counter could
+ never self-correct. A delete that cannot prove the record it removed actually
+ existed (no canonical read, no prior image available) no longer decrements on
+ faith — it marks the ledger `suspect` (narrated once per session) instead of
+ silently drifting, and the next `repairIndex()` clears the flag with a real
+ recount.
+- **One contract for a throwing health probe.** A provider's `validateInvariants()`
+ is documented to never throw — but if one does anyway (a bug, a transient fault),
+ it is now read the same way everywhere: `heal: 'none'`, the error named in the
+ report, never synthesized into a rebuild trigger and never swallowed into "looks
+ fine." A flaky check can no longer buy itself a rebuild. `repairIndex()`'s
+ per-family receipt also gains `missing` (an exact count plus a capped id sample),
+ `rebuilt` (a full rebuild ran, vs. an incremental heal), and `reason`.
+- **The named health report; reads refuse instead of rebuilding.** Any index
+ provider may now expose a synchronous, O(1) `healthReport()` — composed from the
+ provider's own exact ledgers, never a sample — and this is the one signal
+ Brainy's read gate trusts. The first-query lazy-build path is gone: `brain.init()`
+ now runs every needed rebuild to completion before it returns, always, regardless
+ of dataset size. A read that lands on a provider whose health report says it
+ isn't serving throws a typed error instead of triggering a rebuild mid-query —
+ `GraphIndexNotReadyError`, `MetadataIndexNotReadyError`, or
+ `VectorIndexNotReadyError` (all exported from `@soulcraft/brainy`), naming the
+ reasons. `repairIndex({ rebuild: ['metadata' | 'graph' | 'vector'] | 'all' })` is
+ the new explicit operator door: it rebuilds the named family unconditionally, no
+ health check consulted — reach for it when you have independent reason to
+ distrust a family regardless of what it self-reports. Bare `repairIndex()` is
+ unchanged in spirit: report-driven, heals only what its own checks say needs it.
+- New concept doc: [Index Health](docs/concepts/index-health.md) walks the whole
+ story from a consumer's side — degraded-but-serving vs. not-ready, what
+ `repairIndex()` checks and heals per family, what `suspect` counts mean.
+
+**Nothing to change to adopt this.** No API removed, no signature narrowed —
+`repairIndex()` gains an optional options bag and its return value gains fields,
+both additive. The honest notes: if your code ever relied on a `find()` against a
+cold/not-yet-built index quietly triggering a rebuild and returning results a beat
+later, that behavior is gone — it now throws one of the three typed
+`*NotReadyError` classes instead (catch them if you need to distinguish "not ready
+yet" from "no results"). And `disableAutoRebuild: true` no longer defers index
+construction to the first query — a needed rebuild always runs at `open()` now;
+the flag has no effect on timing. Full manual control still lives in
+`repairIndex({ rebuild: [...] })`.
+
+- **Crash-reopen catchup.** After an unclean shutdown, the metadata index now
+ folds the exact fact window it missed — `find()` serves every acked write on
+ reopen, closing the gap where canonical reads and counts recovered a
+ crash-window write but the index kept serving its pre-crash state until the
+ next full rebuild. Related root-cause fixed alongside: `close()` never
+ stamped the index watermarks (only `flush()` did), so a close without a
+ prior flush caused a needless full rescan verdict on the next open.
+- **Relation rows are live in the metadata index.** Previously verb rows
+ entered the metadata index only during a rebuild — so a rebuilt store's
+ relation postings went stale from the first `relate()` after it. Relations
+ are now posted and retracted on the live write path (relate / unrelate /
+ updateRelation / remove's cascade, and their `transact()` forms), in the
+ same commit as the graph leg.
+- **The metadata rebuild is online.** `rebuild()` for the metadata family no
+ longer clears and rebuilds in place (reads went empty for the duration): it
+ builds a complete replacement beside the serving index, mirrors concurrent
+ writes to both, swaps atomically, and persists once after the swap. Reads
+ never observe a partial index. `repairIndex({ rebuild: ['metadata'] })` uses
+ it automatically.
+- **Incremental heal is routed.** A provider invariant that asks for the
+ incremental heal (`heal: 'repair'`) now routes to the provider's own
+ `repair()` when it exposes one — re-posting exactly what its ledger names,
+ never a store-sized rebuild — and the post-heal re-read of the report decides
+ success; a repair that doesn't converge is recorded with the escalation named.
+- **The vector family joins the count ledger.** `getCanonicalCounts()` gains
+ `vectors: { all }` — the count of canonical entities holding a real vector
+ (deferred-embed entities count when their vector lands). And the open gate
+ closes the vector leg: a store whose canonical rows hold vectors but whose
+ derived vector index is empty now builds at `open()` (or refuses with the
+ typed error) instead of silently serving empty vector-search results.
+- **An unknown storage config shape fails loudly.** A nested `config` object
+ carrying a path-shaped key (a shape that was never supported) used to fall
+ through silently to the default shared directory — every instance writing one
+ store while callers believed each had its own. It now throws, naming the
+ canonical `path` key.
+- **Relation index rows are JSON-safe.** Internal endpoint identifiers can no
+ longer ride the metadata-index crossing (a native provider serializes it);
+ they stay on the graph operations where they belong.
+- **A broken accelerator install can never read as "not installed."** The
+ auto-detection free pass now requires the resolution error to name the
+ accelerator package itself, exactly — a missing platform-binary sibling
+ package, an inner file path, or a dependency failure is a broken install and
+ `init()` throws loudly. And a plugin that declines activation is narrated on
+ the always-on log channel, so `silent: true` can no longer hide a fallback
+ to the default engines.
+
+---
+
## v10.3.1 — 2026-08-18 (the fold that behaves)
Three recovery cures from one production first-boot incident (a brain's first
diff --git a/SECURITY.md b/SECURITY.md
index 1f3c4732..91d40d49 100644
--- a/SECURITY.md
+++ b/SECURITY.md
@@ -30,7 +30,7 @@ commit to backporting fixes to unsupported lines.
## Scope
-This policy covers the `@soulcraft/brainy` package itself — the code in
+This policy covers the `@soulcraftlabs/brainy` package itself — the code in
this repository. If you're evaluating a deployment that also uses
`@soulcraft/cor`, report issues in that package the same way, to the same
address; we'll route internally.
diff --git a/bin/brainy-ts.js b/bin/brainy-ts.js
index 4e9aedb8..90a35e98 100644
--- a/bin/brainy-ts.js
+++ b/bin/brainy-ts.js
@@ -3,7 +3,7 @@
/**
* Modern TypeScript CLI Runner
*
- * This is the entry point after npm install @soulcraft/brainy
+ * This is the entry point after npm install @soulcraftlabs/brainy
* It runs the compiled TypeScript CLI code
*/
diff --git a/bun.lock b/bun.lock
index c31b3865..1e3e66e2 100644
--- a/bun.lock
+++ b/bun.lock
@@ -3,7 +3,7 @@
"configVersion": 0,
"workspaces": {
"": {
- "name": "@soulcraft/brainy",
+ "name": "@soulcraftlabs/brainy",
"dependencies": {
"@aws-sdk/client-s3": "^3.540.0",
"@azure/identity": "^4.0.0",
diff --git a/docs/DEVELOPER_LEARNING_PATH.md b/docs/DEVELOPER_LEARNING_PATH.md
index 4134ae63..b2d22fb1 100644
--- a/docs/DEVELOPER_LEARNING_PATH.md
+++ b/docs/DEVELOPER_LEARNING_PATH.md
@@ -25,13 +25,13 @@
### Prerequisites
```bash
-npm install @soulcraft/brainy
+npm install @soulcraftlabs/brainy
```
### Your First Neural Database
```typescript
-import { Brainy, NounType } from '@soulcraft/brainy'
+import { Brainy, NounType } from '@soulcraftlabs/brainy'
// Step 1: Create and initialize Brainy
const brain = new Brainy({
@@ -143,7 +143,7 @@ Once you're comfortable with basic operations, move to **Level 2** to learn abou
### Building a Knowledge Graph
```typescript
-import { Brainy, NounType, VerbType } from '@soulcraft/brainy'
+import { Brainy, NounType, VerbType } from '@soulcraftlabs/brainy'
const brain = new Brainy({ storage: { type: 'memory' } })
await brain.init()
@@ -314,7 +314,7 @@ Ready for AI-powered search and clustering? Move to **Level 3**.
### Triple Intelligence in Action
```typescript
-import { Brainy, NounType, VerbType } from '@soulcraft/brainy'
+import { Brainy, NounType, VerbType } from '@soulcraftlabs/brainy'
const brain = new Brainy({ storage: { type: 'memory' } })
await brain.init()
@@ -529,7 +529,7 @@ Want to treat files as intelligent entities? Learn the **Virtual Filesystem** in
### Files as Intelligent Entities
```typescript
-import { Brainy, NounType, VerbType } from '@soulcraft/brainy'
+import { Brainy, NounType, VerbType } from '@soulcraftlabs/brainy'
const brain = new Brainy({ storage: { type: 'memory' } })
await brain.init()
@@ -832,7 +832,7 @@ Ready for production deployment? Level 5 covers **planet-scale architecture**.
### Production-Ready Deployment
```typescript
-import { Brainy, NounType } from '@soulcraft/brainy'
+import { Brainy, NounType } from '@soulcraftlabs/brainy'
// 1. PRODUCTION STORAGE - Filesystem with off-site snapshots
console.log('Initializing production storage...\n')
diff --git a/docs/FIND_SYSTEM.md b/docs/FIND_SYSTEM.md
index 1cc38ce9..6aa33515 100644
--- a/docs/FIND_SYSTEM.md
+++ b/docs/FIND_SYSTEM.md
@@ -369,6 +369,71 @@ return results.slice(offset, offset + limit)
// → Auto-correction: Use most likely alternative based on affinity data
```
+## Field Projection (`fields`)
+
+`find()` and `get()` accept a `fields` list. Without it they return the whole
+record; with it they return only the fields you name — and, where the index can
+supply them, without opening the canonical record at all.
+
+```ts
+// A list page: two user fields and one engine scalar. No document bodies.
+await brain.find({
+ where: { kind: 'post' },
+ fields: ['title', 'slug', 'system.createdAt'],
+ limit: 50
+})
+
+await brain.get(id, { fields: ['title'] })
+```
+
+### Why it exists
+
+A list view that renders a title and a date does not need the body, but without
+a projection every row hydrates its full record and throws almost all of it
+away. On a posts list that is the dominant cost of the query.
+
+### The rules
+
+| | |
+|---|---|
+| **`fields` absent** | The full record, byte-identical to before. Nothing changes. |
+| **Field names** | The one addressing law: a bare name is user metadata (`'title'`), `system.*` is an engine scalar (`'system.createdAt'`). |
+| **A field the row lacks** | Simply **absent** from the result. Never an error. |
+| **Identity** | Every row keeps its `id` (and `score` on `find`) regardless — a row you cannot identify is not a row. |
+| **Where values come from** | The **column store**, which holds raw values. Never the sparse index, which buckets timestamps for range queries. |
+| **A field the column cannot serve** | The canonical record is read for that field only. Correct, just not free. |
+
+### Missing fields are absent, not errors
+
+This is deliberate and differs from `orderBy`, which throws
+`UnresolvableFieldError` for an unknown field. A typo in `orderBy` silently
+changes the ordering, so it must be loud. A projection asks "give me these if
+you have them", and an optional field must not turn a list into a failure — so
+`fields` uses the permissive path.
+
+### Cost
+
+When every named field is column-served, a projected page performs **zero**
+canonical reads. When one is not, only that read happens and the rest still come
+from the index. Both are pinned by counting reads rather than timing them, in
+`tests/integration/find-fields-projection.test.ts`.
+
+### `related()` takes no `fields`
+
+A `Relation` carries `from` and `to` as **ids** and hydrates no entity record,
+so there is nothing for a projection to trim. Projecting the endpoints would be
+a new capability rather than a projection of an existing one.
+
+### For engine implementers
+
+Projection is served through an optional provider door,
+`getScalarsForIds(ids, fields)` on `MetadataIndexProvider`. The contract is in
+`src/plugin.ts`; the short version is **return only what you can serve exactly,
+and say what you served**. The caller diffs the answer against the request and
+reads records for the remainder, so omission costs a read while a wrong value is
+a wrong answer nobody can see. An engine without the door still works — every
+field falls back to the record.
+
## Performance Characteristics
### Query Performance by Type
@@ -1217,7 +1282,7 @@ where: {
await brain.find({ type: 'Document' })
// ✅ Correct: Use NounType enum
-import { NounType } from '@soulcraft/brainy'
+import { NounType } from '@soulcraftlabs/brainy'
await brain.find({ type: NounType.Document })
// ❌ Error: Operator not recognized
diff --git a/docs/MIGRATION-V3-TO-V4.md b/docs/MIGRATION-V3-TO-V4.md
index 29c409ac..680b6928 100644
--- a/docs/MIGRATION-V3-TO-V4.md
+++ b/docs/MIGRATION-V3-TO-V4.md
@@ -153,13 +153,13 @@ brainy-data/
### Step 1: Update Brainy Package
```bash
-npm install @soulcraft/brainy@latest
+npm install @soulcraftlabs/brainy@latest
```
**Check your version:**
```bash
-npm list @soulcraft/brainy
-# Should show: @soulcraft/brainy@4.0.0
+npm list @soulcraftlabs/brainy
+# Should show: @soulcraftlabs/brainy@4.0.0
```
### Step 2: No Code Changes Required! ✅
@@ -374,7 +374,7 @@ If you encounter issues, you can rollback:
```bash
# Reinstall v3
-npm install @soulcraft/brainy@^3.50.0
+npm install @soulcraftlabs/brainy@^3.50.0
# Restart application
```
@@ -389,7 +389,7 @@ rm -rf ./data
cp -r ./data-backup ./data
# Reinstall v3
-npm install @soulcraft/brainy@^3.50.0
+npm install @soulcraftlabs/brainy@^3.50.0
```
## Common Migration Scenarios
@@ -539,7 +539,7 @@ console.log('Storage type:', status.type)
**Migration Checklist:**
- ✅ Backup data
-- ✅ Update npm package (`npm install @soulcraft/brainy@latest`)
+- ✅ Update npm package (`npm install @soulcraftlabs/brainy@latest`)
- ✅ Restart application (automatic migration)
- ✅ Verify data integrity
- ✅ Enable lifecycle policies
diff --git a/docs/PERFORMANCE.md b/docs/PERFORMANCE.md
index 248a2c70..b543e84a 100644
--- a/docs/PERFORMANCE.md
+++ b/docs/PERFORMANCE.md
@@ -323,58 +323,24 @@ Only the graph adjacency index carries a committed scale assertion:
- ✅ **Single-Node by Design**: One process owns one `path`; scale out at the service layer
- ✅ **Zero Stubs**: Every line of code is production-ready
-## Lazy Loading Performance
+## Index Build at Open (10.4+)
-Brainy supports two initialization modes for optimal performance across different use cases:
+As of 10.4, `brain.init()` runs every needed index rebuild to completion before
+it returns — always, regardless of dataset size. There is no lazy,
+first-query rebuild path: a brain either finishes opening healthy, or `init()`
+fails loudly. `disableAutoRebuild` no longer defers index construction to a
+first query; it has no effect on *when* a rebuild runs. Manual control over
+rebuilds is `repairIndex({ rebuild: [...] })`. See
+[Index Health](concepts/index-health.md) for the full read-gate contract
+(providers self-report readiness via `healthReport()`; a read against a
+not-serving provider throws a typed `*NotReadyError` rather than rebuilding
+mid-query).
-### Mode 1: Auto-Rebuild (Default)
-
-```javascript
-const brain = new Brainy()
-await brain.init() // Rebuilds indexes during init (~500ms-3s for 10K entities)
-```
-
-**Performance:**
-- Init time: 500ms-3s (depends on dataset size)
-- First query: Instant (indexes already loaded)
-- Use case: Traditional applications, long-running servers
-
-### Mode 2: Lazy Loading
-
-```javascript
-const brain = new Brainy({ disableAutoRebuild: true })
-await brain.init() // Returns instantly (0-10ms)
-
-const results = await brain.find({ limit: 10 }) // First query triggers rebuild (~50-200ms)
-const more = await brain.find({ limit: 100 }) // Subsequent queries instant (0ms check)
-```
-
-**Performance:**
-- Init time: 0-10ms (instant)
-- First query: 50-200ms (includes index rebuild for 1K-10K entities)
-- Subsequent queries: 0ms check (instant)
-- Concurrent queries: Wait for same rebuild (mutex prevents duplicates)
-
-**Concurrency Safety:**
-```javascript
-// 100 concurrent queries immediately after init
-await brain.init()
-
-const promises = Array.from({ length: 100 }, () =>
- brain.find({ limit: 10 })
-)
-
-const results = await Promise.all(promises)
-// ✅ Only 1 rebuild triggered (mutex)
-// ✅ All 100 queries return correct results
-// ✅ Total time: ~60ms (not 6000ms!)
-```
-
-**Use Cases for Lazy Loading:**
-- **Serverless/Edge**: Minimize cold start time (0-10ms init)
-- **Development**: Faster restarts during development
-- **Large datasets**: Defer index loading until needed
-- **Read-heavy workloads**: Writes don't wait for index rebuild
+
## Zero Configuration Required
@@ -384,10 +350,6 @@ Brainy is designed to be **smart enough to tune itself dynamically**. No configu
// That's it. Brainy handles everything.
const brain = new Brainy()
await brain.init()
-
-// Or with lazy loading for serverless
-const brain = new Brainy({ disableAutoRebuild: true })
-await brain.init() // Instant (0-10ms)
```
### Automatic Self-Tuning
@@ -395,7 +357,6 @@ await brain.init() // Instant (0-10ms)
- **Metadata Index**: Auto-builds sorted indices for range queries on first use
- **Graph Index**: Auto-flushes every 30 seconds
- **Default Tuning**: Research-based vector index defaults
-- **Lazy Loading**: Indices built only when needed
- **Cache Management**: LRU caches with TTL
### Intelligent Defaults
diff --git a/docs/PLUGINS.md b/docs/PLUGINS.md
index d9a4d3e7..238d2252 100644
--- a/docs/PLUGINS.md
+++ b/docs/PLUGINS.md
@@ -10,7 +10,7 @@ next:
- guides/storage-adapters
---
-# Plugin Development Guide
+# Plugin System
Brainy has a plugin system that allows third-party packages to replace internal subsystems with custom implementations. This is how `@soulcraft/cor` provides optional native acceleration, and it's the same system available to any developer.
@@ -46,7 +46,7 @@ If no plugin provides a given key, brainy uses its built-in JavaScript implement
### 1. Implement the `BrainyPlugin` interface
```typescript
-import type { BrainyPlugin, BrainyPluginContext } from '@soulcraft/brainy/plugin'
+import type { BrainyPlugin, BrainyPluginContext } from '@soulcraftlabs/brainy/plugin'
const myPlugin: BrainyPlugin = {
name: 'my-brainy-plugin', // Must be unique (typically your npm package name)
@@ -90,7 +90,7 @@ await brain.init()
**Programmatic registration:** For plugins not installed as npm packages, use `brain.use()`:
```typescript
-import { Brainy } from '@soulcraft/brainy'
+import { Brainy } from '@soulcraftlabs/brainy'
import myPlugin from './my-plugin.js'
const brain = new Brainy()
@@ -200,15 +200,30 @@ members so a warm reopen never pays a redundant rebuild-from-canonical:
- **`init?(): Promise`** — eager cold-load. Brainy awaits it once during
`brain.init()`, after the metadata provider's `init()` (the id-mapper hydrates first)
and **before the rebuild gate**.
-- **`isReady?(): boolean`** — honest durability signal. `true` ⇔ the persisted index is
- loaded (or cheaply demand-loadable) and consistent with what was last persisted. When
- exposed, the rebuild gate defers to this signal **instead of** the `size() === 0` /
- `totalEntries === 0` heuristics — a disk-native index may report 0 resident entries
- while fully durable. Never return `true` if the durable state failed to load: the
- signal is honest in both directions, and a not-ready provider gets its rebuild even
- when `size() > 0`.
+- **`healthReport?(): HealthReport`** — the PREFERRED signal (10.4+). A named,
+ synchronous, O(1) verdict derived from the provider's own exact ledgers — never a
+ sample, never I/O, must never throw for a well-formed provider. Brainy's read gate
+ (`assessProviderHealth()`) reads this INSTEAD of `isReady()` / size heuristics when
+ present: `serving: false` refuses the read with a typed `*NotReadyError` rather than
+ triggering a rebuild — a read never starts a store walk. `healthy` marks every
+ *verified* invariant holding; a family named in `unledgered` counts as neither
+ healthy nor broken. See `HealthReport` / `LedgerInvariantResult` /
+ `InvariantSource` in `src/plugin.ts`, and
+ [Index Health](concepts/index-health.md) for the consumer-facing story.
+- **`isReady?(): boolean`** — honest durability signal, the fallback when
+ `healthReport()` is absent. `true` ⇔ the persisted index is loaded (or cheaply
+ demand-loadable) and consistent with what was last persisted. When exposed, the
+ gate defers to this signal **instead of** the `size() === 0` / `totalEntries === 0`
+ heuristics — a disk-native index may report 0 resident entries while fully durable.
+ Never return `true` if the durable state failed to load: the signal is honest in
+ both directions, and a not-ready provider gets its rebuild even when `size() > 0`.
- **`isMigrating?(): boolean`** — while `true`, the provider owns its index (background
migration); brainy skips its rebuild entirely.
+- **`validateInvariants?(): Promise`** — the async DEEP
+ diagnostic (full scans allowed), distinct from the bounded, sync `healthReport()`.
+ Must never throw — a failure is `healthy: false` data, not an exception; a provider
+ that throws anyway is read as a loud, unverified failure (never as "healthy") by
+ every caller, never silently retried into a rebuild.
Providers that implement none of these keep the size/count heuristics — correct for
engines whose `rebuild()` *is* their load path (like brainy's built-in JS vector index).
@@ -257,10 +272,10 @@ When provided by an optional native acceleration plugin (such as `@soulcraft/cor
#### `cache`
**Type:** `UnifiedCache`
-Replaces the global `UnifiedCache` singleton used for VFS path resolution, semantic caching, and vector index caching. Must implement the `UnifiedCache` interface (available from `@soulcraft/brainy/internals`).
+Replaces the global `UnifiedCache` singleton used for VFS path resolution, semantic caching, and vector index caching. Must implement the `UnifiedCache` interface (available from `@soulcraftlabs/brainy/internals`).
```typescript
-import type { UnifiedCache } from '@soulcraft/brainy/internals'
+import type { UnifiedCache } from '@soulcraftlabs/brainy/internals'
context.registerProvider('cache', myNativeCache)
```
@@ -310,8 +325,8 @@ Plugins can register custom storage backends that users reference by name.
### Implementing a Storage Adapter
```typescript
-import type { StorageAdapterFactory } from '@soulcraft/brainy/plugin'
-import type { StorageAdapter } from '@soulcraft/brainy'
+import type { StorageAdapterFactory } from '@soulcraftlabs/brainy/plugin'
+import type { StorageAdapter } from '@soulcraftlabs/brainy'
class MyStorageAdapter implements StorageAdapter {
async init(): Promise { /* ... */ }
@@ -345,9 +360,9 @@ Brainy provides three entry points for plugin developers:
| Import Path | Contents | Stability |
|-------------|----------|-----------|
-| `@soulcraft/brainy` | Public API, types, StorageAdapter | Stable (semver) |
-| `@soulcraft/brainy/plugin` | BrainyPlugin, BrainyPluginContext, StorageAdapterFactory | Stable (semver) |
-| `@soulcraft/brainy/internals` | UnifiedCache, EntityIdMapper, logger utilities | Internal (may change between minor versions) |
+| `@soulcraftlabs/brainy` | Public API, types, StorageAdapter | Stable (semver) |
+| `@soulcraftlabs/brainy/plugin` | BrainyPlugin, BrainyPluginContext, StorageAdapterFactory | Stable (semver) |
+| `@soulcraftlabs/brainy/internals` | UnifiedCache, EntityIdMapper, logger utilities | Internal (may change between minor versions) |
## Diagnostics
@@ -425,7 +440,7 @@ A minimal but useful plugin that provides SIMD-accelerated distance calculations
```typescript
// simd-distance-plugin/src/plugin.ts
-import type { BrainyPlugin, BrainyPluginContext } from '@soulcraft/brainy/plugin'
+import type { BrainyPlugin, BrainyPluginContext } from '@soulcraftlabs/brainy/plugin'
// Hypothetical native module
import { simdCosineDistance } from './native.js'
@@ -455,7 +470,7 @@ export default simdDistancePlugin
"main": "./dist/plugin.js",
"types": "./dist/plugin.d.ts",
"peerDependencies": {
- "@soulcraft/brainy": ">=7.0.0"
+ "@soulcraftlabs/brainy": ">=7.0.0"
}
}
```
@@ -463,7 +478,7 @@ export default simdDistancePlugin
Usage:
```typescript
-import { Brainy } from '@soulcraft/brainy'
+import { Brainy } from '@soulcraftlabs/brainy'
const brain = new Brainy({ plugins: ['brainy-simd-distance'] })
await brain.init()
diff --git a/docs/PRODUCTION_SERVICE_ARCHITECTURE.md b/docs/PRODUCTION_SERVICE_ARCHITECTURE.md
index 4568cd31..ad4a4a40 100644
--- a/docs/PRODUCTION_SERVICE_ARCHITECTURE.md
+++ b/docs/PRODUCTION_SERVICE_ARCHITECTURE.md
@@ -54,7 +54,7 @@ After 40 API calls:
```typescript
// server.ts
-import { Brainy } from '@soulcraft/brainy'
+import { Brainy } from '@soulcraftlabs/brainy'
// SINGLETON INSTANCE
let brainInstance: Brainy | null = null
@@ -174,7 +174,7 @@ process.on('SIGTERM', async () => {
```typescript
// server.ts - Clean Bun implementation
-import { Brainy } from '@soulcraft/brainy'
+import { Brainy } from '@soulcraftlabs/brainy'
let brain: Brainy | null = null
diff --git a/docs/README.md b/docs/README.md
index 3290001f..ddb37d20 100644
--- a/docs/README.md
+++ b/docs/README.md
@@ -5,7 +5,7 @@
## Quick Start
```typescript
-import { Brainy, NounType, VerbType } from '@soulcraft/brainy'
+import { Brainy, NounType, VerbType } from '@soulcraftlabs/brainy'
const brain = new Brainy()
await brain.init()
diff --git a/docs/RELEASE-GUIDE.md b/docs/RELEASE-GUIDE.md
index 94c6a6cb..4b120bd8 100644
--- a/docs/RELEASE-GUIDE.md
+++ b/docs/RELEASE-GUIDE.md
@@ -99,7 +99,7 @@ Examples:
```bash
# 1. Deprecate wrong version on npm
-npm deprecate @soulcraft/brainy@X.X.X "Incorrect version - use Y.Y.Y"
+npm deprecate @soulcraftlabs/brainy@X.X.X "Incorrect version - use Y.Y.Y"
# 2. Fix version in package.json
# 3. Republish correct version
diff --git a/docs/SCALING.md b/docs/SCALING.md
index e9ae1136..054d2096 100644
--- a/docs/SCALING.md
+++ b/docs/SCALING.md
@@ -13,7 +13,7 @@
### In-Memory
```typescript
-import Brainy from '@soulcraft/brainy'
+import Brainy from '@soulcraftlabs/brainy'
const brain = new Brainy({ storage: { type: 'memory' } })
```
@@ -43,7 +43,7 @@ The native vector provider (via the optional `@soulcraft/cor` package) extends t
Numbers below are **measured** by `tests/benchmarks/find-composition-scale.js` (a single
Node 22 process, in-memory storage, 384-dim vectors, `balanced` recall). They are the
-open-core (pure-TypeScript) path — what you get from `@soulcraft/brainy` with no native
+open-core (pure-TypeScript) path — what you get from `@soulcraftlabs/brainy` with no native
provider installed. Run it yourself: `node --max-old-space-size=8192 tests/benchmarks/find-composition-scale.js 100000`.
`find()` query latency, p50 / p95 (200 queries each):
diff --git a/docs/api-contract.json b/docs/api-contract.json
new file mode 100644
index 00000000..c4f4e056
--- /dev/null
+++ b/docs/api-contract.json
@@ -0,0 +1,1633 @@
+{
+ "contractVersion": 1,
+ "engine": "@soulcraftlabs/brainy",
+ "compatibility": {
+ "minor": "additive — a new optional door, a new served operator, a new error class; every existing implementation still conforms",
+ "major": "breaking — a door removed, an answer narrowed, an ordering law changed, an optional door promoted to required, or an operator moved from served to refused"
+ },
+ "doors": [
+ {
+ "name": "adaptiveHistoryBudgetBytes",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "add",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "addMany",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "adoptLogAuthority",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "adoptLogAuthorityInner",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "aggViewFromEntity",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "anyProviderMigrating",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "applyFusionScoring",
+ "kind": "method",
+ "arity": 2
+ },
+ {
+ "name": "applyGraphConstraints",
+ "kind": "method",
+ "arity": 2
+ },
+ {
+ "name": "armIdleFlushTimer",
+ "kind": "method",
+ "arity": 2
+ },
+ {
+ "name": "asOf",
+ "kind": "method",
+ "arity": 2
+ },
+ {
+ "name": "assertGenerationStoreReady",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "assertWritable",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "audit",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "auditGraph",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "autoAdoptLegacyVfsBlobsIfNeeded",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "autoAlpha",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "autoCompactHistory",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "awaitMigrationLock",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "awaitPendingEmbeds",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "backfillAggregateIfNeeded",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "batchGet",
+ "kind": "method",
+ "arity": 2
+ },
+ {
+ "name": "brainWideStrictRequiresSubtype",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "bridgeLegacyPendingEmbedSidecars",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "buildAtGenerationVectors",
+ "kind": "method",
+ "arity": 2
+ },
+ {
+ "name": "buildGraphView",
+ "kind": "method",
+ "arity": 4
+ },
+ {
+ "name": "buildMetadataFilter",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "buildMigrationUpdate",
+ "kind": "method",
+ "arity": 5
+ },
+ {
+ "name": "buildRelationMigrationUpdate",
+ "kind": "method",
+ "arity": 5
+ },
+ {
+ "name": "cacheVerbInt",
+ "kind": "method",
+ "arity": 2
+ },
+ {
+ "name": "canServeVectorAtGeneration",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "captureEmbedCheckpoint",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "checkHealth",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "checkMigrations",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "clear",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "clearPendingEmbed",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "close",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "closeDurableSteps",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "cluster",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "collectProviderInvariants",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "compactHistory",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "consumeMetadataWatermarkVerdict",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "convertMetadataToEntity",
+ "kind": "method",
+ "arity": 2
+ },
+ {
+ "name": "convertNounToEntity",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "counts",
+ "kind": "accessor"
+ },
+ {
+ "name": "createGenerationStore",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "createIndex",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "createMigrationBackupIfNeeded",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "createPinnedDb",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "createResult",
+ "kind": "method",
+ "arity": 4
+ },
+ {
+ "name": "dbFinalizationRegistry",
+ "kind": "accessor"
+ },
+ {
+ "name": "dbHost",
+ "kind": "accessor"
+ },
+ {
+ "name": "defineAggregate",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "demoteTornEntityTreeStamp",
+ "kind": "method",
+ "arity": 4
+ },
+ {
+ "name": "detectIdKind",
+ "kind": "method",
+ "arity": 3
+ },
+ {
+ "name": "diagnostics",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "diff",
+ "kind": "method",
+ "arity": 2
+ },
+ {
+ "name": "embed",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "embedBatch",
+ "kind": "method",
+ "arity": 2
+ },
+ {
+ "name": "emitCommitted",
+ "kind": "method",
+ "arity": 4
+ },
+ {
+ "name": "enforceSubtypeOnAdd",
+ "kind": "method",
+ "arity": 4
+ },
+ {
+ "name": "enforceSubtypeOnRelate",
+ "kind": "method",
+ "arity": 4
+ },
+ {
+ "name": "enforceTrackedFieldValues",
+ "kind": "method",
+ "arity": 2
+ },
+ {
+ "name": "enhanceNLPResult",
+ "kind": "method",
+ "arity": 2
+ },
+ {
+ "name": "enqueuePendingEmbed",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "ensureAggregationIndex",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "ensureIndexesLoaded",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "ensureInitialized",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "entityForAggFromRawRecord",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "entityFromGenerationRecord",
+ "kind": "method",
+ "arity": 3
+ },
+ {
+ "name": "entityIntsToUuids",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "entityViewFromRawRecord",
+ "kind": "method",
+ "arity": 2
+ },
+ {
+ "name": "excludedVisibilityTiers",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "executeProximitySearch",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "executeTextSearch",
+ "kind": "method",
+ "arity": 2
+ },
+ {
+ "name": "executeTextSearchScored",
+ "kind": "method",
+ "arity": 3
+ },
+ {
+ "name": "executeVectorSearch",
+ "kind": "method",
+ "arity": 3
+ },
+ {
+ "name": "executeVectorSearchScored",
+ "kind": "method",
+ "arity": 3
+ },
+ {
+ "name": "explain",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "export",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "extract",
+ "kind": "method",
+ "arity": 2
+ },
+ {
+ "name": "extractConcepts",
+ "kind": "method",
+ "arity": 2
+ },
+ {
+ "name": "extractEntities",
+ "kind": "method",
+ "arity": 2
+ },
+ {
+ "name": "factSegmentPaths",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "fieldCountsAggregateName",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "fillSubtypes",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "filterIdsBelted",
+ "kind": "method",
+ "arity": 2
+ },
+ {
+ "name": "filterIdsWithinBelted",
+ "kind": "method",
+ "arity": 2
+ },
+ {
+ "name": "find",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "findAggregate",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "findDuplicates",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "findMatchingWords",
+ "kind": "method",
+ "arity": 3
+ },
+ {
+ "name": "flush",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "formatInfo",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "formatSubtypeError",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "generation",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "generationDigest",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "get",
+ "kind": "method",
+ "arity": 2
+ },
+ {
+ "name": "getActivePlugins",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "getAvailableFields",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "getBackgroundDeduplicator",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "getFieldsForType",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "getFieldStatistics",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "getFieldsWithCardinality",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "getFieldValues",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "getIndexStats",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "getIndexStatus",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "getMemoryStats",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "getNeighborUuids",
+ "kind": "method",
+ "arity": 2
+ },
+ {
+ "name": "getNounCount",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "getOptimalQueryPlan",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "getStats",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "getStorageType",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "getSubtypeRule",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "getTripleIntelligence",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "getTypedNeighbors",
+ "kind": "method",
+ "arity": 4
+ },
+ {
+ "name": "getVerbCount",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "graph",
+ "kind": "accessor"
+ },
+ {
+ "name": "graphAccelerationProvider",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "graphCommunities",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "graphCommunitiesFallback",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "graphCommunitiesNative",
+ "kind": "method",
+ "arity": 2
+ },
+ {
+ "name": "graphEntityInt",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "graphExport",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "graphExportFallback",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "graphExportNative",
+ "kind": "method",
+ "arity": 2
+ },
+ {
+ "name": "graphPath",
+ "kind": "method",
+ "arity": 3
+ },
+ {
+ "name": "graphPathFallback",
+ "kind": "method",
+ "arity": 3
+ },
+ {
+ "name": "graphPathNative",
+ "kind": "method",
+ "arity": 4
+ },
+ {
+ "name": "graphRank",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "graphRankFallback",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "graphRankNative",
+ "kind": "method",
+ "arity": 2
+ },
+ {
+ "name": "graphSubgraph",
+ "kind": "method",
+ "arity": 2
+ },
+ {
+ "name": "graphSubgraphFallback",
+ "kind": "method",
+ "arity": 4
+ },
+ {
+ "name": "graphSubgraphFromQuery",
+ "kind": "method",
+ "arity": 5
+ },
+ {
+ "name": "graphSubgraphNative",
+ "kind": "method",
+ "arity": 5
+ },
+ {
+ "name": "groupByLabel",
+ "kind": "method",
+ "arity": 2
+ },
+ {
+ "name": "hasStorageMethod",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "hasVectorOrTextCriteria",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "health",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "highlight",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "highlightSemanticPhase",
+ "kind": "method",
+ "arity": 5
+ },
+ {
+ "name": "history",
+ "kind": "method",
+ "arity": 2
+ },
+ {
+ "name": "historyStats",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "hub",
+ "kind": "accessor"
+ },
+ {
+ "name": "hydrateIdMapperForGraphRebuild",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "hydrateNativeSubgraph",
+ "kind": "method",
+ "arity": 2
+ },
+ {
+ "name": "hydrateResultPage",
+ "kind": "method",
+ "arity": 2
+ },
+ {
+ "name": "import",
+ "kind": "method",
+ "arity": 2
+ },
+ {
+ "name": "importPluginPackage",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "incidentEdges",
+ "kind": "method",
+ "arity": 3
+ },
+ {
+ "name": "indexStats",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "init",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "insights",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "isClosed",
+ "kind": "accessor"
+ },
+ {
+ "name": "isClosing",
+ "kind": "accessor"
+ },
+ {
+ "name": "isEmbeddingReady",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "isInfrastructureWrite",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "isInitialized",
+ "kind": "accessor"
+ },
+ {
+ "name": "isReadOnly",
+ "kind": "accessor"
+ },
+ {
+ "name": "kickBackgroundFlush",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "kickEmbedWorker",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "legacyLayoutMigrationPhase",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "loadAnalyticsGraph",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "loadPlugins",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "logAuthority",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "maintenanceDebt",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "materializeAtGeneration",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "maybeWriteEmbedCheckpoint",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "maybeWriteEmbedLowWater",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "metadataIndexRetractionOp",
+ "kind": "method",
+ "arity": 3
+ },
+ {
+ "name": "migrate",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "migrateField",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "migrateInternal",
+ "kind": "method",
+ "arity": 2
+ },
+ {
+ "name": "migrateLegacyZeroNormVfsRootIfNeeded",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "migrationSnapshot",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "neededFamiliesMigrating",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "neighbors",
+ "kind": "method",
+ "arity": 2
+ },
+ {
+ "name": "newId",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "nlp",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "normalizeConfig",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "noteEmbedCheckpointCadence",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "noteWriteForPersistence",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "now",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "onChange",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "pageConnectedIds",
+ "kind": "method",
+ "arity": 2
+ },
+ {
+ "name": "pagination",
+ "kind": "accessor"
+ },
+ {
+ "name": "parseMigrationPath",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "parseNaturalQuery",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "pathExists",
+ "kind": "method",
+ "arity": 2
+ },
+ {
+ "name": "pendingEmbedCount",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "pendingResult",
+ "kind": "method",
+ "arity": 2
+ },
+ {
+ "name": "performInit",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "persistPinnedGeneration",
+ "kind": "method",
+ "arity": 2
+ },
+ {
+ "name": "persistSingleOp",
+ "kind": "method",
+ "arity": 6
+ },
+ {
+ "name": "pickMetadataProbe",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "pickVectorProbe",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "pinGeneration",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "planGetEntity",
+ "kind": "method",
+ "arity": 3
+ },
+ {
+ "name": "planTransact",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "planTxAdd",
+ "kind": "method",
+ "arity": 3
+ },
+ {
+ "name": "planTxRelate",
+ "kind": "method",
+ "arity": 3
+ },
+ {
+ "name": "planTxRemove",
+ "kind": "method",
+ "arity": 3
+ },
+ {
+ "name": "planTxUnrelate",
+ "kind": "method",
+ "arity": 3
+ },
+ {
+ "name": "planTxUpdate",
+ "kind": "method",
+ "arity": 3
+ },
+ {
+ "name": "projectionGauges",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "providerForFamily",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "providerIsMigrating",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "providerMigrationStatus",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "queryAggregate",
+ "kind": "method",
+ "arity": 2
+ },
+ {
+ "name": "queryIndexFamilies",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "readPath",
+ "kind": "method",
+ "arity": 2
+ },
+ {
+ "name": "readPendingEmbedBound",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "ready",
+ "kind": "accessor"
+ },
+ {
+ "name": "rebuildIndexesIfNeeded",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "rebuildMetadataIndexOnline",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "reconcileLogDivergence",
+ "kind": "method",
+ "arity": 2
+ },
+ {
+ "name": "reconstructPath",
+ "kind": "method",
+ "arity": 4
+ },
+ {
+ "name": "recordStateAt",
+ "kind": "method",
+ "arity": 3
+ },
+ {
+ "name": "recoverPendingEmbedsFromLog",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "registerShutdownHooks",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "relate",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "related",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "relateMany",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "relationFromGenerationRecord",
+ "kind": "method",
+ "arity": 2
+ },
+ {
+ "name": "relationshipSubtypesOf",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "releaseGeneration",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "remove",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "removeAggregate",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "removeMany",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "removeMigrationBackupSafe",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "repackHistory",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "repairIndex",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "requestFlush",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "requireProviders",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "requireSubtype",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "resolveAsOfGeneration",
+ "kind": "method",
+ "arity": 2
+ },
+ {
+ "name": "resolveConnectedIds",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "resolveDiffEndpoint",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "resolveHiddenIds",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "resolveHNSWPersistMode",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "resolveRawGeneration",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "resolveRetentionPolicy",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "resolveVerbEndpointInts",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "resolveVerbIntsToIds",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "restore",
+ "kind": "method",
+ "arity": 2
+ },
+ {
+ "name": "rrfFusion",
+ "kind": "method",
+ "arity": 3
+ },
+ {
+ "name": "runAggregationBackfillWalk",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "runAggregationCatchUp",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "runEmbedWorker",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "runOracle",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "runRepairIndexPhases",
+ "kind": "method",
+ "arity": 5
+ },
+ {
+ "name": "scanFacts",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "seedIdsToInts",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "selectorToSeedIds",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "setRetentionBudget",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "setupEmbedder",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "setupIndex",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "setupStorage",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "similar",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "similarity",
+ "kind": "method",
+ "arity": 2
+ },
+ {
+ "name": "splitForHighlighting",
+ "kind": "method",
+ "arity": 2
+ },
+ {
+ "name": "stampBrainFormat",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "stampBrainFormatIfNeeded",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "stampEntityTree",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "stampProjectionWatermarks",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "stats",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "storageAdapter",
+ "kind": "accessor"
+ },
+ {
+ "name": "stream",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "streaming",
+ "kind": "accessor"
+ },
+ {
+ "name": "subtypesOf",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "textIdsWithinBelted",
+ "kind": "method",
+ "arity": 2
+ },
+ {
+ "name": "trackField",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "transact",
+ "kind": "method",
+ "arity": 2
+ },
+ {
+ "name": "transactionLog",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "unrelate",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "unvectorNounForRootMigration",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "update",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "updateMany",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "updateRelation",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "upsertMergeParams",
+ "kind": "method",
+ "arity": 2
+ },
+ {
+ "name": "use",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "usesDefaultWasmEmbedder",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "validateIndexConsistency",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "vectorSearchAtGeneration",
+ "kind": "method",
+ "arity": 4
+ },
+ {
+ "name": "verbsToRelations",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "verbToRelationLike",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "verifyEntityTreeStamp",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "verifyGraphAdjacencyLive",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "verifyLogAuthority",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "verifyMetadataLive",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "verifyVectorLive",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "versionedIndexProviders",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "vfs",
+ "kind": "accessor"
+ },
+ {
+ "name": "waitForIndexed",
+ "kind": "method",
+ "arity": 2
+ },
+ {
+ "name": "warm",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "warmupEmbeddings",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "warnIfReadsDegraded",
+ "kind": "method",
+ "arity": 1
+ },
+ {
+ "name": "wireConnectionsCodec",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "wireGraphIdResolver",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "writeEmbedCheckpoint",
+ "kind": "method",
+ "arity": 0
+ },
+ {
+ "name": "writeEmbedLowWater",
+ "kind": "method",
+ "arity": 0
+ }
+ ],
+ "errors": [
+ "BrainyError",
+ "DerivedArtifactMissingError",
+ "GraphIndexNotReadyError",
+ "MetadataArrayTooLargeError",
+ "MetadataIndexNotReadyError",
+ "MigrationInProgressError",
+ "ProtectedArtifactError",
+ "VectorIndexNotReadyError"
+ ],
+ "operators": {
+ "accepted": [
+ "between",
+ "contains",
+ "endsWith",
+ "eq",
+ "equals",
+ "excludes",
+ "exists",
+ "greaterThan",
+ "greaterThanOrEqual",
+ "gt",
+ "gte",
+ "hasAll",
+ "in",
+ "length",
+ "lessThan",
+ "lessThanOrEqual",
+ "lt",
+ "lte",
+ "matches",
+ "missing",
+ "ne",
+ "noneOf",
+ "notEquals",
+ "oneOf",
+ "startsWith"
+ ],
+ "servedOnIndexPath": [
+ "between",
+ "contains",
+ "eq",
+ "equals",
+ "excludes",
+ "exists",
+ "greaterThan",
+ "greaterThanOrEqual",
+ "gt",
+ "gte",
+ "hasAll",
+ "in",
+ "lessThan",
+ "lessThanOrEqual",
+ "lt",
+ "lte",
+ "missing",
+ "ne",
+ "noneOf",
+ "notEquals",
+ "oneOf"
+ ],
+ "refusedByIndexPath": [
+ "endsWith",
+ "length",
+ "matches",
+ "startsWith"
+ ],
+ "combinators": [
+ "allOf",
+ "anyOf",
+ "not"
+ ]
+ },
+ "fieldAddressing": {
+ "systemKeyPrefix": "system.",
+ "systemEntityScalars": [
+ "confidence",
+ "createdAt",
+ "createdBy",
+ "id",
+ "service",
+ "subtype",
+ "type",
+ "updatedAt",
+ "visibility",
+ "weight"
+ ],
+ "systemRelationScalars": [
+ "confidence",
+ "createdAt",
+ "createdBy",
+ "service",
+ "sourceId",
+ "subtype",
+ "targetId",
+ "updatedAt",
+ "verb",
+ "visibility",
+ "weight"
+ ],
+ "plumbingFields": [
+ "_rev",
+ "connections",
+ "data",
+ "level",
+ "vector"
+ ]
+ },
+ "health": {
+ "verdicts": [
+ "pass",
+ "warn",
+ "fail"
+ ],
+ "healKinds": [
+ "none",
+ "repair",
+ "rebuild"
+ ],
+ "servingWithholdingInvariants": [
+ "index-initialized",
+ "durable-state-present",
+ "manifest-residency",
+ "replay-clean",
+ "strand-latch"
+ ]
+ }
+}
diff --git a/docs/api/README.md b/docs/api/README.md
index f5860574..ba49ff48 100644
--- a/docs/api/README.md
+++ b/docs/api/README.md
@@ -24,7 +24,7 @@ next:
## Quick Start
```typescript
-import { Brainy, NounType, VerbType } from '@soulcraft/brainy'
+import { Brainy, NounType, VerbType } from '@soulcraftlabs/brainy'
const brain = new Brainy() // Zero config!
await brain.init() // VFS auto-initialized!
@@ -848,7 +848,6 @@ const db = await brain.transact([
{ op: 'add', id: orderId, type: NounType.Document, subtype: 'order', data: 'Order #1042' },
{ op: 'update', id: customerId, metadata: { lastOrderAt: Date.now() }, ifRev: customer._rev },
{ op: 'relate', from: customerId, to: orderId, type: VerbType.Creates, subtype: 'purchase' },
- { op: 'updateRelation', id: purchaseRelationId, subtype: 'return' },
{ op: 'remove', id: staleDraftId },
{ op: 'unrelate', id: oldRelationId }
], {
@@ -865,7 +864,6 @@ db.receipt.generation // the committed generation
- `{ op: 'update', ... }` — same parameters as `update()`, including per-entity `ifRev` CAS
- `{ op: 'remove', id }` — deletes the entity plus its relationships (same cascade as `delete()`)
- `{ op: 'relate', ... }` — same parameters as `relate()`, including `bidirectional`; duplicates dedupe to the existing relationship id
-- `{ op: 'updateRelation', ... }` — same parameters as `updateRelation()`; a batchable, first-class relationship update (not `unrelate` + `relate` — the relationship id and its edge never change)
- `{ op: 'unrelate', id }` — deletes a relationship by id
Operations may reference ids created earlier in the same batch.
@@ -1012,7 +1010,7 @@ await db.release() // unpin + free cached materialization
### Db API errors
-All exported from `@soulcraft/brainy`:
+All exported from `@soulcraftlabs/brainy`:
| Error | Thrown by | Meaning |
|---|---|---|
@@ -1453,6 +1451,34 @@ const count = await brain.getVerbCount()
---
+### The canonical count ledger (`StorageAdapter.getCanonicalCounts()`)
+
+An OPTIONAL method on the `StorageAdapter` interface (implemented by both
+built-in adapters), not a method on `Brainy` itself — relevant if you're
+writing a custom storage adapter or composing a provider's own
+`healthReport()`. O(1), no I/O. Per family (`nouns`/`verbs`):
+
+```typescript
+interface CanonicalCounts {
+ nouns: { counted: number; all: number }
+ verbs: { counted: number; all: number }
+ suspect: boolean
+}
+```
+
+- `counted` mirrors `getNounCount()` / `getVerbCount()` (public + internal tiers).
+- `all` is the ALL-visibility scalar — every tier, including system/internal
+ records — the denominator a derived index's own coverage math is measured
+ against.
+- `suspect` is `true` when an unprovable delete has left `all` unverified since
+ the last recount; `brain.repairIndex()` clears it with a real canonical walk.
+
+Adapters without the ledger omit the method; treat absence as "no
+denominator," never as zero. See
+**[Index Health](../concepts/index-health.md)** for the full story.
+
+---
+
### Subtype & facet APIs
Full guide: **[Subtypes & Facets](../guides/subtypes-and-facets.md)**.
@@ -1833,6 +1859,104 @@ const semanticOnly = await brain.getStats({ excludeVFS: true })
---
+### `repairIndex(options?)` → `Promise`
+
+The ceremony door for index repair. Bare `repairIndex()` is report-driven: it
+prunes orphaned containers, recomputes count rollups, reconciles VFS
+containment, and rebuilds only a derived-index family whose own health check
+asks for it. Pass `options.rebuild` to force one or more families to rebuild
+UNCONDITIONALLY — no health check is consulted — when an operator has
+independent reason to reconcile a family regardless of what it self-reports.
+
+```typescript
+// Report-driven: only heals what actually needs it
+const report = await brain.repairIndex()
+console.log(report.healedTotal, report.families)
+
+// Explicit: force the graph adjacency to rebuild from canonical, unconditionally
+await brain.repairIndex({ rebuild: ['graph'] })
+
+// Explicit: force all three derived indexes to rebuild
+await brain.repairIndex({ rebuild: 'all' })
+```
+
+**`RepairReport`:**
+- `families: RepairFamilyReport[]` — one row per family checked
+- `healedTotal: number` — items healed across every family
+- `durationMs: number`
+
+**`RepairFamilyReport`** (one row):
+- `family: string` — e.g. `'orphaned-containers'`, `'count-rollups'`,
+ `'vfs-containment'`, `'metadata-corruption'`, `'provider:metadata'`,
+ `'provider:graph'`, `'provider:vector'`
+- `checked: boolean` — was this family actually examined (`false` ⇒ see `skipped`)
+- `healed: number` — items re-posted/corrected in place (the incremental heal count)
+- `missing?: { count: number; sample: string[] }` — exact count plus a capped id
+ sample when the check can name what diverged (never the full list)
+- `rebuilt?: boolean` — a full generational rebuild ran (vs. an incremental heal)
+- `detail?: string` / `reason?: string` — narration
+- `skipped?: string` — why the family wasn't checked
+
+Full walkthrough — what each family checks, degraded-but-serving vs. not-ready,
+and what `suspect` counts mean — in
+**[Index Health](../concepts/index-health.md)**.
+
+---
+
+### Index readiness: typed errors, `healthReport()`, `disableAutoRebuild`
+
+Every derived-index provider (vector, graph, metadata) may expose a named,
+synchronous, O(1) `healthReport()` composed from its own exact ledgers — the
+signal Brainy's read gate trusts over sampling or size heuristics. `init()`
+brings every provider to serving before it returns; there is no first-query
+lazy-rebuild path. A read that reaches a provider whose health report says it
+isn't serving throws instead of rebuilding mid-query:
+
+| Error | Thrown by | Meaning |
+|---|---|---|
+| `GraphIndexNotReadyError` | `find({ connected })`, `neighbors()`, `related()` | Graph adjacency isn't serving |
+| `MetadataIndexNotReadyError` | `find({ where })` | Metadata/field index isn't serving |
+| `VectorIndexNotReadyError` | `find({ query })`, `similar()` | Vector index isn't serving |
+
+All three are exported from `@soulcraftlabs/brainy`. Catch them to distinguish
+"index not ready" from a genuine empty result:
+
+```typescript
+import { MetadataIndexNotReadyError } from '@soulcraftlabs/brainy'
+
+try {
+ const rows = await brain.find({ where: { status: 'active' } })
+} catch (err) {
+ if (err instanceof MetadataIndexNotReadyError) {
+ // reconcile: await brain.repairIndex(), then retry
+ } else {
+ throw err
+ }
+}
+```
+
+**`disableAutoRebuild`** no longer defers index construction to the first
+query. A needed rebuild always runs at `open()`, regardless of this flag or
+dataset size; the flag has no effect on *when* a rebuild runs. Full manual
+control lives in `repairIndex({ rebuild: [...] })`, above.
+
+### `validateIndexConsistency()` → `Promise<...>`
+
+The deep, async diagnostic counterpart to `healthReport()` — safe to run on a
+live brain, but does more work (a provider's `validateInvariants()` may run a
+full scan, not just read a ledger). Aggregates the JS metadata index's own
+consistency check with every derived-index provider's invariant report.
+
+```typescript
+const validation = await brain.validateIndexConsistency()
+if (!validation.healthy) {
+ console.log(validation.recommendation) // what to run, e.g. repairIndex()
+ console.log(validation.providers) // each provider's own invariant report, when exposed
+}
+```
+
+---
+
## Lifecycle
### Initialization
@@ -2084,7 +2208,7 @@ For the full taxonomy with all 169 types and their descriptions, see:
- **📖 Documentation:** [Full Documentation](../)
- **🐛 Issues:** [GitHub Issues](https://github.com/soulcraftlabs/brainy/issues)
- **💬 Discussions:** [GitHub Discussions](https://github.com/soulcraftlabs/brainy/discussions)
-- **📦 NPM:** [@soulcraft/brainy](https://www.npmjs.com/package/@soulcraft/brainy)
+- **📦 NPM:** [@soulcraftlabs/brainy](https://www.npmjs.com/package/@soulcraftlabs/brainy)
- **⭐ GitHub:** [Star us](https://github.com/soulcraftlabs/brainy)
---
diff --git a/docs/architecture/data-storage-architecture.md b/docs/architecture/data-storage-architecture.md
index 48064757..12398747 100644
--- a/docs/architecture/data-storage-architecture.md
+++ b/docs/architecture/data-storage-architecture.md
@@ -217,6 +217,40 @@ membership queries at scale:
`__words__` for tokenized text…).
- `_blobs/_column_index/{field}/L0-NNNNNN.bin` — the actual level-0 run
segments, stored through the shared `_blobs/.bin` binary convention.
+- `_column_index/{field}/k/{kind}/…` — the same two files again, for a
+ **second value kind** on the same field (see below). Absent for a field that
+ holds one kind, which is nearly all of them.
+
+### One posting column per (field, kind)
+
+A field is not obliged to hold one type of value. `category` may carry
+`'electronics'` on some rows and `5` on others, and both are real values of
+that field. A segment, though, has one encoding — i64, f64, UTF-8, or boolean
+— so a field that holds several kinds gets **one column per kind**:
+
+- The first kind a field ever sees owns the plain `_column_index/{field}/`
+ layout above. A single-kind field is therefore byte-identical to what earlier
+ versions wrote, and an index written before typed postings opens unchanged.
+- Every later kind gets its own column beside it at
+ `_column_index/{field}/k/{kind}/`, where `{kind}` is `number`, `string` or
+ `boolean`.
+
+What that buys at query time:
+
+| | |
+|---|---|
+| **Equality** | Answered from the column matching the **query value's own kind**. `where {category: 5}` reads the number postings; `where {category: '5'}` reads the string postings. Neither borrows the other's rows — a row written with the number `5` is not a row whose category is the text `'5'`. |
+| **A kind the field never held** | Matches nothing. That is the true answer, not a coerced one. |
+| **Ranges** | Routed by the kind of the bounds: numeric bounds read the numeric postings and ignore the field's strings. An **unbounded** range is the "has any value here" probe behind `exists`, and reads every kind. |
+| **`orderBy`** | A number and a string have no order between them, so a mixed field orders by kind first (number, string, boolean) and by value within a kind. A single-kind field sorts exactly as it always did. |
+| **Numbers** | One kind, one column: an integer column is written as i64 and widens to f64 the first time a non-integer arrives, so `4.5` is stored as itself rather than rounded. |
+
+`null` and `undefined` are not kinds and are never posted; their absence is
+what the `exists` / `missing` operators read.
+
+Older readers are unaffected by the additional columns: they see the field's
+primary column exactly where it has always been, and a `k/{kind}` directory is
+simply a name they never query.
Sparse per-field indexes, roaring-bitmap chunks, and zone-map/bloom segments
additionally live as bucketed keys under `_system/idx/` (see §3). Which path
@@ -268,7 +302,7 @@ locks/_flush_responses/ # writer answers with .ack
| **Counts/statistics** | Per-type and per-subtype maps | `_system/{type,subtype,verb-subtype}-statistics.json.gz`, `counts.json` | Recomputable by scanning entities (`brainy inspect repair`) |
A pluggable index provider (the 8.0 plugin contract in
-`@soulcraft/brainy/plugin`) may replace any of the JS implementations; the
+`@soulcraftlabs/brainy/plugin`) may replace any of the JS implementations; the
persisted formats above are contract-bound so JS and native implementations
can interleave on the same directory.
diff --git a/docs/architecture/finite-type-system.md b/docs/architecture/finite-type-system.md
index 76492ee5..48a8b1fe 100644
--- a/docs/architecture/finite-type-system.md
+++ b/docs/architecture/finite-type-system.md
@@ -126,7 +126,7 @@ class TypeAwareMetadataIndex {
**The Design**: Specify types clearly in your API calls:
```typescript
-import { Brainy, NounType, VerbType } from '@soulcraft/brainy'
+import { Brainy, NounType, VerbType } from '@soulcraftlabs/brainy'
// Add entity with explicit type
await brain.add({
@@ -231,7 +231,7 @@ class OrgEnrichmentAugmentation {
**Brainy's Approach**: Extract **typed** concepts:
```typescript
-import { NaturalLanguageProcessor } from '@soulcraft/brainy'
+import { NaturalLanguageProcessor } from '@soulcraftlabs/brainy'
const nlp = new NaturalLanguageProcessor()
const concepts = await nlp.extractConcepts("Alice works at Google in San Francisco")
@@ -382,7 +382,7 @@ import {
getVerbTypes,
BrainyTypes,
suggestType
-} from '@soulcraft/brainy'
+} from '@soulcraftlabs/brainy'
// Get all available noun types
const nounTypes = getNounTypes()
diff --git a/docs/architecture/index-architecture.md b/docs/architecture/index-architecture.md
index 8b3dc540..6a754b56 100644
--- a/docs/architecture/index-architecture.md
+++ b/docs/architecture/index-architecture.md
@@ -723,6 +723,14 @@ async stats(): Promise {
### 5. Index Rebuilding (Lazy Loading Support)
+> **Stale as of 10.4 — "Mode 2: Lazy Loading on First Query" below is
+> RETIRED.** `disableAutoRebuild` no longer defers index construction to a
+> first query; `brain.init()` now runs every needed rebuild to completion
+> before it returns, unconditionally, and a read against a not-serving
+> provider throws a typed `*NotReadyError` instead of rebuilding mid-query.
+> See `docs/concepts/index-health.md` for the current contract. Left below
+> as historical background on the rebuild mechanics.
+
**Two modes of index loading:**
#### Mode 1: Auto-Rebuild on init() (default)
diff --git a/docs/architecture/initialization-and-rebuild.md b/docs/architecture/initialization-and-rebuild.md
index a1645744..e19bdd9f 100644
--- a/docs/architecture/initialization-and-rebuild.md
+++ b/docs/architecture/initialization-and-rebuild.md
@@ -1,5 +1,15 @@
# Initialization and Rebuild Processes
+> **Stale as of 10.4 — "Mode 2: Lazy Loading on First Query" below is RETIRED.**
+> `disableAutoRebuild` no longer defers index construction to a first query;
+> `brain.init()` now runs every needed rebuild to completion before it
+> returns, unconditionally. A read against a not-serving provider throws a
+> typed `*NotReadyError` instead of rebuilding mid-query. See
+> `docs/concepts/index-health.md` for the current contract; this document's
+> line-number references to `src/brainy.ts` also predate the file's current
+> size and are unreliable. Left as historical background on the rebuild
+> mechanics, not as a current API description.
+
This document explains how Brainy's four indexes (MetadataIndex, vector index, GraphAdjacencyIndex, DeletedItemsIndex) initialize and rebuild from persisted storage.
## Core Principle: All Indexes Are Disk-Based
diff --git a/docs/architecture/multiprocess-storage-mixin.md b/docs/architecture/multiprocess-storage-mixin.md
index 46f98398..1593bf8f 100644
--- a/docs/architecture/multiprocess-storage-mixin.md
+++ b/docs/architecture/multiprocess-storage-mixin.md
@@ -127,7 +127,7 @@ For reference, a clean migration path:
`isMultiProcessSafe` type-guard. Keep `hasStorageMethod` for
build/install artifact protection.
5. Document the new contract in `concepts/storage-adapters.md`.
-6. Major-version-bump the `@soulcraft/brainy` peerDep range expected by
+6. Major-version-bump the `@soulcraftlabs/brainy` peerDep range expected by
plugins.
Estimated work: ~half a day of code, ~2 hours of doc/example updates,
diff --git a/docs/architecture/noun-verb-taxonomy.md b/docs/architecture/noun-verb-taxonomy.md
index 286464be..3dac6892 100644
--- a/docs/architecture/noun-verb-taxonomy.md
+++ b/docs/architecture/noun-verb-taxonomy.md
@@ -20,7 +20,7 @@ next:
Every example on this page is written against the real Brainy 8.0 API. The setup is always the same:
```typescript
-import { Brainy, NounType, VerbType } from '@soulcraft/brainy'
+import { Brainy, NounType, VerbType } from '@soulcraftlabs/brainy'
const brain = new Brainy()
await brain.init()
@@ -40,7 +40,7 @@ Brainy's **Noun-Verb Taxonomy** achieves broad coverage of human knowledge throu
- **Multi-hop Graph Traversals = Relationship Complexity**
- **Result: Model data across virtually any industry**
-Every piece of information can be represented as entities (nouns) connected by relationships (verbs) carrying properties (metadata). The standardized type system from `@soulcraft/brainy` (`NounType`, `VerbType`) gives those nouns and verbs a stable, shared name.
+Every piece of information can be represented as entities (nouns) connected by relationships (verbs) carrying properties (metadata). The standardized type system from `@soulcraftlabs/brainy` (`NounType`, `VerbType`) gives those nouns and verbs a stable, shared name.
## The Power of Standardization: Universal Interoperability
diff --git a/docs/architecture/zero-config.md b/docs/architecture/zero-config.md
index d42d6784..a35e6416 100644
--- a/docs/architecture/zero-config.md
+++ b/docs/architecture/zero-config.md
@@ -35,7 +35,7 @@ constructor and `init()`.
## Instant Start
```typescript
-import { Brainy } from '@soulcraft/brainy'
+import { Brainy } from '@soulcraftlabs/brainy'
// That's it. No config needed.
const brain = new Brainy()
diff --git a/docs/concepts/field-addressing.md b/docs/concepts/field-addressing.md
index c459021b..d24dd66b 100644
--- a/docs/concepts/field-addressing.md
+++ b/docs/concepts/field-addressing.md
@@ -167,7 +167,7 @@ await brain.find({ orderBy: 'createdAt' })
`UnresolvableFieldError` is exported from the package root:
```typescript
-import { UnresolvableFieldError } from '@soulcraft/brainy'
+import { UnresolvableFieldError } from '@soulcraftlabs/brainy'
try {
await brain.find({ orderBy: 'createdAt' })
diff --git a/docs/concepts/index-health.md b/docs/concepts/index-health.md
new file mode 100644
index 00000000..923267df
--- /dev/null
+++ b/docs/concepts/index-health.md
@@ -0,0 +1,217 @@
+---
+title: Index Health
+slug: concepts/index-health
+public: true
+category: concepts
+template: concept
+order: 8
+description: How Brainy knows whether a derived index can be trusted — exact accounting instead of sampling, the named health report, degraded-but-serving vs. not-ready, and what repairIndex() checks, heals, and rebuilds.
+next:
+ - concepts/generation-fact-log
+ - guides/inspection
+---
+
+# Index Health
+
+Brainy keeps one **canonical** copy of every entity and relationship, and three
+**derived** indexes built from it — vector, metadata, and graph — so `find()` can
+answer semantically, by filter, and by traversal without re-deriving the answer from
+scratch on every query. A derived index is a cache with a serving structure: it can
+be present but stale, present but only partially loaded, or fully out of sync with
+canonical after a crash. This page is about how Brainy decides whether to trust one,
+what it does when it can't, and how you reconcile the two.
+
+## Exact accounting instead of sampling
+
+Older health checks worked by inference: does `size()` return something greater
+than zero, does a spot-check on one known item come back correct. Both are proxies.
+A cold index can report a nonzero count while its actual serving structure never
+loaded, and a spot-check only proves the one item it happened to ask about.
+
+Every derived-index provider may now expose a named, synchronous, O(1)
+`healthReport()` — composed from the provider's own **exact ledgers** (real counters
+it already maintains on the write path), never a sample or a walk. This is the one
+signal Brainy's read gate consults. A provider that doesn't yet expose one falls
+back to an honest `isReady()` boolean, and finally to a size heuristic for engines
+with neither — but wherever a `healthReport()` exists, it wins.
+
+Underneath, storage itself keeps an analogous **canonical count ledger**: a
+`counted` scalar (the user-facing total — what `getNounCount()` / `getVerbCount()`
+return) and an `all` scalar (every tier, including internal records a derived
+index's own coverage math needs to compare against). This is the real denominator
+a provider's `healthReport()` measures itself by, rather than a total that can only
+ever ratchet upward. See [What `suspect` counts mean](#what-suspect-counts-mean)
+below for the one case that ledger can't stay exact through on its own.
+
+## The named report
+
+A `HealthReport` carries, per provider (`'vector'` / `'graph'` / `'metadata'`):
+
+- **`healthy`** — `true` iff every *verified* invariant holds. An invariant whose
+ family has no ledger yet is `unledgered`, never counted either way — unknown,
+ not passing.
+- **`serving`** — can this provider answer a query right now. A failing invariant
+ graded `heal: 'repair'` or `heal: 'none'` still leaves `serving: true` — this is
+ **degraded-but-serving**: something is off (say, a stale rollup on an
+ `employee` record's relationship count) but reads keep working. Only a failure
+ graded `heal: 'rebuild'` flips `serving` to `false` — **not-ready** — because the
+ provider itself is telling you its serving structure cannot answer correctly.
+- **`invariants`** — each checked condition, with its provenance
+ (`source: 'ledger'` — an exact count; `'deep'` — a full scan, diagnostic-only;
+ `'unledgered'` — not yet tracked) and, for a failing one, an exact `missing`
+ count plus a capped sample of the affected ids — a verdict, never a dump.
+- **`generation`** — bumps on every ledger mutation and rebuild, so a caller can
+ cache a verdict per generation instead of re-deriving it.
+
+The distinction that matters day to day: `healthy: false` can be entirely benign —
+a maintenance window, a divergence `repairIndex()` will clean up on its own
+schedule. `serving: false` is not benign. It means this provider is refusing to
+answer, on its own word, right now.
+
+**How a failure gets its grade — the serving law.** A provider grades `heal` by
+one question only: *could an answer be wrong?* — never *how expensive is the
+fix?* A missing-postings shortfall, however large, is `heal: 'repair'` (re-post
+exactly what the ledger names, reads serving throughout); it can never withhold
+serving just because healing it takes work. `serving` is withheld only by a
+small, named set of rebuild-graded conditions — the index not initialized, its
+durable state absent, a manifest naming files that are not resident, a replay
+that did not complete cleanly — the states in which an answer could genuinely be
+wrong. And a read is only ever refused by the family it actually consults: a
+metadata filter is answered by the metadata index alone, vector search by the
+vector index, traversal by the graph index — one family's refusal never blocks
+another family's reads.
+
+## Reads refuse — they never rebuild
+
+A query that reaches a not-serving provider does not trigger a rebuild from inside
+the read. Brainy retired that path deliberately: a rebuild kicked off by an ordinary
+`find({ where: { status: 'active' } })` call is a dark, unpredictable cost hiding
+behind a request that looks like a cheap read. Instead, the read throws a typed,
+catchable error naming the reason:
+
+| Error | Thrown when | Meaning |
+|---|---|---|
+| `GraphIndexNotReadyError` | `find({ connected })`, `neighbors()`, `related()` | The graph adjacency index isn't serving — traversal would otherwise return `[]` indistinguishable from "no relationships" |
+| `MetadataIndexNotReadyError` | `find({ where })` | The metadata/field index isn't serving — a filtered read would otherwise return `[]` indistinguishable from "no matches" |
+| `VectorIndexNotReadyError` | `find({ query })`, `similar()` | The vector index isn't serving — a semantic search would otherwise return `[]` indistinguishable from "nothing similar" |
+
+All three are exported from `@soulcraftlabs/brainy`. Catch them where your application
+needs to distinguish "this index isn't ready yet" from "there's genuinely nothing
+here" — a health dashboard, a retry policy, an operator alert. The fix is always
+the same: reconcile the index, either by reopening the brain (which brings every
+provider to serving before `init()` returns — see the next section) or by calling
+`repairIndex()` explicitly.
+
+```typescript
+try {
+ const active = await brain.find({ where: { status: 'active' } })
+} catch (err) {
+ if (err instanceof MetadataIndexNotReadyError) {
+ // not a "no results" — the index itself refused; alert or retry after repair
+ } else {
+ throw err
+ }
+}
+```
+
+### Rebuilds happen at open, not on first query
+
+`brain.init()` runs every needed rebuild to completion **before it returns**,
+unconditionally, regardless of dataset size. There is no lazy, first-query
+rebuild path anymore — a brain either finishes opening healthy, or it fails
+open loudly. `disableAutoRebuild: true` no longer defers index construction to
+the first query: it has no effect on *when* a needed rebuild runs. Full manual
+control over rebuilds is `repairIndex({ rebuild: [...] })` (below), not this flag.
+
+## `repairIndex()` — checking and healing
+
+Bare `repairIndex()` is **report-driven**: it only heals what its own checks say
+actually needs it, and it always returns a full per-family receipt.
+
+```typescript
+const report = await brain.repairIndex()
+report.healedTotal // total items healed across every family
+report.durationMs
+report.families // one row per family checked
+```
+
+Each `RepairFamilyReport` row names what happened:
+
+- **`checked`** — was this family actually examined (`false` means skipped —
+ see `skipped` for why).
+- **`healed`** — items re-posted or corrected in place.
+- **`missing`** — when the check can name what diverged: an exact `count` plus a
+ capped `sample` of ids.
+- **`rebuilt`** — a full generational rebuild ran (as opposed to an incremental
+ heal).
+- **`detail`** / **`reason`** / **`skipped`** — the receipt's narration; a row is
+ always either checked or explains why it wasn't. Nothing is silent.
+
+On every call, bare `repairIndex()`:
+
+1. Prunes orphaned canonical containers left by a partial delete.
+2. Recomputes the count rollups from one canonical walk (unconditional — this is
+ also what clears a `suspect` ledger; see below).
+3. Reconciles VFS containment edges, if the VFS is initialized.
+4. Runs the metadata index's own corruption detection pass.
+5. Consults each of the three derived-index providers' own health check and
+ rebuilds only a family whose failing invariant actually asks for it
+ (`heal: 'rebuild'`) — never a provider that reports `healthy` or a lesser
+ grade.
+
+### The explicit rebuild door
+
+`options.rebuild` skips the health check and rebuilds one or more families
+**unconditionally** — the operator override for when you have independent reason
+to distrust a family regardless of what it self-reports (a suspicious deploy, a
+storage-layer incident, a support ticket that doesn't match what the health report
+says):
+
+```typescript
+// Force the graph adjacency to rebuild from canonical, no invariant consulted
+await brain.repairIndex({ rebuild: ['graph'] })
+
+// Force all three derived indexes
+await brain.repairIndex({ rebuild: 'all' })
+```
+
+A family named this way is recorded with `rebuilt: true` and
+`reason: 'explicit rebuild requested'`, and is skipped by the normal
+health-driven pass in the same call — it was already rebuilt unconditionally.
+
+Reach for the explicit door when you need certainty regardless of self-report;
+reach for bare `repairIndex()` for routine maintenance and after any incident
+where you're not sure which family (if any) needs it.
+
+## What `suspect` counts mean
+
+Storage's canonical count ledger increments the ALL-visibility total on every new
+record and decrements it on every *proven* delete — one where the record was read,
+or the caller supplied its prior image. A delete that cannot prove what it removed
+existed doesn't guess: it flags the ledger `suspect` (an operator-visible
+`console.warn`, narrated once per session, not once per delete) rather than risk
+decrementing a total that was never incremented for that record in the first
+place. This is intentionally rare — it's a defensive fallback for callers on an
+unusual removal path, not a per-delete cost.
+
+`suspect` is not directly exposed on any `Brainy` method today — it lives on the
+`StorageAdapter`'s optional `getCanonicalCounts()`, primarily consulted by
+`repairIndex()`'s recount step and by custom storage adapters composing their own
+`healthReport()`. What matters for an application: a `suspect` ledger is not
+incorrect, just *unverified since the last recount* — and `repairIndex()`'s
+unconditional count-rollup step (step 2, above) recomputes the ALL scalars from a
+real canonical walk on every call, clearing the flag with proof either way.
+
+## Practical guidance
+
+- **On a normal restart**, do nothing — `init()` brings every provider to
+ serving before it returns, or fails loudly.
+- **On a `*NotReadyError`** from a live read, reconcile with `repairIndex()`
+ (report-driven is almost always sufficient) and retry.
+- **After an incident** where you distrust a specific family regardless of what
+ it reports healthy — a storage-layer fault, a suspicious restore — use the
+ explicit door: `repairIndex({ rebuild: ['metadata' | 'graph' | 'vector'] })`.
+- **To audit before trusting a report**, `brain.auditGraph()` walks every stored
+ relationship and proves (or disproves) that reads return canonical truth,
+ independent of what any provider self-reports — see
+ [Inspecting a Live Brainy](../guides/inspection.md).
diff --git a/docs/concepts/multi-process.md b/docs/concepts/multi-process.md
index 8fda315f..d698eee8 100644
--- a/docs/concepts/multi-process.md
+++ b/docs/concepts/multi-process.md
@@ -95,8 +95,15 @@ The heartbeat interval rewrites the lock file every 10 seconds. The timer
is unref'd, so it does not keep the event loop alive on its own.
On normal shutdown the writer releases the lock in `close()`. The shutdown
-hooks Brainy registers for `SIGTERM`, `SIGINT`, and `beforeExit` also
-release the lock so a container restart doesn't strand the directory.
+hooks Brainy registers for `SIGTERM` and `SIGINT` close every live brain by
+that same `close()`, so a container restart doesn't strand the directory.
+
+`beforeExit` is not one of them. Node emits it whenever the event loop has
+no ref'd work left — a state a healthy script reaches routinely, because
+Brainy's own idle and cadence timers are unref'd — and a drained event loop
+is not a shutdown. That hook only persists derived state with a non-closing
+`flush()`: it closes nothing, releases no lock, and leaves every brain open
+and usable. If you want a shutdown, call `close()` or send `SIGTERM`.
## How to inspect a live writer
diff --git a/docs/concepts/storage-adapters.md b/docs/concepts/storage-adapters.md
index af6d068f..82aa01e8 100644
--- a/docs/concepts/storage-adapters.md
+++ b/docs/concepts/storage-adapters.md
@@ -61,7 +61,7 @@ The only required override is the capability flag. Returning `true` from
to call `acquireWriterLock()` at init.
```typescript
-import { FileSystemStorage } from '@soulcraft/brainy'
+import { FileSystemStorage } from '@soulcraftlabs/brainy'
export class MmapFileSystemStorage extends FileSystemStorage {
public supportsMultiProcessLocking(): boolean {
@@ -79,7 +79,7 @@ If your storage is **not filesystem-backed** (a custom
network backend), extend `BaseStorage` directly:
```typescript
-import { BaseStorage } from '@soulcraft/brainy'
+import { BaseStorage } from '@soulcraftlabs/brainy'
export class MyCloudStorage extends BaseStorage {
// BaseStorage's default no-op implementations of the multi-process
@@ -101,7 +101,7 @@ The defensive check at every new-storage-method call site (`brainy.ts`,
`hasStorageMethod(name)`) does **not** exist to handle "plugin bundles a
stale BaseStorage." Plugins ship a dist that preserves the dynamic ESM
import (verify in your plugin's `dist/`: `import { FileSystemStorage } from
-'@soulcraft/brainy'` is not rewritten to a vendored copy). The prototype
+'@soulcraftlabs/brainy'` is not rewritten to a vendored copy). The prototype
chain at runtime resolves to whatever Brainy version your consumer has
installed.
@@ -109,8 +109,8 @@ installed.
the prototype chain at the consumer-app level:
- **Stale `node_modules`** — a lingering install from before the consumer
- upgraded Brainy. The package.json says `@soulcraft/brainy@7.22.0` but
- `node_modules/@soulcraft/brainy` is still 7.20.x.
+ upgraded Brainy. The package.json says `@soulcraftlabs/brainy@7.22.0` but
+ `node_modules/@soulcraftlabs/brainy` is still 7.20.x.
- **Lockfile drift** — `bun.lockb` / `package-lock.json` pins a brainy
version older than the package.json range, and `bun install` honors the
lockfile.
@@ -131,7 +131,7 @@ and the warning names the adapter class plus a remediation hint:
methods on its prototype chain. Writer locking and the flush-request RPC are
disabled for this directory. Likely fix: clean install (`rm -rf node_modules
bun.lockb && bun install`) or rebuild your container image to refresh
-`@soulcraft/brainy` to ≥7.21. See docs/concepts/storage-adapters.md.
+`@soulcraftlabs/brainy` to ≥7.21. See docs/concepts/storage-adapters.md.
```
## Authoring a new storage adapter — minimum checklist
@@ -168,7 +168,7 @@ bun.lockb && bun install`) or rebuild your container image to refresh
install time — fix install, not your plugin.
6. **Pin your peer dep generously.** `"peerDependencies": {
- "@soulcraft/brainy": "^7.21.0" }` accepts any compatible 7.x. Don't pin
+ "@soulcraftlabs/brainy": "^7.21.0" }` accepts any compatible 7.x. Don't pin
to an exact patch unless you're tracking a known regression.
## Future direction
@@ -185,5 +185,5 @@ follow-up; consumers don't need to anticipate the change.
heartbeat semantics, what the lock protects.
- [`guides/inspection`](../guides/inspection.md) — `brainy inspect` and the
read-only mode.
-- `node_modules/@soulcraft/brainy/dist/storage/baseStorage.d.ts` — the
+- `node_modules/@soulcraftlabs/brainy/dist/storage/baseStorage.d.ts` — the
authoritative type signatures for every method this page references.
diff --git a/docs/guides/aggregation.md b/docs/guides/aggregation.md
index 11d86ec8..616c8fc4 100644
--- a/docs/guides/aggregation.md
+++ b/docs/guides/aggregation.md
@@ -22,7 +22,7 @@ they share a single scan.
## Quick Start
```typescript
-import { Brainy, NounType } from '@soulcraft/brainy'
+import { Brainy, NounType } from '@soulcraftlabs/brainy'
const brain = new Brainy()
await brain.init()
diff --git a/docs/guides/framework-integration.md b/docs/guides/framework-integration.md
index 984466c5..8f85da00 100644
--- a/docs/guides/framework-integration.md
+++ b/docs/guides/framework-integration.md
@@ -8,7 +8,7 @@ Brainy is **framework-friendly** - designed to drop into the server side of any
Brainy embeds an HNSW vector index, a graph engine, and a filesystem-backed persistence layer. These belong on the server:
-- **Zero configuration**: Just `import { Brainy } from '@soulcraft/brainy'`
+- **Zero configuration**: Just `import { Brainy } from '@soulcraftlabs/brainy'`
- **Auto storage detection**: `new Brainy()` auto-selects filesystem persistence on Node
- **Cleaner code**: No browser polyfills, no conditional client/server imports
- **Better DX**: One instance shared across your server routes
@@ -18,13 +18,13 @@ Brainy embeds an HNSW vector index, a graph engine, and a filesystem-backed pers
### Install Brainy
```bash
-npm install @soulcraft/brainy
+npm install @soulcraftlabs/brainy
```
### Basic Integration
```javascript
-import { Brainy } from '@soulcraft/brainy'
+import { Brainy } from '@soulcraftlabs/brainy'
// Run on the server (API route, server component, backend service)
// new Brainy() auto-detects filesystem persistence on Node
@@ -105,7 +105,7 @@ On the server, create one Brainy instance and reuse it across requests. This mod
```javascript
// lib/brain.server.js
-import { Brainy } from '@soulcraft/brainy'
+import { Brainy } from '@soulcraftlabs/brainy'
let brainPromise
@@ -163,7 +163,7 @@ On the server, create one Brainy instance and reuse it across requests:
```javascript
// server/brain.js (server-only module)
-import { Brainy } from '@soulcraft/brainy'
+import { Brainy } from '@soulcraftlabs/brainy'
let brainPromise
@@ -248,7 +248,7 @@ The matching backend endpoint uses Brainy directly (Node/Bun):
```typescript
// server: api/search
-import { Brainy } from '@soulcraft/brainy'
+import { Brainy } from '@soulcraftlabs/brainy'
const brain = new Brainy() // auto-detects filesystem persistence on Node
await brain.init()
@@ -266,7 +266,7 @@ In Next.js, Brainy lives in server code only: API routes, server components, or
```javascript
// lib/brain.server.js (imported only by server code)
-import { Brainy } from '@soulcraft/brainy'
+import { Brainy } from '@soulcraftlabs/brainy'
let brainPromise
@@ -318,7 +318,7 @@ Brainy runs in a server-only module (`*.server.js`); the component fetches resul
```javascript
// src/lib/server/brain.js (server-only — note the .server suffix)
-import { Brainy } from '@soulcraft/brainy'
+import { Brainy } from '@soulcraftlabs/brainy'
let brainPromise
@@ -432,7 +432,7 @@ import { defineConfig } from 'vite'
export default defineConfig({
ssr: {
- external: ['@soulcraft/brainy']
+ external: ['@soulcraftlabs/brainy']
}
})
```
@@ -440,7 +440,7 @@ export default defineConfig({
```javascript
// rollup.config.js (server bundle)
export default {
- external: ['@soulcraft/brainy', 'node:fs', 'node:path', 'node:crypto']
+ external: ['@soulcraftlabs/brainy', 'node:fs', 'node:path', 'node:crypto']
}
```
@@ -466,7 +466,7 @@ export async function load({ url }) {
```javascript
// For build-time usage (runs in Node during the build)
-import { Brainy } from '@soulcraft/brainy'
+import { Brainy } from '@soulcraftlabs/brainy'
export async function generateStaticProps() {
const brain = new Brainy({
@@ -513,7 +513,7 @@ export async function generateStaticProps() {
### Issue: Large client bundle size
**Cause**: A client module is pulling in Brainy.
-**Solution**: Move the `import { Brainy } from '@soulcraft/brainy'` into a server-only module so it never reaches the browser bundle.
+**Solution**: Move the `import { Brainy } from '@soulcraftlabs/brainy'` into a server-only module so it never reaches the browser bundle.
### Issue: SSR hydration mismatch
**Solution**: Run the search on the server (loader / server action / API route) and pass the results down as props, so server and client render the same markup.
diff --git a/docs/guides/import-anything.md b/docs/guides/import-anything.md
index b1bb15ef..ffabe55c 100644
--- a/docs/guides/import-anything.md
+++ b/docs/guides/import-anything.md
@@ -9,7 +9,7 @@ Brainy's import is **ONE magical method** that understands EVERYTHING:
## The Ultimate Simplicity
```javascript
-import { Brainy } from '@soulcraft/brainy'
+import { Brainy } from '@soulcraftlabs/brainy'
const brain = new Brainy()
await brain.init()
diff --git a/docs/guides/import-progress-examples.md b/docs/guides/import-progress-examples.md
index 66f50713..18c3cb9a 100644
--- a/docs/guides/import-progress-examples.md
+++ b/docs/guides/import-progress-examples.md
@@ -13,7 +13,7 @@ Brainy provides real-time progress tracking for **all 7 supported file formats**
### Basic Progress Tracking
```typescript
-import { Brainy } from '@soulcraft/brainy'
+import { Brainy } from '@soulcraftlabs/brainy'
import * as fs from 'fs'
const brain = await Brainy.create()
diff --git a/docs/guides/import-quick-reference.md b/docs/guides/import-quick-reference.md
index 7837d49e..3bc26dae 100644
--- a/docs/guides/import-quick-reference.md
+++ b/docs/guides/import-quick-reference.md
@@ -7,7 +7,7 @@
## Basic Import
```typescript
-import { Brainy } from '@soulcraft/brainy'
+import { Brainy } from '@soulcraftlabs/brainy'
const brain = new Brainy()
await brain.init()
@@ -187,7 +187,7 @@ await brain.import(file, {
## Complete Example
```typescript
-import { Brainy } from '@soulcraft/brainy'
+import { Brainy } from '@soulcraftlabs/brainy'
import * as fs from 'fs'
async function importCatalog() {
diff --git a/docs/guides/inspection.md b/docs/guides/inspection.md
index 240e81ae..8560b543 100644
--- a/docs/guides/inspection.md
+++ b/docs/guides/inspection.md
@@ -108,7 +108,7 @@ check fails — useful for piping into monitoring or CI.
## Programmatic inspection
```typescript
-import { Brainy } from '@soulcraft/brainy'
+import { Brainy } from '@soulcraftlabs/brainy'
const reader = await Brainy.openReadOnly({
storage: { type: 'filesystem', path: '/data/brain' }
diff --git a/docs/guides/installation.md b/docs/guides/installation.md
index 0a36f632..20d40ea2 100644
--- a/docs/guides/installation.md
+++ b/docs/guides/installation.md
@@ -21,21 +21,21 @@ next:
## Install
```bash
-npm install @soulcraft/brainy
+npm install @soulcraftlabs/brainy
```
Or with your preferred package manager:
```bash
-bun add @soulcraft/brainy
-yarn add @soulcraft/brainy
-pnpm add @soulcraft/brainy
+bun add @soulcraftlabs/brainy
+yarn add @soulcraftlabs/brainy
+pnpm add @soulcraftlabs/brainy
```
## Verify
```typescript
-import { Brainy } from '@soulcraft/brainy'
+import { Brainy } from '@soulcraftlabs/brainy'
const brain = new Brainy()
await brain.init()
@@ -52,7 +52,7 @@ npm install @soulcraft/cor
```
```typescript
-import { Brainy } from '@soulcraft/brainy'
+import { Brainy } from '@soulcraftlabs/brainy'
const brain = new Brainy({ plugins: ['@soulcraft/cor'] })
await brain.init() // native providers registered during init
@@ -71,7 +71,7 @@ remains available on npm if you need it.
Brainy ships with full TypeScript types. No `@types/` package needed:
```typescript
-import { Brainy, NounType, VerbType } from '@soulcraft/brainy'
+import { Brainy, NounType, VerbType } from '@soulcraftlabs/brainy'
const brain = new Brainy()
await brain.init()
diff --git a/docs/guides/migration-3.36.0.md b/docs/guides/migration-3.36.0.md
index 8b1f239e..5f00534a 100644
--- a/docs/guides/migration-3.36.0.md
+++ b/docs/guides/migration-3.36.0.md
@@ -66,7 +66,7 @@ const results = await brain.search("query")
**New diagnostics for capacity planning and performance tuning.**
```typescript
-import { Brainy } from '@soulcraft/brainy'
+import { Brainy } from '@soulcraftlabs/brainy'
const brain = new Brainy()
await brain.init()
@@ -112,7 +112,7 @@ Recommendations: ${stats.recommendations.join(', ')}
### Step 1: Update Package
```bash
-npm install @soulcraft/brainy@latest
+npm install @soulcraftlabs/brainy@latest
```
### Step 2: Restart Your Application
@@ -134,7 +134,7 @@ npm run start
### Check Adaptive Sizing is Working
```typescript
-import { Brainy } from '@soulcraft/brainy'
+import { Brainy } from '@soulcraftlabs/brainy'
const brain = new Brainy()
await brain.init()
@@ -218,7 +218,7 @@ For debugging or compatibility testing:
If you need to rollback to v3.35.0:
```bash
-npm install @soulcraft/brainy@3.35.0
+npm install @soulcraftlabs/brainy@3.35.0
```
**Note:** We don't anticipate any issues, but rollback is straightforward if needed.
@@ -367,7 +367,7 @@ if (stats.fairness.fairnessViolation) {
## Next Steps
-1. ✅ **Upgrade:** `npm install @soulcraft/brainy@latest`
+1. ✅ **Upgrade:** `npm install @soulcraftlabs/brainy@latest`
2. 📊 **Monitor:** Use `getCacheStats()` to verify performance improvements
3. 🎯 **Tune:** Adjust based on recommendations (if needed)
4. 📖 **Read:** [Operations Guide](../operations/capacity-planning.md) for capacity planning
diff --git a/docs/guides/model-loading.md b/docs/guides/model-loading.md
index e5b7b1d6..cc1b2b6a 100644
--- a/docs/guides/model-loading.md
+++ b/docs/guides/model-loading.md
@@ -37,7 +37,7 @@ This single WASM file contains everything needed for sentence embeddings.
```bash
# Bun as a runtime — supported and recommended
-bun add @soulcraft/brainy
+bun add @soulcraftlabs/brainy
bun run server.ts
```
diff --git a/docs/guides/namespace-migration.md b/docs/guides/namespace-migration.md
index fad3c766..f7d2c7f7 100644
--- a/docs/guides/namespace-migration.md
+++ b/docs/guides/namespace-migration.md
@@ -80,7 +80,7 @@ If you read raw stored records (fact-log scanners, export tooling), use
the exported shape-aware splitters — they handle both record eras:
```typescript
-import { splitNounMetadataRecord } from '@soulcraft/brainy'
+import { splitNounMetadataRecord } from '@soulcraftlabs/brainy'
const { reserved, custom } = splitNounMetadataRecord(rawRecord)
// reserved = engine fields · custom = the user's bag, ANY names
```
@@ -88,7 +88,7 @@ const { reserved, custom } = splitNounMetadataRecord(rawRecord)
Feature detection (never version-sniff):
```typescript
-import * as brainy from '@soulcraft/brainy'
+import * as brainy from '@soulcraftlabs/brainy'
const lawActive = 'FIELD_ADDRESSING_CAPABILITY' in brainy // 'field-addressing/v1'
```
diff --git a/docs/guides/nextjs-integration.md b/docs/guides/nextjs-integration.md
index ab55e51f..25d6062d 100644
--- a/docs/guides/nextjs-integration.md
+++ b/docs/guides/nextjs-integration.md
@@ -9,7 +9,7 @@ Complete guide to integrating Brainy with Next.js applications, covering App Rou
```bash
npx create-next-app@latest my-brainy-app
cd my-brainy-app
-npm install @soulcraft/brainy
+npm install @soulcraftlabs/brainy
```
### Basic Setup
@@ -18,7 +18,7 @@ npm install @soulcraft/brainy
// app/components/BrainyProvider.jsx
'use client'
import { createContext, useContext, useEffect, useState } from 'react'
-import { Brainy } from '@soulcraft/brainy'
+import { Brainy } from '@soulcraftlabs/brainy'
const BrainyContext = createContext()
@@ -271,7 +271,7 @@ export default function SearchPage() {
```javascript
// app/api/search/route.js (App Router)
-import { Brainy } from '@soulcraft/brainy'
+import { Brainy } from '@soulcraftlabs/brainy'
let brain = null
@@ -332,7 +332,7 @@ export async function GET() {
```javascript
// pages/api/search.js (Pages Router)
-import { Brainy } from '@soulcraft/brainy'
+import { Brainy } from '@soulcraftlabs/brainy'
let brain = null
@@ -374,7 +374,7 @@ export default async function handler(req, res) {
```javascript
// app/api/data/route.js
-import { Brainy } from '@soulcraft/brainy'
+import { Brainy } from '@soulcraftlabs/brainy'
let brain = null
@@ -418,7 +418,7 @@ export async function POST(request) {
```jsx
// app/actions/brainy.js
'use server'
-import { Brainy } from '@soulcraft/brainy'
+import { Brainy } from '@soulcraftlabs/brainy'
let brain = null
@@ -630,7 +630,7 @@ CMD ["npm", "start"]
/** @type {import('next').NextConfig} */
const nextConfig = {
experimental: {
- serverComponentsExternalPackages: ['@soulcraft/brainy']
+ serverComponentsExternalPackages: ['@soulcraftlabs/brainy']
},
webpack: (config, { isServer }) => {
if (!isServer) {
@@ -797,7 +797,7 @@ export function rateLimit(req, limit = 100, window = 60000) {
// app/contexts/BrainyContext.jsx
'use client'
import { createContext, useContext, useReducer, useEffect } from 'react'
-import { Brainy } from '@soulcraft/brainy'
+import { Brainy } from '@soulcraftlabs/brainy'
const BrainyContext = createContext()
@@ -873,7 +873,7 @@ import { BrainyProvider } from '../app/components/BrainyProvider'
import { Search } from '../app/components/Search'
// Mock Brainy
-jest.mock('@soulcraft/brainy', () => ({
+jest.mock('@soulcraftlabs/brainy', () => ({
Brainy: jest.fn().mockImplementation(() => ({
init: jest.fn().mockResolvedValue(undefined),
find: jest.fn().mockResolvedValue([
diff --git a/docs/guides/optimistic-concurrency.md b/docs/guides/optimistic-concurrency.md
index 268bc5fa..2984998b 100644
--- a/docs/guides/optimistic-concurrency.md
+++ b/docs/guides/optimistic-concurrency.md
@@ -32,7 +32,7 @@ Brainy 7.31.0 adds a per-entity revision counter so multiple writers can coordin
Every distributed-job scheduler eventually wants this exact loop:
```ts
-import { Brainy, RevisionConflictError } from '@soulcraft/brainy'
+import { Brainy, RevisionConflictError } from '@soulcraftlabs/brainy'
const LOCK_ID = '...uuid for this job slot...'
@@ -137,7 +137,7 @@ await brain.addIfMissing({ // ← not a real API
It's race-prone as a plain read-then-write: two concurrent imports both see "not found," both insert, you get duplicates. Without a unique-index primitive (which Brainy doesn't have today), close the race with whole-store CAS — read at a pinned generation, then commit only if nothing moved:
```ts
-import { GenerationConflictError } from '@soulcraft/brainy'
+import { GenerationConflictError } from '@soulcraftlabs/brainy'
async function addIfMissingByEmail(email: string, data: string) {
for (let attempt = 0; attempt < 5; attempt++) {
diff --git a/docs/guides/quick-start.md b/docs/guides/quick-start.md
index 097c55fe..d9a4e896 100644
--- a/docs/guides/quick-start.md
+++ b/docs/guides/quick-start.md
@@ -18,13 +18,13 @@ Get Brainy running in under a minute.
## 1. Install
```bash
-npm install @soulcraft/brainy
+npm install @soulcraftlabs/brainy
```
## 2. Initialize
```typescript
-import { Brainy, NounType, VerbType } from '@soulcraft/brainy'
+import { Brainy, NounType, VerbType } from '@soulcraftlabs/brainy'
const brain = new Brainy()
await brain.init()
@@ -67,7 +67,7 @@ await brain.relate({
## 5. Query with Triple Intelligence
```typescript
-import type { Result } from '@soulcraft/brainy'
+import type { Result } from '@soulcraftlabs/brainy'
// All three search paradigms in one call
const results: Result[] = await brain.find({
diff --git a/docs/guides/standard-import-progress.md b/docs/guides/standard-import-progress.md
index 9f2e2e5b..27dabe75 100644
--- a/docs/guides/standard-import-progress.md
+++ b/docs/guides/standard-import-progress.md
@@ -11,7 +11,7 @@
### One Interface for Everything
```typescript
-import { Brainy } from '@soulcraft/brainy'
+import { Brainy } from '@soulcraftlabs/brainy'
const brain = await Brainy.create()
@@ -78,7 +78,7 @@ interface ImportProgress {
```typescript
import { useState } from 'react'
-import { Brainy } from '@soulcraft/brainy'
+import { Brainy } from '@soulcraftlabs/brainy'
function UniversalImportProgress({ file }: { file: File }) {
const [progress, setProgress] = useState({
@@ -177,7 +177,7 @@ function UniversalImportProgress({ file }: { file: File }) {
```typescript
import ora from 'ora'
-import { Brainy } from '@soulcraft/brainy'
+import { Brainy } from '@soulcraftlabs/brainy'
async function importWithProgress(filePath: string) {
const spinner = ora('Starting import...').start()
diff --git a/docs/guides/storage-adapters.md b/docs/guides/storage-adapters.md
index 06ec9f3a..a4224bc8 100644
--- a/docs/guides/storage-adapters.md
+++ b/docs/guides/storage-adapters.md
@@ -28,7 +28,7 @@ on-disk layout (memory's "disk" is a JS Map).
## Quick start
```ts
-import { Brainy } from '@soulcraft/brainy'
+import { Brainy } from '@soulcraftlabs/brainy'
// Filesystem (recommended for any persistent workload):
const brain = new Brainy({
@@ -134,7 +134,7 @@ config; the `type` is optional.
If you want to skip the factory:
```ts
-import { FileSystemStorage, MemoryStorage } from '@soulcraft/brainy'
+import { FileSystemStorage, MemoryStorage } from '@soulcraftlabs/brainy'
const fsStorage = new FileSystemStorage('./brainy-data')
const memStorage = new MemoryStorage()
diff --git a/docs/guides/subtypes-and-facets.md b/docs/guides/subtypes-and-facets.md
index ff5de320..74311528 100644
--- a/docs/guides/subtypes-and-facets.md
+++ b/docs/guides/subtypes-and-facets.md
@@ -34,7 +34,7 @@ Three layers solve this:
### Write
```typescript
-import { Brainy, NounType } from '@soulcraft/brainy'
+import { Brainy, NounType } from '@soulcraftlabs/brainy'
const brain = new Brainy()
await brain.init()
@@ -240,7 +240,7 @@ await brain.migrateField({
A realistic adoption sequence for a brain that started without these primitives:
```typescript
-import { Brainy, NounType } from '@soulcraft/brainy'
+import { Brainy, NounType } from '@soulcraftlabs/brainy'
const brain = new Brainy({ storage: { type: 'filesystem', path: './brain-data' } })
await brain.init()
diff --git a/docs/guides/upgrading-7-to-8.md b/docs/guides/upgrading-7-to-8.md
index a3c64fb9..53aa2a5c 100644
--- a/docs/guides/upgrading-7-to-8.md
+++ b/docs/guides/upgrading-7-to-8.md
@@ -25,7 +25,7 @@ content — and how 8.0 recovers it for you.
## TL;DR
-- **Just upgrade to `@soulcraft/brainy@8.0.12` (or later) and open the store.**
+- **Just upgrade to `@soulcraftlabs/brainy@8.0.12` (or later) and open the store.**
If a previous upgrade left VFS content stranded, 8.0.12 **heals it on open**,
with no operator action.
- Want to force or script it? Call **`await brain.vfs.adoptOrphanedBlobs()`**.
@@ -90,7 +90,7 @@ So the operator action for a stranded store is simply: **upgrade to 8.0.12 and
open it.**
```ts
-import { Brainy } from '@soulcraft/brainy'
+import { Brainy } from '@soulcraftlabs/brainy'
// Opening the store is all that is required — recovery runs during init().
const brain = new Brainy({ storage: { type: 'filesystem', path: '/data/my-store' } })
@@ -182,5 +182,5 @@ and opening each store is sufficient.
The recovery is copy-only, so no rollback of the recovery itself is ever needed.
If you need to roll back the **whole** 7→8 upgrade, restore the directory from
your pre-upgrade backup (retained automatically while recovery is incomplete, or
-your own snapshot) and pin `@soulcraft/brainy@7.x`. 8.0 does not keep the old
+your own snapshot) and pin `@soulcraftlabs/brainy@7.x`. 8.0 does not keep the old
branch layout in place, so a directory-level restore is the rollback path.
diff --git a/docs/guides/vue-integration.md b/docs/guides/vue-integration.md
index 7f7c6a06..34d18ebf 100644
--- a/docs/guides/vue-integration.md
+++ b/docs/guides/vue-integration.md
@@ -12,7 +12,7 @@ Complete guide to integrating Brainy with Vue.js applications, covering Vue 3, N
npm create vue@latest my-brainy-app
cd my-brainy-app
npm install
-npm install @soulcraft/brainy
+npm install @soulcraftlabs/brainy
```
### Basic Setup
@@ -574,7 +574,7 @@ Nuxt's server engine (Nitro) is the natural home for Brainy: it runs on Node/Bun
```javascript
// server/utils/brain.js (server-only — Nitro never bundles this into the client)
-import { Brainy } from '@soulcraft/brainy'
+import { Brainy } from '@soulcraftlabs/brainy'
let brainPromise
@@ -1201,7 +1201,7 @@ import vue from '@vitejs/plugin-vue'
export default defineConfig({
plugins: [vue()],
ssr: {
- external: ['@soulcraft/brainy']
+ external: ['@soulcraftlabs/brainy']
}
})
```
diff --git a/docs/neural-extraction.md b/docs/neural-extraction.md
index 989b1b60..cfb6d764 100644
--- a/docs/neural-extraction.md
+++ b/docs/neural-extraction.md
@@ -24,7 +24,7 @@ Brainy's neural extraction system uses a **4-signal ensemble architecture** to c
### Method 1: Brain Instance (Recommended)
```typescript
-import { Brainy, NounType } from '@soulcraft/brainy'
+import { Brainy, NounType } from '@soulcraftlabs/brainy'
const brain = new Brainy()
await brain.init()
@@ -62,9 +62,9 @@ const people = await brain.extractEntities('...', {
import {
SmartExtractor,
SmartRelationshipExtractor
-} from '@soulcraft/brainy'
+} from '@soulcraftlabs/brainy'
// Or use subpath imports:
-import { SmartExtractor } from '@soulcraft/brainy/neural/SmartExtractor'
+import { SmartExtractor } from '@soulcraftlabs/brainy/neural/SmartExtractor'
const brain = new Brainy()
await brain.init()
@@ -176,7 +176,7 @@ const withVectors = await brain.extractEntities(text, {
**Direct entity type classifier.** Use when you have pre-detected candidates or need custom configuration.
```typescript
-import { SmartExtractor, FormatContext } from '@soulcraft/brainy'
+import { SmartExtractor, FormatContext } from '@soulcraftlabs/brainy'
const extractor = new SmartExtractor(brain, {
minConfidence: 0.7, // Threshold
@@ -229,7 +229,7 @@ interface ExtractionResult {
**Relationship type classifier.** Determines verb/relationship types between entities.
```typescript
-import { SmartRelationshipExtractor } from '@soulcraft/brainy'
+import { SmartRelationshipExtractor } from '@soulcraftlabs/brainy'
const relExtractor = new SmartRelationshipExtractor(brain, {
minConfidence: 0.6,
@@ -286,7 +286,7 @@ const rel = await relExtractor.infer(
**Full extraction orchestrator.** Handles candidate detection, classification, and deduplication.
```typescript
-import { NeuralEntityExtractor } from '@soulcraft/brainy'
+import { NeuralEntityExtractor } from '@soulcraftlabs/brainy'
const extractor = new NeuralEntityExtractor(brain)
@@ -607,7 +607,7 @@ const locations = entities.filter(e => e.type === NounType.Location)
### Example 2: Excel Data Classification
```typescript
-import { SmartExtractor } from '@soulcraft/brainy'
+import { SmartExtractor } from '@soulcraftlabs/brainy'
const extractor = new SmartExtractor(brain)
@@ -629,7 +629,7 @@ for (let i = 0; i < cells.length; i++) {
### Example 3: Relationship Extraction
```typescript
-import { SmartRelationshipExtractor } from '@soulcraft/brainy'
+import { SmartRelationshipExtractor } from '@soulcraftlabs/brainy'
const relExtractor = new SmartRelationshipExtractor(brain)
diff --git a/docs/transactions.md b/docs/transactions.md
index fce7d10e..cbea39c0 100644
--- a/docs/transactions.md
+++ b/docs/transactions.md
@@ -204,8 +204,8 @@ await brain.add({ data: { name: 'Entity' }, type: NounType.Thing })
### Basic Add Operation
```typescript
-import { Brainy } from '@soulcraft/brainy'
-import { NounType } from '@soulcraft/brainy/types'
+import { Brainy } from '@soulcraftlabs/brainy'
+import { NounType } from '@soulcraftlabs/brainy/types'
const brain = new Brainy()
await brain.init()
@@ -428,7 +428,7 @@ await brain.relate({ ... }) // a crash here leaves the entity unlinked
```typescript
import { describe, it, expect } from 'vitest'
-import { Brainy } from '@soulcraft/brainy'
+import { Brainy } from '@soulcraftlabs/brainy'
describe('Transaction Tests', () => {
it('should rollback on failure', async () => {
diff --git a/docs/universal-display-augmentation.md b/docs/universal-display-augmentation.md
index da42874c..464b91fb 100644
--- a/docs/universal-display-augmentation.md
+++ b/docs/universal-display-augmentation.md
@@ -23,7 +23,7 @@ The Universal Display Augmentation is a powerful AI-powered system that automati
### Basic Usage
```typescript
-import { Brainy } from '@soulcraft/brainy'
+import { Brainy } from '@soulcraftlabs/brainy'
const brainy = new Brainy()
await brainy.init()
diff --git a/docs/vfs/PROJECTION_STRATEGY_API.md b/docs/vfs/PROJECTION_STRATEGY_API.md
index 380862e1..f1319d5b 100644
--- a/docs/vfs/PROJECTION_STRATEGY_API.md
+++ b/docs/vfs/PROJECTION_STRATEGY_API.md
@@ -71,9 +71,9 @@ Let's build a projection that organizes files by priority (high, medium, low):
### Step 1: Create the Strategy Class
```typescript
-import { BaseProjectionStrategy } from '@soulcraft/brainy/vfs/semantic'
-import { Brainy } from '@soulcraft/brainy'
-import { VirtualFileSystem, VFSEntity } from '@soulcraft/brainy/vfs'
+import { BaseProjectionStrategy } from '@soulcraftlabs/brainy/vfs/semantic'
+import { Brainy } from '@soulcraftlabs/brainy'
+import { VirtualFileSystem, VFSEntity } from '@soulcraftlabs/brainy/vfs'
export class PriorityProjection extends BaseProjectionStrategy {
readonly name = 'priority'
@@ -141,7 +141,7 @@ export class PriorityProjection extends BaseProjectionStrategy {
### Step 2: Register the Strategy
```typescript
-import { Brainy } from '@soulcraft/brainy'
+import { Brainy } from '@soulcraftlabs/brainy'
import { PriorityProjection } from './PriorityProjection'
const brain = new Brainy()
@@ -537,7 +537,7 @@ Use the projection's resolve cache:
```typescript
import { describe, it, expect, beforeAll } from 'vitest'
-import { Brainy } from '@soulcraft/brainy'
+import { Brainy } from '@soulcraftlabs/brainy'
import { PriorityProjection } from './PriorityProjection'
describe('PriorityProjection', () => {
@@ -714,7 +714,7 @@ async resolve(brain, vfs, value: string) {
3. Use appropriate limits: Don't fetch more than needed
### Type errors
-1. Import correct types: `import { Brainy, VirtualFileSystem } from '@soulcraft/brainy'`
+1. Import correct types: `import { Brainy, VirtualFileSystem } from '@soulcraftlabs/brainy'`
2. Use `as VFSEntity` when mapping results
3. Check BaseProjectionStrategy import
diff --git a/docs/vfs/QUICK_START.md b/docs/vfs/QUICK_START.md
index 8b0efce6..4a1f83dc 100644
--- a/docs/vfs/QUICK_START.md
+++ b/docs/vfs/QUICK_START.md
@@ -14,11 +14,11 @@ A file explorer that:
## ⚡ Step 1: Basic Setup (1 minute)
```bash
-npm install @soulcraft/brainy
+npm install @soulcraftlabs/brainy
```
```typescript
-import { Brainy } from '@soulcraft/brainy'
+import { Brainy } from '@soulcraftlabs/brainy'
// ✅ CORRECT: Use filesystem storage for production
const brain = new Brainy({
@@ -115,7 +115,7 @@ Here's a complete React component using the correct patterns:
```tsx
import React, { useState, useEffect } from 'react'
-import { Brainy } from '@soulcraft/brainy'
+import { Brainy } from '@soulcraftlabs/brainy'
export function FileExplorer() {
const [brain, setBrain] = useState(null)
@@ -288,8 +288,8 @@ Your file explorer is now working! Here's what to explore next:
### "Module not found" errors
```bash
# Make sure you're using the right import
-npm ls @soulcraft/brainy # Check version
-npm install @soulcraft/brainy@latest # Update if needed
+npm ls @soulcraftlabs/brainy # Check version
+npm install @soulcraftlabs/brainy@latest # Update if needed
```
### "VFS not initialized" errors
diff --git a/docs/vfs/README.md b/docs/vfs/README.md
index b95f0d7b..a94910c9 100644
--- a/docs/vfs/README.md
+++ b/docs/vfs/README.md
@@ -24,7 +24,7 @@ Brainy VFS is a revolutionary virtual filesystem that runs on top of Brainy's ne
## Quick Start
```javascript
-import { VirtualFileSystem } from '@soulcraft/brainy/vfs'
+import { VirtualFileSystem } from '@soulcraftlabs/brainy/vfs'
// Initialize the VFS
const vfs = new VirtualFileSystem({
@@ -381,7 +381,7 @@ Brainy VFS fully leverages Brainy's revolutionary Triple Intelligence system:
## Installation
```bash
-npm install @soulcraft/brainy
+npm install @soulcraftlabs/brainy
```
## Requirements
diff --git a/docs/vfs/ROADMAP.md b/docs/vfs/ROADMAP.md
index 93c5b901..c8d15cd2 100644
--- a/docs/vfs/ROADMAP.md
+++ b/docs/vfs/ROADMAP.md
@@ -135,7 +135,7 @@ Mount VFS as a native filesystem on Linux/Mac/Windows.
```typescript
// Planned (research phase)
-import { mountVFS } from '@soulcraft/brainy/vfs/fuse'
+import { mountVFS } from '@soulcraftlabs/brainy/vfs/fuse'
await mountVFS(vfs, {
mountPoint: '/mnt/brainy',
@@ -160,7 +160,7 @@ These features would benefit from community contributions. If you're interested
### Express.js Static Middleware
```typescript
// Wanted: Community contribution
-import { createStaticMiddleware } from '@soulcraft/brainy/vfs/express'
+import { createStaticMiddleware } from '@soulcraftlabs/brainy/vfs/express'
app.use('/files', createStaticMiddleware(vfs, {
index: ['index.html', 'index.md'],
@@ -172,7 +172,7 @@ app.use('/files', createStaticMiddleware(vfs, {
### VSCode Extension
```typescript
// Wanted: Community contribution
-import { VFSProvider } from '@soulcraft/brainy/vfs/vscode'
+import { VFSProvider } from '@soulcraftlabs/brainy/vfs/vscode'
const provider = new VFSProvider(vfs)
vscode.workspace.registerFileSystemProvider('brainy', provider)
diff --git a/docs/vfs/SEMANTIC_VFS.md b/docs/vfs/SEMANTIC_VFS.md
index 9298c822..f34ee9ae 100644
--- a/docs/vfs/SEMANTIC_VFS.md
+++ b/docs/vfs/SEMANTIC_VFS.md
@@ -327,7 +327,7 @@ console.log(id1 === id2 && id2 === id3) // true
Create your own semantic dimensions:
```typescript
-import { BaseProjectionStrategy } from '@soulcraft/brainy/vfs/semantic'
+import { BaseProjectionStrategy } from '@soulcraftlabs/brainy/vfs/semantic'
class PriorityProjection extends BaseProjectionStrategy {
readonly name = 'priority'
diff --git a/docs/vfs/VFS_API_GUIDE.md b/docs/vfs/VFS_API_GUIDE.md
index e0c6a94c..5dcaaeb8 100644
--- a/docs/vfs/VFS_API_GUIDE.md
+++ b/docs/vfs/VFS_API_GUIDE.md
@@ -7,7 +7,7 @@ Brainy's Virtual Filesystem (VFS) provides a POSIX-like filesystem interface tha
## Quick Start
```typescript
-import { Brainy } from '@soulcraft/brainy'
+import { Brainy } from '@soulcraftlabs/brainy'
// Initialize Brainy
const brain = new Brainy({
@@ -598,7 +598,7 @@ const user = await store.findById('users', 'user123')
VFS uses standard POSIX-style errors:
```typescript
-import { VFSError, VFSErrorCode } from '@soulcraft/brainy'
+import { VFSError, VFSErrorCode } from '@soulcraftlabs/brainy'
try {
await vfs.readFile('/nonexistent.txt')
diff --git a/docs/vfs/VFS_CORE.md b/docs/vfs/VFS_CORE.md
index 1eeaf9f8..c1d502c0 100644
--- a/docs/vfs/VFS_CORE.md
+++ b/docs/vfs/VFS_CORE.md
@@ -280,7 +280,7 @@ GitBridge provides Git import/export capabilities:
#### GitBridge Usage
```javascript
// Import and instantiate GitBridge
-import { GitBridge } from '@soulcraft/brainy'
+import { GitBridge } from '@soulcraftlabs/brainy'
const gitBridge = new GitBridge(vfs, brain)
// Export VFS to Git repository structure
@@ -452,7 +452,7 @@ This ordering prevents race conditions where file writes might fail because pare
## Complete Example
```javascript
-import { Brainy } from '@soulcraft/brainy'
+import { Brainy } from '@soulcraftlabs/brainy'
async function vfsExample() {
// Initialize
diff --git a/docs/vfs/VFS_GRAPH_TYPES.md b/docs/vfs/VFS_GRAPH_TYPES.md
index 3c1f30f0..478bef7f 100644
--- a/docs/vfs/VFS_GRAPH_TYPES.md
+++ b/docs/vfs/VFS_GRAPH_TYPES.md
@@ -196,5 +196,5 @@ await brain.relate({
Always import and use the type enums:
```javascript
-import { NounType, VerbType } from '@soulcraft/brainy'
+import { NounType, VerbType } from '@soulcraftlabs/brainy'
```
\ No newline at end of file
diff --git a/docs/vfs/VFS_INITIALIZATION.md b/docs/vfs/VFS_INITIALIZATION.md
index 97e6b0bf..fd12fc71 100644
--- a/docs/vfs/VFS_INITIALIZATION.md
+++ b/docs/vfs/VFS_INITIALIZATION.md
@@ -5,7 +5,7 @@
The Brainy VFS is automatically initialized during `brain.init()`. No separate initialization needed!
```javascript
-import { Brainy } from '@soulcraft/brainy'
+import { Brainy } from '@soulcraftlabs/brainy'
// Create and initialize Brainy
const brain = new Brainy({
@@ -71,7 +71,7 @@ VFS stores files as entities and relationships in the same graph as everything e
## Complete Example
```javascript
-import { Brainy } from '@soulcraft/brainy'
+import { Brainy } from '@soulcraftlabs/brainy'
async function useVFS() {
// Initialize Brainy
@@ -100,7 +100,7 @@ useVFS().catch(console.error)
## TypeScript Usage
```typescript
-import { Brainy, VirtualFileSystem } from '@soulcraft/brainy'
+import { Brainy, VirtualFileSystem } from '@soulcraftlabs/brainy'
class FileManager {
private brain: Brainy
diff --git a/docs/vfs/building-file-explorers.md b/docs/vfs/building-file-explorers.md
index 6bb31871..7514c12e 100644
--- a/docs/vfs/building-file-explorers.md
+++ b/docs/vfs/building-file-explorers.md
@@ -37,7 +37,7 @@ Brainy VFS provides safe, tree-aware methods that prevent these issues:
### Method 1: Use `getDirectChildren()` (Recommended)
```typescript
-import { Brainy, VirtualFileSystem } from '@soulcraft/brainy'
+import { Brainy, VirtualFileSystem } from '@soulcraftlabs/brainy'
const brain = new Brainy()
await brain.init()
@@ -97,7 +97,7 @@ Here's a complete example using React:
```tsx
import React, { useState, useEffect } from 'react'
-import { VirtualFileSystem } from '@soulcraft/brainy'
+import { VirtualFileSystem } from '@soulcraftlabs/brainy'
interface FileNode {
name: string
@@ -177,7 +177,7 @@ function TreeView({ node, onToggle, expanded }) {
If you must build trees manually from flat lists, use the `VFSTreeUtils`:
```typescript
-import { VFSTreeUtils } from '@soulcraft/brainy/vfs'
+import { VFSTreeUtils } from '@soulcraftlabs/brainy/vfs'
// Get all entities somehow
const allEntities = await vfs.getDescendants('/root')
diff --git a/examples/bluesky-distributed-setup.js b/examples/bluesky-distributed-setup.js
index 9e83cf25..e3b33506 100644
--- a/examples/bluesky-distributed-setup.js
+++ b/examples/bluesky-distributed-setup.js
@@ -7,7 +7,7 @@
* the Bluesky firehose with Brainy's distributed architecture
*/
-import { Brainy } from '@soulcraft/brainy'
+import { Brainy } from '@soulcraftlabs/brainy'
import { WebSocket } from 'ws'
// =====================================================
diff --git a/examples/monitor-cache-performance.ts b/examples/monitor-cache-performance.ts
index 9d50d476..87c965a2 100644
--- a/examples/monitor-cache-performance.ts
+++ b/examples/monitor-cache-performance.ts
@@ -14,7 +14,7 @@
* ts-node examples/monitor-cache-performance.ts
*/
-import { Brainy, NounType } from '@soulcraft/brainy'
+import { Brainy, NounType } from '@soulcraftlabs/brainy'
// ANSI color codes for pretty output
const colors = {
diff --git a/integrations/README.md b/integrations/README.md
index aa3d795b..de156623 100644
--- a/integrations/README.md
+++ b/integrations/README.md
@@ -5,7 +5,7 @@ Connect Brainy to spreadsheets, BI tools, and external systems with zero configu
## Quick Start
```typescript
-import { Brainy } from '@soulcraft/brainy'
+import { Brainy } from '@soulcraftlabs/brainy'
const brain = new Brainy({ integrations: true })
await brain.init()
@@ -178,7 +178,7 @@ Webhooks include `X-Brainy-Signature` header with HMAC-SHA256 signature.
### Minimal (in-memory):
```typescript
-import { Brainy } from '@soulcraft/brainy'
+import { Brainy } from '@soulcraftlabs/brainy'
const brain = new Brainy({ integrations: true })
await brain.init()
@@ -194,7 +194,7 @@ console.log(brain.hub.getInstructions())
```typescript
import express from 'express'
-import { Brainy } from '@soulcraft/brainy'
+import { Brainy } from '@soulcraftlabs/brainy'
const app = express()
const brain = new Brainy({
@@ -232,7 +232,7 @@ app.listen(3000, () => {
```typescript
import { Hono } from 'hono'
-import { Brainy } from '@soulcraft/brainy'
+import { Brainy } from '@soulcraftlabs/brainy'
const app = new Hono()
diff --git a/integrations/google-sheets/README.md b/integrations/google-sheets/README.md
index b2b0af3a..8309a30a 100644
--- a/integrations/google-sheets/README.md
+++ b/integrations/google-sheets/README.md
@@ -99,7 +99,7 @@ Add the `BRAINY_URL` script property in Apps Script settings.
The simplest way to enable all integrations:
```javascript
-import { Brainy } from '@soulcraft/brainy'
+import { Brainy } from '@soulcraftlabs/brainy'
const brain = new Brainy({ integrations: true })
await brain.init()
@@ -112,7 +112,7 @@ With Express:
```javascript
import express from 'express'
-import { Brainy } from '@soulcraft/brainy'
+import { Brainy } from '@soulcraftlabs/brainy'
const app = express()
const brain = new Brainy({ integrations: true })
diff --git a/package-lock.json b/package-lock.json
index afce417d..c4757030 100644
--- a/package-lock.json
+++ b/package-lock.json
@@ -1,12 +1,12 @@
{
- "name": "@soulcraft/brainy",
- "version": "10.3.1",
+ "name": "@soulcraftlabs/brainy",
+ "version": "10.4.12",
"lockfileVersion": 3,
"requires": true,
"packages": {
"": {
- "name": "@soulcraft/brainy",
- "version": "10.3.1",
+ "name": "@soulcraftlabs/brainy",
+ "version": "10.4.12",
"license": "MIT",
"dependencies": {
"@msgpack/msgpack": "^3.1.2",
diff --git a/package.json b/package.json
index 75e5bfbc..649f2aaf 100644
--- a/package.json
+++ b/package.json
@@ -1,6 +1,7 @@
{
- "name": "@soulcraft/brainy",
- "version": "10.3.1",
+ "name": "@soulcraftlabs/brainy",
+ "version": "10.4.12",
+ "brainyContract": 1,
"description": "Universal Knowledge Protocol™ - World's first Triple Intelligence database unifying vector, graph, and document search in one API. Stage 3 CANONICAL: 42 nouns × 127 verbs covering 96-97% of all human knowledge.",
"main": "dist/index.js",
"module": "dist/index.js",
@@ -87,7 +88,7 @@
"test:watch": "NODE_OPTIONS='--max-old-space-size=8192' vitest --config tests/configs/vitest.unit.config.ts",
"test:coverage": "NODE_OPTIONS='--max-old-space-size=8192' vitest run --config tests/configs/vitest.unit.config.ts --coverage",
"test:unit": "NODE_OPTIONS='--max-old-space-size=8192' vitest run --config tests/configs/vitest.unit.config.ts",
- "test:perf": "vitest run tests/unit/performance --reporter=basic",
+ "test:perf": "vitest run --config tests/configs/vitest.perf.config.ts",
"test:integration": "NODE_OPTIONS='--max-old-space-size=8192' vitest run --config tests/configs/vitest.integration.config.ts",
"test:semantic": "NODE_OPTIONS='--max-old-space-size=8192' vitest run --config tests/configs/vitest.semantic.config.ts",
"test:all": "npm run test:unit && npm run test:integration",
@@ -126,15 +127,16 @@
"license": "MIT",
"private": false,
"publishConfig": {
- "access": "public"
+ "access": "public",
+ "registry": "https://source.soulcraft.com/api/packages/soulcraftlabs/npm/"
},
- "homepage": "https://source.soulcraft.com/soulcraft/brainy",
+ "homepage": "https://source.soulcraft.com/soulcraftlabs/open-brainy",
"bugs": {
- "url": "https://source.soulcraft.com/soulcraft/brainy/issues"
+ "url": "https://source.soulcraft.com/soulcraftlabs/open-brainy/issues"
},
"repository": {
"type": "git",
- "url": "git+https://source.soulcraft.com/soulcraft/brainy.git"
+ "url": "git+https://source.soulcraft.com/soulcraftlabs/open-brainy.git"
},
"files": [
"dist/**/*.js",
diff --git a/scripts/buildEmbeddedPatterns.ts b/scripts/buildEmbeddedPatterns.ts
index 73e51224..c046df45 100644
--- a/scripts/buildEmbeddedPatterns.ts
+++ b/scripts/buildEmbeddedPatterns.ts
@@ -10,6 +10,7 @@ import { TransformerEmbedding } from '../src/utils/embedding.js'
import * as fs from 'fs/promises'
import * as path from 'path'
import { fileURLToPath } from 'url'
+import { resolveDeterministicStamp } from './lib/deterministicStamp.js'
const __dirname = path.dirname(fileURLToPath(import.meta.url))
@@ -97,13 +98,22 @@ async function buildEmbeddedPatterns() {
// Convert to base64 for embedding in TypeScript
const uint8 = new Uint8Array(buffer)
const base64 = Buffer.from(uint8).toString('base64')
-
+
+ // Deterministic stamp: derived from the git commit time of this
+ // generator's inputs, never from wall-clock time — two builds of the
+ // same source tree must produce byte-identical output.
+ const outputPath = path.join(__dirname, '..', 'src', 'neural', 'embeddedPatterns.ts')
+ const generatedStamp = resolveDeterministicStamp(
+ [path.join(__dirname, 'buildEmbeddedPatterns.ts'), libraryPath],
+ outputPath
+ )
+
// Generate TypeScript file with everything embedded
const tsContent = `/**
* 🧠 BRAINY EMBEDDED PATTERNS
*
* AUTO-GENERATED - DO NOT EDIT
- * Generated: ${new Date().toISOString()}
+ * Generated: ${generatedStamp}
* Patterns: ${libraryData.patterns.length}
* Coverage: 94-98% of all queries
*
@@ -197,7 +207,6 @@ prodLog.info(\`🧠 Brainy Pattern Library loaded: \${EMBEDDED_PATTERNS.length}
`
// Write the TypeScript file
- const outputPath = path.join(__dirname, '..', 'src', 'neural', 'embeddedPatterns.ts')
await fs.writeFile(outputPath, tsContent)
// Report statistics
diff --git a/scripts/buildTypeEmbeddings.ts b/scripts/buildTypeEmbeddings.ts
index 61bcf238..688d6ac1 100644
--- a/scripts/buildTypeEmbeddings.ts
+++ b/scripts/buildTypeEmbeddings.ts
@@ -11,6 +11,7 @@ import * as fs from 'fs/promises'
import * as path from 'path'
import { fileURLToPath } from 'url'
import { NounType, VerbType } from '../src/types/graphTypes.js'
+import { resolveDeterministicStamp } from './lib/deterministicStamp.js'
const __dirname = path.dirname(fileURLToPath(import.meta.url))
@@ -373,12 +374,24 @@ async function buildTypeEmbeddings() {
const uint8 = new Uint8Array(buffer)
const base64 = Buffer.from(uint8).toString('base64')
+ // Deterministic stamp: derived from the git commit time of this
+ // generator's inputs, never from wall-clock time — two builds of the
+ // same source tree must produce byte-identical output.
+ const outputPath = path.join(__dirname, '..', 'src', 'neural', 'embeddedTypeEmbeddings.ts')
+ const generatedStamp = resolveDeterministicStamp(
+ [
+ path.join(__dirname, 'buildTypeEmbeddings.ts'),
+ path.join(__dirname, '..', 'src', 'types', 'graphTypes.ts')
+ ],
+ outputPath
+ )
+
// Generate TypeScript file
const tsContent = `/**
* 🧠 BRAINY EMBEDDED TYPE EMBEDDINGS
*
* AUTO-GENERATED - DO NOT EDIT
- * Generated: ${new Date().toISOString()}
+ * Generated: ${generatedStamp}
* Noun Types: ${nounTypes.length}
* Verb Types: ${verbTypes.length}
*
@@ -395,7 +408,7 @@ export const TYPE_METADATA = {
verbTypes: ${verbTypes.length},
totalTypes: ${totalTypes},
embeddingDimensions: ${embeddingDim},
- generatedAt: "${new Date().toISOString()}",
+ generatedAt: "${generatedStamp}",
sizeBytes: {
embeddings: ${buffer.byteLength},
base64: ${base64.length}
@@ -494,7 +507,6 @@ prodLog.info(\`🧠 Brainy Type Embeddings loaded: \${TYPE_METADATA.nounTypes} n
`
// Write the TypeScript file
- const outputPath = path.join(__dirname, '..', 'src', 'neural', 'embeddedTypeEmbeddings.ts')
await fs.writeFile(outputPath, tsContent)
// Report statistics
diff --git a/scripts/emit-contract-manifest.mjs b/scripts/emit-contract-manifest.mjs
new file mode 100644
index 00000000..be73d4ca
--- /dev/null
+++ b/scripts/emit-contract-manifest.mjs
@@ -0,0 +1,128 @@
+#!/usr/bin/env node
+/**
+ * Emit this build's API-contract manifest to docs/api-contract.json.
+ *
+ * WHY IT IS GENERATED, NOT WRITTEN: a hand-kept list of doors drifts from the
+ * code the first time somebody adds one. This reads the surface the build
+ * actually exposes — the prototype's own methods and accessors, the exported
+ * error classes, the `where` operator sets, the field-addressing vocabulary,
+ * the health verdicts — so a diff between two engines' manifests is a diff
+ * between two engines, never between two authors.
+ *
+ * Requirement marking (required / optional per door) is NOT derivable from the
+ * surface — it is a commitment, recorded with the contract's owner rather than
+ * here. This manifest carries the surface; the promise lives with the contract.
+ *
+ * Usage: node scripts/emit-contract-manifest.mjs [--check]
+ * --check exits non-zero when the committed manifest is stale.
+ */
+
+import { writeFileSync, readFileSync, existsSync } from 'node:fs'
+import { join, dirname } from 'node:path'
+import { fileURLToPath } from 'node:url'
+
+const ROOT = join(dirname(fileURLToPath(import.meta.url)), '..')
+const OUT = join(ROOT, 'docs', 'api-contract.json')
+
+const { Brainy } = await import(join(ROOT, 'dist', 'brainy.js'))
+const errorsModule = await import(join(ROOT, 'dist', 'errors', 'brainyError.js'))
+const versionModule = await import(join(ROOT, 'dist', 'utils', 'version.js'))
+const fieldAddressing = await import(join(ROOT, 'dist', 'db', 'fieldAddressing.js'))
+
+/** Every own method and accessor on the class's prototype, minus the private ones. */
+function surfaceOf(ctor) {
+ const doors = []
+ for (const name of Object.getOwnPropertyNames(ctor.prototype)) {
+ if (name === 'constructor' || name.startsWith('_')) continue
+ const descriptor = Object.getOwnPropertyDescriptor(ctor.prototype, name)
+ if (!descriptor) continue
+ if (typeof descriptor.value === 'function') {
+ doors.push({ name, kind: 'method', arity: descriptor.value.length })
+ } else if (descriptor.get) {
+ doors.push({ name, kind: 'accessor' })
+ }
+ }
+ return doors.sort((a, b) => a.name.localeCompare(b.name))
+}
+
+const errors = Object.entries(errorsModule)
+ .filter(([name, value]) => typeof value === 'function' && /Error$/.test(name))
+ .map(([name]) => name)
+ .sort()
+
+// The operator sets, read from the engine's own refusal message so the
+// manifest can never disagree with the validator.
+const filterSource = readFileSync(join(ROOT, 'src', 'utils', 'metadataFilter.ts'), 'utf-8')
+const acceptedMatch = filterSource.match(/const VALUE_OPERATORS = new Set\(\[([\s\S]*?)\]\)/)
+if (!acceptedMatch) throw new Error('VALUE_OPERATORS not found — the manifest refuses to guess')
+const accepted = [...acceptedMatch[1].matchAll(/'([^']+)'/g)].map((m) => m[1]).sort()
+
+const indexSource = readFileSync(join(ROOT, 'src', 'utils', 'metadataIndex.ts'), 'utf-8')
+const refusedByIndex = ['endsWith', 'length', 'matches', 'startsWith'].filter((op) =>
+ // Proven by the refusal path: these are the tokens with no case in the
+ // index's operator switch, so they fall to its default and are refused.
+ !new RegExp(`case '${op}':`).test(indexSource)
+)
+const servedOnIndex = accepted.filter((op) => !refusedByIndex.includes(op))
+
+const manifest = {
+ contractVersion: versionModule.contractVersion(),
+ engine: '@soulcraftlabs/brainy',
+ compatibility: {
+ minor:
+ 'additive — a new optional door, a new served operator, a new error class; every existing implementation still conforms',
+ major:
+ 'breaking — a door removed, an answer narrowed, an ordering law changed, an optional door promoted to required, or an operator moved from served to refused'
+ },
+ doors: surfaceOf(Brainy),
+ errors,
+ operators: {
+ accepted,
+ servedOnIndexPath: servedOnIndex,
+ refusedByIndexPath: refusedByIndex,
+ combinators: ['allOf', 'anyOf', 'not']
+ },
+ fieldAddressing: {
+ systemKeyPrefix: 'system.',
+ systemEntityScalars: [...(fieldAddressing.SYSTEM_ENTITY_SCALARS ?? [])].sort(),
+ systemRelationScalars: [...(fieldAddressing.SYSTEM_RELATION_SCALARS ?? [])].sort(),
+ plumbingFields: [...(fieldAddressing.PLUMBING_FIELDS ?? [])].sort()
+ },
+ health: {
+ verdicts: ['pass', 'warn', 'fail'],
+ healKinds: ['none', 'repair', 'rebuild'],
+ servingWithholdingInvariants: [
+ 'index-initialized',
+ 'durable-state-present',
+ 'manifest-residency',
+ 'replay-clean',
+ 'strand-latch'
+ ]
+ }
+}
+
+const rendered = `${JSON.stringify(manifest, null, 2)}\n`
+
+if (process.argv.includes('--check')) {
+ if (!existsSync(OUT)) {
+ console.error(`docs/api-contract.json is missing — run: node scripts/emit-contract-manifest.mjs`)
+ process.exit(1)
+ }
+ if (readFileSync(OUT, 'utf-8') !== rendered) {
+ console.error(
+ `docs/api-contract.json is STALE — the public surface changed. Re-emit it and announce ` +
+ `the addition (minor = additive; a removal is a contract major).`
+ )
+ process.exit(1)
+ }
+ console.log(`docs/api-contract.json is current (${manifest.doors.length} doors, contract ${manifest.contractVersion}).`)
+ process.exit(0)
+}
+
+writeFileSync(OUT, rendered)
+console.log(
+ `Wrote docs/api-contract.json — contract ${manifest.contractVersion}, ` +
+ `${manifest.doors.length} doors, ${manifest.errors.length} error classes, ` +
+ `${manifest.operators.accepted.length} operators ` +
+ `(${manifest.operators.refusedByIndexPath.length} refused by the index path).`
+)
diff --git a/scripts/lib/deterministicStamp.ts b/scripts/lib/deterministicStamp.ts
new file mode 100644
index 00000000..c2a66604
--- /dev/null
+++ b/scripts/lib/deterministicStamp.ts
@@ -0,0 +1,118 @@
+/**
+ * Deterministic generation-stamp resolution for Brainy's build-time code
+ * generators.
+ *
+ * Two builds of the same source tree must produce byte-identical output.
+ * A wall-clock stamp (`new Date()`) breaks that guarantee, so every
+ * generator that writes a "Generated:" header or a `generatedAt` field
+ * into its output must resolve the stamp through this module instead.
+ *
+ * Resolution order:
+ * 1. The newest git commit timestamp among the generator's input files
+ * (the generator script itself always counts as an input).
+ * 2. If git metadata is unavailable (for example, building from a
+ * published npm tarball with no `.git` directory), the stamp already
+ * recorded in the previously generated output file.
+ * 3. If neither is available, the fixed epoch string
+ * `1970-01-01T00:00:00.000Z`.
+ *
+ * Every fallback logs a line to stderr — deterministic degradation is
+ * loud, never a silent divergence.
+ */
+
+import { execFileSync } from 'child_process'
+import * as fs from 'fs'
+
+const EPOCH_STAMP = '1970-01-01T00:00:00.000Z'
+const STAMP_PATTERN = /\*\s*Generated:\s*(\S+)/
+
+/**
+ * Resolve the deterministic stamp for a generator run.
+ *
+ * @param inputPaths Absolute paths to every file whose content determines
+ * the generator's output, including the generator script itself.
+ * @param previousOutputPath Absolute path to the previously generated
+ * file, used for the existing-stamp fallback when git is unavailable.
+ * @returns An ISO-8601 timestamp string that is deterministic for a given
+ * source tree.
+ */
+export function resolveDeterministicStamp(
+ inputPaths: string[],
+ previousOutputPath: string
+): string {
+ const gitStamp = newestGitCommitTimestamp(inputPaths)
+ if (gitStamp) {
+ return gitStamp
+ }
+
+ const existingStamp = readExistingStamp(previousOutputPath)
+ if (existingStamp) {
+ process.stderr.write(
+ `[deterministic-stamp] no git commit history found for generator inputs; ` +
+ `reusing existing stamp from ${previousOutputPath}: ${existingStamp}\n`
+ )
+ return existingStamp
+ }
+
+ process.stderr.write(
+ `[deterministic-stamp] no git commit history and no previous output at ` +
+ `${previousOutputPath}; falling back to fixed epoch stamp ${EPOCH_STAMP}\n`
+ )
+ return EPOCH_STAMP
+}
+
+/**
+ * Find the newest git commit timestamp among the given input paths.
+ * Returns null if git is unavailable, the tree is not a git repository,
+ * or none of the inputs have any commit history yet.
+ */
+function newestGitCommitTimestamp(inputPaths: string[]): string | null {
+ let newest: string | null = null
+
+ for (const inputPath of inputPaths) {
+ if (!fs.existsSync(inputPath)) {
+ continue
+ }
+
+ let out: string
+ try {
+ out = execFileSync(
+ 'git',
+ ['log', '-1', '--format=%cI', '--', inputPath],
+ { stdio: ['ignore', 'pipe', 'ignore'] }
+ )
+ .toString()
+ .trim()
+ } catch {
+ // git missing, not a repository, or no permissions — handled by the
+ // caller's fallback chain.
+ continue
+ }
+
+ if (!out) {
+ // Path exists but has no commit history yet (e.g. newly created,
+ // uncommitted file).
+ continue
+ }
+
+ if (!newest || new Date(out).getTime() > new Date(newest).getTime()) {
+ newest = out
+ }
+ }
+
+ return newest
+}
+
+/**
+ * Parse the `* Generated: ` header out of a previously
+ * generated file, if one exists.
+ */
+function readExistingStamp(outputPath: string): string | null {
+ if (!fs.existsSync(outputPath)) {
+ return null
+ }
+
+ const content = fs.readFileSync(outputPath, 'utf-8')
+ const match = content.match(STAMP_PATTERN)
+ return match ? match[1] : null
+}
diff --git a/scripts/release.sh b/scripts/release.sh
index 03d60ac2..a9a1f6e9 100755
--- a/scripts/release.sh
+++ b/scripts/release.sh
@@ -15,6 +15,12 @@ NC='\033[0m' # No Color
RELEASE_TYPE="${1:-patch}" # patch, minor, or major
SKIP_TESTS=false
DRY_RUN=false
+# --source-only is now a no-op: The Source is the one registry, so every
+# release already ships Source-only — tag, CI's publish to The Source, the
+# release page, and the docs push, with no separate storefront leg to skip.
+# The flag is still accepted (for backward-compatible invocations) and just
+# prints a notice; it no longer changes behavior.
+SOURCE_ONLY=false
for arg in "$@"; do
case $arg in
@@ -24,6 +30,9 @@ for arg in "$@"; do
--dry-run)
DRY_RUN=true
;;
+ --source-only)
+ SOURCE_ONLY=true
+ ;;
esac
done
@@ -100,7 +109,7 @@ else
;;
*)
echo -e "${RED}❌ Invalid release type: ${RELEASE_TYPE}${NC}"
- echo "Usage: ./scripts/release.sh [patch|minor|major|] [--dry-run]"
+ echo "Usage: ./scripts/release.sh [patch|minor|major|] [--dry-run] [--source-only (no-op; The Source is the one registry)]"
exit 1
;;
esac
@@ -119,6 +128,9 @@ echo -e "${BLUE}New version: ${NEW_VERSION}${NC}"
if [ "$PRERELEASE" = true ]; then
echo -e "${YELLOW}⚠️ Prerelease → npm dist-tag '${NPM_TAG}', GitHub prerelease${NC}"
fi
+if [ "$SOURCE_ONLY" = true ]; then
+ echo -e "${YELLOW}⚠️ The Source is the one registry; --source-only is implied${NC}"
+fi
echo ""
if [ "$DRY_RUN" = true ]; then
@@ -142,13 +154,26 @@ else
fi
# Create new changelog entry
-CHANGELOG_ENTRY="### [${NEW_VERSION}](https://source.soulcraft.com/soulcraft/brainy/compare/v${CURRENT_VERSION}...v${NEW_VERSION}) ($(date +%Y-%m-%d))
+RELEASE_DATE=$(date +%Y-%m-%d)
+CHANGELOG_ENTRY="### [${NEW_VERSION}](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v${CURRENT_VERSION}...v${NEW_VERSION}) (${RELEASE_DATE})
${COMMITS}
"
+# A CURATED entry wins over the generated one. When a release is cut from a
+# lineage that diverged from the previous tag (a candidate branch carrying
+# main's history), `git log ..HEAD` lists every commit the tag never
+# saw — old notes, already-shipped fixes under new hashes, merge commits — and a
+# wall entry derived from it would misreport the release. If CHANGELOG.md
+# already carries a `### [NEW_VERSION]` heading, it was written on purpose:
+# keep it, and skip the generated prepend entirely.
+CURATED_ENTRY=false
+if grep -qE "^### \[${NEW_VERSION}\]" CHANGELOG.md 2>/dev/null; then
+ CURATED_ENTRY=true
+ echo -e "${YELLOW}CHANGELOG already carries a curated ### [${NEW_VERSION}] entry — keeping it, not generating one from commits${NC}"
+fi
# Prepend to CHANGELOG.md after header
-if [ -f "CHANGELOG.md" ]; then
+if [ "$CURATED_ENTRY" = false ] && [ -f "CHANGELOG.md" ]; then
# Read header (first 4 lines)
HEADER=$(head -n 4 CHANGELOG.md)
# Read rest of file
@@ -162,6 +187,19 @@ if [ -f "CHANGELOG.md" ]; then
fi
echo -e "${GREEN}✅ CHANGELOG updated${NC}\n"
+# Step 6b: Update the releases wall entry — mechanical, derived from the
+# CHANGELOG entry just composed. The fleet's HQ page reads open-brainy.json
+# from the one shared releases repo, soulcraftlabs/releases on The Source —
+# this used to be hand-written after every release (David: never again —
+# make it a step of the rail, landed in the one shared home; this repo no
+# longer hosts its own copy). This step clones/fetches that repo into a
+# local cache, prepends the entry, and pushes it directly — a real
+# cross-repo push, refusing loudly (never skipping) on any
+# clone/validation/commit/push failure.
+echo -e "${BLUE}5️⃣▸ Updating the releases wall...${NC}"
+node scripts/wall-entry.mjs --product open-brainy --version "${NEW_VERSION}" --date "${RELEASE_DATE}" --from-changelog CHANGELOG.md
+echo -e "${GREEN}✅ Releases wall updated${NC}\n"
+
# Step 7: Create release commit
echo -e "${BLUE}6️⃣ Creating release commit...${NC}"
git add package.json package-lock.json CHANGELOG.md
@@ -193,9 +231,9 @@ echo -e "${GREEN}✅ Pushed to origin${NC}\n"
# .forgejo/workflows/publish-source.yml, which builds and publishes on The
# Source's own runner (datacenter-side: seconds, not the laptop's WAN timing
# out on an 87MB tarball PUT). The laptop holds no home-registry publish
-# credential anymore; it only waits for CI's result before trusting the
-# home/npmjs pair enough to publish the storefront leg.
-SOURCE_NPM_REG="https://source.soulcraft.com/api/packages/soulcraft/npm/"
+# credential anymore; it only waits for CI's result before continuing on to
+# the release page and the docs push.
+SOURCE_NPM_REG="https://source.soulcraft.com/api/packages/soulcraftlabs/npm/"
SOURCE_POLL_INTERVAL_S=15
SOURCE_POLL_MAX_ATTEMPTS=200 # 200 × 15s = 50 minutes — the runner is sequential and a busy day's ci.yml
# backlog has twice exceeded the old 20-minute window (8.10.3, 9.0.0);
@@ -203,7 +241,7 @@ SOURCE_POLL_MAX_ATTEMPTS=200 # 200 × 15s = 50 minutes — the runner is sequen
echo -e "${BLUE}9️⃣ Waiting for CI to publish v${NEW_VERSION} to The Source registry (home)...${NC}"
SOURCE_LANDED=false
for ((attempt = 1; attempt <= SOURCE_POLL_MAX_ATTEMPTS; attempt++)); do
- LANDED_VERSION=$(npm view "@soulcraft/brainy@${NEW_VERSION}" version "--@soulcraft:registry=${SOURCE_NPM_REG}" 2>/dev/null || echo "")
+ LANDED_VERSION=$(npm view "@soulcraftlabs/brainy@${NEW_VERSION}" version "--@soulcraftlabs:registry=${SOURCE_NPM_REG}" 2>/dev/null || echo "")
if [ "$LANDED_VERSION" = "$NEW_VERSION" ]; then
SOURCE_LANDED=true
break
@@ -216,50 +254,8 @@ if [ "$SOURCE_LANDED" = true ]; then
echo -e "${GREEN}✅ CI published v${NEW_VERSION} to The Source${NC}\n"
else
echo -e "${RED}❌ CI's home publish did not land — check the workflow run on The Source; the pair must not diverge.${NC}"
- echo -e "${RED} v${NEW_VERSION} was tagged and pushed, but @soulcraft/brainy@${NEW_VERSION} never became visible on the${NC}"
- echo -e "${RED} Source registry after ${SOURCE_POLL_MAX_ATTEMPTS} attempts, ${SOURCE_POLL_INTERVAL_S}s apart. Aborting before npmjs.${NC}"
- exit 1
-fi
-
-echo -e "${BLUE}9️⃣½ Publishing to npmjs (storefront, dist-tag: ${NPM_TAG})...${NC}"
-# BYTE-IDENTITY LAW: the storefront republishes CI's EXACT artifact — download
-# the tarball The Source serves and publish that file, never a fresh local pack
-# (a local rebuild can differ byte-wise, and the fleet verifies the pair by
-# shasum across registries).
-STOREFRONT_TMP="$(mktemp -d)"
-(cd "$STOREFRONT_TMP" && npm pack "@soulcraft/brainy@${NEW_VERSION}" "--@soulcraft:registry=${SOURCE_NPM_REG}" >/dev/null)
-SOURCE_TARBALL="$(ls "$STOREFRONT_TMP"/soulcraft-brainy-*.tgz)"
-echo -e "${BLUE} home artifact: $(sha256sum "$SOURCE_TARBALL" | cut -d' ' -f1)${NC}"
-npm publish "$SOURCE_TARBALL" --tag "$NPM_TAG" "--@soulcraft:registry=https://registry.npmjs.org/"
-rm -rf "$STOREFRONT_TMP"
-# Brainy is the only PUBLIC @soulcraft package — verify visibility after every publish.
-npm access get status @soulcraft/brainy "--@soulcraft:registry=https://registry.npmjs.org/" || true
-# Verify the pair is byte-identical by registry-reported shasum — divergence
-# here means the storefront leg must be treated as failed, loudly. RETRIED
-# with raw curl: npmjs metadata propagates with a lag measured in minutes,
-# and a one-shot npm-view probe fired a false DIVERGENCE on 10.0.0 while a
-# raw curl of the registry document already confirmed byte-identity. The
-# probe now reads the registry JSON directly (no npm cache in the path) and
-# gives propagation up to 5 minutes before calling the pair divergent.
-NPMJS_VERIFY_ATTEMPTS=20
-NPMJS_VERIFY_INTERVAL_S=15 # 20 × 15s = 5 minutes of propagation grace
-SOURCE_SHA=$(npm view "@soulcraft/brainy@${NEW_VERSION}" dist.shasum "--@soulcraft:registry=${SOURCE_NPM_REG}" 2>/dev/null || echo "source-unavailable")
-PAIR_IDENTICAL=false
-for ((attempt = 1; attempt <= NPMJS_VERIFY_ATTEMPTS; attempt++)); do
- NPMJS_SHA=$(curl -fsSL "https://registry.npmjs.org/@soulcraft%2Fbrainy" 2>/dev/null \
- | node -e "let d='';process.stdin.on('data',c=>d+=c).on('end',()=>{try{const v=JSON.parse(d).versions[process.argv[1]];console.log(v?v.dist.shasum:'')}catch{console.log('')}})" "${NEW_VERSION}" \
- || echo "")
- if [ -n "$NPMJS_SHA" ] && [ "$SOURCE_SHA" = "$NPMJS_SHA" ]; then
- PAIR_IDENTICAL=true
- break
- fi
- echo -e "${YELLOW} … npmjs metadata not settled (attempt ${attempt}/${NPMJS_VERIFY_ATTEMPTS}: '${NPMJS_SHA:-absent}' vs '${SOURCE_SHA}'); retrying in ${NPMJS_VERIFY_INTERVAL_S}s${NC}"
- sleep "$NPMJS_VERIFY_INTERVAL_S"
-done
-if [ "$PAIR_IDENTICAL" = true ]; then
- echo -e "${GREEN}✅ Published to npmjs — byte-identical pair (shasum ${NPMJS_SHA})${NC}\n"
-else
- echo -e "${RED}❌ REGISTRY DIVERGENCE: The Source shasum ${SOURCE_SHA} != npmjs shasum ${NPMJS_SHA} after ${NPMJS_VERIFY_ATTEMPTS} attempts — investigate before announcing${NC}\n"
+ echo -e "${RED} v${NEW_VERSION} was tagged and pushed, but @soulcraftlabs/brainy@${NEW_VERSION} never became visible on the${NC}"
+ echo -e "${RED} Source registry after ${SOURCE_POLL_MAX_ATTEMPTS} attempts, ${SOURCE_POLL_INTERVAL_S}s apart. Aborting.${NC}"
exit 1
fi
@@ -267,7 +263,7 @@ fi
# and RELEASES.md are the record; this just gives The Source's UI a release page).
echo -e "${BLUE}🔟 Creating release page on The Source...${NC}"
if [ -n "${FORGEJO_RELEASE_TOKEN:-}" ]; then
- if curl -sf -X POST "https://source.soulcraft.com/api/v1/repos/soulcraft/brainy/releases" \
+ if curl -sf -X POST "https://source.soulcraft.com/api/v1/repos/soulcraftlabs/open-brainy/releases" \
-H "Authorization: token ${FORGEJO_RELEASE_TOKEN}" -H "Content-Type: application/json" \
-d "{\"tag_name\":\"v${NEW_VERSION}\",\"name\":\"v${NEW_VERSION}\",\"prerelease\":${PRERELEASE}}" >/dev/null; then
echo -e "${GREEN}✅ Release page created on The Source${NC}\n"
@@ -278,21 +274,15 @@ else
echo -e "${RED}⚠️ FORGEJO_RELEASE_TOKEN unset — no release page created; tag + CHANGELOG remain the record${NC}\n"
fi
-# Step 12: Push public docs to the soulcraft.com docs ingest door
-# (VENUE-DOCS-RELEASE-PUSH). Skips with a loud warning when
-# DOCS_INGEST_SECRET is unset; fails loudly (without undoing the publish —
-# that already happened) when a push errors, so the docs site never
-# silently trails npm.
-echo -e "${BLUE}1️⃣2️⃣ Pushing public docs to soulcraft.com/docs...${NC}"
-if node scripts/push-docs.js; then
- echo -e "${GREEN}✅ Docs push step done${NC}\n"
-else
- echo -e "${RED}❌ Docs push FAILED — soulcraft.com/docs trails npm until re-run or interim sync${NC}\n"
-fi
+# Step 12 RETIRED (2026-08-31, CORTEX-SITE-BRAINY-RENAME round 12, David-ruled):
+# soulcraft.com/docs carries the paid product's documentation only. This
+# engine's documentation home is THIS repository — README and docs/ — and the
+# site serves 301s for the slugs this rail used to push. The push script stays
+# in the tree for history; the rail no longer calls it.
+echo -e "${BLUE}Docs step: this engine documents itself in its own repo (site push retired 2026-08-31)${NC}"
echo -e "${GREEN}━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━${NC}"
echo -e "${GREEN}🎉 Release ${NEW_VERSION} complete!${NC}"
echo -e "${GREEN}━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━${NC}"
echo ""
-echo -e "📦 npm: ${BLUE}https://www.npmjs.com/package/@soulcraft/brainy/v/${NEW_VERSION}${NC}"
-echo -e "🏠 The Source: ${BLUE}https://source.soulcraft.com/soulcraft/brainy/releases/tag/v${NEW_VERSION}${NC}"
+echo -e "🏠 The Source: ${BLUE}https://source.soulcraft.com/soulcraftlabs/open-brainy/releases/tag/v${NEW_VERSION}${NC}"
diff --git a/scripts/wall-entry.mjs b/scripts/wall-entry.mjs
new file mode 100644
index 00000000..d4ec7ba5
--- /dev/null
+++ b/scripts/wall-entry.mjs
@@ -0,0 +1,504 @@
+#!/usr/bin/env node
+/**
+ * @module scripts/wall-entry
+ * @description The releases-wall entry, made mechanical. The fleet's HQ page
+ * reads one public JSON per product from the ONE releases repo on The Source
+ * (soulcraftlabs/releases, files .json at its root — shape
+ * {product, entries:[{version, date, headline, items, url, thumb?}]}), at
+ * https://source.soulcraft.com/soulcraftlabs/releases/raw/branch/main/.json.
+ * Those entries were hand-written after every release, then briefly written
+ * into this repo's own releases/.json; this script is the one door
+ * that composes an entry and lands it in the shared repo, so it is never
+ * hand-written and never forked across repos again.
+ *
+ * Two modes:
+ *
+ * 1. Generate + publish (default):
+ * node wall-entry.mjs --product --version --date \
+ * --from-changelog
+ * Derives an entry from the CHANGELOG.md entry for (headline = the
+ * entry's first bullet, items = every bullet, trimmed of its trailing
+ * commit hash), then:
+ * - clones (or, if a cached clone already exists, fetches and resets)
+ * the releases repo into a local cache directory,
+ * - prepends the entry to /.json, newest first — replacing
+ * any existing entry for the same version so a re-run is idempotent,
+ * - validates the file's shape before and after,
+ * - commits the change as "chore(wall):
" and pushes main.
+ * A failure at any step (clone, validation, commit, push, a
+ * non-fast-forward remote) exits non-zero naming the cure. Nothing is
+ * ever skipped — the wall either lands correctly or the release fails.
+ *
+ * 2. Dry run:
+ * node wall-entry.mjs --dry-run --product --version \
+ * --date --from-changelog
+ * Derives the entry exactly as above and prints it, along with the file
+ * it would be written to, but touches no clone and no remote — usable
+ * from a fresh checkout with no cache and no network.
+ *
+ * 3. Validate only (--check):
+ * node wall-entry.mjs --check --file
+ * Validates an arbitrary wall file's exact key set (top-level and
+ * per-entry), field types, and strict-descending semver ordering with
+ * no duplicates. Read-only; never writes. Exit 0 = clean, exit 1 =
+ * named violations printed to stderr.
+ *
+ * The remote and the local cache directory are each overridable
+ * (--remote / --cache-dir, or WALL_ENTRY_RELEASES_REMOTE /
+ * WALL_ENTRY_RELEASES_CACHE_DIR) so tests can point at a throwaway local
+ * bare repo and a throwaway cache directory — never the real remote or the
+ * real developer cache.
+ *
+ * No dependencies beyond the system `git` binary — CHANGELOG parsing,
+ * semver comparison, and JSON shape checking are all hand-rolled below.
+ */
+
+import { readFileSync, writeFileSync, existsSync, mkdirSync } from 'node:fs'
+import { execFileSync } from 'node:child_process'
+import { homedir } from 'node:os'
+import { dirname, join } from 'node:path'
+
+const DEFAULT_REMOTE = 'git@source.soulcraft.com:soulcraftlabs/releases.git'
+
+/** @returns {string} */
+function defaultCacheDir() {
+ const base = process.env.XDG_CACHE_HOME || join(homedir(), '.cache')
+ return join(base, 'soulcraft-releases')
+}
+
+// Required on every entry; "thumb" is optional (may be absent, or present as
+// string | null) — matching the HQ contract's {..., thumb?}.
+const ENTRY_REQUIRED_KEYS = ['version', 'date', 'headline', 'items', 'url']
+const ENTRY_OPTIONAL_KEYS = ['thumb']
+const ENTRY_ALLOWED_KEYS = [...ENTRY_REQUIRED_KEYS, ...ENTRY_OPTIONAL_KEYS]
+const FILE_KEYS = ['product', 'entries']
+
+// The public permalink pattern, by product. Every entry MUST carry an https
+// permalink: HQ's parser rejects a wall whose entries carry url: null (the
+// whole feed became unreadable on 2026-09-02). A product whose forge repo is
+// private links its PUBLIC package page on The Source instead of a release
+// page that would 404 for HQ's readers.
+const RELEASE_URL_PATTERNS = {
+ 'open-brainy': (version) => `https://source.soulcraft.com/soulcraftlabs/open-brainy/releases/tag/v${version}`,
+ 'brainy': (version) => `https://source.soulcraft.com/soulcraft/-/packages/npm/@soulcraft%2Fbrainy/${version}`,
+}
+
+/**
+ * Parse argv into a flag map. `--flag value` sets a string; `--flag` alone
+ * (end of argv, or followed by another `--flag`) sets boolean true.
+ * @param {string[]} argv
+ * @returns {Record}
+ */
+function parseArgs(argv) {
+ /** @type {Record} */
+ const args = {}
+ for (let i = 0; i < argv.length; i++) {
+ const a = argv[i]
+ if (!a.startsWith('--')) continue
+ const key = a.slice(2)
+ const next = argv[i + 1]
+ if (next === undefined || next.startsWith('--')) {
+ args[key] = true
+ } else {
+ args[key] = next
+ i++
+ }
+ }
+ return args
+}
+
+/**
+ * Print a loud, named error and exit 1. Every refusal in this script goes
+ * through here so the failure mode is always the same shape: "wall-entry: ".
+ * @param {string} message
+ * @returns {never}
+ */
+function fail(message) {
+ console.error(`wall-entry: ${message}`)
+ process.exit(1)
+}
+
+/**
+ * @param {string} version
+ * @returns {{major: number, minor: number, patch: number, pre: string | null} | null}
+ */
+function parseSemver(version) {
+ const m = /^(\d+)\.(\d+)\.(\d+)(?:-([0-9A-Za-z.-]+))?$/.exec(version)
+ if (!m) return null
+ return { major: Number(m[1]), minor: Number(m[2]), patch: Number(m[3]), pre: m[4] ?? null }
+}
+
+/**
+ * @param {string} a
+ * @param {string} b
+ * @returns {number} positive if a > b, negative if a < b, 0 if equal.
+ */
+function compareSemver(a, b) {
+ const pa = parseSemver(a)
+ const pb = parseSemver(b)
+ if (!pa || !pb) throw new Error(`cannot compare non-semver versions "${a}" vs "${b}"`)
+ if (pa.major !== pb.major) return pa.major - pb.major
+ if (pa.minor !== pb.minor) return pa.minor - pb.minor
+ if (pa.patch !== pb.patch) return pa.patch - pb.patch
+ if (pa.pre === pb.pre) return 0
+ if (pa.pre === null) return 1 // a release outranks any prerelease of the same core version
+ if (pb.pre === null) return -1
+ return pa.pre < pb.pre ? -1 : pa.pre > pb.pre ? 1 : 0
+}
+
+/**
+ * Validate a wall file's full shape: top-level keys ("product", "entries" —
+ * no more, no less), per-entry keys and field types ("thumb" optional), and
+ * strict-descending semver ordering with no duplicates. Collects every
+ * violation instead of failing on the first, so a caller reports the whole
+ * picture in one pass.
+ * @param {unknown} data
+ * @returns {string[]} Violation messages; empty means the file is clean.
+ */
+function validateShape(data) {
+ /** @type {string[]} */
+ const errors = []
+
+ if (typeof data !== 'object' || data === null || Array.isArray(data)) {
+ return ['top level: expected a JSON object']
+ }
+ const obj = /** @type {Record} */ (data)
+
+ const topKeys = Object.keys(obj)
+ const missingTop = FILE_KEYS.filter((k) => !(k in obj))
+ const extraTop = topKeys.filter((k) => !FILE_KEYS.includes(k))
+ if (missingTop.length) errors.push(`top level: missing key(s) ${missingTop.join(', ')}`)
+ if (extraTop.length) errors.push(`top level: unexpected key(s) ${extraTop.join(', ')}`)
+
+ if (typeof obj.product !== 'string' || obj.product.trim() === '') {
+ errors.push('top level: "product" must be a non-empty string')
+ }
+ if (!Array.isArray(obj.entries)) {
+ errors.push('top level: "entries" must be an array')
+ return errors // nothing further to check without an array
+ }
+
+ const entries = /** @type {unknown[]} */ (obj.entries)
+ entries.forEach((rawEntry, i) => {
+ const label = `entries[${i}]`
+ if (typeof rawEntry !== 'object' || rawEntry === null || Array.isArray(rawEntry)) {
+ errors.push(`${label}: expected an object`)
+ return
+ }
+ const entry = /** @type {Record} */ (rawEntry)
+ const keys = Object.keys(entry)
+ const missing = ENTRY_REQUIRED_KEYS.filter((k) => !(k in entry))
+ const extra = keys.filter((k) => !ENTRY_ALLOWED_KEYS.includes(k))
+ if (missing.length) errors.push(`${label}: missing key(s) ${missing.join(', ')}`)
+ if (extra.length) errors.push(`${label}: unexpected key(s) ${extra.join(', ')}`)
+
+ if (typeof entry.version !== 'string' || !parseSemver(entry.version)) {
+ errors.push(`${label}: "version" must be a semver string (got ${JSON.stringify(entry.version)})`)
+ }
+ if (typeof entry.date !== 'string' || !/^\d{4}-\d{2}-\d{2}$/.test(entry.date) || Number.isNaN(Date.parse(entry.date))) {
+ errors.push(`${label}: "date" must be a YYYY-MM-DD string (got ${JSON.stringify(entry.date)})`)
+ }
+ if (typeof entry.headline !== 'string' || entry.headline.trim() === '') {
+ errors.push(`${label}: "headline" must be a non-empty string`)
+ }
+ if (!Array.isArray(entry.items) || entry.items.length === 0 || entry.items.some((it) => typeof it !== 'string' || it.trim() === '')) {
+ errors.push(`${label}: "items" must be a non-empty array of non-empty strings`)
+ }
+ if (typeof entry.url !== 'string' || !/^https:\/\/\S+$/.test(entry.url)) {
+ errors.push(`${label}: "url" must be an https permalink — never null; HQ's parser rejects the whole feed`)
+ }
+ if ('thumb' in entry && !(entry.thumb === null || typeof entry.thumb === 'string')) {
+ errors.push(`${label}: "thumb" must be a string or null when present`)
+ }
+ })
+
+ // Ordering: newest first, strictly descending, no duplicate versions —
+ // checked only over entries whose version parsed (a bad version is
+ // already reported above; comparing it too would just be noise).
+ const versioned = entries
+ .map((e, i) => ({ i, version: /** @type {any} */ (e)?.version }))
+ .filter((e) => typeof e.version === 'string' && parseSemver(e.version))
+ for (let i = 0; i < versioned.length - 1; i++) {
+ const a = versioned[i]
+ const b = versioned[i + 1]
+ const cmp = compareSemver(a.version, b.version)
+ if (cmp === 0) {
+ errors.push(`entries[${a.i}] and entries[${b.i}]: duplicate version ${a.version}`)
+ } else if (cmp < 0) {
+ errors.push(`entries[${a.i}] (${a.version}) sits above entries[${b.i}] (${b.version}) — not newest-first`)
+ }
+ }
+
+ return errors
+}
+
+/**
+ * Extract one version's entry body from a standard-version-style CHANGELOG.md
+ * (headings `### [version](url) (date)`, followed by `- bullet (hash)` lines
+ * until the next heading or EOF).
+ * @param {string} changelog
+ * @param {string} version
+ * @returns {string[]} Bullet lines, trimmed of their leading "- " and
+ * trailing " (hash)".
+ */
+function extractChangelogBullets(changelog, version) {
+ const lines = changelog.split('\n')
+ const headingRe = /^### \[([^\]]+)\]\(.*\)\s*\(\d{4}-\d{2}-\d{2}\)\s*$/
+ let start = -1
+ for (let i = 0; i < lines.length; i++) {
+ const m = headingRe.exec(lines[i])
+ if (m && m[1] === version) {
+ start = i + 1
+ break
+ }
+ }
+ if (start === -1) {
+ fail(
+ `version ${version} has no CHANGELOG entry yet — run this after the CHANGELOG step composes "### [${version}]", not before`,
+ )
+ }
+ /** @type {string[]} */
+ const bullets = []
+ for (let i = start; i < lines.length; i++) {
+ if (headingRe.test(lines[i])) break // next entry starts
+ const bulletMatch = /^- (.+?)(?:\s\(([0-9a-f]{6,40})\))?$/.exec(lines[i].trim())
+ if (lines[i].trim().startsWith('- ') && bulletMatch) {
+ const text = bulletMatch[1].trim()
+ if (text) bullets.push(text)
+ }
+ }
+ if (bullets.length === 0) {
+ fail(`version ${version}'s CHANGELOG entry has no bullets to derive a headline/items from`)
+ }
+ return bullets
+}
+
+/**
+ * Derive a wall entry from a CHANGELOG.md.
+ * @param {{product: string, version: string, date: string, changelogPath: string, url?: string, thumb?: string | null}} opts
+ * @returns {{version: string, date: string, headline: string, items: string[], url: string, thumb: string | null}}
+ */
+function deriveEntry({ product, version, date, changelogPath, url, thumb }) {
+ if (!parseSemver(version)) fail(`--version "${version}" is not a semver string`)
+ if (!/^\d{4}-\d{2}-\d{2}$/.test(date) || Number.isNaN(Date.parse(date))) {
+ fail(`--date "${date}" is not a YYYY-MM-DD date`)
+ }
+ if (!existsSync(changelogPath)) fail(`--from-changelog "${changelogPath}" does not exist`)
+
+ const changelog = readFileSync(changelogPath, 'utf8')
+ const items = extractChangelogBullets(changelog, version)
+ const headline = items[0]
+
+ const pattern = RELEASE_URL_PATTERNS[product]
+ if (url === undefined && pattern === undefined) {
+ throw new Error(`wall-entry: no permalink pattern for product "${product}" — add one to RELEASE_URL_PATTERNS or pass --url; entries never carry url: null`)
+ }
+ const resolvedUrl = url !== undefined ? url : pattern(version)
+ const resolvedThumb = thumb !== undefined ? thumb : null
+
+ return { version, date, headline, items, url: resolvedUrl, thumb: resolvedThumb }
+}
+
+/**
+ * Load and shape-validate a wall file.
+ * @param {string} filePath
+ * @returns {Record}
+ */
+function loadWallFile(filePath) {
+ if (!existsSync(filePath)) fail(`"${filePath}" does not exist`)
+ /** @type {unknown} */
+ let data
+ try {
+ data = JSON.parse(readFileSync(filePath, 'utf8'))
+ } catch (err) {
+ fail(`"${filePath}" is not valid JSON: ${/** @type {Error} */ (err).message}`)
+ }
+ const errors = validateShape(data)
+ if (errors.length) {
+ fail(`"${filePath}" fails shape validation —\n ${errors.join('\n ')}`)
+ }
+ return /** @type {Record} */ (data)
+}
+
+/**
+ * Run a git command, throwing an Error whose message is git's own stderr
+ * (trimmed) on failure — every caller wraps this to name the cure.
+ * @param {string[]} args
+ * @param {string} cwd
+ * @returns {string} stdout, trimmed.
+ */
+function git(args, cwd) {
+ try {
+ return execFileSync('git', args, { cwd, encoding: 'utf8', stdio: ['ignore', 'pipe', 'pipe'] }).trim()
+ } catch (err) {
+ const stderr = /** @type {any} */ (err).stderr
+ const message = (typeof stderr === 'string' && stderr.trim()) || /** @type {Error} */ (err).message
+ throw new Error(message)
+ }
+}
+
+/**
+ * Ensure a clean, up-to-date local clone of the releases repo at
+ * `cacheDir`, checked out on `main` — cloning fresh if `cacheDir` has no
+ * `.git`, otherwise fetching and hard-resetting onto `origin/main` (so a
+ * stray local commit or edit left by a previous failed run can never leak
+ * into the next one).
+ * @param {string} remote
+ * @param {string} cacheDir
+ */
+function ensureReleasesClone(remote, cacheDir) {
+ if (existsSync(join(cacheDir, '.git'))) {
+ try {
+ git(['remote', 'set-url', 'origin', remote], cacheDir)
+ git(['fetch', '--prune', 'origin'], cacheDir)
+ git(['checkout', 'main'], cacheDir)
+ git(['reset', '--hard', 'origin/main'], cacheDir)
+ git(['clean', '-fd'], cacheDir)
+ } catch (err) {
+ fail(
+ `cannot refresh the cached releases checkout at "${cacheDir}" from "${remote}" — ${/** @type {Error} */ (err).message}\n` +
+ ` cure: delete "${cacheDir}" and re-run so it re-clones from scratch, or confirm SSH access with "ssh -T git@source.soulcraft.com"`,
+ )
+ }
+ return
+ }
+
+ mkdirSync(dirname(cacheDir), { recursive: true })
+ try {
+ git(['clone', remote, cacheDir], dirname(cacheDir))
+ } catch (err) {
+ fail(
+ `cannot clone "${remote}" — ${/** @type {Error} */ (err).message}\n` +
+ ` cure: confirm SSH access with "ssh -T git@source.soulcraft.com" and that the soulcraftlabs/releases repo exists yet`,
+ )
+ }
+ try {
+ git(['checkout', 'main'], cacheDir)
+ } catch (err) {
+ fail(
+ `cloned "${remote}" into "${cacheDir}" but could not check out "main" — ${/** @type {Error} */ (err).message}\n` +
+ ` cure: confirm the releases repo's default branch is named "main"`,
+ )
+ }
+}
+
+/**
+ * Prepend `entry` to the wall at `/.json`, replacing any
+ * existing entry for the same version (idempotent re-runs), validating
+ * before and after, committing, and pushing — or refusing loudly, naming
+ * the cure, at whichever step fails.
+ * @param {{version: string, date: string, headline: string, items: string[], url: string, thumb: string | null}} entry
+ * @param {string} product
+ * @param {string} remote
+ * @param {string} cacheDir
+ */
+function publishEntry(entry, product, remote, cacheDir) {
+ ensureReleasesClone(remote, cacheDir)
+
+ const filePath = join(cacheDir, `${product}.json`)
+ if (!existsSync(filePath)) {
+ fail(
+ `"${filePath}" does not exist in the releases repo — cure: seed "${product}.json" at the repo root first (it must exist before any release rail can prepend to it)`,
+ )
+ }
+ const wall = loadWallFile(filePath)
+
+ if (wall.product !== product) {
+ fail(`"${filePath}" has product "${wall.product}", but --product "${product}" was given — refusing a cross-product write`)
+ }
+
+ const replacing = wall.entries.some((e) => e.version === entry.version)
+ wall.entries = [entry, ...wall.entries.filter((e) => e.version !== entry.version)]
+
+ const postErrors = validateShape(wall)
+ if (postErrors.length) {
+ fail(`the entry for ${entry.version} would leave "${filePath}" invalid —\n ${postErrors.join('\n ')}`)
+ }
+
+ writeFileSync(filePath, JSON.stringify(wall, null, 2) + '\n', 'utf8')
+
+ const status = git(['status', '--porcelain', '--', `${product}.json`], cacheDir)
+ if (status === '') {
+ console.log(`wall-entry: "${product}.json" already carries an identical entry for ${entry.version} — nothing to commit or push`)
+ return
+ }
+
+ try {
+ git(['add', `${product}.json`], cacheDir)
+ git(['commit', '-m', `chore(wall): ${product} ${entry.version}`], cacheDir)
+ } catch (err) {
+ fail(`cannot commit the wall entry in "${cacheDir}" — ${/** @type {Error} */ (err).message}\n cure: inspect "${cacheDir}" by hand and re-run once its git state is clean`)
+ }
+
+ try {
+ git(['push', 'origin', 'main'], cacheDir)
+ } catch (err) {
+ fail(
+ `push to "${remote}" failed (likely a non-fast-forward — another release landed on main first) — ${/** @type {Error} */ (err).message}\n` +
+ ` cure: re-run this release step; it re-fetches and resets onto the latest origin/main before retrying`,
+ )
+ }
+
+ const sha = git(['rev-parse', 'HEAD'], cacheDir)
+ console.log(
+ `wall-entry: ${replacing ? 'replaced' : 'wrote'} v${entry.version} in "${product}.json" (${wall.entries.length} entries, newest first) — pushed ${sha} to ${remote} main`,
+ )
+}
+
+function main() {
+ const args = parseArgs(process.argv.slice(2))
+
+ if (args.check) {
+ const filePath = /** @type {string | undefined} */ (args.file)
+ if (!filePath) fail('--check needs --file ')
+ const wall = loadWallFile(/** @type {string} */ (filePath))
+ console.log(`wall-entry --check: "${filePath}" OK — product "${wall.product}", ${wall.entries.length} entries, newest-first, no duplicates`)
+ process.exit(0)
+ }
+
+ // Generate mode (default, also covers --dry-run): --product, --version,
+ // --date, --from-changelog required.
+ const product = /** @type {string | undefined} */ (args.product)
+ const version = /** @type {string | undefined} */ (args.version)
+ const date = /** @type {string | undefined} */ (args.date)
+ const fromChangelog = /** @type {string | undefined} */ (args['from-changelog'])
+
+ const missing = []
+ if (!product) missing.push('--product')
+ if (!version) missing.push('--version')
+ if (!date) missing.push('--date')
+ if (!fromChangelog) missing.push('--from-changelog')
+ if (missing.length) {
+ fail(
+ `missing required flag(s): ${missing.join(', ')}\n` +
+ 'Usage:\n' +
+ ' wall-entry.mjs --product --version --date --from-changelog [--dry-run]\n' +
+ ' wall-entry.mjs --check --file ',
+ )
+ }
+
+ const urlArg = args.url === true ? undefined : /** @type {string | undefined} */ (args.url)
+ const thumbArg = args.thumb === true ? undefined : /** @type {string | undefined} */ (args.thumb)
+
+ const entry = deriveEntry({
+ product: /** @type {string} */ (product),
+ version: /** @type {string} */ (version),
+ date: /** @type {string} */ (date),
+ changelogPath: /** @type {string} */ (fromChangelog),
+ url: urlArg,
+ thumb: thumbArg,
+ })
+
+ const remote = /** @type {string} */ (args.remote ?? process.env.WALL_ENTRY_RELEASES_REMOTE ?? DEFAULT_REMOTE)
+ const cacheDir = /** @type {string} */ (args['cache-dir'] ?? process.env.WALL_ENTRY_RELEASES_CACHE_DIR ?? defaultCacheDir())
+
+ if (args['dry-run']) {
+ console.log(`wall-entry --dry-run: would write to "${join(cacheDir, `${product}.json`)}" in ${remote} (main), pushed as "chore(wall): ${product} ${version}"`)
+ console.log(JSON.stringify(entry, null, 2))
+ process.exit(0)
+ }
+
+ publishEntry(entry, /** @type {string} */ (product), remote, cacheDir)
+}
+
+main()
diff --git a/src/brainy.ts b/src/brainy.ts
index 5a181535..b8eb7f56 100644
--- a/src/brainy.ts
+++ b/src/brainy.ts
@@ -15,6 +15,7 @@ import { JsHnswVectorIndex } from './hnsw/hnswIndex.js'
import { createStorage, resolveFilesystemRoot } from './storage/storageFactory.js'
import type { StorageOptions } from './storage/storageFactory.js'
import { rebuildCounts } from './utils/rebuildCounts.js'
+import { jsonSafeIndexMetadata } from './utils/jsonSafeIndexMetadata.js'
import type { MetadataWriteBuffer } from './utils/metadataWriteBuffer.js'
import { BaseStorage } from './storage/baseStorage.js'
import {
@@ -25,6 +26,7 @@ import {
} from './storage/brainFormat.js'
import type { BrainFormat } from './storage/brainFormat.js'
import { StorageAdapter, Vector, DistanceFunction, EmbeddingFunction, GraphVerb, STANDARD_ENTITY_FIELDS } from './coreTypes.js'
+import { isZeroNormVector } from './utils/distance.js'
import type { HNSWNoun, HNSWNounWithMetadata, HNSWVerbWithMetadata, EntityVisibility } from './coreTypes.js'
import {
defaultEmbeddingFunction,
@@ -103,12 +105,7 @@ import {
UpdateNounMetadataOperation,
UpdateVerbMetadataOperation,
DeleteNounMetadataOperation,
- DeleteVerbMetadataOperation,
- UpdateInMetadataIndexOperation,
- UpdateVerbInGraphIndexOperation,
- metadataUpdateOpProvider,
- graphUpdateOpProvider,
- assertUpdateCapabilityCoherent
+ DeleteVerbMetadataOperation
} from './transaction/operations/index.js'
import {
BaseOperationalMode,
@@ -201,8 +198,13 @@ import {
} from './events/changeFeed.js'
import { isDeterministicEmbedMode } from './embeddings/deterministicEmbedMode.js'
import { GenerationConflictError, StoreInconsistentError } from './db/errors.js'
-import { BrainyError, GraphIndexNotReadyError, MetadataIndexNotReadyError, MigrationInProgressError, VectorIndexNotReadyError, ProviderCapabilityMismatchError } from './errors/brainyError.js'
-import { assessIndexReadiness } from './utils/indexReadiness.js'
+import { BrainyError, GraphIndexNotReadyError, MetadataIndexNotReadyError, MigrationInProgressError, VectorIndexNotReadyError } from './errors/brainyError.js'
+import {
+ assessIndexReadiness,
+ assessProviderHealth,
+ assessProviderRebuild,
+ describeRebuildProgress
+} from './utils/indexReadiness.js'
import { reconstructNounWrapper } from './db/factLog.js'
import { asBrainyFieldRefusal } from './db/fieldAddressing.js'
import {
@@ -400,6 +402,15 @@ interface PlannedTransact {
* marker outlives its write.
*/
markerRecords: FactMarkerRecord[]
+ /**
+ * Ids the batch's `{ op: 'update' }` unvector door (`vector: []`) needs to
+ * decrement on the vectored-noun ledger — consumed by `transact()` with a
+ * proper `await this.storage.noteVectorUnlanded?.(id)` per id, AFTER
+ * `commitTransaction` resolves (never for a rejected batch). Kept separate
+ * from `postCommit` (`Array<() => void>`, called synchronously, fire-and-
+ * forget) because the ledger hook is async and must be awaited.
+ */
+ vectorUnlands: string[]
}
/**
@@ -520,6 +531,19 @@ export class Brainy implements BrainyInterface {
private static sigintListener?: () => void
private static beforeExitListener?: () => void
+ /** True while the `beforeExit` pass is running its flushes. Node re-emits
+ * 'beforeExit' after every loop drain and that pass schedules async work, so
+ * a second emit can arrive on top of the first; it returns instead of
+ * stacking a parallel pass. NOT a one-shot: every genuine drain still gets a
+ * flush. See {@link registerShutdownHooks}. */
+ private static beforeExitFlushInFlight = false
+
+ /** Whether the drained-event-loop notice has been printed for this
+ * registration cycle. Printed ONCE — `console.log` to a pipe is itself
+ * event-loop work, so narrating on every emit would keep the loop turning
+ * and narrate forever. Reset by {@link deregisterShutdownHooksIfIdle}. */
+ private static beforeExitNarrated = false
+
/** Poll cadence (ms) for the migration LOCK when a provider exposes no
* event-driven `whenMigrationComplete()` signal. See {@link awaitMigrationLock}. */
private static readonly MIGRATION_POLL_INTERVAL_MS = 250
@@ -639,8 +663,6 @@ export class Brainy implements BrainyInterface {
/** One-shot guard so the degraded-reads warning fires once per degraded window
* (reset when the degraded state clears). See {@link warnIfReadsDegraded}. */
private _degradedReadWarned = false
- /** One-shot guard so the metadata cold-open consistency probe runs once per brain. */
- private _metadataConsistencyProbed = false
/** Graph-adjacency cold-load consistency: verified-live this session (one-shot). */
private _graphAdjacencyVerified = false
/** Re-entrancy guard: a verify (rebuild → reads) is in flight. */
@@ -737,9 +759,71 @@ export class Brainy implements BrainyInterface {
// Write acks NEVER await it; a failed background flush is LOUD and re-armed.
private _persistDirtyWrites = 0
private _persistLastFlushAt = Date.now()
+ /**
+ * Whether a write has been committed since the last flush that ran. THE
+ * ENGINE DOES NO PERIODIC WORK WITHOUT A CAUSE: a brain nobody has written
+ * to has nothing to make durable, and a flush over it must cost nothing and
+ * say nothing. Before this, a flush called every provider, stamped the
+ * watermarks, persisted the generation counter and re-stamped the entity
+ * tree whether or not anything had changed — roughly 28 writes for a store
+ * that had not moved.
+ *
+ * WHAT THIS DOES NOT EXPLAIN, stated so nobody reads it as solved: a
+ * production process holding 21 brains printed "All indexes flushed to disk
+ * in 216-601ms" per brain every ~35s and idled at 1.26 cores with no writes
+ * for ten minutes. This engine's cadence is WRITE-DRIVEN — every trigger
+ * runs through noteWriteForPersistence, which only a committed write calls —
+ * so something was calling flush() on those brains, and this gate makes such
+ * a call free rather than accounting for it. The caller is still unidentified.
+ */
+ private _dirtySinceLastFlush = false
private _persistIdleTimer: ReturnType | null = null
private _persistBackgroundFlight: Promise | null = null
+ /**
+ * FLUSH IS SINGLE-FLIGHT, AND THE QUEUE IS ONE DEEP. `_flushInFlight` is the
+ * flush body actually running; `_flushFollowUp` is the AT MOST ONE flush
+ * queued behind it. Every caller — the write cadence, the cross-process
+ * flush-request watcher, an application calling `flush()` directly — either
+ * runs (nothing in flight), or joins the single queued follow-up.
+ *
+ * WHY A FOLLOW-UP RATHER THAN JOINING THE RUNNING FLUSH: a caller flushes to
+ * make ITS writes durable, and those writes may have landed after the
+ * running flush read its state. Joining would return "flushed" over data
+ * that was never persisted. Chaining one follow-up costs nothing when there
+ * is nothing new (a clean brain's flush returns immediately — see
+ * `_dirtySinceLastFlush`) and is correct when there is.
+ *
+ * MEASURED, in the production shutdown this was written for: two
+ * "Flushing Brainy indexes and caches to disk..." runs overlapping 3s
+ * apart on one brain, their walls growing 295ms → 4.9s as they contended
+ * for the same providers.
+ *
+ * THE WAITER IS SETTLED BY THE MACHINE, NEVER BY A PROMISE CHAIN. The queue
+ * is a BARE DEFERRED (`_flushQueued` plus its `_flushQueuedSettle` handles),
+ * not `leader.then(() => this.flush())`. A chained follow-up is settled only
+ * by resolving the very promise the leader is being awaited through, so the
+ * moment anything inside a flush body awaits `flush()` the graph closes on
+ * itself and NOBODY resolves — an unbounded hang, not a slow flush. Here the
+ * leader never awaits the queue: its `finally` PROMOTES the waiter to a new
+ * leader and settles the deferred from that run, and the leader's own
+ * promise settles without waiting for it. Every exit — the leader
+ * resolving, the leader REJECTING, the promoted run rejecting — runs the
+ * same promotion, so a queued caller is always settled exactly once.
+ */
+ private _flushInFlight: Promise | null = null
+ private _flushQueued: Promise | null = null
+ private _flushQueuedSettle: {
+ resolve: () => void
+ reject: (error: unknown) => void
+ } | null = null
+ /** Flush bodies that got past the single-flight gate (pinned by tests). */
+ private _flushBodyRuns = 0
+ /** Flush bodies running right now, and the high-water mark — which the
+ * single-flight law requires to stay at 1 (pinned by tests). */
+ private _flushBodiesActive = 0
+ private _flushConcurrencyPeak = 0
+
// DEFERRED EMBEDDING (MT5): pending markers are LOG RECORDS — an
// embed.pending record rides the deferred write's own commit fact and
// embed.landed rides the landing commit; this set is the in-memory
@@ -749,6 +833,59 @@ export class Brainy implements BrainyInterface {
private _pendingEmbedIds = new Set()
private _embedWorkerFlight: Promise | null = null
+ /**
+ * Ids cleared from {@link _pendingEmbedIds} with NO durable disarming record
+ * behind them — today exactly one case: a pending row that still EXISTS but
+ * carries no embeddable data, which the worker reaps in memory only. The log
+ * still says those ids are pending, so the pending-embed CHECKPOINT must
+ * carry them: the checkpoint's contract is "as of generation G the LOG's
+ * pending set was exactly this list", and a checkpoint that quietly dropped
+ * an id the log still arms would make the bounded fold disagree with a full
+ * fold from generation 1 — the one divergence that could lose a vector.
+ * Bounded by the number of such rows; an id leaves when it is re-enqueued or
+ * durably disarmed.
+ */
+ private _pendingEmbedUndurableClears = new Set()
+
+ /**
+ * Pending-set transitions (enqueue/clear) since the last checkpoint attempt —
+ * the checkpoint CADENCE. One mechanism, one hardcoded default, no knob and
+ * no timer (nothing to leave running after close).
+ */
+ private _pendingEmbedCheckpointTransitions = 0
+
+ /**
+ * A checkpoint is OWED: the cadence came due (or the set drained) and no
+ * write has satisfied it yet. It stays armed across attempts the durability
+ * law refuses, so the next transition that CAN be checkpointed is.
+ */
+ private _pendingEmbedCheckpointDue = false
+
+ /** Single-flight guard for the fire-and-forget checkpoint write. */
+ private _pendingEmbedCheckpointFlight: Promise | null = null
+
+ /**
+ * What the last pending-embed recovery fold actually did — the bound it
+ * used, where it started, and how many facts it read. The narration's
+ * source, and the accounting a pin reads instead of a clock.
+ */
+ private _pendingEmbedFoldReport: {
+ bound: 'checkpoint' | 'low-water' | 'genesis'
+ fromGeneration: number
+ factsScanned: number
+ seeded: number
+ pending: number
+ } | null = null
+
+ // OPEN-PATH FIX: the background embedding-engine warm kicked off (never
+ // awaited) by `performInit()` when `eagerEmbeddings` resolves true. Stored
+ // for observability only — `embed()`/`embeddingManager.embed()` already
+ // await the engine's OWN singleton init promise internally, so nothing
+ // needs to explicitly await this field for correctness. Never rejects on
+ // its own: a `.catch` narrates the failure and swallows it so a failed
+ // warm never surfaces as an unhandled rejection.
+ private _embeddingWarmPromise: Promise | null = null
+
/** The stored log-authority switch, read once at open (default: tree). */
private _logAuthority: LogAuthorityRecord = { authority: 'tree' }
// A failed walk latches its error: retries within the cooldown rethrow it
@@ -809,11 +946,49 @@ export class Brainy implements BrainyInterface {
// applies only to instances that were never closed.
private closed = false
- // Lazy rebuild state (Production-scale lazy loading)
- // Prevents race conditions when multiple queries trigger rebuild simultaneously
- private lazyRebuildInProgress = false
+ /**
+ * THE ONE CLOSE. Set SYNCHRONOUSLY by the first `close()` call, before that
+ * call yields, and never cleared — close is terminal. Every later or
+ * concurrent caller receives this same promise, so a shutdown with two
+ * callers (a host's pool close and the engine's own signal handler) runs
+ * ONE teardown, not two.
+ *
+ * MEASURED, the day this was added: a host that owns shutdown called
+ * `close()` on every pooled store at SIGTERM while the engine's signal
+ * handler flushed the same instances in parallel and released their writer
+ * locks in its own `finally`. One store took 149s to close (148s of it
+ * silent) against 24s for its idle siblings, and the same race in a local
+ * reproduction printed `Writer fence lost … the lock file is gone` — the
+ * handler observing a lock the close it was racing had already released.
+ * Two owners of one shutdown; now there is one, whoever calls first.
+ */
+ private _closeInFlight: Promise | null = null
+
+ // Index-build-at-open state. `lazyRebuildCompleted` predates the health-gate
+ // law (it named a first-QUERY lazy rebuild) and stays for `getIndexStatus()`
+ // API compatibility, but its truth changed: a needed rebuild now runs
+ // unconditionally at open() (see `rebuildIndexesIfNeeded`), never deferred to
+ // a read, so this simply flips true once that open-time step has run.
+ // `lazyRebuildInProgress` / `lazyRebuildPromise` (the first-query rebuild's
+ // concurrency guard) are retired with the lazy-build path they served —
+ // `ensureIndexesLoaded()` is a read-time CHECK now, never a build.
private lazyRebuildCompleted = false
- private lazyRebuildPromise: Promise | null = null
+
+ // Read-gate narration dedup: a degraded-but-serving or not-ready health
+ // report narrates via prodLog.warn ONCE per (provider, report.generation) —
+ // never once per read. Keyed on the provider instance itself.
+ /**
+ * The last health narration emitted per provider, keyed by its CONTENT.
+ *
+ * This used to dedupe on the provider's `generation` counter, which bumps on
+ * every ledger mutation and every rebuild boundary — so a provider that
+ * bumps its generation on routine work re-emitted the same unchanged health
+ * line on every read that consulted it, and a provider that never bumped
+ * could suppress a line whose reasons had genuinely changed. The dedupe key
+ * is now what the line SAYS: an unchanged verdict is silent however the
+ * generation moves, and a changed verdict is always heard.
+ */
+ private _lastNarratedHealth = new Map()
constructor(config?: BrainyConfig) {
// The reserved-field write policy died with the field-addressing law:
@@ -931,14 +1106,14 @@ export class Brainy implements BrainyInterface {
* extends FileSystemStorage`) inherit new methods Brainy adds to
* `FileSystemStorage` / `BaseStorage` automatically — `typeof` walks the
* prototype chain, so there's no in-package version skew to worry about as
- * long as the plugin's own dist resolves `@soulcraft/brainy` dynamically
+ * long as the plugin's own dist resolves `@soulcraftlabs/brainy` dynamically
* (which Cortex 2.2.x onward does — see
* `node_modules/@soulcraft/cor/dist/storage/mmapFileSystemStorage.js`).
*
* This helper exists for the **build/install** failure modes the import
* resolution can't catch:
* - Stale `node_modules` left over from a prior `bun install` against
- * `@soulcraft/brainy ≤7.20.x`.
+ * `@soulcraftlabs/brainy ≤7.20.x`.
* - Lockfile drift pinning brainy below the version that introduced the
* method.
* - Docker layer caches that reuse a `node_modules` from an earlier image.
@@ -984,6 +1159,17 @@ export class Brainy implements BrainyInterface {
}
}
+ /**
+ * Factory hook for the generation store, so an engine built on top of this
+ * reference implementation can substitute a `GenerationStore` that keeps
+ * the same behavioural contract (for example, one backed by a native
+ * implementation) — overriding it never changes this engine's own
+ * behaviour, since the default implementation is unchanged.
+ */
+ protected createGenerationStore(storage: BaseStorage): GenerationStore {
+ return new GenerationStore(storage)
+ }
+
/**
* Initialize Brainy.
*
@@ -1077,6 +1263,86 @@ export class Brainy implements BrainyInterface {
configureLogger({ level: LogLevel.DEBUG }) // Enable verbose logging
}
+ // OPEN-PATH NARRATION: phase timing across the five named stretches of
+ // init — storage init / generation-store open+fold / index init+gate /
+ // VFS bootstrap / embedding-warm-started. Each `markPhase()` call records
+ // elapsed ms SINCE THE PREVIOUS checkpoint, so the buckets always sum to
+ // the pre-integration/warmOnOpen total.
+ //
+ // THE LAW THIS ENFORCES: an open is never silent for more than
+ // OPEN_HEARTBEAT_MS. A production service opening a 16 GB store logged
+ // NOTHING for three minutes and then began work — the operator could not
+ // tell a slow open from a hung one, and restarted into the same wall.
+ // Two mechanisms, both on the always-visible narration channel (the old
+ // breakdown used `prodLog.warn`, which production clamps away — that is
+ // why the three minutes were silent):
+ // - a heartbeat that names the phase currently running and its elapsed
+ // wall, every OPEN_HEARTBEAT_MS, for as long as the open lasts;
+ // - one line per phase AS IT ENDS, naming its wall and its cause, for
+ // any phase over OPEN_PHASE_NARRATE_MS.
+ // The heartbeat is unref'd and cleared in the `finally` below, so it can
+ // neither hold the process open nor outlive a failed init. It cannot fire
+ // inside a phase that blocks the event loop synchronously; such a phase
+ // must narrate its own progress (the generation-log fold does).
+ const OPEN_HEARTBEAT_MS = 5_000
+ const OPEN_PHASE_NARRATE_MS = 2_000
+ /** Phase order + what each one is paying for, quoted in its narration. */
+ const OPEN_PHASES: ReadonlyArray<{ name: string; cause: string }> = [
+ { name: 'storage-init', cause: 'opening the store and loading its count ledger' },
+ {
+ name: 'generation-store-open-fold',
+ cause: 'opening the generation store: crash-recovery replay/fold, derived-family registration, format handshake'
+ },
+ { name: 'index-init-gate', cause: 'constructing the derived indexes and gating them for serving' },
+ { name: 'vfs-bootstrap', cause: 'bootstrapping the virtual filesystem' },
+ { name: 'embedding-warm-started', cause: 'starting the background embedding warm' }
+ ]
+ const initStart = Date.now()
+ let lastPhaseCheckpoint = initStart
+ let currentPhaseIndex = 0
+ const phaseTimingsMs: Record = {}
+ const openHeartbeat: ReturnType = setInterval(() => {
+ const phase = OPEN_PHASES[currentPhaseIndex]
+ if (!phase) return
+ prodLog.narrate(
+ `[Brainy] open: still in phase ${currentPhaseIndex + 1}/${OPEN_PHASES.length} ` +
+ `"${phase.name}" after ${Math.round((Date.now() - lastPhaseCheckpoint) / 1000)}s ` +
+ `(${Math.round((Date.now() - initStart) / 1000)}s into the open) — ${phase.cause}`
+ )
+ }, OPEN_HEARTBEAT_MS)
+ if (typeof openHeartbeat.unref === 'function') openHeartbeat.unref()
+ /**
+ * Narrate one STEP inside a phase when it turns out to be expensive.
+ * A phase that costs a minute and names only itself tells an operator
+ * where to look but not what to look at; this names the step. Silent
+ * under OPEN_PHASE_NARRATE_MS, so a fast open says nothing extra.
+ */
+ const step = async (name: string, cause: string, run: () => Promise): Promise => {
+ const startedAt = Date.now()
+ try {
+ return await run()
+ } finally {
+ const elapsed = Date.now() - startedAt
+ if (elapsed >= OPEN_PHASE_NARRATE_MS) {
+ prodLog.narrate(`[Brainy] open: step "${name}" took ${elapsed}ms — ${cause}`)
+ }
+ }
+ }
+ const markPhase = (name: string): void => {
+ const now = Date.now()
+ const elapsed = now - lastPhaseCheckpoint
+ phaseTimingsMs[name] = elapsed
+ lastPhaseCheckpoint = now
+ const finished = OPEN_PHASES[currentPhaseIndex]
+ if (elapsed >= OPEN_PHASE_NARRATE_MS && finished && finished.name === name) {
+ prodLog.narrate(
+ `[Brainy] open: phase ${currentPhaseIndex + 1}/${OPEN_PHASES.length} ` +
+ `"${name}" finished in ${elapsed}ms — ${finished.cause}`
+ )
+ }
+ currentPhaseIndex++
+ }
+
try {
// Auto-detect and activate plugins BEFORE storage setup
// so plugin-provided storage factories (e.g., filesystem override from cor) are available
@@ -1137,7 +1403,7 @@ export class Brainy implements BrainyInterface {
`and the flush-request RPC are disabled for this directory. ` +
`Likely fix: clean install (\`rm -rf node_modules bun.lockb && ` +
`bun install\`) or rebuild your container image to refresh ` +
- `\`@soulcraft/brainy\` to ≥7.21. See docs/concepts/storage-adapters.md.`
+ `\`@soulcraftlabs/brainy\` to ≥7.21. See docs/concepts/storage-adapters.md.`
)
} else {
console.warn(
@@ -1148,6 +1414,12 @@ export class Brainy implements BrainyInterface {
}
}
+ // PHASE 1 of 5 — "storage init": plugin/legacy-layout bootstrap,
+ // storage adapter construction+init, the OS-limit check, and the
+ // writer-lock claim, all folded into one bucket (everything above this
+ // line since performInit started).
+ markPhase('storage-init')
+
// 8.0 generational MVCC: open the record layer BEFORE any index is
// created or loaded. Crash recovery may rewrite canonical entity files
// (restoring before-images of an uncommitted transaction), and every
@@ -1155,10 +1427,13 @@ export class Brainy implements BrainyInterface {
// guarantees indexes never observe rolled-back state. Reader-mode
// instances skip recovery (readers never write; the next writer
// repairs).
- this.generationStore = new GenerationStore(this.storage)
- const generationOpenResult = await this.generationStore.open({
- readOnly: this.config.mode === 'reader'
- })
+ this.generationStore = this.createGenerationStore(this.storage)
+ const generationOpenResult = await step(
+ 'generation-store.open',
+ 'reading the generation manifest and committed ranges, opening the fact log and the ' +
+ 'packed segment tier, and folding any crash-recovery replay',
+ () => this.generationStore.open({ readOnly: this.config.mode === 'reader' })
+ )
// The generation fact log is CANONICAL state, not a derived index — no
// sweeper, GC, or blob-lifecycle path may ever delete under it. Declare
@@ -1196,7 +1471,11 @@ export class Brainy implements BrainyInterface {
// rollup invariants against the log head + live counters. Loud on
// genuine incoherence (repairIndex heals), silent on absent/coherent,
// benign-behind refreshes at the next flush. Never blocks open.
- await this.verifyEntityTreeStamp()
+ await step(
+ 'verify-entity-tree-stamp',
+ 'comparing the entity tree\'s stamped generation and rollups against the store',
+ () => this.verifyEntityTreeStamp()
+ )
// 8.0 ⇄ native-provider version handshake: load the on-disk brain-format
// marker (`_system/brain-format.json`) into an in-memory field NOW —
@@ -1208,7 +1487,11 @@ export class Brainy implements BrainyInterface {
// them from the canonical records and then re-stamps the marker AFTER the
// rebuild verifies (non-destructive: a crash mid-rebuild leaves the old /
// absent marker, so the next open idempotently re-rebuilds).
- this._brainFormat = await readBrainFormat(this.storage)
+ this._brainFormat = await step(
+ 'read-brain-format',
+ 'reading the on-disk format marker that decides whether the derived indexes are stale',
+ () => readBrainFormat(this.storage)
+ )
this._indexEpochStale =
this._brainFormat === null || this._brainFormat.indexEpoch !== EXPECTED_INDEX_EPOCH
@@ -1219,9 +1502,19 @@ export class Brainy implements BrainyInterface {
// upgrade verifies + stamps; retained on failure. No-op for a reader, for
// non-filesystem storage, or for a brain with no persisted data.
if (this._indexEpochStale && this.config.migrationBackup && !this.isReadOnly) {
- await this.createMigrationBackupIfNeeded()
+ await step(
+ 'pre-upgrade-backup',
+ 'snapshotting the brain directory before a one-time format rebuild (migrationBackup)',
+ () => this.createMigrationBackupIfNeeded()
+ )
}
+ // PHASE 2 of 5 — "generation-store open+fold": GenerationStore
+ // construction+open (crash-recovery replay/rollback fold), the
+ // derived-family registration, the fact-scan seam, the entity-tree
+ // stamp check, the brain-format handshake, and the pre-upgrade backup.
+ markPhase('generation-store-open-fold')
+
// Provider: embeddings (reassign embedder if plugin provides one)
const embeddingProvider = this.pluginRegistry.getProvider('embeddings')
if (embeddingProvider) {
@@ -1281,11 +1574,6 @@ export class Brainy implements BrainyInterface {
entityIdMapper: entityIdMapperFactory ? entityIdMapperFactory(this.storage) : undefined,
})
}
- // Registration-time refusal (before any write can run): a provider
- // whose `capabilities` set claims 'update-op' while its instance lacks
- // `updateIndex` is a typed, loud refusal — never a silent fallback
- // discovered only at the first write.
- assertUpdateCapabilityCoherent(this.metadataIndex, 'metadata')
// Provider: graph index factory
const graphFactory = this.pluginRegistry.getProvider<(storage: StorageAdapter) => any>('graphIndex')
@@ -1300,9 +1588,6 @@ export class Brainy implements BrainyInterface {
])
this.graphIndex = graphIndex
}
- // Same registration-time refusal, graph half (see the metadata-index
- // check above).
- assertUpdateCapabilityCoherent(this.graphIndex, 'graph')
// Fact-log v2 mint seam: after-image records carry minted dense ints,
// and the ONE authority for those assignments is the metadata index's
@@ -1384,13 +1669,44 @@ export class Brainy implements BrainyInterface {
`[Brainy] Rebuilding indexes after crash recovery rolled back ` +
`${generationOpenResult.rolledBackGenerations} uncommitted transaction(s)`
)
+ // SELF-REBUILD DEFERENCE, same law as the open gate: a provider that
+ // is already rebuilding itself from canonical is doing exactly this
+ // work. Kicking a second rebuild on top of it is redundant at best.
+ // Safe by ordering: the crash-recovery fold ran in the generation
+ // store's open, BEFORE any provider was constructed, so a provider
+ // rebuilding now is reading the repaired canonical records.
+ const kick = async (leg: string, provider: { rebuild: () => Promise }) => {
+ const rebuilding = assessProviderRebuild(provider)
+ if (rebuilding) {
+ prodLog.narrate(
+ `[Brainy] crash-recovery rebuild: the ${leg} provider is already ` +
+ `${describeRebuildProgress(rebuilding)} from canonical — not kicking a second one.`
+ )
+ return
+ }
+ await provider.rebuild()
+ }
await Promise.all([
- this.metadataIndex.rebuild(),
- this.index.rebuild(),
- this.graphIndex.rebuild()
+ kick('metadata', this.metadataIndex),
+ kick('vector', this.index as unknown as { rebuild: () => Promise }),
+ kick('graph', this.graphIndex)
])
}
+ // METADATA WATERMARK CATCHUP: the JS metadata index computed its
+ // three-way watermark verdict inside metadataIndex.init() above,
+ // against the generation store's now-FINAL committed generation (the
+ // crash-recovery fold above — the durable-at-ack replay of acked
+ // writes whose canonical bytes hadn't reached disk — has already run,
+ // and any rolled-back-transaction rebuild just above already brought
+ // every index current, so the verdict is consumed here whether or not
+ // that rebuild ran). Consumed BEFORE the rebuild gate below and BEFORE
+ // this open serves any read — the cure for the class of bug where
+ // canonical get()/counts recover a crash-window write but find()
+ // keeps serving the metadata index's pre-crash state (the index
+ // flushes only periodically, not per-commit).
+ await this.consumeMetadataWatermarkVerdict(generationOpenResult.rolledBackGenerations > 0)
+
// 8.0 versioned-provider replay-gap check: a provider whose persisted
// index generation is behind the storage layer's committed generation
// replays the gap itself (post-commit applier contract) — surface the
@@ -1457,12 +1773,38 @@ export class Brainy implements BrainyInterface {
}).backfillBlobHistoryRefCountsIfNeeded()
}
- // Rebuild indexes if needed for existing data
- await this.rebuildIndexesIfNeeded()
+ // LEG C (zero-norm/unvector-door law): migrate a legacy zero-norm VFS
+ // root BEFORE the vector-leg open gate below ever compares the
+ // canonical vectored-noun count against the vector index's size — see
+ // migrateLegacyZeroNormVfsRootIfNeeded's JSDoc for why this is a safe
+ // O(1) exception to "nothing at open may scale with brain size", and
+ // why it must run here rather than waiting on VirtualFileSystem's own
+ // (VFS-instance-gated) lazy migration.
+ await this.migrateLegacyZeroNormVfsRootIfNeeded()
+
+ // Rebuild indexes if needed for existing data. Runs to completion before
+ // init() returns — there is no more first-query lazy path, so the flag
+ // below (kept for getIndexStatus() API compatibility) simply flips true
+ // once this open-time step has run.
+ await step(
+ 'rebuild-indexes-if-needed',
+ 'the derived-index gate: each family\'s readiness verdict, and any build it asks for',
+ () => this.rebuildIndexesIfNeeded()
+ )
+ this.lazyRebuildCompleted = true
// Check for pending data migrations
await this.checkMigrations()
+ // PHASE 3 of 5 — "index init+gate": provider wiring (embeddings,
+ // cache, roaring, msgpack, sort:topK, distance), HNSW/metadata/graph
+ // index construction, the eager cold-load, id-resolver + connections-
+ // codec wiring, crash-recovery index rebuild, the replay-gap check,
+ // legacy VFS blob adoption, blob-history backfill, the legacy
+ // zero-norm VFS root migration, and the rebuildIndexesIfNeeded() gate
+ // + migration check.
+ markPhase('index-init-gate')
+
// Register shutdown hooks for graceful count flushing (once globally)
if (!Brainy.shutdownHooksRegisteredGlobally) {
this.registerShutdownHooks()
@@ -1521,7 +1863,11 @@ export class Brainy implements BrainyInterface {
// Initialize VFS: Ensure VFS is ready when accessed as property
// This eliminates need for separate vfs.init() calls - zero additional complexity
this._vfs = new VirtualFileSystem(this)
- await this._vfs.init()
+ await step(
+ 'vfs.init',
+ 'creating or adopting the VFS root and wiring the path resolver',
+ () => this._vfs!.init()
+ )
this._vfsInitialized = true // Mark VFS as fully initialized
// 8.0 MVCC: infrastructure bootstrap (VFS root, etc.) is now the
@@ -1545,7 +1891,11 @@ export class Brainy implements BrainyInterface {
const storedArtifact = await this.storage
.readRawObject(LOG_AUTHORITY_PATH)
.catch(() => null)
- const authority = await readLogAuthority(this.storage)
+ const authority = await step(
+ 'read-log-authority',
+ 'reading the stored storage-authority artifact',
+ () => readLogAuthority(this.storage)
+ )
this._logAuthority = authority
if (authority.authority === 'log') {
this.generationStore.setLogDurability('at-ack')
@@ -1556,7 +1906,12 @@ export class Brainy implements BrainyInterface {
this.generationStore.getFactLog() !== null
) {
try {
- await this.adoptLogAuthority()
+ await step(
+ 'adopt-log-authority',
+ 'the adoption oracle: verifying the log against canonical before flipping this ' +
+ 'brain to durable-at-ack, and backfilling any curable divergence',
+ () => this.adoptLogAuthority()
+ )
prodLog.info(
'[Brainy] storage authority adopted at open: generation log ' +
'(fleet default; oracle green; durable-at-ack enabled)'
@@ -1595,9 +1950,22 @@ export class Brainy implements BrainyInterface {
// a deferred write's ack and its background embed DELAYED a vector;
// this is where it lands.
if (!this.isReadOnly) {
+ // Foreground, as the crash-recovery contract pins it: a reopened brain
+ // has its markers re-armed when open() returns. The low-water mark
+ // bounds this to the log's tail on any brain that has ever drained —
+ // milliseconds — so the foreground cost is the unmarked first open
+ // only, once per upgraded brain.
try {
- await this.bridgeLegacyPendingEmbedSidecars()
- await this.recoverPendingEmbedsFromLog()
+ await step(
+ 'bridge-pending-embed-sidecars',
+ 'migrating any pre-log deferred-embed marker files into the generation log',
+ () => this.bridgeLegacyPendingEmbedSidecars()
+ )
+ await step(
+ 'recover-pending-embeds',
+ 'folding the generation log\'s deferred-embed markers (from the low-water mark) into the pending set',
+ () => this.recoverPendingEmbedsFromLog()
+ )
if (this._pendingEmbedIds.size > 0) {
prodLog.info(
`[Brainy] ${this._pendingEmbedIds.size} deferred embed(s) pending from a previous ` +
@@ -1614,15 +1982,40 @@ export class Brainy implements BrainyInterface {
}
}
- // Eager embedding initialization.
+ // PHASE 4 of 5 — "VFS bootstrap": shutdown-hook registration, blob
+ // storage init, the provider-summary log, flipping `initialized`,
+ // the migration-lock wait, VFS construction+init, flipping generation
+ // stamping active, the log-authority adopt/oracle check, and
+ // pending-embed crash recovery.
+ markPhase('vfs-bootstrap')
+
+ // Eager embedding initialization — BACKGROUND WARM (open-path fix).
//
- // Adaptive default (8.0): the WASM embedding engine eagerly initializes
+ // Adaptive default (8.0): the WASM embedding engine eagerly WARMS
// during init() WHENEVER it is the active embedder — i.e. no native
// 'embeddings' provider has taken over — and the instance is a writer
// (not reader-mode) outside of unit tests. The WASM module (≈93MB with
- // the embedded model) takes 90-140s to compile on throttled CPUs; paying
- // that during boot rather than on the first embed()-driven call is the
- // right default for the overwhelmingly common single-process server.
+ // the embedded model) takes 90-140s to compile on throttled CPUs.
+ //
+ // Historically this AWAITED `embeddingManager.init()` INLINE, so every
+ // writer's open() blocked on the compile — N concurrent opens all
+ // queued on the ONE process-global singleton (an ~80x contention
+ // multiplier measured in a production restart storm: 90,017ms busy vs
+ // 1,117ms quiet). The engine only needs to be ready before the FIRST
+ // REAL embed() call, not before init() returns, so this now only
+ // STARTS the warm and moves on — init() never waits for it.
+ //
+ // No double-await needed for correctness: `this.embed()` (~line 15420)
+ // delegates to `this.embedder`, which for the default engine is
+ // `embeddingManager.getEmbeddingFunction()` → `embeddingManager.embed()`
+ // (src/embeddings/EmbeddingManager.ts). That method calls `await
+ // this.init()` FIRST, and `init()` itself serializes every concurrent
+ // caller onto ONE shared `globalInitPromise` — so the first real
+ // embed() automatically waits for whichever finishes first: this
+ // background warm (if still running) or a fresh init() (if the warm
+ // hasn't reached this code yet, e.g. `eagerEmbeddings: false`).
+ // Verified by reading both call sites; `_embeddingWarmPromise` below
+ // is stored for observability only, never re-awaited by embed().
//
// Skipped automatically when:
// - a native 'embeddings' provider is registered (it owns embeddings;
@@ -1630,8 +2023,8 @@ export class Brainy implements BrainyInterface {
// - reader-mode (readers don't embed — they query existing vectors),
// - unit-test mode (tests must stay fast and use the mock embedder).
//
- // `eagerEmbeddings: false` is the explicit override to force lazy init
- // (first-embed) even when this instance is the active embedder.
+ // `eagerEmbeddings: false` keeps meaning "no warm at all" — fully lazy,
+ // the first embed() call pays the full cost inline, same as before.
const isUnitTestMode = isDeterministicEmbedMode()
const eager = this.config.eagerEmbeddings ?? true
if (
@@ -1640,9 +2033,45 @@ export class Brainy implements BrainyInterface {
this.config.mode !== 'reader' &&
!isUnitTestMode
) {
- console.log('Eager embedding initialization enabled...')
- await embeddingManager.init()
- console.log('Embedding engine ready')
+ const warmStart = Date.now()
+ console.log('Background embedding-engine warm started (init() does not wait for it)...')
+ this._embeddingWarmPromise = embeddingManager
+ .init()
+ .then(() => {
+ prodLog.info(
+ `[Brainy] background embedding-engine warm complete in ${Date.now() - warmStart}ms`
+ )
+ })
+ .catch((err) => {
+ // Loud, never silent: a warm that fails to compile must be
+ // heard NOW, not discovered as a mystery latency spike on
+ // whichever request happens to trigger the first real embed().
+ // That first embed() call still retries init() itself (the
+ // singleton promise contract above) and surfaces its own typed
+ // error to its caller — this is the immediate, background echo.
+ prodLog.warn(
+ `[Brainy] background embedding-engine warm FAILED: ` +
+ `${(err as Error).message} — the first embed() call will retry ` +
+ `initialization and surface the error there`
+ )
+ })
+ }
+
+ // PHASE 5 of 5 — "embedding-warm-started": just the synchronous cost
+ // of kicking off the background warm above (the warm's own compile
+ // time is NOT included — that's the whole point of backgrounding it).
+ markPhase('embedding-warm-started')
+ {
+ const totalOpenMs = Date.now() - initStart
+ if (totalOpenMs > 2000) {
+ const phaseList = Object.entries(phaseTimingsMs)
+ .map(([name, ms]) => `${name}=${ms}ms`)
+ .join(', ')
+ prodLog.narrate(
+ `[Brainy] slow open: ${totalOpenMs}ms total (${phaseList}) — see the ` +
+ `phase breakdown above to find which one to investigate first`
+ )
+ }
}
// Integration Hub initialization
@@ -1695,15 +2124,15 @@ export class Brainy implements BrainyInterface {
if (error instanceof Error && (error as Error & { code?: string }).code === 'BRAINY_WRITER_LOCKED') {
throw error
}
- // Same rationale, provider-capability half: a registration-time refusal
- // (a provider's `capabilities` set lies about implementing 'update-op')
- // carries a machine-readable `.type`/`.family`/`.missingMethod` — the
- // whole point of the typed-error family — so it must not be flattened
- // into a message-only generic Error either.
- if (error instanceof ProviderCapabilityMismatchError) {
- throw error
- }
- throw new Error(`Failed to initialize Brainy: ${error}`)
+ // Wrap with the original as `cause` so the originating frame (a plugin's
+ // own file:line, e.g. a provider boot failure) survives to the caller's
+ // log — a plain string interpolation discards both stack and cause.
+ const message = error instanceof Error ? error.message : String(error)
+ throw new Error(`Failed to initialize Brainy: ${message}`, { cause: error })
+ } finally {
+ // The open is over — succeeded or failed. Stop the heartbeat here so a
+ // failed init never leaves a timer narrating a phase nobody is running.
+ clearInterval(openHeartbeat)
}
}
@@ -1714,83 +2143,204 @@ export class Brainy implements BrainyInterface {
* Critical for Cloud Run, Fargate, Lambda, and other containerized deployments.
*
* Handles:
- * - SIGTERM: Graceful termination (Cloud Run, Fargate, Lambda)
- * - SIGINT: Ctrl+C (development/local testing)
- * - beforeExit: Node.js cleanup hook (fallback)
+ * - SIGTERM: Graceful termination (Cloud Run, Fargate, Lambda) — CLOSES.
+ * - SIGINT: Ctrl+C (development/local testing) — CLOSES.
+ * - beforeExit: the event loop drained — FLUSHES, and closes NOTHING. A
+ * drained loop is not a shutdown; see {@link flushOnDrainedEventLoop}'s
+ * contract below.
*
* NOTE: Registers globally (once for all instances) to avoid MaxListenersExceededWarning
*/
private registerShutdownHooks(): void {
- const flushOnShutdown = async () => {
+ /**
+ * The signal-path shutdown. ONE OWNER PER BRAIN, AND THE PATH IS `close()`.
+ *
+ * WHAT THIS REPLACED, and why. The handler used to run its own shutdown —
+ * a parallel per-component flush, the generation store's close, a second
+ * parallel round of component closes, and a `finally` that stopped the
+ * flush-request watcher and released the writer lock. That is a SECOND
+ * teardown of the same brain, and a host application with its own SIGTERM
+ * handler (the shape every pooled deployment has) ran the FIRST one at the
+ * same moment. MEASURED in production the day this changed: a host closing
+ * seven pooled stores at SIGTERM printed "Shutdown signal received -
+ * flushing pending data...", went silent for 148s, printed "Flushed
+ * successfully (1 instance)", and the host's own close of that same store
+ * returned 1s later — 149s, against 24s for the six stores with no engine
+ * work in flight. The same race reproduced locally as
+ * `Failed to flush one Brainy instance on shutdown: Writer fence lost …
+ * the lock file is gone`: this handler observing a lock that the close it
+ * was racing had already released.
+ *
+ * SO: defer one macrotask, then per instance either STEP ASIDE (a close
+ * has begun or finished — its owner owns the flush, the markers and the
+ * lock) or `await instance.close()` — the one durable path, identical to
+ * what any caller gets. The three laws the old block carried are all
+ * satisfied by `close()`, each verified against its code:
+ *
+ * 1. PER-INSTANCE ISOLATION — kept HERE, in the per-instance try/catch
+ * below: one brain's failed close never aborts the loop over the rest.
+ * (`close()` itself is per-instance by construction.)
+ * 2. THE MARKER IS PART OF SHUTDOWN — `close()` → `closeDurableSteps()`
+ * Phase 1 awaits `this.generationStore.close()`, which persists the
+ * counter, advances the fold checkpoint and stamps the clean-shutdown
+ * marker LAST. That is the step that decides adopt-vs-fold at the next
+ * open, and it is the same call the old block made.
+ * 3. THE LOCK IS ALWAYS GIVEN UP — `close()`'s terminal releases run
+ * whether the durable steps threw or not (its contract: "TWO PARTS, AND
+ * THE SECOND IS UNCONDITIONAL"): `stopFlushRequestWatcher()` then
+ * `releaseWriterLock()`, then the VFS shutdown and the terminal
+ * `closed` flag, and only then is the original failure rethrown.
+ * `close()` releases the lock in MORE cases than the old block did — it
+ * also drains the metadata write buffer first, so no pending write can
+ * land after a successor writer claims the lock.
+ */
+ const closeOnShutdown = async () => {
console.log('Shutdown signal received - flushing pending data...')
+ // DEFER ONE MACROTASK. A host application registers its own listener on
+ // the same signal, and Node runs listeners in registration order — ours
+ // is usually first, because the brain was opened before the host wired
+ // its shutdown. Yielding once lets every other listener for this signal
+ // run its synchronous prologue, so a host that calls close() gets to be
+ // the owner. It is only a courtesy, never the safety: close()'s own
+ // single-flight gate is what makes a lost race harmless.
+ await new Promise((resolve) => setImmediate(resolve))
+
+ let closedCount = 0
+ let deferredCount = 0
+ let failedCount = 0
+ // Snapshot: close() splices Brainy.instances while we iterate.
+ for (const instance of [...Brainy.instances]) {
+ if (!instance.initialized) continue
+ // SOMEONE ELSE OWNS THIS ONE. Not a flush, not a lock release, not a
+ // component close — nothing. Touching a brain whose close is running
+ // is the whole defect this handler was rewritten for.
+ if (instance.closed || instance._closeInFlight !== null) {
+ deferredCount++
+ continue
+ }
+ try {
+ // Law 1: this try/catch is the isolation — the loop continues.
+ await instance.close()
+ closedCount++
+ } catch (error) {
+ failedCount++
+ console.error('Failed to close one Brainy instance on shutdown:', error)
+ }
+ }
+ if (closedCount > 0) {
+ console.log(`Flushed successfully (${closedCount} instance${closedCount > 1 ? 's' : ''})`)
+ }
+ if (deferredCount > 0) {
+ console.log(
+ `${deferredCount} Brainy instance${deferredCount > 1 ? 's are' : ' is'} already ` +
+ `closing — left to the caller that owns that close.`
+ )
+ }
+ if (failedCount > 0) {
+ console.error(
+ `${failedCount} Brainy instance${failedCount > 1 ? 's' : ''} did not complete shutdown — ` +
+ `their writer locks were released, but their next open will run crash recovery.`
+ )
+ }
+ }
+
+ /**
+ * THE DRAINED-EVENT-LOOP PATH. A DRAINED LOOP IS NOT A SHUTDOWN.
+ *
+ * Node emits `'beforeExit'` whenever the event loop has no REF'd work
+ * left — NOT when the process is ending, and with no signal involved. A
+ * perfectly healthy script reaches that state routinely: this engine
+ * unref's its idle and cadence timers ("an idle brain costs nothing"), so
+ * a script awaiting anything those timers drive is, for that instant,
+ * a process with no ref'd work and an open brain.
+ *
+ * MEASURED on the 11.1 rehearsal lane against a copy of a real store: the
+ * `beforeExit` listener was wired to the SIGNAL path, so after the heal
+ * phase the log printed `Shutdown signal received - flushing pending
+ * data...` and `Flushed successfully (1 instance)` with NO signal ever
+ * sent, and the script's very next `add()` threw `Brainy instance is not
+ * initialized: it was closed via close(). Create a new instance.` The
+ * engine had closed a live brain out from under a running script.
+ *
+ * SO, THE LAW: this path NEVER closes, deregisters, tears down or
+ * force-exits anything, and never releases a writer lock. It runs
+ * `flush()` — the engine's own non-closing durability door — on each live
+ * brain, and leaves every one of them open and usable.
+ *
+ * WHY flush() AND NOT NOTHING. Each claim checked against the code it
+ * names:
+ * 1. IT CANNOT CLOSE ANYTHING. `flush()` → `_flushSteps()` persists
+ * DERIVED state only: the count ledger, the metadata/graph/vector
+ * projections, the generation counter, aggregation state, the
+ * entity-tree stamp. It closes no component, deactivates no plugin,
+ * touches neither `initialized` nor `closed`, and never calls
+ * `releaseWriterLock()` — the clean-shutdown marker is written by
+ * `generationStore.close()` alone, reached only from `close()`.
+ * 2. IT CANNOT RACE A LATER WRITE INTO CORRUPTION. A background flush
+ * concurrent with live writes is the engine's ORDINARY steady state:
+ * `noteWriteForPersistence()` kicks exactly this call off an unref'd
+ * timer on every busy brain. `flush()` is single-flight with one queued
+ * follow-up, and a write landing mid-flush re-sets the dirty witness,
+ * so its work is never lost — it belongs to the next flush.
+ * 3. IT CANNOT SPIN. `flush()` on a clean brain returns without touching a
+ * provider or scheduling I/O, so the second emit does no event-loop
+ * work and the process exits. That is also why the listener is NOT
+ * self-deregistered any more: a one-shot listener spent on a spurious
+ * mid-script drain leaves the genuine end-of-script drain with nothing.
+ * 4. A FAILED FLUSH IS SURVIVABLE AND LOUD. The write path is durable at
+ * ack via the fact log; derived state is rebuildable. A throw is
+ * reported per instance and the loop continues — exactly how
+ * `kickBackgroundFlush()` already treats the same failure.
+ *
+ * The one thing lost against a closing handler is the clean-shutdown
+ * marker for a script that opens a brain and never closes it: its next
+ * open folds the log. That is the correct trade — a missing marker costs
+ * a recovery fold, closing a live brain costs the caller its brain — and
+ * the narration below names the cure.
+ */
+ const flushOnDrainedEventLoop = async () => {
+ // A second emit can land on top of the first (this pass schedules async
+ // work, the loop turns, the loop drains again). One pass at a time.
+ if (Brainy.beforeExitFlushInFlight) return
+
+ // Step aside for anyone whose close is running or done — the same
+ // ownership rule the signal path follows.
+ const live = [...Brainy.instances].filter(
+ (instance) => instance.initialized && !instance.closed && instance._closeInFlight === null
+ )
+ if (live.length === 0) return
+
+ // ONCE per registration cycle: a `console.log` to a pipe is itself
+ // event-loop work, so narrating on every emit would keep the loop
+ // turning and narrate forever.
+ if (!Brainy.beforeExitNarrated) {
+ Brainy.beforeExitNarrated = true
+ console.log(
+ `[Brainy] event loop drained with ${live.length} brain${live.length > 1 ? 's' : ''} ` +
+ `open — persisting derived state; NOTHING was closed. A drained loop is not a ` +
+ `shutdown: call close() (or send SIGTERM) when you mean one.`
+ )
+ }
+
+ Brainy.beforeExitFlushInFlight = true
try {
- let flushedCount = 0
- for (const instance of Brainy.instances) {
- if (instance.initialized) {
- // Flush all buffered data, then close to release resources (timers, handles)
- await Promise.all([
- (async () => {
- if (instance.storage && typeof instance.storage.flushCounts === 'function') {
- await instance.storage.flushCounts()
- }
- })(),
- (async () => {
- if (instance.metadataIndex && typeof instance.metadataIndex.flush === 'function') {
- await instance.metadataIndex.flush()
- }
- })(),
- (async () => {
- if (instance.graphIndex && typeof instance.graphIndex.flush === 'function') {
- await instance.graphIndex.flush()
- }
- })(),
- (async () => {
- if (instance.index && typeof instance.index.flush === 'function') {
- await instance.index.flush()
- }
- })()
- ])
- // Close components to stop timers that would prevent clean process exit
- await Promise.all([
- (async () => {
- if (instance.graphIndex && typeof instance.graphIndex.close === 'function') {
- await instance.graphIndex.close()
- }
- })(),
- (async () => {
- const index = instance.index as JsHnswVectorIndex & VectorIndexOptionalHooks
- if (index && typeof index.close === 'function') {
- await index.close()
- }
- })(),
- (async () => {
- const metadataIndex = instance.metadataIndex as MetadataIndexManager & MetadataIndexOptionalHooks
- if (metadataIndex && typeof metadataIndex.close === 'function') {
- await metadataIndex.close()
- }
- })(),
- // Release the writer lock so a successor process can take over.
- // No-op for readers and for backends without locking.
- (async () => {
- if (instance.storage && typeof instance.storage.releaseWriterLock === 'function') {
- await instance.storage.releaseWriterLock()
- }
- })(),
- // Stop the flush-request watcher to release its interval timer.
- (async () => {
- if (instance.storage && typeof instance.storage.stopFlushRequestWatcher === 'function') {
- instance.storage.stopFlushRequestWatcher()
- }
- })(),
- ])
- flushedCount++
+ for (const instance of live) {
+ try {
+ await instance.flush()
+ } catch (error) {
+ // Per-instance isolation, and never fatal: canonical data is
+ // durable at ack, so a failed derived-state flush costs the next
+ // open a rebuild — it must not cost this one its brain.
+ console.error(
+ '[Brainy] flush on a drained event loop failed for one open brain ' +
+ '(the brain stays open and usable; derived-state persistence retries at the ' +
+ 'next flush, and canonical data is unaffected):',
+ error
+ )
}
}
- if (flushedCount > 0) {
- console.log(`Flushed successfully (${flushedCount} instance${flushedCount > 1 ? 's' : ''})`)
- }
- } catch (error) {
- console.error('Failed to flush on shutdown:', error)
+ } finally {
+ Brainy.beforeExitFlushInFlight = false
}
}
@@ -1798,26 +2348,52 @@ export class Brainy implements BrainyInterface {
// kept as statics so the last live instance's close() can deregister them
// — the signal handles they hold are ref'd and would otherwise keep the
// process alive forever after every brain is closed.
+ /**
+ * Exit the process ONLY when Brainy is the sole handler for this signal.
+ *
+ * Registering a signal listener suppresses Node's default terminate
+ * behaviour, so a library that attaches one must either exit or be sure
+ * someone else will. Brainy attaching one AND exiting was the wrong half
+ * of that choice for every host application with its own graceful
+ * shutdown: both handlers run concurrently, and whichever finishes first
+ * wins — a library flush finishing before an application's close()
+ * terminated that close mid-flight, at exit code 0, with locks and
+ * markers unwritten. When the host has its own handler (listener count
+ * above our own), the host owns the exit; Brainy only makes its data
+ * durable and steps aside.
+ *
+ * THE COUNT IS TAKEN WHEN THE SIGNAL ARRIVES, not after the shutdown ran.
+ * "Is anyone else handling this signal?" is a question about the moment
+ * the signal landed. Asking afterwards reads a process that has already
+ * torn itself down: the handler now CLOSES its instances, and closing the
+ * last brain deregisters Brainy's own listeners — so a host application's
+ * single remaining listener would look like `<= 1` and get force-exited
+ * out of its own graceful shutdown, precisely the failure above.
+ *
+ * SIGNALS ONLY — NEVER `beforeExit`. The reasoning above is entirely about
+ * a signal Brainy has suppressed Node's default terminate behaviour for.
+ * `beforeExit` suppresses nothing: Node exits by itself once the loop is
+ * genuinely done, and the script that is still running when it fires is
+ * not shutting down at all. Calling this from that path would end a live
+ * script at exit code 0 mid-work. It is called from the two signal
+ * listeners below and from nowhere else.
+ */
+ const exitIfSoleShutdownOwner = (ownersWhenSignalled: number): void => {
+ if (ownersWhenSignalled <= 1) {
+ process.exit(0)
+ }
+ }
Brainy.sigtermListener = async () => {
- await flushOnShutdown()
- process.exit(0)
+ const owners = process.listenerCount('SIGTERM')
+ await closeOnShutdown()
+ exitIfSoleShutdownOwner(owners)
}
Brainy.sigintListener = async () => {
- await flushOnShutdown()
- process.exit(0)
- }
- Brainy.beforeExitListener = async () => {
- // Self-deregister FIRST: Node re-emits 'beforeExit' after every event-
- // loop drain, and this flush schedules new async work — with the
- // listener still attached, a script that never calls close() would spin
- // flush → drain → flush forever and never exit. One flush, then the
- // next drain finds no listener and the process exits.
- if (Brainy.beforeExitListener) {
- process.off('beforeExit', Brainy.beforeExitListener)
- Brainy.beforeExitListener = undefined
- }
- await flushOnShutdown()
+ const owners = process.listenerCount('SIGINT')
+ await closeOnShutdown()
+ exitIfSoleShutdownOwner(owners)
}
+ Brainy.beforeExitListener = flushOnDrainedEventLoop
process.on('SIGTERM', Brainy.sigtermListener)
process.on('SIGINT', Brainy.sigintListener)
process.on('beforeExit', Brainy.beforeExitListener)
@@ -1839,6 +2415,11 @@ export class Brainy implements BrainyInterface {
Brainy.sigtermListener = undefined
Brainy.sigintListener = undefined
Brainy.beforeExitListener = undefined
+ // A later re-init is a fresh cycle: it may narrate its own drained-loop
+ // notice, and no pass of the previous cycle can still be running (the last
+ // close() drained the flush chain).
+ Brainy.beforeExitNarrated = false
+ Brainy.beforeExitFlushInFlight = false
Brainy.shutdownHooksRegisteredGlobally = false
}
@@ -1889,6 +2470,33 @@ export class Brainy implements BrainyInterface {
return this.initialized
}
+ /**
+ * @description Whether `close()` has BEGUN on this instance — in flight or
+ * already finished. The question a shutdown owner asks: this brain's
+ * teardown belongs to whoever started it, and a second party must not flush
+ * its components or release its writer lock underneath it.
+ *
+ * True from the synchronous moment `close()` is entered, so a listener that
+ * yields a tick and comes back reads the truth, not a stale "not yet".
+ * @returns `true` once a close has started.
+ */
+ get isClosing(): boolean {
+ return this._closeInFlight !== null
+ }
+
+ /**
+ * @description Whether `close()` has FINISHED tearing this instance down —
+ * durable steps attempted, writer lock released, instance terminal. A
+ * closed brain never re-initializes; every operation on it throws.
+ *
+ * True after a close that FAILED partway, too: such a brain still holds no
+ * writer lock and still serves nothing (see {@link close}).
+ * @returns `true` once the teardown has completed.
+ */
+ get isClosed(): boolean {
+ return this.closed
+ }
+
/**
* Promise that resolves when Brainy is fully initialized and ready to use
*
@@ -2059,6 +2667,58 @@ export class Brainy implements BrainyInterface {
*/
private static readonly PENDING_EMBED_PREFIX = '_system/pending_embeds/'
+ /**
+ * Storage-root-relative path of the ADVISORY pending-embed low-water mark:
+ * `{ generation, writtenAt }`, written whenever the pending set drains to
+ * empty (and at clean close when empty). Every marker in facts at or below
+ * `generation` is consumed, so recovery scans from `generation + 1`. The
+ * mark is advisory and monotone-safe: stale-low costs a longer scan, never
+ * a lost marker; it is never required for correctness.
+ */
+ private static readonly PENDING_EMBED_LOWWATER_PATH = '_system/pending_embeds_lowwater.json'
+
+ /**
+ * Storage-root-relative path of the pending-embed CHECKPOINT:
+ * `{ generation, pending: string[], writtenAt }` — "as of durable generation
+ * G the pending set was exactly this list". Open seeds the set from `pending`
+ * and scans the log from `G + 1`, so the fold costs O(facts since G)
+ * REGARDLESS of whether the set ever drains.
+ *
+ * WHY IT REPLACES THE EMPTY-ONLY MARK AS THE BOUND. The low-water mark
+ * ({@link PENDING_EMBED_LOWWATER_PATH}) can only be written when the pending
+ * set is EMPTY, because it carries no set — it means "everything at or below
+ * G is consumed". A brain holding even ONE id that never lands (an embed that
+ * keeps failing; a row reaped in memory only and re-folded every open) never
+ * drains, so it never writes a mark, so the bound never engages on exactly
+ * the brains whose fold is expensive: every open re-reads the whole log. The
+ * checkpoint carries the set, so it needs no drain.
+ *
+ * The mark is still written and still read as the FALLBACK bound (a
+ * checkpoint that is absent, torn, or malformed degrades to it, and then to
+ * generation 1). Correctness over cost in every degradation: a stale or
+ * missing checkpoint only lengthens the scan.
+ */
+ private static readonly PENDING_EMBED_CHECKPOINT_PATH = '_system/pending_embeds_checkpoint.json'
+
+ /**
+ * Checkpoint CADENCE BASE: attempt a checkpoint every N pending-set
+ * transitions (enqueues + clears) while the brain is open, on top of the
+ * drain-to-empty and clean-close writes. Hardcoded 90th-percentile default,
+ * no knob, no timer: 64 transitions is far below the cost of the fold it
+ * bounds and far above the per-write noise floor. An attempt that cannot
+ * satisfy the durability law is SKIPPED, not forced — the next transition
+ * retries.
+ *
+ * The interval ADAPTS to the one signal that matters, the backlog's own
+ * size, because a checkpoint writes the WHOLE pending list: the interval is
+ * `max(64, ceil(|pending| / 64))`, which holds the amortized cost of the
+ * mechanism at ≤ 64 ids written per transition NO MATTER how large the
+ * backlog grows. A term that scales with the store rather than with the
+ * work is exactly the defect class this file is fixing; it must not be
+ * reintroduced by the cure.
+ */
+ private static readonly PENDING_EMBED_CHECKPOINT_EVERY = 64
+
/**
* @description Mark a deferred embed pending (MT5): the id joins the
* in-memory fast-path set and the returned `embed.pending` record is
@@ -2072,6 +2732,9 @@ export class Brainy implements BrainyInterface {
*/
private enqueuePendingEmbed(id: string): FactMarkerRecord {
this._pendingEmbedIds.add(id)
+ // Re-armed for real: any earlier in-memory-only clear is superseded.
+ this._pendingEmbedUndurableClears.delete(id)
+ this.noteEmbedCheckpointCadence()
return { type: 'embed.pending', id, enqueuedAt: Date.now() }
}
@@ -2082,10 +2745,277 @@ export class Brainy implements BrainyInterface {
* fact) — the recovery fold consumes those; nothing here touches storage.
* One honest residue: a pending row whose entity still exists but carries
* no data is reaped in memory only, so it re-folds at the next open and
- * is re-reaped there — a bounded no-op, never a lost vector.
+ * is re-reaped there — a bounded no-op, never a lost vector. That residue
+ * is the ONLY `durability: 'in-memory-only'` caller, and the checkpoint
+ * keeps carrying those ids so the bounded fold and a full fold from
+ * generation 1 agree exactly (see {@link _pendingEmbedUndurableClears}).
+ *
+ * @param id - The pending id to clear.
+ * @param durability - `'durable'` (default) when a record in the log at or
+ * below the current head disarms this id (an `embed.landed` riding the
+ * landing or unvector commit, or the row's tombstone — including the row
+ * simply not being there any more); `'in-memory-only'` when nothing in the
+ * log says so.
*/
- private clearPendingEmbed(id: string): void {
+ private clearPendingEmbed(
+ id: string,
+ durability: 'durable' | 'in-memory-only' = 'durable'
+ ): void {
this._pendingEmbedIds.delete(id)
+ if (durability === 'in-memory-only') this._pendingEmbedUndurableClears.add(id)
+ else this._pendingEmbedUndurableClears.delete(id)
+ if (this._pendingEmbedIds.size === 0) this.maybeWriteEmbedLowWater()
+ this.noteEmbedCheckpointCadence()
+ }
+
+ /**
+ * @description Advance the advisory low-water mark: called at drain-to-empty
+ * (and at clean close when empty), it records the fact log's CURRENT head —
+ * with the set empty, every marker at or below the head has been consumed,
+ * so the next open's recovery fold scans only what comes after. Fire-and-
+ * forget at the drain (close() awaits the core); loud on failure: a missed
+ * write costs the next open a longer scan, never a marker. No-op without a
+ * fact log (no durable markers exist there) and on read-only opens.
+ */
+ private maybeWriteEmbedLowWater(): void {
+ void this.writeEmbedLowWater()
+ }
+
+ /** The awaitable core of {@link maybeWriteEmbedLowWater} — close() awaits it. */
+ private async writeEmbedLowWater(): Promise {
+ if (this.isReadOnly) return
+ const log = this.generationStore ? this.generationStore.getFactLog() : null
+ if (!log) return
+ const generation = log.headGeneration()
+ if (!(generation > 0)) return
+ try {
+ await this.storage.writeRawObject(Brainy.PENDING_EMBED_LOWWATER_PATH, {
+ generation,
+ writtenAt: Date.now()
+ })
+ } catch (err) {
+ prodLog.warn(
+ `[Brainy] pending-embed low-water write failed at generation ${generation}: ` +
+ `${(err as Error).message} — the next open scans from the previous mark`
+ )
+ }
+ }
+
+ /**
+ * @description Capture a pending-embed checkpoint, or refuse.
+ *
+ * THE DURABILITY LAW, satisfied by construction. The checkpoint asserts "as
+ * of generation G the log's pending set was exactly this list", and the next
+ * open TRUSTS it: it seeds the set and never reads a fact at or below G
+ * again. So a checkpoint may only be taken at a G whose facts are DURABLE.
+ * A checkpoint taken at head H while the facts up to H are still buffered
+ * would be read back after a crash that truncated the tail — and an
+ * `embed.landed` in a truncated fact would be gone from the log while the
+ * checkpoint still recorded its id as landed. The row's landing vector went
+ * with the truncated fact, so nothing would ever re-arm it: A LOST VECTOR.
+ *
+ * The gate is therefore `0 < head ≤ committed`. `committed` is the
+ * generation manifest's watermark — the point the store's own recovery
+ * treats as truth, and the point below which `FactLog.open()` never
+ * truncates — and the group-commit flush fsyncs the log BEFORE advancing it
+ * (see `GenerationStore.flushPendingSingleOps`). So every fact at or below
+ * `head` is fsynced and survives the crash exactly as the checkpoint
+ * describes it. Anything else (a head above the manifest, no log, no
+ * generation yet, a read-only or closed brain) REFUSES: skipping a
+ * checkpoint costs a longer scan next open, never a marker.
+ *
+ * The snapshot is taken SYNCHRONOUSLY with reading the two generations — no
+ * `await` between them — so no commit and no worker step can slip between
+ * "the generation I am about to claim" and "the set I claim for it".
+ *
+ * The one asymmetry, deliberately in the safe direction: an id whose
+ * `embed.pending` record has not been appended yet (enqueued in memory, its
+ * commit still in flight) is captured as pending at G although its marker
+ * will land at G+1 or later. Over-stating pending costs one idempotent
+ * re-embed attempt; under-stating it is the shape that loses a vector, and
+ * cannot happen — every clear either rides a durable record at or below the
+ * head, or is carried in {@link _pendingEmbedUndurableClears}.
+ *
+ * @returns The checkpoint payload, or `null` when this instant cannot host
+ * one.
+ */
+ private captureEmbedCheckpoint(): { generation: number; pending: string[] } | null {
+ if (this.isReadOnly || this.closed) return null
+ const store = this.generationStore
+ if (!store) return null
+ const log = store.getFactLog()
+ if (!log) return null
+ // --- ONE SYNCHRONOUS INSTANT: no await until the return. ---
+ const generation = log.headGeneration()
+ const committed = store.committedGeneration()
+ if (!(generation > 0) || generation > committed) return null
+ const pending = new Set(this._pendingEmbedIds)
+ for (const id of this._pendingEmbedUndurableClears) pending.add(id)
+ // --- end of the synchronous instant. ---
+ return { generation, pending: [...pending] }
+ }
+
+ /**
+ * @description Fire-and-forget checkpoint write, single-flight: a burst of
+ * transitions never stacks writes, and because each attempt captures
+ * immediately before it writes, the file always ends up holding the most
+ * recently captured (generation, set) PAIR — and every such pair is
+ * independently true, so even an out-of-order landing is safe.
+ * {@link closeDurableSteps} awaits the flight before taking the final one.
+ */
+ private maybeWriteEmbedCheckpoint(): void {
+ if (this._pendingEmbedCheckpointFlight) return
+ this._pendingEmbedCheckpointFlight = this.writeEmbedCheckpoint()
+ .then((wrote) => {
+ if (wrote) {
+ this._pendingEmbedCheckpointDue = false
+ this._pendingEmbedCheckpointTransitions = 0
+ }
+ })
+ .finally(() => {
+ this._pendingEmbedCheckpointFlight = null
+ })
+ }
+
+ /**
+ * The awaitable core of {@link maybeWriteEmbedCheckpoint}.
+ * @returns `true` when a checkpoint was actually written.
+ */
+ private async writeEmbedCheckpoint(): Promise {
+ const snapshot = this.captureEmbedCheckpoint()
+ if (!snapshot) return false
+ try {
+ // Atomic on disk: the filesystem adapter's writeRawObject is tmp+rename
+ // (see BaseStorage.writeRawObject), so a crash mid-write leaves either
+ // the previous checkpoint or the new one — never a spliced file. And a
+ // file that IS unreadable (a torn gzip, invalid JSON) throws typed on
+ // read and degrades to the fallback bound; it can never parse into a
+ // partial `pending` list.
+ //
+ // The file is NOT separately fsynced, and does not need to be: losing
+ // the rename to a power cut leaves the PREVIOUS checkpoint (or none),
+ // which only lengthens the next scan. The invariant that matters is the
+ // other direction — a checkpoint that IS visible names a generation
+ // whose facts are durable — and that is established by the capture gate
+ // above, not by this write.
+ await this.storage.writeRawObject(Brainy.PENDING_EMBED_CHECKPOINT_PATH, {
+ generation: snapshot.generation,
+ pending: snapshot.pending,
+ writtenAt: Date.now()
+ })
+ return true
+ } catch (err) {
+ prodLog.warn(
+ `[Brainy] pending-embed checkpoint write failed at generation ` +
+ `${snapshot.generation}: ${(err as Error).message} — the next open scans ` +
+ `from the previous checkpoint`
+ )
+ return false
+ }
+ }
+
+ /**
+ * @description The checkpoint cadence tick: count one pending-set transition
+ * and OWE a checkpoint every {@link PENDING_EMBED_CHECKPOINT_EVERY}
+ * transitions, plus on every drain to empty. The debt stays armed across
+ * attempts the durability law refuses — during a write burst the log head
+ * legitimately runs ahead of the manifest, so the first attempt often cannot
+ * be taken — and the next transition retries it. An active brain therefore
+ * checkpoints steadily without ever forcing a flush; an idle one relies on
+ * its clean close. No timer is involved, so nothing survives close().
+ */
+ private noteEmbedCheckpointCadence(): void {
+ if (this.isReadOnly || this.closed) return
+ this._pendingEmbedCheckpointTransitions++
+ const listed = this._pendingEmbedIds.size + this._pendingEmbedUndurableClears.size
+ const every = Math.max(
+ Brainy.PENDING_EMBED_CHECKPOINT_EVERY,
+ Math.ceil(listed / Brainy.PENDING_EMBED_CHECKPOINT_EVERY)
+ )
+ if (
+ this._pendingEmbedIds.size === 0 ||
+ this._pendingEmbedCheckpointTransitions >= every
+ ) {
+ this._pendingEmbedCheckpointDue = true
+ }
+ if (this._pendingEmbedCheckpointDue) this.maybeWriteEmbedCheckpoint()
+ }
+
+ /**
+ * @description Resolve the pending-embed fold's BOUND: the checkpoint first
+ * (a set plus a generation), then the legacy low-water mark (a generation
+ * only), then genesis. Every degradation is loud and lengthens the scan
+ * rather than shortening it — a bound that could skip a marker is never
+ * derived from a value this method could not fully validate.
+ * @returns The bound's name, the first generation to scan, and the ids to
+ * seed the pending set with.
+ */
+ private async readPendingEmbedBound(): Promise<{
+ bound: 'checkpoint' | 'low-water' | 'genesis'
+ fromGeneration: number
+ seeded: string[]
+ }> {
+ let checkpointRejected: string | null = null
+ try {
+ const raw = await this.storage.readRawObject(Brainy.PENDING_EMBED_CHECKPOINT_PATH)
+ if (raw !== null && raw !== undefined) {
+ const parsed = Brainy.parsePendingEmbedCheckpoint(raw)
+ if (parsed) {
+ return {
+ bound: 'checkpoint',
+ fromGeneration: parsed.generation + 1,
+ seeded: parsed.pending
+ }
+ }
+ checkpointRejected = 'its shape is not { generation: number > 0, pending: string[] }'
+ }
+ } catch (err) {
+ // A real storage fault (EIO/EACCES/…). Corruption never lands here: the
+ // adapter maps a torn raw object to `null` AFTER logging it as a
+ // production error, so a torn checkpoint arrives as "absent" — loud at
+ // the adapter, and bounded here by the fallback below.
+ checkpointRejected = `reading it failed: ${(err as Error).message}`
+ }
+ if (checkpointRejected !== null) {
+ prodLog.warn(
+ `[Brainy] pending-embed checkpoint REFUSED (${checkpointRejected}) — falling back ` +
+ `to the low-water mark, else a full fold from generation 1`
+ )
+ }
+
+ try {
+ const mark = (await this.storage.readRawObject(Brainy.PENDING_EMBED_LOWWATER_PATH)) as {
+ generation?: number
+ } | null
+ if (mark && typeof mark.generation === 'number' && mark.generation > 0) {
+ return { bound: 'low-water', fromGeneration: mark.generation + 1, seeded: [] }
+ }
+ } catch {
+ // No mark (or unreadable): scan from 1 — correctness over cost.
+ }
+ return { bound: 'genesis', fromGeneration: 1, seeded: [] }
+ }
+
+ /**
+ * @description Validate a raw checkpoint object STRICTLY. Anything that is
+ * not exactly `{ generation: integer > 0, pending: string[] }` is refused
+ * whole — a partially-usable checkpoint is the one shape that could seed a
+ * short pending set behind a high bound, which is how a vector is lost.
+ * @param raw - The object read back from storage.
+ * @returns The validated checkpoint, or `null`.
+ */
+ private static parsePendingEmbedCheckpoint(
+ raw: unknown
+ ): { generation: number; pending: string[] } | null {
+ if (raw === null || typeof raw !== 'object' || Array.isArray(raw)) return null
+ const { generation, pending } = raw as { generation?: unknown; pending?: unknown }
+ if (typeof generation !== 'number' || !Number.isSafeInteger(generation) || generation <= 0) {
+ return null
+ }
+ if (!Array.isArray(pending) || pending.some((id) => typeof id !== 'string' || id === '')) {
+ return null
+ }
+ return { generation, pending: pending as string[] }
}
/**
@@ -2096,9 +3026,22 @@ export class Brainy implements BrainyInterface {
* survives the fold is exactly the set of acknowledged deferred writes
* whose vectors have not landed.
*
- * BOUND (honest): no durable low-water mark exists for the earliest
- * unconsumed pending, so the fold scans the log's committed facts from
- * generation 1 — a sequential read of the log at open, O(log bytes).
+ * BOUND: the scan starts after the pending-embed CHECKPOINT
+ * ({@link Brainy.PENDING_EMBED_CHECKPOINT_PATH}) — "as of durable generation
+ * G the pending set was exactly this list" — so the fold seeds the set from
+ * that list and reads only the facts after G. O(delta) whether or not the
+ * set ever drains, which is the whole point: the previous bound, the
+ * empty-only low-water mark, could not be written at all by a brain holding
+ * one id that never lands, so those brains re-read their whole log at every
+ * open. The mark remains the FALLBACK bound (checkpoint absent, torn, or
+ * malformed), and generation 1 the fallback below that — a brain opened for
+ * the first time after this change has neither a checkpoint nor, if it never
+ * drained, a mark, so it pays one full fold and writes a checkpoint on the
+ * way out. A stale bound costs a longer scan, never a marker. The fold stays
+ * on the open's foreground — the crash-recovery contract pins that a
+ * reopened brain has its markers re-armed when open() returns — and the
+ * bound is what makes that cheap. What it did (bound, start, facts read) is
+ * narrated and kept in {@link _pendingEmbedFoldReport}.
* It is SKIPPED WHOLESALE when the log has never had a v2 tail
* ({@link FactLog.hasV2History} — v1 facts cannot carry marker records),
* so pre-cutover brains pay nothing; on a mixed log the scan still reads
@@ -2111,9 +3054,13 @@ export class Brainy implements BrainyInterface {
private async recoverPendingEmbedsFromLog(): Promise {
const log = this.generationStore.getFactLog()
if (!log || !log.hasV2History()) return
- const scan = log.scanFacts({ fromGeneration: 1 })
+ const { bound, fromGeneration, seeded } = await this.readPendingEmbedBound()
+ for (const id of seeded) this._pendingEmbedIds.add(id)
+ let factsScanned = 0
+ const scan = log.scanFacts({ fromGeneration })
for await (const batch of scan.batches()) {
for (const fact of batch.facts) {
+ factsScanned++
for (const record of fact.records ?? []) {
if (record.type === 'embed.pending') {
this._pendingEmbedIds.add(record.id)
@@ -2128,6 +3075,21 @@ export class Brainy implements BrainyInterface {
}
}
}
+ this._pendingEmbedFoldReport = {
+ bound,
+ fromGeneration,
+ factsScanned,
+ seeded: seeded.length,
+ pending: this._pendingEmbedIds.size
+ }
+ // The narration channel: an operator is entitled to hear which bound
+ // applied and what it cost, on every open — that is how a bound that
+ // silently stopped engaging (the defect this replaced) becomes visible.
+ prodLog.narrate(
+ `[Brainy] pending-embed fold: ${bound} bound → scanned ${factsScanned} fact(s) ` +
+ `from generation ${fromGeneration}, seeded ${seeded.length} id(s), ` +
+ `${this._pendingEmbedIds.size} pending`
+ )
}
/**
@@ -2219,11 +3181,23 @@ export class Brainy implements BrainyInterface {
for (const id of batch) {
try {
const entity = await this.get(id, { includeVectors: true })
- if (!entity || entity.data === undefined || entity.data === null) {
- // Orphan reap: a deleted row's tombstone fact durably disarms the
- // marker at the next recovery fold; a data-less-but-present row
- // (edge case) re-folds and re-reaps — bounded, never a lost vector.
- this.clearPendingEmbed(id)
+ if (!entity) {
+ // The row is GONE. Either it was deleted — its tombstone fact
+ // durably disarms the marker, at or below the head, exactly as the
+ // fold reads it — or its create never became durable, in which case
+ // the log carries no `embed.pending` for it either. Both are durable
+ // clears: a full fold from generation 1 reaches the same answer.
+ this.clearPendingEmbed(id, 'durable')
+ continue
+ }
+ if (entity.data === undefined || entity.data === null) {
+ // Orphan reap, IN MEMORY ONLY: a data-less-but-present row (edge
+ // case) has nothing to embed, but no record in the log says so, so
+ // the fold would re-arm it. Cleared here and carried in the
+ // checkpoint (see clearPendingEmbed) — it re-folds and re-reaps at
+ // the next open exactly as before: bounded, never a lost vector,
+ // and never a checkpoint that disagrees with the log.
+ this.clearPendingEmbed(id, 'in-memory-only')
continue
}
// Hang guard: a wedged embedder must not block every later pending
@@ -2271,6 +3245,17 @@ export class Brainy implements BrainyInterface {
[{ type: 'embed.landed', id, vector: newVector }],
'system:embed-landing'
)
+ // Vectored-noun ledger: the landing commit above carries a vector
+ // write with NO accompanying metadata operation, so the
+ // saveNounMetadata(..., hasVector) seam never fires for it — the
+ // narrow storage hook is the only seam left. `oldVector.length===0`
+ // (already known for free from the pre-embed read above) proves this
+ // is a GENUINE first landing, not a re-embed of an already-vectored
+ // row (e.g. a deferred update() on a row that already had a real
+ // vector) — the latter must never double-count.
+ if (oldVector.length === 0) {
+ await this.storage.noteVectorLanded?.(id)
+ }
this.clearPendingEmbed(id)
} catch (err) {
prodLog.warn(
@@ -2417,6 +3402,12 @@ export class Brainy implements BrainyInterface {
* engine's own cadence (callers never call flush() in hot paths).
*/
private noteWriteForPersistence(): void {
+ // THE DIRTY WITNESS. Set on every committed write — both commit paths
+ // (single-op and transaction) end here, and the deferred-embed worker
+ // lands its vectors through the single-op path — BEFORE the policy check,
+ // so a `'manual'` consumer's explicit flush() is never skipped either.
+ // Cleared by a flush that actually runs; see flush().
+ this._dirtySinceLastFlush = true
const cfg = this.config.persistence
if (this.isReadOnly || cfg?.policy === 'manual') return
this._persistDirtyWrites++
@@ -2479,9 +3470,18 @@ export class Brainy implements BrainyInterface {
* toward the next trigger. A failure is LOUD and leaves the writes counted
* again — silence is not an option, and neither is a retry storm (the next
* trigger re-attempts).
+ *
+ * COALESCING LIVES IN {@link flush}, NOT HERE. A kick that arrives while a
+ * flush is running used to return without doing anything — the writes it
+ * counted waited for some LATER trigger, and this method's guard also could
+ * not coalesce the flushes it does not start (the cross-process
+ * flush-request watcher and application `flush()` calls both go straight to
+ * `flush()`; two of those overlapping is exactly what production showed).
+ * The gate in `flush()` covers every caller: this kick now either runs the
+ * flush or joins the single queued follow-up, so the writes it counted are
+ * always someone's work, and there is still never a second concurrent run.
*/
private kickBackgroundFlush(reason: 'threshold' | 'idle'): void {
- if (this._persistBackgroundFlight) return
const counted = this._persistDirtyWrites
this._persistDirtyWrites = 0
this._persistLastFlushAt = Date.now()
@@ -2785,13 +3785,43 @@ export class Brainy implements BrainyInterface {
// vector shape, is structurally impossible). The background worker
// embeds + inserts.
const deferringEmbed = params.deferEmbedding === true && !params.vector
- const vector = deferringEmbed
+ let vector = deferringEmbed
? []
: params.vector || (await this.embed(params.data))
+ // THE ZERO-NORM LAW (canonical write side): a zero-norm vector is not a
+ // vector — it never crosses an engine boundary (the engine pair's seam
+ // law). This engine's own cosine distance treats an all-zero vector
+ // safely (a zero-norm operand always scores MAXIMUM distance — see
+ // isZeroNormVector's JSDoc), but a downstream engine serving squared-
+ // euclidean distance cannot tell it apart from a legitimate origin
+ // point — a false attractor that silently darkened 150+ rows in a
+ // production deployment. The index belt (AddToVectorIndexOperation)
+ // already refuses to INDEX a zero-norm vector, but until now the
+ // CANONICAL write still persisted it and the vectored-noun ledger
+ // counted it — so a near-empty store whose only vectored row was
+ // zero-norm read "canonical vectored > 0, index size 0" and threw a
+ // not-ready error at open. Normalize HERE, before the dimension pin,
+ // the vectored-ledger flag (`SaveNounMetadataOperation`'s `hasVector`),
+ // and the index ops below ever see it, so it persists as the sanctioned
+ // "unvectored" `[]` shape instead — the canonical write still succeeds.
+ if (!deferringEmbed && vector.length > 0 && isZeroNormVector(vector)) {
+ prodLog.warn(
+ `[Brainy] add(): entity ${id} was given an explicit all-zero vector — ` +
+ `a zero-norm vector is not a vector; persisted unvectored ([]) instead.`
+ )
+ vector = []
+ }
+
// Ensure dimensions are set (a deferred-embed stub carries no dimension
// information — the worker's real vector goes through the same guard).
- if (!deferringEmbed) {
+ // Gated on `vector.length > 0`, not `!deferringEmbed`: ANY insert whose
+ // vector is the "unvectored" empty-array shape carries no dimension
+ // information, deferred or not — an explicit `vector: []` (e.g. the VFS
+ // root's zero-norm fix, see VirtualFileSystem.doInitializeRoot()) must
+ // never pin `this.dimensions` to 0, which would poison every subsequent
+ // real embed's dimension check for the life of the store.
+ if (!deferringEmbed && vector.length > 0) {
if (!this.dimensions) {
this.dimensions = vector.length
} else if (vector.length !== this.dimensions) {
@@ -2889,8 +3919,11 @@ export class Brainy implements BrainyInterface {
const runInsert: TransactionFunction = async (tx) => {
// Operation 1: Save metadata FIRST (TypeAwareStorage caching)
// isNew=true: skip pre-read for rollback (entity doesn't exist yet)
+ // hasVector: the vectored-noun ledger counts this insert iff its
+ // vector is real/non-empty (never true for a deferred embed, whose
+ // stub `vector` is `[]` — it counts later, at landing).
tx.addOperation(
- new SaveNounMetadataOperation(this.storage, id, storageMetadata, true)
+ new SaveNounMetadataOperation(this.storage, id, storageMetadata, true, vector.length > 0)
)
// Operation 2: Save vector data
@@ -2904,10 +3937,15 @@ export class Brainy implements BrainyInterface {
}, true)
)
- // Operation 3: Add to HNSW index (after entity saved). A deferred
- // embed has nothing to index yet — the worker's atomic update
- // inserts the real vector.
- if (!deferringEmbed) {
+ // Operation 3: Add to HNSW index (after entity saved). Gated on
+ // `vector.length > 0`, not `!deferringEmbed`: a deferred embed has
+ // nothing to index yet (the worker's atomic update inserts the real
+ // vector later), and an explicit `vector: []` insert (the VFS root's
+ // zero-norm fix — permanently unvectored plumbing, never embedded)
+ // is exactly the same "nothing to index yet" shape. The zero-norm
+ // BELT (a real all-zero vector, non-empty) is enforced inside
+ // AddToVectorIndexOperation itself — see its JSDoc.
+ if (vector.length > 0) {
tx.addOperation(
new AddToVectorIndexOperation(this.index, id, vector, this.indexWriteGeneration)
)
@@ -3155,6 +4193,16 @@ export class Brainy implements BrainyInterface {
}
// Route to metadata-only or full entity based on options
+ // A PROJECTED get goes through the same seam every list page uses, so a
+ // detail read of two scalars costs an index read rather than a record read.
+ // It is checked before `includeVectors` because the two are incompatible by
+ // construction: a projection returns the named fields, and a vector is not
+ // one of them unless it was named.
+ if (options?.fields !== undefined && options.fields.length > 0) {
+ const page = await this.#hydratePage([id], options.fields)
+ return page.get(id) ?? null
+ }
+
const includeVectors = options?.includeVectors ?? false // Default: metadata-only (fast)
if (includeVectors) {
@@ -3201,6 +4249,170 @@ export class Brainy implements BrainyInterface {
* const children = childIds.map(id => childrenMap.get(id)).filter(Boolean)
* ```
*/
+ /**
+ * **The projection seam** — hydrate a page of ids under an optional `fields`
+ * projection, opening the canonical record only when the index cannot serve
+ * what was asked for.
+ *
+ * Without a projection this is exactly `batchGet`, byte for byte: the whole
+ * point is that `fields` absent changes nothing.
+ *
+ * With one, the order is: ask the index for the named scalars in a single
+ * batched door; see which requested fields it actually served; and read
+ * records ONLY if something is still missing — and only to fill those fields.
+ * A page whose every requested field is index-served performs zero canonical
+ * reads, which is the whole reason the door exists.
+ *
+ * `guardFields` are fetched ALONGSIDE the projection and trimmed off before
+ * the caller sees them. find()'s index-integrity guard re-validates every row
+ * against its own predicate, and it reads the entity to do so — so a row
+ * projected down to `title` would fail a `where: { kind }` it genuinely
+ * matches, and the whole page would vanish. The fields a filter names are
+ * fields the index can serve by definition, so carrying them costs nothing
+ * and keeps the guard honest.
+ *
+ * A field nothing can supply is simply absent from the row. That is the
+ * permissive law: a projection asks "these, if you have them", and an
+ * optional field must not turn a list into an exception. It deliberately does
+ * NOT route through the strict address resolver, which throws
+ * `UnresolvableFieldError` for an unknown key — that strictness is right for
+ * `orderBy`, where a typo silently changes the order, and wrong here, where
+ * the honest answer is "this row does not have that".
+ *
+ * @param ids - Canonical ids for the page.
+ * @param fields - The projection, or undefined for the full record.
+ * @returns `id → entity`, projected when `fields` was given.
+ */
+ /**
+ * The index keys find()'s integrity guard reads when it re-validates a row.
+ *
+ * The guard calls `entityMatchesFind(entity, params)`, so a projected entity
+ * must still carry whatever the params constrain — otherwise a row that
+ * genuinely matches is dropped for lacking the evidence. These are fetched
+ * with the projection and trimmed off before the caller sees them.
+ *
+ * @param params - The find params.
+ * @returns Index keys to carry through hydration.
+ */
+ #guardFieldsFor(params: FindParams): string[] {
+ const keys: string[] = []
+ if (params.where && typeof params.where === 'object') {
+ // Top-level where keys only: nested `anyOf`/`allOf` branches are carried
+ // by their own keys when the guard walks them, and a filter whose
+ // evidence is missing keeps the row (the guard's own catch) rather than
+ // dropping it.
+ for (const key of Object.keys(params.where as Record)) {
+ if (key === 'anyOf' || key === 'allOf' || key === 'not') continue
+ keys.push(key)
+ }
+ }
+ if (params.type !== undefined) keys.push('system.type')
+ if (params.subtype !== undefined) keys.push('system.subtype')
+ if (params.service !== undefined) keys.push('system.service')
+ if (params.excludeVFS === true) keys.push('vfsType', 'isVFSEntity')
+ return keys
+ }
+
+ async #hydratePage(
+ ids: string[],
+ fields?: readonly string[],
+ guardFields: readonly string[] = []
+ ): Promise