Compare commits
No commits in common. "main" and "v8.3.0" have entirely different histories.
303 changed files with 4339 additions and 47917 deletions
|
|
@ -2,7 +2,7 @@
|
|||
|
||||
## What Is Brainy
|
||||
|
||||
@soulcraftlabs/brainy (v7.17.0) is a Universal Knowledge Protocol -- a Triple Intelligence database combining vector search, graph traversal, and metadata filtering in a single library. Published to npm as a public MIT-licensed package.
|
||||
@soulcraft/brainy (v7.17.0) is a Universal Knowledge Protocol -- a Triple Intelligence database combining vector search, graph traversal, and metadata filtering in a single library. Published to npm as a public MIT-licensed package.
|
||||
|
||||
## Core Architecture
|
||||
|
||||
|
|
|
|||
|
|
@ -1,62 +0,0 @@
|
|||
name: CI
|
||||
|
||||
# Branch pushes only — a release TAG deliberately does not re-run CI: the
|
||||
# tagged commit's CI already ran on its branch push, and the runner is
|
||||
# sequential, so tag-triggered matrix jobs (~22 min) would queue AHEAD of the
|
||||
# tag's publish-source run and starve every release (observed on 8.10.3 and
|
||||
# 9.0.0: the publish sat behind the tag's own redundant CI).
|
||||
on:
|
||||
push:
|
||||
branches: ['**']
|
||||
pull_request:
|
||||
|
||||
jobs:
|
||||
node:
|
||||
name: Node ${{ matrix.node-version }}
|
||||
runs-on: ubuntu-latest
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
node-version: ['22', '24']
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/setup-node@v4
|
||||
with:
|
||||
node-version: ${{ matrix.node-version }}
|
||||
cache: npm
|
||||
- run: npm ci
|
||||
- run: npm run test:unit
|
||||
|
||||
# The correctness plant's full gate: integration + conformance run here on
|
||||
# dedicated iron, on every push, so a release never depends on any other
|
||||
# machine being up. Verdicts live in this run's log (never inferred).
|
||||
integration:
|
||||
name: Integration + conformance (Node 22)
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/setup-node@v4
|
||||
with:
|
||||
node-version: '22'
|
||||
cache: npm
|
||||
- run: npm ci
|
||||
- run: npm run test:ci-integration
|
||||
- run: npx vitest run tests/conformance
|
||||
|
||||
bun:
|
||||
name: Bun (latest)
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/setup-node@v4
|
||||
with:
|
||||
node-version: '22'
|
||||
cache: npm
|
||||
- uses: oven-sh/setup-bun@v2
|
||||
with:
|
||||
bun-version: latest
|
||||
- run: npm ci
|
||||
# test:bun imports the built dist/, so build first.
|
||||
- run: npm run build
|
||||
# Bun as a runtime is the supported Bun story (`bun add` / `bun run`).
|
||||
- run: npm run test:bun
|
||||
|
|
@ -1,79 +0,0 @@
|
|||
name: Publish (The Source)
|
||||
|
||||
# Datacenter-side publish to The Source (source.soulcraft.com — our
|
||||
# self-hosted Forgejo; never call it "the forge", Forge is a different
|
||||
# product), moved off the laptop: an 87MB tarball PUT over the laptop's WAN
|
||||
# times out; The Source's own runner does it in seconds.
|
||||
# scripts/release.sh tags + pushes, then polls this workflow's result (npm
|
||||
# view against The Source's registry) before it ever touches the npmjs leg —
|
||||
# see the "delegation contract" in scripts/release.sh's home-publish step.
|
||||
|
||||
on:
|
||||
push:
|
||||
tags:
|
||||
- 'v*'
|
||||
|
||||
jobs:
|
||||
publish:
|
||||
name: Publish to The Source registry
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/setup-node@v4
|
||||
with:
|
||||
node-version: '22'
|
||||
cache: npm
|
||||
- run: npm ci
|
||||
- run: npm run build
|
||||
- name: Publish + readback-verify on The Source registry
|
||||
env:
|
||||
# The stored repo-settings secret keeps its historical name.
|
||||
FORGE_NPM_TOKEN: ${{ secrets.FORGE_NPM_TOKEN }}
|
||||
run: |
|
||||
set -eo pipefail
|
||||
|
||||
SOURCE_NPM_REG="https://source.soulcraft.com/api/packages/soulcraftlabs/npm/"
|
||||
VERSION="$(node -p "require('./package.json').version")"
|
||||
# The dist-tag follows the version: a prerelease (any hyphen —
|
||||
# 10.4.0-rc.1) publishes under 'rc' and must NEVER move 'latest' —
|
||||
# every consumer resolving 'latest' from this registry would otherwise
|
||||
# be handed a release candidate. Same rule scripts/release.sh applies
|
||||
# to the storefront leg.
|
||||
NPM_TAG="latest"
|
||||
case "$VERSION" in
|
||||
*-*) NPM_TAG="rc" ;;
|
||||
esac
|
||||
echo "Publishing @soulcraftlabs/brainy@${VERSION} to The Source registry (dist-tag: ${NPM_TAG})..."
|
||||
|
||||
TMPRC="$(mktemp)"
|
||||
chmod 600 "$TMPRC"
|
||||
{
|
||||
echo "@soulcraftlabs:registry=${SOURCE_NPM_REG}"
|
||||
echo "//source.soulcraft.com/api/packages/soulcraftlabs/npm/:_authToken=${FORGE_NPM_TOKEN}"
|
||||
} > "$TMPRC"
|
||||
|
||||
# The release script bumps package.json's version before it tags, so
|
||||
# this tag's checkout already carries the version being published —
|
||||
# nothing here re-derives it from the tag name.
|
||||
PUBLISH_OK=true
|
||||
if ! npm publish --tag "$NPM_TAG" --userconfig "$TMPRC"; then
|
||||
PUBLISH_OK=false
|
||||
fi
|
||||
|
||||
# Readback verify is the source of truth, run regardless of the publish
|
||||
# exit code: a benign duplicate publish (a prior run, or a mirror, already
|
||||
# landed this exact version) reports failure even though the registry
|
||||
# already holds the right content.
|
||||
LANDED_VERSION="$(npm view "@soulcraftlabs/brainy@${VERSION}" version --userconfig "$TMPRC" 2>/dev/null || echo "")"
|
||||
rm -f "$TMPRC"
|
||||
|
||||
if [ "$LANDED_VERSION" != "$VERSION" ]; then
|
||||
echo "::error::Readback verify FAILED — The Source registry reports version '${LANDED_VERSION:-<none>}', expected '${VERSION}'. This is a genuine publish failure, not a benign duplicate."
|
||||
exit 1
|
||||
fi
|
||||
|
||||
if [ "$PUBLISH_OK" = true ]; then
|
||||
echo "Published and verified @soulcraftlabs/brainy@${VERSION} on The Source registry."
|
||||
else
|
||||
echo "::warning::npm publish reported failure, but readback confirms @soulcraftlabs/brainy@${VERSION} is already live on The Source (a prior run or mirror landed it) — treating this run as successful, since the registry content is correct. Any OTHER failure mode would have failed the readback check above instead."
|
||||
fi
|
||||
40
.github/workflows/ci.yml
vendored
Normal file
40
.github/workflows/ci.yml
vendored
Normal file
|
|
@ -0,0 +1,40 @@
|
|||
name: CI
|
||||
|
||||
on:
|
||||
push:
|
||||
pull_request:
|
||||
|
||||
jobs:
|
||||
node:
|
||||
name: Node ${{ matrix.node-version }}
|
||||
runs-on: ubuntu-latest
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
node-version: ['22', '24']
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/setup-node@v4
|
||||
with:
|
||||
node-version: ${{ matrix.node-version }}
|
||||
cache: npm
|
||||
- run: npm ci
|
||||
- run: npm run test:unit
|
||||
|
||||
bun:
|
||||
name: Bun (latest)
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/setup-node@v4
|
||||
with:
|
||||
node-version: '22'
|
||||
cache: npm
|
||||
- uses: oven-sh/setup-bun@v2
|
||||
with:
|
||||
bun-version: latest
|
||||
- run: npm ci
|
||||
# test:bun imports the built dist/, so build first.
|
||||
- run: npm run build
|
||||
# Bun as a runtime is the supported Bun story (`bun add` / `bun run`).
|
||||
- run: npm run test:bun
|
||||
334
CHANGELOG.md
334
CHANGELOG.md
|
|
@ -2,340 +2,12 @@
|
|||
|
||||
All notable changes to this project will be documented in this file. See [standard-version](https://github.com/conventional-changelog/standard-version) for commit guidelines.
|
||||
|
||||
### [10.4.4](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v10.4.3...v10.4.4) (2026-08-28)
|
||||
|
||||
- fix(vfs): the old-root sweep narrates only when it has something to say (d49148e1)
|
||||
- fix(tests): the health-gate pin follows the verdict, and the VFS suite uses its own store (42e2da25)
|
||||
- Merge branch 'next/open-lazy-open-and-counts' (5ebd3b40)
|
||||
- docs: the contract manifest stands alone; public docs describe this engine only (a8c724a2)
|
||||
- docs(releases): 10.4.4 consumer notes — correctness and observability, with the performance line stated exactly (61a46927)
|
||||
- docs: measurements in public history carry numbers, not provenance (02c61636)
|
||||
- feat(open): name the two steps that hold the vfs-bootstrap phase (2cf38010)
|
||||
- fix(storage): a dead flush watch falls back to the 500ms poll, not the 30s sweep (5c22f950)
|
||||
- fix(storage): the flush watcher cannot arm twice in its async window (16d2e1a9)
|
||||
- perf(idle): the flush-request watch is event-driven; the heartbeat is observability (fb1da1c5)
|
||||
- perf(open): answer "are there any entities?" with one directory read (417ddb51)
|
||||
- perf(generations): discover generations by directory name, not by walking the log (9dd39921)
|
||||
- fix(flush): clear() and repairIndex() set the dirty witness themselves (e4c27fbc)
|
||||
- feat(open): the open names the STEP that cost the time, not just the phase (5a091cca)
|
||||
- perf(vfs): the old-root sweep runs once per store, not once per open (4a67aa0f)
|
||||
- chore: keep the generated neural stamps at main's values (c1f09723)
|
||||
- feat(contract): declare contract 1, serve three operators, refuse four by name (48802ba3)
|
||||
- fix(open): a provider rebuilding itself is a third state, not a CRITICAL (50676c02)
|
||||
- feat(open): open never waits for a provider that is rebuilding itself (131daa08)
|
||||
- perf(flush): an idle brain does no work — no periodic flush without a write (f5a6cb3f)
|
||||
- feat(repair): repairIndex narrates every phase and its receipt carries the walls (3fffd9c6)
|
||||
- fix(storage): a suspect count ledger heals itself, and counts.json is written atomically (f4e2d34b)
|
||||
- feat(open): the open narrates itself, on a channel production cannot clamp (afe08a1f)
|
||||
- fix(storage): a clean close is recorded, and the writer lock is always given up (e652162c)
|
||||
- docs: repository links point at soulcraftlabs/open-brainy — the soulcraft/brainy path becomes the native engine's repo tonight (38c3397b)
|
||||
|
||||
|
||||
### [10.4.3](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v10.4.2...v10.4.3) (2026-08-27)
|
||||
|
||||
- Merge branch 'next/open-brainy-rename' (a58372f0)
|
||||
- chore: rename to @soulcraftlabs/brainy for Open Brainy on The Source (a99b1e83)
|
||||
- docs(releases): 10.4.3 — Open Brainy's first release under the new name, same engine as 10.4.2; The Source is the one registry (9f248b24)
|
||||
|
||||
|
||||
### [10.4.2](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v10.4.2-rc.1...v10.4.2) (2026-08-27)
|
||||
|
||||
- docs(releases): 10.4.1 and 10.4.2 consumer notes; 10.4.2 is the last MIT release under this name, Open Brainy continues at @soulcraftlabs/brainy (a082e0ef)
|
||||
|
||||
|
||||
### [10.4.2-rc.1](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v10.4.1...v10.4.2-rc.1) (2026-08-27)
|
||||
|
||||
- Merge branch 'next/zero-norm-unvector-door' (9b84ef5b)
|
||||
- fix(vectors): a zero-norm vector is not a vector, canonical side included, plus the sanctioned unvector door (0de76659)
|
||||
- fix(hnsw): skip unvectored rows on rebuild; refuse empty vectors in the index (8fc553b1)
|
||||
- fix(storage): derive the canonical count ledger from identity records, stamp the derivation rule, and mark legacy-derived ledgers suspect at load (fd6b4ce4)
|
||||
- Merge branch 'next/enumeration-identity-rekey' (204d74c1)
|
||||
- fix(storage): enumeration re-keys on the identity record, not the vector leg (f8d8ce16)
|
||||
- fix(init): rethrow plugin activation failures with the original error as cause so the originating frame survives to the caller (2496e09a)
|
||||
- Merge branch 'next/vfs-root-zero-norm' (4c7b0fab)
|
||||
- fix(vfs): the VFS root never persists a zero-norm vector (c6cc0de9)
|
||||
- build: derive generated-file stamps from git commit time, not wall clock (8a5c1245)
|
||||
- Merge remote-tracking branch 'origin/release/10.4.1' (aad9e2ee)
|
||||
- docs(concepts): the serving law — a failure is graded by whether an answer could be wrong, never by the cost of the fix; reads refuse per family (2914e0eb)
|
||||
- chore(release): 10.4.1-rc.1 (7870dc40)
|
||||
|
||||
|
||||
### [10.4.1](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v10.4.0...v10.4.1) (2026-08-26)
|
||||
|
||||
- fix(reads): the read gate is per-family; a write carrying unchanged data never re-embeds (c039411e)
|
||||
- docs(guide): the docs pipeline publishes through the ingest API — the separate deploy step is retired (21e506e8)
|
||||
|
||||
|
||||
### [10.4.0](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v10.4.0-rc.4...v10.4.0) (2026-08-26)
|
||||
|
||||
- docs(releases): the 10.4.0 entry catches up to the late trains — repair routing, the vector ledger and open-gate leg, the loud config guard, the JSON-safe crossing (834149ed)
|
||||
|
||||
|
||||
### [10.4.0-rc.4](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v10.4.0-rc.3...v10.4.0-rc.4) (2026-08-25)
|
||||
|
||||
- feat(vector): the vectored-noun scalar joins the count ledger; the open gate closes the vector leg (9730835b)
|
||||
|
||||
|
||||
### [10.4.0-rc.3](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v10.4.0-rc.2...v10.4.0-rc.3) (2026-08-25)
|
||||
|
||||
- fix(update-seam): the metadata crossing never carries BigInt endpoint ints (f4780c8e)
|
||||
- Merge branch 'worktree-agent-ad3aff0dffd17a6eb' (f14da34b)
|
||||
- fix(add): empty string is real data, not a missing field (258e9042)
|
||||
- feat(vfs): implement readdir's recursive option — typed since 7.30, never read (fc516da6)
|
||||
- feat(open-path): init never gates on the embedding model; open goes concurrent; slow opens narrate (96624f40)
|
||||
|
||||
|
||||
### [10.4.0-rc.2](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v10.4.0-rc.1...v10.4.0-rc.2) (2026-08-25)
|
||||
|
||||
- test(readiness): the report helper's clock freezes — two independently-built reports compared across a millisecond tick made the plant lane red (39b916a3)
|
||||
- feat(repair): a heal:'repair' verdict routes to the provider's own incremental repair() (553e0d97)
|
||||
- fix(storage): an unknown nested storage config can never silently land on the shared default root (ddd5e719)
|
||||
- docs(release): the 10.4.0 entry, the index-health concept doc, and the API surfaces — written from the tree, not the plan (8cced871)
|
||||
- fix(plugins): the silent-degrade doors close — a broken accelerator install can never read as absent (b9ba50fb)
|
||||
- feat(recovery): the catchup verdict is consumed; verb rows go live; the metadata rebuild goes online (18f172e0)
|
||||
- feat(health): the gate reads the named report — reads refuse loudly, never rebuild; open serves before it returns; the ceremony door (f8f64780)
|
||||
|
||||
|
||||
### [10.4.0-rc.1](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v10.3.1...v10.4.0-rc.1) (2026-08-24)
|
||||
|
||||
- ci(publish): the home dist-tag follows the version — a prerelease publishes under 'rc' and never moves 'latest' (a1376e4a)
|
||||
- chore(release): --source-only — a home-only prerelease mode (The Source, never the storefront) (dcbad176)
|
||||
- test(fold-checkpoint): the ARM-AT-FLIP pin arms its crash instead of racing the pending-flush timer (4176439b)
|
||||
- fix(health): one contract for a throwing probe — heal is none, serving is not withheld; repair report gains missing/rebuilt/reason (116550eb)
|
||||
- feat(storage): the canonical count ledger — ALL-visibility scalars, unclamped totals, suspect-on-unprovable-delete (7c8c8be3)
|
||||
- fix(delete): the null-metadata skip closes — index legs run id-keyed or narrate, never silently strand postings (607e9f54)
|
||||
- feat(repair): repairIndex returns the per-family receipt and narrates its summary (8d45f964)
|
||||
- fix(reads): the readiness gate guards every index read surface — serving empty from a not-ready provider is unrepresentable (40e7119b)
|
||||
- ci(gate): the machine-health preflight and the truncation verdict guard (1e046aa1)
|
||||
|
||||
|
||||
### [10.3.1](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v10.3.0...v10.3.1) (2026-08-18)
|
||||
|
||||
- docs(releases): the 10.3.1 consumer entry — the fold that behaves (900cc895)
|
||||
- fix(recovery): the fold streams and narrates; the checkpoint chain arms at the flip (ed7d1db9)
|
||||
|
||||
|
||||
### [10.3.0](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v10.2.0...v10.3.0) (2026-08-18)
|
||||
|
||||
- docs(releases): the 10.3.0 consumer entry — the trust-and-provenance release (97d75649)
|
||||
- fix(locks): the fence keys ownership on pid+hostname — a same-process re-open never fences its predecessor (0991cf28)
|
||||
- test(budgets): iron-honest wall-clock budgets — 3x the worst honest-iron measurement (314e0e6c)
|
||||
- fix(locks): live writers are never auto-evicted; evicted writers are fenced at every commit barrier (292e7c04)
|
||||
- feat(log): system commits carry their origin; the attested per-id reconcile door (9ac9e706)
|
||||
|
||||
|
||||
### [10.2.0](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v10.1.0...v10.2.0) (2026-08-17)
|
||||
|
||||
- docs(releases): the 10.2.0 consumer entry — adoption completes in one call (97538e1f)
|
||||
- ci: the correctness plant runs integration + conformance on every push — a release never waits on a second machine (b17fdc8e)
|
||||
- fix(adoption): the baseline backfill runs to completion — one call adopts a pre-log baseline of any size (a5a18838)
|
||||
|
||||
|
||||
### [10.1.0](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v10.0.0...v10.1.0) (2026-08-13)
|
||||
|
||||
- docs(releases): the 10.1.0 consumer entry — bounded recovery, restore founding, the two write-path cures (7d3c8696)
|
||||
- fix(restore): a restore is an unclean event — the swap runs quiesced and the snapshot's durability stamps never survive it (9ca80667)
|
||||
- feat(recovery): the fold-checkpoint bound — crash folds (checkpoint, head], never the whole log twice (ff43de1a)
|
||||
- fix(log): pad-frame construction is total; the at-ack sync-failure compensation splits by phase — a production adoption's two write-path defects, cured at their roots (cbe34d11)
|
||||
- feat(query): the sparse-store cut — where on a never-carried field serves operator truth, never a refusal (7b67db4d)
|
||||
|
||||
|
||||
### [10.0.0](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v9.0.0...v10.0.0) (2026-08-12)
|
||||
|
||||
- fix(adoption): the baseline backfill cures hydration-law drift — existing brains reach the crash-safe default with zero operator steps (25f0dd96)
|
||||
- fix(adoption): the reserved-root mint exemption — int 0 is legitimate for exactly one id (2abe8b38)
|
||||
- fix(recovery): walks are healers — the typed/tolerant boundary redrawn where block-layer fault injection proved it belonged (0e3facf4)
|
||||
- feat(log): log authority is the fleet default — adopt-at-open, oracle-gated; plus the power-cut throw-site cures and the loud torn-record contract (214c98b4)
|
||||
- fix(durability): three block-layer power-loss findings from the first fault-injection box run — all cured, matrix 15/15 (67c606be)
|
||||
- docs: RELEASES.md frames the release as 10.0.0 — honest major (log format v2 forward-only); comment wording cleanup (d1698fa5)
|
||||
- fix(persistence): the idle flush trigger debounces under load — deferred to the floor, never dropped, never a flush-per-gap amplifier (a50726e6)
|
||||
- feat(reprojection): the one doors-open machinery — budget-capped, yielding, foreground-preempted, atomic-swap; poison records quarantine typed (d1651f98)
|
||||
- feat(embedding): deferred-embed markers become log records — the sidecar recovery path is deleted (b47787bb)
|
||||
- feat(conformance): the golden-log fold oracle — encoder bytes and fold semantics pinned by content hash (c95bea88)
|
||||
- feat(engine): the wiring wave — stamps ride every flush, provider generations, waitForIndexed, adopt-backfill, match-all serves (b53e6e89)
|
||||
- feat(index): watermark stamps on every TS projection — adopt/catchup/rescan verdicts at load, stamp-after-data (b35d87a7)
|
||||
- feat(log): v2 is the LIVE write format — envelope records with minted ints, genesis, sector seals; v1 readable forever (26c60251)
|
||||
- docs: RELEASES.md — the unreleased write-path and lifecycle entry (consumer-facing draft; version set at cut) (73eb88d4)
|
||||
- feat(temporal): as-of semantic recall joins the release contract — past vectors byte-exact, pinned (f7ca0d26)
|
||||
- fix(log): acked writes survive power loss; rejected writes never silently commit — the kill-matrix goes 11/11 with zero .fails debt (13022c51)
|
||||
- feat(plugin): every provider write surface carries the real committed generation (2d532684)
|
||||
- feat(log): fact-log format v2 codec — record envelope, type registry, genesis, sector seals; fault-injection shim (34841074)
|
||||
- feat(log): the guarded log-authority core — group-commit durable-at-ack, the per-brain switch, the verification oracle (65953097)
|
||||
- docs: Path Registry rows DP6/DP8/MT5 flip to contracted+pinned — the deferred-embedding and atomic-update train landed with cited tests (9fda6d95)
|
||||
- feat(embedding): MT5 — deferred embedding with durable markers; write acks never wait on a neural net (287384cf)
|
||||
- fix(index): the flicker window dies — atomic in-place vector update; lazy open honors every provider's not-ready report; the Path Registry twin table (ebe06cdf)
|
||||
- feat(persistence): the engine owns its flush cadence — callers never call flush() in hot paths again (3236a01b)
|
||||
- fix(aggregation): the lifecycle cluster — flush stamps, behind-stamp catches up incrementally, the native rebuild finally gets invoked, deletes are never silently skipped (1dc861d2)
|
||||
- perf(sort): ordered reads never do per-row storage round-trips — the 199-317s production scan class dies structurally (607b6b56)
|
||||
- chore: the home registry is The Source, never 'the forge' — sweep the misnomer out of the release rail, workflows, and release notes (Forge is a different product; the stored CI secret keeps its historical name) (09352c2b)
|
||||
- ci: tags stop triggering the CI matrix (redundant re-run of already-tested commits starved every release's publish run on the sequential runner) + release.sh forge poll window 20→50 min (c6c6ea6b)
|
||||
- test: version-coupling pins go major-agnostic — the 8.x literals broke at the 9.0.0 bump while the coupling law itself behaved correctly (8a6807e8)
|
||||
|
||||
|
||||
### [9.0.0](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v8.11.0...v9.0.0) (2026-08-04)
|
||||
|
||||
- docs: 9.0 namespace-migration guide — the simple story + the mechanical sweep checklist, published for humans and tooling alike (61ab9db2)
|
||||
- fix(release): storefront leg republishes CI's exact forge artifact — byte-identity by construction, verified by cross-registry shasum before the ceremony reports success (d89df2ed)
|
||||
- docs: v9.0.0 release notes — the field-addressing law migration ledger; retitle the shipped 8.11.0 canonical-enumeration entry (header went stale at its cut) (55a7512c)
|
||||
- feat(namespace): merge the field-addressing law train — no special names, system.* scalars, nested-bag storage, epoch-3 index keys (19b477ae)
|
||||
- feat(namespace): NO SPECIAL NAMES + storage fidelity — the ruled completion of the field-addressing law (24bf6cdb)
|
||||
- feat(namespace): write-door forgery refusal (user metadata keys may never start 'system.') + refusal messages name both spellings in every branch (the non-colliding case marks system.<f> honestly as NOT valid) — cross-engine message pin alignment (48a6130a)
|
||||
- feat(namespace): conformance green 19/19 — data-aware did-you-mean on unindexed bare addresses, ordering contract on the column top-K path (never drop, nulls last, ties by id), shape-complete addressed reads (entity views AND raw storage shapes, shadow-proof both scopes), per-key source matching for dotted addresses; refusal classes unified under UnresolvableFieldError (8e962dab)
|
||||
- feat(namespace): aggregation reads under the law + epoch 3 (the key-split rebuild) + THE ARMING COMMIT — the capability constant, the law module, and the typed refusals export from the package root; both engines' conformance suites light on this signal (7492b6cb)
|
||||
- feat(namespace): egress guard + validation speak the law — whereMatcher's resolver reads system.* from the record and bare names from the metadata bag only (the bare-system switch is dead); validateFindParams refuses cursor/includeRelations/writeOnly typed (accepted-and-ignored dies as a class), validates order, and parses every orderBy address (c2fb28a2)
|
||||
- fix(namespace): noun-record updates preserve legacy inline HNSW adjacency — the placeholder-adjacency write stamped out pre-codec records' stored connections (crash-window unreachability); codec-era records were never at risk (empty field is the blob marker); pin covers the legacy shape (4679c894)
|
||||
- feat(namespace): find's own filter builders speak the frozen keys — params.type/subtype/service become system.* index keys at every construction site (three pipelines + the canonical buildMetadataFilter); the where.type→noun alias is dead (bare 'type' belongs to the user now) (7a28a946)
|
||||
- feat(namespace): the index speaks the frozen keys — record-frame scalars index under literal 'system.<field>' (legacy 'noun' spelling folds into system.type; plumbing never indexed from a record frame), user fields stay bare in every shape; filter + sorted paths route every address through parseFieldAddress; storage fallbacks read the addressed side of the record (11c724bc)
|
||||
- docs(namespace): the d.ts JSDoc wave — the sealed field-addressing law on the full find + aggregation surface, present-tense, with the refusal semantics and migration note inline (comment-only; verified zero code lines changed) (fcb24ab6)
|
||||
- test(namespace): unit pins for the pure law — the ruled maps verbatim (incl. the relation mirror, unpinnable via public API), plumbing refusals both kinds, did-you-mean text (5502abcd)
|
||||
- fix(namespace): the JS sorted fallback honors the ruled ordering contract — nulls last in BOTH directions (was nulls-first on desc) + deterministic id-ascending tie-break (56deb2e8)
|
||||
- test(namespace)+docs: the cross-engine conformance suite (self-arming — skips until the resolver exports land) + the public field-addressing docs page; sidebar order deconflicted to 7 (d8d0b55f)
|
||||
- feat(namespace): the one field-addressing law as a single source of truth — parseFieldAddress + the ruled ten-scalar system maps + plumbing invisibility + refusal builders (module only; query surfaces wire in next) (8f9a9989)
|
||||
- docs: port the 8.10.3 backport-release changelog entry to main (f6b14d21)
|
||||
- docs: port the 8.10.2 backport-release changelog entry to main — release branches carry the version bump, main carries the durable record (0b059ac5)
|
||||
- fix: user metadata named 'level' is a real field everywhere — the engine-internal node layer no longer shadows it in sort/filter/aggregation, and the indexing views stop stamping a phantom 0 into its column; index epoch 2 rebuilds existing brains at first open (1a09be06)
|
||||
- fix: metadata-only update() never rewrites the noun record — the unconditional whole-vector save turned per-entity stat touches into full rewrites+fsync, amplifying read-heavy sweeps into disk saturation on a production deployment (cb717be2)
|
||||
- fix(release): double the forge-publish poll budget — the runner executes jobs sequentially and the publish run queues behind the ci matrix (64049631)
|
||||
- Merge branch 'release/8.11.0' (1865f60a)
|
||||
- Merge branch 'release/8.10.1' (fc9f0d72)
|
||||
- chore: the forge is the address — retire the archived mirror from every live surface (415e824a)
|
||||
- Merge remote-tracking branch 'origin/main' (069a8894)
|
||||
- Merge branch 'release/8.10.0' (d918c060)
|
||||
- ci: run the pipeline on the forge (9a5a9ccc)
|
||||
- feat: two-tier history reads + the repacker + generationDigest — D1+D3 wired end-to-end (1201e255)
|
||||
- feat: generation-segment store — the D1+D3 packed-tier file format (d8acb377)
|
||||
- feat: scanFacts liveness contract — first batch or loud failure within a documented bound (f8e6da2b)
|
||||
|
||||
|
||||
### [8.11.0](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v8.10.1...v8.11.0) (2026-07-27)
|
||||
|
||||
- docs: the last two archived-host links point home (91ef1c8b)
|
||||
- feat: includeHidden — export carries every visibility tier for migration-grade canon completeness (63c1eeb9)
|
||||
- feat(release): the forge publish leg moves to CI on the tag push; the laptop verifies by readback and keeps the abort-before-storefront guard (3e4a17dc)
|
||||
- feat: canonical enumeration mode for export — storage-walked, canon-complete, with an index-drift report (4d196af4)
|
||||
- ci: run the pipeline on the forge (999d0ebb)
|
||||
|
||||
|
||||
### [8.10.3](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v8.10.2...v8.10.3) (2026-08-03)
|
||||
|
||||
- docs: dedupe the 8.10.2 release-notes entry the cherry doubled onto the branch (8c956608)
|
||||
- fix: user metadata named 'level' is a real field everywhere — the engine-internal node layer no longer shadows it in sort/filter/aggregation, and the indexing views stop stamping a phantom 0 into its column; index epoch 2 rebuilds existing brains at first open (958a0859)
|
||||
|
||||
|
||||
### [8.10.2](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v8.10.1...v8.10.2) (2026-07-29)
|
||||
|
||||
- docs: 8.10.2 consumer release notes — update() write granularity, PathResolver idle-log fix, graph-lsm key recognition (a0123b5b)
|
||||
- fix: metadata-only update() never rewrites the noun record — the unconditional whole-vector save turned per-entity stat touches into full rewrites+fsync, amplifying read-heavy sweeps into disk saturation on a production deployment (5b65eb82)
|
||||
|
||||
|
||||
### [8.10.1](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v8.10.0...v8.10.1) (2026-07-24)
|
||||
|
||||
- refactor: remove the orphaned transaction-result type left behind by the dead-path removal (edf123a5)
|
||||
- fix: warm() metadata surface routes through the active provider (warm hook added to the metadata contract); add maintenanceDebt() observability surface (5b2cbf74)
|
||||
- fix: transaction timeouts are a typed no-hot-retry contract; engine-side non-retry pinned; dead transaction path removed (003e2a74)
|
||||
- chore: the forge is the address — retire the archived mirror from every live surface (22702b81)
|
||||
|
||||
|
||||
### [8.10.0](https://github.com/soulcraftlabs/brainy/compare/v8.9.0...v8.10.0) (2026-07-23)
|
||||
|
||||
- docs: adoption storefront — contributing guide, security policy, README support + cor section (9a99a7b)
|
||||
- fix(release): push the public mirror explicitly and verify the tag lands at the right commit before publishing (6ba94c8)
|
||||
- docs: project guide version line points at npm instead of a hardcoded stale number (3a1efc9)
|
||||
- feat: vector provider identity is a required name field (hnsw-js), rendered [vector-index:<name>] (3be4ba9)
|
||||
- feat: warm contract (warm/warmOnOpen/provider warm hook), configurable transact budget floor, backend-neutral vector index op names (55b867c)
|
||||
|
||||
|
||||
### [8.9.0](https://github.com/soulcraftlabs/brainy/compare/v8.8.2...v8.9.0) (2026-07-19)
|
||||
|
||||
- docs: measured performance envelopes v1 (per-op p50/p95 at 1k and 10k, pure-JS floor) (5cabd78)
|
||||
- fix: release drains in-flight writer-lock heartbeat — no phantom lock after unlink (70e4bc8)
|
||||
- feat: flush() never compacts — history maintenance moves to close() with bounded passes (300d9f2)
|
||||
|
||||
|
||||
### [8.8.2](https://github.com/soulcraftlabs/brainy/compare/v8.8.1...v8.8.2) (2026-07-19)
|
||||
|
||||
- fix: one field-resolution law across aggregation hooks, source.where, removeMany, and find() spellings (945d92d)
|
||||
- chore: push public docs to the soulcraft.com ingest door on release (42037d0)
|
||||
|
||||
|
||||
### [8.8.1](https://github.com/soulcraftlabs/brainy/compare/v8.8.0...v8.8.1) (2026-07-18)
|
||||
|
||||
- fix: O(1) adaptive retention accounting + historyStats fleet audit (6207e48)
|
||||
- fix: import dedup off-switch honesty + brain-owned lifecycle for the background pass (4fcef7b)
|
||||
|
||||
|
||||
### [8.8.0](https://github.com/soulcraftlabs/brainy/compare/v8.7.1...v8.8.0) (2026-07-17)
|
||||
|
||||
- feat: OS-limit detection for pool-scale deployments (16a73b8)
|
||||
|
||||
|
||||
### [8.7.1](https://github.com/soulcraftlabs/brainy/compare/v8.7.0...v8.7.1) (2026-07-17)
|
||||
|
||||
- fix: race-proof writer-lock acquisition + machine-readable conflict through init (01a3b46)
|
||||
|
||||
|
||||
### [8.7.0](https://github.com/soulcraftlabs/brainy/compare/v8.6.0...v8.7.0) (2026-07-17)
|
||||
|
||||
- feat: scaled transact budgets + labeled timeout diagnostics + envelope docs (6ef9fcb)
|
||||
|
||||
|
||||
### [8.6.0](https://github.com/soulcraftlabs/brainy/compare/v8.5.2...v8.6.0) (2026-07-17)
|
||||
|
||||
- feat: brain.auditGraph() — read-only graph-truth audit (2a03fae)
|
||||
|
||||
|
||||
### [8.5.2](https://github.com/soulcraftlabs/brainy/compare/v8.5.1...v8.5.2) (2026-07-17)
|
||||
|
||||
- fix: exception-safe aggregation backfill + generation-verified adoption + loud open-path guards (a77b064)
|
||||
|
||||
|
||||
### [8.5.1](https://github.com/soulcraftlabs/brainy/compare/v8.5.0...v8.5.1) (2026-07-17)
|
||||
|
||||
- fix: aggregation state adoption on reopen + single-flight backfill + query-cap ratchet removal (da55be7)
|
||||
- docs: external-backups/sparse-storage guide + generation fact log concept (593bb8b)
|
||||
|
||||
|
||||
### [8.5.0](https://github.com/soulcraftlabs/brainy/compare/v8.4.0...v8.5.0) (2026-07-15)
|
||||
|
||||
- test: tolerant timing assertion in the execution-time measure test (4dc0a92)
|
||||
- feat: committedGeneration capability + pinned durability/stability contracts (d1ecee1)
|
||||
- docs: RELEASES.md entry for 8.5.0 (provider fact-log access + shared verifier) (e4f37cd)
|
||||
- feat: provider access to the fact log + shared stamp verifier via internals (352e356)
|
||||
|
||||
|
||||
### [8.4.0](https://github.com/soulcraftlabs/brainy/compare/v8.3.3...v8.4.0) (2026-07-15)
|
||||
|
||||
- docs: RELEASES.md entry for 8.4.0 (generation fact log + family stamp) (4a60b43)
|
||||
- feat: entity-tree family stamp — sourceGeneration + rollup coherence at open (2888ae6)
|
||||
- feat: generation fact log — after-image commit records, dual-written at every commit point (38b0041)
|
||||
|
||||
|
||||
### [8.3.3](https://github.com/soulcraftlabs/brainy/compare/v8.3.2...v8.3.3) (2026-07-15)
|
||||
|
||||
- docs: RELEASES.md entry for 8.3.3 (rename containment fix + repair) (c3feafd)
|
||||
- test: lens-consistency regression — combined vs subtype-only vs canonical ground truth (4fb41f9)
|
||||
- fix: VFS rename moves the containment edge — no ghost in the old directory (af8c179)
|
||||
|
||||
|
||||
### [8.3.2](https://github.com/soulcraftlabs/brainy/compare/v8.3.1...v8.3.2) (2026-07-14)
|
||||
|
||||
- docs: RELEASES.md entry for 8.3.2 (honest counters) (0932ecd)
|
||||
- fix: honest counters — removal never re-reads the removed record + repairIndex recounts and persists all rollups (2e2ba9c)
|
||||
|
||||
|
||||
### [8.3.1](https://github.com/soulcraftlabs/brainy/compare/v8.3.0...v8.3.1) (2026-07-14)
|
||||
|
||||
- docs: RELEASES.md entry for 8.3.1 (full-removal deletes + family-scoped gate) (c0c68ac)
|
||||
- fix: full-removal canonical deletes + family-scoped migration gate (366f9a9)
|
||||
- docs: cite the cross-layer integrity contract generically in comments and notes (1d26988)
|
||||
|
||||
|
||||
### [8.3.0](https://github.com/soulcraftlabs/brainy/compare/v8.2.8...v8.3.0) (2026-07-13)
|
||||
|
||||
- docs: RELEASES.md entry for 8.3.0 (heal-cost + cross-layer integrity contract) (7692c6f)
|
||||
- docs: RELEASES.md entry for 8.3.0 (heal-cost + ADR-004 Pass 2/3) (7692c6f)
|
||||
- perf: parallel + id-only canonical enumeration (heal-cost dominant term) (ec5b933)
|
||||
- feat: registered-blob family contract — declared index blobs are undeletable (bfa1762)
|
||||
- feat: validateIndexConsistency delegates to provider invariants (6bcb54f)
|
||||
- feat: registered-blob family contract — declared index blobs are undeletable (ADR-004 Pass 2) (bfa1762)
|
||||
- feat: validateIndexConsistency delegates to provider invariants (ADR-004 Pass 3) (6bcb54f)
|
||||
|
||||
|
||||
### [8.2.8](https://github.com/soulcraftlabs/brainy/compare/v8.2.7...v8.2.8) (2026-07-13)
|
||||
|
|
|
|||
10
CLAUDE.md
10
CLAUDE.md
|
|
@ -12,13 +12,13 @@ Handoff file: `/home/dpsifr/.strategy/PLATFORM-HANDOFF.md`
|
|||
|
||||
**Brainy's current open actions:** None. MIT open-source — no platform-specific actions.
|
||||
|
||||
**Current version:** run `npm view @soulcraftlabs/brainy version --registry https://source.soulcraft.com/api/packages/soulcraftlabs/npm/` (never trust a hardcoded number here — this line went stale for months); consumer-facing changes tracked in `RELEASES.md`
|
||||
**Current version:** `@soulcraft/brainy@7.31.5` (latest published; 8.0.0 release candidate on `feat/8.0-u64-ids`)
|
||||
|
||||
---
|
||||
|
||||
## Project Overview
|
||||
|
||||
Brainy is a Universal Knowledge Protocol -- a Triple Intelligence database that combines vector similarity search, graph traversal, and metadata filtering into a single TypeScript library. Published as `@soulcraftlabs/brainy` on The Source (source.soulcraft.com registry) under the MIT license.
|
||||
Brainy is a Universal Knowledge Protocol -- a Triple Intelligence database that combines vector similarity search, graph traversal, and metadata filtering into a single TypeScript library. Published as `@soulcraft/brainy` on npm under the MIT license.
|
||||
|
||||
## Getting Started
|
||||
|
||||
|
|
@ -91,7 +91,7 @@ test: add/update tests (patch version bump)
|
|||
|
||||
## Docs Pipeline — soulcraft.com/docs
|
||||
|
||||
Docs in `docs/**/*.md` are published with the npm package (included in `files`) and go live on soulcraft.com/docs via the docs ingest API: the release script's `scripts/push-docs.js` step POSTs every public doc to `https://soulcraft.com/api/docs/ingest` (auth: `DOCS_INGEST_SECRET` in the environment). No separate deploy step is involved (the old deploy-to-publish flow was retired in a platform change, 2026-08). Frontmatter controls what appears publicly.
|
||||
Docs in `docs/**/*.md` are published with the npm package (included in `files`) and synced to soulcraft.com/docs on every portal deploy. Frontmatter controls what appears publicly.
|
||||
|
||||
### Docs check triggers
|
||||
|
||||
|
|
@ -161,9 +161,9 @@ npm run release:major # Breaking changes (rare, manual decision)
|
|||
The script: verifies clean git state, builds, tests, bumps version, updates CHANGELOG.md, commits, tags, pushes, publishes to npm, and creates a GitHub release.
|
||||
|
||||
After a successful release, remind the user:
|
||||
> "Published. Docs are live on soulcraft.com/docs (pushed via the ingest API during the release) — spot-check a changed page with curl."
|
||||
> "Published. Deploy portal to pick up the new docs → go to the portal project and deploy."
|
||||
|
||||
There is no separate deploy step anymore. If the docs push failed (the script warns loudly), re-run `node scripts/push-docs.js` with `DOCS_INGEST_SECRET` set.
|
||||
Do NOT deploy portal from here. Portal is always deployed separately from within the portal project.
|
||||
|
||||
## Closed-Source Product Names — HARD RULE
|
||||
|
||||
|
|
|
|||
329
CONTRIBUTING.md
329
CONTRIBUTING.md
|
|
@ -1,77 +1,298 @@
|
|||
# Contributing to Brainy
|
||||
|
||||
Brainy is MIT-licensed and genuinely open to outside contributions. This page
|
||||
is the honest, current path — please don't rely on older instructions you
|
||||
may find elsewhere in the repo's history.
|
||||
Thank you for your interest in contributing to Brainy! This document provides guidelines and instructions for contributing to the project.
|
||||
|
||||
## Where the project lives
|
||||
## Code of Conduct
|
||||
|
||||
The source of truth is a self-hosted forge: **source.soulcraft.com/soulcraftlabs/open-brainy**.
|
||||
It's anonymously readable and cloneable — no account needed to browse, clone,
|
||||
or build.
|
||||
By participating in this project, you agree to abide by our Code of Conduct:
|
||||
- Be respectful and inclusive
|
||||
- Welcome newcomers and help them get started
|
||||
- Focus on constructive criticism
|
||||
- Respect differing viewpoints and experiences
|
||||
|
||||
## How to contribute
|
||||
## How to Contribute
|
||||
|
||||
**Found a bug, or have an idea?** Email **brainy@soulcraft.com**. No account,
|
||||
no ceremony — you'll get a receipt, and it goes to a human.
|
||||
### Reporting Issues
|
||||
|
||||
**Want to send a patch?** Two ways, both first-class:
|
||||
Before creating an issue, please check existing issues to avoid duplicates.
|
||||
|
||||
- **Email a patch.** Run `git format-patch` against your change and email the
|
||||
output to **brainy@soulcraft.com**. This is a genuinely supported path, not
|
||||
a fallback — plenty of good contributions arrive this way.
|
||||
- **Open a pull request on the forge.** Request an account at
|
||||
**source.soulcraft.com** (registration is request-with-approval, so allow
|
||||
a little lag), clone, push a branch, and open a PR there. Maintainers
|
||||
review and land it.
|
||||
When creating an issue, include:
|
||||
- Clear, descriptive title
|
||||
- Detailed description of the problem
|
||||
- Steps to reproduce
|
||||
- Expected vs actual behavior
|
||||
- System information (OS, Node version, Brainy version)
|
||||
- Code examples if applicable
|
||||
|
||||
Either way, for anything beyond a small fix, opening an issue first (email is
|
||||
fine) to talk through the approach saves everyone rework.
|
||||
### Suggesting Features
|
||||
|
||||
## Development setup
|
||||
Feature requests are welcome! Please provide:
|
||||
- Clear use case
|
||||
- Proposed API/interface
|
||||
- Examples of how it would work
|
||||
- Any potential challenges or considerations
|
||||
|
||||
### Pull Requests
|
||||
|
||||
#### Before Starting
|
||||
|
||||
1. Check existing issues and PRs
|
||||
2. Open an issue to discuss significant changes
|
||||
3. Fork the repository
|
||||
4. Create a feature branch from `main`
|
||||
|
||||
#### Development Setup
|
||||
|
||||
**Quick Setup (Recommended):**
|
||||
```bash
|
||||
git clone https://source.soulcraft.com/soulcraftlabs/open-brainy.git
|
||||
# Clone your fork
|
||||
git clone https://github.com/your-username/brainy.git
|
||||
cd brainy
|
||||
|
||||
# Run setup script (installs all dependencies including Rust)
|
||||
./scripts/setup-dev.sh
|
||||
```
|
||||
|
||||
**Manual Setup:**
|
||||
```bash
|
||||
# Clone your fork
|
||||
git clone https://github.com/your-username/brainy.git
|
||||
cd brainy
|
||||
|
||||
# Install system dependencies (Ubuntu/Debian)
|
||||
sudo apt-get install -y build-essential pkg-config libssl-dev
|
||||
|
||||
# Install Rust (for WASM embedding engine)
|
||||
curl --proto '=https' --tlsv1.2 -sSf https://sh.rustup.rs | sh
|
||||
source ~/.cargo/env
|
||||
rustup target add wasm32-unknown-unknown
|
||||
cargo install wasm-pack
|
||||
|
||||
# Install Node.js dependencies
|
||||
npm install
|
||||
|
||||
# Build Candle WASM embedding engine
|
||||
npm run build:candle
|
||||
|
||||
# Build TypeScript
|
||||
npm run build
|
||||
|
||||
# Run tests
|
||||
npm test
|
||||
```
|
||||
|
||||
Tests run on [Vitest](https://vitest.dev/). `npm test` runs the unit suite;
|
||||
see `package.json` for `test:integration`, `test:coverage`, and friends.
|
||||
#### Making Changes
|
||||
|
||||
## Standards
|
||||
1. **Follow the code style**
|
||||
- TypeScript for all source code
|
||||
- Clear variable and function names
|
||||
- Comments for complex logic
|
||||
- JSDoc for public APIs
|
||||
|
||||
- **Strict TypeScript.** No `any` escape hatches to dodge the type checker.
|
||||
- **Tests exercise real behavior.** No mocking away the thing you're supposed
|
||||
to be testing.
|
||||
- **No stubs, no TODO-code.** If something can't be finished, say so and
|
||||
leave it out — don't merge a placeholder.
|
||||
- **JSDoc on every exported function, class, and type.**
|
||||
- **[Conventional Commits](https://www.conventionalcommits.org/).** `feat:`,
|
||||
`fix:`, `docs:`, `perf:`, `refactor:`, `test:`, `chore:`. Never
|
||||
`BREAKING CHANGE` in a commit message — major version bumps are a separate,
|
||||
deliberate decision.
|
||||
- **Performance claims are measured or labeled projected.** If a PR or its
|
||||
description states a number, cite the benchmark that produced it (see
|
||||
[docs/performance-envelopes.md](docs/performance-envelopes.md) for the
|
||||
pattern). Don't state an estimate as if it were measured.
|
||||
- **Measurements carry numbers, not provenance.** Public commit messages and
|
||||
docs give the SHAPE a number was taken at and never where it was taken: no
|
||||
hostnames, no store or deployment identities, no operational anecdotes about
|
||||
someone's running system. "A 14,056-noun / 72,679-verb production-shaped
|
||||
store, measured solo under an exclusive lock" tells a reader everything the
|
||||
number depends on; the machine it ran on and whose data it was tell them
|
||||
nothing except where somebody's infrastructure lives.
|
||||
- **Documents that answer or reference a confidential specification never enter
|
||||
this repository, even summarized.** The public docs describe THIS engine and
|
||||
the published contract, and nothing else — a summary of a private document is
|
||||
still that document's contents.
|
||||
2. **Write tests**
|
||||
- Add tests for new features
|
||||
- Update tests for changes
|
||||
- Ensure all tests pass
|
||||
|
||||
## License
|
||||
3. **Update documentation**
|
||||
- Update README if needed
|
||||
- Add/update API documentation
|
||||
- Include examples
|
||||
|
||||
Brainy is [MIT licensed](LICENSE). Contributions are accepted under the same
|
||||
license — there's no CLA to sign.
|
||||
#### Commit Guidelines
|
||||
|
||||
Thank you for considering a contribution.
|
||||
Follow conventional commits format:
|
||||
|
||||
```
|
||||
type(scope): description
|
||||
|
||||
[optional body]
|
||||
|
||||
[optional footer]
|
||||
```
|
||||
|
||||
Types:
|
||||
- `feat`: New feature
|
||||
- `fix`: Bug fix
|
||||
- `docs`: Documentation changes
|
||||
- `style`: Code style changes
|
||||
- `refactor`: Code refactoring
|
||||
- `perf`: Performance improvements
|
||||
- `test`: Test changes
|
||||
- `chore`: Build/tooling changes
|
||||
|
||||
Examples:
|
||||
```bash
|
||||
feat(triple): add graph traversal depth limit
|
||||
fix(storage): handle concurrent write conflicts
|
||||
docs(api): update search method documentation
|
||||
```
|
||||
|
||||
#### Submitting PR
|
||||
|
||||
1. Push to your fork
|
||||
2. Create PR against `main` branch
|
||||
3. Fill out PR template
|
||||
4. Ensure CI checks pass
|
||||
5. Wait for review
|
||||
|
||||
### Testing
|
||||
|
||||
#### Running Tests
|
||||
|
||||
```bash
|
||||
# Run all tests
|
||||
npm test
|
||||
|
||||
# Run specific test file
|
||||
npm test tests/core.test.ts
|
||||
|
||||
# Run with coverage
|
||||
npm run test:coverage
|
||||
|
||||
# Watch mode
|
||||
npm run test:watch
|
||||
```
|
||||
|
||||
#### Writing Tests
|
||||
|
||||
```typescript
|
||||
import { describe, it, expect } from 'vitest'
|
||||
import { Brainy } from '../src'
|
||||
|
||||
describe('Feature Name', () => {
|
||||
it('should do something specific', async () => {
|
||||
const brain = new Brainy()
|
||||
await brain.init()
|
||||
|
||||
// Test implementation
|
||||
const result = await brain.search("test")
|
||||
|
||||
expect(result).toBeDefined()
|
||||
expect(result.length).toBeGreaterThan(0)
|
||||
})
|
||||
})
|
||||
```
|
||||
|
||||
## Architecture Guidelines
|
||||
|
||||
### Adding New Features
|
||||
|
||||
1. **Check existing functionality**
|
||||
- Review `ARCHITECTURE.md`
|
||||
- Check if similar features exist
|
||||
- Consider if it should be an augmentation
|
||||
|
||||
2. **Design considerations**
|
||||
- Maintain backward compatibility
|
||||
- Consider performance impact
|
||||
- Think about all storage adapters
|
||||
- Plan for extensibility
|
||||
|
||||
3. **Implementation checklist**
|
||||
- [ ] Core functionality
|
||||
- [ ] Tests (unit and integration)
|
||||
- [ ] Documentation
|
||||
- [ ] TypeScript types
|
||||
- [ ] Examples
|
||||
- [ ] Performance benchmarks (if applicable)
|
||||
|
||||
### Creating Augmentations
|
||||
|
||||
Augmentations extend Brainy's functionality:
|
||||
|
||||
```typescript
|
||||
import { BrainyAugmentation } from '../types'
|
||||
|
||||
export class MyAugmentation extends BrainyAugmentation {
|
||||
name = 'MyAugmentation'
|
||||
|
||||
async onInit(brain: Brainy): Promise<void> {
|
||||
// Initialize augmentation
|
||||
}
|
||||
|
||||
async onAdd(item: any, brain: Brainy): Promise<any> {
|
||||
// Process before adding
|
||||
return item
|
||||
}
|
||||
|
||||
async onSearch(query: any, results: any[], brain: Brainy): Promise<any[]> {
|
||||
// Process search results
|
||||
return results
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
### Performance Considerations
|
||||
|
||||
- Use batch operations where possible
|
||||
- Implement caching strategically
|
||||
- Consider memory usage
|
||||
- Profile performance impacts
|
||||
- Add benchmarks for critical paths
|
||||
|
||||
## Documentation
|
||||
|
||||
### API Documentation
|
||||
|
||||
Use JSDoc for all public APIs:
|
||||
|
||||
```typescript
|
||||
/**
|
||||
* Searches for similar items using vector similarity
|
||||
* @param query - Search query (text or vector)
|
||||
* @param options - Search options
|
||||
* @returns Array of search results with scores
|
||||
* @example
|
||||
* ```typescript
|
||||
* const results = await brain.search("machine learning", { limit: 10 })
|
||||
* ```
|
||||
*/
|
||||
async search(query: string | Vector, options?: SearchOptions): Promise<SearchResult[]> {
|
||||
// Implementation
|
||||
}
|
||||
```
|
||||
|
||||
### Examples
|
||||
|
||||
Add examples for new features:
|
||||
|
||||
```typescript
|
||||
// examples/feature-name.ts
|
||||
import { Brainy } from 'brainy'
|
||||
|
||||
async function exampleUsage() {
|
||||
const brain = new Brainy()
|
||||
await brain.init()
|
||||
|
||||
// Show feature usage
|
||||
// Include comments explaining what's happening
|
||||
// Handle errors appropriately
|
||||
}
|
||||
|
||||
exampleUsage().catch(console.error)
|
||||
```
|
||||
|
||||
## Release Process
|
||||
|
||||
1. **Version bump**: Follow semantic versioning
|
||||
2. **Update CHANGELOG**: Document all changes
|
||||
3. **Run tests**: Ensure all tests pass
|
||||
4. **Build**: Generate distribution files
|
||||
5. **Tag**: Create git tag for version
|
||||
6. **Publish**: Release to npm
|
||||
|
||||
## Getting Help
|
||||
|
||||
- **Discord**: Join our community
|
||||
- **Issues**: Ask questions on GitHub
|
||||
- **Discussions**: Share ideas and get feedback
|
||||
|
||||
## Recognition
|
||||
|
||||
Contributors will be recognized in:
|
||||
- CHANGELOG.md for their contributions
|
||||
- README.md contributors section
|
||||
- GitHub contributors page
|
||||
|
||||
Thank you for contributing to Brainy! 🧠
|
||||
42
README.md
42
README.md
|
|
@ -1,5 +1,5 @@
|
|||
<p align="center">
|
||||
<img src="https://source.soulcraft.com/soulcraftlabs/open-brainy/raw/branch/main/brainy.png" alt="Brainy" width="180">
|
||||
<img src="https://raw.githubusercontent.com/soulcraftlabs/brainy/main/brainy.png" alt="Brainy" width="180">
|
||||
</p>
|
||||
|
||||
<h1 align="center">Brainy</h1>
|
||||
|
|
@ -11,9 +11,9 @@
|
|||
</p>
|
||||
|
||||
<p align="center">
|
||||
<a href="https://source.soulcraft.com/soulcraftlabs/-/packages/npm/brainy"><img src="https://img.shields.io/badge/package-The%20Source-2c3e50.svg" alt="Package on The Source"></a>
|
||||
<a href="https://source.soulcraft.com/soulcraftlabs/open-brainy"><img src="https://img.shields.io/badge/repo-open--brainy-2c3e50.svg" alt="Repository"></a>
|
||||
<a href="https://source.soulcraft.com/soulcraftlabs/open-brainy/actions"><img src="https://source.soulcraft.com/soulcraftlabs/open-brainy/actions/workflows/ci.yml/badge.svg?branch=main" alt="CI"></a>
|
||||
<a href="https://www.npmjs.com/package/@soulcraft/brainy"><img src="https://img.shields.io/npm/v/@soulcraft/brainy.svg" alt="npm version"></a>
|
||||
<a href="https://www.npmjs.com/package/@soulcraft/brainy"><img src="https://img.shields.io/npm/dm/@soulcraft/brainy.svg" alt="npm downloads"></a>
|
||||
<a href="https://github.com/soulcraftlabs/brainy/actions/workflows/ci.yml"><img src="https://github.com/soulcraftlabs/brainy/actions/workflows/ci.yml/badge.svg" alt="CI"></a>
|
||||
<a href="https://soulcraft.com/docs"><img src="https://img.shields.io/badge/docs-soulcraft.com-blue.svg" alt="Documentation"></a>
|
||||
<a href="LICENSE"><img src="https://img.shields.io/badge/license-MIT-blue.svg" alt="MIT License"></a>
|
||||
<a href="https://www.typescriptlang.org/"><img src="https://img.shields.io/badge/%3C%2F%3E-TypeScript-%230074c1.svg" alt="TypeScript"></a>
|
||||
|
|
@ -23,15 +23,12 @@
|
|||
<a href="#quick-start">Quick start</a> ·
|
||||
<a href="#one-query-three-engines">One query</a> ·
|
||||
<a href="#feature-tour">Features</a> ·
|
||||
<a href="#when-you-outgrow-brainy">Scale with Cor</a> ·
|
||||
<a href="#documentation">Docs</a> ·
|
||||
<a href="#support--community">Support</a>
|
||||
<a href="#from-laptop-to-hundreds-of-millions">Scale with Cor</a> ·
|
||||
<a href="#documentation">Docs</a>
|
||||
</p>
|
||||
|
||||
---
|
||||
|
||||
**Open Brainy** is the MIT engine — the open API, client library, types, and protocol; an openly specified canonical on-disk format; and this TypeScript reference engine, scoped as a single-node engine for stores up to roughly one million rows. `@soulcraft/brainy` 10.4.2 was the last release under the old package name — the name passes to the native engine, **Brainy**, at 11.0.0: the same API over the same open format at production scale, and it requires a license.
|
||||
|
||||
Built because we were tired of stitching a vector store to a graph database to a document store — and spending weeks on plumbing before writing a line of business logic. Brainy indexes every fact **three ways at once** and lets one call query them together:
|
||||
|
||||
| You write | Brainy indexes it as | You query it with |
|
||||
|
|
@ -47,14 +44,12 @@ It runs **inside your process** — no server, no Docker, nothing to operate —
|
|||
## Quick start
|
||||
|
||||
```bash
|
||||
bun add @soulcraftlabs/brainy # Bun ≥ 1.1 — recommended
|
||||
npm install @soulcraftlabs/brainy # Node.js ≥ 22
|
||||
bun add @soulcraft/brainy # Bun ≥ 1.1 — recommended
|
||||
npm install @soulcraft/brainy # Node.js ≥ 22
|
||||
```
|
||||
|
||||
> **Registry**: add `@soulcraftlabs:registry=https://source.soulcraft.com/api/packages/soulcraftlabs/npm/` to your `.npmrc` (anonymous read).
|
||||
|
||||
```javascript
|
||||
import { Brainy, NounType, VerbType } from '@soulcraftlabs/brainy'
|
||||
import { Brainy, NounType, VerbType } from '@soulcraft/brainy'
|
||||
|
||||
const brain = new Brainy() // in-memory; one line swaps to disk
|
||||
await brain.init()
|
||||
|
|
@ -177,11 +172,9 @@ await brain.vfs.search('React components with hooks') // semantic file
|
|||
|
||||
**[Multi-process model](docs/concepts/multi-process.md)** · **[Inspection guide](docs/guides/inspection.md)**
|
||||
|
||||
## When you outgrow Brainy
|
||||
## From laptop to hundreds of millions
|
||||
|
||||
Brainy's pure-TypeScript engines carry real workloads a long way on their own — see the measured, per-operation numbers (not marketing figures) in **[docs/performance-envelopes.md](docs/performance-envelopes.md)** for what to expect, unaccelerated, on plain filesystem storage.
|
||||
|
||||
When a deployment needs native-scale vector/graph performance — memory-mapped indexes that don't need your dataset in RAM, billion-scale ambitions — add the native engine. **The API doesn't change:**
|
||||
Brainy's TypeScript engines take you a long way. When you outgrow them, add the native engine — **the API doesn't change**:
|
||||
|
||||
```bash
|
||||
npm install @soulcraft/cor
|
||||
|
|
@ -194,14 +187,13 @@ await brain.init() // @soulcraft/cor detected — same code, native engines un
|
|||
|
||||
Installing the package is the opt-in: if `@soulcraft/cor` is present, it loads and announces itself in the init log; if it's present but broken, `init()` **throws** — an installed accelerator never silently vanishes behind the JS engines. Opt out with `plugins: []`, or pin exactly what loads with `plugins: ['@soulcraft/cor']`. [`@soulcraft/cor`](https://www.npmjs.com/package/@soulcraft/cor) (Brainy 8.x ↔ Cor 3.x, version-matched) registers Rust implementations behind every provider seam: SIMD distance kernels, memory-mapped storage, a disk-native vector index that doesn't need your dataset in RAM, durable LSM field/graph indexes that serve cold opens instantly, and native aggregation. Recall@10 measured **0.99 / 0.96 / 0.96 at 1M / 10M / 100M vectors** in Cor's release gate.
|
||||
|
||||
Open core, commercial accelerator: Brainy is MIT and complete on its own — Cor is more headroom for when you need it, not capability held back to sell you later. Licensing and support: **cor@soulcraft.com**.
|
||||
Open core, commercial accelerator: Brainy is MIT and complete on its own; Cor is licensed and funds both.
|
||||
|
||||
## Performance
|
||||
|
||||
- Per-operation p50/p95 at 1k and 10k entities, pure-JS floor, measured and re-run every release that touches a measured path: **[docs/performance-envelopes.md](docs/performance-envelopes.md)**.
|
||||
- JS distance kernels: **~6× faster cosine, ~1.4× euclidean** than 7.x (measured: [`tests/benchmarks/distance-microbench.mjs`](tests/benchmarks/distance-microbench.mjs), 384-dim, median of 41).
|
||||
- Whole-graph reads are single **O(N + E)** cursor walks — a consumer-measured 19k-edge export dropped from ~27 s of per-node calls to one scan.
|
||||
- Capacity planning and architecture: **[docs/PERFORMANCE.md](docs/PERFORMANCE.md)** · **[docs/SCALING.md](docs/SCALING.md)**
|
||||
- Full numbers and capacity planning: **[docs/PERFORMANCE.md](docs/PERFORMANCE.md)** · **[docs/SCALING.md](docs/SCALING.md)**
|
||||
|
||||
## Use cases
|
||||
|
||||
|
|
@ -220,10 +212,6 @@ Open core, commercial accelerator: Brainy is MIT and complete on its own — Cor
|
|||
|
||||
**Bun ≥ 1.1** (recommended) or **Node.js ≥ 22**. Brainy 8.x is server-only; the 7.x line remains on npm for browser use.
|
||||
|
||||
## Support & community
|
||||
## Contributing & license
|
||||
|
||||
- **Bugs and ideas** → **brainy@soulcraft.com** — no account needed, you'll get a receipt.
|
||||
- **Security reports** → **security@soulcraft.com** — see **[SECURITY.md](SECURITY.md)**.
|
||||
- **Contributing** → see **[CONTRIBUTING.md](CONTRIBUTING.md)**.
|
||||
|
||||
MIT © Brainy Contributors.
|
||||
Contributions welcome — see **[CONTRIBUTING.md](CONTRIBUTING.md)**. MIT © Brainy Contributors.
|
||||
|
|
|
|||
1263
RELEASES.md
1263
RELEASES.md
File diff suppressed because it is too large
Load diff
36
SECURITY.md
36
SECURITY.md
|
|
@ -1,36 +0,0 @@
|
|||
# Security Policy
|
||||
|
||||
## Reporting a vulnerability
|
||||
|
||||
Email **security@soulcraft.com**. That's the one door for security reports
|
||||
across the company, and it works the same way for Brainy: every report is
|
||||
read by a human, you'll get a private receipt, and we'll work with you on
|
||||
coordinated disclosure — please don't open a public issue for anything
|
||||
that isn't already public.
|
||||
|
||||
Include what you'd want if you were on the other end: affected version,
|
||||
how to reproduce, and what you think the impact is. If you have a patch or
|
||||
a suggested fix, send it along — it's welcome but not required.
|
||||
|
||||
There is no bounty program today. We're saying that plainly so you know
|
||||
what to expect going in.
|
||||
|
||||
## Response time
|
||||
|
||||
We respond as fast as truth allows. That means: no fixed SLA, no promise of
|
||||
a reply within a specific number of hours — but a real report from a real
|
||||
person gets read promptly and taken seriously. If you haven't heard anything
|
||||
in a reasonable stretch, a follow-up email is completely fine.
|
||||
|
||||
## Supported versions
|
||||
|
||||
The latest `8.x` minor release line receives security fixes. If you're
|
||||
running an older major version, please upgrade before reporting — we can't
|
||||
commit to backporting fixes to unsupported lines.
|
||||
|
||||
## Scope
|
||||
|
||||
This policy covers the `@soulcraftlabs/brainy` package itself — the code in
|
||||
this repository. If you're evaluating a deployment that also uses
|
||||
`@soulcraft/cor`, report issues in that package the same way, to the same
|
||||
address; we'll route internally.
|
||||
|
|
@ -3,7 +3,7 @@
|
|||
/**
|
||||
* Modern TypeScript CLI Runner
|
||||
*
|
||||
* This is the entry point after npm install @soulcraftlabs/brainy
|
||||
* This is the entry point after npm install @soulcraft/brainy
|
||||
* It runs the compiled TypeScript CLI code
|
||||
*/
|
||||
|
||||
|
|
|
|||
2
bun.lock
2
bun.lock
|
|
@ -3,7 +3,7 @@
|
|||
"configVersion": 0,
|
||||
"workspaces": {
|
||||
"": {
|
||||
"name": "@soulcraftlabs/brainy",
|
||||
"name": "@soulcraft/brainy",
|
||||
"dependencies": {
|
||||
"@aws-sdk/client-s3": "^3.540.0",
|
||||
"@azure/identity": "^4.0.0",
|
||||
|
|
|
|||
|
|
@ -25,13 +25,13 @@
|
|||
|
||||
### Prerequisites
|
||||
```bash
|
||||
npm install @soulcraftlabs/brainy
|
||||
npm install @soulcraft/brainy
|
||||
```
|
||||
|
||||
### Your First Neural Database
|
||||
|
||||
```typescript
|
||||
import { Brainy, NounType } from '@soulcraftlabs/brainy'
|
||||
import { Brainy, NounType } from '@soulcraft/brainy'
|
||||
|
||||
// Step 1: Create and initialize Brainy
|
||||
const brain = new Brainy({
|
||||
|
|
@ -143,7 +143,7 @@ Once you're comfortable with basic operations, move to **Level 2** to learn abou
|
|||
### Building a Knowledge Graph
|
||||
|
||||
```typescript
|
||||
import { Brainy, NounType, VerbType } from '@soulcraftlabs/brainy'
|
||||
import { Brainy, NounType, VerbType } from '@soulcraft/brainy'
|
||||
|
||||
const brain = new Brainy({ storage: { type: 'memory' } })
|
||||
await brain.init()
|
||||
|
|
@ -314,7 +314,7 @@ Ready for AI-powered search and clustering? Move to **Level 3**.
|
|||
### Triple Intelligence in Action
|
||||
|
||||
```typescript
|
||||
import { Brainy, NounType, VerbType } from '@soulcraftlabs/brainy'
|
||||
import { Brainy, NounType, VerbType } from '@soulcraft/brainy'
|
||||
|
||||
const brain = new Brainy({ storage: { type: 'memory' } })
|
||||
await brain.init()
|
||||
|
|
@ -529,7 +529,7 @@ Want to treat files as intelligent entities? Learn the **Virtual Filesystem** in
|
|||
### Files as Intelligent Entities
|
||||
|
||||
```typescript
|
||||
import { Brainy, NounType, VerbType } from '@soulcraftlabs/brainy'
|
||||
import { Brainy, NounType, VerbType } from '@soulcraft/brainy'
|
||||
|
||||
const brain = new Brainy({ storage: { type: 'memory' } })
|
||||
await brain.init()
|
||||
|
|
@ -832,7 +832,7 @@ Ready for production deployment? Level 5 covers **planet-scale architecture**.
|
|||
### Production-Ready Deployment
|
||||
|
||||
```typescript
|
||||
import { Brainy, NounType } from '@soulcraftlabs/brainy'
|
||||
import { Brainy, NounType } from '@soulcraft/brainy'
|
||||
|
||||
// 1. PRODUCTION STORAGE - Filesystem with off-site snapshots
|
||||
console.log('Initializing production storage...\n')
|
||||
|
|
|
|||
|
|
@ -1217,7 +1217,7 @@ where: {
|
|||
await brain.find({ type: 'Document' })
|
||||
|
||||
// ✅ Correct: Use NounType enum
|
||||
import { NounType } from '@soulcraftlabs/brainy'
|
||||
import { NounType } from '@soulcraft/brainy'
|
||||
await brain.find({ type: NounType.Document })
|
||||
|
||||
// ❌ Error: Operator not recognized
|
||||
|
|
|
|||
|
|
@ -153,13 +153,13 @@ brainy-data/
|
|||
### Step 1: Update Brainy Package
|
||||
|
||||
```bash
|
||||
npm install @soulcraftlabs/brainy@latest
|
||||
npm install @soulcraft/brainy@latest
|
||||
```
|
||||
|
||||
**Check your version:**
|
||||
```bash
|
||||
npm list @soulcraftlabs/brainy
|
||||
# Should show: @soulcraftlabs/brainy@4.0.0
|
||||
npm list @soulcraft/brainy
|
||||
# Should show: @soulcraft/brainy@4.0.0
|
||||
```
|
||||
|
||||
### Step 2: No Code Changes Required! ✅
|
||||
|
|
@ -374,7 +374,7 @@ If you encounter issues, you can rollback:
|
|||
|
||||
```bash
|
||||
# Reinstall v3
|
||||
npm install @soulcraftlabs/brainy@^3.50.0
|
||||
npm install @soulcraft/brainy@^3.50.0
|
||||
|
||||
# Restart application
|
||||
```
|
||||
|
|
@ -389,7 +389,7 @@ rm -rf ./data
|
|||
cp -r ./data-backup ./data
|
||||
|
||||
# Reinstall v3
|
||||
npm install @soulcraftlabs/brainy@^3.50.0
|
||||
npm install @soulcraft/brainy@^3.50.0
|
||||
```
|
||||
|
||||
## Common Migration Scenarios
|
||||
|
|
@ -539,7 +539,7 @@ console.log('Storage type:', status.type)
|
|||
|
||||
**Migration Checklist:**
|
||||
- ✅ Backup data
|
||||
- ✅ Update npm package (`npm install @soulcraftlabs/brainy@latest`)
|
||||
- ✅ Update npm package (`npm install @soulcraft/brainy@latest`)
|
||||
- ✅ Restart application (automatic migration)
|
||||
- ✅ Verify data integrity
|
||||
- ✅ Enable lifecycle policies
|
||||
|
|
|
|||
|
|
@ -323,24 +323,58 @@ Only the graph adjacency index carries a committed scale assertion:
|
|||
- ✅ **Single-Node by Design**: One process owns one `path`; scale out at the service layer
|
||||
- ✅ **Zero Stubs**: Every line of code is production-ready
|
||||
|
||||
## Index Build at Open (10.4+)
|
||||
## Lazy Loading Performance
|
||||
|
||||
As of 10.4, `brain.init()` runs every needed index rebuild to completion before
|
||||
it returns — always, regardless of dataset size. There is no lazy,
|
||||
first-query rebuild path: a brain either finishes opening healthy, or `init()`
|
||||
fails loudly. `disableAutoRebuild` no longer defers index construction to a
|
||||
first query; it has no effect on *when* a rebuild runs. Manual control over
|
||||
rebuilds is `repairIndex({ rebuild: [...] })`. See
|
||||
[Index Health](concepts/index-health.md) for the full read-gate contract
|
||||
(providers self-report readiness via `healthReport()`; a read against a
|
||||
not-serving provider throws a typed `*NotReadyError` rather than rebuilding
|
||||
mid-query).
|
||||
Brainy supports two initialization modes for optimal performance across different use cases:
|
||||
|
||||
<!-- The pre-10.4 "Mode 2: Lazy Loading on First Query" section previously
|
||||
documented here (disableAutoRebuild deferring index construction to the
|
||||
first find() call) described a real, now-retired code path. Removed
|
||||
rather than left to mislead; the concept doc above is the current
|
||||
contract. -->
|
||||
### Mode 1: Auto-Rebuild (Default)
|
||||
|
||||
```javascript
|
||||
const brain = new Brainy()
|
||||
await brain.init() // Rebuilds indexes during init (~500ms-3s for 10K entities)
|
||||
```
|
||||
|
||||
**Performance:**
|
||||
- Init time: 500ms-3s (depends on dataset size)
|
||||
- First query: Instant (indexes already loaded)
|
||||
- Use case: Traditional applications, long-running servers
|
||||
|
||||
### Mode 2: Lazy Loading
|
||||
|
||||
```javascript
|
||||
const brain = new Brainy({ disableAutoRebuild: true })
|
||||
await brain.init() // Returns instantly (0-10ms)
|
||||
|
||||
const results = await brain.find({ limit: 10 }) // First query triggers rebuild (~50-200ms)
|
||||
const more = await brain.find({ limit: 100 }) // Subsequent queries instant (0ms check)
|
||||
```
|
||||
|
||||
**Performance:**
|
||||
- Init time: 0-10ms (instant)
|
||||
- First query: 50-200ms (includes index rebuild for 1K-10K entities)
|
||||
- Subsequent queries: 0ms check (instant)
|
||||
- Concurrent queries: Wait for same rebuild (mutex prevents duplicates)
|
||||
|
||||
**Concurrency Safety:**
|
||||
```javascript
|
||||
// 100 concurrent queries immediately after init
|
||||
await brain.init()
|
||||
|
||||
const promises = Array.from({ length: 100 }, () =>
|
||||
brain.find({ limit: 10 })
|
||||
)
|
||||
|
||||
const results = await Promise.all(promises)
|
||||
// ✅ Only 1 rebuild triggered (mutex)
|
||||
// ✅ All 100 queries return correct results
|
||||
// ✅ Total time: ~60ms (not 6000ms!)
|
||||
```
|
||||
|
||||
**Use Cases for Lazy Loading:**
|
||||
- **Serverless/Edge**: Minimize cold start time (0-10ms init)
|
||||
- **Development**: Faster restarts during development
|
||||
- **Large datasets**: Defer index loading until needed
|
||||
- **Read-heavy workloads**: Writes don't wait for index rebuild
|
||||
|
||||
## Zero Configuration Required
|
||||
|
||||
|
|
@ -350,6 +384,10 @@ Brainy is designed to be **smart enough to tune itself dynamically**. No configu
|
|||
// That's it. Brainy handles everything.
|
||||
const brain = new Brainy()
|
||||
await brain.init()
|
||||
|
||||
// Or with lazy loading for serverless
|
||||
const brain = new Brainy({ disableAutoRebuild: true })
|
||||
await brain.init() // Instant (0-10ms)
|
||||
```
|
||||
|
||||
### Automatic Self-Tuning
|
||||
|
|
@ -357,6 +395,7 @@ await brain.init()
|
|||
- **Metadata Index**: Auto-builds sorted indices for range queries on first use
|
||||
- **Graph Index**: Auto-flushes every 30 seconds
|
||||
- **Default Tuning**: Research-based vector index defaults
|
||||
- **Lazy Loading**: Indices built only when needed
|
||||
- **Cache Management**: LRU caches with TTL
|
||||
|
||||
### Intelligent Defaults
|
||||
|
|
|
|||
|
|
@ -10,7 +10,7 @@ next:
|
|||
- guides/storage-adapters
|
||||
---
|
||||
|
||||
# Plugin System
|
||||
# Plugin Development Guide
|
||||
|
||||
Brainy has a plugin system that allows third-party packages to replace internal subsystems with custom implementations. This is how `@soulcraft/cor` provides optional native acceleration, and it's the same system available to any developer.
|
||||
|
||||
|
|
@ -46,7 +46,7 @@ If no plugin provides a given key, brainy uses its built-in JavaScript implement
|
|||
### 1. Implement the `BrainyPlugin` interface
|
||||
|
||||
```typescript
|
||||
import type { BrainyPlugin, BrainyPluginContext } from '@soulcraftlabs/brainy/plugin'
|
||||
import type { BrainyPlugin, BrainyPluginContext } from '@soulcraft/brainy/plugin'
|
||||
|
||||
const myPlugin: BrainyPlugin = {
|
||||
name: 'my-brainy-plugin', // Must be unique (typically your npm package name)
|
||||
|
|
@ -90,7 +90,7 @@ await brain.init()
|
|||
**Programmatic registration:** For plugins not installed as npm packages, use `brain.use()`:
|
||||
|
||||
```typescript
|
||||
import { Brainy } from '@soulcraftlabs/brainy'
|
||||
import { Brainy } from '@soulcraft/brainy'
|
||||
import myPlugin from './my-plugin.js'
|
||||
|
||||
const brain = new Brainy()
|
||||
|
|
@ -200,30 +200,15 @@ members so a warm reopen never pays a redundant rebuild-from-canonical:
|
|||
- **`init?(): Promise<void>`** — eager cold-load. Brainy awaits it once during
|
||||
`brain.init()`, after the metadata provider's `init()` (the id-mapper hydrates first)
|
||||
and **before the rebuild gate**.
|
||||
- **`healthReport?(): HealthReport`** — the PREFERRED signal (10.4+). A named,
|
||||
synchronous, O(1) verdict derived from the provider's own exact ledgers — never a
|
||||
sample, never I/O, must never throw for a well-formed provider. Brainy's read gate
|
||||
(`assessProviderHealth()`) reads this INSTEAD of `isReady()` / size heuristics when
|
||||
present: `serving: false` refuses the read with a typed `*NotReadyError` rather than
|
||||
triggering a rebuild — a read never starts a store walk. `healthy` marks every
|
||||
*verified* invariant holding; a family named in `unledgered` counts as neither
|
||||
healthy nor broken. See `HealthReport` / `LedgerInvariantResult` /
|
||||
`InvariantSource` in `src/plugin.ts`, and
|
||||
[Index Health](concepts/index-health.md) for the consumer-facing story.
|
||||
- **`isReady?(): boolean`** — honest durability signal, the fallback when
|
||||
`healthReport()` is absent. `true` ⇔ the persisted index is loaded (or cheaply
|
||||
demand-loadable) and consistent with what was last persisted. When exposed, the
|
||||
gate defers to this signal **instead of** the `size() === 0` / `totalEntries === 0`
|
||||
heuristics — a disk-native index may report 0 resident entries while fully durable.
|
||||
Never return `true` if the durable state failed to load: the signal is honest in
|
||||
both directions, and a not-ready provider gets its rebuild even when `size() > 0`.
|
||||
- **`isReady?(): boolean`** — honest durability signal. `true` ⇔ the persisted index is
|
||||
loaded (or cheaply demand-loadable) and consistent with what was last persisted. When
|
||||
exposed, the rebuild gate defers to this signal **instead of** the `size() === 0` /
|
||||
`totalEntries === 0` heuristics — a disk-native index may report 0 resident entries
|
||||
while fully durable. Never return `true` if the durable state failed to load: the
|
||||
signal is honest in both directions, and a not-ready provider gets its rebuild even
|
||||
when `size() > 0`.
|
||||
- **`isMigrating?(): boolean`** — while `true`, the provider owns its index (background
|
||||
migration); brainy skips its rebuild entirely.
|
||||
- **`validateInvariants?(): Promise<ProviderInvariantReport>`** — the async DEEP
|
||||
diagnostic (full scans allowed), distinct from the bounded, sync `healthReport()`.
|
||||
Must never throw — a failure is `healthy: false` data, not an exception; a provider
|
||||
that throws anyway is read as a loud, unverified failure (never as "healthy") by
|
||||
every caller, never silently retried into a rebuild.
|
||||
|
||||
Providers that implement none of these keep the size/count heuristics — correct for
|
||||
engines whose `rebuild()` *is* their load path (like brainy's built-in JS vector index).
|
||||
|
|
@ -272,10 +257,10 @@ When provided by an optional native acceleration plugin (such as `@soulcraft/cor
|
|||
#### `cache`
|
||||
**Type:** `UnifiedCache`
|
||||
|
||||
Replaces the global `UnifiedCache` singleton used for VFS path resolution, semantic caching, and vector index caching. Must implement the `UnifiedCache` interface (available from `@soulcraftlabs/brainy/internals`).
|
||||
Replaces the global `UnifiedCache` singleton used for VFS path resolution, semantic caching, and vector index caching. Must implement the `UnifiedCache` interface (available from `@soulcraft/brainy/internals`).
|
||||
|
||||
```typescript
|
||||
import type { UnifiedCache } from '@soulcraftlabs/brainy/internals'
|
||||
import type { UnifiedCache } from '@soulcraft/brainy/internals'
|
||||
|
||||
context.registerProvider('cache', myNativeCache)
|
||||
```
|
||||
|
|
@ -325,8 +310,8 @@ Plugins can register custom storage backends that users reference by name.
|
|||
### Implementing a Storage Adapter
|
||||
|
||||
```typescript
|
||||
import type { StorageAdapterFactory } from '@soulcraftlabs/brainy/plugin'
|
||||
import type { StorageAdapter } from '@soulcraftlabs/brainy'
|
||||
import type { StorageAdapterFactory } from '@soulcraft/brainy/plugin'
|
||||
import type { StorageAdapter } from '@soulcraft/brainy'
|
||||
|
||||
class MyStorageAdapter implements StorageAdapter {
|
||||
async init(): Promise<void> { /* ... */ }
|
||||
|
|
@ -360,9 +345,9 @@ Brainy provides three entry points for plugin developers:
|
|||
|
||||
| Import Path | Contents | Stability |
|
||||
|-------------|----------|-----------|
|
||||
| `@soulcraftlabs/brainy` | Public API, types, StorageAdapter | Stable (semver) |
|
||||
| `@soulcraftlabs/brainy/plugin` | BrainyPlugin, BrainyPluginContext, StorageAdapterFactory | Stable (semver) |
|
||||
| `@soulcraftlabs/brainy/internals` | UnifiedCache, EntityIdMapper, logger utilities | Internal (may change between minor versions) |
|
||||
| `@soulcraft/brainy` | Public API, types, StorageAdapter | Stable (semver) |
|
||||
| `@soulcraft/brainy/plugin` | BrainyPlugin, BrainyPluginContext, StorageAdapterFactory | Stable (semver) |
|
||||
| `@soulcraft/brainy/internals` | UnifiedCache, EntityIdMapper, logger utilities | Internal (may change between minor versions) |
|
||||
|
||||
## Diagnostics
|
||||
|
||||
|
|
@ -440,7 +425,7 @@ A minimal but useful plugin that provides SIMD-accelerated distance calculations
|
|||
|
||||
```typescript
|
||||
// simd-distance-plugin/src/plugin.ts
|
||||
import type { BrainyPlugin, BrainyPluginContext } from '@soulcraftlabs/brainy/plugin'
|
||||
import type { BrainyPlugin, BrainyPluginContext } from '@soulcraft/brainy/plugin'
|
||||
|
||||
// Hypothetical native module
|
||||
import { simdCosineDistance } from './native.js'
|
||||
|
|
@ -470,7 +455,7 @@ export default simdDistancePlugin
|
|||
"main": "./dist/plugin.js",
|
||||
"types": "./dist/plugin.d.ts",
|
||||
"peerDependencies": {
|
||||
"@soulcraftlabs/brainy": ">=7.0.0"
|
||||
"@soulcraft/brainy": ">=7.0.0"
|
||||
}
|
||||
}
|
||||
```
|
||||
|
|
@ -478,7 +463,7 @@ export default simdDistancePlugin
|
|||
Usage:
|
||||
|
||||
```typescript
|
||||
import { Brainy } from '@soulcraftlabs/brainy'
|
||||
import { Brainy } from '@soulcraft/brainy'
|
||||
|
||||
const brain = new Brainy({ plugins: ['brainy-simd-distance'] })
|
||||
await brain.init()
|
||||
|
|
|
|||
|
|
@ -54,7 +54,7 @@ After 40 API calls:
|
|||
|
||||
```typescript
|
||||
// server.ts
|
||||
import { Brainy } from '@soulcraftlabs/brainy'
|
||||
import { Brainy } from '@soulcraft/brainy'
|
||||
|
||||
// SINGLETON INSTANCE
|
||||
let brainInstance: Brainy | null = null
|
||||
|
|
@ -174,7 +174,7 @@ process.on('SIGTERM', async () => {
|
|||
|
||||
```typescript
|
||||
// server.ts - Clean Bun implementation
|
||||
import { Brainy } from '@soulcraftlabs/brainy'
|
||||
import { Brainy } from '@soulcraft/brainy'
|
||||
|
||||
let brain: Brainy | null = null
|
||||
|
||||
|
|
|
|||
|
|
@ -5,7 +5,7 @@
|
|||
## Quick Start
|
||||
|
||||
```typescript
|
||||
import { Brainy, NounType, VerbType } from '@soulcraftlabs/brainy'
|
||||
import { Brainy, NounType, VerbType } from '@soulcraft/brainy'
|
||||
|
||||
const brain = new Brainy()
|
||||
await brain.init()
|
||||
|
|
|
|||
|
|
@ -99,7 +99,7 @@ Examples:
|
|||
|
||||
```bash
|
||||
# 1. Deprecate wrong version on npm
|
||||
npm deprecate @soulcraftlabs/brainy@X.X.X "Incorrect version - use Y.Y.Y"
|
||||
npm deprecate @soulcraft/brainy@X.X.X "Incorrect version - use Y.Y.Y"
|
||||
|
||||
# 2. Fix version in package.json
|
||||
# 3. Republish correct version
|
||||
|
|
|
|||
|
|
@ -13,7 +13,7 @@
|
|||
|
||||
### In-Memory
|
||||
```typescript
|
||||
import Brainy from '@soulcraftlabs/brainy'
|
||||
import Brainy from '@soulcraft/brainy'
|
||||
const brain = new Brainy({ storage: { type: 'memory' } })
|
||||
```
|
||||
|
||||
|
|
@ -43,7 +43,7 @@ The native vector provider (via the optional `@soulcraft/cor` package) extends t
|
|||
|
||||
Numbers below are **measured** by `tests/benchmarks/find-composition-scale.js` (a single
|
||||
Node 22 process, in-memory storage, 384-dim vectors, `balanced` recall). They are the
|
||||
open-core (pure-TypeScript) path — what you get from `@soulcraftlabs/brainy` with no native
|
||||
open-core (pure-TypeScript) path — what you get from `@soulcraft/brainy` with no native
|
||||
provider installed. Run it yourself: `node --max-old-space-size=8192 tests/benchmarks/find-composition-scale.js 100000`.
|
||||
|
||||
`find()` query latency, p50 / p95 (200 queries each):
|
||||
|
|
|
|||
File diff suppressed because it is too large
Load diff
|
|
@ -24,7 +24,7 @@ next:
|
|||
## Quick Start
|
||||
|
||||
```typescript
|
||||
import { Brainy, NounType, VerbType } from '@soulcraftlabs/brainy'
|
||||
import { Brainy, NounType, VerbType } from '@soulcraft/brainy'
|
||||
|
||||
const brain = new Brainy() // Zero config!
|
||||
await brain.init() // VFS auto-initialized!
|
||||
|
|
@ -1010,7 +1010,7 @@ await db.release() // unpin + free cached materialization
|
|||
|
||||
### Db API errors
|
||||
|
||||
All exported from `@soulcraftlabs/brainy`:
|
||||
All exported from `@soulcraft/brainy`:
|
||||
|
||||
| Error | Thrown by | Meaning |
|
||||
|---|---|---|
|
||||
|
|
@ -1451,34 +1451,6 @@ const count = await brain.getVerbCount()
|
|||
|
||||
---
|
||||
|
||||
### The canonical count ledger (`StorageAdapter.getCanonicalCounts()`)
|
||||
|
||||
An OPTIONAL method on the `StorageAdapter` interface (implemented by both
|
||||
built-in adapters), not a method on `Brainy` itself — relevant if you're
|
||||
writing a custom storage adapter or composing a provider's own
|
||||
`healthReport()`. O(1), no I/O. Per family (`nouns`/`verbs`):
|
||||
|
||||
```typescript
|
||||
interface CanonicalCounts {
|
||||
nouns: { counted: number; all: number }
|
||||
verbs: { counted: number; all: number }
|
||||
suspect: boolean
|
||||
}
|
||||
```
|
||||
|
||||
- `counted` mirrors `getNounCount()` / `getVerbCount()` (public + internal tiers).
|
||||
- `all` is the ALL-visibility scalar — every tier, including system/internal
|
||||
records — the denominator a derived index's own coverage math is measured
|
||||
against.
|
||||
- `suspect` is `true` when an unprovable delete has left `all` unverified since
|
||||
the last recount; `brain.repairIndex()` clears it with a real canonical walk.
|
||||
|
||||
Adapters without the ledger omit the method; treat absence as "no
|
||||
denominator," never as zero. See
|
||||
**[Index Health](../concepts/index-health.md)** for the full story.
|
||||
|
||||
---
|
||||
|
||||
### Subtype & facet APIs
|
||||
|
||||
Full guide: **[Subtypes & Facets](../guides/subtypes-and-facets.md)**.
|
||||
|
|
@ -1859,104 +1831,6 @@ const semanticOnly = await brain.getStats({ excludeVFS: true })
|
|||
|
||||
---
|
||||
|
||||
### `repairIndex(options?)` → `Promise<RepairReport>`
|
||||
|
||||
The ceremony door for index repair. Bare `repairIndex()` is report-driven: it
|
||||
prunes orphaned containers, recomputes count rollups, reconciles VFS
|
||||
containment, and rebuilds only a derived-index family whose own health check
|
||||
asks for it. Pass `options.rebuild` to force one or more families to rebuild
|
||||
UNCONDITIONALLY — no health check is consulted — when an operator has
|
||||
independent reason to reconcile a family regardless of what it self-reports.
|
||||
|
||||
```typescript
|
||||
// Report-driven: only heals what actually needs it
|
||||
const report = await brain.repairIndex()
|
||||
console.log(report.healedTotal, report.families)
|
||||
|
||||
// Explicit: force the graph adjacency to rebuild from canonical, unconditionally
|
||||
await brain.repairIndex({ rebuild: ['graph'] })
|
||||
|
||||
// Explicit: force all three derived indexes to rebuild
|
||||
await brain.repairIndex({ rebuild: 'all' })
|
||||
```
|
||||
|
||||
**`RepairReport`:**
|
||||
- `families: RepairFamilyReport[]` — one row per family checked
|
||||
- `healedTotal: number` — items healed across every family
|
||||
- `durationMs: number`
|
||||
|
||||
**`RepairFamilyReport`** (one row):
|
||||
- `family: string` — e.g. `'orphaned-containers'`, `'count-rollups'`,
|
||||
`'vfs-containment'`, `'metadata-corruption'`, `'provider:metadata'`,
|
||||
`'provider:graph'`, `'provider:vector'`
|
||||
- `checked: boolean` — was this family actually examined (`false` ⇒ see `skipped`)
|
||||
- `healed: number` — items re-posted/corrected in place (the incremental heal count)
|
||||
- `missing?: { count: number; sample: string[] }` — exact count plus a capped id
|
||||
sample when the check can name what diverged (never the full list)
|
||||
- `rebuilt?: boolean` — a full generational rebuild ran (vs. an incremental heal)
|
||||
- `detail?: string` / `reason?: string` — narration
|
||||
- `skipped?: string` — why the family wasn't checked
|
||||
|
||||
Full walkthrough — what each family checks, degraded-but-serving vs. not-ready,
|
||||
and what `suspect` counts mean — in
|
||||
**[Index Health](../concepts/index-health.md)**.
|
||||
|
||||
---
|
||||
|
||||
### Index readiness: typed errors, `healthReport()`, `disableAutoRebuild`
|
||||
|
||||
Every derived-index provider (vector, graph, metadata) may expose a named,
|
||||
synchronous, O(1) `healthReport()` composed from its own exact ledgers — the
|
||||
signal Brainy's read gate trusts over sampling or size heuristics. `init()`
|
||||
brings every provider to serving before it returns; there is no first-query
|
||||
lazy-rebuild path. A read that reaches a provider whose health report says it
|
||||
isn't serving throws instead of rebuilding mid-query:
|
||||
|
||||
| Error | Thrown by | Meaning |
|
||||
|---|---|---|
|
||||
| `GraphIndexNotReadyError` | `find({ connected })`, `neighbors()`, `related()` | Graph adjacency isn't serving |
|
||||
| `MetadataIndexNotReadyError` | `find({ where })` | Metadata/field index isn't serving |
|
||||
| `VectorIndexNotReadyError` | `find({ query })`, `similar()` | Vector index isn't serving |
|
||||
|
||||
All three are exported from `@soulcraftlabs/brainy`. Catch them to distinguish
|
||||
"index not ready" from a genuine empty result:
|
||||
|
||||
```typescript
|
||||
import { MetadataIndexNotReadyError } from '@soulcraftlabs/brainy'
|
||||
|
||||
try {
|
||||
const rows = await brain.find({ where: { status: 'active' } })
|
||||
} catch (err) {
|
||||
if (err instanceof MetadataIndexNotReadyError) {
|
||||
// reconcile: await brain.repairIndex(), then retry
|
||||
} else {
|
||||
throw err
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
**`disableAutoRebuild`** no longer defers index construction to the first
|
||||
query. A needed rebuild always runs at `open()`, regardless of this flag or
|
||||
dataset size; the flag has no effect on *when* a rebuild runs. Full manual
|
||||
control lives in `repairIndex({ rebuild: [...] })`, above.
|
||||
|
||||
### `validateIndexConsistency()` → `Promise<...>`
|
||||
|
||||
The deep, async diagnostic counterpart to `healthReport()` — safe to run on a
|
||||
live brain, but does more work (a provider's `validateInvariants()` may run a
|
||||
full scan, not just read a ledger). Aggregates the JS metadata index's own
|
||||
consistency check with every derived-index provider's invariant report.
|
||||
|
||||
```typescript
|
||||
const validation = await brain.validateIndexConsistency()
|
||||
if (!validation.healthy) {
|
||||
console.log(validation.recommendation) // what to run, e.g. repairIndex()
|
||||
console.log(validation.providers) // each provider's own invariant report, when exposed
|
||||
}
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Lifecycle
|
||||
|
||||
### Initialization
|
||||
|
|
@ -2208,7 +2082,7 @@ For the full taxonomy with all 169 types and their descriptions, see:
|
|||
- **📖 Documentation:** [Full Documentation](../)
|
||||
- **🐛 Issues:** [GitHub Issues](https://github.com/soulcraftlabs/brainy/issues)
|
||||
- **💬 Discussions:** [GitHub Discussions](https://github.com/soulcraftlabs/brainy/discussions)
|
||||
- **📦 NPM:** [@soulcraftlabs/brainy](https://www.npmjs.com/package/@soulcraftlabs/brainy)
|
||||
- **📦 NPM:** [@soulcraft/brainy](https://www.npmjs.com/package/@soulcraft/brainy)
|
||||
- **⭐ GitHub:** [Star us](https://github.com/soulcraftlabs/brainy)
|
||||
|
||||
---
|
||||
|
|
|
|||
|
|
@ -268,7 +268,7 @@ locks/_flush_responses/ # writer answers with <uuid>.ack
|
|||
| **Counts/statistics** | Per-type and per-subtype maps | `_system/{type,subtype,verb-subtype}-statistics.json.gz`, `counts.json` | Recomputable by scanning entities (`brainy inspect repair`) |
|
||||
|
||||
A pluggable index provider (the 8.0 plugin contract in
|
||||
`@soulcraftlabs/brainy/plugin`) may replace any of the JS implementations; the
|
||||
`@soulcraft/brainy/plugin`) may replace any of the JS implementations; the
|
||||
persisted formats above are contract-bound so JS and native implementations
|
||||
can interleave on the same directory.
|
||||
|
||||
|
|
|
|||
|
|
@ -126,7 +126,7 @@ class TypeAwareMetadataIndex {
|
|||
**The Design**: Specify types clearly in your API calls:
|
||||
|
||||
```typescript
|
||||
import { Brainy, NounType, VerbType } from '@soulcraftlabs/brainy'
|
||||
import { Brainy, NounType, VerbType } from '@soulcraft/brainy'
|
||||
|
||||
// Add entity with explicit type
|
||||
await brain.add({
|
||||
|
|
@ -231,7 +231,7 @@ class OrgEnrichmentAugmentation {
|
|||
**Brainy's Approach**: Extract **typed** concepts:
|
||||
|
||||
```typescript
|
||||
import { NaturalLanguageProcessor } from '@soulcraftlabs/brainy'
|
||||
import { NaturalLanguageProcessor } from '@soulcraft/brainy'
|
||||
|
||||
const nlp = new NaturalLanguageProcessor()
|
||||
const concepts = await nlp.extractConcepts("Alice works at Google in San Francisco")
|
||||
|
|
@ -382,7 +382,7 @@ import {
|
|||
getVerbTypes,
|
||||
BrainyTypes,
|
||||
suggestType
|
||||
} from '@soulcraftlabs/brainy'
|
||||
} from '@soulcraft/brainy'
|
||||
|
||||
// Get all available noun types
|
||||
const nounTypes = getNounTypes()
|
||||
|
|
|
|||
|
|
@ -723,14 +723,6 @@ async stats(): Promise<Statistics> {
|
|||
|
||||
### 5. Index Rebuilding (Lazy Loading Support)
|
||||
|
||||
> **Stale as of 10.4 — "Mode 2: Lazy Loading on First Query" below is
|
||||
> RETIRED.** `disableAutoRebuild` no longer defers index construction to a
|
||||
> first query; `brain.init()` now runs every needed rebuild to completion
|
||||
> before it returns, unconditionally, and a read against a not-serving
|
||||
> provider throws a typed `*NotReadyError` instead of rebuilding mid-query.
|
||||
> See `docs/concepts/index-health.md` for the current contract. Left below
|
||||
> as historical background on the rebuild mechanics.
|
||||
|
||||
**Two modes of index loading:**
|
||||
|
||||
#### Mode 1: Auto-Rebuild on init() (default)
|
||||
|
|
|
|||
|
|
@ -1,15 +1,5 @@
|
|||
# Initialization and Rebuild Processes
|
||||
|
||||
> **Stale as of 10.4 — "Mode 2: Lazy Loading on First Query" below is RETIRED.**
|
||||
> `disableAutoRebuild` no longer defers index construction to a first query;
|
||||
> `brain.init()` now runs every needed rebuild to completion before it
|
||||
> returns, unconditionally. A read against a not-serving provider throws a
|
||||
> typed `*NotReadyError` instead of rebuilding mid-query. See
|
||||
> `docs/concepts/index-health.md` for the current contract; this document's
|
||||
> line-number references to `src/brainy.ts` also predate the file's current
|
||||
> size and are unreliable. Left as historical background on the rebuild
|
||||
> mechanics, not as a current API description.
|
||||
|
||||
This document explains how Brainy's four indexes (MetadataIndex, vector index, GraphAdjacencyIndex, DeletedItemsIndex) initialize and rebuild from persisted storage.
|
||||
|
||||
## Core Principle: All Indexes Are Disk-Based
|
||||
|
|
|
|||
|
|
@ -127,7 +127,7 @@ For reference, a clean migration path:
|
|||
`isMultiProcessSafe` type-guard. Keep `hasStorageMethod` for
|
||||
build/install artifact protection.
|
||||
5. Document the new contract in `concepts/storage-adapters.md`.
|
||||
6. Major-version-bump the `@soulcraftlabs/brainy` peerDep range expected by
|
||||
6. Major-version-bump the `@soulcraft/brainy` peerDep range expected by
|
||||
plugins.
|
||||
|
||||
Estimated work: ~half a day of code, ~2 hours of doc/example updates,
|
||||
|
|
|
|||
|
|
@ -20,7 +20,7 @@ next:
|
|||
Every example on this page is written against the real Brainy 8.0 API. The setup is always the same:
|
||||
|
||||
```typescript
|
||||
import { Brainy, NounType, VerbType } from '@soulcraftlabs/brainy'
|
||||
import { Brainy, NounType, VerbType } from '@soulcraft/brainy'
|
||||
|
||||
const brain = new Brainy()
|
||||
await brain.init()
|
||||
|
|
@ -40,7 +40,7 @@ Brainy's **Noun-Verb Taxonomy** achieves broad coverage of human knowledge throu
|
|||
- **Multi-hop Graph Traversals = Relationship Complexity**
|
||||
- **Result: Model data across virtually any industry**
|
||||
|
||||
Every piece of information can be represented as entities (nouns) connected by relationships (verbs) carrying properties (metadata). The standardized type system from `@soulcraftlabs/brainy` (`NounType`, `VerbType`) gives those nouns and verbs a stable, shared name.
|
||||
Every piece of information can be represented as entities (nouns) connected by relationships (verbs) carrying properties (metadata). The standardized type system from `@soulcraft/brainy` (`NounType`, `VerbType`) gives those nouns and verbs a stable, shared name.
|
||||
|
||||
## The Power of Standardization: Universal Interoperability
|
||||
|
||||
|
|
|
|||
|
|
@ -35,7 +35,7 @@ constructor and `init()`.
|
|||
## Instant Start
|
||||
|
||||
```typescript
|
||||
import { Brainy } from '@soulcraftlabs/brainy'
|
||||
import { Brainy } from '@soulcraft/brainy'
|
||||
|
||||
// That's it. No config needed.
|
||||
const brain = new Brainy()
|
||||
|
|
|
|||
|
|
@ -1,233 +0,0 @@
|
|||
---
|
||||
title: Field addressing: your fields and system fields
|
||||
slug: concepts/field-addressing
|
||||
public: true
|
||||
category: concepts
|
||||
template: concept
|
||||
order: 7
|
||||
description: The one rule for every query-surface field name — a bare name always means your metadata, system.<field> reaches the ten engine scalars explicitly, and anything else refuses by name.
|
||||
next:
|
||||
- guides/namespace-migration
|
||||
- concepts/consistency-model
|
||||
---
|
||||
|
||||
# Field addressing: your fields and system fields
|
||||
|
||||
Every query surface in Brainy — `find()`'s `where`, `orderBy`, aggregation
|
||||
`groupBy`, and aggregation `source.where` — resolves field names by one rule,
|
||||
with no exceptions:
|
||||
|
||||
> **A bare field name always means your metadata. `system.<field>` reaches an
|
||||
> engine scalar, and only when you spell it explicitly.**
|
||||
|
||||
```typescript
|
||||
await brain.find({ orderBy: 'level' }) // reads entity.metadata.level — YOUR field
|
||||
await brain.find({ orderBy: 'system.createdAt' }) // reads the engine's createdAt scalar
|
||||
await brain.find({ orderBy: 'metadata.level' }) // identical to bare 'level' — explicit scope
|
||||
```
|
||||
|
||||
There is no priority list, no "try the system field, fall back to metadata"
|
||||
behavior, and no name that resolves differently depending on what else
|
||||
happens to exist on your entities. A field called `level`, `score`,
|
||||
`createdAt`, or `type` in your own `metadata` is read as *your* field, every
|
||||
time, by its bare name.
|
||||
|
||||
## Why this rule exists
|
||||
|
||||
An internal report from a production deployment found that a user metadata
|
||||
field literally named `level` was being silently shadowed by the engine's
|
||||
own internal index layer field of the same name — every sort by `level`
|
||||
returned insertion order, with no error raised. This rule makes that class of
|
||||
bug structurally impossible: bare names belong to you, unconditionally, and
|
||||
anything that isn't yours has to be spelled out.
|
||||
|
||||
## The system scalars
|
||||
|
||||
`system.<field>` addresses exactly ten scalars on an entity — no more, no
|
||||
fewer:
|
||||
|
||||
| System field | What it is |
|
||||
|---|---|
|
||||
| `system.id` | The entity's id |
|
||||
| `system.type` | The entity's `NounType` |
|
||||
| `system.subtype` | The per-app sub-classification passed to `add()` |
|
||||
| `system.createdAt` | When the entity was created |
|
||||
| `system.updatedAt` | When the entity was last written |
|
||||
| `system.confidence` | The `confidence` param (0–1) |
|
||||
| `system.weight` | The `weight` param |
|
||||
| `system.visibility` | `'public'` / `'internal'` (see the visibility tiers in [Consistency Model](./consistency-model.md)) |
|
||||
| `system.service` | The multi-tenancy `service` tag |
|
||||
| `system.createdBy` | Who/what created the entity |
|
||||
|
||||
Relationships mirror the same eight shared scalars (`subtype`, `createdAt`,
|
||||
`updatedAt`, `confidence`, `weight`, `visibility`, `service`, `createdBy`)
|
||||
plus three of their own:
|
||||
|
||||
| System field (relationship) | What it is |
|
||||
|---|---|
|
||||
| `system.verb` | The relationship's `VerbType` |
|
||||
| `system.sourceId` | The id of the entity the relationship starts from |
|
||||
| `system.targetId` | The id of the entity the relationship points to |
|
||||
|
||||
Anything not on these two lists is not a system scalar — `system.<name>` for
|
||||
any other name refuses (see "Refusal semantics" below), even if that name
|
||||
sounds like it should be engine-owned.
|
||||
|
||||
## Invisible plumbing — never addressable, in either spelling
|
||||
|
||||
Five names are pure engine internals. They are not reachable as a bare name,
|
||||
and not reachable as `system.<name>` either — they simply have no place on
|
||||
the query surface:
|
||||
|
||||
- **`vector`** — the stored embedding. It participates in similarity search
|
||||
(`query`, `near`, vector `find()`), never in `where`/`orderBy`/`groupBy`.
|
||||
- **`connections`** — graph adjacency. Reached through `connected` and
|
||||
`brain.related()`, not through field addressing.
|
||||
- **`level`** — the internal index layer number used by the nearest-neighbor
|
||||
graph. It is pure index plumbing with no query-surface meaning at all —
|
||||
which is exactly why a user field of the same name must never be shadowed
|
||||
by it. `level` as a bare name is always yours; there is no engine-owned
|
||||
spelling of it to compete with.
|
||||
- **`data`** — your entity's content payload, not a scalar. It can be a
|
||||
string, a number, or an arbitrary object, so sorting or filtering it as a
|
||||
single comparable value would lie about its actual shape. Content is
|
||||
reached through the content/text-search APIs (`query`, `searchMode:
|
||||
'text'`), not through `where`/`orderBy`.
|
||||
- **`_rev`** — the per-entity revision counter used for optimistic
|
||||
concurrency (`ifRev`). It is a CAS token, not a queryable dimension.
|
||||
|
||||
`system.level`, `system.vector`, and `system.data` all refuse for the same
|
||||
reason: they are not in the ten-scalar system map, full stop.
|
||||
|
||||
## `metadata.<field>` — the explicit spelling of "mine"
|
||||
|
||||
Prefix any field with `metadata.` to say the same thing a bare name already
|
||||
says, spelled out. The two are interchangeable everywhere a field name is
|
||||
accepted, including `orderBy`:
|
||||
|
||||
```typescript
|
||||
await brain.find({ where: { 'customer.tier': 'gold' } })
|
||||
await brain.find({ where: { 'metadata.customer.tier': 'gold' } }) // identical
|
||||
await brain.find({ orderBy: 'metadata.score', order: 'desc' }) // identical to orderBy: 'score'
|
||||
```
|
||||
|
||||
Reach for the explicit spelling when it reads more clearly next to a
|
||||
`system.` field in the same query — for example, sorting by your own `score`
|
||||
while filtering on `system.confidence`.
|
||||
|
||||
## No special names — the write side
|
||||
|
||||
The same law governs writes:
|
||||
|
||||
> **Data is either in main space, where developers can use anything, or it
|
||||
> is in `system.*`.**
|
||||
|
||||
There are **no reserved metadata names**. A field called `confidence`,
|
||||
`type`, `id`, `data`, `content`, or anything else inside your `metadata` bag
|
||||
is an ordinary user field: it is stored verbatim, indexed, filterable,
|
||||
sortable, aggregatable, and it survives restarts, index rebuilds, and
|
||||
time-travel (`asOf`) reads exactly as written — even when an engine scalar
|
||||
shares its spelling. The engine's values are written only through their
|
||||
dedicated params (`confidence`, `weight`, `subtype`, `visibility`, …) and
|
||||
read at `system.<field>`; your bag can never touch them and they can never
|
||||
shadow your bag.
|
||||
|
||||
```typescript
|
||||
const id = await brain.add({
|
||||
data: 'Ada Lovelace',
|
||||
type: NounType.Person,
|
||||
confidence: 0.9, // the ENGINE scalar
|
||||
metadata: { confidence: 'self-rated' } // YOUR field, same spelling — both live
|
||||
})
|
||||
|
||||
await brain.find({ where: { confidence: 'self-rated' } }) // finds it (yours)
|
||||
await brain.find({ where: { 'system.confidence': 0.9 } }) // finds it (engine's)
|
||||
```
|
||||
|
||||
The one spelling a write refuses is a metadata key that literally starts
|
||||
with `system.` — the explicit address namespace cannot be forged as a user
|
||||
field name. That refusal is typed and names the fix.
|
||||
|
||||
Value **shape** rules still apply uniformly to every name (they are not name
|
||||
carve-outs): arrays longer than 10 elements are not turned into posting-list
|
||||
scalars, and very long values are indexed by hash.
|
||||
|
||||
## Refusal semantics
|
||||
|
||||
A name that resolves to neither your metadata nor a system scalar is a typed
|
||||
refusal, not a silent empty result and not a guess. Refusals name **both**
|
||||
candidates, so the fix is always in the error text:
|
||||
|
||||
```typescript
|
||||
await brain.find({ orderBy: 'createdAt' })
|
||||
// UnresolvableFieldError: no metadata field 'createdAt' — did you mean
|
||||
// system.createdAt or metadata.createdAt?
|
||||
```
|
||||
|
||||
`UnresolvableFieldError` is exported from the package root:
|
||||
|
||||
```typescript
|
||||
import { UnresolvableFieldError } from '@soulcraftlabs/brainy'
|
||||
|
||||
try {
|
||||
await brain.find({ orderBy: 'createdAt' })
|
||||
} catch (err) {
|
||||
if (err instanceof UnresolvableFieldError) {
|
||||
// err.message names both candidates — usually enough to fix the call site.
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
A handful of `find()` options are not implemented yet: `cursor`,
|
||||
`includeRelations`, and `writeOnly`. Rather than accepting them and quietly
|
||||
ignoring the option, `find()` refuses with `UnsupportedFindOptionError` —
|
||||
also exported from the package root — so a call site can never believe an
|
||||
unimplemented option took effect when it didn't.
|
||||
|
||||
## The ordering contract
|
||||
|
||||
`orderBy` behaves identically regardless of which engine (the pure-TypeScript
|
||||
path or a native accelerator) is serving the query:
|
||||
|
||||
- An entity missing the `orderBy` field, or holding `null` on it, sorts
|
||||
**LAST — in both `asc` and `desc`**. It is never treated as "smaller than
|
||||
everything" in one direction and "larger than everything" in the other; it
|
||||
is simply last, either way.
|
||||
- Rows are **never dropped** from an ordered read because they lack the
|
||||
field — a missing value changes position, never presence.
|
||||
- Ties on the `orderBy` field break by **id ascending**, regardless of the
|
||||
primary sort direction.
|
||||
|
||||
```typescript
|
||||
// employees: [{ score: 9 }, { score: 5 }, { /* no score field */ }]
|
||||
await brain.find({ orderBy: 'score', order: 'desc' }) // [9, 5, missing] — missing is last
|
||||
await brain.find({ orderBy: 'score', order: 'asc' }) // [5, 9, missing] — missing is STILL last
|
||||
```
|
||||
|
||||
## Migrating existing call sites
|
||||
|
||||
If you have call sites written before this rule shipped that rely on a bare
|
||||
system name — `orderBy: 'createdAt'`, `where: { confidence: { greaterThan:
|
||||
0.8 } }`, and similar — they now refuse instead of silently resolving to the
|
||||
engine field. The fix is always in the error: swap the bare name for
|
||||
`system.<field>` (or `metadata.<field>` if you actually meant your own field
|
||||
of that name, and it happens to share a name with a system scalar):
|
||||
|
||||
```typescript
|
||||
// Before: bare 'createdAt' silently meant the engine's timestamp.
|
||||
await brain.find({ orderBy: 'createdAt' })
|
||||
|
||||
// After: say which one you meant.
|
||||
await brain.find({ orderBy: 'system.createdAt' }) // the engine timestamp
|
||||
await brain.find({ orderBy: 'metadata.createdAt' }) // your own field named createdAt, if you have one
|
||||
```
|
||||
|
||||
There is no silent migration path by design — every ambiguous call site
|
||||
surfaces as a refusal naming its own fix, once, the first time it runs
|
||||
against the new rule.
|
||||
|
||||
## Where to go next
|
||||
|
||||
- [Consistency Model](./consistency-model.md) — visibility tiers, revision
|
||||
counters, and the rest of the read/write contract this page's
|
||||
read-time addressing rule.
|
||||
|
|
@ -1,117 +0,0 @@
|
|||
---
|
||||
title: The Generation Fact Log
|
||||
slug: concepts/generation-fact-log
|
||||
public: true
|
||||
category: concepts
|
||||
template: concept
|
||||
order: 6
|
||||
description: Every committed write also appends a self-verifying "fact" — an after-image commit record — to an append-only log. What facts are, the crash-safety model, the scanFacts() streaming surface, family stamps, and how index providers consume the log for sequential heals.
|
||||
next:
|
||||
- concepts/consistency-model
|
||||
- guides/snapshots-and-time-travel
|
||||
---
|
||||
|
||||
# The Generation Fact Log
|
||||
|
||||
Since 8.4.0, every committed generation also appends a **fact** — a compact record of what each
|
||||
touched entity or relationship *became* — to an append-only, checksummed log under
|
||||
`_generations/facts/`. Where the generational history answers *"what did things look like
|
||||
before?"* (before-images, powering `asOf()` and rollback), the fact log answers *"what happened,
|
||||
in order?"* — one sequential, self-verifying stream of the store's present being written.
|
||||
|
||||
Nothing about querying changes. The fact log exists for three consumers:
|
||||
|
||||
1. **Index heals and rebuilds** — one sequential read in commit order replaces a per-entity
|
||||
directory walk over millions of files.
|
||||
2. **Incremental catch-up** — a derived index that knows which generation it reflects reads *just
|
||||
the gap*, instead of rebuilding from scratch.
|
||||
3. **Replay and audit tooling** — anything that wants the store's committed timeline as a stream.
|
||||
|
||||
## What a fact is
|
||||
|
||||
One fact per committed generation:
|
||||
|
||||
- **`generation`** and **`timestamp`** — which commit, when.
|
||||
- **`ops`** — every write in that commit: `{ kind: 'noun' | 'verb', id, record }` where `record`
|
||||
holds the entity's full after-image (both stored legs), or **`null` for a tombstone** — a
|
||||
removal carries no body, by design.
|
||||
- **`meta`** — the transaction metadata `transact()` was submitted with, when present.
|
||||
- **`blobHashes`** — content-blob references, for exact reclamation accounting.
|
||||
|
||||
Facts accumulate **from the first write after upgrading** — pre-existing history is not
|
||||
retroactively converted, and consumers fall back to the enumeration walk when no log exists.
|
||||
|
||||
## Crash safety, in one paragraph
|
||||
|
||||
Facts are appended and fsynced **inside the same durability window as the commit itself**, before
|
||||
the commit point — so after a crash, the log can only ever be *ahead* of committed truth, never
|
||||
behind it with a hole. On open, the store reconciles the log back to the committed watermark:
|
||||
torn tails are detected by per-record checksums and cut; whole records beyond the watermark are
|
||||
truncated. The invariant every reader can rely on: **an absent generation was never committed; a
|
||||
present fact was.** `transact()` facts are durable the moment `transact()` returns; single-op
|
||||
facts share the same group-commit flush as the rest of their generation, so a hard kill loses the
|
||||
fact and the generation *together* — never a torn state.
|
||||
|
||||
## Reading the log
|
||||
|
||||
```typescript
|
||||
const scan = brain.scanFacts({ fromGeneration: 1 })
|
||||
if (scan) {
|
||||
// Telemetry up front — progress bars get a denominator from second zero.
|
||||
console.log(scan.headGeneration, scan.segmentCount, scan.approxFactCount)
|
||||
|
||||
for await (const batch of scan.batches()) {
|
||||
// Each batch: { facts, firstGeneration, lastGeneration, factCount, byteSize, segmentId }
|
||||
for (const fact of batch.facts) {
|
||||
for (const op of fact.ops) {
|
||||
if (op.record === null) {
|
||||
// a tombstone: op.id was removed in this generation
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
console.log(scan.summary()) // { factsYielded, segmentsRead } — the cross-check
|
||||
}
|
||||
```
|
||||
|
||||
- `scanFacts()` returns `null` when the store hosts no fact log (older store, or a storage adapter
|
||||
without binary append support) — fall back to enumerating entities.
|
||||
- Scans run against a **snapshot**: facts appended after the scan opens never bleed in, each fact
|
||||
is yielded exactly once, and a detected gap aborts loudly — never a silent skip.
|
||||
- `brain.factSegmentPaths()` returns the immutable, *sealed* segment files for zero-copy consumers
|
||||
(the append-mutable tail is excluded — read it through `scanFacts()`).
|
||||
|
||||
## Family stamps: how a projection proves it's current
|
||||
|
||||
Anything derived from the store — an index, the entity file tree itself — carries a **family
|
||||
stamp**: a small JSON record of *which committed generation the projection reflects*
|
||||
(`sourceGeneration`) plus the invariants that verify it whole (exact per-file byte sizes for
|
||||
bounded families; rollup invariants like entity counts for unbounded ones). At open, coherence is
|
||||
a **comparison**, not a walk:
|
||||
|
||||
- stamp equals the committed watermark and invariants hold → serve;
|
||||
- stamp is behind → the projection reads just the gap from the fact log;
|
||||
- invariants fail → loud, named divergence — `brain.repairIndex()` rebuilds from canonical and
|
||||
re-stamps.
|
||||
|
||||
The verifier is exported (`verifyFamilyStamp`) so every projection — TypeScript or native — runs
|
||||
literally the same check.
|
||||
|
||||
## For plugin authors: the storage capability
|
||||
|
||||
Index providers receive the storage adapter, not the brain — so the host wires the log onto it.
|
||||
Feature-detect and prefer the stream; fall back to enumeration:
|
||||
|
||||
```typescript
|
||||
const scan = storage.scanFacts?.({ fromGeneration: stamp.sourceGeneration + 1 })
|
||||
if (scan) {
|
||||
// sequential catch-up from the log
|
||||
} else {
|
||||
// enumeration walk (older store or adapter)
|
||||
}
|
||||
const committed = storage.committedGeneration?.() // the watermark stamps compare against
|
||||
```
|
||||
|
||||
Providers must never construct their own reader over the log's files — the open path belongs to
|
||||
the single writer (it reconciles the log at open); the capability is the sanctioned seam.
|
||||
|
|
@ -1,217 +0,0 @@
|
|||
---
|
||||
title: Index Health
|
||||
slug: concepts/index-health
|
||||
public: true
|
||||
category: concepts
|
||||
template: concept
|
||||
order: 8
|
||||
description: How Brainy knows whether a derived index can be trusted — exact accounting instead of sampling, the named health report, degraded-but-serving vs. not-ready, and what repairIndex() checks, heals, and rebuilds.
|
||||
next:
|
||||
- concepts/generation-fact-log
|
||||
- guides/inspection
|
||||
---
|
||||
|
||||
# Index Health
|
||||
|
||||
Brainy keeps one **canonical** copy of every entity and relationship, and three
|
||||
**derived** indexes built from it — vector, metadata, and graph — so `find()` can
|
||||
answer semantically, by filter, and by traversal without re-deriving the answer from
|
||||
scratch on every query. A derived index is a cache with a serving structure: it can
|
||||
be present but stale, present but only partially loaded, or fully out of sync with
|
||||
canonical after a crash. This page is about how Brainy decides whether to trust one,
|
||||
what it does when it can't, and how you reconcile the two.
|
||||
|
||||
## Exact accounting instead of sampling
|
||||
|
||||
Older health checks worked by inference: does `size()` return something greater
|
||||
than zero, does a spot-check on one known item come back correct. Both are proxies.
|
||||
A cold index can report a nonzero count while its actual serving structure never
|
||||
loaded, and a spot-check only proves the one item it happened to ask about.
|
||||
|
||||
Every derived-index provider may now expose a named, synchronous, O(1)
|
||||
`healthReport()` — composed from the provider's own **exact ledgers** (real counters
|
||||
it already maintains on the write path), never a sample or a walk. This is the one
|
||||
signal Brainy's read gate consults. A provider that doesn't yet expose one falls
|
||||
back to an honest `isReady()` boolean, and finally to a size heuristic for engines
|
||||
with neither — but wherever a `healthReport()` exists, it wins.
|
||||
|
||||
Underneath, storage itself keeps an analogous **canonical count ledger**: a
|
||||
`counted` scalar (the user-facing total — what `getNounCount()` / `getVerbCount()`
|
||||
return) and an `all` scalar (every tier, including internal records a derived
|
||||
index's own coverage math needs to compare against). This is the real denominator
|
||||
a provider's `healthReport()` measures itself by, rather than a total that can only
|
||||
ever ratchet upward. See [What `suspect` counts mean](#what-suspect-counts-mean)
|
||||
below for the one case that ledger can't stay exact through on its own.
|
||||
|
||||
## The named report
|
||||
|
||||
A `HealthReport` carries, per provider (`'vector'` / `'graph'` / `'metadata'`):
|
||||
|
||||
- **`healthy`** — `true` iff every *verified* invariant holds. An invariant whose
|
||||
family has no ledger yet is `unledgered`, never counted either way — unknown,
|
||||
not passing.
|
||||
- **`serving`** — can this provider answer a query right now. A failing invariant
|
||||
graded `heal: 'repair'` or `heal: 'none'` still leaves `serving: true` — this is
|
||||
**degraded-but-serving**: something is off (say, a stale rollup on an
|
||||
`employee` record's relationship count) but reads keep working. Only a failure
|
||||
graded `heal: 'rebuild'` flips `serving` to `false` — **not-ready** — because the
|
||||
provider itself is telling you its serving structure cannot answer correctly.
|
||||
- **`invariants`** — each checked condition, with its provenance
|
||||
(`source: 'ledger'` — an exact count; `'deep'` — a full scan, diagnostic-only;
|
||||
`'unledgered'` — not yet tracked) and, for a failing one, an exact `missing`
|
||||
count plus a capped sample of the affected ids — a verdict, never a dump.
|
||||
- **`generation`** — bumps on every ledger mutation and rebuild, so a caller can
|
||||
cache a verdict per generation instead of re-deriving it.
|
||||
|
||||
The distinction that matters day to day: `healthy: false` can be entirely benign —
|
||||
a maintenance window, a divergence `repairIndex()` will clean up on its own
|
||||
schedule. `serving: false` is not benign. It means this provider is refusing to
|
||||
answer, on its own word, right now.
|
||||
|
||||
**How a failure gets its grade — the serving law.** A provider grades `heal` by
|
||||
one question only: *could an answer be wrong?* — never *how expensive is the
|
||||
fix?* A missing-postings shortfall, however large, is `heal: 'repair'` (re-post
|
||||
exactly what the ledger names, reads serving throughout); it can never withhold
|
||||
serving just because healing it takes work. `serving` is withheld only by a
|
||||
small, named set of rebuild-graded conditions — the index not initialized, its
|
||||
durable state absent, a manifest naming files that are not resident, a replay
|
||||
that did not complete cleanly — the states in which an answer could genuinely be
|
||||
wrong. And a read is only ever refused by the family it actually consults: a
|
||||
metadata filter is answered by the metadata index alone, vector search by the
|
||||
vector index, traversal by the graph index — one family's refusal never blocks
|
||||
another family's reads.
|
||||
|
||||
## Reads refuse — they never rebuild
|
||||
|
||||
A query that reaches a not-serving provider does not trigger a rebuild from inside
|
||||
the read. Brainy retired that path deliberately: a rebuild kicked off by an ordinary
|
||||
`find({ where: { status: 'active' } })` call is a dark, unpredictable cost hiding
|
||||
behind a request that looks like a cheap read. Instead, the read throws a typed,
|
||||
catchable error naming the reason:
|
||||
|
||||
| Error | Thrown when | Meaning |
|
||||
|---|---|---|
|
||||
| `GraphIndexNotReadyError` | `find({ connected })`, `neighbors()`, `related()` | The graph adjacency index isn't serving — traversal would otherwise return `[]` indistinguishable from "no relationships" |
|
||||
| `MetadataIndexNotReadyError` | `find({ where })` | The metadata/field index isn't serving — a filtered read would otherwise return `[]` indistinguishable from "no matches" |
|
||||
| `VectorIndexNotReadyError` | `find({ query })`, `similar()` | The vector index isn't serving — a semantic search would otherwise return `[]` indistinguishable from "nothing similar" |
|
||||
|
||||
All three are exported from `@soulcraftlabs/brainy`. Catch them where your application
|
||||
needs to distinguish "this index isn't ready yet" from "there's genuinely nothing
|
||||
here" — a health dashboard, a retry policy, an operator alert. The fix is always
|
||||
the same: reconcile the index, either by reopening the brain (which brings every
|
||||
provider to serving before `init()` returns — see the next section) or by calling
|
||||
`repairIndex()` explicitly.
|
||||
|
||||
```typescript
|
||||
try {
|
||||
const active = await brain.find({ where: { status: 'active' } })
|
||||
} catch (err) {
|
||||
if (err instanceof MetadataIndexNotReadyError) {
|
||||
// not a "no results" — the index itself refused; alert or retry after repair
|
||||
} else {
|
||||
throw err
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
### Rebuilds happen at open, not on first query
|
||||
|
||||
`brain.init()` runs every needed rebuild to completion **before it returns**,
|
||||
unconditionally, regardless of dataset size. There is no lazy, first-query
|
||||
rebuild path anymore — a brain either finishes opening healthy, or it fails
|
||||
open loudly. `disableAutoRebuild: true` no longer defers index construction to
|
||||
the first query: it has no effect on *when* a needed rebuild runs. Full manual
|
||||
control over rebuilds is `repairIndex({ rebuild: [...] })` (below), not this flag.
|
||||
|
||||
## `repairIndex()` — checking and healing
|
||||
|
||||
Bare `repairIndex()` is **report-driven**: it only heals what its own checks say
|
||||
actually needs it, and it always returns a full per-family receipt.
|
||||
|
||||
```typescript
|
||||
const report = await brain.repairIndex()
|
||||
report.healedTotal // total items healed across every family
|
||||
report.durationMs
|
||||
report.families // one row per family checked
|
||||
```
|
||||
|
||||
Each `RepairFamilyReport` row names what happened:
|
||||
|
||||
- **`checked`** — was this family actually examined (`false` means skipped —
|
||||
see `skipped` for why).
|
||||
- **`healed`** — items re-posted or corrected in place.
|
||||
- **`missing`** — when the check can name what diverged: an exact `count` plus a
|
||||
capped `sample` of ids.
|
||||
- **`rebuilt`** — a full generational rebuild ran (as opposed to an incremental
|
||||
heal).
|
||||
- **`detail`** / **`reason`** / **`skipped`** — the receipt's narration; a row is
|
||||
always either checked or explains why it wasn't. Nothing is silent.
|
||||
|
||||
On every call, bare `repairIndex()`:
|
||||
|
||||
1. Prunes orphaned canonical containers left by a partial delete.
|
||||
2. Recomputes the count rollups from one canonical walk (unconditional — this is
|
||||
also what clears a `suspect` ledger; see below).
|
||||
3. Reconciles VFS containment edges, if the VFS is initialized.
|
||||
4. Runs the metadata index's own corruption detection pass.
|
||||
5. Consults each of the three derived-index providers' own health check and
|
||||
rebuilds only a family whose failing invariant actually asks for it
|
||||
(`heal: 'rebuild'`) — never a provider that reports `healthy` or a lesser
|
||||
grade.
|
||||
|
||||
### The explicit rebuild door
|
||||
|
||||
`options.rebuild` skips the health check and rebuilds one or more families
|
||||
**unconditionally** — the operator override for when you have independent reason
|
||||
to distrust a family regardless of what it self-reports (a suspicious deploy, a
|
||||
storage-layer incident, a support ticket that doesn't match what the health report
|
||||
says):
|
||||
|
||||
```typescript
|
||||
// Force the graph adjacency to rebuild from canonical, no invariant consulted
|
||||
await brain.repairIndex({ rebuild: ['graph'] })
|
||||
|
||||
// Force all three derived indexes
|
||||
await brain.repairIndex({ rebuild: 'all' })
|
||||
```
|
||||
|
||||
A family named this way is recorded with `rebuilt: true` and
|
||||
`reason: 'explicit rebuild requested'`, and is skipped by the normal
|
||||
health-driven pass in the same call — it was already rebuilt unconditionally.
|
||||
|
||||
Reach for the explicit door when you need certainty regardless of self-report;
|
||||
reach for bare `repairIndex()` for routine maintenance and after any incident
|
||||
where you're not sure which family (if any) needs it.
|
||||
|
||||
## What `suspect` counts mean
|
||||
|
||||
Storage's canonical count ledger increments the ALL-visibility total on every new
|
||||
record and decrements it on every *proven* delete — one where the record was read,
|
||||
or the caller supplied its prior image. A delete that cannot prove what it removed
|
||||
existed doesn't guess: it flags the ledger `suspect` (an operator-visible
|
||||
`console.warn`, narrated once per session, not once per delete) rather than risk
|
||||
decrementing a total that was never incremented for that record in the first
|
||||
place. This is intentionally rare — it's a defensive fallback for callers on an
|
||||
unusual removal path, not a per-delete cost.
|
||||
|
||||
`suspect` is not directly exposed on any `Brainy` method today — it lives on the
|
||||
`StorageAdapter`'s optional `getCanonicalCounts()`, primarily consulted by
|
||||
`repairIndex()`'s recount step and by custom storage adapters composing their own
|
||||
`healthReport()`. What matters for an application: a `suspect` ledger is not
|
||||
incorrect, just *unverified since the last recount* — and `repairIndex()`'s
|
||||
unconditional count-rollup step (step 2, above) recomputes the ALL scalars from a
|
||||
real canonical walk on every call, clearing the flag with proof either way.
|
||||
|
||||
## Practical guidance
|
||||
|
||||
- **On a normal restart**, do nothing — `init()` brings every provider to
|
||||
serving before it returns, or fails loudly.
|
||||
- **On a `*NotReadyError`** from a live read, reconcile with `repairIndex()`
|
||||
(report-driven is almost always sufficient) and retry.
|
||||
- **After an incident** where you distrust a specific family regardless of what
|
||||
it reports healthy — a storage-layer fault, a suspicious restore — use the
|
||||
explicit door: `repairIndex({ rebuild: ['metadata' | 'graph' | 'vector'] })`.
|
||||
- **To audit before trusting a report**, `brain.auditGraph()` walks every stored
|
||||
relationship and proves (or disproves) that reads return canonical truth,
|
||||
independent of what any provider self-reports — see
|
||||
[Inspecting a Live Brainy](../guides/inspection.md).
|
||||
|
|
@ -61,7 +61,7 @@ The only required override is the capability flag. Returning `true` from
|
|||
to call `acquireWriterLock()` at init.
|
||||
|
||||
```typescript
|
||||
import { FileSystemStorage } from '@soulcraftlabs/brainy'
|
||||
import { FileSystemStorage } from '@soulcraft/brainy'
|
||||
|
||||
export class MmapFileSystemStorage extends FileSystemStorage {
|
||||
public supportsMultiProcessLocking(): boolean {
|
||||
|
|
@ -79,7 +79,7 @@ If your storage is **not filesystem-backed** (a custom
|
|||
network backend), extend `BaseStorage` directly:
|
||||
|
||||
```typescript
|
||||
import { BaseStorage } from '@soulcraftlabs/brainy'
|
||||
import { BaseStorage } from '@soulcraft/brainy'
|
||||
|
||||
export class MyCloudStorage extends BaseStorage {
|
||||
// BaseStorage's default no-op implementations of the multi-process
|
||||
|
|
@ -101,7 +101,7 @@ The defensive check at every new-storage-method call site (`brainy.ts`,
|
|||
`hasStorageMethod(name)`) does **not** exist to handle "plugin bundles a
|
||||
stale BaseStorage." Plugins ship a dist that preserves the dynamic ESM
|
||||
import (verify in your plugin's `dist/`: `import { FileSystemStorage } from
|
||||
'@soulcraftlabs/brainy'` is not rewritten to a vendored copy). The prototype
|
||||
'@soulcraft/brainy'` is not rewritten to a vendored copy). The prototype
|
||||
chain at runtime resolves to whatever Brainy version your consumer has
|
||||
installed.
|
||||
|
||||
|
|
@ -109,8 +109,8 @@ installed.
|
|||
the prototype chain at the consumer-app level:
|
||||
|
||||
- **Stale `node_modules`** — a lingering install from before the consumer
|
||||
upgraded Brainy. The package.json says `@soulcraftlabs/brainy@7.22.0` but
|
||||
`node_modules/@soulcraftlabs/brainy` is still 7.20.x.
|
||||
upgraded Brainy. The package.json says `@soulcraft/brainy@7.22.0` but
|
||||
`node_modules/@soulcraft/brainy` is still 7.20.x.
|
||||
- **Lockfile drift** — `bun.lockb` / `package-lock.json` pins a brainy
|
||||
version older than the package.json range, and `bun install` honors the
|
||||
lockfile.
|
||||
|
|
@ -131,7 +131,7 @@ and the warning names the adapter class plus a remediation hint:
|
|||
methods on its prototype chain. Writer locking and the flush-request RPC are
|
||||
disabled for this directory. Likely fix: clean install (`rm -rf node_modules
|
||||
bun.lockb && bun install`) or rebuild your container image to refresh
|
||||
`@soulcraftlabs/brainy` to ≥7.21. See docs/concepts/storage-adapters.md.
|
||||
`@soulcraft/brainy` to ≥7.21. See docs/concepts/storage-adapters.md.
|
||||
```
|
||||
|
||||
## Authoring a new storage adapter — minimum checklist
|
||||
|
|
@ -168,7 +168,7 @@ bun.lockb && bun install`) or rebuild your container image to refresh
|
|||
install time — fix install, not your plugin.
|
||||
|
||||
6. **Pin your peer dep generously.** `"peerDependencies": {
|
||||
"@soulcraftlabs/brainy": "^7.21.0" }` accepts any compatible 7.x. Don't pin
|
||||
"@soulcraft/brainy": "^7.21.0" }` accepts any compatible 7.x. Don't pin
|
||||
to an exact patch unless you're tracking a known regression.
|
||||
|
||||
## Future direction
|
||||
|
|
@ -185,5 +185,5 @@ follow-up; consumers don't need to anticipate the change.
|
|||
heartbeat semantics, what the lock protects.
|
||||
- [`guides/inspection`](../guides/inspection.md) — `brainy inspect` and the
|
||||
read-only mode.
|
||||
- `node_modules/@soulcraftlabs/brainy/dist/storage/baseStorage.d.ts` — the
|
||||
- `node_modules/@soulcraft/brainy/dist/storage/baseStorage.d.ts` — the
|
||||
authoritative type signatures for every method this page references.
|
||||
|
|
|
|||
|
|
@ -11,18 +11,12 @@ No batch jobs. No scheduled recalculations. Aggregates stay current with every w
|
|||
**Defining over existing data:** if you define an aggregate on a store that already holds
|
||||
matching entities, Brainy backfills it from those entities on the first query (a one-time scan,
|
||||
then purely incremental). So `defineAggregate()` behaves the same whether you define it before
|
||||
or after the data exists.
|
||||
|
||||
**Reopening a persisted brain:** aggregate state persists across restarts. Re-defining the
|
||||
same aggregate at boot (the normal declarative pattern) adopts the persisted state directly —
|
||||
no rescan. A backfill scan runs only when the definition actually changed, when no persisted
|
||||
state exists, or when the state failed to load; and however many aggregates need backfilling,
|
||||
they share a single scan.
|
||||
or after the data exists — including when a persisted brain reopens already populated.
|
||||
|
||||
## Quick Start
|
||||
|
||||
```typescript
|
||||
import { Brainy, NounType } from '@soulcraftlabs/brainy'
|
||||
import { Brainy, NounType } from '@soulcraft/brainy'
|
||||
|
||||
const brain = new Brainy()
|
||||
await brain.init()
|
||||
|
|
|
|||
|
|
@ -1,99 +0,0 @@
|
|||
---
|
||||
title: External Backups & Sparse Storage
|
||||
slug: guides/external-backups
|
||||
public: true
|
||||
category: guides
|
||||
template: guide
|
||||
order: 10
|
||||
description: How to back up a brain directory with external tools (tar, rsync, cp) without exploding sparse files — why a store can show 100+ GB "apparent" size on a small disk, which files are sparse, and how persist()/restore() handle it for you.
|
||||
next:
|
||||
- guides/snapshots-and-time-travel
|
||||
- concepts/storage-adapters
|
||||
---
|
||||
|
||||
# External Backups & Sparse Storage
|
||||
|
||||
The built-in snapshot path — [`db.persist()` and `brain.restore()`](/docs/guides/snapshots-and-time-travel) —
|
||||
already handles everything on this page for you. Read this when you back up a brain directory with
|
||||
**external tools**: `tar`, `rsync`, `cp`, `scp`, or a filesystem-level backup agent.
|
||||
|
||||
## The one-sentence rule
|
||||
|
||||
> **Always use the sparse-aware flag**: `tar czSf` (capital `S`), `rsync --sparse`,
|
||||
> `cp --sparse=always`. A naive copy can turn a 2 GB store into a 100+ GB one — or fail
|
||||
> the disk entirely.
|
||||
|
||||
## Why: some files are sparse
|
||||
|
||||
When a native accelerator plugin is active, parts of the index live in **memory-mapped files**
|
||||
created at a large fixed virtual size — the file's *apparent* size — while the filesystem only
|
||||
allocates blocks that were actually written. A brand-new id-mapper file can report tens of
|
||||
gigabytes in `ls -l` while occupying a few megabytes on disk.
|
||||
|
||||
Check the difference yourself:
|
||||
|
||||
```bash
|
||||
ls -lh brain-data/_id_mapper/ # APPARENT size (can be huge)
|
||||
du -sh brain-data/ # ALLOCATED size (the real footprint)
|
||||
```
|
||||
|
||||
The sparse candidates in a brain directory:
|
||||
|
||||
| Path | What it is |
|
||||
|---|---|
|
||||
| `_id_mapper/` | The native id-mapper's mmap files (large fixed virtual size) |
|
||||
| `_blobs/` | Native index files (vector base, segments) — may be mmap-backed |
|
||||
|
||||
Everything else (entities, `_system`, `_generations`, `_cas` content blobs) is ordinary dense data.
|
||||
|
||||
## Doing it right
|
||||
|
||||
**tar** — the `S` flag detects holes and stores only real data:
|
||||
|
||||
```bash
|
||||
tar czSf brain-backup.tgz /data/brain
|
||||
# restore preserves the holes:
|
||||
tar xzSf brain-backup.tgz -C /data/
|
||||
```
|
||||
|
||||
**rsync**:
|
||||
|
||||
```bash
|
||||
rsync -a --sparse /data/brain/ backup-host:/backups/brain/
|
||||
```
|
||||
|
||||
**cp**:
|
||||
|
||||
```bash
|
||||
cp -a --sparse=always /data/brain /backups/brain
|
||||
```
|
||||
|
||||
**What goes wrong without the flag:** the copy *materializes* every hole as real zero bytes.
|
||||
A store whose apparent size exceeds the target disk fails with `ENOSPC` partway through — and a
|
||||
copy that *does* fit silently costs the full apparent size in storage and transfer time.
|
||||
|
||||
## What the built-in paths do (so you don't have to)
|
||||
|
||||
- **`db.persist(path)`** snapshots via **hard links** — instant and space-shared, since every data
|
||||
file is immutable-by-rename. The handful of append-in-place files (the transaction log, the
|
||||
commit fact log's tail segment) and mmap-mutated directories (`_id_mapper/`) are **byte-copied**
|
||||
instead, so a post-snapshot write can never reach through a shared inode into your backup.
|
||||
- **`brain.restore(path, { confirm: true })`** is **non-destructive and sparse-aware**: the snapshot
|
||||
is copied into a staging area *before* any live data is touched (all-zero blocks stay holes), and
|
||||
only after the copy fully succeeds does an atomic swap move it into place. A failed copy —
|
||||
including `ENOSPC` — leaves the live store exactly as it was. A crash mid-swap completes forward
|
||||
on the next open.
|
||||
|
||||
## Live-store caveats for external tools
|
||||
|
||||
1. **Prefer snapshotting a `persist()` output, not the live directory.** `persist()` produces a
|
||||
crash-consistent, immutable snapshot; running `tar` against a live, actively-written directory
|
||||
can capture a torn mid-write state. If you must archive live, stop writes first (or accept that
|
||||
the archive is only as consistent as the moment's flush state).
|
||||
2. **Never prune or "clean up" files inside a brain directory.** Index files that look stale or
|
||||
redundant are load-bearing; the store protects its declared index families from in-process
|
||||
deletion, but an external `rm` bypasses that fence. If space is the concern, `du -sh` first —
|
||||
the allocated size is usually far smaller than it looks.
|
||||
3. **Verify restores by opening them.** `Brainy.load(path)` opens any snapshot or restored
|
||||
directory read-only — the store verifies its own coherence at open and reports loudly if
|
||||
anything is missing or torn.
|
||||
|
|
@ -40,10 +40,6 @@ Brainy picks `maxLimit` from the first of these that's available:
|
|||
|
||||
Worked example: a 4 GB Cloud Run container picks priority 3 → `floor(4 GB × 0.25 / 25 KB) = floor(40 960) = 40 000` results. A 900 MB free-memory box on priority 4 gets `floor(900 MB / 25 KB) = ~36 000`.
|
||||
|
||||
The cap is fixed at construction and never changes at runtime. Query timing is recorded
|
||||
for diagnostics only — a burst of slow queries cannot silently shrink the cap, and the
|
||||
auto-detected tiers (3 and 4) never go below a floor of 10 000.
|
||||
|
||||
> **Calibration note.** Pre-7.30.2 used 100 KB per result instead of 25 KB, which produced caps that were 4× too tight for typical workloads (an 8 KB / result reality). 7.30.2 recalibrated to match observed entity sizes; existing `limit: 10_000` safety patterns now pass silently on any reasonably-sized box.
|
||||
|
||||
## What happens when you exceed the cap
|
||||
|
|
|
|||
|
|
@ -8,7 +8,7 @@ Brainy is **framework-friendly** - designed to drop into the server side of any
|
|||
|
||||
Brainy embeds an HNSW vector index, a graph engine, and a filesystem-backed persistence layer. These belong on the server:
|
||||
|
||||
- **Zero configuration**: Just `import { Brainy } from '@soulcraftlabs/brainy'`
|
||||
- **Zero configuration**: Just `import { Brainy } from '@soulcraft/brainy'`
|
||||
- **Auto storage detection**: `new Brainy()` auto-selects filesystem persistence on Node
|
||||
- **Cleaner code**: No browser polyfills, no conditional client/server imports
|
||||
- **Better DX**: One instance shared across your server routes
|
||||
|
|
@ -18,13 +18,13 @@ Brainy embeds an HNSW vector index, a graph engine, and a filesystem-backed pers
|
|||
### Install Brainy
|
||||
|
||||
```bash
|
||||
npm install @soulcraftlabs/brainy
|
||||
npm install @soulcraft/brainy
|
||||
```
|
||||
|
||||
### Basic Integration
|
||||
|
||||
```javascript
|
||||
import { Brainy } from '@soulcraftlabs/brainy'
|
||||
import { Brainy } from '@soulcraft/brainy'
|
||||
|
||||
// Run on the server (API route, server component, backend service)
|
||||
// new Brainy() auto-detects filesystem persistence on Node
|
||||
|
|
@ -105,7 +105,7 @@ On the server, create one Brainy instance and reuse it across requests. This mod
|
|||
|
||||
```javascript
|
||||
// lib/brain.server.js
|
||||
import { Brainy } from '@soulcraftlabs/brainy'
|
||||
import { Brainy } from '@soulcraft/brainy'
|
||||
|
||||
let brainPromise
|
||||
|
||||
|
|
@ -163,7 +163,7 @@ On the server, create one Brainy instance and reuse it across requests:
|
|||
|
||||
```javascript
|
||||
// server/brain.js (server-only module)
|
||||
import { Brainy } from '@soulcraftlabs/brainy'
|
||||
import { Brainy } from '@soulcraft/brainy'
|
||||
|
||||
let brainPromise
|
||||
|
||||
|
|
@ -248,7 +248,7 @@ The matching backend endpoint uses Brainy directly (Node/Bun):
|
|||
|
||||
```typescript
|
||||
// server: api/search
|
||||
import { Brainy } from '@soulcraftlabs/brainy'
|
||||
import { Brainy } from '@soulcraft/brainy'
|
||||
|
||||
const brain = new Brainy() // auto-detects filesystem persistence on Node
|
||||
await brain.init()
|
||||
|
|
@ -266,7 +266,7 @@ In Next.js, Brainy lives in server code only: API routes, server components, or
|
|||
|
||||
```javascript
|
||||
// lib/brain.server.js (imported only by server code)
|
||||
import { Brainy } from '@soulcraftlabs/brainy'
|
||||
import { Brainy } from '@soulcraft/brainy'
|
||||
|
||||
let brainPromise
|
||||
|
||||
|
|
@ -318,7 +318,7 @@ Brainy runs in a server-only module (`*.server.js`); the component fetches resul
|
|||
|
||||
```javascript
|
||||
// src/lib/server/brain.js (server-only — note the .server suffix)
|
||||
import { Brainy } from '@soulcraftlabs/brainy'
|
||||
import { Brainy } from '@soulcraft/brainy'
|
||||
|
||||
let brainPromise
|
||||
|
||||
|
|
@ -432,7 +432,7 @@ import { defineConfig } from 'vite'
|
|||
|
||||
export default defineConfig({
|
||||
ssr: {
|
||||
external: ['@soulcraftlabs/brainy']
|
||||
external: ['@soulcraft/brainy']
|
||||
}
|
||||
})
|
||||
```
|
||||
|
|
@ -440,7 +440,7 @@ export default defineConfig({
|
|||
```javascript
|
||||
// rollup.config.js (server bundle)
|
||||
export default {
|
||||
external: ['@soulcraftlabs/brainy', 'node:fs', 'node:path', 'node:crypto']
|
||||
external: ['@soulcraft/brainy', 'node:fs', 'node:path', 'node:crypto']
|
||||
}
|
||||
```
|
||||
|
||||
|
|
@ -466,7 +466,7 @@ export async function load({ url }) {
|
|||
|
||||
```javascript
|
||||
// For build-time usage (runs in Node during the build)
|
||||
import { Brainy } from '@soulcraftlabs/brainy'
|
||||
import { Brainy } from '@soulcraft/brainy'
|
||||
|
||||
export async function generateStaticProps() {
|
||||
const brain = new Brainy({
|
||||
|
|
@ -513,7 +513,7 @@ export async function generateStaticProps() {
|
|||
|
||||
### Issue: Large client bundle size
|
||||
**Cause**: A client module is pulling in Brainy.
|
||||
**Solution**: Move the `import { Brainy } from '@soulcraftlabs/brainy'` into a server-only module so it never reaches the browser bundle.
|
||||
**Solution**: Move the `import { Brainy } from '@soulcraft/brainy'` into a server-only module so it never reaches the browser bundle.
|
||||
|
||||
### Issue: SSR hydration mismatch
|
||||
**Solution**: Run the search on the server (loader / server action / API route) and pass the results down as props, so server and client render the same markup.
|
||||
|
|
|
|||
|
|
@ -9,7 +9,7 @@ Brainy's import is **ONE magical method** that understands EVERYTHING:
|
|||
## The Ultimate Simplicity
|
||||
|
||||
```javascript
|
||||
import { Brainy } from '@soulcraftlabs/brainy'
|
||||
import { Brainy } from '@soulcraft/brainy'
|
||||
|
||||
const brain = new Brainy()
|
||||
await brain.init()
|
||||
|
|
@ -300,10 +300,7 @@ await brain.import(data, {
|
|||
// Deduplication
|
||||
enableDeduplication: true, // Check for duplicate entities (default: true)
|
||||
deduplicationThreshold: 0.85, // Similarity threshold for duplicates (0-1, default: 0.85)
|
||||
// Notes: false disables BOTH the inline merge and the background pass that
|
||||
// runs ~5 min after the last import (merged duplicates are deleted).
|
||||
// The inline pass auto-disables for imports >100 entities (O(n²) cost);
|
||||
// the background pass still covers those unless the flag is false.
|
||||
// Note: Auto-disabled for imports >100 entities
|
||||
|
||||
// Performance
|
||||
chunkSize: 100, // Batch size for processing (default: varies by operation)
|
||||
|
|
|
|||
|
|
@ -13,7 +13,7 @@ Brainy provides real-time progress tracking for **all 7 supported file formats**
|
|||
### Basic Progress Tracking
|
||||
|
||||
```typescript
|
||||
import { Brainy } from '@soulcraftlabs/brainy'
|
||||
import { Brainy } from '@soulcraft/brainy'
|
||||
import * as fs from 'fs'
|
||||
|
||||
const brain = await Brainy.create()
|
||||
|
|
|
|||
|
|
@ -7,7 +7,7 @@
|
|||
## Basic Import
|
||||
|
||||
```typescript
|
||||
import { Brainy } from '@soulcraftlabs/brainy'
|
||||
import { Brainy } from '@soulcraft/brainy'
|
||||
|
||||
const brain = new Brainy()
|
||||
await brain.init()
|
||||
|
|
@ -86,22 +86,11 @@ await brain.import(file, {
|
|||
|
||||
```typescript
|
||||
await brain.import(file, {
|
||||
enableDeduplication: true, // Check for duplicates (default: true)
|
||||
enableDeduplication: true, // Check for duplicates (default: false)
|
||||
deduplicationThreshold: 0.85 // Similarity threshold (default: 0.85)
|
||||
})
|
||||
```
|
||||
|
||||
Deduplication merges entities judged duplicates — the non-primary records are
|
||||
**deleted**. Set `enableDeduplication: false` to disable it entirely: the flag
|
||||
gates both the inline merge during import and the background pass that runs
|
||||
about 5 minutes after the last import.
|
||||
|
||||
```typescript
|
||||
await brain.import(file, {
|
||||
enableDeduplication: false // No merging, inline or background
|
||||
})
|
||||
```
|
||||
|
||||
### Import Tracking
|
||||
|
||||
Track and organize imports by project:
|
||||
|
|
@ -187,7 +176,7 @@ await brain.import(file, {
|
|||
## Complete Example
|
||||
|
||||
```typescript
|
||||
import { Brainy } from '@soulcraftlabs/brainy'
|
||||
import { Brainy } from '@soulcraft/brainy'
|
||||
import * as fs from 'fs'
|
||||
|
||||
async function importCatalog() {
|
||||
|
|
|
|||
|
|
@ -108,7 +108,7 @@ check fails — useful for piping into monitoring or CI.
|
|||
## Programmatic inspection
|
||||
|
||||
```typescript
|
||||
import { Brainy } from '@soulcraftlabs/brainy'
|
||||
import { Brainy } from '@soulcraft/brainy'
|
||||
|
||||
const reader = await Brainy.openReadOnly({
|
||||
storage: { type: 'filesystem', path: '/data/brain' }
|
||||
|
|
@ -166,34 +166,6 @@ brainy inspect diff /data/brain-prod /data/brain-staging
|
|||
Sample-based — for a full diff, dump both with `inspect dump` and compare
|
||||
the JSONL.
|
||||
|
||||
## Auditing graph-read truth
|
||||
|
||||
`brain.auditGraph()` (8.6.0+) proves — or disproves — that relationship reads
|
||||
return canonical truth on a given brain, without mutating anything. It walks
|
||||
every stored relationship record, asks the same read path your application
|
||||
uses (`related()`, VFS `readdir`) with every visibility tier included, and
|
||||
classifies every discrepancy:
|
||||
|
||||
```typescript
|
||||
const report = await brain.auditGraph()
|
||||
|
||||
report.coherent // true = related()/readdir can be trusted on this brain
|
||||
report.missingFromReadsCount // records the read path omits — stale index
|
||||
report.danglingEndpointsCount // relationships whose endpoint entity is gone
|
||||
report.readOnlyCount // read-path edges with NO stored record — ghosts
|
||||
report.visibilityHiddenCount // internal/system edges hidden by design (not a fault)
|
||||
```
|
||||
|
||||
Counts are always exact; the example lists (`missingFromReads`,
|
||||
`danglingEndpoints`, `readOnlyVerbIds`) are capped at `maxExamples`
|
||||
(default 100) and `truncatedExamples` says so when they are.
|
||||
|
||||
Run it after any engine upgrade, restore, or migration. If it reports
|
||||
discrepancies, run `brain.repairIndex()` and audit again — a `coherent`
|
||||
report after the repair is the verified statement that the heal worked.
|
||||
Cost: one relationship-record walk plus one indexed read per distinct
|
||||
source entity — safe on a live brain.
|
||||
|
||||
## Repairing a corrupted store
|
||||
|
||||
If invariants fail and you suspect index corruption, `inspect repair`
|
||||
|
|
|
|||
|
|
@ -21,21 +21,21 @@ next:
|
|||
## Install
|
||||
|
||||
```bash
|
||||
npm install @soulcraftlabs/brainy
|
||||
npm install @soulcraft/brainy
|
||||
```
|
||||
|
||||
Or with your preferred package manager:
|
||||
|
||||
```bash
|
||||
bun add @soulcraftlabs/brainy
|
||||
yarn add @soulcraftlabs/brainy
|
||||
pnpm add @soulcraftlabs/brainy
|
||||
bun add @soulcraft/brainy
|
||||
yarn add @soulcraft/brainy
|
||||
pnpm add @soulcraft/brainy
|
||||
```
|
||||
|
||||
## Verify
|
||||
|
||||
```typescript
|
||||
import { Brainy } from '@soulcraftlabs/brainy'
|
||||
import { Brainy } from '@soulcraft/brainy'
|
||||
|
||||
const brain = new Brainy()
|
||||
await brain.init()
|
||||
|
|
@ -52,7 +52,7 @@ npm install @soulcraft/cor
|
|||
```
|
||||
|
||||
```typescript
|
||||
import { Brainy } from '@soulcraftlabs/brainy'
|
||||
import { Brainy } from '@soulcraft/brainy'
|
||||
|
||||
const brain = new Brainy({ plugins: ['@soulcraft/cor'] })
|
||||
await brain.init() // native providers registered during init
|
||||
|
|
@ -71,7 +71,7 @@ remains available on npm if you need it.
|
|||
Brainy ships with full TypeScript types. No `@types/` package needed:
|
||||
|
||||
```typescript
|
||||
import { Brainy, NounType, VerbType } from '@soulcraftlabs/brainy'
|
||||
import { Brainy, NounType, VerbType } from '@soulcraft/brainy'
|
||||
|
||||
const brain = new Brainy()
|
||||
await brain.init()
|
||||
|
|
|
|||
|
|
@ -66,7 +66,7 @@ const results = await brain.search("query")
|
|||
**New diagnostics for capacity planning and performance tuning.**
|
||||
|
||||
```typescript
|
||||
import { Brainy } from '@soulcraftlabs/brainy'
|
||||
import { Brainy } from '@soulcraft/brainy'
|
||||
|
||||
const brain = new Brainy()
|
||||
await brain.init()
|
||||
|
|
@ -112,7 +112,7 @@ Recommendations: ${stats.recommendations.join(', ')}
|
|||
### Step 1: Update Package
|
||||
|
||||
```bash
|
||||
npm install @soulcraftlabs/brainy@latest
|
||||
npm install @soulcraft/brainy@latest
|
||||
```
|
||||
|
||||
### Step 2: Restart Your Application
|
||||
|
|
@ -134,7 +134,7 @@ npm run start
|
|||
### Check Adaptive Sizing is Working
|
||||
|
||||
```typescript
|
||||
import { Brainy } from '@soulcraftlabs/brainy'
|
||||
import { Brainy } from '@soulcraft/brainy'
|
||||
|
||||
const brain = new Brainy()
|
||||
await brain.init()
|
||||
|
|
@ -218,7 +218,7 @@ For debugging or compatibility testing:
|
|||
If you need to rollback to v3.35.0:
|
||||
|
||||
```bash
|
||||
npm install @soulcraftlabs/brainy@3.35.0
|
||||
npm install @soulcraft/brainy@3.35.0
|
||||
```
|
||||
|
||||
**Note:** We don't anticipate any issues, but rollback is straightforward if needed.
|
||||
|
|
@ -367,7 +367,7 @@ if (stats.fairness.fairnessViolation) {
|
|||
|
||||
## Next Steps
|
||||
|
||||
1. ✅ **Upgrade:** `npm install @soulcraftlabs/brainy@latest`
|
||||
1. ✅ **Upgrade:** `npm install @soulcraft/brainy@latest`
|
||||
2. 📊 **Monitor:** Use `getCacheStats()` to verify performance improvements
|
||||
3. 🎯 **Tune:** Adjust based on recommendations (if needed)
|
||||
4. 📖 **Read:** [Operations Guide](../operations/capacity-planning.md) for capacity planning
|
||||
|
|
|
|||
|
|
@ -37,7 +37,7 @@ This single WASM file contains everything needed for sentence embeddings.
|
|||
|
||||
```bash
|
||||
# Bun as a runtime — supported and recommended
|
||||
bun add @soulcraftlabs/brainy
|
||||
bun add @soulcraft/brainy
|
||||
bun run server.ts
|
||||
```
|
||||
|
||||
|
|
|
|||
|
|
@ -1,99 +0,0 @@
|
|||
---
|
||||
title: Migrating to 9.0 — your fields and system fields
|
||||
slug: guides/namespace-migration
|
||||
public: true
|
||||
category: guides
|
||||
template: guide
|
||||
order: 1
|
||||
description: The simple story of the 9.0 field-addressing change and the mechanical checklist for updating your call sites — every miss fails loudly with the fix in the error.
|
||||
next:
|
||||
- concepts/field-addressing
|
||||
---
|
||||
|
||||
# Migrating to 9.0 — your fields and system fields
|
||||
|
||||
The one-sentence version: **your data's field names are now completely
|
||||
yours, the engine's own fields all live behind one `system.` prefix, and
|
||||
nothing in between can silently go wrong anymore.**
|
||||
|
||||
## What changed, simply
|
||||
|
||||
**1. Any field name just works.** Before 9.0 the engine quietly owned
|
||||
certain names. A field called `level` could be shadowed by the engine's
|
||||
internal index layer of the same name (sorts silently returned insertion
|
||||
order); names like `confidence` or `subtype` were rejected inside
|
||||
`metadata`; names like `content` or `id` were silently never indexed, so
|
||||
filtering on them returned nothing. All of that is gone. Any name —
|
||||
`level`, `confidence`, `type`, `id`, `content`, anything — is stored
|
||||
exactly as written and works with every feature: filtering, sorting,
|
||||
grouping, aggregation, search, and time-travel reads.
|
||||
|
||||
**2. The engine's fields moved behind `system.`.** The engine still keeps
|
||||
its own per-record bookkeeping — creation time, type, confidence, and so
|
||||
on. Those are reached one way only now: spelled out, e.g.
|
||||
`system.createdAt`, `system.type`. They are just as queryable and sortable
|
||||
as before. `orderBy: 'createdAt'` means *your* field named `createdAt`;
|
||||
`orderBy: 'system.createdAt'` means the engine's timestamp. No guessing,
|
||||
no priority rules.
|
||||
|
||||
**3. Storage keeps the two physically separate.** New records store your
|
||||
metadata in its own nested compartment, so a user field named
|
||||
`confidence` and the engine's confidence live side by side, both intact,
|
||||
through restarts, index rebuilds, and `asOf()` history. Old records stay
|
||||
readable forever; nothing rewrites your data.
|
||||
|
||||
**4. Mistakes are loud.** An ambiguous or unknown field name is a typed
|
||||
error naming the fix. Unimplemented options refuse instead of being
|
||||
ignored. The only forbidden name in your metadata is one literally
|
||||
starting with `system.`.
|
||||
|
||||
## The mechanical checklist
|
||||
|
||||
Every missed site fails **loudly** with the correction in the error
|
||||
message — nothing silently changes meaning. Sweep these patterns:
|
||||
|
||||
| Before (8.x) | After (9.0) |
|
||||
|---|---|
|
||||
| `orderBy: 'createdAt'` (meaning the engine timestamp) | `orderBy: 'system.createdAt'` |
|
||||
| `where: { subtype: 'invoice' }` (the engine subtype) | `where: { 'system.subtype': 'invoice' }` |
|
||||
| `where: { confidence: { greaterThan: 0.8 } }` (the engine scalar) | `where: { 'system.confidence': { greaterThan: 0.8 } }` |
|
||||
| `groupBy: ['noun']` or `groupBy: ['type']` | `groupBy: ['system.type']` |
|
||||
| `where: { visibility: 'internal' }` / `{ service: … }` (engine values) | `'system.visibility'` / `'system.service'` |
|
||||
| `metadata: { confidence: 0.9 }` expecting a throw or a lift to the engine scalar | it is YOUR field now — set the engine scalar via the `confidence` param |
|
||||
| `new Brainy({ reservedFieldPolicy: … })` | remove the option (it throws with this note) |
|
||||
| `find({ cursor })` / `includeRelations` / `writeOnly` | refuse with `UnsupportedFindOptionError` — they were silently ignored before |
|
||||
|
||||
If a bare name in a query was genuinely *your* field all along (`orderBy:
|
||||
'score'`, `where: { status: 'active' }`), **change nothing** — bare names
|
||||
mean your fields, always.
|
||||
|
||||
## What happens at first open
|
||||
|
||||
Each existing database rebuilds its derived indexes once, automatically,
|
||||
at the first open on 9.0 (index epoch 3 — the index keys split the two
|
||||
namespaces). One-time cost, observable via `getIndexStatus()`; no manual
|
||||
step, and your stored data is not modified.
|
||||
|
||||
## For tooling and raw-record readers
|
||||
|
||||
If you read raw stored records (fact-log scanners, export tooling), use
|
||||
the exported shape-aware splitters — they handle both record eras:
|
||||
|
||||
```typescript
|
||||
import { splitNounMetadataRecord } from '@soulcraftlabs/brainy'
|
||||
const { reserved, custom } = splitNounMetadataRecord(rawRecord)
|
||||
// reserved = engine fields · custom = the user's bag, ANY names
|
||||
```
|
||||
|
||||
Feature detection (never version-sniff):
|
||||
|
||||
```typescript
|
||||
import * as brainy from '@soulcraftlabs/brainy'
|
||||
const lawActive = 'FIELD_ADDRESSING_CAPABILITY' in brainy // 'field-addressing/v1'
|
||||
```
|
||||
|
||||
## Where to go next
|
||||
|
||||
- [Field addressing](../concepts/field-addressing.md) — the full contract:
|
||||
the ten system scalars, the relation mirror, refusal semantics, and the
|
||||
cross-engine ordering guarantees.
|
||||
|
|
@ -9,7 +9,7 @@ Complete guide to integrating Brainy with Next.js applications, covering App Rou
|
|||
```bash
|
||||
npx create-next-app@latest my-brainy-app
|
||||
cd my-brainy-app
|
||||
npm install @soulcraftlabs/brainy
|
||||
npm install @soulcraft/brainy
|
||||
```
|
||||
|
||||
### Basic Setup
|
||||
|
|
@ -18,7 +18,7 @@ npm install @soulcraftlabs/brainy
|
|||
// app/components/BrainyProvider.jsx
|
||||
'use client'
|
||||
import { createContext, useContext, useEffect, useState } from 'react'
|
||||
import { Brainy } from '@soulcraftlabs/brainy'
|
||||
import { Brainy } from '@soulcraft/brainy'
|
||||
|
||||
const BrainyContext = createContext()
|
||||
|
||||
|
|
@ -271,7 +271,7 @@ export default function SearchPage() {
|
|||
|
||||
```javascript
|
||||
// app/api/search/route.js (App Router)
|
||||
import { Brainy } from '@soulcraftlabs/brainy'
|
||||
import { Brainy } from '@soulcraft/brainy'
|
||||
|
||||
let brain = null
|
||||
|
||||
|
|
@ -332,7 +332,7 @@ export async function GET() {
|
|||
|
||||
```javascript
|
||||
// pages/api/search.js (Pages Router)
|
||||
import { Brainy } from '@soulcraftlabs/brainy'
|
||||
import { Brainy } from '@soulcraft/brainy'
|
||||
|
||||
let brain = null
|
||||
|
||||
|
|
@ -374,7 +374,7 @@ export default async function handler(req, res) {
|
|||
|
||||
```javascript
|
||||
// app/api/data/route.js
|
||||
import { Brainy } from '@soulcraftlabs/brainy'
|
||||
import { Brainy } from '@soulcraft/brainy'
|
||||
|
||||
let brain = null
|
||||
|
||||
|
|
@ -418,7 +418,7 @@ export async function POST(request) {
|
|||
```jsx
|
||||
// app/actions/brainy.js
|
||||
'use server'
|
||||
import { Brainy } from '@soulcraftlabs/brainy'
|
||||
import { Brainy } from '@soulcraft/brainy'
|
||||
|
||||
let brain = null
|
||||
|
||||
|
|
@ -630,7 +630,7 @@ CMD ["npm", "start"]
|
|||
/** @type {import('next').NextConfig} */
|
||||
const nextConfig = {
|
||||
experimental: {
|
||||
serverComponentsExternalPackages: ['@soulcraftlabs/brainy']
|
||||
serverComponentsExternalPackages: ['@soulcraft/brainy']
|
||||
},
|
||||
webpack: (config, { isServer }) => {
|
||||
if (!isServer) {
|
||||
|
|
@ -797,7 +797,7 @@ export function rateLimit(req, limit = 100, window = 60000) {
|
|||
// app/contexts/BrainyContext.jsx
|
||||
'use client'
|
||||
import { createContext, useContext, useReducer, useEffect } from 'react'
|
||||
import { Brainy } from '@soulcraftlabs/brainy'
|
||||
import { Brainy } from '@soulcraft/brainy'
|
||||
|
||||
const BrainyContext = createContext()
|
||||
|
||||
|
|
@ -873,7 +873,7 @@ import { BrainyProvider } from '../app/components/BrainyProvider'
|
|||
import { Search } from '../app/components/Search'
|
||||
|
||||
// Mock Brainy
|
||||
jest.mock('@soulcraftlabs/brainy', () => ({
|
||||
jest.mock('@soulcraft/brainy', () => ({
|
||||
Brainy: jest.fn().mockImplementation(() => ({
|
||||
init: jest.fn().mockResolvedValue(undefined),
|
||||
find: jest.fn().mockResolvedValue([
|
||||
|
|
|
|||
|
|
@ -32,7 +32,7 @@ Brainy 7.31.0 adds a per-entity revision counter so multiple writers can coordin
|
|||
Every distributed-job scheduler eventually wants this exact loop:
|
||||
|
||||
```ts
|
||||
import { Brainy, RevisionConflictError } from '@soulcraftlabs/brainy'
|
||||
import { Brainy, RevisionConflictError } from '@soulcraft/brainy'
|
||||
|
||||
const LOCK_ID = '...uuid for this job slot...'
|
||||
|
||||
|
|
@ -137,7 +137,7 @@ await brain.addIfMissing({ // ← not a real API
|
|||
It's race-prone as a plain read-then-write: two concurrent imports both see "not found," both insert, you get duplicates. Without a unique-index primitive (which Brainy doesn't have today), close the race with whole-store CAS — read at a pinned generation, then commit only if nothing moved:
|
||||
|
||||
```ts
|
||||
import { GenerationConflictError } from '@soulcraftlabs/brainy'
|
||||
import { GenerationConflictError } from '@soulcraft/brainy'
|
||||
|
||||
async function addIfMissingByEmail(email: string, data: string) {
|
||||
for (let attempt = 0; attempt < 5; attempt++) {
|
||||
|
|
@ -180,35 +180,3 @@ Brainy 8.0 has exactly two write-coordination counters, at two granularities:
|
|||
They compose: a `transact()` batch can carry per-entity `ifRev` checks *and* a whole-store `ifAtGeneration`; any failed check rejects the entire batch before anything is staged. Generations also power snapshots and time travel (`brain.now()`, `brain.asOf()`, `db.persist()`) — see the [consistency model](../concepts/consistency-model.md) and [Snapshots & Time Travel](./snapshots-and-time-travel.md).
|
||||
|
||||
A snapshot or historical view captures each entity *including* its `_rev` at that moment, so reading the past and writing back with `ifRev` against the live state works exactly as you'd hope: the write fails if the entity moved since the state you copied from.
|
||||
|
||||
## The transact envelope: batch size, budget, and bulk imports
|
||||
|
||||
`transact()` applies its batch atomically under one commit — which means the whole batch
|
||||
shares one **apply budget**. Since 8.7.0 the budget scales with the batch:
|
||||
`max(30 s, opCount × 2 s)`, or exactly what you pass as `timeoutMs`. A tripped budget rolls
|
||||
the entire batch back (nothing partial survives) and throws a retryable
|
||||
`TransactionTimeoutError` that names the operation it stopped at, the batch size, and the
|
||||
elapsed vs budgeted time — a diagnosis, not just a failure:
|
||||
|
||||
```
|
||||
Transaction timed out at operation 41/120 ('add') — 246012ms elapsed, budget 240000ms.
|
||||
The batch rolled back atomically; retry with a higher timeoutMs or a smaller batch.
|
||||
```
|
||||
|
||||
Practical envelope guidance for bulk work:
|
||||
|
||||
1. **Precompute embeddings outside the commit path.** Embedding inside `transact()` spends
|
||||
the budget on model inference. Use `brain.embedBatch(texts)` and pass each vector via
|
||||
the op's `vector` field — the commit then pays only storage costs, and a retried batch
|
||||
never re-pays inference. (The win is *where* the inference happens, not raw embedding
|
||||
throughput: on the default WASM engine, batch and sequential embedding measure
|
||||
comparably, ~160 ms/text; native embedding providers may batch faster.)
|
||||
2. **Chunk very large imports** into batches of a few hundred ops with one `transact()`
|
||||
each. You lose whole-import atomicity but keep per-chunk atomicity, bounded memory, and
|
||||
resumability — pair with `ifAbsent` upserts so a retried chunk is idempotent.
|
||||
3. **Slow disks change the math, not the contract.** On network-attached storage a single
|
||||
op can cost ~2 s (canonical write + fsync + index maintenance). The scaled default
|
||||
absorbs that; pass an explicit `timeoutMs` only when you know better than the scale.
|
||||
4. **`addMany`/`relateMany` are the convenience tier** — they chunk and batch-embed for
|
||||
you, with per-item error reporting instead of batch atomicity. Choose by what you need:
|
||||
atomic-all-or-nothing → `transact()`; resilient bulk load → `addMany`.
|
||||
|
|
|
|||
|
|
@ -18,13 +18,13 @@ Get Brainy running in under a minute.
|
|||
## 1. Install
|
||||
|
||||
```bash
|
||||
npm install @soulcraftlabs/brainy
|
||||
npm install @soulcraft/brainy
|
||||
```
|
||||
|
||||
## 2. Initialize
|
||||
|
||||
```typescript
|
||||
import { Brainy, NounType, VerbType } from '@soulcraftlabs/brainy'
|
||||
import { Brainy, NounType, VerbType } from '@soulcraft/brainy'
|
||||
|
||||
const brain = new Brainy()
|
||||
await brain.init()
|
||||
|
|
@ -67,7 +67,7 @@ await brain.relate({
|
|||
## 5. Query with Triple Intelligence
|
||||
|
||||
```typescript
|
||||
import type { Result } from '@soulcraftlabs/brainy'
|
||||
import type { Result } from '@soulcraft/brainy'
|
||||
|
||||
// All three search paradigms in one call
|
||||
const results: Result[] = await brain.find({
|
||||
|
|
|
|||
|
|
@ -9,7 +9,6 @@ description: Recipes for the Db API — instant backups with persist(), restore,
|
|||
next:
|
||||
- concepts/consistency-model
|
||||
- guides/optimistic-concurrency
|
||||
- guides/external-backups
|
||||
---
|
||||
|
||||
# Snapshots & Time Travel
|
||||
|
|
@ -42,10 +41,6 @@ bytes. Cross-device targets fall back to per-file byte copies, and
|
|||
persisting an in-memory brain serializes it to the same directory layout —
|
||||
a real, durable store.
|
||||
|
||||
> Archiving a brain directory with **external tools** (`tar`, `rsync`, `cp`)?
|
||||
> Some index files are sparse and can explode to their apparent size under a
|
||||
> naive copy — see [External Backups & Sparse Storage](/docs/guides/external-backups).
|
||||
|
||||
Two things to know:
|
||||
|
||||
- `persist()` requires the view to still be the store's **latest**
|
||||
|
|
@ -344,12 +339,8 @@ For per-entity write coordination (rather than whole-store history), the
|
|||
## Keeping history bounded
|
||||
|
||||
Under Model-B every write is a generation, so history can grow quickly —
|
||||
Brainy auto-compacts at `close()` (time-bounded per pass) under the
|
||||
**`retention`** knob (configured on the constructor). Since 8.9.0, `flush()`
|
||||
never compacts: flushing is durability work and costs only what the current
|
||||
window's writes cost, regardless of history backlog. A long-lived writer that
|
||||
never closes keeps its history until its next explicit `compactHistory()` —
|
||||
schedule one in your maintenance window if you run bounded retention:
|
||||
Brainy auto-compacts on every `flush()`/`close()` under the **`retention`**
|
||||
knob (configured on the constructor):
|
||||
|
||||
```typescript
|
||||
// Zero-config: ADAPTIVE — keep as much history as free disk/RAM allows,
|
||||
|
|
@ -363,13 +354,10 @@ new Brainy({ retention: 'all' })
|
|||
new Brainy({ retention: { maxGenerations: 1000, maxAge: 7 * 86_400_000, maxBytes: 512 * 1024 ** 2 } })
|
||||
```
|
||||
|
||||
Reclaim manually at any time (the same caps, plus an optional per-pass time
|
||||
budget for maintenance windows — an early stop is a consistent prefix and the
|
||||
next pass resumes):
|
||||
Reclaim manually at any time (the same caps):
|
||||
|
||||
```typescript
|
||||
await brain.compactHistory({ maxGenerations: 100, maxAge: 7 * 24 * 60 * 60 * 1000 })
|
||||
await brain.compactHistory({ maxBytes: 512 * 1024 ** 2, timeBudgetMs: 10_000 })
|
||||
```
|
||||
|
||||
Compaction never breaks a pinned read — record-sets are reclaimed only when
|
||||
|
|
|
|||
|
|
@ -11,7 +11,7 @@
|
|||
### One Interface for Everything
|
||||
|
||||
```typescript
|
||||
import { Brainy } from '@soulcraftlabs/brainy'
|
||||
import { Brainy } from '@soulcraft/brainy'
|
||||
|
||||
const brain = await Brainy.create()
|
||||
|
||||
|
|
@ -78,7 +78,7 @@ interface ImportProgress {
|
|||
|
||||
```typescript
|
||||
import { useState } from 'react'
|
||||
import { Brainy } from '@soulcraftlabs/brainy'
|
||||
import { Brainy } from '@soulcraft/brainy'
|
||||
|
||||
function UniversalImportProgress({ file }: { file: File }) {
|
||||
const [progress, setProgress] = useState({
|
||||
|
|
@ -177,7 +177,7 @@ function UniversalImportProgress({ file }: { file: File }) {
|
|||
|
||||
```typescript
|
||||
import ora from 'ora'
|
||||
import { Brainy } from '@soulcraftlabs/brainy'
|
||||
import { Brainy } from '@soulcraft/brainy'
|
||||
|
||||
async function importWithProgress(filePath: string) {
|
||||
const spinner = ora('Starting import...').start()
|
||||
|
|
|
|||
|
|
@ -28,7 +28,7 @@ on-disk layout (memory's "disk" is a JS Map).
|
|||
## Quick start
|
||||
|
||||
```ts
|
||||
import { Brainy } from '@soulcraftlabs/brainy'
|
||||
import { Brainy } from '@soulcraft/brainy'
|
||||
|
||||
// Filesystem (recommended for any persistent workload):
|
||||
const brain = new Brainy({
|
||||
|
|
@ -134,7 +134,7 @@ config; the `type` is optional.
|
|||
If you want to skip the factory:
|
||||
|
||||
```ts
|
||||
import { FileSystemStorage, MemoryStorage } from '@soulcraftlabs/brainy'
|
||||
import { FileSystemStorage, MemoryStorage } from '@soulcraft/brainy'
|
||||
|
||||
const fsStorage = new FileSystemStorage('./brainy-data')
|
||||
const memStorage = new MemoryStorage()
|
||||
|
|
|
|||
|
|
@ -34,7 +34,7 @@ Three layers solve this:
|
|||
### Write
|
||||
|
||||
```typescript
|
||||
import { Brainy, NounType } from '@soulcraftlabs/brainy'
|
||||
import { Brainy, NounType } from '@soulcraft/brainy'
|
||||
|
||||
const brain = new Brainy()
|
||||
await brain.init()
|
||||
|
|
@ -240,7 +240,7 @@ await brain.migrateField({
|
|||
A realistic adoption sequence for a brain that started without these primitives:
|
||||
|
||||
```typescript
|
||||
import { Brainy, NounType } from '@soulcraftlabs/brainy'
|
||||
import { Brainy, NounType } from '@soulcraft/brainy'
|
||||
|
||||
const brain = new Brainy({ storage: { type: 'filesystem', path: './brain-data' } })
|
||||
await brain.init()
|
||||
|
|
|
|||
|
|
@ -25,7 +25,7 @@ content — and how 8.0 recovers it for you.
|
|||
|
||||
## TL;DR
|
||||
|
||||
- **Just upgrade to `@soulcraftlabs/brainy@8.0.12` (or later) and open the store.**
|
||||
- **Just upgrade to `@soulcraft/brainy@8.0.12` (or later) and open the store.**
|
||||
If a previous upgrade left VFS content stranded, 8.0.12 **heals it on open**,
|
||||
with no operator action.
|
||||
- Want to force or script it? Call **`await brain.vfs.adoptOrphanedBlobs()`**.
|
||||
|
|
@ -90,7 +90,7 @@ So the operator action for a stranded store is simply: **upgrade to 8.0.12 and
|
|||
open it.**
|
||||
|
||||
```ts
|
||||
import { Brainy } from '@soulcraftlabs/brainy'
|
||||
import { Brainy } from '@soulcraft/brainy'
|
||||
|
||||
// Opening the store is all that is required — recovery runs during init().
|
||||
const brain = new Brainy({ storage: { type: 'filesystem', path: '/data/my-store' } })
|
||||
|
|
@ -182,5 +182,5 @@ and opening each store is sufficient.
|
|||
The recovery is copy-only, so no rollback of the recovery itself is ever needed.
|
||||
If you need to roll back the **whole** 7→8 upgrade, restore the directory from
|
||||
your pre-upgrade backup (retained automatically while recovery is incomplete, or
|
||||
your own snapshot) and pin `@soulcraftlabs/brainy@7.x`. 8.0 does not keep the old
|
||||
your own snapshot) and pin `@soulcraft/brainy@7.x`. 8.0 does not keep the old
|
||||
branch layout in place, so a directory-level restore is the rollback path.
|
||||
|
|
|
|||
|
|
@ -12,7 +12,7 @@ Complete guide to integrating Brainy with Vue.js applications, covering Vue 3, N
|
|||
npm create vue@latest my-brainy-app
|
||||
cd my-brainy-app
|
||||
npm install
|
||||
npm install @soulcraftlabs/brainy
|
||||
npm install @soulcraft/brainy
|
||||
```
|
||||
|
||||
### Basic Setup
|
||||
|
|
@ -574,7 +574,7 @@ Nuxt's server engine (Nitro) is the natural home for Brainy: it runs on Node/Bun
|
|||
|
||||
```javascript
|
||||
// server/utils/brain.js (server-only — Nitro never bundles this into the client)
|
||||
import { Brainy } from '@soulcraftlabs/brainy'
|
||||
import { Brainy } from '@soulcraft/brainy'
|
||||
|
||||
let brainPromise
|
||||
|
||||
|
|
@ -1201,7 +1201,7 @@ import vue from '@vitejs/plugin-vue'
|
|||
export default defineConfig({
|
||||
plugins: [vue()],
|
||||
ssr: {
|
||||
external: ['@soulcraftlabs/brainy']
|
||||
external: ['@soulcraft/brainy']
|
||||
}
|
||||
})
|
||||
```
|
||||
|
|
|
|||
|
|
@ -24,7 +24,7 @@ Brainy's neural extraction system uses a **4-signal ensemble architecture** to c
|
|||
### Method 1: Brain Instance (Recommended)
|
||||
|
||||
```typescript
|
||||
import { Brainy, NounType } from '@soulcraftlabs/brainy'
|
||||
import { Brainy, NounType } from '@soulcraft/brainy'
|
||||
|
||||
const brain = new Brainy()
|
||||
await brain.init()
|
||||
|
|
@ -62,9 +62,9 @@ const people = await brain.extractEntities('...', {
|
|||
import {
|
||||
SmartExtractor,
|
||||
SmartRelationshipExtractor
|
||||
} from '@soulcraftlabs/brainy'
|
||||
} from '@soulcraft/brainy'
|
||||
// Or use subpath imports:
|
||||
import { SmartExtractor } from '@soulcraftlabs/brainy/neural/SmartExtractor'
|
||||
import { SmartExtractor } from '@soulcraft/brainy/neural/SmartExtractor'
|
||||
|
||||
const brain = new Brainy()
|
||||
await brain.init()
|
||||
|
|
@ -176,7 +176,7 @@ const withVectors = await brain.extractEntities(text, {
|
|||
**Direct entity type classifier.** Use when you have pre-detected candidates or need custom configuration.
|
||||
|
||||
```typescript
|
||||
import { SmartExtractor, FormatContext } from '@soulcraftlabs/brainy'
|
||||
import { SmartExtractor, FormatContext } from '@soulcraft/brainy'
|
||||
|
||||
const extractor = new SmartExtractor(brain, {
|
||||
minConfidence: 0.7, // Threshold
|
||||
|
|
@ -229,7 +229,7 @@ interface ExtractionResult {
|
|||
**Relationship type classifier.** Determines verb/relationship types between entities.
|
||||
|
||||
```typescript
|
||||
import { SmartRelationshipExtractor } from '@soulcraftlabs/brainy'
|
||||
import { SmartRelationshipExtractor } from '@soulcraft/brainy'
|
||||
|
||||
const relExtractor = new SmartRelationshipExtractor(brain, {
|
||||
minConfidence: 0.6,
|
||||
|
|
@ -286,7 +286,7 @@ const rel = await relExtractor.infer(
|
|||
**Full extraction orchestrator.** Handles candidate detection, classification, and deduplication.
|
||||
|
||||
```typescript
|
||||
import { NeuralEntityExtractor } from '@soulcraftlabs/brainy'
|
||||
import { NeuralEntityExtractor } from '@soulcraft/brainy'
|
||||
|
||||
const extractor = new NeuralEntityExtractor(brain)
|
||||
|
||||
|
|
@ -607,7 +607,7 @@ const locations = entities.filter(e => e.type === NounType.Location)
|
|||
### Example 2: Excel Data Classification
|
||||
|
||||
```typescript
|
||||
import { SmartExtractor } from '@soulcraftlabs/brainy'
|
||||
import { SmartExtractor } from '@soulcraft/brainy'
|
||||
|
||||
const extractor = new SmartExtractor(brain)
|
||||
|
||||
|
|
@ -629,7 +629,7 @@ for (let i = 0; i < cells.length; i++) {
|
|||
### Example 3: Relationship Extraction
|
||||
|
||||
```typescript
|
||||
import { SmartRelationshipExtractor } from '@soulcraftlabs/brainy'
|
||||
import { SmartRelationshipExtractor } from '@soulcraft/brainy'
|
||||
|
||||
const relExtractor = new SmartRelationshipExtractor(brain)
|
||||
|
||||
|
|
|
|||
|
|
@ -1,86 +0,0 @@
|
|||
# The Path Registry — brainy's twin table
|
||||
|
||||
The brainy half of the cross-engine Path Registry (the native accelerator
|
||||
maintains the master list; IDs are shared and stable — `LC3`, `DP7`, … are
|
||||
citable in commits, board rounds, release notes, and pins). Every row owes
|
||||
five things: **service class** (INDEX-SERVED | BOUNDED-FALLBACK, announced |
|
||||
TYPED REFUSAL), **latency budget** at 1k/10k/100k/1M (design bar: billions),
|
||||
**lifecycle behavior**, **failure narration**, and a **test pin**. A path not
|
||||
in this registry does not ship; an unregistered path is a red gate in the
|
||||
scan audit.
|
||||
|
||||
**The availability bar governing every row: user-visible downtime is
|
||||
seconds, at restart only.** Migration, heal, compaction, embedding, and
|
||||
retention run behind the doors — yielding, budget-capped, narrated. No path
|
||||
may hold the doors while it does housekeeping.
|
||||
|
||||
Status legend: ✅ contracted + pinned (test cited) · 🟡 partial (what holds
|
||||
and what's missing, stated) · 🔴 owed (named, never silent).
|
||||
|
||||
## LC — Lifecycle
|
||||
|
||||
| ID | Brainy row | Status |
|
||||
|----|-----------|--------|
|
||||
| LC1 | Same-version reopen adopts everything: brain-format epoch match → zero rebuilds; aggregation state adopts by stamp; persisted indexes load. | ✅ `tests/unit/brainy/brain-format-handshake` + `migration-deference` (no-drift reopen never rebuilds) |
|
||||
| LC2 | New empty brain: doors immediate. | ✅ exercised by every suite's setup |
|
||||
| LC3 | Upgrade, same epoch: as LC1 — new code on unchanged formats owes nothing at open. | ✅ same pins as LC1 (epoch equality is the gate) |
|
||||
| LC4 | Upgrade with epoch migration: TODAY brainy's epoch rebuild runs at open before doors. | 🔴 **owed — the sev's lockout row.** The doors-open-serving-old-structures design (yielding installments + atomic swap) lands measured-and-gated behind the service-class pair, per the lifecycle-sprint choreography. Acceptance case: the 9,184-row hours-lockout. |
|
||||
| LC5 | Crash recovery: bounded, resumable, narrated. Aggregation leg ✅ (behind-stamp → incremental catch-up off the fact log + time-travel reconciliation, capped at 5,000 affected before an ANNOUNCED rescan). Vector/metadata legs ride epoch machinery (rebuild-from-canonical, narrated). | 🟡 aggregation pinned (`tests/integration/aggregation-lifecycle-catchup`); the rebuild legs are narrated but not yet installment-yielding (couples to LC4) |
|
||||
| LC6 | Shutdown under load: close() drains the background flush flight, tears down cadence timers, runs ONE time-bounded compaction pass (~5s budget, resumable). | 🟡 pinned for flush/compaction (8.9.0 suites); SIGTERM drain budget not yet declared |
|
||||
| LC7 | Rollback/downgrade: an N−1 build opening an N brain. | 🔴 owed — no declared read-compat window or typed refusal today (epoch mismatch triggers a rebuild, not a refusal; v2 nested-bag records read as a phantom user field on pre-law builds). Needs the declared-window contract. |
|
||||
| LC8 | Relocatable brain directory: no absolute paths in artifacts; persist()/load() round-trips. | 🟡 persist/load pinned; byte-for-byte relocation depot cases are the pair gate's (shared corpora) |
|
||||
| LC9 | Double-open: second writer gets a typed lock refusal (PID-liveness + heartbeat stale detection; `force` escape hatch logs loudly). | ✅ writer-lock suites (8.7.1) |
|
||||
|
||||
## DP — Data plane
|
||||
|
||||
| ID | Brainy row | Status |
|
||||
|----|-----------|--------|
|
||||
| DP1 | `get()` by id: direct storage read + hydrate. INDEX-SERVED (id-mapped). Milliseconds at every scale. | ✅ exercised everywhere; budget rides the pair speed table |
|
||||
| DP2 | `find({query})`: embed + vector search. The embed dominates (native side owns the budget); JS HNSW serves the search leg. | 🟡 300ms-class p95 is the pair speed-table row; brainy-alone budget declared there |
|
||||
| DP3 | Filtered/sorted list: column top-K when the field is columnized (INDEX-SERVED, zero canonical reads on the sorted page — value pairs come from ONE batched metadata-record pass); no-column fallback is BOUNDED-ANNOUNCED (one batch pass, announces once per field past 500 rows); unknown field → TYPED REFUSAL naming both candidate spellings. | ✅ `tests/unit/utils/metadataIndex-sort-callshape` (zero per-row reads, batch-only — latency-blind) + `metadataIndex-nested-orderby` (dotted keys serve-or-refuse) + `tests/integration/orderby-sort-bug` |
|
||||
| DP4 | Aggregation/stats: ALWAYS answers. Write-time incremental; behind-stamp reconciles incrementally; genuine rebuilds go through the native parallel door or the paged JS walk; nothing ever latches off; before-image-less deletes flag a LOUD rescan, never a silent skip. | ✅ `tests/integration/aggregation-lifecycle-catchup` + `tests/unit/aggregation/aggregation-provider-rebuild` |
|
||||
| DP5 | Graph traversal: `related()` paged via adjacency; whole-graph analytics carry declared cost. | 🟡 paged reads pinned; analytics cost-class declaration owed (rides VENUE-GRAPH-TRUST audit tool) |
|
||||
| DP6 | Single write: ack at the canonical commit; visibility committed at ack (the atomic vector update kills the remove→add dark window); maintenance NEVER holds the ack (background flush cadence — THE ACK LAW pins: a hung flush cannot block a write, a hung EMBEDDER cannot block a write). | ✅ `tests/unit/brainy/persistence-policy` + `tests/unit/hnsw/update-item-atomic` + `tests/integration/deferred-embedding` |
|
||||
| DP7 | Bulk ingest: sustained rate holds flat — per-write maintenance taxes must not grow with brain size (A4 removed caller-flush convoys; deferred embedding removes the per-write embed tax where opted). | 🟡 the decay-curve row is a pair speed-table RED GATE; brainy-alone sustained-rate run rides the same corpora |
|
||||
| DP8 | Read under write pressure: no flicker window — a row that exists is never invisible to recall, even transiently (same-vector re-index is a no-op; changed-vector swaps in place, node never leaves the index; deferred updates serve the OLD vector until the atomic swap — stale-beats-absent). | ✅ brainy leg pinned (`tests/unit/hnsw/update-item-atomic` 9/9 + `deferred-embedding` stale-beats-absent); the symmetry property suite + runtime sentinels remain the B4 program |
|
||||
| — | **As-of semantic recall** (time-travel vector search): `asOf(G).find()` serves the vectors AS THEY STOOD at G — byte-exact past vectors, tombstone masking, the deferred-embed cell honest on the vector leg, TYPED refusal beyond the head. Brainy-alone leg = ephemeral at-generation materialization (documented O(n log n at G) build, bounded); the at-scale leg rides the accelerated provider's as-of index. | ✅ `tests/integration/asof-semantic-recall` 4/4 (registry ID pending the master table's mint) |
|
||||
| — | **The lazy-open gate honors EVERY provider's not-ready report** (a not-ready metadata provider can no longer latch the silent-empty state under `disableAutoRebuild`). | ✅ `tests/unit/brainy/lazy-notready-honor` |
|
||||
|
||||
## MT — Maintenance (never in the door path)
|
||||
|
||||
| ID | Brainy row | Status |
|
||||
|----|-----------|--------|
|
||||
| MT1 | Flush/checkpoint: ENGINE-OWNED cadence (write-count/interval/idle triggers, single-flight, background, loud on failure; callers never flush in hot paths; `flush()` stays as an awaitable barrier). | ✅ `tests/unit/brainy/persistence-policy` |
|
||||
| MT2 | Compaction: never on flush (durability-only law, 8.9.0); close-time pass time-budgeted + resumable; explicit `compactHistory({timeBudgetMs})`. | ✅ 8.9.0 suites |
|
||||
| MT3 | Index upkeep (mapper folds, delta promotion): native-side machinery; brainy's JS legs are small and synchronous-cheap. | 🟡 declared; yield audit rides the pair |
|
||||
| MT4 | Heal/rebuild walks (`repairIndex`, backfill walks): paged; failure latches with cooldown; NOT yet yield-to-foreground installments. | 🔴 owed — the priority-isolation clause (couples to LC4; same choreography) |
|
||||
| MT5 | Deferred embedding worker: ack at durability, durable pending markers (written BEFORE the commit — orphan-safe), crash-recovered at open via a bounded prefix listing, single-flight, 60s hang guard, `awaitPendingEmbeds()` barrier + `pendingEmbeds` gauge. VFS write paths adopt it end-to-end. | ✅ `tests/integration/deferred-embedding` 5/5 |
|
||||
| MT6 | Retention/archival walks: retention `'all'` does nothing by design; bounded-retention reclaim is close-time/explicit only. | 🟡 8.9.0 behavior pinned; archival profile is the co-frozen D1+D3 unit |
|
||||
|
||||
## FM — Failure modes
|
||||
|
||||
| ID | Brainy row | Status |
|
||||
|----|-----------|--------|
|
||||
| FM1 | Disk full / IO error mid-op: transaction rollback + typed error; failed rollback → StoreInconsistentError quarantines writes until repairIndex(). | 🟡 rollback paths pinned; explicit disk-full depot case owed |
|
||||
| FM2 | Memory pressure: query limits + reserved-memory config; unified cache eviction. | 🟡 declared budgets; cascade pin owed |
|
||||
| FM3 | Torn/corrupt file on open: malformed brain-format marker → safe rebuild (never trusting a bad epoch); corrupt records surface loudly. | 🟡 marker pin ✅ (`brain-format-handshake`); broader quarantine is native-side |
|
||||
| FM4 | Native module unavailable: plugin load failure is LOUD (version-coupling law throws on range mismatch — never silently version-drifted); JS engine serves with its own declared budgets, named as the active backend in op names. | ✅ `tests/unit/plugin-version-coupling` + op-name stamping |
|
||||
|
||||
## FL — Fleet
|
||||
|
||||
| ID | Brainy row | Status |
|
||||
|----|-----------|--------|
|
||||
| FL1 | Cold open on demand: LC1's adopt-everything open; warm() available for eager paths. | 🟡 open cost pinned at LC1; millisecond budget rides the speed table |
|
||||
| FL2–FL4 | Boot storm / upgrade wave / isolation: fleet-layer policies over LC1/LC4 — engine leg = budgeted opens + LC4's behind-doors migration. | 🔴 owed with LC4 |
|
||||
| FL5 | Brain as product object: create instant (LC2) · erase = `clear()` explicit + complete · export = portable-graph, canon-complete mode available. | ✅ clear-persistence + portable-graph + canonical-enumeration suites |
|
||||
|
||||
## Status summary
|
||||
|
||||
Contracted + pinned this train: **DP3, DP4, DP6, DP8(brainy leg), MT1,
|
||||
MT5, LC5(aggregation), the lazy-open not-ready gate, LC1/LC3/LC9, FM4,
|
||||
FL5** — each with the cited test. Owed, in production-risk order, all
|
||||
coupled to the priority-isolation program the lifecycle sev opened: **LC4
|
||||
(doors-open migration), MT4 (yielding heals), LC7 (downgrade contract),
|
||||
LC6 (SIGTERM budget), FL2–FL4, FM1/FM2 depot cases, B4 symmetry suite +
|
||||
sentinels.** Rows move from owed to contracted only with a cited test —
|
||||
none lands by prose.
|
||||
|
|
@ -1,83 +0,0 @@
|
|||
---
|
||||
title: Performance Envelopes
|
||||
slug: guides/performance-envelopes
|
||||
public: true
|
||||
category: guides
|
||||
template: guide
|
||||
order: 40
|
||||
description: Measured per-operation latency envelopes at stated scales — what to expect, on what hardware, and exactly how each number was produced.
|
||||
next:
|
||||
- guides/find-limits
|
||||
---
|
||||
|
||||
# Performance Envelopes
|
||||
|
||||
Every number on this page is **measured, never projected** — produced by the script
|
||||
cited at the bottom, against the built package (the artifact you install), on the stated
|
||||
hardware. Each entry says what was measured, at what scale, on which storage backend.
|
||||
When a release touches a measured path, that operation is re-measured and this page
|
||||
updates in the same release.
|
||||
|
||||
Two scopes to keep straight:
|
||||
|
||||
- **These envelopes are the pure-JS engine** (no native accelerator registered) on
|
||||
filesystem storage. This is the floor every deployment gets from `npm install` alone.
|
||||
- **Accelerated deployments** (the optional native provider) publish their own numbers —
|
||||
this page never claims them.
|
||||
|
||||
## Read operations
|
||||
|
||||
Reads are where the architecture pays off: after the write path has done its indexing
|
||||
work, queries answer from purpose-built indexes without scanning.
|
||||
|
||||
| Operation | 1,000 entities | 10,000 entities | Notes |
|
||||
|---|---|---|---|
|
||||
| `get(id)` (warm) | p50 < 0.1ms | p50 < 0.1ms | served from cache/metadata index |
|
||||
| `find` (metadata: indexed equality + range, limit 100) | p50 1.0ms · p95 1.8ms | p50 7.0ms · p95 8.9ms | column-store bitmap paths |
|
||||
| `related(id)` (per-node adjacency) | p50 < 0.1ms · p95 0.2ms | p50 < 0.1ms | LSM adjacency index — O(degree), scale-independent |
|
||||
| `find` (semantic: embed + HNSW, 1k docs) | p50 178ms · p95 393ms | — | dominated by WASM query embedding (measured on a machine under concurrent load — treat the p95 as an upper bound); the vector search itself is single-digit ms |
|
||||
|
||||
## Write operations
|
||||
|
||||
Under Model-B **every write is its own durable generation** — a single-op `add` pays
|
||||
serialization, before-image staging, and fsync before it acks. That durability is priced
|
||||
into the write path visibly, by design:
|
||||
|
||||
| Operation | 1,000 entities | 10,000 entities | Notes |
|
||||
|---|---|---|---|
|
||||
| `add` (single-op) | p50 167ms · p95 171ms | p50 165ms · p95 172ms | full durable generation per write — flat across scale |
|
||||
| `addMany` (bulk) | ~163ms/entity | ~187ms/entity | **currently per-item commits** — see the honest note below |
|
||||
| `relateMany` | ~0.8ms/edge | ~0.9ms/edge | edges batch efficiently today |
|
||||
| `flush` (steady-state, 1 pending write) | p50 8ms · p95 10ms | p50 45ms · p95 52ms | durability-only since 8.9.0 — cost no longer depends on history backlog or retention mode |
|
||||
|
||||
**The honest note on bulk writes:** `addMany` today commits each item as its own
|
||||
generation (the same durability as single-op `add`, serialized by the single-writer
|
||||
lock), so bulk-load cost is N × single-op cost. Batched chunk commits (one generation
|
||||
and one fsync window per chunk, as `removeMany` already does) are designed into the
|
||||
unified-commit work on the current roadmap. Until that ships, size bulk imports
|
||||
accordingly — 10k entities is minutes, not seconds, on filesystem storage.
|
||||
|
||||
## Open / close
|
||||
|
||||
| Operation | 1,000 entities | 10,000 entities | Notes |
|
||||
|---|---|---|---|
|
||||
| `open` (empty store) | ~560ms | ~190ms | includes embedder initialization |
|
||||
| `open` (warm, populated, clean shutdown) | 763ms | 4.9s | pure-JS vector index load dominates and grows with entity count; the native accelerator exists precisely to remove this |
|
||||
| `close` | bounded | bounded | auto-compaction pass is time-bounded (~5s max) since 8.9.0 |
|
||||
|
||||
A store that was NOT cleanly closed pays index rebuilds on top of the warm-open
|
||||
number (tens of seconds at 10k) — clean shutdown is worth engineering for.
|
||||
|
||||
## How these were produced
|
||||
|
||||
- **Hardware**: Intel Core i9-14900HX (32 threads), 62GB RAM, NVMe, Linux, Node v22.
|
||||
- **Backend**: `storage: { type: 'filesystem' }`, pure JS (no native providers).
|
||||
- **Embeddings**: deterministic stub for non-semantic ops (isolates engine cost);
|
||||
the real WASM embedder for the semantic row (that's what you'll run).
|
||||
- **Method**: p50/p95 over 50–200 samples per op against the built `dist/`;
|
||||
the measuring script ships in the repo history and re-runs per release.
|
||||
|
||||
Numbers on different hardware will differ; the *shape* (sub-2ms indexed reads,
|
||||
~160ms embedding-bound semantic queries, durability-priced writes) is the envelope
|
||||
you should hold your deployment against. If your measurements diverge from these
|
||||
shapes by an order of magnitude, something is wrong — file it.
|
||||
|
|
@ -204,8 +204,8 @@ await brain.add({ data: { name: 'Entity' }, type: NounType.Thing })
|
|||
### Basic Add Operation
|
||||
|
||||
```typescript
|
||||
import { Brainy } from '@soulcraftlabs/brainy'
|
||||
import { NounType } from '@soulcraftlabs/brainy/types'
|
||||
import { Brainy } from '@soulcraft/brainy'
|
||||
import { NounType } from '@soulcraft/brainy/types'
|
||||
|
||||
const brain = new Brainy()
|
||||
await brain.init()
|
||||
|
|
@ -428,7 +428,7 @@ await brain.relate({ ... }) // a crash here leaves the entity unlinked
|
|||
|
||||
```typescript
|
||||
import { describe, it, expect } from 'vitest'
|
||||
import { Brainy } from '@soulcraftlabs/brainy'
|
||||
import { Brainy } from '@soulcraft/brainy'
|
||||
|
||||
describe('Transaction Tests', () => {
|
||||
it('should rollback on failure', async () => {
|
||||
|
|
|
|||
|
|
@ -23,7 +23,7 @@ The Universal Display Augmentation is a powerful AI-powered system that automati
|
|||
### Basic Usage
|
||||
|
||||
```typescript
|
||||
import { Brainy } from '@soulcraftlabs/brainy'
|
||||
import { Brainy } from '@soulcraft/brainy'
|
||||
|
||||
const brainy = new Brainy()
|
||||
await brainy.init()
|
||||
|
|
|
|||
|
|
@ -71,9 +71,9 @@ Let's build a projection that organizes files by priority (high, medium, low):
|
|||
### Step 1: Create the Strategy Class
|
||||
|
||||
```typescript
|
||||
import { BaseProjectionStrategy } from '@soulcraftlabs/brainy/vfs/semantic'
|
||||
import { Brainy } from '@soulcraftlabs/brainy'
|
||||
import { VirtualFileSystem, VFSEntity } from '@soulcraftlabs/brainy/vfs'
|
||||
import { BaseProjectionStrategy } from '@soulcraft/brainy/vfs/semantic'
|
||||
import { Brainy } from '@soulcraft/brainy'
|
||||
import { VirtualFileSystem, VFSEntity } from '@soulcraft/brainy/vfs'
|
||||
|
||||
export class PriorityProjection extends BaseProjectionStrategy {
|
||||
readonly name = 'priority'
|
||||
|
|
@ -141,7 +141,7 @@ export class PriorityProjection extends BaseProjectionStrategy {
|
|||
### Step 2: Register the Strategy
|
||||
|
||||
```typescript
|
||||
import { Brainy } from '@soulcraftlabs/brainy'
|
||||
import { Brainy } from '@soulcraft/brainy'
|
||||
import { PriorityProjection } from './PriorityProjection'
|
||||
|
||||
const brain = new Brainy()
|
||||
|
|
@ -537,7 +537,7 @@ Use the projection's resolve cache:
|
|||
|
||||
```typescript
|
||||
import { describe, it, expect, beforeAll } from 'vitest'
|
||||
import { Brainy } from '@soulcraftlabs/brainy'
|
||||
import { Brainy } from '@soulcraft/brainy'
|
||||
import { PriorityProjection } from './PriorityProjection'
|
||||
|
||||
describe('PriorityProjection', () => {
|
||||
|
|
@ -714,7 +714,7 @@ async resolve(brain, vfs, value: string) {
|
|||
3. Use appropriate limits: Don't fetch more than needed
|
||||
|
||||
### Type errors
|
||||
1. Import correct types: `import { Brainy, VirtualFileSystem } from '@soulcraftlabs/brainy'`
|
||||
1. Import correct types: `import { Brainy, VirtualFileSystem } from '@soulcraft/brainy'`
|
||||
2. Use `as VFSEntity` when mapping results
|
||||
3. Check BaseProjectionStrategy import
|
||||
|
||||
|
|
|
|||
|
|
@ -14,11 +14,11 @@ A file explorer that:
|
|||
## ⚡ Step 1: Basic Setup (1 minute)
|
||||
|
||||
```bash
|
||||
npm install @soulcraftlabs/brainy
|
||||
npm install @soulcraft/brainy
|
||||
```
|
||||
|
||||
```typescript
|
||||
import { Brainy } from '@soulcraftlabs/brainy'
|
||||
import { Brainy } from '@soulcraft/brainy'
|
||||
|
||||
// ✅ CORRECT: Use filesystem storage for production
|
||||
const brain = new Brainy({
|
||||
|
|
@ -115,7 +115,7 @@ Here's a complete React component using the correct patterns:
|
|||
|
||||
```tsx
|
||||
import React, { useState, useEffect } from 'react'
|
||||
import { Brainy } from '@soulcraftlabs/brainy'
|
||||
import { Brainy } from '@soulcraft/brainy'
|
||||
|
||||
export function FileExplorer() {
|
||||
const [brain, setBrain] = useState(null)
|
||||
|
|
@ -288,8 +288,8 @@ Your file explorer is now working! Here's what to explore next:
|
|||
### "Module not found" errors
|
||||
```bash
|
||||
# Make sure you're using the right import
|
||||
npm ls @soulcraftlabs/brainy # Check version
|
||||
npm install @soulcraftlabs/brainy@latest # Update if needed
|
||||
npm ls @soulcraft/brainy # Check version
|
||||
npm install @soulcraft/brainy@latest # Update if needed
|
||||
```
|
||||
|
||||
### "VFS not initialized" errors
|
||||
|
|
|
|||
|
|
@ -24,7 +24,7 @@ Brainy VFS is a revolutionary virtual filesystem that runs on top of Brainy's ne
|
|||
## Quick Start
|
||||
|
||||
```javascript
|
||||
import { VirtualFileSystem } from '@soulcraftlabs/brainy/vfs'
|
||||
import { VirtualFileSystem } from '@soulcraft/brainy/vfs'
|
||||
|
||||
// Initialize the VFS
|
||||
const vfs = new VirtualFileSystem({
|
||||
|
|
@ -381,7 +381,7 @@ Brainy VFS fully leverages Brainy's revolutionary Triple Intelligence system:
|
|||
## Installation
|
||||
|
||||
```bash
|
||||
npm install @soulcraftlabs/brainy
|
||||
npm install @soulcraft/brainy
|
||||
```
|
||||
|
||||
## Requirements
|
||||
|
|
|
|||
|
|
@ -135,7 +135,7 @@ Mount VFS as a native filesystem on Linux/Mac/Windows.
|
|||
|
||||
```typescript
|
||||
// Planned (research phase)
|
||||
import { mountVFS } from '@soulcraftlabs/brainy/vfs/fuse'
|
||||
import { mountVFS } from '@soulcraft/brainy/vfs/fuse'
|
||||
|
||||
await mountVFS(vfs, {
|
||||
mountPoint: '/mnt/brainy',
|
||||
|
|
@ -160,7 +160,7 @@ These features would benefit from community contributions. If you're interested
|
|||
### Express.js Static Middleware
|
||||
```typescript
|
||||
// Wanted: Community contribution
|
||||
import { createStaticMiddleware } from '@soulcraftlabs/brainy/vfs/express'
|
||||
import { createStaticMiddleware } from '@soulcraft/brainy/vfs/express'
|
||||
|
||||
app.use('/files', createStaticMiddleware(vfs, {
|
||||
index: ['index.html', 'index.md'],
|
||||
|
|
@ -172,7 +172,7 @@ app.use('/files', createStaticMiddleware(vfs, {
|
|||
### VSCode Extension
|
||||
```typescript
|
||||
// Wanted: Community contribution
|
||||
import { VFSProvider } from '@soulcraftlabs/brainy/vfs/vscode'
|
||||
import { VFSProvider } from '@soulcraft/brainy/vfs/vscode'
|
||||
|
||||
const provider = new VFSProvider(vfs)
|
||||
vscode.workspace.registerFileSystemProvider('brainy', provider)
|
||||
|
|
|
|||
|
|
@ -327,7 +327,7 @@ console.log(id1 === id2 && id2 === id3) // true
|
|||
Create your own semantic dimensions:
|
||||
|
||||
```typescript
|
||||
import { BaseProjectionStrategy } from '@soulcraftlabs/brainy/vfs/semantic'
|
||||
import { BaseProjectionStrategy } from '@soulcraft/brainy/vfs/semantic'
|
||||
|
||||
class PriorityProjection extends BaseProjectionStrategy {
|
||||
readonly name = 'priority'
|
||||
|
|
|
|||
|
|
@ -7,7 +7,7 @@ Brainy's Virtual Filesystem (VFS) provides a POSIX-like filesystem interface tha
|
|||
## Quick Start
|
||||
|
||||
```typescript
|
||||
import { Brainy } from '@soulcraftlabs/brainy'
|
||||
import { Brainy } from '@soulcraft/brainy'
|
||||
|
||||
// Initialize Brainy
|
||||
const brain = new Brainy({
|
||||
|
|
@ -598,7 +598,7 @@ const user = await store.findById('users', 'user123')
|
|||
VFS uses standard POSIX-style errors:
|
||||
|
||||
```typescript
|
||||
import { VFSError, VFSErrorCode } from '@soulcraftlabs/brainy'
|
||||
import { VFSError, VFSErrorCode } from '@soulcraft/brainy'
|
||||
|
||||
try {
|
||||
await vfs.readFile('/nonexistent.txt')
|
||||
|
|
|
|||
|
|
@ -280,7 +280,7 @@ GitBridge provides Git import/export capabilities:
|
|||
#### GitBridge Usage
|
||||
```javascript
|
||||
// Import and instantiate GitBridge
|
||||
import { GitBridge } from '@soulcraftlabs/brainy'
|
||||
import { GitBridge } from '@soulcraft/brainy'
|
||||
const gitBridge = new GitBridge(vfs, brain)
|
||||
|
||||
// Export VFS to Git repository structure
|
||||
|
|
@ -452,7 +452,7 @@ This ordering prevents race conditions where file writes might fail because pare
|
|||
## Complete Example
|
||||
|
||||
```javascript
|
||||
import { Brainy } from '@soulcraftlabs/brainy'
|
||||
import { Brainy } from '@soulcraft/brainy'
|
||||
|
||||
async function vfsExample() {
|
||||
// Initialize
|
||||
|
|
|
|||
|
|
@ -196,5 +196,5 @@ await brain.relate({
|
|||
Always import and use the type enums:
|
||||
|
||||
```javascript
|
||||
import { NounType, VerbType } from '@soulcraftlabs/brainy'
|
||||
import { NounType, VerbType } from '@soulcraft/brainy'
|
||||
```
|
||||
|
|
@ -5,7 +5,7 @@
|
|||
The Brainy VFS is automatically initialized during `brain.init()`. No separate initialization needed!
|
||||
|
||||
```javascript
|
||||
import { Brainy } from '@soulcraftlabs/brainy'
|
||||
import { Brainy } from '@soulcraft/brainy'
|
||||
|
||||
// Create and initialize Brainy
|
||||
const brain = new Brainy({
|
||||
|
|
@ -71,7 +71,7 @@ VFS stores files as entities and relationships in the same graph as everything e
|
|||
## Complete Example
|
||||
|
||||
```javascript
|
||||
import { Brainy } from '@soulcraftlabs/brainy'
|
||||
import { Brainy } from '@soulcraft/brainy'
|
||||
|
||||
async function useVFS() {
|
||||
// Initialize Brainy
|
||||
|
|
@ -100,7 +100,7 @@ useVFS().catch(console.error)
|
|||
## TypeScript Usage
|
||||
|
||||
```typescript
|
||||
import { Brainy, VirtualFileSystem } from '@soulcraftlabs/brainy'
|
||||
import { Brainy, VirtualFileSystem } from '@soulcraft/brainy'
|
||||
|
||||
class FileManager {
|
||||
private brain: Brainy
|
||||
|
|
|
|||
|
|
@ -37,7 +37,7 @@ Brainy VFS provides safe, tree-aware methods that prevent these issues:
|
|||
### Method 1: Use `getDirectChildren()` (Recommended)
|
||||
|
||||
```typescript
|
||||
import { Brainy, VirtualFileSystem } from '@soulcraftlabs/brainy'
|
||||
import { Brainy, VirtualFileSystem } from '@soulcraft/brainy'
|
||||
|
||||
const brain = new Brainy()
|
||||
await brain.init()
|
||||
|
|
@ -97,7 +97,7 @@ Here's a complete example using React:
|
|||
|
||||
```tsx
|
||||
import React, { useState, useEffect } from 'react'
|
||||
import { VirtualFileSystem } from '@soulcraftlabs/brainy'
|
||||
import { VirtualFileSystem } from '@soulcraft/brainy'
|
||||
|
||||
interface FileNode {
|
||||
name: string
|
||||
|
|
@ -177,7 +177,7 @@ function TreeView({ node, onToggle, expanded }) {
|
|||
If you must build trees manually from flat lists, use the `VFSTreeUtils`:
|
||||
|
||||
```typescript
|
||||
import { VFSTreeUtils } from '@soulcraftlabs/brainy/vfs'
|
||||
import { VFSTreeUtils } from '@soulcraft/brainy/vfs'
|
||||
|
||||
// Get all entities somehow
|
||||
const allEntities = await vfs.getDescendants('/root')
|
||||
|
|
|
|||
|
|
@ -7,7 +7,7 @@
|
|||
* the Bluesky firehose with Brainy's distributed architecture
|
||||
*/
|
||||
|
||||
import { Brainy } from '@soulcraftlabs/brainy'
|
||||
import { Brainy } from '@soulcraft/brainy'
|
||||
import { WebSocket } from 'ws'
|
||||
|
||||
// =====================================================
|
||||
|
|
|
|||
|
|
@ -14,7 +14,7 @@
|
|||
* ts-node examples/monitor-cache-performance.ts
|
||||
*/
|
||||
|
||||
import { Brainy, NounType } from '@soulcraftlabs/brainy'
|
||||
import { Brainy, NounType } from '@soulcraft/brainy'
|
||||
|
||||
// ANSI color codes for pretty output
|
||||
const colors = {
|
||||
|
|
|
|||
|
|
@ -5,7 +5,7 @@ Connect Brainy to spreadsheets, BI tools, and external systems with zero configu
|
|||
## Quick Start
|
||||
|
||||
```typescript
|
||||
import { Brainy } from '@soulcraftlabs/brainy'
|
||||
import { Brainy } from '@soulcraft/brainy'
|
||||
|
||||
const brain = new Brainy({ integrations: true })
|
||||
await brain.init()
|
||||
|
|
@ -178,7 +178,7 @@ Webhooks include `X-Brainy-Signature` header with HMAC-SHA256 signature.
|
|||
### Minimal (in-memory):
|
||||
|
||||
```typescript
|
||||
import { Brainy } from '@soulcraftlabs/brainy'
|
||||
import { Brainy } from '@soulcraft/brainy'
|
||||
|
||||
const brain = new Brainy({ integrations: true })
|
||||
await brain.init()
|
||||
|
|
@ -194,7 +194,7 @@ console.log(brain.hub.getInstructions())
|
|||
|
||||
```typescript
|
||||
import express from 'express'
|
||||
import { Brainy } from '@soulcraftlabs/brainy'
|
||||
import { Brainy } from '@soulcraft/brainy'
|
||||
|
||||
const app = express()
|
||||
const brain = new Brainy({
|
||||
|
|
@ -232,7 +232,7 @@ app.listen(3000, () => {
|
|||
|
||||
```typescript
|
||||
import { Hono } from 'hono'
|
||||
import { Brainy } from '@soulcraftlabs/brainy'
|
||||
import { Brainy } from '@soulcraft/brainy'
|
||||
|
||||
const app = new Hono()
|
||||
|
||||
|
|
|
|||
|
|
@ -99,7 +99,7 @@ Add the `BRAINY_URL` script property in Apps Script settings.
|
|||
The simplest way to enable all integrations:
|
||||
|
||||
```javascript
|
||||
import { Brainy } from '@soulcraftlabs/brainy'
|
||||
import { Brainy } from '@soulcraft/brainy'
|
||||
|
||||
const brain = new Brainy({ integrations: true })
|
||||
await brain.init()
|
||||
|
|
@ -112,7 +112,7 @@ With Express:
|
|||
|
||||
```javascript
|
||||
import express from 'express'
|
||||
import { Brainy } from '@soulcraftlabs/brainy'
|
||||
import { Brainy } from '@soulcraft/brainy'
|
||||
|
||||
const app = express()
|
||||
const brain = new Brainy({ integrations: true })
|
||||
|
|
|
|||
8
package-lock.json
generated
8
package-lock.json
generated
|
|
@ -1,12 +1,12 @@
|
|||
{
|
||||
"name": "@soulcraftlabs/brainy",
|
||||
"version": "10.4.4",
|
||||
"name": "@soulcraft/brainy",
|
||||
"version": "8.3.0",
|
||||
"lockfileVersion": 3,
|
||||
"requires": true,
|
||||
"packages": {
|
||||
"": {
|
||||
"name": "@soulcraftlabs/brainy",
|
||||
"version": "10.4.4",
|
||||
"name": "@soulcraft/brainy",
|
||||
"version": "8.3.0",
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"@msgpack/msgpack": "^3.1.2",
|
||||
|
|
|
|||
14
package.json
14
package.json
|
|
@ -1,7 +1,6 @@
|
|||
{
|
||||
"name": "@soulcraftlabs/brainy",
|
||||
"version": "10.4.4",
|
||||
"brainyContract": 1,
|
||||
"name": "@soulcraft/brainy",
|
||||
"version": "8.3.0",
|
||||
"description": "Universal Knowledge Protocol™ - World's first Triple Intelligence database unifying vector, graph, and document search in one API. Stage 3 CANONICAL: 42 nouns × 127 verbs covering 96-97% of all human knowledge.",
|
||||
"main": "dist/index.js",
|
||||
"module": "dist/index.js",
|
||||
|
|
@ -127,16 +126,15 @@
|
|||
"license": "MIT",
|
||||
"private": false,
|
||||
"publishConfig": {
|
||||
"access": "public",
|
||||
"registry": "https://source.soulcraft.com/api/packages/soulcraftlabs/npm/"
|
||||
"access": "public"
|
||||
},
|
||||
"homepage": "https://source.soulcraft.com/soulcraftlabs/open-brainy",
|
||||
"homepage": "https://github.com/soulcraftlabs/brainy",
|
||||
"bugs": {
|
||||
"url": "https://source.soulcraft.com/soulcraftlabs/open-brainy/issues"
|
||||
"url": "https://github.com/soulcraftlabs/brainy/issues"
|
||||
},
|
||||
"repository": {
|
||||
"type": "git",
|
||||
"url": "git+https://source.soulcraft.com/soulcraftlabs/open-brainy.git"
|
||||
"url": "git+https://github.com/soulcraftlabs/brainy.git"
|
||||
},
|
||||
"files": [
|
||||
"dist/**/*.js",
|
||||
|
|
|
|||
|
|
@ -1,76 +0,0 @@
|
|||
{
|
||||
"product": "brainy",
|
||||
"entries": [
|
||||
{
|
||||
"version": "11.0.5",
|
||||
"date": "2026-09-02",
|
||||
"headline": "Graph-first finds in production, and opens that stop rescanning history",
|
||||
"items": [
|
||||
"find({ connected, where }) now walks the neighbours first and filters only those rows through a native door — correct at every page and O(neighbours), never the whole store.",
|
||||
"related() with a list of verb types returns every requested kind (a fast path had silently kept only the first).",
|
||||
"Deferred-embedding recovery resumes from a low-water mark instead of rescanning the whole generation log at every open — measured at two minutes on a large brain, now milliseconds."
|
||||
],
|
||||
"url": null,
|
||||
"thumb": null
|
||||
},
|
||||
{
|
||||
"version": "11.0.4",
|
||||
"date": "2026-09-01",
|
||||
"headline": "Closes in milliseconds, index rebuilds without the disk-sync storm",
|
||||
"items": [
|
||||
"close() no longer pays deferred compaction or waits out an in-flight rebuild — measured 8 ms against the 4-minute closes it replaces; deferred work resumes at the next open, in the background.",
|
||||
"The metadata index's rebuild syncs to disk per shard instead of per row, and the durability point moved to the publish step — the same guarantee, a fraction of the disk traffic.",
|
||||
"A new native filter door evaluates queries over exactly the candidate rows a graph walk found, never the whole store."
|
||||
],
|
||||
"url": null,
|
||||
"thumb": null
|
||||
},
|
||||
{
|
||||
"version": "11.0.3",
|
||||
"date": "2026-09-01",
|
||||
"headline": "The embedding upgrade ceremony runs on every brain",
|
||||
"items": [
|
||||
"A brain opened through the standard plugin now carries its embedding-model identity, so the full-precision upgrade ceremony can run on it.",
|
||||
"A one-fix release; nothing else changed."
|
||||
],
|
||||
"url": null,
|
||||
"thumb": null
|
||||
},
|
||||
{
|
||||
"version": "11.0.2",
|
||||
"date": "2026-08-31",
|
||||
"headline": "One embedding quality everywhere, 3–4× faster imports",
|
||||
"items": [
|
||||
"Every runtime embeds with the same full-precision model — search quality no longer depends on where you run.",
|
||||
"Bulk embedding measured 3.1–4.2× faster, and an online re-embed ceremony upgrades existing stores without downtime.",
|
||||
"The engine's change feed is documented, with the SSE/WebSocket fan-out pattern for realtime surfaces."
|
||||
],
|
||||
"url": null,
|
||||
"thumb": null
|
||||
},
|
||||
{
|
||||
"version": "11.0.1",
|
||||
"date": "2026-08-31",
|
||||
"headline": "Deletes inside transactions are safe",
|
||||
"items": [
|
||||
"Deleting relations inside a transact() no longer corrupts index bookkeeping.",
|
||||
"A store that deletes its last relation keeps serving instead of refusing."
|
||||
],
|
||||
"url": null,
|
||||
"thumb": null
|
||||
},
|
||||
{
|
||||
"version": "11.0.0",
|
||||
"date": "2026-08-28",
|
||||
"headline": "One install, one engine — Brainy",
|
||||
"items": [
|
||||
"The former two-package pair is one package: the native engine under the familiar API. One import is the whole install.",
|
||||
"A missing native build refuses loudly with its cures named; nothing falls back silently.",
|
||||
"Stores open in place — no migration."
|
||||
],
|
||||
"url": null,
|
||||
"thumb": null
|
||||
}
|
||||
],
|
||||
"history": "The version line continues from the 4.3.x native-engine releases; their record lives in the product repository's CHANGELOG.md."
|
||||
}
|
||||
|
|
@ -1,110 +0,0 @@
|
|||
{
|
||||
"product": "open-brainy",
|
||||
"entries": [
|
||||
{
|
||||
"version": "10.4.9",
|
||||
"date": "2026-09-02",
|
||||
"headline": "Graph-first finds, honest verb arrays, and opens that stop rescanning history",
|
||||
"items": [
|
||||
"find({ connected, where }) now walks the neighbours first and filters only those rows — correct at every page, and O(neighbours) instead of O(store).",
|
||||
"related() with a list of verb types (or sources, or targets) returns every requested kind — four fast paths silently kept only the first.",
|
||||
"Deferred-embedding recovery resumes from a low-water mark instead of rescanning the whole generation log at every open — measured at two minutes on a large brain, now milliseconds."
|
||||
],
|
||||
"url": "https://source.soulcraft.com/soulcraftlabs/open-brainy/releases/tag/v10.4.9",
|
||||
"thumb": null
|
||||
},
|
||||
{
|
||||
"version": "10.4.7",
|
||||
"date": "2026-09-01",
|
||||
"headline": "Count ledgers can no longer race themselves",
|
||||
"items": [
|
||||
"Concurrent count flushes coalesce into one writer with a trailing pass — parallel flushes can no longer corrupt a store's count ledger.",
|
||||
"Atomic writes carry a per-process sequence, so two processes' temp files can never collide."
|
||||
],
|
||||
"url": "https://source.soulcraft.com/soulcraftlabs/open-brainy/releases/tag/v10.4.7",
|
||||
"thumb": null
|
||||
},
|
||||
{
|
||||
"version": "10.4.6",
|
||||
"date": "2026-08-31",
|
||||
"headline": "Transactions cross the index seam safely",
|
||||
"items": [
|
||||
"Deleting relations inside a transact() no longer fails against the metadata index — operations take a JSON-safe view at the moment they execute.",
|
||||
"Fixes a class of transaction failures on stores with integer-mapped relation endpoints."
|
||||
],
|
||||
"url": "https://source.soulcraft.com/soulcraftlabs/open-brainy/releases/tag/v10.4.6",
|
||||
"thumb": null
|
||||
},
|
||||
{
|
||||
"version": "10.4.5",
|
||||
"date": "2026-08-31",
|
||||
"headline": "Recovery tells the truth, docs live at home",
|
||||
"items": [
|
||||
"A torn generation-log tail is a terminal verdict with a named cure — never an endless wait at open.",
|
||||
"A sealed segment declares only the generations it actually holds.",
|
||||
"The engine's documentation now publishes from its own repository."
|
||||
],
|
||||
"url": "https://source.soulcraft.com/soulcraftlabs/open-brainy/releases/tag/v10.4.5",
|
||||
"thumb": null
|
||||
},
|
||||
{
|
||||
"version": "10.4.4",
|
||||
"date": "2026-08-28",
|
||||
"headline": "Faster opens, quieter idle",
|
||||
"items": [
|
||||
"Opening a store discovers generations from directory names instead of walking the log, and answers \"any entities?\" with one directory read.",
|
||||
"The flush-request watch is event-driven; idle stores stop paying a polling heartbeat.",
|
||||
"A slow open now names the exact step it is in, so operators see what is being paid and why."
|
||||
],
|
||||
"url": "https://source.soulcraft.com/soulcraftlabs/open-brainy/releases/tag/v10.4.4",
|
||||
"thumb": null
|
||||
},
|
||||
{
|
||||
"version": "10.4.3",
|
||||
"date": "2026-08-27",
|
||||
"headline": "Open Brainy, under its own name",
|
||||
"items": [
|
||||
"The same engine as 10.4.2, now published as @soulcraftlabs/brainy — the MIT reference engine, on The Source.",
|
||||
"No code changes; your imports change once and everything else stays put."
|
||||
],
|
||||
"url": "https://source.soulcraft.com/soulcraftlabs/open-brainy/releases/tag/v10.4.3",
|
||||
"thumb": null
|
||||
},
|
||||
{
|
||||
"version": "10.4.2",
|
||||
"date": "2026-08-27",
|
||||
"headline": "Vectors that lie are refused, counts that drift are caught",
|
||||
"items": [
|
||||
"A zero-norm vector is not a vector: the index refuses them, rebuilds skip them, and a sanctioned unvector door removes them cleanly.",
|
||||
"The canonical count ledger derives from identity records and marks legacy-derived ledgers suspect at load.",
|
||||
"Plugin activation failures keep their original error as cause, so the real frame reaches your logs."
|
||||
],
|
||||
"url": "https://source.soulcraft.com/soulcraftlabs/open-brainy/releases/tag/v10.4.2",
|
||||
"thumb": null
|
||||
},
|
||||
{
|
||||
"version": "10.4.1",
|
||||
"date": "2026-08-26",
|
||||
"headline": "Writes that change nothing cost nothing",
|
||||
"items": [
|
||||
"The read gate is per index family, and a write carrying unchanged data never re-embeds.",
|
||||
"The vectored-row count joins the ledger, so vector coverage is a number you can read, not a guess."
|
||||
],
|
||||
"url": "https://source.soulcraft.com/soulcraftlabs/open-brainy/releases/tag/v10.4.1",
|
||||
"thumb": null
|
||||
},
|
||||
{
|
||||
"version": "10.4.0",
|
||||
"date": "2026-08-26",
|
||||
"headline": "Repair routing, the vector ledger, and honest empties",
|
||||
"items": [
|
||||
"Repairs route to the index that owns the damage, and the open gate closes the vector leg until coverage is proven.",
|
||||
"An empty string is real data, not a missing field.",
|
||||
"The metadata crossing never carries raw integer relation endpoints — a whole class of serialization faults closed."
|
||||
],
|
||||
"url": "https://source.soulcraft.com/soulcraftlabs/open-brainy/releases/tag/v10.4.0",
|
||||
"thumb": null
|
||||
}
|
||||
],
|
||||
"history": "Earlier releases are recorded in CHANGELOG.md in this repository."
|
||||
}
|
||||
|
|
@ -10,7 +10,6 @@ import { TransformerEmbedding } from '../src/utils/embedding.js'
|
|||
import * as fs from 'fs/promises'
|
||||
import * as path from 'path'
|
||||
import { fileURLToPath } from 'url'
|
||||
import { resolveDeterministicStamp } from './lib/deterministicStamp.js'
|
||||
|
||||
const __dirname = path.dirname(fileURLToPath(import.meta.url))
|
||||
|
||||
|
|
@ -98,22 +97,13 @@ async function buildEmbeddedPatterns() {
|
|||
// Convert to base64 for embedding in TypeScript
|
||||
const uint8 = new Uint8Array(buffer)
|
||||
const base64 = Buffer.from(uint8).toString('base64')
|
||||
|
||||
// Deterministic stamp: derived from the git commit time of this
|
||||
// generator's inputs, never from wall-clock time — two builds of the
|
||||
// same source tree must produce byte-identical output.
|
||||
const outputPath = path.join(__dirname, '..', 'src', 'neural', 'embeddedPatterns.ts')
|
||||
const generatedStamp = resolveDeterministicStamp(
|
||||
[path.join(__dirname, 'buildEmbeddedPatterns.ts'), libraryPath],
|
||||
outputPath
|
||||
)
|
||||
|
||||
|
||||
// Generate TypeScript file with everything embedded
|
||||
const tsContent = `/**
|
||||
* 🧠 BRAINY EMBEDDED PATTERNS
|
||||
*
|
||||
* AUTO-GENERATED - DO NOT EDIT
|
||||
* Generated: ${generatedStamp}
|
||||
* Generated: ${new Date().toISOString()}
|
||||
* Patterns: ${libraryData.patterns.length}
|
||||
* Coverage: 94-98% of all queries
|
||||
*
|
||||
|
|
@ -207,6 +197,7 @@ prodLog.info(\`🧠 Brainy Pattern Library loaded: \${EMBEDDED_PATTERNS.length}
|
|||
`
|
||||
|
||||
// Write the TypeScript file
|
||||
const outputPath = path.join(__dirname, '..', 'src', 'neural', 'embeddedPatterns.ts')
|
||||
await fs.writeFile(outputPath, tsContent)
|
||||
|
||||
// Report statistics
|
||||
|
|
|
|||
|
|
@ -11,7 +11,6 @@ import * as fs from 'fs/promises'
|
|||
import * as path from 'path'
|
||||
import { fileURLToPath } from 'url'
|
||||
import { NounType, VerbType } from '../src/types/graphTypes.js'
|
||||
import { resolveDeterministicStamp } from './lib/deterministicStamp.js'
|
||||
|
||||
const __dirname = path.dirname(fileURLToPath(import.meta.url))
|
||||
|
||||
|
|
@ -374,24 +373,12 @@ async function buildTypeEmbeddings() {
|
|||
const uint8 = new Uint8Array(buffer)
|
||||
const base64 = Buffer.from(uint8).toString('base64')
|
||||
|
||||
// Deterministic stamp: derived from the git commit time of this
|
||||
// generator's inputs, never from wall-clock time — two builds of the
|
||||
// same source tree must produce byte-identical output.
|
||||
const outputPath = path.join(__dirname, '..', 'src', 'neural', 'embeddedTypeEmbeddings.ts')
|
||||
const generatedStamp = resolveDeterministicStamp(
|
||||
[
|
||||
path.join(__dirname, 'buildTypeEmbeddings.ts'),
|
||||
path.join(__dirname, '..', 'src', 'types', 'graphTypes.ts')
|
||||
],
|
||||
outputPath
|
||||
)
|
||||
|
||||
// Generate TypeScript file
|
||||
const tsContent = `/**
|
||||
* 🧠 BRAINY EMBEDDED TYPE EMBEDDINGS
|
||||
*
|
||||
* AUTO-GENERATED - DO NOT EDIT
|
||||
* Generated: ${generatedStamp}
|
||||
* Generated: ${new Date().toISOString()}
|
||||
* Noun Types: ${nounTypes.length}
|
||||
* Verb Types: ${verbTypes.length}
|
||||
*
|
||||
|
|
@ -408,7 +395,7 @@ export const TYPE_METADATA = {
|
|||
verbTypes: ${verbTypes.length},
|
||||
totalTypes: ${totalTypes},
|
||||
embeddingDimensions: ${embeddingDim},
|
||||
generatedAt: "${generatedStamp}",
|
||||
generatedAt: "${new Date().toISOString()}",
|
||||
sizeBytes: {
|
||||
embeddings: ${buffer.byteLength},
|
||||
base64: ${base64.length}
|
||||
|
|
@ -507,6 +494,7 @@ prodLog.info(\`🧠 Brainy Type Embeddings loaded: \${TYPE_METADATA.nounTypes} n
|
|||
`
|
||||
|
||||
// Write the TypeScript file
|
||||
const outputPath = path.join(__dirname, '..', 'src', 'neural', 'embeddedTypeEmbeddings.ts')
|
||||
await fs.writeFile(outputPath, tsContent)
|
||||
|
||||
// Report statistics
|
||||
|
|
|
|||
|
|
@ -1,128 +0,0 @@
|
|||
#!/usr/bin/env node
|
||||
/**
|
||||
* Emit this build's API-contract manifest to docs/api-contract.json.
|
||||
*
|
||||
* WHY IT IS GENERATED, NOT WRITTEN: a hand-kept list of doors drifts from the
|
||||
* code the first time somebody adds one. This reads the surface the build
|
||||
* actually exposes — the prototype's own methods and accessors, the exported
|
||||
* error classes, the `where` operator sets, the field-addressing vocabulary,
|
||||
* the health verdicts — so a diff between two engines' manifests is a diff
|
||||
* between two engines, never between two authors.
|
||||
*
|
||||
* Requirement marking (required / optional per door) is NOT derivable from the
|
||||
* surface — it is a commitment, recorded with the contract's owner rather than
|
||||
* here. This manifest carries the surface; the promise lives with the contract.
|
||||
*
|
||||
* Usage: node scripts/emit-contract-manifest.mjs [--check]
|
||||
* --check exits non-zero when the committed manifest is stale.
|
||||
*/
|
||||
|
||||
import { writeFileSync, readFileSync, existsSync } from 'node:fs'
|
||||
import { join, dirname } from 'node:path'
|
||||
import { fileURLToPath } from 'node:url'
|
||||
|
||||
const ROOT = join(dirname(fileURLToPath(import.meta.url)), '..')
|
||||
const OUT = join(ROOT, 'docs', 'api-contract.json')
|
||||
|
||||
const { Brainy } = await import(join(ROOT, 'dist', 'brainy.js'))
|
||||
const errorsModule = await import(join(ROOT, 'dist', 'errors', 'brainyError.js'))
|
||||
const versionModule = await import(join(ROOT, 'dist', 'utils', 'version.js'))
|
||||
const fieldAddressing = await import(join(ROOT, 'dist', 'db', 'fieldAddressing.js'))
|
||||
|
||||
/** Every own method and accessor on the class's prototype, minus the private ones. */
|
||||
function surfaceOf(ctor) {
|
||||
const doors = []
|
||||
for (const name of Object.getOwnPropertyNames(ctor.prototype)) {
|
||||
if (name === 'constructor' || name.startsWith('_')) continue
|
||||
const descriptor = Object.getOwnPropertyDescriptor(ctor.prototype, name)
|
||||
if (!descriptor) continue
|
||||
if (typeof descriptor.value === 'function') {
|
||||
doors.push({ name, kind: 'method', arity: descriptor.value.length })
|
||||
} else if (descriptor.get) {
|
||||
doors.push({ name, kind: 'accessor' })
|
||||
}
|
||||
}
|
||||
return doors.sort((a, b) => a.name.localeCompare(b.name))
|
||||
}
|
||||
|
||||
const errors = Object.entries(errorsModule)
|
||||
.filter(([name, value]) => typeof value === 'function' && /Error$/.test(name))
|
||||
.map(([name]) => name)
|
||||
.sort()
|
||||
|
||||
// The operator sets, read from the engine's own refusal message so the
|
||||
// manifest can never disagree with the validator.
|
||||
const filterSource = readFileSync(join(ROOT, 'src', 'utils', 'metadataFilter.ts'), 'utf-8')
|
||||
const acceptedMatch = filterSource.match(/const VALUE_OPERATORS = new Set<string>\(\[([\s\S]*?)\]\)/)
|
||||
if (!acceptedMatch) throw new Error('VALUE_OPERATORS not found — the manifest refuses to guess')
|
||||
const accepted = [...acceptedMatch[1].matchAll(/'([^']+)'/g)].map((m) => m[1]).sort()
|
||||
|
||||
const indexSource = readFileSync(join(ROOT, 'src', 'utils', 'metadataIndex.ts'), 'utf-8')
|
||||
const refusedByIndex = ['endsWith', 'length', 'matches', 'startsWith'].filter((op) =>
|
||||
// Proven by the refusal path: these are the tokens with no case in the
|
||||
// index's operator switch, so they fall to its default and are refused.
|
||||
!new RegExp(`case '${op}':`).test(indexSource)
|
||||
)
|
||||
const servedOnIndex = accepted.filter((op) => !refusedByIndex.includes(op))
|
||||
|
||||
const manifest = {
|
||||
contractVersion: versionModule.contractVersion(),
|
||||
engine: '@soulcraftlabs/brainy',
|
||||
compatibility: {
|
||||
minor:
|
||||
'additive — a new optional door, a new served operator, a new error class; every existing implementation still conforms',
|
||||
major:
|
||||
'breaking — a door removed, an answer narrowed, an ordering law changed, an optional door promoted to required, or an operator moved from served to refused'
|
||||
},
|
||||
doors: surfaceOf(Brainy),
|
||||
errors,
|
||||
operators: {
|
||||
accepted,
|
||||
servedOnIndexPath: servedOnIndex,
|
||||
refusedByIndexPath: refusedByIndex,
|
||||
combinators: ['allOf', 'anyOf', 'not']
|
||||
},
|
||||
fieldAddressing: {
|
||||
systemKeyPrefix: 'system.',
|
||||
systemEntityScalars: [...(fieldAddressing.SYSTEM_ENTITY_SCALARS ?? [])].sort(),
|
||||
systemRelationScalars: [...(fieldAddressing.SYSTEM_RELATION_SCALARS ?? [])].sort(),
|
||||
plumbingFields: [...(fieldAddressing.PLUMBING_FIELDS ?? [])].sort()
|
||||
},
|
||||
health: {
|
||||
verdicts: ['pass', 'warn', 'fail'],
|
||||
healKinds: ['none', 'repair', 'rebuild'],
|
||||
servingWithholdingInvariants: [
|
||||
'index-initialized',
|
||||
'durable-state-present',
|
||||
'manifest-residency',
|
||||
'replay-clean',
|
||||
'strand-latch'
|
||||
]
|
||||
}
|
||||
}
|
||||
|
||||
const rendered = `${JSON.stringify(manifest, null, 2)}\n`
|
||||
|
||||
if (process.argv.includes('--check')) {
|
||||
if (!existsSync(OUT)) {
|
||||
console.error(`docs/api-contract.json is missing — run: node scripts/emit-contract-manifest.mjs`)
|
||||
process.exit(1)
|
||||
}
|
||||
if (readFileSync(OUT, 'utf-8') !== rendered) {
|
||||
console.error(
|
||||
`docs/api-contract.json is STALE — the public surface changed. Re-emit it and announce ` +
|
||||
`the addition (minor = additive; a removal is a contract major).`
|
||||
)
|
||||
process.exit(1)
|
||||
}
|
||||
console.log(`docs/api-contract.json is current (${manifest.doors.length} doors, contract ${manifest.contractVersion}).`)
|
||||
process.exit(0)
|
||||
}
|
||||
|
||||
writeFileSync(OUT, rendered)
|
||||
console.log(
|
||||
`Wrote docs/api-contract.json — contract ${manifest.contractVersion}, ` +
|
||||
`${manifest.doors.length} doors, ${manifest.errors.length} error classes, ` +
|
||||
`${manifest.operators.accepted.length} operators ` +
|
||||
`(${manifest.operators.refusedByIndexPath.length} refused by the index path).`
|
||||
)
|
||||
|
|
@ -1,85 +0,0 @@
|
|||
# Gate Guards
|
||||
|
||||
Two standalone scripts that stand between a test/build gate and a false
|
||||
verdict: one refuses to let the gate start on a noisy machine, the other
|
||||
refuses to let a truncated or crashed vitest run be read as green.
|
||||
|
||||
## Why these exist
|
||||
|
||||
Both guards exist because of the 2026-08-13 lost-day ledger: a gate ran on
|
||||
a machine under load, and separately a vitest worker pool died mid-suite
|
||||
while still printing a plausible-looking summary line, and in both cases
|
||||
the bad result was trusted and acted on for the better part of a day before
|
||||
anyone noticed. Neither failure mode announces itself — a loaded machine
|
||||
still finishes and reports numbers, and a truncated test run still prints a
|
||||
`Test Files` / `Tests` line — so both guards check the evidence explicitly
|
||||
rather than trusting that a gate finishing means the gate was valid.
|
||||
|
||||
## gate-preflight.sh
|
||||
|
||||
Run before any gate lane starts. Exits 1 the moment the machine isn't
|
||||
gate-clean, with one `FATAL:` line per violation naming the exact offender
|
||||
(the pid and command, the path, the measured value). Prints one `OK:` line
|
||||
per check that passes. `WARNING:` lines mark checks that were skipped, not
|
||||
failures.
|
||||
|
||||
Checks:
|
||||
|
||||
| # | Check | Default threshold | Override |
|
||||
|---|-------|--------------------|----------|
|
||||
| a | 1-minute load average | `nproc / 2` | `GATE_MAX_LOAD` |
|
||||
| b | any non-allowlisted process over 50% of one core | 50% | `GATE_ALLOW_REGEX` (extra pattern matched against the process's args) |
|
||||
| c | cpu0 scaling governor must be `performance` | — | none (warns and skips if the sysfs path is absent) |
|
||||
| d | free space on `/` and `/tmp` | 10G each | `GATE_SKIP_DISK_CHECK=1` to skip entirely |
|
||||
|
||||
The allowlist for check (b) is always: this script's own process tree
|
||||
(its ancestors and its direct child processes), `sshd`, `systemd`, and
|
||||
kernel threads (recognizable by args wrapped in brackets, e.g.
|
||||
`[kworker/0:1]`). `GATE_ALLOW_REGEX` extends it — it does not replace it.
|
||||
|
||||
## vitest-verdict-check.sh
|
||||
|
||||
Run after every vitest lane, against that lane's captured log. Fails
|
||||
loudly, quoting the exact line or string that tripped it, when the log's
|
||||
own summary can't be trusted:
|
||||
|
||||
- no `Test Files` (or, in `--count-tests` mode, `Tests`) summary line is
|
||||
present at all
|
||||
- the parenthesized total in that line doesn't match what was expected
|
||||
- fewer files/tests are accounted for (passed + failed + skipped) than the
|
||||
total claims — a truncated run
|
||||
- the log contains `Unhandled Error` or `Timeout calling` anywhere — a dead
|
||||
worker pool, regardless of what the summary line claims
|
||||
|
||||
```
|
||||
vitest-verdict-check.sh <log-file> <expected-file-count>
|
||||
vitest-verdict-check.sh --count-tests <log-file> <minimum-test-count>
|
||||
```
|
||||
|
||||
The first form checks `Test Files` for an exact match. The second checks
|
||||
`Tests` for a minimum (a floor, not an exact count, since the total number
|
||||
of individual tests moves more often than the number of test files).
|
||||
|
||||
## Wiring into a CI lane
|
||||
|
||||
```sh
|
||||
# Before any lane that will report a verdict:
|
||||
scripts/gate/gate-preflight.sh || exit 1
|
||||
|
||||
# Run the suite, capturing its output:
|
||||
npx vitest run tests/unit 2>&1 | tee /tmp/unit.log
|
||||
|
||||
# After every vitest lane, check the log against the actual file count:
|
||||
EXPECTED_FILES=$(ls tests/unit/**/*.test.ts | wc -l)
|
||||
scripts/gate/vitest-verdict-check.sh /tmp/unit.log "$EXPECTED_FILES" || exit 1
|
||||
```
|
||||
|
||||
## Exit-code contract
|
||||
|
||||
| Script | Exit 0 | Exit 1 |
|
||||
|--------|--------|--------|
|
||||
| `gate-preflight.sh` | machine is gate-clean | one or more `FATAL:` violations printed |
|
||||
| `vitest-verdict-check.sh` | log's summary is trustworthy and matches | usage error, missing/unreadable log, or one or more `FATAL:` violations printed |
|
||||
|
||||
Non-zero from either script means: do not trust the gate that was about to
|
||||
run, or the result of the one that just ran.
|
||||
|
|
@ -1,206 +0,0 @@
|
|||
#!/bin/bash
|
||||
set -euo pipefail
|
||||
|
||||
# Brainy Gate Preflight
|
||||
# Refuses to let a test/build gate run on a machine that isn't clean enough
|
||||
# to trust the numbers it produces. See scripts/gate/README.md for why (the
|
||||
# 2026-08-13 lost-day ledger).
|
||||
#
|
||||
# Checks: 1-minute load average, any non-allowlisted process pinning a core,
|
||||
# the cpu0 scaling governor, and free space on / and /tmp.
|
||||
#
|
||||
# Exit 0 and print one OK line per passing check when the machine is clean.
|
||||
# Exit 1 and print one FATAL line per violation, naming the offender, when
|
||||
# it is not.
|
||||
#
|
||||
# Known trap: a helper function whose last executed statement is a `while`
|
||||
# (or any command whose own exit status happens to be nonzero) hands that
|
||||
# status back as the function's return value. Called as a plain statement,
|
||||
# that silently kills this script under `set -e`. Every helper below ends
|
||||
# on an explicit `return 0` as its own statement, never on a loop or test.
|
||||
#
|
||||
# The same failure mode hides in plainer-looking lines too: `var=$(cmd)` is
|
||||
# a bare assignment, so `set -e` DOES treat a nonzero `cmd` (or, under
|
||||
# `pipefail`, a nonzero stage anywhere in `cmd`'s pipeline) as a failure of
|
||||
# that statement and kills the script right there — even mid-loop, even
|
||||
# when the "failure" is routine (a process that exited before a second
|
||||
# lookup, a path that doesn't exist). Every such assignment below is paired
|
||||
# with an explicit `|| var=""` fallback so a routine miss degrades to an
|
||||
# empty value instead of an exit.
|
||||
|
||||
VIOLATIONS=0
|
||||
ANCESTOR_PIDS=""
|
||||
|
||||
fatal() {
|
||||
echo "FATAL: $1"
|
||||
VIOLATIONS=$((VIOLATIONS + 1))
|
||||
}
|
||||
|
||||
ok() {
|
||||
echo "OK: $1"
|
||||
}
|
||||
|
||||
# Walks this process's parent chain up to pid 1, then takes one snapshot of
|
||||
# its direct children (the ps/read pipeline in check_processes), and
|
||||
# records both in ANCESTOR_PIDS — so the process-scan below can recognize
|
||||
# its own tree (the shell/terminal/session that launched it, plus its own
|
||||
# helper commands) instead of flagging it. Children are captured once, up
|
||||
# front, rather than re-queried per row later, so a helper command that has
|
||||
# already exited by the time it's looked up can't be mistaken for a miss.
|
||||
build_ancestor_pids() {
|
||||
local pid="$$"
|
||||
local ppid child
|
||||
ANCESTOR_PIDS=" $pid "
|
||||
while [ "$pid" != "1" ]; do
|
||||
ppid=$(ps -o ppid= -p "$pid" 2>/dev/null | tr -d ' ') || ppid=""
|
||||
if [ -z "$ppid" ]; then
|
||||
break
|
||||
fi
|
||||
ANCESTOR_PIDS="${ANCESTOR_PIDS}${ppid} "
|
||||
pid="$ppid"
|
||||
done
|
||||
|
||||
while IFS= read -r child; do
|
||||
[ -z "$child" ] && continue
|
||||
ANCESTOR_PIDS="${ANCESTOR_PIDS}${child} "
|
||||
done < <(ps --ppid "$$" -o pid= 2>/dev/null || true)
|
||||
|
||||
return 0
|
||||
}
|
||||
|
||||
# (a) 1-minute load average vs. threshold (default: nproc / 2).
|
||||
check_load() {
|
||||
local max_load="${GATE_MAX_LOAD:-}"
|
||||
if [ -z "$max_load" ]; then
|
||||
max_load=$(( $(nproc) / 2 ))
|
||||
if [ "$max_load" -lt 1 ]; then
|
||||
max_load=1
|
||||
fi
|
||||
fi
|
||||
|
||||
local load_1m
|
||||
load_1m=$(cut -d' ' -f1 /proc/loadavg)
|
||||
|
||||
if awk -v l="$load_1m" -v m="$max_load" 'BEGIN { exit !(l > m) }'; then
|
||||
fatal "1-minute load average ${load_1m} exceeds threshold ${max_load} (GATE_MAX_LOAD=${max_load})"
|
||||
else
|
||||
ok "1-minute load average ${load_1m} is within threshold ${max_load}"
|
||||
fi
|
||||
return 0
|
||||
}
|
||||
|
||||
# (b) any process outside the allowlist pinning more than half a core.
|
||||
# Parsed with `read` into named fields, not an awk/cut chain — a fixed-column
|
||||
# awk/cut split on `ps` output duplicated fields the first time this was
|
||||
# tried, because process args vary in word count. `read` with a fixed list
|
||||
# of variables dumps everything left over into the last one (args), which
|
||||
# handles that correctly.
|
||||
check_processes() {
|
||||
local max_pcpu=50
|
||||
local extra_regex="${GATE_ALLOW_REGEX:-}"
|
||||
local violation_found=0
|
||||
local line pcpu pid args pcpu_int
|
||||
|
||||
while IFS= read -r line; do
|
||||
[ -z "$line" ] && continue
|
||||
read -r pcpu pid args <<< "$line"
|
||||
|
||||
# Kernel threads report their comm in brackets, e.g. "[kworker/0:1]".
|
||||
case "$args" in
|
||||
\[*\]) continue ;;
|
||||
esac
|
||||
|
||||
# This script's own tree: its ancestors (shell, terminal, session) and
|
||||
# its direct children, both captured once by build_ancestor_pids.
|
||||
case " $ANCESTOR_PIDS " in
|
||||
*" $pid "*) continue ;;
|
||||
esac
|
||||
|
||||
case "$args" in
|
||||
*sshd*|*systemd*) continue ;;
|
||||
esac
|
||||
|
||||
if [ -n "$extra_regex" ] && [[ "$args" =~ $extra_regex ]]; then
|
||||
continue
|
||||
fi
|
||||
|
||||
pcpu_int="${pcpu%.*}"
|
||||
if [ -z "$pcpu_int" ]; then
|
||||
pcpu_int=0
|
||||
fi
|
||||
if [ "$pcpu_int" -gt "$max_pcpu" ]; then
|
||||
fatal "pid ${pid} ('${args}') is using ${pcpu}% of one core"
|
||||
violation_found=1
|
||||
fi
|
||||
done < <(ps -eo pcpu,pid,args --sort=-pcpu | tail -n +2)
|
||||
|
||||
if [ "$violation_found" -eq 0 ]; then
|
||||
ok "no process outside the allowlist exceeds ${max_pcpu}% of one core"
|
||||
fi
|
||||
return 0
|
||||
}
|
||||
|
||||
# (c) cpu0 scaling governor must be "performance". Skipped with a warning
|
||||
# (not a violation) when the sysfs path doesn't exist on this machine.
|
||||
check_governor() {
|
||||
local gov_path="/sys/devices/system/cpu/cpu0/cpufreq/scaling_governor"
|
||||
if [ ! -r "$gov_path" ]; then
|
||||
echo "WARNING: ${gov_path} not present; skipping governor check"
|
||||
return 0
|
||||
fi
|
||||
|
||||
local governor
|
||||
governor=$(cat "$gov_path" 2>/dev/null) || governor=""
|
||||
if [ "$governor" != "performance" ]; then
|
||||
fatal "cpu0 governor is '${governor}', not 'performance'"
|
||||
else
|
||||
ok "cpu0 governor is 'performance'"
|
||||
fi
|
||||
return 0
|
||||
}
|
||||
|
||||
# (d) free-space floors on / and /tmp (default 10G each). Skip entirely via
|
||||
# GATE_SKIP_DISK_CHECK=1.
|
||||
check_disk() {
|
||||
if [ "${GATE_SKIP_DISK_CHECK:-0}" = "1" ]; then
|
||||
echo "WARNING: disk free-space check skipped (GATE_SKIP_DISK_CHECK=1)"
|
||||
return 0
|
||||
fi
|
||||
|
||||
local floor_gb=10
|
||||
local floor_bytes=$((floor_gb * 1024 * 1024 * 1024))
|
||||
local path avail_bytes avail_gb
|
||||
|
||||
for path in / /tmp; do
|
||||
avail_bytes=$(df --output=avail -B1 "$path" 2>/dev/null | tail -n 1 | tr -d ' ') || avail_bytes=""
|
||||
if [ -z "$avail_bytes" ]; then
|
||||
echo "WARNING: could not determine free space on ${path}; skipping"
|
||||
continue
|
||||
fi
|
||||
if [ "$avail_bytes" -lt "$floor_bytes" ]; then
|
||||
avail_gb=$((avail_bytes / 1024 / 1024 / 1024))
|
||||
fatal "${path} has only ${avail_gb}G free, below the ${floor_gb}G floor"
|
||||
else
|
||||
ok "${path} has enough free space (floor ${floor_gb}G)"
|
||||
fi
|
||||
done
|
||||
return 0
|
||||
}
|
||||
|
||||
echo "Brainy gate preflight"
|
||||
echo "----------------------"
|
||||
|
||||
build_ancestor_pids
|
||||
check_load
|
||||
check_processes
|
||||
check_governor
|
||||
check_disk
|
||||
|
||||
echo "----------------------"
|
||||
if [ "$VIOLATIONS" -gt 0 ]; then
|
||||
echo "FATAL: gate preflight failed with ${VIOLATIONS} violation(s) — machine is not gate-clean"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
echo "gate preflight passed — machine is gate-clean"
|
||||
exit 0
|
||||
|
|
@ -1,158 +0,0 @@
|
|||
#!/bin/bash
|
||||
set -euo pipefail
|
||||
|
||||
# Brainy Vitest Verdict Check
|
||||
# Confirms a vitest run's own summary line is trustworthy before anything
|
||||
# downstream treats a green run as green. See scripts/gate/README.md for why
|
||||
# (the 2026-08-13 lost-day ledger).
|
||||
#
|
||||
# Usage:
|
||||
# vitest-verdict-check.sh <log-file> <expected-file-count>
|
||||
# vitest-verdict-check.sh --count-tests <log-file> <minimum-test-count>
|
||||
#
|
||||
# The first form checks the "Test Files" summary line's total against an
|
||||
# exact expected count. The second checks the "Tests" summary line's total
|
||||
# against a minimum. Both also fail on any sign the worker pool died
|
||||
# mid-run, whether or not a summary line still made it into the log.
|
||||
#
|
||||
# Exit 0 and print one OK line per passing check when the log is clean.
|
||||
# Exit 1 and print one FATAL line per violation, quoting the exact line or
|
||||
# string that tripped it, when it is not.
|
||||
#
|
||||
# Known trap (shared with gate-preflight.sh): every helper below ends on an
|
||||
# explicit `return 0` as its own statement, never on a loop or test, so a
|
||||
# helper's last command can never hand its own exit status back as the
|
||||
# function's under `set -e`. The same applies to `var=$(cmd)` assignments
|
||||
# mid-helper: a bare assignment IS checked by `set -e`, so a `grep` that
|
||||
# legitimately finds nothing (exit 1) would otherwise kill the script
|
||||
# instead of just leaving the variable empty — every such assignment below
|
||||
# is paired with an explicit `|| true` inside the substitution.
|
||||
|
||||
usage() {
|
||||
echo "Usage: $0 <log-file> <expected-file-count>"
|
||||
echo " $0 --count-tests <log-file> <minimum-test-count>"
|
||||
exit 1
|
||||
}
|
||||
|
||||
MODE="files"
|
||||
if [ "${1:-}" = "--count-tests" ]; then
|
||||
MODE="tests"
|
||||
shift
|
||||
fi
|
||||
|
||||
LOG_FILE="${1:-}"
|
||||
THRESHOLD="${2:-}"
|
||||
|
||||
if [ -z "$LOG_FILE" ] || [ -z "$THRESHOLD" ]; then
|
||||
usage
|
||||
fi
|
||||
|
||||
if [ ! -f "$LOG_FILE" ]; then
|
||||
echo "FATAL: log file '${LOG_FILE}' does not exist"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
if ! [[ "$THRESHOLD" =~ ^[0-9]+$ ]]; then
|
||||
echo "FATAL: threshold '${THRESHOLD}' is not a non-negative integer"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
VIOLATIONS=0
|
||||
|
||||
fatal() {
|
||||
echo "FATAL: $1"
|
||||
VIOLATIONS=$((VIOLATIONS + 1))
|
||||
}
|
||||
|
||||
ok() {
|
||||
echo "OK: $1"
|
||||
}
|
||||
|
||||
# Vitest colorizes its summary with ANSI escapes; strip them before parsing
|
||||
# anything, or the color codes end up embedded in the fields we grep for.
|
||||
CLEAN_LOG="$(sed 's/\x1b\[[0-9;]*m//g' "$LOG_FILE")"
|
||||
|
||||
# Worker-pool death: if either string appears, the run's own summary line —
|
||||
# even if present and even if its numbers look fine — cannot be trusted,
|
||||
# because the process died mid-suite and vitest's own accounting is what
|
||||
# died with it.
|
||||
check_worker_death() {
|
||||
if echo "$CLEAN_LOG" | grep -q "Unhandled Error"; then
|
||||
fatal "log contains 'Unhandled Error' — worker pool died mid-run"
|
||||
fi
|
||||
if echo "$CLEAN_LOG" | grep -q "Timeout calling"; then
|
||||
fatal "log contains 'Timeout calling' — worker pool died mid-run"
|
||||
fi
|
||||
return 0
|
||||
}
|
||||
|
||||
# Shared shape between the "Test Files" and "Tests" summary lines:
|
||||
# <label> <n> passed | <n> failed | <n> skipped (<total>)
|
||||
# `compare` is "eq" (total must equal threshold) or "min" (total must be at
|
||||
# least threshold).
|
||||
check_summary_line() {
|
||||
local label="$1"
|
||||
local threshold="$2"
|
||||
local compare="$3"
|
||||
local summary_line total accounted n
|
||||
|
||||
summary_line=$(echo "$CLEAN_LOG" | grep -E "^[[:space:]]*${label}[[:space:]]+" | tail -n 1 || true)
|
||||
|
||||
if [ -z "$summary_line" ]; then
|
||||
fatal "no '${label}' summary line found in ${LOG_FILE}"
|
||||
return 0
|
||||
fi
|
||||
|
||||
total=$(echo "$summary_line" | grep -oE '\([0-9]+\)' | tr -d '()' | tail -n 1 || true)
|
||||
if [ -z "$total" ]; then
|
||||
fatal "'${label}' summary line has no parenthesized total: \"${summary_line}\""
|
||||
return 0
|
||||
fi
|
||||
|
||||
if [ "$compare" = "eq" ]; then
|
||||
if [ "$total" -ne "$threshold" ]; then
|
||||
fatal "'${label}' total is ${total}, expected ${threshold}: \"${summary_line}\""
|
||||
else
|
||||
ok "'${label}' total matches expected ${threshold}"
|
||||
fi
|
||||
else
|
||||
if [ "$total" -lt "$threshold" ]; then
|
||||
fatal "'${label}' total is ${total}, below minimum ${threshold}: \"${summary_line}\""
|
||||
else
|
||||
ok "'${label}' total ${total} meets minimum ${threshold}"
|
||||
fi
|
||||
fi
|
||||
|
||||
accounted=0
|
||||
for n in $(echo "$summary_line" | grep -oE '[0-9]+ (passed|failed|skipped)' | grep -oE '^[0-9]+'); do
|
||||
accounted=$((accounted + n))
|
||||
done
|
||||
|
||||
if [ "$accounted" -lt "$total" ]; then
|
||||
fatal "'${label}' line accounts for only ${accounted} of ${total} — truncated run: \"${summary_line}\""
|
||||
else
|
||||
ok "'${label}' line accounts for all ${total}"
|
||||
fi
|
||||
|
||||
return 0
|
||||
}
|
||||
|
||||
echo "Brainy vitest verdict check: ${LOG_FILE}"
|
||||
echo "----------------------"
|
||||
|
||||
check_worker_death
|
||||
|
||||
if [ "$MODE" = "files" ]; then
|
||||
check_summary_line "Test Files" "$THRESHOLD" "eq"
|
||||
else
|
||||
check_summary_line "Tests" "$THRESHOLD" "min"
|
||||
fi
|
||||
|
||||
echo "----------------------"
|
||||
if [ "$VIOLATIONS" -gt 0 ]; then
|
||||
echo "FATAL: vitest verdict check failed with ${VIOLATIONS} violation(s) for ${LOG_FILE}"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
echo "vitest verdict check passed for ${LOG_FILE}"
|
||||
exit 0
|
||||
|
|
@ -1,118 +0,0 @@
|
|||
/**
|
||||
* Deterministic generation-stamp resolution for Brainy's build-time code
|
||||
* generators.
|
||||
*
|
||||
* Two builds of the same source tree must produce byte-identical output.
|
||||
* A wall-clock stamp (`new Date()`) breaks that guarantee, so every
|
||||
* generator that writes a "Generated:" header or a `generatedAt` field
|
||||
* into its output must resolve the stamp through this module instead.
|
||||
*
|
||||
* Resolution order:
|
||||
* 1. The newest git commit timestamp among the generator's input files
|
||||
* (the generator script itself always counts as an input).
|
||||
* 2. If git metadata is unavailable (for example, building from a
|
||||
* published npm tarball with no `.git` directory), the stamp already
|
||||
* recorded in the previously generated output file.
|
||||
* 3. If neither is available, the fixed epoch string
|
||||
* `1970-01-01T00:00:00.000Z`.
|
||||
*
|
||||
* Every fallback logs a line to stderr — deterministic degradation is
|
||||
* loud, never a silent divergence.
|
||||
*/
|
||||
|
||||
import { execFileSync } from 'child_process'
|
||||
import * as fs from 'fs'
|
||||
|
||||
const EPOCH_STAMP = '1970-01-01T00:00:00.000Z'
|
||||
const STAMP_PATTERN = /\*\s*Generated:\s*(\S+)/
|
||||
|
||||
/**
|
||||
* Resolve the deterministic stamp for a generator run.
|
||||
*
|
||||
* @param inputPaths Absolute paths to every file whose content determines
|
||||
* the generator's output, including the generator script itself.
|
||||
* @param previousOutputPath Absolute path to the previously generated
|
||||
* file, used for the existing-stamp fallback when git is unavailable.
|
||||
* @returns An ISO-8601 timestamp string that is deterministic for a given
|
||||
* source tree.
|
||||
*/
|
||||
export function resolveDeterministicStamp(
|
||||
inputPaths: string[],
|
||||
previousOutputPath: string
|
||||
): string {
|
||||
const gitStamp = newestGitCommitTimestamp(inputPaths)
|
||||
if (gitStamp) {
|
||||
return gitStamp
|
||||
}
|
||||
|
||||
const existingStamp = readExistingStamp(previousOutputPath)
|
||||
if (existingStamp) {
|
||||
process.stderr.write(
|
||||
`[deterministic-stamp] no git commit history found for generator inputs; ` +
|
||||
`reusing existing stamp from ${previousOutputPath}: ${existingStamp}\n`
|
||||
)
|
||||
return existingStamp
|
||||
}
|
||||
|
||||
process.stderr.write(
|
||||
`[deterministic-stamp] no git commit history and no previous output at ` +
|
||||
`${previousOutputPath}; falling back to fixed epoch stamp ${EPOCH_STAMP}\n`
|
||||
)
|
||||
return EPOCH_STAMP
|
||||
}
|
||||
|
||||
/**
|
||||
* Find the newest git commit timestamp among the given input paths.
|
||||
* Returns null if git is unavailable, the tree is not a git repository,
|
||||
* or none of the inputs have any commit history yet.
|
||||
*/
|
||||
function newestGitCommitTimestamp(inputPaths: string[]): string | null {
|
||||
let newest: string | null = null
|
||||
|
||||
for (const inputPath of inputPaths) {
|
||||
if (!fs.existsSync(inputPath)) {
|
||||
continue
|
||||
}
|
||||
|
||||
let out: string
|
||||
try {
|
||||
out = execFileSync(
|
||||
'git',
|
||||
['log', '-1', '--format=%cI', '--', inputPath],
|
||||
{ stdio: ['ignore', 'pipe', 'ignore'] }
|
||||
)
|
||||
.toString()
|
||||
.trim()
|
||||
} catch {
|
||||
// git missing, not a repository, or no permissions — handled by the
|
||||
// caller's fallback chain.
|
||||
continue
|
||||
}
|
||||
|
||||
if (!out) {
|
||||
// Path exists but has no commit history yet (e.g. newly created,
|
||||
// uncommitted file).
|
||||
continue
|
||||
}
|
||||
|
||||
if (!newest || new Date(out).getTime() > new Date(newest).getTime()) {
|
||||
newest = out
|
||||
}
|
||||
}
|
||||
|
||||
return newest
|
||||
}
|
||||
|
||||
/**
|
||||
* Parse the `* Generated: <ISO timestamp>` header out of a previously
|
||||
* generated file, if one exists.
|
||||
*/
|
||||
function readExistingStamp(outputPath: string): string | null {
|
||||
if (!fs.existsSync(outputPath)) {
|
||||
return null
|
||||
}
|
||||
|
||||
const content = fs.readFileSync(outputPath, 'utf-8')
|
||||
const match = content.match(STAMP_PATTERN)
|
||||
return match ? match[1] : null
|
||||
}
|
||||
|
|
@ -1,116 +0,0 @@
|
|||
#!/usr/bin/env node
|
||||
/**
|
||||
* @module scripts/push-docs
|
||||
* @description Push this repo's PUBLIC docs to the soulcraft.com docs ingest
|
||||
* door after an npm publish (VENUE-DOCS-RELEASE-PUSH — retires the old
|
||||
* build-time docs sync).
|
||||
*
|
||||
* Contract (mirrors the reference implementation on the serving side):
|
||||
* POST {base}/api/docs/ingest
|
||||
* headers: x-service-secret: $DOCS_INGEST_SECRET, Content-Type: application/json
|
||||
* body: { docs: [{ slug, title, markdown, nav: { order, section } }] }
|
||||
* batches of 10, idempotent per slug.
|
||||
*
|
||||
* A doc is public iff its frontmatter has `public: true` AND a `slug`. The
|
||||
* frontmatter is stripped; `category` → nav.section, `order` → nav.order.
|
||||
*
|
||||
* Deliberately NOT pushed: the combined /docs landing index. It spans BOTH
|
||||
* engine corpora (this repo's and the native accelerator's), so a per-repo
|
||||
* push would clobber the union — the index is authored on the serving side.
|
||||
*
|
||||
* Env: DOCS_INGEST_SECRET (required), DOCS_INGEST_BASE (default
|
||||
* https://soulcraft.com). Exits 0 with a LOUD warning when the secret is
|
||||
* absent (the npm publish has already happened; the serving side runs its
|
||||
* interim sync on request) and exits 1 when a push actually fails — the docs
|
||||
* site would silently trail npm otherwise, and that must be visible.
|
||||
*/
|
||||
import * as fs from 'node:fs'
|
||||
import * as path from 'node:path'
|
||||
|
||||
const BASE = (process.env.DOCS_INGEST_BASE || 'https://soulcraft.com').replace(/\/+$/, '')
|
||||
const SECRET = process.env.DOCS_INGEST_SECRET
|
||||
const DOCS_DIR = path.join(path.dirname(new URL(import.meta.url).pathname), '..', 'docs')
|
||||
const BATCH = 10
|
||||
|
||||
if (!SECRET) {
|
||||
console.warn(
|
||||
'⚠️ DOCS PUSH SKIPPED: DOCS_INGEST_SECRET is not set.\n' +
|
||||
' soulcraft.com/docs now TRAILS this npm release until docs are pushed.\n' +
|
||||
' Either export DOCS_INGEST_SECRET and re-run `node scripts/push-docs.js`,\n' +
|
||||
' or ping venue on VENUE-DOCS-RELEASE-PUSH for the interim sync.'
|
||||
)
|
||||
process.exit(0)
|
||||
}
|
||||
|
||||
/** Minimal frontmatter split — returns [meta, body] or [null, raw]. */
|
||||
function parseFrontmatter(raw) {
|
||||
const m = raw.match(/^---\n([\s\S]*?)\n---\n([\s\S]*)$/)
|
||||
if (!m) return [null, raw]
|
||||
const meta = {}
|
||||
for (const line of m[1].split('\n')) {
|
||||
const kv = line.match(/^(\w[\w-]*):\s*(.*)$/)
|
||||
if (kv) meta[kv[1]] = kv[2].trim().replace(/^["']|["']$/g, '')
|
||||
}
|
||||
return [meta, m[2]]
|
||||
}
|
||||
|
||||
const docs = []
|
||||
;(function walk(dir) {
|
||||
for (const entry of fs.readdirSync(dir, { withFileTypes: true })) {
|
||||
const full = path.join(dir, entry.name)
|
||||
if (entry.isDirectory()) walk(full)
|
||||
else if (entry.name.endsWith('.md')) {
|
||||
const [meta, body] = parseFrontmatter(fs.readFileSync(full, 'utf-8'))
|
||||
if (!meta || meta.public !== 'true' || !meta.slug) continue
|
||||
docs.push({
|
||||
slug: meta.slug,
|
||||
title: meta.title || meta.slug,
|
||||
markdown: body.trim(),
|
||||
nav: {
|
||||
order: Number.parseInt(meta.order || '99', 10) || 99,
|
||||
section: meta.category || 'guides'
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
})(DOCS_DIR)
|
||||
|
||||
if (docs.length === 0) {
|
||||
console.error('❌ DOCS PUSH FAILED: zero public docs collected — refusing to push an empty corpus.')
|
||||
process.exit(1)
|
||||
}
|
||||
docs.sort((a, b) => a.slug.localeCompare(b.slug))
|
||||
console.log(`Pushing ${docs.length} public docs to ${BASE}/api/docs/ingest …`)
|
||||
|
||||
let failed = false
|
||||
for (let i = 0; i < docs.length; i += BATCH) {
|
||||
const batch = docs.slice(i, i + BATCH)
|
||||
try {
|
||||
const res = await fetch(`${BASE}/api/docs/ingest`, {
|
||||
method: 'POST',
|
||||
headers: {
|
||||
'x-service-secret': SECRET,
|
||||
'Content-Type': 'application/json',
|
||||
'User-Agent': 'brainy-docs-push/1.0'
|
||||
},
|
||||
body: JSON.stringify({ docs: batch }),
|
||||
signal: AbortSignal.timeout(120_000)
|
||||
})
|
||||
if (!res.ok) {
|
||||
throw new Error(`HTTP ${res.status}: ${(await res.text()).slice(0, 300)}`)
|
||||
}
|
||||
console.log(` batch ${i / BATCH + 1}: ${batch.map((d) => d.slug).join(', ')} → ok`)
|
||||
} catch (err) {
|
||||
failed = true
|
||||
console.error(` batch ${i / BATCH + 1} FAILED: ${err instanceof Error ? err.message : err}`)
|
||||
}
|
||||
}
|
||||
|
||||
if (failed) {
|
||||
console.error(
|
||||
'❌ DOCS PUSH INCOMPLETE — soulcraft.com/docs may trail npm. ' +
|
||||
'Re-run `node scripts/push-docs.js` or ping venue on VENUE-DOCS-RELEASE-PUSH.'
|
||||
)
|
||||
process.exit(1)
|
||||
}
|
||||
console.log('✅ Docs pushed.')
|
||||
|
|
@ -15,12 +15,6 @@ NC='\033[0m' # No Color
|
|||
RELEASE_TYPE="${1:-patch}" # patch, minor, or major
|
||||
SKIP_TESTS=false
|
||||
DRY_RUN=false
|
||||
# --source-only is now a no-op: The Source is the one registry, so every
|
||||
# release already ships Source-only — tag, CI's publish to The Source, the
|
||||
# release page, and the docs push, with no separate storefront leg to skip.
|
||||
# The flag is still accepted (for backward-compatible invocations) and just
|
||||
# prints a notice; it no longer changes behavior.
|
||||
SOURCE_ONLY=false
|
||||
|
||||
for arg in "$@"; do
|
||||
case $arg in
|
||||
|
|
@ -30,9 +24,6 @@ for arg in "$@"; do
|
|||
--dry-run)
|
||||
DRY_RUN=true
|
||||
;;
|
||||
--source-only)
|
||||
SOURCE_ONLY=true
|
||||
;;
|
||||
esac
|
||||
done
|
||||
|
||||
|
|
@ -109,7 +100,7 @@ else
|
|||
;;
|
||||
*)
|
||||
echo -e "${RED}❌ Invalid release type: ${RELEASE_TYPE}${NC}"
|
||||
echo "Usage: ./scripts/release.sh [patch|minor|major|<explicit-version>] [--dry-run] [--source-only (no-op; The Source is the one registry)]"
|
||||
echo "Usage: ./scripts/release.sh [patch|minor|major|<explicit-version>] [--dry-run]"
|
||||
exit 1
|
||||
;;
|
||||
esac
|
||||
|
|
@ -128,9 +119,6 @@ echo -e "${BLUE}New version: ${NEW_VERSION}${NC}"
|
|||
if [ "$PRERELEASE" = true ]; then
|
||||
echo -e "${YELLOW}⚠️ Prerelease → npm dist-tag '${NPM_TAG}', GitHub prerelease${NC}"
|
||||
fi
|
||||
if [ "$SOURCE_ONLY" = true ]; then
|
||||
echo -e "${YELLOW}⚠️ The Source is the one registry; --source-only is implied${NC}"
|
||||
fi
|
||||
echo ""
|
||||
|
||||
if [ "$DRY_RUN" = true ]; then
|
||||
|
|
@ -154,7 +142,7 @@ else
|
|||
fi
|
||||
|
||||
# Create new changelog entry
|
||||
CHANGELOG_ENTRY="### [${NEW_VERSION}](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v${CURRENT_VERSION}...v${NEW_VERSION}) ($(date +%Y-%m-%d))
|
||||
CHANGELOG_ENTRY="### [${NEW_VERSION}](https://github.com/soulcraftlabs/brainy/compare/v${CURRENT_VERSION}...v${NEW_VERSION}) ($(date +%Y-%m-%d))
|
||||
|
||||
${COMMITS}
|
||||
"
|
||||
|
|
@ -187,76 +175,30 @@ echo -e "${BLUE}7️⃣ Creating git tag v${NEW_VERSION}...${NC}"
|
|||
git tag -a "v${NEW_VERSION}" -m "Release v${NEW_VERSION}"
|
||||
echo -e "${GREEN}✅ Tag created${NC}\n"
|
||||
|
||||
# Step 9: Push to origin — The Source is the one home (ruled 2026-07-23; the
|
||||
# old public GitHub repo is archived history, no longer part of any release).
|
||||
# TAG FIRST, branch second — deliberately two pushes: the runner is
|
||||
# sequential, and a combined push can queue the release commit's ci.yml run
|
||||
# AHEAD of the tag's publish-source run (observed on 10.0.0: the publish sat
|
||||
# ~37 minutes behind a redundant CI run of the very commit the local gates
|
||||
# had just proven). Pushing the tag alone queues the publish immediately;
|
||||
# the branch push (and its ci.yml run) follows behind it, harmlessly.
|
||||
echo -e "${BLUE}8️⃣ Pushing to origin (tag first — the publish must never queue behind CI)...${NC}"
|
||||
git push origin "v${NEW_VERSION}"
|
||||
git push origin "$CURRENT_BRANCH"
|
||||
echo -e "${GREEN}✅ Pushed to origin${NC}\n"
|
||||
# Step 9: Push to GitHub
|
||||
echo -e "${BLUE}8️⃣ Pushing to GitHub...${NC}"
|
||||
git push --follow-tags origin "$CURRENT_BRANCH"
|
||||
echo -e "${GREEN}✅ Pushed to GitHub${NC}\n"
|
||||
|
||||
# Step 10: The home publish (The Source, source.soulcraft.com) is CI's job
|
||||
# now, not the laptop's — a tag push (just above) triggers
|
||||
# .forgejo/workflows/publish-source.yml, which builds and publishes on The
|
||||
# Source's own runner (datacenter-side: seconds, not the laptop's WAN timing
|
||||
# out on an 87MB tarball PUT). The laptop holds no home-registry publish
|
||||
# credential anymore; it only waits for CI's result before continuing on to
|
||||
# the release page and the docs push.
|
||||
SOURCE_NPM_REG="https://source.soulcraft.com/api/packages/soulcraftlabs/npm/"
|
||||
SOURCE_POLL_INTERVAL_S=15
|
||||
SOURCE_POLL_MAX_ATTEMPTS=200 # 200 × 15s = 50 minutes — the runner is sequential and a busy day's ci.yml
|
||||
# backlog has twice exceeded the old 20-minute window (8.10.3, 9.0.0);
|
||||
# ci.yml no longer runs on tag pushes, but same-day branch pushes still queue ahead
|
||||
echo -e "${BLUE}9️⃣ Waiting for CI to publish v${NEW_VERSION} to The Source registry (home)...${NC}"
|
||||
SOURCE_LANDED=false
|
||||
for ((attempt = 1; attempt <= SOURCE_POLL_MAX_ATTEMPTS; attempt++)); do
|
||||
LANDED_VERSION=$(npm view "@soulcraftlabs/brainy@${NEW_VERSION}" version "--@soulcraftlabs:registry=${SOURCE_NPM_REG}" 2>/dev/null || echo "")
|
||||
if [ "$LANDED_VERSION" = "$NEW_VERSION" ]; then
|
||||
SOURCE_LANDED=true
|
||||
break
|
||||
fi
|
||||
echo -e "${YELLOW} … not yet on The Source (attempt ${attempt}/${SOURCE_POLL_MAX_ATTEMPTS}); retrying in ${SOURCE_POLL_INTERVAL_S}s${NC}"
|
||||
sleep "$SOURCE_POLL_INTERVAL_S"
|
||||
done
|
||||
# Step 10: Publish to npm
|
||||
echo -e "${BLUE}9️⃣ Publishing to npm (dist-tag: ${NPM_TAG})...${NC}"
|
||||
npm publish --tag "$NPM_TAG"
|
||||
# Brainy is the only PUBLIC @soulcraft package — verify visibility after every publish.
|
||||
npm access get status @soulcraft/brainy || true
|
||||
echo -e "${GREEN}✅ Published to npm${NC}\n"
|
||||
|
||||
if [ "$SOURCE_LANDED" = true ]; then
|
||||
echo -e "${GREEN}✅ CI published v${NEW_VERSION} to The Source${NC}\n"
|
||||
# Step 11: Create GitHub release
|
||||
echo -e "${BLUE}🔟 Creating GitHub release...${NC}"
|
||||
if [ "$PRERELEASE" = true ]; then
|
||||
gh release create "v${NEW_VERSION}" --generate-notes --prerelease
|
||||
else
|
||||
echo -e "${RED}❌ CI's home publish did not land — check the workflow run on The Source; the pair must not diverge.${NC}"
|
||||
echo -e "${RED} v${NEW_VERSION} was tagged and pushed, but @soulcraftlabs/brainy@${NEW_VERSION} never became visible on the${NC}"
|
||||
echo -e "${RED} Source registry after ${SOURCE_POLL_MAX_ATTEMPTS} attempts, ${SOURCE_POLL_INTERVAL_S}s apart. Aborting.${NC}"
|
||||
exit 1
|
||||
gh release create "v${NEW_VERSION}" --generate-notes
|
||||
fi
|
||||
|
||||
# Step 11: Release object on The Source (presentational — the tag, CHANGELOG,
|
||||
# and RELEASES.md are the record; this just gives The Source's UI a release page).
|
||||
echo -e "${BLUE}🔟 Creating release page on The Source...${NC}"
|
||||
if [ -n "${FORGEJO_RELEASE_TOKEN:-}" ]; then
|
||||
if curl -sf -X POST "https://source.soulcraft.com/api/v1/repos/soulcraftlabs/open-brainy/releases" \
|
||||
-H "Authorization: token ${FORGEJO_RELEASE_TOKEN}" -H "Content-Type: application/json" \
|
||||
-d "{\"tag_name\":\"v${NEW_VERSION}\",\"name\":\"v${NEW_VERSION}\",\"prerelease\":${PRERELEASE}}" >/dev/null; then
|
||||
echo -e "${GREEN}✅ Release page created on The Source${NC}\n"
|
||||
else
|
||||
echo -e "${RED}⚠️ Release-page API call failed — tag + CHANGELOG remain the record; create the page via The Source's UI if wanted${NC}\n"
|
||||
fi
|
||||
else
|
||||
echo -e "${RED}⚠️ FORGEJO_RELEASE_TOKEN unset — no release page created; tag + CHANGELOG remain the record${NC}\n"
|
||||
fi
|
||||
|
||||
# Step 12 RETIRED (2026-08-31, CORTEX-SITE-BRAINY-RENAME round 12, David-ruled):
|
||||
# soulcraft.com/docs carries the paid product's documentation only. This
|
||||
# engine's documentation home is THIS repository — README and docs/ — and the
|
||||
# site serves 301s for the slugs this rail used to push. The push script stays
|
||||
# in the tree for history; the rail no longer calls it.
|
||||
echo -e "${BLUE}Docs step: this engine documents itself in its own repo (site push retired 2026-08-31)${NC}"
|
||||
echo -e "${GREEN}✅ GitHub release created${NC}\n"
|
||||
|
||||
echo -e "${GREEN}━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━${NC}"
|
||||
echo -e "${GREEN}🎉 Release ${NEW_VERSION} complete!${NC}"
|
||||
echo -e "${GREEN}━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━${NC}"
|
||||
echo ""
|
||||
echo -e "🏠 The Source: ${BLUE}https://source.soulcraft.com/soulcraftlabs/open-brainy/releases/tag/v${NEW_VERSION}${NC}"
|
||||
echo -e "📦 npm: ${BLUE}https://www.npmjs.com/package/@soulcraft/brainy/v/${NEW_VERSION}${NC}"
|
||||
echo -e "🐙 GitHub: ${BLUE}https://github.com/soulcraftlabs/brainy/releases/tag/v${NEW_VERSION}${NC}"
|
||||
|
|
|
|||
|
|
@ -14,22 +14,7 @@
|
|||
*/
|
||||
|
||||
import type { StorageAdapter, HNSWNounWithMetadata } from '../coreTypes.js'
|
||||
import { parseFieldAddress, readEntityFieldAddress } from '../db/fieldAddressing.js'
|
||||
import type { HNSWNounWithMetadata as AddressedEntity } from '../coreTypes.js'
|
||||
|
||||
/**
|
||||
* Read a user-supplied field name under the one addressing law (sealed
|
||||
* 2026-08-03): bare / `metadata.` = the user's metadata field, `system.<x>` =
|
||||
* the ruled engine scalar, malformed = typed refusal. The aggregation engine
|
||||
* NEVER resolves names any other way — the pre-law resolver made bare
|
||||
* `subtype`/`confidence` read engine scalars, silently shadowing user fields.
|
||||
*/
|
||||
function readAddressed(e: unknown, name: string): unknown {
|
||||
return readEntityFieldAddress(
|
||||
e as AddressedEntity,
|
||||
parseFieldAddress(name, 'entity')
|
||||
)
|
||||
}
|
||||
import { resolveEntityField } from '../coreTypes.js'
|
||||
import type {
|
||||
AggregateDefinition,
|
||||
AggregateGroupState,
|
||||
|
|
@ -44,7 +29,6 @@ import { matchesMetadataFilter } from '../utils/metadataFilter.js'
|
|||
import { compareCodePoints } from '../utils/collation.js'
|
||||
import { bucketTimestamp } from './timeWindows.js'
|
||||
import { NounType } from '../types/graphTypes.js'
|
||||
import { prodLog } from '../utils/logger.js'
|
||||
|
||||
/** Persistence key for aggregate definitions */
|
||||
const DEFINITIONS_KEY = '__aggregation_definitions__'
|
||||
|
|
@ -103,22 +87,10 @@ function matchesSource(entity: Record<string, unknown>, source: AggregateDefinit
|
|||
if (entity.service !== source.service) return false
|
||||
}
|
||||
|
||||
// Where filter — resolve each filtered field through resolveEntityField,
|
||||
// the SAME single source of truth groupBy uses (top-level standard fields
|
||||
// + custom metadata). Matching only the metadata sub-object made
|
||||
// where:{subtype}/{visibility}/… a silent no-op: reserved fields never
|
||||
// live in the custom bag, so those filters could never match anything.
|
||||
// Metadata where filter — match against the entity's metadata sub-object
|
||||
if (source.where && Object.keys(source.where).length > 0) {
|
||||
const e = entity as unknown as HNSWNounWithMetadata
|
||||
for (const [key, condition] of Object.entries(source.where)) {
|
||||
// Evaluate ONE field at a time under a neutral key: the address may be
|
||||
// dotted ('system.subtype'), and the filter evaluator would otherwise
|
||||
// walk dots as a nested path instead of treating the key as an address.
|
||||
const value = readAddressed(e, key)
|
||||
if (!matchesMetadataFilter({ v: value }, { v: condition } as Record<string, unknown>)) {
|
||||
return false
|
||||
}
|
||||
}
|
||||
const metadata = (entity.metadata ?? entity) as Record<string, unknown>
|
||||
if (!matchesMetadataFilter(metadata, source.where)) return false
|
||||
}
|
||||
|
||||
return true
|
||||
|
|
@ -148,11 +120,11 @@ function computeGroupKeys(
|
|||
|
||||
for (const dim of groupBy) {
|
||||
if (typeof dim === 'string') {
|
||||
const val = readAddressed(e, dim)
|
||||
const val = resolveEntityField(e, dim)
|
||||
const v = val !== undefined && val !== null ? String(val) : '__null__'
|
||||
for (const k of keys) k[dim] = v
|
||||
} else if ('unnest' in dim) {
|
||||
const val = readAddressed(e, dim.field)
|
||||
const val = resolveEntityField(e, dim.field)
|
||||
const raw = Array.isArray(val) ? val : val !== undefined && val !== null ? [val] : []
|
||||
// Distinct elements: an entity with duplicate tags counts once per distinct tag.
|
||||
const elems = Array.from(new Set(raw.map(x => String(x))))
|
||||
|
|
@ -164,7 +136,7 @@ function computeGroupKeys(
|
|||
keys = next
|
||||
} else {
|
||||
// Time-windowed field
|
||||
const val = readAddressed(e, dim.field)
|
||||
const val = resolveEntityField(e, dim.field)
|
||||
const v = typeof val === 'number' ? bucketTimestamp(val, dim.window) : '__null__'
|
||||
for (const k of keys) k[dim.field] = v
|
||||
}
|
||||
|
|
@ -193,7 +165,7 @@ function computeGroupKey(
|
|||
* in metadata are both handled in one place.
|
||||
*/
|
||||
function getNumericField(entity: Record<string, unknown>, field: string): number | undefined {
|
||||
const val = readAddressed(entity as unknown as HNSWNounWithMetadata, field)
|
||||
const val = resolveEntityField(entity as unknown as HNSWNounWithMetadata, field)
|
||||
if (typeof val === 'number' && !isNaN(val)) return val
|
||||
if (typeof val === 'string') {
|
||||
const num = parseFloat(val)
|
||||
|
|
@ -355,39 +327,6 @@ export class AggregationIndex {
|
|||
/** Track aggregates with stale MIN/MAX (need lazy recompute) */
|
||||
private staleMinMax = new Map<string, Set<string>>()
|
||||
|
||||
/** Resolves when init() has finished loading persisted definitions/state. */
|
||||
private initPromise: Promise<void> | null = null
|
||||
|
||||
/** True once init() has settled (success or failure). */
|
||||
private initDone = false
|
||||
|
||||
/**
|
||||
* Aggregates registered by the app before init() finished loading persisted
|
||||
* state, awaiting reconciliation: init() adopts the persisted state when the
|
||||
* definition hash matches; anything left unadopted when init settles resolves
|
||||
* to a backfill. Deciding backfill eagerly at define time was the boot-order
|
||||
* bug that wiped valid persisted state on every restart — the synchronous
|
||||
* defineAggregate() always beats the async init().
|
||||
*/
|
||||
private pendingAdopt = new Set<string>()
|
||||
|
||||
/**
|
||||
* Aggregates adopted with a BEHIND stamp: name → the exact generation
|
||||
* window `(from, to]` whose writes the adopted state has not seen. The
|
||||
* owner (Brainy) drains this via {@link getPendingCatchUps} +
|
||||
* {@link reconcileEntity} + {@link finishCatchUp} BEFORE serving queries —
|
||||
* cost bounded by the window's affected entities, never store size.
|
||||
*/
|
||||
private pendingCatchUp = new Map<string, { from: number; to: number }>()
|
||||
|
||||
/**
|
||||
* In-flight rescan targets. While a name has a staging map, ALL
|
||||
* contributions (the walk's and concurrent write hooks') land there instead
|
||||
* of the live map; the live map keeps serving until {@link finishBackfill}
|
||||
* swaps the staging map in atomically.
|
||||
*/
|
||||
private backfillStaging = new Map<string, Map<string, AggregateGroupState>>()
|
||||
|
||||
constructor(storage: StorageAdapter, nativeProvider?: AggregationProvider) {
|
||||
this.storage = storage
|
||||
this.nativeProvider = nativeProvider
|
||||
|
|
@ -397,163 +336,28 @@ export class AggregationIndex {
|
|||
|
||||
/**
|
||||
* Initialize: load persisted definitions and state, detect changes, rebuild stale.
|
||||
*
|
||||
* Idempotent — repeated calls return the same promise. Definitions registered
|
||||
* *before* this completes (the normal boot order: `defineAggregate()` is
|
||||
* synchronous and always beats this async load) are reconciled rather than
|
||||
* clobbered: the app's definition wins, and its persisted state is adopted
|
||||
* when the definition hash matches — backfill happens only on a real change.
|
||||
*/
|
||||
init(): Promise<void> {
|
||||
if (!this.initPromise) {
|
||||
this.initPromise = this.loadPersisted().finally(() => {
|
||||
this.resolvePendingAdoptToBackfill()
|
||||
this.initDone = true
|
||||
})
|
||||
}
|
||||
return this.initPromise
|
||||
}
|
||||
|
||||
/**
|
||||
* Await the persisted-state load (if one was started) and settle every
|
||||
* pending adoption decision. After this resolves, `getPendingBackfills()`
|
||||
* is authoritative: a name is listed iff it genuinely needs a rescan.
|
||||
* Query paths must await this before consulting backfill state.
|
||||
*/
|
||||
async ready(): Promise<void> {
|
||||
if (this.initPromise) {
|
||||
try {
|
||||
await this.initPromise
|
||||
} catch {
|
||||
// The owner already surfaced the load failure loudly; backfill covers.
|
||||
}
|
||||
}
|
||||
this.resolvePendingAdoptToBackfill()
|
||||
}
|
||||
|
||||
/**
|
||||
* Any definition still awaiting state adoption has no persisted state to
|
||||
* adopt (or init never ran / failed) — it must backfill.
|
||||
*/
|
||||
private resolvePendingAdoptToBackfill(): void {
|
||||
if (this.pendingAdopt.size > 0) {
|
||||
prodLog.info(
|
||||
`[Aggregation] no adoptable persisted state for: ${Array.from(this.pendingAdopt).join(', ')} — flagged for backfill`
|
||||
)
|
||||
}
|
||||
for (const name of this.pendingAdopt) this.needsBackfill.add(name)
|
||||
this.pendingAdopt.clear()
|
||||
}
|
||||
|
||||
/**
|
||||
* The adoption verdict for persisted state, against the store's committed
|
||||
* watermark (SELF-ENGINE-LIFECYCLE-SPRINT ask (b) — behind-stamp is no
|
||||
* longer a whole-store rescan):
|
||||
*
|
||||
* - `'adopt'` — stamp equals the watermark (clean), or the store has no
|
||||
* watermark capability (hash-only adoption, the pre-stamp behavior).
|
||||
* - `'catchup'` — stamp is BEHIND the watermark (an unclean exit after
|
||||
* later writes, or a long-lived writer whose last flush predates recent
|
||||
* writes). The state is exact AS OF its stamp, so it is adopted and the
|
||||
* missing window `(stamp, committed]` is reconciled INCREMENTALLY per
|
||||
* affected entity via time-travel reads — bounded by writes since the
|
||||
* last flush, never by store size. The owner drains
|
||||
* {@link getPendingCatchUps} before serving queries.
|
||||
* - `'rescan'` — no stamp (pre-stamp state on a stamped store) or stamp
|
||||
* AHEAD of the watermark (e.g. a fact-log truncation on a copied store
|
||||
* pulled the watermark back): the state over-counts unverifiably; one
|
||||
* exact rescan, said out loud.
|
||||
*/
|
||||
private stateAdoptionVerdict(
|
||||
name: string,
|
||||
stateData: unknown
|
||||
): 'adopt' | 'catchup' | 'rescan' {
|
||||
const committed = this.storage.committedGeneration?.() ?? null
|
||||
if (committed === null) return 'adopt'
|
||||
const raw = (stateData as Record<string, unknown>).sourceGeneration
|
||||
const stamped = typeof raw === 'number' ? raw : null
|
||||
if (stamped === committed) return 'adopt'
|
||||
if (stamped !== null && stamped < committed) {
|
||||
this.pendingCatchUp.set(name, { from: stamped, to: committed })
|
||||
prodLog.info(
|
||||
`[Aggregation] '${name}': persisted state is at generation ${stamped}, store is at ` +
|
||||
`${committed} — adopting and reconciling the ${committed - stamped}-generation window ` +
|
||||
`incrementally (no store rescan)`
|
||||
)
|
||||
return 'catchup'
|
||||
}
|
||||
prodLog.warn(
|
||||
`[Aggregation] '${name}': persisted state is at generation ${stamped ?? 'unstamped'} ` +
|
||||
`but the store's committed generation is ${committed} — rescanning instead of adopting`
|
||||
)
|
||||
return 'rescan'
|
||||
}
|
||||
|
||||
private async loadPersisted(): Promise<void> {
|
||||
async init(): Promise<void> {
|
||||
// Load persisted definitions
|
||||
const savedDefs = await this.storage.getMetadata(DEFINITIONS_KEY)
|
||||
if (savedDefs && typeof savedDefs === 'object' && savedDefs.definitions) {
|
||||
const defs = savedDefs.definitions as Array<AggregateDefinition & { _hash?: string }>
|
||||
|
||||
for (const def of defs) {
|
||||
const savedHash = def._hash || ''
|
||||
|
||||
if (this.definitions.has(def.name)) {
|
||||
// The app re-registered this aggregate before the load finished.
|
||||
// The app's definition wins — never clobber it with the persisted
|
||||
// copy. Adopt the persisted state when the definition is unchanged
|
||||
// AND no write has landed for it yet (a landed write would be lost
|
||||
// by adoption; the hook flips such names to backfill).
|
||||
const appHash = this.definitionHashes.get(def.name) || ''
|
||||
if (appHash === savedHash && this.pendingAdopt.has(def.name)) {
|
||||
const stateData = await this.storage.getMetadata(`${STATE_KEY_PREFIX}${def.name}__`)
|
||||
const verdict =
|
||||
stateData && stateData.groups
|
||||
? this.stateAdoptionVerdict(def.name, stateData)
|
||||
: 'rescan'
|
||||
if (verdict !== 'rescan') {
|
||||
const groupMap = new Map<string, AggregateGroupState>()
|
||||
for (const group of stateData!.groups as AggregateGroupState[]) {
|
||||
groupMap.set(serializeGroupKey(group.groupKey), group)
|
||||
}
|
||||
this.states.set(def.name, groupMap)
|
||||
this.pendingAdopt.delete(def.name)
|
||||
this.needsBackfill.delete(def.name)
|
||||
prodLog.info(
|
||||
`[Aggregation] '${def.name}': adopted persisted state (${groupMap.size} groups) — ` +
|
||||
(verdict === 'catchup' ? 'incremental catch-up pending' : 'no rescan')
|
||||
)
|
||||
}
|
||||
// No/invalid persisted state: stays in pendingAdopt and resolves
|
||||
// to backfill when init settles.
|
||||
}
|
||||
continue
|
||||
}
|
||||
|
||||
// Not registered this session — restore definition + state from
|
||||
// persistence.
|
||||
this.definitions.set(def.name, def)
|
||||
const currentHash = hashDefinition(def)
|
||||
const savedHash = def._hash || ''
|
||||
|
||||
// Load persisted state
|
||||
const stateData = await this.storage.getMetadata(`${STATE_KEY_PREFIX}${def.name}__`)
|
||||
const restoreVerdict =
|
||||
stateData && stateData.groups && savedHash === currentHash
|
||||
? this.stateAdoptionVerdict(def.name, stateData)
|
||||
: 'rescan'
|
||||
if (restoreVerdict !== 'rescan') {
|
||||
// Definition unchanged — load state (exact as of its stamp; a
|
||||
// 'catchup' verdict reconciles the missing window incrementally).
|
||||
if (stateData && stateData.groups && savedHash === currentHash) {
|
||||
// Definition unchanged — load state
|
||||
const groupMap = new Map<string, AggregateGroupState>()
|
||||
for (const group of stateData!.groups as AggregateGroupState[]) {
|
||||
for (const group of stateData.groups as AggregateGroupState[]) {
|
||||
const serialized = serializeGroupKey(group.groupKey)
|
||||
groupMap.set(serialized, group)
|
||||
}
|
||||
this.states.set(def.name, groupMap)
|
||||
this.needsBackfill.delete(def.name)
|
||||
prodLog.info(
|
||||
`[Aggregation] '${def.name}': restored definition + adopted persisted state (${groupMap.size} groups)` +
|
||||
(restoreVerdict === 'catchup' ? ' — incremental catch-up pending' : '')
|
||||
)
|
||||
} else {
|
||||
// Definition changed or no saved state — start fresh and backfill from
|
||||
// existing entities (the owner drains needsBackfill on first query).
|
||||
|
|
@ -570,35 +374,15 @@ export class AggregationIndex {
|
|||
}
|
||||
}
|
||||
|
||||
// Restore native provider state from persistence — GATED by the same
|
||||
// adoption verdict as caller-side state (the unconditional adopt was an
|
||||
// asymmetry: a stale native blob restored over a moved store silently
|
||||
// over/under-counted). 'adopt' restores; 'catchup' restores too (the
|
||||
// incremental reconciliation drives the provider through
|
||||
// incrementalUpdate over the exact missing window); 'rescan' SKIPS the
|
||||
// blob — the flagged rebuild repopulates the provider from source.
|
||||
// Legacy unstamped envelopes verdict as rescan, loudly, never silently.
|
||||
// Restore native provider state from persistence
|
||||
if (this.nativeProvider?.restoreState) {
|
||||
const nativeState = await this.storage.getMetadata('__aggregation_native_state__')
|
||||
const blob =
|
||||
nativeState && typeof nativeState === 'string'
|
||||
? nativeState
|
||||
: nativeState && typeof nativeState === 'object' && nativeState.data
|
||||
? (nativeState.data as string)
|
||||
: null
|
||||
if (blob !== null) {
|
||||
const verdict = this.stateAdoptionVerdict(
|
||||
'__native__',
|
||||
nativeState && typeof nativeState === 'object' ? (nativeState as Record<string, unknown>) : {}
|
||||
)
|
||||
if (verdict === 'adopt' || verdict === 'catchup') {
|
||||
this.nativeProvider.restoreState(blob)
|
||||
} else {
|
||||
prodLog.warn(
|
||||
`[Aggregation] native provider state not adopted (verdict: ${verdict}) — ` +
|
||||
`the flagged rescan repopulates the provider from source`
|
||||
)
|
||||
}
|
||||
if (nativeState && typeof nativeState === 'string') {
|
||||
this.nativeProvider.restoreState(nativeState)
|
||||
} else if (nativeState && typeof nativeState === 'object' && nativeState.data) {
|
||||
// flush() persists `{ data: serializeState() }`, so `data` is the
|
||||
// provider's serialized state string.
|
||||
this.nativeProvider.restoreState(nativeState.data as string)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
|
@ -614,37 +398,24 @@ export class AggregationIndex {
|
|||
}))
|
||||
await this.storage.saveMetadata(DEFINITIONS_KEY, { definitions: defsToSave })
|
||||
|
||||
// Persist dirty states, stamped with the committed generation they
|
||||
// reflect. The stamp is what makes reopen-adoption verifiable: state at a
|
||||
// different generation than the store's committed watermark is stale (an
|
||||
// unclean shutdown after later writes) or over-counts (a fact-log
|
||||
// truncation on a copied store pulled the watermark BACK below the
|
||||
// stamp) — either way the answer is one exact rescan, never a silent
|
||||
// adopt. Read the generation after collecting groups so any racing
|
||||
// commit resolves toward rescan, not wrong-adopt.
|
||||
// Persist dirty states
|
||||
for (const name of this.dirty) {
|
||||
const stateMap = this.states.get(name)
|
||||
if (stateMap) {
|
||||
const groups = Array.from(stateMap.values())
|
||||
const sourceGeneration = this.storage.committedGeneration?.() ?? null
|
||||
await this.storage.saveMetadata(
|
||||
`${STATE_KEY_PREFIX}${name}__`,
|
||||
sourceGeneration === null ? { groups } : { groups, sourceGeneration }
|
||||
{ groups }
|
||||
)
|
||||
}
|
||||
}
|
||||
|
||||
// Persist native provider state — stamped. noteSourceGeneration lets the
|
||||
// provider bake the committed watermark into its OWN envelope before
|
||||
// serializing (so a native-side reopen can verify honesty without our
|
||||
// wrapper); the wrapper carries the same stamp for OUR adoption verdict.
|
||||
// Persist native provider state
|
||||
if (this.nativeProvider?.serializeState) {
|
||||
const nativeGen = this.storage.committedGeneration?.() ?? null
|
||||
if (nativeGen !== null) this.nativeProvider.noteSourceGeneration?.(nativeGen)
|
||||
const nativeState = this.nativeProvider.serializeState()
|
||||
await this.storage.saveMetadata(
|
||||
'__aggregation_native_state__',
|
||||
nativeGen === null ? { data: nativeState } : { data: nativeState, sourceGeneration: nativeGen }
|
||||
{ data: nativeState }
|
||||
)
|
||||
}
|
||||
|
||||
|
|
@ -681,19 +452,10 @@ export class AggregationIndex {
|
|||
this.definitions.set(def.name, def)
|
||||
this.definitionHashes.set(def.name, newHash)
|
||||
|
||||
// First sight this session, before init() settled: defer the backfill
|
||||
// decision — init() adopts the persisted state on hash match, and anything
|
||||
// left unadopted resolves to backfill. Deciding eagerly here wiped valid
|
||||
// persisted state on every restart.
|
||||
if (!this.states.has(def.name) && !this.initDone) {
|
||||
this.states.set(def.name, new Map())
|
||||
this.pendingAdopt.add(def.name)
|
||||
}
|
||||
// Reset state if definition changed or doesn't exist yet, and flag it for
|
||||
// backfill so already-stored entities are counted (write-time hooks only see
|
||||
// future writes). The owner drains this on the next query via getPendingBackfills().
|
||||
else if (!this.states.has(def.name) || (oldHash && oldHash !== newHash)) {
|
||||
this.pendingAdopt.delete(def.name)
|
||||
if (!this.states.has(def.name) || (oldHash && oldHash !== newHash)) {
|
||||
this.states.set(def.name, new Map())
|
||||
this.needsBackfill.add(def.name)
|
||||
}
|
||||
|
|
@ -714,8 +476,6 @@ export class AggregationIndex {
|
|||
this.definitionHashes.delete(name)
|
||||
this.states.delete(name)
|
||||
this.staleMinMax.delete(name)
|
||||
this.pendingAdopt.delete(name)
|
||||
this.needsBackfill.delete(name)
|
||||
|
||||
// Notify native provider
|
||||
if (this.nativeProvider?.removeAggregate) {
|
||||
|
|
@ -753,17 +513,9 @@ export class AggregationIndex {
|
|||
return Array.from(this.needsBackfill)
|
||||
}
|
||||
|
||||
/**
|
||||
* Begin a rescan into a STAGING map. The live state is not touched — it
|
||||
* keeps serving (possibly stale, but flagged pending) until the rescan
|
||||
* completes and swaps in atomically. A mid-walk failure drops the staging
|
||||
* map via {@link abortBackfill} and loses nothing: wiping live state before
|
||||
* a scan that could throw was the destructive-before-durable defect.
|
||||
* Contributions (walk + concurrent write hooks) land in staging while it
|
||||
* exists, so the swapped-in result reflects writes that raced the walk.
|
||||
*/
|
||||
/** Clear an aggregate's state so a full rescan cannot double-count. */
|
||||
beginBackfill(name: string): void {
|
||||
this.backfillStaging.set(name, new Map())
|
||||
this.states.set(name, new Map())
|
||||
// Reset native provider state for this aggregate too, if present.
|
||||
const def = this.definitions.get(name)
|
||||
if (def && this.nativeProvider?.removeAggregate && this.nativeProvider?.defineAggregate) {
|
||||
|
|
@ -772,15 +524,6 @@ export class AggregationIndex {
|
|||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Abandon an in-flight rescan after a failure: drop the staging map, keep
|
||||
* the live state serving, leave the aggregate flagged as pending so a later
|
||||
* attempt rescans. The failure itself must be surfaced loudly by the owner.
|
||||
*/
|
||||
abortBackfill(name: string): void {
|
||||
this.backfillStaging.delete(name)
|
||||
}
|
||||
|
||||
/** Feed one already-stored entity into a single aggregate during backfill. */
|
||||
backfillEntity(name: string, entity: Record<string, unknown>): void {
|
||||
if (isAggregateEntity(entity)) return
|
||||
|
|
@ -794,146 +537,14 @@ export class AggregationIndex {
|
|||
}
|
||||
}
|
||||
|
||||
/** Swap the rebuilt staging state in atomically; persists on next flush(). */
|
||||
/** Mark an aggregate's backfill complete; rebuilt state persists on next flush(). */
|
||||
finishBackfill(name: string): void {
|
||||
const staged = this.backfillStaging.get(name)
|
||||
if (staged) {
|
||||
this.states.set(name, staged)
|
||||
this.backfillStaging.delete(name)
|
||||
}
|
||||
this.needsBackfill.delete(name)
|
||||
this.dirty.add(name)
|
||||
}
|
||||
|
||||
// ============= Incremental Catch-Up (behind-stamp adoption) =============
|
||||
|
||||
/** The aggregates adopted behind the watermark, with their exact missing windows. */
|
||||
getPendingCatchUps(): Array<{ name: string; from: number; to: number }> {
|
||||
return Array.from(this.pendingCatchUp, ([name, w]) => ({ name, ...w }))
|
||||
}
|
||||
|
||||
/**
|
||||
* Reconcile ONE entity's contribution across a catch-up window using the
|
||||
* same exact delta algebra the write-time hooks use: remove the
|
||||
* contribution the adopted state counted (the entity AS OF the stamp),
|
||||
* add the contribution it should count (AS OF the window's end). `null`
|
||||
* on either side means the entity did not exist then. Composes exactly
|
||||
* with live hooks because every application is a precise old/new pair —
|
||||
* order between catch-up and post-window writes cannot drift the totals.
|
||||
*/
|
||||
reconcileEntity(
|
||||
name: string,
|
||||
id: string,
|
||||
before: Record<string, unknown> | null,
|
||||
after: Record<string, unknown> | null
|
||||
): void {
|
||||
const def = this.definitions.get(name)
|
||||
if (!def) return
|
||||
if (before && after) {
|
||||
if (isAggregateEntity(after)) return
|
||||
const oldMatches = matchesSource(before, def.source)
|
||||
const newMatches = matchesSource(after, def.source)
|
||||
if (this.nativeProvider && (oldMatches || newMatches)) {
|
||||
this.applyNativeResults(
|
||||
name,
|
||||
this.nativeProvider.incrementalUpdate(name, def, after, 'update', before)
|
||||
)
|
||||
return
|
||||
}
|
||||
if (oldMatches) this.removeContribution(name, def, before)
|
||||
if (newMatches) this.addContribution(name, def, after)
|
||||
return
|
||||
}
|
||||
if (after) {
|
||||
if (isAggregateEntity(after) || !matchesSource(after, def.source)) return
|
||||
if (this.nativeProvider) {
|
||||
this.applyNativeResults(name, this.nativeProvider.incrementalUpdate(name, def, after, 'add'))
|
||||
} else {
|
||||
this.addContribution(name, def, after)
|
||||
}
|
||||
return
|
||||
}
|
||||
if (before) {
|
||||
if (isAggregateEntity(before) || !matchesSource(before, def.source)) return
|
||||
if (this.nativeProvider) {
|
||||
this.applyNativeResults(name, this.nativeProvider.incrementalUpdate(name, def, before, 'delete'))
|
||||
} else {
|
||||
this.removeContribution(name, def, before)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/** Whether the native provider offers the parallel whole-rebuild path. */
|
||||
hasProviderRebuild(): boolean {
|
||||
return typeof this.nativeProvider?.rebuildAggregate === 'function'
|
||||
}
|
||||
|
||||
/** The catch-up window for `name` is fully reconciled; state is current. */
|
||||
finishCatchUp(name: string): void {
|
||||
this.pendingCatchUp.delete(name)
|
||||
this.dirty.add(name)
|
||||
}
|
||||
|
||||
/**
|
||||
* A catch-up could not complete (window unreadable, affected set over the
|
||||
* bound, …): demote to an exact rescan, loudly — never serve un-reconciled.
|
||||
*/
|
||||
demoteCatchUpToBackfill(name: string, reason: string): void {
|
||||
this.pendingCatchUp.delete(name)
|
||||
this.needsBackfill.add(name)
|
||||
prodLog.warn(`[Aggregation] '${name}': catch-up demoted to full rescan — ${reason}`)
|
||||
}
|
||||
|
||||
/**
|
||||
* Rebuild an aggregate through the native provider's parallel path
|
||||
* (SELF-ENGINE-LIFECYCLE-SPRINT ask (c) — `rebuildAggregate` existed on
|
||||
* the provider contract but was never invoked; the JS walk fed
|
||||
* per-entity FFI calls instead). Returns false when no provider rebuild
|
||||
* exists — the caller streams the JS walk as before.
|
||||
*/
|
||||
rebuildWithProvider(name: string, entities: Array<Record<string, unknown>>): boolean {
|
||||
const def = this.definitions.get(name)
|
||||
if (!def || !this.nativeProvider?.rebuildAggregate) return false
|
||||
const rebuilt = this.nativeProvider.rebuildAggregate(
|
||||
def,
|
||||
entities.filter(e => !isAggregateEntity(e) && matchesSource(e, def.source))
|
||||
)
|
||||
this.states.set(name, rebuilt)
|
||||
this.backfillStaging.delete(name)
|
||||
this.needsBackfill.delete(name)
|
||||
this.dirty.add(name)
|
||||
return true
|
||||
}
|
||||
|
||||
/**
|
||||
* A write-path hook could not see the entity it needed (e.g. a delete
|
||||
* whose before-image was unavailable): flag EVERY defined aggregate for
|
||||
* an exact rescan, loudly — the counts must never silently drift
|
||||
* (SELF-ENGINE-LIFECYCLE-SPRINT ask (d): the gated hook used to SKIP).
|
||||
*/
|
||||
flagAllForRescan(reason: string): void {
|
||||
for (const name of this.definitions.keys()) this.needsBackfill.add(name)
|
||||
prodLog.warn(
|
||||
`[Aggregation] all ${this.definitions.size} aggregate(s) flagged for rescan — ${reason}`
|
||||
)
|
||||
}
|
||||
|
||||
// ============= Write-Time Hooks =============
|
||||
|
||||
/**
|
||||
* A write is landing for an aggregate whose persisted-state adoption is still
|
||||
* pending — adopting after this write would lose its contribution. Settle the
|
||||
* decision now: an exact rescan instead of adoption. The window is the few
|
||||
* milliseconds between a boot-time defineAggregate() and init() completing,
|
||||
* so this rarely fires; when it does, correctness wins over the walk.
|
||||
*/
|
||||
private resolveAdoptOnWrite(name: string): void {
|
||||
if (this.pendingAdopt.has(name)) {
|
||||
this.pendingAdopt.delete(name)
|
||||
this.needsBackfill.add(name)
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Called when an entity is added. Updates all matching aggregates.
|
||||
*/
|
||||
|
|
@ -942,7 +553,6 @@ export class AggregationIndex {
|
|||
|
||||
for (const [name, def] of this.definitions) {
|
||||
if (!matchesSource(entity, def.source)) continue
|
||||
this.resolveAdoptOnWrite(name)
|
||||
|
||||
if (this.nativeProvider) {
|
||||
const results = this.nativeProvider.incrementalUpdate(name, def, entity, 'add')
|
||||
|
|
@ -969,10 +579,6 @@ export class AggregationIndex {
|
|||
const oldMatches = matchesSource(oldEntity, def.source)
|
||||
const newMatches = matchesSource(newEntity, def.source)
|
||||
|
||||
if (oldMatches || newMatches) {
|
||||
this.resolveAdoptOnWrite(name)
|
||||
}
|
||||
|
||||
if (this.nativeProvider && (oldMatches || newMatches)) {
|
||||
const results = this.nativeProvider.incrementalUpdate(name, def, newEntity, 'update', oldEntity)
|
||||
this.applyNativeResults(name, results)
|
||||
|
|
@ -999,7 +605,6 @@ export class AggregationIndex {
|
|||
|
||||
for (const [name, def] of this.definitions) {
|
||||
if (!matchesSource(entity, def.source)) continue
|
||||
this.resolveAdoptOnWrite(name)
|
||||
|
||||
if (this.nativeProvider) {
|
||||
const results = this.nativeProvider.incrementalUpdate(name, def, entity, 'delete')
|
||||
|
|
@ -1152,7 +757,7 @@ export class AggregationIndex {
|
|||
def: AggregateDefinition,
|
||||
entity: Record<string, unknown>
|
||||
): void {
|
||||
const stateMap = (this.backfillStaging.get(aggName) ?? this.states.get(aggName))!
|
||||
const stateMap = this.states.get(aggName)!
|
||||
|
||||
// Fan out: an unnest dimension makes one entity contribute to several groups.
|
||||
for (const groupKey of computeGroupKeys(entity, def.groupBy)) {
|
||||
|
|
@ -1180,7 +785,7 @@ export class AggregationIndex {
|
|||
// distinctCount tracks distinct values of ANY type (strings, numbers, booleans),
|
||||
// keyed by their string form — NOT numeric-coerced, since its primary use is
|
||||
// categorical (distinct categories / users / tags), not numeric columns.
|
||||
const raw = readAddressed(entity as unknown as HNSWNounWithMetadata, metricDef.field!)
|
||||
const raw = resolveEntityField(entity as unknown as HNSWNounWithMetadata, metricDef.field!)
|
||||
if (raw !== undefined && raw !== null) {
|
||||
if (!state.valueCounts) state.valueCounts = {}
|
||||
const key = String(raw)
|
||||
|
|
@ -1210,7 +815,7 @@ export class AggregationIndex {
|
|||
def: AggregateDefinition,
|
||||
entity: Record<string, unknown>
|
||||
): void {
|
||||
const stateMap = (this.backfillStaging.get(aggName) ?? this.states.get(aggName))!
|
||||
const stateMap = this.states.get(aggName)!
|
||||
|
||||
// Fan out: reverse the entity's contribution from every group it joined.
|
||||
for (const groupKey of computeGroupKeys(entity, def.groupBy)) {
|
||||
|
|
@ -1224,7 +829,7 @@ export class AggregationIndex {
|
|||
state.count = Math.max(0, state.count - 1)
|
||||
state.sum = Math.max(0, state.sum - 1)
|
||||
} else if (metricDef.op === 'distinctCount') {
|
||||
const raw = readAddressed(entity as unknown as HNSWNounWithMetadata, metricDef.field!)
|
||||
const raw = resolveEntityField(entity as unknown as HNSWNounWithMetadata, metricDef.field!)
|
||||
if (raw !== undefined && raw !== null && state.valueCounts) {
|
||||
const key = String(raw)
|
||||
const c = state.valueCounts[key]
|
||||
|
|
@ -1266,7 +871,7 @@ export class AggregationIndex {
|
|||
* Apply results from native provider back into the state maps.
|
||||
*/
|
||||
private applyNativeResults(aggName: string, results: AggregateGroupState[]): void {
|
||||
const stateMap = (this.backfillStaging.get(aggName) ?? this.states.get(aggName))!
|
||||
const stateMap = this.states.get(aggName)!
|
||||
for (const group of results) {
|
||||
const serialized = serializeGroupKey(group.groupKey)
|
||||
stateMap.set(serialized, group)
|
||||
|
|
|
|||
6648
src/brainy.ts
6648
src/brainy.ts
File diff suppressed because it is too large
Load diff
202
src/coreTypes.ts
202
src/coreTypes.ts
|
|
@ -284,12 +284,7 @@ export const STANDARD_ENTITY_FIELDS: ReadonlySet<string> = new Set([
|
|||
'id',
|
||||
'vector',
|
||||
'connections',
|
||||
// 'level' is deliberately ABSENT: it is HNSW plumbing, not an entity field.
|
||||
// Listing it here made every by-name read of a user metadata field called
|
||||
// `level` resolve to the engine's internal node layer instead — a silent
|
||||
// shadow that broke sort/filter/aggregation on a perfectly natural field
|
||||
// name (VENUE-BRAINY-ORDERBY-NOOP). Engine plumbing is invisible to the
|
||||
// query surface; a bare `level` reads `entity.metadata.level`.
|
||||
'level',
|
||||
'type',
|
||||
'subtype',
|
||||
'visibility',
|
||||
|
|
@ -760,7 +755,7 @@ export interface Change {
|
|||
}
|
||||
|
||||
/**
|
||||
* @description A declared derived-index blob FAMILY (the
|
||||
* @description A declared derived-index blob FAMILY (ADR-004 §7 — the
|
||||
* registered-blob contract). A family names the set of on-disk blobs that a
|
||||
* derived index needs AS A SET (e.g. the vector base = `main.dkann` +
|
||||
* `main.slotmap` + `main.slotrev`): losing ANY member corrupts the index. Once a
|
||||
|
|
@ -792,30 +787,6 @@ export interface DerivedFamilyDeclaration {
|
|||
rebuildable?: boolean
|
||||
}
|
||||
|
||||
/**
|
||||
* @description The canonical count ledger a storage adapter maintains on its
|
||||
* write path: per family, the user-facing `counted` scalar and the
|
||||
* ALL-visibility `all` scalar (every tier — the coverage-ledger denominator).
|
||||
* See {@link StorageAdapter.getCanonicalCounts}.
|
||||
*/
|
||||
export interface CanonicalCounts {
|
||||
nouns: { counted: number; all: number }
|
||||
verbs: { counted: number; all: number }
|
||||
/**
|
||||
* The count of canonical nouns holding a REAL (non-empty) vector — the
|
||||
* coverage denominator a vector index's node-count ledger is measured
|
||||
* against (`nodeCount === vectors.all` is the whole-store coverage
|
||||
* verdict for the vector leg, the vector-side mirror of `nouns.all` for
|
||||
* metadata/graph). A deferred-embed noun (`add({ deferEmbedding: true })`)
|
||||
* counts only once its vector actually LANDS — its canonical record exists
|
||||
* (counted in `nouns.all`) with an empty vector until then, so it is
|
||||
* deliberately NOT counted here in the interim.
|
||||
*/
|
||||
vectors: { all: number }
|
||||
/** An unprovable delete has left the `all` scalars unverified since the last recount. */
|
||||
suspect: boolean
|
||||
}
|
||||
|
||||
export interface StorageAdapter {
|
||||
init(): Promise<void>
|
||||
|
||||
|
|
@ -830,68 +801,14 @@ export interface StorageAdapter {
|
|||
* Save noun metadata separately
|
||||
* @param id Noun ID
|
||||
* @param metadata Noun metadata
|
||||
* @param hasVector - OPTIONAL vectored-noun ledger hint: `true` when this
|
||||
* write is a FRESH insert (`isNew`) whose vector is a real, non-empty
|
||||
* array — the caller already knows this for free (the insert's own
|
||||
* `vector` local), so the increment rides the SAME isNew gate that
|
||||
* already protects `totalNounCountAll` from double-counting on HNSW
|
||||
* neighbor-link re-saves (`saveNoun_internal` re-runs on every link
|
||||
* change; this metadata seam does not). Absent/`false` ⇒ no ledger
|
||||
* action. A deferred-embed insert passes `false` (its vector lands
|
||||
* later — see {@link StorageAdapter.noteVectorLanded}).
|
||||
*/
|
||||
saveNounMetadata(id: string, metadata: NounMetadata, hasVector?: boolean): Promise<void>
|
||||
saveNounMetadata(id: string, metadata: NounMetadata): Promise<void>
|
||||
|
||||
/**
|
||||
* Delete noun metadata
|
||||
* @param id Noun ID
|
||||
* @param priorRecord - OPTIONAL already-known metadata (the caller's
|
||||
* pre-delete read) — see {@link StorageAdapter.deleteNoun}.
|
||||
* @param hadVector - OPTIONAL vectored-noun ledger hint: `true`/`false`
|
||||
* when the caller already knows (read as a side effect of ITS OWN delete
|
||||
* flow — e.g. `remove()`'s pre-read for the vector-index removal — never
|
||||
* a read added FOR this ledger), `undefined` when genuinely unknown. A
|
||||
* known `true` decrements the vectored-noun ledger; a known `false` is a
|
||||
* no-op (it was never counted); `undefined` marks the ledger SUSPECT
|
||||
* rather than guessing — the delete path must never add a canonical read
|
||||
* to answer this question.
|
||||
*/
|
||||
deleteNounMetadata(id: string, priorRecord?: NounMetadata | null, hadVector?: boolean): Promise<void>
|
||||
|
||||
/**
|
||||
* OPTIONAL narrow ledger hook: record that a canonical noun's vector just
|
||||
* LANDED for the first time. Exists ONLY for the deferred-embedding
|
||||
* lifecycle — the landing commit (`system:embed-landing`) carries a vector
|
||||
* write with no accompanying metadata operation, so the normal
|
||||
* `saveNounMetadata(..., hasVector)` seam never fires for it. Callers MUST
|
||||
* call this only when the noun held NO real vector before this write (the
|
||||
* deferred-embed worker already holds that fact for free, from its own
|
||||
* pre-embed read — never an added read). A backend without vectored-noun
|
||||
* tracking is a no-op via this method's absence (feature-detected).
|
||||
* @param id - The noun whose vector just landed.
|
||||
*/
|
||||
noteVectorLanded?(id: string): Promise<void>
|
||||
|
||||
/**
|
||||
* OPTIONAL narrow ledger hook, the mirror of {@link noteVectorLanded}:
|
||||
* record that a canonical noun's vector was just REMOVED — rewritten from
|
||||
* a real (non-empty) vector to the "unvectored" empty-array shape. Exists
|
||||
* for the ONE sanctioned reverse migration this engine supports: the VFS
|
||||
* root's zero-norm fix (see `VirtualFileSystem.doInitializeRoot()` and
|
||||
* `Brainy.unvectorNounForRootMigration()`), which rewrites a pre-fix
|
||||
* store's all-zero placeholder root vector to `[]` and must decrement
|
||||
* `vectors.all` through this hook so the coverage ledger never drifts.
|
||||
* NOT a general-purpose "I removed a vector" callback — ordinary
|
||||
* application data has no sanctioned path from vectored back to
|
||||
* unvectored (`update()` refuses an empty vector as a dimension
|
||||
* mismatch by design). Callers MUST call this only when the noun held a
|
||||
* REAL vector immediately before this write (the caller already holds
|
||||
* that fact for free, from its own pre-write read — never an added read).
|
||||
* A backend without vectored-noun tracking is a no-op via this method's
|
||||
* absence (feature-detected).
|
||||
* @param id - The noun whose vector was just removed.
|
||||
*/
|
||||
noteVectorUnlanded?(id: string): Promise<void>
|
||||
deleteNounMetadata(id: string): Promise<void>
|
||||
|
||||
/**
|
||||
* Get noun with metadata combined
|
||||
|
|
@ -931,20 +848,7 @@ export interface StorageAdapter {
|
|||
*/
|
||||
getNounsByNounType(nounType: string): Promise<HNSWNounWithMetadata[]>
|
||||
|
||||
/**
|
||||
* Delete a noun — FULL canonical removal (both legs + the entity container).
|
||||
*
|
||||
* @param id The entity id.
|
||||
* @param priorMetadata OPTIONAL already-known metadata of the entity being
|
||||
* removed (the caller's pre-delete read). The count decrement must never
|
||||
* REQUIRE re-reading the record being removed: when the internal read
|
||||
* returns `null` (replace race, or a ghost left by an earlier version) the
|
||||
* decrement falls back to this record instead of being silently skipped.
|
||||
* @param hadVector OPTIONAL vectored-noun ledger hint — see
|
||||
* {@link StorageAdapter.deleteNounMetadata}'s `hadVector` param, which
|
||||
* this forwards to unchanged.
|
||||
*/
|
||||
deleteNoun(id: string, priorMetadata?: NounMetadata | null, hadVector?: boolean): Promise<void>
|
||||
deleteNoun(id: string): Promise<void>
|
||||
|
||||
/**
|
||||
* Save verb - Pure HNSW verb with core fields only
|
||||
|
|
@ -1001,15 +905,7 @@ export interface StorageAdapter {
|
|||
*/
|
||||
getVerbsByType(type: string): Promise<HNSWVerbWithMetadata[]>
|
||||
|
||||
/**
|
||||
* Delete a verb — FULL canonical removal (both legs + the container).
|
||||
*
|
||||
* @param id The relationship id.
|
||||
* @param priorMetadata OPTIONAL already-known metadata of the edge being
|
||||
* removed (the caller's pre-delete read); keeps the count decrement honest
|
||||
* when the internal read returns `null` (see `deleteNoun`).
|
||||
*/
|
||||
deleteVerb(id: string, priorMetadata?: VerbMetadata | null): Promise<void>
|
||||
deleteVerb(id: string): Promise<void>
|
||||
|
||||
/**
|
||||
* Save metadata
|
||||
|
|
@ -1187,7 +1083,7 @@ export interface StorageAdapter {
|
|||
getBinaryBlobPath(key: string): string | null
|
||||
|
||||
/**
|
||||
* @description OPTIONAL (registered-blob contract). Declare a
|
||||
* @description OPTIONAL (ADR-004 §7 registered-blob contract). Declare a
|
||||
* derived-index blob {@link DerivedFamilyDeclaration | family} whose members
|
||||
* become UNDELETABLE through this adapter — a subsequent `deleteBinaryBlob` /
|
||||
* `removeRawPrefix` that would remove a declared member throws a
|
||||
|
|
@ -1213,77 +1109,6 @@ export interface StorageAdapter {
|
|||
*/
|
||||
listDerivedFamilies?(): Promise<DerivedFamilyDeclaration[]>
|
||||
|
||||
/**
|
||||
* @description OPTIONAL binary raw-byte primitives — the substrate for
|
||||
* append-only log-structured files (the generation fact log's CRC-framed
|
||||
* segments). Feature-detected: an adapter that omits them simply hosts no
|
||||
* fact log (readers fall back to canonical enumeration). Paths are
|
||||
* storage-root-relative and used VERBATIM (no `.gz`/`.bin` suffixing —
|
||||
* unlike the JSON object and blob primitives).
|
||||
*
|
||||
* Append to a raw binary file, creating it (and parent directories) when
|
||||
* absent. Append durability is the CALLER's job via `syncRawObjects` —
|
||||
* matching the commit protocol, which batches fsyncs at its barrier.
|
||||
*/
|
||||
appendRawBytes?(path: string, bytes: Uint8Array): Promise<void>
|
||||
|
||||
/**
|
||||
* Read a raw binary file whole. Absent → `null`; a real IO fault throws
|
||||
* (never masked as absence).
|
||||
*/
|
||||
readRawBytes?(path: string): Promise<Uint8Array | null>
|
||||
|
||||
/**
|
||||
* Replace a raw binary file atomically (write-new → fsync → rename) — the
|
||||
* reconcile primitive (e.g. truncating a fact-log tail back to committed
|
||||
* truth after a crash).
|
||||
*/
|
||||
writeRawBytes?(path: string, bytes: Uint8Array): Promise<void>
|
||||
|
||||
/**
|
||||
* Byte size of a raw binary file, or `null` when absent.
|
||||
*/
|
||||
rawByteSize?(path: string): Promise<number | null>
|
||||
|
||||
/**
|
||||
* @description OPTIONAL fact-scan capability — how an index provider that
|
||||
* holds only `storage` reaches the generation fact log (the host brain
|
||||
* wires it at init; providers must never construct their own fact-log
|
||||
* reader — the log's open path is writer-side). Present ⟺ this store hosts
|
||||
* a fact log AND the host wired the capability. Returns a scan handle over
|
||||
* committed facts (heal-grade telemetry included), or `null` when no fact
|
||||
* log exists — callers fall back to the canonical enumeration walk.
|
||||
*/
|
||||
scanFacts?(options?: {
|
||||
fromGeneration?: number
|
||||
toGeneration?: number
|
||||
kinds?: Array<'noun' | 'verb'>
|
||||
batchSize?: number
|
||||
}): import('./db/factLog.js').FactScanHandle | null
|
||||
|
||||
/**
|
||||
* @description OPTIONAL (rides the fact-scan capability): the fact log's
|
||||
* head generation — the replay target for `stamp.sourceGeneration + 1`
|
||||
* catch-ups. `null` when no fact log exists.
|
||||
*/
|
||||
factLogHeadGeneration?(): number | null
|
||||
|
||||
/**
|
||||
* @description OPTIONAL (rides the fact-scan capability): the COMMITTED
|
||||
* generation watermark — the manifest truth a projection's
|
||||
* `sourceGeneration` compares against. Exposed as a capability so a
|
||||
* provider never parses the store's private manifest format. `null` when
|
||||
* the capability is unwired.
|
||||
*/
|
||||
committedGeneration?(): number | null
|
||||
|
||||
/**
|
||||
* @description OPTIONAL (rides the fact-scan capability): the immutable,
|
||||
* sealed fact-segment file paths covering `fromGeneration` — the zero-copy
|
||||
* handoff. The mutable tail is excluded (read it via `scanFacts`).
|
||||
*/
|
||||
factSegmentPaths?(options?: { fromGeneration?: number }): string[]
|
||||
|
||||
/**
|
||||
* Save statistics data
|
||||
* @param statistics The statistics data to save
|
||||
|
|
@ -1374,19 +1199,6 @@ export interface StorageAdapter {
|
|||
*/
|
||||
getVerbCount(): Promise<number>
|
||||
|
||||
/**
|
||||
* The canonical count ledger — O(1), no I/O. `counted` mirrors
|
||||
* `getNounCount()` / `getVerbCount()` (public + internal tiers); `all` is
|
||||
* the ALL-visibility scalar every unfiltered storage walk is measured
|
||||
* against — the denominator a derived-index provider's coverage ledger
|
||||
* subtracts from. `suspect` is `true` when an unprovable delete has left
|
||||
* `all` unverified since the last sanctioned recount (`repairIndex()`).
|
||||
* Optional: adapters without the ledger omit it; a consumer treats absence
|
||||
* as "no denominator", never as zero.
|
||||
* @returns Both scalars per family plus the suspect flag.
|
||||
*/
|
||||
getCanonicalCounts?(): Promise<CanonicalCounts>
|
||||
|
||||
/**
|
||||
* OPTIONAL — create a pre-upgrade backup of the whole store and return its
|
||||
* location, or `null` when there is nothing to back up (empty store). On the
|
||||
|
|
|
|||
101
src/db/db.ts
101
src/db/db.ts
|
|
@ -59,10 +59,14 @@ import type {
|
|||
import type { StorageAdapter } from '../coreTypes.js'
|
||||
import { exportGraph } from './portableGraph.js'
|
||||
import type { ExportSelector, ExportOptions, PortableGraph } from './portableGraph.js'
|
||||
import {
|
||||
splitNounMetadataRecord,
|
||||
splitVerbMetadataRecord
|
||||
} from '../types/reservedFields.js'
|
||||
import { v4 as uuidv4 } from '../universal/uuid.js'
|
||||
import { coerceNewEntityId, resolveEntityId, ORIGINAL_ID_KEY } from '../utils/idNormalization.js'
|
||||
import { EntityNotFoundError } from '../errors/notFound.js'
|
||||
import { SpeculativeOverlayError, CanonicalEnumerationUnavailableError } from './errors.js'
|
||||
import { SpeculativeOverlayError } from './errors.js'
|
||||
import type { GenerationStore } from './generationStore.js'
|
||||
import type { ChangedIds, TransactReceipt, TxOperation } from './types.js'
|
||||
import { entityMatchesFind, resolveEntityField, UnsupportedWhereOperatorError } from './whereMatcher.js'
|
||||
|
|
@ -516,37 +520,14 @@ export class Db<T = any> {
|
|||
* (no generation history) — distinct from `persist()` (native whole-brain snapshot
|
||||
* that preserves history). Restore with `brain.import(backup)`.
|
||||
*
|
||||
* `options.enumeration: 'canonical'` (default: `'index'`) walks the storage
|
||||
* adapter's canonical noun/verb layout directly instead of the metadata/graph
|
||||
* indexes, guaranteeing canon-completeness against index corruption — see
|
||||
* {@link ExportOptions.enumeration}. It requires the LIVE, current-generation
|
||||
* view: called on a historical `asOf()` pin or a speculative `with()` overlay it
|
||||
* throws {@link CanonicalEnumerationUnavailableError} rather than silently mixing
|
||||
* generations or missing the overlay's own entities.
|
||||
*
|
||||
* `options.includeHidden: true` admits BOTH hidden visibility tiers
|
||||
* (`'internal'` and `'system'`) into a whole-brain/predicate export, in EITHER
|
||||
* `enumeration` mode — see {@link ExportOptions.includeHidden}. Migration-grade
|
||||
* exports set this; consumer-facing exports leave it off (default: false).
|
||||
*
|
||||
* @param selector - WHAT to export (omit for the whole brain). See {@link ExportSelector}.
|
||||
* @param options - HOW to export (vectors / VFS bytes / edge policy / enumeration mode). See {@link ExportOptions}.
|
||||
* @param options - HOW to export (vectors / VFS bytes / edge policy). See {@link ExportOptions}.
|
||||
* @returns A versioned, portable `PortableGraph` document.
|
||||
* @throws {@link CanonicalEnumerationUnavailableError} if `enumeration:'canonical'` is
|
||||
* requested on a historical or speculative-overlay view.
|
||||
* @example
|
||||
* const backup = await brain.now().export({ collection: id }, { includeVectors: true })
|
||||
* @example
|
||||
* // Canon-complete audit export, with an index-drift report attached.
|
||||
* const audit = await brain.now().export({}, { enumeration: 'canonical', reportIndexDrift: true })
|
||||
* if (audit.drift) console.log(audit.drift.canonicalOnly, audit.drift.indexOnly)
|
||||
*/
|
||||
async export(selector: ExportSelector = {}, options: ExportOptions = {}): Promise<PortableGraph> {
|
||||
this.assertUsable('export')
|
||||
if (options.enumeration === 'canonical') {
|
||||
if (this.overlay) throw new CanonicalEnumerationUnavailableError(this.gen, 'overlay')
|
||||
if (this.isHistorical()) throw new CanonicalEnumerationUnavailableError(this.gen, 'historical')
|
||||
}
|
||||
return exportGraph(this, this.host.storage, selector, options)
|
||||
}
|
||||
|
||||
|
|
@ -701,15 +682,23 @@ export class Db<T = any> {
|
|||
for (const op of ops) {
|
||||
switch (op.op) {
|
||||
case 'add': {
|
||||
// Field-addressing law: the metadata bag is the user's, VERBATIM —
|
||||
// no reserved-name lift, no drops. Engine scalars come ONLY from
|
||||
// their dedicated op fields; a bag field named `confidence` is an
|
||||
// ordinary user field, exactly as on the committed write path.
|
||||
const custom = { ...(op.metadata as Record<string, unknown> | undefined) }
|
||||
const confidence = op.confidence
|
||||
const weight = op.weight
|
||||
const subtype = op.subtype
|
||||
const service = op.service
|
||||
// Reserved-field normalization — mirror of the brain.transact()
|
||||
// write path: user-settable fields lift to their dedicated field
|
||||
// (top-level wins), system-managed fields drop, and the entity's
|
||||
// metadata bag carries ONLY custom fields. Speculative views skip
|
||||
// the one-shot warnings — committing the same ops through
|
||||
// `brain.transact()` warns on the real write path.
|
||||
const { reserved, custom } = splitNounMetadataRecord(
|
||||
op.metadata as Record<string, unknown> | undefined
|
||||
)
|
||||
const confidence =
|
||||
op.confidence ?? (typeof reserved.confidence === 'number' ? reserved.confidence : undefined)
|
||||
const weight =
|
||||
op.weight ?? (typeof reserved.weight === 'number' ? reserved.weight : undefined)
|
||||
const subtype =
|
||||
op.subtype ?? (typeof reserved.subtype === 'string' ? reserved.subtype : undefined)
|
||||
const service =
|
||||
op.service ?? (typeof reserved.service === 'string' ? reserved.service : undefined)
|
||||
|
||||
// Id normalization (8.0) — mirror of the committed transact() add
|
||||
// path: a natural key coerces to a STABLE UUID (v5), preserving the
|
||||
|
|
@ -747,12 +736,16 @@ export class Db<T = any> {
|
|||
`with(): entity ${updateId} not found at generation ${this.gen}`
|
||||
)
|
||||
}
|
||||
// Field-addressing law — mirror of the add case: the patch bag is
|
||||
// the user's verbatim; engine scalars only from dedicated op fields.
|
||||
const custom = { ...(op.metadata as Record<string, unknown> | undefined) }
|
||||
const confidence = op.confidence
|
||||
const weight = op.weight
|
||||
const subtype = op.subtype
|
||||
// Same reserved-field normalization as the committed update path.
|
||||
const { reserved, custom } = splitNounMetadataRecord(
|
||||
op.metadata as Record<string, unknown> | undefined
|
||||
)
|
||||
const confidence =
|
||||
op.confidence ?? (typeof reserved.confidence === 'number' ? reserved.confidence : undefined)
|
||||
const weight =
|
||||
op.weight ?? (typeof reserved.weight === 'number' ? reserved.weight : undefined)
|
||||
const subtype =
|
||||
op.subtype ?? (typeof reserved.subtype === 'string' ? reserved.subtype : undefined)
|
||||
const mergedMetadata =
|
||||
op.merge !== false
|
||||
? ({ ...(base.metadata as object), ...custom } as T)
|
||||
|
|
@ -814,14 +807,19 @@ export class Db<T = any> {
|
|||
}
|
||||
if (duplicate) break
|
||||
|
||||
// Field-addressing law — relationship mirror of the add case: the
|
||||
// edge bag is the user's verbatim; engine scalars only from
|
||||
// dedicated op fields.
|
||||
const custom = { ...(op.metadata as Record<string, unknown> | undefined) }
|
||||
const confidence = op.confidence
|
||||
const weight = op.weight
|
||||
const subtype = op.subtype
|
||||
const service = op.service
|
||||
// Reserved-field normalization — relationship mirror of the add
|
||||
// op above (and of the committed relate() path).
|
||||
const { reserved, custom } = splitVerbMetadataRecord(
|
||||
op.metadata as Record<string, unknown> | undefined
|
||||
)
|
||||
const confidence =
|
||||
op.confidence ?? (typeof reserved.confidence === 'number' ? reserved.confidence : undefined)
|
||||
const weight =
|
||||
op.weight ?? (typeof reserved.weight === 'number' ? reserved.weight : undefined)
|
||||
const subtype =
|
||||
op.subtype ?? (typeof reserved.subtype === 'string' ? reserved.subtype : undefined)
|
||||
const service =
|
||||
op.service ?? (typeof reserved.service === 'string' ? reserved.service : undefined)
|
||||
|
||||
const id = uuidv4()
|
||||
overlay.verbs.set(id, {
|
||||
|
|
@ -947,13 +945,6 @@ export class Db<T = any> {
|
|||
* {@link SpeculativeOverlayError} (commit them with `brain.transact()`
|
||||
* first).
|
||||
*
|
||||
* SPARSE FILES: a native accelerator's mmap index files can be sparse —
|
||||
* huge apparent size, small allocated size. `persist()` handles them
|
||||
* correctly (hard links share the allocation). But if you then archive the
|
||||
* snapshot with EXTERNAL tools, use the sparse-aware flags (`tar czSf`,
|
||||
* `rsync --sparse`, `cp --sparse=always`) or the copy materializes every
|
||||
* hole — see docs/guides/external-backups-and-sparse-storage.md.
|
||||
*
|
||||
* @param path - Absolute directory for the snapshot (created; must be
|
||||
* empty or absent).
|
||||
* @throws GenerationConflictError when this view is no longer the latest
|
||||
|
|
|
|||
|
|
@ -23,12 +23,8 @@
|
|||
* serve the full query surface via at-generation index materialization.
|
||||
* - {@link GenerationCompactedError} — `asOf()` asked for a generation whose
|
||||
* immutable records were reclaimed by `compactHistory()`.
|
||||
* - {@link CanonicalEnumerationUnavailableError} — `export()`'s
|
||||
* `enumeration:'canonical'` mode was called on a historical `asOf()` view or a
|
||||
* speculative `with()` overlay; the canonical storage walk only ever answers
|
||||
* "what is live right now."
|
||||
*
|
||||
* All are exported from the package root (`@soulcraftlabs/brainy`).
|
||||
* All three are exported from the package root (`@soulcraft/brainy`).
|
||||
*/
|
||||
|
||||
/**
|
||||
|
|
@ -164,64 +160,6 @@ export class GenerationCompactedError extends Error {
|
|||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* @description Thrown by `db.export(selector, { enumeration: 'canonical' })` when
|
||||
* the `Db` it is called on is not the live, current-generation view: a historical
|
||||
* `brain.asOf(g)` pin, or a speculative `db.with()` overlay.
|
||||
*
|
||||
* Canonical enumeration mode walks the storage adapter's canonical shard layout
|
||||
* directly (`storage.getNouns()`/`getVerbs()`) instead of the metadata/graph
|
||||
* indexes — but that walk has no generation parameter, it can only ever answer
|
||||
* "what is live right now." Serving it against a historical pin would silently
|
||||
* mix generations (today's canonical records under yesterday's selector), and
|
||||
* against a speculative overlay it would silently miss the overlay's own
|
||||
* in-memory entities (which never touched storage). Both are exactly the kind of
|
||||
* silently-wrong result canonical mode exists to prevent elsewhere — so this
|
||||
* boundary throws instead.
|
||||
*
|
||||
* `enumeration: 'index'` (the default) is unaffected: it composes with
|
||||
* `asOf()`/`with()` exactly as before, via the generation-correct `find()` walk.
|
||||
*
|
||||
* @example
|
||||
* const past = await brain.asOf(g1)
|
||||
* try {
|
||||
* await past.export({}, { enumeration: 'canonical' })
|
||||
* } catch (err) {
|
||||
* if (err instanceof CanonicalEnumerationUnavailableError) {
|
||||
* // Time-travel export: use the default index-based enumeration instead.
|
||||
* await past.export({}, { enumeration: 'index' })
|
||||
* }
|
||||
* }
|
||||
*/
|
||||
export class CanonicalEnumerationUnavailableError extends Error {
|
||||
/** The view's pinned generation. */
|
||||
public readonly generation: number
|
||||
/** Why canonical mode cannot serve this view. */
|
||||
public readonly reason: 'historical' | 'overlay'
|
||||
|
||||
/**
|
||||
* @param generation - The view's pinned generation.
|
||||
* @param reason - `'historical'` (a past `asOf()` pin) or `'overlay'` (a speculative `with()`).
|
||||
*/
|
||||
constructor(generation: number, reason: 'historical' | 'overlay') {
|
||||
const what =
|
||||
reason === 'historical'
|
||||
? `a historical view pinned at generation ${generation}`
|
||||
: `a speculative with() overlay (base generation ${generation})`
|
||||
super(
|
||||
`export()'s enumeration:'canonical' requires the live, current-generation view — ` +
|
||||
`it was called on ${what}. The canonical storage walk has no generation parameter, ` +
|
||||
`so it can only answer "what is live right now"; serving it here would silently ` +
|
||||
`mix generations (historical) or miss the overlay's own in-memory entities ` +
|
||||
`(overlay). Use enumeration:'index' (the default) for a time-travel or what-if ` +
|
||||
`export, or pin brain.now() for a live canonical export.`
|
||||
)
|
||||
this.name = 'CanonicalEnumerationUnavailableError'
|
||||
this.generation = generation
|
||||
this.reason = reason
|
||||
}
|
||||
}
|
||||
|
||||
/** One entity/relationship left in an unreconciled state by a failed rollback. */
|
||||
export interface UnreconciledRecord {
|
||||
/** The entity or relationship id. */
|
||||
|
|
|
|||
1532
src/db/factLog.ts
1532
src/db/factLog.ts
File diff suppressed because it is too large
Load diff
File diff suppressed because it is too large
Load diff
|
|
@ -1,166 +0,0 @@
|
|||
/**
|
||||
* @module db/familyStamp
|
||||
* @description The generalized FAMILY STAMP — one JSON shape that declares,
|
||||
* for any derived projection, WHICH source state it reflects and HOW to verify
|
||||
* it is whole. The entity tree (canonical current-state files) carries the
|
||||
* first brainy-side stamp; native index families carry the same shape. One
|
||||
* verifier reads both member modes:
|
||||
*
|
||||
* - `enumerated` — bounded families: exact byte size per member file,
|
||||
* verified at open.
|
||||
* - `rollup` — unbounded families (the entity tree: millions of files):
|
||||
* the verified surface is a small set of rollup invariants (entity/
|
||||
* relationship counts) plus `sourceGeneration`.
|
||||
*
|
||||
* `sourceGeneration` is the COMMITTED generation of the source-of-truth log
|
||||
* this projection reflects — never the allocated counter, which names a
|
||||
* generation that may never commit (see {@link StampVerdict.torn}) — so
|
||||
* open-time coherence becomes a COMPARISON (stamp vs committed head), not a
|
||||
* walk:
|
||||
*
|
||||
* - equal + invariants hold → coherent, serve.
|
||||
* - behind → the projection missed the tail (crash between commit and stamp);
|
||||
* for the Stage-1 tree this is benign by construction (the tree is written
|
||||
* BY the commit), so the stamp refreshes; a DERIVED projection would replay
|
||||
* the gap instead.
|
||||
* - invariants FAIL at equal generation → genuine incoherence: loud, and the
|
||||
* repair ritual (`repairIndex()`, whose recount rebuilds the rollups from a
|
||||
* canonical walk) heals it.
|
||||
* - AHEAD → a torn generation-log tail: the stamp's fsync outlived the log
|
||||
* tail's. TERMINAL, never a wait — the generation the stamp names does not
|
||||
* exist to arrive.
|
||||
*
|
||||
* Stamps are JSON on purpose — every incident gets debugged by reading a
|
||||
* stamp in a terminal.
|
||||
*/
|
||||
|
||||
/** Storage-root-relative directory holding family stamps. */
|
||||
export const FAMILY_STAMPS_PREFIX = '_system/family-stamps'
|
||||
|
||||
/** The entity tree's stamp path. */
|
||||
export const ENTITY_TREE_STAMP_PATH = `${FAMILY_STAMPS_PREFIX}/entity-tree.json`
|
||||
|
||||
/** One enumerated member: a file and its exact expected byte size. */
|
||||
export interface EnumeratedMember {
|
||||
path: string
|
||||
bytes: number
|
||||
}
|
||||
|
||||
/**
|
||||
* The stamp's verified surface, in one of the two member modes. Rollup
|
||||
* invariant values may be numbers (counts, byte sizes) or strings (content
|
||||
* fingerprints, e.g. a per-tree SHA-256) — the verifier compares by strict
|
||||
* equality either way, so a type mismatch reads as incoherence, never a pass.
|
||||
*/
|
||||
export type StampMembers =
|
||||
| { mode: 'enumerated'; files: EnumeratedMember[] }
|
||||
| { mode: 'rollup'; invariants: Record<string, number | string> }
|
||||
|
||||
/** The generalized family stamp (one shape, one verifier, both engines). */
|
||||
export interface FamilyStamp {
|
||||
/** Which projection this stamps (e.g. `'entity-tree'`). */
|
||||
family: string
|
||||
/** Monotonic per-family stamp generation — bumps on every committed stamp. */
|
||||
generation: number
|
||||
/** ISO timestamp of the stamp write. */
|
||||
committedAt: string
|
||||
/** The source-of-truth generation this projection reflects. */
|
||||
sourceGeneration: number
|
||||
/** The verified surface. */
|
||||
members: StampMembers
|
||||
}
|
||||
|
||||
/** The verdict of an open-time stamp verification. */
|
||||
export type StampVerdict =
|
||||
| { state: 'coherent' }
|
||||
| { state: 'absent' } // legacy store — first stamp writes at the next flush
|
||||
| { state: 'behind'; stampSource: number; head: number }
|
||||
/**
|
||||
* TORN GENERATION-LOG TAIL: the stamp witnesses a source generation the
|
||||
* store's committed watermark can no longer show. TERMINAL — there is no
|
||||
* generation to wait for, so the open demotes (or refuses) and never spins.
|
||||
*/
|
||||
| { state: 'torn'; stampSource: number; head: number }
|
||||
| { state: 'incoherent'; failures: string[] }
|
||||
| { state: 'unverifiable'; reason: string } // a FAULT reading the stamp — never conflated with absence
|
||||
|
||||
/** The narrow storage surface stamps ride (JSON objects + fsync). */
|
||||
export interface StampStorage {
|
||||
readRawObject(path: string): Promise<any | null>
|
||||
writeRawObject(path: string, data: any): Promise<void>
|
||||
syncRawObjects(paths: string[]): Promise<void>
|
||||
}
|
||||
|
||||
/** Read a family's stamp; `null` when none was ever written. */
|
||||
export async function readFamilyStamp(
|
||||
storage: StampStorage,
|
||||
path: string
|
||||
): Promise<FamilyStamp | null> {
|
||||
const stored = (await storage.readRawObject(path)) as FamilyStamp | null
|
||||
if (!stored || typeof stored !== 'object' || typeof stored.family !== 'string') return null
|
||||
return stored
|
||||
}
|
||||
|
||||
/** Write a family's stamp durably (atomic object write + fsync). */
|
||||
export async function writeFamilyStamp(
|
||||
storage: StampStorage,
|
||||
path: string,
|
||||
stamp: Omit<FamilyStamp, 'generation' | 'committedAt'> & { generation?: number }
|
||||
): Promise<void> {
|
||||
const prior = await readFamilyStamp(storage, path)
|
||||
const full: FamilyStamp = {
|
||||
...stamp,
|
||||
generation: (prior?.generation ?? 0) + 1,
|
||||
committedAt: new Date().toISOString()
|
||||
}
|
||||
await storage.writeRawObject(path, full)
|
||||
await storage.syncRawObjects([path])
|
||||
}
|
||||
|
||||
/**
|
||||
* The ONE verifier, both member modes. `actual` supplies the observed rollup
|
||||
* values (rollup mode) or file sizes (enumerated mode, keyed by path);
|
||||
* `head` is the source-of-truth generation now.
|
||||
*/
|
||||
export function verifyFamilyStamp(
|
||||
stamp: FamilyStamp | null,
|
||||
head: number,
|
||||
actual: Record<string, number | string>
|
||||
): StampVerdict {
|
||||
if (stamp === null) return { state: 'absent' }
|
||||
if (stamp.sourceGeneration > head) {
|
||||
// A stamp AHEAD of committed truth witnesses a generation the store can no
|
||||
// longer show: the stamp's fsync survived a crash that the log tail did
|
||||
// not. This is the TORN GENERATION-LOG TAIL — its own class, never folded
|
||||
// in with `incoherent` (a count that drifted at a generation both sides
|
||||
// agree on), because the two have opposite cures: incoherence is recounted,
|
||||
// a tear is DEMOTED. It is also terminal by construction — there is no
|
||||
// generation the open can wait for, because the one the stamp names is
|
||||
// gone.
|
||||
return { state: 'torn', stampSource: stamp.sourceGeneration, head }
|
||||
}
|
||||
if (stamp.sourceGeneration < head) {
|
||||
return { state: 'behind', stampSource: stamp.sourceGeneration, head }
|
||||
}
|
||||
const failures: string[] = []
|
||||
if (stamp.members.mode === 'rollup') {
|
||||
for (const [name, expected] of Object.entries(stamp.members.invariants)) {
|
||||
const observed = actual[name]
|
||||
if (observed === undefined) {
|
||||
failures.push(`rollup invariant '${name}' has no observed value`)
|
||||
} else if (observed !== expected) {
|
||||
failures.push(`rollup invariant '${name}': stamped ${expected}, observed ${observed}`)
|
||||
}
|
||||
}
|
||||
} else {
|
||||
for (const member of stamp.members.files) {
|
||||
const observed = actual[member.path]
|
||||
if (observed === undefined) {
|
||||
failures.push(`member '${member.path}' is missing`)
|
||||
} else if (observed !== member.bytes) {
|
||||
failures.push(`member '${member.path}': stamped ${member.bytes} bytes, observed ${observed}`)
|
||||
}
|
||||
}
|
||||
}
|
||||
return failures.length > 0 ? { state: 'incoherent', failures } : { state: 'coherent' }
|
||||
}
|
||||
|
|
@ -1,164 +0,0 @@
|
|||
/**
|
||||
* @module db/faultInjectionStorage
|
||||
* @description Deterministic fault injection at the fact log's raw-byte
|
||||
* storage surface — the test harness half of the durability protocol. Wraps
|
||||
* any adapter exposing the {@link FactLogStorage} primitives (the exact
|
||||
* surface the fact log appends and syncs through) and injects the three
|
||||
* crash shapes durability tests must prove against:
|
||||
*
|
||||
* - **torn write** ({@link FaultInjectionStorage.tearWriteAtByte}): the next
|
||||
* append persists only its first N bytes, then reports success — the shape
|
||||
* of power loss after a partially-flushed page. The caller-side "crash" is
|
||||
* simulated by abandoning in-memory state and reopening from storage.
|
||||
* - **dropped sync** ({@link FaultInjectionStorage.dropNextSync}): the next
|
||||
* sync becomes a silent no-op — an fsync the device acknowledged into a
|
||||
* volatile cache and lost.
|
||||
* - **failed append** ({@link FaultInjectionStorage.failNextAppend}): the next
|
||||
* append throws {@link FaultInjectedError} without writing a byte — EIO or
|
||||
* a full disk, surfaced to the writer.
|
||||
*
|
||||
* Every injected fault is journaled on {@link FaultInjectionStorage.injectedFaults}
|
||||
* so tests can assert not just the outcome but that the fault actually fired.
|
||||
* Knobs are one-shot (they disarm on firing) and re-arming overwrites the
|
||||
* pending shot. All other operations pass through untouched.
|
||||
*/
|
||||
import type { FactLogStorage } from './factLog.js'
|
||||
|
||||
/** The error a {@link FaultInjectionStorage.failNextAppend} shot throws. */
|
||||
export class FaultInjectedError extends Error {
|
||||
/** The operation the fault fired on. */
|
||||
public readonly operation: 'append'
|
||||
/** The storage path the operation targeted. */
|
||||
public readonly path: string
|
||||
|
||||
constructor(operation: 'append', path: string) {
|
||||
super(`fault injection: ${operation} to ${path} failed by test design`)
|
||||
this.name = 'FaultInjectedError'
|
||||
this.operation = operation
|
||||
this.path = path
|
||||
}
|
||||
}
|
||||
|
||||
/** One journaled fault event — proof the injected fault actually fired. */
|
||||
export interface InjectedFault {
|
||||
kind: 'torn-write' | 'dropped-sync' | 'failed-append'
|
||||
/** The target path (torn-write / failed-append). */
|
||||
path?: string
|
||||
/** The paths a dropped sync was asked to make durable. */
|
||||
paths?: string[]
|
||||
/** Bytes the caller asked to append (torn-write). */
|
||||
requestedBytes?: number
|
||||
/** Bytes actually persisted (torn-write). */
|
||||
writtenBytes?: number
|
||||
}
|
||||
|
||||
/**
|
||||
* A {@link FactLogStorage} wrapper that injects deterministic storage faults.
|
||||
* Construct it around any conforming adapter and hand it wherever a
|
||||
* FactLogStorage is accepted — unarmed, it is a transparent passthrough.
|
||||
*/
|
||||
export class FaultInjectionStorage implements FactLogStorage {
|
||||
private readonly inner: FactLogStorage
|
||||
/** Pending torn-write byte count, or null when unarmed. */
|
||||
private tearAtByte: number | null = null
|
||||
/** Pending dropped-sync shot. */
|
||||
private dropSyncArmed = false
|
||||
/** Pending failed-append shot. */
|
||||
private failAppendArmed = false
|
||||
/** Journal of every fault that fired, in firing order. */
|
||||
public readonly injectedFaults: InjectedFault[] = []
|
||||
|
||||
constructor(inner: FactLogStorage) {
|
||||
this.inner = inner
|
||||
}
|
||||
|
||||
/**
|
||||
* Arm a torn write: the NEXT {@link appendRawBytes} persists only the first
|
||||
* `n` bytes of its buffer (all of it when `n` exceeds the buffer) and then
|
||||
* reports success. One-shot.
|
||||
*/
|
||||
tearWriteAtByte(n: number): void {
|
||||
if (!Number.isInteger(n) || n < 0) {
|
||||
throw new Error(`fault injection: tearWriteAtByte needs a non-negative integer; got ${n}`)
|
||||
}
|
||||
this.tearAtByte = n
|
||||
}
|
||||
|
||||
/** Arm a dropped sync: the NEXT {@link syncRawObjects} silently does nothing. One-shot. */
|
||||
dropNextSync(): void {
|
||||
this.dropSyncArmed = true
|
||||
}
|
||||
|
||||
/**
|
||||
* Arm a failed append: the NEXT {@link appendRawBytes} throws
|
||||
* {@link FaultInjectedError} without writing. One-shot; wins over a
|
||||
* simultaneously-armed torn write (nothing is written at all).
|
||||
*/
|
||||
failNextAppend(): void {
|
||||
this.failAppendArmed = true
|
||||
}
|
||||
|
||||
/** Append bytes — the injection point for torn writes and failed appends. */
|
||||
async appendRawBytes(path: string, bytes: Uint8Array): Promise<void> {
|
||||
if (this.failAppendArmed) {
|
||||
this.failAppendArmed = false
|
||||
this.injectedFaults.push({ kind: 'failed-append', path })
|
||||
throw new FaultInjectedError('append', path)
|
||||
}
|
||||
if (this.tearAtByte !== null) {
|
||||
const writtenBytes = Math.min(this.tearAtByte, bytes.length)
|
||||
this.tearAtByte = null
|
||||
this.injectedFaults.push({
|
||||
kind: 'torn-write',
|
||||
path,
|
||||
requestedBytes: bytes.length,
|
||||
writtenBytes
|
||||
})
|
||||
if (writtenBytes > 0) {
|
||||
await this.inner.appendRawBytes(path, bytes.subarray(0, writtenBytes))
|
||||
}
|
||||
return
|
||||
}
|
||||
return this.inner.appendRawBytes(path, bytes)
|
||||
}
|
||||
|
||||
/** Make paths durable — the injection point for dropped syncs. */
|
||||
async syncRawObjects(paths: string[]): Promise<void> {
|
||||
if (this.dropSyncArmed) {
|
||||
this.dropSyncArmed = false
|
||||
this.injectedFaults.push({ kind: 'dropped-sync', paths: [...paths] })
|
||||
return
|
||||
}
|
||||
return this.inner.syncRawObjects(paths)
|
||||
}
|
||||
|
||||
/** Passthrough. */
|
||||
async readRawBytes(path: string): Promise<Uint8Array | null> {
|
||||
return this.inner.readRawBytes(path)
|
||||
}
|
||||
|
||||
/** Passthrough. */
|
||||
async writeRawBytes(path: string, bytes: Uint8Array): Promise<void> {
|
||||
return this.inner.writeRawBytes(path, bytes)
|
||||
}
|
||||
|
||||
/** Passthrough. */
|
||||
async rawByteSize(path: string): Promise<number | null> {
|
||||
return this.inner.rawByteSize(path)
|
||||
}
|
||||
|
||||
/** Passthrough. */
|
||||
async readRawObject(path: string): Promise<any | null> {
|
||||
return this.inner.readRawObject(path)
|
||||
}
|
||||
|
||||
/** Passthrough. */
|
||||
async writeRawObject(path: string, data: any): Promise<void> {
|
||||
return this.inner.writeRawObject(path, data)
|
||||
}
|
||||
|
||||
/** Passthrough. */
|
||||
async deleteRawObject(path: string): Promise<void> {
|
||||
return this.inner.deleteRawObject(path)
|
||||
}
|
||||
}
|
||||
|
|
@ -1,345 +0,0 @@
|
|||
/**
|
||||
* @module db/fieldAddressing
|
||||
* @description The one field-addressing law for every query surface (find()'s
|
||||
* `where` / `orderBy` / `groupBy`, aggregation `source.where`), ruled
|
||||
* 2026-08-03 after a production incident in which a user metadata field
|
||||
* named `level` was silently shadowed by the engine's internal HNSW node
|
||||
* layer (VENUE-BRAINY-ORDERBY-NOOP — thread id kept verbatim as the audit
|
||||
* key; it names no product):
|
||||
*
|
||||
* 1. A BARE field name addresses the user's metadata field. Always.
|
||||
* No priority resolution, no fallback chain — `orderBy: 'level'`
|
||||
* reads `entity.metadata.level`, full stop.
|
||||
* 2. `system.<field>` addresses an engine scalar, reachable ONLY with the
|
||||
* explicit prefix. The entity map is exactly ten scalars; the relation
|
||||
* map mirrors it with `verb`/`sourceId`/`targetId` as the structural
|
||||
* members.
|
||||
* 3. Engine plumbing (`vector`, `connections`, `level`, `data`, `_rev`) is
|
||||
* INVISIBLE to the query surface in either spelling — `system.level`
|
||||
* refuses; bare `level` is the user's field.
|
||||
* 4. `metadata.<field>` is the explicit spelling of the bare form —
|
||||
* identical semantics on every path.
|
||||
* 5. Anything unresolvable refuses with a TYPED error naming both
|
||||
* candidate spellings — an accepted name either works or refuses;
|
||||
* there is no third state.
|
||||
*
|
||||
* This module is the SINGLE source of truth for the law: parsing, the maps,
|
||||
* and the refusal builders live here so the JS engine, the provider seams,
|
||||
* and the cross-engine conformance suite can never drift on the contract.
|
||||
*/
|
||||
|
||||
import type { HNSWNounWithMetadata, HNSWVerbWithMetadata } from '../coreTypes.js'
|
||||
|
||||
/**
|
||||
* @description The entity-side `system.*` map — EXACTLY the ten engine
|
||||
* scalars David ruled queryable (2026-08-03). Adding a name here is a
|
||||
* cross-engine contract change: the native accelerator's conformance suite
|
||||
* pins this list verbatim, so any edit must ship as a paired release.
|
||||
*/
|
||||
export const SYSTEM_ENTITY_SCALARS: ReadonlySet<string> = new Set([
|
||||
'id',
|
||||
'type',
|
||||
'subtype',
|
||||
'createdAt',
|
||||
'updatedAt',
|
||||
'confidence',
|
||||
'weight',
|
||||
'visibility',
|
||||
'service',
|
||||
'createdBy'
|
||||
])
|
||||
|
||||
/**
|
||||
* @description The relation-side `system.*` map — the verb mirror of
|
||||
* {@link SYSTEM_ENTITY_SCALARS}: `verb`, `sourceId`, `targetId` are the
|
||||
* structural members beside the eight shared scalars. Same one law, same
|
||||
* pairing rule for edits.
|
||||
*/
|
||||
export const SYSTEM_RELATION_SCALARS: ReadonlySet<string> = new Set([
|
||||
'verb',
|
||||
'sourceId',
|
||||
'targetId',
|
||||
'subtype',
|
||||
'createdAt',
|
||||
'updatedAt',
|
||||
'confidence',
|
||||
'weight',
|
||||
'visibility',
|
||||
'service',
|
||||
'createdBy'
|
||||
])
|
||||
|
||||
/**
|
||||
* @description Engine plumbing — never addressable from the query surface in
|
||||
* ANY spelling. `level` is the HNSW node layer (the incident field: listing
|
||||
* it as resolvable shadowed real user data); `data` is the payload container,
|
||||
* not a scalar — content is reached through the content/text-search APIs,
|
||||
* and addressing it as a sortable field would lie about its shape.
|
||||
*/
|
||||
export const PLUMBING_FIELDS: ReadonlySet<string> = new Set([
|
||||
'vector',
|
||||
'connections',
|
||||
'level',
|
||||
'data',
|
||||
'_rev'
|
||||
])
|
||||
|
||||
/** @description Which record kind a field address is being resolved against. */
|
||||
export type FieldAddressKind = 'entity' | 'relation'
|
||||
|
||||
/**
|
||||
* @description A parsed, law-valid field address. `scope` says which side of
|
||||
* the record the name lives on; `field` is the unprefixed name to read.
|
||||
*/
|
||||
export interface FieldAddress {
|
||||
/** 'metadata' = the user's field (bare or `metadata.`-prefixed); 'system' = an engine scalar. */
|
||||
scope: 'metadata' | 'system'
|
||||
/** The field name with any scope prefix removed. */
|
||||
field: string
|
||||
/** The exact spelling the caller used — preserved for error text and telemetry. */
|
||||
raw: string
|
||||
}
|
||||
|
||||
/**
|
||||
* Parse a query-surface field name under the one law. Pure and data-blind:
|
||||
* this validates the ADDRESS (spelling + map membership), not whether any
|
||||
* row actually carries the field — data-aware refusals (the did-you-mean
|
||||
* for a bare system-scalar name no row carries) belong to the query layer,
|
||||
* which calls {@link buildUnresolvableMessage} with index knowledge.
|
||||
*
|
||||
* @param raw - The field name as the caller wrote it (`level`,
|
||||
* `metadata.level`, `system.createdAt`, …)
|
||||
* @param kind - Entity or relation resolution (selects the system map)
|
||||
* @returns The parsed {@link FieldAddress}
|
||||
* @throws {InvalidFieldAddressError} for a `system.*` name outside the ruled
|
||||
* map (including every plumbing field) or a malformed spelling — the error
|
||||
* text enumerates the valid system scalars so the fix is in the message.
|
||||
*
|
||||
* @example
|
||||
* parseFieldAddress('level', 'entity') // { scope: 'metadata', field: 'level' }
|
||||
* parseFieldAddress('metadata.level', 'entity') // { scope: 'metadata', field: 'level' }
|
||||
* parseFieldAddress('system.createdAt', 'entity') // { scope: 'system', field: 'createdAt' }
|
||||
* parseFieldAddress('system.level', 'entity') // throws — plumbing is invisible
|
||||
*/
|
||||
export function parseFieldAddress(
|
||||
raw: string,
|
||||
kind: FieldAddressKind
|
||||
): FieldAddress {
|
||||
const systemMap =
|
||||
kind === 'entity' ? SYSTEM_ENTITY_SCALARS : SYSTEM_RELATION_SCALARS
|
||||
|
||||
if (raw.startsWith('system.')) {
|
||||
const field = raw.slice('system.'.length)
|
||||
if (!systemMap.has(field)) {
|
||||
throw new InvalidFieldAddressError(raw, kind, systemMap)
|
||||
}
|
||||
return { scope: 'system', field, raw }
|
||||
}
|
||||
|
||||
if (raw.startsWith('metadata.')) {
|
||||
const field = raw.slice('metadata.'.length)
|
||||
if (field.length === 0) {
|
||||
throw new InvalidFieldAddressError(raw, kind, systemMap)
|
||||
}
|
||||
return { scope: 'metadata', field, raw }
|
||||
}
|
||||
|
||||
if (raw.length === 0) {
|
||||
throw new InvalidFieldAddressError(raw, kind, systemMap)
|
||||
}
|
||||
|
||||
// Bare name = the user's metadata field. Always. Even when the same name
|
||||
// exists in the system map — `confidence` as a bare name is the user's
|
||||
// metadata field named confidence; the engine scalar is system.confidence.
|
||||
return { scope: 'metadata', field: raw, raw }
|
||||
}
|
||||
|
||||
/**
|
||||
* Read the addressed value off an entity. The ONLY sanctioned way a query
|
||||
* surface turns a {@link FieldAddress} into a value — direct property reads
|
||||
* against records re-create the shadow class this module exists to kill.
|
||||
*
|
||||
* @returns The value, or `undefined` when the record does not carry it
|
||||
* (missing values sort LAST in both directions per the ordering contract —
|
||||
* they are never grounds for dropping a row).
|
||||
*/
|
||||
export function readEntityFieldAddress(
|
||||
entity: HNSWNounWithMetadata,
|
||||
address: FieldAddress
|
||||
): unknown {
|
||||
const rec = entity as unknown as Record<string, unknown>
|
||||
const bag =
|
||||
rec.metadata && typeof rec.metadata === 'object'
|
||||
? (rec.metadata as Record<string, unknown>)
|
||||
: null
|
||||
|
||||
if (address.scope === 'system') {
|
||||
// System scalars live at the record's top level, NEVER in the user's
|
||||
// bag — a user field named `confidence` must be unreachable from
|
||||
// system.confidence (and vice versa). Entity views carry the scalars
|
||||
// top-level directly; record-derived views spell the type `noun`.
|
||||
const top = rec[address.field]
|
||||
if (top !== undefined) return top
|
||||
if (address.field === 'type') return rec.noun
|
||||
return undefined
|
||||
}
|
||||
|
||||
// User scope: the bag IS the user's namespace, authoritative — EVERY name
|
||||
// reads from it, engine spellings included (`bag.confidence` is the user's
|
||||
// confidence field under the field-addressing law).
|
||||
if (bag) return bag[address.field]
|
||||
|
||||
// No bag at all: a LEGACY flat record (pre-nested-bag storage). Its keys
|
||||
// matching system/plumbing names are the ENGINE's — the pre-law write door
|
||||
// refused user colliders — so a bare system name reads as ABSENT rather
|
||||
// than resurrecting the shadow this module exists to kill. Same for the
|
||||
// legacy 'noun' spelling.
|
||||
if (
|
||||
SYSTEM_ENTITY_SCALARS.has(address.field) ||
|
||||
PLUMBING_FIELDS.has(address.field) ||
|
||||
address.field === 'noun'
|
||||
) {
|
||||
return undefined
|
||||
}
|
||||
return rec[address.field]
|
||||
}
|
||||
|
||||
/**
|
||||
* Relation twin of {@link readEntityFieldAddress}. The stored flat record
|
||||
* keys the relation type under `verb`; public Relation shapes may carry it
|
||||
* as `type` — both spellings of the record are read, the ADDRESS is always
|
||||
* `system.verb`.
|
||||
*/
|
||||
export function readRelationFieldAddress(
|
||||
verb: HNSWVerbWithMetadata,
|
||||
address: FieldAddress
|
||||
): unknown {
|
||||
if (address.scope === 'system') {
|
||||
const rec = verb as unknown as Record<string, unknown>
|
||||
if (address.field === 'verb') return rec.verb ?? rec.type
|
||||
return rec[address.field]
|
||||
}
|
||||
return verb.metadata?.[address.field]
|
||||
}
|
||||
|
||||
/**
|
||||
* Build the ruled did-you-mean refusal text for a bare name that resolved to
|
||||
* metadata but is UNKNOWN to the index — the data-aware half of the law,
|
||||
* called by the query layer once it has consulted the known-field set:
|
||||
*
|
||||
* "no metadata field 'createdAt' — did you mean system.createdAt or
|
||||
* metadata.createdAt?"
|
||||
*
|
||||
* When the bare name is NOT a system scalar the system candidate is omitted
|
||||
* (there is only one thing the caller could have meant; the refusal exists
|
||||
* because refusing beats silently sorting nothing).
|
||||
*/
|
||||
export function buildUnresolvableMessage(
|
||||
raw: string,
|
||||
kind: FieldAddressKind
|
||||
): string {
|
||||
const systemMap =
|
||||
kind === 'entity' ? SYSTEM_ENTITY_SCALARS : SYSTEM_RELATION_SCALARS
|
||||
if (systemMap.has(raw)) {
|
||||
return (
|
||||
`no metadata field '${raw}' — did you mean system.${raw} or metadata.${raw}? ` +
|
||||
`(bare names always address your metadata; engine fields need the system. prefix)`
|
||||
)
|
||||
}
|
||||
return (
|
||||
`no metadata field '${raw}' on this store — nothing carries it, so an ordered or ` +
|
||||
`filtered read against it cannot mean anything. Spell it metadata.${raw} once the ` +
|
||||
`field exists, or check the field name (system.${raw} is NOT valid — '${raw}' is ` +
|
||||
`not one of the engine's system scalars).`
|
||||
)
|
||||
}
|
||||
|
||||
/**
|
||||
* @description Refusal for a syntactically valid address that resolves to
|
||||
* NOTHING — a bare name no user field carries. Carries the did-you-mean
|
||||
* (both candidate spellings when the name collides with a system scalar) so
|
||||
* the fix ships inside the error. Thrown by the query layer with index
|
||||
* knowledge, never by the pure parser.
|
||||
*/
|
||||
/**
|
||||
* Cross-package identity normalizer (the seam belt): the native accelerator
|
||||
* throws ITS OWN UnresolvableFieldError class, which fails `instanceof`
|
||||
* against this package's export — consumers were forced to match by name.
|
||||
* Every provider-boundary catch routes suspected field-refusals through
|
||||
* here: a foreign refusal (matched by name, duck fields tolerated) is
|
||||
* rethrown as THIS package's class, so exactly one identity ever reaches
|
||||
* consumers. Anything else returns null (caller rethrows the original).
|
||||
*/
|
||||
export function asBrainyFieldRefusal(err: unknown): UnresolvableFieldError | null {
|
||||
if (err instanceof UnresolvableFieldError) return err
|
||||
const e = err as { name?: string; message?: string; raw?: string; kind?: string } | null
|
||||
if (e && e.name === 'UnresolvableFieldError') {
|
||||
return new UnresolvableFieldError(
|
||||
e.raw ?? 'unknown-field',
|
||||
(e.kind as FieldAddressKind) ?? 'entity',
|
||||
e.message
|
||||
)
|
||||
}
|
||||
return null
|
||||
}
|
||||
|
||||
export class UnresolvableFieldError extends Error {
|
||||
public readonly raw: string
|
||||
public readonly kind: FieldAddressKind
|
||||
|
||||
constructor(raw: string, kind: FieldAddressKind, messageOverride?: string) {
|
||||
super(messageOverride ?? buildUnresolvableMessage(raw, kind))
|
||||
this.name = 'UnresolvableFieldError'
|
||||
this.raw = raw
|
||||
this.kind = kind
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* @description Refusal for a malformed or out-of-map field ADDRESS —
|
||||
* `system.<anything-not-in-the-map>` (including all plumbing), an empty
|
||||
* name, or a bare `metadata.` prefix. The message carries the full valid
|
||||
* system map so the fix never needs a docs lookup.
|
||||
*/
|
||||
export class InvalidFieldAddressError extends UnresolvableFieldError {
|
||||
constructor(raw: string, kind: FieldAddressKind, systemMap: ReadonlySet<string>) {
|
||||
const valid = [...systemMap].map((f) => `system.${f}`).join(', ')
|
||||
super(
|
||||
raw,
|
||||
kind,
|
||||
`'${raw}' is not an addressable ${kind} field. Bare names address your own ` +
|
||||
`metadata fields; engine fields are exactly: ${valid}. Engine plumbing ` +
|
||||
`(vector, connections, level, data, _rev) is not part of the query surface.`
|
||||
)
|
||||
this.name = 'InvalidFieldAddressError'
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* @description Refusal for a find() option that is accepted by the type
|
||||
* surface but NOT implemented — an accepted option must work or refuse;
|
||||
* accepted-and-ignored died as a class (sealed 2026-08-03). Names the
|
||||
* option and the honest state so nobody discovers a no-op by measurement.
|
||||
*/
|
||||
export class UnsupportedFindOptionError extends Error {
|
||||
public readonly option: string
|
||||
|
||||
constructor(option: string) {
|
||||
super(
|
||||
`find() option '${option}' is not implemented — it used to be silently ` +
|
||||
`ignored, which read as working. Remove it from the call (or track the ` +
|
||||
`feature request); it will be honored or refused, never swallowed.`
|
||||
)
|
||||
this.name = 'UnsupportedFindOptionError'
|
||||
this.option = option
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* @description The capability signal both engines' conformance suites arm on
|
||||
* (never a version guess): its presence at the package root means the one
|
||||
* field-addressing law is LIVE on every query surface — bare = user metadata,
|
||||
* `system.*` = the ruled scalars, plumbing invisible, refusals typed.
|
||||
*/
|
||||
export const FIELD_ADDRESSING_CAPABILITY = 'field-addressing/v1'
|
||||
|
|
@ -1,570 +0,0 @@
|
|||
/**
|
||||
* @module db/generationSegments
|
||||
* @description The generation-segment store — Stage-2 D1+D3+repacking's file
|
||||
* format (co-frozen 2026-07-19; design: the d1-d3-repacking spec).
|
||||
*
|
||||
* Packs CONSECUTIVE cold generations' record-sets (before-images + delta)
|
||||
* into append-once segment files with derived sidecar indexes, so history
|
||||
* scales in SEGMENTS (tens) instead of FILES-PER-GENERATION (hundreds of
|
||||
* thousands), and cold-open reads ONE manifest instead of listing the
|
||||
* backlog. Layout under `_generations/segments/`:
|
||||
*
|
||||
* - `seg-<firstGen, zero-padded 20>.bgs` — magic "BGS1", then one frame per
|
||||
* generation: `u32 payloadLen | u32 crc32c | msgpack payload`. Payload is
|
||||
* POSITIONAL: `[generation, timestamp, delta, records[], flags]` with
|
||||
* records `[kindByte, id, record]`. `flags` reserves encoding evolution
|
||||
* (bit 0 = compressed payload — v1 always 0; a future writer upgrade,
|
||||
* never a format break). Sealed segments are IMMUTABLE — the fact log's
|
||||
* own law, generalized.
|
||||
* - `seg-<firstGen>.idx` — DERIVED sidecar (msgpack): per-generation frame
|
||||
* offsets (point reads = one ranged read, never a listing) + per-id
|
||||
* generation postings (per-id chain rebuilds read only what they need).
|
||||
* Corrupt/missing → rebuilt from its segment in one sequential read,
|
||||
* loudly.
|
||||
* - `manifest.json` — the segment catalogue + `compactedBelow` (D3's
|
||||
* horizon marker). Cold-open reads THIS; the packed backlog is never
|
||||
* listed.
|
||||
*
|
||||
* D3 semantics carried here: bounded-retention reclaim drops WHOLE segments
|
||||
* at boundaries (O(1) per segment, no rewrite); under the archival profile
|
||||
* (`retention: 'all'`) nothing here is ever dropped — folding is the only
|
||||
* transform (re-representation, never deletion).
|
||||
*/
|
||||
|
||||
import { encode as msgpackEncode, decode as msgpackDecode } from '@msgpack/msgpack'
|
||||
import { crc32c } from '../utils/crc32c.js'
|
||||
import type { FactLogStorage } from './factLog.js'
|
||||
import { prodLog } from '../utils/logger.js'
|
||||
|
||||
/** Directory for segment files + manifest, under the generations prefix. */
|
||||
export const SEGMENTS_PREFIX = '_generations/segments'
|
||||
|
||||
/** Target sealed-segment size (co-freeze proposal; tunable on evidence). */
|
||||
export const SEGMENT_TARGET_BYTES = 64 * 1024 * 1024
|
||||
|
||||
const MAGIC = new TextEncoder().encode('BGS1')
|
||||
const FRAME_PREFIX_BYTES = 8 // u32 payloadLen + u32 crc32c
|
||||
const MANIFEST_PATH = `${SEGMENTS_PREFIX}/manifest.json`
|
||||
|
||||
/** One generation's fold input — exactly what the live tier holds for it. */
|
||||
export interface FoldGeneration {
|
||||
generation: number
|
||||
timestamp: number
|
||||
/** The tx.json delta object, carried verbatim. */
|
||||
delta: unknown
|
||||
/** The before-image record-set (empty for record-less generations). */
|
||||
records: Array<{ kind: 'noun' | 'verb'; id: string; record: unknown }>
|
||||
}
|
||||
|
||||
/** Manifest entry for one sealed segment. */
|
||||
export interface SegmentMeta {
|
||||
file: string
|
||||
firstGeneration: number
|
||||
lastGeneration: number
|
||||
frames: number
|
||||
bytes: number
|
||||
/** crc32c of the full segment byte stream — the digest chain's link. */
|
||||
checksum: number
|
||||
}
|
||||
|
||||
interface SegmentManifest {
|
||||
version: 1
|
||||
compactedBelow: number
|
||||
segments: SegmentMeta[]
|
||||
}
|
||||
|
||||
interface SidecarIndex {
|
||||
version: 1
|
||||
/** [generation, frameOffset, frameLen] ascending by generation. */
|
||||
generations: Array<[number, number, number]>
|
||||
/** `${kindByte}:${id}` → ascending generations holding a record for it. */
|
||||
ids: Record<string, number[]>
|
||||
}
|
||||
|
||||
const segmentFileName = (firstGeneration: number): string =>
|
||||
`seg-${String(firstGeneration).padStart(20, '0')}.bgs`
|
||||
const sidecarFileName = (firstGeneration: number): string =>
|
||||
`seg-${String(firstGeneration).padStart(20, '0')}.idx`
|
||||
|
||||
/**
|
||||
* The generation-segment store. Owns the packed tier ONLY — the live
|
||||
* per-generation tier and the routing between tiers belong to
|
||||
* `GenerationStore`. All mutating entry points here are called under the
|
||||
* generation store's commit mutex.
|
||||
*/
|
||||
export class GenerationSegmentStore {
|
||||
private readonly storage: FactLogStorage
|
||||
private manifest: SegmentManifest = { version: 1, compactedBelow: 0, segments: [] }
|
||||
/** Sidecar cache — segments are immutable, so entries never invalidate. */
|
||||
private readonly sidecars = new Map<string, SidecarIndex>()
|
||||
|
||||
constructor(storage: FactLogStorage) {
|
||||
this.storage = storage
|
||||
}
|
||||
|
||||
/** Load the manifest (ONE read — never a directory listing). */
|
||||
async open(): Promise<void> {
|
||||
const raw = (await this.storage.readRawObject(MANIFEST_PATH)) as SegmentManifest | null
|
||||
if (raw) {
|
||||
if (raw.version !== 1) {
|
||||
throw new Error(
|
||||
`[GenerationSegments] manifest version ${String(raw.version)} is newer than this ` +
|
||||
`engine understands — refusing to serve partial history. Upgrade the engine.`
|
||||
)
|
||||
}
|
||||
this.manifest = raw
|
||||
}
|
||||
}
|
||||
|
||||
/** The packed tier's catalogue (ascending, immutable snapshot). */
|
||||
segments(): readonly SegmentMeta[] {
|
||||
return this.manifest.segments
|
||||
}
|
||||
|
||||
/** D3's horizon marker: generations below this were reclaimed (bounded profiles only). */
|
||||
compactedBelow(): number {
|
||||
return this.manifest.compactedBelow
|
||||
}
|
||||
|
||||
/** The covering sealed segment for `gen`, or null if it lives outside the packed tier. */
|
||||
private coveringSegment(gen: number): SegmentMeta | null {
|
||||
// Manifest is ascending and ranges never overlap — binary search.
|
||||
const segs = this.manifest.segments
|
||||
let lo = 0
|
||||
let hi = segs.length - 1
|
||||
while (lo <= hi) {
|
||||
const mid = (lo + hi) >> 1
|
||||
const s = segs[mid]
|
||||
if (gen < s.firstGeneration) hi = mid - 1
|
||||
else if (gen > s.lastGeneration) lo = mid + 1
|
||||
else return s
|
||||
}
|
||||
return null
|
||||
}
|
||||
|
||||
/** True when `gen` is packed (readable from this tier). */
|
||||
hasGeneration(gen: number): boolean {
|
||||
return this.coveringSegment(gen) !== null
|
||||
}
|
||||
|
||||
/**
|
||||
* @description True when `meta` declares more generations than it holds
|
||||
* frames — a segment sealed by a writer that folded across a hole. The
|
||||
* manifest records `frames` at fold time, so this is an O(1) comparison
|
||||
* against the declared span and needs no I/O.
|
||||
*/
|
||||
private isSparse(meta: SegmentMeta): boolean {
|
||||
return meta.lastGeneration - meta.firstGeneration + 1 !== meta.frames
|
||||
}
|
||||
|
||||
/**
|
||||
* @description The generations this tier ACTUALLY holds, as coalesced
|
||||
* ascending intervals — not what the segments declare.
|
||||
*
|
||||
* Dense segments (every one a current writer produces) contribute their
|
||||
* declared range with no I/O. A SPARSE segment — one sealed before the
|
||||
* density law was enforced, whose declared range spans generations it has
|
||||
* no frame for — has its real generation list read from its sidecar and
|
||||
* contributed instead, with the discrepancy narrated once.
|
||||
*
|
||||
* This is what keeps a store that already carries the damage from wedging.
|
||||
* `open()` seeds `committedRanges` from these intervals, so a hole is never
|
||||
* re-admitted as a committed generation, and the auto-compaction pass that
|
||||
* used to fail on every run with "packed history is damaged" simply never
|
||||
* asks for the missing frame.
|
||||
*
|
||||
* @returns Ascending, non-overlapping `[first, last]` intervals.
|
||||
*/
|
||||
async actualRanges(): Promise<Array<[number, number]>> {
|
||||
const out: Array<[number, number]> = []
|
||||
for (const meta of this.manifest.segments) {
|
||||
if (!this.isSparse(meta)) {
|
||||
out.push([meta.firstGeneration, meta.lastGeneration])
|
||||
continue
|
||||
}
|
||||
const missing = meta.lastGeneration - meta.firstGeneration + 1 - meta.frames
|
||||
prodLog.warn(
|
||||
`[GenerationSegments] sealed segment ${meta.file} declares generations ` +
|
||||
`${meta.firstGeneration}..${meta.lastGeneration} but holds only ${meta.frames} ` +
|
||||
`frame(s) — ${missing} generation(s) in that span were never folded into it. ` +
|
||||
`Serving the frames it actually holds; the declared span is not treated as ` +
|
||||
`committed history. (Written by a pre-density-law writer that folded across a ` +
|
||||
`gap; the segment itself is intact and no record is lost.)`
|
||||
)
|
||||
const idx = await this.sidecarFor(meta)
|
||||
for (const [gen] of idx.generations) {
|
||||
const last = out[out.length - 1]
|
||||
if (last !== undefined && gen === last[1] + 1) last[1] = gen
|
||||
else out.push([gen, gen])
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
/**
|
||||
* Fold consecutive generations into ONE new sealed segment + sidecar and
|
||||
* append it to the manifest atomically. Caller guarantees: `gens` is
|
||||
* ascending, contiguous with the packed tier (first = last packed + 1 when
|
||||
* segments exist), and already durable in the live tier. Crash between the
|
||||
* segment write and the caller's live-tier delete leaves a DUPLICATE
|
||||
* representation — resolved live-tier-wins by the reader; never a gap.
|
||||
*/
|
||||
async fold(gens: FoldGeneration[]): Promise<SegmentMeta> {
|
||||
if (gens.length === 0) {
|
||||
throw new Error('[GenerationSegments] fold() requires at least one generation')
|
||||
}
|
||||
for (let i = 1; i < gens.length; i++) {
|
||||
if (gens[i].generation <= gens[i - 1].generation) {
|
||||
throw new Error('[GenerationSegments] fold() input must be strictly ascending')
|
||||
}
|
||||
}
|
||||
// THE DENSITY LAW, MADE MECHANICAL.
|
||||
//
|
||||
// A sealed segment declares a CONTIGUOUS range [firstGeneration,
|
||||
// lastGeneration] and every reader treats that range as containment:
|
||||
// `coveringSegment` is an interval test, `hasGeneration` returns true for
|
||||
// anything inside it, and `open()` seeds committedRanges from it. So a
|
||||
// segment folded from a SPARSE input silently claims generations it does
|
||||
// not hold, and the first read of one of those holes throws
|
||||
// "inside sealed segment ... but has no frame — packed history is damaged".
|
||||
//
|
||||
// That is exactly how the damage was produced. `repackHistory` skipped
|
||||
// generations mid-batch — ones absent from committedRanges, ones still in
|
||||
// the pending buffer, ones whose tx.json would not read — and handed the
|
||||
// survivors here, where the range was computed from the first and last of
|
||||
// them. Worse, the mis-declared range was then merged back into
|
||||
// committedRanges at the next open, which is what turned a quiet hole into
|
||||
// a repeating auto-compaction failure on every subsequent run.
|
||||
//
|
||||
// Callers now split at discontinuities; this refusal is what keeps any
|
||||
// future caller from reintroducing the class. A refusal here loses
|
||||
// nothing — the generations stay in the live tier, readable, and the next
|
||||
// pass folds them correctly.
|
||||
for (let i = 1; i < gens.length; i++) {
|
||||
if (gens[i].generation !== gens[i - 1].generation + 1) {
|
||||
throw new Error(
|
||||
`[GenerationSegments] fold() input is not contiguous: ${gens[i - 1].generation} → ` +
|
||||
`${gens[i].generation} skips ${gens[i].generation - gens[i - 1].generation - 1} ` +
|
||||
`generation(s). A sealed segment declares a dense range, so folding a sparse ` +
|
||||
`batch would claim generations it does not hold. Split the batch at the gap.`
|
||||
)
|
||||
}
|
||||
}
|
||||
const last = this.manifest.segments[this.manifest.segments.length - 1]
|
||||
if (last && gens[0].generation <= last.lastGeneration) {
|
||||
throw new Error(
|
||||
`[GenerationSegments] fold() overlaps the packed tier: ${gens[0].generation} ≤ ` +
|
||||
`sealed ${last.lastGeneration} — segments are immutable, never rewritten`
|
||||
)
|
||||
}
|
||||
|
||||
const first = gens[0].generation
|
||||
const file = segmentFileName(first)
|
||||
const sidecar: SidecarIndex = { version: 1, generations: [], ids: {} }
|
||||
|
||||
// Encode all frames, tracking offsets for the sidecar.
|
||||
const parts: Uint8Array[] = [MAGIC]
|
||||
let offset = MAGIC.length
|
||||
for (const g of gens) {
|
||||
const payload = msgpackEncode([
|
||||
g.generation,
|
||||
g.timestamp,
|
||||
g.delta,
|
||||
g.records.map((r) => [r.kind === 'noun' ? 0 : 1, r.id, r.record]),
|
||||
0 // flags: v1 = uncompressed
|
||||
])
|
||||
const frame = new Uint8Array(FRAME_PREFIX_BYTES + payload.length)
|
||||
const view = new DataView(frame.buffer)
|
||||
view.setUint32(0, payload.length, true)
|
||||
view.setUint32(4, crc32c(payload), true)
|
||||
frame.set(payload, FRAME_PREFIX_BYTES)
|
||||
sidecar.generations.push([g.generation, offset, frame.length])
|
||||
for (const r of g.records) {
|
||||
const key = `${r.kind === 'noun' ? 0 : 1}:${r.id}`
|
||||
;(sidecar.ids[key] ??= []).push(g.generation)
|
||||
}
|
||||
parts.push(frame)
|
||||
offset += frame.length
|
||||
}
|
||||
const total = parts.reduce((n, p) => n + p.length, 0)
|
||||
const bytes = new Uint8Array(total)
|
||||
let at = 0
|
||||
for (const p of parts) {
|
||||
bytes.set(p, at)
|
||||
at += p.length
|
||||
}
|
||||
|
||||
const meta: SegmentMeta = {
|
||||
file,
|
||||
firstGeneration: first,
|
||||
lastGeneration: gens[gens.length - 1].generation,
|
||||
frames: gens.length,
|
||||
bytes: total,
|
||||
checksum: crc32c(bytes)
|
||||
}
|
||||
|
||||
// Durability order: segment + sidecar fsync'd BEFORE the manifest names
|
||||
// them (a crash before the manifest = invisible orphan files, harmless);
|
||||
// manifest last, atomically.
|
||||
const segPath = `${SEGMENTS_PREFIX}/${file}`
|
||||
const idxPath = `${SEGMENTS_PREFIX}/${sidecarFileName(first)}`
|
||||
await this.storage.writeRawBytes(segPath, bytes)
|
||||
await this.storage.writeRawBytes(idxPath, msgpackEncode(sidecar))
|
||||
await this.storage.syncRawObjects([segPath, idxPath])
|
||||
const next: SegmentManifest = {
|
||||
...this.manifest,
|
||||
segments: [...this.manifest.segments, meta]
|
||||
}
|
||||
await this.storage.writeRawObject(MANIFEST_PATH, next)
|
||||
await this.storage.syncRawObjects([MANIFEST_PATH])
|
||||
this.manifest = next
|
||||
this.sidecars.set(file, sidecar)
|
||||
return meta
|
||||
}
|
||||
|
||||
/** Load (or rebuild, loudly) a segment's sidecar. */
|
||||
private async sidecarFor(meta: SegmentMeta): Promise<SidecarIndex> {
|
||||
const cached = this.sidecars.get(meta.file)
|
||||
if (cached) return cached
|
||||
const idxPath = `${SEGMENTS_PREFIX}/${sidecarFileName(meta.firstGeneration)}`
|
||||
const raw = await this.storage.readRawBytes(idxPath)
|
||||
if (raw) {
|
||||
try {
|
||||
const idx = msgpackDecode(raw) as SidecarIndex
|
||||
if (idx.version === 1) {
|
||||
this.sidecars.set(meta.file, idx)
|
||||
return idx
|
||||
}
|
||||
} catch {
|
||||
// fall through to rebuild
|
||||
}
|
||||
}
|
||||
// Sidecars are DERIVED: rebuild from the segment, loudly — never serve
|
||||
// wrong offsets silently.
|
||||
prodLog.warn(
|
||||
`[GenerationSegments] sidecar for ${meta.file} missing or unreadable — rebuilding from the segment`
|
||||
)
|
||||
const rebuilt = await this.rebuildSidecar(meta)
|
||||
await this.storage.writeRawBytes(idxPath, msgpackEncode(rebuilt))
|
||||
this.sidecars.set(meta.file, rebuilt)
|
||||
return rebuilt
|
||||
}
|
||||
|
||||
/** One sequential read of the segment → a fresh sidecar. Verifies every frame CRC. */
|
||||
private async rebuildSidecar(meta: SegmentMeta): Promise<SidecarIndex> {
|
||||
const frames = await this.readAllFrames(meta)
|
||||
const idx: SidecarIndex = { version: 1, generations: [], ids: {} }
|
||||
for (const f of frames) {
|
||||
idx.generations.push([f.generation, f.offset, f.frameLen])
|
||||
for (const r of f.records) {
|
||||
const key = `${r.kind === 'noun' ? 0 : 1}:${r.id}`
|
||||
;(idx.ids[key] ??= []).push(f.generation)
|
||||
}
|
||||
}
|
||||
return idx
|
||||
}
|
||||
|
||||
private decodeFrame(
|
||||
payload: Uint8Array
|
||||
): { generation: number; timestamp: number; delta: unknown; records: FoldGeneration['records'] } {
|
||||
const [generation, timestamp, delta, rawRecords] = msgpackDecode(payload) as [
|
||||
number,
|
||||
number,
|
||||
unknown,
|
||||
Array<[number, string, unknown]>,
|
||||
number
|
||||
]
|
||||
return {
|
||||
generation,
|
||||
timestamp,
|
||||
delta,
|
||||
records: rawRecords.map(([kindByte, id, record]) => ({
|
||||
kind: kindByte === 0 ? ('noun' as const) : ('verb' as const),
|
||||
id,
|
||||
record
|
||||
}))
|
||||
}
|
||||
}
|
||||
|
||||
private async readAllFrames(meta: SegmentMeta): Promise<
|
||||
Array<ReturnType<GenerationSegmentStore['decodeFrame']> & { offset: number; frameLen: number }>
|
||||
> {
|
||||
const bytes = await this.storage.readRawBytes(`${SEGMENTS_PREFIX}/${meta.file}`)
|
||||
if (!bytes) {
|
||||
throw new Error(
|
||||
`[GenerationSegments] sealed segment ${meta.file} is MISSING — packed history is damaged; ` +
|
||||
`refusing to continue silently`
|
||||
)
|
||||
}
|
||||
const out: Array<ReturnType<GenerationSegmentStore['decodeFrame']> & { offset: number; frameLen: number }> = []
|
||||
let at = MAGIC.length
|
||||
const view = new DataView(bytes.buffer, bytes.byteOffset, bytes.byteLength)
|
||||
while (at + FRAME_PREFIX_BYTES <= bytes.length) {
|
||||
const payloadLen = view.getUint32(at, true)
|
||||
const crc = view.getUint32(at + 4, true)
|
||||
const payload = bytes.subarray(at + FRAME_PREFIX_BYTES, at + FRAME_PREFIX_BYTES + payloadLen)
|
||||
if (payload.length !== payloadLen || crc32c(payload) !== crc) {
|
||||
throw new Error(
|
||||
`[GenerationSegments] frame CRC mismatch in ${meta.file} at offset ${at} — ` +
|
||||
`packed history is damaged; refusing to serve it`
|
||||
)
|
||||
}
|
||||
out.push({ ...this.decodeFrame(payload), offset: at, frameLen: FRAME_PREFIX_BYTES + payloadLen })
|
||||
at += FRAME_PREFIX_BYTES + payloadLen
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
/** Read one packed generation's frame via its sidecar offset (one ranged read). */
|
||||
private async readFrame(
|
||||
gen: number
|
||||
): Promise<ReturnType<GenerationSegmentStore['decodeFrame']> | null> {
|
||||
const meta = this.coveringSegment(gen)
|
||||
if (!meta) return null
|
||||
const idx = await this.sidecarFor(meta)
|
||||
// generations ascending → binary search.
|
||||
const gens = idx.generations
|
||||
let lo = 0
|
||||
let hi = gens.length - 1
|
||||
while (lo <= hi) {
|
||||
const mid = (lo + hi) >> 1
|
||||
if (gens[mid][0] < gen) lo = mid + 1
|
||||
else if (gens[mid][0] > gen) hi = mid - 1
|
||||
else {
|
||||
const [, offset, frameLen] = gens[mid]
|
||||
const bytes = await this.storage.readRawBytes(`${SEGMENTS_PREFIX}/${meta.file}`)
|
||||
if (!bytes) {
|
||||
throw new Error(`[GenerationSegments] sealed segment ${meta.file} is MISSING`)
|
||||
}
|
||||
const frame = bytes.subarray(offset, offset + frameLen)
|
||||
const view = new DataView(frame.buffer, frame.byteOffset, frame.byteLength)
|
||||
const payloadLen = view.getUint32(0, true)
|
||||
const crc = view.getUint32(4, true)
|
||||
const payload = frame.subarray(FRAME_PREFIX_BYTES, FRAME_PREFIX_BYTES + payloadLen)
|
||||
if (payload.length !== payloadLen || crc32c(payload) !== crc) {
|
||||
throw new Error(
|
||||
`[GenerationSegments] frame CRC mismatch for generation ${gen} in ${meta.file} — ` +
|
||||
`packed history is damaged; refusing to serve it`
|
||||
)
|
||||
}
|
||||
return this.decodeFrame(payload)
|
||||
}
|
||||
}
|
||||
// Inside the covering range but with no frame. Two very different causes,
|
||||
// and conflating them is what made this class wedge every maintenance pass
|
||||
// on the affected stores.
|
||||
//
|
||||
// (1) A SPARSE SEGMENT — the manifest's own `frames` count is smaller than
|
||||
// the span it declares. That segment was sealed by a writer that
|
||||
// folded across a hole (the class this file's density law now bars).
|
||||
// The segment is INTACT and nothing is lost; it simply never held this
|
||||
// generation. Answering "not packed" is the honest answer, and it lets
|
||||
// the caller's two-tier read decide what a genuinely absent generation
|
||||
// means, instead of every compaction pass dying on a repeating throw.
|
||||
// `actualRanges()` keeps such holes out of committedRanges at open, so
|
||||
// in a healed store nobody asks this question in the first place.
|
||||
//
|
||||
// (2) A DENSE SEGMENT missing a frame it says it has — the manifest and
|
||||
// the sidecar disagree about a segment that claims to be complete.
|
||||
// That IS damage, and it stays loud.
|
||||
if (this.isSparse(meta)) {
|
||||
prodLog.warn(
|
||||
`[GenerationSegments] generation ${gen} falls inside sealed segment ${meta.file}'s ` +
|
||||
`declared range ${meta.firstGeneration}..${meta.lastGeneration}, but that segment ` +
|
||||
`holds ${meta.frames} frame(s) for a ${meta.lastGeneration - meta.firstGeneration + 1}` +
|
||||
`-generation span — it was sealed across a gap and never held this generation. ` +
|
||||
`Reporting it as unpacked rather than as damage; no record is lost.`
|
||||
)
|
||||
return null
|
||||
}
|
||||
throw new Error(
|
||||
`[GenerationSegments] generation ${gen} is inside sealed segment ${meta.file}'s declared ` +
|
||||
`range but has no frame, and that segment declares a complete ${meta.frames}-frame ` +
|
||||
`span — the manifest and the sidecar disagree; packed history is damaged`
|
||||
)
|
||||
}
|
||||
|
||||
/** The packed tier's delta for `gen` (null = not packed). */
|
||||
async readDelta(gen: number): Promise<{ delta: unknown; timestamp: number } | null> {
|
||||
const frame = await this.readFrame(gen)
|
||||
return frame ? { delta: frame.delta, timestamp: frame.timestamp } : null
|
||||
}
|
||||
|
||||
/** The packed tier's full record-set for `gen` (null = not packed). */
|
||||
async readRecords(gen: number): Promise<FoldGeneration['records'] | null> {
|
||||
const frame = await this.readFrame(gen)
|
||||
return frame ? frame.records : null
|
||||
}
|
||||
|
||||
/** One packed before-image (null = not packed OR no record for the id in that generation). */
|
||||
async readRecord(gen: number, kind: 'noun' | 'verb', id: string): Promise<unknown | null> {
|
||||
const frame = await this.readFrame(gen)
|
||||
if (!frame) return null
|
||||
const hit = frame.records.find((r) => r.kind === kind && r.id === id)
|
||||
return hit ? hit.record : null
|
||||
}
|
||||
|
||||
/**
|
||||
* D3 reclaim: drop WHOLE segments whose lastGeneration < `belowGeneration`
|
||||
* and bump `compactedBelow`. Partial segments are never dropped — the
|
||||
* boundary waits. NEVER called under the archival profile (the caller
|
||||
* enforces retention semantics; this method only executes boundary drops).
|
||||
*/
|
||||
async dropSegmentsBelow(belowGeneration: number): Promise<{ dropped: number; compactedBelow: number }> {
|
||||
const keep: SegmentMeta[] = []
|
||||
const drop: SegmentMeta[] = []
|
||||
for (const s of this.manifest.segments) {
|
||||
;(s.lastGeneration < belowGeneration ? drop : keep).push(s)
|
||||
}
|
||||
if (drop.length === 0) {
|
||||
return { dropped: 0, compactedBelow: this.manifest.compactedBelow }
|
||||
}
|
||||
const compactedBelow = Math.max(
|
||||
this.manifest.compactedBelow,
|
||||
drop[drop.length - 1].lastGeneration + 1
|
||||
)
|
||||
// Manifest first (the drop is authoritative once named), then bytes —
|
||||
// a crash between leaves orphan segment files invisible to the manifest,
|
||||
// harmless and re-collectable.
|
||||
const next: SegmentManifest = { ...this.manifest, compactedBelow, segments: keep }
|
||||
await this.storage.writeRawObject(MANIFEST_PATH, next)
|
||||
await this.storage.syncRawObjects([MANIFEST_PATH])
|
||||
this.manifest = next
|
||||
for (const s of drop) {
|
||||
await this.storage.deleteRawObject(`${SEGMENTS_PREFIX}/${s.file}`)
|
||||
await this.storage.deleteRawObject(`${SEGMENTS_PREFIX}/${sidecarFileName(s.firstGeneration)}`)
|
||||
this.sidecars.delete(s.file)
|
||||
}
|
||||
return { dropped: drop.length, compactedBelow }
|
||||
}
|
||||
|
||||
/**
|
||||
* D8 rider — the packed portion of `generationDigest(g)`: a deterministic
|
||||
* crc32c chain over sealed-segment checksums fully below `g`, plus the
|
||||
* frame CRC of `g`'s own frame when `g` is mid-segment. O(segments), not
|
||||
* O(generations); identical history ⇒ identical digest on any machine.
|
||||
* The live-tier portion is composed by the caller.
|
||||
*/
|
||||
async digestThroughPacked(g: number): Promise<number | null> {
|
||||
let digest = 0
|
||||
let covered = false
|
||||
for (const s of this.manifest.segments) {
|
||||
if (s.lastGeneration <= g) {
|
||||
digest = crc32c(new TextEncoder().encode(`${digest}:${s.checksum}`))
|
||||
if (s.lastGeneration === g) covered = true
|
||||
} else if (s.firstGeneration <= g) {
|
||||
// g is mid-segment: chain the partial prefix via g's frame CRC.
|
||||
const frame = await this.readFrame(g)
|
||||
if (frame === null) return null
|
||||
const idx = await this.sidecarFor(s)
|
||||
const upTo = idx.generations.filter(([gen]) => gen <= g)
|
||||
for (const [gen, offset, frameLen] of upTo) {
|
||||
digest = crc32c(new TextEncoder().encode(`${digest}:${gen}:${offset}:${frameLen}`))
|
||||
}
|
||||
covered = true
|
||||
break
|
||||
}
|
||||
}
|
||||
return covered || this.manifest.segments.length > 0 ? digest : null
|
||||
}
|
||||
}
|
||||
File diff suppressed because it is too large
Load diff
|
|
@ -1,341 +0,0 @@
|
|||
/**
|
||||
* @module db/logAuthority
|
||||
* @description The per-brain LOG-AUTHORITY SWITCH and its verification
|
||||
* oracle — the guarded adoption path for log-canonical storage.
|
||||
*
|
||||
* Two storage authorities exist during the adoption window:
|
||||
* - `'tree'` (the default, today's behavior): the canonical record tree is
|
||||
* authoritative; the generation log is a complete dual-written journal.
|
||||
* - `'log'`: the generation log is authoritative for this brain; single-op
|
||||
* write acks await a covering log fsync (durable-at-ack), and derived
|
||||
* state treats the log as ground truth.
|
||||
*
|
||||
* THE SWITCH IS PER BRAIN, STORED, CHECKED AT OPEN ONLY, and ONE-DIRECTIONAL
|
||||
* unless explicitly reverted by an operator. A brain flips ONLY when its
|
||||
* verification oracle is green: a full replay-and-diff of the log against
|
||||
* the still-authoritative tree (the read-only witness). The oracle failing
|
||||
* NAMES every divergence — a brain with pre-log history (records the log
|
||||
* never saw) reports them as `pre-log-record` mismatches and needs a
|
||||
* baseline backfill before it can ever flip.
|
||||
*
|
||||
* Nothing in this module mutates data: the oracle is read-only; the flip
|
||||
* writes ONE artifact. Reverting = rewriting the artifact to 'tree' (the
|
||||
* tree remained authoritative-quality throughout the window by dual-write).
|
||||
*/
|
||||
|
||||
import type { FactScanHandle } from './factLog.js'
|
||||
import { prodLog } from '../utils/logger.js'
|
||||
import { createHash } from 'crypto'
|
||||
|
||||
/** Storage-root-relative path of the authority switch artifact. */
|
||||
export const LOG_AUTHORITY_PATH = '_system/log-authority.json'
|
||||
|
||||
/** The persisted shape of the authority switch. */
|
||||
export interface LogAuthorityRecord {
|
||||
/** Which store is authoritative for this brain. */
|
||||
authority: 'tree' | 'log'
|
||||
/** When the flip happened (ms epoch). Absent while authority = 'tree'. */
|
||||
flippedAt?: number
|
||||
/** The oracle verdict that justified the flip (summary, not the full report). */
|
||||
oracle?: {
|
||||
verifiedAt: number
|
||||
generationsScanned: number
|
||||
nounsChecked: number
|
||||
verbsChecked: number
|
||||
}
|
||||
/**
|
||||
* Recorded when an OPEN-TIME adoption attempt (the 10.0.0 fleet default)
|
||||
* was refused — the oracle could not go green. Keeps subsequent opens
|
||||
* cheap; an operator re-runs adoptLogAuthority() after resolving it.
|
||||
*/
|
||||
adoptRefusal?: { at: number; reason: string }
|
||||
}
|
||||
|
||||
/** The narrow storage surface this module needs. */
|
||||
export interface LogAuthorityStorage {
|
||||
readRawObject(path: string): Promise<unknown | null>
|
||||
writeRawObject(path: string, data: unknown): Promise<void>
|
||||
syncRawObjects(paths: string[]): Promise<void>
|
||||
getNouns(opts: {
|
||||
pagination: { limit: number; offset?: number; cursor?: string }
|
||||
}): Promise<{ items: unknown[]; hasMore?: boolean; nextCursor?: string }>
|
||||
getNounMetadata(id: string): Promise<unknown | null>
|
||||
}
|
||||
|
||||
/** One divergence found by the oracle. */
|
||||
export interface OracleMismatch {
|
||||
id: string
|
||||
kind: 'noun' | 'verb'
|
||||
reason:
|
||||
| 'pre-log-record' // canonical row the log never saw — needs baseline backfill
|
||||
| 'state-differs' // latest log after-image ≠ canonical bytes
|
||||
| 'log-live-canonical-absent' // log says live, canonical has no record
|
||||
| 'log-tombstone-canonical-present' // log says deleted, canonical still has it
|
||||
}
|
||||
|
||||
/** The oracle's full report. */
|
||||
export interface OracleReport {
|
||||
verdict: 'green' | 'red'
|
||||
generationsScanned: number
|
||||
nounsChecked: number
|
||||
verbsChecked: number
|
||||
matched: number
|
||||
mismatches: OracleMismatch[]
|
||||
/** Mismatch listing is capped; the counts above are always complete. */
|
||||
mismatchListTruncated: boolean
|
||||
}
|
||||
|
||||
const MISMATCH_LIST_CAP = 200
|
||||
|
||||
/** Read the stored authority (absent artifact = 'tree', the safe default). */
|
||||
export async function readLogAuthority(
|
||||
storage: Pick<LogAuthorityStorage, 'readRawObject'>
|
||||
): Promise<LogAuthorityRecord> {
|
||||
const raw = (await storage
|
||||
.readRawObject(LOG_AUTHORITY_PATH)
|
||||
.catch(() => null)) as LogAuthorityRecord | null
|
||||
if (raw && (raw.authority === 'log' || raw.authority === 'tree')) return raw
|
||||
return { authority: 'tree' }
|
||||
}
|
||||
|
||||
/**
|
||||
* Normalize a canonical noun record to its ENTITY TRUTH before diffing:
|
||||
* the canonical vector-file wrapper denormalizes derived index residue
|
||||
* (`connections` — HNSW graph edges; `level` — the node's random skip-list
|
||||
* level) that the generation log deliberately does NOT carry (projections
|
||||
* own their own rebuild paths). Digesting the residue would report false
|
||||
* `state-differs` on ~any brain whose HNSW assigned a nonzero level. Both
|
||||
* sides of every oracle comparison pass through this normalizer.
|
||||
*/
|
||||
export function nounEntityTruth(record: {
|
||||
metadata: unknown
|
||||
vector: unknown
|
||||
}): { metadata: unknown; vector: unknown } {
|
||||
const v = record.vector
|
||||
if (v && typeof v === 'object' && !Array.isArray(v)) {
|
||||
const { connections: _c, level: _l, ...entity } = v as Record<string, unknown>
|
||||
return { metadata: record.metadata, vector: entity }
|
||||
}
|
||||
return { metadata: record.metadata, vector: v }
|
||||
}
|
||||
|
||||
/**
|
||||
* Stable content hash of a stored record for diffing — key-sorted JSON so
|
||||
* property order can never fake a divergence.
|
||||
*/
|
||||
export function recordDigest(record: unknown): string {
|
||||
const stable = (v: unknown): unknown => {
|
||||
if (Array.isArray(v)) return v.map(stable)
|
||||
if (v && typeof v === 'object') {
|
||||
const out: Record<string, unknown> = {}
|
||||
for (const k of Object.keys(v as Record<string, unknown>).sort()) {
|
||||
out[k] = stable((v as Record<string, unknown>)[k])
|
||||
}
|
||||
return out
|
||||
}
|
||||
return v
|
||||
}
|
||||
return createHash('sha256').update(JSON.stringify(stable(record))).digest('hex')
|
||||
}
|
||||
|
||||
/**
|
||||
* THE VERIFICATION ORACLE: replay the fact log's noun records and diff the
|
||||
* final state per id against the canonical tree (the witness). Read-only;
|
||||
* bounded memory (id → {tombstoned, digest} — digests, never bodies).
|
||||
*
|
||||
* Verdict law: 'green' iff EVERY canonical row's latest state is exactly
|
||||
* reproduced by the log AND the log claims nothing canonical denies. A
|
||||
* brain older than its log reports its unlogged rows as `pre-log-record`
|
||||
* mismatches — the named cure is a baseline backfill, never a silent pass.
|
||||
*/
|
||||
export async function runLogCompletenessOracle(args: {
|
||||
storage: LogAuthorityStorage
|
||||
scanFacts: () => FactScanHandle | null
|
||||
/** Digest the canonical record the same way the log's after-image is digested. */
|
||||
canonicalNounDigest: (id: string) => Promise<string | null>
|
||||
/** Digest a log after-image record's payload. */
|
||||
factRecordDigest: (record: unknown) => string
|
||||
/**
|
||||
* Verb legs (optional until every owner wires them): the canonical verb
|
||||
* digest + the paged verb enumeration. When ABSENT, the oracle counts NO
|
||||
* verbs and says so via verbsChecked = 0 — an honest partial verdict,
|
||||
* never a silent full-pass claim.
|
||||
*/
|
||||
canonicalVerbDigest?: (id: string) => Promise<string | null>
|
||||
getVerbs?: (opts: {
|
||||
pagination: { limit: number; offset?: number; cursor?: string }
|
||||
}) => Promise<{ items: unknown[]; hasMore?: boolean; nextCursor?: string }>
|
||||
/**
|
||||
* Cap on the LISTED mismatches (counts are always complete). Defaults to
|
||||
* the wire-friendly {@link MISMATCH_LIST_CAP}; the adoption backfill passes
|
||||
* `Infinity` so ONE scan yields the ENTIRE curable set — a production
|
||||
* brain with a 12.7k-row pre-log baseline once advanced only 800 rows per
|
||||
* adoption call because each pass could see (and cure) at most 200.
|
||||
*/
|
||||
mismatchListCap?: number
|
||||
}): Promise<OracleReport> {
|
||||
const listCap = args.mismatchListCap ?? MISMATCH_LIST_CAP
|
||||
const report: OracleReport = {
|
||||
verdict: 'red',
|
||||
generationsScanned: 0,
|
||||
nounsChecked: 0,
|
||||
verbsChecked: 0,
|
||||
matched: 0,
|
||||
mismatches: [],
|
||||
mismatchListTruncated: false
|
||||
}
|
||||
const addMismatch = (m: OracleMismatch): void => {
|
||||
if (report.mismatches.length < listCap) report.mismatches.push(m)
|
||||
else report.mismatchListTruncated = true
|
||||
}
|
||||
|
||||
// Pass 1: fold the log — latest state per noun id (digest or tombstone).
|
||||
const scan = args.scanFacts()
|
||||
if (!scan) {
|
||||
// No fact log on this store: nothing can be verified — red, loudly.
|
||||
prodLog.warn('[logAuthority] oracle: this store has no fact log — cannot verify, verdict red')
|
||||
return report
|
||||
}
|
||||
const logState = new Map<string, { tombstoned: boolean; digest: string | null }>()
|
||||
const verbLogState = new Map<string, { tombstoned: boolean; digest: string | null }>()
|
||||
for await (const batch of scan.batches()) {
|
||||
for (const fact of batch.facts) {
|
||||
report.generationsScanned++
|
||||
for (const op of fact.ops) {
|
||||
const state =
|
||||
op.record === null
|
||||
? { tombstoned: true, digest: null }
|
||||
: { tombstoned: false, digest: args.factRecordDigest(op.record) }
|
||||
if (op.kind === 'noun') logState.set(op.id, state)
|
||||
else verbLogState.set(op.id, state)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Pass 2: walk canonical (paged) and diff.
|
||||
const seenCanonical = new Set<string>()
|
||||
const PAGE = 500
|
||||
let offset = 0
|
||||
let cursor: string | undefined
|
||||
for (;;) {
|
||||
const page = await args.storage.getNouns({
|
||||
pagination: cursor ? { limit: PAGE, cursor } : { limit: PAGE, offset }
|
||||
})
|
||||
for (const item of page.items) {
|
||||
const id = (item as { id: string }).id
|
||||
seenCanonical.add(id)
|
||||
report.nounsChecked++
|
||||
const inLog = logState.get(id)
|
||||
if (!inLog) {
|
||||
addMismatch({ id, kind: 'noun', reason: 'pre-log-record' })
|
||||
continue
|
||||
}
|
||||
if (inLog.tombstoned) {
|
||||
addMismatch({ id, kind: 'noun', reason: 'log-tombstone-canonical-present' })
|
||||
continue
|
||||
}
|
||||
const canonicalDigest = await args.canonicalNounDigest(id)
|
||||
if (canonicalDigest === null) {
|
||||
addMismatch({ id, kind: 'noun', reason: 'pre-log-record' })
|
||||
continue
|
||||
}
|
||||
if (canonicalDigest === inLog.digest) report.matched++
|
||||
else addMismatch({ id, kind: 'noun', reason: 'state-differs' })
|
||||
}
|
||||
if (!page.hasMore || page.items.length === 0) break
|
||||
if (page.nextCursor) cursor = page.nextCursor
|
||||
else offset += page.items.length
|
||||
}
|
||||
|
||||
// Pass 3: log-live ids canonical never showed us.
|
||||
for (const [id, state] of logState) {
|
||||
if (!state.tombstoned && !seenCanonical.has(id)) {
|
||||
addMismatch({ id, kind: 'noun', reason: 'log-live-canonical-absent' })
|
||||
}
|
||||
}
|
||||
|
||||
// Verb passes — only when the owner wired the verb legs; otherwise the
|
||||
// report says verbsChecked: 0, an honest partial scope, never a claim.
|
||||
if (args.canonicalVerbDigest && args.getVerbs) {
|
||||
const seenVerbs = new Set<string>()
|
||||
let vOffset = 0
|
||||
let vCursor: string | undefined
|
||||
for (;;) {
|
||||
const page = await args.getVerbs({
|
||||
pagination: vCursor ? { limit: PAGE, cursor: vCursor } : { limit: PAGE, offset: vOffset }
|
||||
})
|
||||
for (const item of page.items) {
|
||||
const id = (item as { id: string }).id
|
||||
seenVerbs.add(id)
|
||||
report.verbsChecked++
|
||||
const inLog = verbLogState.get(id)
|
||||
if (!inLog) {
|
||||
addMismatch({ id, kind: 'verb', reason: 'pre-log-record' })
|
||||
continue
|
||||
}
|
||||
if (inLog.tombstoned) {
|
||||
addMismatch({ id, kind: 'verb', reason: 'log-tombstone-canonical-present' })
|
||||
continue
|
||||
}
|
||||
const canonical = await args.canonicalVerbDigest(id)
|
||||
if (canonical === null) {
|
||||
addMismatch({ id, kind: 'verb', reason: 'pre-log-record' })
|
||||
continue
|
||||
}
|
||||
if (canonical === inLog.digest) report.matched++
|
||||
else addMismatch({ id, kind: 'verb', reason: 'state-differs' })
|
||||
}
|
||||
if (!page.hasMore || page.items.length === 0) break
|
||||
if (page.nextCursor) vCursor = page.nextCursor
|
||||
else vOffset += page.items.length
|
||||
}
|
||||
for (const [id, state] of verbLogState) {
|
||||
if (!state.tombstoned && !seenVerbs.has(id)) {
|
||||
addMismatch({ id, kind: 'verb', reason: 'log-live-canonical-absent' })
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
const totalMismatches =
|
||||
report.mismatches.length + (report.mismatchListTruncated ? 1 : 0)
|
||||
report.verdict = totalMismatches === 0 ? 'green' : 'red'
|
||||
return report
|
||||
}
|
||||
|
||||
/**
|
||||
* Flip this brain's authority to the log — REFUSES unless the supplied
|
||||
* oracle report is green (the caller runs the oracle; the flip records its
|
||||
* summary). Writes + fsyncs the switch artifact; the mode takes full effect
|
||||
* at the NEXT open (checked-at-open-only law), except durable-at-ack which
|
||||
* the owner may enable immediately.
|
||||
*/
|
||||
export async function flipToLogAuthority(
|
||||
storage: Pick<LogAuthorityStorage, 'writeRawObject' | 'syncRawObjects'>,
|
||||
oracle: OracleReport
|
||||
): Promise<LogAuthorityRecord> {
|
||||
if (oracle.verdict !== 'green') {
|
||||
throw new Error(
|
||||
`log-authority flip refused: the verification oracle is RED ` +
|
||||
`(${oracle.mismatches.length}${oracle.mismatchListTruncated ? '+' : ''} mismatches; ` +
|
||||
`first: ${oracle.mismatches[0] ? `${oracle.mismatches[0].reason} on ${oracle.mismatches[0].id}` : 'n/a'}). ` +
|
||||
`A brain flips only on green — fix the divergences (pre-log records need a baseline backfill) and re-run.`
|
||||
)
|
||||
}
|
||||
const record: LogAuthorityRecord = {
|
||||
authority: 'log',
|
||||
flippedAt: Date.now(),
|
||||
oracle: {
|
||||
verifiedAt: Date.now(),
|
||||
generationsScanned: oracle.generationsScanned,
|
||||
nounsChecked: oracle.nounsChecked,
|
||||
verbsChecked: oracle.verbsChecked
|
||||
}
|
||||
}
|
||||
await storage.writeRawObject(LOG_AUTHORITY_PATH, record)
|
||||
await storage.syncRawObjects([LOG_AUTHORITY_PATH])
|
||||
prodLog.info(
|
||||
`[logAuthority] this brain's storage authority is now the generation log ` +
|
||||
`(oracle green over ${oracle.nounsChecked} nouns / ${oracle.generationsScanned} generations)`
|
||||
)
|
||||
return record
|
||||
}
|
||||
Some files were not shown because too many files have changed in this diff Show more
Loading…
Add table
Add a link
Reference in a new issue