Compare commits
170 commits
pair-b/upd
...
main
| Author | SHA1 | Date | |
|---|---|---|---|
| 1b903fe665 | |||
| 842ad44b88 | |||
| 360feaccf8 | |||
| a2ea21b330 | |||
| aac853d3e8 | |||
| 1882532cb7 | |||
| 6eb5e4483d | |||
| ba10aaf52e | |||
| 656d9f6f92 | |||
| 7e1ddee4f7 | |||
| da7d2498bc | |||
| 4e058720b4 | |||
| be307a1579 | |||
| de79d6b5a4 | |||
| d6e7453f1f | |||
| 4c344782a7 | |||
| 4c81d7d4c3 | |||
| dadfa61b5f | |||
| e766ed0a84 | |||
| 87d3a945a5 | |||
| bc82d36294 | |||
| d7444ae804 | |||
| e435da787d | |||
| 97b5ea2d5d | |||
| aa457d7159 | |||
| adcb883e67 | |||
| a2820e81af | |||
| 10a6e81a88 | |||
| e49a73e529 | |||
| 72c8ee6acd | |||
| f27a777615 | |||
| 0d5ab6077d | |||
| a7eb7f5222 | |||
| 5e720d17ae | |||
| 69bda5b7cb | |||
| be77a10bfe | |||
| ad0f493f7a | |||
| 6597c146f7 | |||
| 7932175503 | |||
| 2808398164 | |||
| 6baa4d7f6c | |||
| 85b1fa5c1a | |||
| 8752f11f4d | |||
| 61bc5f423b | |||
| 3835a0e702 | |||
| 27759a1be9 | |||
| 6053f6d423 | |||
| a1423c6da7 | |||
| dea3ec2031 | |||
| 08758c254f | |||
| 3dadbec8f2 | |||
| ebb3a4bf13 | |||
| 2c5e34748e | |||
| 4142f36872 | |||
| 367ca721a5 | |||
| a79db434ac | |||
| da9519903a | |||
| ec644bde56 | |||
| 65493ba2de | |||
| dee46b35c8 | |||
| 1fb5109351 | |||
| 15d4f65dcf | |||
| bc70c43d02 | |||
| 905c267c47 | |||
| b1c7054467 | |||
| 67ae0046de | |||
| 9922631d1f | |||
| 2633e8d5e1 | |||
| f763317af7 | |||
| a8c5fbf9dc | |||
| 34f1886f7c | |||
| 4f1e27c9a0 | |||
| 297a3d7657 | |||
| eec90bdd69 | |||
| 2648f56ddf | |||
| 8a2ebacf02 | |||
| 7ab670b525 | |||
| f097cbf6f2 | |||
| 4d5f823f47 | |||
| 3e60aded36 | |||
| d5147ed608 | |||
| 6a89adc468 | |||
| 88e79729d3 | |||
| e64e2bc175 | |||
| 077cbc0b6f | |||
| 5e3b343a0e | |||
| 4014e0f125 | |||
| 73500e7d10 | |||
| 0f0022b1c9 | |||
| d6bcb14f69 | |||
| a963a744cc | |||
|
|
c99308710a | ||
| 655aa13ea7 | |||
| 39c71ecdac | |||
| 0759c03a82 | |||
| 9a888c37e9 | |||
|
|
298cb6daca | ||
| b8475cc86a | |||
| ff39941b0a | |||
| d49148e140 | |||
| 42e2da259b | |||
| 5ebd3b4061 | |||
| a8c724a202 | |||
| 61a469270e | |||
| 02c6163637 | |||
| 2cf3801007 | |||
| 5c22f9500c | |||
| 16d2e1a97e | |||
| fb1da1c56d | |||
| 417ddb5143 | |||
| 9dd399216b | |||
| e4c27fbca8 | |||
| 5a091ccad9 | |||
| 4a67aa0fb9 | |||
| c1f0972395 | |||
| 48802ba385 | |||
| 50676c02f4 | |||
| 131daa08cd | |||
| f5a6cb3f61 | |||
| 3fffd9c6e6 | |||
| f4e2d34b4e | |||
| afe08a1ff9 | |||
| e652162c1f | |||
| 38c3397b60 | |||
| 384f4b6b9c | |||
| a58372f03f | |||
| a99b1e83c4 | |||
| 9f248b2495 | |||
| 1f34fc6ce4 | |||
| a082e0efdd | |||
| 6f93108648 | |||
| 9b84ef5b02 | |||
| 0de7665930 | |||
| 8fc553b126 | |||
| fd6b4ce4ff | |||
| 204d74c161 | |||
| f8d8ce16b9 | |||
| 2496e09aeb | |||
| 4c7b0fab7a | |||
| c6cc0de955 | |||
| 8a5c1245a7 | |||
| aad9e2eeb1 | |||
| 0a19bbd8a7 | |||
| 2914e0eb42 | |||
| 7870dc4092 | |||
| c039411e08 | |||
| 21e506e802 | |||
| efa042b52c | |||
| 834149ed90 | |||
| d3aacfaf24 | |||
| 9730835bdf | |||
| bce2593e24 | |||
| f4780c8e88 | |||
| f14da34b27 | |||
| 0e1286e321 | |||
| 39b916a3c0 | |||
| 553e0d97ae | |||
| ddd5e71928 | |||
| 258e9042af | |||
| fc516da6eb | |||
| 96624f408c | |||
| 8cced871a0 | |||
| b9ba50fbec | |||
| 18f172e098 | |||
| f8f64780b1 | |||
| a8b5ca0c8f | |||
| a1376e4a2c | |||
| dcbad1765a | |||
| 4176439ba3 | |||
| 116550eb16 |
240 changed files with 26024 additions and 3237 deletions
|
|
@ -2,7 +2,7 @@
|
||||||
|
|
||||||
## What Is Brainy
|
## What Is Brainy
|
||||||
|
|
||||||
@soulcraft/brainy (v7.17.0) is a Universal Knowledge Protocol -- a Triple Intelligence database combining vector search, graph traversal, and metadata filtering in a single library. Published to npm as a public MIT-licensed package.
|
@soulcraftlabs/brainy (v7.17.0) is a Universal Knowledge Protocol -- a Triple Intelligence database combining vector search, graph traversal, and metadata filtering in a single library. Published to npm as a public MIT-licensed package.
|
||||||
|
|
||||||
## Core Architecture
|
## Core Architecture
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -5,6 +5,10 @@ name: CI
|
||||||
# sequential, so tag-triggered matrix jobs (~22 min) would queue AHEAD of the
|
# sequential, so tag-triggered matrix jobs (~22 min) would queue AHEAD of the
|
||||||
# tag's publish-source run and starve every release (observed on 8.10.3 and
|
# tag's publish-source run and starve every release (observed on 8.10.3 and
|
||||||
# 9.0.0: the publish sat behind the tag's own redundant CI).
|
# 9.0.0: the publish sat behind the tag's own redundant CI).
|
||||||
|
concurrency:
|
||||||
|
group: ci-${{ github.ref }}
|
||||||
|
cancel-in-progress: true
|
||||||
|
|
||||||
on:
|
on:
|
||||||
push:
|
push:
|
||||||
branches: ['**']
|
branches: ['**']
|
||||||
|
|
|
||||||
148
.forgejo/workflows/delta-gate.yml
Normal file
148
.forgejo/workflows/delta-gate.yml
Normal file
|
|
@ -0,0 +1,148 @@
|
||||||
|
name: Delta Gate
|
||||||
|
|
||||||
|
# On-demand candidate-vs-control gate on the capped functional CI lane
|
||||||
|
# (label: gate-functional). That lane is Bun-only host-mode — there is no
|
||||||
|
# Node.js runtime available to it, so this workflow deliberately avoids every
|
||||||
|
# JS-based action (checkout/setup-node/setup-bun/upload-artifact all require
|
||||||
|
# one) and does everything with plain git + bun in shell steps instead.
|
||||||
|
#
|
||||||
|
# Verdict lines a caller should grep for in the run log:
|
||||||
|
# COLLECTED patch=<n> control=<n> — collection-truncation guard inputs
|
||||||
|
# NEW-RED-COUNT:<n> — failures on candidate absent from control
|
||||||
|
# DELTA-GATE: CLEAN | NEW REDS | INVALID | STOPPED-BY-REGISTRY-TRIPWIRE
|
||||||
|
#
|
||||||
|
# The lane's own housekeeping stops the runner and drops a marker file when
|
||||||
|
# host pressure (I/O, registry latency, disk budget) trips — never ours to
|
||||||
|
# interpret as a red or a green. The final step checks for that marker before
|
||||||
|
# it says anything about pass/fail.
|
||||||
|
|
||||||
|
on:
|
||||||
|
workflow_dispatch:
|
||||||
|
inputs:
|
||||||
|
candidate:
|
||||||
|
description: 'Candidate ref (branch or sha) to gate'
|
||||||
|
required: true
|
||||||
|
type: string
|
||||||
|
control:
|
||||||
|
description: 'Control sha to diff against'
|
||||||
|
required: true
|
||||||
|
type: string
|
||||||
|
# workflow_dispatch needs Actions-unit write on the dispatching credential;
|
||||||
|
# push does not (it runs from the pushed ref's own tree), so a plain push
|
||||||
|
# to a release or CI branch is the fallback trigger while that grant is
|
||||||
|
# outstanding — see the ref-resolution step below for what it gates against.
|
||||||
|
push:
|
||||||
|
branches: ['rel/**', 'ci/**']
|
||||||
|
|
||||||
|
concurrency:
|
||||||
|
group: delta-gate
|
||||||
|
cancel-in-progress: false
|
||||||
|
|
||||||
|
jobs:
|
||||||
|
delta-gate:
|
||||||
|
name: Delta gate — candidate vs control
|
||||||
|
runs-on: gate-functional
|
||||||
|
timeout-minutes: 120
|
||||||
|
steps:
|
||||||
|
- name: Resolve candidate/control refs
|
||||||
|
id: refs
|
||||||
|
run: |
|
||||||
|
candidate="${{ github.event.inputs.candidate }}"
|
||||||
|
control="${{ github.event.inputs.control }}"
|
||||||
|
# workflow_dispatch supplies both explicitly; a push event carries
|
||||||
|
# neither — fall back to the pushed commit as candidate and the
|
||||||
|
# last released, known-good tip (10.4.9) as control, so a plain
|
||||||
|
# push still produces a meaningful gate instead of an empty ref.
|
||||||
|
if [ -z "$candidate" ]; then candidate="${{ github.sha }}"; fi
|
||||||
|
if [ -z "$control" ]; then control="eec90bdd"; fi
|
||||||
|
echo "candidate=$candidate" >> "$GITHUB_OUTPUT"
|
||||||
|
echo "control=$control" >> "$GITHUB_OUTPUT"
|
||||||
|
echo "Resolved (trigger=${{ github.event_name }}): candidate=$candidate control=$control"
|
||||||
|
|
||||||
|
- name: Clean any residue from a prior run
|
||||||
|
run: rm -rf "ob-cand-${{ github.run_id }}" "ob-ctrl-${{ github.run_id }}" "/tmp/ob-${{ github.run_id }}-"*
|
||||||
|
|
||||||
|
- name: Clone + test — candidate
|
||||||
|
id: patch
|
||||||
|
run: |
|
||||||
|
set -o pipefail
|
||||||
|
git clone --quiet "https://source.soulcraft.com/soulcraftlabs/open-brainy.git" "ob-cand-${{ github.run_id }}"
|
||||||
|
cd "ob-cand-${{ github.run_id }}"
|
||||||
|
git checkout --quiet "${{ steps.refs.outputs.candidate }}"
|
||||||
|
git log --oneline -1
|
||||||
|
bun install
|
||||||
|
rc=0
|
||||||
|
bun x vitest run > "/tmp/ob-${{ github.run_id }}-patch.log" 2>&1 || rc=$?
|
||||||
|
echo "PATCH-RC:$rc"
|
||||||
|
grep -aE "Tests .*(passed|failed)" "/tmp/ob-${{ github.run_id }}-patch.log" | tail -1
|
||||||
|
grep -aE "^ FAIL |^\s+×" "/tmp/ob-${{ github.run_id }}-patch.log" | sed -E "s/ [0-9]+ms$//" | sed -E "s/^\s+//" | sort -u > "/tmp/ob-${{ github.run_id }}-patch.fail"
|
||||||
|
echo "PATCH-FAILING:$(wc -l < "/tmp/ob-${{ github.run_id }}-patch.fail")"
|
||||||
|
|
||||||
|
- name: Clone + test — control
|
||||||
|
id: control
|
||||||
|
run: |
|
||||||
|
set -o pipefail
|
||||||
|
git clone --quiet "https://source.soulcraft.com/soulcraftlabs/open-brainy.git" "ob-ctrl-${{ github.run_id }}"
|
||||||
|
cd "ob-ctrl-${{ github.run_id }}"
|
||||||
|
git checkout --quiet "${{ steps.refs.outputs.control }}"
|
||||||
|
git log --oneline -1
|
||||||
|
bun install
|
||||||
|
rc=0
|
||||||
|
bun x vitest run > "/tmp/ob-${{ github.run_id }}-control.log" 2>&1 || rc=$?
|
||||||
|
echo "CONTROL-RC:$rc"
|
||||||
|
grep -aE "Tests .*(passed|failed)" "/tmp/ob-${{ github.run_id }}-control.log" | tail -1
|
||||||
|
grep -aE "^ FAIL |^\s+×" "/tmp/ob-${{ github.run_id }}-control.log" | sed -E "s/ [0-9]+ms$//" | sed -E "s/^\s+//" | sort -u > "/tmp/ob-${{ github.run_id }}-control.fail"
|
||||||
|
echo "CONTROL-FAILING:$(wc -l < "/tmp/ob-${{ github.run_id }}-control.fail")"
|
||||||
|
|
||||||
|
- name: Delta gate verdict
|
||||||
|
if: always()
|
||||||
|
run: |
|
||||||
|
set -o pipefail
|
||||||
|
|
||||||
|
# The lane's own tripwire wins over anything we would otherwise say:
|
||||||
|
# a bare failure/timeout above with this marker present is host
|
||||||
|
# pressure, never a real red and never a real green.
|
||||||
|
if [ -f /srv/gate-lane/TRIPWIRE-STOPPED ]; then
|
||||||
|
echo "DELTA-GATE: STOPPED-BY-REGISTRY-TRIPWIRE"
|
||||||
|
head -1 /srv/gate-lane/TRIPWIRE-STOPPED
|
||||||
|
exit 3
|
||||||
|
fi
|
||||||
|
|
||||||
|
patch_log="/tmp/ob-${{ github.run_id }}-patch.log"
|
||||||
|
control_log="/tmp/ob-${{ github.run_id }}-control.log"
|
||||||
|
patch_fail="/tmp/ob-${{ github.run_id }}-patch.fail"
|
||||||
|
control_fail="/tmp/ob-${{ github.run_id }}-control.fail"
|
||||||
|
|
||||||
|
if [ ! -s "$patch_log" ] || [ ! -s "$control_log" ]; then
|
||||||
|
echo "DELTA-GATE: INVALID — a leg produced no log (see the two steps above for the real cause)"
|
||||||
|
exit 2
|
||||||
|
fi
|
||||||
|
|
||||||
|
pt=$(grep -aoE "\(([0-9]+)\)$" "$patch_log" | tail -1 | tr -d "()")
|
||||||
|
ct=$(grep -aoE "\(([0-9]+)\)$" "$control_log" | tail -1 | tr -d "()")
|
||||||
|
echo "COLLECTED patch=${pt:-0} control=${ct:-0}"
|
||||||
|
if [ "${pt:-0}" -lt 3000 ] || [ "${ct:-0}" -lt 3000 ]; then
|
||||||
|
echo "DELTA-GATE: INVALID — truncated collection"
|
||||||
|
exit 2
|
||||||
|
fi
|
||||||
|
|
||||||
|
echo "=== NEW REDS ==="
|
||||||
|
comm -23 "$patch_fail" "$control_fail"
|
||||||
|
new=$(comm -23 "$patch_fail" "$control_fail" | wc -l)
|
||||||
|
echo "NEW-RED-COUNT:$new"
|
||||||
|
|
||||||
|
echo "=== full candidate fail list ==="
|
||||||
|
cat "$patch_fail"
|
||||||
|
echo "=== full control fail list ==="
|
||||||
|
cat "$control_fail"
|
||||||
|
|
||||||
|
if [ "$new" -eq 0 ]; then
|
||||||
|
echo "DELTA-GATE: CLEAN"
|
||||||
|
else
|
||||||
|
echo "DELTA-GATE: NEW REDS"
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
- name: Clean up (mind the lane's disk budget)
|
||||||
|
if: always()
|
||||||
|
run: rm -rf "ob-cand-${{ github.run_id }}" "ob-ctrl-${{ github.run_id }}" "/tmp/ob-${{ github.run_id }}-"*
|
||||||
|
|
@ -12,6 +12,11 @@ on:
|
||||||
push:
|
push:
|
||||||
tags:
|
tags:
|
||||||
- 'v*'
|
- 'v*'
|
||||||
|
workflow_dispatch:
|
||||||
|
inputs:
|
||||||
|
ref_reason:
|
||||||
|
description: 'why this manual run (e.g. tag event dropped)'
|
||||||
|
required: false
|
||||||
|
|
||||||
jobs:
|
jobs:
|
||||||
publish:
|
publish:
|
||||||
|
|
@ -32,22 +37,31 @@ jobs:
|
||||||
run: |
|
run: |
|
||||||
set -eo pipefail
|
set -eo pipefail
|
||||||
|
|
||||||
SOURCE_NPM_REG="https://source.soulcraft.com/api/packages/soulcraft/npm/"
|
SOURCE_NPM_REG="https://source.soulcraft.com/api/packages/soulcraftlabs/npm/"
|
||||||
VERSION="$(node -p "require('./package.json').version")"
|
VERSION="$(node -p "require('./package.json').version")"
|
||||||
echo "Publishing @soulcraft/brainy@${VERSION} to The Source registry..."
|
# The dist-tag follows the version: a prerelease (any hyphen —
|
||||||
|
# 10.4.0-rc.1) publishes under 'rc' and must NEVER move 'latest' —
|
||||||
|
# every consumer resolving 'latest' from this registry would otherwise
|
||||||
|
# be handed a release candidate. Same rule scripts/release.sh applies
|
||||||
|
# to the storefront leg.
|
||||||
|
NPM_TAG="latest"
|
||||||
|
case "$VERSION" in
|
||||||
|
*-*) NPM_TAG="rc" ;;
|
||||||
|
esac
|
||||||
|
echo "Publishing @soulcraftlabs/brainy@${VERSION} to The Source registry (dist-tag: ${NPM_TAG})..."
|
||||||
|
|
||||||
TMPRC="$(mktemp)"
|
TMPRC="$(mktemp)"
|
||||||
chmod 600 "$TMPRC"
|
chmod 600 "$TMPRC"
|
||||||
{
|
{
|
||||||
echo "@soulcraft:registry=${SOURCE_NPM_REG}"
|
echo "@soulcraftlabs:registry=${SOURCE_NPM_REG}"
|
||||||
echo "//source.soulcraft.com/api/packages/soulcraft/npm/:_authToken=${FORGE_NPM_TOKEN}"
|
echo "//source.soulcraft.com/api/packages/soulcraftlabs/npm/:_authToken=${FORGE_NPM_TOKEN}"
|
||||||
} > "$TMPRC"
|
} > "$TMPRC"
|
||||||
|
|
||||||
# The release script bumps package.json's version before it tags, so
|
# The release script bumps package.json's version before it tags, so
|
||||||
# this tag's checkout already carries the version being published —
|
# this tag's checkout already carries the version being published —
|
||||||
# nothing here re-derives it from the tag name.
|
# nothing here re-derives it from the tag name.
|
||||||
PUBLISH_OK=true
|
PUBLISH_OK=true
|
||||||
if ! npm publish --tag latest --userconfig "$TMPRC"; then
|
if ! npm publish --tag "$NPM_TAG" --userconfig "$TMPRC"; then
|
||||||
PUBLISH_OK=false
|
PUBLISH_OK=false
|
||||||
fi
|
fi
|
||||||
|
|
||||||
|
|
@ -55,7 +69,7 @@ jobs:
|
||||||
# exit code: a benign duplicate publish (a prior run, or a mirror, already
|
# exit code: a benign duplicate publish (a prior run, or a mirror, already
|
||||||
# landed this exact version) reports failure even though the registry
|
# landed this exact version) reports failure even though the registry
|
||||||
# already holds the right content.
|
# already holds the right content.
|
||||||
LANDED_VERSION="$(npm view "@soulcraft/brainy@${VERSION}" version --userconfig "$TMPRC" 2>/dev/null || echo "")"
|
LANDED_VERSION="$(npm view "@soulcraftlabs/brainy@${VERSION}" version --userconfig "$TMPRC" 2>/dev/null || echo "")"
|
||||||
rm -f "$TMPRC"
|
rm -f "$TMPRC"
|
||||||
|
|
||||||
if [ "$LANDED_VERSION" != "$VERSION" ]; then
|
if [ "$LANDED_VERSION" != "$VERSION" ]; then
|
||||||
|
|
@ -64,7 +78,7 @@ jobs:
|
||||||
fi
|
fi
|
||||||
|
|
||||||
if [ "$PUBLISH_OK" = true ]; then
|
if [ "$PUBLISH_OK" = true ]; then
|
||||||
echo "Published and verified @soulcraft/brainy@${VERSION} on The Source registry."
|
echo "Published and verified @soulcraftlabs/brainy@${VERSION} on The Source registry."
|
||||||
else
|
else
|
||||||
echo "::warning::npm publish reported failure, but readback confirms @soulcraft/brainy@${VERSION} is already live on The Source (a prior run or mirror landed it) — treating this run as successful, since the registry content is correct. Any OTHER failure mode would have failed the readback check above instead."
|
echo "::warning::npm publish reported failure, but readback confirms @soulcraftlabs/brainy@${VERSION} is already live on The Source (a prior run or mirror landed it) — treating this run as successful, since the registry content is correct. Any OTHER failure mode would have failed the readback check above instead."
|
||||||
fi
|
fi
|
||||||
|
|
|
||||||
201
CHANGELOG.md
201
CHANGELOG.md
|
|
@ -2,13 +2,194 @@
|
||||||
|
|
||||||
All notable changes to this project will be documented in this file. See [standard-version](https://github.com/conventional-changelog/standard-version) for commit guidelines.
|
All notable changes to this project will be documented in this file. See [standard-version](https://github.com/conventional-changelog/standard-version) for commit guidelines.
|
||||||
|
|
||||||
### [10.3.1](https://source.soulcraft.com/soulcraft/brainy/compare/v10.3.0...v10.3.1) (2026-08-18)
|
|
||||||
|
### [10.4.13](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v10.4.12...v10.4.13) (2026-09-03)
|
||||||
|
|
||||||
|
- A shutdown that holds its listener until the exit decision, and a test suite that closes every brain it opens
|
||||||
|
- fix(shutdown): the engine's signal handler keeps its listener registered until the exit decision is made — closing the last live instance no longer deregisters the handler mid-run, so a second signal delivery during a clean shutdown can never kill the process after the work is done (a2ea21b3)
|
||||||
|
- fix(release): the release wall entry commits under an explicit git identity read from the developer's checkout; a host with no identity refuses by name instead of failing inside git (aac853d3)
|
||||||
|
- test(hygiene): every brain a test file creates is closed by that file — 40 files fixed, the leaks that let a stray cadence narrate into later files are gone; brains whose init() was expected to fail are closed too (6eb5e448)
|
||||||
|
|
||||||
|
### [10.4.12](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v10.4.11...v10.4.12) (2026-09-03)
|
||||||
|
|
||||||
|
- Mixed-kind fields index exactly, arrays to 256, a drained loop is not a shutdown, and finds project from the column store
|
||||||
|
- fix(index): a metadata field holds every value kind it was written with — one posting column per (field, kind); an equality filter reads the query value's own kind, a range routes by its bounds; nothing is refused and nothing is silently dropped; an index written by the old shape opens unchanged (a128f0ed)
|
||||||
|
- fix(metadata): metadata arrays index up to 256 elements; a longer array refuses at write time by name (MetadataArrayTooLargeError) — a vector parked in metadata now throws; move it to `vector` (e435da78)
|
||||||
|
- fix(shutdown): beforeExit runs a non-closing flush only — a script that never calls close() exits with the writer lock on disk and no clean-shutdown marker, and the next open evicts the stale lock and folds the log, bounded; SIGTERM and SIGINT are unchanged (6baa4d7f)
|
||||||
|
- feat(find): field projection — find({fields}) and get({fields}) resolve scalars from the column store on every leg, including vector-leg finds; absent fields stay absent (ad0f493f)
|
||||||
|
- fix(find): orderBy is the order on every find path, not only the metadata-only one (5e720d17)
|
||||||
|
- fix(metadata): the legacy sparse range path orders values, or refuses by name — never ranks by hash (a7eb7f52)
|
||||||
|
- fix(close): a read-only brain writes nothing under `_system/` (f27a7776)
|
||||||
|
- fix(contract): the flush gate's internals are private, not doors (72c8ee6a)
|
||||||
|
- test(hygiene): the triple-intelligence correctness cases sit in the gate; the idle and connected-find pins name the brain they measure (28083981)
|
||||||
|
- ci(release): the rail writes its own wall entry into the shared releases repo — never hand-written again (adcb883e)
|
||||||
|
|
||||||
|
### [10.4.11](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v10.4.9...v10.4.11) (2026-09-02)
|
||||||
|
|
||||||
|
- ci: superseded pushes cancel their own runs (concurrency per ref) (6053f6d4)
|
||||||
|
- test(batch): the batch-size-limit tests add unvectored items — they test batching, not embedding (a1423c6d)
|
||||||
|
- fix(flush): the gate settles its waiter from the machine, never from a chain (dea3ec20)
|
||||||
|
- test(batch): the batch-vs-individual timing assertion runs in the perf lane, not the correctness gate (ebb3a4bf)
|
||||||
|
- test(gate): the coverage guard counts the perf lane's config as a gate (2c5e3474)
|
||||||
|
- chore(contract): emit the 10.4.11 manifest (4142f368)
|
||||||
|
- fix(close): a read-only brain writes no clean-shutdown evidence — the marker is the writer's word about itself (367ca721)
|
||||||
|
- fix(generation-store): commitTransaction refuses while single-ops are pending — the order invariant is enforced, not assumed (a79db434)
|
||||||
|
- test(shutdown): pin one owner per brain — real processes, real signals (da951990)
|
||||||
|
- fix(shutdown): one owner per brain — the signal handler defers to close(), and flush is single-flight (ec644bde)
|
||||||
|
- fix(vfs): a path-scoped search is a served range over the path, not a refused prefix match (65493ba2)
|
||||||
|
- ci(test): perf and scale benchmarks leave the correctness gate (dee46b35)
|
||||||
|
- test(open): pin the pending-embed checkpoint — stuck id, crash matrix, torn fallback (1fb51093)
|
||||||
|
- perf(open): the pending-embed fold is bounded by a checkpoint of the SET, not an empty-only mark (15d4f65d)
|
||||||
|
- perf(open): a sealed segment the manifest proves is below the bound is never read (bc70c43d)
|
||||||
|
- fix(find): a page the metadata block already cut is not cut again (905c267c)
|
||||||
|
- fix(find): the hybrid legs rank inside the filter, and only the page is read (b1c70544)
|
||||||
|
- ci(delta-gate): add a push fallback trigger alongside workflow_dispatch (67ae0046)
|
||||||
|
- ci: add the delta-gate workflow for the capped functional lane (9922631d)
|
||||||
|
- docs(plugin): the planner door's hiddenIds contract is the answer, not the mechanism (2633e8d5)
|
||||||
|
- feat(engine): a protected factory for the generation store — a subclass may substitute one that keeps the contract (f763317a)
|
||||||
|
- fix(find): near() searches around the anchor's own vector, and refuses by name without one (a8c5fbf9)
|
||||||
|
- Merge remote-tracking branches 'origin/fix/planner-provider-door' and 'origin/fix/containment-batching' into rel/10.4.10-candidate (34f1886f)
|
||||||
|
- feat(plugin): an optional planFindPage door — an index that can plan a find answers it in one call (4d5f823f)
|
||||||
|
- perf(vfs): repairContainment's reconcile is one paged edge walk, not one graph call per file (3e60aded)
|
||||||
|
|
||||||
|
|
||||||
|
### [10.4.9](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v10.4.6...v10.4.9) (2026-09-02)
|
||||||
|
|
||||||
|
- Merge branch 'fix/pending-embed-low-water' into rel/10.4.9-candidate (2648f56d)
|
||||||
|
- fix(open): pending-embed recovery keeps the crash-recovery contract — foreground, bounded by the mark (8a2ebacf)
|
||||||
|
- Merge branches 'fix/connected-find-order', 'fix/pending-embed-low-water' and 'fix/related-verb-array' into rel/10.4.9-candidate (d5147ed6)
|
||||||
|
- fix(graph): the verb fast paths honour every requested type, source, and target (6a89adc4)
|
||||||
|
- perf(open): pending-embed recovery is bounded by a low-water mark and runs behind the doors (88e79729)
|
||||||
|
- fix(find): connected finds are graph-first — neighbours, then the filter over those ids, then the page (077cbc0b)
|
||||||
|
- fix(storage): counts persistence is single-flight, coalesced, and never races its own temp file (5e3b343a)
|
||||||
|
|
||||||
|
|
||||||
|
### [10.4.6](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v10.4.5...v10.4.6) (2026-08-31)
|
||||||
|
|
||||||
|
- fix(transact): metadata-index ops take their JSON-safe view at the crossing, not at construction (73500e7d)
|
||||||
|
|
||||||
|
|
||||||
|
### [10.4.5](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v10.4.4...v10.4.5) (2026-08-31)
|
||||||
|
|
||||||
|
- build(release): the docs-push step retires — this engine documents itself in its own repository (d6bcb14f)
|
||||||
|
- fix(generations): a sealed segment may only declare the generations it holds (a963a744)
|
||||||
|
- fix(recovery): a torn generation-log tail is a terminal verdict, never a wait (c9930871)
|
||||||
|
|
||||||
|
|
||||||
|
### [10.4.4](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v10.4.3...v10.4.4) (2026-08-28)
|
||||||
|
|
||||||
|
- fix(vfs): the old-root sweep narrates only when it has something to say (d49148e1)
|
||||||
|
- fix(tests): the health-gate pin follows the verdict, and the VFS suite uses its own store (42e2da25)
|
||||||
|
- Merge branch 'next/open-lazy-open-and-counts' (5ebd3b40)
|
||||||
|
- docs: the contract manifest stands alone; public docs describe this engine only (a8c724a2)
|
||||||
|
- docs(releases): 10.4.4 consumer notes — correctness and observability, with the performance line stated exactly (61a46927)
|
||||||
|
- docs: measurements in public history carry numbers, not provenance (02c61636)
|
||||||
|
- feat(open): name the two steps that hold the vfs-bootstrap phase (2cf38010)
|
||||||
|
- fix(storage): a dead flush watch falls back to the 500ms poll, not the 30s sweep (5c22f950)
|
||||||
|
- fix(storage): the flush watcher cannot arm twice in its async window (16d2e1a9)
|
||||||
|
- perf(idle): the flush-request watch is event-driven; the heartbeat is observability (fb1da1c5)
|
||||||
|
- perf(open): answer "are there any entities?" with one directory read (417ddb51)
|
||||||
|
- perf(generations): discover generations by directory name, not by walking the log (9dd39921)
|
||||||
|
- fix(flush): clear() and repairIndex() set the dirty witness themselves (e4c27fbc)
|
||||||
|
- feat(open): the open names the STEP that cost the time, not just the phase (5a091cca)
|
||||||
|
- perf(vfs): the old-root sweep runs once per store, not once per open (4a67aa0f)
|
||||||
|
- chore: keep the generated neural stamps at main's values (c1f09723)
|
||||||
|
- feat(contract): declare contract 1, serve three operators, refuse four by name (48802ba3)
|
||||||
|
- fix(open): a provider rebuilding itself is a third state, not a CRITICAL (50676c02)
|
||||||
|
- feat(open): open never waits for a provider that is rebuilding itself (131daa08)
|
||||||
|
- perf(flush): an idle brain does no work — no periodic flush without a write (f5a6cb3f)
|
||||||
|
- feat(repair): repairIndex narrates every phase and its receipt carries the walls (3fffd9c6)
|
||||||
|
- fix(storage): a suspect count ledger heals itself, and counts.json is written atomically (f4e2d34b)
|
||||||
|
- feat(open): the open narrates itself, on a channel production cannot clamp (afe08a1f)
|
||||||
|
- fix(storage): a clean close is recorded, and the writer lock is always given up (e652162c)
|
||||||
|
- docs: repository links point at soulcraftlabs/open-brainy — the soulcraft/brainy path becomes the native engine's repo tonight (38c3397b)
|
||||||
|
|
||||||
|
|
||||||
|
### [10.4.3](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v10.4.2...v10.4.3) (2026-08-27)
|
||||||
|
|
||||||
|
- Merge branch 'next/open-brainy-rename' (a58372f0)
|
||||||
|
- chore: rename to @soulcraftlabs/brainy for Open Brainy on The Source (a99b1e83)
|
||||||
|
- docs(releases): 10.4.3 — Open Brainy's first release under the new name, same engine as 10.4.2; The Source is the one registry (9f248b24)
|
||||||
|
|
||||||
|
|
||||||
|
### [10.4.2](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v10.4.2-rc.1...v10.4.2) (2026-08-27)
|
||||||
|
|
||||||
|
- docs(releases): 10.4.1 and 10.4.2 consumer notes; 10.4.2 is the last MIT release under this name, Open Brainy continues at @soulcraftlabs/brainy (a082e0ef)
|
||||||
|
|
||||||
|
|
||||||
|
### [10.4.2-rc.1](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v10.4.1...v10.4.2-rc.1) (2026-08-27)
|
||||||
|
|
||||||
|
- Merge branch 'next/zero-norm-unvector-door' (9b84ef5b)
|
||||||
|
- fix(vectors): a zero-norm vector is not a vector, canonical side included, plus the sanctioned unvector door (0de76659)
|
||||||
|
- fix(hnsw): skip unvectored rows on rebuild; refuse empty vectors in the index (8fc553b1)
|
||||||
|
- fix(storage): derive the canonical count ledger from identity records, stamp the derivation rule, and mark legacy-derived ledgers suspect at load (fd6b4ce4)
|
||||||
|
- Merge branch 'next/enumeration-identity-rekey' (204d74c1)
|
||||||
|
- fix(storage): enumeration re-keys on the identity record, not the vector leg (f8d8ce16)
|
||||||
|
- fix(init): rethrow plugin activation failures with the original error as cause so the originating frame survives to the caller (2496e09a)
|
||||||
|
- Merge branch 'next/vfs-root-zero-norm' (4c7b0fab)
|
||||||
|
- fix(vfs): the VFS root never persists a zero-norm vector (c6cc0de9)
|
||||||
|
- build: derive generated-file stamps from git commit time, not wall clock (8a5c1245)
|
||||||
|
- Merge remote-tracking branch 'origin/release/10.4.1' (aad9e2ee)
|
||||||
|
- docs(concepts): the serving law — a failure is graded by whether an answer could be wrong, never by the cost of the fix; reads refuse per family (2914e0eb)
|
||||||
|
- chore(release): 10.4.1-rc.1 (7870dc40)
|
||||||
|
|
||||||
|
|
||||||
|
### [10.4.1](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v10.4.0...v10.4.1) (2026-08-26)
|
||||||
|
|
||||||
|
- fix(reads): the read gate is per-family; a write carrying unchanged data never re-embeds (c039411e)
|
||||||
|
- docs(guide): the docs pipeline publishes through the ingest API — the separate deploy step is retired (21e506e8)
|
||||||
|
|
||||||
|
|
||||||
|
### [10.4.0](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v10.4.0-rc.4...v10.4.0) (2026-08-26)
|
||||||
|
|
||||||
|
- docs(releases): the 10.4.0 entry catches up to the late trains — repair routing, the vector ledger and open-gate leg, the loud config guard, the JSON-safe crossing (834149ed)
|
||||||
|
|
||||||
|
|
||||||
|
### [10.4.0-rc.4](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v10.4.0-rc.3...v10.4.0-rc.4) (2026-08-25)
|
||||||
|
|
||||||
|
- feat(vector): the vectored-noun scalar joins the count ledger; the open gate closes the vector leg (9730835b)
|
||||||
|
|
||||||
|
|
||||||
|
### [10.4.0-rc.3](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v10.4.0-rc.2...v10.4.0-rc.3) (2026-08-25)
|
||||||
|
|
||||||
|
- fix(update-seam): the metadata crossing never carries BigInt endpoint ints (f4780c8e)
|
||||||
|
- Merge branch 'worktree-agent-ad3aff0dffd17a6eb' (f14da34b)
|
||||||
|
- fix(add): empty string is real data, not a missing field (258e9042)
|
||||||
|
- feat(vfs): implement readdir's recursive option — typed since 7.30, never read (fc516da6)
|
||||||
|
- feat(open-path): init never gates on the embedding model; open goes concurrent; slow opens narrate (96624f40)
|
||||||
|
|
||||||
|
|
||||||
|
### [10.4.0-rc.2](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v10.4.0-rc.1...v10.4.0-rc.2) (2026-08-25)
|
||||||
|
|
||||||
|
- test(readiness): the report helper's clock freezes — two independently-built reports compared across a millisecond tick made the plant lane red (39b916a3)
|
||||||
|
- feat(repair): a heal:'repair' verdict routes to the provider's own incremental repair() (553e0d97)
|
||||||
|
- fix(storage): an unknown nested storage config can never silently land on the shared default root (ddd5e719)
|
||||||
|
- docs(release): the 10.4.0 entry, the index-health concept doc, and the API surfaces — written from the tree, not the plan (8cced871)
|
||||||
|
- fix(plugins): the silent-degrade doors close — a broken accelerator install can never read as absent (b9ba50fb)
|
||||||
|
- feat(recovery): the catchup verdict is consumed; verb rows go live; the metadata rebuild goes online (18f172e0)
|
||||||
|
- feat(health): the gate reads the named report — reads refuse loudly, never rebuild; open serves before it returns; the ceremony door (f8f64780)
|
||||||
|
|
||||||
|
|
||||||
|
### [10.4.0-rc.1](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v10.3.1...v10.4.0-rc.1) (2026-08-24)
|
||||||
|
|
||||||
|
- ci(publish): the home dist-tag follows the version — a prerelease publishes under 'rc' and never moves 'latest' (a1376e4a)
|
||||||
|
- chore(release): --source-only — a home-only prerelease mode (The Source, never the storefront) (dcbad176)
|
||||||
|
- test(fold-checkpoint): the ARM-AT-FLIP pin arms its crash instead of racing the pending-flush timer (4176439b)
|
||||||
|
- fix(health): one contract for a throwing probe — heal is none, serving is not withheld; repair report gains missing/rebuilt/reason (116550eb)
|
||||||
|
- feat(storage): the canonical count ledger — ALL-visibility scalars, unclamped totals, suspect-on-unprovable-delete (7c8c8be3)
|
||||||
|
- fix(delete): the null-metadata skip closes — index legs run id-keyed or narrate, never silently strand postings (607e9f54)
|
||||||
|
- feat(repair): repairIndex returns the per-family receipt and narrates its summary (8d45f964)
|
||||||
|
- fix(reads): the readiness gate guards every index read surface — serving empty from a not-ready provider is unrepresentable (40e7119b)
|
||||||
|
- ci(gate): the machine-health preflight and the truncation verdict guard (1e046aa1)
|
||||||
|
|
||||||
|
|
||||||
|
### [10.3.1](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v10.3.0...v10.3.1) (2026-08-18)
|
||||||
|
|
||||||
- docs(releases): the 10.3.1 consumer entry — the fold that behaves (900cc895)
|
- docs(releases): the 10.3.1 consumer entry — the fold that behaves (900cc895)
|
||||||
- fix(recovery): the fold streams and narrates; the checkpoint chain arms at the flip (ed7d1db9)
|
- fix(recovery): the fold streams and narrates; the checkpoint chain arms at the flip (ed7d1db9)
|
||||||
|
|
||||||
|
|
||||||
### [10.3.0](https://source.soulcraft.com/soulcraft/brainy/compare/v10.2.0...v10.3.0) (2026-08-18)
|
### [10.3.0](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v10.2.0...v10.3.0) (2026-08-18)
|
||||||
|
|
||||||
- docs(releases): the 10.3.0 consumer entry — the trust-and-provenance release (97d75649)
|
- docs(releases): the 10.3.0 consumer entry — the trust-and-provenance release (97d75649)
|
||||||
- fix(locks): the fence keys ownership on pid+hostname — a same-process re-open never fences its predecessor (0991cf28)
|
- fix(locks): the fence keys ownership on pid+hostname — a same-process re-open never fences its predecessor (0991cf28)
|
||||||
|
|
@ -17,14 +198,14 @@ All notable changes to this project will be documented in this file. See [standa
|
||||||
- feat(log): system commits carry their origin; the attested per-id reconcile door (9ac9e706)
|
- feat(log): system commits carry their origin; the attested per-id reconcile door (9ac9e706)
|
||||||
|
|
||||||
|
|
||||||
### [10.2.0](https://source.soulcraft.com/soulcraft/brainy/compare/v10.1.0...v10.2.0) (2026-08-17)
|
### [10.2.0](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v10.1.0...v10.2.0) (2026-08-17)
|
||||||
|
|
||||||
- docs(releases): the 10.2.0 consumer entry — adoption completes in one call (97538e1f)
|
- docs(releases): the 10.2.0 consumer entry — adoption completes in one call (97538e1f)
|
||||||
- ci: the correctness plant runs integration + conformance on every push — a release never waits on a second machine (b17fdc8e)
|
- ci: the correctness plant runs integration + conformance on every push — a release never waits on a second machine (b17fdc8e)
|
||||||
- fix(adoption): the baseline backfill runs to completion — one call adopts a pre-log baseline of any size (a5a18838)
|
- fix(adoption): the baseline backfill runs to completion — one call adopts a pre-log baseline of any size (a5a18838)
|
||||||
|
|
||||||
|
|
||||||
### [10.1.0](https://source.soulcraft.com/soulcraft/brainy/compare/v10.0.0...v10.1.0) (2026-08-13)
|
### [10.1.0](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v10.0.0...v10.1.0) (2026-08-13)
|
||||||
|
|
||||||
- docs(releases): the 10.1.0 consumer entry — bounded recovery, restore founding, the two write-path cures (7d3c8696)
|
- docs(releases): the 10.1.0 consumer entry — bounded recovery, restore founding, the two write-path cures (7d3c8696)
|
||||||
- fix(restore): a restore is an unclean event — the swap runs quiesced and the snapshot's durability stamps never survive it (9ca80667)
|
- fix(restore): a restore is an unclean event — the swap runs quiesced and the snapshot's durability stamps never survive it (9ca80667)
|
||||||
|
|
@ -33,7 +214,7 @@ All notable changes to this project will be documented in this file. See [standa
|
||||||
- feat(query): the sparse-store cut — where on a never-carried field serves operator truth, never a refusal (7b67db4d)
|
- feat(query): the sparse-store cut — where on a never-carried field serves operator truth, never a refusal (7b67db4d)
|
||||||
|
|
||||||
|
|
||||||
### [10.0.0](https://source.soulcraft.com/soulcraft/brainy/compare/v9.0.0...v10.0.0) (2026-08-12)
|
### [10.0.0](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v9.0.0...v10.0.0) (2026-08-12)
|
||||||
|
|
||||||
- fix(adoption): the baseline backfill cures hydration-law drift — existing brains reach the crash-safe default with zero operator steps (25f0dd96)
|
- fix(adoption): the baseline backfill cures hydration-law drift — existing brains reach the crash-safe default with zero operator steps (25f0dd96)
|
||||||
- fix(adoption): the reserved-root mint exemption — int 0 is legitimate for exactly one id (2abe8b38)
|
- fix(adoption): the reserved-root mint exemption — int 0 is legitimate for exactly one id (2abe8b38)
|
||||||
|
|
@ -65,7 +246,7 @@ All notable changes to this project will be documented in this file. See [standa
|
||||||
- test: version-coupling pins go major-agnostic — the 8.x literals broke at the 9.0.0 bump while the coupling law itself behaved correctly (8a6807e8)
|
- test: version-coupling pins go major-agnostic — the 8.x literals broke at the 9.0.0 bump while the coupling law itself behaved correctly (8a6807e8)
|
||||||
|
|
||||||
|
|
||||||
### [9.0.0](https://source.soulcraft.com/soulcraft/brainy/compare/v8.11.0...v9.0.0) (2026-08-04)
|
### [9.0.0](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v8.11.0...v9.0.0) (2026-08-04)
|
||||||
|
|
||||||
- docs: 9.0 namespace-migration guide — the simple story + the mechanical sweep checklist, published for humans and tooling alike (61ab9db2)
|
- docs: 9.0 namespace-migration guide — the simple story + the mechanical sweep checklist, published for humans and tooling alike (61ab9db2)
|
||||||
- fix(release): storefront leg republishes CI's exact forge artifact — byte-identity by construction, verified by cross-registry shasum before the ceremony reports success (d89df2ed)
|
- fix(release): storefront leg republishes CI's exact forge artifact — byte-identity by construction, verified by cross-registry shasum before the ceremony reports success (d89df2ed)
|
||||||
|
|
@ -100,7 +281,7 @@ All notable changes to this project will be documented in this file. See [standa
|
||||||
- feat: scanFacts liveness contract — first batch or loud failure within a documented bound (f8e6da2b)
|
- feat: scanFacts liveness contract — first batch or loud failure within a documented bound (f8e6da2b)
|
||||||
|
|
||||||
|
|
||||||
### [8.11.0](https://source.soulcraft.com/soulcraft/brainy/compare/v8.10.1...v8.11.0) (2026-07-27)
|
### [8.11.0](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v8.10.1...v8.11.0) (2026-07-27)
|
||||||
|
|
||||||
- docs: the last two archived-host links point home (91ef1c8b)
|
- docs: the last two archived-host links point home (91ef1c8b)
|
||||||
- feat: includeHidden — export carries every visibility tier for migration-grade canon completeness (63c1eeb9)
|
- feat: includeHidden — export carries every visibility tier for migration-grade canon completeness (63c1eeb9)
|
||||||
|
|
@ -109,19 +290,19 @@ All notable changes to this project will be documented in this file. See [standa
|
||||||
- ci: run the pipeline on the forge (999d0ebb)
|
- ci: run the pipeline on the forge (999d0ebb)
|
||||||
|
|
||||||
|
|
||||||
### [8.10.3](https://source.soulcraft.com/soulcraft/brainy/compare/v8.10.2...v8.10.3) (2026-08-03)
|
### [8.10.3](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v8.10.2...v8.10.3) (2026-08-03)
|
||||||
|
|
||||||
- docs: dedupe the 8.10.2 release-notes entry the cherry doubled onto the branch (8c956608)
|
- docs: dedupe the 8.10.2 release-notes entry the cherry doubled onto the branch (8c956608)
|
||||||
- fix: user metadata named 'level' is a real field everywhere — the engine-internal node layer no longer shadows it in sort/filter/aggregation, and the indexing views stop stamping a phantom 0 into its column; index epoch 2 rebuilds existing brains at first open (958a0859)
|
- fix: user metadata named 'level' is a real field everywhere — the engine-internal node layer no longer shadows it in sort/filter/aggregation, and the indexing views stop stamping a phantom 0 into its column; index epoch 2 rebuilds existing brains at first open (958a0859)
|
||||||
|
|
||||||
|
|
||||||
### [8.10.2](https://source.soulcraft.com/soulcraft/brainy/compare/v8.10.1...v8.10.2) (2026-07-29)
|
### [8.10.2](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v8.10.1...v8.10.2) (2026-07-29)
|
||||||
|
|
||||||
- docs: 8.10.2 consumer release notes — update() write granularity, PathResolver idle-log fix, graph-lsm key recognition (a0123b5b)
|
- docs: 8.10.2 consumer release notes — update() write granularity, PathResolver idle-log fix, graph-lsm key recognition (a0123b5b)
|
||||||
- fix: metadata-only update() never rewrites the noun record — the unconditional whole-vector save turned per-entity stat touches into full rewrites+fsync, amplifying read-heavy sweeps into disk saturation on a production deployment (5b65eb82)
|
- fix: metadata-only update() never rewrites the noun record — the unconditional whole-vector save turned per-entity stat touches into full rewrites+fsync, amplifying read-heavy sweeps into disk saturation on a production deployment (5b65eb82)
|
||||||
|
|
||||||
|
|
||||||
### [8.10.1](https://source.soulcraft.com/soulcraft/brainy/compare/v8.10.0...v8.10.1) (2026-07-24)
|
### [8.10.1](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v8.10.0...v8.10.1) (2026-07-24)
|
||||||
|
|
||||||
- refactor: remove the orphaned transaction-result type left behind by the dead-path removal (edf123a5)
|
- refactor: remove the orphaned transaction-result type left behind by the dead-path removal (edf123a5)
|
||||||
- fix: warm() metadata surface routes through the active provider (warm hook added to the metadata contract); add maintenanceDebt() observability surface (5b2cbf74)
|
- fix: warm() metadata surface routes through the active provider (warm hook added to the metadata contract); add maintenanceDebt() observability surface (5b2cbf74)
|
||||||
|
|
|
||||||
10
CLAUDE.md
10
CLAUDE.md
|
|
@ -12,13 +12,13 @@ Handoff file: `/home/dpsifr/.strategy/PLATFORM-HANDOFF.md`
|
||||||
|
|
||||||
**Brainy's current open actions:** None. MIT open-source — no platform-specific actions.
|
**Brainy's current open actions:** None. MIT open-source — no platform-specific actions.
|
||||||
|
|
||||||
**Current version:** run `npm view @soulcraft/brainy version` (never trust a hardcoded number here — this line went stale for months); consumer-facing changes tracked in `RELEASES.md`
|
**Current version:** run `npm view @soulcraftlabs/brainy version --registry https://source.soulcraft.com/api/packages/soulcraftlabs/npm/` (never trust a hardcoded number here — this line went stale for months); consumer-facing changes tracked in `RELEASES.md`
|
||||||
|
|
||||||
---
|
---
|
||||||
|
|
||||||
## Project Overview
|
## Project Overview
|
||||||
|
|
||||||
Brainy is a Universal Knowledge Protocol -- a Triple Intelligence database that combines vector similarity search, graph traversal, and metadata filtering into a single TypeScript library. Published as `@soulcraft/brainy` on npm under the MIT license.
|
Brainy is a Universal Knowledge Protocol -- a Triple Intelligence database that combines vector similarity search, graph traversal, and metadata filtering into a single TypeScript library. Published as `@soulcraftlabs/brainy` on The Source (source.soulcraft.com registry) under the MIT license.
|
||||||
|
|
||||||
## Getting Started
|
## Getting Started
|
||||||
|
|
||||||
|
|
@ -91,7 +91,7 @@ test: add/update tests (patch version bump)
|
||||||
|
|
||||||
## Docs Pipeline — soulcraft.com/docs
|
## Docs Pipeline — soulcraft.com/docs
|
||||||
|
|
||||||
Docs in `docs/**/*.md` are published with the npm package (included in `files`) and synced to soulcraft.com/docs on every portal deploy. Frontmatter controls what appears publicly.
|
Docs in `docs/**/*.md` are published with the npm package (included in `files`) and go live on soulcraft.com/docs via the docs ingest API: the release script's `scripts/push-docs.js` step POSTs every public doc to `https://soulcraft.com/api/docs/ingest` (auth: `DOCS_INGEST_SECRET` in the environment). No separate deploy step is involved (the old deploy-to-publish flow was retired in a platform change, 2026-08). Frontmatter controls what appears publicly.
|
||||||
|
|
||||||
### Docs check triggers
|
### Docs check triggers
|
||||||
|
|
||||||
|
|
@ -161,9 +161,9 @@ npm run release:major # Breaking changes (rare, manual decision)
|
||||||
The script: verifies clean git state, builds, tests, bumps version, updates CHANGELOG.md, commits, tags, pushes, publishes to npm, and creates a GitHub release.
|
The script: verifies clean git state, builds, tests, bumps version, updates CHANGELOG.md, commits, tags, pushes, publishes to npm, and creates a GitHub release.
|
||||||
|
|
||||||
After a successful release, remind the user:
|
After a successful release, remind the user:
|
||||||
> "Published. Deploy portal to pick up the new docs → go to the portal project and deploy."
|
> "Published. Docs are live on soulcraft.com/docs (pushed via the ingest API during the release) — spot-check a changed page with curl."
|
||||||
|
|
||||||
Do NOT deploy portal from here. Portal is always deployed separately from within the portal project.
|
There is no separate deploy step anymore. If the docs push failed (the script warns loudly), re-run `node scripts/push-docs.js` with `DOCS_INGEST_SECRET` set.
|
||||||
|
|
||||||
## Closed-Source Product Names — HARD RULE
|
## Closed-Source Product Names — HARD RULE
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -6,7 +6,7 @@ may find elsewhere in the repo's history.
|
||||||
|
|
||||||
## Where the project lives
|
## Where the project lives
|
||||||
|
|
||||||
The source of truth is a self-hosted forge: **source.soulcraft.com/soulcraft/brainy**.
|
The source of truth is a self-hosted forge: **source.soulcraft.com/soulcraftlabs/open-brainy**.
|
||||||
It's anonymously readable and cloneable — no account needed to browse, clone,
|
It's anonymously readable and cloneable — no account needed to browse, clone,
|
||||||
or build.
|
or build.
|
||||||
|
|
||||||
|
|
@ -31,7 +31,7 @@ fine) to talk through the approach saves everyone rework.
|
||||||
## Development setup
|
## Development setup
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
git clone https://source.soulcraft.com/soulcraft/brainy.git
|
git clone https://source.soulcraft.com/soulcraftlabs/open-brainy.git
|
||||||
cd brainy
|
cd brainy
|
||||||
npm install
|
npm install
|
||||||
npm run build
|
npm run build
|
||||||
|
|
@ -41,6 +41,20 @@ npm test
|
||||||
Tests run on [Vitest](https://vitest.dev/). `npm test` runs the unit suite;
|
Tests run on [Vitest](https://vitest.dev/). `npm test` runs the unit suite;
|
||||||
see `package.json` for `test:integration`, `test:coverage`, and friends.
|
see `package.json` for `test:integration`, `test:coverage`, and friends.
|
||||||
|
|
||||||
|
## Test gate
|
||||||
|
|
||||||
|
The release gate is a bare `vitest run` (no `--config` flag) — the same
|
||||||
|
command the delta gate and CI's checks invoke. It carries the full
|
||||||
|
correctness suite and nothing else: wall-clock/scale benchmarks
|
||||||
|
(`tests/performance/**`, `tests/critical-performance-benchmark.test.ts`,
|
||||||
|
`tests/api/performance-benchmarks.test.ts`) and the two tests whose outcome
|
||||||
|
depends on the host machine or network rather than the code
|
||||||
|
(`tests/package-size-limit.test.ts` shells out to the `npm` CLI;
|
||||||
|
`tests/model-loading.test.ts` makes a real network call to download a model)
|
||||||
|
are excluded from it, because a timing threshold or a flaky network call has
|
||||||
|
no business failing a correctness check. That whole family runs on demand,
|
||||||
|
in its own exclusive slot, via `npm run test:perf`.
|
||||||
|
|
||||||
## Standards
|
## Standards
|
||||||
|
|
||||||
- **Strict TypeScript.** No `any` escape hatches to dodge the type checker.
|
- **Strict TypeScript.** No `any` escape hatches to dodge the type checker.
|
||||||
|
|
@ -57,6 +71,17 @@ see `package.json` for `test:integration`, `test:coverage`, and friends.
|
||||||
description states a number, cite the benchmark that produced it (see
|
description states a number, cite the benchmark that produced it (see
|
||||||
[docs/performance-envelopes.md](docs/performance-envelopes.md) for the
|
[docs/performance-envelopes.md](docs/performance-envelopes.md) for the
|
||||||
pattern). Don't state an estimate as if it were measured.
|
pattern). Don't state an estimate as if it were measured.
|
||||||
|
- **Measurements carry numbers, not provenance.** Public commit messages and
|
||||||
|
docs give the SHAPE a number was taken at and never where it was taken: no
|
||||||
|
hostnames, no store or deployment identities, no operational anecdotes about
|
||||||
|
someone's running system. "A 14,056-noun / 72,679-verb production-shaped
|
||||||
|
store, measured solo under an exclusive lock" tells a reader everything the
|
||||||
|
number depends on; the machine it ran on and whose data it was tell them
|
||||||
|
nothing except where somebody's infrastructure lives.
|
||||||
|
- **Documents that answer or reference a confidential specification never enter
|
||||||
|
this repository, even summarized.** The public docs describe THIS engine and
|
||||||
|
the published contract, and nothing else — a summary of a private document is
|
||||||
|
still that document's contents.
|
||||||
|
|
||||||
## License
|
## License
|
||||||
|
|
||||||
|
|
|
||||||
23
README.md
23
README.md
|
|
@ -1,9 +1,14 @@
|
||||||
<p align="center">
|
<p align="center">
|
||||||
<img src="https://source.soulcraft.com/soulcraft/brainy/raw/branch/main/brainy.png" alt="Brainy" width="180">
|
<img src="https://source.soulcraft.com/soulcraftlabs/open-brainy/raw/branch/main/brainy.png" alt="Brainy" width="180">
|
||||||
</p>
|
</p>
|
||||||
|
|
||||||
<h1 align="center">Brainy</h1>
|
<h1 align="center">Brainy</h1>
|
||||||
|
|
||||||
|
> **Frozen at 10.4.13 (2026-09-03).** This repository is the reference implementation of the Brainy store format and API,
|
||||||
|
> published under the MIT license. Version 10.4.13 is its last release; the repository is read-only from here. The engine
|
||||||
|
> continues as `@soulcraft/brainy`, which bundles this layer as owned code; every published version of this package stays
|
||||||
|
> available on The Source. Use this repository to read a Brainy store independently or to verify the conformance contract.
|
||||||
|
|
||||||
<p align="center">
|
<p align="center">
|
||||||
<b>Three database paradigms. One API. Zero configuration.</b><br>
|
<b>Three database paradigms. One API. Zero configuration.</b><br>
|
||||||
The in-process knowledge database for TypeScript — vector search, graph traversal,<br>
|
The in-process knowledge database for TypeScript — vector search, graph traversal,<br>
|
||||||
|
|
@ -11,9 +16,9 @@
|
||||||
</p>
|
</p>
|
||||||
|
|
||||||
<p align="center">
|
<p align="center">
|
||||||
<a href="https://www.npmjs.com/package/@soulcraft/brainy"><img src="https://img.shields.io/npm/v/@soulcraft/brainy.svg" alt="npm version"></a>
|
<a href="https://source.soulcraft.com/soulcraftlabs/-/packages/npm/brainy"><img src="https://img.shields.io/badge/package-The%20Source-2c3e50.svg" alt="Package on The Source"></a>
|
||||||
<a href="https://www.npmjs.com/package/@soulcraft/brainy"><img src="https://img.shields.io/npm/dm/@soulcraft/brainy.svg" alt="npm downloads"></a>
|
<a href="https://source.soulcraft.com/soulcraftlabs/open-brainy"><img src="https://img.shields.io/badge/repo-open--brainy-2c3e50.svg" alt="Repository"></a>
|
||||||
<a href="https://source.soulcraft.com/soulcraft/brainy/actions"><img src="https://source.soulcraft.com/soulcraft/brainy/actions/workflows/ci.yml/badge.svg?branch=main" alt="CI"></a>
|
<a href="https://source.soulcraft.com/soulcraftlabs/open-brainy/actions"><img src="https://source.soulcraft.com/soulcraftlabs/open-brainy/actions/workflows/ci.yml/badge.svg?branch=main" alt="CI"></a>
|
||||||
<a href="https://soulcraft.com/docs"><img src="https://img.shields.io/badge/docs-soulcraft.com-blue.svg" alt="Documentation"></a>
|
<a href="https://soulcraft.com/docs"><img src="https://img.shields.io/badge/docs-soulcraft.com-blue.svg" alt="Documentation"></a>
|
||||||
<a href="LICENSE"><img src="https://img.shields.io/badge/license-MIT-blue.svg" alt="MIT License"></a>
|
<a href="LICENSE"><img src="https://img.shields.io/badge/license-MIT-blue.svg" alt="MIT License"></a>
|
||||||
<a href="https://www.typescriptlang.org/"><img src="https://img.shields.io/badge/%3C%2F%3E-TypeScript-%230074c1.svg" alt="TypeScript"></a>
|
<a href="https://www.typescriptlang.org/"><img src="https://img.shields.io/badge/%3C%2F%3E-TypeScript-%230074c1.svg" alt="TypeScript"></a>
|
||||||
|
|
@ -30,6 +35,8 @@
|
||||||
|
|
||||||
---
|
---
|
||||||
|
|
||||||
|
**Open Brainy** is the MIT engine — the open API, client library, types, and protocol; an openly specified canonical on-disk format; and this TypeScript reference engine, scoped as a single-node engine for stores up to roughly one million rows. `@soulcraft/brainy` 10.4.2 was the last release under the old package name — the name passes to the native engine, **Brainy**, at 11.0.0: the same API over the same open format at production scale, and it requires a license.
|
||||||
|
|
||||||
Built because we were tired of stitching a vector store to a graph database to a document store — and spending weeks on plumbing before writing a line of business logic. Brainy indexes every fact **three ways at once** and lets one call query them together:
|
Built because we were tired of stitching a vector store to a graph database to a document store — and spending weeks on plumbing before writing a line of business logic. Brainy indexes every fact **three ways at once** and lets one call query them together:
|
||||||
|
|
||||||
| You write | Brainy indexes it as | You query it with |
|
| You write | Brainy indexes it as | You query it with |
|
||||||
|
|
@ -45,12 +52,14 @@ It runs **inside your process** — no server, no Docker, nothing to operate —
|
||||||
## Quick start
|
## Quick start
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
bun add @soulcraft/brainy # Bun ≥ 1.1 — recommended
|
bun add @soulcraftlabs/brainy # Bun ≥ 1.1 — recommended
|
||||||
npm install @soulcraft/brainy # Node.js ≥ 22
|
npm install @soulcraftlabs/brainy # Node.js ≥ 22
|
||||||
```
|
```
|
||||||
|
|
||||||
|
> **Registry**: add `@soulcraftlabs:registry=https://source.soulcraft.com/api/packages/soulcraftlabs/npm/` to your `.npmrc` (anonymous read).
|
||||||
|
|
||||||
```javascript
|
```javascript
|
||||||
import { Brainy, NounType, VerbType } from '@soulcraft/brainy'
|
import { Brainy, NounType, VerbType } from '@soulcraftlabs/brainy'
|
||||||
|
|
||||||
const brain = new Brainy() // in-memory; one line swaps to disk
|
const brain = new Brainy() // in-memory; one line swaps to disk
|
||||||
await brain.init()
|
await brain.init()
|
||||||
|
|
|
||||||
336
RELEASES.md
336
RELEASES.md
|
|
@ -1,7 +1,17 @@
|
||||||
# @soulcraft/brainy — Release Notes for Consumers
|
# @soulcraft/brainy — Release Notes for Consumers
|
||||||
|
|
||||||
|
> **Frozen at 10.4.13 (2026-09-03).** 10.4.13 is the last release of `@soulcraftlabs/brainy`; this repository is read-only from here.
|
||||||
|
> Release notes for the product engine continue on its own wall.
|
||||||
|
|
||||||
|
Machine-readable release notes are published at
|
||||||
|
https://source.soulcraft.com/soulcraftlabs/releases/raw/branch/main/open-brainy.json
|
||||||
|
(this engine) and
|
||||||
|
https://source.soulcraft.com/soulcraftlabs/releases/raw/branch/main/brainy.json
|
||||||
|
(the product engine) — read by HQ's `/hq/releases` door, and the source of
|
||||||
|
truth ahead of this file.
|
||||||
|
|
||||||
This file is the **quick reference for downstream sessions** tracking Brainy changes.
|
This file is the **quick reference for downstream sessions** tracking Brainy changes.
|
||||||
Full auto-generated changelog: `CHANGELOG.md` · Releases: https://source.soulcraft.com/soulcraft/brainy/releases
|
Full auto-generated changelog: `CHANGELOG.md` · Releases: https://source.soulcraft.com/soulcraftlabs/open-brainy/releases
|
||||||
|
|
||||||
**How to use:** Brainy is the underlying data engine for downstream applications. Read this when:
|
**How to use:** Brainy is the underlying data engine for downstream applications. Read this when:
|
||||||
- Upgrading `@soulcraft/brainy` in your application
|
- Upgrading `@soulcraft/brainy` in your application
|
||||||
|
|
@ -31,6 +41,330 @@ is sometimes cited as a 7.x removal — those methods never existed on 7.x; the
|
||||||
|
|
||||||
---
|
---
|
||||||
|
|
||||||
|
## v10.4.4 — 2026-08-28
|
||||||
|
|
||||||
|
**A correctness and observability release.** The headline is not speed: it is that a
|
||||||
|
restart now tells you the truth about itself, a store stops lying about how much it
|
||||||
|
holds, and the engine stops doing work nobody asked for. There is a performance
|
||||||
|
improvement and it is modest; it is stated exactly below rather than rounded up.
|
||||||
|
|
||||||
|
### The dark restart — fixed at the root
|
||||||
|
|
||||||
|
A service could stop cleanly, exit 0, having awaited `close()` on every store it held,
|
||||||
|
and its next boot would announce `Overwriting stale writer lock … appears dead` for
|
||||||
|
every one of them. Nothing had crashed. Two deployments hit this; the same defect also
|
||||||
|
made those boots pay a crash-recovery fold they did not owe.
|
||||||
|
|
||||||
|
The cause was not the lock. `close()` released it correctly — when it got there. A
|
||||||
|
failure part-way through close skipped both the release AND the clean-shutdown marker,
|
||||||
|
and "the recorded pid is gone" reads identically for an orderly restart and a crash.
|
||||||
|
|
||||||
|
- `close()` is now two parts and the second is unconditional: the flush-request watcher,
|
||||||
|
the **writer lock**, the VFS timers and the terminal `closed` flag are released whether
|
||||||
|
the durable steps succeeded or not. The original failure is narrated with what it costs
|
||||||
|
the next open, then rethrown.
|
||||||
|
- Releasing the lock writes a **clean-close record** naming the lock generation it gave
|
||||||
|
up. The next open reads that record instead of guessing: recorded → nothing to recover;
|
||||||
|
absent → it says so, and names the recovery it is about to run. This also ends two
|
||||||
|
long-standing false alarms — a recycled pid locking a store out of its own reopen, and
|
||||||
|
`Re-acquiring writer lock … this is a bug` after a perfectly clean close.
|
||||||
|
- The signal path stopped failing in a batch. One store's failing flush used to strand
|
||||||
|
every remaining store's lock and markers — at exit code 0. Now: per-store isolation, the
|
||||||
|
generation store's close (the marker) is part of shutdown, the lock goes in a `finally`,
|
||||||
|
and the handler no longer calls `process.exit()` when the host application has its own
|
||||||
|
signal handler, a race that truncated the host's own shutdown mid-flight.
|
||||||
|
|
||||||
|
### The count ledger stops lying, and `counts.json` is written atomically
|
||||||
|
|
||||||
|
The all-tier scalars are the denominator a coverage check subtracts against. A ledger
|
||||||
|
derived under the old rule — one entity per id DIRECTORY — counted ghost and scar
|
||||||
|
containers as rows, and was only FLAGGED suspect: it went on serving wrong numbers for
|
||||||
|
the life of the store. Two copies of one archive could disagree, and a downstream index
|
||||||
|
heal reported remaining work that did not exist.
|
||||||
|
|
||||||
|
- Such a ledger now derives itself honestly **in the background** after the open, counting
|
||||||
|
identity records, and persists the correction stamped. Nothing waits for it, because no
|
||||||
|
read is served from a denominator.
|
||||||
|
- A derivation that raced a write refuses to stamp its number: one retry on a quiet store,
|
||||||
|
then the ledger stays SUSPECT and names `repairIndex()` as the door that recounts under
|
||||||
|
a barrier.
|
||||||
|
- `counts.json` is written temp+rename. A truncating write left a window in which a
|
||||||
|
concurrent reader saw the file EMPTY — and an unparseable ledger sends the next open
|
||||||
|
down the full-rescan path, so the cheapest file in the store was buying the most
|
||||||
|
expensive recovery.
|
||||||
|
|
||||||
|
### An open and a repair narrate themselves — on a channel a log level cannot silence
|
||||||
|
|
||||||
|
A store could open for three minutes and print nothing at all. The phase timings existed;
|
||||||
|
they were written to a channel that every production-looking environment clamps away.
|
||||||
|
|
||||||
|
- Narration moved to an always-visible channel. An open now heartbeats the phase it is in,
|
||||||
|
names each phase as it ends with what it was paying for, and names the expensive STEP
|
||||||
|
inside a phase. `repairIndex()` does the same and its receipt carries a per-family
|
||||||
|
`durationMs` — a repair that ran for half an hour with no output could only be watched
|
||||||
|
through `top`.
|
||||||
|
- A brain nobody has written to now does nothing: a flush over a clean store is a no-op
|
||||||
|
and says nothing, the graph index's auto-flush asks before it acts, and the
|
||||||
|
cross-process flush-request watch is **event-driven** (`fs.watch`) instead of polling a
|
||||||
|
directory every 500 ms per store forever, with a slow safety sweep behind it and a
|
||||||
|
narrated fall back to polling where a filesystem cannot be watched.
|
||||||
|
- A provider that is REBUILDING ITSELF is no longer confused with a broken one. `init()`
|
||||||
|
does not wait for it, every other family serves, and that family's doors refuse **by
|
||||||
|
name, carrying the provider's own progress**, saying plainly that they open by
|
||||||
|
themselves and no action is needed. Health narration dedupes by content, so an unchanged
|
||||||
|
verdict is silent however a provider's generation counter moves.
|
||||||
|
|
||||||
|
### For operators — one behaviour change
|
||||||
|
|
||||||
|
**Four `where` operators that previously returned an empty page now raise
|
||||||
|
`INVALID_QUERY`:** `startsWith`, `endsWith`, `matches` and `length`. An equality/range
|
||||||
|
posting index cannot evaluate a substring, a pattern or an array length without reading
|
||||||
|
every row, and it now refuses by name instead of answering with an empty result that
|
||||||
|
looks like an answer.
|
||||||
|
|
||||||
|
**Three that previously returned an empty page are now SERVED:** `hasAll`, `noneOf` and
|
||||||
|
`excludes`. All 25 accepted operator tokens now agree between this engine and its
|
||||||
|
accelerated counterpart.
|
||||||
|
|
||||||
|
### Performance — stated exactly
|
||||||
|
|
||||||
|
Measured on a 14,056-noun / 72,679-verb production-shaped store, both builds solo under
|
||||||
|
an exclusive lock:
|
||||||
|
|
||||||
|
- **Warm reopen after a clean close: 85.7 s → 77.0 s (−10.2%).** The whole of that gain is
|
||||||
|
one fix — generation discovery reads directory NAMES instead of recursively walking the
|
||||||
|
entire generation log (−9.2 s, and it scales with history rather than row count). The
|
||||||
|
VFS phase is **unchanged**.
|
||||||
|
- **Cold open: −31.4 s** (518.1 s → 486.7 s), of which the count-ledger derivation moving
|
||||||
|
off the critical path accounts for storage-init dropping 5,941 ms → 25 ms.
|
||||||
|
- **A dominant ~38 s remains, diagnosed and NOT fixed.** It is not the VFS — the VFS's own
|
||||||
|
init is under 2 s of that phase. It is the log-authority adoption and/or the
|
||||||
|
pending-embed log recovery, both now instrumented so the next measurement names the
|
||||||
|
culprit outright.
|
||||||
|
|
||||||
|
Continuing work, named so nobody has to rediscover it: that ~38 s term; making the
|
||||||
|
generation store's committed-range set lazy; the hydration path that substitutes
|
||||||
|
`Date.now()` for an unreadable stored timestamp (inventing data); and a VFS path-prefix
|
||||||
|
filter built with a `$startsWith` spelling no operator set accepts, so
|
||||||
|
`searchFiles({ path })` throws today.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## v10.4.3 — 2026-08-27 (Open Brainy's first release)
|
||||||
|
|
||||||
|
**`@soulcraftlabs/brainy` 10.4.3 is the same engine as `@soulcraft/brainy` 10.4.2, byte for
|
||||||
|
byte — only the name, the registry, and the pointers changed.** Install:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
npm install @soulcraftlabs/brainy
|
||||||
|
```
|
||||||
|
|
||||||
|
with the registry line in your `.npmrc` (anonymous read):
|
||||||
|
|
||||||
|
```
|
||||||
|
@soulcraftlabs:registry=https://source.soulcraft.com/api/packages/soulcraftlabs/npm/
|
||||||
|
```
|
||||||
|
|
||||||
|
- **The Source is the one registry.** Open Brainy publishes to source.soulcraft.com only; the
|
||||||
|
npmjs republish step is retired from the release rail. Existing npmjs versions of
|
||||||
|
`@soulcraft/brainy` stay as they are and receive no new versions.
|
||||||
|
- **The repository moved** to `soulcraftlabs/open-brainy` on The Source; the old path redirects.
|
||||||
|
- **No engine change.** Everything in the 10.4.2 notes applies unchanged; adoption is one
|
||||||
|
install-line change (`@soulcraft/brainy` → `@soulcraftlabs/brainy`), which downstream
|
||||||
|
applications make together with their native-engine bump.
|
||||||
|
|
||||||
|
## v10.4.2 — 2026-08-27 (a zero-norm vector is not a vector)
|
||||||
|
|
||||||
|
**This is the last release of the MIT engine under the `@soulcraft/brainy` name.**
|
||||||
|
The MIT package continues as **Open Brainy** — `@soulcraftlabs/brainy`: the open API,
|
||||||
|
client library, types and protocol, an openly specified canonical format, and the TypeScript
|
||||||
|
reference engine, scoped honestly as a single-node engine for stores up to roughly one
|
||||||
|
million rows. The `@soulcraft/brainy` name passes to the native engine, **Brainy**, at a
|
||||||
|
major version bump; that engine implements the same API over the same open format at
|
||||||
|
production scale, requires a license, and refuses loudly without one. Nothing changes
|
||||||
|
for existing installs until that major ships; the move is announced with it.
|
||||||
|
|
||||||
|
Six fixes, one law: a vector with no magnitude carries no information, so it must
|
||||||
|
never reach a vector index — in any engine — and the canonical store must say so.
|
||||||
|
|
||||||
|
- **The permanently-unvectored row.** `add({ ..., vector: [] })` (and the same item
|
||||||
|
shape in `addMany` / `transact`) is now the sanctioned "no vector" row: persisted
|
||||||
|
with an empty vector leg, never embedded, never indexed, counted as unvectored in
|
||||||
|
the canonical ledger. Metadata-only rows — telemetry tallies, counters, plumbing —
|
||||||
|
no longer need a placeholder vector and never enter the vector leg. `vector: []`
|
||||||
|
together with `deferEmbedding: true` is refused with a typed error (a supplied
|
||||||
|
vector has nothing to defer). Previously `vector: []` threw a dimension error.
|
||||||
|
- **The unvector door.** `update({ id, vector: [] })` (and its `transact()` twin) is
|
||||||
|
the sanctioned way to strip a vector from an existing row: canonical vector → `[]`,
|
||||||
|
removal from the vector index, the vectored ledger decremented exactly once — and
|
||||||
|
idempotent, so a resumed cleanup pass may simply re-issue. It never re-embeds, and
|
||||||
|
it clears a pending deferred-embed marker durably so the background worker cannot
|
||||||
|
re-vector the row later. Note that a rebuild never sheds vectors (it re-derives the
|
||||||
|
index from canonical rows); shedding historical vectors needs this door.
|
||||||
|
- **Zero-norm vectors are normalized at the write.** An explicit all-zero vector on
|
||||||
|
any write path is persisted as unvectored (`[]`) with one warning naming the row;
|
||||||
|
the vector-index operations keep their own refusal as a second line. The engine's
|
||||||
|
own VFS root, which used to persist a deliberate all-zero placeholder (harmless
|
||||||
|
under cosine distance, a false attractor under a downstream engine's
|
||||||
|
squared-euclidean serving — a production incident this week), is now created
|
||||||
|
unvectored, and an existing store's legacy root is migrated on open by a single
|
||||||
|
fixed-path read before the health gate runs — never a walk.
|
||||||
|
- **Enumeration keys on the identity record.** `getNouns()` / `getVerbs()` and the
|
||||||
|
cursor walks behind them enumerate by the metadata record, the same key the
|
||||||
|
canonical ledger counts by — previously the walk keyed on the vector file, so a
|
||||||
|
row holding metadata but no vector was counted yet never yielded (a permanent
|
||||||
|
"missing" phantom in coverage math), while an orphaned vector-only directory
|
||||||
|
could be yielded as a phantom id. The recovery fold also never deletes an existing
|
||||||
|
vector when it replays a metadata-only after-image (preserve-if-absent). One
|
||||||
|
documented gap remains: a verb's endpoints live only in its vector leg, so a
|
||||||
|
metadata-only verb is counted and loudly skipped, never fabricated — the fix is a
|
||||||
|
canonical-format change and lands with the open format.
|
||||||
|
- **The ledger's one-time derivation counts identity records.** Stores upgraded from
|
||||||
|
pre-ledger versions derived their ALL-visibility scalars once by counting id
|
||||||
|
directories, which included ghost and scar containers left by an old partial-delete
|
||||||
|
defect — an inflated denominator whose coverage row could never reach exact. The
|
||||||
|
derivation now counts only directories holding a metadata record, `counts.json`
|
||||||
|
carries a derivation-rule stamp, and a ledger derived under the old rule is marked
|
||||||
|
`suspect` at open (one O(1) field read, one warning) so the online `repairIndex()`
|
||||||
|
path clears it with a real recount.
|
||||||
|
- **The vector index refuses what it cannot hold.** `rebuild()` skips unvectored and
|
||||||
|
zero-norm rows (one summary line), re-pins the vector dimension from the first real
|
||||||
|
vector after a restart (previously a restart left the pin unset, so a wrong-length
|
||||||
|
insert became the new pin instead of being rejected), and `addItem` / `updateItem`
|
||||||
|
throw a typed `EmptyVectorIndexError` on a length-0 vector instead of ever storing
|
||||||
|
a vector-less node.
|
||||||
|
- **Smaller:** a failing plugin activation now rethrows with the original error as
|
||||||
|
`cause` (the originating file and line survive to the caller's log); build
|
||||||
|
generators stamp from the repository history of their inputs instead of wall clock,
|
||||||
|
so two builds of the same tree are byte-identical.
|
||||||
|
|
||||||
|
Adoption: one restart, paired with its native-engine release. The first open of an
|
||||||
|
existing store runs the legacy-root migration (one narrated line) and, on stores that
|
||||||
|
upgraded from pre-ledger versions, marks the ledger suspect until the next sanctioned
|
||||||
|
recount — no rebuild in either case.
|
||||||
|
|
||||||
|
## v10.4.1 — 2026-08-26 (reads refuse per family; an unchanged write never re-embeds)
|
||||||
|
|
||||||
|
Two production defects from the same week, fixed together as a patch to 10.4.0.
|
||||||
|
|
||||||
|
- **The read gate is per family.** A read now refuses only when the index family it
|
||||||
|
actually consults is unhealthy: a metadata filter is served while the vector leg is
|
||||||
|
rebuilding; a semantic query is refused only by the vector family; a graph
|
||||||
|
traversal only by the graph family. Previously any unhealthy family refused every
|
||||||
|
read on the brain — under a long vector rebuild, a production deployment's
|
||||||
|
metadata-only reads were refused for the duration, and the retries became a write
|
||||||
|
pump of their own.
|
||||||
|
- **Unchanged data never re-embeds.** `update()` compares the incoming `data`
|
||||||
|
structurally with the stored record; an update carrying identical data (a common
|
||||||
|
shape for periodic upserts) no longer embeds again and no longer churns the vector
|
||||||
|
leg. Previously every such update re-embedded and re-inserted, which under load
|
||||||
|
saturated the vector index with near-identical vectors.
|
||||||
|
|
||||||
|
Adoption: one restart, paired with its native-engine release.
|
||||||
|
|
||||||
|
## v10.4.0 — 2026-08-25 (the health report has a name)
|
||||||
|
|
||||||
|
Three related cures, one root cause: an index deciding whether it could be trusted
|
||||||
|
by sampling itself instead of by exact accounting. This release replaces every
|
||||||
|
sampled self-probe with ledger-derived truth, and a read against an unhealthy index
|
||||||
|
now refuses loudly instead of guessing.
|
||||||
|
|
||||||
|
- **The canonical count ledger.** Storage now tracks two scalars per family
|
||||||
|
(nouns/verbs) on the write path: the user-facing `counted` total — unchanged,
|
||||||
|
still what `getNounCount()` / `getVerbCount()` return — and a new ALL-visibility
|
||||||
|
`all` total covering every tier, the real denominator a derived index's own
|
||||||
|
coverage math needs. The unfiltered storage-level `totalCount` returned by
|
||||||
|
`getNouns()` / `getVerbs()` is now this unclamped ALL scalar; previously it could
|
||||||
|
only ever move up (`Math.max(scalar, scanned)`), so an inflated counter could
|
||||||
|
never self-correct. A delete that cannot prove the record it removed actually
|
||||||
|
existed (no canonical read, no prior image available) no longer decrements on
|
||||||
|
faith — it marks the ledger `suspect` (narrated once per session) instead of
|
||||||
|
silently drifting, and the next `repairIndex()` clears the flag with a real
|
||||||
|
recount.
|
||||||
|
- **One contract for a throwing health probe.** A provider's `validateInvariants()`
|
||||||
|
is documented to never throw — but if one does anyway (a bug, a transient fault),
|
||||||
|
it is now read the same way everywhere: `heal: 'none'`, the error named in the
|
||||||
|
report, never synthesized into a rebuild trigger and never swallowed into "looks
|
||||||
|
fine." A flaky check can no longer buy itself a rebuild. `repairIndex()`'s
|
||||||
|
per-family receipt also gains `missing` (an exact count plus a capped id sample),
|
||||||
|
`rebuilt` (a full rebuild ran, vs. an incremental heal), and `reason`.
|
||||||
|
- **The named health report; reads refuse instead of rebuilding.** Any index
|
||||||
|
provider may now expose a synchronous, O(1) `healthReport()` — composed from the
|
||||||
|
provider's own exact ledgers, never a sample — and this is the one signal
|
||||||
|
Brainy's read gate trusts. The first-query lazy-build path is gone: `brain.init()`
|
||||||
|
now runs every needed rebuild to completion before it returns, always, regardless
|
||||||
|
of dataset size. A read that lands on a provider whose health report says it
|
||||||
|
isn't serving throws a typed error instead of triggering a rebuild mid-query —
|
||||||
|
`GraphIndexNotReadyError`, `MetadataIndexNotReadyError`, or
|
||||||
|
`VectorIndexNotReadyError` (all exported from `@soulcraft/brainy`), naming the
|
||||||
|
reasons. `repairIndex({ rebuild: ['metadata' | 'graph' | 'vector'] | 'all' })` is
|
||||||
|
the new explicit operator door: it rebuilds the named family unconditionally, no
|
||||||
|
health check consulted — reach for it when you have independent reason to
|
||||||
|
distrust a family regardless of what it self-reports. Bare `repairIndex()` is
|
||||||
|
unchanged in spirit: report-driven, heals only what its own checks say needs it.
|
||||||
|
- New concept doc: [Index Health](docs/concepts/index-health.md) walks the whole
|
||||||
|
story from a consumer's side — degraded-but-serving vs. not-ready, what
|
||||||
|
`repairIndex()` checks and heals per family, what `suspect` counts mean.
|
||||||
|
|
||||||
|
**Nothing to change to adopt this.** No API removed, no signature narrowed —
|
||||||
|
`repairIndex()` gains an optional options bag and its return value gains fields,
|
||||||
|
both additive. The honest notes: if your code ever relied on a `find()` against a
|
||||||
|
cold/not-yet-built index quietly triggering a rebuild and returning results a beat
|
||||||
|
later, that behavior is gone — it now throws one of the three typed
|
||||||
|
`*NotReadyError` classes instead (catch them if you need to distinguish "not ready
|
||||||
|
yet" from "no results"). And `disableAutoRebuild: true` no longer defers index
|
||||||
|
construction to the first query — a needed rebuild always runs at `open()` now;
|
||||||
|
the flag has no effect on timing. Full manual control still lives in
|
||||||
|
`repairIndex({ rebuild: [...] })`.
|
||||||
|
|
||||||
|
- **Crash-reopen catchup.** After an unclean shutdown, the metadata index now
|
||||||
|
folds the exact fact window it missed — `find()` serves every acked write on
|
||||||
|
reopen, closing the gap where canonical reads and counts recovered a
|
||||||
|
crash-window write but the index kept serving its pre-crash state until the
|
||||||
|
next full rebuild. Related root-cause fixed alongside: `close()` never
|
||||||
|
stamped the index watermarks (only `flush()` did), so a close without a
|
||||||
|
prior flush caused a needless full rescan verdict on the next open.
|
||||||
|
- **Relation rows are live in the metadata index.** Previously verb rows
|
||||||
|
entered the metadata index only during a rebuild — so a rebuilt store's
|
||||||
|
relation postings went stale from the first `relate()` after it. Relations
|
||||||
|
are now posted and retracted on the live write path (relate / unrelate /
|
||||||
|
updateRelation / remove's cascade, and their `transact()` forms), in the
|
||||||
|
same commit as the graph leg.
|
||||||
|
- **The metadata rebuild is online.** `rebuild()` for the metadata family no
|
||||||
|
longer clears and rebuilds in place (reads went empty for the duration): it
|
||||||
|
builds a complete replacement beside the serving index, mirrors concurrent
|
||||||
|
writes to both, swaps atomically, and persists once after the swap. Reads
|
||||||
|
never observe a partial index. `repairIndex({ rebuild: ['metadata'] })` uses
|
||||||
|
it automatically.
|
||||||
|
- **Incremental heal is routed.** A provider invariant that asks for the
|
||||||
|
incremental heal (`heal: 'repair'`) now routes to the provider's own
|
||||||
|
`repair()` when it exposes one — re-posting exactly what its ledger names,
|
||||||
|
never a store-sized rebuild — and the post-heal re-read of the report decides
|
||||||
|
success; a repair that doesn't converge is recorded with the escalation named.
|
||||||
|
- **The vector family joins the count ledger.** `getCanonicalCounts()` gains
|
||||||
|
`vectors: { all }` — the count of canonical entities holding a real vector
|
||||||
|
(deferred-embed entities count when their vector lands). And the open gate
|
||||||
|
closes the vector leg: a store whose canonical rows hold vectors but whose
|
||||||
|
derived vector index is empty now builds at `open()` (or refuses with the
|
||||||
|
typed error) instead of silently serving empty vector-search results.
|
||||||
|
- **An unknown storage config shape fails loudly.** A nested `config` object
|
||||||
|
carrying a path-shaped key (a shape that was never supported) used to fall
|
||||||
|
through silently to the default shared directory — every instance writing one
|
||||||
|
store while callers believed each had its own. It now throws, naming the
|
||||||
|
canonical `path` key.
|
||||||
|
- **Relation index rows are JSON-safe.** Internal endpoint identifiers can no
|
||||||
|
longer ride the metadata-index crossing (a native provider serializes it);
|
||||||
|
they stay on the graph operations where they belong.
|
||||||
|
- **A broken accelerator install can never read as "not installed."** The
|
||||||
|
auto-detection free pass now requires the resolution error to name the
|
||||||
|
accelerator package itself, exactly — a missing platform-binary sibling
|
||||||
|
package, an inner file path, or a dependency failure is a broken install and
|
||||||
|
`init()` throws loudly. And a plugin that declines activation is narrated on
|
||||||
|
the always-on log channel, so `silent: true` can no longer hide a fallback
|
||||||
|
to the default engines.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
## v10.3.1 — 2026-08-18 (the fold that behaves)
|
## v10.3.1 — 2026-08-18 (the fold that behaves)
|
||||||
|
|
||||||
Three recovery cures from one production first-boot incident (a brain's first
|
Three recovery cures from one production first-boot incident (a brain's first
|
||||||
|
|
|
||||||
|
|
@ -30,7 +30,7 @@ commit to backporting fixes to unsupported lines.
|
||||||
|
|
||||||
## Scope
|
## Scope
|
||||||
|
|
||||||
This policy covers the `@soulcraft/brainy` package itself — the code in
|
This policy covers the `@soulcraftlabs/brainy` package itself — the code in
|
||||||
this repository. If you're evaluating a deployment that also uses
|
this repository. If you're evaluating a deployment that also uses
|
||||||
`@soulcraft/cor`, report issues in that package the same way, to the same
|
`@soulcraft/cor`, report issues in that package the same way, to the same
|
||||||
address; we'll route internally.
|
address; we'll route internally.
|
||||||
|
|
|
||||||
|
|
@ -3,7 +3,7 @@
|
||||||
/**
|
/**
|
||||||
* Modern TypeScript CLI Runner
|
* Modern TypeScript CLI Runner
|
||||||
*
|
*
|
||||||
* This is the entry point after npm install @soulcraft/brainy
|
* This is the entry point after npm install @soulcraftlabs/brainy
|
||||||
* It runs the compiled TypeScript CLI code
|
* It runs the compiled TypeScript CLI code
|
||||||
*/
|
*/
|
||||||
|
|
||||||
|
|
|
||||||
2
bun.lock
2
bun.lock
|
|
@ -3,7 +3,7 @@
|
||||||
"configVersion": 0,
|
"configVersion": 0,
|
||||||
"workspaces": {
|
"workspaces": {
|
||||||
"": {
|
"": {
|
||||||
"name": "@soulcraft/brainy",
|
"name": "@soulcraftlabs/brainy",
|
||||||
"dependencies": {
|
"dependencies": {
|
||||||
"@aws-sdk/client-s3": "^3.540.0",
|
"@aws-sdk/client-s3": "^3.540.0",
|
||||||
"@azure/identity": "^4.0.0",
|
"@azure/identity": "^4.0.0",
|
||||||
|
|
|
||||||
|
|
@ -25,13 +25,13 @@
|
||||||
|
|
||||||
### Prerequisites
|
### Prerequisites
|
||||||
```bash
|
```bash
|
||||||
npm install @soulcraft/brainy
|
npm install @soulcraftlabs/brainy
|
||||||
```
|
```
|
||||||
|
|
||||||
### Your First Neural Database
|
### Your First Neural Database
|
||||||
|
|
||||||
```typescript
|
```typescript
|
||||||
import { Brainy, NounType } from '@soulcraft/brainy'
|
import { Brainy, NounType } from '@soulcraftlabs/brainy'
|
||||||
|
|
||||||
// Step 1: Create and initialize Brainy
|
// Step 1: Create and initialize Brainy
|
||||||
const brain = new Brainy({
|
const brain = new Brainy({
|
||||||
|
|
@ -143,7 +143,7 @@ Once you're comfortable with basic operations, move to **Level 2** to learn abou
|
||||||
### Building a Knowledge Graph
|
### Building a Knowledge Graph
|
||||||
|
|
||||||
```typescript
|
```typescript
|
||||||
import { Brainy, NounType, VerbType } from '@soulcraft/brainy'
|
import { Brainy, NounType, VerbType } from '@soulcraftlabs/brainy'
|
||||||
|
|
||||||
const brain = new Brainy({ storage: { type: 'memory' } })
|
const brain = new Brainy({ storage: { type: 'memory' } })
|
||||||
await brain.init()
|
await brain.init()
|
||||||
|
|
@ -314,7 +314,7 @@ Ready for AI-powered search and clustering? Move to **Level 3**.
|
||||||
### Triple Intelligence in Action
|
### Triple Intelligence in Action
|
||||||
|
|
||||||
```typescript
|
```typescript
|
||||||
import { Brainy, NounType, VerbType } from '@soulcraft/brainy'
|
import { Brainy, NounType, VerbType } from '@soulcraftlabs/brainy'
|
||||||
|
|
||||||
const brain = new Brainy({ storage: { type: 'memory' } })
|
const brain = new Brainy({ storage: { type: 'memory' } })
|
||||||
await brain.init()
|
await brain.init()
|
||||||
|
|
@ -529,7 +529,7 @@ Want to treat files as intelligent entities? Learn the **Virtual Filesystem** in
|
||||||
### Files as Intelligent Entities
|
### Files as Intelligent Entities
|
||||||
|
|
||||||
```typescript
|
```typescript
|
||||||
import { Brainy, NounType, VerbType } from '@soulcraft/brainy'
|
import { Brainy, NounType, VerbType } from '@soulcraftlabs/brainy'
|
||||||
|
|
||||||
const brain = new Brainy({ storage: { type: 'memory' } })
|
const brain = new Brainy({ storage: { type: 'memory' } })
|
||||||
await brain.init()
|
await brain.init()
|
||||||
|
|
@ -832,7 +832,7 @@ Ready for production deployment? Level 5 covers **planet-scale architecture**.
|
||||||
### Production-Ready Deployment
|
### Production-Ready Deployment
|
||||||
|
|
||||||
```typescript
|
```typescript
|
||||||
import { Brainy, NounType } from '@soulcraft/brainy'
|
import { Brainy, NounType } from '@soulcraftlabs/brainy'
|
||||||
|
|
||||||
// 1. PRODUCTION STORAGE - Filesystem with off-site snapshots
|
// 1. PRODUCTION STORAGE - Filesystem with off-site snapshots
|
||||||
console.log('Initializing production storage...\n')
|
console.log('Initializing production storage...\n')
|
||||||
|
|
|
||||||
|
|
@ -369,6 +369,71 @@ return results.slice(offset, offset + limit)
|
||||||
// → Auto-correction: Use most likely alternative based on affinity data
|
// → Auto-correction: Use most likely alternative based on affinity data
|
||||||
```
|
```
|
||||||
|
|
||||||
|
## Field Projection (`fields`)
|
||||||
|
|
||||||
|
`find()` and `get()` accept a `fields` list. Without it they return the whole
|
||||||
|
record; with it they return only the fields you name — and, where the index can
|
||||||
|
supply them, without opening the canonical record at all.
|
||||||
|
|
||||||
|
```ts
|
||||||
|
// A list page: two user fields and one engine scalar. No document bodies.
|
||||||
|
await brain.find({
|
||||||
|
where: { kind: 'post' },
|
||||||
|
fields: ['title', 'slug', 'system.createdAt'],
|
||||||
|
limit: 50
|
||||||
|
})
|
||||||
|
|
||||||
|
await brain.get(id, { fields: ['title'] })
|
||||||
|
```
|
||||||
|
|
||||||
|
### Why it exists
|
||||||
|
|
||||||
|
A list view that renders a title and a date does not need the body, but without
|
||||||
|
a projection every row hydrates its full record and throws almost all of it
|
||||||
|
away. On a posts list that is the dominant cost of the query.
|
||||||
|
|
||||||
|
### The rules
|
||||||
|
|
||||||
|
| | |
|
||||||
|
|---|---|
|
||||||
|
| **`fields` absent** | The full record, byte-identical to before. Nothing changes. |
|
||||||
|
| **Field names** | The one addressing law: a bare name is user metadata (`'title'`), `system.*` is an engine scalar (`'system.createdAt'`). |
|
||||||
|
| **A field the row lacks** | Simply **absent** from the result. Never an error. |
|
||||||
|
| **Identity** | Every row keeps its `id` (and `score` on `find`) regardless — a row you cannot identify is not a row. |
|
||||||
|
| **Where values come from** | The **column store**, which holds raw values. Never the sparse index, which buckets timestamps for range queries. |
|
||||||
|
| **A field the column cannot serve** | The canonical record is read for that field only. Correct, just not free. |
|
||||||
|
|
||||||
|
### Missing fields are absent, not errors
|
||||||
|
|
||||||
|
This is deliberate and differs from `orderBy`, which throws
|
||||||
|
`UnresolvableFieldError` for an unknown field. A typo in `orderBy` silently
|
||||||
|
changes the ordering, so it must be loud. A projection asks "give me these if
|
||||||
|
you have them", and an optional field must not turn a list into a failure — so
|
||||||
|
`fields` uses the permissive path.
|
||||||
|
|
||||||
|
### Cost
|
||||||
|
|
||||||
|
When every named field is column-served, a projected page performs **zero**
|
||||||
|
canonical reads. When one is not, only that read happens and the rest still come
|
||||||
|
from the index. Both are pinned by counting reads rather than timing them, in
|
||||||
|
`tests/integration/find-fields-projection.test.ts`.
|
||||||
|
|
||||||
|
### `related()` takes no `fields`
|
||||||
|
|
||||||
|
A `Relation` carries `from` and `to` as **ids** and hydrates no entity record,
|
||||||
|
so there is nothing for a projection to trim. Projecting the endpoints would be
|
||||||
|
a new capability rather than a projection of an existing one.
|
||||||
|
|
||||||
|
### For engine implementers
|
||||||
|
|
||||||
|
Projection is served through an optional provider door,
|
||||||
|
`getScalarsForIds(ids, fields)` on `MetadataIndexProvider`. The contract is in
|
||||||
|
`src/plugin.ts`; the short version is **return only what you can serve exactly,
|
||||||
|
and say what you served**. The caller diffs the answer against the request and
|
||||||
|
reads records for the remainder, so omission costs a read while a wrong value is
|
||||||
|
a wrong answer nobody can see. An engine without the door still works — every
|
||||||
|
field falls back to the record.
|
||||||
|
|
||||||
## Performance Characteristics
|
## Performance Characteristics
|
||||||
|
|
||||||
### Query Performance by Type
|
### Query Performance by Type
|
||||||
|
|
@ -1217,7 +1282,7 @@ where: {
|
||||||
await brain.find({ type: 'Document' })
|
await brain.find({ type: 'Document' })
|
||||||
|
|
||||||
// ✅ Correct: Use NounType enum
|
// ✅ Correct: Use NounType enum
|
||||||
import { NounType } from '@soulcraft/brainy'
|
import { NounType } from '@soulcraftlabs/brainy'
|
||||||
await brain.find({ type: NounType.Document })
|
await brain.find({ type: NounType.Document })
|
||||||
|
|
||||||
// ❌ Error: Operator not recognized
|
// ❌ Error: Operator not recognized
|
||||||
|
|
|
||||||
|
|
@ -153,13 +153,13 @@ brainy-data/
|
||||||
### Step 1: Update Brainy Package
|
### Step 1: Update Brainy Package
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
npm install @soulcraft/brainy@latest
|
npm install @soulcraftlabs/brainy@latest
|
||||||
```
|
```
|
||||||
|
|
||||||
**Check your version:**
|
**Check your version:**
|
||||||
```bash
|
```bash
|
||||||
npm list @soulcraft/brainy
|
npm list @soulcraftlabs/brainy
|
||||||
# Should show: @soulcraft/brainy@4.0.0
|
# Should show: @soulcraftlabs/brainy@4.0.0
|
||||||
```
|
```
|
||||||
|
|
||||||
### Step 2: No Code Changes Required! ✅
|
### Step 2: No Code Changes Required! ✅
|
||||||
|
|
@ -374,7 +374,7 @@ If you encounter issues, you can rollback:
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
# Reinstall v3
|
# Reinstall v3
|
||||||
npm install @soulcraft/brainy@^3.50.0
|
npm install @soulcraftlabs/brainy@^3.50.0
|
||||||
|
|
||||||
# Restart application
|
# Restart application
|
||||||
```
|
```
|
||||||
|
|
@ -389,7 +389,7 @@ rm -rf ./data
|
||||||
cp -r ./data-backup ./data
|
cp -r ./data-backup ./data
|
||||||
|
|
||||||
# Reinstall v3
|
# Reinstall v3
|
||||||
npm install @soulcraft/brainy@^3.50.0
|
npm install @soulcraftlabs/brainy@^3.50.0
|
||||||
```
|
```
|
||||||
|
|
||||||
## Common Migration Scenarios
|
## Common Migration Scenarios
|
||||||
|
|
@ -539,7 +539,7 @@ console.log('Storage type:', status.type)
|
||||||
|
|
||||||
**Migration Checklist:**
|
**Migration Checklist:**
|
||||||
- ✅ Backup data
|
- ✅ Backup data
|
||||||
- ✅ Update npm package (`npm install @soulcraft/brainy@latest`)
|
- ✅ Update npm package (`npm install @soulcraftlabs/brainy@latest`)
|
||||||
- ✅ Restart application (automatic migration)
|
- ✅ Restart application (automatic migration)
|
||||||
- ✅ Verify data integrity
|
- ✅ Verify data integrity
|
||||||
- ✅ Enable lifecycle policies
|
- ✅ Enable lifecycle policies
|
||||||
|
|
|
||||||
|
|
@ -323,58 +323,24 @@ Only the graph adjacency index carries a committed scale assertion:
|
||||||
- ✅ **Single-Node by Design**: One process owns one `path`; scale out at the service layer
|
- ✅ **Single-Node by Design**: One process owns one `path`; scale out at the service layer
|
||||||
- ✅ **Zero Stubs**: Every line of code is production-ready
|
- ✅ **Zero Stubs**: Every line of code is production-ready
|
||||||
|
|
||||||
## Lazy Loading Performance
|
## Index Build at Open (10.4+)
|
||||||
|
|
||||||
Brainy supports two initialization modes for optimal performance across different use cases:
|
As of 10.4, `brain.init()` runs every needed index rebuild to completion before
|
||||||
|
it returns — always, regardless of dataset size. There is no lazy,
|
||||||
|
first-query rebuild path: a brain either finishes opening healthy, or `init()`
|
||||||
|
fails loudly. `disableAutoRebuild` no longer defers index construction to a
|
||||||
|
first query; it has no effect on *when* a rebuild runs. Manual control over
|
||||||
|
rebuilds is `repairIndex({ rebuild: [...] })`. See
|
||||||
|
[Index Health](concepts/index-health.md) for the full read-gate contract
|
||||||
|
(providers self-report readiness via `healthReport()`; a read against a
|
||||||
|
not-serving provider throws a typed `*NotReadyError` rather than rebuilding
|
||||||
|
mid-query).
|
||||||
|
|
||||||
### Mode 1: Auto-Rebuild (Default)
|
<!-- The pre-10.4 "Mode 2: Lazy Loading on First Query" section previously
|
||||||
|
documented here (disableAutoRebuild deferring index construction to the
|
||||||
```javascript
|
first find() call) described a real, now-retired code path. Removed
|
||||||
const brain = new Brainy()
|
rather than left to mislead; the concept doc above is the current
|
||||||
await brain.init() // Rebuilds indexes during init (~500ms-3s for 10K entities)
|
contract. -->
|
||||||
```
|
|
||||||
|
|
||||||
**Performance:**
|
|
||||||
- Init time: 500ms-3s (depends on dataset size)
|
|
||||||
- First query: Instant (indexes already loaded)
|
|
||||||
- Use case: Traditional applications, long-running servers
|
|
||||||
|
|
||||||
### Mode 2: Lazy Loading
|
|
||||||
|
|
||||||
```javascript
|
|
||||||
const brain = new Brainy({ disableAutoRebuild: true })
|
|
||||||
await brain.init() // Returns instantly (0-10ms)
|
|
||||||
|
|
||||||
const results = await brain.find({ limit: 10 }) // First query triggers rebuild (~50-200ms)
|
|
||||||
const more = await brain.find({ limit: 100 }) // Subsequent queries instant (0ms check)
|
|
||||||
```
|
|
||||||
|
|
||||||
**Performance:**
|
|
||||||
- Init time: 0-10ms (instant)
|
|
||||||
- First query: 50-200ms (includes index rebuild for 1K-10K entities)
|
|
||||||
- Subsequent queries: 0ms check (instant)
|
|
||||||
- Concurrent queries: Wait for same rebuild (mutex prevents duplicates)
|
|
||||||
|
|
||||||
**Concurrency Safety:**
|
|
||||||
```javascript
|
|
||||||
// 100 concurrent queries immediately after init
|
|
||||||
await brain.init()
|
|
||||||
|
|
||||||
const promises = Array.from({ length: 100 }, () =>
|
|
||||||
brain.find({ limit: 10 })
|
|
||||||
)
|
|
||||||
|
|
||||||
const results = await Promise.all(promises)
|
|
||||||
// ✅ Only 1 rebuild triggered (mutex)
|
|
||||||
// ✅ All 100 queries return correct results
|
|
||||||
// ✅ Total time: ~60ms (not 6000ms!)
|
|
||||||
```
|
|
||||||
|
|
||||||
**Use Cases for Lazy Loading:**
|
|
||||||
- **Serverless/Edge**: Minimize cold start time (0-10ms init)
|
|
||||||
- **Development**: Faster restarts during development
|
|
||||||
- **Large datasets**: Defer index loading until needed
|
|
||||||
- **Read-heavy workloads**: Writes don't wait for index rebuild
|
|
||||||
|
|
||||||
## Zero Configuration Required
|
## Zero Configuration Required
|
||||||
|
|
||||||
|
|
@ -384,10 +350,6 @@ Brainy is designed to be **smart enough to tune itself dynamically**. No configu
|
||||||
// That's it. Brainy handles everything.
|
// That's it. Brainy handles everything.
|
||||||
const brain = new Brainy()
|
const brain = new Brainy()
|
||||||
await brain.init()
|
await brain.init()
|
||||||
|
|
||||||
// Or with lazy loading for serverless
|
|
||||||
const brain = new Brainy({ disableAutoRebuild: true })
|
|
||||||
await brain.init() // Instant (0-10ms)
|
|
||||||
```
|
```
|
||||||
|
|
||||||
### Automatic Self-Tuning
|
### Automatic Self-Tuning
|
||||||
|
|
@ -395,7 +357,6 @@ await brain.init() // Instant (0-10ms)
|
||||||
- **Metadata Index**: Auto-builds sorted indices for range queries on first use
|
- **Metadata Index**: Auto-builds sorted indices for range queries on first use
|
||||||
- **Graph Index**: Auto-flushes every 30 seconds
|
- **Graph Index**: Auto-flushes every 30 seconds
|
||||||
- **Default Tuning**: Research-based vector index defaults
|
- **Default Tuning**: Research-based vector index defaults
|
||||||
- **Lazy Loading**: Indices built only when needed
|
|
||||||
- **Cache Management**: LRU caches with TTL
|
- **Cache Management**: LRU caches with TTL
|
||||||
|
|
||||||
### Intelligent Defaults
|
### Intelligent Defaults
|
||||||
|
|
|
||||||
|
|
@ -10,7 +10,7 @@ next:
|
||||||
- guides/storage-adapters
|
- guides/storage-adapters
|
||||||
---
|
---
|
||||||
|
|
||||||
# Plugin Development Guide
|
# Plugin System
|
||||||
|
|
||||||
Brainy has a plugin system that allows third-party packages to replace internal subsystems with custom implementations. This is how `@soulcraft/cor` provides optional native acceleration, and it's the same system available to any developer.
|
Brainy has a plugin system that allows third-party packages to replace internal subsystems with custom implementations. This is how `@soulcraft/cor` provides optional native acceleration, and it's the same system available to any developer.
|
||||||
|
|
||||||
|
|
@ -46,7 +46,7 @@ If no plugin provides a given key, brainy uses its built-in JavaScript implement
|
||||||
### 1. Implement the `BrainyPlugin` interface
|
### 1. Implement the `BrainyPlugin` interface
|
||||||
|
|
||||||
```typescript
|
```typescript
|
||||||
import type { BrainyPlugin, BrainyPluginContext } from '@soulcraft/brainy/plugin'
|
import type { BrainyPlugin, BrainyPluginContext } from '@soulcraftlabs/brainy/plugin'
|
||||||
|
|
||||||
const myPlugin: BrainyPlugin = {
|
const myPlugin: BrainyPlugin = {
|
||||||
name: 'my-brainy-plugin', // Must be unique (typically your npm package name)
|
name: 'my-brainy-plugin', // Must be unique (typically your npm package name)
|
||||||
|
|
@ -90,7 +90,7 @@ await brain.init()
|
||||||
**Programmatic registration:** For plugins not installed as npm packages, use `brain.use()`:
|
**Programmatic registration:** For plugins not installed as npm packages, use `brain.use()`:
|
||||||
|
|
||||||
```typescript
|
```typescript
|
||||||
import { Brainy } from '@soulcraft/brainy'
|
import { Brainy } from '@soulcraftlabs/brainy'
|
||||||
import myPlugin from './my-plugin.js'
|
import myPlugin from './my-plugin.js'
|
||||||
|
|
||||||
const brain = new Brainy()
|
const brain = new Brainy()
|
||||||
|
|
@ -200,15 +200,30 @@ members so a warm reopen never pays a redundant rebuild-from-canonical:
|
||||||
- **`init?(): Promise<void>`** — eager cold-load. Brainy awaits it once during
|
- **`init?(): Promise<void>`** — eager cold-load. Brainy awaits it once during
|
||||||
`brain.init()`, after the metadata provider's `init()` (the id-mapper hydrates first)
|
`brain.init()`, after the metadata provider's `init()` (the id-mapper hydrates first)
|
||||||
and **before the rebuild gate**.
|
and **before the rebuild gate**.
|
||||||
- **`isReady?(): boolean`** — honest durability signal. `true` ⇔ the persisted index is
|
- **`healthReport?(): HealthReport`** — the PREFERRED signal (10.4+). A named,
|
||||||
loaded (or cheaply demand-loadable) and consistent with what was last persisted. When
|
synchronous, O(1) verdict derived from the provider's own exact ledgers — never a
|
||||||
exposed, the rebuild gate defers to this signal **instead of** the `size() === 0` /
|
sample, never I/O, must never throw for a well-formed provider. Brainy's read gate
|
||||||
`totalEntries === 0` heuristics — a disk-native index may report 0 resident entries
|
(`assessProviderHealth()`) reads this INSTEAD of `isReady()` / size heuristics when
|
||||||
while fully durable. Never return `true` if the durable state failed to load: the
|
present: `serving: false` refuses the read with a typed `*NotReadyError` rather than
|
||||||
signal is honest in both directions, and a not-ready provider gets its rebuild even
|
triggering a rebuild — a read never starts a store walk. `healthy` marks every
|
||||||
when `size() > 0`.
|
*verified* invariant holding; a family named in `unledgered` counts as neither
|
||||||
|
healthy nor broken. See `HealthReport` / `LedgerInvariantResult` /
|
||||||
|
`InvariantSource` in `src/plugin.ts`, and
|
||||||
|
[Index Health](concepts/index-health.md) for the consumer-facing story.
|
||||||
|
- **`isReady?(): boolean`** — honest durability signal, the fallback when
|
||||||
|
`healthReport()` is absent. `true` ⇔ the persisted index is loaded (or cheaply
|
||||||
|
demand-loadable) and consistent with what was last persisted. When exposed, the
|
||||||
|
gate defers to this signal **instead of** the `size() === 0` / `totalEntries === 0`
|
||||||
|
heuristics — a disk-native index may report 0 resident entries while fully durable.
|
||||||
|
Never return `true` if the durable state failed to load: the signal is honest in
|
||||||
|
both directions, and a not-ready provider gets its rebuild even when `size() > 0`.
|
||||||
- **`isMigrating?(): boolean`** — while `true`, the provider owns its index (background
|
- **`isMigrating?(): boolean`** — while `true`, the provider owns its index (background
|
||||||
migration); brainy skips its rebuild entirely.
|
migration); brainy skips its rebuild entirely.
|
||||||
|
- **`validateInvariants?(): Promise<ProviderInvariantReport>`** — the async DEEP
|
||||||
|
diagnostic (full scans allowed), distinct from the bounded, sync `healthReport()`.
|
||||||
|
Must never throw — a failure is `healthy: false` data, not an exception; a provider
|
||||||
|
that throws anyway is read as a loud, unverified failure (never as "healthy") by
|
||||||
|
every caller, never silently retried into a rebuild.
|
||||||
|
|
||||||
Providers that implement none of these keep the size/count heuristics — correct for
|
Providers that implement none of these keep the size/count heuristics — correct for
|
||||||
engines whose `rebuild()` *is* their load path (like brainy's built-in JS vector index).
|
engines whose `rebuild()` *is* their load path (like brainy's built-in JS vector index).
|
||||||
|
|
@ -257,10 +272,10 @@ When provided by an optional native acceleration plugin (such as `@soulcraft/cor
|
||||||
#### `cache`
|
#### `cache`
|
||||||
**Type:** `UnifiedCache`
|
**Type:** `UnifiedCache`
|
||||||
|
|
||||||
Replaces the global `UnifiedCache` singleton used for VFS path resolution, semantic caching, and vector index caching. Must implement the `UnifiedCache` interface (available from `@soulcraft/brainy/internals`).
|
Replaces the global `UnifiedCache` singleton used for VFS path resolution, semantic caching, and vector index caching. Must implement the `UnifiedCache` interface (available from `@soulcraftlabs/brainy/internals`).
|
||||||
|
|
||||||
```typescript
|
```typescript
|
||||||
import type { UnifiedCache } from '@soulcraft/brainy/internals'
|
import type { UnifiedCache } from '@soulcraftlabs/brainy/internals'
|
||||||
|
|
||||||
context.registerProvider('cache', myNativeCache)
|
context.registerProvider('cache', myNativeCache)
|
||||||
```
|
```
|
||||||
|
|
@ -310,8 +325,8 @@ Plugins can register custom storage backends that users reference by name.
|
||||||
### Implementing a Storage Adapter
|
### Implementing a Storage Adapter
|
||||||
|
|
||||||
```typescript
|
```typescript
|
||||||
import type { StorageAdapterFactory } from '@soulcraft/brainy/plugin'
|
import type { StorageAdapterFactory } from '@soulcraftlabs/brainy/plugin'
|
||||||
import type { StorageAdapter } from '@soulcraft/brainy'
|
import type { StorageAdapter } from '@soulcraftlabs/brainy'
|
||||||
|
|
||||||
class MyStorageAdapter implements StorageAdapter {
|
class MyStorageAdapter implements StorageAdapter {
|
||||||
async init(): Promise<void> { /* ... */ }
|
async init(): Promise<void> { /* ... */ }
|
||||||
|
|
@ -345,9 +360,9 @@ Brainy provides three entry points for plugin developers:
|
||||||
|
|
||||||
| Import Path | Contents | Stability |
|
| Import Path | Contents | Stability |
|
||||||
|-------------|----------|-----------|
|
|-------------|----------|-----------|
|
||||||
| `@soulcraft/brainy` | Public API, types, StorageAdapter | Stable (semver) |
|
| `@soulcraftlabs/brainy` | Public API, types, StorageAdapter | Stable (semver) |
|
||||||
| `@soulcraft/brainy/plugin` | BrainyPlugin, BrainyPluginContext, StorageAdapterFactory | Stable (semver) |
|
| `@soulcraftlabs/brainy/plugin` | BrainyPlugin, BrainyPluginContext, StorageAdapterFactory | Stable (semver) |
|
||||||
| `@soulcraft/brainy/internals` | UnifiedCache, EntityIdMapper, logger utilities | Internal (may change between minor versions) |
|
| `@soulcraftlabs/brainy/internals` | UnifiedCache, EntityIdMapper, logger utilities | Internal (may change between minor versions) |
|
||||||
|
|
||||||
## Diagnostics
|
## Diagnostics
|
||||||
|
|
||||||
|
|
@ -425,7 +440,7 @@ A minimal but useful plugin that provides SIMD-accelerated distance calculations
|
||||||
|
|
||||||
```typescript
|
```typescript
|
||||||
// simd-distance-plugin/src/plugin.ts
|
// simd-distance-plugin/src/plugin.ts
|
||||||
import type { BrainyPlugin, BrainyPluginContext } from '@soulcraft/brainy/plugin'
|
import type { BrainyPlugin, BrainyPluginContext } from '@soulcraftlabs/brainy/plugin'
|
||||||
|
|
||||||
// Hypothetical native module
|
// Hypothetical native module
|
||||||
import { simdCosineDistance } from './native.js'
|
import { simdCosineDistance } from './native.js'
|
||||||
|
|
@ -455,7 +470,7 @@ export default simdDistancePlugin
|
||||||
"main": "./dist/plugin.js",
|
"main": "./dist/plugin.js",
|
||||||
"types": "./dist/plugin.d.ts",
|
"types": "./dist/plugin.d.ts",
|
||||||
"peerDependencies": {
|
"peerDependencies": {
|
||||||
"@soulcraft/brainy": ">=7.0.0"
|
"@soulcraftlabs/brainy": ">=7.0.0"
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
```
|
```
|
||||||
|
|
@ -463,7 +478,7 @@ export default simdDistancePlugin
|
||||||
Usage:
|
Usage:
|
||||||
|
|
||||||
```typescript
|
```typescript
|
||||||
import { Brainy } from '@soulcraft/brainy'
|
import { Brainy } from '@soulcraftlabs/brainy'
|
||||||
|
|
||||||
const brain = new Brainy({ plugins: ['brainy-simd-distance'] })
|
const brain = new Brainy({ plugins: ['brainy-simd-distance'] })
|
||||||
await brain.init()
|
await brain.init()
|
||||||
|
|
|
||||||
|
|
@ -54,7 +54,7 @@ After 40 API calls:
|
||||||
|
|
||||||
```typescript
|
```typescript
|
||||||
// server.ts
|
// server.ts
|
||||||
import { Brainy } from '@soulcraft/brainy'
|
import { Brainy } from '@soulcraftlabs/brainy'
|
||||||
|
|
||||||
// SINGLETON INSTANCE
|
// SINGLETON INSTANCE
|
||||||
let brainInstance: Brainy | null = null
|
let brainInstance: Brainy | null = null
|
||||||
|
|
@ -174,7 +174,7 @@ process.on('SIGTERM', async () => {
|
||||||
|
|
||||||
```typescript
|
```typescript
|
||||||
// server.ts - Clean Bun implementation
|
// server.ts - Clean Bun implementation
|
||||||
import { Brainy } from '@soulcraft/brainy'
|
import { Brainy } from '@soulcraftlabs/brainy'
|
||||||
|
|
||||||
let brain: Brainy | null = null
|
let brain: Brainy | null = null
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -5,7 +5,7 @@
|
||||||
## Quick Start
|
## Quick Start
|
||||||
|
|
||||||
```typescript
|
```typescript
|
||||||
import { Brainy, NounType, VerbType } from '@soulcraft/brainy'
|
import { Brainy, NounType, VerbType } from '@soulcraftlabs/brainy'
|
||||||
|
|
||||||
const brain = new Brainy()
|
const brain = new Brainy()
|
||||||
await brain.init()
|
await brain.init()
|
||||||
|
|
|
||||||
|
|
@ -99,7 +99,7 @@ Examples:
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
# 1. Deprecate wrong version on npm
|
# 1. Deprecate wrong version on npm
|
||||||
npm deprecate @soulcraft/brainy@X.X.X "Incorrect version - use Y.Y.Y"
|
npm deprecate @soulcraftlabs/brainy@X.X.X "Incorrect version - use Y.Y.Y"
|
||||||
|
|
||||||
# 2. Fix version in package.json
|
# 2. Fix version in package.json
|
||||||
# 3. Republish correct version
|
# 3. Republish correct version
|
||||||
|
|
|
||||||
|
|
@ -13,7 +13,7 @@
|
||||||
|
|
||||||
### In-Memory
|
### In-Memory
|
||||||
```typescript
|
```typescript
|
||||||
import Brainy from '@soulcraft/brainy'
|
import Brainy from '@soulcraftlabs/brainy'
|
||||||
const brain = new Brainy({ storage: { type: 'memory' } })
|
const brain = new Brainy({ storage: { type: 'memory' } })
|
||||||
```
|
```
|
||||||
|
|
||||||
|
|
@ -43,7 +43,7 @@ The native vector provider (via the optional `@soulcraft/cor` package) extends t
|
||||||
|
|
||||||
Numbers below are **measured** by `tests/benchmarks/find-composition-scale.js` (a single
|
Numbers below are **measured** by `tests/benchmarks/find-composition-scale.js` (a single
|
||||||
Node 22 process, in-memory storage, 384-dim vectors, `balanced` recall). They are the
|
Node 22 process, in-memory storage, 384-dim vectors, `balanced` recall). They are the
|
||||||
open-core (pure-TypeScript) path — what you get from `@soulcraft/brainy` with no native
|
open-core (pure-TypeScript) path — what you get from `@soulcraftlabs/brainy` with no native
|
||||||
provider installed. Run it yourself: `node --max-old-space-size=8192 tests/benchmarks/find-composition-scale.js 100000`.
|
provider installed. Run it yourself: `node --max-old-space-size=8192 tests/benchmarks/find-composition-scale.js 100000`.
|
||||||
|
|
||||||
`find()` query latency, p50 / p95 (200 queries each):
|
`find()` query latency, p50 / p95 (200 queries each):
|
||||||
|
|
|
||||||
1633
docs/api-contract.json
Normal file
1633
docs/api-contract.json
Normal file
File diff suppressed because it is too large
Load diff
|
|
@ -24,7 +24,7 @@ next:
|
||||||
## Quick Start
|
## Quick Start
|
||||||
|
|
||||||
```typescript
|
```typescript
|
||||||
import { Brainy, NounType, VerbType } from '@soulcraft/brainy'
|
import { Brainy, NounType, VerbType } from '@soulcraftlabs/brainy'
|
||||||
|
|
||||||
const brain = new Brainy() // Zero config!
|
const brain = new Brainy() // Zero config!
|
||||||
await brain.init() // VFS auto-initialized!
|
await brain.init() // VFS auto-initialized!
|
||||||
|
|
@ -848,7 +848,6 @@ const db = await brain.transact([
|
||||||
{ op: 'add', id: orderId, type: NounType.Document, subtype: 'order', data: 'Order #1042' },
|
{ op: 'add', id: orderId, type: NounType.Document, subtype: 'order', data: 'Order #1042' },
|
||||||
{ op: 'update', id: customerId, metadata: { lastOrderAt: Date.now() }, ifRev: customer._rev },
|
{ op: 'update', id: customerId, metadata: { lastOrderAt: Date.now() }, ifRev: customer._rev },
|
||||||
{ op: 'relate', from: customerId, to: orderId, type: VerbType.Creates, subtype: 'purchase' },
|
{ op: 'relate', from: customerId, to: orderId, type: VerbType.Creates, subtype: 'purchase' },
|
||||||
{ op: 'updateRelation', id: purchaseRelationId, subtype: 'return' },
|
|
||||||
{ op: 'remove', id: staleDraftId },
|
{ op: 'remove', id: staleDraftId },
|
||||||
{ op: 'unrelate', id: oldRelationId }
|
{ op: 'unrelate', id: oldRelationId }
|
||||||
], {
|
], {
|
||||||
|
|
@ -865,7 +864,6 @@ db.receipt.generation // the committed generation
|
||||||
- `{ op: 'update', ... }` — same parameters as `update()`, including per-entity `ifRev` CAS
|
- `{ op: 'update', ... }` — same parameters as `update()`, including per-entity `ifRev` CAS
|
||||||
- `{ op: 'remove', id }` — deletes the entity plus its relationships (same cascade as `delete()`)
|
- `{ op: 'remove', id }` — deletes the entity plus its relationships (same cascade as `delete()`)
|
||||||
- `{ op: 'relate', ... }` — same parameters as `relate()`, including `bidirectional`; duplicates dedupe to the existing relationship id
|
- `{ op: 'relate', ... }` — same parameters as `relate()`, including `bidirectional`; duplicates dedupe to the existing relationship id
|
||||||
- `{ op: 'updateRelation', ... }` — same parameters as `updateRelation()`; a batchable, first-class relationship update (not `unrelate` + `relate` — the relationship id and its edge never change)
|
|
||||||
- `{ op: 'unrelate', id }` — deletes a relationship by id
|
- `{ op: 'unrelate', id }` — deletes a relationship by id
|
||||||
|
|
||||||
Operations may reference ids created earlier in the same batch.
|
Operations may reference ids created earlier in the same batch.
|
||||||
|
|
@ -1012,7 +1010,7 @@ await db.release() // unpin + free cached materialization
|
||||||
|
|
||||||
### Db API errors
|
### Db API errors
|
||||||
|
|
||||||
All exported from `@soulcraft/brainy`:
|
All exported from `@soulcraftlabs/brainy`:
|
||||||
|
|
||||||
| Error | Thrown by | Meaning |
|
| Error | Thrown by | Meaning |
|
||||||
|---|---|---|
|
|---|---|---|
|
||||||
|
|
@ -1453,6 +1451,34 @@ const count = await brain.getVerbCount()
|
||||||
|
|
||||||
---
|
---
|
||||||
|
|
||||||
|
### The canonical count ledger (`StorageAdapter.getCanonicalCounts()`)
|
||||||
|
|
||||||
|
An OPTIONAL method on the `StorageAdapter` interface (implemented by both
|
||||||
|
built-in adapters), not a method on `Brainy` itself — relevant if you're
|
||||||
|
writing a custom storage adapter or composing a provider's own
|
||||||
|
`healthReport()`. O(1), no I/O. Per family (`nouns`/`verbs`):
|
||||||
|
|
||||||
|
```typescript
|
||||||
|
interface CanonicalCounts {
|
||||||
|
nouns: { counted: number; all: number }
|
||||||
|
verbs: { counted: number; all: number }
|
||||||
|
suspect: boolean
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
- `counted` mirrors `getNounCount()` / `getVerbCount()` (public + internal tiers).
|
||||||
|
- `all` is the ALL-visibility scalar — every tier, including system/internal
|
||||||
|
records — the denominator a derived index's own coverage math is measured
|
||||||
|
against.
|
||||||
|
- `suspect` is `true` when an unprovable delete has left `all` unverified since
|
||||||
|
the last recount; `brain.repairIndex()` clears it with a real canonical walk.
|
||||||
|
|
||||||
|
Adapters without the ledger omit the method; treat absence as "no
|
||||||
|
denominator," never as zero. See
|
||||||
|
**[Index Health](../concepts/index-health.md)** for the full story.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
### Subtype & facet APIs
|
### Subtype & facet APIs
|
||||||
|
|
||||||
Full guide: **[Subtypes & Facets](../guides/subtypes-and-facets.md)**.
|
Full guide: **[Subtypes & Facets](../guides/subtypes-and-facets.md)**.
|
||||||
|
|
@ -1833,6 +1859,104 @@ const semanticOnly = await brain.getStats({ excludeVFS: true })
|
||||||
|
|
||||||
---
|
---
|
||||||
|
|
||||||
|
### `repairIndex(options?)` → `Promise<RepairReport>`
|
||||||
|
|
||||||
|
The ceremony door for index repair. Bare `repairIndex()` is report-driven: it
|
||||||
|
prunes orphaned containers, recomputes count rollups, reconciles VFS
|
||||||
|
containment, and rebuilds only a derived-index family whose own health check
|
||||||
|
asks for it. Pass `options.rebuild` to force one or more families to rebuild
|
||||||
|
UNCONDITIONALLY — no health check is consulted — when an operator has
|
||||||
|
independent reason to reconcile a family regardless of what it self-reports.
|
||||||
|
|
||||||
|
```typescript
|
||||||
|
// Report-driven: only heals what actually needs it
|
||||||
|
const report = await brain.repairIndex()
|
||||||
|
console.log(report.healedTotal, report.families)
|
||||||
|
|
||||||
|
// Explicit: force the graph adjacency to rebuild from canonical, unconditionally
|
||||||
|
await brain.repairIndex({ rebuild: ['graph'] })
|
||||||
|
|
||||||
|
// Explicit: force all three derived indexes to rebuild
|
||||||
|
await brain.repairIndex({ rebuild: 'all' })
|
||||||
|
```
|
||||||
|
|
||||||
|
**`RepairReport`:**
|
||||||
|
- `families: RepairFamilyReport[]` — one row per family checked
|
||||||
|
- `healedTotal: number` — items healed across every family
|
||||||
|
- `durationMs: number`
|
||||||
|
|
||||||
|
**`RepairFamilyReport`** (one row):
|
||||||
|
- `family: string` — e.g. `'orphaned-containers'`, `'count-rollups'`,
|
||||||
|
`'vfs-containment'`, `'metadata-corruption'`, `'provider:metadata'`,
|
||||||
|
`'provider:graph'`, `'provider:vector'`
|
||||||
|
- `checked: boolean` — was this family actually examined (`false` ⇒ see `skipped`)
|
||||||
|
- `healed: number` — items re-posted/corrected in place (the incremental heal count)
|
||||||
|
- `missing?: { count: number; sample: string[] }` — exact count plus a capped id
|
||||||
|
sample when the check can name what diverged (never the full list)
|
||||||
|
- `rebuilt?: boolean` — a full generational rebuild ran (vs. an incremental heal)
|
||||||
|
- `detail?: string` / `reason?: string` — narration
|
||||||
|
- `skipped?: string` — why the family wasn't checked
|
||||||
|
|
||||||
|
Full walkthrough — what each family checks, degraded-but-serving vs. not-ready,
|
||||||
|
and what `suspect` counts mean — in
|
||||||
|
**[Index Health](../concepts/index-health.md)**.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
### Index readiness: typed errors, `healthReport()`, `disableAutoRebuild`
|
||||||
|
|
||||||
|
Every derived-index provider (vector, graph, metadata) may expose a named,
|
||||||
|
synchronous, O(1) `healthReport()` composed from its own exact ledgers — the
|
||||||
|
signal Brainy's read gate trusts over sampling or size heuristics. `init()`
|
||||||
|
brings every provider to serving before it returns; there is no first-query
|
||||||
|
lazy-rebuild path. A read that reaches a provider whose health report says it
|
||||||
|
isn't serving throws instead of rebuilding mid-query:
|
||||||
|
|
||||||
|
| Error | Thrown by | Meaning |
|
||||||
|
|---|---|---|
|
||||||
|
| `GraphIndexNotReadyError` | `find({ connected })`, `neighbors()`, `related()` | Graph adjacency isn't serving |
|
||||||
|
| `MetadataIndexNotReadyError` | `find({ where })` | Metadata/field index isn't serving |
|
||||||
|
| `VectorIndexNotReadyError` | `find({ query })`, `similar()` | Vector index isn't serving |
|
||||||
|
|
||||||
|
All three are exported from `@soulcraftlabs/brainy`. Catch them to distinguish
|
||||||
|
"index not ready" from a genuine empty result:
|
||||||
|
|
||||||
|
```typescript
|
||||||
|
import { MetadataIndexNotReadyError } from '@soulcraftlabs/brainy'
|
||||||
|
|
||||||
|
try {
|
||||||
|
const rows = await brain.find({ where: { status: 'active' } })
|
||||||
|
} catch (err) {
|
||||||
|
if (err instanceof MetadataIndexNotReadyError) {
|
||||||
|
// reconcile: await brain.repairIndex(), then retry
|
||||||
|
} else {
|
||||||
|
throw err
|
||||||
|
}
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
**`disableAutoRebuild`** no longer defers index construction to the first
|
||||||
|
query. A needed rebuild always runs at `open()`, regardless of this flag or
|
||||||
|
dataset size; the flag has no effect on *when* a rebuild runs. Full manual
|
||||||
|
control lives in `repairIndex({ rebuild: [...] })`, above.
|
||||||
|
|
||||||
|
### `validateIndexConsistency()` → `Promise<...>`
|
||||||
|
|
||||||
|
The deep, async diagnostic counterpart to `healthReport()` — safe to run on a
|
||||||
|
live brain, but does more work (a provider's `validateInvariants()` may run a
|
||||||
|
full scan, not just read a ledger). Aggregates the JS metadata index's own
|
||||||
|
consistency check with every derived-index provider's invariant report.
|
||||||
|
|
||||||
|
```typescript
|
||||||
|
const validation = await brain.validateIndexConsistency()
|
||||||
|
if (!validation.healthy) {
|
||||||
|
console.log(validation.recommendation) // what to run, e.g. repairIndex()
|
||||||
|
console.log(validation.providers) // each provider's own invariant report, when exposed
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
## Lifecycle
|
## Lifecycle
|
||||||
|
|
||||||
### Initialization
|
### Initialization
|
||||||
|
|
@ -2084,7 +2208,7 @@ For the full taxonomy with all 169 types and their descriptions, see:
|
||||||
- **📖 Documentation:** [Full Documentation](../)
|
- **📖 Documentation:** [Full Documentation](../)
|
||||||
- **🐛 Issues:** [GitHub Issues](https://github.com/soulcraftlabs/brainy/issues)
|
- **🐛 Issues:** [GitHub Issues](https://github.com/soulcraftlabs/brainy/issues)
|
||||||
- **💬 Discussions:** [GitHub Discussions](https://github.com/soulcraftlabs/brainy/discussions)
|
- **💬 Discussions:** [GitHub Discussions](https://github.com/soulcraftlabs/brainy/discussions)
|
||||||
- **📦 NPM:** [@soulcraft/brainy](https://www.npmjs.com/package/@soulcraft/brainy)
|
- **📦 NPM:** [@soulcraftlabs/brainy](https://www.npmjs.com/package/@soulcraftlabs/brainy)
|
||||||
- **⭐ GitHub:** [Star us](https://github.com/soulcraftlabs/brainy)
|
- **⭐ GitHub:** [Star us](https://github.com/soulcraftlabs/brainy)
|
||||||
|
|
||||||
---
|
---
|
||||||
|
|
|
||||||
|
|
@ -217,6 +217,40 @@ membership queries at scale:
|
||||||
`__words__` for tokenized text…).
|
`__words__` for tokenized text…).
|
||||||
- `_blobs/_column_index/{field}/L0-NNNNNN.bin` — the actual level-0 run
|
- `_blobs/_column_index/{field}/L0-NNNNNN.bin` — the actual level-0 run
|
||||||
segments, stored through the shared `_blobs/<key>.bin` binary convention.
|
segments, stored through the shared `_blobs/<key>.bin` binary convention.
|
||||||
|
- `_column_index/{field}/k/{kind}/…` — the same two files again, for a
|
||||||
|
**second value kind** on the same field (see below). Absent for a field that
|
||||||
|
holds one kind, which is nearly all of them.
|
||||||
|
|
||||||
|
### One posting column per (field, kind)
|
||||||
|
|
||||||
|
A field is not obliged to hold one type of value. `category` may carry
|
||||||
|
`'electronics'` on some rows and `5` on others, and both are real values of
|
||||||
|
that field. A segment, though, has one encoding — i64, f64, UTF-8, or boolean
|
||||||
|
— so a field that holds several kinds gets **one column per kind**:
|
||||||
|
|
||||||
|
- The first kind a field ever sees owns the plain `_column_index/{field}/`
|
||||||
|
layout above. A single-kind field is therefore byte-identical to what earlier
|
||||||
|
versions wrote, and an index written before typed postings opens unchanged.
|
||||||
|
- Every later kind gets its own column beside it at
|
||||||
|
`_column_index/{field}/k/{kind}/`, where `{kind}` is `number`, `string` or
|
||||||
|
`boolean`.
|
||||||
|
|
||||||
|
What that buys at query time:
|
||||||
|
|
||||||
|
| | |
|
||||||
|
|---|---|
|
||||||
|
| **Equality** | Answered from the column matching the **query value's own kind**. `where {category: 5}` reads the number postings; `where {category: '5'}` reads the string postings. Neither borrows the other's rows — a row written with the number `5` is not a row whose category is the text `'5'`. |
|
||||||
|
| **A kind the field never held** | Matches nothing. That is the true answer, not a coerced one. |
|
||||||
|
| **Ranges** | Routed by the kind of the bounds: numeric bounds read the numeric postings and ignore the field's strings. An **unbounded** range is the "has any value here" probe behind `exists`, and reads every kind. |
|
||||||
|
| **`orderBy`** | A number and a string have no order between them, so a mixed field orders by kind first (number, string, boolean) and by value within a kind. A single-kind field sorts exactly as it always did. |
|
||||||
|
| **Numbers** | One kind, one column: an integer column is written as i64 and widens to f64 the first time a non-integer arrives, so `4.5` is stored as itself rather than rounded. |
|
||||||
|
|
||||||
|
`null` and `undefined` are not kinds and are never posted; their absence is
|
||||||
|
what the `exists` / `missing` operators read.
|
||||||
|
|
||||||
|
Older readers are unaffected by the additional columns: they see the field's
|
||||||
|
primary column exactly where it has always been, and a `k/{kind}` directory is
|
||||||
|
simply a name they never query.
|
||||||
|
|
||||||
Sparse per-field indexes, roaring-bitmap chunks, and zone-map/bloom segments
|
Sparse per-field indexes, roaring-bitmap chunks, and zone-map/bloom segments
|
||||||
additionally live as bucketed keys under `_system/idx/` (see §3). Which path
|
additionally live as bucketed keys under `_system/idx/` (see §3). Which path
|
||||||
|
|
@ -268,7 +302,7 @@ locks/_flush_responses/ # writer answers with <uuid>.ack
|
||||||
| **Counts/statistics** | Per-type and per-subtype maps | `_system/{type,subtype,verb-subtype}-statistics.json.gz`, `counts.json` | Recomputable by scanning entities (`brainy inspect repair`) |
|
| **Counts/statistics** | Per-type and per-subtype maps | `_system/{type,subtype,verb-subtype}-statistics.json.gz`, `counts.json` | Recomputable by scanning entities (`brainy inspect repair`) |
|
||||||
|
|
||||||
A pluggable index provider (the 8.0 plugin contract in
|
A pluggable index provider (the 8.0 plugin contract in
|
||||||
`@soulcraft/brainy/plugin`) may replace any of the JS implementations; the
|
`@soulcraftlabs/brainy/plugin`) may replace any of the JS implementations; the
|
||||||
persisted formats above are contract-bound so JS and native implementations
|
persisted formats above are contract-bound so JS and native implementations
|
||||||
can interleave on the same directory.
|
can interleave on the same directory.
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -126,7 +126,7 @@ class TypeAwareMetadataIndex {
|
||||||
**The Design**: Specify types clearly in your API calls:
|
**The Design**: Specify types clearly in your API calls:
|
||||||
|
|
||||||
```typescript
|
```typescript
|
||||||
import { Brainy, NounType, VerbType } from '@soulcraft/brainy'
|
import { Brainy, NounType, VerbType } from '@soulcraftlabs/brainy'
|
||||||
|
|
||||||
// Add entity with explicit type
|
// Add entity with explicit type
|
||||||
await brain.add({
|
await brain.add({
|
||||||
|
|
@ -231,7 +231,7 @@ class OrgEnrichmentAugmentation {
|
||||||
**Brainy's Approach**: Extract **typed** concepts:
|
**Brainy's Approach**: Extract **typed** concepts:
|
||||||
|
|
||||||
```typescript
|
```typescript
|
||||||
import { NaturalLanguageProcessor } from '@soulcraft/brainy'
|
import { NaturalLanguageProcessor } from '@soulcraftlabs/brainy'
|
||||||
|
|
||||||
const nlp = new NaturalLanguageProcessor()
|
const nlp = new NaturalLanguageProcessor()
|
||||||
const concepts = await nlp.extractConcepts("Alice works at Google in San Francisco")
|
const concepts = await nlp.extractConcepts("Alice works at Google in San Francisco")
|
||||||
|
|
@ -382,7 +382,7 @@ import {
|
||||||
getVerbTypes,
|
getVerbTypes,
|
||||||
BrainyTypes,
|
BrainyTypes,
|
||||||
suggestType
|
suggestType
|
||||||
} from '@soulcraft/brainy'
|
} from '@soulcraftlabs/brainy'
|
||||||
|
|
||||||
// Get all available noun types
|
// Get all available noun types
|
||||||
const nounTypes = getNounTypes()
|
const nounTypes = getNounTypes()
|
||||||
|
|
|
||||||
|
|
@ -723,6 +723,14 @@ async stats(): Promise<Statistics> {
|
||||||
|
|
||||||
### 5. Index Rebuilding (Lazy Loading Support)
|
### 5. Index Rebuilding (Lazy Loading Support)
|
||||||
|
|
||||||
|
> **Stale as of 10.4 — "Mode 2: Lazy Loading on First Query" below is
|
||||||
|
> RETIRED.** `disableAutoRebuild` no longer defers index construction to a
|
||||||
|
> first query; `brain.init()` now runs every needed rebuild to completion
|
||||||
|
> before it returns, unconditionally, and a read against a not-serving
|
||||||
|
> provider throws a typed `*NotReadyError` instead of rebuilding mid-query.
|
||||||
|
> See `docs/concepts/index-health.md` for the current contract. Left below
|
||||||
|
> as historical background on the rebuild mechanics.
|
||||||
|
|
||||||
**Two modes of index loading:**
|
**Two modes of index loading:**
|
||||||
|
|
||||||
#### Mode 1: Auto-Rebuild on init() (default)
|
#### Mode 1: Auto-Rebuild on init() (default)
|
||||||
|
|
|
||||||
|
|
@ -1,5 +1,15 @@
|
||||||
# Initialization and Rebuild Processes
|
# Initialization and Rebuild Processes
|
||||||
|
|
||||||
|
> **Stale as of 10.4 — "Mode 2: Lazy Loading on First Query" below is RETIRED.**
|
||||||
|
> `disableAutoRebuild` no longer defers index construction to a first query;
|
||||||
|
> `brain.init()` now runs every needed rebuild to completion before it
|
||||||
|
> returns, unconditionally. A read against a not-serving provider throws a
|
||||||
|
> typed `*NotReadyError` instead of rebuilding mid-query. See
|
||||||
|
> `docs/concepts/index-health.md` for the current contract; this document's
|
||||||
|
> line-number references to `src/brainy.ts` also predate the file's current
|
||||||
|
> size and are unreliable. Left as historical background on the rebuild
|
||||||
|
> mechanics, not as a current API description.
|
||||||
|
|
||||||
This document explains how Brainy's four indexes (MetadataIndex, vector index, GraphAdjacencyIndex, DeletedItemsIndex) initialize and rebuild from persisted storage.
|
This document explains how Brainy's four indexes (MetadataIndex, vector index, GraphAdjacencyIndex, DeletedItemsIndex) initialize and rebuild from persisted storage.
|
||||||
|
|
||||||
## Core Principle: All Indexes Are Disk-Based
|
## Core Principle: All Indexes Are Disk-Based
|
||||||
|
|
|
||||||
|
|
@ -127,7 +127,7 @@ For reference, a clean migration path:
|
||||||
`isMultiProcessSafe` type-guard. Keep `hasStorageMethod` for
|
`isMultiProcessSafe` type-guard. Keep `hasStorageMethod` for
|
||||||
build/install artifact protection.
|
build/install artifact protection.
|
||||||
5. Document the new contract in `concepts/storage-adapters.md`.
|
5. Document the new contract in `concepts/storage-adapters.md`.
|
||||||
6. Major-version-bump the `@soulcraft/brainy` peerDep range expected by
|
6. Major-version-bump the `@soulcraftlabs/brainy` peerDep range expected by
|
||||||
plugins.
|
plugins.
|
||||||
|
|
||||||
Estimated work: ~half a day of code, ~2 hours of doc/example updates,
|
Estimated work: ~half a day of code, ~2 hours of doc/example updates,
|
||||||
|
|
|
||||||
|
|
@ -20,7 +20,7 @@ next:
|
||||||
Every example on this page is written against the real Brainy 8.0 API. The setup is always the same:
|
Every example on this page is written against the real Brainy 8.0 API. The setup is always the same:
|
||||||
|
|
||||||
```typescript
|
```typescript
|
||||||
import { Brainy, NounType, VerbType } from '@soulcraft/brainy'
|
import { Brainy, NounType, VerbType } from '@soulcraftlabs/brainy'
|
||||||
|
|
||||||
const brain = new Brainy()
|
const brain = new Brainy()
|
||||||
await brain.init()
|
await brain.init()
|
||||||
|
|
@ -40,7 +40,7 @@ Brainy's **Noun-Verb Taxonomy** achieves broad coverage of human knowledge throu
|
||||||
- **Multi-hop Graph Traversals = Relationship Complexity**
|
- **Multi-hop Graph Traversals = Relationship Complexity**
|
||||||
- **Result: Model data across virtually any industry**
|
- **Result: Model data across virtually any industry**
|
||||||
|
|
||||||
Every piece of information can be represented as entities (nouns) connected by relationships (verbs) carrying properties (metadata). The standardized type system from `@soulcraft/brainy` (`NounType`, `VerbType`) gives those nouns and verbs a stable, shared name.
|
Every piece of information can be represented as entities (nouns) connected by relationships (verbs) carrying properties (metadata). The standardized type system from `@soulcraftlabs/brainy` (`NounType`, `VerbType`) gives those nouns and verbs a stable, shared name.
|
||||||
|
|
||||||
## The Power of Standardization: Universal Interoperability
|
## The Power of Standardization: Universal Interoperability
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -35,7 +35,7 @@ constructor and `init()`.
|
||||||
## Instant Start
|
## Instant Start
|
||||||
|
|
||||||
```typescript
|
```typescript
|
||||||
import { Brainy } from '@soulcraft/brainy'
|
import { Brainy } from '@soulcraftlabs/brainy'
|
||||||
|
|
||||||
// That's it. No config needed.
|
// That's it. No config needed.
|
||||||
const brain = new Brainy()
|
const brain = new Brainy()
|
||||||
|
|
|
||||||
|
|
@ -167,7 +167,7 @@ await brain.find({ orderBy: 'createdAt' })
|
||||||
`UnresolvableFieldError` is exported from the package root:
|
`UnresolvableFieldError` is exported from the package root:
|
||||||
|
|
||||||
```typescript
|
```typescript
|
||||||
import { UnresolvableFieldError } from '@soulcraft/brainy'
|
import { UnresolvableFieldError } from '@soulcraftlabs/brainy'
|
||||||
|
|
||||||
try {
|
try {
|
||||||
await brain.find({ orderBy: 'createdAt' })
|
await brain.find({ orderBy: 'createdAt' })
|
||||||
|
|
|
||||||
217
docs/concepts/index-health.md
Normal file
217
docs/concepts/index-health.md
Normal file
|
|
@ -0,0 +1,217 @@
|
||||||
|
---
|
||||||
|
title: Index Health
|
||||||
|
slug: concepts/index-health
|
||||||
|
public: true
|
||||||
|
category: concepts
|
||||||
|
template: concept
|
||||||
|
order: 8
|
||||||
|
description: How Brainy knows whether a derived index can be trusted — exact accounting instead of sampling, the named health report, degraded-but-serving vs. not-ready, and what repairIndex() checks, heals, and rebuilds.
|
||||||
|
next:
|
||||||
|
- concepts/generation-fact-log
|
||||||
|
- guides/inspection
|
||||||
|
---
|
||||||
|
|
||||||
|
# Index Health
|
||||||
|
|
||||||
|
Brainy keeps one **canonical** copy of every entity and relationship, and three
|
||||||
|
**derived** indexes built from it — vector, metadata, and graph — so `find()` can
|
||||||
|
answer semantically, by filter, and by traversal without re-deriving the answer from
|
||||||
|
scratch on every query. A derived index is a cache with a serving structure: it can
|
||||||
|
be present but stale, present but only partially loaded, or fully out of sync with
|
||||||
|
canonical after a crash. This page is about how Brainy decides whether to trust one,
|
||||||
|
what it does when it can't, and how you reconcile the two.
|
||||||
|
|
||||||
|
## Exact accounting instead of sampling
|
||||||
|
|
||||||
|
Older health checks worked by inference: does `size()` return something greater
|
||||||
|
than zero, does a spot-check on one known item come back correct. Both are proxies.
|
||||||
|
A cold index can report a nonzero count while its actual serving structure never
|
||||||
|
loaded, and a spot-check only proves the one item it happened to ask about.
|
||||||
|
|
||||||
|
Every derived-index provider may now expose a named, synchronous, O(1)
|
||||||
|
`healthReport()` — composed from the provider's own **exact ledgers** (real counters
|
||||||
|
it already maintains on the write path), never a sample or a walk. This is the one
|
||||||
|
signal Brainy's read gate consults. A provider that doesn't yet expose one falls
|
||||||
|
back to an honest `isReady()` boolean, and finally to a size heuristic for engines
|
||||||
|
with neither — but wherever a `healthReport()` exists, it wins.
|
||||||
|
|
||||||
|
Underneath, storage itself keeps an analogous **canonical count ledger**: a
|
||||||
|
`counted` scalar (the user-facing total — what `getNounCount()` / `getVerbCount()`
|
||||||
|
return) and an `all` scalar (every tier, including internal records a derived
|
||||||
|
index's own coverage math needs to compare against). This is the real denominator
|
||||||
|
a provider's `healthReport()` measures itself by, rather than a total that can only
|
||||||
|
ever ratchet upward. See [What `suspect` counts mean](#what-suspect-counts-mean)
|
||||||
|
below for the one case that ledger can't stay exact through on its own.
|
||||||
|
|
||||||
|
## The named report
|
||||||
|
|
||||||
|
A `HealthReport` carries, per provider (`'vector'` / `'graph'` / `'metadata'`):
|
||||||
|
|
||||||
|
- **`healthy`** — `true` iff every *verified* invariant holds. An invariant whose
|
||||||
|
family has no ledger yet is `unledgered`, never counted either way — unknown,
|
||||||
|
not passing.
|
||||||
|
- **`serving`** — can this provider answer a query right now. A failing invariant
|
||||||
|
graded `heal: 'repair'` or `heal: 'none'` still leaves `serving: true` — this is
|
||||||
|
**degraded-but-serving**: something is off (say, a stale rollup on an
|
||||||
|
`employee` record's relationship count) but reads keep working. Only a failure
|
||||||
|
graded `heal: 'rebuild'` flips `serving` to `false` — **not-ready** — because the
|
||||||
|
provider itself is telling you its serving structure cannot answer correctly.
|
||||||
|
- **`invariants`** — each checked condition, with its provenance
|
||||||
|
(`source: 'ledger'` — an exact count; `'deep'` — a full scan, diagnostic-only;
|
||||||
|
`'unledgered'` — not yet tracked) and, for a failing one, an exact `missing`
|
||||||
|
count plus a capped sample of the affected ids — a verdict, never a dump.
|
||||||
|
- **`generation`** — bumps on every ledger mutation and rebuild, so a caller can
|
||||||
|
cache a verdict per generation instead of re-deriving it.
|
||||||
|
|
||||||
|
The distinction that matters day to day: `healthy: false` can be entirely benign —
|
||||||
|
a maintenance window, a divergence `repairIndex()` will clean up on its own
|
||||||
|
schedule. `serving: false` is not benign. It means this provider is refusing to
|
||||||
|
answer, on its own word, right now.
|
||||||
|
|
||||||
|
**How a failure gets its grade — the serving law.** A provider grades `heal` by
|
||||||
|
one question only: *could an answer be wrong?* — never *how expensive is the
|
||||||
|
fix?* A missing-postings shortfall, however large, is `heal: 'repair'` (re-post
|
||||||
|
exactly what the ledger names, reads serving throughout); it can never withhold
|
||||||
|
serving just because healing it takes work. `serving` is withheld only by a
|
||||||
|
small, named set of rebuild-graded conditions — the index not initialized, its
|
||||||
|
durable state absent, a manifest naming files that are not resident, a replay
|
||||||
|
that did not complete cleanly — the states in which an answer could genuinely be
|
||||||
|
wrong. And a read is only ever refused by the family it actually consults: a
|
||||||
|
metadata filter is answered by the metadata index alone, vector search by the
|
||||||
|
vector index, traversal by the graph index — one family's refusal never blocks
|
||||||
|
another family's reads.
|
||||||
|
|
||||||
|
## Reads refuse — they never rebuild
|
||||||
|
|
||||||
|
A query that reaches a not-serving provider does not trigger a rebuild from inside
|
||||||
|
the read. Brainy retired that path deliberately: a rebuild kicked off by an ordinary
|
||||||
|
`find({ where: { status: 'active' } })` call is a dark, unpredictable cost hiding
|
||||||
|
behind a request that looks like a cheap read. Instead, the read throws a typed,
|
||||||
|
catchable error naming the reason:
|
||||||
|
|
||||||
|
| Error | Thrown when | Meaning |
|
||||||
|
|---|---|---|
|
||||||
|
| `GraphIndexNotReadyError` | `find({ connected })`, `neighbors()`, `related()` | The graph adjacency index isn't serving — traversal would otherwise return `[]` indistinguishable from "no relationships" |
|
||||||
|
| `MetadataIndexNotReadyError` | `find({ where })` | The metadata/field index isn't serving — a filtered read would otherwise return `[]` indistinguishable from "no matches" |
|
||||||
|
| `VectorIndexNotReadyError` | `find({ query })`, `similar()` | The vector index isn't serving — a semantic search would otherwise return `[]` indistinguishable from "nothing similar" |
|
||||||
|
|
||||||
|
All three are exported from `@soulcraftlabs/brainy`. Catch them where your application
|
||||||
|
needs to distinguish "this index isn't ready yet" from "there's genuinely nothing
|
||||||
|
here" — a health dashboard, a retry policy, an operator alert. The fix is always
|
||||||
|
the same: reconcile the index, either by reopening the brain (which brings every
|
||||||
|
provider to serving before `init()` returns — see the next section) or by calling
|
||||||
|
`repairIndex()` explicitly.
|
||||||
|
|
||||||
|
```typescript
|
||||||
|
try {
|
||||||
|
const active = await brain.find({ where: { status: 'active' } })
|
||||||
|
} catch (err) {
|
||||||
|
if (err instanceof MetadataIndexNotReadyError) {
|
||||||
|
// not a "no results" — the index itself refused; alert or retry after repair
|
||||||
|
} else {
|
||||||
|
throw err
|
||||||
|
}
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
### Rebuilds happen at open, not on first query
|
||||||
|
|
||||||
|
`brain.init()` runs every needed rebuild to completion **before it returns**,
|
||||||
|
unconditionally, regardless of dataset size. There is no lazy, first-query
|
||||||
|
rebuild path anymore — a brain either finishes opening healthy, or it fails
|
||||||
|
open loudly. `disableAutoRebuild: true` no longer defers index construction to
|
||||||
|
the first query: it has no effect on *when* a needed rebuild runs. Full manual
|
||||||
|
control over rebuilds is `repairIndex({ rebuild: [...] })` (below), not this flag.
|
||||||
|
|
||||||
|
## `repairIndex()` — checking and healing
|
||||||
|
|
||||||
|
Bare `repairIndex()` is **report-driven**: it only heals what its own checks say
|
||||||
|
actually needs it, and it always returns a full per-family receipt.
|
||||||
|
|
||||||
|
```typescript
|
||||||
|
const report = await brain.repairIndex()
|
||||||
|
report.healedTotal // total items healed across every family
|
||||||
|
report.durationMs
|
||||||
|
report.families // one row per family checked
|
||||||
|
```
|
||||||
|
|
||||||
|
Each `RepairFamilyReport` row names what happened:
|
||||||
|
|
||||||
|
- **`checked`** — was this family actually examined (`false` means skipped —
|
||||||
|
see `skipped` for why).
|
||||||
|
- **`healed`** — items re-posted or corrected in place.
|
||||||
|
- **`missing`** — when the check can name what diverged: an exact `count` plus a
|
||||||
|
capped `sample` of ids.
|
||||||
|
- **`rebuilt`** — a full generational rebuild ran (as opposed to an incremental
|
||||||
|
heal).
|
||||||
|
- **`detail`** / **`reason`** / **`skipped`** — the receipt's narration; a row is
|
||||||
|
always either checked or explains why it wasn't. Nothing is silent.
|
||||||
|
|
||||||
|
On every call, bare `repairIndex()`:
|
||||||
|
|
||||||
|
1. Prunes orphaned canonical containers left by a partial delete.
|
||||||
|
2. Recomputes the count rollups from one canonical walk (unconditional — this is
|
||||||
|
also what clears a `suspect` ledger; see below).
|
||||||
|
3. Reconciles VFS containment edges, if the VFS is initialized.
|
||||||
|
4. Runs the metadata index's own corruption detection pass.
|
||||||
|
5. Consults each of the three derived-index providers' own health check and
|
||||||
|
rebuilds only a family whose failing invariant actually asks for it
|
||||||
|
(`heal: 'rebuild'`) — never a provider that reports `healthy` or a lesser
|
||||||
|
grade.
|
||||||
|
|
||||||
|
### The explicit rebuild door
|
||||||
|
|
||||||
|
`options.rebuild` skips the health check and rebuilds one or more families
|
||||||
|
**unconditionally** — the operator override for when you have independent reason
|
||||||
|
to distrust a family regardless of what it self-reports (a suspicious deploy, a
|
||||||
|
storage-layer incident, a support ticket that doesn't match what the health report
|
||||||
|
says):
|
||||||
|
|
||||||
|
```typescript
|
||||||
|
// Force the graph adjacency to rebuild from canonical, no invariant consulted
|
||||||
|
await brain.repairIndex({ rebuild: ['graph'] })
|
||||||
|
|
||||||
|
// Force all three derived indexes
|
||||||
|
await brain.repairIndex({ rebuild: 'all' })
|
||||||
|
```
|
||||||
|
|
||||||
|
A family named this way is recorded with `rebuilt: true` and
|
||||||
|
`reason: 'explicit rebuild requested'`, and is skipped by the normal
|
||||||
|
health-driven pass in the same call — it was already rebuilt unconditionally.
|
||||||
|
|
||||||
|
Reach for the explicit door when you need certainty regardless of self-report;
|
||||||
|
reach for bare `repairIndex()` for routine maintenance and after any incident
|
||||||
|
where you're not sure which family (if any) needs it.
|
||||||
|
|
||||||
|
## What `suspect` counts mean
|
||||||
|
|
||||||
|
Storage's canonical count ledger increments the ALL-visibility total on every new
|
||||||
|
record and decrements it on every *proven* delete — one where the record was read,
|
||||||
|
or the caller supplied its prior image. A delete that cannot prove what it removed
|
||||||
|
existed doesn't guess: it flags the ledger `suspect` (an operator-visible
|
||||||
|
`console.warn`, narrated once per session, not once per delete) rather than risk
|
||||||
|
decrementing a total that was never incremented for that record in the first
|
||||||
|
place. This is intentionally rare — it's a defensive fallback for callers on an
|
||||||
|
unusual removal path, not a per-delete cost.
|
||||||
|
|
||||||
|
`suspect` is not directly exposed on any `Brainy` method today — it lives on the
|
||||||
|
`StorageAdapter`'s optional `getCanonicalCounts()`, primarily consulted by
|
||||||
|
`repairIndex()`'s recount step and by custom storage adapters composing their own
|
||||||
|
`healthReport()`. What matters for an application: a `suspect` ledger is not
|
||||||
|
incorrect, just *unverified since the last recount* — and `repairIndex()`'s
|
||||||
|
unconditional count-rollup step (step 2, above) recomputes the ALL scalars from a
|
||||||
|
real canonical walk on every call, clearing the flag with proof either way.
|
||||||
|
|
||||||
|
## Practical guidance
|
||||||
|
|
||||||
|
- **On a normal restart**, do nothing — `init()` brings every provider to
|
||||||
|
serving before it returns, or fails loudly.
|
||||||
|
- **On a `*NotReadyError`** from a live read, reconcile with `repairIndex()`
|
||||||
|
(report-driven is almost always sufficient) and retry.
|
||||||
|
- **After an incident** where you distrust a specific family regardless of what
|
||||||
|
it reports healthy — a storage-layer fault, a suspicious restore — use the
|
||||||
|
explicit door: `repairIndex({ rebuild: ['metadata' | 'graph' | 'vector'] })`.
|
||||||
|
- **To audit before trusting a report**, `brain.auditGraph()` walks every stored
|
||||||
|
relationship and proves (or disproves) that reads return canonical truth,
|
||||||
|
independent of what any provider self-reports — see
|
||||||
|
[Inspecting a Live Brainy](../guides/inspection.md).
|
||||||
|
|
@ -95,8 +95,15 @@ The heartbeat interval rewrites the lock file every 10 seconds. The timer
|
||||||
is unref'd, so it does not keep the event loop alive on its own.
|
is unref'd, so it does not keep the event loop alive on its own.
|
||||||
|
|
||||||
On normal shutdown the writer releases the lock in `close()`. The shutdown
|
On normal shutdown the writer releases the lock in `close()`. The shutdown
|
||||||
hooks Brainy registers for `SIGTERM`, `SIGINT`, and `beforeExit` also
|
hooks Brainy registers for `SIGTERM` and `SIGINT` close every live brain by
|
||||||
release the lock so a container restart doesn't strand the directory.
|
that same `close()`, so a container restart doesn't strand the directory.
|
||||||
|
|
||||||
|
`beforeExit` is not one of them. Node emits it whenever the event loop has
|
||||||
|
no ref'd work left — a state a healthy script reaches routinely, because
|
||||||
|
Brainy's own idle and cadence timers are unref'd — and a drained event loop
|
||||||
|
is not a shutdown. That hook only persists derived state with a non-closing
|
||||||
|
`flush()`: it closes nothing, releases no lock, and leaves every brain open
|
||||||
|
and usable. If you want a shutdown, call `close()` or send `SIGTERM`.
|
||||||
|
|
||||||
## How to inspect a live writer
|
## How to inspect a live writer
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -61,7 +61,7 @@ The only required override is the capability flag. Returning `true` from
|
||||||
to call `acquireWriterLock()` at init.
|
to call `acquireWriterLock()` at init.
|
||||||
|
|
||||||
```typescript
|
```typescript
|
||||||
import { FileSystemStorage } from '@soulcraft/brainy'
|
import { FileSystemStorage } from '@soulcraftlabs/brainy'
|
||||||
|
|
||||||
export class MmapFileSystemStorage extends FileSystemStorage {
|
export class MmapFileSystemStorage extends FileSystemStorage {
|
||||||
public supportsMultiProcessLocking(): boolean {
|
public supportsMultiProcessLocking(): boolean {
|
||||||
|
|
@ -79,7 +79,7 @@ If your storage is **not filesystem-backed** (a custom
|
||||||
network backend), extend `BaseStorage` directly:
|
network backend), extend `BaseStorage` directly:
|
||||||
|
|
||||||
```typescript
|
```typescript
|
||||||
import { BaseStorage } from '@soulcraft/brainy'
|
import { BaseStorage } from '@soulcraftlabs/brainy'
|
||||||
|
|
||||||
export class MyCloudStorage extends BaseStorage {
|
export class MyCloudStorage extends BaseStorage {
|
||||||
// BaseStorage's default no-op implementations of the multi-process
|
// BaseStorage's default no-op implementations of the multi-process
|
||||||
|
|
@ -101,7 +101,7 @@ The defensive check at every new-storage-method call site (`brainy.ts`,
|
||||||
`hasStorageMethod(name)`) does **not** exist to handle "plugin bundles a
|
`hasStorageMethod(name)`) does **not** exist to handle "plugin bundles a
|
||||||
stale BaseStorage." Plugins ship a dist that preserves the dynamic ESM
|
stale BaseStorage." Plugins ship a dist that preserves the dynamic ESM
|
||||||
import (verify in your plugin's `dist/`: `import { FileSystemStorage } from
|
import (verify in your plugin's `dist/`: `import { FileSystemStorage } from
|
||||||
'@soulcraft/brainy'` is not rewritten to a vendored copy). The prototype
|
'@soulcraftlabs/brainy'` is not rewritten to a vendored copy). The prototype
|
||||||
chain at runtime resolves to whatever Brainy version your consumer has
|
chain at runtime resolves to whatever Brainy version your consumer has
|
||||||
installed.
|
installed.
|
||||||
|
|
||||||
|
|
@ -109,8 +109,8 @@ installed.
|
||||||
the prototype chain at the consumer-app level:
|
the prototype chain at the consumer-app level:
|
||||||
|
|
||||||
- **Stale `node_modules`** — a lingering install from before the consumer
|
- **Stale `node_modules`** — a lingering install from before the consumer
|
||||||
upgraded Brainy. The package.json says `@soulcraft/brainy@7.22.0` but
|
upgraded Brainy. The package.json says `@soulcraftlabs/brainy@7.22.0` but
|
||||||
`node_modules/@soulcraft/brainy` is still 7.20.x.
|
`node_modules/@soulcraftlabs/brainy` is still 7.20.x.
|
||||||
- **Lockfile drift** — `bun.lockb` / `package-lock.json` pins a brainy
|
- **Lockfile drift** — `bun.lockb` / `package-lock.json` pins a brainy
|
||||||
version older than the package.json range, and `bun install` honors the
|
version older than the package.json range, and `bun install` honors the
|
||||||
lockfile.
|
lockfile.
|
||||||
|
|
@ -131,7 +131,7 @@ and the warning names the adapter class plus a remediation hint:
|
||||||
methods on its prototype chain. Writer locking and the flush-request RPC are
|
methods on its prototype chain. Writer locking and the flush-request RPC are
|
||||||
disabled for this directory. Likely fix: clean install (`rm -rf node_modules
|
disabled for this directory. Likely fix: clean install (`rm -rf node_modules
|
||||||
bun.lockb && bun install`) or rebuild your container image to refresh
|
bun.lockb && bun install`) or rebuild your container image to refresh
|
||||||
`@soulcraft/brainy` to ≥7.21. See docs/concepts/storage-adapters.md.
|
`@soulcraftlabs/brainy` to ≥7.21. See docs/concepts/storage-adapters.md.
|
||||||
```
|
```
|
||||||
|
|
||||||
## Authoring a new storage adapter — minimum checklist
|
## Authoring a new storage adapter — minimum checklist
|
||||||
|
|
@ -168,7 +168,7 @@ bun.lockb && bun install`) or rebuild your container image to refresh
|
||||||
install time — fix install, not your plugin.
|
install time — fix install, not your plugin.
|
||||||
|
|
||||||
6. **Pin your peer dep generously.** `"peerDependencies": {
|
6. **Pin your peer dep generously.** `"peerDependencies": {
|
||||||
"@soulcraft/brainy": "^7.21.0" }` accepts any compatible 7.x. Don't pin
|
"@soulcraftlabs/brainy": "^7.21.0" }` accepts any compatible 7.x. Don't pin
|
||||||
to an exact patch unless you're tracking a known regression.
|
to an exact patch unless you're tracking a known regression.
|
||||||
|
|
||||||
## Future direction
|
## Future direction
|
||||||
|
|
@ -185,5 +185,5 @@ follow-up; consumers don't need to anticipate the change.
|
||||||
heartbeat semantics, what the lock protects.
|
heartbeat semantics, what the lock protects.
|
||||||
- [`guides/inspection`](../guides/inspection.md) — `brainy inspect` and the
|
- [`guides/inspection`](../guides/inspection.md) — `brainy inspect` and the
|
||||||
read-only mode.
|
read-only mode.
|
||||||
- `node_modules/@soulcraft/brainy/dist/storage/baseStorage.d.ts` — the
|
- `node_modules/@soulcraftlabs/brainy/dist/storage/baseStorage.d.ts` — the
|
||||||
authoritative type signatures for every method this page references.
|
authoritative type signatures for every method this page references.
|
||||||
|
|
|
||||||
|
|
@ -22,7 +22,7 @@ they share a single scan.
|
||||||
## Quick Start
|
## Quick Start
|
||||||
|
|
||||||
```typescript
|
```typescript
|
||||||
import { Brainy, NounType } from '@soulcraft/brainy'
|
import { Brainy, NounType } from '@soulcraftlabs/brainy'
|
||||||
|
|
||||||
const brain = new Brainy()
|
const brain = new Brainy()
|
||||||
await brain.init()
|
await brain.init()
|
||||||
|
|
|
||||||
|
|
@ -8,7 +8,7 @@ Brainy is **framework-friendly** - designed to drop into the server side of any
|
||||||
|
|
||||||
Brainy embeds an HNSW vector index, a graph engine, and a filesystem-backed persistence layer. These belong on the server:
|
Brainy embeds an HNSW vector index, a graph engine, and a filesystem-backed persistence layer. These belong on the server:
|
||||||
|
|
||||||
- **Zero configuration**: Just `import { Brainy } from '@soulcraft/brainy'`
|
- **Zero configuration**: Just `import { Brainy } from '@soulcraftlabs/brainy'`
|
||||||
- **Auto storage detection**: `new Brainy()` auto-selects filesystem persistence on Node
|
- **Auto storage detection**: `new Brainy()` auto-selects filesystem persistence on Node
|
||||||
- **Cleaner code**: No browser polyfills, no conditional client/server imports
|
- **Cleaner code**: No browser polyfills, no conditional client/server imports
|
||||||
- **Better DX**: One instance shared across your server routes
|
- **Better DX**: One instance shared across your server routes
|
||||||
|
|
@ -18,13 +18,13 @@ Brainy embeds an HNSW vector index, a graph engine, and a filesystem-backed pers
|
||||||
### Install Brainy
|
### Install Brainy
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
npm install @soulcraft/brainy
|
npm install @soulcraftlabs/brainy
|
||||||
```
|
```
|
||||||
|
|
||||||
### Basic Integration
|
### Basic Integration
|
||||||
|
|
||||||
```javascript
|
```javascript
|
||||||
import { Brainy } from '@soulcraft/brainy'
|
import { Brainy } from '@soulcraftlabs/brainy'
|
||||||
|
|
||||||
// Run on the server (API route, server component, backend service)
|
// Run on the server (API route, server component, backend service)
|
||||||
// new Brainy() auto-detects filesystem persistence on Node
|
// new Brainy() auto-detects filesystem persistence on Node
|
||||||
|
|
@ -105,7 +105,7 @@ On the server, create one Brainy instance and reuse it across requests. This mod
|
||||||
|
|
||||||
```javascript
|
```javascript
|
||||||
// lib/brain.server.js
|
// lib/brain.server.js
|
||||||
import { Brainy } from '@soulcraft/brainy'
|
import { Brainy } from '@soulcraftlabs/brainy'
|
||||||
|
|
||||||
let brainPromise
|
let brainPromise
|
||||||
|
|
||||||
|
|
@ -163,7 +163,7 @@ On the server, create one Brainy instance and reuse it across requests:
|
||||||
|
|
||||||
```javascript
|
```javascript
|
||||||
// server/brain.js (server-only module)
|
// server/brain.js (server-only module)
|
||||||
import { Brainy } from '@soulcraft/brainy'
|
import { Brainy } from '@soulcraftlabs/brainy'
|
||||||
|
|
||||||
let brainPromise
|
let brainPromise
|
||||||
|
|
||||||
|
|
@ -248,7 +248,7 @@ The matching backend endpoint uses Brainy directly (Node/Bun):
|
||||||
|
|
||||||
```typescript
|
```typescript
|
||||||
// server: api/search
|
// server: api/search
|
||||||
import { Brainy } from '@soulcraft/brainy'
|
import { Brainy } from '@soulcraftlabs/brainy'
|
||||||
|
|
||||||
const brain = new Brainy() // auto-detects filesystem persistence on Node
|
const brain = new Brainy() // auto-detects filesystem persistence on Node
|
||||||
await brain.init()
|
await brain.init()
|
||||||
|
|
@ -266,7 +266,7 @@ In Next.js, Brainy lives in server code only: API routes, server components, or
|
||||||
|
|
||||||
```javascript
|
```javascript
|
||||||
// lib/brain.server.js (imported only by server code)
|
// lib/brain.server.js (imported only by server code)
|
||||||
import { Brainy } from '@soulcraft/brainy'
|
import { Brainy } from '@soulcraftlabs/brainy'
|
||||||
|
|
||||||
let brainPromise
|
let brainPromise
|
||||||
|
|
||||||
|
|
@ -318,7 +318,7 @@ Brainy runs in a server-only module (`*.server.js`); the component fetches resul
|
||||||
|
|
||||||
```javascript
|
```javascript
|
||||||
// src/lib/server/brain.js (server-only — note the .server suffix)
|
// src/lib/server/brain.js (server-only — note the .server suffix)
|
||||||
import { Brainy } from '@soulcraft/brainy'
|
import { Brainy } from '@soulcraftlabs/brainy'
|
||||||
|
|
||||||
let brainPromise
|
let brainPromise
|
||||||
|
|
||||||
|
|
@ -432,7 +432,7 @@ import { defineConfig } from 'vite'
|
||||||
|
|
||||||
export default defineConfig({
|
export default defineConfig({
|
||||||
ssr: {
|
ssr: {
|
||||||
external: ['@soulcraft/brainy']
|
external: ['@soulcraftlabs/brainy']
|
||||||
}
|
}
|
||||||
})
|
})
|
||||||
```
|
```
|
||||||
|
|
@ -440,7 +440,7 @@ export default defineConfig({
|
||||||
```javascript
|
```javascript
|
||||||
// rollup.config.js (server bundle)
|
// rollup.config.js (server bundle)
|
||||||
export default {
|
export default {
|
||||||
external: ['@soulcraft/brainy', 'node:fs', 'node:path', 'node:crypto']
|
external: ['@soulcraftlabs/brainy', 'node:fs', 'node:path', 'node:crypto']
|
||||||
}
|
}
|
||||||
```
|
```
|
||||||
|
|
||||||
|
|
@ -466,7 +466,7 @@ export async function load({ url }) {
|
||||||
|
|
||||||
```javascript
|
```javascript
|
||||||
// For build-time usage (runs in Node during the build)
|
// For build-time usage (runs in Node during the build)
|
||||||
import { Brainy } from '@soulcraft/brainy'
|
import { Brainy } from '@soulcraftlabs/brainy'
|
||||||
|
|
||||||
export async function generateStaticProps() {
|
export async function generateStaticProps() {
|
||||||
const brain = new Brainy({
|
const brain = new Brainy({
|
||||||
|
|
@ -513,7 +513,7 @@ export async function generateStaticProps() {
|
||||||
|
|
||||||
### Issue: Large client bundle size
|
### Issue: Large client bundle size
|
||||||
**Cause**: A client module is pulling in Brainy.
|
**Cause**: A client module is pulling in Brainy.
|
||||||
**Solution**: Move the `import { Brainy } from '@soulcraft/brainy'` into a server-only module so it never reaches the browser bundle.
|
**Solution**: Move the `import { Brainy } from '@soulcraftlabs/brainy'` into a server-only module so it never reaches the browser bundle.
|
||||||
|
|
||||||
### Issue: SSR hydration mismatch
|
### Issue: SSR hydration mismatch
|
||||||
**Solution**: Run the search on the server (loader / server action / API route) and pass the results down as props, so server and client render the same markup.
|
**Solution**: Run the search on the server (loader / server action / API route) and pass the results down as props, so server and client render the same markup.
|
||||||
|
|
|
||||||
|
|
@ -9,7 +9,7 @@ Brainy's import is **ONE magical method** that understands EVERYTHING:
|
||||||
## The Ultimate Simplicity
|
## The Ultimate Simplicity
|
||||||
|
|
||||||
```javascript
|
```javascript
|
||||||
import { Brainy } from '@soulcraft/brainy'
|
import { Brainy } from '@soulcraftlabs/brainy'
|
||||||
|
|
||||||
const brain = new Brainy()
|
const brain = new Brainy()
|
||||||
await brain.init()
|
await brain.init()
|
||||||
|
|
|
||||||
|
|
@ -13,7 +13,7 @@ Brainy provides real-time progress tracking for **all 7 supported file formats**
|
||||||
### Basic Progress Tracking
|
### Basic Progress Tracking
|
||||||
|
|
||||||
```typescript
|
```typescript
|
||||||
import { Brainy } from '@soulcraft/brainy'
|
import { Brainy } from '@soulcraftlabs/brainy'
|
||||||
import * as fs from 'fs'
|
import * as fs from 'fs'
|
||||||
|
|
||||||
const brain = await Brainy.create()
|
const brain = await Brainy.create()
|
||||||
|
|
|
||||||
|
|
@ -7,7 +7,7 @@
|
||||||
## Basic Import
|
## Basic Import
|
||||||
|
|
||||||
```typescript
|
```typescript
|
||||||
import { Brainy } from '@soulcraft/brainy'
|
import { Brainy } from '@soulcraftlabs/brainy'
|
||||||
|
|
||||||
const brain = new Brainy()
|
const brain = new Brainy()
|
||||||
await brain.init()
|
await brain.init()
|
||||||
|
|
@ -187,7 +187,7 @@ await brain.import(file, {
|
||||||
## Complete Example
|
## Complete Example
|
||||||
|
|
||||||
```typescript
|
```typescript
|
||||||
import { Brainy } from '@soulcraft/brainy'
|
import { Brainy } from '@soulcraftlabs/brainy'
|
||||||
import * as fs from 'fs'
|
import * as fs from 'fs'
|
||||||
|
|
||||||
async function importCatalog() {
|
async function importCatalog() {
|
||||||
|
|
|
||||||
|
|
@ -108,7 +108,7 @@ check fails — useful for piping into monitoring or CI.
|
||||||
## Programmatic inspection
|
## Programmatic inspection
|
||||||
|
|
||||||
```typescript
|
```typescript
|
||||||
import { Brainy } from '@soulcraft/brainy'
|
import { Brainy } from '@soulcraftlabs/brainy'
|
||||||
|
|
||||||
const reader = await Brainy.openReadOnly({
|
const reader = await Brainy.openReadOnly({
|
||||||
storage: { type: 'filesystem', path: '/data/brain' }
|
storage: { type: 'filesystem', path: '/data/brain' }
|
||||||
|
|
|
||||||
|
|
@ -21,21 +21,21 @@ next:
|
||||||
## Install
|
## Install
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
npm install @soulcraft/brainy
|
npm install @soulcraftlabs/brainy
|
||||||
```
|
```
|
||||||
|
|
||||||
Or with your preferred package manager:
|
Or with your preferred package manager:
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
bun add @soulcraft/brainy
|
bun add @soulcraftlabs/brainy
|
||||||
yarn add @soulcraft/brainy
|
yarn add @soulcraftlabs/brainy
|
||||||
pnpm add @soulcraft/brainy
|
pnpm add @soulcraftlabs/brainy
|
||||||
```
|
```
|
||||||
|
|
||||||
## Verify
|
## Verify
|
||||||
|
|
||||||
```typescript
|
```typescript
|
||||||
import { Brainy } from '@soulcraft/brainy'
|
import { Brainy } from '@soulcraftlabs/brainy'
|
||||||
|
|
||||||
const brain = new Brainy()
|
const brain = new Brainy()
|
||||||
await brain.init()
|
await brain.init()
|
||||||
|
|
@ -52,7 +52,7 @@ npm install @soulcraft/cor
|
||||||
```
|
```
|
||||||
|
|
||||||
```typescript
|
```typescript
|
||||||
import { Brainy } from '@soulcraft/brainy'
|
import { Brainy } from '@soulcraftlabs/brainy'
|
||||||
|
|
||||||
const brain = new Brainy({ plugins: ['@soulcraft/cor'] })
|
const brain = new Brainy({ plugins: ['@soulcraft/cor'] })
|
||||||
await brain.init() // native providers registered during init
|
await brain.init() // native providers registered during init
|
||||||
|
|
@ -71,7 +71,7 @@ remains available on npm if you need it.
|
||||||
Brainy ships with full TypeScript types. No `@types/` package needed:
|
Brainy ships with full TypeScript types. No `@types/` package needed:
|
||||||
|
|
||||||
```typescript
|
```typescript
|
||||||
import { Brainy, NounType, VerbType } from '@soulcraft/brainy'
|
import { Brainy, NounType, VerbType } from '@soulcraftlabs/brainy'
|
||||||
|
|
||||||
const brain = new Brainy()
|
const brain = new Brainy()
|
||||||
await brain.init()
|
await brain.init()
|
||||||
|
|
|
||||||
|
|
@ -66,7 +66,7 @@ const results = await brain.search("query")
|
||||||
**New diagnostics for capacity planning and performance tuning.**
|
**New diagnostics for capacity planning and performance tuning.**
|
||||||
|
|
||||||
```typescript
|
```typescript
|
||||||
import { Brainy } from '@soulcraft/brainy'
|
import { Brainy } from '@soulcraftlabs/brainy'
|
||||||
|
|
||||||
const brain = new Brainy()
|
const brain = new Brainy()
|
||||||
await brain.init()
|
await brain.init()
|
||||||
|
|
@ -112,7 +112,7 @@ Recommendations: ${stats.recommendations.join(', ')}
|
||||||
### Step 1: Update Package
|
### Step 1: Update Package
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
npm install @soulcraft/brainy@latest
|
npm install @soulcraftlabs/brainy@latest
|
||||||
```
|
```
|
||||||
|
|
||||||
### Step 2: Restart Your Application
|
### Step 2: Restart Your Application
|
||||||
|
|
@ -134,7 +134,7 @@ npm run start
|
||||||
### Check Adaptive Sizing is Working
|
### Check Adaptive Sizing is Working
|
||||||
|
|
||||||
```typescript
|
```typescript
|
||||||
import { Brainy } from '@soulcraft/brainy'
|
import { Brainy } from '@soulcraftlabs/brainy'
|
||||||
|
|
||||||
const brain = new Brainy()
|
const brain = new Brainy()
|
||||||
await brain.init()
|
await brain.init()
|
||||||
|
|
@ -218,7 +218,7 @@ For debugging or compatibility testing:
|
||||||
If you need to rollback to v3.35.0:
|
If you need to rollback to v3.35.0:
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
npm install @soulcraft/brainy@3.35.0
|
npm install @soulcraftlabs/brainy@3.35.0
|
||||||
```
|
```
|
||||||
|
|
||||||
**Note:** We don't anticipate any issues, but rollback is straightforward if needed.
|
**Note:** We don't anticipate any issues, but rollback is straightforward if needed.
|
||||||
|
|
@ -367,7 +367,7 @@ if (stats.fairness.fairnessViolation) {
|
||||||
|
|
||||||
## Next Steps
|
## Next Steps
|
||||||
|
|
||||||
1. ✅ **Upgrade:** `npm install @soulcraft/brainy@latest`
|
1. ✅ **Upgrade:** `npm install @soulcraftlabs/brainy@latest`
|
||||||
2. 📊 **Monitor:** Use `getCacheStats()` to verify performance improvements
|
2. 📊 **Monitor:** Use `getCacheStats()` to verify performance improvements
|
||||||
3. 🎯 **Tune:** Adjust based on recommendations (if needed)
|
3. 🎯 **Tune:** Adjust based on recommendations (if needed)
|
||||||
4. 📖 **Read:** [Operations Guide](../operations/capacity-planning.md) for capacity planning
|
4. 📖 **Read:** [Operations Guide](../operations/capacity-planning.md) for capacity planning
|
||||||
|
|
|
||||||
|
|
@ -37,7 +37,7 @@ This single WASM file contains everything needed for sentence embeddings.
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
# Bun as a runtime — supported and recommended
|
# Bun as a runtime — supported and recommended
|
||||||
bun add @soulcraft/brainy
|
bun add @soulcraftlabs/brainy
|
||||||
bun run server.ts
|
bun run server.ts
|
||||||
```
|
```
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -80,7 +80,7 @@ If you read raw stored records (fact-log scanners, export tooling), use
|
||||||
the exported shape-aware splitters — they handle both record eras:
|
the exported shape-aware splitters — they handle both record eras:
|
||||||
|
|
||||||
```typescript
|
```typescript
|
||||||
import { splitNounMetadataRecord } from '@soulcraft/brainy'
|
import { splitNounMetadataRecord } from '@soulcraftlabs/brainy'
|
||||||
const { reserved, custom } = splitNounMetadataRecord(rawRecord)
|
const { reserved, custom } = splitNounMetadataRecord(rawRecord)
|
||||||
// reserved = engine fields · custom = the user's bag, ANY names
|
// reserved = engine fields · custom = the user's bag, ANY names
|
||||||
```
|
```
|
||||||
|
|
@ -88,7 +88,7 @@ const { reserved, custom } = splitNounMetadataRecord(rawRecord)
|
||||||
Feature detection (never version-sniff):
|
Feature detection (never version-sniff):
|
||||||
|
|
||||||
```typescript
|
```typescript
|
||||||
import * as brainy from '@soulcraft/brainy'
|
import * as brainy from '@soulcraftlabs/brainy'
|
||||||
const lawActive = 'FIELD_ADDRESSING_CAPABILITY' in brainy // 'field-addressing/v1'
|
const lawActive = 'FIELD_ADDRESSING_CAPABILITY' in brainy // 'field-addressing/v1'
|
||||||
```
|
```
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -9,7 +9,7 @@ Complete guide to integrating Brainy with Next.js applications, covering App Rou
|
||||||
```bash
|
```bash
|
||||||
npx create-next-app@latest my-brainy-app
|
npx create-next-app@latest my-brainy-app
|
||||||
cd my-brainy-app
|
cd my-brainy-app
|
||||||
npm install @soulcraft/brainy
|
npm install @soulcraftlabs/brainy
|
||||||
```
|
```
|
||||||
|
|
||||||
### Basic Setup
|
### Basic Setup
|
||||||
|
|
@ -18,7 +18,7 @@ npm install @soulcraft/brainy
|
||||||
// app/components/BrainyProvider.jsx
|
// app/components/BrainyProvider.jsx
|
||||||
'use client'
|
'use client'
|
||||||
import { createContext, useContext, useEffect, useState } from 'react'
|
import { createContext, useContext, useEffect, useState } from 'react'
|
||||||
import { Brainy } from '@soulcraft/brainy'
|
import { Brainy } from '@soulcraftlabs/brainy'
|
||||||
|
|
||||||
const BrainyContext = createContext()
|
const BrainyContext = createContext()
|
||||||
|
|
||||||
|
|
@ -271,7 +271,7 @@ export default function SearchPage() {
|
||||||
|
|
||||||
```javascript
|
```javascript
|
||||||
// app/api/search/route.js (App Router)
|
// app/api/search/route.js (App Router)
|
||||||
import { Brainy } from '@soulcraft/brainy'
|
import { Brainy } from '@soulcraftlabs/brainy'
|
||||||
|
|
||||||
let brain = null
|
let brain = null
|
||||||
|
|
||||||
|
|
@ -332,7 +332,7 @@ export async function GET() {
|
||||||
|
|
||||||
```javascript
|
```javascript
|
||||||
// pages/api/search.js (Pages Router)
|
// pages/api/search.js (Pages Router)
|
||||||
import { Brainy } from '@soulcraft/brainy'
|
import { Brainy } from '@soulcraftlabs/brainy'
|
||||||
|
|
||||||
let brain = null
|
let brain = null
|
||||||
|
|
||||||
|
|
@ -374,7 +374,7 @@ export default async function handler(req, res) {
|
||||||
|
|
||||||
```javascript
|
```javascript
|
||||||
// app/api/data/route.js
|
// app/api/data/route.js
|
||||||
import { Brainy } from '@soulcraft/brainy'
|
import { Brainy } from '@soulcraftlabs/brainy'
|
||||||
|
|
||||||
let brain = null
|
let brain = null
|
||||||
|
|
||||||
|
|
@ -418,7 +418,7 @@ export async function POST(request) {
|
||||||
```jsx
|
```jsx
|
||||||
// app/actions/brainy.js
|
// app/actions/brainy.js
|
||||||
'use server'
|
'use server'
|
||||||
import { Brainy } from '@soulcraft/brainy'
|
import { Brainy } from '@soulcraftlabs/brainy'
|
||||||
|
|
||||||
let brain = null
|
let brain = null
|
||||||
|
|
||||||
|
|
@ -630,7 +630,7 @@ CMD ["npm", "start"]
|
||||||
/** @type {import('next').NextConfig} */
|
/** @type {import('next').NextConfig} */
|
||||||
const nextConfig = {
|
const nextConfig = {
|
||||||
experimental: {
|
experimental: {
|
||||||
serverComponentsExternalPackages: ['@soulcraft/brainy']
|
serverComponentsExternalPackages: ['@soulcraftlabs/brainy']
|
||||||
},
|
},
|
||||||
webpack: (config, { isServer }) => {
|
webpack: (config, { isServer }) => {
|
||||||
if (!isServer) {
|
if (!isServer) {
|
||||||
|
|
@ -797,7 +797,7 @@ export function rateLimit(req, limit = 100, window = 60000) {
|
||||||
// app/contexts/BrainyContext.jsx
|
// app/contexts/BrainyContext.jsx
|
||||||
'use client'
|
'use client'
|
||||||
import { createContext, useContext, useReducer, useEffect } from 'react'
|
import { createContext, useContext, useReducer, useEffect } from 'react'
|
||||||
import { Brainy } from '@soulcraft/brainy'
|
import { Brainy } from '@soulcraftlabs/brainy'
|
||||||
|
|
||||||
const BrainyContext = createContext()
|
const BrainyContext = createContext()
|
||||||
|
|
||||||
|
|
@ -873,7 +873,7 @@ import { BrainyProvider } from '../app/components/BrainyProvider'
|
||||||
import { Search } from '../app/components/Search'
|
import { Search } from '../app/components/Search'
|
||||||
|
|
||||||
// Mock Brainy
|
// Mock Brainy
|
||||||
jest.mock('@soulcraft/brainy', () => ({
|
jest.mock('@soulcraftlabs/brainy', () => ({
|
||||||
Brainy: jest.fn().mockImplementation(() => ({
|
Brainy: jest.fn().mockImplementation(() => ({
|
||||||
init: jest.fn().mockResolvedValue(undefined),
|
init: jest.fn().mockResolvedValue(undefined),
|
||||||
find: jest.fn().mockResolvedValue([
|
find: jest.fn().mockResolvedValue([
|
||||||
|
|
|
||||||
|
|
@ -32,7 +32,7 @@ Brainy 7.31.0 adds a per-entity revision counter so multiple writers can coordin
|
||||||
Every distributed-job scheduler eventually wants this exact loop:
|
Every distributed-job scheduler eventually wants this exact loop:
|
||||||
|
|
||||||
```ts
|
```ts
|
||||||
import { Brainy, RevisionConflictError } from '@soulcraft/brainy'
|
import { Brainy, RevisionConflictError } from '@soulcraftlabs/brainy'
|
||||||
|
|
||||||
const LOCK_ID = '...uuid for this job slot...'
|
const LOCK_ID = '...uuid for this job slot...'
|
||||||
|
|
||||||
|
|
@ -137,7 +137,7 @@ await brain.addIfMissing({ // ← not a real API
|
||||||
It's race-prone as a plain read-then-write: two concurrent imports both see "not found," both insert, you get duplicates. Without a unique-index primitive (which Brainy doesn't have today), close the race with whole-store CAS — read at a pinned generation, then commit only if nothing moved:
|
It's race-prone as a plain read-then-write: two concurrent imports both see "not found," both insert, you get duplicates. Without a unique-index primitive (which Brainy doesn't have today), close the race with whole-store CAS — read at a pinned generation, then commit only if nothing moved:
|
||||||
|
|
||||||
```ts
|
```ts
|
||||||
import { GenerationConflictError } from '@soulcraft/brainy'
|
import { GenerationConflictError } from '@soulcraftlabs/brainy'
|
||||||
|
|
||||||
async function addIfMissingByEmail(email: string, data: string) {
|
async function addIfMissingByEmail(email: string, data: string) {
|
||||||
for (let attempt = 0; attempt < 5; attempt++) {
|
for (let attempt = 0; attempt < 5; attempt++) {
|
||||||
|
|
|
||||||
|
|
@ -18,13 +18,13 @@ Get Brainy running in under a minute.
|
||||||
## 1. Install
|
## 1. Install
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
npm install @soulcraft/brainy
|
npm install @soulcraftlabs/brainy
|
||||||
```
|
```
|
||||||
|
|
||||||
## 2. Initialize
|
## 2. Initialize
|
||||||
|
|
||||||
```typescript
|
```typescript
|
||||||
import { Brainy, NounType, VerbType } from '@soulcraft/brainy'
|
import { Brainy, NounType, VerbType } from '@soulcraftlabs/brainy'
|
||||||
|
|
||||||
const brain = new Brainy()
|
const brain = new Brainy()
|
||||||
await brain.init()
|
await brain.init()
|
||||||
|
|
@ -67,7 +67,7 @@ await brain.relate({
|
||||||
## 5. Query with Triple Intelligence
|
## 5. Query with Triple Intelligence
|
||||||
|
|
||||||
```typescript
|
```typescript
|
||||||
import type { Result } from '@soulcraft/brainy'
|
import type { Result } from '@soulcraftlabs/brainy'
|
||||||
|
|
||||||
// All three search paradigms in one call
|
// All three search paradigms in one call
|
||||||
const results: Result[] = await brain.find({
|
const results: Result[] = await brain.find({
|
||||||
|
|
|
||||||
|
|
@ -11,7 +11,7 @@
|
||||||
### One Interface for Everything
|
### One Interface for Everything
|
||||||
|
|
||||||
```typescript
|
```typescript
|
||||||
import { Brainy } from '@soulcraft/brainy'
|
import { Brainy } from '@soulcraftlabs/brainy'
|
||||||
|
|
||||||
const brain = await Brainy.create()
|
const brain = await Brainy.create()
|
||||||
|
|
||||||
|
|
@ -78,7 +78,7 @@ interface ImportProgress {
|
||||||
|
|
||||||
```typescript
|
```typescript
|
||||||
import { useState } from 'react'
|
import { useState } from 'react'
|
||||||
import { Brainy } from '@soulcraft/brainy'
|
import { Brainy } from '@soulcraftlabs/brainy'
|
||||||
|
|
||||||
function UniversalImportProgress({ file }: { file: File }) {
|
function UniversalImportProgress({ file }: { file: File }) {
|
||||||
const [progress, setProgress] = useState({
|
const [progress, setProgress] = useState({
|
||||||
|
|
@ -177,7 +177,7 @@ function UniversalImportProgress({ file }: { file: File }) {
|
||||||
|
|
||||||
```typescript
|
```typescript
|
||||||
import ora from 'ora'
|
import ora from 'ora'
|
||||||
import { Brainy } from '@soulcraft/brainy'
|
import { Brainy } from '@soulcraftlabs/brainy'
|
||||||
|
|
||||||
async function importWithProgress(filePath: string) {
|
async function importWithProgress(filePath: string) {
|
||||||
const spinner = ora('Starting import...').start()
|
const spinner = ora('Starting import...').start()
|
||||||
|
|
|
||||||
|
|
@ -28,7 +28,7 @@ on-disk layout (memory's "disk" is a JS Map).
|
||||||
## Quick start
|
## Quick start
|
||||||
|
|
||||||
```ts
|
```ts
|
||||||
import { Brainy } from '@soulcraft/brainy'
|
import { Brainy } from '@soulcraftlabs/brainy'
|
||||||
|
|
||||||
// Filesystem (recommended for any persistent workload):
|
// Filesystem (recommended for any persistent workload):
|
||||||
const brain = new Brainy({
|
const brain = new Brainy({
|
||||||
|
|
@ -134,7 +134,7 @@ config; the `type` is optional.
|
||||||
If you want to skip the factory:
|
If you want to skip the factory:
|
||||||
|
|
||||||
```ts
|
```ts
|
||||||
import { FileSystemStorage, MemoryStorage } from '@soulcraft/brainy'
|
import { FileSystemStorage, MemoryStorage } from '@soulcraftlabs/brainy'
|
||||||
|
|
||||||
const fsStorage = new FileSystemStorage('./brainy-data')
|
const fsStorage = new FileSystemStorage('./brainy-data')
|
||||||
const memStorage = new MemoryStorage()
|
const memStorage = new MemoryStorage()
|
||||||
|
|
|
||||||
|
|
@ -34,7 +34,7 @@ Three layers solve this:
|
||||||
### Write
|
### Write
|
||||||
|
|
||||||
```typescript
|
```typescript
|
||||||
import { Brainy, NounType } from '@soulcraft/brainy'
|
import { Brainy, NounType } from '@soulcraftlabs/brainy'
|
||||||
|
|
||||||
const brain = new Brainy()
|
const brain = new Brainy()
|
||||||
await brain.init()
|
await brain.init()
|
||||||
|
|
@ -240,7 +240,7 @@ await brain.migrateField({
|
||||||
A realistic adoption sequence for a brain that started without these primitives:
|
A realistic adoption sequence for a brain that started without these primitives:
|
||||||
|
|
||||||
```typescript
|
```typescript
|
||||||
import { Brainy, NounType } from '@soulcraft/brainy'
|
import { Brainy, NounType } from '@soulcraftlabs/brainy'
|
||||||
|
|
||||||
const brain = new Brainy({ storage: { type: 'filesystem', path: './brain-data' } })
|
const brain = new Brainy({ storage: { type: 'filesystem', path: './brain-data' } })
|
||||||
await brain.init()
|
await brain.init()
|
||||||
|
|
|
||||||
|
|
@ -25,7 +25,7 @@ content — and how 8.0 recovers it for you.
|
||||||
|
|
||||||
## TL;DR
|
## TL;DR
|
||||||
|
|
||||||
- **Just upgrade to `@soulcraft/brainy@8.0.12` (or later) and open the store.**
|
- **Just upgrade to `@soulcraftlabs/brainy@8.0.12` (or later) and open the store.**
|
||||||
If a previous upgrade left VFS content stranded, 8.0.12 **heals it on open**,
|
If a previous upgrade left VFS content stranded, 8.0.12 **heals it on open**,
|
||||||
with no operator action.
|
with no operator action.
|
||||||
- Want to force or script it? Call **`await brain.vfs.adoptOrphanedBlobs()`**.
|
- Want to force or script it? Call **`await brain.vfs.adoptOrphanedBlobs()`**.
|
||||||
|
|
@ -90,7 +90,7 @@ So the operator action for a stranded store is simply: **upgrade to 8.0.12 and
|
||||||
open it.**
|
open it.**
|
||||||
|
|
||||||
```ts
|
```ts
|
||||||
import { Brainy } from '@soulcraft/brainy'
|
import { Brainy } from '@soulcraftlabs/brainy'
|
||||||
|
|
||||||
// Opening the store is all that is required — recovery runs during init().
|
// Opening the store is all that is required — recovery runs during init().
|
||||||
const brain = new Brainy({ storage: { type: 'filesystem', path: '/data/my-store' } })
|
const brain = new Brainy({ storage: { type: 'filesystem', path: '/data/my-store' } })
|
||||||
|
|
@ -182,5 +182,5 @@ and opening each store is sufficient.
|
||||||
The recovery is copy-only, so no rollback of the recovery itself is ever needed.
|
The recovery is copy-only, so no rollback of the recovery itself is ever needed.
|
||||||
If you need to roll back the **whole** 7→8 upgrade, restore the directory from
|
If you need to roll back the **whole** 7→8 upgrade, restore the directory from
|
||||||
your pre-upgrade backup (retained automatically while recovery is incomplete, or
|
your pre-upgrade backup (retained automatically while recovery is incomplete, or
|
||||||
your own snapshot) and pin `@soulcraft/brainy@7.x`. 8.0 does not keep the old
|
your own snapshot) and pin `@soulcraftlabs/brainy@7.x`. 8.0 does not keep the old
|
||||||
branch layout in place, so a directory-level restore is the rollback path.
|
branch layout in place, so a directory-level restore is the rollback path.
|
||||||
|
|
|
||||||
|
|
@ -12,7 +12,7 @@ Complete guide to integrating Brainy with Vue.js applications, covering Vue 3, N
|
||||||
npm create vue@latest my-brainy-app
|
npm create vue@latest my-brainy-app
|
||||||
cd my-brainy-app
|
cd my-brainy-app
|
||||||
npm install
|
npm install
|
||||||
npm install @soulcraft/brainy
|
npm install @soulcraftlabs/brainy
|
||||||
```
|
```
|
||||||
|
|
||||||
### Basic Setup
|
### Basic Setup
|
||||||
|
|
@ -574,7 +574,7 @@ Nuxt's server engine (Nitro) is the natural home for Brainy: it runs on Node/Bun
|
||||||
|
|
||||||
```javascript
|
```javascript
|
||||||
// server/utils/brain.js (server-only — Nitro never bundles this into the client)
|
// server/utils/brain.js (server-only — Nitro never bundles this into the client)
|
||||||
import { Brainy } from '@soulcraft/brainy'
|
import { Brainy } from '@soulcraftlabs/brainy'
|
||||||
|
|
||||||
let brainPromise
|
let brainPromise
|
||||||
|
|
||||||
|
|
@ -1201,7 +1201,7 @@ import vue from '@vitejs/plugin-vue'
|
||||||
export default defineConfig({
|
export default defineConfig({
|
||||||
plugins: [vue()],
|
plugins: [vue()],
|
||||||
ssr: {
|
ssr: {
|
||||||
external: ['@soulcraft/brainy']
|
external: ['@soulcraftlabs/brainy']
|
||||||
}
|
}
|
||||||
})
|
})
|
||||||
```
|
```
|
||||||
|
|
|
||||||
|
|
@ -24,7 +24,7 @@ Brainy's neural extraction system uses a **4-signal ensemble architecture** to c
|
||||||
### Method 1: Brain Instance (Recommended)
|
### Method 1: Brain Instance (Recommended)
|
||||||
|
|
||||||
```typescript
|
```typescript
|
||||||
import { Brainy, NounType } from '@soulcraft/brainy'
|
import { Brainy, NounType } from '@soulcraftlabs/brainy'
|
||||||
|
|
||||||
const brain = new Brainy()
|
const brain = new Brainy()
|
||||||
await brain.init()
|
await brain.init()
|
||||||
|
|
@ -62,9 +62,9 @@ const people = await brain.extractEntities('...', {
|
||||||
import {
|
import {
|
||||||
SmartExtractor,
|
SmartExtractor,
|
||||||
SmartRelationshipExtractor
|
SmartRelationshipExtractor
|
||||||
} from '@soulcraft/brainy'
|
} from '@soulcraftlabs/brainy'
|
||||||
// Or use subpath imports:
|
// Or use subpath imports:
|
||||||
import { SmartExtractor } from '@soulcraft/brainy/neural/SmartExtractor'
|
import { SmartExtractor } from '@soulcraftlabs/brainy/neural/SmartExtractor'
|
||||||
|
|
||||||
const brain = new Brainy()
|
const brain = new Brainy()
|
||||||
await brain.init()
|
await brain.init()
|
||||||
|
|
@ -176,7 +176,7 @@ const withVectors = await brain.extractEntities(text, {
|
||||||
**Direct entity type classifier.** Use when you have pre-detected candidates or need custom configuration.
|
**Direct entity type classifier.** Use when you have pre-detected candidates or need custom configuration.
|
||||||
|
|
||||||
```typescript
|
```typescript
|
||||||
import { SmartExtractor, FormatContext } from '@soulcraft/brainy'
|
import { SmartExtractor, FormatContext } from '@soulcraftlabs/brainy'
|
||||||
|
|
||||||
const extractor = new SmartExtractor(brain, {
|
const extractor = new SmartExtractor(brain, {
|
||||||
minConfidence: 0.7, // Threshold
|
minConfidence: 0.7, // Threshold
|
||||||
|
|
@ -229,7 +229,7 @@ interface ExtractionResult {
|
||||||
**Relationship type classifier.** Determines verb/relationship types between entities.
|
**Relationship type classifier.** Determines verb/relationship types between entities.
|
||||||
|
|
||||||
```typescript
|
```typescript
|
||||||
import { SmartRelationshipExtractor } from '@soulcraft/brainy'
|
import { SmartRelationshipExtractor } from '@soulcraftlabs/brainy'
|
||||||
|
|
||||||
const relExtractor = new SmartRelationshipExtractor(brain, {
|
const relExtractor = new SmartRelationshipExtractor(brain, {
|
||||||
minConfidence: 0.6,
|
minConfidence: 0.6,
|
||||||
|
|
@ -286,7 +286,7 @@ const rel = await relExtractor.infer(
|
||||||
**Full extraction orchestrator.** Handles candidate detection, classification, and deduplication.
|
**Full extraction orchestrator.** Handles candidate detection, classification, and deduplication.
|
||||||
|
|
||||||
```typescript
|
```typescript
|
||||||
import { NeuralEntityExtractor } from '@soulcraft/brainy'
|
import { NeuralEntityExtractor } from '@soulcraftlabs/brainy'
|
||||||
|
|
||||||
const extractor = new NeuralEntityExtractor(brain)
|
const extractor = new NeuralEntityExtractor(brain)
|
||||||
|
|
||||||
|
|
@ -607,7 +607,7 @@ const locations = entities.filter(e => e.type === NounType.Location)
|
||||||
### Example 2: Excel Data Classification
|
### Example 2: Excel Data Classification
|
||||||
|
|
||||||
```typescript
|
```typescript
|
||||||
import { SmartExtractor } from '@soulcraft/brainy'
|
import { SmartExtractor } from '@soulcraftlabs/brainy'
|
||||||
|
|
||||||
const extractor = new SmartExtractor(brain)
|
const extractor = new SmartExtractor(brain)
|
||||||
|
|
||||||
|
|
@ -629,7 +629,7 @@ for (let i = 0; i < cells.length; i++) {
|
||||||
### Example 3: Relationship Extraction
|
### Example 3: Relationship Extraction
|
||||||
|
|
||||||
```typescript
|
```typescript
|
||||||
import { SmartRelationshipExtractor } from '@soulcraft/brainy'
|
import { SmartRelationshipExtractor } from '@soulcraftlabs/brainy'
|
||||||
|
|
||||||
const relExtractor = new SmartRelationshipExtractor(brain)
|
const relExtractor = new SmartRelationshipExtractor(brain)
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -204,8 +204,8 @@ await brain.add({ data: { name: 'Entity' }, type: NounType.Thing })
|
||||||
### Basic Add Operation
|
### Basic Add Operation
|
||||||
|
|
||||||
```typescript
|
```typescript
|
||||||
import { Brainy } from '@soulcraft/brainy'
|
import { Brainy } from '@soulcraftlabs/brainy'
|
||||||
import { NounType } from '@soulcraft/brainy/types'
|
import { NounType } from '@soulcraftlabs/brainy/types'
|
||||||
|
|
||||||
const brain = new Brainy()
|
const brain = new Brainy()
|
||||||
await brain.init()
|
await brain.init()
|
||||||
|
|
@ -428,7 +428,7 @@ await brain.relate({ ... }) // a crash here leaves the entity unlinked
|
||||||
|
|
||||||
```typescript
|
```typescript
|
||||||
import { describe, it, expect } from 'vitest'
|
import { describe, it, expect } from 'vitest'
|
||||||
import { Brainy } from '@soulcraft/brainy'
|
import { Brainy } from '@soulcraftlabs/brainy'
|
||||||
|
|
||||||
describe('Transaction Tests', () => {
|
describe('Transaction Tests', () => {
|
||||||
it('should rollback on failure', async () => {
|
it('should rollback on failure', async () => {
|
||||||
|
|
|
||||||
|
|
@ -23,7 +23,7 @@ The Universal Display Augmentation is a powerful AI-powered system that automati
|
||||||
### Basic Usage
|
### Basic Usage
|
||||||
|
|
||||||
```typescript
|
```typescript
|
||||||
import { Brainy } from '@soulcraft/brainy'
|
import { Brainy } from '@soulcraftlabs/brainy'
|
||||||
|
|
||||||
const brainy = new Brainy()
|
const brainy = new Brainy()
|
||||||
await brainy.init()
|
await brainy.init()
|
||||||
|
|
|
||||||
|
|
@ -71,9 +71,9 @@ Let's build a projection that organizes files by priority (high, medium, low):
|
||||||
### Step 1: Create the Strategy Class
|
### Step 1: Create the Strategy Class
|
||||||
|
|
||||||
```typescript
|
```typescript
|
||||||
import { BaseProjectionStrategy } from '@soulcraft/brainy/vfs/semantic'
|
import { BaseProjectionStrategy } from '@soulcraftlabs/brainy/vfs/semantic'
|
||||||
import { Brainy } from '@soulcraft/brainy'
|
import { Brainy } from '@soulcraftlabs/brainy'
|
||||||
import { VirtualFileSystem, VFSEntity } from '@soulcraft/brainy/vfs'
|
import { VirtualFileSystem, VFSEntity } from '@soulcraftlabs/brainy/vfs'
|
||||||
|
|
||||||
export class PriorityProjection extends BaseProjectionStrategy {
|
export class PriorityProjection extends BaseProjectionStrategy {
|
||||||
readonly name = 'priority'
|
readonly name = 'priority'
|
||||||
|
|
@ -141,7 +141,7 @@ export class PriorityProjection extends BaseProjectionStrategy {
|
||||||
### Step 2: Register the Strategy
|
### Step 2: Register the Strategy
|
||||||
|
|
||||||
```typescript
|
```typescript
|
||||||
import { Brainy } from '@soulcraft/brainy'
|
import { Brainy } from '@soulcraftlabs/brainy'
|
||||||
import { PriorityProjection } from './PriorityProjection'
|
import { PriorityProjection } from './PriorityProjection'
|
||||||
|
|
||||||
const brain = new Brainy()
|
const brain = new Brainy()
|
||||||
|
|
@ -537,7 +537,7 @@ Use the projection's resolve cache:
|
||||||
|
|
||||||
```typescript
|
```typescript
|
||||||
import { describe, it, expect, beforeAll } from 'vitest'
|
import { describe, it, expect, beforeAll } from 'vitest'
|
||||||
import { Brainy } from '@soulcraft/brainy'
|
import { Brainy } from '@soulcraftlabs/brainy'
|
||||||
import { PriorityProjection } from './PriorityProjection'
|
import { PriorityProjection } from './PriorityProjection'
|
||||||
|
|
||||||
describe('PriorityProjection', () => {
|
describe('PriorityProjection', () => {
|
||||||
|
|
@ -714,7 +714,7 @@ async resolve(brain, vfs, value: string) {
|
||||||
3. Use appropriate limits: Don't fetch more than needed
|
3. Use appropriate limits: Don't fetch more than needed
|
||||||
|
|
||||||
### Type errors
|
### Type errors
|
||||||
1. Import correct types: `import { Brainy, VirtualFileSystem } from '@soulcraft/brainy'`
|
1. Import correct types: `import { Brainy, VirtualFileSystem } from '@soulcraftlabs/brainy'`
|
||||||
2. Use `as VFSEntity` when mapping results
|
2. Use `as VFSEntity` when mapping results
|
||||||
3. Check BaseProjectionStrategy import
|
3. Check BaseProjectionStrategy import
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -14,11 +14,11 @@ A file explorer that:
|
||||||
## ⚡ Step 1: Basic Setup (1 minute)
|
## ⚡ Step 1: Basic Setup (1 minute)
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
npm install @soulcraft/brainy
|
npm install @soulcraftlabs/brainy
|
||||||
```
|
```
|
||||||
|
|
||||||
```typescript
|
```typescript
|
||||||
import { Brainy } from '@soulcraft/brainy'
|
import { Brainy } from '@soulcraftlabs/brainy'
|
||||||
|
|
||||||
// ✅ CORRECT: Use filesystem storage for production
|
// ✅ CORRECT: Use filesystem storage for production
|
||||||
const brain = new Brainy({
|
const brain = new Brainy({
|
||||||
|
|
@ -115,7 +115,7 @@ Here's a complete React component using the correct patterns:
|
||||||
|
|
||||||
```tsx
|
```tsx
|
||||||
import React, { useState, useEffect } from 'react'
|
import React, { useState, useEffect } from 'react'
|
||||||
import { Brainy } from '@soulcraft/brainy'
|
import { Brainy } from '@soulcraftlabs/brainy'
|
||||||
|
|
||||||
export function FileExplorer() {
|
export function FileExplorer() {
|
||||||
const [brain, setBrain] = useState(null)
|
const [brain, setBrain] = useState(null)
|
||||||
|
|
@ -288,8 +288,8 @@ Your file explorer is now working! Here's what to explore next:
|
||||||
### "Module not found" errors
|
### "Module not found" errors
|
||||||
```bash
|
```bash
|
||||||
# Make sure you're using the right import
|
# Make sure you're using the right import
|
||||||
npm ls @soulcraft/brainy # Check version
|
npm ls @soulcraftlabs/brainy # Check version
|
||||||
npm install @soulcraft/brainy@latest # Update if needed
|
npm install @soulcraftlabs/brainy@latest # Update if needed
|
||||||
```
|
```
|
||||||
|
|
||||||
### "VFS not initialized" errors
|
### "VFS not initialized" errors
|
||||||
|
|
|
||||||
|
|
@ -24,7 +24,7 @@ Brainy VFS is a revolutionary virtual filesystem that runs on top of Brainy's ne
|
||||||
## Quick Start
|
## Quick Start
|
||||||
|
|
||||||
```javascript
|
```javascript
|
||||||
import { VirtualFileSystem } from '@soulcraft/brainy/vfs'
|
import { VirtualFileSystem } from '@soulcraftlabs/brainy/vfs'
|
||||||
|
|
||||||
// Initialize the VFS
|
// Initialize the VFS
|
||||||
const vfs = new VirtualFileSystem({
|
const vfs = new VirtualFileSystem({
|
||||||
|
|
@ -381,7 +381,7 @@ Brainy VFS fully leverages Brainy's revolutionary Triple Intelligence system:
|
||||||
## Installation
|
## Installation
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
npm install @soulcraft/brainy
|
npm install @soulcraftlabs/brainy
|
||||||
```
|
```
|
||||||
|
|
||||||
## Requirements
|
## Requirements
|
||||||
|
|
|
||||||
|
|
@ -135,7 +135,7 @@ Mount VFS as a native filesystem on Linux/Mac/Windows.
|
||||||
|
|
||||||
```typescript
|
```typescript
|
||||||
// Planned (research phase)
|
// Planned (research phase)
|
||||||
import { mountVFS } from '@soulcraft/brainy/vfs/fuse'
|
import { mountVFS } from '@soulcraftlabs/brainy/vfs/fuse'
|
||||||
|
|
||||||
await mountVFS(vfs, {
|
await mountVFS(vfs, {
|
||||||
mountPoint: '/mnt/brainy',
|
mountPoint: '/mnt/brainy',
|
||||||
|
|
@ -160,7 +160,7 @@ These features would benefit from community contributions. If you're interested
|
||||||
### Express.js Static Middleware
|
### Express.js Static Middleware
|
||||||
```typescript
|
```typescript
|
||||||
// Wanted: Community contribution
|
// Wanted: Community contribution
|
||||||
import { createStaticMiddleware } from '@soulcraft/brainy/vfs/express'
|
import { createStaticMiddleware } from '@soulcraftlabs/brainy/vfs/express'
|
||||||
|
|
||||||
app.use('/files', createStaticMiddleware(vfs, {
|
app.use('/files', createStaticMiddleware(vfs, {
|
||||||
index: ['index.html', 'index.md'],
|
index: ['index.html', 'index.md'],
|
||||||
|
|
@ -172,7 +172,7 @@ app.use('/files', createStaticMiddleware(vfs, {
|
||||||
### VSCode Extension
|
### VSCode Extension
|
||||||
```typescript
|
```typescript
|
||||||
// Wanted: Community contribution
|
// Wanted: Community contribution
|
||||||
import { VFSProvider } from '@soulcraft/brainy/vfs/vscode'
|
import { VFSProvider } from '@soulcraftlabs/brainy/vfs/vscode'
|
||||||
|
|
||||||
const provider = new VFSProvider(vfs)
|
const provider = new VFSProvider(vfs)
|
||||||
vscode.workspace.registerFileSystemProvider('brainy', provider)
|
vscode.workspace.registerFileSystemProvider('brainy', provider)
|
||||||
|
|
|
||||||
|
|
@ -327,7 +327,7 @@ console.log(id1 === id2 && id2 === id3) // true
|
||||||
Create your own semantic dimensions:
|
Create your own semantic dimensions:
|
||||||
|
|
||||||
```typescript
|
```typescript
|
||||||
import { BaseProjectionStrategy } from '@soulcraft/brainy/vfs/semantic'
|
import { BaseProjectionStrategy } from '@soulcraftlabs/brainy/vfs/semantic'
|
||||||
|
|
||||||
class PriorityProjection extends BaseProjectionStrategy {
|
class PriorityProjection extends BaseProjectionStrategy {
|
||||||
readonly name = 'priority'
|
readonly name = 'priority'
|
||||||
|
|
|
||||||
|
|
@ -7,7 +7,7 @@ Brainy's Virtual Filesystem (VFS) provides a POSIX-like filesystem interface tha
|
||||||
## Quick Start
|
## Quick Start
|
||||||
|
|
||||||
```typescript
|
```typescript
|
||||||
import { Brainy } from '@soulcraft/brainy'
|
import { Brainy } from '@soulcraftlabs/brainy'
|
||||||
|
|
||||||
// Initialize Brainy
|
// Initialize Brainy
|
||||||
const brain = new Brainy({
|
const brain = new Brainy({
|
||||||
|
|
@ -598,7 +598,7 @@ const user = await store.findById('users', 'user123')
|
||||||
VFS uses standard POSIX-style errors:
|
VFS uses standard POSIX-style errors:
|
||||||
|
|
||||||
```typescript
|
```typescript
|
||||||
import { VFSError, VFSErrorCode } from '@soulcraft/brainy'
|
import { VFSError, VFSErrorCode } from '@soulcraftlabs/brainy'
|
||||||
|
|
||||||
try {
|
try {
|
||||||
await vfs.readFile('/nonexistent.txt')
|
await vfs.readFile('/nonexistent.txt')
|
||||||
|
|
|
||||||
|
|
@ -280,7 +280,7 @@ GitBridge provides Git import/export capabilities:
|
||||||
#### GitBridge Usage
|
#### GitBridge Usage
|
||||||
```javascript
|
```javascript
|
||||||
// Import and instantiate GitBridge
|
// Import and instantiate GitBridge
|
||||||
import { GitBridge } from '@soulcraft/brainy'
|
import { GitBridge } from '@soulcraftlabs/brainy'
|
||||||
const gitBridge = new GitBridge(vfs, brain)
|
const gitBridge = new GitBridge(vfs, brain)
|
||||||
|
|
||||||
// Export VFS to Git repository structure
|
// Export VFS to Git repository structure
|
||||||
|
|
@ -452,7 +452,7 @@ This ordering prevents race conditions where file writes might fail because pare
|
||||||
## Complete Example
|
## Complete Example
|
||||||
|
|
||||||
```javascript
|
```javascript
|
||||||
import { Brainy } from '@soulcraft/brainy'
|
import { Brainy } from '@soulcraftlabs/brainy'
|
||||||
|
|
||||||
async function vfsExample() {
|
async function vfsExample() {
|
||||||
// Initialize
|
// Initialize
|
||||||
|
|
|
||||||
|
|
@ -196,5 +196,5 @@ await brain.relate({
|
||||||
Always import and use the type enums:
|
Always import and use the type enums:
|
||||||
|
|
||||||
```javascript
|
```javascript
|
||||||
import { NounType, VerbType } from '@soulcraft/brainy'
|
import { NounType, VerbType } from '@soulcraftlabs/brainy'
|
||||||
```
|
```
|
||||||
|
|
@ -5,7 +5,7 @@
|
||||||
The Brainy VFS is automatically initialized during `brain.init()`. No separate initialization needed!
|
The Brainy VFS is automatically initialized during `brain.init()`. No separate initialization needed!
|
||||||
|
|
||||||
```javascript
|
```javascript
|
||||||
import { Brainy } from '@soulcraft/brainy'
|
import { Brainy } from '@soulcraftlabs/brainy'
|
||||||
|
|
||||||
// Create and initialize Brainy
|
// Create and initialize Brainy
|
||||||
const brain = new Brainy({
|
const brain = new Brainy({
|
||||||
|
|
@ -71,7 +71,7 @@ VFS stores files as entities and relationships in the same graph as everything e
|
||||||
## Complete Example
|
## Complete Example
|
||||||
|
|
||||||
```javascript
|
```javascript
|
||||||
import { Brainy } from '@soulcraft/brainy'
|
import { Brainy } from '@soulcraftlabs/brainy'
|
||||||
|
|
||||||
async function useVFS() {
|
async function useVFS() {
|
||||||
// Initialize Brainy
|
// Initialize Brainy
|
||||||
|
|
@ -100,7 +100,7 @@ useVFS().catch(console.error)
|
||||||
## TypeScript Usage
|
## TypeScript Usage
|
||||||
|
|
||||||
```typescript
|
```typescript
|
||||||
import { Brainy, VirtualFileSystem } from '@soulcraft/brainy'
|
import { Brainy, VirtualFileSystem } from '@soulcraftlabs/brainy'
|
||||||
|
|
||||||
class FileManager {
|
class FileManager {
|
||||||
private brain: Brainy
|
private brain: Brainy
|
||||||
|
|
|
||||||
|
|
@ -37,7 +37,7 @@ Brainy VFS provides safe, tree-aware methods that prevent these issues:
|
||||||
### Method 1: Use `getDirectChildren()` (Recommended)
|
### Method 1: Use `getDirectChildren()` (Recommended)
|
||||||
|
|
||||||
```typescript
|
```typescript
|
||||||
import { Brainy, VirtualFileSystem } from '@soulcraft/brainy'
|
import { Brainy, VirtualFileSystem } from '@soulcraftlabs/brainy'
|
||||||
|
|
||||||
const brain = new Brainy()
|
const brain = new Brainy()
|
||||||
await brain.init()
|
await brain.init()
|
||||||
|
|
@ -97,7 +97,7 @@ Here's a complete example using React:
|
||||||
|
|
||||||
```tsx
|
```tsx
|
||||||
import React, { useState, useEffect } from 'react'
|
import React, { useState, useEffect } from 'react'
|
||||||
import { VirtualFileSystem } from '@soulcraft/brainy'
|
import { VirtualFileSystem } from '@soulcraftlabs/brainy'
|
||||||
|
|
||||||
interface FileNode {
|
interface FileNode {
|
||||||
name: string
|
name: string
|
||||||
|
|
@ -177,7 +177,7 @@ function TreeView({ node, onToggle, expanded }) {
|
||||||
If you must build trees manually from flat lists, use the `VFSTreeUtils`:
|
If you must build trees manually from flat lists, use the `VFSTreeUtils`:
|
||||||
|
|
||||||
```typescript
|
```typescript
|
||||||
import { VFSTreeUtils } from '@soulcraft/brainy/vfs'
|
import { VFSTreeUtils } from '@soulcraftlabs/brainy/vfs'
|
||||||
|
|
||||||
// Get all entities somehow
|
// Get all entities somehow
|
||||||
const allEntities = await vfs.getDescendants('/root')
|
const allEntities = await vfs.getDescendants('/root')
|
||||||
|
|
|
||||||
|
|
@ -7,7 +7,7 @@
|
||||||
* the Bluesky firehose with Brainy's distributed architecture
|
* the Bluesky firehose with Brainy's distributed architecture
|
||||||
*/
|
*/
|
||||||
|
|
||||||
import { Brainy } from '@soulcraft/brainy'
|
import { Brainy } from '@soulcraftlabs/brainy'
|
||||||
import { WebSocket } from 'ws'
|
import { WebSocket } from 'ws'
|
||||||
|
|
||||||
// =====================================================
|
// =====================================================
|
||||||
|
|
|
||||||
|
|
@ -14,7 +14,7 @@
|
||||||
* ts-node examples/monitor-cache-performance.ts
|
* ts-node examples/monitor-cache-performance.ts
|
||||||
*/
|
*/
|
||||||
|
|
||||||
import { Brainy, NounType } from '@soulcraft/brainy'
|
import { Brainy, NounType } from '@soulcraftlabs/brainy'
|
||||||
|
|
||||||
// ANSI color codes for pretty output
|
// ANSI color codes for pretty output
|
||||||
const colors = {
|
const colors = {
|
||||||
|
|
|
||||||
|
|
@ -5,7 +5,7 @@ Connect Brainy to spreadsheets, BI tools, and external systems with zero configu
|
||||||
## Quick Start
|
## Quick Start
|
||||||
|
|
||||||
```typescript
|
```typescript
|
||||||
import { Brainy } from '@soulcraft/brainy'
|
import { Brainy } from '@soulcraftlabs/brainy'
|
||||||
|
|
||||||
const brain = new Brainy({ integrations: true })
|
const brain = new Brainy({ integrations: true })
|
||||||
await brain.init()
|
await brain.init()
|
||||||
|
|
@ -178,7 +178,7 @@ Webhooks include `X-Brainy-Signature` header with HMAC-SHA256 signature.
|
||||||
### Minimal (in-memory):
|
### Minimal (in-memory):
|
||||||
|
|
||||||
```typescript
|
```typescript
|
||||||
import { Brainy } from '@soulcraft/brainy'
|
import { Brainy } from '@soulcraftlabs/brainy'
|
||||||
|
|
||||||
const brain = new Brainy({ integrations: true })
|
const brain = new Brainy({ integrations: true })
|
||||||
await brain.init()
|
await brain.init()
|
||||||
|
|
@ -194,7 +194,7 @@ console.log(brain.hub.getInstructions())
|
||||||
|
|
||||||
```typescript
|
```typescript
|
||||||
import express from 'express'
|
import express from 'express'
|
||||||
import { Brainy } from '@soulcraft/brainy'
|
import { Brainy } from '@soulcraftlabs/brainy'
|
||||||
|
|
||||||
const app = express()
|
const app = express()
|
||||||
const brain = new Brainy({
|
const brain = new Brainy({
|
||||||
|
|
@ -232,7 +232,7 @@ app.listen(3000, () => {
|
||||||
|
|
||||||
```typescript
|
```typescript
|
||||||
import { Hono } from 'hono'
|
import { Hono } from 'hono'
|
||||||
import { Brainy } from '@soulcraft/brainy'
|
import { Brainy } from '@soulcraftlabs/brainy'
|
||||||
|
|
||||||
const app = new Hono()
|
const app = new Hono()
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -99,7 +99,7 @@ Add the `BRAINY_URL` script property in Apps Script settings.
|
||||||
The simplest way to enable all integrations:
|
The simplest way to enable all integrations:
|
||||||
|
|
||||||
```javascript
|
```javascript
|
||||||
import { Brainy } from '@soulcraft/brainy'
|
import { Brainy } from '@soulcraftlabs/brainy'
|
||||||
|
|
||||||
const brain = new Brainy({ integrations: true })
|
const brain = new Brainy({ integrations: true })
|
||||||
await brain.init()
|
await brain.init()
|
||||||
|
|
@ -112,7 +112,7 @@ With Express:
|
||||||
|
|
||||||
```javascript
|
```javascript
|
||||||
import express from 'express'
|
import express from 'express'
|
||||||
import { Brainy } from '@soulcraft/brainy'
|
import { Brainy } from '@soulcraftlabs/brainy'
|
||||||
|
|
||||||
const app = express()
|
const app = express()
|
||||||
const brain = new Brainy({ integrations: true })
|
const brain = new Brainy({ integrations: true })
|
||||||
|
|
|
||||||
8
package-lock.json
generated
8
package-lock.json
generated
|
|
@ -1,12 +1,12 @@
|
||||||
{
|
{
|
||||||
"name": "@soulcraft/brainy",
|
"name": "@soulcraftlabs/brainy",
|
||||||
"version": "10.3.1",
|
"version": "10.4.13",
|
||||||
"lockfileVersion": 3,
|
"lockfileVersion": 3,
|
||||||
"requires": true,
|
"requires": true,
|
||||||
"packages": {
|
"packages": {
|
||||||
"": {
|
"": {
|
||||||
"name": "@soulcraft/brainy",
|
"name": "@soulcraftlabs/brainy",
|
||||||
"version": "10.3.1",
|
"version": "10.4.13",
|
||||||
"license": "MIT",
|
"license": "MIT",
|
||||||
"dependencies": {
|
"dependencies": {
|
||||||
"@msgpack/msgpack": "^3.1.2",
|
"@msgpack/msgpack": "^3.1.2",
|
||||||
|
|
|
||||||
16
package.json
16
package.json
|
|
@ -1,6 +1,7 @@
|
||||||
{
|
{
|
||||||
"name": "@soulcraft/brainy",
|
"name": "@soulcraftlabs/brainy",
|
||||||
"version": "10.3.1",
|
"version": "10.4.13",
|
||||||
|
"brainyContract": 1,
|
||||||
"description": "Universal Knowledge Protocol™ - World's first Triple Intelligence database unifying vector, graph, and document search in one API. Stage 3 CANONICAL: 42 nouns × 127 verbs covering 96-97% of all human knowledge.",
|
"description": "Universal Knowledge Protocol™ - World's first Triple Intelligence database unifying vector, graph, and document search in one API. Stage 3 CANONICAL: 42 nouns × 127 verbs covering 96-97% of all human knowledge.",
|
||||||
"main": "dist/index.js",
|
"main": "dist/index.js",
|
||||||
"module": "dist/index.js",
|
"module": "dist/index.js",
|
||||||
|
|
@ -87,7 +88,7 @@
|
||||||
"test:watch": "NODE_OPTIONS='--max-old-space-size=8192' vitest --config tests/configs/vitest.unit.config.ts",
|
"test:watch": "NODE_OPTIONS='--max-old-space-size=8192' vitest --config tests/configs/vitest.unit.config.ts",
|
||||||
"test:coverage": "NODE_OPTIONS='--max-old-space-size=8192' vitest run --config tests/configs/vitest.unit.config.ts --coverage",
|
"test:coverage": "NODE_OPTIONS='--max-old-space-size=8192' vitest run --config tests/configs/vitest.unit.config.ts --coverage",
|
||||||
"test:unit": "NODE_OPTIONS='--max-old-space-size=8192' vitest run --config tests/configs/vitest.unit.config.ts",
|
"test:unit": "NODE_OPTIONS='--max-old-space-size=8192' vitest run --config tests/configs/vitest.unit.config.ts",
|
||||||
"test:perf": "vitest run tests/unit/performance --reporter=basic",
|
"test:perf": "vitest run --config tests/configs/vitest.perf.config.ts",
|
||||||
"test:integration": "NODE_OPTIONS='--max-old-space-size=8192' vitest run --config tests/configs/vitest.integration.config.ts",
|
"test:integration": "NODE_OPTIONS='--max-old-space-size=8192' vitest run --config tests/configs/vitest.integration.config.ts",
|
||||||
"test:semantic": "NODE_OPTIONS='--max-old-space-size=8192' vitest run --config tests/configs/vitest.semantic.config.ts",
|
"test:semantic": "NODE_OPTIONS='--max-old-space-size=8192' vitest run --config tests/configs/vitest.semantic.config.ts",
|
||||||
"test:all": "npm run test:unit && npm run test:integration",
|
"test:all": "npm run test:unit && npm run test:integration",
|
||||||
|
|
@ -126,15 +127,16 @@
|
||||||
"license": "MIT",
|
"license": "MIT",
|
||||||
"private": false,
|
"private": false,
|
||||||
"publishConfig": {
|
"publishConfig": {
|
||||||
"access": "public"
|
"access": "public",
|
||||||
|
"registry": "https://source.soulcraft.com/api/packages/soulcraftlabs/npm/"
|
||||||
},
|
},
|
||||||
"homepage": "https://source.soulcraft.com/soulcraft/brainy",
|
"homepage": "https://source.soulcraft.com/soulcraftlabs/open-brainy",
|
||||||
"bugs": {
|
"bugs": {
|
||||||
"url": "https://source.soulcraft.com/soulcraft/brainy/issues"
|
"url": "https://source.soulcraft.com/soulcraftlabs/open-brainy/issues"
|
||||||
},
|
},
|
||||||
"repository": {
|
"repository": {
|
||||||
"type": "git",
|
"type": "git",
|
||||||
"url": "git+https://source.soulcraft.com/soulcraft/brainy.git"
|
"url": "git+https://source.soulcraft.com/soulcraftlabs/open-brainy.git"
|
||||||
},
|
},
|
||||||
"files": [
|
"files": [
|
||||||
"dist/**/*.js",
|
"dist/**/*.js",
|
||||||
|
|
|
||||||
|
|
@ -10,6 +10,7 @@ import { TransformerEmbedding } from '../src/utils/embedding.js'
|
||||||
import * as fs from 'fs/promises'
|
import * as fs from 'fs/promises'
|
||||||
import * as path from 'path'
|
import * as path from 'path'
|
||||||
import { fileURLToPath } from 'url'
|
import { fileURLToPath } from 'url'
|
||||||
|
import { resolveDeterministicStamp } from './lib/deterministicStamp.js'
|
||||||
|
|
||||||
const __dirname = path.dirname(fileURLToPath(import.meta.url))
|
const __dirname = path.dirname(fileURLToPath(import.meta.url))
|
||||||
|
|
||||||
|
|
@ -98,12 +99,21 @@ async function buildEmbeddedPatterns() {
|
||||||
const uint8 = new Uint8Array(buffer)
|
const uint8 = new Uint8Array(buffer)
|
||||||
const base64 = Buffer.from(uint8).toString('base64')
|
const base64 = Buffer.from(uint8).toString('base64')
|
||||||
|
|
||||||
|
// Deterministic stamp: derived from the git commit time of this
|
||||||
|
// generator's inputs, never from wall-clock time — two builds of the
|
||||||
|
// same source tree must produce byte-identical output.
|
||||||
|
const outputPath = path.join(__dirname, '..', 'src', 'neural', 'embeddedPatterns.ts')
|
||||||
|
const generatedStamp = resolveDeterministicStamp(
|
||||||
|
[path.join(__dirname, 'buildEmbeddedPatterns.ts'), libraryPath],
|
||||||
|
outputPath
|
||||||
|
)
|
||||||
|
|
||||||
// Generate TypeScript file with everything embedded
|
// Generate TypeScript file with everything embedded
|
||||||
const tsContent = `/**
|
const tsContent = `/**
|
||||||
* 🧠 BRAINY EMBEDDED PATTERNS
|
* 🧠 BRAINY EMBEDDED PATTERNS
|
||||||
*
|
*
|
||||||
* AUTO-GENERATED - DO NOT EDIT
|
* AUTO-GENERATED - DO NOT EDIT
|
||||||
* Generated: ${new Date().toISOString()}
|
* Generated: ${generatedStamp}
|
||||||
* Patterns: ${libraryData.patterns.length}
|
* Patterns: ${libraryData.patterns.length}
|
||||||
* Coverage: 94-98% of all queries
|
* Coverage: 94-98% of all queries
|
||||||
*
|
*
|
||||||
|
|
@ -197,7 +207,6 @@ prodLog.info(\`🧠 Brainy Pattern Library loaded: \${EMBEDDED_PATTERNS.length}
|
||||||
`
|
`
|
||||||
|
|
||||||
// Write the TypeScript file
|
// Write the TypeScript file
|
||||||
const outputPath = path.join(__dirname, '..', 'src', 'neural', 'embeddedPatterns.ts')
|
|
||||||
await fs.writeFile(outputPath, tsContent)
|
await fs.writeFile(outputPath, tsContent)
|
||||||
|
|
||||||
// Report statistics
|
// Report statistics
|
||||||
|
|
|
||||||
|
|
@ -11,6 +11,7 @@ import * as fs from 'fs/promises'
|
||||||
import * as path from 'path'
|
import * as path from 'path'
|
||||||
import { fileURLToPath } from 'url'
|
import { fileURLToPath } from 'url'
|
||||||
import { NounType, VerbType } from '../src/types/graphTypes.js'
|
import { NounType, VerbType } from '../src/types/graphTypes.js'
|
||||||
|
import { resolveDeterministicStamp } from './lib/deterministicStamp.js'
|
||||||
|
|
||||||
const __dirname = path.dirname(fileURLToPath(import.meta.url))
|
const __dirname = path.dirname(fileURLToPath(import.meta.url))
|
||||||
|
|
||||||
|
|
@ -373,12 +374,24 @@ async function buildTypeEmbeddings() {
|
||||||
const uint8 = new Uint8Array(buffer)
|
const uint8 = new Uint8Array(buffer)
|
||||||
const base64 = Buffer.from(uint8).toString('base64')
|
const base64 = Buffer.from(uint8).toString('base64')
|
||||||
|
|
||||||
|
// Deterministic stamp: derived from the git commit time of this
|
||||||
|
// generator's inputs, never from wall-clock time — two builds of the
|
||||||
|
// same source tree must produce byte-identical output.
|
||||||
|
const outputPath = path.join(__dirname, '..', 'src', 'neural', 'embeddedTypeEmbeddings.ts')
|
||||||
|
const generatedStamp = resolveDeterministicStamp(
|
||||||
|
[
|
||||||
|
path.join(__dirname, 'buildTypeEmbeddings.ts'),
|
||||||
|
path.join(__dirname, '..', 'src', 'types', 'graphTypes.ts')
|
||||||
|
],
|
||||||
|
outputPath
|
||||||
|
)
|
||||||
|
|
||||||
// Generate TypeScript file
|
// Generate TypeScript file
|
||||||
const tsContent = `/**
|
const tsContent = `/**
|
||||||
* 🧠 BRAINY EMBEDDED TYPE EMBEDDINGS
|
* 🧠 BRAINY EMBEDDED TYPE EMBEDDINGS
|
||||||
*
|
*
|
||||||
* AUTO-GENERATED - DO NOT EDIT
|
* AUTO-GENERATED - DO NOT EDIT
|
||||||
* Generated: ${new Date().toISOString()}
|
* Generated: ${generatedStamp}
|
||||||
* Noun Types: ${nounTypes.length}
|
* Noun Types: ${nounTypes.length}
|
||||||
* Verb Types: ${verbTypes.length}
|
* Verb Types: ${verbTypes.length}
|
||||||
*
|
*
|
||||||
|
|
@ -395,7 +408,7 @@ export const TYPE_METADATA = {
|
||||||
verbTypes: ${verbTypes.length},
|
verbTypes: ${verbTypes.length},
|
||||||
totalTypes: ${totalTypes},
|
totalTypes: ${totalTypes},
|
||||||
embeddingDimensions: ${embeddingDim},
|
embeddingDimensions: ${embeddingDim},
|
||||||
generatedAt: "${new Date().toISOString()}",
|
generatedAt: "${generatedStamp}",
|
||||||
sizeBytes: {
|
sizeBytes: {
|
||||||
embeddings: ${buffer.byteLength},
|
embeddings: ${buffer.byteLength},
|
||||||
base64: ${base64.length}
|
base64: ${base64.length}
|
||||||
|
|
@ -494,7 +507,6 @@ prodLog.info(\`🧠 Brainy Type Embeddings loaded: \${TYPE_METADATA.nounTypes} n
|
||||||
`
|
`
|
||||||
|
|
||||||
// Write the TypeScript file
|
// Write the TypeScript file
|
||||||
const outputPath = path.join(__dirname, '..', 'src', 'neural', 'embeddedTypeEmbeddings.ts')
|
|
||||||
await fs.writeFile(outputPath, tsContent)
|
await fs.writeFile(outputPath, tsContent)
|
||||||
|
|
||||||
// Report statistics
|
// Report statistics
|
||||||
|
|
|
||||||
128
scripts/emit-contract-manifest.mjs
Normal file
128
scripts/emit-contract-manifest.mjs
Normal file
|
|
@ -0,0 +1,128 @@
|
||||||
|
#!/usr/bin/env node
|
||||||
|
/**
|
||||||
|
* Emit this build's API-contract manifest to docs/api-contract.json.
|
||||||
|
*
|
||||||
|
* WHY IT IS GENERATED, NOT WRITTEN: a hand-kept list of doors drifts from the
|
||||||
|
* code the first time somebody adds one. This reads the surface the build
|
||||||
|
* actually exposes — the prototype's own methods and accessors, the exported
|
||||||
|
* error classes, the `where` operator sets, the field-addressing vocabulary,
|
||||||
|
* the health verdicts — so a diff between two engines' manifests is a diff
|
||||||
|
* between two engines, never between two authors.
|
||||||
|
*
|
||||||
|
* Requirement marking (required / optional per door) is NOT derivable from the
|
||||||
|
* surface — it is a commitment, recorded with the contract's owner rather than
|
||||||
|
* here. This manifest carries the surface; the promise lives with the contract.
|
||||||
|
*
|
||||||
|
* Usage: node scripts/emit-contract-manifest.mjs [--check]
|
||||||
|
* --check exits non-zero when the committed manifest is stale.
|
||||||
|
*/
|
||||||
|
|
||||||
|
import { writeFileSync, readFileSync, existsSync } from 'node:fs'
|
||||||
|
import { join, dirname } from 'node:path'
|
||||||
|
import { fileURLToPath } from 'node:url'
|
||||||
|
|
||||||
|
const ROOT = join(dirname(fileURLToPath(import.meta.url)), '..')
|
||||||
|
const OUT = join(ROOT, 'docs', 'api-contract.json')
|
||||||
|
|
||||||
|
const { Brainy } = await import(join(ROOT, 'dist', 'brainy.js'))
|
||||||
|
const errorsModule = await import(join(ROOT, 'dist', 'errors', 'brainyError.js'))
|
||||||
|
const versionModule = await import(join(ROOT, 'dist', 'utils', 'version.js'))
|
||||||
|
const fieldAddressing = await import(join(ROOT, 'dist', 'db', 'fieldAddressing.js'))
|
||||||
|
|
||||||
|
/** Every own method and accessor on the class's prototype, minus the private ones. */
|
||||||
|
function surfaceOf(ctor) {
|
||||||
|
const doors = []
|
||||||
|
for (const name of Object.getOwnPropertyNames(ctor.prototype)) {
|
||||||
|
if (name === 'constructor' || name.startsWith('_')) continue
|
||||||
|
const descriptor = Object.getOwnPropertyDescriptor(ctor.prototype, name)
|
||||||
|
if (!descriptor) continue
|
||||||
|
if (typeof descriptor.value === 'function') {
|
||||||
|
doors.push({ name, kind: 'method', arity: descriptor.value.length })
|
||||||
|
} else if (descriptor.get) {
|
||||||
|
doors.push({ name, kind: 'accessor' })
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return doors.sort((a, b) => a.name.localeCompare(b.name))
|
||||||
|
}
|
||||||
|
|
||||||
|
const errors = Object.entries(errorsModule)
|
||||||
|
.filter(([name, value]) => typeof value === 'function' && /Error$/.test(name))
|
||||||
|
.map(([name]) => name)
|
||||||
|
.sort()
|
||||||
|
|
||||||
|
// The operator sets, read from the engine's own refusal message so the
|
||||||
|
// manifest can never disagree with the validator.
|
||||||
|
const filterSource = readFileSync(join(ROOT, 'src', 'utils', 'metadataFilter.ts'), 'utf-8')
|
||||||
|
const acceptedMatch = filterSource.match(/const VALUE_OPERATORS = new Set<string>\(\[([\s\S]*?)\]\)/)
|
||||||
|
if (!acceptedMatch) throw new Error('VALUE_OPERATORS not found — the manifest refuses to guess')
|
||||||
|
const accepted = [...acceptedMatch[1].matchAll(/'([^']+)'/g)].map((m) => m[1]).sort()
|
||||||
|
|
||||||
|
const indexSource = readFileSync(join(ROOT, 'src', 'utils', 'metadataIndex.ts'), 'utf-8')
|
||||||
|
const refusedByIndex = ['endsWith', 'length', 'matches', 'startsWith'].filter((op) =>
|
||||||
|
// Proven by the refusal path: these are the tokens with no case in the
|
||||||
|
// index's operator switch, so they fall to its default and are refused.
|
||||||
|
!new RegExp(`case '${op}':`).test(indexSource)
|
||||||
|
)
|
||||||
|
const servedOnIndex = accepted.filter((op) => !refusedByIndex.includes(op))
|
||||||
|
|
||||||
|
const manifest = {
|
||||||
|
contractVersion: versionModule.contractVersion(),
|
||||||
|
engine: '@soulcraftlabs/brainy',
|
||||||
|
compatibility: {
|
||||||
|
minor:
|
||||||
|
'additive — a new optional door, a new served operator, a new error class; every existing implementation still conforms',
|
||||||
|
major:
|
||||||
|
'breaking — a door removed, an answer narrowed, an ordering law changed, an optional door promoted to required, or an operator moved from served to refused'
|
||||||
|
},
|
||||||
|
doors: surfaceOf(Brainy),
|
||||||
|
errors,
|
||||||
|
operators: {
|
||||||
|
accepted,
|
||||||
|
servedOnIndexPath: servedOnIndex,
|
||||||
|
refusedByIndexPath: refusedByIndex,
|
||||||
|
combinators: ['allOf', 'anyOf', 'not']
|
||||||
|
},
|
||||||
|
fieldAddressing: {
|
||||||
|
systemKeyPrefix: 'system.',
|
||||||
|
systemEntityScalars: [...(fieldAddressing.SYSTEM_ENTITY_SCALARS ?? [])].sort(),
|
||||||
|
systemRelationScalars: [...(fieldAddressing.SYSTEM_RELATION_SCALARS ?? [])].sort(),
|
||||||
|
plumbingFields: [...(fieldAddressing.PLUMBING_FIELDS ?? [])].sort()
|
||||||
|
},
|
||||||
|
health: {
|
||||||
|
verdicts: ['pass', 'warn', 'fail'],
|
||||||
|
healKinds: ['none', 'repair', 'rebuild'],
|
||||||
|
servingWithholdingInvariants: [
|
||||||
|
'index-initialized',
|
||||||
|
'durable-state-present',
|
||||||
|
'manifest-residency',
|
||||||
|
'replay-clean',
|
||||||
|
'strand-latch'
|
||||||
|
]
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
const rendered = `${JSON.stringify(manifest, null, 2)}\n`
|
||||||
|
|
||||||
|
if (process.argv.includes('--check')) {
|
||||||
|
if (!existsSync(OUT)) {
|
||||||
|
console.error(`docs/api-contract.json is missing — run: node scripts/emit-contract-manifest.mjs`)
|
||||||
|
process.exit(1)
|
||||||
|
}
|
||||||
|
if (readFileSync(OUT, 'utf-8') !== rendered) {
|
||||||
|
console.error(
|
||||||
|
`docs/api-contract.json is STALE — the public surface changed. Re-emit it and announce ` +
|
||||||
|
`the addition (minor = additive; a removal is a contract major).`
|
||||||
|
)
|
||||||
|
process.exit(1)
|
||||||
|
}
|
||||||
|
console.log(`docs/api-contract.json is current (${manifest.doors.length} doors, contract ${manifest.contractVersion}).`)
|
||||||
|
process.exit(0)
|
||||||
|
}
|
||||||
|
|
||||||
|
writeFileSync(OUT, rendered)
|
||||||
|
console.log(
|
||||||
|
`Wrote docs/api-contract.json — contract ${manifest.contractVersion}, ` +
|
||||||
|
`${manifest.doors.length} doors, ${manifest.errors.length} error classes, ` +
|
||||||
|
`${manifest.operators.accepted.length} operators ` +
|
||||||
|
`(${manifest.operators.refusedByIndexPath.length} refused by the index path).`
|
||||||
|
)
|
||||||
118
scripts/lib/deterministicStamp.ts
Normal file
118
scripts/lib/deterministicStamp.ts
Normal file
|
|
@ -0,0 +1,118 @@
|
||||||
|
/**
|
||||||
|
* Deterministic generation-stamp resolution for Brainy's build-time code
|
||||||
|
* generators.
|
||||||
|
*
|
||||||
|
* Two builds of the same source tree must produce byte-identical output.
|
||||||
|
* A wall-clock stamp (`new Date()`) breaks that guarantee, so every
|
||||||
|
* generator that writes a "Generated:" header or a `generatedAt` field
|
||||||
|
* into its output must resolve the stamp through this module instead.
|
||||||
|
*
|
||||||
|
* Resolution order:
|
||||||
|
* 1. The newest git commit timestamp among the generator's input files
|
||||||
|
* (the generator script itself always counts as an input).
|
||||||
|
* 2. If git metadata is unavailable (for example, building from a
|
||||||
|
* published npm tarball with no `.git` directory), the stamp already
|
||||||
|
* recorded in the previously generated output file.
|
||||||
|
* 3. If neither is available, the fixed epoch string
|
||||||
|
* `1970-01-01T00:00:00.000Z`.
|
||||||
|
*
|
||||||
|
* Every fallback logs a line to stderr — deterministic degradation is
|
||||||
|
* loud, never a silent divergence.
|
||||||
|
*/
|
||||||
|
|
||||||
|
import { execFileSync } from 'child_process'
|
||||||
|
import * as fs from 'fs'
|
||||||
|
|
||||||
|
const EPOCH_STAMP = '1970-01-01T00:00:00.000Z'
|
||||||
|
const STAMP_PATTERN = /\*\s*Generated:\s*(\S+)/
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Resolve the deterministic stamp for a generator run.
|
||||||
|
*
|
||||||
|
* @param inputPaths Absolute paths to every file whose content determines
|
||||||
|
* the generator's output, including the generator script itself.
|
||||||
|
* @param previousOutputPath Absolute path to the previously generated
|
||||||
|
* file, used for the existing-stamp fallback when git is unavailable.
|
||||||
|
* @returns An ISO-8601 timestamp string that is deterministic for a given
|
||||||
|
* source tree.
|
||||||
|
*/
|
||||||
|
export function resolveDeterministicStamp(
|
||||||
|
inputPaths: string[],
|
||||||
|
previousOutputPath: string
|
||||||
|
): string {
|
||||||
|
const gitStamp = newestGitCommitTimestamp(inputPaths)
|
||||||
|
if (gitStamp) {
|
||||||
|
return gitStamp
|
||||||
|
}
|
||||||
|
|
||||||
|
const existingStamp = readExistingStamp(previousOutputPath)
|
||||||
|
if (existingStamp) {
|
||||||
|
process.stderr.write(
|
||||||
|
`[deterministic-stamp] no git commit history found for generator inputs; ` +
|
||||||
|
`reusing existing stamp from ${previousOutputPath}: ${existingStamp}\n`
|
||||||
|
)
|
||||||
|
return existingStamp
|
||||||
|
}
|
||||||
|
|
||||||
|
process.stderr.write(
|
||||||
|
`[deterministic-stamp] no git commit history and no previous output at ` +
|
||||||
|
`${previousOutputPath}; falling back to fixed epoch stamp ${EPOCH_STAMP}\n`
|
||||||
|
)
|
||||||
|
return EPOCH_STAMP
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Find the newest git commit timestamp among the given input paths.
|
||||||
|
* Returns null if git is unavailable, the tree is not a git repository,
|
||||||
|
* or none of the inputs have any commit history yet.
|
||||||
|
*/
|
||||||
|
function newestGitCommitTimestamp(inputPaths: string[]): string | null {
|
||||||
|
let newest: string | null = null
|
||||||
|
|
||||||
|
for (const inputPath of inputPaths) {
|
||||||
|
if (!fs.existsSync(inputPath)) {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
|
||||||
|
let out: string
|
||||||
|
try {
|
||||||
|
out = execFileSync(
|
||||||
|
'git',
|
||||||
|
['log', '-1', '--format=%cI', '--', inputPath],
|
||||||
|
{ stdio: ['ignore', 'pipe', 'ignore'] }
|
||||||
|
)
|
||||||
|
.toString()
|
||||||
|
.trim()
|
||||||
|
} catch {
|
||||||
|
// git missing, not a repository, or no permissions — handled by the
|
||||||
|
// caller's fallback chain.
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
|
||||||
|
if (!out) {
|
||||||
|
// Path exists but has no commit history yet (e.g. newly created,
|
||||||
|
// uncommitted file).
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
|
||||||
|
if (!newest || new Date(out).getTime() > new Date(newest).getTime()) {
|
||||||
|
newest = out
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
return newest
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Parse the `* Generated: <ISO timestamp>` header out of a previously
|
||||||
|
* generated file, if one exists.
|
||||||
|
*/
|
||||||
|
function readExistingStamp(outputPath: string): string | null {
|
||||||
|
if (!fs.existsSync(outputPath)) {
|
||||||
|
return null
|
||||||
|
}
|
||||||
|
|
||||||
|
const content = fs.readFileSync(outputPath, 'utf-8')
|
||||||
|
const match = content.match(STAMP_PATTERN)
|
||||||
|
return match ? match[1] : null
|
||||||
|
}
|
||||||
|
|
@ -15,6 +15,12 @@ NC='\033[0m' # No Color
|
||||||
RELEASE_TYPE="${1:-patch}" # patch, minor, or major
|
RELEASE_TYPE="${1:-patch}" # patch, minor, or major
|
||||||
SKIP_TESTS=false
|
SKIP_TESTS=false
|
||||||
DRY_RUN=false
|
DRY_RUN=false
|
||||||
|
# --source-only is now a no-op: The Source is the one registry, so every
|
||||||
|
# release already ships Source-only — tag, CI's publish to The Source, the
|
||||||
|
# release page, and the docs push, with no separate storefront leg to skip.
|
||||||
|
# The flag is still accepted (for backward-compatible invocations) and just
|
||||||
|
# prints a notice; it no longer changes behavior.
|
||||||
|
SOURCE_ONLY=false
|
||||||
|
|
||||||
for arg in "$@"; do
|
for arg in "$@"; do
|
||||||
case $arg in
|
case $arg in
|
||||||
|
|
@ -24,6 +30,9 @@ for arg in "$@"; do
|
||||||
--dry-run)
|
--dry-run)
|
||||||
DRY_RUN=true
|
DRY_RUN=true
|
||||||
;;
|
;;
|
||||||
|
--source-only)
|
||||||
|
SOURCE_ONLY=true
|
||||||
|
;;
|
||||||
esac
|
esac
|
||||||
done
|
done
|
||||||
|
|
||||||
|
|
@ -100,7 +109,7 @@ else
|
||||||
;;
|
;;
|
||||||
*)
|
*)
|
||||||
echo -e "${RED}❌ Invalid release type: ${RELEASE_TYPE}${NC}"
|
echo -e "${RED}❌ Invalid release type: ${RELEASE_TYPE}${NC}"
|
||||||
echo "Usage: ./scripts/release.sh [patch|minor|major|<explicit-version>] [--dry-run]"
|
echo "Usage: ./scripts/release.sh [patch|minor|major|<explicit-version>] [--dry-run] [--source-only (no-op; The Source is the one registry)]"
|
||||||
exit 1
|
exit 1
|
||||||
;;
|
;;
|
||||||
esac
|
esac
|
||||||
|
|
@ -119,6 +128,9 @@ echo -e "${BLUE}New version: ${NEW_VERSION}${NC}"
|
||||||
if [ "$PRERELEASE" = true ]; then
|
if [ "$PRERELEASE" = true ]; then
|
||||||
echo -e "${YELLOW}⚠️ Prerelease → npm dist-tag '${NPM_TAG}', GitHub prerelease${NC}"
|
echo -e "${YELLOW}⚠️ Prerelease → npm dist-tag '${NPM_TAG}', GitHub prerelease${NC}"
|
||||||
fi
|
fi
|
||||||
|
if [ "$SOURCE_ONLY" = true ]; then
|
||||||
|
echo -e "${YELLOW}⚠️ The Source is the one registry; --source-only is implied${NC}"
|
||||||
|
fi
|
||||||
echo ""
|
echo ""
|
||||||
|
|
||||||
if [ "$DRY_RUN" = true ]; then
|
if [ "$DRY_RUN" = true ]; then
|
||||||
|
|
@ -142,13 +154,26 @@ else
|
||||||
fi
|
fi
|
||||||
|
|
||||||
# Create new changelog entry
|
# Create new changelog entry
|
||||||
CHANGELOG_ENTRY="### [${NEW_VERSION}](https://source.soulcraft.com/soulcraft/brainy/compare/v${CURRENT_VERSION}...v${NEW_VERSION}) ($(date +%Y-%m-%d))
|
RELEASE_DATE=$(date +%Y-%m-%d)
|
||||||
|
CHANGELOG_ENTRY="### [${NEW_VERSION}](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v${CURRENT_VERSION}...v${NEW_VERSION}) (${RELEASE_DATE})
|
||||||
|
|
||||||
${COMMITS}
|
${COMMITS}
|
||||||
"
|
"
|
||||||
|
# A CURATED entry wins over the generated one. When a release is cut from a
|
||||||
|
# lineage that diverged from the previous tag (a candidate branch carrying
|
||||||
|
# main's history), `git log <last-tag>..HEAD` lists every commit the tag never
|
||||||
|
# saw — old notes, already-shipped fixes under new hashes, merge commits — and a
|
||||||
|
# wall entry derived from it would misreport the release. If CHANGELOG.md
|
||||||
|
# already carries a `### [NEW_VERSION]` heading, it was written on purpose:
|
||||||
|
# keep it, and skip the generated prepend entirely.
|
||||||
|
CURATED_ENTRY=false
|
||||||
|
if grep -qE "^### \[${NEW_VERSION}\]" CHANGELOG.md 2>/dev/null; then
|
||||||
|
CURATED_ENTRY=true
|
||||||
|
echo -e "${YELLOW}CHANGELOG already carries a curated ### [${NEW_VERSION}] entry — keeping it, not generating one from commits${NC}"
|
||||||
|
fi
|
||||||
|
|
||||||
# Prepend to CHANGELOG.md after header
|
# Prepend to CHANGELOG.md after header
|
||||||
if [ -f "CHANGELOG.md" ]; then
|
if [ "$CURATED_ENTRY" = false ] && [ -f "CHANGELOG.md" ]; then
|
||||||
# Read header (first 4 lines)
|
# Read header (first 4 lines)
|
||||||
HEADER=$(head -n 4 CHANGELOG.md)
|
HEADER=$(head -n 4 CHANGELOG.md)
|
||||||
# Read rest of file
|
# Read rest of file
|
||||||
|
|
@ -162,6 +187,19 @@ if [ -f "CHANGELOG.md" ]; then
|
||||||
fi
|
fi
|
||||||
echo -e "${GREEN}✅ CHANGELOG updated${NC}\n"
|
echo -e "${GREEN}✅ CHANGELOG updated${NC}\n"
|
||||||
|
|
||||||
|
# Step 6b: Update the releases wall entry — mechanical, derived from the
|
||||||
|
# CHANGELOG entry just composed. The fleet's HQ page reads open-brainy.json
|
||||||
|
# from the one shared releases repo, soulcraftlabs/releases on The Source —
|
||||||
|
# this used to be hand-written after every release (David: never again —
|
||||||
|
# make it a step of the rail, landed in the one shared home; this repo no
|
||||||
|
# longer hosts its own copy). This step clones/fetches that repo into a
|
||||||
|
# local cache, prepends the entry, and pushes it directly — a real
|
||||||
|
# cross-repo push, refusing loudly (never skipping) on any
|
||||||
|
# clone/validation/commit/push failure.
|
||||||
|
echo -e "${BLUE}5️⃣▸ Updating the releases wall...${NC}"
|
||||||
|
node scripts/wall-entry.mjs --product open-brainy --version "${NEW_VERSION}" --date "${RELEASE_DATE}" --from-changelog CHANGELOG.md
|
||||||
|
echo -e "${GREEN}✅ Releases wall updated${NC}\n"
|
||||||
|
|
||||||
# Step 7: Create release commit
|
# Step 7: Create release commit
|
||||||
echo -e "${BLUE}6️⃣ Creating release commit...${NC}"
|
echo -e "${BLUE}6️⃣ Creating release commit...${NC}"
|
||||||
git add package.json package-lock.json CHANGELOG.md
|
git add package.json package-lock.json CHANGELOG.md
|
||||||
|
|
@ -193,9 +231,9 @@ echo -e "${GREEN}✅ Pushed to origin${NC}\n"
|
||||||
# .forgejo/workflows/publish-source.yml, which builds and publishes on The
|
# .forgejo/workflows/publish-source.yml, which builds and publishes on The
|
||||||
# Source's own runner (datacenter-side: seconds, not the laptop's WAN timing
|
# Source's own runner (datacenter-side: seconds, not the laptop's WAN timing
|
||||||
# out on an 87MB tarball PUT). The laptop holds no home-registry publish
|
# out on an 87MB tarball PUT). The laptop holds no home-registry publish
|
||||||
# credential anymore; it only waits for CI's result before trusting the
|
# credential anymore; it only waits for CI's result before continuing on to
|
||||||
# home/npmjs pair enough to publish the storefront leg.
|
# the release page and the docs push.
|
||||||
SOURCE_NPM_REG="https://source.soulcraft.com/api/packages/soulcraft/npm/"
|
SOURCE_NPM_REG="https://source.soulcraft.com/api/packages/soulcraftlabs/npm/"
|
||||||
SOURCE_POLL_INTERVAL_S=15
|
SOURCE_POLL_INTERVAL_S=15
|
||||||
SOURCE_POLL_MAX_ATTEMPTS=200 # 200 × 15s = 50 minutes — the runner is sequential and a busy day's ci.yml
|
SOURCE_POLL_MAX_ATTEMPTS=200 # 200 × 15s = 50 minutes — the runner is sequential and a busy day's ci.yml
|
||||||
# backlog has twice exceeded the old 20-minute window (8.10.3, 9.0.0);
|
# backlog has twice exceeded the old 20-minute window (8.10.3, 9.0.0);
|
||||||
|
|
@ -203,7 +241,7 @@ SOURCE_POLL_MAX_ATTEMPTS=200 # 200 × 15s = 50 minutes — the runner is sequen
|
||||||
echo -e "${BLUE}9️⃣ Waiting for CI to publish v${NEW_VERSION} to The Source registry (home)...${NC}"
|
echo -e "${BLUE}9️⃣ Waiting for CI to publish v${NEW_VERSION} to The Source registry (home)...${NC}"
|
||||||
SOURCE_LANDED=false
|
SOURCE_LANDED=false
|
||||||
for ((attempt = 1; attempt <= SOURCE_POLL_MAX_ATTEMPTS; attempt++)); do
|
for ((attempt = 1; attempt <= SOURCE_POLL_MAX_ATTEMPTS; attempt++)); do
|
||||||
LANDED_VERSION=$(npm view "@soulcraft/brainy@${NEW_VERSION}" version "--@soulcraft:registry=${SOURCE_NPM_REG}" 2>/dev/null || echo "")
|
LANDED_VERSION=$(npm view "@soulcraftlabs/brainy@${NEW_VERSION}" version "--@soulcraftlabs:registry=${SOURCE_NPM_REG}" 2>/dev/null || echo "")
|
||||||
if [ "$LANDED_VERSION" = "$NEW_VERSION" ]; then
|
if [ "$LANDED_VERSION" = "$NEW_VERSION" ]; then
|
||||||
SOURCE_LANDED=true
|
SOURCE_LANDED=true
|
||||||
break
|
break
|
||||||
|
|
@ -216,50 +254,8 @@ if [ "$SOURCE_LANDED" = true ]; then
|
||||||
echo -e "${GREEN}✅ CI published v${NEW_VERSION} to The Source${NC}\n"
|
echo -e "${GREEN}✅ CI published v${NEW_VERSION} to The Source${NC}\n"
|
||||||
else
|
else
|
||||||
echo -e "${RED}❌ CI's home publish did not land — check the workflow run on The Source; the pair must not diverge.${NC}"
|
echo -e "${RED}❌ CI's home publish did not land — check the workflow run on The Source; the pair must not diverge.${NC}"
|
||||||
echo -e "${RED} v${NEW_VERSION} was tagged and pushed, but @soulcraft/brainy@${NEW_VERSION} never became visible on the${NC}"
|
echo -e "${RED} v${NEW_VERSION} was tagged and pushed, but @soulcraftlabs/brainy@${NEW_VERSION} never became visible on the${NC}"
|
||||||
echo -e "${RED} Source registry after ${SOURCE_POLL_MAX_ATTEMPTS} attempts, ${SOURCE_POLL_INTERVAL_S}s apart. Aborting before npmjs.${NC}"
|
echo -e "${RED} Source registry after ${SOURCE_POLL_MAX_ATTEMPTS} attempts, ${SOURCE_POLL_INTERVAL_S}s apart. Aborting.${NC}"
|
||||||
exit 1
|
|
||||||
fi
|
|
||||||
|
|
||||||
echo -e "${BLUE}9️⃣½ Publishing to npmjs (storefront, dist-tag: ${NPM_TAG})...${NC}"
|
|
||||||
# BYTE-IDENTITY LAW: the storefront republishes CI's EXACT artifact — download
|
|
||||||
# the tarball The Source serves and publish that file, never a fresh local pack
|
|
||||||
# (a local rebuild can differ byte-wise, and the fleet verifies the pair by
|
|
||||||
# shasum across registries).
|
|
||||||
STOREFRONT_TMP="$(mktemp -d)"
|
|
||||||
(cd "$STOREFRONT_TMP" && npm pack "@soulcraft/brainy@${NEW_VERSION}" "--@soulcraft:registry=${SOURCE_NPM_REG}" >/dev/null)
|
|
||||||
SOURCE_TARBALL="$(ls "$STOREFRONT_TMP"/soulcraft-brainy-*.tgz)"
|
|
||||||
echo -e "${BLUE} home artifact: $(sha256sum "$SOURCE_TARBALL" | cut -d' ' -f1)${NC}"
|
|
||||||
npm publish "$SOURCE_TARBALL" --tag "$NPM_TAG" "--@soulcraft:registry=https://registry.npmjs.org/"
|
|
||||||
rm -rf "$STOREFRONT_TMP"
|
|
||||||
# Brainy is the only PUBLIC @soulcraft package — verify visibility after every publish.
|
|
||||||
npm access get status @soulcraft/brainy "--@soulcraft:registry=https://registry.npmjs.org/" || true
|
|
||||||
# Verify the pair is byte-identical by registry-reported shasum — divergence
|
|
||||||
# here means the storefront leg must be treated as failed, loudly. RETRIED
|
|
||||||
# with raw curl: npmjs metadata propagates with a lag measured in minutes,
|
|
||||||
# and a one-shot npm-view probe fired a false DIVERGENCE on 10.0.0 while a
|
|
||||||
# raw curl of the registry document already confirmed byte-identity. The
|
|
||||||
# probe now reads the registry JSON directly (no npm cache in the path) and
|
|
||||||
# gives propagation up to 5 minutes before calling the pair divergent.
|
|
||||||
NPMJS_VERIFY_ATTEMPTS=20
|
|
||||||
NPMJS_VERIFY_INTERVAL_S=15 # 20 × 15s = 5 minutes of propagation grace
|
|
||||||
SOURCE_SHA=$(npm view "@soulcraft/brainy@${NEW_VERSION}" dist.shasum "--@soulcraft:registry=${SOURCE_NPM_REG}" 2>/dev/null || echo "source-unavailable")
|
|
||||||
PAIR_IDENTICAL=false
|
|
||||||
for ((attempt = 1; attempt <= NPMJS_VERIFY_ATTEMPTS; attempt++)); do
|
|
||||||
NPMJS_SHA=$(curl -fsSL "https://registry.npmjs.org/@soulcraft%2Fbrainy" 2>/dev/null \
|
|
||||||
| node -e "let d='';process.stdin.on('data',c=>d+=c).on('end',()=>{try{const v=JSON.parse(d).versions[process.argv[1]];console.log(v?v.dist.shasum:'')}catch{console.log('')}})" "${NEW_VERSION}" \
|
|
||||||
|| echo "")
|
|
||||||
if [ -n "$NPMJS_SHA" ] && [ "$SOURCE_SHA" = "$NPMJS_SHA" ]; then
|
|
||||||
PAIR_IDENTICAL=true
|
|
||||||
break
|
|
||||||
fi
|
|
||||||
echo -e "${YELLOW} … npmjs metadata not settled (attempt ${attempt}/${NPMJS_VERIFY_ATTEMPTS}: '${NPMJS_SHA:-absent}' vs '${SOURCE_SHA}'); retrying in ${NPMJS_VERIFY_INTERVAL_S}s${NC}"
|
|
||||||
sleep "$NPMJS_VERIFY_INTERVAL_S"
|
|
||||||
done
|
|
||||||
if [ "$PAIR_IDENTICAL" = true ]; then
|
|
||||||
echo -e "${GREEN}✅ Published to npmjs — byte-identical pair (shasum ${NPMJS_SHA})${NC}\n"
|
|
||||||
else
|
|
||||||
echo -e "${RED}❌ REGISTRY DIVERGENCE: The Source shasum ${SOURCE_SHA} != npmjs shasum ${NPMJS_SHA} after ${NPMJS_VERIFY_ATTEMPTS} attempts — investigate before announcing${NC}\n"
|
|
||||||
exit 1
|
exit 1
|
||||||
fi
|
fi
|
||||||
|
|
||||||
|
|
@ -267,7 +263,7 @@ fi
|
||||||
# and RELEASES.md are the record; this just gives The Source's UI a release page).
|
# and RELEASES.md are the record; this just gives The Source's UI a release page).
|
||||||
echo -e "${BLUE}🔟 Creating release page on The Source...${NC}"
|
echo -e "${BLUE}🔟 Creating release page on The Source...${NC}"
|
||||||
if [ -n "${FORGEJO_RELEASE_TOKEN:-}" ]; then
|
if [ -n "${FORGEJO_RELEASE_TOKEN:-}" ]; then
|
||||||
if curl -sf -X POST "https://source.soulcraft.com/api/v1/repos/soulcraft/brainy/releases" \
|
if curl -sf -X POST "https://source.soulcraft.com/api/v1/repos/soulcraftlabs/open-brainy/releases" \
|
||||||
-H "Authorization: token ${FORGEJO_RELEASE_TOKEN}" -H "Content-Type: application/json" \
|
-H "Authorization: token ${FORGEJO_RELEASE_TOKEN}" -H "Content-Type: application/json" \
|
||||||
-d "{\"tag_name\":\"v${NEW_VERSION}\",\"name\":\"v${NEW_VERSION}\",\"prerelease\":${PRERELEASE}}" >/dev/null; then
|
-d "{\"tag_name\":\"v${NEW_VERSION}\",\"name\":\"v${NEW_VERSION}\",\"prerelease\":${PRERELEASE}}" >/dev/null; then
|
||||||
echo -e "${GREEN}✅ Release page created on The Source${NC}\n"
|
echo -e "${GREEN}✅ Release page created on The Source${NC}\n"
|
||||||
|
|
@ -278,21 +274,15 @@ else
|
||||||
echo -e "${RED}⚠️ FORGEJO_RELEASE_TOKEN unset — no release page created; tag + CHANGELOG remain the record${NC}\n"
|
echo -e "${RED}⚠️ FORGEJO_RELEASE_TOKEN unset — no release page created; tag + CHANGELOG remain the record${NC}\n"
|
||||||
fi
|
fi
|
||||||
|
|
||||||
# Step 12: Push public docs to the soulcraft.com docs ingest door
|
# Step 12 RETIRED (2026-08-31, CORTEX-SITE-BRAINY-RENAME round 12, David-ruled):
|
||||||
# (VENUE-DOCS-RELEASE-PUSH). Skips with a loud warning when
|
# soulcraft.com/docs carries the paid product's documentation only. This
|
||||||
# DOCS_INGEST_SECRET is unset; fails loudly (without undoing the publish —
|
# engine's documentation home is THIS repository — README and docs/ — and the
|
||||||
# that already happened) when a push errors, so the docs site never
|
# site serves 301s for the slugs this rail used to push. The push script stays
|
||||||
# silently trails npm.
|
# in the tree for history; the rail no longer calls it.
|
||||||
echo -e "${BLUE}1️⃣2️⃣ Pushing public docs to soulcraft.com/docs...${NC}"
|
echo -e "${BLUE}Docs step: this engine documents itself in its own repo (site push retired 2026-08-31)${NC}"
|
||||||
if node scripts/push-docs.js; then
|
|
||||||
echo -e "${GREEN}✅ Docs push step done${NC}\n"
|
|
||||||
else
|
|
||||||
echo -e "${RED}❌ Docs push FAILED — soulcraft.com/docs trails npm until re-run or interim sync${NC}\n"
|
|
||||||
fi
|
|
||||||
|
|
||||||
echo -e "${GREEN}━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━${NC}"
|
echo -e "${GREEN}━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━${NC}"
|
||||||
echo -e "${GREEN}🎉 Release ${NEW_VERSION} complete!${NC}"
|
echo -e "${GREEN}🎉 Release ${NEW_VERSION} complete!${NC}"
|
||||||
echo -e "${GREEN}━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━${NC}"
|
echo -e "${GREEN}━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━${NC}"
|
||||||
echo ""
|
echo ""
|
||||||
echo -e "📦 npm: ${BLUE}https://www.npmjs.com/package/@soulcraft/brainy/v/${NEW_VERSION}${NC}"
|
echo -e "🏠 The Source: ${BLUE}https://source.soulcraft.com/soulcraftlabs/open-brainy/releases/tag/v${NEW_VERSION}${NC}"
|
||||||
echo -e "🏠 The Source: ${BLUE}https://source.soulcraft.com/soulcraft/brainy/releases/tag/v${NEW_VERSION}${NC}"
|
|
||||||
|
|
|
||||||
539
scripts/wall-entry.mjs
Normal file
539
scripts/wall-entry.mjs
Normal file
|
|
@ -0,0 +1,539 @@
|
||||||
|
#!/usr/bin/env node
|
||||||
|
/**
|
||||||
|
* @module scripts/wall-entry
|
||||||
|
* @description The releases-wall entry, made mechanical. The fleet's HQ page
|
||||||
|
* reads one public JSON per product from the ONE releases repo on The Source
|
||||||
|
* (soulcraftlabs/releases, files <product>.json at its root — shape
|
||||||
|
* {product, entries:[{version, date, headline, items, url, thumb?}]}), at
|
||||||
|
* https://source.soulcraft.com/soulcraftlabs/releases/raw/branch/main/<product>.json.
|
||||||
|
* Those entries were hand-written after every release, then briefly written
|
||||||
|
* into this repo's own releases/<product>.json; this script is the one door
|
||||||
|
* that composes an entry and lands it in the shared repo, so it is never
|
||||||
|
* hand-written and never forked across repos again.
|
||||||
|
*
|
||||||
|
* Two modes:
|
||||||
|
*
|
||||||
|
* 1. Generate + publish (default):
|
||||||
|
* node wall-entry.mjs --product <p> --version <v> --date <YYYY-MM-DD> \
|
||||||
|
* --from-changelog <CHANGELOG.md>
|
||||||
|
* Derives an entry from the CHANGELOG.md entry for <v> (headline = the
|
||||||
|
* entry's first bullet, items = every bullet, trimmed of its trailing
|
||||||
|
* commit hash), then:
|
||||||
|
* - clones (or, if a cached clone already exists, fetches and resets)
|
||||||
|
* the releases repo into a local cache directory,
|
||||||
|
* - prepends the entry to <cache>/<p>.json, newest first — replacing
|
||||||
|
* any existing entry for the same version so a re-run is idempotent,
|
||||||
|
* - validates the file's shape before and after,
|
||||||
|
* - commits the change as "chore(wall): <p> <v>" and pushes main.
|
||||||
|
* A failure at any step (clone, validation, commit, push, a
|
||||||
|
* non-fast-forward remote) exits non-zero naming the cure. Nothing is
|
||||||
|
* ever skipped — the wall either lands correctly or the release fails.
|
||||||
|
*
|
||||||
|
* 2. Dry run:
|
||||||
|
* node wall-entry.mjs --dry-run --product <p> --version <v> \
|
||||||
|
* --date <YYYY-MM-DD> --from-changelog <CHANGELOG.md>
|
||||||
|
* Derives the entry exactly as above and prints it, along with the file
|
||||||
|
* it would be written to, but touches no clone and no remote — usable
|
||||||
|
* from a fresh checkout with no cache and no network.
|
||||||
|
*
|
||||||
|
* 3. Validate only (--check):
|
||||||
|
* node wall-entry.mjs --check --file <path/to/product.json>
|
||||||
|
* Validates an arbitrary wall file's exact key set (top-level and
|
||||||
|
* per-entry), field types, and strict-descending semver ordering with
|
||||||
|
* no duplicates. Read-only; never writes. Exit 0 = clean, exit 1 =
|
||||||
|
* named violations printed to stderr.
|
||||||
|
*
|
||||||
|
* The remote and the local cache directory are each overridable
|
||||||
|
* (--remote / --cache-dir, or WALL_ENTRY_RELEASES_REMOTE /
|
||||||
|
* WALL_ENTRY_RELEASES_CACHE_DIR) so tests can point at a throwaway local
|
||||||
|
* bare repo and a throwaway cache directory — never the real remote or the
|
||||||
|
* real developer cache.
|
||||||
|
*
|
||||||
|
* No dependencies beyond the system `git` binary — CHANGELOG parsing,
|
||||||
|
* semver comparison, and JSON shape checking are all hand-rolled below.
|
||||||
|
*/
|
||||||
|
|
||||||
|
import { readFileSync, writeFileSync, existsSync, mkdirSync } from 'node:fs'
|
||||||
|
import { execFileSync } from 'node:child_process'
|
||||||
|
import { homedir } from 'node:os'
|
||||||
|
import { dirname, join } from 'node:path'
|
||||||
|
|
||||||
|
const DEFAULT_REMOTE = 'git@source.soulcraft.com:soulcraftlabs/releases.git'
|
||||||
|
|
||||||
|
/** @returns {string} */
|
||||||
|
function defaultCacheDir() {
|
||||||
|
const base = process.env.XDG_CACHE_HOME || join(homedir(), '.cache')
|
||||||
|
return join(base, 'soulcraft-releases')
|
||||||
|
}
|
||||||
|
|
||||||
|
// Required on every entry; "thumb" is optional (may be absent, or present as
|
||||||
|
// string | null) — matching the HQ contract's {..., thumb?}.
|
||||||
|
const ENTRY_REQUIRED_KEYS = ['version', 'date', 'headline', 'items', 'url']
|
||||||
|
const ENTRY_OPTIONAL_KEYS = ['thumb']
|
||||||
|
const ENTRY_ALLOWED_KEYS = [...ENTRY_REQUIRED_KEYS, ...ENTRY_OPTIONAL_KEYS]
|
||||||
|
const FILE_KEYS = ['product', 'entries']
|
||||||
|
|
||||||
|
// The public permalink pattern, by product. Every entry MUST carry an https
|
||||||
|
// permalink: HQ's parser rejects a wall whose entries carry url: null (the
|
||||||
|
// whole feed became unreadable on 2026-09-02). A product whose forge repo is
|
||||||
|
// private links its PUBLIC package page on The Source instead of a release
|
||||||
|
// page that would 404 for HQ's readers.
|
||||||
|
const RELEASE_URL_PATTERNS = {
|
||||||
|
'open-brainy': (version) => `https://source.soulcraft.com/soulcraftlabs/open-brainy/releases/tag/v${version}`,
|
||||||
|
'brainy': (version) => `https://source.soulcraft.com/soulcraft/-/packages/npm/@soulcraft%2Fbrainy/${version}`,
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Parse argv into a flag map. `--flag value` sets a string; `--flag` alone
|
||||||
|
* (end of argv, or followed by another `--flag`) sets boolean true.
|
||||||
|
* @param {string[]} argv
|
||||||
|
* @returns {Record<string, string | true>}
|
||||||
|
*/
|
||||||
|
function parseArgs(argv) {
|
||||||
|
/** @type {Record<string, string | true>} */
|
||||||
|
const args = {}
|
||||||
|
for (let i = 0; i < argv.length; i++) {
|
||||||
|
const a = argv[i]
|
||||||
|
if (!a.startsWith('--')) continue
|
||||||
|
const key = a.slice(2)
|
||||||
|
const next = argv[i + 1]
|
||||||
|
if (next === undefined || next.startsWith('--')) {
|
||||||
|
args[key] = true
|
||||||
|
} else {
|
||||||
|
args[key] = next
|
||||||
|
i++
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return args
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Print a loud, named error and exit 1. Every refusal in this script goes
|
||||||
|
* through here so the failure mode is always the same shape: "wall-entry: <what>".
|
||||||
|
* @param {string} message
|
||||||
|
* @returns {never}
|
||||||
|
*/
|
||||||
|
function fail(message) {
|
||||||
|
console.error(`wall-entry: ${message}`)
|
||||||
|
process.exit(1)
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* @param {string} version
|
||||||
|
* @returns {{major: number, minor: number, patch: number, pre: string | null} | null}
|
||||||
|
*/
|
||||||
|
function parseSemver(version) {
|
||||||
|
const m = /^(\d+)\.(\d+)\.(\d+)(?:-([0-9A-Za-z.-]+))?$/.exec(version)
|
||||||
|
if (!m) return null
|
||||||
|
return { major: Number(m[1]), minor: Number(m[2]), patch: Number(m[3]), pre: m[4] ?? null }
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* @param {string} a
|
||||||
|
* @param {string} b
|
||||||
|
* @returns {number} positive if a > b, negative if a < b, 0 if equal.
|
||||||
|
*/
|
||||||
|
function compareSemver(a, b) {
|
||||||
|
const pa = parseSemver(a)
|
||||||
|
const pb = parseSemver(b)
|
||||||
|
if (!pa || !pb) throw new Error(`cannot compare non-semver versions "${a}" vs "${b}"`)
|
||||||
|
if (pa.major !== pb.major) return pa.major - pb.major
|
||||||
|
if (pa.minor !== pb.minor) return pa.minor - pb.minor
|
||||||
|
if (pa.patch !== pb.patch) return pa.patch - pb.patch
|
||||||
|
if (pa.pre === pb.pre) return 0
|
||||||
|
if (pa.pre === null) return 1 // a release outranks any prerelease of the same core version
|
||||||
|
if (pb.pre === null) return -1
|
||||||
|
return pa.pre < pb.pre ? -1 : pa.pre > pb.pre ? 1 : 0
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Validate a wall file's full shape: top-level keys ("product", "entries" —
|
||||||
|
* no more, no less), per-entry keys and field types ("thumb" optional), and
|
||||||
|
* strict-descending semver ordering with no duplicates. Collects every
|
||||||
|
* violation instead of failing on the first, so a caller reports the whole
|
||||||
|
* picture in one pass.
|
||||||
|
* @param {unknown} data
|
||||||
|
* @returns {string[]} Violation messages; empty means the file is clean.
|
||||||
|
*/
|
||||||
|
function validateShape(data) {
|
||||||
|
/** @type {string[]} */
|
||||||
|
const errors = []
|
||||||
|
|
||||||
|
if (typeof data !== 'object' || data === null || Array.isArray(data)) {
|
||||||
|
return ['top level: expected a JSON object']
|
||||||
|
}
|
||||||
|
const obj = /** @type {Record<string, unknown>} */ (data)
|
||||||
|
|
||||||
|
const topKeys = Object.keys(obj)
|
||||||
|
const missingTop = FILE_KEYS.filter((k) => !(k in obj))
|
||||||
|
const extraTop = topKeys.filter((k) => !FILE_KEYS.includes(k))
|
||||||
|
if (missingTop.length) errors.push(`top level: missing key(s) ${missingTop.join(', ')}`)
|
||||||
|
if (extraTop.length) errors.push(`top level: unexpected key(s) ${extraTop.join(', ')}`)
|
||||||
|
|
||||||
|
if (typeof obj.product !== 'string' || obj.product.trim() === '') {
|
||||||
|
errors.push('top level: "product" must be a non-empty string')
|
||||||
|
}
|
||||||
|
if (!Array.isArray(obj.entries)) {
|
||||||
|
errors.push('top level: "entries" must be an array')
|
||||||
|
return errors // nothing further to check without an array
|
||||||
|
}
|
||||||
|
|
||||||
|
const entries = /** @type {unknown[]} */ (obj.entries)
|
||||||
|
entries.forEach((rawEntry, i) => {
|
||||||
|
const label = `entries[${i}]`
|
||||||
|
if (typeof rawEntry !== 'object' || rawEntry === null || Array.isArray(rawEntry)) {
|
||||||
|
errors.push(`${label}: expected an object`)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
const entry = /** @type {Record<string, unknown>} */ (rawEntry)
|
||||||
|
const keys = Object.keys(entry)
|
||||||
|
const missing = ENTRY_REQUIRED_KEYS.filter((k) => !(k in entry))
|
||||||
|
const extra = keys.filter((k) => !ENTRY_ALLOWED_KEYS.includes(k))
|
||||||
|
if (missing.length) errors.push(`${label}: missing key(s) ${missing.join(', ')}`)
|
||||||
|
if (extra.length) errors.push(`${label}: unexpected key(s) ${extra.join(', ')}`)
|
||||||
|
|
||||||
|
if (typeof entry.version !== 'string' || !parseSemver(entry.version)) {
|
||||||
|
errors.push(`${label}: "version" must be a semver string (got ${JSON.stringify(entry.version)})`)
|
||||||
|
}
|
||||||
|
if (typeof entry.date !== 'string' || !/^\d{4}-\d{2}-\d{2}$/.test(entry.date) || Number.isNaN(Date.parse(entry.date))) {
|
||||||
|
errors.push(`${label}: "date" must be a YYYY-MM-DD string (got ${JSON.stringify(entry.date)})`)
|
||||||
|
}
|
||||||
|
if (typeof entry.headline !== 'string' || entry.headline.trim() === '') {
|
||||||
|
errors.push(`${label}: "headline" must be a non-empty string`)
|
||||||
|
}
|
||||||
|
if (!Array.isArray(entry.items) || entry.items.length === 0 || entry.items.some((it) => typeof it !== 'string' || it.trim() === '')) {
|
||||||
|
errors.push(`${label}: "items" must be a non-empty array of non-empty strings`)
|
||||||
|
}
|
||||||
|
if (typeof entry.url !== 'string' || !/^https:\/\/\S+$/.test(entry.url)) {
|
||||||
|
errors.push(`${label}: "url" must be an https permalink — never null; HQ's parser rejects the whole feed`)
|
||||||
|
}
|
||||||
|
if ('thumb' in entry && !(entry.thumb === null || typeof entry.thumb === 'string')) {
|
||||||
|
errors.push(`${label}: "thumb" must be a string or null when present`)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
|
||||||
|
// Ordering: newest first, strictly descending, no duplicate versions —
|
||||||
|
// checked only over entries whose version parsed (a bad version is
|
||||||
|
// already reported above; comparing it too would just be noise).
|
||||||
|
const versioned = entries
|
||||||
|
.map((e, i) => ({ i, version: /** @type {any} */ (e)?.version }))
|
||||||
|
.filter((e) => typeof e.version === 'string' && parseSemver(e.version))
|
||||||
|
for (let i = 0; i < versioned.length - 1; i++) {
|
||||||
|
const a = versioned[i]
|
||||||
|
const b = versioned[i + 1]
|
||||||
|
const cmp = compareSemver(a.version, b.version)
|
||||||
|
if (cmp === 0) {
|
||||||
|
errors.push(`entries[${a.i}] and entries[${b.i}]: duplicate version ${a.version}`)
|
||||||
|
} else if (cmp < 0) {
|
||||||
|
errors.push(`entries[${a.i}] (${a.version}) sits above entries[${b.i}] (${b.version}) — not newest-first`)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
return errors
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Extract one version's entry body from a standard-version-style CHANGELOG.md
|
||||||
|
* (headings `### [version](url) (date)`, followed by `- bullet (hash)` lines
|
||||||
|
* until the next heading or EOF).
|
||||||
|
* @param {string} changelog
|
||||||
|
* @param {string} version
|
||||||
|
* @returns {string[]} Bullet lines, trimmed of their leading "- " and
|
||||||
|
* trailing " (hash)".
|
||||||
|
*/
|
||||||
|
function extractChangelogBullets(changelog, version) {
|
||||||
|
const lines = changelog.split('\n')
|
||||||
|
const headingRe = /^### \[([^\]]+)\]\(.*\)\s*\(\d{4}-\d{2}-\d{2}\)\s*$/
|
||||||
|
let start = -1
|
||||||
|
for (let i = 0; i < lines.length; i++) {
|
||||||
|
const m = headingRe.exec(lines[i])
|
||||||
|
if (m && m[1] === version) {
|
||||||
|
start = i + 1
|
||||||
|
break
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if (start === -1) {
|
||||||
|
fail(
|
||||||
|
`version ${version} has no CHANGELOG entry yet — run this after the CHANGELOG step composes "### [${version}]", not before`,
|
||||||
|
)
|
||||||
|
}
|
||||||
|
/** @type {string[]} */
|
||||||
|
const bullets = []
|
||||||
|
for (let i = start; i < lines.length; i++) {
|
||||||
|
if (headingRe.test(lines[i])) break // next entry starts
|
||||||
|
const bulletMatch = /^- (.+?)(?:\s\(([0-9a-f]{6,40})\))?$/.exec(lines[i].trim())
|
||||||
|
if (lines[i].trim().startsWith('- ') && bulletMatch) {
|
||||||
|
const text = bulletMatch[1].trim()
|
||||||
|
if (text) bullets.push(text)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if (bullets.length === 0) {
|
||||||
|
fail(`version ${version}'s CHANGELOG entry has no bullets to derive a headline/items from`)
|
||||||
|
}
|
||||||
|
return bullets
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Derive a wall entry from a CHANGELOG.md.
|
||||||
|
* @param {{product: string, version: string, date: string, changelogPath: string, url?: string, thumb?: string | null}} opts
|
||||||
|
* @returns {{version: string, date: string, headline: string, items: string[], url: string, thumb: string | null}}
|
||||||
|
*/
|
||||||
|
function deriveEntry({ product, version, date, changelogPath, url, thumb }) {
|
||||||
|
if (!parseSemver(version)) fail(`--version "${version}" is not a semver string`)
|
||||||
|
if (!/^\d{4}-\d{2}-\d{2}$/.test(date) || Number.isNaN(Date.parse(date))) {
|
||||||
|
fail(`--date "${date}" is not a YYYY-MM-DD date`)
|
||||||
|
}
|
||||||
|
if (!existsSync(changelogPath)) fail(`--from-changelog "${changelogPath}" does not exist`)
|
||||||
|
|
||||||
|
const changelog = readFileSync(changelogPath, 'utf8')
|
||||||
|
const items = extractChangelogBullets(changelog, version)
|
||||||
|
const headline = items[0]
|
||||||
|
|
||||||
|
const pattern = RELEASE_URL_PATTERNS[product]
|
||||||
|
if (url === undefined && pattern === undefined) {
|
||||||
|
throw new Error(`wall-entry: no permalink pattern for product "${product}" — add one to RELEASE_URL_PATTERNS or pass --url; entries never carry url: null`)
|
||||||
|
}
|
||||||
|
const resolvedUrl = url !== undefined ? url : pattern(version)
|
||||||
|
const resolvedThumb = thumb !== undefined ? thumb : null
|
||||||
|
|
||||||
|
return { version, date, headline, items, url: resolvedUrl, thumb: resolvedThumb }
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Load and shape-validate a wall file.
|
||||||
|
* @param {string} filePath
|
||||||
|
* @returns {Record<string, any>}
|
||||||
|
*/
|
||||||
|
function loadWallFile(filePath) {
|
||||||
|
if (!existsSync(filePath)) fail(`"${filePath}" does not exist`)
|
||||||
|
/** @type {unknown} */
|
||||||
|
let data
|
||||||
|
try {
|
||||||
|
data = JSON.parse(readFileSync(filePath, 'utf8'))
|
||||||
|
} catch (err) {
|
||||||
|
fail(`"${filePath}" is not valid JSON: ${/** @type {Error} */ (err).message}`)
|
||||||
|
}
|
||||||
|
const errors = validateShape(data)
|
||||||
|
if (errors.length) {
|
||||||
|
fail(`"${filePath}" fails shape validation —\n ${errors.join('\n ')}`)
|
||||||
|
}
|
||||||
|
return /** @type {Record<string, any>} */ (data)
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Run a git command, throwing an Error whose message is git's own stderr
|
||||||
|
* (trimmed) on failure — every caller wraps this to name the cure.
|
||||||
|
* @param {string[]} args
|
||||||
|
* @param {string} cwd
|
||||||
|
* @returns {string} stdout, trimmed.
|
||||||
|
*/
|
||||||
|
function git(args, cwd) {
|
||||||
|
try {
|
||||||
|
return execFileSync('git', args, { cwd, encoding: 'utf8', stdio: ['ignore', 'pipe', 'pipe'] }).trim()
|
||||||
|
} catch (err) {
|
||||||
|
const stderr = /** @type {any} */ (err).stderr
|
||||||
|
const message = (typeof stderr === 'string' && stderr.trim()) || /** @type {Error} */ (err).message
|
||||||
|
throw new Error(message)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Resolve the git identity for the wall commit from the repository the rail
|
||||||
|
* is actually running in — the developer's own checkout (`process.cwd()`;
|
||||||
|
* `release.sh` invokes this script from the repo root with no `cd`), via
|
||||||
|
* git's normal config precedence (repo-local, then global, then system).
|
||||||
|
* Never guessed and never left to git's own "who are you?" prompt: a host
|
||||||
|
* with no configured identity anywhere (a bare CI box, say) must refuse
|
||||||
|
* loudly rather than have git manufacture a placeholder identity or hang.
|
||||||
|
* @returns {{name: string, email: string}}
|
||||||
|
*/
|
||||||
|
function resolveWallCommitIdentity() {
|
||||||
|
const repo = process.cwd()
|
||||||
|
let name = ''
|
||||||
|
let email = ''
|
||||||
|
try {
|
||||||
|
name = git(['config', 'user.name'], repo)
|
||||||
|
} catch {
|
||||||
|
name = ''
|
||||||
|
}
|
||||||
|
try {
|
||||||
|
email = git(['config', 'user.email'], repo)
|
||||||
|
} catch {
|
||||||
|
email = ''
|
||||||
|
}
|
||||||
|
if (!name || !email) {
|
||||||
|
fail('no git identity for the wall commit — set user.name/user.email')
|
||||||
|
}
|
||||||
|
return { name, email }
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Ensure a clean, up-to-date local clone of the releases repo at
|
||||||
|
* `cacheDir`, checked out on `main` — cloning fresh if `cacheDir` has no
|
||||||
|
* `.git`, otherwise fetching and hard-resetting onto `origin/main` (so a
|
||||||
|
* stray local commit or edit left by a previous failed run can never leak
|
||||||
|
* into the next one).
|
||||||
|
* @param {string} remote
|
||||||
|
* @param {string} cacheDir
|
||||||
|
*/
|
||||||
|
function ensureReleasesClone(remote, cacheDir) {
|
||||||
|
if (existsSync(join(cacheDir, '.git'))) {
|
||||||
|
try {
|
||||||
|
git(['remote', 'set-url', 'origin', remote], cacheDir)
|
||||||
|
git(['fetch', '--prune', 'origin'], cacheDir)
|
||||||
|
git(['checkout', 'main'], cacheDir)
|
||||||
|
git(['reset', '--hard', 'origin/main'], cacheDir)
|
||||||
|
git(['clean', '-fd'], cacheDir)
|
||||||
|
} catch (err) {
|
||||||
|
fail(
|
||||||
|
`cannot refresh the cached releases checkout at "${cacheDir}" from "${remote}" — ${/** @type {Error} */ (err).message}\n` +
|
||||||
|
` cure: delete "${cacheDir}" and re-run so it re-clones from scratch, or confirm SSH access with "ssh -T git@source.soulcraft.com"`,
|
||||||
|
)
|
||||||
|
}
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
mkdirSync(dirname(cacheDir), { recursive: true })
|
||||||
|
try {
|
||||||
|
git(['clone', remote, cacheDir], dirname(cacheDir))
|
||||||
|
} catch (err) {
|
||||||
|
fail(
|
||||||
|
`cannot clone "${remote}" — ${/** @type {Error} */ (err).message}\n` +
|
||||||
|
` cure: confirm SSH access with "ssh -T git@source.soulcraft.com" and that the soulcraftlabs/releases repo exists yet`,
|
||||||
|
)
|
||||||
|
}
|
||||||
|
try {
|
||||||
|
git(['checkout', 'main'], cacheDir)
|
||||||
|
} catch (err) {
|
||||||
|
fail(
|
||||||
|
`cloned "${remote}" into "${cacheDir}" but could not check out "main" — ${/** @type {Error} */ (err).message}\n` +
|
||||||
|
` cure: confirm the releases repo's default branch is named "main"`,
|
||||||
|
)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Prepend `entry` to the wall at `<cacheDir>/<product>.json`, replacing any
|
||||||
|
* existing entry for the same version (idempotent re-runs), validating
|
||||||
|
* before and after, committing, and pushing — or refusing loudly, naming
|
||||||
|
* the cure, at whichever step fails.
|
||||||
|
* @param {{version: string, date: string, headline: string, items: string[], url: string, thumb: string | null}} entry
|
||||||
|
* @param {string} product
|
||||||
|
* @param {string} remote
|
||||||
|
* @param {string} cacheDir
|
||||||
|
*/
|
||||||
|
function publishEntry(entry, product, remote, cacheDir) {
|
||||||
|
ensureReleasesClone(remote, cacheDir)
|
||||||
|
|
||||||
|
const filePath = join(cacheDir, `${product}.json`)
|
||||||
|
if (!existsSync(filePath)) {
|
||||||
|
fail(
|
||||||
|
`"${filePath}" does not exist in the releases repo — cure: seed "${product}.json" at the repo root first (it must exist before any release rail can prepend to it)`,
|
||||||
|
)
|
||||||
|
}
|
||||||
|
const wall = loadWallFile(filePath)
|
||||||
|
|
||||||
|
if (wall.product !== product) {
|
||||||
|
fail(`"${filePath}" has product "${wall.product}", but --product "${product}" was given — refusing a cross-product write`)
|
||||||
|
}
|
||||||
|
|
||||||
|
const replacing = wall.entries.some((e) => e.version === entry.version)
|
||||||
|
wall.entries = [entry, ...wall.entries.filter((e) => e.version !== entry.version)]
|
||||||
|
|
||||||
|
const postErrors = validateShape(wall)
|
||||||
|
if (postErrors.length) {
|
||||||
|
fail(`the entry for ${entry.version} would leave "${filePath}" invalid —\n ${postErrors.join('\n ')}`)
|
||||||
|
}
|
||||||
|
|
||||||
|
writeFileSync(filePath, JSON.stringify(wall, null, 2) + '\n', 'utf8')
|
||||||
|
|
||||||
|
const status = git(['status', '--porcelain', '--', `${product}.json`], cacheDir)
|
||||||
|
if (status === '') {
|
||||||
|
console.log(`wall-entry: "${product}.json" already carries an identical entry for ${entry.version} — nothing to commit or push`)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
const identity = resolveWallCommitIdentity()
|
||||||
|
|
||||||
|
try {
|
||||||
|
git(['add', `${product}.json`], cacheDir)
|
||||||
|
git(
|
||||||
|
['-c', `user.name=${identity.name}`, '-c', `user.email=${identity.email}`, 'commit', '-m', `chore(wall): ${product} ${entry.version}`],
|
||||||
|
cacheDir,
|
||||||
|
)
|
||||||
|
} catch (err) {
|
||||||
|
fail(`cannot commit the wall entry in "${cacheDir}" — ${/** @type {Error} */ (err).message}\n cure: inspect "${cacheDir}" by hand and re-run once its git state is clean`)
|
||||||
|
}
|
||||||
|
|
||||||
|
try {
|
||||||
|
git(['push', 'origin', 'main'], cacheDir)
|
||||||
|
} catch (err) {
|
||||||
|
fail(
|
||||||
|
`push to "${remote}" failed (likely a non-fast-forward — another release landed on main first) — ${/** @type {Error} */ (err).message}\n` +
|
||||||
|
` cure: re-run this release step; it re-fetches and resets onto the latest origin/main before retrying`,
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
const sha = git(['rev-parse', 'HEAD'], cacheDir)
|
||||||
|
console.log(
|
||||||
|
`wall-entry: ${replacing ? 'replaced' : 'wrote'} v${entry.version} in "${product}.json" (${wall.entries.length} entries, newest first) — pushed ${sha} to ${remote} main`,
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
function main() {
|
||||||
|
const args = parseArgs(process.argv.slice(2))
|
||||||
|
|
||||||
|
if (args.check) {
|
||||||
|
const filePath = /** @type {string | undefined} */ (args.file)
|
||||||
|
if (!filePath) fail('--check needs --file <path>')
|
||||||
|
const wall = loadWallFile(/** @type {string} */ (filePath))
|
||||||
|
console.log(`wall-entry --check: "${filePath}" OK — product "${wall.product}", ${wall.entries.length} entries, newest-first, no duplicates`)
|
||||||
|
process.exit(0)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Generate mode (default, also covers --dry-run): --product, --version,
|
||||||
|
// --date, --from-changelog required.
|
||||||
|
const product = /** @type {string | undefined} */ (args.product)
|
||||||
|
const version = /** @type {string | undefined} */ (args.version)
|
||||||
|
const date = /** @type {string | undefined} */ (args.date)
|
||||||
|
const fromChangelog = /** @type {string | undefined} */ (args['from-changelog'])
|
||||||
|
|
||||||
|
const missing = []
|
||||||
|
if (!product) missing.push('--product')
|
||||||
|
if (!version) missing.push('--version')
|
||||||
|
if (!date) missing.push('--date')
|
||||||
|
if (!fromChangelog) missing.push('--from-changelog')
|
||||||
|
if (missing.length) {
|
||||||
|
fail(
|
||||||
|
`missing required flag(s): ${missing.join(', ')}\n` +
|
||||||
|
'Usage:\n' +
|
||||||
|
' wall-entry.mjs --product <p> --version <v> --date <YYYY-MM-DD> --from-changelog <CHANGELOG.md> [--dry-run]\n' +
|
||||||
|
' wall-entry.mjs --check --file <path/to/product.json>',
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
const urlArg = args.url === true ? undefined : /** @type {string | undefined} */ (args.url)
|
||||||
|
const thumbArg = args.thumb === true ? undefined : /** @type {string | undefined} */ (args.thumb)
|
||||||
|
|
||||||
|
const entry = deriveEntry({
|
||||||
|
product: /** @type {string} */ (product),
|
||||||
|
version: /** @type {string} */ (version),
|
||||||
|
date: /** @type {string} */ (date),
|
||||||
|
changelogPath: /** @type {string} */ (fromChangelog),
|
||||||
|
url: urlArg,
|
||||||
|
thumb: thumbArg,
|
||||||
|
})
|
||||||
|
|
||||||
|
const remote = /** @type {string} */ (args.remote ?? process.env.WALL_ENTRY_RELEASES_REMOTE ?? DEFAULT_REMOTE)
|
||||||
|
const cacheDir = /** @type {string} */ (args['cache-dir'] ?? process.env.WALL_ENTRY_RELEASES_CACHE_DIR ?? defaultCacheDir())
|
||||||
|
|
||||||
|
if (args['dry-run']) {
|
||||||
|
console.log(`wall-entry --dry-run: would write to "${join(cacheDir, `${product}.json`)}" in ${remote} (main), pushed as "chore(wall): ${product} ${version}"`)
|
||||||
|
console.log(JSON.stringify(entry, null, 2))
|
||||||
|
process.exit(0)
|
||||||
|
}
|
||||||
|
|
||||||
|
publishEntry(entry, /** @type {string} */ (product), remote, cacheDir)
|
||||||
|
}
|
||||||
|
|
||||||
|
main()
|
||||||
5076
src/brainy.ts
5076
src/brainy.ts
File diff suppressed because it is too large
Load diff
|
|
@ -801,6 +801,17 @@ export interface DerivedFamilyDeclaration {
|
||||||
export interface CanonicalCounts {
|
export interface CanonicalCounts {
|
||||||
nouns: { counted: number; all: number }
|
nouns: { counted: number; all: number }
|
||||||
verbs: { counted: number; all: number }
|
verbs: { counted: number; all: number }
|
||||||
|
/**
|
||||||
|
* The count of canonical nouns holding a REAL (non-empty) vector — the
|
||||||
|
* coverage denominator a vector index's node-count ledger is measured
|
||||||
|
* against (`nodeCount === vectors.all` is the whole-store coverage
|
||||||
|
* verdict for the vector leg, the vector-side mirror of `nouns.all` for
|
||||||
|
* metadata/graph). A deferred-embed noun (`add({ deferEmbedding: true })`)
|
||||||
|
* counts only once its vector actually LANDS — its canonical record exists
|
||||||
|
* (counted in `nouns.all`) with an empty vector until then, so it is
|
||||||
|
* deliberately NOT counted here in the interim.
|
||||||
|
*/
|
||||||
|
vectors: { all: number }
|
||||||
/** An unprovable delete has left the `all` scalars unverified since the last recount. */
|
/** An unprovable delete has left the `all` scalars unverified since the last recount. */
|
||||||
suspect: boolean
|
suspect: boolean
|
||||||
}
|
}
|
||||||
|
|
@ -819,14 +830,68 @@ export interface StorageAdapter {
|
||||||
* Save noun metadata separately
|
* Save noun metadata separately
|
||||||
* @param id Noun ID
|
* @param id Noun ID
|
||||||
* @param metadata Noun metadata
|
* @param metadata Noun metadata
|
||||||
|
* @param hasVector - OPTIONAL vectored-noun ledger hint: `true` when this
|
||||||
|
* write is a FRESH insert (`isNew`) whose vector is a real, non-empty
|
||||||
|
* array — the caller already knows this for free (the insert's own
|
||||||
|
* `vector` local), so the increment rides the SAME isNew gate that
|
||||||
|
* already protects `totalNounCountAll` from double-counting on HNSW
|
||||||
|
* neighbor-link re-saves (`saveNoun_internal` re-runs on every link
|
||||||
|
* change; this metadata seam does not). Absent/`false` ⇒ no ledger
|
||||||
|
* action. A deferred-embed insert passes `false` (its vector lands
|
||||||
|
* later — see {@link StorageAdapter.noteVectorLanded}).
|
||||||
*/
|
*/
|
||||||
saveNounMetadata(id: string, metadata: NounMetadata): Promise<void>
|
saveNounMetadata(id: string, metadata: NounMetadata, hasVector?: boolean): Promise<void>
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Delete noun metadata
|
* Delete noun metadata
|
||||||
* @param id Noun ID
|
* @param id Noun ID
|
||||||
|
* @param priorRecord - OPTIONAL already-known metadata (the caller's
|
||||||
|
* pre-delete read) — see {@link StorageAdapter.deleteNoun}.
|
||||||
|
* @param hadVector - OPTIONAL vectored-noun ledger hint: `true`/`false`
|
||||||
|
* when the caller already knows (read as a side effect of ITS OWN delete
|
||||||
|
* flow — e.g. `remove()`'s pre-read for the vector-index removal — never
|
||||||
|
* a read added FOR this ledger), `undefined` when genuinely unknown. A
|
||||||
|
* known `true` decrements the vectored-noun ledger; a known `false` is a
|
||||||
|
* no-op (it was never counted); `undefined` marks the ledger SUSPECT
|
||||||
|
* rather than guessing — the delete path must never add a canonical read
|
||||||
|
* to answer this question.
|
||||||
*/
|
*/
|
||||||
deleteNounMetadata(id: string): Promise<void>
|
deleteNounMetadata(id: string, priorRecord?: NounMetadata | null, hadVector?: boolean): Promise<void>
|
||||||
|
|
||||||
|
/**
|
||||||
|
* OPTIONAL narrow ledger hook: record that a canonical noun's vector just
|
||||||
|
* LANDED for the first time. Exists ONLY for the deferred-embedding
|
||||||
|
* lifecycle — the landing commit (`system:embed-landing`) carries a vector
|
||||||
|
* write with no accompanying metadata operation, so the normal
|
||||||
|
* `saveNounMetadata(..., hasVector)` seam never fires for it. Callers MUST
|
||||||
|
* call this only when the noun held NO real vector before this write (the
|
||||||
|
* deferred-embed worker already holds that fact for free, from its own
|
||||||
|
* pre-embed read — never an added read). A backend without vectored-noun
|
||||||
|
* tracking is a no-op via this method's absence (feature-detected).
|
||||||
|
* @param id - The noun whose vector just landed.
|
||||||
|
*/
|
||||||
|
noteVectorLanded?(id: string): Promise<void>
|
||||||
|
|
||||||
|
/**
|
||||||
|
* OPTIONAL narrow ledger hook, the mirror of {@link noteVectorLanded}:
|
||||||
|
* record that a canonical noun's vector was just REMOVED — rewritten from
|
||||||
|
* a real (non-empty) vector to the "unvectored" empty-array shape. Exists
|
||||||
|
* for the ONE sanctioned reverse migration this engine supports: the VFS
|
||||||
|
* root's zero-norm fix (see `VirtualFileSystem.doInitializeRoot()` and
|
||||||
|
* `Brainy.unvectorNounForRootMigration()`), which rewrites a pre-fix
|
||||||
|
* store's all-zero placeholder root vector to `[]` and must decrement
|
||||||
|
* `vectors.all` through this hook so the coverage ledger never drifts.
|
||||||
|
* NOT a general-purpose "I removed a vector" callback — ordinary
|
||||||
|
* application data has no sanctioned path from vectored back to
|
||||||
|
* unvectored (`update()` refuses an empty vector as a dimension
|
||||||
|
* mismatch by design). Callers MUST call this only when the noun held a
|
||||||
|
* REAL vector immediately before this write (the caller already holds
|
||||||
|
* that fact for free, from its own pre-write read — never an added read).
|
||||||
|
* A backend without vectored-noun tracking is a no-op via this method's
|
||||||
|
* absence (feature-detected).
|
||||||
|
* @param id - The noun whose vector was just removed.
|
||||||
|
*/
|
||||||
|
noteVectorUnlanded?(id: string): Promise<void>
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Get noun with metadata combined
|
* Get noun with metadata combined
|
||||||
|
|
@ -875,8 +940,11 @@ export interface StorageAdapter {
|
||||||
* REQUIRE re-reading the record being removed: when the internal read
|
* REQUIRE re-reading the record being removed: when the internal read
|
||||||
* returns `null` (replace race, or a ghost left by an earlier version) the
|
* returns `null` (replace race, or a ghost left by an earlier version) the
|
||||||
* decrement falls back to this record instead of being silently skipped.
|
* decrement falls back to this record instead of being silently skipped.
|
||||||
|
* @param hadVector OPTIONAL vectored-noun ledger hint — see
|
||||||
|
* {@link StorageAdapter.deleteNounMetadata}'s `hadVector` param, which
|
||||||
|
* this forwards to unchanged.
|
||||||
*/
|
*/
|
||||||
deleteNoun(id: string, priorMetadata?: NounMetadata | null): Promise<void>
|
deleteNoun(id: string, priorMetadata?: NounMetadata | null, hadVector?: boolean): Promise<void>
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Save verb - Pure HNSW verb with core fields only
|
* Save verb - Pure HNSW verb with core fields only
|
||||||
|
|
|
||||||
64
src/db/db.ts
64
src/db/db.ts
|
|
@ -61,7 +61,7 @@ import { exportGraph } from './portableGraph.js'
|
||||||
import type { ExportSelector, ExportOptions, PortableGraph } from './portableGraph.js'
|
import type { ExportSelector, ExportOptions, PortableGraph } from './portableGraph.js'
|
||||||
import { v4 as uuidv4 } from '../universal/uuid.js'
|
import { v4 as uuidv4 } from '../universal/uuid.js'
|
||||||
import { coerceNewEntityId, resolveEntityId, ORIGINAL_ID_KEY } from '../utils/idNormalization.js'
|
import { coerceNewEntityId, resolveEntityId, ORIGINAL_ID_KEY } from '../utils/idNormalization.js'
|
||||||
import { EntityNotFoundError, RelationNotFoundError } from '../errors/notFound.js'
|
import { EntityNotFoundError } from '../errors/notFound.js'
|
||||||
import { SpeculativeOverlayError, CanonicalEnumerationUnavailableError } from './errors.js'
|
import { SpeculativeOverlayError, CanonicalEnumerationUnavailableError } from './errors.js'
|
||||||
import type { GenerationStore } from './generationStore.js'
|
import type { GenerationStore } from './generationStore.js'
|
||||||
import type { ChangedIds, TransactReceipt, TxOperation } from './types.js'
|
import type { ChangedIds, TransactReceipt, TxOperation } from './types.js'
|
||||||
|
|
@ -698,39 +698,6 @@ export class Db<T = any> {
|
||||||
return this.get(id)
|
return this.get(id)
|
||||||
}
|
}
|
||||||
|
|
||||||
// The verb counterpart of `speculativeGet`. No `Db.getRelation(id)`
|
|
||||||
// exists (only the array-returning `related()`), so this replicates its
|
|
||||||
// generation-aware single-id resolution: the overlay first, then the
|
|
||||||
// generational before-image if something after this pin touched the
|
|
||||||
// verb (the same `resolveAt('verb', …)` + `relationFromRecord` pairing
|
|
||||||
// `related()` uses for its own changed-but-not-overlaid merge), else the
|
|
||||||
// live stored verb — nothing after this pin touched it, so the live
|
|
||||||
// state IS this view's state.
|
|
||||||
const speculativeGetRelation = async (id: string): Promise<Relation<T> | null> => {
|
|
||||||
if (overlay.verbs.has(id)) return overlay.verbs.get(id) ?? null
|
|
||||||
const resolved = await this.host.store.resolveAt('verb', id, this.gen)
|
|
||||||
if (resolved.source === 'absent') return null
|
|
||||||
if (resolved.source === 'record') {
|
|
||||||
return this.host.relationFromRecord(id, { metadata: resolved.metadata, vector: resolved.vector })
|
|
||||||
}
|
|
||||||
const stored = await this.host.storage.getVerb(id)
|
|
||||||
if (!stored) return null
|
|
||||||
return {
|
|
||||||
id: stored.id,
|
|
||||||
from: stored.sourceId,
|
|
||||||
to: stored.targetId,
|
|
||||||
type: stored.verb,
|
|
||||||
...(stored.subtype !== undefined && { subtype: stored.subtype }),
|
|
||||||
...(stored.visibility !== undefined && { visibility: stored.visibility }),
|
|
||||||
weight: stored.weight ?? 1.0,
|
|
||||||
...(stored.confidence !== undefined && { confidence: stored.confidence }),
|
|
||||||
data: stored.data,
|
|
||||||
metadata: (stored.metadata ?? {}) as T,
|
|
||||||
...(stored.service !== undefined && { service: stored.service }),
|
|
||||||
createdAt: stored.createdAt
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
for (const op of ops) {
|
for (const op of ops) {
|
||||||
switch (op.op) {
|
switch (op.op) {
|
||||||
case 'add': {
|
case 'add': {
|
||||||
|
|
@ -888,35 +855,6 @@ export class Db<T = any> {
|
||||||
}
|
}
|
||||||
break
|
break
|
||||||
}
|
}
|
||||||
case 'updateRelation': {
|
|
||||||
const base = await speculativeGetRelation(op.id)
|
|
||||||
if (!base) {
|
|
||||||
throw new RelationNotFoundError(
|
|
||||||
op.id,
|
|
||||||
`with(): relationship ${op.id} not found at generation ${this.gen}`
|
|
||||||
)
|
|
||||||
}
|
|
||||||
// Field-addressing law — mirror of the 'update' case: the patch
|
|
||||||
// bag is the user's verbatim; engine scalars only from dedicated
|
|
||||||
// op fields.
|
|
||||||
const custom = { ...(op.metadata as Record<string, unknown> | undefined) }
|
|
||||||
const mergedMetadata =
|
|
||||||
op.merge !== false
|
|
||||||
? ({ ...(base.metadata as object), ...custom } as T)
|
|
||||||
: ((op.metadata !== undefined ? custom : base.metadata) as T)
|
|
||||||
overlay.verbs.set(op.id, {
|
|
||||||
...base,
|
|
||||||
...(op.type !== undefined && { type: op.type }),
|
|
||||||
...(op.subtype !== undefined && { subtype: op.subtype }),
|
|
||||||
...(op.visibility !== undefined && { visibility: op.visibility }),
|
|
||||||
...(op.weight !== undefined && { weight: op.weight }),
|
|
||||||
...(op.confidence !== undefined && { confidence: op.confidence }),
|
|
||||||
...(op.data !== undefined && { data: op.data }),
|
|
||||||
metadata: mergedMetadata,
|
|
||||||
updatedAt: Date.now()
|
|
||||||
})
|
|
||||||
break
|
|
||||||
}
|
|
||||||
case 'unrelate': {
|
case 'unrelate': {
|
||||||
overlay.verbs.set(op.id, null)
|
overlay.verbs.set(op.id, null)
|
||||||
break
|
break
|
||||||
|
|
|
||||||
|
|
@ -28,7 +28,7 @@
|
||||||
* speculative `with()` overlay; the canonical storage walk only ever answers
|
* speculative `with()` overlay; the canonical storage walk only ever answers
|
||||||
* "what is live right now."
|
* "what is live right now."
|
||||||
*
|
*
|
||||||
* All are exported from the package root (`@soulcraft/brainy`).
|
* All are exported from the package root (`@soulcraftlabs/brainy`).
|
||||||
*/
|
*/
|
||||||
|
|
||||||
/**
|
/**
|
||||||
|
|
@ -351,3 +351,63 @@ export class PendingFlushDurabilityError extends Error {
|
||||||
this.failedAttempts = failedAttempts
|
this.failedAttempts = failedAttempts
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* @description Thrown by {@link GenerationStore.commitTransaction} when the
|
||||||
|
* PENDING single-op tier is non-empty — i.e. one or more `commitSingleOp()`
|
||||||
|
* generations are buffered in memory, not yet flushed to
|
||||||
|
* `committedRanges` via `flushPendingSingleOps()`.
|
||||||
|
*
|
||||||
|
* The invariant `reservedGensAsc()` (and everything built on it —
|
||||||
|
* `resolveManyAt`, `resolveAt`, `changedBetween`, the hot-tail window) relies
|
||||||
|
* on is documented, not enforced by types: pending generations must always be
|
||||||
|
* numerically greater than every committed one, because the ONLY sanctioned
|
||||||
|
* callers of `commitTransaction()` — `Brainy.transact()` and
|
||||||
|
* `Brainy.compactHistory()` — flush the pending tier FIRST. A caller that
|
||||||
|
* invokes `commitTransaction()` directly while single-ops are still pending
|
||||||
|
* breaks that invariant: the new commit lands in `committedRanges` ABOVE
|
||||||
|
* generations still sitting in `pendingGens`, so the committed-then-pending
|
||||||
|
* concatenation `reservedGensAsc()` yields is no longer ascending. The
|
||||||
|
* concrete failure this produces is silent, not a crash: `resolveManyAt`
|
||||||
|
* walks committed ranges before pending ones, so it can report a NEWER
|
||||||
|
* generation as the "first after" a pin than an older, still-pending one that
|
||||||
|
* actually touched the id first — a wrong before-image at a point-in-time
|
||||||
|
* read, without a compensating error to warn a caller anything went wrong.
|
||||||
|
*
|
||||||
|
* This error refuses the commit outright, before any staging I/O: nothing is
|
||||||
|
* written, the generation counter reservation is untouched, and
|
||||||
|
* `committedRanges`/`pendingGens` are exactly as they were. Call
|
||||||
|
* `flushPendingSingleOps()` first (or go through `Brainy.transact()`, which
|
||||||
|
* already does).
|
||||||
|
*
|
||||||
|
* @example
|
||||||
|
* try {
|
||||||
|
* await generationStore.commitTransaction({ touched, execute })
|
||||||
|
* } catch (err) {
|
||||||
|
* if (err instanceof PendingSingleOpsUnflushedError) {
|
||||||
|
* await generationStore.flushPendingSingleOps()
|
||||||
|
* await generationStore.commitTransaction({ touched, execute }) // now safe
|
||||||
|
* }
|
||||||
|
* }
|
||||||
|
*/
|
||||||
|
export class PendingSingleOpsUnflushedError extends Error {
|
||||||
|
/** How many un-flushed single-op generations were buffered at refusal time. */
|
||||||
|
public readonly pendingCount: number
|
||||||
|
|
||||||
|
/**
|
||||||
|
* @param pendingCount - `pendingGens.length` at the moment of refusal (always ≥ 1).
|
||||||
|
*/
|
||||||
|
constructor(pendingCount: number) {
|
||||||
|
super(
|
||||||
|
`commitTransaction() refused: ${pendingCount} pending single-op generation(s) ` +
|
||||||
|
`are still buffered and un-flushed. Flush the pending single-op tier before ` +
|
||||||
|
`committing a transaction — Brainy.transact() does this automatically; a ` +
|
||||||
|
`direct commitTransaction() call with pending generations would leave the ` +
|
||||||
|
`generation order unsorted (committed generations landing above lower, ` +
|
||||||
|
`still-pending ones) and make point-in-time reads (resolveManyAt/resolveAt) ` +
|
||||||
|
`return the wrong before-image. Call flushPendingSingleOps() first, then retry.`
|
||||||
|
)
|
||||||
|
this.name = 'PendingSingleOpsUnflushedError'
|
||||||
|
this.pendingCount = pendingCount
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
|
||||||
|
|
@ -40,7 +40,10 @@
|
||||||
* The manifest (`_generations/facts/manifest.json`, JSON — forensics stay
|
* The manifest (`_generations/facts/manifest.json`, JSON — forensics stay
|
||||||
* terminal-readable) is the single source of truth for the segment SET;
|
* terminal-readable) is the single source of truth for the segment SET;
|
||||||
* rotation flips it atomically (write-new → fsync → rename) BEFORE the new
|
* rotation flips it atomically (write-new → fsync → rename) BEFORE the new
|
||||||
* tail's first byte exists, so no segment file is ever unaccounted for.
|
* tail's first byte exists, so no segment file is ever unaccounted for. Its
|
||||||
|
* per-segment `firstGeneration`/`lastGeneration` are LOAD-BEARING at open: a
|
||||||
|
* recovery pass looking for facts above a bound reads only the segments those
|
||||||
|
* bounds cannot rule out (the prune law — see `segmentsHoldingFactsAbove`).
|
||||||
*
|
*
|
||||||
* ## Mixed-version logs (the v2 live-write cutover)
|
* ## Mixed-version logs (the v2 live-write cutover)
|
||||||
*
|
*
|
||||||
|
|
@ -689,6 +692,74 @@ function parseSegment(
|
||||||
return { facts, validBytes: offset, formatVersion: FACT_LOG_FORMAT_V1 }
|
return { facts, validBytes: offset, formatVersion: FACT_LOG_FORMAT_V1 }
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* THE PRUNE LAW — which segment files a pass looking for facts ABOVE
|
||||||
|
* `committedGeneration` actually has to read, and how many the manifest's own
|
||||||
|
* recorded bounds took off the table.
|
||||||
|
*
|
||||||
|
* A sealed segment's `lastGeneration` is written at SEAL time and never
|
||||||
|
* mutated upward afterwards ({@link FactLog.rotate}, unchanged since the log
|
||||||
|
* was introduced): the tail's bytes are fsynced FIRST (`await this.sync()` —
|
||||||
|
* "sealed segments are always fully durable"), the entry is then built from
|
||||||
|
* the content that fsync covered, and only then does the manifest flip —
|
||||||
|
* atomically (tmp+rename) and fsynced — which in the SAME write re-points
|
||||||
|
* `tailSegment` at a new file, so the sealed file is never appended to again.
|
||||||
|
* A crash anywhere in that order is safe in the pruning direction: crash
|
||||||
|
* before the manifest write and the segment is still the TAIL (read whole);
|
||||||
|
* crash after it and the entry describes bytes that were already durable. The
|
||||||
|
* only later mutation of a sealed segment is `open()`'s straddle truncation,
|
||||||
|
* which REMOVES facts and re-derives the entry from the actual bytes — so a
|
||||||
|
* recorded bound can drift DOWN with its file, never up.
|
||||||
|
*
|
||||||
|
* Therefore: `lastGeneration = L` proves the file holds no fact above L, and
|
||||||
|
* a pass above `committedGeneration >= L` can skip it whole — no read, no
|
||||||
|
* CRC decode, no msgpack. What the manifest cannot PROVE is never pruned: an
|
||||||
|
* entry with no numeric `lastGeneration` (a legacy or hand-repaired manifest)
|
||||||
|
* is read, and the unsealed tail is always read.
|
||||||
|
*
|
||||||
|
* This is the difference between an open that costs O(whole fact log) and one
|
||||||
|
* that costs O(the facts that could matter). MEASURED in production: a 16k-row
|
||||||
|
* brain at generation ~478,819 paid 34-37s of segment reads and CRC decoding
|
||||||
|
* in `generation-store-open-fold` on EVERY open — to answer a question whose
|
||||||
|
* answer, after a clean close, is always "nothing".
|
||||||
|
*/
|
||||||
|
function segmentsHoldingFactsAbove(
|
||||||
|
stored: FactsManifest,
|
||||||
|
committedGeneration: number
|
||||||
|
): { files: string[]; pruned: number } {
|
||||||
|
const files: string[] = []
|
||||||
|
let pruned = 0
|
||||||
|
for (const entry of stored.segments) {
|
||||||
|
const last = (entry as Partial<SegmentEntry>).lastGeneration
|
||||||
|
if (typeof last === 'number' && Number.isFinite(last) && last <= committedGeneration) {
|
||||||
|
pruned++
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
files.push(entry.file)
|
||||||
|
}
|
||||||
|
if (stored.tailSegment) files.push(stored.tailSegment)
|
||||||
|
return { files, pruned }
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Say what the open actually read. One line, and only when the log holds more
|
||||||
|
* than one segment (a single-segment log has nothing to prune and nothing to
|
||||||
|
* report) — the operator's receipt that the open is paying for the tail, not
|
||||||
|
* for the whole history.
|
||||||
|
*/
|
||||||
|
function narrateAboveScan(
|
||||||
|
pass: string,
|
||||||
|
committedGeneration: number,
|
||||||
|
read: number,
|
||||||
|
pruned: number
|
||||||
|
): void {
|
||||||
|
if (read + pruned <= 1) return
|
||||||
|
prodLog.narrate(
|
||||||
|
`[FactLog] ${pass} above generation ${committedGeneration}: ${read} segment(s) read, ` +
|
||||||
|
`${pruned} pruned of ${read + pruned} (sealed at or below the bound)`
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* The generation fact log. One instance per open store; every method assumes
|
* The generation fact log. One instance per open store; every method assumes
|
||||||
* the single-writer discipline the generation store already enforces (calls
|
* the single-writer discipline the generation store already enforces (calls
|
||||||
|
|
@ -754,22 +825,6 @@ export class FactLog {
|
||||||
return this.manifest.brainId !== undefined || this.tailVersion === FACT_LOG_FORMAT_V2
|
return this.manifest.brainId !== undefined || this.tailVersion === FACT_LOG_FORMAT_V2
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
|
||||||
* Open the log and reconcile it to committed truth: read the manifest,
|
|
||||||
* establish the tail's intact content (torn-tail scan), then TRUNCATE any
|
|
||||||
* fact with `generation > committedGeneration` — those never committed (a
|
|
||||||
* crash between fact-append and the commit point). After open, the log is
|
|
||||||
* exactly the committed prefix.
|
|
||||||
*/
|
|
||||||
/**
|
|
||||||
* Read (without truncating) every intact fact ABOVE a generation — the
|
|
||||||
* log-authority recovery surface: after a crash, facts beyond the
|
|
||||||
* manifest watermark that survived with valid CRCs are ACKED writes in
|
|
||||||
* durable-at-ack mode, and the owner REPLAYS them instead of letting
|
|
||||||
* open() truncate them. Must be called BEFORE open() (it reads the raw
|
|
||||||
* segments directly; the torn tail's invalid suffix is ignored exactly
|
|
||||||
* like open() would).
|
|
||||||
*/
|
|
||||||
/**
|
/**
|
||||||
* STREAMING twin of {@link FactLog.peekFactsAbove} for the recovery fold:
|
* STREAMING twin of {@link FactLog.peekFactsAbove} for the recovery fold:
|
||||||
* yields facts above the bound one SEGMENT at a time, ascending, without
|
* yields facts above the bound one SEGMENT at a time, ascending, without
|
||||||
|
|
@ -779,13 +834,18 @@ export class FactLog {
|
||||||
* Works manifest-direct (safe before {@link FactLog.open}). Ordering is
|
* Works manifest-direct (safe before {@link FactLog.open}). Ordering is
|
||||||
* structural (segments rotate in order; appends are ordered within one) and
|
* structural (segments rotate in order; appends are ordered within one) and
|
||||||
* ASSERTED — a violation aborts loudly, never a silent misordered replay.
|
* ASSERTED — a violation aborts loudly, never a silent misordered replay.
|
||||||
|
*
|
||||||
|
* Reads only the segments that CAN hold a fact above the bound — see
|
||||||
|
* {@link segmentsHoldingFactsAbove}. A bounded fold above a high checkpoint
|
||||||
|
* therefore reads its own tail, not the whole history it already proved
|
||||||
|
* durable.
|
||||||
*/
|
*/
|
||||||
async *streamFactsAbove(committedGeneration: number): AsyncGenerator<CommitFact[], void> {
|
async *streamFactsAbove(committedGeneration: number): AsyncGenerator<CommitFact[], void> {
|
||||||
const stored = (await this.storage.readRawObject(FACTS_MANIFEST_PATH)) as FactsManifest | null
|
const stored = (await this.storage.readRawObject(FACTS_MANIFEST_PATH)) as FactsManifest | null
|
||||||
if (!stored || typeof stored !== 'object' || !Array.isArray(stored.segments)) return
|
if (!stored || typeof stored !== 'object' || !Array.isArray(stored.segments)) return
|
||||||
if (stored.formatVersion !== FACTS_FORMAT_VERSION) return
|
if (stored.formatVersion !== FACTS_FORMAT_VERSION) return
|
||||||
const files = [...stored.segments.map((s) => s.file)]
|
const { files, pruned } = segmentsHoldingFactsAbove(stored, committedGeneration)
|
||||||
if (stored.tailSegment) files.push(stored.tailSegment)
|
narrateAboveScan('recovery fold', committedGeneration, files.length, pruned)
|
||||||
let lastGen = committedGeneration
|
let lastGen = committedGeneration
|
||||||
for (const file of files) {
|
for (const file of files) {
|
||||||
const bytes = await this.storage.readRawBytes(`${FACTS_PREFIX}/${file}`)
|
const bytes = await this.storage.readRawBytes(`${FACTS_PREFIX}/${file}`)
|
||||||
|
|
@ -807,13 +867,27 @@ export class FactLog {
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Read (without truncating) every intact fact ABOVE a generation — the
|
||||||
|
* log-authority recovery surface: after a crash, facts beyond the
|
||||||
|
* manifest watermark that survived with valid CRCs are ACKED writes in
|
||||||
|
* durable-at-ack mode, and the owner REPLAYS them instead of letting
|
||||||
|
* open() truncate them. Must be called BEFORE open() (it reads the raw
|
||||||
|
* segments directly; the torn tail's invalid suffix is ignored exactly
|
||||||
|
* like open() would).
|
||||||
|
*
|
||||||
|
* Reads only the segments that CAN hold such a fact — see
|
||||||
|
* {@link segmentsHoldingFactsAbove}. This runs on EVERY log-authority open,
|
||||||
|
* including the clean one where the answer is always empty, so the segments
|
||||||
|
* the manifest already proves irrelevant are never opened at all.
|
||||||
|
*/
|
||||||
async peekFactsAbove(committedGeneration: number): Promise<CommitFact[]> {
|
async peekFactsAbove(committedGeneration: number): Promise<CommitFact[]> {
|
||||||
const stored = (await this.storage.readRawObject(FACTS_MANIFEST_PATH)) as FactsManifest | null
|
const stored = (await this.storage.readRawObject(FACTS_MANIFEST_PATH)) as FactsManifest | null
|
||||||
if (!stored || typeof stored !== 'object' || !Array.isArray(stored.segments)) return []
|
if (!stored || typeof stored !== 'object' || !Array.isArray(stored.segments)) return []
|
||||||
if (stored.formatVersion !== FACTS_FORMAT_VERSION) return []
|
if (stored.formatVersion !== FACTS_FORMAT_VERSION) return []
|
||||||
const out: CommitFact[] = []
|
const out: CommitFact[] = []
|
||||||
const files = [...stored.segments.map((s) => s.file)]
|
const { files, pruned } = segmentsHoldingFactsAbove(stored, committedGeneration)
|
||||||
if (stored.tailSegment) files.push(stored.tailSegment)
|
narrateAboveScan('above-manifest peek', committedGeneration, files.length, pruned)
|
||||||
for (const file of files) {
|
for (const file of files) {
|
||||||
const bytes = await this.storage.readRawBytes(`${FACTS_PREFIX}/${file}`)
|
const bytes = await this.storage.readRawBytes(`${FACTS_PREFIX}/${file}`)
|
||||||
if (bytes === null) continue
|
if (bytes === null) continue
|
||||||
|
|
@ -826,6 +900,13 @@ export class FactLog {
|
||||||
return out
|
return out
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Open the log and reconcile it to committed truth: read the manifest,
|
||||||
|
* establish the tail's intact content (torn-tail scan), then TRUNCATE any
|
||||||
|
* fact with `generation > committedGeneration` — those never committed (a
|
||||||
|
* crash between fact-append and the commit point). After open, the log is
|
||||||
|
* exactly the committed prefix.
|
||||||
|
*/
|
||||||
async open(committedGeneration: number): Promise<void> {
|
async open(committedGeneration: number): Promise<void> {
|
||||||
const stored = (await this.storage.readRawObject(FACTS_MANIFEST_PATH)) as FactsManifest | null
|
const stored = (await this.storage.readRawObject(FACTS_MANIFEST_PATH)) as FactsManifest | null
|
||||||
if (stored && typeof stored === 'object' && Array.isArray(stored.segments)) {
|
if (stored && typeof stored === 'object' && Array.isArray(stored.segments)) {
|
||||||
|
|
|
||||||
|
|
@ -12,9 +12,11 @@
|
||||||
* the verified surface is a small set of rollup invariants (entity/
|
* the verified surface is a small set of rollup invariants (entity/
|
||||||
* relationship counts) plus `sourceGeneration`.
|
* relationship counts) plus `sourceGeneration`.
|
||||||
*
|
*
|
||||||
* `sourceGeneration` is the generation of the source-of-truth log this
|
* `sourceGeneration` is the COMMITTED generation of the source-of-truth log
|
||||||
* projection reflects — open-time coherence becomes a COMPARISON (stamp vs
|
* this projection reflects — never the allocated counter, which names a
|
||||||
* log head), not a walk:
|
* generation that may never commit (see {@link StampVerdict.torn}) — so
|
||||||
|
* open-time coherence becomes a COMPARISON (stamp vs committed head), not a
|
||||||
|
* walk:
|
||||||
*
|
*
|
||||||
* - equal + invariants hold → coherent, serve.
|
* - equal + invariants hold → coherent, serve.
|
||||||
* - behind → the projection missed the tail (crash between commit and stamp);
|
* - behind → the projection missed the tail (crash between commit and stamp);
|
||||||
|
|
@ -24,6 +26,9 @@
|
||||||
* - invariants FAIL at equal generation → genuine incoherence: loud, and the
|
* - invariants FAIL at equal generation → genuine incoherence: loud, and the
|
||||||
* repair ritual (`repairIndex()`, whose recount rebuilds the rollups from a
|
* repair ritual (`repairIndex()`, whose recount rebuilds the rollups from a
|
||||||
* canonical walk) heals it.
|
* canonical walk) heals it.
|
||||||
|
* - AHEAD → a torn generation-log tail: the stamp's fsync outlived the log
|
||||||
|
* tail's. TERMINAL, never a wait — the generation the stamp names does not
|
||||||
|
* exist to arrive.
|
||||||
*
|
*
|
||||||
* Stamps are JSON on purpose — every incident gets debugged by reading a
|
* Stamps are JSON on purpose — every incident gets debugged by reading a
|
||||||
* stamp in a terminal.
|
* stamp in a terminal.
|
||||||
|
|
@ -70,6 +75,12 @@ export type StampVerdict =
|
||||||
| { state: 'coherent' }
|
| { state: 'coherent' }
|
||||||
| { state: 'absent' } // legacy store — first stamp writes at the next flush
|
| { state: 'absent' } // legacy store — first stamp writes at the next flush
|
||||||
| { state: 'behind'; stampSource: number; head: number }
|
| { state: 'behind'; stampSource: number; head: number }
|
||||||
|
/**
|
||||||
|
* TORN GENERATION-LOG TAIL: the stamp witnesses a source generation the
|
||||||
|
* store's committed watermark can no longer show. TERMINAL — there is no
|
||||||
|
* generation to wait for, so the open demotes (or refuses) and never spins.
|
||||||
|
*/
|
||||||
|
| { state: 'torn'; stampSource: number; head: number }
|
||||||
| { state: 'incoherent'; failures: string[] }
|
| { state: 'incoherent'; failures: string[] }
|
||||||
| { state: 'unverifiable'; reason: string } // a FAULT reading the stamp — never conflated with absence
|
| { state: 'unverifiable'; reason: string } // a FAULT reading the stamp — never conflated with absence
|
||||||
|
|
||||||
|
|
@ -118,12 +129,15 @@ export function verifyFamilyStamp(
|
||||||
): StampVerdict {
|
): StampVerdict {
|
||||||
if (stamp === null) return { state: 'absent' }
|
if (stamp === null) return { state: 'absent' }
|
||||||
if (stamp.sourceGeneration > head) {
|
if (stamp.sourceGeneration > head) {
|
||||||
// A stamp AHEAD of the log claims state that never committed — the
|
// A stamp AHEAD of committed truth witnesses a generation the store can no
|
||||||
// projection was stamped against truth that a crash rolled back.
|
// longer show: the stamp's fsync survived a crash that the log tail did
|
||||||
return {
|
// not. This is the TORN GENERATION-LOG TAIL — its own class, never folded
|
||||||
state: 'incoherent',
|
// in with `incoherent` (a count that drifted at a generation both sides
|
||||||
failures: [`sourceGeneration ${stamp.sourceGeneration} is ahead of the log head ${head}`]
|
// agree on), because the two have opposite cures: incoherence is recounted,
|
||||||
}
|
// a tear is DEMOTED. It is also terminal by construction — there is no
|
||||||
|
// generation the open can wait for, because the one the stamp names is
|
||||||
|
// gone.
|
||||||
|
return { state: 'torn', stampSource: stamp.sourceGeneration, head }
|
||||||
}
|
}
|
||||||
if (stamp.sourceGeneration < head) {
|
if (stamp.sourceGeneration < head) {
|
||||||
return { state: 'behind', stampSource: stamp.sourceGeneration, head }
|
return { state: 'behind', stampSource: stamp.sourceGeneration, head }
|
||||||
|
|
|
||||||
|
|
@ -147,6 +147,60 @@ export class GenerationSegmentStore {
|
||||||
return this.coveringSegment(gen) !== null
|
return this.coveringSegment(gen) !== null
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* @description True when `meta` declares more generations than it holds
|
||||||
|
* frames — a segment sealed by a writer that folded across a hole. The
|
||||||
|
* manifest records `frames` at fold time, so this is an O(1) comparison
|
||||||
|
* against the declared span and needs no I/O.
|
||||||
|
*/
|
||||||
|
private isSparse(meta: SegmentMeta): boolean {
|
||||||
|
return meta.lastGeneration - meta.firstGeneration + 1 !== meta.frames
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* @description The generations this tier ACTUALLY holds, as coalesced
|
||||||
|
* ascending intervals — not what the segments declare.
|
||||||
|
*
|
||||||
|
* Dense segments (every one a current writer produces) contribute their
|
||||||
|
* declared range with no I/O. A SPARSE segment — one sealed before the
|
||||||
|
* density law was enforced, whose declared range spans generations it has
|
||||||
|
* no frame for — has its real generation list read from its sidecar and
|
||||||
|
* contributed instead, with the discrepancy narrated once.
|
||||||
|
*
|
||||||
|
* This is what keeps a store that already carries the damage from wedging.
|
||||||
|
* `open()` seeds `committedRanges` from these intervals, so a hole is never
|
||||||
|
* re-admitted as a committed generation, and the auto-compaction pass that
|
||||||
|
* used to fail on every run with "packed history is damaged" simply never
|
||||||
|
* asks for the missing frame.
|
||||||
|
*
|
||||||
|
* @returns Ascending, non-overlapping `[first, last]` intervals.
|
||||||
|
*/
|
||||||
|
async actualRanges(): Promise<Array<[number, number]>> {
|
||||||
|
const out: Array<[number, number]> = []
|
||||||
|
for (const meta of this.manifest.segments) {
|
||||||
|
if (!this.isSparse(meta)) {
|
||||||
|
out.push([meta.firstGeneration, meta.lastGeneration])
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
const missing = meta.lastGeneration - meta.firstGeneration + 1 - meta.frames
|
||||||
|
prodLog.warn(
|
||||||
|
`[GenerationSegments] sealed segment ${meta.file} declares generations ` +
|
||||||
|
`${meta.firstGeneration}..${meta.lastGeneration} but holds only ${meta.frames} ` +
|
||||||
|
`frame(s) — ${missing} generation(s) in that span were never folded into it. ` +
|
||||||
|
`Serving the frames it actually holds; the declared span is not treated as ` +
|
||||||
|
`committed history. (Written by a pre-density-law writer that folded across a ` +
|
||||||
|
`gap; the segment itself is intact and no record is lost.)`
|
||||||
|
)
|
||||||
|
const idx = await this.sidecarFor(meta)
|
||||||
|
for (const [gen] of idx.generations) {
|
||||||
|
const last = out[out.length - 1]
|
||||||
|
if (last !== undefined && gen === last[1] + 1) last[1] = gen
|
||||||
|
else out.push([gen, gen])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return out
|
||||||
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Fold consecutive generations into ONE new sealed segment + sidecar and
|
* Fold consecutive generations into ONE new sealed segment + sidecar and
|
||||||
* append it to the manifest atomically. Caller guarantees: `gens` is
|
* append it to the manifest atomically. Caller guarantees: `gens` is
|
||||||
|
|
@ -164,6 +218,38 @@ export class GenerationSegmentStore {
|
||||||
throw new Error('[GenerationSegments] fold() input must be strictly ascending')
|
throw new Error('[GenerationSegments] fold() input must be strictly ascending')
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
// THE DENSITY LAW, MADE MECHANICAL.
|
||||||
|
//
|
||||||
|
// A sealed segment declares a CONTIGUOUS range [firstGeneration,
|
||||||
|
// lastGeneration] and every reader treats that range as containment:
|
||||||
|
// `coveringSegment` is an interval test, `hasGeneration` returns true for
|
||||||
|
// anything inside it, and `open()` seeds committedRanges from it. So a
|
||||||
|
// segment folded from a SPARSE input silently claims generations it does
|
||||||
|
// not hold, and the first read of one of those holes throws
|
||||||
|
// "inside sealed segment ... but has no frame — packed history is damaged".
|
||||||
|
//
|
||||||
|
// That is exactly how the damage was produced. `repackHistory` skipped
|
||||||
|
// generations mid-batch — ones absent from committedRanges, ones still in
|
||||||
|
// the pending buffer, ones whose tx.json would not read — and handed the
|
||||||
|
// survivors here, where the range was computed from the first and last of
|
||||||
|
// them. Worse, the mis-declared range was then merged back into
|
||||||
|
// committedRanges at the next open, which is what turned a quiet hole into
|
||||||
|
// a repeating auto-compaction failure on every subsequent run.
|
||||||
|
//
|
||||||
|
// Callers now split at discontinuities; this refusal is what keeps any
|
||||||
|
// future caller from reintroducing the class. A refusal here loses
|
||||||
|
// nothing — the generations stay in the live tier, readable, and the next
|
||||||
|
// pass folds them correctly.
|
||||||
|
for (let i = 1; i < gens.length; i++) {
|
||||||
|
if (gens[i].generation !== gens[i - 1].generation + 1) {
|
||||||
|
throw new Error(
|
||||||
|
`[GenerationSegments] fold() input is not contiguous: ${gens[i - 1].generation} → ` +
|
||||||
|
`${gens[i].generation} skips ${gens[i].generation - gens[i - 1].generation - 1} ` +
|
||||||
|
`generation(s). A sealed segment declares a dense range, so folding a sparse ` +
|
||||||
|
`batch would claim generations it does not hold. Split the batch at the gap.`
|
||||||
|
)
|
||||||
|
}
|
||||||
|
}
|
||||||
const last = this.manifest.segments[this.manifest.segments.length - 1]
|
const last = this.manifest.segments[this.manifest.segments.length - 1]
|
||||||
if (last && gens[0].generation <= last.lastGeneration) {
|
if (last && gens[0].generation <= last.lastGeneration) {
|
||||||
throw new Error(
|
throw new Error(
|
||||||
|
|
@ -364,12 +450,37 @@ export class GenerationSegmentStore {
|
||||||
return this.decodeFrame(payload)
|
return this.decodeFrame(payload)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
// In the covering range but not present: the packed tier is dense by
|
// Inside the covering range but with no frame. Two very different causes,
|
||||||
// construction (fold packs every generation it is handed, including
|
// and conflating them is what made this class wedge every maintenance pass
|
||||||
// record-less ones) — absence inside a sealed range is damage.
|
// on the affected stores.
|
||||||
|
//
|
||||||
|
// (1) A SPARSE SEGMENT — the manifest's own `frames` count is smaller than
|
||||||
|
// the span it declares. That segment was sealed by a writer that
|
||||||
|
// folded across a hole (the class this file's density law now bars).
|
||||||
|
// The segment is INTACT and nothing is lost; it simply never held this
|
||||||
|
// generation. Answering "not packed" is the honest answer, and it lets
|
||||||
|
// the caller's two-tier read decide what a genuinely absent generation
|
||||||
|
// means, instead of every compaction pass dying on a repeating throw.
|
||||||
|
// `actualRanges()` keeps such holes out of committedRanges at open, so
|
||||||
|
// in a healed store nobody asks this question in the first place.
|
||||||
|
//
|
||||||
|
// (2) A DENSE SEGMENT missing a frame it says it has — the manifest and
|
||||||
|
// the sidecar disagree about a segment that claims to be complete.
|
||||||
|
// That IS damage, and it stays loud.
|
||||||
|
if (this.isSparse(meta)) {
|
||||||
|
prodLog.warn(
|
||||||
|
`[GenerationSegments] generation ${gen} falls inside sealed segment ${meta.file}'s ` +
|
||||||
|
`declared range ${meta.firstGeneration}..${meta.lastGeneration}, but that segment ` +
|
||||||
|
`holds ${meta.frames} frame(s) for a ${meta.lastGeneration - meta.firstGeneration + 1}` +
|
||||||
|
`-generation span — it was sealed across a gap and never held this generation. ` +
|
||||||
|
`Reporting it as unpacked rather than as damage; no record is lost.`
|
||||||
|
)
|
||||||
|
return null
|
||||||
|
}
|
||||||
throw new Error(
|
throw new Error(
|
||||||
`[GenerationSegments] generation ${gen} is inside sealed segment ${meta.file}'s declared ` +
|
`[GenerationSegments] generation ${gen} is inside sealed segment ${meta.file}'s declared ` +
|
||||||
`range but has no frame — packed history is damaged`
|
`range but has no frame, and that segment declares a complete ${meta.frames}-frame ` +
|
||||||
|
`span — the manifest and the sidecar disagree; packed history is damaged`
|
||||||
)
|
)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -32,7 +32,13 @@
|
||||||
*/
|
*/
|
||||||
|
|
||||||
import { prodLog } from '../utils/logger.js'
|
import { prodLog } from '../utils/logger.js'
|
||||||
import { GenerationCompactedError, GenerationConflictError, PendingFlushDurabilityError, StoreInconsistentError } from './errors.js'
|
import {
|
||||||
|
GenerationCompactedError,
|
||||||
|
GenerationConflictError,
|
||||||
|
PendingFlushDurabilityError,
|
||||||
|
PendingSingleOpsUnflushedError,
|
||||||
|
StoreInconsistentError
|
||||||
|
} from './errors.js'
|
||||||
import type { UnreconciledRecord } from './errors.js'
|
import type { UnreconciledRecord } from './errors.js'
|
||||||
import { TransactionRollbackError } from '../transaction/errors.js'
|
import { TransactionRollbackError } from '../transaction/errors.js'
|
||||||
import type {
|
import type {
|
||||||
|
|
@ -96,6 +102,35 @@ export const FOLD_CHECKPOINT_PATH = '_system/fold-checkpoint.json'
|
||||||
/** Storage-root-relative prefix of the per-generation record directories. */
|
/** Storage-root-relative prefix of the per-generation record directories. */
|
||||||
export const GENERATIONS_PREFIX = '_generations'
|
export const GENERATIONS_PREFIX = '_generations'
|
||||||
|
|
||||||
|
/**
|
||||||
|
* @description Split an ascending list of fold candidates into maximal
|
||||||
|
* CONTIGUOUS runs — `[7,8,9,12,13]` becomes `[[7,8,9],[12,13]]`.
|
||||||
|
*
|
||||||
|
* A sealed segment declares one dense range `[firstGeneration,
|
||||||
|
* lastGeneration]`, and every reader treats that range as containment. So a
|
||||||
|
* batch with a hole in it must never become one segment: it would claim a
|
||||||
|
* generation it does not hold, and the first read of that hole reports the
|
||||||
|
* packed history as damaged. One run, one segment — the ranges then describe
|
||||||
|
* exactly what the segments contain.
|
||||||
|
*
|
||||||
|
* @param gens - Fold candidates, strictly ascending by generation.
|
||||||
|
* @returns One array per contiguous run, in ascending order. Empty in, empty out.
|
||||||
|
*/
|
||||||
|
export function contiguousRuns(gens: FoldGeneration[]): FoldGeneration[][] {
|
||||||
|
const runs: FoldGeneration[][] = []
|
||||||
|
let run: FoldGeneration[] = []
|
||||||
|
for (const g of gens) {
|
||||||
|
const prev = run[run.length - 1]
|
||||||
|
if (prev !== undefined && g.generation !== prev.generation + 1) {
|
||||||
|
runs.push(run)
|
||||||
|
run = []
|
||||||
|
}
|
||||||
|
run.push(g)
|
||||||
|
}
|
||||||
|
if (run.length > 0) runs.push(run)
|
||||||
|
return runs
|
||||||
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* @description Phases of the {@link GenerationStore.commitTransaction} commit
|
* @description Phases of the {@link GenerationStore.commitTransaction} commit
|
||||||
* protocol at which a test-only fault injector can simulate a process crash.
|
* protocol at which a test-only fault injector can simulate a process crash.
|
||||||
|
|
@ -537,12 +572,29 @@ export class GenerationStore {
|
||||||
this.horizonGen = finiteGen(manifest?.horizon, 'manifest horizon')
|
this.horizonGen = finiteGen(manifest?.horizon, 'manifest horizon')
|
||||||
this.counter = Math.max(finiteGen(counterFile?.generation, 'generation counter'), this.committed)
|
this.counter = Math.max(finiteGen(counterFile?.generation, 'generation counter'), this.committed)
|
||||||
|
|
||||||
// Discover existing generation record directories.
|
// Discover existing generation record directories — BY DIRECTORY NAME.
|
||||||
const recordPaths = await this.storage.listRawObjects(GENERATIONS_PREFIX)
|
// This used to call listRawObjects(), which recurses the whole
|
||||||
|
// `_generations/` tree and returns every file in every generation, to
|
||||||
|
// extract a set of integers the top-level directory names already spell.
|
||||||
|
// MEASURED on a real store with an 11 GB generation history: the phase
|
||||||
|
// this sits in cost 55,538 ms of a WARM REOPEN after a clean close, with
|
||||||
|
// no fold to blame — this walk is what it was doing. An adapter without
|
||||||
|
// the one-level door falls back to the recursive listing, unchanged.
|
||||||
const seenGens = new Set<number>()
|
const seenGens = new Set<number>()
|
||||||
for (const p of recordPaths) {
|
const oneLevel = (
|
||||||
const gen = parseGenerationFromPath(p)
|
this.storage as { listRawPrefixes?: (prefix: string) => Promise<string[]> }
|
||||||
if (gen !== null) seenGens.add(gen)
|
).listRawPrefixes
|
||||||
|
if (typeof oneLevel === 'function') {
|
||||||
|
for (const name of await oneLevel.call(this.storage, GENERATIONS_PREFIX)) {
|
||||||
|
const gen = Number(name)
|
||||||
|
if (Number.isSafeInteger(gen) && gen >= 0) seenGens.add(gen)
|
||||||
|
}
|
||||||
|
} else {
|
||||||
|
const recordPaths = await this.storage.listRawObjects(GENERATIONS_PREFIX)
|
||||||
|
for (const p of recordPaths) {
|
||||||
|
const gen = parseGenerationFromPath(p)
|
||||||
|
if (gen !== null) seenGens.add(gen)
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
let rolledBack = 0
|
let rolledBack = 0
|
||||||
|
|
@ -652,21 +704,56 @@ export class GenerationStore {
|
||||||
: 'WHOLE-LOG fold'
|
: 'WHOLE-LOG fold'
|
||||||
: 'above-manifest replay'
|
: 'above-manifest replay'
|
||||||
let replayed = 0
|
let replayed = 0
|
||||||
|
const foldStartedAt = Date.now()
|
||||||
const replayFact = async (fact: CommitFact): Promise<void> => {
|
const replayFact = async (fact: CommitFact): Promise<void> => {
|
||||||
for (const op of fact.ops) {
|
for (const op of fact.ops) {
|
||||||
const image =
|
let image: { metadata: unknown | null; vector: unknown | null }
|
||||||
op.record === null
|
if (op.record === null) {
|
||||||
? { metadata: null, vector: null }
|
// A genuine tombstone (both legs absent) — the fold removes
|
||||||
: { metadata: op.record.metadata, vector: op.record.vector }
|
// both legs, exactly like `writeNounRaw`/`writeVerbRaw`'s raw
|
||||||
|
// exact-restore contract.
|
||||||
|
image = { metadata: null, vector: null }
|
||||||
|
} else if (
|
||||||
|
op.record.metadata !== null &&
|
||||||
|
(op.record.vector === null || op.record.vector === undefined)
|
||||||
|
) {
|
||||||
|
// PRESERVE-IF-ABSENT (population law, ADR-008 G1 — the fold's
|
||||||
|
// half): a metadata-only after-image must never DELETE an
|
||||||
|
// existing vector leg through the fold. `writeNounRaw`/
|
||||||
|
// `writeVerbRaw` are exact-restore primitives — a `vector:
|
||||||
|
// null` there means "delete", which is exactly right for
|
||||||
|
// `rollBackUncommittedGeneration`'s before-image restore (a
|
||||||
|
// transaction abort legitimately un-writes a vector the failed
|
||||||
|
// transaction added). It is NOT right here: this fold replays
|
||||||
|
// AFTER-IMAGES, and re-applying an already-intact record must
|
||||||
|
// be byte-safe (this module's own invariant, see the log-authority
|
||||||
|
// comment above) — silently erasing a landed vector because one
|
||||||
|
// replayed fact's vector leg came back null is the exact defect
|
||||||
|
// that left metadata-counted, never-enumerated rows in a
|
||||||
|
// production store (confirmed root cause: the enumeration walk
|
||||||
|
// used to key on the vector leg, so a preserved-but-then-deleted
|
||||||
|
// vector made the row invisible while the ledger still counted
|
||||||
|
// it by metadata). A genuine "unvector" has its own sanctioned,
|
||||||
|
// ledger-correct path (`Brainy.unvectorNounForRootMigration`) —
|
||||||
|
// never this raw primitive, and never the fold.
|
||||||
|
const current =
|
||||||
|
op.kind === 'verb'
|
||||||
|
? await this.storage.readVerbRaw(op.id)
|
||||||
|
: await this.storage.readNounRaw(op.id)
|
||||||
|
image = { metadata: op.record.metadata, vector: current.vector ?? null }
|
||||||
|
} else {
|
||||||
|
image = { metadata: op.record.metadata, vector: op.record.vector }
|
||||||
|
}
|
||||||
if (op.kind === 'verb') await this.storage.writeVerbRaw(op.id, image)
|
if (op.kind === 'verb') await this.storage.writeVerbRaw(op.id, image)
|
||||||
else await this.storage.writeNounRaw(op.id, image)
|
else await this.storage.writeNounRaw(op.id, image)
|
||||||
this.noteCheckpointDirty(op.kind, op.id)
|
this.noteCheckpointDirty(op.kind, op.id)
|
||||||
}
|
}
|
||||||
replayed++
|
replayed++
|
||||||
if (replayed % 1000 === 0) {
|
if (replayed % 1000 === 0) {
|
||||||
prodLog.warn(
|
prodLog.narrate(
|
||||||
`[GenerationStore] recovery fold in progress — ${replayed} fact(s) folded ` +
|
`[GenerationStore] recovery fold in progress — ${replayed} fact(s) folded ` +
|
||||||
`(at generation ${fact.generation}); do not restart, the fold is finite`
|
`in ${Date.now() - foldStartedAt}ms (at generation ${fact.generation}); ` +
|
||||||
|
`do not restart, the fold is finite`
|
||||||
)
|
)
|
||||||
}
|
}
|
||||||
if (fact.generation > this.committed) {
|
if (fact.generation > this.committed) {
|
||||||
|
|
@ -681,7 +768,7 @@ export class GenerationStore {
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
if (uncleanOpen) {
|
if (uncleanOpen) {
|
||||||
prodLog.warn(
|
prodLog.narrate(
|
||||||
`[GenerationStore] log-authority recovery: ${foldKind} beginning ` +
|
`[GenerationStore] log-authority recovery: ${foldKind} beginning ` +
|
||||||
`(unclean shutdown detected) — streaming replay, bounded memory, ` +
|
`(unclean shutdown detected) — streaming replay, bounded memory, ` +
|
||||||
`progress every 1000 facts. Do not restart the process; a restart ` +
|
`progress every 1000 facts. Do not restart the process; a restart ` +
|
||||||
|
|
@ -704,9 +791,10 @@ export class GenerationStore {
|
||||||
}
|
}
|
||||||
await this.storage.writeRawObject(MANIFEST_PATH, manifest)
|
await this.storage.writeRawObject(MANIFEST_PATH, manifest)
|
||||||
await this.storage.syncRawObjects([MANIFEST_PATH])
|
await this.storage.syncRawObjects([MANIFEST_PATH])
|
||||||
prodLog.warn(
|
prodLog.narrate(
|
||||||
`[GenerationStore] log-authority recovery replayed ${replayed} fact(s) into ` +
|
`[GenerationStore] log-authority recovery replayed ${replayed} fact(s) into ` +
|
||||||
`canonical (${foldKind}; committed at ${this.committed}) — an acked write is never lost`
|
`canonical in ${Date.now() - foldStartedAt}ms (${foldKind}; committed at ` +
|
||||||
|
`${this.committed}) — an acked write is never lost`
|
||||||
)
|
)
|
||||||
}
|
}
|
||||||
// A recovery fold re-applied (and the barrier below re-syncs) every
|
// A recovery fold re-applied (and the barrier below re-syncs) every
|
||||||
|
|
@ -717,7 +805,16 @@ export class GenerationStore {
|
||||||
if (uncleanOpen) await this.advanceFoldCheckpointUnlocked()
|
if (uncleanOpen) await this.advanceFoldCheckpointUnlocked()
|
||||||
// The marker is consumed: any session that can write invalidates it
|
// The marker is consumed: any session that can write invalidates it
|
||||||
// at first commit (see the commit paths); a clean close re-writes it.
|
// at first commit (see the commit paths); a clean close re-writes it.
|
||||||
await this.clearCleanShutdownMarker()
|
// A READER NEVER CONSUMES IT. The marker is the writer's own evidence
|
||||||
|
// about the writer's own process — clearing it here exists so that
|
||||||
|
// if THIS session goes on to write and then dies before its next
|
||||||
|
// clean close, the marker's absence correctly reads as unclean. A
|
||||||
|
// reader can never write, so it can never leave the store in a state
|
||||||
|
// its own crash would mis-describe; clearing the marker for it would
|
||||||
|
// only cost the store's actual writer a needless whole-log fold on
|
||||||
|
// its next open, for a generation the reader merely observed. Leave
|
||||||
|
// `_system/` exactly as found.
|
||||||
|
if (!options?.readOnly) await this.clearCleanShutdownMarker()
|
||||||
}
|
}
|
||||||
await this.factLog.open(this.committed)
|
await this.factLog.open(this.committed)
|
||||||
} else {
|
} else {
|
||||||
|
|
@ -731,9 +828,15 @@ export class GenerationStore {
|
||||||
if (storageSupportsFactLog(this.storage)) {
|
if (storageSupportsFactLog(this.storage)) {
|
||||||
this.segments = new GenerationSegmentStore(this.storage)
|
this.segments = new GenerationSegmentStore(this.storage)
|
||||||
await this.segments.open()
|
await this.segments.open()
|
||||||
const packedRanges = this.segments
|
// ACTUAL ranges, not declared ones. A segment sealed by a pre-density-law
|
||||||
.segments()
|
// writer can declare a span wider than the frames it holds; seeding
|
||||||
.map((s): [number, number] => [s.firstGeneration, Math.min(s.lastGeneration, this.committed)])
|
// committedRanges from the declared span re-admits those holes as
|
||||||
|
// committed generations, and every later maintenance pass then asks for a
|
||||||
|
// frame that was never written. `actualRanges()` reads the real
|
||||||
|
// generation list from the sidecar for exactly those segments (and does
|
||||||
|
// no I/O for the dense ones, which is all of them on a healthy store).
|
||||||
|
const packedRanges = (await this.segments.actualRanges())
|
||||||
|
.map((r): [number, number] => [r[0], Math.min(r[1], this.committed)])
|
||||||
.filter(([lo, hi]) => lo <= hi)
|
.filter(([lo, hi]) => lo <= hi)
|
||||||
if (packedRanges.length > 0) {
|
if (packedRanges.length > 0) {
|
||||||
// Merge packed (older) + live (newer) interval sets — both ascending;
|
// Merge packed (older) + live (newer) interval sets — both ascending;
|
||||||
|
|
@ -801,7 +904,11 @@ export class GenerationStore {
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/** Consume the clean-shutdown marker (every open; a clean close re-writes it). */
|
/**
|
||||||
|
* Consume the clean-shutdown marker (every WRITER open; a clean close
|
||||||
|
* re-writes it). Callers must gate this on `!options.readOnly` — a reader
|
||||||
|
* never consumes the marker, see the call site in {@link open}.
|
||||||
|
*/
|
||||||
private async clearCleanShutdownMarker(): Promise<void> {
|
private async clearCleanShutdownMarker(): Promise<void> {
|
||||||
try {
|
try {
|
||||||
await this.storage.deleteRawObject(CLEAN_SHUTDOWN_PATH)
|
await this.storage.deleteRawObject(CLEAN_SHUTDOWN_PATH)
|
||||||
|
|
@ -1263,6 +1370,9 @@ export class GenerationStore {
|
||||||
* @param args.execute - Runs the planned operation batch atomically.
|
* @param args.execute - Runs the planned operation batch atomically.
|
||||||
* @returns The committed generation and its commit timestamp.
|
* @returns The committed generation and its commit timestamp.
|
||||||
* @throws GenerationConflictError when the CAS expectation fails.
|
* @throws GenerationConflictError when the CAS expectation fails.
|
||||||
|
* @throws PendingSingleOpsUnflushedError when the pending single-op tier is
|
||||||
|
* non-empty — call `flushPendingSingleOps()` first (both `Brainy.transact()`
|
||||||
|
* and `Brainy.compactHistory()` already do).
|
||||||
*/
|
*/
|
||||||
/**
|
/**
|
||||||
* The generation fact log, or `null` when the storage layer cannot host one.
|
* The generation fact log, or `null` when the storage layer cannot host one.
|
||||||
|
|
@ -1337,6 +1447,13 @@ export class GenerationStore {
|
||||||
execute: () => Promise<void>
|
execute: () => Promise<void>
|
||||||
}): Promise<{ generation: number; timestamp: number }> {
|
}): Promise<{ generation: number; timestamp: number }> {
|
||||||
return this.withMutex(async () => {
|
return this.withMutex(async () => {
|
||||||
|
// The generation-order guard (see assertPendingSingleOpsFlushed): a
|
||||||
|
// direct commitTransaction() call while single-ops are still pending
|
||||||
|
// would commit above them, unsorting reservedGensAsc() and corrupting
|
||||||
|
// point-in-time reads. Both sanctioned callers (Brainy.transact(),
|
||||||
|
// Brainy.compactHistory()) already flush first, so this is
|
||||||
|
// behavior-neutral on every real path.
|
||||||
|
this.assertPendingSingleOpsFlushed()
|
||||||
// A latched history-durability failure compromises the whole generation
|
// A latched history-durability failure compromises the whole generation
|
||||||
// chain — refuse a transact too (advancing the manifest past stuck,
|
// chain — refuse a transact too (advancing the manifest past stuck,
|
||||||
// un-durable single-op generations would be inconsistent). Same loud
|
// un-durable single-op generations would be inconsistent). Same loud
|
||||||
|
|
@ -2206,6 +2323,37 @@ export class GenerationStore {
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* @description Throw if the pending single-op tier is non-empty. Called at
|
||||||
|
* the top of {@link commitTransaction} (the ONLY method that appends a
|
||||||
|
* fresh commit directly into {@link committedRanges} outside recovery) so
|
||||||
|
* the ordering invariant {@link reservedGensAsc}'s own doc comment states —
|
||||||
|
* "pending generations are always greater than every committed one" — is
|
||||||
|
* ENFORCED there rather than merely assumed.
|
||||||
|
*
|
||||||
|
* That invariant holds today only because both sanctioned callers flush the
|
||||||
|
* pending tier before committing: `Brainy.transact()` (src/brainy.ts,
|
||||||
|
* `await this.generationStore.flushPendingSingleOps()` immediately before
|
||||||
|
* its `commitTransaction()` call) and `Brainy.compactHistory()`
|
||||||
|
* (src/brainy.ts, the same flush immediately before its `compact()` call —
|
||||||
|
* `compact()` itself only ever RECLAIMS an existing committed prefix, so it
|
||||||
|
* cannot land a commit out of order and needs no guard of its own). A
|
||||||
|
* caller that reaches `commitTransaction()` by any other path — bypassing
|
||||||
|
* that flush — would commit a new generation into `committedRanges` ABOVE
|
||||||
|
* generations still sitting in `pendingGens`, breaking `reservedGensAsc`'s
|
||||||
|
* "committed-then-pending is already sorted" assumption and making
|
||||||
|
* `resolveManyAt`'s single ascending pass (and `resolveAt`'s consumers)
|
||||||
|
* return the WRONG before-image for a point-in-time read — silently, no
|
||||||
|
* compensating error. Refusing here, before any staging I/O, keeps the
|
||||||
|
* store untouched (nothing committed, nothing staged, the generation
|
||||||
|
* counter reservation unaffected) on every path that already flushes.
|
||||||
|
*/
|
||||||
|
private assertPendingSingleOpsFlushed(): void {
|
||||||
|
if (this.pendingGens.length > 0) {
|
||||||
|
throw new PendingSingleOpsUnflushedError(this.pendingGens.length)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
/** Schedule a coalesced pending-tier flush (size trigger fires immediately on
|
/** Schedule a coalesced pending-tier flush (size trigger fires immediately on
|
||||||
* the next microtask; otherwise a {@link PENDING_FLUSH_DELAY_MS} timer). Both
|
* the next microtask; otherwise a {@link PENDING_FLUSH_DELAY_MS} timer). Both
|
||||||
* defer outside the current mutex section so the flush can re-acquire it. A
|
* defer outside the current mutex section so the flush can re-acquire it. A
|
||||||
|
|
@ -2289,6 +2437,13 @@ export class GenerationStore {
|
||||||
* committed-then-pending concatenation is already sorted — identical to the old
|
* committed-then-pending concatenation is already sorted — identical to the old
|
||||||
* `[...committedGens, ...pendingGens]`. This is the union historical reads
|
* `[...committedGens, ...pendingGens]`. This is the union historical reads
|
||||||
* resolve over so un-flushed single-ops are visible to pins/`asOf`.
|
* resolve over so un-flushed single-ops are visible to pins/`asOf`.
|
||||||
|
*
|
||||||
|
* The "flush first" half of that invariant is ENFORCED, not just documented:
|
||||||
|
* {@link commitTransaction} — the only method that lands a fresh commit into
|
||||||
|
* {@link committedRanges} outside crash recovery — refuses via
|
||||||
|
* {@link assertPendingSingleOpsFlushed} whenever {@link pendingGens} is
|
||||||
|
* non-empty, so a committed generation can never land above a still-pending
|
||||||
|
* one and break this ordering.
|
||||||
*/
|
*/
|
||||||
private *reservedGensAsc(): IterableIterator<number> {
|
private *reservedGensAsc(): IterableIterator<number> {
|
||||||
yield* this.committedGensAsc()
|
yield* this.committedGensAsc()
|
||||||
|
|
@ -3068,13 +3223,26 @@ export class GenerationStore {
|
||||||
foldInput.push({ generation: gen, timestamp: delta.timestamp, delta, records })
|
foldInput.push({ generation: gen, timestamp: delta.timestamp, delta, records })
|
||||||
}
|
}
|
||||||
if (foldInput.length === 0) continue
|
if (foldInput.length === 0) continue
|
||||||
await segments.fold(foldInput)
|
// SPLIT AT DISCONTINUITIES. `eligible` is NOT contiguous — three
|
||||||
segmentsCreated++
|
// filters above punch holes in it: a generation missing from
|
||||||
// Segment + manifest durable → the live copies retire.
|
// committedRanges never appears, one still in the pending buffer is
|
||||||
for (const g of foldInput) {
|
// skipped, and one whose tx.json will not read is skipped. A sealed
|
||||||
await this.storage.removeRawPrefix(`${GENERATIONS_PREFIX}/${g.generation}`)
|
// segment declares a DENSE range, so folding across such a hole makes
|
||||||
|
// the segment claim a generation it does not hold; the next open
|
||||||
|
// merges that mis-declared range into committedRanges, and every
|
||||||
|
// subsequent auto-compaction pass then asks for the missing frame and
|
||||||
|
// fails with "packed history is damaged". Fold each contiguous RUN as
|
||||||
|
// its own segment instead — same bytes, honest ranges.
|
||||||
|
for (const run of contiguousRuns(foldInput)) {
|
||||||
|
if (deadline !== undefined && Date.now() >= deadline) break
|
||||||
|
await segments.fold(run)
|
||||||
|
segmentsCreated++
|
||||||
|
// Segment + manifest durable → the live copies retire.
|
||||||
|
for (const g of run) {
|
||||||
|
await this.storage.removeRawPrefix(`${GENERATIONS_PREFIX}/${g.generation}`)
|
||||||
|
}
|
||||||
|
folded += run.length
|
||||||
}
|
}
|
||||||
folded += foldInput.length
|
|
||||||
}
|
}
|
||||||
if (folded > 0) {
|
if (folded > 0) {
|
||||||
prodLog.info(
|
prodLog.info(
|
||||||
|
|
|
||||||
|
|
@ -25,7 +25,7 @@
|
||||||
* `docs/ADR-001-generational-mvcc.md` for the full justification.
|
* `docs/ADR-001-generational-mvcc.md` for the full justification.
|
||||||
*/
|
*/
|
||||||
|
|
||||||
import type { AddParams, UpdateParams, RelateParams, UpdateRelationParams, Entity, Relation } from '../types/brainy.types.js'
|
import type { AddParams, UpdateParams, RelateParams, Entity, Relation } from '../types/brainy.types.js'
|
||||||
|
|
||||||
// ============================================================================
|
// ============================================================================
|
||||||
// Transaction operations (brain.transact input)
|
// Transaction operations (brain.transact input)
|
||||||
|
|
@ -75,18 +75,6 @@ export interface TxRelateOperation<T = any> extends RelateParams<T> {
|
||||||
op: 'relate'
|
op: 'relate'
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
|
||||||
* @description Update a relationship. Carries the same parameters as
|
|
||||||
* `brain.updateRelation()` — a first-class batch op (not merely `unrelate`
|
|
||||||
* + `relate`), so a type change re-indexes the SAME relationship id rather
|
|
||||||
* than minting a new one, and a batch containing it is rejected atomically
|
|
||||||
* (like every other op here) if the relationship id does not exist.
|
|
||||||
*/
|
|
||||||
export interface TxUpdateRelationOperation<T = any> extends UpdateRelationParams<T> {
|
|
||||||
/** Discriminator. */
|
|
||||||
op: 'updateRelation'
|
|
||||||
}
|
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* @description Delete a relationship by id (mirror of `brain.unrelate()`).
|
* @description Delete a relationship by id (mirror of `brain.unrelate()`).
|
||||||
*/
|
*/
|
||||||
|
|
@ -108,7 +96,6 @@ export type TxOperation<T = any> =
|
||||||
| TxUpdateOperation<T>
|
| TxUpdateOperation<T>
|
||||||
| TxRemoveOperation
|
| TxRemoveOperation
|
||||||
| TxRelateOperation<T>
|
| TxRelateOperation<T>
|
||||||
| TxUpdateRelationOperation<T>
|
|
||||||
| TxUnrelateOperation
|
| TxUnrelateOperation
|
||||||
|
|
||||||
/**
|
/**
|
||||||
|
|
@ -463,6 +450,21 @@ export interface GenerationStorage {
|
||||||
deleteRawObject(path: string): Promise<void>
|
deleteRawObject(path: string): Promise<void>
|
||||||
/** List raw object paths under a prefix (normalized, `.gz`-stripped). */
|
/** List raw object paths under a prefix (normalized, `.gz`-stripped). */
|
||||||
listRawObjects(prefix: string): Promise<string[]>
|
listRawObjects(prefix: string): Promise<string[]>
|
||||||
|
/**
|
||||||
|
* OPTIONAL: the IMMEDIATE child directory names under a prefix — one level,
|
||||||
|
* no recursion, no file paths.
|
||||||
|
*
|
||||||
|
* Why it exists: discovering which generations are on disk needs only the
|
||||||
|
* top-level directory NAMES under `_generations/`, but the only door for it
|
||||||
|
* was `listRawObjects`, which recurses the whole tree and returns every file
|
||||||
|
* in every generation. On a store with a long history that is a full walk of
|
||||||
|
* the entire generation log, paid on EVERY open, to learn a set of integers
|
||||||
|
* the directory names already spell out.
|
||||||
|
*
|
||||||
|
* An adapter without this door keeps working — the caller falls back to the
|
||||||
|
* recursive listing.
|
||||||
|
*/
|
||||||
|
listRawPrefixes?(prefix: string): Promise<string[]>
|
||||||
/** Remove every object under a prefix (and the directory itself on disk). */
|
/** Remove every object under a prefix (and the directory itself on disk). */
|
||||||
removeRawPrefix(prefix: string): Promise<void>
|
removeRawPrefix(prefix: string): Promise<void>
|
||||||
/** Durability barrier: fsync the given object paths (no-op in memory). */
|
/** Durability barrier: fsync the given object paths (no-op in memory). */
|
||||||
|
|
|
||||||
|
|
@ -128,7 +128,7 @@ async function loadBunAssets(): Promise<ModelAssets> {
|
||||||
}
|
}
|
||||||
|
|
||||||
// Strategy 2: node_modules path relative to CWD (for installed packages)
|
// Strategy 2: node_modules path relative to CWD (for installed packages)
|
||||||
const nmPath = './node_modules/@soulcraft/brainy/assets/models/all-MiniLM-L6-v2'
|
const nmPath = './node_modules/@soulcraftlabs/brainy/assets/models/all-MiniLM-L6-v2'
|
||||||
pathsToTry.push([
|
pathsToTry.push([
|
||||||
`${nmPath}/model.safetensors`,
|
`${nmPath}/model.safetensors`,
|
||||||
`${nmPath}/tokenizer.json`,
|
`${nmPath}/tokenizer.json`,
|
||||||
|
|
@ -168,9 +168,9 @@ async function loadBunAssets(): Promise<ModelAssets> {
|
||||||
// If all strategies fail, provide helpful error message
|
// If all strategies fail, provide helpful error message
|
||||||
throw new Error(
|
throw new Error(
|
||||||
'Could not load model assets. For bun --compile, ensure model files are accessible:\n' +
|
'Could not load model assets. For bun --compile, ensure model files are accessible:\n' +
|
||||||
' Option 1: Keep node_modules/@soulcraft/brainy/assets/ alongside your binary\n' +
|
' Option 1: Keep node_modules/@soulcraftlabs/brainy/assets/ alongside your binary\n' +
|
||||||
' Option 2: Copy assets/ folder to your working directory\n' +
|
' Option 2: Copy assets/ folder to your working directory\n' +
|
||||||
' Option 3: Use --asset flag: bun build --compile --asset="./node_modules/@soulcraft/brainy/assets/**/*"'
|
' Option 3: Use --asset flag: bun build --compile --asset="./node_modules/@soulcraftlabs/brainy/assets/**/*"'
|
||||||
)
|
)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
@ -190,7 +190,7 @@ async function loadNodeAssets(): Promise<ModelAssets> {
|
||||||
if (!fs.existsSync(assetsDir)) {
|
if (!fs.existsSync(assetsDir)) {
|
||||||
throw new Error(
|
throw new Error(
|
||||||
`Model assets not found: ${assetsDir}\n` +
|
`Model assets not found: ${assetsDir}\n` +
|
||||||
`Ensure @soulcraft/brainy is installed correctly.`
|
`Ensure @soulcraftlabs/brainy is installed correctly.`
|
||||||
)
|
)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -18,7 +18,6 @@ export type BrainyErrorType =
|
||||||
| 'PROTECTED_ARTIFACT'
|
| 'PROTECTED_ARTIFACT'
|
||||||
| 'DERIVED_ARTIFACT_MISSING'
|
| 'DERIVED_ARTIFACT_MISSING'
|
||||||
| 'MIGRATION_IN_PROGRESS'
|
| 'MIGRATION_IN_PROGRESS'
|
||||||
| 'PROVIDER_CAPABILITY_MISMATCH'
|
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Custom error class for Brainy operations
|
* Custom error class for Brainy operations
|
||||||
|
|
@ -408,41 +407,71 @@ export class MigrationInProgressError extends BrainyError {
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Thrown at PROVIDER REGISTRATION when an index-provider instance (the
|
* THE INDEXABLE-ARRAY BOUND. An array-valued metadata field indexes one posting
|
||||||
* `'metadataIndex'` or `'graphIndex'` provider a native accelerator
|
* per element, so an unbounded array is an unbounded write — a 384-float
|
||||||
* registers) announces a capability in its `capabilities` set that its own
|
* embedding parked in the metadata bag would mint 384 postings for one row.
|
||||||
* methods do not actually implement — e.g. the set contains `'update-op'`
|
* The bound exists to keep that out of the index.
|
||||||
* but the instance has no `updateIndex`/`updateVerb` function. A provider
|
|
||||||
* must never claim more than it delivers: honoring the announcement would
|
|
||||||
* let brainy emit an update op the provider cannot execute, discovered only
|
|
||||||
* at the first write instead of at startup. This is the TYPED REFUSAL AT
|
|
||||||
* REGISTRATION — loud and immediate, never a silent fallback to the legacy
|
|
||||||
* remove+add pair for a provider that lied about its capabilities.
|
|
||||||
*
|
*
|
||||||
* Raised by `assertUpdateCapabilityCoherent`
|
* 256 is hardcoded on purpose (the zero-config law: no knob). It sits far above
|
||||||
* (`src/transaction/operations/updateCapability.ts`), called once per
|
* every legitimate multi-value field the engine has seen — tags, authors,
|
||||||
* provider at adoption time, before any write can run.
|
* categories, labels, keyword lists, participant lists — and still below the
|
||||||
|
* narrowest embedding this engine will ever meet (384 dimensions, the smallest
|
||||||
|
* model it ships), so the two populations do not overlap and no caller has to
|
||||||
|
* tune it. A vector parked in metadata is refused; a long keyword list is not.
|
||||||
|
*
|
||||||
|
* It replaces a limit of 10 that was applied SILENTLY: a row whose `tags` array
|
||||||
|
* held eleven entries had that field skipped entirely and dropped out of every
|
||||||
|
* filtered search on it, with no error, no warning and no way to tell the
|
||||||
|
* difference from "no row matches". A rule this consequential is a law with a
|
||||||
|
* name and a refusal, not a `continue`.
|
||||||
|
*
|
||||||
|
* This is the ONE place the number lives. Every message, warning, doc line and
|
||||||
|
* pin derives it from here — never a literal.
|
||||||
*/
|
*/
|
||||||
export class ProviderCapabilityMismatchError extends BrainyError {
|
export const MAX_INDEXED_ARRAY_LENGTH = 256
|
||||||
/** Which provider family failed the check. */
|
|
||||||
public readonly family: 'metadata' | 'graph'
|
|
||||||
/** The method the announced capability required but the provider lacks. */
|
|
||||||
public readonly missingMethod: string
|
|
||||||
|
|
||||||
constructor(family: 'metadata' | 'graph', missingMethod: string) {
|
/**
|
||||||
|
* A metadata field carries an array longer than {@link MAX_INDEXED_ARRAY_LENGTH}.
|
||||||
|
*
|
||||||
|
* Thrown at the WRITE door (`add` / `update` / `relate` / `updateRelation`), so
|
||||||
|
* the caller learns at the moment of writing that the field will not be
|
||||||
|
* searchable — rather than discovering it later as rows that quietly fail to
|
||||||
|
* match. Carries the field, its length and the bound so a handler can report
|
||||||
|
* or repair without parsing the message.
|
||||||
|
*
|
||||||
|
* The cure is one of: store the long array outside the indexed bag (`data`
|
||||||
|
* carries arbitrary content and is not indexed element-wise); pass an embedding
|
||||||
|
* as the first-class `vector` parameter, which is where a vector belongs; or
|
||||||
|
* shorten the field to the values that are actually queried.
|
||||||
|
*/
|
||||||
|
export class MetadataArrayTooLargeError extends BrainyError {
|
||||||
|
/** The metadata field whose array is too long (its full dotted address). */
|
||||||
|
public readonly field: string
|
||||||
|
/** How many elements that array holds. */
|
||||||
|
public readonly length: number
|
||||||
|
/** The bound it exceeded — {@link MAX_INDEXED_ARRAY_LENGTH}. */
|
||||||
|
public readonly limit: number
|
||||||
|
|
||||||
|
constructor(site: string, field: string, length: number, limit: number) {
|
||||||
super(
|
super(
|
||||||
`Provider capability mismatch: the '${family}' index provider's ` +
|
`${site}: metadata field '${field}' holds ${length} array elements, ` +
|
||||||
`\`capabilities\` set claims 'update-op' but does not expose a ` +
|
`over the ${limit}-element indexing bound. An array field indexes one ` +
|
||||||
`\`${missingMethod}\` method — a provider must not announce a ` +
|
`posting per element, so an unbounded array is an unbounded write. ` +
|
||||||
`capability it does not implement. Registration refused.`,
|
`This write is refused rather than indexed partially or skipped silently ` +
|
||||||
'PROVIDER_CAPABILITY_MISMATCH',
|
`— a skipped field drops the row out of every filtered search on '${field}' ` +
|
||||||
|
`with no way to tell that from "nothing matched". ` +
|
||||||
|
`Cures: put the long array in 'data' (stored, not indexed element-wise); ` +
|
||||||
|
`pass an embedding as the first-class 'vector' parameter; or keep only ` +
|
||||||
|
`the values you actually query in '${field}'.`,
|
||||||
|
'VALIDATION',
|
||||||
false
|
false
|
||||||
)
|
)
|
||||||
this.name = 'ProviderCapabilityMismatchError'
|
this.name = 'MetadataArrayTooLargeError'
|
||||||
this.family = family
|
this.field = field
|
||||||
this.missingMethod = missingMethod
|
this.length = length
|
||||||
|
this.limit = limit
|
||||||
if (Error.captureStackTrace) {
|
if (Error.captureStackTrace) {
|
||||||
Error.captureStackTrace(this, ProviderCapabilityMismatchError)
|
Error.captureStackTrace(this, MetadataArrayTooLargeError)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
|
||||||
|
|
@ -14,7 +14,7 @@
|
||||||
* - {@link RelationNotFoundError} — a referenced relationship (verb) does
|
* - {@link RelationNotFoundError} — a referenced relationship (verb) does
|
||||||
* not exist.
|
* not exist.
|
||||||
*
|
*
|
||||||
* Both are exported from the package root (`@soulcraft/brainy`).
|
* Both are exported from the package root (`@soulcraftlabs/brainy`).
|
||||||
*/
|
*/
|
||||||
|
|
||||||
/**
|
/**
|
||||||
|
|
|
||||||
|
|
@ -1052,6 +1052,17 @@ export class GraphAdjacencyIndex implements GraphIndexProvider {
|
||||||
*/
|
*/
|
||||||
private startAutoFlush(): void {
|
private startAutoFlush(): void {
|
||||||
this.flushTimer = setInterval(async () => {
|
this.flushTimer = setInterval(async () => {
|
||||||
|
// NO PERIODIC WORK WITHOUT A CAUSE. Ask first, in two O(1) reads: an
|
||||||
|
// index nobody has written to since the last flush has nothing to
|
||||||
|
// write, and calling into the trees (and their logging) on a cadence
|
||||||
|
// over a quiet store is exactly the idle cost this law exists to
|
||||||
|
// remove.
|
||||||
|
if (
|
||||||
|
!this.lsmTreeVerbsBySource.hasPendingWrites() &&
|
||||||
|
!this.lsmTreeVerbsByTarget.hasPendingWrites()
|
||||||
|
) {
|
||||||
|
return
|
||||||
|
}
|
||||||
await this.flush()
|
await this.flush()
|
||||||
}, this.config.flushInterval)
|
}, this.config.flushInterval)
|
||||||
// Background maintenance must never keep the host process alive —
|
// Background maintenance must never keep the host process alive —
|
||||||
|
|
@ -1094,13 +1105,31 @@ export class GraphAdjacencyIndex implements GraphIndexProvider {
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Clean shutdown
|
* Stop the auto-flush interval WITHOUT writing anything.
|
||||||
|
*
|
||||||
|
* The non-writing half of {@link close}, for a shutdown that must leave the
|
||||||
|
* store byte-identical — a read-only brain's close. `close()` itself is a
|
||||||
|
* writer: it drains both LSM MemTables to SSTables and stamps the watermark,
|
||||||
|
* which is exactly right for a writer and forbidden for a reader. A reader
|
||||||
|
* still has to release this interval, though: it is the one piece of this
|
||||||
|
* index that outlives the close and could fire against a store the session no
|
||||||
|
* longer owns.
|
||||||
|
*
|
||||||
|
* @returns Nothing.
|
||||||
*/
|
*/
|
||||||
async close(): Promise<void> {
|
stopBackgroundFlush(): void {
|
||||||
if (this.flushTimer) {
|
if (this.flushTimer) {
|
||||||
clearInterval(this.flushTimer)
|
clearInterval(this.flushTimer)
|
||||||
this.flushTimer = undefined
|
this.flushTimer = undefined
|
||||||
}
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Clean shutdown — drains both trees and stamps the watermark. THIS WRITES;
|
||||||
|
* a read-only brain must call {@link stopBackgroundFlush} instead.
|
||||||
|
*/
|
||||||
|
async close(): Promise<void> {
|
||||||
|
this.stopBackgroundFlush()
|
||||||
|
|
||||||
// Close both LSM-trees (will flush MemTables to SSTables)
|
// Close both LSM-trees (will flush MemTables to SSTables)
|
||||||
if (this.initialized) {
|
if (this.initialized) {
|
||||||
|
|
|
||||||
|
|
@ -687,6 +687,17 @@ export class LSMTree {
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* @description Whether this tree holds anything a flush would write —
|
||||||
|
* the MemTable is non-empty. Synchronous and O(1), so a background cadence
|
||||||
|
* can ask before it does anything at all: the engine does no periodic work
|
||||||
|
* without a cause.
|
||||||
|
* @returns true when a flush would write; false when it would be a no-op.
|
||||||
|
*/
|
||||||
|
hasPendingWrites(): boolean {
|
||||||
|
return !this.memTable.isEmpty()
|
||||||
|
}
|
||||||
|
|
||||||
async close(): Promise<void> {
|
async close(): Promise<void> {
|
||||||
this.stopCompactionTimer()
|
this.stopCompactionTimer()
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -10,7 +10,7 @@ import {
|
||||||
Vector,
|
Vector,
|
||||||
VectorDocument
|
VectorDocument
|
||||||
} from '../coreTypes.js'
|
} from '../coreTypes.js'
|
||||||
import { euclideanDistance, calculateDistancesBatch } from '../utils/index.js'
|
import { euclideanDistance, calculateDistancesBatch, isZeroNormVector } from '../utils/index.js'
|
||||||
import type { BaseStorage } from '../storage/baseStorage.js'
|
import type { BaseStorage } from '../storage/baseStorage.js'
|
||||||
import { getGlobalCache, UnifiedCache } from '../utils/unifiedCache.js'
|
import { getGlobalCache, UnifiedCache } from '../utils/unifiedCache.js'
|
||||||
import { prodLog } from '../utils/logger.js'
|
import { prodLog } from '../utils/logger.js'
|
||||||
|
|
@ -64,6 +64,34 @@ export class HnswFlushError extends Error {
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* @description Thrown by {@link JsHnswVectorIndex.addItem} / {@link
|
||||||
|
* JsHnswVectorIndex.updateItem} when handed a length-0 vector. A length-0
|
||||||
|
* vector is the sanctioned "unvectored" shape for a canonical noun record
|
||||||
|
* (class-J: a VFS-system row, a deferred embed not yet landed, or any other
|
||||||
|
* legitimately-vector-less row) — but it is NEVER a legal INDEX insert. The
|
||||||
|
* index itself has no concept of "unvectored"; deciding that a row is
|
||||||
|
* unvectored and therefore skippable is the FILL/REBUILD/LOAD consumer's job
|
||||||
|
* (see {@link JsHnswVectorIndex.rebuild}), done BEFORE ever calling addItem.
|
||||||
|
* A length-0 vector reaching this point is a caller bug: silently accepting
|
||||||
|
* it would pin `this.dimension = 0` on an empty index (poisoning every real
|
||||||
|
* insert thereafter with a dimension mismatch) or store a vector-less node
|
||||||
|
* that a distance calculation can never safely compare against. Loud errors,
|
||||||
|
* never quiet losses — this throws instead of either.
|
||||||
|
*/
|
||||||
|
export class EmptyVectorIndexError extends Error {
|
||||||
|
constructor(public readonly id: string, operation: 'addItem' | 'updateItem') {
|
||||||
|
super(
|
||||||
|
`${operation}(${id}): refusing to index a length-0 vector — a length-0 vector is the ` +
|
||||||
|
`sanctioned "unvectored" shape for a canonical row, but it is never a legal index ` +
|
||||||
|
`insert. Callers that fill/rebuild/load the index must skip vector.length === 0 rows ` +
|
||||||
|
`themselves (unvectored = nothing to index, not an error at that layer); reaching ` +
|
||||||
|
`here with one is a caller bug.`
|
||||||
|
)
|
||||||
|
this.name = 'EmptyVectorIndexError'
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Implements {@link VectorIndexProvider}: the vector-index surface Brainy calls
|
* Implements {@link VectorIndexProvider}: the vector-index surface Brainy calls
|
||||||
* on whatever the `'vector'` factory returns (its own `JsHnswVectorIndex`, or a native
|
* on whatever the `'vector'` factory returns (its own `JsHnswVectorIndex`, or a native
|
||||||
|
|
@ -580,6 +608,15 @@ export class JsHnswVectorIndex implements VectorIndexProvider {
|
||||||
throw new Error('Vector is undefined or null')
|
throw new Error('Vector is undefined or null')
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// THE INDEX REFUSES A LENGTH-0 VECTOR (see EmptyVectorIndexError's JSDoc):
|
||||||
|
// an empty vector is the sanctioned "unvectored" shape at the canonical
|
||||||
|
// layer, never a legal index member. Refusing here — loudly, before the
|
||||||
|
// dimension pin below — means no future fill/rebuild/load path can ever
|
||||||
|
// poison `this.dimension` to 0 or park a vector-less node in the graph.
|
||||||
|
if (vector.length === 0) {
|
||||||
|
throw new EmptyVectorIndexError(id, 'addItem')
|
||||||
|
}
|
||||||
|
|
||||||
// Set dimension on first insert
|
// Set dimension on first insert
|
||||||
if (this.dimension === null) {
|
if (this.dimension === null) {
|
||||||
this.dimension = vector.length
|
this.dimension = vector.length
|
||||||
|
|
@ -954,6 +991,13 @@ export class JsHnswVectorIndex implements VectorIndexProvider {
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Same refusal as addItem (see EmptyVectorIndexError's JSDoc) — an
|
||||||
|
// in-place relink must never rewrite an already-indexed node down to the
|
||||||
|
// unvectored shape or poison the pinned dimension.
|
||||||
|
if (vector.length === 0) {
|
||||||
|
throw new EmptyVectorIndexError(id, 'updateItem')
|
||||||
|
}
|
||||||
|
|
||||||
if (this.dimension === null) {
|
if (this.dimension === null) {
|
||||||
this.dimension = vector.length
|
this.dimension = vector.length
|
||||||
} else if (vector.length !== this.dimension) {
|
} else if (vector.length !== this.dimension) {
|
||||||
|
|
@ -1555,7 +1599,15 @@ export class JsHnswVectorIndex implements VectorIndexProvider {
|
||||||
}
|
}
|
||||||
|
|
||||||
const loaded = await this.storage.getNounVector(noun.id)
|
const loaded = await this.storage.getNounVector(noun.id)
|
||||||
if (!loaded) {
|
// `loaded` is a length-0 array (not null/undefined) for a canonical row
|
||||||
|
// that is legitimately unvectored — `![]` is FALSE (an empty array is
|
||||||
|
// truthy), so the bare `!loaded` check below would silently accept it
|
||||||
|
// as "found" and hand a dimension-0 vector to a distance calculation.
|
||||||
|
// A node only reaches this lazy-load path because it is a MEMBER of
|
||||||
|
// the live index (rebuild() now refuses to admit unvectored rows — see
|
||||||
|
// its JSDoc), so an empty vector here is never legitimate: treat it
|
||||||
|
// exactly like "not found", loudly.
|
||||||
|
if (!loaded || loaded.length === 0) {
|
||||||
throw new Error(`Vector not found for noun ${noun.id}`)
|
throw new Error(`Vector not found for noun ${noun.id}`)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
@ -1765,9 +1817,56 @@ export class JsHnswVectorIndex implements VectorIndexProvider {
|
||||||
|
|
||||||
totalCount = result.totalCount || result.items.length
|
totalCount = result.totalCount || result.items.length
|
||||||
|
|
||||||
|
// UNVECTORED ROWS ARE NOT AN INDEX MEMBER (the class-J law): a canonical
|
||||||
|
// noun whose vector leg is `[]` (a VFS-root-style system row, a
|
||||||
|
// deferred embed not yet landed, or a best-effort fallback for an
|
||||||
|
// unreadable vector leg) is a normal, enumerable, countable row — it
|
||||||
|
// is simply not indexed. `storage.getVectorIndexData()` derives its
|
||||||
|
// {level, connections} answer straight from the noun's OWN record, so
|
||||||
|
// it returns non-null for every existing noun regardless of whether
|
||||||
|
// that noun ever actually reached `addItem()` — it cannot be used to
|
||||||
|
// decide indexability. `nounData.vector.length === 0` is the one
|
||||||
|
// truthful signal (mirrors the `noun.vector.length > 0` guards in
|
||||||
|
// {@link getVectorSafe} / {@link getVectorSync}): skip here, counted
|
||||||
|
// once in a summary line, never per-row spam.
|
||||||
|
let skippedUnvectored = 0
|
||||||
|
|
||||||
// Process all nouns at once
|
// Process all nouns at once
|
||||||
for (const nounData of result.items) {
|
for (const nounData of result.items) {
|
||||||
try {
|
try {
|
||||||
|
if (!Array.isArray(nounData.vector) || nounData.vector.length === 0) {
|
||||||
|
skippedUnvectored++
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
// THE ZERO-NORM LAW — bulk-rebuild leg: a persisted zero-norm
|
||||||
|
// vector (a pre-10.4.2 row the canonical write has not yet
|
||||||
|
// normalized) must never enter the index either, mirroring the
|
||||||
|
// belt AddToVectorIndexOperation enforces on the live write path.
|
||||||
|
// Only the canonical vector is authoritative here — persisted
|
||||||
|
// HNSW graph metadata (level/connections) can outlive an unvector.
|
||||||
|
if (isZeroNormVector(nounData.vector)) {
|
||||||
|
prodLog.warn(
|
||||||
|
`[HNSW] rebuild(): skipping entity ${nounData.id} — persisted vector is ` +
|
||||||
|
`zero-norm (a zero-norm vector is not a vector and never crosses an ` +
|
||||||
|
`engine boundary)`
|
||||||
|
)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
|
||||||
|
// Restore the pinned dimension from the first real vector this
|
||||||
|
// rebuild loads. `addItem`/`updateItem` only pin `this.dimension`
|
||||||
|
// on a LIVE insert — a fresh rebuild from storage never goes
|
||||||
|
// through either, so without this the pin stays `null` across a
|
||||||
|
// restart. A `null` pin means the very next insert (correct OR
|
||||||
|
// wrong length) silently BECOMES the new pin instead of being
|
||||||
|
// checked against the store's real dimension — the wrong-length
|
||||||
|
// case then fails much later and less clearly, inside a distance
|
||||||
|
// calculation against an already-loaded node, instead of here,
|
||||||
|
// immediately, with a named expected-vs-got mismatch.
|
||||||
|
if (this.dimension === null) {
|
||||||
|
this.dimension = nounData.vector.length
|
||||||
|
}
|
||||||
|
|
||||||
// Load HNSW graph data for this entity
|
// Load HNSW graph data for this entity
|
||||||
const hnswData = await this.storage.getVectorIndexData(nounData.id)
|
const hnswData = await this.storage.getVectorIndexData(nounData.id)
|
||||||
|
|
||||||
|
|
@ -1815,7 +1914,10 @@ export class JsHnswVectorIndex implements VectorIndexProvider {
|
||||||
options.onProgress(loadedCount, totalCount)
|
options.onProgress(loadedCount, totalCount)
|
||||||
}
|
}
|
||||||
|
|
||||||
prodLog.info(`HNSW: Loaded ${loadedCount.toLocaleString()} nodes (${storageType})`)
|
prodLog.info(
|
||||||
|
`HNSW: Loaded ${loadedCount.toLocaleString()} nodes (${storageType})` +
|
||||||
|
(skippedUnvectored > 0 ? ` — ${skippedUnvectored.toLocaleString()} unvectored row(s) skipped` : '')
|
||||||
|
)
|
||||||
}
|
}
|
||||||
|
|
||||||
// Step 5: CRITICAL - Recover entry point if missing)
|
// Step 5: CRITICAL - Recover entry point if missing)
|
||||||
|
|
|
||||||
15
src/index.ts
15
src/index.ts
|
|
@ -184,6 +184,7 @@ export {
|
||||||
|
|
||||||
// Export version utilities
|
// Export version utilities
|
||||||
export { getBrainyVersion } from './utils/version.js'
|
export { getBrainyVersion } from './utils/version.js'
|
||||||
|
export { contractVersion, BRAINY_CONTRACT_VERSION } from './utils/version.js'
|
||||||
|
|
||||||
// Export plugin system
|
// Export plugin system
|
||||||
export type { BrainyPlugin, BrainyPluginContext, StorageAdapterFactory } from './plugin.js'
|
export type { BrainyPlugin, BrainyPluginContext, StorageAdapterFactory } from './plugin.js'
|
||||||
|
|
@ -202,7 +203,7 @@ export { EntityNotFoundError, RelationNotFoundError } from './errors/notFound.js
|
||||||
|
|
||||||
// Base error + typed migration-lock error — thrown by any data-plane call while a
|
// Base error + typed migration-lock error — thrown by any data-plane call while a
|
||||||
// brain runs its one-time 7.x→8.0 upgrade; catch to answer HTTP 503 + Retry-After.
|
// brain runs its one-time 7.x→8.0 upgrade; catch to answer HTTP 503 + Retry-After.
|
||||||
export { BrainyError, MigrationInProgressError, GraphIndexNotReadyError, MetadataIndexNotReadyError, VectorIndexNotReadyError, ProtectedArtifactError, DerivedArtifactMissingError, ProviderCapabilityMismatchError } from './errors/brainyError.js'
|
export { BrainyError, MigrationInProgressError, GraphIndexNotReadyError, MetadataIndexNotReadyError, VectorIndexNotReadyError, ProtectedArtifactError, DerivedArtifactMissingError, MetadataArrayTooLargeError, MAX_INDEXED_ARRAY_LENGTH } from './errors/brainyError.js'
|
||||||
export type { BrainyErrorType } from './errors/brainyError.js'
|
export type { BrainyErrorType } from './errors/brainyError.js'
|
||||||
|
|
||||||
// ============= 8.0 Db API — generational MVCC =============
|
// ============= 8.0 Db API — generational MVCC =============
|
||||||
|
|
@ -230,7 +231,8 @@ export {
|
||||||
GenerationCompactedError,
|
GenerationCompactedError,
|
||||||
StoreInconsistentError,
|
StoreInconsistentError,
|
||||||
PendingFlushDurabilityError,
|
PendingFlushDurabilityError,
|
||||||
CanonicalEnumerationUnavailableError
|
CanonicalEnumerationUnavailableError,
|
||||||
|
PendingSingleOpsUnflushedError
|
||||||
} from './db/errors.js'
|
} from './db/errors.js'
|
||||||
export type { UnreconciledRecord } from './db/errors.js'
|
export type { UnreconciledRecord } from './db/errors.js'
|
||||||
export type {
|
export type {
|
||||||
|
|
@ -269,6 +271,10 @@ export type { FamilyStamp, StampMembers, StampVerdict } from './db/familyStamp.j
|
||||||
export { isVersionedIndexProvider } from './plugin.js'
|
export { isVersionedIndexProvider } from './plugin.js'
|
||||||
export type { VersionedIndexProvider } from './plugin.js'
|
export type { VersionedIndexProvider } from './plugin.js'
|
||||||
export type { ProviderInvariantReport, InvariantResult, InvariantHeal } from './plugin.js'
|
export type { ProviderInvariantReport, InvariantResult, InvariantHeal } from './plugin.js'
|
||||||
|
// The named, synchronous, O(1) health-report contract (the read gate's ONLY
|
||||||
|
// source of truth for "can I serve right now") — see HealthReport's
|
||||||
|
// derivation laws in plugin.ts.
|
||||||
|
export type { HealthReport, LedgerInvariantResult, InvariantSource } from './plugin.js'
|
||||||
// Optional provider self-report of outstanding background maintenance work
|
// Optional provider self-report of outstanding background maintenance work
|
||||||
// (compaction, deferred writes, etc.) — the payload type for
|
// (compaction, deferred writes, etc.) — the payload type for
|
||||||
// brain.maintenanceDebt(). See the measure-only-what-you-track contract on
|
// brain.maintenanceDebt(). See the measure-only-what-you-track contract on
|
||||||
|
|
@ -385,7 +391,10 @@ import type {
|
||||||
HNSWVerb,
|
HNSWVerb,
|
||||||
HNSWConfig,
|
HNSWConfig,
|
||||||
StorageAdapter,
|
StorageAdapter,
|
||||||
DerivedFamilyDeclaration
|
DerivedFamilyDeclaration,
|
||||||
|
// The canonical count ledger a storage adapter maintains (counted + ALL-visibility
|
||||||
|
// scalars per family, the coverage-ledger denominators) — see StorageAdapter.getCanonicalCounts.
|
||||||
|
CanonicalCounts
|
||||||
} from './coreTypes.js'
|
} from './coreTypes.js'
|
||||||
|
|
||||||
// Export vector index implementation (the JS HNSW path)
|
// Export vector index implementation (the JS HNSW path)
|
||||||
|
|
|
||||||
|
|
@ -23,7 +23,10 @@ import type { ColumnStoreProvider, SegmentMeta } from './types.js'
|
||||||
import {
|
import {
|
||||||
ValueType,
|
ValueType,
|
||||||
DEFAULT_FLUSH_THRESHOLD,
|
DEFAULT_FLUSH_THRESHOLD,
|
||||||
FLAG_MULTI_VALUE
|
FLAG_MULTI_VALUE,
|
||||||
|
POSTING_KINDS,
|
||||||
|
KIND_PATH_SEGMENT,
|
||||||
|
type PostingKind
|
||||||
} from './types.js'
|
} from './types.js'
|
||||||
import { ColumnTailBuffer } from './ColumnTailBuffer.js'
|
import { ColumnTailBuffer } from './ColumnTailBuffer.js'
|
||||||
import { ColumnManifest } from './ColumnManifest.js'
|
import { ColumnManifest } from './ColumnManifest.js'
|
||||||
|
|
@ -52,10 +55,89 @@ interface HeapEntry {
|
||||||
value: number | string
|
value: number | string
|
||||||
entityIntId: number
|
entityIntId: number
|
||||||
cursorIndex: number
|
cursorIndex: number
|
||||||
|
/**
|
||||||
|
* Rank of the posting kind this entry came from, from {@link POSTING_KINDS}.
|
||||||
|
* A mixed-kind field has no natural total order, so the merge orders by kind
|
||||||
|
* first and by value within a kind.
|
||||||
|
*/
|
||||||
|
kindRank: number
|
||||||
/** Iterator for the cursor — call next() to advance */
|
/** Iterator for the cursor — call next() to advance */
|
||||||
iterator: Generator<CursorEntry>
|
iterator: Generator<CursorEntry>
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* One physical posting column: a (field, kind) pair and the key every internal
|
||||||
|
* map and every storage path uses for it.
|
||||||
|
*/
|
||||||
|
interface KindColumn {
|
||||||
|
/** The field as the query language names it. */
|
||||||
|
field: string
|
||||||
|
/** The kind of value this column holds. */
|
||||||
|
kind: PostingKind
|
||||||
|
/**
|
||||||
|
* Internal map / storage key. The field's PRIMARY kind uses the bare field
|
||||||
|
* name — the historical layout — and every other kind uses
|
||||||
|
* `<field>/<KIND_PATH_SEGMENT>/<kind>`.
|
||||||
|
*/
|
||||||
|
key: string
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* The KIND a value indexes under — its JavaScript `typeof` class, not its
|
||||||
|
* storage encoding.
|
||||||
|
*
|
||||||
|
* Anything that is not a number, string or boolean indexes as a string, which
|
||||||
|
* is the `String(value)` treatment those values already received. `null` and
|
||||||
|
* `undefined` never reach here: `addEntity` skips them, and their absence is
|
||||||
|
* what the `exists` / `missing` operators read.
|
||||||
|
*
|
||||||
|
* @param value - The value about to be indexed or queried
|
||||||
|
* @returns The posting kind that owns this value
|
||||||
|
*/
|
||||||
|
function kindOfValue(value: unknown): PostingKind {
|
||||||
|
const t = typeof value
|
||||||
|
if (t === 'number') return 'number'
|
||||||
|
if (t === 'boolean') return 'boolean'
|
||||||
|
return 'string'
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* The segment encoding a fresh column of this kind starts with.
|
||||||
|
*
|
||||||
|
* Only the number kind has a choice: an integer column starts as i64 and
|
||||||
|
* widens to f64 the first time a non-integer arrives
|
||||||
|
* ({@link ColumnTailBuffer.promoteToFloat}).
|
||||||
|
*/
|
||||||
|
function initialValueTypeFor(kind: PostingKind, firstValue: unknown): ValueType {
|
||||||
|
switch (kind) {
|
||||||
|
case 'boolean':
|
||||||
|
return ValueType.Boolean
|
||||||
|
case 'string':
|
||||||
|
return ValueType.String
|
||||||
|
case 'number':
|
||||||
|
return Number.isInteger(firstValue) ? ValueType.Number : ValueType.Float
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* The kind a column of this encoding holds — the inverse of
|
||||||
|
* {@link initialValueTypeFor}, used to read a kind back off a manifest written
|
||||||
|
* before typed postings existed.
|
||||||
|
*/
|
||||||
|
function kindOfValueType(valueType: ValueType): PostingKind {
|
||||||
|
switch (valueType) {
|
||||||
|
case ValueType.Boolean:
|
||||||
|
return 'boolean'
|
||||||
|
case ValueType.String:
|
||||||
|
return 'string'
|
||||||
|
case ValueType.Number:
|
||||||
|
case ValueType.Float:
|
||||||
|
return 'number'
|
||||||
|
default:
|
||||||
|
throw new Error(`Unknown ValueType: ${valueType}`)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Unified column store coordinator.
|
* Unified column store coordinator.
|
||||||
*
|
*
|
||||||
|
|
@ -121,9 +203,19 @@ export class ColumnStore implements ColumnStoreProvider {
|
||||||
*/
|
*/
|
||||||
private deletedEntities: Map<string, RoaringBitmap32> = new Map()
|
private deletedEntities: Map<string, RoaringBitmap32> = new Map()
|
||||||
|
|
||||||
/** Known field value types (inferred from first write). */
|
/** Segment encoding per COLUMN key (not per field — a field has one per kind). */
|
||||||
private fieldTypes: Map<string, ValueType> = new Map()
|
private fieldTypes: Map<string, ValueType> = new Map()
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Every posting column a field owns: field → kind → column key.
|
||||||
|
*
|
||||||
|
* This is the map that ends the first-writer type freeze. A field's first
|
||||||
|
* kind takes the bare field name as its column key, keeping the historical
|
||||||
|
* on-disk layout; each later kind takes its own column beside it. Nothing is
|
||||||
|
* coerced across kinds and nothing is dropped for being the wrong type.
|
||||||
|
*/
|
||||||
|
private fieldColumns: Map<string, Map<PostingKind, string>> = new Map()
|
||||||
|
|
||||||
/** Whether init() has completed. */
|
/** Whether init() has completed. */
|
||||||
private initialized = false
|
private initialized = false
|
||||||
|
|
||||||
|
|
@ -140,6 +232,128 @@ export class ColumnStore implements ColumnStoreProvider {
|
||||||
this.l0CompactionTrigger = config?.l0CompactionTrigger ?? 4
|
this.l0CompactionTrigger = config?.l0CompactionTrigger ?? 4
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// =========================================================================
|
||||||
|
// Posting columns: (field, kind) → one physical column
|
||||||
|
// =========================================================================
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Storage / map key for a (field, kind) column.
|
||||||
|
*
|
||||||
|
* `primary` is the kind that owns the bare field name. It is whichever kind
|
||||||
|
* the field saw first, which for an index written before typed postings is
|
||||||
|
* simply the kind of its single manifest — so the historical layout is
|
||||||
|
* preserved rather than migrated.
|
||||||
|
*/
|
||||||
|
private static columnKeyFor(field: string, kind: PostingKind, primary: PostingKind | null): string {
|
||||||
|
return primary === null || kind === primary
|
||||||
|
? field
|
||||||
|
: `${field}/${KIND_PATH_SEGMENT}/${kind}`
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Split a discovered manifest path back into its (field, kind) column, or
|
||||||
|
* `null` when the path names a field's primary column rather than a kind
|
||||||
|
* column. `<field>/k/<kind>` is the only shape that reads as a kind column,
|
||||||
|
* and only for a `<kind>` this version knows.
|
||||||
|
*/
|
||||||
|
private static parseKindColumnKey(key: string): { field: string; kind: PostingKind } | null {
|
||||||
|
const marker = `/${KIND_PATH_SEGMENT}/`
|
||||||
|
const at = key.lastIndexOf(marker)
|
||||||
|
if (at <= 0) return null
|
||||||
|
const kind = key.slice(at + marker.length)
|
||||||
|
if (!POSTING_KINDS.includes(kind as PostingKind)) return null
|
||||||
|
return { field: key.slice(0, at), kind: kind as PostingKind }
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Record a discovered or freshly created column against its field. */
|
||||||
|
private registerColumn(field: string, kind: PostingKind, key: string): void {
|
||||||
|
let byKind = this.fieldColumns.get(field)
|
||||||
|
if (!byKind) {
|
||||||
|
byKind = new Map()
|
||||||
|
this.fieldColumns.set(field, byKind)
|
||||||
|
}
|
||||||
|
const existing = byKind.get(kind)
|
||||||
|
if (existing !== undefined && existing !== key) {
|
||||||
|
// Two columns claiming one (field, kind) means the layout on disk is not
|
||||||
|
// one this writer could have produced. Serving it would silently answer
|
||||||
|
// from half the postings, so say which two and stop.
|
||||||
|
throw new Error(
|
||||||
|
`ColumnStore: field '${field}' has two '${kind}' posting columns on ` +
|
||||||
|
`disk ('${existing}' and '${key}'). The column index layout is ` +
|
||||||
|
`inconsistent — rebuild/repair the metadata index rather than ` +
|
||||||
|
`serving from one half of it.`
|
||||||
|
)
|
||||||
|
}
|
||||||
|
byKind.set(kind, key)
|
||||||
|
}
|
||||||
|
|
||||||
|
/** The column key for this (field, kind), or `null` if the field has no such kind. */
|
||||||
|
private columnKey(field: string, kind: PostingKind): string | null {
|
||||||
|
return this.fieldColumns.get(field)?.get(kind) ?? null
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* The column key for this (field, kind), creating the registration if the
|
||||||
|
* field has not seen this kind before. Write path only.
|
||||||
|
*/
|
||||||
|
private ensureColumnKey(field: string, kind: PostingKind): string {
|
||||||
|
const byKind = this.fieldColumns.get(field)
|
||||||
|
const existing = byKind?.get(kind)
|
||||||
|
if (existing !== undefined) return existing
|
||||||
|
|
||||||
|
// The primary kind is the one already holding the bare field name, if any.
|
||||||
|
let primary: PostingKind | null = null
|
||||||
|
if (byKind) {
|
||||||
|
for (const [k, key] of byKind) {
|
||||||
|
if (key === field) { primary = k; break }
|
||||||
|
}
|
||||||
|
}
|
||||||
|
const key = ColumnStore.columnKeyFor(field, kind, primary)
|
||||||
|
this.registerColumn(field, kind, key)
|
||||||
|
return key
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Every posting column this field owns, in {@link POSTING_KINDS} order.
|
||||||
|
*
|
||||||
|
* Read doors that are not about one particular value — an unbounded range
|
||||||
|
* used as an "any value present" probe, distinct values, sorting — fan out
|
||||||
|
* over all of them.
|
||||||
|
*/
|
||||||
|
private columnsForField(field: string): KindColumn[] {
|
||||||
|
const byKind = this.fieldColumns.get(field)
|
||||||
|
if (!byKind) return []
|
||||||
|
const out: KindColumn[] = []
|
||||||
|
for (const kind of POSTING_KINDS) {
|
||||||
|
const key = byKind.get(kind)
|
||||||
|
if (key !== undefined) out.push({ field, kind, key })
|
||||||
|
}
|
||||||
|
return out
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Which value kinds this field actually holds, in {@link POSTING_KINDS}
|
||||||
|
* order — the honest answer to "what type is this field?".
|
||||||
|
*
|
||||||
|
* A field that carries both `'electronics'` and `5` reports
|
||||||
|
* `['number', 'string']`, not whichever of them was written first.
|
||||||
|
*
|
||||||
|
* @param field - Field name
|
||||||
|
* @returns Every kind with at least one posting, or `[]` for an unknown field
|
||||||
|
*/
|
||||||
|
getFieldKinds(field: string): PostingKind[] {
|
||||||
|
return this.columnsForField(field)
|
||||||
|
.filter((c) => this.columnHasData(c.key))
|
||||||
|
.map((c) => c.kind)
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Does this physical column hold any postings (persisted or buffered)? */
|
||||||
|
private columnHasData(key: string): boolean {
|
||||||
|
const manifest = this.manifests.get(key)
|
||||||
|
const buffer = this.tailBuffers.get(key)
|
||||||
|
return (manifest !== undefined && !manifest.isEmpty()) || (buffer !== undefined && buffer.size > 0)
|
||||||
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Initialize the column store: discover existing field manifests.
|
* Initialize the column store: discover existing field manifests.
|
||||||
*/
|
*/
|
||||||
|
|
@ -157,11 +371,23 @@ export class ColumnStore implements ColumnStoreProvider {
|
||||||
}).listObjectsUnderPath(this.basePath + '/')
|
}).listObjectsUnderPath(this.basePath + '/')
|
||||||
for (const path of paths) {
|
for (const path of paths) {
|
||||||
if (path.endsWith('/MANIFEST.json')) {
|
if (path.endsWith('/MANIFEST.json')) {
|
||||||
const fieldName = path.replace(this.basePath + '/', '').replace('/MANIFEST.json', '')
|
// The discovered name is a COLUMN key: either a bare field (that
|
||||||
const manifest = new ColumnManifest(fieldName, this.basePath)
|
// field's primary kind, which is every column an index written
|
||||||
|
// before typed postings has) or `<field>/k/<kind>` for a second
|
||||||
|
// kind that arrived on a field later.
|
||||||
|
const columnKey = path.replace(this.basePath + '/', '').replace('/MANIFEST.json', '')
|
||||||
|
const manifest = new ColumnManifest(columnKey, this.basePath)
|
||||||
await manifest.load(storage)
|
await manifest.load(storage)
|
||||||
this.manifests.set(fieldName, manifest)
|
this.manifests.set(columnKey, manifest)
|
||||||
this.fieldTypes.set(fieldName, manifest.valueType)
|
this.fieldTypes.set(columnKey, manifest.valueType)
|
||||||
|
|
||||||
|
const parsed = ColumnStore.parseKindColumnKey(columnKey)
|
||||||
|
if (parsed) {
|
||||||
|
this.registerColumn(parsed.field, parsed.kind, columnKey)
|
||||||
|
} else {
|
||||||
|
this.registerColumn(columnKey, kindOfValueType(manifest.valueType), columnKey)
|
||||||
|
}
|
||||||
|
const fieldName = columnKey
|
||||||
|
|
||||||
// Load global deleted bitmap if it exists. Raw blob preferred
|
// Load global deleted bitmap if it exists. Raw blob preferred
|
||||||
// (2.4.0 #4 cortex-shared format); legacy envelope fallback for
|
// (2.4.0 #4 cortex-shared format); legacy envelope fallback for
|
||||||
|
|
@ -264,26 +490,43 @@ export class ColumnStore implements ColumnStoreProvider {
|
||||||
/**
|
/**
|
||||||
* Point filter: find entities where field equals value.
|
* Point filter: find entities where field equals value.
|
||||||
*
|
*
|
||||||
* Searches all segments + tail buffer, returns union as roaring bitmap.
|
* The QUERY VALUE'S OWN KIND picks the posting column, and only that column
|
||||||
* Excludes globally deleted entities.
|
* is read. `where {category: 5}` answers from the number postings and
|
||||||
|
* `where {category: '5'}` from the string postings — neither borrows the
|
||||||
|
* other's rows, because a row written with the number `5` is not a row whose
|
||||||
|
* category is the text `'5'`.
|
||||||
|
*
|
||||||
|
* A field that has never seen this kind matches nothing, which is the true
|
||||||
|
* answer rather than a coerced one.
|
||||||
|
*
|
||||||
|
* Searches all segments + tail buffer of that column, returns the union as a
|
||||||
|
* roaring bitmap. Excludes globally deleted entities.
|
||||||
*/
|
*/
|
||||||
async filter(field: string, value: unknown): Promise<RoaringBitmap32> {
|
async filter(field: string, value: unknown): Promise<RoaringBitmap32> {
|
||||||
const result = new RoaringBitmap32()
|
const result = new RoaringBitmap32()
|
||||||
const deleted = this.deletedEntities.get(field)
|
const columnKey = this.columnKey(field, kindOfValue(value))
|
||||||
|
if (columnKey === null) return result
|
||||||
|
|
||||||
|
// The query value takes the column's encoding — a boolean queried against
|
||||||
|
// a boolean column has to become the 1/0 the column stores.
|
||||||
|
const encoded = this.normalizeValue(value, this.fieldTypes.get(columnKey) ?? ValueType.String)
|
||||||
|
if (encoded === undefined) return result
|
||||||
|
|
||||||
|
const deleted = this.deletedEntities.get(columnKey)
|
||||||
|
|
||||||
// Search segments
|
// Search segments
|
||||||
const cursors = await this.getSegmentCursors(field)
|
const cursors = await this.getSegmentCursors(columnKey)
|
||||||
for (const cursor of cursors) {
|
for (const cursor of cursors) {
|
||||||
const ids = cursor.getEntityIdsForValue(value as number | string)
|
const ids = cursor.getEntityIdsForValue(encoded)
|
||||||
for (const id of ids) {
|
for (const id of ids) {
|
||||||
if (!deleted || !deleted.has(id)) result.add(id)
|
if (!deleted || !deleted.has(id)) result.add(id)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// Search tail buffer
|
// Search tail buffer
|
||||||
const tailCursor = this.getTailBufferCursor(field)
|
const tailCursor = this.getTailBufferCursor(columnKey)
|
||||||
if (tailCursor) {
|
if (tailCursor) {
|
||||||
const ids = tailCursor.getEntityIdsForValue(value as number | string)
|
const ids = tailCursor.getEntityIdsForValue(encoded)
|
||||||
for (const id of ids) {
|
for (const id of ids) {
|
||||||
if (!deleted || !deleted.has(id)) result.add(id)
|
if (!deleted || !deleted.has(id)) result.add(id)
|
||||||
}
|
}
|
||||||
|
|
@ -292,6 +535,62 @@ export class ColumnStore implements ColumnStoreProvider {
|
||||||
return result
|
return result
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Read this column's value for each of `entityIntIds` — the per-id read
|
||||||
|
* behind `find({ fields })`.
|
||||||
|
*
|
||||||
|
* Every other read door here answers "which entities have this value". A
|
||||||
|
* projection asks the opposite — "what value does this entity have" — and
|
||||||
|
* without it a projection has to go to the canonical record for a field the
|
||||||
|
* column is already holding.
|
||||||
|
*
|
||||||
|
* The column is walked ONCE and the wanted ids are picked out as they pass,
|
||||||
|
* so the cost is O(column) per field rather than O(ids x column). Later
|
||||||
|
* sources win: the tail buffer holds writes newer than any segment, and
|
||||||
|
* within the segments a later one supersedes an earlier, exactly as `filter`
|
||||||
|
* treats them.
|
||||||
|
*
|
||||||
|
* Values are EXACT — this store keeps raw values, not the bucketed form the
|
||||||
|
* sparse index uses for range queries — which is what makes it safe to
|
||||||
|
* project from. Deleted entities are skipped; an id with no value in this
|
||||||
|
* column is simply absent from the result.
|
||||||
|
*
|
||||||
|
* @param field - Field name to read.
|
||||||
|
* @param entityIntIds - Entity integer ids to read values for.
|
||||||
|
* @returns `entityIntId -> value` for the ids this column holds.
|
||||||
|
*/
|
||||||
|
async valuesForIds(
|
||||||
|
field: string,
|
||||||
|
entityIntIds: Iterable<number>
|
||||||
|
): Promise<Map<number, number | string>> {
|
||||||
|
const wanted = new Set<number>(entityIntIds)
|
||||||
|
const out = new Map<number, number | string>()
|
||||||
|
if (wanted.size === 0 || !this.hasField(field)) return out
|
||||||
|
|
||||||
|
// Every kind the field holds is read, in POSTING_KINDS order — a value an
|
||||||
|
// entity wrote as a string is still that entity's value for this field.
|
||||||
|
for (const column of this.columnsForField(field)) {
|
||||||
|
const deleted = this.deletedEntities.get(column.key)
|
||||||
|
const take = (entry: { value: number | string; entityIntId: number }): void => {
|
||||||
|
if (!wanted.has(entry.entityIntId)) return
|
||||||
|
if (deleted && deleted.has(entry.entityIntId)) return
|
||||||
|
out.set(entry.entityIntId, entry.value)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Segments oldest -> newest, then the tail: a later write overwrites an
|
||||||
|
// earlier one for the same id.
|
||||||
|
const cursors = await this.getSegmentCursors(column.key)
|
||||||
|
for (const cursor of cursors) {
|
||||||
|
for (const entry of cursor.iterateForward()) take(entry)
|
||||||
|
}
|
||||||
|
const tailCursor = this.getTailBufferCursor(column.key)
|
||||||
|
if (tailCursor) {
|
||||||
|
for (const entry of tailCursor.iterateForward()) take(entry)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return out
|
||||||
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Range filter: find entities where field is within the bounds.
|
* Range filter: find entities where field is within the bounds.
|
||||||
*
|
*
|
||||||
|
|
@ -311,41 +610,59 @@ export class ColumnStore implements ColumnStoreProvider {
|
||||||
includeMax: boolean = true
|
includeMax: boolean = true
|
||||||
): Promise<RoaringBitmap32> {
|
): Promise<RoaringBitmap32> {
|
||||||
const result = new RoaringBitmap32()
|
const result = new RoaringBitmap32()
|
||||||
const cursors = await this.getSegmentCursors(field)
|
|
||||||
|
|
||||||
const hasMin = min !== undefined && min !== null
|
const hasMin = min !== undefined && min !== null
|
||||||
const hasMax = max !== undefined && max !== null
|
const hasMax = max !== undefined && max !== null
|
||||||
|
|
||||||
for (const cursor of cursors) {
|
// The BOUNDS pick the column: numeric bounds read the numeric postings,
|
||||||
const lo = hasMin ? min as number | string : cursor.minValue
|
// string bounds the string postings. An unbounded call is not a range at
|
||||||
const hi = hasMax ? max as number | string : cursor.maxValue
|
// all — it is the "has any value here" probe behind `exists` — so it fans
|
||||||
if (lo === undefined || hi === undefined) continue
|
// out over every kind the field holds.
|
||||||
// Exclusivity applies only to an explicitly provided bound. A bound taken
|
const columns: KindColumn[] = hasMin
|
||||||
// from the segment's own min/max is a real stored value and must stay
|
? this.columnsForKind(field, kindOfValue(min))
|
||||||
// inclusive, or the segment's boundary entities would be wrongly dropped.
|
: hasMax
|
||||||
const ids = cursor.getEntityIdsInRange(
|
? this.columnsForKind(field, kindOfValue(max))
|
||||||
lo,
|
: this.columnsForField(field)
|
||||||
hi,
|
|
||||||
hasMin ? includeMin : true,
|
|
||||||
hasMax ? includeMax : true
|
|
||||||
)
|
|
||||||
for (const id of ids) result.add(id)
|
|
||||||
}
|
|
||||||
|
|
||||||
// Tail buffer range: linear scan (tail is small)
|
for (const column of columns) {
|
||||||
const tailCursor = this.getTailBufferCursor(field)
|
const cursors = await this.getSegmentCursors(column.key)
|
||||||
if (tailCursor) {
|
for (const cursor of cursors) {
|
||||||
for (const entry of tailCursor.iterateForward()) {
|
const lo = hasMin ? min as number | string : cursor.minValue
|
||||||
const v = entry.value as any
|
const hi = hasMax ? max as number | string : cursor.maxValue
|
||||||
const loOk = !hasMin || (includeMin ? v >= (min as any) : v > (min as any))
|
if (lo === undefined || hi === undefined) continue
|
||||||
const hiOk = !hasMax || (includeMax ? v <= (max as any) : v < (max as any))
|
// Exclusivity applies only to an explicitly provided bound. A bound taken
|
||||||
if (loOk && hiOk) result.add(entry.entityIntId)
|
// from the segment's own min/max is a real stored value and must stay
|
||||||
|
// inclusive, or the segment's boundary entities would be wrongly dropped.
|
||||||
|
const ids = cursor.getEntityIdsInRange(
|
||||||
|
lo,
|
||||||
|
hi,
|
||||||
|
hasMin ? includeMin : true,
|
||||||
|
hasMax ? includeMax : true
|
||||||
|
)
|
||||||
|
for (const id of ids) result.add(id)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Tail buffer range: linear scan (tail is small)
|
||||||
|
const tailCursor = this.getTailBufferCursor(column.key)
|
||||||
|
if (tailCursor) {
|
||||||
|
for (const entry of tailCursor.iterateForward()) {
|
||||||
|
const v = entry.value as any
|
||||||
|
const loOk = !hasMin || (includeMin ? v >= (min as any) : v > (min as any))
|
||||||
|
const hiOk = !hasMax || (includeMax ? v <= (max as any) : v < (max as any))
|
||||||
|
if (loOk && hiOk) result.add(entry.entityIntId)
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
return result
|
return result
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/** The single column for this (field, kind), as a list, or empty if absent. */
|
||||||
|
private columnsForKind(field: string, kind: PostingKind): KindColumn[] {
|
||||||
|
const key = this.columnKey(field, kind)
|
||||||
|
return key === null ? [] : [{ field, kind, key }]
|
||||||
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Sort top-K: return K entity int IDs in sorted order (u64-safe BigInt).
|
* Sort top-K: return K entity int IDs in sorted order (u64-safe BigInt).
|
||||||
*
|
*
|
||||||
|
|
@ -376,18 +693,21 @@ export class ColumnStore implements ColumnStoreProvider {
|
||||||
*/
|
*/
|
||||||
async getFilterValues(field: string): Promise<string[]> {
|
async getFilterValues(field: string): Promise<string[]> {
|
||||||
const valueSet = new Set<string>()
|
const valueSet = new Set<string>()
|
||||||
const cursors = await this.getSegmentCursors(field)
|
|
||||||
|
|
||||||
for (const cursor of cursors) {
|
for (const column of this.columnsForField(field)) {
|
||||||
for (const entry of cursor.iterateForward()) {
|
const cursors = await this.getSegmentCursors(column.key)
|
||||||
valueSet.add(String(entry.value))
|
|
||||||
|
for (const cursor of cursors) {
|
||||||
|
for (const entry of cursor.iterateForward()) {
|
||||||
|
valueSet.add(String(entry.value))
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
|
||||||
|
|
||||||
const tailCursor = this.getTailBufferCursor(field)
|
const tailCursor = this.getTailBufferCursor(column.key)
|
||||||
if (tailCursor) {
|
if (tailCursor) {
|
||||||
for (const entry of tailCursor.iterateForward()) {
|
for (const entry of tailCursor.iterateForward()) {
|
||||||
valueSet.add(String(entry.value))
|
valueSet.add(String(entry.value))
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
@ -398,9 +718,7 @@ export class ColumnStore implements ColumnStoreProvider {
|
||||||
* Check if a field has any indexed data.
|
* Check if a field has any indexed data.
|
||||||
*/
|
*/
|
||||||
hasField(field: string): boolean {
|
hasField(field: string): boolean {
|
||||||
const manifest = this.manifests.get(field)
|
return this.columnsForField(field).some((c) => this.columnHasData(c.key))
|
||||||
const buffer = this.tailBuffers.get(field)
|
|
||||||
return (manifest !== undefined && !manifest.isEmpty()) || (buffer !== undefined && buffer.size > 0)
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
|
|
@ -410,12 +728,11 @@ export class ColumnStore implements ColumnStoreProvider {
|
||||||
* store will actually serve queries from.
|
* store will actually serve queries from.
|
||||||
*/
|
*/
|
||||||
getIndexedFields(): string[] {
|
getIndexedFields(): string[] {
|
||||||
|
// Names FIELDS, not columns: a field carrying two kinds is one name here,
|
||||||
|
// the same name a caller queries with.
|
||||||
const fields = new Set<string>()
|
const fields = new Set<string>()
|
||||||
for (const [field, manifest] of this.manifests) {
|
for (const [field] of this.fieldColumns) {
|
||||||
if (!manifest.isEmpty()) fields.add(field)
|
if (this.hasField(field)) fields.add(field)
|
||||||
}
|
|
||||||
for (const [field, buffer] of this.tailBuffers) {
|
|
||||||
if (buffer.size > 0) fields.add(field)
|
|
||||||
}
|
}
|
||||||
return Array.from(fields).sort()
|
return Array.from(fields).sort()
|
||||||
}
|
}
|
||||||
|
|
@ -430,12 +747,16 @@ export class ColumnStore implements ColumnStoreProvider {
|
||||||
getFieldSizeSummary(): Array<{ field: string; segmentCount: number; tailSize: number }> {
|
getFieldSizeSummary(): Array<{ field: string; segmentCount: number; tailSize: number }> {
|
||||||
const summary: Array<{ field: string; segmentCount: number; tailSize: number }> = []
|
const summary: Array<{ field: string; segmentCount: number; tailSize: number }> = []
|
||||||
for (const field of this.getIndexedFields()) {
|
for (const field of this.getIndexedFields()) {
|
||||||
const manifest = this.manifests.get(field)
|
// Summed across the field's kind columns — the caller asked about a
|
||||||
const buffer = this.tailBuffers.get(field)
|
// field, and a field's size is all of the postings under its name.
|
||||||
const segmentCount = manifest && !manifest.isEmpty()
|
let segmentCount = 0
|
||||||
? manifest.getAllSegments().length
|
let tailSize = 0
|
||||||
: 0
|
for (const column of this.columnsForField(field)) {
|
||||||
const tailSize = buffer ? buffer.size : 0
|
const manifest = this.manifests.get(column.key)
|
||||||
|
const buffer = this.tailBuffers.get(column.key)
|
||||||
|
if (manifest && !manifest.isEmpty()) segmentCount += manifest.getAllSegments().length
|
||||||
|
if (buffer) tailSize += buffer.size
|
||||||
|
}
|
||||||
summary.push({ field, segmentCount, tailSize })
|
summary.push({ field, segmentCount, tailSize })
|
||||||
}
|
}
|
||||||
return summary
|
return summary
|
||||||
|
|
@ -463,6 +784,8 @@ export class ColumnStore implements ColumnStoreProvider {
|
||||||
this.segmentCache.clear()
|
this.segmentCache.clear()
|
||||||
this.manifests.clear()
|
this.manifests.clear()
|
||||||
this.deletedEntities.clear()
|
this.deletedEntities.clear()
|
||||||
|
this.fieldColumns.clear()
|
||||||
|
this.fieldTypes.clear()
|
||||||
this.initialized = false
|
this.initialized = false
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
@ -471,32 +794,64 @@ export class ColumnStore implements ColumnStoreProvider {
|
||||||
// =========================================================================
|
// =========================================================================
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Push a single value to a field's tail buffer.
|
* Push a single value to the posting column for its (field, KIND).
|
||||||
* Creates the buffer and manifest if first write to this field.
|
*
|
||||||
* Infers ValueType from the first value seen.
|
* The value's own kind picks the column — a string goes to the field's
|
||||||
|
* string postings, a number to its number postings — so a field carrying
|
||||||
|
* `'electronics'` and `5` keeps both, each answerable by an equality filter
|
||||||
|
* of its own kind. Under the first-writer type freeze this method replaced,
|
||||||
|
* the first value's type became the field's type and every later value of
|
||||||
|
* another kind was coerced to it or, when coercion failed, dropped with no
|
||||||
|
* error at all.
|
||||||
|
*
|
||||||
|
* Creates the column's buffer and manifest on its first value.
|
||||||
*/
|
*/
|
||||||
private pushToBuffer(field: string, value: unknown, entityIntId: number, isMultiValue: boolean): void {
|
private pushToBuffer(field: string, value: unknown, entityIntId: number, isMultiValue: boolean): void {
|
||||||
let buffer = this.tailBuffers.get(field)
|
const kind = kindOfValue(value)
|
||||||
|
const columnKey = this.ensureColumnKey(field, kind)
|
||||||
|
|
||||||
|
let buffer = this.tailBuffers.get(columnKey)
|
||||||
if (!buffer) {
|
if (!buffer) {
|
||||||
const valueType = this.inferValueType(value)
|
// A reopened column takes its encoding from its manifest — an integer
|
||||||
buffer = new ColumnTailBuffer(field, valueType, this.flushThreshold)
|
// column that widened to f64 in an earlier session stays widened.
|
||||||
this.tailBuffers.set(field, buffer)
|
const valueType =
|
||||||
this.fieldTypes.set(field, valueType)
|
this.manifests.get(columnKey)?.valueType ?? initialValueTypeFor(kind, value)
|
||||||
|
buffer = new ColumnTailBuffer(columnKey, valueType, this.flushThreshold)
|
||||||
|
this.tailBuffers.set(columnKey, buffer)
|
||||||
|
this.fieldTypes.set(columnKey, valueType)
|
||||||
|
|
||||||
// Ensure manifest exists
|
// Ensure manifest exists
|
||||||
if (!this.manifests.has(field)) {
|
if (!this.manifests.has(columnKey)) {
|
||||||
const manifest = new ColumnManifest(field, this.basePath)
|
const manifest = new ColumnManifest(columnKey, this.basePath)
|
||||||
manifest.valueType = valueType
|
manifest.valueType = valueType
|
||||||
manifest.multiValue = isMultiValue
|
manifest.multiValue = isMultiValue
|
||||||
this.manifests.set(field, manifest)
|
this.manifests.set(columnKey, manifest)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// Normalize value to the column type
|
// An integer column widens the first time a non-integer number arrives, so
|
||||||
const normalizedValue = this.normalizeValue(value, buffer.valueType)
|
// the value is stored as itself instead of rounded to the nearest integer.
|
||||||
if (normalizedValue !== undefined) {
|
if (kind === 'number' && buffer.valueType === ValueType.Number && !Number.isInteger(value)) {
|
||||||
buffer.add(normalizedValue, entityIntId)
|
buffer.promoteToFloat()
|
||||||
|
this.fieldTypes.set(columnKey, ValueType.Float)
|
||||||
|
const manifest = this.manifests.get(columnKey)
|
||||||
|
if (manifest) manifest.valueType = ValueType.Float
|
||||||
}
|
}
|
||||||
|
|
||||||
|
const normalizedValue = this.normalizeValue(value, buffer.valueType)
|
||||||
|
if (normalizedValue === undefined) {
|
||||||
|
// Unreachable by construction: the column was chosen BY this value's
|
||||||
|
// kind, so the encoding always accepts it. Reaching here would mean a
|
||||||
|
// value had been silently dropped from the index — the exact failure
|
||||||
|
// typed postings exist to end — so it is an error, never a skip.
|
||||||
|
throw new Error(
|
||||||
|
`ColumnStore: field '${field}' rejected a ${kind} value for its own ` +
|
||||||
|
`${ValueType[buffer.valueType]} posting column. The value would have ` +
|
||||||
|
`been dropped from the index while the row stayed readable by id — ` +
|
||||||
|
`this is a kind-routing bug, not a value the caller may ignore.`
|
||||||
|
)
|
||||||
|
}
|
||||||
|
buffer.add(normalizedValue, entityIntId)
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
|
|
@ -625,8 +980,15 @@ export class ColumnStore implements ColumnStoreProvider {
|
||||||
/** Torn-segment quarantine entries for a field (observability + heal input). */
|
/** Torn-segment quarantine entries for a field (observability + heal input). */
|
||||||
quarantinedSegments(field: string): Array<{ segment: string; error: string; hits: number }> {
|
quarantinedSegments(field: string): Array<{ segment: string; error: string; hits: number }> {
|
||||||
const out: Array<{ segment: string; error: string; hits: number }> = []
|
const out: Array<{ segment: string; error: string; hits: number }> = []
|
||||||
for (const [key, q] of this.segmentQuarantine) {
|
// Across every kind column of the field — a torn segment in the string
|
||||||
if (key.startsWith(`${field}:`)) out.push({ segment: key.slice(field.length + 1), error: q.error, hits: q.hits })
|
// postings is this field's torn segment as much as one in the numbers.
|
||||||
|
for (const column of this.columnsForField(field)) {
|
||||||
|
const prefix = `${column.key}:`
|
||||||
|
for (const [key, q] of this.segmentQuarantine) {
|
||||||
|
if (key.startsWith(prefix)) {
|
||||||
|
out.push({ segment: key.slice(prefix.length), error: q.error, hits: q.hits })
|
||||||
|
}
|
||||||
|
}
|
||||||
}
|
}
|
||||||
return out
|
return out
|
||||||
}
|
}
|
||||||
|
|
@ -798,17 +1160,22 @@ export class ColumnStore implements ColumnStoreProvider {
|
||||||
k: number,
|
k: number,
|
||||||
filterBitmap: RoaringBitmap32 | null
|
filterBitmap: RoaringBitmap32 | null
|
||||||
): Promise<number[]> {
|
): Promise<number[]> {
|
||||||
// Collect all cursors (segments + tail buffer)
|
// Collect cursors across EVERY kind the field holds. A single-kind field —
|
||||||
const segCursors = await this.getSegmentCursors(field)
|
// nearly all of them — merges exactly the cursors it always did.
|
||||||
const tailCursor = this.getTailBufferCursor(field)
|
|
||||||
|
|
||||||
// Create iterators for each cursor in the specified direction
|
|
||||||
const iterators: Generator<CursorEntry>[] = []
|
const iterators: Generator<CursorEntry>[] = []
|
||||||
for (const cursor of segCursors) {
|
const iteratorKindRank: number[] = []
|
||||||
iterators.push(order === 'asc' ? cursor.iterateForward() : cursor.iterateBackward())
|
for (const column of this.columnsForField(field)) {
|
||||||
}
|
const kindRank = POSTING_KINDS.indexOf(column.kind)
|
||||||
if (tailCursor) {
|
const segCursors = await this.getSegmentCursors(column.key)
|
||||||
iterators.push(order === 'asc' ? tailCursor.iterateForward() : tailCursor.iterateBackward())
|
for (const cursor of segCursors) {
|
||||||
|
iterators.push(order === 'asc' ? cursor.iterateForward() : cursor.iterateBackward())
|
||||||
|
iteratorKindRank.push(kindRank)
|
||||||
|
}
|
||||||
|
const tailCursor = this.getTailBufferCursor(column.key)
|
||||||
|
if (tailCursor) {
|
||||||
|
iterators.push(order === 'asc' ? tailCursor.iterateForward() : tailCursor.iterateBackward())
|
||||||
|
iteratorKindRank.push(kindRank)
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
if (iterators.length === 0) return []
|
if (iterators.length === 0) return []
|
||||||
|
|
@ -822,16 +1189,21 @@ export class ColumnStore implements ColumnStoreProvider {
|
||||||
value: next.value.value,
|
value: next.value.value,
|
||||||
entityIntId: next.value.entityIntId,
|
entityIntId: next.value.entityIntId,
|
||||||
cursorIndex: i,
|
cursorIndex: i,
|
||||||
|
kindRank: iteratorKindRank[i],
|
||||||
iterator: iterators[i]
|
iterator: iterators[i]
|
||||||
})
|
})
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// Heapify
|
// Heapify. A number and a string have no ordering between them, so a
|
||||||
const isString = (this.fieldTypes.get(field) ?? ValueType.Number) === ValueType.String
|
// mixed-kind field orders by KIND first (POSTING_KINDS order) and by value
|
||||||
|
// within a kind — one defined total order instead of a comparison whose
|
||||||
|
// answer depends on which value happened to be on the left.
|
||||||
const compare = (a: HeapEntry, b: HeapEntry): number => {
|
const compare = (a: HeapEntry, b: HeapEntry): number => {
|
||||||
let cmp: number
|
let cmp: number
|
||||||
if (isString) {
|
if (a.kindRank !== b.kindRank) {
|
||||||
|
cmp = a.kindRank - b.kindRank
|
||||||
|
} else if (POSTING_KINDS[a.kindRank] === 'string') {
|
||||||
cmp = compareCodePoints(String(a.value), String(b.value))
|
cmp = compareCodePoints(String(a.value), String(b.value))
|
||||||
} else {
|
} else {
|
||||||
cmp = (a.value as number) - (b.value as number)
|
cmp = (a.value as number) - (b.value as number)
|
||||||
|
|
@ -863,6 +1235,7 @@ export class ColumnStore implements ColumnStoreProvider {
|
||||||
value: next.value.value,
|
value: next.value.value,
|
||||||
entityIntId: next.value.entityIntId,
|
entityIntId: next.value.entityIntId,
|
||||||
cursorIndex: top.cursorIndex,
|
cursorIndex: top.cursorIndex,
|
||||||
|
kindRank: top.kindRank,
|
||||||
iterator: top.iterator
|
iterator: top.iterator
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
@ -870,8 +1243,11 @@ export class ColumnStore implements ColumnStoreProvider {
|
||||||
this.heapDown(heap, 0, compare)
|
this.heapDown(heap, 0, compare)
|
||||||
}
|
}
|
||||||
|
|
||||||
// Apply global deleted check, filter, and dedup
|
// Apply global deleted check, filter, and dedup. The deleted bitmap is
|
||||||
const deleted = this.deletedEntities.get(field)
|
// per COLUMN, and the entry came from the column its kind names.
|
||||||
|
const deleted = this.deletedEntities.get(
|
||||||
|
this.columnKey(field, POSTING_KINDS[top.kindRank]) ?? field
|
||||||
|
)
|
||||||
if (deleted && deleted.has(top.entityIntId)) continue
|
if (deleted && deleted.has(top.entityIntId)) continue
|
||||||
if (seen.has(top.entityIntId)) continue
|
if (seen.has(top.entityIntId)) continue
|
||||||
if (filterBitmap && !filterBitmap.has(top.entityIntId)) continue
|
if (filterBitmap && !filterBitmap.has(top.entityIntId)) continue
|
||||||
|
|
@ -913,35 +1289,31 @@ export class ColumnStore implements ColumnStoreProvider {
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Infer ValueType from a JavaScript value.
|
* Encode a value for the column its own kind selected.
|
||||||
*/
|
*
|
||||||
private inferValueType(value: unknown): ValueType {
|
* This does NOT convert between kinds. It used to: a string reaching a
|
||||||
if (typeof value === 'boolean') return ValueType.Boolean
|
* numeric column was run through `Number(value)`, and a number reaching a
|
||||||
if (typeof value === 'number') {
|
* numeric column was run through `Math.round`, so `'electronics'` became
|
||||||
return Number.isInteger(value) ? ValueType.Number : ValueType.Float
|
* `NaN` and vanished while `4.5` became `5` and answered the wrong query.
|
||||||
}
|
* Kind routing removes the need for either — the only work left is picking
|
||||||
return ValueType.String
|
* the encoding the column already committed to.
|
||||||
}
|
*
|
||||||
|
* @returns The encoded value, or `undefined` if the value does not belong in
|
||||||
/**
|
* this column at all — which the caller treats as a routing bug and
|
||||||
* Normalize a JavaScript value to the column's ValueType.
|
* raises, never as a value to skip.
|
||||||
*/
|
*/
|
||||||
private normalizeValue(value: unknown, type: ValueType): number | string | undefined {
|
private normalizeValue(value: unknown, type: ValueType): number | string | undefined {
|
||||||
switch (type) {
|
switch (type) {
|
||||||
case ValueType.Number:
|
case ValueType.Number:
|
||||||
if (typeof value === 'number') return Math.round(value)
|
// Integer column. Non-integers widen it to Float before reaching here.
|
||||||
if (typeof value === 'string') { const n = Number(value); return isNaN(n) ? undefined : Math.round(n) }
|
return typeof value === 'number' && Number.isInteger(value) ? value : undefined
|
||||||
if (typeof value === 'boolean') return value ? 1 : 0
|
|
||||||
return undefined
|
|
||||||
case ValueType.Float:
|
case ValueType.Float:
|
||||||
if (typeof value === 'number') return value
|
return typeof value === 'number' ? value : undefined
|
||||||
if (typeof value === 'string') { const n = Number(value); return isNaN(n) ? undefined : n }
|
|
||||||
return undefined
|
|
||||||
case ValueType.Boolean:
|
case ValueType.Boolean:
|
||||||
if (typeof value === 'boolean') return value ? 1 : 0
|
return typeof value === 'boolean' ? (value ? 1 : 0) : undefined
|
||||||
if (typeof value === 'number') return value ? 1 : 0
|
|
||||||
return undefined
|
|
||||||
case ValueType.String:
|
case ValueType.String:
|
||||||
|
// The string kind is also where objects and bigints land, exactly as
|
||||||
|
// they always did.
|
||||||
return String(value)
|
return String(value)
|
||||||
default:
|
default:
|
||||||
return undefined
|
return undefined
|
||||||
|
|
|
||||||
|
|
@ -55,8 +55,12 @@ export class ColumnTailBuffer {
|
||||||
/** Field name this buffer is for. */
|
/** Field name this buffer is for. */
|
||||||
readonly fieldName: string
|
readonly fieldName: string
|
||||||
|
|
||||||
/** Value type determines sort comparator. */
|
/**
|
||||||
readonly valueType: ValueType
|
* Value type determines sort comparator and segment encoding.
|
||||||
|
*
|
||||||
|
* Widened in place by {@link promoteToFloat} — never otherwise reassigned.
|
||||||
|
*/
|
||||||
|
valueType: ValueType
|
||||||
|
|
||||||
/** Flush threshold. */
|
/** Flush threshold. */
|
||||||
readonly threshold: number
|
readonly threshold: number
|
||||||
|
|
@ -81,6 +85,38 @@ export class ColumnTailBuffer {
|
||||||
this.threshold = threshold
|
this.threshold = threshold
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Widen an integer column to floating point, losslessly and in place.
|
||||||
|
*
|
||||||
|
* The number posting kind holds every JavaScript number, but a segment picks
|
||||||
|
* ONE encoding: i64 for integers, f64 for the rest. A column that has only
|
||||||
|
* ever seen integers is written as i64; the first non-integer to arrive
|
||||||
|
* widens it here, so that value is stored as itself instead of being rounded
|
||||||
|
* to the nearest integer with no error — the rounding that made `4.5` and
|
||||||
|
* `5.5` both answer `where {score: 5}` and neither answer its own value.
|
||||||
|
*
|
||||||
|
* Widening is lossless in both directions it has to be: every value already
|
||||||
|
* buffered is an integer, and every integer is exactly representable as f64.
|
||||||
|
* Segments already on disk keep their own i64 encoding in their own headers
|
||||||
|
* and keep decoding by it — only segments written from here on are f64.
|
||||||
|
*
|
||||||
|
* @throws Error if called on a column that is not an integer column — the
|
||||||
|
* only legal widening is Number → Float, and any other request is a bug in
|
||||||
|
* the caller's kind routing rather than something to absorb quietly.
|
||||||
|
*/
|
||||||
|
promoteToFloat(): void {
|
||||||
|
if (this.valueType === ValueType.Float) return
|
||||||
|
if (this.valueType !== ValueType.Number) {
|
||||||
|
throw new Error(
|
||||||
|
`ColumnTailBuffer '${this.fieldName}': cannot widen a ` +
|
||||||
|
`${ValueType[this.valueType]} column to Float — only an integer ` +
|
||||||
|
`(Number) column widens, and this call means a value reached the ` +
|
||||||
|
`wrong kind's column`
|
||||||
|
)
|
||||||
|
}
|
||||||
|
this.valueType = ValueType.Float
|
||||||
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Add a (value, entityIntId) entry to the buffer.
|
* Add a (value, entityIntId) entry to the buffer.
|
||||||
*
|
*
|
||||||
|
|
|
||||||
|
|
@ -58,6 +58,53 @@ export enum ValueType {
|
||||||
Boolean = 3
|
Boolean = 3
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* The KIND of a value, as the query language sees it.
|
||||||
|
*
|
||||||
|
* A kind is a JavaScript `typeof` class, not a storage encoding: `5` and `5.5`
|
||||||
|
* are one kind (`'number'`) held in one posting column, even though they need
|
||||||
|
* different segment encodings (i64 vs f64 — see {@link ValueType}).
|
||||||
|
*
|
||||||
|
* A field holds ONE POSTING COLUMN PER KIND, so `category` may carry string
|
||||||
|
* values and number values at the same time and answer equality on each. This
|
||||||
|
* replaces the first-writer type freeze, under which the first value's type
|
||||||
|
* became the field's type and every later value of another kind was coerced —
|
||||||
|
* or, when coercion failed (`Number('electronics')`), dropped from the index
|
||||||
|
* with no error: the row stayed readable by id and by vector but vanished from
|
||||||
|
* every equality filter on that field.
|
||||||
|
*
|
||||||
|
* Kinds do not coerce into one another at query time either: `where {c: 5}`
|
||||||
|
* matches rows written with the NUMBER `5`, and `where {c: '5'}` matches rows
|
||||||
|
* written with the STRING `'5'`. Neither ever matches the other.
|
||||||
|
*
|
||||||
|
* Values that are none of these three (objects, bigints) index as strings —
|
||||||
|
* the same `String(value)` treatment they received before.
|
||||||
|
*/
|
||||||
|
export type PostingKind = 'number' | 'string' | 'boolean'
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Every posting kind, in the order that defines cross-kind sort position.
|
||||||
|
*
|
||||||
|
* A mixed-kind field has no natural total order — a number does not compare
|
||||||
|
* with a string — so `sortTopK` orders by KIND first (numbers, then strings,
|
||||||
|
* then booleans) and by value within a kind. A single-kind field, which is
|
||||||
|
* nearly every field, sorts exactly as it always did.
|
||||||
|
*/
|
||||||
|
export const POSTING_KINDS: readonly PostingKind[] = ['number', 'string', 'boolean']
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Path segment marking a field's NON-PRIMARY kind columns on disk.
|
||||||
|
*
|
||||||
|
* The first kind a field ever sees keeps the historical layout —
|
||||||
|
* `<base>/<field>/MANIFEST.json` and `<base>/<field>/L0-NNNNNN` — so every
|
||||||
|
* index written before typed postings opens unchanged, and the byte-for-byte
|
||||||
|
* interchange with the native column store is untouched for the single-kind
|
||||||
|
* fields that are nearly all of them. A second kind arriving on the same field
|
||||||
|
* gets its own column at `<base>/<field>/k/<kind>/…` rather than overwriting or
|
||||||
|
* being coerced into the first.
|
||||||
|
*/
|
||||||
|
export const KIND_PATH_SEGMENT = 'k'
|
||||||
|
|
||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
// Segment header and footer
|
// Segment header and footer
|
||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
|
|
@ -267,6 +314,19 @@ export interface ColumnStoreProvider {
|
||||||
*/
|
*/
|
||||||
hasField(field: string): boolean
|
hasField(field: string): boolean
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Which value KINDS this field actually holds, in {@link POSTING_KINDS}
|
||||||
|
* order — the honest answer to "what type is this field?" for a field that
|
||||||
|
* carries more than one.
|
||||||
|
*
|
||||||
|
* OPTIONAL so an implementation written against the pre-typed-postings
|
||||||
|
* contract still satisfies this interface; feature-detect before calling.
|
||||||
|
*
|
||||||
|
* @param field - Field name
|
||||||
|
* @returns Every kind with at least one posting, or `[]` for an unknown field
|
||||||
|
*/
|
||||||
|
getFieldKinds?(field: string): PostingKind[]
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Flush all in-memory tail buffers to L0 segments on disk.
|
* Flush all in-memory tail buffers to L0 segments on disk.
|
||||||
* Saves all manifests.
|
* Saves all manifests.
|
||||||
|
|
|
||||||
|
|
@ -9,7 +9,7 @@
|
||||||
*
|
*
|
||||||
* @example Enable integrations (recommended)
|
* @example Enable integrations (recommended)
|
||||||
* ```typescript
|
* ```typescript
|
||||||
* import { Brainy } from '@soulcraft/brainy'
|
* import { Brainy } from '@soulcraftlabs/brainy'
|
||||||
*
|
*
|
||||||
* const brain = new Brainy({ integrations: true })
|
* const brain = new Brainy({ integrations: true })
|
||||||
* await brain.init()
|
* await brain.init()
|
||||||
|
|
|
||||||
|
|
@ -41,7 +41,7 @@ The `BrainyMCPService` has been refactored to separate the core functionality fr
|
||||||
### In Any Environment (Browser, Node.js, Server)
|
### In Any Environment (Browser, Node.js, Server)
|
||||||
|
|
||||||
```typescript
|
```typescript
|
||||||
import { Brainy, BrainyMCPAdapter, MCPAugmentationToolset } from '@soulcraft/brainy'
|
import { Brainy, BrainyMCPAdapter, MCPAugmentationToolset } from '@soulcraftlabs/brainy'
|
||||||
|
|
||||||
// Create a Brainy instance
|
// Create a Brainy instance
|
||||||
const brainyData = new Brainy()
|
const brainyData = new Brainy()
|
||||||
|
|
@ -81,7 +81,7 @@ const toolResponse = await toolset.handleRequest({
|
||||||
### In Browser Environment (Core Functionality Only)
|
### In Browser Environment (Core Functionality Only)
|
||||||
|
|
||||||
```typescript
|
```typescript
|
||||||
import { Brainy, BrainyMCPService } from '@soulcraft/brainy'
|
import { Brainy, BrainyMCPService } from '@soulcraftlabs/brainy'
|
||||||
|
|
||||||
// Create a Brainy instance
|
// Create a Brainy instance
|
||||||
const brainyData = new Brainy()
|
const brainyData = new Brainy()
|
||||||
|
|
|
||||||
|
|
@ -2,7 +2,7 @@
|
||||||
* 🧠 BRAINY EMBEDDED PATTERNS
|
* 🧠 BRAINY EMBEDDED PATTERNS
|
||||||
*
|
*
|
||||||
* AUTO-GENERATED - DO NOT EDIT
|
* AUTO-GENERATED - DO NOT EDIT
|
||||||
* Generated: 2026-07-02T21:43:26.976Z
|
* Generated: 2026-08-27T09:18:45-07:00
|
||||||
* Patterns: 220
|
* Patterns: 220
|
||||||
* Coverage: 94-98% of all queries
|
* Coverage: 94-98% of all queries
|
||||||
*
|
*
|
||||||
|
|
|
||||||
|
|
@ -2,7 +2,7 @@
|
||||||
* 🧠 BRAINY EMBEDDED TYPE EMBEDDINGS
|
* 🧠 BRAINY EMBEDDED TYPE EMBEDDINGS
|
||||||
*
|
*
|
||||||
* AUTO-GENERATED - DO NOT EDIT
|
* AUTO-GENERATED - DO NOT EDIT
|
||||||
* Generated: 2026-02-09T16:59:48.867Z
|
* Generated: 2026-08-27T09:18:45-07:00
|
||||||
* Noun Types: 42
|
* Noun Types: 42
|
||||||
* Verb Types: 127
|
* Verb Types: 127
|
||||||
*
|
*
|
||||||
|
|
@ -19,7 +19,7 @@ export const TYPE_METADATA = {
|
||||||
verbTypes: 127,
|
verbTypes: 127,
|
||||||
totalTypes: 169,
|
totalTypes: 169,
|
||||||
embeddingDimensions: 384,
|
embeddingDimensions: 384,
|
||||||
generatedAt: "2026-02-09T16:59:48.867Z",
|
generatedAt: "2026-08-27T09:18:45-07:00",
|
||||||
sizeBytes: {
|
sizeBytes: {
|
||||||
embeddings: 259584,
|
embeddings: 259584,
|
||||||
base64: 346112
|
base64: 346112
|
||||||
|
|
|
||||||
Some files were not shown because too many files have changed in this diff Show more
Reference in a new issue