Compare commits
40 commits
main
...
rel/10.4.1
| Author | SHA1 | Date | |
|---|---|---|---|
| 27759a1be9 | |||
| 6053f6d423 | |||
| a1423c6da7 | |||
| dea3ec2031 | |||
| ebb3a4bf13 | |||
| 2c5e34748e | |||
| 4142f36872 | |||
| 367ca721a5 | |||
| a79db434ac | |||
| da9519903a | |||
| ec644bde56 | |||
| 65493ba2de | |||
| dee46b35c8 | |||
| 1fb5109351 | |||
| 15d4f65dcf | |||
| bc70c43d02 | |||
| 905c267c47 | |||
| b1c7054467 | |||
| 67ae0046de | |||
| 9922631d1f | |||
| 2633e8d5e1 | |||
| f763317af7 | |||
| a8c5fbf9dc | |||
| 34f1886f7c | |||
| eec90bdd69 | |||
| 2648f56ddf | |||
| 8a2ebacf02 | |||
| 4d5f823f47 | |||
| 3e60aded36 | |||
| d5147ed608 | |||
| 6a89adc468 | |||
| 88e79729d3 | |||
| 077cbc0b6f | |||
| 5e3b343a0e | |||
| 4014e0f125 | |||
| 73500e7d10 | |||
| 0f0022b1c9 | |||
| d6bcb14f69 | |||
| a963a744cc | |||
|
|
c99308710a |
49 changed files with 6875 additions and 436 deletions
|
|
@ -5,6 +5,10 @@ name: CI
|
|||
# sequential, so tag-triggered matrix jobs (~22 min) would queue AHEAD of the
|
||||
# tag's publish-source run and starve every release (observed on 8.10.3 and
|
||||
# 9.0.0: the publish sat behind the tag's own redundant CI).
|
||||
concurrency:
|
||||
group: ci-${{ github.ref }}
|
||||
cancel-in-progress: true
|
||||
|
||||
on:
|
||||
push:
|
||||
branches: ['**']
|
||||
|
|
|
|||
148
.forgejo/workflows/delta-gate.yml
Normal file
148
.forgejo/workflows/delta-gate.yml
Normal file
|
|
@ -0,0 +1,148 @@
|
|||
name: Delta Gate
|
||||
|
||||
# On-demand candidate-vs-control gate on the capped functional CI lane
|
||||
# (label: gate-functional). That lane is Bun-only host-mode — there is no
|
||||
# Node.js runtime available to it, so this workflow deliberately avoids every
|
||||
# JS-based action (checkout/setup-node/setup-bun/upload-artifact all require
|
||||
# one) and does everything with plain git + bun in shell steps instead.
|
||||
#
|
||||
# Verdict lines a caller should grep for in the run log:
|
||||
# COLLECTED patch=<n> control=<n> — collection-truncation guard inputs
|
||||
# NEW-RED-COUNT:<n> — failures on candidate absent from control
|
||||
# DELTA-GATE: CLEAN | NEW REDS | INVALID | STOPPED-BY-REGISTRY-TRIPWIRE
|
||||
#
|
||||
# The lane's own housekeeping stops the runner and drops a marker file when
|
||||
# host pressure (I/O, registry latency, disk budget) trips — never ours to
|
||||
# interpret as a red or a green. The final step checks for that marker before
|
||||
# it says anything about pass/fail.
|
||||
|
||||
on:
|
||||
workflow_dispatch:
|
||||
inputs:
|
||||
candidate:
|
||||
description: 'Candidate ref (branch or sha) to gate'
|
||||
required: true
|
||||
type: string
|
||||
control:
|
||||
description: 'Control sha to diff against'
|
||||
required: true
|
||||
type: string
|
||||
# workflow_dispatch needs Actions-unit write on the dispatching credential;
|
||||
# push does not (it runs from the pushed ref's own tree), so a plain push
|
||||
# to a release or CI branch is the fallback trigger while that grant is
|
||||
# outstanding — see the ref-resolution step below for what it gates against.
|
||||
push:
|
||||
branches: ['rel/**', 'ci/**']
|
||||
|
||||
concurrency:
|
||||
group: delta-gate
|
||||
cancel-in-progress: false
|
||||
|
||||
jobs:
|
||||
delta-gate:
|
||||
name: Delta gate — candidate vs control
|
||||
runs-on: gate-functional
|
||||
timeout-minutes: 120
|
||||
steps:
|
||||
- name: Resolve candidate/control refs
|
||||
id: refs
|
||||
run: |
|
||||
candidate="${{ github.event.inputs.candidate }}"
|
||||
control="${{ github.event.inputs.control }}"
|
||||
# workflow_dispatch supplies both explicitly; a push event carries
|
||||
# neither — fall back to the pushed commit as candidate and the
|
||||
# last released, known-good tip (10.4.9) as control, so a plain
|
||||
# push still produces a meaningful gate instead of an empty ref.
|
||||
if [ -z "$candidate" ]; then candidate="${{ github.sha }}"; fi
|
||||
if [ -z "$control" ]; then control="eec90bdd"; fi
|
||||
echo "candidate=$candidate" >> "$GITHUB_OUTPUT"
|
||||
echo "control=$control" >> "$GITHUB_OUTPUT"
|
||||
echo "Resolved (trigger=${{ github.event_name }}): candidate=$candidate control=$control"
|
||||
|
||||
- name: Clean any residue from a prior run
|
||||
run: rm -rf "ob-cand-${{ github.run_id }}" "ob-ctrl-${{ github.run_id }}" "/tmp/ob-${{ github.run_id }}-"*
|
||||
|
||||
- name: Clone + test — candidate
|
||||
id: patch
|
||||
run: |
|
||||
set -o pipefail
|
||||
git clone --quiet "https://source.soulcraft.com/soulcraftlabs/open-brainy.git" "ob-cand-${{ github.run_id }}"
|
||||
cd "ob-cand-${{ github.run_id }}"
|
||||
git checkout --quiet "${{ steps.refs.outputs.candidate }}"
|
||||
git log --oneline -1
|
||||
bun install
|
||||
rc=0
|
||||
bun x vitest run > "/tmp/ob-${{ github.run_id }}-patch.log" 2>&1 || rc=$?
|
||||
echo "PATCH-RC:$rc"
|
||||
grep -aE "Tests .*(passed|failed)" "/tmp/ob-${{ github.run_id }}-patch.log" | tail -1
|
||||
grep -aE "^ FAIL |^\s+×" "/tmp/ob-${{ github.run_id }}-patch.log" | sed -E "s/ [0-9]+ms$//" | sed -E "s/^\s+//" | sort -u > "/tmp/ob-${{ github.run_id }}-patch.fail"
|
||||
echo "PATCH-FAILING:$(wc -l < "/tmp/ob-${{ github.run_id }}-patch.fail")"
|
||||
|
||||
- name: Clone + test — control
|
||||
id: control
|
||||
run: |
|
||||
set -o pipefail
|
||||
git clone --quiet "https://source.soulcraft.com/soulcraftlabs/open-brainy.git" "ob-ctrl-${{ github.run_id }}"
|
||||
cd "ob-ctrl-${{ github.run_id }}"
|
||||
git checkout --quiet "${{ steps.refs.outputs.control }}"
|
||||
git log --oneline -1
|
||||
bun install
|
||||
rc=0
|
||||
bun x vitest run > "/tmp/ob-${{ github.run_id }}-control.log" 2>&1 || rc=$?
|
||||
echo "CONTROL-RC:$rc"
|
||||
grep -aE "Tests .*(passed|failed)" "/tmp/ob-${{ github.run_id }}-control.log" | tail -1
|
||||
grep -aE "^ FAIL |^\s+×" "/tmp/ob-${{ github.run_id }}-control.log" | sed -E "s/ [0-9]+ms$//" | sed -E "s/^\s+//" | sort -u > "/tmp/ob-${{ github.run_id }}-control.fail"
|
||||
echo "CONTROL-FAILING:$(wc -l < "/tmp/ob-${{ github.run_id }}-control.fail")"
|
||||
|
||||
- name: Delta gate verdict
|
||||
if: always()
|
||||
run: |
|
||||
set -o pipefail
|
||||
|
||||
# The lane's own tripwire wins over anything we would otherwise say:
|
||||
# a bare failure/timeout above with this marker present is host
|
||||
# pressure, never a real red and never a real green.
|
||||
if [ -f /srv/gate-lane/TRIPWIRE-STOPPED ]; then
|
||||
echo "DELTA-GATE: STOPPED-BY-REGISTRY-TRIPWIRE"
|
||||
head -1 /srv/gate-lane/TRIPWIRE-STOPPED
|
||||
exit 3
|
||||
fi
|
||||
|
||||
patch_log="/tmp/ob-${{ github.run_id }}-patch.log"
|
||||
control_log="/tmp/ob-${{ github.run_id }}-control.log"
|
||||
patch_fail="/tmp/ob-${{ github.run_id }}-patch.fail"
|
||||
control_fail="/tmp/ob-${{ github.run_id }}-control.fail"
|
||||
|
||||
if [ ! -s "$patch_log" ] || [ ! -s "$control_log" ]; then
|
||||
echo "DELTA-GATE: INVALID — a leg produced no log (see the two steps above for the real cause)"
|
||||
exit 2
|
||||
fi
|
||||
|
||||
pt=$(grep -aoE "\(([0-9]+)\)$" "$patch_log" | tail -1 | tr -d "()")
|
||||
ct=$(grep -aoE "\(([0-9]+)\)$" "$control_log" | tail -1 | tr -d "()")
|
||||
echo "COLLECTED patch=${pt:-0} control=${ct:-0}"
|
||||
if [ "${pt:-0}" -lt 3000 ] || [ "${ct:-0}" -lt 3000 ]; then
|
||||
echo "DELTA-GATE: INVALID — truncated collection"
|
||||
exit 2
|
||||
fi
|
||||
|
||||
echo "=== NEW REDS ==="
|
||||
comm -23 "$patch_fail" "$control_fail"
|
||||
new=$(comm -23 "$patch_fail" "$control_fail" | wc -l)
|
||||
echo "NEW-RED-COUNT:$new"
|
||||
|
||||
echo "=== full candidate fail list ==="
|
||||
cat "$patch_fail"
|
||||
echo "=== full control fail list ==="
|
||||
cat "$control_fail"
|
||||
|
||||
if [ "$new" -eq 0 ]; then
|
||||
echo "DELTA-GATE: CLEAN"
|
||||
else
|
||||
echo "DELTA-GATE: NEW REDS"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
- name: Clean up (mind the lane's disk budget)
|
||||
if: always()
|
||||
run: rm -rf "ob-cand-${{ github.run_id }}" "ob-ctrl-${{ github.run_id }}" "/tmp/ob-${{ github.run_id }}-"*
|
||||
52
CHANGELOG.md
52
CHANGELOG.md
|
|
@ -2,6 +2,58 @@
|
|||
|
||||
All notable changes to this project will be documented in this file. See [standard-version](https://github.com/conventional-changelog/standard-version) for commit guidelines.
|
||||
|
||||
### [10.4.11](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v10.4.9...v10.4.11) (2026-09-02)
|
||||
|
||||
- ci: superseded pushes cancel their own runs (concurrency per ref) (6053f6d4)
|
||||
- test(batch): the batch-size-limit tests add unvectored items — they test batching, not embedding (a1423c6d)
|
||||
- fix(flush): the gate settles its waiter from the machine, never from a chain (dea3ec20)
|
||||
- test(batch): the batch-vs-individual timing assertion runs in the perf lane, not the correctness gate (ebb3a4bf)
|
||||
- test(gate): the coverage guard counts the perf lane's config as a gate (2c5e3474)
|
||||
- chore(contract): emit the 10.4.11 manifest (4142f368)
|
||||
- fix(close): a read-only brain writes no clean-shutdown evidence — the marker is the writer's word about itself (367ca721)
|
||||
- fix(generation-store): commitTransaction refuses while single-ops are pending — the order invariant is enforced, not assumed (a79db434)
|
||||
- test(shutdown): pin one owner per brain — real processes, real signals (da951990)
|
||||
- fix(shutdown): one owner per brain — the signal handler defers to close(), and flush is single-flight (ec644bde)
|
||||
- fix(vfs): a path-scoped search is a served range over the path, not a refused prefix match (65493ba2)
|
||||
- ci(test): perf and scale benchmarks leave the correctness gate (dee46b35)
|
||||
- test(open): pin the pending-embed checkpoint — stuck id, crash matrix, torn fallback (1fb51093)
|
||||
- perf(open): the pending-embed fold is bounded by a checkpoint of the SET, not an empty-only mark (15d4f65d)
|
||||
- perf(open): a sealed segment the manifest proves is below the bound is never read (bc70c43d)
|
||||
- fix(find): a page the metadata block already cut is not cut again (905c267c)
|
||||
- fix(find): the hybrid legs rank inside the filter, and only the page is read (b1c70544)
|
||||
- ci(delta-gate): add a push fallback trigger alongside workflow_dispatch (67ae0046)
|
||||
- ci: add the delta-gate workflow for the capped functional lane (9922631d)
|
||||
- docs(plugin): the planner door's hiddenIds contract is the answer, not the mechanism (2633e8d5)
|
||||
- feat(engine): a protected factory for the generation store — a subclass may substitute one that keeps the contract (f763317a)
|
||||
- fix(find): near() searches around the anchor's own vector, and refuses by name without one (a8c5fbf9)
|
||||
- Merge remote-tracking branches 'origin/fix/planner-provider-door' and 'origin/fix/containment-batching' into rel/10.4.10-candidate (34f1886f)
|
||||
- feat(plugin): an optional planFindPage door — an index that can plan a find answers it in one call (4d5f823f)
|
||||
- perf(vfs): repairContainment's reconcile is one paged edge walk, not one graph call per file (3e60aded)
|
||||
|
||||
|
||||
### [10.4.9](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v10.4.6...v10.4.9) (2026-09-02)
|
||||
|
||||
- Merge branch 'fix/pending-embed-low-water' into rel/10.4.9-candidate (2648f56d)
|
||||
- fix(open): pending-embed recovery keeps the crash-recovery contract — foreground, bounded by the mark (8a2ebacf)
|
||||
- Merge branches 'fix/connected-find-order', 'fix/pending-embed-low-water' and 'fix/related-verb-array' into rel/10.4.9-candidate (d5147ed6)
|
||||
- fix(graph): the verb fast paths honour every requested type, source, and target (6a89adc4)
|
||||
- perf(open): pending-embed recovery is bounded by a low-water mark and runs behind the doors (88e79729)
|
||||
- fix(find): connected finds are graph-first — neighbours, then the filter over those ids, then the page (077cbc0b)
|
||||
- fix(storage): counts persistence is single-flight, coalesced, and never races its own temp file (5e3b343a)
|
||||
|
||||
|
||||
### [10.4.6](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v10.4.5...v10.4.6) (2026-08-31)
|
||||
|
||||
- fix(transact): metadata-index ops take their JSON-safe view at the crossing, not at construction (73500e7d)
|
||||
|
||||
|
||||
### [10.4.5](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v10.4.4...v10.4.5) (2026-08-31)
|
||||
|
||||
- build(release): the docs-push step retires — this engine documents itself in its own repository (d6bcb14f)
|
||||
- fix(generations): a sealed segment may only declare the generations it holds (a963a744)
|
||||
- fix(recovery): a torn generation-log tail is a terminal verdict, never a wait (c9930871)
|
||||
|
||||
|
||||
### [10.4.4](https://source.soulcraft.com/soulcraftlabs/open-brainy/compare/v10.4.3...v10.4.4) (2026-08-28)
|
||||
|
||||
- fix(vfs): the old-root sweep narrates only when it has something to say (d49148e1)
|
||||
|
|
|
|||
|
|
@ -41,6 +41,20 @@ npm test
|
|||
Tests run on [Vitest](https://vitest.dev/). `npm test` runs the unit suite;
|
||||
see `package.json` for `test:integration`, `test:coverage`, and friends.
|
||||
|
||||
## Test gate
|
||||
|
||||
The release gate is a bare `vitest run` (no `--config` flag) — the same
|
||||
command the delta gate and CI's checks invoke. It carries the full
|
||||
correctness suite and nothing else: wall-clock/scale benchmarks
|
||||
(`tests/performance/**`, `tests/critical-performance-benchmark.test.ts`,
|
||||
`tests/api/performance-benchmarks.test.ts`) and the two tests whose outcome
|
||||
depends on the host machine or network rather than the code
|
||||
(`tests/package-size-limit.test.ts` shells out to the `npm` CLI;
|
||||
`tests/model-loading.test.ts` makes a real network call to download a model)
|
||||
are excluded from it, because a timing threshold or a flaky network call has
|
||||
no business failing a correctness check. That whole family runs on demand,
|
||||
in its own exclusive slot, via `npm run test:perf`.
|
||||
|
||||
## Standards
|
||||
|
||||
- **Strict TypeScript.** No `any` escape hatches to dodge the type checker.
|
||||
|
|
|
|||
|
|
@ -161,6 +161,11 @@
|
|||
"kind": "method",
|
||||
"arity": 1
|
||||
},
|
||||
{
|
||||
"name": "captureEmbedCheckpoint",
|
||||
"kind": "method",
|
||||
"arity": 0
|
||||
},
|
||||
{
|
||||
"name": "checkHealth",
|
||||
"kind": "method",
|
||||
|
|
@ -225,6 +230,11 @@
|
|||
"name": "counts",
|
||||
"kind": "accessor"
|
||||
},
|
||||
{
|
||||
"name": "createGenerationStore",
|
||||
"kind": "method",
|
||||
"arity": 1
|
||||
},
|
||||
{
|
||||
"name": "createIndex",
|
||||
"kind": "method",
|
||||
|
|
@ -258,6 +268,11 @@
|
|||
"kind": "method",
|
||||
"arity": 1
|
||||
},
|
||||
{
|
||||
"name": "demoteTornEntityTreeStamp",
|
||||
"kind": "method",
|
||||
"arity": 4
|
||||
},
|
||||
{
|
||||
"name": "detectIdKind",
|
||||
"kind": "method",
|
||||
|
|
@ -353,11 +368,6 @@
|
|||
"kind": "method",
|
||||
"arity": 1
|
||||
},
|
||||
{
|
||||
"name": "executeGraphSearch",
|
||||
"kind": "method",
|
||||
"arity": 2
|
||||
},
|
||||
{
|
||||
"name": "executeProximitySearch",
|
||||
"kind": "method",
|
||||
|
|
@ -368,11 +378,21 @@
|
|||
"kind": "method",
|
||||
"arity": 2
|
||||
},
|
||||
{
|
||||
"name": "executeTextSearchScored",
|
||||
"kind": "method",
|
||||
"arity": 3
|
||||
},
|
||||
{
|
||||
"name": "executeVectorSearch",
|
||||
"kind": "method",
|
||||
"arity": 3
|
||||
},
|
||||
{
|
||||
"name": "executeVectorSearchScored",
|
||||
"kind": "method",
|
||||
"arity": 3
|
||||
},
|
||||
{
|
||||
"name": "explain",
|
||||
"kind": "method",
|
||||
|
|
@ -418,6 +438,11 @@
|
|||
"kind": "method",
|
||||
"arity": 2
|
||||
},
|
||||
{
|
||||
"name": "filterIdsWithinBelted",
|
||||
"kind": "method",
|
||||
"arity": 2
|
||||
},
|
||||
{
|
||||
"name": "find",
|
||||
"kind": "method",
|
||||
|
|
@ -711,6 +736,11 @@
|
|||
"kind": "method",
|
||||
"arity": 2
|
||||
},
|
||||
{
|
||||
"name": "hydrateResultPage",
|
||||
"kind": "method",
|
||||
"arity": 2
|
||||
},
|
||||
{
|
||||
"name": "import",
|
||||
"kind": "method",
|
||||
|
|
@ -741,6 +771,14 @@
|
|||
"kind": "method",
|
||||
"arity": 0
|
||||
},
|
||||
{
|
||||
"name": "isClosed",
|
||||
"kind": "accessor"
|
||||
},
|
||||
{
|
||||
"name": "isClosing",
|
||||
"kind": "accessor"
|
||||
},
|
||||
{
|
||||
"name": "isEmbeddingReady",
|
||||
"kind": "method",
|
||||
|
|
@ -799,6 +837,16 @@
|
|||
"kind": "method",
|
||||
"arity": 1
|
||||
},
|
||||
{
|
||||
"name": "maybeWriteEmbedCheckpoint",
|
||||
"kind": "method",
|
||||
"arity": 0
|
||||
},
|
||||
{
|
||||
"name": "maybeWriteEmbedLowWater",
|
||||
"kind": "method",
|
||||
"arity": 0
|
||||
},
|
||||
{
|
||||
"name": "metadataIndexRetractionOp",
|
||||
"kind": "method",
|
||||
|
|
@ -854,6 +902,11 @@
|
|||
"kind": "method",
|
||||
"arity": 1
|
||||
},
|
||||
{
|
||||
"name": "noteEmbedCheckpointCadence",
|
||||
"kind": "method",
|
||||
"arity": 0
|
||||
},
|
||||
{
|
||||
"name": "noteWriteForPersistence",
|
||||
"kind": "method",
|
||||
|
|
@ -869,6 +922,11 @@
|
|||
"kind": "method",
|
||||
"arity": 1
|
||||
},
|
||||
{
|
||||
"name": "pageConnectedIds",
|
||||
"kind": "method",
|
||||
"arity": 2
|
||||
},
|
||||
{
|
||||
"name": "pagination",
|
||||
"kind": "accessor"
|
||||
|
|
@ -893,6 +951,11 @@
|
|||
"kind": "method",
|
||||
"arity": 0
|
||||
},
|
||||
{
|
||||
"name": "pendingResult",
|
||||
"kind": "method",
|
||||
"arity": 2
|
||||
},
|
||||
{
|
||||
"name": "performInit",
|
||||
"kind": "method",
|
||||
|
|
@ -993,6 +1056,11 @@
|
|||
"kind": "method",
|
||||
"arity": 2
|
||||
},
|
||||
{
|
||||
"name": "readPendingEmbedBound",
|
||||
"kind": "method",
|
||||
"arity": 0
|
||||
},
|
||||
{
|
||||
"name": "ready",
|
||||
"kind": "accessor"
|
||||
|
|
@ -1112,6 +1180,11 @@
|
|||
"kind": "method",
|
||||
"arity": 2
|
||||
},
|
||||
{
|
||||
"name": "resolveConnectedIds",
|
||||
"kind": "method",
|
||||
"arity": 1
|
||||
},
|
||||
{
|
||||
"name": "resolveDiffEndpoint",
|
||||
"kind": "method",
|
||||
|
|
@ -1155,7 +1228,7 @@
|
|||
{
|
||||
"name": "rrfFusion",
|
||||
"kind": "method",
|
||||
"arity": 4
|
||||
"arity": 3
|
||||
},
|
||||
{
|
||||
"name": "runAggregationBackfillWalk",
|
||||
|
|
@ -1275,6 +1348,11 @@
|
|||
"kind": "method",
|
||||
"arity": 1
|
||||
},
|
||||
{
|
||||
"name": "textIdsWithinBelted",
|
||||
"kind": "method",
|
||||
"arity": 2
|
||||
},
|
||||
{
|
||||
"name": "trackField",
|
||||
"kind": "method",
|
||||
|
|
@ -1413,6 +1491,16 @@
|
|||
"name": "wireGraphIdResolver",
|
||||
"kind": "method",
|
||||
"arity": 0
|
||||
},
|
||||
{
|
||||
"name": "writeEmbedCheckpoint",
|
||||
"kind": "method",
|
||||
"arity": 0
|
||||
},
|
||||
{
|
||||
"name": "writeEmbedLowWater",
|
||||
"kind": "method",
|
||||
"arity": 0
|
||||
}
|
||||
],
|
||||
"errors": [
|
||||
|
|
|
|||
4
package-lock.json
generated
4
package-lock.json
generated
|
|
@ -1,12 +1,12 @@
|
|||
{
|
||||
"name": "@soulcraftlabs/brainy",
|
||||
"version": "10.4.4",
|
||||
"version": "10.4.11",
|
||||
"lockfileVersion": 3,
|
||||
"requires": true,
|
||||
"packages": {
|
||||
"": {
|
||||
"name": "@soulcraftlabs/brainy",
|
||||
"version": "10.4.4",
|
||||
"version": "10.4.11",
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"@msgpack/msgpack": "^3.1.2",
|
||||
|
|
|
|||
|
|
@ -1,6 +1,6 @@
|
|||
{
|
||||
"name": "@soulcraftlabs/brainy",
|
||||
"version": "10.4.4",
|
||||
"version": "10.4.11",
|
||||
"brainyContract": 1,
|
||||
"description": "Universal Knowledge Protocol™ - World's first Triple Intelligence database unifying vector, graph, and document search in one API. Stage 3 CANONICAL: 42 nouns × 127 verbs covering 96-97% of all human knowledge.",
|
||||
"main": "dist/index.js",
|
||||
|
|
@ -88,7 +88,7 @@
|
|||
"test:watch": "NODE_OPTIONS='--max-old-space-size=8192' vitest --config tests/configs/vitest.unit.config.ts",
|
||||
"test:coverage": "NODE_OPTIONS='--max-old-space-size=8192' vitest run --config tests/configs/vitest.unit.config.ts --coverage",
|
||||
"test:unit": "NODE_OPTIONS='--max-old-space-size=8192' vitest run --config tests/configs/vitest.unit.config.ts",
|
||||
"test:perf": "vitest run tests/unit/performance --reporter=basic",
|
||||
"test:perf": "vitest run --config tests/configs/vitest.perf.config.ts",
|
||||
"test:integration": "NODE_OPTIONS='--max-old-space-size=8192' vitest run --config tests/configs/vitest.integration.config.ts",
|
||||
"test:semantic": "NODE_OPTIONS='--max-old-space-size=8192' vitest run --config tests/configs/vitest.semantic.config.ts",
|
||||
"test:all": "npm run test:unit && npm run test:integration",
|
||||
|
|
|
|||
|
|
@ -248,17 +248,12 @@ else
|
|||
echo -e "${RED}⚠️ FORGEJO_RELEASE_TOKEN unset — no release page created; tag + CHANGELOG remain the record${NC}\n"
|
||||
fi
|
||||
|
||||
# Step 12: Push public docs to the soulcraft.com docs ingest door
|
||||
# (VENUE-DOCS-RELEASE-PUSH). Skips with a loud warning when
|
||||
# DOCS_INGEST_SECRET is unset; fails loudly (without undoing the publish —
|
||||
# that already happened) when a push errors, so the docs site never
|
||||
# silently trails npm.
|
||||
echo -e "${BLUE}1️⃣2️⃣ Pushing public docs to soulcraft.com/docs...${NC}"
|
||||
if node scripts/push-docs.js; then
|
||||
echo -e "${GREEN}✅ Docs push step done${NC}\n"
|
||||
else
|
||||
echo -e "${RED}❌ Docs push FAILED — soulcraft.com/docs trails npm until re-run or interim sync${NC}\n"
|
||||
fi
|
||||
# Step 12 RETIRED (2026-08-31, CORTEX-SITE-BRAINY-RENAME round 12, David-ruled):
|
||||
# soulcraft.com/docs carries the paid product's documentation only. This
|
||||
# engine's documentation home is THIS repository — README and docs/ — and the
|
||||
# site serves 301s for the slugs this rail used to push. The push script stays
|
||||
# in the tree for history; the rail no longer calls it.
|
||||
echo -e "${BLUE}Docs step: this engine documents itself in its own repo (site push retired 2026-08-31)${NC}"
|
||||
|
||||
echo -e "${GREEN}━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━${NC}"
|
||||
echo -e "${GREEN}🎉 Release ${NEW_VERSION} complete!${NC}"
|
||||
|
|
|
|||
1590
src/brainy.ts
1590
src/brainy.ts
File diff suppressed because it is too large
Load diff
|
|
@ -351,3 +351,63 @@ export class PendingFlushDurabilityError extends Error {
|
|||
this.failedAttempts = failedAttempts
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* @description Thrown by {@link GenerationStore.commitTransaction} when the
|
||||
* PENDING single-op tier is non-empty — i.e. one or more `commitSingleOp()`
|
||||
* generations are buffered in memory, not yet flushed to
|
||||
* `committedRanges` via `flushPendingSingleOps()`.
|
||||
*
|
||||
* The invariant `reservedGensAsc()` (and everything built on it —
|
||||
* `resolveManyAt`, `resolveAt`, `changedBetween`, the hot-tail window) relies
|
||||
* on is documented, not enforced by types: pending generations must always be
|
||||
* numerically greater than every committed one, because the ONLY sanctioned
|
||||
* callers of `commitTransaction()` — `Brainy.transact()` and
|
||||
* `Brainy.compactHistory()` — flush the pending tier FIRST. A caller that
|
||||
* invokes `commitTransaction()` directly while single-ops are still pending
|
||||
* breaks that invariant: the new commit lands in `committedRanges` ABOVE
|
||||
* generations still sitting in `pendingGens`, so the committed-then-pending
|
||||
* concatenation `reservedGensAsc()` yields is no longer ascending. The
|
||||
* concrete failure this produces is silent, not a crash: `resolveManyAt`
|
||||
* walks committed ranges before pending ones, so it can report a NEWER
|
||||
* generation as the "first after" a pin than an older, still-pending one that
|
||||
* actually touched the id first — a wrong before-image at a point-in-time
|
||||
* read, without a compensating error to warn a caller anything went wrong.
|
||||
*
|
||||
* This error refuses the commit outright, before any staging I/O: nothing is
|
||||
* written, the generation counter reservation is untouched, and
|
||||
* `committedRanges`/`pendingGens` are exactly as they were. Call
|
||||
* `flushPendingSingleOps()` first (or go through `Brainy.transact()`, which
|
||||
* already does).
|
||||
*
|
||||
* @example
|
||||
* try {
|
||||
* await generationStore.commitTransaction({ touched, execute })
|
||||
* } catch (err) {
|
||||
* if (err instanceof PendingSingleOpsUnflushedError) {
|
||||
* await generationStore.flushPendingSingleOps()
|
||||
* await generationStore.commitTransaction({ touched, execute }) // now safe
|
||||
* }
|
||||
* }
|
||||
*/
|
||||
export class PendingSingleOpsUnflushedError extends Error {
|
||||
/** How many un-flushed single-op generations were buffered at refusal time. */
|
||||
public readonly pendingCount: number
|
||||
|
||||
/**
|
||||
* @param pendingCount - `pendingGens.length` at the moment of refusal (always ≥ 1).
|
||||
*/
|
||||
constructor(pendingCount: number) {
|
||||
super(
|
||||
`commitTransaction() refused: ${pendingCount} pending single-op generation(s) ` +
|
||||
`are still buffered and un-flushed. Flush the pending single-op tier before ` +
|
||||
`committing a transaction — Brainy.transact() does this automatically; a ` +
|
||||
`direct commitTransaction() call with pending generations would leave the ` +
|
||||
`generation order unsorted (committed generations landing above lower, ` +
|
||||
`still-pending ones) and make point-in-time reads (resolveManyAt/resolveAt) ` +
|
||||
`return the wrong before-image. Call flushPendingSingleOps() first, then retry.`
|
||||
)
|
||||
this.name = 'PendingSingleOpsUnflushedError'
|
||||
this.pendingCount = pendingCount
|
||||
}
|
||||
}
|
||||
|
|
|
|||
|
|
@ -40,7 +40,10 @@
|
|||
* The manifest (`_generations/facts/manifest.json`, JSON — forensics stay
|
||||
* terminal-readable) is the single source of truth for the segment SET;
|
||||
* rotation flips it atomically (write-new → fsync → rename) BEFORE the new
|
||||
* tail's first byte exists, so no segment file is ever unaccounted for.
|
||||
* tail's first byte exists, so no segment file is ever unaccounted for. Its
|
||||
* per-segment `firstGeneration`/`lastGeneration` are LOAD-BEARING at open: a
|
||||
* recovery pass looking for facts above a bound reads only the segments those
|
||||
* bounds cannot rule out (the prune law — see `segmentsHoldingFactsAbove`).
|
||||
*
|
||||
* ## Mixed-version logs (the v2 live-write cutover)
|
||||
*
|
||||
|
|
@ -689,6 +692,74 @@ function parseSegment(
|
|||
return { facts, validBytes: offset, formatVersion: FACT_LOG_FORMAT_V1 }
|
||||
}
|
||||
|
||||
/**
|
||||
* THE PRUNE LAW — which segment files a pass looking for facts ABOVE
|
||||
* `committedGeneration` actually has to read, and how many the manifest's own
|
||||
* recorded bounds took off the table.
|
||||
*
|
||||
* A sealed segment's `lastGeneration` is written at SEAL time and never
|
||||
* mutated upward afterwards ({@link FactLog.rotate}, unchanged since the log
|
||||
* was introduced): the tail's bytes are fsynced FIRST (`await this.sync()` —
|
||||
* "sealed segments are always fully durable"), the entry is then built from
|
||||
* the content that fsync covered, and only then does the manifest flip —
|
||||
* atomically (tmp+rename) and fsynced — which in the SAME write re-points
|
||||
* `tailSegment` at a new file, so the sealed file is never appended to again.
|
||||
* A crash anywhere in that order is safe in the pruning direction: crash
|
||||
* before the manifest write and the segment is still the TAIL (read whole);
|
||||
* crash after it and the entry describes bytes that were already durable. The
|
||||
* only later mutation of a sealed segment is `open()`'s straddle truncation,
|
||||
* which REMOVES facts and re-derives the entry from the actual bytes — so a
|
||||
* recorded bound can drift DOWN with its file, never up.
|
||||
*
|
||||
* Therefore: `lastGeneration = L` proves the file holds no fact above L, and
|
||||
* a pass above `committedGeneration >= L` can skip it whole — no read, no
|
||||
* CRC decode, no msgpack. What the manifest cannot PROVE is never pruned: an
|
||||
* entry with no numeric `lastGeneration` (a legacy or hand-repaired manifest)
|
||||
* is read, and the unsealed tail is always read.
|
||||
*
|
||||
* This is the difference between an open that costs O(whole fact log) and one
|
||||
* that costs O(the facts that could matter). MEASURED in production: a 16k-row
|
||||
* brain at generation ~478,819 paid 34-37s of segment reads and CRC decoding
|
||||
* in `generation-store-open-fold` on EVERY open — to answer a question whose
|
||||
* answer, after a clean close, is always "nothing".
|
||||
*/
|
||||
function segmentsHoldingFactsAbove(
|
||||
stored: FactsManifest,
|
||||
committedGeneration: number
|
||||
): { files: string[]; pruned: number } {
|
||||
const files: string[] = []
|
||||
let pruned = 0
|
||||
for (const entry of stored.segments) {
|
||||
const last = (entry as Partial<SegmentEntry>).lastGeneration
|
||||
if (typeof last === 'number' && Number.isFinite(last) && last <= committedGeneration) {
|
||||
pruned++
|
||||
continue
|
||||
}
|
||||
files.push(entry.file)
|
||||
}
|
||||
if (stored.tailSegment) files.push(stored.tailSegment)
|
||||
return { files, pruned }
|
||||
}
|
||||
|
||||
/**
|
||||
* Say what the open actually read. One line, and only when the log holds more
|
||||
* than one segment (a single-segment log has nothing to prune and nothing to
|
||||
* report) — the operator's receipt that the open is paying for the tail, not
|
||||
* for the whole history.
|
||||
*/
|
||||
function narrateAboveScan(
|
||||
pass: string,
|
||||
committedGeneration: number,
|
||||
read: number,
|
||||
pruned: number
|
||||
): void {
|
||||
if (read + pruned <= 1) return
|
||||
prodLog.narrate(
|
||||
`[FactLog] ${pass} above generation ${committedGeneration}: ${read} segment(s) read, ` +
|
||||
`${pruned} pruned of ${read + pruned} (sealed at or below the bound)`
|
||||
)
|
||||
}
|
||||
|
||||
/**
|
||||
* The generation fact log. One instance per open store; every method assumes
|
||||
* the single-writer discipline the generation store already enforces (calls
|
||||
|
|
@ -754,22 +825,6 @@ export class FactLog {
|
|||
return this.manifest.brainId !== undefined || this.tailVersion === FACT_LOG_FORMAT_V2
|
||||
}
|
||||
|
||||
/**
|
||||
* Open the log and reconcile it to committed truth: read the manifest,
|
||||
* establish the tail's intact content (torn-tail scan), then TRUNCATE any
|
||||
* fact with `generation > committedGeneration` — those never committed (a
|
||||
* crash between fact-append and the commit point). After open, the log is
|
||||
* exactly the committed prefix.
|
||||
*/
|
||||
/**
|
||||
* Read (without truncating) every intact fact ABOVE a generation — the
|
||||
* log-authority recovery surface: after a crash, facts beyond the
|
||||
* manifest watermark that survived with valid CRCs are ACKED writes in
|
||||
* durable-at-ack mode, and the owner REPLAYS them instead of letting
|
||||
* open() truncate them. Must be called BEFORE open() (it reads the raw
|
||||
* segments directly; the torn tail's invalid suffix is ignored exactly
|
||||
* like open() would).
|
||||
*/
|
||||
/**
|
||||
* STREAMING twin of {@link FactLog.peekFactsAbove} for the recovery fold:
|
||||
* yields facts above the bound one SEGMENT at a time, ascending, without
|
||||
|
|
@ -779,13 +834,18 @@ export class FactLog {
|
|||
* Works manifest-direct (safe before {@link FactLog.open}). Ordering is
|
||||
* structural (segments rotate in order; appends are ordered within one) and
|
||||
* ASSERTED — a violation aborts loudly, never a silent misordered replay.
|
||||
*
|
||||
* Reads only the segments that CAN hold a fact above the bound — see
|
||||
* {@link segmentsHoldingFactsAbove}. A bounded fold above a high checkpoint
|
||||
* therefore reads its own tail, not the whole history it already proved
|
||||
* durable.
|
||||
*/
|
||||
async *streamFactsAbove(committedGeneration: number): AsyncGenerator<CommitFact[], void> {
|
||||
const stored = (await this.storage.readRawObject(FACTS_MANIFEST_PATH)) as FactsManifest | null
|
||||
if (!stored || typeof stored !== 'object' || !Array.isArray(stored.segments)) return
|
||||
if (stored.formatVersion !== FACTS_FORMAT_VERSION) return
|
||||
const files = [...stored.segments.map((s) => s.file)]
|
||||
if (stored.tailSegment) files.push(stored.tailSegment)
|
||||
const { files, pruned } = segmentsHoldingFactsAbove(stored, committedGeneration)
|
||||
narrateAboveScan('recovery fold', committedGeneration, files.length, pruned)
|
||||
let lastGen = committedGeneration
|
||||
for (const file of files) {
|
||||
const bytes = await this.storage.readRawBytes(`${FACTS_PREFIX}/${file}`)
|
||||
|
|
@ -807,13 +867,27 @@ export class FactLog {
|
|||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Read (without truncating) every intact fact ABOVE a generation — the
|
||||
* log-authority recovery surface: after a crash, facts beyond the
|
||||
* manifest watermark that survived with valid CRCs are ACKED writes in
|
||||
* durable-at-ack mode, and the owner REPLAYS them instead of letting
|
||||
* open() truncate them. Must be called BEFORE open() (it reads the raw
|
||||
* segments directly; the torn tail's invalid suffix is ignored exactly
|
||||
* like open() would).
|
||||
*
|
||||
* Reads only the segments that CAN hold such a fact — see
|
||||
* {@link segmentsHoldingFactsAbove}. This runs on EVERY log-authority open,
|
||||
* including the clean one where the answer is always empty, so the segments
|
||||
* the manifest already proves irrelevant are never opened at all.
|
||||
*/
|
||||
async peekFactsAbove(committedGeneration: number): Promise<CommitFact[]> {
|
||||
const stored = (await this.storage.readRawObject(FACTS_MANIFEST_PATH)) as FactsManifest | null
|
||||
if (!stored || typeof stored !== 'object' || !Array.isArray(stored.segments)) return []
|
||||
if (stored.formatVersion !== FACTS_FORMAT_VERSION) return []
|
||||
const out: CommitFact[] = []
|
||||
const files = [...stored.segments.map((s) => s.file)]
|
||||
if (stored.tailSegment) files.push(stored.tailSegment)
|
||||
const { files, pruned } = segmentsHoldingFactsAbove(stored, committedGeneration)
|
||||
narrateAboveScan('above-manifest peek', committedGeneration, files.length, pruned)
|
||||
for (const file of files) {
|
||||
const bytes = await this.storage.readRawBytes(`${FACTS_PREFIX}/${file}`)
|
||||
if (bytes === null) continue
|
||||
|
|
@ -826,6 +900,13 @@ export class FactLog {
|
|||
return out
|
||||
}
|
||||
|
||||
/**
|
||||
* Open the log and reconcile it to committed truth: read the manifest,
|
||||
* establish the tail's intact content (torn-tail scan), then TRUNCATE any
|
||||
* fact with `generation > committedGeneration` — those never committed (a
|
||||
* crash between fact-append and the commit point). After open, the log is
|
||||
* exactly the committed prefix.
|
||||
*/
|
||||
async open(committedGeneration: number): Promise<void> {
|
||||
const stored = (await this.storage.readRawObject(FACTS_MANIFEST_PATH)) as FactsManifest | null
|
||||
if (stored && typeof stored === 'object' && Array.isArray(stored.segments)) {
|
||||
|
|
|
|||
|
|
@ -12,9 +12,11 @@
|
|||
* the verified surface is a small set of rollup invariants (entity/
|
||||
* relationship counts) plus `sourceGeneration`.
|
||||
*
|
||||
* `sourceGeneration` is the generation of the source-of-truth log this
|
||||
* projection reflects — open-time coherence becomes a COMPARISON (stamp vs
|
||||
* log head), not a walk:
|
||||
* `sourceGeneration` is the COMMITTED generation of the source-of-truth log
|
||||
* this projection reflects — never the allocated counter, which names a
|
||||
* generation that may never commit (see {@link StampVerdict.torn}) — so
|
||||
* open-time coherence becomes a COMPARISON (stamp vs committed head), not a
|
||||
* walk:
|
||||
*
|
||||
* - equal + invariants hold → coherent, serve.
|
||||
* - behind → the projection missed the tail (crash between commit and stamp);
|
||||
|
|
@ -24,6 +26,9 @@
|
|||
* - invariants FAIL at equal generation → genuine incoherence: loud, and the
|
||||
* repair ritual (`repairIndex()`, whose recount rebuilds the rollups from a
|
||||
* canonical walk) heals it.
|
||||
* - AHEAD → a torn generation-log tail: the stamp's fsync outlived the log
|
||||
* tail's. TERMINAL, never a wait — the generation the stamp names does not
|
||||
* exist to arrive.
|
||||
*
|
||||
* Stamps are JSON on purpose — every incident gets debugged by reading a
|
||||
* stamp in a terminal.
|
||||
|
|
@ -70,6 +75,12 @@ export type StampVerdict =
|
|||
| { state: 'coherent' }
|
||||
| { state: 'absent' } // legacy store — first stamp writes at the next flush
|
||||
| { state: 'behind'; stampSource: number; head: number }
|
||||
/**
|
||||
* TORN GENERATION-LOG TAIL: the stamp witnesses a source generation the
|
||||
* store's committed watermark can no longer show. TERMINAL — there is no
|
||||
* generation to wait for, so the open demotes (or refuses) and never spins.
|
||||
*/
|
||||
| { state: 'torn'; stampSource: number; head: number }
|
||||
| { state: 'incoherent'; failures: string[] }
|
||||
| { state: 'unverifiable'; reason: string } // a FAULT reading the stamp — never conflated with absence
|
||||
|
||||
|
|
@ -118,12 +129,15 @@ export function verifyFamilyStamp(
|
|||
): StampVerdict {
|
||||
if (stamp === null) return { state: 'absent' }
|
||||
if (stamp.sourceGeneration > head) {
|
||||
// A stamp AHEAD of the log claims state that never committed — the
|
||||
// projection was stamped against truth that a crash rolled back.
|
||||
return {
|
||||
state: 'incoherent',
|
||||
failures: [`sourceGeneration ${stamp.sourceGeneration} is ahead of the log head ${head}`]
|
||||
}
|
||||
// A stamp AHEAD of committed truth witnesses a generation the store can no
|
||||
// longer show: the stamp's fsync survived a crash that the log tail did
|
||||
// not. This is the TORN GENERATION-LOG TAIL — its own class, never folded
|
||||
// in with `incoherent` (a count that drifted at a generation both sides
|
||||
// agree on), because the two have opposite cures: incoherence is recounted,
|
||||
// a tear is DEMOTED. It is also terminal by construction — there is no
|
||||
// generation the open can wait for, because the one the stamp names is
|
||||
// gone.
|
||||
return { state: 'torn', stampSource: stamp.sourceGeneration, head }
|
||||
}
|
||||
if (stamp.sourceGeneration < head) {
|
||||
return { state: 'behind', stampSource: stamp.sourceGeneration, head }
|
||||
|
|
|
|||
|
|
@ -147,6 +147,60 @@ export class GenerationSegmentStore {
|
|||
return this.coveringSegment(gen) !== null
|
||||
}
|
||||
|
||||
/**
|
||||
* @description True when `meta` declares more generations than it holds
|
||||
* frames — a segment sealed by a writer that folded across a hole. The
|
||||
* manifest records `frames` at fold time, so this is an O(1) comparison
|
||||
* against the declared span and needs no I/O.
|
||||
*/
|
||||
private isSparse(meta: SegmentMeta): boolean {
|
||||
return meta.lastGeneration - meta.firstGeneration + 1 !== meta.frames
|
||||
}
|
||||
|
||||
/**
|
||||
* @description The generations this tier ACTUALLY holds, as coalesced
|
||||
* ascending intervals — not what the segments declare.
|
||||
*
|
||||
* Dense segments (every one a current writer produces) contribute their
|
||||
* declared range with no I/O. A SPARSE segment — one sealed before the
|
||||
* density law was enforced, whose declared range spans generations it has
|
||||
* no frame for — has its real generation list read from its sidecar and
|
||||
* contributed instead, with the discrepancy narrated once.
|
||||
*
|
||||
* This is what keeps a store that already carries the damage from wedging.
|
||||
* `open()` seeds `committedRanges` from these intervals, so a hole is never
|
||||
* re-admitted as a committed generation, and the auto-compaction pass that
|
||||
* used to fail on every run with "packed history is damaged" simply never
|
||||
* asks for the missing frame.
|
||||
*
|
||||
* @returns Ascending, non-overlapping `[first, last]` intervals.
|
||||
*/
|
||||
async actualRanges(): Promise<Array<[number, number]>> {
|
||||
const out: Array<[number, number]> = []
|
||||
for (const meta of this.manifest.segments) {
|
||||
if (!this.isSparse(meta)) {
|
||||
out.push([meta.firstGeneration, meta.lastGeneration])
|
||||
continue
|
||||
}
|
||||
const missing = meta.lastGeneration - meta.firstGeneration + 1 - meta.frames
|
||||
prodLog.warn(
|
||||
`[GenerationSegments] sealed segment ${meta.file} declares generations ` +
|
||||
`${meta.firstGeneration}..${meta.lastGeneration} but holds only ${meta.frames} ` +
|
||||
`frame(s) — ${missing} generation(s) in that span were never folded into it. ` +
|
||||
`Serving the frames it actually holds; the declared span is not treated as ` +
|
||||
`committed history. (Written by a pre-density-law writer that folded across a ` +
|
||||
`gap; the segment itself is intact and no record is lost.)`
|
||||
)
|
||||
const idx = await this.sidecarFor(meta)
|
||||
for (const [gen] of idx.generations) {
|
||||
const last = out[out.length - 1]
|
||||
if (last !== undefined && gen === last[1] + 1) last[1] = gen
|
||||
else out.push([gen, gen])
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
/**
|
||||
* Fold consecutive generations into ONE new sealed segment + sidecar and
|
||||
* append it to the manifest atomically. Caller guarantees: `gens` is
|
||||
|
|
@ -164,6 +218,38 @@ export class GenerationSegmentStore {
|
|||
throw new Error('[GenerationSegments] fold() input must be strictly ascending')
|
||||
}
|
||||
}
|
||||
// THE DENSITY LAW, MADE MECHANICAL.
|
||||
//
|
||||
// A sealed segment declares a CONTIGUOUS range [firstGeneration,
|
||||
// lastGeneration] and every reader treats that range as containment:
|
||||
// `coveringSegment` is an interval test, `hasGeneration` returns true for
|
||||
// anything inside it, and `open()` seeds committedRanges from it. So a
|
||||
// segment folded from a SPARSE input silently claims generations it does
|
||||
// not hold, and the first read of one of those holes throws
|
||||
// "inside sealed segment ... but has no frame — packed history is damaged".
|
||||
//
|
||||
// That is exactly how the damage was produced. `repackHistory` skipped
|
||||
// generations mid-batch — ones absent from committedRanges, ones still in
|
||||
// the pending buffer, ones whose tx.json would not read — and handed the
|
||||
// survivors here, where the range was computed from the first and last of
|
||||
// them. Worse, the mis-declared range was then merged back into
|
||||
// committedRanges at the next open, which is what turned a quiet hole into
|
||||
// a repeating auto-compaction failure on every subsequent run.
|
||||
//
|
||||
// Callers now split at discontinuities; this refusal is what keeps any
|
||||
// future caller from reintroducing the class. A refusal here loses
|
||||
// nothing — the generations stay in the live tier, readable, and the next
|
||||
// pass folds them correctly.
|
||||
for (let i = 1; i < gens.length; i++) {
|
||||
if (gens[i].generation !== gens[i - 1].generation + 1) {
|
||||
throw new Error(
|
||||
`[GenerationSegments] fold() input is not contiguous: ${gens[i - 1].generation} → ` +
|
||||
`${gens[i].generation} skips ${gens[i].generation - gens[i - 1].generation - 1} ` +
|
||||
`generation(s). A sealed segment declares a dense range, so folding a sparse ` +
|
||||
`batch would claim generations it does not hold. Split the batch at the gap.`
|
||||
)
|
||||
}
|
||||
}
|
||||
const last = this.manifest.segments[this.manifest.segments.length - 1]
|
||||
if (last && gens[0].generation <= last.lastGeneration) {
|
||||
throw new Error(
|
||||
|
|
@ -364,12 +450,37 @@ export class GenerationSegmentStore {
|
|||
return this.decodeFrame(payload)
|
||||
}
|
||||
}
|
||||
// In the covering range but not present: the packed tier is dense by
|
||||
// construction (fold packs every generation it is handed, including
|
||||
// record-less ones) — absence inside a sealed range is damage.
|
||||
// Inside the covering range but with no frame. Two very different causes,
|
||||
// and conflating them is what made this class wedge every maintenance pass
|
||||
// on the affected stores.
|
||||
//
|
||||
// (1) A SPARSE SEGMENT — the manifest's own `frames` count is smaller than
|
||||
// the span it declares. That segment was sealed by a writer that
|
||||
// folded across a hole (the class this file's density law now bars).
|
||||
// The segment is INTACT and nothing is lost; it simply never held this
|
||||
// generation. Answering "not packed" is the honest answer, and it lets
|
||||
// the caller's two-tier read decide what a genuinely absent generation
|
||||
// means, instead of every compaction pass dying on a repeating throw.
|
||||
// `actualRanges()` keeps such holes out of committedRanges at open, so
|
||||
// in a healed store nobody asks this question in the first place.
|
||||
//
|
||||
// (2) A DENSE SEGMENT missing a frame it says it has — the manifest and
|
||||
// the sidecar disagree about a segment that claims to be complete.
|
||||
// That IS damage, and it stays loud.
|
||||
if (this.isSparse(meta)) {
|
||||
prodLog.warn(
|
||||
`[GenerationSegments] generation ${gen} falls inside sealed segment ${meta.file}'s ` +
|
||||
`declared range ${meta.firstGeneration}..${meta.lastGeneration}, but that segment ` +
|
||||
`holds ${meta.frames} frame(s) for a ${meta.lastGeneration - meta.firstGeneration + 1}` +
|
||||
`-generation span — it was sealed across a gap and never held this generation. ` +
|
||||
`Reporting it as unpacked rather than as damage; no record is lost.`
|
||||
)
|
||||
return null
|
||||
}
|
||||
throw new Error(
|
||||
`[GenerationSegments] generation ${gen} is inside sealed segment ${meta.file}'s declared ` +
|
||||
`range but has no frame — packed history is damaged`
|
||||
`range but has no frame, and that segment declares a complete ${meta.frames}-frame ` +
|
||||
`span — the manifest and the sidecar disagree; packed history is damaged`
|
||||
)
|
||||
}
|
||||
|
||||
|
|
|
|||
|
|
@ -32,7 +32,13 @@
|
|||
*/
|
||||
|
||||
import { prodLog } from '../utils/logger.js'
|
||||
import { GenerationCompactedError, GenerationConflictError, PendingFlushDurabilityError, StoreInconsistentError } from './errors.js'
|
||||
import {
|
||||
GenerationCompactedError,
|
||||
GenerationConflictError,
|
||||
PendingFlushDurabilityError,
|
||||
PendingSingleOpsUnflushedError,
|
||||
StoreInconsistentError
|
||||
} from './errors.js'
|
||||
import type { UnreconciledRecord } from './errors.js'
|
||||
import { TransactionRollbackError } from '../transaction/errors.js'
|
||||
import type {
|
||||
|
|
@ -96,6 +102,35 @@ export const FOLD_CHECKPOINT_PATH = '_system/fold-checkpoint.json'
|
|||
/** Storage-root-relative prefix of the per-generation record directories. */
|
||||
export const GENERATIONS_PREFIX = '_generations'
|
||||
|
||||
/**
|
||||
* @description Split an ascending list of fold candidates into maximal
|
||||
* CONTIGUOUS runs — `[7,8,9,12,13]` becomes `[[7,8,9],[12,13]]`.
|
||||
*
|
||||
* A sealed segment declares one dense range `[firstGeneration,
|
||||
* lastGeneration]`, and every reader treats that range as containment. So a
|
||||
* batch with a hole in it must never become one segment: it would claim a
|
||||
* generation it does not hold, and the first read of that hole reports the
|
||||
* packed history as damaged. One run, one segment — the ranges then describe
|
||||
* exactly what the segments contain.
|
||||
*
|
||||
* @param gens - Fold candidates, strictly ascending by generation.
|
||||
* @returns One array per contiguous run, in ascending order. Empty in, empty out.
|
||||
*/
|
||||
export function contiguousRuns(gens: FoldGeneration[]): FoldGeneration[][] {
|
||||
const runs: FoldGeneration[][] = []
|
||||
let run: FoldGeneration[] = []
|
||||
for (const g of gens) {
|
||||
const prev = run[run.length - 1]
|
||||
if (prev !== undefined && g.generation !== prev.generation + 1) {
|
||||
runs.push(run)
|
||||
run = []
|
||||
}
|
||||
run.push(g)
|
||||
}
|
||||
if (run.length > 0) runs.push(run)
|
||||
return runs
|
||||
}
|
||||
|
||||
/**
|
||||
* @description Phases of the {@link GenerationStore.commitTransaction} commit
|
||||
* protocol at which a test-only fault injector can simulate a process crash.
|
||||
|
|
@ -770,7 +805,16 @@ export class GenerationStore {
|
|||
if (uncleanOpen) await this.advanceFoldCheckpointUnlocked()
|
||||
// The marker is consumed: any session that can write invalidates it
|
||||
// at first commit (see the commit paths); a clean close re-writes it.
|
||||
await this.clearCleanShutdownMarker()
|
||||
// A READER NEVER CONSUMES IT. The marker is the writer's own evidence
|
||||
// about the writer's own process — clearing it here exists so that
|
||||
// if THIS session goes on to write and then dies before its next
|
||||
// clean close, the marker's absence correctly reads as unclean. A
|
||||
// reader can never write, so it can never leave the store in a state
|
||||
// its own crash would mis-describe; clearing the marker for it would
|
||||
// only cost the store's actual writer a needless whole-log fold on
|
||||
// its next open, for a generation the reader merely observed. Leave
|
||||
// `_system/` exactly as found.
|
||||
if (!options?.readOnly) await this.clearCleanShutdownMarker()
|
||||
}
|
||||
await this.factLog.open(this.committed)
|
||||
} else {
|
||||
|
|
@ -784,9 +828,15 @@ export class GenerationStore {
|
|||
if (storageSupportsFactLog(this.storage)) {
|
||||
this.segments = new GenerationSegmentStore(this.storage)
|
||||
await this.segments.open()
|
||||
const packedRanges = this.segments
|
||||
.segments()
|
||||
.map((s): [number, number] => [s.firstGeneration, Math.min(s.lastGeneration, this.committed)])
|
||||
// ACTUAL ranges, not declared ones. A segment sealed by a pre-density-law
|
||||
// writer can declare a span wider than the frames it holds; seeding
|
||||
// committedRanges from the declared span re-admits those holes as
|
||||
// committed generations, and every later maintenance pass then asks for a
|
||||
// frame that was never written. `actualRanges()` reads the real
|
||||
// generation list from the sidecar for exactly those segments (and does
|
||||
// no I/O for the dense ones, which is all of them on a healthy store).
|
||||
const packedRanges = (await this.segments.actualRanges())
|
||||
.map((r): [number, number] => [r[0], Math.min(r[1], this.committed)])
|
||||
.filter(([lo, hi]) => lo <= hi)
|
||||
if (packedRanges.length > 0) {
|
||||
// Merge packed (older) + live (newer) interval sets — both ascending;
|
||||
|
|
@ -854,7 +904,11 @@ export class GenerationStore {
|
|||
}
|
||||
}
|
||||
|
||||
/** Consume the clean-shutdown marker (every open; a clean close re-writes it). */
|
||||
/**
|
||||
* Consume the clean-shutdown marker (every WRITER open; a clean close
|
||||
* re-writes it). Callers must gate this on `!options.readOnly` — a reader
|
||||
* never consumes the marker, see the call site in {@link open}.
|
||||
*/
|
||||
private async clearCleanShutdownMarker(): Promise<void> {
|
||||
try {
|
||||
await this.storage.deleteRawObject(CLEAN_SHUTDOWN_PATH)
|
||||
|
|
@ -1316,6 +1370,9 @@ export class GenerationStore {
|
|||
* @param args.execute - Runs the planned operation batch atomically.
|
||||
* @returns The committed generation and its commit timestamp.
|
||||
* @throws GenerationConflictError when the CAS expectation fails.
|
||||
* @throws PendingSingleOpsUnflushedError when the pending single-op tier is
|
||||
* non-empty — call `flushPendingSingleOps()` first (both `Brainy.transact()`
|
||||
* and `Brainy.compactHistory()` already do).
|
||||
*/
|
||||
/**
|
||||
* The generation fact log, or `null` when the storage layer cannot host one.
|
||||
|
|
@ -1390,6 +1447,13 @@ export class GenerationStore {
|
|||
execute: () => Promise<void>
|
||||
}): Promise<{ generation: number; timestamp: number }> {
|
||||
return this.withMutex(async () => {
|
||||
// The generation-order guard (see assertPendingSingleOpsFlushed): a
|
||||
// direct commitTransaction() call while single-ops are still pending
|
||||
// would commit above them, unsorting reservedGensAsc() and corrupting
|
||||
// point-in-time reads. Both sanctioned callers (Brainy.transact(),
|
||||
// Brainy.compactHistory()) already flush first, so this is
|
||||
// behavior-neutral on every real path.
|
||||
this.assertPendingSingleOpsFlushed()
|
||||
// A latched history-durability failure compromises the whole generation
|
||||
// chain — refuse a transact too (advancing the manifest past stuck,
|
||||
// un-durable single-op generations would be inconsistent). Same loud
|
||||
|
|
@ -2259,6 +2323,37 @@ export class GenerationStore {
|
|||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* @description Throw if the pending single-op tier is non-empty. Called at
|
||||
* the top of {@link commitTransaction} (the ONLY method that appends a
|
||||
* fresh commit directly into {@link committedRanges} outside recovery) so
|
||||
* the ordering invariant {@link reservedGensAsc}'s own doc comment states —
|
||||
* "pending generations are always greater than every committed one" — is
|
||||
* ENFORCED there rather than merely assumed.
|
||||
*
|
||||
* That invariant holds today only because both sanctioned callers flush the
|
||||
* pending tier before committing: `Brainy.transact()` (src/brainy.ts,
|
||||
* `await this.generationStore.flushPendingSingleOps()` immediately before
|
||||
* its `commitTransaction()` call) and `Brainy.compactHistory()`
|
||||
* (src/brainy.ts, the same flush immediately before its `compact()` call —
|
||||
* `compact()` itself only ever RECLAIMS an existing committed prefix, so it
|
||||
* cannot land a commit out of order and needs no guard of its own). A
|
||||
* caller that reaches `commitTransaction()` by any other path — bypassing
|
||||
* that flush — would commit a new generation into `committedRanges` ABOVE
|
||||
* generations still sitting in `pendingGens`, breaking `reservedGensAsc`'s
|
||||
* "committed-then-pending is already sorted" assumption and making
|
||||
* `resolveManyAt`'s single ascending pass (and `resolveAt`'s consumers)
|
||||
* return the WRONG before-image for a point-in-time read — silently, no
|
||||
* compensating error. Refusing here, before any staging I/O, keeps the
|
||||
* store untouched (nothing committed, nothing staged, the generation
|
||||
* counter reservation unaffected) on every path that already flushes.
|
||||
*/
|
||||
private assertPendingSingleOpsFlushed(): void {
|
||||
if (this.pendingGens.length > 0) {
|
||||
throw new PendingSingleOpsUnflushedError(this.pendingGens.length)
|
||||
}
|
||||
}
|
||||
|
||||
/** Schedule a coalesced pending-tier flush (size trigger fires immediately on
|
||||
* the next microtask; otherwise a {@link PENDING_FLUSH_DELAY_MS} timer). Both
|
||||
* defer outside the current mutex section so the flush can re-acquire it. A
|
||||
|
|
@ -2342,6 +2437,13 @@ export class GenerationStore {
|
|||
* committed-then-pending concatenation is already sorted — identical to the old
|
||||
* `[...committedGens, ...pendingGens]`. This is the union historical reads
|
||||
* resolve over so un-flushed single-ops are visible to pins/`asOf`.
|
||||
*
|
||||
* The "flush first" half of that invariant is ENFORCED, not just documented:
|
||||
* {@link commitTransaction} — the only method that lands a fresh commit into
|
||||
* {@link committedRanges} outside crash recovery — refuses via
|
||||
* {@link assertPendingSingleOpsFlushed} whenever {@link pendingGens} is
|
||||
* non-empty, so a committed generation can never land above a still-pending
|
||||
* one and break this ordering.
|
||||
*/
|
||||
private *reservedGensAsc(): IterableIterator<number> {
|
||||
yield* this.committedGensAsc()
|
||||
|
|
@ -3121,13 +3223,26 @@ export class GenerationStore {
|
|||
foldInput.push({ generation: gen, timestamp: delta.timestamp, delta, records })
|
||||
}
|
||||
if (foldInput.length === 0) continue
|
||||
await segments.fold(foldInput)
|
||||
segmentsCreated++
|
||||
// Segment + manifest durable → the live copies retire.
|
||||
for (const g of foldInput) {
|
||||
await this.storage.removeRawPrefix(`${GENERATIONS_PREFIX}/${g.generation}`)
|
||||
// SPLIT AT DISCONTINUITIES. `eligible` is NOT contiguous — three
|
||||
// filters above punch holes in it: a generation missing from
|
||||
// committedRanges never appears, one still in the pending buffer is
|
||||
// skipped, and one whose tx.json will not read is skipped. A sealed
|
||||
// segment declares a DENSE range, so folding across such a hole makes
|
||||
// the segment claim a generation it does not hold; the next open
|
||||
// merges that mis-declared range into committedRanges, and every
|
||||
// subsequent auto-compaction pass then asks for the missing frame and
|
||||
// fails with "packed history is damaged". Fold each contiguous RUN as
|
||||
// its own segment instead — same bytes, honest ranges.
|
||||
for (const run of contiguousRuns(foldInput)) {
|
||||
if (deadline !== undefined && Date.now() >= deadline) break
|
||||
await segments.fold(run)
|
||||
segmentsCreated++
|
||||
// Segment + manifest durable → the live copies retire.
|
||||
for (const g of run) {
|
||||
await this.storage.removeRawPrefix(`${GENERATIONS_PREFIX}/${g.generation}`)
|
||||
}
|
||||
folded += run.length
|
||||
}
|
||||
folded += foldInput.length
|
||||
}
|
||||
if (folded > 0) {
|
||||
prodLog.info(
|
||||
|
|
|
|||
|
|
@ -231,7 +231,8 @@ export {
|
|||
GenerationCompactedError,
|
||||
StoreInconsistentError,
|
||||
PendingFlushDurabilityError,
|
||||
CanonicalEnumerationUnavailableError
|
||||
CanonicalEnumerationUnavailableError,
|
||||
PendingSingleOpsUnflushedError
|
||||
} from './db/errors.js'
|
||||
export type { UnreconciledRecord } from './db/errors.js'
|
||||
export type {
|
||||
|
|
|
|||
|
|
@ -2,7 +2,7 @@
|
|||
* 🧠 BRAINY EMBEDDED TYPE EMBEDDINGS
|
||||
*
|
||||
* AUTO-GENERATED - DO NOT EDIT
|
||||
* Generated: 2026-06-29T10:04:19-07:00
|
||||
* Generated: 2026-08-27T09:18:45-07:00
|
||||
* Noun Types: 42
|
||||
* Verb Types: 127
|
||||
*
|
||||
|
|
@ -19,7 +19,7 @@ export const TYPE_METADATA = {
|
|||
verbTypes: 127,
|
||||
totalTypes: 169,
|
||||
embeddingDimensions: 384,
|
||||
generatedAt: "2026-06-29T10:04:19-07:00",
|
||||
generatedAt: "2026-08-27T09:18:45-07:00",
|
||||
sizeBytes: {
|
||||
embeddings: 259584,
|
||||
base64: 346112
|
||||
|
|
|
|||
|
|
@ -411,7 +411,90 @@ export interface MetadataIndexProvider {
|
|||
* @returns The matching id universe as an opaque set.
|
||||
*/
|
||||
getIdSetForFilter?(filter: any): Promise<OpaqueIdSet>
|
||||
/**
|
||||
* @description OPTIONAL: evaluate `filter` over `ids` ONLY and return the
|
||||
* survivors in the caller's order — the door a graph-first
|
||||
* `find({ connected, where })` walks. The neighbour set is the universe there,
|
||||
* so the filter must cost O(|ids|) membership checks, never a whole-store
|
||||
* materialization. A native index answers from its roaring filter result
|
||||
* (membership by entity int); the reference index answers from its own
|
||||
* `getIdsForFilter`, so the two doors can never disagree. Absent → Brainy
|
||||
* intersects `getIdsForFilter`'s answer with `ids` itself (correct, O(store)).
|
||||
* @param filter - The same filter shape accepted by `getIdsForFilter`.
|
||||
* @param ids - The candidate ids (canonical). The answer is a subsequence.
|
||||
*/
|
||||
filterIdsWithin?(filter: any, ids: readonly string[]): Promise<string[]>
|
||||
/**
|
||||
* @description OPTIONAL: plan and execute a WHOLE `find()` — the graph
|
||||
* traversal, the metadata filter, the ordering and the page — and answer the
|
||||
* page's ids, or `null` for a shape this index does not plan.
|
||||
*
|
||||
* The doors above each serve one stage, so a `find()` that consults three of
|
||||
* them crosses into the index three times and marshals a result set at every
|
||||
* crossing. An index that can decide the stage ORDER itself does the whole
|
||||
* thing in one call and materializes ids only for the page — a filter
|
||||
* matching a hundred thousand rows then builds twenty-five id strings instead
|
||||
* of a hundred thousand.
|
||||
*
|
||||
* The contract this door must keep, because Brainy cannot check it:
|
||||
*
|
||||
* - **The same answer.** Identical rows, in identical order, to what the
|
||||
* stage doors would have produced for the same params. This door changes
|
||||
* which code runs, never what the answer is.
|
||||
* - **The law of the stages** (`find({ connected })` is graph-first): the
|
||||
* neighbour set is the candidate universe, the filter is evaluated over
|
||||
* those ids only, `orderBy` sorts the whole candidate set, and the page is
|
||||
* cut LAST.
|
||||
* - **`null` before work, not instead of an answer.** A shape the index does
|
||||
* not plan must be handed back BEFORE any evaluation, so Brainy serves it
|
||||
* through the stage doors exactly as it always has. Returning `null` after
|
||||
* partial work, or an empty page for a shape it could not evaluate, is a
|
||||
* silent wrong answer.
|
||||
* - **`emptyAt` names the stage** that produced an empty page — `'graph'`,
|
||||
* `'filter'`, `'visibility'` or `'none'` — so Brainy can apply its serving
|
||||
* law to the right index. An empty answer from an index that is not
|
||||
* serving must refuse loudly, and Brainy can only re-verify what it is told.
|
||||
*
|
||||
* Absent → every `find()` is served by the stage doors, which is Brainy's
|
||||
* own behaviour and the ordering oracle for any implementation of this one.
|
||||
* @param params - The find params, already normalized by `find()`
|
||||
* (natural-language parsed, `connected` anchors resolved to canonical ids,
|
||||
* an empty `where` dropped).
|
||||
* @param hiddenIds - Ids this read must not return. The contract is the ANSWER, not the
|
||||
* mechanism: a provider may subtract this set before paging, or derive the
|
||||
* same exclusion from the params' visibility tiers itself — either way the
|
||||
* page must equal the engine's own answer with none of these ids in it.
|
||||
* @param graphIndex - The active graph provider, for a `connected` plan.
|
||||
* @returns The page's ids plus the stage that emptied it, or `null`.
|
||||
*/
|
||||
planFindPage?(
|
||||
params: any,
|
||||
hiddenIds: readonly string[],
|
||||
graphIndex: unknown
|
||||
): Promise<{ ids: string[]; emptyAt: 'graph' | 'filter' | 'visibility' | 'none' } | null>
|
||||
getIdsForTextQuery(query: string): Promise<Array<{ id: string; matchCount: number }>>
|
||||
/**
|
||||
* @description OPTIONAL: score `query` over `ids` ONLY — the text-leg twin of
|
||||
* {@link filterIdsWithin}, and the door a hybrid `find({ query, where })`
|
||||
* walks. The metadata filter's universe is the candidate set there, so the
|
||||
* text leg must cost O(|ids|) membership checks and marshal at most `|ids|`
|
||||
* rows, never the whole posting list of every query word. A native index
|
||||
* intersects its own postings with the candidate set (membership by entity
|
||||
* int) before any string crosses the boundary; the reference index answers
|
||||
* from its own `getIdsForTextQuery`, so the two doors can never disagree.
|
||||
* Absent → Brainy intersects `getIdsForTextQuery`'s answer with `ids` itself
|
||||
* (correct, and still hydrate-last, but it marshals the whole answer).
|
||||
*
|
||||
* The answer keeps `getIdsForTextQuery`'s contract: `{ id, matchCount }`
|
||||
* sorted by `matchCount` descending, ties in the order the whole-store answer
|
||||
* would have produced. Only rows in `ids` may appear.
|
||||
* @param query - The same text query accepted by `getIdsForTextQuery`.
|
||||
* @param ids - The candidate ids (canonical). The answer is a subset.
|
||||
*/
|
||||
getIdsForTextQueryWithin?(
|
||||
query: string,
|
||||
ids: readonly string[]
|
||||
): Promise<Array<{ id: string; matchCount: number }>>
|
||||
getSortedIdsForFilter(filter: any, orderBy: string, order?: 'asc' | 'desc', topK?: number): Promise<string[]>
|
||||
getFilterValues(field: string): Promise<string[]>
|
||||
getFilterFields(): Promise<string[]>
|
||||
|
|
|
|||
|
|
@ -1089,6 +1089,10 @@ export abstract class BaseStorageAdapter implements StorageAdapter {
|
|||
|
||||
// Counts changed since the last persist? Drives the write-through flush.
|
||||
protected pendingCountPersist = false
|
||||
/** The one persist running right now, if any (single-flight law — see flushCounts). */
|
||||
private countPersistInFlight: Promise<void> | null = null
|
||||
/** The one trailing persist a burst has queued behind the in-flight one. */
|
||||
private countPersistTrailing: Promise<void> | null = null
|
||||
|
||||
/**
|
||||
* Get total noun count - O(1) operation
|
||||
|
|
@ -1341,15 +1345,46 @@ export abstract class BaseStorageAdapter implements StorageAdapter {
|
|||
return
|
||||
}
|
||||
|
||||
try {
|
||||
// Persist to storage (implemented by subclass)
|
||||
await this.persistCounts()
|
||||
this.pendingCountPersist = false
|
||||
} catch (error) {
|
||||
console.error('CRITICAL: Failed to flush counts to storage:', error)
|
||||
// Keep pending flag set so we retry on next operation
|
||||
throw error
|
||||
// SINGLE-FLIGHT, COALESCED. Counts are write-through on every change, so
|
||||
// a burst of writes used to launch one persist per change, all in flight
|
||||
// together. Two of them inside the same millisecond shared the atomic
|
||||
// writer's temp path (`.tmp-<pid>-<ms>`): both wrote it, the first rename
|
||||
// consumed it, the second rename found nothing — ENOENT, ~1,500 times a
|
||||
// day on a busy production brain, with a full ledger write per change
|
||||
// behind it. Now exactly one persist runs at a time; requests that arrive
|
||||
// while it runs collapse into ONE trailing persist that carries the final
|
||||
// state. A burst of N changes costs at most two writes and never races
|
||||
// itself.
|
||||
if (this.countPersistInFlight) {
|
||||
// The in-flight write may have already serialised a stale snapshot —
|
||||
// ask for one more pass after it, and let every caller in this burst
|
||||
// await that same pass.
|
||||
if (!this.countPersistTrailing) {
|
||||
this.countPersistTrailing = this.countPersistInFlight
|
||||
.catch(() => undefined)
|
||||
.then(() => {
|
||||
this.countPersistTrailing = null
|
||||
return this.flushCounts()
|
||||
})
|
||||
}
|
||||
return this.countPersistTrailing
|
||||
}
|
||||
|
||||
this.countPersistInFlight = (async () => {
|
||||
try {
|
||||
// Persist to storage (implemented by subclass)
|
||||
this.pendingCountPersist = false
|
||||
await this.persistCounts()
|
||||
} catch (error) {
|
||||
// Keep the flag set so the next operation retries.
|
||||
this.pendingCountPersist = true
|
||||
console.error('CRITICAL: Failed to flush counts to storage:', error)
|
||||
throw error
|
||||
} finally {
|
||||
this.countPersistInFlight = null
|
||||
}
|
||||
})()
|
||||
return this.countPersistInFlight
|
||||
}
|
||||
|
||||
/**
|
||||
|
|
|
|||
|
|
@ -2400,8 +2400,15 @@ export class FileSystemStorage extends BaseStorage {
|
|||
* Atomic write via temp-file-then-rename so concurrent readers never see a
|
||||
* half-written lock JSON. Reused by writer-lock writes + heartbeat.
|
||||
*/
|
||||
/** Monotonic per-process sequence so two atomic writes never share a temp path. */
|
||||
private static atomicWriteSeq = 0
|
||||
|
||||
private async writeFileAtomic(filePath: string, contents: string): Promise<void> {
|
||||
const tmp = `${filePath}.tmp-${process.pid}-${Date.now()}`
|
||||
// pid + timestamp alone collided: two writers of the same target inside
|
||||
// one millisecond shared this path, and the loser's rename found the
|
||||
// winner had already moved it (ENOENT). The sequence makes every call's
|
||||
// temp path its own.
|
||||
const tmp = `${filePath}.tmp-${process.pid}-${Date.now()}-${++FileSystemStorage.atomicWriteSeq}`
|
||||
await fs.promises.writeFile(tmp, contents)
|
||||
await fs.promises.rename(tmp, filePath)
|
||||
}
|
||||
|
|
|
|||
|
|
@ -2942,19 +2942,33 @@ export abstract class BaseStorage extends BaseStorageAdapter {
|
|||
!options.filter.service &&
|
||||
!options.filter.metadata
|
||||
) {
|
||||
const sourceId = Array.isArray(options.filter.sourceId)
|
||||
? options.filter.sourceId[0]
|
||||
: options.filter.sourceId
|
||||
const sourceIds = Array.isArray(options.filter.sourceId)
|
||||
? options.filter.sourceId
|
||||
: [options.filter.sourceId]
|
||||
|
||||
const verbType = Array.isArray(options.filter.verbType)
|
||||
? options.filter.verbType[0]
|
||||
: options.filter.verbType
|
||||
// EVERY requested verb type is honoured — an array used to collapse to
|
||||
// its first element here, silently dropping the rest of the ask.
|
||||
const verbTypes = new Set(
|
||||
Array.isArray(options.filter.verbType)
|
||||
? options.filter.verbType
|
||||
: [options.filter.verbType]
|
||||
)
|
||||
|
||||
// Get verbs by source, then filter by type (O(1) graph lookup + O(n) type filter),
|
||||
// then apply the subtype / visibility metadata filters on the candidate set.
|
||||
const verbsBySource = await this.getVerbsBySource_internal(sourceId)
|
||||
// Get verbs by source (union over every requested source), filter by the
|
||||
// requested type SET (O(1) graph lookup + O(n) type filter), then apply
|
||||
// the subtype / visibility metadata filters on the candidate set.
|
||||
const bySource: HNSWVerbWithMetadata[] = []
|
||||
const seenVerbIds = new Set<string>()
|
||||
for (const oneSource of sourceIds) {
|
||||
for (const v of await this.getVerbsBySource_internal(oneSource)) {
|
||||
if (!seenVerbIds.has(v.id)) {
|
||||
seenVerbIds.add(v.id)
|
||||
bySource.push(v)
|
||||
}
|
||||
}
|
||||
}
|
||||
const filteredVerbs = this.applyVerbMetadataFilters(
|
||||
verbsBySource.filter(v => v.verb === verbType),
|
||||
bySource.filter(v => verbTypes.has(v.verb)),
|
||||
options.filter
|
||||
)
|
||||
|
||||
|
|
@ -2985,16 +2999,22 @@ export abstract class BaseStorage extends BaseStorageAdapter {
|
|||
!options.filter.service &&
|
||||
!options.filter.metadata
|
||||
) {
|
||||
const sourceId = Array.isArray(options.filter.sourceId)
|
||||
? options.filter.sourceId[0]
|
||||
: options.filter.sourceId
|
||||
|
||||
// Get verbs by source directly (hydrated with metadata), then apply the
|
||||
// subtype / visibility metadata filters on the O(degree) candidate set.
|
||||
const verbsBySource = this.applyVerbMetadataFilters(
|
||||
await this.getVerbsBySource_internal(sourceId),
|
||||
options.filter
|
||||
)
|
||||
// EVERY requested source is honoured — an array used to collapse to
|
||||
// its first element here, silently dropping the rest of the ask.
|
||||
const onlySourceIds = Array.isArray(options.filter.sourceId)
|
||||
? options.filter.sourceId
|
||||
: [options.filter.sourceId]
|
||||
const sourceUnion: HNSWVerbWithMetadata[] = []
|
||||
const seenSourceVerbIds = new Set<string>()
|
||||
for (const oneSource of onlySourceIds) {
|
||||
for (const v of await this.getVerbsBySource_internal(oneSource)) {
|
||||
if (!seenSourceVerbIds.has(v.id)) {
|
||||
seenSourceVerbIds.add(v.id)
|
||||
sourceUnion.push(v)
|
||||
}
|
||||
}
|
||||
}
|
||||
const verbsBySource = this.applyVerbMetadataFilters(sourceUnion, options.filter)
|
||||
|
||||
// Apply pagination
|
||||
const paginatedVerbs = verbsBySource.slice(offset, offset + limit)
|
||||
|
|
@ -3023,16 +3043,22 @@ export abstract class BaseStorage extends BaseStorageAdapter {
|
|||
!options.filter.service &&
|
||||
!options.filter.metadata
|
||||
) {
|
||||
const targetId = Array.isArray(options.filter.targetId)
|
||||
? options.filter.targetId[0]
|
||||
: options.filter.targetId
|
||||
|
||||
// Get verbs by target directly (hydrated with metadata), then apply the
|
||||
// subtype / visibility metadata filters on the O(degree) candidate set.
|
||||
const verbsByTarget = this.applyVerbMetadataFilters(
|
||||
await this.getVerbsByTarget_internal(targetId),
|
||||
options.filter
|
||||
)
|
||||
// EVERY requested target is honoured — an array used to collapse to
|
||||
// its first element here, silently dropping the rest of the ask.
|
||||
const onlyTargetIds = Array.isArray(options.filter.targetId)
|
||||
? options.filter.targetId
|
||||
: [options.filter.targetId]
|
||||
const targetUnion: HNSWVerbWithMetadata[] = []
|
||||
const seenTargetVerbIds = new Set<string>()
|
||||
for (const oneTarget of onlyTargetIds) {
|
||||
for (const v of await this.getVerbsByTarget_internal(oneTarget)) {
|
||||
if (!seenTargetVerbIds.has(v.id)) {
|
||||
seenTargetVerbIds.add(v.id)
|
||||
targetUnion.push(v)
|
||||
}
|
||||
}
|
||||
}
|
||||
const verbsByTarget = this.applyVerbMetadataFilters(targetUnion, options.filter)
|
||||
|
||||
// Apply pagination
|
||||
const paginatedVerbs = verbsByTarget.slice(offset, offset + limit)
|
||||
|
|
@ -3061,16 +3087,25 @@ export abstract class BaseStorage extends BaseStorageAdapter {
|
|||
!options.filter.service &&
|
||||
!options.filter.metadata
|
||||
) {
|
||||
const verbType = Array.isArray(options.filter.verbType)
|
||||
? options.filter.verbType[0]
|
||||
: options.filter.verbType
|
||||
// EVERY requested verb type is honoured — an array used to collapse to
|
||||
// its first element here, silently dropping the rest of the ask.
|
||||
const verbTypes = Array.isArray(options.filter.verbType)
|
||||
? options.filter.verbType
|
||||
: [options.filter.verbType]
|
||||
|
||||
// Get verbs by type directly (hydrated with metadata), then apply the
|
||||
// subtype / visibility metadata filters on the candidate set.
|
||||
const verbsByType = this.applyVerbMetadataFilters(
|
||||
await this.getVerbsByType_internal(verbType),
|
||||
options.filter
|
||||
)
|
||||
// Get verbs by each requested type (hydrated with metadata), deduped by
|
||||
// id, then apply the subtype / visibility metadata filters on the set.
|
||||
const byType: HNSWVerbWithMetadata[] = []
|
||||
const seenTypeVerbIds = new Set<string>()
|
||||
for (const oneType of verbTypes) {
|
||||
for (const v of await this.getVerbsByType_internal(oneType)) {
|
||||
if (!seenTypeVerbIds.has(v.id)) {
|
||||
seenTypeVerbIds.add(v.id)
|
||||
byType.push(v)
|
||||
}
|
||||
}
|
||||
}
|
||||
const verbsByType = this.applyVerbMetadataFilters(byType, options.filter)
|
||||
|
||||
// Apply pagination
|
||||
const paginatedVerbs = verbsByType.slice(offset, offset + limit)
|
||||
|
|
|
|||
|
|
@ -14,6 +14,7 @@ import type { MetadataIndexManager } from '../../utils/metadataIndex.js'
|
|||
import type { GraphVerb } from '../../coreTypes.js'
|
||||
import type { Operation, RollbackAction } from '../types.js'
|
||||
import { isZeroNormVector } from '../../utils/distance.js'
|
||||
import { jsonSafeIndexMetadata } from '../../utils/jsonSafeIndexMetadata.js'
|
||||
import { prodLog } from '../../utils/logger.js'
|
||||
|
||||
/**
|
||||
|
|
@ -390,13 +391,21 @@ export class AddToMetadataIndexOperation implements Operation {
|
|||
// rollback so add + undo reference the same watermark.
|
||||
const generation = this.generationFn?.()
|
||||
|
||||
// Add to metadata index (skipFlush=true for transaction atomicity)
|
||||
await this.index.addToIndex(this.id, this.entity, true, false, generation)
|
||||
// The JSON-safe view is taken HERE, per crossing, never at construction:
|
||||
// the entity reference this op holds can be mutated between plan and
|
||||
// execute (a graph op's execute-time endpoint-int resolution mirrors
|
||||
// BigInts onto a shared verb object) — see jsonSafeIndexMetadata's
|
||||
// module doc.
|
||||
await this.index.addToIndex(
|
||||
this.id, jsonSafeIndexMetadata(this.entity), true, false, generation
|
||||
)
|
||||
|
||||
// Return rollback action
|
||||
return async () => {
|
||||
// Remove from metadata index
|
||||
await this.index.removeFromIndex(this.id, this.entity, generation)
|
||||
await this.index.removeFromIndex(
|
||||
this.id, jsonSafeIndexMetadata(this.entity), generation
|
||||
)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
|
@ -432,13 +441,21 @@ export class RemoveFromMetadataIndexOperation implements Operation {
|
|||
// Resolve the removal generation once; reuse it for the rollback re-add.
|
||||
const generation = this.generationFn?.()
|
||||
|
||||
// Remove from metadata index
|
||||
await this.index.removeFromIndex(this.id, this.entity, generation)
|
||||
// Sanitized per crossing, never at construction — transact()'s delete
|
||||
// legs hand this op the SAME verb object the graph-retraction op's
|
||||
// execute-time endpoint resolution mutates (BigInt sourceInt/targetInt),
|
||||
// so a plan-time view aliases the pollution. See jsonSafeIndexMetadata's
|
||||
// module doc.
|
||||
await this.index.removeFromIndex(
|
||||
this.id, jsonSafeIndexMetadata(this.entity), generation
|
||||
)
|
||||
|
||||
// Return rollback action
|
||||
return async () => {
|
||||
// Re-add with original metadata (skipFlush=true)
|
||||
await this.index.addToIndex(this.id, this.entity, true, false, generation)
|
||||
await this.index.addToIndex(
|
||||
this.id, jsonSafeIndexMetadata(this.entity), true, false, generation
|
||||
)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
|
|
|||
47
src/utils/jsonSafeIndexMetadata.ts
Normal file
47
src/utils/jsonSafeIndexMetadata.ts
Normal file
|
|
@ -0,0 +1,47 @@
|
|||
/**
|
||||
* @module utils/jsonSafeIndexMetadata
|
||||
* @description The metadata-index crossing's JSON-safety law, as a leaf
|
||||
* function both the coordinator and the transaction operations share.
|
||||
*
|
||||
* The seam's metadata is JSON-safe BY CONTRACT (a native provider serializes
|
||||
* it; u64 ints as Number corrupt above 2^53) — but `resolveVerbEndpointInts`
|
||||
* MIRRORS the resolved endpoint ints onto the verb object itself as BigInt
|
||||
* (`verb.sourceInt`/`targetInt`), so a verb object reused as index metadata
|
||||
* carries BigInts into JSON.stringify, which throws, aborting the whole
|
||||
* transaction. Endpoint ints ride their OWN op params on the graph legs — the
|
||||
* metadata crossing drops every BigInt-valued top-level key instead of
|
||||
* guessing at a lossy numeric encoding.
|
||||
*
|
||||
* WHY THIS IS A LEAF MODULE, ENFORCED AT THE CROSSING: sanitizing only at
|
||||
* operation-construction time is not enough. `transact()`'s delete legs pass
|
||||
* the SAME verb object to both the graph-retraction op (whose endpoint-int
|
||||
* thunk deliberately resolves at EXECUTE time, for same-batch forward refs)
|
||||
* and the metadata-retraction op. At plan time the verb is still clean, so a
|
||||
* plan-time sanitize returns the same reference — then the graph op executes
|
||||
* first, mirrors the BigInt ints onto the shared object, and the metadata op
|
||||
* crosses the seam with them (found by the first fleet adoption of the native
|
||||
* pair: every transact-wrapped edge delete aborted). The crossing itself is
|
||||
* the only place ordering cannot bypass.
|
||||
*/
|
||||
|
||||
/**
|
||||
* A JSON-safe view of a record bound for the metadata-index crossing.
|
||||
*
|
||||
* @param metadata - The candidate index-metadata record.
|
||||
* @returns The same object when already JSON-safe, else a shallow copy
|
||||
* without the BigInt-valued keys.
|
||||
*/
|
||||
export function jsonSafeIndexMetadata(metadata: unknown): unknown {
|
||||
if (metadata === null || typeof metadata !== 'object') return metadata
|
||||
const rec = metadata as Record<string, unknown>
|
||||
let hasBigint = false
|
||||
for (const k in rec) {
|
||||
if (typeof rec[k] === 'bigint') { hasBigint = true; break }
|
||||
}
|
||||
if (!hasBigint) return metadata
|
||||
const out: Record<string, unknown> = {}
|
||||
for (const k in rec) {
|
||||
if (typeof rec[k] !== 'bigint') out[k] = rec[k]
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
|
@ -1509,11 +1509,56 @@ export class MetadataIndexManager implements MetadataIndexProvider {
|
|||
* @returns Array of { id, matchCount } sorted by matchCount descending
|
||||
*/
|
||||
async getIdsForTextQuery(query: string): Promise<Array<{ id: string; matchCount: number }>> {
|
||||
return this.scoreTextQuery(query)
|
||||
}
|
||||
|
||||
/**
|
||||
* Score a text query over `ids` ONLY — the reference implementation of the
|
||||
* optional `getIdsForTextQueryWithin` door (see
|
||||
* {@link import('../plugin.js').MetadataIndexProvider}). The hybrid
|
||||
* `find({ query, where })` path passes the metadata filter's universe here so
|
||||
* the text leg ranks INSIDE that universe instead of ranking the whole store
|
||||
* and discarding the rows the filter would have dropped.
|
||||
*
|
||||
* It answers from the same posting-list merge as {@link getIdsForTextQuery},
|
||||
* with the candidate membership applied as each word's postings are counted,
|
||||
* so the two doors can never disagree: the answer is exactly the whole-store
|
||||
* answer restricted to `ids`, in the same order.
|
||||
*
|
||||
* @param query - Text query to search for.
|
||||
* @param ids - Candidate entity ids; only these may appear in the answer.
|
||||
* @returns Array of { id, matchCount } sorted by matchCount descending.
|
||||
*/
|
||||
async getIdsForTextQueryWithin(
|
||||
query: string,
|
||||
ids: readonly string[]
|
||||
): Promise<Array<{ id: string; matchCount: number }>> {
|
||||
if (ids.length === 0) return []
|
||||
return this.scoreTextQuery(query, new Set(ids))
|
||||
}
|
||||
|
||||
/**
|
||||
* The one posting-list merge behind both text doors.
|
||||
*
|
||||
* Each query word contributes AT MOST one match per entity (a posting list
|
||||
* can name an id more than once), and entities are ranked by how many of the
|
||||
* query's words they matched. `within`, when given, restricts the count to
|
||||
* those candidates — applied during the merge, so a restricted call never
|
||||
* materializes a whole-store match map.
|
||||
*
|
||||
* @param query - Text query to search for.
|
||||
* @param within - Optional candidate universe; absent = the whole store.
|
||||
* @returns Array of { id, matchCount } sorted by matchCount descending.
|
||||
*/
|
||||
private async scoreTextQuery(
|
||||
query: string,
|
||||
within?: ReadonlySet<string>
|
||||
): Promise<Array<{ id: string; matchCount: number }>> {
|
||||
const queryWords = this.tokenize(query)
|
||||
if (queryWords.length === 0) return []
|
||||
|
||||
// Get IDs for each word hash
|
||||
const wordIdSets: Map<string, number>[] = []
|
||||
// Count matches per entity, one word's postings at a time.
|
||||
const matchCounts = new Map<string, number>()
|
||||
for (const word of queryWords) {
|
||||
const wordHash = this.hashWord(word)
|
||||
let ids: string[]
|
||||
|
|
@ -1529,19 +1574,12 @@ export class MetadataIndexManager implements MetadataIndexProvider {
|
|||
throw err
|
||||
}
|
||||
}
|
||||
const idSet = new Map<string, number>()
|
||||
// One count per (word, entity) — dedupe this word's postings first.
|
||||
const counted = new Set<string>()
|
||||
for (const id of ids) {
|
||||
idSet.set(id, 1)
|
||||
}
|
||||
wordIdSets.push(idSet)
|
||||
}
|
||||
|
||||
if (wordIdSets.length === 0) return []
|
||||
|
||||
// Count matches per entity
|
||||
const matchCounts = new Map<string, number>()
|
||||
for (const idSet of wordIdSets) {
|
||||
for (const [id] of idSet) {
|
||||
if (counted.has(id)) continue
|
||||
counted.add(id)
|
||||
if (within && !within.has(id)) continue
|
||||
matchCounts.set(id, (matchCounts.get(id) || 0) + 1)
|
||||
}
|
||||
}
|
||||
|
|
@ -2575,6 +2613,19 @@ export class MetadataIndexManager implements MetadataIndexProvider {
|
|||
/** Once-per-field flag for the fallback-degradation announcement. */
|
||||
private static announcedFallbackSorts = new Set<string>()
|
||||
|
||||
/**
|
||||
* Evaluate `filter` over `ids` only — the graph-first find's door (the
|
||||
* neighbour set filtered by id, never the store filtered and then
|
||||
* intersected). This index answers from its own `getIdsForFilter`, so the
|
||||
* two doors cannot disagree; the cost is that of the filter over this
|
||||
* in-memory index, and the answer keeps the caller's order.
|
||||
*/
|
||||
async filterIdsWithin(filter: any, ids: readonly string[]): Promise<string[]> {
|
||||
if (ids.length === 0) return []
|
||||
const matched = new Set(await this.getIdsForFilter(filter))
|
||||
return ids.filter((id) => matched.has(id))
|
||||
}
|
||||
|
||||
async getSortedIdsForFilter(
|
||||
filter: any,
|
||||
orderBy: string,
|
||||
|
|
|
|||
|
|
@ -1572,7 +1572,19 @@ export class VirtualFileSystem implements IVirtualFileSystem {
|
|||
// ============= Semantic Operations =============
|
||||
|
||||
/**
|
||||
* Search files with natural language
|
||||
* Search files with natural language.
|
||||
*
|
||||
* `options.path` scopes the search to a directory: its whole subtree by
|
||||
* default, its immediate children when `recursive` is `false`. Both scopes
|
||||
* are metadata filters the index SERVES, so the scope narrows the search
|
||||
* before it runs — no tree walk, and never an over-fetch filtered afterwards.
|
||||
*
|
||||
* @param query - The natural-language query.
|
||||
* @param options - Scope, metadata filters and paging (see {@link SearchOptions}).
|
||||
* @returns The matching files, best first.
|
||||
* @throws {VFSError} ENOENT when `recursive: false` names a path that does
|
||||
* not exist (the non-recursive scope is the directory's own identity, so
|
||||
* the directory has to be there).
|
||||
*/
|
||||
async search(query: string, options?: SearchOptions): Promise<SearchResult[]> {
|
||||
await this.ensureInitialized()
|
||||
|
|
@ -1588,11 +1600,26 @@ export class VirtualFileSystem implements IVirtualFileSystem {
|
|||
}
|
||||
}
|
||||
|
||||
// Add path filter if specified
|
||||
// Scope to a directory, if asked. This used to emit
|
||||
// `path: { $startsWith }` — an operator that is not in the filter
|
||||
// vocabulary at all, and whose `$`-less spelling the metadata index
|
||||
// REFUSES by the served-operator law (an equality/range posting index
|
||||
// cannot evaluate a substring without reading every row). Every
|
||||
// path-scoped VFS search therefore threw, and none has ever worked on
|
||||
// this engine line. Both scopes below are served shapes.
|
||||
if (options?.path) {
|
||||
params.where = {
|
||||
...params.where,
|
||||
path: { $startsWith: options.path }
|
||||
if (options.recursive === false) {
|
||||
// Immediate children only: the directory's identity IS the scope, and
|
||||
// `parent` is an indexed equality on every VFS entity.
|
||||
params.where = {
|
||||
...params.where,
|
||||
parent: await this.pathResolver.resolve(options.path)
|
||||
}
|
||||
} else {
|
||||
const scope = this.descendantPathScope(options.path)
|
||||
if (scope) {
|
||||
params.where = { ...params.where, path: scope }
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
|
@ -1754,6 +1781,42 @@ export class VirtualFileSystem implements IVirtualFileSystem {
|
|||
return entity as VFSEntity
|
||||
}
|
||||
|
||||
/**
|
||||
* The SERVED metadata shape for "everything under this directory".
|
||||
*
|
||||
* `metadata.path` is the VFS's truth — write and rename maintain it, and the
|
||||
* `Contains` edges are a projection of it (see {@link repairContainment}) —
|
||||
* it is indexed on every VFS entity, and the metadata index serves ordered
|
||||
* range operators. So a subtree scope is a half-open range over the path
|
||||
* column: O(log n + matches), no tree walk, and nothing fetched that the
|
||||
* scope then discards.
|
||||
*
|
||||
* The range is `[dir + '/', dir + <successor of '/'>)`. Every descendant path
|
||||
* begins with `dir + '/'`, and '0' is the code point directly after '/', so a
|
||||
* string lies in the range EXACTLY when it carries that prefix. The two
|
||||
* bounds differ at a single ASCII position, so the answer is the same under
|
||||
* code-unit and code-point collation alike — no dependence on how the store
|
||||
* orders the rest of the string.
|
||||
*
|
||||
* Sibling exclusion falls out of the same fact and is worth stating, because
|
||||
* it is where a naive prefix test goes wrong: for `dir = '/scope'`,
|
||||
* `/scope-sibling/x` sorts BELOW the lower bound ('-' precedes '/') and
|
||||
* `/scope0` sits at the open upper bound — both outside, while
|
||||
* `/scope/sub/deep/c.txt` is inside at any depth.
|
||||
*
|
||||
* @param path - The directory to scope to.
|
||||
* @returns The `where` fragment for the `path` field, or `null` for the root
|
||||
* — every VFS entity is under it, so no clause narrows the search.
|
||||
*/
|
||||
private descendantPathScope(path: string): { gte: string; lt: string } | null {
|
||||
const dir = path.replace(/\/+/g, '/').replace(/\/$/, '') || '/'
|
||||
if (dir === '/') return null
|
||||
// Computed, so the bound carries its own reason: the first string that can
|
||||
// no longer share the `dir + '/'` prefix.
|
||||
const separatorSuccessor = String.fromCharCode('/'.charCodeAt(0) + 1)
|
||||
return { gte: `${dir}/`, lt: `${dir}${separatorSuccessor}` }
|
||||
}
|
||||
|
||||
private getParentPath(path: string): string {
|
||||
const normalized = path.replace(/\/+/g, '/').replace(/\/$/, '')
|
||||
const lastSlash = normalized.lastIndexOf('/')
|
||||
|
|
@ -2295,6 +2358,31 @@ export class VirtualFileSystem implements IVirtualFileSystem {
|
|||
cursor = page.nextCursor
|
||||
}
|
||||
|
||||
// Pass 2: ONE paged walk over every Contains edge, grouped by target in
|
||||
// memory. The earlier shape issued one awaited related({ to }) per VFS
|
||||
// entity — O(entities) serialized graph calls, measured in whole minutes
|
||||
// on large brains. This shape is O(edges / page) calls regardless of how
|
||||
// many entities exist; mutations alone stay per-defect.
|
||||
const incomingByTarget = new Map<string, Relation<any>[]>()
|
||||
{
|
||||
const pageSize = 1000
|
||||
let pageOffset = 0
|
||||
for (;;) {
|
||||
const page = await this.brain.related({
|
||||
type: VerbType.Contains,
|
||||
limit: pageSize,
|
||||
offset: pageOffset
|
||||
})
|
||||
for (const edge of page) {
|
||||
const bucket = incomingByTarget.get(edge.to)
|
||||
if (bucket) bucket.push(edge)
|
||||
else incomingByTarget.set(edge.to, [edge])
|
||||
}
|
||||
if (page.length < pageSize) break
|
||||
pageOffset += pageSize
|
||||
}
|
||||
}
|
||||
|
||||
let removed = 0
|
||||
let restored = 0
|
||||
for (const { id, path } of vfsEntities) {
|
||||
|
|
@ -2307,7 +2395,7 @@ export class VirtualFileSystem implements IVirtualFileSystem {
|
|||
continue
|
||||
}
|
||||
|
||||
const incoming = await this.brain.related({ to: id, type: VerbType.Contains })
|
||||
const incoming = incomingByTarget.get(id) ?? []
|
||||
let expectedSeen = false
|
||||
for (const edge of incoming) {
|
||||
const isVfsEdge = edge.subtype === 'vfs-contains' || (edge.metadata as any)?.isVFS === true
|
||||
|
|
|
|||
68
tests/configs/vitest.perf.config.ts
Normal file
68
tests/configs/vitest.perf.config.ts
Normal file
|
|
@ -0,0 +1,68 @@
|
|||
import { defineConfig } from 'vitest/config'
|
||||
|
||||
/**
|
||||
* Perf/scale + environment-dependent test configuration.
|
||||
*
|
||||
* The exclusive on-demand slot for everything the correctness gate
|
||||
* (`vitest.config.ts`, the config a bare `vitest run` picks up) excludes:
|
||||
* wall-clock/scale benchmarks and the two tests whose outcome depends on
|
||||
* the host machine or network rather than the code. See CONTRIBUTING.md's
|
||||
* "Test gate" section and the exclude list in `vitest.config.ts` (root) for
|
||||
* why each file lives here instead of the gate.
|
||||
*
|
||||
* `include` names this set explicitly — it is the mirror image of the
|
||||
* root config's exclude list, not an independent glob, so the two stay in
|
||||
* sync by inspection. Longer timeouts than the gate's 120s/60s: one case in
|
||||
* tests/critical-performance-benchmark.test.ts measures ~128s of real work.
|
||||
*/
|
||||
export default defineConfig({
|
||||
test: {
|
||||
globals: true,
|
||||
setupFiles: ['./tests/setup.ts'],
|
||||
environment: 'node',
|
||||
|
||||
// The marker a test uses to tell it is running under this lane (see
|
||||
// tests/integration/storage-batch-operations.test.ts's batch-vs-
|
||||
// individual timing case) — a wall-clock RATIO assertion self-skips
|
||||
// with a reason when this is absent, rather than flaking the
|
||||
// correctness gate on whichever path happens to be faster this build.
|
||||
env: { BRAINY_PERF_LANE: '1' },
|
||||
|
||||
// Sequential, single fork — same isolation the gate uses, so a perf
|
||||
// measurement isn't skewed by sibling test contention.
|
||||
pool: 'forks',
|
||||
poolOptions: {
|
||||
forks: {
|
||||
maxForks: 1,
|
||||
minForks: 1,
|
||||
singleFork: true,
|
||||
isolate: true
|
||||
}
|
||||
},
|
||||
|
||||
testTimeout: 300000, // 5 minutes per test (the 128s case plus headroom)
|
||||
hookTimeout: 120000,
|
||||
teardownTimeout: 10000,
|
||||
|
||||
maxConcurrency: 1,
|
||||
fileParallelism: false,
|
||||
|
||||
include: [
|
||||
'tests/performance/**/*.{test,spec}.{js,ts}',
|
||||
'tests/critical-performance-benchmark.test.ts',
|
||||
'tests/api/performance-benchmarks.test.ts',
|
||||
'tests/package-size-limit.test.ts',
|
||||
'tests/model-loading.test.ts',
|
||||
// Not a whole perf file — one wall-clock-ratio case inside an
|
||||
// otherwise-correctness integration suite (self-skipped everywhere
|
||||
// else via BRAINY_PERF_LANE). Stays in the integration gate's
|
||||
// include too, so every OTHER test in the file keeps running there.
|
||||
'tests/integration/storage-batch-operations.test.ts'
|
||||
],
|
||||
|
||||
reporters: process.env.CI ? ['dot'] : ['basic'],
|
||||
|
||||
retry: process.env.CI ? 1 : 0,
|
||||
shard: process.env.VITEST_SHARD
|
||||
}
|
||||
})
|
||||
111
tests/integration/counts-persist-single-flight.test.ts
Normal file
111
tests/integration/counts-persist-single-flight.test.ts
Normal file
|
|
@ -0,0 +1,111 @@
|
|||
/**
|
||||
* @module tests/integration/counts-persist-single-flight
|
||||
* @description Regression for a production race in FileSystemStorage's
|
||||
* counts ledger: `persistCounts()` was write-through on every count change
|
||||
* with no serialization, and the atomic writer named its temp file with
|
||||
* millisecond granularity (`.tmp-<pid>-<ms>`). Two persists inside one
|
||||
* millisecond shared the temp path — both wrote it, the first rename
|
||||
* consumed it, the second rename found nothing: ENOENT, ~1,500 times a day
|
||||
* on a busy production brain, with a full ledger write per change behind it.
|
||||
*
|
||||
* Under pin: persists are single-flight and coalesced — one in flight, at
|
||||
* most one trailing pass carrying the burst's final state — and every atomic
|
||||
* write owns a unique temp path. A burst of N count changes costs at most
|
||||
* two ledger writes, never errors, and leaves a ledger equal to memory.
|
||||
*/
|
||||
import { describe, it, expect, beforeEach, afterEach, vi } from 'vitest'
|
||||
import * as fs from 'node:fs'
|
||||
import * as os from 'node:os'
|
||||
import * as path from 'node:path'
|
||||
import { Brainy } from '../../src/brainy.js'
|
||||
import { NounType } from '../../src/types/graphTypes.js'
|
||||
|
||||
describe('counts persistence is single-flight, coalesced, and never races its own temp file', () => {
|
||||
let dir: string
|
||||
let brain: any
|
||||
|
||||
beforeEach(async () => {
|
||||
process.env.BRAINY_DETERMINISTIC_EMBEDDINGS = 'true'
|
||||
dir = fs.mkdtempSync(path.join(os.tmpdir(), 'brainy-counts-race-'))
|
||||
brain = new Brainy({
|
||||
requireSubtype: false,
|
||||
storage: { type: 'filesystem', path: dir },
|
||||
dimensions: 384,
|
||||
silent: true
|
||||
})
|
||||
await brain.init()
|
||||
})
|
||||
|
||||
afterEach(async () => {
|
||||
vi.restoreAllMocks()
|
||||
await brain.close()
|
||||
fs.rmSync(dir, { recursive: true, force: true })
|
||||
})
|
||||
|
||||
it('a burst of concurrent count changes → at most two ledger writes, zero errors, ledger == memory', async () => {
|
||||
const storage = brain.storage
|
||||
const countsPath: string = storage.countsFilePath
|
||||
expect(countsPath, 'the filesystem adapter persists a counts ledger').toBeTruthy()
|
||||
|
||||
// Let init's own persists settle so the burst is measured alone.
|
||||
await storage.flushCounts?.()
|
||||
|
||||
const renameSpy = vi.spyOn(fs.promises, 'rename')
|
||||
const errorSpy = vi.spyOn(console, 'error')
|
||||
|
||||
// Twenty-five concurrent count changes — the shape of a write burst; each
|
||||
// used to launch its own persist.
|
||||
const BURST = 25
|
||||
await Promise.all(
|
||||
Array.from({ length: BURST }, () => storage.scheduleCountPersist())
|
||||
)
|
||||
|
||||
const ledgerRenames = renameSpy.mock.calls.filter(([, to]) => String(to) === countsPath)
|
||||
expect(ledgerRenames.length, 'single-flight + one trailing pass').toBeLessThanOrEqual(2)
|
||||
expect(ledgerRenames.length, 'the burst was persisted at all').toBeGreaterThanOrEqual(1)
|
||||
|
||||
const persistErrors = errorSpy.mock.calls.filter((args) => String(args[0]).includes('persisting counts'))
|
||||
expect(persistErrors).toEqual([])
|
||||
|
||||
const ledger = JSON.parse(fs.readFileSync(countsPath, 'utf-8'))
|
||||
expect(ledger.totalNounCount).toBe(storage.totalNounCount)
|
||||
expect(ledger.totalVerbCount).toBe(storage.totalVerbCount)
|
||||
})
|
||||
|
||||
it('real writes in parallel: the ledger lands complete and no persist error is logged', async () => {
|
||||
const storage = brain.storage
|
||||
const countsPath: string = storage.countsFilePath
|
||||
const errorSpy = vi.spyOn(console, 'error')
|
||||
|
||||
await Promise.all(
|
||||
Array.from({ length: 12 }, (_, i) =>
|
||||
brain.add({ data: `burst row ${i}`, type: NounType.Thing })
|
||||
)
|
||||
)
|
||||
await storage.flushCounts?.()
|
||||
|
||||
const persistErrors = errorSpy.mock.calls.filter((args) => String(args[0]).includes('persisting counts'))
|
||||
expect(persistErrors).toEqual([])
|
||||
const ledger = JSON.parse(fs.readFileSync(countsPath, 'utf-8'))
|
||||
expect(ledger.totalNounCount).toBe(storage.totalNounCount)
|
||||
expect(await brain.getNounCount()).toBe(ledger.totalNounCount)
|
||||
})
|
||||
|
||||
it('every atomic write owns its own temp path — two writes in one millisecond never collide', async () => {
|
||||
const storage = brain.storage
|
||||
const tmpNames: string[] = []
|
||||
vi.spyOn(fs.promises, 'writeFile').mockImplementation(async (p: any) => {
|
||||
tmpNames.push(String(p))
|
||||
})
|
||||
vi.spyOn(fs.promises, 'rename').mockImplementation(async () => undefined)
|
||||
const target = path.join(dir, 'probe.json')
|
||||
await Promise.all([
|
||||
storage.writeFileAtomic(target, '{"a":1}'),
|
||||
storage.writeFileAtomic(target, '{"a":2}'),
|
||||
storage.writeFileAtomic(target, '{"a":3}')
|
||||
])
|
||||
const probeTmps = tmpNames.filter((n) => n.startsWith(`${target}.tmp-`))
|
||||
expect(probeTmps.length).toBe(3)
|
||||
expect(new Set(probeTmps).size, 'no two writes shared a temp path').toBe(3)
|
||||
})
|
||||
})
|
||||
|
|
@ -57,7 +57,11 @@ describe('entity-tree family stamp', () => {
|
|||
const invariants = (stamp.members as any).invariants
|
||||
expect(invariants.nounCount).toBe(await brain.storage.getNounCount())
|
||||
expect(invariants.verbCount).toBe(await brain.storage.getVerbCount())
|
||||
expect(stamp.sourceGeneration).toBe(brain.generation())
|
||||
// THE SOURCE IS COMMITTED TRUTH, never the allocated counter. Stamping the
|
||||
// counter labelled the stamp with a generation a write in flight had merely
|
||||
// claimed, so every crash inside a write window produced a spurious verdict
|
||||
// at the next open (see the torn-tail pins below).
|
||||
expect(stamp.sourceGeneration).toBe(brain.generationStore.committedGeneration())
|
||||
expect(stamp.generation).toBeGreaterThanOrEqual(1)
|
||||
})
|
||||
|
||||
|
|
@ -112,6 +116,96 @@ describe('entity-tree family stamp', () => {
|
|||
expect(stillIncoherent).toEqual([])
|
||||
})
|
||||
|
||||
/**
|
||||
* Rewrite the on-disk stamp so its `sourceGeneration` sits ABOVE the store's
|
||||
* committed watermark — the durable shape a torn generation-log tail leaves
|
||||
* behind (the stamp's fsync outlived the tail's). Fabricated rather than
|
||||
* crash-produced so the pin is deterministic; the seeded-SIGKILL lane
|
||||
* (`scripts/crash-consistency.mjs` in the engine repo) produces the same
|
||||
* shape from a real abrupt termination.
|
||||
*/
|
||||
const fabricateTear = (ahead: number): FamilyStamp => {
|
||||
const file = path.join(dir, `${ENTITY_TREE_STAMP_PATH}.gz`)
|
||||
const zlib = require('node:zlib')
|
||||
const raw = JSON.parse(zlib.gunzipSync(fs.readFileSync(file)).toString('utf-8')) as FamilyStamp
|
||||
const torn: FamilyStamp = { ...raw, sourceGeneration: raw.sourceGeneration + ahead }
|
||||
fs.writeFileSync(file, zlib.gzipSync(JSON.stringify(torn)))
|
||||
return torn
|
||||
}
|
||||
|
||||
it('a torn generation-log tail is a TERMINAL VERDICT at open: narrated, demoted, never a wait', async () => {
|
||||
for (let i = 0; i < 3; i++)
|
||||
await brain.add({ data: `torn${i}`, type: 'document', metadata: { i } })
|
||||
await brain.close()
|
||||
const torn = fabricateTear(5)
|
||||
|
||||
const warn = vi.spyOn(prodLog, 'warn')
|
||||
const startedAt = Date.now()
|
||||
brain = await open()
|
||||
const openMs = Date.now() - startedAt
|
||||
|
||||
const tearLines = warn.mock.calls.filter((c) => String(c[0]).includes('TORN GENERATION-LOG TAIL'))
|
||||
expect(tearLines.length).toBe(1)
|
||||
const said = String(tearLines[0][0])
|
||||
// Narrated PRECISELY: both generations, the file, and the named cure.
|
||||
expect(said).toContain(`source generation ${torn.sourceGeneration}`)
|
||||
expect(said).toContain(`committed generation ${brain.generationStore.committedGeneration()}`)
|
||||
expect(said).toContain(ENTITY_TREE_STAMP_PATH)
|
||||
expect(said).toContain('DEMOTED')
|
||||
expect(said).toMatch(/repairIndex\(\)/)
|
||||
// Terminal, not a wait: the demotion is O(1) straight-line work, so a tear
|
||||
// cannot turn an open into the 8-minute spin this class was reported as.
|
||||
expect(openMs).toBeLessThan(30_000)
|
||||
|
||||
// The store SERVES — a tear in a stamp never locks an owner out of the
|
||||
// canonical tree the stamp merely describes.
|
||||
expect((await brain.find({ type: 'document', limit: 100 })).length).toBe(3)
|
||||
|
||||
// The demotion CONVERGED: the stamp now names committed truth, and the
|
||||
// next open is quiet. A verdict that re-narrates every open is a wait
|
||||
// wearing a different hat.
|
||||
const restamped = (await readFamilyStamp(brain.storage, ENTITY_TREE_STAMP_PATH)) as FamilyStamp
|
||||
expect(restamped.sourceGeneration).toBe(brain.generationStore.committedGeneration())
|
||||
await brain.close()
|
||||
const warn2 = vi.spyOn(prodLog, 'warn')
|
||||
brain = await open()
|
||||
expect(warn2.mock.calls.filter((c) => String(c[0]).includes('TORN'))).toEqual([])
|
||||
})
|
||||
|
||||
it('a READ-ONLY open on a torn tail refuses to guess: terminal verdict + named cure, no re-stamp', async () => {
|
||||
await brain.add({ data: 'ro', type: 'document', metadata: {} })
|
||||
await brain.close()
|
||||
const torn = fabricateTear(3)
|
||||
|
||||
const warn = vi.spyOn(prodLog, 'warn')
|
||||
const reader: any = await Brainy.openReadOnly({
|
||||
requireSubtype: false,
|
||||
storage: { type: 'filesystem', path: dir },
|
||||
silent: true,
|
||||
dimensions: 384
|
||||
})
|
||||
const tearLines = warn.mock.calls.filter((c) => String(c[0]).includes('TORN GENERATION-LOG TAIL'))
|
||||
expect(tearLines.length).toBe(1)
|
||||
const said = String(tearLines[0][0])
|
||||
expect(said).toContain('READ-ONLY')
|
||||
expect(said).toContain('UNVERIFIED')
|
||||
expect(said).toMatch(/repairIndex\(\)/)
|
||||
await reader.close()
|
||||
|
||||
// A reader never rewrites the store: read the bytes back off disk (not
|
||||
// through a writer open, which would demote them) — the torn stamp is
|
||||
// exactly as it was found.
|
||||
const onDisk = JSON.parse(
|
||||
require('node:zlib')
|
||||
.gunzipSync(fs.readFileSync(path.join(dir, `${ENTITY_TREE_STAMP_PATH}.gz`)))
|
||||
.toString('utf-8')
|
||||
) as FamilyStamp
|
||||
expect(onDisk.sourceGeneration).toBe(torn.sourceGeneration)
|
||||
expect(onDisk.generation).toBe(torn.generation)
|
||||
|
||||
brain = await open()
|
||||
})
|
||||
|
||||
it('the one verifier handles both member modes', () => {
|
||||
const rollup: FamilyStamp = {
|
||||
family: 'x',
|
||||
|
|
@ -127,7 +221,13 @@ describe('entity-tree family stamp', () => {
|
|||
stampSource: 5,
|
||||
head: 9
|
||||
})
|
||||
expect(verifyFamilyStamp(rollup, 3, { nounCount: 10 }).state).toBe('incoherent') // ahead of head
|
||||
// AHEAD is its own class — a torn generation-log tail, never folded in
|
||||
// with `incoherent`: the two have opposite cures (recount vs demote).
|
||||
expect(verifyFamilyStamp(rollup, 3, { nounCount: 10 })).toEqual({
|
||||
state: 'torn',
|
||||
stampSource: 5,
|
||||
head: 3
|
||||
})
|
||||
expect(verifyFamilyStamp(null, 5, {})).toEqual({ state: 'absent' })
|
||||
|
||||
const enumerated: FamilyStamp = {
|
||||
|
|
|
|||
360
tests/integration/factlog-open-prune.test.ts
Normal file
360
tests/integration/factlog-open-prune.test.ts
Normal file
|
|
@ -0,0 +1,360 @@
|
|||
/**
|
||||
* @module tests/integration/factlog-open-prune
|
||||
* @description THE OPEN READS THE TAIL, NOT THE HISTORY.
|
||||
*
|
||||
* Every log-authority open asks the fact log one question — "is there a fact
|
||||
* above the committed pointer?" — and until this lane existed it answered by
|
||||
* reading and CRC-decoding EVERY segment file the manifest names. MEASURED in
|
||||
* production on a 16k-row brain at generation ~478,819: 34-37 seconds inside
|
||||
* the `generation-store-open-fold` phase, on every open, including the clean
|
||||
* one where the answer is always "nothing".
|
||||
*
|
||||
* The manifest already records each sealed segment's `lastGeneration`, written
|
||||
* at seal time AFTER the segment's bytes are fsynced and into a manifest that
|
||||
* is itself written atomically and fsynced — and a sealed file is never
|
||||
* appended to again (the same manifest flip re-points `tailSegment`). So an
|
||||
* entry recording `lastGeneration ≤ committed` PROVES its file holds nothing
|
||||
* above the bound, and the open can skip it whole.
|
||||
*
|
||||
* Pinned here, from the log's own counters (the narration line), never a clock:
|
||||
*
|
||||
* 1. A clean close and reopen on a log with ≥4 sealed segments reads
|
||||
* EXACTLY the tail (1 of 6), prunes the rest, and finds nothing.
|
||||
* 2. A real SIGKILLed process that sealed segments holding facts ABOVE the
|
||||
* committed pointer: the reopen READS those sealed segments and recovers
|
||||
* byte-identically to an unpruned open (differential — the same store,
|
||||
* with the provable field stripped from its manifest, takes the full-scan
|
||||
* path and must agree fact for fact, before and after `open()`).
|
||||
* 3. A manifest entry with no `lastGeneration` (legacy, or hand-repaired) is
|
||||
* READ. Never prune what the manifest cannot prove.
|
||||
*/
|
||||
import { describe, it, expect, afterEach } from 'vitest'
|
||||
import * as fs from 'node:fs'
|
||||
import * as os from 'node:os'
|
||||
import * as path from 'node:path'
|
||||
import { spawn } from 'node:child_process'
|
||||
import {
|
||||
FactLog,
|
||||
FACTS_MANIFEST_PATH,
|
||||
type CommitFact,
|
||||
type FactLogStorage
|
||||
} from '../../src/db/factLog.js'
|
||||
import { FileSystemStorage } from '../../src/storage/adapters/fileSystemStorage.js'
|
||||
|
||||
const REPO_ROOT = process.cwd()
|
||||
const TSX = path.join(REPO_ROOT, 'node_modules', '.bin', 'tsx')
|
||||
/** ~1KB frames against a 4KB rotation threshold: ~5 facts per segment. */
|
||||
const ROTATE_BYTES = 4096
|
||||
|
||||
const tmpDirs: string[] = []
|
||||
function makeTempDir(): string {
|
||||
const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'brainy-factlog-prune-'))
|
||||
tmpDirs.push(dir)
|
||||
return dir
|
||||
}
|
||||
|
||||
afterEach(() => {
|
||||
for (const dir of tmpDirs.splice(0)) {
|
||||
try {
|
||||
fs.rmSync(dir, { recursive: true, force: true })
|
||||
} catch {
|
||||
/* best effort */
|
||||
}
|
||||
try {
|
||||
fs.rmSync(`${dir}.ready.json`, { force: true })
|
||||
} catch {
|
||||
/* best effort */
|
||||
}
|
||||
}
|
||||
})
|
||||
|
||||
const UUID = (n: number): string => `00000000-0000-4000-8000-${String(n).padStart(12, '0')}`
|
||||
|
||||
/** One ~1KB fact — the padding is what makes rotation cheap to provoke. */
|
||||
function fact(generation: number): CommitFact {
|
||||
return {
|
||||
generation,
|
||||
timestamp: 1_700_000_000_000 + generation,
|
||||
ops: [
|
||||
{
|
||||
kind: 'noun',
|
||||
id: UUID(generation),
|
||||
record: {
|
||||
metadata: { noun: 'document', pad: 'x'.repeat(900), g: generation },
|
||||
vector: null
|
||||
}
|
||||
}
|
||||
]
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* A deterministic int minter so the log writes the V2 format production
|
||||
* writes (the prune is a manifest-level decision and never touches segment
|
||||
* bytes — but the pins should run against the bytes the fleet actually has).
|
||||
*/
|
||||
function makeMinter(): (kind: 'noun' | 'verb', id: string) => bigint {
|
||||
const ints = new Map<string, bigint>()
|
||||
return (kind, id) => {
|
||||
const key = `${kind}:${id}`
|
||||
let minted = ints.get(key)
|
||||
if (minted === undefined) {
|
||||
minted = BigInt(ints.size + 1)
|
||||
ints.set(key, minted)
|
||||
}
|
||||
return minted
|
||||
}
|
||||
}
|
||||
|
||||
/** Open a fact log over a store directory (a fresh adapter each time — this is
|
||||
* what a reopen actually does). */
|
||||
async function openStore(dir: string): Promise<{ storage: any; log: FactLog }> {
|
||||
const storage: any = new FileSystemStorage(dir)
|
||||
await storage.init()
|
||||
const log = new FactLog(storage as FactLogStorage, { rotateBytes: ROTATE_BYTES })
|
||||
log.setIntMinter(makeMinter())
|
||||
return { storage, log }
|
||||
}
|
||||
|
||||
/** Build a log of `count` facts (rotating every ~5), left durable, not closed. */
|
||||
async function buildLog(dir: string, count: number): Promise<number> {
|
||||
const { log } = await openStore(dir)
|
||||
await log.open(0)
|
||||
for (let g = 1; g <= count; g++) await log.append(fact(g))
|
||||
await log.sync()
|
||||
return log.headGeneration()
|
||||
}
|
||||
|
||||
/** Capture the narration channel (`prodLog.narrate` → console.warn). */
|
||||
async function captureNarration<T>(
|
||||
fn: () => Promise<T>
|
||||
): Promise<{ result: T; lines: string[] }> {
|
||||
const lines: string[] = []
|
||||
const original = console.warn
|
||||
console.warn = ((...args: unknown[]) => {
|
||||
lines.push(args.map((a) => String(a)).join(' '))
|
||||
}) as typeof console.warn
|
||||
try {
|
||||
return { result: await fn(), lines }
|
||||
} finally {
|
||||
console.warn = original
|
||||
}
|
||||
}
|
||||
|
||||
/** The counters the open narrated — the pin's only source of truth for what
|
||||
* was read (a wall-clock assertion could pass on a warm page cache). */
|
||||
function scanCounts(lines: string[]): { read: number; pruned: number; total: number } {
|
||||
const line = lines.find((l) => l.includes('[FactLog] above-manifest peek above generation'))
|
||||
if (!line) {
|
||||
throw new Error(`no peek narration in:\n${lines.join('\n')}`)
|
||||
}
|
||||
const match = /(\d+) segment\(s\) read, (\d+) pruned of (\d+)/.exec(line)
|
||||
if (!match) throw new Error(`unparsable peek narration: ${line}`)
|
||||
return { read: Number(match[1]), pruned: Number(match[2]), total: Number(match[3]) }
|
||||
}
|
||||
|
||||
interface SegmentEntryOnDisk {
|
||||
file: string
|
||||
firstGeneration: number
|
||||
lastGeneration?: number
|
||||
facts: number
|
||||
bytes: number
|
||||
}
|
||||
|
||||
async function readManifest(dir: string): Promise<{
|
||||
segments: SegmentEntryOnDisk[]
|
||||
tailSegment: string | null
|
||||
}> {
|
||||
const storage: any = new FileSystemStorage(dir)
|
||||
await storage.init()
|
||||
return (await storage.readRawObject(FACTS_MANIFEST_PATH)) as any
|
||||
}
|
||||
|
||||
async function rewriteManifest(
|
||||
dir: string,
|
||||
mutate: (manifest: any) => void
|
||||
): Promise<void> {
|
||||
const storage: any = new FileSystemStorage(dir)
|
||||
await storage.init()
|
||||
const manifest = await storage.readRawObject(FACTS_MANIFEST_PATH)
|
||||
mutate(manifest)
|
||||
await storage.writeRawObject(FACTS_MANIFEST_PATH, manifest)
|
||||
await storage.syncRawObjects([FACTS_MANIFEST_PATH])
|
||||
}
|
||||
|
||||
/** Every fact the log holds, in order — the recovered state, read back. */
|
||||
async function allFacts(log: FactLog): Promise<CommitFact[]> {
|
||||
const out: CommitFact[] = []
|
||||
const handle = log.scanFacts()
|
||||
for await (const batch of handle.batches()) out.push(...batch.facts)
|
||||
return out
|
||||
}
|
||||
|
||||
describe('fact log — the open reads only the segments that can hold facts above the bound', () => {
|
||||
it('a clean close + reopen over ≥4 sealed segments reads exactly the tail and finds nothing', async () => {
|
||||
const dir = makeTempDir()
|
||||
const head = await buildLog(dir, 30)
|
||||
|
||||
const manifest = await readManifest(dir)
|
||||
expect(manifest.segments.length).toBeGreaterThanOrEqual(4) // the fixture is real
|
||||
expect(manifest.tailSegment).not.toBeNull()
|
||||
|
||||
// The reopen: a clean close means committed === the log's head.
|
||||
const { log } = await openStore(dir)
|
||||
const { result: orphans, lines } = await captureNarration(() => log.peekFactsAbove(head))
|
||||
|
||||
expect(orphans).toEqual([]) // the fold finds nothing, as it always does after a clean close
|
||||
const counts = scanCounts(lines)
|
||||
expect(counts.read).toBe(1) // EXACTLY the tail
|
||||
expect(counts.total).toBe(manifest.segments.length + 1)
|
||||
expect(counts.pruned).toBe(manifest.segments.length)
|
||||
|
||||
// And the reconciling open still lands on the same committed prefix.
|
||||
await log.open(head)
|
||||
expect(log.headGeneration()).toBe(head)
|
||||
expect((await allFacts(log)).map((f) => f.generation)).toEqual(
|
||||
Array.from({ length: head }, (_, i) => i + 1)
|
||||
)
|
||||
})
|
||||
|
||||
it('a manifest entry with no lastGeneration is READ — never prune what you cannot prove', async () => {
|
||||
const dir = makeTempDir()
|
||||
const head = await buildLog(dir, 30)
|
||||
const before = await readManifest(dir)
|
||||
expect(before.segments.length).toBeGreaterThanOrEqual(4)
|
||||
|
||||
// A legacy/hand-repaired entry: the field the prune needs is simply absent.
|
||||
await rewriteManifest(dir, (m) => {
|
||||
delete m.segments[0].lastGeneration
|
||||
})
|
||||
|
||||
const { log } = await openStore(dir)
|
||||
const { result: orphans, lines } = await captureNarration(() => log.peekFactsAbove(head))
|
||||
|
||||
expect(orphans).toEqual([]) // still nothing above the bound — it was READ to find out
|
||||
const counts = scanCounts(lines)
|
||||
expect(counts.read).toBe(2) // the unprovable entry + the tail
|
||||
expect(counts.pruned).toBe(before.segments.length - 1)
|
||||
expect(counts.total).toBe(before.segments.length + 1)
|
||||
})
|
||||
|
||||
it(
|
||||
'a SIGKILLed writer that sealed segments above the committed pointer recovers identically to an unpruned open',
|
||||
async () => {
|
||||
const dir = makeTempDir()
|
||||
const readyPath = `${dir}.ready.json`
|
||||
// A real process death: the child fsyncs its segments, records what it
|
||||
// reached, and SIGKILLs ITSELF — no close, no unwind, no chance to tidy.
|
||||
const script = `
|
||||
import * as fs from 'node:fs'
|
||||
import { FactLog } from ${JSON.stringify(path.join(REPO_ROOT, 'src', 'db', 'factLog.ts'))}
|
||||
import { FileSystemStorage } from ${JSON.stringify(path.join(REPO_ROOT, 'src', 'storage', 'adapters', 'fileSystemStorage.ts'))}
|
||||
const UUID = (n) => '00000000-0000-4000-8000-' + String(n).padStart(12, '0')
|
||||
const fact = (g) => ({
|
||||
generation: g,
|
||||
timestamp: 1700000000000 + g,
|
||||
ops: [{ kind: 'noun', id: UUID(g), record: { metadata: { noun: 'document', pad: 'x'.repeat(900), g }, vector: null } }]
|
||||
})
|
||||
const ints = new Map()
|
||||
const storage = new FileSystemStorage(${JSON.stringify(dir)})
|
||||
await storage.init()
|
||||
const log = new FactLog(storage, { rotateBytes: ${ROTATE_BYTES} })
|
||||
log.setIntMinter((kind, id) => {
|
||||
const key = kind + ':' + id
|
||||
if (!ints.has(key)) ints.set(key, BigInt(ints.size + 1))
|
||||
return ints.get(key)
|
||||
})
|
||||
await log.open(0)
|
||||
for (let g = 1; g <= 30; g++) await log.append(fact(g))
|
||||
await log.sync()
|
||||
fs.writeFileSync(${JSON.stringify(readyPath)}, JSON.stringify({ head: log.headGeneration() }))
|
||||
process.kill(process.pid, 'SIGKILL')
|
||||
`
|
||||
const scriptPath = path.join(dir, 'crash-writer.mts')
|
||||
fs.writeFileSync(scriptPath, script)
|
||||
const child = spawn(TSX, [scriptPath], { cwd: REPO_ROOT, stdio: ['ignore', 'pipe', 'pipe'] })
|
||||
let output = ''
|
||||
child.stdout.on('data', (d) => { output += String(d) })
|
||||
child.stderr.on('data', (d) => { output += String(d) })
|
||||
const exit = await new Promise<{ code: number | null; signal: string | null }>((resolve) =>
|
||||
child.on('exit', (code, signal) => resolve({ code, signal }))
|
||||
)
|
||||
if (!fs.existsSync(readyPath)) {
|
||||
throw new Error(`the crash writer never reached its kill point:\n${output}`)
|
||||
}
|
||||
// Death, not a shutdown: no close(), no unwind, no orderly exit code.
|
||||
expect(exit.signal ?? `code ${exit.code}`).not.toBe('code 0')
|
||||
const head = JSON.parse(fs.readFileSync(readyPath, 'utf8')).head as number
|
||||
expect(head).toBe(30)
|
||||
|
||||
// The committed pointer the survivor comes back on: mid-log, so sealed
|
||||
// segments hold facts ABOVE it — the exact shape the prune must not skip.
|
||||
const committed = 12
|
||||
const manifest = await readManifest(dir)
|
||||
const straddling = manifest.segments.filter(
|
||||
(s) => s.firstGeneration <= committed && (s.lastGeneration ?? 0) > committed
|
||||
)
|
||||
const entirelyAbove = manifest.segments.filter((s) => s.firstGeneration > committed)
|
||||
expect(straddling.length).toBeGreaterThanOrEqual(1)
|
||||
expect(entirelyAbove.length).toBeGreaterThanOrEqual(1)
|
||||
|
||||
// THE DIFFERENTIAL. The unpruned answer, through the SAME code on the
|
||||
// SAME bytes: a peek above generation 0 can prune nothing (no sealed
|
||||
// segment ends at or below 0), so it reads every segment file and
|
||||
// decodes every frame — exactly what this open used to do — and its
|
||||
// facts above the pointer are what the fold is entitled to replay.
|
||||
const { log } = await openStore(dir)
|
||||
const { result: fullScan, lines: fullLines } = await captureNarration(() =>
|
||||
log.peekFactsAbove(0)
|
||||
)
|
||||
expect(scanCounts(fullLines)).toEqual({
|
||||
read: manifest.segments.length + 1,
|
||||
pruned: 0,
|
||||
total: manifest.segments.length + 1
|
||||
})
|
||||
const unprunedAnswer = fullScan.filter((f) => f.generation > committed)
|
||||
|
||||
const { result: prunedAnswer, lines } = await captureNarration(() =>
|
||||
log.peekFactsAbove(committed)
|
||||
)
|
||||
|
||||
// The sealed segments above the bound were READ, not skipped.
|
||||
const counts = scanCounts(lines)
|
||||
expect(counts.read).toBe(straddling.length + entirelyAbove.length + 1)
|
||||
expect(counts.pruned).toBe(manifest.segments.length - straddling.length - entirelyAbove.length)
|
||||
expect(counts.pruned).toBeGreaterThan(0) // the prune did engage, and was still right
|
||||
expect(prunedAnswer.map((f) => f.generation)).toEqual(
|
||||
Array.from({ length: head - committed }, (_, i) => committed + 1 + i)
|
||||
)
|
||||
// Facts that live in a SEALED segment (not the tail) came back.
|
||||
expect(prunedAnswer.some((f) => f.generation <= (straddling[0].lastGeneration ?? 0))).toBe(
|
||||
true
|
||||
)
|
||||
// Fact for fact, the pruned answer IS the unpruned answer — so whatever
|
||||
// the recovery replays, it replays identically.
|
||||
expect(prunedAnswer).toEqual(unprunedAnswer)
|
||||
|
||||
// The fold's streaming twin (the unclean-open path) agrees too.
|
||||
const streamed: CommitFact[] = []
|
||||
for await (const batch of log.streamFactsAbove(committed)) streamed.push(...batch)
|
||||
expect(streamed).toEqual(unprunedAnswer)
|
||||
|
||||
// And the reconciling open rolls back exactly as it always did: the two
|
||||
// never-committed sealed segments dropped, the straddling one cut, the
|
||||
// tail truncated — the log left as the committed prefix.
|
||||
await log.open(committed)
|
||||
expect(log.headGeneration()).toBe(committed)
|
||||
expect((await allFacts(log)).map((f) => f.generation)).toEqual(
|
||||
Array.from({ length: committed }, (_, i) => i + 1)
|
||||
)
|
||||
const after = await readManifest(dir)
|
||||
expect(after.segments.map((s) => s.file)).toEqual(
|
||||
manifest.segments
|
||||
.filter((s) => s.firstGeneration <= committed)
|
||||
.map((s) => s.file)
|
||||
)
|
||||
expect(after.segments[after.segments.length - 1].lastGeneration).toBe(committed)
|
||||
},
|
||||
120_000
|
||||
)
|
||||
})
|
||||
165
tests/integration/find-connected-order.test.ts
Normal file
165
tests/integration/find-connected-order.test.ts
Normal file
|
|
@ -0,0 +1,165 @@
|
|||
/**
|
||||
* @module tests/integration/find-connected-order
|
||||
* @description The graph-first law for `find({ connected })` (10.4.8).
|
||||
*
|
||||
* With `connected` present the neighbour set is the candidate universe: it is
|
||||
* resolved from the adjacency first, the metadata filter is evaluated over
|
||||
* those ids only, and the page is cut last. The earlier order materialized the
|
||||
* whole-store filtered id list, paged it, hydrated the page, and only then
|
||||
* intersected with the neighbours — so a neighbour outside the first page of
|
||||
* the filtered STORE was silently dropped, and every call paid O(store).
|
||||
*
|
||||
* These pins hold both halves. The answer: every matching neighbour is
|
||||
* reachable by paging, a non-neighbour never appears, a negation (`missing`)
|
||||
* is evaluated over the neighbours, `orderBy` sorts the whole neighbour set
|
||||
* before the page is cut, and the vector leg walks the neighbours only. The
|
||||
* cost shape: the metadata index is asked about the neighbour ids only, and
|
||||
* hydration is one page — never the store.
|
||||
*/
|
||||
import { describe, it, expect, beforeAll, afterAll, vi } from 'vitest'
|
||||
import { Brainy } from '../../src/brainy'
|
||||
import { NounType, VerbType } from '../../src/types/graphTypes'
|
||||
import { v5 } from '../../src/universal/uuid'
|
||||
import { generateTestVector } from '../helpers/test-factory'
|
||||
|
||||
/** Matching rows that are NOT neighbours — added FIRST, so the whole-store filtered list leads with them. */
|
||||
const NOISE = 120
|
||||
/** Matching rows that ARE neighbours of the anchor. */
|
||||
const NEIGHBOURS = 30
|
||||
/** Neighbours carrying `retracted: true` — excluded by the `missing` negation. */
|
||||
const RETRACTED = 4
|
||||
|
||||
describe('find({ connected }) is graph-first: neighbours → filter → page', () => {
|
||||
let brain: Brainy<any>
|
||||
const anchor = 'anchor'
|
||||
const sharedVector = generateTestVector()
|
||||
const neighbourIds = new Set(Array.from({ length: NEIGHBOURS }, (_, i) => v5(`nb-${i}`)))
|
||||
|
||||
beforeAll(async () => {
|
||||
brain = new Brainy({ requireSubtype: false, storage: { type: 'memory' } })
|
||||
await brain.init()
|
||||
await brain.add({
|
||||
id: anchor,
|
||||
data: 'the anchor',
|
||||
type: NounType.Person,
|
||||
metadata: { kind: 'anchor' },
|
||||
vector: generateTestVector()
|
||||
})
|
||||
for (let i = 0; i < NOISE; i++) {
|
||||
await brain.add({
|
||||
id: `noise-${i}`,
|
||||
data: `noise ${i}`,
|
||||
type: NounType.Person,
|
||||
metadata: { kind: 'note', rank: 1000 + i },
|
||||
vector: sharedVector
|
||||
})
|
||||
}
|
||||
for (let i = 0; i < NEIGHBOURS; i++) {
|
||||
await brain.add({
|
||||
id: `nb-${i}`,
|
||||
data: `neighbour ${i}`,
|
||||
type: NounType.Person,
|
||||
metadata: { kind: 'note', rank: i + 1, ...(i < RETRACTED ? { retracted: true } : {}) },
|
||||
vector: sharedVector
|
||||
})
|
||||
await brain.relate({ from: anchor, to: `nb-${i}`, type: VerbType.Knows })
|
||||
}
|
||||
})
|
||||
|
||||
afterAll(async () => {
|
||||
brain = null as any
|
||||
})
|
||||
|
||||
it('returns the matching neighbours page by page — none dropped, never a non-neighbour', async () => {
|
||||
const seen = new Set<string>()
|
||||
for (let offset = 0; offset <= NEIGHBOURS; offset += 10) {
|
||||
const page = await brain.find({
|
||||
connected: { from: anchor, direction: 'out' },
|
||||
where: { kind: 'note' },
|
||||
limit: 10,
|
||||
offset
|
||||
})
|
||||
expect(page).toHaveLength(offset < NEIGHBOURS ? 10 : 0)
|
||||
for (const r of page) {
|
||||
expect(neighbourIds.has(r.entity.id)).toBe(true)
|
||||
expect(seen.has(r.entity.id)).toBe(false)
|
||||
seen.add(r.entity.id)
|
||||
}
|
||||
}
|
||||
expect(seen.size).toBe(NEIGHBOURS)
|
||||
})
|
||||
|
||||
it('evaluates a negation (`missing`) over the neighbour set, not the store', async () => {
|
||||
const results = await brain.find({
|
||||
connected: { from: anchor, direction: 'out' },
|
||||
where: { kind: 'note', retracted: { missing: true } },
|
||||
limit: 100
|
||||
})
|
||||
expect(results).toHaveLength(NEIGHBOURS - RETRACTED)
|
||||
for (const r of results) {
|
||||
expect(neighbourIds.has(r.entity.id)).toBe(true)
|
||||
expect(r.entity.metadata.retracted).toBeUndefined()
|
||||
}
|
||||
})
|
||||
|
||||
it('asks the metadata index about the neighbour ids only, and hydrates one page', async () => {
|
||||
const index = (brain as any).metadataIndex
|
||||
const within = vi.spyOn(index, 'filterIdsWithin')
|
||||
const hydrate = vi.spyOn(brain as any, 'batchGet')
|
||||
try {
|
||||
const results = await brain.find({
|
||||
connected: { from: anchor, direction: 'out' },
|
||||
where: { kind: 'note' },
|
||||
limit: 10
|
||||
})
|
||||
expect(results).toHaveLength(10)
|
||||
expect(within).toHaveBeenCalledTimes(1)
|
||||
const askedIds = within.mock.calls[0][1] as string[]
|
||||
expect(askedIds).toHaveLength(NEIGHBOURS)
|
||||
for (const id of askedIds) expect(neighbourIds.has(id)).toBe(true)
|
||||
expect(hydrate).toHaveBeenCalledTimes(1)
|
||||
expect(hydrate.mock.calls[0][0]).toHaveLength(10)
|
||||
} finally {
|
||||
within.mockRestore()
|
||||
hydrate.mockRestore()
|
||||
}
|
||||
})
|
||||
|
||||
it('orders the WHOLE neighbour set before cutting the page', async () => {
|
||||
const results = await brain.find({
|
||||
connected: { from: anchor, direction: 'out' },
|
||||
where: { kind: 'note' },
|
||||
orderBy: 'rank',
|
||||
order: 'desc',
|
||||
limit: 5
|
||||
})
|
||||
expect(results.map((r) => r.entity.metadata.rank)).toEqual([30, 29, 28, 27, 26])
|
||||
})
|
||||
|
||||
it('walks the vector leg over the neighbours only', async () => {
|
||||
const results = await brain.find({
|
||||
vector: sharedVector,
|
||||
connected: { from: anchor, direction: 'out' },
|
||||
where: { kind: 'note' },
|
||||
limit: 5
|
||||
})
|
||||
expect(results).toHaveLength(5)
|
||||
for (const r of results) expect(neighbourIds.has(r.entity.id)).toBe(true)
|
||||
})
|
||||
|
||||
it('an anchor without neighbours answers [] before the filter is asked', async () => {
|
||||
const index = (brain as any).metadataIndex
|
||||
const within = vi.spyOn(index, 'filterIdsWithin')
|
||||
try {
|
||||
const results = await brain.find({
|
||||
connected: { from: 'noise-0', direction: 'out' },
|
||||
where: { kind: 'note' },
|
||||
limit: 10
|
||||
})
|
||||
expect(results).toEqual([])
|
||||
expect(within).not.toHaveBeenCalled()
|
||||
} finally {
|
||||
within.mockRestore()
|
||||
}
|
||||
})
|
||||
})
|
||||
635
tests/integration/find-hybrid-filter-before-hydrate.test.ts
Normal file
635
tests/integration/find-hybrid-filter-before-hydrate.test.ts
Normal file
|
|
@ -0,0 +1,635 @@
|
|||
/**
|
||||
* @module tests/integration/find-hybrid-filter-before-hydrate
|
||||
* @description FILTER BEFORE HYDRATE, applied to the hybrid `find({ query })` path.
|
||||
*
|
||||
* A hybrid find fuses two legs. The semantic leg already walked only the
|
||||
* metadata filter's universe (`candidateIds` / `allowedIds`). The TEXT leg did
|
||||
* not: it ranked the WHOLE store, took the top `limit * 4`, read every one of
|
||||
* those rows from canonical, and only then intersected with the filter — so a
|
||||
* filtered hybrid find on a large store read hundreds of rows to return a
|
||||
* handful of them, and a matching row outside the store-wide text prefix was
|
||||
* silently dropped. That is the same defect `find({ connected })` carried
|
||||
* before the graph-first law, one leg over.
|
||||
*
|
||||
* Both halves are pinned here.
|
||||
*
|
||||
* THE ANSWER. Where the filter did not truncate the text leg — the universe
|
||||
* covers every text match, so both orders rank the same rows — the new
|
||||
* pipeline's answer is IDENTICAL to the old one's: same rows, same order, same
|
||||
* scores, same match visibility, same row shape. The oracle below is the
|
||||
* pre-change pipeline itself, replayed on the same brain through the same
|
||||
* doors, so the comparison is against what actually ran, not a remembered
|
||||
* expectation.
|
||||
*
|
||||
* THE CORRECTION. Where the filter DID truncate it — the query's words are
|
||||
* common outside the universe — the old order let the text leg contribute
|
||||
* nothing at all: every row it ranked was discarded by the filter, and the
|
||||
* answer came from the semantic leg alone. The new order ranks inside the
|
||||
* universe, so the text leg contributes the rows it always should have.
|
||||
*
|
||||
* THE COST. Canonical is read for exactly the page: one batch, `limit` rows,
|
||||
* never the legs. And the text leg is asked about the universe's ids only —
|
||||
* what it marshals is bounded by the universe, not by the store.
|
||||
*/
|
||||
import { describe, it, expect, beforeAll, vi } from 'vitest'
|
||||
import { Brainy } from '../../src/brainy'
|
||||
import { NounType, VerbType } from '../../src/types/graphTypes'
|
||||
import { rankIndicesByScore, reorderByIndices } from '../../src/utils/resultRanking'
|
||||
import { resolveEntityId } from '../../src/utils/idNormalization'
|
||||
|
||||
/** Embedding width of the default model — the row vectors must match it. */
|
||||
const DIM = 384
|
||||
|
||||
/**
|
||||
* A deterministic, per-row-distinct unit vector. Distinct so the semantic leg
|
||||
* has a real ranking to produce (identical vectors would make its order a tie
|
||||
* break), deterministic so the oracle and the pipeline see the same one.
|
||||
*/
|
||||
function seededVector(seed: number): number[] {
|
||||
const v = new Array<number>(DIM)
|
||||
for (let i = 0; i < DIM; i++) {
|
||||
v[i] = Math.sin((i + 1) * 0.11 + seed * 0.37) * 0.5 + Math.cos((i + 1) * 0.05 + seed * 0.13) * 0.3
|
||||
}
|
||||
const magnitude = Math.sqrt(v.reduce((sum, x) => sum + x * x, 0))
|
||||
return v.map((x) => x / magnitude)
|
||||
}
|
||||
|
||||
/** The fields a caller reads off a hybrid row — the whole comparable surface. */
|
||||
function project(rows: any[]): any[] {
|
||||
return rows.map((r) => ({
|
||||
id: r.id,
|
||||
score: r.score,
|
||||
type: r.type,
|
||||
metadata: r.metadata,
|
||||
textMatches: r.textMatches,
|
||||
textScore: r.textScore,
|
||||
semanticScore: r.semanticScore,
|
||||
matchSource: r.matchSource
|
||||
}))
|
||||
}
|
||||
|
||||
/**
|
||||
* The PRE-CHANGE hybrid pipeline, replayed on a live brain through the same
|
||||
* provider doors it used: whole-store text ranking with both legs hydrated in
|
||||
* full, RRF fusion, then the metadata intersection, then the page.
|
||||
*
|
||||
* Supports the shapes these pins exercise (query + where/type/excludeVFS +
|
||||
* connected + offset); `orderBy`, `fusion` and `near` are not replayed.
|
||||
*/
|
||||
async function legacyHybridFind(brain: any, params: any): Promise<any[]> {
|
||||
const index = brain.metadataIndex
|
||||
const limit = params.limit ?? 10
|
||||
const offset = params.offset ?? 0
|
||||
const hasFilter = Boolean(
|
||||
params.where || params.type || params.subtype || params.service || params.excludeVFS
|
||||
)
|
||||
|
||||
let preResolvedMetadataIds: string[] | null = null
|
||||
let preResolvedFilter: any = null
|
||||
let graphFirstIds: string[] | null = null
|
||||
|
||||
if (params.connected) {
|
||||
// find() normalizes the anchors to canonical ids before this stage runs.
|
||||
const anchored = {
|
||||
...params,
|
||||
connected: {
|
||||
...params.connected,
|
||||
...(params.connected.from && { from: resolveEntityId(params.connected.from) }),
|
||||
...(params.connected.to && { to: resolveEntityId(params.connected.to) })
|
||||
}
|
||||
}
|
||||
graphFirstIds = await brain.resolveConnectedIds(anchored)
|
||||
if (graphFirstIds!.length > 0 && hasFilter) {
|
||||
preResolvedFilter = brain.buildMetadataFilter(params)
|
||||
graphFirstIds = await brain.filterIdsWithinBelted(preResolvedFilter, graphFirstIds)
|
||||
}
|
||||
if (graphFirstIds!.length === 0) return []
|
||||
preResolvedMetadataIds = graphFirstIds
|
||||
} else if (hasFilter) {
|
||||
preResolvedFilter = brain.buildMetadataFilter(params)
|
||||
preResolvedMetadataIds = await brain.filterIdsBelted(preResolvedFilter)
|
||||
if (preResolvedMetadataIds!.length === 0) return []
|
||||
}
|
||||
|
||||
// Text leg — the whole store, then the top `limit * 4`, hydrated in full.
|
||||
const allTextMatches = await index.getIdsForTextQuery(params.query)
|
||||
const topMatches = allTextMatches.slice(0, limit * 2 * 2)
|
||||
const maxMatches = topMatches[0]?.matchCount || 1
|
||||
const textEntities = await brain.batchGet(topMatches.map((m: any) => m.id))
|
||||
const textResults = topMatches
|
||||
.filter((m: any) => textEntities.has(m.id))
|
||||
.map((m: any) => ({ id: m.id, score: m.matchCount / maxMatches }))
|
||||
|
||||
// Semantic leg — the beam walk over the universe, hydrated in full.
|
||||
const vector = await brain.embed(params.query)
|
||||
const searchOptions = preResolvedMetadataIds ? { candidateIds: preResolvedMetadataIds } : undefined
|
||||
const searchResults: [string, number][] = await brain.index.search(
|
||||
vector,
|
||||
limit * 2,
|
||||
undefined,
|
||||
searchOptions
|
||||
)
|
||||
const semanticEntities = await brain.batchGet(searchResults.map(([id]) => id))
|
||||
const semanticResults = searchResults
|
||||
.filter(([id]) => semanticEntities.has(id))
|
||||
.map(([id, distance]) => ({ id, score: Math.max(0, Math.min(1, 1 / (1 + distance))) }))
|
||||
|
||||
// RRF fusion, with the match visibility the rows carried.
|
||||
const alpha = params.hybridAlpha ?? brain.autoAlpha(params.query)
|
||||
const k = 60
|
||||
const matchData = new Map<string, any>()
|
||||
const textWeight = 1 - alpha
|
||||
textResults.forEach((r: any, rank: number) => {
|
||||
const existing = matchData.get(r.id) || { rrf: 0, hasText: false, hasSemantic: false }
|
||||
existing.rrf += textWeight * (1 / (k + rank + 1))
|
||||
existing.textScore = r.score
|
||||
existing.hasText = true
|
||||
matchData.set(r.id, existing)
|
||||
})
|
||||
semanticResults.forEach((r: any, rank: number) => {
|
||||
const existing = matchData.get(r.id) || { rrf: 0, hasText: false, hasSemantic: false }
|
||||
existing.rrf += alpha * (1 / (k + rank + 1))
|
||||
existing.semanticScore = r.score
|
||||
existing.hasSemantic = true
|
||||
matchData.set(r.id, existing)
|
||||
})
|
||||
|
||||
const queryWords: string[] = index.tokenize(params.query)
|
||||
const textResultIds = new Set(textResults.map((r: any) => r.id))
|
||||
const fusedIds = Array.from(matchData.entries())
|
||||
.sort((a, b) => b[1].rrf - a[1].rrf)
|
||||
.map(([id, data]) => ({ id, data }))
|
||||
|
||||
const allEntities = await brain.batchGet(fusedIds.map((f) => f.id))
|
||||
let rows: any[] = []
|
||||
for (const { id, data } of fusedIds) {
|
||||
const entity = allEntities.get(id)
|
||||
if (!entity) continue
|
||||
const textContent = textResultIds.has(id)
|
||||
? index.extractTextContent({ data: entity.data, metadata: entity.metadata }).toLowerCase()
|
||||
: null
|
||||
rows.push({
|
||||
id,
|
||||
score: data.rrf,
|
||||
type: entity.type,
|
||||
metadata: entity.metadata,
|
||||
textMatches:
|
||||
textContent === null ? [] : queryWords.filter((w) => textContent.includes(w.toLowerCase())),
|
||||
textScore: data.textScore,
|
||||
semanticScore: data.semanticScore,
|
||||
matchSource: data.hasText && data.hasSemantic ? 'both' : data.hasText ? 'text' : 'semantic'
|
||||
})
|
||||
}
|
||||
|
||||
// The metadata intersection — after the legs, as it was.
|
||||
if (preResolvedMetadataIds && preResolvedFilter) {
|
||||
const filteredIdSet = new Set(preResolvedMetadataIds)
|
||||
rows = rows.filter((r) => filteredIdSet.has(r.id))
|
||||
}
|
||||
if (graphFirstIds !== null) {
|
||||
const neighbourSet = new Set(graphFirstIds)
|
||||
rows = rows.filter((r) => neighbourSet.has(r.id))
|
||||
}
|
||||
|
||||
// Rank to the page, then cut it.
|
||||
const order = rankIndicesByScore(
|
||||
rows.map((r) => r.score),
|
||||
offset + limit,
|
||||
true
|
||||
)
|
||||
return reorderByIndices(rows, order).slice(offset, offset + limit)
|
||||
}
|
||||
|
||||
/**
|
||||
* FIXTURE A — the filter's universe covers every text match, so the two orders
|
||||
* rank exactly the same rows and the answers must be identical.
|
||||
*/
|
||||
describe('hybrid find: filter before hydrate — the answer is unchanged', () => {
|
||||
let brain: Brainy<any>
|
||||
const QUERY = 'orbital telemetry'
|
||||
const MATCHES = 24
|
||||
const FILLER = 120
|
||||
const OUTSIDE = 30
|
||||
const VFS = 10
|
||||
const RETRACTED = 6
|
||||
const anchor = 'array-anchor'
|
||||
const matchIds: string[] = []
|
||||
|
||||
beforeAll(async () => {
|
||||
brain = new Brainy({ requireSubtype: false, storage: { type: 'memory' } })
|
||||
await brain.init()
|
||||
|
||||
let seed = 1
|
||||
await brain.add({
|
||||
id: anchor,
|
||||
data: 'ground station anchor record',
|
||||
type: NounType.Thing,
|
||||
metadata: { lane: 'alpha', role: 'anchor' },
|
||||
vector: seededVector(seed++)
|
||||
})
|
||||
|
||||
// Rows the query's words actually match — all inside every filter below.
|
||||
for (let i = 0; i < MATCHES; i++) {
|
||||
const id = `match-${i}`
|
||||
await brain.add({
|
||||
id,
|
||||
data: `orbital telemetry packet ${i} recorded downlink`,
|
||||
type: NounType.Document,
|
||||
metadata: { lane: 'alpha', rank: i },
|
||||
vector: seededVector(seed++)
|
||||
})
|
||||
matchIds.push(resolveEntityId(id))
|
||||
await brain.relate({ from: anchor, to: id, type: VerbType.RelatedTo })
|
||||
}
|
||||
// Rows inside the universe that the query's words do NOT match.
|
||||
for (let i = 0; i < FILLER; i++) {
|
||||
await brain.add({
|
||||
id: `filler-${i}`,
|
||||
data: `cistern ledger entry ${i} archived`,
|
||||
type: NounType.Document,
|
||||
metadata: { lane: 'alpha', rank: 1000 + i },
|
||||
vector: seededVector(seed++)
|
||||
})
|
||||
}
|
||||
// Rows outside the universe.
|
||||
for (let i = 0; i < OUTSIDE; i++) {
|
||||
await brain.add({
|
||||
id: `outside-${i}`,
|
||||
data: `unrelated dossier ${i}`,
|
||||
type: NounType.Person,
|
||||
metadata: { lane: 'beta' },
|
||||
vector: seededVector(seed++)
|
||||
})
|
||||
}
|
||||
// VFS infrastructure rows — excluded by excludeVFS.
|
||||
for (let i = 0; i < VFS; i++) {
|
||||
await brain.add({
|
||||
id: `vfs-${i}`,
|
||||
data: `mounted path ${i}`,
|
||||
type: NounType.Document,
|
||||
metadata: { lane: 'alpha', vfsType: 'file' },
|
||||
vector: seededVector(seed++)
|
||||
})
|
||||
}
|
||||
// Retracted rows — excluded by a `missing` negation.
|
||||
for (let i = 0; i < RETRACTED; i++) {
|
||||
await brain.add({
|
||||
id: `retracted-${i}`,
|
||||
data: `withdrawn note ${i}`,
|
||||
type: NounType.Document,
|
||||
metadata: { lane: 'alpha', retracted: true },
|
||||
vector: seededVector(seed++)
|
||||
})
|
||||
}
|
||||
|
||||
// The reference index has no opaque-set door, so the pipeline and the
|
||||
// oracle both restrict the beam walk with the materialized candidate ids.
|
||||
expect(typeof (brain as any).metadataIndex.getIdSetForFilter).not.toBe('function')
|
||||
})
|
||||
|
||||
it('the fixture does not truncate the text leg — the universe covers every text match', async () => {
|
||||
const index = (brain as any).metadataIndex
|
||||
const textMatches = await index.getIdsForTextQuery(QUERY)
|
||||
expect(textMatches).toHaveLength(MATCHES)
|
||||
const universe = await (brain as any).filterIdsBelted({ lane: 'alpha' })
|
||||
const inUniverse = new Set(universe)
|
||||
for (const m of textMatches) expect(inUniverse.has(m.id)).toBe(true)
|
||||
})
|
||||
|
||||
it('hybrid + where: identical rows, identical order, identical scores', async () => {
|
||||
const params = { query: QUERY, where: { lane: 'alpha' }, limit: 8 }
|
||||
const expected = await legacyHybridFind(brain as any, params)
|
||||
const actual = await brain.find(params as any)
|
||||
expect(actual.length).toBe(expected.length)
|
||||
expect(project(actual)).toEqual(expected)
|
||||
})
|
||||
|
||||
it('hybrid + where + offset: identical page two', async () => {
|
||||
const params = { query: QUERY, where: { lane: 'alpha' }, limit: 6, offset: 6 }
|
||||
const expected = await legacyHybridFind(brain as any, params)
|
||||
const actual = await brain.find(params as any)
|
||||
expect(actual.length).toBe(expected.length)
|
||||
expect(project(actual)).toEqual(expected)
|
||||
})
|
||||
|
||||
it('hybrid + type list + excludeVFS + a `missing` negation: identical', async () => {
|
||||
const params = {
|
||||
query: QUERY,
|
||||
type: [NounType.Document, NounType.Person],
|
||||
excludeVFS: true,
|
||||
where: { lane: 'alpha', retracted: { missing: true } },
|
||||
limit: 8
|
||||
}
|
||||
const expected = await legacyHybridFind(brain as any, params)
|
||||
const actual = await brain.find(params as any)
|
||||
expect(actual.length).toBe(expected.length)
|
||||
expect(project(actual)).toEqual(expected)
|
||||
for (const r of actual) {
|
||||
expect(r.metadata.retracted).toBeUndefined()
|
||||
expect(r.metadata.vfsType).toBeUndefined()
|
||||
}
|
||||
})
|
||||
|
||||
it('hybrid + type list + excludeVFS + a `missing` negation, offset: identical', async () => {
|
||||
const params = {
|
||||
query: QUERY,
|
||||
type: [NounType.Document, NounType.Person],
|
||||
excludeVFS: true,
|
||||
where: { lane: 'alpha', retracted: { missing: true } },
|
||||
limit: 5,
|
||||
offset: 5
|
||||
}
|
||||
const expected = await legacyHybridFind(brain as any, params)
|
||||
const actual = await brain.find(params as any)
|
||||
expect(actual.length).toBe(expected.length)
|
||||
expect(project(actual)).toEqual(expected)
|
||||
})
|
||||
|
||||
it('hybrid + connected: identical, and never a non-neighbour', async () => {
|
||||
const params = {
|
||||
query: QUERY,
|
||||
connected: { from: anchor, direction: 'out' as const },
|
||||
where: { lane: 'alpha' },
|
||||
limit: 8
|
||||
}
|
||||
const expected = await legacyHybridFind(brain as any, params)
|
||||
const actual = await brain.find(params as any)
|
||||
expect(actual.length).toBe(expected.length)
|
||||
expect(project(actual)).toEqual(expected)
|
||||
const neighbours = new Set(matchIds)
|
||||
for (const r of actual) expect(neighbours.has(r.id)).toBe(true)
|
||||
})
|
||||
|
||||
it('hybrid + connected + offset: page two is the page, not an empty answer', async () => {
|
||||
const params = {
|
||||
query: QUERY,
|
||||
connected: { from: anchor, direction: 'out' as const },
|
||||
where: { lane: 'alpha' },
|
||||
limit: 5,
|
||||
offset: 5
|
||||
}
|
||||
const expected = await legacyHybridFind(brain as any, params)
|
||||
expect(expected).toHaveLength(5)
|
||||
const actual = await brain.find(params as any)
|
||||
expect(actual.length).toBe(expected.length)
|
||||
expect(project(actual)).toEqual(expected)
|
||||
})
|
||||
|
||||
it('hybrid + connected: paging reaches every matching neighbour exactly once', async () => {
|
||||
const seen = new Set<string>()
|
||||
for (let offset = 0; offset < MATCHES; offset += 6) {
|
||||
const page = await brain.find({
|
||||
query: QUERY,
|
||||
connected: { from: anchor, direction: 'out' as const },
|
||||
where: { lane: 'alpha' },
|
||||
limit: 6,
|
||||
offset
|
||||
} as any)
|
||||
for (const r of page) {
|
||||
expect(seen.has(r.id)).toBe(false)
|
||||
seen.add(r.id)
|
||||
}
|
||||
}
|
||||
// Every row the fused candidate set holds is reachable by paging, and the
|
||||
// neighbour set is the ceiling.
|
||||
expect(seen.size).toBeGreaterThanOrEqual(MATCHES)
|
||||
const neighbours = new Set(matchIds)
|
||||
for (const id of seen) expect(neighbours.has(id)).toBe(true)
|
||||
})
|
||||
|
||||
it('hybrid + fusion + offset: page two is the page', async () => {
|
||||
const plain = await brain.find({
|
||||
query: QUERY,
|
||||
where: { lane: 'alpha' },
|
||||
limit: 5,
|
||||
offset: 5
|
||||
} as any)
|
||||
const fused = await brain.find({
|
||||
query: QUERY,
|
||||
where: { lane: 'alpha' },
|
||||
fusion: 'weighted',
|
||||
limit: 5,
|
||||
offset: 5
|
||||
} as any)
|
||||
expect(fused).toHaveLength(plain.length)
|
||||
expect(fused.map((r) => r.id)).toEqual(plain.map((r) => r.id))
|
||||
})
|
||||
|
||||
it('a hydrated hybrid row is shaped exactly as an eagerly-built one', async () => {
|
||||
const rows = await brain.find({ query: QUERY, where: { lane: 'alpha' }, limit: 8 } as any)
|
||||
const row = rows[0]
|
||||
expect(Object.keys(row)).toEqual([
|
||||
'id',
|
||||
'score',
|
||||
'type',
|
||||
'subtype',
|
||||
'visibility',
|
||||
'metadata',
|
||||
'data',
|
||||
'confidence',
|
||||
'weight',
|
||||
'_rev',
|
||||
'entity',
|
||||
'textMatches',
|
||||
'textScore',
|
||||
'semanticScore',
|
||||
'matchSource'
|
||||
])
|
||||
// The flattened fields are projections of the entity, as always.
|
||||
expect(row.entity).toBeDefined()
|
||||
expect(row.type).toBe(row.entity.type)
|
||||
expect(row.metadata).toBe(row.entity.metadata)
|
||||
expect(row.data).toBe(row.entity.data)
|
||||
expect(row._rev).toBe(row.entity._rev)
|
||||
// The match visibility survives the deferral — every leg's fields, on the
|
||||
// rows that leg contributed, exactly as the eager pipeline set them.
|
||||
expect(['text', 'semantic', 'both']).toContain(row.matchSource)
|
||||
for (const r of rows) {
|
||||
if (r.matchSource === 'semantic') {
|
||||
expect(r.textMatches).toEqual([])
|
||||
expect(r.textScore).toBeUndefined()
|
||||
} else {
|
||||
expect(r.textMatches).toEqual(['orbital', 'telemetry'])
|
||||
expect(typeof r.textScore).toBe('number')
|
||||
}
|
||||
if (r.matchSource === 'text') {
|
||||
expect(r.semanticScore).toBeUndefined()
|
||||
} else {
|
||||
expect(typeof r.semanticScore).toBe('number')
|
||||
}
|
||||
}
|
||||
})
|
||||
|
||||
it('reads canonical for the page only — one batch, `limit` rows', async () => {
|
||||
// Warm any first-read verification before the counters are read.
|
||||
await brain.find({ query: QUERY, where: { lane: 'alpha' }, limit: 1 } as any)
|
||||
|
||||
const hydrate = vi.spyOn(brain as any, 'batchGet')
|
||||
try {
|
||||
const results = await brain.find({ query: QUERY, where: { lane: 'alpha' }, limit: 10 } as any)
|
||||
expect(results).toHaveLength(10)
|
||||
expect(hydrate).toHaveBeenCalledTimes(1)
|
||||
expect((hydrate.mock.calls[0][0] as string[]).length).toBe(10)
|
||||
} finally {
|
||||
hydrate.mockRestore()
|
||||
}
|
||||
})
|
||||
|
||||
it('asks the text index about the universe only, never the whole store', async () => {
|
||||
const index = (brain as any).metadataIndex
|
||||
const wholeStore = vi.spyOn(index, 'getIdsForTextQuery')
|
||||
const within = vi.spyOn(index, 'getIdsForTextQueryWithin')
|
||||
try {
|
||||
await brain.find({ query: QUERY, where: { lane: 'alpha' }, limit: 10 } as any)
|
||||
expect(wholeStore).not.toHaveBeenCalled()
|
||||
expect(within).toHaveBeenCalledTimes(1)
|
||||
|
||||
const askedIds = within.mock.calls[0][1] as string[]
|
||||
const universe = await (brain as any).filterIdsBelted({ lane: 'alpha' })
|
||||
expect(askedIds).toHaveLength(universe.length)
|
||||
|
||||
// What the text leg marshals is bounded by the universe, not the store.
|
||||
const marshalled = (await within.mock.results[0].value) as unknown[]
|
||||
expect(marshalled.length).toBeLessThanOrEqual(universe.length)
|
||||
expect(marshalled).toHaveLength(MATCHES)
|
||||
} finally {
|
||||
wholeStore.mockRestore()
|
||||
within.mockRestore()
|
||||
}
|
||||
})
|
||||
|
||||
it('the two text doors agree: within is the whole-store answer restricted', async () => {
|
||||
const index = (brain as any).metadataIndex
|
||||
const universe: string[] = await (brain as any).filterIdsBelted({
|
||||
lane: 'alpha',
|
||||
retracted: { missing: true }
|
||||
})
|
||||
const inUniverse = new Set(universe)
|
||||
const whole = await index.getIdsForTextQuery(QUERY)
|
||||
const within = await index.getIdsForTextQueryWithin(QUERY, universe)
|
||||
expect(within).toEqual(whole.filter((m: any) => inUniverse.has(m.id)))
|
||||
expect(await index.getIdsForTextQueryWithin(QUERY, [])).toEqual([])
|
||||
})
|
||||
})
|
||||
|
||||
/**
|
||||
* FIXTURE B — the query's words are common OUTSIDE the universe, so the old
|
||||
* order's text leg was entirely consumed by rows the filter then discarded.
|
||||
* This is the corrected behaviour, held by name.
|
||||
*/
|
||||
describe('hybrid find: the text leg ranks inside the filter, not around it', () => {
|
||||
let brain: Brainy<any>
|
||||
const QUERY = 'orbital telemetry drift'
|
||||
const NOISE = 150
|
||||
const KEEP = 15
|
||||
const keepIds: string[] = []
|
||||
|
||||
beforeAll(async () => {
|
||||
brain = new Brainy({ requireSubtype: false, storage: { type: 'memory' } })
|
||||
await brain.init()
|
||||
|
||||
let seed = 5000
|
||||
// Added FIRST and matching one more query word, so they lead the
|
||||
// store-wide text ranking outright — and none of them pass the filter.
|
||||
for (let i = 0; i < NOISE; i++) {
|
||||
await brain.add({
|
||||
id: `noise-${i}`,
|
||||
data: `orbital telemetry drift report ${i}`,
|
||||
type: NounType.Document,
|
||||
metadata: { lane: 'beta' },
|
||||
vector: seededVector(seed++)
|
||||
})
|
||||
}
|
||||
for (let i = 0; i < KEEP; i++) {
|
||||
const id = `keep-${i}`
|
||||
await brain.add({
|
||||
id,
|
||||
data: `orbital telemetry summary ${i}`,
|
||||
type: NounType.Document,
|
||||
metadata: { lane: 'alpha' },
|
||||
vector: seededVector(seed++)
|
||||
})
|
||||
keepIds.push(resolveEntityId(id))
|
||||
}
|
||||
})
|
||||
|
||||
it('the old order let the filter consume the whole text leg', async () => {
|
||||
const index = (brain as any).metadataIndex
|
||||
const universe: string[] = await (brain as any).filterIdsBelted({ lane: 'alpha' })
|
||||
expect(universe).toHaveLength(KEEP)
|
||||
const inUniverse = new Set(universe)
|
||||
|
||||
// The store-wide prefix the old text leg took (limit 10 → limit * 4).
|
||||
const prefix = (await index.getIdsForTextQuery(QUERY)).slice(0, 40)
|
||||
expect(prefix).toHaveLength(40)
|
||||
expect(prefix.filter((m: any) => inUniverse.has(m.id))).toHaveLength(0)
|
||||
|
||||
// Every row the old text leg ranked was then discarded by the filter, so
|
||||
// the old answer carried NO text contribution at all — fifteen rows that
|
||||
// match the query's words exactly, and not one of them reached the page
|
||||
// through the text leg. What the old order returned was whatever the
|
||||
// semantic leg alone happened to reach.
|
||||
const legacy = await legacyHybridFind(brain as any, {
|
||||
query: QUERY,
|
||||
where: { lane: 'alpha' },
|
||||
limit: 10
|
||||
})
|
||||
for (const r of legacy) {
|
||||
expect(r.matchSource).toBe('semantic')
|
||||
expect(r.textScore).toBeUndefined()
|
||||
expect(r.textMatches).toEqual([])
|
||||
}
|
||||
})
|
||||
|
||||
it('the new order ranks the text leg inside the universe', async () => {
|
||||
const results = await brain.find({
|
||||
query: QUERY,
|
||||
where: { lane: 'alpha' },
|
||||
limit: 10
|
||||
} as any)
|
||||
|
||||
expect(results).toHaveLength(10)
|
||||
const keeps = new Set(keepIds)
|
||||
for (const r of results) {
|
||||
expect(keeps.has(r.id)).toBe(true)
|
||||
expect(r.metadata.lane).toBe('alpha')
|
||||
// The text leg is the contributor the old order threw away.
|
||||
expect(['text', 'both']).toContain(r.matchSource)
|
||||
expect(r.textScore).toBe(1)
|
||||
expect(r.textMatches).toEqual(['orbital', 'telemetry'])
|
||||
}
|
||||
})
|
||||
|
||||
it('paging reaches every matching row the old order could not see', async () => {
|
||||
const seen = new Set<string>()
|
||||
for (let offset = 0; offset < KEEP; offset += 5) {
|
||||
const page = await brain.find({
|
||||
query: QUERY,
|
||||
where: { lane: 'alpha' },
|
||||
limit: 5,
|
||||
offset
|
||||
} as any)
|
||||
expect(page).toHaveLength(5)
|
||||
for (const r of page) {
|
||||
expect(seen.has(r.id)).toBe(false)
|
||||
seen.add(r.id)
|
||||
}
|
||||
}
|
||||
expect(seen.size).toBe(KEEP)
|
||||
expect([...seen].sort()).toEqual([...keepIds].sort())
|
||||
})
|
||||
|
||||
it('reads canonical for the page only, on the truncating shape too', async () => {
|
||||
await brain.find({ query: QUERY, where: { lane: 'alpha' }, limit: 1 } as any)
|
||||
|
||||
const hydrate = vi.spyOn(brain as any, 'batchGet')
|
||||
try {
|
||||
const results = await brain.find({ query: QUERY, where: { lane: 'alpha' }, limit: 10 } as any)
|
||||
expect(results).toHaveLength(10)
|
||||
expect(hydrate).toHaveBeenCalledTimes(1)
|
||||
expect((hydrate.mock.calls[0][0] as string[]).length).toBe(10)
|
||||
} finally {
|
||||
hydrate.mockRestore()
|
||||
}
|
||||
})
|
||||
})
|
||||
48
tests/integration/find-near.test.ts
Normal file
48
tests/integration/find-near.test.ts
Normal file
|
|
@ -0,0 +1,48 @@
|
|||
/**
|
||||
* @module tests/integration/find-near
|
||||
* @description find({ near }) searches around the anchor's OWN vector (10.4.10).
|
||||
*
|
||||
* The proximity search fetched its anchor without vectors and fed a
|
||||
* zero-length vector to the index — every near() refused with a dimension
|
||||
* mismatch, for every caller. Found by the Rust planner's first-contact pins
|
||||
* (the planner declines `near`; the pin compared outcomes with and without
|
||||
* it). Now the anchor is fetched with its vector, and an anchor without one
|
||||
* refuses by name instead of failing inside the index.
|
||||
*/
|
||||
import { describe, it, expect, beforeAll } from 'vitest'
|
||||
import { Brainy } from '../../src/brainy'
|
||||
import { NounType } from '../../src/types/graphTypes'
|
||||
import { v5 } from '../../src/universal/uuid'
|
||||
import { generateTestVector } from '../helpers/test-factory'
|
||||
|
||||
describe('find({ near }) uses the anchor vector', () => {
|
||||
let brain: Brainy<any>
|
||||
const anchorVector = generateTestVector()
|
||||
|
||||
beforeAll(async () => {
|
||||
brain = new Brainy({ requireSubtype: false, storage: { type: 'memory' } })
|
||||
await brain.init()
|
||||
await brain.add({ id: 'anchor', data: 'anchor row', type: NounType.Thing, vector: anchorVector })
|
||||
// A twin with the identical vector and a far row.
|
||||
await brain.add({ id: 'twin', data: 'twin row', type: NounType.Thing, vector: [...anchorVector] })
|
||||
await brain.add({ id: 'far', data: 'far row', type: NounType.Thing, vector: generateTestVector() })
|
||||
})
|
||||
|
||||
it('returns the anchor\'s neighbours by its own vector', async () => {
|
||||
const results = await brain.find({ near: { id: 'anchor' }, limit: 3 })
|
||||
expect(results.length).toBeGreaterThan(0)
|
||||
const ids = results.map((r) => r.entity.id)
|
||||
expect(ids).toContain(v5('twin'))
|
||||
})
|
||||
|
||||
it('refuses by name when the anchor has no vector', async () => {
|
||||
await brain.add({
|
||||
id: 'unvectored',
|
||||
data: 'no vector here',
|
||||
type: NounType.Thing,
|
||||
deferEmbedding: true
|
||||
})
|
||||
;(brain as any).kickEmbedWorker = () => {}
|
||||
await expect(brain.find({ near: { id: 'unvectored' }, limit: 3 })).rejects.toThrow(/has no vector to search around/)
|
||||
})
|
||||
})
|
||||
137
tests/integration/find-planner-door.test.ts
Normal file
137
tests/integration/find-planner-door.test.ts
Normal file
|
|
@ -0,0 +1,137 @@
|
|||
/**
|
||||
* @module tests/integration/find-planner-door
|
||||
* @description The optional `MetadataIndexProvider.planFindPage` door.
|
||||
*
|
||||
* The stage doors each serve one stage, so a `find()` that consults three of
|
||||
* them crosses into the index three times and marshals a result set at every
|
||||
* crossing — a filter matching a hundred thousand rows builds a hundred
|
||||
* thousand id strings to return a page of twenty-five. An index that can decide
|
||||
* the stage order itself answers the page in one call.
|
||||
*
|
||||
* These pins hold the three properties that make such a door safe to add:
|
||||
*
|
||||
* 1. **Absent, nothing changes.** The reference index has no planner, and every
|
||||
* find is served by the stage doors exactly as before. That is also what
|
||||
* makes this engine the ordering oracle for any index that implements one.
|
||||
* 2. **Present, it is asked first and its answer is used** — above the branch
|
||||
* selection, with the params already normalized, the hidden ids passed, and
|
||||
* the graph provider handed over.
|
||||
* 3. **`null` is routing, not an answer.** A door that declines a shape leaves
|
||||
* it to the path that always served it, and the result is unchanged.
|
||||
*
|
||||
* Plus the serving law: an empty page stamped `emptyAt: 'graph'` is re-verified
|
||||
* against the adjacency before it is believed, so a not-serving graph refuses
|
||||
* loudly instead of answering `[]` as truth.
|
||||
*/
|
||||
import { describe, it, expect, beforeAll, vi } from 'vitest'
|
||||
import { Brainy } from '../../src/brainy'
|
||||
import { NounType, VerbType } from '../../src/types/graphTypes'
|
||||
import { generateTestVector } from '../helpers/test-factory'
|
||||
|
||||
describe('find(): the optional planner door', () => {
|
||||
let brain: Brainy<any>
|
||||
const anchor = 'planner-anchor'
|
||||
let neighbourId = ''
|
||||
|
||||
beforeAll(async () => {
|
||||
brain = new Brainy({ requireSubtype: false, storage: { type: 'memory' } })
|
||||
await brain.init()
|
||||
await brain.add({
|
||||
id: anchor,
|
||||
data: 'anchor',
|
||||
type: NounType.Person,
|
||||
metadata: { kind: 'anchor' },
|
||||
vector: generateTestVector()
|
||||
})
|
||||
for (let i = 0; i < 12; i++) {
|
||||
const id = await brain.add({
|
||||
id: `row-${i}`,
|
||||
data: `row ${i}`,
|
||||
type: NounType.Person,
|
||||
metadata: { kind: 'note', rank: i },
|
||||
vector: generateTestVector()
|
||||
})
|
||||
if (i === 0) neighbourId = id
|
||||
await brain.relate({ from: anchor, to: id, type: VerbType.Knows })
|
||||
}
|
||||
})
|
||||
|
||||
/** Install a planner door for one call, then remove it. */
|
||||
const withDoor = async <T>(
|
||||
door: (...a: any[]) => Promise<any>,
|
||||
body: () => Promise<T>
|
||||
): Promise<T> => {
|
||||
const index = (brain as any).metadataIndex
|
||||
index.planFindPage = door
|
||||
try {
|
||||
return await body()
|
||||
} finally {
|
||||
delete index.planFindPage
|
||||
}
|
||||
}
|
||||
|
||||
it('is absent on the reference index — every find is served by the stage doors', async () => {
|
||||
expect((brain as any).metadataIndex.planFindPage).toBeUndefined()
|
||||
const results = await brain.find({ where: { kind: 'note' }, limit: 5 })
|
||||
expect(results).toHaveLength(5)
|
||||
})
|
||||
|
||||
it('is asked before the branches, with normalized params and the graph provider', async () => {
|
||||
const door = vi.fn(async () => null)
|
||||
await withDoor(door, async () => {
|
||||
await brain.find({ where: { kind: 'note' }, limit: 5 })
|
||||
})
|
||||
expect(door).toHaveBeenCalledTimes(1)
|
||||
const [params, hidden, graph] = door.mock.calls[0] as any[]
|
||||
expect(params.where).toEqual({ kind: 'note' })
|
||||
expect(Array.isArray(hidden)).toBe(true)
|
||||
expect(graph).toBe((brain as any).graphIndex)
|
||||
})
|
||||
|
||||
it('uses the page it answers, hydrated and in the door\'s order', async () => {
|
||||
const results = await withDoor(
|
||||
async () => ({ ids: [neighbourId], emptyAt: 'none' as const }),
|
||||
async () => brain.find({ where: { kind: 'note' }, limit: 5 })
|
||||
)
|
||||
expect(results).toHaveLength(1)
|
||||
expect(results[0].entity.id).toBe(neighbourId)
|
||||
})
|
||||
|
||||
it('a declining door changes nothing — the shape is served as it always was', async () => {
|
||||
const withoutDoor = await brain.find({ where: { kind: 'note' }, orderBy: 'rank', limit: 4 })
|
||||
const declined = await withDoor(
|
||||
async () => null,
|
||||
async () => brain.find({ where: { kind: 'note' }, orderBy: 'rank', limit: 4 })
|
||||
)
|
||||
expect(declined.map((r) => r.entity.id)).toEqual(withoutDoor.map((r) => r.entity.id))
|
||||
})
|
||||
|
||||
it('re-verifies the adjacency before believing an empty graph answer', async () => {
|
||||
const verify = vi.spyOn(brain as any, 'verifyGraphAdjacencyLive')
|
||||
try {
|
||||
const results = await withDoor(
|
||||
async () => ({ ids: [], emptyAt: 'graph' as const }),
|
||||
async () => brain.find({ connected: { from: anchor }, where: { kind: 'note' }, limit: 5 })
|
||||
)
|
||||
expect(results).toEqual([])
|
||||
expect(verify).toHaveBeenCalled()
|
||||
} finally {
|
||||
verify.mockRestore()
|
||||
}
|
||||
})
|
||||
|
||||
it('does not re-verify the adjacency for an empty the FILTER produced', async () => {
|
||||
const verify = vi.spyOn(brain as any, 'verifyGraphAdjacencyLive')
|
||||
verify.mockClear()
|
||||
try {
|
||||
const results = await withDoor(
|
||||
async () => ({ ids: [], emptyAt: 'filter' as const }),
|
||||
async () => brain.find({ where: { kind: 'note' }, limit: 5 })
|
||||
)
|
||||
expect(results).toEqual([])
|
||||
expect(verify).not.toHaveBeenCalled()
|
||||
} finally {
|
||||
verify.mockRestore()
|
||||
}
|
||||
})
|
||||
})
|
||||
101
tests/integration/generation-store-factory.test.ts
Normal file
101
tests/integration/generation-store-factory.test.ts
Normal file
|
|
@ -0,0 +1,101 @@
|
|||
/**
|
||||
* @module tests/integration/generation-store-factory
|
||||
* @description Pins the `createGenerationStore` protected factory hook on
|
||||
* `Brainy` ({@link Brainy.createGenerationStore}). The hook exists so an
|
||||
* engine built on top of this reference implementation can substitute a
|
||||
* `GenerationStore` that keeps the same behavioural contract; this suite
|
||||
* proves two things:
|
||||
*
|
||||
* 1. A subclass overriding the hook is the ONLY path that constructs the
|
||||
* generation store — it is called exactly once, with the same storage
|
||||
* instance `performInit` holds — and the store the brain actually uses
|
||||
* is the one the override returned.
|
||||
* 2. The default (non-overridden) path is unaffected — proven here by
|
||||
* confirming the base class still produces a plain `GenerationStore`
|
||||
* wired to `brain.storage`, and separately by running the existing
|
||||
* `db-mvcc` and `brainy-core.integration` suites unmodified against this
|
||||
* change (they exercise generation-store behaviour end to end).
|
||||
*/
|
||||
|
||||
import { describe, it, expect, afterEach } from 'vitest'
|
||||
import { Brainy } from '../../src/brainy.js'
|
||||
import { GenerationStore } from '../../src/db/generationStore.js'
|
||||
import type { BaseStorage } from '../../src/storage/baseStorage.js'
|
||||
|
||||
/** Typed access to the brain's private storage + generation-store fields (test injection point). */
|
||||
function internalsOf(brain: Brainy): { storage: BaseStorage; generationStore: GenerationStore } {
|
||||
return brain as unknown as { storage: BaseStorage; generationStore: GenerationStore }
|
||||
}
|
||||
|
||||
/**
|
||||
* A `GenerationStore` subclass that counts its own construction and
|
||||
* remembers the storage instance it was built with, so the test can prove
|
||||
* the hook is the sole construction path without mocking the module.
|
||||
*/
|
||||
class SpyGenerationStore extends GenerationStore {
|
||||
static constructCount = 0
|
||||
static lastStorage: BaseStorage | undefined
|
||||
|
||||
constructor(storage: BaseStorage) {
|
||||
super(storage)
|
||||
SpyGenerationStore.constructCount++
|
||||
SpyGenerationStore.lastStorage = storage
|
||||
}
|
||||
}
|
||||
|
||||
/** A Brainy subclass overriding the factory hook — stands in for an engine built on the reference. */
|
||||
class BrainyWithSpyStore extends Brainy {
|
||||
hookCallCount = 0
|
||||
hookStorageArg: BaseStorage | undefined
|
||||
|
||||
protected override createGenerationStore(storage: BaseStorage): GenerationStore {
|
||||
this.hookCallCount++
|
||||
this.hookStorageArg = storage
|
||||
return new SpyGenerationStore(storage)
|
||||
}
|
||||
}
|
||||
|
||||
describe('Brainy.createGenerationStore — protected factory hook', () => {
|
||||
const brains: Brainy[] = []
|
||||
|
||||
afterEach(async () => {
|
||||
SpyGenerationStore.constructCount = 0
|
||||
SpyGenerationStore.lastStorage = undefined
|
||||
for (const brain of brains.splice(0)) {
|
||||
try {
|
||||
await brain.close()
|
||||
} catch {
|
||||
// already closed by the test
|
||||
}
|
||||
}
|
||||
})
|
||||
|
||||
it('a subclass override is the sole construction path: called once, same storage instance, its store is the one the brain uses', async () => {
|
||||
const brain = new BrainyWithSpyStore({ requireSubtype: false, storage: { type: 'memory' } })
|
||||
await brain.init()
|
||||
brains.push(brain)
|
||||
|
||||
// Called exactly once, through the hook.
|
||||
expect(brain.hookCallCount).toBe(1)
|
||||
expect(SpyGenerationStore.constructCount).toBe(1)
|
||||
|
||||
// Same storage instance the base class holds — not a copy, not a different adapter.
|
||||
const { storage, generationStore } = internalsOf(brain)
|
||||
expect(brain.hookStorageArg).toBe(storage)
|
||||
expect(SpyGenerationStore.lastStorage).toBe(storage)
|
||||
|
||||
// The store the brain actually uses is the one the override returned.
|
||||
expect(generationStore).toBeInstanceOf(SpyGenerationStore)
|
||||
})
|
||||
|
||||
it('the default (non-overridden) path still produces a plain GenerationStore wired to the same storage', async () => {
|
||||
const brain = new Brainy({ requireSubtype: false, storage: { type: 'memory' } })
|
||||
await brain.init()
|
||||
brains.push(brain)
|
||||
|
||||
const { storage, generationStore } = internalsOf(brain)
|
||||
expect(generationStore).toBeInstanceOf(GenerationStore)
|
||||
// The default implementation constructs from the same storage the brain holds.
|
||||
expect((generationStore as unknown as { storage: BaseStorage }).storage).toBe(storage)
|
||||
})
|
||||
})
|
||||
|
|
@ -16,6 +16,7 @@ import { describe, it, expect, afterEach } from 'vitest'
|
|||
import * as fs from 'node:fs'
|
||||
import * as path from 'node:path'
|
||||
import * as os from 'node:os'
|
||||
import * as zlib from 'node:zlib'
|
||||
import { Brainy } from '../../src/brainy.js'
|
||||
import { NounType } from '../../src/types/graphTypes.js'
|
||||
import { GenerationStore } from '../../src/db/generationStore.js'
|
||||
|
|
@ -57,6 +58,107 @@ describe('history repacking — the two-tier lifecycle', () => {
|
|||
}
|
||||
})
|
||||
|
||||
/**
|
||||
* THE HOLE, END TO END — the shape a real store carries.
|
||||
*
|
||||
* A forensic fixture was measured with generation directories 1..2503
|
||||
* present except for exactly one: 1416. Its fact-log segment already showed
|
||||
* the tell — `seg-...1410.bfl` declaring firstGeneration 1410, lastGeneration
|
||||
* 1940 (531 generations) while recording only 530 facts.
|
||||
*
|
||||
* Before the fix, repacking such a store folded ACROSS that hole: the batch
|
||||
* skipped 1416 (no readable delta) and the sealed segment declared a range
|
||||
* spanning it anyway. The next open merged that declared range back into
|
||||
* committedRanges, re-admitting 1416 as committed history, and every
|
||||
* subsequent auto-compaction pass then asked the packed tier for a frame
|
||||
* that was never written — producing, on EVERY run, the non-fatal narration
|
||||
*
|
||||
* Auto-compaction of generational history failed (non-fatal): generation
|
||||
* N is inside sealed segment seg-....bgs's declared range but has no frame
|
||||
* — packed history is damaged
|
||||
*
|
||||
* This pin removes a generation directory to make the same hole, then
|
||||
* requires repack + reopen + compaction to complete cleanly.
|
||||
*/
|
||||
it('a missing generation directory does not poison the packed tier', async () => {
|
||||
const dir = tempDir()
|
||||
// `retention: 'all'` throughout: close() otherwise auto-compacts the
|
||||
// history away, and this pin needs the cold generations still on disk so
|
||||
// there is something to punch a hole in. The live window stays at its
|
||||
// production default for the build phase, so nothing folds yet.
|
||||
const archival = async (): Promise<Brainy> => {
|
||||
const b = new Brainy({
|
||||
requireSubtype: false,
|
||||
storage: { type: 'filesystem', path: dir },
|
||||
embeddingFunction: stub,
|
||||
retention: 'all'
|
||||
})
|
||||
await b.init()
|
||||
return b
|
||||
}
|
||||
const brain = await archival()
|
||||
|
||||
const id = await brain.add({
|
||||
data: 'holed-entity',
|
||||
type: NounType.Document,
|
||||
metadata: { v: 0 }
|
||||
})
|
||||
// One flush per update: single-op writes coalesce inside a flush window,
|
||||
// so a history deep enough to have a middle needs the windows separated.
|
||||
for (let v = 1; v <= 12; v++) {
|
||||
await brain.update({ id, metadata: { v } })
|
||||
await brain.flush()
|
||||
}
|
||||
await brain.close()
|
||||
|
||||
// Punch the hole: delete ONE generation directory in the middle of the
|
||||
// cold range, exactly as the real store presents it.
|
||||
const genRoot = path.join(dir, '_generations')
|
||||
const numeric = fs
|
||||
.readdirSync(genRoot, { withFileTypes: true })
|
||||
.filter((e) => e.isDirectory() && /^\d+$/.test(e.name))
|
||||
.map((e) => Number(e.name))
|
||||
.sort((a, b) => a - b)
|
||||
expect(numeric.length).toBeGreaterThan(6)
|
||||
const victim = numeric[Math.floor(numeric.length / 2)]
|
||||
fs.rmSync(path.join(genRoot, String(victim)), { recursive: true, force: true })
|
||||
|
||||
// Now shrink the live window and reopen. close() repacks automatically
|
||||
// (brainy.ts phase 0b), so this is the production sequence exactly: a
|
||||
// store with a hole in its history gets folded by ordinary housekeeping,
|
||||
// with nobody asking for it.
|
||||
;(GenerationStore as any).REPACK_LIVE_WINDOW = 3
|
||||
const reopened = await archival()
|
||||
const result = await reopened.repackHistory()
|
||||
expect(result.foldedGenerations).toBeGreaterThan(0)
|
||||
|
||||
const segDir = path.join(dir, SEGMENTS_PREFIX)
|
||||
const manifestPath = ['manifest.json', 'manifest.json.gz']
|
||||
.map((f) => path.join(segDir, f))
|
||||
.find((p) => fs.existsSync(p))!
|
||||
const raw = manifestPath.endsWith('.gz')
|
||||
? zlib.gunzipSync(fs.readFileSync(manifestPath)).toString('utf8')
|
||||
: fs.readFileSync(manifestPath, 'utf8')
|
||||
const manifest = JSON.parse(raw) as {
|
||||
segments: Array<{ firstGeneration: number; lastGeneration: number; frames: number }>
|
||||
}
|
||||
|
||||
// THE LAW: every sealed segment declares exactly as many generations as it
|
||||
// holds frames, and none of them spans the victim.
|
||||
for (const s of manifest.segments) {
|
||||
expect(s.lastGeneration - s.firstGeneration + 1).toBe(s.frames)
|
||||
expect(victim >= s.firstGeneration && victim <= s.lastGeneration).toBe(false)
|
||||
}
|
||||
|
||||
await reopened.close()
|
||||
|
||||
// And the pass that used to fail on every run now completes: reopen (which
|
||||
// re-seeds committedRanges from the packed tier) then compact history.
|
||||
const third = await openBrain(dir)
|
||||
await expect(third.compactHistory({ maxGenerations: 2 })).resolves.toBeDefined()
|
||||
await third.close()
|
||||
})
|
||||
|
||||
it('repack preserves every historical read across cold reopen; folded dirs are gone', async () => {
|
||||
;(GenerationStore as any).REPACK_LIVE_WINDOW = 3
|
||||
const dir = tempDir()
|
||||
|
|
|
|||
547
tests/integration/pending-embed-checkpoint.test.ts
Normal file
547
tests/integration/pending-embed-checkpoint.test.ts
Normal file
|
|
@ -0,0 +1,547 @@
|
|||
/**
|
||||
* @module tests/integration/pending-embed-checkpoint
|
||||
* @description THE PENDING-EMBED CHECKPOINT — the bound that engages on the
|
||||
* brains that need it.
|
||||
*
|
||||
* 10.4.9 bounded the open-path `recover-pending-embeds` fold with a LOW-WATER
|
||||
* MARK: the log head at which the pending set last drained to EMPTY. That mark
|
||||
* carries no set, so it can only be written when the set is empty — and a brain
|
||||
* holding even ONE id that never lands (an embed that keeps failing, a worker
|
||||
* that never gets to it, a row reaped in memory only and re-folded every open)
|
||||
* never drains, therefore never writes a mark, therefore re-reads its WHOLE
|
||||
* fact log on every single open. The bound was absent from exactly the brains
|
||||
* whose fold is expensive: a silent scaling defect.
|
||||
*
|
||||
* The cure is a CHECKPOINT of the pending set —
|
||||
* `_system/pending_embeds_checkpoint.json` = `{ generation, pending, writtenAt }`,
|
||||
* meaning "as of durable generation G the pending set was exactly this list".
|
||||
* Open seeds the set from `pending` and scans only from `G + 1`, so the fold is
|
||||
* O(facts since G) whether or not the set ever drains.
|
||||
*
|
||||
* What this suite pins:
|
||||
* 1. A brain with one permanently-stuck pending id, closed cleanly and
|
||||
* reopened, scans ONLY the facts after the checkpoint — asserted from the
|
||||
* fold's own accounting, never a clock. The same fixture pins the DEFECT:
|
||||
* no low-water mark exists on that brain, because it never drained.
|
||||
* 2. A crash matrix in a REAL child process (SIGKILL, no close), for kills
|
||||
* before a checkpoint write, after one with embeds landed and flushed
|
||||
* after it, and after one with an UN-FLUSHED tail at the moment of death.
|
||||
* The invariant in every row is differential: the checkpoint-bounded fold
|
||||
* the reopened brain actually ran ≡ a full fold from generation 1 over the
|
||||
* same recovered log.
|
||||
* 3. A torn checkpoint falls back — loudly (the adapter's torn-record gauge
|
||||
* plus the fold's own narration of which bound applied) and correctly.
|
||||
* 4. The existing low-water pins keep passing unchanged
|
||||
* (`pending-embed-low-water.test.ts`): the mark is still written and is
|
||||
* still read, now as the FALLBACK bound beneath the checkpoint.
|
||||
*
|
||||
* The crash-recovery contract is untouched: the fold runs on the open's
|
||||
* foreground, so a reopened brain has its markers re-armed when open() returns.
|
||||
*/
|
||||
import { describe, it, expect, afterEach } from 'vitest'
|
||||
import { mkdtempSync, rmSync, existsSync, readFileSync, writeFileSync } from 'node:fs'
|
||||
import { spawn } from 'node:child_process'
|
||||
import { gunzipSync } from 'node:zlib'
|
||||
import { tmpdir } from 'node:os'
|
||||
import { join } from 'node:path'
|
||||
import { Brainy } from '../../src/brainy.js'
|
||||
import { NounType } from '../../src/types/graphTypes.js'
|
||||
import { getTornRecordGauge } from '../../src/storage/tornRecordError.js'
|
||||
|
||||
const CHECKPOINT_PATH = '_system/pending_embeds_checkpoint.json'
|
||||
const LOWWATER_PATH = '_system/pending_embeds_lowwater.json'
|
||||
const REPO_ROOT = process.cwd()
|
||||
const TSX = join(REPO_ROOT, 'node_modules', '.bin', 'tsx')
|
||||
|
||||
/** The fold's own accounting for the most recent open. */
|
||||
interface FoldReport {
|
||||
bound: 'checkpoint' | 'low-water' | 'genesis'
|
||||
fromGeneration: number
|
||||
factsScanned: number
|
||||
seeded: number
|
||||
pending: number
|
||||
}
|
||||
|
||||
const roots: string[] = []
|
||||
const liveBrains: Brainy<any>[] = []
|
||||
|
||||
function dir(): string {
|
||||
const d = mkdtempSync(join(tmpdir(), 'brainy-embed-ckpt-'))
|
||||
roots.push(d)
|
||||
return d
|
||||
}
|
||||
|
||||
async function open(root: string, opts?: { blockWorker?: boolean }): Promise<Brainy<any>> {
|
||||
const brain = new Brainy<any>({
|
||||
requireSubtype: false,
|
||||
storage: { type: 'filesystem', path: root }
|
||||
})
|
||||
// Blocking the worker BEFORE init() is how a "permanently stuck" pending id
|
||||
// is built deterministically: the state under test is "an id the fold keeps
|
||||
// re-arming and nothing ever disarms", and its production causes (a failing
|
||||
// embedder, a wedged model, a data-less row) all reduce to exactly that.
|
||||
if (opts?.blockWorker) (brain as unknown as { kickEmbedWorker: () => void }).kickEmbedWorker = () => {}
|
||||
await brain.init()
|
||||
liveBrains.push(brain)
|
||||
return brain
|
||||
}
|
||||
|
||||
function foldReport(brain: Brainy<any>): FoldReport {
|
||||
const report = (brain as unknown as { _pendingEmbedFoldReport: FoldReport | null })
|
||||
._pendingEmbedFoldReport
|
||||
if (report === null) throw new Error('the open ran no pending-embed fold')
|
||||
return report
|
||||
}
|
||||
|
||||
function pendingIds(brain: Brainy<any>): string[] {
|
||||
return [
|
||||
...(brain as unknown as { _pendingEmbedIds: Set<string> })._pendingEmbedIds
|
||||
].sort()
|
||||
}
|
||||
|
||||
/** Read an artifact straight off disk (the adapter gzips raw objects). */
|
||||
function readArtifact(root: string, path: string): Record<string, unknown> | null {
|
||||
const plain = join(root, ...path.split('/'))
|
||||
const gz = `${plain}.gz`
|
||||
if (existsSync(gz)) return JSON.parse(gunzipSync(readFileSync(gz)).toString('utf-8'))
|
||||
if (existsSync(plain)) return JSON.parse(readFileSync(plain, 'utf-8'))
|
||||
return null
|
||||
}
|
||||
|
||||
/** The on-disk path the adapter actually used for an artifact. */
|
||||
function artifactPath(root: string, path: string): string | null {
|
||||
const plain = join(root, ...path.split('/'))
|
||||
const gz = `${plain}.gz`
|
||||
if (existsSync(gz)) return gz
|
||||
if (existsSync(plain)) return plain
|
||||
return null
|
||||
}
|
||||
|
||||
/**
|
||||
* THE DIFFERENTIAL ORACLE: fold the log from generation 1 with exactly the
|
||||
* engine's own rules. This is what the bounded fold must agree with, and its
|
||||
* fact count is what the unbounded fold used to read at every open.
|
||||
*/
|
||||
async function fullFold(brain: Brainy<any>): Promise<{ ids: string[]; facts: number }> {
|
||||
const log = (
|
||||
brain as unknown as { generationStore: { getFactLog(): any } }
|
||||
).generationStore.getFactLog()
|
||||
const pending = new Set<string>()
|
||||
let facts = 0
|
||||
const scan = log.scanFacts({ fromGeneration: 1 })
|
||||
for await (const batch of scan.batches()) {
|
||||
for (const fact of batch.facts) {
|
||||
facts++
|
||||
for (const record of fact.records ?? []) {
|
||||
if (record.type === 'embed.pending') pending.add(record.id)
|
||||
else if (record.type === 'embed.landed') pending.delete(record.id)
|
||||
}
|
||||
for (const op of fact.ops) {
|
||||
if (op.kind === 'noun' && op.record === null) pending.delete(op.id)
|
||||
}
|
||||
}
|
||||
}
|
||||
return { ids: [...pending].sort(), facts }
|
||||
}
|
||||
|
||||
/** Capture every console.warn/error line emitted while `fn` runs. */
|
||||
async function captureConsole<T>(fn: () => Promise<T>): Promise<{ result: T; lines: string[] }> {
|
||||
const lines: string[] = []
|
||||
const origWarn = console.warn
|
||||
const origError = console.error
|
||||
const sink = (...args: unknown[]) => {
|
||||
lines.push(args.map((a) => String(a)).join(' '))
|
||||
}
|
||||
console.warn = sink as typeof console.warn
|
||||
console.error = sink as typeof console.error
|
||||
try {
|
||||
const result = await fn()
|
||||
return { result, lines }
|
||||
} finally {
|
||||
console.warn = origWarn
|
||||
console.error = origError
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Run a child process that arranges a store and then waits forever, so the
|
||||
* parent can SIGKILL it. A real process death is the only honest way to pin
|
||||
* "no close ran, no shutdown hook ran, RAM is gone".
|
||||
*
|
||||
* `detached` puts the child in its own process GROUP: tsx runs the script in a
|
||||
* grandchild, and only a group-wide signal reaches the process holding the
|
||||
* writer lock.
|
||||
*/
|
||||
function spawnArranger(root: string, body: string): Promise<{
|
||||
child: ReturnType<typeof spawn>
|
||||
output: () => string
|
||||
}> {
|
||||
const scriptPath = join(root, 'arrange.mts')
|
||||
writeFileSync(scriptPath, body)
|
||||
const child = spawn(TSX, [scriptPath], {
|
||||
cwd: REPO_ROOT,
|
||||
stdio: ['ignore', 'pipe', 'pipe'],
|
||||
detached: true
|
||||
})
|
||||
let out = ''
|
||||
child.stdout!.on('data', (d) => { out += String(d) })
|
||||
child.stderr!.on('data', (d) => { out += String(d) })
|
||||
return new Promise((resolvePromise, rejectPromise) => {
|
||||
const timer = setTimeout(
|
||||
() => rejectPromise(new Error(`arranger never became READY:\n${out}`)),
|
||||
180_000
|
||||
)
|
||||
child.stdout!.on('data', () => {
|
||||
if (out.includes('READY')) {
|
||||
clearTimeout(timer)
|
||||
resolvePromise({ child, output: () => out })
|
||||
}
|
||||
})
|
||||
child.on('exit', (code) => {
|
||||
clearTimeout(timer)
|
||||
if (!out.includes('READY')) rejectPromise(new Error(`arranger exited ${code}:\n${out}`))
|
||||
})
|
||||
})
|
||||
}
|
||||
|
||||
/** Parse the `IDS:{...}` line an arranger prints — supplied ids are normalised
|
||||
* to canonical uuids, and the markers, checkpoint and fold all speak those. */
|
||||
function childIds(output: string): Record<string, string> {
|
||||
const line = output.split('\n').find((l) => l.startsWith('IDS:'))
|
||||
if (!line) throw new Error(`arranger printed no IDS line:\n${output}`)
|
||||
return JSON.parse(line.slice('IDS:'.length))
|
||||
}
|
||||
|
||||
/** SIGKILL the whole group and wait for the grandchild's death to settle. */
|
||||
async function sigkill(child: ReturnType<typeof spawn>): Promise<void> {
|
||||
process.kill(-(child.pid as number), 'SIGKILL')
|
||||
await new Promise<void>((r) => child.on('exit', () => r()))
|
||||
await new Promise<void>((r) => setTimeout(r, 500))
|
||||
}
|
||||
|
||||
/** The preamble every arranger child shares. */
|
||||
function childPreamble(root: string): string {
|
||||
return `
|
||||
import { Brainy } from ${JSON.stringify(join(REPO_ROOT, 'src', 'brainy.ts'))}
|
||||
const ROOT = ${JSON.stringify(root)}
|
||||
const brain = new Brainy<any>({ requireSubtype: false, storage: { type: 'filesystem', path: ROOT } })
|
||||
const block = () => { (brain as any).kickEmbedWorker = () => {} }
|
||||
const settleCheckpoint = async () => {
|
||||
// The cadence write is fire-and-forget; wait for the single flight.
|
||||
for (let i = 0; i < 200; i++) {
|
||||
if (!(brain as any)._pendingEmbedCheckpointFlight) break
|
||||
await (brain as any)._pendingEmbedCheckpointFlight.catch(() => {})
|
||||
}
|
||||
}
|
||||
`
|
||||
}
|
||||
|
||||
afterEach(async () => {
|
||||
for (const brain of liveBrains.splice(0)) {
|
||||
try { await brain.close() } catch { /* already closed / crashed — teardown only */ }
|
||||
}
|
||||
for (const d of roots.splice(0)) rmSync(d, { recursive: true, force: true })
|
||||
})
|
||||
|
||||
// ===========================================================================
|
||||
// 1. The stuck-id brain — the defect, and the bound that now engages on it
|
||||
// ===========================================================================
|
||||
|
||||
describe('pending-embed checkpoint — a brain whose pending set never drains', () => {
|
||||
it('a permanently-stuck pending id: the reopen scans only the facts after the checkpoint', async () => {
|
||||
const root = dir()
|
||||
const first = await open(root, { blockWorker: true })
|
||||
// add() returns the CANONICAL id (supplied ids are normalised), and that is
|
||||
// the id the markers, the checkpoint and the fold all speak.
|
||||
const stuck = await first.add({
|
||||
id: 'stuck',
|
||||
data: 'a deferred row whose embed never lands',
|
||||
type: NounType.Thing,
|
||||
deferEmbedding: true
|
||||
})
|
||||
expect(first.pendingEmbedCount()).toBe(1)
|
||||
// Ordinary traffic after it — every one of these is a fact the unbounded
|
||||
// fold had to re-read at every open, forever, because of that one id.
|
||||
for (let i = 0; i < 12; i++) {
|
||||
await first.add({ id: `row-${i}`, data: `row ${i}`, type: NounType.Thing })
|
||||
}
|
||||
await first.close()
|
||||
liveBrains.splice(liveBrains.indexOf(first), 1)
|
||||
|
||||
// THE DEFECT, PINNED: the pending set never drained, so the old bound was
|
||||
// never written — nothing on this brain could have shortened its fold.
|
||||
expect(readArtifact(root, LOWWATER_PATH)).toBeNull()
|
||||
// The checkpoint IS written at the clean close, set non-empty and all.
|
||||
const checkpoint = readArtifact(root, CHECKPOINT_PATH) as {
|
||||
generation: number
|
||||
pending: string[]
|
||||
} | null
|
||||
expect(checkpoint).not.toBeNull()
|
||||
expect(checkpoint!.generation).toBeGreaterThan(0)
|
||||
expect(checkpoint!.pending).toEqual([stuck])
|
||||
|
||||
const second = await open(root, { blockWorker: true })
|
||||
const report = foldReport(second)
|
||||
// THE FIX, from the fold's own counter — not the clock.
|
||||
expect(report.bound).toBe('checkpoint')
|
||||
expect(report.fromGeneration).toBe(checkpoint!.generation + 1)
|
||||
expect(report.factsScanned).toBe(0)
|
||||
expect(report.seeded).toBe(1)
|
||||
// The crash-recovery contract is intact: the marker is re-armed by open().
|
||||
expect(pendingIds(second)).toEqual([stuck])
|
||||
expect(second.pendingEmbedCount()).toBe(1)
|
||||
|
||||
// The differential: the bounded answer is the full-fold answer, and the
|
||||
// full fold is what the previous bound would have had to read.
|
||||
const full = await fullFold(second)
|
||||
expect(full.ids).toEqual([stuck])
|
||||
expect(full.facts).toBeGreaterThanOrEqual(13)
|
||||
expect(report.factsScanned).toBeLessThan(full.facts)
|
||||
}, 180_000)
|
||||
|
||||
it('the bound stays O(delta) across repeated opens while the id is still stuck', async () => {
|
||||
const root = dir()
|
||||
const first = await open(root, { blockWorker: true })
|
||||
const stuck = await first.add({
|
||||
id: 'stuck',
|
||||
data: 'never lands',
|
||||
type: NounType.Thing,
|
||||
deferEmbedding: true
|
||||
})
|
||||
for (let i = 0; i < 6; i++) {
|
||||
await first.add({ id: `a-${i}`, data: `a ${i}`, type: NounType.Thing })
|
||||
}
|
||||
await first.close()
|
||||
liveBrains.splice(liveBrains.indexOf(first), 1)
|
||||
|
||||
const second = await open(root, { blockWorker: true })
|
||||
expect(foldReport(second).factsScanned).toBe(0)
|
||||
// More history under the same stuck id.
|
||||
for (let i = 0; i < 9; i++) {
|
||||
await second.add({ id: `b-${i}`, data: `b ${i}`, type: NounType.Thing })
|
||||
}
|
||||
await second.close()
|
||||
liveBrains.splice(liveBrains.indexOf(second), 1)
|
||||
|
||||
const third = await open(root, { blockWorker: true })
|
||||
const report = foldReport(third)
|
||||
const full = await fullFold(third)
|
||||
expect(report.bound).toBe('checkpoint')
|
||||
expect(report.factsScanned).toBe(0)
|
||||
// The unbounded fold grew with the store; the bounded one did not.
|
||||
expect(full.facts).toBeGreaterThanOrEqual(16)
|
||||
expect(pendingIds(third)).toEqual([stuck])
|
||||
expect(full.ids).toEqual([stuck])
|
||||
}, 180_000)
|
||||
})
|
||||
|
||||
// ===========================================================================
|
||||
// 2. Torn checkpoint — falls back, loudly, correctly
|
||||
// ===========================================================================
|
||||
|
||||
describe('pending-embed checkpoint — a torn checkpoint never shortens the fold', () => {
|
||||
it('an undecodable checkpoint file degrades to the next bound, loudly, with the right pending set', async () => {
|
||||
const root = dir()
|
||||
const first = await open(root, { blockWorker: true })
|
||||
const stuck = await first.add({
|
||||
id: 'stuck',
|
||||
data: 'never lands',
|
||||
type: NounType.Thing,
|
||||
deferEmbedding: true
|
||||
})
|
||||
for (let i = 0; i < 5; i++) {
|
||||
await first.add({ id: `row-${i}`, data: `row ${i}`, type: NounType.Thing })
|
||||
}
|
||||
await first.close()
|
||||
liveBrains.splice(liveBrains.indexOf(first), 1)
|
||||
|
||||
const onDisk = artifactPath(root, CHECKPOINT_PATH)
|
||||
expect(onDisk).not.toBeNull()
|
||||
// Tear it: bytes that are neither valid gzip nor valid JSON. A torn file
|
||||
// must THROW on read — never parse into a partial `pending` list.
|
||||
writeFileSync(onDisk!, 'not a checkpoint at all {{{')
|
||||
|
||||
const before = getTornRecordGauge().count
|
||||
const { result: second, lines } = await captureConsole(async () =>
|
||||
open(root, { blockWorker: true })
|
||||
)
|
||||
const report = foldReport(second)
|
||||
// Fell back — never to a shorter bound, and never silently.
|
||||
expect(report.bound).not.toBe('checkpoint')
|
||||
expect(report.seeded).toBe(0)
|
||||
expect(report.fromGeneration).toBe(1) // no mark either: this brain never drained
|
||||
// LOUD, two ways: the adapter's torn-record gauge and its production error…
|
||||
expect(getTornRecordGauge().count).toBeGreaterThan(before)
|
||||
expect(getTornRecordGauge().lastPath).toContain('pending_embeds_checkpoint')
|
||||
expect(lines.some((l) => /TORN RECORD/.test(l))).toBe(true)
|
||||
// …and the fold's own narration of which bound it actually used.
|
||||
expect(lines.some((l) => /pending-embed fold: genesis bound/.test(l))).toBe(true)
|
||||
|
||||
// CORRECT: the marker is still recovered, from the log itself.
|
||||
expect(pendingIds(second)).toEqual([stuck])
|
||||
const full = await fullFold(second)
|
||||
expect(full.ids).toEqual([stuck])
|
||||
expect(report.factsScanned).toBe(full.facts)
|
||||
}, 180_000)
|
||||
|
||||
it('a well-formed but shape-invalid checkpoint is refused whole, never partially trusted', async () => {
|
||||
const root = dir()
|
||||
const first = await open(root, { blockWorker: true })
|
||||
const stuck = await first.add({
|
||||
id: 'stuck',
|
||||
data: 'never lands',
|
||||
type: NounType.Thing,
|
||||
deferEmbedding: true
|
||||
})
|
||||
await first.add({ id: 'other', data: 'ordinary row', type: NounType.Thing })
|
||||
await first.close()
|
||||
liveBrains.splice(liveBrains.indexOf(first), 1)
|
||||
|
||||
// A checkpoint with a plausible generation but a `pending` that is not a
|
||||
// list of ids: trusting the generation alone would bound the scan behind a
|
||||
// set that was never recovered — the exact shape that loses a vector.
|
||||
const onDisk = artifactPath(root, CHECKPOINT_PATH)!
|
||||
const good = readArtifact(root, CHECKPOINT_PATH) as { generation: number }
|
||||
rmSync(onDisk)
|
||||
writeFileSync(
|
||||
join(root, '_system', 'pending_embeds_checkpoint.json'),
|
||||
JSON.stringify({ generation: good.generation, pending: { stuck: true }, writtenAt: 1 })
|
||||
)
|
||||
|
||||
const { result: second, lines } = await captureConsole(async () =>
|
||||
open(root, { blockWorker: true })
|
||||
)
|
||||
expect(lines.some((l) => /pending-embed checkpoint REFUSED/.test(l))).toBe(true)
|
||||
const report = foldReport(second)
|
||||
expect(report.bound).not.toBe('checkpoint')
|
||||
expect(report.seeded).toBe(0)
|
||||
expect(pendingIds(second)).toEqual([stuck])
|
||||
}, 180_000)
|
||||
})
|
||||
|
||||
// ===========================================================================
|
||||
// 3. The crash matrix — real processes, real SIGKILL, differential invariant
|
||||
// ===========================================================================
|
||||
|
||||
describe('pending-embed checkpoint — crash matrix (real child process, SIGKILL)', () => {
|
||||
/**
|
||||
* The invariant every row shares: whatever the reopened brain's fold did with
|
||||
* whatever bound survived the crash, its pending set must equal the truth a
|
||||
* full fold from generation 1 derives from the SAME recovered log.
|
||||
*/
|
||||
async function assertDifferentialAfterCrash(root: string): Promise<{
|
||||
report: FoldReport
|
||||
full: { ids: string[]; facts: number }
|
||||
pending: string[]
|
||||
}> {
|
||||
const reopened = await open(root, { blockWorker: true })
|
||||
const report = foldReport(reopened)
|
||||
const full = await fullFold(reopened)
|
||||
const pending = pendingIds(reopened)
|
||||
expect(pending).toEqual(full.ids)
|
||||
return { report, full, pending }
|
||||
}
|
||||
|
||||
it('killed BEFORE any checkpoint was written — falls back and recovers the marker from the log', async () => {
|
||||
const root = dir()
|
||||
const { child, output } = await spawnArranger(
|
||||
root,
|
||||
`${childPreamble(root)}
|
||||
block()
|
||||
await brain.init()
|
||||
await brain.add({ id: 'landed-row', data: 'an ordinary row', type: 'thing' })
|
||||
const stuck = await brain.add({ id: 'stuck-1', data: 'deferred, never lands', type: 'thing', deferEmbedding: true })
|
||||
await brain.flush()
|
||||
console.log('IDS:' + JSON.stringify({ stuck }))
|
||||
console.log('READY')
|
||||
setInterval(() => {}, 1000)
|
||||
`
|
||||
)
|
||||
const ids = childIds(output())
|
||||
// One enqueue is well under the cadence and the set never drained, so no
|
||||
// checkpoint exists — this is the pre-checkpoint crash.
|
||||
expect(readArtifact(root, CHECKPOINT_PATH)).toBeNull()
|
||||
await sigkill(child)
|
||||
|
||||
const { report, pending } = await assertDifferentialAfterCrash(root)
|
||||
expect(report.bound).toBe('genesis')
|
||||
expect(pending).toEqual([ids.stuck])
|
||||
}, 300_000)
|
||||
|
||||
it('killed AFTER a checkpoint, with an embed landed and flushed after it — the post-checkpoint facts carry the disarm', async () => {
|
||||
const root = dir()
|
||||
const { child, output } = await spawnArranger(
|
||||
root,
|
||||
`${childPreamble(root)}
|
||||
await brain.init()
|
||||
// Land one deferred embed: the drain arms the checkpoint debt.
|
||||
await brain.add({ id: 'seed', data: 'lands first', type: 'thing', deferEmbedding: true })
|
||||
await brain.awaitPendingEmbeds()
|
||||
await brain.flush()
|
||||
// A second deferred write pays the debt (the head is at the manifest now),
|
||||
// then LANDS — its embed.landed rides a fact ABOVE the checkpoint.
|
||||
const landsAfter = await brain.add({ id: 'lands-after', data: 'lands after the checkpoint', type: 'thing', deferEmbedding: true })
|
||||
await settleCheckpoint()
|
||||
await brain.awaitPendingEmbeds()
|
||||
// …and one that never will.
|
||||
block()
|
||||
const stuck = await brain.add({ id: 'stuck-1', data: 'deferred, never lands', type: 'thing', deferEmbedding: true })
|
||||
await brain.add({ id: 'plain', data: 'more history', type: 'thing' })
|
||||
await brain.flush()
|
||||
console.log('IDS:' + JSON.stringify({ stuck, landsAfter }))
|
||||
console.log('READY')
|
||||
setInterval(() => {}, 1000)
|
||||
`
|
||||
)
|
||||
const ids = childIds(output())
|
||||
const checkpoint = readArtifact(root, CHECKPOINT_PATH) as {
|
||||
generation: number
|
||||
pending: string[]
|
||||
} | null
|
||||
expect(checkpoint).not.toBeNull()
|
||||
await sigkill(child)
|
||||
|
||||
const { report, full, pending } = await assertDifferentialAfterCrash(root)
|
||||
expect(report.bound).toBe('checkpoint')
|
||||
expect(report.fromGeneration).toBe(checkpoint!.generation + 1)
|
||||
// The bound really bounded: fewer facts than the whole log.
|
||||
expect(report.factsScanned).toBeLessThan(full.facts)
|
||||
// A landed embed above the checkpoint is disarmed by the scan, not lost;
|
||||
// the stuck one is re-armed.
|
||||
expect(pending).toEqual([ids.stuck])
|
||||
expect(pending).not.toContain(ids.landsAfter)
|
||||
}, 300_000)
|
||||
|
||||
it('killed AFTER a checkpoint with an UN-FLUSHED tail — truncated facts and the bounded fold still agree', async () => {
|
||||
const root = dir()
|
||||
const { child } = await spawnArranger(
|
||||
root,
|
||||
`${childPreamble(root)}
|
||||
await brain.init()
|
||||
await brain.add({ id: 'seed', data: 'lands first', type: 'thing', deferEmbedding: true })
|
||||
await brain.awaitPendingEmbeds()
|
||||
await brain.flush()
|
||||
await brain.add({ id: 'lands-after', data: 'lands after the checkpoint', type: 'thing', deferEmbedding: true })
|
||||
await settleCheckpoint()
|
||||
await brain.awaitPendingEmbeds()
|
||||
await brain.flush()
|
||||
// Now write PAST the manifest and never flush: these facts are the tail a
|
||||
// crash truncates. Whatever survives, the two folds must agree on it.
|
||||
block()
|
||||
await brain.add({ id: 'stuck-tail', data: 'deferred, never lands', type: 'thing', deferEmbedding: true })
|
||||
await brain.add({ id: 'plain-tail', data: 'unflushed history', type: 'thing' })
|
||||
console.log('READY')
|
||||
setInterval(() => {}, 1000)
|
||||
`
|
||||
)
|
||||
const checkpoint = readArtifact(root, CHECKPOINT_PATH) as { generation: number } | null
|
||||
expect(checkpoint).not.toBeNull()
|
||||
await sigkill(child)
|
||||
|
||||
const { report } = await assertDifferentialAfterCrash(root)
|
||||
// The checkpoint's generation is at or below the manifest by construction,
|
||||
// so it survived the truncation and still bounds the fold.
|
||||
expect(report.bound).toBe('checkpoint')
|
||||
expect(report.fromGeneration).toBe(checkpoint!.generation + 1)
|
||||
}, 300_000)
|
||||
})
|
||||
141
tests/integration/pending-embed-low-water.test.ts
Normal file
141
tests/integration/pending-embed-low-water.test.ts
Normal file
|
|
@ -0,0 +1,141 @@
|
|||
/**
|
||||
* @module tests/integration/pending-embed-low-water
|
||||
* @description The pending-embed recovery fold is bounded and background (10.4.9).
|
||||
*
|
||||
* The fold used to scan the generation log from generation 1 at EVERY open,
|
||||
* on the open's foreground — O(whole history) per open on long-lived brains.
|
||||
* Now: an advisory low-water mark (`_system/pending_embeds_lowwater.json`)
|
||||
* records the committed generation whenever the pending set drains to empty,
|
||||
* recovery scans from `mark + 1` on the open's foreground — the crash-recovery
|
||||
* contract keeps markers re-armed when open() returns. The mark is advisory: stale-low costs a longer scan, never a
|
||||
* marker — a pending embed enqueued before a crash is still recovered.
|
||||
*/
|
||||
import { describe, it, expect, afterEach, vi } from 'vitest'
|
||||
import { mkdtempSync, rmSync } from 'node:fs'
|
||||
import { tmpdir } from 'node:os'
|
||||
import { join } from 'node:path'
|
||||
import { Brainy } from '../../src/brainy'
|
||||
import { NounType } from '../../src/types/graphTypes'
|
||||
|
||||
const LOWWATER_PATH = '_system/pending_embeds_lowwater.json'
|
||||
|
||||
describe('pending-embed recovery: bounded by the low-water mark', () => {
|
||||
const roots: string[] = []
|
||||
const dir = (): string => {
|
||||
const d = mkdtempSync(join(tmpdir(), 'brainy-lowwater-'))
|
||||
roots.push(d)
|
||||
return d
|
||||
}
|
||||
const open = async (root: string): Promise<Brainy<any>> => {
|
||||
const brain = new Brainy<any>({
|
||||
requireSubtype: false,
|
||||
storage: { type: 'filesystem', path: root }
|
||||
})
|
||||
await brain.init()
|
||||
return brain
|
||||
}
|
||||
|
||||
afterEach(() => {
|
||||
for (const d of roots.splice(0)) rmSync(d, { recursive: true, force: true })
|
||||
})
|
||||
|
||||
it('drain-to-empty writes the mark, and the next open scans from mark + 1', async () => {
|
||||
const root = dir()
|
||||
const brain = await open(root)
|
||||
// Hold the worker so the pending state is observable, then release it.
|
||||
const realKick = (brain as any).kickEmbedWorker.bind(brain)
|
||||
;(brain as any).kickEmbedWorker = () => {}
|
||||
await brain.add({
|
||||
id: 'row-1',
|
||||
data: 'the first deferred row',
|
||||
type: NounType.Thing,
|
||||
deferEmbedding: true
|
||||
})
|
||||
expect(brain.pendingEmbedCount()).toBeGreaterThan(0)
|
||||
;(brain as any).kickEmbedWorker = realKick
|
||||
await brain.awaitPendingEmbeds()
|
||||
// The drain wrote the advisory mark (fire-and-forget: settle the microtask).
|
||||
await new Promise((r) => setTimeout(r, 50))
|
||||
const mark = (await (brain as any).storage.readRawObject(LOWWATER_PATH)) as {
|
||||
generation: number
|
||||
} | null
|
||||
expect(mark).not.toBeNull()
|
||||
expect(mark!.generation).toBeGreaterThan(0)
|
||||
await brain.close()
|
||||
|
||||
const brain2 = await open(root)
|
||||
const log = (brain2 as any).generationStore.getFactLog()
|
||||
const scanSpy = vi.spyOn(log, 'scanFacts')
|
||||
try {
|
||||
await (brain2 as any).recoverPendingEmbedsFromLog()
|
||||
expect(scanSpy).toHaveBeenCalledTimes(1)
|
||||
const opts = scanSpy.mock.calls[0][0] as { fromGeneration?: number }
|
||||
expect(opts.fromGeneration).toBeGreaterThanOrEqual(mark!.generation + 1)
|
||||
} finally {
|
||||
scanSpy.mockRestore()
|
||||
await brain2.close()
|
||||
}
|
||||
})
|
||||
|
||||
it('a pending embed enqueued after the mark survives an unclean stop', async () => {
|
||||
const root = dir()
|
||||
const brain = await open(root)
|
||||
await brain.add({ id: 'settled', data: 'lands before the mark', type: NounType.Thing })
|
||||
await brain.awaitPendingEmbeds()
|
||||
await new Promise((r) => setTimeout(r, 50))
|
||||
|
||||
// A deferred write whose embed never lands: block the worker, then drop
|
||||
// the instance without close() — the unclean-stop shape.
|
||||
;(brain as any).kickEmbedWorker = () => {}
|
||||
await brain.add({
|
||||
id: 'orphan',
|
||||
data: 'enqueued then abandoned',
|
||||
type: NounType.Thing,
|
||||
deferEmbedding: true
|
||||
})
|
||||
expect(brain.pendingEmbedCount()).toBeGreaterThan(0)
|
||||
// No close(): simulate the crash by releasing only the writer lock so the
|
||||
// next open can proceed.
|
||||
await (brain as any).storage.releaseWriterLock()
|
||||
|
||||
const brain2 = await open(root)
|
||||
expect(brain2.pendingEmbedCount()).toBeGreaterThan(0)
|
||||
await brain2.awaitPendingEmbeds()
|
||||
expect(brain2.pendingEmbedCount()).toBe(0)
|
||||
await brain2.close()
|
||||
// Reap the crashed instance: its fence is gone, so close() fails loudly —
|
||||
// swallow that here; the point is clearing its watchers and registry entry.
|
||||
await brain.close().catch(() => undefined)
|
||||
})
|
||||
|
||||
it('a reopened brain has its pending set settled when open() returns', async () => {
|
||||
const root = dir()
|
||||
const brain = await open(root)
|
||||
await brain.add({ id: 'a-row', data: 'some data', type: NounType.Thing })
|
||||
await brain.awaitPendingEmbeds()
|
||||
await brain.close()
|
||||
|
||||
const brain2 = await open(root)
|
||||
// The crash-recovery contract: markers are re-armed by open itself —
|
||||
// no latch, no background race. (Here the drain landed, so zero.)
|
||||
expect(brain2.pendingEmbedCount()).toBe(0)
|
||||
await brain2.close()
|
||||
})
|
||||
|
||||
it('a clean close with an empty set writes the mark even if no drain happened', async () => {
|
||||
const root = dir()
|
||||
const brain = await open(root)
|
||||
await brain.add({ id: 'r1', data: 'row one', type: NounType.Thing })
|
||||
await brain.awaitPendingEmbeds()
|
||||
await brain.close()
|
||||
// Read the mark back through the storage door (the adapter owns the
|
||||
// on-disk encoding), on a fresh instance.
|
||||
const brain2 = await open(root)
|
||||
const mark = (await (brain2 as any).storage.readRawObject(LOWWATER_PATH)) as {
|
||||
generation: number
|
||||
} | null
|
||||
expect(mark).not.toBeNull()
|
||||
expect(mark!.generation).toBeGreaterThan(0)
|
||||
await brain2.close()
|
||||
})
|
||||
})
|
||||
250
tests/integration/readonly-close-no-marker.test.ts
Normal file
250
tests/integration/readonly-close-no-marker.test.ts
Normal file
|
|
@ -0,0 +1,250 @@
|
|||
/**
|
||||
* @module tests/integration/readonly-close-no-marker
|
||||
* @description A READ-ONLY BRAIN WRITES NO CLEAN-SHUTDOWN EVIDENCE.
|
||||
*
|
||||
* `_system/clean-shutdown.json` is the WRITER's own word about the writer's
|
||||
* own process: "everything above this line, from THIS session, is durable."
|
||||
* Two call sites treated a reader exactly like a writer:
|
||||
*
|
||||
* 1. `Brainy#closeDurableSteps()` called `generationStore.close()`
|
||||
* unconditionally — a reader's close re-stamped the marker at the
|
||||
* generation the reader merely OBSERVED, never committed.
|
||||
* 2. `GenerationStore#open()` consumed (deleted) the marker on every open,
|
||||
* reader or writer alike, so a reader that never got to a matching
|
||||
* close left the store looking crashed to the next writer.
|
||||
*
|
||||
* Both are fixed by making a read-only brain leave `_system/` exactly as it
|
||||
* found it — at open AND at close. Pinned here:
|
||||
*
|
||||
* 1. `_system/` is byte-for-byte identical (file set + contents) before and
|
||||
* after a reader opens a cleanly-closed store, reads it, and closes.
|
||||
* 2. After the reader's close, the next WRITER open adopts the marker as
|
||||
* clean — no recovery fold narrates.
|
||||
* 3. A reader creates no file under `_system/` merely by opening (before it
|
||||
* ever closes).
|
||||
* 4. A reader that opens and is then abandoned (crash-style, no close) does
|
||||
* not force the next writer to pay a recovery fold — the concrete harm
|
||||
* the fix closes.
|
||||
*/
|
||||
|
||||
import { describe, it, expect, beforeEach, afterEach } from 'vitest'
|
||||
import { mkdtempSync, rmSync, readdirSync, readFileSync, statSync } from 'node:fs'
|
||||
import { createHash } from 'node:crypto'
|
||||
import { tmpdir } from 'node:os'
|
||||
import { join } from 'node:path'
|
||||
import { Brainy } from '../../src/brainy.js'
|
||||
import { NounType } from '../../src/types/graphTypes.js'
|
||||
import { abandonAsCrashed } from '../helpers/durabilityKillMatrix.js'
|
||||
|
||||
function makeTempDir(): string {
|
||||
return mkdtempSync(join(tmpdir(), 'brainy-readonly-close-'))
|
||||
}
|
||||
|
||||
/** Recursively hash every regular file under `dir`, keyed by its path relative to `dir`. */
|
||||
function snapshotDir(dir: string): Map<string, string> {
|
||||
const out = new Map<string, string>()
|
||||
const walk = (rel: string): void => {
|
||||
const abs = rel ? join(dir, rel) : dir
|
||||
let entries: string[]
|
||||
try {
|
||||
entries = readdirSync(abs)
|
||||
} catch {
|
||||
return
|
||||
}
|
||||
for (const name of entries) {
|
||||
const childRel = rel ? join(rel, name) : name
|
||||
const childAbs = join(dir, childRel)
|
||||
const st = statSync(childAbs)
|
||||
if (st.isDirectory()) {
|
||||
walk(childRel)
|
||||
} else if (st.isFile()) {
|
||||
const hash = createHash('sha256').update(readFileSync(childAbs)).digest('hex')
|
||||
out.set(childRel, hash)
|
||||
}
|
||||
}
|
||||
}
|
||||
walk('')
|
||||
return out
|
||||
}
|
||||
|
||||
/** Capture console.warn lines (the narration channel — see `prodLog.narrate`) while `fn` runs. */
|
||||
async function captureWarn<T>(fn: () => Promise<T>): Promise<{ result: T; lines: string[] }> {
|
||||
const lines: string[] = []
|
||||
const orig = console.warn
|
||||
console.warn = ((...args: unknown[]) => {
|
||||
lines.push(args.map((a) => String(a)).join(' '))
|
||||
}) as typeof console.warn
|
||||
try {
|
||||
return { result: await fn(), lines }
|
||||
} finally {
|
||||
console.warn = orig
|
||||
}
|
||||
}
|
||||
|
||||
describe('a read-only brain writes no clean-shutdown evidence', () => {
|
||||
let dir: string
|
||||
let brain: Brainy | null = null
|
||||
|
||||
beforeEach(() => {
|
||||
dir = makeTempDir()
|
||||
})
|
||||
|
||||
afterEach(async () => {
|
||||
if (brain) {
|
||||
try {
|
||||
await brain.close()
|
||||
} catch {
|
||||
/* already closed */
|
||||
}
|
||||
brain = null
|
||||
}
|
||||
try {
|
||||
rmSync(dir, { recursive: true, force: true })
|
||||
} catch {
|
||||
/* ignore */
|
||||
}
|
||||
})
|
||||
|
||||
const systemDir = () => join(dir, '_system')
|
||||
/**
|
||||
* The marker file's actual on-disk name — `clean-shutdown.json` or, under
|
||||
* FileSystemStorage's default gzip compression, `clean-shutdown.json.gz`.
|
||||
* Returns null when absent.
|
||||
*/
|
||||
const findMarkerPath = (): string | null => {
|
||||
let entries: string[]
|
||||
try {
|
||||
entries = readdirSync(systemDir())
|
||||
} catch {
|
||||
return null
|
||||
}
|
||||
const name = entries.find((n) => n.startsWith('clean-shutdown.json'))
|
||||
return name ? join(systemDir(), name) : null
|
||||
}
|
||||
|
||||
it('leaves `_system/`\'s file set and the clean-shutdown marker\'s bytes identical across a reader open → read → close', async () => {
|
||||
// A writer opens, writes, and closes cleanly — the marker lands at
|
||||
// whatever generation the writer actually committed.
|
||||
const writer = new Brainy({ requireSubtype: false, storage: { type: 'filesystem', path: dir } })
|
||||
await writer.init()
|
||||
await writer.add({ data: 'seed entity', type: NounType.Concept })
|
||||
await writer.add({ data: 'second entity', type: NounType.Concept })
|
||||
await writer.flush()
|
||||
await writer.close()
|
||||
|
||||
const markerBeforePath = findMarkerPath()
|
||||
expect(markerBeforePath, 'the writer left a clean-shutdown marker').not.toBeNull()
|
||||
const before = snapshotDir(systemDir())
|
||||
expect(before.size).toBeGreaterThan(0)
|
||||
const markerBeforeHash = before.get(
|
||||
(markerBeforePath as string).slice(systemDir().length + 1)
|
||||
)
|
||||
expect(markerBeforeHash).toBeTruthy()
|
||||
|
||||
// A reader opens the same store, reads, and closes.
|
||||
brain = await Brainy.openReadOnly({ storage: { type: 'filesystem', path: dir } })
|
||||
expect(brain.isReadOnly).toBe(true)
|
||||
await brain.stats()
|
||||
await brain.close()
|
||||
brain = null
|
||||
|
||||
// The FILE SET under `_system/` is unchanged — a reader creates and
|
||||
// removes nothing. (Other files under `_system/` — e.g. the metadata
|
||||
// field registry, which stamps its own `lastUpdated` on every persist —
|
||||
// are a pre-existing, separate concern outside this fix's scope: this
|
||||
// pin is specifically about the generation store's clean-shutdown
|
||||
// evidence, not about every subsystem's close() being a true no-op for
|
||||
// a reader.)
|
||||
const after = snapshotDir(systemDir())
|
||||
expect([...after.keys()].sort()).toEqual([...before.keys()].sort())
|
||||
|
||||
// The MARKER's bytes are byte-for-byte identical — the reader neither
|
||||
// consumed it at open nor re-stamped it at close.
|
||||
const markerAfterPath = findMarkerPath()
|
||||
expect(markerAfterPath, 'the marker must still exist, under the same name').toBe(markerBeforePath)
|
||||
const markerAfterHash = after.get((markerAfterPath as string).slice(systemDir().length + 1))
|
||||
expect(markerAfterHash).toBe(markerBeforeHash)
|
||||
}, 120_000)
|
||||
|
||||
it('creates no file under `_system/` merely by opening read-only', async () => {
|
||||
const writer = new Brainy({ requireSubtype: false, storage: { type: 'filesystem', path: dir } })
|
||||
await writer.init()
|
||||
await writer.add({ data: 'seed entity', type: NounType.Concept })
|
||||
await writer.flush()
|
||||
await writer.close()
|
||||
|
||||
const baselineNames = [...snapshotDir(systemDir()).keys()].sort()
|
||||
expect(baselineNames.length).toBeGreaterThan(0)
|
||||
|
||||
// Open the reader and inspect `_system/` BEFORE it ever closes — open()
|
||||
// alone must create nothing.
|
||||
brain = await Brainy.openReadOnly({ storage: { type: 'filesystem', path: dir } })
|
||||
const whileOpenNames = [...snapshotDir(systemDir()).keys()].sort()
|
||||
expect(whileOpenNames).toEqual(baselineNames)
|
||||
|
||||
await brain.close()
|
||||
brain = null
|
||||
}, 120_000)
|
||||
|
||||
it('a writer reopening after the reader closes adopts the marker — no recovery fold', async () => {
|
||||
const writer1 = new Brainy({ requireSubtype: false, storage: { type: 'filesystem', path: dir } })
|
||||
await writer1.init()
|
||||
await writer1.add({ data: 'seed entity', type: NounType.Concept })
|
||||
await writer1.flush()
|
||||
await writer1.close()
|
||||
|
||||
// A reader opens and closes in between — must not disturb the marker.
|
||||
const reader = await Brainy.openReadOnly({ storage: { type: 'filesystem', path: dir } })
|
||||
await reader.stats()
|
||||
await reader.close()
|
||||
|
||||
// The next writer open must be a clean, no-fold open: no
|
||||
// "log-authority recovery" / "WHOLE-LOG fold" narration line.
|
||||
const { result: writer2, lines } = await captureWarn(async () => {
|
||||
const w = new Brainy({ requireSubtype: false, storage: { type: 'filesystem', path: dir } })
|
||||
await w.init()
|
||||
return w
|
||||
})
|
||||
brain = writer2
|
||||
|
||||
const foldLines = lines.filter((l) => /log-authority recovery|WHOLE-LOG fold|recovery fold/i.test(l))
|
||||
expect(foldLines, `unexpected recovery narration:\n${foldLines.join('\n')}`).toEqual([])
|
||||
|
||||
// And the store is exactly what the first writer left — the seed row is
|
||||
// still there, nothing was rolled back or re-derived.
|
||||
const found = await writer2.find({ where: {} } as any)
|
||||
expect(found.length).toBeGreaterThanOrEqual(1)
|
||||
}, 120_000)
|
||||
|
||||
it('a reader that opens and is then abandoned (never closes) does not force the next writer to fold', async () => {
|
||||
// This is the concrete harm the fix closes: pre-fix, a reader's open()
|
||||
// unconditionally DELETED the marker (consuming it as if it were the
|
||||
// writer). A reader that opened and then died — no close, exactly like
|
||||
// a killed process — left the marker gone, so the actual writer's next
|
||||
// open read the store as crashed and paid a full recovery fold for a
|
||||
// "crash" that was really just a reader that came and went.
|
||||
const writer1 = new Brainy({ requireSubtype: false, storage: { type: 'filesystem', path: dir } })
|
||||
await writer1.init()
|
||||
await writer1.add({ data: 'seed entity', type: NounType.Concept })
|
||||
await writer1.flush()
|
||||
await writer1.close()
|
||||
|
||||
const reader = await Brainy.openReadOnly({ storage: { type: 'filesystem', path: dir } })
|
||||
await reader.stats()
|
||||
// NEVER calls reader.close() — abandon it exactly like a killed process.
|
||||
await abandonAsCrashed(reader)
|
||||
|
||||
const { result: writer2, lines } = await captureWarn(async () => {
|
||||
const w = new Brainy({ requireSubtype: false, storage: { type: 'filesystem', path: dir } })
|
||||
await w.init()
|
||||
return w
|
||||
})
|
||||
brain = writer2
|
||||
|
||||
const foldLines = lines.filter((l) => /log-authority recovery|WHOLE-LOG fold|recovery fold/i.test(l))
|
||||
expect(
|
||||
foldLines,
|
||||
`an abandoned READER forced a recovery fold on the next writer open:\n${foldLines.join('\n')}`
|
||||
).toEqual([])
|
||||
}, 120_000)
|
||||
})
|
||||
89
tests/integration/related-verb-array.test.ts
Normal file
89
tests/integration/related-verb-array.test.ts
Normal file
|
|
@ -0,0 +1,89 @@
|
|||
/**
|
||||
* @module tests/integration/related-verb-array
|
||||
* @description related() honours EVERY verb type in an array (10.4.9).
|
||||
*
|
||||
* The storage fast paths for `sourceId + verbType` and `verbType` collapsed a
|
||||
* verb-type ARRAY to its first element — `related({ from, type: [a, b] })`
|
||||
* silently returned only `a` edges, whichever order the array came in. The
|
||||
* same quiet-loss class as the graph-first paging defect, one seam over.
|
||||
* These pins seed a store where the SECOND requested type's edge must come
|
||||
* back, on every path the collapse lived in.
|
||||
*/
|
||||
import { describe, it, expect, beforeAll, afterAll } from 'vitest'
|
||||
import { Brainy } from '../../src/brainy'
|
||||
import { NounType, VerbType } from '../../src/types/graphTypes'
|
||||
import { v5 } from '../../src/universal/uuid'
|
||||
|
||||
describe('related() with a verb-type array returns every requested type', () => {
|
||||
let brain: Brainy<any>
|
||||
|
||||
beforeAll(async () => {
|
||||
brain = new Brainy({ requireSubtype: false, storage: { type: 'memory' } })
|
||||
await brain.init()
|
||||
for (const id of ['a', 'b', 'c', 'd']) {
|
||||
await brain.add({ id, data: `node ${id}`, type: NounType.Person })
|
||||
}
|
||||
await brain.relate({ from: 'a', to: 'b', type: VerbType.Supports })
|
||||
await brain.relate({ from: 'a', to: 'c', type: VerbType.RelatedTo })
|
||||
await brain.relate({ from: 'a', to: 'd', type: VerbType.Knows })
|
||||
await brain.relate({ from: 'b', to: 'c', type: VerbType.RelatedTo })
|
||||
})
|
||||
|
||||
afterAll(async () => {
|
||||
brain = null as any
|
||||
})
|
||||
|
||||
it('from + type array: the second type\'s edge comes back, both orders', async () => {
|
||||
for (const types of [
|
||||
[VerbType.Supports, VerbType.RelatedTo],
|
||||
[VerbType.RelatedTo, VerbType.Supports]
|
||||
]) {
|
||||
const edges = await brain.related({ from: 'a', type: types })
|
||||
const targets = new Set(edges.map((e) => e.to))
|
||||
expect(targets.has(v5('b')), `types [${types}] missing Supports edge`).toBe(true)
|
||||
expect(targets.has(v5('c')), `types [${types}] missing RelatedTo edge`).toBe(true)
|
||||
expect(targets.has(v5('d'))).toBe(false)
|
||||
expect(edges).toHaveLength(2)
|
||||
}
|
||||
})
|
||||
|
||||
it('a single-element array behaves exactly like the scalar', async () => {
|
||||
const scalar = await brain.related({ from: 'a', type: VerbType.Supports })
|
||||
const array = await brain.related({ from: 'a', type: [VerbType.Supports] })
|
||||
expect(array.map((e) => e.id).sort()).toEqual(scalar.map((e) => e.id).sort())
|
||||
expect(array).toHaveLength(1)
|
||||
})
|
||||
|
||||
it('no duplicate edges when types overlap the same edge set', async () => {
|
||||
const edges = await brain.related({
|
||||
from: 'a',
|
||||
type: [VerbType.Supports, VerbType.RelatedTo, VerbType.Knows]
|
||||
})
|
||||
const ids = edges.map((e) => e.id)
|
||||
expect(new Set(ids).size).toBe(ids.length)
|
||||
expect(edges).toHaveLength(3)
|
||||
})
|
||||
|
||||
it('type-only asks (no anchor) honour the whole array too', async () => {
|
||||
const edges = await brain.related({ type: [VerbType.Supports, VerbType.Knows] })
|
||||
const verbs = new Set(edges.map((e) => e.type))
|
||||
expect(verbs.has(VerbType.Supports)).toBe(true)
|
||||
expect(verbs.has(VerbType.Knows)).toBe(true)
|
||||
expect(edges).toHaveLength(2)
|
||||
})
|
||||
|
||||
it('to + type array: the target side honours every type too', async () => {
|
||||
const edges = await brain.related({ to: 'c', type: [VerbType.RelatedTo, VerbType.Supports] })
|
||||
const froms = new Set(edges.map((e) => e.from))
|
||||
expect(froms.has(v5('a'))).toBe(true)
|
||||
expect(froms.has(v5('b'))).toBe(true)
|
||||
expect(edges).toHaveLength(2)
|
||||
})
|
||||
|
||||
it('pagination stays consistent across the union', async () => {
|
||||
const page1 = await brain.related({ from: 'a', type: [VerbType.Supports, VerbType.RelatedTo, VerbType.Knows], limit: 2 })
|
||||
const page2 = await brain.related({ from: 'a', type: [VerbType.Supports, VerbType.RelatedTo, VerbType.Knows], limit: 2, offset: 2 })
|
||||
const all = [...page1, ...page2].map((e) => e.id)
|
||||
expect(new Set(all).size).toBe(3)
|
||||
})
|
||||
})
|
||||
405
tests/integration/shutdown-single-owner.test.ts
Normal file
405
tests/integration/shutdown-single-owner.test.ts
Normal file
|
|
@ -0,0 +1,405 @@
|
|||
/**
|
||||
* @module tests/integration/shutdown-single-owner
|
||||
* @description ONE SHUTDOWN, ONE OWNER.
|
||||
*
|
||||
* MEASURED IN PRODUCTION. A host that owns its own shutdown — one SIGTERM
|
||||
* listener calling `close()` on every pooled store — ran head-on into the
|
||||
* engine's own signal handler, which iterated every live instance, flushed its
|
||||
* components in parallel, and released its writer lock in a `finally`. Two
|
||||
* teardowns of the same brain at the same moment. The log shape:
|
||||
*
|
||||
* "Shutdown signal received - flushing pending data..." (SIGTERM)
|
||||
* ...148 seconds of silence...
|
||||
* "Flushed successfully (1 instance)"
|
||||
* ...the host's pool close of that same store returns 1s later
|
||||
*
|
||||
* 149s for the one store with engine work in flight, against 24s for its six
|
||||
* idle siblings. The same race in a local reproduction printed
|
||||
* `Failed to flush one Brainy instance on shutdown: Writer fence lost … the
|
||||
* lock file is gone` — the handler observing a lock the close it was racing
|
||||
* had already released.
|
||||
*
|
||||
* The contract pinned here:
|
||||
* (a) A host owner and the engine's hooks both live: EXACTLY ONE close runs
|
||||
* per brain, no fence is lost, both durability markers are written, the
|
||||
* process exits 0, and the reopen adopts rather than folding.
|
||||
* (b) No host owner: the engine's handler closes every instance by the same
|
||||
* `close()` path — markers written, clean exit.
|
||||
* (c) `close()` is idempotent and re-entrant: concurrent callers share ONE
|
||||
* execution and all of them settle.
|
||||
* (d) Flush is single-flight: N kicks during a running flush arm exactly one
|
||||
* follow-up, and two flush bodies never overlap.
|
||||
*/
|
||||
|
||||
import { describe, it, expect, beforeEach, afterEach } from 'vitest'
|
||||
import { mkdtempSync, rmSync, existsSync, readFileSync, writeFileSync } from 'node:fs'
|
||||
import { spawn } from 'node:child_process'
|
||||
import { tmpdir } from 'node:os'
|
||||
import { join } from 'node:path'
|
||||
import { Brainy } from '../../src/brainy.js'
|
||||
import { NounType } from '../../src/types/graphTypes.js'
|
||||
|
||||
const REPO_ROOT = process.cwd()
|
||||
const TSX = join(REPO_ROOT, 'node_modules', '.bin', 'tsx')
|
||||
const BRAINY_SRC = join(REPO_ROOT, 'src', 'brainy.ts')
|
||||
|
||||
function makeTempDir(prefix: string): string {
|
||||
return mkdtempSync(join(tmpdir(), prefix))
|
||||
}
|
||||
|
||||
/** The writer lock's clean-close record — written by `releaseWriterLock()`. */
|
||||
const closeRecordPath = (dir: string) => join(dir, 'locks', '_writer.close')
|
||||
/**
|
||||
* The generation store's clean-shutdown marker — the adopt-vs-fold gate.
|
||||
* (`FileSystemStorage` gzips raw objects, so the file on disk carries `.gz`;
|
||||
* both spellings are accepted so the pin survives a compression change.)
|
||||
*/
|
||||
const cleanShutdownWritten = (dir: string) =>
|
||||
existsSync(join(dir, '_system', 'clean-shutdown.json.gz')) ||
|
||||
existsSync(join(dir, '_system', 'clean-shutdown.json'))
|
||||
|
||||
/**
|
||||
* Write a child script and start it under tsx, in its OWN process group so a
|
||||
* group-wide signal reaches the grandchild that actually holds the writer
|
||||
* lock. (A file, not `tsx -e`: the eval form compiles to CommonJS, which has
|
||||
* no top-level await.)
|
||||
*/
|
||||
function startChild(scriptDir: string, body: string): ReturnType<typeof spawn> {
|
||||
const scriptPath = join(scriptDir, 'child-process.mts')
|
||||
writeFileSync(scriptPath, body)
|
||||
return spawn(TSX, [scriptPath], {
|
||||
cwd: REPO_ROOT,
|
||||
stdio: ['ignore', 'pipe', 'pipe'],
|
||||
detached: true
|
||||
})
|
||||
}
|
||||
|
||||
/** Start a child and resolve once it prints READY, collecting all its output. */
|
||||
function startAndAwaitReady(
|
||||
scriptDir: string,
|
||||
body: string
|
||||
): Promise<{ child: ReturnType<typeof spawn>; output: () => string }> {
|
||||
const child = startChild(scriptDir, body)
|
||||
let out = ''
|
||||
child.stdout?.on('data', (d) => { out += String(d) })
|
||||
child.stderr?.on('data', (d) => { out += String(d) })
|
||||
return new Promise((resolvePromise, rejectPromise) => {
|
||||
const timer = setTimeout(
|
||||
() => rejectPromise(new Error(`child never became READY:\n${out}`)),
|
||||
120_000
|
||||
)
|
||||
child.stdout?.on('data', () => {
|
||||
if (out.includes('READY')) {
|
||||
clearTimeout(timer)
|
||||
resolvePromise({ child, output: () => out })
|
||||
}
|
||||
})
|
||||
child.on('exit', (code) => {
|
||||
clearTimeout(timer)
|
||||
if (!out.includes('READY')) rejectPromise(new Error(`child exited ${code} before READY:\n${out}`))
|
||||
})
|
||||
})
|
||||
}
|
||||
|
||||
/** Capture console.warn/error/log lines emitted while `fn` runs. */
|
||||
async function captureConsole<T>(fn: () => Promise<T>): Promise<{ result: T; lines: string[] }> {
|
||||
const lines: string[] = []
|
||||
const orig = { log: console.log, warn: console.warn, error: console.error }
|
||||
const sink = (...args: unknown[]) => { lines.push(args.map((a) => String(a)).join(' ')) }
|
||||
console.log = sink as typeof console.log
|
||||
console.warn = sink as typeof console.warn
|
||||
console.error = sink as typeof console.error
|
||||
try {
|
||||
return { result: await fn(), lines }
|
||||
} finally {
|
||||
console.log = orig.log
|
||||
console.warn = orig.warn
|
||||
console.error = orig.error
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Reopen a store and assert the open ADOPTED: no crash-recovery fold, no
|
||||
* stale-lock verdict. This is the whole point of a close having run exactly
|
||||
* once — a fold is measured in tens of seconds on a real store.
|
||||
*/
|
||||
async function expectCleanReopen(dir: string): Promise<void> {
|
||||
const { result, lines } = await captureConsole(async () => {
|
||||
const next = new Brainy({ requireSubtype: false, storage: { type: 'filesystem', path: dir } })
|
||||
await next.init()
|
||||
return next
|
||||
})
|
||||
try {
|
||||
expect(lines.filter((l) => /log-authority recovery|unclean shutdown detected/i.test(l))).toEqual([])
|
||||
expect(lines.filter((l) => /Overwriting stale writer lock|appears dead/i.test(l))).toEqual([])
|
||||
} finally {
|
||||
await result.close()
|
||||
}
|
||||
}
|
||||
|
||||
/** The child's counts of closes entered and close bodies run, per brain. */
|
||||
function readResult(
|
||||
resultPath: string,
|
||||
out: string
|
||||
): { entries: Record<string, number>; bodies: Record<string, number>; releases: Record<string, number> } {
|
||||
if (!existsSync(resultPath)) throw new Error(`child wrote no result file:\n${out}`)
|
||||
return JSON.parse(readFileSync(resultPath, 'utf-8'))
|
||||
}
|
||||
|
||||
/**
|
||||
* The child-side instrumentation, shared by (a) and (b): count how many times
|
||||
* `close()` is ENTERED per brain and how many times its body actually RUNS.
|
||||
* The counting wrapper is an OWN property, so it shadows the prototype for
|
||||
* every caller — including the engine's own signal handler, which calls
|
||||
* `instance.close()`.
|
||||
*
|
||||
* `report()` writes SYNCHRONOUSLY to a file: it runs on the way out of the
|
||||
* process (the engine's handler calls `process.exit(0)` when it is the sole
|
||||
* shutdown owner), and a `console.log` to a pipe is asynchronous and can be
|
||||
* dropped by that exit.
|
||||
*/
|
||||
function childCounters(resultPath: string): string {
|
||||
return `
|
||||
const entries = {}
|
||||
const bodies = {}
|
||||
const releases = {}
|
||||
function instrument(name, brain) {
|
||||
entries[name] = 0
|
||||
bodies[name] = 0
|
||||
releases[name] = 0
|
||||
const enter = brain.close.bind(brain)
|
||||
brain.close = () => { entries[name]++; return enter() }
|
||||
const durable = brain.closeDurableSteps.bind(brain)
|
||||
brain.closeDurableSteps = () => { bodies[name]++; return durable() }
|
||||
// The writer lock is the ownership witness: the old handler released it
|
||||
// in its own finally, on top of the owner's close doing the same.
|
||||
const storage = brain.storage
|
||||
const release = storage.releaseWriterLock.bind(storage)
|
||||
storage.releaseWriterLock = () => { releases[name]++; return release() }
|
||||
}
|
||||
const report = () => {
|
||||
__writeFileSync(${JSON.stringify(resultPath)}, JSON.stringify({ entries, bodies, releases }))
|
||||
}
|
||||
`
|
||||
}
|
||||
|
||||
describe('shutdown has exactly one owner', () => {
|
||||
let dirA: string
|
||||
let dirB: string
|
||||
let scriptDir: string
|
||||
let resultPath: string
|
||||
|
||||
beforeEach(() => {
|
||||
dirA = makeTempDir('brainy-shutdown-owner-a-')
|
||||
dirB = makeTempDir('brainy-shutdown-owner-b-')
|
||||
scriptDir = makeTempDir('brainy-shutdown-owner-script-')
|
||||
resultPath = join(scriptDir, 'result.json')
|
||||
})
|
||||
|
||||
afterEach(() => {
|
||||
for (const d of [dirA, dirB, scriptDir]) {
|
||||
try { rmSync(d, { recursive: true, force: true }) } catch { /* ignore */ }
|
||||
}
|
||||
})
|
||||
|
||||
it('(a) a host owner closes both brains and the engine handler steps aside', async () => {
|
||||
const script = `
|
||||
import { writeFileSync as __writeFileSync } from 'node:fs'
|
||||
import { Brainy } from ${JSON.stringify(BRAINY_SRC)}
|
||||
const a = new Brainy({ requireSubtype: false, storage: { type: 'filesystem', path: ${JSON.stringify(dirA)} } })
|
||||
const b = new Brainy({ requireSubtype: false, storage: { type: 'filesystem', path: ${JSON.stringify(dirB)} } })
|
||||
await a.init()
|
||||
await b.init()
|
||||
await a.add({ data: 'row in brain a', type: 'concept' })
|
||||
await b.add({ data: 'row in brain b', type: 'concept' })
|
||||
${childCounters(resultPath)}
|
||||
instrument('a', a)
|
||||
instrument('b', b)
|
||||
// THE HOST'S OWN SHUTDOWN OWNER, registered after the engine's hooks —
|
||||
// the ordinary shape: the pool was built before the signal wiring.
|
||||
process.on('SIGTERM', async () => {
|
||||
await Promise.all([a.close(), b.close()])
|
||||
// Stay alive a beat so the engine's deferred handler gets its turn and
|
||||
// has to decide what to do about two already-closed brains.
|
||||
await new Promise((r) => setTimeout(r, 1500))
|
||||
report()
|
||||
process.exit(0)
|
||||
})
|
||||
console.log('READY')
|
||||
setInterval(() => {}, 1000)
|
||||
`
|
||||
const { child, output } = await startAndAwaitReady(scriptDir, script)
|
||||
|
||||
process.kill(-(child.pid as number), 'SIGTERM')
|
||||
const code = await new Promise<number | null>((r) => child.on('exit', (c) => r(c)))
|
||||
// The tsx wrapper's exit event and the grandchild that actually held the
|
||||
// locks are asynchronous with each other — let its last writes land.
|
||||
await new Promise<void>((r) => setTimeout(r, 750))
|
||||
const out = output()
|
||||
|
||||
// The process shut down cleanly.
|
||||
expect(code, `child output:\n${out}`).toBe(0)
|
||||
|
||||
// EXACTLY ONE close per brain — entered once, body run once. A second
|
||||
// entry would mean the engine's handler closed a brain its owner was
|
||||
// already closing; a second body would mean close() is not single-flight.
|
||||
const { entries, bodies, releases } = readResult(resultPath, out)
|
||||
expect(entries).toEqual({ a: 1, b: 1 })
|
||||
expect(bodies).toEqual({ a: 1, b: 1 })
|
||||
// ...and the writer lock was given up exactly once per brain. This is the
|
||||
// assertion that fails on the old handler, which released the lock in its
|
||||
// own `finally` on top of the owner's close doing the same — two owners.
|
||||
expect(releases).toEqual({ a: 1, b: 1 })
|
||||
|
||||
// The engine's handler ran (it announced the signal) and stepped aside for
|
||||
// both brains rather than touching them. setImmediate lands in the check
|
||||
// phase of the same loop turn, so a close that has begun cannot have
|
||||
// finished — it is still in flight when the handler looks.
|
||||
expect(out).toContain('Shutdown signal received')
|
||||
expect(out).toMatch(/2 Brainy instances are already closing/)
|
||||
|
||||
// Nothing was taken out from under the owner, and nothing failed.
|
||||
expect(out).not.toMatch(/Writer fence lost/i)
|
||||
expect(out).not.toMatch(/Failed to (flush|close) one Brainy instance/i)
|
||||
|
||||
// Both durability markers, both brains: the writer lock's clean-close
|
||||
// record and the generation store's clean-shutdown marker.
|
||||
for (const dir of [dirA, dirB]) {
|
||||
expect(existsSync(closeRecordPath(dir)), `clean-close record missing in ${dir}`).toBe(true)
|
||||
expect(cleanShutdownWritten(dir), `clean-shutdown marker missing in ${dir}`).toBe(true)
|
||||
}
|
||||
|
||||
// And the next open adopts instead of folding.
|
||||
await expectCleanReopen(dirA)
|
||||
await expectCleanReopen(dirB)
|
||||
}, 240_000)
|
||||
|
||||
it('(b) with no host owner the engine closes every instance the same way', async () => {
|
||||
const script = `
|
||||
import { writeFileSync as __writeFileSync } from 'node:fs'
|
||||
import { Brainy } from ${JSON.stringify(BRAINY_SRC)}
|
||||
const a = new Brainy({ requireSubtype: false, storage: { type: 'filesystem', path: ${JSON.stringify(dirA)} } })
|
||||
const b = new Brainy({ requireSubtype: false, storage: { type: 'filesystem', path: ${JSON.stringify(dirB)} } })
|
||||
await a.init()
|
||||
await b.init()
|
||||
await a.add({ data: 'row in brain a', type: 'concept' })
|
||||
await b.add({ data: 'row in brain b', type: 'concept' })
|
||||
${childCounters(resultPath)}
|
||||
instrument('a', a)
|
||||
instrument('b', b)
|
||||
process.on('exit', report)
|
||||
console.log('READY')
|
||||
setInterval(() => {}, 1000)
|
||||
`
|
||||
const { child, output } = await startAndAwaitReady(scriptDir, script)
|
||||
|
||||
process.kill(-(child.pid as number), 'SIGTERM')
|
||||
const code = await new Promise<number | null>((r) => child.on('exit', (c) => r(c)))
|
||||
// The tsx wrapper's exit event and the grandchild that actually held the
|
||||
// locks are asynchronous with each other — let its last writes land.
|
||||
await new Promise<void>((r) => setTimeout(r, 750))
|
||||
const out = output()
|
||||
|
||||
expect(code, `child output:\n${out}`).toBe(0)
|
||||
|
||||
// The engine owned this shutdown: one close per brain, through close().
|
||||
const { entries, bodies, releases } = readResult(resultPath, out)
|
||||
expect(entries).toEqual({ a: 1, b: 1 })
|
||||
expect(bodies).toEqual({ a: 1, b: 1 })
|
||||
expect(releases).toEqual({ a: 1, b: 1 })
|
||||
expect(out).toContain('Shutdown signal received')
|
||||
expect(out).toMatch(/Flushed successfully \(2 instances\)/)
|
||||
expect(out).not.toMatch(/Writer fence lost/i)
|
||||
expect(out).not.toMatch(/Failed to (flush|close) one Brainy instance/i)
|
||||
|
||||
for (const dir of [dirA, dirB]) {
|
||||
expect(existsSync(closeRecordPath(dir)), `clean-close record missing in ${dir}`).toBe(true)
|
||||
expect(cleanShutdownWritten(dir), `clean-shutdown marker missing in ${dir}`).toBe(true)
|
||||
}
|
||||
|
||||
await expectCleanReopen(dirA)
|
||||
await expectCleanReopen(dirB)
|
||||
}, 240_000)
|
||||
|
||||
it('(c) two concurrent close() callers share ONE execution, and both settle', async () => {
|
||||
const brain = new Brainy({ requireSubtype: false, storage: { type: 'filesystem', path: dirA } })
|
||||
await brain.init()
|
||||
await brain.add({ data: 'one row', type: NounType.Concept })
|
||||
|
||||
const inner = brain as unknown as { closeDurableSteps: () => Promise<void> }
|
||||
const durable = inner.closeDurableSteps.bind(inner)
|
||||
let bodies = 0
|
||||
inner.closeDurableSteps = () => { bodies++; return durable() }
|
||||
|
||||
expect(brain.isClosing).toBe(false)
|
||||
expect(brain.isClosed).toBe(false)
|
||||
|
||||
const first = brain.close()
|
||||
// The state is observable IMMEDIATELY — a signal handler that yields a
|
||||
// tick and comes back must not read a stale "not yet".
|
||||
expect(brain.isClosing).toBe(true)
|
||||
const second = brain.close()
|
||||
expect(first === second, 'concurrent callers must share the one promise').toBe(true)
|
||||
|
||||
await Promise.all([first, second])
|
||||
expect(bodies).toBe(1)
|
||||
expect(brain.isClosed).toBe(true)
|
||||
|
||||
// A caller arriving after the close finished gets the same settled answer,
|
||||
// and nothing runs again.
|
||||
await brain.close()
|
||||
expect(bodies).toBe(1)
|
||||
|
||||
expect(existsSync(closeRecordPath(dirA))).toBe(true)
|
||||
expect(cleanShutdownWritten(dirA)).toBe(true)
|
||||
}, 120_000)
|
||||
|
||||
it('(d) N kicks during a running flush arm exactly one follow-up, never a second flush', async () => {
|
||||
const brain = new Brainy({ requireSubtype: false, storage: { type: 'filesystem', path: dirA } })
|
||||
await brain.init()
|
||||
|
||||
const inner = brain as unknown as {
|
||||
_flushBodyRuns: number
|
||||
_flushConcurrencyPeak: number
|
||||
_flushInFlight: Promise<void> | null
|
||||
_flushQueued: Promise<void> | null
|
||||
_persistBackgroundFlight: Promise<void> | null
|
||||
metadataIndex: { flush: () => Promise<void> }
|
||||
kickBackgroundFlush: (reason: 'threshold' | 'idle') => void
|
||||
}
|
||||
|
||||
// Widen the flush body's window so the kicks land INSIDE it — the
|
||||
// production shape, where two flushes overlapped 3s apart.
|
||||
const metaFlush = inner.metadataIndex.flush.bind(inner.metadataIndex)
|
||||
inner.metadataIndex.flush = async () => {
|
||||
await new Promise((r) => setTimeout(r, 400))
|
||||
return metaFlush()
|
||||
}
|
||||
|
||||
await brain.add({ data: 'a write to flush', type: NounType.Concept })
|
||||
const runsBefore = inner._flushBodyRuns
|
||||
|
||||
const leader = brain.flush()
|
||||
await new Promise((r) => setTimeout(r, 50)) // the leader is inside its body
|
||||
expect(inner._flushInFlight, 'a flush is running').not.toBeNull()
|
||||
|
||||
// The cadence kicks — the door named in the defect — plus direct callers
|
||||
// (an application flush, the cross-process flush-request watcher).
|
||||
for (let i = 0; i < 5; i++) inner.kickBackgroundFlush('threshold')
|
||||
const direct = [brain.flush(), brain.flush(), brain.flush()]
|
||||
|
||||
// EXACTLY ONE follow-up is armed, however many callers arrived.
|
||||
expect(inner._flushQueued, 'the eight kicks armed one follow-up').not.toBeNull()
|
||||
|
||||
await Promise.all([leader, ...direct, inner._persistBackgroundFlight ?? Promise.resolve()])
|
||||
|
||||
// One leader + one follow-up. Not nine, and never two at once.
|
||||
expect(inner._flushBodyRuns - runsBefore).toBe(2)
|
||||
expect(inner._flushConcurrencyPeak).toBe(1)
|
||||
expect(inner._flushInFlight).toBeNull()
|
||||
expect(inner._flushQueued).toBeNull()
|
||||
|
||||
inner.metadataIndex.flush = metaFlush
|
||||
await brain.close()
|
||||
}, 120_000)
|
||||
})
|
||||
|
|
@ -95,7 +95,13 @@ describe('Storage-Level Batch Operations v5.12.0', () => {
|
|||
expect(entity?.vector?.length).toBeGreaterThan(0)
|
||||
})
|
||||
|
||||
it('should be faster than individual gets for large batches', async () => {
|
||||
it('should be faster than individual gets for large batches', async (ctx) => {
|
||||
// Wall-clock RATIO assertion — belongs to the perf lane (npm run
|
||||
// test:perf), not the correctness gate: under the exclusive release
|
||||
// gate this flaked when individual gets got faster on their own
|
||||
// (open-path/hydration changes), not because batchGet regressed.
|
||||
ctx.skip(!process.env.BRAINY_PERF_LANE, 'timing-ratio assertion — runs only under the perf lane (npm run test:perf)')
|
||||
|
||||
// Create 100 entities
|
||||
const ids: string[] = []
|
||||
for (let i = 0; i < 100; i++) {
|
||||
|
|
|
|||
184
tests/integration/transact-edge-delete-bigint-aliasing.test.ts
Normal file
184
tests/integration/transact-edge-delete-bigint-aliasing.test.ts
Normal file
|
|
@ -0,0 +1,184 @@
|
|||
/**
|
||||
* @module tests/integration/transact-edge-delete-bigint-aliasing
|
||||
* @description Regression for a fleet-adoption blocker: ANY edge delete
|
||||
* inside `transact()` — a direct unrelate or a noun-remove's cascade —
|
||||
* aborted with the metadata seam's BigInt JSON-guard error on a strict
|
||||
* (native) metadata provider.
|
||||
*
|
||||
* The aliasing chain: `planTxUnrelate`/the remove-cascade pass the SAME verb
|
||||
* object to the graph-retraction op and the metadata-retraction op. The
|
||||
* metadata leg's JSON-safe wrap ran at PLAN time, when the verb was still
|
||||
* clean — so it returned the same reference. At EXECUTE time the graph op
|
||||
* runs first and `resolveVerbEndpointInts` mirrors BigInt
|
||||
* `sourceInt`/`targetInt` onto the shared object (deliberately deferred for
|
||||
* same-batch forward refs — see transact-forward-ref-graph.test.ts); the
|
||||
* metadata op then crossed the seam with the polluted object. Direct
|
||||
* `unrelate()` resolves ints at BUILD time, before its sanitize, which is why
|
||||
* only the transact() shapes ever hit it.
|
||||
*
|
||||
* Fix under pin: the JSON-safe view is taken AT THE CROSSING — inside the
|
||||
* metadata-index operations' execute/rollback — so no plan-vs-execute
|
||||
* ordering can bypass it. The JS baseline index tolerates BigInts (it would
|
||||
* mask the bug), so these pins SPY on the seam and assert what actually
|
||||
* crossed, exactly as a strict native provider would judge it.
|
||||
*/
|
||||
import { describe, it, expect, beforeEach, afterEach } from 'vitest'
|
||||
import * as fs from 'node:fs'
|
||||
import * as os from 'node:os'
|
||||
import * as path from 'node:path'
|
||||
import { Brainy } from '../../src/brainy.js'
|
||||
import { NounType, VerbType } from '../../src/types/graphTypes.js'
|
||||
import {
|
||||
AddToMetadataIndexOperation,
|
||||
RemoveFromMetadataIndexOperation
|
||||
} from '../../src/transaction/operations/index.js'
|
||||
|
||||
let seq = 0
|
||||
const freshId = (): string =>
|
||||
`00000000-0000-4000-8000-${(++seq).toString(16).padStart(12, '0')}`
|
||||
|
||||
/** Top-level BigInt-valued keys of a candidate seam crossing (the guard's law). */
|
||||
const bigintKeys = (metadata: unknown): string[] => {
|
||||
if (metadata === null || typeof metadata !== 'object') return []
|
||||
return Object.entries(metadata as Record<string, unknown>)
|
||||
.filter(([, v]) => typeof v === 'bigint')
|
||||
.map(([k]) => k)
|
||||
}
|
||||
|
||||
describe('transact() edge deletes never carry BigInt across the metadata seam', () => {
|
||||
let dir: string
|
||||
let brain: any
|
||||
let crossings: Array<{ door: string; id: string; keys: string[] }>
|
||||
|
||||
beforeEach(async () => {
|
||||
process.env.BRAINY_DETERMINISTIC_EMBEDDINGS = 'true'
|
||||
dir = fs.mkdtempSync(path.join(os.tmpdir(), 'brainy-tx-bigint-'))
|
||||
brain = new Brainy({
|
||||
requireSubtype: false,
|
||||
storage: { type: 'filesystem', path: dir },
|
||||
dimensions: 384,
|
||||
silent: true
|
||||
})
|
||||
await brain.init()
|
||||
|
||||
// Spy on the seam the way a strict native provider judges it: record the
|
||||
// BigInt-valued top-level keys of every metadata argument that crosses.
|
||||
// The JS baseline index tolerates BigInts, so without this the baseline
|
||||
// run would green a shape the native pair aborts on.
|
||||
crossings = []
|
||||
const index = brain.metadataIndex
|
||||
for (const door of ['addToIndex', 'removeFromIndex'] as const) {
|
||||
const real = index[door].bind(index)
|
||||
index[door] = (id: string, metadata: unknown, ...rest: unknown[]) => {
|
||||
crossings.push({ door, id, keys: bigintKeys(metadata) })
|
||||
return real(id, metadata, ...rest)
|
||||
}
|
||||
}
|
||||
})
|
||||
|
||||
afterEach(async () => {
|
||||
await brain.close()
|
||||
fs.rmSync(dir, { recursive: true, force: true })
|
||||
})
|
||||
|
||||
it('CASE 1 (the fleet repro): relate, then transact([{op: unrelate}])', async () => {
|
||||
const a = await brain.add({ id: freshId(), data: 'a', type: NounType.Thing })
|
||||
const b = await brain.add({ id: freshId(), data: 'b', type: NounType.Thing })
|
||||
const verbId = await brain.relate({ from: a, to: b, type: VerbType.RelatedTo })
|
||||
|
||||
crossings.length = 0
|
||||
await brain.transact([{ op: 'unrelate', id: verbId }])
|
||||
|
||||
const polluted = crossings.filter((c) => c.keys.length > 0)
|
||||
expect(polluted).toEqual([])
|
||||
expect(await brain.storage.getVerb(verbId)).toBeFalsy()
|
||||
})
|
||||
|
||||
it('CASE 2 (the cascade shape): transact([{op: remove}]) cascading edge deletes', async () => {
|
||||
const a = await brain.add({ id: freshId(), data: 'a', type: NounType.Thing })
|
||||
const b = await brain.add({ id: freshId(), data: 'b', type: NounType.Thing })
|
||||
const c = await brain.add({ id: freshId(), data: 'c', type: NounType.Thing })
|
||||
const ab = await brain.relate({ from: a, to: b, type: VerbType.RelatedTo })
|
||||
const ca = await brain.relate({ from: c, to: a, type: VerbType.RelatedTo })
|
||||
|
||||
crossings.length = 0
|
||||
await brain.transact([{ op: 'remove', id: a }])
|
||||
|
||||
const polluted = crossings.filter((c2) => c2.keys.length > 0)
|
||||
expect(polluted).toEqual([])
|
||||
expect(await brain.get(a)).toBeFalsy()
|
||||
expect(await brain.storage.getVerb(ab)).toBeFalsy()
|
||||
expect(await brain.storage.getVerb(ca)).toBeFalsy()
|
||||
})
|
||||
|
||||
it('CASE 3 (one batch, both legs): adds + relate + unrelate of a pre-existing edge', async () => {
|
||||
const a = await brain.add({ id: freshId(), data: 'a', type: NounType.Thing })
|
||||
const b = await brain.add({ id: freshId(), data: 'b', type: NounType.Thing })
|
||||
const old = await brain.relate({ from: a, to: b, type: VerbType.RelatedTo })
|
||||
|
||||
const x = freshId()
|
||||
crossings.length = 0
|
||||
await brain.transact([
|
||||
{ op: 'add', id: x, data: 'x', type: NounType.Thing },
|
||||
{ op: 'relate', from: a, to: x, type: VerbType.RelatedTo },
|
||||
{ op: 'unrelate', id: old }
|
||||
])
|
||||
|
||||
const polluted = crossings.filter((c) => c.keys.length > 0)
|
||||
expect(polluted).toEqual([])
|
||||
expect(await brain.storage.getVerb(old)).toBeFalsy()
|
||||
const edges = await brain.related({ from: a })
|
||||
expect(edges.length).toBe(1)
|
||||
expect(edges[0].id).not.toBe(old)
|
||||
})
|
||||
})
|
||||
|
||||
describe('the metadata-index operations sanitize at the crossing, not at construction', () => {
|
||||
/** A strict seam: refuses BigInts exactly as the native provider does. */
|
||||
const strictIndex = () => {
|
||||
const seen: Array<{ door: string; keys: string[] }> = []
|
||||
const judge = (door: string, metadata: unknown) => {
|
||||
const keys = bigintKeys(metadata)
|
||||
seen.push({ door, keys })
|
||||
if (keys.length > 0) {
|
||||
throw new Error(
|
||||
`${door}: the metadata object violates the provider seam's JSON ` +
|
||||
`contract — BigInt at ${keys.join(', ')}.`
|
||||
)
|
||||
}
|
||||
}
|
||||
return {
|
||||
seen,
|
||||
addToIndex: async (_id: string, metadata: unknown) => judge('addToIndex', metadata),
|
||||
removeFromIndex: async (_id: string, metadata: unknown) => judge('removeFromIndex', metadata)
|
||||
}
|
||||
}
|
||||
|
||||
it('RemoveFromMetadataIndexOperation: entity mutated AFTER construction still crosses clean', async () => {
|
||||
const index = strictIndex()
|
||||
const verb: Record<string, unknown> = { id: 'v1', sourceId: 'a', targetId: 'b' }
|
||||
const op = new RemoveFromMetadataIndexOperation(index as any, 'v1', verb, () => 7n)
|
||||
|
||||
// The graph leg's execute-time endpoint resolution, simulated: the shared
|
||||
// object is polluted between plan and execute.
|
||||
verb.sourceInt = 800_000n
|
||||
verb.targetInt = 800_001n
|
||||
|
||||
const rollback = await op.execute()
|
||||
await rollback()
|
||||
expect(index.seen.map((s) => s.keys)).toEqual([[], []])
|
||||
})
|
||||
|
||||
it('AddToMetadataIndexOperation: same law on the add leg and its rollback', async () => {
|
||||
const index = strictIndex()
|
||||
const verb: Record<string, unknown> = { id: 'v2', sourceId: 'a', targetId: 'b' }
|
||||
const op = new AddToMetadataIndexOperation(index as any, 'v2', verb, () => 7n)
|
||||
|
||||
verb.sourceInt = 800_000n
|
||||
verb.targetInt = 800_001n
|
||||
|
||||
const rollback = await op.execute()
|
||||
await rollback()
|
||||
expect(index.seen.map((s) => s.keys)).toEqual([[], []])
|
||||
})
|
||||
})
|
||||
115
tests/integration/vfs-containment-batched.test.ts
Normal file
115
tests/integration/vfs-containment-batched.test.ts
Normal file
|
|
@ -0,0 +1,115 @@
|
|||
/**
|
||||
* @module tests/integration/vfs-containment-batched
|
||||
* @description repairContainment costs O(edges/page) graph calls, not O(entities) (10.4.9 train).
|
||||
*
|
||||
* Pass 2 used to issue one awaited `related({ to })` per VFS entity — minutes
|
||||
* of serialized graph calls on large brains. Now one paged walk over every
|
||||
* Contains edge feeds an in-memory group-by-target, and only actual defects
|
||||
* mutate. These pins hold the verdicts (duplicate removed, stale parent
|
||||
* removed, missing edge restored, user knowledge edges untouched) AND the
|
||||
* cost shape (related() call count independent of the entity count).
|
||||
*/
|
||||
import { describe, it, expect, beforeAll, afterAll, vi } from 'vitest'
|
||||
import { Brainy } from '../../src/brainy'
|
||||
import { NounType, VerbType } from '../../src/types/graphTypes'
|
||||
|
||||
const FILES = 60
|
||||
|
||||
describe('repairContainment: batched pass 2', () => {
|
||||
let brain: Brainy<any>
|
||||
let result: { removed: number; restored: number }
|
||||
let relatedCalls = 0
|
||||
|
||||
beforeAll(async () => {
|
||||
brain = new Brainy({ requireSubtype: false, storage: { type: 'memory' } })
|
||||
await brain.init()
|
||||
const vfs = (brain as any).vfs ?? (brain as any)._vfs
|
||||
expect(vfs).toBeTruthy()
|
||||
await vfs.init()
|
||||
|
||||
// A directory and FILES entries under it, wired as real VFS rows.
|
||||
const mkNode = async (id: string, path: string, vfsType: string): Promise<void> => {
|
||||
await brain.add({
|
||||
id,
|
||||
data: `vfs node ${path}`,
|
||||
type: NounType.File,
|
||||
visibility: 'system',
|
||||
metadata: { vfsType, path }
|
||||
})
|
||||
}
|
||||
await mkNode('dir', '/docs', 'directory')
|
||||
const rootId = vfs.rootEntityId ?? (await vfs.initializeRoot?.())
|
||||
if (rootId) {
|
||||
await brain.relate({
|
||||
from: rootId,
|
||||
to: 'dir',
|
||||
type: VerbType.Contains,
|
||||
subtype: 'vfs-contains',
|
||||
metadata: { isVFS: true }
|
||||
})
|
||||
}
|
||||
for (let i = 0; i < FILES; i++) {
|
||||
await mkNode(`f-${i}`, `/docs/f-${i}.md`, 'file')
|
||||
if (i === 0) continue // f-0: MISSING edge — must be restored
|
||||
await brain.relate({
|
||||
from: 'dir',
|
||||
to: `f-${i}`,
|
||||
type: VerbType.Contains,
|
||||
subtype: 'vfs-contains',
|
||||
metadata: { isVFS: true }
|
||||
})
|
||||
}
|
||||
// NOTE: relate() is idempotent for an identical from/to/type, so a true
|
||||
// duplicate (a concurrent-writer artifact) cannot be seeded through the
|
||||
// public API — the duplicate branch is covered by the tree-correctness
|
||||
// pin below, which proves at most one vfs edge survives per file.
|
||||
// f-2: STALE parent edge (from a sibling file) — must be removed.
|
||||
await brain.relate({
|
||||
from: 'f-3',
|
||||
to: 'f-2',
|
||||
type: VerbType.Contains,
|
||||
subtype: 'vfs-contains',
|
||||
metadata: { isVFS: true }
|
||||
})
|
||||
// A USER knowledge Contains edge (not vfs-flagged) — must be untouched.
|
||||
await brain.relate({ from: 'f-4', to: 'f-5', type: VerbType.Contains })
|
||||
|
||||
const spy = vi.spyOn(brain, 'related')
|
||||
result = await vfs.repairContainment()
|
||||
relatedCalls = spy.mock.calls.length
|
||||
spy.mockRestore()
|
||||
})
|
||||
|
||||
afterAll(async () => {
|
||||
brain = null as any
|
||||
})
|
||||
|
||||
it('restores the missing edge and removes the stale parent — exactly', () => {
|
||||
expect(result.restored).toBe(1) // f-0's missing edge
|
||||
expect(result.removed).toBe(1) // f-2's stale parent (f-3 → f-2)
|
||||
})
|
||||
|
||||
it('the repaired tree is correct: every file has exactly one vfs edge from its dir', async () => {
|
||||
for (let i = 0; i < 6; i++) {
|
||||
const incoming = await brain.related({ to: `f-${i}`, type: VerbType.Contains })
|
||||
const vfsEdges = incoming.filter(
|
||||
(e) => e.subtype === 'vfs-contains' || (e.metadata as any)?.isVFS === true
|
||||
)
|
||||
expect(vfsEdges, `f-${i}`).toHaveLength(1)
|
||||
}
|
||||
})
|
||||
|
||||
it('never touches user knowledge edges', async () => {
|
||||
const incoming = await brain.related({ to: 'f-5', type: VerbType.Contains })
|
||||
const user = incoming.filter(
|
||||
(e) => e.subtype !== 'vfs-contains' && (e.metadata as any)?.isVFS !== true
|
||||
)
|
||||
expect(user).toHaveLength(1)
|
||||
})
|
||||
|
||||
it('cost shape: related() calls do not scale with the entity count', () => {
|
||||
// One paged type-only walk (~E/1000 pages) — with 60+ entities the old
|
||||
// shape issued 60+ calls; the new one a handful. Bound generously.
|
||||
expect(relatedCalls).toBeLessThanOrEqual(5)
|
||||
})
|
||||
})
|
||||
|
|
@ -113,7 +113,12 @@ describe('Brainy Batch Operations', () => {
|
|||
items: Array.from({ length: 100 }, (_, i) => ({
|
||||
data: `Bulk ${i}`,
|
||||
type: NounType.Thing,
|
||||
metadata: { counter: 0 }
|
||||
metadata: { counter: 0 },
|
||||
// This test exercises updateMany's batching, not embedding — the
|
||||
// sanctioned "unvectored" `[]` shape (see
|
||||
// tests/integration/index-skips-unvectored.test.ts) skips the
|
||||
// real embedder entirely.
|
||||
vector: []
|
||||
}))
|
||||
})
|
||||
const manyIds = manyResult.successful
|
||||
|
|
@ -274,7 +279,12 @@ describe('Brainy Batch Operations', () => {
|
|||
const manyResult = await brain.addMany({
|
||||
items: Array.from({ length: 100 }, (_, i) => ({
|
||||
data: `Bulk Delete ${i}`,
|
||||
type: NounType.Thing
|
||||
type: NounType.Thing,
|
||||
// This test exercises removeMany's batching, not embedding — the
|
||||
// sanctioned "unvectored" `[]` shape (see
|
||||
// tests/integration/index-skips-unvectored.test.ts) skips the
|
||||
// real embedder entirely.
|
||||
vector: []
|
||||
}))
|
||||
})
|
||||
const manyIds = manyResult.successful
|
||||
|
|
@ -545,10 +555,18 @@ describe('Brainy Batch Operations', () => {
|
|||
|
||||
it('should validate batch size limits', async () => {
|
||||
// Try to add a large batch (reduced from 10000 to 1000 for reasonable test time)
|
||||
// This test validates the batch SIZE law, not embeddings — items carry
|
||||
// the sanctioned "unvectored" `[]` shape (see
|
||||
// tests/integration/index-skips-unvectored.test.ts) so addMany's batch
|
||||
// embedder is never invoked; 1000 real embeddings under the root
|
||||
// vitest config (which does not mock the embedder) is a 60-180s
|
||||
// budget flake waiting to happen, not a defect in what this test
|
||||
// actually asserts.
|
||||
const largeCount = 1000
|
||||
const largeItems = Array.from({ length: largeCount }, (_, i) => ({
|
||||
data: `Large ${i}`,
|
||||
type: NounType.Thing
|
||||
type: NounType.Thing,
|
||||
vector: []
|
||||
}))
|
||||
|
||||
try {
|
||||
|
|
@ -560,12 +578,7 @@ describe('Brainy Batch Operations', () => {
|
|||
// Might throw if there's a limit
|
||||
expect(error).toBeDefined()
|
||||
}
|
||||
// order-of-magnitude guard: this test batches 20x the item count of the
|
||||
// sibling "perform better" test above (worst measured 11.9s for 50
|
||||
// items on CPU-only honest iron); the prior 60s timeout was itself
|
||||
// observed being hit, so this is 3x that floor rather than a scaled
|
||||
// extrapolation, to leave real headroom for run-to-run variance
|
||||
}, 180000)
|
||||
})
|
||||
|
||||
it('should provide meaningful error messages', async () => {
|
||||
try {
|
||||
|
|
|
|||
175
tests/unit/brainy/flush-single-flight.test.ts
Normal file
175
tests/unit/brainy/flush-single-flight.test.ts
Normal file
|
|
@ -0,0 +1,175 @@
|
|||
/**
|
||||
* @module tests/unit/brainy/flush-single-flight
|
||||
* @description THE FLUSH GATE NEVER STRANDS A WAITER.
|
||||
*
|
||||
* The gate serialises flushes: one body runs, at most one waits. The failure
|
||||
* mode that shape invites is a promise CYCLE — a queued follow-up expressed as
|
||||
* `leader.then(() => this.flush())` is settled only by resolving the promise
|
||||
* the leader is being awaited through, so anything that awaits `flush()` from
|
||||
* inside a flush body closes the graph on itself and nobody ever resolves.
|
||||
* That is an unbounded hang, not a slow flush, and it presents exactly like a
|
||||
* test timing out inside a bulk write.
|
||||
*
|
||||
* The gate therefore settles its waiter from the MACHINE (a bare deferred
|
||||
* promoted in the leader's `finally`), never from a chain. The laws pinned
|
||||
* here, each on a path that must settle the waiter:
|
||||
*
|
||||
* (a) many callers during one running flush → one body, one follow-up, and
|
||||
* EVERY caller resolves within a bound;
|
||||
* (b) the leader REJECTS → its own caller rejects, and the queued caller is
|
||||
* still run and still settled;
|
||||
* (c) the promoted follow-up itself rejects → its waiter rejects (settled,
|
||||
* not stranded) and the gate is left open for the next flush;
|
||||
* (d) the leader's promise does not wait for its follower.
|
||||
*/
|
||||
|
||||
import { describe, it, expect, afterEach } from 'vitest'
|
||||
import { Brainy } from '../../../src/brainy'
|
||||
import { NounType } from '../../../src/types/graphTypes'
|
||||
|
||||
type GateInternals = {
|
||||
_flushInFlight: Promise<void> | null
|
||||
_flushQueued: Promise<void> | null
|
||||
_flushBodyRuns: number
|
||||
_flushConcurrencyPeak: number
|
||||
_flushSteps: () => Promise<void>
|
||||
kickBackgroundFlush: (reason: 'threshold' | 'idle') => void
|
||||
}
|
||||
|
||||
/** Fail loudly rather than hanging the suite: a stranded waiter never settles. */
|
||||
function withinBound<T>(p: Promise<T>, ms: number, what: string): Promise<T> {
|
||||
let timer: ReturnType<typeof setTimeout>
|
||||
return Promise.race([
|
||||
p,
|
||||
new Promise<never>((_, reject) => {
|
||||
timer = setTimeout(() => reject(new Error(`${what} did not settle within ${ms}ms`)), ms)
|
||||
})
|
||||
]).finally(() => clearTimeout(timer)) as Promise<T>
|
||||
}
|
||||
|
||||
describe('the flush gate settles every waiter', () => {
|
||||
const brains: Brainy<any>[] = []
|
||||
|
||||
afterEach(async () => {
|
||||
for (const b of brains.splice(0)) {
|
||||
try { await b.close() } catch { /* already closed */ }
|
||||
}
|
||||
})
|
||||
|
||||
async function openBrain(): Promise<Brainy<any>> {
|
||||
const brain = new Brainy({ requireSubtype: false, storage: { type: 'memory' } })
|
||||
brains.push(brain)
|
||||
await brain.init()
|
||||
await brain.add({ data: 'a write, so a flush has work', type: NounType.Thing })
|
||||
return brain
|
||||
}
|
||||
|
||||
it('(a) every caller arriving during one flush resolves, and only one follows', async () => {
|
||||
const brain = await openBrain()
|
||||
const inner = brain as unknown as GateInternals
|
||||
|
||||
const realSteps = inner._flushSteps.bind(inner)
|
||||
inner._flushSteps = async () => {
|
||||
await new Promise((r) => setTimeout(r, 120))
|
||||
return realSteps()
|
||||
}
|
||||
|
||||
const runsBefore = inner._flushBodyRuns
|
||||
const leader = brain.flush()
|
||||
await new Promise((r) => setTimeout(r, 20))
|
||||
|
||||
const joiners = [brain.flush(), brain.flush(), brain.flush(), brain.flush()]
|
||||
for (let i = 0; i < 4; i++) inner.kickBackgroundFlush('threshold')
|
||||
expect(inner._flushQueued, 'exactly one waiter is queued').not.toBeNull()
|
||||
|
||||
await withinBound(Promise.all([leader, ...joiners]), 15_000, 'the flush callers')
|
||||
|
||||
expect(inner._flushBodyRuns - runsBefore).toBe(2)
|
||||
expect(inner._flushConcurrencyPeak).toBe(1)
|
||||
expect(inner._flushQueued).toBeNull()
|
||||
})
|
||||
|
||||
it('(b) a leader that REJECTS still runs and settles the queued waiter', async () => {
|
||||
const brain = await openBrain()
|
||||
const inner = brain as unknown as GateInternals
|
||||
|
||||
const realSteps = inner._flushSteps.bind(inner)
|
||||
let call = 0
|
||||
inner._flushSteps = async () => {
|
||||
call++
|
||||
await new Promise((r) => setTimeout(r, 80))
|
||||
if (call === 1) throw new Error('injected: the leader flush failed')
|
||||
return realSteps()
|
||||
}
|
||||
|
||||
const leader = brain.flush()
|
||||
await new Promise((r) => setTimeout(r, 20))
|
||||
const queued = brain.flush()
|
||||
|
||||
await expect(leader).rejects.toThrow(/injected: the leader flush failed/)
|
||||
// The waiter is NOT collateral damage of the leader's failure: it gets its
|
||||
// own run, and it settles.
|
||||
await withinBound(queued, 15_000, 'the queued waiter after a failed leader')
|
||||
expect(call).toBe(2)
|
||||
expect(inner._flushQueued).toBeNull()
|
||||
expect(inner._flushInFlight).toBeNull()
|
||||
})
|
||||
|
||||
it('(c) a promoted follow-up that rejects settles its waiter and opens the gate', async () => {
|
||||
const brain = await openBrain()
|
||||
const inner = brain as unknown as GateInternals
|
||||
|
||||
const realSteps = inner._flushSteps.bind(inner)
|
||||
let call = 0
|
||||
inner._flushSteps = async () => {
|
||||
call++
|
||||
await new Promise((r) => setTimeout(r, 80))
|
||||
if (call === 2) throw new Error('injected: the follow-up flush failed')
|
||||
return realSteps()
|
||||
}
|
||||
|
||||
const leader = brain.flush()
|
||||
await new Promise((r) => setTimeout(r, 20))
|
||||
const queued = brain.flush()
|
||||
|
||||
await withinBound(leader, 15_000, 'the leader')
|
||||
await withinBound(
|
||||
expect(queued).rejects.toThrow(/injected: the follow-up flush failed/),
|
||||
15_000,
|
||||
'the rejected follow-up'
|
||||
)
|
||||
// The gate is open: a later flush still runs.
|
||||
inner._flushSteps = realSteps
|
||||
await brain.add({ data: 'another write', type: NounType.Thing })
|
||||
await withinBound(brain.flush(), 15_000, 'the flush after a failed follow-up')
|
||||
expect(inner._flushInFlight).toBeNull()
|
||||
expect(inner._flushQueued).toBeNull()
|
||||
})
|
||||
|
||||
it('(d) the leader does not wait for its follower', async () => {
|
||||
const brain = await openBrain()
|
||||
const inner = brain as unknown as GateInternals
|
||||
|
||||
const realSteps = inner._flushSteps.bind(inner)
|
||||
let call = 0
|
||||
inner._flushSteps = async () => {
|
||||
call++
|
||||
// The follow-up is deliberately far slower than the leader.
|
||||
await new Promise((r) => setTimeout(r, call === 1 ? 60 : 600))
|
||||
return realSteps()
|
||||
}
|
||||
|
||||
const leader = brain.flush()
|
||||
await new Promise((r) => setTimeout(r, 20))
|
||||
const queued = brain.flush()
|
||||
|
||||
const t0 = Date.now()
|
||||
await withinBound(leader, 15_000, 'the leader')
|
||||
const leaderWall = Date.now() - t0
|
||||
// If the leader awaited its follower it could not return before the
|
||||
// follower's own 600ms body had run.
|
||||
expect(leaderWall).toBeLessThan(500)
|
||||
|
||||
await withinBound(queued, 15_000, 'the follower')
|
||||
})
|
||||
})
|
||||
|
|
@ -147,4 +147,119 @@ describe('db/GenerationSegmentStore — the D1+D3 packed tier', () => {
|
|||
await expect(store.fold([gen(4), gen(4)])).rejects.toThrow(/strictly ascending/)
|
||||
await expect(store.fold([])).rejects.toThrow(/at least one generation/)
|
||||
})
|
||||
|
||||
// ==========================================================================
|
||||
// THE DENSITY LAW
|
||||
// ==========================================================================
|
||||
//
|
||||
// A sealed segment declares a CONTIGUOUS range and every reader treats that
|
||||
// range as containment. Folding a sparse batch therefore makes the segment
|
||||
// claim generations it does not hold — and because `open()` merges declared
|
||||
// ranges back into committedRanges, the hole is re-admitted as committed
|
||||
// history and every later maintenance pass fails asking for a frame that was
|
||||
// never written. That is the "generation N is inside sealed segment
|
||||
// seg-....bgs's declared range but has no frame — packed history is damaged"
|
||||
// narration seen on every run of the affected stores.
|
||||
|
||||
it('fold REFUSES a batch with a hole — a dense range may not be declared over sparse input', async () => {
|
||||
await expect(store.fold([gen(1), gen(2), gen(4)])).rejects.toThrow(
|
||||
/not contiguous: 2 → 4 skips 1 generation/
|
||||
)
|
||||
// The refusal loses nothing: no segment was sealed, so the generations
|
||||
// stay in the live tier and the next pass folds them correctly.
|
||||
expect(store.segments()).toHaveLength(0)
|
||||
expect(store.hasGeneration(1)).toBe(false)
|
||||
})
|
||||
|
||||
it('a wider gap names how many generations it would have swallowed', async () => {
|
||||
await expect(store.fold([gen(10), gen(20)])).rejects.toThrow(
|
||||
/not contiguous: 10 → 20 skips 9 generation\(s\)/
|
||||
)
|
||||
})
|
||||
|
||||
it('two contiguous runs folded separately declare honest ranges', async () => {
|
||||
// What the caller now does instead of folding across the gap.
|
||||
const a = await store.fold([gen(1), gen(2), gen(3)])
|
||||
const b = await store.fold([gen(7), gen(8)])
|
||||
expect(a).toMatchObject({ firstGeneration: 1, lastGeneration: 3, frames: 3 })
|
||||
expect(b).toMatchObject({ firstGeneration: 7, lastGeneration: 8, frames: 2 })
|
||||
// The gap is honestly outside the packed tier.
|
||||
for (const g of [4, 5, 6]) expect(store.hasGeneration(g)).toBe(false)
|
||||
for (const g of [1, 2, 3, 7, 8]) expect(store.hasGeneration(g)).toBe(true)
|
||||
expect(await store.actualRanges()).toEqual([
|
||||
[1, 3],
|
||||
[7, 8]
|
||||
])
|
||||
})
|
||||
|
||||
it('actualRanges() is exact and I/O-free for dense segments', async () => {
|
||||
await store.fold([gen(1), gen(2)])
|
||||
await store.fold([gen(3), gen(4)])
|
||||
// Adjacent dense segments each contribute their declared range.
|
||||
expect(await store.actualRanges()).toEqual([
|
||||
[1, 2],
|
||||
[3, 4]
|
||||
])
|
||||
})
|
||||
|
||||
// ---- pre-existing damage: a store sealed by the old writer ----------------
|
||||
|
||||
/**
|
||||
* Seal a SPARSE segment the way the pre-fix writer did: write the bytes and
|
||||
* sidecar for a contiguous run, then rewrite the manifest so the segment
|
||||
* declares a wider range than the frames it holds. This reproduces on disk
|
||||
* exactly what the affected stores carry, without needing the old code.
|
||||
*/
|
||||
const sealSparseSegment = async (): Promise<void> => {
|
||||
await store.fold([gen(1), gen(2), gen(3)])
|
||||
const manifest = (await storage.readRawObject(`${SEGMENTS_PREFIX}/manifest.json`)) as any
|
||||
// Declare 1..5 while holding frames for 1..3 — generations 4 and 5 become
|
||||
// holes inside a sealed range.
|
||||
manifest.segments[0].lastGeneration = 5
|
||||
await storage.writeRawObject(`${SEGMENTS_PREFIX}/manifest.json`, manifest)
|
||||
}
|
||||
|
||||
it('a pre-existing sparse segment reports its holes as UNPACKED, not as damage', async () => {
|
||||
await sealSparseSegment()
|
||||
const reopened = new GenerationSegmentStore(storage as any)
|
||||
await reopened.open()
|
||||
|
||||
// The frames it really holds still serve, byte-faithfully.
|
||||
expect((await reopened.readDelta(2))?.timestamp).toBe(1_700_000_000_002)
|
||||
expect(await reopened.readRecords(3)).toHaveLength(2)
|
||||
|
||||
// The holes answer "not packed" instead of throwing. This is the fix for
|
||||
// the wedge: the old reader threw here on EVERY maintenance pass.
|
||||
expect(await reopened.readDelta(4)).toBeNull()
|
||||
expect(await reopened.readRecords(5)).toBeNull()
|
||||
})
|
||||
|
||||
it('actualRanges() excludes the holes so they are never re-admitted as committed', async () => {
|
||||
await sealSparseSegment()
|
||||
const reopened = new GenerationSegmentStore(storage as any)
|
||||
await reopened.open()
|
||||
// Declared 1..5; actually holds 1..3. The store seeds committedRanges from
|
||||
// THIS, so generations 4 and 5 never become committed history again.
|
||||
expect(await reopened.actualRanges()).toEqual([[1, 3]])
|
||||
})
|
||||
|
||||
it('a DENSE segment missing a frame is still loud damage', async () => {
|
||||
// The other side of the branch: when the manifest claims a complete span,
|
||||
// a missing frame means the manifest and sidecar disagree — real damage,
|
||||
// and it must not be quietly downgraded to "unpacked".
|
||||
await store.fold([gen(1), gen(2), gen(3)])
|
||||
const idxPath = `${SEGMENTS_PREFIX}/seg-${String(1).padStart(20, '0')}.idx`
|
||||
const raw = (await storage.readRawBytes(idxPath))!
|
||||
const { decode, encode } = await import('@msgpack/msgpack')
|
||||
const idx = decode(raw) as any
|
||||
// Drop generation 2's entry while the manifest still declares 3 frames.
|
||||
idx.generations = idx.generations.filter(([g]: [number]) => g !== 2)
|
||||
await storage.writeRawBytes(idxPath, encode(idx))
|
||||
|
||||
const reopened = new GenerationSegmentStore(storage as any)
|
||||
await reopened.open()
|
||||
await expect(reopened.readDelta(2)).rejects.toThrow(
|
||||
/manifest and the sidecar disagree; packed history is damaged/
|
||||
)
|
||||
})
|
||||
})
|
||||
|
|
|
|||
254
tests/unit/db/generationStore-commit-guard.test.ts
Normal file
254
tests/unit/db/generationStore-commit-guard.test.ts
Normal file
|
|
@ -0,0 +1,254 @@
|
|||
/**
|
||||
* @module tests/unit/db/generationStore-commit-guard
|
||||
* @description Pins the commit-order guard on
|
||||
* `GenerationStore.commitTransaction()` (`src/db/generationStore.ts`).
|
||||
*
|
||||
* `reservedGensAsc()`'s own doc comment states an invariant it never
|
||||
* enforced: pending single-op generations are always greater than every
|
||||
* committed one, because the store's only two sanctioned callers —
|
||||
* `Brainy.transact()` and `Brainy.compactHistory()` — flush the pending tier
|
||||
* before committing. Nothing stopped a caller from invoking
|
||||
* `commitTransaction()` directly while single-ops were still buffered: the
|
||||
* fresh commit would land in `committedRanges` ABOVE those lower,
|
||||
* still-pending generations, so the committed-then-pending concatenation
|
||||
* `reservedGensAsc()` yields is no longer ascending — and `resolveManyAt`
|
||||
* (which walks committed ranges before pending ones) would silently report a
|
||||
* WRONG before-image for a point-in-time read. `commitTransaction()` now
|
||||
* refuses loudly (`PendingSingleOpsUnflushedError`) instead of assuming.
|
||||
*
|
||||
* Four pins:
|
||||
* 1. A direct `commitTransaction()` call while single-ops are pending throws
|
||||
* and commits NOTHING.
|
||||
* 2. The same commit succeeds once the pending tier is flushed first.
|
||||
* 3. `Brainy.transact()` — which already flushes first — is unaffected
|
||||
* (mirrors `tests/unit/db/generation-chain.test.ts`'s `seedX()`/`bumpX()`
|
||||
* transact pin: add, then transact-update, generation advances by one
|
||||
* each time, the update lands).
|
||||
* 4. `reservedGensAsc()` stays ascending across a real add+transact+delete
|
||||
* workload — proven by point-in-time reads (`asOf`) staying correct
|
||||
* throughout, which is exactly what an ordering break would corrupt.
|
||||
*/
|
||||
|
||||
import { describe, it, expect, beforeEach, afterEach } from 'vitest'
|
||||
import { MemoryStorage } from '../../../src/storage/adapters/memoryStorage.js'
|
||||
import {
|
||||
GenerationStore,
|
||||
GENERATIONS_PREFIX,
|
||||
MANIFEST_PATH
|
||||
} from '../../../src/db/generationStore.js'
|
||||
import { PendingSingleOpsUnflushedError } from '../../../src/db/errors.js'
|
||||
import { Brainy } from '../../../src/index.js'
|
||||
import { NounType } from '../../../src/types/graphTypes.js'
|
||||
import { createTestConfig, generateTestVector } from '../../helpers/test-factory.js'
|
||||
|
||||
/** Precomputed embedding so Brainy-level adds skip the (slow) embedding model —
|
||||
* these tests exercise the generation layer, not semantics. */
|
||||
const VEC = generateTestVector()
|
||||
|
||||
// Entity ids must be UUID-shaped (the sharded storage layout derives the
|
||||
// shard from the UUID hex) — same fixture convention as generationStore.test.ts.
|
||||
const ID_A = '00000000-0000-4000-8000-0000000000aa'
|
||||
const ID_B = '00000000-0000-4000-8000-0000000000bb'
|
||||
|
||||
/** Stored-metadata fixture in the canonical shape the live write paths use
|
||||
* (matches generationStore.test.ts's fixture exactly). */
|
||||
function metadataFixture(version: number): Record<string, unknown> {
|
||||
return {
|
||||
noun: NounType.Document,
|
||||
subtype: 'note',
|
||||
data: `payload-v${version}`,
|
||||
version,
|
||||
createdAt: 1000,
|
||||
updatedAt: 1000 + version,
|
||||
_rev: version
|
||||
}
|
||||
}
|
||||
|
||||
describe('db/GenerationStore — commitTransaction pending-tier guard (store level)', () => {
|
||||
let storage: MemoryStorage
|
||||
let store: GenerationStore
|
||||
|
||||
beforeEach(async () => {
|
||||
storage = new MemoryStorage()
|
||||
await storage.init()
|
||||
store = new GenerationStore(storage)
|
||||
await store.open()
|
||||
})
|
||||
|
||||
/** Buffer one single-op generation via commitSingleOp WITHOUT flushing —
|
||||
* the pending tier that must be drained before commitTransaction(). */
|
||||
async function pendingSingleOp(id: string, version: number): Promise<number> {
|
||||
const { generation } = await store.commitSingleOp({
|
||||
touched: { nouns: [id] },
|
||||
execute: async () => {
|
||||
await storage.saveNounMetadata(id, metadataFixture(version))
|
||||
}
|
||||
})
|
||||
return generation
|
||||
}
|
||||
|
||||
/** A direct transact commit — exactly what a caller bypassing
|
||||
* Brainy.transact()'s flush-first step would issue. */
|
||||
function directCommit(id: string, version: number): Promise<{ generation: number; timestamp: number }> {
|
||||
return store.commitTransaction({
|
||||
touched: { nouns: [id], verbs: [] },
|
||||
execute: async () => {
|
||||
await storage.saveNounMetadata(id, metadataFixture(version))
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
it('PIN 1: refuses a direct commitTransaction() while single-ops are pending, and commits NOTHING', async () => {
|
||||
const g1 = await pendingSingleOp(ID_A, 1)
|
||||
expect(g1).toBe(1)
|
||||
expect(store.committedGeneration()).toBe(0) // nothing flushed to disk yet
|
||||
|
||||
let caught: unknown
|
||||
try {
|
||||
await directCommit(ID_B, 1)
|
||||
expect.unreachable('should have thrown PendingSingleOpsUnflushedError')
|
||||
} catch (err) {
|
||||
caught = err
|
||||
}
|
||||
expect(caught).toBeInstanceOf(PendingSingleOpsUnflushedError)
|
||||
expect((caught as PendingSingleOpsUnflushedError).pendingCount).toBe(1)
|
||||
|
||||
// Nothing committed: the head + committed ranges are unchanged, and the
|
||||
// counter never advanced for the refused attempt (the guard fires before
|
||||
// a generation is even reserved).
|
||||
expect(store.committedGeneration()).toBe(0)
|
||||
expect(store.generation()).toBe(1) // still just the pending single-op's gen
|
||||
expect(await storage.readRawObject(MANIFEST_PATH)).toBeNull()
|
||||
// The guard fires BEFORE a generation is reserved (`gen = ++this.counter`
|
||||
// never runs), so the refused attempt's would-be directory (generation 2,
|
||||
// the next number after the pending single-op's 1) was never created.
|
||||
expect(await storage.listRawObjects(`${GENERATIONS_PREFIX}/2`)).toEqual([])
|
||||
|
||||
// The refused write never touched canonical storage.
|
||||
expect((await storage.readNounRaw(ID_B)).metadata).toBeNull()
|
||||
|
||||
// The pending tier itself is untouched by the refused attempt — flushing
|
||||
// now still commits the ORIGINAL single-op cleanly.
|
||||
await store.flushPendingSingleOps()
|
||||
expect(store.committedGeneration()).toBe(1)
|
||||
const atG0 = await store.resolveAt('noun', ID_A, 0)
|
||||
expect(atG0).toEqual({ source: 'absent' }) // the create sentinel before g1's write
|
||||
})
|
||||
|
||||
it('PIN 2: the same commit succeeds once the pending tier is flushed first', async () => {
|
||||
await pendingSingleOp(ID_A, 1)
|
||||
await expect(directCommit(ID_B, 1)).rejects.toBeInstanceOf(PendingSingleOpsUnflushedError)
|
||||
|
||||
await store.flushPendingSingleOps()
|
||||
expect(store.committedGeneration()).toBe(1)
|
||||
|
||||
const { generation } = await directCommit(ID_B, 1)
|
||||
expect(generation).toBe(2)
|
||||
expect(store.committedGeneration()).toBe(2)
|
||||
expect((await storage.readNounRaw(ID_B)).metadata).toMatchObject({ version: 1 })
|
||||
})
|
||||
})
|
||||
|
||||
describe('Brainy public API — commitTransaction pending-tier guard is behavior-neutral', () => {
|
||||
let brain: Brainy
|
||||
|
||||
beforeEach(async () => {
|
||||
brain = new Brainy(createTestConfig())
|
||||
await brain.init()
|
||||
})
|
||||
afterEach(async () => {
|
||||
await brain.close()
|
||||
})
|
||||
|
||||
it('PIN 3: Brainy.transact() still commits normally over pending single-ops (mirrors generation-chain.test.ts\'s seedX()/bumpX() transact pin)', async () => {
|
||||
const store = (brain as any).generationStore as GenerationStore
|
||||
// Relative, not absolute: under the adopt-at-open default the open-time
|
||||
// baseline backfill takes a generation of its own (see
|
||||
// bounded-chains.test.ts's identical note), so the first user add is not
|
||||
// necessarily generation 1.
|
||||
const baseGen = brain.generation()
|
||||
const baseCommitted = store.committedGeneration()
|
||||
|
||||
const id = await brain.add({
|
||||
data: 'x',
|
||||
type: NounType.Document,
|
||||
subtype: 'note',
|
||||
metadata: { v: 1 },
|
||||
vector: VEC
|
||||
})
|
||||
// The add is a pending single-op generation — NOT yet flushed.
|
||||
expect(brain.generation()).toBe(baseGen + 1)
|
||||
expect(store.committedGeneration()).toBe(baseCommitted)
|
||||
|
||||
// Brainy.transact() flushes the pending tier FIRST (src/brainy.ts:
|
||||
// `await this.generationStore.flushPendingSingleOps()`, immediately
|
||||
// before its `generationStore.commitTransaction()` call), so the guard
|
||||
// never fires on this path — same shape as generation-chain.test.ts's
|
||||
// seedX() (add) → bumpX() (transact update) → generation advances by one.
|
||||
const db = await brain.transact([{ op: 'update', id, metadata: { v: 2 } }])
|
||||
await db.release()
|
||||
|
||||
expect(brain.generation()).toBe(baseGen + 2)
|
||||
expect(store.committedGeneration()).toBe(baseGen + 2) // the flushed add + the transact update
|
||||
const entity = (await brain.get(id)) as any
|
||||
expect(entity.metadata.v).toBe(2)
|
||||
})
|
||||
|
||||
it('PIN 4: reservedGensAsc() stays ascending across a real add+transact+delete workload — point-in-time reads stay correct', async () => {
|
||||
const store = (brain as any).generationStore as GenerationStore
|
||||
const baseGen = brain.generation()
|
||||
const baseCommitted = store.committedGeneration()
|
||||
|
||||
const idX = await brain.add({
|
||||
data: 'x',
|
||||
type: NounType.Document,
|
||||
subtype: 'note',
|
||||
metadata: { v: 1 },
|
||||
vector: VEC
|
||||
})
|
||||
expect(brain.generation()).toBe(baseGen + 1) // pending (un-flushed)
|
||||
|
||||
const idY = await brain.add({
|
||||
data: 'y',
|
||||
type: NounType.Document,
|
||||
subtype: 'note',
|
||||
metadata: { v: 1 },
|
||||
vector: VEC
|
||||
})
|
||||
// Pin right after BOTH adds — before the transact update — so X reads v1
|
||||
// and Y still exists at this pin, unlike the live head after the rest of
|
||||
// the workload runs.
|
||||
const pinAfterBothAdds = brain.generation()
|
||||
expect(pinAfterBothAdds).toBe(baseGen + 2) // ALSO pending — two un-flushed single-ops
|
||||
expect(store.committedGeneration()).toBe(baseCommitted)
|
||||
|
||||
// A transact() flushes baseGen+1 and baseGen+2 first, then commits its
|
||||
// own update as baseGen+3. If committed-vs-pending ordering ever broke,
|
||||
// this is exactly the step that would land a commit ABOVE still-pending
|
||||
// generations.
|
||||
const db = await brain.transact([{ op: 'update', id: idX, metadata: { v: 3 } }])
|
||||
await db.release()
|
||||
expect(brain.generation()).toBe(baseGen + 3)
|
||||
expect(store.committedGeneration()).toBe(baseGen + 3)
|
||||
|
||||
// A single-op delete, pending again (un-flushed).
|
||||
await brain.remove(idY)
|
||||
expect(brain.generation()).toBe(baseGen + 4)
|
||||
|
||||
// A point-in-time read pinned right after the two adds (before the
|
||||
// transact update) must see X's PRE-update value and Y still present.
|
||||
// This is precisely what resolveManyAt/resolveAt get WRONG if committed
|
||||
// and pending generations were ever interleaved out of ascending order.
|
||||
const past = await brain.asOf(pinAfterBothAdds)
|
||||
const xAtPin = (await past.get(idX)) as any
|
||||
expect(xAtPin?.metadata?.v).toBe(1)
|
||||
const yAtPin = (await past.get(idY)) as any
|
||||
expect(yAtPin?.metadata?.v).toBe(1) // not yet removed, as of this pin
|
||||
await past.release()
|
||||
|
||||
// Live state reflects every later write, in the right order.
|
||||
const xNow = (await brain.get(idX)) as any
|
||||
expect(xNow.metadata.v).toBe(3)
|
||||
expect(await brain.get(idY)).toBeNull()
|
||||
})
|
||||
})
|
||||
|
|
@ -4,7 +4,8 @@
|
|||
* config (so it never runs and gives false coverage confidence — the exact drift
|
||||
* that left ~27 test files un-run before 8.0). Every `*.test.ts` must either match
|
||||
* a gate config (`tests/unit/**`, `tests/integration/**`, `*.unit.test.ts`,
|
||||
* `*.integration.test.ts`) or be explicitly listed in MANUAL_ONLY below.
|
||||
* `*.integration.test.ts`, or the perf lane's `tests/configs/vitest.perf.config.ts`
|
||||
* — see PERF_LANE_FILES below) or be explicitly listed in MANUAL_ONLY below.
|
||||
*/
|
||||
import { describe, it, expect } from 'vitest'
|
||||
import { readdirSync } from 'node:fs'
|
||||
|
|
@ -24,10 +25,12 @@ function allTestFiles(dir: string, out: string[] = []): string[] {
|
|||
}
|
||||
|
||||
/**
|
||||
* Test files INTENTIONALLY excluded from the unit/integration gate: benchmarks,
|
||||
* scale/perf measurements, package-size checks, and real-model-load checks. They
|
||||
* are run manually (slow / need real resources), not in CI. Every entry is a
|
||||
* conscious decision — a NEW orphan not listed here fails the guard below.
|
||||
* Test files INTENTIONALLY excluded from every automated gate — conformance
|
||||
* suites invoked directly, and checks that need real resources (network,
|
||||
* unusual scale) no CI lane provides. Wall-clock/scale benchmarks that DO
|
||||
* run automatically belong to the perf lane (PERF_LANE_FILES / inGate
|
||||
* below), not here. Every entry is a conscious decision — a NEW orphan not
|
||||
* listed here fails the guard below.
|
||||
*/
|
||||
const MANUAL_ONLY = new Set<string>([
|
||||
// Conformance suites run as an explicit gate stage (both engines run them
|
||||
|
|
@ -40,15 +43,11 @@ const MANUAL_ONLY = new Set<string>([
|
|||
// The sparse-store cut's shared operator rows (both engines run these):
|
||||
// explicit conformance-gate invocation, like its siblings.
|
||||
'tests/conformance/sparse-store-cut.test.ts',
|
||||
'tests/api/performance-benchmarks.test.ts',
|
||||
// NOT the perf lane: no wall-clock/scale assertion, so it does not belong
|
||||
// in tests/configs/vitest.perf.config.ts's include list — genuinely run
|
||||
// by hand only.
|
||||
'tests/critical-neural-validation.test.ts',
|
||||
'tests/critical-performance-benchmark.test.ts',
|
||||
'tests/model-loading.test.ts',
|
||||
'tests/package-size-breakdown.test.ts',
|
||||
'tests/package-size-limit.test.ts',
|
||||
'tests/performance/graph-scale-performance.test.ts',
|
||||
'tests/performance/triple-intelligence-scale.test.ts',
|
||||
'tests/performance/typeAware.bench.test.ts',
|
||||
// Cross-engine field-addressing conformance suite: pinned bit-for-bit against
|
||||
// the native accelerator's implementation of the SAME contract, and invoked
|
||||
// directly (`npx vitest run tests/conformance/namespace-law.test.ts`), never
|
||||
|
|
@ -59,6 +58,21 @@ const MANUAL_ONLY = new Set<string>([
|
|||
'tests/conformance/namespace-law.test.ts'
|
||||
])
|
||||
|
||||
/**
|
||||
* The perf lane's own gate: `tests/configs/vitest.perf.config.ts`, run by
|
||||
* `npm run test:perf`. Mirrors that config's `include` list — kept in sync
|
||||
* by inspection, the same convention that config uses against the root
|
||||
* gate's exclude list (see its own header comment). A file that runs here
|
||||
* is GATED, not manual: it belongs in this set (or the `tests/performance/`
|
||||
* prefix below), never in MANUAL_ONLY.
|
||||
*/
|
||||
const PERF_LANE_FILES = new Set<string>([
|
||||
'tests/critical-performance-benchmark.test.ts',
|
||||
'tests/api/performance-benchmarks.test.ts',
|
||||
'tests/package-size-limit.test.ts',
|
||||
'tests/model-loading.test.ts'
|
||||
])
|
||||
|
||||
function inGate(rel: string): boolean {
|
||||
return (
|
||||
rel.startsWith('tests/unit/') ||
|
||||
|
|
@ -67,7 +81,11 @@ function inGate(rel: string): boolean {
|
|||
// ('tests/lifecycle/**/*.test.ts'; see tests/lifecycle/README.md).
|
||||
rel.startsWith('tests/lifecycle/') ||
|
||||
rel.endsWith('.unit.test.ts') ||
|
||||
rel.endsWith('.integration.test.ts')
|
||||
rel.endsWith('.integration.test.ts') ||
|
||||
// The perf lane (see PERF_LANE_FILES above) — mirrors
|
||||
// tests/configs/vitest.perf.config.ts's `tests/performance/**` glob.
|
||||
rel.startsWith('tests/performance/') ||
|
||||
PERF_LANE_FILES.has(rel)
|
||||
)
|
||||
}
|
||||
|
||||
|
|
|
|||
165
tests/vfs/vfs-search-path-scope.unit.test.ts
Normal file
165
tests/vfs/vfs-search-path-scope.unit.test.ts
Normal file
|
|
@ -0,0 +1,165 @@
|
|||
/**
|
||||
* @module tests/vfs/vfs-search-path-scope.unit
|
||||
* @description `vfs.search({ path })` scopes with a SERVED filter.
|
||||
*
|
||||
* The scope used to be emitted as `path: { $startsWith }` — an operator that is
|
||||
* not in the filter vocabulary at all, and whose `$`-less spelling the metadata
|
||||
* index refuses by the served-operator law (an equality/range posting index
|
||||
* cannot evaluate a substring without reading every row). Every path-scoped VFS
|
||||
* search threw; none has ever worked on this engine line.
|
||||
*
|
||||
* The scope is now a half-open range over `metadata.path`, which is the VFS's
|
||||
* truth, is indexed on every VFS entity, and is served by the ordered range
|
||||
* operators: `[dir + '/', dir + '0')` — '0' being the code point after '/', so
|
||||
* membership in the range is EXACTLY "carries the prefix `dir/`". The
|
||||
* non-recursive scope is the directory's own identity, `parent`, an equality.
|
||||
*
|
||||
* These pins hold the answer (descendants at every depth, siblings never — the
|
||||
* `/scope-sibling` trap included), the shape (the operators the search emits
|
||||
* are answered by the index's own door, never refused), and the law that the
|
||||
* scope narrows the search BEFORE it runs rather than filtering an over-fetch.
|
||||
*/
|
||||
import { describe, it, expect, beforeAll, afterAll, vi } from 'vitest'
|
||||
import { VirtualFileSystem } from '../../src/vfs/VirtualFileSystem.js'
|
||||
import { Brainy } from '../../src/brainy.js'
|
||||
import { VFSErrorCode } from '../../src/vfs/types.js'
|
||||
|
||||
/** A word every fixture file carries, so the text leg reaches all of them. */
|
||||
const TOKEN = 'quasar'
|
||||
|
||||
describe('vfs.search({ path }) scopes with a served filter', () => {
|
||||
let brain: Brainy
|
||||
let vfs: VirtualFileSystem
|
||||
|
||||
/** In scope for '/scope', at three depths. */
|
||||
const inScope = ['/scope/a.txt', '/scope/sub/b.txt', '/scope/sub/deep/c.txt']
|
||||
/** Out of scope — including the two prefix traps a naive test misses. */
|
||||
const outOfScope = ['/scope-sibling/d.txt', '/scope0/e.txt', '/elsewhere/f.txt', '/g.txt']
|
||||
|
||||
beforeAll(async () => {
|
||||
brain = new Brainy({ requireSubtype: false, storage: { type: 'memory' }, silent: true })
|
||||
await brain.init()
|
||||
vfs = brain.vfs
|
||||
await vfs.init()
|
||||
|
||||
await vfs.mkdir('/scope/sub/deep', { recursive: true })
|
||||
await vfs.mkdir('/scope-sibling', { recursive: true })
|
||||
await vfs.mkdir('/scope0', { recursive: true })
|
||||
await vfs.mkdir('/elsewhere', { recursive: true })
|
||||
|
||||
for (const path of [...inScope, ...outOfScope]) {
|
||||
await vfs.writeFile(path, `${TOKEN} content for ${path}`)
|
||||
}
|
||||
})
|
||||
|
||||
afterAll(async () => {
|
||||
await vfs?.close()
|
||||
await brain?.close()
|
||||
})
|
||||
|
||||
it('includes every descendant depth and excludes every sibling', async () => {
|
||||
const results = await vfs.search(TOKEN, { path: '/scope', limit: 50 })
|
||||
const paths = results.map((r) => r.path).sort()
|
||||
|
||||
expect(paths).toEqual([...inScope].sort())
|
||||
for (const path of outOfScope) expect(paths).not.toContain(path)
|
||||
})
|
||||
|
||||
it('a trailing slash and a doubled slash name the same scope', async () => {
|
||||
const plain = await vfs.search(TOKEN, { path: '/scope', limit: 50 })
|
||||
const trailing = await vfs.search(TOKEN, { path: '/scope/', limit: 50 })
|
||||
const doubled = await vfs.search(TOKEN, { path: '//scope//', limit: 50 })
|
||||
|
||||
const ids = (rs: Array<{ entityId: string }>) => rs.map((r) => r.entityId).sort()
|
||||
expect(ids(trailing)).toEqual(ids(plain))
|
||||
expect(ids(doubled)).toEqual(ids(plain))
|
||||
})
|
||||
|
||||
it('the root scope is every VFS file — it adds no clause to narrow with', async () => {
|
||||
const rooted = await vfs.search(TOKEN, { path: '/', limit: 50 })
|
||||
const unscoped = await vfs.search(TOKEN, { limit: 50 })
|
||||
|
||||
const paths = rooted.map((r) => r.path).sort()
|
||||
expect(paths).toEqual([...inScope, ...outOfScope].sort())
|
||||
expect(paths).toEqual(unscoped.map((r) => r.path).sort())
|
||||
})
|
||||
|
||||
it('recursive: false is the immediate children, not the subtree', async () => {
|
||||
const results = await vfs.search(TOKEN, { path: '/scope', recursive: false, limit: 50 })
|
||||
expect(results.map((r) => r.path)).toEqual(['/scope/a.txt'])
|
||||
})
|
||||
|
||||
it('recursive: false on a path that does not exist refuses by name', async () => {
|
||||
await expect(
|
||||
vfs.search(TOKEN, { path: '/no-such-dir', recursive: false, limit: 50 })
|
||||
).rejects.toMatchObject({ code: VFSErrorCode.ENOENT })
|
||||
})
|
||||
|
||||
it('every operator the search emits is ANSWERED by the index door, never refused', async () => {
|
||||
const index = (brain as any).metadataIndex
|
||||
const emitted: any[] = []
|
||||
const find = vi.spyOn(brain as any, 'find')
|
||||
try {
|
||||
await vfs.search(TOKEN, { path: '/scope', limit: 50 })
|
||||
await vfs.search(TOKEN, { path: '/scope/sub', where: { mimeType: 'text/plain' }, limit: 50 })
|
||||
await vfs.search(TOKEN, { path: '/scope', recursive: false, limit: 50 })
|
||||
await vfs.search(TOKEN, { path: '/', limit: 50 })
|
||||
for (const call of find.mock.calls) emitted.push((call[0] as any).where)
|
||||
} finally {
|
||||
find.mockRestore()
|
||||
}
|
||||
|
||||
expect(emitted).toHaveLength(4)
|
||||
for (const where of emitted) {
|
||||
// The door itself is the judge: an operator outside the served set is
|
||||
// REFUSED here (BrainyError INVALID_QUERY), never answered.
|
||||
await expect(index.getIdsForFilter(where)).resolves.toBeInstanceOf(Array)
|
||||
}
|
||||
|
||||
// And the scope really is a range on the path — the shape this fix chose.
|
||||
expect(emitted[0].path).toEqual({ gte: '/scope/', lt: '/scope0' })
|
||||
expect(emitted[3].path).toBeUndefined()
|
||||
})
|
||||
|
||||
it('the scope narrows the search before it runs — no over-fetch to filter', async () => {
|
||||
const index = (brain as any).metadataIndex
|
||||
const filter = vi.spyOn(index, 'getIdsForFilter')
|
||||
let universe: string[] = []
|
||||
try {
|
||||
await vfs.search(TOKEN, { path: '/scope', limit: 50 })
|
||||
// The search's own call — the one carrying the scope. (Path resolution
|
||||
// asks this same door for the root, before the search is built.)
|
||||
const scoped = filter.mock.calls.findIndex(
|
||||
(c) => (c[0] as any)?.path?.gte === '/scope/'
|
||||
)
|
||||
expect(scoped).toBeGreaterThanOrEqual(0)
|
||||
universe = (await filter.mock.results[scoped].value) as string[]
|
||||
} finally {
|
||||
filter.mockRestore()
|
||||
}
|
||||
|
||||
// The id universe the index resolved for the search is already the scope:
|
||||
// three files, and not one row from outside it.
|
||||
const rows = await brain.batchGet(universe)
|
||||
const paths = [...rows.values()].map((e: any) => e.metadata.path).sort()
|
||||
expect(paths).toEqual([...inScope].sort())
|
||||
})
|
||||
|
||||
it('the range answers the same ids as walking the tree', async () => {
|
||||
// The path is the truth and the Contains edges are its projection; a scope
|
||||
// read from the truth must agree with one walked over the projection.
|
||||
const walked: string[] = []
|
||||
const walk = async (dir: string): Promise<void> => {
|
||||
for (const name of await vfs.readdir(dir)) {
|
||||
const child = dir === '/' ? `/${name}` : `${dir}/${name}`
|
||||
const stat = await vfs.stat(child)
|
||||
if (stat.isDirectory()) await walk(child)
|
||||
else walked.push(child)
|
||||
}
|
||||
}
|
||||
await walk('/scope')
|
||||
|
||||
const searched = await vfs.search(TOKEN, { path: '/scope', limit: 50 })
|
||||
expect(searched.map((r) => r.path).sort()).toEqual(walked.sort())
|
||||
})
|
||||
})
|
||||
|
|
@ -2,9 +2,16 @@ import { defineConfig } from 'vitest/config'
|
|||
|
||||
/**
|
||||
* Vitest Configuration - Optimized for Memory-Intensive Tests
|
||||
*
|
||||
*
|
||||
* Handles ONNX transformer model testing (4-8GB memory requirement)
|
||||
* Based on 2024-2025 best practices
|
||||
*
|
||||
* THE CORRECTNESS GATE: this is the config a bare `vitest run` (no
|
||||
* `--config` flag) picks up — the delta gate and CI both invoke it that
|
||||
* way. See CONTRIBUTING.md's "Test gate" section for the full picture.
|
||||
* Wall-clock/scale benchmarks and tests whose outcome depends on the host
|
||||
* machine or network rather than the code are excluded below and run on
|
||||
* demand instead, in their own slot: `npm run test:perf`.
|
||||
*/
|
||||
export default defineConfig({
|
||||
test: {
|
||||
|
|
@ -38,7 +45,29 @@ export default defineConfig({
|
|||
'node_modules/**',
|
||||
'dist/**',
|
||||
'scripts/**',
|
||||
'**/*.browser.test.ts'
|
||||
'**/*.browser.test.ts',
|
||||
|
||||
// Wall-clock/scale benchmark family — timing assertions and scale
|
||||
// sweeps whose pass/fail depends on the host machine's speed, not on
|
||||
// the code. Whole files only (a file that mixes correctness describes
|
||||
// with a perf describe stays in the gate). Run on demand via
|
||||
// `npm run test:perf`, which targets exactly this list.
|
||||
'tests/performance/**',
|
||||
'tests/critical-performance-benchmark.test.ts',
|
||||
'tests/api/performance-benchmarks.test.ts',
|
||||
|
||||
// Environment-dependent by construction, not timing-based:
|
||||
// package-size-limit shells out to the `npm` CLI (not guaranteed
|
||||
// present — the functional gate lane is Bun-only host-mode with no
|
||||
// Node.js runtime) and parses npm-version-specific `npm pack` notice
|
||||
// text; model-loading's "Real Model Download Integration" case makes
|
||||
// a genuine, unmocked network call to HuggingFace (its own header
|
||||
// says "Uses REAL transformer models - NO MOCKING"), and the whole
|
||||
// file imports `../src/embeddings/model-manager.js`, which no longer
|
||||
// exists anywhere under src/ — neither belongs in a gate that must be
|
||||
// deterministic.
|
||||
'tests/package-size-limit.test.ts',
|
||||
'tests/model-loading.test.ts'
|
||||
],
|
||||
|
||||
// REPORTERS: Dot for CI, verbose for local
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue