feat: multi-process safety + read-only inspector mode

Filesystem storage now enforces single-writer, many-reader semantics.
A second writer on the same data directory throws at init time with the
holder's PID, hostname, and heartbeat — replacing the previous silent
stale-reads failure mode.

- New: `Brainy.openReadOnly()` — coexists with a live writer, every
  mutation throws clearly.
- New: writer lock at `<rootDir>/locks/_writer.lock` with 10s heartbeat
  and stale-detection (PID liveness + heartbeat freshness).
- New: cross-process flush-request RPC (filesystem-based, no signals)
  so inspectors can force fresh state on demand.
- New: `brain.stats()`, `brain.explain(findParams)`, `brain.health()`
  for operator-facing introspection.
- New: `brainy inspect` CLI with 13 subcommands (stats, find, get,
  relations, explain, health, sample, fields, dump, watch, backup,
  repair, diff), all read-only by default.
- Same-PID re-opens allowed with a warning (preserves test "simulate
  restart" patterns).
- Storage instances passed directly via `storage: new MemoryStorage()`
  are now honoured instead of silently falling through to the
  filesystem auto-detect path.

Brainy + Cortex compose under this model — the lock covers both because
they share `rootDir`, Cortex segments are immutable mmap files, and
MANIFEST updates use atomic-rename.
This commit is contained in:
David Snelling 2026-05-15 11:25:05 -07:00
parent 1bc6a430c7
commit 4fcdc0fef3
12 changed files with 2342 additions and 7 deletions

View file

@ -18,6 +18,7 @@ import { nlpCommands } from './commands/nlp.js'
import { insightsCommands } from './commands/insights.js'
import { importCommands } from './commands/import.js'
import { cowCommands } from './commands/cow.js'
import { inspectCommands } from './commands/inspect.js'
import { readFileSync } from 'fs'
import { fileURLToPath } from 'url'
import { dirname, join } from 'path'
@ -551,6 +552,143 @@ program
.description('Show detailed database statistics')
.action(dataCommands.stats)
// ===== Inspect Commands =====
// Out-of-process diagnostics. Every subcommand opens the store via
// Brainy.openReadOnly() so a live writer can keep running. `--fresh`
// (default) asks the writer to flush before opening.
program
.command('inspect')
.description('🔍 Out-of-process diagnostics on a Brainy data directory')
.addCommand(
new Command('stats')
.argument('<path>', 'Path to the Brainy data directory')
.description('Counts, mode, indexed fields, writer lock info')
.option('--no-fresh', 'Skip the writer flush request (faster, but state may be slightly stale)')
.option('--json', 'Output as JSON')
.option('--pretty', 'Pretty-print JSON')
.action((path, options) => inspectCommands.stats(path, options))
)
.addCommand(
new Command('find')
.argument('<path>', 'Path to the Brainy data directory')
.description('Find entities matching a where-clause filter')
.option('--type <type>', 'Filter by entity type')
.option('--where <json>', 'Metadata filter (JSON object)')
.option('--limit <n>', 'Max results', '20')
.option('--offset <n>', 'Skip N results (pagination)')
.option('--no-fresh', 'Skip the writer flush request')
.option('--json', 'Output as JSON')
.option('--pretty', 'Pretty-print JSON')
.action((path, options) => inspectCommands.find(path, options))
)
.addCommand(
new Command('get')
.argument('<path>', 'Path to the Brainy data directory')
.argument('<id>', 'Entity ID')
.description('Fetch a single entity by ID')
.option('--no-fresh', 'Skip the writer flush request')
.option('--json', 'Output as JSON')
.option('--pretty', 'Pretty-print JSON')
.action((path, id, options) => inspectCommands.get(path, id, options))
)
.addCommand(
new Command('relations')
.argument('<path>', 'Path to the Brainy data directory')
.argument('<id>', 'Entity ID')
.description('Show inbound/outbound relationships for an entity')
.option('--direction <dir>', 'in | out | both', 'both')
.option('--type <type>', 'Filter by verb type')
.option('--limit <n>', 'Max relationships', '50')
.option('--no-fresh', 'Skip the writer flush request')
.option('--json', 'Output as JSON')
.option('--pretty', 'Pretty-print JSON')
.action((path, id, options) => inspectCommands.relations(path, id, options))
)
.addCommand(
new Command('explain')
.argument('<path>', 'Path to the Brainy data directory')
.description('Show which index path will serve each where-clause field (column-store / sparse / none)')
.option('--type <type>', 'Filter by entity type')
.option('--where <json>', 'Metadata filter to plan (JSON object)')
.option('--no-fresh', 'Skip the writer flush request')
.option('--json', 'Output as JSON')
.option('--pretty', 'Pretty-print JSON')
.action((path, options) => inspectCommands.explain(path, options))
)
.addCommand(
new Command('health')
.argument('<path>', 'Path to the Brainy data directory')
.description('Run invariant checks (index parity, field registry, _seeded sweep, writer heartbeat)')
.option('--no-fresh', 'Skip the writer flush request')
.option('--json', 'Output as JSON')
.option('--pretty', 'Pretty-print JSON')
.action((path, options) => inspectCommands.health(path, options))
)
.addCommand(
new Command('sample')
.argument('<path>', 'Path to the Brainy data directory')
.description('Random N-entity sample')
.option('--type <type>', 'Filter by entity type')
.option('--n <n>', 'Sample size', '10')
.option('--no-fresh', 'Skip the writer flush request')
.option('--json', 'Output as JSON')
.option('--pretty', 'Pretty-print JSON')
.action((path, options) => inspectCommands.sample(path, options))
)
.addCommand(
new Command('fields')
.argument('<path>', 'Path to the Brainy data directory')
.description('List indexed metadata fields')
.option('--no-fresh', 'Skip the writer flush request')
.option('--json', 'Output as JSON')
.option('--pretty', 'Pretty-print JSON')
.action((path, options) => inspectCommands.fields(path, options))
)
.addCommand(
new Command('dump')
.argument('<path>', 'Path to the Brainy data directory')
.description('Dump all entities of a type as JSONL (one per line) to stdout')
.option('--type <type>', 'Filter by entity type')
.option('--batch <n>', 'Page size', '500')
.option('--no-fresh', 'Skip the writer flush request')
.action((path, options) => inspectCommands.dump(path, options))
)
.addCommand(
new Command('watch')
.argument('<path>', 'Path to the Brainy data directory')
.description('Tail newly-written entities')
.option('--type <type>', 'Filter by entity type')
.option('--interval <ms>', 'Poll interval', '1000')
.action((path, options) => inspectCommands.watch(path, options))
)
.addCommand(
new Command('backup')
.argument('<path>', 'Path to the Brainy data directory')
.argument('<dest>', 'Destination tarball')
.description('Atomic flush-then-tar snapshot of the data directory')
.action((path, dest, options) => inspectCommands.backup(path, dest, options))
)
.addCommand(
new Command('repair')
.argument('<path>', 'Path to the Brainy data directory')
.description('Rebuild indexes from raw storage (writer-mode — stop the live writer first)')
.option('--force', 'Override the writer lock if you are sure no other writer is running')
.option('--json', 'Output as JSON')
.option('--pretty', 'Pretty-print JSON')
.action((path, options) => inspectCommands.repair(path, options))
)
.addCommand(
new Command('diff')
.argument('<pathA>', 'First Brainy data directory')
.argument('<pathB>', 'Second Brainy data directory')
.description('Compare counts and a sample of entity IDs between two stores')
.option('--sample <n>', 'Sample size per side', '100')
.option('--json', 'Output as JSON')
.option('--pretty', 'Pretty-print JSON')
.action((pathA, pathB, options) => inspectCommands.diff(pathA, pathB, options))
)
// ===== NLP Commands =====
program