perf(8.0): drop O(N)-resident id-keyed storage caches; source counts from the record

The storage layer held five id-keyed Maps (nounTypeByIdCache + the subtype and
visibility caches) resident for the writer's lifetime — one entry per live
entity, present even when a native provider is registered (it swaps the indexes,
not the storage adapter). At billion scale that is hundreds of GB of writer RAM,
breaking the "nothing O(N)-resident" invariant.

Eliminate all five. Per-type/subtype counts are attributed at the metadata-save
site where the type is already in hand, and the delete path re-derives the prior
values by reading the record before removing it. O(1) writer RAM on both the
native and standalone paths; the latent rebuildTypeCounts staleness edge is gone
by construction. Unit 1718 + integration 607 green.
This commit is contained in:
David Snelling 2026-06-29 16:05:03 -07:00
parent 855298ab79
commit b6beb7f96a

View file

@ -16,7 +16,6 @@ import {
HNSWVerbWithMetadata, HNSWVerbWithMetadata,
StatisticsData StatisticsData
} from '../coreTypes.js' } from '../coreTypes.js'
import type { EntityVisibility } from '../coreTypes.js'
import { BaseStorageAdapter } from './adapters/baseStorageAdapter.js' import { BaseStorageAdapter } from './adapters/baseStorageAdapter.js'
import { validateNounType, validateVerbType } from '../utils/typeValidation.js' import { validateNounType, validateVerbType } from '../utils/typeValidation.js'
import { import {
@ -208,18 +207,6 @@ function getVerbMetadataPath(id: string): string {
return `entities/verbs/${shard}/${id}/metadata.json` return `entities/verbs/${shard}/${id}/metadata.json`
} }
/**
* Extract the entity id from an ID-first metadata path. Both noun and verb
* metadata live at `entities/<kind>/<shard>/<id>/metadata.json`, so the id is
* the second-to-last path segment.
* @param path - A metadata.json path produced by `getNounMetadataPath` / `getVerbMetadataPath`.
* @returns The id segment, or `undefined` if the path doesn't have the expected shape.
*/
function idFromMetadataPath(path: string): string | undefined {
const segments = path.split('/')
return segments[segments.length - 2]
}
/** /**
* Optional count capabilities probed via duck typing by getNouns()/getVerbs(). * Optional count capabilities probed via duck typing by getNouns()/getVerbs().
* Adapters with a native O(1) count API may implement these; they are not part * Adapters with a native O(1) count API may implement these; they are not part
@ -322,54 +309,16 @@ export abstract class BaseStorage extends BaseStorageAdapter {
*/ */
protected verbSubtypeCountsByType = new Map<number, Map<string, number>>() protected verbSubtypeCountsByType = new Map<number, Map<string, number>>()
/** // Count attribution (type / subtype / visibility) is sourced directly from the
* In-memory map from noun id its `noun` (NounType) value, populated when // canonical metadata RECORD, never from id-keyed in-memory caches. The metadata
* `saveNounMetadata_internal()` runs. `saveNoun_internal()` consults this // save/delete paths already read the prior record (`existingMetadata` on write,
* to attribute the entity to the correct slot in `nounCountsByType` so the // read-before-delete on remove), so the entity's `noun`/`verb` type, `subtype`,
* persisted `_system/type-statistics.json` is honest. // and `visibility` are in hand exactly where a count must change — there is no
* // need for a parallel O(N) `id → type/subtype/visibility` map resident on the
* History: a previous cache was removed in commit `42ae5be` and replaced // writer. The five such caches that used to live here were removed in the 8.0
* with a hardcoded `return 'thing'` from `getNounType()`, which silently // billion-scale RAM pass; `nounCountsByType` / `verbCountsByType` (fixed
* poisoned the on-disk type statistics. This cache restores the correct // type-indexed Uint32Arrays) and `subtypeCountsByType` / `verbSubtypeCountsByType`
* behavior. Memory footprint is one Map entry per live noun id (typically // (bounded by distinct subtype labels) remain because they are NOT id-keyed.
* tens of bytes each); the writer process keeps it for the duration of
* its lifetime and prunes on delete.
*/
protected nounTypeByIdCache = new Map<string, NounType>()
/**
* In-memory map from noun id its `subtype` string (when set). Parallel to
* `nounTypeByIdCache`. Lets `deleteNounMetadata()` decrement the correct
* subtype bucket without re-reading metadata. Sparse only ids with a
* non-empty subtype get an entry.
*/
protected nounSubtypeByIdCache = new Map<string, string>()
/**
* In-memory map from verb id `{ verb, subtype }` pair. Verb-side mirror of
* `nounTypeByIdCache` + `nounSubtypeByIdCache`. Lets `deleteVerbMetadata` and
* `updateRelation` decrement the right bucket without a re-read of metadata.
* Sparse only ids with a non-empty subtype get an entry.
*/
protected verbSubtypeByIdCache = new Map<string, { verb: VerbType; subtype: string }>()
// Total: 676 bytes (99.2% reduction vs Map-based tracking)
/**
* In-memory map from noun id its non-public `visibility` tier ('internal' |
* 'system'). Parallel to `nounTypeByIdCache`. Lets `saveNoun_internal()`
* (which has no metadata access in its signature) skip the user-facing
* `nounCountsByType` increment for hidden entities, and lets
* `deleteNounMetadata()` skip the matching decrement keeping the counts
* symmetric. Sparse by design: public entities (the common case) get NO
* entry, so the absence of an id means "public, counted".
*/
protected nounVisibilityByIdCache = new Map<string, EntityVisibility>()
/**
* In-memory map from verb id its non-public `visibility` tier. Verb-side
* mirror of `nounVisibilityByIdCache`. Sparse public edges get no entry.
*/
protected verbVisibilityByIdCache = new Map<string, EntityVisibility>()
// Type caches REMOVED - ID-first paths eliminate need for type lookups! // Type caches REMOVED - ID-first paths eliminate need for type lookups!
// With ID-first architecture, we construct paths directly from IDs: {SHARD}/{ID}/metadata.json // With ID-first architecture, we construct paths directly from IDs: {SHARD}/{ID}/metadata.json
@ -1037,15 +986,13 @@ export abstract class BaseStorage extends BaseStorageAdapter {
/** /**
* Reset and reload every piece of adapter-internal derived state after the * Reset and reload every piece of adapter-internal derived state after the
* underlying objects changed wholesale (restore-from-snapshot): the * underlying objects changed wholesale (restore-from-snapshot): the
* write-through cache, the idtype/subtype caches, type/subtype statistics, * write-through cache, type/subtype statistics, total counts, and the
* total counts, and the graph-index singleton (invalidated so the next * graph-index singleton (invalidated so the next accessor rebuilds from the
* accessor rebuilds from the restored verbs). * restored verbs). Count attribution reads the canonical metadata record, so
* there are no id-keyed caches to clear here.
*/ */
protected async reloadDerivedState(): Promise<void> { protected async reloadDerivedState(): Promise<void> {
this.clearWriteCache() this.clearWriteCache()
this.nounTypeByIdCache.clear()
this.nounSubtypeByIdCache.clear()
this.verbSubtypeByIdCache.clear()
this.nounCountsByType.fill(0) this.nounCountsByType.fill(0)
this.verbCountsByType.fill(0) this.verbCountsByType.fill(0)
this.subtypeCountsByType.clear() this.subtypeCountsByType.clear()
@ -2633,19 +2580,16 @@ export abstract class BaseStorage extends BaseStorageAdapter {
// Save the metadata (write-cache coherent canonical write) // Save the metadata (write-cache coherent canonical write)
await this.writeCanonicalObject(path, metadata) await this.writeCanonicalObject(path, metadata)
// Record the id→type mapping so `saveNoun_internal()` (which receives only
// an HNSWNoun and has no metadata access in its signature) can attribute
// the entity to the right slot in `nounCountsByType`. This is what makes
// `_system/type-statistics.json` honest. Updates the cache even for
// existing entities so a type change via `update()` is reflected.
if (metadata.noun) {
this.nounTypeByIdCache.set(id, metadata.noun as NounType)
}
// Track subtype changes: on type or subtype change via update(), decrement // Track subtype changes: on type or subtype change via update(), decrement
// the prior bucket before incrementing the new one. Symmetric with the // the prior bucket before incrementing the new one. The prior (type, subtype)
// delete-path decrement in `deleteNounMetadata()`. // comes straight from the canonical record (`existingMetadata`, already loaded
const priorSubtype = this.nounSubtypeByIdCache.get(id) // above) — there is no id-keyed subtype cache. Symmetric with the delete-path
// decrement in `deleteNounMetadata()`.
const priorSubtype = isNew
? undefined
: (typeof existingMetadata?.subtype === 'string' && (existingMetadata.subtype as string).length > 0
? (existingMetadata.subtype as string)
: undefined)
const priorTypeForSubtype = isNew ? undefined : (existingMetadata?.noun as NounType | undefined) const priorTypeForSubtype = isNew ? undefined : (existingMetadata?.noun as NounType | undefined)
const newSubtype = typeof metadata.subtype === 'string' && metadata.subtype.length > 0 const newSubtype = typeof metadata.subtype === 'string' && metadata.subtype.length > 0
? metadata.subtype as string ? metadata.subtype as string
@ -2657,24 +2601,14 @@ export abstract class BaseStorage extends BaseStorageAdapter {
} }
if (newSubtype && newType && (isNew || priorSubtype !== newSubtype || priorTypeForSubtype !== newType)) { if (newSubtype && newType && (isNew || priorSubtype !== newSubtype || priorTypeForSubtype !== newType)) {
this.incrementSubtypeCount(newType, newSubtype) this.incrementSubtypeCount(newType, newSubtype)
this.nounSubtypeByIdCache.set(id, newSubtype)
} else if (!newSubtype && priorSubtype) {
// Subtype cleared by an update — drop the cache entry (decrement already done above)
this.nounSubtypeByIdCache.delete(id)
} }
// Visibility (8.0): only public entities count toward the user-facing totals. // Visibility (8.0): only public entities count toward the user-facing totals.
// Warm `nounVisibilityByIdCache` so `saveNoun_internal()` (no metadata access) // The gate reads `metadata.visibility` (new) and `existingMetadata?.visibility`
// can gate the `nounCountsByType` increment, and `deleteNounMetadata()` can gate // (prior) directly off the record — no id-keyed visibility cache.
// the matching decrement. Sparse — public ids get no entry.
const newVisibility = metadata.visibility const newVisibility = metadata.visibility
const wasCounted = isNew ? false : isCountedVisibility(existingMetadata?.visibility) const wasCounted = isNew ? false : isCountedVisibility(existingMetadata?.visibility)
const isCounted = isCountedVisibility(newVisibility) const isCounted = isCountedVisibility(newVisibility)
if (isCountedVisibility(newVisibility)) {
this.nounVisibilityByIdCache.delete(id)
} else {
this.nounVisibilityByIdCache.set(id, newVisibility as EntityVisibility)
}
// CRITICAL FIX: Increment count for new entities // CRITICAL FIX: Increment count for new entities
// This runs AFTER metadata is saved, guaranteeing type information is available // This runs AFTER metadata is saved, guaranteeing type information is available
@ -2688,11 +2622,21 @@ export abstract class BaseStorage extends BaseStorageAdapter {
// used to be bumped unconditionally in saveNoun_internal(), but the HNSW index // used to be bumped unconditionally in saveNoun_internal(), but the HNSW index
// re-saves a node on every neighbor-link change, so that inflated the per-type // re-saves a node on every neighbor-link change, so that inflated the per-type
// counts with graph connectivity (e.g. 8 documents could read as 44). // counts with graph connectivity (e.g. 8 documents could read as 44).
this.nounCountsByType[TypeUtils.getNounIndex(metadata.noun as NounType)]++ const typeIdx = TypeUtils.getNounIndex(metadata.noun as NounType)
this.nounCountsByType[typeIdx]++
// Persist counts asynchronously (fire and forget) // Persist counts asynchronously (fire and forget)
this.scheduleCountPersist().catch(() => { this.scheduleCountPersist().catch(() => {
// Ignore persist errors - will retry on next operation // Ignore persist errors - will retry on next operation
}) })
// Persist type-statistics on the first entity of a type and every 100th
// thereafter. This trigger used to live in saveNoun_internal(), which had to
// call getNounType() purely to recover the type index; sourcing the type from
// the metadata record here keeps the hot vector-save path free of any type
// lookup. The "only when counted" half of the heuristic holds by construction
// inside this branch.
if (this.nounCountsByType[typeIdx] === 1 || this.nounCountsByType[typeIdx] % 100 === 0) {
await this.saveTypeStatistics()
}
} else if (!isNew && metadata.noun && wasCounted !== isCounted) { } else if (!isNew && metadata.noun && wasCounted !== isCounted) {
// Visibility flipped on update(): move the entity in/out of the user-facing // Visibility flipped on update(): move the entity in/out of the user-facing
// total (counts.json / getNounCount()) AND the per-type counter together, so // total (counts.json / getNounCount()) AND the per-type counter together, so
@ -2701,6 +2645,12 @@ export abstract class BaseStorage extends BaseStorageAdapter {
if (isCounted) { if (isCounted) {
this.incrementEntityCount(metadata.noun) this.incrementEntityCount(metadata.noun)
this.nounCountsByType[typeIdx]++ this.nounCountsByType[typeIdx]++
// Same cadence-gated type-statistics persist as the fresh-add branch — only
// fires when the entity is now counted (public), matching the original
// `counted && (count === 1 || count % 100 === 0)` heuristic.
if (this.nounCountsByType[typeIdx] === 1 || this.nounCountsByType[typeIdx] % 100 === 0) {
await this.saveTypeStatistics()
}
} else { } else {
this.decrementEntityCount(metadata.noun) this.decrementEntityCount(metadata.noun)
if (this.nounCountsByType[typeIdx] > 0) this.nounCountsByType[typeIdx]-- if (this.nounCountsByType[typeIdx] > 0) this.nounCountsByType[typeIdx]--
@ -3027,21 +2977,20 @@ export abstract class BaseStorage extends BaseStorageAdapter {
public async deleteNounMetadata(id: string): Promise<void> { public async deleteNounMetadata(id: string): Promise<void> {
await this.ensureInitialized() await this.ensureInitialized()
// Direct O(1) delete with ID-first path // Direct O(1) delete with ID-first path. Read the canonical record BEFORE
// removing it: the per-type and subtype decrements are sourced from the
// entity's own metadata (`noun` type, `subtype`, `visibility`) rather than an
// id-keyed cache, keeping type-statistics honest across deletes — symmetric
// with the increments in `saveNounMetadata_internal()`.
const path = getNounMetadataPath(id) const path = getNounMetadataPath(id)
const record = await this.readCanonicalObject(path)
await this.deleteCanonicalObject(path) await this.deleteCanonicalObject(path)
// Prune the id→type cache so a future re-add of the same id (e.g. churn const priorType = record?.noun as NounType | undefined
// during tests) doesn't see a stale type. Lookup the prior type and
// decrement `nounCountsByType` so type-statistics stay honest across
// deletes; symmetric with the increment in `saveNounMetadata_internal()`.
const priorType = this.nounTypeByIdCache.get(id)
// 8.0 visibility: an internal/system entity was never added to `nounCountsByType` // 8.0 visibility: an internal/system entity was never added to `nounCountsByType`
// (gated in `saveNoun_internal()`), so it must not be decremented here either. // (gated in `saveNounMetadata_internal()`), so it must not be decremented here either.
const priorCounted = isCountedVisibility(this.nounVisibilityByIdCache.get(id)) const priorCounted = isCountedVisibility(record?.visibility)
this.nounVisibilityByIdCache.delete(id)
if (priorType) { if (priorType) {
this.nounTypeByIdCache.delete(id)
if (priorCounted) { if (priorCounted) {
const idx = TypeUtils.getNounIndex(priorType) const idx = TypeUtils.getNounIndex(priorType)
if (this.nounCountsByType[idx] > 0) { if (this.nounCountsByType[idx] > 0) {
@ -3049,10 +2998,11 @@ export abstract class BaseStorage extends BaseStorageAdapter {
} }
} }
// Symmetric subtype decrement // Symmetric subtype decrement — same non-empty-string guard as the write path.
const priorSubtype = this.nounSubtypeByIdCache.get(id) const priorSubtype = typeof record?.subtype === 'string' && (record.subtype as string).length > 0
? (record.subtype as string)
: undefined
if (priorSubtype) { if (priorSubtype) {
this.nounSubtypeByIdCache.delete(id)
this.decrementSubtypeCount(priorType, priorSubtype) this.decrementSubtypeCount(priorType, priorSubtype)
} }
} }
@ -3108,39 +3058,31 @@ export abstract class BaseStorage extends BaseStorageAdapter {
// Save the metadata (write-cache coherent canonical write) // Save the metadata (write-cache coherent canonical write)
await this.writeCanonicalObject(path, metadata) await this.writeCanonicalObject(path, metadata)
// Cache verb type for faster lookups
// Track verb subtype changes: on type or subtype change via updateRelation(), // Track verb subtype changes: on type or subtype change via updateRelation(),
// decrement the prior bucket before incrementing the new one. Symmetric with // decrement the prior bucket before incrementing the new one. The prior
// the delete-path decrement in `deleteVerbMetadata()`. // (verb, subtype) is read straight from the canonical record (`existingMetadata`,
const priorEntry = this.verbSubtypeByIdCache.get(id) // loaded above) — there is no id-keyed verb-subtype cache. Symmetric with the
// delete-path decrement in `deleteVerbMetadata()`.
const priorVerbForSubtype = isNew ? undefined : (existingMetadata?.verb as VerbType | undefined) const priorVerbForSubtype = isNew ? undefined : (existingMetadata?.verb as VerbType | undefined)
const priorSubtype = isNew
? undefined
: (typeof existingMetadata?.subtype === 'string' && (existingMetadata.subtype as string).length > 0
? (existingMetadata.subtype as string)
: undefined)
const newSubtype = typeof metadata.subtype === 'string' && metadata.subtype.length > 0 const newSubtype = typeof metadata.subtype === 'string' && metadata.subtype.length > 0
? metadata.subtype as string ? metadata.subtype as string
: undefined : undefined
if (priorEntry && (priorEntry.subtype !== newSubtype || priorEntry.verb !== verbType)) { if (priorSubtype && priorVerbForSubtype && (priorSubtype !== newSubtype || priorVerbForSubtype !== verbType)) {
this.decrementVerbSubtypeCount(priorEntry.verb, priorEntry.subtype) this.decrementVerbSubtypeCount(priorVerbForSubtype, priorSubtype)
} else if (!priorEntry && priorVerbForSubtype && !isNew) {
// Edge case: cache miss but metadata existed with a subtype (e.g. reader process startup).
const priorSubFromMeta = existingMetadata?.subtype as string | undefined
if (priorSubFromMeta && (priorSubFromMeta !== newSubtype || priorVerbForSubtype !== verbType)) {
this.decrementVerbSubtypeCount(priorVerbForSubtype, priorSubFromMeta)
} }
} if (newSubtype && (isNew || priorSubtype !== newSubtype || priorVerbForSubtype !== verbType)) {
if (newSubtype) {
if (!priorEntry || priorEntry.subtype !== newSubtype || priorEntry.verb !== verbType) {
this.incrementVerbSubtypeCount(verbType, newSubtype) this.incrementVerbSubtypeCount(verbType, newSubtype)
} }
this.verbSubtypeByIdCache.set(id, { verb: verbType, subtype: newSubtype })
} else if (priorEntry) {
// Subtype cleared by an update — drop the cache entry (decrement done above)
this.verbSubtypeByIdCache.delete(id)
}
// Visibility (8.0): verb mirror of the noun count gating. Warm // Visibility (8.0): verb mirror of the noun count gating. The gate reads
// `verbVisibilityByIdCache` so `deleteVerbMetadata()` can prune it. Sparse — // `metadata.visibility` (new) and `existingMetadata?.visibility` (prior)
// public edges get no entry. // directly off the record — no id-keyed visibility cache.
// //
// NOTE on `verbCountsByType`: unlike the noun path, `saveVerb_internal()` runs // NOTE on `verbCountsByType`: unlike the noun path, `saveVerb_internal()` runs
// BEFORE this method (relate() saves the verb vector first) and has already done // BEFORE this method (relate() saves the verb vector first) and has already done
@ -3150,11 +3092,6 @@ export abstract class BaseStorage extends BaseStorageAdapter {
const newVisibility = metadata.visibility const newVisibility = metadata.visibility
const wasCounted = isNew ? false : isCountedVisibility(existingMetadata?.visibility) const wasCounted = isNew ? false : isCountedVisibility(existingMetadata?.visibility)
const isCounted = isCountedVisibility(newVisibility) const isCounted = isCountedVisibility(newVisibility)
if (isCounted) {
this.verbVisibilityByIdCache.delete(id)
} else {
this.verbVisibilityByIdCache.set(id, newVisibility as EntityVisibility)
}
const verbTypeIdx = TypeUtils.getVerbIndex(verbType) const verbTypeIdx = TypeUtils.getVerbIndex(verbType)
// CRITICAL FIX: Increment verb count for new relationships // CRITICAL FIX: Increment verb count for new relationships
@ -3212,19 +3149,23 @@ export abstract class BaseStorage extends BaseStorageAdapter {
public async deleteVerbMetadata(id: string): Promise<void> { public async deleteVerbMetadata(id: string): Promise<void> {
await this.ensureInitialized() await this.ensureInitialized()
// Direct O(1) delete with ID-first path // Direct O(1) delete with ID-first path. Read the canonical record BEFORE
// removing it so the verb-subtype decrement is sourced from the edge's own
// metadata (`verb` type + `subtype`) rather than an id-keyed cache — symmetric
// with the increment in `saveVerbMetadata_internal()`. Verb deletes do not
// touch `verbCountsByType` in this path (matching prior behavior), so no
// visibility read is needed here.
const path = getVerbMetadataPath(id) const path = getVerbMetadataPath(id)
const record = await this.readCanonicalObject(path)
await this.deleteCanonicalObject(path) await this.deleteCanonicalObject(path)
// Symmetric verb subtype decrement const priorVerb = record?.verb as VerbType | undefined
const priorEntry = this.verbSubtypeByIdCache.get(id) const priorSubtype = typeof record?.subtype === 'string' && (record.subtype as string).length > 0
if (priorEntry) { ? (record.subtype as string)
this.verbSubtypeByIdCache.delete(id) : undefined
this.decrementVerbSubtypeCount(priorEntry.verb, priorEntry.subtype) if (priorVerb && priorSubtype) {
this.decrementVerbSubtypeCount(priorVerb, priorSubtype)
} }
// 8.0 visibility: prune the cache entry (verb deletes don't touch verbCountsByType
// in this path, mirroring the existing verb-subtype delete semantics).
this.verbVisibilityByIdCache.delete(id)
// 8.0 MVCC: entity-visible write — advance the generation watermark // 8.0 MVCC: entity-visible write — advance the generation watermark
// (suppressed inside transact batches by the generation store). // (suppressed inside transact batches by the generation store).
@ -3421,7 +3362,6 @@ export abstract class BaseStorage extends BaseStorageAdapter {
public async rebuildSubtypeCounts(): Promise<void> { public async rebuildSubtypeCounts(): Promise<void> {
prodLog.info('[BaseStorage] Rebuilding subtype counts from storage...') prodLog.info('[BaseStorage] Rebuilding subtype counts from storage...')
this.subtypeCountsByType.clear() this.subtypeCountsByType.clear()
this.nounSubtypeByIdCache.clear()
for (let shard = 0; shard < 256; shard++) { for (let shard = 0; shard < 256; shard++) {
const shardHex = shard.toString(16).padStart(2, '0') const shardHex = shard.toString(16).padStart(2, '0')
@ -3434,10 +3374,6 @@ export abstract class BaseStorage extends BaseStorageAdapter {
const metadata = await this.readCanonicalObject(path) const metadata = await this.readCanonicalObject(path)
if (metadata && metadata.noun && typeof metadata.subtype === 'string' && metadata.subtype.length > 0) { if (metadata && metadata.noun && typeof metadata.subtype === 'string' && metadata.subtype.length > 0) {
this.incrementSubtypeCount(metadata.noun as NounType, metadata.subtype) this.incrementSubtypeCount(metadata.noun as NounType, metadata.subtype)
// Path is `entities/nouns/<shard>/<id>/metadata.json` — extract id segment.
const segments = path.split('/')
const idSeg = segments[segments.length - 2]
if (idSeg) this.nounSubtypeByIdCache.set(idSeg, metadata.subtype)
} }
} catch { /* skip unreadable entities */ } } catch { /* skip unreadable entities */ }
} }
@ -3538,7 +3474,6 @@ export abstract class BaseStorage extends BaseStorageAdapter {
public async rebuildVerbSubtypeCounts(): Promise<void> { public async rebuildVerbSubtypeCounts(): Promise<void> {
prodLog.info('[BaseStorage] Rebuilding verb subtype counts from storage...') prodLog.info('[BaseStorage] Rebuilding verb subtype counts from storage...')
this.verbSubtypeCountsByType.clear() this.verbSubtypeCountsByType.clear()
this.verbSubtypeByIdCache.clear()
for (let shard = 0; shard < 256; shard++) { for (let shard = 0; shard < 256; shard++) {
const shardHex = shard.toString(16).padStart(2, '0') const shardHex = shard.toString(16).padStart(2, '0')
@ -3553,9 +3488,6 @@ export abstract class BaseStorage extends BaseStorageAdapter {
const verb = metadata.verb as VerbType const verb = metadata.verb as VerbType
const subtype = metadata.subtype as string const subtype = metadata.subtype as string
this.incrementVerbSubtypeCount(verb, subtype) this.incrementVerbSubtypeCount(verb, subtype)
const segments = path.split('/')
const idSeg = segments[segments.length - 2]
if (idSeg) this.verbSubtypeByIdCache.set(idSeg, { verb, subtype })
} }
} catch { /* skip unreadable verbs */ } } catch { /* skip unreadable verbs */ }
} }
@ -3660,17 +3592,12 @@ export abstract class BaseStorage extends BaseStorageAdapter {
try { try {
const metadata = await this.readCanonicalObject(path) const metadata = await this.readCanonicalObject(path)
if (metadata && metadata.noun) { if (metadata && metadata.noun) {
// 8.0 visibility: rebuild only public entities into the user-facing // 8.0 visibility: rebuild only public entities into the user-facing per-type stat.
// per-type stat; warm the cache so later writes/deletes stay symmetric.
const id = idFromMetadataPath(path)
if (isCountedVisibility(metadata.visibility)) { if (isCountedVisibility(metadata.visibility)) {
const typeIndex = TypeUtils.getNounIndex(metadata.noun) const typeIndex = TypeUtils.getNounIndex(metadata.noun)
if (typeIndex >= 0 && typeIndex < NOUN_TYPE_COUNT) { if (typeIndex >= 0 && typeIndex < NOUN_TYPE_COUNT) {
this.nounCountsByType[typeIndex]++ this.nounCountsByType[typeIndex]++
} }
if (id) this.nounVisibilityByIdCache.delete(id)
} else if (id) {
this.nounVisibilityByIdCache.set(id, metadata.visibility as EntityVisibility)
} }
} }
} catch (error) { } catch (error) {
@ -3697,15 +3624,11 @@ export abstract class BaseStorage extends BaseStorageAdapter {
const metadata = await this.readCanonicalObject(path) const metadata = await this.readCanonicalObject(path)
if (metadata && metadata.verb) { if (metadata && metadata.verb) {
// 8.0 visibility: rebuild only public edges into the user-facing per-type stat. // 8.0 visibility: rebuild only public edges into the user-facing per-type stat.
const id = idFromMetadataPath(path)
if (isCountedVisibility(metadata.visibility)) { if (isCountedVisibility(metadata.visibility)) {
const typeIndex = TypeUtils.getVerbIndex(metadata.verb) const typeIndex = TypeUtils.getVerbIndex(metadata.verb)
if (typeIndex >= 0 && typeIndex < VERB_TYPE_COUNT) { if (typeIndex >= 0 && typeIndex < VERB_TYPE_COUNT) {
this.verbCountsByType[typeIndex]++ this.verbCountsByType[typeIndex]++
} }
if (id) this.verbVisibilityByIdCache.delete(id)
} else if (id) {
this.verbVisibilityByIdCache.set(id, metadata.visibility as EntityVisibility)
} }
} }
} catch (error) { } catch (error) {
@ -3726,56 +3649,23 @@ export abstract class BaseStorage extends BaseStorageAdapter {
} }
/** /**
* Resolve the `NounType` for a noun, used to attribute the entity to the * Resolve a noun's `NounType` straight from its canonical metadata record on
* correct slot in `nounCountsByType` (which backs `_system/type-statistics.json` * disk. Used by the poisoned-statistics detector (`detectPoisonedTypeStatistics()`)
* and `brain.stats().entitiesByType`). * to confirm whether on-disk types disagree with a `'thing'`-only persisted
* rollup before triggering a full `rebuildTypeCounts()`. Returns `null` when
* metadata genuinely doesn't exist or the read fails; the caller decides whether
* to skip the entity. There is no in-memory idtype cache the record is the
* single source of truth.
* *
* Lookup order: * @param id - The noun id whose type to resolve.
* 1. `nounTypeByIdCache`, populated by `saveNounMetadata_internal()` * @returns The stored `NounType`, or `null` if absent/unreadable.
* whenever a noun's metadata is written. This is the common path
* add() saves metadata before the HNSW noun, so by the time we get
* here the cache is warm.
* 2. `'thing'` as a final fallback when metadata genuinely doesn't exist
* (a noun added without metadata irregular but tolerated for back-
* compat). Logged so a real bug doesn't go unnoticed.
*
* Note this is intentionally synchronous because `saveNoun_internal()` and
* its callers are not async-friendly at this layer. For cases where only
* an id is known after a process restart and the cache is cold,
* `getNounTypeFromStorageAsync()` is available for callers that can await.
*/
protected getNounType(noun: HNSWNoun): NounType {
const cached = this.nounTypeByIdCache.get(noun.id)
if (cached) return cached
// Cache miss on a noun we're about to write means metadata was not saved
// first. Brainy.add() always saves metadata before HNSW, so this only
// fires in unusual code paths (raw `storage.saveNoun()` without metadata).
prodLog.warn(
`[BaseStorage] getNounType: no type cached for noun ${noun.id}. ` +
`Type-statistics will attribute this entity to 'thing'. ` +
`Call saveNounMetadata() before saveNoun() to avoid stat drift.`
)
return 'thing'
}
/**
* Async variant of `getNounType()` that consults disk if the in-memory
* cache is cold (e.g. after a process restart with deferred work). Used by
* `rebuildTypeCounts()` and other callers that can await an IO. Returns
* `null` if metadata genuinely doesn't exist; callers decide whether to
* fall back to 'thing' or skip the entity.
*/ */
protected async getNounTypeFromStorageAsync(id: string): Promise<NounType | null> { protected async getNounTypeFromStorageAsync(id: string): Promise<NounType | null> {
const cached = this.nounTypeByIdCache.get(id)
if (cached) return cached
try { try {
const metadataPath = getNounMetadataPath(id) const metadataPath = getNounMetadataPath(id)
const metadata = await this.readCanonicalObject(metadataPath) const metadata = await this.readCanonicalObject(metadataPath)
if (metadata && (metadata as NounMetadata).noun) { if (metadata && (metadata as NounMetadata).noun) {
const type = (metadata as NounMetadata).noun as NounType return (metadata as NounMetadata).noun as NounType
// Warm the cache so subsequent sync calls hit fast.
this.nounTypeByIdCache.set(id, type)
return type
} }
} catch { } catch {
// Storage error — treat as unknown. // Storage error — treat as unknown.
@ -3879,28 +3769,16 @@ export abstract class BaseStorage extends BaseStorageAdapter {
* Save a noun to storage (ID-first path) * Save a noun to storage (ID-first path)
*/ */
protected async saveNoun_internal(noun: HNSWNoun): Promise<void> { protected async saveNoun_internal(noun: HNSWNoun): Promise<void> {
const type = this.getNounType(noun)
const path = getNounVectorPath(noun.id) const path = getNounVectorPath(noun.id)
// Per-type stats counter (`nounCountsByType`, read by stats().entitiesByType) // Hot path: write the vector record only. Per-type counters
// is maintained in saveNounMetadata_internal(), gated on isNew + visibility — // (`nounCountsByType`, read by stats().entitiesByType) AND the periodic
// NOT here. saveNoun_internal() also runs on HNSW neighbor-link re-saves, so // type-statistics persist trigger are both maintained in
// incrementing here inflated the per-type counts with graph connectivity. We // saveNounMetadata_internal(), which holds the canonical metadata record —
// only READ the (externally-maintained) count below to decide when to persist. // and therefore the NounType + visibility — at the exact point a count
const typeIndex = TypeUtils.getNounIndex(type) // changes. saveNoun_internal() also re-runs on every HNSW neighbor-link
const counted = isCountedVisibility(this.nounVisibilityByIdCache.get(noun.id)) // re-save, so it deliberately performs NO type lookup and NO count work here.
// Write-cache coherent canonical write
await this.writeCanonicalObject(path, noun) await this.writeCanonicalObject(path, noun)
// Periodically save statistics
// Also save on first noun of each type to ensure low-count types are tracked
const shouldSave = counted &&
(this.nounCountsByType[typeIndex] === 1 || // First noun of type
this.nounCountsByType[typeIndex] % 100 === 0) // Every 100th
if (shouldSave) {
await this.saveTypeStatistics()
}
} }
/** /**