fix: metadata-only update() never rewrites the noun record — the unconditional whole-vector save turned per-entity stat touches into full rewrites+fsync, amplifying read-heavy sweeps into disk saturation on a production deployment
Also: idle PathResolver stats tick no longer logs NaN% every minute (logs only on new traffic, via prodLog); graph-lsm-* key family recognized as system resources (kills the per-boot unknown-key warning on provider-backed brains). Four regression pins in tests/integration/update-write-granularity.
This commit is contained in:
parent
64049631bc
commit
cb717be275
4 changed files with 155 additions and 15 deletions
|
|
@ -3161,18 +3161,23 @@ export class Brainy<T = any> implements BrainyInterface<T> {
|
|||
new UpdateNounMetadataOperation(this.storage, params.id, updatedMetadata)
|
||||
)
|
||||
|
||||
// Operation 2: Update vector data (will use updated type cache)
|
||||
tx.addOperation(
|
||||
new SaveNounOperation(this.storage, {
|
||||
id: params.id,
|
||||
vector,
|
||||
connections: new Map(),
|
||||
level: 0
|
||||
})
|
||||
)
|
||||
|
||||
// Operation 3-4: Update HNSW index (remove and re-add if reindexing needed)
|
||||
// Operations 2-4: vector-record write + HNSW reindex — ONLY when the
|
||||
// vector side actually changed (new data/vector/type). A metadata-only
|
||||
// update must never rewrite the noun record: the record carries the
|
||||
// full vector, so an unconditional save turned every metadata touch
|
||||
// into a whole-vector rewrite + fsync — under a read-heavy consumer
|
||||
// sweep that bumps per-entity stats, this amplified into disk
|
||||
// saturation on a production deployment (SELF-ENGINE-RESTART-GRIND,
|
||||
// 2026-07-29: 5.8GB written in 40min from ~50 recalls/min).
|
||||
if (needsReindexing) {
|
||||
tx.addOperation(
|
||||
new SaveNounOperation(this.storage, {
|
||||
id: params.id,
|
||||
vector,
|
||||
connections: new Map(),
|
||||
level: 0
|
||||
})
|
||||
)
|
||||
tx.addOperation(
|
||||
new RemoveFromVectorIndexOperation(this.index, params.id, existing.vector)
|
||||
)
|
||||
|
|
|
|||
|
|
@ -382,6 +382,10 @@ export abstract class BaseStorage extends BaseStorageAdapter {
|
|||
// identical to the unknown-key fallback these keys hit
|
||||
// before being listed here — this only kills the
|
||||
// per-boot "Unknown key format" warning)
|
||||
id.startsWith('graph-lsm-') || // Graph-LSM store manifests written through storage by
|
||||
// an active native graph provider — same
|
||||
// warn-then-route fallback as above; listing the family
|
||||
// silences the per-boot warning on provider-backed brains
|
||||
isSingletonSystemKey(id) // Known singletons (e.g. brainy:entityIdMapper) hit the
|
||||
// same warn-then-route fallback without this — the
|
||||
// routing below already handles them identically
|
||||
|
|
|
|||
|
|
@ -57,6 +57,7 @@ export class PathResolver {
|
|||
// Statistics
|
||||
private cacheHits = 0
|
||||
private cacheMisses = 0
|
||||
private lastLoggedLookups = 0 // last total the maintenance tick logged stats at
|
||||
private metadataIndexHits = 0
|
||||
private metadataIndexMisses = 0
|
||||
private graphTraversalFallbacks = 0
|
||||
|
|
@ -519,10 +520,14 @@ export class PathResolver {
|
|||
}
|
||||
}
|
||||
|
||||
// Log cache statistics (in production, send to monitoring)
|
||||
const hitRate = this.cacheHits / (this.cacheHits + this.cacheMisses)
|
||||
if ((this.cacheHits + this.cacheMisses) % 1000 === 0) {
|
||||
console.log(`[PathResolver] Cache stats: ${Math.round(hitRate * 100)}% hit rate, ${this.pathCache.size} entries, ${this.hotPaths.size} hot paths`)
|
||||
// Log cache statistics only when there is new traffic to report — an
|
||||
// idle resolver stays silent. 0/0 lookups previously rendered
|
||||
// "NaN% hit rate" (and the %1000 gate passes at zero), which spammed
|
||||
// production journals once a minute on every idle VFS.
|
||||
const totalLookups = this.cacheHits + this.cacheMisses
|
||||
if (totalLookups > 0 && totalLookups !== this.lastLoggedLookups && totalLookups % 1000 === 0) {
|
||||
this.lastLoggedLookups = totalLookups
|
||||
prodLog.debug(`[PathResolver] Cache stats: ${Math.round((this.cacheHits / totalLookups) * 100)}% hit rate, ${this.pathCache.size} entries, ${this.hotPaths.size} hot paths`)
|
||||
}
|
||||
}, 60000) // Every minute
|
||||
// Cache maintenance must never keep the host process alive.
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue