fix(8.0): distinctCount aggregates distinct values of any type + edge-case regression tests
distinctCount previously routed through getNumericField, so it silently returned 0 for
string/categorical fields — its primary use case (distinct categories / users / tags).
It now tracks the raw value (any type, keyed by string form) on both the add and remove
contribution paths; numeric ops (sum/avg/min/max/stddev/variance/percentile) are unchanged.
Also adds regression coverage confirming two long-standing query behaviors hold on the 8.0
engine: find({ where: { field: { missing: true } } }) matches a never-registered field, and
find({ type: [...], orderBy, limit }) returns the full set on the first call after
mutate+query+get. (percentile/median were already correct.)
Tests: tests/unit/brainy/find-agg-edge-cases.test.ts + existing aggregation suite green.
This commit is contained in:
parent
7aad80395e
commit
574a8b147c
2 changed files with 130 additions and 4 deletions
|
|
@ -200,8 +200,9 @@ function updateMetricAdd(state: MetricState, val: number, op: AggregationOp): vo
|
||||||
state.m2 = (state.m2 ?? 0) + (val - oldMean) * (val - mean)
|
state.m2 = (state.m2 ?? 0) + (val - oldMean) * (val - mean)
|
||||||
}
|
}
|
||||||
|
|
||||||
// Percentile/distinctCount: track the value multiset (matches Cortex's value_counts).
|
// Percentile: track the numeric value multiset (matches Cortex's value_counts) for the
|
||||||
if (op === 'percentile' || op === 'distinctCount') {
|
// exact-percentile computation. (distinctCount is tracked separately, on raw values.)
|
||||||
|
if (op === 'percentile') {
|
||||||
if (!state.valueCounts) state.valueCounts = {}
|
if (!state.valueCounts) state.valueCounts = {}
|
||||||
const key = String(val)
|
const key = String(val)
|
||||||
state.valueCounts[key] = (state.valueCounts[key] ?? 0) + 1
|
state.valueCounts[key] = (state.valueCounts[key] ?? 0) + 1
|
||||||
|
|
@ -213,8 +214,9 @@ function updateMetricAdd(state: MetricState, val: number, op: AggregationOp): vo
|
||||||
* Note: removing from Welford's is the inverse update.
|
* Note: removing from Welford's is the inverse update.
|
||||||
*/
|
*/
|
||||||
function updateMetricRemove(state: MetricState, val: number, op: AggregationOp): void {
|
function updateMetricRemove(state: MetricState, val: number, op: AggregationOp): void {
|
||||||
// Percentile/distinctCount: decrement the value multiset (drop the key at zero).
|
// Percentile: decrement the numeric value multiset (drop the key at zero).
|
||||||
if ((op === 'percentile' || op === 'distinctCount') && state.valueCounts) {
|
// (distinctCount is decremented separately, on raw values.)
|
||||||
|
if (op === 'percentile' && state.valueCounts) {
|
||||||
const key = String(val)
|
const key = String(val)
|
||||||
const c = state.valueCounts[key]
|
const c = state.valueCounts[key]
|
||||||
if (c !== undefined) {
|
if (c !== undefined) {
|
||||||
|
|
@ -741,6 +743,17 @@ export class AggregationIndex {
|
||||||
if (metricDef.op === 'count') {
|
if (metricDef.op === 'count') {
|
||||||
state.count++
|
state.count++
|
||||||
state.sum++
|
state.sum++
|
||||||
|
} else if (metricDef.op === 'distinctCount') {
|
||||||
|
// distinctCount tracks distinct values of ANY type (strings, numbers, booleans),
|
||||||
|
// keyed by their string form — NOT numeric-coerced, since its primary use is
|
||||||
|
// categorical (distinct categories / users / tags), not numeric columns.
|
||||||
|
const raw = resolveEntityField(entity as unknown as HNSWNounWithMetadata, metricDef.field!)
|
||||||
|
if (raw !== undefined && raw !== null) {
|
||||||
|
if (!state.valueCounts) state.valueCounts = {}
|
||||||
|
const key = String(raw)
|
||||||
|
state.valueCounts[key] = (state.valueCounts[key] ?? 0) + 1
|
||||||
|
state.count++
|
||||||
|
}
|
||||||
} else {
|
} else {
|
||||||
const val = getNumericField(entity, metricDef.field!)
|
const val = getNumericField(entity, metricDef.field!)
|
||||||
if (val !== undefined) {
|
if (val !== undefined) {
|
||||||
|
|
@ -777,6 +790,17 @@ export class AggregationIndex {
|
||||||
if (metricDef.op === 'count') {
|
if (metricDef.op === 'count') {
|
||||||
state.count = Math.max(0, state.count - 1)
|
state.count = Math.max(0, state.count - 1)
|
||||||
state.sum = Math.max(0, state.sum - 1)
|
state.sum = Math.max(0, state.sum - 1)
|
||||||
|
} else if (metricDef.op === 'distinctCount') {
|
||||||
|
const raw = resolveEntityField(entity as unknown as HNSWNounWithMetadata, metricDef.field!)
|
||||||
|
if (raw !== undefined && raw !== null && state.valueCounts) {
|
||||||
|
const key = String(raw)
|
||||||
|
const c = state.valueCounts[key]
|
||||||
|
if (c !== undefined) {
|
||||||
|
if (c <= 1) delete state.valueCounts[key]
|
||||||
|
else state.valueCounts[key] = c - 1
|
||||||
|
}
|
||||||
|
state.count = Math.max(0, state.count - 1)
|
||||||
|
}
|
||||||
} else {
|
} else {
|
||||||
const val = getNumericField(entity, metricDef.field!)
|
const val = getNumericField(entity, metricDef.field!)
|
||||||
if (val !== undefined) {
|
if (val !== undefined) {
|
||||||
|
|
|
||||||
102
tests/unit/brainy/find-agg-edge-cases.test.ts
Normal file
102
tests/unit/brainy/find-agg-edge-cases.test.ts
Normal file
|
|
@ -0,0 +1,102 @@
|
||||||
|
/**
|
||||||
|
* Edge-case regression tests for three long-standing query/aggregation items
|
||||||
|
* (#26 missing-field predicate, #27 type[]+orderBy first-call completeness,
|
||||||
|
* #24 percentile/median/distinctCount aggregation ops) — verified against the 8.0 engine.
|
||||||
|
*/
|
||||||
|
|
||||||
|
import { describe, it, expect, beforeEach, afterEach } from 'vitest'
|
||||||
|
import { Brainy } from '../../../src/brainy'
|
||||||
|
import { createTestConfig } from '../../helpers/test-factory'
|
||||||
|
import { NounType } from '../../../src/types/graphTypes'
|
||||||
|
|
||||||
|
describe('find + aggregation edge cases', () => {
|
||||||
|
let brain: Brainy
|
||||||
|
|
||||||
|
beforeEach(async () => {
|
||||||
|
brain = new Brainy(createTestConfig())
|
||||||
|
await brain.init()
|
||||||
|
})
|
||||||
|
|
||||||
|
afterEach(async () => {
|
||||||
|
await brain.close()
|
||||||
|
})
|
||||||
|
|
||||||
|
// #26
|
||||||
|
it('find({ where: { field: { missing: true } } }) matches a never-registered field', async () => {
|
||||||
|
const a = await brain.add({ data: 'a', type: NounType.Thing, subtype: 'x', metadata: { name: 'A' } })
|
||||||
|
const b = await brain.add({
|
||||||
|
data: 'b',
|
||||||
|
type: NounType.Thing,
|
||||||
|
subtype: 'x',
|
||||||
|
metadata: { name: 'B', status: 'open' }
|
||||||
|
})
|
||||||
|
|
||||||
|
// Never-registered field → every entity "misses" it.
|
||||||
|
const neverReg = await brain.find({ where: { neverRegistered: { missing: true } }, limit: 100 })
|
||||||
|
const neverRegIds = neverReg.map((r) => r.id)
|
||||||
|
expect(neverRegIds).toContain(a)
|
||||||
|
expect(neverRegIds).toContain(b)
|
||||||
|
|
||||||
|
// Registered field, absent on one entity.
|
||||||
|
const noStatus = await brain.find({ where: { status: { missing: true } }, limit: 100 })
|
||||||
|
const noStatusIds = noStatus.map((r) => r.id)
|
||||||
|
expect(noStatusIds).toContain(a) // a has no status
|
||||||
|
expect(noStatusIds).not.toContain(b) // b has status
|
||||||
|
})
|
||||||
|
|
||||||
|
// #27
|
||||||
|
it('find({ type: [...], orderBy, limit }) returns the full set on the first call after mutate+query+get', async () => {
|
||||||
|
const ids: string[] = []
|
||||||
|
for (let i = 0; i < 6; i++) {
|
||||||
|
ids.push(
|
||||||
|
await brain.add({
|
||||||
|
data: `e${i}`,
|
||||||
|
type: i % 2 === 0 ? NounType.Document : NounType.Person,
|
||||||
|
subtype: 'x',
|
||||||
|
metadata: { n: i }
|
||||||
|
})
|
||||||
|
)
|
||||||
|
}
|
||||||
|
await brain.update({ id: ids[0], metadata: { n: 100 }, merge: true }) // mutate
|
||||||
|
await brain.find({ query: 'e' }) // query
|
||||||
|
await brain.get(ids[1]) // get
|
||||||
|
|
||||||
|
const r = await brain.find({
|
||||||
|
type: [NounType.Document, NounType.Person],
|
||||||
|
orderBy: 'n',
|
||||||
|
limit: 10
|
||||||
|
})
|
||||||
|
expect(r).toHaveLength(6) // all six, not a partial page
|
||||||
|
})
|
||||||
|
|
||||||
|
// #24
|
||||||
|
it('queryAggregate supports percentile (median) and distinctCount', async () => {
|
||||||
|
brain.defineAggregate({
|
||||||
|
name: 'amounts',
|
||||||
|
source: { type: NounType.Event },
|
||||||
|
groupBy: ['subtype'], // all 'txn' → one group
|
||||||
|
metrics: {
|
||||||
|
median: { op: 'percentile', field: 'amount', p: 0.5 },
|
||||||
|
distinctCats: { op: 'distinctCount', field: 'category' }
|
||||||
|
}
|
||||||
|
})
|
||||||
|
|
||||||
|
const rows0: Array<[number, string]> = [
|
||||||
|
[10, 'a'],
|
||||||
|
[20, 'b'],
|
||||||
|
[30, 'a'],
|
||||||
|
[40, 'c'],
|
||||||
|
[50, 'b']
|
||||||
|
]
|
||||||
|
for (const [amount, category] of rows0) {
|
||||||
|
await brain.add({ data: 'evt', type: NounType.Event, subtype: 'txn', metadata: { amount, category } })
|
||||||
|
}
|
||||||
|
|
||||||
|
const rows = await brain.queryAggregate('amounts', {})
|
||||||
|
expect(rows.length).toBeGreaterThanOrEqual(1)
|
||||||
|
const m = (rows[0] as any).metrics ?? rows[0]
|
||||||
|
expect(m.median).toBeGreaterThanOrEqual(20)
|
||||||
|
expect(m.median).toBeLessThanOrEqual(40)
|
||||||
|
expect(m.distinctCats).toBe(3) // a, b, c
|
||||||
|
})
|
||||||
|
})
|
||||||
Loading…
Add table
Add a link
Reference in a new issue