2025-10-09 18:08:57 -07:00
/ * *
* 🧠 BRAINY EMBEDDED TYPE EMBEDDINGS
*
* AUTO - GENERATED - DO NOT EDIT
feat: production-ready value-based temporal field detection
Replaces unreliable field name pattern matching with DuckDB-inspired value analysis.
### Critical Bug Fix
- Fixes 618k file explosion from false positive temporal field detection
- Field name patterns like `.endsWith('at')` incorrectly flagged non-temporal fields
- Example: "cat", "bat", "hat" were treated as timestamps, creating millions of files
### New System: FieldTypeInference
- Analyzes actual data VALUES, not field names
- Unix timestamp detection: checks if numbers fall in 2000-2100 range
- ISO 8601 datetime detection: pattern matching for date strings
- 11 field types: TIMESTAMP_MS, TIMESTAMP_S, DATE_ISO8601, DATETIME_ISO8601, BOOLEAN, INTEGER, FLOAT, UUID, ARRAY, OBJECT, STRING
- Persistent caching for O(1) lookups at billion scale
- 95%+ accuracy vs 70% with pattern matching
### Architecture
- Zero configuration required
- No fallbacks - pure value-based detection only
- Progressive refinement as more data arrives
- Production patterns from DuckDB, Apache Arrow, Parquet
### Tests
- 39 comprehensive unit tests (all passing)
- Real-world scenarios including exact bug reproduction
- Full coverage: all types, cache, edge cases
### Performance
- Cache hit: 0.1-0.5ms (O(1))
- Cache miss: 5-10ms (analyze 100 samples)
- Memory: ~500 bytes per field
🤖 Generated with [Claude Code](https://claude.com/claude-code)
Co-Authored-By: Claude <noreply@anthropic.com>
2025-10-16 13:58:57 -07:00
* Generated : 2025 - 10 - 16T20 :17 : 08.371 Z
2025-10-09 18:08:57 -07:00
* Noun Types : 31
* Verb Types : 40
*
* This file contains pre - computed embeddings for all NounTypes and VerbTypes .
* No runtime computation needed , instant availability !
* /
import { NounType , VerbType } from '../types/graphTypes.js'
import { Vector } from '../coreTypes.js'
// Type metadata
export const TYPE_METADATA = {
nounTypes : 31 ,
verbTypes : 40 ,
totalTypes : 71 ,
embeddingDimensions : 384 ,
feat: production-ready value-based temporal field detection
Replaces unreliable field name pattern matching with DuckDB-inspired value analysis.
### Critical Bug Fix
- Fixes 618k file explosion from false positive temporal field detection
- Field name patterns like `.endsWith('at')` incorrectly flagged non-temporal fields
- Example: "cat", "bat", "hat" were treated as timestamps, creating millions of files
### New System: FieldTypeInference
- Analyzes actual data VALUES, not field names
- Unix timestamp detection: checks if numbers fall in 2000-2100 range
- ISO 8601 datetime detection: pattern matching for date strings
- 11 field types: TIMESTAMP_MS, TIMESTAMP_S, DATE_ISO8601, DATETIME_ISO8601, BOOLEAN, INTEGER, FLOAT, UUID, ARRAY, OBJECT, STRING
- Persistent caching for O(1) lookups at billion scale
- 95%+ accuracy vs 70% with pattern matching
### Architecture
- Zero configuration required
- No fallbacks - pure value-based detection only
- Progressive refinement as more data arrives
- Production patterns from DuckDB, Apache Arrow, Parquet
### Tests
- 39 comprehensive unit tests (all passing)
- Real-world scenarios including exact bug reproduction
- Full coverage: all types, cache, edge cases
### Performance
- Cache hit: 0.1-0.5ms (O(1))
- Cache miss: 5-10ms (analyze 100 samples)
- Memory: ~500 bytes per field
🤖 Generated with [Claude Code](https://claude.com/claude-code)
Co-Authored-By: Claude <noreply@anthropic.com>
2025-10-16 13:58:57 -07:00
generatedAt : "2025-10-16T20:17:08.371Z" ,
2025-10-09 18:08:57 -07:00
sizeBytes : {
embeddings : 109056 ,
base64 : 145408
}
}
// All noun types in order
const NOUN_TYPE_ORDER : NounType [ ] = [ "person" , "organization" , "location" , "thing" , "concept" , "event" , "document" , "media" , "file" , "message" , "content" , "collection" , "dataset" , "product" , "service" , "user" , "task" , "project" , "process" , "state" , "role" , "topic" , "language" , "currency" , "measurement" , "hypothesis" , "experiment" , "contract" , "regulation" , "interface" , "resource" ]
// All verb types in order
const VERB_TYPE_ORDER : VerbType [ ] = [ "relatedTo" , "contains" , "partOf" , "locatedAt" , "references" , "precedes" , "succeeds" , "causes" , "dependsOn" , "requires" , "creates" , "transforms" , "becomes" , "modifies" , "consumes" , "owns" , "attributedTo" , "createdBy" , "belongsTo" , "memberOf" , "worksWith" , "friendOf" , "follows" , "likes" , "reportsTo" , "supervises" , "mentors" , "communicates" , "describes" , "defines" , "categorizes" , "measures" , "evaluates" , "uses" , "implements" , "extends" , "inherits" , "conflicts" , "synchronizes" , "competes" ]
// Pre-computed embeddings (142.0KB base64)
const EMBEDDINGS_BASE64 = " O0Q + vZSXRL15 + VC9oLmdvUUFbb1Hjk46csIJPpIizbx3c8Y8LWbDPDU0hzw4VKi8JzXGOwwRuLuAJi89h5fHO / zgYT0fry49ovu8PC6PbDx58y + 94 eg1vZNooL03wve7PGHGvJbqp72bHXo9gevtvPJyBD31lcK6DB8HPcWroDznk4o9lIGFPb0kRT39rF09XJE8uSZHBD2sdJO8rxqNPO0juTwXHla9qNahvJvRjrzaWoo9peOxvCxglDuZUji7wvlPvHDRQT2WW7a9AhYMPRvZ0jxlxNy8rNteO0lEiL0lQQk9Wespu4ZjMjzNmPm8xpwcPIFklr02r645z4DZOxPiBz0h7WU8SMSRvbwtp73MjgO9Q77GvYSs / ryYT6K81hzOvHLKDb0p + RQ7HDaGvTJs + LswwR + 7 HmjMPfKvLL01C2c8cbcgveOV2Tx5VyM9i65uvVWcvz3rN389eyc2PQE34rxc0Uc9egT9vP6Xbz1Asf49ZuaAvSIperyInxc9BSMpvX3GLLwqAxu8PWF6PcIrrr2WlI08Sf29u1kFbzz0oA08twjNPK1OKjtIYd28IdCjPM5TpTznOvC8hap5PBv53b2FtRm99zcjPhKyAbwaNSK9uGK4PW0NsDtOXsq9Tm78PNoRNz0kqmA8cQbaO0pclDzYt4I68wA2PQyojAmiSiS9q + ewPezyMj2Mz8I8yPWTOzWlIj15sj69doL3PHEbFjw8oJe9NxQ4PZ8AJLzHCFw8 + nOTPQXIsr1MZ + 883 EOfvVE + lbzf / KS9ri1bPc5zfz272eA8hLgWvRbsjT2woYe9N848vG / 0 AD1r5tc7ZI + WPcZZp7vabyG7vdNcPe6ayTzK0Ys7194dvBYvhTwR4cq81QxKvU / OOj3kQ / g8Fy58vVsPfD3tWtU9AqeBvY7fLL7ixYq70GwHPbvfjDxoeFE9hRqQPTFlZDzZgpA9x / cSPXu1FT3JjlU9T62Mva / QBrxR5da8aTibu88FJb3z07u7Q6HKPeGobT0CnNI9tWRhvN56XL24zAw9X908vOOuqj2rX / 28 aoqKPd34iD3xhGI8EKsEvUPLI75Y8ie9KrcWvKGPGz3FWWe8tL + qPQZ6LL25obY875n5PPyZyL3rsiq9MNMgvekGSrzsr4i96oQAPa6tkT296aM9d / 3 YvLahN7y / ohc + aoxXvSBmz4lHNeS9Zu6UvTPLZb2s2269YgsyPfqXj708wuy8fXKDuZVKLLwmvEQ8HbmfvVHOPb37RL + 7 DQKdu00jDT01I / G81PWpu / r4v7sB2Sy8mB0jvC0Y / Dx0fuY7mPOKvUsBCT1QErI9USRKu / PfxT0O8m89zxn1uzGeo70vgh + 7 TVx3PD5 + yr1s7pq9msMhuwFqIjt + Zoe9lPfbuz7egLpiZK68XeSBPR9tJbzuC2q9 + ZPPvF8RQL1ytVK86W / LPOXa8L1y6FC9fPfePI6VTr3zJzG9V3aSPDbyar2rlK29ea / 9 PEkC8T05Ke + 8 HwIBPg0WQjyyYgk9VSxuvFmitLrUm9Y99jg9vZeV6rwpwjQ9sdZEPRmi / rwJVSW9Jo3OPOtquL0FpcG9TCT3vD1xNTuz2hC9Gjgivf + 4 R71eWhM9uAPhvdQYiL1avwg9qFqMPHgUKj2LB5c97prKPAczUbyLM + k8PEGkvMEqpD1aTuQ7OOAiPZ7Al70h9bu75FmivaUls7KwRra9rO4wvJHnsDzZfg27cXDRvODJyLwYMy29geuFvaTsCDuT16E988vNPOpOBb1opwQ9 / adrO0Ne5D2uT7e8632ZPI / GSz1Car691OP3vAgaFbs6pyI9dtaxOhK7i70X + YA4 / PGQPfPWer25XYU8S89wvRQo9jz9 + xu9FACiPScLjbwEE1q8Iyd1PV51CD2qQK47 / IfYuwZxJ7yXOEa9HuhMPG1 + 7 DzYuaW7gttBPcKL + j206NQ8E7ohvT0EuL2 / JQg8kuaXOyxW2bwLl2a9hKUUvT4PuDzE3ae6LvgxvcAwQT3 + 1 m89nqGwPYm077x2Hks8VDA9PawxhTxwndG7XS2YvErDh72 / GZ69iqRBuwya / Lx6ROm6g2jhu6jOkr0li8A9 + joIvciADDzya3 + 6 W + LAve1cQL23Y6y8LfqPvPjT / Tz + Fqs8 + rOIvIl1n72d30K8L4KLvds0 / btkwq + 887 Livdqt27wd / 4 K95QslvIb7Hr3zFX29KqNtvC5 / VD3NYAs + ddh8OdOKnz2 + hpo9NqPUO0QAerzWGYs9RHToOncPYL3LnuU82JjYPLEpmTy9otG8vFjtPDJgNby0 + JU8OXn3Ow4I2T02aUQ9zClhvU + U3Dzlw3o9wVdQvGvVvTykaJO9Kj2CvGq3V7q1N2S9UiZEPc5ogDxLz109qCOBPaP6Fz16sYo9DC / LvDJJwLyVFR68LHwyvrzhLj00ZQe9qWt2va / 8 zTstSNY8j2TbPUq5s7ybCYQ9dQzbPSBKlb0UqlA9kdOcPUG7jjxH0pw9G36SvaOL6jyFhIA9JsRBPZfFQ71H3sw94w2 / PHlcszwkhn09r / rovLgHkL1j / iQ9rjQPvZY6SzvHArg8smdzPbP5gbzvZTA9KYsRvUi5SzhFlfS9PBUAveCgQz0yiVM7eK6KPaN6gj3R4SI8LMZGPYBDgb1sh4 + 9 uMBHvfmN + r2e91 + 8 rYDbPJ7RQrxoZ8y9vp2FPaEf4zxn9r28KLpgvf5jCz3nsys9681EvCZVngnNC2y74 + kKPYjcuTxWi1a9NIGgO7lVfDvz + Yi8rUbuPNik1L3pepQ9AyoTviuxYDujMzo9Iw / mvUpVyjtuEk29kU36O7LKRDzRefo8wIPjPDkiobx5Qls90g + EPAQCs7xIB4M9rS + cPaXSfr147788Wrn8PN + qxDxCnZg9HalnvTsZcb04tya9D9PUun6yHLye89m7q70OvAqZNT23q4W8Y16 + vFThBT20HTo9vrNhPX5hubx2pf286bAZPJHUCbhYcxA + CuV9vDBjJ70pIEG9HdGHPZUKHb1wdVA9pIORvW5X6TwgLLU8G4ufu8zxCT0eDIc9fEQiPp37Dr1xKU49GQuvvEF2Er3UkwW9c4DQvDND7j0XTPu9HlVWPSjHG736SRw9Mr8HPaiD0r0FiPq7 / Qr4O + 99 qDx5Ozi84KBUvK + dZryPOje9T93BO + wqFryV / HA9Wb + xPNkrjz2ulIM5K5PtvKOrST1ws5y9Tw0KPfqyvDzgiR4 + 5 OCku9cY + In / DW49v7BYvWpRMjw6gby9XdcPPcJ0Vbx0UU68o4WMvcXmoDwEaJk95rlGPB + Oq7zk0ai9i8AvPOUgMDxUWbq86btMPb + siDsflqC815duvQI8OL3ngGk9VopSPUwczDxiiPa8fEAkvMzCvT1wtDO9XwEuPZT5WTylEPY6Mq08vIgmer2lTJU9pfa5vF0YC73dB / u85kA7vbQjYDxGlTa9ke / GPGARWD2Y3Iu9hfhIPR2V1zyBjI68lZWGPSTnWzui2la9PpkMPINj4708hKm9CH3wuxD2g7pxSJ89zagKPuOrej2m9S69aFjzPMmyRbx5zCE94B23OwAQ1DxBoG89uy9WPQUydz2DI7a7S7YoPXNIwzvNJC47v7z9PIZ5Xr0iwo693G0FPan7 / 7 yRpxI9zMMpvVOhHb1xBI + 9 LMpwvFNAv7w6b6 + 8 + 6 AlvWY0rT0FGyK9tIs2Pfx / pz2 / j6e7wIlZvE8JaT1o3iO9LFzSvVVUqT0uQrw8 + t6rvNdLx7Lo4vK86fOFPC6xxTygHvC8NE9BPXHNLL5HMIS8U4ZIvTtWhLyMjNC8KV5oPZBki7u7op + 9 U / CSPHoGsDwjvMc8w7OevXUhh7rpGgq9WK0uunI5QbxIv6O9TjKkvMaZ1TltIIO8wrQZva5RFL1YXF28by01Oxse2TsVoQ49EY5DPVROl7zHm2q7FgnUPFlqfL30VTe9o1BPPLMpCzyg3gS9C295vU96F702U9K7YG4hPeoTpjwbnwI6h2IKveeiXDvyJU69SdG4vfPolTwOgHY6k6GZvWSvOj3ZIUe9egxAvSGjbDz8lXs8Oc6T
// Decode embeddings at startup (happens once, <10ms)
function decodeEmbeddings ( ) : Uint8Array {
if ( typeof Buffer !== 'undefined' ) {
// Node.js environment
return Buffer . from ( EMBEDDINGS_BASE64 , 'base64' )
} else if ( typeof atob !== 'undefined' ) {
// Browser environment
const binaryString = atob ( EMBEDDINGS_BASE64 )
const bytes = new Uint8Array ( binaryString . length )
for ( let i = 0 ; i < binaryString . length ; i ++ ) {
bytes [ i ] = binaryString . charCodeAt ( i )
}
return bytes
}
return new Uint8Array ( 0 )
}
// Cached decoded embeddings
let decodedEmbeddings : Uint8Array | null = null
/ * *
* Get noun type embeddings as a Map for fast lookup
* This is called once and cached
* /
export function getNounTypeEmbeddings ( ) : Map < NounType , Vector > {
if ( ! decodedEmbeddings ) {
decodedEmbeddings = decodeEmbeddings ( )
}
const embeddings = new Map < NounType , Vector > ( )
const view = new DataView ( decodedEmbeddings . buffer )
const embeddingSize = 384
NOUN_TYPE_ORDER . forEach ( ( type , index ) = > {
const offset = index * embeddingSize * 4
const embedding = new Float32Array ( embeddingSize )
for ( let i = 0 ; i < embeddingSize ; i ++ ) {
embedding [ i ] = view . getFloat32 ( offset + i * 4 , true )
}
embeddings . set ( type , Array . from ( embedding ) )
} )
return embeddings
}
/ * *
* Get verb type embeddings as a Map for fast lookup
* This is called once and cached
* /
export function getVerbTypeEmbeddings ( ) : Map < VerbType , Vector > {
if ( ! decodedEmbeddings ) {
decodedEmbeddings = decodeEmbeddings ( )
}
const embeddings = new Map < VerbType , Vector > ( )
const view = new DataView ( decodedEmbeddings . buffer )
const embeddingSize = 384
// Verb embeddings start after noun embeddings
const verbStartOffset = 31 * embeddingSize * 4
VERB_TYPE_ORDER . forEach ( ( type , index ) = > {
const offset = verbStartOffset + index * embeddingSize * 4
const embedding = new Float32Array ( embeddingSize )
for ( let i = 0 ; i < embeddingSize ; i ++ ) {
embedding [ i ] = view . getFloat32 ( offset + i * 4 , true )
}
embeddings . set ( type , Array . from ( embedding ) )
} )
return embeddings
}
// Import logging
import { prodLog } from '../utils/logger.js'
prodLog . info ( ` 🧠 Brainy Type Embeddings loaded: ${ TYPE_METADATA . nounTypes } nouns, ${ TYPE_METADATA . verbTypes } verbs, ${ ( TYPE_METADATA . sizeBytes . embeddings / 1024 ) . toFixed ( 1 ) } KB ` )