brainy/src/import/ImportCoordinator.ts
David Snelling 970e08c466 feat(8.0): reserved-field contract — one canonical location, typed prevention, unified read/write
Brainy-owned field names (noun/verb, subtype, createdAt, updatedAt,
confidence, weight, service, data, createdBy, _rev) now have exactly one
home — top level — enforced by three layers driven from a single source of
truth, src/types/reservedFields.ts (RESERVED_ENTITY_FIELDS /
RESERVED_RELATION_FIELDS, exported):

1. Compile time — AddParams/UpdateParams/RelateParams/UpdateRelationParams
   metadata (and the transact() ops that extend them) reject a literal
   reserved key as a TypeScript error while keeping generic T ergonomics
   (typed bags, untyped brains, index-signature shapes, and a documented
   exemption for T-declared reserved keys). Pinned by @ts-expect-error
   type tests run under vitest typecheck mode on every unit run.

2. Write time — the 7.x update() remap is ported to 8.0 and extended to
   every write path: add/update/relate/updateRelation, their transact()
   mirrors, and db.with() overlays. User-settable fields lift to their
   dedicated param (top-level wins when both are supplied — closes the 7.x
   trap where update({metadata:{confidence}}) silently no-oped), and
   system-managed fields drop with a one-shot warning naming the right
   path. A remapped subtype satisfies subtype-pairing enforcement exactly
   like a top-level one.

3. Read time — every storage combine goes through one canonical hydration
   helper (hydrateNounWithMetadata / hydrateVerbWithMetadata over
   splitNoun/VerbMetadataRecord), so reserved fields surface ONLY top-level
   and entity/relation.metadata carry ONLY custom fields on live reads,
   batch reads, paginated listings, getRelations by source/target, streamed
   verbs, and historical asOf() materialization alike.

Read-path echoes found and fixed (previously the full stored record —
including the verb type key — leaked inside metadata): noun pagination,
verb pagination, getVerbsBySource/ByTarget (adjacency + shard fallback),
getVerbsBySourceBatch (which also dropped subtype/data), and the
filesystem verb stream. getRelations() results now surface
confidence/updatedAt top-level via verbsToRelations, updateRelation() no
longer erases service/createdBy, relate() persists its top-level
confidence/service params, and the dead convertHNSWVerbToGraphVerb echo
path is deleted. Import paths (CLI extract, deduplicator, coordinators,
neural import) write confidence through the dedicated param instead of the
bag. UpdateRelationParams is now exported from the package root.

Documented for consumers in docs/concepts/consistency-model.md ("Reserved
fields") and RELEASES.md. Regression tests ported from the 7.x fix and
extended to the full 8.0 contract (17 runtime tests + 41 type-level
assertions); full unit suite 1427/1427, db-mvcc integration 24/24.
2026-06-11 13:13:09 -07:00

1779 lines
60 KiB
TypeScript

/**
* Import Coordinator
*
* Unified import orchestrator that:
* - Auto-detects file formats
* - Routes to appropriate handlers
* - Coordinates dual storage (VFS + Graph)
* - Provides simple, unified API
*
* NO MOCKS - Production-ready implementation
*/
import { Brainy } from '../brainy.js'
import { FormatDetector, SupportedFormat } from './FormatDetector.js'
import { ImportHistory } from './ImportHistory.js'
import { BackgroundDeduplicator } from './BackgroundDeduplicator.js'
import { SmartExcelImporter } from '../importers/SmartExcelImporter.js'
import { SmartPDFImporter } from '../importers/SmartPDFImporter.js'
import { SmartCSVImporter } from '../importers/SmartCSVImporter.js'
import { SmartJSONImporter } from '../importers/SmartJSONImporter.js'
import { SmartMarkdownImporter } from '../importers/SmartMarkdownImporter.js'
import { SmartYAMLImporter } from '../importers/SmartYAMLImporter.js'
import { SmartDOCXImporter } from '../importers/SmartDOCXImporter.js'
import { VFSStructureGenerator } from '../importers/VFSStructureGenerator.js'
import { NounType, VerbType } from '../types/graphTypes.js'
import { v4 as uuidv4 } from '../universal/uuid.js'
import * as fs from 'fs'
import * as path from 'path'
export interface ImportSource {
/** Source type */
type: 'buffer' | 'path' | 'string' | 'object' | 'url'
/** Source data */
data: Buffer | string | object
/** Optional filename hint */
filename?: string
/** HTTP headers for URL imports */
headers?: Record<string, string>
/** Basic authentication for URL imports */
auth?: {
username: string
password: string
}
}
/**
* Tracking context for import operations
* Contains metadata that should be attached to all created entities/relationships
*/
export interface TrackingContext {
/** Unique identifier for this import operation */
importId: string
/** Project identifier grouping related imports */
projectId: string
/** Timestamp when import started */
importedAt: number
/** Format of imported data */
importFormat: string
/** Source filename or URL */
importSource: string
/** Custom metadata from user */
customMetadata: Record<string, any>
}
/**
* Valid import options for v4.x
*/
export interface ValidImportOptions {
/** Force specific format (skip auto-detection) */
format?: SupportedFormat
/** VFS root path for imported files */
vfsPath?: string
/** Grouping strategy for VFS */
groupBy?: 'type' | 'sheet' | 'flat' | 'custom'
/** Custom grouping function */
customGrouping?: (entity: any) => string
/** Create entities in knowledge graph */
createEntities?: boolean
/** Create relationships in knowledge graph */
createRelationships?: boolean
/** Create provenance relationships (document → entity) */
createProvenanceLinks?: boolean
/** Preserve source file in VFS */
preserveSource?: boolean
/** Enable neural entity extraction */
enableNeuralExtraction?: boolean
/** Enable relationship inference */
enableRelationshipInference?: boolean
/** Enable concept extraction */
enableConceptExtraction?: boolean
/** Confidence threshold for entities */
confidenceThreshold?: number
/** Enable entity deduplication across imports */
enableDeduplication?: boolean
/** Similarity threshold for deduplication (0-1) */
deduplicationThreshold?: number
/** Enable import history tracking */
enableHistory?: boolean
/** Chunk size for streaming large imports (0 = no streaming) */
chunkSize?: number
/**
* Unique identifier for this import operation (auto-generated if not provided)
* Used to track all entities/relationships created in this import
* Note: Entities can belong to multiple imports (stored as array)
*/
importId?: string
/**
* Project identifier (user-specified or derived from vfsPath)
* Groups multiple imports under a common project
* If not specified, defaults to sanitized vfsPath
*/
projectId?: string
/**
* Custom metadata to attach to all created entities
* Merged with import/project tracking metadata
*/
customMetadata?: Record<string, any>
/**
* Default subtype for imported entities when the extractor doesn't set one.
*
* The importer resolves subtype in this precedence order:
*
* 1. Extractor-set subtype on the extracted entity (highest priority — the
* extractor knows the entity's true sub-classification).
* 2. `defaultSubtype` from this option (caller's choice — useful for tagging
* a whole import batch, e.g. `'customer-upload-2026q2'`).
* 3. Brainy-default `'imported'` (lowest priority — safety net so enforcement
* doesn't fire on entities the consumer forgot to classify).
*
* Added 7.30.1 so importers behave correctly under brain-wide strict mode and
* SDK_CORE_VOCABULARY-style enforcement consumers register.
*/
defaultSubtype?: string
/**
* Progress callback for tracking import progress
*
* **Streaming Architecture** (always enabled):
* - Indexes are flushed periodically during import (adaptive intervals)
* - Data is queryable progressively as import proceeds
* - `progress.queryable` is `true` after each flush
* - Provides crash resilience and live monitoring
*
* **Adaptive Flush Intervals**:
* - <1K entities: Flush every 100 entities (max 10 flushes)
* - 1K-10K entities: Flush every 1000 entities (10-100 flushes)
* - >10K entities: Flush every 5000 entities (low overhead)
*
* **Performance**:
* - Flush overhead: ~5-50ms per flush (~0.3% total time)
* - No configuration needed - works optimally out of the box
*
* @example
* ```typescript
* // Monitor import progress with live queries
* await brain.import(file, {
* onProgress: async (progress) => {
* console.log(`${progress.processed}/${progress.total}`)
*
* // Query data as it's imported!
* if (progress.queryable) {
* const count = await brain.count({ type: 'Product' })
* console.log(`${count} products imported so far`)
* }
* }
* })
* ```
*/
onProgress?: (progress: ImportProgress) => void | Promise<void>
}
/**
* Complete import options interface.
*/
export type ImportOptions = ValidImportOptions
export interface ImportProgress {
stage: 'detecting' | 'extracting' | 'storing-vfs' | 'storing-graph' | 'relationships' | 'complete'
/** Phase of import - extraction or relationship building */
phase?: 'extraction' | 'relationships'
message: string
processed?: number
/** Alias for processed, used in relationship phase */
current?: number
total?: number
entities?: number
relationships?: number
/** Rows per second */
throughput?: number
/** Estimated time remaining in ms */
eta?: number
/**
* Whether data is queryable at this point
*
* When true, indexes have been flushed and queries will return up-to-date results.
* When false, data exists in storage but indexes may not be current (queries may be slower/incomplete).
*
* Only present during streaming imports with flushInterval > 0.
*/
queryable?: boolean
}
export interface ImportResult {
/** Import ID for history tracking */
importId: string
/** Detected format */
format: SupportedFormat
/** Format detection confidence */
formatConfidence: number
/** VFS paths created */
vfs: {
rootPath: string
directories: string[]
files: Array<{
path: string
entityId?: string
type: 'entity' | 'metadata' | 'source' | 'relationships'
}>
}
/** Knowledge graph entities created */
entities: Array<{
id: string
name: string
type: NounType
vfsPath?: string
}>
/** Knowledge graph relationships created */
relationships: Array<{
id: string
from: string
to: string
type: VerbType
}>
/** Import statistics */
stats: {
entitiesExtracted: number
relationshipsInferred: number
vfsFilesCreated: number
graphNodesCreated: number
graphEdgesCreated: number
entitiesMerged: number
entitiesNew: number
processingTime: number
}
}
/**
* ImportCoordinator - Main entry point for all imports
*/
export class ImportCoordinator {
private brain: Brainy
private detector: FormatDetector
private history: ImportHistory
private backgroundDedup: BackgroundDeduplicator
private excelImporter: SmartExcelImporter
private pdfImporter: SmartPDFImporter
private csvImporter: SmartCSVImporter
private jsonImporter: SmartJSONImporter
private markdownImporter: SmartMarkdownImporter
private yamlImporter: SmartYAMLImporter
private docxImporter: SmartDOCXImporter
private vfsGenerator: VFSStructureGenerator
constructor(brain: Brainy) {
this.brain = brain
this.detector = new FormatDetector()
this.history = new ImportHistory(brain)
this.backgroundDedup = new BackgroundDeduplicator(brain)
this.excelImporter = new SmartExcelImporter(brain)
this.pdfImporter = new SmartPDFImporter(brain)
this.csvImporter = new SmartCSVImporter(brain)
this.jsonImporter = new SmartJSONImporter(brain)
this.markdownImporter = new SmartMarkdownImporter(brain)
this.yamlImporter = new SmartYAMLImporter(brain)
this.docxImporter = new SmartDOCXImporter(brain)
this.vfsGenerator = new VFSStructureGenerator(brain)
}
/**
* Initialize all importers
*/
async init(): Promise<void> {
await this.excelImporter.init()
await this.pdfImporter.init()
await this.csvImporter.init()
await this.jsonImporter.init()
await this.markdownImporter.init()
await this.yamlImporter.init()
await this.docxImporter.init()
await this.vfsGenerator.init()
await this.history.init()
}
/**
* Get import history
*/
getHistory() {
return this.history
}
/**
* Import from any source with auto-detection
* Now supports URL imports with authentication
*/
async import(
source: Buffer | string | object | ImportSource,
options: ImportOptions = {}
): Promise<ImportResult> {
const startTime = Date.now()
// Validate options (Reject deprecated options)
this.validateOptions(options)
// Normalize source (handles URL fetching)
const normalizedSource = await this.normalizeSource(source, options.format)
// Report detection stage
options.onProgress?.({
stage: 'detecting',
message: 'Detecting format...'
})
// Detect format
const detection = options.format
? { format: options.format, confidence: 1.0, evidence: ['Explicitly specified'] }
: this.detectFormat(normalizedSource)
if (!detection) {
throw new Error('Unable to detect file format. Please specify format explicitly.')
}
// Set defaults early (needed for tracking context)
// CRITICAL FIX: Spread options FIRST, then apply defaults
// Previously: ...options at the end overwrote normalized defaults with undefined
// Now: Defaults properly override undefined values
// Enable AI features by default for smarter imports
const opts = {
...options, // Spread first to get all options
vfsPath: options.vfsPath || `/imports/${Date.now()}`,
groupBy: options.groupBy || 'type',
createEntities: options.createEntities !== false,
createRelationships: options.createRelationships !== false,
preserveSource: options.preserveSource !== false,
enableDeduplication: options.enableDeduplication !== false,
enableNeuralExtraction: options.enableNeuralExtraction !== false, // Default true
enableRelationshipInference: options.enableRelationshipInference !== false, // Default true
enableConceptExtraction: options.enableConceptExtraction !== false, // Already defaults to true
deduplicationThreshold: options.deduplicationThreshold || 0.85
}
// Generate tracking context (Unified import/project tracking)
const importId = options.importId || uuidv4()
const projectId = options.projectId || this.deriveProjectId(opts.vfsPath)
const trackingContext: TrackingContext = {
importId,
projectId,
importedAt: Date.now(),
importFormat: detection.format,
importSource: normalizedSource.filename || 'unknown',
customMetadata: options.customMetadata || {}
}
// Report extraction stage
options.onProgress?.({
stage: 'extracting',
message: `Extracting entities from ${detection.format}...`
})
// Extract entities and relationships
const extractionResult = await this.extract(normalizedSource, detection.format, options)
// Report VFS storage stage
options.onProgress?.({
stage: 'storing-vfs',
message: 'Creating VFS structure...'
})
// Normalize extraction result to unified format
const normalizedResult = this.normalizeExtractionResult(extractionResult, detection.format)
// Create VFS structure
const vfsResult = await this.vfsGenerator.generate(normalizedResult, {
rootPath: opts.vfsPath,
groupBy: opts.groupBy,
customGrouping: opts.customGrouping,
preserveSource: opts.preserveSource,
// Fix sourceBuffer for file paths - type is 'path' not 'buffer' from normalizeSource()
sourceBuffer: Buffer.isBuffer(normalizedSource.data) ? normalizedSource.data as Buffer : undefined,
sourceFilename: normalizedSource.filename || `import.${detection.format}`,
createRelationshipFile: true,
createMetadataFile: true,
trackingContext, // Pass tracking metadata to VFS
// Pass progress callback for VFS creation updates
onProgress: (vfsProgress) => {
options.onProgress?.({
stage: 'storing-vfs',
message: vfsProgress.message,
processed: vfsProgress.processed,
total: vfsProgress.total
})
}
})
// Report graph storage stage
options.onProgress?.({
stage: 'storing-graph',
message: 'Creating knowledge graph...'
})
// Create entities and relationships in graph
const graphResult = await this.createGraphEntities(
normalizedResult,
vfsResult,
opts,
{
sourceFilename: normalizedSource.filename || `import.${detection.format}`,
format: detection.format
},
trackingContext // Pass tracking metadata to graph creation
)
// Report complete
options.onProgress?.({
stage: 'complete',
message: 'Import complete',
entities: graphResult.entities.length,
relationships: graphResult.relationships.length
})
const result: ImportResult = {
importId,
format: detection.format,
formatConfidence: detection.confidence,
vfs: {
rootPath: vfsResult.rootPath,
directories: vfsResult.directories,
files: vfsResult.files
},
entities: graphResult.entities,
relationships: graphResult.relationships,
stats: {
entitiesExtracted: extractionResult.entitiesExtracted,
relationshipsInferred: extractionResult.relationshipsInferred,
vfsFilesCreated: vfsResult.files.length,
graphNodesCreated: graphResult.entities.length,
graphEdgesCreated: graphResult.relationships.length,
entitiesMerged: graphResult.merged || 0,
entitiesNew: graphResult.newEntities || 0,
processingTime: Date.now() - startTime
}
}
// Record in history if enabled
if (options.enableHistory !== false) {
await this.history.recordImport(
importId,
{
type: normalizedSource.type === 'path' ? 'file' : normalizedSource.type as any,
filename: normalizedSource.filename,
format: detection.format
},
result
)
}
// CRITICAL FIX: Auto-flush all indexes before returning
// Ensures imported data survives server restarts
// Bug #5: Import data was only in memory, lost on restart
options.onProgress?.({
stage: 'complete',
message: 'Flushing indexes to disk...'
})
await this.brain.flush()
return result
}
/**
* Normalize source to ImportSource
* Now async to support URL fetching
*/
private async normalizeSource(
source: Buffer | string | object | ImportSource,
formatHint?: SupportedFormat
): Promise<ImportSource> {
// If already an ImportSource, handle URL fetching if needed
if (this.isImportSource(source)) {
if (source.type === 'url') {
return await this.fetchUrl(source)
}
return source
}
// Buffer
if (Buffer.isBuffer(source)) {
return {
type: 'buffer',
data: source
}
}
// String - could be URL, path, or content
if (typeof source === 'string') {
// Check if it's a URL
if (this.isUrl(source)) {
return await this.fetchUrl({
type: 'url',
data: source
})
}
// Check if it's a file path
if (this.isFilePath(source)) {
const buffer = fs.readFileSync(source)
return {
type: 'path',
data: buffer,
filename: path.basename(source)
}
}
// Otherwise treat as content
return {
type: 'string',
data: source
}
}
// Object
if (typeof source === 'object' && source !== null) {
return {
type: 'object',
data: source
}
}
throw new Error('Invalid source type. Expected Buffer, string, object, or ImportSource.')
}
/**
* Check if value is an ImportSource object
*/
private isImportSource(value: any): value is ImportSource {
return value && typeof value === 'object' && 'type' in value && 'data' in value
}
/**
* Check if string is a URL
*/
private isUrl(str: string): boolean {
try {
const url = new URL(str)
return url.protocol === 'http:' || url.protocol === 'https:'
} catch {
return false
}
}
/**
* Fetch content from URL
* Supports authentication and custom headers
*/
private async fetchUrl(source: ImportSource): Promise<ImportSource> {
const url = typeof source.data === 'string' ? source.data : String(source.data)
// Build headers
const headers: Record<string, string> = {
'User-Agent': 'Brainy/4.2.0',
...(source.headers || {})
}
// Add basic auth if provided
if (source.auth) {
const credentials = Buffer.from(`${source.auth.username}:${source.auth.password}`).toString('base64')
headers['Authorization'] = `Basic ${credentials}`
}
try {
const response = await fetch(url, { headers })
if (!response.ok) {
throw new Error(`HTTP ${response.status}: ${response.statusText}`)
}
// Get filename from URL or Content-Disposition header
const contentDisposition = response.headers.get('content-disposition')
let filename = source.filename
if (contentDisposition) {
const match = contentDisposition.match(/filename=["']?([^"';]+)["']?/)
if (match) filename = match[1]
}
if (!filename) {
filename = new URL(url).pathname.split('/').pop() || 'download'
}
// Get content type for format hint
const contentType = response.headers.get('content-type')
// Convert response to buffer
const arrayBuffer = await response.arrayBuffer()
const buffer = Buffer.from(arrayBuffer)
return {
type: 'buffer',
data: buffer,
filename,
headers: { 'content-type': contentType || 'application/octet-stream' }
}
} catch (error: any) {
throw new Error(`Failed to fetch URL ${url}: ${error.message}`)
}
}
/**
* Check if string is a file path
*/
private isFilePath(str: string): boolean {
// Check if file exists
try {
return fs.existsSync(str) && fs.statSync(str).isFile()
} catch {
return false
}
}
/**
* Detect format from source
*/
private detectFormat(source: ImportSource): { format: SupportedFormat; confidence: number; evidence: string[] } | null {
switch (source.type) {
case 'buffer':
case 'path':
const buffer = source.data as Buffer
let result = this.detector.detectFromBuffer(buffer)
// Try filename hint if buffer detection fails
if (!result && source.filename) {
result = this.detector.detectFromPath(source.filename)
}
return result
case 'string':
return this.detector.detectFromString(source.data as string)
case 'object':
return this.detector.detectFromObject(source.data)
case 'url':
// URL sources are converted to buffers in normalizeSource()
// This should never be reached, but included for type safety
return null
default:
return null
}
}
/**
* Extract entities using format-specific importer
*/
private async extract(
source: ImportSource,
format: SupportedFormat,
options: ImportOptions
): Promise<any> {
// Check if IntelligentImportAugmentation already extracted data
if ((options as any)._intelligentImport && (options as any)._extractedData) {
const extractedData = (options as any)._extractedData
// Convert extracted data to ExtractedRow format
const rows = extractedData.map((item: any) => ({
entity: {
id: item.id || `entity-${Date.now()}-${Math.random()}`,
name: item.name || item.type || 'Unnamed',
type: item.type || 'unknown',
description: item.description || '',
confidence: 1.0,
metadata: item.metadata || {}
},
relatedEntities: [],
relationships: []
}))
return {
rows,
entities: extractedData,
relationships: [],
metadata: (options as any)._metadata?.intelligentImport || {},
stats: {
byType: {},
byConfidence: {}
},
rowsProcessed: extractedData.length,
entitiesExtracted: extractedData.length,
relationshipsInferred: 0,
processingTime: 0
}
}
const extractOptions = {
enableNeuralExtraction: options.enableNeuralExtraction !== false,
enableRelationshipInference: options.enableRelationshipInference !== false,
enableConceptExtraction: options.enableConceptExtraction !== false,
confidenceThreshold: options.confidenceThreshold || 0.6,
onProgress: (stats: any) => {
// Enhanced progress reporting with throughput and ETA
const message = stats.throughput
? `Extracting entities from ${format} (${stats.throughput} rows/sec, ETA: ${Math.round(stats.eta / 1000)}s)...`
: `Extracting entities from ${format}...`
options.onProgress?.({
stage: 'extracting',
message,
processed: stats.processed,
total: stats.total,
entities: stats.entities,
relationships: stats.relationships,
// Pass through enhanced metrics if available
throughput: stats.throughput,
eta: stats.eta
})
}
}
switch (format) {
case 'excel':
const buffer = source.type === 'buffer' || source.type === 'path'
? source.data as Buffer
: Buffer.from(JSON.stringify(source.data))
return await this.excelImporter.extract(buffer, extractOptions)
case 'pdf':
const pdfBuffer = source.data as Buffer
return await this.pdfImporter.extract(pdfBuffer, extractOptions)
case 'csv':
const csvBuffer = source.type === 'buffer' || source.type === 'path'
? source.data as Buffer
: Buffer.from(source.data as string)
return await this.csvImporter.extract(csvBuffer, extractOptions)
case 'json':
const jsonData = source.type === 'object'
? source.data
: source.type === 'string'
? source.data as string
: (source.data as Buffer).toString('utf8')
return await this.jsonImporter.extract(jsonData, extractOptions)
case 'markdown':
const mdContent = source.type === 'string'
? source.data as string
: (source.data as Buffer).toString('utf8')
return await this.markdownImporter.extract(mdContent, extractOptions)
case 'yaml':
const yamlContent = source.type === 'string'
? source.data as string
: source.type === 'buffer' || source.type === 'path'
? (source.data as Buffer).toString('utf8')
: JSON.stringify(source.data)
return await this.yamlImporter.extract(yamlContent, extractOptions)
case 'docx':
const docxBuffer = source.type === 'buffer' || source.type === 'path'
? source.data as Buffer
: Buffer.from(JSON.stringify(source.data))
return await this.docxImporter.extract(docxBuffer, extractOptions)
case 'image':
// Images are handled by IntelligentImportAugmentation
// If we reach here, augmentation didn't process it - return minimal result
const imageName = source.filename || 'image'
const imageId = `image-${Date.now()}`
return {
rows: [{
entity: {
id: imageId,
name: imageName,
type: 'media' as any,
description: '',
confidence: 1.0,
metadata: { subtype: 'image' }
},
relatedEntities: [],
relationships: []
}],
entities: [{
id: imageId,
name: imageName,
type: 'media',
metadata: { subtype: 'image' }
}],
relationships: [],
metadata: {},
stats: {
byType: { media: 1 },
byConfidence: { high: 1 }
},
rowsProcessed: 1,
entitiesExtracted: 1,
relationshipsInferred: 0,
processingTime: 0
}
default:
throw new Error(`Unsupported format: ${format}`)
}
}
/**
* Create entities and relationships in knowledge graph
* Added sourceInfo parameter for document entity creation
*/
private async createGraphEntities(
extractionResult: any,
vfsResult: any,
options: ImportOptions,
sourceInfo?: {
sourceFilename: string
format: string
},
trackingContext?: TrackingContext // Import/project tracking
): Promise<{
entities: Array<{ id: string; name: string; type: NounType; vfsPath?: string; metadata?: Record<string, any> }>
relationships: Array<{ id: string; from: string; to: string; type: VerbType }>
merged: number
newEntities: number
documentEntity?: string
provenanceCount?: number
}> {
const entities: Array<{ id: string; name: string; type: NounType; vfsPath?: string; metadata?: Record<string, any> }> = []
const relationships: Array<{ id: string; from: string; to: string; type: VerbType }> = []
let mergedCount = 0
let newCount = 0
// CRITICAL FIX: Default to true when undefined
// Previously: if (!options.createEntities) treated undefined as false
// Now: Only skip when explicitly set to false
if (options.createEntities === false) {
return {
entities,
relationships,
merged: 0,
newEntities: 0,
documentEntity: undefined,
provenanceCount: 0
}
}
// Extract rows/sections/entities from result (unified across formats)
const rows = extractionResult.rows || extractionResult.sections || extractionResult.entities || []
// Progressive flush interval - adjusts based on current count
// Starts at 100, increases to 1000 at 1K entities, then 5000 at 10K
// This works for both known totals (files) and unknown totals (streaming APIs)
let currentFlushInterval = 100 // Start with frequent updates for better UX
let entitiesSinceFlush = 0
let totalFlushes = 0
console.log(
`📊 Streaming Import: Progressive flush intervals\n` +
` Starting interval: Every ${currentFlushInterval} entities\n` +
` Auto-adjusts: 100 → 1000 (at 1K entities) → 5000 (at 10K entities)\n` +
` Benefits: Live queries, crash resilience, frequent early updates\n` +
` Works with: Known totals (files) and unknown totals (streaming APIs)`
)
// Smart deduplication auto-disable for large imports (prevents O(n²) performance)
const DEDUPLICATION_AUTO_DISABLE_THRESHOLD = 100
let actuallyEnableDeduplication = options.enableDeduplication
if (options.enableDeduplication && rows.length > DEDUPLICATION_AUTO_DISABLE_THRESHOLD) {
actuallyEnableDeduplication = false
console.log(
`📊 Smart Import: Auto-disabled deduplication for large import (${rows.length} entities > ${DEDUPLICATION_AUTO_DISABLE_THRESHOLD} threshold)\n` +
` Reason: Deduplication performs O(n²) vector searches which is too slow for large datasets\n` +
` Tip: For large imports, deduplicate manually after import or use smaller batches\n` +
` Override: Set deduplicationThreshold to force enable (not recommended for >500 entities)`
)
}
// ============================================
// Create document entity for import source
// ============================================
let documentEntityId: string | null = null
let provenanceCount = 0
if (sourceInfo && options.createProvenanceLinks !== false) {
console.log(`📄 Creating document entity for import source: ${sourceInfo.sourceFilename}`)
// Subtype `import-source` distinguishes the synthetic Document entity that
// represents the import operation itself (the file being imported) from
// entities extracted from its contents. Also satisfies enforcement when a
// consumer registers a vocabulary on NounType.Document (added 7.30.1).
documentEntityId = await this.brain.add({
data: sourceInfo.sourceFilename,
type: NounType.Document,
subtype: 'import-source',
metadata: {
name: sourceInfo.sourceFilename,
sourceFile: sourceInfo.sourceFilename,
format: sourceInfo.format,
importSource: true,
vfsPath: vfsResult.rootPath,
totalRows: rows.length,
byType: this.countByType(rows),
// Import tracking metadata
...(trackingContext && {
importIds: [trackingContext.importId],
projectId: trackingContext.projectId,
importedAt: trackingContext.importedAt,
importFormat: trackingContext.importFormat,
importSource: trackingContext.importSource,
...trackingContext.customMetadata
})
}
})
console.log(`✅ Document entity created: ${documentEntityId}`)
}
// ============================================
// Batch entity creation using addMany()
// Replaces entity-by-entity loop for 10-100x performance improvement on cloud storage
// ============================================
if (!actuallyEnableDeduplication) {
// FAST PATH: Batch creation without deduplication (recommended for imports > 100 entities)
const importSource = vfsResult.rootPath
// Prepare all entity parameters upfront. Mirror the subtype resolution from
// the deduplication path above: preserve extractor-set subtype if any, else
// fall back to caller-supplied default, else `'imported'` (added 7.30.1).
const entityParams = rows.map((row: any) => {
const entity = row.entity || row
const vfsFile = vfsResult.files.find((f: any) => f.entityId === entity.id)
return {
data: entity.description || entity.name,
type: entity.type,
subtype: entity.subtype ?? options.defaultSubtype ?? 'imported',
metadata: {
...entity.metadata,
name: entity.name,
confidence: entity.confidence,
vfsPath: vfsFile?.path,
importedFrom: 'import-coordinator',
imports: [importSource],
...(trackingContext && {
importIds: [trackingContext.importId],
projectId: trackingContext.projectId,
importedAt: trackingContext.importedAt,
importFormat: trackingContext.importFormat,
importSource: trackingContext.importSource,
sourceRow: row.rowNumber,
sourceSheet: row.sheet,
...trackingContext.customMetadata
})
}
}
})
// Batch create all entities (storage-aware batching handles rate limits automatically)
const addResult = await this.brain.addMany({
items: entityParams,
continueOnError: true,
onProgress: (done, total) => {
options.onProgress?.({
stage: 'storing-graph',
message: `Creating entities: ${done}/${total}`,
processed: done,
total,
entities: done
})
}
})
// Map results to entities array and update rows with new IDs
for (let i = 0; i < addResult.successful.length; i++) {
const entityId = addResult.successful[i]
const row = rows[i]
const entity = row.entity || row
const vfsFile = vfsResult.files.find((f: any) => f.entityId === entity.id)
entity.id = entityId
entities.push({
id: entityId,
name: entity.name,
type: entity.type,
vfsPath: vfsFile?.path,
metadata: entity.metadata // Include metadata in return (for ImageHandler, etc)
})
newCount++
}
// Handle failed entities
if (addResult.failed.length > 0) {
console.warn(`⚠️ ${addResult.failed.length} entities failed to create`)
}
// Create provenance links in batch
if (documentEntityId && options.createProvenanceLinks !== false && entities.length > 0) {
const provenanceParams = entities.map((entity, idx) => {
const row = rows[idx]
return {
from: documentEntityId,
to: entity.id,
type: VerbType.Contains,
metadata: {
relationshipType: 'provenance',
evidence: `Extracted from ${sourceInfo?.sourceFilename}`,
sheet: row?.sheet,
rowNumber: row?.rowNumber,
extractedAt: Date.now(),
format: sourceInfo?.format,
...(trackingContext && {
importIds: [trackingContext.importId],
projectId: trackingContext.projectId,
importFormat: trackingContext.importFormat,
...trackingContext.customMetadata
})
}
}
})
await this.brain.relateMany({
items: provenanceParams,
continueOnError: true
})
provenanceCount = provenanceParams.length
}
} else {
// SLOW PATH: Entity-by-entity with deduplication (only for small imports < 100 entities)
for (const row of rows) {
const entity = row.entity || row
const vfsFile = vfsResult.files.find((f: any) => f.entityId === entity.id)
try {
const importSource = vfsResult.rootPath
let entityId: string
// No deduplication during import (12-24x speedup)
// Background deduplication runs 5 minutes after import completes.
// Preserves any subtype the extractor already set on the entity; falls back
// to the caller-supplied `options.defaultSubtype` or to the Brainy-default
// `'imported'` so enforcement doesn't fire (added 7.30.1).
entityId = await this.brain.add({
data: entity.description || entity.name,
type: entity.type,
subtype: entity.subtype ?? options.defaultSubtype ?? 'imported',
metadata: {
...entity.metadata,
name: entity.name,
confidence: entity.confidence,
vfsPath: vfsFile?.path,
importedFrom: 'import-coordinator',
// Import tracking metadata
...(trackingContext && {
importId: trackingContext.importId, // Used for background dedup
importIds: [trackingContext.importId],
projectId: trackingContext.projectId,
importedAt: trackingContext.importedAt,
importFormat: trackingContext.importFormat,
importSource: trackingContext.importSource,
sourceRow: row.rowNumber,
sourceSheet: row.sheet,
...trackingContext.customMetadata
})
}
})
newCount++
// Update entity ID in extraction result
entity.id = entityId
entities.push({
id: entityId,
name: entity.name,
type: entity.type,
vfsPath: vfsFile?.path,
metadata: entity.metadata // Include metadata in return (for ImageHandler, etc)
})
// ============================================
// Create provenance relationship (document → entity)
// ============================================
if (documentEntityId && options.createProvenanceLinks !== false) {
await this.brain.relate({
from: documentEntityId,
to: entityId,
type: VerbType.Contains,
metadata: {
relationshipType: 'provenance',
evidence: `Extracted from ${sourceInfo?.sourceFilename}`,
sheet: row.sheet,
rowNumber: row.rowNumber,
extractedAt: Date.now(),
format: sourceInfo?.format,
// Import tracking metadata (`createdAt` is reserved — the
// relationship's own creation time is system-managed, and the
// import timestamp already travels as `extractedAt`)
...(trackingContext && {
importIds: [trackingContext.importId],
projectId: trackingContext.projectId,
importFormat: trackingContext.importFormat,
...trackingContext.customMetadata
})
}
})
provenanceCount++
}
// Collect relationships for batch creation
if (options.createRelationships && row.relationships) {
for (const rel of row.relationships) {
try {
// CRITICAL FIX: Prevent infinite placeholder creation loop
// Find or create target entity using EXACT matching only
let targetEntityId: string | undefined
// STEP 1: Check if target already exists in entities list (includes placeholders)
// This prevents creating duplicate placeholders - the root cause of Bug #1
const existingTarget = entities.find(e =>
e.name.toLowerCase() === rel.to.toLowerCase()
)
if (existingTarget) {
targetEntityId = existingTarget.id
} else {
// STEP 2: Try to find in extraction results (rows)
// FIX: Use EXACT matching instead of fuzzy .includes()
// Fuzzy matching caused false matches (e.g., "Entity_29" matching "Entity_297")
for (const otherRow of rows) {
const otherEntity = otherRow.entity || otherRow
if (otherEntity.name.toLowerCase() === rel.to.toLowerCase()) {
targetEntityId = otherEntity.id
break
}
}
// STEP 3: If still not found, create placeholder entity ONCE
// The placeholder is added to entities array, so future searches will find it.
// Subtype `import-placeholder` marks these as synthetic targets (not real
// imports) so downstream queries can distinguish them and dedup runs can
// safely consolidate them with real entities later (added 7.30.1).
if (!targetEntityId) {
targetEntityId = await this.brain.add({
data: rel.to,
type: NounType.Thing,
subtype: 'import-placeholder',
metadata: {
name: rel.to,
placeholder: true,
inferredFrom: entity.name,
// Import tracking metadata
...(trackingContext && {
importIds: [trackingContext.importId],
projectId: trackingContext.projectId,
importedAt: trackingContext.importedAt,
importFormat: trackingContext.importFormat,
...trackingContext.customMetadata
})
}
})
// CRITICAL: Add to entities array so future searches find it
entities.push({
id: targetEntityId,
name: rel.to,
type: NounType.Thing
})
}
}
// Add to relationships array with target ID for batch processing
relationships.push({
id: '', // Will be assigned after batch creation
from: entityId,
to: targetEntityId,
type: rel.type,
confidence: rel.confidence, // Top-level field
weight: rel.weight || 1.0, // Top-level field
metadata: {
evidence: rel.evidence,
// Import tracking metadata (will be merged in batch creation)
...(trackingContext && {
importIds: [trackingContext.importId],
projectId: trackingContext.projectId,
importedAt: trackingContext.importedAt,
importFormat: trackingContext.importFormat,
...trackingContext.customMetadata
})
}
} as any)
} catch (error) {
// Skip relationship collection errors (entity might not exist, etc.)
continue
}
}
}
// Streaming import: Progressive flush with dynamic interval adjustment
entitiesSinceFlush++
if (entitiesSinceFlush >= currentFlushInterval) {
const flushStart = Date.now()
await this.brain.flush()
const flushDuration = Date.now() - flushStart
totalFlushes++
// Reset counter
entitiesSinceFlush = 0
// Recalculate flush interval based on current entity count
const newInterval = this.getProgressiveFlushInterval(entities.length)
if (newInterval !== currentFlushInterval) {
console.log(
`📊 Flush interval adjusted: ${currentFlushInterval}${newInterval}\n` +
` Reason: Reached ${entities.length} entities (threshold for next tier)\n` +
` Impact: ${newInterval > currentFlushInterval ? 'Fewer' : 'More'} flushes = ${newInterval > currentFlushInterval ? 'Better performance' : 'More frequent updates'}`
)
currentFlushInterval = newInterval
}
// Notify progress callback that data is now queryable
await options.onProgress?.({
stage: 'storing-graph',
message: `Flushed indexes (${entities.length}/${rows.length} entities, ${flushDuration}ms)`,
processed: entities.length,
total: rows.length,
entities: entities.length,
queryable: true // ← Indexes are flushed, data is queryable!
})
}
} catch (error) {
// Skip entity creation errors (might already exist, etc.)
continue
}
}
} // End of deduplication else block
// Final flush for any remaining entities
if (entitiesSinceFlush > 0) {
const flushStart = Date.now()
await this.brain.flush()
const flushDuration = Date.now() - flushStart
totalFlushes++
console.log(
`✅ Import complete: ${entities.length} entities processed\n` +
` Total flushes: ${totalFlushes}\n` +
` Final flush: ${flushDuration}ms\n` +
` Average overhead: ~${((totalFlushes * 50) / (entities.length * 100) * 100).toFixed(2)}%`
)
await options.onProgress?.({
stage: 'storing-graph',
message: `Final flush complete (${entities.length} entities)`,
processed: entities.length,
total: rows.length,
entities: entities.length,
queryable: true
})
}
// Batch create all relationships using brain.relateMany() for performance
// Enhanced with type-based inference and semantic metadata
if (options.createRelationships && relationships.length > 0) {
try {
const relationshipParams = relationships.map(rel => {
// Get entity types for inference
const sourceEntity = entities.find(e => e.id === rel.from)
const targetEntity = entities.find(e => e.id === rel.to)
// Infer better relationship type if generic and we have entity types
let verbType = rel.type
if (verbType === VerbType.RelatedTo && sourceEntity && targetEntity) {
verbType = this.inferRelationshipType(
sourceEntity.type,
targetEntity.type,
(rel as any).metadata?.evidence
)
}
return {
from: rel.from,
to: rel.to,
type: verbType, // Enhanced type
metadata: {
...((rel as any).metadata || {}),
relationshipType: 'semantic', // Distinguish from VFS/provenance
inferredType: verbType !== rel.type, // Track if type was enhanced
originalType: rel.type
}
}
})
const relationshipIds = await this.brain.relateMany({
items: relationshipParams,
parallel: true,
chunkSize: 100,
continueOnError: true,
onProgress: (done, total) => {
options.onProgress?.({
stage: 'storing-graph',
phase: 'relationships',
message: `Building relationships: ${done}/${total}`,
current: done,
processed: done,
total: total,
entities: entities.length,
relationships: done
})
}
})
// Update relationship IDs
relationshipIds.forEach((id, index) => {
if (id && relationships[index]) {
relationships[index].id = id
}
})
} catch (error) {
console.warn('Error creating relationships in batch:', error)
// Continue - relationships are optional
}
}
// Schedule background deduplication (debounced 5 minutes)
if (trackingContext && trackingContext.importId) {
this.backgroundDedup.scheduleDedup(trackingContext.importId)
}
return {
entities,
relationships,
merged: mergedCount,
newEntities: newCount,
documentEntity: documentEntityId || undefined,
provenanceCount
}
}
/**
* Normalize extraction result to unified format (Excel-like structure)
*/
private normalizeExtractionResult(result: any, format: SupportedFormat): any {
// Excel and CSV already have the right format
if (format === 'excel' || format === 'csv') {
return result
}
// PDF: sections -> rows
if (format === 'pdf') {
const rows = result.sections.flatMap((section: any) =>
section.entities.map((entity: any) => ({
entity,
relatedEntities: [],
relationships: section.relationships.filter((r: any) => r.from === entity.id),
concepts: section.concepts || []
}))
)
return {
rowsProcessed: result.sectionsProcessed,
entitiesExtracted: result.entitiesExtracted,
relationshipsInferred: result.relationshipsInferred,
rows,
entityMap: result.entityMap,
processingTime: result.processingTime,
stats: result.stats
}
}
// JSON: entities -> rows
if (format === 'json') {
const rows = result.entities.map((entity: any) => ({
entity,
relatedEntities: [],
relationships: result.relationships.filter((r: any) => r.from === entity.id),
concepts: entity.metadata?.concepts || []
}))
return {
rowsProcessed: result.nodesProcessed,
entitiesExtracted: result.entitiesExtracted,
relationshipsInferred: result.relationshipsInferred,
rows,
entityMap: result.entityMap,
processingTime: result.processingTime,
stats: result.stats
}
}
// Markdown: sections -> rows
if (format === 'markdown') {
const rows = result.sections.flatMap((section: any) =>
section.entities.map((entity: any) => ({
entity,
relatedEntities: [],
relationships: section.relationships.filter((r: any) => r.from === entity.id),
concepts: section.concepts || []
}))
)
return {
rowsProcessed: result.sectionsProcessed,
entitiesExtracted: result.entitiesExtracted,
relationshipsInferred: result.relationshipsInferred,
rows,
entityMap: result.entityMap,
processingTime: result.processingTime,
stats: result.stats
}
}
// YAML: entities -> rows
if (format === 'yaml') {
const rows = result.entities.map((entity: any) => ({
entity,
relatedEntities: [],
relationships: result.relationships.filter((r: any) => r.from === entity.id),
concepts: entity.metadata?.concepts || []
}))
return {
rowsProcessed: result.nodesProcessed,
entitiesExtracted: result.entitiesExtracted,
relationshipsInferred: result.relationshipsInferred,
rows,
entityMap: result.entityMap,
processingTime: result.processingTime,
stats: result.stats
}
}
// DOCX: entities -> rows
if (format === 'docx') {
const rows = result.entities.map((entity: any) => ({
entity,
relatedEntities: [],
relationships: result.relationships.filter((r: any) => r.from === entity.id),
concepts: entity.metadata?.concepts || []
}))
return {
rowsProcessed: result.paragraphsProcessed,
entitiesExtracted: result.entitiesExtracted,
relationshipsInferred: result.relationshipsInferred,
rows,
entityMap: result.entityMap,
processingTime: result.processingTime,
stats: result.stats
}
}
// Fallback: return as-is
return result
}
/**
* Validate options and reject deprecated v3.x options
* Throws clear errors with migration guidance
*/
private validateOptions(options: any): void {
const invalidOptions: Array<{ old: string; new: string; message: string }> = []
// Check for v3.x deprecated options
if ('extractRelationships' in options) {
invalidOptions.push({
old: 'extractRelationships',
new: 'enableRelationshipInference',
message: 'Option renamed for clarity in v4.x - explicitly indicates AI-powered relationship inference'
})
}
if ('autoDetect' in options) {
invalidOptions.push({
old: 'autoDetect',
new: '(removed)',
message: 'Auto-detection is now always enabled - no need to specify this option'
})
}
if ('createFileStructure' in options) {
invalidOptions.push({
old: 'createFileStructure',
new: 'vfsPath',
message: 'Use vfsPath to explicitly specify the virtual filesystem directory path'
})
}
if ('excelSheets' in options) {
invalidOptions.push({
old: 'excelSheets',
new: '(removed)',
message: 'All sheets are now processed automatically - no configuration needed'
})
}
if ('pdfExtractTables' in options) {
invalidOptions.push({
old: 'pdfExtractTables',
new: '(removed)',
message: 'Table extraction is now automatic for PDF imports'
})
}
// If invalid options found, throw error with detailed message
if (invalidOptions.length > 0) {
const errorMessage = this.buildValidationErrorMessage(invalidOptions)
throw new Error(errorMessage)
}
}
/**
* Build detailed error message for invalid options
* Respects LOG_LEVEL for verbosity (detailed in dev, concise in prod)
*/
private buildValidationErrorMessage(
invalidOptions: Array<{ old: string; new: string; message: string }>
): string {
// Check environment for verbosity level
const verbose =
process.env.LOG_LEVEL === 'debug' ||
process.env.LOG_LEVEL === 'verbose' ||
process.env.NODE_ENV === 'development' ||
process.env.NODE_ENV === 'dev'
if (verbose) {
// DETAILED mode (development)
const optionDetails = invalidOptions
.map(
(opt) => `
${opt.old}
→ Use: ${opt.new}
→ Why: ${opt.message}`
)
.join('\n')
return `
❌ Invalid import options detected (Brainy v4.x breaking changes)
The following v3.x options are no longer supported:
${optionDetails}
📖 Migration Guide: https://brainy.dev/docs/guides/migrating-to-v4
💡 Quick Fix Examples:
Before (v3.x):
await brain.import(file, {
extractRelationships: true,
createFileStructure: true
})
After (v4.x):
await brain.import(file, {
enableRelationshipInference: true,
vfsPath: '/imports/my-data'
})
🔗 Full API docs: https://brainy.dev/docs/api/import
`.trim()
} else {
// CONCISE mode (production)
const optionsList = invalidOptions.map((o) => `'${o.old}'`).join(', ')
return `Invalid import options: ${optionsList}. See https://brainy.dev/docs/guides/migrating-to-v4`
}
}
/**
* Derive project ID from VFS path
* Extracts meaningful project name from path, avoiding timestamps
*
* Examples:
* - /imports/myproject → "myproject"
* - /imports/2024-01-15/myproject → "myproject"
* - /imports/1234567890 → "import_1234567890"
* - /my-game/characters → "my-game"
*
* @param vfsPath - VFS path to derive project ID from
* @returns Derived project identifier
*/
private deriveProjectId(vfsPath: string): string {
// Extract meaningful project name from vfsPath
const segments = vfsPath.split('/').filter(s => s.length > 0)
if (segments.length === 0) {
return 'default_project'
}
// If path starts with /imports/, look for meaningful segment
if (segments[0] === 'imports') {
if (segments.length === 1) {
return 'default_project'
}
const lastSegment = segments[segments.length - 1]
// If last segment looks like a timestamp, use parent
if (/^\d{4}-\d{2}-\d{2}$/.test(lastSegment) || /^\d{10,}$/.test(lastSegment)) {
// Use parent segment if available
if (segments.length >= 3) {
return segments[segments.length - 2]
}
return `import_${lastSegment}`
}
return lastSegment
}
// For non-/imports/ paths, use first segment as project
return segments[0]
}
/**
* Get progressive flush interval based on CURRENT entity count
*
* Unlike adaptive intervals (which require knowing total count upfront),
* progressive intervals adjust dynamically as import proceeds.
*
* Thresholds:
* - 0-999 entities: Flush every 100 (frequent updates for better UX)
* - 1K-9.9K entities: Flush every 1000 (balanced performance/responsiveness)
* - 10K+ entities: Flush every 5000 (performance focused, minimal overhead)
*
* Benefits:
* - Works with known totals (file imports)
* - Works with unknown totals (streaming APIs, database cursors)
* - Frequent updates early when user is watching
* - Efficient processing later when performance matters
* - Low overhead (~0.3% for large imports)
* - No configuration required
*
* Example:
* - Import with 50K entities:
* - Flushes at: 100, 200, ..., 900 (9 flushes with interval=100)
* - Interval increases to 1000 at entity #1000
* - Flushes at: 1000, 2000, ..., 9000 (9 more flushes)
* - Interval increases to 5000 at entity #10000
* - Flushes at: 10000, 15000, ..., 50000 (8 more flushes)
* - Total: ~26 flushes = ~1.3s overhead = 0.026% of import time
*
* @param currentEntityCount - Current number of entities imported so far
* @returns Current optimal flush interval
*/
private getProgressiveFlushInterval(currentEntityCount: number): number {
if (currentEntityCount < 1000) {
return 100 // Frequent updates for small imports and early stages
} else if (currentEntityCount < 10000) {
return 1000 // Balanced interval for medium-sized imports
} else {
return 5000 // Performance-focused interval for large imports
}
}
/**
* Infer relationship type based on entity types and context
* Semantic relationship enhancement
*
* @param sourceType - Type of source entity
* @param targetType - Type of target entity
* @param context - Optional context string for additional hints
* @returns Inferred verb type
*/
private inferRelationshipType(
sourceType: NounType,
targetType: NounType,
context?: string
): VerbType {
// Context-based inference (highest priority)
if (context) {
const lowerContext = context.toLowerCase()
if (lowerContext.includes('live') || lowerContext.includes('reside') || lowerContext.includes('dwell')) {
return VerbType.LocatedAt
}
if (lowerContext.includes('create') || lowerContext.includes('invent') || lowerContext.includes('make')) {
return VerbType.Creates
}
if (lowerContext.includes('own') || lowerContext.includes('possess') || lowerContext.includes('belong')) {
return VerbType.PartOf
}
if (lowerContext.includes('work') || lowerContext.includes('collaborate') || lowerContext.includes('team')) {
return VerbType.WorksWith
}
if (lowerContext.includes('use') || lowerContext.includes('wield') || lowerContext.includes('employ')) {
return VerbType.Uses
}
if (lowerContext.includes('know') || lowerContext.includes('friend') || lowerContext.includes('ally')) {
return VerbType.FriendOf
}
}
// Type-based inference (fallback)
// Sort types for consistent lookup
const sortedTypes = [sourceType, targetType].sort()
const typeKey = `${sortedTypes[0]}+${sortedTypes[1]}`
const typeMapping: Record<string, VerbType> = {
// Person relationships
[`${NounType.Person}+${NounType.Location}`]: VerbType.LocatedAt,
[`${NounType.Person}+${NounType.Thing}`]: VerbType.Uses,
[`${NounType.Person}+${NounType.Person}`]: VerbType.FriendOf,
[`${NounType.Person}+${NounType.Concept}`]: VerbType.RelatedTo,
[`${NounType.Person}+${NounType.Event}`]: VerbType.RelatedTo,
// Location relationships
[`${NounType.Location}+${NounType.Thing}`]: VerbType.Contains,
[`${NounType.Location}+${NounType.Concept}`]: VerbType.RelatedTo,
[`${NounType.Location}+${NounType.Event}`]: VerbType.LocatedAt,
// Thing relationships
[`${NounType.Thing}+${NounType.Concept}`]: VerbType.RelatedTo,
[`${NounType.Thing}+${NounType.Event}`]: VerbType.RelatedTo,
// Concept relationships
[`${NounType.Concept}+${NounType.Concept}`]: VerbType.RelatedTo,
[`${NounType.Concept}+${NounType.Event}`]: VerbType.RelatedTo,
// Event relationships
[`${NounType.Event}+${NounType.Event}`]: VerbType.Precedes
}
return typeMapping[typeKey] || VerbType.RelatedTo
}
/**
* Count entities by type for document metadata
* Used for document entity statistics
*
* @param rows - Extracted rows from import
* @returns Record of entity type counts
*/
private countByType(rows: any[]): Record<string, number> {
const counts: Record<string, number> = {}
for (const row of rows) {
const entity = row.entity || row
const type = entity.type || NounType.Thing
counts[type] = (counts[type] || 0) + 1
}
return counts
}
}