2025-08-29 15:39:07 -07:00
/ * *
* Model Configuration Auto - Selection
* Intelligently selects model precision based on environment
* while allowing manual override
* /
import { isBrowser , isNode } from '../utils/environment.js'
2025-09-02 10:00:52 -07:00
import { setModelPrecision } from './modelPrecisionManager.js'
2025-08-29 15:39:07 -07:00
export type ModelPrecision = 'fp32' | 'q8'
export type ModelPreset = 'fast' | 'small' | 'auto'
interface ModelConfigResult {
precision : ModelPrecision
reason : string
autoSelected : boolean
}
/ * *
* Auto - select model precision based on environment and resources
2025-09-02 10:00:52 -07:00
* DEFAULT : Q8 for optimal size / performance balance
2025-08-29 15:39:07 -07:00
* @param override - Manual override : 'fp32' , 'q8' , 'fast' ( fp32 ) , 'small' ( q8 ) , or 'auto'
* /
export function autoSelectModelPrecision ( override? : ModelPrecision | ModelPreset ) : ModelConfigResult {
// Handle direct precision override
if ( override === 'fp32' || override === 'q8' ) {
2025-09-02 10:00:52 -07:00
setModelPrecision ( override ) // Update central config
2025-08-29 15:39:07 -07:00
return {
precision : override ,
reason : ` Manually specified: ${ override } ` ,
autoSelected : false
}
}
// Handle preset overrides
if ( override === 'fast' ) {
2025-09-02 10:00:52 -07:00
setModelPrecision ( 'fp32' ) // Update central config
2025-08-29 15:39:07 -07:00
return {
precision : 'fp32' ,
reason : 'Preset: fast (fp32 for best quality)' ,
autoSelected : false
}
}
if ( override === 'small' ) {
2025-09-02 10:00:52 -07:00
setModelPrecision ( 'q8' ) // Update central config
2025-08-29 15:39:07 -07:00
return {
precision : 'q8' ,
reason : 'Preset: small (q8 for reduced size)' ,
autoSelected : false
}
}
// Auto-selection logic
return autoDetectBestPrecision ( )
}
/ * *
* Automatically detect the best model precision for the environment
2025-09-02 10:00:52 -07:00
* NEW DEFAULT : Q8 for optimal size / performance ( 75 % smaller , 99 % accuracy )
2025-08-29 15:39:07 -07:00
* /
function autoDetectBestPrecision ( ) : ModelConfigResult {
2025-09-02 10:00:52 -07:00
// Check if user explicitly wants FP32 via environment variable
if ( process . env . BRAINY_FORCE_FP32 === 'true' ) {
setModelPrecision ( 'fp32' )
return {
precision : 'fp32' ,
reason : 'FP32 forced via BRAINY_FORCE_FP32 environment variable' ,
autoSelected : false
}
}
2025-08-29 15:39:07 -07:00
// Browser environment - use Q8 for smaller download/memory
if ( isBrowser ( ) ) {
2025-09-02 10:00:52 -07:00
setModelPrecision ( 'q8' )
2025-08-29 15:39:07 -07:00
return {
precision : 'q8' ,
2025-09-02 10:00:52 -07:00
reason : 'Browser environment - using Q8 (23MB vs 90MB)' ,
2025-08-29 15:39:07 -07:00
autoSelected : true
}
}
// Serverless environments - use Q8 for faster cold starts
if ( isServerlessEnvironment ( ) ) {
2025-09-02 10:00:52 -07:00
setModelPrecision ( 'q8' )
2025-08-29 15:39:07 -07:00
return {
precision : 'q8' ,
2025-09-02 10:00:52 -07:00
reason : 'Serverless environment - using Q8 for 75% faster cold starts' ,
2025-08-29 15:39:07 -07:00
autoSelected : true
}
}
// Check available memory
const memoryMB = getAvailableMemoryMB ( )
2025-09-02 10:00:52 -07:00
// Only use FP32 if explicitly high memory AND user opts in
if ( memoryMB >= 4096 && process . env . BRAINY_PREFER_QUALITY === 'true' ) {
setModelPrecision ( 'fp32' )
2025-08-29 15:39:07 -07:00
return {
precision : 'fp32' ,
2025-09-02 10:00:52 -07:00
reason : ` High memory ( ${ memoryMB } MB) + quality preference - using FP32 ` ,
2025-08-29 15:39:07 -07:00
autoSelected : true
}
}
2025-09-02 10:00:52 -07:00
// DEFAULT TO Q8 - Optimal for 99% of use cases
// Q8 provides 99% accuracy at 25% of the size
setModelPrecision ( 'q8' )
2025-08-29 15:39:07 -07:00
return {
precision : 'q8' ,
2025-09-02 10:00:52 -07:00
reason : 'Default: Q8 model (23MB, 99% accuracy, 4x faster loads)' ,
2025-08-29 15:39:07 -07:00
autoSelected : true
}
}
/ * *
* Check if running in a serverless environment
* /
function isServerlessEnvironment ( ) : boolean {
if ( ! isNode ( ) ) return false
return ! ! (
process . env . AWS_LAMBDA_FUNCTION_NAME ||
process . env . VERCEL ||
process . env . NETLIFY ||
process . env . CLOUDFLARE_WORKERS ||
process . env . FUNCTIONS_WORKER_RUNTIME ||
process . env . K_SERVICE // Google Cloud Run
)
}
/ * *
* Get available memory in MB
* /
function getAvailableMemoryMB ( ) : number {
if ( isBrowser ( ) ) {
// @ts-ignore - navigator.deviceMemory is experimental
if ( navigator . deviceMemory ) {
// @ts-ignore
return navigator . deviceMemory * 1024 // Device memory in GB
}
return 256 // Conservative default for browsers
}
if ( isNode ( ) ) {
try {
// Try to get memory info synchronously for Node.js
// This will be available in Node.js environments
if ( typeof process !== 'undefined' && process . memoryUsage ) {
// Use RSS (Resident Set Size) as a proxy for available memory
const rss = process . memoryUsage ( ) . rss
// Assume we can use up to 4GB or 50% more than current usage
return Math . min ( 4096 , Math . floor ( rss / ( 1024 * 1024 ) * 1.5 ) )
}
} catch {
// Fall through to default
}
return 1024 // Default 1GB for Node.js
}
return 512 // Conservative default
}
/ * *
* Convenience function to check if models need to be downloaded
* This replaces the need for BRAINY_ALLOW_REMOTE_MODELS
* /
export function shouldAutoDownloadModels ( ) : boolean {
// Always allow downloads unless explicitly disabled
// This eliminates the need for BRAINY_ALLOW_REMOTE_MODELS
const explicitlyDisabled = process . env . BRAINY_ALLOW_REMOTE_MODELS === 'false'
if ( explicitlyDisabled ) {
console . warn ( 'Model downloads disabled via BRAINY_ALLOW_REMOTE_MODELS=false' )
return false
}
// In production, always allow downloads for seamless operation
if ( process . env . NODE_ENV === 'production' ) {
return true
}
// In development, allow downloads with a one-time notice
if ( process . env . NODE_ENV === 'development' ) {
return true
}
// Default: allow downloads
return true
}
/ * *
* Get the model path with intelligent defaults
* This replaces the need for BRAINY_MODELS_PATH env var
* /
export function getModelPath ( ) : string {
// Check if user explicitly set a path (keeping this for advanced users)
if ( process . env . BRAINY_MODELS_PATH ) {
return process . env . BRAINY_MODELS_PATH
}
// Browser - use cache API or IndexedDB (handled by transformers.js)
if ( isBrowser ( ) ) {
return 'browser-cache'
}
// Serverless - use /tmp for ephemeral storage
if ( isServerlessEnvironment ( ) ) {
return '/tmp/.brainy/models'
}
// Node.js - use home directory for persistent storage
if ( isNode ( ) ) {
// Use process.env.HOME as a fallback
const homeDir = process . env . HOME || process . env . USERPROFILE || '~'
return ` ${ homeDir } /.brainy/models `
}
// Fallback
return './.brainy/models'
}
/ * *
* Log model configuration decision ( only in verbose mode )
* /
export function logModelConfig ( config : ModelConfigResult , verbose : boolean = false ) : void {
if ( ! verbose && process . env . NODE_ENV === 'production' ) {
return // Silent in production unless verbose
}
const icon = config . autoSelected ? '🤖' : '👤'
console . log ( ` ${ icon } Model: ${ config . precision . toUpperCase ( ) } - ${ config . reason } ` )
}