/** * Ollama runtime configuration helpers. * * Reads env vars and CPU count to produce sensible defaults * for num_thread / num_ctx so that Ollama uses the server efficiently. */ import { cpus } from "os"; /** CPU cores visible to Node (host cores when no Docker limit is set). */ function detectCpuCount(): number { try { return cpus().length; } catch { return 1; } } /** * Number of threads Ollama should use for inference. * Priority: * 1. OLLAMA_NUM_THREAD env var * 2. cpus().length (leave 1 core for OS / other containers when > 1) */ export function getOllamaNumThread(): number { if (process.env.OLLAMA_NUM_THREAD) { const parsed = parseInt(process.env.OLLAMA_NUM_THREAD, 10); if (!Number.isNaN(parsed) && parsed > 0) { return parsed; } } const count = detectCpuCount(); // Leave one core for the OS / DB / app when we have > 1 core return Math.max(1, count > 1 ? count - 1 : count); } /** * Context window size for Ollama models. * Priority: * 1. OLLAMA_NUM_CTX env var * 2. Default 2048 (safe for CPU-only setups) */ export function getOllamaNumCtx(): number { if (process.env.OLLAMA_NUM_CTX) { const parsed = parseInt(process.env.OLLAMA_NUM_CTX, 10); if (!Number.isNaN(parsed) && parsed >= 512) { return parsed; } } return 2048; }