/** * LLM Provider Factory * * Returns the right adapter (embedding + summarisation) based on * the organisation's RAG settings (provider, baseUrl, model, etc.). * * Supported providers: * openai — OpenAI API (api.openai.com) * openai_compatible — Any OpenAI-compatible endpoint with a custom baseUrl * ollama — Local Ollama server (no API key required) */ import type { OrgEmbeddingConfig } from "./embedding.service"; import { getOllamaNumThread } from "../utils/ollama-config"; const DEFAULT_OPENAI_EMBEDDING_MODEL = "text-embedding-3-small"; const DEFAULT_OPENAI_CHAT_MODEL = "gpt-4o-mini"; const DEFAULT_OLLAMA_CHAT_MODEL = "llama3"; const DEFAULT_OLLAMA_BASE_URL = process.env.OLLAMA_BASE_URL ?? "http://localhost:11434"; // Default max chunk characters per provider export const DEFAULT_MAX_CHUNK_CHARS: Record = { openai: 8000, openai_compatible: 8000, ollama: 2048, }; /** Resolve the API base URL for a given config */ export function resolveApiBase(config: OrgEmbeddingConfig): string { if (config.provider === "ollama") { return (config.baseUrl ?? DEFAULT_OLLAMA_BASE_URL).replace(/\/$/, ""); } if (config.provider === "openai_compatible" && config.baseUrl) { return config.baseUrl.replace(/\/$/, ""); } return "https://api.openai.com"; } function apiUrl(apiBase: string, path: string): string { const hasVersion = /\/v\d+$/.test(apiBase); return hasVersion ? `${apiBase}${path}` : `${apiBase}/v1${path}`; } /** Max chunk size for this config (falls back to provider default) */ export function resolveMaxChunkChars(config: LlmProviderConfig): number { if (config.maxChunkChars && config.maxChunkChars > 0) { return config.maxChunkChars; } return DEFAULT_MAX_CHUNK_CHARS[config.provider] ?? 8000; } export interface LlmProviderConfig extends OrgEmbeddingConfig { chatModel: string | null; maxChunkChars: number | null; summarizationEnabled: boolean; botId?: number; } export interface EmbeddingResult { embedding: number[] | null; } export interface SummaryResult { text: string | null; } /** * Generate an embedding vector for the given text using the configured provider. */ export async function generateEmbedding( text: string, config: LlmProviderConfig ): Promise { const apiBase = resolveApiBase(config); const maxChars = resolveMaxChunkChars(config); const truncated = text.slice(0, maxChars); if (config.provider === "ollama") { try { const response = await fetch(`${apiBase}/api/embeddings`, { method: "POST", headers: { "Content-Type": "application/json", ...config.customHeaders }, body: JSON.stringify({ model: config.embeddingModel, prompt: truncated }), }); if (!response.ok) { console.error("[LLM] Ollama embedding error:", response.status, await response.text()); return null; } const data = await response.json() as { embedding: number[] }; return data.embedding ?? null; } catch (err) { console.error("[LLM] Ollama embedding request failed:", err); return null; } } if (!config.apiKey) return null; try { const response = await fetch(apiUrl(apiBase, '/embeddings'), { method: "POST", headers: { "Content-Type": "application/json", Authorization: `Bearer ${config.apiKey}`, ...config.customHeaders, }, body: JSON.stringify({ input: truncated, model: config.embeddingModel }), signal: AbortSignal.timeout(15000), }); if (!response.ok) { console.error("[LLM] OpenAI embedding error:", response.status, await response.text()); return null; } const data = await response.json() as { data: Array<{ embedding: number[] }> }; return data.data[0]?.embedding ?? null; } catch (err) { console.error("[LLM] OpenAI embedding request failed:", err); return null; } } /** * Generate a batch of embeddings. Ollama doesn't support batch, so we serialise. */ export async function generateEmbeddingBatch( texts: string[], config: LlmProviderConfig ): Promise<(number[] | null)[]> { if (config.provider === "ollama") { return Promise.all(texts.map((t) => generateEmbedding(t, config))); } if (!config.apiKey) return texts.map(() => null); const apiBase = resolveApiBase(config); const maxChars = resolveMaxChunkChars(config); const truncated = texts.map((t) => t.slice(0, maxChars)); try { const response = await fetch(apiUrl(apiBase, '/embeddings'), { method: "POST", headers: { "Content-Type": "application/json", Authorization: `Bearer ${config.apiKey}`, ...config.customHeaders, }, body: JSON.stringify({ input: truncated, model: config.embeddingModel }), signal: AbortSignal.timeout(30000), }); if (!response.ok) { console.error("[LLM] OpenAI batch embedding error:", response.status, await response.text()); return texts.map(() => null); } const data = await response.json() as { data: Array<{ embedding: number[]; index: number }> }; const result: (number[] | null)[] = texts.map(() => null); for (const item of data.data) { result[item.index] = item.embedding; } return result; } catch (err) { console.error("[LLM] OpenAI batch embedding request failed:", err); return texts.map(() => null); } } /** * Generate a text summary using the configured chat model. * Returns null when the provider is not configured or the request fails. */ export async function generateSummary( prompt: string, config: LlmProviderConfig ): Promise { const apiBase = resolveApiBase(config); if (config.provider === "ollama") { const model = config.chatModel || DEFAULT_OLLAMA_CHAT_MODEL; try { const response = await fetch(`${apiBase}/api/generate`, { method: "POST", headers: { "Content-Type": "application/json", ...config.customHeaders }, body: JSON.stringify({ model, prompt, stream: false, options: { num_thread: getOllamaNumThread() } }), signal: AbortSignal.timeout(120000), }); if (!response.ok) { console.error("[LLM] Ollama generate error:", response.status, await response.text()); return null; } const data = await response.json() as { response?: string }; return data.response?.trim() ?? null; } catch (err) { console.error("[LLM] Ollama generate request failed:", err); return null; } } if (!config.apiKey) return null; const chatModel = config.chatModel || DEFAULT_OPENAI_CHAT_MODEL; try { const response = await fetch(apiUrl(apiBase, '/chat/completions'), { method: "POST", headers: { "Content-Type": "application/json", Authorization: `Bearer ${config.apiKey}`, ...config.customHeaders, }, body: JSON.stringify({ model: chatModel, messages: [{ role: "user", content: prompt }], max_tokens: 2000, temperature: 0.3, }), signal: AbortSignal.timeout(120000), }); if (!response.ok) { console.error("[LLM] OpenAI chat error:", response.status, await response.text()); return null; } const data = await response.json() as { choices: Array<{ message: { content: string; reasoning_content?: string } }> }; const choice = data.choices[0]; return (choice?.message?.content?.trim() || choice?.message?.reasoning_content?.trim()) ?? null; } catch (err) { console.error("[LLM] OpenAI chat request failed:", err); return null; } } /** * Lookup table of well-known embedding models and their vector dimensions. * Used at startup to detect dimension mismatches without making network calls. */ export const KNOWN_EMBEDDING_DIMS: Record = { // OpenAI "text-embedding-3-small": 1536, "text-embedding-3-large": 3072, "text-embedding-ada-002": 1536, // BAAI / Ollama "bge-m3": 1024, "bge-large-en-v1.5": 1024, "bge-base-en-v1.5": 768, "bge-small-en-v1.5": 384, // Nomic "nomic-embed-text": 768, // MixedBread "mxbai-embed-large": 1024, // SentenceTransformers / all-MiniLM "all-minilm": 384, "all-minilm-l6-v2": 384, "all-minilm-l12-v2": 384, // Snowflake "snowflake-arctic-embed": 1024, "snowflake-arctic-embed2": 1024, // E5 "multilingual-e5-large": 1024, "e5-mistral-7b-instruct": 4096, }; /** * Return the known vector dimension for the given model name. * Strips the ":tag" suffix (e.g. "bge-m3:latest" → "bge-m3") before lookup. * Returns null when the model is not in the lookup table. */ export function getKnownEmbeddingDim(model: string): number | null { if (!model) return null; const direct = KNOWN_EMBEDDING_DIMS[model]; if (direct) return direct; const base = model.split(":")[0]; return KNOWN_EMBEDDING_DIMS[base] ?? null; } /** * Detect the actual embedding dimension by generating a test embedding. * Falls back to null when the provider is unreachable or returns no data. */ export async function detectEmbeddingDimension( config: LlmProviderConfig ): Promise { const knownDim = config.embeddingModel ? getKnownEmbeddingDim(config.embeddingModel) : null; if (knownDim) return knownDim; const vec = await generateEmbedding("test", config); return vec ? vec.length : null; } export async function testOllamaConnection( baseUrl: string, embeddingModel: string ): Promise<{ ok: true } | { ok: false; error: string }> { const base = baseUrl.replace(/\/$/, ""); try { const res = await fetch(`${base}/api/tags`, { signal: AbortSignal.timeout(8000), }); if (!res.ok) { return { ok: false, error: `Ollama недоступен (${res.status})` }; } const data = await res.json() as { models?: Array<{ name: string }> }; const models = data.models ?? []; // Check model availability — Ollama names can be "nomic-embed-text:latest" etc. const modelAvailable = models.some( (m) => m.name === embeddingModel || m.name.startsWith(`${embeddingModel}:`) ); if (!modelAvailable && models.length > 0) { const available = models.map((m) => m.name).join(", "); return { ok: false, error: `Модель «${embeddingModel}» не найдена. Доступные: ${available}. Выполните: ollama pull ${embeddingModel}`, }; } return { ok: true }; } catch (err: any) { const msg = err?.message ?? String(err); if (msg.includes("fetch") || msg.includes("ECONNREFUSED") || msg.includes("timeout")) { return { ok: false, error: `Не удаётся подключиться к Ollama по адресу ${base}. Убедитесь, что сервер запущен.` }; } return { ok: false, error: `Ошибка соединения с Ollama: ${msg.slice(0, 200)}` }; } }