- number-поля теперь рендерятся как text + inputMode=numeric, чтобы браузер не округлял значения через input type=number - пробелы при вставке в number-поля удаляются - бэкенд нормализует значения number-полей в строку перед сохранением - добавлен хелпер normalizeFieldValueForStorage Closes: искажение расчётного счёта и других длинных числовых полей
324 lines
11 KiB
TypeScript
324 lines
11 KiB
TypeScript
/**
|
||
* LLM Provider Factory
|
||
*
|
||
* Returns the right adapter (embedding + summarisation) based on
|
||
* the organisation's RAG settings (provider, baseUrl, model, etc.).
|
||
*
|
||
* Supported providers:
|
||
* openai — OpenAI API (api.openai.com)
|
||
* openai_compatible — Any OpenAI-compatible endpoint with a custom baseUrl
|
||
* ollama — Local Ollama server (no API key required)
|
||
*/
|
||
|
||
import type { OrgEmbeddingConfig } from "./embedding.service";
|
||
import { getOllamaNumThread } from "../utils/ollama-config";
|
||
|
||
const DEFAULT_OPENAI_EMBEDDING_MODEL = "text-embedding-3-small";
|
||
const DEFAULT_OPENAI_CHAT_MODEL = "gpt-4o-mini";
|
||
const DEFAULT_OLLAMA_CHAT_MODEL = "llama3";
|
||
const DEFAULT_OLLAMA_BASE_URL = process.env.OLLAMA_BASE_URL ?? "http://localhost:11434";
|
||
|
||
// Default max chunk characters per provider
|
||
export const DEFAULT_MAX_CHUNK_CHARS: Record<string, number> = {
|
||
openai: 8000,
|
||
openai_compatible: 8000,
|
||
ollama: 2048,
|
||
};
|
||
|
||
/** Resolve the API base URL for a given config */
|
||
export function resolveApiBase(config: OrgEmbeddingConfig): string {
|
||
if (config.provider === "ollama") {
|
||
return (config.baseUrl ?? DEFAULT_OLLAMA_BASE_URL).replace(/\/$/, "");
|
||
}
|
||
if (config.provider === "openai_compatible" && config.baseUrl) {
|
||
return config.baseUrl.replace(/\/$/, "");
|
||
}
|
||
return "https://api.openai.com";
|
||
}
|
||
|
||
function apiUrl(apiBase: string, path: string): string {
|
||
const hasVersion = /\/v\d+$/.test(apiBase);
|
||
return hasVersion ? `${apiBase}${path}` : `${apiBase}/v1${path}`;
|
||
}
|
||
|
||
/** Max chunk size for this config (falls back to provider default) */
|
||
export function resolveMaxChunkChars(config: LlmProviderConfig): number {
|
||
if (config.maxChunkChars && config.maxChunkChars > 0) {
|
||
return config.maxChunkChars;
|
||
}
|
||
return DEFAULT_MAX_CHUNK_CHARS[config.provider] ?? 8000;
|
||
}
|
||
|
||
export interface LlmProviderConfig extends OrgEmbeddingConfig {
|
||
chatModel: string | null;
|
||
maxChunkChars: number | null;
|
||
summarizationEnabled: boolean;
|
||
botId?: number;
|
||
}
|
||
|
||
export interface EmbeddingResult {
|
||
embedding: number[] | null;
|
||
}
|
||
|
||
export interface SummaryResult {
|
||
text: string | null;
|
||
}
|
||
|
||
/**
|
||
* Generate an embedding vector for the given text using the configured provider.
|
||
*/
|
||
export async function generateEmbedding(
|
||
text: string,
|
||
config: LlmProviderConfig
|
||
): Promise<number[] | null> {
|
||
const apiBase = resolveApiBase(config);
|
||
const maxChars = resolveMaxChunkChars(config);
|
||
const truncated = text.slice(0, maxChars);
|
||
|
||
if (config.provider === "ollama") {
|
||
try {
|
||
const response = await fetch(`${apiBase}/api/embeddings`, {
|
||
method: "POST",
|
||
headers: { "Content-Type": "application/json", ...config.customHeaders },
|
||
body: JSON.stringify({ model: config.embeddingModel, prompt: truncated }),
|
||
});
|
||
if (!response.ok) {
|
||
console.error("[LLM] Ollama embedding error:", response.status, await response.text());
|
||
return null;
|
||
}
|
||
const data = await response.json() as { embedding: number[] };
|
||
return data.embedding ?? null;
|
||
} catch (err) {
|
||
console.error("[LLM] Ollama embedding request failed:", err);
|
||
return null;
|
||
}
|
||
}
|
||
|
||
if (!config.apiKey) return null;
|
||
|
||
try {
|
||
const response = await fetch(apiUrl(apiBase, '/embeddings'), {
|
||
method: "POST",
|
||
headers: {
|
||
"Content-Type": "application/json",
|
||
Authorization: `Bearer ${config.apiKey}`,
|
||
...config.customHeaders,
|
||
},
|
||
body: JSON.stringify({ input: truncated, model: config.embeddingModel }),
|
||
signal: AbortSignal.timeout(15000),
|
||
});
|
||
if (!response.ok) {
|
||
console.error("[LLM] OpenAI embedding error:", response.status, await response.text());
|
||
return null;
|
||
}
|
||
const data = await response.json() as { data: Array<{ embedding: number[] }> };
|
||
return data.data[0]?.embedding ?? null;
|
||
} catch (err) {
|
||
console.error("[LLM] OpenAI embedding request failed:", err);
|
||
return null;
|
||
}
|
||
}
|
||
|
||
/**
|
||
* Generate a batch of embeddings. Ollama doesn't support batch, so we serialise.
|
||
*/
|
||
export async function generateEmbeddingBatch(
|
||
texts: string[],
|
||
config: LlmProviderConfig
|
||
): Promise<(number[] | null)[]> {
|
||
if (config.provider === "ollama") {
|
||
return Promise.all(texts.map((t) => generateEmbedding(t, config)));
|
||
}
|
||
|
||
if (!config.apiKey) return texts.map(() => null);
|
||
|
||
const apiBase = resolveApiBase(config);
|
||
const maxChars = resolveMaxChunkChars(config);
|
||
const truncated = texts.map((t) => t.slice(0, maxChars));
|
||
|
||
try {
|
||
const response = await fetch(apiUrl(apiBase, '/embeddings'), {
|
||
method: "POST",
|
||
headers: {
|
||
"Content-Type": "application/json",
|
||
Authorization: `Bearer ${config.apiKey}`,
|
||
...config.customHeaders,
|
||
},
|
||
body: JSON.stringify({ input: truncated, model: config.embeddingModel }),
|
||
signal: AbortSignal.timeout(30000),
|
||
});
|
||
if (!response.ok) {
|
||
console.error("[LLM] OpenAI batch embedding error:", response.status, await response.text());
|
||
return texts.map(() => null);
|
||
}
|
||
const data = await response.json() as { data: Array<{ embedding: number[]; index: number }> };
|
||
const result: (number[] | null)[] = texts.map(() => null);
|
||
for (const item of data.data) {
|
||
result[item.index] = item.embedding;
|
||
}
|
||
return result;
|
||
} catch (err) {
|
||
console.error("[LLM] OpenAI batch embedding request failed:", err);
|
||
return texts.map(() => null);
|
||
}
|
||
}
|
||
|
||
/**
|
||
* Generate a text summary using the configured chat model.
|
||
* Returns null when the provider is not configured or the request fails.
|
||
*/
|
||
export async function generateSummary(
|
||
prompt: string,
|
||
config: LlmProviderConfig
|
||
): Promise<string | null> {
|
||
const apiBase = resolveApiBase(config);
|
||
|
||
if (config.provider === "ollama") {
|
||
const model = config.chatModel || DEFAULT_OLLAMA_CHAT_MODEL;
|
||
try {
|
||
const response = await fetch(`${apiBase}/api/generate`, {
|
||
method: "POST",
|
||
headers: { "Content-Type": "application/json", ...config.customHeaders },
|
||
body: JSON.stringify({ model, prompt, stream: false, options: { num_thread: getOllamaNumThread() } }),
|
||
signal: AbortSignal.timeout(120000),
|
||
});
|
||
if (!response.ok) {
|
||
console.error("[LLM] Ollama generate error:", response.status, await response.text());
|
||
return null;
|
||
}
|
||
const data = await response.json() as { response?: string };
|
||
return data.response?.trim() ?? null;
|
||
} catch (err) {
|
||
console.error("[LLM] Ollama generate request failed:", err);
|
||
return null;
|
||
}
|
||
}
|
||
|
||
if (!config.apiKey) return null;
|
||
|
||
const chatModel = config.chatModel || DEFAULT_OPENAI_CHAT_MODEL;
|
||
|
||
try {
|
||
const response = await fetch(apiUrl(apiBase, '/chat/completions'), {
|
||
method: "POST",
|
||
headers: {
|
||
"Content-Type": "application/json",
|
||
Authorization: `Bearer ${config.apiKey}`,
|
||
...config.customHeaders,
|
||
},
|
||
body: JSON.stringify({
|
||
model: chatModel,
|
||
messages: [{ role: "user", content: prompt }],
|
||
max_tokens: 2000,
|
||
temperature: 0.3,
|
||
}),
|
||
signal: AbortSignal.timeout(120000),
|
||
});
|
||
if (!response.ok) {
|
||
console.error("[LLM] OpenAI chat error:", response.status, await response.text());
|
||
return null;
|
||
}
|
||
const data = await response.json() as { choices: Array<{ message: { content: string; reasoning_content?: string } }> };
|
||
const choice = data.choices[0];
|
||
return (choice?.message?.content?.trim() || choice?.message?.reasoning_content?.trim()) ?? null;
|
||
} catch (err) {
|
||
console.error("[LLM] OpenAI chat request failed:", err);
|
||
return null;
|
||
}
|
||
}
|
||
|
||
/**
|
||
* Lookup table of well-known embedding models and their vector dimensions.
|
||
* Used at startup to detect dimension mismatches without making network calls.
|
||
*/
|
||
export const KNOWN_EMBEDDING_DIMS: Record<string, number> = {
|
||
// OpenAI
|
||
"text-embedding-3-small": 1536,
|
||
"text-embedding-3-large": 3072,
|
||
"text-embedding-ada-002": 1536,
|
||
// BAAI / Ollama
|
||
"bge-m3": 1024,
|
||
"bge-large-en-v1.5": 1024,
|
||
"bge-base-en-v1.5": 768,
|
||
"bge-small-en-v1.5": 384,
|
||
// Nomic
|
||
"nomic-embed-text": 768,
|
||
// MixedBread
|
||
"mxbai-embed-large": 1024,
|
||
// SentenceTransformers / all-MiniLM
|
||
"all-minilm": 384,
|
||
"all-minilm-l6-v2": 384,
|
||
"all-minilm-l12-v2": 384,
|
||
// Snowflake
|
||
"snowflake-arctic-embed": 1024,
|
||
"snowflake-arctic-embed2": 1024,
|
||
// E5
|
||
"multilingual-e5-large": 1024,
|
||
"e5-mistral-7b-instruct": 4096,
|
||
};
|
||
|
||
/**
|
||
* Return the known vector dimension for the given model name.
|
||
* Strips the ":tag" suffix (e.g. "bge-m3:latest" → "bge-m3") before lookup.
|
||
* Returns null when the model is not in the lookup table.
|
||
*/
|
||
export function getKnownEmbeddingDim(model: string): number | null {
|
||
if (!model) return null;
|
||
const direct = KNOWN_EMBEDDING_DIMS[model];
|
||
if (direct) return direct;
|
||
const base = model.split(":")[0];
|
||
return KNOWN_EMBEDDING_DIMS[base] ?? null;
|
||
}
|
||
|
||
/**
|
||
* Detect the actual embedding dimension by generating a test embedding.
|
||
* Falls back to null when the provider is unreachable or returns no data.
|
||
*/
|
||
export async function detectEmbeddingDimension(
|
||
config: LlmProviderConfig
|
||
): Promise<number | null> {
|
||
const knownDim = config.embeddingModel ? getKnownEmbeddingDim(config.embeddingModel) : null;
|
||
if (knownDim) return knownDim;
|
||
const vec = await generateEmbedding("test", config);
|
||
return vec ? vec.length : null;
|
||
}
|
||
|
||
export async function testOllamaConnection(
|
||
baseUrl: string,
|
||
embeddingModel: string
|
||
): Promise<{ ok: true } | { ok: false; error: string }> {
|
||
const base = baseUrl.replace(/\/$/, "");
|
||
|
||
try {
|
||
const res = await fetch(`${base}/api/tags`, {
|
||
signal: AbortSignal.timeout(8000),
|
||
});
|
||
if (!res.ok) {
|
||
return { ok: false, error: `Ollama недоступен (${res.status})` };
|
||
}
|
||
const data = await res.json() as { models?: Array<{ name: string }> };
|
||
const models = data.models ?? [];
|
||
|
||
// Check model availability — Ollama names can be "nomic-embed-text:latest" etc.
|
||
const modelAvailable = models.some(
|
||
(m) => m.name === embeddingModel || m.name.startsWith(`${embeddingModel}:`)
|
||
);
|
||
|
||
if (!modelAvailable && models.length > 0) {
|
||
const available = models.map((m) => m.name).join(", ");
|
||
return {
|
||
ok: false,
|
||
error: `Модель «${embeddingModel}» не найдена. Доступные: ${available}. Выполните: ollama pull ${embeddingModel}`,
|
||
};
|
||
}
|
||
|
||
return { ok: true };
|
||
} catch (err: any) {
|
||
const msg = err?.message ?? String(err);
|
||
if (msg.includes("fetch") || msg.includes("ECONNREFUSED") || msg.includes("timeout")) {
|
||
return { ok: false, error: `Не удаётся подключиться к Ollama по адресу ${base}. Убедитесь, что сервер запущен.` };
|
||
}
|
||
return { ok: false, error: `Ошибка соединения с Ollama: ${msg.slice(0, 200)}` };
|
||
}
|
||
}
|