freegate 0.6.15 → 0.6.17

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/server.js CHANGED
@@ -3,18 +3,26 @@ const http = require('http');
3
3
  const fs = require('fs');
4
4
  const path = require('path');
5
5
  const { LRUCache } = require('./lib/cache');
6
- const { PROVIDERS, MODEL_MAP, callProvider } = require('./lib/providers');
7
- const { loadState, initHealth, isCircuitOpen, recordSuccess, recordFailure, recordRequest, recordTokens, getHealth, getStats, recordRecent, recordRpm, getRecent, getRpm, recordSelection, getLastSelection, getBandit, recordBandit } = require('./lib/health');
6
+ const { PROVIDERS, MODEL_MAP, callProvider, reloadProviders } = require('./lib/providers');
7
+ const { loadState, initHealth, isCircuitOpen, recordSuccess, recordFailure, recordRequest, recordTokens, getHealth, getStats, getReliability, recordRecent, recordRpm, getRecent, getRpm, recordSelection, getLastSelection, getBandit, recordBandit, warmBanditPriors, getContextStats } = require('./lib/health');
8
8
  const { checkRateLimit } = require('./lib/rateLimit');
9
9
  const { handleDashboard } = require('./lib/dashboard');
10
10
  const { acquire, stats: poolStats } = require('./lib/pool');
11
+ const { aggregateSavings } = require('./lib/economics');
11
12
  const { stripThink, cleanDelta, cleanMessage, fixReasoningMessage, isTooShort, MIN_ANSWER_LEN } = require('./lib/clean');
12
- const { classifyComplexity, maybeUpgradeTier, classifyVisionComplexity } = require('./lib/routing');
13
+ const { classifyComplexity, maybeUpgradeTier, classifyVisionComplexity, needsWindowUpgrade } = require('./lib/routing');
13
14
  const { bucket, pick: banditPick, isTransientLimit } = require('./lib/bandit');
15
+ const { StringDecoder } = require('string_decoder');
14
16
  const logger = require('./lib/logger');
17
+ const { KEY_GROUPS, readKeys, saveKeys, validateKey, getStoredKey } = require('./lib/setup');
18
+ const { prepareMessages, estimateTokens, setMemory: compactorSetMemory } = require('./lib/compactor');
19
+ const { classify: classifyTask } = require('./lib/taskclassify');
20
+ const { injectMethodology, enabledByDefault: methodEnabledByDefault } = require('./lib/methodology');
21
+ const { create: createMemoryStore } = require('./lib/memory-store');
15
22
 
16
23
  // Load persisted state
17
24
  loadState();
25
+ const contextStats = getContextStats();
18
26
 
19
27
  // Drop stale health entries for providers that no longer exist (e.g. auto-disabled)
20
28
  const activeKeys = new Set(Object.keys(PROVIDERS));
@@ -26,6 +34,14 @@ if (stale.length > 0) {
26
34
  logger.info('Cleaned stale health entries', { removed: stale });
27
35
  }
28
36
 
37
+ // Warm bandit priors for enabled providers with no/low history so brand-new
38
+ // models (e.g. or-minimax-m3-free) are explored promptly instead of ignored.
39
+ const warmed = warmBanditPriors(
40
+ Object.keys(PROVIDERS).filter(k => PROVIDERS[k].enabled !== false),
41
+ ['low', 'med', 'high']
42
+ );
43
+ if (warmed > 0) logger.info('Warmed bandit priors', { warmed });
44
+
29
45
  const cache = new LRUCache(500, 3600000, false, true); // 4th arg: semantic normalize ON
30
46
  require('./lib/cache')._activeCache = cache;
31
47
 
@@ -33,7 +49,7 @@ require('./lib/cache')._activeCache = cache;
33
49
  // Prefer cwd config.json (user's project) over the package dir.
34
50
  const CONFIG_CANDIDATES = [path.join(process.cwd(), 'config.json'), path.join(__dirname, 'config.json')];
35
51
  const CONFIG_PATH = CONFIG_CANDIDATES.find(p => fs.existsSync(p)) || CONFIG_CANDIDATES[1];
36
- let config = { port: 4000, auth: '', rateLimit: { maxRequests: 100, windowMs: 60000 } };
52
+ let config = { port: 4000, auth: '', rateLimit: { maxRequests: 100, windowMs: 60000 }, methodology: true };
37
53
  try {
38
54
  const parsed = JSON.parse(fs.readFileSync(CONFIG_PATH, 'utf8'));
39
55
  if (parsed && typeof parsed === 'object') config = { ...config, ...parsed };
@@ -51,6 +67,75 @@ const PORT = parseInt(process.env.PORT || getArg('port', config.port || '4000'))
51
67
  const AUTH_KEY = process.env.AUTH || getArg('auth', config.auth || '');
52
68
  const RATE_LIMIT = config.rateLimit || { maxRequests: 100, windowMs: 60000 };
53
69
 
70
+ // --- Долговременная память (vector memory) ---
71
+ // По умолчанию включена: факты берутся побочно от компакции (без лишних вызовов
72
+ // LLM) и подмешиваются в контекст по релевантности. config.memory.enabled=false —
73
+ // слой отключается целиком.
74
+ const MEMORY_CONFIG = Object.assign(
75
+ { enabled: true, topK: 3, minSimilarity: 0.05, coverageForSkip: 0.6 },
76
+ (config.memory && typeof config.memory === 'object') ? config.memory : {}
77
+ );
78
+ const memStore = MEMORY_CONFIG.enabled ? createMemoryStore({ filePath: path.join(__dirname, 'memory.json') }) : null;
79
+ if (memStore) compactorSetMemory(memStore);
80
+
81
+ // --- Семантический кэш ---
82
+ // На промахе точного ключа ищет закэшированный диалог с похожим нормализованным
83
+ // текстом (Dice по символьным триграммам). config.semcache.enabled=false отключает.
84
+ const SEMCACHE_CONFIG = Object.assign(
85
+ { enabled: true, minSimilarity: 0.85 },
86
+ (config.semcache && typeof config.semcache === 'object') ? config.semcache : {}
87
+ );
88
+
89
+ // --- Методолог (инженерная дисциплина в промпте) ---
90
+ // config.methodology.enabled=false отключает. По умолчанию включён.
91
+ // config.methodology.prompts.{category} переопределяет текст промпта для
92
+ // конкретной категории (незаданные категории остаются в дефолтах).
93
+ const METHODOLOGY_CONFIG = Object.assign(
94
+ { enabled: methodEnabledByDefault(), prompts: {} },
95
+ (config.methodology && typeof config.methodology === 'object') ? config.methodology : {}
96
+ );
97
+
98
+ // --- Самообновляющаяся база моделей (Model Discovery Engine) ---
99
+ // Планировщик живёт внутри сервера: работает «всегда» у всех пользователей
100
+ // пакета без cron/launchd. config.modelManager.enabled=false отключает.
101
+ function _loadEnvFile() {
102
+ try {
103
+ const envPath = path.join(__dirname, '.env');
104
+ if (!fs.existsSync(envPath)) return;
105
+ for (const line of fs.readFileSync(envPath, 'utf8').split('\n')) {
106
+ const t = line.trim();
107
+ if (!t || t.startsWith('#')) continue;
108
+ const eq = t.indexOf('=');
109
+ if (eq > 0 && !process.env[t.slice(0, eq).trim()]) process.env[t.slice(0, eq).trim()] = t.slice(eq + 1).trim();
110
+ }
111
+ } catch {}
112
+ }
113
+ _loadEnvFile();
114
+ const { ModelManager } = require('./lib/modelmanager');
115
+ const MODEL_MANAGER_CONFIG = Object.assign(
116
+ {},
117
+ (config.modelManager && typeof config.modelManager === 'object') ? config.modelManager : {}
118
+ );
119
+ const modelManager = new ModelManager({
120
+ dbPath: path.join(__dirname, 'models-db.json'),
121
+ catalogPath: path.join(__dirname, 'providers.json'),
122
+ configPath: path.join(__dirname, 'config.json'),
123
+ keys: {
124
+ openrouter: process.env.PROVIDER_OPENROUTER_APIKEY || '',
125
+ huggingface: process.env.HF_TOKEN || process.env.PROVIDER_HF_APIKEY || '',
126
+ groq: process.env.PROVIDER_GROQ_APIKEY || '',
127
+ mistral: process.env.PROVIDER_MISTRAL_APIKEY || '',
128
+ gemini: process.env.PROVIDER_GEMINI_APIKEY || '',
129
+ cerebras: process.env.PROVIDER_CEREBRAS_APIKEY || '',
130
+ deepseek: process.env.PROVIDER_DEEPSEEK_APIKEY || '',
131
+ nim: process.env.PROVIDER_NIM_APIKEY || '',
132
+ },
133
+ fetchImpl: (url, opts) => fetch(url, opts),
134
+ config: MODEL_MANAGER_CONFIG,
135
+ reload: () => { try { reloadProviders(); } catch {} },
136
+ log: (msg) => logger.info('[modelManager] ' + msg),
137
+ });
138
+
54
139
  // Health check
55
140
  const healthIntervals = {}; // key -> { nextCheck, backoff }
56
141
 
@@ -123,6 +208,26 @@ async function healthCheck() {
123
208
  setInterval(healthCheck, 30000);
124
209
  setTimeout(healthCheck, 1000);
125
210
 
211
+ // Периодический сейв долговременной памяти (вдобавок к shutdown).
212
+ if (memStore) setInterval(() => memStore.save(), 60000);
213
+
214
+ // Извлекает usage из SSE-чанка (если провайдер шлёт его в последнем чанке).
215
+ function collectReasonUsage(str, usageObj) {
216
+ if (!str || !/data: /.test(str)) return;
217
+ for (const line of str.split('\n')) {
218
+ const m = line.match(/^data: (.+)$/);
219
+ if (!m || m[1].trim() === '[DONE]') continue;
220
+ try {
221
+ const obj = JSON.parse(m[1]);
222
+ if (obj.usage && (obj.usage.prompt_tokens || obj.usage.completion_tokens)) {
223
+ usageObj.prompt_tokens = obj.usage.prompt_tokens;
224
+ usageObj.completion_tokens = obj.usage.completion_tokens;
225
+ usageObj.total_tokens = obj.usage.total_tokens;
226
+ }
227
+ } catch {}
228
+ }
229
+ }
230
+
126
231
  // Извлекает все значения content из SSE-чанка. Возвращает true, если есть
127
232
  // хотя бы одно непустое (реальный токен, а не пустая дельта).
128
233
  function chunkHasToken(str) {
@@ -134,6 +239,22 @@ function chunkHasToken(str) {
134
239
  return false;
135
240
  }
136
241
 
242
+ // Последний user-текст — для оценки, насколько короткий ответ легитимен.
243
+ function lastUserText(messages) {
244
+ if (!Array.isArray(messages)) return '';
245
+ for (let i = messages.length - 1; i >= 0; i--) {
246
+ const m = messages[i];
247
+ if (m && m.role === 'user') {
248
+ if (typeof m.content === 'string') return m.content;
249
+ if (Array.isArray(m.content)) {
250
+ const t = m.content.filter(c => c && c.type === 'text').map(c => c.text || '').join(' ');
251
+ if (t) return t;
252
+ }
253
+ }
254
+ }
255
+ return '';
256
+ }
257
+
137
258
  // Chat completion handler
138
259
  async function handleChatCompletion(req, res, body) {
139
260
  const requestedModel = body.model || 'tier-splus';
@@ -144,6 +265,12 @@ async function handleChatCompletion(req, res, body) {
144
265
  let effectiveModel = maybeUpgradeTier(requestedModel, complexity);
145
266
  let targetProviderKey = MODEL_MAP[effectiveModel] || MODEL_MAP[requestedModel] || 'zai';
146
267
  const isStreaming = body.stream === true;
268
+ // --- Контекстная телеметрия: один measure на запрос, record только в терминальной точке. ---
269
+ const measure = { ts: Date.now(), provider: targetProviderKey, cacheType: 'miss', status: 0, win: PROVIDERS[targetProviderKey]?.context_window || 0, upgraded: 0 };
270
+ const commit = (status) => {
271
+ measure.status = status;
272
+ contextStats.record(measure);
273
+ };
147
274
 
148
275
  // Vision detection: if the request contains images, route to a vision provider.
149
276
  // TWO-STAGE pipeline:
@@ -182,29 +309,41 @@ async function handleChatCompletion(req, res, body) {
182
309
  if (visionChain.length > 0) {
183
310
  logger.info('Vision pipeline: распознаю скриншот', { chain: visionChain.map(p => p.key).join(',') });
184
311
  let extracted = '';
185
- for (const visionProvider of visionChain) {
186
- try {
187
- const visionBody = {
188
- model: visionProvider.model,
189
- messages: [{
190
- role: 'user',
191
- content: [
192
- { type: 'text', text: 'Распознай и извлеки ВЕСЬ текст с изображения (ошибка, код, сообщение). Верни только содержимое, без комментариев. Если это код — верни код как есть.' },
193
- ...(Array.isArray(body.messages) ? body.messages.flatMap((m) => (Array.isArray(m.content) ? m.content.filter((c) => c && (c.type === 'image_url' || c.type === 'image' || c.type === 'input_image')).map((c) => {
194
- // Normalize any image part to the universal image_url format
195
- const url = c.image_url?.url || c.image?.url || c.image?.data || (c.image && typeof c.image === 'string' ? c.image : null) || c.url;
196
- return url ? { type: 'image_url', image_url: { url } } : null;
197
- }).filter(Boolean) : [])) : []),
198
- ],
199
- }],
200
- max_tokens: 2000,
201
- };
202
- const visionRes = await callProvider(visionProvider, visionBody);
203
- extracted = visionRes.data?.choices?.[0]?.message?.content || visionRes.data?.choices?.[0]?.message?.reasoning || '';
204
- if (extracted) { logger.info('Vision pipeline: распознал ' + visionProvider.key); break; }
205
- } catch (err) {
206
- logger.warn('Vision pipeline: ' + visionProvider.key + ' не сработал', { error: err.message.slice(0, 80) });
312
+ // PARALLEL vision attempt: fire all vision providers at once and take the
313
+ // first one that extracts text. Previously each was tried IN SEQUENCE,
314
+ // so a slow/failed first provider meant the pipeline waited provider
315
+ // after provider — the "two chats think forever" pattern for screenshots.
316
+ const visionAttempts = visionChain.map((visionProvider) => (async () => {
317
+ const visionBody = {
318
+ model: visionProvider.model,
319
+ messages: [{
320
+ role: 'user',
321
+ content: [
322
+ { type: 'text', text: 'Распознай и извлеки ВЕСЬ текст с изображения (ошибка, код, сообщение). Верни только содержимое, без комментариев. Если это код — верни код как есть.' },
323
+ ...(Array.isArray(body.messages) ? body.messages.flatMap((m) => (Array.isArray(m.content) ? m.content.filter((c) => c && (c.type === 'image_url' || c.type === 'image' || c.type === 'input_image')).map((c) => {
324
+ // Normalize any image part to the universal image_url format
325
+ const url = c.image_url?.url || c.image?.url || c.image?.data || (c.image && typeof c.image === 'string' ? c.image : null) || c.url;
326
+ return url ? { type: 'image_url', image_url: { url } } : null;
327
+ }).filter(Boolean) : [])) : []),
328
+ ],
329
+ }],
330
+ max_tokens: 2000,
331
+ };
332
+ const visionRes = await callProvider(visionProvider, visionBody);
333
+ const text = visionRes.data?.choices?.[0]?.message?.content || visionRes.data?.choices?.[0]?.message?.reasoning || '';
334
+ if (text) {
335
+ logger.info('Vision pipeline: распознал ' + visionProvider.key);
336
+ return text;
207
337
  }
338
+ throw new Error(visionProvider.key + ': пустой OCR');
339
+ })());
340
+ // First provider to yield text wins; hard failures (429/5xx) are skipped
341
+ // without blocking the others. If ALL fail, extracted stays '' and the
342
+ // request proceeds without vision context (as before).
343
+ try {
344
+ extracted = await Promise.any(visionAttempts);
345
+ } catch {
346
+ logger.warn('Vision pipeline: все вижн-провайдеры не сработали', { tried: visionChain.map(p => p.key) });
208
347
  }
209
348
  const cleaned = stripThink(extracted, true);
210
349
  logger.info('Vision pipeline: скриншот распознан', { chars: cleaned.length });
@@ -233,13 +372,94 @@ async function handleChatCompletion(req, res, body) {
233
372
  }
234
373
  }
235
374
 
236
- // Check cache (works for both streaming and non-streaming)
237
- const cached = cache.get(effectiveModel, body.messages, body.temperature);
238
- if (cached) {
239
- logger.request({ model: requestedModel, provider: 'cache', status: 200, cached: true });
240
- recordRecent({ model: requestedModel, provider: 'cache', status: 200, latency: 0, cached: true });
375
+ // Window-aware upgrade: если запрос не влезает в окно целевой модели (после
376
+ // vision-апгрейда target), компакция НЕ запускается — суммаризатор не должен
377
+ // сжимать контекст, который провайдер с большим окном возьмёт целиком.
378
+ const windowUpgraded = Array.isArray(body.messages) && body.messages.length > 0 &&
379
+ needsWindowUpgrade(PROVIDERS[targetProviderKey]?.context_window || 0, estimateTokens(body.messages));
380
+
381
+ // Compact overly large conversations so free models don't reject on context.
382
+ // Runs AFTER the vision pipeline (images already converted to text above).
383
+ // Skipped in window-upgrade mode — the big provider takes the raw context.
384
+ if (Array.isArray(body.messages)) measure.origTokens = estimateTokens(body.messages);
385
+ if (Array.isArray(body.messages) && body.messages.length > 0 && !windowUpgraded) {
386
+ body.messages = await prepareMessages(body.messages, { contextWindow: PROVIDERS[targetProviderKey]?.context_window || 0 });
387
+ }
388
+ // Контекстная телеметрия: токены после компакции + доля системного промпта.
389
+ if (Array.isArray(body.messages)) {
390
+ measure.sentTokens = estimateTokens(body.messages);
391
+ measure.est = measure.sentTokens;
392
+ measure.compacted = measure.sentTokens < measure.origTokens;
393
+ const sysChars = body.messages.filter(m => m && m.role === 'system').reduce((a, m) => a + (typeof m.content === 'string' ? m.content.length : 0), 0);
394
+ const allChars = body.messages.reduce((a, m) => a + (typeof (m && m.content) === 'string' ? m.content.length : 0), 0);
395
+ if (allChars > 0) measure.sysShare = sysChars / allChars;
396
+ }
397
+
398
+ // --- Long-term memory recall ---
399
+ // Подмешиваем релевантные факты из прошлых сессий (векторная память) как
400
+ // user-сообщение В НАЧАЛЕ диалога — после системных правил, до кэша. Так
401
+ // ключ кэша (normalize игнорирует system, но учитывает user) различает разные
402
+ // наборы фактов, и ответы не отравляются чужим кэшем.
403
+ if (memStore) {
404
+ try {
405
+ const userTexts = [];
406
+ const userMsgs = body.messages.filter(m => m && m.role === 'user');
407
+ for (const m of userMsgs.slice(-3)) {
408
+ if (typeof m.content === 'string') userTexts.push(m.content);
409
+ else if (Array.isArray(m.content)) userTexts.push(m.content.filter(c => c && c.type === 'text').map(c => c.text || '').join(' '));
410
+ }
411
+ const query = userTexts.join('\n').trim();
412
+ if (query.length > 20) {
413
+ const hits = memStore.recall(query, { topK: MEMORY_CONFIG.topK, minSimilarity: MEMORY_CONFIG.minSimilarity });
414
+ if (hits.length > 0) {
415
+ // Не подмешиваем факт, который уже покрыт резюме компактора или
416
+ // недавними сообщениями этой же сессии (защита от дублей).
417
+ const existing = body.messages
418
+ .filter(m => m && typeof m.content === 'string')
419
+ .map(m => m.content)
420
+ .concat(userTexts);
421
+ const fresh = hits.filter(f => !memStore.isCovered(f.text, existing));
422
+ if (fresh.length > 0) {
423
+ let memoryMsg = fresh.map(f => f.text).join('\n');
424
+ if (memoryMsg) {
425
+ const insertAt = body.messages.findIndex(m => m && m.role !== 'system');
426
+ const block = { role: 'user', content: '[Память: релевантные факты из прошлого]\n' + memoryMsg };
427
+ if (insertAt === -1) body.messages.unshift(block);
428
+ else body.messages.splice(insertAt, 0, block);
429
+ measure.memory = true;
430
+ logger.info('Memory recall', { facts: fresh.length, covered: hits.length - fresh.length });
431
+ }
432
+ }
433
+ }
434
+ }
435
+ } catch (err) {
436
+ logger.error('Memory recall error', { message: err.message });
437
+ }
438
+ }
439
+
440
+ // --- Методолог (инженерная дисциплина) ---
441
+ // Классифицируем задачу (coding/reasoning/search/chat) и вставляем короткий
442
+ // системный промпт-методолог после памяти, до кэша. System-сообщение
443
+ // игнорируется normalize, поэтому кэш-ключ не меняется. Методолог влияет
444
+ // только на реальные запросы к провайдеру (кэш-хиты его не видят).
445
+ let taskCategory = 'chat';
446
+ if (METHODOLOGY_CONFIG.enabled && Array.isArray(body.messages)) {
447
+ try {
448
+ taskCategory = classifyTask(body.messages);
449
+ const injected = injectMethodology(body.messages, taskCategory, METHODOLOGY_CONFIG);
450
+ if (injected !== body.messages) {
451
+ body.messages = injected;
452
+ measure.taskCategory = taskCategory;
453
+ }
454
+ } catch (err) {
455
+ logger.error('Methodology error', { message: err.message });
456
+ }
457
+ }
458
+
459
+ // Replays a previously cached completion (exact or semantic hit), preserving
460
+ // the stream/non-stream shape the client asked for.
461
+ function serveCached(res, cached, isStreaming) {
241
462
  if (isStreaming) {
242
- // Replay cached answer as an SSE stream
243
463
  res.writeHead(200, { 'Content-Type': 'text/event-stream', 'Cache-Control': 'no-cache', 'Connection': 'keep-alive' });
244
464
  const content = cached.choices?.[0]?.message?.content || '';
245
465
  res.write(`data: ${JSON.stringify({ id: 'chatcmpl-cached', object: 'chat.completion.chunk', created: Math.floor(Date.now() / 1000), model: cached.model, choices: [{ index: 0, delta: { role: 'assistant', content: '' }, finish_reason: null }] })}\n\n`);
@@ -247,13 +467,40 @@ async function handleChatCompletion(req, res, body) {
247
467
  res.write(`data: ${JSON.stringify({ id: 'chatcmpl-cached', object: 'chat.completion.chunk', created: Math.floor(Date.now() / 1000), model: cached.model, choices: [{ index: 0, delta: {}, finish_reason: 'stop' }] })}\n\n`);
248
468
  res.write('data: [DONE]\n\n');
249
469
  res.end();
250
- return;
470
+ } else {
471
+ res.writeHead(200, { 'Content-Type': 'application/json' });
472
+ res.end(JSON.stringify(cached));
251
473
  }
252
- res.writeHead(200, { 'Content-Type': 'application/json' });
253
- res.end(JSON.stringify(cached));
474
+ }
475
+
476
+ // Check cache (works for both streaming and non-streaming)
477
+ const cached = cache.get(effectiveModel, body.messages, body.temperature);
478
+ if (cached) {
479
+ logger.request({ model: requestedModel, provider: 'cache', status: 200, cached: true });
480
+ recordRecent({ model: requestedModel, provider: 'cache', status: 200, latency: 0, cached: true });
481
+ measure.cacheType = 'exact';
482
+ commit(200);
483
+ serveCached(res, cached, isStreaming);
254
484
  return;
255
485
  }
256
486
 
487
+ // Semantic cache: same intent, rephrased wording → replay without a new LLM call.
488
+ if (SEMCACHE_CONFIG.enabled) {
489
+ const semantic = cache.getSemantic(effectiveModel, body.messages, body.temperature, SEMCACHE_CONFIG.minSimilarity);
490
+ if (semantic) {
491
+ logger.request({ model: requestedModel, provider: 'semcache', status: 200, cached: true });
492
+ recordRecent({ model: requestedModel, provider: 'semcache', status: 200, latency: 0, cached: true });
493
+ measure.cacheType = 'semcache';
494
+ commit(200);
495
+ serveCached(res, semantic.value, isStreaming);
496
+ return;
497
+ }
498
+ }
499
+
500
+ // Запрос реально идёт к провайдеру (кэш промахнулся) — только теперь помечаем
501
+ // апгрейд: кэш-хит апгрейдом не считается (апстрим-вызова не было).
502
+ if (windowUpgraded) measure.upgraded = 1;
503
+
257
504
  // Weighted selection among healthy providers
258
505
  const today = new Date().toISOString().slice(0, 10);
259
506
  // 'ratelimited' providers are alive but temporarily limited — include them
@@ -264,44 +511,102 @@ async function handleChatCompletion(req, res, body) {
264
511
  .filter(([_, p]) => p.enabled && !isCircuitOpen(p.key) && p.vision !== true &&
265
512
  (getHealth()[p.key]?.status === 'up' || getHealth()[p.key]?.status === 'ratelimited'));
266
513
 
514
+ // Window-aware routing: estimate the request size and only consider providers
515
+ // whose context window can actually hold it. This stops large requests from
516
+ // burning time falling through lfm (65k) / groq (131k) providers that reject
517
+ // them — they go straight to nemotron-35/dots-3/minimax (1M/512k windows).
518
+ // Only applies when the request is big enough to matter, so small/typical
519
+ // requests keep the full fast pool.
520
+ const requestTokens = estimateTokens(body.messages);
521
+ const MIN_WINDOW = 50000; // below this we don't filter (typical requests)
522
+ let windowPool = healthyProviders;
523
+ let upgradeNoCapable = false; // апгрейд, но ни один здоровый провайдер не держит запрос
524
+ if (requestTokens > MIN_WINDOW || windowUpgraded) {
525
+ const capable = healthyProviders.filter(([_, p]) => {
526
+ const win = p.context_window || 0;
527
+ // Unknown/0 window providers are kept (heuristic) — better to try than drop.
528
+ return win === 0 || win >= requestTokens;
529
+ });
530
+ if (capable.length > 0) {
531
+ windowPool = capable;
532
+ } else if (windowUpgraded) {
533
+ // Best-effort: запрос больше окна любого провайдера (minimax 1M не держит).
534
+ // Всё равно переполним — выберем самое большое окно ниже, минимизируя
535
+ // потери контекста; target исключён; OVERFLOW поймает телеметрия.
536
+ upgradeNoCapable = true;
537
+ }
538
+ }
539
+
267
540
  // Prefer providers below 90% of their daily limit; only fall back to
268
541
  // near-exhausted ones if that leaves nothing (avoids avoidable 429s).
269
- let pool = healthyProviders;
270
- const underLimit = healthyProviders.filter(([_, p]) => {
271
- const limit = p.dailyLimit || 1000;
272
- const used = (getStats().dailyUsage?.[p.key]?.[today]) || 0;
273
- return used < limit * 0.9;
274
- });
275
- if (underLimit.length > 0) pool = underLimit;
542
+ // Крутящийся пул по дневным лимитам: провайдер, исчерпавший дневной лимит,
543
+ // исключается из выбора и не возвращается до сброса. Это лечит 429-спираль
544
+ // (модель с лимитом 50/день сгорает к обеду, дальше каждая попытка на ней —
545
+ // бесполезный 429). Остаёмся на живых; выгоревшие — только как крайний резерв.
546
+ const usedTodayFor = (key) => (getStats().dailyUsage?.[key]?.[today]) || 0;
547
+ const pool = (() => {
548
+ // Провайдеры, у которых дневной лимит ещё не исчерпан (strict < limit).
549
+ const exemptables = windowPool.filter(([_, p]) => {
550
+ const limit = p.dailyLimit || 0;
551
+ if (limit <= 0) return true; // нет лимита — считаем «бесконечным»
552
+ return usedTodayFor(p.key) < limit;
553
+ });
554
+ // Есть живые → только они. Все выгорели → вернуть весь пул (best-effort,
555
+ // bandit-штраф ниже сделает их маловероятными, но не невозможными).
556
+ return exemptables.length > 0 ? exemptables : windowPool;
557
+ })();
276
558
 
277
559
  let selected = [];
278
560
  if (pool.length > 0) {
279
- const scored = pool.map(([key, provider]) => {
280
- const h = getHealth()[key];
281
- let score = h.score || 50;
282
- const rawLat = h.latency || 0;
283
- const lat = rawLat > 0 ? Math.max(rawLat, 100) : 500;
284
- let weight = score / lat;
285
- if (h.status === 'ratelimited') weight *= 0.05;
286
- if (key === targetProviderKey) weight *= 1.15;
287
- const dailyLimit = provider.dailyLimit || 1000;
288
- const usedToday = getStats().providerUsage[key] || 0;
289
- if (usedToday >= dailyLimit * 0.9) weight *= 0.5;
290
- return { key, provider, weight };
291
- });
561
+ if (upgradeNoCapable) {
562
+ // Bandit здесь бессилен: все провайдеры в пуле переполнят окно (запрос
563
+ // больше самого большого). Берём самое большое окно — наименьшие потери.
564
+ selected = pool
565
+ .map(([k, p]) => ({ key: k, provider: p }))
566
+ .sort((a, b) => (b.provider.context_window || 0) - (a.provider.context_window || 0))
567
+ .slice(0, 1);
568
+ } else {
569
+ const scored = pool.map(([key, provider]) => {
570
+ const h = getHealth()[key];
571
+ let score = h.score || 50;
572
+ const rawLat = h.latency || 0;
573
+ const lat = rawLat > 0 ? Math.max(rawLat, 100) : 500;
574
+ let weight = score / lat;
575
+ if (h.status === 'ratelimited') weight *= 0.05;
576
+ if (key === targetProviderKey) weight *= 1.15;
577
+ // Методолог: категория задачи задаёт буст моделям подходящей категории,
578
+ // не исключая fallback. coding→coding, reasoning→reasoning, chat/search→general.
579
+ const cat = provider.category || 'general';
580
+ if (taskCategory === 'coding' && cat === 'coding') weight *= 1.5;
581
+ else if (taskCategory === 'reasoning' && cat === 'reasoning') weight *= 1.5;
582
+ else if ((taskCategory === 'chat' || taskCategory === 'search') && cat === 'general') weight *= 1.2;
583
+ // Простой факт/поиск не должен «думать вслух» на reasoning-модели:
584
+ // это дорого по лимитам и даёт размышления вместо краткого ответа.
585
+ if ((taskCategory === 'chat' || taskCategory === 'search') && cat === 'reasoning') weight *= 0.15;
586
+ else if ((taskCategory === 'chat' || taskCategory === 'search') && cat === 'vision') weight *= 0.3;
587
+ const dailyLimit = provider.dailyLimit || 0;
588
+ // Провайдер, исчерпавший дневной лимит (попал в пул лишь как крайний
589
+ // резерв, когда живы все выгорели), — сильно штрафуем, чтобы выбрать
590
+ // его только в безвыходной ситуации, а не в первой же попытке.
591
+ const usedToday = usedTodayFor(key);
592
+ if (dailyLimit > 0 && usedToday >= dailyLimit) weight *= 0.03;
593
+ else if (dailyLimit > 0 && usedToday >= dailyLimit * 0.9) weight *= 0.4;
594
+ return { key, provider, weight };
595
+ });
292
596
 
293
- // Bandit weight contract: bandit's pick() multiplies the Beta sample by
294
- // `weight`, so safety-штрафы (ratelimited ×0.05, target ×1.15) действуют и
295
- // при холодном старте. score/latency держит вес ~0.01-1.0; приоры bandit'а
296
- // (a,b ~1+) со временем начинают доминировать. Не добавляй нормализацию
297
- // здесь, пока измеренные веса не превысят ~5.
298
- // Thompson sampling: рисуем сэмпл Beta(a+1, b+1) для каждого, умножаем на
299
- // weight, выбираем максимум. Приоры из бакета сложности (bandit обучается).
300
- const priors = getBandit()[complexityBucket] || {};
301
- const bestKey = banditPick(scored, priors);
302
- const bestProvider = scored.find((p) => p.key === bestKey);
303
- if (bestProvider) selected = [bestProvider];
304
- else if (scored.length > 0) selected = [scored[0]];
597
+ // Bandit weight contract: bandit's pick() multiplies the Beta sample by
598
+ // `weight`, so safety-штрафы (ratelimited ×0.05, target ×1.15) действуют и
599
+ // при холодном старте. score/latency держит вес ~0.01-1.0; приоры bandit'а
600
+ // (a,b ~1+) со временем начинают доминировать. Не добавляй нормализацию
601
+ // здесь, пока измеренные веса не превысят ~5.
602
+ // Thompson sampling: рисуем сэмпл Beta(a+1, b+1) для каждого, умножаем на
603
+ // weight, выбираем максимум. Приоры из бакета сложности (bandit обучается).
604
+ const priors = getBandit()[complexityBucket] || {};
605
+ const bestKey = banditPick(scored, priors);
606
+ const bestProvider = scored.find((p) => p.key === bestKey);
607
+ if (bestProvider) selected = [bestProvider];
608
+ else if (scored.length > 0) selected = [scored[0]];
609
+ }
305
610
  }
306
611
 
307
612
  // Weighted-random picked ONE provider as the primary; append the rest of the
@@ -310,21 +615,37 @@ async function handleChatCompletion(req, res, body) {
310
615
  ? pool.map(([k, p]) => ({ key: k, provider: p })).filter(s => s.key !== selected[0].key)
311
616
  .sort((a, b) => (getHealth()[b.key]?.score || 0) - (getHealth()[a.key]?.score || 0))
312
617
  : [];
313
- const enabledProviders = selected.length > 0
618
+ let enabledProviders = selected.length > 0
314
619
  ? [selected[0]].concat(restOfPool).map(s => [s.key, s.provider])
315
620
  : Object.entries(PROVIDERS).filter(([_, p]) => p.enabled)
316
621
  .sort((a, b) => (getHealth()[b[0]]?.score || 50) - (getHealth()[a[0]]?.score || 50));
317
622
 
318
- // Ensure the requested model's mapped provider is at least IN the candidate
319
- // list (it may have been filtered out), but DON'T force it to the front —
320
- // the weighted selection above should pick the fastest/healthiest provider.
321
- if (MODEL_MAP[requestedModel] && PROVIDERS[targetProviderKey]) {
322
- if (!enabledProviders.some(([k]) => k === targetProviderKey)) {
323
- enabledProviders.unshift([targetProviderKey, PROVIDERS[targetProviderKey]]);
623
+ // Put the requested model's mapped provider FIRST. It's the only provider
624
+ // guaranteed to accept this tier/model — the rest are fallbacks (many reject
625
+ // tier-* requests with 400/422). Trying them before the target produced huge
626
+ // serial fallback chains (10+ sequential HTTP calls per request), which looked
627
+ // like the "model thinking forever". Correct mapping beats weighted guessing.
628
+ // Skipped in window-upgrade mode: the target doesn't fit the request anyway.
629
+ // If the target is rate-limited or near/over its daily limit, DON'T put it
630
+ // first — otherwise every request burns a doomed 429 attempt on it and the
631
+ // pool collapses into a rate-limit spiral. Skip straight to the healthy pool.
632
+ if (!windowUpgraded && MODEL_MAP[requestedModel] && PROVIDERS[targetProviderKey]) {
633
+ const tHealth = getHealth()[targetProviderKey];
634
+ const tLimit = (getStats().dailyUsage?.[targetProviderKey]?.[today]) || 0;
635
+ const tCap = PROVIDERS[targetProviderKey].dailyLimit || 0;
636
+ const targetBurned = tHealth?.status === 'ratelimited' || (tCap > 0 && tLimit >= tCap * 0.9);
637
+ if (!targetBurned) {
638
+ enabledProviders = [
639
+ [targetProviderKey, PROVIDERS[targetProviderKey]],
640
+ ...enabledProviders.filter(([k]) => k !== targetProviderKey),
641
+ ];
642
+ } else {
643
+ logger.info('Target-first skip', { key: targetProviderKey, status: tHealth?.status, used: tLimit, cap: tCap });
324
644
  }
325
645
  }
326
646
 
327
647
  if (enabledProviders.length === 0) {
648
+ commit(503);
328
649
  res.writeHead(503, { 'Content-Type': 'application/json' });
329
650
  res.end(JSON.stringify({ error: 'No providers available' }));
330
651
  return;
@@ -332,7 +653,13 @@ async function handleChatCompletion(req, res, body) {
332
653
 
333
654
  const errors = [];
334
655
 
335
- for (const [key, provider] of enabledProviders) {
656
+ // Cap serial fallback attempts. Trying provider after provider sequentially
657
+ // made a single tier-* request walk 10+ providers (each a real HTTP call),
658
+ // looking like the model "thinks forever". target-first above fixes the common
659
+ // case (right provider immediately); this cap bounds the worst case.
660
+ const fallbackProviders = enabledProviders.slice(0, 5);
661
+
662
+ for (const [key, provider] of fallbackProviders) {
336
663
  if (isCircuitOpen(key)) {
337
664
  errors.push(key + ': circuit breaker open');
338
665
  continue;
@@ -354,14 +681,13 @@ async function handleChatCompletion(req, res, body) {
354
681
  getHealth()[key].latency = Math.min(result.latency || 0, 60000);
355
682
  getHealth()[key].lastCheck = Date.now();
356
683
 
357
- // For non-stream, verify the response isn't empty BEFORE recording success.
358
684
  if (!isStreaming && result.data) {
359
- delete result.data.nvext;
360
- if (result.data.choices?.[0]) {
361
- fixReasoningMessage(result.data.choices[0].message);
362
- cleanMessage(result.data.choices[0].message);
363
- }
364
- if (isTooShort(result.data)) {
685
+ delete result.data.nvext;
686
+ if (result.data.choices?.[0]) {
687
+ fixReasoningMessage(result.data.choices[0].message);
688
+ cleanMessage(result.data.choices[0].message);
689
+ }
690
+ if (isTooShort(result.data, lastUserText(body.messages))) {
365
691
  // Пустой/мусорный ответ (провайдер-глитч) НЕ считается успехом — пробуем следующего.
366
692
  const msg = key + ': empty or too short response';
367
693
  errors.push(msg);
@@ -375,6 +701,7 @@ async function handleChatCompletion(req, res, body) {
375
701
  }
376
702
 
377
703
  if (isStreaming && result.stream) {
704
+ measure.provider = key;
378
705
  const chunks = [];
379
706
 
380
707
  // Очистка SSE-строки: убрать nvext, logprobs, think-блоки из дельт.
@@ -392,6 +719,7 @@ async function handleChatCompletion(req, res, body) {
392
719
  });
393
720
 
394
721
  // Сбор контент-токенов для кэша (strip think).
722
+ const streamUsage = {};
395
723
  const collect = (str) => {
396
724
  const lines = str.split('\n');
397
725
  for (const line of lines) {
@@ -401,6 +729,12 @@ async function handleChatCompletion(req, res, body) {
401
729
  const obj = JSON.parse(m[1]);
402
730
  const delta = obj.choices?.[0]?.delta?.content;
403
731
  if (typeof delta === 'string') chunks.push(stripThink(delta, false));
732
+ // OpenRouter / Nebius etc. put usage in a final chunk. Keep it.
733
+ if (obj.usage && (obj.usage.prompt_tokens || obj.usage.completion_tokens)) {
734
+ streamUsage.prompt_tokens = obj.usage.prompt_tokens;
735
+ streamUsage.completion_tokens = obj.usage.completion_tokens;
736
+ streamUsage.total_tokens = obj.usage.total_tokens;
737
+ }
404
738
  } catch {}
405
739
  }
406
740
  };
@@ -416,17 +750,29 @@ async function handleChatCompletion(req, res, body) {
416
750
  recordRecent({ model: requestedModel, provider: key, status: 200, latency: result.latency, cached: false });
417
751
  recordSelection(key, provider.model, requestedModel);
418
752
  const { Transform } = require('stream');
753
+ const reasonDec = new StringDecoder('utf8');
754
+ const reasonUsage = {};
419
755
  const cleaner = new Transform({
420
756
  transform(chunk, encoding, callback) {
421
- const str = chunk.toString();
757
+ const str = reasonDec.write(chunk);
758
+ collectReasonUsage(str, reasonUsage);
422
759
  collect(str);
423
760
  callback(null, cleanStr(str));
761
+ },
762
+ flush(callback) {
763
+ const tail = reasonDec.end();
764
+ if (tail) { collectReasonUsage(tail, reasonUsage); collect(tail); const tailStr = cleanStr(tail); if (tailStr) this.push(tailStr); }
765
+ callback();
424
766
  }
425
767
  });
426
768
  result.stream.on('end', () => {
427
769
  const full = chunks.join('');
428
770
  // Bandit учится по качеству: пустой/мусорный стрим = фейл.
429
771
  recordBandit(complexityBucket, key, full.trim().length >= MIN_ANSWER_LEN);
772
+ if (Object.keys(reasonUsage).length > 0) {
773
+ recordTokens(key, reasonUsage);
774
+ if (reasonUsage.prompt_tokens) measure.real = reasonUsage.prompt_tokens;
775
+ }
430
776
  if (full.trim().length >= MIN_ANSWER_LEN) {
431
777
  cache.set(effectiveModel, body.messages, body.temperature, {
432
778
  id: 'chatcmpl-cached',
@@ -437,11 +783,13 @@ async function handleChatCompletion(req, res, body) {
437
783
  usage: { prompt_tokens: 0, completion_tokens: 0, total_tokens: 0 },
438
784
  });
439
785
  }
786
+ commit(200);
440
787
  res.end();
441
788
  });
442
789
  result.stream.on('error', (err) => {
443
790
  logger.error('Stream error', { key, error: err.message });
444
791
  if (!isTransientLimit(err.statusCode)) recordBandit(complexityBucket, key, false);
792
+ commit(err.statusCode || 502);
445
793
  res.end();
446
794
  });
447
795
  result.stream.pipe(cleaner).pipe(res);
@@ -450,13 +798,24 @@ async function handleChatCompletion(req, res, body) {
450
798
 
451
799
  // Обычные модели: буферизуем до первого токена (макс 5 сек).
452
800
  // Заголовки не пишем сразу — если токена нет за 5 сек, fallback.
801
+ // StringDecoder держит частично пришедший multi-byte UTF-8 между чанками —
802
+ // иначе русский текст дробится на '' символ.
453
803
  const rawBuf = [];
804
+ const streamDec = new StringDecoder('utf8');
805
+ // Адаптивный таймаут первого токена: используем измеренную скорость
806
+ // провайдера (latency от реальных запросов). Быстрые провайдеры не ждут
807
+ // полные 5с перед fallback'ом, а медленные (но рабочие) не отбрасываются
808
+ // слишком рано. Диапазон 2.5-8с для защиты от обоих крайностей.
809
+ const knownLat = getHealth()[key]?.latency || 0;
810
+ const firstTokenWait = knownLat > 0
811
+ ? Math.max(2500, Math.min(8000, Math.round(knownLat * 2)))
812
+ : 5000;
454
813
  const firstToken = new Promise((resolve) => {
455
814
  let done = false;
456
- const timer = setTimeout(() => { if (!done) { done = true; resolve(false); } }, 5000);
815
+ const timer = setTimeout(() => { if (!done) { done = true; resolve(false); } }, firstTokenWait);
457
816
  const finish = (ok) => { if (!done) { done = true; clearTimeout(timer); resolve(ok); } };
458
817
  result.stream.on('data', (chunk) => {
459
- const str = chunk.toString();
818
+ const str = streamDec.write(chunk);
460
819
  rawBuf.push(str);
461
820
  collect(str);
462
821
  // Первый контент-токен: хотя бы одно непустое `"content":"..."` в чанке.
@@ -503,16 +862,27 @@ async function handleChatCompletion(req, res, body) {
503
862
 
504
863
  // Убираем наш 'data'-слушатель (он больше не нужен — данные уже
505
864
  // буферизованы в rawBuf и промыты). Дальше обрабатываем вручную.
865
+ // Продолжаем использовать ТОТ ЖЕ streamDec — иначе multi-byte UTF-8,
866
+ // разделённый границей буфера, превратится в '' .
506
867
  result.stream.removeAllListeners('data');
507
868
  result.stream.on('data', (chunk) => {
508
- const str = chunk.toString();
869
+ const str = streamDec.write(chunk);
509
870
  collect(str);
510
871
  res.write(cleanStr(str));
511
872
  });
512
873
  result.stream.on('end', () => {
874
+ const tail = streamDec.end();
875
+ if (tail) {
876
+ collect(tail);
877
+ res.write(cleanStr(tail));
878
+ }
513
879
  const full = chunks.join('');
514
880
  // Bandit учится по качеству: обрыв/мусорный стрим = фейл.
515
881
  recordBandit(complexityBucket, key, full.trim().length >= MIN_ANSWER_LEN);
882
+ if (Object.keys(streamUsage).length > 0) {
883
+ recordTokens(key, streamUsage);
884
+ if (streamUsage.prompt_tokens) measure.real = streamUsage.prompt_tokens;
885
+ }
516
886
  if (full.trim().length >= MIN_ANSWER_LEN) {
517
887
  cache.set(effectiveModel, body.messages, body.temperature, {
518
888
  id: 'chatcmpl-cached',
@@ -523,11 +893,13 @@ async function handleChatCompletion(req, res, body) {
523
893
  usage: { prompt_tokens: 0, completion_tokens: 0, total_tokens: 0 },
524
894
  });
525
895
  }
896
+ commit(200);
526
897
  res.end();
527
898
  });
528
899
  result.stream.on('error', (err) => {
529
900
  logger.error('Stream error', { key, error: err.message });
530
901
  if (!isTransientLimit(err.statusCode)) recordBandit(complexityBucket, key, false);
902
+ commit(err.statusCode || 502);
531
903
  res.end();
532
904
  });
533
905
  return;
@@ -541,6 +913,10 @@ async function handleChatCompletion(req, res, body) {
541
913
  logger.request({ model: requestedModel, provider: key, status: 200, latency: result.latency, stream: isStreaming });
542
914
  recordRecent({ model: requestedModel, provider: key, status: 200, latency: result.latency, cached: false });
543
915
  recordSelection(key, provider.model, requestedModel);
916
+ measure.provider = key;
917
+ measure.real = (result.data.usage && result.data.usage.prompt_tokens) ? result.data.usage.prompt_tokens : measure.sentTokens || 0;
918
+ measure.win = PROVIDERS[key]?.context_window || 0;
919
+ commit(200);
544
920
  cache.set(effectiveModel, body.messages, body.temperature, result.data);
545
921
  recordTokens(key, result.usage);
546
922
  res.writeHead(200, { 'Content-Type': 'application/json' });
@@ -569,13 +945,24 @@ async function handleChatCompletion(req, res, body) {
569
945
  getHealth()[key].reason = 'не отвечает';
570
946
  }
571
947
  getHealth()[key].score = Math.max(0, (getHealth()[key].score || 50) - (statusCode === 429 ? 5 : 10));
572
- recordFailure(key, statusCode);
573
- // 404 = model not available for this account — disable permanently
948
+ recordFailure(key, statusCode, statusCode === 404 && err.providerSide ? { providerSide: true } : undefined);
949
+ // 404 = model not available. Two distinct flavors:
950
+ // - plain 404 («does not exist») → disable permanently (model gone from OpenRouter).
951
+ // - provider-side 404 (Nvidia quota/upstream failing) → NOT permanent — the model
952
+ // may recover. Mark it down hard so routing prefers others, and let the periodic
953
+ // health-check re-enable it when it comes back.
574
954
  if (statusCode === 404) {
575
- provider.enabled = false;
576
- getHealth()[key].status = 'disabled';
577
- getHealth()[key].reason = 'отключён автоматически (404)';
578
- logger.warn('Provider auto-disabled (404)', { key, model: provider.model });
955
+ if (err.providerSide) {
956
+ getHealth()[key].status = 'error';
957
+ getHealth()[key].reason = 'провайдер временно недоступен (404)';
958
+ getHealth()[key].score = Math.max(0, (getHealth()[key].score || 50) - 25);
959
+ logger.warn('Provider temporarily down (provider-side 404)', { key, model: provider.model });
960
+ } else {
961
+ provider.enabled = false;
962
+ getHealth()[key].status = 'disabled';
963
+ getHealth()[key].reason = 'отключён автоматически (404)';
964
+ logger.warn('Provider auto-disabled (404)', { key, model: provider.model });
965
+ }
579
966
  }
580
967
  }
581
968
  }
@@ -585,7 +972,7 @@ async function handleChatCompletion(req, res, body) {
585
972
  const allSoft = errors.length > 0 && errors.every(e => !/429|401|403|404/.test(e));
586
973
  if (allSoft && enabledProviders.length > 1) {
587
974
  await new Promise(r => setTimeout(r, 1500));
588
- for (const [key, provider] of enabledProviders) {
975
+ for (const [key, provider] of fallbackProviders) {
589
976
  if (isCircuitOpen(key)) continue;
590
977
  try {
591
978
  const release = await acquire(key);
@@ -601,7 +988,7 @@ async function handleChatCompletion(req, res, body) {
601
988
  fixReasoningMessage(result.data.choices[0].message);
602
989
  cleanMessage(result.data.choices[0].message);
603
990
  }
604
- if (isTooShort(result.data)) {
991
+ if (isTooShort(result.data, lastUserText(body.messages))) {
605
992
  recordFailure(key, 0);
606
993
  recordRequest(key, false, key + ': empty or too short response (retry)');
607
994
  recordBandit(complexityBucket, key, false);
@@ -613,6 +1000,10 @@ async function handleChatCompletion(req, res, body) {
613
1000
  recordBandit(complexityBucket, key, true);
614
1001
  recordRecent({ model: requestedModel, provider: key, status: 200, latency: result.latency, cached: false });
615
1002
  recordSelection(key, provider.model, requestedModel);
1003
+ measure.provider = key;
1004
+ measure.real = (result.data.usage && result.data.usage.prompt_tokens) ? result.data.usage.prompt_tokens : measure.sentTokens || 0;
1005
+ measure.win = PROVIDERS[key]?.context_window || 0;
1006
+ commit(200);
616
1007
  cache.set(effectiveModel, body.messages, body.temperature, result.data);
617
1008
  recordTokens(key, result.usage);
618
1009
  res.writeHead(200, { 'Content-Type': 'application/json' });
@@ -620,6 +1011,7 @@ async function handleChatCompletion(req, res, body) {
620
1011
  return;
621
1012
  }
622
1013
  if (body.stream && result.stream) {
1014
+ measure.provider = key;
623
1015
  recordSuccess(key);
624
1016
  recordRequest(key, true);
625
1017
  recordRecent({ model: requestedModel, provider: key, status: 200, latency: result.latency, cached: false });
@@ -650,12 +1042,15 @@ async function handleChatCompletion(req, res, body) {
650
1042
  return 'data: ' + JSON.stringify(obj);
651
1043
  } catch { return match; }
652
1044
  });
1045
+ const retryDec = new StringDecoder('utf8');
653
1046
  result.stream.on('data', (chunk) => {
654
- const str = chunk.toString();
1047
+ const str = retryDec.write(chunk);
655
1048
  collectRetry(str);
656
1049
  res.write(cleanRetry(str));
657
1050
  });
658
1051
  result.stream.on('end', () => {
1052
+ const tail = retryDec.end();
1053
+ if (tail) { collectRetry(tail); res.write(cleanRetry(tail)); }
659
1054
  const full = chunks.join('');
660
1055
  // Bandit учится по качеству в ретрае тоже.
661
1056
  recordBandit(complexityBucket, key, full.trim().length >= MIN_ANSWER_LEN);
@@ -669,11 +1064,13 @@ async function handleChatCompletion(req, res, body) {
669
1064
  usage: { prompt_tokens: 0, completion_tokens: 0, total_tokens: 0 },
670
1065
  });
671
1066
  }
1067
+ commit(200);
672
1068
  res.end();
673
1069
  });
674
1070
  result.stream.on('error', (err) => {
675
1071
  logger.error('Stream error (retry)', { key, error: err.message });
676
1072
  if (!isTransientLimit(err.statusCode)) recordBandit(complexityBucket, key, false);
1073
+ commit(err.statusCode || 502);
677
1074
  res.end();
678
1075
  });
679
1076
  return;
@@ -686,6 +1083,7 @@ async function handleChatCompletion(req, res, body) {
686
1083
  }
687
1084
  }
688
1085
 
1086
+ commit(502);
689
1087
  res.writeHead(502, { 'Content-Type': 'application/json' });
690
1088
  res.end(JSON.stringify({ error: { message: 'All providers failed', type: 'api_error', code: 'all_providers_failed', details: errors } }));
691
1089
  }
@@ -715,12 +1113,145 @@ const server = http.createServer(async (req, res) => {
715
1113
  return;
716
1114
  }
717
1115
 
1116
+ if (parsedUrl.pathname === '/v1/reload' && req.method === 'POST') {
1117
+ // Hot-reload providers.json + config.json without restarting the server.
1118
+ // Auth-protected like the other admin endpoints.
1119
+ if (AUTH_KEY) {
1120
+ const apiKey = (req.headers.authorization || '').replace('Bearer ', '').trim();
1121
+ const keyFromQuery = parsedUrl.searchParams.get('key');
1122
+ if (apiKey !== AUTH_KEY && keyFromQuery !== AUTH_KEY) {
1123
+ res.writeHead(401, { 'Content-Type': 'application/json' });
1124
+ res.end(JSON.stringify({ error: { message: 'Invalid API key' } }));
1125
+ return;
1126
+ }
1127
+ }
1128
+ try {
1129
+ const result = reloadProviders();
1130
+ res.writeHead(200, { 'Content-Type': 'application/json' });
1131
+ res.end(JSON.stringify({ ok: true, ...result }));
1132
+ } catch (err) {
1133
+ res.writeHead(500, { 'Content-Type': 'application/json' });
1134
+ res.end(JSON.stringify({ error: { message: 'Reload failed: ' + err.message } }));
1135
+ }
1136
+ return;
1137
+ }
1138
+
1139
+ if (parsedUrl.pathname === '/v1/models-db' && req.method === 'GET') {
1140
+ // Структурированная база моделей: паспорта + статистика + топ по скору.
1141
+ if (AUTH_KEY) {
1142
+ const apiKey = (req.headers.authorization || '').replace('Bearer ', '').trim();
1143
+ const keyFromQuery = parsedUrl.searchParams.get('key');
1144
+ if (apiKey !== AUTH_KEY && keyFromQuery !== AUTH_KEY) {
1145
+ res.writeHead(401, { 'Content-Type': 'application/json' });
1146
+ res.end(JSON.stringify({ error: { message: 'Invalid API key' } }));
1147
+ return;
1148
+ }
1149
+ }
1150
+ try {
1151
+ const models = modelManager.db.all()
1152
+ .map(m => ({
1153
+ key: m.key, model: m.model, source: m.source, category: m.category,
1154
+ contextWindow: m.contextWindow, dailyLimit: m.dailyLimit,
1155
+ score: m.score || 0, status: m.status,
1156
+ lastCheckedAt: m.lastCheckedAt || null, lastOkAt: m.lastOkAt || null,
1157
+ }))
1158
+ .sort((a, b) => b.score - a.score);
1159
+ res.writeHead(200, { 'Content-Type': 'application/json' });
1160
+ res.end(JSON.stringify({
1161
+ stats: modelManager.db.stats(),
1162
+ top: models.slice(0, 10),
1163
+ models,
1164
+ manager: { enabled: MODEL_MANAGER_CONFIG.enabled !== false, intervalHours: modelManager.config.intervalHours, running: modelManager._running },
1165
+ }));
1166
+ } catch (err) {
1167
+ res.writeHead(500, { 'Content-Type': 'application/json' });
1168
+ res.end(JSON.stringify({ error: { message: err.message } }));
1169
+ }
1170
+ return;
1171
+ }
1172
+
1173
+ // Парсим /v1/models/{key}/toggle и /v1/models/{key}/test
1174
+ const modelActionMatch = parsedUrl.pathname.match(/^\/v1\/models\/([^/]+)\/(toggle|test)$/);
1175
+ if (modelActionMatch && req.method === 'POST') {
1176
+ if (AUTH_KEY) {
1177
+ const apiKey = (req.headers.authorization || '').replace('Bearer ', '').trim();
1178
+ const keyFromQuery = parsedUrl.searchParams.get('key');
1179
+ if (apiKey !== AUTH_KEY && keyFromQuery !== AUTH_KEY) {
1180
+ res.writeHead(401, { 'Content-Type': 'application/json' });
1181
+ res.end(JSON.stringify({ error: { message: 'Invalid API key' } }));
1182
+ return;
1183
+ }
1184
+ }
1185
+ const modelKey = decodeURIComponent(modelActionMatch[1]);
1186
+ const action = modelActionMatch[2];
1187
+
1188
+ if (action === 'test') {
1189
+ // Живой "hi"-тест: работает ли модель с нашим ключом. Без изменения каталога.
1190
+ const dbEntry = modelManager.db.get(modelKey);
1191
+ const prov = PROVIDERS[modelKey];
1192
+ const endpoint = (dbEntry && dbEntry.endpoint) || (prov && prov.endpoint);
1193
+ const model = (dbEntry && dbEntry.model) || (prov && prov.model);
1194
+ if (!endpoint || !model) {
1195
+ res.writeHead(404, { 'Content-Type': 'application/json' });
1196
+ res.end(JSON.stringify({ ok: false, error: 'Модель ' + modelKey + ' не найдена' }));
1197
+ return;
1198
+ }
1199
+ const source = (dbEntry && dbEntry.source) || 'unknown';
1200
+ const apiKey = modelManager.apiKeyFor(source);
1201
+ try {
1202
+ const t0 = Date.now();
1203
+ const r = await fetch(endpoint, {
1204
+ method: 'POST',
1205
+ headers: { 'Content-Type': 'application/json', ...(apiKey ? { Authorization: 'Bearer ' + apiKey } : {}) },
1206
+ body: JSON.stringify({ model, messages: [{ role: 'user', content: 'hi' }], max_tokens: 5 }),
1207
+ signal: AbortSignal.timeout(20000),
1208
+ });
1209
+ const latencyMs = Date.now() - t0;
1210
+ // Обновляем базу результатом проверки (но не «активируем» принудительно).
1211
+ modelManager.db.markChecked(modelKey, { ok: r.ok, status: r.status, latencyMs }, { now: Date.now() });
1212
+ modelManager.db.save();
1213
+ res.writeHead(200, { 'Content-Type': 'application/json' });
1214
+ res.end(JSON.stringify({ ok: r.ok, status: r.status, latencyMs, model: { key: modelKey, status: modelManager.db.get(modelKey).status } }));
1215
+ } catch (err) {
1216
+ res.writeHead(200, { 'Content-Type': 'application/json' });
1217
+ res.end(JSON.stringify({ ok: false, status: 0, error: err.message }));
1218
+ }
1219
+ return;
1220
+ }
1221
+
1222
+ // action === 'toggle': вкл/выкл провайдера в config.json (только этот ключ) + hot-reload.
1223
+ try {
1224
+ const userCfg = JSON.parse(fs.readFileSync(CONFIG_PATH, 'utf8'));
1225
+ if (!userCfg.providers) userCfg.providers = {};
1226
+ if (!userCfg.providers[modelKey]) userCfg.providers[modelKey] = {};
1227
+ // Flip: было включено → выключить, было выключено → включить.
1228
+ const currentlyEnabled = !(userCfg.providers[modelKey].enabled === false);
1229
+ const newEnabled = !currentlyEnabled;
1230
+ userCfg.providers[modelKey].enabled = newEnabled;
1231
+ fs.writeFileSync(CONFIG_PATH, JSON.stringify(userCfg, null, 2));
1232
+ reloadProviders();
1233
+ // Статус в базе: включили → untested (проверится в след. цикле), выключили → user-disabled.
1234
+ const entry = modelManager.db.get(modelKey);
1235
+ if (entry) {
1236
+ modelManager.db.setStatus(modelKey, newEnabled ? 'untested' : 'user-disabled');
1237
+ modelManager.db.save();
1238
+ }
1239
+ res.writeHead(200, { 'Content-Type': 'application/json' });
1240
+ res.end(JSON.stringify({ ok: true, key: modelKey, enabled: newEnabled, model: entry ? { key: modelKey, status: modelManager.db.get(modelKey).status } : null }));
1241
+ } catch (err) {
1242
+ res.writeHead(500, { 'Content-Type': 'application/json' });
1243
+ res.end(JSON.stringify({ error: { message: 'Toggle failed: ' + err.message } }));
1244
+ }
1245
+ return;
1246
+ }
1247
+
718
1248
  if (parsedUrl.pathname === '/health') {
719
1249
  const h = getHealth();
720
1250
  const upCount = Object.values(h).filter(v => v.status === 'up').length;
721
1251
  const totalCount = Object.entries(PROVIDERS).filter(([_, p]) => p.enabled).length;
1252
+ const contextSummary = (() => { try { return contextStats.summary(); } catch { return null; } })();
722
1253
  res.writeHead(upCount > 0 ? 200 : 503, { 'Content-Type': 'application/json' });
723
- res.end(JSON.stringify({ status: upCount > 0 ? 'ok' : 'degraded', providers: { up: upCount, total: totalCount } }));
1254
+ res.end(JSON.stringify({ status: upCount > 0 ? 'ok' : 'degraded', providers: { up: upCount, total: totalCount }, context: contextSummary }));
724
1255
  return;
725
1256
  }
726
1257
 
@@ -738,6 +1269,14 @@ const server = http.createServer(async (req, res) => {
738
1269
  res.end(JSON.stringify({
739
1270
  total_requests: s.totalRequests, successful_requests: s.successfulRequests, failed_requests: s.failedRequests,
740
1271
  provider_usage: s.providerUsage, token_usage: s.tokenUsage, errors: s.errors, uptime_seconds: Math.floor((Date.now() - s.startTime) / 1000),
1272
+ today: (() => {
1273
+ const r = getReliability();
1274
+ let success = 0, fail = 0;
1275
+ for (const [_, v] of Object.entries(r)) { if (v && v.day === today) { success += v.success || 0; fail += v.fail || 0; } }
1276
+ const total = success + fail;
1277
+ return { requests: total, success, failed: fail, successRate: total > 0 ? Math.round((success / total) * 100) : 0 };
1278
+ })(),
1279
+ savings: (() => { try { return aggregateSavings(s.tokenUsage || {}); } catch { return null; } })(),
741
1280
  health: Object.fromEntries(Object.entries(getHealth()).map(([k, v]) => {
742
1281
  const limit = limits[k];
743
1282
  const err = s.errors[k] || 0;
@@ -756,6 +1295,7 @@ const server = http.createServer(async (req, res) => {
756
1295
  pool: poolStats(),
757
1296
  last_selection: getLastSelection(),
758
1297
  bandit: getBandit(),
1298
+ context_summary: (() => { try { return contextStats.summary(); } catch { return null; } })(),
759
1299
  }));
760
1300
  return;
761
1301
  }
@@ -836,6 +1376,25 @@ const server = http.createServer(async (req, res) => {
836
1376
  return;
837
1377
  }
838
1378
 
1379
+ // POST /v1/cache/clear — drop the in-memory semantic cache without a restart.
1380
+ // Handy when you tweaked providers/models and don't want stale answers served.
1381
+ if (parsedUrl.pathname === '/v1/cache/clear' && req.method === 'POST') {
1382
+ if (AUTH_KEY) {
1383
+ const apiKey = (req.headers.authorization || '').replace('Bearer ', '').trim();
1384
+ const keyFromQuery = parsedUrl.searchParams.get('key');
1385
+ if (apiKey !== AUTH_KEY && keyFromQuery !== AUTH_KEY) {
1386
+ res.writeHead(401, { 'Content-Type': 'application/json' });
1387
+ res.end(JSON.stringify({ error: { message: 'Invalid API key' } }));
1388
+ return;
1389
+ }
1390
+ }
1391
+ const before = cache.stats().size || 0;
1392
+ cache.clear();
1393
+ res.writeHead(200, { 'Content-Type': 'application/json' });
1394
+ res.end(JSON.stringify({ ok: true, cleared: before }));
1395
+ return;
1396
+ }
1397
+
839
1398
  // POST /v1/shorts — generate a vertical short video via the tools generator.
840
1399
  // Body: { prompt, duration?, format? ("9:16"/"16:9"/"1:1"), steps? }
841
1400
  if (parsedUrl.pathname === '/v1/shorts' && req.method === 'POST') {
@@ -889,6 +1448,87 @@ const server = http.createServer(async (req, res) => {
889
1448
  return;
890
1449
  }
891
1450
 
1451
+ // --- Setup Dashboard API ---
1452
+ if (parsedUrl.pathname === '/v1/setup/keys' && req.method === 'GET') {
1453
+ const { keys } = readKeys();
1454
+ // Mask keys for display; inputs stay EMPTY so we never send masked values back.
1455
+ const masked = {};
1456
+ const empty = {};
1457
+ for (const [k, v] of Object.entries(keys)) {
1458
+ empty[k] = '';
1459
+ if (!v) { masked[k] = ''; continue; }
1460
+ if (v.length <= 10) { masked[k] = v.slice(0, 2) + '***' + v.slice(-2); continue; }
1461
+ masked[k] = v.slice(0, 4) + '***' + v.slice(-4);
1462
+ }
1463
+ res.writeHead(200, { 'Content-Type': 'application/json' });
1464
+ res.end(JSON.stringify({ groups: KEY_GROUPS, keys: empty, masked }));
1465
+ return;
1466
+ }
1467
+
1468
+ if (parsedUrl.pathname === '/v1/setup/keys' && req.method === 'POST') {
1469
+ if (AUTH_KEY) {
1470
+ const apiKey = (req.headers.authorization || '').replace('Bearer ', '').trim();
1471
+ const keyFromQuery = parsedUrl.searchParams.get('key');
1472
+ if (apiKey !== AUTH_KEY && keyFromQuery !== AUTH_KEY) {
1473
+ res.writeHead(401, { 'Content-Type': 'application/json' });
1474
+ res.end(JSON.stringify({ error: { message: 'Invalid API key' } }));
1475
+ return;
1476
+ }
1477
+ }
1478
+ let body = '';
1479
+ req.on('data', (c) => body += c);
1480
+ req.on('end', () => {
1481
+ try {
1482
+ const newKeys = JSON.parse(body);
1483
+ // Only accept known env vars
1484
+ const filtered = {};
1485
+ for (const k of Object.keys(KEY_GROUPS)) {
1486
+ if (typeof newKeys[k] === 'string') filtered[k] = newKeys[k];
1487
+ }
1488
+ const result = saveKeys(filtered);
1489
+ res.writeHead(200, { 'Content-Type': 'application/json' });
1490
+ res.end(JSON.stringify({ ok: true, ...result }));
1491
+ } catch (e) {
1492
+ res.writeHead(400, { 'Content-Type': 'application/json' });
1493
+ res.end(JSON.stringify({ error: { message: 'Invalid request: ' + e.message } }));
1494
+ }
1495
+ });
1496
+ return;
1497
+ }
1498
+
1499
+ if (parsedUrl.pathname === '/v1/setup/validate') {
1500
+ if (AUTH_KEY) {
1501
+ const apiKey = (req.headers.authorization || '').replace('Bearer ', '').trim();
1502
+ const keyFromQuery = parsedUrl.searchParams.get('key');
1503
+ if (apiKey !== AUTH_KEY && keyFromQuery !== AUTH_KEY) {
1504
+ res.writeHead(401, { 'Content-Type': 'application/json' });
1505
+ res.end(JSON.stringify({ error: { message: 'Invalid API key' } }));
1506
+ return;
1507
+ }
1508
+ }
1509
+ const envVar = parsedUrl.searchParams.get('envVar');
1510
+ let testKey = parsedUrl.searchParams.get('apiKey');
1511
+ if (!envVar) {
1512
+ res.writeHead(400, { 'Content-Type': 'application/json' });
1513
+ res.end(JSON.stringify({ error: { message: 'envVar required' } }));
1514
+ return;
1515
+ }
1516
+ // If apiKey param is empty/absent, validate the real stored key from .env.
1517
+ if (!testKey) {
1518
+ testKey = getStoredKey(envVar);
1519
+ }
1520
+ if (!testKey) {
1521
+ res.writeHead(200, { 'Content-Type': 'application/json' });
1522
+ res.end(JSON.stringify({ valid: false, error: 'Нет сохранённого ключа' }));
1523
+ return;
1524
+ }
1525
+ validateKey(envVar, testKey).then((result) => {
1526
+ res.writeHead(200, { 'Content-Type': 'application/json' });
1527
+ res.end(JSON.stringify(result));
1528
+ });
1529
+ return;
1530
+ }
1531
+
892
1532
  res.writeHead(404, { 'Content-Type': 'application/json' });
893
1533
  res.end(JSON.stringify({ error: 'Not found' }));
894
1534
  });
@@ -896,7 +1536,19 @@ const server = http.createServer(async (req, res) => {
896
1536
  server.listen(PORT, process.env.HOST || '127.0.0.1', () => {
897
1537
  logger.info('Freegate started', { port: PORT });
898
1538
  console.log('Dashboard: http://localhost:' + PORT + '/');
1539
+ // Самообновляющаяся база моделей: первый цикл через 2 мин, далее по интервалу.
1540
+ if (MODEL_MANAGER_CONFIG.enabled !== false) {
1541
+ modelManager.start();
1542
+ logger.info('ModelManager started', { intervalHours: modelManager.config.intervalHours });
1543
+ }
899
1544
  });
900
1545
 
901
- process.on('SIGINT', () => { require('./lib/health').saveState(); cache.persist(); server.close(() => process.exit(0)); });
902
- process.on('SIGTERM', () => { require('./lib/health').saveState(); cache.persist(); server.close(() => process.exit(0)); });
1546
+ const _shutdown = () => {
1547
+ if (memStore) { memStore.stopTimer(); memStore.save(); }
1548
+ require('./lib/health').saveState();
1549
+ cache.persist();
1550
+ try { modelManager.stop(); } catch {}
1551
+ server.close(() => process.exit(0));
1552
+ };
1553
+ process.on('SIGINT', _shutdown);
1554
+ process.on('SIGTERM', _shutdown);