freegate 0.6.14 → 0.6.16

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/server.js CHANGED
@@ -3,18 +3,26 @@ const http = require('http');
3
3
  const fs = require('fs');
4
4
  const path = require('path');
5
5
  const { LRUCache } = require('./lib/cache');
6
- const { PROVIDERS, MODEL_MAP, callProvider } = require('./lib/providers');
7
- const { loadState, initHealth, isCircuitOpen, recordSuccess, recordFailure, recordRequest, recordTokens, getHealth, getStats, recordRecent, recordRpm, getRecent, getRpm, recordSelection, getLastSelection, getBandit, recordBandit } = require('./lib/health');
6
+ const { PROVIDERS, MODEL_MAP, callProvider, reloadProviders } = require('./lib/providers');
7
+ const { loadState, initHealth, isCircuitOpen, recordSuccess, recordFailure, recordRequest, recordTokens, getHealth, getStats, recordRecent, recordRpm, getRecent, getRpm, recordSelection, getLastSelection, getBandit, recordBandit, warmBanditPriors, getContextStats } = require('./lib/health');
8
8
  const { checkRateLimit } = require('./lib/rateLimit');
9
9
  const { handleDashboard } = require('./lib/dashboard');
10
10
  const { acquire, stats: poolStats } = require('./lib/pool');
11
+ const { aggregateSavings } = require('./lib/economics');
11
12
  const { stripThink, cleanDelta, cleanMessage, fixReasoningMessage, isTooShort, MIN_ANSWER_LEN } = require('./lib/clean');
12
- const { classifyComplexity, maybeUpgradeTier, classifyVisionComplexity } = require('./lib/routing');
13
- const { bucket, pick: banditPick } = require('./lib/bandit');
13
+ const { classifyComplexity, maybeUpgradeTier, classifyVisionComplexity, needsWindowUpgrade } = require('./lib/routing');
14
+ const { bucket, pick: banditPick, isTransientLimit } = require('./lib/bandit');
15
+ const { StringDecoder } = require('string_decoder');
14
16
  const logger = require('./lib/logger');
17
+ const { KEY_GROUPS, readKeys, saveKeys, validateKey, getStoredKey } = require('./lib/setup');
18
+ const { prepareMessages, estimateTokens, setMemory: compactorSetMemory } = require('./lib/compactor');
19
+ const { classify: classifyTask } = require('./lib/taskclassify');
20
+ const { injectMethodology, enabledByDefault: methodEnabledByDefault } = require('./lib/methodology');
21
+ const { create: createMemoryStore } = require('./lib/memory-store');
15
22
 
16
23
  // Load persisted state
17
24
  loadState();
25
+ const contextStats = getContextStats();
18
26
 
19
27
  // Drop stale health entries for providers that no longer exist (e.g. auto-disabled)
20
28
  const activeKeys = new Set(Object.keys(PROVIDERS));
@@ -26,6 +34,14 @@ if (stale.length > 0) {
26
34
  logger.info('Cleaned stale health entries', { removed: stale });
27
35
  }
28
36
 
37
+ // Warm bandit priors for enabled providers with no/low history so brand-new
38
+ // models (e.g. or-minimax-m3-free) are explored promptly instead of ignored.
39
+ const warmed = warmBanditPriors(
40
+ Object.keys(PROVIDERS).filter(k => PROVIDERS[k].enabled !== false),
41
+ ['low', 'med', 'high']
42
+ );
43
+ if (warmed > 0) logger.info('Warmed bandit priors', { warmed });
44
+
29
45
  const cache = new LRUCache(500, 3600000, false, true); // 4th arg: semantic normalize ON
30
46
  require('./lib/cache')._activeCache = cache;
31
47
 
@@ -33,7 +49,7 @@ require('./lib/cache')._activeCache = cache;
33
49
  // Prefer cwd config.json (user's project) over the package dir.
34
50
  const CONFIG_CANDIDATES = [path.join(process.cwd(), 'config.json'), path.join(__dirname, 'config.json')];
35
51
  const CONFIG_PATH = CONFIG_CANDIDATES.find(p => fs.existsSync(p)) || CONFIG_CANDIDATES[1];
36
- let config = { port: 4000, auth: '', rateLimit: { maxRequests: 100, windowMs: 60000 } };
52
+ let config = { port: 4000, auth: '', rateLimit: { maxRequests: 100, windowMs: 60000 }, methodology: true };
37
53
  try {
38
54
  const parsed = JSON.parse(fs.readFileSync(CONFIG_PATH, 'utf8'));
39
55
  if (parsed && typeof parsed === 'object') config = { ...config, ...parsed };
@@ -51,6 +67,75 @@ const PORT = parseInt(process.env.PORT || getArg('port', config.port || '4000'))
51
67
  const AUTH_KEY = process.env.AUTH || getArg('auth', config.auth || '');
52
68
  const RATE_LIMIT = config.rateLimit || { maxRequests: 100, windowMs: 60000 };
53
69
 
70
+ // --- Долговременная память (vector memory) ---
71
+ // По умолчанию включена: факты берутся побочно от компакции (без лишних вызовов
72
+ // LLM) и подмешиваются в контекст по релевантности. config.memory.enabled=false —
73
+ // слой отключается целиком.
74
+ const MEMORY_CONFIG = Object.assign(
75
+ { enabled: true, topK: 3, minSimilarity: 0.05, coverageForSkip: 0.6 },
76
+ (config.memory && typeof config.memory === 'object') ? config.memory : {}
77
+ );
78
+ const memStore = MEMORY_CONFIG.enabled ? createMemoryStore({ filePath: path.join(__dirname, 'memory.json') }) : null;
79
+ if (memStore) compactorSetMemory(memStore);
80
+
81
+ // --- Семантический кэш ---
82
+ // На промахе точного ключа ищет закэшированный диалог с похожим нормализованным
83
+ // текстом (Dice по символьным триграммам). config.semcache.enabled=false отключает.
84
+ const SEMCACHE_CONFIG = Object.assign(
85
+ { enabled: true, minSimilarity: 0.85 },
86
+ (config.semcache && typeof config.semcache === 'object') ? config.semcache : {}
87
+ );
88
+
89
+ // --- Методолог (инженерная дисциплина в промпте) ---
90
+ // config.methodology.enabled=false отключает. По умолчанию включён.
91
+ // config.methodology.prompts.{category} переопределяет текст промпта для
92
+ // конкретной категории (незаданные категории остаются в дефолтах).
93
+ const METHODOLOGY_CONFIG = Object.assign(
94
+ { enabled: methodEnabledByDefault(), prompts: {} },
95
+ (config.methodology && typeof config.methodology === 'object') ? config.methodology : {}
96
+ );
97
+
98
+ // --- Самообновляющаяся база моделей (Model Discovery Engine) ---
99
+ // Планировщик живёт внутри сервера: работает «всегда» у всех пользователей
100
+ // пакета без cron/launchd. config.modelManager.enabled=false отключает.
101
+ function _loadEnvFile() {
102
+ try {
103
+ const envPath = path.join(__dirname, '.env');
104
+ if (!fs.existsSync(envPath)) return;
105
+ for (const line of fs.readFileSync(envPath, 'utf8').split('\n')) {
106
+ const t = line.trim();
107
+ if (!t || t.startsWith('#')) continue;
108
+ const eq = t.indexOf('=');
109
+ if (eq > 0 && !process.env[t.slice(0, eq).trim()]) process.env[t.slice(0, eq).trim()] = t.slice(eq + 1).trim();
110
+ }
111
+ } catch {}
112
+ }
113
+ _loadEnvFile();
114
+ const { ModelManager } = require('./lib/modelmanager');
115
+ const MODEL_MANAGER_CONFIG = Object.assign(
116
+ {},
117
+ (config.modelManager && typeof config.modelManager === 'object') ? config.modelManager : {}
118
+ );
119
+ const modelManager = new ModelManager({
120
+ dbPath: path.join(__dirname, 'models-db.json'),
121
+ catalogPath: path.join(__dirname, 'providers.json'),
122
+ configPath: path.join(__dirname, 'config.json'),
123
+ keys: {
124
+ openrouter: process.env.PROVIDER_OPENROUTER_APIKEY || '',
125
+ huggingface: process.env.HF_TOKEN || process.env.PROVIDER_HF_APIKEY || '',
126
+ groq: process.env.PROVIDER_GROQ_APIKEY || '',
127
+ mistral: process.env.PROVIDER_MISTRAL_APIKEY || '',
128
+ gemini: process.env.PROVIDER_GEMINI_APIKEY || '',
129
+ cerebras: process.env.PROVIDER_CEREBRAS_APIKEY || '',
130
+ deepseek: process.env.PROVIDER_DEEPSEEK_APIKEY || '',
131
+ nim: process.env.PROVIDER_NIM_APIKEY || '',
132
+ },
133
+ fetchImpl: (url, opts) => fetch(url, opts),
134
+ config: MODEL_MANAGER_CONFIG,
135
+ reload: () => { try { reloadProviders(); } catch {} },
136
+ log: (msg) => logger.info('[modelManager] ' + msg),
137
+ });
138
+
54
139
  // Health check
55
140
  const healthIntervals = {}; // key -> { nextCheck, backoff }
56
141
 
@@ -123,6 +208,26 @@ async function healthCheck() {
123
208
  setInterval(healthCheck, 30000);
124
209
  setTimeout(healthCheck, 1000);
125
210
 
211
+ // Периодический сейв долговременной памяти (вдобавок к shutdown).
212
+ if (memStore) setInterval(() => memStore.save(), 60000);
213
+
214
+ // Извлекает usage из SSE-чанка (если провайдер шлёт его в последнем чанке).
215
+ function collectReasonUsage(str, usageObj) {
216
+ if (!str || !/data: /.test(str)) return;
217
+ for (const line of str.split('\n')) {
218
+ const m = line.match(/^data: (.+)$/);
219
+ if (!m || m[1].trim() === '[DONE]') continue;
220
+ try {
221
+ const obj = JSON.parse(m[1]);
222
+ if (obj.usage && (obj.usage.prompt_tokens || obj.usage.completion_tokens)) {
223
+ usageObj.prompt_tokens = obj.usage.prompt_tokens;
224
+ usageObj.completion_tokens = obj.usage.completion_tokens;
225
+ usageObj.total_tokens = obj.usage.total_tokens;
226
+ }
227
+ } catch {}
228
+ }
229
+ }
230
+
126
231
  // Извлекает все значения content из SSE-чанка. Возвращает true, если есть
127
232
  // хотя бы одно непустое (реальный токен, а не пустая дельта).
128
233
  function chunkHasToken(str) {
@@ -134,6 +239,22 @@ function chunkHasToken(str) {
134
239
  return false;
135
240
  }
136
241
 
242
+ // Последний user-текст — для оценки, насколько короткий ответ легитимен.
243
+ function lastUserText(messages) {
244
+ if (!Array.isArray(messages)) return '';
245
+ for (let i = messages.length - 1; i >= 0; i--) {
246
+ const m = messages[i];
247
+ if (m && m.role === 'user') {
248
+ if (typeof m.content === 'string') return m.content;
249
+ if (Array.isArray(m.content)) {
250
+ const t = m.content.filter(c => c && c.type === 'text').map(c => c.text || '').join(' ');
251
+ if (t) return t;
252
+ }
253
+ }
254
+ }
255
+ return '';
256
+ }
257
+
137
258
  // Chat completion handler
138
259
  async function handleChatCompletion(req, res, body) {
139
260
  const requestedModel = body.model || 'tier-splus';
@@ -144,6 +265,12 @@ async function handleChatCompletion(req, res, body) {
144
265
  let effectiveModel = maybeUpgradeTier(requestedModel, complexity);
145
266
  let targetProviderKey = MODEL_MAP[effectiveModel] || MODEL_MAP[requestedModel] || 'zai';
146
267
  const isStreaming = body.stream === true;
268
+ // --- Контекстная телеметрия: один measure на запрос, record только в терминальной точке. ---
269
+ const measure = { ts: Date.now(), provider: targetProviderKey, cacheType: 'miss', status: 0, win: PROVIDERS[targetProviderKey]?.context_window || 0, upgraded: 0 };
270
+ const commit = (status) => {
271
+ measure.status = status;
272
+ contextStats.record(measure);
273
+ };
147
274
 
148
275
  // Vision detection: if the request contains images, route to a vision provider.
149
276
  // TWO-STAGE pipeline:
@@ -182,29 +309,41 @@ async function handleChatCompletion(req, res, body) {
182
309
  if (visionChain.length > 0) {
183
310
  logger.info('Vision pipeline: распознаю скриншот', { chain: visionChain.map(p => p.key).join(',') });
184
311
  let extracted = '';
185
- for (const visionProvider of visionChain) {
186
- try {
187
- const visionBody = {
188
- model: visionProvider.model,
189
- messages: [{
190
- role: 'user',
191
- content: [
192
- { type: 'text', text: 'Распознай и извлеки ВЕСЬ текст с изображения (ошибка, код, сообщение). Верни только содержимое, без комментариев. Если это код — верни код как есть.' },
193
- ...(Array.isArray(body.messages) ? body.messages.flatMap((m) => (Array.isArray(m.content) ? m.content.filter((c) => c && (c.type === 'image_url' || c.type === 'image' || c.type === 'input_image')).map((c) => {
194
- // Normalize any image part to the universal image_url format
195
- const url = c.image_url?.url || c.image?.url || c.image?.data || (c.image && typeof c.image === 'string' ? c.image : null) || c.url;
196
- return url ? { type: 'image_url', image_url: { url } } : null;
197
- }).filter(Boolean) : [])) : []),
198
- ],
199
- }],
200
- max_tokens: 2000,
201
- };
202
- const visionRes = await callProvider(visionProvider, visionBody);
203
- extracted = visionRes.data?.choices?.[0]?.message?.content || visionRes.data?.choices?.[0]?.message?.reasoning || '';
204
- if (extracted) { logger.info('Vision pipeline: распознал ' + visionProvider.key); break; }
205
- } catch (err) {
206
- logger.warn('Vision pipeline: ' + visionProvider.key + ' не сработал', { error: err.message.slice(0, 80) });
312
+ // PARALLEL vision attempt: fire all vision providers at once and take the
313
+ // first one that extracts text. Previously each was tried IN SEQUENCE,
314
+ // so a slow/failed first provider meant the pipeline waited provider
315
+ // after provider — the "two chats think forever" pattern for screenshots.
316
+ const visionAttempts = visionChain.map((visionProvider) => (async () => {
317
+ const visionBody = {
318
+ model: visionProvider.model,
319
+ messages: [{
320
+ role: 'user',
321
+ content: [
322
+ { type: 'text', text: 'Распознай и извлеки ВЕСЬ текст с изображения (ошибка, код, сообщение). Верни только содержимое, без комментариев. Если это код — верни код как есть.' },
323
+ ...(Array.isArray(body.messages) ? body.messages.flatMap((m) => (Array.isArray(m.content) ? m.content.filter((c) => c && (c.type === 'image_url' || c.type === 'image' || c.type === 'input_image')).map((c) => {
324
+ // Normalize any image part to the universal image_url format
325
+ const url = c.image_url?.url || c.image?.url || c.image?.data || (c.image && typeof c.image === 'string' ? c.image : null) || c.url;
326
+ return url ? { type: 'image_url', image_url: { url } } : null;
327
+ }).filter(Boolean) : [])) : []),
328
+ ],
329
+ }],
330
+ max_tokens: 2000,
331
+ };
332
+ const visionRes = await callProvider(visionProvider, visionBody);
333
+ const text = visionRes.data?.choices?.[0]?.message?.content || visionRes.data?.choices?.[0]?.message?.reasoning || '';
334
+ if (text) {
335
+ logger.info('Vision pipeline: распознал ' + visionProvider.key);
336
+ return text;
207
337
  }
338
+ throw new Error(visionProvider.key + ': пустой OCR');
339
+ })());
340
+ // First provider to yield text wins; hard failures (429/5xx) are skipped
341
+ // without blocking the others. If ALL fail, extracted stays '' and the
342
+ // request proceeds without vision context (as before).
343
+ try {
344
+ extracted = await Promise.any(visionAttempts);
345
+ } catch {
346
+ logger.warn('Vision pipeline: все вижн-провайдеры не сработали', { tried: visionChain.map(p => p.key) });
208
347
  }
209
348
  const cleaned = stripThink(extracted, true);
210
349
  logger.info('Vision pipeline: скриншот распознан', { chars: cleaned.length });
@@ -233,13 +372,94 @@ async function handleChatCompletion(req, res, body) {
233
372
  }
234
373
  }
235
374
 
236
- // Check cache (works for both streaming and non-streaming)
237
- const cached = cache.get(effectiveModel, body.messages, body.temperature);
238
- if (cached) {
239
- logger.request({ model: requestedModel, provider: 'cache', status: 200, cached: true });
240
- recordRecent({ model: requestedModel, provider: 'cache', status: 200, latency: 0, cached: true });
375
+ // Window-aware upgrade: если запрос не влезает в окно целевой модели (после
376
+ // vision-апгрейда target), компакция НЕ запускается — суммаризатор не должен
377
+ // сжимать контекст, который провайдер с большим окном возьмёт целиком.
378
+ const windowUpgraded = Array.isArray(body.messages) && body.messages.length > 0 &&
379
+ needsWindowUpgrade(PROVIDERS[targetProviderKey]?.context_window || 0, estimateTokens(body.messages));
380
+
381
+ // Compact overly large conversations so free models don't reject on context.
382
+ // Runs AFTER the vision pipeline (images already converted to text above).
383
+ // Skipped in window-upgrade mode — the big provider takes the raw context.
384
+ if (Array.isArray(body.messages)) measure.origTokens = estimateTokens(body.messages);
385
+ if (Array.isArray(body.messages) && body.messages.length > 0 && !windowUpgraded) {
386
+ body.messages = await prepareMessages(body.messages, { contextWindow: PROVIDERS[targetProviderKey]?.context_window || 0 });
387
+ }
388
+ // Контекстная телеметрия: токены после компакции + доля системного промпта.
389
+ if (Array.isArray(body.messages)) {
390
+ measure.sentTokens = estimateTokens(body.messages);
391
+ measure.est = measure.sentTokens;
392
+ measure.compacted = measure.sentTokens < measure.origTokens;
393
+ const sysChars = body.messages.filter(m => m && m.role === 'system').reduce((a, m) => a + (typeof m.content === 'string' ? m.content.length : 0), 0);
394
+ const allChars = body.messages.reduce((a, m) => a + (typeof (m && m.content) === 'string' ? m.content.length : 0), 0);
395
+ if (allChars > 0) measure.sysShare = sysChars / allChars;
396
+ }
397
+
398
+ // --- Long-term memory recall ---
399
+ // Подмешиваем релевантные факты из прошлых сессий (векторная память) как
400
+ // user-сообщение В НАЧАЛЕ диалога — после системных правил, до кэша. Так
401
+ // ключ кэша (normalize игнорирует system, но учитывает user) различает разные
402
+ // наборы фактов, и ответы не отравляются чужим кэшем.
403
+ if (memStore) {
404
+ try {
405
+ const userTexts = [];
406
+ const userMsgs = body.messages.filter(m => m && m.role === 'user');
407
+ for (const m of userMsgs.slice(-3)) {
408
+ if (typeof m.content === 'string') userTexts.push(m.content);
409
+ else if (Array.isArray(m.content)) userTexts.push(m.content.filter(c => c && c.type === 'text').map(c => c.text || '').join(' '));
410
+ }
411
+ const query = userTexts.join('\n').trim();
412
+ if (query.length > 20) {
413
+ const hits = memStore.recall(query, { topK: MEMORY_CONFIG.topK, minSimilarity: MEMORY_CONFIG.minSimilarity });
414
+ if (hits.length > 0) {
415
+ // Не подмешиваем факт, который уже покрыт резюме компактора или
416
+ // недавними сообщениями этой же сессии (защита от дублей).
417
+ const existing = body.messages
418
+ .filter(m => m && typeof m.content === 'string')
419
+ .map(m => m.content)
420
+ .concat(userTexts);
421
+ const fresh = hits.filter(f => !memStore.isCovered(f.text, existing));
422
+ if (fresh.length > 0) {
423
+ let memoryMsg = fresh.map(f => f.text).join('\n');
424
+ if (memoryMsg) {
425
+ const insertAt = body.messages.findIndex(m => m && m.role !== 'system');
426
+ const block = { role: 'user', content: '[Память: релевантные факты из прошлого]\n' + memoryMsg };
427
+ if (insertAt === -1) body.messages.unshift(block);
428
+ else body.messages.splice(insertAt, 0, block);
429
+ measure.memory = true;
430
+ logger.info('Memory recall', { facts: fresh.length, covered: hits.length - fresh.length });
431
+ }
432
+ }
433
+ }
434
+ }
435
+ } catch (err) {
436
+ logger.error('Memory recall error', { message: err.message });
437
+ }
438
+ }
439
+
440
+ // --- Методолог (инженерная дисциплина) ---
441
+ // Классифицируем задачу (coding/reasoning/search/chat) и вставляем короткий
442
+ // системный промпт-методолог после памяти, до кэша. System-сообщение
443
+ // игнорируется normalize, поэтому кэш-ключ не меняется. Методолог влияет
444
+ // только на реальные запросы к провайдеру (кэш-хиты его не видят).
445
+ let taskCategory = 'chat';
446
+ if (METHODOLOGY_CONFIG.enabled && Array.isArray(body.messages)) {
447
+ try {
448
+ taskCategory = classifyTask(body.messages);
449
+ const injected = injectMethodology(body.messages, taskCategory, METHODOLOGY_CONFIG);
450
+ if (injected !== body.messages) {
451
+ body.messages = injected;
452
+ measure.taskCategory = taskCategory;
453
+ }
454
+ } catch (err) {
455
+ logger.error('Methodology error', { message: err.message });
456
+ }
457
+ }
458
+
459
+ // Replays a previously cached completion (exact or semantic hit), preserving
460
+ // the stream/non-stream shape the client asked for.
461
+ function serveCached(res, cached, isStreaming) {
241
462
  if (isStreaming) {
242
- // Replay cached answer as an SSE stream
243
463
  res.writeHead(200, { 'Content-Type': 'text/event-stream', 'Cache-Control': 'no-cache', 'Connection': 'keep-alive' });
244
464
  const content = cached.choices?.[0]?.message?.content || '';
245
465
  res.write(`data: ${JSON.stringify({ id: 'chatcmpl-cached', object: 'chat.completion.chunk', created: Math.floor(Date.now() / 1000), model: cached.model, choices: [{ index: 0, delta: { role: 'assistant', content: '' }, finish_reason: null }] })}\n\n`);
@@ -247,13 +467,40 @@ async function handleChatCompletion(req, res, body) {
247
467
  res.write(`data: ${JSON.stringify({ id: 'chatcmpl-cached', object: 'chat.completion.chunk', created: Math.floor(Date.now() / 1000), model: cached.model, choices: [{ index: 0, delta: {}, finish_reason: 'stop' }] })}\n\n`);
248
468
  res.write('data: [DONE]\n\n');
249
469
  res.end();
250
- return;
470
+ } else {
471
+ res.writeHead(200, { 'Content-Type': 'application/json' });
472
+ res.end(JSON.stringify(cached));
251
473
  }
252
- res.writeHead(200, { 'Content-Type': 'application/json' });
253
- res.end(JSON.stringify(cached));
474
+ }
475
+
476
+ // Check cache (works for both streaming and non-streaming)
477
+ const cached = cache.get(effectiveModel, body.messages, body.temperature);
478
+ if (cached) {
479
+ logger.request({ model: requestedModel, provider: 'cache', status: 200, cached: true });
480
+ recordRecent({ model: requestedModel, provider: 'cache', status: 200, latency: 0, cached: true });
481
+ measure.cacheType = 'exact';
482
+ commit(200);
483
+ serveCached(res, cached, isStreaming);
254
484
  return;
255
485
  }
256
486
 
487
+ // Semantic cache: same intent, rephrased wording → replay without a new LLM call.
488
+ if (SEMCACHE_CONFIG.enabled) {
489
+ const semantic = cache.getSemantic(effectiveModel, body.messages, body.temperature, SEMCACHE_CONFIG.minSimilarity);
490
+ if (semantic) {
491
+ logger.request({ model: requestedModel, provider: 'semcache', status: 200, cached: true });
492
+ recordRecent({ model: requestedModel, provider: 'semcache', status: 200, latency: 0, cached: true });
493
+ measure.cacheType = 'semcache';
494
+ commit(200);
495
+ serveCached(res, semantic.value, isStreaming);
496
+ return;
497
+ }
498
+ }
499
+
500
+ // Запрос реально идёт к провайдеру (кэш промахнулся) — только теперь помечаем
501
+ // апгрейд: кэш-хит апгрейдом не считается (апстрим-вызова не было).
502
+ if (windowUpgraded) measure.upgraded = 1;
503
+
257
504
  // Weighted selection among healthy providers
258
505
  const today = new Date().toISOString().slice(0, 10);
259
506
  // 'ratelimited' providers are alive but temporarily limited — include them
@@ -264,10 +511,36 @@ async function handleChatCompletion(req, res, body) {
264
511
  .filter(([_, p]) => p.enabled && !isCircuitOpen(p.key) && p.vision !== true &&
265
512
  (getHealth()[p.key]?.status === 'up' || getHealth()[p.key]?.status === 'ratelimited'));
266
513
 
514
+ // Window-aware routing: estimate the request size and only consider providers
515
+ // whose context window can actually hold it. This stops large requests from
516
+ // burning time falling through lfm (65k) / groq (131k) providers that reject
517
+ // them — they go straight to nemotron-35/dots-3/minimax (1M/512k windows).
518
+ // Only applies when the request is big enough to matter, so small/typical
519
+ // requests keep the full fast pool.
520
+ const requestTokens = estimateTokens(body.messages);
521
+ const MIN_WINDOW = 50000; // below this we don't filter (typical requests)
522
+ let windowPool = healthyProviders;
523
+ let upgradeNoCapable = false; // апгрейд, но ни один здоровый провайдер не держит запрос
524
+ if (requestTokens > MIN_WINDOW || windowUpgraded) {
525
+ const capable = healthyProviders.filter(([_, p]) => {
526
+ const win = p.context_window || 0;
527
+ // Unknown/0 window providers are kept (heuristic) — better to try than drop.
528
+ return win === 0 || win >= requestTokens;
529
+ });
530
+ if (capable.length > 0) {
531
+ windowPool = capable;
532
+ } else if (windowUpgraded) {
533
+ // Best-effort: запрос больше окна любого провайдера (minimax 1M не держит).
534
+ // Всё равно переполним — выберем самое большое окно ниже, минимизируя
535
+ // потери контекста; target исключён; OVERFLOW поймает телеметрия.
536
+ upgradeNoCapable = true;
537
+ }
538
+ }
539
+
267
540
  // Prefer providers below 90% of their daily limit; only fall back to
268
541
  // near-exhausted ones if that leaves nothing (avoids avoidable 429s).
269
- let pool = healthyProviders;
270
- const underLimit = healthyProviders.filter(([_, p]) => {
542
+ let pool = windowPool;
543
+ const underLimit = windowPool.filter(([_, p]) => {
271
544
  const limit = p.dailyLimit || 1000;
272
545
  const used = (getStats().dailyUsage?.[p.key]?.[today]) || 0;
273
546
  return used < limit * 0.9;
@@ -276,32 +549,47 @@ async function handleChatCompletion(req, res, body) {
276
549
 
277
550
  let selected = [];
278
551
  if (pool.length > 0) {
279
- const scored = pool.map(([key, provider]) => {
280
- const h = getHealth()[key];
281
- let score = h.score || 50;
282
- const rawLat = h.latency || 0;
283
- const lat = rawLat > 0 ? Math.max(rawLat, 100) : 500;
284
- let weight = score / lat;
285
- if (h.status === 'ratelimited') weight *= 0.05;
286
- if (key === targetProviderKey) weight *= 1.15;
287
- const dailyLimit = provider.dailyLimit || 1000;
288
- const usedToday = getStats().providerUsage[key] || 0;
289
- if (usedToday >= dailyLimit * 0.9) weight *= 0.5;
290
- return { key, provider, weight };
291
- });
552
+ if (upgradeNoCapable) {
553
+ // Bandit здесь бессилен: все провайдеры в пуле переполнят окно (запрос
554
+ // больше самого большого). Берём самое большое окно — наименьшие потери.
555
+ selected = pool
556
+ .map(([k, p]) => ({ key: k, provider: p }))
557
+ .sort((a, b) => (b.provider.context_window || 0) - (a.provider.context_window || 0))
558
+ .slice(0, 1);
559
+ } else {
560
+ const scored = pool.map(([key, provider]) => {
561
+ const h = getHealth()[key];
562
+ let score = h.score || 50;
563
+ const rawLat = h.latency || 0;
564
+ const lat = rawLat > 0 ? Math.max(rawLat, 100) : 500;
565
+ let weight = score / lat;
566
+ if (h.status === 'ratelimited') weight *= 0.05;
567
+ if (key === targetProviderKey) weight *= 1.15;
568
+ // Методолог: категория задачи задаёт буст моделям подходящей категории,
569
+ // не исключая fallback. coding→coding, reasoning→reasoning, chat/search→general.
570
+ const cat = provider.category || 'general';
571
+ if (taskCategory === 'coding' && cat === 'coding') weight *= 1.5;
572
+ else if (taskCategory === 'reasoning' && cat === 'reasoning') weight *= 1.5;
573
+ else if ((taskCategory === 'chat' || taskCategory === 'search') && cat === 'general') weight *= 1.2;
574
+ const dailyLimit = provider.dailyLimit || 1000;
575
+ const usedToday = getStats().providerUsage[key] || 0;
576
+ if (usedToday >= dailyLimit * 0.9) weight *= 0.5;
577
+ return { key, provider, weight };
578
+ });
292
579
 
293
- // Bandit weight contract: bandit's pick() multiplies the Beta sample by
294
- // `weight`, so safety-штрафы (ratelimited ×0.05, target ×1.15) действуют и
295
- // при холодном старте. score/latency держит вес ~0.01-1.0; приоры bandit'а
296
- // (a,b ~1+) со временем начинают доминировать. Не добавляй нормализацию
297
- // здесь, пока измеренные веса не превысят ~5.
298
- // Thompson sampling: рисуем сэмпл Beta(a+1, b+1) для каждого, умножаем на
299
- // weight, выбираем максимум. Приоры из бакета сложности (bandit обучается).
300
- const priors = getBandit()[complexityBucket] || {};
301
- const bestKey = banditPick(scored, priors);
302
- const bestProvider = scored.find((p) => p.key === bestKey);
303
- if (bestProvider) selected = [bestProvider];
304
- else if (scored.length > 0) selected = [scored[0]];
580
+ // Bandit weight contract: bandit's pick() multiplies the Beta sample by
581
+ // `weight`, so safety-штрафы (ratelimited ×0.05, target ×1.15) действуют и
582
+ // при холодном старте. score/latency держит вес ~0.01-1.0; приоры bandit'а
583
+ // (a,b ~1+) со временем начинают доминировать. Не добавляй нормализацию
584
+ // здесь, пока измеренные веса не превысят ~5.
585
+ // Thompson sampling: рисуем сэмпл Beta(a+1, b+1) для каждого, умножаем на
586
+ // weight, выбираем максимум. Приоры из бакета сложности (bandit обучается).
587
+ const priors = getBandit()[complexityBucket] || {};
588
+ const bestKey = banditPick(scored, priors);
589
+ const bestProvider = scored.find((p) => p.key === bestKey);
590
+ if (bestProvider) selected = [bestProvider];
591
+ else if (scored.length > 0) selected = [scored[0]];
592
+ }
305
593
  }
306
594
 
307
595
  // Weighted-random picked ONE provider as the primary; append the rest of the
@@ -310,21 +598,37 @@ async function handleChatCompletion(req, res, body) {
310
598
  ? pool.map(([k, p]) => ({ key: k, provider: p })).filter(s => s.key !== selected[0].key)
311
599
  .sort((a, b) => (getHealth()[b.key]?.score || 0) - (getHealth()[a.key]?.score || 0))
312
600
  : [];
313
- const enabledProviders = selected.length > 0
601
+ let enabledProviders = selected.length > 0
314
602
  ? [selected[0]].concat(restOfPool).map(s => [s.key, s.provider])
315
603
  : Object.entries(PROVIDERS).filter(([_, p]) => p.enabled)
316
604
  .sort((a, b) => (getHealth()[b[0]]?.score || 50) - (getHealth()[a[0]]?.score || 50));
317
605
 
318
- // Ensure the requested model's mapped provider is at least IN the candidate
319
- // list (it may have been filtered out), but DON'T force it to the front —
320
- // the weighted selection above should pick the fastest/healthiest provider.
321
- if (MODEL_MAP[requestedModel] && PROVIDERS[targetProviderKey]) {
322
- if (!enabledProviders.some(([k]) => k === targetProviderKey)) {
323
- enabledProviders.unshift([targetProviderKey, PROVIDERS[targetProviderKey]]);
606
+ // Put the requested model's mapped provider FIRST. It's the only provider
607
+ // guaranteed to accept this tier/model — the rest are fallbacks (many reject
608
+ // tier-* requests with 400/422). Trying them before the target produced huge
609
+ // serial fallback chains (10+ sequential HTTP calls per request), which looked
610
+ // like the "model thinking forever". Correct mapping beats weighted guessing.
611
+ // Skipped in window-upgrade mode: the target doesn't fit the request anyway.
612
+ // If the target is rate-limited or near/over its daily limit, DON'T put it
613
+ // first — otherwise every request burns a doomed 429 attempt on it and the
614
+ // pool collapses into a rate-limit spiral. Skip straight to the healthy pool.
615
+ if (!windowUpgraded && MODEL_MAP[requestedModel] && PROVIDERS[targetProviderKey]) {
616
+ const tHealth = getHealth()[targetProviderKey];
617
+ const tLimit = (getStats().dailyUsage?.[targetProviderKey]?.[today]) || 0;
618
+ const tCap = PROVIDERS[targetProviderKey].dailyLimit || 0;
619
+ const targetBurned = tHealth?.status === 'ratelimited' || (tCap > 0 && tLimit >= tCap * 0.9);
620
+ if (!targetBurned) {
621
+ enabledProviders = [
622
+ [targetProviderKey, PROVIDERS[targetProviderKey]],
623
+ ...enabledProviders.filter(([k]) => k !== targetProviderKey),
624
+ ];
625
+ } else {
626
+ logger.info('Target-first skip', { key: targetProviderKey, status: tHealth?.status, used: tLimit, cap: tCap });
324
627
  }
325
628
  }
326
629
 
327
630
  if (enabledProviders.length === 0) {
631
+ commit(503);
328
632
  res.writeHead(503, { 'Content-Type': 'application/json' });
329
633
  res.end(JSON.stringify({ error: 'No providers available' }));
330
634
  return;
@@ -332,7 +636,13 @@ async function handleChatCompletion(req, res, body) {
332
636
 
333
637
  const errors = [];
334
638
 
335
- for (const [key, provider] of enabledProviders) {
639
+ // Cap serial fallback attempts. Trying provider after provider sequentially
640
+ // made a single tier-* request walk 10+ providers (each a real HTTP call),
641
+ // looking like the model "thinks forever". target-first above fixes the common
642
+ // case (right provider immediately); this cap bounds the worst case.
643
+ const fallbackProviders = enabledProviders.slice(0, 5);
644
+
645
+ for (const [key, provider] of fallbackProviders) {
336
646
  if (isCircuitOpen(key)) {
337
647
  errors.push(key + ': circuit breaker open');
338
648
  continue;
@@ -354,14 +664,13 @@ async function handleChatCompletion(req, res, body) {
354
664
  getHealth()[key].latency = Math.min(result.latency || 0, 60000);
355
665
  getHealth()[key].lastCheck = Date.now();
356
666
 
357
- // For non-stream, verify the response isn't empty BEFORE recording success.
358
667
  if (!isStreaming && result.data) {
359
- delete result.data.nvext;
360
- if (result.data.choices?.[0]) {
361
- fixReasoningMessage(result.data.choices[0].message);
362
- cleanMessage(result.data.choices[0].message);
363
- }
364
- if (isTooShort(result.data)) {
668
+ delete result.data.nvext;
669
+ if (result.data.choices?.[0]) {
670
+ fixReasoningMessage(result.data.choices[0].message);
671
+ cleanMessage(result.data.choices[0].message);
672
+ }
673
+ if (isTooShort(result.data, lastUserText(body.messages))) {
365
674
  // Пустой/мусорный ответ (провайдер-глитч) НЕ считается успехом — пробуем следующего.
366
675
  const msg = key + ': empty or too short response';
367
676
  errors.push(msg);
@@ -375,6 +684,7 @@ async function handleChatCompletion(req, res, body) {
375
684
  }
376
685
 
377
686
  if (isStreaming && result.stream) {
687
+ measure.provider = key;
378
688
  const chunks = [];
379
689
 
380
690
  // Очистка SSE-строки: убрать nvext, logprobs, think-блоки из дельт.
@@ -392,6 +702,7 @@ async function handleChatCompletion(req, res, body) {
392
702
  });
393
703
 
394
704
  // Сбор контент-токенов для кэша (strip think).
705
+ const streamUsage = {};
395
706
  const collect = (str) => {
396
707
  const lines = str.split('\n');
397
708
  for (const line of lines) {
@@ -401,6 +712,12 @@ async function handleChatCompletion(req, res, body) {
401
712
  const obj = JSON.parse(m[1]);
402
713
  const delta = obj.choices?.[0]?.delta?.content;
403
714
  if (typeof delta === 'string') chunks.push(stripThink(delta, false));
715
+ // OpenRouter / Nebius etc. put usage in a final chunk. Keep it.
716
+ if (obj.usage && (obj.usage.prompt_tokens || obj.usage.completion_tokens)) {
717
+ streamUsage.prompt_tokens = obj.usage.prompt_tokens;
718
+ streamUsage.completion_tokens = obj.usage.completion_tokens;
719
+ streamUsage.total_tokens = obj.usage.total_tokens;
720
+ }
404
721
  } catch {}
405
722
  }
406
723
  };
@@ -416,17 +733,29 @@ async function handleChatCompletion(req, res, body) {
416
733
  recordRecent({ model: requestedModel, provider: key, status: 200, latency: result.latency, cached: false });
417
734
  recordSelection(key, provider.model, requestedModel);
418
735
  const { Transform } = require('stream');
736
+ const reasonDec = new StringDecoder('utf8');
737
+ const reasonUsage = {};
419
738
  const cleaner = new Transform({
420
739
  transform(chunk, encoding, callback) {
421
- const str = chunk.toString();
740
+ const str = reasonDec.write(chunk);
741
+ collectReasonUsage(str, reasonUsage);
422
742
  collect(str);
423
743
  callback(null, cleanStr(str));
744
+ },
745
+ flush(callback) {
746
+ const tail = reasonDec.end();
747
+ if (tail) { collectReasonUsage(tail, reasonUsage); collect(tail); const tailStr = cleanStr(tail); if (tailStr) this.push(tailStr); }
748
+ callback();
424
749
  }
425
750
  });
426
751
  result.stream.on('end', () => {
427
752
  const full = chunks.join('');
428
753
  // Bandit учится по качеству: пустой/мусорный стрим = фейл.
429
754
  recordBandit(complexityBucket, key, full.trim().length >= MIN_ANSWER_LEN);
755
+ if (Object.keys(reasonUsage).length > 0) {
756
+ recordTokens(key, reasonUsage);
757
+ if (reasonUsage.prompt_tokens) measure.real = reasonUsage.prompt_tokens;
758
+ }
430
759
  if (full.trim().length >= MIN_ANSWER_LEN) {
431
760
  cache.set(effectiveModel, body.messages, body.temperature, {
432
761
  id: 'chatcmpl-cached',
@@ -437,11 +766,13 @@ async function handleChatCompletion(req, res, body) {
437
766
  usage: { prompt_tokens: 0, completion_tokens: 0, total_tokens: 0 },
438
767
  });
439
768
  }
769
+ commit(200);
440
770
  res.end();
441
771
  });
442
772
  result.stream.on('error', (err) => {
443
773
  logger.error('Stream error', { key, error: err.message });
444
- recordBandit(complexityBucket, key, false);
774
+ if (!isTransientLimit(err.statusCode)) recordBandit(complexityBucket, key, false);
775
+ commit(err.statusCode || 502);
445
776
  res.end();
446
777
  });
447
778
  result.stream.pipe(cleaner).pipe(res);
@@ -450,13 +781,24 @@ async function handleChatCompletion(req, res, body) {
450
781
 
451
782
  // Обычные модели: буферизуем до первого токена (макс 5 сек).
452
783
  // Заголовки не пишем сразу — если токена нет за 5 сек, fallback.
784
+ // StringDecoder держит частично пришедший multi-byte UTF-8 между чанками —
785
+ // иначе русский текст дробится на '' символ.
453
786
  const rawBuf = [];
787
+ const streamDec = new StringDecoder('utf8');
788
+ // Адаптивный таймаут первого токена: используем измеренную скорость
789
+ // провайдера (latency от реальных запросов). Быстрые провайдеры не ждут
790
+ // полные 5с перед fallback'ом, а медленные (но рабочие) не отбрасываются
791
+ // слишком рано. Диапазон 2.5-8с для защиты от обоих крайностей.
792
+ const knownLat = getHealth()[key]?.latency || 0;
793
+ const firstTokenWait = knownLat > 0
794
+ ? Math.max(2500, Math.min(8000, Math.round(knownLat * 2)))
795
+ : 5000;
454
796
  const firstToken = new Promise((resolve) => {
455
797
  let done = false;
456
- const timer = setTimeout(() => { if (!done) { done = true; resolve(false); } }, 5000);
798
+ const timer = setTimeout(() => { if (!done) { done = true; resolve(false); } }, firstTokenWait);
457
799
  const finish = (ok) => { if (!done) { done = true; clearTimeout(timer); resolve(ok); } };
458
800
  result.stream.on('data', (chunk) => {
459
- const str = chunk.toString();
801
+ const str = streamDec.write(chunk);
460
802
  rawBuf.push(str);
461
803
  collect(str);
462
804
  // Первый контент-токен: хотя бы одно непустое `"content":"..."` в чанке.
@@ -503,16 +845,27 @@ async function handleChatCompletion(req, res, body) {
503
845
 
504
846
  // Убираем наш 'data'-слушатель (он больше не нужен — данные уже
505
847
  // буферизованы в rawBuf и промыты). Дальше обрабатываем вручную.
848
+ // Продолжаем использовать ТОТ ЖЕ streamDec — иначе multi-byte UTF-8,
849
+ // разделённый границей буфера, превратится в '' .
506
850
  result.stream.removeAllListeners('data');
507
851
  result.stream.on('data', (chunk) => {
508
- const str = chunk.toString();
852
+ const str = streamDec.write(chunk);
509
853
  collect(str);
510
854
  res.write(cleanStr(str));
511
855
  });
512
856
  result.stream.on('end', () => {
857
+ const tail = streamDec.end();
858
+ if (tail) {
859
+ collect(tail);
860
+ res.write(cleanStr(tail));
861
+ }
513
862
  const full = chunks.join('');
514
863
  // Bandit учится по качеству: обрыв/мусорный стрим = фейл.
515
864
  recordBandit(complexityBucket, key, full.trim().length >= MIN_ANSWER_LEN);
865
+ if (Object.keys(streamUsage).length > 0) {
866
+ recordTokens(key, streamUsage);
867
+ if (streamUsage.prompt_tokens) measure.real = streamUsage.prompt_tokens;
868
+ }
516
869
  if (full.trim().length >= MIN_ANSWER_LEN) {
517
870
  cache.set(effectiveModel, body.messages, body.temperature, {
518
871
  id: 'chatcmpl-cached',
@@ -523,11 +876,13 @@ async function handleChatCompletion(req, res, body) {
523
876
  usage: { prompt_tokens: 0, completion_tokens: 0, total_tokens: 0 },
524
877
  });
525
878
  }
879
+ commit(200);
526
880
  res.end();
527
881
  });
528
882
  result.stream.on('error', (err) => {
529
883
  logger.error('Stream error', { key, error: err.message });
530
- recordBandit(complexityBucket, key, false);
884
+ if (!isTransientLimit(err.statusCode)) recordBandit(complexityBucket, key, false);
885
+ commit(err.statusCode || 502);
531
886
  res.end();
532
887
  });
533
888
  return;
@@ -541,6 +896,10 @@ async function handleChatCompletion(req, res, body) {
541
896
  logger.request({ model: requestedModel, provider: key, status: 200, latency: result.latency, stream: isStreaming });
542
897
  recordRecent({ model: requestedModel, provider: key, status: 200, latency: result.latency, cached: false });
543
898
  recordSelection(key, provider.model, requestedModel);
899
+ measure.provider = key;
900
+ measure.real = (result.data.usage && result.data.usage.prompt_tokens) ? result.data.usage.prompt_tokens : measure.sentTokens || 0;
901
+ measure.win = PROVIDERS[key]?.context_window || 0;
902
+ commit(200);
544
903
  cache.set(effectiveModel, body.messages, body.temperature, result.data);
545
904
  recordTokens(key, result.usage);
546
905
  res.writeHead(200, { 'Content-Type': 'application/json' });
@@ -551,7 +910,8 @@ async function handleChatCompletion(req, res, body) {
551
910
  const statusCode = err.statusCode || 502;
552
911
  errors.push(err.message);
553
912
  recordRequest(key, false, err.message);
554
- recordBandit(complexityBucket, key, false);
913
+ // Временные лимиты (429/403/402) — не наказываем провайдера в bandit.
914
+ if (!isTransientLimit(statusCode)) recordBandit(complexityBucket, key, false);
555
915
  recordRecent({ model: requestedModel, provider: key, status: statusCode, latency: 0, cached: false });
556
916
  initHealth(key);
557
917
  // Do NOT flip provider to 'error' on a single failed request — transient
@@ -568,13 +928,24 @@ async function handleChatCompletion(req, res, body) {
568
928
  getHealth()[key].reason = 'не отвечает';
569
929
  }
570
930
  getHealth()[key].score = Math.max(0, (getHealth()[key].score || 50) - (statusCode === 429 ? 5 : 10));
571
- recordFailure(key, statusCode);
572
- // 404 = model not available for this account — disable permanently
931
+ recordFailure(key, statusCode, statusCode === 404 && err.providerSide ? { providerSide: true } : undefined);
932
+ // 404 = model not available. Two distinct flavors:
933
+ // - plain 404 («does not exist») → disable permanently (model gone from OpenRouter).
934
+ // - provider-side 404 (Nvidia quota/upstream failing) → NOT permanent — the model
935
+ // may recover. Mark it down hard so routing prefers others, and let the periodic
936
+ // health-check re-enable it when it comes back.
573
937
  if (statusCode === 404) {
574
- provider.enabled = false;
575
- getHealth()[key].status = 'disabled';
576
- getHealth()[key].reason = 'отключён автоматически (404)';
577
- logger.warn('Provider auto-disabled (404)', { key, model: provider.model });
938
+ if (err.providerSide) {
939
+ getHealth()[key].status = 'error';
940
+ getHealth()[key].reason = 'провайдер временно недоступен (404)';
941
+ getHealth()[key].score = Math.max(0, (getHealth()[key].score || 50) - 25);
942
+ logger.warn('Provider temporarily down (provider-side 404)', { key, model: provider.model });
943
+ } else {
944
+ provider.enabled = false;
945
+ getHealth()[key].status = 'disabled';
946
+ getHealth()[key].reason = 'отключён автоматически (404)';
947
+ logger.warn('Provider auto-disabled (404)', { key, model: provider.model });
948
+ }
578
949
  }
579
950
  }
580
951
  }
@@ -584,7 +955,7 @@ async function handleChatCompletion(req, res, body) {
584
955
  const allSoft = errors.length > 0 && errors.every(e => !/429|401|403|404/.test(e));
585
956
  if (allSoft && enabledProviders.length > 1) {
586
957
  await new Promise(r => setTimeout(r, 1500));
587
- for (const [key, provider] of enabledProviders) {
958
+ for (const [key, provider] of fallbackProviders) {
588
959
  if (isCircuitOpen(key)) continue;
589
960
  try {
590
961
  const release = await acquire(key);
@@ -600,7 +971,7 @@ async function handleChatCompletion(req, res, body) {
600
971
  fixReasoningMessage(result.data.choices[0].message);
601
972
  cleanMessage(result.data.choices[0].message);
602
973
  }
603
- if (isTooShort(result.data)) {
974
+ if (isTooShort(result.data, lastUserText(body.messages))) {
604
975
  recordFailure(key, 0);
605
976
  recordRequest(key, false, key + ': empty or too short response (retry)');
606
977
  recordBandit(complexityBucket, key, false);
@@ -612,6 +983,10 @@ async function handleChatCompletion(req, res, body) {
612
983
  recordBandit(complexityBucket, key, true);
613
984
  recordRecent({ model: requestedModel, provider: key, status: 200, latency: result.latency, cached: false });
614
985
  recordSelection(key, provider.model, requestedModel);
986
+ measure.provider = key;
987
+ measure.real = (result.data.usage && result.data.usage.prompt_tokens) ? result.data.usage.prompt_tokens : measure.sentTokens || 0;
988
+ measure.win = PROVIDERS[key]?.context_window || 0;
989
+ commit(200);
615
990
  cache.set(effectiveModel, body.messages, body.temperature, result.data);
616
991
  recordTokens(key, result.usage);
617
992
  res.writeHead(200, { 'Content-Type': 'application/json' });
@@ -619,6 +994,7 @@ async function handleChatCompletion(req, res, body) {
619
994
  return;
620
995
  }
621
996
  if (body.stream && result.stream) {
997
+ measure.provider = key;
622
998
  recordSuccess(key);
623
999
  recordRequest(key, true);
624
1000
  recordRecent({ model: requestedModel, provider: key, status: 200, latency: result.latency, cached: false });
@@ -649,12 +1025,15 @@ async function handleChatCompletion(req, res, body) {
649
1025
  return 'data: ' + JSON.stringify(obj);
650
1026
  } catch { return match; }
651
1027
  });
1028
+ const retryDec = new StringDecoder('utf8');
652
1029
  result.stream.on('data', (chunk) => {
653
- const str = chunk.toString();
1030
+ const str = retryDec.write(chunk);
654
1031
  collectRetry(str);
655
1032
  res.write(cleanRetry(str));
656
1033
  });
657
1034
  result.stream.on('end', () => {
1035
+ const tail = retryDec.end();
1036
+ if (tail) { collectRetry(tail); res.write(cleanRetry(tail)); }
658
1037
  const full = chunks.join('');
659
1038
  // Bandit учится по качеству в ретрае тоже.
660
1039
  recordBandit(complexityBucket, key, full.trim().length >= MIN_ANSWER_LEN);
@@ -668,23 +1047,26 @@ async function handleChatCompletion(req, res, body) {
668
1047
  usage: { prompt_tokens: 0, completion_tokens: 0, total_tokens: 0 },
669
1048
  });
670
1049
  }
1050
+ commit(200);
671
1051
  res.end();
672
1052
  });
673
1053
  result.stream.on('error', (err) => {
674
1054
  logger.error('Stream error (retry)', { key, error: err.message });
675
- recordBandit(complexityBucket, key, false);
1055
+ if (!isTransientLimit(err.statusCode)) recordBandit(complexityBucket, key, false);
1056
+ commit(err.statusCode || 502);
676
1057
  res.end();
677
1058
  });
678
1059
  return;
679
1060
  }
680
1061
  } catch (err2) {
681
1062
  recordRequest(key, false, err2.message);
682
- recordBandit(complexityBucket, key, false);
1063
+ if (!isTransientLimit(err2.statusCode)) recordBandit(complexityBucket, key, false);
683
1064
  recordFailure(key, err2.statusCode);
684
1065
  }
685
1066
  }
686
1067
  }
687
1068
 
1069
+ commit(502);
688
1070
  res.writeHead(502, { 'Content-Type': 'application/json' });
689
1071
  res.end(JSON.stringify({ error: { message: 'All providers failed', type: 'api_error', code: 'all_providers_failed', details: errors } }));
690
1072
  }
@@ -714,12 +1096,145 @@ const server = http.createServer(async (req, res) => {
714
1096
  return;
715
1097
  }
716
1098
 
1099
+ if (parsedUrl.pathname === '/v1/reload' && req.method === 'POST') {
1100
+ // Hot-reload providers.json + config.json without restarting the server.
1101
+ // Auth-protected like the other admin endpoints.
1102
+ if (AUTH_KEY) {
1103
+ const apiKey = (req.headers.authorization || '').replace('Bearer ', '').trim();
1104
+ const keyFromQuery = parsedUrl.searchParams.get('key');
1105
+ if (apiKey !== AUTH_KEY && keyFromQuery !== AUTH_KEY) {
1106
+ res.writeHead(401, { 'Content-Type': 'application/json' });
1107
+ res.end(JSON.stringify({ error: { message: 'Invalid API key' } }));
1108
+ return;
1109
+ }
1110
+ }
1111
+ try {
1112
+ const result = reloadProviders();
1113
+ res.writeHead(200, { 'Content-Type': 'application/json' });
1114
+ res.end(JSON.stringify({ ok: true, ...result }));
1115
+ } catch (err) {
1116
+ res.writeHead(500, { 'Content-Type': 'application/json' });
1117
+ res.end(JSON.stringify({ error: { message: 'Reload failed: ' + err.message } }));
1118
+ }
1119
+ return;
1120
+ }
1121
+
1122
+ if (parsedUrl.pathname === '/v1/models-db' && req.method === 'GET') {
1123
+ // Структурированная база моделей: паспорта + статистика + топ по скору.
1124
+ if (AUTH_KEY) {
1125
+ const apiKey = (req.headers.authorization || '').replace('Bearer ', '').trim();
1126
+ const keyFromQuery = parsedUrl.searchParams.get('key');
1127
+ if (apiKey !== AUTH_KEY && keyFromQuery !== AUTH_KEY) {
1128
+ res.writeHead(401, { 'Content-Type': 'application/json' });
1129
+ res.end(JSON.stringify({ error: { message: 'Invalid API key' } }));
1130
+ return;
1131
+ }
1132
+ }
1133
+ try {
1134
+ const models = modelManager.db.all()
1135
+ .map(m => ({
1136
+ key: m.key, model: m.model, source: m.source, category: m.category,
1137
+ contextWindow: m.contextWindow, dailyLimit: m.dailyLimit,
1138
+ score: m.score || 0, status: m.status,
1139
+ lastCheckedAt: m.lastCheckedAt || null, lastOkAt: m.lastOkAt || null,
1140
+ }))
1141
+ .sort((a, b) => b.score - a.score);
1142
+ res.writeHead(200, { 'Content-Type': 'application/json' });
1143
+ res.end(JSON.stringify({
1144
+ stats: modelManager.db.stats(),
1145
+ top: models.slice(0, 10),
1146
+ models,
1147
+ manager: { enabled: MODEL_MANAGER_CONFIG.enabled !== false, intervalHours: modelManager.config.intervalHours, running: modelManager._running },
1148
+ }));
1149
+ } catch (err) {
1150
+ res.writeHead(500, { 'Content-Type': 'application/json' });
1151
+ res.end(JSON.stringify({ error: { message: err.message } }));
1152
+ }
1153
+ return;
1154
+ }
1155
+
1156
+ // Парсим /v1/models/{key}/toggle и /v1/models/{key}/test
1157
+ const modelActionMatch = parsedUrl.pathname.match(/^\/v1\/models\/([^/]+)\/(toggle|test)$/);
1158
+ if (modelActionMatch && req.method === 'POST') {
1159
+ if (AUTH_KEY) {
1160
+ const apiKey = (req.headers.authorization || '').replace('Bearer ', '').trim();
1161
+ const keyFromQuery = parsedUrl.searchParams.get('key');
1162
+ if (apiKey !== AUTH_KEY && keyFromQuery !== AUTH_KEY) {
1163
+ res.writeHead(401, { 'Content-Type': 'application/json' });
1164
+ res.end(JSON.stringify({ error: { message: 'Invalid API key' } }));
1165
+ return;
1166
+ }
1167
+ }
1168
+ const modelKey = decodeURIComponent(modelActionMatch[1]);
1169
+ const action = modelActionMatch[2];
1170
+
1171
+ if (action === 'test') {
1172
+ // Живой "hi"-тест: работает ли модель с нашим ключом. Без изменения каталога.
1173
+ const dbEntry = modelManager.db.get(modelKey);
1174
+ const prov = PROVIDERS[modelKey];
1175
+ const endpoint = (dbEntry && dbEntry.endpoint) || (prov && prov.endpoint);
1176
+ const model = (dbEntry && dbEntry.model) || (prov && prov.model);
1177
+ if (!endpoint || !model) {
1178
+ res.writeHead(404, { 'Content-Type': 'application/json' });
1179
+ res.end(JSON.stringify({ ok: false, error: 'Модель ' + modelKey + ' не найдена' }));
1180
+ return;
1181
+ }
1182
+ const source = (dbEntry && dbEntry.source) || 'unknown';
1183
+ const apiKey = modelManager.apiKeyFor(source);
1184
+ try {
1185
+ const t0 = Date.now();
1186
+ const r = await fetch(endpoint, {
1187
+ method: 'POST',
1188
+ headers: { 'Content-Type': 'application/json', ...(apiKey ? { Authorization: 'Bearer ' + apiKey } : {}) },
1189
+ body: JSON.stringify({ model, messages: [{ role: 'user', content: 'hi' }], max_tokens: 5 }),
1190
+ signal: AbortSignal.timeout(20000),
1191
+ });
1192
+ const latencyMs = Date.now() - t0;
1193
+ // Обновляем базу результатом проверки (но не «активируем» принудительно).
1194
+ modelManager.db.markChecked(modelKey, { ok: r.ok, status: r.status, latencyMs }, { now: Date.now() });
1195
+ modelManager.db.save();
1196
+ res.writeHead(200, { 'Content-Type': 'application/json' });
1197
+ res.end(JSON.stringify({ ok: r.ok, status: r.status, latencyMs, model: { key: modelKey, status: modelManager.db.get(modelKey).status } }));
1198
+ } catch (err) {
1199
+ res.writeHead(200, { 'Content-Type': 'application/json' });
1200
+ res.end(JSON.stringify({ ok: false, status: 0, error: err.message }));
1201
+ }
1202
+ return;
1203
+ }
1204
+
1205
+ // action === 'toggle': вкл/выкл провайдера в config.json (только этот ключ) + hot-reload.
1206
+ try {
1207
+ const userCfg = JSON.parse(fs.readFileSync(CONFIG_PATH, 'utf8'));
1208
+ if (!userCfg.providers) userCfg.providers = {};
1209
+ if (!userCfg.providers[modelKey]) userCfg.providers[modelKey] = {};
1210
+ // Flip: было включено → выключить, было выключено → включить.
1211
+ const currentlyEnabled = !(userCfg.providers[modelKey].enabled === false);
1212
+ const newEnabled = !currentlyEnabled;
1213
+ userCfg.providers[modelKey].enabled = newEnabled;
1214
+ fs.writeFileSync(CONFIG_PATH, JSON.stringify(userCfg, null, 2));
1215
+ reloadProviders();
1216
+ // Статус в базе: включили → untested (проверится в след. цикле), выключили → user-disabled.
1217
+ const entry = modelManager.db.get(modelKey);
1218
+ if (entry) {
1219
+ modelManager.db.setStatus(modelKey, newEnabled ? 'untested' : 'user-disabled');
1220
+ modelManager.db.save();
1221
+ }
1222
+ res.writeHead(200, { 'Content-Type': 'application/json' });
1223
+ res.end(JSON.stringify({ ok: true, key: modelKey, enabled: newEnabled, model: entry ? { key: modelKey, status: modelManager.db.get(modelKey).status } : null }));
1224
+ } catch (err) {
1225
+ res.writeHead(500, { 'Content-Type': 'application/json' });
1226
+ res.end(JSON.stringify({ error: { message: 'Toggle failed: ' + err.message } }));
1227
+ }
1228
+ return;
1229
+ }
1230
+
717
1231
  if (parsedUrl.pathname === '/health') {
718
1232
  const h = getHealth();
719
1233
  const upCount = Object.values(h).filter(v => v.status === 'up').length;
720
1234
  const totalCount = Object.entries(PROVIDERS).filter(([_, p]) => p.enabled).length;
1235
+ const contextSummary = (() => { try { return contextStats.summary(); } catch { return null; } })();
721
1236
  res.writeHead(upCount > 0 ? 200 : 503, { 'Content-Type': 'application/json' });
722
- res.end(JSON.stringify({ status: upCount > 0 ? 'ok' : 'degraded', providers: { up: upCount, total: totalCount } }));
1237
+ res.end(JSON.stringify({ status: upCount > 0 ? 'ok' : 'degraded', providers: { up: upCount, total: totalCount }, context: contextSummary }));
723
1238
  return;
724
1239
  }
725
1240
 
@@ -737,6 +1252,7 @@ const server = http.createServer(async (req, res) => {
737
1252
  res.end(JSON.stringify({
738
1253
  total_requests: s.totalRequests, successful_requests: s.successfulRequests, failed_requests: s.failedRequests,
739
1254
  provider_usage: s.providerUsage, token_usage: s.tokenUsage, errors: s.errors, uptime_seconds: Math.floor((Date.now() - s.startTime) / 1000),
1255
+ savings: (() => { try { return aggregateSavings(s.tokenUsage || {}); } catch { return null; } })(),
740
1256
  health: Object.fromEntries(Object.entries(getHealth()).map(([k, v]) => {
741
1257
  const limit = limits[k];
742
1258
  const err = s.errors[k] || 0;
@@ -755,6 +1271,7 @@ const server = http.createServer(async (req, res) => {
755
1271
  pool: poolStats(),
756
1272
  last_selection: getLastSelection(),
757
1273
  bandit: getBandit(),
1274
+ context_summary: (() => { try { return contextStats.summary(); } catch { return null; } })(),
758
1275
  }));
759
1276
  return;
760
1277
  }
@@ -835,6 +1352,25 @@ const server = http.createServer(async (req, res) => {
835
1352
  return;
836
1353
  }
837
1354
 
1355
+ // POST /v1/cache/clear — drop the in-memory semantic cache without a restart.
1356
+ // Handy when you tweaked providers/models and don't want stale answers served.
1357
+ if (parsedUrl.pathname === '/v1/cache/clear' && req.method === 'POST') {
1358
+ if (AUTH_KEY) {
1359
+ const apiKey = (req.headers.authorization || '').replace('Bearer ', '').trim();
1360
+ const keyFromQuery = parsedUrl.searchParams.get('key');
1361
+ if (apiKey !== AUTH_KEY && keyFromQuery !== AUTH_KEY) {
1362
+ res.writeHead(401, { 'Content-Type': 'application/json' });
1363
+ res.end(JSON.stringify({ error: { message: 'Invalid API key' } }));
1364
+ return;
1365
+ }
1366
+ }
1367
+ const before = cache.stats().size || 0;
1368
+ cache.clear();
1369
+ res.writeHead(200, { 'Content-Type': 'application/json' });
1370
+ res.end(JSON.stringify({ ok: true, cleared: before }));
1371
+ return;
1372
+ }
1373
+
838
1374
  // POST /v1/shorts — generate a vertical short video via the tools generator.
839
1375
  // Body: { prompt, duration?, format? ("9:16"/"16:9"/"1:1"), steps? }
840
1376
  if (parsedUrl.pathname === '/v1/shorts' && req.method === 'POST') {
@@ -888,6 +1424,87 @@ const server = http.createServer(async (req, res) => {
888
1424
  return;
889
1425
  }
890
1426
 
1427
+ // --- Setup Dashboard API ---
1428
+ if (parsedUrl.pathname === '/v1/setup/keys' && req.method === 'GET') {
1429
+ const { keys } = readKeys();
1430
+ // Mask keys for display; inputs stay EMPTY so we never send masked values back.
1431
+ const masked = {};
1432
+ const empty = {};
1433
+ for (const [k, v] of Object.entries(keys)) {
1434
+ empty[k] = '';
1435
+ if (!v) { masked[k] = ''; continue; }
1436
+ if (v.length <= 10) { masked[k] = v.slice(0, 2) + '***' + v.slice(-2); continue; }
1437
+ masked[k] = v.slice(0, 4) + '***' + v.slice(-4);
1438
+ }
1439
+ res.writeHead(200, { 'Content-Type': 'application/json' });
1440
+ res.end(JSON.stringify({ groups: KEY_GROUPS, keys: empty, masked }));
1441
+ return;
1442
+ }
1443
+
1444
+ if (parsedUrl.pathname === '/v1/setup/keys' && req.method === 'POST') {
1445
+ if (AUTH_KEY) {
1446
+ const apiKey = (req.headers.authorization || '').replace('Bearer ', '').trim();
1447
+ const keyFromQuery = parsedUrl.searchParams.get('key');
1448
+ if (apiKey !== AUTH_KEY && keyFromQuery !== AUTH_KEY) {
1449
+ res.writeHead(401, { 'Content-Type': 'application/json' });
1450
+ res.end(JSON.stringify({ error: { message: 'Invalid API key' } }));
1451
+ return;
1452
+ }
1453
+ }
1454
+ let body = '';
1455
+ req.on('data', (c) => body += c);
1456
+ req.on('end', () => {
1457
+ try {
1458
+ const newKeys = JSON.parse(body);
1459
+ // Only accept known env vars
1460
+ const filtered = {};
1461
+ for (const k of Object.keys(KEY_GROUPS)) {
1462
+ if (typeof newKeys[k] === 'string') filtered[k] = newKeys[k];
1463
+ }
1464
+ const result = saveKeys(filtered);
1465
+ res.writeHead(200, { 'Content-Type': 'application/json' });
1466
+ res.end(JSON.stringify({ ok: true, ...result }));
1467
+ } catch (e) {
1468
+ res.writeHead(400, { 'Content-Type': 'application/json' });
1469
+ res.end(JSON.stringify({ error: { message: 'Invalid request: ' + e.message } }));
1470
+ }
1471
+ });
1472
+ return;
1473
+ }
1474
+
1475
+ if (parsedUrl.pathname === '/v1/setup/validate') {
1476
+ if (AUTH_KEY) {
1477
+ const apiKey = (req.headers.authorization || '').replace('Bearer ', '').trim();
1478
+ const keyFromQuery = parsedUrl.searchParams.get('key');
1479
+ if (apiKey !== AUTH_KEY && keyFromQuery !== AUTH_KEY) {
1480
+ res.writeHead(401, { 'Content-Type': 'application/json' });
1481
+ res.end(JSON.stringify({ error: { message: 'Invalid API key' } }));
1482
+ return;
1483
+ }
1484
+ }
1485
+ const envVar = parsedUrl.searchParams.get('envVar');
1486
+ let testKey = parsedUrl.searchParams.get('apiKey');
1487
+ if (!envVar) {
1488
+ res.writeHead(400, { 'Content-Type': 'application/json' });
1489
+ res.end(JSON.stringify({ error: { message: 'envVar required' } }));
1490
+ return;
1491
+ }
1492
+ // If apiKey param is empty/absent, validate the real stored key from .env.
1493
+ if (!testKey) {
1494
+ testKey = getStoredKey(envVar);
1495
+ }
1496
+ if (!testKey) {
1497
+ res.writeHead(200, { 'Content-Type': 'application/json' });
1498
+ res.end(JSON.stringify({ valid: false, error: 'Нет сохранённого ключа' }));
1499
+ return;
1500
+ }
1501
+ validateKey(envVar, testKey).then((result) => {
1502
+ res.writeHead(200, { 'Content-Type': 'application/json' });
1503
+ res.end(JSON.stringify(result));
1504
+ });
1505
+ return;
1506
+ }
1507
+
891
1508
  res.writeHead(404, { 'Content-Type': 'application/json' });
892
1509
  res.end(JSON.stringify({ error: 'Not found' }));
893
1510
  });
@@ -895,7 +1512,19 @@ const server = http.createServer(async (req, res) => {
895
1512
  server.listen(PORT, process.env.HOST || '127.0.0.1', () => {
896
1513
  logger.info('Freegate started', { port: PORT });
897
1514
  console.log('Dashboard: http://localhost:' + PORT + '/');
1515
+ // Самообновляющаяся база моделей: первый цикл через 2 мин, далее по интервалу.
1516
+ if (MODEL_MANAGER_CONFIG.enabled !== false) {
1517
+ modelManager.start();
1518
+ logger.info('ModelManager started', { intervalHours: modelManager.config.intervalHours });
1519
+ }
898
1520
  });
899
1521
 
900
- process.on('SIGINT', () => { require('./lib/health').saveState(); cache.persist(); server.close(() => process.exit(0)); });
901
- process.on('SIGTERM', () => { require('./lib/health').saveState(); cache.persist(); server.close(() => process.exit(0)); });
1522
+ const _shutdown = () => {
1523
+ if (memStore) { memStore.stopTimer(); memStore.save(); }
1524
+ require('./lib/health').saveState();
1525
+ cache.persist();
1526
+ try { modelManager.stop(); } catch {}
1527
+ server.close(() => process.exit(0));
1528
+ };
1529
+ process.on('SIGINT', _shutdown);
1530
+ process.on('SIGTERM', _shutdown);