freegate 0.6.15 → 0.6.17
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +72 -7
- package/README.ru.md +93 -7
- package/assets/dashboard-models.png +0 -0
- package/assets/dashboard.png +0 -0
- package/bin/freegate.js +32 -5
- package/config.example.json +13 -0
- package/lib/cache.js +54 -5
- package/lib/clean.js +7 -2
- package/lib/compactor.js +220 -0
- package/lib/contextstats.js +224 -0
- package/lib/dashboard.js +791 -127
- package/lib/economics.js +43 -0
- package/lib/health.js +56 -9
- package/lib/logger.js +1 -1
- package/lib/memory-store.js +68 -0
- package/lib/memory.js +167 -0
- package/lib/methodology.js +70 -0
- package/lib/modeldb.js +125 -0
- package/lib/modelmanager.js +335 -0
- package/lib/modelscan.js +190 -0
- package/lib/normalize.js +14 -4
- package/lib/providers.js +68 -5
- package/lib/routing.js +14 -2
- package/lib/semcache.js +31 -0
- package/lib/setup.js +309 -0
- package/lib/taskclassify.js +59 -0
- package/package.json +1 -1
- package/providers.json +113 -137
- package/server.js +750 -98
package/server.js
CHANGED
|
@@ -3,18 +3,26 @@ const http = require('http');
|
|
|
3
3
|
const fs = require('fs');
|
|
4
4
|
const path = require('path');
|
|
5
5
|
const { LRUCache } = require('./lib/cache');
|
|
6
|
-
const { PROVIDERS, MODEL_MAP, callProvider } = require('./lib/providers');
|
|
7
|
-
const { loadState, initHealth, isCircuitOpen, recordSuccess, recordFailure, recordRequest, recordTokens, getHealth, getStats, recordRecent, recordRpm, getRecent, getRpm, recordSelection, getLastSelection, getBandit, recordBandit } = require('./lib/health');
|
|
6
|
+
const { PROVIDERS, MODEL_MAP, callProvider, reloadProviders } = require('./lib/providers');
|
|
7
|
+
const { loadState, initHealth, isCircuitOpen, recordSuccess, recordFailure, recordRequest, recordTokens, getHealth, getStats, getReliability, recordRecent, recordRpm, getRecent, getRpm, recordSelection, getLastSelection, getBandit, recordBandit, warmBanditPriors, getContextStats } = require('./lib/health');
|
|
8
8
|
const { checkRateLimit } = require('./lib/rateLimit');
|
|
9
9
|
const { handleDashboard } = require('./lib/dashboard');
|
|
10
10
|
const { acquire, stats: poolStats } = require('./lib/pool');
|
|
11
|
+
const { aggregateSavings } = require('./lib/economics');
|
|
11
12
|
const { stripThink, cleanDelta, cleanMessage, fixReasoningMessage, isTooShort, MIN_ANSWER_LEN } = require('./lib/clean');
|
|
12
|
-
const { classifyComplexity, maybeUpgradeTier, classifyVisionComplexity } = require('./lib/routing');
|
|
13
|
+
const { classifyComplexity, maybeUpgradeTier, classifyVisionComplexity, needsWindowUpgrade } = require('./lib/routing');
|
|
13
14
|
const { bucket, pick: banditPick, isTransientLimit } = require('./lib/bandit');
|
|
15
|
+
const { StringDecoder } = require('string_decoder');
|
|
14
16
|
const logger = require('./lib/logger');
|
|
17
|
+
const { KEY_GROUPS, readKeys, saveKeys, validateKey, getStoredKey } = require('./lib/setup');
|
|
18
|
+
const { prepareMessages, estimateTokens, setMemory: compactorSetMemory } = require('./lib/compactor');
|
|
19
|
+
const { classify: classifyTask } = require('./lib/taskclassify');
|
|
20
|
+
const { injectMethodology, enabledByDefault: methodEnabledByDefault } = require('./lib/methodology');
|
|
21
|
+
const { create: createMemoryStore } = require('./lib/memory-store');
|
|
15
22
|
|
|
16
23
|
// Load persisted state
|
|
17
24
|
loadState();
|
|
25
|
+
const contextStats = getContextStats();
|
|
18
26
|
|
|
19
27
|
// Drop stale health entries for providers that no longer exist (e.g. auto-disabled)
|
|
20
28
|
const activeKeys = new Set(Object.keys(PROVIDERS));
|
|
@@ -26,6 +34,14 @@ if (stale.length > 0) {
|
|
|
26
34
|
logger.info('Cleaned stale health entries', { removed: stale });
|
|
27
35
|
}
|
|
28
36
|
|
|
37
|
+
// Warm bandit priors for enabled providers with no/low history so brand-new
|
|
38
|
+
// models (e.g. or-minimax-m3-free) are explored promptly instead of ignored.
|
|
39
|
+
const warmed = warmBanditPriors(
|
|
40
|
+
Object.keys(PROVIDERS).filter(k => PROVIDERS[k].enabled !== false),
|
|
41
|
+
['low', 'med', 'high']
|
|
42
|
+
);
|
|
43
|
+
if (warmed > 0) logger.info('Warmed bandit priors', { warmed });
|
|
44
|
+
|
|
29
45
|
const cache = new LRUCache(500, 3600000, false, true); // 4th arg: semantic normalize ON
|
|
30
46
|
require('./lib/cache')._activeCache = cache;
|
|
31
47
|
|
|
@@ -33,7 +49,7 @@ require('./lib/cache')._activeCache = cache;
|
|
|
33
49
|
// Prefer cwd config.json (user's project) over the package dir.
|
|
34
50
|
const CONFIG_CANDIDATES = [path.join(process.cwd(), 'config.json'), path.join(__dirname, 'config.json')];
|
|
35
51
|
const CONFIG_PATH = CONFIG_CANDIDATES.find(p => fs.existsSync(p)) || CONFIG_CANDIDATES[1];
|
|
36
|
-
let config = { port: 4000, auth: '', rateLimit: { maxRequests: 100, windowMs: 60000 } };
|
|
52
|
+
let config = { port: 4000, auth: '', rateLimit: { maxRequests: 100, windowMs: 60000 }, methodology: true };
|
|
37
53
|
try {
|
|
38
54
|
const parsed = JSON.parse(fs.readFileSync(CONFIG_PATH, 'utf8'));
|
|
39
55
|
if (parsed && typeof parsed === 'object') config = { ...config, ...parsed };
|
|
@@ -51,6 +67,75 @@ const PORT = parseInt(process.env.PORT || getArg('port', config.port || '4000'))
|
|
|
51
67
|
const AUTH_KEY = process.env.AUTH || getArg('auth', config.auth || '');
|
|
52
68
|
const RATE_LIMIT = config.rateLimit || { maxRequests: 100, windowMs: 60000 };
|
|
53
69
|
|
|
70
|
+
// --- Долговременная память (vector memory) ---
|
|
71
|
+
// По умолчанию включена: факты берутся побочно от компакции (без лишних вызовов
|
|
72
|
+
// LLM) и подмешиваются в контекст по релевантности. config.memory.enabled=false —
|
|
73
|
+
// слой отключается целиком.
|
|
74
|
+
const MEMORY_CONFIG = Object.assign(
|
|
75
|
+
{ enabled: true, topK: 3, minSimilarity: 0.05, coverageForSkip: 0.6 },
|
|
76
|
+
(config.memory && typeof config.memory === 'object') ? config.memory : {}
|
|
77
|
+
);
|
|
78
|
+
const memStore = MEMORY_CONFIG.enabled ? createMemoryStore({ filePath: path.join(__dirname, 'memory.json') }) : null;
|
|
79
|
+
if (memStore) compactorSetMemory(memStore);
|
|
80
|
+
|
|
81
|
+
// --- Семантический кэш ---
|
|
82
|
+
// На промахе точного ключа ищет закэшированный диалог с похожим нормализованным
|
|
83
|
+
// текстом (Dice по символьным триграммам). config.semcache.enabled=false отключает.
|
|
84
|
+
const SEMCACHE_CONFIG = Object.assign(
|
|
85
|
+
{ enabled: true, minSimilarity: 0.85 },
|
|
86
|
+
(config.semcache && typeof config.semcache === 'object') ? config.semcache : {}
|
|
87
|
+
);
|
|
88
|
+
|
|
89
|
+
// --- Методолог (инженерная дисциплина в промпте) ---
|
|
90
|
+
// config.methodology.enabled=false отключает. По умолчанию включён.
|
|
91
|
+
// config.methodology.prompts.{category} переопределяет текст промпта для
|
|
92
|
+
// конкретной категории (незаданные категории остаются в дефолтах).
|
|
93
|
+
const METHODOLOGY_CONFIG = Object.assign(
|
|
94
|
+
{ enabled: methodEnabledByDefault(), prompts: {} },
|
|
95
|
+
(config.methodology && typeof config.methodology === 'object') ? config.methodology : {}
|
|
96
|
+
);
|
|
97
|
+
|
|
98
|
+
// --- Самообновляющаяся база моделей (Model Discovery Engine) ---
|
|
99
|
+
// Планировщик живёт внутри сервера: работает «всегда» у всех пользователей
|
|
100
|
+
// пакета без cron/launchd. config.modelManager.enabled=false отключает.
|
|
101
|
+
function _loadEnvFile() {
|
|
102
|
+
try {
|
|
103
|
+
const envPath = path.join(__dirname, '.env');
|
|
104
|
+
if (!fs.existsSync(envPath)) return;
|
|
105
|
+
for (const line of fs.readFileSync(envPath, 'utf8').split('\n')) {
|
|
106
|
+
const t = line.trim();
|
|
107
|
+
if (!t || t.startsWith('#')) continue;
|
|
108
|
+
const eq = t.indexOf('=');
|
|
109
|
+
if (eq > 0 && !process.env[t.slice(0, eq).trim()]) process.env[t.slice(0, eq).trim()] = t.slice(eq + 1).trim();
|
|
110
|
+
}
|
|
111
|
+
} catch {}
|
|
112
|
+
}
|
|
113
|
+
_loadEnvFile();
|
|
114
|
+
const { ModelManager } = require('./lib/modelmanager');
|
|
115
|
+
const MODEL_MANAGER_CONFIG = Object.assign(
|
|
116
|
+
{},
|
|
117
|
+
(config.modelManager && typeof config.modelManager === 'object') ? config.modelManager : {}
|
|
118
|
+
);
|
|
119
|
+
const modelManager = new ModelManager({
|
|
120
|
+
dbPath: path.join(__dirname, 'models-db.json'),
|
|
121
|
+
catalogPath: path.join(__dirname, 'providers.json'),
|
|
122
|
+
configPath: path.join(__dirname, 'config.json'),
|
|
123
|
+
keys: {
|
|
124
|
+
openrouter: process.env.PROVIDER_OPENROUTER_APIKEY || '',
|
|
125
|
+
huggingface: process.env.HF_TOKEN || process.env.PROVIDER_HF_APIKEY || '',
|
|
126
|
+
groq: process.env.PROVIDER_GROQ_APIKEY || '',
|
|
127
|
+
mistral: process.env.PROVIDER_MISTRAL_APIKEY || '',
|
|
128
|
+
gemini: process.env.PROVIDER_GEMINI_APIKEY || '',
|
|
129
|
+
cerebras: process.env.PROVIDER_CEREBRAS_APIKEY || '',
|
|
130
|
+
deepseek: process.env.PROVIDER_DEEPSEEK_APIKEY || '',
|
|
131
|
+
nim: process.env.PROVIDER_NIM_APIKEY || '',
|
|
132
|
+
},
|
|
133
|
+
fetchImpl: (url, opts) => fetch(url, opts),
|
|
134
|
+
config: MODEL_MANAGER_CONFIG,
|
|
135
|
+
reload: () => { try { reloadProviders(); } catch {} },
|
|
136
|
+
log: (msg) => logger.info('[modelManager] ' + msg),
|
|
137
|
+
});
|
|
138
|
+
|
|
54
139
|
// Health check
|
|
55
140
|
const healthIntervals = {}; // key -> { nextCheck, backoff }
|
|
56
141
|
|
|
@@ -123,6 +208,26 @@ async function healthCheck() {
|
|
|
123
208
|
setInterval(healthCheck, 30000);
|
|
124
209
|
setTimeout(healthCheck, 1000);
|
|
125
210
|
|
|
211
|
+
// Периодический сейв долговременной памяти (вдобавок к shutdown).
|
|
212
|
+
if (memStore) setInterval(() => memStore.save(), 60000);
|
|
213
|
+
|
|
214
|
+
// Извлекает usage из SSE-чанка (если провайдер шлёт его в последнем чанке).
|
|
215
|
+
function collectReasonUsage(str, usageObj) {
|
|
216
|
+
if (!str || !/data: /.test(str)) return;
|
|
217
|
+
for (const line of str.split('\n')) {
|
|
218
|
+
const m = line.match(/^data: (.+)$/);
|
|
219
|
+
if (!m || m[1].trim() === '[DONE]') continue;
|
|
220
|
+
try {
|
|
221
|
+
const obj = JSON.parse(m[1]);
|
|
222
|
+
if (obj.usage && (obj.usage.prompt_tokens || obj.usage.completion_tokens)) {
|
|
223
|
+
usageObj.prompt_tokens = obj.usage.prompt_tokens;
|
|
224
|
+
usageObj.completion_tokens = obj.usage.completion_tokens;
|
|
225
|
+
usageObj.total_tokens = obj.usage.total_tokens;
|
|
226
|
+
}
|
|
227
|
+
} catch {}
|
|
228
|
+
}
|
|
229
|
+
}
|
|
230
|
+
|
|
126
231
|
// Извлекает все значения content из SSE-чанка. Возвращает true, если есть
|
|
127
232
|
// хотя бы одно непустое (реальный токен, а не пустая дельта).
|
|
128
233
|
function chunkHasToken(str) {
|
|
@@ -134,6 +239,22 @@ function chunkHasToken(str) {
|
|
|
134
239
|
return false;
|
|
135
240
|
}
|
|
136
241
|
|
|
242
|
+
// Последний user-текст — для оценки, насколько короткий ответ легитимен.
|
|
243
|
+
function lastUserText(messages) {
|
|
244
|
+
if (!Array.isArray(messages)) return '';
|
|
245
|
+
for (let i = messages.length - 1; i >= 0; i--) {
|
|
246
|
+
const m = messages[i];
|
|
247
|
+
if (m && m.role === 'user') {
|
|
248
|
+
if (typeof m.content === 'string') return m.content;
|
|
249
|
+
if (Array.isArray(m.content)) {
|
|
250
|
+
const t = m.content.filter(c => c && c.type === 'text').map(c => c.text || '').join(' ');
|
|
251
|
+
if (t) return t;
|
|
252
|
+
}
|
|
253
|
+
}
|
|
254
|
+
}
|
|
255
|
+
return '';
|
|
256
|
+
}
|
|
257
|
+
|
|
137
258
|
// Chat completion handler
|
|
138
259
|
async function handleChatCompletion(req, res, body) {
|
|
139
260
|
const requestedModel = body.model || 'tier-splus';
|
|
@@ -144,6 +265,12 @@ async function handleChatCompletion(req, res, body) {
|
|
|
144
265
|
let effectiveModel = maybeUpgradeTier(requestedModel, complexity);
|
|
145
266
|
let targetProviderKey = MODEL_MAP[effectiveModel] || MODEL_MAP[requestedModel] || 'zai';
|
|
146
267
|
const isStreaming = body.stream === true;
|
|
268
|
+
// --- Контекстная телеметрия: один measure на запрос, record только в терминальной точке. ---
|
|
269
|
+
const measure = { ts: Date.now(), provider: targetProviderKey, cacheType: 'miss', status: 0, win: PROVIDERS[targetProviderKey]?.context_window || 0, upgraded: 0 };
|
|
270
|
+
const commit = (status) => {
|
|
271
|
+
measure.status = status;
|
|
272
|
+
contextStats.record(measure);
|
|
273
|
+
};
|
|
147
274
|
|
|
148
275
|
// Vision detection: if the request contains images, route to a vision provider.
|
|
149
276
|
// TWO-STAGE pipeline:
|
|
@@ -182,29 +309,41 @@ async function handleChatCompletion(req, res, body) {
|
|
|
182
309
|
if (visionChain.length > 0) {
|
|
183
310
|
logger.info('Vision pipeline: распознаю скриншот', { chain: visionChain.map(p => p.key).join(',') });
|
|
184
311
|
let extracted = '';
|
|
185
|
-
|
|
186
|
-
|
|
187
|
-
|
|
188
|
-
|
|
189
|
-
|
|
190
|
-
|
|
191
|
-
|
|
192
|
-
|
|
193
|
-
|
|
194
|
-
|
|
195
|
-
|
|
196
|
-
|
|
197
|
-
|
|
198
|
-
|
|
199
|
-
|
|
200
|
-
|
|
201
|
-
|
|
202
|
-
|
|
203
|
-
|
|
204
|
-
|
|
205
|
-
|
|
206
|
-
|
|
312
|
+
// PARALLEL vision attempt: fire all vision providers at once and take the
|
|
313
|
+
// first one that extracts text. Previously each was tried IN SEQUENCE,
|
|
314
|
+
// so a slow/failed first provider meant the pipeline waited provider
|
|
315
|
+
// after provider — the "two chats think forever" pattern for screenshots.
|
|
316
|
+
const visionAttempts = visionChain.map((visionProvider) => (async () => {
|
|
317
|
+
const visionBody = {
|
|
318
|
+
model: visionProvider.model,
|
|
319
|
+
messages: [{
|
|
320
|
+
role: 'user',
|
|
321
|
+
content: [
|
|
322
|
+
{ type: 'text', text: 'Распознай и извлеки ВЕСЬ текст с изображения (ошибка, код, сообщение). Верни только содержимое, без комментариев. Если это код — верни код как есть.' },
|
|
323
|
+
...(Array.isArray(body.messages) ? body.messages.flatMap((m) => (Array.isArray(m.content) ? m.content.filter((c) => c && (c.type === 'image_url' || c.type === 'image' || c.type === 'input_image')).map((c) => {
|
|
324
|
+
// Normalize any image part to the universal image_url format
|
|
325
|
+
const url = c.image_url?.url || c.image?.url || c.image?.data || (c.image && typeof c.image === 'string' ? c.image : null) || c.url;
|
|
326
|
+
return url ? { type: 'image_url', image_url: { url } } : null;
|
|
327
|
+
}).filter(Boolean) : [])) : []),
|
|
328
|
+
],
|
|
329
|
+
}],
|
|
330
|
+
max_tokens: 2000,
|
|
331
|
+
};
|
|
332
|
+
const visionRes = await callProvider(visionProvider, visionBody);
|
|
333
|
+
const text = visionRes.data?.choices?.[0]?.message?.content || visionRes.data?.choices?.[0]?.message?.reasoning || '';
|
|
334
|
+
if (text) {
|
|
335
|
+
logger.info('Vision pipeline: распознал ' + visionProvider.key);
|
|
336
|
+
return text;
|
|
207
337
|
}
|
|
338
|
+
throw new Error(visionProvider.key + ': пустой OCR');
|
|
339
|
+
})());
|
|
340
|
+
// First provider to yield text wins; hard failures (429/5xx) are skipped
|
|
341
|
+
// without blocking the others. If ALL fail, extracted stays '' and the
|
|
342
|
+
// request proceeds without vision context (as before).
|
|
343
|
+
try {
|
|
344
|
+
extracted = await Promise.any(visionAttempts);
|
|
345
|
+
} catch {
|
|
346
|
+
logger.warn('Vision pipeline: все вижн-провайдеры не сработали', { tried: visionChain.map(p => p.key) });
|
|
208
347
|
}
|
|
209
348
|
const cleaned = stripThink(extracted, true);
|
|
210
349
|
logger.info('Vision pipeline: скриншот распознан', { chars: cleaned.length });
|
|
@@ -233,13 +372,94 @@ async function handleChatCompletion(req, res, body) {
|
|
|
233
372
|
}
|
|
234
373
|
}
|
|
235
374
|
|
|
236
|
-
//
|
|
237
|
-
|
|
238
|
-
|
|
239
|
-
|
|
240
|
-
|
|
375
|
+
// Window-aware upgrade: если запрос не влезает в окно целевой модели (после
|
|
376
|
+
// vision-апгрейда target), компакция НЕ запускается — суммаризатор не должен
|
|
377
|
+
// сжимать контекст, который провайдер с большим окном возьмёт целиком.
|
|
378
|
+
const windowUpgraded = Array.isArray(body.messages) && body.messages.length > 0 &&
|
|
379
|
+
needsWindowUpgrade(PROVIDERS[targetProviderKey]?.context_window || 0, estimateTokens(body.messages));
|
|
380
|
+
|
|
381
|
+
// Compact overly large conversations so free models don't reject on context.
|
|
382
|
+
// Runs AFTER the vision pipeline (images already converted to text above).
|
|
383
|
+
// Skipped in window-upgrade mode — the big provider takes the raw context.
|
|
384
|
+
if (Array.isArray(body.messages)) measure.origTokens = estimateTokens(body.messages);
|
|
385
|
+
if (Array.isArray(body.messages) && body.messages.length > 0 && !windowUpgraded) {
|
|
386
|
+
body.messages = await prepareMessages(body.messages, { contextWindow: PROVIDERS[targetProviderKey]?.context_window || 0 });
|
|
387
|
+
}
|
|
388
|
+
// Контекстная телеметрия: токены после компакции + доля системного промпта.
|
|
389
|
+
if (Array.isArray(body.messages)) {
|
|
390
|
+
measure.sentTokens = estimateTokens(body.messages);
|
|
391
|
+
measure.est = measure.sentTokens;
|
|
392
|
+
measure.compacted = measure.sentTokens < measure.origTokens;
|
|
393
|
+
const sysChars = body.messages.filter(m => m && m.role === 'system').reduce((a, m) => a + (typeof m.content === 'string' ? m.content.length : 0), 0);
|
|
394
|
+
const allChars = body.messages.reduce((a, m) => a + (typeof (m && m.content) === 'string' ? m.content.length : 0), 0);
|
|
395
|
+
if (allChars > 0) measure.sysShare = sysChars / allChars;
|
|
396
|
+
}
|
|
397
|
+
|
|
398
|
+
// --- Long-term memory recall ---
|
|
399
|
+
// Подмешиваем релевантные факты из прошлых сессий (векторная память) как
|
|
400
|
+
// user-сообщение В НАЧАЛЕ диалога — после системных правил, до кэша. Так
|
|
401
|
+
// ключ кэша (normalize игнорирует system, но учитывает user) различает разные
|
|
402
|
+
// наборы фактов, и ответы не отравляются чужим кэшем.
|
|
403
|
+
if (memStore) {
|
|
404
|
+
try {
|
|
405
|
+
const userTexts = [];
|
|
406
|
+
const userMsgs = body.messages.filter(m => m && m.role === 'user');
|
|
407
|
+
for (const m of userMsgs.slice(-3)) {
|
|
408
|
+
if (typeof m.content === 'string') userTexts.push(m.content);
|
|
409
|
+
else if (Array.isArray(m.content)) userTexts.push(m.content.filter(c => c && c.type === 'text').map(c => c.text || '').join(' '));
|
|
410
|
+
}
|
|
411
|
+
const query = userTexts.join('\n').trim();
|
|
412
|
+
if (query.length > 20) {
|
|
413
|
+
const hits = memStore.recall(query, { topK: MEMORY_CONFIG.topK, minSimilarity: MEMORY_CONFIG.minSimilarity });
|
|
414
|
+
if (hits.length > 0) {
|
|
415
|
+
// Не подмешиваем факт, который уже покрыт резюме компактора или
|
|
416
|
+
// недавними сообщениями этой же сессии (защита от дублей).
|
|
417
|
+
const existing = body.messages
|
|
418
|
+
.filter(m => m && typeof m.content === 'string')
|
|
419
|
+
.map(m => m.content)
|
|
420
|
+
.concat(userTexts);
|
|
421
|
+
const fresh = hits.filter(f => !memStore.isCovered(f.text, existing));
|
|
422
|
+
if (fresh.length > 0) {
|
|
423
|
+
let memoryMsg = fresh.map(f => f.text).join('\n');
|
|
424
|
+
if (memoryMsg) {
|
|
425
|
+
const insertAt = body.messages.findIndex(m => m && m.role !== 'system');
|
|
426
|
+
const block = { role: 'user', content: '[Память: релевантные факты из прошлого]\n' + memoryMsg };
|
|
427
|
+
if (insertAt === -1) body.messages.unshift(block);
|
|
428
|
+
else body.messages.splice(insertAt, 0, block);
|
|
429
|
+
measure.memory = true;
|
|
430
|
+
logger.info('Memory recall', { facts: fresh.length, covered: hits.length - fresh.length });
|
|
431
|
+
}
|
|
432
|
+
}
|
|
433
|
+
}
|
|
434
|
+
}
|
|
435
|
+
} catch (err) {
|
|
436
|
+
logger.error('Memory recall error', { message: err.message });
|
|
437
|
+
}
|
|
438
|
+
}
|
|
439
|
+
|
|
440
|
+
// --- Методолог (инженерная дисциплина) ---
|
|
441
|
+
// Классифицируем задачу (coding/reasoning/search/chat) и вставляем короткий
|
|
442
|
+
// системный промпт-методолог после памяти, до кэша. System-сообщение
|
|
443
|
+
// игнорируется normalize, поэтому кэш-ключ не меняется. Методолог влияет
|
|
444
|
+
// только на реальные запросы к провайдеру (кэш-хиты его не видят).
|
|
445
|
+
let taskCategory = 'chat';
|
|
446
|
+
if (METHODOLOGY_CONFIG.enabled && Array.isArray(body.messages)) {
|
|
447
|
+
try {
|
|
448
|
+
taskCategory = classifyTask(body.messages);
|
|
449
|
+
const injected = injectMethodology(body.messages, taskCategory, METHODOLOGY_CONFIG);
|
|
450
|
+
if (injected !== body.messages) {
|
|
451
|
+
body.messages = injected;
|
|
452
|
+
measure.taskCategory = taskCategory;
|
|
453
|
+
}
|
|
454
|
+
} catch (err) {
|
|
455
|
+
logger.error('Methodology error', { message: err.message });
|
|
456
|
+
}
|
|
457
|
+
}
|
|
458
|
+
|
|
459
|
+
// Replays a previously cached completion (exact or semantic hit), preserving
|
|
460
|
+
// the stream/non-stream shape the client asked for.
|
|
461
|
+
function serveCached(res, cached, isStreaming) {
|
|
241
462
|
if (isStreaming) {
|
|
242
|
-
// Replay cached answer as an SSE stream
|
|
243
463
|
res.writeHead(200, { 'Content-Type': 'text/event-stream', 'Cache-Control': 'no-cache', 'Connection': 'keep-alive' });
|
|
244
464
|
const content = cached.choices?.[0]?.message?.content || '';
|
|
245
465
|
res.write(`data: ${JSON.stringify({ id: 'chatcmpl-cached', object: 'chat.completion.chunk', created: Math.floor(Date.now() / 1000), model: cached.model, choices: [{ index: 0, delta: { role: 'assistant', content: '' }, finish_reason: null }] })}\n\n`);
|
|
@@ -247,13 +467,40 @@ async function handleChatCompletion(req, res, body) {
|
|
|
247
467
|
res.write(`data: ${JSON.stringify({ id: 'chatcmpl-cached', object: 'chat.completion.chunk', created: Math.floor(Date.now() / 1000), model: cached.model, choices: [{ index: 0, delta: {}, finish_reason: 'stop' }] })}\n\n`);
|
|
248
468
|
res.write('data: [DONE]\n\n');
|
|
249
469
|
res.end();
|
|
250
|
-
|
|
470
|
+
} else {
|
|
471
|
+
res.writeHead(200, { 'Content-Type': 'application/json' });
|
|
472
|
+
res.end(JSON.stringify(cached));
|
|
251
473
|
}
|
|
252
|
-
|
|
253
|
-
|
|
474
|
+
}
|
|
475
|
+
|
|
476
|
+
// Check cache (works for both streaming and non-streaming)
|
|
477
|
+
const cached = cache.get(effectiveModel, body.messages, body.temperature);
|
|
478
|
+
if (cached) {
|
|
479
|
+
logger.request({ model: requestedModel, provider: 'cache', status: 200, cached: true });
|
|
480
|
+
recordRecent({ model: requestedModel, provider: 'cache', status: 200, latency: 0, cached: true });
|
|
481
|
+
measure.cacheType = 'exact';
|
|
482
|
+
commit(200);
|
|
483
|
+
serveCached(res, cached, isStreaming);
|
|
254
484
|
return;
|
|
255
485
|
}
|
|
256
486
|
|
|
487
|
+
// Semantic cache: same intent, rephrased wording → replay without a new LLM call.
|
|
488
|
+
if (SEMCACHE_CONFIG.enabled) {
|
|
489
|
+
const semantic = cache.getSemantic(effectiveModel, body.messages, body.temperature, SEMCACHE_CONFIG.minSimilarity);
|
|
490
|
+
if (semantic) {
|
|
491
|
+
logger.request({ model: requestedModel, provider: 'semcache', status: 200, cached: true });
|
|
492
|
+
recordRecent({ model: requestedModel, provider: 'semcache', status: 200, latency: 0, cached: true });
|
|
493
|
+
measure.cacheType = 'semcache';
|
|
494
|
+
commit(200);
|
|
495
|
+
serveCached(res, semantic.value, isStreaming);
|
|
496
|
+
return;
|
|
497
|
+
}
|
|
498
|
+
}
|
|
499
|
+
|
|
500
|
+
// Запрос реально идёт к провайдеру (кэш промахнулся) — только теперь помечаем
|
|
501
|
+
// апгрейд: кэш-хит апгрейдом не считается (апстрим-вызова не было).
|
|
502
|
+
if (windowUpgraded) measure.upgraded = 1;
|
|
503
|
+
|
|
257
504
|
// Weighted selection among healthy providers
|
|
258
505
|
const today = new Date().toISOString().slice(0, 10);
|
|
259
506
|
// 'ratelimited' providers are alive but temporarily limited — include them
|
|
@@ -264,44 +511,102 @@ async function handleChatCompletion(req, res, body) {
|
|
|
264
511
|
.filter(([_, p]) => p.enabled && !isCircuitOpen(p.key) && p.vision !== true &&
|
|
265
512
|
(getHealth()[p.key]?.status === 'up' || getHealth()[p.key]?.status === 'ratelimited'));
|
|
266
513
|
|
|
514
|
+
// Window-aware routing: estimate the request size and only consider providers
|
|
515
|
+
// whose context window can actually hold it. This stops large requests from
|
|
516
|
+
// burning time falling through lfm (65k) / groq (131k) providers that reject
|
|
517
|
+
// them — they go straight to nemotron-35/dots-3/minimax (1M/512k windows).
|
|
518
|
+
// Only applies when the request is big enough to matter, so small/typical
|
|
519
|
+
// requests keep the full fast pool.
|
|
520
|
+
const requestTokens = estimateTokens(body.messages);
|
|
521
|
+
const MIN_WINDOW = 50000; // below this we don't filter (typical requests)
|
|
522
|
+
let windowPool = healthyProviders;
|
|
523
|
+
let upgradeNoCapable = false; // апгрейд, но ни один здоровый провайдер не держит запрос
|
|
524
|
+
if (requestTokens > MIN_WINDOW || windowUpgraded) {
|
|
525
|
+
const capable = healthyProviders.filter(([_, p]) => {
|
|
526
|
+
const win = p.context_window || 0;
|
|
527
|
+
// Unknown/0 window providers are kept (heuristic) — better to try than drop.
|
|
528
|
+
return win === 0 || win >= requestTokens;
|
|
529
|
+
});
|
|
530
|
+
if (capable.length > 0) {
|
|
531
|
+
windowPool = capable;
|
|
532
|
+
} else if (windowUpgraded) {
|
|
533
|
+
// Best-effort: запрос больше окна любого провайдера (minimax 1M не держит).
|
|
534
|
+
// Всё равно переполним — выберем самое большое окно ниже, минимизируя
|
|
535
|
+
// потери контекста; target исключён; OVERFLOW поймает телеметрия.
|
|
536
|
+
upgradeNoCapable = true;
|
|
537
|
+
}
|
|
538
|
+
}
|
|
539
|
+
|
|
267
540
|
// Prefer providers below 90% of their daily limit; only fall back to
|
|
268
541
|
// near-exhausted ones if that leaves nothing (avoids avoidable 429s).
|
|
269
|
-
|
|
270
|
-
|
|
271
|
-
|
|
272
|
-
|
|
273
|
-
|
|
274
|
-
|
|
275
|
-
|
|
542
|
+
// Крутящийся пул по дневным лимитам: провайдер, исчерпавший дневной лимит,
|
|
543
|
+
// исключается из выбора и не возвращается до сброса. Это лечит 429-спираль
|
|
544
|
+
// (модель с лимитом 50/день сгорает к обеду, дальше каждая попытка на ней —
|
|
545
|
+
// бесполезный 429). Остаёмся на живых; выгоревшие — только как крайний резерв.
|
|
546
|
+
const usedTodayFor = (key) => (getStats().dailyUsage?.[key]?.[today]) || 0;
|
|
547
|
+
const pool = (() => {
|
|
548
|
+
// Провайдеры, у которых дневной лимит ещё не исчерпан (strict < limit).
|
|
549
|
+
const exemptables = windowPool.filter(([_, p]) => {
|
|
550
|
+
const limit = p.dailyLimit || 0;
|
|
551
|
+
if (limit <= 0) return true; // нет лимита — считаем «бесконечным»
|
|
552
|
+
return usedTodayFor(p.key) < limit;
|
|
553
|
+
});
|
|
554
|
+
// Есть живые → только они. Все выгорели → вернуть весь пул (best-effort,
|
|
555
|
+
// bandit-штраф ниже сделает их маловероятными, но не невозможными).
|
|
556
|
+
return exemptables.length > 0 ? exemptables : windowPool;
|
|
557
|
+
})();
|
|
276
558
|
|
|
277
559
|
let selected = [];
|
|
278
560
|
if (pool.length > 0) {
|
|
279
|
-
|
|
280
|
-
|
|
281
|
-
|
|
282
|
-
|
|
283
|
-
|
|
284
|
-
|
|
285
|
-
|
|
286
|
-
|
|
287
|
-
const
|
|
288
|
-
|
|
289
|
-
|
|
290
|
-
|
|
291
|
-
|
|
561
|
+
if (upgradeNoCapable) {
|
|
562
|
+
// Bandit здесь бессилен: все провайдеры в пуле переполнят окно (запрос
|
|
563
|
+
// больше самого большого). Берём самое большое окно — наименьшие потери.
|
|
564
|
+
selected = pool
|
|
565
|
+
.map(([k, p]) => ({ key: k, provider: p }))
|
|
566
|
+
.sort((a, b) => (b.provider.context_window || 0) - (a.provider.context_window || 0))
|
|
567
|
+
.slice(0, 1);
|
|
568
|
+
} else {
|
|
569
|
+
const scored = pool.map(([key, provider]) => {
|
|
570
|
+
const h = getHealth()[key];
|
|
571
|
+
let score = h.score || 50;
|
|
572
|
+
const rawLat = h.latency || 0;
|
|
573
|
+
const lat = rawLat > 0 ? Math.max(rawLat, 100) : 500;
|
|
574
|
+
let weight = score / lat;
|
|
575
|
+
if (h.status === 'ratelimited') weight *= 0.05;
|
|
576
|
+
if (key === targetProviderKey) weight *= 1.15;
|
|
577
|
+
// Методолог: категория задачи задаёт буст моделям подходящей категории,
|
|
578
|
+
// не исключая fallback. coding→coding, reasoning→reasoning, chat/search→general.
|
|
579
|
+
const cat = provider.category || 'general';
|
|
580
|
+
if (taskCategory === 'coding' && cat === 'coding') weight *= 1.5;
|
|
581
|
+
else if (taskCategory === 'reasoning' && cat === 'reasoning') weight *= 1.5;
|
|
582
|
+
else if ((taskCategory === 'chat' || taskCategory === 'search') && cat === 'general') weight *= 1.2;
|
|
583
|
+
// Простой факт/поиск не должен «думать вслух» на reasoning-модели:
|
|
584
|
+
// это дорого по лимитам и даёт размышления вместо краткого ответа.
|
|
585
|
+
if ((taskCategory === 'chat' || taskCategory === 'search') && cat === 'reasoning') weight *= 0.15;
|
|
586
|
+
else if ((taskCategory === 'chat' || taskCategory === 'search') && cat === 'vision') weight *= 0.3;
|
|
587
|
+
const dailyLimit = provider.dailyLimit || 0;
|
|
588
|
+
// Провайдер, исчерпавший дневной лимит (попал в пул лишь как крайний
|
|
589
|
+
// резерв, когда живы все выгорели), — сильно штрафуем, чтобы выбрать
|
|
590
|
+
// его только в безвыходной ситуации, а не в первой же попытке.
|
|
591
|
+
const usedToday = usedTodayFor(key);
|
|
592
|
+
if (dailyLimit > 0 && usedToday >= dailyLimit) weight *= 0.03;
|
|
593
|
+
else if (dailyLimit > 0 && usedToday >= dailyLimit * 0.9) weight *= 0.4;
|
|
594
|
+
return { key, provider, weight };
|
|
595
|
+
});
|
|
292
596
|
|
|
293
|
-
|
|
294
|
-
|
|
295
|
-
|
|
296
|
-
|
|
297
|
-
|
|
298
|
-
|
|
299
|
-
|
|
300
|
-
|
|
301
|
-
|
|
302
|
-
|
|
303
|
-
|
|
304
|
-
|
|
597
|
+
// Bandit weight contract: bandit's pick() multiplies the Beta sample by
|
|
598
|
+
// `weight`, so safety-штрафы (ratelimited ×0.05, target ×1.15) действуют и
|
|
599
|
+
// при холодном старте. score/latency держит вес ~0.01-1.0; приоры bandit'а
|
|
600
|
+
// (a,b ~1+) со временем начинают доминировать. Не добавляй нормализацию
|
|
601
|
+
// здесь, пока измеренные веса не превысят ~5.
|
|
602
|
+
// Thompson sampling: рисуем сэмпл Beta(a+1, b+1) для каждого, умножаем на
|
|
603
|
+
// weight, выбираем максимум. Приоры из бакета сложности (bandit обучается).
|
|
604
|
+
const priors = getBandit()[complexityBucket] || {};
|
|
605
|
+
const bestKey = banditPick(scored, priors);
|
|
606
|
+
const bestProvider = scored.find((p) => p.key === bestKey);
|
|
607
|
+
if (bestProvider) selected = [bestProvider];
|
|
608
|
+
else if (scored.length > 0) selected = [scored[0]];
|
|
609
|
+
}
|
|
305
610
|
}
|
|
306
611
|
|
|
307
612
|
// Weighted-random picked ONE provider as the primary; append the rest of the
|
|
@@ -310,21 +615,37 @@ async function handleChatCompletion(req, res, body) {
|
|
|
310
615
|
? pool.map(([k, p]) => ({ key: k, provider: p })).filter(s => s.key !== selected[0].key)
|
|
311
616
|
.sort((a, b) => (getHealth()[b.key]?.score || 0) - (getHealth()[a.key]?.score || 0))
|
|
312
617
|
: [];
|
|
313
|
-
|
|
618
|
+
let enabledProviders = selected.length > 0
|
|
314
619
|
? [selected[0]].concat(restOfPool).map(s => [s.key, s.provider])
|
|
315
620
|
: Object.entries(PROVIDERS).filter(([_, p]) => p.enabled)
|
|
316
621
|
.sort((a, b) => (getHealth()[b[0]]?.score || 50) - (getHealth()[a[0]]?.score || 50));
|
|
317
622
|
|
|
318
|
-
//
|
|
319
|
-
//
|
|
320
|
-
//
|
|
321
|
-
|
|
322
|
-
|
|
323
|
-
|
|
623
|
+
// Put the requested model's mapped provider FIRST. It's the only provider
|
|
624
|
+
// guaranteed to accept this tier/model — the rest are fallbacks (many reject
|
|
625
|
+
// tier-* requests with 400/422). Trying them before the target produced huge
|
|
626
|
+
// serial fallback chains (10+ sequential HTTP calls per request), which looked
|
|
627
|
+
// like the "model thinking forever". Correct mapping beats weighted guessing.
|
|
628
|
+
// Skipped in window-upgrade mode: the target doesn't fit the request anyway.
|
|
629
|
+
// If the target is rate-limited or near/over its daily limit, DON'T put it
|
|
630
|
+
// first — otherwise every request burns a doomed 429 attempt on it and the
|
|
631
|
+
// pool collapses into a rate-limit spiral. Skip straight to the healthy pool.
|
|
632
|
+
if (!windowUpgraded && MODEL_MAP[requestedModel] && PROVIDERS[targetProviderKey]) {
|
|
633
|
+
const tHealth = getHealth()[targetProviderKey];
|
|
634
|
+
const tLimit = (getStats().dailyUsage?.[targetProviderKey]?.[today]) || 0;
|
|
635
|
+
const tCap = PROVIDERS[targetProviderKey].dailyLimit || 0;
|
|
636
|
+
const targetBurned = tHealth?.status === 'ratelimited' || (tCap > 0 && tLimit >= tCap * 0.9);
|
|
637
|
+
if (!targetBurned) {
|
|
638
|
+
enabledProviders = [
|
|
639
|
+
[targetProviderKey, PROVIDERS[targetProviderKey]],
|
|
640
|
+
...enabledProviders.filter(([k]) => k !== targetProviderKey),
|
|
641
|
+
];
|
|
642
|
+
} else {
|
|
643
|
+
logger.info('Target-first skip', { key: targetProviderKey, status: tHealth?.status, used: tLimit, cap: tCap });
|
|
324
644
|
}
|
|
325
645
|
}
|
|
326
646
|
|
|
327
647
|
if (enabledProviders.length === 0) {
|
|
648
|
+
commit(503);
|
|
328
649
|
res.writeHead(503, { 'Content-Type': 'application/json' });
|
|
329
650
|
res.end(JSON.stringify({ error: 'No providers available' }));
|
|
330
651
|
return;
|
|
@@ -332,7 +653,13 @@ async function handleChatCompletion(req, res, body) {
|
|
|
332
653
|
|
|
333
654
|
const errors = [];
|
|
334
655
|
|
|
335
|
-
|
|
656
|
+
// Cap serial fallback attempts. Trying provider after provider sequentially
|
|
657
|
+
// made a single tier-* request walk 10+ providers (each a real HTTP call),
|
|
658
|
+
// looking like the model "thinks forever". target-first above fixes the common
|
|
659
|
+
// case (right provider immediately); this cap bounds the worst case.
|
|
660
|
+
const fallbackProviders = enabledProviders.slice(0, 5);
|
|
661
|
+
|
|
662
|
+
for (const [key, provider] of fallbackProviders) {
|
|
336
663
|
if (isCircuitOpen(key)) {
|
|
337
664
|
errors.push(key + ': circuit breaker open');
|
|
338
665
|
continue;
|
|
@@ -354,14 +681,13 @@ async function handleChatCompletion(req, res, body) {
|
|
|
354
681
|
getHealth()[key].latency = Math.min(result.latency || 0, 60000);
|
|
355
682
|
getHealth()[key].lastCheck = Date.now();
|
|
356
683
|
|
|
357
|
-
// For non-stream, verify the response isn't empty BEFORE recording success.
|
|
358
684
|
if (!isStreaming && result.data) {
|
|
359
|
-
|
|
360
|
-
|
|
361
|
-
|
|
362
|
-
|
|
363
|
-
|
|
364
|
-
|
|
685
|
+
delete result.data.nvext;
|
|
686
|
+
if (result.data.choices?.[0]) {
|
|
687
|
+
fixReasoningMessage(result.data.choices[0].message);
|
|
688
|
+
cleanMessage(result.data.choices[0].message);
|
|
689
|
+
}
|
|
690
|
+
if (isTooShort(result.data, lastUserText(body.messages))) {
|
|
365
691
|
// Пустой/мусорный ответ (провайдер-глитч) НЕ считается успехом — пробуем следующего.
|
|
366
692
|
const msg = key + ': empty or too short response';
|
|
367
693
|
errors.push(msg);
|
|
@@ -375,6 +701,7 @@ async function handleChatCompletion(req, res, body) {
|
|
|
375
701
|
}
|
|
376
702
|
|
|
377
703
|
if (isStreaming && result.stream) {
|
|
704
|
+
measure.provider = key;
|
|
378
705
|
const chunks = [];
|
|
379
706
|
|
|
380
707
|
// Очистка SSE-строки: убрать nvext, logprobs, think-блоки из дельт.
|
|
@@ -392,6 +719,7 @@ async function handleChatCompletion(req, res, body) {
|
|
|
392
719
|
});
|
|
393
720
|
|
|
394
721
|
// Сбор контент-токенов для кэша (strip think).
|
|
722
|
+
const streamUsage = {};
|
|
395
723
|
const collect = (str) => {
|
|
396
724
|
const lines = str.split('\n');
|
|
397
725
|
for (const line of lines) {
|
|
@@ -401,6 +729,12 @@ async function handleChatCompletion(req, res, body) {
|
|
|
401
729
|
const obj = JSON.parse(m[1]);
|
|
402
730
|
const delta = obj.choices?.[0]?.delta?.content;
|
|
403
731
|
if (typeof delta === 'string') chunks.push(stripThink(delta, false));
|
|
732
|
+
// OpenRouter / Nebius etc. put usage in a final chunk. Keep it.
|
|
733
|
+
if (obj.usage && (obj.usage.prompt_tokens || obj.usage.completion_tokens)) {
|
|
734
|
+
streamUsage.prompt_tokens = obj.usage.prompt_tokens;
|
|
735
|
+
streamUsage.completion_tokens = obj.usage.completion_tokens;
|
|
736
|
+
streamUsage.total_tokens = obj.usage.total_tokens;
|
|
737
|
+
}
|
|
404
738
|
} catch {}
|
|
405
739
|
}
|
|
406
740
|
};
|
|
@@ -416,17 +750,29 @@ async function handleChatCompletion(req, res, body) {
|
|
|
416
750
|
recordRecent({ model: requestedModel, provider: key, status: 200, latency: result.latency, cached: false });
|
|
417
751
|
recordSelection(key, provider.model, requestedModel);
|
|
418
752
|
const { Transform } = require('stream');
|
|
753
|
+
const reasonDec = new StringDecoder('utf8');
|
|
754
|
+
const reasonUsage = {};
|
|
419
755
|
const cleaner = new Transform({
|
|
420
756
|
transform(chunk, encoding, callback) {
|
|
421
|
-
const str =
|
|
757
|
+
const str = reasonDec.write(chunk);
|
|
758
|
+
collectReasonUsage(str, reasonUsage);
|
|
422
759
|
collect(str);
|
|
423
760
|
callback(null, cleanStr(str));
|
|
761
|
+
},
|
|
762
|
+
flush(callback) {
|
|
763
|
+
const tail = reasonDec.end();
|
|
764
|
+
if (tail) { collectReasonUsage(tail, reasonUsage); collect(tail); const tailStr = cleanStr(tail); if (tailStr) this.push(tailStr); }
|
|
765
|
+
callback();
|
|
424
766
|
}
|
|
425
767
|
});
|
|
426
768
|
result.stream.on('end', () => {
|
|
427
769
|
const full = chunks.join('');
|
|
428
770
|
// Bandit учится по качеству: пустой/мусорный стрим = фейл.
|
|
429
771
|
recordBandit(complexityBucket, key, full.trim().length >= MIN_ANSWER_LEN);
|
|
772
|
+
if (Object.keys(reasonUsage).length > 0) {
|
|
773
|
+
recordTokens(key, reasonUsage);
|
|
774
|
+
if (reasonUsage.prompt_tokens) measure.real = reasonUsage.prompt_tokens;
|
|
775
|
+
}
|
|
430
776
|
if (full.trim().length >= MIN_ANSWER_LEN) {
|
|
431
777
|
cache.set(effectiveModel, body.messages, body.temperature, {
|
|
432
778
|
id: 'chatcmpl-cached',
|
|
@@ -437,11 +783,13 @@ async function handleChatCompletion(req, res, body) {
|
|
|
437
783
|
usage: { prompt_tokens: 0, completion_tokens: 0, total_tokens: 0 },
|
|
438
784
|
});
|
|
439
785
|
}
|
|
786
|
+
commit(200);
|
|
440
787
|
res.end();
|
|
441
788
|
});
|
|
442
789
|
result.stream.on('error', (err) => {
|
|
443
790
|
logger.error('Stream error', { key, error: err.message });
|
|
444
791
|
if (!isTransientLimit(err.statusCode)) recordBandit(complexityBucket, key, false);
|
|
792
|
+
commit(err.statusCode || 502);
|
|
445
793
|
res.end();
|
|
446
794
|
});
|
|
447
795
|
result.stream.pipe(cleaner).pipe(res);
|
|
@@ -450,13 +798,24 @@ async function handleChatCompletion(req, res, body) {
|
|
|
450
798
|
|
|
451
799
|
// Обычные модели: буферизуем до первого токена (макс 5 сек).
|
|
452
800
|
// Заголовки не пишем сразу — если токена нет за 5 сек, fallback.
|
|
801
|
+
// StringDecoder держит частично пришедший multi-byte UTF-8 между чанками —
|
|
802
|
+
// иначе русский текст дробится на '' символ.
|
|
453
803
|
const rawBuf = [];
|
|
804
|
+
const streamDec = new StringDecoder('utf8');
|
|
805
|
+
// Адаптивный таймаут первого токена: используем измеренную скорость
|
|
806
|
+
// провайдера (latency от реальных запросов). Быстрые провайдеры не ждут
|
|
807
|
+
// полные 5с перед fallback'ом, а медленные (но рабочие) не отбрасываются
|
|
808
|
+
// слишком рано. Диапазон 2.5-8с для защиты от обоих крайностей.
|
|
809
|
+
const knownLat = getHealth()[key]?.latency || 0;
|
|
810
|
+
const firstTokenWait = knownLat > 0
|
|
811
|
+
? Math.max(2500, Math.min(8000, Math.round(knownLat * 2)))
|
|
812
|
+
: 5000;
|
|
454
813
|
const firstToken = new Promise((resolve) => {
|
|
455
814
|
let done = false;
|
|
456
|
-
const timer = setTimeout(() => { if (!done) { done = true; resolve(false); } },
|
|
815
|
+
const timer = setTimeout(() => { if (!done) { done = true; resolve(false); } }, firstTokenWait);
|
|
457
816
|
const finish = (ok) => { if (!done) { done = true; clearTimeout(timer); resolve(ok); } };
|
|
458
817
|
result.stream.on('data', (chunk) => {
|
|
459
|
-
const str =
|
|
818
|
+
const str = streamDec.write(chunk);
|
|
460
819
|
rawBuf.push(str);
|
|
461
820
|
collect(str);
|
|
462
821
|
// Первый контент-токен: хотя бы одно непустое `"content":"..."` в чанке.
|
|
@@ -503,16 +862,27 @@ async function handleChatCompletion(req, res, body) {
|
|
|
503
862
|
|
|
504
863
|
// Убираем наш 'data'-слушатель (он больше не нужен — данные уже
|
|
505
864
|
// буферизованы в rawBuf и промыты). Дальше обрабатываем вручную.
|
|
865
|
+
// Продолжаем использовать ТОТ ЖЕ streamDec — иначе multi-byte UTF-8,
|
|
866
|
+
// разделённый границей буфера, превратится в '' .
|
|
506
867
|
result.stream.removeAllListeners('data');
|
|
507
868
|
result.stream.on('data', (chunk) => {
|
|
508
|
-
const str =
|
|
869
|
+
const str = streamDec.write(chunk);
|
|
509
870
|
collect(str);
|
|
510
871
|
res.write(cleanStr(str));
|
|
511
872
|
});
|
|
512
873
|
result.stream.on('end', () => {
|
|
874
|
+
const tail = streamDec.end();
|
|
875
|
+
if (tail) {
|
|
876
|
+
collect(tail);
|
|
877
|
+
res.write(cleanStr(tail));
|
|
878
|
+
}
|
|
513
879
|
const full = chunks.join('');
|
|
514
880
|
// Bandit учится по качеству: обрыв/мусорный стрим = фейл.
|
|
515
881
|
recordBandit(complexityBucket, key, full.trim().length >= MIN_ANSWER_LEN);
|
|
882
|
+
if (Object.keys(streamUsage).length > 0) {
|
|
883
|
+
recordTokens(key, streamUsage);
|
|
884
|
+
if (streamUsage.prompt_tokens) measure.real = streamUsage.prompt_tokens;
|
|
885
|
+
}
|
|
516
886
|
if (full.trim().length >= MIN_ANSWER_LEN) {
|
|
517
887
|
cache.set(effectiveModel, body.messages, body.temperature, {
|
|
518
888
|
id: 'chatcmpl-cached',
|
|
@@ -523,11 +893,13 @@ async function handleChatCompletion(req, res, body) {
|
|
|
523
893
|
usage: { prompt_tokens: 0, completion_tokens: 0, total_tokens: 0 },
|
|
524
894
|
});
|
|
525
895
|
}
|
|
896
|
+
commit(200);
|
|
526
897
|
res.end();
|
|
527
898
|
});
|
|
528
899
|
result.stream.on('error', (err) => {
|
|
529
900
|
logger.error('Stream error', { key, error: err.message });
|
|
530
901
|
if (!isTransientLimit(err.statusCode)) recordBandit(complexityBucket, key, false);
|
|
902
|
+
commit(err.statusCode || 502);
|
|
531
903
|
res.end();
|
|
532
904
|
});
|
|
533
905
|
return;
|
|
@@ -541,6 +913,10 @@ async function handleChatCompletion(req, res, body) {
|
|
|
541
913
|
logger.request({ model: requestedModel, provider: key, status: 200, latency: result.latency, stream: isStreaming });
|
|
542
914
|
recordRecent({ model: requestedModel, provider: key, status: 200, latency: result.latency, cached: false });
|
|
543
915
|
recordSelection(key, provider.model, requestedModel);
|
|
916
|
+
measure.provider = key;
|
|
917
|
+
measure.real = (result.data.usage && result.data.usage.prompt_tokens) ? result.data.usage.prompt_tokens : measure.sentTokens || 0;
|
|
918
|
+
measure.win = PROVIDERS[key]?.context_window || 0;
|
|
919
|
+
commit(200);
|
|
544
920
|
cache.set(effectiveModel, body.messages, body.temperature, result.data);
|
|
545
921
|
recordTokens(key, result.usage);
|
|
546
922
|
res.writeHead(200, { 'Content-Type': 'application/json' });
|
|
@@ -569,13 +945,24 @@ async function handleChatCompletion(req, res, body) {
|
|
|
569
945
|
getHealth()[key].reason = 'не отвечает';
|
|
570
946
|
}
|
|
571
947
|
getHealth()[key].score = Math.max(0, (getHealth()[key].score || 50) - (statusCode === 429 ? 5 : 10));
|
|
572
|
-
recordFailure(key, statusCode);
|
|
573
|
-
// 404 = model not available
|
|
948
|
+
recordFailure(key, statusCode, statusCode === 404 && err.providerSide ? { providerSide: true } : undefined);
|
|
949
|
+
// 404 = model not available. Two distinct flavors:
|
|
950
|
+
// - plain 404 («does not exist») → disable permanently (model gone from OpenRouter).
|
|
951
|
+
// - provider-side 404 (Nvidia quota/upstream failing) → NOT permanent — the model
|
|
952
|
+
// may recover. Mark it down hard so routing prefers others, and let the periodic
|
|
953
|
+
// health-check re-enable it when it comes back.
|
|
574
954
|
if (statusCode === 404) {
|
|
575
|
-
|
|
576
|
-
|
|
577
|
-
|
|
578
|
-
|
|
955
|
+
if (err.providerSide) {
|
|
956
|
+
getHealth()[key].status = 'error';
|
|
957
|
+
getHealth()[key].reason = 'провайдер временно недоступен (404)';
|
|
958
|
+
getHealth()[key].score = Math.max(0, (getHealth()[key].score || 50) - 25);
|
|
959
|
+
logger.warn('Provider temporarily down (provider-side 404)', { key, model: provider.model });
|
|
960
|
+
} else {
|
|
961
|
+
provider.enabled = false;
|
|
962
|
+
getHealth()[key].status = 'disabled';
|
|
963
|
+
getHealth()[key].reason = 'отключён автоматически (404)';
|
|
964
|
+
logger.warn('Provider auto-disabled (404)', { key, model: provider.model });
|
|
965
|
+
}
|
|
579
966
|
}
|
|
580
967
|
}
|
|
581
968
|
}
|
|
@@ -585,7 +972,7 @@ async function handleChatCompletion(req, res, body) {
|
|
|
585
972
|
const allSoft = errors.length > 0 && errors.every(e => !/429|401|403|404/.test(e));
|
|
586
973
|
if (allSoft && enabledProviders.length > 1) {
|
|
587
974
|
await new Promise(r => setTimeout(r, 1500));
|
|
588
|
-
for (const [key, provider] of
|
|
975
|
+
for (const [key, provider] of fallbackProviders) {
|
|
589
976
|
if (isCircuitOpen(key)) continue;
|
|
590
977
|
try {
|
|
591
978
|
const release = await acquire(key);
|
|
@@ -601,7 +988,7 @@ async function handleChatCompletion(req, res, body) {
|
|
|
601
988
|
fixReasoningMessage(result.data.choices[0].message);
|
|
602
989
|
cleanMessage(result.data.choices[0].message);
|
|
603
990
|
}
|
|
604
|
-
|
|
991
|
+
if (isTooShort(result.data, lastUserText(body.messages))) {
|
|
605
992
|
recordFailure(key, 0);
|
|
606
993
|
recordRequest(key, false, key + ': empty or too short response (retry)');
|
|
607
994
|
recordBandit(complexityBucket, key, false);
|
|
@@ -613,6 +1000,10 @@ async function handleChatCompletion(req, res, body) {
|
|
|
613
1000
|
recordBandit(complexityBucket, key, true);
|
|
614
1001
|
recordRecent({ model: requestedModel, provider: key, status: 200, latency: result.latency, cached: false });
|
|
615
1002
|
recordSelection(key, provider.model, requestedModel);
|
|
1003
|
+
measure.provider = key;
|
|
1004
|
+
measure.real = (result.data.usage && result.data.usage.prompt_tokens) ? result.data.usage.prompt_tokens : measure.sentTokens || 0;
|
|
1005
|
+
measure.win = PROVIDERS[key]?.context_window || 0;
|
|
1006
|
+
commit(200);
|
|
616
1007
|
cache.set(effectiveModel, body.messages, body.temperature, result.data);
|
|
617
1008
|
recordTokens(key, result.usage);
|
|
618
1009
|
res.writeHead(200, { 'Content-Type': 'application/json' });
|
|
@@ -620,6 +1011,7 @@ async function handleChatCompletion(req, res, body) {
|
|
|
620
1011
|
return;
|
|
621
1012
|
}
|
|
622
1013
|
if (body.stream && result.stream) {
|
|
1014
|
+
measure.provider = key;
|
|
623
1015
|
recordSuccess(key);
|
|
624
1016
|
recordRequest(key, true);
|
|
625
1017
|
recordRecent({ model: requestedModel, provider: key, status: 200, latency: result.latency, cached: false });
|
|
@@ -650,12 +1042,15 @@ async function handleChatCompletion(req, res, body) {
|
|
|
650
1042
|
return 'data: ' + JSON.stringify(obj);
|
|
651
1043
|
} catch { return match; }
|
|
652
1044
|
});
|
|
1045
|
+
const retryDec = new StringDecoder('utf8');
|
|
653
1046
|
result.stream.on('data', (chunk) => {
|
|
654
|
-
const str =
|
|
1047
|
+
const str = retryDec.write(chunk);
|
|
655
1048
|
collectRetry(str);
|
|
656
1049
|
res.write(cleanRetry(str));
|
|
657
1050
|
});
|
|
658
1051
|
result.stream.on('end', () => {
|
|
1052
|
+
const tail = retryDec.end();
|
|
1053
|
+
if (tail) { collectRetry(tail); res.write(cleanRetry(tail)); }
|
|
659
1054
|
const full = chunks.join('');
|
|
660
1055
|
// Bandit учится по качеству в ретрае тоже.
|
|
661
1056
|
recordBandit(complexityBucket, key, full.trim().length >= MIN_ANSWER_LEN);
|
|
@@ -669,11 +1064,13 @@ async function handleChatCompletion(req, res, body) {
|
|
|
669
1064
|
usage: { prompt_tokens: 0, completion_tokens: 0, total_tokens: 0 },
|
|
670
1065
|
});
|
|
671
1066
|
}
|
|
1067
|
+
commit(200);
|
|
672
1068
|
res.end();
|
|
673
1069
|
});
|
|
674
1070
|
result.stream.on('error', (err) => {
|
|
675
1071
|
logger.error('Stream error (retry)', { key, error: err.message });
|
|
676
1072
|
if (!isTransientLimit(err.statusCode)) recordBandit(complexityBucket, key, false);
|
|
1073
|
+
commit(err.statusCode || 502);
|
|
677
1074
|
res.end();
|
|
678
1075
|
});
|
|
679
1076
|
return;
|
|
@@ -686,6 +1083,7 @@ async function handleChatCompletion(req, res, body) {
|
|
|
686
1083
|
}
|
|
687
1084
|
}
|
|
688
1085
|
|
|
1086
|
+
commit(502);
|
|
689
1087
|
res.writeHead(502, { 'Content-Type': 'application/json' });
|
|
690
1088
|
res.end(JSON.stringify({ error: { message: 'All providers failed', type: 'api_error', code: 'all_providers_failed', details: errors } }));
|
|
691
1089
|
}
|
|
@@ -715,12 +1113,145 @@ const server = http.createServer(async (req, res) => {
|
|
|
715
1113
|
return;
|
|
716
1114
|
}
|
|
717
1115
|
|
|
1116
|
+
if (parsedUrl.pathname === '/v1/reload' && req.method === 'POST') {
|
|
1117
|
+
// Hot-reload providers.json + config.json without restarting the server.
|
|
1118
|
+
// Auth-protected like the other admin endpoints.
|
|
1119
|
+
if (AUTH_KEY) {
|
|
1120
|
+
const apiKey = (req.headers.authorization || '').replace('Bearer ', '').trim();
|
|
1121
|
+
const keyFromQuery = parsedUrl.searchParams.get('key');
|
|
1122
|
+
if (apiKey !== AUTH_KEY && keyFromQuery !== AUTH_KEY) {
|
|
1123
|
+
res.writeHead(401, { 'Content-Type': 'application/json' });
|
|
1124
|
+
res.end(JSON.stringify({ error: { message: 'Invalid API key' } }));
|
|
1125
|
+
return;
|
|
1126
|
+
}
|
|
1127
|
+
}
|
|
1128
|
+
try {
|
|
1129
|
+
const result = reloadProviders();
|
|
1130
|
+
res.writeHead(200, { 'Content-Type': 'application/json' });
|
|
1131
|
+
res.end(JSON.stringify({ ok: true, ...result }));
|
|
1132
|
+
} catch (err) {
|
|
1133
|
+
res.writeHead(500, { 'Content-Type': 'application/json' });
|
|
1134
|
+
res.end(JSON.stringify({ error: { message: 'Reload failed: ' + err.message } }));
|
|
1135
|
+
}
|
|
1136
|
+
return;
|
|
1137
|
+
}
|
|
1138
|
+
|
|
1139
|
+
if (parsedUrl.pathname === '/v1/models-db' && req.method === 'GET') {
|
|
1140
|
+
// Структурированная база моделей: паспорта + статистика + топ по скору.
|
|
1141
|
+
if (AUTH_KEY) {
|
|
1142
|
+
const apiKey = (req.headers.authorization || '').replace('Bearer ', '').trim();
|
|
1143
|
+
const keyFromQuery = parsedUrl.searchParams.get('key');
|
|
1144
|
+
if (apiKey !== AUTH_KEY && keyFromQuery !== AUTH_KEY) {
|
|
1145
|
+
res.writeHead(401, { 'Content-Type': 'application/json' });
|
|
1146
|
+
res.end(JSON.stringify({ error: { message: 'Invalid API key' } }));
|
|
1147
|
+
return;
|
|
1148
|
+
}
|
|
1149
|
+
}
|
|
1150
|
+
try {
|
|
1151
|
+
const models = modelManager.db.all()
|
|
1152
|
+
.map(m => ({
|
|
1153
|
+
key: m.key, model: m.model, source: m.source, category: m.category,
|
|
1154
|
+
contextWindow: m.contextWindow, dailyLimit: m.dailyLimit,
|
|
1155
|
+
score: m.score || 0, status: m.status,
|
|
1156
|
+
lastCheckedAt: m.lastCheckedAt || null, lastOkAt: m.lastOkAt || null,
|
|
1157
|
+
}))
|
|
1158
|
+
.sort((a, b) => b.score - a.score);
|
|
1159
|
+
res.writeHead(200, { 'Content-Type': 'application/json' });
|
|
1160
|
+
res.end(JSON.stringify({
|
|
1161
|
+
stats: modelManager.db.stats(),
|
|
1162
|
+
top: models.slice(0, 10),
|
|
1163
|
+
models,
|
|
1164
|
+
manager: { enabled: MODEL_MANAGER_CONFIG.enabled !== false, intervalHours: modelManager.config.intervalHours, running: modelManager._running },
|
|
1165
|
+
}));
|
|
1166
|
+
} catch (err) {
|
|
1167
|
+
res.writeHead(500, { 'Content-Type': 'application/json' });
|
|
1168
|
+
res.end(JSON.stringify({ error: { message: err.message } }));
|
|
1169
|
+
}
|
|
1170
|
+
return;
|
|
1171
|
+
}
|
|
1172
|
+
|
|
1173
|
+
// Парсим /v1/models/{key}/toggle и /v1/models/{key}/test
|
|
1174
|
+
const modelActionMatch = parsedUrl.pathname.match(/^\/v1\/models\/([^/]+)\/(toggle|test)$/);
|
|
1175
|
+
if (modelActionMatch && req.method === 'POST') {
|
|
1176
|
+
if (AUTH_KEY) {
|
|
1177
|
+
const apiKey = (req.headers.authorization || '').replace('Bearer ', '').trim();
|
|
1178
|
+
const keyFromQuery = parsedUrl.searchParams.get('key');
|
|
1179
|
+
if (apiKey !== AUTH_KEY && keyFromQuery !== AUTH_KEY) {
|
|
1180
|
+
res.writeHead(401, { 'Content-Type': 'application/json' });
|
|
1181
|
+
res.end(JSON.stringify({ error: { message: 'Invalid API key' } }));
|
|
1182
|
+
return;
|
|
1183
|
+
}
|
|
1184
|
+
}
|
|
1185
|
+
const modelKey = decodeURIComponent(modelActionMatch[1]);
|
|
1186
|
+
const action = modelActionMatch[2];
|
|
1187
|
+
|
|
1188
|
+
if (action === 'test') {
|
|
1189
|
+
// Живой "hi"-тест: работает ли модель с нашим ключом. Без изменения каталога.
|
|
1190
|
+
const dbEntry = modelManager.db.get(modelKey);
|
|
1191
|
+
const prov = PROVIDERS[modelKey];
|
|
1192
|
+
const endpoint = (dbEntry && dbEntry.endpoint) || (prov && prov.endpoint);
|
|
1193
|
+
const model = (dbEntry && dbEntry.model) || (prov && prov.model);
|
|
1194
|
+
if (!endpoint || !model) {
|
|
1195
|
+
res.writeHead(404, { 'Content-Type': 'application/json' });
|
|
1196
|
+
res.end(JSON.stringify({ ok: false, error: 'Модель ' + modelKey + ' не найдена' }));
|
|
1197
|
+
return;
|
|
1198
|
+
}
|
|
1199
|
+
const source = (dbEntry && dbEntry.source) || 'unknown';
|
|
1200
|
+
const apiKey = modelManager.apiKeyFor(source);
|
|
1201
|
+
try {
|
|
1202
|
+
const t0 = Date.now();
|
|
1203
|
+
const r = await fetch(endpoint, {
|
|
1204
|
+
method: 'POST',
|
|
1205
|
+
headers: { 'Content-Type': 'application/json', ...(apiKey ? { Authorization: 'Bearer ' + apiKey } : {}) },
|
|
1206
|
+
body: JSON.stringify({ model, messages: [{ role: 'user', content: 'hi' }], max_tokens: 5 }),
|
|
1207
|
+
signal: AbortSignal.timeout(20000),
|
|
1208
|
+
});
|
|
1209
|
+
const latencyMs = Date.now() - t0;
|
|
1210
|
+
// Обновляем базу результатом проверки (но не «активируем» принудительно).
|
|
1211
|
+
modelManager.db.markChecked(modelKey, { ok: r.ok, status: r.status, latencyMs }, { now: Date.now() });
|
|
1212
|
+
modelManager.db.save();
|
|
1213
|
+
res.writeHead(200, { 'Content-Type': 'application/json' });
|
|
1214
|
+
res.end(JSON.stringify({ ok: r.ok, status: r.status, latencyMs, model: { key: modelKey, status: modelManager.db.get(modelKey).status } }));
|
|
1215
|
+
} catch (err) {
|
|
1216
|
+
res.writeHead(200, { 'Content-Type': 'application/json' });
|
|
1217
|
+
res.end(JSON.stringify({ ok: false, status: 0, error: err.message }));
|
|
1218
|
+
}
|
|
1219
|
+
return;
|
|
1220
|
+
}
|
|
1221
|
+
|
|
1222
|
+
// action === 'toggle': вкл/выкл провайдера в config.json (только этот ключ) + hot-reload.
|
|
1223
|
+
try {
|
|
1224
|
+
const userCfg = JSON.parse(fs.readFileSync(CONFIG_PATH, 'utf8'));
|
|
1225
|
+
if (!userCfg.providers) userCfg.providers = {};
|
|
1226
|
+
if (!userCfg.providers[modelKey]) userCfg.providers[modelKey] = {};
|
|
1227
|
+
// Flip: было включено → выключить, было выключено → включить.
|
|
1228
|
+
const currentlyEnabled = !(userCfg.providers[modelKey].enabled === false);
|
|
1229
|
+
const newEnabled = !currentlyEnabled;
|
|
1230
|
+
userCfg.providers[modelKey].enabled = newEnabled;
|
|
1231
|
+
fs.writeFileSync(CONFIG_PATH, JSON.stringify(userCfg, null, 2));
|
|
1232
|
+
reloadProviders();
|
|
1233
|
+
// Статус в базе: включили → untested (проверится в след. цикле), выключили → user-disabled.
|
|
1234
|
+
const entry = modelManager.db.get(modelKey);
|
|
1235
|
+
if (entry) {
|
|
1236
|
+
modelManager.db.setStatus(modelKey, newEnabled ? 'untested' : 'user-disabled');
|
|
1237
|
+
modelManager.db.save();
|
|
1238
|
+
}
|
|
1239
|
+
res.writeHead(200, { 'Content-Type': 'application/json' });
|
|
1240
|
+
res.end(JSON.stringify({ ok: true, key: modelKey, enabled: newEnabled, model: entry ? { key: modelKey, status: modelManager.db.get(modelKey).status } : null }));
|
|
1241
|
+
} catch (err) {
|
|
1242
|
+
res.writeHead(500, { 'Content-Type': 'application/json' });
|
|
1243
|
+
res.end(JSON.stringify({ error: { message: 'Toggle failed: ' + err.message } }));
|
|
1244
|
+
}
|
|
1245
|
+
return;
|
|
1246
|
+
}
|
|
1247
|
+
|
|
718
1248
|
if (parsedUrl.pathname === '/health') {
|
|
719
1249
|
const h = getHealth();
|
|
720
1250
|
const upCount = Object.values(h).filter(v => v.status === 'up').length;
|
|
721
1251
|
const totalCount = Object.entries(PROVIDERS).filter(([_, p]) => p.enabled).length;
|
|
1252
|
+
const contextSummary = (() => { try { return contextStats.summary(); } catch { return null; } })();
|
|
722
1253
|
res.writeHead(upCount > 0 ? 200 : 503, { 'Content-Type': 'application/json' });
|
|
723
|
-
res.end(JSON.stringify({ status: upCount > 0 ? 'ok' : 'degraded', providers: { up: upCount, total: totalCount } }));
|
|
1254
|
+
res.end(JSON.stringify({ status: upCount > 0 ? 'ok' : 'degraded', providers: { up: upCount, total: totalCount }, context: contextSummary }));
|
|
724
1255
|
return;
|
|
725
1256
|
}
|
|
726
1257
|
|
|
@@ -738,6 +1269,14 @@ const server = http.createServer(async (req, res) => {
|
|
|
738
1269
|
res.end(JSON.stringify({
|
|
739
1270
|
total_requests: s.totalRequests, successful_requests: s.successfulRequests, failed_requests: s.failedRequests,
|
|
740
1271
|
provider_usage: s.providerUsage, token_usage: s.tokenUsage, errors: s.errors, uptime_seconds: Math.floor((Date.now() - s.startTime) / 1000),
|
|
1272
|
+
today: (() => {
|
|
1273
|
+
const r = getReliability();
|
|
1274
|
+
let success = 0, fail = 0;
|
|
1275
|
+
for (const [_, v] of Object.entries(r)) { if (v && v.day === today) { success += v.success || 0; fail += v.fail || 0; } }
|
|
1276
|
+
const total = success + fail;
|
|
1277
|
+
return { requests: total, success, failed: fail, successRate: total > 0 ? Math.round((success / total) * 100) : 0 };
|
|
1278
|
+
})(),
|
|
1279
|
+
savings: (() => { try { return aggregateSavings(s.tokenUsage || {}); } catch { return null; } })(),
|
|
741
1280
|
health: Object.fromEntries(Object.entries(getHealth()).map(([k, v]) => {
|
|
742
1281
|
const limit = limits[k];
|
|
743
1282
|
const err = s.errors[k] || 0;
|
|
@@ -756,6 +1295,7 @@ const server = http.createServer(async (req, res) => {
|
|
|
756
1295
|
pool: poolStats(),
|
|
757
1296
|
last_selection: getLastSelection(),
|
|
758
1297
|
bandit: getBandit(),
|
|
1298
|
+
context_summary: (() => { try { return contextStats.summary(); } catch { return null; } })(),
|
|
759
1299
|
}));
|
|
760
1300
|
return;
|
|
761
1301
|
}
|
|
@@ -836,6 +1376,25 @@ const server = http.createServer(async (req, res) => {
|
|
|
836
1376
|
return;
|
|
837
1377
|
}
|
|
838
1378
|
|
|
1379
|
+
// POST /v1/cache/clear — drop the in-memory semantic cache without a restart.
|
|
1380
|
+
// Handy when you tweaked providers/models and don't want stale answers served.
|
|
1381
|
+
if (parsedUrl.pathname === '/v1/cache/clear' && req.method === 'POST') {
|
|
1382
|
+
if (AUTH_KEY) {
|
|
1383
|
+
const apiKey = (req.headers.authorization || '').replace('Bearer ', '').trim();
|
|
1384
|
+
const keyFromQuery = parsedUrl.searchParams.get('key');
|
|
1385
|
+
if (apiKey !== AUTH_KEY && keyFromQuery !== AUTH_KEY) {
|
|
1386
|
+
res.writeHead(401, { 'Content-Type': 'application/json' });
|
|
1387
|
+
res.end(JSON.stringify({ error: { message: 'Invalid API key' } }));
|
|
1388
|
+
return;
|
|
1389
|
+
}
|
|
1390
|
+
}
|
|
1391
|
+
const before = cache.stats().size || 0;
|
|
1392
|
+
cache.clear();
|
|
1393
|
+
res.writeHead(200, { 'Content-Type': 'application/json' });
|
|
1394
|
+
res.end(JSON.stringify({ ok: true, cleared: before }));
|
|
1395
|
+
return;
|
|
1396
|
+
}
|
|
1397
|
+
|
|
839
1398
|
// POST /v1/shorts — generate a vertical short video via the tools generator.
|
|
840
1399
|
// Body: { prompt, duration?, format? ("9:16"/"16:9"/"1:1"), steps? }
|
|
841
1400
|
if (parsedUrl.pathname === '/v1/shorts' && req.method === 'POST') {
|
|
@@ -889,6 +1448,87 @@ const server = http.createServer(async (req, res) => {
|
|
|
889
1448
|
return;
|
|
890
1449
|
}
|
|
891
1450
|
|
|
1451
|
+
// --- Setup Dashboard API ---
|
|
1452
|
+
if (parsedUrl.pathname === '/v1/setup/keys' && req.method === 'GET') {
|
|
1453
|
+
const { keys } = readKeys();
|
|
1454
|
+
// Mask keys for display; inputs stay EMPTY so we never send masked values back.
|
|
1455
|
+
const masked = {};
|
|
1456
|
+
const empty = {};
|
|
1457
|
+
for (const [k, v] of Object.entries(keys)) {
|
|
1458
|
+
empty[k] = '';
|
|
1459
|
+
if (!v) { masked[k] = ''; continue; }
|
|
1460
|
+
if (v.length <= 10) { masked[k] = v.slice(0, 2) + '***' + v.slice(-2); continue; }
|
|
1461
|
+
masked[k] = v.slice(0, 4) + '***' + v.slice(-4);
|
|
1462
|
+
}
|
|
1463
|
+
res.writeHead(200, { 'Content-Type': 'application/json' });
|
|
1464
|
+
res.end(JSON.stringify({ groups: KEY_GROUPS, keys: empty, masked }));
|
|
1465
|
+
return;
|
|
1466
|
+
}
|
|
1467
|
+
|
|
1468
|
+
if (parsedUrl.pathname === '/v1/setup/keys' && req.method === 'POST') {
|
|
1469
|
+
if (AUTH_KEY) {
|
|
1470
|
+
const apiKey = (req.headers.authorization || '').replace('Bearer ', '').trim();
|
|
1471
|
+
const keyFromQuery = parsedUrl.searchParams.get('key');
|
|
1472
|
+
if (apiKey !== AUTH_KEY && keyFromQuery !== AUTH_KEY) {
|
|
1473
|
+
res.writeHead(401, { 'Content-Type': 'application/json' });
|
|
1474
|
+
res.end(JSON.stringify({ error: { message: 'Invalid API key' } }));
|
|
1475
|
+
return;
|
|
1476
|
+
}
|
|
1477
|
+
}
|
|
1478
|
+
let body = '';
|
|
1479
|
+
req.on('data', (c) => body += c);
|
|
1480
|
+
req.on('end', () => {
|
|
1481
|
+
try {
|
|
1482
|
+
const newKeys = JSON.parse(body);
|
|
1483
|
+
// Only accept known env vars
|
|
1484
|
+
const filtered = {};
|
|
1485
|
+
for (const k of Object.keys(KEY_GROUPS)) {
|
|
1486
|
+
if (typeof newKeys[k] === 'string') filtered[k] = newKeys[k];
|
|
1487
|
+
}
|
|
1488
|
+
const result = saveKeys(filtered);
|
|
1489
|
+
res.writeHead(200, { 'Content-Type': 'application/json' });
|
|
1490
|
+
res.end(JSON.stringify({ ok: true, ...result }));
|
|
1491
|
+
} catch (e) {
|
|
1492
|
+
res.writeHead(400, { 'Content-Type': 'application/json' });
|
|
1493
|
+
res.end(JSON.stringify({ error: { message: 'Invalid request: ' + e.message } }));
|
|
1494
|
+
}
|
|
1495
|
+
});
|
|
1496
|
+
return;
|
|
1497
|
+
}
|
|
1498
|
+
|
|
1499
|
+
if (parsedUrl.pathname === '/v1/setup/validate') {
|
|
1500
|
+
if (AUTH_KEY) {
|
|
1501
|
+
const apiKey = (req.headers.authorization || '').replace('Bearer ', '').trim();
|
|
1502
|
+
const keyFromQuery = parsedUrl.searchParams.get('key');
|
|
1503
|
+
if (apiKey !== AUTH_KEY && keyFromQuery !== AUTH_KEY) {
|
|
1504
|
+
res.writeHead(401, { 'Content-Type': 'application/json' });
|
|
1505
|
+
res.end(JSON.stringify({ error: { message: 'Invalid API key' } }));
|
|
1506
|
+
return;
|
|
1507
|
+
}
|
|
1508
|
+
}
|
|
1509
|
+
const envVar = parsedUrl.searchParams.get('envVar');
|
|
1510
|
+
let testKey = parsedUrl.searchParams.get('apiKey');
|
|
1511
|
+
if (!envVar) {
|
|
1512
|
+
res.writeHead(400, { 'Content-Type': 'application/json' });
|
|
1513
|
+
res.end(JSON.stringify({ error: { message: 'envVar required' } }));
|
|
1514
|
+
return;
|
|
1515
|
+
}
|
|
1516
|
+
// If apiKey param is empty/absent, validate the real stored key from .env.
|
|
1517
|
+
if (!testKey) {
|
|
1518
|
+
testKey = getStoredKey(envVar);
|
|
1519
|
+
}
|
|
1520
|
+
if (!testKey) {
|
|
1521
|
+
res.writeHead(200, { 'Content-Type': 'application/json' });
|
|
1522
|
+
res.end(JSON.stringify({ valid: false, error: 'Нет сохранённого ключа' }));
|
|
1523
|
+
return;
|
|
1524
|
+
}
|
|
1525
|
+
validateKey(envVar, testKey).then((result) => {
|
|
1526
|
+
res.writeHead(200, { 'Content-Type': 'application/json' });
|
|
1527
|
+
res.end(JSON.stringify(result));
|
|
1528
|
+
});
|
|
1529
|
+
return;
|
|
1530
|
+
}
|
|
1531
|
+
|
|
892
1532
|
res.writeHead(404, { 'Content-Type': 'application/json' });
|
|
893
1533
|
res.end(JSON.stringify({ error: 'Not found' }));
|
|
894
1534
|
});
|
|
@@ -896,7 +1536,19 @@ const server = http.createServer(async (req, res) => {
|
|
|
896
1536
|
server.listen(PORT, process.env.HOST || '127.0.0.1', () => {
|
|
897
1537
|
logger.info('Freegate started', { port: PORT });
|
|
898
1538
|
console.log('Dashboard: http://localhost:' + PORT + '/');
|
|
1539
|
+
// Самообновляющаяся база моделей: первый цикл через 2 мин, далее по интервалу.
|
|
1540
|
+
if (MODEL_MANAGER_CONFIG.enabled !== false) {
|
|
1541
|
+
modelManager.start();
|
|
1542
|
+
logger.info('ModelManager started', { intervalHours: modelManager.config.intervalHours });
|
|
1543
|
+
}
|
|
899
1544
|
});
|
|
900
1545
|
|
|
901
|
-
|
|
902
|
-
|
|
1546
|
+
const _shutdown = () => {
|
|
1547
|
+
if (memStore) { memStore.stopTimer(); memStore.save(); }
|
|
1548
|
+
require('./lib/health').saveState();
|
|
1549
|
+
cache.persist();
|
|
1550
|
+
try { modelManager.stop(); } catch {}
|
|
1551
|
+
server.close(() => process.exit(0));
|
|
1552
|
+
};
|
|
1553
|
+
process.on('SIGINT', _shutdown);
|
|
1554
|
+
process.on('SIGTERM', _shutdown);
|