@arcaneorion/dsh-model-channel-manager 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/index.js ADDED
@@ -0,0 +1,903 @@
1
+ import z from "@deepseek-ai/schemastery";
2
+ export const name = 'model-channel-manager';
3
+ export const inject = ['llm', 'timer'];
4
+ const ROUTE_PREFIX = 'roundrobin/';
5
+ const GROUP_ID_RE = /^[a-z0-9][a-z0-9-]*$/;
6
+ const NS_CONFIG = 'model-channels';
7
+ const NS_HEALTH = 'model-channel-health';
8
+ const CONFIG_SCHEMA = z.object({
9
+ groups: z.array(z.object({
10
+ id: z.string(),
11
+ }).loose(true)).default([]),
12
+ // 面板供应商顺序的持久化载体:settings-file 的 patchNode 对 map 键序是盲的
13
+ // (纯重排 = 零 diff),但数组是 deepEqual 不等即整值 setIn(原位、保序)。
14
+ // 顺序存成数组数据而不是对象键序元数据,重启后从文件真实还原。引擎不消费它。
15
+ providerOrder: z.array(z.string()).default([]),
16
+ }).loose(true);
17
+ const HEALTH_SCHEMA = z.object({
18
+ records: z.dict(z.array(z.any())).default({}),
19
+ speedResults: z.dict(z.array(z.any())).default({}),
20
+ runtime: z.dict(z.any()).default({}),
21
+ speedRequest: z.any(),
22
+ lastHandledNonce: z.number().default(0),
23
+ testRequest: z.any(),
24
+ testResults: z.dict(z.any()).default({}),
25
+ lastTestHandledNonce: z.number().default(0),
26
+ legacyMigrated: z.any(),
27
+ }).loose(true);
28
+ export function apply(ctx) {
29
+ const isContentChunk = (c) => {
30
+ if (!c) return false;
31
+ if (c.type === 'text-delta' || c.type === 'reasoning-delta') return c.text !== '';
32
+ if (c.type === 'tool-call-delta') return c.argumentsDelta !== '' || c.name !== undefined;
33
+ return false;
34
+ };
35
+ const isTerminalChunk = (c) => c && c.type === 'finish';
36
+ const isSuccessReason = (r) => r && (r.kind === 'stop' || r.kind === 'max-tokens' || r.kind === 'tool-calls');
37
+ const failChunk = (message, code) => ({ type: 'finish', reason: { kind: 'error', failure: { message, code } } });
38
+ const abortedChunk = (message) => ({ type: 'finish', reason: { kind: 'aborted', failure: { message, code: 'ABORTED' } } });
39
+ const candKey = (c) => c.provider + '::' + c.model;
40
+ const clone = (v) => (v == null ? v : JSON.parse(JSON.stringify(v)));
41
+ const groupOfRoute = (p) => (p != null && p.startsWith(ROUTE_PREFIX) ? p.slice(ROUTE_PREFIX.length) : null);
42
+ const routeOfGroup = (id) => ROUTE_PREFIX + id;
43
+ const runtime = new Map();
44
+ const state = { config: { groups: [] }, records: {}, speedResults: {}, testResults: {} };
45
+ const groupRuntime = (id) => {
46
+ let g = runtime.get(id);
47
+ if (g === undefined) {
48
+ g = { currentIndex: 0, cooldowns: new Map(), lastSpeedTestAt: 0, speedTestRunning: false, events: [] };
49
+ runtime.set(id, g);
50
+ }
51
+ return g;
52
+ };
53
+ // ---------- 可取消超时(避免 Promise.race 留下孤儿定时器) ----------
54
+ const timed = (ms, makeError) => {
55
+ let rejectFn;
56
+ const promise = new Promise((_, reject) => { rejectFn = reject; });
57
+ const dispose = ctx.effect(() => {
58
+ const timer = setTimeout(() => rejectFn(makeError()), ms);
59
+ return () => clearTimeout(timer);
60
+ }, 'model-channel timeout guard');
61
+ return { promise, dispose };
62
+ };
63
+ // ---------- 配置归一化(与动态原型逐行一致) ----------
64
+ function normalizeConfig(raw) {
65
+ const groups = Array.isArray(raw && raw.groups) ? raw.groups : [];
66
+ const out = [];
67
+ const seen = new Set();
68
+ for (const rg of groups) {
69
+ const id = rg && typeof rg.id === 'string' ? rg.id : null;
70
+ if (id === null || !GROUP_ID_RE.test(id) || seen.has(id))
71
+ continue;
72
+ seen.add(id);
73
+ const vm = rg.virtualModel || {};
74
+ const st = rg.speedTest || {};
75
+ out.push({
76
+ id,
77
+ virtualModel: {
78
+ name: typeof vm.name === 'string' && vm.name.length > 0 ? vm.name : id,
79
+ reasoning: vm.reasoning !== false,
80
+ input: Array.isArray(vm.input) && vm.input.length > 0 && vm.input.every((x) => x === 'text' || x === 'image') ? vm.input.slice() : ['text'],
81
+ contextWindow: Number.isSafeInteger(vm.contextWindow) && vm.contextWindow > 0 ? vm.contextWindow : 200000,
82
+ maxTokens: Number.isSafeInteger(vm.maxTokens) && vm.maxTokens > 0 ? vm.maxTokens : 16384,
83
+ },
84
+ candidates: Array.isArray(rg.candidates)
85
+ ? rg.candidates.filter((c) => c && typeof c.provider === 'string' && c.provider.length > 0 && typeof c.model === 'string' && c.model.length > 0).map((c) => ({ provider: c.provider, model: c.model }))
86
+ : [],
87
+ presets: Array.isArray(rg.presets)
88
+ ? rg.presets.filter((p) => p && typeof p.id === 'string' && Array.isArray(p.candidates)).map((p) => ({
89
+ id: p.id, name: typeof p.name === 'string' ? p.name : p.id,
90
+ candidates: p.candidates.filter((c) => c && typeof c.provider === 'string' && typeof c.model === 'string').map((c) => ({ provider: c.provider, model: c.model })),
91
+ }))
92
+ : [],
93
+ activePreset: typeof rg.activePreset === 'string' ? rg.activePreset : null,
94
+ strategy: ['sticky', 'round-robin', 'primary'].includes(rg.strategy) ? rg.strategy : 'sticky',
95
+ timeoutMs: Number.isFinite(rg.timeoutMs) && rg.timeoutMs > 0 ? rg.timeoutMs : 30000,
96
+ cooldownMs: Number.isFinite(rg.cooldownMs) && rg.cooldownMs >= 0 ? rg.cooldownMs : 60000,
97
+ maxRetriesPerCandidate: Number.isSafeInteger(rg.maxRetriesPerCandidate) && rg.maxRetriesPerCandidate >= 0 ? rg.maxRetriesPerCandidate : 2,
98
+ speedTest: {
99
+ enabled: st.enabled !== false,
100
+ sortKey: ['ttft', 'latency', 'hybrid', 'smart'].includes(st.sortKey) ? st.sortKey : 'ttft',
101
+ prompt: typeof st.prompt === 'string' && st.prompt.length > 0 ? st.prompt : '欧拉函数的意义?',
102
+ maxTokens: Number.isFinite(st.maxTokens) && st.maxTokens > 0 ? st.maxTokens : 2048,
103
+ timeoutMs: Number.isFinite(st.timeoutMs) && st.timeoutMs > 0 ? st.timeoutMs : 60000,
104
+ concurrency: Number.isSafeInteger(st.concurrency) && st.concurrency > 0 ? st.concurrency : 3,
105
+ retries: Number.isSafeInteger(st.retries) && st.retries >= 0 ? st.retries : 2,
106
+ onFirstUse: st.onFirstUse === true,
107
+ },
108
+ });
109
+ }
110
+ return { groups: out };
111
+ }
112
+ function pullConfig() {
113
+ return normalizeConfig(state.config);
114
+ }
115
+ function pullRecords() { return state.records; }
116
+ function pullSpeedResults() { return state.speedResults; }
117
+ // ---------- 健康聚合 ----------
118
+ function healthAggregate() {
119
+ const recordsMap = pullRecords();
120
+ // 聚合所有真实渠道(provider::model)的请求健康流水;记录由全局 llm/stream 拦截器按 provider 键写入
121
+ const allEvents = Object.values(recordsMap).flat();
122
+ const by = new Map();
123
+ for (const e of allEvents) {
124
+ const key = e.provider + '::' + e.model;
125
+ let agg = by.get(key);
126
+ if (agg === undefined) {
127
+ agg = { provider: e.provider, model: e.model, total: 0, success: 0, ttftSum: 0, latSum: 0, lastTs: 0, lastCode: null, lastOk: null };
128
+ by.set(key, agg);
129
+ }
130
+ agg.total++;
131
+ if (e.ok)
132
+ agg.success++;
133
+ if (e.ttftMs != null)
134
+ agg.ttftSum += e.ttftMs;
135
+ if (e.latencyMs != null)
136
+ agg.latSum += e.latencyMs;
137
+ if (e.ts > agg.lastTs) {
138
+ agg.lastTs = e.ts;
139
+ agg.lastCode = e.code || null;
140
+ agg.lastOk = e.ok;
141
+ }
142
+ }
143
+ return [...by.values()].map((a) => ({
144
+ provider: a.provider, model: a.model, total: a.total, success: a.success,
145
+ successRate: a.total > 0 ? a.success / a.total : null,
146
+ reliability: (a.success + 2.5) / (a.total + 5),
147
+ avgTtftMs: a.success > 0 ? a.ttftSum / a.success : null,
148
+ avgLatencyMs: a.success > 0 ? a.latSum / a.success : null,
149
+ lastTs: a.lastTs, lastCode: a.lastCode, lastOk: a.lastOk,
150
+ }));
151
+ }
152
+ function activeCandidates(cfg) {
153
+ if (cfg.activePreset) {
154
+ const preset = cfg.presets.find((p) => p.id === cfg.activePreset);
155
+ if (preset && preset.candidates.length > 0)
156
+ return preset.candidates;
157
+ }
158
+ return cfg.candidates;
159
+ }
160
+ function speedResultOf(cfg, key) {
161
+ const r = (pullSpeedResults()[cfg.id] || []).find((s) => candKey(s) === key);
162
+ return r && r.ok ? r : null;
163
+ }
164
+ function dynamicTimeoutMs(cfg, key) {
165
+ const ttft = cfg.speedTest.enabled ? (speedResultOf(cfg, key) || {}).ttft : null;
166
+ if (ttft && ttft > 0)
167
+ return Math.max(cfg.timeoutMs, Math.min(120000, Math.round(ttft * 2)));
168
+ return cfg.timeoutMs;
169
+ }
170
+ function median(nums) {
171
+ if (!nums || nums.length === 0)
172
+ return null;
173
+ const s = nums.slice().sort((a, b) => a - b);
174
+ const m = Math.floor(s.length / 2);
175
+ return s.length % 2 === 0 ? (s[m - 1] + s[m]) / 2 : s[m];
176
+ }
177
+ function orderedCandidates(cfg) {
178
+ const rt = groupRuntime(cfg.id);
179
+ const base = activeCandidates(cfg);
180
+ const aggList = healthAggregate();
181
+ const speedRows = pullSpeedResults()[cfg.id] || [];
182
+ const entries = base.map((cand) => {
183
+ const key = candKey(cand);
184
+ return { cand, key, speed: speedRows.find((s) => candKey(s) === key), agg: aggList.find((a) => a.provider === cand.provider && a.model === cand.model) };
185
+ });
186
+ let scored;
187
+ if (cfg.speedTest.enabled && speedRows.length > 0) {
188
+ const measured = entries.filter((e) => e.speed && e.speed.ok && (e.speed.ttft != null || e.speed.latency != null));
189
+ const ttfts = measured.map((e) => e.speed.ttft).filter((v) => v != null);
190
+ const lats = measured.map((e) => e.speed.latency).filter((v) => v != null);
191
+ const ttftMedian = median(ttfts);
192
+ const latMedian = median(lats);
193
+ const ttftRange = ttfts.length > 1 ? Math.max(...ttfts) - Math.min(...ttfts) : 0;
194
+ const latRange = lats.length > 1 ? Math.max(...lats) - Math.min(...lats) : 0;
195
+ scored = entries.map((e) => {
196
+ if (e.speed && e.speed.ok && (e.speed.ttft != null || e.speed.latency != null)) {
197
+ const ttft = e.speed.ttft != null ? e.speed.ttft : ttftMedian;
198
+ const lat = e.speed.latency != null ? e.speed.latency : latMedian;
199
+ const reliability = e.agg ? e.agg.reliability : 0.5;
200
+ let score;
201
+ if (cfg.speedTest.sortKey === 'latency')
202
+ score = lat;
203
+ else if (cfg.speedTest.sortKey === 'hybrid')
204
+ score = 0.7 * ttft + 0.3 * lat;
205
+ else if (cfg.speedTest.sortKey === 'smart')
206
+ score = 0.5 * (ttft / (ttftRange || 1)) + 0.3 * (1 - reliability) + 0.2 * (lat / (latRange || 1));
207
+ else
208
+ score = ttft;
209
+ return { ...e, score, failed: false };
210
+ }
211
+ return { ...e, score: Infinity, failed: true };
212
+ });
213
+ }
214
+ else {
215
+ scored = entries.map((e, i) => ({ ...e, score: i, failed: false }));
216
+ }
217
+ const out = scored.slice();
218
+ out.sort((a, b) => {
219
+ if (a.failed !== b.failed)
220
+ return a.failed ? 1 : -1;
221
+ if (a.score !== b.score)
222
+ return a.score - b.score;
223
+ return entries.indexOf(a) - entries.indexOf(b);
224
+ });
225
+ return out.map((e) => e.cand);
226
+ }
227
+ function pushEvent(cfg, kind, payload) {
228
+ const rt = groupRuntime(cfg.id);
229
+ rt.events.push(Object.assign({ ts: Date.now(), group: cfg.id, kind }, payload || {}));
230
+ if (rt.events.length > 200)
231
+ rt.events.splice(0, rt.events.length - 200);
232
+ }
233
+ // ---------- 持久化:settings 总线 ----------
234
+ const persistRuntime = async () => {
235
+ const out = {};
236
+ for (const [id, rt] of runtime) {
237
+ const now = Date.now();
238
+ out[id] = {
239
+ currentIndex: rt.currentIndex,
240
+ cooldowns: Array.from(rt.cooldowns.entries()).map(([key, until]) => ({ key, until, active: until > now })),
241
+ events: rt.events.slice(-40),
242
+ lastSpeedTestAt: rt.lastSpeedTestAt,
243
+ };
244
+ }
245
+ return out;
246
+ };
247
+ // ---------- 运行时态恢复(重启后还原冷却与 sticky 指针) ----------
248
+ function restoreRuntime(saved) {
249
+ if (!saved || typeof saved !== 'object') return;
250
+ const now = Date.now();
251
+ for (const [id, data] of Object.entries(saved)) {
252
+ if (!data || typeof data !== 'object') continue;
253
+ const rt = groupRuntime(id);
254
+ if (Number.isSafeInteger(data.currentIndex)) rt.currentIndex = data.currentIndex;
255
+ if (Number.isSafeInteger(data.lastSpeedTestAt)) rt.lastSpeedTestAt = data.lastSpeedTestAt;
256
+ if (Array.isArray(data.cooldowns)) {
257
+ for (const c of data.cooldowns) {
258
+ if (c && typeof c.key === 'string' && Number.isFinite(c.until) && c.until > now) {
259
+ rt.cooldowns.set(c.key, c.until);
260
+ }
261
+ }
262
+ }
263
+ }
264
+ }
265
+ const recordHealth = (gid, cand, entry) => {
266
+ const now = Date.now();
267
+ const key = gid || 'global';
268
+ const list = state.records[key] || (state.records[key] = []);
269
+ list.push({ ts: now, provider: cand.provider, model: cand.model, ok: entry.ok, ttftMs: entry.ttftMs, latencyMs: entry.latencyMs, code: entry.code || null });
270
+ const cutoff = now - 7 * 24 * 3600 * 1000;
271
+ state.records[key] = list.filter((e) => e.ts >= cutoff).slice(-2000);
272
+ persistHealth();
273
+ };
274
+ const persistSpeedResults = async (r) => {
275
+ if (bus === null) return;
276
+ await bus.settings.update(NS_HEALTH, { speedResults: clone(r), runtime: await persistRuntime() });
277
+ };
278
+ // ---------- 健康持久化节流:合并 2s 窗口内的写入,避免每个请求全量重写 settings ----------
279
+ let healthFlushHandle = null;
280
+ let healthDirty = false;
281
+ const flushHealthNow = async () => {
282
+ if (bus === null) return;
283
+ await bus.settings.update(NS_HEALTH, { records: clone(pullRecords()), speedResults: clone(pullSpeedResults()), runtime: await persistRuntime() });
284
+ };
285
+ const persistHealth = () => {
286
+ healthDirty = true;
287
+ if (healthFlushHandle === null) {
288
+ healthFlushHandle = ctx.timeout(() => {
289
+ healthFlushHandle = null;
290
+ if (!healthDirty) return;
291
+ healthDirty = false;
292
+ flushHealthNow().catch(() => { });
293
+ }, 2000);
294
+ }
295
+ };
296
+ ctx.effect(() => () => {
297
+ // 插件卸载时尽力冲刷最后一批健康数据(settings 服务仍在,失败静默)
298
+ if (healthFlushHandle !== null) {
299
+ healthFlushHandle();
300
+ healthFlushHandle = null;
301
+ }
302
+ if (healthDirty) {
303
+ healthDirty = false;
304
+ flushHealthNow().catch(() => { });
305
+ }
306
+ }, 'model-channel health flush');
307
+ // ---------- 引擎:单候选尝试 ----------
308
+ async function* streamAttempt(cfg, cand, options, attemptStart) {
309
+ if (cand.provider.startsWith(ROUTE_PREFIX))
310
+ throw { code: 'INVALID_CANDIDATE', message: 'candidate provider must not be a virtual route' };
311
+ const llm = ctx.llm;
312
+ const inner = llm.stream(Object.assign({}, options, { provider: cand.provider, model: cand.model }));
313
+ const timeoutMs = dynamicTimeoutMs(cfg, candKey(cand));
314
+ let buffer = [];
315
+ let started = false;
316
+ let ttftMs = null;
317
+ let lastAt = Date.now();
318
+ const closeInner = async () => { try {
319
+ const c = inner.return ? inner.return() : null;
320
+ if (c && c.then)
321
+ await c;
322
+ }
323
+ catch (_e) { } };
324
+ try {
325
+ while (true) {
326
+ if (options.signal && options.signal.aborted)
327
+ throw { code: 'ABORTED', message: 'request aborted' };
328
+ const remaining = Math.max(0, timeoutMs - (Date.now() - (started ? lastAt : attemptStart)));
329
+ if (remaining <= 0)
330
+ throw { code: 'TIMEOUT', message: 'channel candidate timed out after ' + timeoutMs + 'ms' };
331
+ // guard 覆盖全量 remaining:不能 cap 到 30s,否则 timeoutMs>30s 的首响应
332
+ // 超时与动态超时(min(120s, ttft×2))在 >30s 区间全部退化为 30s 切候选
333
+ const guard = timed(Math.max(1, remaining), () => ({ code: 'TIMEOUT', message: 'channel candidate timed out after ' + timeoutMs + 'ms' }));
334
+ let next;
335
+ try {
336
+ next = await Promise.race([inner.next(), guard.promise]);
337
+ }
338
+ finally {
339
+ // 超时路径同样要 dispose:ctx.effect 的注册表条目只有显式调用 disposer 才会移除
340
+ guard.dispose();
341
+ }
342
+ lastAt = Date.now();
343
+ const chunk = next.value;
344
+ if (chunk === undefined)
345
+ throw { code: 'STREAM_CLOSED', message: 'channel stream ended without a terminal chunk' };
346
+ if (!started) {
347
+ if (isContentChunk(chunk)) {
348
+ started = true;
349
+ ttftMs = Date.now() - attemptStart;
350
+ for (const b of buffer)
351
+ yield b;
352
+ yield chunk;
353
+ }
354
+ else if (isTerminalChunk(chunk)) {
355
+ const reason = chunk.reason || {};
356
+ // 引擎提前终止的内层流不会被全局拦截器完整排水记账,这里自行记录
357
+ if (isSuccessReason(reason)) {
358
+ recordHealth(cand.provider, cand, { ok: false, ttftMs: null, latencyMs: null, code: 'EMPTY_RESPONSE' });
359
+ throw { code: 'EMPTY_RESPONSE', message: 'model returned a completed response with no content' };
360
+ }
361
+ const preCode = (reason.failure && reason.failure.code) || 'STREAM_ERROR';
362
+ recordHealth(cand.provider, cand, { ok: false, ttftMs: null, latencyMs: null, code: preCode });
363
+ throw { code: preCode, message: (reason.failure && reason.failure.message) || 'candidate stream failed' };
364
+ }
365
+ else {
366
+ buffer.push(chunk);
367
+ }
368
+ }
369
+ else {
370
+ if (isTerminalChunk(chunk)) {
371
+ const reason = chunk.reason || {};
372
+ if (isSuccessReason(reason)) {
373
+ recordHealth(cand.provider, cand, { ok: true, ttftMs, latencyMs: Date.now() - attemptStart, code: null });
374
+ yield chunk;
375
+ return;
376
+ }
377
+ // 内容已向下游输出:绝不能再下发该失败终止块(否则出现 finish 后继续输出/双 finish),
378
+ // 改为抛错并标记 emitted,由组层直接终结本次请求
379
+ const midCode = (reason.failure && reason.failure.code) || 'STREAM_ERROR';
380
+ recordHealth(cand.provider, cand, { ok: false, ttftMs, latencyMs: null, code: midCode });
381
+ throw { code: midCode, message: (reason.failure && reason.failure.message) || 'candidate stream failed mid-stream' };
382
+ }
383
+ yield chunk;
384
+ }
385
+ }
386
+ }
387
+ catch (err) {
388
+ await closeInner();
389
+ if (started && err && typeof err === 'object')
390
+ err.emitted = true;
391
+ if (err && err.code === 'TIMEOUT')
392
+ recordHealth(cand.provider, cand, { ok: false, ttftMs, latencyMs: null, code: 'TIMEOUT' });
393
+ throw err;
394
+ }
395
+ }
396
+ // ---------- 引擎:组级故障转移循环 ----------
397
+ async function* streamGroup(cfg, options) {
398
+ const rt = groupRuntime(cfg.id);
399
+ const order = orderedCandidates(cfg);
400
+ if (order.length === 0) {
401
+ pushEvent(cfg, 'no-candidates', {});
402
+ yield failChunk('channel group "' + cfg.id + '" has no candidates', 'NO_CANDIDATES');
403
+ return;
404
+ }
405
+ const strategy = cfg.strategy || 'sticky';
406
+ let cursor = strategy === 'primary' ? 0 : rt.currentIndex % order.length;
407
+ let fullRounds = 0;
408
+ let lastFail = null;
409
+ const failedKeys = [];
410
+ while (true) {
411
+ if (options.signal && options.signal.aborted) {
412
+ yield abortedChunk('request aborted');
413
+ return;
414
+ }
415
+ const round = [];
416
+ for (let i = 0; i < order.length; i++)
417
+ round.push(order[(cursor + i) % order.length]);
418
+ const now = Date.now();
419
+ const allCooled = round.every((cand) => (rt.cooldowns.get(candKey(cand)) || 0) > now);
420
+ let succeeded = false;
421
+ lastFail = null;
422
+ for (const cand of round) {
423
+ const key = candKey(cand);
424
+ const cooledUntil = rt.cooldowns.get(key) || 0;
425
+ if (cooledUntil > now && !allCooled)
426
+ continue;
427
+ if (options.signal && options.signal.aborted) {
428
+ yield abortedChunk('request aborted');
429
+ return;
430
+ }
431
+ let candidateOk = false;
432
+ for (let r = 0; r <= (cfg.maxRetriesPerCandidate || 0); r++) {
433
+ if (options.signal && options.signal.aborted) {
434
+ yield abortedChunk('request aborted');
435
+ return;
436
+ }
437
+ const attemptStart = Date.now();
438
+ try {
439
+ const attempt = streamAttempt(cfg, cand, options, attemptStart);
440
+ for await (const chunk of attempt)
441
+ yield chunk;
442
+ succeeded = true;
443
+ candidateOk = true;
444
+ rt.cooldowns.delete(key);
445
+ const pos = order.findIndex((c) => candKey(c) === key);
446
+ if (strategy === 'round-robin')
447
+ rt.currentIndex = (pos + 1) % order.length;
448
+ else if (strategy === 'sticky')
449
+ rt.currentIndex = pos;
450
+ else
451
+ rt.currentIndex = 0;
452
+ if (failedKeys.length > 0)
453
+ pushEvent(cfg, 'failover', { from: failedKeys[failedKeys.length - 1], to: key, reason: 'candidate change after ' + failedKeys.length + ' failure(s)' });
454
+ break;
455
+ }
456
+ catch (err) {
457
+ const message = (err && err.message) || 'candidate failed';
458
+ lastFail = message;
459
+ if (options.signal && options.signal.aborted) {
460
+ yield abortedChunk('request aborted');
461
+ return;
462
+ }
463
+ if (err && err.emitted) {
464
+ // 内容已流出后失败:无法干净重放到下一候选(会拼接两个模型的输出/产生第二个 finish),
465
+ // 置冷却并直接以失败终结本次请求
466
+ rt.cooldowns.set(key, Date.now() + (cfg.cooldownMs || 0));
467
+ failedKeys.push(key);
468
+ pushEvent(cfg, 'midstream-fail', { candidate: key, reason: message });
469
+ yield failChunk('channel candidate failed after content was already delivered: ' + message, (err && err.code) || 'CHANNEL_MIDSTREAM_FAIL');
470
+ return;
471
+ }
472
+ if (r < (cfg.maxRetriesPerCandidate || 0)) {
473
+ pushEvent(cfg, 'retry', { candidate: key, attempt: r + 1, reason: message });
474
+ await ctx.timeout(Math.min(4000, 250 * Math.pow(2, r)));
475
+ continue;
476
+ }
477
+ rt.cooldowns.set(key, Date.now() + (cfg.cooldownMs || 0));
478
+ failedKeys.push(key);
479
+ break;
480
+ }
481
+ }
482
+ if (candidateOk)
483
+ break;
484
+ }
485
+ if (succeeded)
486
+ return;
487
+ pushEvent(cfg, 'full-fail', { candidates: failedKeys.slice(), reason: lastFail });
488
+ if (fullRounds < 1) {
489
+ fullRounds++;
490
+ rt.cooldowns.clear();
491
+ rt.currentIndex = 0;
492
+ cursor = 0;
493
+ continue;
494
+ }
495
+ if (options.signal && options.signal.aborted) {
496
+ yield abortedChunk('request aborted');
497
+ return;
498
+ }
499
+ yield failChunk('all channel candidates failed: ' + (lastFail || 'unknown error'), 'CHANNEL_FULL_FAIL');
500
+ return;
501
+ }
502
+ }
503
+ // ---------- 测速 ----------
504
+ async function measureCandidate(cfg, cand, st) {
505
+ const llm = ctx.llm;
506
+ const inner = llm.stream({ provider: cand.provider, model: cand.model, messages: [{ role: 'user', content: [{ type: 'text', text: st.prompt }] }], maxTokens: st.maxTokens });
507
+ const start = Date.now();
508
+ let ttft = null;
509
+ const closeInner = async () => { try {
510
+ const c = inner.return ? inner.return() : null;
511
+ if (c && c.then)
512
+ await c;
513
+ }
514
+ catch (_e) { } };
515
+ try {
516
+ while (true) {
517
+ const guard = timed(Math.max(1, st.timeoutMs), () => ({ code: 'TIMEOUT', message: 'speedtest timed out after ' + st.timeoutMs + 'ms' }));
518
+ let next;
519
+ try {
520
+ next = await Promise.race([inner.next(), guard.promise]);
521
+ }
522
+ finally {
523
+ guard.dispose();
524
+ }
525
+ const chunk = next.value;
526
+ if (chunk === undefined)
527
+ throw { code: 'STREAM_CLOSED', message: 'speedtest stream ended early' };
528
+ if (ttft === null && isContentChunk(chunk))
529
+ ttft = Date.now() - start;
530
+ if (isTerminalChunk(chunk)) {
531
+ const reason = chunk.reason || {};
532
+ if (isSuccessReason(reason))
533
+ return { provider: cand.provider, model: cand.model, ok: true, ttft, latency: Date.now() - start };
534
+ throw { code: (reason.failure && reason.failure.code) || 'STREAM_ERROR', message: (reason.failure && reason.failure.message) || 'speedtest failed' };
535
+ }
536
+ }
537
+ }
538
+ catch (err) {
539
+ await closeInner();
540
+ throw { code: (err && err.code) || 'STREAM_ERROR', message: (err && err.message) || 'speedtest failed' };
541
+ }
542
+ }
543
+ async function runSpeedTest(cfg) {
544
+ const rt = groupRuntime(cfg.id);
545
+ if (rt.speedTestRunning)
546
+ return { ok: false, reason: 'already-running' };
547
+ rt.speedTestRunning = true;
548
+ pushEvent(cfg, 'speedtest-start', {});
549
+ try {
550
+ const st = cfg.speedTest;
551
+ const order = orderedCandidates(cfg);
552
+ const results = [];
553
+ const concurrency = Math.max(1, st.concurrency || 3);
554
+ for (let i = 0; i < order.length; i += concurrency) {
555
+ const wave = order.slice(i, i + concurrency);
556
+ if (wave.length === 0)
557
+ break;
558
+ const measured = await Promise.all(wave.map(async (cand) => {
559
+ let lastErr = null;
560
+ for (let t = 0; t <= (st.retries || 0); t++) {
561
+ try {
562
+ return await measureCandidate(cfg, cand, st);
563
+ }
564
+ catch (err) {
565
+ lastErr = err;
566
+ if (t < (st.retries || 0))
567
+ await ctx.timeout(1500);
568
+ }
569
+ }
570
+ return { provider: cand.provider, model: cand.model, ok: false, ttft: null, latency: null, failure: (lastErr && lastErr.message) || 'speedtest failed' };
571
+ }));
572
+ results.push(...measured);
573
+ }
574
+ const rows = results.map((r) => ({ provider: r.provider, model: r.model, ok: r.ok, ttft: r.ttft, latency: r.latency, at: Date.now(), failure: r.failure || null }));
575
+ state.speedResults[cfg.id] = rows;
576
+ await persistSpeedResults(state.speedResults);
577
+ const now = Date.now();
578
+ for (const r of rows)
579
+ if (!r.ok)
580
+ rt.cooldowns.set(candKey(r), now + (cfg.cooldownMs || 0));
581
+ rt.currentIndex = 0;
582
+ rt.lastSpeedTestAt = now;
583
+ pushEvent(cfg, 'speedtest-done', { ok: results.filter((r) => r.ok).length, fail: results.filter((r) => !r.ok).length });
584
+ return { ok: true, results };
585
+ }
586
+ finally {
587
+ rt.speedTestRunning = false;
588
+ }
589
+ }
590
+ // ---------- LLM 适配器(虚拟 route 注册) ----------
591
+ // 模型元数据解析。config 显式传入:resolveModel 用现势 pullConfig(),
592
+ // prepareCall 用准备时刻的快照——「prepare 与 dispatch 间 settings 变化不得混代」。
593
+ function resolveModelWith(config, provider, model) {
594
+ const id = groupOfRoute(provider) || provider;
595
+ const cfg = config.groups.find((g) => g.id === id);
596
+ if (!cfg)
597
+ return { provider, id: model, name: model };
598
+ const levels = cfg.virtualModel.reasoning ? ['off', 'minimal', 'low', 'medium', 'high', 'xhigh', 'max'] : [];
599
+ return {
600
+ provider, id: model, name: cfg.virtualModel.name,
601
+ context: { contextWindow: cfg.virtualModel.contextWindow },
602
+ defaultMaxTokens: cfg.virtualModel.maxTokens,
603
+ inputModalities: cfg.virtualModel.input.slice(),
604
+ reasoning: levels.length > 0 ? { efforts: levels.map((l) => ({ id: l, name: l })), defaultEffort: 'medium' } : undefined,
605
+ };
606
+ }
607
+ const adapter = {
608
+ providerInfo(provider) {
609
+ const id = groupOfRoute(provider) || provider;
610
+ const cfg = pullConfig().groups.find((g) => g.id === id);
611
+ return { id: provider, name: cfg ? cfg.virtualModel.name : provider };
612
+ },
613
+ providerRetryPolicy() { return undefined; },
614
+ async listModels(provider) {
615
+ const id = groupOfRoute(provider) || provider;
616
+ const cfg = pullConfig().groups.find((g) => g.id === id);
617
+ if (!cfg)
618
+ return [];
619
+ return [{ provider, id: cfg.id, name: cfg.virtualModel.name, inputModalities: cfg.virtualModel.input.slice() }];
620
+ },
621
+ async resolveModel(provider, model, _signal) {
622
+ return resolveModelWith(pullConfig(), provider, model);
623
+ },
624
+ // rc.2 运行时契约:主分发路径(llm.stream / llm.prepareCall)都先调
625
+ // adapter.prepareCall(provider, model, signal) 拿 {model, stream}——
626
+ // 缺失会在真实发对话时报 `registration.adapter.prepareCall is not a function`
627
+ // (注册/目录/菜单不经过它,所以此前未暴露)。快照绑定对齐 llm-pi-ai 的
628
+ // current() 模式:元数据与本次 dispatch 都用同一份组配置。
629
+ prepareCall(provider, model, _signal) {
630
+ const snapshot = pullConfig();
631
+ const cfg = snapshot.groups.find((g) => g.id === groupOfRoute(provider));
632
+ return Promise.resolve({
633
+ model: resolveModelWith(snapshot, provider, model),
634
+ stream: (options) => cfg === undefined
635
+ ? (async function* () { yield failChunk('unknown channel group "' + provider + '"', 'NO_ADAPTER'); })()
636
+ : streamGroup(cfg, options),
637
+ });
638
+ },
639
+ stream(options) {
640
+ const gid = groupOfRoute(options.provider);
641
+ const cfg = pullConfig().groups.find((g) => g.id === gid);
642
+ if (!cfg) {
643
+ return (async function* () { yield failChunk('unknown channel group "' + options.provider + '"', 'NO_ADAPTER'); })();
644
+ }
645
+ return streamGroup(cfg, options);
646
+ },
647
+ };
648
+ let adapterHandle = null;
649
+ // ---------- 单模型真实请求测试(client 经 settings 总线下发,走 DSH 真实 llm.stream 链路) ----------
650
+ async function runModelTest(provider, model, prompt, maxTokens) {
651
+ const llm = ctx.llm;
652
+ const inner = llm.stream({ provider, model, messages: [{ role: 'user', content: [{ type: 'text', text: prompt || '你好' }] }], maxTokens: maxTokens || 512 });
653
+ const start = Date.now();
654
+ let ttft = null;
655
+ let text = '';
656
+ let reasoning = '';
657
+ const closeInner = async () => { try {
658
+ const c = inner.return ? inner.return() : null;
659
+ if (c && c.then)
660
+ await c;
661
+ }
662
+ catch (_e) { } };
663
+ try {
664
+ while (true) {
665
+ const guard = timed(60000, () => ({ code: 'TIMEOUT', message: 'model test timed out after 60000ms' }));
666
+ let next;
667
+ try {
668
+ next = await Promise.race([inner.next(), guard.promise]);
669
+ }
670
+ finally {
671
+ guard.dispose();
672
+ }
673
+ const chunk = next.value;
674
+ if (chunk === undefined)
675
+ throw { code: 'STREAM_CLOSED', message: 'model test stream ended early' };
676
+ if (ttft === null && isContentChunk(chunk)) {
677
+ ttft = Date.now() - start;
678
+ }
679
+ if (chunk && chunk.type === 'text-delta' && typeof chunk.text === 'string') {
680
+ text += chunk.text;
681
+ }
682
+ else if (chunk && chunk.type === 'reasoning-delta' && typeof chunk.text === 'string') {
683
+ reasoning += chunk.text;
684
+ }
685
+ if (isTerminalChunk(chunk)) {
686
+ const reason = chunk.reason || {};
687
+ if (isSuccessReason(reason))
688
+ return { ok: true, ttftMs: ttft, latencyMs: Date.now() - start, text: text.slice(0, 2000), reasoning: reasoning.slice(0, 2000) };
689
+ throw { code: (reason.failure && reason.failure.code) || 'STREAM_ERROR', message: (reason.failure && reason.failure.message) || 'model test failed' };
690
+ }
691
+ }
692
+ }
693
+ catch (err) {
694
+ await closeInner();
695
+ throw err;
696
+ }
697
+ }
698
+ function handleTestRequest(settings, next) {
699
+ const req = next.testRequest;
700
+ const last = next.lastTestHandledNonce || 0;
701
+ if (!req || typeof req.provider !== 'string' || typeof req.model !== 'string' || typeof req.nonce !== 'number' || req.nonce === last)
702
+ return;
703
+ settings.update(NS_HEALTH, { lastTestHandledNonce: req.nonce }).catch(() => { });
704
+ // 结果写入走 mutate path-ops 原子设置单 nonce:read-merge-write 整包 update 在
705
+ // 两个测试并发完成时会互相覆盖(读旧包→写入丢掉对方 nonce);path-ops 对
706
+ // settings 队列中的 section 现势作用,不再依赖调用方快照
707
+ const setResult = (entry) => {
708
+ settings.mutate(NS_HEALTH, [{ op: 'set', path: ['testResults', String(req.nonce)], value: Object.assign({}, entry, { nonce: req.nonce, provider: req.provider, model: req.model }) }]).catch(() => { });
709
+ };
710
+ // 修剪低频执行:只在条目数超限时砍到 50(不在每次写入时整包重写)
711
+ const maybePrune = () => {
712
+ const all = (bus && bus.healthScope ? bus.healthScope.get().testResults : null) || {};
713
+ const entries = Object.entries(all);
714
+ if (entries.length <= 50) return;
715
+ const drop = entries.sort((a, b) => ((b[1] && b[1].finishedAt) || 0) - ((a[1] && a[1].finishedAt) || 0)).slice(50).map(([k]) => ({ op: 'unset', path: ['testResults', String(k)] }));
716
+ if (drop.length > 0)
717
+ settings.mutate(NS_HEALTH, drop).catch(() => { });
718
+ };
719
+ const done = async (r) => {
720
+ try {
721
+ setResult(Object.assign({}, r, { finishedAt: Date.now() }));
722
+ maybePrune();
723
+ }
724
+ catch (_e) { }
725
+ };
726
+ setResult({ status: 'running', startedAt: Date.now() });
727
+ runModelTest(req.provider, req.model, req.prompt, req.maxTokens).then(async (r) => {
728
+ await done(Object.assign({ status: 'ok' }, r));
729
+ }).catch(async (e) => {
730
+ await done({ status: 'error', code: (e && e.code) || 'STREAM_ERROR', error: (e && e.message) || String(e) });
731
+ });
732
+ }
733
+ function rewireRoutes() {
734
+ const llm = ctx.llm;
735
+ const next = pullConfig().groups.map((g) => routeOfGroup(g.id));
736
+ const current = llm.listProviders().map((p) => p.id).filter((id) => id.startsWith(ROUTE_PREFIX));
737
+ if (next.length === 0) {
738
+ // 空配置:LLM 服务拒绝注册零 provider 的适配器——推迟首次注册;已注册则清空路由
739
+ if (adapterHandle !== null)
740
+ adapterHandle.replace(next);
741
+ return;
742
+ }
743
+ if (adapterHandle === null)
744
+ adapterHandle = llm.registerAdapter(next, adapter);
745
+ else if (JSON.stringify(current) !== JSON.stringify(next))
746
+ adapterHandle.replace(next);
747
+ }
748
+ // ---------- settings 总线接入(响应式:settings 服务异步初始化,apply 时查询太早) ----------
749
+ let bus = null;
750
+ ctx.inject(['settings'], (sctx) => {
751
+ const settings = sctx.settings;
752
+ bus = { settings };
753
+ const cfgScope = settings.register(NS_CONFIG, CONFIG_SCHEMA, { base: {} });
754
+ const healthScope = settings.register(NS_HEALTH, HEALTH_SCHEMA, { base: {} });
755
+ bus.cfgScope = cfgScope;
756
+ bus.healthScope = healthScope;
757
+ state.config = clone(cfgScope.get() || { groups: [] });
758
+ state.records = clone(healthScope.get().records || {});
759
+ state.speedResults = clone(healthScope.get().speedResults || {});
760
+ state.testResults = clone(healthScope.get().testResults || {});
761
+ restoreRuntime(healthScope.get().runtime || {});
762
+ cfgScope.watch(() => {
763
+ state.config = clone(cfgScope.get() || { groups: [] });
764
+ const live = pullConfig().groups;
765
+ for (const g of live)
766
+ groupRuntime(g.id);
767
+ // 清理已删除组的运行时残留,避免 runtime 持久化无限累积
768
+ for (const id of Array.from(runtime.keys()))
769
+ if (!live.some((g) => g.id === id))
770
+ runtime.delete(id);
771
+ rewireRoutes();
772
+ console.log('[model-channel-manager] config hot-reloaded, routes:', pullConfig().groups.map((g) => g.id).join(', ') || '(none)');
773
+ });
774
+ healthScope.watch((next) => {
775
+ state.records = clone(next.records || {});
776
+ state.speedResults = clone(next.speedResults || {});
777
+ state.testResults = clone(next.testResults || {});
778
+ const req = next.speedRequest;
779
+ const last = next.lastHandledNonce || 0;
780
+ if (req && typeof req.group === 'string' && typeof req.nonce === 'number' && req.nonce !== last) {
781
+ settings.update(NS_HEALTH, { lastHandledNonce: req.nonce }).catch(() => { });
782
+ const cfg = pullConfig().groups.find((g) => g.id === req.group);
783
+ if (cfg)
784
+ runSpeedTest(cfg).catch((e) => console.error('[model-channel-manager] speedtest error:', e));
785
+ }
786
+ handleTestRequest(settings, next);
787
+ });
788
+ boot();
789
+ });
790
+ // ---------- 全局 LLM 请求健康拦截 (涵盖所有非虚拟路由的真实渠道模型调用) ----------
791
+ ctx.on('llm/stream', async function* (options, next) {
792
+ // 如果 options.provider 是虚拟轮询路由(以 roundrobin/ 开头),则跳过被动采集(避免双计)
793
+ if (options && typeof options.provider === 'string' && options.provider.startsWith(ROUTE_PREFIX)) {
794
+ for await (const chunk of next()) {
795
+ yield chunk;
796
+ }
797
+ return;
798
+ }
799
+ const startTs = Date.now();
800
+ let ttft = null;
801
+ let lastError = null;
802
+ let isSuccess = false;
803
+ // 用户主动中止不是渠道故障:AbortError / code ABORTED / signal 已 aborted
804
+ // 三种形态都不进健康流水,否则污染成功率与 smart 键的 reliability 权重
805
+ const isAbortLike = (err) => {
806
+ if (options && options.signal && options.signal.aborted)
807
+ return true;
808
+ if (!err)
809
+ return false;
810
+ if (err.code === 'ABORTED' || err.name === 'AbortError')
811
+ return true;
812
+ return typeof err.message === 'string' && /abort/i.test(err.message);
813
+ };
814
+ try {
815
+ for await (const chunk of next()) {
816
+ if (ttft === null && isContentChunk(chunk)) {
817
+ ttft = Date.now() - startTs;
818
+ }
819
+ if (isTerminalChunk(chunk)) {
820
+ const reason = chunk.reason || {};
821
+ if (isSuccessReason(reason)) {
822
+ isSuccess = true;
823
+ } else {
824
+ lastError = (reason.failure && (reason.failure.code || reason.failure.message)) || 'ERROR';
825
+ }
826
+ }
827
+ yield chunk;
828
+ }
829
+ if (options && options.provider && options.model && lastError !== 'ABORTED') {
830
+ // lastError 'ABORTED' = 终止块 kind aborted(用户中止而非渠道故障),不进健康流水
831
+ const latency = Date.now() - startTs;
832
+ const p = options.provider;
833
+ recordHealth(p, { provider: p, model: options.model }, {
834
+ ok: isSuccess,
835
+ ttftMs: isSuccess ? ttft : null,
836
+ latencyMs: isSuccess ? latency : null,
837
+ code: isSuccess ? null : (lastError || 'UNKNOWN_TERMINAL'),
838
+ });
839
+ }
840
+ } catch (err) {
841
+ if (options && options.provider && options.model && !isAbortLike(err)) {
842
+ const p = options.provider;
843
+ recordHealth(p, { provider: p, model: options.model }, {
844
+ ok: false,
845
+ code: (err && err.code) || (err && err.message) || 'EXCEPTION',
846
+ });
847
+ }
848
+ throw err;
849
+ }
850
+ }, { global: true, prepend: true });
851
+
852
+ // ---------- 启动 ----------
853
+ async function boot() {
854
+ // 动态原型迁移异步进行,不阻塞路由注册(迁移落盘后 cfg watcher 会热重建路由)
855
+ void migrateLegacyConfig();
856
+ rewireRoutes();
857
+ for (const g of pullConfig().groups) {
858
+ if (g.speedTest.enabled && g.speedTest.onFirstUse) {
859
+ const rt = groupRuntime(g.id);
860
+ if ((state.speedResults[g.id] || []).length === 0)
861
+ runSpeedTest(g).catch(() => { });
862
+ }
863
+ }
864
+ console.log('[model-channel-manager] booted, groups:', pullConfig().groups.map((g) => g.id).join(', ') || '(none)');
865
+ }
866
+ // 工作区 .channel-manager/config.json -> settings 命名空间(一次性)。
867
+ // 完成后写 legacyMigrated 哨兵,防止「用户清空全部组 → 重启 → 旧配置复活」;
868
+ // fs/sandboxPolicy 未就绪时短暂重试,而不是静默放弃直到下次重启。
869
+ async function migrateLegacyConfig() {
870
+ const settings = bus && bus.settings;
871
+ if (settings === undefined || pullConfig().groups.length !== 0)
872
+ return;
873
+ if ((bus && bus.healthScope ? bus.healthScope.get().legacyMigrated : false) === true)
874
+ return;
875
+ for (let attempt = 0; attempt < 6; attempt++) {
876
+ const fsSvc = ctx.get('fs');
877
+ const sp = ctx.get('sandboxPolicy');
878
+ const root = sp ? sp.workspaceRoot : null;
879
+ if (fsSvc !== undefined && root) {
880
+ try {
881
+ const target = await fsSvc.resolve(root + '/.channel-manager/config.json');
882
+ const text = await fsSvc.readText(target);
883
+ const legacy = JSON.parse(text);
884
+ const migrated = normalizeConfig(legacy);
885
+ if (migrated.groups.length > 0) {
886
+ await settings.replace(NS_CONFIG, { groups: migrated.groups });
887
+ state.config = { groups: migrated.groups };
888
+ console.log('[model-channel-manager] migrated legacy config from workspace .channel-manager/config.json');
889
+ }
890
+ }
891
+ catch (_e) { /* 无遗留配置,忽略 */ }
892
+ settings.update(NS_HEALTH, { legacyMigrated: true }).catch(() => { });
893
+ return;
894
+ }
895
+ try {
896
+ await ctx.timeout(5000);
897
+ }
898
+ catch (_e) {
899
+ return; // 插件已卸载,停止重试
900
+ }
901
+ }
902
+ }
903
+ }