mslxdff 0.1.156 → 0.1.159

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (97) hide show
  1. package/bin/mslxdff.js +4 -4
  2. package/docs/cli_help_mini.md +133 -0
  3. package/package.json +3 -3
  4. package/src/auto.js +254 -254
  5. package/src/bench/cline-bench.js +42 -42
  6. package/src/bench/probe.js +70 -70
  7. package/src/bench/report.js +162 -162
  8. package/src/bench/runner.js +77 -77
  9. package/src/bench/via-probe.js +124 -124
  10. package/src/bench/via-routes.js +87 -87
  11. package/src/bench/workbuddy-bench.js +54 -54
  12. package/src/chat/engine.js +160 -160
  13. package/src/chat/gateway.js +163 -163
  14. package/src/chat/orchestrator.js +234 -234
  15. package/src/chat/prompt.js +70 -70
  16. package/src/chat/repl.js +88 -88
  17. package/src/chat/terminal.js +135 -135
  18. package/src/chat/tools.js +306 -306
  19. package/src/chat-pipeline/index.js +123 -123
  20. package/src/chat-pipeline/policy.js +76 -76
  21. package/src/chat-pipeline/serial-trial.js +210 -210
  22. package/src/cli/commands/group.js +249 -249
  23. package/src/cli/commands/model/list-providers.js +1 -1
  24. package/src/cli/commands/model/picks.js +50 -50
  25. package/src/cli/commands/provider/bench-via.js +247 -247
  26. package/src/cli/commands/provider/bench.js +141 -141
  27. package/src/cli/commands/provider/index.js +124 -116
  28. package/src/cli/commands/provider/models.js +139 -128
  29. package/src/cli/commands/provider/qwenwork-login.js +119 -0
  30. package/src/cli/commands/provider/zcode-login.js +77 -0
  31. package/src/cli/commands/provider/zcode-quota.js +55 -0
  32. package/src/cli/commands/sync.js +232 -232
  33. package/src/cli/provider-row.js +2 -2
  34. package/src/cli/status.js +279 -279
  35. package/src/daemon.js +96 -96
  36. package/src/model-capabilities/enrich.js +86 -86
  37. package/src/model-capabilities/index.js +183 -183
  38. package/src/model-capabilities/parse.js +70 -70
  39. package/src/models.js +225 -225
  40. package/src/providers/classify.js +1 -1
  41. package/src/providers/cline/auth.js +228 -228
  42. package/src/providers/cline/chat.js +307 -307
  43. package/src/providers/cline.js +2 -2
  44. package/src/providers/keyring.js +60 -56
  45. package/src/providers/qoder/chat.js +183 -174
  46. package/src/providers/qoder/index.js +230 -185
  47. package/src/providers/qoder/sse.js +103 -62
  48. package/src/providers/qwenwork/account-store.js +133 -0
  49. package/src/providers/qwenwork/constants.js +67 -0
  50. package/src/providers/qwenwork/cosy.js +120 -0
  51. package/src/providers/qwenwork/crypto.js +218 -0
  52. package/src/providers/qwenwork/http.js +20 -0
  53. package/src/providers/qwenwork/index.js +327 -0
  54. package/src/providers/qwenwork/payload.js +142 -0
  55. package/src/providers/qwenwork/rsa.js +54 -0
  56. package/src/providers/qwenwork/sse.js +268 -0
  57. package/src/providers/qwenwork/stream.js +130 -0
  58. package/src/providers/qwenwork/upstream.js +120 -0
  59. package/src/providers/qwenwork.js +1 -0
  60. package/src/providers/registry.js +66 -56
  61. package/src/providers/share-keys.js +2 -2
  62. package/src/providers/workbuddy/chat.js +248 -248
  63. package/src/providers/workbuddy/reshape.js +152 -152
  64. package/src/providers/workbuddy.js +2 -2
  65. package/src/providers/zcode/account-store.js +129 -0
  66. package/src/providers/zcode/auth.js +28 -0
  67. package/src/providers/zcode/chat.js +171 -0
  68. package/src/providers/zcode/const.js +54 -0
  69. package/src/providers/zcode/headers.js +55 -0
  70. package/src/providers/zcode/index.js +124 -0
  71. package/src/providers/zcode/models.js +65 -0
  72. package/src/providers/zcode/oauth.js +120 -0
  73. package/src/providers/zcode/quota.js +176 -0
  74. package/src/providers/zcode/sse.js +179 -0
  75. package/src/reasoning.js +32 -32
  76. package/src/routes/chat/gateway.js +46 -46
  77. package/src/routes/chat/relay-pipeline.js +250 -250
  78. package/src/routes/chat/via-route-handler.js +144 -144
  79. package/src/routes/hedge.js +255 -255
  80. package/src/routes/models-route.js +167 -167
  81. package/src/routes/peers.js +273 -273
  82. package/src/routes/stream.js +438 -438
  83. package/src/runtime/bootstrap.js +45 -45
  84. package/src/runtime/provider-gate.js +33 -30
  85. package/src/runtime/providers-setup.js +165 -165
  86. package/src/server.js +64 -64
  87. package/src/state/schemas/allowlist.js +92 -92
  88. package/src/sync-opencode.js +280 -280
  89. package/src/transport/index.js +244 -244
  90. package/src/transport/pool.js +56 -56
  91. package/src/transport/retry.js +24 -24
  92. package/src/transport/sse.js +93 -93
  93. package/src/upstream-probe/display.js +52 -52
  94. package/src/upstream-probe/probe.js +49 -49
  95. package/src/upstream-probe/rotate.js +110 -110
  96. package/src/upstream-probe/start.js +45 -45
  97. package/src/upstream.js +289 -289
package/src/auto.js CHANGED
@@ -1,254 +1,254 @@
1
- import { statSync } from "node:fs";
2
- import { loadModelErrors, saveModelErrors, loadModelLatencies, saveModelLatencies, loadPreferredModel, loadModelPicks, saveModelPicks, defaultStateFile } from "./state.js";
3
-
4
- // 出厂默认首选模型(state.json 的 preferredModel / env MSLXDFF_PREFERRED_MODEL 可覆盖)
5
- export const DEFAULT_PREFERRED_MODEL = "big-pickle";
6
- // 兼容旧导出名:语义为"出厂默认",当前生效值请用 getPreferredModel()
7
- export const PREFERRED_MODEL = DEFAULT_PREFERRED_MODEL;
8
-
9
- // 当前生效的首选模型:state.json > env > 出厂默认;mtime 缓存保证 daemon 热生效
10
- const _prefCache = { mtimeMs: -1, file: null, value: null };
11
- export function getPreferredModel({ file = defaultStateFile() } = {}) {
12
- try {
13
- const st = statSync(file);
14
- if (_prefCache.file !== file || st.mtimeMs !== _prefCache.mtimeMs) {
15
- _prefCache.file = file;
16
- _prefCache.mtimeMs = st.mtimeMs;
17
- _prefCache.value = loadPreferredModel({ file });
18
- }
19
- } catch {
20
- _prefCache.file = file;
21
- _prefCache.mtimeMs = -1;
22
- _prefCache.value = null;
23
- }
24
- const env = (process.env.MSLXDFF_PREFERRED_MODEL || "").trim();
25
- return _prefCache.value || env || DEFAULT_PREFERRED_MODEL;
26
- }
27
-
28
- export const DEFAULT_AUTO_MODELS = [
29
- PREFERRED_MODEL,
30
- "mimo-v2.5-free",
31
- "deepseek-v4-flash-free",
32
- "ling-3.0-flash-fin-free",
33
- "nemotron-3-ultra-free",
34
- "nemotron-3.5-lightning-free",
35
- "muse-spark-1.3-contributor-free",
36
- ].filter((id, i, arr) => id && arr.indexOf(id) === i);
37
-
38
- export function isAutoModel(model) {
39
- return !model || model === "auto";
40
- }
41
-
42
- export const MODEL_STATUS = Object.freeze({
43
- NORMAL: "normal",
44
- LIMIT: "limit",
45
- ERROR: "error",
46
- });
47
-
48
- // Legacy modelErrors entries are bare timestamps ({id: ts}); newer ones are
49
- // objects ({id: {status, at, code}}). Normalize both to an entry object.
50
- // `slow` flags a model whose last request was slow (over the wall-clock
51
- // threshold) — those get a longer cooldown so they lie low until they recover.
52
- function normEntry(e) {
53
- if (typeof e === "number") return { status: MODEL_STATUS.ERROR, at: e, code: null, slow: false };
54
- if (e && typeof e === "object") {
55
- return {
56
- status: e.status || MODEL_STATUS.ERROR,
57
- at: typeof e.at === "number" ? e.at : 0,
58
- code: e.code ?? null,
59
- slow: Boolean(e.slow),
60
- };
61
- }
62
- return null;
63
- }
64
-
65
- export function classifyErrorEvent(evt = {}) {
66
- if (evt.slow) return MODEL_STATUS.ERROR;
67
- const code = Number(evt.status);
68
- if (code === 429) return MODEL_STATUS.LIMIT;
69
- const msg = String(evt.message || evt.note || "").toLowerCase();
70
- if (msg.includes("rate limit") || msg.includes("limit exceeded") || msg.includes("429")) {
71
- return MODEL_STATUS.LIMIT;
72
- }
73
- return MODEL_STATUS.ERROR;
74
- }
75
-
76
- export const DEFAULT_COOLDOWN_MS = 60_000;
77
- export const DEFAULT_SLOW_COOLDOWN_MS = 5 * 60_000;
78
- export const DEFAULT_LATENCY_ALPHA = 0.3;
79
-
80
- function effectiveCooldown(entry, slowCooldownMs, cooldownMs) {
81
- if (entry && entry.slow) return slowCooldownMs || 0;
82
- return cooldownMs || 0;
83
- }
84
-
85
- function inCooldown(id, errors, now, cooldownMs, slowCooldownMs) {
86
- const e = normEntry(errors[id]);
87
- if (!e || !(e.at > 0)) return false;
88
- if (e.status === MODEL_STATUS.NORMAL) return false;
89
- const cd = effectiveCooldown(e, slowCooldownMs, cooldownMs);
90
- return cd > 0 && now - e.at < cd;
91
- }
92
-
93
- // Latency EMA helpers
94
- function normLatency(e) {
95
- if (!e || typeof e !== "object") return null;
96
- const ema = Number(e.emaMs);
97
- return Number.isFinite(ema) && ema > 0 ? ema : null;
98
- }
99
-
100
- export function rankModels(ids, errors = {}, { now = Date.now(), cooldownMs = 0, slowCooldownMs = 0, latencies = {}, preferred } = {}) {
101
- const pref = preferred ?? getPreferredModel();
102
- // 最近一次成功的模型(NORMAL 且非慢且 at 最大),用于“上次成功优先”——慢模型即使刚成功也不应钉死
103
- let lastSuccessId = null;
104
- let lastSuccessAt = 0;
105
- for (const [id, e] of Object.entries(errors)) {
106
- const ne = normEntry(e);
107
- if (ne && ne.status === MODEL_STATUS.NORMAL && !ne.slow && ne.at > lastSuccessAt) {
108
- lastSuccessAt = ne.at;
109
- lastSuccessId = id;
110
- }
111
- }
112
- return [...new Set(ids)]
113
- .filter(Boolean)
114
- .map((id) => ({
115
- id,
116
- e: normEntry(errors[id]),
117
- err: normEntry(errors[id])?.at ?? 0,
118
- isPreferred: id === pref,
119
- isLastSuccess: id === lastSuccessId,
120
- cooling: inCooldown(id, errors, now, cooldownMs, slowCooldownMs),
121
- latency: normLatency(latencies[id]) ?? Number.MAX_SAFE_INTEGER,
122
- }))
123
- .sort(
124
- (a, b) =>
125
- (a.cooling ? 1 : 0) - (b.cooling ? 1 : 0) ||
126
- (b.isLastSuccess ? 1 : 0) - (a.isLastSuccess ? 1 : 0) ||
127
- (b.isPreferred ? 1 : 0) - (a.isPreferred ? 1 : 0) ||
128
- a.latency - b.latency ||
129
- a.err - b.err
130
- )
131
- .map((x) => x.id);
132
- }
133
-
134
- export function createAutoSelector({
135
- loadCandidates,
136
- file,
137
- now = () => Date.now(),
138
- cooldownMs = DEFAULT_COOLDOWN_MS,
139
- slowCooldownMs = DEFAULT_SLOW_COOLDOWN_MS,
140
- latencyAlpha = DEFAULT_LATENCY_ALPHA,
141
- errors: seedErrors,
142
- latencies: seedLatencies,
143
- persist = (errors, f = file) => saveModelErrors(errors, f ? { file: f } : {}),
144
- persistLatencies = (latencies, f = file) => saveModelLatencies(latencies, f ? { file: f } : {}),
145
- loadPicks = () => (file ? loadModelPicks({ file }) : []),
146
- persistPicks = (picks) => (file ? saveModelPicks(picks, { file }) : picks),
147
- } = {}) {
148
- const lastErrorAt = { ...(seedErrors ?? loadModelErrors(file ? { file } : {})) };
149
- const latencies = { ...(seedLatencies ?? loadModelLatencies(file ? { file } : {})) };
150
-
151
- async function loadList() {
152
- let list;
153
- try {
154
- const loaded = await loadCandidates?.();
155
- list = Array.isArray(loaded) && loaded.length ? loaded : DEFAULT_AUTO_MODELS;
156
- } catch {
157
- list = DEFAULT_AUTO_MODELS;
158
- }
159
- return [...new Set(list)].filter(Boolean);
160
- }
161
-
162
- // 勾选集 = auto 候选池白名单:只在勾选的模型里择优;空勾选或勾选中无可用模型时回退全量
163
- async function pickedPool(list) {
164
- const picks = loadPicks();
165
- if (!picks.length) return list;
166
- const pickedSet = new Set(picks);
167
- const filtered = list.filter((id) => pickedSet.has(id));
168
- return filtered.length ? filtered : list;
169
- }
170
-
171
- async function candidates() {
172
- const list = await loadList();
173
- const pool = await pickedPool(list);
174
- return rankModels(pool, lastErrorAt, { now: now(), cooldownMs, slowCooldownMs, latencies, preferred: getPreferredModel({ file: file ?? undefined }) });
175
- }
176
-
177
- async function candidatesFor(requested) {
178
- if (!requested) return candidates();
179
- const list = await loadList();
180
- // 显式指定某模型 = 认可它,自动加入勾选集(仅当它是真实上游 free 模型时,避免垃圾 id 污染)
181
- const picks = loadPicks();
182
- if (list.includes(requested) && !picks.includes(requested) && persistPicks) {
183
- await persistPicks([...picks, requested]);
184
- }
185
- const all = list.includes(requested) ? list : [requested, ...list];
186
- const others = rankModels(all.filter((id) => id !== requested), lastErrorAt, {
187
- now: now(),
188
- cooldownMs,
189
- slowCooldownMs,
190
- latencies,
191
- preferred: getPreferredModel({ file: file ?? undefined }),
192
- });
193
- // 显式指定模型:严格优先,永不因冷却被挤到最后(原设计:A deepseek 失败 → B/D deepseek 并发 → 都失败才 fallback)
194
- // 冷却仅影响 auto 的择优,不影响指定模型的“很难被更改”语义
195
- return [requested, ...others];
196
- }
197
-
198
- function isCooling(id) {
199
- return inCooldown(id, lastErrorAt, now(), cooldownMs, slowCooldownMs);
200
- }
201
-
202
- async function recordError(id, evt = {}) {
203
- if (!id) return;
204
- lastErrorAt[id] = {
205
- status: classifyErrorEvent(evt),
206
- at: now(),
207
- code: Number.isInteger(Number(evt.status)) ? Number(evt.status) : null,
208
- slow: Boolean(evt.slow),
209
- };
210
- await persist({ ...lastErrorAt });
211
- }
212
-
213
- async function recordOk(id, evt = {}) {
214
- if (!id) return;
215
- lastErrorAt[id] = { status: MODEL_STATUS.NORMAL, at: now(), code: 200, slow: false };
216
- await persist({ ...lastErrorAt });
217
- const ms = Number(evt.latencyMs ?? evt.totalMs ?? evt.elapsedMs);
218
- if (Number.isFinite(ms) && ms > 0) {
219
- const prev = latencies[id]?.emaMs;
220
- const ema = prev ? Math.round(prev * (1 - latencyAlpha) + ms * latencyAlpha) : Math.round(ms);
221
- latencies[id] = { emaMs: ema, lastMs: Math.round(ms), at: now(), count: (latencies[id]?.count ?? 0) + 1 };
222
- await persistLatencies({ ...latencies });
223
- }
224
- }
225
-
226
- async function recordLatency(id, ms) {
227
- if (!id || !Number.isFinite(ms) || ms <= 0) return;
228
- const prev = latencies[id]?.emaMs;
229
- const ema = prev ? Math.round(prev * (1 - latencyAlpha) + ms * latencyAlpha) : Math.round(ms);
230
- latencies[id] = { emaMs: ema, lastMs: Math.round(ms), at: now(), count: (latencies[id]?.count ?? 0) + 1 };
231
- await persistLatencies({ ...latencies });
232
- }
233
-
234
- function statuses() {
235
- return { ...lastErrorAt };
236
- }
237
-
238
- function latencyStatuses() {
239
- return { ...latencies };
240
- }
241
-
242
- return {
243
- candidates,
244
- candidatesFor,
245
- recordError,
246
- recordOk,
247
- recordLatency,
248
- statuses,
249
- latencyStatuses,
250
- isCooling,
251
- errors: () => ({ ...lastErrorAt }),
252
- latencies: () => ({ ...latencies }),
253
- };
254
- }
1
+ import { statSync } from "node:fs";
2
+ import { loadModelErrors, saveModelErrors, loadModelLatencies, saveModelLatencies, loadPreferredModel, loadModelPicks, saveModelPicks, defaultStateFile } from "./state.js";
3
+
4
+ // 出厂默认首选模型(state.json 的 preferredModel / env MSLXDFF_PREFERRED_MODEL 可覆盖)
5
+ export const DEFAULT_PREFERRED_MODEL = "big-pickle";
6
+ // 兼容旧导出名:语义为"出厂默认",当前生效值请用 getPreferredModel()
7
+ export const PREFERRED_MODEL = DEFAULT_PREFERRED_MODEL;
8
+
9
+ // 当前生效的首选模型:state.json > env > 出厂默认;mtime 缓存保证 daemon 热生效
10
+ const _prefCache = { mtimeMs: -1, file: null, value: null };
11
+ export function getPreferredModel({ file = defaultStateFile() } = {}) {
12
+ try {
13
+ const st = statSync(file);
14
+ if (_prefCache.file !== file || st.mtimeMs !== _prefCache.mtimeMs) {
15
+ _prefCache.file = file;
16
+ _prefCache.mtimeMs = st.mtimeMs;
17
+ _prefCache.value = loadPreferredModel({ file });
18
+ }
19
+ } catch {
20
+ _prefCache.file = file;
21
+ _prefCache.mtimeMs = -1;
22
+ _prefCache.value = null;
23
+ }
24
+ const env = (process.env.MSLXDFF_PREFERRED_MODEL || "").trim();
25
+ return _prefCache.value || env || DEFAULT_PREFERRED_MODEL;
26
+ }
27
+
28
+ export const DEFAULT_AUTO_MODELS = [
29
+ PREFERRED_MODEL,
30
+ "mimo-v2.5-free",
31
+ "deepseek-v4-flash-free",
32
+ "ling-3.0-flash-fin-free",
33
+ "nemotron-3-ultra-free",
34
+ "nemotron-3.5-lightning-free",
35
+ "muse-spark-1.3-contributor-free",
36
+ ].filter((id, i, arr) => id && arr.indexOf(id) === i);
37
+
38
+ export function isAutoModel(model) {
39
+ return !model || model === "auto";
40
+ }
41
+
42
+ export const MODEL_STATUS = Object.freeze({
43
+ NORMAL: "normal",
44
+ LIMIT: "limit",
45
+ ERROR: "error",
46
+ });
47
+
48
+ // Legacy modelErrors entries are bare timestamps ({id: ts}); newer ones are
49
+ // objects ({id: {status, at, code}}). Normalize both to an entry object.
50
+ // `slow` flags a model whose last request was slow (over the wall-clock
51
+ // threshold) — those get a longer cooldown so they lie low until they recover.
52
+ function normEntry(e) {
53
+ if (typeof e === "number") return { status: MODEL_STATUS.ERROR, at: e, code: null, slow: false };
54
+ if (e && typeof e === "object") {
55
+ return {
56
+ status: e.status || MODEL_STATUS.ERROR,
57
+ at: typeof e.at === "number" ? e.at : 0,
58
+ code: e.code ?? null,
59
+ slow: Boolean(e.slow),
60
+ };
61
+ }
62
+ return null;
63
+ }
64
+
65
+ export function classifyErrorEvent(evt = {}) {
66
+ if (evt.slow) return MODEL_STATUS.ERROR;
67
+ const code = Number(evt.status);
68
+ if (code === 429) return MODEL_STATUS.LIMIT;
69
+ const msg = String(evt.message || evt.note || "").toLowerCase();
70
+ if (msg.includes("rate limit") || msg.includes("limit exceeded") || msg.includes("429")) {
71
+ return MODEL_STATUS.LIMIT;
72
+ }
73
+ return MODEL_STATUS.ERROR;
74
+ }
75
+
76
+ export const DEFAULT_COOLDOWN_MS = 60_000;
77
+ export const DEFAULT_SLOW_COOLDOWN_MS = 5 * 60_000;
78
+ export const DEFAULT_LATENCY_ALPHA = 0.3;
79
+
80
+ function effectiveCooldown(entry, slowCooldownMs, cooldownMs) {
81
+ if (entry && entry.slow) return slowCooldownMs || 0;
82
+ return cooldownMs || 0;
83
+ }
84
+
85
+ function inCooldown(id, errors, now, cooldownMs, slowCooldownMs) {
86
+ const e = normEntry(errors[id]);
87
+ if (!e || !(e.at > 0)) return false;
88
+ if (e.status === MODEL_STATUS.NORMAL) return false;
89
+ const cd = effectiveCooldown(e, slowCooldownMs, cooldownMs);
90
+ return cd > 0 && now - e.at < cd;
91
+ }
92
+
93
+ // Latency EMA helpers
94
+ function normLatency(e) {
95
+ if (!e || typeof e !== "object") return null;
96
+ const ema = Number(e.emaMs);
97
+ return Number.isFinite(ema) && ema > 0 ? ema : null;
98
+ }
99
+
100
+ export function rankModels(ids, errors = {}, { now = Date.now(), cooldownMs = 0, slowCooldownMs = 0, latencies = {}, preferred } = {}) {
101
+ const pref = preferred ?? getPreferredModel();
102
+ // 最近一次成功的模型(NORMAL 且非慢且 at 最大),用于“上次成功优先”——慢模型即使刚成功也不应钉死
103
+ let lastSuccessId = null;
104
+ let lastSuccessAt = 0;
105
+ for (const [id, e] of Object.entries(errors)) {
106
+ const ne = normEntry(e);
107
+ if (ne && ne.status === MODEL_STATUS.NORMAL && !ne.slow && ne.at > lastSuccessAt) {
108
+ lastSuccessAt = ne.at;
109
+ lastSuccessId = id;
110
+ }
111
+ }
112
+ return [...new Set(ids)]
113
+ .filter(Boolean)
114
+ .map((id) => ({
115
+ id,
116
+ e: normEntry(errors[id]),
117
+ err: normEntry(errors[id])?.at ?? 0,
118
+ isPreferred: id === pref,
119
+ isLastSuccess: id === lastSuccessId,
120
+ cooling: inCooldown(id, errors, now, cooldownMs, slowCooldownMs),
121
+ latency: normLatency(latencies[id]) ?? Number.MAX_SAFE_INTEGER,
122
+ }))
123
+ .sort(
124
+ (a, b) =>
125
+ (a.cooling ? 1 : 0) - (b.cooling ? 1 : 0) ||
126
+ (b.isLastSuccess ? 1 : 0) - (a.isLastSuccess ? 1 : 0) ||
127
+ (b.isPreferred ? 1 : 0) - (a.isPreferred ? 1 : 0) ||
128
+ a.latency - b.latency ||
129
+ a.err - b.err
130
+ )
131
+ .map((x) => x.id);
132
+ }
133
+
134
+ export function createAutoSelector({
135
+ loadCandidates,
136
+ file,
137
+ now = () => Date.now(),
138
+ cooldownMs = DEFAULT_COOLDOWN_MS,
139
+ slowCooldownMs = DEFAULT_SLOW_COOLDOWN_MS,
140
+ latencyAlpha = DEFAULT_LATENCY_ALPHA,
141
+ errors: seedErrors,
142
+ latencies: seedLatencies,
143
+ persist = (errors, f = file) => saveModelErrors(errors, f ? { file: f } : {}),
144
+ persistLatencies = (latencies, f = file) => saveModelLatencies(latencies, f ? { file: f } : {}),
145
+ loadPicks = () => (file ? loadModelPicks({ file }) : []),
146
+ persistPicks = (picks) => (file ? saveModelPicks(picks, { file }) : picks),
147
+ } = {}) {
148
+ const lastErrorAt = { ...(seedErrors ?? loadModelErrors(file ? { file } : {})) };
149
+ const latencies = { ...(seedLatencies ?? loadModelLatencies(file ? { file } : {})) };
150
+
151
+ async function loadList() {
152
+ let list;
153
+ try {
154
+ const loaded = await loadCandidates?.();
155
+ list = Array.isArray(loaded) && loaded.length ? loaded : DEFAULT_AUTO_MODELS;
156
+ } catch {
157
+ list = DEFAULT_AUTO_MODELS;
158
+ }
159
+ return [...new Set(list)].filter(Boolean);
160
+ }
161
+
162
+ // 勾选集 = auto 候选池白名单:只在勾选的模型里择优;空勾选或勾选中无可用模型时回退全量
163
+ async function pickedPool(list) {
164
+ const picks = loadPicks();
165
+ if (!picks.length) return list;
166
+ const pickedSet = new Set(picks);
167
+ const filtered = list.filter((id) => pickedSet.has(id));
168
+ return filtered.length ? filtered : list;
169
+ }
170
+
171
+ async function candidates() {
172
+ const list = await loadList();
173
+ const pool = await pickedPool(list);
174
+ return rankModels(pool, lastErrorAt, { now: now(), cooldownMs, slowCooldownMs, latencies, preferred: getPreferredModel({ file: file ?? undefined }) });
175
+ }
176
+
177
+ async function candidatesFor(requested) {
178
+ if (!requested) return candidates();
179
+ const list = await loadList();
180
+ // 显式指定某模型 = 认可它,自动加入勾选集(仅当它是真实上游 free 模型时,避免垃圾 id 污染)
181
+ const picks = loadPicks();
182
+ if (list.includes(requested) && !picks.includes(requested) && persistPicks) {
183
+ await persistPicks([...picks, requested]);
184
+ }
185
+ const all = list.includes(requested) ? list : [requested, ...list];
186
+ const others = rankModels(all.filter((id) => id !== requested), lastErrorAt, {
187
+ now: now(),
188
+ cooldownMs,
189
+ slowCooldownMs,
190
+ latencies,
191
+ preferred: getPreferredModel({ file: file ?? undefined }),
192
+ });
193
+ // 显式指定模型:严格优先,永不因冷却被挤到最后(原设计:A deepseek 失败 → B/D deepseek 并发 → 都失败才 fallback)
194
+ // 冷却仅影响 auto 的择优,不影响指定模型的“很难被更改”语义
195
+ return [requested, ...others];
196
+ }
197
+
198
+ function isCooling(id) {
199
+ return inCooldown(id, lastErrorAt, now(), cooldownMs, slowCooldownMs);
200
+ }
201
+
202
+ async function recordError(id, evt = {}) {
203
+ if (!id) return;
204
+ lastErrorAt[id] = {
205
+ status: classifyErrorEvent(evt),
206
+ at: now(),
207
+ code: Number.isInteger(Number(evt.status)) ? Number(evt.status) : null,
208
+ slow: Boolean(evt.slow),
209
+ };
210
+ await persist({ ...lastErrorAt });
211
+ }
212
+
213
+ async function recordOk(id, evt = {}) {
214
+ if (!id) return;
215
+ lastErrorAt[id] = { status: MODEL_STATUS.NORMAL, at: now(), code: 200, slow: false };
216
+ await persist({ ...lastErrorAt });
217
+ const ms = Number(evt.latencyMs ?? evt.totalMs ?? evt.elapsedMs);
218
+ if (Number.isFinite(ms) && ms > 0) {
219
+ const prev = latencies[id]?.emaMs;
220
+ const ema = prev ? Math.round(prev * (1 - latencyAlpha) + ms * latencyAlpha) : Math.round(ms);
221
+ latencies[id] = { emaMs: ema, lastMs: Math.round(ms), at: now(), count: (latencies[id]?.count ?? 0) + 1 };
222
+ await persistLatencies({ ...latencies });
223
+ }
224
+ }
225
+
226
+ async function recordLatency(id, ms) {
227
+ if (!id || !Number.isFinite(ms) || ms <= 0) return;
228
+ const prev = latencies[id]?.emaMs;
229
+ const ema = prev ? Math.round(prev * (1 - latencyAlpha) + ms * latencyAlpha) : Math.round(ms);
230
+ latencies[id] = { emaMs: ema, lastMs: Math.round(ms), at: now(), count: (latencies[id]?.count ?? 0) + 1 };
231
+ await persistLatencies({ ...latencies });
232
+ }
233
+
234
+ function statuses() {
235
+ return { ...lastErrorAt };
236
+ }
237
+
238
+ function latencyStatuses() {
239
+ return { ...latencies };
240
+ }
241
+
242
+ return {
243
+ candidates,
244
+ candidatesFor,
245
+ recordError,
246
+ recordOk,
247
+ recordLatency,
248
+ statuses,
249
+ latencyStatuses,
250
+ isCooling,
251
+ errors: () => ({ ...lastErrorAt }),
252
+ latencies: () => ({ ...latencies }),
253
+ };
254
+ }
@@ -1,42 +1,42 @@
1
- import { clineHeaders } from "../providers/cline/headers.js";
2
- import { computeMetrics } from "../metrics.js";
3
- import { createTransport } from "../transport/index.js";
4
-
5
- function sseContent(obj) {
6
- const c = obj?.choices?.[0]?.delta?.content || obj?.choices?.[0]?.message?.content || "";
7
- return typeof c === "string" ? c : "";
8
- }
9
-
10
- export async function clineBenchOne({ baseUrl, model, accessToken, prompt, maxTokens, timeoutMs, fetchImpl }) {
11
- const tr = createTransport({ fetchImpl, keepAlive: false, retry: {}, timeoutMs });
12
- let ttfbMs = null;
13
- let totalMs = null;
14
- let content = "";
15
- try {
16
- const sid = `sess_bench_${Date.now()}`;
17
- const res = await tr.request({
18
- url: `${baseUrl}/api/v1/chat/completions`,
19
- method: "POST",
20
- headers: { ...clineHeaders(sid, accessToken), Accept: "text/event-stream" },
21
- body: { model, messages: [{ role: "user", content: prompt }], stream: true, max_tokens: maxTokens, session_id: sid, reasoning_effort: "high" },
22
- stream: true,
23
- });
24
- if (!res.ok) {
25
- let txt = "";
26
- try { txt = await res.text(); } catch {}
27
- const label = res.status === 401 ? "鉴权失败" : res.status === 429 ? "限流" : res.status >= 500 ? `上游错误 ${res.status}` : `HTTP ${res.status}`;
28
- return { id: model, ok: false, status: res.status, label, error: txt.slice(0, 300), ttfbMs, totalMs: res.totalMs, tps: null, charsPerSec: null, tokens: null };
29
- }
30
- for await (const ev of res.stream()) {
31
- try { content += sseContent(JSON.parse(ev)); } catch {}
32
- }
33
- ttfbMs = res.ttfbMs;
34
- totalMs = res.totalMs;
35
- const chars = content.length;
36
- const { tps, charsPerSec } = computeMetrics({ ttfbMs, totalMs, promptTokens: null, completionTokens: null, chars });
37
- return { id: model, ok: true, status: 200, label: "成功", ttfbMs, totalMs, tps, charsPerSec, tokens: null, chars };
38
- } catch (e) {
39
- const msg = e?.message || String(e);
40
- return { id: model, ok: false, label: /timeout|abort/i.test(msg) ? "超时" : "网络错误", error: msg.slice(0, 300), ttfbMs, totalMs, tps: null, charsPerSec: null, tokens: null };
41
- }
42
- }
1
+ import { clineHeaders } from "../providers/cline/headers.js";
2
+ import { computeMetrics } from "../metrics.js";
3
+ import { createTransport } from "../transport/index.js";
4
+
5
+ function sseContent(obj) {
6
+ const c = obj?.choices?.[0]?.delta?.content || obj?.choices?.[0]?.message?.content || "";
7
+ return typeof c === "string" ? c : "";
8
+ }
9
+
10
+ export async function clineBenchOne({ baseUrl, model, accessToken, prompt, maxTokens, timeoutMs, fetchImpl }) {
11
+ const tr = createTransport({ fetchImpl, keepAlive: false, retry: {}, timeoutMs });
12
+ let ttfbMs = null;
13
+ let totalMs = null;
14
+ let content = "";
15
+ try {
16
+ const sid = `sess_bench_${Date.now()}`;
17
+ const res = await tr.request({
18
+ url: `${baseUrl}/api/v1/chat/completions`,
19
+ method: "POST",
20
+ headers: { ...clineHeaders(sid, accessToken), Accept: "text/event-stream" },
21
+ body: { model, messages: [{ role: "user", content: prompt }], stream: true, max_tokens: maxTokens, session_id: sid, reasoning_effort: "high" },
22
+ stream: true,
23
+ });
24
+ if (!res.ok) {
25
+ let txt = "";
26
+ try { txt = await res.text(); } catch {}
27
+ const label = res.status === 401 ? "鉴权失败" : res.status === 429 ? "限流" : res.status >= 500 ? `上游错误 ${res.status}` : `HTTP ${res.status}`;
28
+ return { id: model, ok: false, status: res.status, label, error: txt.slice(0, 300), ttfbMs, totalMs: res.totalMs, tps: null, charsPerSec: null, tokens: null };
29
+ }
30
+ for await (const ev of res.stream()) {
31
+ try { content += sseContent(JSON.parse(ev)); } catch {}
32
+ }
33
+ ttfbMs = res.ttfbMs;
34
+ totalMs = res.totalMs;
35
+ const chars = content.length;
36
+ const { tps, charsPerSec } = computeMetrics({ ttfbMs, totalMs, promptTokens: null, completionTokens: null, chars });
37
+ return { id: model, ok: true, status: 200, label: "成功", ttfbMs, totalMs, tps, charsPerSec, tokens: null, chars };
38
+ } catch (e) {
39
+ const msg = e?.message || String(e);
40
+ return { id: model, ok: false, label: /timeout|abort/i.test(msg) ? "超时" : "网络错误", error: msg.slice(0, 300), ttfbMs, totalMs, tps: null, charsPerSec: null, tokens: null };
41
+ }
42
+ }