mslxdff 0.1.158 → 0.1.160

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (142) hide show
  1. package/bin/mslxdff.js +4 -4
  2. package/package.json +3 -3
  3. package/src/auto.js +254 -254
  4. package/src/bench/cline-bench.js +42 -42
  5. package/src/bench/probe.js +70 -70
  6. package/src/bench/report.js +162 -162
  7. package/src/bench/runner.js +77 -77
  8. package/src/bench/via-probe.js +124 -124
  9. package/src/bench/via-routes.js +87 -87
  10. package/src/bench/workbuddy-bench.js +54 -54
  11. package/src/chat/engine.js +160 -160
  12. package/src/chat/gateway.js +163 -163
  13. package/src/chat/orchestrator.js +234 -234
  14. package/src/chat/prompt.js +70 -70
  15. package/src/chat/repl.js +88 -88
  16. package/src/chat/terminal.js +135 -135
  17. package/src/chat/tools.js +306 -306
  18. package/src/chat-pipeline/auto-race.js +1 -1
  19. package/src/chat-pipeline/index.js +123 -123
  20. package/src/chat-pipeline/policy.js +76 -76
  21. package/src/chat-pipeline/serial-trial.js +210 -210
  22. package/src/cli/commands/group.js +249 -249
  23. package/src/cli/commands/model/picks.js +50 -50
  24. package/src/cli/commands/provider/bench-via.js +247 -247
  25. package/src/cli/commands/provider/bench.js +141 -141
  26. package/src/cli/commands/provider/index.js +124 -124
  27. package/src/cli/commands/provider/models.js +139 -139
  28. package/src/cli/commands/provider/qwenwork-login.js +119 -119
  29. package/src/cli/commands/stats.js +20 -5
  30. package/src/cli/commands/sync.js +232 -232
  31. package/src/cli/provider-row.js +2 -2
  32. package/src/cli/status.js +279 -279
  33. package/src/daemon.js +96 -96
  34. package/src/model-capabilities/enrich.js +86 -86
  35. package/src/model-capabilities/index.js +183 -183
  36. package/src/model-capabilities/parse.js +70 -70
  37. package/src/models.js +225 -225
  38. package/src/providers/cline/auth.js +228 -228
  39. package/src/providers/cline/chat.js +307 -307
  40. package/src/providers/cline.js +2 -2
  41. package/src/providers/keyring.js +60 -60
  42. package/src/providers/qoder/chat.js +183 -183
  43. package/src/providers/qoder/index.js +230 -230
  44. package/src/providers/qoder/sse.js +103 -103
  45. package/src/providers/qwenwork/account-store.js +133 -133
  46. package/src/providers/qwenwork/constants.js +67 -67
  47. package/src/providers/qwenwork/cosy.js +120 -120
  48. package/src/providers/qwenwork/crypto.js +218 -218
  49. package/src/providers/qwenwork/http.js +20 -20
  50. package/src/providers/qwenwork/index.js +327 -327
  51. package/src/providers/qwenwork/payload.js +142 -142
  52. package/src/providers/qwenwork/rsa.js +54 -54
  53. package/src/providers/qwenwork/sse.js +268 -268
  54. package/src/providers/qwenwork/stream.js +130 -130
  55. package/src/providers/qwenwork/upstream.js +120 -120
  56. package/src/providers/qwenwork.js +1 -1
  57. package/src/providers/registry.js +66 -66
  58. package/src/providers/workbuddy/chat.js +248 -248
  59. package/src/providers/workbuddy/reshape.js +152 -152
  60. package/src/providers/workbuddy.js +2 -2
  61. package/src/providers/zcode/chat.js +11 -10
  62. package/src/providers/zcode/const.js +1 -1
  63. package/src/providers/zcode/context-shape.js +162 -0
  64. package/src/providers/zcode/headers.js +38 -0
  65. package/src/reasoning.js +32 -32
  66. package/src/routes/chat/broadband-handler.js +3 -2
  67. package/src/routes/chat/gateway.js +46 -46
  68. package/src/routes/chat/hedge-handler.js +3 -0
  69. package/src/routes/chat/local-handler.js +2 -0
  70. package/src/routes/chat/peer-handler.js +2 -0
  71. package/src/routes/chat/relay-pipeline.js +264 -250
  72. package/src/routes/chat/via-route-handler.js +146 -144
  73. package/src/routes/hedge.js +255 -255
  74. package/src/routes/models-route.js +167 -167
  75. package/src/routes/peers.js +273 -273
  76. package/src/routes/stream-scan.js +86 -0
  77. package/src/routes/stream.js +393 -438
  78. package/src/runtime/bootstrap.js +45 -45
  79. package/src/runtime/provider-gate.js +33 -33
  80. package/src/runtime/providers-setup.js +165 -165
  81. package/src/server.js +64 -64
  82. package/src/state/schemas/allowlist.js +92 -92
  83. package/src/sync-opencode.js +280 -280
  84. package/src/transport/index.js +244 -244
  85. package/src/transport/pool.js +56 -56
  86. package/src/transport/retry.js +24 -24
  87. package/src/transport/sse.js +93 -93
  88. package/src/upstream-probe/display.js +52 -52
  89. package/src/upstream-probe/probe.js +49 -49
  90. package/src/upstream-probe/rotate.js +110 -110
  91. package/src/upstream-probe/start.js +45 -45
  92. package/src/upstream.js +289 -289
  93. package/src/usage/record.js +8 -1
  94. package/src/usage/report.js +39 -2
  95. package/docs/ARCHITECTURE.md +0 -426
  96. package/docs/FEATURE_TREE.md +0 -164
  97. package/docs/MOBILE.md +0 -82
  98. package/docs/adr/0001-reasoning-content-injection.md +0 -14
  99. package/docs/adr/0002-models-free-filter.md +0 -12
  100. package/docs/adr/0003-zero-state-no-auth.md +0 -10
  101. package/docs/adr/0004-bearer-token.md +0 -18
  102. package/docs/adr/0005-peer-mesh.md +0 -53
  103. package/docs/adr/0006-broadband-member.md +0 -103
  104. package/docs/adr/0007-multi-provider-prefix.md +0 -25
  105. package/docs/adr/0008-share-keys-to-peers.md +0 -52
  106. package/docs/adr/0009-chat-repl.md +0 -37
  107. package/docs/adr/0010-allowlist.md +0 -36
  108. package/docs/adr/0011-broadband-stream.md +0 -30
  109. package/docs/adr/0012-responses-endpoint-codex-sync.md +0 -48
  110. package/docs/adr/0013-node16-compat.md +0 -41
  111. package/docs/adr/0014-deepseek-provider.md +0 -52
  112. package/docs/adr/0015-upstream-probe-routing.md +0 -50
  113. package/docs/adr/0016-model-capabilities.md +0 -27
  114. package/docs/adr/0017-ai-sdk-upstream-engine.md +0 -59
  115. package/docs/adr/0018-zen-client-identity.md +0 -52
  116. package/docs/adr/0019-share-keys-always-lend.md +0 -59
  117. package/docs/adr/0020-zen-free-lane-agent-shape.md +0 -52
  118. package/docs/adr/0021-usage-report-jsonl.md +0 -47
  119. package/docs/adr/0022-models-capability-merge.md +0 -61
  120. package/docs/adr/0023-key-provider-default-direct.md +0 -58
  121. package/docs/adr/0024-node18-baseline.md +0 -63
  122. package/docs/adr/0025-workbuddy-authdir-follows-state.md +0 -71
  123. package/docs/adr/0026-cline-provider-id-unify.md +0 -67
  124. package/docs/adr/0027-codearts-provider.md +0 -48
  125. package/docs/adr/0028-traework-provider.md +0 -34
  126. package/docs/adr/0029-qoder-native-provider.md +0 -60
  127. package/docs/adr/0030-models-list-scoped-by-picks.md +0 -53
  128. package/docs/adr/0031-qoder-true-streaming.md +0 -50
  129. package/docs/adr/0032-generic-responses-channel.md +0 -72
  130. package/docs/adr/0033-cline-allowlist-auto-sync.md +0 -75
  131. package/docs/adr/0034-request-level-human-readable-observability.md +0 -60
  132. package/docs/adr/0035-sdk-channel-headers-timeout.md +0 -49
  133. package/docs/adr/0036-qoder-per-request-sticky-account.md +0 -82
  134. package/docs/adr/0037-qwenwork-independent-provider.md +0 -82
  135. package/docs/adr/0038-zcode-provider.md +0 -140
  136. package/docs/agents/domain.md +0 -51
  137. package/docs/agents/issue-tracker.md +0 -30
  138. package/docs/agents/triage-labels.md +0 -15
  139. package/docs/cli_help.md +0 -1391
  140. package/docs/cli_help_mini.md +0 -133
  141. package/docs/plans/bench-via-latency-2026-09-01.md +0 -215
  142. package/docs/plugins.md +0 -187
package/bin/mslxdff.js CHANGED
@@ -1,4 +1,4 @@
1
- #!/usr/bin/env node
2
- // thin adapter — all logic lives in src/cli (deep module, single inlet run)
3
- import { run } from "../src/cli/index.js";
4
- await run(process.argv.slice(2));
1
+ #!/usr/bin/env node
2
+ // thin adapter — all logic lives in src/cli (deep module, single inlet run)
3
+ import { run } from "../src/cli/index.js";
4
+ await run(process.argv.slice(2));
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "mslxdff",
3
- "version": "0.1.158",
3
+ "version": "0.1.160",
4
4
  "description": "测试项目,请勿使用。",
5
5
  "type": "module",
6
6
  "bin": {
@@ -18,8 +18,8 @@
18
18
  "bin/",
19
19
  "src/",
20
20
  "plugins/",
21
- "docs/",
22
- "README.md"
21
+ "README.md",
22
+ "docs/cli_help_mini.md"
23
23
  ],
24
24
  "keywords": [
25
25
  "test",
package/src/auto.js CHANGED
@@ -1,254 +1,254 @@
1
- import { statSync } from "node:fs";
2
- import { loadModelErrors, saveModelErrors, loadModelLatencies, saveModelLatencies, loadPreferredModel, loadModelPicks, saveModelPicks, defaultStateFile } from "./state.js";
3
-
4
- // 出厂默认首选模型(state.json 的 preferredModel / env MSLXDFF_PREFERRED_MODEL 可覆盖)
5
- export const DEFAULT_PREFERRED_MODEL = "big-pickle";
6
- // 兼容旧导出名:语义为"出厂默认",当前生效值请用 getPreferredModel()
7
- export const PREFERRED_MODEL = DEFAULT_PREFERRED_MODEL;
8
-
9
- // 当前生效的首选模型:state.json > env > 出厂默认;mtime 缓存保证 daemon 热生效
10
- const _prefCache = { mtimeMs: -1, file: null, value: null };
11
- export function getPreferredModel({ file = defaultStateFile() } = {}) {
12
- try {
13
- const st = statSync(file);
14
- if (_prefCache.file !== file || st.mtimeMs !== _prefCache.mtimeMs) {
15
- _prefCache.file = file;
16
- _prefCache.mtimeMs = st.mtimeMs;
17
- _prefCache.value = loadPreferredModel({ file });
18
- }
19
- } catch {
20
- _prefCache.file = file;
21
- _prefCache.mtimeMs = -1;
22
- _prefCache.value = null;
23
- }
24
- const env = (process.env.MSLXDFF_PREFERRED_MODEL || "").trim();
25
- return _prefCache.value || env || DEFAULT_PREFERRED_MODEL;
26
- }
27
-
28
- export const DEFAULT_AUTO_MODELS = [
29
- PREFERRED_MODEL,
30
- "mimo-v2.5-free",
31
- "deepseek-v4-flash-free",
32
- "ling-3.0-flash-fin-free",
33
- "nemotron-3-ultra-free",
34
- "nemotron-3.5-lightning-free",
35
- "muse-spark-1.3-contributor-free",
36
- ].filter((id, i, arr) => id && arr.indexOf(id) === i);
37
-
38
- export function isAutoModel(model) {
39
- return !model || model === "auto";
40
- }
41
-
42
- export const MODEL_STATUS = Object.freeze({
43
- NORMAL: "normal",
44
- LIMIT: "limit",
45
- ERROR: "error",
46
- });
47
-
48
- // Legacy modelErrors entries are bare timestamps ({id: ts}); newer ones are
49
- // objects ({id: {status, at, code}}). Normalize both to an entry object.
50
- // `slow` flags a model whose last request was slow (over the wall-clock
51
- // threshold) — those get a longer cooldown so they lie low until they recover.
52
- function normEntry(e) {
53
- if (typeof e === "number") return { status: MODEL_STATUS.ERROR, at: e, code: null, slow: false };
54
- if (e && typeof e === "object") {
55
- return {
56
- status: e.status || MODEL_STATUS.ERROR,
57
- at: typeof e.at === "number" ? e.at : 0,
58
- code: e.code ?? null,
59
- slow: Boolean(e.slow),
60
- };
61
- }
62
- return null;
63
- }
64
-
65
- export function classifyErrorEvent(evt = {}) {
66
- if (evt.slow) return MODEL_STATUS.ERROR;
67
- const code = Number(evt.status);
68
- if (code === 429) return MODEL_STATUS.LIMIT;
69
- const msg = String(evt.message || evt.note || "").toLowerCase();
70
- if (msg.includes("rate limit") || msg.includes("limit exceeded") || msg.includes("429")) {
71
- return MODEL_STATUS.LIMIT;
72
- }
73
- return MODEL_STATUS.ERROR;
74
- }
75
-
76
- export const DEFAULT_COOLDOWN_MS = 60_000;
77
- export const DEFAULT_SLOW_COOLDOWN_MS = 5 * 60_000;
78
- export const DEFAULT_LATENCY_ALPHA = 0.3;
79
-
80
- function effectiveCooldown(entry, slowCooldownMs, cooldownMs) {
81
- if (entry && entry.slow) return slowCooldownMs || 0;
82
- return cooldownMs || 0;
83
- }
84
-
85
- function inCooldown(id, errors, now, cooldownMs, slowCooldownMs) {
86
- const e = normEntry(errors[id]);
87
- if (!e || !(e.at > 0)) return false;
88
- if (e.status === MODEL_STATUS.NORMAL) return false;
89
- const cd = effectiveCooldown(e, slowCooldownMs, cooldownMs);
90
- return cd > 0 && now - e.at < cd;
91
- }
92
-
93
- // Latency EMA helpers
94
- function normLatency(e) {
95
- if (!e || typeof e !== "object") return null;
96
- const ema = Number(e.emaMs);
97
- return Number.isFinite(ema) && ema > 0 ? ema : null;
98
- }
99
-
100
- export function rankModels(ids, errors = {}, { now = Date.now(), cooldownMs = 0, slowCooldownMs = 0, latencies = {}, preferred } = {}) {
101
- const pref = preferred ?? getPreferredModel();
102
- // 最近一次成功的模型(NORMAL 且非慢且 at 最大),用于“上次成功优先”——慢模型即使刚成功也不应钉死
103
- let lastSuccessId = null;
104
- let lastSuccessAt = 0;
105
- for (const [id, e] of Object.entries(errors)) {
106
- const ne = normEntry(e);
107
- if (ne && ne.status === MODEL_STATUS.NORMAL && !ne.slow && ne.at > lastSuccessAt) {
108
- lastSuccessAt = ne.at;
109
- lastSuccessId = id;
110
- }
111
- }
112
- return [...new Set(ids)]
113
- .filter(Boolean)
114
- .map((id) => ({
115
- id,
116
- e: normEntry(errors[id]),
117
- err: normEntry(errors[id])?.at ?? 0,
118
- isPreferred: id === pref,
119
- isLastSuccess: id === lastSuccessId,
120
- cooling: inCooldown(id, errors, now, cooldownMs, slowCooldownMs),
121
- latency: normLatency(latencies[id]) ?? Number.MAX_SAFE_INTEGER,
122
- }))
123
- .sort(
124
- (a, b) =>
125
- (a.cooling ? 1 : 0) - (b.cooling ? 1 : 0) ||
126
- (b.isLastSuccess ? 1 : 0) - (a.isLastSuccess ? 1 : 0) ||
127
- (b.isPreferred ? 1 : 0) - (a.isPreferred ? 1 : 0) ||
128
- a.latency - b.latency ||
129
- a.err - b.err
130
- )
131
- .map((x) => x.id);
132
- }
133
-
134
- export function createAutoSelector({
135
- loadCandidates,
136
- file,
137
- now = () => Date.now(),
138
- cooldownMs = DEFAULT_COOLDOWN_MS,
139
- slowCooldownMs = DEFAULT_SLOW_COOLDOWN_MS,
140
- latencyAlpha = DEFAULT_LATENCY_ALPHA,
141
- errors: seedErrors,
142
- latencies: seedLatencies,
143
- persist = (errors, f = file) => saveModelErrors(errors, f ? { file: f } : {}),
144
- persistLatencies = (latencies, f = file) => saveModelLatencies(latencies, f ? { file: f } : {}),
145
- loadPicks = () => (file ? loadModelPicks({ file }) : []),
146
- persistPicks = (picks) => (file ? saveModelPicks(picks, { file }) : picks),
147
- } = {}) {
148
- const lastErrorAt = { ...(seedErrors ?? loadModelErrors(file ? { file } : {})) };
149
- const latencies = { ...(seedLatencies ?? loadModelLatencies(file ? { file } : {})) };
150
-
151
- async function loadList() {
152
- let list;
153
- try {
154
- const loaded = await loadCandidates?.();
155
- list = Array.isArray(loaded) && loaded.length ? loaded : DEFAULT_AUTO_MODELS;
156
- } catch {
157
- list = DEFAULT_AUTO_MODELS;
158
- }
159
- return [...new Set(list)].filter(Boolean);
160
- }
161
-
162
- // 勾选集 = auto 候选池白名单:只在勾选的模型里择优;空勾选或勾选中无可用模型时回退全量
163
- async function pickedPool(list) {
164
- const picks = loadPicks();
165
- if (!picks.length) return list;
166
- const pickedSet = new Set(picks);
167
- const filtered = list.filter((id) => pickedSet.has(id));
168
- return filtered.length ? filtered : list;
169
- }
170
-
171
- async function candidates() {
172
- const list = await loadList();
173
- const pool = await pickedPool(list);
174
- return rankModels(pool, lastErrorAt, { now: now(), cooldownMs, slowCooldownMs, latencies, preferred: getPreferredModel({ file: file ?? undefined }) });
175
- }
176
-
177
- async function candidatesFor(requested) {
178
- if (!requested) return candidates();
179
- const list = await loadList();
180
- // 显式指定某模型 = 认可它,自动加入勾选集(仅当它是真实上游 free 模型时,避免垃圾 id 污染)
181
- const picks = loadPicks();
182
- if (list.includes(requested) && !picks.includes(requested) && persistPicks) {
183
- await persistPicks([...picks, requested]);
184
- }
185
- const all = list.includes(requested) ? list : [requested, ...list];
186
- const others = rankModels(all.filter((id) => id !== requested), lastErrorAt, {
187
- now: now(),
188
- cooldownMs,
189
- slowCooldownMs,
190
- latencies,
191
- preferred: getPreferredModel({ file: file ?? undefined }),
192
- });
193
- // 显式指定模型:严格优先,永不因冷却被挤到最后(原设计:A deepseek 失败 → B/D deepseek 并发 → 都失败才 fallback)
194
- // 冷却仅影响 auto 的择优,不影响指定模型的“很难被更改”语义
195
- return [requested, ...others];
196
- }
197
-
198
- function isCooling(id) {
199
- return inCooldown(id, lastErrorAt, now(), cooldownMs, slowCooldownMs);
200
- }
201
-
202
- async function recordError(id, evt = {}) {
203
- if (!id) return;
204
- lastErrorAt[id] = {
205
- status: classifyErrorEvent(evt),
206
- at: now(),
207
- code: Number.isInteger(Number(evt.status)) ? Number(evt.status) : null,
208
- slow: Boolean(evt.slow),
209
- };
210
- await persist({ ...lastErrorAt });
211
- }
212
-
213
- async function recordOk(id, evt = {}) {
214
- if (!id) return;
215
- lastErrorAt[id] = { status: MODEL_STATUS.NORMAL, at: now(), code: 200, slow: false };
216
- await persist({ ...lastErrorAt });
217
- const ms = Number(evt.latencyMs ?? evt.totalMs ?? evt.elapsedMs);
218
- if (Number.isFinite(ms) && ms > 0) {
219
- const prev = latencies[id]?.emaMs;
220
- const ema = prev ? Math.round(prev * (1 - latencyAlpha) + ms * latencyAlpha) : Math.round(ms);
221
- latencies[id] = { emaMs: ema, lastMs: Math.round(ms), at: now(), count: (latencies[id]?.count ?? 0) + 1 };
222
- await persistLatencies({ ...latencies });
223
- }
224
- }
225
-
226
- async function recordLatency(id, ms) {
227
- if (!id || !Number.isFinite(ms) || ms <= 0) return;
228
- const prev = latencies[id]?.emaMs;
229
- const ema = prev ? Math.round(prev * (1 - latencyAlpha) + ms * latencyAlpha) : Math.round(ms);
230
- latencies[id] = { emaMs: ema, lastMs: Math.round(ms), at: now(), count: (latencies[id]?.count ?? 0) + 1 };
231
- await persistLatencies({ ...latencies });
232
- }
233
-
234
- function statuses() {
235
- return { ...lastErrorAt };
236
- }
237
-
238
- function latencyStatuses() {
239
- return { ...latencies };
240
- }
241
-
242
- return {
243
- candidates,
244
- candidatesFor,
245
- recordError,
246
- recordOk,
247
- recordLatency,
248
- statuses,
249
- latencyStatuses,
250
- isCooling,
251
- errors: () => ({ ...lastErrorAt }),
252
- latencies: () => ({ ...latencies }),
253
- };
254
- }
1
+ import { statSync } from "node:fs";
2
+ import { loadModelErrors, saveModelErrors, loadModelLatencies, saveModelLatencies, loadPreferredModel, loadModelPicks, saveModelPicks, defaultStateFile } from "./state.js";
3
+
4
+ // 出厂默认首选模型(state.json 的 preferredModel / env MSLXDFF_PREFERRED_MODEL 可覆盖)
5
+ export const DEFAULT_PREFERRED_MODEL = "big-pickle";
6
+ // 兼容旧导出名:语义为"出厂默认",当前生效值请用 getPreferredModel()
7
+ export const PREFERRED_MODEL = DEFAULT_PREFERRED_MODEL;
8
+
9
+ // 当前生效的首选模型:state.json > env > 出厂默认;mtime 缓存保证 daemon 热生效
10
+ const _prefCache = { mtimeMs: -1, file: null, value: null };
11
+ export function getPreferredModel({ file = defaultStateFile() } = {}) {
12
+ try {
13
+ const st = statSync(file);
14
+ if (_prefCache.file !== file || st.mtimeMs !== _prefCache.mtimeMs) {
15
+ _prefCache.file = file;
16
+ _prefCache.mtimeMs = st.mtimeMs;
17
+ _prefCache.value = loadPreferredModel({ file });
18
+ }
19
+ } catch {
20
+ _prefCache.file = file;
21
+ _prefCache.mtimeMs = -1;
22
+ _prefCache.value = null;
23
+ }
24
+ const env = (process.env.MSLXDFF_PREFERRED_MODEL || "").trim();
25
+ return _prefCache.value || env || DEFAULT_PREFERRED_MODEL;
26
+ }
27
+
28
+ export const DEFAULT_AUTO_MODELS = [
29
+ PREFERRED_MODEL,
30
+ "mimo-v2.5-free",
31
+ "deepseek-v4-flash-free",
32
+ "ling-3.0-flash-fin-free",
33
+ "nemotron-3-ultra-free",
34
+ "nemotron-3.5-lightning-free",
35
+ "muse-spark-1.3-contributor-free",
36
+ ].filter((id, i, arr) => id && arr.indexOf(id) === i);
37
+
38
+ export function isAutoModel(model) {
39
+ return !model || model === "auto";
40
+ }
41
+
42
+ export const MODEL_STATUS = Object.freeze({
43
+ NORMAL: "normal",
44
+ LIMIT: "limit",
45
+ ERROR: "error",
46
+ });
47
+
48
+ // Legacy modelErrors entries are bare timestamps ({id: ts}); newer ones are
49
+ // objects ({id: {status, at, code}}). Normalize both to an entry object.
50
+ // `slow` flags a model whose last request was slow (over the wall-clock
51
+ // threshold) — those get a longer cooldown so they lie low until they recover.
52
+ function normEntry(e) {
53
+ if (typeof e === "number") return { status: MODEL_STATUS.ERROR, at: e, code: null, slow: false };
54
+ if (e && typeof e === "object") {
55
+ return {
56
+ status: e.status || MODEL_STATUS.ERROR,
57
+ at: typeof e.at === "number" ? e.at : 0,
58
+ code: e.code ?? null,
59
+ slow: Boolean(e.slow),
60
+ };
61
+ }
62
+ return null;
63
+ }
64
+
65
+ export function classifyErrorEvent(evt = {}) {
66
+ if (evt.slow) return MODEL_STATUS.ERROR;
67
+ const code = Number(evt.status);
68
+ if (code === 429) return MODEL_STATUS.LIMIT;
69
+ const msg = String(evt.message || evt.note || "").toLowerCase();
70
+ if (msg.includes("rate limit") || msg.includes("limit exceeded") || msg.includes("429")) {
71
+ return MODEL_STATUS.LIMIT;
72
+ }
73
+ return MODEL_STATUS.ERROR;
74
+ }
75
+
76
+ export const DEFAULT_COOLDOWN_MS = 60_000;
77
+ export const DEFAULT_SLOW_COOLDOWN_MS = 5 * 60_000;
78
+ export const DEFAULT_LATENCY_ALPHA = 0.3;
79
+
80
+ function effectiveCooldown(entry, slowCooldownMs, cooldownMs) {
81
+ if (entry && entry.slow) return slowCooldownMs || 0;
82
+ return cooldownMs || 0;
83
+ }
84
+
85
+ function inCooldown(id, errors, now, cooldownMs, slowCooldownMs) {
86
+ const e = normEntry(errors[id]);
87
+ if (!e || !(e.at > 0)) return false;
88
+ if (e.status === MODEL_STATUS.NORMAL) return false;
89
+ const cd = effectiveCooldown(e, slowCooldownMs, cooldownMs);
90
+ return cd > 0 && now - e.at < cd;
91
+ }
92
+
93
+ // Latency EMA helpers
94
+ function normLatency(e) {
95
+ if (!e || typeof e !== "object") return null;
96
+ const ema = Number(e.emaMs);
97
+ return Number.isFinite(ema) && ema > 0 ? ema : null;
98
+ }
99
+
100
+ export function rankModels(ids, errors = {}, { now = Date.now(), cooldownMs = 0, slowCooldownMs = 0, latencies = {}, preferred } = {}) {
101
+ const pref = preferred ?? getPreferredModel();
102
+ // 最近一次成功的模型(NORMAL 且非慢且 at 最大),用于“上次成功优先”——慢模型即使刚成功也不应钉死
103
+ let lastSuccessId = null;
104
+ let lastSuccessAt = 0;
105
+ for (const [id, e] of Object.entries(errors)) {
106
+ const ne = normEntry(e);
107
+ if (ne && ne.status === MODEL_STATUS.NORMAL && !ne.slow && ne.at > lastSuccessAt) {
108
+ lastSuccessAt = ne.at;
109
+ lastSuccessId = id;
110
+ }
111
+ }
112
+ return [...new Set(ids)]
113
+ .filter(Boolean)
114
+ .map((id) => ({
115
+ id,
116
+ e: normEntry(errors[id]),
117
+ err: normEntry(errors[id])?.at ?? 0,
118
+ isPreferred: id === pref,
119
+ isLastSuccess: id === lastSuccessId,
120
+ cooling: inCooldown(id, errors, now, cooldownMs, slowCooldownMs),
121
+ latency: normLatency(latencies[id]) ?? Number.MAX_SAFE_INTEGER,
122
+ }))
123
+ .sort(
124
+ (a, b) =>
125
+ (a.cooling ? 1 : 0) - (b.cooling ? 1 : 0) ||
126
+ (b.isLastSuccess ? 1 : 0) - (a.isLastSuccess ? 1 : 0) ||
127
+ (b.isPreferred ? 1 : 0) - (a.isPreferred ? 1 : 0) ||
128
+ a.latency - b.latency ||
129
+ a.err - b.err
130
+ )
131
+ .map((x) => x.id);
132
+ }
133
+
134
+ export function createAutoSelector({
135
+ loadCandidates,
136
+ file,
137
+ now = () => Date.now(),
138
+ cooldownMs = DEFAULT_COOLDOWN_MS,
139
+ slowCooldownMs = DEFAULT_SLOW_COOLDOWN_MS,
140
+ latencyAlpha = DEFAULT_LATENCY_ALPHA,
141
+ errors: seedErrors,
142
+ latencies: seedLatencies,
143
+ persist = (errors, f = file) => saveModelErrors(errors, f ? { file: f } : {}),
144
+ persistLatencies = (latencies, f = file) => saveModelLatencies(latencies, f ? { file: f } : {}),
145
+ loadPicks = () => (file ? loadModelPicks({ file }) : []),
146
+ persistPicks = (picks) => (file ? saveModelPicks(picks, { file }) : picks),
147
+ } = {}) {
148
+ const lastErrorAt = { ...(seedErrors ?? loadModelErrors(file ? { file } : {})) };
149
+ const latencies = { ...(seedLatencies ?? loadModelLatencies(file ? { file } : {})) };
150
+
151
+ async function loadList() {
152
+ let list;
153
+ try {
154
+ const loaded = await loadCandidates?.();
155
+ list = Array.isArray(loaded) && loaded.length ? loaded : DEFAULT_AUTO_MODELS;
156
+ } catch {
157
+ list = DEFAULT_AUTO_MODELS;
158
+ }
159
+ return [...new Set(list)].filter(Boolean);
160
+ }
161
+
162
+ // 勾选集 = auto 候选池白名单:只在勾选的模型里择优;空勾选或勾选中无可用模型时回退全量
163
+ async function pickedPool(list) {
164
+ const picks = loadPicks();
165
+ if (!picks.length) return list;
166
+ const pickedSet = new Set(picks);
167
+ const filtered = list.filter((id) => pickedSet.has(id));
168
+ return filtered.length ? filtered : list;
169
+ }
170
+
171
+ async function candidates() {
172
+ const list = await loadList();
173
+ const pool = await pickedPool(list);
174
+ return rankModels(pool, lastErrorAt, { now: now(), cooldownMs, slowCooldownMs, latencies, preferred: getPreferredModel({ file: file ?? undefined }) });
175
+ }
176
+
177
+ async function candidatesFor(requested) {
178
+ if (!requested) return candidates();
179
+ const list = await loadList();
180
+ // 显式指定某模型 = 认可它,自动加入勾选集(仅当它是真实上游 free 模型时,避免垃圾 id 污染)
181
+ const picks = loadPicks();
182
+ if (list.includes(requested) && !picks.includes(requested) && persistPicks) {
183
+ await persistPicks([...picks, requested]);
184
+ }
185
+ const all = list.includes(requested) ? list : [requested, ...list];
186
+ const others = rankModels(all.filter((id) => id !== requested), lastErrorAt, {
187
+ now: now(),
188
+ cooldownMs,
189
+ slowCooldownMs,
190
+ latencies,
191
+ preferred: getPreferredModel({ file: file ?? undefined }),
192
+ });
193
+ // 显式指定模型:严格优先,永不因冷却被挤到最后(原设计:A deepseek 失败 → B/D deepseek 并发 → 都失败才 fallback)
194
+ // 冷却仅影响 auto 的择优,不影响指定模型的“很难被更改”语义
195
+ return [requested, ...others];
196
+ }
197
+
198
+ function isCooling(id) {
199
+ return inCooldown(id, lastErrorAt, now(), cooldownMs, slowCooldownMs);
200
+ }
201
+
202
+ async function recordError(id, evt = {}) {
203
+ if (!id) return;
204
+ lastErrorAt[id] = {
205
+ status: classifyErrorEvent(evt),
206
+ at: now(),
207
+ code: Number.isInteger(Number(evt.status)) ? Number(evt.status) : null,
208
+ slow: Boolean(evt.slow),
209
+ };
210
+ await persist({ ...lastErrorAt });
211
+ }
212
+
213
+ async function recordOk(id, evt = {}) {
214
+ if (!id) return;
215
+ lastErrorAt[id] = { status: MODEL_STATUS.NORMAL, at: now(), code: 200, slow: false };
216
+ await persist({ ...lastErrorAt });
217
+ const ms = Number(evt.latencyMs ?? evt.totalMs ?? evt.elapsedMs);
218
+ if (Number.isFinite(ms) && ms > 0) {
219
+ const prev = latencies[id]?.emaMs;
220
+ const ema = prev ? Math.round(prev * (1 - latencyAlpha) + ms * latencyAlpha) : Math.round(ms);
221
+ latencies[id] = { emaMs: ema, lastMs: Math.round(ms), at: now(), count: (latencies[id]?.count ?? 0) + 1 };
222
+ await persistLatencies({ ...latencies });
223
+ }
224
+ }
225
+
226
+ async function recordLatency(id, ms) {
227
+ if (!id || !Number.isFinite(ms) || ms <= 0) return;
228
+ const prev = latencies[id]?.emaMs;
229
+ const ema = prev ? Math.round(prev * (1 - latencyAlpha) + ms * latencyAlpha) : Math.round(ms);
230
+ latencies[id] = { emaMs: ema, lastMs: Math.round(ms), at: now(), count: (latencies[id]?.count ?? 0) + 1 };
231
+ await persistLatencies({ ...latencies });
232
+ }
233
+
234
+ function statuses() {
235
+ return { ...lastErrorAt };
236
+ }
237
+
238
+ function latencyStatuses() {
239
+ return { ...latencies };
240
+ }
241
+
242
+ return {
243
+ candidates,
244
+ candidatesFor,
245
+ recordError,
246
+ recordOk,
247
+ recordLatency,
248
+ statuses,
249
+ latencyStatuses,
250
+ isCooling,
251
+ errors: () => ({ ...lastErrorAt }),
252
+ latencies: () => ({ ...latencies }),
253
+ };
254
+ }
@@ -1,42 +1,42 @@
1
- import { clineHeaders } from "../providers/cline/headers.js";
2
- import { computeMetrics } from "../metrics.js";
3
- import { createTransport } from "../transport/index.js";
4
-
5
- function sseContent(obj) {
6
- const c = obj?.choices?.[0]?.delta?.content || obj?.choices?.[0]?.message?.content || "";
7
- return typeof c === "string" ? c : "";
8
- }
9
-
10
- export async function clineBenchOne({ baseUrl, model, accessToken, prompt, maxTokens, timeoutMs, fetchImpl }) {
11
- const tr = createTransport({ fetchImpl, keepAlive: false, retry: {}, timeoutMs });
12
- let ttfbMs = null;
13
- let totalMs = null;
14
- let content = "";
15
- try {
16
- const sid = `sess_bench_${Date.now()}`;
17
- const res = await tr.request({
18
- url: `${baseUrl}/api/v1/chat/completions`,
19
- method: "POST",
20
- headers: { ...clineHeaders(sid, accessToken), Accept: "text/event-stream" },
21
- body: { model, messages: [{ role: "user", content: prompt }], stream: true, max_tokens: maxTokens, session_id: sid, reasoning_effort: "high" },
22
- stream: true,
23
- });
24
- if (!res.ok) {
25
- let txt = "";
26
- try { txt = await res.text(); } catch {}
27
- const label = res.status === 401 ? "鉴权失败" : res.status === 429 ? "限流" : res.status >= 500 ? `上游错误 ${res.status}` : `HTTP ${res.status}`;
28
- return { id: model, ok: false, status: res.status, label, error: txt.slice(0, 300), ttfbMs, totalMs: res.totalMs, tps: null, charsPerSec: null, tokens: null };
29
- }
30
- for await (const ev of res.stream()) {
31
- try { content += sseContent(JSON.parse(ev)); } catch {}
32
- }
33
- ttfbMs = res.ttfbMs;
34
- totalMs = res.totalMs;
35
- const chars = content.length;
36
- const { tps, charsPerSec } = computeMetrics({ ttfbMs, totalMs, promptTokens: null, completionTokens: null, chars });
37
- return { id: model, ok: true, status: 200, label: "成功", ttfbMs, totalMs, tps, charsPerSec, tokens: null, chars };
38
- } catch (e) {
39
- const msg = e?.message || String(e);
40
- return { id: model, ok: false, label: /timeout|abort/i.test(msg) ? "超时" : "网络错误", error: msg.slice(0, 300), ttfbMs, totalMs, tps: null, charsPerSec: null, tokens: null };
41
- }
42
- }
1
+ import { clineHeaders } from "../providers/cline/headers.js";
2
+ import { computeMetrics } from "../metrics.js";
3
+ import { createTransport } from "../transport/index.js";
4
+
5
+ function sseContent(obj) {
6
+ const c = obj?.choices?.[0]?.delta?.content || obj?.choices?.[0]?.message?.content || "";
7
+ return typeof c === "string" ? c : "";
8
+ }
9
+
10
+ export async function clineBenchOne({ baseUrl, model, accessToken, prompt, maxTokens, timeoutMs, fetchImpl }) {
11
+ const tr = createTransport({ fetchImpl, keepAlive: false, retry: {}, timeoutMs });
12
+ let ttfbMs = null;
13
+ let totalMs = null;
14
+ let content = "";
15
+ try {
16
+ const sid = `sess_bench_${Date.now()}`;
17
+ const res = await tr.request({
18
+ url: `${baseUrl}/api/v1/chat/completions`,
19
+ method: "POST",
20
+ headers: { ...clineHeaders(sid, accessToken), Accept: "text/event-stream" },
21
+ body: { model, messages: [{ role: "user", content: prompt }], stream: true, max_tokens: maxTokens, session_id: sid, reasoning_effort: "high" },
22
+ stream: true,
23
+ });
24
+ if (!res.ok) {
25
+ let txt = "";
26
+ try { txt = await res.text(); } catch {}
27
+ const label = res.status === 401 ? "鉴权失败" : res.status === 429 ? "限流" : res.status >= 500 ? `上游错误 ${res.status}` : `HTTP ${res.status}`;
28
+ return { id: model, ok: false, status: res.status, label, error: txt.slice(0, 300), ttfbMs, totalMs: res.totalMs, tps: null, charsPerSec: null, tokens: null };
29
+ }
30
+ for await (const ev of res.stream()) {
31
+ try { content += sseContent(JSON.parse(ev)); } catch {}
32
+ }
33
+ ttfbMs = res.ttfbMs;
34
+ totalMs = res.totalMs;
35
+ const chars = content.length;
36
+ const { tps, charsPerSec } = computeMetrics({ ttfbMs, totalMs, promptTokens: null, completionTokens: null, chars });
37
+ return { id: model, ok: true, status: 200, label: "成功", ttfbMs, totalMs, tps, charsPerSec, tokens: null, chars };
38
+ } catch (e) {
39
+ const msg = e?.message || String(e);
40
+ return { id: model, ok: false, label: /timeout|abort/i.test(msg) ? "超时" : "网络错误", error: msg.slice(0, 300), ttfbMs, totalMs, tps: null, charsPerSec: null, tokens: null };
41
+ }
42
+ }