mslxdff 0.1.156 → 0.1.159

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (97) hide show
  1. package/bin/mslxdff.js +4 -4
  2. package/docs/cli_help_mini.md +133 -0
  3. package/package.json +3 -3
  4. package/src/auto.js +254 -254
  5. package/src/bench/cline-bench.js +42 -42
  6. package/src/bench/probe.js +70 -70
  7. package/src/bench/report.js +162 -162
  8. package/src/bench/runner.js +77 -77
  9. package/src/bench/via-probe.js +124 -124
  10. package/src/bench/via-routes.js +87 -87
  11. package/src/bench/workbuddy-bench.js +54 -54
  12. package/src/chat/engine.js +160 -160
  13. package/src/chat/gateway.js +163 -163
  14. package/src/chat/orchestrator.js +234 -234
  15. package/src/chat/prompt.js +70 -70
  16. package/src/chat/repl.js +88 -88
  17. package/src/chat/terminal.js +135 -135
  18. package/src/chat/tools.js +306 -306
  19. package/src/chat-pipeline/index.js +123 -123
  20. package/src/chat-pipeline/policy.js +76 -76
  21. package/src/chat-pipeline/serial-trial.js +210 -210
  22. package/src/cli/commands/group.js +249 -249
  23. package/src/cli/commands/model/list-providers.js +1 -1
  24. package/src/cli/commands/model/picks.js +50 -50
  25. package/src/cli/commands/provider/bench-via.js +247 -247
  26. package/src/cli/commands/provider/bench.js +141 -141
  27. package/src/cli/commands/provider/index.js +124 -116
  28. package/src/cli/commands/provider/models.js +139 -128
  29. package/src/cli/commands/provider/qwenwork-login.js +119 -0
  30. package/src/cli/commands/provider/zcode-login.js +77 -0
  31. package/src/cli/commands/provider/zcode-quota.js +55 -0
  32. package/src/cli/commands/sync.js +232 -232
  33. package/src/cli/provider-row.js +2 -2
  34. package/src/cli/status.js +279 -279
  35. package/src/daemon.js +96 -96
  36. package/src/model-capabilities/enrich.js +86 -86
  37. package/src/model-capabilities/index.js +183 -183
  38. package/src/model-capabilities/parse.js +70 -70
  39. package/src/models.js +225 -225
  40. package/src/providers/classify.js +1 -1
  41. package/src/providers/cline/auth.js +228 -228
  42. package/src/providers/cline/chat.js +307 -307
  43. package/src/providers/cline.js +2 -2
  44. package/src/providers/keyring.js +60 -56
  45. package/src/providers/qoder/chat.js +183 -174
  46. package/src/providers/qoder/index.js +230 -185
  47. package/src/providers/qoder/sse.js +103 -62
  48. package/src/providers/qwenwork/account-store.js +133 -0
  49. package/src/providers/qwenwork/constants.js +67 -0
  50. package/src/providers/qwenwork/cosy.js +120 -0
  51. package/src/providers/qwenwork/crypto.js +218 -0
  52. package/src/providers/qwenwork/http.js +20 -0
  53. package/src/providers/qwenwork/index.js +327 -0
  54. package/src/providers/qwenwork/payload.js +142 -0
  55. package/src/providers/qwenwork/rsa.js +54 -0
  56. package/src/providers/qwenwork/sse.js +268 -0
  57. package/src/providers/qwenwork/stream.js +130 -0
  58. package/src/providers/qwenwork/upstream.js +120 -0
  59. package/src/providers/qwenwork.js +1 -0
  60. package/src/providers/registry.js +66 -56
  61. package/src/providers/share-keys.js +2 -2
  62. package/src/providers/workbuddy/chat.js +248 -248
  63. package/src/providers/workbuddy/reshape.js +152 -152
  64. package/src/providers/workbuddy.js +2 -2
  65. package/src/providers/zcode/account-store.js +129 -0
  66. package/src/providers/zcode/auth.js +28 -0
  67. package/src/providers/zcode/chat.js +171 -0
  68. package/src/providers/zcode/const.js +54 -0
  69. package/src/providers/zcode/headers.js +55 -0
  70. package/src/providers/zcode/index.js +124 -0
  71. package/src/providers/zcode/models.js +65 -0
  72. package/src/providers/zcode/oauth.js +120 -0
  73. package/src/providers/zcode/quota.js +176 -0
  74. package/src/providers/zcode/sse.js +179 -0
  75. package/src/reasoning.js +32 -32
  76. package/src/routes/chat/gateway.js +46 -46
  77. package/src/routes/chat/relay-pipeline.js +250 -250
  78. package/src/routes/chat/via-route-handler.js +144 -144
  79. package/src/routes/hedge.js +255 -255
  80. package/src/routes/models-route.js +167 -167
  81. package/src/routes/peers.js +273 -273
  82. package/src/routes/stream.js +438 -438
  83. package/src/runtime/bootstrap.js +45 -45
  84. package/src/runtime/provider-gate.js +33 -30
  85. package/src/runtime/providers-setup.js +165 -165
  86. package/src/server.js +64 -64
  87. package/src/state/schemas/allowlist.js +92 -92
  88. package/src/sync-opencode.js +280 -280
  89. package/src/transport/index.js +244 -244
  90. package/src/transport/pool.js +56 -56
  91. package/src/transport/retry.js +24 -24
  92. package/src/transport/sse.js +93 -93
  93. package/src/upstream-probe/display.js +52 -52
  94. package/src/upstream-probe/probe.js +49 -49
  95. package/src/upstream-probe/rotate.js +110 -110
  96. package/src/upstream-probe/start.js +45 -45
  97. package/src/upstream.js +289 -289
@@ -1,250 +1,250 @@
1
- import { runHook } from "../../plugins.js";
2
- import { recordModelStats } from "../../state.js";
3
- import { normalizeFullId } from "../../providers/model-id.js";
4
- import { computeMetrics } from "../../metrics.js";
5
- import { recordChatUsage } from "../../usage/record.js"; // 窗口报表唯一写入点(canonical 名单记防双计)— 见 .agents/notes/implemented/feature/2026-09-19-usage-report-jsonl.md
6
- import { recordOutput, computeOutputRow } from "../../providers/cline/usage.js"; // cline 专属旁路统计(账号×模型,流式也覆盖)
7
- import { upstreamEcho } from "../../model-trace.js"; // 回显头→日志字段(谁上的/哪个号/为什么/是否冷却)单一来源
8
-
9
- // 唯一/最后候选没有 failover 去向:首块闸门退化为纯"防连接泄漏",放宽避免误杀慢模型
10
- // (参考 opencode:zen 通道不设超时;openai responses 硬编码 300s headerTimeout)
11
- export const LAST_CANDIDATE_TIMEOUT_MS = (() => {
12
- const n = Number(process.env.MSLXDFF_LAST_CANDIDATE_TIMEOUT_MS);
13
- return Number.isInteger(n) && n >= 0 ? n : 120_000;
14
- })();
15
-
16
- /** 空转 200 判定(serial-trial 同模型重试用):与 execute 内 _emptyTurn 产生的 lastErr 同源 */
17
- export function isEmptyTurnError(err) {
18
- return Number(err?.status) === 502 && String(err?.message || "").startsWith("EMPTY_MODEL_RESPONSE");
19
- }
20
-
21
- /**
22
- * RelayPipeline 深模块
23
- * 把 5 个 handler 各自的 fallback→relay→scoring→事件 6段流水收敛为单一真相。
24
- * 对外 1 接口:createRelayPipeline(deps) => { execute(ctx) }
25
- * 设计要点:全部外部可注入,便于在 pipeline seam 上做行为测试;constants 注入便于单测加速。
26
- */
27
- export function createRelayPipeline({
28
- relay,
29
- buildFallbackInfo,
30
- auto,
31
- plugins,
32
- evt,
33
- mark,
34
- logCall,
35
- logError,
36
- constants,
37
- startedAt: defaultStartedAt,
38
- stages: defaultStages,
39
- perfNow,
40
- } = {}) {
41
- const C = {
42
- STREAM_TIMEOUT_MS: 25_000,
43
- SLOW_TOTAL_MS: 20_000,
44
- STALL_TIMEOUT_MS: 0,
45
- SCORE_STALL_MS: 15_000,
46
- ...(constants || {}),
47
- };
48
- const _relay = relay;
49
- const _build = buildFallbackInfo;
50
- const _evt = evt || (() => {});
51
- const _mark = mark || (() => {});
52
- const _logCall = logCall || (() => {});
53
- const _logError = logError || (() => {});
54
- const _perfNow = perfNow || (() => Date.now());
55
-
56
- async function execute({
57
- res,
58
- upRes,
59
- body,
60
- requested,
61
- actual,
62
- lastErr,
63
- via,
64
- lockModel,
65
- useAuto,
66
- handlerCtx,
67
- mark: m2,
68
- perf0,
69
- stages: s2,
70
- startedAt: sa2,
71
- streamTimeoutMs: ctxStreamTimeoutMs,
72
- } = {}) {
73
- const markFn = m2 || _mark;
74
- const curStartedAt = sa2 ?? defaultStartedAt ?? Date.now();
75
- const curStages = s2 ?? defaultStages ?? [];
76
- const reqId = handlerCtx?.reqId;
77
- const hops = handlerCtx?.hops;
78
-
79
- // 1. logCall(pre) — 保持原 handler 的 logCall→fallback→relay-start 时序
80
- try { _logCall(actual, upRes?.status); } catch {}
81
- // 2. fallback + relay-start
82
- let fallback = null;
83
- try {
84
- if (_build) fallback = _build({ requested, actual, lastErr, via, useAuto, lockModel });
85
- } catch {}
86
- if (fallback?.fallback) {
87
- _evt("fallback-notice", { reqId, requested, actual, reason: fallback.reason, notice: fallback.notice, via, fallback: true });
88
- }
89
- _evt("relay-start", { reqId, model: actual, via, isStream: Boolean(body?.stream), fallback });
90
-
91
- // 3. relay(唯一/最后候选:无 failover 去向 → 闸门放宽到防泄漏级别;显式 streamTimeoutMs 优先)
92
- const orderLen = handlerCtx?.orderLen;
93
- const curIdx = handlerCtx?.idx;
94
- const isLastCandidate =
95
- Number.isInteger(orderLen) && orderLen > 0 &&
96
- (orderLen === 1 || (Number.isInteger(curIdx) && curIdx >= orderLen - 1));
97
- const streamTimeoutMs = Number.isInteger(ctxStreamTimeoutMs) && ctxStreamTimeoutMs >= 0
98
- ? ctxStreamTimeoutMs
99
- : (isLastCandidate ? LAST_CANDIDATE_TIMEOUT_MS : C.STREAM_TIMEOUT_MS);
100
- const out = await _relay(res, upRes, body, {
101
- fallback,
102
- streamTimeoutMs,
103
- onFirstChunk: (delta) => {
104
- try { markFn(`ttf-${actual}`); } catch {}
105
- _evt("relay-first-chunk", { reqId, model: actual, ttfMs: delta, via });
106
- if (plugins?.length) runHook(plugins, "relay:first-chunk", { reqId, requested, model: actual, via, ttfMs: delta }).catch(() => {});
107
- },
108
- onDownstreamAbort: () => {
109
- _evt("client-abort", { reqId, model: actual, totalMs: Math.round(_perfNow() - (perf0 ?? 0)), stages: [...curStages] });
110
- },
111
- });
112
-
113
- // 4. relay-done
114
- _evt("relay-done", {
115
- reqId,
116
- model: actual,
117
- via,
118
- status: out.status,
119
- ttfMs: out.ttfMs,
120
- totalMs: out.totalMs,
121
- aborted: out.aborted,
122
- interrupted: out.interrupted ?? false,
123
- timedOut: out.timedOut ?? false,
124
- ...upstreamEcho(upRes),
125
- detail: out.detail ?? null,
126
- });
127
-
128
- // 5a0. 空转 200:流正常结束但零正文零工具调用 → 客户端会报 EMPTY_MODEL_RESPONSE
129
- // ("The model ended its turn without producing any output");转 failover 而不是
130
- // 把空轮递给客户端。chatShaped 是前置证据:只有看得出是 chat 轮才判空,
131
- // 非 chat SSE/无 choices JSON 透传是正式契约(chat-route 单测锁死),一律放行。
132
- // 工具轮豁免:tool_calls 无正文是合法 agent 形态;
133
- // finish=tool_calls/function_call 兜底豁免(防计数漏检误杀);下游已断开不重试(写给谁看)。
134
- // 对标 dsh-cline-pass 的 EMPTY_RESPONSE 语义;不记 auto 冷却(空转≠模型坏,重试多半能好)。
135
- const _d = out.detail || {};
136
- const _emptyTurn = out.status === 200 && !out.timedOut && !out.interrupted && !_d.downstreamClosed &&
137
- _d.chatShaped === true &&
138
- (Number(_d.chars) || 0) === 0 && (Number(_d.toolCalls) || 0) === 0 &&
139
- !["tool_calls", "function_call"].includes(_d.sawFinishReason);
140
- if (_emptyTurn) {
141
- const _why = _d.sawFinishReason ? ` (finish_reason=${_d.sawFinishReason})` : " (no content, no tool calls)";
142
- // 错误包络暂扣后下游不再直观看到上游原文:把摘要带进最终报错(截断 200 字),排障不断线。
143
- const _err = _d.upstreamErrorText ? ` upstream=${String(_d.upstreamErrorText).slice(0, 200)}` : "";
144
- try { _logError(actual, 502, `empty turn${_why}${_err}`); } catch {}
145
- _evt("upstream-error", { reqId, model: actual, status: 502, message: "empty turn", timing: null, ...upstreamEcho(upRes) });
146
- _evt("fallback", { reqId, from: actual, to: null, reason: "empty turn" });
147
- let _errMsg = `EMPTY_MODEL_RESPONSE: upstream returned 200 with no content${_why}${_err} — retry or rephrase`;
148
- if (_d.upstreamErrorText && _d.upstreamErrorText.includes("retryAfterSeconds")) {
149
- _errMsg = _d.upstreamErrorText;
150
- }
151
- return { handled: false, upRes: null, lastErr: { model: actual, upstream: null, status: 502, message: _errMsg } };
152
- }
153
-
154
- // 5a. 首块超时未写字节 → 回退(显式 timedOut 字段,status 只是 HTTP 语义展示)
155
- if (out.timedOut === true) {
156
- const why = out.detail?.upstreamError ? ` (upstream read error: ${out.detail.upstreamError})` : "";
157
- if (auto) try { await auto.recordError(actual, { status: 502, slow: true, note: `stream timeout ${streamTimeoutMs}ms` }); } catch {}
158
- try { _logError(actual, 502, `stream timeout ${streamTimeoutMs}ms${why}`); } catch {}
159
- _evt("upstream-error", { reqId, model: actual, status: 502, message: "stream timeout", timing: null, ...upstreamEcho(upRes) });
160
- _evt("fallback", { reqId, from: actual, to: null, reason: "stream timeout" });
161
- return { handled: false, upRes: null, lastErr: { model: actual, upstream: null, status: 502, message: `stream timed out after ${streamTimeoutMs}ms${why}` } };
162
- }
163
-
164
- // 5b. 中断(stall 超时 / max 流时长)
165
- if (out.interrupted) {
166
- if (auto) {
167
- try { await auto.recordError(actual, { status: 200, slow: true, note: `stall ${C.STALL_TIMEOUT_MS}ms` }); } catch {}
168
- try { await auto.recordLatency(actual, out.totalMs ?? (Date.now() - curStartedAt)); } catch {}
169
- }
170
- _evt("slow-model", { reqId, model: actual, elapsedMs: out.totalMs ?? (Date.now() - curStartedAt), threshold: C.STALL_TIMEOUT_MS, interrupted: true, detail: out.detail ?? null, ...upstreamEcho(upRes) });
171
- try { _logCall(actual, 200); } catch {}
172
- // interrupted 的 200 也是真实消耗(最贵的长生成)——照常落 usage 标 interrupted:1,口径与 5c 一致 — 见 .agents/notes/implemented/feature/2026-09-19-usage-report-jsonl.md
173
- if (out.status === 200) {
174
- try {
175
- const u = out.detail?.usage || null;
176
- const t1 = Number.isFinite(out.totalMs) && out.totalMs > 0 ? out.totalMs : (Date.now() - curStartedAt);
177
- const t0 = Number.isFinite(out.ttfMs) && out.ttfMs > 0 ? out.ttfMs : null;
178
- recordChatUsage({ model: normalizeFullId(actual), via, usage: u, interrupted: 1, ttfbMs: t0, totalMs: t1, tps: null }).catch(() => {});
179
- } catch {}
180
- }
181
- _evt("result", { reqId, model: actual, status: out.status, via, timing: upRes?._t ?? null, ttfMs: out.ttfMs, totalMs: out.totalMs, interrupted: true, ...upstreamEcho(upRes), detail: out.detail ?? null, fallback, requested, actual });
182
- _evt("client-response", { requested, actual, via, fallback, status: out.status, reqId, interrupted: true });
183
- if (plugins?.length) runHook(plugins, "request:completed", { reqId, requested, useAuto, hops, stream: Boolean(body?.stream), durationMs: Date.now() - curStartedAt, via, status: out.status, actual, interrupted: true, fallback }).catch(() => {});
184
- return { handled: true };
185
- }
186
-
187
- // 5c. 慢速计分 + ok
188
- const elapsed = Date.now() - curStartedAt;
189
- const latencyMs = out.totalMs ?? elapsed;
190
- let scoredSlow = false;
191
-
192
- if (C.SLOW_TOTAL_MS && auto && elapsed > C.SLOW_TOTAL_MS && out.status === 200) {
193
- try { await auto.recordError(actual, { status: 200, slow: true, note: `slow ${elapsed}ms` }); } catch {}
194
- try { await auto.recordLatency(actual, latencyMs); } catch {}
195
- _evt("slow-model", { reqId, model: actual, elapsedMs: elapsed, threshold: C.SLOW_TOTAL_MS, reason: "total", detail: out.detail ?? null, ...upstreamEcho(upRes) });
196
- scoredSlow = true;
197
- }
198
- if (out.detail?.stallHits > 0 && auto && out.status === 200) {
199
- try { await auto.recordError(actual, { status: 200, slow: true, note: `stall ${out.detail.stallHits}x gap>${C.SCORE_STALL_MS}ms maxGap ${out.detail.maxGapMs}ms` }); } catch {}
200
- try { await auto.recordLatency(actual, latencyMs); } catch {}
201
- _evt("slow-model", { reqId, model: actual, elapsedMs: elapsed, threshold: C.SCORE_STALL_MS, reason: "stall", stallHits: out.detail.stallHits, maxGapMs: out.detail.maxGapMs, detail: out.detail ?? null, ...upstreamEcho(upRes) });
202
- scoredSlow = true;
203
- }
204
- if (!scoredSlow && auto && out.status === 200) {
205
- try { await auto.recordOk(actual, { latencyMs }); } catch {}
206
- } else if (!scoredSlow && auto) {
207
- try { await auto.recordLatency(actual, latencyMs); } catch {}
208
- }
209
-
210
- // 每次 8989 正常返回都落体检:count/首字/总耗时/速度(供 -status TopN)
211
- if (out.status === 200) {
212
- try {
213
- const isStream = Boolean(body?.stream);
214
- let ttfb = isStream ? (out.ttfMs ?? upRes?._t?.ttfbMs ?? null) : null;
215
- // out.totalMs 为 0 时(非流式 <1ms 四舍五入)回退到 elapsed/durationMs
216
- const elapsedFallback = Date.now() - curStartedAt;
217
- let total = out.totalMs;
218
- if (!Number.isFinite(total) || total <= 0) total = elapsedFallback;
219
- if (!Number.isFinite(total) || total <= 0) total = latencyMs;
220
- if (!Number.isFinite(total) || total <= 0) total = Date.now() - curStartedAt;
221
- if (isStream && (!Number.isFinite(ttfb) || ttfb <= 0)) ttfb = null;
222
- const usage = out.detail?.usage || null;
223
- const chars = out.detail?.chars ?? null;
224
- const compTok = usage?.completion_tokens ?? null;
225
- const m = computeMetrics({ ttfbMs: ttfb, totalMs: total, completionTokens: compTok, chars });
226
- const tps = m.tps ?? m.charsPerSec ?? null;
227
- const fullId = normalizeFullId(actual);
228
- recordModelStats(fullId, { ttfbMs: ttfb, totalMs: total, tps, completionTokens: compTok });
229
- if (fullId !== actual) recordModelStats(actual, { ttfbMs: ttfb, totalMs: total, tps, completionTokens: compTok });
230
- // 窗口报表:逐请求落 usage(行形状由 usage/record.js 拥有,含 prompt/total ——
231
- // state 的 modelStats 只存 completion 的 EMA)。只按 canonical 名记一次,避免双计。
232
- recordChatUsage({ model: fullId, via, usage, ttfbMs: ttfb, totalMs: total, tps: m.tps }).catch(() => {});
233
- // cline 旁路记账:流式时 provider 已把账号哈希挂在 upRes 上,这里用消费完的 usage 记一笔。
234
- // 纯旁路:只调 recordOutput,不改转发/切号/重试;无账号或非 cline 直接跳过。
235
- if (upRes?.clineAccountId) {
236
- const clineModel = upRes.clineModel || String(actual).replace(/^cline\//, "");
237
- const orow = computeOutputRow({ model: clineModel, accountId: upRes.clineAccountId, usage, chars });
238
- if (orow) recordOutput(orow).catch(() => {});
239
- }
240
- } catch {}
241
- }
242
-
243
- _evt("result", { reqId, model: actual, status: out.status, via, timing: upRes?._t ?? null, ttfMs: out.ttfMs, totalMs: out.totalMs, ...upstreamEcho(upRes), detail: out.detail ?? null, fallback, requested, actual });
244
- _evt("client-response", { requested, actual, via, fallback, status: out.status, reqId });
245
- if (plugins?.length) runHook(plugins, "request:completed", { reqId, requested, useAuto, hops, stream: Boolean(body?.stream), durationMs: Date.now() - curStartedAt, via, status: out.status, actual, fallback }).catch(() => {});
246
- return { handled: true };
247
- }
248
-
249
- return { execute };
250
- }
1
+ import { runHook } from "../../plugins.js";
2
+ import { recordModelStats } from "../../state.js";
3
+ import { normalizeFullId } from "../../providers/model-id.js";
4
+ import { computeMetrics } from "../../metrics.js";
5
+ import { recordChatUsage } from "../../usage/record.js"; // 窗口报表唯一写入点(canonical 名单记防双计)— 见 .agents/notes/implemented/feature/2026-09-19-usage-report-jsonl.md
6
+ import { recordOutput, computeOutputRow } from "../../providers/cline/usage.js"; // cline 专属旁路统计(账号×模型,流式也覆盖)
7
+ import { upstreamEcho } from "../../model-trace.js"; // 回显头→日志字段(谁上的/哪个号/为什么/是否冷却)单一来源
8
+
9
+ // 唯一/最后候选没有 failover 去向:首块闸门退化为纯"防连接泄漏",放宽避免误杀慢模型
10
+ // (参考 opencode:zen 通道不设超时;openai responses 硬编码 300s headerTimeout)
11
+ export const LAST_CANDIDATE_TIMEOUT_MS = (() => {
12
+ const n = Number(process.env.MSLXDFF_LAST_CANDIDATE_TIMEOUT_MS);
13
+ return Number.isInteger(n) && n >= 0 ? n : 120_000;
14
+ })();
15
+
16
+ /** 空转 200 判定(serial-trial 同模型重试用):与 execute 内 _emptyTurn 产生的 lastErr 同源 */
17
+ export function isEmptyTurnError(err) {
18
+ return Number(err?.status) === 502 && String(err?.message || "").startsWith("EMPTY_MODEL_RESPONSE");
19
+ }
20
+
21
+ /**
22
+ * RelayPipeline 深模块
23
+ * 把 5 个 handler 各自的 fallback→relay→scoring→事件 6段流水收敛为单一真相。
24
+ * 对外 1 接口:createRelayPipeline(deps) => { execute(ctx) }
25
+ * 设计要点:全部外部可注入,便于在 pipeline seam 上做行为测试;constants 注入便于单测加速。
26
+ */
27
+ export function createRelayPipeline({
28
+ relay,
29
+ buildFallbackInfo,
30
+ auto,
31
+ plugins,
32
+ evt,
33
+ mark,
34
+ logCall,
35
+ logError,
36
+ constants,
37
+ startedAt: defaultStartedAt,
38
+ stages: defaultStages,
39
+ perfNow,
40
+ } = {}) {
41
+ const C = {
42
+ STREAM_TIMEOUT_MS: 25_000,
43
+ SLOW_TOTAL_MS: 20_000,
44
+ STALL_TIMEOUT_MS: 0,
45
+ SCORE_STALL_MS: 15_000,
46
+ ...(constants || {}),
47
+ };
48
+ const _relay = relay;
49
+ const _build = buildFallbackInfo;
50
+ const _evt = evt || (() => {});
51
+ const _mark = mark || (() => {});
52
+ const _logCall = logCall || (() => {});
53
+ const _logError = logError || (() => {});
54
+ const _perfNow = perfNow || (() => Date.now());
55
+
56
+ async function execute({
57
+ res,
58
+ upRes,
59
+ body,
60
+ requested,
61
+ actual,
62
+ lastErr,
63
+ via,
64
+ lockModel,
65
+ useAuto,
66
+ handlerCtx,
67
+ mark: m2,
68
+ perf0,
69
+ stages: s2,
70
+ startedAt: sa2,
71
+ streamTimeoutMs: ctxStreamTimeoutMs,
72
+ } = {}) {
73
+ const markFn = m2 || _mark;
74
+ const curStartedAt = sa2 ?? defaultStartedAt ?? Date.now();
75
+ const curStages = s2 ?? defaultStages ?? [];
76
+ const reqId = handlerCtx?.reqId;
77
+ const hops = handlerCtx?.hops;
78
+
79
+ // 1. logCall(pre) — 保持原 handler 的 logCall→fallback→relay-start 时序
80
+ try { _logCall(actual, upRes?.status); } catch {}
81
+ // 2. fallback + relay-start
82
+ let fallback = null;
83
+ try {
84
+ if (_build) fallback = _build({ requested, actual, lastErr, via, useAuto, lockModel });
85
+ } catch {}
86
+ if (fallback?.fallback) {
87
+ _evt("fallback-notice", { reqId, requested, actual, reason: fallback.reason, notice: fallback.notice, via, fallback: true });
88
+ }
89
+ _evt("relay-start", { reqId, model: actual, via, isStream: Boolean(body?.stream), fallback });
90
+
91
+ // 3. relay(唯一/最后候选:无 failover 去向 → 闸门放宽到防泄漏级别;显式 streamTimeoutMs 优先)
92
+ const orderLen = handlerCtx?.orderLen;
93
+ const curIdx = handlerCtx?.idx;
94
+ const isLastCandidate =
95
+ Number.isInteger(orderLen) && orderLen > 0 &&
96
+ (orderLen === 1 || (Number.isInteger(curIdx) && curIdx >= orderLen - 1));
97
+ const streamTimeoutMs = Number.isInteger(ctxStreamTimeoutMs) && ctxStreamTimeoutMs >= 0
98
+ ? ctxStreamTimeoutMs
99
+ : (isLastCandidate ? LAST_CANDIDATE_TIMEOUT_MS : C.STREAM_TIMEOUT_MS);
100
+ const out = await _relay(res, upRes, body, {
101
+ fallback,
102
+ streamTimeoutMs,
103
+ onFirstChunk: (delta) => {
104
+ try { markFn(`ttf-${actual}`); } catch {}
105
+ _evt("relay-first-chunk", { reqId, model: actual, ttfMs: delta, via });
106
+ if (plugins?.length) runHook(plugins, "relay:first-chunk", { reqId, requested, model: actual, via, ttfMs: delta }).catch(() => {});
107
+ },
108
+ onDownstreamAbort: () => {
109
+ _evt("client-abort", { reqId, model: actual, totalMs: Math.round(_perfNow() - (perf0 ?? 0)), stages: [...curStages] });
110
+ },
111
+ });
112
+
113
+ // 4. relay-done
114
+ _evt("relay-done", {
115
+ reqId,
116
+ model: actual,
117
+ via,
118
+ status: out.status,
119
+ ttfMs: out.ttfMs,
120
+ totalMs: out.totalMs,
121
+ aborted: out.aborted,
122
+ interrupted: out.interrupted ?? false,
123
+ timedOut: out.timedOut ?? false,
124
+ ...upstreamEcho(upRes),
125
+ detail: out.detail ?? null,
126
+ });
127
+
128
+ // 5a0. 空转 200:流正常结束但零正文零工具调用 → 客户端会报 EMPTY_MODEL_RESPONSE
129
+ // ("The model ended its turn without producing any output");转 failover 而不是
130
+ // 把空轮递给客户端。chatShaped 是前置证据:只有看得出是 chat 轮才判空,
131
+ // 非 chat SSE/无 choices JSON 透传是正式契约(chat-route 单测锁死),一律放行。
132
+ // 工具轮豁免:tool_calls 无正文是合法 agent 形态;
133
+ // finish=tool_calls/function_call 兜底豁免(防计数漏检误杀);下游已断开不重试(写给谁看)。
134
+ // 对标 dsh-cline-pass 的 EMPTY_RESPONSE 语义;不记 auto 冷却(空转≠模型坏,重试多半能好)。
135
+ const _d = out.detail || {};
136
+ const _emptyTurn = out.status === 200 && !out.timedOut && !out.interrupted && !_d.downstreamClosed &&
137
+ _d.chatShaped === true &&
138
+ (Number(_d.chars) || 0) === 0 && (Number(_d.toolCalls) || 0) === 0 &&
139
+ !["tool_calls", "function_call"].includes(_d.sawFinishReason);
140
+ if (_emptyTurn) {
141
+ const _why = _d.sawFinishReason ? ` (finish_reason=${_d.sawFinishReason})` : " (no content, no tool calls)";
142
+ // 错误包络暂扣后下游不再直观看到上游原文:把摘要带进最终报错(截断 200 字),排障不断线。
143
+ const _err = _d.upstreamErrorText ? ` upstream=${String(_d.upstreamErrorText).slice(0, 200)}` : "";
144
+ try { _logError(actual, 502, `empty turn${_why}${_err}`); } catch {}
145
+ _evt("upstream-error", { reqId, model: actual, status: 502, message: "empty turn", timing: null, ...upstreamEcho(upRes) });
146
+ _evt("fallback", { reqId, from: actual, to: null, reason: "empty turn" });
147
+ let _errMsg = `EMPTY_MODEL_RESPONSE: upstream returned 200 with no content${_why}${_err} — retry or rephrase`;
148
+ if (_d.upstreamErrorText && _d.upstreamErrorText.includes("retryAfterSeconds")) {
149
+ _errMsg = _d.upstreamErrorText;
150
+ }
151
+ return { handled: false, upRes: null, lastErr: { model: actual, upstream: null, status: 502, message: _errMsg } };
152
+ }
153
+
154
+ // 5a. 首块超时未写字节 → 回退(显式 timedOut 字段,status 只是 HTTP 语义展示)
155
+ if (out.timedOut === true) {
156
+ const why = out.detail?.upstreamError ? ` (upstream read error: ${out.detail.upstreamError})` : "";
157
+ if (auto) try { await auto.recordError(actual, { status: 502, slow: true, note: `stream timeout ${streamTimeoutMs}ms` }); } catch {}
158
+ try { _logError(actual, 502, `stream timeout ${streamTimeoutMs}ms${why}`); } catch {}
159
+ _evt("upstream-error", { reqId, model: actual, status: 502, message: "stream timeout", timing: null, ...upstreamEcho(upRes) });
160
+ _evt("fallback", { reqId, from: actual, to: null, reason: "stream timeout" });
161
+ return { handled: false, upRes: null, lastErr: { model: actual, upstream: null, status: 502, message: `stream timed out after ${streamTimeoutMs}ms${why}` } };
162
+ }
163
+
164
+ // 5b. 中断(stall 超时 / max 流时长)
165
+ if (out.interrupted) {
166
+ if (auto) {
167
+ try { await auto.recordError(actual, { status: 200, slow: true, note: `stall ${C.STALL_TIMEOUT_MS}ms` }); } catch {}
168
+ try { await auto.recordLatency(actual, out.totalMs ?? (Date.now() - curStartedAt)); } catch {}
169
+ }
170
+ _evt("slow-model", { reqId, model: actual, elapsedMs: out.totalMs ?? (Date.now() - curStartedAt), threshold: C.STALL_TIMEOUT_MS, interrupted: true, detail: out.detail ?? null, ...upstreamEcho(upRes) });
171
+ try { _logCall(actual, 200); } catch {}
172
+ // interrupted 的 200 也是真实消耗(最贵的长生成)——照常落 usage 标 interrupted:1,口径与 5c 一致 — 见 .agents/notes/implemented/feature/2026-09-19-usage-report-jsonl.md
173
+ if (out.status === 200) {
174
+ try {
175
+ const u = out.detail?.usage || null;
176
+ const t1 = Number.isFinite(out.totalMs) && out.totalMs > 0 ? out.totalMs : (Date.now() - curStartedAt);
177
+ const t0 = Number.isFinite(out.ttfMs) && out.ttfMs > 0 ? out.ttfMs : null;
178
+ recordChatUsage({ model: normalizeFullId(actual), via, usage: u, interrupted: 1, ttfbMs: t0, totalMs: t1, tps: null }).catch(() => {});
179
+ } catch {}
180
+ }
181
+ _evt("result", { reqId, model: actual, status: out.status, via, timing: upRes?._t ?? null, ttfMs: out.ttfMs, totalMs: out.totalMs, interrupted: true, ...upstreamEcho(upRes), detail: out.detail ?? null, fallback, requested, actual });
182
+ _evt("client-response", { requested, actual, via, fallback, status: out.status, reqId, interrupted: true });
183
+ if (plugins?.length) runHook(plugins, "request:completed", { reqId, requested, useAuto, hops, stream: Boolean(body?.stream), durationMs: Date.now() - curStartedAt, via, status: out.status, actual, interrupted: true, fallback }).catch(() => {});
184
+ return { handled: true };
185
+ }
186
+
187
+ // 5c. 慢速计分 + ok
188
+ const elapsed = Date.now() - curStartedAt;
189
+ const latencyMs = out.totalMs ?? elapsed;
190
+ let scoredSlow = false;
191
+
192
+ if (C.SLOW_TOTAL_MS && auto && elapsed > C.SLOW_TOTAL_MS && out.status === 200) {
193
+ try { await auto.recordError(actual, { status: 200, slow: true, note: `slow ${elapsed}ms` }); } catch {}
194
+ try { await auto.recordLatency(actual, latencyMs); } catch {}
195
+ _evt("slow-model", { reqId, model: actual, elapsedMs: elapsed, threshold: C.SLOW_TOTAL_MS, reason: "total", detail: out.detail ?? null, ...upstreamEcho(upRes) });
196
+ scoredSlow = true;
197
+ }
198
+ if (out.detail?.stallHits > 0 && auto && out.status === 200) {
199
+ try { await auto.recordError(actual, { status: 200, slow: true, note: `stall ${out.detail.stallHits}x gap>${C.SCORE_STALL_MS}ms maxGap ${out.detail.maxGapMs}ms` }); } catch {}
200
+ try { await auto.recordLatency(actual, latencyMs); } catch {}
201
+ _evt("slow-model", { reqId, model: actual, elapsedMs: elapsed, threshold: C.SCORE_STALL_MS, reason: "stall", stallHits: out.detail.stallHits, maxGapMs: out.detail.maxGapMs, detail: out.detail ?? null, ...upstreamEcho(upRes) });
202
+ scoredSlow = true;
203
+ }
204
+ if (!scoredSlow && auto && out.status === 200) {
205
+ try { await auto.recordOk(actual, { latencyMs }); } catch {}
206
+ } else if (!scoredSlow && auto) {
207
+ try { await auto.recordLatency(actual, latencyMs); } catch {}
208
+ }
209
+
210
+ // 每次 8989 正常返回都落体检:count/首字/总耗时/速度(供 -status TopN)
211
+ if (out.status === 200) {
212
+ try {
213
+ const isStream = Boolean(body?.stream);
214
+ let ttfb = isStream ? (out.ttfMs ?? upRes?._t?.ttfbMs ?? null) : null;
215
+ // out.totalMs 为 0 时(非流式 <1ms 四舍五入)回退到 elapsed/durationMs
216
+ const elapsedFallback = Date.now() - curStartedAt;
217
+ let total = out.totalMs;
218
+ if (!Number.isFinite(total) || total <= 0) total = elapsedFallback;
219
+ if (!Number.isFinite(total) || total <= 0) total = latencyMs;
220
+ if (!Number.isFinite(total) || total <= 0) total = Date.now() - curStartedAt;
221
+ if (isStream && (!Number.isFinite(ttfb) || ttfb <= 0)) ttfb = null;
222
+ const usage = out.detail?.usage || null;
223
+ const chars = out.detail?.chars ?? null;
224
+ const compTok = usage?.completion_tokens ?? null;
225
+ const m = computeMetrics({ ttfbMs: ttfb, totalMs: total, completionTokens: compTok, chars });
226
+ const tps = m.tps ?? m.charsPerSec ?? null;
227
+ const fullId = normalizeFullId(actual);
228
+ recordModelStats(fullId, { ttfbMs: ttfb, totalMs: total, tps, completionTokens: compTok });
229
+ if (fullId !== actual) recordModelStats(actual, { ttfbMs: ttfb, totalMs: total, tps, completionTokens: compTok });
230
+ // 窗口报表:逐请求落 usage(行形状由 usage/record.js 拥有,含 prompt/total ——
231
+ // state 的 modelStats 只存 completion 的 EMA)。只按 canonical 名记一次,避免双计。
232
+ recordChatUsage({ model: fullId, via, usage, ttfbMs: ttfb, totalMs: total, tps: m.tps }).catch(() => {});
233
+ // cline 旁路记账:流式时 provider 已把账号哈希挂在 upRes 上,这里用消费完的 usage 记一笔。
234
+ // 纯旁路:只调 recordOutput,不改转发/切号/重试;无账号或非 cline 直接跳过。
235
+ if (upRes?.clineAccountId) {
236
+ const clineModel = upRes.clineModel || String(actual).replace(/^cline\//, "");
237
+ const orow = computeOutputRow({ model: clineModel, accountId: upRes.clineAccountId, usage, chars });
238
+ if (orow) recordOutput(orow).catch(() => {});
239
+ }
240
+ } catch {}
241
+ }
242
+
243
+ _evt("result", { reqId, model: actual, status: out.status, via, timing: upRes?._t ?? null, ttfMs: out.ttfMs, totalMs: out.totalMs, ...upstreamEcho(upRes), detail: out.detail ?? null, fallback, requested, actual });
244
+ _evt("client-response", { requested, actual, via, fallback, status: out.status, reqId });
245
+ if (plugins?.length) runHook(plugins, "request:completed", { reqId, requested, useAuto, hops, stream: Boolean(body?.stream), durationMs: Date.now() - curStartedAt, via, status: out.status, actual, fallback }).catch(() => {});
246
+ return { handled: true };
247
+ }
248
+
249
+ return { execute };
250
+ }