mslxdff 0.1.151 → 0.1.153

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "mslxdff",
3
- "version": "0.1.151",
3
+ "version": "0.1.153",
4
4
  "description": "测试项目,请勿使用。",
5
5
  "type": "module",
6
6
  "bin": {
@@ -10,6 +10,20 @@ import { handleBroadbandRelay } from "../routes/chat/broadband-handler.js";
10
10
  import { handleViaRoute } from "../routes/chat/via-route-handler.js";
11
11
  import { handleExhaustedLocal, handleExhaustedAll } from "../routes/chat/exhausted-handler.js";
12
12
  import { shouldUseGroupForModel, isHardLocalOnly, isKeyProviderDirectOnly } from "../state/schemas/use-group.js";
13
+ import { isEmptyTurnError } from "../routes/chat/relay-pipeline.js";
14
+
15
+ const sleep = (ms) => new Promise((r) => setTimeout(r, ms));
16
+
17
+ // 空转重试档位(每次请求读取 env,单测可覆盖):默认同模型最多重试 2 次、间隔 1s;
18
+ // MSLXDFF_EMPTY_TURN_RETRIES=0 关闭(回旧行为:空转直接换候选/终结)。
19
+ function emptyRetryCfg() {
20
+ const r = Number(process.env.MSLXDFF_EMPTY_TURN_RETRIES);
21
+ const d = Number(process.env.MSLXDFF_EMPTY_TURN_RETRY_DELAY_MS);
22
+ return {
23
+ max: Number.isInteger(r) && r >= 0 ? r : 2,
24
+ delayMs: Number.isFinite(d) && d >= 0 ? d : 1000,
25
+ };
26
+ }
13
27
 
14
28
  function groupSkipReason(model) {
15
29
  if (isHardLocalOnly(model)) return "provider local-only(禁组员,仅本机直连)";
@@ -54,7 +68,7 @@ export async function runSerialTrial(ctx, deps = {}) {
54
68
  }
55
69
 
56
70
  let lastErr = viaRouteLastErr;
57
- for (let idx = 0; idx < order.length; idx++) {
71
+ candidate: for (let idx = 0; idx < order.length; idx++) {
58
72
  const model = order[idx];
59
73
  handlerCtx.model = model;
60
74
  handlerCtx.orderLen = order.length;
@@ -72,82 +86,99 @@ export async function runSerialTrial(ctx, deps = {}) {
72
86
  for (const e of ur.errors) evt("plugin-hook-error", { reqId, hook: "upstream:request", plugin: e.plugin, error: e.error });
73
87
  if (ur.changed && ur.value?.payload && typeof ur.value.payload === "object") { forwarded = ur.value.payload; evt("plugin-hook", { reqId, hook: "upstream:request", applied: true, model, rewrittenModel: forwarded.model ?? null }); }
74
88
  }
75
- const tUp = performance.now();
76
- evt("upstream-try", { reqId, model, attempt: idx + 1 });
77
- try {
78
- const chatOpts = {};
79
- if (Object.keys(shareKeys).length) chatOpts.shareKeys = shareKeys;
80
- if (workbuddyUid) chatOpts.workbuddyUid = workbuddyUid;
81
- if (handlerCtx?.sessionId) chatOpts.sessionId = handlerCtx.sessionId;
82
- upRes = await upstream.chat(forwarded, Object.keys(chatOpts).length ? chatOpts : undefined);
83
- evt("upstream-done", { reqId, model, ok: !(upRes instanceof Error) && upRes.status < 400, status: upRes instanceof Error ? null : upRes.status, timing: upRes._t ?? null, error: null });
84
- } catch (err) {
85
- if (auto) await auto.recordError(model, { message: errMsg(err) });
86
- lastErr = { model, upstream: null, status: 502, message: errMsg(err) };
87
- logError(model, 502, errMsg(err));
88
- evt("upstream-error", { reqId, model, status: 502, message: errMsg(err), timing: err._t ?? { attempts: [], waitMs: 0, totalMs: Math.round(performance.now() - tUp) } });
89
- }
90
- if (plugins?.length) {
91
- runHook(plugins, "upstream:response", {
92
- reqId, requested, model,
93
- status: upRes instanceof Error ? null : upRes instanceof Object ? (upRes.status ?? null) : null,
94
- ok: !(upRes instanceof Error) && upRes ? upRes.status < 400 : false,
95
- error: upRes instanceof Error ? errMsg(upRes) : null,
96
- timing: upRes?._t ?? null,
97
- }).catch(() => {});
98
- }
99
- mark(`up-${model}`);
100
- if (upRes && upRes.status >= 400) {
101
- const isAllowlistBlock = upRes.status === 403 && (upRes.headers?.get?.("x-mslxdff-allowlist") === "1");
102
- if (isAllowlistBlock) {
103
- let bodyText = null; try { bodyText = await upRes.clone().text(); } catch {}
104
- let errBody = { error: `model not allowed for provider` };
105
- try { errBody = bodyText ? JSON.parse(bodyText) : errBody; } catch { errBody = { error: bodyText || "model not allowed" }; }
106
- if (useAuto) {
89
+ const chatOpts = {};
90
+ if (Object.keys(shareKeys).length) chatOpts.shareKeys = shareKeys;
91
+ if (workbuddyUid) chatOpts.workbuddyUid = workbuddyUid;
92
+ if (handlerCtx?.sessionId) chatOpts.sessionId = handlerCtx.sessionId;
93
+ const chatOptsArg = Object.keys(chatOpts).length ? chatOpts : undefined;
94
+ // 空转 200(模型无输出)同模型暂停重试:默认 2 次、间隔 1s;仅 EMPTY_MODEL_RESPONSE,
95
+ // 429/403/500 与 fetch 异常走原有切号/failover(防烧额度)。MSLXDFF_EMPTY_TURN_RETRIES=0 关闭。
96
+ const emptyCfg = emptyRetryCfg();
97
+ let emptyRetried = 0;
98
+ for (;;) {
99
+ const tUp = performance.now();
100
+ evt("upstream-try", { reqId, model, attempt: idx + 1, emptyRetry: emptyRetried });
101
+ try {
102
+ upRes = await upstream.chat(forwarded, chatOptsArg);
103
+ evt("upstream-done", { reqId, model, ok: !(upRes instanceof Error) && upRes.status < 400, status: upRes instanceof Error ? null : upRes.status, timing: upRes._t ?? null, error: null });
104
+ } catch (err) {
105
+ if (auto) await auto.recordError(model, { message: errMsg(err) });
106
+ lastErr = { model, upstream: null, status: 502, message: errMsg(err) };
107
+ logError(model, 502, errMsg(err));
108
+ evt("upstream-error", { reqId, model, status: 502, message: errMsg(err), timing: err._t ?? { attempts: [], waitMs: 0, totalMs: Math.round(performance.now() - tUp) } });
109
+ upRes = null;
110
+ }
111
+ if (plugins?.length) {
112
+ runHook(plugins, "upstream:response", {
113
+ reqId, requested, model,
114
+ status: upRes instanceof Error ? null : upRes instanceof Object ? (upRes.status ?? null) : null,
115
+ ok: !(upRes instanceof Error) && upRes ? upRes.status < 400 : false,
116
+ error: upRes instanceof Error ? errMsg(upRes) : null,
117
+ timing: upRes?._t ?? null,
118
+ }).catch(() => {});
119
+ }
120
+ mark(`up-${model}`);
121
+ if (upRes && upRes.status >= 400) {
122
+ const isAllowlistBlock = upRes.status === 403 && (upRes.headers?.get?.("x-mslxdff-allowlist") === "1");
123
+ if (isAllowlistBlock) {
124
+ let bodyText = null; try { bodyText = await upRes.clone().text(); } catch {}
125
+ let errBody = { error: `model not allowed for provider` };
126
+ try { errBody = bodyText ? JSON.parse(bodyText) : errBody; } catch { errBody = { error: bodyText || "model not allowed" }; }
127
+ if (useAuto) {
128
+ logError(model, 403, errBody.error || "model not allowed");
129
+ evt("upstream-error", { reqId, model, status: 403, message: errBody.error, timing: upRes._t ?? null, allowlist: true, skipped: true });
130
+ lastErr = { model, upstream: upRes, status: 403, message: errBody.error || "model not allowed" };
131
+ if (canFallback && idx < order.length - 1) { evt("fallback", { reqId, from: model, to: order[idx + 1] ?? null, reason: `allowlist skip ${errBody.error || "blocked"}` }); continue candidate; }
132
+ return json(res, 403, errBody);
133
+ }
107
134
  logError(model, 403, errBody.error || "model not allowed");
108
- evt("upstream-error", { reqId, model, status: 403, message: errBody.error, timing: upRes._t ?? null, allowlist: true, skipped: true });
109
- lastErr = { model, upstream: upRes, status: 403, message: errBody.error || "model not allowed" };
110
- if (canFallback && idx < order.length - 1) { evt("fallback", { reqId, from: model, to: order[idx + 1] ?? null, reason: `allowlist skip ${errBody.error || "blocked"}` }); continue; }
135
+ evt("upstream-error", { reqId, model, status: 403, message: errBody.error, timing: upRes._t ?? null, allowlist: true });
111
136
  return json(res, 403, errBody);
112
137
  }
113
- logError(model, 403, errBody.error || "model not allowed");
114
- evt("upstream-error", { reqId, model, status: 403, message: errBody.error, timing: upRes._t ?? null, allowlist: true });
115
- return json(res, 403, errBody);
116
- }
117
- if (auto) await auto.recordError(model, { status: upRes.status });
118
- // 读失败响应体(clone 不影响后续 relay 转发原响应;1s 上限防流式错误体拖慢)
119
- let upBody = "";
120
- try {
121
- upBody = String(await Promise.race([
122
- upRes.clone().text(),
123
- new Promise((r) => { const t = setTimeout(() => r(""), 1000); t.unref?.(); }),
124
- ])).replace(/\s+/g, " ").slice(0, 400);
125
- } catch {}
126
- const upMsg = upBody || `upstream ${upRes.status}`;
127
- lastErr = { model, upstream: upRes, status: upRes.status, message: upMsg };
128
- logError(model, upRes.status, `upstream ${upRes.status}${upBody ? ` body=${upBody.slice(0, 300)}` : ""}`);
129
- evt("upstream-error", { reqId, model, status: upRes.status, message: upMsg.slice(0, 300), timing: upRes._t ?? null });
130
- upRes = null;
131
- }
132
- if (upRes) {
133
- const isStream = Boolean(body.stream);
134
- const d = hedgeDelayMs();
135
- const hasPeers = Boolean(peers) && peers.ordered().length > 0;
136
- const canUseGroup = shouldUseGroupForModel(model);
137
- const doHedge = canUseGroup && shouldHedge({ isStream, canForwardPeers, hedgeDelayMs: d, hasPeers, model }) && upRes.status === 200 && upRes.body;
138
- if (doHedge) {
139
- const hr = await hedge({ upRes, model, body, order, idx, lastErr, requested, useAuto, lockModel, auto, peers, handlerCtx, evt, logCall, logError, mark, perf0, stages, startedAt, plugins, res, hedgeDelayMs: d });
140
- if (hr.handled) return { done: true };
141
- if (hr.lastErr) lastErr = hr.lastErr;
142
- if (hr.upRes === null) upRes = null;
143
- else if (hr.upRes) upRes = hr.upRes;
138
+ if (auto) await auto.recordError(model, { status: upRes.status });
139
+ // 读失败响应体(clone 不影响后续 relay 转发原响应;1s 上限防流式错误体拖慢)
140
+ let upBody = "";
141
+ try {
142
+ upBody = String(await Promise.race([
143
+ upRes.clone().text(),
144
+ new Promise((r) => { const t = setTimeout(() => r(""), 1000); t.unref?.(); }),
145
+ ])).replace(/\s+/g, " ").slice(0, 400);
146
+ } catch {}
147
+ const upMsg = upBody || `upstream ${upRes.status}`;
148
+ lastErr = { model, upstream: upRes, status: upRes.status, message: upMsg };
149
+ logError(model, upRes.status, `upstream ${upRes.status}${upBody ? ` body=${upBody.slice(0, 300)}` : ""}`);
150
+ evt("upstream-error", { reqId, model, status: upRes.status, message: upMsg.slice(0, 300), timing: upRes._t ?? null });
151
+ upRes = null;
152
+ break;
144
153
  }
145
154
  if (upRes) {
146
- const lr = await localRelay({ upRes, model, body, order, idx, lastErr, requested, useAuto, lockModel, auto, handlerCtx, evt, logCall, logError, mark, perf0, stages, startedAt, plugins, res });
147
- if (lr.handled) return { done: true };
148
- if (lr.lastErr) { lastErr = lr.lastErr; continue; }
149
- return { done: true };
155
+ const isStream = Boolean(body.stream);
156
+ const d = hedgeDelayMs();
157
+ const hasPeers = Boolean(peers) && peers.ordered().length > 0;
158
+ const canUseGroup = shouldUseGroupForModel(model);
159
+ const doHedge = canUseGroup && shouldHedge({ isStream, canForwardPeers, hedgeDelayMs: d, hasPeers, model }) && upRes.status === 200 && upRes.body;
160
+ if (doHedge) {
161
+ const hr = await hedge({ upRes, model, body, order, idx, lastErr, requested, useAuto, lockModel, auto, peers, handlerCtx, evt, logCall, logError, mark, perf0, stages, startedAt, plugins, res, hedgeDelayMs: d });
162
+ if (hr.handled) return { done: true };
163
+ if (hr.lastErr) lastErr = hr.lastErr;
164
+ if (hr.upRes === null) upRes = null;
165
+ else if (hr.upRes) upRes = hr.upRes;
166
+ }
167
+ if (upRes) {
168
+ const lr = await localRelay({ upRes, model, body, order, idx, lastErr, requested, useAuto, lockModel, auto, handlerCtx, evt, logCall, logError, mark, perf0, stages, startedAt, plugins, res });
169
+ if (lr.handled) return { done: true };
170
+ if (lr.lastErr && isEmptyTurnError(lr.lastErr) && emptyRetried < emptyCfg.max) {
171
+ emptyRetried++;
172
+ evt("empty-turn-retry", { reqId, model, retry: emptyRetried, max: emptyCfg.max, delayMs: emptyCfg.delayMs });
173
+ await sleep(emptyCfg.delayMs);
174
+ continue;
175
+ }
176
+ if (lr.lastErr) { lastErr = lr.lastErr; continue candidate; }
177
+ return { done: true };
178
+ }
179
+ break;
150
180
  }
181
+ break;
151
182
  }
152
183
  if (canForwardPeers) {
153
184
  if (!shouldUseGroupForModel(model)) {
@@ -12,6 +12,11 @@ export const LAST_CANDIDATE_TIMEOUT_MS = (() => {
12
12
  return Number.isInteger(n) && n >= 0 ? n : 120_000;
13
13
  })();
14
14
 
15
+ /** 空转 200 判定(serial-trial 同模型重试用):与 execute 内 _emptyTurn 产生的 lastErr 同源 */
16
+ export function isEmptyTurnError(err) {
17
+ return Number(err?.status) === 502 && String(err?.message || "").startsWith("EMPTY_MODEL_RESPONSE");
18
+ }
19
+
15
20
  /**
16
21
  * RelayPipeline 深模块
17
22
  * 把 5 个 handler 各自的 fallback→relay→scoring→事件 6段流水收敛为单一真相。
@@ -132,10 +137,12 @@ export function createRelayPipeline({
132
137
  !["tool_calls", "function_call"].includes(_d.sawFinishReason);
133
138
  if (_emptyTurn) {
134
139
  const _why = _d.sawFinishReason ? ` (finish_reason=${_d.sawFinishReason})` : " (no content, no tool calls)";
135
- try { _logError(actual, 502, `empty turn${_why}`); } catch {}
140
+ // 错误包络暂扣后下游不再直观看到上游原文:把摘要带进最终报错(截断 200 字),排障不断线。
141
+ const _err = _d.upstreamErrorText ? ` upstream=${String(_d.upstreamErrorText).slice(0, 200)}` : "";
142
+ try { _logError(actual, 502, `empty turn${_why}${_err}`); } catch {}
136
143
  _evt("upstream-error", { reqId, model: actual, status: 502, message: "empty turn", timing: null });
137
144
  _evt("fallback", { reqId, from: actual, to: null, reason: "empty turn" });
138
- return { handled: false, upRes: null, lastErr: { model: actual, upstream: null, status: 502, message: `EMPTY_MODEL_RESPONSE: upstream returned 200 with no content${_why} — retry or rephrase` } };
145
+ return { handled: false, upRes: null, lastErr: { model: actual, upstream: null, status: 502, message: `EMPTY_MODEL_RESPONSE: upstream returned 200 with no content${_why}${_err} — retry or rephrase` } };
139
146
  }
140
147
 
141
148
  // 5a. 首块超时未写字节 → 回退(显式 timedOut 字段,status 只是 HTTP 语义展示)
@@ -30,6 +30,40 @@ function hasPayload(text) {
30
30
  return false;
31
31
  }
32
32
 
33
+ // 错误包络 chunk 判定(纯函数):data 行 JSON 含顶层 .error 对象、或只有 finish_reason=error
34
+ // 的空帧,且整 chunk 无任何 content/tool_calls/reasoning 输出 → 返回错误摘要(建议暂扣),
35
+ // 否则返回 null(透传)。解析失败一律透传(默认保安全)。
36
+ // 背景:网关把上游失败包成 HTTP 200 SSE;暂扣后下游流式 UI 不再展示瞬时错误
37
+ //(如 Vertex 503 + google fallback 400),本轮走空转重试,客户端只看到最终结果。
38
+ function holdableChunk(txt) {
39
+ if (typeof txt !== "string" || !txt.includes("data:")) return null;
40
+ let sawData = false;
41
+ const errs = [];
42
+ for (const line of txt.split("\n")) {
43
+ const t = line.trim();
44
+ if (!t.startsWith("data:")) continue;
45
+ const d = t.slice(5).trim();
46
+ if (!d || d === "[DONE]") continue;
47
+ let j = null;
48
+ try { j = JSON.parse(d); } catch { return null; }
49
+ if (!j || typeof j !== "object") return null;
50
+ sawData = true;
51
+ const c0 = (Array.isArray(j.choices) && j.choices[0]) || {};
52
+ const delta = c0.delta || {};
53
+ const msg = c0.message || {};
54
+ const out = delta.content || msg.content || delta.tool_calls || msg.tool_calls || delta.reasoning || msg.reasoning;
55
+ if (out && !(Array.isArray(out) && out.length === 0)) return null;
56
+ if (j.error && typeof j.error === "object") {
57
+ const m = j.error.message || j.error.code || "";
58
+ if (m) errs.push(String(m).slice(0, 300));
59
+ } else if (c0.finish_reason !== "error") {
60
+ return null;
61
+ }
62
+ }
63
+ if (!sawData) return null;
64
+ return errs.length ? errs.join(" | ") : "finish_reason=error 空帧";
65
+ }
66
+
33
67
  // 自持 reader 优先:for-await 会锁定 ReadableStream,使 body.cancel() 必 reject(真流上等于空操作),
34
68
  // 超时/断下游要真能掐上游必须走 reader.cancel()
35
69
  async function* bodyChunks(body, reader) {
@@ -116,6 +150,8 @@ export async function relay(res, upRes, body, { onFirstChunk, onDownstreamAbort,
116
150
  chars: 0,
117
151
  toolCalls: 0,
118
152
  chatShaped: false,
153
+ heldErrorChunks: 0,
154
+ upstreamErrorText: null,
119
155
  recoveries: 0,
120
156
  };
121
157
  let prevChunkAt = t0;
@@ -206,6 +242,13 @@ export async function relay(res, upRes, body, { onFirstChunk, onDownstreamAbort,
206
242
  if (gap > SCORE_STALL_MS) detail.stallHits += 1;
207
243
  prevChunkAt = now;
208
244
  const txt = chunkText(chunk);
245
+ // 错误包络暂扣:本轮尚未写出真实输出时错误帧不写下游(下游 UI 不再展示瞬时错误),
246
+ // 摘要记 detail.upstreamErrorText 供空转判定与最终报错;已写出真实内容后的错误帧照常透传。
247
+ const holdErr = !wrotePayload ? holdableChunk(txt) : null;
248
+ if (holdErr) {
249
+ detail.heldErrorChunks += 1;
250
+ if (!detail.upstreamErrorText) detail.upstreamErrorText = String(holdErr).slice(0, 500);
251
+ }
209
252
  const isPayload = hasPayload(txt);
210
253
  try {
211
254
  if (txt.includes("[DONE]")) detail.sawDone = true;
@@ -252,7 +295,7 @@ export async function relay(res, upRes, body, { onFirstChunk, onDownstreamAbort,
252
295
  // 首块/空闲超时后上游仍吐出了真实数据 → 只是慢,不是死:撤销超时判定,照常转发
253
296
  //(cancel 是异步的,竞态窗口内已到达的数据是纯收益;丢掉是纯损失)
254
297
  // 注释帧(keepalive)不算:否则对端只要在发心跳,闸门就永远解除
255
- if (isPayload && (timedOut || stalled)) {
298
+ if (isPayload && !holdErr && (timedOut || stalled)) {
256
299
  timedOut = false;
257
300
  stalled = false;
258
301
  detail.recoveries = (detail.recoveries || 0) + 1;
@@ -260,7 +303,7 @@ export async function relay(res, upRes, body, { onFirstChunk, onDownstreamAbort,
260
303
  if (firstTimer) { clearTimeout(firstTimer); firstTimer = null; }
261
304
  }
262
305
  if (timedOut || stalled || tooLong) break;
263
- if (isPayload && first) {
306
+ if (isPayload && first && !holdErr) {
264
307
  first = false;
265
308
  ttf = Math.round(now - t0);
266
309
  onFirstChunk?.(ttf);
@@ -279,11 +322,13 @@ export async function relay(res, upRes, body, { onFirstChunk, onDownstreamAbort,
279
322
  }
280
323
  } catch {}
281
324
  }
282
- wroteAny = true;
283
- if (isPayload) wrotePayload = true;
284
- detail.wroteChunks += 1;
285
- detail.wroteBytes += Buffer.isBuffer(outChunk) ? outChunk.length : (outChunk?.length ?? len);
286
- try { res.write(outChunk); } catch { /* 下游已断开:onClose 已掐上游 */ }
325
+ if (!holdErr) {
326
+ wroteAny = true;
327
+ if (isPayload) wrotePayload = true;
328
+ detail.wroteChunks += 1;
329
+ detail.wroteBytes += Buffer.isBuffer(outChunk) ? outChunk.length : (outChunk?.length ?? len);
330
+ try { res.write(outChunk); } catch { /* 下游已断开:onClose 已掐上游 */ }
331
+ }
287
332
  armStall();
288
333
  }
289
334
  if (!detail.exitReason) detail.exitReason = detail.downstreamClosed ? "downstream-closed" : "normal";