mslxdff 0.1.151 → 0.1.152
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json
CHANGED
|
@@ -10,6 +10,20 @@ import { handleBroadbandRelay } from "../routes/chat/broadband-handler.js";
|
|
|
10
10
|
import { handleViaRoute } from "../routes/chat/via-route-handler.js";
|
|
11
11
|
import { handleExhaustedLocal, handleExhaustedAll } from "../routes/chat/exhausted-handler.js";
|
|
12
12
|
import { shouldUseGroupForModel, isHardLocalOnly, isKeyProviderDirectOnly } from "../state/schemas/use-group.js";
|
|
13
|
+
import { isEmptyTurnError } from "../routes/chat/relay-pipeline.js";
|
|
14
|
+
|
|
15
|
+
const sleep = (ms) => new Promise((r) => setTimeout(r, ms));
|
|
16
|
+
|
|
17
|
+
// 空转重试档位(每次请求读取 env,单测可覆盖):默认同模型最多重试 2 次、间隔 1s;
|
|
18
|
+
// MSLXDFF_EMPTY_TURN_RETRIES=0 关闭(回旧行为:空转直接换候选/终结)。
|
|
19
|
+
function emptyRetryCfg() {
|
|
20
|
+
const r = Number(process.env.MSLXDFF_EMPTY_TURN_RETRIES);
|
|
21
|
+
const d = Number(process.env.MSLXDFF_EMPTY_TURN_RETRY_DELAY_MS);
|
|
22
|
+
return {
|
|
23
|
+
max: Number.isInteger(r) && r >= 0 ? r : 2,
|
|
24
|
+
delayMs: Number.isFinite(d) && d >= 0 ? d : 1000,
|
|
25
|
+
};
|
|
26
|
+
}
|
|
13
27
|
|
|
14
28
|
function groupSkipReason(model) {
|
|
15
29
|
if (isHardLocalOnly(model)) return "provider local-only(禁组员,仅本机直连)";
|
|
@@ -54,7 +68,7 @@ export async function runSerialTrial(ctx, deps = {}) {
|
|
|
54
68
|
}
|
|
55
69
|
|
|
56
70
|
let lastErr = viaRouteLastErr;
|
|
57
|
-
for (let idx = 0; idx < order.length; idx++) {
|
|
71
|
+
candidate: for (let idx = 0; idx < order.length; idx++) {
|
|
58
72
|
const model = order[idx];
|
|
59
73
|
handlerCtx.model = model;
|
|
60
74
|
handlerCtx.orderLen = order.length;
|
|
@@ -72,82 +86,99 @@ export async function runSerialTrial(ctx, deps = {}) {
|
|
|
72
86
|
for (const e of ur.errors) evt("plugin-hook-error", { reqId, hook: "upstream:request", plugin: e.plugin, error: e.error });
|
|
73
87
|
if (ur.changed && ur.value?.payload && typeof ur.value.payload === "object") { forwarded = ur.value.payload; evt("plugin-hook", { reqId, hook: "upstream:request", applied: true, model, rewrittenModel: forwarded.model ?? null }); }
|
|
74
88
|
}
|
|
75
|
-
const
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
|
|
89
|
+
const chatOpts = {};
|
|
90
|
+
if (Object.keys(shareKeys).length) chatOpts.shareKeys = shareKeys;
|
|
91
|
+
if (workbuddyUid) chatOpts.workbuddyUid = workbuddyUid;
|
|
92
|
+
if (handlerCtx?.sessionId) chatOpts.sessionId = handlerCtx.sessionId;
|
|
93
|
+
const chatOptsArg = Object.keys(chatOpts).length ? chatOpts : undefined;
|
|
94
|
+
// 空转 200(模型无输出)同模型暂停重试:默认 2 次、间隔 1s;仅 EMPTY_MODEL_RESPONSE,
|
|
95
|
+
// 429/403/500 与 fetch 异常走原有切号/failover(防烧额度)。MSLXDFF_EMPTY_TURN_RETRIES=0 关闭。
|
|
96
|
+
const emptyCfg = emptyRetryCfg();
|
|
97
|
+
let emptyRetried = 0;
|
|
98
|
+
for (;;) {
|
|
99
|
+
const tUp = performance.now();
|
|
100
|
+
evt("upstream-try", { reqId, model, attempt: idx + 1, emptyRetry: emptyRetried });
|
|
101
|
+
try {
|
|
102
|
+
upRes = await upstream.chat(forwarded, chatOptsArg);
|
|
103
|
+
evt("upstream-done", { reqId, model, ok: !(upRes instanceof Error) && upRes.status < 400, status: upRes instanceof Error ? null : upRes.status, timing: upRes._t ?? null, error: null });
|
|
104
|
+
} catch (err) {
|
|
105
|
+
if (auto) await auto.recordError(model, { message: errMsg(err) });
|
|
106
|
+
lastErr = { model, upstream: null, status: 502, message: errMsg(err) };
|
|
107
|
+
logError(model, 502, errMsg(err));
|
|
108
|
+
evt("upstream-error", { reqId, model, status: 502, message: errMsg(err), timing: err._t ?? { attempts: [], waitMs: 0, totalMs: Math.round(performance.now() - tUp) } });
|
|
109
|
+
upRes = null;
|
|
110
|
+
}
|
|
111
|
+
if (plugins?.length) {
|
|
112
|
+
runHook(plugins, "upstream:response", {
|
|
113
|
+
reqId, requested, model,
|
|
114
|
+
status: upRes instanceof Error ? null : upRes instanceof Object ? (upRes.status ?? null) : null,
|
|
115
|
+
ok: !(upRes instanceof Error) && upRes ? upRes.status < 400 : false,
|
|
116
|
+
error: upRes instanceof Error ? errMsg(upRes) : null,
|
|
117
|
+
timing: upRes?._t ?? null,
|
|
118
|
+
}).catch(() => {});
|
|
119
|
+
}
|
|
120
|
+
mark(`up-${model}`);
|
|
121
|
+
if (upRes && upRes.status >= 400) {
|
|
122
|
+
const isAllowlistBlock = upRes.status === 403 && (upRes.headers?.get?.("x-mslxdff-allowlist") === "1");
|
|
123
|
+
if (isAllowlistBlock) {
|
|
124
|
+
let bodyText = null; try { bodyText = await upRes.clone().text(); } catch {}
|
|
125
|
+
let errBody = { error: `model not allowed for provider` };
|
|
126
|
+
try { errBody = bodyText ? JSON.parse(bodyText) : errBody; } catch { errBody = { error: bodyText || "model not allowed" }; }
|
|
127
|
+
if (useAuto) {
|
|
128
|
+
logError(model, 403, errBody.error || "model not allowed");
|
|
129
|
+
evt("upstream-error", { reqId, model, status: 403, message: errBody.error, timing: upRes._t ?? null, allowlist: true, skipped: true });
|
|
130
|
+
lastErr = { model, upstream: upRes, status: 403, message: errBody.error || "model not allowed" };
|
|
131
|
+
if (canFallback && idx < order.length - 1) { evt("fallback", { reqId, from: model, to: order[idx + 1] ?? null, reason: `allowlist skip ${errBody.error || "blocked"}` }); continue candidate; }
|
|
132
|
+
return json(res, 403, errBody);
|
|
133
|
+
}
|
|
107
134
|
logError(model, 403, errBody.error || "model not allowed");
|
|
108
|
-
evt("upstream-error", { reqId, model, status: 403, message: errBody.error, timing: upRes._t ?? null, allowlist: true
|
|
109
|
-
lastErr = { model, upstream: upRes, status: 403, message: errBody.error || "model not allowed" };
|
|
110
|
-
if (canFallback && idx < order.length - 1) { evt("fallback", { reqId, from: model, to: order[idx + 1] ?? null, reason: `allowlist skip ${errBody.error || "blocked"}` }); continue; }
|
|
135
|
+
evt("upstream-error", { reqId, model, status: 403, message: errBody.error, timing: upRes._t ?? null, allowlist: true });
|
|
111
136
|
return json(res, 403, errBody);
|
|
112
137
|
}
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
|
|
118
|
-
|
|
119
|
-
|
|
120
|
-
|
|
121
|
-
|
|
122
|
-
|
|
123
|
-
|
|
124
|
-
|
|
125
|
-
|
|
126
|
-
|
|
127
|
-
|
|
128
|
-
logError(model, upRes.status, `upstream ${upRes.status}${upBody ? ` body=${upBody.slice(0, 300)}` : ""}`);
|
|
129
|
-
evt("upstream-error", { reqId, model, status: upRes.status, message: upMsg.slice(0, 300), timing: upRes._t ?? null });
|
|
130
|
-
upRes = null;
|
|
131
|
-
}
|
|
132
|
-
if (upRes) {
|
|
133
|
-
const isStream = Boolean(body.stream);
|
|
134
|
-
const d = hedgeDelayMs();
|
|
135
|
-
const hasPeers = Boolean(peers) && peers.ordered().length > 0;
|
|
136
|
-
const canUseGroup = shouldUseGroupForModel(model);
|
|
137
|
-
const doHedge = canUseGroup && shouldHedge({ isStream, canForwardPeers, hedgeDelayMs: d, hasPeers, model }) && upRes.status === 200 && upRes.body;
|
|
138
|
-
if (doHedge) {
|
|
139
|
-
const hr = await hedge({ upRes, model, body, order, idx, lastErr, requested, useAuto, lockModel, auto, peers, handlerCtx, evt, logCall, logError, mark, perf0, stages, startedAt, plugins, res, hedgeDelayMs: d });
|
|
140
|
-
if (hr.handled) return { done: true };
|
|
141
|
-
if (hr.lastErr) lastErr = hr.lastErr;
|
|
142
|
-
if (hr.upRes === null) upRes = null;
|
|
143
|
-
else if (hr.upRes) upRes = hr.upRes;
|
|
138
|
+
if (auto) await auto.recordError(model, { status: upRes.status });
|
|
139
|
+
// 读失败响应体(clone 不影响后续 relay 转发原响应;1s 上限防流式错误体拖慢)
|
|
140
|
+
let upBody = "";
|
|
141
|
+
try {
|
|
142
|
+
upBody = String(await Promise.race([
|
|
143
|
+
upRes.clone().text(),
|
|
144
|
+
new Promise((r) => { const t = setTimeout(() => r(""), 1000); t.unref?.(); }),
|
|
145
|
+
])).replace(/\s+/g, " ").slice(0, 400);
|
|
146
|
+
} catch {}
|
|
147
|
+
const upMsg = upBody || `upstream ${upRes.status}`;
|
|
148
|
+
lastErr = { model, upstream: upRes, status: upRes.status, message: upMsg };
|
|
149
|
+
logError(model, upRes.status, `upstream ${upRes.status}${upBody ? ` body=${upBody.slice(0, 300)}` : ""}`);
|
|
150
|
+
evt("upstream-error", { reqId, model, status: upRes.status, message: upMsg.slice(0, 300), timing: upRes._t ?? null });
|
|
151
|
+
upRes = null;
|
|
152
|
+
break;
|
|
144
153
|
}
|
|
145
154
|
if (upRes) {
|
|
146
|
-
const
|
|
147
|
-
|
|
148
|
-
|
|
149
|
-
|
|
155
|
+
const isStream = Boolean(body.stream);
|
|
156
|
+
const d = hedgeDelayMs();
|
|
157
|
+
const hasPeers = Boolean(peers) && peers.ordered().length > 0;
|
|
158
|
+
const canUseGroup = shouldUseGroupForModel(model);
|
|
159
|
+
const doHedge = canUseGroup && shouldHedge({ isStream, canForwardPeers, hedgeDelayMs: d, hasPeers, model }) && upRes.status === 200 && upRes.body;
|
|
160
|
+
if (doHedge) {
|
|
161
|
+
const hr = await hedge({ upRes, model, body, order, idx, lastErr, requested, useAuto, lockModel, auto, peers, handlerCtx, evt, logCall, logError, mark, perf0, stages, startedAt, plugins, res, hedgeDelayMs: d });
|
|
162
|
+
if (hr.handled) return { done: true };
|
|
163
|
+
if (hr.lastErr) lastErr = hr.lastErr;
|
|
164
|
+
if (hr.upRes === null) upRes = null;
|
|
165
|
+
else if (hr.upRes) upRes = hr.upRes;
|
|
166
|
+
}
|
|
167
|
+
if (upRes) {
|
|
168
|
+
const lr = await localRelay({ upRes, model, body, order, idx, lastErr, requested, useAuto, lockModel, auto, handlerCtx, evt, logCall, logError, mark, perf0, stages, startedAt, plugins, res });
|
|
169
|
+
if (lr.handled) return { done: true };
|
|
170
|
+
if (lr.lastErr && isEmptyTurnError(lr.lastErr) && emptyRetried < emptyCfg.max) {
|
|
171
|
+
emptyRetried++;
|
|
172
|
+
evt("empty-turn-retry", { reqId, model, retry: emptyRetried, max: emptyCfg.max, delayMs: emptyCfg.delayMs });
|
|
173
|
+
await sleep(emptyCfg.delayMs);
|
|
174
|
+
continue;
|
|
175
|
+
}
|
|
176
|
+
if (lr.lastErr) { lastErr = lr.lastErr; continue candidate; }
|
|
177
|
+
return { done: true };
|
|
178
|
+
}
|
|
179
|
+
break;
|
|
150
180
|
}
|
|
181
|
+
break;
|
|
151
182
|
}
|
|
152
183
|
if (canForwardPeers) {
|
|
153
184
|
if (!shouldUseGroupForModel(model)) {
|
|
@@ -12,6 +12,11 @@ export const LAST_CANDIDATE_TIMEOUT_MS = (() => {
|
|
|
12
12
|
return Number.isInteger(n) && n >= 0 ? n : 120_000;
|
|
13
13
|
})();
|
|
14
14
|
|
|
15
|
+
/** 空转 200 判定(serial-trial 同模型重试用):与 execute 内 _emptyTurn 产生的 lastErr 同源 */
|
|
16
|
+
export function isEmptyTurnError(err) {
|
|
17
|
+
return Number(err?.status) === 502 && String(err?.message || "").startsWith("EMPTY_MODEL_RESPONSE");
|
|
18
|
+
}
|
|
19
|
+
|
|
15
20
|
/**
|
|
16
21
|
* RelayPipeline 深模块
|
|
17
22
|
* 把 5 个 handler 各自的 fallback→relay→scoring→事件 6段流水收敛为单一真相。
|