mslxdff 0.1.120 → 0.1.122
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/src/chat-pipeline/serial-trial.js +13 -3
- package/src/cli/commands/daemon.js +21 -0
- package/src/cli/commands/provider/add.js +1 -0
- package/src/cli/policy.js +4 -0
- package/src/peers.js +36 -4
- package/src/providers/workbuddy/chat.js +7 -3
- package/src/providers/workbuddy/sanitize-tools.js +40 -34
- package/src/providers/workbuddy/sdk-chat.js +2 -1
- package/src/routes/chat/peer-handler.js +11 -3
- package/src/routes/chat/via-route-handler.js +1 -1
- package/src/routes/hedge.js +2 -1
- package/src/routes/peers.js +22 -3
- package/src/runtime/auto-update.js +6 -1
- package/src/runtime/providers-setup.js +2 -2
- package/src/runtime/server-lifecycle.js +9 -1
- package/src/upstream-engine/sdk/attempt.js +28 -2
- package/src/upstream-engine/sdk/diagnose.js +61 -0
package/package.json
CHANGED
|
@@ -107,9 +107,18 @@ export async function runSerialTrial(ctx, deps = {}) {
|
|
|
107
107
|
return json(res, 403, errBody);
|
|
108
108
|
}
|
|
109
109
|
if (auto) await auto.recordError(model, { status: upRes.status });
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
110
|
+
// 读失败响应体(clone 不影响后续 relay 转发原响应;1s 上限防流式错误体拖慢)
|
|
111
|
+
let upBody = "";
|
|
112
|
+
try {
|
|
113
|
+
upBody = String(await Promise.race([
|
|
114
|
+
upRes.clone().text(),
|
|
115
|
+
new Promise((r) => { const t = setTimeout(() => r(""), 1000); t.unref?.(); }),
|
|
116
|
+
])).replace(/\s+/g, " ").slice(0, 400);
|
|
117
|
+
} catch {}
|
|
118
|
+
const upMsg = upBody || `upstream ${upRes.status}`;
|
|
119
|
+
lastErr = { model, upstream: upRes, status: upRes.status, message: upMsg };
|
|
120
|
+
logError(model, upRes.status, `upstream ${upRes.status}${upBody ? ` body=${upBody.slice(0, 300)}` : ""}`);
|
|
121
|
+
evt("upstream-error", { reqId, model, status: upRes.status, message: upMsg.slice(0, 300), timing: upRes._t ?? null });
|
|
113
122
|
upRes = null;
|
|
114
123
|
}
|
|
115
124
|
if (upRes) {
|
|
@@ -138,6 +147,7 @@ export async function runSerialTrial(ctx, deps = {}) {
|
|
|
138
147
|
} else {
|
|
139
148
|
const pr = await peerRelay({ model, body, lastErr, requested, useAuto, lockModel, auto, peers, handlerCtx, evt, logCall, mark, perf0, stages, startedAt, plugins, res });
|
|
140
149
|
if (pr.handled) return { done: true };
|
|
150
|
+
if (pr.lastErr) lastErr = pr.lastErr;
|
|
141
151
|
}
|
|
142
152
|
}
|
|
143
153
|
if (groups) {
|
|
@@ -95,8 +95,29 @@ export async function handleDaemonFlag(args, VERSION) {
|
|
|
95
95
|
|
|
96
96
|
export async function handleDebug(args) {
|
|
97
97
|
if (!(args.includes("-debug") || args.includes("--debug"))) return false;
|
|
98
|
+
// systemd 自启协同:若后台 daemon 由 user service(Restart=always, RestartSec=3)托管,
|
|
99
|
+
// 仅 stopDaemon() 会让 systemd 3 秒后拉起新实例 → 抢 8989 → EADDRINUSE 自愈反杀本 debug 前台。
|
|
100
|
+
// 必须先 systemctl stop(主动停止不会被 Restart 拉起)。
|
|
101
|
+
if (process.platform === "linux") {
|
|
102
|
+
try {
|
|
103
|
+
const { execFile } = await import("node:child_process");
|
|
104
|
+
const active = await new Promise((res) => execFile("systemctl", ["--user", "is-active", "mslxdff"], { windowsHide: true, timeout: 4000 }, (e, so) => res(String(so || "").trim())));
|
|
105
|
+
if (active === "active") {
|
|
106
|
+
await new Promise((res) => execFile("systemctl", ["--user", "stop", "mslxdff"], { windowsHide: true, timeout: 6000 }, () => res()));
|
|
107
|
+
console.log("[debug] stopped systemd user service (mslxdff) — it will be restarted on exit");
|
|
108
|
+
}
|
|
109
|
+
} catch {}
|
|
110
|
+
}
|
|
98
111
|
const { stopped, pid } = stopDaemon();
|
|
99
112
|
if (stopped) console.log(`[debug] stopped background daemon (pid ${pid})`);
|
|
113
|
+
// 等旧 daemon 真正退出再抢端口(Windows 端口释放有延迟,否则 EADDRINUSE 会让 debug 立即崩)
|
|
114
|
+
if (stopped && pid) {
|
|
115
|
+
const t0 = Date.now();
|
|
116
|
+
while (isPidAlive(pid) && Date.now() - t0 < 4000) {
|
|
117
|
+
await new Promise((r) => setTimeout(r, 100));
|
|
118
|
+
}
|
|
119
|
+
await new Promise((r) => setTimeout(r, 150));
|
|
120
|
+
}
|
|
100
121
|
try {
|
|
101
122
|
const dir = logDir();
|
|
102
123
|
const toClear = [eventsFile(), callsFile(), errorsFile(), logFile()];
|
|
@@ -101,6 +101,7 @@ export async function handleProviderAdd(id, sub, rest) {
|
|
|
101
101
|
console.log(` share: ${loadProviderShareKeys(nid) ? "ON" : "off"} (mslxdff -provider ${nid} share on|off)`);
|
|
102
102
|
console.log(` allowAny: OFF (secure, empty allowlist = 403 block before upstream) — enable via: mslxdff -provider ${nid} allowAny on`);
|
|
103
103
|
console.log(` use as: ${nid}/<model-id> — restart daemon to activate`);
|
|
104
|
+
if (allowedModels.length) console.log(` opencode 中使用: mslxdff -setto opencode ${nid}/${allowedModels[0]} (同步进 opencode.json,会一并设为默认模型)`);
|
|
104
105
|
console.log(` NOTE: empty allowlist = 403 before upstream, no cost — must set allowlist to use`);
|
|
105
106
|
process.exit(0);
|
|
106
107
|
}
|
package/src/cli/policy.js
CHANGED
|
@@ -89,6 +89,10 @@ export function peerCooldownMs() {
|
|
|
89
89
|
const n = Number(process.env.MSLXDFF_PEER_COOLDOWN_MS);
|
|
90
90
|
return Number.isInteger(n) && n > 0 ? n : 30_000;
|
|
91
91
|
}
|
|
92
|
+
export function peerLimitCooldownMs() {
|
|
93
|
+
const n = Number(process.env.MSLXDFF_PEER_LIMIT_COOLDOWN_MS);
|
|
94
|
+
return Number.isInteger(n) && n > 0 ? n : 5 * 60_000;
|
|
95
|
+
}
|
|
92
96
|
export function peerHeatMs() {
|
|
93
97
|
const n = Number(process.env.MSLXDFF_PEER_HEAT_MS);
|
|
94
98
|
return Number.isInteger(n) && n > 0 ? n : 5 * 60_000;
|
package/src/peers.js
CHANGED
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
import { loadPeers, savePeers, loadPeerErrors, savePeerErrors, loadPeerStats, savePeerStats } from "./state.js";
|
|
2
2
|
|
|
3
3
|
export const DEFAULT_PEER_COOLDOWN_MS = 30_000;
|
|
4
|
+
export const DEFAULT_PEER_LIMIT_COOLDOWN_MS = 5 * 60_000;
|
|
4
5
|
export const DEFAULT_PEER_HEAT_MS = 5 * 60_000;
|
|
5
6
|
export const DEFAULT_MAX_HOPS = 3;
|
|
6
7
|
export const DEFAULT_BROADBAND_STALE_MS = 90_000;
|
|
@@ -28,6 +29,7 @@ export function createPeersService({
|
|
|
28
29
|
file,
|
|
29
30
|
now = () => Date.now(),
|
|
30
31
|
cooldownMs = DEFAULT_PEER_COOLDOWN_MS,
|
|
32
|
+
limitCooldownMs = DEFAULT_PEER_LIMIT_COOLDOWN_MS,
|
|
31
33
|
heatMs = DEFAULT_PEER_HEAT_MS,
|
|
32
34
|
peers: seedPeers,
|
|
33
35
|
errors: seedErrors,
|
|
@@ -75,10 +77,18 @@ export function createPeersService({
|
|
|
75
77
|
return before - list.length;
|
|
76
78
|
}
|
|
77
79
|
|
|
80
|
+
// 分级冷却:连续 429(streak >= 2)→ limitCooldownMs(限流恢复慢,5min 内不再考虑);
|
|
81
|
+
// 其他失败(网络抖动、5xx)→ cooldownMs(可能几秒恢复,给快速重试机会)。
|
|
82
|
+
function coolingWindowMs(url) {
|
|
83
|
+
const streak = stats[url]?.streak429 ?? 0;
|
|
84
|
+
return streak >= 2 ? limitCooldownMs : cooldownMs;
|
|
85
|
+
}
|
|
86
|
+
|
|
78
87
|
function isCooling(url) {
|
|
79
|
-
|
|
88
|
+
const win = coolingWindowMs(url);
|
|
89
|
+
if (!win) return false;
|
|
80
90
|
const err = lastErrorAt[url];
|
|
81
|
-
if (typeof err === "number" && now() - err <
|
|
91
|
+
if (typeof err === "number" && now() - err < win) return true;
|
|
82
92
|
const peer = list.find((p) => p.url === url);
|
|
83
93
|
if (peer && isBroadbandStale(peer, now(), broadbandStaleMs())) return true;
|
|
84
94
|
return false;
|
|
@@ -148,6 +158,21 @@ export function createPeersService({
|
|
|
148
158
|
|
|
149
159
|
let cursor = 0;
|
|
150
160
|
|
|
161
|
+
// 兜底重试集合:仅"因错误冷却中"的 peer(按最早失败优先)。
|
|
162
|
+
// 场景:唯一可用节点偶发失败被冷却锁死 → 主链零候选/全败时,兜底给它一次机会;
|
|
163
|
+
// 成功会被 recordResult 清除错误记录、回归热路径,避免"明明能用的节点被 30s 冷却全灭"。
|
|
164
|
+
function coolingByLastError() {
|
|
165
|
+
const t = now();
|
|
166
|
+
return list
|
|
167
|
+
.filter((p) => {
|
|
168
|
+
const err = lastErrorAt[p.url];
|
|
169
|
+
if (typeof err !== "number") return false;
|
|
170
|
+
if ((stats[p.url]?.streak429 ?? 0) >= 2) return false; // 429 长冷却:兜底也不考虑
|
|
171
|
+
return t - err < cooldownMs;
|
|
172
|
+
})
|
|
173
|
+
.sort((a, b) => (lastErrorAt[a.url] ?? 0) - (lastErrorAt[b.url] ?? 0));
|
|
174
|
+
}
|
|
175
|
+
|
|
151
176
|
function next() {
|
|
152
177
|
const avail = available();
|
|
153
178
|
if (!avail.length) return null;
|
|
@@ -158,10 +183,16 @@ export function createPeersService({
|
|
|
158
183
|
// Long-lived error memory: a peer keeps its last-error timestamp until a
|
|
159
184
|
// subsequent success resets it (success clears the failure record) or the
|
|
160
185
|
// error is no longer in the persist store on next load.
|
|
161
|
-
async function recordError(url) {
|
|
186
|
+
async function recordError(url, { status } = {}) {
|
|
162
187
|
if (!url) return;
|
|
163
188
|
lastErrorAt[url] = now();
|
|
164
189
|
await persistErrors({ ...lastErrorAt });
|
|
190
|
+
// 连续 429 计数(仅成功清零):第 2 次起进入长冷却——该 peer 大概率真被限流,不再频繁试它
|
|
191
|
+
if (Number(status) === 429) {
|
|
192
|
+
const prev = stats[url] || {};
|
|
193
|
+
stats[url] = { ...prev, streak429: (prev.streak429 || 0) + 1 };
|
|
194
|
+
await persistStats({ ...stats });
|
|
195
|
+
}
|
|
165
196
|
}
|
|
166
197
|
|
|
167
198
|
// Outcome of a forwarded request: ok updates the hot-cache (EMA latency,
|
|
@@ -182,6 +213,7 @@ export function createPeersService({
|
|
|
182
213
|
: (typeof latencyMs === "number" ? latencyMs : prev.latencyMs ?? 0),
|
|
183
214
|
fails: 0,
|
|
184
215
|
model: model || prev.model || "",
|
|
216
|
+
streak429: 0,
|
|
185
217
|
};
|
|
186
218
|
} else {
|
|
187
219
|
const prev = stats[url] || {};
|
|
@@ -194,7 +226,7 @@ export function createPeersService({
|
|
|
194
226
|
all, add, remove, removeByGroup, isCooling, isBroadbandCooling, isBroadbandStale: (url) => {
|
|
195
227
|
const peer = list.find((p) => p.url === url);
|
|
196
228
|
return peer ? isBroadbandStale(peer, now(), broadbandStaleMs()) : false;
|
|
197
|
-
}, isHot, stat, ordered, orderedByLastError, available, next,
|
|
229
|
+
}, isHot, stat, ordered, orderedByLastError, coolingByLastError, available, next,
|
|
198
230
|
recordError, recordResult, errors: () => ({ ...lastErrorAt }), stats: () => ({ ...stats }),
|
|
199
231
|
};
|
|
200
232
|
}
|
|
@@ -2,6 +2,7 @@ import { joinUrl } from "../base.js";
|
|
|
2
2
|
import { isAuthError, isInsufficientStatus } from "./auth.js";
|
|
3
3
|
import { appendRotationLog as defaultAppend } from "./rotation-log.js";
|
|
4
4
|
import { createTransport } from "../../transport/index.js";
|
|
5
|
+
import { dispatcherFetch } from "../../upstream-engine/sdk/attempt.js";
|
|
5
6
|
import { reshapeWorkbuddySse } from "./reshape.js";
|
|
6
7
|
import { attemptOnceSdk } from "./sdk-chat.js";
|
|
7
8
|
import { sanitizeToolSequence } from "./sanitize-tools.js";
|
|
@@ -88,6 +89,9 @@ export function createChatService({
|
|
|
88
89
|
try { defaultAppend(opts); } catch {}
|
|
89
90
|
};
|
|
90
91
|
const transport = createTransport({ fetchImpl, dispatcher, keepAlive: !!dispatcher, timeoutMs: connectTimeoutMs, retry: {} });
|
|
92
|
+
// SDK 通道与 legacy 同享 keep-alive 连接池(dispatcher → fetch 注入);同时该 fetch 被 attempt 层
|
|
93
|
+
// 包装用于错误路径捕获请求体(上游 4xx 时打印 tool 序列断裂诊断,见 sdk/attempt.js)。
|
|
94
|
+
const sdkFetch = dispatcher ? dispatcherFetch(dispatcher) : fetchImpl;
|
|
91
95
|
|
|
92
96
|
function authForKey(key) {
|
|
93
97
|
const idx = keys.indexOf(key);
|
|
@@ -109,7 +113,7 @@ export function createChatService({
|
|
|
109
113
|
const payload = rewriteWorkbuddyPayload(body);
|
|
110
114
|
if (engineMode === "sdk") {
|
|
111
115
|
try {
|
|
112
|
-
return await attemptOnceSdk({ url, body: payload, key, auth, buildHeaders: buildAuthHeaders });
|
|
116
|
+
return await attemptOnceSdk({ url, body: payload, key, auth, buildHeaders: buildAuthHeaders, ...(sdkFetch ? { fetchImpl: sdkFetch } : {}) });
|
|
113
117
|
} catch (e) {
|
|
114
118
|
if (!e || !e._sdkLoadFailed) throw e;
|
|
115
119
|
if (!sdkFallbackLogged) {
|
|
@@ -157,8 +161,8 @@ export function createChatService({
|
|
|
157
161
|
const url = joinUrl(baseUrl, chatPath);
|
|
158
162
|
const t0 = nowMs(clock);
|
|
159
163
|
const clean = sanitizeToolSequence(body?.messages);
|
|
160
|
-
if (clean.droppedCalls || clean.droppedResults) {
|
|
161
|
-
try { console.error(`[workbuddy] tool-sequence sanitized: dropped ${clean.droppedCalls} call(s), ${clean.droppedResults} result(s), model=${body?.model || ""}`); } catch {}
|
|
164
|
+
if (clean.droppedCalls || clean.droppedResults || clean.movedResults || clean.injectedHead) {
|
|
165
|
+
try { console.error(`[workbuddy] tool-sequence sanitized: dropped ${clean.droppedCalls} call(s), ${clean.droppedResults} result(s), moved ${clean.movedResults} result(s), head+${clean.injectedHead}, model=${body?.model || ""}`); } catch {}
|
|
162
166
|
body = { ...body, messages: clean.messages };
|
|
163
167
|
}
|
|
164
168
|
_t0Val = t0;
|
|
@@ -1,55 +1,61 @@
|
|
|
1
|
-
// workbuddy 上游对 tool 序列严格校验(400 code 11148 tool_call_sequence_broken
|
|
2
|
-
// assistant.tool_calls
|
|
3
|
-
//
|
|
4
|
-
//
|
|
5
|
-
//
|
|
6
|
-
//
|
|
1
|
+
// workbuddy 上游对 tool 序列严格校验(400 code 11148 tool_call_sequence_broken),实测三条规则:
|
|
2
|
+
// ① assistant.tool_calls 与结果必须按 id 一一配对(孤立调用/结果均拒);
|
|
3
|
+
// ② 结果必须紧跟对应 assistant(配对不能被 user/assistant 消息打断);
|
|
4
|
+
// ③ 序列不能以"带 tool_calls 的 assistant"开头(tool call 无前置上下文即拒;compaction 裁剪点
|
|
5
|
+
// 落在工具链中间时最易触发)。
|
|
6
|
+
// 这里在 workbuddy 出口做防御性清洗:结果前移到对应调用之后(组内按 calls 顺序)、剔无法配对的
|
|
7
|
+
// 调用/结果、首条为 assistant/tool 时注入占位 user("continue")。其他上游宽容不受影响——只挂 workbuddy。
|
|
7
8
|
export function sanitizeToolSequence(messages) {
|
|
8
9
|
const list = Array.isArray(messages) ? messages : [];
|
|
9
10
|
|
|
10
|
-
const
|
|
11
|
-
|
|
12
|
-
if (m?.role
|
|
13
|
-
|
|
14
|
-
|
|
15
|
-
|
|
16
|
-
}
|
|
17
|
-
}
|
|
18
|
-
}
|
|
19
|
-
|
|
20
|
-
const resultIds = new Set();
|
|
21
|
-
for (const m of list) {
|
|
22
|
-
if (m?.role === "tool") {
|
|
23
|
-
const id = String(m.tool_call_id || "");
|
|
24
|
-
if (id) resultIds.add(id);
|
|
25
|
-
}
|
|
26
|
-
}
|
|
27
|
-
|
|
28
|
-
const paired = new Set();
|
|
29
|
-
for (const id of callIds) if (resultIds.has(id)) paired.add(id);
|
|
11
|
+
const resultById = new Map();
|
|
12
|
+
list.forEach((m, idx) => {
|
|
13
|
+
if (m?.role !== "tool") return;
|
|
14
|
+
const id = String(m.tool_call_id || "");
|
|
15
|
+
if (id && !resultById.has(id)) resultById.set(id, { msg: m, idx, used: false });
|
|
16
|
+
});
|
|
30
17
|
|
|
31
18
|
let droppedCalls = 0;
|
|
32
19
|
let droppedResults = 0;
|
|
20
|
+
let movedResults = 0;
|
|
33
21
|
const out = [];
|
|
34
|
-
|
|
22
|
+
list.forEach((m, idx) => {
|
|
35
23
|
if (m?.role === "assistant" && Array.isArray(m.tool_calls) && m.tool_calls.length) {
|
|
36
|
-
const kept =
|
|
37
|
-
|
|
24
|
+
const kept = [];
|
|
25
|
+
for (const tc of m.tool_calls) {
|
|
26
|
+
const r = resultById.get(String(tc?.id || ""));
|
|
27
|
+
if (r && !r.used) kept.push(tc);
|
|
28
|
+
else droppedCalls++;
|
|
29
|
+
}
|
|
38
30
|
if (kept.length) {
|
|
39
31
|
out.push({ ...m, tool_calls: kept });
|
|
32
|
+
for (const tc of kept) {
|
|
33
|
+
const r = resultById.get(String(tc.id));
|
|
34
|
+
r.used = true;
|
|
35
|
+
if (!(r.idx >= idx + 1 && r.idx <= idx + kept.length)) movedResults++;
|
|
36
|
+
out.push(r.msg);
|
|
37
|
+
}
|
|
40
38
|
} else if (typeof m.content === "string" && m.content) {
|
|
41
39
|
const { tool_calls: _dropped, ...rest } = m;
|
|
42
40
|
out.push(rest);
|
|
43
41
|
}
|
|
44
|
-
|
|
42
|
+
return;
|
|
45
43
|
}
|
|
46
44
|
if (m?.role === "tool") {
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
45
|
+
const id = String(m.tool_call_id || "");
|
|
46
|
+
const r = resultById.get(id);
|
|
47
|
+
if (r && r.idx === idx && r.used) return; // 已前移消费,跳过原位置
|
|
48
|
+
droppedResults++; // 孤儿或重复
|
|
49
|
+
return;
|
|
50
50
|
}
|
|
51
51
|
out.push(m);
|
|
52
|
+
});
|
|
53
|
+
|
|
54
|
+
let injectedHead = 0;
|
|
55
|
+
if (out.length && out[0]?.role !== "user" && out[0]?.role !== "system" && out[0]?.role !== "developer") {
|
|
56
|
+
out.unshift({ role: "user", content: "continue" });
|
|
57
|
+
injectedHead = 1;
|
|
52
58
|
}
|
|
53
59
|
|
|
54
|
-
return { messages: out, droppedCalls, droppedResults };
|
|
60
|
+
return { messages: out, droppedCalls, droppedResults, movedResults, injectedHead };
|
|
55
61
|
}
|
|
@@ -8,7 +8,7 @@ export { sdkBaseFromUrl };
|
|
|
8
8
|
|
|
9
9
|
let channelLogged = false;
|
|
10
10
|
|
|
11
|
-
export async function attemptOnceSdk({ url, body, key, auth, buildHeaders, clock = Date.now } = {}) {
|
|
11
|
+
export async function attemptOnceSdk({ url, body, key, auth, buildHeaders, clock = Date.now, fetchImpl } = {}) {
|
|
12
12
|
const out = await attemptGeneric({
|
|
13
13
|
url,
|
|
14
14
|
body,
|
|
@@ -16,6 +16,7 @@ export async function attemptOnceSdk({ url, body, key, auth, buildHeaders, clock
|
|
|
16
16
|
providerName: "workbuddy",
|
|
17
17
|
marker: { name: "x-mslxdff-workbuddy-channel", value: "sdk" },
|
|
18
18
|
clock,
|
|
19
|
+
...(fetchImpl ? { fetchImpl } : {}),
|
|
19
20
|
});
|
|
20
21
|
if (!channelLogged) {
|
|
21
22
|
channelLogged = true;
|
|
@@ -24,12 +24,20 @@ export async function handlePeerRelay({
|
|
|
24
24
|
res,
|
|
25
25
|
}) {
|
|
26
26
|
evt("peer-race-start", { reqId: handlerCtx.reqId, model, peers: peers.ordered().length });
|
|
27
|
+
const peerErrors = [];
|
|
28
|
+
const pctx = { ...handlerCtx, peerErrors };
|
|
27
29
|
const win =
|
|
28
|
-
(await racePeerCandidates(peers.ordered(),
|
|
29
|
-
(await racePeerCandidates(peers.orderedByLastError(),
|
|
30
|
+
(await racePeerCandidates(peers.ordered(), pctx)) ||
|
|
31
|
+
(await racePeerCandidates(peers.orderedByLastError(), pctx)) ||
|
|
32
|
+
(await racePeerCandidates(peers.coolingByLastError(), pctx));
|
|
30
33
|
if (!win) {
|
|
31
34
|
evt("peer-race-lose", { reqId: handlerCtx.reqId, model });
|
|
32
|
-
|
|
35
|
+
// 组员全失败:把每个组员的真实返回汇总给调用者(否则只剩一个无信息的 429/502)
|
|
36
|
+
const detail = peerErrors.length
|
|
37
|
+
? peerErrors.map((e) => `${e.peer} -> ${e.status}${e.message ? ` (${String(e.message).slice(0, 160)})` : ""}`).join("; ")
|
|
38
|
+
: "no peers available";
|
|
39
|
+
const allLimit = peerErrors.length > 0 && peerErrors.every((e) => e.status === 429);
|
|
40
|
+
return { handled: false, lastErr: { model, upstream: null, status: allLimit ? 429 : 502, message: `peers failed: ${detail}` } };
|
|
33
41
|
}
|
|
34
42
|
evt("peer-race-win", { reqId: handlerCtx.reqId, model, winPeer: win.peer.url, winTarget: win.target, latencyMs: win.latencyMs });
|
|
35
43
|
await peers.recordResult(win.peer.url, { ok: true, latencyMs: win.latencyMs, model: win.target });
|
|
@@ -104,7 +104,7 @@ export async function handleViaRoute({
|
|
|
104
104
|
try { bodyText = await upRes.clone().text(); } catch {}
|
|
105
105
|
const msg = bodyText.slice(0, 300) || errMsg(upRes) || `peer ${status}`;
|
|
106
106
|
evt("via-route-peer-error", { reqId: handlerCtx.reqId, peer: peer.url, model, status, message: msg.slice(0, 200) });
|
|
107
|
-
try { await peers.recordError(peer.url); } catch {}
|
|
107
|
+
try { await peers.recordError(peer.url, { status }); } catch {}
|
|
108
108
|
try { await peers.recordResult(peer.url, { ok: false }); } catch {}
|
|
109
109
|
// 502/429 等可 fallback 到 direct
|
|
110
110
|
return { handled: false, lastErr: { model, upstream: upRes, status, message: msg } };
|
package/src/routes/hedge.js
CHANGED
|
@@ -136,7 +136,8 @@ export async function hedgedFirstChunkRace({
|
|
|
136
136
|
try {
|
|
137
137
|
peerWin =
|
|
138
138
|
(await racePeerCandidates(candidates, handlerCtx)) ||
|
|
139
|
-
(await racePeerCandidates(handlerCtx.peers.orderedByLastError(), handlerCtx))
|
|
139
|
+
(await racePeerCandidates(handlerCtx.peers.orderedByLastError(), handlerCtx)) ||
|
|
140
|
+
(await racePeerCandidates(handlerCtx.peers.coolingByLastError(), handlerCtx));
|
|
140
141
|
} catch (_) {
|
|
141
142
|
peerWin = null;
|
|
142
143
|
}
|
package/src/routes/peers.js
CHANGED
|
@@ -142,7 +142,9 @@ export async function racePeerCandidates(candidates, ctx) {
|
|
|
142
142
|
}
|
|
143
143
|
if (failed) {
|
|
144
144
|
const status = res instanceof Error ? 502 : res.status;
|
|
145
|
-
|
|
145
|
+
const failRec = { peer: peer.url, status, message: res instanceof Error ? errMsg(res) : null };
|
|
146
|
+
if (Array.isArray(ctx.peerErrors)) ctx.peerErrors.push(failRec);
|
|
147
|
+
ctx.logError(ctx.model, status, res instanceof Error ? `peer ${peer.url} ${errMsg(res)}` : `peer ${peer.url} ${status}`);
|
|
146
148
|
ctx.evt("peer-error", { peer: peer.url, model: target, status, message: res instanceof Error ? errMsg(res) : null });
|
|
147
149
|
order.push({ ok: false, peer, target, res, status });
|
|
148
150
|
} else {
|
|
@@ -157,7 +159,7 @@ export async function racePeerCandidates(candidates, ctx) {
|
|
|
157
159
|
for (const o of completed) {
|
|
158
160
|
if (o === winner) continue;
|
|
159
161
|
if (!o.ok) {
|
|
160
|
-
await ctx.peers.recordError(o.peer.url);
|
|
162
|
+
await ctx.peers.recordError(o.peer.url, { status: o.status });
|
|
161
163
|
await ctx.peers.recordResult(o.peer.url, { ok: false });
|
|
162
164
|
} else {
|
|
163
165
|
await ctx.peers.recordResult(o.peer.url, { ok: true, latencyMs: o.latencyMs, model: o.target });
|
|
@@ -165,8 +167,25 @@ export async function racePeerCandidates(candidates, ctx) {
|
|
|
165
167
|
}
|
|
166
168
|
return { peer: winner.peer, target: winner.target, res: winner.res, latencyMs: winner.latencyMs };
|
|
167
169
|
}
|
|
170
|
+
// 全失败:补读失败响应体(诊断 + 调用者详情)。仅在"本轮无 winner"时执行,
|
|
171
|
+
// 不拖慢成功路径;并行读、每 peer 上限 600ms,body 里才有 400/429 的真实原因。
|
|
172
|
+
await Promise.all(completed.map(async (o) => {
|
|
173
|
+
if (o.ok || !o.res || typeof o.res !== "object" || o.res instanceof Error) return;
|
|
174
|
+
try {
|
|
175
|
+
const snip = String(await Promise.race([
|
|
176
|
+
o.res.clone().text(),
|
|
177
|
+
new Promise((r) => { const t = setTimeout(() => r(""), 600); t.unref?.(); }),
|
|
178
|
+
])).replace(/\s+/g, " ").slice(0, 300);
|
|
179
|
+
if (!snip) return;
|
|
180
|
+
if (Array.isArray(ctx.peerErrors)) {
|
|
181
|
+
const rec = ctx.peerErrors.find((x) => x.peer === o.peer.url && x.status === o.status && !x.message);
|
|
182
|
+
if (rec) rec.message = snip;
|
|
183
|
+
}
|
|
184
|
+
ctx.logError(ctx.model, o.status, `peer ${o.peer.url} ${o.status} body=${snip}`);
|
|
185
|
+
} catch {}
|
|
186
|
+
}));
|
|
168
187
|
for (const o of completed) {
|
|
169
|
-
await ctx.peers.recordError(o.peer.url);
|
|
188
|
+
await ctx.peers.recordError(o.peer.url, { status: o.status });
|
|
170
189
|
await ctx.peers.recordResult(o.peer.url, { ok: false });
|
|
171
190
|
}
|
|
172
191
|
}
|
|
@@ -11,7 +11,12 @@ export function setupAutoUpdate({ VERSION, bus, logs }) {
|
|
|
11
11
|
const line = `[auto-update] ${type} ${JSON.stringify(data)}`;
|
|
12
12
|
console.log(line);
|
|
13
13
|
}
|
|
14
|
-
|
|
14
|
+
// debug 会话不自动升级:debug 前台会把自己的 pid 写入 daemon.pid,
|
|
15
|
+
// auto-update 的 stopDaemon() 会把它自己停掉(现象:-debug 跑一会儿就"自己退出")
|
|
16
|
+
if (autoUpdateMs && process.env.MSLXDFF_DEBUG === "1") {
|
|
17
|
+
console.log(`auto-update: skipped (debug session)`);
|
|
18
|
+
emitAutoUpdate("auto-update-skipped", { intervalMs: autoUpdateMs, current: VERSION, reason: "debug" });
|
|
19
|
+
} else if (autoUpdateMs) {
|
|
15
20
|
console.log(`auto-update enabled: checking every ${Math.round(autoUpdateMs / 60000)}m`);
|
|
16
21
|
emitAutoUpdate("auto-update-enabled", { intervalMs: autoUpdateMs, current: VERSION });
|
|
17
22
|
setTimeout(() => {
|
|
@@ -11,7 +11,7 @@ import { logDir, appendEvent } from "../logs.js";
|
|
|
11
11
|
import { loadPlugins, runHook, resolvePluginDirs } from "../plugins.js";
|
|
12
12
|
import { createOpenCodeProvider } from "../providers/opencode.js";
|
|
13
13
|
import { loadProviderKeys, loadProviderAuths, loadProviderConfigs } from "../state.js";
|
|
14
|
-
import { refreshIntervalMs, modelCooldownMs, slowCooldownMs, peerCooldownMs, peerHeatMs, banWindowMs, banThreshold } from "../cli/policy.js";
|
|
14
|
+
import { refreshIntervalMs, modelCooldownMs, slowCooldownMs, peerCooldownMs, peerLimitCooldownMs, peerHeatMs, banWindowMs, banThreshold } from "../cli/policy.js";
|
|
15
15
|
import { errMsg } from "../cli/util.js";
|
|
16
16
|
|
|
17
17
|
/**
|
|
@@ -140,7 +140,7 @@ export async function setupProviders() {
|
|
|
140
140
|
}
|
|
141
141
|
},
|
|
142
142
|
});
|
|
143
|
-
const peers = createPeersService({ cooldownMs: peerCooldownMs(), heatMs: peerHeatMs() });
|
|
143
|
+
const peers = createPeersService({ cooldownMs: peerCooldownMs(), limitCooldownMs: peerLimitCooldownMs(), heatMs: peerHeatMs() });
|
|
144
144
|
const groups = createGroupsService({});
|
|
145
145
|
const bans = createBansService({ windowMs: banWindowMs(), threshold: banThreshold() });
|
|
146
146
|
|
|
@@ -37,8 +37,16 @@ export async function startServerLifecycle({ VERSION, token, created, upstream,
|
|
|
37
37
|
} catch {}
|
|
38
38
|
});
|
|
39
39
|
const { startDaemon: sd } = await import("../daemon.js");
|
|
40
|
-
const restore2 = () => {
|
|
40
|
+
const restore2 = async () => {
|
|
41
41
|
console.log("\n[debug] restoring background daemon...");
|
|
42
|
+
try { await srv.close(); } catch {} // 先释放端口,避免恢复的 daemon 抢端口互杀
|
|
43
|
+
if (process.platform === "linux") {
|
|
44
|
+
try {
|
|
45
|
+
const { execFile } = await import("node:child_process");
|
|
46
|
+
const ok = await new Promise((res) => execFile("systemctl", ["--user", "start", "mslxdff"], { windowsHide: true, timeout: 8000 }, (e) => res(!e)));
|
|
47
|
+
if (ok) { console.log("[debug] systemd user service restarted (mslxdff)"); setTimeout(() => process.exit(0), 300); return; }
|
|
48
|
+
} catch {}
|
|
49
|
+
}
|
|
42
50
|
try {
|
|
43
51
|
const restoredPid = sd([]);
|
|
44
52
|
console.log(`[debug] daemon restored (pid ${restoredPid})`);
|
|
@@ -6,6 +6,7 @@
|
|
|
6
6
|
// 见 .scratch/ai-sdk-upstream/{SPEC.md,SPEC-p3-responses.md} 与 docs/adr/0017。
|
|
7
7
|
import { toModelPrompt, toModelTools, toModelToolChoice, toModelParams } from "./convert.js";
|
|
8
8
|
import { createSseSerializer } from "./sse.js";
|
|
9
|
+
import { diagnoseToolSequence, compactSequence } from "./diagnose.js";
|
|
9
10
|
import { getUndici } from "../../compat.js";
|
|
10
11
|
|
|
11
12
|
let sdkPromise = null;
|
|
@@ -44,6 +45,20 @@ export function markerHeaders(marker) {
|
|
|
44
45
|
return marker && marker.name ? { [marker.name]: marker.value } : {};
|
|
45
46
|
}
|
|
46
47
|
|
|
48
|
+
// 错误路径诊断:AI SDK 的 APICallError 对 HTTP 4xx 的 requestBodyValues 是空对象(provider-utils 实测),
|
|
49
|
+
// 故用包装 fetch 捕获真实发出的 body,在上游非 2xx 时打印 tool 序列断裂点与尾部序列。
|
|
50
|
+
function logRequestDiagnosis(bodyText, status) {
|
|
51
|
+
if (!bodyText || typeof bodyText !== "string") return;
|
|
52
|
+
let parsed = null;
|
|
53
|
+
try { parsed = JSON.parse(bodyText); } catch { return; }
|
|
54
|
+
const msgs = parsed?.messages;
|
|
55
|
+
if (!Array.isArray(msgs)) return;
|
|
56
|
+
const { issues, summary } = diagnoseToolSequence(msgs);
|
|
57
|
+
console.error(`[sdk-upstream] ${status} ${summary} issues=${issues.length ? issues.join(" | ") : "none"}`);
|
|
58
|
+
console.error(`[sdk-upstream] head: ${compactSequence(msgs, 0, 6)}`);
|
|
59
|
+
console.error(`[sdk-upstream] tail: ${compactSequence(msgs, Math.max(0, msgs.length - 40))}`);
|
|
60
|
+
}
|
|
61
|
+
|
|
47
62
|
// SDK 抛出的 APICallError → 带状态码的 Response;非 HTTP 错误返回 null 由调用方 rethrow。
|
|
48
63
|
export function errorResponseFromSdkError(e, { marker = null } = {}) {
|
|
49
64
|
const status = Number(e?.statusCode ?? e?.status);
|
|
@@ -125,11 +140,19 @@ export async function attemptOnceSdk({
|
|
|
125
140
|
err._sdkLoadFailed = true;
|
|
126
141
|
throw err;
|
|
127
142
|
}
|
|
143
|
+
// 包装注入的 fetch 以捕获真实请求体(诊断用):不改变调用语义,仅在非 2xx 时被读取。
|
|
144
|
+
let lastBodyText = null;
|
|
145
|
+
const capturedFetch = fetchImpl
|
|
146
|
+
? async (u, init) => {
|
|
147
|
+
try { if (init && typeof init.body === "string") lastBodyText = init.body; } catch {}
|
|
148
|
+
return fetchImpl(u, init);
|
|
149
|
+
}
|
|
150
|
+
: undefined;
|
|
128
151
|
const provider = createOpenAICompatible({
|
|
129
152
|
name: providerName,
|
|
130
153
|
baseURL,
|
|
131
154
|
headers: sanitizeHeaders(headers),
|
|
132
|
-
...(
|
|
155
|
+
...(capturedFetch ? { fetch: capturedFetch } : {}),
|
|
133
156
|
});
|
|
134
157
|
const model = provider.chatModel(String(body?.model || ""));
|
|
135
158
|
let res;
|
|
@@ -142,7 +165,10 @@ export async function attemptOnceSdk({
|
|
|
142
165
|
});
|
|
143
166
|
} catch (e) {
|
|
144
167
|
const mapped = errorResponseFromSdkError(e, { marker });
|
|
145
|
-
if (mapped)
|
|
168
|
+
if (mapped) {
|
|
169
|
+
try { logRequestDiagnosis(lastBodyText, mapped.status); } catch {}
|
|
170
|
+
return mapped;
|
|
171
|
+
}
|
|
146
172
|
throw e;
|
|
147
173
|
}
|
|
148
174
|
return streamResponseFromParts(res.stream, { marker, clock, t0 });
|
|
@@ -0,0 +1,61 @@
|
|
|
1
|
+
// 请求消息序列诊断(纯函数):上游 11148 tool_call_sequence_broken 时定位断裂点。
|
|
2
|
+
// 上游校验规则:assistant.tool_calls 必须与紧随其后的 tool 结果按 id 严格配对,
|
|
3
|
+
// 且配对过程不能被其他消息(user/assistant/system)打断——仅集合级 id 配对(sanitize-tools)不够。
|
|
4
|
+
// 在 SDK 错误路径对 AI SDK 实际发出的请求体做本诊断,异常位置用消息下标指认。
|
|
5
|
+
export function compactSequence(messages, from = 0, limit = Infinity) {
|
|
6
|
+
const list = Array.isArray(messages) ? messages : [];
|
|
7
|
+
const end = Number.isFinite(limit) ? Math.min(list.length, from + limit) : list.length;
|
|
8
|
+
const out = [];
|
|
9
|
+
for (let i = Math.max(0, from); i < end; i++) {
|
|
10
|
+
const m = list[i] || {};
|
|
11
|
+
const role = String(m.role || "?");
|
|
12
|
+
if (role === "assistant") {
|
|
13
|
+
const calls = Array.isArray(m.tool_calls) ? m.tool_calls : [];
|
|
14
|
+
out.push(calls.length ? `A{${calls.map((t) => String(t?.id || "empty")).join(",")}}` : "A");
|
|
15
|
+
} else if (role === "tool") out.push(`T{${String(m.tool_call_id || "empty")}}`);
|
|
16
|
+
else if (role === "user") out.push("U");
|
|
17
|
+
else if (role === "system" || role === "developer") out.push("S");
|
|
18
|
+
else out.push(String(role)[0] || "?");
|
|
19
|
+
}
|
|
20
|
+
return out.join(" ");
|
|
21
|
+
}
|
|
22
|
+
|
|
23
|
+
export function diagnoseToolSequence(messages) {
|
|
24
|
+
const list = Array.isArray(messages) ? messages : [];
|
|
25
|
+
const issues = [];
|
|
26
|
+
const open = new Map();
|
|
27
|
+
let calls = 0;
|
|
28
|
+
let results = 0;
|
|
29
|
+
const first = list[0];
|
|
30
|
+
if (first && first.role !== "user" && first.role !== "system" && first.role !== "developer") {
|
|
31
|
+
issues.push(`#0 首条为 ${String(first.role || "?")}(上游拒带工具链的 assistant/tool 开头)`);
|
|
32
|
+
}
|
|
33
|
+
list.forEach((m, i) => {
|
|
34
|
+
const role = m?.role;
|
|
35
|
+
if (role === "assistant") {
|
|
36
|
+
const tcs = Array.isArray(m.tool_calls) ? m.tool_calls : [];
|
|
37
|
+
if (tcs.length) {
|
|
38
|
+
if (open.size) issues.push(`#${i} assistant 打断未闭合{${[...open.keys()].join(",")}}`);
|
|
39
|
+
for (const tc of tcs) {
|
|
40
|
+
const id = String(tc?.id || "");
|
|
41
|
+
calls++;
|
|
42
|
+
if (!id) issues.push(`#${i} tool_call 空 id`);
|
|
43
|
+
else if (open.has(id)) issues.push(`#${i} call id 重复:${id}`);
|
|
44
|
+
else open.set(id, i);
|
|
45
|
+
}
|
|
46
|
+
} else if (open.size) {
|
|
47
|
+
issues.push(`#${i} assistant(无调用) 打断{${[...open.keys()].join(",")}}`);
|
|
48
|
+
}
|
|
49
|
+
} else if (role === "tool") {
|
|
50
|
+
results++;
|
|
51
|
+
const id = String(m.tool_call_id || "");
|
|
52
|
+
if (!id) issues.push(`#${i} tool 结果空 id`);
|
|
53
|
+
else if (!open.has(id)) issues.push(`#${i} 孤立结果:${id}`);
|
|
54
|
+
else open.delete(id);
|
|
55
|
+
} else if (open.size) {
|
|
56
|
+
issues.push(`#${i} ${role} 打断{${[...open.keys()].join(",")}}`);
|
|
57
|
+
}
|
|
58
|
+
});
|
|
59
|
+
for (const [id, at] of open) issues.push(`#${at} 悬空调用:${id}`);
|
|
60
|
+
return { issues, summary: `msgs=${list.length} calls=${calls} results=${results}` };
|
|
61
|
+
}
|