mslxdff 0.1.121 → 0.1.123
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/src/chat-pipeline/serial-trial.js +13 -3
- package/src/cli/commands/daemon.js +21 -0
- package/src/cli/commands/provider/add.js +1 -0
- package/src/cli/policy.js +4 -0
- package/src/peers.js +36 -4
- package/src/routes/chat/peer-handler.js +11 -3
- package/src/routes/chat/via-route-handler.js +1 -1
- package/src/routes/hedge.js +2 -1
- package/src/routes/peers.js +96 -21
- package/src/runtime/auto-update.js +6 -1
- package/src/runtime/providers-setup.js +2 -2
- package/src/runtime/server-lifecycle.js +9 -1
package/package.json
CHANGED
|
@@ -107,9 +107,18 @@ export async function runSerialTrial(ctx, deps = {}) {
|
|
|
107
107
|
return json(res, 403, errBody);
|
|
108
108
|
}
|
|
109
109
|
if (auto) await auto.recordError(model, { status: upRes.status });
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
110
|
+
// 读失败响应体(clone 不影响后续 relay 转发原响应;1s 上限防流式错误体拖慢)
|
|
111
|
+
let upBody = "";
|
|
112
|
+
try {
|
|
113
|
+
upBody = String(await Promise.race([
|
|
114
|
+
upRes.clone().text(),
|
|
115
|
+
new Promise((r) => { const t = setTimeout(() => r(""), 1000); t.unref?.(); }),
|
|
116
|
+
])).replace(/\s+/g, " ").slice(0, 400);
|
|
117
|
+
} catch {}
|
|
118
|
+
const upMsg = upBody || `upstream ${upRes.status}`;
|
|
119
|
+
lastErr = { model, upstream: upRes, status: upRes.status, message: upMsg };
|
|
120
|
+
logError(model, upRes.status, `upstream ${upRes.status}${upBody ? ` body=${upBody.slice(0, 300)}` : ""}`);
|
|
121
|
+
evt("upstream-error", { reqId, model, status: upRes.status, message: upMsg.slice(0, 300), timing: upRes._t ?? null });
|
|
113
122
|
upRes = null;
|
|
114
123
|
}
|
|
115
124
|
if (upRes) {
|
|
@@ -138,6 +147,7 @@ export async function runSerialTrial(ctx, deps = {}) {
|
|
|
138
147
|
} else {
|
|
139
148
|
const pr = await peerRelay({ model, body, lastErr, requested, useAuto, lockModel, auto, peers, handlerCtx, evt, logCall, mark, perf0, stages, startedAt, plugins, res });
|
|
140
149
|
if (pr.handled) return { done: true };
|
|
150
|
+
if (pr.lastErr) lastErr = pr.lastErr;
|
|
141
151
|
}
|
|
142
152
|
}
|
|
143
153
|
if (groups) {
|
|
@@ -95,8 +95,29 @@ export async function handleDaemonFlag(args, VERSION) {
|
|
|
95
95
|
|
|
96
96
|
export async function handleDebug(args) {
|
|
97
97
|
if (!(args.includes("-debug") || args.includes("--debug"))) return false;
|
|
98
|
+
// systemd 自启协同:若后台 daemon 由 user service(Restart=always, RestartSec=3)托管,
|
|
99
|
+
// 仅 stopDaemon() 会让 systemd 3 秒后拉起新实例 → 抢 8989 → EADDRINUSE 自愈反杀本 debug 前台。
|
|
100
|
+
// 必须先 systemctl stop(主动停止不会被 Restart 拉起)。
|
|
101
|
+
if (process.platform === "linux") {
|
|
102
|
+
try {
|
|
103
|
+
const { execFile } = await import("node:child_process");
|
|
104
|
+
const active = await new Promise((res) => execFile("systemctl", ["--user", "is-active", "mslxdff"], { windowsHide: true, timeout: 4000 }, (e, so) => res(String(so || "").trim())));
|
|
105
|
+
if (active === "active") {
|
|
106
|
+
await new Promise((res) => execFile("systemctl", ["--user", "stop", "mslxdff"], { windowsHide: true, timeout: 6000 }, () => res()));
|
|
107
|
+
console.log("[debug] stopped systemd user service (mslxdff) — it will be restarted on exit");
|
|
108
|
+
}
|
|
109
|
+
} catch {}
|
|
110
|
+
}
|
|
98
111
|
const { stopped, pid } = stopDaemon();
|
|
99
112
|
if (stopped) console.log(`[debug] stopped background daemon (pid ${pid})`);
|
|
113
|
+
// 等旧 daemon 真正退出再抢端口(Windows 端口释放有延迟,否则 EADDRINUSE 会让 debug 立即崩)
|
|
114
|
+
if (stopped && pid) {
|
|
115
|
+
const t0 = Date.now();
|
|
116
|
+
while (isPidAlive(pid) && Date.now() - t0 < 4000) {
|
|
117
|
+
await new Promise((r) => setTimeout(r, 100));
|
|
118
|
+
}
|
|
119
|
+
await new Promise((r) => setTimeout(r, 150));
|
|
120
|
+
}
|
|
100
121
|
try {
|
|
101
122
|
const dir = logDir();
|
|
102
123
|
const toClear = [eventsFile(), callsFile(), errorsFile(), logFile()];
|
|
@@ -101,6 +101,7 @@ export async function handleProviderAdd(id, sub, rest) {
|
|
|
101
101
|
console.log(` share: ${loadProviderShareKeys(nid) ? "ON" : "off"} (mslxdff -provider ${nid} share on|off)`);
|
|
102
102
|
console.log(` allowAny: OFF (secure, empty allowlist = 403 block before upstream) — enable via: mslxdff -provider ${nid} allowAny on`);
|
|
103
103
|
console.log(` use as: ${nid}/<model-id> — restart daemon to activate`);
|
|
104
|
+
if (allowedModels.length) console.log(` opencode 中使用: mslxdff -setto opencode ${nid}/${allowedModels[0]} (同步进 opencode.json,会一并设为默认模型)`);
|
|
104
105
|
console.log(` NOTE: empty allowlist = 403 before upstream, no cost — must set allowlist to use`);
|
|
105
106
|
process.exit(0);
|
|
106
107
|
}
|
package/src/cli/policy.js
CHANGED
|
@@ -89,6 +89,10 @@ export function peerCooldownMs() {
|
|
|
89
89
|
const n = Number(process.env.MSLXDFF_PEER_COOLDOWN_MS);
|
|
90
90
|
return Number.isInteger(n) && n > 0 ? n : 30_000;
|
|
91
91
|
}
|
|
92
|
+
export function peerLimitCooldownMs() {
|
|
93
|
+
const n = Number(process.env.MSLXDFF_PEER_LIMIT_COOLDOWN_MS);
|
|
94
|
+
return Number.isInteger(n) && n > 0 ? n : 5 * 60_000;
|
|
95
|
+
}
|
|
92
96
|
export function peerHeatMs() {
|
|
93
97
|
const n = Number(process.env.MSLXDFF_PEER_HEAT_MS);
|
|
94
98
|
return Number.isInteger(n) && n > 0 ? n : 5 * 60_000;
|
package/src/peers.js
CHANGED
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
import { loadPeers, savePeers, loadPeerErrors, savePeerErrors, loadPeerStats, savePeerStats } from "./state.js";
|
|
2
2
|
|
|
3
3
|
export const DEFAULT_PEER_COOLDOWN_MS = 30_000;
|
|
4
|
+
export const DEFAULT_PEER_LIMIT_COOLDOWN_MS = 5 * 60_000;
|
|
4
5
|
export const DEFAULT_PEER_HEAT_MS = 5 * 60_000;
|
|
5
6
|
export const DEFAULT_MAX_HOPS = 3;
|
|
6
7
|
export const DEFAULT_BROADBAND_STALE_MS = 90_000;
|
|
@@ -28,6 +29,7 @@ export function createPeersService({
|
|
|
28
29
|
file,
|
|
29
30
|
now = () => Date.now(),
|
|
30
31
|
cooldownMs = DEFAULT_PEER_COOLDOWN_MS,
|
|
32
|
+
limitCooldownMs = DEFAULT_PEER_LIMIT_COOLDOWN_MS,
|
|
31
33
|
heatMs = DEFAULT_PEER_HEAT_MS,
|
|
32
34
|
peers: seedPeers,
|
|
33
35
|
errors: seedErrors,
|
|
@@ -75,10 +77,18 @@ export function createPeersService({
|
|
|
75
77
|
return before - list.length;
|
|
76
78
|
}
|
|
77
79
|
|
|
80
|
+
// 分级冷却:连续 429(streak >= 2)→ limitCooldownMs(限流恢复慢,5min 内不再考虑);
|
|
81
|
+
// 其他失败(网络抖动、5xx)→ cooldownMs(可能几秒恢复,给快速重试机会)。
|
|
82
|
+
function coolingWindowMs(url) {
|
|
83
|
+
const streak = stats[url]?.streak429 ?? 0;
|
|
84
|
+
return streak >= 2 ? limitCooldownMs : cooldownMs;
|
|
85
|
+
}
|
|
86
|
+
|
|
78
87
|
function isCooling(url) {
|
|
79
|
-
|
|
88
|
+
const win = coolingWindowMs(url);
|
|
89
|
+
if (!win) return false;
|
|
80
90
|
const err = lastErrorAt[url];
|
|
81
|
-
if (typeof err === "number" && now() - err <
|
|
91
|
+
if (typeof err === "number" && now() - err < win) return true;
|
|
82
92
|
const peer = list.find((p) => p.url === url);
|
|
83
93
|
if (peer && isBroadbandStale(peer, now(), broadbandStaleMs())) return true;
|
|
84
94
|
return false;
|
|
@@ -148,6 +158,21 @@ export function createPeersService({
|
|
|
148
158
|
|
|
149
159
|
let cursor = 0;
|
|
150
160
|
|
|
161
|
+
// 兜底重试集合:仅"因错误冷却中"的 peer(按最早失败优先)。
|
|
162
|
+
// 场景:唯一可用节点偶发失败被冷却锁死 → 主链零候选/全败时,兜底给它一次机会;
|
|
163
|
+
// 成功会被 recordResult 清除错误记录、回归热路径,避免"明明能用的节点被 30s 冷却全灭"。
|
|
164
|
+
function coolingByLastError() {
|
|
165
|
+
const t = now();
|
|
166
|
+
return list
|
|
167
|
+
.filter((p) => {
|
|
168
|
+
const err = lastErrorAt[p.url];
|
|
169
|
+
if (typeof err !== "number") return false;
|
|
170
|
+
if ((stats[p.url]?.streak429 ?? 0) >= 2) return false; // 429 长冷却:兜底也不考虑
|
|
171
|
+
return t - err < cooldownMs;
|
|
172
|
+
})
|
|
173
|
+
.sort((a, b) => (lastErrorAt[a.url] ?? 0) - (lastErrorAt[b.url] ?? 0));
|
|
174
|
+
}
|
|
175
|
+
|
|
151
176
|
function next() {
|
|
152
177
|
const avail = available();
|
|
153
178
|
if (!avail.length) return null;
|
|
@@ -158,10 +183,16 @@ export function createPeersService({
|
|
|
158
183
|
// Long-lived error memory: a peer keeps its last-error timestamp until a
|
|
159
184
|
// subsequent success resets it (success clears the failure record) or the
|
|
160
185
|
// error is no longer in the persist store on next load.
|
|
161
|
-
async function recordError(url) {
|
|
186
|
+
async function recordError(url, { status } = {}) {
|
|
162
187
|
if (!url) return;
|
|
163
188
|
lastErrorAt[url] = now();
|
|
164
189
|
await persistErrors({ ...lastErrorAt });
|
|
190
|
+
// 连续 429 计数(仅成功清零):第 2 次起进入长冷却——该 peer 大概率真被限流,不再频繁试它
|
|
191
|
+
if (Number(status) === 429) {
|
|
192
|
+
const prev = stats[url] || {};
|
|
193
|
+
stats[url] = { ...prev, streak429: (prev.streak429 || 0) + 1 };
|
|
194
|
+
await persistStats({ ...stats });
|
|
195
|
+
}
|
|
165
196
|
}
|
|
166
197
|
|
|
167
198
|
// Outcome of a forwarded request: ok updates the hot-cache (EMA latency,
|
|
@@ -182,6 +213,7 @@ export function createPeersService({
|
|
|
182
213
|
: (typeof latencyMs === "number" ? latencyMs : prev.latencyMs ?? 0),
|
|
183
214
|
fails: 0,
|
|
184
215
|
model: model || prev.model || "",
|
|
216
|
+
streak429: 0,
|
|
185
217
|
};
|
|
186
218
|
} else {
|
|
187
219
|
const prev = stats[url] || {};
|
|
@@ -194,7 +226,7 @@ export function createPeersService({
|
|
|
194
226
|
all, add, remove, removeByGroup, isCooling, isBroadbandCooling, isBroadbandStale: (url) => {
|
|
195
227
|
const peer = list.find((p) => p.url === url);
|
|
196
228
|
return peer ? isBroadbandStale(peer, now(), broadbandStaleMs()) : false;
|
|
197
|
-
}, isHot, stat, ordered, orderedByLastError, available, next,
|
|
229
|
+
}, isHot, stat, ordered, orderedByLastError, coolingByLastError, available, next,
|
|
198
230
|
recordError, recordResult, errors: () => ({ ...lastErrorAt }), stats: () => ({ ...stats }),
|
|
199
231
|
};
|
|
200
232
|
}
|
|
@@ -24,12 +24,20 @@ export async function handlePeerRelay({
|
|
|
24
24
|
res,
|
|
25
25
|
}) {
|
|
26
26
|
evt("peer-race-start", { reqId: handlerCtx.reqId, model, peers: peers.ordered().length });
|
|
27
|
+
const peerErrors = [];
|
|
28
|
+
const pctx = { ...handlerCtx, peerErrors };
|
|
27
29
|
const win =
|
|
28
|
-
(await racePeerCandidates(peers.ordered(),
|
|
29
|
-
(await racePeerCandidates(peers.orderedByLastError(),
|
|
30
|
+
(await racePeerCandidates(peers.ordered(), pctx)) ||
|
|
31
|
+
(await racePeerCandidates(peers.orderedByLastError(), pctx)) ||
|
|
32
|
+
(await racePeerCandidates(peers.coolingByLastError(), pctx));
|
|
30
33
|
if (!win) {
|
|
31
34
|
evt("peer-race-lose", { reqId: handlerCtx.reqId, model });
|
|
32
|
-
|
|
35
|
+
// 组员全失败:把每个组员的真实返回汇总给调用者(否则只剩一个无信息的 429/502)
|
|
36
|
+
const detail = peerErrors.length
|
|
37
|
+
? peerErrors.map((e) => `${e.peer} -> ${e.status}${e.message ? ` (${String(e.message).slice(0, 160)})` : ""}`).join("; ")
|
|
38
|
+
: "no peers available";
|
|
39
|
+
const allLimit = peerErrors.length > 0 && peerErrors.every((e) => e.status === 429);
|
|
40
|
+
return { handled: false, lastErr: { model, upstream: null, status: allLimit ? 429 : 502, message: `peers failed: ${detail}` } };
|
|
33
41
|
}
|
|
34
42
|
evt("peer-race-win", { reqId: handlerCtx.reqId, model, winPeer: win.peer.url, winTarget: win.target, latencyMs: win.latencyMs });
|
|
35
43
|
await peers.recordResult(win.peer.url, { ok: true, latencyMs: win.latencyMs, model: win.target });
|
|
@@ -104,7 +104,7 @@ export async function handleViaRoute({
|
|
|
104
104
|
try { bodyText = await upRes.clone().text(); } catch {}
|
|
105
105
|
const msg = bodyText.slice(0, 300) || errMsg(upRes) || `peer ${status}`;
|
|
106
106
|
evt("via-route-peer-error", { reqId: handlerCtx.reqId, peer: peer.url, model, status, message: msg.slice(0, 200) });
|
|
107
|
-
try { await peers.recordError(peer.url); } catch {}
|
|
107
|
+
try { await peers.recordError(peer.url, { status }); } catch {}
|
|
108
108
|
try { await peers.recordResult(peer.url, { ok: false }); } catch {}
|
|
109
109
|
// 502/429 等可 fallback 到 direct
|
|
110
110
|
return { handled: false, lastErr: { model, upstream: upRes, status, message: msg } };
|
package/src/routes/hedge.js
CHANGED
|
@@ -136,7 +136,8 @@ export async function hedgedFirstChunkRace({
|
|
|
136
136
|
try {
|
|
137
137
|
peerWin =
|
|
138
138
|
(await racePeerCandidates(candidates, handlerCtx)) ||
|
|
139
|
-
(await racePeerCandidates(handlerCtx.peers.orderedByLastError(), handlerCtx))
|
|
139
|
+
(await racePeerCandidates(handlerCtx.peers.orderedByLastError(), handlerCtx)) ||
|
|
140
|
+
(await racePeerCandidates(handlerCtx.peers.coolingByLastError(), handlerCtx));
|
|
140
141
|
} catch (_) {
|
|
141
142
|
peerWin = null;
|
|
142
143
|
}
|
package/src/routes/peers.js
CHANGED
|
@@ -3,10 +3,53 @@ import { isAutoModel } from "../auto.js";
|
|
|
3
3
|
import { errMsg } from "./helpers.js";
|
|
4
4
|
import { runHook } from "../plugins.js";
|
|
5
5
|
import { buildShareKeysHeader, SHARE_KEYS_HEADER } from "../providers/share-keys.js";
|
|
6
|
-
import { compatFetch, timeoutSignal } from "../compat.js";
|
|
6
|
+
import { compatFetch, timeoutSignal, getUndici } from "../compat.js";
|
|
7
7
|
|
|
8
8
|
const PEER_TIMEOUT_MS = 30_000;
|
|
9
9
|
const PEER_STATUS_TIMEOUT_MS = 2_000;
|
|
10
|
+
const DEFAULT_PEER_CONNECT_TIMEOUT_MS = 3_000;
|
|
11
|
+
|
|
12
|
+
function peerConnectTimeoutMs() {
|
|
13
|
+
const n = Number(process.env.MSLXDFF_PEER_CONNECT_TIMEOUT_MS);
|
|
14
|
+
return Number.isInteger(n) && n > 0 ? n : DEFAULT_PEER_CONNECT_TIMEOUT_MS;
|
|
15
|
+
}
|
|
16
|
+
|
|
17
|
+
// 连接级超时(DNS + TCP/TLS 握手):黑洞节点(端口挂起)必须快速失败,
|
|
18
|
+
// 不能占用 30s 的响应超时(曾把一次请求拖到 92s)。仅 undici 可用时生效。
|
|
19
|
+
let peerDispatcher = null;
|
|
20
|
+
function getPeerDispatcher() {
|
|
21
|
+
if (peerDispatcher) return peerDispatcher;
|
|
22
|
+
const { Agent } = getUndici();
|
|
23
|
+
if (!Agent) return null;
|
|
24
|
+
try {
|
|
25
|
+
peerDispatcher = new Agent({
|
|
26
|
+
connect: { timeout: peerConnectTimeoutMs() },
|
|
27
|
+
headersTimeout: PEER_TIMEOUT_MS,
|
|
28
|
+
bodyTimeout: PEER_TIMEOUT_MS,
|
|
29
|
+
keepAliveTimeout: 30_000,
|
|
30
|
+
keepAliveMaxTimeout: 60_000,
|
|
31
|
+
});
|
|
32
|
+
} catch { peerDispatcher = null; }
|
|
33
|
+
return peerDispatcher;
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
// 跨 caller 中继必须剥离 reasoning 加密态:encrypted_content 由上游按 caller(出口)签发,
|
|
37
|
+
// 换个节点转发会被拒(400 "reasoning encrypted_content was not issued to this caller")。
|
|
38
|
+
// 明文 reasoning_content 保留(不绑定 caller,thinking 模式需要)。
|
|
39
|
+
export function stripCallerBoundReasoning(body) {
|
|
40
|
+
const messages = body?.messages;
|
|
41
|
+
if (!Array.isArray(messages)) return body;
|
|
42
|
+
let changed = false;
|
|
43
|
+
const out = messages.map((m) => {
|
|
44
|
+
if (!m || m.role !== "assistant") return m;
|
|
45
|
+
if (!Array.isArray(m.reasoning_items) || !m.reasoning_items.length) return m;
|
|
46
|
+
changed = true;
|
|
47
|
+
const copy = { ...m };
|
|
48
|
+
delete copy.reasoning_items;
|
|
49
|
+
return copy;
|
|
50
|
+
});
|
|
51
|
+
return changed ? { ...body, messages: out } : body;
|
|
52
|
+
}
|
|
10
53
|
|
|
11
54
|
function peerHealthTtlMs() {
|
|
12
55
|
const n = Number(process.env.MSLXDFF_PEER_HEALTH_TTL_MS);
|
|
@@ -75,11 +118,13 @@ async function forwardToPeer(peer, body, model, hops) {
|
|
|
75
118
|
// ADR-0008:该模型命中的供应商若开启 share → 附带瞬时 key 给组员借用(opencode 恒排除)
|
|
76
119
|
const shareHeader = buildShareKeysHeader(model);
|
|
77
120
|
if (shareHeader) headers[SHARE_KEYS_HEADER] = shareHeader;
|
|
121
|
+
const dispatcher = getPeerDispatcher();
|
|
78
122
|
return await compatFetch(`${peer.url}/v1/chat/completions`, {
|
|
79
123
|
method: "POST",
|
|
80
124
|
headers,
|
|
81
|
-
body: JSON.stringify({ ...body, model }),
|
|
125
|
+
body: JSON.stringify({ ...stripCallerBoundReasoning(body), model }),
|
|
82
126
|
signal: controller.signal,
|
|
127
|
+
...(dispatcher ? { dispatcher } : {}),
|
|
83
128
|
});
|
|
84
129
|
} catch (err) {
|
|
85
130
|
return err;
|
|
@@ -117,15 +162,30 @@ export const PEER_RACE_LIMIT = Number(process.env.MSLXDFF_PEER_RACE_LIMIT) > 0
|
|
|
117
162
|
? Number(process.env.MSLXDFF_PEER_RACE_LIMIT)
|
|
118
163
|
: 3;
|
|
119
164
|
|
|
165
|
+
// 串行记账队列:失败/迟到成功的记录不阻塞赢家返回,同时避免并发写盘互相覆盖。
|
|
166
|
+
let peerRecordChain = Promise.resolve();
|
|
167
|
+
function recordLater(fn) {
|
|
168
|
+
peerRecordChain = peerRecordChain.then(fn).catch(() => {});
|
|
169
|
+
}
|
|
170
|
+
|
|
120
171
|
export async function racePeerCandidates(candidates, ctx) {
|
|
121
|
-
|
|
122
|
-
|
|
172
|
+
const tried = (ctx.triedUrls ??= new Set());
|
|
173
|
+
const fresh = candidates.filter((p) => !tried.has(p.url));
|
|
174
|
+
for (let i = 0; i < fresh.length; i += PEER_RACE_LIMIT) {
|
|
175
|
+
const batch = fresh.slice(i, i + PEER_RACE_LIMIT);
|
|
123
176
|
const prepared = (await Promise.all(batch.map((peer) => resolvePeerTarget(ctx, peer)))).filter(Boolean);
|
|
124
177
|
if (!prepared.length) continue;
|
|
125
178
|
const completed = await new Promise((resolve) => {
|
|
126
179
|
const order = [];
|
|
127
180
|
const total = prepared.length;
|
|
181
|
+
let settled = false;
|
|
182
|
+
const finish = (winner) => {
|
|
183
|
+
if (settled) return;
|
|
184
|
+
settled = true;
|
|
185
|
+
resolve({ list: order, winner });
|
|
186
|
+
};
|
|
128
187
|
for (const { peer, target } of prepared) {
|
|
188
|
+
tried.add(peer.url);
|
|
129
189
|
ctx.evt("peer-request", { peer: peer.url, model: target, hops: ctx.hops + 1 });
|
|
130
190
|
// 插件 hook:peer:beforeForward — 转发给组员前观察
|
|
131
191
|
if (ctx.plugins?.length) {
|
|
@@ -142,33 +202,48 @@ export async function racePeerCandidates(candidates, ctx) {
|
|
|
142
202
|
}
|
|
143
203
|
if (failed) {
|
|
144
204
|
const status = res instanceof Error ? 502 : res.status;
|
|
145
|
-
|
|
205
|
+
const failRec = { peer: peer.url, status, message: res instanceof Error ? errMsg(res) : null };
|
|
206
|
+
if (Array.isArray(ctx.peerErrors)) ctx.peerErrors.push(failRec);
|
|
207
|
+
ctx.logError(ctx.model, status, res instanceof Error ? `peer ${peer.url} ${errMsg(res)}` : `peer ${peer.url} ${status}`);
|
|
146
208
|
ctx.evt("peer-error", { peer: peer.url, model: target, status, message: res instanceof Error ? errMsg(res) : null });
|
|
147
209
|
order.push({ ok: false, peer, target, res, status });
|
|
210
|
+
recordLater(() => ctx.peers.recordError(peer.url, { status }));
|
|
211
|
+
recordLater(() => ctx.peers.recordResult(peer.url, { ok: false }));
|
|
212
|
+
if (order.length === total) finish(null);
|
|
148
213
|
} else {
|
|
149
|
-
|
|
214
|
+
const entry = { ok: true, peer, target, res, latencyMs };
|
|
215
|
+
order.push(entry);
|
|
216
|
+
if (!settled) {
|
|
217
|
+
// 第一个成功立即返回:不等慢/黑洞候选(迟到者的记账由各分支自理)
|
|
218
|
+
finish(entry);
|
|
219
|
+
} else {
|
|
220
|
+
recordLater(() => ctx.peers.recordResult(peer.url, { ok: true, latencyMs, model: target }));
|
|
221
|
+
}
|
|
150
222
|
}
|
|
151
|
-
if (order.length === total) resolve(order);
|
|
152
223
|
});
|
|
153
224
|
}
|
|
154
225
|
});
|
|
155
|
-
const winner = completed.
|
|
226
|
+
const winner = completed.winner;
|
|
156
227
|
if (winner) {
|
|
157
|
-
for (const o of completed) {
|
|
158
|
-
if (o === winner) continue;
|
|
159
|
-
if (!o.ok) {
|
|
160
|
-
await ctx.peers.recordError(o.peer.url);
|
|
161
|
-
await ctx.peers.recordResult(o.peer.url, { ok: false });
|
|
162
|
-
} else {
|
|
163
|
-
await ctx.peers.recordResult(o.peer.url, { ok: true, latencyMs: o.latencyMs, model: o.target });
|
|
164
|
-
}
|
|
165
|
-
}
|
|
166
228
|
return { peer: winner.peer, target: winner.target, res: winner.res, latencyMs: winner.latencyMs };
|
|
167
229
|
}
|
|
168
|
-
|
|
169
|
-
|
|
170
|
-
|
|
171
|
-
|
|
230
|
+
// 全失败:补读失败响应体(诊断 + 调用者详情)。仅在"本轮无 winner"时执行,
|
|
231
|
+
// 不拖慢成功路径;并行读、每 peer 上限 600ms,body 里才有 400/429 的真实原因。
|
|
232
|
+
await Promise.all(completed.list.map(async (o) => {
|
|
233
|
+
if (o.ok || !o.res || typeof o.res !== "object" || o.res instanceof Error) return;
|
|
234
|
+
try {
|
|
235
|
+
const snip = String(await Promise.race([
|
|
236
|
+
o.res.clone().text(),
|
|
237
|
+
new Promise((r) => { const t = setTimeout(() => r(""), 600); t.unref?.(); }),
|
|
238
|
+
])).replace(/\s+/g, " ").slice(0, 300);
|
|
239
|
+
if (!snip) return;
|
|
240
|
+
if (Array.isArray(ctx.peerErrors)) {
|
|
241
|
+
const rec = ctx.peerErrors.find((x) => x.peer === o.peer.url && x.status === o.status && !x.message);
|
|
242
|
+
if (rec) rec.message = snip;
|
|
243
|
+
}
|
|
244
|
+
ctx.logError(ctx.model, o.status, `peer ${o.peer.url} ${o.status} body=${snip}`);
|
|
245
|
+
} catch {}
|
|
246
|
+
}));
|
|
172
247
|
}
|
|
173
248
|
return null;
|
|
174
249
|
}
|
|
@@ -11,7 +11,12 @@ export function setupAutoUpdate({ VERSION, bus, logs }) {
|
|
|
11
11
|
const line = `[auto-update] ${type} ${JSON.stringify(data)}`;
|
|
12
12
|
console.log(line);
|
|
13
13
|
}
|
|
14
|
-
|
|
14
|
+
// debug 会话不自动升级:debug 前台会把自己的 pid 写入 daemon.pid,
|
|
15
|
+
// auto-update 的 stopDaemon() 会把它自己停掉(现象:-debug 跑一会儿就"自己退出")
|
|
16
|
+
if (autoUpdateMs && process.env.MSLXDFF_DEBUG === "1") {
|
|
17
|
+
console.log(`auto-update: skipped (debug session)`);
|
|
18
|
+
emitAutoUpdate("auto-update-skipped", { intervalMs: autoUpdateMs, current: VERSION, reason: "debug" });
|
|
19
|
+
} else if (autoUpdateMs) {
|
|
15
20
|
console.log(`auto-update enabled: checking every ${Math.round(autoUpdateMs / 60000)}m`);
|
|
16
21
|
emitAutoUpdate("auto-update-enabled", { intervalMs: autoUpdateMs, current: VERSION });
|
|
17
22
|
setTimeout(() => {
|
|
@@ -11,7 +11,7 @@ import { logDir, appendEvent } from "../logs.js";
|
|
|
11
11
|
import { loadPlugins, runHook, resolvePluginDirs } from "../plugins.js";
|
|
12
12
|
import { createOpenCodeProvider } from "../providers/opencode.js";
|
|
13
13
|
import { loadProviderKeys, loadProviderAuths, loadProviderConfigs } from "../state.js";
|
|
14
|
-
import { refreshIntervalMs, modelCooldownMs, slowCooldownMs, peerCooldownMs, peerHeatMs, banWindowMs, banThreshold } from "../cli/policy.js";
|
|
14
|
+
import { refreshIntervalMs, modelCooldownMs, slowCooldownMs, peerCooldownMs, peerLimitCooldownMs, peerHeatMs, banWindowMs, banThreshold } from "../cli/policy.js";
|
|
15
15
|
import { errMsg } from "../cli/util.js";
|
|
16
16
|
|
|
17
17
|
/**
|
|
@@ -140,7 +140,7 @@ export async function setupProviders() {
|
|
|
140
140
|
}
|
|
141
141
|
},
|
|
142
142
|
});
|
|
143
|
-
const peers = createPeersService({ cooldownMs: peerCooldownMs(), heatMs: peerHeatMs() });
|
|
143
|
+
const peers = createPeersService({ cooldownMs: peerCooldownMs(), limitCooldownMs: peerLimitCooldownMs(), heatMs: peerHeatMs() });
|
|
144
144
|
const groups = createGroupsService({});
|
|
145
145
|
const bans = createBansService({ windowMs: banWindowMs(), threshold: banThreshold() });
|
|
146
146
|
|
|
@@ -37,8 +37,16 @@ export async function startServerLifecycle({ VERSION, token, created, upstream,
|
|
|
37
37
|
} catch {}
|
|
38
38
|
});
|
|
39
39
|
const { startDaemon: sd } = await import("../daemon.js");
|
|
40
|
-
const restore2 = () => {
|
|
40
|
+
const restore2 = async () => {
|
|
41
41
|
console.log("\n[debug] restoring background daemon...");
|
|
42
|
+
try { await srv.close(); } catch {} // 先释放端口,避免恢复的 daemon 抢端口互杀
|
|
43
|
+
if (process.platform === "linux") {
|
|
44
|
+
try {
|
|
45
|
+
const { execFile } = await import("node:child_process");
|
|
46
|
+
const ok = await new Promise((res) => execFile("systemctl", ["--user", "start", "mslxdff"], { windowsHide: true, timeout: 8000 }, (e) => res(!e)));
|
|
47
|
+
if (ok) { console.log("[debug] systemd user service restarted (mslxdff)"); setTimeout(() => process.exit(0), 300); return; }
|
|
48
|
+
} catch {}
|
|
49
|
+
}
|
|
42
50
|
try {
|
|
43
51
|
const restoredPid = sd([]);
|
|
44
52
|
console.log(`[debug] daemon restored (pid ${restoredPid})`);
|