mslxdff 0.1.121 → 0.1.122

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "mslxdff",
3
- "version": "0.1.121",
3
+ "version": "0.1.122",
4
4
  "description": "测试项目,请勿使用。",
5
5
  "type": "module",
6
6
  "bin": {
@@ -107,9 +107,18 @@ export async function runSerialTrial(ctx, deps = {}) {
107
107
  return json(res, 403, errBody);
108
108
  }
109
109
  if (auto) await auto.recordError(model, { status: upRes.status });
110
- lastErr = { model, upstream: upRes, status: upRes.status, message: null };
111
- logError(model, upRes.status, `upstream ${upRes.status}`);
112
- evt("upstream-error", { reqId, model, status: upRes.status, message: null, timing: upRes._t ?? null });
110
+ // 读失败响应体(clone 不影响后续 relay 转发原响应;1s 上限防流式错误体拖慢)
111
+ let upBody = "";
112
+ try {
113
+ upBody = String(await Promise.race([
114
+ upRes.clone().text(),
115
+ new Promise((r) => { const t = setTimeout(() => r(""), 1000); t.unref?.(); }),
116
+ ])).replace(/\s+/g, " ").slice(0, 400);
117
+ } catch {}
118
+ const upMsg = upBody || `upstream ${upRes.status}`;
119
+ lastErr = { model, upstream: upRes, status: upRes.status, message: upMsg };
120
+ logError(model, upRes.status, `upstream ${upRes.status}${upBody ? ` body=${upBody.slice(0, 300)}` : ""}`);
121
+ evt("upstream-error", { reqId, model, status: upRes.status, message: upMsg.slice(0, 300), timing: upRes._t ?? null });
113
122
  upRes = null;
114
123
  }
115
124
  if (upRes) {
@@ -138,6 +147,7 @@ export async function runSerialTrial(ctx, deps = {}) {
138
147
  } else {
139
148
  const pr = await peerRelay({ model, body, lastErr, requested, useAuto, lockModel, auto, peers, handlerCtx, evt, logCall, mark, perf0, stages, startedAt, plugins, res });
140
149
  if (pr.handled) return { done: true };
150
+ if (pr.lastErr) lastErr = pr.lastErr;
141
151
  }
142
152
  }
143
153
  if (groups) {
@@ -95,8 +95,29 @@ export async function handleDaemonFlag(args, VERSION) {
95
95
 
96
96
  export async function handleDebug(args) {
97
97
  if (!(args.includes("-debug") || args.includes("--debug"))) return false;
98
+ // systemd 自启协同:若后台 daemon 由 user service(Restart=always, RestartSec=3)托管,
99
+ // 仅 stopDaemon() 会让 systemd 3 秒后拉起新实例 → 抢 8989 → EADDRINUSE 自愈反杀本 debug 前台。
100
+ // 必须先 systemctl stop(主动停止不会被 Restart 拉起)。
101
+ if (process.platform === "linux") {
102
+ try {
103
+ const { execFile } = await import("node:child_process");
104
+ const active = await new Promise((res) => execFile("systemctl", ["--user", "is-active", "mslxdff"], { windowsHide: true, timeout: 4000 }, (e, so) => res(String(so || "").trim())));
105
+ if (active === "active") {
106
+ await new Promise((res) => execFile("systemctl", ["--user", "stop", "mslxdff"], { windowsHide: true, timeout: 6000 }, () => res()));
107
+ console.log("[debug] stopped systemd user service (mslxdff) — it will be restarted on exit");
108
+ }
109
+ } catch {}
110
+ }
98
111
  const { stopped, pid } = stopDaemon();
99
112
  if (stopped) console.log(`[debug] stopped background daemon (pid ${pid})`);
113
+ // 等旧 daemon 真正退出再抢端口(Windows 端口释放有延迟,否则 EADDRINUSE 会让 debug 立即崩)
114
+ if (stopped && pid) {
115
+ const t0 = Date.now();
116
+ while (isPidAlive(pid) && Date.now() - t0 < 4000) {
117
+ await new Promise((r) => setTimeout(r, 100));
118
+ }
119
+ await new Promise((r) => setTimeout(r, 150));
120
+ }
100
121
  try {
101
122
  const dir = logDir();
102
123
  const toClear = [eventsFile(), callsFile(), errorsFile(), logFile()];
@@ -101,6 +101,7 @@ export async function handleProviderAdd(id, sub, rest) {
101
101
  console.log(` share: ${loadProviderShareKeys(nid) ? "ON" : "off"} (mslxdff -provider ${nid} share on|off)`);
102
102
  console.log(` allowAny: OFF (secure, empty allowlist = 403 block before upstream) — enable via: mslxdff -provider ${nid} allowAny on`);
103
103
  console.log(` use as: ${nid}/<model-id> — restart daemon to activate`);
104
+ if (allowedModels.length) console.log(` opencode 中使用: mslxdff -setto opencode ${nid}/${allowedModels[0]} (同步进 opencode.json,会一并设为默认模型)`);
104
105
  console.log(` NOTE: empty allowlist = 403 before upstream, no cost — must set allowlist to use`);
105
106
  process.exit(0);
106
107
  }
package/src/cli/policy.js CHANGED
@@ -89,6 +89,10 @@ export function peerCooldownMs() {
89
89
  const n = Number(process.env.MSLXDFF_PEER_COOLDOWN_MS);
90
90
  return Number.isInteger(n) && n > 0 ? n : 30_000;
91
91
  }
92
+ export function peerLimitCooldownMs() {
93
+ const n = Number(process.env.MSLXDFF_PEER_LIMIT_COOLDOWN_MS);
94
+ return Number.isInteger(n) && n > 0 ? n : 5 * 60_000;
95
+ }
92
96
  export function peerHeatMs() {
93
97
  const n = Number(process.env.MSLXDFF_PEER_HEAT_MS);
94
98
  return Number.isInteger(n) && n > 0 ? n : 5 * 60_000;
package/src/peers.js CHANGED
@@ -1,6 +1,7 @@
1
1
  import { loadPeers, savePeers, loadPeerErrors, savePeerErrors, loadPeerStats, savePeerStats } from "./state.js";
2
2
 
3
3
  export const DEFAULT_PEER_COOLDOWN_MS = 30_000;
4
+ export const DEFAULT_PEER_LIMIT_COOLDOWN_MS = 5 * 60_000;
4
5
  export const DEFAULT_PEER_HEAT_MS = 5 * 60_000;
5
6
  export const DEFAULT_MAX_HOPS = 3;
6
7
  export const DEFAULT_BROADBAND_STALE_MS = 90_000;
@@ -28,6 +29,7 @@ export function createPeersService({
28
29
  file,
29
30
  now = () => Date.now(),
30
31
  cooldownMs = DEFAULT_PEER_COOLDOWN_MS,
32
+ limitCooldownMs = DEFAULT_PEER_LIMIT_COOLDOWN_MS,
31
33
  heatMs = DEFAULT_PEER_HEAT_MS,
32
34
  peers: seedPeers,
33
35
  errors: seedErrors,
@@ -75,10 +77,18 @@ export function createPeersService({
75
77
  return before - list.length;
76
78
  }
77
79
 
80
+ // 分级冷却:连续 429(streak >= 2)→ limitCooldownMs(限流恢复慢,5min 内不再考虑);
81
+ // 其他失败(网络抖动、5xx)→ cooldownMs(可能几秒恢复,给快速重试机会)。
82
+ function coolingWindowMs(url) {
83
+ const streak = stats[url]?.streak429 ?? 0;
84
+ return streak >= 2 ? limitCooldownMs : cooldownMs;
85
+ }
86
+
78
87
  function isCooling(url) {
79
- if (!cooldownMs) return false;
88
+ const win = coolingWindowMs(url);
89
+ if (!win) return false;
80
90
  const err = lastErrorAt[url];
81
- if (typeof err === "number" && now() - err < cooldownMs) return true;
91
+ if (typeof err === "number" && now() - err < win) return true;
82
92
  const peer = list.find((p) => p.url === url);
83
93
  if (peer && isBroadbandStale(peer, now(), broadbandStaleMs())) return true;
84
94
  return false;
@@ -148,6 +158,21 @@ export function createPeersService({
148
158
 
149
159
  let cursor = 0;
150
160
 
161
+ // 兜底重试集合:仅"因错误冷却中"的 peer(按最早失败优先)。
162
+ // 场景:唯一可用节点偶发失败被冷却锁死 → 主链零候选/全败时,兜底给它一次机会;
163
+ // 成功会被 recordResult 清除错误记录、回归热路径,避免"明明能用的节点被 30s 冷却全灭"。
164
+ function coolingByLastError() {
165
+ const t = now();
166
+ return list
167
+ .filter((p) => {
168
+ const err = lastErrorAt[p.url];
169
+ if (typeof err !== "number") return false;
170
+ if ((stats[p.url]?.streak429 ?? 0) >= 2) return false; // 429 长冷却:兜底也不考虑
171
+ return t - err < cooldownMs;
172
+ })
173
+ .sort((a, b) => (lastErrorAt[a.url] ?? 0) - (lastErrorAt[b.url] ?? 0));
174
+ }
175
+
151
176
  function next() {
152
177
  const avail = available();
153
178
  if (!avail.length) return null;
@@ -158,10 +183,16 @@ export function createPeersService({
158
183
  // Long-lived error memory: a peer keeps its last-error timestamp until a
159
184
  // subsequent success resets it (success clears the failure record) or the
160
185
  // error is no longer in the persist store on next load.
161
- async function recordError(url) {
186
+ async function recordError(url, { status } = {}) {
162
187
  if (!url) return;
163
188
  lastErrorAt[url] = now();
164
189
  await persistErrors({ ...lastErrorAt });
190
+ // 连续 429 计数(仅成功清零):第 2 次起进入长冷却——该 peer 大概率真被限流,不再频繁试它
191
+ if (Number(status) === 429) {
192
+ const prev = stats[url] || {};
193
+ stats[url] = { ...prev, streak429: (prev.streak429 || 0) + 1 };
194
+ await persistStats({ ...stats });
195
+ }
165
196
  }
166
197
 
167
198
  // Outcome of a forwarded request: ok updates the hot-cache (EMA latency,
@@ -182,6 +213,7 @@ export function createPeersService({
182
213
  : (typeof latencyMs === "number" ? latencyMs : prev.latencyMs ?? 0),
183
214
  fails: 0,
184
215
  model: model || prev.model || "",
216
+ streak429: 0,
185
217
  };
186
218
  } else {
187
219
  const prev = stats[url] || {};
@@ -194,7 +226,7 @@ export function createPeersService({
194
226
  all, add, remove, removeByGroup, isCooling, isBroadbandCooling, isBroadbandStale: (url) => {
195
227
  const peer = list.find((p) => p.url === url);
196
228
  return peer ? isBroadbandStale(peer, now(), broadbandStaleMs()) : false;
197
- }, isHot, stat, ordered, orderedByLastError, available, next,
229
+ }, isHot, stat, ordered, orderedByLastError, coolingByLastError, available, next,
198
230
  recordError, recordResult, errors: () => ({ ...lastErrorAt }), stats: () => ({ ...stats }),
199
231
  };
200
232
  }
@@ -24,12 +24,20 @@ export async function handlePeerRelay({
24
24
  res,
25
25
  }) {
26
26
  evt("peer-race-start", { reqId: handlerCtx.reqId, model, peers: peers.ordered().length });
27
+ const peerErrors = [];
28
+ const pctx = { ...handlerCtx, peerErrors };
27
29
  const win =
28
- (await racePeerCandidates(peers.ordered(), handlerCtx)) ||
29
- (await racePeerCandidates(peers.orderedByLastError(), handlerCtx));
30
+ (await racePeerCandidates(peers.ordered(), pctx)) ||
31
+ (await racePeerCandidates(peers.orderedByLastError(), pctx)) ||
32
+ (await racePeerCandidates(peers.coolingByLastError(), pctx));
30
33
  if (!win) {
31
34
  evt("peer-race-lose", { reqId: handlerCtx.reqId, model });
32
- return { handled: false };
35
+ // 组员全失败:把每个组员的真实返回汇总给调用者(否则只剩一个无信息的 429/502)
36
+ const detail = peerErrors.length
37
+ ? peerErrors.map((e) => `${e.peer} -> ${e.status}${e.message ? ` (${String(e.message).slice(0, 160)})` : ""}`).join("; ")
38
+ : "no peers available";
39
+ const allLimit = peerErrors.length > 0 && peerErrors.every((e) => e.status === 429);
40
+ return { handled: false, lastErr: { model, upstream: null, status: allLimit ? 429 : 502, message: `peers failed: ${detail}` } };
33
41
  }
34
42
  evt("peer-race-win", { reqId: handlerCtx.reqId, model, winPeer: win.peer.url, winTarget: win.target, latencyMs: win.latencyMs });
35
43
  await peers.recordResult(win.peer.url, { ok: true, latencyMs: win.latencyMs, model: win.target });
@@ -104,7 +104,7 @@ export async function handleViaRoute({
104
104
  try { bodyText = await upRes.clone().text(); } catch {}
105
105
  const msg = bodyText.slice(0, 300) || errMsg(upRes) || `peer ${status}`;
106
106
  evt("via-route-peer-error", { reqId: handlerCtx.reqId, peer: peer.url, model, status, message: msg.slice(0, 200) });
107
- try { await peers.recordError(peer.url); } catch {}
107
+ try { await peers.recordError(peer.url, { status }); } catch {}
108
108
  try { await peers.recordResult(peer.url, { ok: false }); } catch {}
109
109
  // 502/429 等可 fallback 到 direct
110
110
  return { handled: false, lastErr: { model, upstream: upRes, status, message: msg } };
@@ -136,7 +136,8 @@ export async function hedgedFirstChunkRace({
136
136
  try {
137
137
  peerWin =
138
138
  (await racePeerCandidates(candidates, handlerCtx)) ||
139
- (await racePeerCandidates(handlerCtx.peers.orderedByLastError(), handlerCtx));
139
+ (await racePeerCandidates(handlerCtx.peers.orderedByLastError(), handlerCtx)) ||
140
+ (await racePeerCandidates(handlerCtx.peers.coolingByLastError(), handlerCtx));
140
141
  } catch (_) {
141
142
  peerWin = null;
142
143
  }
@@ -142,7 +142,9 @@ export async function racePeerCandidates(candidates, ctx) {
142
142
  }
143
143
  if (failed) {
144
144
  const status = res instanceof Error ? 502 : res.status;
145
- ctx.logError(ctx.model, status, res instanceof Error ? errMsg(res) : `peer ${status}`);
145
+ const failRec = { peer: peer.url, status, message: res instanceof Error ? errMsg(res) : null };
146
+ if (Array.isArray(ctx.peerErrors)) ctx.peerErrors.push(failRec);
147
+ ctx.logError(ctx.model, status, res instanceof Error ? `peer ${peer.url} ${errMsg(res)}` : `peer ${peer.url} ${status}`);
146
148
  ctx.evt("peer-error", { peer: peer.url, model: target, status, message: res instanceof Error ? errMsg(res) : null });
147
149
  order.push({ ok: false, peer, target, res, status });
148
150
  } else {
@@ -157,7 +159,7 @@ export async function racePeerCandidates(candidates, ctx) {
157
159
  for (const o of completed) {
158
160
  if (o === winner) continue;
159
161
  if (!o.ok) {
160
- await ctx.peers.recordError(o.peer.url);
162
+ await ctx.peers.recordError(o.peer.url, { status: o.status });
161
163
  await ctx.peers.recordResult(o.peer.url, { ok: false });
162
164
  } else {
163
165
  await ctx.peers.recordResult(o.peer.url, { ok: true, latencyMs: o.latencyMs, model: o.target });
@@ -165,8 +167,25 @@ export async function racePeerCandidates(candidates, ctx) {
165
167
  }
166
168
  return { peer: winner.peer, target: winner.target, res: winner.res, latencyMs: winner.latencyMs };
167
169
  }
170
+ // 全失败:补读失败响应体(诊断 + 调用者详情)。仅在"本轮无 winner"时执行,
171
+ // 不拖慢成功路径;并行读、每 peer 上限 600ms,body 里才有 400/429 的真实原因。
172
+ await Promise.all(completed.map(async (o) => {
173
+ if (o.ok || !o.res || typeof o.res !== "object" || o.res instanceof Error) return;
174
+ try {
175
+ const snip = String(await Promise.race([
176
+ o.res.clone().text(),
177
+ new Promise((r) => { const t = setTimeout(() => r(""), 600); t.unref?.(); }),
178
+ ])).replace(/\s+/g, " ").slice(0, 300);
179
+ if (!snip) return;
180
+ if (Array.isArray(ctx.peerErrors)) {
181
+ const rec = ctx.peerErrors.find((x) => x.peer === o.peer.url && x.status === o.status && !x.message);
182
+ if (rec) rec.message = snip;
183
+ }
184
+ ctx.logError(ctx.model, o.status, `peer ${o.peer.url} ${o.status} body=${snip}`);
185
+ } catch {}
186
+ }));
168
187
  for (const o of completed) {
169
- await ctx.peers.recordError(o.peer.url);
188
+ await ctx.peers.recordError(o.peer.url, { status: o.status });
170
189
  await ctx.peers.recordResult(o.peer.url, { ok: false });
171
190
  }
172
191
  }
@@ -11,7 +11,12 @@ export function setupAutoUpdate({ VERSION, bus, logs }) {
11
11
  const line = `[auto-update] ${type} ${JSON.stringify(data)}`;
12
12
  console.log(line);
13
13
  }
14
- if (autoUpdateMs) {
14
+ // debug 会话不自动升级:debug 前台会把自己的 pid 写入 daemon.pid,
15
+ // auto-update 的 stopDaemon() 会把它自己停掉(现象:-debug 跑一会儿就"自己退出")
16
+ if (autoUpdateMs && process.env.MSLXDFF_DEBUG === "1") {
17
+ console.log(`auto-update: skipped (debug session)`);
18
+ emitAutoUpdate("auto-update-skipped", { intervalMs: autoUpdateMs, current: VERSION, reason: "debug" });
19
+ } else if (autoUpdateMs) {
15
20
  console.log(`auto-update enabled: checking every ${Math.round(autoUpdateMs / 60000)}m`);
16
21
  emitAutoUpdate("auto-update-enabled", { intervalMs: autoUpdateMs, current: VERSION });
17
22
  setTimeout(() => {
@@ -11,7 +11,7 @@ import { logDir, appendEvent } from "../logs.js";
11
11
  import { loadPlugins, runHook, resolvePluginDirs } from "../plugins.js";
12
12
  import { createOpenCodeProvider } from "../providers/opencode.js";
13
13
  import { loadProviderKeys, loadProviderAuths, loadProviderConfigs } from "../state.js";
14
- import { refreshIntervalMs, modelCooldownMs, slowCooldownMs, peerCooldownMs, peerHeatMs, banWindowMs, banThreshold } from "../cli/policy.js";
14
+ import { refreshIntervalMs, modelCooldownMs, slowCooldownMs, peerCooldownMs, peerLimitCooldownMs, peerHeatMs, banWindowMs, banThreshold } from "../cli/policy.js";
15
15
  import { errMsg } from "../cli/util.js";
16
16
 
17
17
  /**
@@ -140,7 +140,7 @@ export async function setupProviders() {
140
140
  }
141
141
  },
142
142
  });
143
- const peers = createPeersService({ cooldownMs: peerCooldownMs(), heatMs: peerHeatMs() });
143
+ const peers = createPeersService({ cooldownMs: peerCooldownMs(), limitCooldownMs: peerLimitCooldownMs(), heatMs: peerHeatMs() });
144
144
  const groups = createGroupsService({});
145
145
  const bans = createBansService({ windowMs: banWindowMs(), threshold: banThreshold() });
146
146
 
@@ -37,8 +37,16 @@ export async function startServerLifecycle({ VERSION, token, created, upstream,
37
37
  } catch {}
38
38
  });
39
39
  const { startDaemon: sd } = await import("../daemon.js");
40
- const restore2 = () => {
40
+ const restore2 = async () => {
41
41
  console.log("\n[debug] restoring background daemon...");
42
+ try { await srv.close(); } catch {} // 先释放端口,避免恢复的 daemon 抢端口互杀
43
+ if (process.platform === "linux") {
44
+ try {
45
+ const { execFile } = await import("node:child_process");
46
+ const ok = await new Promise((res) => execFile("systemctl", ["--user", "start", "mslxdff"], { windowsHide: true, timeout: 8000 }, (e) => res(!e)));
47
+ if (ok) { console.log("[debug] systemd user service restarted (mslxdff)"); setTimeout(() => process.exit(0), 300); return; }
48
+ } catch {}
49
+ }
42
50
  try {
43
51
  const restoredPid = sd([]);
44
52
  console.log(`[debug] daemon restored (pid ${restoredPid})`);