mslxdff 0.1.65 → 0.1.67

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (65) hide show
  1. package/bin/mslxdff.js +3 -3159
  2. package/package.json +1 -1
  3. package/src/chat/cooling.js +99 -0
  4. package/src/chat/direct.js +57 -0
  5. package/src/chat/gateway.js +148 -0
  6. package/src/chat/orchestrator.js +218 -0
  7. package/src/chat/sse.js +69 -0
  8. package/src/chat/upstream.js +59 -501
  9. package/src/cli/bootstrap.js +1 -0
  10. package/src/cli/commands/daemon.js +151 -0
  11. package/src/cli/commands/group.js +272 -0
  12. package/src/cli/commands/model.js +405 -0
  13. package/src/cli/commands/provider/add.js +106 -0
  14. package/src/cli/commands/provider/allowlist.js +99 -0
  15. package/src/cli/commands/provider/config.js +82 -0
  16. package/src/cli/commands/provider/index.js +193 -0
  17. package/src/cli/commands/provider/keys.js +135 -0
  18. package/src/cli/commands/provider/models.js +99 -0
  19. package/src/cli/commands/provider.js +1 -0
  20. package/src/cli/commands/sync.js +143 -0
  21. package/src/cli/commands/system.js +236 -0
  22. package/src/cli/commands/workbuddy.js +91 -0
  23. package/src/cli/format.js +119 -0
  24. package/src/cli/group-helpers.js +69 -0
  25. package/src/cli/help.js +66 -0
  26. package/src/cli/index.js +59 -0
  27. package/src/cli/interactive.js +84 -0
  28. package/src/cli/policy.js +119 -0
  29. package/src/cli/provider-row.js +73 -0
  30. package/src/cli/status.js +283 -0
  31. package/src/cli/util.js +24 -0
  32. package/src/providers/base.js +159 -0
  33. package/src/providers/dispatcher.js +18 -6
  34. package/src/providers/generic.js +28 -166
  35. package/src/providers/openrouter.js +19 -205
  36. package/src/providers/workbuddy/auth.js +175 -0
  37. package/src/providers/workbuddy/balance.js +84 -0
  38. package/src/providers/workbuddy/chat.js +310 -0
  39. package/src/providers/workbuddy/index.js +263 -0
  40. package/src/providers/workbuddy/models.js +111 -0
  41. package/src/providers/workbuddy/rotation-log.js +54 -0
  42. package/src/providers/workbuddy.js +2 -677
  43. package/src/routes/chat/broadband-handler.js +25 -46
  44. package/src/routes/chat/exhausted-handler.js +2 -2
  45. package/src/routes/chat/gateway.js +301 -0
  46. package/src/routes/chat/hedge-handler.js +65 -83
  47. package/src/routes/chat/index.js +1 -384
  48. package/src/routes/chat/local-handler.js +32 -61
  49. package/src/routes/chat/peer-handler.js +32 -23
  50. package/src/routes/chat/relay-pipeline.js +151 -0
  51. package/src/runtime/bootstrap.js +408 -0
  52. package/src/state/facade.js +57 -0
  53. package/src/state/memory.js +161 -0
  54. package/src/state/merge.js +26 -0
  55. package/src/state/persist.js +42 -0
  56. package/src/state/provider-config.js +143 -0
  57. package/src/state/schemas/allowlist.js +87 -0
  58. package/src/state/schemas/group.js +31 -0
  59. package/src/state/schemas/model.js +53 -0
  60. package/src/state/schemas/peer.js +31 -0
  61. package/src/state/schemas/port.js +10 -0
  62. package/src/state/schemas/provider.js +204 -0
  63. package/src/state/schemas/token.js +43 -0
  64. package/src/state/store.js +176 -0
  65. package/src/state.js +1 -711
@@ -1,8 +1,9 @@
1
+ import { relay, SLOW_TOTAL_MS, STREAM_TIMEOUT_MS, STALL_TIMEOUT_MS, SCORE_STALL_MS } from "../stream.js";
1
2
  import { buildFallbackInfo } from "../fallback.js";
2
- import { relay, SLOW_TOTAL_MS } from "../stream.js";
3
+ import { createRelayPipeline } from "./relay-pipeline.js";
3
4
  import { tryBroadbandRelay } from "../relay-queue.js";
4
- import { runHook } from "../../plugins.js";
5
5
 
6
+ /** 薄适配:broadband 两种形态合一经由 pipeline */
6
7
  export async function handleBroadbandRelay({
7
8
  model,
8
9
  body,
@@ -30,57 +31,35 @@ export async function handleBroadbandRelay({
30
31
  evt("relay-miss", { reqId: handlerCtx.reqId, model });
31
32
  return { handled: false };
32
33
  }
34
+ const pipeline = createRelayPipeline({
35
+ relay,
36
+ buildFallbackInfo,
37
+ auto,
38
+ plugins,
39
+ evt,
40
+ mark,
41
+ logCall: () => {},
42
+ logError: () => {},
43
+ constants: { STREAM_TIMEOUT_MS, SLOW_TOTAL_MS, STALL_TIMEOUT_MS, SCORE_STALL_MS },
44
+ startedAt,
45
+ stages,
46
+ });
33
47
  const isResponse = bb.result && typeof bb.result.status === "number" && typeof bb.result.headers?.get === "function";
34
48
  if (isResponse) {
35
- const bbFallback = buildFallbackInfo({ requested, actual: model, lastErr, via: "broadband", useAuto, lockModel });
36
- if (bbFallback?.fallback) evt("fallback-notice", { reqId: handlerCtx.reqId, requested, actual: model, reason: bbFallback.reason, notice: bbFallback.notice, via: "broadband" });
37
- evt("relay-start", { reqId: handlerCtx.reqId, model, via: "broadband", target: bb.target, group: bb.group, fallback: bbFallback });
38
- const out = await relay(res, bb.result, body, {
39
- fallback: bbFallback,
40
- onFirstChunk: (d) => mark(`ttf-bb-${model}`),
41
- onDownstreamAbort: () => evt("client-abort", { reqId: handlerCtx.reqId, model, totalMs: Math.round(Date.now() - startedAt), stages: [...stages] }),
42
- });
43
- evt("relay-done", { reqId: handlerCtx.reqId, model, via: "broadband", status: out.status, ttfMs: out.ttfMs, totalMs: out.totalMs, aborted: out.aborted, interrupted: out.interrupted ?? false, detail: out.detail ?? null });
44
- if (auto && out.status === 200) {
45
- const latencyMs = out.totalMs ?? 0;
46
- if (out.detail?.stallHits > 0 || (latencyMs && latencyMs > SLOW_TOTAL_MS)) {
47
- void auto.recordError(model, { status: 200, slow: true, note: `broadband slow ${latencyMs}ms` });
48
- void auto.recordLatency(model, latencyMs);
49
- } else {
50
- await auto.recordOk(model, { latencyMs });
51
- }
52
- }
53
- evt("result", { model, status: out.status, via: "broadband", timing: bb.result._t ?? null, ttfMs: out.ttfMs, totalMs: out.totalMs, detail: out.detail ?? null, fallback: bbFallback, requested, actual: model });
54
- evt("client-response", { requested, actual: model, via: "broadband", fallback: bbFallback, status: out.status, reqId: handlerCtx.reqId });
55
- if (plugins?.length) runHook(plugins, "request:completed", { reqId: handlerCtx.reqId, requested, useAuto, hops: handlerCtx.hops, stream: Boolean(body.stream), durationMs: Date.now() - startedAt, via: "broadband", status: out.status, actual: model, fallback: bbFallback }).catch(() => {});
49
+ await pipeline.execute({ res, upRes: bb.result, body, requested, actual: model, lastErr, via: "broadband", lockModel, useAuto, handlerCtx, mark, perf0, stages, startedAt });
56
50
  return { handled: true };
57
- } else if (bb.result && typeof bb.result.status === "number") {
51
+ }
52
+ if (bb.result && typeof bb.result.status === "number") {
53
+ const b = bb.result.body || "";
54
+ const str = typeof b === "string" ? b : JSON.stringify(b);
55
+ const isSSE = bb.result.headers?.["Content-Type"]?.includes("text/event-stream");
58
56
  const fakeRes = {
59
57
  status: bb.result.status,
60
58
  headers: { get: (k) => bb.result.headers?.[k] || bb.result.headers?.[k.toLowerCase()] || null },
61
- text: async () => typeof bb.result.body === "string" ? bb.result.body : JSON.stringify(bb.result.body),
62
- body: (() => {
63
- const b = bb.result.body || "";
64
- const str = typeof b === "string" ? b : JSON.stringify(b);
65
- const isSSE = bb.result.headers?.["Content-Type"]?.includes("text/event-stream");
66
- if (isSSE) {
67
- return (async function* () { yield Buffer.from(str); })();
68
- }
69
- return null;
70
- })(),
59
+ text: async () => str,
60
+ body: isSSE ? (async function* () { yield Buffer.from(str); })() : null,
71
61
  };
72
- const bbLocalFallback = buildFallbackInfo({ requested, actual: model, lastErr, via: "broadband", useAuto, lockModel });
73
- if (bbLocalFallback?.fallback) evt("fallback-notice", { reqId: handlerCtx.reqId, requested, actual: model, reason: bbLocalFallback.reason, notice: bbLocalFallback.notice, via: "broadband" });
74
- evt("relay-start", { reqId: handlerCtx.reqId, model, via: "broadband-local", target: bb.target, group: bb.group, fallback: bbLocalFallback });
75
- const out = await relay(res, fakeRes, body, {
76
- fallback: bbLocalFallback,
77
- onFirstChunk: (d) => mark(`ttf-bb-${model}`),
78
- onDownstreamAbort: () => evt("client-abort", { reqId: handlerCtx.reqId, model, totalMs: Math.round(Date.now() - startedAt), stages: [...stages] }),
79
- });
80
- evt("relay-done", { reqId: handlerCtx.reqId, model, via: "broadband-local", status: out.status, ttfMs: out.ttfMs, totalMs: out.totalMs, aborted: out.aborted, interrupted: out.interrupted ?? false, detail: out.detail ?? null });
81
- evt("result", { model, status: out.status, via: "broadband", timing: null, ttfMs: out.ttfMs, totalMs: out.totalMs, detail: out.detail ?? null, fallback: bbLocalFallback, requested, actual: model });
82
- evt("client-response", { requested, actual: model, via: "broadband", fallback: bbLocalFallback, status: out.status, reqId: handlerCtx.reqId });
83
- if (plugins?.length) runHook(plugins, "request:completed", { reqId: handlerCtx.reqId, requested, useAuto, hops: handlerCtx.hops, stream: Boolean(body.stream), durationMs: Date.now() - startedAt, via: "broadband", status: out.status, actual: model, fallback: bbLocalFallback }).catch(() => {});
62
+ await pipeline.execute({ res, upRes: fakeRes, body, requested, actual: model, lastErr, via: "broadband-local", lockModel, useAuto, handlerCtx, mark, perf0, stages, startedAt });
84
63
  return { handled: true };
85
64
  }
86
65
  evt("relay-miss", { reqId: handlerCtx.reqId, model });
@@ -1,6 +1,6 @@
1
- import { performance } from "node:perf_hooks";
2
- import { json } from "../helpers.js";
3
1
  import { relay } from "../stream.js";
2
+ import { json } from "../helpers.js";
3
+ import { performance } from "node:perf_hooks";
4
4
 
5
5
  export async function handleExhaustedLocal({ res, body, lastErr, order, handlerCtx, evt, logCall, mark, perf0, stages, done, requested, useAuto }) {
6
6
  const model = lastErr?.model ?? handlerCtx.model;
@@ -0,0 +1,301 @@
1
+ import { performance } from "node:perf_hooks";
2
+ import { injectReasoningContent, normalizeModel } from "../../reasoning.js";
3
+ import { isAutoModel } from "../../auto.js";
4
+ import { toInternalId as aliasToInternal } from "../../sync-opencode.js";
5
+ import { clientIp, json, readBody, parseHops, summarizePrompt, errMsg } from "../helpers.js";
6
+ import { hedgeDelayMs, shouldHedge } from "../hedge.js";
7
+ import { runHook } from "../../plugins.js";
8
+ import { parseShareKeysHeader, SHARE_KEYS_HEADER } from "../../providers/share-keys.js";
9
+ import { handleHedge } from "./hedge-handler.js";
10
+ import { handleLocalRelay } from "./local-handler.js";
11
+ import { handlePeerRelay } from "./peer-handler.js";
12
+ import { handleBroadbandRelay } from "./broadband-handler.js";
13
+ import { handleExhaustedLocal, handleExhaustedAll } from "./exhausted-handler.js";
14
+ import { normalizeFullId, getModelAlias } from "../../providers/model-id.js";
15
+
16
+ /**
17
+ * ChatGateway 深模块:对外 1 handle,内部 Policy→Selector→Executor 三段编排
18
+ * Policy: 别名/allowlist/header 透传
19
+ * Selector: order 推导 + 并发择优 + 排序
20
+ * Executor: 串行 trial → hedge → local → peer → broadband → exhausted
21
+ * 两 adapter:Provider(upstream.chat) + Clock/Latency(auto) 可注入 fake
22
+ */
23
+ export function createChatGateway({ upstream, auto, logs, peers, maxHops, groups, bus, token, plugins }) {
24
+ async function handle({ req, res }) {
25
+ let body;
26
+ try {
27
+ body = await readBody(req);
28
+ } catch {
29
+ return json(res, 400, { error: "Invalid JSON body" });
30
+ }
31
+ if (plugins?.length) {
32
+ const rc = await runHook(plugins, "request:received", { ip: clientIp(req), hops: parseHops(req.headers["x-mslxdff-hops"]), headers: { "content-type": req.headers["content-type"] }, body });
33
+ for (const e of rc.errors) logs?.appendEvent?.({ ts: Date.now(), type: "plugin-hook-error", hook: "request:received", plugin: e.plugin, error: e.error });
34
+ const respond = rc.value?.respond;
35
+ if (respond && typeof respond === "object") return json(res, respond.status || 200, respond.body ?? {});
36
+ }
37
+
38
+ const startedAt = Date.now();
39
+ const reqId = `${Date.now().toString(36)}-${Math.random().toString(36).slice(2, 6)}`;
40
+ const perf0 = performance.now();
41
+ const stages = [];
42
+ const mark = (name) => stages.push([name, Math.round(performance.now() - perf0)]);
43
+
44
+ // ===== Policy =====
45
+ const hops = parseHops(req.headers["x-mslxdff-hops"]);
46
+ const shareKeys = parseShareKeysHeader(req.headers[SHARE_KEYS_HEADER] || "");
47
+ const workbuddyUid = (req.headers["x-mslxdff-workbuddy-uid"] || req.headers["x-workbuddy-uid"] || "").toString().trim();
48
+ const lockModel = req.headers["x-mslxdff-model-lock"] || "";
49
+ const rawModel = body.model || "";
50
+ let normalizedRequested = normalizeModel(lockModel || rawModel || "");
51
+ const aliasResolved = getModelAlias(normalizedRequested);
52
+ if (aliasResolved) { normalizedRequested = aliasResolved; body = { ...body, model: aliasResolved }; }
53
+ let requested = normalizedRequested;
54
+ let aliasInfo = null;
55
+ if (requested.startsWith("mslxdff-")) {
56
+ const internal = aliasToInternal(requested);
57
+ if (internal) { aliasInfo = `${requested} -> ${internal}`; requested = internal; }
58
+ } else if (requested.includes("/")) {
59
+ const slashIdx = requested.indexOf("/");
60
+ const rawPart = requested.slice(slashIdx + 1);
61
+ const providerPart = requested.slice(0, slashIdx);
62
+ if (rawPart.startsWith("mslxdff-")) {
63
+ const internal = aliasToInternal(rawPart);
64
+ if (internal) {
65
+ aliasInfo = `${requested} -> ${providerPart}/${internal} (alias stripped)`;
66
+ requested = `${providerPart}/${internal}`;
67
+ if (providerPart === "mslxdff") { requested = internal; aliasInfo = `${rawModel} -> ${internal} (mslxdff alias stripped)`; }
68
+ }
69
+ } else if (providerPart === "mslxdff") {
70
+ aliasInfo = `${requested} -> ${rawPart} (mslxdff provider stripped, 原名兼容)`;
71
+ requested = rawPart;
72
+ }
73
+ }
74
+ const useAuto = isAutoModel(requested);
75
+ mark("parsed");
76
+ if (aliasInfo) { try { res.setHeader("x-mslxdff-alias", aliasInfo); } catch {} }
77
+
78
+ // ===== Selector: order 推导 =====
79
+ let order;
80
+ if (lockModel) order = [requested];
81
+ else if (useAuto) order = auto ? await auto.candidates() : [""];
82
+ else order = auto ? await auto.candidatesFor(requested) : [requested];
83
+ if (!order.length) order = [""];
84
+ const canFallback = order.length > 1;
85
+ const canForwardPeers = Boolean(peers) && hops < maxHops;
86
+ mark("ordered");
87
+
88
+ const logCall = (model, status) => logs?.appendCall({ reqId, model, auto: useAuto, status, durationMs: Date.now() - startedAt, stream: Boolean(body.stream), stages });
89
+ const logError = (model, status, message) => logs?.appendError({ reqId, model, auto: useAuto, status, message, stages });
90
+ const evt = (type, data) => {
91
+ const entry = { ts: Date.now(), reqId, type, ...data, model: data.model ?? requested, auto: useAuto, durationMs: Date.now() - startedAt, stages: [...stages] };
92
+ if (bus) bus.emit(entry);
93
+ logs?.appendEvent?.(entry);
94
+ };
95
+ const done = (info) => {
96
+ if (!plugins?.length) return;
97
+ runHook(plugins, "request:completed", { reqId, requested, useAuto, hops, stream: Boolean(body.stream), durationMs: Date.now() - startedAt, ...info }).catch(() => {});
98
+ };
99
+ evt("request", { reqId, hops, ip: clientIp(req), stream: Boolean(body.stream), prompt: summarizePrompt(body), rawModel, requested, lockModel: lockModel || null });
100
+ if (aliasInfo) evt("alias", { reqId, alias: aliasInfo, rawModel, requested });
101
+ if (Object.keys(shareKeys).length) evt("share-keys", { reqId, providers: Object.keys(shareKeys) });
102
+ evt("ordered", { reqId, order, canFallback, canForwardPeers, useAuto, statuses: auto?.statuses?.() ?? null });
103
+
104
+ if (plugins?.length && !lockModel) {
105
+ const sel = await runHook(plugins, "model:select", { reqId, requested, useAuto, order: [...order], hops, stream: Boolean(body.stream) });
106
+ if (sel.changed && Array.isArray(sel.value) && sel.value.length) {
107
+ order = sel.value.filter(Boolean);
108
+ if (!order.length) order = [requested];
109
+ evt("plugin-hook", { reqId, hook: "model:select", applied: true, order: [...order] });
110
+ }
111
+ for (const e of sel.errors) evt("plugin-hook-error", { reqId, hook: "model:select", plugin: e.plugin, error: e.error });
112
+ }
113
+
114
+ const handlerCtx = { reqId, model: null, body, hops, peers, plugins, evt, logError, logCall, logs };
115
+
116
+ // ===== Selector: 首次 auto 并发择优 =====
117
+ if (useAuto && order.length > 1 && auto && !lockModel) {
118
+ const statuses = auto.statuses?.() ?? {};
119
+ const hasPriorSuccess = Object.values(statuses).some((e) => e && typeof e === "object" && e.status === "normal");
120
+ const nonCoolingOrder = order.filter((m) => { try { return !auto.isCooling(m); } catch { return true; } });
121
+ if (!hasPriorSuccess && nonCoolingOrder.length > 1) {
122
+ const concLimit = (() => {
123
+ const v = Number(process.env.MSLXDFF_AUTO_CONCURRENT);
124
+ if (Number.isInteger(v) && v > 0) return Math.min(v, nonCoolingOrder.length);
125
+ return Math.min(nonCoolingOrder.length, 5);
126
+ })();
127
+ const raceModels = nonCoolingOrder.slice(0, concLimit);
128
+ evt("auto-concurrent-race", { reqId, models: raceModels, skippedFaulty: order.length - nonCoolingOrder.length, limit: concLimit });
129
+ const raceStart = performance.now();
130
+ const attempts = raceModels.map(async (m) => {
131
+ const fwd = { ...injectReasoningContent(m, body), model: m };
132
+ let r = null;
133
+ try {
134
+ const chatOpts = {};
135
+ if (Object.keys(shareKeys).length) chatOpts.shareKeys = shareKeys;
136
+ if (workbuddyUid) chatOpts.workbuddyUid = workbuddyUid;
137
+ r = await upstream.chat(fwd, Object.keys(chatOpts).length ? chatOpts : undefined);
138
+ } catch (err) {
139
+ return { model: m, ok: false, error: errMsg(err), status: 502, timing: err?._t ?? null };
140
+ }
141
+ if (r && r.status >= 400) {
142
+ const isAllow = r.status === 403 && r.headers?.get?.("x-mslxdff-allowlist") === "1";
143
+ if (isAllow) return { model: m, ok: false, error: "allowlist", status: 403, allowlist: true };
144
+ return { model: m, ok: false, error: `upstream ${r.status}`, status: r.status, res: r, timing: r._t ?? null };
145
+ }
146
+ if (r instanceof Error) return { model: m, ok: false, error: errMsg(r), status: 502 };
147
+ return { model: m, ok: true, res: r, status: r.status, timing: r._t ?? null };
148
+ });
149
+ const results = await Promise.allSettled(attempts);
150
+ const okList = results.map((r, i) => ({ r, i, model: raceModels[i] }))
151
+ .filter(({ r }) => r.status === "fulfilled" && r.value?.ok)
152
+ .map(({ r, i, model }) => ({ model, idx: i, val: r.value, t: r.value.timing?.totalMs ?? r.value.timing?.ms ?? Number.MAX_SAFE_INTEGER }));
153
+ if (okList.length) {
154
+ okList.sort((a, b) => a.t - b.t);
155
+ const best = okList[0];
156
+ const winModel = best.model;
157
+ evt("auto-concurrent-win", { reqId, model: winModel, timing: best.val.timing, totalMs: Math.round(performance.now() - raceStart), tried: raceModels.length });
158
+ for (const { r, i } of results.map((r, i) => ({ r, i }))) {
159
+ const m = raceModels[i];
160
+ if (r.status === "fulfilled" && r.value?.ok) {
161
+ if (m === winModel) {
162
+ const latencyMs = r.value.timing?.totalMs ?? Math.round(performance.now() - raceStart);
163
+ await auto.recordOk(m, { latencyMs });
164
+ try { const { savePreferredModel } = await import("../../state.js"); savePreferredModel(m); evt("auto-concurrent-preferred", { reqId, model: m }); } catch {}
165
+ }
166
+ } else if (r.status === "fulfilled" && !r.value?.ok && !r.value?.allowlist) await auto.recordError(m, { status: r.value.status || 502 });
167
+ else if (r.status === "rejected") await auto.recordError(m, { status: 502 });
168
+ }
169
+ handlerCtx.model = winModel;
170
+ const { handleLocalRelay: _relay } = await import("./local-handler.js");
171
+ const lr = await _relay({
172
+ upRes: best.val.res, model: winModel, body, order: raceModels, idx: best.idx,
173
+ lastErr: null, requested, useAuto, lockModel, auto, handlerCtx, evt, logCall, logError, mark, perf0, stages, startedAt, plugins, res,
174
+ });
175
+ if (lr.handled) return;
176
+ if (!lr.lastErr) return;
177
+ } else {
178
+ evt("auto-concurrent-all-fail", { reqId, tried: raceModels.length, totalMs: Math.round(performance.now() - raceStart) });
179
+ for (const { r, i } of results.map((r, i) => ({ r, i }))) {
180
+ const m = raceModels[i];
181
+ if (r.status === "fulfilled" && !r.value?.ok && !r.value?.allowlist) await auto.recordError(m, { status: r.value.status || 502 });
182
+ else if (r.status === "rejected") await auto.recordError(m, { status: 502 });
183
+ }
184
+ }
185
+ const triedSet = new Set(raceModels);
186
+ order = order.filter((m) => !triedSet.has(m));
187
+ if (!order.length) {
188
+ const last = { model: raceModels[0] || requested, status: 502, message: "all concurrent candidates failed" };
189
+ await handleExhaustedAll({ res, body, lastErr: last, order: raceModels, requested, handlerCtx: { ...handlerCtx, reqId, startedAt }, evt, logCall, mark, perf0, stages });
190
+ return;
191
+ }
192
+ }
193
+ }
194
+
195
+ // ===== Executor: 串行 trial =====
196
+ let lastErr = null;
197
+ for (let idx = 0; idx < order.length; idx++) {
198
+ const model = order[idx];
199
+ handlerCtx.model = model;
200
+ evt("model-try", { reqId, model, idx, remaining: order.length - idx });
201
+ if (plugins?.length) {
202
+ const bt = await runHook(plugins, "model:beforeTry", { reqId, requested, model, idx, hops });
203
+ for (const e of bt.errors) evt("plugin-hook-error", { reqId, hook: "model:beforeTry", plugin: e.plugin, error: e.error });
204
+ if (bt.value === false || bt.value?.skip === true) { evt("plugin-hook", { reqId, hook: "model:beforeTry", applied: true, skipped: model }); continue; }
205
+ }
206
+ let upRes = null;
207
+ let forwarded = { ...injectReasoningContent(model, body), model };
208
+ if (plugins?.length) {
209
+ const ur = await runHook(plugins, "upstream:request", { reqId, requested, model, payload: forwarded, stream: Boolean(body.stream) });
210
+ for (const e of ur.errors) evt("plugin-hook-error", { reqId, hook: "upstream:request", plugin: e.plugin, error: e.error });
211
+ if (ur.changed && ur.value?.payload && typeof ur.value.payload === "object") { forwarded = ur.value.payload; evt("plugin-hook", { reqId, hook: "upstream:request", applied: true, model, rewrittenModel: forwarded.model ?? null }); }
212
+ }
213
+ const tUp = performance.now();
214
+ evt("upstream-try", { reqId, model, attempt: idx + 1 });
215
+ try {
216
+ const chatOpts = {};
217
+ if (Object.keys(shareKeys).length) chatOpts.shareKeys = shareKeys;
218
+ if (workbuddyUid) chatOpts.workbuddyUid = workbuddyUid;
219
+ upRes = await upstream.chat(forwarded, Object.keys(chatOpts).length ? chatOpts : undefined);
220
+ evt("upstream-done", { reqId, model, ok: !(upRes instanceof Error) && upRes.status < 400, status: upRes instanceof Error ? null : upRes.status, timing: upRes._t ?? null, error: null });
221
+ } catch (err) {
222
+ if (auto) await auto.recordError(model, { message: errMsg(err) });
223
+ lastErr = { model, upstream: null, status: 502, message: errMsg(err) };
224
+ logError(model, 502, errMsg(err));
225
+ evt("upstream-error", { reqId, model, status: 502, message: errMsg(err), timing: err._t ?? { attempts: [], waitMs: 0, totalMs: Math.round(performance.now() - tUp) } });
226
+ }
227
+ if (plugins?.length) {
228
+ runHook(plugins, "upstream:response", {
229
+ reqId, requested, model,
230
+ status: upRes instanceof Error ? null : upRes instanceof Object ? (upRes.status ?? null) : null,
231
+ ok: !(upRes instanceof Error) && upRes ? upRes.status < 400 : false,
232
+ error: upRes instanceof Error ? errMsg(upRes) : null,
233
+ timing: upRes?._t ?? null,
234
+ }).catch(() => {});
235
+ }
236
+ mark(`up-${model}`);
237
+ if (upRes && upRes.status >= 400) {
238
+ const isAllowlistBlock = upRes.status === 403 && (upRes.headers?.get?.("x-mslxdff-allowlist") === "1");
239
+ if (isAllowlistBlock) {
240
+ let bodyText = null; try { bodyText = await upRes.clone().text(); } catch {}
241
+ let errBody = { error: `model not allowed for provider` };
242
+ try { errBody = bodyText ? JSON.parse(bodyText) : errBody; } catch { errBody = { error: bodyText || "model not allowed" }; }
243
+ if (useAuto) {
244
+ logError(model, 403, errBody.error || "model not allowed");
245
+ evt("upstream-error", { reqId, model, status: 403, message: errBody.error, timing: upRes._t ?? null, allowlist: true, skipped: true });
246
+ lastErr = { model, upstream: upRes, status: 403, message: errBody.error || "model not allowed" };
247
+ if (canFallback && idx < order.length - 1) { evt("fallback", { reqId, from: model, to: order[idx + 1] ?? null, reason: `allowlist skip ${errBody.error || "blocked"}` }); continue; }
248
+ return json(res, 403, errBody);
249
+ }
250
+ logError(model, 403, errBody.error || "model not allowed");
251
+ evt("upstream-error", { reqId, model, status: 403, message: errBody.error, timing: upRes._t ?? null, allowlist: true });
252
+ return json(res, 403, errBody);
253
+ }
254
+ if (auto) await auto.recordError(model, { status: upRes.status });
255
+ lastErr = { model, upstream: upRes, status: upRes.status, message: null };
256
+ logError(model, upRes.status, `upstream ${upRes.status}`);
257
+ evt("upstream-error", { reqId, model, status: upRes.status, message: null, timing: upRes._t ?? null });
258
+ upRes = null;
259
+ }
260
+ if (upRes) {
261
+ const isStream = Boolean(body.stream);
262
+ const d = hedgeDelayMs();
263
+ const hasPeers = Boolean(peers) && peers.ordered().length > 0;
264
+ const doHedge = shouldHedge({ isStream, canForwardPeers, hedgeDelayMs: d, hasPeers }) && upRes.status === 200 && upRes.body;
265
+ if (doHedge) {
266
+ const hr = await handleHedge({ upRes, model, body, order, idx, lastErr, requested, useAuto, lockModel, auto, peers, handlerCtx, evt, logCall, logError, mark, perf0, stages, startedAt, plugins, res, hedgeDelayMs: d });
267
+ if (hr.handled) return;
268
+ if (hr.lastErr) lastErr = hr.lastErr;
269
+ if (hr.upRes === null) upRes = null;
270
+ else if (hr.upRes) upRes = hr.upRes;
271
+ }
272
+ if (upRes) {
273
+ const lr = await handleLocalRelay({ upRes, model, body, order, idx, lastErr, requested, useAuto, lockModel, auto, handlerCtx, evt, logCall, logError, mark, perf0, stages, startedAt, plugins, res });
274
+ if (lr.handled) return;
275
+ if (lr.lastErr) { lastErr = lr.lastErr; continue; }
276
+ return;
277
+ }
278
+ }
279
+ if (canForwardPeers) {
280
+ const pr = await handlePeerRelay({ model, body, lastErr, requested, useAuto, lockModel, auto, peers, handlerCtx, evt, logCall, mark, perf0, stages, startedAt, plugins, res });
281
+ if (pr.handled) return;
282
+ }
283
+ if (groups) {
284
+ const br = await handleBroadbandRelay({ model, body, hops, lastErr, requested, useAuto, lockModel, auto, groups, token, bus, logs, handlerCtx, evt, mark, perf0, stages, res, startedAt, plugins });
285
+ if (br.handled) return;
286
+ }
287
+ if (canFallback) { evt("fallback", { reqId, from: model, to: order[idx + 1] ?? null, reason: lastErr?.message || `upstream ${lastErr?.status ?? 502}` }); continue; }
288
+ await handleExhaustedLocal({ res, body, lastErr, order, handlerCtx: { ...handlerCtx, model, reqId }, evt, logCall, mark, perf0, stages, done, requested, useAuto });
289
+ return;
290
+ }
291
+ await handleExhaustedAll({ res, body, lastErr, order, requested, handlerCtx: { ...handlerCtx, reqId, startedAt }, evt, logCall, mark, perf0, stages });
292
+ }
293
+
294
+ return { handle };
295
+ }
296
+
297
+ // 薄适配:保持原 chatHandler 签名兼容
298
+ export async function chatHandler(ctx) {
299
+ const gw = createChatGateway(ctx);
300
+ return gw.handle(ctx);
301
+ }
@@ -1,8 +1,9 @@
1
- import { buildFallbackInfo } from "../fallback.js";
2
1
  import { relay, SLOW_TOTAL_MS, STREAM_TIMEOUT_MS, STALL_TIMEOUT_MS, SCORE_STALL_MS } from "../stream.js";
2
+ import { buildFallbackInfo } from "../fallback.js";
3
+ import { createRelayPipeline } from "./relay-pipeline.js";
3
4
  import { hedgedFirstChunkRace } from "../hedge.js";
4
- import { runHook } from "../../plugins.js";
5
5
 
6
+ /** 薄适配:hedge 赛跑后按 winner 调 pipeline(复用 relay-pipeline 深模块) */
6
7
  export async function handleHedge({
7
8
  upRes,
8
9
  model,
@@ -27,104 +28,86 @@ export async function handleHedge({
27
28
  res,
28
29
  hedgeDelayMs,
29
30
  }) {
30
- const isStream = Boolean(body.stream);
31
31
  const d = hedgeDelayMs;
32
- // hedge 已在外层判断 doHedge,这里直接执行赛跑
33
32
  try {
34
33
  const hedged = await hedgedFirstChunkRace({ localUpRes: upRes, peers, handlerCtx, hedgeDelayMs: d, evt });
35
34
  if (hedged && hedged.winner) {
36
35
  if (hedged.winner === "local") {
37
- const fallback = buildFallbackInfo({ requested, actual: model, lastErr, via: "local", useAuto, lockModel });
38
- if (fallback?.fallback) evt("fallback-notice", { reqId: handlerCtx.reqId, requested, actual: model, reason: fallback.reason, notice: fallback.notice, via: "local" });
39
- evt("relay-start", { reqId: handlerCtx.reqId, model, via: "local", isStream, fallback, hedged: true });
40
36
  const bufferedUpRes = { ...upRes, body: hedged.bufferedBody, headers: upRes.headers, status: upRes.status, _t: upRes._t };
41
- logCall(model, bufferedUpRes.status);
42
- const out = await relay(res, bufferedUpRes, body, {
43
- fallback,
44
- onFirstChunk: (delta) => {
45
- mark(`ttf-${model}`);
46
- evt("relay-first-chunk", { reqId: handlerCtx.reqId, model, ttfMs: delta, hedged: true, via: "local" });
47
- if (plugins?.length) runHook(plugins, "relay:first-chunk", { reqId: handlerCtx.reqId, requested, model, via: "local", ttfMs: delta }).catch(() => {});
48
- },
49
- onDownstreamAbort: () => {
50
- evt("client-abort", { reqId: handlerCtx.reqId, model, totalMs: Math.round(perf0 ? (Date.now() - startedAt) : 0), stages: [...stages] });
51
- },
37
+ const pipeline = createRelayPipeline({
38
+ relay,
39
+ buildFallbackInfo,
40
+ auto,
41
+ plugins,
42
+ evt,
43
+ mark,
44
+ logCall,
45
+ logError,
46
+ constants: { STREAM_TIMEOUT_MS, SLOW_TOTAL_MS, STALL_TIMEOUT_MS, SCORE_STALL_MS },
47
+ startedAt,
48
+ stages,
49
+ });
50
+ const r = await pipeline.execute({
51
+ res,
52
+ upRes: bufferedUpRes,
53
+ body,
54
+ requested,
55
+ actual: model,
56
+ lastErr,
57
+ via: "local",
58
+ lockModel,
59
+ useAuto,
60
+ handlerCtx,
61
+ mark,
62
+ perf0,
63
+ stages,
64
+ startedAt,
52
65
  });
53
- evt("relay-done", { reqId: handlerCtx.reqId, model, via: "local", status: out.status, ttfMs: out.ttfMs ?? hedged.ttfMs, totalMs: out.totalMs, aborted: out.aborted, interrupted: out.interrupted ?? false, detail: out.detail ?? null, hedged: true });
54
- if (out.status === STREAM_TIMEOUT_MS) {
55
- if (auto) await auto.recordError(model, { status: 502, slow: true, note: `stream timeout ${STREAM_TIMEOUT_MS}ms` });
56
- const err = { model, upstream: null, status: 502, message: `stream timed out after ${STREAM_TIMEOUT_MS}ms` };
57
- logError(model, 502, `stream timeout ${STREAM_TIMEOUT_MS}ms`);
58
- evt("upstream-error", { reqId: handlerCtx.reqId, model, status: 502, message: "stream timeout", timing: null });
59
- evt("fallback", { reqId: handlerCtx.reqId, from: model, to: order[idx + 1] ?? null, reason: "stream timeout" });
60
- return { handled: false, upRes: null, lastErr: err };
61
- }
62
- if (out.interrupted) {
63
- if (auto) {
64
- await auto.recordError(model, { status: 200, slow: true, note: `stall ${STALL_TIMEOUT_MS}ms` });
65
- await auto.recordLatency(model, out.totalMs ?? (Date.now() - startedAt));
66
- }
67
- evt("slow-model", { model, elapsedMs: out.totalMs ?? (Date.now() - startedAt), threshold: STALL_TIMEOUT_MS, interrupted: true, detail: out.detail ?? null });
68
- logCall(model, 200);
69
- evt("result", { model, status: out.status, via: "local", timing: upRes._t ?? null, ttfMs: out.ttfMs, totalMs: out.totalMs, interrupted: true, detail: out.detail ?? null, fallback, requested, actual: model });
70
- evt("client-response", { requested, actual: model, via: "local", fallback, status: out.status, reqId: handlerCtx.reqId });
71
- if (plugins?.length) runHook(plugins, "request:completed", { reqId: handlerCtx.reqId, requested, useAuto, hops: handlerCtx.hops, stream: isStream, durationMs: Date.now() - startedAt, via: "local", status: out.status, actual: model, interrupted: true, fallback }).catch(() => {});
72
- return { handled: true };
73
- }
74
- const elapsed = Date.now() - startedAt;
75
- const latencyMs = out.totalMs ?? elapsed;
76
- let scoredSlow = false;
77
- if (SLOW_TOTAL_MS && auto && elapsed > SLOW_TOTAL_MS && out.status === 200) {
78
- void auto.recordError(model, { status: 200, slow: true, note: `slow ${elapsed}ms` });
79
- void auto.recordLatency(model, latencyMs);
80
- evt("slow-model", { model, elapsedMs: elapsed, threshold: SLOW_TOTAL_MS, reason: "total", detail: out.detail ?? null });
81
- scoredSlow = true;
82
- }
83
- if (out.detail?.stallHits > 0 && auto && out.status === 200) {
84
- void auto.recordError(model, { status: 200, slow: true, note: `stall ${out.detail.stallHits}x gap>${SCORE_STALL_MS}ms maxGap ${out.detail.maxGapMs}ms` });
85
- void auto.recordLatency(model, latencyMs);
86
- evt("slow-model", { model, elapsedMs: elapsed, threshold: SCORE_STALL_MS, reason: "stall", stallHits: out.detail.stallHits, maxGapMs: out.detail.maxGapMs, detail: out.detail ?? null });
87
- scoredSlow = true;
88
- }
89
- if (!scoredSlow && auto && out.status === 200) await auto.recordOk(model, { latencyMs });
90
- else if (!scoredSlow && auto) await auto.recordLatency(model, latencyMs);
91
- evt("result", { model, status: out.status, via: "local", timing: upRes._t ?? null, ttfMs: out.ttfMs, totalMs: out.totalMs, detail: out.detail ?? null, fallback, requested, actual: model, hedged: true });
92
- evt("client-response", { requested, actual: model, via: "local", fallback, status: out.status, reqId: handlerCtx.reqId });
93
- if (plugins?.length) runHook(plugins, "request:completed", { reqId: handlerCtx.reqId, requested, useAuto, hops: handlerCtx.hops, stream: isStream, durationMs: Date.now() - startedAt, via: "local", status: out.status, actual: model, fallback }).catch(() => {});
66
+ // 保留 hedged 语义:若超时则透传 needs fallback
67
+ if (!r.handled) return { handled: false, upRes: null, lastErr: r.lastErr };
94
68
  return { handled: true };
95
- } else if (hedged.winner === "peer" && hedged.peerInfo) {
69
+ }
70
+ if (hedged.winner === "peer" && hedged.peerInfo) {
96
71
  const win = hedged.peerInfo;
97
72
  evt("peer-race-win", { reqId: handlerCtx.reqId, model, winPeer: win.peer.url, winTarget: win.target, latencyMs: win.latencyMs, hedged: true, ttfMs: hedged.ttfMs });
98
73
  await peers.recordResult(win.peer.url, { ok: true, latencyMs: win.latencyMs, model: win.target });
99
- logCall(win.target, win.res.status);
100
- const peerFallback = buildFallbackInfo({ requested, actual: win.target, lastErr, via: "peer", useAuto, lockModel });
101
- if (peerFallback?.fallback) evt("fallback-notice", { reqId: handlerCtx.reqId, requested, actual: win.target, reason: peerFallback.reason, notice: peerFallback.notice, via: "peer" });
102
- evt("relay-start", { reqId: handlerCtx.reqId, model: win.target, via: "peer", isStream, fallback: peerFallback, hedged: true });
103
74
  const bufferedPeerRes = { ...win.res, body: hedged.bufferedBody, headers: win.res.headers, status: win.res.status, _t: win.res._t };
104
- const out = await relay(res, bufferedPeerRes, body, {
105
- fallback: peerFallback,
106
- onFirstChunk: (d) => mark(`ttf-peer-${win.target}`),
107
- onDownstreamAbort: () => evt("client-abort", { reqId: handlerCtx.reqId, model: win.target, totalMs: Math.round(perf0 ? (Date.now() - startedAt) : 0), stages: [...stages] }),
75
+ const pipeline = createRelayPipeline({
76
+ relay,
77
+ buildFallbackInfo,
78
+ auto,
79
+ plugins,
80
+ evt,
81
+ mark,
82
+ logCall,
83
+ logError: () => {},
84
+ constants: { STREAM_TIMEOUT_MS, SLOW_TOTAL_MS, STALL_TIMEOUT_MS, SCORE_STALL_MS },
85
+ startedAt,
86
+ stages,
87
+ });
88
+ await pipeline.execute({
89
+ res,
90
+ upRes: bufferedPeerRes,
91
+ body,
92
+ requested,
93
+ actual: win.target,
94
+ lastErr,
95
+ via: "peer",
96
+ lockModel,
97
+ useAuto,
98
+ handlerCtx: { ...handlerCtx, model: win.target },
99
+ mark: (n) => mark(n),
100
+ perf0,
101
+ stages,
102
+ startedAt,
108
103
  });
109
- evt("relay-done", { reqId: handlerCtx.reqId, model: win.target, via: "peer", status: out.status, ttfMs: out.ttfMs ?? hedged.ttfMs, totalMs: out.totalMs, aborted: out.aborted, interrupted: out.interrupted ?? false, detail: out.detail ?? null, hedged: true });
110
- if (auto && out.status === 200) {
111
- const latencyMs = out.totalMs ?? win.latencyMs;
112
- if (out.detail?.stallHits > 0 || (latencyMs && latencyMs > SLOW_TOTAL_MS)) {
113
- void auto.recordError(win.target, { status: 200, slow: true, note: `peer slow ${latencyMs}ms` });
114
- void auto.recordLatency(win.target, latencyMs);
115
- } else {
116
- await auto.recordOk(win.target, { latencyMs });
117
- }
118
- }
119
- evt("result", { model: win.target, status: out.status, via: "peer", timing: win.res._t ?? null, ttfMs: out.ttfMs, totalMs: out.totalMs, detail: out.detail ?? null, fallback: peerFallback, requested, actual: win.target, hedged: true });
120
- evt("client-response", { requested, actual: win.target, via: "peer", fallback: peerFallback, status: out.status, reqId: handlerCtx.reqId });
121
- if (plugins?.length) runHook(plugins, "request:completed", { reqId: handlerCtx.reqId, requested, useAuto, hops: handlerCtx.hops, stream: isStream, durationMs: Date.now() - startedAt, via: "peer", status: out.status, actual: win.target, fallback: peerFallback }).catch(() => {});
122
104
  return { handled: true };
123
105
  }
124
106
  }
125
107
  if (hedged && hedged.needsPeer) {
126
108
  return { handled: false, upRes: null, lastErr, needsPeer: true };
127
- } else if (!hedged || !hedged.winner) {
109
+ }
110
+ if (!hedged || !hedged.winner) {
128
111
  evt("hedge-both-fail", { reqId: handlerCtx.reqId, model });
129
112
  if (!hedged) {
130
113
  if (auto) await auto.recordError(model, { status: 502, slow: false, note: "hedge both fail" });
@@ -136,7 +119,6 @@ export async function handleHedge({
136
119
  return { handled: false, upRes: null, lastErr };
137
120
  } catch (hedgeErr) {
138
121
  evt("hedge-error", { reqId: handlerCtx.reqId, model, error: String(hedgeErr?.message || hedgeErr).slice(0, 300) });
139
- // 回退到串行
140
122
  return { handled: false, upRes, lastErr };
141
123
  }
142
124
  }