mslxdff 0.1.66 → 0.1.68

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (38) hide show
  1. package/package.json +1 -1
  2. package/src/bench/probe.js +68 -0
  3. package/src/bench/report.js +56 -0
  4. package/src/bench/runner.js +74 -0
  5. package/src/chat/cooling.js +99 -0
  6. package/src/chat/direct.js +57 -0
  7. package/src/chat/gateway.js +148 -0
  8. package/src/chat/orchestrator.js +218 -0
  9. package/src/chat/sse.js +69 -0
  10. package/src/chat/upstream.js +59 -501
  11. package/src/cli/bootstrap.js +1 -408
  12. package/src/cli/commands/provider/bench.js +140 -0
  13. package/src/cli/commands/provider/index.js +2 -0
  14. package/src/cli/provider-row.js +73 -0
  15. package/src/cli/status.js +3 -52
  16. package/src/metrics.js +63 -0
  17. package/src/providers/dispatcher.js +18 -6
  18. package/src/providers/generic.js +2 -0
  19. package/src/providers/openrouter.js +16 -72
  20. package/src/providers/workbuddy/auth.js +175 -0
  21. package/src/providers/workbuddy/balance.js +84 -0
  22. package/src/providers/workbuddy/chat.js +310 -0
  23. package/src/providers/workbuddy/index.js +263 -0
  24. package/src/providers/workbuddy/models.js +111 -0
  25. package/src/providers/workbuddy/rotation-log.js +54 -0
  26. package/src/providers/workbuddy.js +2 -652
  27. package/src/routes/chat/broadband-handler.js +25 -46
  28. package/src/routes/chat/exhausted-handler.js +2 -2
  29. package/src/routes/chat/hedge-handler.js +65 -83
  30. package/src/routes/chat/local-handler.js +32 -61
  31. package/src/routes/chat/peer-handler.js +32 -23
  32. package/src/routes/chat/relay-pipeline.js +151 -0
  33. package/src/runtime/bootstrap.js +408 -0
  34. package/src/state/memory.js +2 -21
  35. package/src/state/merge.js +26 -0
  36. package/src/state/provider-config.js +143 -0
  37. package/src/state/schemas/provider.js +83 -133
  38. package/src/state/store.js +2 -28
@@ -1,8 +1,9 @@
1
+ import { relay, SLOW_TOTAL_MS, STREAM_TIMEOUT_MS, STALL_TIMEOUT_MS, SCORE_STALL_MS } from "../stream.js";
1
2
  import { buildFallbackInfo } from "../fallback.js";
2
- import { relay, SLOW_TOTAL_MS } from "../stream.js";
3
+ import { createRelayPipeline } from "./relay-pipeline.js";
3
4
  import { tryBroadbandRelay } from "../relay-queue.js";
4
- import { runHook } from "../../plugins.js";
5
5
 
6
+ /** 薄适配:broadband 两种形态合一经由 pipeline */
6
7
  export async function handleBroadbandRelay({
7
8
  model,
8
9
  body,
@@ -30,57 +31,35 @@ export async function handleBroadbandRelay({
30
31
  evt("relay-miss", { reqId: handlerCtx.reqId, model });
31
32
  return { handled: false };
32
33
  }
34
+ const pipeline = createRelayPipeline({
35
+ relay,
36
+ buildFallbackInfo,
37
+ auto,
38
+ plugins,
39
+ evt,
40
+ mark,
41
+ logCall: () => {},
42
+ logError: () => {},
43
+ constants: { STREAM_TIMEOUT_MS, SLOW_TOTAL_MS, STALL_TIMEOUT_MS, SCORE_STALL_MS },
44
+ startedAt,
45
+ stages,
46
+ });
33
47
  const isResponse = bb.result && typeof bb.result.status === "number" && typeof bb.result.headers?.get === "function";
34
48
  if (isResponse) {
35
- const bbFallback = buildFallbackInfo({ requested, actual: model, lastErr, via: "broadband", useAuto, lockModel });
36
- if (bbFallback?.fallback) evt("fallback-notice", { reqId: handlerCtx.reqId, requested, actual: model, reason: bbFallback.reason, notice: bbFallback.notice, via: "broadband" });
37
- evt("relay-start", { reqId: handlerCtx.reqId, model, via: "broadband", target: bb.target, group: bb.group, fallback: bbFallback });
38
- const out = await relay(res, bb.result, body, {
39
- fallback: bbFallback,
40
- onFirstChunk: (d) => mark(`ttf-bb-${model}`),
41
- onDownstreamAbort: () => evt("client-abort", { reqId: handlerCtx.reqId, model, totalMs: Math.round(Date.now() - startedAt), stages: [...stages] }),
42
- });
43
- evt("relay-done", { reqId: handlerCtx.reqId, model, via: "broadband", status: out.status, ttfMs: out.ttfMs, totalMs: out.totalMs, aborted: out.aborted, interrupted: out.interrupted ?? false, detail: out.detail ?? null });
44
- if (auto && out.status === 200) {
45
- const latencyMs = out.totalMs ?? 0;
46
- if (out.detail?.stallHits > 0 || (latencyMs && latencyMs > SLOW_TOTAL_MS)) {
47
- void auto.recordError(model, { status: 200, slow: true, note: `broadband slow ${latencyMs}ms` });
48
- void auto.recordLatency(model, latencyMs);
49
- } else {
50
- await auto.recordOk(model, { latencyMs });
51
- }
52
- }
53
- evt("result", { model, status: out.status, via: "broadband", timing: bb.result._t ?? null, ttfMs: out.ttfMs, totalMs: out.totalMs, detail: out.detail ?? null, fallback: bbFallback, requested, actual: model });
54
- evt("client-response", { requested, actual: model, via: "broadband", fallback: bbFallback, status: out.status, reqId: handlerCtx.reqId });
55
- if (plugins?.length) runHook(plugins, "request:completed", { reqId: handlerCtx.reqId, requested, useAuto, hops: handlerCtx.hops, stream: Boolean(body.stream), durationMs: Date.now() - startedAt, via: "broadband", status: out.status, actual: model, fallback: bbFallback }).catch(() => {});
49
+ await pipeline.execute({ res, upRes: bb.result, body, requested, actual: model, lastErr, via: "broadband", lockModel, useAuto, handlerCtx, mark, perf0, stages, startedAt });
56
50
  return { handled: true };
57
- } else if (bb.result && typeof bb.result.status === "number") {
51
+ }
52
+ if (bb.result && typeof bb.result.status === "number") {
53
+ const b = bb.result.body || "";
54
+ const str = typeof b === "string" ? b : JSON.stringify(b);
55
+ const isSSE = bb.result.headers?.["Content-Type"]?.includes("text/event-stream");
58
56
  const fakeRes = {
59
57
  status: bb.result.status,
60
58
  headers: { get: (k) => bb.result.headers?.[k] || bb.result.headers?.[k.toLowerCase()] || null },
61
- text: async () => typeof bb.result.body === "string" ? bb.result.body : JSON.stringify(bb.result.body),
62
- body: (() => {
63
- const b = bb.result.body || "";
64
- const str = typeof b === "string" ? b : JSON.stringify(b);
65
- const isSSE = bb.result.headers?.["Content-Type"]?.includes("text/event-stream");
66
- if (isSSE) {
67
- return (async function* () { yield Buffer.from(str); })();
68
- }
69
- return null;
70
- })(),
59
+ text: async () => str,
60
+ body: isSSE ? (async function* () { yield Buffer.from(str); })() : null,
71
61
  };
72
- const bbLocalFallback = buildFallbackInfo({ requested, actual: model, lastErr, via: "broadband", useAuto, lockModel });
73
- if (bbLocalFallback?.fallback) evt("fallback-notice", { reqId: handlerCtx.reqId, requested, actual: model, reason: bbLocalFallback.reason, notice: bbLocalFallback.notice, via: "broadband" });
74
- evt("relay-start", { reqId: handlerCtx.reqId, model, via: "broadband-local", target: bb.target, group: bb.group, fallback: bbLocalFallback });
75
- const out = await relay(res, fakeRes, body, {
76
- fallback: bbLocalFallback,
77
- onFirstChunk: (d) => mark(`ttf-bb-${model}`),
78
- onDownstreamAbort: () => evt("client-abort", { reqId: handlerCtx.reqId, model, totalMs: Math.round(Date.now() - startedAt), stages: [...stages] }),
79
- });
80
- evt("relay-done", { reqId: handlerCtx.reqId, model, via: "broadband-local", status: out.status, ttfMs: out.ttfMs, totalMs: out.totalMs, aborted: out.aborted, interrupted: out.interrupted ?? false, detail: out.detail ?? null });
81
- evt("result", { model, status: out.status, via: "broadband", timing: null, ttfMs: out.ttfMs, totalMs: out.totalMs, detail: out.detail ?? null, fallback: bbLocalFallback, requested, actual: model });
82
- evt("client-response", { requested, actual: model, via: "broadband", fallback: bbLocalFallback, status: out.status, reqId: handlerCtx.reqId });
83
- if (plugins?.length) runHook(plugins, "request:completed", { reqId: handlerCtx.reqId, requested, useAuto, hops: handlerCtx.hops, stream: Boolean(body.stream), durationMs: Date.now() - startedAt, via: "broadband", status: out.status, actual: model, fallback: bbLocalFallback }).catch(() => {});
62
+ await pipeline.execute({ res, upRes: fakeRes, body, requested, actual: model, lastErr, via: "broadband-local", lockModel, useAuto, handlerCtx, mark, perf0, stages, startedAt });
84
63
  return { handled: true };
85
64
  }
86
65
  evt("relay-miss", { reqId: handlerCtx.reqId, model });
@@ -1,6 +1,6 @@
1
- import { performance } from "node:perf_hooks";
2
- import { json } from "../helpers.js";
3
1
  import { relay } from "../stream.js";
2
+ import { json } from "../helpers.js";
3
+ import { performance } from "node:perf_hooks";
4
4
 
5
5
  export async function handleExhaustedLocal({ res, body, lastErr, order, handlerCtx, evt, logCall, mark, perf0, stages, done, requested, useAuto }) {
6
6
  const model = lastErr?.model ?? handlerCtx.model;
@@ -1,8 +1,9 @@
1
- import { buildFallbackInfo } from "../fallback.js";
2
1
  import { relay, SLOW_TOTAL_MS, STREAM_TIMEOUT_MS, STALL_TIMEOUT_MS, SCORE_STALL_MS } from "../stream.js";
2
+ import { buildFallbackInfo } from "../fallback.js";
3
+ import { createRelayPipeline } from "./relay-pipeline.js";
3
4
  import { hedgedFirstChunkRace } from "../hedge.js";
4
- import { runHook } from "../../plugins.js";
5
5
 
6
+ /** 薄适配:hedge 赛跑后按 winner 调 pipeline(复用 relay-pipeline 深模块) */
6
7
  export async function handleHedge({
7
8
  upRes,
8
9
  model,
@@ -27,104 +28,86 @@ export async function handleHedge({
27
28
  res,
28
29
  hedgeDelayMs,
29
30
  }) {
30
- const isStream = Boolean(body.stream);
31
31
  const d = hedgeDelayMs;
32
- // hedge 已在外层判断 doHedge,这里直接执行赛跑
33
32
  try {
34
33
  const hedged = await hedgedFirstChunkRace({ localUpRes: upRes, peers, handlerCtx, hedgeDelayMs: d, evt });
35
34
  if (hedged && hedged.winner) {
36
35
  if (hedged.winner === "local") {
37
- const fallback = buildFallbackInfo({ requested, actual: model, lastErr, via: "local", useAuto, lockModel });
38
- if (fallback?.fallback) evt("fallback-notice", { reqId: handlerCtx.reqId, requested, actual: model, reason: fallback.reason, notice: fallback.notice, via: "local" });
39
- evt("relay-start", { reqId: handlerCtx.reqId, model, via: "local", isStream, fallback, hedged: true });
40
36
  const bufferedUpRes = { ...upRes, body: hedged.bufferedBody, headers: upRes.headers, status: upRes.status, _t: upRes._t };
41
- logCall(model, bufferedUpRes.status);
42
- const out = await relay(res, bufferedUpRes, body, {
43
- fallback,
44
- onFirstChunk: (delta) => {
45
- mark(`ttf-${model}`);
46
- evt("relay-first-chunk", { reqId: handlerCtx.reqId, model, ttfMs: delta, hedged: true, via: "local" });
47
- if (plugins?.length) runHook(plugins, "relay:first-chunk", { reqId: handlerCtx.reqId, requested, model, via: "local", ttfMs: delta }).catch(() => {});
48
- },
49
- onDownstreamAbort: () => {
50
- evt("client-abort", { reqId: handlerCtx.reqId, model, totalMs: Math.round(perf0 ? (Date.now() - startedAt) : 0), stages: [...stages] });
51
- },
37
+ const pipeline = createRelayPipeline({
38
+ relay,
39
+ buildFallbackInfo,
40
+ auto,
41
+ plugins,
42
+ evt,
43
+ mark,
44
+ logCall,
45
+ logError,
46
+ constants: { STREAM_TIMEOUT_MS, SLOW_TOTAL_MS, STALL_TIMEOUT_MS, SCORE_STALL_MS },
47
+ startedAt,
48
+ stages,
49
+ });
50
+ const r = await pipeline.execute({
51
+ res,
52
+ upRes: bufferedUpRes,
53
+ body,
54
+ requested,
55
+ actual: model,
56
+ lastErr,
57
+ via: "local",
58
+ lockModel,
59
+ useAuto,
60
+ handlerCtx,
61
+ mark,
62
+ perf0,
63
+ stages,
64
+ startedAt,
52
65
  });
53
- evt("relay-done", { reqId: handlerCtx.reqId, model, via: "local", status: out.status, ttfMs: out.ttfMs ?? hedged.ttfMs, totalMs: out.totalMs, aborted: out.aborted, interrupted: out.interrupted ?? false, detail: out.detail ?? null, hedged: true });
54
- if (out.status === STREAM_TIMEOUT_MS) {
55
- if (auto) await auto.recordError(model, { status: 502, slow: true, note: `stream timeout ${STREAM_TIMEOUT_MS}ms` });
56
- const err = { model, upstream: null, status: 502, message: `stream timed out after ${STREAM_TIMEOUT_MS}ms` };
57
- logError(model, 502, `stream timeout ${STREAM_TIMEOUT_MS}ms`);
58
- evt("upstream-error", { reqId: handlerCtx.reqId, model, status: 502, message: "stream timeout", timing: null });
59
- evt("fallback", { reqId: handlerCtx.reqId, from: model, to: order[idx + 1] ?? null, reason: "stream timeout" });
60
- return { handled: false, upRes: null, lastErr: err };
61
- }
62
- if (out.interrupted) {
63
- if (auto) {
64
- await auto.recordError(model, { status: 200, slow: true, note: `stall ${STALL_TIMEOUT_MS}ms` });
65
- await auto.recordLatency(model, out.totalMs ?? (Date.now() - startedAt));
66
- }
67
- evt("slow-model", { model, elapsedMs: out.totalMs ?? (Date.now() - startedAt), threshold: STALL_TIMEOUT_MS, interrupted: true, detail: out.detail ?? null });
68
- logCall(model, 200);
69
- evt("result", { model, status: out.status, via: "local", timing: upRes._t ?? null, ttfMs: out.ttfMs, totalMs: out.totalMs, interrupted: true, detail: out.detail ?? null, fallback, requested, actual: model });
70
- evt("client-response", { requested, actual: model, via: "local", fallback, status: out.status, reqId: handlerCtx.reqId });
71
- if (plugins?.length) runHook(plugins, "request:completed", { reqId: handlerCtx.reqId, requested, useAuto, hops: handlerCtx.hops, stream: isStream, durationMs: Date.now() - startedAt, via: "local", status: out.status, actual: model, interrupted: true, fallback }).catch(() => {});
72
- return { handled: true };
73
- }
74
- const elapsed = Date.now() - startedAt;
75
- const latencyMs = out.totalMs ?? elapsed;
76
- let scoredSlow = false;
77
- if (SLOW_TOTAL_MS && auto && elapsed > SLOW_TOTAL_MS && out.status === 200) {
78
- void auto.recordError(model, { status: 200, slow: true, note: `slow ${elapsed}ms` });
79
- void auto.recordLatency(model, latencyMs);
80
- evt("slow-model", { model, elapsedMs: elapsed, threshold: SLOW_TOTAL_MS, reason: "total", detail: out.detail ?? null });
81
- scoredSlow = true;
82
- }
83
- if (out.detail?.stallHits > 0 && auto && out.status === 200) {
84
- void auto.recordError(model, { status: 200, slow: true, note: `stall ${out.detail.stallHits}x gap>${SCORE_STALL_MS}ms maxGap ${out.detail.maxGapMs}ms` });
85
- void auto.recordLatency(model, latencyMs);
86
- evt("slow-model", { model, elapsedMs: elapsed, threshold: SCORE_STALL_MS, reason: "stall", stallHits: out.detail.stallHits, maxGapMs: out.detail.maxGapMs, detail: out.detail ?? null });
87
- scoredSlow = true;
88
- }
89
- if (!scoredSlow && auto && out.status === 200) await auto.recordOk(model, { latencyMs });
90
- else if (!scoredSlow && auto) await auto.recordLatency(model, latencyMs);
91
- evt("result", { model, status: out.status, via: "local", timing: upRes._t ?? null, ttfMs: out.ttfMs, totalMs: out.totalMs, detail: out.detail ?? null, fallback, requested, actual: model, hedged: true });
92
- evt("client-response", { requested, actual: model, via: "local", fallback, status: out.status, reqId: handlerCtx.reqId });
93
- if (plugins?.length) runHook(plugins, "request:completed", { reqId: handlerCtx.reqId, requested, useAuto, hops: handlerCtx.hops, stream: isStream, durationMs: Date.now() - startedAt, via: "local", status: out.status, actual: model, fallback }).catch(() => {});
66
+ // 保留 hedged 语义:若超时则透传 needs fallback
67
+ if (!r.handled) return { handled: false, upRes: null, lastErr: r.lastErr };
94
68
  return { handled: true };
95
- } else if (hedged.winner === "peer" && hedged.peerInfo) {
69
+ }
70
+ if (hedged.winner === "peer" && hedged.peerInfo) {
96
71
  const win = hedged.peerInfo;
97
72
  evt("peer-race-win", { reqId: handlerCtx.reqId, model, winPeer: win.peer.url, winTarget: win.target, latencyMs: win.latencyMs, hedged: true, ttfMs: hedged.ttfMs });
98
73
  await peers.recordResult(win.peer.url, { ok: true, latencyMs: win.latencyMs, model: win.target });
99
- logCall(win.target, win.res.status);
100
- const peerFallback = buildFallbackInfo({ requested, actual: win.target, lastErr, via: "peer", useAuto, lockModel });
101
- if (peerFallback?.fallback) evt("fallback-notice", { reqId: handlerCtx.reqId, requested, actual: win.target, reason: peerFallback.reason, notice: peerFallback.notice, via: "peer" });
102
- evt("relay-start", { reqId: handlerCtx.reqId, model: win.target, via: "peer", isStream, fallback: peerFallback, hedged: true });
103
74
  const bufferedPeerRes = { ...win.res, body: hedged.bufferedBody, headers: win.res.headers, status: win.res.status, _t: win.res._t };
104
- const out = await relay(res, bufferedPeerRes, body, {
105
- fallback: peerFallback,
106
- onFirstChunk: (d) => mark(`ttf-peer-${win.target}`),
107
- onDownstreamAbort: () => evt("client-abort", { reqId: handlerCtx.reqId, model: win.target, totalMs: Math.round(perf0 ? (Date.now() - startedAt) : 0), stages: [...stages] }),
75
+ const pipeline = createRelayPipeline({
76
+ relay,
77
+ buildFallbackInfo,
78
+ auto,
79
+ plugins,
80
+ evt,
81
+ mark,
82
+ logCall,
83
+ logError: () => {},
84
+ constants: { STREAM_TIMEOUT_MS, SLOW_TOTAL_MS, STALL_TIMEOUT_MS, SCORE_STALL_MS },
85
+ startedAt,
86
+ stages,
87
+ });
88
+ await pipeline.execute({
89
+ res,
90
+ upRes: bufferedPeerRes,
91
+ body,
92
+ requested,
93
+ actual: win.target,
94
+ lastErr,
95
+ via: "peer",
96
+ lockModel,
97
+ useAuto,
98
+ handlerCtx: { ...handlerCtx, model: win.target },
99
+ mark: (n) => mark(n),
100
+ perf0,
101
+ stages,
102
+ startedAt,
108
103
  });
109
- evt("relay-done", { reqId: handlerCtx.reqId, model: win.target, via: "peer", status: out.status, ttfMs: out.ttfMs ?? hedged.ttfMs, totalMs: out.totalMs, aborted: out.aborted, interrupted: out.interrupted ?? false, detail: out.detail ?? null, hedged: true });
110
- if (auto && out.status === 200) {
111
- const latencyMs = out.totalMs ?? win.latencyMs;
112
- if (out.detail?.stallHits > 0 || (latencyMs && latencyMs > SLOW_TOTAL_MS)) {
113
- void auto.recordError(win.target, { status: 200, slow: true, note: `peer slow ${latencyMs}ms` });
114
- void auto.recordLatency(win.target, latencyMs);
115
- } else {
116
- await auto.recordOk(win.target, { latencyMs });
117
- }
118
- }
119
- evt("result", { model: win.target, status: out.status, via: "peer", timing: win.res._t ?? null, ttfMs: out.ttfMs, totalMs: out.totalMs, detail: out.detail ?? null, fallback: peerFallback, requested, actual: win.target, hedged: true });
120
- evt("client-response", { requested, actual: win.target, via: "peer", fallback: peerFallback, status: out.status, reqId: handlerCtx.reqId });
121
- if (plugins?.length) runHook(plugins, "request:completed", { reqId: handlerCtx.reqId, requested, useAuto, hops: handlerCtx.hops, stream: isStream, durationMs: Date.now() - startedAt, via: "peer", status: out.status, actual: win.target, fallback: peerFallback }).catch(() => {});
122
104
  return { handled: true };
123
105
  }
124
106
  }
125
107
  if (hedged && hedged.needsPeer) {
126
108
  return { handled: false, upRes: null, lastErr, needsPeer: true };
127
- } else if (!hedged || !hedged.winner) {
109
+ }
110
+ if (!hedged || !hedged.winner) {
128
111
  evt("hedge-both-fail", { reqId: handlerCtx.reqId, model });
129
112
  if (!hedged) {
130
113
  if (auto) await auto.recordError(model, { status: 502, slow: false, note: "hedge both fail" });
@@ -136,7 +119,6 @@ export async function handleHedge({
136
119
  return { handled: false, upRes: null, lastErr };
137
120
  } catch (hedgeErr) {
138
121
  evt("hedge-error", { reqId: handlerCtx.reqId, model, error: String(hedgeErr?.message || hedgeErr).slice(0, 300) });
139
- // 回退到串行
140
122
  return { handled: false, upRes, lastErr };
141
123
  }
142
124
  }
@@ -1,8 +1,8 @@
1
- import { buildFallbackInfo } from "../fallback.js";
2
1
  import { relay, SLOW_TOTAL_MS, STREAM_TIMEOUT_MS, STALL_TIMEOUT_MS, SCORE_STALL_MS } from "../stream.js";
3
- import { runHook } from "../../plugins.js";
4
- import { performance } from "node:perf_hooks";
2
+ import { buildFallbackInfo } from "../fallback.js";
3
+ import { createRelayPipeline } from "./relay-pipeline.js";
5
4
 
5
+ /** 薄适配:本地透传唯一经由 relay-pipeline */
6
6
  export async function handleLocalRelay({
7
7
  upRes,
8
8
  model,
@@ -25,64 +25,35 @@ export async function handleLocalRelay({
25
25
  plugins,
26
26
  res,
27
27
  }) {
28
- logCall(model, upRes.status);
29
- const fallback = buildFallbackInfo({ requested, actual: model, lastErr, via: "local", useAuto, lockModel });
30
- if (fallback?.fallback) evt("fallback-notice", { reqId: handlerCtx.reqId, requested, actual: model, reason: fallback.reason, notice: fallback.notice, via: "local" });
31
- evt("relay-start", { reqId: handlerCtx.reqId, model, via: "local", isStream: Boolean(body.stream), fallback });
32
- const out = await relay(res, upRes, body, {
33
- fallback,
34
- onFirstChunk: (delta) => {
35
- mark(`ttf-${model}`);
36
- evt("relay-first-chunk", { reqId: handlerCtx.reqId, model, ttfMs: delta });
37
- if (plugins?.length) runHook(plugins, "relay:first-chunk", { reqId: handlerCtx.reqId, requested, model, via: "local", ttfMs: delta }).catch(() => {});
38
- },
39
- onDownstreamAbort: () => {
40
- evt("client-abort", { reqId: handlerCtx.reqId, model, totalMs: Math.round(performance.now() - perf0), stages: [...stages] });
41
- },
28
+ const pipeline = createRelayPipeline({
29
+ relay,
30
+ buildFallbackInfo,
31
+ auto,
32
+ plugins,
33
+ evt,
34
+ mark,
35
+ logCall,
36
+ logError,
37
+ constants: { STREAM_TIMEOUT_MS, SLOW_TOTAL_MS, STALL_TIMEOUT_MS, SCORE_STALL_MS },
38
+ startedAt,
39
+ stages,
40
+ });
41
+ const r = await pipeline.execute({
42
+ res,
43
+ upRes,
44
+ body,
45
+ requested,
46
+ actual: model,
47
+ lastErr,
48
+ via: "local",
49
+ lockModel,
50
+ useAuto,
51
+ handlerCtx,
52
+ mark,
53
+ perf0,
54
+ stages,
55
+ startedAt,
42
56
  });
43
- evt("relay-done", { reqId: handlerCtx.reqId, model, via: "local", status: out.status, ttfMs: out.ttfMs, totalMs: out.totalMs, aborted: out.aborted, interrupted: out.interrupted ?? false, detail: out.detail ?? null });
44
- if (out.status === STREAM_TIMEOUT_MS) {
45
- if (auto) await auto.recordError(model, { status: 502, slow: true, note: `stream timeout ${STREAM_TIMEOUT_MS}ms` });
46
- const err = { model, upstream: null, status: 502, message: `stream timed out after ${STREAM_TIMEOUT_MS}ms` };
47
- logError(model, 502, `stream timeout ${STREAM_TIMEOUT_MS}ms`);
48
- evt("upstream-error", { reqId: handlerCtx.reqId, model, status: 502, message: "stream timeout", timing: null });
49
- evt("fallback", { reqId: handlerCtx.reqId, from: model, to: order[idx + 1] ?? null, reason: "stream timeout" });
50
- return { handled: false, upRes: null, lastErr: err };
51
- }
52
- if (out.interrupted) {
53
- if (auto) {
54
- await auto.recordError(model, { status: 200, slow: true, note: `stall ${STALL_TIMEOUT_MS}ms` });
55
- await auto.recordLatency(model, out.totalMs ?? (Date.now() - startedAt));
56
- }
57
- evt("slow-model", { model, elapsedMs: out.totalMs ?? (Date.now() - startedAt), threshold: STALL_TIMEOUT_MS, interrupted: true, detail: out.detail ?? null });
58
- logCall(model, 200);
59
- evt("result", { model, status: out.status, via: "local", timing: upRes._t ?? null, ttfMs: out.ttfMs, totalMs: out.totalMs, interrupted: true, detail: out.detail ?? null, fallback, requested, actual: model });
60
- evt("client-response", { requested, actual: model, via: "local", fallback, status: out.status, interrupted: true, reqId: handlerCtx.reqId });
61
- if (plugins?.length) runHook(plugins, "request:completed", { reqId: handlerCtx.reqId, requested, useAuto, hops: handlerCtx.hops, stream: Boolean(body.stream), durationMs: Date.now() - startedAt, via: "local", status: out.status, actual: model, interrupted: true, fallback }).catch(() => {});
62
- return { handled: true };
63
- }
64
- const elapsed = Date.now() - startedAt;
65
- const latencyMs = out.totalMs ?? elapsed;
66
- let scoredSlow = false;
67
- if (SLOW_TOTAL_MS && auto && elapsed > SLOW_TOTAL_MS && out.status === 200) {
68
- void auto.recordError(model, { status: 200, slow: true, note: `slow ${elapsed}ms` });
69
- void auto.recordLatency(model, latencyMs);
70
- evt("slow-model", { model, elapsedMs: elapsed, threshold: SLOW_TOTAL_MS, reason: "total", detail: out.detail ?? null });
71
- scoredSlow = true;
72
- }
73
- if (out.detail?.stallHits > 0 && auto && out.status === 200) {
74
- void auto.recordError(model, { status: 200, slow: true, note: `stall ${out.detail.stallHits}x gap>${SCORE_STALL_MS}ms maxGap ${out.detail.maxGapMs}ms` });
75
- void auto.recordLatency(model, latencyMs);
76
- evt("slow-model", { model, elapsedMs: elapsed, threshold: SCORE_STALL_MS, reason: "stall", stallHits: out.detail.stallHits, maxGapMs: out.detail.maxGapMs, detail: out.detail ?? null });
77
- scoredSlow = true;
78
- }
79
- if (!scoredSlow && auto && out.status === 200) {
80
- await auto.recordOk(model, { latencyMs });
81
- } else if (!scoredSlow && auto) {
82
- await auto.recordLatency(model, latencyMs);
83
- }
84
- evt("result", { model, status: out.status, via: "local", timing: upRes._t ?? null, ttfMs: out.ttfMs, totalMs: out.totalMs, detail: out.detail ?? null, fallback, requested, actual: model });
85
- evt("client-response", { requested, actual: model, via: "local", fallback, status: out.status, reqId: handlerCtx.reqId });
86
- if (plugins?.length) runHook(plugins, "request:completed", { reqId: handlerCtx.reqId, requested, useAuto, hops: handlerCtx.hops, stream: Boolean(body.stream), durationMs: Date.now() - startedAt, via: "local", status: out.status, actual: model, fallback }).catch(() => {});
57
+ if (!r.handled) return { handled: false, upRes: null, lastErr: r.lastErr };
87
58
  return { handled: true };
88
59
  }
@@ -1,8 +1,9 @@
1
+ import { relay, SLOW_TOTAL_MS, STREAM_TIMEOUT_MS, STALL_TIMEOUT_MS, SCORE_STALL_MS } from "../stream.js";
1
2
  import { buildFallbackInfo } from "../fallback.js";
2
- import { relay, SLOW_TOTAL_MS } from "../stream.js";
3
+ import { createRelayPipeline } from "./relay-pipeline.js";
3
4
  import { racePeerCandidates } from "../peers.js";
4
- import { runHook } from "../../plugins.js";
5
5
 
6
+ /** 薄适配:peer 赛跑后经由 pipeline(via=peer) */
6
7
  export async function handlePeerRelay({
7
8
  model,
8
9
  body,
@@ -32,27 +33,35 @@ export async function handlePeerRelay({
32
33
  }
33
34
  evt("peer-race-win", { reqId: handlerCtx.reqId, model, winPeer: win.peer.url, winTarget: win.target, latencyMs: win.latencyMs });
34
35
  await peers.recordResult(win.peer.url, { ok: true, latencyMs: win.latencyMs, model: win.target });
35
- logCall(win.target, win.res.status);
36
- const peerFallback = buildFallbackInfo({ requested, actual: win.target, lastErr, via: "peer", useAuto, lockModel });
37
- if (peerFallback?.fallback) evt("fallback-notice", { reqId: handlerCtx.reqId, requested, actual: win.target, reason: peerFallback.reason, notice: peerFallback.notice, via: "peer" });
38
- evt("relay-start", { reqId: handlerCtx.reqId, model: win.target, via: "peer", isStream: Boolean(body.stream), fallback: peerFallback });
39
- const out = await relay(res, win.res, body, {
40
- fallback: peerFallback,
41
- onFirstChunk: (d) => mark(`ttf-peer-${win.target}`),
42
- onDownstreamAbort: () => evt("client-abort", { reqId: handlerCtx.reqId, model: win.target, totalMs: Math.round(Date.now() - startedAt), stages: [...stages] }),
36
+
37
+ const pipeline = createRelayPipeline({
38
+ relay,
39
+ buildFallbackInfo,
40
+ auto,
41
+ plugins,
42
+ evt,
43
+ mark,
44
+ logCall,
45
+ logError: () => {},
46
+ constants: { STREAM_TIMEOUT_MS, SLOW_TOTAL_MS, STALL_TIMEOUT_MS, SCORE_STALL_MS },
47
+ startedAt,
48
+ stages,
49
+ });
50
+ await pipeline.execute({
51
+ res,
52
+ upRes: win.res,
53
+ body,
54
+ requested,
55
+ actual: win.target,
56
+ lastErr,
57
+ via: "peer",
58
+ lockModel,
59
+ useAuto,
60
+ handlerCtx: { ...handlerCtx, model: win.target },
61
+ mark,
62
+ perf0,
63
+ stages,
64
+ startedAt,
43
65
  });
44
- evt("relay-done", { reqId: handlerCtx.reqId, model: win.target, via: "peer", status: out.status, ttfMs: out.ttfMs, totalMs: out.totalMs, aborted: out.aborted, interrupted: out.interrupted ?? false, detail: out.detail ?? null });
45
- if (auto && out.status === 200) {
46
- const latencyMs = out.totalMs ?? win.latencyMs;
47
- if (out.detail?.stallHits > 0 || (latencyMs && latencyMs > SLOW_TOTAL_MS)) {
48
- void auto.recordError(win.target, { status: 200, slow: true, note: `peer slow ${latencyMs}ms` });
49
- void auto.recordLatency(win.target, latencyMs);
50
- } else {
51
- await auto.recordOk(win.target, { latencyMs });
52
- }
53
- }
54
- evt("result", { model: win.target, status: out.status, via: "peer", timing: win.res._t ?? null, ttfMs: out.ttfMs, totalMs: out.totalMs, detail: out.detail ?? null, fallback: peerFallback, requested, actual: win.target });
55
- evt("client-response", { requested, actual: win.target, via: "peer", fallback: peerFallback, status: out.status, reqId: handlerCtx.reqId });
56
- if (plugins?.length) runHook(plugins, "request:completed", { reqId: handlerCtx.reqId, requested, useAuto, hops: handlerCtx.hops, stream: Boolean(body.stream), durationMs: Date.now() - startedAt, via: "peer", status: out.status, actual: win.target, fallback: peerFallback }).catch(() => {});
57
66
  return { handled: true };
58
67
  }
@@ -0,0 +1,151 @@
1
+ import { runHook } from "../../plugins.js";
2
+
3
+ /**
4
+ * RelayPipeline 深模块
5
+ * 把 5 个 handler 各自的 fallback→relay→scoring→事件 6段流水收敛为单一真相。
6
+ * 对外 1 接口:createRelayPipeline(deps) => { execute(ctx) }
7
+ * 设计要点:全部外部可注入,便于在 pipeline seam 上做行为测试;constants 注入便于单测加速。
8
+ */
9
+ export function createRelayPipeline({
10
+ relay,
11
+ buildFallbackInfo,
12
+ auto,
13
+ plugins,
14
+ evt,
15
+ mark,
16
+ logCall,
17
+ logError,
18
+ constants,
19
+ startedAt: defaultStartedAt,
20
+ stages: defaultStages,
21
+ perfNow,
22
+ } = {}) {
23
+ const C = {
24
+ STREAM_TIMEOUT_MS: 25_000,
25
+ SLOW_TOTAL_MS: 20_000,
26
+ STALL_TIMEOUT_MS: 0,
27
+ SCORE_STALL_MS: 15_000,
28
+ ...(constants || {}),
29
+ };
30
+ const _relay = relay;
31
+ const _build = buildFallbackInfo;
32
+ const _evt = evt || (() => {});
33
+ const _mark = mark || (() => {});
34
+ const _logCall = logCall || (() => {});
35
+ const _logError = logError || (() => {});
36
+ const _perfNow = perfNow || (() => Date.now());
37
+
38
+ async function execute({
39
+ res,
40
+ upRes,
41
+ body,
42
+ requested,
43
+ actual,
44
+ lastErr,
45
+ via,
46
+ lockModel,
47
+ useAuto,
48
+ handlerCtx,
49
+ mark: m2,
50
+ perf0,
51
+ stages: s2,
52
+ startedAt: sa2,
53
+ } = {}) {
54
+ const markFn = m2 || _mark;
55
+ const curStartedAt = sa2 ?? defaultStartedAt ?? Date.now();
56
+ const curStages = s2 ?? defaultStages ?? [];
57
+ const reqId = handlerCtx?.reqId;
58
+ const hops = handlerCtx?.hops;
59
+
60
+ // 1. logCall(pre) — 保持原 handler 的 logCall→fallback→relay-start 时序
61
+ try { _logCall(actual, upRes?.status); } catch {}
62
+ // 2. fallback + relay-start
63
+ let fallback = null;
64
+ try {
65
+ if (_build) fallback = _build({ requested, actual, lastErr, via, useAuto, lockModel });
66
+ } catch {}
67
+ if (fallback?.fallback) {
68
+ _evt("fallback-notice", { reqId, requested, actual, reason: fallback.reason, notice: fallback.notice, via, fallback: true });
69
+ }
70
+ _evt("relay-start", { reqId, model: actual, via, isStream: Boolean(body?.stream), fallback });
71
+
72
+ // 3. relay
73
+ const out = await _relay(res, upRes, body, {
74
+ fallback,
75
+ onFirstChunk: (delta) => {
76
+ try { markFn(`ttf-${actual}`); } catch {}
77
+ _evt("relay-first-chunk", { reqId, model: actual, ttfMs: delta, via });
78
+ if (plugins?.length) runHook(plugins, "relay:first-chunk", { reqId, requested, model: actual, via, ttfMs: delta }).catch(() => {});
79
+ },
80
+ onDownstreamAbort: () => {
81
+ _evt("client-abort", { reqId, model: actual, totalMs: Math.round(_perfNow() - (perf0 ?? 0)), stages: [...curStages] });
82
+ },
83
+ });
84
+
85
+ // 4. relay-done
86
+ _evt("relay-done", {
87
+ reqId,
88
+ model: actual,
89
+ via,
90
+ status: out.status,
91
+ ttfMs: out.ttfMs,
92
+ totalMs: out.totalMs,
93
+ aborted: out.aborted,
94
+ interrupted: out.interrupted ?? false,
95
+ detail: out.detail ?? null,
96
+ });
97
+
98
+ // 5a. 首块超时未写字节 → 回退
99
+ if (out.status === C.STREAM_TIMEOUT_MS) {
100
+ if (auto) try { await auto.recordError(actual, { status: 502, slow: true, note: `stream timeout ${C.STREAM_TIMEOUT_MS}ms` }); } catch {}
101
+ try { _logError(actual, 502, `stream timeout ${C.STREAM_TIMEOUT_MS}ms`); } catch {}
102
+ _evt("upstream-error", { reqId, model: actual, status: 502, message: "stream timeout", timing: null });
103
+ _evt("fallback", { reqId, from: actual, to: null, reason: "stream timeout" });
104
+ return { handled: false, upRes: null, lastErr: { model: actual, upstream: null, status: 502, message: `stream timed out after ${C.STREAM_TIMEOUT_MS}ms` } };
105
+ }
106
+
107
+ // 5b. 中断(stall 超时 / max 流时长)
108
+ if (out.interrupted) {
109
+ if (auto) {
110
+ try { await auto.recordError(actual, { status: 200, slow: true, note: `stall ${C.STALL_TIMEOUT_MS}ms` }); } catch {}
111
+ try { await auto.recordLatency(actual, out.totalMs ?? (Date.now() - curStartedAt)); } catch {}
112
+ }
113
+ _evt("slow-model", { reqId, model: actual, elapsedMs: out.totalMs ?? (Date.now() - curStartedAt), threshold: C.STALL_TIMEOUT_MS, interrupted: true, detail: out.detail ?? null });
114
+ try { _logCall(actual, 200); } catch {}
115
+ _evt("result", { reqId, model: actual, status: out.status, via, timing: upRes?._t ?? null, ttfMs: out.ttfMs, totalMs: out.totalMs, interrupted: true, detail: out.detail ?? null, fallback, requested, actual });
116
+ _evt("client-response", { requested, actual, via, fallback, status: out.status, reqId, interrupted: true });
117
+ if (plugins?.length) runHook(plugins, "request:completed", { reqId, requested, useAuto, hops, stream: Boolean(body?.stream), durationMs: Date.now() - curStartedAt, via, status: out.status, actual, interrupted: true, fallback }).catch(() => {});
118
+ return { handled: true };
119
+ }
120
+
121
+ // 5c. 慢速计分 + ok
122
+ const elapsed = Date.now() - curStartedAt;
123
+ const latencyMs = out.totalMs ?? elapsed;
124
+ let scoredSlow = false;
125
+
126
+ if (C.SLOW_TOTAL_MS && auto && elapsed > C.SLOW_TOTAL_MS && out.status === 200) {
127
+ try { await auto.recordError(actual, { status: 200, slow: true, note: `slow ${elapsed}ms` }); } catch {}
128
+ try { await auto.recordLatency(actual, latencyMs); } catch {}
129
+ _evt("slow-model", { reqId, model: actual, elapsedMs: elapsed, threshold: C.SLOW_TOTAL_MS, reason: "total", detail: out.detail ?? null });
130
+ scoredSlow = true;
131
+ }
132
+ if (out.detail?.stallHits > 0 && auto && out.status === 200) {
133
+ try { await auto.recordError(actual, { status: 200, slow: true, note: `stall ${out.detail.stallHits}x gap>${C.SCORE_STALL_MS}ms maxGap ${out.detail.maxGapMs}ms` }); } catch {}
134
+ try { await auto.recordLatency(actual, latencyMs); } catch {}
135
+ _evt("slow-model", { reqId, model: actual, elapsedMs: elapsed, threshold: C.SCORE_STALL_MS, reason: "stall", stallHits: out.detail.stallHits, maxGapMs: out.detail.maxGapMs, detail: out.detail ?? null });
136
+ scoredSlow = true;
137
+ }
138
+ if (!scoredSlow && auto && out.status === 200) {
139
+ try { await auto.recordOk(actual, { latencyMs }); } catch {}
140
+ } else if (!scoredSlow && auto) {
141
+ try { await auto.recordLatency(actual, latencyMs); } catch {}
142
+ }
143
+
144
+ _evt("result", { reqId, model: actual, status: out.status, via, timing: upRes?._t ?? null, ttfMs: out.ttfMs, totalMs: out.totalMs, detail: out.detail ?? null, fallback, requested, actual });
145
+ _evt("client-response", { requested, actual, via, fallback, status: out.status, reqId });
146
+ if (plugins?.length) runHook(plugins, "request:completed", { reqId, requested, useAuto, hops, stream: Boolean(body?.stream), durationMs: Date.now() - curStartedAt, via, status: out.status, actual, fallback }).catch(() => {});
147
+ return { handled: true };
148
+ }
149
+
150
+ return { execute };
151
+ }