mslxdff 0.1.66 → 0.1.68
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/src/bench/probe.js +68 -0
- package/src/bench/report.js +56 -0
- package/src/bench/runner.js +74 -0
- package/src/chat/cooling.js +99 -0
- package/src/chat/direct.js +57 -0
- package/src/chat/gateway.js +148 -0
- package/src/chat/orchestrator.js +218 -0
- package/src/chat/sse.js +69 -0
- package/src/chat/upstream.js +59 -501
- package/src/cli/bootstrap.js +1 -408
- package/src/cli/commands/provider/bench.js +140 -0
- package/src/cli/commands/provider/index.js +2 -0
- package/src/cli/provider-row.js +73 -0
- package/src/cli/status.js +3 -52
- package/src/metrics.js +63 -0
- package/src/providers/dispatcher.js +18 -6
- package/src/providers/generic.js +2 -0
- package/src/providers/openrouter.js +16 -72
- package/src/providers/workbuddy/auth.js +175 -0
- package/src/providers/workbuddy/balance.js +84 -0
- package/src/providers/workbuddy/chat.js +310 -0
- package/src/providers/workbuddy/index.js +263 -0
- package/src/providers/workbuddy/models.js +111 -0
- package/src/providers/workbuddy/rotation-log.js +54 -0
- package/src/providers/workbuddy.js +2 -652
- package/src/routes/chat/broadband-handler.js +25 -46
- package/src/routes/chat/exhausted-handler.js +2 -2
- package/src/routes/chat/hedge-handler.js +65 -83
- package/src/routes/chat/local-handler.js +32 -61
- package/src/routes/chat/peer-handler.js +32 -23
- package/src/routes/chat/relay-pipeline.js +151 -0
- package/src/runtime/bootstrap.js +408 -0
- package/src/state/memory.js +2 -21
- package/src/state/merge.js +26 -0
- package/src/state/provider-config.js +143 -0
- package/src/state/schemas/provider.js +83 -133
- package/src/state/store.js +2 -28
|
@@ -1,8 +1,9 @@
|
|
|
1
|
+
import { relay, SLOW_TOTAL_MS, STREAM_TIMEOUT_MS, STALL_TIMEOUT_MS, SCORE_STALL_MS } from "../stream.js";
|
|
1
2
|
import { buildFallbackInfo } from "../fallback.js";
|
|
2
|
-
import {
|
|
3
|
+
import { createRelayPipeline } from "./relay-pipeline.js";
|
|
3
4
|
import { tryBroadbandRelay } from "../relay-queue.js";
|
|
4
|
-
import { runHook } from "../../plugins.js";
|
|
5
5
|
|
|
6
|
+
/** 薄适配:broadband 两种形态合一经由 pipeline */
|
|
6
7
|
export async function handleBroadbandRelay({
|
|
7
8
|
model,
|
|
8
9
|
body,
|
|
@@ -30,57 +31,35 @@ export async function handleBroadbandRelay({
|
|
|
30
31
|
evt("relay-miss", { reqId: handlerCtx.reqId, model });
|
|
31
32
|
return { handled: false };
|
|
32
33
|
}
|
|
34
|
+
const pipeline = createRelayPipeline({
|
|
35
|
+
relay,
|
|
36
|
+
buildFallbackInfo,
|
|
37
|
+
auto,
|
|
38
|
+
plugins,
|
|
39
|
+
evt,
|
|
40
|
+
mark,
|
|
41
|
+
logCall: () => {},
|
|
42
|
+
logError: () => {},
|
|
43
|
+
constants: { STREAM_TIMEOUT_MS, SLOW_TOTAL_MS, STALL_TIMEOUT_MS, SCORE_STALL_MS },
|
|
44
|
+
startedAt,
|
|
45
|
+
stages,
|
|
46
|
+
});
|
|
33
47
|
const isResponse = bb.result && typeof bb.result.status === "number" && typeof bb.result.headers?.get === "function";
|
|
34
48
|
if (isResponse) {
|
|
35
|
-
|
|
36
|
-
if (bbFallback?.fallback) evt("fallback-notice", { reqId: handlerCtx.reqId, requested, actual: model, reason: bbFallback.reason, notice: bbFallback.notice, via: "broadband" });
|
|
37
|
-
evt("relay-start", { reqId: handlerCtx.reqId, model, via: "broadband", target: bb.target, group: bb.group, fallback: bbFallback });
|
|
38
|
-
const out = await relay(res, bb.result, body, {
|
|
39
|
-
fallback: bbFallback,
|
|
40
|
-
onFirstChunk: (d) => mark(`ttf-bb-${model}`),
|
|
41
|
-
onDownstreamAbort: () => evt("client-abort", { reqId: handlerCtx.reqId, model, totalMs: Math.round(Date.now() - startedAt), stages: [...stages] }),
|
|
42
|
-
});
|
|
43
|
-
evt("relay-done", { reqId: handlerCtx.reqId, model, via: "broadband", status: out.status, ttfMs: out.ttfMs, totalMs: out.totalMs, aborted: out.aborted, interrupted: out.interrupted ?? false, detail: out.detail ?? null });
|
|
44
|
-
if (auto && out.status === 200) {
|
|
45
|
-
const latencyMs = out.totalMs ?? 0;
|
|
46
|
-
if (out.detail?.stallHits > 0 || (latencyMs && latencyMs > SLOW_TOTAL_MS)) {
|
|
47
|
-
void auto.recordError(model, { status: 200, slow: true, note: `broadband slow ${latencyMs}ms` });
|
|
48
|
-
void auto.recordLatency(model, latencyMs);
|
|
49
|
-
} else {
|
|
50
|
-
await auto.recordOk(model, { latencyMs });
|
|
51
|
-
}
|
|
52
|
-
}
|
|
53
|
-
evt("result", { model, status: out.status, via: "broadband", timing: bb.result._t ?? null, ttfMs: out.ttfMs, totalMs: out.totalMs, detail: out.detail ?? null, fallback: bbFallback, requested, actual: model });
|
|
54
|
-
evt("client-response", { requested, actual: model, via: "broadband", fallback: bbFallback, status: out.status, reqId: handlerCtx.reqId });
|
|
55
|
-
if (plugins?.length) runHook(plugins, "request:completed", { reqId: handlerCtx.reqId, requested, useAuto, hops: handlerCtx.hops, stream: Boolean(body.stream), durationMs: Date.now() - startedAt, via: "broadband", status: out.status, actual: model, fallback: bbFallback }).catch(() => {});
|
|
49
|
+
await pipeline.execute({ res, upRes: bb.result, body, requested, actual: model, lastErr, via: "broadband", lockModel, useAuto, handlerCtx, mark, perf0, stages, startedAt });
|
|
56
50
|
return { handled: true };
|
|
57
|
-
}
|
|
51
|
+
}
|
|
52
|
+
if (bb.result && typeof bb.result.status === "number") {
|
|
53
|
+
const b = bb.result.body || "";
|
|
54
|
+
const str = typeof b === "string" ? b : JSON.stringify(b);
|
|
55
|
+
const isSSE = bb.result.headers?.["Content-Type"]?.includes("text/event-stream");
|
|
58
56
|
const fakeRes = {
|
|
59
57
|
status: bb.result.status,
|
|
60
58
|
headers: { get: (k) => bb.result.headers?.[k] || bb.result.headers?.[k.toLowerCase()] || null },
|
|
61
|
-
text: async () =>
|
|
62
|
-
body: (()
|
|
63
|
-
const b = bb.result.body || "";
|
|
64
|
-
const str = typeof b === "string" ? b : JSON.stringify(b);
|
|
65
|
-
const isSSE = bb.result.headers?.["Content-Type"]?.includes("text/event-stream");
|
|
66
|
-
if (isSSE) {
|
|
67
|
-
return (async function* () { yield Buffer.from(str); })();
|
|
68
|
-
}
|
|
69
|
-
return null;
|
|
70
|
-
})(),
|
|
59
|
+
text: async () => str,
|
|
60
|
+
body: isSSE ? (async function* () { yield Buffer.from(str); })() : null,
|
|
71
61
|
};
|
|
72
|
-
|
|
73
|
-
if (bbLocalFallback?.fallback) evt("fallback-notice", { reqId: handlerCtx.reqId, requested, actual: model, reason: bbLocalFallback.reason, notice: bbLocalFallback.notice, via: "broadband" });
|
|
74
|
-
evt("relay-start", { reqId: handlerCtx.reqId, model, via: "broadband-local", target: bb.target, group: bb.group, fallback: bbLocalFallback });
|
|
75
|
-
const out = await relay(res, fakeRes, body, {
|
|
76
|
-
fallback: bbLocalFallback,
|
|
77
|
-
onFirstChunk: (d) => mark(`ttf-bb-${model}`),
|
|
78
|
-
onDownstreamAbort: () => evt("client-abort", { reqId: handlerCtx.reqId, model, totalMs: Math.round(Date.now() - startedAt), stages: [...stages] }),
|
|
79
|
-
});
|
|
80
|
-
evt("relay-done", { reqId: handlerCtx.reqId, model, via: "broadband-local", status: out.status, ttfMs: out.ttfMs, totalMs: out.totalMs, aborted: out.aborted, interrupted: out.interrupted ?? false, detail: out.detail ?? null });
|
|
81
|
-
evt("result", { model, status: out.status, via: "broadband", timing: null, ttfMs: out.ttfMs, totalMs: out.totalMs, detail: out.detail ?? null, fallback: bbLocalFallback, requested, actual: model });
|
|
82
|
-
evt("client-response", { requested, actual: model, via: "broadband", fallback: bbLocalFallback, status: out.status, reqId: handlerCtx.reqId });
|
|
83
|
-
if (plugins?.length) runHook(plugins, "request:completed", { reqId: handlerCtx.reqId, requested, useAuto, hops: handlerCtx.hops, stream: Boolean(body.stream), durationMs: Date.now() - startedAt, via: "broadband", status: out.status, actual: model, fallback: bbLocalFallback }).catch(() => {});
|
|
62
|
+
await pipeline.execute({ res, upRes: fakeRes, body, requested, actual: model, lastErr, via: "broadband-local", lockModel, useAuto, handlerCtx, mark, perf0, stages, startedAt });
|
|
84
63
|
return { handled: true };
|
|
85
64
|
}
|
|
86
65
|
evt("relay-miss", { reqId: handlerCtx.reqId, model });
|
|
@@ -1,6 +1,6 @@
|
|
|
1
|
-
import { performance } from "node:perf_hooks";
|
|
2
|
-
import { json } from "../helpers.js";
|
|
3
1
|
import { relay } from "../stream.js";
|
|
2
|
+
import { json } from "../helpers.js";
|
|
3
|
+
import { performance } from "node:perf_hooks";
|
|
4
4
|
|
|
5
5
|
export async function handleExhaustedLocal({ res, body, lastErr, order, handlerCtx, evt, logCall, mark, perf0, stages, done, requested, useAuto }) {
|
|
6
6
|
const model = lastErr?.model ?? handlerCtx.model;
|
|
@@ -1,8 +1,9 @@
|
|
|
1
|
-
import { buildFallbackInfo } from "../fallback.js";
|
|
2
1
|
import { relay, SLOW_TOTAL_MS, STREAM_TIMEOUT_MS, STALL_TIMEOUT_MS, SCORE_STALL_MS } from "../stream.js";
|
|
2
|
+
import { buildFallbackInfo } from "../fallback.js";
|
|
3
|
+
import { createRelayPipeline } from "./relay-pipeline.js";
|
|
3
4
|
import { hedgedFirstChunkRace } from "../hedge.js";
|
|
4
|
-
import { runHook } from "../../plugins.js";
|
|
5
5
|
|
|
6
|
+
/** 薄适配:hedge 赛跑后按 winner 调 pipeline(复用 relay-pipeline 深模块) */
|
|
6
7
|
export async function handleHedge({
|
|
7
8
|
upRes,
|
|
8
9
|
model,
|
|
@@ -27,104 +28,86 @@ export async function handleHedge({
|
|
|
27
28
|
res,
|
|
28
29
|
hedgeDelayMs,
|
|
29
30
|
}) {
|
|
30
|
-
const isStream = Boolean(body.stream);
|
|
31
31
|
const d = hedgeDelayMs;
|
|
32
|
-
// hedge 已在外层判断 doHedge,这里直接执行赛跑
|
|
33
32
|
try {
|
|
34
33
|
const hedged = await hedgedFirstChunkRace({ localUpRes: upRes, peers, handlerCtx, hedgeDelayMs: d, evt });
|
|
35
34
|
if (hedged && hedged.winner) {
|
|
36
35
|
if (hedged.winner === "local") {
|
|
37
|
-
const fallback = buildFallbackInfo({ requested, actual: model, lastErr, via: "local", useAuto, lockModel });
|
|
38
|
-
if (fallback?.fallback) evt("fallback-notice", { reqId: handlerCtx.reqId, requested, actual: model, reason: fallback.reason, notice: fallback.notice, via: "local" });
|
|
39
|
-
evt("relay-start", { reqId: handlerCtx.reqId, model, via: "local", isStream, fallback, hedged: true });
|
|
40
36
|
const bufferedUpRes = { ...upRes, body: hedged.bufferedBody, headers: upRes.headers, status: upRes.status, _t: upRes._t };
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
37
|
+
const pipeline = createRelayPipeline({
|
|
38
|
+
relay,
|
|
39
|
+
buildFallbackInfo,
|
|
40
|
+
auto,
|
|
41
|
+
plugins,
|
|
42
|
+
evt,
|
|
43
|
+
mark,
|
|
44
|
+
logCall,
|
|
45
|
+
logError,
|
|
46
|
+
constants: { STREAM_TIMEOUT_MS, SLOW_TOTAL_MS, STALL_TIMEOUT_MS, SCORE_STALL_MS },
|
|
47
|
+
startedAt,
|
|
48
|
+
stages,
|
|
49
|
+
});
|
|
50
|
+
const r = await pipeline.execute({
|
|
51
|
+
res,
|
|
52
|
+
upRes: bufferedUpRes,
|
|
53
|
+
body,
|
|
54
|
+
requested,
|
|
55
|
+
actual: model,
|
|
56
|
+
lastErr,
|
|
57
|
+
via: "local",
|
|
58
|
+
lockModel,
|
|
59
|
+
useAuto,
|
|
60
|
+
handlerCtx,
|
|
61
|
+
mark,
|
|
62
|
+
perf0,
|
|
63
|
+
stages,
|
|
64
|
+
startedAt,
|
|
52
65
|
});
|
|
53
|
-
|
|
54
|
-
if (
|
|
55
|
-
if (auto) await auto.recordError(model, { status: 502, slow: true, note: `stream timeout ${STREAM_TIMEOUT_MS}ms` });
|
|
56
|
-
const err = { model, upstream: null, status: 502, message: `stream timed out after ${STREAM_TIMEOUT_MS}ms` };
|
|
57
|
-
logError(model, 502, `stream timeout ${STREAM_TIMEOUT_MS}ms`);
|
|
58
|
-
evt("upstream-error", { reqId: handlerCtx.reqId, model, status: 502, message: "stream timeout", timing: null });
|
|
59
|
-
evt("fallback", { reqId: handlerCtx.reqId, from: model, to: order[idx + 1] ?? null, reason: "stream timeout" });
|
|
60
|
-
return { handled: false, upRes: null, lastErr: err };
|
|
61
|
-
}
|
|
62
|
-
if (out.interrupted) {
|
|
63
|
-
if (auto) {
|
|
64
|
-
await auto.recordError(model, { status: 200, slow: true, note: `stall ${STALL_TIMEOUT_MS}ms` });
|
|
65
|
-
await auto.recordLatency(model, out.totalMs ?? (Date.now() - startedAt));
|
|
66
|
-
}
|
|
67
|
-
evt("slow-model", { model, elapsedMs: out.totalMs ?? (Date.now() - startedAt), threshold: STALL_TIMEOUT_MS, interrupted: true, detail: out.detail ?? null });
|
|
68
|
-
logCall(model, 200);
|
|
69
|
-
evt("result", { model, status: out.status, via: "local", timing: upRes._t ?? null, ttfMs: out.ttfMs, totalMs: out.totalMs, interrupted: true, detail: out.detail ?? null, fallback, requested, actual: model });
|
|
70
|
-
evt("client-response", { requested, actual: model, via: "local", fallback, status: out.status, reqId: handlerCtx.reqId });
|
|
71
|
-
if (plugins?.length) runHook(plugins, "request:completed", { reqId: handlerCtx.reqId, requested, useAuto, hops: handlerCtx.hops, stream: isStream, durationMs: Date.now() - startedAt, via: "local", status: out.status, actual: model, interrupted: true, fallback }).catch(() => {});
|
|
72
|
-
return { handled: true };
|
|
73
|
-
}
|
|
74
|
-
const elapsed = Date.now() - startedAt;
|
|
75
|
-
const latencyMs = out.totalMs ?? elapsed;
|
|
76
|
-
let scoredSlow = false;
|
|
77
|
-
if (SLOW_TOTAL_MS && auto && elapsed > SLOW_TOTAL_MS && out.status === 200) {
|
|
78
|
-
void auto.recordError(model, { status: 200, slow: true, note: `slow ${elapsed}ms` });
|
|
79
|
-
void auto.recordLatency(model, latencyMs);
|
|
80
|
-
evt("slow-model", { model, elapsedMs: elapsed, threshold: SLOW_TOTAL_MS, reason: "total", detail: out.detail ?? null });
|
|
81
|
-
scoredSlow = true;
|
|
82
|
-
}
|
|
83
|
-
if (out.detail?.stallHits > 0 && auto && out.status === 200) {
|
|
84
|
-
void auto.recordError(model, { status: 200, slow: true, note: `stall ${out.detail.stallHits}x gap>${SCORE_STALL_MS}ms maxGap ${out.detail.maxGapMs}ms` });
|
|
85
|
-
void auto.recordLatency(model, latencyMs);
|
|
86
|
-
evt("slow-model", { model, elapsedMs: elapsed, threshold: SCORE_STALL_MS, reason: "stall", stallHits: out.detail.stallHits, maxGapMs: out.detail.maxGapMs, detail: out.detail ?? null });
|
|
87
|
-
scoredSlow = true;
|
|
88
|
-
}
|
|
89
|
-
if (!scoredSlow && auto && out.status === 200) await auto.recordOk(model, { latencyMs });
|
|
90
|
-
else if (!scoredSlow && auto) await auto.recordLatency(model, latencyMs);
|
|
91
|
-
evt("result", { model, status: out.status, via: "local", timing: upRes._t ?? null, ttfMs: out.ttfMs, totalMs: out.totalMs, detail: out.detail ?? null, fallback, requested, actual: model, hedged: true });
|
|
92
|
-
evt("client-response", { requested, actual: model, via: "local", fallback, status: out.status, reqId: handlerCtx.reqId });
|
|
93
|
-
if (plugins?.length) runHook(plugins, "request:completed", { reqId: handlerCtx.reqId, requested, useAuto, hops: handlerCtx.hops, stream: isStream, durationMs: Date.now() - startedAt, via: "local", status: out.status, actual: model, fallback }).catch(() => {});
|
|
66
|
+
// 保留 hedged 语义:若超时则透传 needs fallback
|
|
67
|
+
if (!r.handled) return { handled: false, upRes: null, lastErr: r.lastErr };
|
|
94
68
|
return { handled: true };
|
|
95
|
-
}
|
|
69
|
+
}
|
|
70
|
+
if (hedged.winner === "peer" && hedged.peerInfo) {
|
|
96
71
|
const win = hedged.peerInfo;
|
|
97
72
|
evt("peer-race-win", { reqId: handlerCtx.reqId, model, winPeer: win.peer.url, winTarget: win.target, latencyMs: win.latencyMs, hedged: true, ttfMs: hedged.ttfMs });
|
|
98
73
|
await peers.recordResult(win.peer.url, { ok: true, latencyMs: win.latencyMs, model: win.target });
|
|
99
|
-
logCall(win.target, win.res.status);
|
|
100
|
-
const peerFallback = buildFallbackInfo({ requested, actual: win.target, lastErr, via: "peer", useAuto, lockModel });
|
|
101
|
-
if (peerFallback?.fallback) evt("fallback-notice", { reqId: handlerCtx.reqId, requested, actual: win.target, reason: peerFallback.reason, notice: peerFallback.notice, via: "peer" });
|
|
102
|
-
evt("relay-start", { reqId: handlerCtx.reqId, model: win.target, via: "peer", isStream, fallback: peerFallback, hedged: true });
|
|
103
74
|
const bufferedPeerRes = { ...win.res, body: hedged.bufferedBody, headers: win.res.headers, status: win.res.status, _t: win.res._t };
|
|
104
|
-
const
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
|
|
75
|
+
const pipeline = createRelayPipeline({
|
|
76
|
+
relay,
|
|
77
|
+
buildFallbackInfo,
|
|
78
|
+
auto,
|
|
79
|
+
plugins,
|
|
80
|
+
evt,
|
|
81
|
+
mark,
|
|
82
|
+
logCall,
|
|
83
|
+
logError: () => {},
|
|
84
|
+
constants: { STREAM_TIMEOUT_MS, SLOW_TOTAL_MS, STALL_TIMEOUT_MS, SCORE_STALL_MS },
|
|
85
|
+
startedAt,
|
|
86
|
+
stages,
|
|
87
|
+
});
|
|
88
|
+
await pipeline.execute({
|
|
89
|
+
res,
|
|
90
|
+
upRes: bufferedPeerRes,
|
|
91
|
+
body,
|
|
92
|
+
requested,
|
|
93
|
+
actual: win.target,
|
|
94
|
+
lastErr,
|
|
95
|
+
via: "peer",
|
|
96
|
+
lockModel,
|
|
97
|
+
useAuto,
|
|
98
|
+
handlerCtx: { ...handlerCtx, model: win.target },
|
|
99
|
+
mark: (n) => mark(n),
|
|
100
|
+
perf0,
|
|
101
|
+
stages,
|
|
102
|
+
startedAt,
|
|
108
103
|
});
|
|
109
|
-
evt("relay-done", { reqId: handlerCtx.reqId, model: win.target, via: "peer", status: out.status, ttfMs: out.ttfMs ?? hedged.ttfMs, totalMs: out.totalMs, aborted: out.aborted, interrupted: out.interrupted ?? false, detail: out.detail ?? null, hedged: true });
|
|
110
|
-
if (auto && out.status === 200) {
|
|
111
|
-
const latencyMs = out.totalMs ?? win.latencyMs;
|
|
112
|
-
if (out.detail?.stallHits > 0 || (latencyMs && latencyMs > SLOW_TOTAL_MS)) {
|
|
113
|
-
void auto.recordError(win.target, { status: 200, slow: true, note: `peer slow ${latencyMs}ms` });
|
|
114
|
-
void auto.recordLatency(win.target, latencyMs);
|
|
115
|
-
} else {
|
|
116
|
-
await auto.recordOk(win.target, { latencyMs });
|
|
117
|
-
}
|
|
118
|
-
}
|
|
119
|
-
evt("result", { model: win.target, status: out.status, via: "peer", timing: win.res._t ?? null, ttfMs: out.ttfMs, totalMs: out.totalMs, detail: out.detail ?? null, fallback: peerFallback, requested, actual: win.target, hedged: true });
|
|
120
|
-
evt("client-response", { requested, actual: win.target, via: "peer", fallback: peerFallback, status: out.status, reqId: handlerCtx.reqId });
|
|
121
|
-
if (plugins?.length) runHook(plugins, "request:completed", { reqId: handlerCtx.reqId, requested, useAuto, hops: handlerCtx.hops, stream: isStream, durationMs: Date.now() - startedAt, via: "peer", status: out.status, actual: win.target, fallback: peerFallback }).catch(() => {});
|
|
122
104
|
return { handled: true };
|
|
123
105
|
}
|
|
124
106
|
}
|
|
125
107
|
if (hedged && hedged.needsPeer) {
|
|
126
108
|
return { handled: false, upRes: null, lastErr, needsPeer: true };
|
|
127
|
-
}
|
|
109
|
+
}
|
|
110
|
+
if (!hedged || !hedged.winner) {
|
|
128
111
|
evt("hedge-both-fail", { reqId: handlerCtx.reqId, model });
|
|
129
112
|
if (!hedged) {
|
|
130
113
|
if (auto) await auto.recordError(model, { status: 502, slow: false, note: "hedge both fail" });
|
|
@@ -136,7 +119,6 @@ export async function handleHedge({
|
|
|
136
119
|
return { handled: false, upRes: null, lastErr };
|
|
137
120
|
} catch (hedgeErr) {
|
|
138
121
|
evt("hedge-error", { reqId: handlerCtx.reqId, model, error: String(hedgeErr?.message || hedgeErr).slice(0, 300) });
|
|
139
|
-
// 回退到串行
|
|
140
122
|
return { handled: false, upRes, lastErr };
|
|
141
123
|
}
|
|
142
124
|
}
|
|
@@ -1,8 +1,8 @@
|
|
|
1
|
-
import { buildFallbackInfo } from "../fallback.js";
|
|
2
1
|
import { relay, SLOW_TOTAL_MS, STREAM_TIMEOUT_MS, STALL_TIMEOUT_MS, SCORE_STALL_MS } from "../stream.js";
|
|
3
|
-
import {
|
|
4
|
-
import {
|
|
2
|
+
import { buildFallbackInfo } from "../fallback.js";
|
|
3
|
+
import { createRelayPipeline } from "./relay-pipeline.js";
|
|
5
4
|
|
|
5
|
+
/** 薄适配:本地透传唯一经由 relay-pipeline */
|
|
6
6
|
export async function handleLocalRelay({
|
|
7
7
|
upRes,
|
|
8
8
|
model,
|
|
@@ -25,64 +25,35 @@ export async function handleLocalRelay({
|
|
|
25
25
|
plugins,
|
|
26
26
|
res,
|
|
27
27
|
}) {
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
28
|
+
const pipeline = createRelayPipeline({
|
|
29
|
+
relay,
|
|
30
|
+
buildFallbackInfo,
|
|
31
|
+
auto,
|
|
32
|
+
plugins,
|
|
33
|
+
evt,
|
|
34
|
+
mark,
|
|
35
|
+
logCall,
|
|
36
|
+
logError,
|
|
37
|
+
constants: { STREAM_TIMEOUT_MS, SLOW_TOTAL_MS, STALL_TIMEOUT_MS, SCORE_STALL_MS },
|
|
38
|
+
startedAt,
|
|
39
|
+
stages,
|
|
40
|
+
});
|
|
41
|
+
const r = await pipeline.execute({
|
|
42
|
+
res,
|
|
43
|
+
upRes,
|
|
44
|
+
body,
|
|
45
|
+
requested,
|
|
46
|
+
actual: model,
|
|
47
|
+
lastErr,
|
|
48
|
+
via: "local",
|
|
49
|
+
lockModel,
|
|
50
|
+
useAuto,
|
|
51
|
+
handlerCtx,
|
|
52
|
+
mark,
|
|
53
|
+
perf0,
|
|
54
|
+
stages,
|
|
55
|
+
startedAt,
|
|
42
56
|
});
|
|
43
|
-
|
|
44
|
-
if (out.status === STREAM_TIMEOUT_MS) {
|
|
45
|
-
if (auto) await auto.recordError(model, { status: 502, slow: true, note: `stream timeout ${STREAM_TIMEOUT_MS}ms` });
|
|
46
|
-
const err = { model, upstream: null, status: 502, message: `stream timed out after ${STREAM_TIMEOUT_MS}ms` };
|
|
47
|
-
logError(model, 502, `stream timeout ${STREAM_TIMEOUT_MS}ms`);
|
|
48
|
-
evt("upstream-error", { reqId: handlerCtx.reqId, model, status: 502, message: "stream timeout", timing: null });
|
|
49
|
-
evt("fallback", { reqId: handlerCtx.reqId, from: model, to: order[idx + 1] ?? null, reason: "stream timeout" });
|
|
50
|
-
return { handled: false, upRes: null, lastErr: err };
|
|
51
|
-
}
|
|
52
|
-
if (out.interrupted) {
|
|
53
|
-
if (auto) {
|
|
54
|
-
await auto.recordError(model, { status: 200, slow: true, note: `stall ${STALL_TIMEOUT_MS}ms` });
|
|
55
|
-
await auto.recordLatency(model, out.totalMs ?? (Date.now() - startedAt));
|
|
56
|
-
}
|
|
57
|
-
evt("slow-model", { model, elapsedMs: out.totalMs ?? (Date.now() - startedAt), threshold: STALL_TIMEOUT_MS, interrupted: true, detail: out.detail ?? null });
|
|
58
|
-
logCall(model, 200);
|
|
59
|
-
evt("result", { model, status: out.status, via: "local", timing: upRes._t ?? null, ttfMs: out.ttfMs, totalMs: out.totalMs, interrupted: true, detail: out.detail ?? null, fallback, requested, actual: model });
|
|
60
|
-
evt("client-response", { requested, actual: model, via: "local", fallback, status: out.status, interrupted: true, reqId: handlerCtx.reqId });
|
|
61
|
-
if (plugins?.length) runHook(plugins, "request:completed", { reqId: handlerCtx.reqId, requested, useAuto, hops: handlerCtx.hops, stream: Boolean(body.stream), durationMs: Date.now() - startedAt, via: "local", status: out.status, actual: model, interrupted: true, fallback }).catch(() => {});
|
|
62
|
-
return { handled: true };
|
|
63
|
-
}
|
|
64
|
-
const elapsed = Date.now() - startedAt;
|
|
65
|
-
const latencyMs = out.totalMs ?? elapsed;
|
|
66
|
-
let scoredSlow = false;
|
|
67
|
-
if (SLOW_TOTAL_MS && auto && elapsed > SLOW_TOTAL_MS && out.status === 200) {
|
|
68
|
-
void auto.recordError(model, { status: 200, slow: true, note: `slow ${elapsed}ms` });
|
|
69
|
-
void auto.recordLatency(model, latencyMs);
|
|
70
|
-
evt("slow-model", { model, elapsedMs: elapsed, threshold: SLOW_TOTAL_MS, reason: "total", detail: out.detail ?? null });
|
|
71
|
-
scoredSlow = true;
|
|
72
|
-
}
|
|
73
|
-
if (out.detail?.stallHits > 0 && auto && out.status === 200) {
|
|
74
|
-
void auto.recordError(model, { status: 200, slow: true, note: `stall ${out.detail.stallHits}x gap>${SCORE_STALL_MS}ms maxGap ${out.detail.maxGapMs}ms` });
|
|
75
|
-
void auto.recordLatency(model, latencyMs);
|
|
76
|
-
evt("slow-model", { model, elapsedMs: elapsed, threshold: SCORE_STALL_MS, reason: "stall", stallHits: out.detail.stallHits, maxGapMs: out.detail.maxGapMs, detail: out.detail ?? null });
|
|
77
|
-
scoredSlow = true;
|
|
78
|
-
}
|
|
79
|
-
if (!scoredSlow && auto && out.status === 200) {
|
|
80
|
-
await auto.recordOk(model, { latencyMs });
|
|
81
|
-
} else if (!scoredSlow && auto) {
|
|
82
|
-
await auto.recordLatency(model, latencyMs);
|
|
83
|
-
}
|
|
84
|
-
evt("result", { model, status: out.status, via: "local", timing: upRes._t ?? null, ttfMs: out.ttfMs, totalMs: out.totalMs, detail: out.detail ?? null, fallback, requested, actual: model });
|
|
85
|
-
evt("client-response", { requested, actual: model, via: "local", fallback, status: out.status, reqId: handlerCtx.reqId });
|
|
86
|
-
if (plugins?.length) runHook(plugins, "request:completed", { reqId: handlerCtx.reqId, requested, useAuto, hops: handlerCtx.hops, stream: Boolean(body.stream), durationMs: Date.now() - startedAt, via: "local", status: out.status, actual: model, fallback }).catch(() => {});
|
|
57
|
+
if (!r.handled) return { handled: false, upRes: null, lastErr: r.lastErr };
|
|
87
58
|
return { handled: true };
|
|
88
59
|
}
|
|
@@ -1,8 +1,9 @@
|
|
|
1
|
+
import { relay, SLOW_TOTAL_MS, STREAM_TIMEOUT_MS, STALL_TIMEOUT_MS, SCORE_STALL_MS } from "../stream.js";
|
|
1
2
|
import { buildFallbackInfo } from "../fallback.js";
|
|
2
|
-
import {
|
|
3
|
+
import { createRelayPipeline } from "./relay-pipeline.js";
|
|
3
4
|
import { racePeerCandidates } from "../peers.js";
|
|
4
|
-
import { runHook } from "../../plugins.js";
|
|
5
5
|
|
|
6
|
+
/** 薄适配:peer 赛跑后经由 pipeline(via=peer) */
|
|
6
7
|
export async function handlePeerRelay({
|
|
7
8
|
model,
|
|
8
9
|
body,
|
|
@@ -32,27 +33,35 @@ export async function handlePeerRelay({
|
|
|
32
33
|
}
|
|
33
34
|
evt("peer-race-win", { reqId: handlerCtx.reqId, model, winPeer: win.peer.url, winTarget: win.target, latencyMs: win.latencyMs });
|
|
34
35
|
await peers.recordResult(win.peer.url, { ok: true, latencyMs: win.latencyMs, model: win.target });
|
|
35
|
-
|
|
36
|
-
const
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
36
|
+
|
|
37
|
+
const pipeline = createRelayPipeline({
|
|
38
|
+
relay,
|
|
39
|
+
buildFallbackInfo,
|
|
40
|
+
auto,
|
|
41
|
+
plugins,
|
|
42
|
+
evt,
|
|
43
|
+
mark,
|
|
44
|
+
logCall,
|
|
45
|
+
logError: () => {},
|
|
46
|
+
constants: { STREAM_TIMEOUT_MS, SLOW_TOTAL_MS, STALL_TIMEOUT_MS, SCORE_STALL_MS },
|
|
47
|
+
startedAt,
|
|
48
|
+
stages,
|
|
49
|
+
});
|
|
50
|
+
await pipeline.execute({
|
|
51
|
+
res,
|
|
52
|
+
upRes: win.res,
|
|
53
|
+
body,
|
|
54
|
+
requested,
|
|
55
|
+
actual: win.target,
|
|
56
|
+
lastErr,
|
|
57
|
+
via: "peer",
|
|
58
|
+
lockModel,
|
|
59
|
+
useAuto,
|
|
60
|
+
handlerCtx: { ...handlerCtx, model: win.target },
|
|
61
|
+
mark,
|
|
62
|
+
perf0,
|
|
63
|
+
stages,
|
|
64
|
+
startedAt,
|
|
43
65
|
});
|
|
44
|
-
evt("relay-done", { reqId: handlerCtx.reqId, model: win.target, via: "peer", status: out.status, ttfMs: out.ttfMs, totalMs: out.totalMs, aborted: out.aborted, interrupted: out.interrupted ?? false, detail: out.detail ?? null });
|
|
45
|
-
if (auto && out.status === 200) {
|
|
46
|
-
const latencyMs = out.totalMs ?? win.latencyMs;
|
|
47
|
-
if (out.detail?.stallHits > 0 || (latencyMs && latencyMs > SLOW_TOTAL_MS)) {
|
|
48
|
-
void auto.recordError(win.target, { status: 200, slow: true, note: `peer slow ${latencyMs}ms` });
|
|
49
|
-
void auto.recordLatency(win.target, latencyMs);
|
|
50
|
-
} else {
|
|
51
|
-
await auto.recordOk(win.target, { latencyMs });
|
|
52
|
-
}
|
|
53
|
-
}
|
|
54
|
-
evt("result", { model: win.target, status: out.status, via: "peer", timing: win.res._t ?? null, ttfMs: out.ttfMs, totalMs: out.totalMs, detail: out.detail ?? null, fallback: peerFallback, requested, actual: win.target });
|
|
55
|
-
evt("client-response", { requested, actual: win.target, via: "peer", fallback: peerFallback, status: out.status, reqId: handlerCtx.reqId });
|
|
56
|
-
if (plugins?.length) runHook(plugins, "request:completed", { reqId: handlerCtx.reqId, requested, useAuto, hops: handlerCtx.hops, stream: Boolean(body.stream), durationMs: Date.now() - startedAt, via: "peer", status: out.status, actual: win.target, fallback: peerFallback }).catch(() => {});
|
|
57
66
|
return { handled: true };
|
|
58
67
|
}
|
|
@@ -0,0 +1,151 @@
|
|
|
1
|
+
import { runHook } from "../../plugins.js";
|
|
2
|
+
|
|
3
|
+
/**
|
|
4
|
+
* RelayPipeline 深模块
|
|
5
|
+
* 把 5 个 handler 各自的 fallback→relay→scoring→事件 6段流水收敛为单一真相。
|
|
6
|
+
* 对外 1 接口:createRelayPipeline(deps) => { execute(ctx) }
|
|
7
|
+
* 设计要点:全部外部可注入,便于在 pipeline seam 上做行为测试;constants 注入便于单测加速。
|
|
8
|
+
*/
|
|
9
|
+
export function createRelayPipeline({
|
|
10
|
+
relay,
|
|
11
|
+
buildFallbackInfo,
|
|
12
|
+
auto,
|
|
13
|
+
plugins,
|
|
14
|
+
evt,
|
|
15
|
+
mark,
|
|
16
|
+
logCall,
|
|
17
|
+
logError,
|
|
18
|
+
constants,
|
|
19
|
+
startedAt: defaultStartedAt,
|
|
20
|
+
stages: defaultStages,
|
|
21
|
+
perfNow,
|
|
22
|
+
} = {}) {
|
|
23
|
+
const C = {
|
|
24
|
+
STREAM_TIMEOUT_MS: 25_000,
|
|
25
|
+
SLOW_TOTAL_MS: 20_000,
|
|
26
|
+
STALL_TIMEOUT_MS: 0,
|
|
27
|
+
SCORE_STALL_MS: 15_000,
|
|
28
|
+
...(constants || {}),
|
|
29
|
+
};
|
|
30
|
+
const _relay = relay;
|
|
31
|
+
const _build = buildFallbackInfo;
|
|
32
|
+
const _evt = evt || (() => {});
|
|
33
|
+
const _mark = mark || (() => {});
|
|
34
|
+
const _logCall = logCall || (() => {});
|
|
35
|
+
const _logError = logError || (() => {});
|
|
36
|
+
const _perfNow = perfNow || (() => Date.now());
|
|
37
|
+
|
|
38
|
+
async function execute({
|
|
39
|
+
res,
|
|
40
|
+
upRes,
|
|
41
|
+
body,
|
|
42
|
+
requested,
|
|
43
|
+
actual,
|
|
44
|
+
lastErr,
|
|
45
|
+
via,
|
|
46
|
+
lockModel,
|
|
47
|
+
useAuto,
|
|
48
|
+
handlerCtx,
|
|
49
|
+
mark: m2,
|
|
50
|
+
perf0,
|
|
51
|
+
stages: s2,
|
|
52
|
+
startedAt: sa2,
|
|
53
|
+
} = {}) {
|
|
54
|
+
const markFn = m2 || _mark;
|
|
55
|
+
const curStartedAt = sa2 ?? defaultStartedAt ?? Date.now();
|
|
56
|
+
const curStages = s2 ?? defaultStages ?? [];
|
|
57
|
+
const reqId = handlerCtx?.reqId;
|
|
58
|
+
const hops = handlerCtx?.hops;
|
|
59
|
+
|
|
60
|
+
// 1. logCall(pre) — 保持原 handler 的 logCall→fallback→relay-start 时序
|
|
61
|
+
try { _logCall(actual, upRes?.status); } catch {}
|
|
62
|
+
// 2. fallback + relay-start
|
|
63
|
+
let fallback = null;
|
|
64
|
+
try {
|
|
65
|
+
if (_build) fallback = _build({ requested, actual, lastErr, via, useAuto, lockModel });
|
|
66
|
+
} catch {}
|
|
67
|
+
if (fallback?.fallback) {
|
|
68
|
+
_evt("fallback-notice", { reqId, requested, actual, reason: fallback.reason, notice: fallback.notice, via, fallback: true });
|
|
69
|
+
}
|
|
70
|
+
_evt("relay-start", { reqId, model: actual, via, isStream: Boolean(body?.stream), fallback });
|
|
71
|
+
|
|
72
|
+
// 3. relay
|
|
73
|
+
const out = await _relay(res, upRes, body, {
|
|
74
|
+
fallback,
|
|
75
|
+
onFirstChunk: (delta) => {
|
|
76
|
+
try { markFn(`ttf-${actual}`); } catch {}
|
|
77
|
+
_evt("relay-first-chunk", { reqId, model: actual, ttfMs: delta, via });
|
|
78
|
+
if (plugins?.length) runHook(plugins, "relay:first-chunk", { reqId, requested, model: actual, via, ttfMs: delta }).catch(() => {});
|
|
79
|
+
},
|
|
80
|
+
onDownstreamAbort: () => {
|
|
81
|
+
_evt("client-abort", { reqId, model: actual, totalMs: Math.round(_perfNow() - (perf0 ?? 0)), stages: [...curStages] });
|
|
82
|
+
},
|
|
83
|
+
});
|
|
84
|
+
|
|
85
|
+
// 4. relay-done
|
|
86
|
+
_evt("relay-done", {
|
|
87
|
+
reqId,
|
|
88
|
+
model: actual,
|
|
89
|
+
via,
|
|
90
|
+
status: out.status,
|
|
91
|
+
ttfMs: out.ttfMs,
|
|
92
|
+
totalMs: out.totalMs,
|
|
93
|
+
aborted: out.aborted,
|
|
94
|
+
interrupted: out.interrupted ?? false,
|
|
95
|
+
detail: out.detail ?? null,
|
|
96
|
+
});
|
|
97
|
+
|
|
98
|
+
// 5a. 首块超时未写字节 → 回退
|
|
99
|
+
if (out.status === C.STREAM_TIMEOUT_MS) {
|
|
100
|
+
if (auto) try { await auto.recordError(actual, { status: 502, slow: true, note: `stream timeout ${C.STREAM_TIMEOUT_MS}ms` }); } catch {}
|
|
101
|
+
try { _logError(actual, 502, `stream timeout ${C.STREAM_TIMEOUT_MS}ms`); } catch {}
|
|
102
|
+
_evt("upstream-error", { reqId, model: actual, status: 502, message: "stream timeout", timing: null });
|
|
103
|
+
_evt("fallback", { reqId, from: actual, to: null, reason: "stream timeout" });
|
|
104
|
+
return { handled: false, upRes: null, lastErr: { model: actual, upstream: null, status: 502, message: `stream timed out after ${C.STREAM_TIMEOUT_MS}ms` } };
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
// 5b. 中断(stall 超时 / max 流时长)
|
|
108
|
+
if (out.interrupted) {
|
|
109
|
+
if (auto) {
|
|
110
|
+
try { await auto.recordError(actual, { status: 200, slow: true, note: `stall ${C.STALL_TIMEOUT_MS}ms` }); } catch {}
|
|
111
|
+
try { await auto.recordLatency(actual, out.totalMs ?? (Date.now() - curStartedAt)); } catch {}
|
|
112
|
+
}
|
|
113
|
+
_evt("slow-model", { reqId, model: actual, elapsedMs: out.totalMs ?? (Date.now() - curStartedAt), threshold: C.STALL_TIMEOUT_MS, interrupted: true, detail: out.detail ?? null });
|
|
114
|
+
try { _logCall(actual, 200); } catch {}
|
|
115
|
+
_evt("result", { reqId, model: actual, status: out.status, via, timing: upRes?._t ?? null, ttfMs: out.ttfMs, totalMs: out.totalMs, interrupted: true, detail: out.detail ?? null, fallback, requested, actual });
|
|
116
|
+
_evt("client-response", { requested, actual, via, fallback, status: out.status, reqId, interrupted: true });
|
|
117
|
+
if (plugins?.length) runHook(plugins, "request:completed", { reqId, requested, useAuto, hops, stream: Boolean(body?.stream), durationMs: Date.now() - curStartedAt, via, status: out.status, actual, interrupted: true, fallback }).catch(() => {});
|
|
118
|
+
return { handled: true };
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
// 5c. 慢速计分 + ok
|
|
122
|
+
const elapsed = Date.now() - curStartedAt;
|
|
123
|
+
const latencyMs = out.totalMs ?? elapsed;
|
|
124
|
+
let scoredSlow = false;
|
|
125
|
+
|
|
126
|
+
if (C.SLOW_TOTAL_MS && auto && elapsed > C.SLOW_TOTAL_MS && out.status === 200) {
|
|
127
|
+
try { await auto.recordError(actual, { status: 200, slow: true, note: `slow ${elapsed}ms` }); } catch {}
|
|
128
|
+
try { await auto.recordLatency(actual, latencyMs); } catch {}
|
|
129
|
+
_evt("slow-model", { reqId, model: actual, elapsedMs: elapsed, threshold: C.SLOW_TOTAL_MS, reason: "total", detail: out.detail ?? null });
|
|
130
|
+
scoredSlow = true;
|
|
131
|
+
}
|
|
132
|
+
if (out.detail?.stallHits > 0 && auto && out.status === 200) {
|
|
133
|
+
try { await auto.recordError(actual, { status: 200, slow: true, note: `stall ${out.detail.stallHits}x gap>${C.SCORE_STALL_MS}ms maxGap ${out.detail.maxGapMs}ms` }); } catch {}
|
|
134
|
+
try { await auto.recordLatency(actual, latencyMs); } catch {}
|
|
135
|
+
_evt("slow-model", { reqId, model: actual, elapsedMs: elapsed, threshold: C.SCORE_STALL_MS, reason: "stall", stallHits: out.detail.stallHits, maxGapMs: out.detail.maxGapMs, detail: out.detail ?? null });
|
|
136
|
+
scoredSlow = true;
|
|
137
|
+
}
|
|
138
|
+
if (!scoredSlow && auto && out.status === 200) {
|
|
139
|
+
try { await auto.recordOk(actual, { latencyMs }); } catch {}
|
|
140
|
+
} else if (!scoredSlow && auto) {
|
|
141
|
+
try { await auto.recordLatency(actual, latencyMs); } catch {}
|
|
142
|
+
}
|
|
143
|
+
|
|
144
|
+
_evt("result", { reqId, model: actual, status: out.status, via, timing: upRes?._t ?? null, ttfMs: out.ttfMs, totalMs: out.totalMs, detail: out.detail ?? null, fallback, requested, actual });
|
|
145
|
+
_evt("client-response", { requested, actual, via, fallback, status: out.status, reqId });
|
|
146
|
+
if (plugins?.length) runHook(plugins, "request:completed", { reqId, requested, useAuto, hops, stream: Boolean(body?.stream), durationMs: Date.now() - curStartedAt, via, status: out.status, actual, fallback }).catch(() => {});
|
|
147
|
+
return { handled: true };
|
|
148
|
+
}
|
|
149
|
+
|
|
150
|
+
return { execute };
|
|
151
|
+
}
|