mslxdff 0.1.65 → 0.1.67
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bin/mslxdff.js +3 -3159
- package/package.json +1 -1
- package/src/chat/cooling.js +99 -0
- package/src/chat/direct.js +57 -0
- package/src/chat/gateway.js +148 -0
- package/src/chat/orchestrator.js +218 -0
- package/src/chat/sse.js +69 -0
- package/src/chat/upstream.js +59 -501
- package/src/cli/bootstrap.js +1 -0
- package/src/cli/commands/daemon.js +151 -0
- package/src/cli/commands/group.js +272 -0
- package/src/cli/commands/model.js +405 -0
- package/src/cli/commands/provider/add.js +106 -0
- package/src/cli/commands/provider/allowlist.js +99 -0
- package/src/cli/commands/provider/config.js +82 -0
- package/src/cli/commands/provider/index.js +193 -0
- package/src/cli/commands/provider/keys.js +135 -0
- package/src/cli/commands/provider/models.js +99 -0
- package/src/cli/commands/provider.js +1 -0
- package/src/cli/commands/sync.js +143 -0
- package/src/cli/commands/system.js +236 -0
- package/src/cli/commands/workbuddy.js +91 -0
- package/src/cli/format.js +119 -0
- package/src/cli/group-helpers.js +69 -0
- package/src/cli/help.js +66 -0
- package/src/cli/index.js +59 -0
- package/src/cli/interactive.js +84 -0
- package/src/cli/policy.js +119 -0
- package/src/cli/provider-row.js +73 -0
- package/src/cli/status.js +283 -0
- package/src/cli/util.js +24 -0
- package/src/providers/base.js +159 -0
- package/src/providers/dispatcher.js +18 -6
- package/src/providers/generic.js +28 -166
- package/src/providers/openrouter.js +19 -205
- package/src/providers/workbuddy/auth.js +175 -0
- package/src/providers/workbuddy/balance.js +84 -0
- package/src/providers/workbuddy/chat.js +310 -0
- package/src/providers/workbuddy/index.js +263 -0
- package/src/providers/workbuddy/models.js +111 -0
- package/src/providers/workbuddy/rotation-log.js +54 -0
- package/src/providers/workbuddy.js +2 -677
- package/src/routes/chat/broadband-handler.js +25 -46
- package/src/routes/chat/exhausted-handler.js +2 -2
- package/src/routes/chat/gateway.js +301 -0
- package/src/routes/chat/hedge-handler.js +65 -83
- package/src/routes/chat/index.js +1 -384
- package/src/routes/chat/local-handler.js +32 -61
- package/src/routes/chat/peer-handler.js +32 -23
- package/src/routes/chat/relay-pipeline.js +151 -0
- package/src/runtime/bootstrap.js +408 -0
- package/src/state/facade.js +57 -0
- package/src/state/memory.js +161 -0
- package/src/state/merge.js +26 -0
- package/src/state/persist.js +42 -0
- package/src/state/provider-config.js +143 -0
- package/src/state/schemas/allowlist.js +87 -0
- package/src/state/schemas/group.js +31 -0
- package/src/state/schemas/model.js +53 -0
- package/src/state/schemas/peer.js +31 -0
- package/src/state/schemas/port.js +10 -0
- package/src/state/schemas/provider.js +204 -0
- package/src/state/schemas/token.js +43 -0
- package/src/state/store.js +176 -0
- package/src/state.js +1 -711
|
@@ -1,8 +1,9 @@
|
|
|
1
|
+
import { relay, SLOW_TOTAL_MS, STREAM_TIMEOUT_MS, STALL_TIMEOUT_MS, SCORE_STALL_MS } from "../stream.js";
|
|
1
2
|
import { buildFallbackInfo } from "../fallback.js";
|
|
2
|
-
import {
|
|
3
|
+
import { createRelayPipeline } from "./relay-pipeline.js";
|
|
3
4
|
import { tryBroadbandRelay } from "../relay-queue.js";
|
|
4
|
-
import { runHook } from "../../plugins.js";
|
|
5
5
|
|
|
6
|
+
/** 薄适配:broadband 两种形态合一经由 pipeline */
|
|
6
7
|
export async function handleBroadbandRelay({
|
|
7
8
|
model,
|
|
8
9
|
body,
|
|
@@ -30,57 +31,35 @@ export async function handleBroadbandRelay({
|
|
|
30
31
|
evt("relay-miss", { reqId: handlerCtx.reqId, model });
|
|
31
32
|
return { handled: false };
|
|
32
33
|
}
|
|
34
|
+
const pipeline = createRelayPipeline({
|
|
35
|
+
relay,
|
|
36
|
+
buildFallbackInfo,
|
|
37
|
+
auto,
|
|
38
|
+
plugins,
|
|
39
|
+
evt,
|
|
40
|
+
mark,
|
|
41
|
+
logCall: () => {},
|
|
42
|
+
logError: () => {},
|
|
43
|
+
constants: { STREAM_TIMEOUT_MS, SLOW_TOTAL_MS, STALL_TIMEOUT_MS, SCORE_STALL_MS },
|
|
44
|
+
startedAt,
|
|
45
|
+
stages,
|
|
46
|
+
});
|
|
33
47
|
const isResponse = bb.result && typeof bb.result.status === "number" && typeof bb.result.headers?.get === "function";
|
|
34
48
|
if (isResponse) {
|
|
35
|
-
|
|
36
|
-
if (bbFallback?.fallback) evt("fallback-notice", { reqId: handlerCtx.reqId, requested, actual: model, reason: bbFallback.reason, notice: bbFallback.notice, via: "broadband" });
|
|
37
|
-
evt("relay-start", { reqId: handlerCtx.reqId, model, via: "broadband", target: bb.target, group: bb.group, fallback: bbFallback });
|
|
38
|
-
const out = await relay(res, bb.result, body, {
|
|
39
|
-
fallback: bbFallback,
|
|
40
|
-
onFirstChunk: (d) => mark(`ttf-bb-${model}`),
|
|
41
|
-
onDownstreamAbort: () => evt("client-abort", { reqId: handlerCtx.reqId, model, totalMs: Math.round(Date.now() - startedAt), stages: [...stages] }),
|
|
42
|
-
});
|
|
43
|
-
evt("relay-done", { reqId: handlerCtx.reqId, model, via: "broadband", status: out.status, ttfMs: out.ttfMs, totalMs: out.totalMs, aborted: out.aborted, interrupted: out.interrupted ?? false, detail: out.detail ?? null });
|
|
44
|
-
if (auto && out.status === 200) {
|
|
45
|
-
const latencyMs = out.totalMs ?? 0;
|
|
46
|
-
if (out.detail?.stallHits > 0 || (latencyMs && latencyMs > SLOW_TOTAL_MS)) {
|
|
47
|
-
void auto.recordError(model, { status: 200, slow: true, note: `broadband slow ${latencyMs}ms` });
|
|
48
|
-
void auto.recordLatency(model, latencyMs);
|
|
49
|
-
} else {
|
|
50
|
-
await auto.recordOk(model, { latencyMs });
|
|
51
|
-
}
|
|
52
|
-
}
|
|
53
|
-
evt("result", { model, status: out.status, via: "broadband", timing: bb.result._t ?? null, ttfMs: out.ttfMs, totalMs: out.totalMs, detail: out.detail ?? null, fallback: bbFallback, requested, actual: model });
|
|
54
|
-
evt("client-response", { requested, actual: model, via: "broadband", fallback: bbFallback, status: out.status, reqId: handlerCtx.reqId });
|
|
55
|
-
if (plugins?.length) runHook(plugins, "request:completed", { reqId: handlerCtx.reqId, requested, useAuto, hops: handlerCtx.hops, stream: Boolean(body.stream), durationMs: Date.now() - startedAt, via: "broadband", status: out.status, actual: model, fallback: bbFallback }).catch(() => {});
|
|
49
|
+
await pipeline.execute({ res, upRes: bb.result, body, requested, actual: model, lastErr, via: "broadband", lockModel, useAuto, handlerCtx, mark, perf0, stages, startedAt });
|
|
56
50
|
return { handled: true };
|
|
57
|
-
}
|
|
51
|
+
}
|
|
52
|
+
if (bb.result && typeof bb.result.status === "number") {
|
|
53
|
+
const b = bb.result.body || "";
|
|
54
|
+
const str = typeof b === "string" ? b : JSON.stringify(b);
|
|
55
|
+
const isSSE = bb.result.headers?.["Content-Type"]?.includes("text/event-stream");
|
|
58
56
|
const fakeRes = {
|
|
59
57
|
status: bb.result.status,
|
|
60
58
|
headers: { get: (k) => bb.result.headers?.[k] || bb.result.headers?.[k.toLowerCase()] || null },
|
|
61
|
-
text: async () =>
|
|
62
|
-
body: (()
|
|
63
|
-
const b = bb.result.body || "";
|
|
64
|
-
const str = typeof b === "string" ? b : JSON.stringify(b);
|
|
65
|
-
const isSSE = bb.result.headers?.["Content-Type"]?.includes("text/event-stream");
|
|
66
|
-
if (isSSE) {
|
|
67
|
-
return (async function* () { yield Buffer.from(str); })();
|
|
68
|
-
}
|
|
69
|
-
return null;
|
|
70
|
-
})(),
|
|
59
|
+
text: async () => str,
|
|
60
|
+
body: isSSE ? (async function* () { yield Buffer.from(str); })() : null,
|
|
71
61
|
};
|
|
72
|
-
|
|
73
|
-
if (bbLocalFallback?.fallback) evt("fallback-notice", { reqId: handlerCtx.reqId, requested, actual: model, reason: bbLocalFallback.reason, notice: bbLocalFallback.notice, via: "broadband" });
|
|
74
|
-
evt("relay-start", { reqId: handlerCtx.reqId, model, via: "broadband-local", target: bb.target, group: bb.group, fallback: bbLocalFallback });
|
|
75
|
-
const out = await relay(res, fakeRes, body, {
|
|
76
|
-
fallback: bbLocalFallback,
|
|
77
|
-
onFirstChunk: (d) => mark(`ttf-bb-${model}`),
|
|
78
|
-
onDownstreamAbort: () => evt("client-abort", { reqId: handlerCtx.reqId, model, totalMs: Math.round(Date.now() - startedAt), stages: [...stages] }),
|
|
79
|
-
});
|
|
80
|
-
evt("relay-done", { reqId: handlerCtx.reqId, model, via: "broadband-local", status: out.status, ttfMs: out.ttfMs, totalMs: out.totalMs, aborted: out.aborted, interrupted: out.interrupted ?? false, detail: out.detail ?? null });
|
|
81
|
-
evt("result", { model, status: out.status, via: "broadband", timing: null, ttfMs: out.ttfMs, totalMs: out.totalMs, detail: out.detail ?? null, fallback: bbLocalFallback, requested, actual: model });
|
|
82
|
-
evt("client-response", { requested, actual: model, via: "broadband", fallback: bbLocalFallback, status: out.status, reqId: handlerCtx.reqId });
|
|
83
|
-
if (plugins?.length) runHook(plugins, "request:completed", { reqId: handlerCtx.reqId, requested, useAuto, hops: handlerCtx.hops, stream: Boolean(body.stream), durationMs: Date.now() - startedAt, via: "broadband", status: out.status, actual: model, fallback: bbLocalFallback }).catch(() => {});
|
|
62
|
+
await pipeline.execute({ res, upRes: fakeRes, body, requested, actual: model, lastErr, via: "broadband-local", lockModel, useAuto, handlerCtx, mark, perf0, stages, startedAt });
|
|
84
63
|
return { handled: true };
|
|
85
64
|
}
|
|
86
65
|
evt("relay-miss", { reqId: handlerCtx.reqId, model });
|
|
@@ -1,6 +1,6 @@
|
|
|
1
|
-
import { performance } from "node:perf_hooks";
|
|
2
|
-
import { json } from "../helpers.js";
|
|
3
1
|
import { relay } from "../stream.js";
|
|
2
|
+
import { json } from "../helpers.js";
|
|
3
|
+
import { performance } from "node:perf_hooks";
|
|
4
4
|
|
|
5
5
|
export async function handleExhaustedLocal({ res, body, lastErr, order, handlerCtx, evt, logCall, mark, perf0, stages, done, requested, useAuto }) {
|
|
6
6
|
const model = lastErr?.model ?? handlerCtx.model;
|
|
@@ -0,0 +1,301 @@
|
|
|
1
|
+
import { performance } from "node:perf_hooks";
|
|
2
|
+
import { injectReasoningContent, normalizeModel } from "../../reasoning.js";
|
|
3
|
+
import { isAutoModel } from "../../auto.js";
|
|
4
|
+
import { toInternalId as aliasToInternal } from "../../sync-opencode.js";
|
|
5
|
+
import { clientIp, json, readBody, parseHops, summarizePrompt, errMsg } from "../helpers.js";
|
|
6
|
+
import { hedgeDelayMs, shouldHedge } from "../hedge.js";
|
|
7
|
+
import { runHook } from "../../plugins.js";
|
|
8
|
+
import { parseShareKeysHeader, SHARE_KEYS_HEADER } from "../../providers/share-keys.js";
|
|
9
|
+
import { handleHedge } from "./hedge-handler.js";
|
|
10
|
+
import { handleLocalRelay } from "./local-handler.js";
|
|
11
|
+
import { handlePeerRelay } from "./peer-handler.js";
|
|
12
|
+
import { handleBroadbandRelay } from "./broadband-handler.js";
|
|
13
|
+
import { handleExhaustedLocal, handleExhaustedAll } from "./exhausted-handler.js";
|
|
14
|
+
import { normalizeFullId, getModelAlias } from "../../providers/model-id.js";
|
|
15
|
+
|
|
16
|
+
/**
|
|
17
|
+
* ChatGateway 深模块:对外 1 handle,内部 Policy→Selector→Executor 三段编排
|
|
18
|
+
* Policy: 别名/allowlist/header 透传
|
|
19
|
+
* Selector: order 推导 + 并发择优 + 排序
|
|
20
|
+
* Executor: 串行 trial → hedge → local → peer → broadband → exhausted
|
|
21
|
+
* 两 adapter:Provider(upstream.chat) + Clock/Latency(auto) 可注入 fake
|
|
22
|
+
*/
|
|
23
|
+
export function createChatGateway({ upstream, auto, logs, peers, maxHops, groups, bus, token, plugins }) {
|
|
24
|
+
async function handle({ req, res }) {
|
|
25
|
+
let body;
|
|
26
|
+
try {
|
|
27
|
+
body = await readBody(req);
|
|
28
|
+
} catch {
|
|
29
|
+
return json(res, 400, { error: "Invalid JSON body" });
|
|
30
|
+
}
|
|
31
|
+
if (plugins?.length) {
|
|
32
|
+
const rc = await runHook(plugins, "request:received", { ip: clientIp(req), hops: parseHops(req.headers["x-mslxdff-hops"]), headers: { "content-type": req.headers["content-type"] }, body });
|
|
33
|
+
for (const e of rc.errors) logs?.appendEvent?.({ ts: Date.now(), type: "plugin-hook-error", hook: "request:received", plugin: e.plugin, error: e.error });
|
|
34
|
+
const respond = rc.value?.respond;
|
|
35
|
+
if (respond && typeof respond === "object") return json(res, respond.status || 200, respond.body ?? {});
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
const startedAt = Date.now();
|
|
39
|
+
const reqId = `${Date.now().toString(36)}-${Math.random().toString(36).slice(2, 6)}`;
|
|
40
|
+
const perf0 = performance.now();
|
|
41
|
+
const stages = [];
|
|
42
|
+
const mark = (name) => stages.push([name, Math.round(performance.now() - perf0)]);
|
|
43
|
+
|
|
44
|
+
// ===== Policy =====
|
|
45
|
+
const hops = parseHops(req.headers["x-mslxdff-hops"]);
|
|
46
|
+
const shareKeys = parseShareKeysHeader(req.headers[SHARE_KEYS_HEADER] || "");
|
|
47
|
+
const workbuddyUid = (req.headers["x-mslxdff-workbuddy-uid"] || req.headers["x-workbuddy-uid"] || "").toString().trim();
|
|
48
|
+
const lockModel = req.headers["x-mslxdff-model-lock"] || "";
|
|
49
|
+
const rawModel = body.model || "";
|
|
50
|
+
let normalizedRequested = normalizeModel(lockModel || rawModel || "");
|
|
51
|
+
const aliasResolved = getModelAlias(normalizedRequested);
|
|
52
|
+
if (aliasResolved) { normalizedRequested = aliasResolved; body = { ...body, model: aliasResolved }; }
|
|
53
|
+
let requested = normalizedRequested;
|
|
54
|
+
let aliasInfo = null;
|
|
55
|
+
if (requested.startsWith("mslxdff-")) {
|
|
56
|
+
const internal = aliasToInternal(requested);
|
|
57
|
+
if (internal) { aliasInfo = `${requested} -> ${internal}`; requested = internal; }
|
|
58
|
+
} else if (requested.includes("/")) {
|
|
59
|
+
const slashIdx = requested.indexOf("/");
|
|
60
|
+
const rawPart = requested.slice(slashIdx + 1);
|
|
61
|
+
const providerPart = requested.slice(0, slashIdx);
|
|
62
|
+
if (rawPart.startsWith("mslxdff-")) {
|
|
63
|
+
const internal = aliasToInternal(rawPart);
|
|
64
|
+
if (internal) {
|
|
65
|
+
aliasInfo = `${requested} -> ${providerPart}/${internal} (alias stripped)`;
|
|
66
|
+
requested = `${providerPart}/${internal}`;
|
|
67
|
+
if (providerPart === "mslxdff") { requested = internal; aliasInfo = `${rawModel} -> ${internal} (mslxdff alias stripped)`; }
|
|
68
|
+
}
|
|
69
|
+
} else if (providerPart === "mslxdff") {
|
|
70
|
+
aliasInfo = `${requested} -> ${rawPart} (mslxdff provider stripped, 原名兼容)`;
|
|
71
|
+
requested = rawPart;
|
|
72
|
+
}
|
|
73
|
+
}
|
|
74
|
+
const useAuto = isAutoModel(requested);
|
|
75
|
+
mark("parsed");
|
|
76
|
+
if (aliasInfo) { try { res.setHeader("x-mslxdff-alias", aliasInfo); } catch {} }
|
|
77
|
+
|
|
78
|
+
// ===== Selector: order 推导 =====
|
|
79
|
+
let order;
|
|
80
|
+
if (lockModel) order = [requested];
|
|
81
|
+
else if (useAuto) order = auto ? await auto.candidates() : [""];
|
|
82
|
+
else order = auto ? await auto.candidatesFor(requested) : [requested];
|
|
83
|
+
if (!order.length) order = [""];
|
|
84
|
+
const canFallback = order.length > 1;
|
|
85
|
+
const canForwardPeers = Boolean(peers) && hops < maxHops;
|
|
86
|
+
mark("ordered");
|
|
87
|
+
|
|
88
|
+
const logCall = (model, status) => logs?.appendCall({ reqId, model, auto: useAuto, status, durationMs: Date.now() - startedAt, stream: Boolean(body.stream), stages });
|
|
89
|
+
const logError = (model, status, message) => logs?.appendError({ reqId, model, auto: useAuto, status, message, stages });
|
|
90
|
+
const evt = (type, data) => {
|
|
91
|
+
const entry = { ts: Date.now(), reqId, type, ...data, model: data.model ?? requested, auto: useAuto, durationMs: Date.now() - startedAt, stages: [...stages] };
|
|
92
|
+
if (bus) bus.emit(entry);
|
|
93
|
+
logs?.appendEvent?.(entry);
|
|
94
|
+
};
|
|
95
|
+
const done = (info) => {
|
|
96
|
+
if (!plugins?.length) return;
|
|
97
|
+
runHook(plugins, "request:completed", { reqId, requested, useAuto, hops, stream: Boolean(body.stream), durationMs: Date.now() - startedAt, ...info }).catch(() => {});
|
|
98
|
+
};
|
|
99
|
+
evt("request", { reqId, hops, ip: clientIp(req), stream: Boolean(body.stream), prompt: summarizePrompt(body), rawModel, requested, lockModel: lockModel || null });
|
|
100
|
+
if (aliasInfo) evt("alias", { reqId, alias: aliasInfo, rawModel, requested });
|
|
101
|
+
if (Object.keys(shareKeys).length) evt("share-keys", { reqId, providers: Object.keys(shareKeys) });
|
|
102
|
+
evt("ordered", { reqId, order, canFallback, canForwardPeers, useAuto, statuses: auto?.statuses?.() ?? null });
|
|
103
|
+
|
|
104
|
+
if (plugins?.length && !lockModel) {
|
|
105
|
+
const sel = await runHook(plugins, "model:select", { reqId, requested, useAuto, order: [...order], hops, stream: Boolean(body.stream) });
|
|
106
|
+
if (sel.changed && Array.isArray(sel.value) && sel.value.length) {
|
|
107
|
+
order = sel.value.filter(Boolean);
|
|
108
|
+
if (!order.length) order = [requested];
|
|
109
|
+
evt("plugin-hook", { reqId, hook: "model:select", applied: true, order: [...order] });
|
|
110
|
+
}
|
|
111
|
+
for (const e of sel.errors) evt("plugin-hook-error", { reqId, hook: "model:select", plugin: e.plugin, error: e.error });
|
|
112
|
+
}
|
|
113
|
+
|
|
114
|
+
const handlerCtx = { reqId, model: null, body, hops, peers, plugins, evt, logError, logCall, logs };
|
|
115
|
+
|
|
116
|
+
// ===== Selector: 首次 auto 并发择优 =====
|
|
117
|
+
if (useAuto && order.length > 1 && auto && !lockModel) {
|
|
118
|
+
const statuses = auto.statuses?.() ?? {};
|
|
119
|
+
const hasPriorSuccess = Object.values(statuses).some((e) => e && typeof e === "object" && e.status === "normal");
|
|
120
|
+
const nonCoolingOrder = order.filter((m) => { try { return !auto.isCooling(m); } catch { return true; } });
|
|
121
|
+
if (!hasPriorSuccess && nonCoolingOrder.length > 1) {
|
|
122
|
+
const concLimit = (() => {
|
|
123
|
+
const v = Number(process.env.MSLXDFF_AUTO_CONCURRENT);
|
|
124
|
+
if (Number.isInteger(v) && v > 0) return Math.min(v, nonCoolingOrder.length);
|
|
125
|
+
return Math.min(nonCoolingOrder.length, 5);
|
|
126
|
+
})();
|
|
127
|
+
const raceModels = nonCoolingOrder.slice(0, concLimit);
|
|
128
|
+
evt("auto-concurrent-race", { reqId, models: raceModels, skippedFaulty: order.length - nonCoolingOrder.length, limit: concLimit });
|
|
129
|
+
const raceStart = performance.now();
|
|
130
|
+
const attempts = raceModels.map(async (m) => {
|
|
131
|
+
const fwd = { ...injectReasoningContent(m, body), model: m };
|
|
132
|
+
let r = null;
|
|
133
|
+
try {
|
|
134
|
+
const chatOpts = {};
|
|
135
|
+
if (Object.keys(shareKeys).length) chatOpts.shareKeys = shareKeys;
|
|
136
|
+
if (workbuddyUid) chatOpts.workbuddyUid = workbuddyUid;
|
|
137
|
+
r = await upstream.chat(fwd, Object.keys(chatOpts).length ? chatOpts : undefined);
|
|
138
|
+
} catch (err) {
|
|
139
|
+
return { model: m, ok: false, error: errMsg(err), status: 502, timing: err?._t ?? null };
|
|
140
|
+
}
|
|
141
|
+
if (r && r.status >= 400) {
|
|
142
|
+
const isAllow = r.status === 403 && r.headers?.get?.("x-mslxdff-allowlist") === "1";
|
|
143
|
+
if (isAllow) return { model: m, ok: false, error: "allowlist", status: 403, allowlist: true };
|
|
144
|
+
return { model: m, ok: false, error: `upstream ${r.status}`, status: r.status, res: r, timing: r._t ?? null };
|
|
145
|
+
}
|
|
146
|
+
if (r instanceof Error) return { model: m, ok: false, error: errMsg(r), status: 502 };
|
|
147
|
+
return { model: m, ok: true, res: r, status: r.status, timing: r._t ?? null };
|
|
148
|
+
});
|
|
149
|
+
const results = await Promise.allSettled(attempts);
|
|
150
|
+
const okList = results.map((r, i) => ({ r, i, model: raceModels[i] }))
|
|
151
|
+
.filter(({ r }) => r.status === "fulfilled" && r.value?.ok)
|
|
152
|
+
.map(({ r, i, model }) => ({ model, idx: i, val: r.value, t: r.value.timing?.totalMs ?? r.value.timing?.ms ?? Number.MAX_SAFE_INTEGER }));
|
|
153
|
+
if (okList.length) {
|
|
154
|
+
okList.sort((a, b) => a.t - b.t);
|
|
155
|
+
const best = okList[0];
|
|
156
|
+
const winModel = best.model;
|
|
157
|
+
evt("auto-concurrent-win", { reqId, model: winModel, timing: best.val.timing, totalMs: Math.round(performance.now() - raceStart), tried: raceModels.length });
|
|
158
|
+
for (const { r, i } of results.map((r, i) => ({ r, i }))) {
|
|
159
|
+
const m = raceModels[i];
|
|
160
|
+
if (r.status === "fulfilled" && r.value?.ok) {
|
|
161
|
+
if (m === winModel) {
|
|
162
|
+
const latencyMs = r.value.timing?.totalMs ?? Math.round(performance.now() - raceStart);
|
|
163
|
+
await auto.recordOk(m, { latencyMs });
|
|
164
|
+
try { const { savePreferredModel } = await import("../../state.js"); savePreferredModel(m); evt("auto-concurrent-preferred", { reqId, model: m }); } catch {}
|
|
165
|
+
}
|
|
166
|
+
} else if (r.status === "fulfilled" && !r.value?.ok && !r.value?.allowlist) await auto.recordError(m, { status: r.value.status || 502 });
|
|
167
|
+
else if (r.status === "rejected") await auto.recordError(m, { status: 502 });
|
|
168
|
+
}
|
|
169
|
+
handlerCtx.model = winModel;
|
|
170
|
+
const { handleLocalRelay: _relay } = await import("./local-handler.js");
|
|
171
|
+
const lr = await _relay({
|
|
172
|
+
upRes: best.val.res, model: winModel, body, order: raceModels, idx: best.idx,
|
|
173
|
+
lastErr: null, requested, useAuto, lockModel, auto, handlerCtx, evt, logCall, logError, mark, perf0, stages, startedAt, plugins, res,
|
|
174
|
+
});
|
|
175
|
+
if (lr.handled) return;
|
|
176
|
+
if (!lr.lastErr) return;
|
|
177
|
+
} else {
|
|
178
|
+
evt("auto-concurrent-all-fail", { reqId, tried: raceModels.length, totalMs: Math.round(performance.now() - raceStart) });
|
|
179
|
+
for (const { r, i } of results.map((r, i) => ({ r, i }))) {
|
|
180
|
+
const m = raceModels[i];
|
|
181
|
+
if (r.status === "fulfilled" && !r.value?.ok && !r.value?.allowlist) await auto.recordError(m, { status: r.value.status || 502 });
|
|
182
|
+
else if (r.status === "rejected") await auto.recordError(m, { status: 502 });
|
|
183
|
+
}
|
|
184
|
+
}
|
|
185
|
+
const triedSet = new Set(raceModels);
|
|
186
|
+
order = order.filter((m) => !triedSet.has(m));
|
|
187
|
+
if (!order.length) {
|
|
188
|
+
const last = { model: raceModels[0] || requested, status: 502, message: "all concurrent candidates failed" };
|
|
189
|
+
await handleExhaustedAll({ res, body, lastErr: last, order: raceModels, requested, handlerCtx: { ...handlerCtx, reqId, startedAt }, evt, logCall, mark, perf0, stages });
|
|
190
|
+
return;
|
|
191
|
+
}
|
|
192
|
+
}
|
|
193
|
+
}
|
|
194
|
+
|
|
195
|
+
// ===== Executor: 串行 trial =====
|
|
196
|
+
let lastErr = null;
|
|
197
|
+
for (let idx = 0; idx < order.length; idx++) {
|
|
198
|
+
const model = order[idx];
|
|
199
|
+
handlerCtx.model = model;
|
|
200
|
+
evt("model-try", { reqId, model, idx, remaining: order.length - idx });
|
|
201
|
+
if (plugins?.length) {
|
|
202
|
+
const bt = await runHook(plugins, "model:beforeTry", { reqId, requested, model, idx, hops });
|
|
203
|
+
for (const e of bt.errors) evt("plugin-hook-error", { reqId, hook: "model:beforeTry", plugin: e.plugin, error: e.error });
|
|
204
|
+
if (bt.value === false || bt.value?.skip === true) { evt("plugin-hook", { reqId, hook: "model:beforeTry", applied: true, skipped: model }); continue; }
|
|
205
|
+
}
|
|
206
|
+
let upRes = null;
|
|
207
|
+
let forwarded = { ...injectReasoningContent(model, body), model };
|
|
208
|
+
if (plugins?.length) {
|
|
209
|
+
const ur = await runHook(plugins, "upstream:request", { reqId, requested, model, payload: forwarded, stream: Boolean(body.stream) });
|
|
210
|
+
for (const e of ur.errors) evt("plugin-hook-error", { reqId, hook: "upstream:request", plugin: e.plugin, error: e.error });
|
|
211
|
+
if (ur.changed && ur.value?.payload && typeof ur.value.payload === "object") { forwarded = ur.value.payload; evt("plugin-hook", { reqId, hook: "upstream:request", applied: true, model, rewrittenModel: forwarded.model ?? null }); }
|
|
212
|
+
}
|
|
213
|
+
const tUp = performance.now();
|
|
214
|
+
evt("upstream-try", { reqId, model, attempt: idx + 1 });
|
|
215
|
+
try {
|
|
216
|
+
const chatOpts = {};
|
|
217
|
+
if (Object.keys(shareKeys).length) chatOpts.shareKeys = shareKeys;
|
|
218
|
+
if (workbuddyUid) chatOpts.workbuddyUid = workbuddyUid;
|
|
219
|
+
upRes = await upstream.chat(forwarded, Object.keys(chatOpts).length ? chatOpts : undefined);
|
|
220
|
+
evt("upstream-done", { reqId, model, ok: !(upRes instanceof Error) && upRes.status < 400, status: upRes instanceof Error ? null : upRes.status, timing: upRes._t ?? null, error: null });
|
|
221
|
+
} catch (err) {
|
|
222
|
+
if (auto) await auto.recordError(model, { message: errMsg(err) });
|
|
223
|
+
lastErr = { model, upstream: null, status: 502, message: errMsg(err) };
|
|
224
|
+
logError(model, 502, errMsg(err));
|
|
225
|
+
evt("upstream-error", { reqId, model, status: 502, message: errMsg(err), timing: err._t ?? { attempts: [], waitMs: 0, totalMs: Math.round(performance.now() - tUp) } });
|
|
226
|
+
}
|
|
227
|
+
if (plugins?.length) {
|
|
228
|
+
runHook(plugins, "upstream:response", {
|
|
229
|
+
reqId, requested, model,
|
|
230
|
+
status: upRes instanceof Error ? null : upRes instanceof Object ? (upRes.status ?? null) : null,
|
|
231
|
+
ok: !(upRes instanceof Error) && upRes ? upRes.status < 400 : false,
|
|
232
|
+
error: upRes instanceof Error ? errMsg(upRes) : null,
|
|
233
|
+
timing: upRes?._t ?? null,
|
|
234
|
+
}).catch(() => {});
|
|
235
|
+
}
|
|
236
|
+
mark(`up-${model}`);
|
|
237
|
+
if (upRes && upRes.status >= 400) {
|
|
238
|
+
const isAllowlistBlock = upRes.status === 403 && (upRes.headers?.get?.("x-mslxdff-allowlist") === "1");
|
|
239
|
+
if (isAllowlistBlock) {
|
|
240
|
+
let bodyText = null; try { bodyText = await upRes.clone().text(); } catch {}
|
|
241
|
+
let errBody = { error: `model not allowed for provider` };
|
|
242
|
+
try { errBody = bodyText ? JSON.parse(bodyText) : errBody; } catch { errBody = { error: bodyText || "model not allowed" }; }
|
|
243
|
+
if (useAuto) {
|
|
244
|
+
logError(model, 403, errBody.error || "model not allowed");
|
|
245
|
+
evt("upstream-error", { reqId, model, status: 403, message: errBody.error, timing: upRes._t ?? null, allowlist: true, skipped: true });
|
|
246
|
+
lastErr = { model, upstream: upRes, status: 403, message: errBody.error || "model not allowed" };
|
|
247
|
+
if (canFallback && idx < order.length - 1) { evt("fallback", { reqId, from: model, to: order[idx + 1] ?? null, reason: `allowlist skip ${errBody.error || "blocked"}` }); continue; }
|
|
248
|
+
return json(res, 403, errBody);
|
|
249
|
+
}
|
|
250
|
+
logError(model, 403, errBody.error || "model not allowed");
|
|
251
|
+
evt("upstream-error", { reqId, model, status: 403, message: errBody.error, timing: upRes._t ?? null, allowlist: true });
|
|
252
|
+
return json(res, 403, errBody);
|
|
253
|
+
}
|
|
254
|
+
if (auto) await auto.recordError(model, { status: upRes.status });
|
|
255
|
+
lastErr = { model, upstream: upRes, status: upRes.status, message: null };
|
|
256
|
+
logError(model, upRes.status, `upstream ${upRes.status}`);
|
|
257
|
+
evt("upstream-error", { reqId, model, status: upRes.status, message: null, timing: upRes._t ?? null });
|
|
258
|
+
upRes = null;
|
|
259
|
+
}
|
|
260
|
+
if (upRes) {
|
|
261
|
+
const isStream = Boolean(body.stream);
|
|
262
|
+
const d = hedgeDelayMs();
|
|
263
|
+
const hasPeers = Boolean(peers) && peers.ordered().length > 0;
|
|
264
|
+
const doHedge = shouldHedge({ isStream, canForwardPeers, hedgeDelayMs: d, hasPeers }) && upRes.status === 200 && upRes.body;
|
|
265
|
+
if (doHedge) {
|
|
266
|
+
const hr = await handleHedge({ upRes, model, body, order, idx, lastErr, requested, useAuto, lockModel, auto, peers, handlerCtx, evt, logCall, logError, mark, perf0, stages, startedAt, plugins, res, hedgeDelayMs: d });
|
|
267
|
+
if (hr.handled) return;
|
|
268
|
+
if (hr.lastErr) lastErr = hr.lastErr;
|
|
269
|
+
if (hr.upRes === null) upRes = null;
|
|
270
|
+
else if (hr.upRes) upRes = hr.upRes;
|
|
271
|
+
}
|
|
272
|
+
if (upRes) {
|
|
273
|
+
const lr = await handleLocalRelay({ upRes, model, body, order, idx, lastErr, requested, useAuto, lockModel, auto, handlerCtx, evt, logCall, logError, mark, perf0, stages, startedAt, plugins, res });
|
|
274
|
+
if (lr.handled) return;
|
|
275
|
+
if (lr.lastErr) { lastErr = lr.lastErr; continue; }
|
|
276
|
+
return;
|
|
277
|
+
}
|
|
278
|
+
}
|
|
279
|
+
if (canForwardPeers) {
|
|
280
|
+
const pr = await handlePeerRelay({ model, body, lastErr, requested, useAuto, lockModel, auto, peers, handlerCtx, evt, logCall, mark, perf0, stages, startedAt, plugins, res });
|
|
281
|
+
if (pr.handled) return;
|
|
282
|
+
}
|
|
283
|
+
if (groups) {
|
|
284
|
+
const br = await handleBroadbandRelay({ model, body, hops, lastErr, requested, useAuto, lockModel, auto, groups, token, bus, logs, handlerCtx, evt, mark, perf0, stages, res, startedAt, plugins });
|
|
285
|
+
if (br.handled) return;
|
|
286
|
+
}
|
|
287
|
+
if (canFallback) { evt("fallback", { reqId, from: model, to: order[idx + 1] ?? null, reason: lastErr?.message || `upstream ${lastErr?.status ?? 502}` }); continue; }
|
|
288
|
+
await handleExhaustedLocal({ res, body, lastErr, order, handlerCtx: { ...handlerCtx, model, reqId }, evt, logCall, mark, perf0, stages, done, requested, useAuto });
|
|
289
|
+
return;
|
|
290
|
+
}
|
|
291
|
+
await handleExhaustedAll({ res, body, lastErr, order, requested, handlerCtx: { ...handlerCtx, reqId, startedAt }, evt, logCall, mark, perf0, stages });
|
|
292
|
+
}
|
|
293
|
+
|
|
294
|
+
return { handle };
|
|
295
|
+
}
|
|
296
|
+
|
|
297
|
+
// 薄适配:保持原 chatHandler 签名兼容
|
|
298
|
+
export async function chatHandler(ctx) {
|
|
299
|
+
const gw = createChatGateway(ctx);
|
|
300
|
+
return gw.handle(ctx);
|
|
301
|
+
}
|
|
@@ -1,8 +1,9 @@
|
|
|
1
|
-
import { buildFallbackInfo } from "../fallback.js";
|
|
2
1
|
import { relay, SLOW_TOTAL_MS, STREAM_TIMEOUT_MS, STALL_TIMEOUT_MS, SCORE_STALL_MS } from "../stream.js";
|
|
2
|
+
import { buildFallbackInfo } from "../fallback.js";
|
|
3
|
+
import { createRelayPipeline } from "./relay-pipeline.js";
|
|
3
4
|
import { hedgedFirstChunkRace } from "../hedge.js";
|
|
4
|
-
import { runHook } from "../../plugins.js";
|
|
5
5
|
|
|
6
|
+
/** 薄适配:hedge 赛跑后按 winner 调 pipeline(复用 relay-pipeline 深模块) */
|
|
6
7
|
export async function handleHedge({
|
|
7
8
|
upRes,
|
|
8
9
|
model,
|
|
@@ -27,104 +28,86 @@ export async function handleHedge({
|
|
|
27
28
|
res,
|
|
28
29
|
hedgeDelayMs,
|
|
29
30
|
}) {
|
|
30
|
-
const isStream = Boolean(body.stream);
|
|
31
31
|
const d = hedgeDelayMs;
|
|
32
|
-
// hedge 已在外层判断 doHedge,这里直接执行赛跑
|
|
33
32
|
try {
|
|
34
33
|
const hedged = await hedgedFirstChunkRace({ localUpRes: upRes, peers, handlerCtx, hedgeDelayMs: d, evt });
|
|
35
34
|
if (hedged && hedged.winner) {
|
|
36
35
|
if (hedged.winner === "local") {
|
|
37
|
-
const fallback = buildFallbackInfo({ requested, actual: model, lastErr, via: "local", useAuto, lockModel });
|
|
38
|
-
if (fallback?.fallback) evt("fallback-notice", { reqId: handlerCtx.reqId, requested, actual: model, reason: fallback.reason, notice: fallback.notice, via: "local" });
|
|
39
|
-
evt("relay-start", { reqId: handlerCtx.reqId, model, via: "local", isStream, fallback, hedged: true });
|
|
40
36
|
const bufferedUpRes = { ...upRes, body: hedged.bufferedBody, headers: upRes.headers, status: upRes.status, _t: upRes._t };
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
37
|
+
const pipeline = createRelayPipeline({
|
|
38
|
+
relay,
|
|
39
|
+
buildFallbackInfo,
|
|
40
|
+
auto,
|
|
41
|
+
plugins,
|
|
42
|
+
evt,
|
|
43
|
+
mark,
|
|
44
|
+
logCall,
|
|
45
|
+
logError,
|
|
46
|
+
constants: { STREAM_TIMEOUT_MS, SLOW_TOTAL_MS, STALL_TIMEOUT_MS, SCORE_STALL_MS },
|
|
47
|
+
startedAt,
|
|
48
|
+
stages,
|
|
49
|
+
});
|
|
50
|
+
const r = await pipeline.execute({
|
|
51
|
+
res,
|
|
52
|
+
upRes: bufferedUpRes,
|
|
53
|
+
body,
|
|
54
|
+
requested,
|
|
55
|
+
actual: model,
|
|
56
|
+
lastErr,
|
|
57
|
+
via: "local",
|
|
58
|
+
lockModel,
|
|
59
|
+
useAuto,
|
|
60
|
+
handlerCtx,
|
|
61
|
+
mark,
|
|
62
|
+
perf0,
|
|
63
|
+
stages,
|
|
64
|
+
startedAt,
|
|
52
65
|
});
|
|
53
|
-
|
|
54
|
-
if (
|
|
55
|
-
if (auto) await auto.recordError(model, { status: 502, slow: true, note: `stream timeout ${STREAM_TIMEOUT_MS}ms` });
|
|
56
|
-
const err = { model, upstream: null, status: 502, message: `stream timed out after ${STREAM_TIMEOUT_MS}ms` };
|
|
57
|
-
logError(model, 502, `stream timeout ${STREAM_TIMEOUT_MS}ms`);
|
|
58
|
-
evt("upstream-error", { reqId: handlerCtx.reqId, model, status: 502, message: "stream timeout", timing: null });
|
|
59
|
-
evt("fallback", { reqId: handlerCtx.reqId, from: model, to: order[idx + 1] ?? null, reason: "stream timeout" });
|
|
60
|
-
return { handled: false, upRes: null, lastErr: err };
|
|
61
|
-
}
|
|
62
|
-
if (out.interrupted) {
|
|
63
|
-
if (auto) {
|
|
64
|
-
await auto.recordError(model, { status: 200, slow: true, note: `stall ${STALL_TIMEOUT_MS}ms` });
|
|
65
|
-
await auto.recordLatency(model, out.totalMs ?? (Date.now() - startedAt));
|
|
66
|
-
}
|
|
67
|
-
evt("slow-model", { model, elapsedMs: out.totalMs ?? (Date.now() - startedAt), threshold: STALL_TIMEOUT_MS, interrupted: true, detail: out.detail ?? null });
|
|
68
|
-
logCall(model, 200);
|
|
69
|
-
evt("result", { model, status: out.status, via: "local", timing: upRes._t ?? null, ttfMs: out.ttfMs, totalMs: out.totalMs, interrupted: true, detail: out.detail ?? null, fallback, requested, actual: model });
|
|
70
|
-
evt("client-response", { requested, actual: model, via: "local", fallback, status: out.status, reqId: handlerCtx.reqId });
|
|
71
|
-
if (plugins?.length) runHook(plugins, "request:completed", { reqId: handlerCtx.reqId, requested, useAuto, hops: handlerCtx.hops, stream: isStream, durationMs: Date.now() - startedAt, via: "local", status: out.status, actual: model, interrupted: true, fallback }).catch(() => {});
|
|
72
|
-
return { handled: true };
|
|
73
|
-
}
|
|
74
|
-
const elapsed = Date.now() - startedAt;
|
|
75
|
-
const latencyMs = out.totalMs ?? elapsed;
|
|
76
|
-
let scoredSlow = false;
|
|
77
|
-
if (SLOW_TOTAL_MS && auto && elapsed > SLOW_TOTAL_MS && out.status === 200) {
|
|
78
|
-
void auto.recordError(model, { status: 200, slow: true, note: `slow ${elapsed}ms` });
|
|
79
|
-
void auto.recordLatency(model, latencyMs);
|
|
80
|
-
evt("slow-model", { model, elapsedMs: elapsed, threshold: SLOW_TOTAL_MS, reason: "total", detail: out.detail ?? null });
|
|
81
|
-
scoredSlow = true;
|
|
82
|
-
}
|
|
83
|
-
if (out.detail?.stallHits > 0 && auto && out.status === 200) {
|
|
84
|
-
void auto.recordError(model, { status: 200, slow: true, note: `stall ${out.detail.stallHits}x gap>${SCORE_STALL_MS}ms maxGap ${out.detail.maxGapMs}ms` });
|
|
85
|
-
void auto.recordLatency(model, latencyMs);
|
|
86
|
-
evt("slow-model", { model, elapsedMs: elapsed, threshold: SCORE_STALL_MS, reason: "stall", stallHits: out.detail.stallHits, maxGapMs: out.detail.maxGapMs, detail: out.detail ?? null });
|
|
87
|
-
scoredSlow = true;
|
|
88
|
-
}
|
|
89
|
-
if (!scoredSlow && auto && out.status === 200) await auto.recordOk(model, { latencyMs });
|
|
90
|
-
else if (!scoredSlow && auto) await auto.recordLatency(model, latencyMs);
|
|
91
|
-
evt("result", { model, status: out.status, via: "local", timing: upRes._t ?? null, ttfMs: out.ttfMs, totalMs: out.totalMs, detail: out.detail ?? null, fallback, requested, actual: model, hedged: true });
|
|
92
|
-
evt("client-response", { requested, actual: model, via: "local", fallback, status: out.status, reqId: handlerCtx.reqId });
|
|
93
|
-
if (plugins?.length) runHook(plugins, "request:completed", { reqId: handlerCtx.reqId, requested, useAuto, hops: handlerCtx.hops, stream: isStream, durationMs: Date.now() - startedAt, via: "local", status: out.status, actual: model, fallback }).catch(() => {});
|
|
66
|
+
// 保留 hedged 语义:若超时则透传 needs fallback
|
|
67
|
+
if (!r.handled) return { handled: false, upRes: null, lastErr: r.lastErr };
|
|
94
68
|
return { handled: true };
|
|
95
|
-
}
|
|
69
|
+
}
|
|
70
|
+
if (hedged.winner === "peer" && hedged.peerInfo) {
|
|
96
71
|
const win = hedged.peerInfo;
|
|
97
72
|
evt("peer-race-win", { reqId: handlerCtx.reqId, model, winPeer: win.peer.url, winTarget: win.target, latencyMs: win.latencyMs, hedged: true, ttfMs: hedged.ttfMs });
|
|
98
73
|
await peers.recordResult(win.peer.url, { ok: true, latencyMs: win.latencyMs, model: win.target });
|
|
99
|
-
logCall(win.target, win.res.status);
|
|
100
|
-
const peerFallback = buildFallbackInfo({ requested, actual: win.target, lastErr, via: "peer", useAuto, lockModel });
|
|
101
|
-
if (peerFallback?.fallback) evt("fallback-notice", { reqId: handlerCtx.reqId, requested, actual: win.target, reason: peerFallback.reason, notice: peerFallback.notice, via: "peer" });
|
|
102
|
-
evt("relay-start", { reqId: handlerCtx.reqId, model: win.target, via: "peer", isStream, fallback: peerFallback, hedged: true });
|
|
103
74
|
const bufferedPeerRes = { ...win.res, body: hedged.bufferedBody, headers: win.res.headers, status: win.res.status, _t: win.res._t };
|
|
104
|
-
const
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
|
|
75
|
+
const pipeline = createRelayPipeline({
|
|
76
|
+
relay,
|
|
77
|
+
buildFallbackInfo,
|
|
78
|
+
auto,
|
|
79
|
+
plugins,
|
|
80
|
+
evt,
|
|
81
|
+
mark,
|
|
82
|
+
logCall,
|
|
83
|
+
logError: () => {},
|
|
84
|
+
constants: { STREAM_TIMEOUT_MS, SLOW_TOTAL_MS, STALL_TIMEOUT_MS, SCORE_STALL_MS },
|
|
85
|
+
startedAt,
|
|
86
|
+
stages,
|
|
87
|
+
});
|
|
88
|
+
await pipeline.execute({
|
|
89
|
+
res,
|
|
90
|
+
upRes: bufferedPeerRes,
|
|
91
|
+
body,
|
|
92
|
+
requested,
|
|
93
|
+
actual: win.target,
|
|
94
|
+
lastErr,
|
|
95
|
+
via: "peer",
|
|
96
|
+
lockModel,
|
|
97
|
+
useAuto,
|
|
98
|
+
handlerCtx: { ...handlerCtx, model: win.target },
|
|
99
|
+
mark: (n) => mark(n),
|
|
100
|
+
perf0,
|
|
101
|
+
stages,
|
|
102
|
+
startedAt,
|
|
108
103
|
});
|
|
109
|
-
evt("relay-done", { reqId: handlerCtx.reqId, model: win.target, via: "peer", status: out.status, ttfMs: out.ttfMs ?? hedged.ttfMs, totalMs: out.totalMs, aborted: out.aborted, interrupted: out.interrupted ?? false, detail: out.detail ?? null, hedged: true });
|
|
110
|
-
if (auto && out.status === 200) {
|
|
111
|
-
const latencyMs = out.totalMs ?? win.latencyMs;
|
|
112
|
-
if (out.detail?.stallHits > 0 || (latencyMs && latencyMs > SLOW_TOTAL_MS)) {
|
|
113
|
-
void auto.recordError(win.target, { status: 200, slow: true, note: `peer slow ${latencyMs}ms` });
|
|
114
|
-
void auto.recordLatency(win.target, latencyMs);
|
|
115
|
-
} else {
|
|
116
|
-
await auto.recordOk(win.target, { latencyMs });
|
|
117
|
-
}
|
|
118
|
-
}
|
|
119
|
-
evt("result", { model: win.target, status: out.status, via: "peer", timing: win.res._t ?? null, ttfMs: out.ttfMs, totalMs: out.totalMs, detail: out.detail ?? null, fallback: peerFallback, requested, actual: win.target, hedged: true });
|
|
120
|
-
evt("client-response", { requested, actual: win.target, via: "peer", fallback: peerFallback, status: out.status, reqId: handlerCtx.reqId });
|
|
121
|
-
if (plugins?.length) runHook(plugins, "request:completed", { reqId: handlerCtx.reqId, requested, useAuto, hops: handlerCtx.hops, stream: isStream, durationMs: Date.now() - startedAt, via: "peer", status: out.status, actual: win.target, fallback: peerFallback }).catch(() => {});
|
|
122
104
|
return { handled: true };
|
|
123
105
|
}
|
|
124
106
|
}
|
|
125
107
|
if (hedged && hedged.needsPeer) {
|
|
126
108
|
return { handled: false, upRes: null, lastErr, needsPeer: true };
|
|
127
|
-
}
|
|
109
|
+
}
|
|
110
|
+
if (!hedged || !hedged.winner) {
|
|
128
111
|
evt("hedge-both-fail", { reqId: handlerCtx.reqId, model });
|
|
129
112
|
if (!hedged) {
|
|
130
113
|
if (auto) await auto.recordError(model, { status: 502, slow: false, note: "hedge both fail" });
|
|
@@ -136,7 +119,6 @@ export async function handleHedge({
|
|
|
136
119
|
return { handled: false, upRes: null, lastErr };
|
|
137
120
|
} catch (hedgeErr) {
|
|
138
121
|
evt("hedge-error", { reqId: handlerCtx.reqId, model, error: String(hedgeErr?.message || hedgeErr).slice(0, 300) });
|
|
139
|
-
// 回退到串行
|
|
140
122
|
return { handled: false, upRes, lastErr };
|
|
141
123
|
}
|
|
142
124
|
}
|