mslxdff 0.1.112 → 0.1.114
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json
CHANGED
|
@@ -49,6 +49,8 @@ export async function runSerialTrial(ctx, deps = {}) {
|
|
|
49
49
|
for (let idx = 0; idx < order.length; idx++) {
|
|
50
50
|
const model = order[idx];
|
|
51
51
|
handlerCtx.model = model;
|
|
52
|
+
handlerCtx.orderLen = order.length;
|
|
53
|
+
handlerCtx.idx = idx;
|
|
52
54
|
evt("model-try", { reqId, model, idx, remaining: order.length - idx });
|
|
53
55
|
if (plugins?.length) {
|
|
54
56
|
const bt = await runHook(plugins, "model:beforeTry", { reqId, requested, model, idx, hops });
|
|
@@ -3,6 +3,13 @@ import { recordModelStats } from "../../state.js";
|
|
|
3
3
|
import { normalizeFullId } from "../../providers/model-id.js";
|
|
4
4
|
import { computeMetrics, extractUsageFromJson, extractUsageFromSseText } from "../../metrics.js";
|
|
5
5
|
|
|
6
|
+
// 唯一/最后候选没有 failover 去向:首块闸门退化为纯"防连接泄漏",放宽避免误杀慢模型
|
|
7
|
+
// (参考 opencode:zen 通道不设超时;openai responses 硬编码 300s headerTimeout)
|
|
8
|
+
const LAST_CANDIDATE_TIMEOUT_MS = (() => {
|
|
9
|
+
const n = Number(process.env.MSLXDFF_LAST_CANDIDATE_TIMEOUT_MS);
|
|
10
|
+
return Number.isInteger(n) && n >= 0 ? n : 120_000;
|
|
11
|
+
})();
|
|
12
|
+
|
|
6
13
|
/**
|
|
7
14
|
* RelayPipeline 深模块
|
|
8
15
|
* 把 5 个 handler 各自的 fallback→relay→scoring→事件 6段流水收敛为单一真相。
|
|
@@ -72,9 +79,16 @@ export function createRelayPipeline({
|
|
|
72
79
|
}
|
|
73
80
|
_evt("relay-start", { reqId, model: actual, via, isStream: Boolean(body?.stream), fallback });
|
|
74
81
|
|
|
75
|
-
// 3. relay
|
|
82
|
+
// 3. relay(唯一/最后候选:无 failover 去向 → 闸门放宽到防泄漏级别)
|
|
83
|
+
const orderLen = handlerCtx?.orderLen;
|
|
84
|
+
const curIdx = handlerCtx?.idx;
|
|
85
|
+
const isLastCandidate =
|
|
86
|
+
Number.isInteger(orderLen) && orderLen > 0 &&
|
|
87
|
+
(orderLen === 1 || (Number.isInteger(curIdx) && curIdx >= orderLen - 1));
|
|
88
|
+
const streamTimeoutMs = isLastCandidate ? LAST_CANDIDATE_TIMEOUT_MS : C.STREAM_TIMEOUT_MS;
|
|
76
89
|
const out = await _relay(res, upRes, body, {
|
|
77
90
|
fallback,
|
|
91
|
+
streamTimeoutMs,
|
|
78
92
|
onFirstChunk: (delta) => {
|
|
79
93
|
try { markFn(`ttf-${actual}`); } catch {}
|
|
80
94
|
_evt("relay-first-chunk", { reqId, model: actual, ttfMs: delta, via });
|
|
@@ -99,12 +113,12 @@ export function createRelayPipeline({
|
|
|
99
113
|
});
|
|
100
114
|
|
|
101
115
|
// 5a. 首块超时未写字节 → 回退
|
|
102
|
-
if (out.status ===
|
|
103
|
-
if (auto) try { await auto.recordError(actual, { status: 502, slow: true, note: `stream timeout ${
|
|
104
|
-
try { _logError(actual, 502, `stream timeout ${
|
|
116
|
+
if (streamTimeoutMs > 0 && out.status === streamTimeoutMs) {
|
|
117
|
+
if (auto) try { await auto.recordError(actual, { status: 502, slow: true, note: `stream timeout ${streamTimeoutMs}ms` }); } catch {}
|
|
118
|
+
try { _logError(actual, 502, `stream timeout ${streamTimeoutMs}ms`); } catch {}
|
|
105
119
|
_evt("upstream-error", { reqId, model: actual, status: 502, message: "stream timeout", timing: null });
|
|
106
120
|
_evt("fallback", { reqId, from: actual, to: null, reason: "stream timeout" });
|
|
107
|
-
return { handled: false, upRes: null, lastErr: { model: actual, upstream: null, status: 502, message: `stream timed out after ${
|
|
121
|
+
return { handled: false, upRes: null, lastErr: { model: actual, upstream: null, status: 502, message: `stream timed out after ${streamTimeoutMs}ms` } };
|
|
108
122
|
}
|
|
109
123
|
|
|
110
124
|
// 5b. 中断(stall 超时 / max 流时长)
|
package/src/routes/stream.js
CHANGED
|
@@ -2,6 +2,23 @@ import { performance } from "node:perf_hooks";
|
|
|
2
2
|
import { applyFallbackHeaders, enrichNonStreamJson, enrichSseChunkText } from "./fallback.js";
|
|
3
3
|
import { json } from "./helpers.js";
|
|
4
4
|
|
|
5
|
+
// SDK 通道(TextEncoder)产出 Uint8Array,legacy 通道为 Buffer;
|
|
6
|
+
// 统一转文本,避免 [DONE]/finish_reason/usage/chars 统计在 SDK 路径下静默失效。
|
|
7
|
+
function chunkText(chunk) {
|
|
8
|
+
if (typeof chunk === "string") return chunk;
|
|
9
|
+
if (Buffer.isBuffer(chunk)) return chunk.toString("utf8");
|
|
10
|
+
if (chunk instanceof Uint8Array) return Buffer.from(chunk).toString("utf8");
|
|
11
|
+
return "";
|
|
12
|
+
}
|
|
13
|
+
|
|
14
|
+
// body.cancel() 可能返回非 Promise(自定义/AI SDK 流)——同步异常与 rejection 双路径都要吞掉
|
|
15
|
+
function cancelBody(body) {
|
|
16
|
+
try {
|
|
17
|
+
const p = typeof body?.cancel === "function" ? body.cancel() : null;
|
|
18
|
+
if (p && typeof p.catch === "function") p.catch(() => {});
|
|
19
|
+
} catch { /* ignore */ }
|
|
20
|
+
}
|
|
21
|
+
|
|
5
22
|
export const SLOW_TOTAL_MS = (() => {
|
|
6
23
|
const n = Number(process.env.MSLXDFF_SLOW_TOTAL_MS);
|
|
7
24
|
return Number.isInteger(n) && n > 0 ? n : 20_000;
|
|
@@ -9,7 +26,8 @@ export const SLOW_TOTAL_MS = (() => {
|
|
|
9
26
|
|
|
10
27
|
export const STREAM_TIMEOUT_MS = (() => {
|
|
11
28
|
const n = Number(process.env.MSLXDFF_STREAM_TIMEOUT_MS);
|
|
12
|
-
|
|
29
|
+
// 0 = 显式关闭首块超时(慢思考模型专用);未设/非法值 → 默认 25s
|
|
30
|
+
return Number.isInteger(n) && n >= 0 ? n : 25_000;
|
|
13
31
|
})();
|
|
14
32
|
|
|
15
33
|
export const STALL_TIMEOUT_MS = (() => {
|
|
@@ -42,6 +60,8 @@ export async function relay(res, upRes, body, { onFirstChunk, onDownstreamAbort,
|
|
|
42
60
|
if (reason) res.setHeader("x-mslxdff-workbuddy-reason", reason);
|
|
43
61
|
const allow = upRes.headers.get("x-mslxdff-allowlist");
|
|
44
62
|
if (allow) res.setHeader("x-mslxdff-allowlist", allow);
|
|
63
|
+
const engine = upRes.headers.get("x-mslxdff-upstream-engine");
|
|
64
|
+
if (engine) res.setHeader("x-mslxdff-upstream-engine", engine);
|
|
45
65
|
} catch {}
|
|
46
66
|
if (fallback) applyFallbackHeaders(res, fallback);
|
|
47
67
|
|
|
@@ -64,6 +84,7 @@ export async function relay(res, upRes, body, { onFirstChunk, onDownstreamAbort,
|
|
|
64
84
|
downstreamClosed: false,
|
|
65
85
|
usage: null,
|
|
66
86
|
chars: 0,
|
|
87
|
+
recoveries: 0,
|
|
67
88
|
};
|
|
68
89
|
let prevChunkAt = t0;
|
|
69
90
|
const onClose = () => {
|
|
@@ -95,20 +116,22 @@ export async function relay(res, upRes, body, { onFirstChunk, onDownstreamAbort,
|
|
|
95
116
|
? setTimeout(() => {
|
|
96
117
|
stalled = true;
|
|
97
118
|
detail.exitReason = "stall";
|
|
98
|
-
|
|
119
|
+
cancelBody(upRes.body);
|
|
99
120
|
}, STALL_TIMEOUT_MS)
|
|
100
121
|
: null;
|
|
101
122
|
};
|
|
102
|
-
let firstTimer =
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
|
|
123
|
+
let firstTimer = streamTimeoutMs > 0
|
|
124
|
+
? setTimeout(() => {
|
|
125
|
+
timedOut = true;
|
|
126
|
+
detail.exitReason = "first-timeout";
|
|
127
|
+
cancelBody(upRes.body);
|
|
128
|
+
}, streamTimeoutMs)
|
|
129
|
+
: null;
|
|
107
130
|
const maxTimer = MAX_STREAM_MS
|
|
108
131
|
? setTimeout(() => {
|
|
109
132
|
tooLong = true;
|
|
110
133
|
detail.exitReason = "max";
|
|
111
|
-
|
|
134
|
+
cancelBody(upRes.body);
|
|
112
135
|
}, MAX_STREAM_MS)
|
|
113
136
|
: null;
|
|
114
137
|
try {
|
|
@@ -124,7 +147,7 @@ export async function relay(res, upRes, body, { onFirstChunk, onDownstreamAbort,
|
|
|
124
147
|
if (gap > SCORE_STALL_MS) detail.stallHits += 1;
|
|
125
148
|
prevChunkAt = now;
|
|
126
149
|
try {
|
|
127
|
-
const txt =
|
|
150
|
+
const txt = chunkText(chunk);
|
|
128
151
|
if (txt.includes("[DONE]")) detail.sawDone = true;
|
|
129
152
|
const m = txt.match(/"finish_reason"\s*:\s*"([^"]+)"/);
|
|
130
153
|
if (m) detail.sawFinishReason = m[1];
|
|
@@ -157,7 +180,7 @@ export async function relay(res, upRes, body, { onFirstChunk, onDownstreamAbort,
|
|
|
157
180
|
} else {
|
|
158
181
|
// 非 usage 的普通 delta 也累 chars
|
|
159
182
|
try {
|
|
160
|
-
const txt2 =
|
|
183
|
+
const txt2 = chunkText(chunk);
|
|
161
184
|
const ms = txt2.match(/"content"\s*:\s*"([^"]*)"/g);
|
|
162
185
|
if (ms) for (const mm of ms) {
|
|
163
186
|
const c = JSON.parse(`{${mm}}`);
|
|
@@ -166,6 +189,15 @@ export async function relay(res, upRes, body, { onFirstChunk, onDownstreamAbort,
|
|
|
166
189
|
} catch {}
|
|
167
190
|
}
|
|
168
191
|
} catch { /* ignore */ }
|
|
192
|
+
// 首块/空闲超时后上游仍吐出了数据 → 只是慢,不是死:撤销超时判定,照常转发
|
|
193
|
+
//(cancel 是协作式的,缓冲数据仍会到达;丢掉已到达的数据是纯损失)
|
|
194
|
+
if (timedOut || stalled) {
|
|
195
|
+
timedOut = false;
|
|
196
|
+
stalled = false;
|
|
197
|
+
detail.recoveries = (detail.recoveries || 0) + 1;
|
|
198
|
+
if (detail.exitReason === "first-timeout" || detail.exitReason === "stall") detail.exitReason = null;
|
|
199
|
+
if (firstTimer) { clearTimeout(firstTimer); firstTimer = null; }
|
|
200
|
+
}
|
|
169
201
|
if (timedOut || stalled || tooLong) break;
|
|
170
202
|
if (first) {
|
|
171
203
|
first = false;
|
|
@@ -205,7 +237,7 @@ export async function relay(res, upRes, body, { onFirstChunk, onDownstreamAbort,
|
|
|
205
237
|
}
|
|
206
238
|
if (timedOut && !wroteAny) {
|
|
207
239
|
res.removeListener("close", onClose);
|
|
208
|
-
return { status:
|
|
240
|
+
return { status: streamTimeoutMs, ttfMs: null, totalMs: Math.round(performance.now() - t0), aborted: true, interrupted: false, detail };
|
|
209
241
|
}
|
|
210
242
|
if ((stalled || tooLong) && wroteAny) {
|
|
211
243
|
interrupted = true;
|
|
@@ -2,12 +2,34 @@
|
|
|
2
2
|
// 纯函数序列化器:push(part) 返回该 part 产生的 SSE 文本(可能为空串/组合帧),end() 返回 [DONE]。
|
|
3
3
|
// 见 .scratch/workbuddy-sdk-channel/SPEC.md。
|
|
4
4
|
|
|
5
|
+
// 数值兼容:扁平 number / { total } 嵌套
|
|
6
|
+
function pickTokens(v) {
|
|
7
|
+
if (typeof v === "number") return v;
|
|
8
|
+
if (v && typeof v === "object" && typeof v.total === "number") return v.total;
|
|
9
|
+
return undefined;
|
|
10
|
+
}
|
|
11
|
+
function num(v) {
|
|
12
|
+
const n = Number(v);
|
|
13
|
+
return Number.isFinite(n) ? n : undefined;
|
|
14
|
+
}
|
|
15
|
+
|
|
5
16
|
export function usageToOpenAI(u) {
|
|
6
17
|
if (!u || typeof u !== "object") return undefined;
|
|
7
|
-
|
|
8
|
-
|
|
9
|
-
|
|
10
|
-
|
|
18
|
+
const raw = u.raw && typeof u.raw === "object" ? u.raw : null;
|
|
19
|
+
// raw 已是 OpenAI chat 口径 → 直接用(最保真)
|
|
20
|
+
if (raw && (raw.prompt_tokens != null || raw.completion_tokens != null || raw.total_tokens != null)) return raw;
|
|
21
|
+
// 否则从 raw(Responses 口径 input_tokens/output_tokens)或 V2 扁平 / V3 嵌套标准字段归一
|
|
22
|
+
const src = raw || u;
|
|
23
|
+
const prompt = num(src.input_tokens) ?? pickTokens(src.inputTokens) ?? pickTokens(u.inputTokens) ?? 0;
|
|
24
|
+
const completion = num(src.output_tokens) ?? pickTokens(src.outputTokens) ?? pickTokens(u.outputTokens) ?? 0;
|
|
25
|
+
const total = num(src.total_tokens) ?? num(src.totalTokens) ?? pickTokens(u.totalTokens) ?? (prompt + completion);
|
|
26
|
+
const out = { prompt_tokens: prompt, completion_tokens: completion, total_tokens: total };
|
|
27
|
+
// Responses 口径的 details 映射到 OpenAI chat details(有则带上)
|
|
28
|
+
const cached = num(src.input_tokens_details?.cached_tokens ?? src.prompt_tokens_details?.cached_tokens);
|
|
29
|
+
if (cached != null) out.prompt_tokens_details = { cached_tokens: cached };
|
|
30
|
+
const reasoningTokens = num(src.output_tokens_details?.reasoning_tokens ?? src.completion_tokens_details?.reasoning_tokens);
|
|
31
|
+
if (reasoningTokens != null) out.completion_tokens_details = { reasoning_tokens: reasoningTokens };
|
|
32
|
+
return out;
|
|
11
33
|
}
|
|
12
34
|
|
|
13
35
|
export function createSseSerializer() {
|
|
@@ -72,7 +94,10 @@ export function createSseSerializer() {
|
|
|
72
94
|
return ensureRole() + frame({ tool_calls: [{ index: idx, id, type: "function", function: { name: part.toolName || "", arguments: args } }] });
|
|
73
95
|
}
|
|
74
96
|
case "finish": {
|
|
75
|
-
|
|
97
|
+
// V2 spec 的 finishReason 是字符串;V3 spec 是 { unified, raw }
|
|
98
|
+
const fr = part.finishReason;
|
|
99
|
+
const raw = typeof fr === "string" ? fr : (fr?.raw ?? fr?.unified);
|
|
100
|
+
const reason = raw === "tool-calls" ? "tool_calls" : raw === "content-filter" ? "content_filter" : (raw ?? "stop");
|
|
76
101
|
return frame({}, { finishReason: reason, usage: usageToOpenAI(part.usage) });
|
|
77
102
|
}
|
|
78
103
|
case "error": {
|