mslxdff 0.1.113 → 0.1.115
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/src/chat-pipeline/auto-race.js +1 -0
- package/src/chat-pipeline/index.js +5 -1
- package/src/chat-pipeline/serial-trial.js +3 -0
- package/src/model-capabilities/index.js +9 -2
- package/src/model-capabilities/parse.js +2 -0
- package/src/providers/dispatcher.js +2 -2
- package/src/providers/opencode.js +1 -1
- package/src/responses/translate.js +72 -20
- package/src/routes/chat/relay-pipeline.js +19 -5
- package/src/routes/stream.js +44 -10
- package/src/runtime/server-lifecycle.js +18 -0
- package/src/upstream-engine/sdk/attempt.js +8 -3
- package/src/upstream-engine/sdk/convert.js +15 -1
- package/src/upstream-engine/sdk/responses.js +62 -3
- package/src/upstream-engine/sdk/sse.js +39 -5
- package/src/upstream-responses.js +19 -1
- package/src/upstream.js +28 -5
package/package.json
CHANGED
|
@@ -65,6 +65,7 @@ export async function runAutoRace(ctx, deps = {}) {
|
|
|
65
65
|
const o = {};
|
|
66
66
|
if (Object.keys(shareKeys).length) o.shareKeys = shareKeys;
|
|
67
67
|
if (workbuddyUid) o.workbuddyUid = workbuddyUid;
|
|
68
|
+
if (handlerCtx?.sessionId) o.sessionId = handlerCtx.sessionId;
|
|
68
69
|
r = await upstream.chat(f, Object.keys(o).length ? o : undefined);
|
|
69
70
|
} catch (e) {
|
|
70
71
|
if (plugins?.length) runHook(plugins, "upstream:response", { reqId, requested, model: m, status: null, ok: false, error: errMsg(e), timing: e?._t ?? null }).catch(() => {});
|
|
@@ -82,7 +82,11 @@ export function createChatPipeline({ upstream, auto, logs, peers, groups, bus, t
|
|
|
82
82
|
for (const e of sel.errors) evt("plugin-hook-error", { reqId, hook: "model:select", plugin: e.plugin, error: e.error });
|
|
83
83
|
}
|
|
84
84
|
|
|
85
|
-
|
|
85
|
+
// 客户端会话标识(opencode 插件 chat.headers 注入)→ 透传上游做粘性路由/缓存亲和;
|
|
86
|
+
// 无头时由 upstream.js 按对话首两条消息哈希兜底(不再每请求随机)
|
|
87
|
+
const clientSession = String(req?.headers?.["x-session-affinity"] || req?.headers?.["x-session-id"] || "").trim() || null;
|
|
88
|
+
const handlerCtx = { reqId, model: null, body: req?.body, hops, peers, plugins, evt, logError, logCall, logs, workbuddyUid, sessionId: clientSession };
|
|
89
|
+
if (clientSession) evt("client-session", { reqId, sessionId: clientSession.slice(0, 24) });
|
|
86
90
|
|
|
87
91
|
const plan = planRoute(policy, {
|
|
88
92
|
candidates: order,
|
|
@@ -49,6 +49,8 @@ export async function runSerialTrial(ctx, deps = {}) {
|
|
|
49
49
|
for (let idx = 0; idx < order.length; idx++) {
|
|
50
50
|
const model = order[idx];
|
|
51
51
|
handlerCtx.model = model;
|
|
52
|
+
handlerCtx.orderLen = order.length;
|
|
53
|
+
handlerCtx.idx = idx;
|
|
52
54
|
evt("model-try", { reqId, model, idx, remaining: order.length - idx });
|
|
53
55
|
if (plugins?.length) {
|
|
54
56
|
const bt = await runHook(plugins, "model:beforeTry", { reqId, requested, model, idx, hops });
|
|
@@ -68,6 +70,7 @@ export async function runSerialTrial(ctx, deps = {}) {
|
|
|
68
70
|
const chatOpts = {};
|
|
69
71
|
if (Object.keys(shareKeys).length) chatOpts.shareKeys = shareKeys;
|
|
70
72
|
if (workbuddyUid) chatOpts.workbuddyUid = workbuddyUid;
|
|
73
|
+
if (handlerCtx?.sessionId) chatOpts.sessionId = handlerCtx.sessionId;
|
|
71
74
|
upRes = await upstream.chat(forwarded, Object.keys(chatOpts).length ? chatOpts : undefined);
|
|
72
75
|
evt("upstream-done", { reqId, model, ok: !(upRes instanceof Error) && upRes.status < 400, status: upRes instanceof Error ? null : upRes.status, timing: upRes._t ?? null, error: null });
|
|
73
76
|
} catch (err) {
|
|
@@ -28,15 +28,22 @@ export function createCapabilitiesService({
|
|
|
28
28
|
} = {}) {
|
|
29
29
|
let raw = null; // 原始目录(全 provider)
|
|
30
30
|
let capsIndex = new Map(); // providerId -> { modelId -> caps }
|
|
31
|
+
let npmIndex = new Map(); // opencode 裸 modelId -> provider.npm(responses 判定用;null = 继承默认)
|
|
31
32
|
let loadedAt = 0;
|
|
32
33
|
let inflight = null;
|
|
33
34
|
|
|
34
35
|
function buildIndex(data) {
|
|
35
36
|
const idx = new Map();
|
|
37
|
+
const npm = new Map();
|
|
36
38
|
for (const [pid, p] of Object.entries(data || {})) {
|
|
37
39
|
if (!p || typeof p !== "object") continue;
|
|
38
|
-
|
|
40
|
+
const caps = normalizeProviderModels(p.models || {});
|
|
41
|
+
idx.set(pid, caps);
|
|
42
|
+
if (String(pid).toLowerCase() === "opencode") {
|
|
43
|
+
for (const [mid, c] of Object.entries(caps)) npm.set(mid, c.npm ?? null);
|
|
44
|
+
}
|
|
39
45
|
}
|
|
46
|
+
npmIndex = npm;
|
|
40
47
|
return idx;
|
|
41
48
|
}
|
|
42
49
|
|
|
@@ -111,7 +118,7 @@ export function createCapabilitiesService({
|
|
|
111
118
|
return [...capsIndex.keys()].sort();
|
|
112
119
|
}
|
|
113
120
|
|
|
114
|
-
return { ready, get, list, providers };
|
|
121
|
+
return { ready, get, list, providers, npmIndex: () => new Map(npmIndex) };
|
|
115
122
|
}
|
|
116
123
|
|
|
117
124
|
// 模块级单例(与 globalDedup 同模式):HTTP handler 懒加载,测试 _reset 后注入
|
|
@@ -24,6 +24,8 @@ export function normalizeModelCaps(_id, m) {
|
|
|
24
24
|
releaseDate: typeof m?.release_date === "string" && m.release_date ? m.release_date : null,
|
|
25
25
|
inputModalities: input.length ? input : ["text"],
|
|
26
26
|
outputModalities: Array.isArray(m?.modalities?.output) && m.modalities.output.length ? m.modalities.output : ["text"],
|
|
27
|
+
// 模型级 SDK 覆盖(models.dev provider.npm):@ai-sdk/openai → responses 端点;null = 继承 provider 默认
|
|
28
|
+
npm: typeof m?.provider?.npm === "string" && m.provider.npm ? m.provider.npm : null,
|
|
27
29
|
};
|
|
28
30
|
}
|
|
29
31
|
|
|
@@ -53,9 +53,9 @@ export function createProviderDispatcher(providers = [], opts = {}) {
|
|
|
53
53
|
return provider.chatWithKeys(forwarded, sharedKeys);
|
|
54
54
|
}
|
|
55
55
|
if (provider.id === "workbuddy" && workbuddyUid) {
|
|
56
|
-
return provider.chat(forwarded, { workbuddyUid });
|
|
56
|
+
return provider.chat(forwarded, { ...opts, workbuddyUid });
|
|
57
57
|
}
|
|
58
|
-
return provider.chat(forwarded);
|
|
58
|
+
return provider.chat(forwarded, opts);
|
|
59
59
|
}
|
|
60
60
|
|
|
61
61
|
// 聚合所有供应商的模型列表;默认供应商(opencode)裸 id,其它带前缀
|
|
@@ -8,7 +8,7 @@ export function createOpenCodeProvider({ upstream, modelsService, baseUrl, authT
|
|
|
8
8
|
return {
|
|
9
9
|
id: "opencode",
|
|
10
10
|
upstream: client,
|
|
11
|
-
chat: (body) => client.chat(body),
|
|
11
|
+
chat: (body, opts) => client.chat(body, opts),
|
|
12
12
|
preheat: (args) => client.preheat(args),
|
|
13
13
|
close: () => client.close(),
|
|
14
14
|
async listModels() {
|
|
@@ -33,15 +33,27 @@ export function responsesToChatBody(req = {}) {
|
|
|
33
33
|
if (req.instructions) messages.push({ role: "system", content: String(req.instructions) });
|
|
34
34
|
const input = req.input;
|
|
35
35
|
const items = typeof input === "string" ? [{ type: "message", role: "user", content: input }] : Array.isArray(input) ? input : [];
|
|
36
|
+
// 加密思考往返:input 里的 reasoning item 挂到下一条 assistant 消息(thinking 跨轮必需,不能丢)
|
|
37
|
+
let pendingReasoning = [];
|
|
36
38
|
for (const it of items) {
|
|
37
39
|
if (!it || typeof it !== "object") continue;
|
|
40
|
+
if (it.type === "reasoning") {
|
|
41
|
+
if (it.id || it.encrypted_content) {
|
|
42
|
+
pendingReasoning.push({ id: it.id, encrypted_content: it.encrypted_content, summary: it.summary });
|
|
43
|
+
}
|
|
44
|
+
continue;
|
|
45
|
+
}
|
|
38
46
|
if (it.type === "message") {
|
|
39
|
-
|
|
47
|
+
const msg = { role: it.role || "user", content: inputTextOf(it.content) };
|
|
48
|
+
if (pendingReasoning.length && msg.role === "assistant") { msg.reasoning_items = pendingReasoning; pendingReasoning = []; }
|
|
49
|
+
messages.push(msg);
|
|
40
50
|
} else if (it.type === "function_call") {
|
|
41
|
-
|
|
51
|
+
const msg = {
|
|
42
52
|
role: "assistant", content: "",
|
|
43
53
|
tool_calls: [{ id: it.call_id || it.id || "", type: "function", function: { name: it.name || "", arguments: it.arguments || "" } }],
|
|
44
|
-
}
|
|
54
|
+
};
|
|
55
|
+
if (pendingReasoning.length) { msg.reasoning_items = pendingReasoning; pendingReasoning = []; }
|
|
56
|
+
messages.push(msg);
|
|
45
57
|
} else if (it.type === "function_call_output") {
|
|
46
58
|
messages.push({ role: "tool", tool_call_id: it.call_id || "", content: typeof it.output === "string" ? it.output : JSON.stringify(it.output ?? "") });
|
|
47
59
|
}
|
|
@@ -115,7 +127,16 @@ export function createChunkTranslator(model = "") {
|
|
|
115
127
|
let textLen = 0;
|
|
116
128
|
let lastFinish = "stop";
|
|
117
129
|
let lastUsage = null;
|
|
118
|
-
const tools = new Map(); // index → {id, name, args, announced}
|
|
130
|
+
const tools = new Map(); // index → {id, name, args, announced, outputIndex}
|
|
131
|
+
// output_index 按 item 出现顺序动态分配(reasoning 可能先于 message/tool 出现)
|
|
132
|
+
let nextIdx = 0;
|
|
133
|
+
let textIdx = null;
|
|
134
|
+
// 加密思考(thinking 跨轮):sse.js 首帧带 x_reasoning_item(含加密态),translator 据此建 reasoning item
|
|
135
|
+
let reasoningOpen = false;
|
|
136
|
+
let reasoningId = null;
|
|
137
|
+
let reasoningEncrypted = null;
|
|
138
|
+
let reasoningIdx = null;
|
|
139
|
+
let reasoningText = "";
|
|
119
140
|
const ev = (type, extra = {}) => ({ type, ...extra });
|
|
120
141
|
|
|
121
142
|
function begin() {
|
|
@@ -125,8 +146,9 @@ export function createChunkTranslator(model = "") {
|
|
|
125
146
|
function ensureTextItem() {
|
|
126
147
|
if (textItemOpen) return [];
|
|
127
148
|
textItemOpen = true;
|
|
149
|
+
textIdx = nextIdx++;
|
|
128
150
|
const item = { type: "message", id: `msg_${id}`, status: "in_progress", role: "assistant", content: [{ type: "output_text", text: "", annotations: [] }] };
|
|
129
|
-
return [ev("response.output_item.added", { output_index:
|
|
151
|
+
return [ev("response.output_item.added", { output_index: textIdx, item }), ev("response.content_part.added", { item_id: item.id, output_index: textIdx, content_index: 0, part: { type: "output_text", text: "", annotations: [] } })];
|
|
130
152
|
}
|
|
131
153
|
|
|
132
154
|
let fullText = "";
|
|
@@ -135,26 +157,51 @@ export function createChunkTranslator(model = "") {
|
|
|
135
157
|
const out = ensureTextItem();
|
|
136
158
|
textLen += delta.length;
|
|
137
159
|
fullText += delta;
|
|
138
|
-
out.push(ev("response.output_text.delta", { item_id: `msg_${id}`, output_index:
|
|
160
|
+
out.push(ev("response.output_text.delta", { item_id: `msg_${id}`, output_index: textIdx, content_index: 0, delta }));
|
|
161
|
+
return out;
|
|
162
|
+
}
|
|
163
|
+
|
|
164
|
+
function pushReasoning(delta, meta) {
|
|
165
|
+
const out = ensureReasoningItem(meta);
|
|
166
|
+
reasoningText += delta;
|
|
167
|
+
if (delta) out.push(ev("response.reasoning_summary_text.delta", { item_id: reasoningId, output_index: reasoningIdx, summary_index: 0, delta }));
|
|
139
168
|
return out;
|
|
140
169
|
}
|
|
141
170
|
|
|
171
|
+
function ensureReasoningItem(meta) {
|
|
172
|
+
if (reasoningOpen) {
|
|
173
|
+
// 补丁:加密态可能晚到(sse.js 在收尾帧补发无文本的 x_reasoning_item)
|
|
174
|
+
if (!reasoningEncrypted && typeof meta?.encrypted_content === "string") reasoningEncrypted = meta.encrypted_content;
|
|
175
|
+
return [];
|
|
176
|
+
}
|
|
177
|
+
reasoningOpen = true;
|
|
178
|
+
reasoningId = meta?.id || `rs_${id}`;
|
|
179
|
+
reasoningEncrypted = typeof meta?.encrypted_content === "string" ? meta.encrypted_content : null;
|
|
180
|
+
reasoningIdx = nextIdx++;
|
|
181
|
+
const item = { type: "reasoning", id: reasoningId, summary: [], ...(reasoningEncrypted ? { encrypted_content: reasoningEncrypted } : {}) };
|
|
182
|
+
return [
|
|
183
|
+
ev("response.output_item.added", { output_index: reasoningIdx, item }),
|
|
184
|
+
ev("response.reasoning_summary_part.added", { item_id: reasoningId, output_index: reasoningIdx, summary_index: 0, part: { type: "summary_text", text: "" } }),
|
|
185
|
+
];
|
|
186
|
+
}
|
|
187
|
+
|
|
142
188
|
function pushTool(tc) {
|
|
143
189
|
const idx = Number(tc.index ?? 0);
|
|
144
190
|
let t = tools.get(idx);
|
|
145
|
-
if (!t) { t = { id: "", name: "", args: "", announced: false }; tools.set(idx, t); }
|
|
191
|
+
if (!t) { t = { id: "", name: "", args: "", announced: false, outputIndex: null }; tools.set(idx, t); }
|
|
146
192
|
if (tc.id) t.id = tc.id;
|
|
147
193
|
if (tc.function?.name) t.name += tc.function.name;
|
|
148
194
|
const frag = tc.function?.arguments || "";
|
|
149
195
|
const out = [];
|
|
150
196
|
if (!t.announced && t.id && t.name) {
|
|
151
197
|
t.announced = true;
|
|
152
|
-
|
|
198
|
+
t.outputIndex = nextIdx++;
|
|
199
|
+
out.push(ev("response.output_item.added", { output_index: t.outputIndex, item: { type: "function_call", id: `fc_${id}_${idx}`, call_id: t.id, name: t.name, arguments: "" } }));
|
|
153
200
|
}
|
|
154
201
|
if (frag) {
|
|
155
202
|
if (!t.announced) { t.args += frag; return out; } // id/name 未到先攒着
|
|
156
203
|
t.args += frag;
|
|
157
|
-
out.push(ev("response.function_call_arguments.delta", { item_id: `fc_${id}_${idx}`, output_index:
|
|
204
|
+
out.push(ev("response.function_call_arguments.delta", { item_id: `fc_${id}_${idx}`, output_index: t.outputIndex, delta: frag }));
|
|
158
205
|
}
|
|
159
206
|
return out;
|
|
160
207
|
}
|
|
@@ -181,29 +228,34 @@ export function createChunkTranslator(model = "") {
|
|
|
181
228
|
if (typeof delta.content === "string" && delta.content) { dbg.textChars += delta.content.length; out.push(...pushText(delta.content)); }
|
|
182
229
|
for (const tc of delta.tool_calls || []) { dbg.toolDeltas++; out.push(...pushTool(tc)); }
|
|
183
230
|
const rc = typeof delta.reasoning_content === "string" ? delta.reasoning_content : "";
|
|
184
|
-
if (rc) { dbg.reasoningChars += rc.length; out.push(...
|
|
231
|
+
if (rc || chunk.x_reasoning_item) { if (rc) dbg.reasoningChars += rc.length; out.push(...pushReasoning(rc, chunk.x_reasoning_item)); }
|
|
185
232
|
}
|
|
186
233
|
return out;
|
|
187
234
|
}
|
|
188
235
|
|
|
189
236
|
function end({ finish = "stop", usage = null } = {}) {
|
|
190
237
|
const out = [];
|
|
238
|
+
if (reasoningOpen) {
|
|
239
|
+
out.push(ev("response.reasoning_summary_text.done", { item_id: reasoningId, output_index: reasoningIdx, summary_index: 0, text: reasoningText }));
|
|
240
|
+
out.push(ev("response.reasoning_summary_part.done", { item_id: reasoningId, output_index: reasoningIdx, summary_index: 0, part: { type: "summary_text", text: reasoningText } }));
|
|
241
|
+
out.push(ev("response.output_item.done", { output_index: reasoningIdx, item: { type: "reasoning", id: reasoningId, summary: [{ type: "summary_text", text: reasoningText }], ...(reasoningEncrypted ? { encrypted_content: reasoningEncrypted } : {}) } }));
|
|
242
|
+
}
|
|
191
243
|
if (textItemOpen) {
|
|
192
|
-
out.push(ev("response.output_text.done", { item_id: `msg_${id}`, output_index:
|
|
193
|
-
out.push(ev("response.content_part.done", { item_id: `msg_${id}`, output_index:
|
|
194
|
-
out.push(ev("response.output_item.done", { output_index:
|
|
244
|
+
out.push(ev("response.output_text.done", { item_id: `msg_${id}`, output_index: textIdx, content_index: 0, text: fullText }));
|
|
245
|
+
out.push(ev("response.content_part.done", { item_id: `msg_${id}`, output_index: textIdx, content_index: 0, part: { type: "output_text", text: fullText, annotations: [] } }));
|
|
246
|
+
out.push(ev("response.output_item.done", { output_index: textIdx, item: { type: "message", id: `msg_${id}`, status: "completed", role: "assistant", content: [{ type: "output_text", text: fullText, annotations: [] }] } }));
|
|
195
247
|
}
|
|
196
|
-
let i = 0;
|
|
197
248
|
for (const [idx, t] of tools) {
|
|
198
|
-
if (!t.id && !t.name && !t.args)
|
|
249
|
+
if (!t.id && !t.name && !t.args) continue;
|
|
199
250
|
if (!t.announced) {
|
|
200
|
-
|
|
251
|
+
t.announced = true;
|
|
252
|
+
t.outputIndex = nextIdx++;
|
|
253
|
+
out.push(ev("response.output_item.added", { output_index: t.outputIndex, item: { type: "function_call", id: `fc_${id}_${idx}`, call_id: t.id, name: t.name, arguments: "" } }));
|
|
201
254
|
}
|
|
202
|
-
if (t.args) out.push(ev("response.function_call_arguments.done", { item_id: `fc_${id}_${idx}`, output_index:
|
|
203
|
-
out.push(ev("response.output_item.done", { output_index:
|
|
204
|
-
i++;
|
|
255
|
+
if (t.args) out.push(ev("response.function_call_arguments.done", { item_id: `fc_${id}_${idx}`, output_index: t.outputIndex, arguments: t.args }));
|
|
256
|
+
out.push(ev("response.output_item.done", { output_index: t.outputIndex, item: { type: "function_call", id: `fc_${id}_${idx}`, call_id: t.id, name: t.name, arguments: t.args, status: "completed" } }));
|
|
205
257
|
}
|
|
206
|
-
void
|
|
258
|
+
void textLen;
|
|
207
259
|
out.push(ev("response.completed", { response: { id, object: "response", created_at: createdAt, status: finish === "length" ? "incomplete" : "completed", model, output: [], usage: toResponsesUsage(usage) } }));
|
|
208
260
|
return out;
|
|
209
261
|
}
|
|
@@ -3,6 +3,13 @@ import { recordModelStats } from "../../state.js";
|
|
|
3
3
|
import { normalizeFullId } from "../../providers/model-id.js";
|
|
4
4
|
import { computeMetrics, extractUsageFromJson, extractUsageFromSseText } from "../../metrics.js";
|
|
5
5
|
|
|
6
|
+
// 唯一/最后候选没有 failover 去向:首块闸门退化为纯"防连接泄漏",放宽避免误杀慢模型
|
|
7
|
+
// (参考 opencode:zen 通道不设超时;openai responses 硬编码 300s headerTimeout)
|
|
8
|
+
const LAST_CANDIDATE_TIMEOUT_MS = (() => {
|
|
9
|
+
const n = Number(process.env.MSLXDFF_LAST_CANDIDATE_TIMEOUT_MS);
|
|
10
|
+
return Number.isInteger(n) && n >= 0 ? n : 120_000;
|
|
11
|
+
})();
|
|
12
|
+
|
|
6
13
|
/**
|
|
7
14
|
* RelayPipeline 深模块
|
|
8
15
|
* 把 5 个 handler 各自的 fallback→relay→scoring→事件 6段流水收敛为单一真相。
|
|
@@ -72,9 +79,16 @@ export function createRelayPipeline({
|
|
|
72
79
|
}
|
|
73
80
|
_evt("relay-start", { reqId, model: actual, via, isStream: Boolean(body?.stream), fallback });
|
|
74
81
|
|
|
75
|
-
// 3. relay
|
|
82
|
+
// 3. relay(唯一/最后候选:无 failover 去向 → 闸门放宽到防泄漏级别)
|
|
83
|
+
const orderLen = handlerCtx?.orderLen;
|
|
84
|
+
const curIdx = handlerCtx?.idx;
|
|
85
|
+
const isLastCandidate =
|
|
86
|
+
Number.isInteger(orderLen) && orderLen > 0 &&
|
|
87
|
+
(orderLen === 1 || (Number.isInteger(curIdx) && curIdx >= orderLen - 1));
|
|
88
|
+
const streamTimeoutMs = isLastCandidate ? LAST_CANDIDATE_TIMEOUT_MS : C.STREAM_TIMEOUT_MS;
|
|
76
89
|
const out = await _relay(res, upRes, body, {
|
|
77
90
|
fallback,
|
|
91
|
+
streamTimeoutMs,
|
|
78
92
|
onFirstChunk: (delta) => {
|
|
79
93
|
try { markFn(`ttf-${actual}`); } catch {}
|
|
80
94
|
_evt("relay-first-chunk", { reqId, model: actual, ttfMs: delta, via });
|
|
@@ -99,12 +113,12 @@ export function createRelayPipeline({
|
|
|
99
113
|
});
|
|
100
114
|
|
|
101
115
|
// 5a. 首块超时未写字节 → 回退
|
|
102
|
-
if (out.status ===
|
|
103
|
-
if (auto) try { await auto.recordError(actual, { status: 502, slow: true, note: `stream timeout ${
|
|
104
|
-
try { _logError(actual, 502, `stream timeout ${
|
|
116
|
+
if (streamTimeoutMs > 0 && out.status === streamTimeoutMs) {
|
|
117
|
+
if (auto) try { await auto.recordError(actual, { status: 502, slow: true, note: `stream timeout ${streamTimeoutMs}ms` }); } catch {}
|
|
118
|
+
try { _logError(actual, 502, `stream timeout ${streamTimeoutMs}ms`); } catch {}
|
|
105
119
|
_evt("upstream-error", { reqId, model: actual, status: 502, message: "stream timeout", timing: null });
|
|
106
120
|
_evt("fallback", { reqId, from: actual, to: null, reason: "stream timeout" });
|
|
107
|
-
return { handled: false, upRes: null, lastErr: { model: actual, upstream: null, status: 502, message: `stream timed out after ${
|
|
121
|
+
return { handled: false, upRes: null, lastErr: { model: actual, upstream: null, status: 502, message: `stream timed out after ${streamTimeoutMs}ms` } };
|
|
108
122
|
}
|
|
109
123
|
|
|
110
124
|
// 5b. 中断(stall 超时 / max 流时长)
|
package/src/routes/stream.js
CHANGED
|
@@ -11,6 +11,14 @@ function chunkText(chunk) {
|
|
|
11
11
|
return "";
|
|
12
12
|
}
|
|
13
13
|
|
|
14
|
+
// body.cancel() 可能返回非 Promise(自定义/AI SDK 流)——同步异常与 rejection 双路径都要吞掉
|
|
15
|
+
function cancelBody(body) {
|
|
16
|
+
try {
|
|
17
|
+
const p = typeof body?.cancel === "function" ? body.cancel() : null;
|
|
18
|
+
if (p && typeof p.catch === "function") p.catch(() => {});
|
|
19
|
+
} catch { /* ignore */ }
|
|
20
|
+
}
|
|
21
|
+
|
|
14
22
|
export const SLOW_TOTAL_MS = (() => {
|
|
15
23
|
const n = Number(process.env.MSLXDFF_SLOW_TOTAL_MS);
|
|
16
24
|
return Number.isInteger(n) && n > 0 ? n : 20_000;
|
|
@@ -18,7 +26,14 @@ export const SLOW_TOTAL_MS = (() => {
|
|
|
18
26
|
|
|
19
27
|
export const STREAM_TIMEOUT_MS = (() => {
|
|
20
28
|
const n = Number(process.env.MSLXDFF_STREAM_TIMEOUT_MS);
|
|
21
|
-
|
|
29
|
+
// 0 = 显式关闭首块超时(慢思考模型专用);未设/非法值 → 默认 25s
|
|
30
|
+
return Number.isInteger(n) && n >= 0 ? n : 25_000;
|
|
31
|
+
})();
|
|
32
|
+
|
|
33
|
+
// 等首块期间的心跳间隔(SSE 注释帧,标准客户端忽略):上游偶发卡 90s+,避免客户端误判卡死/断连
|
|
34
|
+
export const KEEPALIVE_MS = (() => {
|
|
35
|
+
const n = Number(process.env.MSLXDFF_KEEPALIVE_MS);
|
|
36
|
+
return Number.isInteger(n) && n >= 0 ? n : 10_000;
|
|
22
37
|
})();
|
|
23
38
|
|
|
24
39
|
export const STALL_TIMEOUT_MS = (() => {
|
|
@@ -37,7 +52,7 @@ export const MAX_STREAM_MS = (() => {
|
|
|
37
52
|
return Number.isInteger(n) && n > 0 ? n : 0;
|
|
38
53
|
})();
|
|
39
54
|
|
|
40
|
-
export async function relay(res, upRes, body, { onFirstChunk, onDownstreamAbort, streamTimeoutMs = STREAM_TIMEOUT_MS, fallback } = {}) {
|
|
55
|
+
export async function relay(res, upRes, body, { onFirstChunk, onDownstreamAbort, streamTimeoutMs = STREAM_TIMEOUT_MS, keepaliveMs = KEEPALIVE_MS, fallback } = {}) {
|
|
41
56
|
const t0 = performance.now();
|
|
42
57
|
const contentType = upRes.headers.get("content-type") || "";
|
|
43
58
|
// 需同时满足:客户端要流 + 上游真的是 SSE;避免 muse-spark 聚合 JSON 被误判为流式,或 workbuddy SSE 被聚合
|
|
@@ -75,6 +90,7 @@ export async function relay(res, upRes, body, { onFirstChunk, onDownstreamAbort,
|
|
|
75
90
|
downstreamClosed: false,
|
|
76
91
|
usage: null,
|
|
77
92
|
chars: 0,
|
|
93
|
+
recoveries: 0,
|
|
78
94
|
};
|
|
79
95
|
let prevChunkAt = t0;
|
|
80
96
|
const onClose = () => {
|
|
@@ -100,26 +116,34 @@ export async function relay(res, upRes, body, { onFirstChunk, onDownstreamAbort,
|
|
|
100
116
|
let stalled = false;
|
|
101
117
|
let tooLong = false;
|
|
102
118
|
let stallTimer = null;
|
|
119
|
+
let pingTimer = keepaliveMs > 0
|
|
120
|
+
? setInterval(() => {
|
|
121
|
+
if (wroteAny) return;
|
|
122
|
+
try { res.write(": keepalive\n\n"); } catch { /* ignore */ }
|
|
123
|
+
}, keepaliveMs)
|
|
124
|
+
: null;
|
|
103
125
|
const armStall = () => {
|
|
104
126
|
if (stallTimer) clearTimeout(stallTimer);
|
|
105
127
|
stallTimer = STALL_TIMEOUT_MS
|
|
106
128
|
? setTimeout(() => {
|
|
107
129
|
stalled = true;
|
|
108
130
|
detail.exitReason = "stall";
|
|
109
|
-
|
|
131
|
+
cancelBody(upRes.body);
|
|
110
132
|
}, STALL_TIMEOUT_MS)
|
|
111
133
|
: null;
|
|
112
134
|
};
|
|
113
|
-
let firstTimer =
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
|
|
135
|
+
let firstTimer = streamTimeoutMs > 0
|
|
136
|
+
? setTimeout(() => {
|
|
137
|
+
timedOut = true;
|
|
138
|
+
detail.exitReason = "first-timeout";
|
|
139
|
+
cancelBody(upRes.body);
|
|
140
|
+
}, streamTimeoutMs)
|
|
141
|
+
: null;
|
|
118
142
|
const maxTimer = MAX_STREAM_MS
|
|
119
143
|
? setTimeout(() => {
|
|
120
144
|
tooLong = true;
|
|
121
145
|
detail.exitReason = "max";
|
|
122
|
-
|
|
146
|
+
cancelBody(upRes.body);
|
|
123
147
|
}, MAX_STREAM_MS)
|
|
124
148
|
: null;
|
|
125
149
|
try {
|
|
@@ -177,6 +201,15 @@ export async function relay(res, upRes, body, { onFirstChunk, onDownstreamAbort,
|
|
|
177
201
|
} catch {}
|
|
178
202
|
}
|
|
179
203
|
} catch { /* ignore */ }
|
|
204
|
+
// 首块/空闲超时后上游仍吐出了数据 → 只是慢,不是死:撤销超时判定,照常转发
|
|
205
|
+
//(cancel 是协作式的,缓冲数据仍会到达;丢掉已到达的数据是纯损失)
|
|
206
|
+
if (timedOut || stalled) {
|
|
207
|
+
timedOut = false;
|
|
208
|
+
stalled = false;
|
|
209
|
+
detail.recoveries = (detail.recoveries || 0) + 1;
|
|
210
|
+
if (detail.exitReason === "first-timeout" || detail.exitReason === "stall") detail.exitReason = null;
|
|
211
|
+
if (firstTimer) { clearTimeout(firstTimer); firstTimer = null; }
|
|
212
|
+
}
|
|
180
213
|
if (timedOut || stalled || tooLong) break;
|
|
181
214
|
if (first) {
|
|
182
215
|
first = false;
|
|
@@ -213,10 +246,11 @@ export async function relay(res, upRes, body, { onFirstChunk, onDownstreamAbort,
|
|
|
213
246
|
if (firstTimer) clearTimeout(firstTimer);
|
|
214
247
|
if (maxTimer) clearTimeout(maxTimer);
|
|
215
248
|
if (stallTimer) clearTimeout(stallTimer);
|
|
249
|
+
if (pingTimer) clearInterval(pingTimer);
|
|
216
250
|
}
|
|
217
251
|
if (timedOut && !wroteAny) {
|
|
218
252
|
res.removeListener("close", onClose);
|
|
219
|
-
return { status:
|
|
253
|
+
return { status: streamTimeoutMs, ttfMs: null, totalMs: Math.round(performance.now() - t0), aborted: true, interrupted: false, detail };
|
|
220
254
|
}
|
|
221
255
|
if ((stalled || tooLong) && wroteAny) {
|
|
222
256
|
interrupted = true;
|
|
@@ -115,6 +115,24 @@ export async function startServerLifecycle({ VERSION, token, created, upstream,
|
|
|
115
115
|
}).catch(() => {});
|
|
116
116
|
}, 100).unref?.();
|
|
117
117
|
|
|
118
|
+
// responses 模型判定改元数据驱动(models.dev provider.npm):就绪后注入,每小时重查使新模型自动识别
|
|
119
|
+
void (async () => {
|
|
120
|
+
try {
|
|
121
|
+
const { globalCapabilities } = await import("../model-capabilities/index.js");
|
|
122
|
+
const { setResponsesNpmIndex } = await import("../upstream-responses.js");
|
|
123
|
+
const svc = globalCapabilities();
|
|
124
|
+
const apply = () => setResponsesNpmIndex(svc.npmIndex());
|
|
125
|
+
await svc.ready();
|
|
126
|
+
apply();
|
|
127
|
+
const entry = { ts: Date.now(), type: "responses-npm-index", models: svc.npmIndex().size };
|
|
128
|
+
try { bus.emit(entry); } catch {}
|
|
129
|
+
try { logs.appendEvent(entry); } catch {}
|
|
130
|
+
setInterval(() => { svc.ready().then(apply).catch(() => {}); }, 60 * 60 * 1000).unref?.();
|
|
131
|
+
} catch (e) {
|
|
132
|
+
try { logs.appendEvent({ ts: Date.now(), type: "responses-npm-index-failed", error: String(e?.message || e).slice(0, 200) }); } catch {}
|
|
133
|
+
}
|
|
134
|
+
})();
|
|
135
|
+
|
|
118
136
|
models.startAutoRefresh();
|
|
119
137
|
if (process.env.MSLXDFF_DAEMON) {
|
|
120
138
|
writePid(process.pid, VERSION);
|
|
@@ -57,9 +57,9 @@ export function errorResponseFromSdkError(e, { marker = null } = {}) {
|
|
|
57
57
|
});
|
|
58
58
|
}
|
|
59
59
|
|
|
60
|
-
// SDK parts → OpenAI SSE Response(chat/responses
|
|
61
|
-
export function streamResponseFromParts(parts, { marker = null, clock = Date.now, t0 = clock() } = {}) {
|
|
62
|
-
const ser = createSseSerializer();
|
|
60
|
+
// SDK parts → OpenAI SSE Response(chat/responses 共用同一序列化器;captured 为 responses 侧信道)。
|
|
61
|
+
export function streamResponseFromParts(parts, { marker = null, clock = Date.now, t0 = clock(), captured = null } = {}) {
|
|
62
|
+
const ser = createSseSerializer(captured);
|
|
63
63
|
const enc = new TextEncoder();
|
|
64
64
|
let cancelled = false;
|
|
65
65
|
const stream = new ReadableStream({
|
|
@@ -67,6 +67,11 @@ export function streamResponseFromParts(parts, { marker = null, clock = Date.now
|
|
|
67
67
|
try {
|
|
68
68
|
for await (const part of parts) {
|
|
69
69
|
if (cancelled) break;
|
|
70
|
+
// 侧信道(captured)可能落后几毫秒:收尾前给 encrypted 一点就绪时间(最多 150ms)
|
|
71
|
+
if (captured && !captured.reasoning && part?.type === "finish") {
|
|
72
|
+
const tw = Date.now();
|
|
73
|
+
while (!captured.reasoning && Date.now() - tw < 150) await new Promise((r) => setTimeout(r, 10));
|
|
74
|
+
}
|
|
70
75
|
const text = ser.push(part);
|
|
71
76
|
if (text) controller.enqueue(enc.encode(text));
|
|
72
77
|
}
|
|
@@ -52,7 +52,21 @@ export function toModelPrompt(messages) {
|
|
|
52
52
|
const parts = [];
|
|
53
53
|
const text = textOf(m.content);
|
|
54
54
|
if (text) parts.push({ type: "text", text });
|
|
55
|
-
|
|
55
|
+
// 带加密态的 reasoning items(responses 通道思考跨轮):AI SDK 会转成上游要的 encrypted reasoning item。
|
|
56
|
+
// 无加密态时才退回纯文本 reasoning(chat 通道的 reasoning_content)。
|
|
57
|
+
const items = Array.isArray(m.reasoning_items) ? m.reasoning_items : [];
|
|
58
|
+
let pushedEncrypted = false;
|
|
59
|
+
for (const r of items) {
|
|
60
|
+
if (!r || typeof r !== "object") continue;
|
|
61
|
+
const summaryText = Array.isArray(r.summary) ? r.summary.map((s) => s?.text || "").join("\n") : "";
|
|
62
|
+
parts.push({
|
|
63
|
+
type: "reasoning",
|
|
64
|
+
text: summaryText || " ",
|
|
65
|
+
providerOptions: { openai: { itemId: r.id, reasoningEncryptedContent: r.encrypted_content } },
|
|
66
|
+
});
|
|
67
|
+
pushedEncrypted = true;
|
|
68
|
+
}
|
|
69
|
+
if (!pushedEncrypted && m.reasoning_content) parts.push({ type: "reasoning", text: String(m.reasoning_content) });
|
|
56
70
|
for (const tc of Array.isArray(m.tool_calls) ? m.tool_calls : []) {
|
|
57
71
|
if (!tc || typeof tc !== "object") continue;
|
|
58
72
|
parts.push({
|
|
@@ -8,6 +8,50 @@ import { ENGINE_MARKER } from "./chat.js";
|
|
|
8
8
|
|
|
9
9
|
export const RESPONSES_CHAT_PATH = "/zen/v1/responses";
|
|
10
10
|
|
|
11
|
+
// 上游 encrypted reasoning 只在 output_item.done 里给,而 AI SDK 仅在有 summary 文本时才透出到 parts
|
|
12
|
+
// (muse-spark 这类无 summary 的思考模型会被吞掉)。这里在 fetch 层 tee 一份原始 SSE 自行解析,
|
|
13
|
+
// 侧信道把加密思考交给序列化器,收尾帧补发。
|
|
14
|
+
function captureReasoningFetch(baseFetch, sink) {
|
|
15
|
+
const dbg = process.env.MSLXDFF_RESPONSES_DEBUG === "1";
|
|
16
|
+
return async (input, init) => {
|
|
17
|
+
const res = await baseFetch(input, init);
|
|
18
|
+
const ct = res.headers.get("content-type") || "";
|
|
19
|
+
if (dbg) console.log(`[capture] enter ct=${ct} hasBody=${Boolean(res.body)}`);
|
|
20
|
+
if (!ct.includes("text/event-stream") || !res.body) return res;
|
|
21
|
+
const [passthrough, probe] = res.body.tee();
|
|
22
|
+
(async () => {
|
|
23
|
+
const reader = probe.getReader();
|
|
24
|
+
const dec = new TextDecoder();
|
|
25
|
+
let buf = "";
|
|
26
|
+
try {
|
|
27
|
+
for (;;) {
|
|
28
|
+
const { done, value } = await reader.read();
|
|
29
|
+
if (done) break;
|
|
30
|
+
buf += dec.decode(value, { stream: true });
|
|
31
|
+
let nl;
|
|
32
|
+
while ((nl = buf.indexOf("\n")) >= 0) {
|
|
33
|
+
const line = buf.slice(0, nl).trim();
|
|
34
|
+
buf = buf.slice(nl + 1);
|
|
35
|
+
if (!line.startsWith("data:")) continue;
|
|
36
|
+
const d = line.slice(5).trim();
|
|
37
|
+
if (!d || d === "[DONE]") continue;
|
|
38
|
+
try {
|
|
39
|
+
const j = JSON.parse(d);
|
|
40
|
+
if (dbg && j.type === "response.output_item.done") console.log(`[capture] item.done type=${j.item?.type} enc=${typeof j.item?.encrypted_content}`);
|
|
41
|
+
if (j.type === "response.output_item.done" && j.item?.type === "reasoning" && typeof j.item.encrypted_content === "string") {
|
|
42
|
+
sink.reasoning = { id: j.item.id, encrypted: j.item.encrypted_content, summary: j.item.summary ?? [] };
|
|
43
|
+
if (dbg) console.log(`[capture] hit id=${String(j.item.id).slice(0, 24)} len=${j.item.encrypted_content.length}`);
|
|
44
|
+
}
|
|
45
|
+
} catch { /* ignore */ }
|
|
46
|
+
}
|
|
47
|
+
}
|
|
48
|
+
} catch { /* ignore */ }
|
|
49
|
+
if (dbg) console.log(`[capture] stream end captured=${sink.reasoning ? "yes" : "no"}`);
|
|
50
|
+
})();
|
|
51
|
+
return new Response(passthrough, { status: res.status, statusText: res.statusText, headers: res.headers });
|
|
52
|
+
};
|
|
53
|
+
}
|
|
54
|
+
|
|
11
55
|
let sdkPromise = null;
|
|
12
56
|
|
|
13
57
|
export function loadOpenAISdk() {
|
|
@@ -37,19 +81,34 @@ export async function attemptOnceResponsesSdk({
|
|
|
37
81
|
throw err;
|
|
38
82
|
}
|
|
39
83
|
// headers 由调用方构造(含 Authorization),headers 优先于 apiKey 默认头
|
|
84
|
+
const captured = { reasoning: null };
|
|
85
|
+
const baseFetch = fetchImpl ?? ((u, i) => globalThis.fetch(u, i));
|
|
40
86
|
const provider = createOpenAI({
|
|
41
87
|
name: providerName,
|
|
42
88
|
baseURL,
|
|
43
89
|
apiKey: "public",
|
|
44
90
|
headers: sanitizeHeaders(headers),
|
|
45
|
-
|
|
91
|
+
fetch: captureReasoningFetch(baseFetch, captured),
|
|
46
92
|
});
|
|
47
93
|
const model = provider.responses(String(body?.model || ""));
|
|
94
|
+
const params = toModelParams(body, providerName);
|
|
95
|
+
// 无状态 + 加密思考回传(thinking 跨轮):上游把思考以加密块发回,客户端持有并每轮带回。
|
|
96
|
+
// AI SDK responses 的 providerOptionsName 对非 azure 硬编码为 "openai"(dist/index.js:5240)。
|
|
97
|
+
const providerOptions = {
|
|
98
|
+
...(params.providerOptions || {}),
|
|
99
|
+
openai: {
|
|
100
|
+
store: false,
|
|
101
|
+
include: ["reasoning.encrypted_content"],
|
|
102
|
+
reasoningSummary: "auto",
|
|
103
|
+
...((params.providerOptions || {}).openai || {}),
|
|
104
|
+
},
|
|
105
|
+
};
|
|
48
106
|
let res;
|
|
49
107
|
try {
|
|
50
108
|
res = await model.doStream({
|
|
51
109
|
prompt: toModelPrompt(body?.messages),
|
|
52
|
-
...
|
|
110
|
+
...params,
|
|
111
|
+
providerOptions,
|
|
53
112
|
tools: toModelTools(body?.tools),
|
|
54
113
|
toolChoice: toModelToolChoice(body?.tool_choice),
|
|
55
114
|
});
|
|
@@ -58,7 +117,7 @@ export async function attemptOnceResponsesSdk({
|
|
|
58
117
|
if (mapped) return mapped;
|
|
59
118
|
throw e;
|
|
60
119
|
}
|
|
61
|
-
return streamResponseFromParts(res.stream, { marker, clock, t0 });
|
|
120
|
+
return streamResponseFromParts(res.stream, { marker, clock, t0, captured });
|
|
62
121
|
}
|
|
63
122
|
|
|
64
123
|
export function createSdkResponses({
|
|
@@ -32,14 +32,18 @@ export function usageToOpenAI(u) {
|
|
|
32
32
|
return out;
|
|
33
33
|
}
|
|
34
34
|
|
|
35
|
-
export function createSseSerializer() {
|
|
35
|
+
export function createSseSerializer(captured = null) {
|
|
36
36
|
const meta = { id: "chatcmpl-wb-sdk", model: "", created: Math.floor(Date.now() / 1000) };
|
|
37
37
|
let roleSent = false;
|
|
38
38
|
let nextIndex = 0;
|
|
39
39
|
const toolIndex = new Map();
|
|
40
40
|
const toolDeltaIds = new Set();
|
|
41
|
+
// 加密思考往返:reasoning-start 的 providerMetadata 带 itemId + 加密态,随首帧透出
|
|
42
|
+
let reasoningMeta = null;
|
|
43
|
+
let reasoningFrameSent = false;
|
|
44
|
+
let reasoningEncSent = false;
|
|
41
45
|
|
|
42
|
-
function frame(delta, { finishReason = null, usage } = {}) {
|
|
46
|
+
function frame(delta, { finishReason = null, usage, extra } = {}) {
|
|
43
47
|
const obj = {
|
|
44
48
|
id: meta.id,
|
|
45
49
|
object: "chat.completion.chunk",
|
|
@@ -48,6 +52,7 @@ export function createSseSerializer() {
|
|
|
48
52
|
choices: [{ index: 0, delta, finish_reason: finishReason }],
|
|
49
53
|
};
|
|
50
54
|
if (usage !== undefined) obj.usage = usage;
|
|
55
|
+
if (extra) Object.assign(obj, extra);
|
|
51
56
|
return `data: ${JSON.stringify(obj)}\n\n`;
|
|
52
57
|
}
|
|
53
58
|
|
|
@@ -69,8 +74,30 @@ export function createSseSerializer() {
|
|
|
69
74
|
}
|
|
70
75
|
return null;
|
|
71
76
|
}
|
|
72
|
-
case "reasoning-
|
|
73
|
-
|
|
77
|
+
case "reasoning-start": {
|
|
78
|
+
const pm = part.providerMetadata?.openai || part.providerMetadata || {};
|
|
79
|
+
reasoningMeta = {
|
|
80
|
+
id: String(pm.itemId ?? part.id ?? "reasoning"),
|
|
81
|
+
encrypted: typeof pm.reasoningEncryptedContent === "string" ? pm.reasoningEncryptedContent : null,
|
|
82
|
+
};
|
|
83
|
+
// 即时透出 item 元数据(含加密态):上游可能只给 encrypted 不给 summary 文本,不能等 delta
|
|
84
|
+
reasoningFrameSent = true;
|
|
85
|
+
if (reasoningMeta.encrypted) reasoningEncSent = true;
|
|
86
|
+
return ensureRole() + frame({ reasoning_content: "" }, { extra: { x_reasoning_item: { id: reasoningMeta.id, encrypted_content: reasoningMeta.encrypted } } });
|
|
87
|
+
}
|
|
88
|
+
case "reasoning-delta": {
|
|
89
|
+
const delta = String(part.delta ?? "");
|
|
90
|
+
let extra;
|
|
91
|
+
if (reasoningMeta && !reasoningFrameSent) {
|
|
92
|
+
reasoningFrameSent = true;
|
|
93
|
+
if (reasoningMeta.encrypted) reasoningEncSent = true;
|
|
94
|
+
// x_reasoning_item:responses translator 据此建 reasoning item(chat 客户端忽略未知顶层字段)
|
|
95
|
+
extra = { x_reasoning_item: { id: reasoningMeta.id, encrypted_content: reasoningMeta.encrypted } };
|
|
96
|
+
} else if (reasoningMeta) {
|
|
97
|
+
extra = { x_reasoning_id: reasoningMeta.id };
|
|
98
|
+
}
|
|
99
|
+
return ensureRole() + frame({ reasoning_content: delta }, { extra });
|
|
100
|
+
}
|
|
74
101
|
case "text-delta":
|
|
75
102
|
return ensureRole() + frame({ content: String(part.delta ?? "") });
|
|
76
103
|
case "tool-input-start": {
|
|
@@ -98,7 +125,14 @@ export function createSseSerializer() {
|
|
|
98
125
|
const fr = part.finishReason;
|
|
99
126
|
const raw = typeof fr === "string" ? fr : (fr?.raw ?? fr?.unified);
|
|
100
127
|
const reason = raw === "tool-calls" ? "tool_calls" : raw === "content-filter" ? "content_filter" : (raw ?? "stop");
|
|
101
|
-
|
|
128
|
+
// 加密思考补发:上游 encrypted_content 只在流末尾可得(fetch 侧信道捕获),
|
|
129
|
+
// 无 summary 文本时 AI SDK parts 不会带出 → 收尾帧前补一帧
|
|
130
|
+
let pre = "";
|
|
131
|
+
if (captured?.reasoning && !reasoningEncSent) {
|
|
132
|
+
reasoningEncSent = true;
|
|
133
|
+
pre = ensureRole() + frame({ reasoning_content: "" }, { extra: { x_reasoning_item: { id: captured.reasoning.id, encrypted_content: captured.reasoning.encrypted } } });
|
|
134
|
+
}
|
|
135
|
+
return pre + frame({}, { finishReason: reason, usage: usageToOpenAI(part.usage) });
|
|
102
136
|
}
|
|
103
137
|
case "error": {
|
|
104
138
|
const message = part.error?.message
|
|
@@ -2,8 +2,26 @@
|
|
|
2
2
|
* responses 转换层 — 从 upstream.js 抽出的 muse-spark 专用形状转换。
|
|
3
3
|
* chat ⇄ responses 互转纯函数,无网络、无副作用。
|
|
4
4
|
*/
|
|
5
|
+
// responses 判定索引(models.dev 模型级 provider.npm,启动时注入):
|
|
6
|
+
// "@ai-sdk/openai" → responses 端点;未注入/未命中 → 前缀兜底(新模型早于 models.dev 刷新时仍可用)。
|
|
7
|
+
let responsesNpmIndex = null;
|
|
8
|
+
|
|
9
|
+
export function setResponsesNpmIndex(idx) {
|
|
10
|
+
responsesNpmIndex = idx instanceof Map ? idx : null;
|
|
11
|
+
}
|
|
12
|
+
|
|
13
|
+
export function _resetResponsesNpmIndex() {
|
|
14
|
+
responsesNpmIndex = null;
|
|
15
|
+
}
|
|
16
|
+
|
|
5
17
|
export function isResponsesModel(model) {
|
|
6
|
-
|
|
18
|
+
const m = String(model || "").toLowerCase().trim();
|
|
19
|
+
if (!m) return false;
|
|
20
|
+
if (responsesNpmIndex) {
|
|
21
|
+
const bare = m.replace(/^opencode\//, "");
|
|
22
|
+
if (responsesNpmIndex.has(bare)) return responsesNpmIndex.get(bare) === "@ai-sdk/openai";
|
|
23
|
+
}
|
|
24
|
+
return m.startsWith("muse-spark");
|
|
7
25
|
}
|
|
8
26
|
|
|
9
27
|
export function chatToResponsesBody(chatBody) {
|
package/src/upstream.js
CHANGED
|
@@ -12,6 +12,24 @@ import { uuid } from "./compat.js";
|
|
|
12
12
|
function genId(prefix) {
|
|
13
13
|
return `${prefix}${uuid().replace(/-/g, "")}`;
|
|
14
14
|
}
|
|
15
|
+
// 上游按 session 做粘性路由(实测:固定 session 两次请求均 ~1.2s;每次随机时可能撞冷机器 26s+)。
|
|
16
|
+
// 客户端(opencode AI SDK 路径)不带会话标识 → 用对话首两条消息(system + 首条 user)哈希做稳定会话:
|
|
17
|
+
// 同一会话多轮里这两条不变 ⇒ 路由亲和稳定;不同会话天然分散。
|
|
18
|
+
function sessionFromMessages(messages) {
|
|
19
|
+
try {
|
|
20
|
+
const msgs = Array.isArray(messages) ? messages : [];
|
|
21
|
+
const pick = (role) => {
|
|
22
|
+
const m = msgs.find((x) => x?.role === role);
|
|
23
|
+
if (!m) return "";
|
|
24
|
+
return typeof m.content === "string" ? m.content : JSON.stringify(m.content ?? "");
|
|
25
|
+
};
|
|
26
|
+
const seed = `${pick("system")}|${pick("user")}`.slice(0, 4000);
|
|
27
|
+
if (seed === "|") return null;
|
|
28
|
+
return `ses_${crypto.createHash("sha1").update(seed).digest("hex").slice(0, 32)}`;
|
|
29
|
+
} catch {
|
|
30
|
+
return null;
|
|
31
|
+
}
|
|
32
|
+
}
|
|
15
33
|
function envInt(name, fallback) {
|
|
16
34
|
const v = Number(process.env[name]);
|
|
17
35
|
return Number.isInteger(v) && v > 0 ? v : fallback;
|
|
@@ -47,13 +65,16 @@ export function createOpencodeHeaderBuilder({ authToken = "public", env = proces
|
|
|
47
65
|
Authorization: `Bearer ${authToken}`,
|
|
48
66
|
"x-opencode-client": "desktop",
|
|
49
67
|
};
|
|
50
|
-
|
|
68
|
+
// 无 messages 可哈希时的进程级兜底:至少 daemon 生命周期内稳定,不再每请求随机
|
|
69
|
+
const FALLBACK_SESSION = genId("ses_");
|
|
70
|
+
function buildHeaders(body, { anonymous = anonFirst, sessionId = null } = {}) {
|
|
51
71
|
const isStream = body?.stream !== false;
|
|
72
|
+
const session = sessionId || sessionFromMessages(body?.messages) || FALLBACK_SESSION;
|
|
52
73
|
const base = {
|
|
53
74
|
...baseHeaders,
|
|
54
75
|
Accept: isStream ? "text/event-stream" : "*/*",
|
|
55
76
|
"User-Agent": "opencode",
|
|
56
|
-
"x-opencode-session":
|
|
77
|
+
"x-opencode-session": session,
|
|
57
78
|
"x-opencode-request": genId("msg_"),
|
|
58
79
|
"x-opencode-project": "global",
|
|
59
80
|
};
|
|
@@ -111,8 +132,10 @@ export function createUpstreamClient({
|
|
|
111
132
|
hooks,
|
|
112
133
|
});
|
|
113
134
|
|
|
114
|
-
async function chat(body) {
|
|
135
|
+
async function chat(body, opts = {}) {
|
|
115
136
|
const isResp = isResponsesModel(body?.model);
|
|
137
|
+
// responses 路径的 reqBody 无 messages,会话哈希必须基于原始 chat body 计算
|
|
138
|
+
const sessionId = opts?.sessionId || sessionFromMessages(body?.messages) || null;
|
|
116
139
|
const url = isResp ? `${baseUrl}/zen/v1/responses` : `${baseUrl}/zen/v1/chat/completions`;
|
|
117
140
|
const reqBody = isResp ? chatToResponsesBody(body) : body;
|
|
118
141
|
const t0 = performance.now();
|
|
@@ -123,7 +146,7 @@ export function createUpstreamClient({
|
|
|
123
146
|
res = await transport.request({
|
|
124
147
|
url,
|
|
125
148
|
method: "POST",
|
|
126
|
-
headers: buildHeaders(reqBody),
|
|
149
|
+
headers: buildHeaders(reqBody, { sessionId }),
|
|
127
150
|
body: reqBody,
|
|
128
151
|
stream: body?.stream !== false,
|
|
129
152
|
timeoutMs: connectTimeoutMs,
|
|
@@ -145,7 +168,7 @@ export function createUpstreamClient({
|
|
|
145
168
|
anonRes = await transport.request({
|
|
146
169
|
url,
|
|
147
170
|
method: "POST",
|
|
148
|
-
headers: buildHeaders(reqBody, { anonymous: true }),
|
|
171
|
+
headers: buildHeaders(reqBody, { anonymous: true, sessionId }),
|
|
149
172
|
body: reqBody,
|
|
150
173
|
stream: body?.stream !== false,
|
|
151
174
|
timeoutMs: connectTimeoutMs,
|