mslxdff 0.1.114 → 0.1.116
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/src/chat-pipeline/auto-race.js +1 -0
- package/src/chat-pipeline/index.js +5 -1
- package/src/chat-pipeline/serial-trial.js +1 -0
- package/src/model-capabilities/index.js +9 -2
- package/src/model-capabilities/parse.js +2 -0
- package/src/providers/dispatcher.js +2 -2
- package/src/providers/opencode.js +1 -1
- package/src/responses/translate.js +74 -21
- package/src/routes/responses-route.js +9 -0
- package/src/routes/stream.js +14 -1
- package/src/runtime/server-lifecycle.js +18 -0
- package/src/upstream-engine/sdk/attempt.js +8 -3
- package/src/upstream-engine/sdk/convert.js +15 -1
- package/src/upstream-engine/sdk/responses.js +62 -3
- package/src/upstream-engine/sdk/sse.js +39 -5
- package/src/upstream-responses.js +19 -1
- package/src/upstream.js +28 -5
package/package.json
CHANGED
|
@@ -65,6 +65,7 @@ export async function runAutoRace(ctx, deps = {}) {
|
|
|
65
65
|
const o = {};
|
|
66
66
|
if (Object.keys(shareKeys).length) o.shareKeys = shareKeys;
|
|
67
67
|
if (workbuddyUid) o.workbuddyUid = workbuddyUid;
|
|
68
|
+
if (handlerCtx?.sessionId) o.sessionId = handlerCtx.sessionId;
|
|
68
69
|
r = await upstream.chat(f, Object.keys(o).length ? o : undefined);
|
|
69
70
|
} catch (e) {
|
|
70
71
|
if (plugins?.length) runHook(plugins, "upstream:response", { reqId, requested, model: m, status: null, ok: false, error: errMsg(e), timing: e?._t ?? null }).catch(() => {});
|
|
@@ -82,7 +82,11 @@ export function createChatPipeline({ upstream, auto, logs, peers, groups, bus, t
|
|
|
82
82
|
for (const e of sel.errors) evt("plugin-hook-error", { reqId, hook: "model:select", plugin: e.plugin, error: e.error });
|
|
83
83
|
}
|
|
84
84
|
|
|
85
|
-
|
|
85
|
+
// 客户端会话标识(opencode 插件 chat.headers 注入)→ 透传上游做粘性路由/缓存亲和;
|
|
86
|
+
// 无头时由 upstream.js 按对话首两条消息哈希兜底(不再每请求随机)
|
|
87
|
+
const clientSession = String(req?.headers?.["x-session-affinity"] || req?.headers?.["x-session-id"] || "").trim() || null;
|
|
88
|
+
const handlerCtx = { reqId, model: null, body: req?.body, hops, peers, plugins, evt, logError, logCall, logs, workbuddyUid, sessionId: clientSession };
|
|
89
|
+
if (clientSession) evt("client-session", { reqId, sessionId: clientSession.slice(0, 24) });
|
|
86
90
|
|
|
87
91
|
const plan = planRoute(policy, {
|
|
88
92
|
candidates: order,
|
|
@@ -70,6 +70,7 @@ export async function runSerialTrial(ctx, deps = {}) {
|
|
|
70
70
|
const chatOpts = {};
|
|
71
71
|
if (Object.keys(shareKeys).length) chatOpts.shareKeys = shareKeys;
|
|
72
72
|
if (workbuddyUid) chatOpts.workbuddyUid = workbuddyUid;
|
|
73
|
+
if (handlerCtx?.sessionId) chatOpts.sessionId = handlerCtx.sessionId;
|
|
73
74
|
upRes = await upstream.chat(forwarded, Object.keys(chatOpts).length ? chatOpts : undefined);
|
|
74
75
|
evt("upstream-done", { reqId, model, ok: !(upRes instanceof Error) && upRes.status < 400, status: upRes instanceof Error ? null : upRes.status, timing: upRes._t ?? null, error: null });
|
|
75
76
|
} catch (err) {
|
|
@@ -28,15 +28,22 @@ export function createCapabilitiesService({
|
|
|
28
28
|
} = {}) {
|
|
29
29
|
let raw = null; // 原始目录(全 provider)
|
|
30
30
|
let capsIndex = new Map(); // providerId -> { modelId -> caps }
|
|
31
|
+
let npmIndex = new Map(); // opencode 裸 modelId -> provider.npm(responses 判定用;null = 继承默认)
|
|
31
32
|
let loadedAt = 0;
|
|
32
33
|
let inflight = null;
|
|
33
34
|
|
|
34
35
|
function buildIndex(data) {
|
|
35
36
|
const idx = new Map();
|
|
37
|
+
const npm = new Map();
|
|
36
38
|
for (const [pid, p] of Object.entries(data || {})) {
|
|
37
39
|
if (!p || typeof p !== "object") continue;
|
|
38
|
-
|
|
40
|
+
const caps = normalizeProviderModels(p.models || {});
|
|
41
|
+
idx.set(pid, caps);
|
|
42
|
+
if (String(pid).toLowerCase() === "opencode") {
|
|
43
|
+
for (const [mid, c] of Object.entries(caps)) npm.set(mid, c.npm ?? null);
|
|
44
|
+
}
|
|
39
45
|
}
|
|
46
|
+
npmIndex = npm;
|
|
40
47
|
return idx;
|
|
41
48
|
}
|
|
42
49
|
|
|
@@ -111,7 +118,7 @@ export function createCapabilitiesService({
|
|
|
111
118
|
return [...capsIndex.keys()].sort();
|
|
112
119
|
}
|
|
113
120
|
|
|
114
|
-
return { ready, get, list, providers };
|
|
121
|
+
return { ready, get, list, providers, npmIndex: () => new Map(npmIndex) };
|
|
115
122
|
}
|
|
116
123
|
|
|
117
124
|
// 模块级单例(与 globalDedup 同模式):HTTP handler 懒加载,测试 _reset 后注入
|
|
@@ -24,6 +24,8 @@ export function normalizeModelCaps(_id, m) {
|
|
|
24
24
|
releaseDate: typeof m?.release_date === "string" && m.release_date ? m.release_date : null,
|
|
25
25
|
inputModalities: input.length ? input : ["text"],
|
|
26
26
|
outputModalities: Array.isArray(m?.modalities?.output) && m.modalities.output.length ? m.modalities.output : ["text"],
|
|
27
|
+
// 模型级 SDK 覆盖(models.dev provider.npm):@ai-sdk/openai → responses 端点;null = 继承 provider 默认
|
|
28
|
+
npm: typeof m?.provider?.npm === "string" && m.provider.npm ? m.provider.npm : null,
|
|
27
29
|
};
|
|
28
30
|
}
|
|
29
31
|
|
|
@@ -53,9 +53,9 @@ export function createProviderDispatcher(providers = [], opts = {}) {
|
|
|
53
53
|
return provider.chatWithKeys(forwarded, sharedKeys);
|
|
54
54
|
}
|
|
55
55
|
if (provider.id === "workbuddy" && workbuddyUid) {
|
|
56
|
-
return provider.chat(forwarded, { workbuddyUid });
|
|
56
|
+
return provider.chat(forwarded, { ...opts, workbuddyUid });
|
|
57
57
|
}
|
|
58
|
-
return provider.chat(forwarded);
|
|
58
|
+
return provider.chat(forwarded, opts);
|
|
59
59
|
}
|
|
60
60
|
|
|
61
61
|
// 聚合所有供应商的模型列表;默认供应商(opencode)裸 id,其它带前缀
|
|
@@ -8,7 +8,7 @@ export function createOpenCodeProvider({ upstream, modelsService, baseUrl, authT
|
|
|
8
8
|
return {
|
|
9
9
|
id: "opencode",
|
|
10
10
|
upstream: client,
|
|
11
|
-
chat: (body) => client.chat(body),
|
|
11
|
+
chat: (body, opts) => client.chat(body, opts),
|
|
12
12
|
preheat: (args) => client.preheat(args),
|
|
13
13
|
close: () => client.close(),
|
|
14
14
|
async listModels() {
|
|
@@ -33,15 +33,28 @@ export function responsesToChatBody(req = {}) {
|
|
|
33
33
|
if (req.instructions) messages.push({ role: "system", content: String(req.instructions) });
|
|
34
34
|
const input = req.input;
|
|
35
35
|
const items = typeof input === "string" ? [{ type: "message", role: "user", content: input }] : Array.isArray(input) ? input : [];
|
|
36
|
+
// 加密思考往返:input 里的 reasoning item 挂到下一条 assistant 消息(thinking 跨轮必需,不能丢)
|
|
37
|
+
let pendingReasoning = [];
|
|
36
38
|
for (const it of items) {
|
|
37
39
|
if (!it || typeof it !== "object") continue;
|
|
38
|
-
if (it.type === "
|
|
39
|
-
|
|
40
|
+
if (it.type === "reasoning") {
|
|
41
|
+
if (it.id || it.encrypted_content) {
|
|
42
|
+
pendingReasoning.push({ id: it.id, encrypted_content: it.encrypted_content, summary: it.summary });
|
|
43
|
+
}
|
|
44
|
+
continue;
|
|
45
|
+
}
|
|
46
|
+
// responses 规范:message item 的 type 可省(AI SDK/opencode 就不发)→ 有 role 即按 message 处理
|
|
47
|
+
if (it.type === "message" || (!it.type && it.role)) {
|
|
48
|
+
const msg = { role: it.role || "user", content: inputTextOf(it.content) };
|
|
49
|
+
if (pendingReasoning.length && msg.role === "assistant") { msg.reasoning_items = pendingReasoning; pendingReasoning = []; }
|
|
50
|
+
messages.push(msg);
|
|
40
51
|
} else if (it.type === "function_call") {
|
|
41
|
-
|
|
52
|
+
const msg = {
|
|
42
53
|
role: "assistant", content: "",
|
|
43
54
|
tool_calls: [{ id: it.call_id || it.id || "", type: "function", function: { name: it.name || "", arguments: it.arguments || "" } }],
|
|
44
|
-
}
|
|
55
|
+
};
|
|
56
|
+
if (pendingReasoning.length) { msg.reasoning_items = pendingReasoning; pendingReasoning = []; }
|
|
57
|
+
messages.push(msg);
|
|
45
58
|
} else if (it.type === "function_call_output") {
|
|
46
59
|
messages.push({ role: "tool", tool_call_id: it.call_id || "", content: typeof it.output === "string" ? it.output : JSON.stringify(it.output ?? "") });
|
|
47
60
|
}
|
|
@@ -115,7 +128,16 @@ export function createChunkTranslator(model = "") {
|
|
|
115
128
|
let textLen = 0;
|
|
116
129
|
let lastFinish = "stop";
|
|
117
130
|
let lastUsage = null;
|
|
118
|
-
const tools = new Map(); // index → {id, name, args, announced}
|
|
131
|
+
const tools = new Map(); // index → {id, name, args, announced, outputIndex}
|
|
132
|
+
// output_index 按 item 出现顺序动态分配(reasoning 可能先于 message/tool 出现)
|
|
133
|
+
let nextIdx = 0;
|
|
134
|
+
let textIdx = null;
|
|
135
|
+
// 加密思考(thinking 跨轮):sse.js 首帧带 x_reasoning_item(含加密态),translator 据此建 reasoning item
|
|
136
|
+
let reasoningOpen = false;
|
|
137
|
+
let reasoningId = null;
|
|
138
|
+
let reasoningEncrypted = null;
|
|
139
|
+
let reasoningIdx = null;
|
|
140
|
+
let reasoningText = "";
|
|
119
141
|
const ev = (type, extra = {}) => ({ type, ...extra });
|
|
120
142
|
|
|
121
143
|
function begin() {
|
|
@@ -125,8 +147,9 @@ export function createChunkTranslator(model = "") {
|
|
|
125
147
|
function ensureTextItem() {
|
|
126
148
|
if (textItemOpen) return [];
|
|
127
149
|
textItemOpen = true;
|
|
150
|
+
textIdx = nextIdx++;
|
|
128
151
|
const item = { type: "message", id: `msg_${id}`, status: "in_progress", role: "assistant", content: [{ type: "output_text", text: "", annotations: [] }] };
|
|
129
|
-
return [ev("response.output_item.added", { output_index:
|
|
152
|
+
return [ev("response.output_item.added", { output_index: textIdx, item }), ev("response.content_part.added", { item_id: item.id, output_index: textIdx, content_index: 0, part: { type: "output_text", text: "", annotations: [] } })];
|
|
130
153
|
}
|
|
131
154
|
|
|
132
155
|
let fullText = "";
|
|
@@ -135,26 +158,51 @@ export function createChunkTranslator(model = "") {
|
|
|
135
158
|
const out = ensureTextItem();
|
|
136
159
|
textLen += delta.length;
|
|
137
160
|
fullText += delta;
|
|
138
|
-
out.push(ev("response.output_text.delta", { item_id: `msg_${id}`, output_index:
|
|
161
|
+
out.push(ev("response.output_text.delta", { item_id: `msg_${id}`, output_index: textIdx, content_index: 0, delta }));
|
|
162
|
+
return out;
|
|
163
|
+
}
|
|
164
|
+
|
|
165
|
+
function pushReasoning(delta, meta) {
|
|
166
|
+
const out = ensureReasoningItem(meta);
|
|
167
|
+
reasoningText += delta;
|
|
168
|
+
if (delta) out.push(ev("response.reasoning_summary_text.delta", { item_id: reasoningId, output_index: reasoningIdx, summary_index: 0, delta }));
|
|
139
169
|
return out;
|
|
140
170
|
}
|
|
141
171
|
|
|
172
|
+
function ensureReasoningItem(meta) {
|
|
173
|
+
if (reasoningOpen) {
|
|
174
|
+
// 补丁:加密态可能晚到(sse.js 在收尾帧补发无文本的 x_reasoning_item)
|
|
175
|
+
if (!reasoningEncrypted && typeof meta?.encrypted_content === "string") reasoningEncrypted = meta.encrypted_content;
|
|
176
|
+
return [];
|
|
177
|
+
}
|
|
178
|
+
reasoningOpen = true;
|
|
179
|
+
reasoningId = meta?.id || `rs_${id}`;
|
|
180
|
+
reasoningEncrypted = typeof meta?.encrypted_content === "string" ? meta.encrypted_content : null;
|
|
181
|
+
reasoningIdx = nextIdx++;
|
|
182
|
+
const item = { type: "reasoning", id: reasoningId, summary: [], ...(reasoningEncrypted ? { encrypted_content: reasoningEncrypted } : {}) };
|
|
183
|
+
return [
|
|
184
|
+
ev("response.output_item.added", { output_index: reasoningIdx, item }),
|
|
185
|
+
ev("response.reasoning_summary_part.added", { item_id: reasoningId, output_index: reasoningIdx, summary_index: 0, part: { type: "summary_text", text: "" } }),
|
|
186
|
+
];
|
|
187
|
+
}
|
|
188
|
+
|
|
142
189
|
function pushTool(tc) {
|
|
143
190
|
const idx = Number(tc.index ?? 0);
|
|
144
191
|
let t = tools.get(idx);
|
|
145
|
-
if (!t) { t = { id: "", name: "", args: "", announced: false }; tools.set(idx, t); }
|
|
192
|
+
if (!t) { t = { id: "", name: "", args: "", announced: false, outputIndex: null }; tools.set(idx, t); }
|
|
146
193
|
if (tc.id) t.id = tc.id;
|
|
147
194
|
if (tc.function?.name) t.name += tc.function.name;
|
|
148
195
|
const frag = tc.function?.arguments || "";
|
|
149
196
|
const out = [];
|
|
150
197
|
if (!t.announced && t.id && t.name) {
|
|
151
198
|
t.announced = true;
|
|
152
|
-
|
|
199
|
+
t.outputIndex = nextIdx++;
|
|
200
|
+
out.push(ev("response.output_item.added", { output_index: t.outputIndex, item: { type: "function_call", id: `fc_${id}_${idx}`, call_id: t.id, name: t.name, arguments: "" } }));
|
|
153
201
|
}
|
|
154
202
|
if (frag) {
|
|
155
203
|
if (!t.announced) { t.args += frag; return out; } // id/name 未到先攒着
|
|
156
204
|
t.args += frag;
|
|
157
|
-
out.push(ev("response.function_call_arguments.delta", { item_id: `fc_${id}_${idx}`, output_index:
|
|
205
|
+
out.push(ev("response.function_call_arguments.delta", { item_id: `fc_${id}_${idx}`, output_index: t.outputIndex, delta: frag }));
|
|
158
206
|
}
|
|
159
207
|
return out;
|
|
160
208
|
}
|
|
@@ -181,29 +229,34 @@ export function createChunkTranslator(model = "") {
|
|
|
181
229
|
if (typeof delta.content === "string" && delta.content) { dbg.textChars += delta.content.length; out.push(...pushText(delta.content)); }
|
|
182
230
|
for (const tc of delta.tool_calls || []) { dbg.toolDeltas++; out.push(...pushTool(tc)); }
|
|
183
231
|
const rc = typeof delta.reasoning_content === "string" ? delta.reasoning_content : "";
|
|
184
|
-
if (rc) { dbg.reasoningChars += rc.length; out.push(...
|
|
232
|
+
if (rc || chunk.x_reasoning_item) { if (rc) dbg.reasoningChars += rc.length; out.push(...pushReasoning(rc, chunk.x_reasoning_item)); }
|
|
185
233
|
}
|
|
186
234
|
return out;
|
|
187
235
|
}
|
|
188
236
|
|
|
189
237
|
function end({ finish = "stop", usage = null } = {}) {
|
|
190
238
|
const out = [];
|
|
239
|
+
if (reasoningOpen) {
|
|
240
|
+
out.push(ev("response.reasoning_summary_text.done", { item_id: reasoningId, output_index: reasoningIdx, summary_index: 0, text: reasoningText }));
|
|
241
|
+
out.push(ev("response.reasoning_summary_part.done", { item_id: reasoningId, output_index: reasoningIdx, summary_index: 0, part: { type: "summary_text", text: reasoningText } }));
|
|
242
|
+
out.push(ev("response.output_item.done", { output_index: reasoningIdx, item: { type: "reasoning", id: reasoningId, summary: [{ type: "summary_text", text: reasoningText }], ...(reasoningEncrypted ? { encrypted_content: reasoningEncrypted } : {}) } }));
|
|
243
|
+
}
|
|
191
244
|
if (textItemOpen) {
|
|
192
|
-
out.push(ev("response.output_text.done", { item_id: `msg_${id}`, output_index:
|
|
193
|
-
out.push(ev("response.content_part.done", { item_id: `msg_${id}`, output_index:
|
|
194
|
-
out.push(ev("response.output_item.done", { output_index:
|
|
245
|
+
out.push(ev("response.output_text.done", { item_id: `msg_${id}`, output_index: textIdx, content_index: 0, text: fullText }));
|
|
246
|
+
out.push(ev("response.content_part.done", { item_id: `msg_${id}`, output_index: textIdx, content_index: 0, part: { type: "output_text", text: fullText, annotations: [] } }));
|
|
247
|
+
out.push(ev("response.output_item.done", { output_index: textIdx, item: { type: "message", id: `msg_${id}`, status: "completed", role: "assistant", content: [{ type: "output_text", text: fullText, annotations: [] }] } }));
|
|
195
248
|
}
|
|
196
|
-
let i = 0;
|
|
197
249
|
for (const [idx, t] of tools) {
|
|
198
|
-
if (!t.id && !t.name && !t.args)
|
|
250
|
+
if (!t.id && !t.name && !t.args) continue;
|
|
199
251
|
if (!t.announced) {
|
|
200
|
-
|
|
252
|
+
t.announced = true;
|
|
253
|
+
t.outputIndex = nextIdx++;
|
|
254
|
+
out.push(ev("response.output_item.added", { output_index: t.outputIndex, item: { type: "function_call", id: `fc_${id}_${idx}`, call_id: t.id, name: t.name, arguments: "" } }));
|
|
201
255
|
}
|
|
202
|
-
if (t.args) out.push(ev("response.function_call_arguments.done", { item_id: `fc_${id}_${idx}`, output_index:
|
|
203
|
-
out.push(ev("response.output_item.done", { output_index:
|
|
204
|
-
i++;
|
|
256
|
+
if (t.args) out.push(ev("response.function_call_arguments.done", { item_id: `fc_${id}_${idx}`, output_index: t.outputIndex, arguments: t.args }));
|
|
257
|
+
out.push(ev("response.output_item.done", { output_index: t.outputIndex, item: { type: "function_call", id: `fc_${id}_${idx}`, call_id: t.id, name: t.name, arguments: t.args, status: "completed" } }));
|
|
205
258
|
}
|
|
206
|
-
void
|
|
259
|
+
void textLen;
|
|
207
260
|
out.push(ev("response.completed", { response: { id, object: "response", created_at: createdAt, status: finish === "length" ? "incomplete" : "completed", model, output: [], usage: toResponsesUsage(usage) } }));
|
|
208
261
|
return out;
|
|
209
262
|
}
|
|
@@ -112,6 +112,15 @@ export async function responsesHandler(ctx) {
|
|
|
112
112
|
inputKinds: Array.isArray(body?.input) ? [...new Set(body.input.map((i) => i?.type))] : null,
|
|
113
113
|
tools: Array.isArray(body?.tools) ? body.tools.length : 0,
|
|
114
114
|
instructionsLen: String(body?.instructions || "").length,
|
|
115
|
+
// 文本丢失定位:各 message item 的 content 形状(part 类型 + 文本长度 + 头部)
|
|
116
|
+
contentShapes: Array.isArray(body?.input)
|
|
117
|
+
? body.input.filter((i) => i && (i.type === "message" || i.role)).map((m) => ({
|
|
118
|
+
type: m.type ?? "(无type)", role: m.role,
|
|
119
|
+
cType: Array.isArray(m.content) ? "array" : typeof m.content,
|
|
120
|
+
parts: Array.isArray(m.content) ? m.content.map((p) => `${p?.type ?? "?"}:${typeof p?.text === "string" ? p.text.length : "-"}`).slice(0, 6) : null,
|
|
121
|
+
textHead: typeof m.content === "string" ? m.content.slice(0, 50) : null,
|
|
122
|
+
})).slice(0, 10)
|
|
123
|
+
: null,
|
|
115
124
|
}));
|
|
116
125
|
let chatBody;
|
|
117
126
|
try {
|
package/src/routes/stream.js
CHANGED
|
@@ -30,6 +30,12 @@ export const STREAM_TIMEOUT_MS = (() => {
|
|
|
30
30
|
return Number.isInteger(n) && n >= 0 ? n : 25_000;
|
|
31
31
|
})();
|
|
32
32
|
|
|
33
|
+
// 等首块期间的心跳间隔(SSE 注释帧,标准客户端忽略):上游偶发卡 90s+,避免客户端误判卡死/断连
|
|
34
|
+
export const KEEPALIVE_MS = (() => {
|
|
35
|
+
const n = Number(process.env.MSLXDFF_KEEPALIVE_MS);
|
|
36
|
+
return Number.isInteger(n) && n >= 0 ? n : 10_000;
|
|
37
|
+
})();
|
|
38
|
+
|
|
33
39
|
export const STALL_TIMEOUT_MS = (() => {
|
|
34
40
|
const n = Number(process.env.MSLXDFF_STALL_TIMEOUT_MS);
|
|
35
41
|
return Number.isInteger(n) && n > 0 ? n : 0;
|
|
@@ -46,7 +52,7 @@ export const MAX_STREAM_MS = (() => {
|
|
|
46
52
|
return Number.isInteger(n) && n > 0 ? n : 0;
|
|
47
53
|
})();
|
|
48
54
|
|
|
49
|
-
export async function relay(res, upRes, body, { onFirstChunk, onDownstreamAbort, streamTimeoutMs = STREAM_TIMEOUT_MS, fallback } = {}) {
|
|
55
|
+
export async function relay(res, upRes, body, { onFirstChunk, onDownstreamAbort, streamTimeoutMs = STREAM_TIMEOUT_MS, keepaliveMs = KEEPALIVE_MS, fallback } = {}) {
|
|
50
56
|
const t0 = performance.now();
|
|
51
57
|
const contentType = upRes.headers.get("content-type") || "";
|
|
52
58
|
// 需同时满足:客户端要流 + 上游真的是 SSE;避免 muse-spark 聚合 JSON 被误判为流式,或 workbuddy SSE 被聚合
|
|
@@ -110,6 +116,12 @@ export async function relay(res, upRes, body, { onFirstChunk, onDownstreamAbort,
|
|
|
110
116
|
let stalled = false;
|
|
111
117
|
let tooLong = false;
|
|
112
118
|
let stallTimer = null;
|
|
119
|
+
let pingTimer = keepaliveMs > 0
|
|
120
|
+
? setInterval(() => {
|
|
121
|
+
if (wroteAny) return;
|
|
122
|
+
try { res.write(": keepalive\n\n"); } catch { /* ignore */ }
|
|
123
|
+
}, keepaliveMs)
|
|
124
|
+
: null;
|
|
113
125
|
const armStall = () => {
|
|
114
126
|
if (stallTimer) clearTimeout(stallTimer);
|
|
115
127
|
stallTimer = STALL_TIMEOUT_MS
|
|
@@ -234,6 +246,7 @@ export async function relay(res, upRes, body, { onFirstChunk, onDownstreamAbort,
|
|
|
234
246
|
if (firstTimer) clearTimeout(firstTimer);
|
|
235
247
|
if (maxTimer) clearTimeout(maxTimer);
|
|
236
248
|
if (stallTimer) clearTimeout(stallTimer);
|
|
249
|
+
if (pingTimer) clearInterval(pingTimer);
|
|
237
250
|
}
|
|
238
251
|
if (timedOut && !wroteAny) {
|
|
239
252
|
res.removeListener("close", onClose);
|
|
@@ -115,6 +115,24 @@ export async function startServerLifecycle({ VERSION, token, created, upstream,
|
|
|
115
115
|
}).catch(() => {});
|
|
116
116
|
}, 100).unref?.();
|
|
117
117
|
|
|
118
|
+
// responses 模型判定改元数据驱动(models.dev provider.npm):就绪后注入,每小时重查使新模型自动识别
|
|
119
|
+
void (async () => {
|
|
120
|
+
try {
|
|
121
|
+
const { globalCapabilities } = await import("../model-capabilities/index.js");
|
|
122
|
+
const { setResponsesNpmIndex } = await import("../upstream-responses.js");
|
|
123
|
+
const svc = globalCapabilities();
|
|
124
|
+
const apply = () => setResponsesNpmIndex(svc.npmIndex());
|
|
125
|
+
await svc.ready();
|
|
126
|
+
apply();
|
|
127
|
+
const entry = { ts: Date.now(), type: "responses-npm-index", models: svc.npmIndex().size };
|
|
128
|
+
try { bus.emit(entry); } catch {}
|
|
129
|
+
try { logs.appendEvent(entry); } catch {}
|
|
130
|
+
setInterval(() => { svc.ready().then(apply).catch(() => {}); }, 60 * 60 * 1000).unref?.();
|
|
131
|
+
} catch (e) {
|
|
132
|
+
try { logs.appendEvent({ ts: Date.now(), type: "responses-npm-index-failed", error: String(e?.message || e).slice(0, 200) }); } catch {}
|
|
133
|
+
}
|
|
134
|
+
})();
|
|
135
|
+
|
|
118
136
|
models.startAutoRefresh();
|
|
119
137
|
if (process.env.MSLXDFF_DAEMON) {
|
|
120
138
|
writePid(process.pid, VERSION);
|
|
@@ -57,9 +57,9 @@ export function errorResponseFromSdkError(e, { marker = null } = {}) {
|
|
|
57
57
|
});
|
|
58
58
|
}
|
|
59
59
|
|
|
60
|
-
// SDK parts → OpenAI SSE Response(chat/responses
|
|
61
|
-
export function streamResponseFromParts(parts, { marker = null, clock = Date.now, t0 = clock() } = {}) {
|
|
62
|
-
const ser = createSseSerializer();
|
|
60
|
+
// SDK parts → OpenAI SSE Response(chat/responses 共用同一序列化器;captured 为 responses 侧信道)。
|
|
61
|
+
export function streamResponseFromParts(parts, { marker = null, clock = Date.now, t0 = clock(), captured = null } = {}) {
|
|
62
|
+
const ser = createSseSerializer(captured);
|
|
63
63
|
const enc = new TextEncoder();
|
|
64
64
|
let cancelled = false;
|
|
65
65
|
const stream = new ReadableStream({
|
|
@@ -67,6 +67,11 @@ export function streamResponseFromParts(parts, { marker = null, clock = Date.now
|
|
|
67
67
|
try {
|
|
68
68
|
for await (const part of parts) {
|
|
69
69
|
if (cancelled) break;
|
|
70
|
+
// 侧信道(captured)可能落后几毫秒:收尾前给 encrypted 一点就绪时间(最多 150ms)
|
|
71
|
+
if (captured && !captured.reasoning && part?.type === "finish") {
|
|
72
|
+
const tw = Date.now();
|
|
73
|
+
while (!captured.reasoning && Date.now() - tw < 150) await new Promise((r) => setTimeout(r, 10));
|
|
74
|
+
}
|
|
70
75
|
const text = ser.push(part);
|
|
71
76
|
if (text) controller.enqueue(enc.encode(text));
|
|
72
77
|
}
|
|
@@ -52,7 +52,21 @@ export function toModelPrompt(messages) {
|
|
|
52
52
|
const parts = [];
|
|
53
53
|
const text = textOf(m.content);
|
|
54
54
|
if (text) parts.push({ type: "text", text });
|
|
55
|
-
|
|
55
|
+
// 带加密态的 reasoning items(responses 通道思考跨轮):AI SDK 会转成上游要的 encrypted reasoning item。
|
|
56
|
+
// 无加密态时才退回纯文本 reasoning(chat 通道的 reasoning_content)。
|
|
57
|
+
const items = Array.isArray(m.reasoning_items) ? m.reasoning_items : [];
|
|
58
|
+
let pushedEncrypted = false;
|
|
59
|
+
for (const r of items) {
|
|
60
|
+
if (!r || typeof r !== "object") continue;
|
|
61
|
+
const summaryText = Array.isArray(r.summary) ? r.summary.map((s) => s?.text || "").join("\n") : "";
|
|
62
|
+
parts.push({
|
|
63
|
+
type: "reasoning",
|
|
64
|
+
text: summaryText || " ",
|
|
65
|
+
providerOptions: { openai: { itemId: r.id, reasoningEncryptedContent: r.encrypted_content } },
|
|
66
|
+
});
|
|
67
|
+
pushedEncrypted = true;
|
|
68
|
+
}
|
|
69
|
+
if (!pushedEncrypted && m.reasoning_content) parts.push({ type: "reasoning", text: String(m.reasoning_content) });
|
|
56
70
|
for (const tc of Array.isArray(m.tool_calls) ? m.tool_calls : []) {
|
|
57
71
|
if (!tc || typeof tc !== "object") continue;
|
|
58
72
|
parts.push({
|
|
@@ -8,6 +8,50 @@ import { ENGINE_MARKER } from "./chat.js";
|
|
|
8
8
|
|
|
9
9
|
export const RESPONSES_CHAT_PATH = "/zen/v1/responses";
|
|
10
10
|
|
|
11
|
+
// 上游 encrypted reasoning 只在 output_item.done 里给,而 AI SDK 仅在有 summary 文本时才透出到 parts
|
|
12
|
+
// (muse-spark 这类无 summary 的思考模型会被吞掉)。这里在 fetch 层 tee 一份原始 SSE 自行解析,
|
|
13
|
+
// 侧信道把加密思考交给序列化器,收尾帧补发。
|
|
14
|
+
function captureReasoningFetch(baseFetch, sink) {
|
|
15
|
+
const dbg = process.env.MSLXDFF_RESPONSES_DEBUG === "1";
|
|
16
|
+
return async (input, init) => {
|
|
17
|
+
const res = await baseFetch(input, init);
|
|
18
|
+
const ct = res.headers.get("content-type") || "";
|
|
19
|
+
if (dbg) console.log(`[capture] enter ct=${ct} hasBody=${Boolean(res.body)}`);
|
|
20
|
+
if (!ct.includes("text/event-stream") || !res.body) return res;
|
|
21
|
+
const [passthrough, probe] = res.body.tee();
|
|
22
|
+
(async () => {
|
|
23
|
+
const reader = probe.getReader();
|
|
24
|
+
const dec = new TextDecoder();
|
|
25
|
+
let buf = "";
|
|
26
|
+
try {
|
|
27
|
+
for (;;) {
|
|
28
|
+
const { done, value } = await reader.read();
|
|
29
|
+
if (done) break;
|
|
30
|
+
buf += dec.decode(value, { stream: true });
|
|
31
|
+
let nl;
|
|
32
|
+
while ((nl = buf.indexOf("\n")) >= 0) {
|
|
33
|
+
const line = buf.slice(0, nl).trim();
|
|
34
|
+
buf = buf.slice(nl + 1);
|
|
35
|
+
if (!line.startsWith("data:")) continue;
|
|
36
|
+
const d = line.slice(5).trim();
|
|
37
|
+
if (!d || d === "[DONE]") continue;
|
|
38
|
+
try {
|
|
39
|
+
const j = JSON.parse(d);
|
|
40
|
+
if (dbg && j.type === "response.output_item.done") console.log(`[capture] item.done type=${j.item?.type} enc=${typeof j.item?.encrypted_content}`);
|
|
41
|
+
if (j.type === "response.output_item.done" && j.item?.type === "reasoning" && typeof j.item.encrypted_content === "string") {
|
|
42
|
+
sink.reasoning = { id: j.item.id, encrypted: j.item.encrypted_content, summary: j.item.summary ?? [] };
|
|
43
|
+
if (dbg) console.log(`[capture] hit id=${String(j.item.id).slice(0, 24)} len=${j.item.encrypted_content.length}`);
|
|
44
|
+
}
|
|
45
|
+
} catch { /* ignore */ }
|
|
46
|
+
}
|
|
47
|
+
}
|
|
48
|
+
} catch { /* ignore */ }
|
|
49
|
+
if (dbg) console.log(`[capture] stream end captured=${sink.reasoning ? "yes" : "no"}`);
|
|
50
|
+
})();
|
|
51
|
+
return new Response(passthrough, { status: res.status, statusText: res.statusText, headers: res.headers });
|
|
52
|
+
};
|
|
53
|
+
}
|
|
54
|
+
|
|
11
55
|
let sdkPromise = null;
|
|
12
56
|
|
|
13
57
|
export function loadOpenAISdk() {
|
|
@@ -37,19 +81,34 @@ export async function attemptOnceResponsesSdk({
|
|
|
37
81
|
throw err;
|
|
38
82
|
}
|
|
39
83
|
// headers 由调用方构造(含 Authorization),headers 优先于 apiKey 默认头
|
|
84
|
+
const captured = { reasoning: null };
|
|
85
|
+
const baseFetch = fetchImpl ?? ((u, i) => globalThis.fetch(u, i));
|
|
40
86
|
const provider = createOpenAI({
|
|
41
87
|
name: providerName,
|
|
42
88
|
baseURL,
|
|
43
89
|
apiKey: "public",
|
|
44
90
|
headers: sanitizeHeaders(headers),
|
|
45
|
-
|
|
91
|
+
fetch: captureReasoningFetch(baseFetch, captured),
|
|
46
92
|
});
|
|
47
93
|
const model = provider.responses(String(body?.model || ""));
|
|
94
|
+
const params = toModelParams(body, providerName);
|
|
95
|
+
// 无状态 + 加密思考回传(thinking 跨轮):上游把思考以加密块发回,客户端持有并每轮带回。
|
|
96
|
+
// AI SDK responses 的 providerOptionsName 对非 azure 硬编码为 "openai"(dist/index.js:5240)。
|
|
97
|
+
const providerOptions = {
|
|
98
|
+
...(params.providerOptions || {}),
|
|
99
|
+
openai: {
|
|
100
|
+
store: false,
|
|
101
|
+
include: ["reasoning.encrypted_content"],
|
|
102
|
+
reasoningSummary: "auto",
|
|
103
|
+
...((params.providerOptions || {}).openai || {}),
|
|
104
|
+
},
|
|
105
|
+
};
|
|
48
106
|
let res;
|
|
49
107
|
try {
|
|
50
108
|
res = await model.doStream({
|
|
51
109
|
prompt: toModelPrompt(body?.messages),
|
|
52
|
-
...
|
|
110
|
+
...params,
|
|
111
|
+
providerOptions,
|
|
53
112
|
tools: toModelTools(body?.tools),
|
|
54
113
|
toolChoice: toModelToolChoice(body?.tool_choice),
|
|
55
114
|
});
|
|
@@ -58,7 +117,7 @@ export async function attemptOnceResponsesSdk({
|
|
|
58
117
|
if (mapped) return mapped;
|
|
59
118
|
throw e;
|
|
60
119
|
}
|
|
61
|
-
return streamResponseFromParts(res.stream, { marker, clock, t0 });
|
|
120
|
+
return streamResponseFromParts(res.stream, { marker, clock, t0, captured });
|
|
62
121
|
}
|
|
63
122
|
|
|
64
123
|
export function createSdkResponses({
|
|
@@ -32,14 +32,18 @@ export function usageToOpenAI(u) {
|
|
|
32
32
|
return out;
|
|
33
33
|
}
|
|
34
34
|
|
|
35
|
-
export function createSseSerializer() {
|
|
35
|
+
export function createSseSerializer(captured = null) {
|
|
36
36
|
const meta = { id: "chatcmpl-wb-sdk", model: "", created: Math.floor(Date.now() / 1000) };
|
|
37
37
|
let roleSent = false;
|
|
38
38
|
let nextIndex = 0;
|
|
39
39
|
const toolIndex = new Map();
|
|
40
40
|
const toolDeltaIds = new Set();
|
|
41
|
+
// 加密思考往返:reasoning-start 的 providerMetadata 带 itemId + 加密态,随首帧透出
|
|
42
|
+
let reasoningMeta = null;
|
|
43
|
+
let reasoningFrameSent = false;
|
|
44
|
+
let reasoningEncSent = false;
|
|
41
45
|
|
|
42
|
-
function frame(delta, { finishReason = null, usage } = {}) {
|
|
46
|
+
function frame(delta, { finishReason = null, usage, extra } = {}) {
|
|
43
47
|
const obj = {
|
|
44
48
|
id: meta.id,
|
|
45
49
|
object: "chat.completion.chunk",
|
|
@@ -48,6 +52,7 @@ export function createSseSerializer() {
|
|
|
48
52
|
choices: [{ index: 0, delta, finish_reason: finishReason }],
|
|
49
53
|
};
|
|
50
54
|
if (usage !== undefined) obj.usage = usage;
|
|
55
|
+
if (extra) Object.assign(obj, extra);
|
|
51
56
|
return `data: ${JSON.stringify(obj)}\n\n`;
|
|
52
57
|
}
|
|
53
58
|
|
|
@@ -69,8 +74,30 @@ export function createSseSerializer() {
|
|
|
69
74
|
}
|
|
70
75
|
return null;
|
|
71
76
|
}
|
|
72
|
-
case "reasoning-
|
|
73
|
-
|
|
77
|
+
case "reasoning-start": {
|
|
78
|
+
const pm = part.providerMetadata?.openai || part.providerMetadata || {};
|
|
79
|
+
reasoningMeta = {
|
|
80
|
+
id: String(pm.itemId ?? part.id ?? "reasoning"),
|
|
81
|
+
encrypted: typeof pm.reasoningEncryptedContent === "string" ? pm.reasoningEncryptedContent : null,
|
|
82
|
+
};
|
|
83
|
+
// 即时透出 item 元数据(含加密态):上游可能只给 encrypted 不给 summary 文本,不能等 delta
|
|
84
|
+
reasoningFrameSent = true;
|
|
85
|
+
if (reasoningMeta.encrypted) reasoningEncSent = true;
|
|
86
|
+
return ensureRole() + frame({ reasoning_content: "" }, { extra: { x_reasoning_item: { id: reasoningMeta.id, encrypted_content: reasoningMeta.encrypted } } });
|
|
87
|
+
}
|
|
88
|
+
case "reasoning-delta": {
|
|
89
|
+
const delta = String(part.delta ?? "");
|
|
90
|
+
let extra;
|
|
91
|
+
if (reasoningMeta && !reasoningFrameSent) {
|
|
92
|
+
reasoningFrameSent = true;
|
|
93
|
+
if (reasoningMeta.encrypted) reasoningEncSent = true;
|
|
94
|
+
// x_reasoning_item:responses translator 据此建 reasoning item(chat 客户端忽略未知顶层字段)
|
|
95
|
+
extra = { x_reasoning_item: { id: reasoningMeta.id, encrypted_content: reasoningMeta.encrypted } };
|
|
96
|
+
} else if (reasoningMeta) {
|
|
97
|
+
extra = { x_reasoning_id: reasoningMeta.id };
|
|
98
|
+
}
|
|
99
|
+
return ensureRole() + frame({ reasoning_content: delta }, { extra });
|
|
100
|
+
}
|
|
74
101
|
case "text-delta":
|
|
75
102
|
return ensureRole() + frame({ content: String(part.delta ?? "") });
|
|
76
103
|
case "tool-input-start": {
|
|
@@ -98,7 +125,14 @@ export function createSseSerializer() {
|
|
|
98
125
|
const fr = part.finishReason;
|
|
99
126
|
const raw = typeof fr === "string" ? fr : (fr?.raw ?? fr?.unified);
|
|
100
127
|
const reason = raw === "tool-calls" ? "tool_calls" : raw === "content-filter" ? "content_filter" : (raw ?? "stop");
|
|
101
|
-
|
|
128
|
+
// 加密思考补发:上游 encrypted_content 只在流末尾可得(fetch 侧信道捕获),
|
|
129
|
+
// 无 summary 文本时 AI SDK parts 不会带出 → 收尾帧前补一帧
|
|
130
|
+
let pre = "";
|
|
131
|
+
if (captured?.reasoning && !reasoningEncSent) {
|
|
132
|
+
reasoningEncSent = true;
|
|
133
|
+
pre = ensureRole() + frame({ reasoning_content: "" }, { extra: { x_reasoning_item: { id: captured.reasoning.id, encrypted_content: captured.reasoning.encrypted } } });
|
|
134
|
+
}
|
|
135
|
+
return pre + frame({}, { finishReason: reason, usage: usageToOpenAI(part.usage) });
|
|
102
136
|
}
|
|
103
137
|
case "error": {
|
|
104
138
|
const message = part.error?.message
|
|
@@ -2,8 +2,26 @@
|
|
|
2
2
|
* responses 转换层 — 从 upstream.js 抽出的 muse-spark 专用形状转换。
|
|
3
3
|
* chat ⇄ responses 互转纯函数,无网络、无副作用。
|
|
4
4
|
*/
|
|
5
|
+
// responses 判定索引(models.dev 模型级 provider.npm,启动时注入):
|
|
6
|
+
// "@ai-sdk/openai" → responses 端点;未注入/未命中 → 前缀兜底(新模型早于 models.dev 刷新时仍可用)。
|
|
7
|
+
let responsesNpmIndex = null;
|
|
8
|
+
|
|
9
|
+
export function setResponsesNpmIndex(idx) {
|
|
10
|
+
responsesNpmIndex = idx instanceof Map ? idx : null;
|
|
11
|
+
}
|
|
12
|
+
|
|
13
|
+
export function _resetResponsesNpmIndex() {
|
|
14
|
+
responsesNpmIndex = null;
|
|
15
|
+
}
|
|
16
|
+
|
|
5
17
|
export function isResponsesModel(model) {
|
|
6
|
-
|
|
18
|
+
const m = String(model || "").toLowerCase().trim();
|
|
19
|
+
if (!m) return false;
|
|
20
|
+
if (responsesNpmIndex) {
|
|
21
|
+
const bare = m.replace(/^opencode\//, "");
|
|
22
|
+
if (responsesNpmIndex.has(bare)) return responsesNpmIndex.get(bare) === "@ai-sdk/openai";
|
|
23
|
+
}
|
|
24
|
+
return m.startsWith("muse-spark");
|
|
7
25
|
}
|
|
8
26
|
|
|
9
27
|
export function chatToResponsesBody(chatBody) {
|
package/src/upstream.js
CHANGED
|
@@ -12,6 +12,24 @@ import { uuid } from "./compat.js";
|
|
|
12
12
|
function genId(prefix) {
|
|
13
13
|
return `${prefix}${uuid().replace(/-/g, "")}`;
|
|
14
14
|
}
|
|
15
|
+
// 上游按 session 做粘性路由(实测:固定 session 两次请求均 ~1.2s;每次随机时可能撞冷机器 26s+)。
|
|
16
|
+
// 客户端(opencode AI SDK 路径)不带会话标识 → 用对话首两条消息(system + 首条 user)哈希做稳定会话:
|
|
17
|
+
// 同一会话多轮里这两条不变 ⇒ 路由亲和稳定;不同会话天然分散。
|
|
18
|
+
function sessionFromMessages(messages) {
|
|
19
|
+
try {
|
|
20
|
+
const msgs = Array.isArray(messages) ? messages : [];
|
|
21
|
+
const pick = (role) => {
|
|
22
|
+
const m = msgs.find((x) => x?.role === role);
|
|
23
|
+
if (!m) return "";
|
|
24
|
+
return typeof m.content === "string" ? m.content : JSON.stringify(m.content ?? "");
|
|
25
|
+
};
|
|
26
|
+
const seed = `${pick("system")}|${pick("user")}`.slice(0, 4000);
|
|
27
|
+
if (seed === "|") return null;
|
|
28
|
+
return `ses_${crypto.createHash("sha1").update(seed).digest("hex").slice(0, 32)}`;
|
|
29
|
+
} catch {
|
|
30
|
+
return null;
|
|
31
|
+
}
|
|
32
|
+
}
|
|
15
33
|
function envInt(name, fallback) {
|
|
16
34
|
const v = Number(process.env[name]);
|
|
17
35
|
return Number.isInteger(v) && v > 0 ? v : fallback;
|
|
@@ -47,13 +65,16 @@ export function createOpencodeHeaderBuilder({ authToken = "public", env = proces
|
|
|
47
65
|
Authorization: `Bearer ${authToken}`,
|
|
48
66
|
"x-opencode-client": "desktop",
|
|
49
67
|
};
|
|
50
|
-
|
|
68
|
+
// 无 messages 可哈希时的进程级兜底:至少 daemon 生命周期内稳定,不再每请求随机
|
|
69
|
+
const FALLBACK_SESSION = genId("ses_");
|
|
70
|
+
function buildHeaders(body, { anonymous = anonFirst, sessionId = null } = {}) {
|
|
51
71
|
const isStream = body?.stream !== false;
|
|
72
|
+
const session = sessionId || sessionFromMessages(body?.messages) || FALLBACK_SESSION;
|
|
52
73
|
const base = {
|
|
53
74
|
...baseHeaders,
|
|
54
75
|
Accept: isStream ? "text/event-stream" : "*/*",
|
|
55
76
|
"User-Agent": "opencode",
|
|
56
|
-
"x-opencode-session":
|
|
77
|
+
"x-opencode-session": session,
|
|
57
78
|
"x-opencode-request": genId("msg_"),
|
|
58
79
|
"x-opencode-project": "global",
|
|
59
80
|
};
|
|
@@ -111,8 +132,10 @@ export function createUpstreamClient({
|
|
|
111
132
|
hooks,
|
|
112
133
|
});
|
|
113
134
|
|
|
114
|
-
async function chat(body) {
|
|
135
|
+
async function chat(body, opts = {}) {
|
|
115
136
|
const isResp = isResponsesModel(body?.model);
|
|
137
|
+
// responses 路径的 reqBody 无 messages,会话哈希必须基于原始 chat body 计算
|
|
138
|
+
const sessionId = opts?.sessionId || sessionFromMessages(body?.messages) || null;
|
|
116
139
|
const url = isResp ? `${baseUrl}/zen/v1/responses` : `${baseUrl}/zen/v1/chat/completions`;
|
|
117
140
|
const reqBody = isResp ? chatToResponsesBody(body) : body;
|
|
118
141
|
const t0 = performance.now();
|
|
@@ -123,7 +146,7 @@ export function createUpstreamClient({
|
|
|
123
146
|
res = await transport.request({
|
|
124
147
|
url,
|
|
125
148
|
method: "POST",
|
|
126
|
-
headers: buildHeaders(reqBody),
|
|
149
|
+
headers: buildHeaders(reqBody, { sessionId }),
|
|
127
150
|
body: reqBody,
|
|
128
151
|
stream: body?.stream !== false,
|
|
129
152
|
timeoutMs: connectTimeoutMs,
|
|
@@ -145,7 +168,7 @@ export function createUpstreamClient({
|
|
|
145
168
|
anonRes = await transport.request({
|
|
146
169
|
url,
|
|
147
170
|
method: "POST",
|
|
148
|
-
headers: buildHeaders(reqBody, { anonymous: true }),
|
|
171
|
+
headers: buildHeaders(reqBody, { anonymous: true, sessionId }),
|
|
149
172
|
body: reqBody,
|
|
150
173
|
stream: body?.stream !== false,
|
|
151
174
|
timeoutMs: connectTimeoutMs,
|