mslxdff 0.1.114 → 0.1.115
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/src/chat-pipeline/auto-race.js +1 -0
- package/src/chat-pipeline/index.js +5 -1
- package/src/chat-pipeline/serial-trial.js +1 -0
- package/src/model-capabilities/index.js +9 -2
- package/src/model-capabilities/parse.js +2 -0
- package/src/providers/dispatcher.js +2 -2
- package/src/providers/opencode.js +1 -1
- package/src/responses/translate.js +72 -20
- package/src/routes/stream.js +14 -1
- package/src/runtime/server-lifecycle.js +18 -0
- package/src/upstream-engine/sdk/attempt.js +8 -3
- package/src/upstream-engine/sdk/convert.js +15 -1
- package/src/upstream-engine/sdk/responses.js +62 -3
- package/src/upstream-engine/sdk/sse.js +39 -5
- package/src/upstream-responses.js +19 -1
- package/src/upstream.js +28 -5
package/package.json
CHANGED
|
@@ -65,6 +65,7 @@ export async function runAutoRace(ctx, deps = {}) {
|
|
|
65
65
|
const o = {};
|
|
66
66
|
if (Object.keys(shareKeys).length) o.shareKeys = shareKeys;
|
|
67
67
|
if (workbuddyUid) o.workbuddyUid = workbuddyUid;
|
|
68
|
+
if (handlerCtx?.sessionId) o.sessionId = handlerCtx.sessionId;
|
|
68
69
|
r = await upstream.chat(f, Object.keys(o).length ? o : undefined);
|
|
69
70
|
} catch (e) {
|
|
70
71
|
if (plugins?.length) runHook(plugins, "upstream:response", { reqId, requested, model: m, status: null, ok: false, error: errMsg(e), timing: e?._t ?? null }).catch(() => {});
|
|
@@ -82,7 +82,11 @@ export function createChatPipeline({ upstream, auto, logs, peers, groups, bus, t
|
|
|
82
82
|
for (const e of sel.errors) evt("plugin-hook-error", { reqId, hook: "model:select", plugin: e.plugin, error: e.error });
|
|
83
83
|
}
|
|
84
84
|
|
|
85
|
-
|
|
85
|
+
// 客户端会话标识(opencode 插件 chat.headers 注入)→ 透传上游做粘性路由/缓存亲和;
|
|
86
|
+
// 无头时由 upstream.js 按对话首两条消息哈希兜底(不再每请求随机)
|
|
87
|
+
const clientSession = String(req?.headers?.["x-session-affinity"] || req?.headers?.["x-session-id"] || "").trim() || null;
|
|
88
|
+
const handlerCtx = { reqId, model: null, body: req?.body, hops, peers, plugins, evt, logError, logCall, logs, workbuddyUid, sessionId: clientSession };
|
|
89
|
+
if (clientSession) evt("client-session", { reqId, sessionId: clientSession.slice(0, 24) });
|
|
86
90
|
|
|
87
91
|
const plan = planRoute(policy, {
|
|
88
92
|
candidates: order,
|
|
@@ -70,6 +70,7 @@ export async function runSerialTrial(ctx, deps = {}) {
|
|
|
70
70
|
const chatOpts = {};
|
|
71
71
|
if (Object.keys(shareKeys).length) chatOpts.shareKeys = shareKeys;
|
|
72
72
|
if (workbuddyUid) chatOpts.workbuddyUid = workbuddyUid;
|
|
73
|
+
if (handlerCtx?.sessionId) chatOpts.sessionId = handlerCtx.sessionId;
|
|
73
74
|
upRes = await upstream.chat(forwarded, Object.keys(chatOpts).length ? chatOpts : undefined);
|
|
74
75
|
evt("upstream-done", { reqId, model, ok: !(upRes instanceof Error) && upRes.status < 400, status: upRes instanceof Error ? null : upRes.status, timing: upRes._t ?? null, error: null });
|
|
75
76
|
} catch (err) {
|
|
@@ -28,15 +28,22 @@ export function createCapabilitiesService({
|
|
|
28
28
|
} = {}) {
|
|
29
29
|
let raw = null; // 原始目录(全 provider)
|
|
30
30
|
let capsIndex = new Map(); // providerId -> { modelId -> caps }
|
|
31
|
+
let npmIndex = new Map(); // opencode 裸 modelId -> provider.npm(responses 判定用;null = 继承默认)
|
|
31
32
|
let loadedAt = 0;
|
|
32
33
|
let inflight = null;
|
|
33
34
|
|
|
34
35
|
function buildIndex(data) {
|
|
35
36
|
const idx = new Map();
|
|
37
|
+
const npm = new Map();
|
|
36
38
|
for (const [pid, p] of Object.entries(data || {})) {
|
|
37
39
|
if (!p || typeof p !== "object") continue;
|
|
38
|
-
|
|
40
|
+
const caps = normalizeProviderModels(p.models || {});
|
|
41
|
+
idx.set(pid, caps);
|
|
42
|
+
if (String(pid).toLowerCase() === "opencode") {
|
|
43
|
+
for (const [mid, c] of Object.entries(caps)) npm.set(mid, c.npm ?? null);
|
|
44
|
+
}
|
|
39
45
|
}
|
|
46
|
+
npmIndex = npm;
|
|
40
47
|
return idx;
|
|
41
48
|
}
|
|
42
49
|
|
|
@@ -111,7 +118,7 @@ export function createCapabilitiesService({
|
|
|
111
118
|
return [...capsIndex.keys()].sort();
|
|
112
119
|
}
|
|
113
120
|
|
|
114
|
-
return { ready, get, list, providers };
|
|
121
|
+
return { ready, get, list, providers, npmIndex: () => new Map(npmIndex) };
|
|
115
122
|
}
|
|
116
123
|
|
|
117
124
|
// 模块级单例(与 globalDedup 同模式):HTTP handler 懒加载,测试 _reset 后注入
|
|
@@ -24,6 +24,8 @@ export function normalizeModelCaps(_id, m) {
|
|
|
24
24
|
releaseDate: typeof m?.release_date === "string" && m.release_date ? m.release_date : null,
|
|
25
25
|
inputModalities: input.length ? input : ["text"],
|
|
26
26
|
outputModalities: Array.isArray(m?.modalities?.output) && m.modalities.output.length ? m.modalities.output : ["text"],
|
|
27
|
+
// 模型级 SDK 覆盖(models.dev provider.npm):@ai-sdk/openai → responses 端点;null = 继承 provider 默认
|
|
28
|
+
npm: typeof m?.provider?.npm === "string" && m.provider.npm ? m.provider.npm : null,
|
|
27
29
|
};
|
|
28
30
|
}
|
|
29
31
|
|
|
@@ -53,9 +53,9 @@ export function createProviderDispatcher(providers = [], opts = {}) {
|
|
|
53
53
|
return provider.chatWithKeys(forwarded, sharedKeys);
|
|
54
54
|
}
|
|
55
55
|
if (provider.id === "workbuddy" && workbuddyUid) {
|
|
56
|
-
return provider.chat(forwarded, { workbuddyUid });
|
|
56
|
+
return provider.chat(forwarded, { ...opts, workbuddyUid });
|
|
57
57
|
}
|
|
58
|
-
return provider.chat(forwarded);
|
|
58
|
+
return provider.chat(forwarded, opts);
|
|
59
59
|
}
|
|
60
60
|
|
|
61
61
|
// 聚合所有供应商的模型列表;默认供应商(opencode)裸 id,其它带前缀
|
|
@@ -8,7 +8,7 @@ export function createOpenCodeProvider({ upstream, modelsService, baseUrl, authT
|
|
|
8
8
|
return {
|
|
9
9
|
id: "opencode",
|
|
10
10
|
upstream: client,
|
|
11
|
-
chat: (body) => client.chat(body),
|
|
11
|
+
chat: (body, opts) => client.chat(body, opts),
|
|
12
12
|
preheat: (args) => client.preheat(args),
|
|
13
13
|
close: () => client.close(),
|
|
14
14
|
async listModels() {
|
|
@@ -33,15 +33,27 @@ export function responsesToChatBody(req = {}) {
|
|
|
33
33
|
if (req.instructions) messages.push({ role: "system", content: String(req.instructions) });
|
|
34
34
|
const input = req.input;
|
|
35
35
|
const items = typeof input === "string" ? [{ type: "message", role: "user", content: input }] : Array.isArray(input) ? input : [];
|
|
36
|
+
// 加密思考往返:input 里的 reasoning item 挂到下一条 assistant 消息(thinking 跨轮必需,不能丢)
|
|
37
|
+
let pendingReasoning = [];
|
|
36
38
|
for (const it of items) {
|
|
37
39
|
if (!it || typeof it !== "object") continue;
|
|
40
|
+
if (it.type === "reasoning") {
|
|
41
|
+
if (it.id || it.encrypted_content) {
|
|
42
|
+
pendingReasoning.push({ id: it.id, encrypted_content: it.encrypted_content, summary: it.summary });
|
|
43
|
+
}
|
|
44
|
+
continue;
|
|
45
|
+
}
|
|
38
46
|
if (it.type === "message") {
|
|
39
|
-
|
|
47
|
+
const msg = { role: it.role || "user", content: inputTextOf(it.content) };
|
|
48
|
+
if (pendingReasoning.length && msg.role === "assistant") { msg.reasoning_items = pendingReasoning; pendingReasoning = []; }
|
|
49
|
+
messages.push(msg);
|
|
40
50
|
} else if (it.type === "function_call") {
|
|
41
|
-
|
|
51
|
+
const msg = {
|
|
42
52
|
role: "assistant", content: "",
|
|
43
53
|
tool_calls: [{ id: it.call_id || it.id || "", type: "function", function: { name: it.name || "", arguments: it.arguments || "" } }],
|
|
44
|
-
}
|
|
54
|
+
};
|
|
55
|
+
if (pendingReasoning.length) { msg.reasoning_items = pendingReasoning; pendingReasoning = []; }
|
|
56
|
+
messages.push(msg);
|
|
45
57
|
} else if (it.type === "function_call_output") {
|
|
46
58
|
messages.push({ role: "tool", tool_call_id: it.call_id || "", content: typeof it.output === "string" ? it.output : JSON.stringify(it.output ?? "") });
|
|
47
59
|
}
|
|
@@ -115,7 +127,16 @@ export function createChunkTranslator(model = "") {
|
|
|
115
127
|
let textLen = 0;
|
|
116
128
|
let lastFinish = "stop";
|
|
117
129
|
let lastUsage = null;
|
|
118
|
-
const tools = new Map(); // index → {id, name, args, announced}
|
|
130
|
+
const tools = new Map(); // index → {id, name, args, announced, outputIndex}
|
|
131
|
+
// output_index 按 item 出现顺序动态分配(reasoning 可能先于 message/tool 出现)
|
|
132
|
+
let nextIdx = 0;
|
|
133
|
+
let textIdx = null;
|
|
134
|
+
// 加密思考(thinking 跨轮):sse.js 首帧带 x_reasoning_item(含加密态),translator 据此建 reasoning item
|
|
135
|
+
let reasoningOpen = false;
|
|
136
|
+
let reasoningId = null;
|
|
137
|
+
let reasoningEncrypted = null;
|
|
138
|
+
let reasoningIdx = null;
|
|
139
|
+
let reasoningText = "";
|
|
119
140
|
const ev = (type, extra = {}) => ({ type, ...extra });
|
|
120
141
|
|
|
121
142
|
function begin() {
|
|
@@ -125,8 +146,9 @@ export function createChunkTranslator(model = "") {
|
|
|
125
146
|
function ensureTextItem() {
|
|
126
147
|
if (textItemOpen) return [];
|
|
127
148
|
textItemOpen = true;
|
|
149
|
+
textIdx = nextIdx++;
|
|
128
150
|
const item = { type: "message", id: `msg_${id}`, status: "in_progress", role: "assistant", content: [{ type: "output_text", text: "", annotations: [] }] };
|
|
129
|
-
return [ev("response.output_item.added", { output_index:
|
|
151
|
+
return [ev("response.output_item.added", { output_index: textIdx, item }), ev("response.content_part.added", { item_id: item.id, output_index: textIdx, content_index: 0, part: { type: "output_text", text: "", annotations: [] } })];
|
|
130
152
|
}
|
|
131
153
|
|
|
132
154
|
let fullText = "";
|
|
@@ -135,26 +157,51 @@ export function createChunkTranslator(model = "") {
|
|
|
135
157
|
const out = ensureTextItem();
|
|
136
158
|
textLen += delta.length;
|
|
137
159
|
fullText += delta;
|
|
138
|
-
out.push(ev("response.output_text.delta", { item_id: `msg_${id}`, output_index:
|
|
160
|
+
out.push(ev("response.output_text.delta", { item_id: `msg_${id}`, output_index: textIdx, content_index: 0, delta }));
|
|
161
|
+
return out;
|
|
162
|
+
}
|
|
163
|
+
|
|
164
|
+
function pushReasoning(delta, meta) {
|
|
165
|
+
const out = ensureReasoningItem(meta);
|
|
166
|
+
reasoningText += delta;
|
|
167
|
+
if (delta) out.push(ev("response.reasoning_summary_text.delta", { item_id: reasoningId, output_index: reasoningIdx, summary_index: 0, delta }));
|
|
139
168
|
return out;
|
|
140
169
|
}
|
|
141
170
|
|
|
171
|
+
function ensureReasoningItem(meta) {
|
|
172
|
+
if (reasoningOpen) {
|
|
173
|
+
// 补丁:加密态可能晚到(sse.js 在收尾帧补发无文本的 x_reasoning_item)
|
|
174
|
+
if (!reasoningEncrypted && typeof meta?.encrypted_content === "string") reasoningEncrypted = meta.encrypted_content;
|
|
175
|
+
return [];
|
|
176
|
+
}
|
|
177
|
+
reasoningOpen = true;
|
|
178
|
+
reasoningId = meta?.id || `rs_${id}`;
|
|
179
|
+
reasoningEncrypted = typeof meta?.encrypted_content === "string" ? meta.encrypted_content : null;
|
|
180
|
+
reasoningIdx = nextIdx++;
|
|
181
|
+
const item = { type: "reasoning", id: reasoningId, summary: [], ...(reasoningEncrypted ? { encrypted_content: reasoningEncrypted } : {}) };
|
|
182
|
+
return [
|
|
183
|
+
ev("response.output_item.added", { output_index: reasoningIdx, item }),
|
|
184
|
+
ev("response.reasoning_summary_part.added", { item_id: reasoningId, output_index: reasoningIdx, summary_index: 0, part: { type: "summary_text", text: "" } }),
|
|
185
|
+
];
|
|
186
|
+
}
|
|
187
|
+
|
|
142
188
|
function pushTool(tc) {
|
|
143
189
|
const idx = Number(tc.index ?? 0);
|
|
144
190
|
let t = tools.get(idx);
|
|
145
|
-
if (!t) { t = { id: "", name: "", args: "", announced: false }; tools.set(idx, t); }
|
|
191
|
+
if (!t) { t = { id: "", name: "", args: "", announced: false, outputIndex: null }; tools.set(idx, t); }
|
|
146
192
|
if (tc.id) t.id = tc.id;
|
|
147
193
|
if (tc.function?.name) t.name += tc.function.name;
|
|
148
194
|
const frag = tc.function?.arguments || "";
|
|
149
195
|
const out = [];
|
|
150
196
|
if (!t.announced && t.id && t.name) {
|
|
151
197
|
t.announced = true;
|
|
152
|
-
|
|
198
|
+
t.outputIndex = nextIdx++;
|
|
199
|
+
out.push(ev("response.output_item.added", { output_index: t.outputIndex, item: { type: "function_call", id: `fc_${id}_${idx}`, call_id: t.id, name: t.name, arguments: "" } }));
|
|
153
200
|
}
|
|
154
201
|
if (frag) {
|
|
155
202
|
if (!t.announced) { t.args += frag; return out; } // id/name 未到先攒着
|
|
156
203
|
t.args += frag;
|
|
157
|
-
out.push(ev("response.function_call_arguments.delta", { item_id: `fc_${id}_${idx}`, output_index:
|
|
204
|
+
out.push(ev("response.function_call_arguments.delta", { item_id: `fc_${id}_${idx}`, output_index: t.outputIndex, delta: frag }));
|
|
158
205
|
}
|
|
159
206
|
return out;
|
|
160
207
|
}
|
|
@@ -181,29 +228,34 @@ export function createChunkTranslator(model = "") {
|
|
|
181
228
|
if (typeof delta.content === "string" && delta.content) { dbg.textChars += delta.content.length; out.push(...pushText(delta.content)); }
|
|
182
229
|
for (const tc of delta.tool_calls || []) { dbg.toolDeltas++; out.push(...pushTool(tc)); }
|
|
183
230
|
const rc = typeof delta.reasoning_content === "string" ? delta.reasoning_content : "";
|
|
184
|
-
if (rc) { dbg.reasoningChars += rc.length; out.push(...
|
|
231
|
+
if (rc || chunk.x_reasoning_item) { if (rc) dbg.reasoningChars += rc.length; out.push(...pushReasoning(rc, chunk.x_reasoning_item)); }
|
|
185
232
|
}
|
|
186
233
|
return out;
|
|
187
234
|
}
|
|
188
235
|
|
|
189
236
|
function end({ finish = "stop", usage = null } = {}) {
|
|
190
237
|
const out = [];
|
|
238
|
+
if (reasoningOpen) {
|
|
239
|
+
out.push(ev("response.reasoning_summary_text.done", { item_id: reasoningId, output_index: reasoningIdx, summary_index: 0, text: reasoningText }));
|
|
240
|
+
out.push(ev("response.reasoning_summary_part.done", { item_id: reasoningId, output_index: reasoningIdx, summary_index: 0, part: { type: "summary_text", text: reasoningText } }));
|
|
241
|
+
out.push(ev("response.output_item.done", { output_index: reasoningIdx, item: { type: "reasoning", id: reasoningId, summary: [{ type: "summary_text", text: reasoningText }], ...(reasoningEncrypted ? { encrypted_content: reasoningEncrypted } : {}) } }));
|
|
242
|
+
}
|
|
191
243
|
if (textItemOpen) {
|
|
192
|
-
out.push(ev("response.output_text.done", { item_id: `msg_${id}`, output_index:
|
|
193
|
-
out.push(ev("response.content_part.done", { item_id: `msg_${id}`, output_index:
|
|
194
|
-
out.push(ev("response.output_item.done", { output_index:
|
|
244
|
+
out.push(ev("response.output_text.done", { item_id: `msg_${id}`, output_index: textIdx, content_index: 0, text: fullText }));
|
|
245
|
+
out.push(ev("response.content_part.done", { item_id: `msg_${id}`, output_index: textIdx, content_index: 0, part: { type: "output_text", text: fullText, annotations: [] } }));
|
|
246
|
+
out.push(ev("response.output_item.done", { output_index: textIdx, item: { type: "message", id: `msg_${id}`, status: "completed", role: "assistant", content: [{ type: "output_text", text: fullText, annotations: [] }] } }));
|
|
195
247
|
}
|
|
196
|
-
let i = 0;
|
|
197
248
|
for (const [idx, t] of tools) {
|
|
198
|
-
if (!t.id && !t.name && !t.args)
|
|
249
|
+
if (!t.id && !t.name && !t.args) continue;
|
|
199
250
|
if (!t.announced) {
|
|
200
|
-
|
|
251
|
+
t.announced = true;
|
|
252
|
+
t.outputIndex = nextIdx++;
|
|
253
|
+
out.push(ev("response.output_item.added", { output_index: t.outputIndex, item: { type: "function_call", id: `fc_${id}_${idx}`, call_id: t.id, name: t.name, arguments: "" } }));
|
|
201
254
|
}
|
|
202
|
-
if (t.args) out.push(ev("response.function_call_arguments.done", { item_id: `fc_${id}_${idx}`, output_index:
|
|
203
|
-
out.push(ev("response.output_item.done", { output_index:
|
|
204
|
-
i++;
|
|
255
|
+
if (t.args) out.push(ev("response.function_call_arguments.done", { item_id: `fc_${id}_${idx}`, output_index: t.outputIndex, arguments: t.args }));
|
|
256
|
+
out.push(ev("response.output_item.done", { output_index: t.outputIndex, item: { type: "function_call", id: `fc_${id}_${idx}`, call_id: t.id, name: t.name, arguments: t.args, status: "completed" } }));
|
|
205
257
|
}
|
|
206
|
-
void
|
|
258
|
+
void textLen;
|
|
207
259
|
out.push(ev("response.completed", { response: { id, object: "response", created_at: createdAt, status: finish === "length" ? "incomplete" : "completed", model, output: [], usage: toResponsesUsage(usage) } }));
|
|
208
260
|
return out;
|
|
209
261
|
}
|
package/src/routes/stream.js
CHANGED
|
@@ -30,6 +30,12 @@ export const STREAM_TIMEOUT_MS = (() => {
|
|
|
30
30
|
return Number.isInteger(n) && n >= 0 ? n : 25_000;
|
|
31
31
|
})();
|
|
32
32
|
|
|
33
|
+
// 等首块期间的心跳间隔(SSE 注释帧,标准客户端忽略):上游偶发卡 90s+,避免客户端误判卡死/断连
|
|
34
|
+
export const KEEPALIVE_MS = (() => {
|
|
35
|
+
const n = Number(process.env.MSLXDFF_KEEPALIVE_MS);
|
|
36
|
+
return Number.isInteger(n) && n >= 0 ? n : 10_000;
|
|
37
|
+
})();
|
|
38
|
+
|
|
33
39
|
export const STALL_TIMEOUT_MS = (() => {
|
|
34
40
|
const n = Number(process.env.MSLXDFF_STALL_TIMEOUT_MS);
|
|
35
41
|
return Number.isInteger(n) && n > 0 ? n : 0;
|
|
@@ -46,7 +52,7 @@ export const MAX_STREAM_MS = (() => {
|
|
|
46
52
|
return Number.isInteger(n) && n > 0 ? n : 0;
|
|
47
53
|
})();
|
|
48
54
|
|
|
49
|
-
export async function relay(res, upRes, body, { onFirstChunk, onDownstreamAbort, streamTimeoutMs = STREAM_TIMEOUT_MS, fallback } = {}) {
|
|
55
|
+
export async function relay(res, upRes, body, { onFirstChunk, onDownstreamAbort, streamTimeoutMs = STREAM_TIMEOUT_MS, keepaliveMs = KEEPALIVE_MS, fallback } = {}) {
|
|
50
56
|
const t0 = performance.now();
|
|
51
57
|
const contentType = upRes.headers.get("content-type") || "";
|
|
52
58
|
// 需同时满足:客户端要流 + 上游真的是 SSE;避免 muse-spark 聚合 JSON 被误判为流式,或 workbuddy SSE 被聚合
|
|
@@ -110,6 +116,12 @@ export async function relay(res, upRes, body, { onFirstChunk, onDownstreamAbort,
|
|
|
110
116
|
let stalled = false;
|
|
111
117
|
let tooLong = false;
|
|
112
118
|
let stallTimer = null;
|
|
119
|
+
let pingTimer = keepaliveMs > 0
|
|
120
|
+
? setInterval(() => {
|
|
121
|
+
if (wroteAny) return;
|
|
122
|
+
try { res.write(": keepalive\n\n"); } catch { /* ignore */ }
|
|
123
|
+
}, keepaliveMs)
|
|
124
|
+
: null;
|
|
113
125
|
const armStall = () => {
|
|
114
126
|
if (stallTimer) clearTimeout(stallTimer);
|
|
115
127
|
stallTimer = STALL_TIMEOUT_MS
|
|
@@ -234,6 +246,7 @@ export async function relay(res, upRes, body, { onFirstChunk, onDownstreamAbort,
|
|
|
234
246
|
if (firstTimer) clearTimeout(firstTimer);
|
|
235
247
|
if (maxTimer) clearTimeout(maxTimer);
|
|
236
248
|
if (stallTimer) clearTimeout(stallTimer);
|
|
249
|
+
if (pingTimer) clearInterval(pingTimer);
|
|
237
250
|
}
|
|
238
251
|
if (timedOut && !wroteAny) {
|
|
239
252
|
res.removeListener("close", onClose);
|
|
@@ -115,6 +115,24 @@ export async function startServerLifecycle({ VERSION, token, created, upstream,
|
|
|
115
115
|
}).catch(() => {});
|
|
116
116
|
}, 100).unref?.();
|
|
117
117
|
|
|
118
|
+
// responses 模型判定改元数据驱动(models.dev provider.npm):就绪后注入,每小时重查使新模型自动识别
|
|
119
|
+
void (async () => {
|
|
120
|
+
try {
|
|
121
|
+
const { globalCapabilities } = await import("../model-capabilities/index.js");
|
|
122
|
+
const { setResponsesNpmIndex } = await import("../upstream-responses.js");
|
|
123
|
+
const svc = globalCapabilities();
|
|
124
|
+
const apply = () => setResponsesNpmIndex(svc.npmIndex());
|
|
125
|
+
await svc.ready();
|
|
126
|
+
apply();
|
|
127
|
+
const entry = { ts: Date.now(), type: "responses-npm-index", models: svc.npmIndex().size };
|
|
128
|
+
try { bus.emit(entry); } catch {}
|
|
129
|
+
try { logs.appendEvent(entry); } catch {}
|
|
130
|
+
setInterval(() => { svc.ready().then(apply).catch(() => {}); }, 60 * 60 * 1000).unref?.();
|
|
131
|
+
} catch (e) {
|
|
132
|
+
try { logs.appendEvent({ ts: Date.now(), type: "responses-npm-index-failed", error: String(e?.message || e).slice(0, 200) }); } catch {}
|
|
133
|
+
}
|
|
134
|
+
})();
|
|
135
|
+
|
|
118
136
|
models.startAutoRefresh();
|
|
119
137
|
if (process.env.MSLXDFF_DAEMON) {
|
|
120
138
|
writePid(process.pid, VERSION);
|
|
@@ -57,9 +57,9 @@ export function errorResponseFromSdkError(e, { marker = null } = {}) {
|
|
|
57
57
|
});
|
|
58
58
|
}
|
|
59
59
|
|
|
60
|
-
// SDK parts → OpenAI SSE Response(chat/responses
|
|
61
|
-
export function streamResponseFromParts(parts, { marker = null, clock = Date.now, t0 = clock() } = {}) {
|
|
62
|
-
const ser = createSseSerializer();
|
|
60
|
+
// SDK parts → OpenAI SSE Response(chat/responses 共用同一序列化器;captured 为 responses 侧信道)。
|
|
61
|
+
export function streamResponseFromParts(parts, { marker = null, clock = Date.now, t0 = clock(), captured = null } = {}) {
|
|
62
|
+
const ser = createSseSerializer(captured);
|
|
63
63
|
const enc = new TextEncoder();
|
|
64
64
|
let cancelled = false;
|
|
65
65
|
const stream = new ReadableStream({
|
|
@@ -67,6 +67,11 @@ export function streamResponseFromParts(parts, { marker = null, clock = Date.now
|
|
|
67
67
|
try {
|
|
68
68
|
for await (const part of parts) {
|
|
69
69
|
if (cancelled) break;
|
|
70
|
+
// 侧信道(captured)可能落后几毫秒:收尾前给 encrypted 一点就绪时间(最多 150ms)
|
|
71
|
+
if (captured && !captured.reasoning && part?.type === "finish") {
|
|
72
|
+
const tw = Date.now();
|
|
73
|
+
while (!captured.reasoning && Date.now() - tw < 150) await new Promise((r) => setTimeout(r, 10));
|
|
74
|
+
}
|
|
70
75
|
const text = ser.push(part);
|
|
71
76
|
if (text) controller.enqueue(enc.encode(text));
|
|
72
77
|
}
|
|
@@ -52,7 +52,21 @@ export function toModelPrompt(messages) {
|
|
|
52
52
|
const parts = [];
|
|
53
53
|
const text = textOf(m.content);
|
|
54
54
|
if (text) parts.push({ type: "text", text });
|
|
55
|
-
|
|
55
|
+
// 带加密态的 reasoning items(responses 通道思考跨轮):AI SDK 会转成上游要的 encrypted reasoning item。
|
|
56
|
+
// 无加密态时才退回纯文本 reasoning(chat 通道的 reasoning_content)。
|
|
57
|
+
const items = Array.isArray(m.reasoning_items) ? m.reasoning_items : [];
|
|
58
|
+
let pushedEncrypted = false;
|
|
59
|
+
for (const r of items) {
|
|
60
|
+
if (!r || typeof r !== "object") continue;
|
|
61
|
+
const summaryText = Array.isArray(r.summary) ? r.summary.map((s) => s?.text || "").join("\n") : "";
|
|
62
|
+
parts.push({
|
|
63
|
+
type: "reasoning",
|
|
64
|
+
text: summaryText || " ",
|
|
65
|
+
providerOptions: { openai: { itemId: r.id, reasoningEncryptedContent: r.encrypted_content } },
|
|
66
|
+
});
|
|
67
|
+
pushedEncrypted = true;
|
|
68
|
+
}
|
|
69
|
+
if (!pushedEncrypted && m.reasoning_content) parts.push({ type: "reasoning", text: String(m.reasoning_content) });
|
|
56
70
|
for (const tc of Array.isArray(m.tool_calls) ? m.tool_calls : []) {
|
|
57
71
|
if (!tc || typeof tc !== "object") continue;
|
|
58
72
|
parts.push({
|
|
@@ -8,6 +8,50 @@ import { ENGINE_MARKER } from "./chat.js";
|
|
|
8
8
|
|
|
9
9
|
export const RESPONSES_CHAT_PATH = "/zen/v1/responses";
|
|
10
10
|
|
|
11
|
+
// 上游 encrypted reasoning 只在 output_item.done 里给,而 AI SDK 仅在有 summary 文本时才透出到 parts
|
|
12
|
+
// (muse-spark 这类无 summary 的思考模型会被吞掉)。这里在 fetch 层 tee 一份原始 SSE 自行解析,
|
|
13
|
+
// 侧信道把加密思考交给序列化器,收尾帧补发。
|
|
14
|
+
function captureReasoningFetch(baseFetch, sink) {
|
|
15
|
+
const dbg = process.env.MSLXDFF_RESPONSES_DEBUG === "1";
|
|
16
|
+
return async (input, init) => {
|
|
17
|
+
const res = await baseFetch(input, init);
|
|
18
|
+
const ct = res.headers.get("content-type") || "";
|
|
19
|
+
if (dbg) console.log(`[capture] enter ct=${ct} hasBody=${Boolean(res.body)}`);
|
|
20
|
+
if (!ct.includes("text/event-stream") || !res.body) return res;
|
|
21
|
+
const [passthrough, probe] = res.body.tee();
|
|
22
|
+
(async () => {
|
|
23
|
+
const reader = probe.getReader();
|
|
24
|
+
const dec = new TextDecoder();
|
|
25
|
+
let buf = "";
|
|
26
|
+
try {
|
|
27
|
+
for (;;) {
|
|
28
|
+
const { done, value } = await reader.read();
|
|
29
|
+
if (done) break;
|
|
30
|
+
buf += dec.decode(value, { stream: true });
|
|
31
|
+
let nl;
|
|
32
|
+
while ((nl = buf.indexOf("\n")) >= 0) {
|
|
33
|
+
const line = buf.slice(0, nl).trim();
|
|
34
|
+
buf = buf.slice(nl + 1);
|
|
35
|
+
if (!line.startsWith("data:")) continue;
|
|
36
|
+
const d = line.slice(5).trim();
|
|
37
|
+
if (!d || d === "[DONE]") continue;
|
|
38
|
+
try {
|
|
39
|
+
const j = JSON.parse(d);
|
|
40
|
+
if (dbg && j.type === "response.output_item.done") console.log(`[capture] item.done type=${j.item?.type} enc=${typeof j.item?.encrypted_content}`);
|
|
41
|
+
if (j.type === "response.output_item.done" && j.item?.type === "reasoning" && typeof j.item.encrypted_content === "string") {
|
|
42
|
+
sink.reasoning = { id: j.item.id, encrypted: j.item.encrypted_content, summary: j.item.summary ?? [] };
|
|
43
|
+
if (dbg) console.log(`[capture] hit id=${String(j.item.id).slice(0, 24)} len=${j.item.encrypted_content.length}`);
|
|
44
|
+
}
|
|
45
|
+
} catch { /* ignore */ }
|
|
46
|
+
}
|
|
47
|
+
}
|
|
48
|
+
} catch { /* ignore */ }
|
|
49
|
+
if (dbg) console.log(`[capture] stream end captured=${sink.reasoning ? "yes" : "no"}`);
|
|
50
|
+
})();
|
|
51
|
+
return new Response(passthrough, { status: res.status, statusText: res.statusText, headers: res.headers });
|
|
52
|
+
};
|
|
53
|
+
}
|
|
54
|
+
|
|
11
55
|
let sdkPromise = null;
|
|
12
56
|
|
|
13
57
|
export function loadOpenAISdk() {
|
|
@@ -37,19 +81,34 @@ export async function attemptOnceResponsesSdk({
|
|
|
37
81
|
throw err;
|
|
38
82
|
}
|
|
39
83
|
// headers 由调用方构造(含 Authorization),headers 优先于 apiKey 默认头
|
|
84
|
+
const captured = { reasoning: null };
|
|
85
|
+
const baseFetch = fetchImpl ?? ((u, i) => globalThis.fetch(u, i));
|
|
40
86
|
const provider = createOpenAI({
|
|
41
87
|
name: providerName,
|
|
42
88
|
baseURL,
|
|
43
89
|
apiKey: "public",
|
|
44
90
|
headers: sanitizeHeaders(headers),
|
|
45
|
-
|
|
91
|
+
fetch: captureReasoningFetch(baseFetch, captured),
|
|
46
92
|
});
|
|
47
93
|
const model = provider.responses(String(body?.model || ""));
|
|
94
|
+
const params = toModelParams(body, providerName);
|
|
95
|
+
// 无状态 + 加密思考回传(thinking 跨轮):上游把思考以加密块发回,客户端持有并每轮带回。
|
|
96
|
+
// AI SDK responses 的 providerOptionsName 对非 azure 硬编码为 "openai"(dist/index.js:5240)。
|
|
97
|
+
const providerOptions = {
|
|
98
|
+
...(params.providerOptions || {}),
|
|
99
|
+
openai: {
|
|
100
|
+
store: false,
|
|
101
|
+
include: ["reasoning.encrypted_content"],
|
|
102
|
+
reasoningSummary: "auto",
|
|
103
|
+
...((params.providerOptions || {}).openai || {}),
|
|
104
|
+
},
|
|
105
|
+
};
|
|
48
106
|
let res;
|
|
49
107
|
try {
|
|
50
108
|
res = await model.doStream({
|
|
51
109
|
prompt: toModelPrompt(body?.messages),
|
|
52
|
-
...
|
|
110
|
+
...params,
|
|
111
|
+
providerOptions,
|
|
53
112
|
tools: toModelTools(body?.tools),
|
|
54
113
|
toolChoice: toModelToolChoice(body?.tool_choice),
|
|
55
114
|
});
|
|
@@ -58,7 +117,7 @@ export async function attemptOnceResponsesSdk({
|
|
|
58
117
|
if (mapped) return mapped;
|
|
59
118
|
throw e;
|
|
60
119
|
}
|
|
61
|
-
return streamResponseFromParts(res.stream, { marker, clock, t0 });
|
|
120
|
+
return streamResponseFromParts(res.stream, { marker, clock, t0, captured });
|
|
62
121
|
}
|
|
63
122
|
|
|
64
123
|
export function createSdkResponses({
|
|
@@ -32,14 +32,18 @@ export function usageToOpenAI(u) {
|
|
|
32
32
|
return out;
|
|
33
33
|
}
|
|
34
34
|
|
|
35
|
-
export function createSseSerializer() {
|
|
35
|
+
export function createSseSerializer(captured = null) {
|
|
36
36
|
const meta = { id: "chatcmpl-wb-sdk", model: "", created: Math.floor(Date.now() / 1000) };
|
|
37
37
|
let roleSent = false;
|
|
38
38
|
let nextIndex = 0;
|
|
39
39
|
const toolIndex = new Map();
|
|
40
40
|
const toolDeltaIds = new Set();
|
|
41
|
+
// 加密思考往返:reasoning-start 的 providerMetadata 带 itemId + 加密态,随首帧透出
|
|
42
|
+
let reasoningMeta = null;
|
|
43
|
+
let reasoningFrameSent = false;
|
|
44
|
+
let reasoningEncSent = false;
|
|
41
45
|
|
|
42
|
-
function frame(delta, { finishReason = null, usage } = {}) {
|
|
46
|
+
function frame(delta, { finishReason = null, usage, extra } = {}) {
|
|
43
47
|
const obj = {
|
|
44
48
|
id: meta.id,
|
|
45
49
|
object: "chat.completion.chunk",
|
|
@@ -48,6 +52,7 @@ export function createSseSerializer() {
|
|
|
48
52
|
choices: [{ index: 0, delta, finish_reason: finishReason }],
|
|
49
53
|
};
|
|
50
54
|
if (usage !== undefined) obj.usage = usage;
|
|
55
|
+
if (extra) Object.assign(obj, extra);
|
|
51
56
|
return `data: ${JSON.stringify(obj)}\n\n`;
|
|
52
57
|
}
|
|
53
58
|
|
|
@@ -69,8 +74,30 @@ export function createSseSerializer() {
|
|
|
69
74
|
}
|
|
70
75
|
return null;
|
|
71
76
|
}
|
|
72
|
-
case "reasoning-
|
|
73
|
-
|
|
77
|
+
case "reasoning-start": {
|
|
78
|
+
const pm = part.providerMetadata?.openai || part.providerMetadata || {};
|
|
79
|
+
reasoningMeta = {
|
|
80
|
+
id: String(pm.itemId ?? part.id ?? "reasoning"),
|
|
81
|
+
encrypted: typeof pm.reasoningEncryptedContent === "string" ? pm.reasoningEncryptedContent : null,
|
|
82
|
+
};
|
|
83
|
+
// 即时透出 item 元数据(含加密态):上游可能只给 encrypted 不给 summary 文本,不能等 delta
|
|
84
|
+
reasoningFrameSent = true;
|
|
85
|
+
if (reasoningMeta.encrypted) reasoningEncSent = true;
|
|
86
|
+
return ensureRole() + frame({ reasoning_content: "" }, { extra: { x_reasoning_item: { id: reasoningMeta.id, encrypted_content: reasoningMeta.encrypted } } });
|
|
87
|
+
}
|
|
88
|
+
case "reasoning-delta": {
|
|
89
|
+
const delta = String(part.delta ?? "");
|
|
90
|
+
let extra;
|
|
91
|
+
if (reasoningMeta && !reasoningFrameSent) {
|
|
92
|
+
reasoningFrameSent = true;
|
|
93
|
+
if (reasoningMeta.encrypted) reasoningEncSent = true;
|
|
94
|
+
// x_reasoning_item:responses translator 据此建 reasoning item(chat 客户端忽略未知顶层字段)
|
|
95
|
+
extra = { x_reasoning_item: { id: reasoningMeta.id, encrypted_content: reasoningMeta.encrypted } };
|
|
96
|
+
} else if (reasoningMeta) {
|
|
97
|
+
extra = { x_reasoning_id: reasoningMeta.id };
|
|
98
|
+
}
|
|
99
|
+
return ensureRole() + frame({ reasoning_content: delta }, { extra });
|
|
100
|
+
}
|
|
74
101
|
case "text-delta":
|
|
75
102
|
return ensureRole() + frame({ content: String(part.delta ?? "") });
|
|
76
103
|
case "tool-input-start": {
|
|
@@ -98,7 +125,14 @@ export function createSseSerializer() {
|
|
|
98
125
|
const fr = part.finishReason;
|
|
99
126
|
const raw = typeof fr === "string" ? fr : (fr?.raw ?? fr?.unified);
|
|
100
127
|
const reason = raw === "tool-calls" ? "tool_calls" : raw === "content-filter" ? "content_filter" : (raw ?? "stop");
|
|
101
|
-
|
|
128
|
+
// 加密思考补发:上游 encrypted_content 只在流末尾可得(fetch 侧信道捕获),
|
|
129
|
+
// 无 summary 文本时 AI SDK parts 不会带出 → 收尾帧前补一帧
|
|
130
|
+
let pre = "";
|
|
131
|
+
if (captured?.reasoning && !reasoningEncSent) {
|
|
132
|
+
reasoningEncSent = true;
|
|
133
|
+
pre = ensureRole() + frame({ reasoning_content: "" }, { extra: { x_reasoning_item: { id: captured.reasoning.id, encrypted_content: captured.reasoning.encrypted } } });
|
|
134
|
+
}
|
|
135
|
+
return pre + frame({}, { finishReason: reason, usage: usageToOpenAI(part.usage) });
|
|
102
136
|
}
|
|
103
137
|
case "error": {
|
|
104
138
|
const message = part.error?.message
|
|
@@ -2,8 +2,26 @@
|
|
|
2
2
|
* responses 转换层 — 从 upstream.js 抽出的 muse-spark 专用形状转换。
|
|
3
3
|
* chat ⇄ responses 互转纯函数,无网络、无副作用。
|
|
4
4
|
*/
|
|
5
|
+
// responses 判定索引(models.dev 模型级 provider.npm,启动时注入):
|
|
6
|
+
// "@ai-sdk/openai" → responses 端点;未注入/未命中 → 前缀兜底(新模型早于 models.dev 刷新时仍可用)。
|
|
7
|
+
let responsesNpmIndex = null;
|
|
8
|
+
|
|
9
|
+
export function setResponsesNpmIndex(idx) {
|
|
10
|
+
responsesNpmIndex = idx instanceof Map ? idx : null;
|
|
11
|
+
}
|
|
12
|
+
|
|
13
|
+
export function _resetResponsesNpmIndex() {
|
|
14
|
+
responsesNpmIndex = null;
|
|
15
|
+
}
|
|
16
|
+
|
|
5
17
|
export function isResponsesModel(model) {
|
|
6
|
-
|
|
18
|
+
const m = String(model || "").toLowerCase().trim();
|
|
19
|
+
if (!m) return false;
|
|
20
|
+
if (responsesNpmIndex) {
|
|
21
|
+
const bare = m.replace(/^opencode\//, "");
|
|
22
|
+
if (responsesNpmIndex.has(bare)) return responsesNpmIndex.get(bare) === "@ai-sdk/openai";
|
|
23
|
+
}
|
|
24
|
+
return m.startsWith("muse-spark");
|
|
7
25
|
}
|
|
8
26
|
|
|
9
27
|
export function chatToResponsesBody(chatBody) {
|
package/src/upstream.js
CHANGED
|
@@ -12,6 +12,24 @@ import { uuid } from "./compat.js";
|
|
|
12
12
|
function genId(prefix) {
|
|
13
13
|
return `${prefix}${uuid().replace(/-/g, "")}`;
|
|
14
14
|
}
|
|
15
|
+
// 上游按 session 做粘性路由(实测:固定 session 两次请求均 ~1.2s;每次随机时可能撞冷机器 26s+)。
|
|
16
|
+
// 客户端(opencode AI SDK 路径)不带会话标识 → 用对话首两条消息(system + 首条 user)哈希做稳定会话:
|
|
17
|
+
// 同一会话多轮里这两条不变 ⇒ 路由亲和稳定;不同会话天然分散。
|
|
18
|
+
function sessionFromMessages(messages) {
|
|
19
|
+
try {
|
|
20
|
+
const msgs = Array.isArray(messages) ? messages : [];
|
|
21
|
+
const pick = (role) => {
|
|
22
|
+
const m = msgs.find((x) => x?.role === role);
|
|
23
|
+
if (!m) return "";
|
|
24
|
+
return typeof m.content === "string" ? m.content : JSON.stringify(m.content ?? "");
|
|
25
|
+
};
|
|
26
|
+
const seed = `${pick("system")}|${pick("user")}`.slice(0, 4000);
|
|
27
|
+
if (seed === "|") return null;
|
|
28
|
+
return `ses_${crypto.createHash("sha1").update(seed).digest("hex").slice(0, 32)}`;
|
|
29
|
+
} catch {
|
|
30
|
+
return null;
|
|
31
|
+
}
|
|
32
|
+
}
|
|
15
33
|
function envInt(name, fallback) {
|
|
16
34
|
const v = Number(process.env[name]);
|
|
17
35
|
return Number.isInteger(v) && v > 0 ? v : fallback;
|
|
@@ -47,13 +65,16 @@ export function createOpencodeHeaderBuilder({ authToken = "public", env = proces
|
|
|
47
65
|
Authorization: `Bearer ${authToken}`,
|
|
48
66
|
"x-opencode-client": "desktop",
|
|
49
67
|
};
|
|
50
|
-
|
|
68
|
+
// 无 messages 可哈希时的进程级兜底:至少 daemon 生命周期内稳定,不再每请求随机
|
|
69
|
+
const FALLBACK_SESSION = genId("ses_");
|
|
70
|
+
function buildHeaders(body, { anonymous = anonFirst, sessionId = null } = {}) {
|
|
51
71
|
const isStream = body?.stream !== false;
|
|
72
|
+
const session = sessionId || sessionFromMessages(body?.messages) || FALLBACK_SESSION;
|
|
52
73
|
const base = {
|
|
53
74
|
...baseHeaders,
|
|
54
75
|
Accept: isStream ? "text/event-stream" : "*/*",
|
|
55
76
|
"User-Agent": "opencode",
|
|
56
|
-
"x-opencode-session":
|
|
77
|
+
"x-opencode-session": session,
|
|
57
78
|
"x-opencode-request": genId("msg_"),
|
|
58
79
|
"x-opencode-project": "global",
|
|
59
80
|
};
|
|
@@ -111,8 +132,10 @@ export function createUpstreamClient({
|
|
|
111
132
|
hooks,
|
|
112
133
|
});
|
|
113
134
|
|
|
114
|
-
async function chat(body) {
|
|
135
|
+
async function chat(body, opts = {}) {
|
|
115
136
|
const isResp = isResponsesModel(body?.model);
|
|
137
|
+
// responses 路径的 reqBody 无 messages,会话哈希必须基于原始 chat body 计算
|
|
138
|
+
const sessionId = opts?.sessionId || sessionFromMessages(body?.messages) || null;
|
|
116
139
|
const url = isResp ? `${baseUrl}/zen/v1/responses` : `${baseUrl}/zen/v1/chat/completions`;
|
|
117
140
|
const reqBody = isResp ? chatToResponsesBody(body) : body;
|
|
118
141
|
const t0 = performance.now();
|
|
@@ -123,7 +146,7 @@ export function createUpstreamClient({
|
|
|
123
146
|
res = await transport.request({
|
|
124
147
|
url,
|
|
125
148
|
method: "POST",
|
|
126
|
-
headers: buildHeaders(reqBody),
|
|
149
|
+
headers: buildHeaders(reqBody, { sessionId }),
|
|
127
150
|
body: reqBody,
|
|
128
151
|
stream: body?.stream !== false,
|
|
129
152
|
timeoutMs: connectTimeoutMs,
|
|
@@ -145,7 +168,7 @@ export function createUpstreamClient({
|
|
|
145
168
|
anonRes = await transport.request({
|
|
146
169
|
url,
|
|
147
170
|
method: "POST",
|
|
148
|
-
headers: buildHeaders(reqBody, { anonymous: true }),
|
|
171
|
+
headers: buildHeaders(reqBody, { anonymous: true, sessionId }),
|
|
149
172
|
body: reqBody,
|
|
150
173
|
stream: body?.stream !== false,
|
|
151
174
|
timeoutMs: connectTimeoutMs,
|