mslxdff 0.1.114 → 0.1.116

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "mslxdff",
3
- "version": "0.1.114",
3
+ "version": "0.1.116",
4
4
  "description": "测试项目,请勿使用。",
5
5
  "type": "module",
6
6
  "bin": {
@@ -65,6 +65,7 @@ export async function runAutoRace(ctx, deps = {}) {
65
65
  const o = {};
66
66
  if (Object.keys(shareKeys).length) o.shareKeys = shareKeys;
67
67
  if (workbuddyUid) o.workbuddyUid = workbuddyUid;
68
+ if (handlerCtx?.sessionId) o.sessionId = handlerCtx.sessionId;
68
69
  r = await upstream.chat(f, Object.keys(o).length ? o : undefined);
69
70
  } catch (e) {
70
71
  if (plugins?.length) runHook(plugins, "upstream:response", { reqId, requested, model: m, status: null, ok: false, error: errMsg(e), timing: e?._t ?? null }).catch(() => {});
@@ -82,7 +82,11 @@ export function createChatPipeline({ upstream, auto, logs, peers, groups, bus, t
82
82
  for (const e of sel.errors) evt("plugin-hook-error", { reqId, hook: "model:select", plugin: e.plugin, error: e.error });
83
83
  }
84
84
 
85
- const handlerCtx = { reqId, model: null, body: req?.body, hops, peers, plugins, evt, logError, logCall, logs, workbuddyUid };
85
+ // 客户端会话标识(opencode 插件 chat.headers 注入)→ 透传上游做粘性路由/缓存亲和;
86
+ // 无头时由 upstream.js 按对话首两条消息哈希兜底(不再每请求随机)
87
+ const clientSession = String(req?.headers?.["x-session-affinity"] || req?.headers?.["x-session-id"] || "").trim() || null;
88
+ const handlerCtx = { reqId, model: null, body: req?.body, hops, peers, plugins, evt, logError, logCall, logs, workbuddyUid, sessionId: clientSession };
89
+ if (clientSession) evt("client-session", { reqId, sessionId: clientSession.slice(0, 24) });
86
90
 
87
91
  const plan = planRoute(policy, {
88
92
  candidates: order,
@@ -70,6 +70,7 @@ export async function runSerialTrial(ctx, deps = {}) {
70
70
  const chatOpts = {};
71
71
  if (Object.keys(shareKeys).length) chatOpts.shareKeys = shareKeys;
72
72
  if (workbuddyUid) chatOpts.workbuddyUid = workbuddyUid;
73
+ if (handlerCtx?.sessionId) chatOpts.sessionId = handlerCtx.sessionId;
73
74
  upRes = await upstream.chat(forwarded, Object.keys(chatOpts).length ? chatOpts : undefined);
74
75
  evt("upstream-done", { reqId, model, ok: !(upRes instanceof Error) && upRes.status < 400, status: upRes instanceof Error ? null : upRes.status, timing: upRes._t ?? null, error: null });
75
76
  } catch (err) {
@@ -28,15 +28,22 @@ export function createCapabilitiesService({
28
28
  } = {}) {
29
29
  let raw = null; // 原始目录(全 provider)
30
30
  let capsIndex = new Map(); // providerId -> { modelId -> caps }
31
+ let npmIndex = new Map(); // opencode 裸 modelId -> provider.npm(responses 判定用;null = 继承默认)
31
32
  let loadedAt = 0;
32
33
  let inflight = null;
33
34
 
34
35
  function buildIndex(data) {
35
36
  const idx = new Map();
37
+ const npm = new Map();
36
38
  for (const [pid, p] of Object.entries(data || {})) {
37
39
  if (!p || typeof p !== "object") continue;
38
- idx.set(pid, normalizeProviderModels(p.models || {}));
40
+ const caps = normalizeProviderModels(p.models || {});
41
+ idx.set(pid, caps);
42
+ if (String(pid).toLowerCase() === "opencode") {
43
+ for (const [mid, c] of Object.entries(caps)) npm.set(mid, c.npm ?? null);
44
+ }
39
45
  }
46
+ npmIndex = npm;
40
47
  return idx;
41
48
  }
42
49
 
@@ -111,7 +118,7 @@ export function createCapabilitiesService({
111
118
  return [...capsIndex.keys()].sort();
112
119
  }
113
120
 
114
- return { ready, get, list, providers };
121
+ return { ready, get, list, providers, npmIndex: () => new Map(npmIndex) };
115
122
  }
116
123
 
117
124
  // 模块级单例(与 globalDedup 同模式):HTTP handler 懒加载,测试 _reset 后注入
@@ -24,6 +24,8 @@ export function normalizeModelCaps(_id, m) {
24
24
  releaseDate: typeof m?.release_date === "string" && m.release_date ? m.release_date : null,
25
25
  inputModalities: input.length ? input : ["text"],
26
26
  outputModalities: Array.isArray(m?.modalities?.output) && m.modalities.output.length ? m.modalities.output : ["text"],
27
+ // 模型级 SDK 覆盖(models.dev provider.npm):@ai-sdk/openai → responses 端点;null = 继承 provider 默认
28
+ npm: typeof m?.provider?.npm === "string" && m.provider.npm ? m.provider.npm : null,
27
29
  };
28
30
  }
29
31
 
@@ -53,9 +53,9 @@ export function createProviderDispatcher(providers = [], opts = {}) {
53
53
  return provider.chatWithKeys(forwarded, sharedKeys);
54
54
  }
55
55
  if (provider.id === "workbuddy" && workbuddyUid) {
56
- return provider.chat(forwarded, { workbuddyUid });
56
+ return provider.chat(forwarded, { ...opts, workbuddyUid });
57
57
  }
58
- return provider.chat(forwarded);
58
+ return provider.chat(forwarded, opts);
59
59
  }
60
60
 
61
61
  // 聚合所有供应商的模型列表;默认供应商(opencode)裸 id,其它带前缀
@@ -8,7 +8,7 @@ export function createOpenCodeProvider({ upstream, modelsService, baseUrl, authT
8
8
  return {
9
9
  id: "opencode",
10
10
  upstream: client,
11
- chat: (body) => client.chat(body),
11
+ chat: (body, opts) => client.chat(body, opts),
12
12
  preheat: (args) => client.preheat(args),
13
13
  close: () => client.close(),
14
14
  async listModels() {
@@ -33,15 +33,28 @@ export function responsesToChatBody(req = {}) {
33
33
  if (req.instructions) messages.push({ role: "system", content: String(req.instructions) });
34
34
  const input = req.input;
35
35
  const items = typeof input === "string" ? [{ type: "message", role: "user", content: input }] : Array.isArray(input) ? input : [];
36
+ // 加密思考往返:input 里的 reasoning item 挂到下一条 assistant 消息(thinking 跨轮必需,不能丢)
37
+ let pendingReasoning = [];
36
38
  for (const it of items) {
37
39
  if (!it || typeof it !== "object") continue;
38
- if (it.type === "message") {
39
- messages.push({ role: it.role || "user", content: inputTextOf(it.content) });
40
+ if (it.type === "reasoning") {
41
+ if (it.id || it.encrypted_content) {
42
+ pendingReasoning.push({ id: it.id, encrypted_content: it.encrypted_content, summary: it.summary });
43
+ }
44
+ continue;
45
+ }
46
+ // responses 规范:message item 的 type 可省(AI SDK/opencode 就不发)→ 有 role 即按 message 处理
47
+ if (it.type === "message" || (!it.type && it.role)) {
48
+ const msg = { role: it.role || "user", content: inputTextOf(it.content) };
49
+ if (pendingReasoning.length && msg.role === "assistant") { msg.reasoning_items = pendingReasoning; pendingReasoning = []; }
50
+ messages.push(msg);
40
51
  } else if (it.type === "function_call") {
41
- messages.push({
52
+ const msg = {
42
53
  role: "assistant", content: "",
43
54
  tool_calls: [{ id: it.call_id || it.id || "", type: "function", function: { name: it.name || "", arguments: it.arguments || "" } }],
44
- });
55
+ };
56
+ if (pendingReasoning.length) { msg.reasoning_items = pendingReasoning; pendingReasoning = []; }
57
+ messages.push(msg);
45
58
  } else if (it.type === "function_call_output") {
46
59
  messages.push({ role: "tool", tool_call_id: it.call_id || "", content: typeof it.output === "string" ? it.output : JSON.stringify(it.output ?? "") });
47
60
  }
@@ -115,7 +128,16 @@ export function createChunkTranslator(model = "") {
115
128
  let textLen = 0;
116
129
  let lastFinish = "stop";
117
130
  let lastUsage = null;
118
- const tools = new Map(); // index → {id, name, args, announced}
131
+ const tools = new Map(); // index → {id, name, args, announced, outputIndex}
132
+ // output_index 按 item 出现顺序动态分配(reasoning 可能先于 message/tool 出现)
133
+ let nextIdx = 0;
134
+ let textIdx = null;
135
+ // 加密思考(thinking 跨轮):sse.js 首帧带 x_reasoning_item(含加密态),translator 据此建 reasoning item
136
+ let reasoningOpen = false;
137
+ let reasoningId = null;
138
+ let reasoningEncrypted = null;
139
+ let reasoningIdx = null;
140
+ let reasoningText = "";
119
141
  const ev = (type, extra = {}) => ({ type, ...extra });
120
142
 
121
143
  function begin() {
@@ -125,8 +147,9 @@ export function createChunkTranslator(model = "") {
125
147
  function ensureTextItem() {
126
148
  if (textItemOpen) return [];
127
149
  textItemOpen = true;
150
+ textIdx = nextIdx++;
128
151
  const item = { type: "message", id: `msg_${id}`, status: "in_progress", role: "assistant", content: [{ type: "output_text", text: "", annotations: [] }] };
129
- return [ev("response.output_item.added", { output_index: 0, item }), ev("response.content_part.added", { item_id: item.id, output_index: 0, content_index: 0, part: { type: "output_text", text: "", annotations: [] } })];
152
+ return [ev("response.output_item.added", { output_index: textIdx, item }), ev("response.content_part.added", { item_id: item.id, output_index: textIdx, content_index: 0, part: { type: "output_text", text: "", annotations: [] } })];
130
153
  }
131
154
 
132
155
  let fullText = "";
@@ -135,26 +158,51 @@ export function createChunkTranslator(model = "") {
135
158
  const out = ensureTextItem();
136
159
  textLen += delta.length;
137
160
  fullText += delta;
138
- out.push(ev("response.output_text.delta", { item_id: `msg_${id}`, output_index: 0, content_index: 0, delta }));
161
+ out.push(ev("response.output_text.delta", { item_id: `msg_${id}`, output_index: textIdx, content_index: 0, delta }));
162
+ return out;
163
+ }
164
+
165
+ function pushReasoning(delta, meta) {
166
+ const out = ensureReasoningItem(meta);
167
+ reasoningText += delta;
168
+ if (delta) out.push(ev("response.reasoning_summary_text.delta", { item_id: reasoningId, output_index: reasoningIdx, summary_index: 0, delta }));
139
169
  return out;
140
170
  }
141
171
 
172
+ function ensureReasoningItem(meta) {
173
+ if (reasoningOpen) {
174
+ // 补丁:加密态可能晚到(sse.js 在收尾帧补发无文本的 x_reasoning_item)
175
+ if (!reasoningEncrypted && typeof meta?.encrypted_content === "string") reasoningEncrypted = meta.encrypted_content;
176
+ return [];
177
+ }
178
+ reasoningOpen = true;
179
+ reasoningId = meta?.id || `rs_${id}`;
180
+ reasoningEncrypted = typeof meta?.encrypted_content === "string" ? meta.encrypted_content : null;
181
+ reasoningIdx = nextIdx++;
182
+ const item = { type: "reasoning", id: reasoningId, summary: [], ...(reasoningEncrypted ? { encrypted_content: reasoningEncrypted } : {}) };
183
+ return [
184
+ ev("response.output_item.added", { output_index: reasoningIdx, item }),
185
+ ev("response.reasoning_summary_part.added", { item_id: reasoningId, output_index: reasoningIdx, summary_index: 0, part: { type: "summary_text", text: "" } }),
186
+ ];
187
+ }
188
+
142
189
  function pushTool(tc) {
143
190
  const idx = Number(tc.index ?? 0);
144
191
  let t = tools.get(idx);
145
- if (!t) { t = { id: "", name: "", args: "", announced: false }; tools.set(idx, t); }
192
+ if (!t) { t = { id: "", name: "", args: "", announced: false, outputIndex: null }; tools.set(idx, t); }
146
193
  if (tc.id) t.id = tc.id;
147
194
  if (tc.function?.name) t.name += tc.function.name;
148
195
  const frag = tc.function?.arguments || "";
149
196
  const out = [];
150
197
  if (!t.announced && t.id && t.name) {
151
198
  t.announced = true;
152
- out.push(ev("response.output_item.added", { output_index: idx + 1, item: { type: "function_call", id: `fc_${id}_${idx}`, call_id: t.id, name: t.name, arguments: "" } }));
199
+ t.outputIndex = nextIdx++;
200
+ out.push(ev("response.output_item.added", { output_index: t.outputIndex, item: { type: "function_call", id: `fc_${id}_${idx}`, call_id: t.id, name: t.name, arguments: "" } }));
153
201
  }
154
202
  if (frag) {
155
203
  if (!t.announced) { t.args += frag; return out; } // id/name 未到先攒着
156
204
  t.args += frag;
157
- out.push(ev("response.function_call_arguments.delta", { item_id: `fc_${id}_${idx}`, output_index: idx + 1, delta: frag }));
205
+ out.push(ev("response.function_call_arguments.delta", { item_id: `fc_${id}_${idx}`, output_index: t.outputIndex, delta: frag }));
158
206
  }
159
207
  return out;
160
208
  }
@@ -181,29 +229,34 @@ export function createChunkTranslator(model = "") {
181
229
  if (typeof delta.content === "string" && delta.content) { dbg.textChars += delta.content.length; out.push(...pushText(delta.content)); }
182
230
  for (const tc of delta.tool_calls || []) { dbg.toolDeltas++; out.push(...pushTool(tc)); }
183
231
  const rc = typeof delta.reasoning_content === "string" ? delta.reasoning_content : "";
184
- if (rc) { dbg.reasoningChars += rc.length; out.push(...pushText(rc)); }
232
+ if (rc || chunk.x_reasoning_item) { if (rc) dbg.reasoningChars += rc.length; out.push(...pushReasoning(rc, chunk.x_reasoning_item)); }
185
233
  }
186
234
  return out;
187
235
  }
188
236
 
189
237
  function end({ finish = "stop", usage = null } = {}) {
190
238
  const out = [];
239
+ if (reasoningOpen) {
240
+ out.push(ev("response.reasoning_summary_text.done", { item_id: reasoningId, output_index: reasoningIdx, summary_index: 0, text: reasoningText }));
241
+ out.push(ev("response.reasoning_summary_part.done", { item_id: reasoningId, output_index: reasoningIdx, summary_index: 0, part: { type: "summary_text", text: reasoningText } }));
242
+ out.push(ev("response.output_item.done", { output_index: reasoningIdx, item: { type: "reasoning", id: reasoningId, summary: [{ type: "summary_text", text: reasoningText }], ...(reasoningEncrypted ? { encrypted_content: reasoningEncrypted } : {}) } }));
243
+ }
191
244
  if (textItemOpen) {
192
- out.push(ev("response.output_text.done", { item_id: `msg_${id}`, output_index: 0, content_index: 0, text: fullText }));
193
- out.push(ev("response.content_part.done", { item_id: `msg_${id}`, output_index: 0, content_index: 0, part: { type: "output_text", text: fullText, annotations: [] } }));
194
- out.push(ev("response.output_item.done", { output_index: 0, item: { type: "message", id: `msg_${id}`, status: "completed", role: "assistant", content: [{ type: "output_text", text: fullText, annotations: [] }] } }));
245
+ out.push(ev("response.output_text.done", { item_id: `msg_${id}`, output_index: textIdx, content_index: 0, text: fullText }));
246
+ out.push(ev("response.content_part.done", { item_id: `msg_${id}`, output_index: textIdx, content_index: 0, part: { type: "output_text", text: fullText, annotations: [] } }));
247
+ out.push(ev("response.output_item.done", { output_index: textIdx, item: { type: "message", id: `msg_${id}`, status: "completed", role: "assistant", content: [{ type: "output_text", text: fullText, annotations: [] }] } }));
195
248
  }
196
- let i = 0;
197
249
  for (const [idx, t] of tools) {
198
- if (!t.id && !t.name && !t.args) { i++; continue; }
250
+ if (!t.id && !t.name && !t.args) continue;
199
251
  if (!t.announced) {
200
- out.push(ev("response.output_item.added", { output_index: idx + 1, item: { type: "function_call", id: `fc_${id}_${idx}`, call_id: t.id, name: t.name, arguments: "" } }));
252
+ t.announced = true;
253
+ t.outputIndex = nextIdx++;
254
+ out.push(ev("response.output_item.added", { output_index: t.outputIndex, item: { type: "function_call", id: `fc_${id}_${idx}`, call_id: t.id, name: t.name, arguments: "" } }));
201
255
  }
202
- if (t.args) out.push(ev("response.function_call_arguments.done", { item_id: `fc_${id}_${idx}`, output_index: idx + 1, arguments: t.args }));
203
- out.push(ev("response.output_item.done", { output_index: idx + 1, item: { type: "function_call", id: `fc_${id}_${idx}`, call_id: t.id, name: t.name, arguments: t.args, status: "completed" } }));
204
- i++;
256
+ if (t.args) out.push(ev("response.function_call_arguments.done", { item_id: `fc_${id}_${idx}`, output_index: t.outputIndex, arguments: t.args }));
257
+ out.push(ev("response.output_item.done", { output_index: t.outputIndex, item: { type: "function_call", id: `fc_${id}_${idx}`, call_id: t.id, name: t.name, arguments: t.args, status: "completed" } }));
205
258
  }
206
- void i; void textLen;
259
+ void textLen;
207
260
  out.push(ev("response.completed", { response: { id, object: "response", created_at: createdAt, status: finish === "length" ? "incomplete" : "completed", model, output: [], usage: toResponsesUsage(usage) } }));
208
261
  return out;
209
262
  }
@@ -112,6 +112,15 @@ export async function responsesHandler(ctx) {
112
112
  inputKinds: Array.isArray(body?.input) ? [...new Set(body.input.map((i) => i?.type))] : null,
113
113
  tools: Array.isArray(body?.tools) ? body.tools.length : 0,
114
114
  instructionsLen: String(body?.instructions || "").length,
115
+ // 文本丢失定位:各 message item 的 content 形状(part 类型 + 文本长度 + 头部)
116
+ contentShapes: Array.isArray(body?.input)
117
+ ? body.input.filter((i) => i && (i.type === "message" || i.role)).map((m) => ({
118
+ type: m.type ?? "(无type)", role: m.role,
119
+ cType: Array.isArray(m.content) ? "array" : typeof m.content,
120
+ parts: Array.isArray(m.content) ? m.content.map((p) => `${p?.type ?? "?"}:${typeof p?.text === "string" ? p.text.length : "-"}`).slice(0, 6) : null,
121
+ textHead: typeof m.content === "string" ? m.content.slice(0, 50) : null,
122
+ })).slice(0, 10)
123
+ : null,
115
124
  }));
116
125
  let chatBody;
117
126
  try {
@@ -30,6 +30,12 @@ export const STREAM_TIMEOUT_MS = (() => {
30
30
  return Number.isInteger(n) && n >= 0 ? n : 25_000;
31
31
  })();
32
32
 
33
+ // 等首块期间的心跳间隔(SSE 注释帧,标准客户端忽略):上游偶发卡 90s+,避免客户端误判卡死/断连
34
+ export const KEEPALIVE_MS = (() => {
35
+ const n = Number(process.env.MSLXDFF_KEEPALIVE_MS);
36
+ return Number.isInteger(n) && n >= 0 ? n : 10_000;
37
+ })();
38
+
33
39
  export const STALL_TIMEOUT_MS = (() => {
34
40
  const n = Number(process.env.MSLXDFF_STALL_TIMEOUT_MS);
35
41
  return Number.isInteger(n) && n > 0 ? n : 0;
@@ -46,7 +52,7 @@ export const MAX_STREAM_MS = (() => {
46
52
  return Number.isInteger(n) && n > 0 ? n : 0;
47
53
  })();
48
54
 
49
- export async function relay(res, upRes, body, { onFirstChunk, onDownstreamAbort, streamTimeoutMs = STREAM_TIMEOUT_MS, fallback } = {}) {
55
+ export async function relay(res, upRes, body, { onFirstChunk, onDownstreamAbort, streamTimeoutMs = STREAM_TIMEOUT_MS, keepaliveMs = KEEPALIVE_MS, fallback } = {}) {
50
56
  const t0 = performance.now();
51
57
  const contentType = upRes.headers.get("content-type") || "";
52
58
  // 需同时满足:客户端要流 + 上游真的是 SSE;避免 muse-spark 聚合 JSON 被误判为流式,或 workbuddy SSE 被聚合
@@ -110,6 +116,12 @@ export async function relay(res, upRes, body, { onFirstChunk, onDownstreamAbort,
110
116
  let stalled = false;
111
117
  let tooLong = false;
112
118
  let stallTimer = null;
119
+ let pingTimer = keepaliveMs > 0
120
+ ? setInterval(() => {
121
+ if (wroteAny) return;
122
+ try { res.write(": keepalive\n\n"); } catch { /* ignore */ }
123
+ }, keepaliveMs)
124
+ : null;
113
125
  const armStall = () => {
114
126
  if (stallTimer) clearTimeout(stallTimer);
115
127
  stallTimer = STALL_TIMEOUT_MS
@@ -234,6 +246,7 @@ export async function relay(res, upRes, body, { onFirstChunk, onDownstreamAbort,
234
246
  if (firstTimer) clearTimeout(firstTimer);
235
247
  if (maxTimer) clearTimeout(maxTimer);
236
248
  if (stallTimer) clearTimeout(stallTimer);
249
+ if (pingTimer) clearInterval(pingTimer);
237
250
  }
238
251
  if (timedOut && !wroteAny) {
239
252
  res.removeListener("close", onClose);
@@ -115,6 +115,24 @@ export async function startServerLifecycle({ VERSION, token, created, upstream,
115
115
  }).catch(() => {});
116
116
  }, 100).unref?.();
117
117
 
118
+ // responses 模型判定改元数据驱动(models.dev provider.npm):就绪后注入,每小时重查使新模型自动识别
119
+ void (async () => {
120
+ try {
121
+ const { globalCapabilities } = await import("../model-capabilities/index.js");
122
+ const { setResponsesNpmIndex } = await import("../upstream-responses.js");
123
+ const svc = globalCapabilities();
124
+ const apply = () => setResponsesNpmIndex(svc.npmIndex());
125
+ await svc.ready();
126
+ apply();
127
+ const entry = { ts: Date.now(), type: "responses-npm-index", models: svc.npmIndex().size };
128
+ try { bus.emit(entry); } catch {}
129
+ try { logs.appendEvent(entry); } catch {}
130
+ setInterval(() => { svc.ready().then(apply).catch(() => {}); }, 60 * 60 * 1000).unref?.();
131
+ } catch (e) {
132
+ try { logs.appendEvent({ ts: Date.now(), type: "responses-npm-index-failed", error: String(e?.message || e).slice(0, 200) }); } catch {}
133
+ }
134
+ })();
135
+
118
136
  models.startAutoRefresh();
119
137
  if (process.env.MSLXDFF_DAEMON) {
120
138
  writePid(process.pid, VERSION);
@@ -57,9 +57,9 @@ export function errorResponseFromSdkError(e, { marker = null } = {}) {
57
57
  });
58
58
  }
59
59
 
60
- // SDK parts → OpenAI SSE Response(chat/responses 共用同一序列化器)。
61
- export function streamResponseFromParts(parts, { marker = null, clock = Date.now, t0 = clock() } = {}) {
62
- const ser = createSseSerializer();
60
+ // SDK parts → OpenAI SSE Response(chat/responses 共用同一序列化器;captured 为 responses 侧信道)。
61
+ export function streamResponseFromParts(parts, { marker = null, clock = Date.now, t0 = clock(), captured = null } = {}) {
62
+ const ser = createSseSerializer(captured);
63
63
  const enc = new TextEncoder();
64
64
  let cancelled = false;
65
65
  const stream = new ReadableStream({
@@ -67,6 +67,11 @@ export function streamResponseFromParts(parts, { marker = null, clock = Date.now
67
67
  try {
68
68
  for await (const part of parts) {
69
69
  if (cancelled) break;
70
+ // 侧信道(captured)可能落后几毫秒:收尾前给 encrypted 一点就绪时间(最多 150ms)
71
+ if (captured && !captured.reasoning && part?.type === "finish") {
72
+ const tw = Date.now();
73
+ while (!captured.reasoning && Date.now() - tw < 150) await new Promise((r) => setTimeout(r, 10));
74
+ }
70
75
  const text = ser.push(part);
71
76
  if (text) controller.enqueue(enc.encode(text));
72
77
  }
@@ -52,7 +52,21 @@ export function toModelPrompt(messages) {
52
52
  const parts = [];
53
53
  const text = textOf(m.content);
54
54
  if (text) parts.push({ type: "text", text });
55
- if (m.reasoning_content) parts.push({ type: "reasoning", text: String(m.reasoning_content) });
55
+ // 带加密态的 reasoning items(responses 通道思考跨轮):AI SDK 会转成上游要的 encrypted reasoning item。
56
+ // 无加密态时才退回纯文本 reasoning(chat 通道的 reasoning_content)。
57
+ const items = Array.isArray(m.reasoning_items) ? m.reasoning_items : [];
58
+ let pushedEncrypted = false;
59
+ for (const r of items) {
60
+ if (!r || typeof r !== "object") continue;
61
+ const summaryText = Array.isArray(r.summary) ? r.summary.map((s) => s?.text || "").join("\n") : "";
62
+ parts.push({
63
+ type: "reasoning",
64
+ text: summaryText || " ",
65
+ providerOptions: { openai: { itemId: r.id, reasoningEncryptedContent: r.encrypted_content } },
66
+ });
67
+ pushedEncrypted = true;
68
+ }
69
+ if (!pushedEncrypted && m.reasoning_content) parts.push({ type: "reasoning", text: String(m.reasoning_content) });
56
70
  for (const tc of Array.isArray(m.tool_calls) ? m.tool_calls : []) {
57
71
  if (!tc || typeof tc !== "object") continue;
58
72
  parts.push({
@@ -8,6 +8,50 @@ import { ENGINE_MARKER } from "./chat.js";
8
8
 
9
9
  export const RESPONSES_CHAT_PATH = "/zen/v1/responses";
10
10
 
11
+ // 上游 encrypted reasoning 只在 output_item.done 里给,而 AI SDK 仅在有 summary 文本时才透出到 parts
12
+ // (muse-spark 这类无 summary 的思考模型会被吞掉)。这里在 fetch 层 tee 一份原始 SSE 自行解析,
13
+ // 侧信道把加密思考交给序列化器,收尾帧补发。
14
+ function captureReasoningFetch(baseFetch, sink) {
15
+ const dbg = process.env.MSLXDFF_RESPONSES_DEBUG === "1";
16
+ return async (input, init) => {
17
+ const res = await baseFetch(input, init);
18
+ const ct = res.headers.get("content-type") || "";
19
+ if (dbg) console.log(`[capture] enter ct=${ct} hasBody=${Boolean(res.body)}`);
20
+ if (!ct.includes("text/event-stream") || !res.body) return res;
21
+ const [passthrough, probe] = res.body.tee();
22
+ (async () => {
23
+ const reader = probe.getReader();
24
+ const dec = new TextDecoder();
25
+ let buf = "";
26
+ try {
27
+ for (;;) {
28
+ const { done, value } = await reader.read();
29
+ if (done) break;
30
+ buf += dec.decode(value, { stream: true });
31
+ let nl;
32
+ while ((nl = buf.indexOf("\n")) >= 0) {
33
+ const line = buf.slice(0, nl).trim();
34
+ buf = buf.slice(nl + 1);
35
+ if (!line.startsWith("data:")) continue;
36
+ const d = line.slice(5).trim();
37
+ if (!d || d === "[DONE]") continue;
38
+ try {
39
+ const j = JSON.parse(d);
40
+ if (dbg && j.type === "response.output_item.done") console.log(`[capture] item.done type=${j.item?.type} enc=${typeof j.item?.encrypted_content}`);
41
+ if (j.type === "response.output_item.done" && j.item?.type === "reasoning" && typeof j.item.encrypted_content === "string") {
42
+ sink.reasoning = { id: j.item.id, encrypted: j.item.encrypted_content, summary: j.item.summary ?? [] };
43
+ if (dbg) console.log(`[capture] hit id=${String(j.item.id).slice(0, 24)} len=${j.item.encrypted_content.length}`);
44
+ }
45
+ } catch { /* ignore */ }
46
+ }
47
+ }
48
+ } catch { /* ignore */ }
49
+ if (dbg) console.log(`[capture] stream end captured=${sink.reasoning ? "yes" : "no"}`);
50
+ })();
51
+ return new Response(passthrough, { status: res.status, statusText: res.statusText, headers: res.headers });
52
+ };
53
+ }
54
+
11
55
  let sdkPromise = null;
12
56
 
13
57
  export function loadOpenAISdk() {
@@ -37,19 +81,34 @@ export async function attemptOnceResponsesSdk({
37
81
  throw err;
38
82
  }
39
83
  // headers 由调用方构造(含 Authorization),headers 优先于 apiKey 默认头
84
+ const captured = { reasoning: null };
85
+ const baseFetch = fetchImpl ?? ((u, i) => globalThis.fetch(u, i));
40
86
  const provider = createOpenAI({
41
87
  name: providerName,
42
88
  baseURL,
43
89
  apiKey: "public",
44
90
  headers: sanitizeHeaders(headers),
45
- ...(fetchImpl ? { fetch: fetchImpl } : {}),
91
+ fetch: captureReasoningFetch(baseFetch, captured),
46
92
  });
47
93
  const model = provider.responses(String(body?.model || ""));
94
+ const params = toModelParams(body, providerName);
95
+ // 无状态 + 加密思考回传(thinking 跨轮):上游把思考以加密块发回,客户端持有并每轮带回。
96
+ // AI SDK responses 的 providerOptionsName 对非 azure 硬编码为 "openai"(dist/index.js:5240)。
97
+ const providerOptions = {
98
+ ...(params.providerOptions || {}),
99
+ openai: {
100
+ store: false,
101
+ include: ["reasoning.encrypted_content"],
102
+ reasoningSummary: "auto",
103
+ ...((params.providerOptions || {}).openai || {}),
104
+ },
105
+ };
48
106
  let res;
49
107
  try {
50
108
  res = await model.doStream({
51
109
  prompt: toModelPrompt(body?.messages),
52
- ...toModelParams(body, providerName),
110
+ ...params,
111
+ providerOptions,
53
112
  tools: toModelTools(body?.tools),
54
113
  toolChoice: toModelToolChoice(body?.tool_choice),
55
114
  });
@@ -58,7 +117,7 @@ export async function attemptOnceResponsesSdk({
58
117
  if (mapped) return mapped;
59
118
  throw e;
60
119
  }
61
- return streamResponseFromParts(res.stream, { marker, clock, t0 });
120
+ return streamResponseFromParts(res.stream, { marker, clock, t0, captured });
62
121
  }
63
122
 
64
123
  export function createSdkResponses({
@@ -32,14 +32,18 @@ export function usageToOpenAI(u) {
32
32
  return out;
33
33
  }
34
34
 
35
- export function createSseSerializer() {
35
+ export function createSseSerializer(captured = null) {
36
36
  const meta = { id: "chatcmpl-wb-sdk", model: "", created: Math.floor(Date.now() / 1000) };
37
37
  let roleSent = false;
38
38
  let nextIndex = 0;
39
39
  const toolIndex = new Map();
40
40
  const toolDeltaIds = new Set();
41
+ // 加密思考往返:reasoning-start 的 providerMetadata 带 itemId + 加密态,随首帧透出
42
+ let reasoningMeta = null;
43
+ let reasoningFrameSent = false;
44
+ let reasoningEncSent = false;
41
45
 
42
- function frame(delta, { finishReason = null, usage } = {}) {
46
+ function frame(delta, { finishReason = null, usage, extra } = {}) {
43
47
  const obj = {
44
48
  id: meta.id,
45
49
  object: "chat.completion.chunk",
@@ -48,6 +52,7 @@ export function createSseSerializer() {
48
52
  choices: [{ index: 0, delta, finish_reason: finishReason }],
49
53
  };
50
54
  if (usage !== undefined) obj.usage = usage;
55
+ if (extra) Object.assign(obj, extra);
51
56
  return `data: ${JSON.stringify(obj)}\n\n`;
52
57
  }
53
58
 
@@ -69,8 +74,30 @@ export function createSseSerializer() {
69
74
  }
70
75
  return null;
71
76
  }
72
- case "reasoning-delta":
73
- return ensureRole() + frame({ reasoning_content: String(part.delta ?? "") });
77
+ case "reasoning-start": {
78
+ const pm = part.providerMetadata?.openai || part.providerMetadata || {};
79
+ reasoningMeta = {
80
+ id: String(pm.itemId ?? part.id ?? "reasoning"),
81
+ encrypted: typeof pm.reasoningEncryptedContent === "string" ? pm.reasoningEncryptedContent : null,
82
+ };
83
+ // 即时透出 item 元数据(含加密态):上游可能只给 encrypted 不给 summary 文本,不能等 delta
84
+ reasoningFrameSent = true;
85
+ if (reasoningMeta.encrypted) reasoningEncSent = true;
86
+ return ensureRole() + frame({ reasoning_content: "" }, { extra: { x_reasoning_item: { id: reasoningMeta.id, encrypted_content: reasoningMeta.encrypted } } });
87
+ }
88
+ case "reasoning-delta": {
89
+ const delta = String(part.delta ?? "");
90
+ let extra;
91
+ if (reasoningMeta && !reasoningFrameSent) {
92
+ reasoningFrameSent = true;
93
+ if (reasoningMeta.encrypted) reasoningEncSent = true;
94
+ // x_reasoning_item:responses translator 据此建 reasoning item(chat 客户端忽略未知顶层字段)
95
+ extra = { x_reasoning_item: { id: reasoningMeta.id, encrypted_content: reasoningMeta.encrypted } };
96
+ } else if (reasoningMeta) {
97
+ extra = { x_reasoning_id: reasoningMeta.id };
98
+ }
99
+ return ensureRole() + frame({ reasoning_content: delta }, { extra });
100
+ }
74
101
  case "text-delta":
75
102
  return ensureRole() + frame({ content: String(part.delta ?? "") });
76
103
  case "tool-input-start": {
@@ -98,7 +125,14 @@ export function createSseSerializer() {
98
125
  const fr = part.finishReason;
99
126
  const raw = typeof fr === "string" ? fr : (fr?.raw ?? fr?.unified);
100
127
  const reason = raw === "tool-calls" ? "tool_calls" : raw === "content-filter" ? "content_filter" : (raw ?? "stop");
101
- return frame({}, { finishReason: reason, usage: usageToOpenAI(part.usage) });
128
+ // 加密思考补发:上游 encrypted_content 只在流末尾可得(fetch 侧信道捕获),
129
+ // 无 summary 文本时 AI SDK parts 不会带出 → 收尾帧前补一帧
130
+ let pre = "";
131
+ if (captured?.reasoning && !reasoningEncSent) {
132
+ reasoningEncSent = true;
133
+ pre = ensureRole() + frame({ reasoning_content: "" }, { extra: { x_reasoning_item: { id: captured.reasoning.id, encrypted_content: captured.reasoning.encrypted } } });
134
+ }
135
+ return pre + frame({}, { finishReason: reason, usage: usageToOpenAI(part.usage) });
102
136
  }
103
137
  case "error": {
104
138
  const message = part.error?.message
@@ -2,8 +2,26 @@
2
2
  * responses 转换层 — 从 upstream.js 抽出的 muse-spark 专用形状转换。
3
3
  * chat ⇄ responses 互转纯函数,无网络、无副作用。
4
4
  */
5
+ // responses 判定索引(models.dev 模型级 provider.npm,启动时注入):
6
+ // "@ai-sdk/openai" → responses 端点;未注入/未命中 → 前缀兜底(新模型早于 models.dev 刷新时仍可用)。
7
+ let responsesNpmIndex = null;
8
+
9
+ export function setResponsesNpmIndex(idx) {
10
+ responsesNpmIndex = idx instanceof Map ? idx : null;
11
+ }
12
+
13
+ export function _resetResponsesNpmIndex() {
14
+ responsesNpmIndex = null;
15
+ }
16
+
5
17
  export function isResponsesModel(model) {
6
- return String(model || "").toLowerCase().startsWith("muse-spark");
18
+ const m = String(model || "").toLowerCase().trim();
19
+ if (!m) return false;
20
+ if (responsesNpmIndex) {
21
+ const bare = m.replace(/^opencode\//, "");
22
+ if (responsesNpmIndex.has(bare)) return responsesNpmIndex.get(bare) === "@ai-sdk/openai";
23
+ }
24
+ return m.startsWith("muse-spark");
7
25
  }
8
26
 
9
27
  export function chatToResponsesBody(chatBody) {
package/src/upstream.js CHANGED
@@ -12,6 +12,24 @@ import { uuid } from "./compat.js";
12
12
  function genId(prefix) {
13
13
  return `${prefix}${uuid().replace(/-/g, "")}`;
14
14
  }
15
+ // 上游按 session 做粘性路由(实测:固定 session 两次请求均 ~1.2s;每次随机时可能撞冷机器 26s+)。
16
+ // 客户端(opencode AI SDK 路径)不带会话标识 → 用对话首两条消息(system + 首条 user)哈希做稳定会话:
17
+ // 同一会话多轮里这两条不变 ⇒ 路由亲和稳定;不同会话天然分散。
18
+ function sessionFromMessages(messages) {
19
+ try {
20
+ const msgs = Array.isArray(messages) ? messages : [];
21
+ const pick = (role) => {
22
+ const m = msgs.find((x) => x?.role === role);
23
+ if (!m) return "";
24
+ return typeof m.content === "string" ? m.content : JSON.stringify(m.content ?? "");
25
+ };
26
+ const seed = `${pick("system")}|${pick("user")}`.slice(0, 4000);
27
+ if (seed === "|") return null;
28
+ return `ses_${crypto.createHash("sha1").update(seed).digest("hex").slice(0, 32)}`;
29
+ } catch {
30
+ return null;
31
+ }
32
+ }
15
33
  function envInt(name, fallback) {
16
34
  const v = Number(process.env[name]);
17
35
  return Number.isInteger(v) && v > 0 ? v : fallback;
@@ -47,13 +65,16 @@ export function createOpencodeHeaderBuilder({ authToken = "public", env = proces
47
65
  Authorization: `Bearer ${authToken}`,
48
66
  "x-opencode-client": "desktop",
49
67
  };
50
- function buildHeaders(body, { anonymous = anonFirst } = {}) {
68
+ // 无 messages 可哈希时的进程级兜底:至少 daemon 生命周期内稳定,不再每请求随机
69
+ const FALLBACK_SESSION = genId("ses_");
70
+ function buildHeaders(body, { anonymous = anonFirst, sessionId = null } = {}) {
51
71
  const isStream = body?.stream !== false;
72
+ const session = sessionId || sessionFromMessages(body?.messages) || FALLBACK_SESSION;
52
73
  const base = {
53
74
  ...baseHeaders,
54
75
  Accept: isStream ? "text/event-stream" : "*/*",
55
76
  "User-Agent": "opencode",
56
- "x-opencode-session": genId("ses_"),
77
+ "x-opencode-session": session,
57
78
  "x-opencode-request": genId("msg_"),
58
79
  "x-opencode-project": "global",
59
80
  };
@@ -111,8 +132,10 @@ export function createUpstreamClient({
111
132
  hooks,
112
133
  });
113
134
 
114
- async function chat(body) {
135
+ async function chat(body, opts = {}) {
115
136
  const isResp = isResponsesModel(body?.model);
137
+ // responses 路径的 reqBody 无 messages,会话哈希必须基于原始 chat body 计算
138
+ const sessionId = opts?.sessionId || sessionFromMessages(body?.messages) || null;
116
139
  const url = isResp ? `${baseUrl}/zen/v1/responses` : `${baseUrl}/zen/v1/chat/completions`;
117
140
  const reqBody = isResp ? chatToResponsesBody(body) : body;
118
141
  const t0 = performance.now();
@@ -123,7 +146,7 @@ export function createUpstreamClient({
123
146
  res = await transport.request({
124
147
  url,
125
148
  method: "POST",
126
- headers: buildHeaders(reqBody),
149
+ headers: buildHeaders(reqBody, { sessionId }),
127
150
  body: reqBody,
128
151
  stream: body?.stream !== false,
129
152
  timeoutMs: connectTimeoutMs,
@@ -145,7 +168,7 @@ export function createUpstreamClient({
145
168
  anonRes = await transport.request({
146
169
  url,
147
170
  method: "POST",
148
- headers: buildHeaders(reqBody, { anonymous: true }),
171
+ headers: buildHeaders(reqBody, { anonymous: true, sessionId }),
149
172
  body: reqBody,
150
173
  stream: body?.stream !== false,
151
174
  timeoutMs: connectTimeoutMs,