mslxdff 0.1.113 → 0.1.115

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "mslxdff",
3
- "version": "0.1.113",
3
+ "version": "0.1.115",
4
4
  "description": "测试项目,请勿使用。",
5
5
  "type": "module",
6
6
  "bin": {
@@ -65,6 +65,7 @@ export async function runAutoRace(ctx, deps = {}) {
65
65
  const o = {};
66
66
  if (Object.keys(shareKeys).length) o.shareKeys = shareKeys;
67
67
  if (workbuddyUid) o.workbuddyUid = workbuddyUid;
68
+ if (handlerCtx?.sessionId) o.sessionId = handlerCtx.sessionId;
68
69
  r = await upstream.chat(f, Object.keys(o).length ? o : undefined);
69
70
  } catch (e) {
70
71
  if (plugins?.length) runHook(plugins, "upstream:response", { reqId, requested, model: m, status: null, ok: false, error: errMsg(e), timing: e?._t ?? null }).catch(() => {});
@@ -82,7 +82,11 @@ export function createChatPipeline({ upstream, auto, logs, peers, groups, bus, t
82
82
  for (const e of sel.errors) evt("plugin-hook-error", { reqId, hook: "model:select", plugin: e.plugin, error: e.error });
83
83
  }
84
84
 
85
- const handlerCtx = { reqId, model: null, body: req?.body, hops, peers, plugins, evt, logError, logCall, logs, workbuddyUid };
85
+ // 客户端会话标识(opencode 插件 chat.headers 注入)→ 透传上游做粘性路由/缓存亲和;
86
+ // 无头时由 upstream.js 按对话首两条消息哈希兜底(不再每请求随机)
87
+ const clientSession = String(req?.headers?.["x-session-affinity"] || req?.headers?.["x-session-id"] || "").trim() || null;
88
+ const handlerCtx = { reqId, model: null, body: req?.body, hops, peers, plugins, evt, logError, logCall, logs, workbuddyUid, sessionId: clientSession };
89
+ if (clientSession) evt("client-session", { reqId, sessionId: clientSession.slice(0, 24) });
86
90
 
87
91
  const plan = planRoute(policy, {
88
92
  candidates: order,
@@ -49,6 +49,8 @@ export async function runSerialTrial(ctx, deps = {}) {
49
49
  for (let idx = 0; idx < order.length; idx++) {
50
50
  const model = order[idx];
51
51
  handlerCtx.model = model;
52
+ handlerCtx.orderLen = order.length;
53
+ handlerCtx.idx = idx;
52
54
  evt("model-try", { reqId, model, idx, remaining: order.length - idx });
53
55
  if (plugins?.length) {
54
56
  const bt = await runHook(plugins, "model:beforeTry", { reqId, requested, model, idx, hops });
@@ -68,6 +70,7 @@ export async function runSerialTrial(ctx, deps = {}) {
68
70
  const chatOpts = {};
69
71
  if (Object.keys(shareKeys).length) chatOpts.shareKeys = shareKeys;
70
72
  if (workbuddyUid) chatOpts.workbuddyUid = workbuddyUid;
73
+ if (handlerCtx?.sessionId) chatOpts.sessionId = handlerCtx.sessionId;
71
74
  upRes = await upstream.chat(forwarded, Object.keys(chatOpts).length ? chatOpts : undefined);
72
75
  evt("upstream-done", { reqId, model, ok: !(upRes instanceof Error) && upRes.status < 400, status: upRes instanceof Error ? null : upRes.status, timing: upRes._t ?? null, error: null });
73
76
  } catch (err) {
@@ -28,15 +28,22 @@ export function createCapabilitiesService({
28
28
  } = {}) {
29
29
  let raw = null; // 原始目录(全 provider)
30
30
  let capsIndex = new Map(); // providerId -> { modelId -> caps }
31
+ let npmIndex = new Map(); // opencode 裸 modelId -> provider.npm(responses 判定用;null = 继承默认)
31
32
  let loadedAt = 0;
32
33
  let inflight = null;
33
34
 
34
35
  function buildIndex(data) {
35
36
  const idx = new Map();
37
+ const npm = new Map();
36
38
  for (const [pid, p] of Object.entries(data || {})) {
37
39
  if (!p || typeof p !== "object") continue;
38
- idx.set(pid, normalizeProviderModels(p.models || {}));
40
+ const caps = normalizeProviderModels(p.models || {});
41
+ idx.set(pid, caps);
42
+ if (String(pid).toLowerCase() === "opencode") {
43
+ for (const [mid, c] of Object.entries(caps)) npm.set(mid, c.npm ?? null);
44
+ }
39
45
  }
46
+ npmIndex = npm;
40
47
  return idx;
41
48
  }
42
49
 
@@ -111,7 +118,7 @@ export function createCapabilitiesService({
111
118
  return [...capsIndex.keys()].sort();
112
119
  }
113
120
 
114
- return { ready, get, list, providers };
121
+ return { ready, get, list, providers, npmIndex: () => new Map(npmIndex) };
115
122
  }
116
123
 
117
124
  // 模块级单例(与 globalDedup 同模式):HTTP handler 懒加载,测试 _reset 后注入
@@ -24,6 +24,8 @@ export function normalizeModelCaps(_id, m) {
24
24
  releaseDate: typeof m?.release_date === "string" && m.release_date ? m.release_date : null,
25
25
  inputModalities: input.length ? input : ["text"],
26
26
  outputModalities: Array.isArray(m?.modalities?.output) && m.modalities.output.length ? m.modalities.output : ["text"],
27
+ // 模型级 SDK 覆盖(models.dev provider.npm):@ai-sdk/openai → responses 端点;null = 继承 provider 默认
28
+ npm: typeof m?.provider?.npm === "string" && m.provider.npm ? m.provider.npm : null,
27
29
  };
28
30
  }
29
31
 
@@ -53,9 +53,9 @@ export function createProviderDispatcher(providers = [], opts = {}) {
53
53
  return provider.chatWithKeys(forwarded, sharedKeys);
54
54
  }
55
55
  if (provider.id === "workbuddy" && workbuddyUid) {
56
- return provider.chat(forwarded, { workbuddyUid });
56
+ return provider.chat(forwarded, { ...opts, workbuddyUid });
57
57
  }
58
- return provider.chat(forwarded);
58
+ return provider.chat(forwarded, opts);
59
59
  }
60
60
 
61
61
  // 聚合所有供应商的模型列表;默认供应商(opencode)裸 id,其它带前缀
@@ -8,7 +8,7 @@ export function createOpenCodeProvider({ upstream, modelsService, baseUrl, authT
8
8
  return {
9
9
  id: "opencode",
10
10
  upstream: client,
11
- chat: (body) => client.chat(body),
11
+ chat: (body, opts) => client.chat(body, opts),
12
12
  preheat: (args) => client.preheat(args),
13
13
  close: () => client.close(),
14
14
  async listModels() {
@@ -33,15 +33,27 @@ export function responsesToChatBody(req = {}) {
33
33
  if (req.instructions) messages.push({ role: "system", content: String(req.instructions) });
34
34
  const input = req.input;
35
35
  const items = typeof input === "string" ? [{ type: "message", role: "user", content: input }] : Array.isArray(input) ? input : [];
36
+ // 加密思考往返:input 里的 reasoning item 挂到下一条 assistant 消息(thinking 跨轮必需,不能丢)
37
+ let pendingReasoning = [];
36
38
  for (const it of items) {
37
39
  if (!it || typeof it !== "object") continue;
40
+ if (it.type === "reasoning") {
41
+ if (it.id || it.encrypted_content) {
42
+ pendingReasoning.push({ id: it.id, encrypted_content: it.encrypted_content, summary: it.summary });
43
+ }
44
+ continue;
45
+ }
38
46
  if (it.type === "message") {
39
- messages.push({ role: it.role || "user", content: inputTextOf(it.content) });
47
+ const msg = { role: it.role || "user", content: inputTextOf(it.content) };
48
+ if (pendingReasoning.length && msg.role === "assistant") { msg.reasoning_items = pendingReasoning; pendingReasoning = []; }
49
+ messages.push(msg);
40
50
  } else if (it.type === "function_call") {
41
- messages.push({
51
+ const msg = {
42
52
  role: "assistant", content: "",
43
53
  tool_calls: [{ id: it.call_id || it.id || "", type: "function", function: { name: it.name || "", arguments: it.arguments || "" } }],
44
- });
54
+ };
55
+ if (pendingReasoning.length) { msg.reasoning_items = pendingReasoning; pendingReasoning = []; }
56
+ messages.push(msg);
45
57
  } else if (it.type === "function_call_output") {
46
58
  messages.push({ role: "tool", tool_call_id: it.call_id || "", content: typeof it.output === "string" ? it.output : JSON.stringify(it.output ?? "") });
47
59
  }
@@ -115,7 +127,16 @@ export function createChunkTranslator(model = "") {
115
127
  let textLen = 0;
116
128
  let lastFinish = "stop";
117
129
  let lastUsage = null;
118
- const tools = new Map(); // index → {id, name, args, announced}
130
+ const tools = new Map(); // index → {id, name, args, announced, outputIndex}
131
+ // output_index 按 item 出现顺序动态分配(reasoning 可能先于 message/tool 出现)
132
+ let nextIdx = 0;
133
+ let textIdx = null;
134
+ // 加密思考(thinking 跨轮):sse.js 首帧带 x_reasoning_item(含加密态),translator 据此建 reasoning item
135
+ let reasoningOpen = false;
136
+ let reasoningId = null;
137
+ let reasoningEncrypted = null;
138
+ let reasoningIdx = null;
139
+ let reasoningText = "";
119
140
  const ev = (type, extra = {}) => ({ type, ...extra });
120
141
 
121
142
  function begin() {
@@ -125,8 +146,9 @@ export function createChunkTranslator(model = "") {
125
146
  function ensureTextItem() {
126
147
  if (textItemOpen) return [];
127
148
  textItemOpen = true;
149
+ textIdx = nextIdx++;
128
150
  const item = { type: "message", id: `msg_${id}`, status: "in_progress", role: "assistant", content: [{ type: "output_text", text: "", annotations: [] }] };
129
- return [ev("response.output_item.added", { output_index: 0, item }), ev("response.content_part.added", { item_id: item.id, output_index: 0, content_index: 0, part: { type: "output_text", text: "", annotations: [] } })];
151
+ return [ev("response.output_item.added", { output_index: textIdx, item }), ev("response.content_part.added", { item_id: item.id, output_index: textIdx, content_index: 0, part: { type: "output_text", text: "", annotations: [] } })];
130
152
  }
131
153
 
132
154
  let fullText = "";
@@ -135,26 +157,51 @@ export function createChunkTranslator(model = "") {
135
157
  const out = ensureTextItem();
136
158
  textLen += delta.length;
137
159
  fullText += delta;
138
- out.push(ev("response.output_text.delta", { item_id: `msg_${id}`, output_index: 0, content_index: 0, delta }));
160
+ out.push(ev("response.output_text.delta", { item_id: `msg_${id}`, output_index: textIdx, content_index: 0, delta }));
161
+ return out;
162
+ }
163
+
164
+ function pushReasoning(delta, meta) {
165
+ const out = ensureReasoningItem(meta);
166
+ reasoningText += delta;
167
+ if (delta) out.push(ev("response.reasoning_summary_text.delta", { item_id: reasoningId, output_index: reasoningIdx, summary_index: 0, delta }));
139
168
  return out;
140
169
  }
141
170
 
171
+ function ensureReasoningItem(meta) {
172
+ if (reasoningOpen) {
173
+ // 补丁:加密态可能晚到(sse.js 在收尾帧补发无文本的 x_reasoning_item)
174
+ if (!reasoningEncrypted && typeof meta?.encrypted_content === "string") reasoningEncrypted = meta.encrypted_content;
175
+ return [];
176
+ }
177
+ reasoningOpen = true;
178
+ reasoningId = meta?.id || `rs_${id}`;
179
+ reasoningEncrypted = typeof meta?.encrypted_content === "string" ? meta.encrypted_content : null;
180
+ reasoningIdx = nextIdx++;
181
+ const item = { type: "reasoning", id: reasoningId, summary: [], ...(reasoningEncrypted ? { encrypted_content: reasoningEncrypted } : {}) };
182
+ return [
183
+ ev("response.output_item.added", { output_index: reasoningIdx, item }),
184
+ ev("response.reasoning_summary_part.added", { item_id: reasoningId, output_index: reasoningIdx, summary_index: 0, part: { type: "summary_text", text: "" } }),
185
+ ];
186
+ }
187
+
142
188
  function pushTool(tc) {
143
189
  const idx = Number(tc.index ?? 0);
144
190
  let t = tools.get(idx);
145
- if (!t) { t = { id: "", name: "", args: "", announced: false }; tools.set(idx, t); }
191
+ if (!t) { t = { id: "", name: "", args: "", announced: false, outputIndex: null }; tools.set(idx, t); }
146
192
  if (tc.id) t.id = tc.id;
147
193
  if (tc.function?.name) t.name += tc.function.name;
148
194
  const frag = tc.function?.arguments || "";
149
195
  const out = [];
150
196
  if (!t.announced && t.id && t.name) {
151
197
  t.announced = true;
152
- out.push(ev("response.output_item.added", { output_index: idx + 1, item: { type: "function_call", id: `fc_${id}_${idx}`, call_id: t.id, name: t.name, arguments: "" } }));
198
+ t.outputIndex = nextIdx++;
199
+ out.push(ev("response.output_item.added", { output_index: t.outputIndex, item: { type: "function_call", id: `fc_${id}_${idx}`, call_id: t.id, name: t.name, arguments: "" } }));
153
200
  }
154
201
  if (frag) {
155
202
  if (!t.announced) { t.args += frag; return out; } // id/name 未到先攒着
156
203
  t.args += frag;
157
- out.push(ev("response.function_call_arguments.delta", { item_id: `fc_${id}_${idx}`, output_index: idx + 1, delta: frag }));
204
+ out.push(ev("response.function_call_arguments.delta", { item_id: `fc_${id}_${idx}`, output_index: t.outputIndex, delta: frag }));
158
205
  }
159
206
  return out;
160
207
  }
@@ -181,29 +228,34 @@ export function createChunkTranslator(model = "") {
181
228
  if (typeof delta.content === "string" && delta.content) { dbg.textChars += delta.content.length; out.push(...pushText(delta.content)); }
182
229
  for (const tc of delta.tool_calls || []) { dbg.toolDeltas++; out.push(...pushTool(tc)); }
183
230
  const rc = typeof delta.reasoning_content === "string" ? delta.reasoning_content : "";
184
- if (rc) { dbg.reasoningChars += rc.length; out.push(...pushText(rc)); }
231
+ if (rc || chunk.x_reasoning_item) { if (rc) dbg.reasoningChars += rc.length; out.push(...pushReasoning(rc, chunk.x_reasoning_item)); }
185
232
  }
186
233
  return out;
187
234
  }
188
235
 
189
236
  function end({ finish = "stop", usage = null } = {}) {
190
237
  const out = [];
238
+ if (reasoningOpen) {
239
+ out.push(ev("response.reasoning_summary_text.done", { item_id: reasoningId, output_index: reasoningIdx, summary_index: 0, text: reasoningText }));
240
+ out.push(ev("response.reasoning_summary_part.done", { item_id: reasoningId, output_index: reasoningIdx, summary_index: 0, part: { type: "summary_text", text: reasoningText } }));
241
+ out.push(ev("response.output_item.done", { output_index: reasoningIdx, item: { type: "reasoning", id: reasoningId, summary: [{ type: "summary_text", text: reasoningText }], ...(reasoningEncrypted ? { encrypted_content: reasoningEncrypted } : {}) } }));
242
+ }
191
243
  if (textItemOpen) {
192
- out.push(ev("response.output_text.done", { item_id: `msg_${id}`, output_index: 0, content_index: 0, text: fullText }));
193
- out.push(ev("response.content_part.done", { item_id: `msg_${id}`, output_index: 0, content_index: 0, part: { type: "output_text", text: fullText, annotations: [] } }));
194
- out.push(ev("response.output_item.done", { output_index: 0, item: { type: "message", id: `msg_${id}`, status: "completed", role: "assistant", content: [{ type: "output_text", text: fullText, annotations: [] }] } }));
244
+ out.push(ev("response.output_text.done", { item_id: `msg_${id}`, output_index: textIdx, content_index: 0, text: fullText }));
245
+ out.push(ev("response.content_part.done", { item_id: `msg_${id}`, output_index: textIdx, content_index: 0, part: { type: "output_text", text: fullText, annotations: [] } }));
246
+ out.push(ev("response.output_item.done", { output_index: textIdx, item: { type: "message", id: `msg_${id}`, status: "completed", role: "assistant", content: [{ type: "output_text", text: fullText, annotations: [] }] } }));
195
247
  }
196
- let i = 0;
197
248
  for (const [idx, t] of tools) {
198
- if (!t.id && !t.name && !t.args) { i++; continue; }
249
+ if (!t.id && !t.name && !t.args) continue;
199
250
  if (!t.announced) {
200
- out.push(ev("response.output_item.added", { output_index: idx + 1, item: { type: "function_call", id: `fc_${id}_${idx}`, call_id: t.id, name: t.name, arguments: "" } }));
251
+ t.announced = true;
252
+ t.outputIndex = nextIdx++;
253
+ out.push(ev("response.output_item.added", { output_index: t.outputIndex, item: { type: "function_call", id: `fc_${id}_${idx}`, call_id: t.id, name: t.name, arguments: "" } }));
201
254
  }
202
- if (t.args) out.push(ev("response.function_call_arguments.done", { item_id: `fc_${id}_${idx}`, output_index: idx + 1, arguments: t.args }));
203
- out.push(ev("response.output_item.done", { output_index: idx + 1, item: { type: "function_call", id: `fc_${id}_${idx}`, call_id: t.id, name: t.name, arguments: t.args, status: "completed" } }));
204
- i++;
255
+ if (t.args) out.push(ev("response.function_call_arguments.done", { item_id: `fc_${id}_${idx}`, output_index: t.outputIndex, arguments: t.args }));
256
+ out.push(ev("response.output_item.done", { output_index: t.outputIndex, item: { type: "function_call", id: `fc_${id}_${idx}`, call_id: t.id, name: t.name, arguments: t.args, status: "completed" } }));
205
257
  }
206
- void i; void textLen;
258
+ void textLen;
207
259
  out.push(ev("response.completed", { response: { id, object: "response", created_at: createdAt, status: finish === "length" ? "incomplete" : "completed", model, output: [], usage: toResponsesUsage(usage) } }));
208
260
  return out;
209
261
  }
@@ -3,6 +3,13 @@ import { recordModelStats } from "../../state.js";
3
3
  import { normalizeFullId } from "../../providers/model-id.js";
4
4
  import { computeMetrics, extractUsageFromJson, extractUsageFromSseText } from "../../metrics.js";
5
5
 
6
+ // 唯一/最后候选没有 failover 去向:首块闸门退化为纯"防连接泄漏",放宽避免误杀慢模型
7
+ // (参考 opencode:zen 通道不设超时;openai responses 硬编码 300s headerTimeout)
8
+ const LAST_CANDIDATE_TIMEOUT_MS = (() => {
9
+ const n = Number(process.env.MSLXDFF_LAST_CANDIDATE_TIMEOUT_MS);
10
+ return Number.isInteger(n) && n >= 0 ? n : 120_000;
11
+ })();
12
+
6
13
  /**
7
14
  * RelayPipeline 深模块
8
15
  * 把 5 个 handler 各自的 fallback→relay→scoring→事件 6段流水收敛为单一真相。
@@ -72,9 +79,16 @@ export function createRelayPipeline({
72
79
  }
73
80
  _evt("relay-start", { reqId, model: actual, via, isStream: Boolean(body?.stream), fallback });
74
81
 
75
- // 3. relay
82
+ // 3. relay(唯一/最后候选:无 failover 去向 → 闸门放宽到防泄漏级别)
83
+ const orderLen = handlerCtx?.orderLen;
84
+ const curIdx = handlerCtx?.idx;
85
+ const isLastCandidate =
86
+ Number.isInteger(orderLen) && orderLen > 0 &&
87
+ (orderLen === 1 || (Number.isInteger(curIdx) && curIdx >= orderLen - 1));
88
+ const streamTimeoutMs = isLastCandidate ? LAST_CANDIDATE_TIMEOUT_MS : C.STREAM_TIMEOUT_MS;
76
89
  const out = await _relay(res, upRes, body, {
77
90
  fallback,
91
+ streamTimeoutMs,
78
92
  onFirstChunk: (delta) => {
79
93
  try { markFn(`ttf-${actual}`); } catch {}
80
94
  _evt("relay-first-chunk", { reqId, model: actual, ttfMs: delta, via });
@@ -99,12 +113,12 @@ export function createRelayPipeline({
99
113
  });
100
114
 
101
115
  // 5a. 首块超时未写字节 → 回退
102
- if (out.status === C.STREAM_TIMEOUT_MS) {
103
- if (auto) try { await auto.recordError(actual, { status: 502, slow: true, note: `stream timeout ${C.STREAM_TIMEOUT_MS}ms` }); } catch {}
104
- try { _logError(actual, 502, `stream timeout ${C.STREAM_TIMEOUT_MS}ms`); } catch {}
116
+ if (streamTimeoutMs > 0 && out.status === streamTimeoutMs) {
117
+ if (auto) try { await auto.recordError(actual, { status: 502, slow: true, note: `stream timeout ${streamTimeoutMs}ms` }); } catch {}
118
+ try { _logError(actual, 502, `stream timeout ${streamTimeoutMs}ms`); } catch {}
105
119
  _evt("upstream-error", { reqId, model: actual, status: 502, message: "stream timeout", timing: null });
106
120
  _evt("fallback", { reqId, from: actual, to: null, reason: "stream timeout" });
107
- return { handled: false, upRes: null, lastErr: { model: actual, upstream: null, status: 502, message: `stream timed out after ${C.STREAM_TIMEOUT_MS}ms` } };
121
+ return { handled: false, upRes: null, lastErr: { model: actual, upstream: null, status: 502, message: `stream timed out after ${streamTimeoutMs}ms` } };
108
122
  }
109
123
 
110
124
  // 5b. 中断(stall 超时 / max 流时长)
@@ -11,6 +11,14 @@ function chunkText(chunk) {
11
11
  return "";
12
12
  }
13
13
 
14
+ // body.cancel() 可能返回非 Promise(自定义/AI SDK 流)——同步异常与 rejection 双路径都要吞掉
15
+ function cancelBody(body) {
16
+ try {
17
+ const p = typeof body?.cancel === "function" ? body.cancel() : null;
18
+ if (p && typeof p.catch === "function") p.catch(() => {});
19
+ } catch { /* ignore */ }
20
+ }
21
+
14
22
  export const SLOW_TOTAL_MS = (() => {
15
23
  const n = Number(process.env.MSLXDFF_SLOW_TOTAL_MS);
16
24
  return Number.isInteger(n) && n > 0 ? n : 20_000;
@@ -18,7 +26,14 @@ export const SLOW_TOTAL_MS = (() => {
18
26
 
19
27
  export const STREAM_TIMEOUT_MS = (() => {
20
28
  const n = Number(process.env.MSLXDFF_STREAM_TIMEOUT_MS);
21
- return Number.isInteger(n) && n > 0 ? n : 25_000;
29
+ // 0 = 显式关闭首块超时(慢思考模型专用);未设/非法值 → 默认 25s
30
+ return Number.isInteger(n) && n >= 0 ? n : 25_000;
31
+ })();
32
+
33
+ // 等首块期间的心跳间隔(SSE 注释帧,标准客户端忽略):上游偶发卡 90s+,避免客户端误判卡死/断连
34
+ export const KEEPALIVE_MS = (() => {
35
+ const n = Number(process.env.MSLXDFF_KEEPALIVE_MS);
36
+ return Number.isInteger(n) && n >= 0 ? n : 10_000;
22
37
  })();
23
38
 
24
39
  export const STALL_TIMEOUT_MS = (() => {
@@ -37,7 +52,7 @@ export const MAX_STREAM_MS = (() => {
37
52
  return Number.isInteger(n) && n > 0 ? n : 0;
38
53
  })();
39
54
 
40
- export async function relay(res, upRes, body, { onFirstChunk, onDownstreamAbort, streamTimeoutMs = STREAM_TIMEOUT_MS, fallback } = {}) {
55
+ export async function relay(res, upRes, body, { onFirstChunk, onDownstreamAbort, streamTimeoutMs = STREAM_TIMEOUT_MS, keepaliveMs = KEEPALIVE_MS, fallback } = {}) {
41
56
  const t0 = performance.now();
42
57
  const contentType = upRes.headers.get("content-type") || "";
43
58
  // 需同时满足:客户端要流 + 上游真的是 SSE;避免 muse-spark 聚合 JSON 被误判为流式,或 workbuddy SSE 被聚合
@@ -75,6 +90,7 @@ export async function relay(res, upRes, body, { onFirstChunk, onDownstreamAbort,
75
90
  downstreamClosed: false,
76
91
  usage: null,
77
92
  chars: 0,
93
+ recoveries: 0,
78
94
  };
79
95
  let prevChunkAt = t0;
80
96
  const onClose = () => {
@@ -100,26 +116,34 @@ export async function relay(res, upRes, body, { onFirstChunk, onDownstreamAbort,
100
116
  let stalled = false;
101
117
  let tooLong = false;
102
118
  let stallTimer = null;
119
+ let pingTimer = keepaliveMs > 0
120
+ ? setInterval(() => {
121
+ if (wroteAny) return;
122
+ try { res.write(": keepalive\n\n"); } catch { /* ignore */ }
123
+ }, keepaliveMs)
124
+ : null;
103
125
  const armStall = () => {
104
126
  if (stallTimer) clearTimeout(stallTimer);
105
127
  stallTimer = STALL_TIMEOUT_MS
106
128
  ? setTimeout(() => {
107
129
  stalled = true;
108
130
  detail.exitReason = "stall";
109
- if (typeof upRes.body.cancel === "function") upRes.body.cancel().catch(() => {});
131
+ cancelBody(upRes.body);
110
132
  }, STALL_TIMEOUT_MS)
111
133
  : null;
112
134
  };
113
- let firstTimer = setTimeout(() => {
114
- timedOut = true;
115
- detail.exitReason = "first-timeout";
116
- if (typeof upRes.body.cancel === "function") upRes.body.cancel().catch(() => {});
117
- }, streamTimeoutMs);
135
+ let firstTimer = streamTimeoutMs > 0
136
+ ? setTimeout(() => {
137
+ timedOut = true;
138
+ detail.exitReason = "first-timeout";
139
+ cancelBody(upRes.body);
140
+ }, streamTimeoutMs)
141
+ : null;
118
142
  const maxTimer = MAX_STREAM_MS
119
143
  ? setTimeout(() => {
120
144
  tooLong = true;
121
145
  detail.exitReason = "max";
122
- if (typeof upRes.body.cancel === "function") upRes.body.cancel().catch(() => {});
146
+ cancelBody(upRes.body);
123
147
  }, MAX_STREAM_MS)
124
148
  : null;
125
149
  try {
@@ -177,6 +201,15 @@ export async function relay(res, upRes, body, { onFirstChunk, onDownstreamAbort,
177
201
  } catch {}
178
202
  }
179
203
  } catch { /* ignore */ }
204
+ // 首块/空闲超时后上游仍吐出了数据 → 只是慢,不是死:撤销超时判定,照常转发
205
+ //(cancel 是协作式的,缓冲数据仍会到达;丢掉已到达的数据是纯损失)
206
+ if (timedOut || stalled) {
207
+ timedOut = false;
208
+ stalled = false;
209
+ detail.recoveries = (detail.recoveries || 0) + 1;
210
+ if (detail.exitReason === "first-timeout" || detail.exitReason === "stall") detail.exitReason = null;
211
+ if (firstTimer) { clearTimeout(firstTimer); firstTimer = null; }
212
+ }
180
213
  if (timedOut || stalled || tooLong) break;
181
214
  if (first) {
182
215
  first = false;
@@ -213,10 +246,11 @@ export async function relay(res, upRes, body, { onFirstChunk, onDownstreamAbort,
213
246
  if (firstTimer) clearTimeout(firstTimer);
214
247
  if (maxTimer) clearTimeout(maxTimer);
215
248
  if (stallTimer) clearTimeout(stallTimer);
249
+ if (pingTimer) clearInterval(pingTimer);
216
250
  }
217
251
  if (timedOut && !wroteAny) {
218
252
  res.removeListener("close", onClose);
219
- return { status: STREAM_TIMEOUT_MS, ttfMs: null, totalMs: Math.round(performance.now() - t0), aborted: true, interrupted: false, detail };
253
+ return { status: streamTimeoutMs, ttfMs: null, totalMs: Math.round(performance.now() - t0), aborted: true, interrupted: false, detail };
220
254
  }
221
255
  if ((stalled || tooLong) && wroteAny) {
222
256
  interrupted = true;
@@ -115,6 +115,24 @@ export async function startServerLifecycle({ VERSION, token, created, upstream,
115
115
  }).catch(() => {});
116
116
  }, 100).unref?.();
117
117
 
118
+ // responses 模型判定改元数据驱动(models.dev provider.npm):就绪后注入,每小时重查使新模型自动识别
119
+ void (async () => {
120
+ try {
121
+ const { globalCapabilities } = await import("../model-capabilities/index.js");
122
+ const { setResponsesNpmIndex } = await import("../upstream-responses.js");
123
+ const svc = globalCapabilities();
124
+ const apply = () => setResponsesNpmIndex(svc.npmIndex());
125
+ await svc.ready();
126
+ apply();
127
+ const entry = { ts: Date.now(), type: "responses-npm-index", models: svc.npmIndex().size };
128
+ try { bus.emit(entry); } catch {}
129
+ try { logs.appendEvent(entry); } catch {}
130
+ setInterval(() => { svc.ready().then(apply).catch(() => {}); }, 60 * 60 * 1000).unref?.();
131
+ } catch (e) {
132
+ try { logs.appendEvent({ ts: Date.now(), type: "responses-npm-index-failed", error: String(e?.message || e).slice(0, 200) }); } catch {}
133
+ }
134
+ })();
135
+
118
136
  models.startAutoRefresh();
119
137
  if (process.env.MSLXDFF_DAEMON) {
120
138
  writePid(process.pid, VERSION);
@@ -57,9 +57,9 @@ export function errorResponseFromSdkError(e, { marker = null } = {}) {
57
57
  });
58
58
  }
59
59
 
60
- // SDK parts → OpenAI SSE Response(chat/responses 共用同一序列化器)。
61
- export function streamResponseFromParts(parts, { marker = null, clock = Date.now, t0 = clock() } = {}) {
62
- const ser = createSseSerializer();
60
+ // SDK parts → OpenAI SSE Response(chat/responses 共用同一序列化器;captured 为 responses 侧信道)。
61
+ export function streamResponseFromParts(parts, { marker = null, clock = Date.now, t0 = clock(), captured = null } = {}) {
62
+ const ser = createSseSerializer(captured);
63
63
  const enc = new TextEncoder();
64
64
  let cancelled = false;
65
65
  const stream = new ReadableStream({
@@ -67,6 +67,11 @@ export function streamResponseFromParts(parts, { marker = null, clock = Date.now
67
67
  try {
68
68
  for await (const part of parts) {
69
69
  if (cancelled) break;
70
+ // 侧信道(captured)可能落后几毫秒:收尾前给 encrypted 一点就绪时间(最多 150ms)
71
+ if (captured && !captured.reasoning && part?.type === "finish") {
72
+ const tw = Date.now();
73
+ while (!captured.reasoning && Date.now() - tw < 150) await new Promise((r) => setTimeout(r, 10));
74
+ }
70
75
  const text = ser.push(part);
71
76
  if (text) controller.enqueue(enc.encode(text));
72
77
  }
@@ -52,7 +52,21 @@ export function toModelPrompt(messages) {
52
52
  const parts = [];
53
53
  const text = textOf(m.content);
54
54
  if (text) parts.push({ type: "text", text });
55
- if (m.reasoning_content) parts.push({ type: "reasoning", text: String(m.reasoning_content) });
55
+ // 带加密态的 reasoning items(responses 通道思考跨轮):AI SDK 会转成上游要的 encrypted reasoning item。
56
+ // 无加密态时才退回纯文本 reasoning(chat 通道的 reasoning_content)。
57
+ const items = Array.isArray(m.reasoning_items) ? m.reasoning_items : [];
58
+ let pushedEncrypted = false;
59
+ for (const r of items) {
60
+ if (!r || typeof r !== "object") continue;
61
+ const summaryText = Array.isArray(r.summary) ? r.summary.map((s) => s?.text || "").join("\n") : "";
62
+ parts.push({
63
+ type: "reasoning",
64
+ text: summaryText || " ",
65
+ providerOptions: { openai: { itemId: r.id, reasoningEncryptedContent: r.encrypted_content } },
66
+ });
67
+ pushedEncrypted = true;
68
+ }
69
+ if (!pushedEncrypted && m.reasoning_content) parts.push({ type: "reasoning", text: String(m.reasoning_content) });
56
70
  for (const tc of Array.isArray(m.tool_calls) ? m.tool_calls : []) {
57
71
  if (!tc || typeof tc !== "object") continue;
58
72
  parts.push({
@@ -8,6 +8,50 @@ import { ENGINE_MARKER } from "./chat.js";
8
8
 
9
9
  export const RESPONSES_CHAT_PATH = "/zen/v1/responses";
10
10
 
11
+ // 上游 encrypted reasoning 只在 output_item.done 里给,而 AI SDK 仅在有 summary 文本时才透出到 parts
12
+ // (muse-spark 这类无 summary 的思考模型会被吞掉)。这里在 fetch 层 tee 一份原始 SSE 自行解析,
13
+ // 侧信道把加密思考交给序列化器,收尾帧补发。
14
+ function captureReasoningFetch(baseFetch, sink) {
15
+ const dbg = process.env.MSLXDFF_RESPONSES_DEBUG === "1";
16
+ return async (input, init) => {
17
+ const res = await baseFetch(input, init);
18
+ const ct = res.headers.get("content-type") || "";
19
+ if (dbg) console.log(`[capture] enter ct=${ct} hasBody=${Boolean(res.body)}`);
20
+ if (!ct.includes("text/event-stream") || !res.body) return res;
21
+ const [passthrough, probe] = res.body.tee();
22
+ (async () => {
23
+ const reader = probe.getReader();
24
+ const dec = new TextDecoder();
25
+ let buf = "";
26
+ try {
27
+ for (;;) {
28
+ const { done, value } = await reader.read();
29
+ if (done) break;
30
+ buf += dec.decode(value, { stream: true });
31
+ let nl;
32
+ while ((nl = buf.indexOf("\n")) >= 0) {
33
+ const line = buf.slice(0, nl).trim();
34
+ buf = buf.slice(nl + 1);
35
+ if (!line.startsWith("data:")) continue;
36
+ const d = line.slice(5).trim();
37
+ if (!d || d === "[DONE]") continue;
38
+ try {
39
+ const j = JSON.parse(d);
40
+ if (dbg && j.type === "response.output_item.done") console.log(`[capture] item.done type=${j.item?.type} enc=${typeof j.item?.encrypted_content}`);
41
+ if (j.type === "response.output_item.done" && j.item?.type === "reasoning" && typeof j.item.encrypted_content === "string") {
42
+ sink.reasoning = { id: j.item.id, encrypted: j.item.encrypted_content, summary: j.item.summary ?? [] };
43
+ if (dbg) console.log(`[capture] hit id=${String(j.item.id).slice(0, 24)} len=${j.item.encrypted_content.length}`);
44
+ }
45
+ } catch { /* ignore */ }
46
+ }
47
+ }
48
+ } catch { /* ignore */ }
49
+ if (dbg) console.log(`[capture] stream end captured=${sink.reasoning ? "yes" : "no"}`);
50
+ })();
51
+ return new Response(passthrough, { status: res.status, statusText: res.statusText, headers: res.headers });
52
+ };
53
+ }
54
+
11
55
  let sdkPromise = null;
12
56
 
13
57
  export function loadOpenAISdk() {
@@ -37,19 +81,34 @@ export async function attemptOnceResponsesSdk({
37
81
  throw err;
38
82
  }
39
83
  // headers 由调用方构造(含 Authorization),headers 优先于 apiKey 默认头
84
+ const captured = { reasoning: null };
85
+ const baseFetch = fetchImpl ?? ((u, i) => globalThis.fetch(u, i));
40
86
  const provider = createOpenAI({
41
87
  name: providerName,
42
88
  baseURL,
43
89
  apiKey: "public",
44
90
  headers: sanitizeHeaders(headers),
45
- ...(fetchImpl ? { fetch: fetchImpl } : {}),
91
+ fetch: captureReasoningFetch(baseFetch, captured),
46
92
  });
47
93
  const model = provider.responses(String(body?.model || ""));
94
+ const params = toModelParams(body, providerName);
95
+ // 无状态 + 加密思考回传(thinking 跨轮):上游把思考以加密块发回,客户端持有并每轮带回。
96
+ // AI SDK responses 的 providerOptionsName 对非 azure 硬编码为 "openai"(dist/index.js:5240)。
97
+ const providerOptions = {
98
+ ...(params.providerOptions || {}),
99
+ openai: {
100
+ store: false,
101
+ include: ["reasoning.encrypted_content"],
102
+ reasoningSummary: "auto",
103
+ ...((params.providerOptions || {}).openai || {}),
104
+ },
105
+ };
48
106
  let res;
49
107
  try {
50
108
  res = await model.doStream({
51
109
  prompt: toModelPrompt(body?.messages),
52
- ...toModelParams(body, providerName),
110
+ ...params,
111
+ providerOptions,
53
112
  tools: toModelTools(body?.tools),
54
113
  toolChoice: toModelToolChoice(body?.tool_choice),
55
114
  });
@@ -58,7 +117,7 @@ export async function attemptOnceResponsesSdk({
58
117
  if (mapped) return mapped;
59
118
  throw e;
60
119
  }
61
- return streamResponseFromParts(res.stream, { marker, clock, t0 });
120
+ return streamResponseFromParts(res.stream, { marker, clock, t0, captured });
62
121
  }
63
122
 
64
123
  export function createSdkResponses({
@@ -32,14 +32,18 @@ export function usageToOpenAI(u) {
32
32
  return out;
33
33
  }
34
34
 
35
- export function createSseSerializer() {
35
+ export function createSseSerializer(captured = null) {
36
36
  const meta = { id: "chatcmpl-wb-sdk", model: "", created: Math.floor(Date.now() / 1000) };
37
37
  let roleSent = false;
38
38
  let nextIndex = 0;
39
39
  const toolIndex = new Map();
40
40
  const toolDeltaIds = new Set();
41
+ // 加密思考往返:reasoning-start 的 providerMetadata 带 itemId + 加密态,随首帧透出
42
+ let reasoningMeta = null;
43
+ let reasoningFrameSent = false;
44
+ let reasoningEncSent = false;
41
45
 
42
- function frame(delta, { finishReason = null, usage } = {}) {
46
+ function frame(delta, { finishReason = null, usage, extra } = {}) {
43
47
  const obj = {
44
48
  id: meta.id,
45
49
  object: "chat.completion.chunk",
@@ -48,6 +52,7 @@ export function createSseSerializer() {
48
52
  choices: [{ index: 0, delta, finish_reason: finishReason }],
49
53
  };
50
54
  if (usage !== undefined) obj.usage = usage;
55
+ if (extra) Object.assign(obj, extra);
51
56
  return `data: ${JSON.stringify(obj)}\n\n`;
52
57
  }
53
58
 
@@ -69,8 +74,30 @@ export function createSseSerializer() {
69
74
  }
70
75
  return null;
71
76
  }
72
- case "reasoning-delta":
73
- return ensureRole() + frame({ reasoning_content: String(part.delta ?? "") });
77
+ case "reasoning-start": {
78
+ const pm = part.providerMetadata?.openai || part.providerMetadata || {};
79
+ reasoningMeta = {
80
+ id: String(pm.itemId ?? part.id ?? "reasoning"),
81
+ encrypted: typeof pm.reasoningEncryptedContent === "string" ? pm.reasoningEncryptedContent : null,
82
+ };
83
+ // 即时透出 item 元数据(含加密态):上游可能只给 encrypted 不给 summary 文本,不能等 delta
84
+ reasoningFrameSent = true;
85
+ if (reasoningMeta.encrypted) reasoningEncSent = true;
86
+ return ensureRole() + frame({ reasoning_content: "" }, { extra: { x_reasoning_item: { id: reasoningMeta.id, encrypted_content: reasoningMeta.encrypted } } });
87
+ }
88
+ case "reasoning-delta": {
89
+ const delta = String(part.delta ?? "");
90
+ let extra;
91
+ if (reasoningMeta && !reasoningFrameSent) {
92
+ reasoningFrameSent = true;
93
+ if (reasoningMeta.encrypted) reasoningEncSent = true;
94
+ // x_reasoning_item:responses translator 据此建 reasoning item(chat 客户端忽略未知顶层字段)
95
+ extra = { x_reasoning_item: { id: reasoningMeta.id, encrypted_content: reasoningMeta.encrypted } };
96
+ } else if (reasoningMeta) {
97
+ extra = { x_reasoning_id: reasoningMeta.id };
98
+ }
99
+ return ensureRole() + frame({ reasoning_content: delta }, { extra });
100
+ }
74
101
  case "text-delta":
75
102
  return ensureRole() + frame({ content: String(part.delta ?? "") });
76
103
  case "tool-input-start": {
@@ -98,7 +125,14 @@ export function createSseSerializer() {
98
125
  const fr = part.finishReason;
99
126
  const raw = typeof fr === "string" ? fr : (fr?.raw ?? fr?.unified);
100
127
  const reason = raw === "tool-calls" ? "tool_calls" : raw === "content-filter" ? "content_filter" : (raw ?? "stop");
101
- return frame({}, { finishReason: reason, usage: usageToOpenAI(part.usage) });
128
+ // 加密思考补发:上游 encrypted_content 只在流末尾可得(fetch 侧信道捕获),
129
+ // 无 summary 文本时 AI SDK parts 不会带出 → 收尾帧前补一帧
130
+ let pre = "";
131
+ if (captured?.reasoning && !reasoningEncSent) {
132
+ reasoningEncSent = true;
133
+ pre = ensureRole() + frame({ reasoning_content: "" }, { extra: { x_reasoning_item: { id: captured.reasoning.id, encrypted_content: captured.reasoning.encrypted } } });
134
+ }
135
+ return pre + frame({}, { finishReason: reason, usage: usageToOpenAI(part.usage) });
102
136
  }
103
137
  case "error": {
104
138
  const message = part.error?.message
@@ -2,8 +2,26 @@
2
2
  * responses 转换层 — 从 upstream.js 抽出的 muse-spark 专用形状转换。
3
3
  * chat ⇄ responses 互转纯函数,无网络、无副作用。
4
4
  */
5
+ // responses 判定索引(models.dev 模型级 provider.npm,启动时注入):
6
+ // "@ai-sdk/openai" → responses 端点;未注入/未命中 → 前缀兜底(新模型早于 models.dev 刷新时仍可用)。
7
+ let responsesNpmIndex = null;
8
+
9
+ export function setResponsesNpmIndex(idx) {
10
+ responsesNpmIndex = idx instanceof Map ? idx : null;
11
+ }
12
+
13
+ export function _resetResponsesNpmIndex() {
14
+ responsesNpmIndex = null;
15
+ }
16
+
5
17
  export function isResponsesModel(model) {
6
- return String(model || "").toLowerCase().startsWith("muse-spark");
18
+ const m = String(model || "").toLowerCase().trim();
19
+ if (!m) return false;
20
+ if (responsesNpmIndex) {
21
+ const bare = m.replace(/^opencode\//, "");
22
+ if (responsesNpmIndex.has(bare)) return responsesNpmIndex.get(bare) === "@ai-sdk/openai";
23
+ }
24
+ return m.startsWith("muse-spark");
7
25
  }
8
26
 
9
27
  export function chatToResponsesBody(chatBody) {
package/src/upstream.js CHANGED
@@ -12,6 +12,24 @@ import { uuid } from "./compat.js";
12
12
  function genId(prefix) {
13
13
  return `${prefix}${uuid().replace(/-/g, "")}`;
14
14
  }
15
+ // 上游按 session 做粘性路由(实测:固定 session 两次请求均 ~1.2s;每次随机时可能撞冷机器 26s+)。
16
+ // 客户端(opencode AI SDK 路径)不带会话标识 → 用对话首两条消息(system + 首条 user)哈希做稳定会话:
17
+ // 同一会话多轮里这两条不变 ⇒ 路由亲和稳定;不同会话天然分散。
18
+ function sessionFromMessages(messages) {
19
+ try {
20
+ const msgs = Array.isArray(messages) ? messages : [];
21
+ const pick = (role) => {
22
+ const m = msgs.find((x) => x?.role === role);
23
+ if (!m) return "";
24
+ return typeof m.content === "string" ? m.content : JSON.stringify(m.content ?? "");
25
+ };
26
+ const seed = `${pick("system")}|${pick("user")}`.slice(0, 4000);
27
+ if (seed === "|") return null;
28
+ return `ses_${crypto.createHash("sha1").update(seed).digest("hex").slice(0, 32)}`;
29
+ } catch {
30
+ return null;
31
+ }
32
+ }
15
33
  function envInt(name, fallback) {
16
34
  const v = Number(process.env[name]);
17
35
  return Number.isInteger(v) && v > 0 ? v : fallback;
@@ -47,13 +65,16 @@ export function createOpencodeHeaderBuilder({ authToken = "public", env = proces
47
65
  Authorization: `Bearer ${authToken}`,
48
66
  "x-opencode-client": "desktop",
49
67
  };
50
- function buildHeaders(body, { anonymous = anonFirst } = {}) {
68
+ // 无 messages 可哈希时的进程级兜底:至少 daemon 生命周期内稳定,不再每请求随机
69
+ const FALLBACK_SESSION = genId("ses_");
70
+ function buildHeaders(body, { anonymous = anonFirst, sessionId = null } = {}) {
51
71
  const isStream = body?.stream !== false;
72
+ const session = sessionId || sessionFromMessages(body?.messages) || FALLBACK_SESSION;
52
73
  const base = {
53
74
  ...baseHeaders,
54
75
  Accept: isStream ? "text/event-stream" : "*/*",
55
76
  "User-Agent": "opencode",
56
- "x-opencode-session": genId("ses_"),
77
+ "x-opencode-session": session,
57
78
  "x-opencode-request": genId("msg_"),
58
79
  "x-opencode-project": "global",
59
80
  };
@@ -111,8 +132,10 @@ export function createUpstreamClient({
111
132
  hooks,
112
133
  });
113
134
 
114
- async function chat(body) {
135
+ async function chat(body, opts = {}) {
115
136
  const isResp = isResponsesModel(body?.model);
137
+ // responses 路径的 reqBody 无 messages,会话哈希必须基于原始 chat body 计算
138
+ const sessionId = opts?.sessionId || sessionFromMessages(body?.messages) || null;
116
139
  const url = isResp ? `${baseUrl}/zen/v1/responses` : `${baseUrl}/zen/v1/chat/completions`;
117
140
  const reqBody = isResp ? chatToResponsesBody(body) : body;
118
141
  const t0 = performance.now();
@@ -123,7 +146,7 @@ export function createUpstreamClient({
123
146
  res = await transport.request({
124
147
  url,
125
148
  method: "POST",
126
- headers: buildHeaders(reqBody),
149
+ headers: buildHeaders(reqBody, { sessionId }),
127
150
  body: reqBody,
128
151
  stream: body?.stream !== false,
129
152
  timeoutMs: connectTimeoutMs,
@@ -145,7 +168,7 @@ export function createUpstreamClient({
145
168
  anonRes = await transport.request({
146
169
  url,
147
170
  method: "POST",
148
- headers: buildHeaders(reqBody, { anonymous: true }),
171
+ headers: buildHeaders(reqBody, { anonymous: true, sessionId }),
149
172
  body: reqBody,
150
173
  stream: body?.stream !== false,
151
174
  timeoutMs: connectTimeoutMs,