mslxdff 0.1.145 → 0.1.148

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "mslxdff",
3
- "version": "0.1.145",
3
+ "version": "0.1.148",
4
4
  "description": "测试项目,请勿使用。",
5
5
  "type": "module",
6
6
  "bin": {
@@ -170,18 +170,21 @@ export function createChatService({
170
170
  const sessionId = genSessionId();
171
171
  const isStream = body?.stream === true;
172
172
  const upstreamModel = stripProviderPrefix(model, id);
173
- // token 口径双写:对标官方 withMaxCompletionTokensForReasoningModels——
174
- // cline 上游默认 reasoning_effort high,推理模型认 max_completion_tokens,
175
- // 只发 max_tokens 会被部分通道拒;双写兼容最稳。
176
- const tokLimit = body?.max_tokens || body?.max_completion_tokens || 4096;
173
+ // token 上限对标 dsh-cline-pass(DEFAULT_MAX_TOKENS=32000):客户端显式 max 优先,
174
+ // 缺省给 32000 而非 4096——旧 4096 默认会在长文处 finish=length 拦腰截断(2026-09-23 实测)。
175
+ // reasoning_effort 缺省不发(dsh 同款:调用方自选 none→max,不替上游做 high 假设;
176
+ // 高推理+小预算双挤是超大上下文秒回超短答的推手之一);max_completion_tokens 双写保留
177
+ //(只发 max_tokens 会被部分通道拒,历史教训)。
178
+ const tokLimit = body?.max_tokens || body?.max_completion_tokens || 32000;
177
179
  const upstreamBody = {
178
180
  model: upstreamModel,
179
181
  max_tokens: tokLimit,
180
182
  max_completion_tokens: tokLimit,
181
183
  session_id: sessionId,
182
- reasoning_effort: body?.reasoning_effort || body?.reasoningEffort || "high",
183
184
  messages: body?.messages || [],
184
185
  };
186
+ const effort = body?.reasoning_effort || body?.reasoningEffort;
187
+ if (effort) upstreamBody.reasoning_effort = effort;
185
188
  // 部分免费通道(deepseek 家族,含 cline-free/deepseek-*)原生非流式不可靠
186
189
  //(500 empty response):这类模型在内部走 stream+聚合成 JSON,对外仍按请求方
187
190
  // stream 标志返回(true 直透,false 聚合)。其它模型完全尊重请求方,不强制。
@@ -81,20 +81,63 @@ export function createQoderProvider({
81
81
  return chatSvc.runChat(stripped, picked.sess, picked.region);
82
82
  }
83
83
 
84
- async function listModels() {
85
- const picked = pickSession();
86
- if (!picked) return [];
87
- try { return await modelsSvc.listModels(picked.sess, picked.region); } catch { return []; }
88
- }
89
-
90
- async function preheat() {
91
- try {
92
- const picked = pickSession();
93
- if (!picked) return { ok: false, error: "no account" };
94
- await modelsSvc.listModels(picked.sess, picked.region);
95
- return { ok: true };
96
- } catch (e) { return { ok: false, error: String(e?.message || e).slice(0, 120) }; }
97
- }
84
+ // 全号聚合:双号分属 cn/global 两区,模型表各不同(cn 14 个/global 15 个);
85
+ // 只取单号会漏另一区(如轮询到 global 就看不到 cn 独有的 q37fmodel/gm51model)。
86
+ // 按 id 并集去重,只返回 enable=true 的可调用模型(过滤已下沉到 models.js)。
87
+ function eachCred() {
88
+ const out = [];
89
+ const seen = new Set();
90
+ const push = (blob, auth) => {
91
+ if (!blob?.deviceToken || seen.has(blob.deviceToken)) return;
92
+ seen.add(blob.deviceToken);
93
+ out.push({ blob, auth: auth || {} });
94
+ };
95
+ for (const k of keys) {
96
+ const blob = accountFromBlob(k);
97
+ if (!blob?.deviceToken) continue;
98
+ const auth = authList.find((a) => String(a?.refreshToken || "") === String(blob.refreshToken || "")) || authList[0] || {};
99
+ push(blob, auth);
100
+ }
101
+ if (!out.length) {
102
+ // keys 为空但 auth 目录有号(如 state 被外部改写):从落盘 doc 合成凭据,目录不断即可出列表
103
+ try {
104
+ for (const { uid, doc } of listAccountDocs()) {
105
+ const a = doc?.auth || {};
106
+ if (!a.deviceToken) continue;
107
+ push({ deviceToken: a.deviceToken, refreshToken: a.refreshToken || "" },
108
+ { uid, name: doc?.account?.name || "", region: a.region || "global", refreshToken: a.refreshToken || "" });
109
+ }
110
+ } catch {}
111
+ }
112
+ return out;
113
+ }
114
+
115
+ async function listModels() {
116
+ const creds = eachCred();
117
+ if (!creds.length) return [];
118
+ const seen = new Set();
119
+ const out = [];
120
+ for (const { blob, auth } of creds) {
121
+ const region = normalizeRegion(region0 || auth.region || "global");
122
+ const sess = buildSessionFor({ ...blob, uid: auth.uid, name: auth.name, region });
123
+ try {
124
+ const list = await modelsSvc.listModels(sess, region);
125
+ for (const m of list) {
126
+ if (!m?.id || seen.has(m.id)) continue;
127
+ seen.add(m.id);
128
+ out.push(m);
129
+ }
130
+ } catch {}
131
+ }
132
+ return out;
133
+ }
134
+
135
+ async function preheat() {
136
+ try {
137
+ const list = await listModels();
138
+ return list.length ? { ok: true } : { ok: false, error: "no account" };
139
+ } catch (e) { return { ok: false, error: String(e?.message || e).slice(0, 120) }; }
140
+ }
98
141
 
99
142
  async function close() {}
100
143
  async function chatWithKeys(body, keysOverride) {
@@ -1,5 +1,5 @@
1
- // qoder 模型服务(转译 bridge.go ListModels/extractModels):原生 COSY 签名拉 /model/list,
2
- // 10min 缓存;暴露全部 15 个 key(含 enable=false 的展示,allowlist 由上层过滤)。
1
+ // qoder 模型服务(转译 bridge.go ListModels/extractModels):原生 COSY 签名拉 /model/list,
2
+ // 10min 缓存(按 region 分桶);只暴露 enable=true 的可用模型(enable=false 调对话必挂,不展示)。
3
3
  import { compatFetch, timeoutSignal } from "../../compat.js";
4
4
  import { getEndpoints, normalizeRegion } from "./constants.js";
5
5
  import { buildCosyHeaders, pathSigFrom } from "./session.js";
@@ -29,34 +29,45 @@ export function extractModels(chatList) {
29
29
  export function createModelsService({ id = "qoder", fetchImpl, clock = Date.now } = {}) {
30
30
  let cache = null, fetchedAt = 0;
31
31
 
32
- async function listModels(sess, region = "global") {
33
- const normRegion = normalizeRegion(region);
34
- const now = clock();
35
- if (cache && now - fetchedAt < CACHE_TTL_MS) return cache;
36
- const ep = getEndpoints(normRegion);
37
- const url = ep.modelListURL;
38
- const headers = buildCosyHeaders(sess, pathSigFrom(url), "", "application/json");
39
- const res = await fetchImpl(url, { headers, signal: timeoutSignal(15000) });
40
- if (!res.ok) throw new Error(`models http ${res.status}`);
41
- const j = await res.json().catch(() => ({}));
42
- const models = extractModels(j.chat) || extractModels(j.assistant);
43
- if (!models.length) throw new Error("models list empty");
44
- cache = models.map((m) => ({
45
- id: joinModelId(id, m.id),
46
- object: "model",
47
- created: CREATED,
48
- owned_by: OWNER,
49
- name: m.name,
50
- context_length: m.contextLength,
51
- enable: m.enable,
52
- is_reasoning: m.isReasoning,
53
- price_factor: m.priceFactor,
54
- }));
55
- fetchedAt = now;
56
- return cache;
57
- }
58
-
59
- function clearCache() { cache = null; fetchedAt = 0; }
60
-
61
- return { listModels, clearCache, _getCache: () => cache };
62
- }
32
+ // 按 region 分桶缓存:同一进程可能有 cn + global 双号,混用单桶会串区(cn 14 个/global 15 个各不同)。
33
+ const cacheByRegion = new Map();
34
+ async function listModels(sess, region = "global") {
35
+ const normRegion = normalizeRegion(region);
36
+ const now = clock();
37
+ const hit = cacheByRegion.get(normRegion);
38
+ if (hit && now - hit.at < CACHE_TTL_MS) return hit.list;
39
+ const ep = getEndpoints(normRegion);
40
+ const url = ep.modelListURL;
41
+ const headers = buildCosyHeaders(sess, pathSigFrom(url), "", "application/json");
42
+ const res = await fetchImpl(url, { headers, signal: timeoutSignal(15000) });
43
+ if (!res.ok) throw new Error(`models http ${res.status}`);
44
+ const j = await res.json().catch(() => ({}));
45
+ // extractModels 恒返回数组(空数组亦 truthy),必须按长度选 chat/assistant 分支
46
+ const chatModels = extractModels(j.chat);
47
+ const models = chatModels.length ? chatModels : extractModels(j.assistant);
48
+ if (!models.length) throw new Error("models list empty");
49
+ // 只留可用:enable=false 的 key(如 global 的 auto/ultimate,cn 的绝大多数)展示即误导
50
+ const usable = models.filter((m) => m.enable);
51
+ if (!usable.length) throw new Error("models list empty (no enabled models)");
52
+ const list = usable.map((m) => ({
53
+ id: joinModelId(id, m.id),
54
+ object: "model",
55
+ created: CREATED,
56
+ owned_by: OWNER,
57
+ name: m.name,
58
+ context_length: m.contextLength,
59
+ enable: true,
60
+ is_reasoning: m.isReasoning,
61
+ price_factor: m.priceFactor,
62
+ }));
63
+ cacheByRegion.set(normRegion, { list, at: now });
64
+ // 兼容旧单测 _getCache:默认读 global 桶
65
+ cache = normRegion === "global" ? list : cache;
66
+ fetchedAt = normRegion === "global" ? now : fetchedAt;
67
+ return list;
68
+ }
69
+
70
+ function clearCache() { cache = null; fetchedAt = 0; cacheByRegion.clear(); }
71
+
72
+ return { listModels, clearCache, _getCache: () => cache ?? cacheByRegion.get("global")?.list ?? null };
73
+ }
@@ -54,6 +54,9 @@ export function responsesToChatBody(req = {}) {
54
54
  // 客户端重试/重放历史时会把同一 item 存两份),故按 id 保首个、丢后续——重复项无合法语义,删了不吃亏。
55
55
  let pendingReasoning = [];
56
56
  const seenReasoningIds = new Set();
57
+ // 同 call_id 的 function_call_output 重复出现 → 上游 400 "Duplicate function_call_output"
58
+ // (2026-09-23 实测,客户端重试/重放历史时会把同一工具结果存两份),故同样按 call_id 保首个、丢后续。
59
+ const seenToolResultIds = new Set();
57
60
  for (const it of items) {
58
61
  if (!it || typeof it !== "object") continue;
59
62
  if (it.type === "reasoning") {
@@ -81,6 +84,8 @@ export function responsesToChatBody(req = {}) {
81
84
  if (pendingReasoning.length) { msg.reasoning_items = pendingReasoning; pendingReasoning = []; }
82
85
  messages.push(msg);
83
86
  } else if (it.type === "function_call_output") {
87
+ const _tkey = String(it.call_id || "");
88
+ if (_tkey) { if (seenToolResultIds.has(_tkey)) continue; seenToolResultIds.add(_tkey); }
84
89
  messages.push({ role: "tool", tool_call_id: it.call_id || "", content: typeof it.output === "string" ? it.output : JSON.stringify(it.output ?? "") });
85
90
  }
86
91
  }
@@ -117,6 +117,26 @@ export function createRelayPipeline({
117
117
  detail: out.detail ?? null,
118
118
  });
119
119
 
120
+ // 5a0. 空转 200:流正常结束但零正文零工具调用 → 客户端会报 EMPTY_MODEL_RESPONSE
121
+ // ("The model ended its turn without producing any output");转 failover 而不是
122
+ // 把空轮递给客户端。chatShaped 是前置证据:只有看得出是 chat 轮才判空,
123
+ // 非 chat SSE/无 choices JSON 透传是正式契约(chat-route 单测锁死),一律放行。
124
+ // 工具轮豁免:tool_calls 无正文是合法 agent 形态;
125
+ // finish=tool_calls/function_call 兜底豁免(防计数漏检误杀);下游已断开不重试(写给谁看)。
126
+ // 对标 dsh-cline-pass 的 EMPTY_RESPONSE 语义;不记 auto 冷却(空转≠模型坏,重试多半能好)。
127
+ const _d = out.detail || {};
128
+ const _emptyTurn = out.status === 200 && !out.timedOut && !out.interrupted && !_d.downstreamClosed &&
129
+ _d.chatShaped === true &&
130
+ (Number(_d.chars) || 0) === 0 && (Number(_d.toolCalls) || 0) === 0 &&
131
+ !["tool_calls", "function_call"].includes(_d.sawFinishReason);
132
+ if (_emptyTurn) {
133
+ const _why = _d.sawFinishReason ? ` (finish_reason=${_d.sawFinishReason})` : " (no content, no tool calls)";
134
+ try { _logError(actual, 502, `empty turn${_why}`); } catch {}
135
+ _evt("upstream-error", { reqId, model: actual, status: 502, message: "empty turn", timing: null });
136
+ _evt("fallback", { reqId, from: actual, to: null, reason: "empty turn" });
137
+ return { handled: false, upRes: null, lastErr: { model: actual, upstream: null, status: 502, message: `EMPTY_MODEL_RESPONSE: upstream returned 200 with no content${_why} — retry or rephrase` } };
138
+ }
139
+
120
140
  // 5a. 首块超时未写字节 → 回退(显式 timedOut 字段,status 只是 HTTP 语义展示)
121
141
  if (out.timedOut === true) {
122
142
  const why = out.detail?.upstreamError ? ` (upstream read error: ${out.detail.upstreamError})` : "";
@@ -114,6 +114,8 @@ export async function relay(res, upRes, body, { onFirstChunk, onDownstreamAbort,
114
114
  downstreamClosed: false,
115
115
  usage: null,
116
116
  chars: 0,
117
+ toolCalls: 0,
118
+ chatShaped: false,
117
119
  recoveries: 0,
118
120
  };
119
121
  let prevChunkAt = t0;
@@ -209,6 +211,12 @@ export async function relay(res, upRes, body, { onFirstChunk, onDownstreamAbort,
209
211
  if (txt.includes("[DONE]")) detail.sawDone = true;
210
212
  const m = txt.match(/"finish_reason"\s*:\s*"([^"]+)"/);
211
213
  if (m) detail.sawFinishReason = m[1];
214
+ // chat 形状证据:只有看得出是 chat 轮才配判空(非 chat SSE/JSON 透传是正式契约,不得误伤)
215
+ if (!detail.chatShaped && (txt.includes('"choices"') || txt.includes('"delta"') || txt.includes('"finish_reason"') || txt.includes('"usage"') || txt.includes('"prompt_tokens"') || txt.includes("[DONE]"))) detail.chatShaped = true;
216
+ // 工具调用计数(空数组不算):tool_calls 无正文是合法 agent 轮,
217
+ // 空转闸门必须豁免它,否则所有工具轮都会被误判为空轮——见 relay-pipeline 4b。
218
+ const tc = txt.match(/"tool_calls"\s*:\s*\[\s*\{/g);
219
+ if (tc) detail.toolCalls = (detail.toolCalls || 0) + tc.length;
212
220
  // 尝试提取 usage(流式末帧):口径收口到 metrics.js,与未流式分支共用
213
221
  if (txt.includes("\"usage\"") || txt.includes("\"prompt_tokens\"")) {
214
222
  try {
@@ -326,6 +334,10 @@ export async function relay(res, upRes, body, { onFirstChunk, onDownstreamAbort,
326
334
  if (u) detail.usage = u;
327
335
  if (parsed.choices?.[0]?.message?.content) detail.chars = String(parsed.choices[0].message.content).length;
328
336
  else if (parsed.choices?.[0]?.text) detail.chars = String(parsed.choices[0].text).length;
337
+ // 非流式同样只判 chat 形状:无 choices 的任意 JSON 是透传契约(chat-route 单测锁死),不得判空
338
+ if (Array.isArray(parsed?.choices)) detail.chatShaped = true;
339
+ const _tcList = parsed.choices?.[0]?.message?.tool_calls;
340
+ if (Array.isArray(_tcList) && _tcList.length) detail.toolCalls = _tcList.length;
329
341
  const enriched = enrichNonStreamJson(parsed, fallback);
330
342
  json(res, upRes.status, enriched);
331
343
  } catch {
@@ -62,6 +62,9 @@ export function toModelPrompt(messages, { dropEncrypted = false } = {}) {
62
62
  // (上游报错原文即要求 Remove duplicate items)。这里按 itemId 保首个、丢后续,
63
63
  // 防客户端重放/其他生产者把同一加密 item 组装两次。
64
64
  const seenReasoningIds = new Set();
65
+ // 同 toolCallId 的 tool 结果重复出现 → 上游 400 "Duplicate function_call_output"
66
+ // (2026-09-23 实测:responses 入站虽已去重,此处仍加防线,防其他生产者直接组装 chat messages 重放)。
67
+ const seenToolResultIds = new Set();
65
68
  for (const m of Array.isArray(messages) ? messages : []) {
66
69
  if (!m || typeof m !== "object") continue;
67
70
  const role = String(m.role || "");
@@ -112,6 +115,8 @@ export function toModelPrompt(messages, { dropEncrypted = false } = {}) {
112
115
  }
113
116
  out.push({ role: "assistant", content: parts });
114
117
  } else if (role === "tool") {
118
+ const _tkey = String(m.tool_call_id || "");
119
+ if (_tkey) { if (seenToolResultIds.has(_tkey)) continue; seenToolResultIds.add(_tkey); }
115
120
  out.push({
116
121
  role: "tool",
117
122
  content: [{
@@ -24,6 +24,7 @@ export function diagnoseToolSequence(messages) {
24
24
  const list = Array.isArray(messages) ? messages : [];
25
25
  const issues = [];
26
26
  const open = new Map();
27
+ const seenResults = new Set();
27
28
  let calls = 0;
28
29
  let results = 0;
29
30
  const first = list[0];
@@ -50,8 +51,8 @@ export function diagnoseToolSequence(messages) {
50
51
  results++;
51
52
  const id = String(m.tool_call_id || "");
52
53
  if (!id) issues.push(`#${i} tool 结果空 id`);
53
- else if (!open.has(id)) issues.push(`#${i} 孤立结果:${id}`);
54
- else open.delete(id);
54
+ else if (seenResults.has(id)) issues.push(`#${i} 结果重复:${id}(上游按 call_id 查重会 400 Duplicate function_call_output,去重后只应出现一次)`);
55
+ else { seenResults.add(id); if (!open.has(id)) issues.push(`#${i} 孤立结果:${id}`); else open.delete(id); }
55
56
  } else if (open.size) {
56
57
  issues.push(`#${i} ${role} 打断{${[...open.keys()].join(",")}}`);
57
58
  }