mslxdff 0.1.145 → 0.1.147

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "mslxdff",
3
- "version": "0.1.145",
3
+ "version": "0.1.147",
4
4
  "description": "测试项目,请勿使用。",
5
5
  "type": "module",
6
6
  "bin": {
@@ -170,18 +170,21 @@ export function createChatService({
170
170
  const sessionId = genSessionId();
171
171
  const isStream = body?.stream === true;
172
172
  const upstreamModel = stripProviderPrefix(model, id);
173
- // token 口径双写:对标官方 withMaxCompletionTokensForReasoningModels——
174
- // cline 上游默认 reasoning_effort high,推理模型认 max_completion_tokens,
175
- // 只发 max_tokens 会被部分通道拒;双写兼容最稳。
176
- const tokLimit = body?.max_tokens || body?.max_completion_tokens || 4096;
173
+ // token 上限对标 dsh-cline-pass(DEFAULT_MAX_TOKENS=32000):客户端显式 max 优先,
174
+ // 缺省给 32000 而非 4096——旧 4096 默认会在长文处 finish=length 拦腰截断(2026-09-23 实测)。
175
+ // reasoning_effort 缺省不发(dsh 同款:调用方自选 none→max,不替上游做 high 假设;
176
+ // 高推理+小预算双挤是超大上下文秒回超短答的推手之一);max_completion_tokens 双写保留
177
+ //(只发 max_tokens 会被部分通道拒,历史教训)。
178
+ const tokLimit = body?.max_tokens || body?.max_completion_tokens || 32000;
177
179
  const upstreamBody = {
178
180
  model: upstreamModel,
179
181
  max_tokens: tokLimit,
180
182
  max_completion_tokens: tokLimit,
181
183
  session_id: sessionId,
182
- reasoning_effort: body?.reasoning_effort || body?.reasoningEffort || "high",
183
184
  messages: body?.messages || [],
184
185
  };
186
+ const effort = body?.reasoning_effort || body?.reasoningEffort;
187
+ if (effort) upstreamBody.reasoning_effort = effort;
185
188
  // 部分免费通道(deepseek 家族,含 cline-free/deepseek-*)原生非流式不可靠
186
189
  //(500 empty response):这类模型在内部走 stream+聚合成 JSON,对外仍按请求方
187
190
  // stream 标志返回(true 直透,false 聚合)。其它模型完全尊重请求方,不强制。
@@ -81,20 +81,63 @@ export function createQoderProvider({
81
81
  return chatSvc.runChat(stripped, picked.sess, picked.region);
82
82
  }
83
83
 
84
- async function listModels() {
85
- const picked = pickSession();
86
- if (!picked) return [];
87
- try { return await modelsSvc.listModels(picked.sess, picked.region); } catch { return []; }
88
- }
89
-
90
- async function preheat() {
91
- try {
92
- const picked = pickSession();
93
- if (!picked) return { ok: false, error: "no account" };
94
- await modelsSvc.listModels(picked.sess, picked.region);
95
- return { ok: true };
96
- } catch (e) { return { ok: false, error: String(e?.message || e).slice(0, 120) }; }
97
- }
84
+ // 全号聚合:双号分属 cn/global 两区,模型表各不同(cn 14 个/global 15 个);
85
+ // 只取单号会漏另一区(如轮询到 global 就看不到 cn 独有的 q37fmodel/gm51model)。
86
+ // 按 id 并集去重,只返回 enable=true 的可调用模型(过滤已下沉到 models.js)。
87
+ function eachCred() {
88
+ const out = [];
89
+ const seen = new Set();
90
+ const push = (blob, auth) => {
91
+ if (!blob?.deviceToken || seen.has(blob.deviceToken)) return;
92
+ seen.add(blob.deviceToken);
93
+ out.push({ blob, auth: auth || {} });
94
+ };
95
+ for (const k of keys) {
96
+ const blob = accountFromBlob(k);
97
+ if (!blob?.deviceToken) continue;
98
+ const auth = authList.find((a) => String(a?.refreshToken || "") === String(blob.refreshToken || "")) || authList[0] || {};
99
+ push(blob, auth);
100
+ }
101
+ if (!out.length) {
102
+ // keys 为空但 auth 目录有号(如 state 被外部改写):从落盘 doc 合成凭据,目录不断即可出列表
103
+ try {
104
+ for (const { uid, doc } of listAccountDocs()) {
105
+ const a = doc?.auth || {};
106
+ if (!a.deviceToken) continue;
107
+ push({ deviceToken: a.deviceToken, refreshToken: a.refreshToken || "" },
108
+ { uid, name: doc?.account?.name || "", region: a.region || "global", refreshToken: a.refreshToken || "" });
109
+ }
110
+ } catch {}
111
+ }
112
+ return out;
113
+ }
114
+
115
+ async function listModels() {
116
+ const creds = eachCred();
117
+ if (!creds.length) return [];
118
+ const seen = new Set();
119
+ const out = [];
120
+ for (const { blob, auth } of creds) {
121
+ const region = normalizeRegion(region0 || auth.region || "global");
122
+ const sess = buildSessionFor({ ...blob, uid: auth.uid, name: auth.name, region });
123
+ try {
124
+ const list = await modelsSvc.listModels(sess, region);
125
+ for (const m of list) {
126
+ if (!m?.id || seen.has(m.id)) continue;
127
+ seen.add(m.id);
128
+ out.push(m);
129
+ }
130
+ } catch {}
131
+ }
132
+ return out;
133
+ }
134
+
135
+ async function preheat() {
136
+ try {
137
+ const list = await listModels();
138
+ return list.length ? { ok: true } : { ok: false, error: "no account" };
139
+ } catch (e) { return { ok: false, error: String(e?.message || e).slice(0, 120) }; }
140
+ }
98
141
 
99
142
  async function close() {}
100
143
  async function chatWithKeys(body, keysOverride) {
@@ -1,5 +1,5 @@
1
- // qoder 模型服务(转译 bridge.go ListModels/extractModels):原生 COSY 签名拉 /model/list,
2
- // 10min 缓存;暴露全部 15 个 key(含 enable=false 的展示,allowlist 由上层过滤)。
1
+ // qoder 模型服务(转译 bridge.go ListModels/extractModels):原生 COSY 签名拉 /model/list,
2
+ // 10min 缓存(按 region 分桶);只暴露 enable=true 的可用模型(enable=false 调对话必挂,不展示)。
3
3
  import { compatFetch, timeoutSignal } from "../../compat.js";
4
4
  import { getEndpoints, normalizeRegion } from "./constants.js";
5
5
  import { buildCosyHeaders, pathSigFrom } from "./session.js";
@@ -29,34 +29,45 @@ export function extractModels(chatList) {
29
29
  export function createModelsService({ id = "qoder", fetchImpl, clock = Date.now } = {}) {
30
30
  let cache = null, fetchedAt = 0;
31
31
 
32
- async function listModels(sess, region = "global") {
33
- const normRegion = normalizeRegion(region);
34
- const now = clock();
35
- if (cache && now - fetchedAt < CACHE_TTL_MS) return cache;
36
- const ep = getEndpoints(normRegion);
37
- const url = ep.modelListURL;
38
- const headers = buildCosyHeaders(sess, pathSigFrom(url), "", "application/json");
39
- const res = await fetchImpl(url, { headers, signal: timeoutSignal(15000) });
40
- if (!res.ok) throw new Error(`models http ${res.status}`);
41
- const j = await res.json().catch(() => ({}));
42
- const models = extractModels(j.chat) || extractModels(j.assistant);
43
- if (!models.length) throw new Error("models list empty");
44
- cache = models.map((m) => ({
45
- id: joinModelId(id, m.id),
46
- object: "model",
47
- created: CREATED,
48
- owned_by: OWNER,
49
- name: m.name,
50
- context_length: m.contextLength,
51
- enable: m.enable,
52
- is_reasoning: m.isReasoning,
53
- price_factor: m.priceFactor,
54
- }));
55
- fetchedAt = now;
56
- return cache;
57
- }
58
-
59
- function clearCache() { cache = null; fetchedAt = 0; }
60
-
61
- return { listModels, clearCache, _getCache: () => cache };
62
- }
32
+ // 按 region 分桶缓存:同一进程可能有 cn + global 双号,混用单桶会串区(cn 14 个/global 15 个各不同)。
33
+ const cacheByRegion = new Map();
34
+ async function listModels(sess, region = "global") {
35
+ const normRegion = normalizeRegion(region);
36
+ const now = clock();
37
+ const hit = cacheByRegion.get(normRegion);
38
+ if (hit && now - hit.at < CACHE_TTL_MS) return hit.list;
39
+ const ep = getEndpoints(normRegion);
40
+ const url = ep.modelListURL;
41
+ const headers = buildCosyHeaders(sess, pathSigFrom(url), "", "application/json");
42
+ const res = await fetchImpl(url, { headers, signal: timeoutSignal(15000) });
43
+ if (!res.ok) throw new Error(`models http ${res.status}`);
44
+ const j = await res.json().catch(() => ({}));
45
+ // extractModels 恒返回数组(空数组亦 truthy),必须按长度选 chat/assistant 分支
46
+ const chatModels = extractModels(j.chat);
47
+ const models = chatModels.length ? chatModels : extractModels(j.assistant);
48
+ if (!models.length) throw new Error("models list empty");
49
+ // 只留可用:enable=false 的 key(如 global 的 auto/ultimate,cn 的绝大多数)展示即误导
50
+ const usable = models.filter((m) => m.enable);
51
+ if (!usable.length) throw new Error("models list empty (no enabled models)");
52
+ const list = usable.map((m) => ({
53
+ id: joinModelId(id, m.id),
54
+ object: "model",
55
+ created: CREATED,
56
+ owned_by: OWNER,
57
+ name: m.name,
58
+ context_length: m.contextLength,
59
+ enable: true,
60
+ is_reasoning: m.isReasoning,
61
+ price_factor: m.priceFactor,
62
+ }));
63
+ cacheByRegion.set(normRegion, { list, at: now });
64
+ // 兼容旧单测 _getCache:默认读 global 桶
65
+ cache = normRegion === "global" ? list : cache;
66
+ fetchedAt = normRegion === "global" ? now : fetchedAt;
67
+ return list;
68
+ }
69
+
70
+ function clearCache() { cache = null; fetchedAt = 0; cacheByRegion.clear(); }
71
+
72
+ return { listModels, clearCache, _getCache: () => cache ?? cacheByRegion.get("global")?.list ?? null };
73
+ }
@@ -117,6 +117,26 @@ export function createRelayPipeline({
117
117
  detail: out.detail ?? null,
118
118
  });
119
119
 
120
+ // 5a0. 空转 200:流正常结束但零正文零工具调用 → 客户端会报 EMPTY_MODEL_RESPONSE
121
+ // ("The model ended its turn without producing any output");转 failover 而不是
122
+ // 把空轮递给客户端。chatShaped 是前置证据:只有看得出是 chat 轮才判空,
123
+ // 非 chat SSE/无 choices JSON 透传是正式契约(chat-route 单测锁死),一律放行。
124
+ // 工具轮豁免:tool_calls 无正文是合法 agent 形态;
125
+ // finish=tool_calls/function_call 兜底豁免(防计数漏检误杀);下游已断开不重试(写给谁看)。
126
+ // 对标 dsh-cline-pass 的 EMPTY_RESPONSE 语义;不记 auto 冷却(空转≠模型坏,重试多半能好)。
127
+ const _d = out.detail || {};
128
+ const _emptyTurn = out.status === 200 && !out.timedOut && !out.interrupted && !_d.downstreamClosed &&
129
+ _d.chatShaped === true &&
130
+ (Number(_d.chars) || 0) === 0 && (Number(_d.toolCalls) || 0) === 0 &&
131
+ !["tool_calls", "function_call"].includes(_d.sawFinishReason);
132
+ if (_emptyTurn) {
133
+ const _why = _d.sawFinishReason ? ` (finish_reason=${_d.sawFinishReason})` : " (no content, no tool calls)";
134
+ try { _logError(actual, 502, `empty turn${_why}`); } catch {}
135
+ _evt("upstream-error", { reqId, model: actual, status: 502, message: "empty turn", timing: null });
136
+ _evt("fallback", { reqId, from: actual, to: null, reason: "empty turn" });
137
+ return { handled: false, upRes: null, lastErr: { model: actual, upstream: null, status: 502, message: `EMPTY_MODEL_RESPONSE: upstream returned 200 with no content${_why} — retry or rephrase` } };
138
+ }
139
+
120
140
  // 5a. 首块超时未写字节 → 回退(显式 timedOut 字段,status 只是 HTTP 语义展示)
121
141
  if (out.timedOut === true) {
122
142
  const why = out.detail?.upstreamError ? ` (upstream read error: ${out.detail.upstreamError})` : "";
@@ -114,6 +114,8 @@ export async function relay(res, upRes, body, { onFirstChunk, onDownstreamAbort,
114
114
  downstreamClosed: false,
115
115
  usage: null,
116
116
  chars: 0,
117
+ toolCalls: 0,
118
+ chatShaped: false,
117
119
  recoveries: 0,
118
120
  };
119
121
  let prevChunkAt = t0;
@@ -209,6 +211,12 @@ export async function relay(res, upRes, body, { onFirstChunk, onDownstreamAbort,
209
211
  if (txt.includes("[DONE]")) detail.sawDone = true;
210
212
  const m = txt.match(/"finish_reason"\s*:\s*"([^"]+)"/);
211
213
  if (m) detail.sawFinishReason = m[1];
214
+ // chat 形状证据:只有看得出是 chat 轮才配判空(非 chat SSE/JSON 透传是正式契约,不得误伤)
215
+ if (!detail.chatShaped && (txt.includes('"choices"') || txt.includes('"delta"') || txt.includes('"finish_reason"') || txt.includes('"usage"') || txt.includes('"prompt_tokens"') || txt.includes("[DONE]"))) detail.chatShaped = true;
216
+ // 工具调用计数(空数组不算):tool_calls 无正文是合法 agent 轮,
217
+ // 空转闸门必须豁免它,否则所有工具轮都会被误判为空轮——见 relay-pipeline 4b。
218
+ const tc = txt.match(/"tool_calls"\s*:\s*\[\s*\{/g);
219
+ if (tc) detail.toolCalls = (detail.toolCalls || 0) + tc.length;
212
220
  // 尝试提取 usage(流式末帧):口径收口到 metrics.js,与未流式分支共用
213
221
  if (txt.includes("\"usage\"") || txt.includes("\"prompt_tokens\"")) {
214
222
  try {
@@ -326,6 +334,10 @@ export async function relay(res, upRes, body, { onFirstChunk, onDownstreamAbort,
326
334
  if (u) detail.usage = u;
327
335
  if (parsed.choices?.[0]?.message?.content) detail.chars = String(parsed.choices[0].message.content).length;
328
336
  else if (parsed.choices?.[0]?.text) detail.chars = String(parsed.choices[0].text).length;
337
+ // 非流式同样只判 chat 形状:无 choices 的任意 JSON 是透传契约(chat-route 单测锁死),不得判空
338
+ if (Array.isArray(parsed?.choices)) detail.chatShaped = true;
339
+ const _tcList = parsed.choices?.[0]?.message?.tool_calls;
340
+ if (Array.isArray(_tcList) && _tcList.length) detail.toolCalls = _tcList.length;
329
341
  const enriched = enrichNonStreamJson(parsed, fallback);
330
342
  json(res, upRes.status, enriched);
331
343
  } catch {