mslxdff 0.1.160 → 0.1.161

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (117) hide show
  1. package/bin/mslxdff.js +4 -4
  2. package/docs/cli_help_mini.md +135 -0
  3. package/package.json +1 -1
  4. package/src/auto.js +254 -254
  5. package/src/autostart.js +3 -0
  6. package/src/bench/cline-bench.js +42 -42
  7. package/src/bench/probe.js +70 -70
  8. package/src/bench/report.js +162 -162
  9. package/src/bench/runner.js +77 -77
  10. package/src/bench/via-probe.js +124 -124
  11. package/src/bench/via-routes.js +87 -87
  12. package/src/bench/workbuddy-bench.js +54 -54
  13. package/src/chat/engine.js +160 -160
  14. package/src/chat/gateway.js +163 -163
  15. package/src/chat/orchestrator.js +234 -234
  16. package/src/chat/prompt.js +70 -70
  17. package/src/chat/repl.js +88 -88
  18. package/src/chat/terminal.js +135 -135
  19. package/src/chat/tools.js +306 -306
  20. package/src/chat-pipeline/empty-turn.js +119 -0
  21. package/src/chat-pipeline/index.js +126 -123
  22. package/src/chat-pipeline/policy.js +76 -76
  23. package/src/chat-pipeline/serial-trial.js +53 -19
  24. package/src/cli/commands/daemon.js +4 -4
  25. package/src/cli/commands/group.js +249 -249
  26. package/src/cli/commands/model/list-providers.js +1 -1
  27. package/src/cli/commands/model/picks.js +50 -50
  28. package/src/cli/commands/provider/bench-via.js +247 -247
  29. package/src/cli/commands/provider/bench.js +141 -141
  30. package/src/cli/commands/provider/globalqwenwork-login.js +126 -0
  31. package/src/cli/commands/provider/index.js +126 -124
  32. package/src/cli/commands/provider/models.js +142 -139
  33. package/src/cli/commands/provider/qwenwork-login.js +119 -119
  34. package/src/cli/commands/sync.js +232 -232
  35. package/src/cli/commands/system.js +2 -2
  36. package/src/cli/format.js +6 -0
  37. package/src/cli/policy.js +2 -2
  38. package/src/cli/provider-row.js +2 -2
  39. package/src/cli/status.js +279 -279
  40. package/src/daemon.js +101 -96
  41. package/src/logs.js +14 -1
  42. package/src/model-capabilities/enrich.js +86 -86
  43. package/src/model-capabilities/index.js +183 -183
  44. package/src/model-capabilities/parse.js +70 -70
  45. package/src/model-trace.js +5 -2
  46. package/src/models.js +225 -225
  47. package/src/providers/AGENTS.md +46 -0
  48. package/src/providers/classify.js +1 -1
  49. package/src/providers/cline/auth.js +228 -228
  50. package/src/providers/cline/chat.js +307 -307
  51. package/src/providers/cline.js +2 -2
  52. package/src/providers/globalqwenwork/account-store.js +141 -0
  53. package/src/providers/globalqwenwork/constants.js +73 -0
  54. package/src/providers/globalqwenwork/cosy.js +123 -0
  55. package/src/providers/globalqwenwork/crypto.js +220 -0
  56. package/src/providers/globalqwenwork/http.js +22 -0
  57. package/src/providers/globalqwenwork/index.js +331 -0
  58. package/src/providers/globalqwenwork/payload.js +145 -0
  59. package/src/providers/globalqwenwork/rsa.js +56 -0
  60. package/src/providers/globalqwenwork/sse.js +270 -0
  61. package/src/providers/globalqwenwork/stream.js +132 -0
  62. package/src/providers/globalqwenwork/upstream.js +122 -0
  63. package/src/providers/globalqwenwork.js +1 -0
  64. package/src/providers/keyring.js +60 -60
  65. package/src/providers/qoder/chat.js +183 -183
  66. package/src/providers/qoder/index.js +230 -230
  67. package/src/providers/qoder/sse.js +103 -103
  68. package/src/providers/qwenwork/account-store.js +133 -133
  69. package/src/providers/qwenwork/constants.js +67 -67
  70. package/src/providers/qwenwork/cosy.js +120 -120
  71. package/src/providers/qwenwork/crypto.js +218 -218
  72. package/src/providers/qwenwork/http.js +20 -20
  73. package/src/providers/qwenwork/index.js +327 -327
  74. package/src/providers/qwenwork/payload.js +142 -142
  75. package/src/providers/qwenwork/rsa.js +54 -54
  76. package/src/providers/qwenwork/sse.js +268 -268
  77. package/src/providers/qwenwork/stream.js +130 -130
  78. package/src/providers/qwenwork/upstream.js +120 -120
  79. package/src/providers/qwenwork.js +1 -1
  80. package/src/providers/registry.js +74 -66
  81. package/src/providers/workbuddy/chat.js +248 -248
  82. package/src/providers/workbuddy/reshape.js +152 -152
  83. package/src/providers/workbuddy.js +2 -2
  84. package/src/providers/zcode/sse.js +19 -2
  85. package/src/reasoning.js +32 -32
  86. package/src/routes/AGENTS.md +37 -0
  87. package/src/routes/chat/exhausted-handler.js +23 -8
  88. package/src/routes/chat/gateway.js +46 -46
  89. package/src/routes/chat/local-handler.js +3 -1
  90. package/src/routes/chat/relay-pipeline.js +276 -264
  91. package/src/routes/chat/via-route-handler.js +146 -146
  92. package/src/routes/hedge.js +255 -255
  93. package/src/routes/helpers.js +20 -4
  94. package/src/routes/models-route.js +167 -167
  95. package/src/routes/peers.js +273 -273
  96. package/src/routes/stream-hold.js +138 -0
  97. package/src/routes/stream-scan.js +192 -0
  98. package/src/routes/stream.js +386 -393
  99. package/src/runtime/bootstrap.js +45 -45
  100. package/src/runtime/lifecycle-forensics.js +116 -0
  101. package/src/runtime/lifecycle-log.js +23 -0
  102. package/src/runtime/provider-gate.js +34 -33
  103. package/src/runtime/providers-setup.js +165 -165
  104. package/src/server.js +64 -64
  105. package/src/state/schemas/allowlist.js +92 -92
  106. package/src/sync-opencode.js +280 -280
  107. package/src/talk-log.js +226 -0
  108. package/src/timeline.js +5 -2
  109. package/src/transport/index.js +244 -244
  110. package/src/transport/pool.js +56 -56
  111. package/src/transport/retry.js +24 -24
  112. package/src/transport/sse.js +93 -93
  113. package/src/upstream-probe/display.js +52 -52
  114. package/src/upstream-probe/probe.js +49 -49
  115. package/src/upstream-probe/rotate.js +110 -110
  116. package/src/upstream-probe/start.js +45 -45
  117. package/src/upstream.js +289 -289
@@ -1,234 +1,234 @@
1
- import { performance as nodePerf } from "node:perf_hooks";
2
-
3
- const ORIG_PREFERRED = "mimo-v2.5-free";
4
- const ORIG_FALLBACK = "big-pickle";
5
-
6
- /**
7
- * 编排深模块:mimo → pickle → gateway 三级降级 + 800ms 对冲
8
- * 注入化:chatOnce / chatViaGateway / cooling / config / env / performance
9
- */
10
- export function createOrchestrator({
11
- chatOnce,
12
- chatViaGateway,
13
- cooling,
14
- config = {},
15
- env = process.env,
16
- performance: perf = nodePerf,
17
- chatWithFallbackImpl,
18
- } = {}) {
19
- const CHAT_PREFERRED = config.CHAT_PREFERRED || ORIG_PREFERRED;
20
- const CHAT_FALLBACK = config.CHAT_FALLBACK || ORIG_FALLBACK;
21
- const CHAT_GATEWAY_TIMEOUT_MS = config.CHAT_GATEWAY_TIMEOUT_MS || 25000;
22
- const CHAT_COOLDOWN_MS = 10 * 60 * 1000;
23
- const CHAT_SLOW_COOLDOWN_MS = 10 * 60 * 1000;
24
-
25
- const _chatOnce = chatOnce || (async () => ({ ok: false, error: "no chatOnce", status: 500 }));
26
- const _gateway = chatViaGateway || (async () => ({ ok: false, error: "no gateway", status: 500 }));
27
- const _cooling = cooling || {
28
- isCooling: async () => false,
29
- recordError: async () => {},
30
- recordOk: async () => {},
31
- };
32
-
33
- async function isCoolingAsync(id) {
34
- try { return await _cooling.isCooling(id); } catch { return false; }
35
- }
36
- async function recordChatError(id, status, opts) {
37
- try { await _cooling.recordError(id, status, opts); } catch {}
38
- }
39
- async function recordChatOk(id, latencyMs) {
40
- try { await _cooling.recordOk(id, latencyMs); } catch {}
41
- }
42
-
43
- async function safeChatOnce(opts, model) {
44
- try {
45
- const r = await _chatOnce({ ...opts, model });
46
- return r;
47
- } catch (err) {
48
- const msg = String(err?.message || err).slice(0, 800);
49
- const status = err?._t ? 502 : 502;
50
- return { ok: false, error: msg, status, _thrown: err };
51
- }
52
- }
53
-
54
- async function chatWithFallback(opts) {
55
- if (chatWithFallbackImpl) return chatWithFallbackImpl(opts);
56
- // 严格模式:显式指定了具体模型(非 auto/空)→ 只用该模型,失败即报错,绝不降级到其他模型
57
- const pinned = String(opts?.model || "").trim();
58
- if (pinned && pinned.toLowerCase() !== "auto") {
59
- const r = await safeChatOnce(opts, pinned);
60
- if (r.ok) return { ...r, model: pinned };
61
- return { ok: false, error: `指定模型 ${pinned} 失败:${r.error}(已锁定不自动换模型;如需自动择优请输入 /model auto)`, status: r.status || 502, pinnedModel: pinned };
62
- }
63
- const TRACE = env.MSLXDFF_CHAT_TRACE !== "0";
64
- const HEDGE_MS = (() => {
65
- const v = Number(env.MSLXDFF_HEDGE_DELAY_MS);
66
- return Number.isInteger(v) && v >= 0 ? v : 800;
67
- })();
68
- const t0 = TRACE ? perf.now() : 0;
69
-
70
- const firstCooling = await isCoolingAsync(CHAT_PREFERRED);
71
- let first;
72
- let firstMs = 0;
73
- if (firstCooling) {
74
- if (TRACE) console.log(`\x1b[90m· [LLM] ${CHAT_PREFERRED} 跳过(冷却 10min)· 额度用完,准备走网关 auto(将尝试其他可用模型)\x1b[0m`);
75
- first = { ok: false, error: "skip cooling", status: 429 };
76
- } else {
77
- const t = perf.now();
78
- first = await safeChatOnce(opts, CHAT_PREFERRED);
79
- firstMs = Math.round(perf.now() - t);
80
- if (TRACE) console.log(`\x1b[90m· [LLM] ${CHAT_PREFERRED} ${first.ok ? "OK" : "FAIL"} · ${Math.round(perf.now() - t0)}ms${first.ok ? "" : ` · ${String(first.error).slice(0, 80)}`}\x1b[0m`);
81
- if (first.ok) {
82
- await recordChatOk(CHAT_PREFERRED, firstMs);
83
- return { ...first, model: CHAT_PREFERRED };
84
- } else {
85
- const slow = firstMs > 20000;
86
- await recordChatError(CHAT_PREFERRED, first.status, { slow, latencyMs: firstMs });
87
- }
88
- }
89
-
90
- if (firstCooling) {
91
- if (TRACE) console.log(`\x1b[90m· [LLM] ${CHAT_PREFERRED} 冷却中(${CHAT_COOLDOWN_MS / 60000}min)· 直接走网关 auto(:8989),按 auto 择优尝试其他可用模型,跳过 ${CHAT_FALLBACK} 直连\x1b[0m`);
92
- const t2 = TRACE ? perf.now() : 0;
93
- const third = await _gateway(opts);
94
- if (TRACE) console.log(`\x1b[90m· [LLM] gateway auto ${third.ok ? "OK" : "FAIL"} · ${Math.round(perf.now() - t2)}ms · 总 ${Math.round(perf.now() - t0)}ms (gateway-fallback)\x1b[0m`);
95
- if (third.ok) return { ...third, model: third.model || "auto", fallbackGateway: true, fallback: true, firstError: first.error, secondError: "skip big-pickle (mimo cooling)", viaGateway: true };
96
- const isGwDown = third.code === "GATEWAY_NOT_RUNNING" || /本地(网关未运行|服务没有启动)/.test(String(third.error || ""));
97
- if (isGwDown) {
98
- return { ok: false, error: `${CHAT_PREFERRED} 冷却跳过(额度用完 10min);${third.error} —— 本次 auto 未能尝试任何其他模型,因网关未就绪(3ms 失败即证明未触达上游,请先启动网关再重试)`, status: 502, code: "GATEWAY_NOT_RUNNING" };
99
- }
100
- return { ok: false, error: `${CHAT_PREFERRED} 冷却跳过(额度用完);网关 auto 已尝试其他可用模型但均失败:${third.error}(总 ${Math.round(perf.now() - t0)}ms)`, status: third.status || 429 };
101
- }
102
-
103
- const secondCooling = await isCoolingAsync(CHAT_FALLBACK);
104
- const doGateway = () => _gateway(opts);
105
- const doSecond = async () => {
106
- const t = perf.now();
107
- const r = await safeChatOnce(opts, CHAT_FALLBACK);
108
- const ms = Math.round(perf.now() - t);
109
- if (r.ok) await recordChatOk(CHAT_FALLBACK, ms);
110
- else await recordChatError(CHAT_FALLBACK, r.status, { slow: ms > 20000, latencyMs: ms });
111
- return { res: r, ms };
112
- };
113
-
114
- if (secondCooling) {
115
- if (TRACE) console.log(`\x1b[90m· [LLM] ${CHAT_FALLBACK} 跳过(冷却中)· 直接走网关 auto(将尝试其他可用模型)\x1b[0m`);
116
- const t2 = perf.now();
117
- const third = await doGateway();
118
- if (TRACE) console.log(`\x1b[90m· [LLM] gateway auto ${third.ok ? "OK" : "FAIL"} · ${Math.round(perf.now() - t2)}ms · 总 ${Math.round(perf.now() - t0)}ms (gateway-fallback)\x1b[0m`);
119
- if (third.ok) return { ...third, model: third.model || "auto", fallbackGateway: true, fallback: true, firstError: first.error, secondError: "skip cooling", viaGateway: true };
120
- const isGwDown2 = third.code === "GATEWAY_NOT_RUNNING" || /本地(网关未运行|服务没有启动)/.test(String(third.error || ""));
121
- if (isGwDown2) return { ok: false, error: `${CHAT_PREFERRED} 失败:${first.error};${CHAT_FALLBACK} 冷却跳过;${third.error} —— auto 未能尝试其他模型(网关未就绪)`, status: 502, code: "GATEWAY_NOT_RUNNING" };
122
- return { ok: false, error: `${CHAT_PREFERRED} 失败:${first.error};${CHAT_FALLBACK} 冷却跳过;网关 auto 已尝试其他模型但均失败:${third.error}(总 ${Math.round(perf.now() - t0)}ms)`, status: third.status || first.status };
123
- }
124
-
125
- if (TRACE) console.log(`\x1b[90m· [LLM] ${CHAT_PREFERRED} 失败,${CHAT_FALLBACK} + gateway 对冲中(${HEDGE_MS}ms)\x1b[0m`);
126
- const t1 = TRACE ? perf.now() : 0;
127
- let secondRes = null;
128
- let secondMs = 0;
129
- let gatewayRes = null;
130
-
131
- const secondPromise = doSecond().then(({ res, ms }) => {
132
- secondRes = res; secondMs = ms;
133
- if (TRACE) console.log(`\x1b[90m· [LLM] ${CHAT_FALLBACK} ${res.ok ? "OK" : "FAIL"} · ${Math.round(perf.now() - t1)}ms · 总 ${Math.round(perf.now() - t0)}ms (hedge)\x1b[0m`);
134
- return res;
135
- });
136
-
137
- let gatewayPromise = null;
138
- const gatewayDelay = HEDGE_MS > 0 ? HEDGE_MS : 0;
139
- if (gatewayDelay > 0) {
140
- gatewayPromise = new Promise((resolve) => {
141
- setTimeout(async () => {
142
- const r = await doGateway();
143
- gatewayRes = r;
144
- resolve(r);
145
- }, gatewayDelay);
146
- });
147
- } else {
148
- gatewayPromise = doGateway().then((r) => { gatewayRes = r; return r; });
149
- }
150
-
151
- const raceFirstOk = async () => {
152
- const secondOrTimeout = await Promise.race([
153
- secondPromise.then((r) => ({ kind: "second", r })),
154
- new Promise((resolve) => setTimeout(() => resolve({ kind: "timeout" }), gatewayDelay)),
155
- ]);
156
- if (secondOrTimeout.kind === "second" && secondOrTimeout.r?.ok) {
157
- return { ok: true, res: secondOrTimeout.r, model: CHAT_FALLBACK, fallback: true, firstError: first.error };
158
- }
159
- if (!gatewayPromise || secondOrTimeout.kind === "timeout") {
160
- const immediate = doGateway().then((r) => { gatewayRes = r; return r; });
161
- if (gatewayPromise) {
162
- gatewayRes = await Promise.race([gatewayPromise, immediate]);
163
- } else {
164
- gatewayRes = await immediate;
165
- }
166
- } else {
167
- gatewayRes = await gatewayPromise;
168
- }
169
- if (gatewayRes?.ok) return { ok: true, res: gatewayRes, model: gatewayRes.model || "auto", fallbackGateway: true, fallback: true, firstError: first.error, secondError: secondRes?.error, viaGateway: true };
170
- if (secondRes?.ok) return { ok: true, res: secondRes, model: CHAT_FALLBACK, fallback: true, firstError: first.error };
171
- return { ok: false, gatewayRes, secondRes };
172
- };
173
-
174
- if (gatewayDelay === 0) {
175
- const [sRes, gRes] = await Promise.all([
176
- secondPromise.catch((e) => ({ ok: false, error: String(e), status: 502 })),
177
- doGateway().catch((e) => ({ ok: false, error: String(e), status: 502 })),
178
- ]);
179
- secondRes = sRes; gatewayRes = gRes;
180
- if (sRes?.ok) return { ...sRes, model: CHAT_FALLBACK, fallback: true, firstError: first.error };
181
- if (gRes?.ok) return { ...gRes, model: gRes.model || "auto", fallbackGateway: true, fallback: true, firstError: first.error, secondError: sRes?.error, viaGateway: true };
182
- const gwDown0 = gRes?.code === "GATEWAY_NOT_RUNNING" || /本地(网关未运行|服务没有启动)/.test(String(gRes?.error || ""));
183
- if (gwDown0) return { ok: false, error: `${CHAT_PREFERRED} 失败:${first.error};${CHAT_FALLBACK} 失败:${sRes?.error};${gRes?.error} —— gateway 侧 auto 未能尝试其他模型(网关未就绪)`, status: 502, code: "GATEWAY_NOT_RUNNING" };
184
- return { ok: false, error: `${CHAT_PREFERRED} 失败:${first.error};${CHAT_FALLBACK} 失败:${sRes?.error};网关 auto 已尝试其他模型但均失败:${gRes?.error}(总 ${Math.round(perf.now() - t0)}ms)`, status: gRes?.status || sRes?.status || first.status };
185
- }
186
-
187
- const raced = await raceFirstOk();
188
- if (raced.ok) {
189
- const r = raced.res;
190
- return { ...r, model: raced.model, fallback: raced.fallback, fallbackGateway: raced.fallbackGateway, firstError: raced.firstError, secondError: raced.secondError, viaGateway: raced.viaGateway };
191
- }
192
- if (!secondRes) {
193
- try { secondRes = await secondPromise; } catch (e) { secondRes = { ok: false, error: String(e), status: 502 }; }
194
- }
195
- if (secondRes?.ok) return { ...secondRes, model: CHAT_FALLBACK, fallback: true, firstError: first.error };
196
- if (!gatewayRes) {
197
- try { gatewayRes = await (gatewayPromise || doGateway()); } catch (e) { gatewayRes = { ok: false, error: String(e), status: 502 }; }
198
- }
199
- if (gatewayRes?.ok) return { ...gatewayRes, model: gatewayRes.model || "auto", fallbackGateway: true, fallback: true, firstError: first.error, secondError: secondRes?.error, viaGateway: true };
200
- const gwDown = gatewayRes?.code === "GATEWAY_NOT_RUNNING" || /本地(网关未运行|服务没有启动)/.test(String(gatewayRes?.error || ""));
201
- if (gwDown) return { ok: false, error: `${CHAT_PREFERRED} 失败:${first.error};${CHAT_FALLBACK} 失败:${secondRes?.error};${gatewayRes?.error} —— gateway 侧 auto 未能尝试其他模型(网关未就绪)`, status: 502, code: "GATEWAY_NOT_RUNNING" };
202
- return { ok: false, error: `${CHAT_PREFERRED} 失败:${first.error};${CHAT_FALLBACK} 失败:${secondRes?.error};网关 auto 已尝试其他模型但均失败:${gatewayRes?.error}(总 ${Math.round(perf.now() - t0)}ms)`, status: gatewayRes?.status || secondRes?.status || first.status };
203
- }
204
-
205
- async function summarizeHistory(messages) {
206
- const prompt = [
207
- { role: "system", content: "你是对话压缩助手,把以下历史对话压缩成 800 字以内的中文摘要,保留关键操作与结果、用户的偏好与待办、模型设置与群组操作及时间线,不要遗漏重要细节。" },
208
- { role: "user", content: messages.map((m) => `${m.role}: ${m.content || JSON.stringify(m.tool_calls || "")}`).join("\n").slice(0, 90000) },
209
- ];
210
- const r = await chatWithFallback({ messages: prompt });
211
- if (!r.ok) return null;
212
- const txt = String(r.message?.content || "").trim();
213
- return txt ? `【历史摘要】${txt}` : null;
214
- }
215
-
216
- // 兼容测试的额外覆盖参数
217
- if (chatWithFallbackImpl) {
218
- const orig = chatWithFallback;
219
- // 允许测试覆盖 summarize 内部的 chatWithFallback
220
- const wrapped = async (opts) => chatWithFallbackImpl(opts);
221
- return { chatWithFallback: wrapped, summarizeHistory: async (msgs) => {
222
- const prompt = [
223
- { role: "system", content: "你是对话压缩助手,把以下历史对话压缩成 800 字以内的中文摘要,保留关键操作与结果、用户的偏好与待办、模型设置与群组操作及时间线,不要遗漏重要细节。" },
224
- { role: "user", content: msgs.map((m) => `${m.role}: ${m.content || JSON.stringify(m.tool_calls || "")}`).join("\n").slice(0, 90000) },
225
- ];
226
- const r = await wrapped({ messages: prompt });
227
- if (!r.ok) return null;
228
- const txt = String(r.message?.content || "").trim();
229
- return txt ? `【历史摘要】${txt}` : null;
230
- } };
231
- }
232
-
233
- return { chatWithFallback, summarizeHistory };
234
- }
1
+ import { performance as nodePerf } from "node:perf_hooks";
2
+
3
+ const ORIG_PREFERRED = "mimo-v2.5-free";
4
+ const ORIG_FALLBACK = "big-pickle";
5
+
6
+ /**
7
+ * 编排深模块:mimo → pickle → gateway 三级降级 + 800ms 对冲
8
+ * 注入化:chatOnce / chatViaGateway / cooling / config / env / performance
9
+ */
10
+ export function createOrchestrator({
11
+ chatOnce,
12
+ chatViaGateway,
13
+ cooling,
14
+ config = {},
15
+ env = process.env,
16
+ performance: perf = nodePerf,
17
+ chatWithFallbackImpl,
18
+ } = {}) {
19
+ const CHAT_PREFERRED = config.CHAT_PREFERRED || ORIG_PREFERRED;
20
+ const CHAT_FALLBACK = config.CHAT_FALLBACK || ORIG_FALLBACK;
21
+ const CHAT_GATEWAY_TIMEOUT_MS = config.CHAT_GATEWAY_TIMEOUT_MS || 25000;
22
+ const CHAT_COOLDOWN_MS = 10 * 60 * 1000;
23
+ const CHAT_SLOW_COOLDOWN_MS = 10 * 60 * 1000;
24
+
25
+ const _chatOnce = chatOnce || (async () => ({ ok: false, error: "no chatOnce", status: 500 }));
26
+ const _gateway = chatViaGateway || (async () => ({ ok: false, error: "no gateway", status: 500 }));
27
+ const _cooling = cooling || {
28
+ isCooling: async () => false,
29
+ recordError: async () => {},
30
+ recordOk: async () => {},
31
+ };
32
+
33
+ async function isCoolingAsync(id) {
34
+ try { return await _cooling.isCooling(id); } catch { return false; }
35
+ }
36
+ async function recordChatError(id, status, opts) {
37
+ try { await _cooling.recordError(id, status, opts); } catch {}
38
+ }
39
+ async function recordChatOk(id, latencyMs) {
40
+ try { await _cooling.recordOk(id, latencyMs); } catch {}
41
+ }
42
+
43
+ async function safeChatOnce(opts, model) {
44
+ try {
45
+ const r = await _chatOnce({ ...opts, model });
46
+ return r;
47
+ } catch (err) {
48
+ const msg = String(err?.message || err).slice(0, 800);
49
+ const status = err?._t ? 502 : 502;
50
+ return { ok: false, error: msg, status, _thrown: err };
51
+ }
52
+ }
53
+
54
+ async function chatWithFallback(opts) {
55
+ if (chatWithFallbackImpl) return chatWithFallbackImpl(opts);
56
+ // 严格模式:显式指定了具体模型(非 auto/空)→ 只用该模型,失败即报错,绝不降级到其他模型
57
+ const pinned = String(opts?.model || "").trim();
58
+ if (pinned && pinned.toLowerCase() !== "auto") {
59
+ const r = await safeChatOnce(opts, pinned);
60
+ if (r.ok) return { ...r, model: pinned };
61
+ return { ok: false, error: `指定模型 ${pinned} 失败:${r.error}(已锁定不自动换模型;如需自动择优请输入 /model auto)`, status: r.status || 502, pinnedModel: pinned };
62
+ }
63
+ const TRACE = env.MSLXDFF_CHAT_TRACE !== "0";
64
+ const HEDGE_MS = (() => {
65
+ const v = Number(env.MSLXDFF_HEDGE_DELAY_MS);
66
+ return Number.isInteger(v) && v >= 0 ? v : 800;
67
+ })();
68
+ const t0 = TRACE ? perf.now() : 0;
69
+
70
+ const firstCooling = await isCoolingAsync(CHAT_PREFERRED);
71
+ let first;
72
+ let firstMs = 0;
73
+ if (firstCooling) {
74
+ if (TRACE) console.log(`\x1b[90m· [LLM] ${CHAT_PREFERRED} 跳过(冷却 10min)· 额度用完,准备走网关 auto(将尝试其他可用模型)\x1b[0m`);
75
+ first = { ok: false, error: "skip cooling", status: 429 };
76
+ } else {
77
+ const t = perf.now();
78
+ first = await safeChatOnce(opts, CHAT_PREFERRED);
79
+ firstMs = Math.round(perf.now() - t);
80
+ if (TRACE) console.log(`\x1b[90m· [LLM] ${CHAT_PREFERRED} ${first.ok ? "OK" : "FAIL"} · ${Math.round(perf.now() - t0)}ms${first.ok ? "" : ` · ${String(first.error).slice(0, 80)}`}\x1b[0m`);
81
+ if (first.ok) {
82
+ await recordChatOk(CHAT_PREFERRED, firstMs);
83
+ return { ...first, model: CHAT_PREFERRED };
84
+ } else {
85
+ const slow = firstMs > 20000;
86
+ await recordChatError(CHAT_PREFERRED, first.status, { slow, latencyMs: firstMs });
87
+ }
88
+ }
89
+
90
+ if (firstCooling) {
91
+ if (TRACE) console.log(`\x1b[90m· [LLM] ${CHAT_PREFERRED} 冷却中(${CHAT_COOLDOWN_MS / 60000}min)· 直接走网关 auto(:8989),按 auto 择优尝试其他可用模型,跳过 ${CHAT_FALLBACK} 直连\x1b[0m`);
92
+ const t2 = TRACE ? perf.now() : 0;
93
+ const third = await _gateway(opts);
94
+ if (TRACE) console.log(`\x1b[90m· [LLM] gateway auto ${third.ok ? "OK" : "FAIL"} · ${Math.round(perf.now() - t2)}ms · 总 ${Math.round(perf.now() - t0)}ms (gateway-fallback)\x1b[0m`);
95
+ if (third.ok) return { ...third, model: third.model || "auto", fallbackGateway: true, fallback: true, firstError: first.error, secondError: "skip big-pickle (mimo cooling)", viaGateway: true };
96
+ const isGwDown = third.code === "GATEWAY_NOT_RUNNING" || /本地(网关未运行|服务没有启动)/.test(String(third.error || ""));
97
+ if (isGwDown) {
98
+ return { ok: false, error: `${CHAT_PREFERRED} 冷却跳过(额度用完 10min);${third.error} —— 本次 auto 未能尝试任何其他模型,因网关未就绪(3ms 失败即证明未触达上游,请先启动网关再重试)`, status: 502, code: "GATEWAY_NOT_RUNNING" };
99
+ }
100
+ return { ok: false, error: `${CHAT_PREFERRED} 冷却跳过(额度用完);网关 auto 已尝试其他可用模型但均失败:${third.error}(总 ${Math.round(perf.now() - t0)}ms)`, status: third.status || 429 };
101
+ }
102
+
103
+ const secondCooling = await isCoolingAsync(CHAT_FALLBACK);
104
+ const doGateway = () => _gateway(opts);
105
+ const doSecond = async () => {
106
+ const t = perf.now();
107
+ const r = await safeChatOnce(opts, CHAT_FALLBACK);
108
+ const ms = Math.round(perf.now() - t);
109
+ if (r.ok) await recordChatOk(CHAT_FALLBACK, ms);
110
+ else await recordChatError(CHAT_FALLBACK, r.status, { slow: ms > 20000, latencyMs: ms });
111
+ return { res: r, ms };
112
+ };
113
+
114
+ if (secondCooling) {
115
+ if (TRACE) console.log(`\x1b[90m· [LLM] ${CHAT_FALLBACK} 跳过(冷却中)· 直接走网关 auto(将尝试其他可用模型)\x1b[0m`);
116
+ const t2 = perf.now();
117
+ const third = await doGateway();
118
+ if (TRACE) console.log(`\x1b[90m· [LLM] gateway auto ${third.ok ? "OK" : "FAIL"} · ${Math.round(perf.now() - t2)}ms · 总 ${Math.round(perf.now() - t0)}ms (gateway-fallback)\x1b[0m`);
119
+ if (third.ok) return { ...third, model: third.model || "auto", fallbackGateway: true, fallback: true, firstError: first.error, secondError: "skip cooling", viaGateway: true };
120
+ const isGwDown2 = third.code === "GATEWAY_NOT_RUNNING" || /本地(网关未运行|服务没有启动)/.test(String(third.error || ""));
121
+ if (isGwDown2) return { ok: false, error: `${CHAT_PREFERRED} 失败:${first.error};${CHAT_FALLBACK} 冷却跳过;${third.error} —— auto 未能尝试其他模型(网关未就绪)`, status: 502, code: "GATEWAY_NOT_RUNNING" };
122
+ return { ok: false, error: `${CHAT_PREFERRED} 失败:${first.error};${CHAT_FALLBACK} 冷却跳过;网关 auto 已尝试其他模型但均失败:${third.error}(总 ${Math.round(perf.now() - t0)}ms)`, status: third.status || first.status };
123
+ }
124
+
125
+ if (TRACE) console.log(`\x1b[90m· [LLM] ${CHAT_PREFERRED} 失败,${CHAT_FALLBACK} + gateway 对冲中(${HEDGE_MS}ms)\x1b[0m`);
126
+ const t1 = TRACE ? perf.now() : 0;
127
+ let secondRes = null;
128
+ let secondMs = 0;
129
+ let gatewayRes = null;
130
+
131
+ const secondPromise = doSecond().then(({ res, ms }) => {
132
+ secondRes = res; secondMs = ms;
133
+ if (TRACE) console.log(`\x1b[90m· [LLM] ${CHAT_FALLBACK} ${res.ok ? "OK" : "FAIL"} · ${Math.round(perf.now() - t1)}ms · 总 ${Math.round(perf.now() - t0)}ms (hedge)\x1b[0m`);
134
+ return res;
135
+ });
136
+
137
+ let gatewayPromise = null;
138
+ const gatewayDelay = HEDGE_MS > 0 ? HEDGE_MS : 0;
139
+ if (gatewayDelay > 0) {
140
+ gatewayPromise = new Promise((resolve) => {
141
+ setTimeout(async () => {
142
+ const r = await doGateway();
143
+ gatewayRes = r;
144
+ resolve(r);
145
+ }, gatewayDelay);
146
+ });
147
+ } else {
148
+ gatewayPromise = doGateway().then((r) => { gatewayRes = r; return r; });
149
+ }
150
+
151
+ const raceFirstOk = async () => {
152
+ const secondOrTimeout = await Promise.race([
153
+ secondPromise.then((r) => ({ kind: "second", r })),
154
+ new Promise((resolve) => setTimeout(() => resolve({ kind: "timeout" }), gatewayDelay)),
155
+ ]);
156
+ if (secondOrTimeout.kind === "second" && secondOrTimeout.r?.ok) {
157
+ return { ok: true, res: secondOrTimeout.r, model: CHAT_FALLBACK, fallback: true, firstError: first.error };
158
+ }
159
+ if (!gatewayPromise || secondOrTimeout.kind === "timeout") {
160
+ const immediate = doGateway().then((r) => { gatewayRes = r; return r; });
161
+ if (gatewayPromise) {
162
+ gatewayRes = await Promise.race([gatewayPromise, immediate]);
163
+ } else {
164
+ gatewayRes = await immediate;
165
+ }
166
+ } else {
167
+ gatewayRes = await gatewayPromise;
168
+ }
169
+ if (gatewayRes?.ok) return { ok: true, res: gatewayRes, model: gatewayRes.model || "auto", fallbackGateway: true, fallback: true, firstError: first.error, secondError: secondRes?.error, viaGateway: true };
170
+ if (secondRes?.ok) return { ok: true, res: secondRes, model: CHAT_FALLBACK, fallback: true, firstError: first.error };
171
+ return { ok: false, gatewayRes, secondRes };
172
+ };
173
+
174
+ if (gatewayDelay === 0) {
175
+ const [sRes, gRes] = await Promise.all([
176
+ secondPromise.catch((e) => ({ ok: false, error: String(e), status: 502 })),
177
+ doGateway().catch((e) => ({ ok: false, error: String(e), status: 502 })),
178
+ ]);
179
+ secondRes = sRes; gatewayRes = gRes;
180
+ if (sRes?.ok) return { ...sRes, model: CHAT_FALLBACK, fallback: true, firstError: first.error };
181
+ if (gRes?.ok) return { ...gRes, model: gRes.model || "auto", fallbackGateway: true, fallback: true, firstError: first.error, secondError: sRes?.error, viaGateway: true };
182
+ const gwDown0 = gRes?.code === "GATEWAY_NOT_RUNNING" || /本地(网关未运行|服务没有启动)/.test(String(gRes?.error || ""));
183
+ if (gwDown0) return { ok: false, error: `${CHAT_PREFERRED} 失败:${first.error};${CHAT_FALLBACK} 失败:${sRes?.error};${gRes?.error} —— gateway 侧 auto 未能尝试其他模型(网关未就绪)`, status: 502, code: "GATEWAY_NOT_RUNNING" };
184
+ return { ok: false, error: `${CHAT_PREFERRED} 失败:${first.error};${CHAT_FALLBACK} 失败:${sRes?.error};网关 auto 已尝试其他模型但均失败:${gRes?.error}(总 ${Math.round(perf.now() - t0)}ms)`, status: gRes?.status || sRes?.status || first.status };
185
+ }
186
+
187
+ const raced = await raceFirstOk();
188
+ if (raced.ok) {
189
+ const r = raced.res;
190
+ return { ...r, model: raced.model, fallback: raced.fallback, fallbackGateway: raced.fallbackGateway, firstError: raced.firstError, secondError: raced.secondError, viaGateway: raced.viaGateway };
191
+ }
192
+ if (!secondRes) {
193
+ try { secondRes = await secondPromise; } catch (e) { secondRes = { ok: false, error: String(e), status: 502 }; }
194
+ }
195
+ if (secondRes?.ok) return { ...secondRes, model: CHAT_FALLBACK, fallback: true, firstError: first.error };
196
+ if (!gatewayRes) {
197
+ try { gatewayRes = await (gatewayPromise || doGateway()); } catch (e) { gatewayRes = { ok: false, error: String(e), status: 502 }; }
198
+ }
199
+ if (gatewayRes?.ok) return { ...gatewayRes, model: gatewayRes.model || "auto", fallbackGateway: true, fallback: true, firstError: first.error, secondError: secondRes?.error, viaGateway: true };
200
+ const gwDown = gatewayRes?.code === "GATEWAY_NOT_RUNNING" || /本地(网关未运行|服务没有启动)/.test(String(gatewayRes?.error || ""));
201
+ if (gwDown) return { ok: false, error: `${CHAT_PREFERRED} 失败:${first.error};${CHAT_FALLBACK} 失败:${secondRes?.error};${gatewayRes?.error} —— gateway 侧 auto 未能尝试其他模型(网关未就绪)`, status: 502, code: "GATEWAY_NOT_RUNNING" };
202
+ return { ok: false, error: `${CHAT_PREFERRED} 失败:${first.error};${CHAT_FALLBACK} 失败:${secondRes?.error};网关 auto 已尝试其他模型但均失败:${gatewayRes?.error}(总 ${Math.round(perf.now() - t0)}ms)`, status: gatewayRes?.status || secondRes?.status || first.status };
203
+ }
204
+
205
+ async function summarizeHistory(messages) {
206
+ const prompt = [
207
+ { role: "system", content: "你是对话压缩助手,把以下历史对话压缩成 800 字以内的中文摘要,保留关键操作与结果、用户的偏好与待办、模型设置与群组操作及时间线,不要遗漏重要细节。" },
208
+ { role: "user", content: messages.map((m) => `${m.role}: ${m.content || JSON.stringify(m.tool_calls || "")}`).join("\n").slice(0, 90000) },
209
+ ];
210
+ const r = await chatWithFallback({ messages: prompt });
211
+ if (!r.ok) return null;
212
+ const txt = String(r.message?.content || "").trim();
213
+ return txt ? `【历史摘要】${txt}` : null;
214
+ }
215
+
216
+ // 兼容测试的额外覆盖参数
217
+ if (chatWithFallbackImpl) {
218
+ const orig = chatWithFallback;
219
+ // 允许测试覆盖 summarize 内部的 chatWithFallback
220
+ const wrapped = async (opts) => chatWithFallbackImpl(opts);
221
+ return { chatWithFallback: wrapped, summarizeHistory: async (msgs) => {
222
+ const prompt = [
223
+ { role: "system", content: "你是对话压缩助手,把以下历史对话压缩成 800 字以内的中文摘要,保留关键操作与结果、用户的偏好与待办、模型设置与群组操作及时间线,不要遗漏重要细节。" },
224
+ { role: "user", content: msgs.map((m) => `${m.role}: ${m.content || JSON.stringify(m.tool_calls || "")}`).join("\n").slice(0, 90000) },
225
+ ];
226
+ const r = await wrapped({ messages: prompt });
227
+ if (!r.ok) return null;
228
+ const txt = String(r.message?.content || "").trim();
229
+ return txt ? `【历史摘要】${txt}` : null;
230
+ } };
231
+ }
232
+
233
+ return { chatWithFallback, summarizeHistory };
234
+ }
@@ -1,70 +1,70 @@
1
- import { readFileSync, existsSync } from "node:fs";
2
- import { join, dirname } from "node:path";
3
- import { fileURLToPath } from "node:url";
4
- import { logDir } from "../logs.js";
5
- import { nowShanghaiYMDHM } from "../time.js";
6
-
7
- const pkgRoot = join(dirname(fileURLToPath(import.meta.url)), "../..");
8
-
9
- function readMini() {
10
- try {
11
- const p = join(pkgRoot, "docs/cli_help_mini.md");
12
- return readFileSync(p, "utf8");
13
- } catch { return ""; }
14
- }
15
-
16
- function readModels() {
17
- try {
18
- const cache = join(logDir(), "models.json");
19
- if (existsSync(cache)) {
20
- const j = JSON.parse(readFileSync(cache, "utf8"));
21
- const ids = (j.data || []).map((m) => m.id).filter(Boolean);
22
- if (ids.length) return ids;
23
- }
24
- } catch {}
25
- // 兜底:硬编码常见 free 模型,供离线时参考(2026-09 实测 /zen/v1/models free 池)
26
- return ["big-pickle", "mimo-v2.5-free", "ling-3.0-flash-fin-free", "deepseek-v4-flash-free", "nemotron-3.5-lightning-free", "nemotron-3-ultra-free", "muse-spark-1.3-contributor-free", "muse-spark-1.2-contributor-free"];
27
- }
28
-
29
- export function buildSystemPrompt({ modelsOverride } = {}) {
30
- const mini = readMini();
31
- const models = modelsOverride || readModels();
32
- const now = nowShanghaiYMDHM();
33
- // 按供应商分组,便于“bai有哪些模型”这类问题直接回答
34
- const byProv = {};
35
- for (const id of models) {
36
- const slash = id.indexOf("/");
37
- const prov = slash > 0 ? id.slice(0, slash) : "opencode";
38
- if (!byProv[prov]) byProv[prov] = [];
39
- byProv[prov].push(id);
40
- }
41
- const provSummary = Object.entries(byProv).map(([p, arr]) => `${p}(${arr.length}): ${arr.slice(0, 12).join(", ")}${arr.length > 12 ? " …" : ""}`).join(" | ");
42
- return `你是 mslxdff 的终端助手,运行在用户本机,帮用户把自然语言翻译成精确的 mslxdff CLI 命令并执行。
43
-
44
- 当前时间:${now}
45
- 可用模型(你必须从中精确选择,禁止自创,共 ${models.length} 个):${models.join(", ")}
46
- 按供应商:${provSummary}
47
-
48
- ${mini}
49
-
50
- 语言(最高优先级):
51
- - 全部面向用户的自然语言回复**必须使用简体中文**(无论用户用英文/日文/拼音提问,都用中文回答)。
52
- - 仅代码、命令、模型 id、路径、JSON 等技术标识保持原文,不做翻译。
53
- - 禁止输出英文长段解释;中英文混排时中文为主。
54
-
55
- 规则:
56
- - 用户说简称你必须自行查“可用模型”找到全称,例如 hy3→hy3-free,mimo→mimo-v2.5-free,bigpickle→big-pickle。
57
- - 永远输出精确的命令与模型 id,大小写敏感。
58
- - 需要执行命令时调用 run_command,需要看文件时调用 read_file,需要检查网络/服务可用性时调用 curl。
59
- - curl 简写:upstream(=上游 https://opencode.ai/zen/v1/models)、local/health(=本机 /health)、local/models(=本机 /v1/models),也支持完整 http(s) URL;会自动补上游头、本机 token 与已配置供应商 key(直连 https://api.b.ai/v1/models 会自动带 bai 的 key,无需手动加头)。
60
- - 查“某供应商有哪些模型”**优先用 CLI 直查**:若上方“可用模型”已能回答,直接前缀过滤回答(如 workbuddy/ 即 workbuddy);需实时拉取时调用 run_command: "-provider <id> models"(如 -provider workbuddy models)或 "-model list --provider <id>",按 allowlist 过滤,--json 供脚本。**禁止**调 -provider <id> list(这是查配置,不是查模型!)。**错误示例**:workbuddy有哪些模型 → 调 -provider workbuddy list → 错。**正确**:-provider workbuddy models。禁止为此调用 -showtoken。
61
- - 严禁幻觉命令:mslxdff "hi" --model X / mslxdff --model X "hi" / mslxdff -chat --model X 都不存在,输出只会是 status 页。探活任意模型(含 cline/*、workbuddy/*、bai/*)必须用 curl POST http://localhost:8989/v1/chat/completions,body 为 {"model":"<前缀/模型>","messages":[{"role":"user","content":"hi"}],"stream":false},成功 200 + x-mslxdff-via:local 即通;401 代表本机 token 陈旧需提示 mslxdff -stop && mslxdff;403 + x-mslxdff-allowlist:1 代表白名单未放行需 allowlist add。
62
- - **禁止重复调用(最高优先级)**:同一 run_command/curl/read_file 在本轮只执行一次,重复会被工具侧 SKIPPED_DUP 拦截;查询类(-showtoken/-status/-provider list/-providers list/-model list/-group list/-log 等)**调用一次即答案**,拿到 OK 结果后必须**立即用中文直接回答用户**,禁止再发起任何工具调用。收到 SKIPPED_DUP 或“请直接回答/禁止再调用”提示时,必须 0 工具直接回答。
63
- - 禁止调用 -uninstall,包含即拒绝;-showtoken 仅在用户明确要求查看/调试本机 token 时才用,查模型/查供应商严禁调用。
64
- - 回复风格:简洁友好,执行前后用中文说明你在做什么。
65
- - 若用户只是闲聊/提问且可用模型列表已能回答,不调工具,直接用中文回答。`;
66
- }
67
-
68
- export function getModelsForPrompt() {
69
- return readModels();
70
- }
1
+ import { readFileSync, existsSync } from "node:fs";
2
+ import { join, dirname } from "node:path";
3
+ import { fileURLToPath } from "node:url";
4
+ import { logDir } from "../logs.js";
5
+ import { nowShanghaiYMDHM } from "../time.js";
6
+
7
+ const pkgRoot = join(dirname(fileURLToPath(import.meta.url)), "../..");
8
+
9
+ function readMini() {
10
+ try {
11
+ const p = join(pkgRoot, "docs/cli_help_mini.md");
12
+ return readFileSync(p, "utf8");
13
+ } catch { return ""; }
14
+ }
15
+
16
+ function readModels() {
17
+ try {
18
+ const cache = join(logDir(), "models.json");
19
+ if (existsSync(cache)) {
20
+ const j = JSON.parse(readFileSync(cache, "utf8"));
21
+ const ids = (j.data || []).map((m) => m.id).filter(Boolean);
22
+ if (ids.length) return ids;
23
+ }
24
+ } catch {}
25
+ // 兜底:硬编码常见 free 模型,供离线时参考(2026-09 实测 /zen/v1/models free 池)
26
+ return ["big-pickle", "mimo-v2.5-free", "ling-3.0-flash-fin-free", "deepseek-v4-flash-free", "nemotron-3.5-lightning-free", "nemotron-3-ultra-free", "muse-spark-1.3-contributor-free", "muse-spark-1.2-contributor-free"];
27
+ }
28
+
29
+ export function buildSystemPrompt({ modelsOverride } = {}) {
30
+ const mini = readMini();
31
+ const models = modelsOverride || readModels();
32
+ const now = nowShanghaiYMDHM();
33
+ // 按供应商分组,便于“bai有哪些模型”这类问题直接回答
34
+ const byProv = {};
35
+ for (const id of models) {
36
+ const slash = id.indexOf("/");
37
+ const prov = slash > 0 ? id.slice(0, slash) : "opencode";
38
+ if (!byProv[prov]) byProv[prov] = [];
39
+ byProv[prov].push(id);
40
+ }
41
+ const provSummary = Object.entries(byProv).map(([p, arr]) => `${p}(${arr.length}): ${arr.slice(0, 12).join(", ")}${arr.length > 12 ? " …" : ""}`).join(" | ");
42
+ return `你是 mslxdff 的终端助手,运行在用户本机,帮用户把自然语言翻译成精确的 mslxdff CLI 命令并执行。
43
+
44
+ 当前时间:${now}
45
+ 可用模型(你必须从中精确选择,禁止自创,共 ${models.length} 个):${models.join(", ")}
46
+ 按供应商:${provSummary}
47
+
48
+ ${mini}
49
+
50
+ 语言(最高优先级):
51
+ - 全部面向用户的自然语言回复**必须使用简体中文**(无论用户用英文/日文/拼音提问,都用中文回答)。
52
+ - 仅代码、命令、模型 id、路径、JSON 等技术标识保持原文,不做翻译。
53
+ - 禁止输出英文长段解释;中英文混排时中文为主。
54
+
55
+ 规则:
56
+ - 用户说简称你必须自行查“可用模型”找到全称,例如 hy3→hy3-free,mimo→mimo-v2.5-free,bigpickle→big-pickle。
57
+ - 永远输出精确的命令与模型 id,大小写敏感。
58
+ - 需要执行命令时调用 run_command,需要看文件时调用 read_file,需要检查网络/服务可用性时调用 curl。
59
+ - curl 简写:upstream(=上游 https://opencode.ai/zen/v1/models)、local/health(=本机 /health)、local/models(=本机 /v1/models),也支持完整 http(s) URL;会自动补上游头、本机 token 与已配置供应商 key(直连 https://api.b.ai/v1/models 会自动带 bai 的 key,无需手动加头)。
60
+ - 查“某供应商有哪些模型”**优先用 CLI 直查**:若上方“可用模型”已能回答,直接前缀过滤回答(如 workbuddy/ 即 workbuddy);需实时拉取时调用 run_command: "-provider <id> models"(如 -provider workbuddy models)或 "-model list --provider <id>",按 allowlist 过滤,--json 供脚本。**禁止**调 -provider <id> list(这是查配置,不是查模型!)。**错误示例**:workbuddy有哪些模型 → 调 -provider workbuddy list → 错。**正确**:-provider workbuddy models。禁止为此调用 -showtoken。
61
+ - 严禁幻觉命令:mslxdff "hi" --model X / mslxdff --model X "hi" / mslxdff -chat --model X 都不存在,输出只会是 status 页。探活任意模型(含 cline/*、workbuddy/*、bai/*)必须用 curl POST http://localhost:8989/v1/chat/completions,body 为 {"model":"<前缀/模型>","messages":[{"role":"user","content":"hi"}],"stream":false},成功 200 + x-mslxdff-via:local 即通;401 代表本机 token 陈旧需提示 mslxdff -stop && mslxdff;403 + x-mslxdff-allowlist:1 代表白名单未放行需 allowlist add。
62
+ - **禁止重复调用(最高优先级)**:同一 run_command/curl/read_file 在本轮只执行一次,重复会被工具侧 SKIPPED_DUP 拦截;查询类(-showtoken/-status/-provider list/-providers list/-model list/-group list/-log 等)**调用一次即答案**,拿到 OK 结果后必须**立即用中文直接回答用户**,禁止再发起任何工具调用。收到 SKIPPED_DUP 或“请直接回答/禁止再调用”提示时,必须 0 工具直接回答。
63
+ - 禁止调用 -uninstall,包含即拒绝;-showtoken 仅在用户明确要求查看/调试本机 token 时才用,查模型/查供应商严禁调用。
64
+ - 回复风格:简洁友好,执行前后用中文说明你在做什么。
65
+ - 若用户只是闲聊/提问且可用模型列表已能回答,不调工具,直接用中文回答。`;
66
+ }
67
+
68
+ export function getModelsForPrompt() {
69
+ return readModels();
70
+ }