mslxdff 0.1.160 → 0.1.161
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bin/mslxdff.js +4 -4
- package/docs/cli_help_mini.md +135 -0
- package/package.json +1 -1
- package/src/auto.js +254 -254
- package/src/autostart.js +3 -0
- package/src/bench/cline-bench.js +42 -42
- package/src/bench/probe.js +70 -70
- package/src/bench/report.js +162 -162
- package/src/bench/runner.js +77 -77
- package/src/bench/via-probe.js +124 -124
- package/src/bench/via-routes.js +87 -87
- package/src/bench/workbuddy-bench.js +54 -54
- package/src/chat/engine.js +160 -160
- package/src/chat/gateway.js +163 -163
- package/src/chat/orchestrator.js +234 -234
- package/src/chat/prompt.js +70 -70
- package/src/chat/repl.js +88 -88
- package/src/chat/terminal.js +135 -135
- package/src/chat/tools.js +306 -306
- package/src/chat-pipeline/empty-turn.js +119 -0
- package/src/chat-pipeline/index.js +126 -123
- package/src/chat-pipeline/policy.js +76 -76
- package/src/chat-pipeline/serial-trial.js +53 -19
- package/src/cli/commands/daemon.js +4 -4
- package/src/cli/commands/group.js +249 -249
- package/src/cli/commands/model/list-providers.js +1 -1
- package/src/cli/commands/model/picks.js +50 -50
- package/src/cli/commands/provider/bench-via.js +247 -247
- package/src/cli/commands/provider/bench.js +141 -141
- package/src/cli/commands/provider/globalqwenwork-login.js +126 -0
- package/src/cli/commands/provider/index.js +126 -124
- package/src/cli/commands/provider/models.js +142 -139
- package/src/cli/commands/provider/qwenwork-login.js +119 -119
- package/src/cli/commands/sync.js +232 -232
- package/src/cli/commands/system.js +2 -2
- package/src/cli/format.js +6 -0
- package/src/cli/policy.js +2 -2
- package/src/cli/provider-row.js +2 -2
- package/src/cli/status.js +279 -279
- package/src/daemon.js +101 -96
- package/src/logs.js +14 -1
- package/src/model-capabilities/enrich.js +86 -86
- package/src/model-capabilities/index.js +183 -183
- package/src/model-capabilities/parse.js +70 -70
- package/src/model-trace.js +5 -2
- package/src/models.js +225 -225
- package/src/providers/AGENTS.md +46 -0
- package/src/providers/classify.js +1 -1
- package/src/providers/cline/auth.js +228 -228
- package/src/providers/cline/chat.js +307 -307
- package/src/providers/cline.js +2 -2
- package/src/providers/globalqwenwork/account-store.js +141 -0
- package/src/providers/globalqwenwork/constants.js +73 -0
- package/src/providers/globalqwenwork/cosy.js +123 -0
- package/src/providers/globalqwenwork/crypto.js +220 -0
- package/src/providers/globalqwenwork/http.js +22 -0
- package/src/providers/globalqwenwork/index.js +331 -0
- package/src/providers/globalqwenwork/payload.js +145 -0
- package/src/providers/globalqwenwork/rsa.js +56 -0
- package/src/providers/globalqwenwork/sse.js +270 -0
- package/src/providers/globalqwenwork/stream.js +132 -0
- package/src/providers/globalqwenwork/upstream.js +122 -0
- package/src/providers/globalqwenwork.js +1 -0
- package/src/providers/keyring.js +60 -60
- package/src/providers/qoder/chat.js +183 -183
- package/src/providers/qoder/index.js +230 -230
- package/src/providers/qoder/sse.js +103 -103
- package/src/providers/qwenwork/account-store.js +133 -133
- package/src/providers/qwenwork/constants.js +67 -67
- package/src/providers/qwenwork/cosy.js +120 -120
- package/src/providers/qwenwork/crypto.js +218 -218
- package/src/providers/qwenwork/http.js +20 -20
- package/src/providers/qwenwork/index.js +327 -327
- package/src/providers/qwenwork/payload.js +142 -142
- package/src/providers/qwenwork/rsa.js +54 -54
- package/src/providers/qwenwork/sse.js +268 -268
- package/src/providers/qwenwork/stream.js +130 -130
- package/src/providers/qwenwork/upstream.js +120 -120
- package/src/providers/qwenwork.js +1 -1
- package/src/providers/registry.js +74 -66
- package/src/providers/workbuddy/chat.js +248 -248
- package/src/providers/workbuddy/reshape.js +152 -152
- package/src/providers/workbuddy.js +2 -2
- package/src/providers/zcode/sse.js +19 -2
- package/src/reasoning.js +32 -32
- package/src/routes/AGENTS.md +37 -0
- package/src/routes/chat/exhausted-handler.js +23 -8
- package/src/routes/chat/gateway.js +46 -46
- package/src/routes/chat/local-handler.js +3 -1
- package/src/routes/chat/relay-pipeline.js +276 -264
- package/src/routes/chat/via-route-handler.js +146 -146
- package/src/routes/hedge.js +255 -255
- package/src/routes/helpers.js +20 -4
- package/src/routes/models-route.js +167 -167
- package/src/routes/peers.js +273 -273
- package/src/routes/stream-hold.js +138 -0
- package/src/routes/stream-scan.js +192 -0
- package/src/routes/stream.js +386 -393
- package/src/runtime/bootstrap.js +45 -45
- package/src/runtime/lifecycle-forensics.js +116 -0
- package/src/runtime/lifecycle-log.js +23 -0
- package/src/runtime/provider-gate.js +34 -33
- package/src/runtime/providers-setup.js +165 -165
- package/src/server.js +64 -64
- package/src/state/schemas/allowlist.js +92 -92
- package/src/sync-opencode.js +280 -280
- package/src/talk-log.js +226 -0
- package/src/timeline.js +5 -2
- package/src/transport/index.js +244 -244
- package/src/transport/pool.js +56 -56
- package/src/transport/retry.js +24 -24
- package/src/transport/sse.js +93 -93
- package/src/upstream-probe/display.js +52 -52
- package/src/upstream-probe/probe.js +49 -49
- package/src/upstream-probe/rotate.js +110 -110
- package/src/upstream-probe/start.js +45 -45
- package/src/upstream.js +289 -289
package/src/auto.js
CHANGED
|
@@ -1,254 +1,254 @@
|
|
|
1
|
-
import { statSync } from "node:fs";
|
|
2
|
-
import { loadModelErrors, saveModelErrors, loadModelLatencies, saveModelLatencies, loadPreferredModel, loadModelPicks, saveModelPicks, defaultStateFile } from "./state.js";
|
|
3
|
-
|
|
4
|
-
// 出厂默认首选模型(state.json 的 preferredModel / env MSLXDFF_PREFERRED_MODEL 可覆盖)
|
|
5
|
-
export const DEFAULT_PREFERRED_MODEL = "big-pickle";
|
|
6
|
-
// 兼容旧导出名:语义为"出厂默认",当前生效值请用 getPreferredModel()
|
|
7
|
-
export const PREFERRED_MODEL = DEFAULT_PREFERRED_MODEL;
|
|
8
|
-
|
|
9
|
-
// 当前生效的首选模型:state.json > env > 出厂默认;mtime 缓存保证 daemon 热生效
|
|
10
|
-
const _prefCache = { mtimeMs: -1, file: null, value: null };
|
|
11
|
-
export function getPreferredModel({ file = defaultStateFile() } = {}) {
|
|
12
|
-
try {
|
|
13
|
-
const st = statSync(file);
|
|
14
|
-
if (_prefCache.file !== file || st.mtimeMs !== _prefCache.mtimeMs) {
|
|
15
|
-
_prefCache.file = file;
|
|
16
|
-
_prefCache.mtimeMs = st.mtimeMs;
|
|
17
|
-
_prefCache.value = loadPreferredModel({ file });
|
|
18
|
-
}
|
|
19
|
-
} catch {
|
|
20
|
-
_prefCache.file = file;
|
|
21
|
-
_prefCache.mtimeMs = -1;
|
|
22
|
-
_prefCache.value = null;
|
|
23
|
-
}
|
|
24
|
-
const env = (process.env.MSLXDFF_PREFERRED_MODEL || "").trim();
|
|
25
|
-
return _prefCache.value || env || DEFAULT_PREFERRED_MODEL;
|
|
26
|
-
}
|
|
27
|
-
|
|
28
|
-
export const DEFAULT_AUTO_MODELS = [
|
|
29
|
-
PREFERRED_MODEL,
|
|
30
|
-
"mimo-v2.5-free",
|
|
31
|
-
"deepseek-v4-flash-free",
|
|
32
|
-
"ling-3.0-flash-fin-free",
|
|
33
|
-
"nemotron-3-ultra-free",
|
|
34
|
-
"nemotron-3.5-lightning-free",
|
|
35
|
-
"muse-spark-1.3-contributor-free",
|
|
36
|
-
].filter((id, i, arr) => id && arr.indexOf(id) === i);
|
|
37
|
-
|
|
38
|
-
export function isAutoModel(model) {
|
|
39
|
-
return !model || model === "auto";
|
|
40
|
-
}
|
|
41
|
-
|
|
42
|
-
export const MODEL_STATUS = Object.freeze({
|
|
43
|
-
NORMAL: "normal",
|
|
44
|
-
LIMIT: "limit",
|
|
45
|
-
ERROR: "error",
|
|
46
|
-
});
|
|
47
|
-
|
|
48
|
-
// Legacy modelErrors entries are bare timestamps ({id: ts}); newer ones are
|
|
49
|
-
// objects ({id: {status, at, code}}). Normalize both to an entry object.
|
|
50
|
-
// `slow` flags a model whose last request was slow (over the wall-clock
|
|
51
|
-
// threshold) — those get a longer cooldown so they lie low until they recover.
|
|
52
|
-
function normEntry(e) {
|
|
53
|
-
if (typeof e === "number") return { status: MODEL_STATUS.ERROR, at: e, code: null, slow: false };
|
|
54
|
-
if (e && typeof e === "object") {
|
|
55
|
-
return {
|
|
56
|
-
status: e.status || MODEL_STATUS.ERROR,
|
|
57
|
-
at: typeof e.at === "number" ? e.at : 0,
|
|
58
|
-
code: e.code ?? null,
|
|
59
|
-
slow: Boolean(e.slow),
|
|
60
|
-
};
|
|
61
|
-
}
|
|
62
|
-
return null;
|
|
63
|
-
}
|
|
64
|
-
|
|
65
|
-
export function classifyErrorEvent(evt = {}) {
|
|
66
|
-
if (evt.slow) return MODEL_STATUS.ERROR;
|
|
67
|
-
const code = Number(evt.status);
|
|
68
|
-
if (code === 429) return MODEL_STATUS.LIMIT;
|
|
69
|
-
const msg = String(evt.message || evt.note || "").toLowerCase();
|
|
70
|
-
if (msg.includes("rate limit") || msg.includes("limit exceeded") || msg.includes("429")) {
|
|
71
|
-
return MODEL_STATUS.LIMIT;
|
|
72
|
-
}
|
|
73
|
-
return MODEL_STATUS.ERROR;
|
|
74
|
-
}
|
|
75
|
-
|
|
76
|
-
export const DEFAULT_COOLDOWN_MS = 60_000;
|
|
77
|
-
export const DEFAULT_SLOW_COOLDOWN_MS = 5 * 60_000;
|
|
78
|
-
export const DEFAULT_LATENCY_ALPHA = 0.3;
|
|
79
|
-
|
|
80
|
-
function effectiveCooldown(entry, slowCooldownMs, cooldownMs) {
|
|
81
|
-
if (entry && entry.slow) return slowCooldownMs || 0;
|
|
82
|
-
return cooldownMs || 0;
|
|
83
|
-
}
|
|
84
|
-
|
|
85
|
-
function inCooldown(id, errors, now, cooldownMs, slowCooldownMs) {
|
|
86
|
-
const e = normEntry(errors[id]);
|
|
87
|
-
if (!e || !(e.at > 0)) return false;
|
|
88
|
-
if (e.status === MODEL_STATUS.NORMAL) return false;
|
|
89
|
-
const cd = effectiveCooldown(e, slowCooldownMs, cooldownMs);
|
|
90
|
-
return cd > 0 && now - e.at < cd;
|
|
91
|
-
}
|
|
92
|
-
|
|
93
|
-
// Latency EMA helpers
|
|
94
|
-
function normLatency(e) {
|
|
95
|
-
if (!e || typeof e !== "object") return null;
|
|
96
|
-
const ema = Number(e.emaMs);
|
|
97
|
-
return Number.isFinite(ema) && ema > 0 ? ema : null;
|
|
98
|
-
}
|
|
99
|
-
|
|
100
|
-
export function rankModels(ids, errors = {}, { now = Date.now(), cooldownMs = 0, slowCooldownMs = 0, latencies = {}, preferred } = {}) {
|
|
101
|
-
const pref = preferred ?? getPreferredModel();
|
|
102
|
-
// 最近一次成功的模型(NORMAL 且非慢且 at 最大),用于“上次成功优先”——慢模型即使刚成功也不应钉死
|
|
103
|
-
let lastSuccessId = null;
|
|
104
|
-
let lastSuccessAt = 0;
|
|
105
|
-
for (const [id, e] of Object.entries(errors)) {
|
|
106
|
-
const ne = normEntry(e);
|
|
107
|
-
if (ne && ne.status === MODEL_STATUS.NORMAL && !ne.slow && ne.at > lastSuccessAt) {
|
|
108
|
-
lastSuccessAt = ne.at;
|
|
109
|
-
lastSuccessId = id;
|
|
110
|
-
}
|
|
111
|
-
}
|
|
112
|
-
return [...new Set(ids)]
|
|
113
|
-
.filter(Boolean)
|
|
114
|
-
.map((id) => ({
|
|
115
|
-
id,
|
|
116
|
-
e: normEntry(errors[id]),
|
|
117
|
-
err: normEntry(errors[id])?.at ?? 0,
|
|
118
|
-
isPreferred: id === pref,
|
|
119
|
-
isLastSuccess: id === lastSuccessId,
|
|
120
|
-
cooling: inCooldown(id, errors, now, cooldownMs, slowCooldownMs),
|
|
121
|
-
latency: normLatency(latencies[id]) ?? Number.MAX_SAFE_INTEGER,
|
|
122
|
-
}))
|
|
123
|
-
.sort(
|
|
124
|
-
(a, b) =>
|
|
125
|
-
(a.cooling ? 1 : 0) - (b.cooling ? 1 : 0) ||
|
|
126
|
-
(b.isLastSuccess ? 1 : 0) - (a.isLastSuccess ? 1 : 0) ||
|
|
127
|
-
(b.isPreferred ? 1 : 0) - (a.isPreferred ? 1 : 0) ||
|
|
128
|
-
a.latency - b.latency ||
|
|
129
|
-
a.err - b.err
|
|
130
|
-
)
|
|
131
|
-
.map((x) => x.id);
|
|
132
|
-
}
|
|
133
|
-
|
|
134
|
-
export function createAutoSelector({
|
|
135
|
-
loadCandidates,
|
|
136
|
-
file,
|
|
137
|
-
now = () => Date.now(),
|
|
138
|
-
cooldownMs = DEFAULT_COOLDOWN_MS,
|
|
139
|
-
slowCooldownMs = DEFAULT_SLOW_COOLDOWN_MS,
|
|
140
|
-
latencyAlpha = DEFAULT_LATENCY_ALPHA,
|
|
141
|
-
errors: seedErrors,
|
|
142
|
-
latencies: seedLatencies,
|
|
143
|
-
persist = (errors, f = file) => saveModelErrors(errors, f ? { file: f } : {}),
|
|
144
|
-
persistLatencies = (latencies, f = file) => saveModelLatencies(latencies, f ? { file: f } : {}),
|
|
145
|
-
loadPicks = () => (file ? loadModelPicks({ file }) : []),
|
|
146
|
-
persistPicks = (picks) => (file ? saveModelPicks(picks, { file }) : picks),
|
|
147
|
-
} = {}) {
|
|
148
|
-
const lastErrorAt = { ...(seedErrors ?? loadModelErrors(file ? { file } : {})) };
|
|
149
|
-
const latencies = { ...(seedLatencies ?? loadModelLatencies(file ? { file } : {})) };
|
|
150
|
-
|
|
151
|
-
async function loadList() {
|
|
152
|
-
let list;
|
|
153
|
-
try {
|
|
154
|
-
const loaded = await loadCandidates?.();
|
|
155
|
-
list = Array.isArray(loaded) && loaded.length ? loaded : DEFAULT_AUTO_MODELS;
|
|
156
|
-
} catch {
|
|
157
|
-
list = DEFAULT_AUTO_MODELS;
|
|
158
|
-
}
|
|
159
|
-
return [...new Set(list)].filter(Boolean);
|
|
160
|
-
}
|
|
161
|
-
|
|
162
|
-
// 勾选集 = auto 候选池白名单:只在勾选的模型里择优;空勾选或勾选中无可用模型时回退全量
|
|
163
|
-
async function pickedPool(list) {
|
|
164
|
-
const picks = loadPicks();
|
|
165
|
-
if (!picks.length) return list;
|
|
166
|
-
const pickedSet = new Set(picks);
|
|
167
|
-
const filtered = list.filter((id) => pickedSet.has(id));
|
|
168
|
-
return filtered.length ? filtered : list;
|
|
169
|
-
}
|
|
170
|
-
|
|
171
|
-
async function candidates() {
|
|
172
|
-
const list = await loadList();
|
|
173
|
-
const pool = await pickedPool(list);
|
|
174
|
-
return rankModels(pool, lastErrorAt, { now: now(), cooldownMs, slowCooldownMs, latencies, preferred: getPreferredModel({ file: file ?? undefined }) });
|
|
175
|
-
}
|
|
176
|
-
|
|
177
|
-
async function candidatesFor(requested) {
|
|
178
|
-
if (!requested) return candidates();
|
|
179
|
-
const list = await loadList();
|
|
180
|
-
// 显式指定某模型 = 认可它,自动加入勾选集(仅当它是真实上游 free 模型时,避免垃圾 id 污染)
|
|
181
|
-
const picks = loadPicks();
|
|
182
|
-
if (list.includes(requested) && !picks.includes(requested) && persistPicks) {
|
|
183
|
-
await persistPicks([...picks, requested]);
|
|
184
|
-
}
|
|
185
|
-
const all = list.includes(requested) ? list : [requested, ...list];
|
|
186
|
-
const others = rankModels(all.filter((id) => id !== requested), lastErrorAt, {
|
|
187
|
-
now: now(),
|
|
188
|
-
cooldownMs,
|
|
189
|
-
slowCooldownMs,
|
|
190
|
-
latencies,
|
|
191
|
-
preferred: getPreferredModel({ file: file ?? undefined }),
|
|
192
|
-
});
|
|
193
|
-
// 显式指定模型:严格优先,永不因冷却被挤到最后(原设计:A deepseek 失败 → B/D deepseek 并发 → 都失败才 fallback)
|
|
194
|
-
// 冷却仅影响 auto 的择优,不影响指定模型的“很难被更改”语义
|
|
195
|
-
return [requested, ...others];
|
|
196
|
-
}
|
|
197
|
-
|
|
198
|
-
function isCooling(id) {
|
|
199
|
-
return inCooldown(id, lastErrorAt, now(), cooldownMs, slowCooldownMs);
|
|
200
|
-
}
|
|
201
|
-
|
|
202
|
-
async function recordError(id, evt = {}) {
|
|
203
|
-
if (!id) return;
|
|
204
|
-
lastErrorAt[id] = {
|
|
205
|
-
status: classifyErrorEvent(evt),
|
|
206
|
-
at: now(),
|
|
207
|
-
code: Number.isInteger(Number(evt.status)) ? Number(evt.status) : null,
|
|
208
|
-
slow: Boolean(evt.slow),
|
|
209
|
-
};
|
|
210
|
-
await persist({ ...lastErrorAt });
|
|
211
|
-
}
|
|
212
|
-
|
|
213
|
-
async function recordOk(id, evt = {}) {
|
|
214
|
-
if (!id) return;
|
|
215
|
-
lastErrorAt[id] = { status: MODEL_STATUS.NORMAL, at: now(), code: 200, slow: false };
|
|
216
|
-
await persist({ ...lastErrorAt });
|
|
217
|
-
const ms = Number(evt.latencyMs ?? evt.totalMs ?? evt.elapsedMs);
|
|
218
|
-
if (Number.isFinite(ms) && ms > 0) {
|
|
219
|
-
const prev = latencies[id]?.emaMs;
|
|
220
|
-
const ema = prev ? Math.round(prev * (1 - latencyAlpha) + ms * latencyAlpha) : Math.round(ms);
|
|
221
|
-
latencies[id] = { emaMs: ema, lastMs: Math.round(ms), at: now(), count: (latencies[id]?.count ?? 0) + 1 };
|
|
222
|
-
await persistLatencies({ ...latencies });
|
|
223
|
-
}
|
|
224
|
-
}
|
|
225
|
-
|
|
226
|
-
async function recordLatency(id, ms) {
|
|
227
|
-
if (!id || !Number.isFinite(ms) || ms <= 0) return;
|
|
228
|
-
const prev = latencies[id]?.emaMs;
|
|
229
|
-
const ema = prev ? Math.round(prev * (1 - latencyAlpha) + ms * latencyAlpha) : Math.round(ms);
|
|
230
|
-
latencies[id] = { emaMs: ema, lastMs: Math.round(ms), at: now(), count: (latencies[id]?.count ?? 0) + 1 };
|
|
231
|
-
await persistLatencies({ ...latencies });
|
|
232
|
-
}
|
|
233
|
-
|
|
234
|
-
function statuses() {
|
|
235
|
-
return { ...lastErrorAt };
|
|
236
|
-
}
|
|
237
|
-
|
|
238
|
-
function latencyStatuses() {
|
|
239
|
-
return { ...latencies };
|
|
240
|
-
}
|
|
241
|
-
|
|
242
|
-
return {
|
|
243
|
-
candidates,
|
|
244
|
-
candidatesFor,
|
|
245
|
-
recordError,
|
|
246
|
-
recordOk,
|
|
247
|
-
recordLatency,
|
|
248
|
-
statuses,
|
|
249
|
-
latencyStatuses,
|
|
250
|
-
isCooling,
|
|
251
|
-
errors: () => ({ ...lastErrorAt }),
|
|
252
|
-
latencies: () => ({ ...latencies }),
|
|
253
|
-
};
|
|
254
|
-
}
|
|
1
|
+
import { statSync } from "node:fs";
|
|
2
|
+
import { loadModelErrors, saveModelErrors, loadModelLatencies, saveModelLatencies, loadPreferredModel, loadModelPicks, saveModelPicks, defaultStateFile } from "./state.js";
|
|
3
|
+
|
|
4
|
+
// 出厂默认首选模型(state.json 的 preferredModel / env MSLXDFF_PREFERRED_MODEL 可覆盖)
|
|
5
|
+
export const DEFAULT_PREFERRED_MODEL = "big-pickle";
|
|
6
|
+
// 兼容旧导出名:语义为"出厂默认",当前生效值请用 getPreferredModel()
|
|
7
|
+
export const PREFERRED_MODEL = DEFAULT_PREFERRED_MODEL;
|
|
8
|
+
|
|
9
|
+
// 当前生效的首选模型:state.json > env > 出厂默认;mtime 缓存保证 daemon 热生效
|
|
10
|
+
const _prefCache = { mtimeMs: -1, file: null, value: null };
|
|
11
|
+
export function getPreferredModel({ file = defaultStateFile() } = {}) {
|
|
12
|
+
try {
|
|
13
|
+
const st = statSync(file);
|
|
14
|
+
if (_prefCache.file !== file || st.mtimeMs !== _prefCache.mtimeMs) {
|
|
15
|
+
_prefCache.file = file;
|
|
16
|
+
_prefCache.mtimeMs = st.mtimeMs;
|
|
17
|
+
_prefCache.value = loadPreferredModel({ file });
|
|
18
|
+
}
|
|
19
|
+
} catch {
|
|
20
|
+
_prefCache.file = file;
|
|
21
|
+
_prefCache.mtimeMs = -1;
|
|
22
|
+
_prefCache.value = null;
|
|
23
|
+
}
|
|
24
|
+
const env = (process.env.MSLXDFF_PREFERRED_MODEL || "").trim();
|
|
25
|
+
return _prefCache.value || env || DEFAULT_PREFERRED_MODEL;
|
|
26
|
+
}
|
|
27
|
+
|
|
28
|
+
export const DEFAULT_AUTO_MODELS = [
|
|
29
|
+
PREFERRED_MODEL,
|
|
30
|
+
"mimo-v2.5-free",
|
|
31
|
+
"deepseek-v4-flash-free",
|
|
32
|
+
"ling-3.0-flash-fin-free",
|
|
33
|
+
"nemotron-3-ultra-free",
|
|
34
|
+
"nemotron-3.5-lightning-free",
|
|
35
|
+
"muse-spark-1.3-contributor-free",
|
|
36
|
+
].filter((id, i, arr) => id && arr.indexOf(id) === i);
|
|
37
|
+
|
|
38
|
+
export function isAutoModel(model) {
|
|
39
|
+
return !model || model === "auto";
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
export const MODEL_STATUS = Object.freeze({
|
|
43
|
+
NORMAL: "normal",
|
|
44
|
+
LIMIT: "limit",
|
|
45
|
+
ERROR: "error",
|
|
46
|
+
});
|
|
47
|
+
|
|
48
|
+
// Legacy modelErrors entries are bare timestamps ({id: ts}); newer ones are
|
|
49
|
+
// objects ({id: {status, at, code}}). Normalize both to an entry object.
|
|
50
|
+
// `slow` flags a model whose last request was slow (over the wall-clock
|
|
51
|
+
// threshold) — those get a longer cooldown so they lie low until they recover.
|
|
52
|
+
function normEntry(e) {
|
|
53
|
+
if (typeof e === "number") return { status: MODEL_STATUS.ERROR, at: e, code: null, slow: false };
|
|
54
|
+
if (e && typeof e === "object") {
|
|
55
|
+
return {
|
|
56
|
+
status: e.status || MODEL_STATUS.ERROR,
|
|
57
|
+
at: typeof e.at === "number" ? e.at : 0,
|
|
58
|
+
code: e.code ?? null,
|
|
59
|
+
slow: Boolean(e.slow),
|
|
60
|
+
};
|
|
61
|
+
}
|
|
62
|
+
return null;
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
export function classifyErrorEvent(evt = {}) {
|
|
66
|
+
if (evt.slow) return MODEL_STATUS.ERROR;
|
|
67
|
+
const code = Number(evt.status);
|
|
68
|
+
if (code === 429) return MODEL_STATUS.LIMIT;
|
|
69
|
+
const msg = String(evt.message || evt.note || "").toLowerCase();
|
|
70
|
+
if (msg.includes("rate limit") || msg.includes("limit exceeded") || msg.includes("429")) {
|
|
71
|
+
return MODEL_STATUS.LIMIT;
|
|
72
|
+
}
|
|
73
|
+
return MODEL_STATUS.ERROR;
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
export const DEFAULT_COOLDOWN_MS = 60_000;
|
|
77
|
+
export const DEFAULT_SLOW_COOLDOWN_MS = 5 * 60_000;
|
|
78
|
+
export const DEFAULT_LATENCY_ALPHA = 0.3;
|
|
79
|
+
|
|
80
|
+
function effectiveCooldown(entry, slowCooldownMs, cooldownMs) {
|
|
81
|
+
if (entry && entry.slow) return slowCooldownMs || 0;
|
|
82
|
+
return cooldownMs || 0;
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
function inCooldown(id, errors, now, cooldownMs, slowCooldownMs) {
|
|
86
|
+
const e = normEntry(errors[id]);
|
|
87
|
+
if (!e || !(e.at > 0)) return false;
|
|
88
|
+
if (e.status === MODEL_STATUS.NORMAL) return false;
|
|
89
|
+
const cd = effectiveCooldown(e, slowCooldownMs, cooldownMs);
|
|
90
|
+
return cd > 0 && now - e.at < cd;
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
// Latency EMA helpers
|
|
94
|
+
function normLatency(e) {
|
|
95
|
+
if (!e || typeof e !== "object") return null;
|
|
96
|
+
const ema = Number(e.emaMs);
|
|
97
|
+
return Number.isFinite(ema) && ema > 0 ? ema : null;
|
|
98
|
+
}
|
|
99
|
+
|
|
100
|
+
export function rankModels(ids, errors = {}, { now = Date.now(), cooldownMs = 0, slowCooldownMs = 0, latencies = {}, preferred } = {}) {
|
|
101
|
+
const pref = preferred ?? getPreferredModel();
|
|
102
|
+
// 最近一次成功的模型(NORMAL 且非慢且 at 最大),用于“上次成功优先”——慢模型即使刚成功也不应钉死
|
|
103
|
+
let lastSuccessId = null;
|
|
104
|
+
let lastSuccessAt = 0;
|
|
105
|
+
for (const [id, e] of Object.entries(errors)) {
|
|
106
|
+
const ne = normEntry(e);
|
|
107
|
+
if (ne && ne.status === MODEL_STATUS.NORMAL && !ne.slow && ne.at > lastSuccessAt) {
|
|
108
|
+
lastSuccessAt = ne.at;
|
|
109
|
+
lastSuccessId = id;
|
|
110
|
+
}
|
|
111
|
+
}
|
|
112
|
+
return [...new Set(ids)]
|
|
113
|
+
.filter(Boolean)
|
|
114
|
+
.map((id) => ({
|
|
115
|
+
id,
|
|
116
|
+
e: normEntry(errors[id]),
|
|
117
|
+
err: normEntry(errors[id])?.at ?? 0,
|
|
118
|
+
isPreferred: id === pref,
|
|
119
|
+
isLastSuccess: id === lastSuccessId,
|
|
120
|
+
cooling: inCooldown(id, errors, now, cooldownMs, slowCooldownMs),
|
|
121
|
+
latency: normLatency(latencies[id]) ?? Number.MAX_SAFE_INTEGER,
|
|
122
|
+
}))
|
|
123
|
+
.sort(
|
|
124
|
+
(a, b) =>
|
|
125
|
+
(a.cooling ? 1 : 0) - (b.cooling ? 1 : 0) ||
|
|
126
|
+
(b.isLastSuccess ? 1 : 0) - (a.isLastSuccess ? 1 : 0) ||
|
|
127
|
+
(b.isPreferred ? 1 : 0) - (a.isPreferred ? 1 : 0) ||
|
|
128
|
+
a.latency - b.latency ||
|
|
129
|
+
a.err - b.err
|
|
130
|
+
)
|
|
131
|
+
.map((x) => x.id);
|
|
132
|
+
}
|
|
133
|
+
|
|
134
|
+
export function createAutoSelector({
|
|
135
|
+
loadCandidates,
|
|
136
|
+
file,
|
|
137
|
+
now = () => Date.now(),
|
|
138
|
+
cooldownMs = DEFAULT_COOLDOWN_MS,
|
|
139
|
+
slowCooldownMs = DEFAULT_SLOW_COOLDOWN_MS,
|
|
140
|
+
latencyAlpha = DEFAULT_LATENCY_ALPHA,
|
|
141
|
+
errors: seedErrors,
|
|
142
|
+
latencies: seedLatencies,
|
|
143
|
+
persist = (errors, f = file) => saveModelErrors(errors, f ? { file: f } : {}),
|
|
144
|
+
persistLatencies = (latencies, f = file) => saveModelLatencies(latencies, f ? { file: f } : {}),
|
|
145
|
+
loadPicks = () => (file ? loadModelPicks({ file }) : []),
|
|
146
|
+
persistPicks = (picks) => (file ? saveModelPicks(picks, { file }) : picks),
|
|
147
|
+
} = {}) {
|
|
148
|
+
const lastErrorAt = { ...(seedErrors ?? loadModelErrors(file ? { file } : {})) };
|
|
149
|
+
const latencies = { ...(seedLatencies ?? loadModelLatencies(file ? { file } : {})) };
|
|
150
|
+
|
|
151
|
+
async function loadList() {
|
|
152
|
+
let list;
|
|
153
|
+
try {
|
|
154
|
+
const loaded = await loadCandidates?.();
|
|
155
|
+
list = Array.isArray(loaded) && loaded.length ? loaded : DEFAULT_AUTO_MODELS;
|
|
156
|
+
} catch {
|
|
157
|
+
list = DEFAULT_AUTO_MODELS;
|
|
158
|
+
}
|
|
159
|
+
return [...new Set(list)].filter(Boolean);
|
|
160
|
+
}
|
|
161
|
+
|
|
162
|
+
// 勾选集 = auto 候选池白名单:只在勾选的模型里择优;空勾选或勾选中无可用模型时回退全量
|
|
163
|
+
async function pickedPool(list) {
|
|
164
|
+
const picks = loadPicks();
|
|
165
|
+
if (!picks.length) return list;
|
|
166
|
+
const pickedSet = new Set(picks);
|
|
167
|
+
const filtered = list.filter((id) => pickedSet.has(id));
|
|
168
|
+
return filtered.length ? filtered : list;
|
|
169
|
+
}
|
|
170
|
+
|
|
171
|
+
async function candidates() {
|
|
172
|
+
const list = await loadList();
|
|
173
|
+
const pool = await pickedPool(list);
|
|
174
|
+
return rankModels(pool, lastErrorAt, { now: now(), cooldownMs, slowCooldownMs, latencies, preferred: getPreferredModel({ file: file ?? undefined }) });
|
|
175
|
+
}
|
|
176
|
+
|
|
177
|
+
async function candidatesFor(requested) {
|
|
178
|
+
if (!requested) return candidates();
|
|
179
|
+
const list = await loadList();
|
|
180
|
+
// 显式指定某模型 = 认可它,自动加入勾选集(仅当它是真实上游 free 模型时,避免垃圾 id 污染)
|
|
181
|
+
const picks = loadPicks();
|
|
182
|
+
if (list.includes(requested) && !picks.includes(requested) && persistPicks) {
|
|
183
|
+
await persistPicks([...picks, requested]);
|
|
184
|
+
}
|
|
185
|
+
const all = list.includes(requested) ? list : [requested, ...list];
|
|
186
|
+
const others = rankModels(all.filter((id) => id !== requested), lastErrorAt, {
|
|
187
|
+
now: now(),
|
|
188
|
+
cooldownMs,
|
|
189
|
+
slowCooldownMs,
|
|
190
|
+
latencies,
|
|
191
|
+
preferred: getPreferredModel({ file: file ?? undefined }),
|
|
192
|
+
});
|
|
193
|
+
// 显式指定模型:严格优先,永不因冷却被挤到最后(原设计:A deepseek 失败 → B/D deepseek 并发 → 都失败才 fallback)
|
|
194
|
+
// 冷却仅影响 auto 的择优,不影响指定模型的“很难被更改”语义
|
|
195
|
+
return [requested, ...others];
|
|
196
|
+
}
|
|
197
|
+
|
|
198
|
+
function isCooling(id) {
|
|
199
|
+
return inCooldown(id, lastErrorAt, now(), cooldownMs, slowCooldownMs);
|
|
200
|
+
}
|
|
201
|
+
|
|
202
|
+
async function recordError(id, evt = {}) {
|
|
203
|
+
if (!id) return;
|
|
204
|
+
lastErrorAt[id] = {
|
|
205
|
+
status: classifyErrorEvent(evt),
|
|
206
|
+
at: now(),
|
|
207
|
+
code: Number.isInteger(Number(evt.status)) ? Number(evt.status) : null,
|
|
208
|
+
slow: Boolean(evt.slow),
|
|
209
|
+
};
|
|
210
|
+
await persist({ ...lastErrorAt });
|
|
211
|
+
}
|
|
212
|
+
|
|
213
|
+
async function recordOk(id, evt = {}) {
|
|
214
|
+
if (!id) return;
|
|
215
|
+
lastErrorAt[id] = { status: MODEL_STATUS.NORMAL, at: now(), code: 200, slow: false };
|
|
216
|
+
await persist({ ...lastErrorAt });
|
|
217
|
+
const ms = Number(evt.latencyMs ?? evt.totalMs ?? evt.elapsedMs);
|
|
218
|
+
if (Number.isFinite(ms) && ms > 0) {
|
|
219
|
+
const prev = latencies[id]?.emaMs;
|
|
220
|
+
const ema = prev ? Math.round(prev * (1 - latencyAlpha) + ms * latencyAlpha) : Math.round(ms);
|
|
221
|
+
latencies[id] = { emaMs: ema, lastMs: Math.round(ms), at: now(), count: (latencies[id]?.count ?? 0) + 1 };
|
|
222
|
+
await persistLatencies({ ...latencies });
|
|
223
|
+
}
|
|
224
|
+
}
|
|
225
|
+
|
|
226
|
+
async function recordLatency(id, ms) {
|
|
227
|
+
if (!id || !Number.isFinite(ms) || ms <= 0) return;
|
|
228
|
+
const prev = latencies[id]?.emaMs;
|
|
229
|
+
const ema = prev ? Math.round(prev * (1 - latencyAlpha) + ms * latencyAlpha) : Math.round(ms);
|
|
230
|
+
latencies[id] = { emaMs: ema, lastMs: Math.round(ms), at: now(), count: (latencies[id]?.count ?? 0) + 1 };
|
|
231
|
+
await persistLatencies({ ...latencies });
|
|
232
|
+
}
|
|
233
|
+
|
|
234
|
+
function statuses() {
|
|
235
|
+
return { ...lastErrorAt };
|
|
236
|
+
}
|
|
237
|
+
|
|
238
|
+
function latencyStatuses() {
|
|
239
|
+
return { ...latencies };
|
|
240
|
+
}
|
|
241
|
+
|
|
242
|
+
return {
|
|
243
|
+
candidates,
|
|
244
|
+
candidatesFor,
|
|
245
|
+
recordError,
|
|
246
|
+
recordOk,
|
|
247
|
+
recordLatency,
|
|
248
|
+
statuses,
|
|
249
|
+
latencyStatuses,
|
|
250
|
+
isCooling,
|
|
251
|
+
errors: () => ({ ...lastErrorAt }),
|
|
252
|
+
latencies: () => ({ ...latencies }),
|
|
253
|
+
};
|
|
254
|
+
}
|
package/src/autostart.js
CHANGED
|
@@ -3,6 +3,8 @@ import { existsSync, mkdirSync, writeFileSync, readFileSync, rmSync } from "node
|
|
|
3
3
|
import { homedir, userInfo } from "node:os";
|
|
4
4
|
import { join, dirname } from "node:path";
|
|
5
5
|
import { fileURLToPath } from "node:url";
|
|
6
|
+
import { daemonDir, readPid } from "./daemon.js";
|
|
7
|
+
import { writeExitMarker } from "./runtime/lifecycle-forensics.js";
|
|
6
8
|
|
|
7
9
|
const TASK_NAME = "mslxdff";
|
|
8
10
|
const SERVICE_NAME = "mslxdff";
|
|
@@ -203,6 +205,7 @@ async function linuxEnable() {
|
|
|
203
205
|
}
|
|
204
206
|
for (const p of pids) {
|
|
205
207
|
if (p === process.pid) continue;
|
|
208
|
+
if (p === readPid()) { try { writeExitMarker(daemonDir(), { reason: "autostart-cleanup", prevPid: p, prevVersion: null, byPid: process.pid }); } catch {} }
|
|
206
209
|
try { process.kill(p, "SIGTERM"); killed++; } catch {}
|
|
207
210
|
}
|
|
208
211
|
if (killed) await new Promise((r2) => setTimeout(r2, 400));
|
package/src/bench/cline-bench.js
CHANGED
|
@@ -1,42 +1,42 @@
|
|
|
1
|
-
import { clineHeaders } from "../providers/cline/headers.js";
|
|
2
|
-
import { computeMetrics } from "../metrics.js";
|
|
3
|
-
import { createTransport } from "../transport/index.js";
|
|
4
|
-
|
|
5
|
-
function sseContent(obj) {
|
|
6
|
-
const c = obj?.choices?.[0]?.delta?.content || obj?.choices?.[0]?.message?.content || "";
|
|
7
|
-
return typeof c === "string" ? c : "";
|
|
8
|
-
}
|
|
9
|
-
|
|
10
|
-
export async function clineBenchOne({ baseUrl, model, accessToken, prompt, maxTokens, timeoutMs, fetchImpl }) {
|
|
11
|
-
const tr = createTransport({ fetchImpl, keepAlive: false, retry: {}, timeoutMs });
|
|
12
|
-
let ttfbMs = null;
|
|
13
|
-
let totalMs = null;
|
|
14
|
-
let content = "";
|
|
15
|
-
try {
|
|
16
|
-
const sid = `sess_bench_${Date.now()}`;
|
|
17
|
-
const res = await tr.request({
|
|
18
|
-
url: `${baseUrl}/api/v1/chat/completions`,
|
|
19
|
-
method: "POST",
|
|
20
|
-
headers: { ...clineHeaders(sid, accessToken), Accept: "text/event-stream" },
|
|
21
|
-
body: { model, messages: [{ role: "user", content: prompt }], stream: true, max_tokens: maxTokens, session_id: sid, reasoning_effort: "high" },
|
|
22
|
-
stream: true,
|
|
23
|
-
});
|
|
24
|
-
if (!res.ok) {
|
|
25
|
-
let txt = "";
|
|
26
|
-
try { txt = await res.text(); } catch {}
|
|
27
|
-
const label = res.status === 401 ? "鉴权失败" : res.status === 429 ? "限流" : res.status >= 500 ? `上游错误 ${res.status}` : `HTTP ${res.status}`;
|
|
28
|
-
return { id: model, ok: false, status: res.status, label, error: txt.slice(0, 300), ttfbMs, totalMs: res.totalMs, tps: null, charsPerSec: null, tokens: null };
|
|
29
|
-
}
|
|
30
|
-
for await (const ev of res.stream()) {
|
|
31
|
-
try { content += sseContent(JSON.parse(ev)); } catch {}
|
|
32
|
-
}
|
|
33
|
-
ttfbMs = res.ttfbMs;
|
|
34
|
-
totalMs = res.totalMs;
|
|
35
|
-
const chars = content.length;
|
|
36
|
-
const { tps, charsPerSec } = computeMetrics({ ttfbMs, totalMs, promptTokens: null, completionTokens: null, chars });
|
|
37
|
-
return { id: model, ok: true, status: 200, label: "成功", ttfbMs, totalMs, tps, charsPerSec, tokens: null, chars };
|
|
38
|
-
} catch (e) {
|
|
39
|
-
const msg = e?.message || String(e);
|
|
40
|
-
return { id: model, ok: false, label: /timeout|abort/i.test(msg) ? "超时" : "网络错误", error: msg.slice(0, 300), ttfbMs, totalMs, tps: null, charsPerSec: null, tokens: null };
|
|
41
|
-
}
|
|
42
|
-
}
|
|
1
|
+
import { clineHeaders } from "../providers/cline/headers.js";
|
|
2
|
+
import { computeMetrics } from "../metrics.js";
|
|
3
|
+
import { createTransport } from "../transport/index.js";
|
|
4
|
+
|
|
5
|
+
function sseContent(obj) {
|
|
6
|
+
const c = obj?.choices?.[0]?.delta?.content || obj?.choices?.[0]?.message?.content || "";
|
|
7
|
+
return typeof c === "string" ? c : "";
|
|
8
|
+
}
|
|
9
|
+
|
|
10
|
+
export async function clineBenchOne({ baseUrl, model, accessToken, prompt, maxTokens, timeoutMs, fetchImpl }) {
|
|
11
|
+
const tr = createTransport({ fetchImpl, keepAlive: false, retry: {}, timeoutMs });
|
|
12
|
+
let ttfbMs = null;
|
|
13
|
+
let totalMs = null;
|
|
14
|
+
let content = "";
|
|
15
|
+
try {
|
|
16
|
+
const sid = `sess_bench_${Date.now()}`;
|
|
17
|
+
const res = await tr.request({
|
|
18
|
+
url: `${baseUrl}/api/v1/chat/completions`,
|
|
19
|
+
method: "POST",
|
|
20
|
+
headers: { ...clineHeaders(sid, accessToken), Accept: "text/event-stream" },
|
|
21
|
+
body: { model, messages: [{ role: "user", content: prompt }], stream: true, max_tokens: maxTokens, session_id: sid, reasoning_effort: "high" },
|
|
22
|
+
stream: true,
|
|
23
|
+
});
|
|
24
|
+
if (!res.ok) {
|
|
25
|
+
let txt = "";
|
|
26
|
+
try { txt = await res.text(); } catch {}
|
|
27
|
+
const label = res.status === 401 ? "鉴权失败" : res.status === 429 ? "限流" : res.status >= 500 ? `上游错误 ${res.status}` : `HTTP ${res.status}`;
|
|
28
|
+
return { id: model, ok: false, status: res.status, label, error: txt.slice(0, 300), ttfbMs, totalMs: res.totalMs, tps: null, charsPerSec: null, tokens: null };
|
|
29
|
+
}
|
|
30
|
+
for await (const ev of res.stream()) {
|
|
31
|
+
try { content += sseContent(JSON.parse(ev)); } catch {}
|
|
32
|
+
}
|
|
33
|
+
ttfbMs = res.ttfbMs;
|
|
34
|
+
totalMs = res.totalMs;
|
|
35
|
+
const chars = content.length;
|
|
36
|
+
const { tps, charsPerSec } = computeMetrics({ ttfbMs, totalMs, promptTokens: null, completionTokens: null, chars });
|
|
37
|
+
return { id: model, ok: true, status: 200, label: "成功", ttfbMs, totalMs, tps, charsPerSec, tokens: null, chars };
|
|
38
|
+
} catch (e) {
|
|
39
|
+
const msg = e?.message || String(e);
|
|
40
|
+
return { id: model, ok: false, label: /timeout|abort/i.test(msg) ? "超时" : "网络错误", error: msg.slice(0, 300), ttfbMs, totalMs, tps: null, charsPerSec: null, tokens: null };
|
|
41
|
+
}
|
|
42
|
+
}
|