dsh-livebench-panel 0.1.11 → 0.1.13
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/lib/client.js +130 -72
- package/lib/index.js +95 -56
- package/package.json +1 -1
package/lib/client.js
CHANGED
|
@@ -92,11 +92,12 @@ window.__ModuleLoader__.load({
|
|
|
92
92
|
function LiveBenchView() {
|
|
93
93
|
const [config, setConfig] = useState(null);
|
|
94
94
|
const [configError, setConfigError] = useState(null);
|
|
95
|
-
const [sel, setSel] = useState({
|
|
95
|
+
const [sel, setSel] = useState({ models: [], reasoning: "default", cats: [], tasks: [], release: "2024-11-25", begin: "", end: "", maxTokens: "32000" });
|
|
96
|
+
const [modelDdOpen, setModelDdOpen] = useState(false);
|
|
97
|
+
const modelDdRef = useRef(null);
|
|
96
98
|
const [busy, setBusy] = useState(false);
|
|
97
99
|
const [running, setRunning] = useState(false);
|
|
98
|
-
const [
|
|
99
|
-
const [exitCode, setExitCode] = useState(null);
|
|
100
|
+
const [runsList, setRunsList] = useState([]);
|
|
100
101
|
const [startError, setStartError] = useState(null);
|
|
101
102
|
const [homeInput, setHomeInput] = useState("");
|
|
102
103
|
const [homeBusy, setHomeBusy] = useState(false);
|
|
@@ -108,6 +109,16 @@ window.__ModuleLoader__.load({
|
|
|
108
109
|
const dragKey = useRef(null);
|
|
109
110
|
const logRef = useRef(null);
|
|
110
111
|
|
|
112
|
+
// 点击面板外部时关闭模型下拉
|
|
113
|
+
useEffect(() => {
|
|
114
|
+
if (!modelDdOpen) return;
|
|
115
|
+
const onDocDown = (event) => {
|
|
116
|
+
if (modelDdRef.current && !modelDdRef.current.contains(event.target)) setModelDdOpen(false);
|
|
117
|
+
};
|
|
118
|
+
document.addEventListener("mousedown", onDocDown);
|
|
119
|
+
return () => document.removeEventListener("mousedown", onDocDown);
|
|
120
|
+
}, [modelDdOpen]);
|
|
121
|
+
|
|
111
122
|
const loadConfig = useCallback(async () => {
|
|
112
123
|
const payload = await api("/config");
|
|
113
124
|
if (payload.ok) {
|
|
@@ -115,14 +126,12 @@ window.__ModuleLoader__.load({
|
|
|
115
126
|
setConfigError(null);
|
|
116
127
|
setSel((prev) => {
|
|
117
128
|
const next = { ...prev, release: payload.releases.includes(prev.release) ? prev.release : "2024-11-25" };
|
|
118
|
-
|
|
119
|
-
|
|
120
|
-
|
|
121
|
-
|
|
122
|
-
|
|
123
|
-
|
|
124
|
-
}
|
|
125
|
-
if (!payload.releases.includes(next.release)) next.release = payload.releases[0] ?? "2024-11-25";
|
|
129
|
+
// 保留仍存在的选择,剔除失效项
|
|
130
|
+
next.models = prev.models.filter((v) => {
|
|
131
|
+
const [pid] = v.split("::");
|
|
132
|
+
const p = payload.providers.find((x) => x.id === pid);
|
|
133
|
+
return p ? true : false;
|
|
134
|
+
});
|
|
126
135
|
return next;
|
|
127
136
|
});
|
|
128
137
|
} else {
|
|
@@ -138,11 +147,8 @@ window.__ModuleLoader__.load({
|
|
|
138
147
|
const refreshStatus = useCallback(async () => {
|
|
139
148
|
const payload = await api("/status");
|
|
140
149
|
if (payload.ok) {
|
|
150
|
+
setRunsList(payload.runs ?? []);
|
|
141
151
|
setRunning(payload.running === true);
|
|
142
|
-
if (payload.hasRun) {
|
|
143
|
-
setLog(payload.log ?? "");
|
|
144
|
-
setExitCode(payload.exitCode);
|
|
145
|
-
}
|
|
146
152
|
return payload.running === true;
|
|
147
153
|
}
|
|
148
154
|
return false;
|
|
@@ -165,10 +171,6 @@ window.__ModuleLoader__.load({
|
|
|
165
171
|
return () => clearInterval(timer);
|
|
166
172
|
}, [refreshStatus, loadResults]);
|
|
167
173
|
|
|
168
|
-
useEffect(() => {
|
|
169
|
-
if (logRef.current) logRef.current.scrollTop = logRef.current.scrollHeight;
|
|
170
|
-
}, [log]);
|
|
171
|
-
|
|
172
174
|
const categories = useMemo(() => (config ? Object.keys(config.tasks) : []), [config]);
|
|
173
175
|
// 有效题数:LiveBench 会丢弃「发布晚于所选 release」和「在所选 release
|
|
174
176
|
// 前已退役」的题目,这里按题目桶 (发布日, 移除日) 精确复算。
|
|
@@ -197,21 +199,31 @@ window.__ModuleLoader__.load({
|
|
|
197
199
|
const next = list.includes(value) ? list.filter((v) => v !== value) : [...list, value];
|
|
198
200
|
return { ...prev, [key]: next };
|
|
199
201
|
});
|
|
200
|
-
|
|
201
|
-
|
|
202
|
-
[
|
|
203
|
-
|
|
204
|
-
|
|
205
|
-
|
|
206
|
-
|
|
207
|
-
);
|
|
202
|
+
// 已选模型(provider::model 值列表)的解析与 effort 并集
|
|
203
|
+
const selectedModelEntries = useMemo(() => sel.models.map((value) => {
|
|
204
|
+
const [pid, ...rest] = value.split("::");
|
|
205
|
+
const mid = rest.join("::");
|
|
206
|
+
const provider = config?.providers.find((p) => p.id === pid) ?? null;
|
|
207
|
+
const model = provider?.models.find((m) => m.id === mid) ?? null;
|
|
208
|
+
return { providerId: pid, modelId: mid, efforts: model?.efforts ?? [] };
|
|
209
|
+
}).filter((entry) => entry.modelId !== ""), [sel.models, config]);
|
|
208
210
|
const effortOptions = useMemo(() => {
|
|
209
|
-
const
|
|
210
|
-
|
|
211
|
-
|
|
211
|
+
const seen = new Set();
|
|
212
|
+
const union = [];
|
|
213
|
+
for (const entry of selectedModelEntries) {
|
|
214
|
+
for (const e of entry.efforts) {
|
|
215
|
+
if (!seen.has(e)) { seen.add(e); union.push({ value: e, label: e === "off" ? "off(关闭思考)" : e }); }
|
|
216
|
+
}
|
|
217
|
+
}
|
|
218
|
+
return union;
|
|
219
|
+
}, [selectedModelEntries]);
|
|
212
220
|
|
|
213
221
|
const onStart = async () => {
|
|
214
222
|
setStartError(null);
|
|
223
|
+
if (selectedModelEntries.length === 0) {
|
|
224
|
+
setStartError("请先在下拉框中至少选择一个模型。");
|
|
225
|
+
return;
|
|
226
|
+
}
|
|
215
227
|
const benchNames = [];
|
|
216
228
|
if (sel.cats.length === 0) {
|
|
217
229
|
benchNames.push("live_bench");
|
|
@@ -227,22 +239,28 @@ window.__ModuleLoader__.load({
|
|
|
227
239
|
}
|
|
228
240
|
setBusy(true);
|
|
229
241
|
try {
|
|
230
|
-
|
|
231
|
-
|
|
232
|
-
|
|
233
|
-
|
|
234
|
-
|
|
235
|
-
|
|
236
|
-
|
|
237
|
-
|
|
238
|
-
|
|
239
|
-
|
|
240
|
-
|
|
241
|
-
|
|
242
|
-
|
|
242
|
+
// 每个 selected 模型并发启动一个评测(服务端有并发上限保护)
|
|
243
|
+
const failures = [];
|
|
244
|
+
const launches = selectedModelEntries.map(async (entry) => {
|
|
245
|
+
const payload = await api("/start", {
|
|
246
|
+
method: "POST",
|
|
247
|
+
headers: { "content-type": "application/json" },
|
|
248
|
+
body: JSON.stringify({
|
|
249
|
+
provider: entry.providerId,
|
|
250
|
+
model: entry.modelId,
|
|
251
|
+
reasoningEffort: sel.reasoning,
|
|
252
|
+
benchNames,
|
|
253
|
+
release: sel.release,
|
|
254
|
+
begin: sel.begin,
|
|
255
|
+
end: sel.end,
|
|
256
|
+
maxTokens: sel.maxTokens,
|
|
257
|
+
}),
|
|
258
|
+
});
|
|
259
|
+
if (!payload.ok) failures.push(`${entry.modelId}: ${payload.error ?? "启动失败"}`);
|
|
243
260
|
});
|
|
244
|
-
|
|
245
|
-
|
|
261
|
+
await Promise.all(launches);
|
|
262
|
+
if (failures.length > 0) setStartError(failures.join(";"));
|
|
263
|
+
await refreshStatus();
|
|
246
264
|
} catch (error) {
|
|
247
265
|
setStartError(String(error.message ?? error));
|
|
248
266
|
} finally {
|
|
@@ -275,14 +293,7 @@ window.__ModuleLoader__.load({
|
|
|
275
293
|
|
|
276
294
|
const setField = (key) => (event) => {
|
|
277
295
|
const value = event.target.value;
|
|
278
|
-
setSel((prev) => {
|
|
279
|
-
const next = { ...prev, [key]: value };
|
|
280
|
-
if (key === "provider") {
|
|
281
|
-
const provider = config?.providers.find((p) => p.id === value) ?? null;
|
|
282
|
-
next.model = provider?.models[0]?.id ?? "";
|
|
283
|
-
}
|
|
284
|
-
return next;
|
|
285
|
-
});
|
|
296
|
+
setSel((prev) => ({ ...prev, [key]: value }));
|
|
286
297
|
};
|
|
287
298
|
|
|
288
299
|
// ---- 成绩矩阵:model 为行标识、category/task 为列标识 ----
|
|
@@ -385,18 +396,48 @@ window.__ModuleLoader__.load({
|
|
|
385
396
|
configError !== null && h("p", { className: c("error") }, `加载配置失败:${configError}`),
|
|
386
397
|
h("div", { className: c("card") },
|
|
387
398
|
h("div", { className: c("grid") },
|
|
388
|
-
h("div", { className: c("field") },
|
|
389
|
-
h("label", { className: c("label") }, "模型(harness
|
|
390
|
-
h("
|
|
391
|
-
|
|
392
|
-
|
|
393
|
-
|
|
394
|
-
|
|
399
|
+
h("div", { className: c("field"), ref: modelDdRef, style: { position: "relative" } },
|
|
400
|
+
h("label", { className: c("label") }, "模型(harness 全部模型,可多选)"),
|
|
401
|
+
h("button", {
|
|
402
|
+
type: "button", className: c("select"), style: { textAlign: "left", width: "100%", overflow: "hidden", textOverflow: "ellipsis", whiteSpace: "nowrap" },
|
|
403
|
+
onClick: () => setModelDdOpen((v) => !v),
|
|
404
|
+
}, sel.models.length === 0
|
|
405
|
+
? "点击选择模型…"
|
|
406
|
+
: sel.models.length === 1
|
|
407
|
+
? sel.models[0].split("::")[1]
|
|
408
|
+
: `已选 ${sel.models.length} 个模型`),
|
|
409
|
+
modelDdOpen && h("div", {
|
|
410
|
+
style: {
|
|
411
|
+
position: "absolute", zIndex: 30, top: "calc(100% + 4px)", left: 0, right: 0,
|
|
412
|
+
maxHeight: "320px", overflowY: "auto", border: "1px solid var(--dsw-alias-border-l2)",
|
|
413
|
+
borderRadius: "8px", background: "var(--dsw-alias-bg-layer-1)", padding: "6px",
|
|
414
|
+
boxShadow: "var(--dsw-shadow-lv1, 0 4px 16px rgba(0,0,0,.25))",
|
|
415
|
+
},
|
|
416
|
+
},
|
|
417
|
+
(config?.providers ?? []).map((p) => h("div", { key: p.id, style: { marginBottom: "4px" } },
|
|
418
|
+
h("div", { style: { fontSize: "11px", color: "var(--dsw-alias-label-tertiary)", padding: "3px 4px" } },
|
|
419
|
+
p.name + (p.routable ? "" : " ·未接LiveBench")),
|
|
420
|
+
p.models.map((m) => {
|
|
421
|
+
const value = p.id + "::" + m.id;
|
|
422
|
+
const checked = sel.models.includes(value);
|
|
423
|
+
return h("label", {
|
|
424
|
+
key: value,
|
|
425
|
+
style: { display: "flex", alignItems: "center", gap: "6px", padding: "3px 4px", cursor: "pointer", borderRadius: "6px" },
|
|
426
|
+
},
|
|
427
|
+
h("input", { type: "checkbox", checked, onChange: () => {
|
|
428
|
+
setSel((prev) => {
|
|
429
|
+
const set = new Set(prev.models);
|
|
430
|
+
if (set.has(value)) set.delete(value); else set.add(value);
|
|
431
|
+
return { ...prev, models: [...set] };
|
|
432
|
+
});
|
|
433
|
+
} }),
|
|
434
|
+
h("span", { style: { fontSize: "12.5px" } }, m.name));
|
|
435
|
+
}))),
|
|
395
436
|
),
|
|
396
437
|
h("div", { className: c("field") },
|
|
397
438
|
h("label", { className: c("label") }, "推理强度"),
|
|
398
439
|
h("select", { className: c("select"), value: sel.reasoning, onChange: setField("reasoning"), disabled: effortOptions.length === 0 },
|
|
399
|
-
h("option", { value: "default" }, effortOptions.length === 0 ? "
|
|
440
|
+
h("option", { value: "default" }, effortOptions.length === 0 ? "(按模型默认)" : "(模型默认)"),
|
|
400
441
|
effortOptions.map((e) => h("option", { key: e.value, value: e.value }, e.label))),
|
|
401
442
|
),
|
|
402
443
|
h("div", { className: c("field") },
|
|
@@ -443,22 +484,25 @@ window.__ModuleLoader__.load({
|
|
|
443
484
|
),
|
|
444
485
|
),
|
|
445
486
|
h("div", { className: c("row") },
|
|
446
|
-
h("button", { className: c("btn"), onClick: onStart, disabled: busy || running || !config?.available || sel.
|
|
447
|
-
running ?
|
|
448
|
-
running && h("button", { className: c("btn") + " " + c("btnDanger"), onClick: onStop }, "
|
|
487
|
+
h("button", { className: c("btn"), onClick: onStart, disabled: busy || running || !config?.available || sel.models.length === 0 },
|
|
488
|
+
running ? `评测运行中(可另起)…` : busy ? "启动中…" : `开始评测(${sel.models.length} 个模型并发)`),
|
|
489
|
+
running && h("button", { className: c("btn") + " " + c("btnDanger"), onClick: onStop }, "停止全部"),
|
|
449
490
|
h("button", { className: c("btnGhost") + " " + c("btn"), onClick: () => { loadConfig(); loadResults(); } }, "刷新"),
|
|
450
|
-
selectedProvider && !selectedProvider.routable && h("span", { className: c("hint") },
|
|
451
|
-
"该 provider 的协议或端点未知,无法自动路由:LiveBench 将按模型名原生尝试,未注册的模型名会失败。"),
|
|
452
491
|
),
|
|
453
492
|
startError !== null && h("p", { className: c("error") }, startError),
|
|
454
493
|
),
|
|
455
|
-
|
|
494
|
+
runsList.length > 0 && h("div", { className: c("card") },
|
|
456
495
|
h("div", { className: c("row") },
|
|
457
496
|
h("span", { className: c("badge"), "data-ok": running ? "1" : "0" },
|
|
458
|
-
running ? "
|
|
497
|
+
running ? "有评测运行中" : "无运行中的评测"),
|
|
459
498
|
),
|
|
460
|
-
h("
|
|
461
|
-
|
|
499
|
+
runsList.map((r) => h("details", { key: r.runId, open: r.running },
|
|
500
|
+
h("summary", { style: { cursor: "pointer", fontSize: "12px", color: "var(--dsw-alias-label-secondary)" } },
|
|
501
|
+
h("span", { className: c("badge"), "data-ok": r.running ? "1" : "0" }, r.running ? "运行中" : r.exitCode === 0 ? "完成" : `退出(${r.exitCode})`),
|
|
502
|
+
" ",
|
|
503
|
+
r.displayName),
|
|
504
|
+
h("pre", { className: c("log"), ref: r.running ? logRef : null }, r.log || "(暂无输出)"),
|
|
505
|
+
)))),
|
|
462
506
|
sortedModels.length > 0 && h("div", { className: c("card") },
|
|
463
507
|
h("div", { className: c("row") },
|
|
464
508
|
h("span", { className: c("label") }, "评测成绩(model 为行;分数 = 平均分 ×100;行首可拖动排序,顺序保存在本机)"),
|
|
@@ -498,8 +542,22 @@ window.__ModuleLoader__.load({
|
|
|
498
542
|
h("td", null, model),
|
|
499
543
|
matrix.tasks.map((taskKeyCol) => {
|
|
500
544
|
const row = matrix.cells.get(`${model}\u0000${taskKeyCol}`);
|
|
501
|
-
|
|
502
|
-
|
|
545
|
+
// ×= 该模型未做过此任务;E = 做了但全部 API 失败
|
|
546
|
+
let cellText = "×";
|
|
547
|
+
let title = "该模型未执行此任务";
|
|
548
|
+
if (row) {
|
|
549
|
+
if (row.answered > 0 && row.errors >= row.answered) {
|
|
550
|
+
cellText = `E (${row.errors})`;
|
|
551
|
+
title = "该任务全部回答均 API 失败($ERROR$),计 0 分";
|
|
552
|
+
} else if (row.judged === 0) {
|
|
553
|
+
cellText = "未判分";
|
|
554
|
+
title = "有答案但尚未判分";
|
|
555
|
+
} else {
|
|
556
|
+
cellText = `${row.score.toFixed(1)} (${row.judged}/${row.total})`;
|
|
557
|
+
if (row.errors > 0) title += ` · 其中 ${row.errors} 题 API 失败计 0 分`;
|
|
558
|
+
}
|
|
559
|
+
}
|
|
560
|
+
return h("td", { key: taskKeyCol, className: c("score"), title }, cellText);
|
|
503
561
|
}),
|
|
504
562
|
h("td", null, h("button", {
|
|
505
563
|
className: c("btnGhost") + " " + c("btn"), title: "删除该模型的所有评测成绩",
|
package/lib/index.js
CHANGED
|
@@ -102,8 +102,8 @@ const RELEASES = [
|
|
|
102
102
|
];
|
|
103
103
|
/** Sane cap so a runaway run cannot eat memory with its log. */
|
|
104
104
|
const LOG_MAX_LINES = 600;
|
|
105
|
-
/** One evaluation at a time
|
|
106
|
-
const MAX_CONCURRENT_RUNS =
|
|
105
|
+
/** One evaluation at a time per model, but several models may run concurrently. */
|
|
106
|
+
const MAX_CONCURRENT_RUNS = 6;
|
|
107
107
|
|
|
108
108
|
/** Resolve the profile directory from the config-tree anchor (plugin-market pattern). */
|
|
109
109
|
function resolveProfileDir(ctx) {
|
|
@@ -434,6 +434,7 @@ function readJsonl(path) {
|
|
|
434
434
|
function computeResults(dataDir) {
|
|
435
435
|
const judged = new Map(); // `${model}\u0000${task}` -> {model, category, task, sum, n, time}
|
|
436
436
|
const totals = new Map(); // task -> question count
|
|
437
|
+
const answers = new Map(); // `${model}\u0000${task}` -> {answered, errors}
|
|
437
438
|
if (!existsSync(dataDir)) return { rows: [], taskTotals: {} };
|
|
438
439
|
for (const category of readDirSafe(dataDir)) {
|
|
439
440
|
const catDir = join(dataDir, category);
|
|
@@ -441,6 +442,23 @@ function computeResults(dataDir) {
|
|
|
441
442
|
const taskDir = join(catDir, task);
|
|
442
443
|
const questions = readJsonl(join(taskDir, "question.jsonl"));
|
|
443
444
|
totals.set(task, questions.length);
|
|
445
|
+
// answer side: how many answers each model produced, and how many of
|
|
446
|
+
// them were $ERROR$ (API call failures) — distinguishes "didn't run"
|
|
447
|
+
// from "ran and failed" from "ran and scored".
|
|
448
|
+
const answerDir = join(taskDir, "model_answer");
|
|
449
|
+
for (const file of readDirSafe(answerDir)) {
|
|
450
|
+
if (!file.endsWith(".jsonl")) continue;
|
|
451
|
+
for (const line of readJsonl(join(answerDir, file))) {
|
|
452
|
+
const model = typeof line.model_id === "string" ? line.model_id : file.slice(0, -".jsonl".length);
|
|
453
|
+
const turns = line.choices?.[0]?.turns;
|
|
454
|
+
const text = Array.isArray(turns) ? String(turns[0] ?? "") : "";
|
|
455
|
+
const key = `${model}\u0000${task}`;
|
|
456
|
+
const stat = answers.get(key) ?? { answered: 0, errors: 0 };
|
|
457
|
+
stat.answered += 1;
|
|
458
|
+
if (text.startsWith("$ERROR$")) stat.errors += 1;
|
|
459
|
+
answers.set(key, stat);
|
|
460
|
+
}
|
|
461
|
+
}
|
|
444
462
|
const judgments = readJsonl(join(taskDir, "model_judgment", "ground_truth_judgment.jsonl"));
|
|
445
463
|
for (const row of judgments) {
|
|
446
464
|
const model = typeof row.model === "string" ? row.model : null;
|
|
@@ -456,15 +474,29 @@ function computeResults(dataDir) {
|
|
|
456
474
|
}
|
|
457
475
|
}
|
|
458
476
|
}
|
|
459
|
-
const
|
|
460
|
-
|
|
461
|
-
|
|
462
|
-
|
|
463
|
-
|
|
464
|
-
|
|
465
|
-
|
|
466
|
-
|
|
467
|
-
|
|
477
|
+
const keys = new Set([...judged.keys(), ...answers.keys()]);
|
|
478
|
+
const rows = [...keys].map((key) => {
|
|
479
|
+
const entry = judged.get(key);
|
|
480
|
+
const stat = answers.get(key) ?? { answered: 0, errors: 0 };
|
|
481
|
+
const [model, task] = key.split("\u0000");
|
|
482
|
+
let category = entry?.category;
|
|
483
|
+
if (!category) {
|
|
484
|
+
for (const cat of readDirSafe(dataDir)) {
|
|
485
|
+
if (existsSync(join(dataDir, cat, task))) { category = cat; break; }
|
|
486
|
+
}
|
|
487
|
+
}
|
|
488
|
+
return {
|
|
489
|
+
model,
|
|
490
|
+
category: category ?? "",
|
|
491
|
+
task,
|
|
492
|
+
score: entry && entry.n > 0 ? (entry.sum / entry.n) * 100 : null,
|
|
493
|
+
judged: entry?.n ?? 0,
|
|
494
|
+
total: totals.get(task) ?? 0,
|
|
495
|
+
time: entry && entry.time > 0 ? entry.time : null,
|
|
496
|
+
answered: stat.answered,
|
|
497
|
+
errors: stat.errors,
|
|
498
|
+
};
|
|
499
|
+
});
|
|
468
500
|
const taskTotals = Object.fromEntries(totals);
|
|
469
501
|
return { rows, taskTotals };
|
|
470
502
|
}
|
|
@@ -519,16 +551,17 @@ function validateBenchName(bn) {
|
|
|
519
551
|
}
|
|
520
552
|
|
|
521
553
|
function apply(ctx) {
|
|
522
|
-
/**
|
|
523
|
-
|
|
554
|
+
/** All runs: runId -> record. Several evaluations may run concurrently. */
|
|
555
|
+
const runs = new Map();
|
|
524
556
|
|
|
525
557
|
const startRun = async (body) => {
|
|
526
558
|
const layout = livebenchLayout();
|
|
527
559
|
if (!layout.available) {
|
|
528
560
|
return { status: 409, payload: { ok: false, error: `LiveBench venv not found under ${layout.root}` } };
|
|
529
561
|
}
|
|
530
|
-
|
|
531
|
-
|
|
562
|
+
const runningCount = [...runs.values()].filter((r) => r.exitCode === null).length;
|
|
563
|
+
if (runningCount >= MAX_CONCURRENT_RUNS) {
|
|
564
|
+
return { status: 409, payload: { ok: false, error: `已有 ${runningCount} 个评测在并发运行(上限 ${MAX_CONCURRENT_RUNS}),请等待或停止部分评测` } };
|
|
532
565
|
}
|
|
533
566
|
const providerId = typeof body.provider === "string" ? body.provider : "";
|
|
534
567
|
const modelId = typeof body.model === "string" ? body.model.trim() : "";
|
|
@@ -567,11 +600,16 @@ function apply(ctx) {
|
|
|
567
600
|
}
|
|
568
601
|
}
|
|
569
602
|
|
|
570
|
-
|
|
571
|
-
|
|
603
|
+
// A global effort only applies to models whose own effort ladder
|
|
604
|
+
// includes it; others silently fall back to their default behaviour.
|
|
605
|
+
const modelMeta = provider?.models.find((m) => m.id === modelId) ?? null;
|
|
606
|
+
const effort = reasoningEffort !== null && modelMeta?.efforts?.length > 0 && !modelMeta.efforts.includes(reasoningEffort)
|
|
607
|
+
? null
|
|
608
|
+
: reasoningEffort;
|
|
609
|
+
const hasEffortSuffix = effort !== null && effort !== "off";
|
|
610
|
+
const displayName = displayModelName(providerId || "direct", modelId) + (hasEffortSuffix ? "@" + effort : "");
|
|
572
611
|
let cliModel = modelId;
|
|
573
|
-
let writeError = null;
|
|
574
|
-
// Anthropic-protocol proxies cannot go through --api-base (that path
|
|
612
|
+
let writeError = null; // Anthropic-protocol proxies cannot go through --api-base (that path
|
|
575
613
|
// speaks OpenAI Chat Completions). Instead the generated model config
|
|
576
614
|
// selects LiveBench's native anthropic client and the spawn env points
|
|
577
615
|
// the SDK at the proxy (ANTHROPIC_BASE_URL / ANTHROPIC_API_KEY).
|
|
@@ -587,7 +625,7 @@ function apply(ctx) {
|
|
|
587
625
|
writeError = writeGeneratedModelConfig(layout, loadYaml(profileDir), {
|
|
588
626
|
displayName,
|
|
589
627
|
modelId,
|
|
590
|
-
reasoningEffort: isAnthropicRoute ? null :
|
|
628
|
+
reasoningEffort: isAnthropicRoute ? null : effort,
|
|
591
629
|
protocol: isAnthropicRoute ? "anthropic" : isOpenAIResponsesRoute ? "openai_responses" : "openai",
|
|
592
630
|
});
|
|
593
631
|
if (writeError) return { status: 500, payload: { ok: false, error: writeError } };
|
|
@@ -657,14 +695,21 @@ function apply(ctx) {
|
|
|
657
695
|
});
|
|
658
696
|
|
|
659
697
|
const record = {
|
|
660
|
-
id: `${Date.now()}`,
|
|
698
|
+
id: `${Date.now()}-${Math.random().toString(36).slice(2, 6)}`,
|
|
661
699
|
proc: child,
|
|
700
|
+
displayName,
|
|
662
701
|
exitCode: null,
|
|
663
702
|
startedAt: new Date().toISOString(),
|
|
664
703
|
command: [layout.pythonExe, ...args].join(" "),
|
|
665
704
|
log: [`$ ${[...args].join(" ")}`],
|
|
666
705
|
};
|
|
667
|
-
|
|
706
|
+
runs.set(record.id, record);
|
|
707
|
+
// keep history bounded: drop oldest finished runs beyond 20 entries
|
|
708
|
+
if (runs.size > 20) {
|
|
709
|
+
for (const [id, r] of runs) {
|
|
710
|
+
if (r.exitCode !== null && runs.size > 20) runs.delete(id);
|
|
711
|
+
}
|
|
712
|
+
}
|
|
668
713
|
|
|
669
714
|
child.stdout.on("data", (chunk) => appendLog(record, chunk));
|
|
670
715
|
child.stderr.on("data", (chunk) => appendLog(record, chunk));
|
|
@@ -747,8 +792,9 @@ function apply(ctx) {
|
|
|
747
792
|
sendJson(res, 400, { ok: false, error: "invalid request body" });
|
|
748
793
|
return;
|
|
749
794
|
}
|
|
750
|
-
|
|
751
|
-
|
|
795
|
+
const runningCount = [...runs.values()].filter((r) => r.exitCode === null).length;
|
|
796
|
+
if (runningCount >= MAX_CONCURRENT_RUNS) {
|
|
797
|
+
sendJson(res, 409, { ok: false, error: `已有 ${runningCount} 个评测在并发运行(上限 ${MAX_CONCURRENT_RUNS})` });
|
|
752
798
|
return;
|
|
753
799
|
}
|
|
754
800
|
const result = await startRun(body);
|
|
@@ -764,20 +810,15 @@ function apply(ctx) {
|
|
|
764
810
|
sendJson(res, 405, { ok: false, error: "method not allowed" });
|
|
765
811
|
return;
|
|
766
812
|
}
|
|
767
|
-
|
|
768
|
-
|
|
769
|
-
|
|
770
|
-
|
|
771
|
-
|
|
772
|
-
|
|
773
|
-
|
|
774
|
-
|
|
775
|
-
|
|
776
|
-
exitCode: run.exitCode,
|
|
777
|
-
startedAt: run.startedAt,
|
|
778
|
-
command: run.command,
|
|
779
|
-
log: run.log.slice(-120).join("\n"),
|
|
780
|
-
});
|
|
813
|
+
const list = [...runs.values()].reverse().map((r) => ({
|
|
814
|
+
runId: r.id,
|
|
815
|
+
displayName: r.displayName,
|
|
816
|
+
running: r.exitCode === null,
|
|
817
|
+
exitCode: r.exitCode,
|
|
818
|
+
startedAt: r.startedAt,
|
|
819
|
+
log: r.log.slice(-80).join("\n"),
|
|
820
|
+
}));
|
|
821
|
+
sendJson(res, 200, { ok: true, running: list.some((r) => r.running), runs: list });
|
|
781
822
|
},
|
|
782
823
|
}), `${name}: status route`);
|
|
783
824
|
|
|
@@ -793,24 +834,27 @@ function apply(ctx) {
|
|
|
793
834
|
sendJson(res, 403, { ok: false, error: "forbidden origin" });
|
|
794
835
|
return;
|
|
795
836
|
}
|
|
796
|
-
|
|
837
|
+
const running = [...runs.values()].filter((r) => r.exitCode === null);
|
|
838
|
+
if (running.length === 0) {
|
|
797
839
|
sendJson(res, 200, { ok: true, stopped: false });
|
|
798
840
|
return;
|
|
799
841
|
}
|
|
800
|
-
|
|
801
|
-
|
|
802
|
-
|
|
803
|
-
|
|
804
|
-
|
|
805
|
-
|
|
806
|
-
|
|
807
|
-
|
|
842
|
+
for (const record of running) {
|
|
843
|
+
try {
|
|
844
|
+
if (process.platform === "win32") {
|
|
845
|
+
// proc.kill() only terminates run_livebench.py itself; its child
|
|
846
|
+
// (cmd → gen_api_answer.py → …) would survive. taskkill /T /F
|
|
847
|
+
// takes down the whole tree.
|
|
848
|
+
spawn("taskkill", ["/PID", String(record.proc.pid), "/T", "/F"], { windowsHide: true });
|
|
849
|
+
} else {
|
|
850
|
+
record.proc.kill();
|
|
851
|
+
}
|
|
852
|
+
} catch (error) {
|
|
853
|
+
sendJson(res, 500, { ok: false, error: String(error.message ?? error) });
|
|
854
|
+
return;
|
|
808
855
|
}
|
|
809
|
-
} catch (error) {
|
|
810
|
-
sendJson(res, 500, { ok: false, error: String(error.message ?? error) });
|
|
811
|
-
return;
|
|
812
856
|
}
|
|
813
|
-
sendJson(res, 200, { ok: true, stopped:
|
|
857
|
+
sendJson(res, 200, { ok: true, stopped: running.length });
|
|
814
858
|
},
|
|
815
859
|
}), `${name}: stop route`);
|
|
816
860
|
|
|
@@ -908,9 +952,4 @@ function apply(ctx) {
|
|
|
908
952
|
}), `${name}: delete route`);
|
|
909
953
|
}
|
|
910
954
|
|
|
911
|
-
/** Count alive runs (at most one today, kept for future parallel lanes). */
|
|
912
|
-
function runsAlive() {
|
|
913
|
-
return 1;
|
|
914
|
-
}
|
|
915
|
-
|
|
916
955
|
export { name, inject, apply, writeGeneratedModelConfig, readProviders, validateBenchName };
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "dsh-livebench-panel",
|
|
3
|
-
"version": "0.1.
|
|
3
|
+
"version": "0.1.13",
|
|
4
4
|
"description": "DSH web plugin: a LiveBench tab in the Trajectory view (right of 对话/轨迹). Run LiveBench evaluations against every model configured in the DeepSeek Harness — pick provider/model, category, task, release and question range from dropdowns, watch progress, and read scores in place.",
|
|
5
5
|
"license": "MIT",
|
|
6
6
|
"type": "module",
|