dsh-livebench-panel 0.1.11 → 0.1.13

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (3) hide show
  1. package/lib/client.js +130 -72
  2. package/lib/index.js +95 -56
  3. package/package.json +1 -1
package/lib/client.js CHANGED
@@ -92,11 +92,12 @@ window.__ModuleLoader__.load({
92
92
  function LiveBenchView() {
93
93
  const [config, setConfig] = useState(null);
94
94
  const [configError, setConfigError] = useState(null);
95
- const [sel, setSel] = useState({ provider: "", model: "", reasoning: "default", cats: [], tasks: [], release: "2024-11-25", begin: "", end: "", maxTokens: "32000" });
95
+ const [sel, setSel] = useState({ models: [], reasoning: "default", cats: [], tasks: [], release: "2024-11-25", begin: "", end: "", maxTokens: "32000" });
96
+ const [modelDdOpen, setModelDdOpen] = useState(false);
97
+ const modelDdRef = useRef(null);
96
98
  const [busy, setBusy] = useState(false);
97
99
  const [running, setRunning] = useState(false);
98
- const [log, setLog] = useState("");
99
- const [exitCode, setExitCode] = useState(null);
100
+ const [runsList, setRunsList] = useState([]);
100
101
  const [startError, setStartError] = useState(null);
101
102
  const [homeInput, setHomeInput] = useState("");
102
103
  const [homeBusy, setHomeBusy] = useState(false);
@@ -108,6 +109,16 @@ window.__ModuleLoader__.load({
108
109
  const dragKey = useRef(null);
109
110
  const logRef = useRef(null);
110
111
 
112
+ // 点击面板外部时关闭模型下拉
113
+ useEffect(() => {
114
+ if (!modelDdOpen) return;
115
+ const onDocDown = (event) => {
116
+ if (modelDdRef.current && !modelDdRef.current.contains(event.target)) setModelDdOpen(false);
117
+ };
118
+ document.addEventListener("mousedown", onDocDown);
119
+ return () => document.removeEventListener("mousedown", onDocDown);
120
+ }, [modelDdOpen]);
121
+
111
122
  const loadConfig = useCallback(async () => {
112
123
  const payload = await api("/config");
113
124
  if (payload.ok) {
@@ -115,14 +126,12 @@ window.__ModuleLoader__.load({
115
126
  setConfigError(null);
116
127
  setSel((prev) => {
117
128
  const next = { ...prev, release: payload.releases.includes(prev.release) ? prev.release : "2024-11-25" };
118
- const provider = payload.providers.find((p) => p.id === prev.provider) ?? payload.providers.find((p) => p.models.length > 0) ?? null;
119
- if (provider) {
120
- next.provider = provider.id;
121
- if (!provider.models.some((m) => m.id === prev.model)) {
122
- next.model = provider.models[0]?.id ?? "";
123
- }
124
- }
125
- if (!payload.releases.includes(next.release)) next.release = payload.releases[0] ?? "2024-11-25";
129
+ // 保留仍存在的选择,剔除失效项
130
+ next.models = prev.models.filter((v) => {
131
+ const [pid] = v.split("::");
132
+ const p = payload.providers.find((x) => x.id === pid);
133
+ return p ? true : false;
134
+ });
126
135
  return next;
127
136
  });
128
137
  } else {
@@ -138,11 +147,8 @@ window.__ModuleLoader__.load({
138
147
  const refreshStatus = useCallback(async () => {
139
148
  const payload = await api("/status");
140
149
  if (payload.ok) {
150
+ setRunsList(payload.runs ?? []);
141
151
  setRunning(payload.running === true);
142
- if (payload.hasRun) {
143
- setLog(payload.log ?? "");
144
- setExitCode(payload.exitCode);
145
- }
146
152
  return payload.running === true;
147
153
  }
148
154
  return false;
@@ -165,10 +171,6 @@ window.__ModuleLoader__.load({
165
171
  return () => clearInterval(timer);
166
172
  }, [refreshStatus, loadResults]);
167
173
 
168
- useEffect(() => {
169
- if (logRef.current) logRef.current.scrollTop = logRef.current.scrollHeight;
170
- }, [log]);
171
-
172
174
  const categories = useMemo(() => (config ? Object.keys(config.tasks) : []), [config]);
173
175
  // 有效题数:LiveBench 会丢弃「发布晚于所选 release」和「在所选 release
174
176
  // 前已退役」的题目,这里按题目桶 (发布日, 移除日) 精确复算。
@@ -197,21 +199,31 @@ window.__ModuleLoader__.load({
197
199
  const next = list.includes(value) ? list.filter((v) => v !== value) : [...list, value];
198
200
  return { ...prev, [key]: next };
199
201
  });
200
- const selectedProvider = useMemo(
201
- () => (config ? config.providers.find((p) => p.id === sel.provider) ?? null : null),
202
- [config, sel.provider],
203
- );
204
- const selectedModel = useMemo(
205
- () => selectedProvider?.models.find((m) => m.id === sel.model) ?? null,
206
- [selectedProvider, sel.model],
207
- );
202
+ // 已选模型(provider::model 值列表)的解析与 effort 并集
203
+ const selectedModelEntries = useMemo(() => sel.models.map((value) => {
204
+ const [pid, ...rest] = value.split("::");
205
+ const mid = rest.join("::");
206
+ const provider = config?.providers.find((p) => p.id === pid) ?? null;
207
+ const model = provider?.models.find((m) => m.id === mid) ?? null;
208
+ return { providerId: pid, modelId: mid, efforts: model?.efforts ?? [] };
209
+ }).filter((entry) => entry.modelId !== ""), [sel.models, config]);
208
210
  const effortOptions = useMemo(() => {
209
- const efforts = selectedModel?.efforts ?? [];
210
- return efforts.map((e) => ({ value: e, label: e === "off" ? "off(关闭思考)" : e }));
211
- }, [selectedModel]);
211
+ const seen = new Set();
212
+ const union = [];
213
+ for (const entry of selectedModelEntries) {
214
+ for (const e of entry.efforts) {
215
+ if (!seen.has(e)) { seen.add(e); union.push({ value: e, label: e === "off" ? "off(关闭思考)" : e }); }
216
+ }
217
+ }
218
+ return union;
219
+ }, [selectedModelEntries]);
212
220
 
213
221
  const onStart = async () => {
214
222
  setStartError(null);
223
+ if (selectedModelEntries.length === 0) {
224
+ setStartError("请先在下拉框中至少选择一个模型。");
225
+ return;
226
+ }
215
227
  const benchNames = [];
216
228
  if (sel.cats.length === 0) {
217
229
  benchNames.push("live_bench");
@@ -227,22 +239,28 @@ window.__ModuleLoader__.load({
227
239
  }
228
240
  setBusy(true);
229
241
  try {
230
- const payload = await api("/start", {
231
- method: "POST",
232
- headers: { "content-type": "application/json" },
233
- body: JSON.stringify({
234
- provider: sel.provider,
235
- model: sel.model,
236
- reasoningEffort: sel.reasoning,
237
- benchNames,
238
- release: sel.release,
239
- begin: sel.begin,
240
- end: sel.end,
241
- maxTokens: sel.maxTokens,
242
- }),
242
+ // 每个 selected 模型并发启动一个评测(服务端有并发上限保护)
243
+ const failures = [];
244
+ const launches = selectedModelEntries.map(async (entry) => {
245
+ const payload = await api("/start", {
246
+ method: "POST",
247
+ headers: { "content-type": "application/json" },
248
+ body: JSON.stringify({
249
+ provider: entry.providerId,
250
+ model: entry.modelId,
251
+ reasoningEffort: sel.reasoning,
252
+ benchNames,
253
+ release: sel.release,
254
+ begin: sel.begin,
255
+ end: sel.end,
256
+ maxTokens: sel.maxTokens,
257
+ }),
258
+ });
259
+ if (!payload.ok) failures.push(`${entry.modelId}: ${payload.error ?? "启动失败"}`);
243
260
  });
244
- if (!payload.ok) setStartError(payload.error ?? "启动失败");
245
- else await refreshStatus();
261
+ await Promise.all(launches);
262
+ if (failures.length > 0) setStartError(failures.join(";"));
263
+ await refreshStatus();
246
264
  } catch (error) {
247
265
  setStartError(String(error.message ?? error));
248
266
  } finally {
@@ -275,14 +293,7 @@ window.__ModuleLoader__.load({
275
293
 
276
294
  const setField = (key) => (event) => {
277
295
  const value = event.target.value;
278
- setSel((prev) => {
279
- const next = { ...prev, [key]: value };
280
- if (key === "provider") {
281
- const provider = config?.providers.find((p) => p.id === value) ?? null;
282
- next.model = provider?.models[0]?.id ?? "";
283
- }
284
- return next;
285
- });
296
+ setSel((prev) => ({ ...prev, [key]: value }));
286
297
  };
287
298
 
288
299
  // ---- 成绩矩阵:model 为行标识、category/task 为列标识 ----
@@ -385,18 +396,48 @@ window.__ModuleLoader__.load({
385
396
  configError !== null && h("p", { className: c("error") }, `加载配置失败:${configError}`),
386
397
  h("div", { className: c("card") },
387
398
  h("div", { className: c("grid") },
388
- h("div", { className: c("field") },
389
- h("label", { className: c("label") }, "模型(harness 全部模型)"),
390
- h("select", { className: c("select"), value: sel.provider + "::" + sel.model, onChange: (event) => {
391
- const [providerId, ...rest] = event.target.value.split("::");
392
- const modelId = rest.join("::");
393
- setSel((prev) => ({ ...prev, provider: providerId, model: modelId, reasoning: "default" }));
394
- } }, modelOptions(config?.providers ?? [])),
399
+ h("div", { className: c("field"), ref: modelDdRef, style: { position: "relative" } },
400
+ h("label", { className: c("label") }, "模型(harness 全部模型,可多选)"),
401
+ h("button", {
402
+ type: "button", className: c("select"), style: { textAlign: "left", width: "100%", overflow: "hidden", textOverflow: "ellipsis", whiteSpace: "nowrap" },
403
+ onClick: () => setModelDdOpen((v) => !v),
404
+ }, sel.models.length === 0
405
+ ? "点击选择模型…"
406
+ : sel.models.length === 1
407
+ ? sel.models[0].split("::")[1]
408
+ : `已选 ${sel.models.length} 个模型`),
409
+ modelDdOpen && h("div", {
410
+ style: {
411
+ position: "absolute", zIndex: 30, top: "calc(100% + 4px)", left: 0, right: 0,
412
+ maxHeight: "320px", overflowY: "auto", border: "1px solid var(--dsw-alias-border-l2)",
413
+ borderRadius: "8px", background: "var(--dsw-alias-bg-layer-1)", padding: "6px",
414
+ boxShadow: "var(--dsw-shadow-lv1, 0 4px 16px rgba(0,0,0,.25))",
415
+ },
416
+ },
417
+ (config?.providers ?? []).map((p) => h("div", { key: p.id, style: { marginBottom: "4px" } },
418
+ h("div", { style: { fontSize: "11px", color: "var(--dsw-alias-label-tertiary)", padding: "3px 4px" } },
419
+ p.name + (p.routable ? "" : " ·未接LiveBench")),
420
+ p.models.map((m) => {
421
+ const value = p.id + "::" + m.id;
422
+ const checked = sel.models.includes(value);
423
+ return h("label", {
424
+ key: value,
425
+ style: { display: "flex", alignItems: "center", gap: "6px", padding: "3px 4px", cursor: "pointer", borderRadius: "6px" },
426
+ },
427
+ h("input", { type: "checkbox", checked, onChange: () => {
428
+ setSel((prev) => {
429
+ const set = new Set(prev.models);
430
+ if (set.has(value)) set.delete(value); else set.add(value);
431
+ return { ...prev, models: [...set] };
432
+ });
433
+ } }),
434
+ h("span", { style: { fontSize: "12.5px" } }, m.name));
435
+ }))),
395
436
  ),
396
437
  h("div", { className: c("field") },
397
438
  h("label", { className: c("label") }, "推理强度"),
398
439
  h("select", { className: c("select"), value: sel.reasoning, onChange: setField("reasoning"), disabled: effortOptions.length === 0 },
399
- h("option", { value: "default" }, effortOptions.length === 0 ? "(该模型不支持)" : "(模型默认)"),
440
+ h("option", { value: "default" }, effortOptions.length === 0 ? "(按模型默认)" : "(模型默认)"),
400
441
  effortOptions.map((e) => h("option", { key: e.value, value: e.value }, e.label))),
401
442
  ),
402
443
  h("div", { className: c("field") },
@@ -443,22 +484,25 @@ window.__ModuleLoader__.load({
443
484
  ),
444
485
  ),
445
486
  h("div", { className: c("row") },
446
- h("button", { className: c("btn"), onClick: onStart, disabled: busy || running || !config?.available || sel.model === "" },
447
- running ? "评测运行中…" : busy ? "启动中…" : "开始评测"),
448
- running && h("button", { className: c("btn") + " " + c("btnDanger"), onClick: onStop }, "停止"),
487
+ h("button", { className: c("btn"), onClick: onStart, disabled: busy || running || !config?.available || sel.models.length === 0 },
488
+ running ? `评测运行中(可另起)…` : busy ? "启动中…" : `开始评测(${sel.models.length} 个模型并发)`),
489
+ running && h("button", { className: c("btn") + " " + c("btnDanger"), onClick: onStop }, "停止全部"),
449
490
  h("button", { className: c("btnGhost") + " " + c("btn"), onClick: () => { loadConfig(); loadResults(); } }, "刷新"),
450
- selectedProvider && !selectedProvider.routable && h("span", { className: c("hint") },
451
- "该 provider 的协议或端点未知,无法自动路由:LiveBench 将按模型名原生尝试,未注册的模型名会失败。"),
452
491
  ),
453
492
  startError !== null && h("p", { className: c("error") }, startError),
454
493
  ),
455
- (log.length > 0 || running) && h("div", { className: c("card") },
494
+ runsList.length > 0 && h("div", { className: c("card") },
456
495
  h("div", { className: c("row") },
457
496
  h("span", { className: c("badge"), "data-ok": running ? "1" : "0" },
458
- running ? "运行中" : exitCode === 0 ? "已完成" : exitCode === null ? "待运行" : `已退出 (${exitCode})`),
497
+ running ? "有评测运行中" : "无运行中的评测"),
459
498
  ),
460
- h("pre", { className: c("log"), ref: logRef }, log || "(暂无输出)"),
461
- ),
499
+ runsList.map((r) => h("details", { key: r.runId, open: r.running },
500
+ h("summary", { style: { cursor: "pointer", fontSize: "12px", color: "var(--dsw-alias-label-secondary)" } },
501
+ h("span", { className: c("badge"), "data-ok": r.running ? "1" : "0" }, r.running ? "运行中" : r.exitCode === 0 ? "完成" : `退出(${r.exitCode})`),
502
+ " ",
503
+ r.displayName),
504
+ h("pre", { className: c("log"), ref: r.running ? logRef : null }, r.log || "(暂无输出)"),
505
+ )))),
462
506
  sortedModels.length > 0 && h("div", { className: c("card") },
463
507
  h("div", { className: c("row") },
464
508
  h("span", { className: c("label") }, "评测成绩(model 为行;分数 = 平均分 ×100;行首可拖动排序,顺序保存在本机)"),
@@ -498,8 +542,22 @@ window.__ModuleLoader__.load({
498
542
  h("td", null, model),
499
543
  matrix.tasks.map((taskKeyCol) => {
500
544
  const row = matrix.cells.get(`${model}\u0000${taskKeyCol}`);
501
- return h("td", { key: taskKeyCol, className: c("score") },
502
- row ? `${row.score.toFixed(1)} (${row.judged}/${row.total})` : "");
545
+ // ×= 该模型未做过此任务;E = 做了但全部 API 失败
546
+ let cellText = "×";
547
+ let title = "该模型未执行此任务";
548
+ if (row) {
549
+ if (row.answered > 0 && row.errors >= row.answered) {
550
+ cellText = `E (${row.errors})`;
551
+ title = "该任务全部回答均 API 失败($ERROR$),计 0 分";
552
+ } else if (row.judged === 0) {
553
+ cellText = "未判分";
554
+ title = "有答案但尚未判分";
555
+ } else {
556
+ cellText = `${row.score.toFixed(1)} (${row.judged}/${row.total})`;
557
+ if (row.errors > 0) title += ` · 其中 ${row.errors} 题 API 失败计 0 分`;
558
+ }
559
+ }
560
+ return h("td", { key: taskKeyCol, className: c("score"), title }, cellText);
503
561
  }),
504
562
  h("td", null, h("button", {
505
563
  className: c("btnGhost") + " " + c("btn"), title: "删除该模型的所有评测成绩",
package/lib/index.js CHANGED
@@ -102,8 +102,8 @@ const RELEASES = [
102
102
  ];
103
103
  /** Sane cap so a runaway run cannot eat memory with its log. */
104
104
  const LOG_MAX_LINES = 600;
105
- /** One evaluation at a time LiveBench grading is not cheap. */
106
- const MAX_CONCURRENT_RUNS = 1;
105
+ /** One evaluation at a time per model, but several models may run concurrently. */
106
+ const MAX_CONCURRENT_RUNS = 6;
107
107
 
108
108
  /** Resolve the profile directory from the config-tree anchor (plugin-market pattern). */
109
109
  function resolveProfileDir(ctx) {
@@ -434,6 +434,7 @@ function readJsonl(path) {
434
434
  function computeResults(dataDir) {
435
435
  const judged = new Map(); // `${model}\u0000${task}` -> {model, category, task, sum, n, time}
436
436
  const totals = new Map(); // task -> question count
437
+ const answers = new Map(); // `${model}\u0000${task}` -> {answered, errors}
437
438
  if (!existsSync(dataDir)) return { rows: [], taskTotals: {} };
438
439
  for (const category of readDirSafe(dataDir)) {
439
440
  const catDir = join(dataDir, category);
@@ -441,6 +442,23 @@ function computeResults(dataDir) {
441
442
  const taskDir = join(catDir, task);
442
443
  const questions = readJsonl(join(taskDir, "question.jsonl"));
443
444
  totals.set(task, questions.length);
445
+ // answer side: how many answers each model produced, and how many of
446
+ // them were $ERROR$ (API call failures) — distinguishes "didn't run"
447
+ // from "ran and failed" from "ran and scored".
448
+ const answerDir = join(taskDir, "model_answer");
449
+ for (const file of readDirSafe(answerDir)) {
450
+ if (!file.endsWith(".jsonl")) continue;
451
+ for (const line of readJsonl(join(answerDir, file))) {
452
+ const model = typeof line.model_id === "string" ? line.model_id : file.slice(0, -".jsonl".length);
453
+ const turns = line.choices?.[0]?.turns;
454
+ const text = Array.isArray(turns) ? String(turns[0] ?? "") : "";
455
+ const key = `${model}\u0000${task}`;
456
+ const stat = answers.get(key) ?? { answered: 0, errors: 0 };
457
+ stat.answered += 1;
458
+ if (text.startsWith("$ERROR$")) stat.errors += 1;
459
+ answers.set(key, stat);
460
+ }
461
+ }
444
462
  const judgments = readJsonl(join(taskDir, "model_judgment", "ground_truth_judgment.jsonl"));
445
463
  for (const row of judgments) {
446
464
  const model = typeof row.model === "string" ? row.model : null;
@@ -456,15 +474,29 @@ function computeResults(dataDir) {
456
474
  }
457
475
  }
458
476
  }
459
- const rows = [...judged.values()].map((entry) => ({
460
- model: entry.model,
461
- category: entry.category,
462
- task: entry.task,
463
- score: entry.n > 0 ? (entry.sum / entry.n) * 100 : 0,
464
- judged: entry.n,
465
- total: totals.get(entry.task) ?? 0,
466
- time: entry.time > 0 ? entry.time : null,
467
- }));
477
+ const keys = new Set([...judged.keys(), ...answers.keys()]);
478
+ const rows = [...keys].map((key) => {
479
+ const entry = judged.get(key);
480
+ const stat = answers.get(key) ?? { answered: 0, errors: 0 };
481
+ const [model, task] = key.split("\u0000");
482
+ let category = entry?.category;
483
+ if (!category) {
484
+ for (const cat of readDirSafe(dataDir)) {
485
+ if (existsSync(join(dataDir, cat, task))) { category = cat; break; }
486
+ }
487
+ }
488
+ return {
489
+ model,
490
+ category: category ?? "",
491
+ task,
492
+ score: entry && entry.n > 0 ? (entry.sum / entry.n) * 100 : null,
493
+ judged: entry?.n ?? 0,
494
+ total: totals.get(task) ?? 0,
495
+ time: entry && entry.time > 0 ? entry.time : null,
496
+ answered: stat.answered,
497
+ errors: stat.errors,
498
+ };
499
+ });
468
500
  const taskTotals = Object.fromEntries(totals);
469
501
  return { rows, taskTotals };
470
502
  }
@@ -519,16 +551,17 @@ function validateBenchName(bn) {
519
551
  }
520
552
 
521
553
  function apply(ctx) {
522
- /** The single run slot. */
523
- let run = null;
554
+ /** All runs: runId -> record. Several evaluations may run concurrently. */
555
+ const runs = new Map();
524
556
 
525
557
  const startRun = async (body) => {
526
558
  const layout = livebenchLayout();
527
559
  if (!layout.available) {
528
560
  return { status: 409, payload: { ok: false, error: `LiveBench venv not found under ${layout.root}` } };
529
561
  }
530
- if (run && run.exitCode === null) {
531
- return { status: 409, payload: { ok: false, error: "another evaluation is already running" } };
562
+ const runningCount = [...runs.values()].filter((r) => r.exitCode === null).length;
563
+ if (runningCount >= MAX_CONCURRENT_RUNS) {
564
+ return { status: 409, payload: { ok: false, error: `已有 ${runningCount} 个评测在并发运行(上限 ${MAX_CONCURRENT_RUNS}),请等待或停止部分评测` } };
532
565
  }
533
566
  const providerId = typeof body.provider === "string" ? body.provider : "";
534
567
  const modelId = typeof body.model === "string" ? body.model.trim() : "";
@@ -567,11 +600,16 @@ function apply(ctx) {
567
600
  }
568
601
  }
569
602
 
570
- const hasEffortSuffix = reasoningEffort !== null && reasoningEffort !== "off";
571
- const displayName = displayModelName(providerId || "direct", modelId) + (hasEffortSuffix ? "@" + reasoningEffort : "");
603
+ // A global effort only applies to models whose own effort ladder
604
+ // includes it; others silently fall back to their default behaviour.
605
+ const modelMeta = provider?.models.find((m) => m.id === modelId) ?? null;
606
+ const effort = reasoningEffort !== null && modelMeta?.efforts?.length > 0 && !modelMeta.efforts.includes(reasoningEffort)
607
+ ? null
608
+ : reasoningEffort;
609
+ const hasEffortSuffix = effort !== null && effort !== "off";
610
+ const displayName = displayModelName(providerId || "direct", modelId) + (hasEffortSuffix ? "@" + effort : "");
572
611
  let cliModel = modelId;
573
- let writeError = null;
574
- // Anthropic-protocol proxies cannot go through --api-base (that path
612
+ let writeError = null; // Anthropic-protocol proxies cannot go through --api-base (that path
575
613
  // speaks OpenAI Chat Completions). Instead the generated model config
576
614
  // selects LiveBench's native anthropic client and the spawn env points
577
615
  // the SDK at the proxy (ANTHROPIC_BASE_URL / ANTHROPIC_API_KEY).
@@ -587,7 +625,7 @@ function apply(ctx) {
587
625
  writeError = writeGeneratedModelConfig(layout, loadYaml(profileDir), {
588
626
  displayName,
589
627
  modelId,
590
- reasoningEffort: isAnthropicRoute ? null : reasoningEffort,
628
+ reasoningEffort: isAnthropicRoute ? null : effort,
591
629
  protocol: isAnthropicRoute ? "anthropic" : isOpenAIResponsesRoute ? "openai_responses" : "openai",
592
630
  });
593
631
  if (writeError) return { status: 500, payload: { ok: false, error: writeError } };
@@ -657,14 +695,21 @@ function apply(ctx) {
657
695
  });
658
696
 
659
697
  const record = {
660
- id: `${Date.now()}`,
698
+ id: `${Date.now()}-${Math.random().toString(36).slice(2, 6)}`,
661
699
  proc: child,
700
+ displayName,
662
701
  exitCode: null,
663
702
  startedAt: new Date().toISOString(),
664
703
  command: [layout.pythonExe, ...args].join(" "),
665
704
  log: [`$ ${[...args].join(" ")}`],
666
705
  };
667
- run = record;
706
+ runs.set(record.id, record);
707
+ // keep history bounded: drop oldest finished runs beyond 20 entries
708
+ if (runs.size > 20) {
709
+ for (const [id, r] of runs) {
710
+ if (r.exitCode !== null && runs.size > 20) runs.delete(id);
711
+ }
712
+ }
668
713
 
669
714
  child.stdout.on("data", (chunk) => appendLog(record, chunk));
670
715
  child.stderr.on("data", (chunk) => appendLog(record, chunk));
@@ -747,8 +792,9 @@ function apply(ctx) {
747
792
  sendJson(res, 400, { ok: false, error: "invalid request body" });
748
793
  return;
749
794
  }
750
- if (run !== null && run.exitCode === null && runsAlive() >= MAX_CONCURRENT_RUNS) {
751
- sendJson(res, 409, { ok: false, error: "another evaluation is already running" });
795
+ const runningCount = [...runs.values()].filter((r) => r.exitCode === null).length;
796
+ if (runningCount >= MAX_CONCURRENT_RUNS) {
797
+ sendJson(res, 409, { ok: false, error: `已有 ${runningCount} 个评测在并发运行(上限 ${MAX_CONCURRENT_RUNS})` });
752
798
  return;
753
799
  }
754
800
  const result = await startRun(body);
@@ -764,20 +810,15 @@ function apply(ctx) {
764
810
  sendJson(res, 405, { ok: false, error: "method not allowed" });
765
811
  return;
766
812
  }
767
- if (run === null) {
768
- sendJson(res, 200, { ok: true, running: false, hasRun: false });
769
- return;
770
- }
771
- sendJson(res, 200, {
772
- ok: true,
773
- running: run.exitCode === null,
774
- hasRun: true,
775
- runId: run.id,
776
- exitCode: run.exitCode,
777
- startedAt: run.startedAt,
778
- command: run.command,
779
- log: run.log.slice(-120).join("\n"),
780
- });
813
+ const list = [...runs.values()].reverse().map((r) => ({
814
+ runId: r.id,
815
+ displayName: r.displayName,
816
+ running: r.exitCode === null,
817
+ exitCode: r.exitCode,
818
+ startedAt: r.startedAt,
819
+ log: r.log.slice(-80).join("\n"),
820
+ }));
821
+ sendJson(res, 200, { ok: true, running: list.some((r) => r.running), runs: list });
781
822
  },
782
823
  }), `${name}: status route`);
783
824
 
@@ -793,24 +834,27 @@ function apply(ctx) {
793
834
  sendJson(res, 403, { ok: false, error: "forbidden origin" });
794
835
  return;
795
836
  }
796
- if (run === null || run.exitCode !== null) {
837
+ const running = [...runs.values()].filter((r) => r.exitCode === null);
838
+ if (running.length === 0) {
797
839
  sendJson(res, 200, { ok: true, stopped: false });
798
840
  return;
799
841
  }
800
- try {
801
- if (process.platform === "win32") {
802
- // proc.kill() only terminates run_livebench.py itself; its child
803
- // (cmd gen_api_answer.py …) would survive. taskkill /T /F
804
- // takes down the whole tree.
805
- spawn("taskkill", ["/PID", String(run.proc.pid), "/T", "/F"], { windowsHide: true });
806
- } else {
807
- run.proc.kill();
842
+ for (const record of running) {
843
+ try {
844
+ if (process.platform === "win32") {
845
+ // proc.kill() only terminates run_livebench.py itself; its child
846
+ // (cmd gen_api_answer.py …) would survive. taskkill /T /F
847
+ // takes down the whole tree.
848
+ spawn("taskkill", ["/PID", String(record.proc.pid), "/T", "/F"], { windowsHide: true });
849
+ } else {
850
+ record.proc.kill();
851
+ }
852
+ } catch (error) {
853
+ sendJson(res, 500, { ok: false, error: String(error.message ?? error) });
854
+ return;
808
855
  }
809
- } catch (error) {
810
- sendJson(res, 500, { ok: false, error: String(error.message ?? error) });
811
- return;
812
856
  }
813
- sendJson(res, 200, { ok: true, stopped: true });
857
+ sendJson(res, 200, { ok: true, stopped: running.length });
814
858
  },
815
859
  }), `${name}: stop route`);
816
860
 
@@ -908,9 +952,4 @@ function apply(ctx) {
908
952
  }), `${name}: delete route`);
909
953
  }
910
954
 
911
- /** Count alive runs (at most one today, kept for future parallel lanes). */
912
- function runsAlive() {
913
- return 1;
914
- }
915
-
916
955
  export { name, inject, apply, writeGeneratedModelConfig, readProviders, validateBenchName };
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "dsh-livebench-panel",
3
- "version": "0.1.11",
3
+ "version": "0.1.13",
4
4
  "description": "DSH web plugin: a LiveBench tab in the Trajectory view (right of 对话/轨迹). Run LiveBench evaluations against every model configured in the DeepSeek Harness — pick provider/model, category, task, release and question range from dropdowns, watch progress, and read scores in place.",
5
5
  "license": "MIT",
6
6
  "type": "module",