dsh-livebench-panel 0.2.6 → 0.2.8

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (2) hide show
  1. package/lib/index.js +10 -5
  2. package/package.json +1 -1
package/lib/index.js CHANGED
@@ -691,7 +691,12 @@ function apply(ctx) {
691
691
  ? null
692
692
  : reasoningEffort;
693
693
  const hasEffortSuffix = effort !== null && effort !== "off";
694
- const displayName = displayModelName(providerId || "direct", modelId) + (hasEffortSuffix ? "@" + effort : "");
694
+ // 每次评测使用带时间戳的独立 display-name:答案/判分文件各自独立,
695
+ // 成绩表里每次评测是独立一行,可对比模型的历史变化(不再覆盖旧数据)。
696
+ const _now = new Date();
697
+ const _pad = (n) => String(n).padStart(2, "0");
698
+ const runStamp = `r${_now.getFullYear()}${_pad(_now.getMonth() + 1)}${_pad(_now.getDate())}-${_pad(_now.getHours())}${_pad(_now.getMinutes())}${_pad(_now.getSeconds())}`;
699
+ const displayName = displayModelName(providerId || "direct", modelId) + (hasEffortSuffix ? "@" + effort : "") + "__" + runStamp;
695
700
  let cliModel = modelId;
696
701
  let writeError = null; // Anthropic-protocol proxies cannot go through --api-base (that path
697
702
  // speaks OpenAI Chat Completions). Instead the generated model config
@@ -725,7 +730,7 @@ function apply(ctx) {
725
730
  ...(benchNames !== null ? benchNames : [benchParts.join("/")]),
726
731
  "--livebench-release-option", release,
727
732
  "--max-tokens", String(asInt(body.maxTokens, 256, 32768, 32000)),
728
- "--parallel-requests", String(asInt(body.parallel, 1, 8, 2)),
733
+ "--parallel-requests", String(asInt(body.parallel, 1, 8, 1)),
729
734
  // 流式:中转网关(Cloudflare 等)对非流式请求有 ~100s 超时(524),
730
735
  // 高推理强度模型思考数分钟必然超时。流式保持字节流动可规避。
731
736
  "--stream",
@@ -831,18 +836,18 @@ function apply(ctx) {
831
836
  // 自动补跑:正常结束但存在 $ERROR$ 答案时,用 --resume --retry-failures
832
837
  // 只重跑失败题(最多 2 轮)。统计改为异步流式(同步版曾阻塞事件循环,
833
838
  // 且函数缺失会在 close 回调抛未捕获异常 → DSH 假死断连)。
834
- if (record.exitCode === 0 && record.autoRetries < 2 && record.body) {
839
+ if (record.exitCode === 0 && record.autoRetries < 3 && record.body) {
835
840
  (async () => {
836
841
  try {
837
842
  const errors = await countErrorAnswersAsync(layout.dataDir, record.displayName, record.body);
838
843
  if (errors > 0) {
839
- record.log.push(`[自动补跑] 检测到 ${errors} 条失败答案,自动重试(第 ${record.autoRetries + 1}/2 轮)`);
844
+ record.log.push(`[自动补跑] 检测到 ${errors} 条失败答案,自动重试(第 ${record.autoRetries + 1}/3 轮)`);
840
845
  const retryBody = { ...record.body, resume: true, retryFailures: true };
841
846
  setTimeout(() => {
842
847
  startRun(retryBody, record.autoRetries + 1).catch((e) => {
843
848
  appendLog(record, `[自动补跑启动失败] ${e.message}`);
844
849
  });
845
- }, 3000);
850
+ }, 15000);
846
851
  }
847
852
  } catch (e) {
848
853
  record.log.push(`[自动补跑检查失败] ${e.message}`);
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "dsh-livebench-panel",
3
- "version": "0.2.6",
3
+ "version": "0.2.8",
4
4
  "description": "DSH web plugin: a LiveBench tab in the Trajectory view (right of 对话/轨迹). Run LiveBench evaluations against every model configured in the DeepSeek Harness — pick provider/model, category, task, release and question range from dropdowns, watch progress, and read scores in place.",
5
5
  "license": "MIT",
6
6
  "type": "module",