dsh-livebench-panel 0.2.32 → 0.2.33

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (3) hide show
  1. package/README.md +13 -12
  2. package/lib/index.js +14 -37
  3. package/package.json +1 -1
package/README.md CHANGED
@@ -87,19 +87,20 @@ cd livebench
87
87
  - 插件向 `livebench/model/model_configs/dsh_panel_generated__<display-name>.yaml` 写入一条模型配置(每个模型一个文件,避免并发写同名文件互相覆盖),经 LiveBench 的 `api_kwargs.default.reasoning_effort` 透传给 API(`off` 表示不透传、由后端走默认);
88
88
  - 强度编码进 display-name(如 `code-gpt__gpt-5.6-sol@high`),**不同强度在成绩表中是独立条目**,可直接对比。
89
89
  - **参数下拉框**:题集 release(LiveBench 全部 releases)、分类(coding/math/reasoning/language/data_analysis/instruction_following)、任务(随分类联动)、题目序号范围、max-tokens(**默认 32000**,自动夹到模型声明的上限;LiveBench 自带默认只有 4096,推理模型会思考吃满、正文为空,难题可手动调到 65536)、**并发请求数**(`--parallel-requests`,1/2/3/4/6/8,默认 1=串行;模型单题思考慢时并发多题是唯一能等比缩短总时长的开关)。
90
- - **协议路由**:**Anthropic/Claude 系模型一律走 `anthropic-messages`**(POST `<baseURL>/v1/messages`,
91
- 与会话窗口同协议;`baseURL` 会自动剥掉 `/v1`、`/v1/messages` 后缀,因为 anthropic SDK 自己会拼);
90
+ - **协议路由**:**Anthropic/Claude 系模型一律走 `anthropic-messages`**(POST `<baseURL>/v1/messages`;
91
+ `baseURL` 会自动剥掉 `/v1`、`/v1/messages` 后缀,因为 anthropic SDK 自己会拼);
92
92
  其余模型按 settings.yaml 的 `api` 走 `openai-completions` / `openai-responses` / 内置端点。
93
- 运行日志首行打印 `[route] …`,写明协议、目标地址、`max_tokens` 与 thinking 形状。
94
- - **请求形状对齐会话窗口**:anthropic 通道按强度档附上
95
- `thinking: {type: enabled, budget_tokens: N}`(minimal 1024 / low 2048 / medium 8192 /
96
- high·xhigh·max 16384;`off` → `{type: disabled}`),并因此**不发 `temperature`**;
97
- 非 anthropic 通道按渠道习惯选长度上限字段(aiportx 这类中转用 `max_completion_tokens`),
98
- 且 `--max-tokens` 会夹到该模型在 settings.yaml 里声明的上限。
99
- - **流式看门狗不会误杀健康请求**:判活标准是"连接上是否还有字节"(含网关每 3 秒的 `ping`),
100
- 默认 600s 无字节才断开重试(可用渠道声明的 `streamIdleTimeoutMs` 覆写);
101
- 另有两道硬边界——单次请求总时长上限(`LIVEBENCH_STREAM_MAX_SECONDS`,默认 1800s)
102
- 与首个字节等待上限(`LIVEBENCH_FIRST_BYTE_TIMEOUT`,默认 600s)。
93
+ 运行日志首行打印 `[route] …`,写明协议、目标地址与 `max_tokens`。
94
+ - **不替上游做形状决定**(实测结论):面板**不注入 `thinking`**、**不注入长度上限字段**。
95
+ 中转对这两样都不当真 —— 注入 `thinking` 后 aiportx 只会回空的 thinking 片段(一题 11~22 分钟、
96
+ 正文 0~75 字符、判 0);`max_tokens` / `max_completion_tokens` 写 512 实测吐 4062 / 4737 token。
97
+ 保持 LiveBench 的原生形状最稳;`--max-tokens` 仍会夹到模型在 settings.yaml 里声明的上限。
98
+ - **流式看门狗三层判死,且不会误杀健康请求**:① 连接上任何字节都算活(含每 3 秒的 `ping`),
99
+ `LIVEBENCH_STREAM_IDLE_TIMEOUT` 默认 600s(可用渠道声明的 `streamIdleTimeoutMs` 覆写);
100
+ ② **内容**维度:`LIVEBENCH_CONTENT_IDLE_SECONDS` 默认 480s 没有任何正文/思考片段就断开
101
+ —— 中转会连发 ping 十几分钟不给一个字,只按字节判活对它无效;
102
+ ③ 总时长上限 `LIVEBENCH_STREAM_MAX_SECONDS` 默认 900s;首个字节等待上限
103
+ `LIVEBENCH_FIRST_BYTE_TIMEOUT` 默认 600s。
103
104
  - **进度心跳**:模型长时间思考时网关只发 `ping`、正文迟迟不来,日志里会每 30 秒出现一行
104
105
  `仍在接收: 已 120s,连接累计 12345 字节(模型思考中,正文未到)`,长思考不会"看起来卡死"
105
106
  (`LIVEBENCH_PROGRESS_SECONDS`,0 = 关闭)。
package/lib/index.js CHANGED
@@ -614,40 +614,21 @@ function writeGeneratedModelConfig(layout, YAML, { displayName, modelId, reasoni
614
614
  `display_name: ${displayName}`,
615
615
  "api_name:",
616
616
  ];
617
- // 输出上限的字段名:写进 api_kwargs 后,LiveBench 的
618
- // `if 'max_tokens' not in api_kwargs and 'max_completion_tokens' not in api_kwargs`
619
- // 分支会被跳过,于是"用哪个字段"完全由面板决定(见 maxTokensFieldFor)。
620
- const limitLine = protocol === "openai" && maxTokensField === "max_completion_tokens"
621
- && Number.isFinite(Number(maxTokens)) && Number(maxTokens) > 0
622
- ? ` max_completion_tokens: ${Number(maxTokens)}`
623
- : null;
617
+ // 长度上限:**不再由面板注入字段名**。
618
+ // 早先为对齐会话窗口写过 `max_completion_tokens`,但实测 aiportx 这类中转对
619
+ // `max_tokens` 与 `max_completion_tokens` **两个字段都不执行**(请求写 512,
620
+ // 实测吐 4062 / 4737 token),写进去只是"假装有上限",还容易让人误判。
621
+ // 交回 LiveBench 自己的默认(非 gpt 模型发 max_tokens)即可。
622
+ const limitLine = null;
624
623
  if (protocol === "anthropic") {
625
624
  // Route through LiveBench's native anthropic client; the endpoint is
626
625
  // pointed at the provider proxy via ANTHROPIC_BASE_URL (spawn env).
627
626
  lines.push(` anthropic: ${modelId}`, "default_provider: anthropic");
628
- // 会话窗口在这个通道上固定带 thinking(并因此不带 temperature),面板照抄,
629
- // 否则上游会走"自适应思考",一题能安静地想十几分钟(见 thinkingBudgetForEffort)。
630
- if (Number.isFinite(Number(thinkingBudget)) && Number(thinkingBudget) > 0) {
631
- lines.push(
632
- "api_kwargs:",
633
- " default:",
634
- " thinking:",
635
- " type: enabled",
636
- ` budget_tokens: ${Number(thinkingBudget)}`,
637
- // thinking 与 temperature 互斥:显式置 None → LiveBench 映射成 NOT_GIVEN 而不发送
638
- " temperature: null",
639
- );
640
- } else if (thinkingDisabled) {
641
- // 用户把强度选成 off:会话窗口发的是 {type: disabled},不带 thinking 反而会让
642
- // 上游走默认(实测默认是"开思考")。
643
- lines.push(
644
- "api_kwargs:",
645
- " default:",
646
- " thinking:",
647
- " type: disabled",
648
- " temperature: null",
649
- );
650
- }
627
+ // 注:**不注入 thinking**。曾经照抄会话窗口写过 thinking={enabled,16384},
628
+ // 但实测 aiportx-claude 在这之后每题要 11~22 分钟、正文 0~75 字符(思考吃满预算、
629
+ // 判 0 分);而同一渠道 09-12 的旧形状(temperature=0、不带 thinking、32000)
630
+ // 只要 139~286 秒、正文 1.9k 字符、75.5 分。中转对 thinking 预算并不当真,
631
+ // 所以保持 LiveBench 原生形状更稳。
651
632
  } else if (protocol === "openai_responses") {
652
633
  // OpenAI Responses API proxy: provider name matches
653
634
  // get_api_function('openai_responses') → chat_completion_openai_responses,
@@ -1388,17 +1369,13 @@ function apply(ctx) {
1388
1369
  writeError = writeGeneratedModelConfig(layout, loadYaml(profileDir), {
1389
1370
  displayName,
1390
1371
  modelId,
1372
+ // anthropic 通道不转发 reasoning_effort(那是 OpenAI 风格的开关),
1373
+ // 也不注入 thinking —— 实测中转对 thinking 预算并不当真,注入后反而
1374
+ // "只想不答"(见 writeGeneratedModelConfig 里的说明)。
1391
1375
  reasoningEffort: isAnthropicRoute ? null : effort,
1392
1376
  protocol: isAnthropicRoute ? "anthropic" : isOpenAIResponsesRoute ? "openai_responses" : "openai",
1393
- // 输出上限与字段名:让 LiveBench 用**会话窗口同样的字段**发出去
1394
- // (aiportx 这类中转要 max_completion_tokens;发 max_tokens 会慢数倍甚至卡死)
1395
1377
  maxTokens,
1396
- // provider 上已经算好(providerCapsFromSettings 里调 maxTokensFieldFor,
1397
- // 会尊重 settings.yaml 显式声明的 compat.maxTokensField)
1398
1378
  maxTokensField: provider !== null ? provider.maxTokensField : null,
1399
- // anthropic 通道:把会话窗口那份 thinking 预算一起发出去(见 thinkingBudgetForEffort)
1400
- thinkingBudget: isAnthropicRoute ? thinkingBudgetForEffort(effort) : null,
1401
- thinkingDisabled: isAnthropicRoute && effort === "off",
1402
1379
  });
1403
1380
  if (writeError) return { status: 500, payload: { ok: false, error: writeError } };
1404
1381
 
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "dsh-livebench-panel",
3
- "version": "0.2.32",
3
+ "version": "0.2.33",
4
4
  "description": "DSH web plugin: a LiveBench tab in the Trajectory view (right of 对话/轨迹). Run LiveBench evaluations against every model configured in the DeepSeek Harness — pick provider/model, category, task, release and question range from dropdowns, watch progress, and read scores in place. Ships a Baseline probe set built from the questions a reference model cannot solve.",
5
5
  "license": "MIT",
6
6
  "author": "cszr (Vithrive)",