dsh-livebench-panel 0.2.30 → 0.2.33
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +17 -13
- package/lib/client.js +14 -6
- package/lib/index.js +21 -43
- package/package.json +1 -1
package/README.md
CHANGED
|
@@ -86,20 +86,24 @@ cd livebench
|
|
|
86
86
|
- **推理强度下拉框**(模型右侧):来自模型配置的 `reasoningEfforts` 映射(如 gpt-5.6 系 off/low/medium/high/xhigh/max,DeepSeek off/low/high/max)。选定后:
|
|
87
87
|
- 插件向 `livebench/model/model_configs/dsh_panel_generated__<display-name>.yaml` 写入一条模型配置(每个模型一个文件,避免并发写同名文件互相覆盖),经 LiveBench 的 `api_kwargs.default.reasoning_effort` 透传给 API(`off` 表示不透传、由后端走默认);
|
|
88
88
|
- 强度编码进 display-name(如 `code-gpt__gpt-5.6-sol@high`),**不同强度在成绩表中是独立条目**,可直接对比。
|
|
89
|
-
- **参数下拉框**:题集 release(LiveBench 全部 releases)、分类(coding/math/reasoning/language/data_analysis/instruction_following)、任务(随分类联动)、题目序号范围、max-tokens
|
|
90
|
-
- **协议路由**:**Anthropic/Claude 系模型一律走 `anthropic-messages`**(POST `<baseURL>/v1/messages
|
|
91
|
-
|
|
89
|
+
- **参数下拉框**:题集 release(LiveBench 全部 releases)、分类(coding/math/reasoning/language/data_analysis/instruction_following)、任务(随分类联动)、题目序号范围、max-tokens(**默认 32000**,自动夹到模型声明的上限;LiveBench 自带默认只有 4096,推理模型会思考吃满、正文为空,难题可手动调到 65536)、**并发请求数**(`--parallel-requests`,1/2/3/4/6/8,默认 1=串行;模型单题思考慢时并发多题是唯一能等比缩短总时长的开关)。
|
|
90
|
+
- **协议路由**:**Anthropic/Claude 系模型一律走 `anthropic-messages`**(POST `<baseURL>/v1/messages`;
|
|
91
|
+
`baseURL` 会自动剥掉 `/v1`、`/v1/messages` 后缀,因为 anthropic SDK 自己会拼);
|
|
92
92
|
其余模型按 settings.yaml 的 `api` 走 `openai-completions` / `openai-responses` / 内置端点。
|
|
93
|
-
运行日志首行打印 `[route]
|
|
94
|
-
-
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
|
|
93
|
+
运行日志首行打印 `[route] …`,写明协议、目标地址与 `max_tokens`。
|
|
94
|
+
- **不替上游做形状决定**(实测结论):面板**不注入 `thinking`**、**不注入长度上限字段**。
|
|
95
|
+
中转对这两样都不当真 —— 注入 `thinking` 后 aiportx 只会回空的 thinking 片段(一题 11~22 分钟、
|
|
96
|
+
正文 0~75 字符、判 0);`max_tokens` / `max_completion_tokens` 写 512 实测吐 4062 / 4737 token。
|
|
97
|
+
保持 LiveBench 的原生形状最稳;`--max-tokens` 仍会夹到模型在 settings.yaml 里声明的上限。
|
|
98
|
+
- **流式看门狗三层判死,且不会误杀健康请求**:① 连接上任何字节都算活(含每 3 秒的 `ping`),
|
|
99
|
+
`LIVEBENCH_STREAM_IDLE_TIMEOUT` 默认 600s(可用渠道声明的 `streamIdleTimeoutMs` 覆写);
|
|
100
|
+
② **内容**维度:`LIVEBENCH_CONTENT_IDLE_SECONDS` 默认 480s 没有任何正文/思考片段就断开
|
|
101
|
+
—— 中转会连发 ping 十几分钟不给一个字,只按字节判活对它无效;
|
|
102
|
+
③ 总时长上限 `LIVEBENCH_STREAM_MAX_SECONDS` 默认 900s;首个字节等待上限
|
|
103
|
+
`LIVEBENCH_FIRST_BYTE_TIMEOUT` 默认 600s。
|
|
104
|
+
- **进度心跳**:模型长时间思考时网关只发 `ping`、正文迟迟不来,日志里会每 30 秒出现一行
|
|
105
|
+
`仍在接收: 已 120s,连接累计 12345 字节(模型思考中,正文未到)`,长思考不会"看起来卡死"
|
|
106
|
+
(`LIVEBENCH_PROGRESS_SECONDS`,0 = 关闭)。
|
|
103
107
|
- **运行控制**:开始 / 停止 / 刷新;最多 9 个模型并发评测(每个模型同时只跑 1 次);实时滚动日志(每 2.5s 轮询);「刷新」会清掉已结束的运行日志。
|
|
104
108
|
- **自动补跑(API 抖动兜底)**:一次评测要连续打几百次 API,中转站 502/限流/超时是常态。
|
|
105
109
|
跑完后若答案文件里还有 `$ERROR$`,面板会自动用 `--resume --retry-failures`
|
package/lib/client.js
CHANGED
|
@@ -121,10 +121,10 @@ window.__ModuleLoader__.load({
|
|
|
121
121
|
function LiveBenchView() {
|
|
122
122
|
const [config, setConfig] = useState(null);
|
|
123
123
|
const [configError, setConfigError] = useState(null);
|
|
124
|
-
// maxTokens 默认
|
|
125
|
-
//
|
|
126
|
-
//
|
|
127
|
-
const [sel, setSel] = useState({ models: [], efforts: {}, cats: [], tasks: [], release: "2024-11-25", begin: "", end: "", maxTokens: "
|
|
124
|
+
// maxTokens 默认 32000(用户口径):LiveBench 自带默认只有 4096,跑推理模型
|
|
125
|
+
// 会"思考吃满、正文为空";32k 是够用与耗时之间的折中。难题若出现
|
|
126
|
+
// "思考占满 max-tokens、未产出答案",把它调到 65536 重跑即可(上限 200k)。
|
|
127
|
+
const [sel, setSel] = useState({ models: [], efforts: {}, cats: [], tasks: [], release: "2024-11-25", begin: "", end: "", maxTokens: "32000", parallel: "1", baseline: null });
|
|
128
128
|
const [modelDdOpen, setModelDdOpen] = useState(false);
|
|
129
129
|
const [busy, setBusy] = useState(false);
|
|
130
130
|
const [running, setRunning] = useState(false);
|
|
@@ -314,6 +314,7 @@ window.__ModuleLoader__.load({
|
|
|
314
314
|
begin: sel.begin,
|
|
315
315
|
end: sel.end,
|
|
316
316
|
maxTokens: sel.maxTokens,
|
|
317
|
+
parallel: sel.parallel,
|
|
317
318
|
}),
|
|
318
319
|
});
|
|
319
320
|
if (!payload.ok) failures.push(`${entry.modelId}: ${payload.error ?? "启动失败"}`);
|
|
@@ -580,8 +581,15 @@ window.__ModuleLoader__.load({
|
|
|
580
581
|
),
|
|
581
582
|
),
|
|
582
583
|
h("div", { className: c("field") },
|
|
583
|
-
h("label", { className: c("label") }, "max-tokens
|
|
584
|
-
h("input", { className: c("input"), type: "number", min: 256, max: 200000, value: sel.maxTokens, onChange: setField("maxTokens"), title: "
|
|
584
|
+
h("label", { className: c("label") }, "max-tokens"),
|
|
585
|
+
h("input", { className: c("input"), type: "number", min: 256, max: 200000, value: sel.maxTokens, onChange: setField("maxTokens"), title: "单题输出上限(默认 32000,会自动夹到该模型在 settings.yaml 里声明的上限)。LiveBench 自带默认只有 4096,推理模型会思考吃满、正文为空(成绩表里记「没做出来」);难题遇到这种情况就调到 65536 重跑。" }),
|
|
586
|
+
),
|
|
587
|
+
// Anthropic/Claude 这类模型答一道难题常常要思考几分钟(网关期间只发 ping、
|
|
588
|
+
// 正文不来),单题串行会把总时长拉得很长;这里允许同一次评测内并发多题。
|
|
589
|
+
h("div", { className: c("field") },
|
|
590
|
+
h("label", { className: c("label") }, "并发请求数"),
|
|
591
|
+
h("select", { className: c("select"), value: sel.parallel, onChange: setField("parallel"), title: "同一次评测内同时请求几道题(--parallel-requests)。默认 1 = 串行;模型思考慢时调大可以显著缩短总时长。中转限流/502 会由容错层退避重试。" },
|
|
592
|
+
["1", "2", "3", "4", "6", "8"].map((n) => h("option", { key: n, value: n }, n === "1" ? "1(串行)" : n))),
|
|
585
593
|
),
|
|
586
594
|
h("div", { className: c("field") },
|
|
587
595
|
h("label", { className: c("label") }, "题目序号范围(可选,0 起,含首尾)"),
|
package/lib/index.js
CHANGED
|
@@ -185,7 +185,7 @@ function loadYaml(profileDir) {
|
|
|
185
185
|
*
|
|
186
186
|
* 除了 id/name/api/baseURL,还要带上两个**渠道约束**,面板必须遵守:
|
|
187
187
|
* - `models[].maxTokens`:该模型单次输出上限(如 claude-fable-5-1 = 64000);
|
|
188
|
-
* 面板的 max-tokens 默认
|
|
188
|
+
* 面板的 max-tokens 默认 32000,不夹住就可能超过渠道上限(bearlab 这类渠道上限就是 32000);
|
|
189
189
|
* - `streamIdleTimeoutMs`:harness 自己给这个渠道设的"流多久没数据就掐断"
|
|
190
190
|
* (如 code-claude = 120000);面板把它透给 LiveBench 的静默看门狗,
|
|
191
191
|
* 这样"能长思考的渠道"和"确实会挂住的渠道"用的是各自合适的阈值。
|
|
@@ -220,7 +220,7 @@ function readProviders(profileDir) {
|
|
|
220
220
|
*
|
|
221
221
|
* 除了 id/name/api/baseURL,还要带上两个**渠道约束**,面板必须遵守:
|
|
222
222
|
* - `models[].maxTokens`:该模型单次输出上限(如 claude-fable-5-1 = 64000);
|
|
223
|
-
* 面板的 max-tokens 默认
|
|
223
|
+
* 面板的 max-tokens 默认 32000,不夹住就可能超过渠道上限(bearlab 这类渠道上限就是 32000);
|
|
224
224
|
* - `streamIdleTimeoutMs`:harness 自己给这个渠道设的"流多久没数据就掐断"
|
|
225
225
|
* (如 code-claude = 120000);面板把它透给 LiveBench 的静默看门狗,
|
|
226
226
|
* 这样"能长思考的渠道"和"确实会挂住的渠道"各自用合适的阈值。
|
|
@@ -614,40 +614,21 @@ function writeGeneratedModelConfig(layout, YAML, { displayName, modelId, reasoni
|
|
|
614
614
|
`display_name: ${displayName}`,
|
|
615
615
|
"api_name:",
|
|
616
616
|
];
|
|
617
|
-
//
|
|
618
|
-
// `
|
|
619
|
-
//
|
|
620
|
-
|
|
621
|
-
|
|
622
|
-
|
|
623
|
-
: null;
|
|
617
|
+
// 长度上限:**不再由面板注入字段名**。
|
|
618
|
+
// 早先为对齐会话窗口写过 `max_completion_tokens`,但实测 aiportx 这类中转对
|
|
619
|
+
// `max_tokens` 与 `max_completion_tokens` **两个字段都不执行**(请求写 512,
|
|
620
|
+
// 实测吐 4062 / 4737 token),写进去只是"假装有上限",还容易让人误判。
|
|
621
|
+
// 交回 LiveBench 自己的默认(非 gpt 模型发 max_tokens)即可。
|
|
622
|
+
const limitLine = null;
|
|
624
623
|
if (protocol === "anthropic") {
|
|
625
624
|
// Route through LiveBench's native anthropic client; the endpoint is
|
|
626
625
|
// pointed at the provider proxy via ANTHROPIC_BASE_URL (spawn env).
|
|
627
626
|
lines.push(` anthropic: ${modelId}`, "default_provider: anthropic");
|
|
628
|
-
//
|
|
629
|
-
//
|
|
630
|
-
|
|
631
|
-
|
|
632
|
-
|
|
633
|
-
" default:",
|
|
634
|
-
" thinking:",
|
|
635
|
-
" type: enabled",
|
|
636
|
-
` budget_tokens: ${Number(thinkingBudget)}`,
|
|
637
|
-
// thinking 与 temperature 互斥:显式置 None → LiveBench 映射成 NOT_GIVEN 而不发送
|
|
638
|
-
" temperature: null",
|
|
639
|
-
);
|
|
640
|
-
} else if (thinkingDisabled) {
|
|
641
|
-
// 用户把强度选成 off:会话窗口发的是 {type: disabled},不带 thinking 反而会让
|
|
642
|
-
// 上游走默认(实测默认是"开思考")。
|
|
643
|
-
lines.push(
|
|
644
|
-
"api_kwargs:",
|
|
645
|
-
" default:",
|
|
646
|
-
" thinking:",
|
|
647
|
-
" type: disabled",
|
|
648
|
-
" temperature: null",
|
|
649
|
-
);
|
|
650
|
-
}
|
|
627
|
+
// 注:**不注入 thinking**。曾经照抄会话窗口写过 thinking={enabled,16384},
|
|
628
|
+
// 但实测 aiportx-claude 在这之后每题要 11~22 分钟、正文 0~75 字符(思考吃满预算、
|
|
629
|
+
// 判 0 分);而同一渠道 09-12 的旧形状(temperature=0、不带 thinking、32000)
|
|
630
|
+
// 只要 139~286 秒、正文 1.9k 字符、75.5 分。中转对 thinking 预算并不当真,
|
|
631
|
+
// 所以保持 LiveBench 原生形状更稳。
|
|
651
632
|
} else if (protocol === "openai_responses") {
|
|
652
633
|
// OpenAI Responses API proxy: provider name matches
|
|
653
634
|
// get_api_function('openai_responses') → chat_completion_openai_responses,
|
|
@@ -1378,26 +1359,23 @@ function apply(ctx) {
|
|
|
1378
1359
|
// provider / api_kwargs / max_tokens —— 实测会把 temperature 悄悄变成 1.0、
|
|
1379
1360
|
// max_tokens 变成 131072,等于用户选的 provider 被换掉。写了自己的配置,
|
|
1380
1361
|
// 解析结果就是确定的 {local: <modelId>} + --api-base。
|
|
1381
|
-
// max-tokens
|
|
1382
|
-
//
|
|
1383
|
-
//
|
|
1384
|
-
|
|
1362
|
+
// max-tokens:面板默认 32000(LiveBench 自带默认只有 4096,推理模型会思考吃满、
|
|
1363
|
+
// 正文为空 → 记「没做出来」;32k 是够用与耗时之间的折中,难题可手动调到 65536)。
|
|
1364
|
+
// 另外不能超过 harness 给这个模型声明的上限(settings.yaml 的 models[].maxTokens,
|
|
1365
|
+
// 例如 claude-fable-5-1 = 64000、bearlab-claude = 32000)。
|
|
1366
|
+
const requestedMaxTokens = asInt(body.maxTokens, 256, 200000, 32000);
|
|
1385
1367
|
const modelMaxTokens = modelMeta !== null && Number(modelMeta.maxTokens) > 0 ? Number(modelMeta.maxTokens) : null;
|
|
1386
1368
|
const maxTokens = modelMaxTokens !== null ? Math.min(requestedMaxTokens, modelMaxTokens) : requestedMaxTokens;
|
|
1387
1369
|
writeError = writeGeneratedModelConfig(layout, loadYaml(profileDir), {
|
|
1388
1370
|
displayName,
|
|
1389
1371
|
modelId,
|
|
1372
|
+
// anthropic 通道不转发 reasoning_effort(那是 OpenAI 风格的开关),
|
|
1373
|
+
// 也不注入 thinking —— 实测中转对 thinking 预算并不当真,注入后反而
|
|
1374
|
+
// "只想不答"(见 writeGeneratedModelConfig 里的说明)。
|
|
1390
1375
|
reasoningEffort: isAnthropicRoute ? null : effort,
|
|
1391
1376
|
protocol: isAnthropicRoute ? "anthropic" : isOpenAIResponsesRoute ? "openai_responses" : "openai",
|
|
1392
|
-
// 输出上限与字段名:让 LiveBench 用**会话窗口同样的字段**发出去
|
|
1393
|
-
// (aiportx 这类中转要 max_completion_tokens;发 max_tokens 会慢数倍甚至卡死)
|
|
1394
1377
|
maxTokens,
|
|
1395
|
-
// provider 上已经算好(providerCapsFromSettings 里调 maxTokensFieldFor,
|
|
1396
|
-
// 会尊重 settings.yaml 显式声明的 compat.maxTokensField)
|
|
1397
1378
|
maxTokensField: provider !== null ? provider.maxTokensField : null,
|
|
1398
|
-
// anthropic 通道:把会话窗口那份 thinking 预算一起发出去(见 thinkingBudgetForEffort)
|
|
1399
|
-
thinkingBudget: isAnthropicRoute ? thinkingBudgetForEffort(effort) : null,
|
|
1400
|
-
thinkingDisabled: isAnthropicRoute && effort === "off",
|
|
1401
1379
|
});
|
|
1402
1380
|
if (writeError) return { status: 500, payload: { ok: false, error: writeError } };
|
|
1403
1381
|
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "dsh-livebench-panel",
|
|
3
|
-
"version": "0.2.
|
|
3
|
+
"version": "0.2.33",
|
|
4
4
|
"description": "DSH web plugin: a LiveBench tab in the Trajectory view (right of 对话/轨迹). Run LiveBench evaluations against every model configured in the DeepSeek Harness — pick provider/model, category, task, release and question range from dropdowns, watch progress, and read scores in place. Ships a Baseline probe set built from the questions a reference model cannot solve.",
|
|
5
5
|
"license": "MIT",
|
|
6
6
|
"author": "cszr (Vithrive)",
|