dsh-livebench-panel 0.2.30 → 0.2.32
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +4 -1
- package/lib/client.js +14 -6
- package/lib/index.js +7 -6
- package/package.json +1 -1
package/README.md
CHANGED
|
@@ -86,7 +86,7 @@ cd livebench
|
|
|
86
86
|
- **推理强度下拉框**(模型右侧):来自模型配置的 `reasoningEfforts` 映射(如 gpt-5.6 系 off/low/medium/high/xhigh/max,DeepSeek off/low/high/max)。选定后:
|
|
87
87
|
- 插件向 `livebench/model/model_configs/dsh_panel_generated__<display-name>.yaml` 写入一条模型配置(每个模型一个文件,避免并发写同名文件互相覆盖),经 LiveBench 的 `api_kwargs.default.reasoning_effort` 透传给 API(`off` 表示不透传、由后端走默认);
|
|
88
88
|
- 强度编码进 display-name(如 `code-gpt__gpt-5.6-sol@high`),**不同强度在成绩表中是独立条目**,可直接对比。
|
|
89
|
-
- **参数下拉框**:题集 release(LiveBench 全部 releases)、分类(coding/math/reasoning/language/data_analysis/instruction_following)、任务(随分类联动)、题目序号范围、max-tokens
|
|
89
|
+
- **参数下拉框**:题集 release(LiveBench 全部 releases)、分类(coding/math/reasoning/language/data_analysis/instruction_following)、任务(随分类联动)、题目序号范围、max-tokens(**默认 32000**,自动夹到模型声明的上限;LiveBench 自带默认只有 4096,推理模型会思考吃满、正文为空,难题可手动调到 65536)、**并发请求数**(`--parallel-requests`,1/2/3/4/6/8,默认 1=串行;模型单题思考慢时并发多题是唯一能等比缩短总时长的开关)。
|
|
90
90
|
- **协议路由**:**Anthropic/Claude 系模型一律走 `anthropic-messages`**(POST `<baseURL>/v1/messages`,
|
|
91
91
|
与会话窗口同协议;`baseURL` 会自动剥掉 `/v1`、`/v1/messages` 后缀,因为 anthropic SDK 自己会拼);
|
|
92
92
|
其余模型按 settings.yaml 的 `api` 走 `openai-completions` / `openai-responses` / 内置端点。
|
|
@@ -100,6 +100,9 @@ cd livebench
|
|
|
100
100
|
默认 600s 无字节才断开重试(可用渠道声明的 `streamIdleTimeoutMs` 覆写);
|
|
101
101
|
另有两道硬边界——单次请求总时长上限(`LIVEBENCH_STREAM_MAX_SECONDS`,默认 1800s)
|
|
102
102
|
与首个字节等待上限(`LIVEBENCH_FIRST_BYTE_TIMEOUT`,默认 600s)。
|
|
103
|
+
- **进度心跳**:模型长时间思考时网关只发 `ping`、正文迟迟不来,日志里会每 30 秒出现一行
|
|
104
|
+
`仍在接收: 已 120s,连接累计 12345 字节(模型思考中,正文未到)`,长思考不会"看起来卡死"
|
|
105
|
+
(`LIVEBENCH_PROGRESS_SECONDS`,0 = 关闭)。
|
|
103
106
|
- **运行控制**:开始 / 停止 / 刷新;最多 9 个模型并发评测(每个模型同时只跑 1 次);实时滚动日志(每 2.5s 轮询);「刷新」会清掉已结束的运行日志。
|
|
104
107
|
- **自动补跑(API 抖动兜底)**:一次评测要连续打几百次 API,中转站 502/限流/超时是常态。
|
|
105
108
|
跑完后若答案文件里还有 `$ERROR$`,面板会自动用 `--resume --retry-failures`
|
package/lib/client.js
CHANGED
|
@@ -121,10 +121,10 @@ window.__ModuleLoader__.load({
|
|
|
121
121
|
function LiveBenchView() {
|
|
122
122
|
const [config, setConfig] = useState(null);
|
|
123
123
|
const [configError, setConfigError] = useState(null);
|
|
124
|
-
// maxTokens 默认
|
|
125
|
-
//
|
|
126
|
-
//
|
|
127
|
-
const [sel, setSel] = useState({ models: [], efforts: {}, cats: [], tasks: [], release: "2024-11-25", begin: "", end: "", maxTokens: "
|
|
124
|
+
// maxTokens 默认 32000(用户口径):LiveBench 自带默认只有 4096,跑推理模型
|
|
125
|
+
// 会"思考吃满、正文为空";32k 是够用与耗时之间的折中。难题若出现
|
|
126
|
+
// "思考占满 max-tokens、未产出答案",把它调到 65536 重跑即可(上限 200k)。
|
|
127
|
+
const [sel, setSel] = useState({ models: [], efforts: {}, cats: [], tasks: [], release: "2024-11-25", begin: "", end: "", maxTokens: "32000", parallel: "1", baseline: null });
|
|
128
128
|
const [modelDdOpen, setModelDdOpen] = useState(false);
|
|
129
129
|
const [busy, setBusy] = useState(false);
|
|
130
130
|
const [running, setRunning] = useState(false);
|
|
@@ -314,6 +314,7 @@ window.__ModuleLoader__.load({
|
|
|
314
314
|
begin: sel.begin,
|
|
315
315
|
end: sel.end,
|
|
316
316
|
maxTokens: sel.maxTokens,
|
|
317
|
+
parallel: sel.parallel,
|
|
317
318
|
}),
|
|
318
319
|
});
|
|
319
320
|
if (!payload.ok) failures.push(`${entry.modelId}: ${payload.error ?? "启动失败"}`);
|
|
@@ -580,8 +581,15 @@ window.__ModuleLoader__.load({
|
|
|
580
581
|
),
|
|
581
582
|
),
|
|
582
583
|
h("div", { className: c("field") },
|
|
583
|
-
h("label", { className: c("label") }, "max-tokens
|
|
584
|
-
h("input", { className: c("input"), type: "number", min: 256, max: 200000, value: sel.maxTokens, onChange: setField("maxTokens"), title: "
|
|
584
|
+
h("label", { className: c("label") }, "max-tokens"),
|
|
585
|
+
h("input", { className: c("input"), type: "number", min: 256, max: 200000, value: sel.maxTokens, onChange: setField("maxTokens"), title: "单题输出上限(默认 32000,会自动夹到该模型在 settings.yaml 里声明的上限)。LiveBench 自带默认只有 4096,推理模型会思考吃满、正文为空(成绩表里记「没做出来」);难题遇到这种情况就调到 65536 重跑。" }),
|
|
586
|
+
),
|
|
587
|
+
// Anthropic/Claude 这类模型答一道难题常常要思考几分钟(网关期间只发 ping、
|
|
588
|
+
// 正文不来),单题串行会把总时长拉得很长;这里允许同一次评测内并发多题。
|
|
589
|
+
h("div", { className: c("field") },
|
|
590
|
+
h("label", { className: c("label") }, "并发请求数"),
|
|
591
|
+
h("select", { className: c("select"), value: sel.parallel, onChange: setField("parallel"), title: "同一次评测内同时请求几道题(--parallel-requests)。默认 1 = 串行;模型思考慢时调大可以显著缩短总时长。中转限流/502 会由容错层退避重试。" },
|
|
592
|
+
["1", "2", "3", "4", "6", "8"].map((n) => h("option", { key: n, value: n }, n === "1" ? "1(串行)" : n))),
|
|
585
593
|
),
|
|
586
594
|
h("div", { className: c("field") },
|
|
587
595
|
h("label", { className: c("label") }, "题目序号范围(可选,0 起,含首尾)"),
|
package/lib/index.js
CHANGED
|
@@ -185,7 +185,7 @@ function loadYaml(profileDir) {
|
|
|
185
185
|
*
|
|
186
186
|
* 除了 id/name/api/baseURL,还要带上两个**渠道约束**,面板必须遵守:
|
|
187
187
|
* - `models[].maxTokens`:该模型单次输出上限(如 claude-fable-5-1 = 64000);
|
|
188
|
-
* 面板的 max-tokens 默认
|
|
188
|
+
* 面板的 max-tokens 默认 32000,不夹住就可能超过渠道上限(bearlab 这类渠道上限就是 32000);
|
|
189
189
|
* - `streamIdleTimeoutMs`:harness 自己给这个渠道设的"流多久没数据就掐断"
|
|
190
190
|
* (如 code-claude = 120000);面板把它透给 LiveBench 的静默看门狗,
|
|
191
191
|
* 这样"能长思考的渠道"和"确实会挂住的渠道"用的是各自合适的阈值。
|
|
@@ -220,7 +220,7 @@ function readProviders(profileDir) {
|
|
|
220
220
|
*
|
|
221
221
|
* 除了 id/name/api/baseURL,还要带上两个**渠道约束**,面板必须遵守:
|
|
222
222
|
* - `models[].maxTokens`:该模型单次输出上限(如 claude-fable-5-1 = 64000);
|
|
223
|
-
* 面板的 max-tokens 默认
|
|
223
|
+
* 面板的 max-tokens 默认 32000,不夹住就可能超过渠道上限(bearlab 这类渠道上限就是 32000);
|
|
224
224
|
* - `streamIdleTimeoutMs`:harness 自己给这个渠道设的"流多久没数据就掐断"
|
|
225
225
|
* (如 code-claude = 120000);面板把它透给 LiveBench 的静默看门狗,
|
|
226
226
|
* 这样"能长思考的渠道"和"确实会挂住的渠道"各自用合适的阈值。
|
|
@@ -1378,10 +1378,11 @@ function apply(ctx) {
|
|
|
1378
1378
|
// provider / api_kwargs / max_tokens —— 实测会把 temperature 悄悄变成 1.0、
|
|
1379
1379
|
// max_tokens 变成 131072,等于用户选的 provider 被换掉。写了自己的配置,
|
|
1380
1380
|
// 解析结果就是确定的 {local: <modelId>} + --api-base。
|
|
1381
|
-
// max-tokens
|
|
1382
|
-
//
|
|
1383
|
-
//
|
|
1384
|
-
|
|
1381
|
+
// max-tokens:面板默认 32000(LiveBench 自带默认只有 4096,推理模型会思考吃满、
|
|
1382
|
+
// 正文为空 → 记「没做出来」;32k 是够用与耗时之间的折中,难题可手动调到 65536)。
|
|
1383
|
+
// 另外不能超过 harness 给这个模型声明的上限(settings.yaml 的 models[].maxTokens,
|
|
1384
|
+
// 例如 claude-fable-5-1 = 64000、bearlab-claude = 32000)。
|
|
1385
|
+
const requestedMaxTokens = asInt(body.maxTokens, 256, 200000, 32000);
|
|
1385
1386
|
const modelMaxTokens = modelMeta !== null && Number(modelMeta.maxTokens) > 0 ? Number(modelMeta.maxTokens) : null;
|
|
1386
1387
|
const maxTokens = modelMaxTokens !== null ? Math.min(requestedMaxTokens, modelMaxTokens) : requestedMaxTokens;
|
|
1387
1388
|
writeError = writeGeneratedModelConfig(layout, loadYaml(profileDir), {
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "dsh-livebench-panel",
|
|
3
|
-
"version": "0.2.
|
|
3
|
+
"version": "0.2.32",
|
|
4
4
|
"description": "DSH web plugin: a LiveBench tab in the Trajectory view (right of 对话/轨迹). Run LiveBench evaluations against every model configured in the DeepSeek Harness — pick provider/model, category, task, release and question range from dropdowns, watch progress, and read scores in place. Ships a Baseline probe set built from the questions a reference model cannot solve.",
|
|
5
5
|
"license": "MIT",
|
|
6
6
|
"author": "cszr (Vithrive)",
|