dsh-livebench-panel 0.2.24 → 0.2.30
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +25 -2
- package/lib/client.js +4 -1
- package/lib/index.js +493 -108
- package/package.json +1 -1
package/README.md
CHANGED
|
@@ -46,7 +46,7 @@ cd livebench
|
|
|
46
46
|
| 分类 | 六大类:coding / math / reasoning / language / data_analysis / instruction_following | 首次验证选 `language` |
|
|
47
47
|
| 任务 | 分类下的具体任务(如 language/typos 拼写纠错) | 首次验证选 `typos`(短平快) |
|
|
48
48
|
| 题目序号范围 | 起止下标(0 起,**含首尾**);选 Baseline 时序号相对该题库 | 冒烟测试填 0–1(只跑 2 题) |
|
|
49
|
-
| max-tokens | 单次回答的 token
|
|
49
|
+
| max-tokens | 单次回答的 token 上限(默认 65536,上限 200000) | 推理模型别低于 32768,否则思考吃满预算、正文为空,这题会被记成「没做出来」 |
|
|
50
50
|
|
|
51
51
|
## Baseline 题库(探针)
|
|
52
52
|
|
|
@@ -87,8 +87,31 @@ cd livebench
|
|
|
87
87
|
- 插件向 `livebench/model/model_configs/dsh_panel_generated__<display-name>.yaml` 写入一条模型配置(每个模型一个文件,避免并发写同名文件互相覆盖),经 LiveBench 的 `api_kwargs.default.reasoning_effort` 透传给 API(`off` 表示不透传、由后端走默认);
|
|
88
88
|
- 强度编码进 display-name(如 `code-gpt__gpt-5.6-sol@high`),**不同强度在成绩表中是独立条目**,可直接对比。
|
|
89
89
|
- **参数下拉框**:题集 release(LiveBench 全部 releases)、分类(coding/math/reasoning/language/data_analysis/instruction_following)、任务(随分类联动)、题目序号范围、max-tokens。
|
|
90
|
+
- **协议路由**:**Anthropic/Claude 系模型一律走 `anthropic-messages`**(POST `<baseURL>/v1/messages`,
|
|
91
|
+
与会话窗口同协议;`baseURL` 会自动剥掉 `/v1`、`/v1/messages` 后缀,因为 anthropic SDK 自己会拼);
|
|
92
|
+
其余模型按 settings.yaml 的 `api` 走 `openai-completions` / `openai-responses` / 内置端点。
|
|
93
|
+
运行日志首行打印 `[route] …`,写明协议、目标地址、`max_tokens` 与 thinking 形状。
|
|
94
|
+
- **请求形状对齐会话窗口**:anthropic 通道按强度档附上
|
|
95
|
+
`thinking: {type: enabled, budget_tokens: N}`(minimal 1024 / low 2048 / medium 8192 /
|
|
96
|
+
high·xhigh·max 16384;`off` → `{type: disabled}`),并因此**不发 `temperature`**;
|
|
97
|
+
非 anthropic 通道按渠道习惯选长度上限字段(aiportx 这类中转用 `max_completion_tokens`),
|
|
98
|
+
且 `--max-tokens` 会夹到该模型在 settings.yaml 里声明的上限。
|
|
99
|
+
- **流式看门狗不会误杀健康请求**:判活标准是"连接上是否还有字节"(含网关每 3 秒的 `ping`),
|
|
100
|
+
默认 600s 无字节才断开重试(可用渠道声明的 `streamIdleTimeoutMs` 覆写);
|
|
101
|
+
另有两道硬边界——单次请求总时长上限(`LIVEBENCH_STREAM_MAX_SECONDS`,默认 1800s)
|
|
102
|
+
与首个字节等待上限(`LIVEBENCH_FIRST_BYTE_TIMEOUT`,默认 600s)。
|
|
90
103
|
- **运行控制**:开始 / 停止 / 刷新;最多 9 个模型并发评测(每个模型同时只跑 1 次);实时滚动日志(每 2.5s 轮询);「刷新」会清掉已结束的运行日志。
|
|
91
|
-
-
|
|
104
|
+
- **自动补跑(API 抖动兜底)**:一次评测要连续打几百次 API,中转站 502/限流/超时是常态。
|
|
105
|
+
跑完后若答案文件里还有 `$ERROR$`,面板会自动用 `--resume --retry-failures`
|
|
106
|
+
**只重跑失败的那几题**(最多 3 轮,间隔 15s / 60s / 3min),并复用同一个条目名 ——
|
|
107
|
+
已成功题目的答案与判分都保留,成绩表里仍然只有一行。Baseline 评测同样适用。
|
|
108
|
+
日志里会打印 `[自动补跑] …`。
|
|
109
|
+
注意:思考吃满 max-tokens 的空答案(不是 `$ERROR$`)不会自动补跑——那不是抖动,
|
|
110
|
+
把 max-tokens 调大后手动重跑才有意义。
|
|
111
|
+
- **题目数据集离线读取**:huggingface.co 在国内常常连不通,而已缓存的分类根本不需要联网。
|
|
112
|
+
面板检测到本次要跑的分类都在 `.hf_cache` 里时,直接以 `HF_HUB_OFFLINE=1` 启动,
|
|
113
|
+
省掉 5 轮 HEAD 超时重试(每次白等 20~60s);缺缓存的分类保持联网,不影响新 release。
|
|
114
|
+
- **成绩表**:直接读取 `data/live_bench/**/model_judgment/ground_truth_judgment.jsonl` 计算 模型 × 任务 平均分(分数 = 正确率 ×100,单元格下方括号标注「**本次没做或没做完的题数 / 用户选择的题数**」,即题目序号范围的题数;不参与正确率计算的总题数不展示),无需等 LiveBench 自己出榜。**判分与答案按 `answer_id` 配对**:补跑换过答案的题只认新判分,旧的 0 分不会把成绩拉低;答案文件里同一题出现多行时按最后一行取值(与 LiveBench 自身口径一致)。支持按模型名/时间排序、拖 ⠿ 自定义行序、勾选或单行按钮删除成绩(删除会同时清掉该模型的答案与判分行;只有空壳文件的“幽灵行”不会再被扫描出来,删除结果稳定持久)。
|
|
92
115
|
- **题目序号范围按闭区间处理**:LiveBench 的 `--question-end` 是开区间(`questions[begin:end]`),面板按用户直觉采用闭区间(传参时 `止 + 1`),因此界面上填的题数与实际跑的题数一致。
|
|
93
116
|
|
|
94
117
|
## 依赖
|
package/lib/client.js
CHANGED
|
@@ -121,7 +121,10 @@ window.__ModuleLoader__.load({
|
|
|
121
121
|
function LiveBenchView() {
|
|
122
122
|
const [config, setConfig] = useState(null);
|
|
123
123
|
const [configError, setConfigError] = useState(null);
|
|
124
|
-
|
|
124
|
+
// maxTokens 默认 65536:推理模型在 olympiad 这类难题上会把预算全烧在
|
|
125
|
+
// 思考里(eval_status=token_exhaustion,正文为空 → 记「没做出来」),
|
|
126
|
+
// 32k 明显不够;上限 200k,可自行调。
|
|
127
|
+
const [sel, setSel] = useState({ models: [], efforts: {}, cats: [], tasks: [], release: "2024-11-25", begin: "", end: "", maxTokens: "65536", baseline: null });
|
|
125
128
|
const [modelDdOpen, setModelDdOpen] = useState(false);
|
|
126
129
|
const [busy, setBusy] = useState(false);
|
|
127
130
|
const [running, setRunning] = useState(false);
|
package/lib/index.js
CHANGED
|
@@ -182,13 +182,52 @@ function loadYaml(profileDir) {
|
|
|
182
182
|
* Read harness providers from settings.yaml, plus the built-in
|
|
183
183
|
* `deepseek-official` provider (registered by @deepseek-ai/dsh-llm-deepseek,
|
|
184
184
|
* which never appears under llm-pi-ai.providers).
|
|
185
|
-
*
|
|
185
|
+
*
|
|
186
|
+
* 除了 id/name/api/baseURL,还要带上两个**渠道约束**,面板必须遵守:
|
|
187
|
+
* - `models[].maxTokens`:该模型单次输出上限(如 claude-fable-5-1 = 64000);
|
|
188
|
+
* 面板的 max-tokens 默认 65536,不夹住就可能超过渠道上限;
|
|
189
|
+
* - `streamIdleTimeoutMs`:harness 自己给这个渠道设的"流多久没数据就掐断"
|
|
190
|
+
* (如 code-claude = 120000);面板把它透给 LiveBench 的静默看门狗,
|
|
191
|
+
* 这样"能长思考的渠道"和"确实会挂住的渠道"用的是各自合适的阈值。
|
|
192
|
+
* @returns {Array<{id: string, name: string, api: string|null, baseURL: string|null, keyEnv: string|null, streamIdleTimeoutMs: number|null, models: Array<{id: string, name: string, efforts: string[], maxTokens: number|null}>}>}
|
|
186
193
|
*/
|
|
187
194
|
function readProviders(profileDir) {
|
|
188
195
|
// profileDir = ~/.dsh/profiles/web → settings.yaml = ~/.dsh/settings.yaml
|
|
189
196
|
const settingsPath = join(profileDir.replace(/[\\/]+$/, ""), "..", "..", "settings.yaml");
|
|
190
197
|
const settings = existsSync(settingsPath) ? loadYaml(profileDir).parse(readFileSync(settingsPath, "utf8")) ?? {} : {};
|
|
198
|
+
const providers = providerCapsFromSettings(settings);
|
|
191
199
|
|
|
200
|
+
// Providers without an explicit baseURL may still be OpenAI-compatible with
|
|
201
|
+
// a well-known endpoint: resolve it from the pi-ai provider registry
|
|
202
|
+
// (@earendil-works/pi-ai, the same source the harness serves from).
|
|
203
|
+
const piAiData = resolvePiAiDataDir(profileDir);
|
|
204
|
+
if (piAiData !== null) {
|
|
205
|
+
for (const provider of providers) {
|
|
206
|
+
if (provider.baseURL === null) {
|
|
207
|
+
const info = builtinProtocolInfo(piAiData, provider.id);
|
|
208
|
+
if (info) {
|
|
209
|
+
provider.baseURL = info.baseURL;
|
|
210
|
+
provider.api = info.api;
|
|
211
|
+
}
|
|
212
|
+
}
|
|
213
|
+
}
|
|
214
|
+
}
|
|
215
|
+
return providers;
|
|
216
|
+
}
|
|
217
|
+
|
|
218
|
+
/**
|
|
219
|
+
* settings.yaml(已解析)→ 面板用的 provider/模型清单。纯函数,便于回归测试。
|
|
220
|
+
*
|
|
221
|
+
* 除了 id/name/api/baseURL,还要带上两个**渠道约束**,面板必须遵守:
|
|
222
|
+
* - `models[].maxTokens`:该模型单次输出上限(如 claude-fable-5-1 = 64000);
|
|
223
|
+
* 面板的 max-tokens 默认 65536,不夹住就可能超过渠道上限;
|
|
224
|
+
* - `streamIdleTimeoutMs`:harness 自己给这个渠道设的"流多久没数据就掐断"
|
|
225
|
+
* (如 code-claude = 120000);面板把它透给 LiveBench 的静默看门狗,
|
|
226
|
+
* 这样"能长思考的渠道"和"确实会挂住的渠道"各自用合适的阈值。
|
|
227
|
+
* @param {object} settings 解析后的 settings.yaml
|
|
228
|
+
*/
|
|
229
|
+
function providerCapsFromSettings(settings) {
|
|
230
|
+
const positiveNumber = (value) => (Number.isFinite(Number(value)) && Number(value) > 0 ? Number(value) : null);
|
|
192
231
|
const providers = [];
|
|
193
232
|
const piAi = settings?.["llm-pi-ai"]?.providers ?? {};
|
|
194
233
|
for (const [id, cfg] of Object.entries(piAi)) {
|
|
@@ -198,10 +237,18 @@ function readProviders(profileDir) {
|
|
|
198
237
|
api: typeof cfg?.api === "string" ? cfg.api : null,
|
|
199
238
|
baseURL: typeof cfg?.baseURL === "string" ? cfg.baseURL : null,
|
|
200
239
|
keyEnv: typeof cfg?.apiKeyEnv === "string" ? cfg.apiKeyEnv : null,
|
|
240
|
+
streamIdleTimeoutMs: positiveNumber(cfg?.streamIdleTimeoutMs),
|
|
241
|
+
// 会话窗口(pi-ai)发输出上限时用的字段名 —— 面板要让 LiveBench 用同一个
|
|
242
|
+
maxTokensField: maxTokensFieldFor({ id, baseURL: cfg?.baseURL, compat: cfg?.compat }),
|
|
201
243
|
models: Array.isArray(cfg?.models)
|
|
202
244
|
? cfg.models
|
|
203
245
|
.filter((m) => m && typeof m.id === "string" && m.id.trim().length > 0)
|
|
204
|
-
.map((m) => ({
|
|
246
|
+
.map((m) => ({
|
|
247
|
+
id: m.id.trim(),
|
|
248
|
+
name: String(m.name ?? m.id).trim(),
|
|
249
|
+
efforts: effortsOfModel(m),
|
|
250
|
+
maxTokens: positiveNumber(m?.maxTokens),
|
|
251
|
+
}))
|
|
205
252
|
: [],
|
|
206
253
|
});
|
|
207
254
|
}
|
|
@@ -213,32 +260,24 @@ function readProviders(profileDir) {
|
|
|
213
260
|
const dsModels = Array.isArray(dsCfg.models) && dsCfg.models.length > 0
|
|
214
261
|
? dsCfg.models
|
|
215
262
|
.filter((m) => m && typeof m.id === "string" && m.id.trim().length > 0)
|
|
216
|
-
.map((m) => ({
|
|
217
|
-
|
|
263
|
+
.map((m) => ({
|
|
264
|
+
id: m.id.trim(),
|
|
265
|
+
name: String(m.name ?? m.id).trim(),
|
|
266
|
+
efforts: [...DEEPSEEK_EFFORTS],
|
|
267
|
+
maxTokens: positiveNumber(m?.maxTokens),
|
|
268
|
+
}))
|
|
269
|
+
: DEFAULT_DEEPSEEK_MODELS.map((m) => ({ ...m, efforts: [...DEEPSEEK_EFFORTS], maxTokens: null }));
|
|
218
270
|
providers.push({
|
|
219
271
|
id: "deepseek-official",
|
|
220
272
|
name: "DeepSeek 官方",
|
|
221
273
|
api: "openai-completions",
|
|
222
274
|
baseURL: typeof dsCfg.baseURL === "string" && dsCfg.baseURL.length > 0 ? dsCfg.baseURL : "https://api.deepseek.com",
|
|
223
275
|
keyEnv: typeof dsCfg.apiKeyEnv === "string" && dsCfg.apiKeyEnv.length > 0 ? dsCfg.apiKeyEnv : "DEEPSEEK_API_KEY",
|
|
276
|
+
streamIdleTimeoutMs: positiveNumber(dsCfg?.streamIdleTimeoutMs),
|
|
277
|
+
// DeepSeek 官方在 pi-ai 的 detectCompat 里属于 useMaxTokens → max_tokens
|
|
278
|
+
maxTokensField: "max_tokens",
|
|
224
279
|
models: dsModels,
|
|
225
280
|
});
|
|
226
|
-
|
|
227
|
-
// Providers without an explicit baseURL may still be OpenAI-compatible with
|
|
228
|
-
// a well-known endpoint: resolve it from the pi-ai provider registry
|
|
229
|
-
// (@earendil-works/pi-ai, the same source the harness serves from).
|
|
230
|
-
const piAiData = resolvePiAiDataDir(profileDir);
|
|
231
|
-
if (piAiData !== null) {
|
|
232
|
-
for (const provider of providers) {
|
|
233
|
-
if (provider.baseURL === null) {
|
|
234
|
-
const info = builtinProtocolInfo(piAiData, provider.id);
|
|
235
|
-
if (info) {
|
|
236
|
-
provider.baseURL = info.baseURL;
|
|
237
|
-
provider.api = info.api;
|
|
238
|
-
}
|
|
239
|
-
}
|
|
240
|
-
}
|
|
241
|
-
}
|
|
242
281
|
return providers;
|
|
243
282
|
}
|
|
244
283
|
|
|
@@ -472,6 +511,90 @@ function displayModelName(providerId, modelId) {
|
|
|
472
511
|
return `${String(providerId).replace(/[^A-Za-z0-9._-]/g, "-")}__${String(modelId).replace(/[^A-Za-z0-9._-]/g, "-")}`;
|
|
473
512
|
}
|
|
474
513
|
|
|
514
|
+
/**
|
|
515
|
+
* 这个 模型/provider 是不是 Anthropic(Claude) 系?—— 决定要不要走 anthropic-messages 协议。
|
|
516
|
+
* 名字里带 claude / anthropic 就算(provider id 或 model id 任一命中)。
|
|
517
|
+
*/
|
|
518
|
+
function isAnthropicModel(providerId, modelId) {
|
|
519
|
+
return /claude|anthropic/i.test(String(modelId ?? "")) || /claude|anthropic/i.test(String(providerId ?? ""));
|
|
520
|
+
}
|
|
521
|
+
|
|
522
|
+
/**
|
|
523
|
+
* anthropic SDK 会在 baseURL 后面自己拼 `/v1/messages`,所以 ANTHROPIC_BASE_URL
|
|
524
|
+
* 必须是**站点根**:把用户可能写的 `/v1`、`/v1/messages`、结尾斜杠都剥掉。
|
|
525
|
+
* 踩过的坑:settings.yaml 里 openai 兼容渠道习惯写成 `https://host/v1`,
|
|
526
|
+
* 直接拿它当 ANTHROPIC_BASE_URL 会 POST 到 `https://host/v1/v1/messages` → 404。
|
|
527
|
+
*/
|
|
528
|
+
function anthropicBaseUrl(baseURL) {
|
|
529
|
+
return String(baseURL ?? "")
|
|
530
|
+
.trim()
|
|
531
|
+
.replace(/\/+$/, "")
|
|
532
|
+
.replace(/\/v1\/messages$/i, "")
|
|
533
|
+
.replace(/\/v1$/i, "")
|
|
534
|
+
.replace(/\/+$/, "");
|
|
535
|
+
}
|
|
536
|
+
|
|
537
|
+
/**
|
|
538
|
+
* 请求形状的"思考预算"档位(对齐 pi-ai):
|
|
539
|
+
* - 有预算 → thinking enabled + budget_tokens
|
|
540
|
+
* - effort 明确是 off → thinking disabled
|
|
541
|
+
* - 其它(没选强度)→ 不写,交给上游默认
|
|
542
|
+
*/
|
|
543
|
+
function thinkingModeForEffort(effort) {
|
|
544
|
+
const budget = thinkingBudgetForEffort(effort);
|
|
545
|
+
if (budget !== null) return { type: "enabled", budget_tokens: budget };
|
|
546
|
+
if (effort === "off") return { type: "disabled" };
|
|
547
|
+
return null;
|
|
548
|
+
}
|
|
549
|
+
|
|
550
|
+
/**
|
|
551
|
+
* LiveBench 发"单次输出上限"时该用哪个字段?
|
|
552
|
+
*
|
|
553
|
+
* 会话窗口(harness)走 pi-ai,会按渠道自动挑字段:大部分中转(aiportx、code28 等)
|
|
554
|
+
* 用 `max_completion_tokens`;deepseek / moonshot / z.ai / together / nvidia /
|
|
555
|
+
* ant-ling / chutes / Cloudflare 网关用 `max_tokens`
|
|
556
|
+
* (对应 pi-ai `openai-completions.js` 的 `detectCompat().maxTokensField`)。
|
|
557
|
+
* 而 LiveBench 的 `chat_completion_openai` 只看模型名里有没有 "gpt":非 gpt 模型一律发
|
|
558
|
+
* `max_tokens`。**实测 aiportx-claude 因此慢 5 倍以上**:同一道 olympiad、同一
|
|
559
|
+
* `reasoning_effort=max`,发 `max_completion_tokens=64000` 用 224s 跑完(最大空档 106s);
|
|
560
|
+
* 发 `max_tokens=64000` 要 1232s,中途 449s 一个 chunk 都没有 —— 面板里就是"卡死"。
|
|
561
|
+
* 所以面板在生成的模型配置里显式写死用哪个字段,LiveBench 看到就不再自己猜。
|
|
562
|
+
* @param {{id?: string, baseURL?: string|null, compat?: object}} provider
|
|
563
|
+
* @returns {"max_tokens"|"max_completion_tokens"}
|
|
564
|
+
*/
|
|
565
|
+
function maxTokensFieldFor(provider) {
|
|
566
|
+
const explicit = provider?.compat?.maxTokensField;
|
|
567
|
+
if (explicit === "max_tokens" || explicit === "max_completion_tokens") return explicit;
|
|
568
|
+
const id = String(provider?.id ?? "");
|
|
569
|
+
const url = String(provider?.baseURL ?? "").toLowerCase();
|
|
570
|
+
const useMaxTokens = url.includes("chutes.ai")
|
|
571
|
+
|| id === "deepseek" || url.includes("deepseek.com")
|
|
572
|
+
|| id === "moonshotai" || id === "moonshotai-cn" || url.includes("api.moonshot.")
|
|
573
|
+
|| id === "cloudflare-ai-gateway" || url.includes("gateway.ai.cloudflare.com")
|
|
574
|
+
|| id === "together" || url.includes("api.together.ai") || url.includes("api.together.xyz")
|
|
575
|
+
|| id === "nvidia" || url.includes("integrate.api.nvidia.com")
|
|
576
|
+
|| id === "ant-ling" || url.includes("api.ant-ling.com")
|
|
577
|
+
|| id === "zai" || id === "zai-coding-cn" || url.includes("api.z.ai") || url.includes("open.bigmodel.cn");
|
|
578
|
+
return useMaxTokens ? "max_tokens" : "max_completion_tokens";
|
|
579
|
+
}
|
|
580
|
+
|
|
581
|
+
/**
|
|
582
|
+
* 会话窗口(pi-ai 的 anthropic-messages 通道)给每个推理档位的 thinking 预算:
|
|
583
|
+
* minimal 1024 / low 2048 / medium 8192 / high 16384,`xhigh` 与 `max` 都按 `high` 处理
|
|
584
|
+
* (对应 `simple-options.js` 的 `DEFAULT_THINKING_BUDGETS` + `clampReasoning`)。
|
|
585
|
+
*
|
|
586
|
+
* **这个预算必须转发**:会话窗口在 anthropic 通道上固定发
|
|
587
|
+
* `thinking: {type: enabled, budget_tokens: N}`(并且因为开了 thinking,
|
|
588
|
+
* 就**完全不发 temperature**)。LiveBench 的 `chat_completion_anthropic` 不会自己发
|
|
589
|
+
* `thinking`,于是上游走"自适应思考":一道 olympiad 能安静地想十几分钟、
|
|
590
|
+
* 中途一个 chunk 都没有(面板里表现为卡死 + 一堆 RemoteProtocolError)。
|
|
591
|
+
* 面板把同一份预算写进生成的模型配置,让两边请求形状一致。
|
|
592
|
+
*/
|
|
593
|
+
const THINKING_BUDGETS = { minimal: 1024, low: 2048, medium: 8192, high: 16384, xhigh: 16384, max: 16384 };
|
|
594
|
+
function thinkingBudgetForEffort(effort) {
|
|
595
|
+
return Number.isFinite(THINKING_BUDGETS[effort]) ? THINKING_BUDGETS[effort] : null;
|
|
596
|
+
}
|
|
597
|
+
|
|
475
598
|
/**
|
|
476
599
|
* Write a single-entry model config into LiveBench's model_configs directory so
|
|
477
600
|
* `get_model_config(displayName)` resolves it (instead of the bare custom-model
|
|
@@ -479,7 +602,7 @@ function displayModelName(providerId, modelId) {
|
|
|
479
602
|
* The file is regenerated on every start; secrets never go in here.
|
|
480
603
|
* @returns {string|null} error message, or null on success.
|
|
481
604
|
*/
|
|
482
|
-
function writeGeneratedModelConfig(layout, YAML, { displayName, modelId, reasoningEffort, protocol }) {
|
|
605
|
+
function writeGeneratedModelConfig(layout, YAML, { displayName, modelId, reasoningEffort, protocol, maxTokens = null, maxTokensField = null, thinkingBudget = null, thinkingDisabled = false }) {
|
|
483
606
|
const configDir = join(layout.livebenchDir, "model", "model_configs");
|
|
484
607
|
// 每个模型一个独立配置文件:并发启动多模型时共享单文件会互相覆盖,
|
|
485
608
|
// 导致先启动的进程读到别人的配置 → anthropic/responses 路由失效 → 全 $ERROR$。
|
|
@@ -491,26 +614,56 @@ function writeGeneratedModelConfig(layout, YAML, { displayName, modelId, reasoni
|
|
|
491
614
|
`display_name: ${displayName}`,
|
|
492
615
|
"api_name:",
|
|
493
616
|
];
|
|
617
|
+
// 输出上限的字段名:写进 api_kwargs 后,LiveBench 的
|
|
618
|
+
// `if 'max_tokens' not in api_kwargs and 'max_completion_tokens' not in api_kwargs`
|
|
619
|
+
// 分支会被跳过,于是"用哪个字段"完全由面板决定(见 maxTokensFieldFor)。
|
|
620
|
+
const limitLine = protocol === "openai" && maxTokensField === "max_completion_tokens"
|
|
621
|
+
&& Number.isFinite(Number(maxTokens)) && Number(maxTokens) > 0
|
|
622
|
+
? ` max_completion_tokens: ${Number(maxTokens)}`
|
|
623
|
+
: null;
|
|
494
624
|
if (protocol === "anthropic") {
|
|
495
625
|
// Route through LiveBench's native anthropic client; the endpoint is
|
|
496
626
|
// pointed at the provider proxy via ANTHROPIC_BASE_URL (spawn env).
|
|
497
627
|
lines.push(` anthropic: ${modelId}`, "default_provider: anthropic");
|
|
498
|
-
|
|
499
|
-
|
|
500
|
-
|
|
628
|
+
// 会话窗口在这个通道上固定带 thinking(并因此不带 temperature),面板照抄,
|
|
629
|
+
// 否则上游会走"自适应思考",一题能安静地想十几分钟(见 thinkingBudgetForEffort)。
|
|
630
|
+
if (Number.isFinite(Number(thinkingBudget)) && Number(thinkingBudget) > 0) {
|
|
631
|
+
lines.push(
|
|
632
|
+
"api_kwargs:",
|
|
633
|
+
" default:",
|
|
634
|
+
" thinking:",
|
|
635
|
+
" type: enabled",
|
|
636
|
+
` budget_tokens: ${Number(thinkingBudget)}`,
|
|
637
|
+
// thinking 与 temperature 互斥:显式置 None → LiveBench 映射成 NOT_GIVEN 而不发送
|
|
638
|
+
" temperature: null",
|
|
639
|
+
);
|
|
640
|
+
} else if (thinkingDisabled) {
|
|
641
|
+
// 用户把强度选成 off:会话窗口发的是 {type: disabled},不带 thinking 反而会让
|
|
642
|
+
// 上游走默认(实测默认是"开思考")。
|
|
643
|
+
lines.push(
|
|
644
|
+
"api_kwargs:",
|
|
645
|
+
" default:",
|
|
646
|
+
" thinking:",
|
|
647
|
+
" type: disabled",
|
|
648
|
+
" temperature: null",
|
|
649
|
+
);
|
|
501
650
|
}
|
|
502
651
|
} else if (protocol === "openai_responses") {
|
|
503
652
|
// OpenAI Responses API proxy: provider name matches
|
|
504
653
|
// get_api_function('openai_responses') → chat_completion_openai_responses,
|
|
505
654
|
// which converts reasoning_effort to reasoning:{effort} itself.
|
|
506
655
|
lines.push(` openai_responses: ${modelId}`);
|
|
507
|
-
if (reasoningEffort) {
|
|
508
|
-
lines.push("api_kwargs:", " default:"
|
|
656
|
+
if (reasoningEffort || limitLine !== null) {
|
|
657
|
+
lines.push("api_kwargs:", " default:");
|
|
658
|
+
if (reasoningEffort) lines.push(` reasoning_effort: ${reasoningEffort}`);
|
|
659
|
+
if (limitLine !== null) lines.push(limitLine);
|
|
509
660
|
}
|
|
510
661
|
} else {
|
|
511
662
|
lines.push(` local: ${modelId}`);
|
|
512
|
-
if (reasoningEffort) {
|
|
513
|
-
lines.push("api_kwargs:", " default:"
|
|
663
|
+
if (reasoningEffort || limitLine !== null) {
|
|
664
|
+
lines.push("api_kwargs:", " default:");
|
|
665
|
+
if (reasoningEffort) lines.push(` reasoning_effort: ${reasoningEffort}`);
|
|
666
|
+
if (limitLine !== null) lines.push(limitLine);
|
|
514
667
|
}
|
|
515
668
|
}
|
|
516
669
|
lines.push("");
|
|
@@ -584,19 +737,119 @@ async function streamJsonl(path, onRow) {
|
|
|
584
737
|
}
|
|
585
738
|
}
|
|
586
739
|
|
|
740
|
+
/** 从答案行里抽 answer_id(同样只做正则,不 JSON.parse)。 */
|
|
741
|
+
const ANSWER_ID_RE = /"answer_id"\s*:\s*"([A-Za-z0-9_.:-]{6,80})"/;
|
|
742
|
+
|
|
743
|
+
/** 一行答案归并后的状态:这一题最终算不算"做出来了"。 */
|
|
744
|
+
function newAnswerScan() {
|
|
745
|
+
return { byQid: new Map(), anon: 0 };
|
|
746
|
+
}
|
|
747
|
+
|
|
587
748
|
/**
|
|
588
|
-
*
|
|
589
|
-
*
|
|
749
|
+
* 把一行答案并进扫描状态(同一 question_id 后写的覆盖先写的)。
|
|
750
|
+
*
|
|
751
|
+
* 为什么必须按 question_id 归并:断点补跑(--resume --retry-failures)重跑出来的
|
|
752
|
+
* 答案会**追加**到同一个文件里,旧行不会被删(LiveBench 自己靠 reorg_answer_file
|
|
753
|
+
* 取每个 question_id 的最后一行)。不归并就会把同一题算两次:作答数虚高、
|
|
754
|
+
* $ERROR$ 虚高,括号里的「没做出来 / 选择题数」跟着一起错。
|
|
590
755
|
*/
|
|
591
|
-
function
|
|
592
|
-
|
|
593
|
-
|
|
594
|
-
|
|
595
|
-
|
|
596
|
-
|
|
597
|
-
|
|
598
|
-
|
|
756
|
+
function scanAnswerLine(scan, line) {
|
|
757
|
+
if (line.trim().length === 0) return;
|
|
758
|
+
const idMatch = line.match(QUESTION_ID_RE);
|
|
759
|
+
// 没有 question_id 的行(理论上不该有)用序号占位,避免互相覆盖
|
|
760
|
+
const qid = idMatch !== null ? idMatch[1] : `\u0000#${scan.anon++}`;
|
|
761
|
+
const aMatch = line.match(ANSWER_ID_RE);
|
|
762
|
+
const isError = line.includes('"$ERROR$"');
|
|
763
|
+
scan.byQid.set(qid, {
|
|
764
|
+
isError,
|
|
765
|
+
isEmpty: !isError && isEmptyAnswerLine(line),
|
|
766
|
+
// 每次生成都会换一个 answer_id(gen_api_answer: shortuuid.uuid()),
|
|
767
|
+
// 判分侧靠它区分"这条判分属于哪一次答案",见 scoreJudgments。
|
|
768
|
+
answerId: aMatch !== null ? aMatch[1] : null,
|
|
769
|
+
});
|
|
770
|
+
}
|
|
771
|
+
|
|
772
|
+
/**
|
|
773
|
+
* 收尾统计:作答数 / API 失败数 / 空答案数 / 有有效答案的题号。
|
|
774
|
+
* @param {{byQid: Map<string,{isError:boolean,isEmpty:boolean,answerId:string|null}>}} scan
|
|
775
|
+
*/
|
|
776
|
+
function answerScanStats(scan) {
|
|
777
|
+
const okIds = new Set();
|
|
778
|
+
let errors = 0;
|
|
779
|
+
let empty = 0;
|
|
780
|
+
for (const [qid, rec] of scan.byQid) {
|
|
781
|
+
if (rec.isError) errors += 1;
|
|
782
|
+
else if (rec.isEmpty) empty += 1;
|
|
783
|
+
else okIds.add(qid);
|
|
784
|
+
}
|
|
785
|
+
return { answered: scan.byQid.size, errors, empty, okIds, byQid: scan.byQid };
|
|
786
|
+
}
|
|
787
|
+
|
|
788
|
+
/** 答案文件全文 → 统计(scanAnswerLine + answerScanStats 的便捷封装)。 */
|
|
789
|
+
function reduceAnswerLines(text) {
|
|
790
|
+
const scan = newAnswerScan();
|
|
791
|
+
for (const line of text.split("\n")) scanAnswerLine(scan, line);
|
|
792
|
+
return answerScanStats(scan);
|
|
793
|
+
}
|
|
794
|
+
|
|
795
|
+
/**
|
|
796
|
+
* 判分归并:**只保留属于本次答案的那一条判分**。
|
|
797
|
+
*
|
|
798
|
+
* 补跑会换新 answer_id,而旧答案(例如 $ERROR$ 那一版)的 0 分判分仍留在
|
|
799
|
+
* ground_truth_judgment.jsonl 里。老实现把所有判分行直接求和,于是
|
|
800
|
+
* "补跑后每题都做对"也会被旧 0 分拉低(8 题里 4 题满分 + 4 条旧 0 分 → 50%)。
|
|
801
|
+
*
|
|
802
|
+
* 规则:
|
|
803
|
+
* - 答案文件里这一题是 $ERROR$ / 空答案 → 判分一律不计(没做出来,不进分母);
|
|
804
|
+
* - 判分 answer_id 与当前答案一致 → 计入;
|
|
805
|
+
* - 判分/答案缺 answer_id(老数据)→ 无法区分,沿用旧行为计入;
|
|
806
|
+
* - 只剩旧答案的判分 → 丢弃(宁可少一题,也不把旧判分算进新成绩);
|
|
807
|
+
* - 答案文件里根本没有这一题(只有判分)→ 沿用旧行为计入。
|
|
808
|
+
*
|
|
809
|
+
* @param {Array<{questionId:string|null,answerId:string|null,score:number,tstamp:number}>} judgments
|
|
810
|
+
* @param {Map<string,{isError:boolean,isEmpty:boolean,answerId:string|null}>} byQid
|
|
811
|
+
* @returns {{judged:number, sum:number, time:number}}
|
|
812
|
+
*/
|
|
813
|
+
function scoreJudgments(judgments, byQid) {
|
|
814
|
+
const byQuestion = new Map();
|
|
815
|
+
const orphans = [];
|
|
816
|
+
for (const j of judgments) {
|
|
817
|
+
if (j.questionId === null) {
|
|
818
|
+
orphans.push(j);
|
|
819
|
+
continue;
|
|
820
|
+
}
|
|
821
|
+
const list = byQuestion.get(j.questionId);
|
|
822
|
+
if (list === undefined) byQuestion.set(j.questionId, [j]);
|
|
823
|
+
else list.push(j);
|
|
824
|
+
}
|
|
825
|
+
let sum = 0;
|
|
826
|
+
let judged = 0;
|
|
827
|
+
let time = 0;
|
|
828
|
+
const accept = (j) => {
|
|
829
|
+
sum += j.score;
|
|
830
|
+
judged += 1;
|
|
831
|
+
if (Number.isFinite(j.tstamp) && j.tstamp > time) time = j.tstamp;
|
|
832
|
+
};
|
|
833
|
+
for (const [qid, list] of byQuestion) {
|
|
834
|
+
const ans = byQid.get(qid);
|
|
835
|
+
if (ans !== undefined && (ans.isError || ans.isEmpty)) continue;
|
|
836
|
+
if (ans === undefined || ans.answerId === null) {
|
|
837
|
+
accept(list[list.length - 1]);
|
|
838
|
+
continue;
|
|
839
|
+
}
|
|
840
|
+
let picked = null;
|
|
841
|
+
for (let i = list.length - 1; i >= 0; i -= 1) {
|
|
842
|
+
if (list[i].answerId === ans.answerId) { picked = list[i]; break; }
|
|
843
|
+
}
|
|
844
|
+
if (picked === null) {
|
|
845
|
+
for (let i = list.length - 1; i >= 0; i -= 1) {
|
|
846
|
+
if (list[i].answerId === null) { picked = list[i]; break; }
|
|
847
|
+
}
|
|
848
|
+
}
|
|
849
|
+
if (picked !== null) accept(picked);
|
|
599
850
|
}
|
|
851
|
+
for (const j of orphans) accept(j);
|
|
852
|
+
return { judged, sum, time };
|
|
600
853
|
}
|
|
601
854
|
|
|
602
855
|
/**
|
|
@@ -706,9 +959,9 @@ async function readTextFileAsync(path) {
|
|
|
706
959
|
async function computeResults(dataDir) {
|
|
707
960
|
const now = Date.now();
|
|
708
961
|
if (resultsCache.data !== null && now - resultsCache.at < 5000) return resultsCache.data;
|
|
709
|
-
const judged = new Map(); // `${model}\u0000${task}` -> {model, category, task,
|
|
962
|
+
const judged = new Map(); // `${model}\u0000${task}` -> {model, category, task, items: []}
|
|
710
963
|
const totals = new Map(); // task -> question count
|
|
711
|
-
const answers = new Map(); // `${model}\u0000${task}` ->
|
|
964
|
+
const answers = new Map(); // `${model}\u0000${task}` -> 答案扫描状态(按 question_id 归并)
|
|
712
965
|
const catByTask = new Map();
|
|
713
966
|
if (!existsSync(dataDir)) return { rows: [], taskTotals: {} };
|
|
714
967
|
// 读取每次评测的配置范围(displayName -> {benchNames, begin, end})
|
|
@@ -735,29 +988,13 @@ async function computeResults(dataDir) {
|
|
|
735
988
|
const text = await readTextFileAsync(join(answerDir, file));
|
|
736
989
|
if (text === null) continue;
|
|
737
990
|
const key = `${model}\u0000${task}`;
|
|
738
|
-
|
|
739
|
-
//
|
|
740
|
-
|
|
741
|
-
|
|
742
|
-
// 只用正则抽 id(不 JSON.parse),对几百 KB~几 MB 的答案文件也够快。
|
|
743
|
-
const runMetaEntry = runMeta.get(model);
|
|
744
|
-
const trackIds = baselineIdsFor(runMetaEntry) !== null;
|
|
745
|
-
if (trackIds && stat.okIds === null) stat.okIds = new Set();
|
|
746
|
-
for (const line of text.split("\n")) {
|
|
747
|
-
if (line.trim().length === 0) continue;
|
|
748
|
-
stat.answered += 1;
|
|
749
|
-
const isError = line.includes('"$ERROR$"');
|
|
750
|
-
const isEmpty = !isError && isEmptyAnswerLine(line);
|
|
751
|
-
if (isError) stat.errors += 1;
|
|
752
|
-
else if (isEmpty) stat.empty += 1;
|
|
753
|
-
if (trackIds && !isError && !isEmpty) {
|
|
754
|
-
const idMatch = line.match(QUESTION_ID_RE);
|
|
755
|
-
if (idMatch !== null) stat.okIds.add(idMatch[1]);
|
|
756
|
-
}
|
|
757
|
-
}
|
|
991
|
+
// 逐行归并(不 JSON.parse 大答案行):同一题保留最后一行,
|
|
992
|
+
// 也就是补跑后的新答案 —— 与 LiveBench reorg_answer_file 的口径一致。
|
|
993
|
+
const scan = answers.get(key) ?? newAnswerScan();
|
|
994
|
+
for (const line of text.split("\n")) scanAnswerLine(scan, line);
|
|
758
995
|
// 空文件(尚未产出答案,或成绩已被删除只留下 0 字节壳)不计入结果:
|
|
759
996
|
// 否则删掉的记录会在下一次扫描时凭“文件名仍在”重新冒出来。
|
|
760
|
-
if (
|
|
997
|
+
if (scan.byQid.size > 0) answers.set(key, scan);
|
|
761
998
|
}
|
|
762
999
|
const jText = await readTextFileAsync(join(taskDir, "model_judgment", "ground_truth_judgment.jsonl"));
|
|
763
1000
|
if (jText !== null) {
|
|
@@ -769,11 +1006,13 @@ async function computeResults(dataDir) {
|
|
|
769
1006
|
const score = typeof row.score === "number" ? row.score : Number(row.score);
|
|
770
1007
|
if (model === null || !Number.isFinite(score) || score < 0) continue;
|
|
771
1008
|
const key = `${model}\u0000${task}`;
|
|
772
|
-
const entry = judged.get(key) ?? { model, category, task,
|
|
773
|
-
entry.
|
|
774
|
-
|
|
775
|
-
|
|
776
|
-
|
|
1009
|
+
const entry = judged.get(key) ?? { model, category, task, items: [] };
|
|
1010
|
+
entry.items.push({
|
|
1011
|
+
questionId: typeof row.question_id === "string" ? row.question_id : null,
|
|
1012
|
+
answerId: typeof row.answer_id === "string" ? row.answer_id : null,
|
|
1013
|
+
score,
|
|
1014
|
+
tstamp: Number(row.tstamp),
|
|
1015
|
+
});
|
|
777
1016
|
judged.set(key, entry);
|
|
778
1017
|
}
|
|
779
1018
|
}
|
|
@@ -784,16 +1023,16 @@ async function computeResults(dataDir) {
|
|
|
784
1023
|
const keys = new Set([...judged.keys(), ...answers.keys()]);
|
|
785
1024
|
const rows = [...keys].map((key) => {
|
|
786
1025
|
const entry = judged.get(key);
|
|
787
|
-
const
|
|
1026
|
+
const scan = answers.get(key) ?? null;
|
|
788
1027
|
const [model, task] = key.split("\u0000");
|
|
789
1028
|
const category = entry?.category ?? catByTask.get(task) ?? "";
|
|
790
1029
|
const total = totals.get(task) ?? 0;
|
|
791
|
-
|
|
792
|
-
// 「做出来」的题 =
|
|
793
|
-
//
|
|
794
|
-
|
|
795
|
-
|
|
796
|
-
const empty = stat.empty
|
|
1030
|
+
// 答案先归并到「每题一条最新记录」,再用它给判分定性:
|
|
1031
|
+
// 「做出来」的题 = 有有效答案的题;思考吃满 max_tokens(正文为空)、
|
|
1032
|
+
// API 失败都不是"做错了",不能进正确率分母,也不能被旧判分拉低。
|
|
1033
|
+
const stat = scan === null ? { answered: 0, errors: 0, empty: 0, okIds: new Set(), byQid: new Map() } : answerScanStats(scan);
|
|
1034
|
+
const { judged: judgedCount, sum, time: judgedTime } = scoreJudgments(entry?.items ?? [], stat.byQid);
|
|
1035
|
+
const empty = stat.empty;
|
|
797
1036
|
const notProduced = stat.errors + empty;
|
|
798
1037
|
const done = Math.max(0, stat.answered - notProduced);
|
|
799
1038
|
// 选择题数 = 用户在「题目序号范围」里选择的题数(end 含端点)。
|
|
@@ -834,20 +1073,21 @@ async function computeResults(dataDir) {
|
|
|
834
1073
|
// 实际尝试数比设定还多(断点重跑等)时,以实际为准
|
|
835
1074
|
if (configured < stat.answered) configured = stat.answered;
|
|
836
1075
|
const notDone = Math.max(0, configured - done);
|
|
837
|
-
// 正确率 = 做对的题 / **做出来的题**(API 失败、思考吃满 token
|
|
838
|
-
|
|
839
|
-
|
|
1076
|
+
// 正确率 = 做对的题 / **做出来的题**(API 失败、思考吃满 token 没产出答案的都不进分母)。
|
|
1077
|
+
// judgedCount 已经只统计"属于本次答案"的判分,所以这里不再减 notProduced:
|
|
1078
|
+
// 那些题的判分(如果有)在 scoreJudgments 里就已经被剔除了。
|
|
1079
|
+
const score = judgedCount > 0 ? (sum / judgedCount) * 100 : null;
|
|
840
1080
|
// Baseline 评测:算出**具体哪几个题号没做出来**(题号 = 该题库内的 0 起下标,
|
|
841
1081
|
// 与「题目序号范围」的编号一致,方便直接复制去重跑)。
|
|
842
|
-
const missingIndexes = missingBaselineIndexes(meta, stat.okIds
|
|
1082
|
+
const missingIndexes = missingBaselineIndexes(meta, stat.okIds);
|
|
843
1083
|
return {
|
|
844
1084
|
model,
|
|
845
1085
|
category,
|
|
846
1086
|
task,
|
|
847
1087
|
score,
|
|
848
|
-
judged:
|
|
1088
|
+
judged: judgedCount,
|
|
849
1089
|
total,
|
|
850
|
-
time:
|
|
1090
|
+
time: judgedTime > 0 ? judgedTime : null,
|
|
851
1091
|
answered: stat.answered,
|
|
852
1092
|
errors: stat.errors,
|
|
853
1093
|
empty,
|
|
@@ -872,28 +1112,34 @@ async function computeResults(dataDir) {
|
|
|
872
1112
|
}
|
|
873
1113
|
|
|
874
1114
|
/**
|
|
875
|
-
* 异步统计某模型在指定 bench
|
|
876
|
-
*
|
|
1115
|
+
* 异步统计某模型在指定 bench 范围内"最终仍是 $ERROR$"的题数(用于自动补跑判断)。
|
|
1116
|
+
*
|
|
1117
|
+
* benchPaths 必须是**服务端解析后的** bench 列表(`live_bench/<cat>[/<task>]`)。
|
|
1118
|
+
* 早期版本直接读 body.benchNames,而 Baseline 评测根本不送这个字段(它送的是
|
|
1119
|
+
* baseline/baselineTask/baselinePicks),于是 baseline 跑完全军覆没也不会自动补跑 ——
|
|
1120
|
+
* 这正是 kimi-k3 那次 8 题全 502、面板只留一行 `—`、没有任何重试的原因。
|
|
1121
|
+
*
|
|
1122
|
+
* 按 question_id 归并后统计(同一题补跑成功过就不再算失败),流式读取不阻塞事件循环。
|
|
1123
|
+
* @returns {Promise<{errors:number, failedIds:string[]}>}
|
|
877
1124
|
*/
|
|
878
|
-
async function countErrorAnswersAsync(dataDir, displayName,
|
|
879
|
-
if (!/^[A-Za-z0-9._@-]{1,160}$/.test(displayName)) return 0;
|
|
880
|
-
const benches = Array.isArray(
|
|
881
|
-
|
|
1125
|
+
async function countErrorAnswersAsync(dataDir, displayName, benchPaths) {
|
|
1126
|
+
if (!/^[A-Za-z0-9._@-]{1,160}$/.test(displayName)) return { errors: 0, failedIds: [] };
|
|
1127
|
+
const benches = Array.isArray(benchPaths) ? benchPaths : [];
|
|
1128
|
+
const scan = newAnswerScan();
|
|
882
1129
|
for (const bn of benches) {
|
|
883
|
-
const parts = bn.split("/").filter((s) => s.length > 0); // live_bench/<cat>[/<task>]
|
|
1130
|
+
const parts = String(bn).split("/").filter((s) => s.length > 0); // live_bench/<cat>[/<task>]
|
|
884
1131
|
if (parts.length < 2) continue;
|
|
885
1132
|
const category = parts[1];
|
|
886
1133
|
const tasks = parts.length >= 3 ? [parts[2]] : readDirSafe(join(dataDir, category));
|
|
887
1134
|
for (const task of tasks) {
|
|
888
1135
|
if (!/^[A-Za-z0-9_]{1,80}$/.test(task)) continue;
|
|
889
1136
|
const file = join(dataDir, category, task, "model_answer", displayName + ".jsonl");
|
|
890
|
-
await streamJsonl(file, (line) =>
|
|
891
|
-
const parsed = parseAnswerLineLight(line);
|
|
892
|
-
if (parsed && parsed.is_error) errors += 1;
|
|
893
|
-
});
|
|
1137
|
+
await streamJsonl(file, (line) => scanAnswerLine(scan, line));
|
|
894
1138
|
}
|
|
895
1139
|
}
|
|
896
|
-
|
|
1140
|
+
const failedIds = [];
|
|
1141
|
+
for (const [qid, rec] of scan.byQid) if (rec.isError) failedIds.push(qid);
|
|
1142
|
+
return { errors: failedIds.length, failedIds };
|
|
897
1143
|
}
|
|
898
1144
|
|
|
899
1145
|
/**
|
|
@@ -956,11 +1202,62 @@ function validateBenchName(bn) {
|
|
|
956
1202
|
return null;
|
|
957
1203
|
}
|
|
958
1204
|
|
|
1205
|
+
/**
|
|
1206
|
+
* LiveBench 的分类列表(权威来源是 LiveBench 自己的 `LIVE_BENCH_CATEGORIES`)。
|
|
1207
|
+
*
|
|
1208
|
+
* 面板要判断"本次要跑的题目数据集是否都已缓存",而 `--bench-name live_bench`
|
|
1209
|
+
* (全部分类)的路径里没有分类段 —— 不补上分类列表就会误判成"没缓存",
|
|
1210
|
+
* 于是每个分类都要走一遍 huggingface.co 的 HEAD 超时重试(WinError 10060 刷屏)。
|
|
1211
|
+
* 所以直接从 LiveBench 源码里读那份列表,读不到再用兜底常量。
|
|
1212
|
+
* @param {string} root LiveBench 根目录
|
|
1213
|
+
* @returns {string[]} 分类名(如 coding / math / reasoning …)
|
|
1214
|
+
*/
|
|
1215
|
+
const LIVEBENCH_CATEGORIES_FALLBACK = ["coding", "data_analysis", "instruction_following", "math", "reasoning", "language"];
|
|
1216
|
+
function livebenchCategories(root) {
|
|
1217
|
+
try {
|
|
1218
|
+
const text = readFileSync(join(root, "livebench", "common.py"), "utf8");
|
|
1219
|
+
const block = text.match(/LIVE_BENCH_CATEGORIES\s*=\s*\[([^\]]*)\]/);
|
|
1220
|
+
if (block !== null) {
|
|
1221
|
+
const names = [...block[1].matchAll(/"([A-Za-z0-9_]{1,40})"/g)].map((m) => m[1]);
|
|
1222
|
+
if (names.length > 0) return names;
|
|
1223
|
+
}
|
|
1224
|
+
} catch { /* 源码读不到(改过结构/权限)时用兜底列表 */ }
|
|
1225
|
+
return LIVEBENCH_CATEGORIES_FALLBACK;
|
|
1226
|
+
}
|
|
1227
|
+
|
|
1228
|
+
/**
|
|
1229
|
+
* 一次运行需要哪些题目数据集(= 分类)?
|
|
1230
|
+
*
|
|
1231
|
+
* bench 路径带分类(`live_bench/<分类>[/<任务>]`)就用它;
|
|
1232
|
+
* 只给 `live_bench`(全跑)时就是**全部分类** —— 这里漏掉分类是实测踩过的坑:
|
|
1233
|
+
* 答案为空的集合 → 判成"没缓存" → 每个分类白刷 5 轮 HF 超时重试。
|
|
1234
|
+
* @param {string[]} benchPaths 服务端解析后的 bench 列表
|
|
1235
|
+
* @param {string[]} categories LiveBench 全部分类
|
|
1236
|
+
* @returns {Set<string>}
|
|
1237
|
+
*/
|
|
1238
|
+
function hfCategoriesFor(benchPaths, categories) {
|
|
1239
|
+
const out = new Set();
|
|
1240
|
+
for (const bn of benchPaths) {
|
|
1241
|
+
const parts = String(bn).split("/").filter((s) => s.length > 0);
|
|
1242
|
+
if (parts.length >= 2 && /^[A-Za-z0-9_]{1,80}$/.test(parts[1])) out.add(parts[1]);
|
|
1243
|
+
else for (const cat of categories) out.add(cat);
|
|
1244
|
+
}
|
|
1245
|
+
return out;
|
|
1246
|
+
}
|
|
1247
|
+
|
|
959
1248
|
function apply(ctx) {
|
|
960
1249
|
/** All runs: runId -> record. Several evaluations may run concurrently. */
|
|
961
1250
|
const runs = new Map();
|
|
962
1251
|
|
|
963
|
-
|
|
1252
|
+
/**
|
|
1253
|
+
* 启动一次评测。
|
|
1254
|
+
* @param {object} body 面板送来的请求体
|
|
1255
|
+
* @param {number} autoRetries 已经自动补跑过几轮
|
|
1256
|
+
* @param {string|null} displayNameOverride 自动补跑时复用首次运行的 display-name:
|
|
1257
|
+
* answers 文件同名 → `--resume --retry-failures` 只重跑失败题(其它题的答案与判分都留着),
|
|
1258
|
+
* 成绩表里也仍然只有一行,而不是每次补跑多冒出一行。
|
|
1259
|
+
*/
|
|
1260
|
+
const startRun = async (body, autoRetries = 0, displayNameOverride = null) => {
|
|
964
1261
|
const layout = livebenchLayout();
|
|
965
1262
|
if (!layout.available) {
|
|
966
1263
|
return { status: 409, payload: { ok: false, error: `LiveBench venv not found under ${layout.root}` } };
|
|
@@ -1054,12 +1351,21 @@ function apply(ctx) {
|
|
|
1054
1351
|
const _now = new Date();
|
|
1055
1352
|
const _pad = (n) => String(n).padStart(2, "0");
|
|
1056
1353
|
const runStamp = `r${_now.getFullYear()}${_pad(_now.getMonth() + 1)}${_pad(_now.getDate())}-${_pad(_now.getHours())}${_pad(_now.getMinutes())}${_pad(_now.getSeconds())}`;
|
|
1057
|
-
const displayName =
|
|
1354
|
+
const displayName = typeof displayNameOverride === "string" && displayNameOverride.length > 0
|
|
1355
|
+
? displayNameOverride
|
|
1356
|
+
: displayModelName(providerId || "direct", modelId) + (hasEffortSuffix ? "@" + effort : "") + "__" + runStamp;
|
|
1058
1357
|
let writeError = null; // Anthropic-protocol proxies cannot go through --api-base (that path
|
|
1059
1358
|
// speaks OpenAI Chat Completions). Instead the generated model config
|
|
1060
1359
|
// selects LiveBench's native anthropic client and the spawn env points
|
|
1061
1360
|
// the SDK at the proxy (ANTHROPIC_BASE_URL / ANTHROPIC_API_KEY).
|
|
1062
|
-
|
|
1361
|
+
//
|
|
1362
|
+
// 规则(用户要求 + 实测):**只要是 Anthropic/Claude 系模型,一律走 anthropic-messages**,
|
|
1363
|
+
// 不再看 settings.yaml 里写的是 openai-completions 还是 anthropic-messages ——
|
|
1364
|
+
// 会话窗口对这类模型用的就是 /v1/messages,两边协议一致才谈得上"形状一致"。
|
|
1365
|
+
// baseURL 会被规范化(剥掉 /v1、/v1/messages),因为 anthropic SDK 自己拼 /v1/messages。
|
|
1366
|
+
const anthropicRouteBase = provider?.baseURL ? anthropicBaseUrl(provider.baseURL) : "";
|
|
1367
|
+
const isAnthropicRoute = anthropicRouteBase.length > 0
|
|
1368
|
+
&& (provider.api === "anthropic-messages" || isAnthropicModel(providerId, modelId));
|
|
1063
1369
|
// OpenAI Responses-API proxies (api: openai-responses) select LiveBench's
|
|
1064
1370
|
// openai_responses client; --api-base + LIVEBENCH_API_KEY route them the
|
|
1065
1371
|
// usual way, and reasoning_effort converts to reasoning:{effort} inside.
|
|
@@ -1072,26 +1378,44 @@ function apply(ctx) {
|
|
|
1072
1378
|
// provider / api_kwargs / max_tokens —— 实测会把 temperature 悄悄变成 1.0、
|
|
1073
1379
|
// max_tokens 变成 131072,等于用户选的 provider 被换掉。写了自己的配置,
|
|
1074
1380
|
// 解析结果就是确定的 {local: <modelId>} + --api-base。
|
|
1381
|
+
// max-tokens:不能超过 harness 给这个模型声明的上限(settings.yaml 的
|
|
1382
|
+
// models[].maxTokens,例如 claude-fable-5-1 = 64000)。面板默认 65536 就是为了
|
|
1383
|
+
// 让推理模型别把预算烧在思考里,但超过渠道上限容易被上游挂住或直接拒。
|
|
1384
|
+
const requestedMaxTokens = asInt(body.maxTokens, 256, 200000, 65536);
|
|
1385
|
+
const modelMaxTokens = modelMeta !== null && Number(modelMeta.maxTokens) > 0 ? Number(modelMeta.maxTokens) : null;
|
|
1386
|
+
const maxTokens = modelMaxTokens !== null ? Math.min(requestedMaxTokens, modelMaxTokens) : requestedMaxTokens;
|
|
1075
1387
|
writeError = writeGeneratedModelConfig(layout, loadYaml(profileDir), {
|
|
1076
1388
|
displayName,
|
|
1077
1389
|
modelId,
|
|
1078
1390
|
reasoningEffort: isAnthropicRoute ? null : effort,
|
|
1079
1391
|
protocol: isAnthropicRoute ? "anthropic" : isOpenAIResponsesRoute ? "openai_responses" : "openai",
|
|
1392
|
+
// 输出上限与字段名:让 LiveBench 用**会话窗口同样的字段**发出去
|
|
1393
|
+
// (aiportx 这类中转要 max_completion_tokens;发 max_tokens 会慢数倍甚至卡死)
|
|
1394
|
+
maxTokens,
|
|
1395
|
+
// provider 上已经算好(providerCapsFromSettings 里调 maxTokensFieldFor,
|
|
1396
|
+
// 会尊重 settings.yaml 显式声明的 compat.maxTokensField)
|
|
1397
|
+
maxTokensField: provider !== null ? provider.maxTokensField : null,
|
|
1398
|
+
// anthropic 通道:把会话窗口那份 thinking 预算一起发出去(见 thinkingBudgetForEffort)
|
|
1399
|
+
thinkingBudget: isAnthropicRoute ? thinkingBudgetForEffort(effort) : null,
|
|
1400
|
+
thinkingDisabled: isAnthropicRoute && effort === "off",
|
|
1080
1401
|
});
|
|
1081
1402
|
if (writeError) return { status: 500, payload: { ok: false, error: writeError } };
|
|
1082
1403
|
|
|
1083
1404
|
const benchParts = ["live_bench", ...(category ? [category] : []), ...(task ? [task] : [])];
|
|
1405
|
+
// 服务端最终使用的 bench 列表:自动补跑扫描(找 $ERROR$)和 HF 缓存判断都用它,
|
|
1406
|
+
// 不能用 body.benchNames —— baseline 评测不送这个字段。
|
|
1407
|
+
const benchPaths = benchNames !== null ? benchNames : [benchParts.join("/")];
|
|
1084
1408
|
const args = [
|
|
1085
1409
|
"run_livebench.py",
|
|
1086
1410
|
"--model", displayName,
|
|
1087
1411
|
"--model-display-name", displayName,
|
|
1088
1412
|
"--bench-name",
|
|
1089
|
-
...
|
|
1413
|
+
...benchPaths,
|
|
1090
1414
|
"--livebench-release-option", release,
|
|
1091
1415
|
// 上限放到 200k:推理模型在难题上会把 max_tokens 全烧在思考里,正文为空
|
|
1092
1416
|
// (eval_status=token_exhaustion),此时"做不出来"其实是预算不够而不是能力不够。
|
|
1093
1417
|
// 旧上限 32768 在 olympiad 这类题上会直接把参考模型卡死。
|
|
1094
|
-
"--max-tokens", String(
|
|
1418
|
+
"--max-tokens", String(maxTokens),
|
|
1095
1419
|
"--parallel-requests", String(asInt(body.parallel, 1, 8, 1)),
|
|
1096
1420
|
// 流式:中转网关(Cloudflare 等)对非流式请求有 ~100s 超时(524),
|
|
1097
1421
|
// 高推理强度模型思考数分钟必然超时。流式保持字节流动可规避。
|
|
@@ -1116,10 +1440,38 @@ function apply(ctx) {
|
|
|
1116
1440
|
if (body.resume === true) args.push("--resume");
|
|
1117
1441
|
if (body.retryFailures === true) args.push("--retry-failures");
|
|
1118
1442
|
|
|
1443
|
+
// HuggingFace 只用来取题目数据集。国内网络下 huggingface.co 常常连不通:
|
|
1444
|
+
// 每次评测都要先 HEAD 请求超时重试 5 轮(1+2+4+8+8s 退避 + 每次约 21s 的连接超时,
|
|
1445
|
+
// 实测刷一屏 WinError 10060),再回落到本地缓存。
|
|
1446
|
+
// 本次要跑的分类**都已在缓存里**时直接离线启动:题目一样,省掉整段等待。
|
|
1447
|
+
// 缺任何一个分类的缓存就不设离线 —— 保留联网下载能力,避免"新 release
|
|
1448
|
+
// 的题目还没下过"这种正常场景被离线模式直接判死。
|
|
1449
|
+
//
|
|
1450
|
+
// 注意 `--bench-name live_bench`(全部分类)时 bench 路径里没有分类段,
|
|
1451
|
+
// 必须自己补上分类列表,否则会被误判成"没缓存"→ 6 个分类各刷一遍超时重试
|
|
1452
|
+
// (实测:跑到第 3 个分类已经白等 10 分钟,用户看到的"总是报错"就是这个)。
|
|
1453
|
+
const hfCategories = hfCategoriesFor(benchPaths, livebenchCategories(layout.root));
|
|
1454
|
+
const hfMissing = [...hfCategories].filter((cat) => !existsSync(join(layout.root, ".hf_cache", "datasets", `livebench___${cat}`)));
|
|
1455
|
+
const hfCached = hfCategories.size > 0 && hfMissing.length === 0;
|
|
1456
|
+
// harness 给这个渠道声明的"流多久没数据就掐断"(毫秒)→ 秒
|
|
1457
|
+
const providerIdleSeconds = Number.isFinite(Number(provider?.streamIdleTimeoutMs)) && Number(provider.streamIdleTimeoutMs) > 0
|
|
1458
|
+
? Math.max(30, Math.round(Number(provider.streamIdleTimeoutMs) / 1000))
|
|
1459
|
+
: null;
|
|
1460
|
+
|
|
1119
1461
|
const env = {
|
|
1120
1462
|
...process.env,
|
|
1121
1463
|
HF_HOME: join(layout.root, ".hf_cache"),
|
|
1122
1464
|
HF_HUB_DISABLE_SYMLINKS_WARNING: "1",
|
|
1465
|
+
// 真要联网时也别卡太久:把 etag/下载超时压低,失败得快一点。
|
|
1466
|
+
HF_HUB_ETAG_TIMEOUT: "5",
|
|
1467
|
+
HF_HUB_DOWNLOAD_TIMEOUT: "30",
|
|
1468
|
+
...(hfCached ? { HF_HUB_OFFLINE: "1", HF_DATASETS_OFFLINE: "1" } : {}),
|
|
1469
|
+
// 静默看门狗阈值:优先用 harness 给这个渠道声明的 streamIdleTimeoutMs
|
|
1470
|
+
// (code-claude = 120000 → 120s),没声明就不设,由 LiveBench 用自己的
|
|
1471
|
+
// 保守默认(600s)。**不能一律压到 120s** —— Claude 这类模型答一道
|
|
1472
|
+
// olympiad 本来就要 2~5 分钟、输出 1~2 万 token,阈值太小会把正常长思考
|
|
1473
|
+
// 当成"挂住"反复掐断(实测:一题被掐 5 次、每次 120s,最后记 $ERROR$)。
|
|
1474
|
+
...(providerIdleSeconds !== null ? { LIVEBENCH_STREAM_IDLE_TIMEOUT: String(providerIdleSeconds) } : {}),
|
|
1123
1475
|
// 限制评测进程的数值库线程(numpy/pyarrow/MKL 默认吃满全核,
|
|
1124
1476
|
// 多模型并发时会把 node 事件循环饿死 → websocket 断连)。
|
|
1125
1477
|
OMP_NUM_THREADS: "1",
|
|
@@ -1161,16 +1513,21 @@ function apply(ctx) {
|
|
|
1161
1513
|
} catch { /* credential seam unavailable — fall through to env */ }
|
|
1162
1514
|
if (!key && process.env[provider.keyEnv]) key = process.env[provider.keyEnv];
|
|
1163
1515
|
}
|
|
1164
|
-
if (
|
|
1516
|
+
if (isAnthropicRoute) {
|
|
1517
|
+
// The generated model config selects the native anthropic client;
|
|
1518
|
+
// the Anthropic SDK picks endpoint+key up from these env vars and
|
|
1519
|
+
// appends /v1/messages itself —— 所以这里给的是**站点根**,
|
|
1520
|
+
// 不再给 --api-base(那条路只讲 OpenAI Chat Completions)。
|
|
1521
|
+
if (key) env.ANTHROPIC_API_KEY = key;
|
|
1522
|
+
env.ANTHROPIC_BASE_URL = anthropicRouteBase;
|
|
1523
|
+
} else if (provider.api === "openai-completions" || provider.api === "openai-responses" || provider.api === "openai_responses") {
|
|
1165
1524
|
// openai-responses 与 openai-completions 一样走 --api-base + LIVEBENCH_API_KEY
|
|
1166
1525
|
// (Responses 客户端同样从 api_dict 读 base_url/api_key)。
|
|
1167
1526
|
args.push("--api-base", provider.baseURL);
|
|
1168
1527
|
if (key) env.LIVEBENCH_API_KEY = key;
|
|
1169
1528
|
} else if (provider.api === "anthropic-messages") {
|
|
1170
|
-
// The generated model config selects the native anthropic client;
|
|
1171
|
-
// the Anthropic SDK picks endpoint+key up from these env vars.
|
|
1172
1529
|
if (key) env.ANTHROPIC_API_KEY = key;
|
|
1173
|
-
env.ANTHROPIC_BASE_URL =
|
|
1530
|
+
env.ANTHROPIC_BASE_URL = anthropicRouteBase;
|
|
1174
1531
|
}
|
|
1175
1532
|
}
|
|
1176
1533
|
|
|
@@ -1222,11 +1579,29 @@ function apply(ctx) {
|
|
|
1222
1579
|
proc: child,
|
|
1223
1580
|
displayName,
|
|
1224
1581
|
body,
|
|
1582
|
+
// 服务端解析后的 bench 列表:自动补跑扫描要用(baseline 评测没有 body.benchNames)
|
|
1583
|
+
benchPaths,
|
|
1225
1584
|
autoRetries,
|
|
1226
1585
|
exitCode: null,
|
|
1227
1586
|
startedAt: new Date().toISOString(),
|
|
1228
1587
|
command: [layout.pythonExe, ...args].join(" "),
|
|
1229
|
-
log: [
|
|
1588
|
+
log: [
|
|
1589
|
+
`$ ${[...args].join(" ")}`,
|
|
1590
|
+
...(modelMaxTokens !== null && requestedMaxTokens > modelMaxTokens
|
|
1591
|
+
? [`[cap] 面板请求的 max-tokens ${requestedMaxTokens} 超过该模型在 settings.yaml 里声明的上限 ${modelMaxTokens},本次按 ${modelMaxTokens} 发送`]
|
|
1592
|
+
: []),
|
|
1593
|
+
...(hfCached
|
|
1594
|
+
? [`[hf] 题目数据集已在本地缓存(${[...hfCategories].join(", ")}),本次离线读取,跳过 huggingface.co 联网检查`]
|
|
1595
|
+
: hfMissing.length > 0
|
|
1596
|
+
? [`[hf] 以下分类的数据集还没有本地缓存:${hfMissing.join(", ")} —— 本次会联网向 huggingface.co 取;连不通时会先超时重试 5 轮再失败`]
|
|
1597
|
+
: []),
|
|
1598
|
+
// 路由与请求形状:出问题时这段是唯一能自证"面板到底发了什么"的地方
|
|
1599
|
+
...(isAnthropicRoute
|
|
1600
|
+
? [`[route] Anthropic 系模型 → anthropic-messages(POST ${anthropicRouteBase}/v1/messages);`
|
|
1601
|
+
+ `max_tokens=${maxTokens};thinking=${JSON.stringify(thinkingModeForEffort(effort) ?? "不写(交给上游默认)")}`]
|
|
1602
|
+
: [`[route] ${provider !== null ? provider.api ?? "未知协议" : "直连"} → ${args.includes("--api-base") ? `--api-base ${provider?.baseURL}` : "内置默认端点"};`
|
|
1603
|
+
+ `max_tokens=${maxTokens}(字段 ${provider?.maxTokensField ?? "默认"})`]),
|
|
1604
|
+
],
|
|
1230
1605
|
};
|
|
1231
1606
|
runs.set(record.id, record);
|
|
1232
1607
|
// keep history bounded: drop oldest finished runs beyond 20 entries
|
|
@@ -1245,21 +1620,26 @@ function apply(ctx) {
|
|
|
1245
1620
|
child.on("close", (code) => {
|
|
1246
1621
|
record.exitCode = code === null ? -1 : code;
|
|
1247
1622
|
record.log.push(`[exit ${record.exitCode}]`);
|
|
1248
|
-
//
|
|
1249
|
-
// 只重跑失败题(最多
|
|
1623
|
+
// 自动补跑:正常结束但仍有 $ERROR$ 答案时,用 --resume --retry-failures
|
|
1624
|
+
// 只重跑失败题(最多 3 轮)。统计改为异步流式(同步版曾阻塞事件循环,
|
|
1250
1625
|
// 且函数缺失会在 close 回调抛未捕获异常 → DSH 假死断连)。
|
|
1626
|
+
// 复用同一个 display-name:答案写回同一个文件,只补失败的题;
|
|
1627
|
+
// 换了名字就会整套重跑一遍,还在成绩表里多出一行。
|
|
1251
1628
|
if (record.exitCode === 0 && record.autoRetries < 3 && record.body) {
|
|
1252
1629
|
(async () => {
|
|
1253
1630
|
try {
|
|
1254
|
-
const errors = await countErrorAnswersAsync(layout.dataDir, record.displayName, record.
|
|
1631
|
+
const { errors, failedIds } = await countErrorAnswersAsync(layout.dataDir, record.displayName, record.benchPaths ?? []);
|
|
1255
1632
|
if (errors > 0) {
|
|
1256
|
-
record.log.push(`[自动补跑] 检测到 ${errors}
|
|
1633
|
+
record.log.push(`[自动补跑] 检测到 ${errors} 题 API 失败($ERROR$),`
|
|
1634
|
+
+ `第 ${record.autoRetries + 1}/3 轮:只重跑这几题,其余答案与判分保留`);
|
|
1257
1635
|
const retryBody = { ...record.body, resume: true, retryFailures: true };
|
|
1258
1636
|
setTimeout(() => {
|
|
1259
|
-
startRun(retryBody, record.autoRetries + 1).catch((e) => {
|
|
1637
|
+
startRun(retryBody, record.autoRetries + 1, record.displayName).catch((e) => {
|
|
1260
1638
|
appendLog(record, `[自动补跑启动失败] ${e.message}`);
|
|
1261
1639
|
});
|
|
1262
1640
|
}, [15000, 60000, 180000][Math.min(record.autoRetries, 2)]);
|
|
1641
|
+
} else {
|
|
1642
|
+
record.log.push(`[自动补跑] 检查通过:${(record.benchPaths ?? []).join(", ")} 没有 $ERROR$ 答案`);
|
|
1263
1643
|
}
|
|
1264
1644
|
} catch (e) {
|
|
1265
1645
|
record.log.push(`[自动补跑检查失败] ${e.message}`);
|
|
@@ -1322,12 +1702,17 @@ function apply(ctx) {
|
|
|
1322
1702
|
failedCount: t.failed.length,
|
|
1323
1703
|
})),
|
|
1324
1704
|
})),
|
|
1325
|
-
providers: providers.map(({ id, name: pname, models, api, baseURL }) => ({
|
|
1705
|
+
providers: providers.map(({ id, name: pname, models, api, baseURL, streamIdleTimeoutMs, maxTokensField }) => ({
|
|
1326
1706
|
id,
|
|
1327
1707
|
name: pname,
|
|
1328
1708
|
routable: ["openai-completions", "openai-responses", "openai_responses", "anthropic-messages"].includes(api)
|
|
1329
1709
|
&& typeof baseURL === "string" && baseURL.length > 0,
|
|
1330
1710
|
baseURL: baseURL ?? null,
|
|
1711
|
+
// harness 给这个渠道设的流空闲阈值(毫秒);面板会透给 LiveBench 的静默看门狗
|
|
1712
|
+
streamIdleTimeoutMs: streamIdleTimeoutMs ?? null,
|
|
1713
|
+
// 会话窗口(pi-ai)发输出上限用的字段名;面板让 LiveBench 也用同一个
|
|
1714
|
+
maxTokensField: maxTokensField ?? null,
|
|
1715
|
+
// models 里带 maxTokens(该模型单次输出上限),面板用它夹 --max-tokens
|
|
1331
1716
|
models,
|
|
1332
1717
|
})),
|
|
1333
1718
|
});
|
|
@@ -1539,4 +1924,4 @@ function apply(ctx) {
|
|
|
1539
1924
|
}), `${name}: delete route`);
|
|
1540
1925
|
}
|
|
1541
1926
|
|
|
1542
|
-
export { name, inject, apply, writeGeneratedModelConfig, readProviders, validateBenchName, clampedRangeLength, releaseQuestionCount, isEmptyAnswerLine, missingBaselineIndexes, baselineIdsFor };
|
|
1927
|
+
export { name, inject, apply, writeGeneratedModelConfig, readProviders, providerCapsFromSettings, maxTokensFieldFor, thinkingBudgetForEffort, thinkingModeForEffort, isAnthropicModel, anthropicBaseUrl, validateBenchName, clampedRangeLength, releaseQuestionCount, isEmptyAnswerLine, missingBaselineIndexes, baselineIdsFor, reduceAnswerLines, scoreJudgments, countErrorAnswersAsync, livebenchCategories, hfCategoriesFor };
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "dsh-livebench-panel",
|
|
3
|
-
"version": "0.2.
|
|
3
|
+
"version": "0.2.30",
|
|
4
4
|
"description": "DSH web plugin: a LiveBench tab in the Trajectory view (right of 对话/轨迹). Run LiveBench evaluations against every model configured in the DeepSeek Harness — pick provider/model, category, task, release and question range from dropdowns, watch progress, and read scores in place. Ships a Baseline probe set built from the questions a reference model cannot solve.",
|
|
5
5
|
"license": "MIT",
|
|
6
6
|
"author": "cszr (Vithrive)",
|