oh-my-knowledge 0.23.0 → 0.25.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +146 -84
- package/README.zh.md +144 -83
- package/dist/src/analysis/report-diagnostics.d.ts +3 -2
- package/dist/src/analysis/report-diagnostics.d.ts.map +1 -1
- package/dist/src/analysis/report-diagnostics.js +107 -80
- package/dist/src/analysis/report-diagnostics.js.map +1 -1
- package/dist/src/analysis/sample-diagnostics.d.ts +4 -1
- package/dist/src/analysis/sample-diagnostics.d.ts.map +1 -1
- package/dist/src/analysis/sample-diagnostics.js +123 -29
- package/dist/src/analysis/sample-diagnostics.js.map +1 -1
- package/dist/src/analysis/saturation.d.ts +2 -2
- package/dist/src/analysis/saturation.js +2 -2
- package/dist/src/authoring/evolver.d.ts +12 -4
- package/dist/src/authoring/evolver.d.ts.map +1 -1
- package/dist/src/authoring/evolver.js +35 -10
- package/dist/src/authoring/evolver.js.map +1 -1
- package/dist/src/cli/i18n-dict.d.ts +1 -1
- package/dist/src/cli/i18n-dict.d.ts.map +1 -1
- package/dist/src/cli/i18n-dict.js +241 -64
- package/dist/src/cli/i18n-dict.js.map +1 -1
- package/dist/src/cli/index.js +267 -152
- package/dist/src/cli/index.js.map +1 -1
- package/dist/src/cli/parse-run-config.d.ts +32 -9
- package/dist/src/cli/parse-run-config.d.ts.map +1 -1
- package/dist/src/cli/parse-run-config.js +80 -25
- package/dist/src/cli/parse-run-config.js.map +1 -1
- package/dist/src/cli/parse-strict.d.ts +20 -0
- package/dist/src/cli/parse-strict.d.ts.map +1 -0
- package/dist/src/cli/parse-strict.js +25 -0
- package/dist/src/cli/parse-strict.js.map +1 -0
- package/dist/src/cli/progress.d.ts +1 -0
- package/dist/src/cli/progress.d.ts.map +1 -1
- package/dist/src/cli/progress.js +8 -1
- package/dist/src/cli/progress.js.map +1 -1
- package/dist/src/doctor/index.d.ts +19 -0
- package/dist/src/doctor/index.d.ts.map +1 -0
- package/dist/src/doctor/index.js +182 -0
- package/dist/src/doctor/index.js.map +1 -0
- package/dist/src/doctor/preflight.d.ts +32 -0
- package/dist/src/doctor/preflight.d.ts.map +1 -0
- package/dist/src/doctor/preflight.js +32 -0
- package/dist/src/doctor/preflight.js.map +1 -0
- package/dist/src/doctor/renderer.d.ts +13 -0
- package/dist/src/doctor/renderer.d.ts.map +1 -0
- package/dist/src/doctor/renderer.js +69 -0
- package/dist/src/doctor/renderer.js.map +1 -0
- package/dist/src/doctor/rules.d.ts +30 -0
- package/dist/src/doctor/rules.d.ts.map +1 -0
- package/dist/src/doctor/rules.js +216 -0
- package/dist/src/doctor/rules.js.map +1 -0
- package/dist/src/eval-core/bootstrap.d.ts +1 -1
- package/dist/src/eval-core/bootstrap.js +1 -1
- package/dist/src/eval-core/cache.d.ts +7 -5
- package/dist/src/eval-core/cache.d.ts.map +1 -1
- package/dist/src/eval-core/cache.js +11 -7
- package/dist/src/eval-core/cache.js.map +1 -1
- package/dist/src/eval-core/comparability.d.ts +11 -0
- package/dist/src/eval-core/comparability.d.ts.map +1 -0
- package/dist/src/eval-core/comparability.js +296 -0
- package/dist/src/eval-core/comparability.js.map +1 -0
- package/dist/src/eval-core/dependency-checker.js +1 -1
- package/dist/src/eval-core/dependency-checker.js.map +1 -1
- package/dist/src/eval-core/evaluation-execution.d.ts +23 -7
- package/dist/src/eval-core/evaluation-execution.d.ts.map +1 -1
- package/dist/src/eval-core/evaluation-execution.js +53 -21
- package/dist/src/eval-core/evaluation-execution.js.map +1 -1
- package/dist/src/eval-core/evaluation-job.d.ts +3 -5
- package/dist/src/eval-core/evaluation-job.d.ts.map +1 -1
- package/dist/src/eval-core/evaluation-job.js +2 -4
- package/dist/src/eval-core/evaluation-job.js.map +1 -1
- package/dist/src/eval-core/evaluation-reporting.d.ts +3 -1
- package/dist/src/eval-core/evaluation-reporting.d.ts.map +1 -1
- package/dist/src/eval-core/evaluation-reporting.js +67 -9
- package/dist/src/eval-core/evaluation-reporting.js.map +1 -1
- package/dist/src/eval-core/execution-strategy.js +2 -2
- package/dist/src/eval-core/execution-strategy.js.map +1 -1
- package/dist/src/eval-core/schema.d.ts.map +1 -1
- package/dist/src/eval-core/schema.js +20 -2
- package/dist/src/eval-core/schema.js.map +1 -1
- package/dist/src/eval-workflows/batch-evaluation-workflow.d.ts +113 -0
- package/dist/src/eval-workflows/batch-evaluation-workflow.d.ts.map +1 -0
- package/dist/src/eval-workflows/batch-evaluation-workflow.js +217 -0
- package/dist/src/eval-workflows/batch-evaluation-workflow.js.map +1 -0
- package/dist/src/eval-workflows/evaluation-pipeline.d.ts +6 -4
- package/dist/src/eval-workflows/evaluation-pipeline.d.ts.map +1 -1
- package/dist/src/eval-workflows/evaluation-pipeline.js +44 -38
- package/dist/src/eval-workflows/evaluation-pipeline.js.map +1 -1
- package/dist/src/eval-workflows/evaluation-preparation.d.ts +4 -17
- package/dist/src/eval-workflows/evaluation-preparation.d.ts.map +1 -1
- package/dist/src/eval-workflows/evaluation-preparation.js +3 -18
- package/dist/src/eval-workflows/evaluation-preparation.js.map +1 -1
- package/dist/src/eval-workflows/run-evaluation.d.ts +27 -23
- package/dist/src/eval-workflows/run-evaluation.d.ts.map +1 -1
- package/dist/src/eval-workflows/run-evaluation.js +137 -20
- package/dist/src/eval-workflows/run-evaluation.js.map +1 -1
- package/dist/src/executors/claude-cli.d.ts.map +1 -1
- package/dist/src/executors/claude-cli.js +11 -6
- package/dist/src/executors/claude-cli.js.map +1 -1
- package/dist/src/executors/codex-cli-trace.d.ts +10 -0
- package/dist/src/executors/codex-cli-trace.d.ts.map +1 -0
- package/dist/src/executors/codex-cli-trace.js +123 -0
- package/dist/src/executors/codex-cli-trace.js.map +1 -0
- package/dist/src/executors/codex-cli.d.ts +18 -0
- package/dist/src/executors/codex-cli.d.ts.map +1 -0
- package/dist/src/executors/codex-cli.js +254 -0
- package/dist/src/executors/codex-cli.js.map +1 -0
- package/dist/src/executors/codex-sdk.d.ts +18 -0
- package/dist/src/executors/codex-sdk.d.ts.map +1 -0
- package/dist/src/executors/codex-sdk.js +214 -0
- package/dist/src/executors/codex-sdk.js.map +1 -0
- package/dist/src/executors/gemini.d.ts.map +1 -1
- package/dist/src/executors/gemini.js +28 -24
- package/dist/src/executors/gemini.js.map +1 -1
- package/dist/src/executors/index.d.ts.map +1 -1
- package/dist/src/executors/index.js +7 -2
- package/dist/src/executors/index.js.map +1 -1
- package/dist/src/executors/runtime-fingerprint.d.ts +7 -0
- package/dist/src/executors/runtime-fingerprint.d.ts.map +1 -0
- package/dist/src/executors/runtime-fingerprint.js +277 -0
- package/dist/src/executors/runtime-fingerprint.js.map +1 -0
- package/dist/src/executors/script.d.ts.map +1 -1
- package/dist/src/executors/script.js +47 -55
- package/dist/src/executors/script.js.map +1 -1
- package/dist/src/executors/shared.d.ts +78 -1
- package/dist/src/executors/shared.d.ts.map +1 -1
- package/dist/src/executors/shared.js +203 -1
- package/dist/src/executors/shared.js.map +1 -1
- package/dist/src/grading/assertions.d.ts.map +1 -1
- package/dist/src/grading/assertions.js +22 -7
- package/dist/src/grading/assertions.js.map +1 -1
- package/dist/src/grading/gold-cli.js +4 -4
- package/dist/src/grading/gold-cli.js.map +1 -1
- package/dist/src/grading/human-gold.d.ts +1 -1
- package/dist/src/grading/human-gold.js +1 -1
- package/dist/src/grading/index.d.ts +20 -15
- package/dist/src/grading/index.d.ts.map +1 -1
- package/dist/src/grading/index.js +40 -16
- package/dist/src/grading/index.js.map +1 -1
- package/dist/src/grading/judge.d.ts +1 -1
- package/dist/src/grading/judge.d.ts.map +1 -1
- package/dist/src/grading/judge.js +76 -7
- package/dist/src/grading/judge.js.map +1 -1
- package/dist/src/inputs/eval-config.js +65 -7
- package/dist/src/inputs/eval-config.js.map +1 -1
- package/dist/src/inputs/skill-loader.d.ts +1 -1
- package/dist/src/inputs/skill-loader.d.ts.map +1 -1
- package/dist/src/inputs/skill-loader.js +1 -1
- package/dist/src/inputs/skill-loader.js.map +1 -1
- package/dist/src/renderer/html-renderer.d.ts +5 -4
- package/dist/src/renderer/html-renderer.d.ts.map +1 -1
- package/dist/src/renderer/html-renderer.js +240 -96
- package/dist/src/renderer/html-renderer.js.map +1 -1
- package/dist/src/renderer/layout.d.ts +2 -1
- package/dist/src/renderer/layout.d.ts.map +1 -1
- package/dist/src/renderer/layout.js +30 -42
- package/dist/src/renderer/layout.js.map +1 -1
- package/dist/src/renderer/summary.d.ts +3 -3
- package/dist/src/renderer/summary.d.ts.map +1 -1
- package/dist/src/renderer/summary.js +233 -43
- package/dist/src/renderer/summary.js.map +1 -1
- package/dist/src/renderer/trends.d.ts.map +1 -1
- package/dist/src/renderer/trends.js +5 -3
- package/dist/src/renderer/trends.js.map +1 -1
- package/dist/src/server/report-server.js +4 -4
- package/dist/src/server/report-server.js.map +1 -1
- package/dist/src/server/report-store.d.ts +7 -5
- package/dist/src/server/report-store.d.ts.map +1 -1
- package/dist/src/server/report-store.js +39 -11
- package/dist/src/server/report-store.js.map +1 -1
- package/dist/src/types/doctor.d.ts +95 -0
- package/dist/src/types/doctor.d.ts.map +1 -0
- package/dist/src/types/doctor.js +2 -0
- package/dist/src/types/doctor.js.map +1 -0
- package/dist/src/types/eval.d.ts +40 -19
- package/dist/src/types/eval.d.ts.map +1 -1
- package/dist/src/types/executor.d.ts +38 -0
- package/dist/src/types/executor.d.ts.map +1 -1
- package/dist/src/types/index.d.ts +1 -0
- package/dist/src/types/index.d.ts.map +1 -1
- package/dist/src/types/index.js +1 -0
- package/dist/src/types/index.js.map +1 -1
- package/dist/src/types/judge.d.ts +21 -0
- package/dist/src/types/judge.d.ts.map +1 -1
- package/dist/src/types/report.d.ts +100 -35
- package/dist/src/types/report.d.ts.map +1 -1
- package/dist/src/types/storage.d.ts +7 -7
- package/dist/src/types/storage.d.ts.map +1 -1
- package/package.json +6 -5
- package/dist/src/eval-workflows/each-evaluation-workflow.d.ts +0 -153
- package/dist/src/eval-workflows/each-evaluation-workflow.d.ts.map +0 -1
- package/dist/src/eval-workflows/each-evaluation-workflow.js +0 -178
- package/dist/src/eval-workflows/each-evaluation-workflow.js.map +0 -1
- package/dist/src/executors/openai-cli.d.ts +0 -3
- package/dist/src/executors/openai-cli.d.ts.map +0 -1
- package/dist/src/executors/openai-cli.js +0 -60
- package/dist/src/executors/openai-cli.js.map +0 -1
package/dist/src/cli/index.js
CHANGED
|
@@ -1,5 +1,4 @@
|
|
|
1
1
|
#!/usr/bin/env node
|
|
2
|
-
import { parseArgs } from 'node:util';
|
|
3
2
|
import { resolve } from 'node:path';
|
|
4
3
|
import { join } from 'node:path';
|
|
5
4
|
import { existsSync } from 'node:fs';
|
|
@@ -7,6 +6,20 @@ import { tCli, getCliLang, parseLangFromArgv, langFromArgv } from './i18n.js';
|
|
|
7
6
|
import { parseRunConfig, DEFAULT_REPORTS_DIR, COMMON_OPTIONS, } from './parse-run-config.js';
|
|
8
7
|
import { makeOnProgress } from './progress.js';
|
|
9
8
|
import { checkUpdate } from './update-check.js';
|
|
9
|
+
import { parseArgsStrictOrExit } from './parse-strict.js';
|
|
10
|
+
function requireEvaluationReport(report, id, lang) {
|
|
11
|
+
if (!report) {
|
|
12
|
+
console.error(tCli('cli.common.report_not_found', lang, { id }));
|
|
13
|
+
process.exit(1);
|
|
14
|
+
}
|
|
15
|
+
if (report.kind === 'batch-evaluation') {
|
|
16
|
+
console.error(lang === 'zh'
|
|
17
|
+
? `报告 ${id} 是 BatchEvaluationReport。该命令需要单次 EvaluationReport;请使用其中的 child reportId。`
|
|
18
|
+
: `Report ${id} is a BatchEvaluationReport. This command requires an EvaluationReport; use a child reportId from the batch.`);
|
|
19
|
+
process.exit(1);
|
|
20
|
+
}
|
|
21
|
+
return report;
|
|
22
|
+
}
|
|
10
23
|
// ---------------------------------------------------------------------------
|
|
11
24
|
// Main
|
|
12
25
|
// ---------------------------------------------------------------------------
|
|
@@ -23,6 +36,11 @@ async function main() {
|
|
|
23
36
|
await handleAnalyze(args);
|
|
24
37
|
return;
|
|
25
38
|
}
|
|
39
|
+
if (domain === 'doctor') {
|
|
40
|
+
const args = command ? [command, ...rest] : [];
|
|
41
|
+
await handleDoctor(args);
|
|
42
|
+
return;
|
|
43
|
+
}
|
|
26
44
|
if (domain !== 'bench') {
|
|
27
45
|
console.error(tCli('cli.common.unknown_domain', lang, { domain }));
|
|
28
46
|
process.exit(1);
|
|
@@ -88,64 +106,46 @@ async function main() {
|
|
|
88
106
|
// ---------------------------------------------------------------------------
|
|
89
107
|
async function handleRun(argv) {
|
|
90
108
|
const lang = langFromArgv(argv);
|
|
91
|
-
|
|
109
|
+
if (argv.includes('--help') || argv.includes('-h')) {
|
|
110
|
+
console.log(tCli('cli.help.main', lang).trim());
|
|
111
|
+
process.exit(0);
|
|
112
|
+
}
|
|
113
|
+
// 注: 这里**不**给 parseArgs default 值, 否则 values.xxx 永远不为 undefined,
|
|
114
|
+
// CLI > eval.yaml > hardcoded-default 三级 fallback 区分不开 ("用户没传" vs "用户传了等于 default 值")。
|
|
115
|
+
// hardcoded default 在下面处理 undefined 时显式给。
|
|
116
|
+
const { values, config, evalConfig } = parseRunConfig(argv, {
|
|
92
117
|
blind: { type: 'boolean' },
|
|
93
|
-
repeat: { type: 'string'
|
|
94
|
-
'judge-repeat': { type: 'string'
|
|
95
|
-
|
|
96
|
-
bootstrap: { type: '
|
|
97
|
-
'bootstrap-samples': { type: 'string', default: '1000' },
|
|
118
|
+
repeat: { type: 'string' },
|
|
119
|
+
'judge-repeat': { type: 'string' },
|
|
120
|
+
bootstrap: { type: 'boolean' },
|
|
121
|
+
'bootstrap-samples': { type: 'string' },
|
|
98
122
|
'gold-dir': { type: 'string' },
|
|
99
|
-
'no-debias-length': { type: 'boolean'
|
|
123
|
+
'no-debias-length': { type: 'boolean' },
|
|
100
124
|
'budget-usd': { type: 'string' },
|
|
101
125
|
'budget-per-sample-usd': { type: 'string' },
|
|
102
126
|
'budget-per-sample-ms': { type: 'string' },
|
|
103
127
|
});
|
|
104
|
-
const { runEvaluation, runMultiple,
|
|
128
|
+
const { runEvaluation, runMultiple, runBatchEvaluation } = await import('../eval-workflows/run-evaluation.js');
|
|
105
129
|
if (values.blind !== undefined) {
|
|
106
130
|
config.blind = values.blind;
|
|
107
131
|
}
|
|
108
132
|
config.onProgress = makeOnProgress(lang);
|
|
109
|
-
// --repeat
|
|
110
|
-
// 提前到 --each 分支之前, 保证 each 模式也能读到 repeat (曾经 bug: --each 吞 --repeat)。
|
|
133
|
+
// --repeat: CLI > eval.yaml > 1. 非 ≥1 整数时提示并钳到 1。
|
|
111
134
|
const repeatRaw = values.repeat;
|
|
112
|
-
const parsedRepeat = repeatRaw !== undefined ? Number(repeatRaw) : 1;
|
|
135
|
+
const parsedRepeat = repeatRaw !== undefined ? Number(repeatRaw) : (evalConfig?.repeat ?? 1);
|
|
113
136
|
if (repeatRaw !== undefined && (!Number.isFinite(parsedRepeat) || parsedRepeat < 1)) {
|
|
114
137
|
process.stderr.write(tCli('cli.run.invalid_repeat', lang, { value: repeatRaw }));
|
|
115
138
|
}
|
|
116
139
|
const repeatCount = Math.max(1, Math.floor(parsedRepeat) || 1);
|
|
117
|
-
// --judge-repeat
|
|
140
|
+
// --judge-repeat: CLI > eval.yaml > 1.
|
|
118
141
|
const judgeRepeatRaw = values['judge-repeat'];
|
|
119
|
-
const parsedJudgeRepeat = judgeRepeatRaw !== undefined ? Number(judgeRepeatRaw) : 1;
|
|
142
|
+
const parsedJudgeRepeat = judgeRepeatRaw !== undefined ? Number(judgeRepeatRaw) : (evalConfig?.judgeRepeat ?? 1);
|
|
120
143
|
if (judgeRepeatRaw !== undefined && (!Number.isFinite(parsedJudgeRepeat) || parsedJudgeRepeat < 1)) {
|
|
121
144
|
process.stderr.write(tCli('cli.run.invalid_judge_repeat', lang, { value: judgeRepeatRaw }));
|
|
122
145
|
}
|
|
123
146
|
const judgeRepeatCount = Math.max(1, Math.floor(parsedJudgeRepeat) || 1);
|
|
124
147
|
if (judgeRepeatCount > 1)
|
|
125
148
|
config.judgeRepeat = judgeRepeatCount;
|
|
126
|
-
// --judge-models executor:model,executor:model,... -> JudgeConfig[]
|
|
127
|
-
// 至少 2 个才进 ensemble 模式, 1 个等同于 --judge-model
|
|
128
|
-
const judgeModelsRaw = values['judge-models'];
|
|
129
|
-
if (judgeModelsRaw) {
|
|
130
|
-
const parts = judgeModelsRaw.split(',').map((s) => s.trim()).filter(Boolean);
|
|
131
|
-
const judges = parts.map((p) => {
|
|
132
|
-
const [executor, ...modelParts] = p.split(':');
|
|
133
|
-
const model = modelParts.join(':');
|
|
134
|
-
if (!executor || !model) {
|
|
135
|
-
throw new Error(tCli('cli.run.invalid_judge_models_format', lang, { part: p }));
|
|
136
|
-
}
|
|
137
|
-
return { executor, model };
|
|
138
|
-
});
|
|
139
|
-
if (judges.length >= 2) {
|
|
140
|
-
config.judgeModels = judges;
|
|
141
|
-
}
|
|
142
|
-
else if (judges.length === 1) {
|
|
143
|
-
// 单 judge 不走 ensemble, 但允许这样写, 等同于 --judge-model + --executor
|
|
144
|
-
process.stderr.write(tCli('cli.run.judge_models_single_warning', lang, {
|
|
145
|
-
executor: judges[0].executor, model: judges[0].model,
|
|
146
|
-
}));
|
|
147
|
-
}
|
|
148
|
-
}
|
|
149
149
|
// --budget-usd / --budget-per-sample-usd / --budget-per-sample-ms:
|
|
150
150
|
// hard budget caps. CLI flags override config-file values. When the
|
|
151
151
|
// total-USD cap is exceeded mid-run, remaining tasks are skipped and a
|
|
@@ -160,18 +160,22 @@ async function handleRun(argv) {
|
|
|
160
160
|
...(budgetPerSampleMs !== undefined && Number.isFinite(budgetPerSampleMs) && budgetPerSampleMs >= 0 ? { perSampleMs: budgetPerSampleMs } : {}),
|
|
161
161
|
};
|
|
162
162
|
}
|
|
163
|
-
// --no-debias-length: opt out of
|
|
164
|
-
// Default
|
|
165
|
-
//
|
|
166
|
-
|
|
163
|
+
// --no-debias-length / eval.yaml `lengthDebias: false`: opt out of length-controlled prompt。
|
|
164
|
+
// Default debias-on (judge prompt v3-cot-length); flip off only to reproduce historical reports。
|
|
165
|
+
// CLI 显式 --no-debias-length > eval.yaml lengthDebias > 默认 true。
|
|
166
|
+
const lengthDebiasOff = values['no-debias-length'] === true
|
|
167
|
+
|| (values['no-debias-length'] === undefined && evalConfig?.lengthDebias === false);
|
|
168
|
+
if (lengthDebiasOff) {
|
|
167
169
|
config.lengthDebias = false;
|
|
168
170
|
process.stderr.write(tCli('cli.run.no_debias_length_active', lang));
|
|
169
171
|
}
|
|
170
|
-
// --bootstrap / --bootstrap-samples
|
|
171
|
-
|
|
172
|
+
// --bootstrap / --bootstrap-samples: CLI > eval.yaml > default(off / 1000)。
|
|
173
|
+
const bootstrapEnabled = values.bootstrap === true
|
|
174
|
+
|| (values.bootstrap === undefined && evalConfig?.bootstrap === true);
|
|
175
|
+
if (bootstrapEnabled) {
|
|
172
176
|
config.bootstrap = true;
|
|
173
177
|
const bsRaw = values['bootstrap-samples'];
|
|
174
|
-
const parsedBs = bsRaw !== undefined ? Number(bsRaw) : 1000;
|
|
178
|
+
const parsedBs = bsRaw !== undefined ? Number(bsRaw) : (evalConfig?.bootstrapSamples ?? 1000);
|
|
175
179
|
if (bsRaw !== undefined && (!Number.isFinite(parsedBs) || parsedBs < 100)) {
|
|
176
180
|
process.stderr.write(tCli('cli.run.invalid_bootstrap_samples', lang, { value: bsRaw }));
|
|
177
181
|
}
|
|
@@ -181,10 +185,15 @@ async function handleRun(argv) {
|
|
|
181
185
|
}
|
|
182
186
|
config.bootstrapSamples = bsCount;
|
|
183
187
|
}
|
|
188
|
+
// 注入 lang 让 evaluation pipeline 能渲染 doctor 报告(失败时)。
|
|
189
|
+
config.lang = lang;
|
|
190
|
+
if (values['skip-connectivity']) {
|
|
191
|
+
process.stderr.write(tCli('cli.run.skip_connectivity_warning', lang) + '\n');
|
|
192
|
+
}
|
|
184
193
|
try {
|
|
185
|
-
// --
|
|
186
|
-
if (values.
|
|
187
|
-
const { report, filePath } = await
|
|
194
|
+
// --batch mode: evaluate each skill independently
|
|
195
|
+
if (values.batch) {
|
|
196
|
+
const { report, filePath } = await runBatchEvaluation({
|
|
188
197
|
...config,
|
|
189
198
|
repeat: repeatCount,
|
|
190
199
|
onSkillProgress({ phase, skill, current, total }) {
|
|
@@ -237,8 +246,8 @@ async function handleRun(argv) {
|
|
|
237
246
|
report = result.report;
|
|
238
247
|
filePath = result.filePath;
|
|
239
248
|
}
|
|
240
|
-
// --gold-dir: compute α/κ/Pearson against gold annotations and re-persist.
|
|
241
|
-
const goldDir = values['gold-dir'];
|
|
249
|
+
// --gold-dir / eval.yaml goldDir: compute α/κ/Pearson against gold annotations and re-persist.
|
|
250
|
+
const goldDir = values['gold-dir'] ?? evalConfig?.goldDir;
|
|
242
251
|
if (goldDir && filePath) {
|
|
243
252
|
const { attachGoldAgreementToReport, formatGoldCompare } = await import('../grading/gold-cli.js');
|
|
244
253
|
const out = attachGoldAgreementToReport({
|
|
@@ -299,7 +308,11 @@ async function handleRun(argv) {
|
|
|
299
308
|
// ---------------------------------------------------------------------------
|
|
300
309
|
async function handleReport(argv) {
|
|
301
310
|
const lang = langFromArgv(argv);
|
|
302
|
-
|
|
311
|
+
if (argv.includes('--help') || argv.includes('-h')) {
|
|
312
|
+
console.log(tCli('cli.help.main', lang).trim());
|
|
313
|
+
process.exit(0);
|
|
314
|
+
}
|
|
315
|
+
const { values } = parseArgsStrictOrExit({
|
|
303
316
|
args: argv,
|
|
304
317
|
options: {
|
|
305
318
|
...COMMON_OPTIONS,
|
|
@@ -308,7 +321,6 @@ async function handleReport(argv) {
|
|
|
308
321
|
export: { type: 'string' },
|
|
309
322
|
dev: { type: 'boolean', default: false },
|
|
310
323
|
},
|
|
311
|
-
strict: false,
|
|
312
324
|
});
|
|
313
325
|
// Dev mode: restart server on file changes via node --watch
|
|
314
326
|
if (values.dev && !process.env.__OMK_DEV_CHILD) {
|
|
@@ -330,7 +342,7 @@ async function handleReport(argv) {
|
|
|
330
342
|
}
|
|
331
343
|
if (values.export) {
|
|
332
344
|
const { createFileStore } = await import('../server/report-store.js');
|
|
333
|
-
const {
|
|
345
|
+
const { renderReportDocumentDetail } = await import('../renderer/html-renderer.js');
|
|
334
346
|
const { writeFileSync } = await import('node:fs');
|
|
335
347
|
const store = createFileStore(resolve(values['reports-dir']));
|
|
336
348
|
const report = await store.get(values.export);
|
|
@@ -338,7 +350,7 @@ async function handleReport(argv) {
|
|
|
338
350
|
console.error(tCli('cli.common.report_not_found', lang, { id: values.export }));
|
|
339
351
|
process.exit(1);
|
|
340
352
|
}
|
|
341
|
-
const html =
|
|
353
|
+
const html = renderReportDocumentDetail(report);
|
|
342
354
|
const outPath = resolve(`${values.export}.html`);
|
|
343
355
|
writeFileSync(outPath, html);
|
|
344
356
|
console.log(`Exported to: ${outPath}`);
|
|
@@ -429,9 +441,76 @@ function parseLastWindow(spec) {
|
|
|
429
441
|
const ms = unit === 'd' ? n * 86400_000 : unit === 'h' ? n * 3600_000 : n * 60_000;
|
|
430
442
|
return new Date(Date.now() - ms).toISOString();
|
|
431
443
|
}
|
|
444
|
+
async function handleDoctor(argv) {
|
|
445
|
+
const lang = langFromArgv(argv);
|
|
446
|
+
if (argv.includes('--help') || argv.includes('-h')) {
|
|
447
|
+
console.log(tCli('cli.help.doctor_usage', lang));
|
|
448
|
+
process.exit(0);
|
|
449
|
+
}
|
|
450
|
+
const { values, positionals } = parseArgsStrictOrExit({
|
|
451
|
+
args: argv,
|
|
452
|
+
allowPositionals: true,
|
|
453
|
+
options: {
|
|
454
|
+
...COMMON_OPTIONS,
|
|
455
|
+
json: { type: 'boolean', default: false },
|
|
456
|
+
gate: { type: 'boolean', default: false },
|
|
457
|
+
executor: { type: 'string' },
|
|
458
|
+
model: { type: 'string' },
|
|
459
|
+
timeout: { type: 'string' },
|
|
460
|
+
},
|
|
461
|
+
});
|
|
462
|
+
const target = positionals[0] ?? null;
|
|
463
|
+
const executorName = values.executor ?? 'claude';
|
|
464
|
+
const model = values.model ?? 'sonnet';
|
|
465
|
+
const timeoutRaw = values.timeout;
|
|
466
|
+
const timeoutSec = timeoutRaw != null ? Number(timeoutRaw) : 8;
|
|
467
|
+
const timeoutMs = Math.max(1000, Math.floor((Number.isFinite(timeoutSec) ? timeoutSec : 8) * 1000));
|
|
468
|
+
const cwd = process.cwd();
|
|
469
|
+
const { runDoctor } = await import('../doctor/index.js');
|
|
470
|
+
const { renderDoctorReportText, renderDoctorReportJson } = await import('../doctor/renderer.js');
|
|
471
|
+
let report;
|
|
472
|
+
try {
|
|
473
|
+
report = await runDoctor({
|
|
474
|
+
target,
|
|
475
|
+
cwd,
|
|
476
|
+
executorName,
|
|
477
|
+
model,
|
|
478
|
+
timeoutMs,
|
|
479
|
+
lang,
|
|
480
|
+
});
|
|
481
|
+
}
|
|
482
|
+
catch (err) {
|
|
483
|
+
const msg = err instanceof Error ? err.message : String(err);
|
|
484
|
+
console.error(tCli('cli.doctor.no_skill_found', lang, { path: target ?? cwd }));
|
|
485
|
+
console.error(`(${msg})`);
|
|
486
|
+
process.exit(1);
|
|
487
|
+
}
|
|
488
|
+
if (report.skills.length === 0) {
|
|
489
|
+
console.error(tCli('cli.doctor.no_skill_found', lang, { path: target ?? cwd }));
|
|
490
|
+
process.exit(1);
|
|
491
|
+
}
|
|
492
|
+
const isJson = values.json;
|
|
493
|
+
const isGate = values.gate;
|
|
494
|
+
if (isJson) {
|
|
495
|
+
console.log(renderDoctorReportJson(report));
|
|
496
|
+
}
|
|
497
|
+
else if (isGate) {
|
|
498
|
+
// gate 模式: 静默 stdout, fail 时简短 stderr 摘要(供 CI 抓 exit code)
|
|
499
|
+
if (report.failed) {
|
|
500
|
+
const summary = lang === 'zh'
|
|
501
|
+
? `doctor failed: ${report.totals.fail} 个 skill 未通过 (${report.totals.warn} warn / ${report.totals.pass} pass)`
|
|
502
|
+
: `doctor failed: ${report.totals.fail} skills did not pass (${report.totals.warn} warn / ${report.totals.pass} pass)`;
|
|
503
|
+
console.error(summary);
|
|
504
|
+
}
|
|
505
|
+
}
|
|
506
|
+
else {
|
|
507
|
+
renderDoctorReportText(report, lang);
|
|
508
|
+
}
|
|
509
|
+
process.exit(report.failed ? 1 : 0);
|
|
510
|
+
}
|
|
432
511
|
async function handleAnalyze(argv) {
|
|
433
512
|
const lang = langFromArgv(argv);
|
|
434
|
-
const { values, positionals } =
|
|
513
|
+
const { values: rawValues, positionals } = parseArgsStrictOrExit({
|
|
435
514
|
args: argv,
|
|
436
515
|
allowPositionals: true,
|
|
437
516
|
options: {
|
|
@@ -444,6 +523,8 @@ async function handleAnalyze(argv) {
|
|
|
444
523
|
'output-dir': { type: 'string' },
|
|
445
524
|
},
|
|
446
525
|
});
|
|
526
|
+
// 该 handler options 全是 string-typed (无 boolean), 收紧 cast 让 caller 直接 use values.xxx 当 string 用。
|
|
527
|
+
const values = rawValues;
|
|
447
528
|
const dir = positionals[0];
|
|
448
529
|
if (!dir) {
|
|
449
530
|
console.error(tCli('cli.help.analyze_usage', lang));
|
|
@@ -499,7 +580,14 @@ async function handleAnalyze(argv) {
|
|
|
499
580
|
}
|
|
500
581
|
async function handleInit(argv) {
|
|
501
582
|
const lang = langFromArgv(argv);
|
|
502
|
-
|
|
583
|
+
// 走 helper 让未知 option fail-fast (e.g. `omk bench init --bogus`),
|
|
584
|
+
// 否则 argv[0] 直接当目录名, --bogus / --lang 都会被当成 dir 写文件。
|
|
585
|
+
const { positionals } = parseArgsStrictOrExit({
|
|
586
|
+
args: argv,
|
|
587
|
+
allowPositionals: true,
|
|
588
|
+
options: { ...COMMON_OPTIONS },
|
|
589
|
+
});
|
|
590
|
+
const targetDir = resolve(positionals[0] || '.');
|
|
503
591
|
const { writeFileSync, mkdirSync } = await import('node:fs');
|
|
504
592
|
mkdirSync(join(targetDir, 'skills'), { recursive: true });
|
|
505
593
|
writeFileSync(join(targetDir, 'eval-samples.json'), INIT_SAMPLES);
|
|
@@ -517,23 +605,26 @@ async function handleInit(argv) {
|
|
|
517
605
|
// ---------------------------------------------------------------------------
|
|
518
606
|
async function handleGenSamples(argv) {
|
|
519
607
|
const lang = langFromArgv(argv);
|
|
520
|
-
|
|
608
|
+
if (argv.includes('--help') || argv.includes('-h')) {
|
|
609
|
+
console.log(tCli('cli.help.main', lang).trim());
|
|
610
|
+
process.exit(0);
|
|
611
|
+
}
|
|
612
|
+
const { values } = parseArgsStrictOrExit({
|
|
521
613
|
args: argv,
|
|
522
614
|
options: {
|
|
523
615
|
...COMMON_OPTIONS,
|
|
524
|
-
|
|
616
|
+
batch: { type: 'boolean', default: false },
|
|
525
617
|
count: { type: 'string', default: '5' },
|
|
526
618
|
model: { type: 'string', default: 'sonnet' },
|
|
527
619
|
'skill-dir': { type: 'string', default: 'skills' },
|
|
528
620
|
},
|
|
529
|
-
strict: false,
|
|
530
621
|
allowPositionals: true,
|
|
531
622
|
});
|
|
532
623
|
const { generateSamples } = await import('../authoring/generator.js');
|
|
533
624
|
const { readFileSync, writeFileSync } = await import('node:fs');
|
|
534
625
|
const count = Math.max(1, Number(values.count) || 5);
|
|
535
626
|
const model = values.model;
|
|
536
|
-
if (values.
|
|
627
|
+
if (values.batch) {
|
|
537
628
|
// Batch mode: generate for all skills missing eval-samples
|
|
538
629
|
const skillDir = resolve(values['skill-dir']);
|
|
539
630
|
if (!existsSync(skillDir)) {
|
|
@@ -631,7 +722,7 @@ async function handleGenSamples(argv) {
|
|
|
631
722
|
// ---------------------------------------------------------------------------
|
|
632
723
|
async function handleEvolve(argv) {
|
|
633
724
|
const lang = langFromArgv(argv);
|
|
634
|
-
const { values } =
|
|
725
|
+
const { values, positionals } = parseArgsStrictOrExit({
|
|
635
726
|
args: argv,
|
|
636
727
|
options: {
|
|
637
728
|
...COMMON_OPTIONS,
|
|
@@ -639,17 +730,18 @@ async function handleEvolve(argv) {
|
|
|
639
730
|
target: { type: 'string' },
|
|
640
731
|
samples: { type: 'string', default: 'eval-samples.json' },
|
|
641
732
|
model: { type: 'string', default: 'sonnet' },
|
|
642
|
-
'judge-
|
|
733
|
+
'judge-models': { type: 'string', default: 'claude:haiku' },
|
|
643
734
|
'improve-model': { type: 'string', default: 'sonnet' },
|
|
644
735
|
concurrency: { type: 'string', default: '1' },
|
|
645
736
|
timeout: { type: 'string', default: '120' },
|
|
646
737
|
executor: { type: 'string', default: 'claude' },
|
|
647
|
-
'skip-
|
|
738
|
+
'skip-connectivity': { type: 'boolean', default: false },
|
|
648
739
|
},
|
|
649
|
-
strict: false,
|
|
650
740
|
allowPositionals: true,
|
|
651
741
|
});
|
|
652
|
-
|
|
742
|
+
// skill path 走 parseArgs 的 positionals (避免 raw argv.find 把 flag value
|
|
743
|
+
// 当成 path 误识别 — 例如 `evolve --judge-models openai-api:gpt-4o foo.md`)。
|
|
744
|
+
const skillPath = positionals[0];
|
|
653
745
|
if (!skillPath) {
|
|
654
746
|
console.error(tCli('cli.evolve.specify_skill_path', lang));
|
|
655
747
|
process.exit(1);
|
|
@@ -662,6 +754,12 @@ async function handleEvolve(argv) {
|
|
|
662
754
|
samplesFile = 'eval-samples.yml';
|
|
663
755
|
}
|
|
664
756
|
const { evolveSkill } = await import('../authoring/evolver.js');
|
|
757
|
+
const { parseJudgeModelsArgOrExit } = await import('./parse-run-config.js');
|
|
758
|
+
const evolveJudges = parseJudgeModelsArgOrExit(values['judge-models']);
|
|
759
|
+
if (evolveJudges.length > 1) {
|
|
760
|
+
console.error(tCli('cli.common.judge_models_single_only', lang, { cmd: 'evolve' }));
|
|
761
|
+
process.exit(2);
|
|
762
|
+
}
|
|
665
763
|
process.stderr.write(tCli('cli.evolve.section_header', lang, { path: skillPath }));
|
|
666
764
|
try {
|
|
667
765
|
const result = await evolveSkill({
|
|
@@ -670,17 +768,20 @@ async function handleEvolve(argv) {
|
|
|
670
768
|
rounds: Math.max(1, Number(values.rounds) || 5),
|
|
671
769
|
target: values.target ? Number(values.target) : null,
|
|
672
770
|
model: values.model,
|
|
673
|
-
|
|
771
|
+
judgeModels: evolveJudges,
|
|
674
772
|
improveModel: values['improve-model'],
|
|
675
773
|
executorName: values.executor,
|
|
676
774
|
concurrency: Math.max(1, Number(values.concurrency) || 1),
|
|
677
775
|
timeoutMs: Math.max(1, Number(values.timeout) || 120) * 1000,
|
|
678
|
-
|
|
776
|
+
skipConnectivity: values['skip-connectivity'],
|
|
679
777
|
onProgress: makeOnProgress(lang),
|
|
680
|
-
onRoundProgress({ round, totalRounds: _totalRounds, phase, score, delta, accepted, costUSD, error }) {
|
|
778
|
+
onRoundProgress({ round, totalRounds: _totalRounds, phase, score, delta, accepted, costUSD, costReported, error }) {
|
|
779
|
+
// costReported=false 时显示「—」而不是 $0.0000(executor 不报 cost,如 codex)。
|
|
780
|
+
// 缺位 / true 当 reported 走旧格式。
|
|
781
|
+
const fmtRoundCost = (c, r) => r ? `$${c.toFixed(4)}` : '—';
|
|
681
782
|
if (phase === 'baseline') {
|
|
682
783
|
process.stderr.write(tCli('cli.evolve.round_baseline', lang, {
|
|
683
|
-
score: score.toFixed(2), cost: costUSD
|
|
784
|
+
score: score.toFixed(2), cost: fmtRoundCost(costUSD, costReported !== false),
|
|
684
785
|
}));
|
|
685
786
|
}
|
|
686
787
|
else if (phase === 'error') {
|
|
@@ -692,7 +793,7 @@ async function handleEvolve(argv) {
|
|
|
692
793
|
const delta_ = delta >= 0 ? `+${delta.toFixed(2)}` : delta.toFixed(2);
|
|
693
794
|
const status = accepted ? '✓ ACCEPT' : '✗ REJECT';
|
|
694
795
|
process.stderr.write(tCli('cli.evolve.round_done', lang, {
|
|
695
|
-
round, score: score.toFixed(2), delta: delta_, status, cost: costUSD
|
|
796
|
+
round, score: score.toFixed(2), delta: delta_, status, cost: fmtRoundCost(costUSD, costReported !== false),
|
|
696
797
|
}));
|
|
697
798
|
}
|
|
698
799
|
},
|
|
@@ -700,9 +801,12 @@ async function handleEvolve(argv) {
|
|
|
700
801
|
const improvement = result.startScore > 0
|
|
701
802
|
? ((result.finalScore - result.startScore) / result.startScore * 100).toFixed(1)
|
|
702
803
|
: '0';
|
|
804
|
+
const totalCostStr = result.costReported === false
|
|
805
|
+
? '—' // 任一轮的 executor 不报 cost → totalCostUSD 是 lower-bound
|
|
806
|
+
: `$${result.totalCostUSD.toFixed(4)}`;
|
|
703
807
|
process.stderr.write(tCli('cli.evolve.summary', lang, {
|
|
704
808
|
start: result.startScore.toFixed(2), final: result.finalScore.toFixed(2),
|
|
705
|
-
percent: improvement, rounds: result.totalRounds, cost:
|
|
809
|
+
percent: improvement, rounds: result.totalRounds, cost: totalCostStr,
|
|
706
810
|
}));
|
|
707
811
|
process.stderr.write(tCli('cli.evolve.best_path', lang, {
|
|
708
812
|
best: result.bestSkillPath, target: resolve(skillPath),
|
|
@@ -737,12 +841,18 @@ async function handleGate(argv) {
|
|
|
737
841
|
});
|
|
738
842
|
const { runEvaluation } = await import('../eval-workflows/run-evaluation.js');
|
|
739
843
|
config.onProgress = makeOnProgress(lang);
|
|
844
|
+
// 注入 lang + skip-connectivity warning(若 flag set);doctor 由 evaluation 强制调, 无 skip 选项。
|
|
845
|
+
config.lang = lang;
|
|
846
|
+
if (values['skip-connectivity']) {
|
|
847
|
+
process.stderr.write(tCli('cli.run.skip_connectivity_warning', lang) + '\n');
|
|
848
|
+
}
|
|
740
849
|
try {
|
|
741
|
-
const { report } = (await runEvaluation(config));
|
|
742
|
-
if (
|
|
850
|
+
const { report: document } = (await runEvaluation(config));
|
|
851
|
+
if (document.dryRun) {
|
|
743
852
|
console.log('Gate dry-run: no scores to check');
|
|
744
853
|
process.exit(0);
|
|
745
854
|
}
|
|
855
|
+
const report = requireEvaluationReport(document, 'current run', lang);
|
|
746
856
|
// gate 内核 = run + verdict, 自动覆盖 omk 全部决策维度(三层 layer-gate /
|
|
747
857
|
// bootstrap diff CI / saturation / Krippendorff α)。computeVerdict 是单一
|
|
748
858
|
// 决策源, exit code 跟 verdict.level 走 — 数据 underpowered 直接 FAIL,
|
|
@@ -798,7 +908,7 @@ async function handleDiff(argv) {
|
|
|
798
908
|
console.error(tCli('cli.help.diff_usage', lang));
|
|
799
909
|
process.exit(positional.length === 0 ? 1 : 0);
|
|
800
910
|
}
|
|
801
|
-
const { values } =
|
|
911
|
+
const { values } = parseArgsStrictOrExit({
|
|
802
912
|
args: flagArgs,
|
|
803
913
|
options: {
|
|
804
914
|
...COMMON_OPTIONS,
|
|
@@ -807,7 +917,6 @@ async function handleDiff(argv) {
|
|
|
807
917
|
variant: { type: 'string' },
|
|
808
918
|
top: { type: 'string' },
|
|
809
919
|
},
|
|
810
|
-
strict: false,
|
|
811
920
|
});
|
|
812
921
|
const { createFileStore } = await import('../server/report-store.js');
|
|
813
922
|
const store = createFileStore(resolve(DEFAULT_REPORTS_DIR));
|
|
@@ -816,16 +925,8 @@ async function handleDiff(argv) {
|
|
|
816
925
|
return;
|
|
817
926
|
}
|
|
818
927
|
const [id1, id2] = positional;
|
|
819
|
-
const r1 = await store.get(id1);
|
|
820
|
-
const r2 = await store.get(id2);
|
|
821
|
-
if (!r1) {
|
|
822
|
-
console.error(tCli('cli.common.report_not_found', lang, { id: id1 }));
|
|
823
|
-
process.exit(1);
|
|
824
|
-
}
|
|
825
|
-
if (!r2) {
|
|
826
|
-
console.error(tCli('cli.common.report_not_found', lang, { id: id2 }));
|
|
827
|
-
process.exit(1);
|
|
828
|
-
}
|
|
928
|
+
const r1 = requireEvaluationReport(await store.get(id1), id1, lang);
|
|
929
|
+
const r2 = requireEvaluationReport(await store.get(id2), id2, lang);
|
|
829
930
|
console.log(`\n Diff: ${id1} → ${id2}\n`);
|
|
830
931
|
// Git info — r1/r2 are guaranteed non-null after process.exit() guards above
|
|
831
932
|
const g1 = r1.meta?.gitInfo;
|
|
@@ -833,6 +934,10 @@ async function handleDiff(argv) {
|
|
|
833
934
|
if (g1 || g2) {
|
|
834
935
|
console.log(` Git: ${g1?.commitShort || '?'}${g1?.dirty ? '*' : ''} (${g1?.branch || '?'}) → ${g2?.commitShort || '?'}${g2?.dirty ? '*' : ''} (${g2?.branch || '?'})`);
|
|
835
936
|
}
|
|
937
|
+
const { crossReportComparabilityWarnings, formatComparabilityWarnings } = await import('../eval-core/comparability.js');
|
|
938
|
+
const comparability = formatComparabilityWarnings(crossReportComparabilityWarnings(r1, r2), lang);
|
|
939
|
+
if (comparability)
|
|
940
|
+
process.stderr.write(`\n${comparability}\n\n`);
|
|
836
941
|
// Per-variant comparison
|
|
837
942
|
const variants = [...new Set([...(r1.meta?.variants || []), ...(r2.meta?.variants || [])])];
|
|
838
943
|
for (const v of variants) {
|
|
@@ -861,8 +966,14 @@ async function handleDiff(argv) {
|
|
|
861
966
|
}
|
|
862
967
|
const cost1 = s1?.avgCostPerSample ?? 0;
|
|
863
968
|
const cost2 = s2?.avgCostPerSample ?? 0;
|
|
864
|
-
const
|
|
865
|
-
|
|
969
|
+
const reported1 = s1?.execCostReported !== false;
|
|
970
|
+
const reported2 = s2?.execCostReported !== false;
|
|
971
|
+
const fmt = (c, r) => r ? `$${c.toFixed(4)}` : '—';
|
|
972
|
+
// 任一边 not reported 就不报增减百分比(没意义)
|
|
973
|
+
const costPct = (reported1 && reported2 && cost1 > 0)
|
|
974
|
+
? ` (${cost2 > cost1 ? '+' : ''}${(((cost2 - cost1) / cost1) * 100).toFixed(0)}%)`
|
|
975
|
+
: '';
|
|
976
|
+
console.log(` Cost: ${fmt(cost1, reported1)} → ${fmt(cost2, reported2)}${costPct}`);
|
|
866
977
|
// Skill hash change
|
|
867
978
|
const h1 = r1.meta?.artifactHashes?.[v];
|
|
868
979
|
const h2 = r2.meta?.artifactHashes?.[v];
|
|
@@ -880,11 +991,11 @@ async function handleDiff(argv) {
|
|
|
880
991
|
* `--variant` overrides which variant is the "treatment" side.
|
|
881
992
|
*/
|
|
882
993
|
async function runSampleLevelDiff(reportId, store, flags, lang) {
|
|
883
|
-
const report = await store.get(reportId);
|
|
884
|
-
|
|
885
|
-
|
|
886
|
-
|
|
887
|
-
|
|
994
|
+
const report = requireEvaluationReport(await store.get(reportId), reportId, lang);
|
|
995
|
+
const { reportComparabilityWarnings, formatComparabilityWarnings } = await import('../eval-core/comparability.js');
|
|
996
|
+
const comparability = formatComparabilityWarnings(reportComparabilityWarnings(report), lang);
|
|
997
|
+
if (comparability)
|
|
998
|
+
process.stderr.write(`\n${comparability}\n\n`);
|
|
888
999
|
const variants = report.meta?.variants ?? [];
|
|
889
1000
|
if (variants.length < 2) {
|
|
890
1001
|
console.error('Sample-level diff needs at least 2 variants in the report.');
|
|
@@ -967,14 +1078,13 @@ async function handleGold(argv) {
|
|
|
967
1078
|
process.exit(sub ? 0 : 1);
|
|
968
1079
|
}
|
|
969
1080
|
if (sub === 'init') {
|
|
970
|
-
const { values } =
|
|
1081
|
+
const { values } = parseArgsStrictOrExit({
|
|
971
1082
|
args: rest,
|
|
972
1083
|
options: {
|
|
973
1084
|
...COMMON_OPTIONS,
|
|
974
1085
|
out: { type: 'string', default: './gold-dataset' },
|
|
975
1086
|
annotator: { type: 'string' },
|
|
976
1087
|
},
|
|
977
|
-
strict: false,
|
|
978
1088
|
});
|
|
979
1089
|
const { initGoldDataset } = await import('../grading/gold-cli.js');
|
|
980
1090
|
try {
|
|
@@ -995,7 +1105,14 @@ async function handleGold(argv) {
|
|
|
995
1105
|
return;
|
|
996
1106
|
}
|
|
997
1107
|
if (sub === 'validate') {
|
|
998
|
-
|
|
1108
|
+
// 走 helper 让 `omk bench gold validate <dir> --bogus` 走 unknown option 路径,
|
|
1109
|
+
// 而不是直接执行 validate 后再报 dataset 错。
|
|
1110
|
+
const { positionals } = parseArgsStrictOrExit({
|
|
1111
|
+
args: rest,
|
|
1112
|
+
allowPositionals: true,
|
|
1113
|
+
options: { ...COMMON_OPTIONS },
|
|
1114
|
+
});
|
|
1115
|
+
const dir = positionals[0];
|
|
999
1116
|
if (!dir) {
|
|
1000
1117
|
console.error(tCli('cli.common.usage_gold_validate', lang));
|
|
1001
1118
|
process.exit(1);
|
|
@@ -1017,7 +1134,7 @@ async function handleGold(argv) {
|
|
|
1017
1134
|
console.error('Usage: omk bench gold compare <reportId> --gold-dir <dir>');
|
|
1018
1135
|
process.exit(1);
|
|
1019
1136
|
}
|
|
1020
|
-
const { values } =
|
|
1137
|
+
const { values } = parseArgsStrictOrExit({
|
|
1021
1138
|
args: rest.slice(1),
|
|
1022
1139
|
options: {
|
|
1023
1140
|
...COMMON_OPTIONS,
|
|
@@ -1027,7 +1144,6 @@ async function handleGold(argv) {
|
|
|
1027
1144
|
'bootstrap-samples': { type: 'string', default: '1000' },
|
|
1028
1145
|
seed: { type: 'string' },
|
|
1029
1146
|
},
|
|
1030
|
-
strict: false,
|
|
1031
1147
|
});
|
|
1032
1148
|
const goldDir = values['gold-dir'];
|
|
1033
1149
|
if (!goldDir) {
|
|
@@ -1050,15 +1166,11 @@ async function handleGold(argv) {
|
|
|
1050
1166
|
console.error(`warn: ${i.message}`);
|
|
1051
1167
|
}
|
|
1052
1168
|
const store = createFileStore(resolve(values['reports-dir']));
|
|
1053
|
-
const report = await store.get(reportId);
|
|
1054
|
-
if (!report) {
|
|
1055
|
-
console.error(tCli('cli.common.report_not_found', lang, { id: reportId }));
|
|
1056
|
-
process.exit(1);
|
|
1057
|
-
}
|
|
1169
|
+
const report = requireEvaluationReport(await store.get(reportId), reportId, lang);
|
|
1058
1170
|
const samples = Math.max(100, Number(values['bootstrap-samples']) || 1000);
|
|
1059
1171
|
const seedVal = values.seed != null ? Number(values.seed) : undefined;
|
|
1060
1172
|
const result = compareGoldToReport({
|
|
1061
|
-
report
|
|
1173
|
+
report,
|
|
1062
1174
|
gold: dataset,
|
|
1063
1175
|
variant: values.variant,
|
|
1064
1176
|
samples,
|
|
@@ -1071,7 +1183,7 @@ async function handleGold(argv) {
|
|
|
1071
1183
|
process.exit(1);
|
|
1072
1184
|
}
|
|
1073
1185
|
// ---------------------------------------------------------------------------
|
|
1074
|
-
// handleDebiasValidate — measure length-debias prompt sensitivity
|
|
1186
|
+
// handleDebiasValidate — measure length-debias prompt sensitivity
|
|
1075
1187
|
// ---------------------------------------------------------------------------
|
|
1076
1188
|
async function handleDebiasValidate(argv) {
|
|
1077
1189
|
const lang = langFromArgv(argv);
|
|
@@ -1090,27 +1202,31 @@ async function handleDebiasValidate(argv) {
|
|
|
1090
1202
|
console.error('Usage: omk bench debias-validate length <reportId>');
|
|
1091
1203
|
process.exit(1);
|
|
1092
1204
|
}
|
|
1093
|
-
const { values } =
|
|
1205
|
+
const { values } = parseArgsStrictOrExit({
|
|
1094
1206
|
args: rest.slice(1),
|
|
1095
1207
|
options: {
|
|
1096
1208
|
...COMMON_OPTIONS,
|
|
1097
1209
|
'reports-dir': { type: 'string', default: DEFAULT_REPORTS_DIR },
|
|
1098
1210
|
samples: { type: 'string' },
|
|
1099
1211
|
variant: { type: 'string' },
|
|
1100
|
-
'judge-
|
|
1101
|
-
'judge-model': { type: 'string' },
|
|
1212
|
+
'judge-models': { type: 'string' },
|
|
1102
1213
|
'bootstrap-samples': { type: 'string', default: '1000' },
|
|
1103
1214
|
seed: { type: 'string' },
|
|
1104
1215
|
},
|
|
1105
|
-
strict: false,
|
|
1106
1216
|
});
|
|
1217
|
+
// Parse --judge-models 在 load report 之前 fail-fast。重复 entry / 缺 executor /
|
|
1218
|
+
// 空串等参数错误应立即给 friendly error: + exit 2,不要等到 store IO 完成才暴露。
|
|
1219
|
+
const { parseJudgeModelsArgOrExit: parseJudgesA } = await import('./parse-run-config.js');
|
|
1220
|
+
const cliJudgeModelsA = values['judge-models'] !== undefined
|
|
1221
|
+
? parseJudgesA(values['judge-models'])
|
|
1222
|
+
: undefined;
|
|
1223
|
+
if (cliJudgeModelsA && cliJudgeModelsA.length > 1) {
|
|
1224
|
+
console.error(tCli('cli.common.judge_models_single_only', lang, { cmd: 'debias-validate' }));
|
|
1225
|
+
process.exit(2);
|
|
1226
|
+
}
|
|
1107
1227
|
const { createFileStore } = await import('../server/report-store.js');
|
|
1108
1228
|
const store = createFileStore(resolve(values['reports-dir']));
|
|
1109
|
-
const report = await store.get(reportId);
|
|
1110
|
-
if (!report) {
|
|
1111
|
-
console.error(tCli('cli.common.report_not_found', lang, { id: reportId }));
|
|
1112
|
-
process.exit(1);
|
|
1113
|
-
}
|
|
1229
|
+
const report = requireEvaluationReport(await store.get(reportId), reportId, lang);
|
|
1114
1230
|
// Resolve samples path: --samples overrides; otherwise read from report.meta.request.
|
|
1115
1231
|
const samplesPath = values.samples
|
|
1116
1232
|
?? report.meta?.request?.samplesPath;
|
|
@@ -1120,20 +1236,23 @@ async function handleDebiasValidate(argv) {
|
|
|
1120
1236
|
}
|
|
1121
1237
|
const { loadSamples } = await import('../inputs/load-samples.js');
|
|
1122
1238
|
const { samples } = loadSamples(samplesPath);
|
|
1123
|
-
const
|
|
1124
|
-
?? report.meta?.
|
|
1125
|
-
|
|
1239
|
+
const debiasJudges = cliJudgeModelsA
|
|
1240
|
+
?? (report.meta?.judgeModels?.[0]
|
|
1241
|
+
? [{ executor: report.meta.judgeModels[0].executor, model: report.meta.judgeModels[0].model }]
|
|
1242
|
+
: []);
|
|
1243
|
+
if (debiasJudges.length === 0) {
|
|
1126
1244
|
console.error(tCli('cli.common.no_judge_model', lang));
|
|
1127
1245
|
process.exit(1);
|
|
1128
1246
|
}
|
|
1129
1247
|
process.stderr.write(tCli('cli.debias.warn_cost_doubles', lang));
|
|
1130
1248
|
const { createExecutor } = await import('../executors/index.js');
|
|
1131
|
-
const judgeExecutor = createExecutor(
|
|
1249
|
+
const judgeExecutor = createExecutor(debiasJudges[0].executor);
|
|
1250
|
+
const judgeModel = debiasJudges[0].model;
|
|
1132
1251
|
const { validateLengthDebias, formatDebiasValidate } = await import('../grading/debias-validate.js');
|
|
1133
1252
|
const seedVal = values.seed != null ? Number(values.seed) : undefined;
|
|
1134
1253
|
const bsRaw = Number(values['bootstrap-samples']) || 1000;
|
|
1135
1254
|
const result = await validateLengthDebias({
|
|
1136
|
-
report
|
|
1255
|
+
report,
|
|
1137
1256
|
samples,
|
|
1138
1257
|
judgeExecutor,
|
|
1139
1258
|
judgeModel,
|
|
@@ -1156,22 +1275,17 @@ async function handleSaturation(argv) {
|
|
|
1156
1275
|
console.log(tCli('cli.help.saturation', lang));
|
|
1157
1276
|
process.exit(reportId ? 0 : 1);
|
|
1158
1277
|
}
|
|
1159
|
-
const { values } =
|
|
1278
|
+
const { values } = parseArgsStrictOrExit({
|
|
1160
1279
|
args: argv.slice(1),
|
|
1161
1280
|
options: {
|
|
1162
1281
|
...COMMON_OPTIONS,
|
|
1163
1282
|
'reports-dir': { type: 'string', default: DEFAULT_REPORTS_DIR },
|
|
1164
1283
|
variant: { type: 'string' },
|
|
1165
1284
|
},
|
|
1166
|
-
strict: false,
|
|
1167
1285
|
});
|
|
1168
1286
|
const { createFileStore } = await import('../server/report-store.js');
|
|
1169
1287
|
const store = createFileStore(resolve(values['reports-dir']));
|
|
1170
|
-
const report = await store.get(reportId);
|
|
1171
|
-
if (!report) {
|
|
1172
|
-
console.error(tCli('cli.common.report_not_found', lang, { id: reportId }));
|
|
1173
|
-
process.exit(1);
|
|
1174
|
-
}
|
|
1288
|
+
const report = requireEvaluationReport(await store.get(reportId), reportId, lang);
|
|
1175
1289
|
const saturation = report.variance?.saturation;
|
|
1176
1290
|
if (!saturation) {
|
|
1177
1291
|
console.error(tCli('cli.saturation.no_data', lang));
|
|
@@ -1225,7 +1339,7 @@ async function handleVerdict(argv) {
|
|
|
1225
1339
|
console.log(tCli('cli.help.verdict', lang));
|
|
1226
1340
|
process.exit(reportId ? 0 : 1);
|
|
1227
1341
|
}
|
|
1228
|
-
const { values } =
|
|
1342
|
+
const { values } = parseArgsStrictOrExit({
|
|
1229
1343
|
args: argv.slice(1),
|
|
1230
1344
|
options: {
|
|
1231
1345
|
...COMMON_OPTIONS,
|
|
@@ -1234,16 +1348,15 @@ async function handleVerdict(argv) {
|
|
|
1234
1348
|
'trivial-diff': { type: 'string' },
|
|
1235
1349
|
verbose: { type: 'boolean', default: false },
|
|
1236
1350
|
},
|
|
1237
|
-
strict: false,
|
|
1238
1351
|
});
|
|
1239
1352
|
const { createFileStore } = await import('../server/report-store.js');
|
|
1240
1353
|
const store = createFileStore(resolve(values['reports-dir']));
|
|
1241
|
-
const report = await store.get(reportId);
|
|
1242
|
-
if (!report) {
|
|
1243
|
-
console.error(tCli('cli.common.report_not_found', lang, { id: reportId }));
|
|
1244
|
-
process.exit(1);
|
|
1245
|
-
}
|
|
1354
|
+
const report = requireEvaluationReport(await store.get(reportId), reportId, lang);
|
|
1246
1355
|
const { computeVerdict, formatVerdictText } = await import('../eval-core/verdict.js');
|
|
1356
|
+
const { reportComparabilityWarnings, formatComparabilityWarnings } = await import('../eval-core/comparability.js');
|
|
1357
|
+
const comparability = formatComparabilityWarnings(reportComparabilityWarnings(report), lang);
|
|
1358
|
+
if (comparability)
|
|
1359
|
+
process.stderr.write(`${comparability}\n`);
|
|
1247
1360
|
const result = computeVerdict(report, {
|
|
1248
1361
|
gateThreshold: values.threshold != null ? Number(values.threshold) : undefined,
|
|
1249
1362
|
triviallySmallDiff: values['trivial-diff'] != null ? Number(values['trivial-diff']) : undefined,
|
|
@@ -1270,7 +1383,7 @@ async function handleDiagnose(argv) {
|
|
|
1270
1383
|
console.log(tCli('cli.help.diagnose', lang));
|
|
1271
1384
|
process.exit(reportId ? 0 : 1);
|
|
1272
1385
|
}
|
|
1273
|
-
const { values } =
|
|
1386
|
+
const { values } = parseArgsStrictOrExit({
|
|
1274
1387
|
args: argv.slice(1),
|
|
1275
1388
|
options: {
|
|
1276
1389
|
...COMMON_OPTIONS,
|
|
@@ -1283,15 +1396,10 @@ async function handleDiagnose(argv) {
|
|
|
1283
1396
|
'latency-k': { type: 'string' },
|
|
1284
1397
|
flat: { type: 'string' },
|
|
1285
1398
|
},
|
|
1286
|
-
strict: false,
|
|
1287
1399
|
});
|
|
1288
1400
|
const { createFileStore } = await import('../server/report-store.js');
|
|
1289
1401
|
const store = createFileStore(resolve(values['reports-dir']));
|
|
1290
|
-
const report = await store.get(reportId);
|
|
1291
|
-
if (!report) {
|
|
1292
|
-
console.error(tCli('cli.common.report_not_found', lang, { id: reportId }));
|
|
1293
|
-
process.exit(1);
|
|
1294
|
-
}
|
|
1402
|
+
const report = requireEvaluationReport(await store.get(reportId), reportId, lang);
|
|
1295
1403
|
// Try to read the samples file for near-duplicate detection. Source order:
|
|
1296
1404
|
// 1. --samples <path> override
|
|
1297
1405
|
// 2. report.meta.request.samplesPath (recorded at run time)
|
|
@@ -1320,7 +1428,7 @@ async function handleDiagnose(argv) {
|
|
|
1320
1428
|
latencyOutlierK: values['latency-k'] != null ? Number(values['latency-k']) : undefined,
|
|
1321
1429
|
flatThreshold: values.flat != null ? Number(values.flat) : undefined,
|
|
1322
1430
|
});
|
|
1323
|
-
console.log(formatSampleDiagnostics(diag, { topN }));
|
|
1431
|
+
console.log(formatSampleDiagnostics(diag, { topN, lang }));
|
|
1324
1432
|
// Sample design science coverage block. Render after diagnose 主体,因为
|
|
1325
1433
|
// coverage 是声明式元数据(capability/difficulty/construct/provenance)的整体分布,
|
|
1326
1434
|
// 跟 issue list 是不同视角的两件事。优先从 samples (现场加载) 算,fallback 到
|
|
@@ -1345,36 +1453,43 @@ async function handleFailures(argv) {
|
|
|
1345
1453
|
console.log(tCli('cli.help.failures', lang));
|
|
1346
1454
|
process.exit(reportId ? 0 : 1);
|
|
1347
1455
|
}
|
|
1348
|
-
const { values } =
|
|
1456
|
+
const { values } = parseArgsStrictOrExit({
|
|
1349
1457
|
args: argv.slice(1),
|
|
1350
1458
|
options: {
|
|
1351
1459
|
...COMMON_OPTIONS,
|
|
1352
1460
|
'reports-dir': { type: 'string', default: DEFAULT_REPORTS_DIR },
|
|
1353
|
-
'judge-
|
|
1354
|
-
'judge-model': { type: 'string' },
|
|
1461
|
+
'judge-models': { type: 'string' },
|
|
1355
1462
|
'max-clusters': { type: 'string', default: '5' },
|
|
1356
1463
|
threshold: { type: 'string', default: '3' },
|
|
1357
1464
|
'max-feed': { type: 'string', default: '50' },
|
|
1358
1465
|
},
|
|
1359
|
-
strict: false,
|
|
1360
1466
|
});
|
|
1467
|
+
// Parse --judge-models 在 load report 之前 fail-fast(同 debias-validate)。
|
|
1468
|
+
const { parseJudgeModelsArgOrExit: parseJudgesB } = await import('./parse-run-config.js');
|
|
1469
|
+
const cliJudgeModelsB = values['judge-models'] !== undefined
|
|
1470
|
+
? parseJudgesB(values['judge-models'])
|
|
1471
|
+
: undefined;
|
|
1472
|
+
if (cliJudgeModelsB && cliJudgeModelsB.length > 1) {
|
|
1473
|
+
console.error(tCli('cli.common.judge_models_single_only', lang, { cmd: 'failures' }));
|
|
1474
|
+
process.exit(2);
|
|
1475
|
+
}
|
|
1361
1476
|
const { createFileStore } = await import('../server/report-store.js');
|
|
1362
1477
|
const store = createFileStore(resolve(values['reports-dir']));
|
|
1363
|
-
const report = await store.get(reportId);
|
|
1364
|
-
|
|
1365
|
-
|
|
1366
|
-
|
|
1367
|
-
|
|
1368
|
-
|
|
1369
|
-
if (!judgeModel) {
|
|
1478
|
+
const report = requireEvaluationReport(await store.get(reportId), reportId, lang);
|
|
1479
|
+
const failuresJudges = cliJudgeModelsB
|
|
1480
|
+
?? (report.meta?.judgeModels?.[0]
|
|
1481
|
+
? [{ executor: report.meta.judgeModels[0].executor, model: report.meta.judgeModels[0].model }]
|
|
1482
|
+
: []);
|
|
1483
|
+
if (failuresJudges.length === 0) {
|
|
1370
1484
|
console.error(tCli('cli.common.no_judge_model', lang));
|
|
1371
1485
|
process.exit(1);
|
|
1372
1486
|
}
|
|
1373
1487
|
const { createExecutor } = await import('../executors/index.js');
|
|
1374
|
-
const executor = createExecutor(
|
|
1488
|
+
const executor = createExecutor(failuresJudges[0].executor);
|
|
1489
|
+
const judgeModel = failuresJudges[0].model;
|
|
1375
1490
|
const { clusterFailures, formatFailureClusterReport } = await import('../analysis/failure-clusterer.js');
|
|
1376
1491
|
const out = await clusterFailures({
|
|
1377
|
-
report
|
|
1492
|
+
report,
|
|
1378
1493
|
executor,
|
|
1379
1494
|
judgeModel,
|
|
1380
1495
|
maxClusters: Number(values['max-clusters']) || 5,
|