oh-my-knowledge 0.26.0 → 0.28.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +105 -346
- package/README.zh.md +142 -375
- package/dist/src/analysis/report-diagnostics.d.ts +2 -2
- package/dist/src/analysis/report-diagnostics.js +2 -2
- package/dist/src/analysis/sample-diagnostics.d.ts +3 -3
- package/dist/src/analysis/sample-diagnostics.js +3 -3
- package/dist/src/authoring/generator.d.ts.map +1 -1
- package/dist/src/authoring/generator.js +9 -8
- package/dist/src/authoring/generator.js.map +1 -1
- package/dist/src/cli/cli-exit.d.ts +15 -0
- package/dist/src/cli/cli-exit.d.ts.map +1 -0
- package/dist/src/cli/cli-exit.js +19 -0
- package/dist/src/cli/cli-exit.js.map +1 -0
- package/dist/src/cli/commands/_shared.d.ts +12 -0
- package/dist/src/cli/commands/_shared.d.ts.map +1 -0
- package/dist/src/cli/commands/_shared.js +26 -0
- package/dist/src/cli/commands/_shared.js.map +1 -0
- package/dist/src/cli/commands/doctor.d.ts +2 -0
- package/dist/src/cli/commands/doctor.d.ts.map +1 -0
- package/dist/src/cli/commands/doctor.js +174 -0
- package/dist/src/cli/commands/doctor.js.map +1 -0
- package/dist/src/cli/commands/eval-gold.d.ts +2 -0
- package/dist/src/cli/commands/eval-gold.d.ts.map +1 -0
- package/dist/src/cli/commands/eval-gold.js +137 -0
- package/dist/src/cli/commands/eval-gold.js.map +1 -0
- package/dist/src/cli/commands/eval-runner.d.ts +2 -0
- package/dist/src/cli/commands/eval-runner.d.ts.map +1 -0
- package/dist/src/cli/commands/eval-runner.js +299 -0
- package/dist/src/cli/commands/eval-runner.js.map +1 -0
- package/dist/src/cli/commands/eval.d.ts +2 -0
- package/dist/src/cli/commands/eval.d.ts.map +1 -0
- package/dist/src/cli/commands/eval.js +11 -0
- package/dist/src/cli/commands/eval.js.map +1 -0
- package/dist/src/cli/commands/evolve.d.ts +2 -0
- package/dist/src/cli/commands/evolve.d.ts.map +1 -0
- package/dist/src/cli/commands/evolve.js +115 -0
- package/dist/src/cli/commands/evolve.js.map +1 -0
- package/dist/src/cli/commands/init.d.ts +2 -0
- package/dist/src/cli/commands/init.d.ts.map +1 -0
- package/dist/src/cli/commands/init.js +110 -0
- package/dist/src/cli/commands/init.js.map +1 -0
- package/dist/src/cli/commands/observe.d.ts +2 -0
- package/dist/src/cli/commands/observe.d.ts.map +1 -0
- package/dist/src/cli/commands/observe.js +77 -0
- package/dist/src/cli/commands/observe.js.map +1 -0
- package/dist/src/cli/commands/registry.d.ts +11 -0
- package/dist/src/cli/commands/registry.d.ts.map +1 -0
- package/dist/src/cli/commands/registry.js +24 -0
- package/dist/src/cli/commands/registry.js.map +1 -0
- package/dist/src/cli/commands/sample.d.ts +2 -0
- package/dist/src/cli/commands/sample.d.ts.map +1 -0
- package/dist/src/cli/commands/sample.js +121 -0
- package/dist/src/cli/commands/sample.js.map +1 -0
- package/dist/src/cli/commands/studio.d.ts +2 -0
- package/dist/src/cli/commands/studio.d.ts.map +1 -0
- package/dist/src/cli/commands/studio.js +76 -0
- package/dist/src/cli/commands/studio.js.map +1 -0
- package/dist/src/cli/i18n-dict.d.ts +2 -4
- package/dist/src/cli/i18n-dict.d.ts.map +1 -1
- package/dist/src/cli/i18n-dict.js +461 -675
- package/dist/src/cli/i18n-dict.js.map +1 -1
- package/dist/src/cli/index.js +43 -1516
- package/dist/src/cli/index.js.map +1 -1
- package/dist/src/cli/parse-run-config.d.ts +3 -4
- package/dist/src/cli/parse-run-config.d.ts.map +1 -1
- package/dist/src/cli/parse-run-config.js +5 -5
- package/dist/src/cli/parse-run-config.js.map +1 -1
- package/dist/src/cli/parse-strict.d.ts +0 -13
- package/dist/src/cli/parse-strict.d.ts.map +1 -1
- package/dist/src/cli/parse-strict.js +2 -1
- package/dist/src/cli/parse-strict.js.map +1 -1
- package/dist/src/doctor/health/builtin-dimensions.d.ts +11 -0
- package/dist/src/doctor/health/builtin-dimensions.d.ts.map +1 -0
- package/dist/src/doctor/health/builtin-dimensions.js +94 -0
- package/dist/src/doctor/health/builtin-dimensions.js.map +1 -0
- package/dist/src/doctor/health/composer.d.ts +19 -0
- package/dist/src/doctor/health/composer.d.ts.map +1 -0
- package/dist/src/doctor/health/composer.js +289 -0
- package/dist/src/doctor/health/composer.js.map +1 -0
- package/dist/src/doctor/health/dimension-registry.d.ts +13 -0
- package/dist/src/doctor/health/dimension-registry.d.ts.map +1 -0
- package/dist/src/doctor/health/dimension-registry.js +28 -0
- package/dist/src/doctor/health/dimension-registry.js.map +1 -0
- package/dist/src/doctor/health/dimension-spec.d.ts +46 -0
- package/dist/src/doctor/health/dimension-spec.d.ts.map +1 -0
- package/dist/src/doctor/health/dimension-spec.js +12 -0
- package/dist/src/doctor/health/dimension-spec.js.map +1 -0
- package/dist/src/doctor/health/parser.d.ts +27 -0
- package/dist/src/doctor/health/parser.d.ts.map +1 -0
- package/dist/src/doctor/health/parser.js +190 -0
- package/dist/src/doctor/health/parser.js.map +1 -0
- package/dist/src/doctor/health/prompt-builder.d.ts +22 -0
- package/dist/src/doctor/health/prompt-builder.d.ts.map +1 -0
- package/dist/src/doctor/health/prompt-builder.js +162 -0
- package/dist/src/doctor/health/prompt-builder.js.map +1 -0
- package/dist/src/doctor/health/register.d.ts +13 -0
- package/dist/src/doctor/health/register.d.ts.map +1 -0
- package/dist/src/doctor/health/register.js +20 -0
- package/dist/src/doctor/health/register.js.map +1 -0
- package/dist/src/doctor/html-renderer.d.ts +20 -0
- package/dist/src/doctor/html-renderer.d.ts.map +1 -0
- package/dist/src/doctor/html-renderer.js +366 -0
- package/dist/src/doctor/html-renderer.js.map +1 -0
- package/dist/src/doctor/index.d.ts +1 -1
- package/dist/src/doctor/index.d.ts.map +1 -1
- package/dist/src/doctor/index.js +57 -8
- package/dist/src/doctor/index.js.map +1 -1
- package/dist/src/doctor/preflight.d.ts +2 -2
- package/dist/src/doctor/preflight.js +2 -2
- package/dist/src/doctor/renderer.d.ts +4 -0
- package/dist/src/doctor/renderer.d.ts.map +1 -1
- package/dist/src/doctor/renderer.js +72 -18
- package/dist/src/doctor/renderer.js.map +1 -1
- package/dist/src/doctor/rules.d.ts +6 -5
- package/dist/src/doctor/rules.d.ts.map +1 -1
- package/dist/src/doctor/rules.js +4 -3
- package/dist/src/doctor/rules.js.map +1 -1
- package/dist/src/eval-core/fact-checker.js +1 -1
- package/dist/src/eval-core/fact-checker.js.map +1 -1
- package/dist/src/eval-core/layer-gates.d.ts +1 -1
- package/dist/src/eval-core/layer-gates.js +1 -1
- package/dist/src/eval-core/verdict.d.ts +4 -4
- package/dist/src/eval-core/verdict.d.ts.map +1 -1
- package/dist/src/eval-core/verdict.js +2 -2
- package/dist/src/eval-workflows/batch-evaluation-workflow.d.ts +15 -2
- package/dist/src/eval-workflows/batch-evaluation-workflow.d.ts.map +1 -1
- package/dist/src/eval-workflows/batch-evaluation-workflow.js +13 -2
- package/dist/src/eval-workflows/batch-evaluation-workflow.js.map +1 -1
- package/dist/src/eval-workflows/evaluation-pipeline.d.ts +6 -3
- package/dist/src/eval-workflows/evaluation-pipeline.d.ts.map +1 -1
- package/dist/src/eval-workflows/evaluation-pipeline.js +10 -9
- package/dist/src/eval-workflows/evaluation-pipeline.js.map +1 -1
- package/dist/src/eval-workflows/run-evaluation.d.ts +1 -1
- package/dist/src/eval-workflows/run-evaluation.d.ts.map +1 -1
- package/dist/src/eval-workflows/run-evaluation.js +13 -6
- package/dist/src/eval-workflows/run-evaluation.js.map +1 -1
- package/dist/src/executors/script.d.ts.map +1 -1
- package/dist/src/executors/script.js +16 -3
- package/dist/src/executors/script.js.map +1 -1
- package/dist/src/grading/debias-validate.d.ts +2 -2
- package/dist/src/grading/debias-validate.js +2 -2
- package/dist/src/grading/gold-cli.d.ts +2 -5
- package/dist/src/grading/gold-cli.d.ts.map +1 -1
- package/dist/src/grading/gold-cli.js +4 -8
- package/dist/src/grading/gold-cli.js.map +1 -1
- package/dist/src/grading/judge.d.ts +1 -1
- package/dist/src/renderer/html-renderer.js +1 -1
- package/dist/src/renderer/layout.js +5 -5
- package/dist/src/renderer/layout.js.map +1 -1
- package/dist/src/renderer/skill-health-renderer.d.ts +1 -1
- package/dist/src/renderer/skill-health-renderer.js +1 -1
- package/dist/src/renderer/summary.js +8 -8
- package/dist/src/server/report-server.d.ts.map +1 -1
- package/dist/src/server/report-server.js +8 -7
- package/dist/src/server/report-server.js.map +1 -1
- package/dist/src/types/doctor.d.ts +44 -6
- package/dist/src/types/doctor.d.ts.map +1 -1
- package/dist/src/types/doctor.js +3 -0
- package/dist/src/types/doctor.js.map +1 -1
- package/dist/src/types/eval.d.ts +1 -1
- package/dist/src/types/report.d.ts +2 -2
- package/dist/src/types/report.d.ts.map +1 -1
- package/package.json +1 -1
- package/dist/src/cli/coverage-renderer.d.ts +0 -15
- package/dist/src/cli/coverage-renderer.d.ts.map +0 -1
- package/dist/src/cli/coverage-renderer.js +0 -74
- package/dist/src/cli/coverage-renderer.js.map +0 -1
package/dist/src/cli/index.js
CHANGED
|
@@ -1,1534 +1,61 @@
|
|
|
1
1
|
#!/usr/bin/env node
|
|
2
|
-
import {
|
|
3
|
-
import { join } from 'node:path';
|
|
4
|
-
import { existsSync } from 'node:fs';
|
|
5
|
-
import { tCli, getCliLang, parseLangFromArgv, langFromArgv } from './i18n.js';
|
|
6
|
-
import { parseRunConfig, DEFAULT_REPORTS_DIR, COMMON_OPTIONS, } from './parse-run-config.js';
|
|
7
|
-
import { makeOnProgress } from './progress.js';
|
|
8
|
-
import { computeRunTally } from './run-tally.js';
|
|
2
|
+
import { tCli, getCliLang, parseLangFromArgv } from './i18n.js';
|
|
9
3
|
import { checkUpdate } from './update-check.js';
|
|
10
|
-
import {
|
|
11
|
-
|
|
12
|
-
if (!report) {
|
|
13
|
-
console.error(tCli('cli.common.report_not_found', lang, { id }));
|
|
14
|
-
process.exit(1);
|
|
15
|
-
}
|
|
16
|
-
if (report.kind === 'batch-evaluation') {
|
|
17
|
-
console.error(lang === 'zh'
|
|
18
|
-
? `报告 ${id} 是 BatchEvaluationReport。该命令需要单次 EvaluationReport;请使用其中的 child reportId。`
|
|
19
|
-
: `Report ${id} is a BatchEvaluationReport. This command requires an EvaluationReport; use a child reportId from the batch.`);
|
|
20
|
-
process.exit(1);
|
|
21
|
-
}
|
|
22
|
-
return report;
|
|
23
|
-
}
|
|
24
|
-
// ---------------------------------------------------------------------------
|
|
25
|
-
// Main
|
|
26
|
-
// ---------------------------------------------------------------------------
|
|
27
|
-
async function main() {
|
|
28
|
-
const lang = getCliLang(parseLangFromArgv(process.argv));
|
|
29
|
-
checkUpdate(lang);
|
|
30
|
-
const [domain, command, ...rest] = process.argv.slice(2);
|
|
31
|
-
if (!domain || domain === '--help' || domain === '-h') {
|
|
32
|
-
console.log(tCli('cli.help.main', lang).trim());
|
|
33
|
-
process.exit(0);
|
|
34
|
-
}
|
|
35
|
-
if (domain === 'analyze') {
|
|
36
|
-
const args = command ? [command, ...rest] : [];
|
|
37
|
-
await handleAnalyze(args);
|
|
38
|
-
return;
|
|
39
|
-
}
|
|
40
|
-
if (domain === 'doctor') {
|
|
41
|
-
const args = command ? [command, ...rest] : [];
|
|
42
|
-
await handleDoctor(args);
|
|
43
|
-
return;
|
|
44
|
-
}
|
|
45
|
-
if (domain !== 'bench') {
|
|
46
|
-
console.error(tCli('cli.common.unknown_domain', lang, { domain }));
|
|
47
|
-
process.exit(1);
|
|
48
|
-
}
|
|
49
|
-
if (!command || command === '--help' || command === '-h') {
|
|
50
|
-
console.log(tCli('cli.help.main', lang).trim());
|
|
51
|
-
process.exit(0);
|
|
52
|
-
}
|
|
53
|
-
switch (command) {
|
|
54
|
-
case 'run':
|
|
55
|
-
await handleRun(rest);
|
|
56
|
-
break;
|
|
57
|
-
case 'report':
|
|
58
|
-
await handleReport(rest);
|
|
59
|
-
break;
|
|
60
|
-
case 'init':
|
|
61
|
-
await handleInit(rest);
|
|
62
|
-
break;
|
|
63
|
-
case 'gate':
|
|
64
|
-
await handleGate(rest);
|
|
65
|
-
break;
|
|
66
|
-
case 'gen-samples':
|
|
67
|
-
await handleGenSamples(rest);
|
|
68
|
-
break;
|
|
69
|
-
case 'evolve':
|
|
70
|
-
await handleEvolve(rest);
|
|
71
|
-
break;
|
|
72
|
-
case 'diff':
|
|
73
|
-
await handleDiff(rest);
|
|
74
|
-
break;
|
|
75
|
-
case 'gold':
|
|
76
|
-
await handleGold(rest);
|
|
77
|
-
break;
|
|
78
|
-
case 'debias-validate':
|
|
79
|
-
await handleDebiasValidate(rest);
|
|
80
|
-
break;
|
|
81
|
-
case 'saturation':
|
|
82
|
-
await handleSaturation(rest);
|
|
83
|
-
break;
|
|
84
|
-
case 'verdict':
|
|
85
|
-
await handleVerdict(rest);
|
|
86
|
-
break;
|
|
87
|
-
case 'diagnose':
|
|
88
|
-
await handleDiagnose(rest);
|
|
89
|
-
break;
|
|
90
|
-
case 'failures':
|
|
91
|
-
await handleFailures(rest);
|
|
92
|
-
break;
|
|
93
|
-
default:
|
|
94
|
-
console.error(tCli('cli.common.unknown_bench_command', lang, { command }));
|
|
95
|
-
process.exit(1);
|
|
96
|
-
}
|
|
97
|
-
}
|
|
98
|
-
// ---------------------------------------------------------------------------
|
|
99
|
-
// Progress callback
|
|
100
|
-
// ---------------------------------------------------------------------------
|
|
4
|
+
import { CliExit } from './cli-exit.js';
|
|
5
|
+
import { PRODUCT_COMMANDS } from './commands/registry.js';
|
|
101
6
|
/**
|
|
102
|
-
*
|
|
103
|
-
*
|
|
7
|
+
* --help / -h 在 argv 任意位置都打印对应 helpKey 内容并 exit 0。
|
|
8
|
+
* 集中在 dispatcher 处理,因为下游 execute 走 parseArgsStrictOrExit,
|
|
9
|
+
* 那一层 strict:true 不识别 --help 会当 unknown option 报错。
|
|
104
10
|
*/
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
async function handleRun(argv) {
|
|
109
|
-
const lang = langFromArgv(argv);
|
|
110
|
-
if (argv.includes('--help') || argv.includes('-h')) {
|
|
111
|
-
console.log(tCli('cli.help.main', lang).trim());
|
|
112
|
-
process.exit(0);
|
|
113
|
-
}
|
|
114
|
-
// 注: 这里**不**给 parseArgs default 值, 否则 values.xxx 永远不为 undefined,
|
|
115
|
-
// CLI > eval.yaml > hardcoded-default 三级 fallback 区分不开 ("用户没传" vs "用户传了等于 default 值")。
|
|
116
|
-
// hardcoded default 在下面处理 undefined 时显式给。
|
|
117
|
-
const { values, config, evalConfig } = parseRunConfig(argv, {
|
|
118
|
-
blind: { type: 'boolean' },
|
|
119
|
-
repeat: { type: 'string' },
|
|
120
|
-
'judge-repeat': { type: 'string' },
|
|
121
|
-
bootstrap: { type: 'boolean' },
|
|
122
|
-
'bootstrap-samples': { type: 'string' },
|
|
123
|
-
'gold-dir': { type: 'string' },
|
|
124
|
-
'no-debias-length': { type: 'boolean' },
|
|
125
|
-
'budget-usd': { type: 'string' },
|
|
126
|
-
'budget-per-sample-usd': { type: 'string' },
|
|
127
|
-
'budget-per-sample-ms': { type: 'string' },
|
|
128
|
-
});
|
|
129
|
-
const { runEvaluation, runMultiple, runBatchEvaluation } = await import('../eval-workflows/run-evaluation.js');
|
|
130
|
-
if (values.blind !== undefined) {
|
|
131
|
-
config.blind = values.blind;
|
|
132
|
-
}
|
|
133
|
-
config.onProgress = makeOnProgress(lang);
|
|
134
|
-
// --repeat: CLI > eval.yaml > 1. 非 ≥1 整数时提示并钳到 1。
|
|
135
|
-
const repeatRaw = values.repeat;
|
|
136
|
-
const parsedRepeat = repeatRaw !== undefined ? Number(repeatRaw) : (evalConfig?.repeat ?? 1);
|
|
137
|
-
if (repeatRaw !== undefined && (!Number.isFinite(parsedRepeat) || parsedRepeat < 1)) {
|
|
138
|
-
process.stderr.write(tCli('cli.run.invalid_repeat', lang, { value: repeatRaw }));
|
|
139
|
-
}
|
|
140
|
-
const repeatCount = Math.max(1, Math.floor(parsedRepeat) || 1);
|
|
141
|
-
// --judge-repeat: CLI > eval.yaml > 1.
|
|
142
|
-
const judgeRepeatRaw = values['judge-repeat'];
|
|
143
|
-
const parsedJudgeRepeat = judgeRepeatRaw !== undefined ? Number(judgeRepeatRaw) : (evalConfig?.judgeRepeat ?? 1);
|
|
144
|
-
if (judgeRepeatRaw !== undefined && (!Number.isFinite(parsedJudgeRepeat) || parsedJudgeRepeat < 1)) {
|
|
145
|
-
process.stderr.write(tCli('cli.run.invalid_judge_repeat', lang, { value: judgeRepeatRaw }));
|
|
146
|
-
}
|
|
147
|
-
const judgeRepeatCount = Math.max(1, Math.floor(parsedJudgeRepeat) || 1);
|
|
148
|
-
if (judgeRepeatCount > 1)
|
|
149
|
-
config.judgeRepeat = judgeRepeatCount;
|
|
150
|
-
// --budget-usd / --budget-per-sample-usd / --budget-per-sample-ms:
|
|
151
|
-
// hard budget caps. CLI flags override config-file values. When the
|
|
152
|
-
// total-USD cap is exceeded mid-run, remaining tasks are skipped and a
|
|
153
|
-
// partial report is persisted with meta.budgetExhausted=true.
|
|
154
|
-
const budgetUSD = values['budget-usd'] != null ? Number(values['budget-usd']) : undefined;
|
|
155
|
-
const budgetPerSampleUSD = values['budget-per-sample-usd'] != null ? Number(values['budget-per-sample-usd']) : undefined;
|
|
156
|
-
const budgetPerSampleMs = values['budget-per-sample-ms'] != null ? Number(values['budget-per-sample-ms']) : undefined;
|
|
157
|
-
if (budgetUSD !== undefined || budgetPerSampleUSD !== undefined || budgetPerSampleMs !== undefined) {
|
|
158
|
-
config.budget = {
|
|
159
|
-
...(budgetUSD !== undefined && Number.isFinite(budgetUSD) && budgetUSD >= 0 ? { totalUSD: budgetUSD } : {}),
|
|
160
|
-
...(budgetPerSampleUSD !== undefined && Number.isFinite(budgetPerSampleUSD) && budgetPerSampleUSD >= 0 ? { perSampleUSD: budgetPerSampleUSD } : {}),
|
|
161
|
-
...(budgetPerSampleMs !== undefined && Number.isFinite(budgetPerSampleMs) && budgetPerSampleMs >= 0 ? { perSampleMs: budgetPerSampleMs } : {}),
|
|
162
|
-
};
|
|
163
|
-
}
|
|
164
|
-
// --no-debias-length / eval.yaml `lengthDebias: false`: opt out of length-controlled prompt。
|
|
165
|
-
// Default debias-on (judge prompt v3-cot-length); flip off only to reproduce historical reports。
|
|
166
|
-
// CLI 显式 --no-debias-length > eval.yaml lengthDebias > 默认 true。
|
|
167
|
-
const lengthDebiasOff = values['no-debias-length'] === true
|
|
168
|
-
|| (values['no-debias-length'] === undefined && evalConfig?.lengthDebias === false);
|
|
169
|
-
if (lengthDebiasOff) {
|
|
170
|
-
config.lengthDebias = false;
|
|
171
|
-
process.stderr.write(tCli('cli.run.no_debias_length_active', lang));
|
|
172
|
-
}
|
|
173
|
-
// --bootstrap / --bootstrap-samples: CLI > eval.yaml > default(off / 1000)。
|
|
174
|
-
const bootstrapEnabled = values.bootstrap === true
|
|
175
|
-
|| (values.bootstrap === undefined && evalConfig?.bootstrap === true);
|
|
176
|
-
if (bootstrapEnabled) {
|
|
177
|
-
config.bootstrap = true;
|
|
178
|
-
const bsRaw = values['bootstrap-samples'];
|
|
179
|
-
const parsedBs = bsRaw !== undefined ? Number(bsRaw) : (evalConfig?.bootstrapSamples ?? 1000);
|
|
180
|
-
if (bsRaw !== undefined && (!Number.isFinite(parsedBs) || parsedBs < 100)) {
|
|
181
|
-
process.stderr.write(tCli('cli.run.invalid_bootstrap_samples', lang, { value: bsRaw }));
|
|
182
|
-
}
|
|
183
|
-
const bsCount = Math.max(100, Math.floor(parsedBs) || 1000);
|
|
184
|
-
if (bsCount > 10000) {
|
|
185
|
-
process.stderr.write(tCli('cli.run.bootstrap_samples_too_large', lang, { n: bsCount }));
|
|
186
|
-
}
|
|
187
|
-
config.bootstrapSamples = bsCount;
|
|
188
|
-
}
|
|
189
|
-
// 注入 lang 让 evaluation pipeline 能渲染 doctor 报告(失败时)。
|
|
190
|
-
config.lang = lang;
|
|
191
|
-
if (values['skip-connectivity']) {
|
|
192
|
-
process.stderr.write(tCli('cli.run.skip_connectivity_warning', lang) + '\n');
|
|
193
|
-
}
|
|
194
|
-
try {
|
|
195
|
-
// --batch mode: evaluate each skill independently
|
|
196
|
-
if (values.batch) {
|
|
197
|
-
const { report, filePath } = await runBatchEvaluation({
|
|
198
|
-
...config,
|
|
199
|
-
repeat: repeatCount,
|
|
200
|
-
onSkillProgress({ phase, skill, current, total }) {
|
|
201
|
-
if (phase === 'start') {
|
|
202
|
-
process.stderr.write(tCli('cli.run.skill_section', lang, {
|
|
203
|
-
i: current ?? '', n: total ?? '', skill: skill ?? '',
|
|
204
|
-
}));
|
|
205
|
-
}
|
|
206
|
-
},
|
|
207
|
-
});
|
|
208
|
-
console.log(JSON.stringify(report, null, 2));
|
|
209
|
-
if (filePath) {
|
|
210
|
-
process.stderr.write(tCli('cli.run.batch_complete', lang));
|
|
211
|
-
const tally = computeRunTally(report);
|
|
212
|
-
process.stderr.write(tCli('cli.run.tally', lang, tally));
|
|
213
|
-
process.stderr.write(tCli('cli.run.report_saved', lang, { path: filePath }));
|
|
214
|
-
if (!values['no-serve'] && process.stdout.isTTY) {
|
|
215
|
-
const { createReportServer } = await import('../server/report-server.js');
|
|
216
|
-
const server = createReportServer({ reportsDir: config.outputDir });
|
|
217
|
-
const serverUrl = await server.start();
|
|
218
|
-
const reportUrl = `${serverUrl}/reports/${report.id}`;
|
|
219
|
-
process.stderr.write(tCli('cli.run.report_server_running', lang, { url: serverUrl }));
|
|
220
|
-
process.stderr.write(tCli('cli.run.report_server_view', lang, { url: reportUrl }));
|
|
221
|
-
process.stderr.write(tCli('cli.run.report_server_stop', lang));
|
|
222
|
-
const { platform } = await import('node:os');
|
|
223
|
-
const openCmd = platform() === 'darwin' ? 'open' : platform() === 'win32' ? 'start' : 'xdg-open';
|
|
224
|
-
const { execFile: execFileCb } = await import('node:child_process');
|
|
225
|
-
execFileCb(openCmd, [reportUrl], () => { });
|
|
226
|
-
}
|
|
227
|
-
else if (!values['no-serve']) {
|
|
228
|
-
process.stderr.write(tCli('cli.run.no_serve_in_non_tty', lang));
|
|
229
|
-
process.stderr.write(tCli('cli.run.no_serve_view_hint', lang, { dir: config.outputDir }));
|
|
230
|
-
}
|
|
231
|
-
}
|
|
232
|
-
return;
|
|
233
|
-
}
|
|
234
|
-
let report;
|
|
235
|
-
let filePath;
|
|
236
|
-
if (repeatCount > 1) {
|
|
237
|
-
const result = await runMultiple({
|
|
238
|
-
...config,
|
|
239
|
-
repeat: repeatCount,
|
|
240
|
-
onRepeatProgress({ run, total }) {
|
|
241
|
-
process.stderr.write(tCli('cli.run.run_section', lang, { i: run, n: total }));
|
|
242
|
-
},
|
|
243
|
-
});
|
|
244
|
-
report = result.report;
|
|
245
|
-
filePath = null;
|
|
246
|
-
}
|
|
247
|
-
else {
|
|
248
|
-
const result = (await runEvaluation(config));
|
|
249
|
-
report = result.report;
|
|
250
|
-
filePath = result.filePath;
|
|
251
|
-
}
|
|
252
|
-
// --gold-dir / eval.yaml goldDir: compute α/κ/Pearson against gold annotations and re-persist.
|
|
253
|
-
const goldDir = values['gold-dir'] ?? evalConfig?.goldDir;
|
|
254
|
-
if (goldDir && filePath) {
|
|
255
|
-
const { attachGoldAgreementToReport, formatGoldCompare } = await import('../grading/gold-cli.js');
|
|
256
|
-
const out = attachGoldAgreementToReport({
|
|
257
|
-
report,
|
|
258
|
-
goldDir,
|
|
259
|
-
outputDir: config.outputDir,
|
|
260
|
-
samples: config.bootstrapSamples,
|
|
261
|
-
});
|
|
262
|
-
if (out.result && out.gold) {
|
|
263
|
-
process.stderr.write(formatGoldCompare(out.result, out.gold));
|
|
264
|
-
if (out.result.contaminationWarning) {
|
|
265
|
-
process.stderr.write(tCli('cli.run.contamination_warning', lang, {
|
|
266
|
-
warning: out.result.contaminationWarning,
|
|
267
|
-
}));
|
|
268
|
-
}
|
|
269
|
-
}
|
|
270
|
-
else {
|
|
271
|
-
process.stderr.write(tCli('cli.run.gold_load_failed', lang, { dir: goldDir }));
|
|
272
|
-
for (const m of out.loadIssues) {
|
|
273
|
-
process.stderr.write(tCli('cli.run.gold_load_issue', lang, { message: m }));
|
|
274
|
-
}
|
|
275
|
-
}
|
|
276
|
-
}
|
|
277
|
-
console.log(JSON.stringify(report, null, 2));
|
|
278
|
-
if (filePath) {
|
|
279
|
-
process.stderr.write(tCli('cli.run.eval_complete', lang));
|
|
280
|
-
const tally = computeRunTally(report);
|
|
281
|
-
process.stderr.write(tCli('cli.run.tally', lang, tally));
|
|
282
|
-
process.stderr.write(tCli('cli.run.report_saved', lang, { path: filePath }));
|
|
283
|
-
if (!values['no-serve'] && process.stdout.isTTY) {
|
|
284
|
-
// Auto-start report server
|
|
285
|
-
const { createReportServer } = await import('../server/report-server.js');
|
|
286
|
-
const server = createReportServer({
|
|
287
|
-
reportsDir: config.outputDir,
|
|
288
|
-
});
|
|
289
|
-
const serverUrl = await server.start();
|
|
290
|
-
const reportUrl = `${serverUrl}/reports/${report.id}`;
|
|
291
|
-
process.stderr.write(tCli('cli.run.report_server_running', lang, { url: serverUrl }));
|
|
292
|
-
process.stderr.write(tCli('cli.run.report_server_view', lang, { url: reportUrl }));
|
|
293
|
-
process.stderr.write(tCli('cli.run.report_server_stop', lang));
|
|
294
|
-
// Auto-open report in browser
|
|
295
|
-
const { platform } = await import('node:os');
|
|
296
|
-
const openCmd = platform() === 'darwin' ? 'open' : platform() === 'win32' ? 'start' : 'xdg-open';
|
|
297
|
-
const { execFile: execFileCb } = await import('node:child_process');
|
|
298
|
-
execFileCb(openCmd, [reportUrl], () => { });
|
|
299
|
-
}
|
|
300
|
-
else if (!values['no-serve']) {
|
|
301
|
-
process.stderr.write(tCli('cli.run.no_serve_in_non_tty', lang));
|
|
302
|
-
process.stderr.write(tCli('cli.run.no_serve_view_hint', lang, { dir: config.outputDir }));
|
|
303
|
-
}
|
|
304
|
-
}
|
|
305
|
-
}
|
|
306
|
-
catch (err) {
|
|
307
|
-
console.error(tCli('cli.common.error_prefix', lang, { message: err.message }));
|
|
308
|
-
process.exit(1);
|
|
309
|
-
}
|
|
310
|
-
}
|
|
311
|
-
// ---------------------------------------------------------------------------
|
|
312
|
-
// handleReport
|
|
313
|
-
// ---------------------------------------------------------------------------
|
|
314
|
-
async function handleReport(argv) {
|
|
315
|
-
const lang = langFromArgv(argv);
|
|
316
|
-
if (argv.includes('--help') || argv.includes('-h')) {
|
|
317
|
-
console.log(tCli('cli.help.main', lang).trim());
|
|
318
|
-
process.exit(0);
|
|
319
|
-
}
|
|
320
|
-
const { values } = parseArgsStrictOrExit({
|
|
321
|
-
args: argv,
|
|
322
|
-
options: {
|
|
323
|
-
...COMMON_OPTIONS,
|
|
324
|
-
port: { type: 'string', default: '7799' },
|
|
325
|
-
'reports-dir': { type: 'string', default: DEFAULT_REPORTS_DIR },
|
|
326
|
-
export: { type: 'string' },
|
|
327
|
-
dev: { type: 'boolean', default: false },
|
|
328
|
-
},
|
|
329
|
-
});
|
|
330
|
-
// Dev mode: restart server on file changes via node --watch
|
|
331
|
-
if (values.dev && !process.env.__OMK_DEV_CHILD) {
|
|
332
|
-
const { spawn } = await import('node:child_process');
|
|
333
|
-
const { fileURLToPath } = await import('node:url');
|
|
334
|
-
const cliPath = fileURLToPath(import.meta.url);
|
|
335
|
-
const libDir = resolve(cliPath, '..', 'lib');
|
|
336
|
-
const args = [
|
|
337
|
-
'--watch-path', libDir, cliPath, 'bench', 'report',
|
|
338
|
-
'--port', values.port,
|
|
339
|
-
'--reports-dir', values['reports-dir'],
|
|
340
|
-
];
|
|
341
|
-
const child = spawn(process.execPath, args, {
|
|
342
|
-
stdio: 'inherit',
|
|
343
|
-
env: { ...process.env, __OMK_DEV_CHILD: '1' },
|
|
344
|
-
});
|
|
345
|
-
child.on('exit', (code) => process.exit(code || 0));
|
|
346
|
-
return;
|
|
347
|
-
}
|
|
348
|
-
if (values.export) {
|
|
349
|
-
const { createFileStore } = await import('../server/report-store.js');
|
|
350
|
-
const { renderReportDocumentDetail } = await import('../renderer/html-renderer.js');
|
|
351
|
-
const { writeFileSync } = await import('node:fs');
|
|
352
|
-
const store = createFileStore(resolve(values['reports-dir']));
|
|
353
|
-
const report = await store.get(values.export);
|
|
354
|
-
if (!report) {
|
|
355
|
-
console.error(tCli('cli.common.report_not_found', lang, { id: values.export }));
|
|
356
|
-
process.exit(1);
|
|
357
|
-
}
|
|
358
|
-
const html = renderReportDocumentDetail(report);
|
|
359
|
-
const outPath = resolve(`${values.export}.html`);
|
|
360
|
-
writeFileSync(outPath, html);
|
|
361
|
-
console.log(`Exported to: ${outPath}`);
|
|
362
|
-
console.log('Open in browser, or Ctrl+P to save as PDF');
|
|
363
|
-
return;
|
|
364
|
-
}
|
|
365
|
-
const { createReportServer } = await import('../server/report-server.js');
|
|
366
|
-
const server = createReportServer({
|
|
367
|
-
port: Number(values.port),
|
|
368
|
-
reportsDir: resolve(values['reports-dir']),
|
|
369
|
-
});
|
|
370
|
-
const url = await server.start();
|
|
371
|
-
console.log(`Report server running at ${url}`);
|
|
372
|
-
console.log('Press Ctrl+C to stop');
|
|
373
|
-
}
|
|
374
|
-
// ---------------------------------------------------------------------------
|
|
375
|
-
// handleInit
|
|
376
|
-
// ---------------------------------------------------------------------------
|
|
377
|
-
const INIT_SAMPLES = `[
|
|
378
|
-
{
|
|
379
|
-
"sample_id": "s001",
|
|
380
|
-
"prompt": "审查以下代码",
|
|
381
|
-
"context": "function authenticate(username, password) {\\n const query = \`SELECT * FROM users WHERE name='\${username}' AND pass='\${password}'\`;\\n return db.execute(query);\\n}",
|
|
382
|
-
"rubric": "应识别 SQL 注入风险,建议使用参数化查询",
|
|
383
|
-
"assertions": [
|
|
384
|
-
{ "type": "contains", "value": "SQL", "weight": 1 },
|
|
385
|
-
{ "type": "contains", "value": "注入", "weight": 1 },
|
|
386
|
-
{ "type": "contains", "value": "参数化", "weight": 0.5 },
|
|
387
|
-
{ "type": "not_contains", "value": "没有问题", "weight": 0.5 }
|
|
388
|
-
],
|
|
389
|
-
"dimensions": {
|
|
390
|
-
"security": "是否准确识别出 SQL 注入漏洞并说明其危害",
|
|
391
|
-
"actionability": "是否给出可直接使用的参数化查询修复代码"
|
|
392
|
-
}
|
|
393
|
-
},
|
|
394
|
-
{
|
|
395
|
-
"sample_id": "s002",
|
|
396
|
-
"prompt": "审查以下代码",
|
|
397
|
-
"context": "async function fetchData(url) {\\n const res = await fetch(url);\\n const data = await res.json();\\n return data;\\n}",
|
|
398
|
-
"rubric": "应指出缺少错误处理(网络异常、非 JSON 响应、HTTP 错误状态码)",
|
|
399
|
-
"assertions": [
|
|
400
|
-
{ "type": "contains", "value": "错误处理", "weight": 1 },
|
|
401
|
-
{ "type": "regex", "pattern": "try[\\\\s\\\\S]*catch|错误|异常|error", "flags": "i", "weight": 1 },
|
|
402
|
-
{ "type": "contains", "value": "status", "weight": 0.5 }
|
|
403
|
-
],
|
|
404
|
-
"dimensions": {
|
|
405
|
-
"robustness": "是否指出了所有缺失的错误处理场景",
|
|
406
|
-
"actionability": "是否给出了完整的 try-catch 修复代码"
|
|
407
|
-
}
|
|
408
|
-
},
|
|
409
|
-
{
|
|
410
|
-
"sample_id": "s003",
|
|
411
|
-
"prompt": "审查以下代码",
|
|
412
|
-
"context": "function renderComment(comment) {\\n document.getElementById('output').innerHTML = '<p>' + comment + '</p>';\\n}",
|
|
413
|
-
"rubric": "应识别 XSS 风险,建议使用 textContent 或转义 HTML",
|
|
414
|
-
"assertions": [
|
|
415
|
-
{ "type": "contains", "value": "XSS", "weight": 1 },
|
|
416
|
-
{ "type": "regex", "pattern": "textContent|转义|escape|sanitize", "flags": "i", "weight": 1 },
|
|
417
|
-
{ "type": "contains", "value": "innerHTML", "weight": 0.5 }
|
|
418
|
-
],
|
|
419
|
-
"dimensions": {
|
|
420
|
-
"security": "是否准确识别出 XSS 漏洞并说明攻击方式",
|
|
421
|
-
"actionability": "是否给出使用 textContent 或转义的修复代码"
|
|
422
|
-
}
|
|
423
|
-
}
|
|
424
|
-
]
|
|
425
|
-
`;
|
|
426
|
-
// 模板带 Claude Code SKILL.md 兼容 frontmatter(name + description),让用户
|
|
427
|
-
// 可以把 init 出来的 SKILL.md 直接 deploy 到 ~/.claude/skills/ 给 Claude Code 用,
|
|
428
|
-
// 一份文件双向 dogfood(omk 评测 + Claude 部署)。omk 当前不 strip frontmatter,
|
|
429
|
-
// 它会跟着 leak 进 system prompt — 在 model 行为层面是无害噪声,跨 executor 一致。
|
|
430
|
-
const INIT_SKILL_V1 = `---
|
|
431
|
-
name: code-review-v1
|
|
432
|
-
description: 简单代码审查 skill,识别明显问题
|
|
433
|
-
---
|
|
434
|
-
|
|
435
|
-
# Code review v1
|
|
436
|
-
|
|
437
|
-
你是一个代码审查助手。请审查用户提供的代码,指出潜在问题。
|
|
438
|
-
`;
|
|
439
|
-
const INIT_SKILL_V2 = `---
|
|
440
|
-
name: code-review-v2
|
|
441
|
-
description: 多维度代码审查,覆盖安全 / 健壮 / 可维护 / 性能,带严重程度标注
|
|
442
|
-
---
|
|
443
|
-
|
|
444
|
-
# Code review v2
|
|
445
|
-
|
|
446
|
-
你是一个高级代码审查专家。请从以下维度审查用户提供的代码:
|
|
447
|
-
|
|
448
|
-
1. 安全性:是否存在注入、XSS、敏感信息泄露等风险
|
|
449
|
-
2. 健壮性:是否有适当的错误处理和边界检查
|
|
450
|
-
3. 可维护性:命名是否清晰、结构是否合理
|
|
451
|
-
4. 性能:是否存在明显的性能瓶颈
|
|
452
|
-
|
|
453
|
-
对每个维度给出具体的改进建议,并标注严重程度(高/中/低)。
|
|
454
|
-
`;
|
|
455
|
-
// ---------------------------------------------------------------------------
|
|
456
|
-
// handleAnalyze (v0.18 skill 健康度日报)
|
|
457
|
-
// ---------------------------------------------------------------------------
|
|
458
|
-
function parseLastWindow(spec) {
|
|
459
|
-
// "7d" / "24h" / "30m" → ISO timestamp (from = now - spec)
|
|
460
|
-
const m = /^(\d+)([dhm])$/.exec(spec);
|
|
461
|
-
if (!m)
|
|
462
|
-
return null;
|
|
463
|
-
const n = Number(m[1]);
|
|
464
|
-
const unit = m[2];
|
|
465
|
-
const ms = unit === 'd' ? n * 86400_000 : unit === 'h' ? n * 3600_000 : n * 60_000;
|
|
466
|
-
return new Date(Date.now() - ms).toISOString();
|
|
467
|
-
}
|
|
468
|
-
async function handleDoctor(argv) {
|
|
469
|
-
const lang = langFromArgv(argv);
|
|
470
|
-
if (argv.includes('--help') || argv.includes('-h')) {
|
|
471
|
-
console.log(tCli('cli.help.doctor_usage', lang));
|
|
472
|
-
process.exit(0);
|
|
473
|
-
}
|
|
474
|
-
const { values, positionals } = parseArgsStrictOrExit({
|
|
475
|
-
args: argv,
|
|
476
|
-
allowPositionals: true,
|
|
477
|
-
options: {
|
|
478
|
-
...COMMON_OPTIONS,
|
|
479
|
-
json: { type: 'boolean', default: false },
|
|
480
|
-
gate: { type: 'boolean', default: false },
|
|
481
|
-
executor: { type: 'string' },
|
|
482
|
-
model: { type: 'string' },
|
|
483
|
-
timeout: { type: 'string' },
|
|
484
|
-
},
|
|
485
|
-
});
|
|
486
|
-
const target = positionals[0] ?? null;
|
|
487
|
-
const executorName = values.executor ?? 'claude';
|
|
488
|
-
const model = values.model ?? 'sonnet';
|
|
489
|
-
const timeoutRaw = values.timeout;
|
|
490
|
-
const timeoutSec = timeoutRaw != null ? Number(timeoutRaw) : 8;
|
|
491
|
-
const timeoutMs = Math.max(1000, Math.floor((Number.isFinite(timeoutSec) ? timeoutSec : 8) * 1000));
|
|
492
|
-
const cwd = process.cwd();
|
|
493
|
-
const { runDoctor } = await import('../doctor/index.js');
|
|
494
|
-
const { renderDoctorReportText, renderDoctorReportJson } = await import('../doctor/renderer.js');
|
|
495
|
-
let report;
|
|
496
|
-
try {
|
|
497
|
-
report = await runDoctor({
|
|
498
|
-
target,
|
|
499
|
-
cwd,
|
|
500
|
-
executorName,
|
|
501
|
-
model,
|
|
502
|
-
timeoutMs,
|
|
503
|
-
lang,
|
|
504
|
-
});
|
|
505
|
-
}
|
|
506
|
-
catch (err) {
|
|
507
|
-
const msg = err instanceof Error ? err.message : String(err);
|
|
508
|
-
console.error(tCli('cli.doctor.no_skill_found', lang, { path: target ?? cwd }));
|
|
509
|
-
console.error(`(${msg})`);
|
|
510
|
-
process.exit(1);
|
|
511
|
-
}
|
|
512
|
-
if (report.skills.length === 0) {
|
|
513
|
-
console.error(tCli('cli.doctor.no_skill_found', lang, { path: target ?? cwd }));
|
|
514
|
-
process.exit(1);
|
|
515
|
-
}
|
|
516
|
-
const isJson = values.json;
|
|
517
|
-
const isGate = values.gate;
|
|
518
|
-
if (isJson) {
|
|
519
|
-
console.log(renderDoctorReportJson(report));
|
|
520
|
-
}
|
|
521
|
-
else if (isGate) {
|
|
522
|
-
// gate 模式: 静默 stdout, fail 时简短 stderr 摘要(供 CI 抓 exit code)
|
|
523
|
-
if (report.outcome === 'failed') {
|
|
524
|
-
const summary = lang === 'zh'
|
|
525
|
-
? `doctor failed: ${report.totals.fail} 个 skill 未通过 (${report.totals.warn} warn / ${report.totals.pass} pass)`
|
|
526
|
-
: `doctor failed: ${report.totals.fail} skills did not pass (${report.totals.warn} warn / ${report.totals.pass} pass)`;
|
|
527
|
-
console.error(summary);
|
|
528
|
-
}
|
|
529
|
-
}
|
|
530
|
-
else {
|
|
531
|
-
renderDoctorReportText(report, lang);
|
|
532
|
-
}
|
|
533
|
-
process.exit(report.outcome === 'failed' ? 1 : 0);
|
|
534
|
-
}
|
|
535
|
-
async function handleAnalyze(argv) {
|
|
536
|
-
const lang = langFromArgv(argv);
|
|
537
|
-
const { values: rawValues, positionals } = parseArgsStrictOrExit({
|
|
538
|
-
args: argv,
|
|
539
|
-
allowPositionals: true,
|
|
540
|
-
options: {
|
|
541
|
-
...COMMON_OPTIONS,
|
|
542
|
-
kb: { type: 'string' },
|
|
543
|
-
last: { type: 'string' },
|
|
544
|
-
from: { type: 'string' },
|
|
545
|
-
to: { type: 'string' },
|
|
546
|
-
skills: { type: 'string' },
|
|
547
|
-
'output-dir': { type: 'string' },
|
|
548
|
-
},
|
|
549
|
-
});
|
|
550
|
-
// 该 handler options 全是 string-typed (无 boolean), 收紧 cast 让 caller 直接 use values.xxx 当 string 用。
|
|
551
|
-
const values = rawValues;
|
|
552
|
-
const dir = positionals[0];
|
|
553
|
-
if (!dir) {
|
|
554
|
-
console.error(tCli('cli.help.analyze_usage', lang));
|
|
555
|
-
process.exit(1);
|
|
556
|
-
}
|
|
557
|
-
const tracePath = resolve(dir);
|
|
558
|
-
const { existsSync, mkdirSync, writeFileSync } = await import('node:fs');
|
|
559
|
-
if (!existsSync(tracePath)) {
|
|
560
|
-
console.error(`Trace path does not exist: ${tracePath}`);
|
|
561
|
-
process.exit(1);
|
|
562
|
-
}
|
|
563
|
-
// 时间窗: --from/--to 优先, --last fallback
|
|
564
|
-
let from = values.from;
|
|
565
|
-
if (!from && values.last) {
|
|
566
|
-
const inferred = parseLastWindow(values.last);
|
|
567
|
-
if (!inferred) {
|
|
568
|
-
console.error(`Invalid --last format: "${values.last}". Expected e.g. "7d" / "24h" / "30m".`);
|
|
569
|
-
process.exit(1);
|
|
570
|
-
}
|
|
571
|
-
from = inferred;
|
|
572
|
-
}
|
|
573
|
-
const to = values.to;
|
|
574
|
-
const skills = values.skills ? values.skills.split(',').map((s) => s.trim()).filter(Boolean) : undefined;
|
|
575
|
-
console.log(`[omk] analyzing ${tracePath}...`);
|
|
576
|
-
const { computeSkillHealthReport } = await import('../observability/skill-health-analyzer.js');
|
|
577
|
-
const report = computeSkillHealthReport(tracePath, {
|
|
578
|
-
kbRoot: values.kb ? resolve(values.kb) : undefined,
|
|
579
|
-
from,
|
|
580
|
-
to,
|
|
581
|
-
skills,
|
|
582
|
-
});
|
|
583
|
-
// JSON 是主产物; HTML 由 report server 的 /analyses/:id 按需渲染 (和 bench run 一致)
|
|
584
|
-
const outDir = resolve(values['output-dir'] || join(process.env.HOME || '.', '.oh-my-knowledge', 'analyses'));
|
|
585
|
-
mkdirSync(outDir, { recursive: true });
|
|
586
|
-
const timestamp = new Date().toISOString().replace(/[:.]/g, '-').slice(0, 19);
|
|
587
|
-
const jsonPath = join(outDir, `${timestamp}-skill-health.json`);
|
|
588
|
-
writeFileSync(jsonPath, JSON.stringify(report, null, 2));
|
|
589
|
-
// 控制台摘要
|
|
590
|
-
const { sessionCount, segmentCount, toolCallCount, toolFailureRate } = report.meta;
|
|
591
|
-
console.log('');
|
|
592
|
-
console.log(`sessions: ${sessionCount} · segments: ${segmentCount} · tool calls: ${toolCallCount} · fail rate: ${(toolFailureRate * 100).toFixed(1)}%`);
|
|
593
|
-
console.log(`overall: gapRate ${(report.overall.gapRate * 100).toFixed(1)}% · weightedGapRate ${(report.overall.weightedGapRate * 100).toFixed(1)}% · health: ${report.overall.healthBand}`);
|
|
594
|
-
console.log('');
|
|
595
|
-
const skillRows = Object.values(report.bySkill)
|
|
596
|
-
.sort((a, b) => b.segmentCount - a.segmentCount)
|
|
597
|
-
.slice(0, 10)
|
|
598
|
-
.map((s) => ` ${s.skillName.padEnd(24)} segs=${String(s.segmentCount).padStart(4)} gapRate=${String(Math.round(s.gap.gapRate * 100) + '%').padStart(4)} weighted=${String(Math.round(s.gap.weightedGapRate * 100) + '%').padStart(4)}${s.coverage ? ` cov=${Math.round(s.coverage.fileCoverageRate * 100)}%` : ''}`);
|
|
599
|
-
console.log('top skills:');
|
|
600
|
-
console.log(skillRows.join('\n'));
|
|
601
|
-
console.log('');
|
|
602
|
-
console.log(`report written to: ${jsonPath}`);
|
|
603
|
-
console.log(tCli('cli.analyze.view_in_browser', lang));
|
|
604
|
-
}
|
|
605
|
-
async function handleInit(argv) {
|
|
606
|
-
const lang = langFromArgv(argv);
|
|
607
|
-
// 走 helper 让未知 option fail-fast (e.g. `omk bench init --bogus`),
|
|
608
|
-
// 否则 argv[0] 直接当目录名, --bogus / --lang 都会被当成 dir 写文件。
|
|
609
|
-
const { positionals } = parseArgsStrictOrExit({
|
|
610
|
-
args: argv,
|
|
611
|
-
allowPositionals: true,
|
|
612
|
-
options: { ...COMMON_OPTIONS },
|
|
613
|
-
});
|
|
614
|
-
const targetDir = resolve(positionals[0] || '.');
|
|
615
|
-
const { writeFileSync, mkdirSync } = await import('node:fs');
|
|
616
|
-
// omk skill loader 把 `skills/<name>/SKILL.md` 子目录识别为 directory-skill,
|
|
617
|
-
// cwd 默认锚到 skill 根目录,后续可在同目录下放 assets / 子文档。
|
|
618
|
-
// 子目录主题化命名(code-review-v1 / code-review-v2)比泛 v1.md / v2.md 心智模型更清晰。
|
|
619
|
-
mkdirSync(join(targetDir, 'skills', 'code-review-v1'), { recursive: true });
|
|
620
|
-
mkdirSync(join(targetDir, 'skills', 'code-review-v2'), { recursive: true });
|
|
621
|
-
writeFileSync(join(targetDir, 'eval-samples.json'), INIT_SAMPLES);
|
|
622
|
-
writeFileSync(join(targetDir, 'skills', 'code-review-v1', 'SKILL.md'), INIT_SKILL_V1);
|
|
623
|
-
writeFileSync(join(targetDir, 'skills', 'code-review-v2', 'SKILL.md'), INIT_SKILL_V2);
|
|
624
|
-
console.log(tCli('cli.init.scaffolded', lang, { dir: targetDir }));
|
|
625
|
-
console.log('');
|
|
626
|
-
console.log(tCli('cli.init.next_steps_title', lang));
|
|
627
|
-
console.log(tCli('cli.init.next_step_edit_samples', lang));
|
|
628
|
-
console.log(tCli('cli.init.next_step_edit_skills', lang));
|
|
629
|
-
console.log(tCli('cli.init.next_step_run', lang));
|
|
630
|
-
console.log(tCli('cli.init.note_codex_executor', lang));
|
|
631
|
-
}
|
|
632
|
-
// ---------------------------------------------------------------------------
|
|
633
|
-
// handleGenSamples
|
|
634
|
-
// ---------------------------------------------------------------------------
|
|
635
|
-
async function handleGenSamples(argv) {
|
|
636
|
-
const lang = langFromArgv(argv);
|
|
637
|
-
if (argv.includes('--help') || argv.includes('-h')) {
|
|
638
|
-
console.log(tCli('cli.help.main', lang).trim());
|
|
639
|
-
process.exit(0);
|
|
640
|
-
}
|
|
641
|
-
const { values } = parseArgsStrictOrExit({
|
|
642
|
-
args: argv,
|
|
643
|
-
options: {
|
|
644
|
-
...COMMON_OPTIONS,
|
|
645
|
-
batch: { type: 'boolean', default: false },
|
|
646
|
-
count: { type: 'string', default: '5' },
|
|
647
|
-
model: { type: 'string', default: 'sonnet' },
|
|
648
|
-
'skill-dir': { type: 'string', default: 'skills' },
|
|
649
|
-
},
|
|
650
|
-
allowPositionals: true,
|
|
651
|
-
});
|
|
652
|
-
const { generateSamples } = await import('../authoring/generator.js');
|
|
653
|
-
const { readFileSync, writeFileSync } = await import('node:fs');
|
|
654
|
-
const count = Math.max(1, Number(values.count) || 5);
|
|
655
|
-
const model = values.model;
|
|
656
|
-
if (values.batch) {
|
|
657
|
-
// Batch mode: generate for all skills missing eval-samples
|
|
658
|
-
const skillDir = resolve(values['skill-dir']);
|
|
659
|
-
if (!existsSync(skillDir)) {
|
|
660
|
-
console.error(tCli('cli.common.skill_dir_not_found', lang, { path: skillDir }));
|
|
661
|
-
process.exit(1);
|
|
662
|
-
}
|
|
663
|
-
const { readdirSync, statSync } = await import('node:fs');
|
|
664
|
-
const entries = readdirSync(skillDir);
|
|
665
|
-
let generated = 0;
|
|
666
|
-
for (const entry of entries) {
|
|
667
|
-
let name;
|
|
668
|
-
let skillPath;
|
|
669
|
-
let samplesPath;
|
|
670
|
-
const fullPath = join(skillDir, entry);
|
|
671
|
-
if (entry.endsWith('.md') && !entry.endsWith('.eval-samples.json')) {
|
|
672
|
-
name = entry.slice(0, -3);
|
|
673
|
-
skillPath = fullPath;
|
|
674
|
-
samplesPath = join(skillDir, `${name}.eval-samples.json`);
|
|
675
|
-
}
|
|
676
|
-
else if (statSync(fullPath).isDirectory()) {
|
|
677
|
-
const skillMd = join(fullPath, 'SKILL.md');
|
|
678
|
-
if (!existsSync(skillMd))
|
|
679
|
-
continue;
|
|
680
|
-
name = entry;
|
|
681
|
-
skillPath = skillMd;
|
|
682
|
-
samplesPath = join(fullPath, 'eval-samples.json');
|
|
683
|
-
}
|
|
684
|
-
else {
|
|
685
|
-
continue;
|
|
686
|
-
}
|
|
687
|
-
if (existsSync(samplesPath)) {
|
|
688
|
-
process.stderr.write(tCli('cli.gen.skill_skipped_existing', lang, { name }));
|
|
689
|
-
continue;
|
|
690
|
-
}
|
|
691
|
-
process.stderr.write(tCli('cli.gen.skill_generating', lang, { name, count }));
|
|
692
|
-
try {
|
|
693
|
-
const skillContent = readFileSync(skillPath, 'utf-8');
|
|
694
|
-
const { samples, costUSD } = await generateSamples({ skillContent, count, model });
|
|
695
|
-
writeFileSync(samplesPath, JSON.stringify(samples, null, 2));
|
|
696
|
-
const cost = costUSD > 0 ? ` $${costUSD.toFixed(4)}` : '';
|
|
697
|
-
process.stderr.write(tCli('cli.gen.skill_done', lang, {
|
|
698
|
-
name, n: samples.length, path: samplesPath, cost,
|
|
699
|
-
}));
|
|
700
|
-
generated++;
|
|
701
|
-
}
|
|
702
|
-
catch (err) {
|
|
703
|
-
process.stderr.write(tCli('cli.gen.skill_failed', lang, {
|
|
704
|
-
name, message: err.message,
|
|
705
|
-
}));
|
|
706
|
-
}
|
|
707
|
-
}
|
|
708
|
-
if (generated === 0) {
|
|
709
|
-
console.log(tCli('cli.gen.batch_none_needed', lang));
|
|
710
|
-
}
|
|
711
|
-
else {
|
|
712
|
-
console.log(tCli('cli.gen.batch_summary', lang, { n: generated }));
|
|
713
|
-
}
|
|
714
|
-
}
|
|
715
|
-
else {
|
|
716
|
-
// Single skill mode
|
|
717
|
-
const skillPath = argv.find((a) => !a.startsWith('-'));
|
|
718
|
-
if (!skillPath) {
|
|
719
|
-
console.error(tCli('cli.gen.specify_skill_path', lang));
|
|
720
|
-
process.exit(1);
|
|
721
|
-
}
|
|
722
|
-
const resolvedPath = resolve(skillPath);
|
|
723
|
-
if (!existsSync(resolvedPath)) {
|
|
724
|
-
console.error(tCli('cli.common.skill_file_not_found', lang, { path: resolvedPath }));
|
|
725
|
-
process.exit(1);
|
|
726
|
-
}
|
|
727
|
-
const skillContent = readFileSync(resolvedPath, 'utf-8');
|
|
728
|
-
const outputPath = resolve('eval-samples.json');
|
|
729
|
-
if (existsSync(outputPath)) {
|
|
730
|
-
console.error(tCli('cli.gen.samples_already_exists', lang));
|
|
731
|
-
process.exit(1);
|
|
732
|
-
}
|
|
733
|
-
process.stderr.write(tCli('cli.gen.single_generating', lang, { count }));
|
|
734
|
-
try {
|
|
735
|
-
const { samples, costUSD } = await generateSamples({ skillContent, count, model });
|
|
736
|
-
writeFileSync(outputPath, JSON.stringify(samples, null, 2));
|
|
737
|
-
const cost = costUSD > 0 ? ` $${costUSD.toFixed(4)}` : '';
|
|
738
|
-
process.stderr.write(tCli('cli.gen.single_done', lang, {
|
|
739
|
-
n: samples.length, path: outputPath, cost,
|
|
740
|
-
}));
|
|
741
|
-
console.log(tCli('cli.gen.review_hint', lang));
|
|
742
|
-
}
|
|
743
|
-
catch (err) {
|
|
744
|
-
console.error(tCli('cli.gen.failed', lang, { message: err.message }));
|
|
745
|
-
process.exit(1);
|
|
746
|
-
}
|
|
747
|
-
}
|
|
748
|
-
}
|
|
749
|
-
// ---------------------------------------------------------------------------
|
|
750
|
-
// handleEvolve
|
|
751
|
-
// ---------------------------------------------------------------------------
|
|
752
|
-
async function handleEvolve(argv) {
|
|
753
|
-
const lang = langFromArgv(argv);
|
|
754
|
-
const { values, positionals } = parseArgsStrictOrExit({
|
|
755
|
-
args: argv,
|
|
756
|
-
options: {
|
|
757
|
-
...COMMON_OPTIONS,
|
|
758
|
-
rounds: { type: 'string', default: '5' },
|
|
759
|
-
target: { type: 'string' },
|
|
760
|
-
samples: { type: 'string', default: 'eval-samples.json' },
|
|
761
|
-
model: { type: 'string', default: 'sonnet' },
|
|
762
|
-
'judge-models': { type: 'string', default: 'claude:haiku' },
|
|
763
|
-
'improve-model': { type: 'string', default: 'sonnet' },
|
|
764
|
-
concurrency: { type: 'string', default: '1' },
|
|
765
|
-
timeout: { type: 'string', default: '120' },
|
|
766
|
-
executor: { type: 'string', default: 'claude' },
|
|
767
|
-
'skip-connectivity': { type: 'boolean', default: false },
|
|
768
|
-
},
|
|
769
|
-
allowPositionals: true,
|
|
770
|
-
});
|
|
771
|
-
// skill path 走 parseArgs 的 positionals (避免 raw argv.find 把 flag value
|
|
772
|
-
// 当成 path 误识别 — 例如 `evolve --judge-models openai-api:gpt-4o foo.md`)。
|
|
773
|
-
const skillPath = positionals[0];
|
|
774
|
-
if (!skillPath) {
|
|
775
|
-
console.error(tCli('cli.evolve.specify_skill_path', lang));
|
|
776
|
-
process.exit(1);
|
|
777
|
-
}
|
|
778
|
-
let samplesFile = values.samples ?? 'eval-samples.json';
|
|
779
|
-
if (samplesFile === 'eval-samples.json' && !existsSync(resolve(samplesFile))) {
|
|
780
|
-
if (existsSync(resolve('eval-samples.yaml')))
|
|
781
|
-
samplesFile = 'eval-samples.yaml';
|
|
782
|
-
else if (existsSync(resolve('eval-samples.yml')))
|
|
783
|
-
samplesFile = 'eval-samples.yml';
|
|
784
|
-
}
|
|
785
|
-
const { evolveSkill } = await import('../authoring/evolver.js');
|
|
786
|
-
const { parseJudgeModelsArgOrExit } = await import('./parse-run-config.js');
|
|
787
|
-
const evolveJudges = parseJudgeModelsArgOrExit(values['judge-models']);
|
|
788
|
-
if (evolveJudges.length > 1) {
|
|
789
|
-
console.error(tCli('cli.common.judge_models_single_only', lang, { cmd: 'evolve' }));
|
|
790
|
-
process.exit(2);
|
|
791
|
-
}
|
|
792
|
-
process.stderr.write(tCli('cli.evolve.section_header', lang, { path: skillPath }));
|
|
793
|
-
try {
|
|
794
|
-
const result = await evolveSkill({
|
|
795
|
-
skillPath: resolve(skillPath),
|
|
796
|
-
samplesPath: resolve(samplesFile),
|
|
797
|
-
rounds: Math.max(1, Number(values.rounds) || 5),
|
|
798
|
-
target: values.target ? Number(values.target) : null,
|
|
799
|
-
model: values.model,
|
|
800
|
-
judgeModels: evolveJudges,
|
|
801
|
-
improveModel: values['improve-model'],
|
|
802
|
-
executorName: values.executor,
|
|
803
|
-
concurrency: Math.max(1, Number(values.concurrency) || 1),
|
|
804
|
-
timeoutMs: Math.max(1, Number(values.timeout) || 120) * 1000,
|
|
805
|
-
skipConnectivity: values['skip-connectivity'],
|
|
806
|
-
onProgress: makeOnProgress(lang),
|
|
807
|
-
onRoundProgress({ round, totalRounds: _totalRounds, phase, score, delta, accepted, costUSD, costReported, error }) {
|
|
808
|
-
// costReported=false 时显示「—」而不是 $0.0000(executor 不报 cost,如 codex)。
|
|
809
|
-
// 缺位 / true 当 reported 走旧格式。
|
|
810
|
-
const fmtRoundCost = (c, r) => r ? `$${c.toFixed(4)}` : '—';
|
|
811
|
-
if (phase === 'baseline') {
|
|
812
|
-
process.stderr.write(tCli('cli.evolve.round_baseline', lang, {
|
|
813
|
-
score: score.toFixed(2), cost: fmtRoundCost(costUSD, costReported !== false),
|
|
814
|
-
}));
|
|
815
|
-
}
|
|
816
|
-
else if (phase === 'error') {
|
|
817
|
-
process.stderr.write(tCli('cli.evolve.round_error', lang, {
|
|
818
|
-
round, error: String(error ?? ''),
|
|
819
|
-
}));
|
|
820
|
-
}
|
|
821
|
-
else if (phase === 'done') {
|
|
822
|
-
const delta_ = delta >= 0 ? `+${delta.toFixed(2)}` : delta.toFixed(2);
|
|
823
|
-
const status = accepted ? '✓ ACCEPT' : '✗ REJECT';
|
|
824
|
-
process.stderr.write(tCli('cli.evolve.round_done', lang, {
|
|
825
|
-
round, score: score.toFixed(2), delta: delta_, status, cost: fmtRoundCost(costUSD, costReported !== false),
|
|
826
|
-
}));
|
|
827
|
-
}
|
|
828
|
-
},
|
|
829
|
-
});
|
|
830
|
-
const improvement = result.startScore > 0
|
|
831
|
-
? ((result.finalScore - result.startScore) / result.startScore * 100).toFixed(1)
|
|
832
|
-
: '0';
|
|
833
|
-
const totalCostStr = result.costReported === false
|
|
834
|
-
? '—' // 任一轮的 executor 不报 cost → totalCostUSD 是 lower-bound
|
|
835
|
-
: `$${result.totalCostUSD.toFixed(4)}`;
|
|
836
|
-
process.stderr.write(tCli('cli.evolve.summary', lang, {
|
|
837
|
-
start: result.startScore.toFixed(2), final: result.finalScore.toFixed(2),
|
|
838
|
-
percent: improvement, rounds: result.totalRounds, cost: totalCostStr,
|
|
839
|
-
}));
|
|
840
|
-
process.stderr.write(tCli('cli.evolve.best_path', lang, {
|
|
841
|
-
best: result.bestSkillPath, target: resolve(skillPath),
|
|
842
|
-
}));
|
|
843
|
-
process.stderr.write(tCli('cli.evolve.versions_saved', lang, {
|
|
844
|
-
dir: join(resolve(skillPath, '..'), 'evolve'),
|
|
845
|
-
}));
|
|
846
|
-
if (result.reportId) {
|
|
847
|
-
process.stderr.write(tCli('cli.evolve.report_link', lang, { id: result.reportId }));
|
|
848
|
-
}
|
|
849
|
-
console.log(JSON.stringify(result, null, 2));
|
|
850
|
-
}
|
|
851
|
-
catch (err) {
|
|
852
|
-
console.error(tCli('cli.common.error_prefix', lang, { message: err.message }));
|
|
853
|
-
process.exit(1);
|
|
854
|
-
}
|
|
855
|
-
}
|
|
856
|
-
// ---------------------------------------------------------------------------
|
|
857
|
-
// handleGate — 跑评测 + 应用 gate, exit code 0/1 适合 CI/CD pipeline 调用。
|
|
858
|
-
// 内部 = runEvaluation + computeVerdict + formatVerdictText, 与 bench verdict
|
|
859
|
-
// 共用决策内核(只是 verdict 读已有报告, gate 跑完再判)。
|
|
860
|
-
// ---------------------------------------------------------------------------
|
|
861
|
-
async function handleGate(argv) {
|
|
862
|
-
const lang = langFromArgv(argv);
|
|
863
|
-
if (argv[0] === '--help' || argv[0] === '-h') {
|
|
864
|
-
console.log(tCli('cli.help.main', lang).trim());
|
|
865
|
-
process.exit(0);
|
|
866
|
-
}
|
|
867
|
-
const { values, config } = parseRunConfig(argv, {
|
|
868
|
-
threshold: { type: 'string', default: '3.5' },
|
|
869
|
-
'trivial-diff': { type: 'string' },
|
|
870
|
-
});
|
|
871
|
-
const { runEvaluation } = await import('../eval-workflows/run-evaluation.js');
|
|
872
|
-
config.onProgress = makeOnProgress(lang);
|
|
873
|
-
// 注入 lang + skip-connectivity warning(若 flag set);doctor 由 evaluation 强制调, 无 skip 选项。
|
|
874
|
-
config.lang = lang;
|
|
875
|
-
if (values['skip-connectivity']) {
|
|
876
|
-
process.stderr.write(tCli('cli.run.skip_connectivity_warning', lang) + '\n');
|
|
877
|
-
}
|
|
878
|
-
try {
|
|
879
|
-
const { report: document } = (await runEvaluation(config));
|
|
880
|
-
if (document.dryRun) {
|
|
881
|
-
console.log('Gate dry-run: no scores to check');
|
|
882
|
-
process.exit(0);
|
|
883
|
-
}
|
|
884
|
-
const report = requireEvaluationReport(document, 'current run', lang);
|
|
885
|
-
// gate 内核 = run + verdict, 自动覆盖 omk 全部决策维度(三层 layer-gate /
|
|
886
|
-
// bootstrap diff CI / saturation / Krippendorff α)。computeVerdict 是单一
|
|
887
|
-
// 决策源, exit code 跟 verdict.level 走 — 数据 underpowered 直接 FAIL,
|
|
888
|
-
// 堵住"过 PASS 就 deploy"的漏洞。
|
|
889
|
-
const { computeVerdict, formatVerdictText } = await import('../eval-core/verdict.js');
|
|
890
|
-
const result = computeVerdict(report, {
|
|
891
|
-
gateThreshold: Number(values.threshold),
|
|
892
|
-
triviallySmallDiff: values['trivial-diff'] != null ? Number(values['trivial-diff']) : undefined,
|
|
893
|
-
});
|
|
894
|
-
console.log(formatVerdictText(result, { verbose: true }));
|
|
895
|
-
// exit code 与 handleVerdict 对齐:只有 PROGRESS / SOLO-pass 才 0,
|
|
896
|
-
// NOISE / UNDERPOWERED / CAUTIOUS / REGRESS 全 1。pipeline `omk bench gate
|
|
897
|
-
// && deploy` 数据不显著就不会误 deploy。
|
|
898
|
-
if (result.level === 'PROGRESS') {
|
|
899
|
-
process.exit(0);
|
|
900
|
-
}
|
|
901
|
-
if (result.level === 'SOLO' && result.headline.includes('PASS')) {
|
|
902
|
-
process.exit(0);
|
|
903
|
-
}
|
|
904
|
-
process.exit(1);
|
|
905
|
-
}
|
|
906
|
-
catch (err) {
|
|
907
|
-
console.error(`Error: ${err.message}`);
|
|
908
|
-
process.exit(1);
|
|
909
|
-
}
|
|
910
|
-
}
|
|
911
|
-
// ---------------------------------------------------------------------------
|
|
912
|
-
// handleDiff
|
|
913
|
-
// ---------------------------------------------------------------------------
|
|
914
|
-
async function handleDiff(argv) {
|
|
915
|
-
const lang = langFromArgv(argv);
|
|
916
|
-
// Flag-aware split: separate positional report IDs from flags so we can support
|
|
917
|
-
// omk bench diff <id> — within-report sample-level
|
|
918
|
-
// omk bench diff <id1> <id2> — cross-report variant-level (legacy)
|
|
919
|
-
// both with optional --regressions-only / --threshold / --variant flags.
|
|
920
|
-
const positional = [];
|
|
921
|
-
const flagArgs = [];
|
|
11
|
+
function helpKeyFor(cmd, argv) {
|
|
12
|
+
if (!cmd.subHelp)
|
|
13
|
+
return cmd.helpKey;
|
|
922
14
|
for (let i = 0; i < argv.length; i++) {
|
|
923
|
-
const
|
|
924
|
-
if (
|
|
925
|
-
flagArgs.push(a);
|
|
926
|
-
const next = argv[i + 1];
|
|
927
|
-
if (next !== undefined && !next.startsWith('--')) {
|
|
928
|
-
flagArgs.push(next);
|
|
929
|
-
i++;
|
|
930
|
-
}
|
|
931
|
-
}
|
|
932
|
-
else {
|
|
933
|
-
positional.push(a);
|
|
934
|
-
}
|
|
935
|
-
}
|
|
936
|
-
if (positional.length === 0) {
|
|
937
|
-
console.error(tCli('cli.help.diff_usage', lang));
|
|
938
|
-
process.exit(positional.length === 0 ? 1 : 0);
|
|
939
|
-
}
|
|
940
|
-
const { values } = parseArgsStrictOrExit({
|
|
941
|
-
args: flagArgs,
|
|
942
|
-
options: {
|
|
943
|
-
...COMMON_OPTIONS,
|
|
944
|
-
'regressions-only': { type: 'boolean', default: false },
|
|
945
|
-
threshold: { type: 'string' },
|
|
946
|
-
variant: { type: 'string' },
|
|
947
|
-
top: { type: 'string' },
|
|
948
|
-
},
|
|
949
|
-
});
|
|
950
|
-
const { createFileStore } = await import('../server/report-store.js');
|
|
951
|
-
const store = createFileStore(resolve(DEFAULT_REPORTS_DIR));
|
|
952
|
-
if (positional.length === 1) {
|
|
953
|
-
await runSampleLevelDiff(positional[0], store, values, lang);
|
|
954
|
-
return;
|
|
955
|
-
}
|
|
956
|
-
const [id1, id2] = positional;
|
|
957
|
-
const r1 = requireEvaluationReport(await store.get(id1), id1, lang);
|
|
958
|
-
const r2 = requireEvaluationReport(await store.get(id2), id2, lang);
|
|
959
|
-
console.log(`\n Diff: ${id1} → ${id2}\n`);
|
|
960
|
-
// Git info — r1/r2 are guaranteed non-null after process.exit() guards above
|
|
961
|
-
const g1 = r1.meta?.gitInfo;
|
|
962
|
-
const g2 = r2.meta?.gitInfo;
|
|
963
|
-
if (g1 || g2) {
|
|
964
|
-
console.log(` Git: ${g1?.commitShort || '?'}${g1?.dirty ? '*' : ''} (${g1?.branch || '?'}) → ${g2?.commitShort || '?'}${g2?.dirty ? '*' : ''} (${g2?.branch || '?'})`);
|
|
965
|
-
}
|
|
966
|
-
const { crossReportComparabilityWarnings, formatComparabilityWarnings } = await import('../eval-core/comparability.js');
|
|
967
|
-
const comparability = formatComparabilityWarnings(crossReportComparabilityWarnings(r1, r2), lang);
|
|
968
|
-
if (comparability)
|
|
969
|
-
process.stderr.write(`\n${comparability}\n\n`);
|
|
970
|
-
// Per-variant comparison
|
|
971
|
-
const variants = [...new Set([...(r1.meta?.variants || []), ...(r2.meta?.variants || [])])];
|
|
972
|
-
for (const v of variants) {
|
|
973
|
-
const s1 = r1.summary?.[v];
|
|
974
|
-
const s2 = r2.summary?.[v];
|
|
975
|
-
if (!s1 && !s2)
|
|
15
|
+
const token = argv[i];
|
|
16
|
+
if (!token || token === '--help' || token === '-h')
|
|
976
17
|
continue;
|
|
977
|
-
|
|
978
|
-
|
|
979
|
-
const score2 = s2?.avgCompositeScore ?? '-';
|
|
980
|
-
const scoreDelta = typeof score1 === 'number' && typeof score2 === 'number'
|
|
981
|
-
? ` (${score2 > score1 ? '+' : ''}${(score2 - score1).toFixed(2)})`
|
|
982
|
-
: '';
|
|
983
|
-
console.log(` Score: ${score1} → ${score2}${scoreDelta}`);
|
|
984
|
-
const turns1 = s1?.avgNumTurns ?? '-';
|
|
985
|
-
const turns2 = s2?.avgNumTurns ?? '-';
|
|
986
|
-
console.log(` Turns: ${turns1} → ${turns2}`);
|
|
987
|
-
// Tool calls comparison (agent metrics)
|
|
988
|
-
if (s1?.avgToolCalls != null || s2?.avgToolCalls != null) {
|
|
989
|
-
const tc1 = s1?.avgToolCalls ?? '-';
|
|
990
|
-
const tc2 = s2?.avgToolCalls ?? '-';
|
|
991
|
-
console.log(` Tools: ${tc1} → ${tc2}`);
|
|
992
|
-
const sr1 = s1?.toolSuccessRate != null ? `${(s1.toolSuccessRate * 100).toFixed(0)}%` : '-';
|
|
993
|
-
const sr2 = s2?.toolSuccessRate != null ? `${(s2.toolSuccessRate * 100).toFixed(0)}%` : '-';
|
|
994
|
-
console.log(` ToolOK: ${sr1} → ${sr2}`);
|
|
995
|
-
}
|
|
996
|
-
const cost1 = s1?.avgCostPerSample ?? 0;
|
|
997
|
-
const cost2 = s2?.avgCostPerSample ?? 0;
|
|
998
|
-
const reported1 = s1?.execCostReported !== false;
|
|
999
|
-
const reported2 = s2?.execCostReported !== false;
|
|
1000
|
-
const fmt = (c, r) => r ? `$${c.toFixed(4)}` : '—';
|
|
1001
|
-
// 任一边 not reported 就不报增减百分比(没意义)
|
|
1002
|
-
const costPct = (reported1 && reported2 && cost1 > 0)
|
|
1003
|
-
? ` (${cost2 > cost1 ? '+' : ''}${(((cost2 - cost1) / cost1) * 100).toFixed(0)}%)`
|
|
1004
|
-
: '';
|
|
1005
|
-
console.log(` Cost: ${fmt(cost1, reported1)} → ${fmt(cost2, reported2)}${costPct}`);
|
|
1006
|
-
// Skill hash change
|
|
1007
|
-
const h1 = r1.meta?.artifactHashes?.[v];
|
|
1008
|
-
const h2 = r2.meta?.artifactHashes?.[v];
|
|
1009
|
-
if (h1 && h2 && h1 !== h2) {
|
|
1010
|
-
console.log(` Skill: ${h1.slice(0, 8)} → ${h2.slice(0, 8)} (changed)`);
|
|
1011
|
-
}
|
|
1012
|
-
}
|
|
1013
|
-
console.log('');
|
|
1014
|
-
}
|
|
1015
|
-
/**
|
|
1016
|
-
* Within-report sample-level diff. Compares two variants' scores on
|
|
1017
|
-
* each shared sample and surfaces the worst regressions / biggest wins.
|
|
1018
|
-
*
|
|
1019
|
-
* Default focus is variants[0] (control) vs variants[1] (treatment), but
|
|
1020
|
-
* `--variant` overrides which variant is the "treatment" side.
|
|
1021
|
-
*/
|
|
1022
|
-
async function runSampleLevelDiff(reportId, store, flags, lang) {
|
|
1023
|
-
const report = requireEvaluationReport(await store.get(reportId), reportId, lang);
|
|
1024
|
-
const { reportComparabilityWarnings, formatComparabilityWarnings } = await import('../eval-core/comparability.js');
|
|
1025
|
-
const comparability = formatComparabilityWarnings(reportComparabilityWarnings(report), lang);
|
|
1026
|
-
if (comparability)
|
|
1027
|
-
process.stderr.write(`\n${comparability}\n\n`);
|
|
1028
|
-
const variants = report.meta?.variants ?? [];
|
|
1029
|
-
if (variants.length < 2) {
|
|
1030
|
-
console.error('Sample-level diff needs at least 2 variants in the report.');
|
|
1031
|
-
process.exit(1);
|
|
1032
|
-
}
|
|
1033
|
-
const control = variants[0];
|
|
1034
|
-
const treatment = flags.variant ?? variants[1];
|
|
1035
|
-
if (!variants.includes(treatment)) {
|
|
1036
|
-
console.error(`Variant "${treatment}" not in report. Available: ${variants.join(', ')}`);
|
|
1037
|
-
process.exit(1);
|
|
1038
|
-
}
|
|
1039
|
-
const threshold = flags.threshold != null ? Number(flags.threshold) : 0;
|
|
1040
|
-
const regressionsOnly = Boolean(flags['regressions-only']);
|
|
1041
|
-
const topN = flags.top != null ? Math.max(1, Number(flags.top) || 0) : undefined;
|
|
1042
|
-
const rows = [];
|
|
1043
|
-
for (const entry of report.results ?? []) {
|
|
1044
|
-
const c = entry.variants?.[control];
|
|
1045
|
-
const t = entry.variants?.[treatment];
|
|
1046
|
-
if (!c || !t)
|
|
18
|
+
if (token === '--lang') {
|
|
19
|
+
i++;
|
|
1047
20
|
continue;
|
|
1048
|
-
const cComp = c.compositeScore ?? c.llmScore ?? 0;
|
|
1049
|
-
const tComp = t.compositeScore ?? t.llmScore ?? 0;
|
|
1050
|
-
const delta = Number((tComp - cComp).toFixed(3));
|
|
1051
|
-
rows.push({
|
|
1052
|
-
id: entry.sample_id,
|
|
1053
|
-
cFact: c.layeredScores?.factScore, tFact: t.layeredScores?.factScore,
|
|
1054
|
-
cBeh: c.layeredScores?.behaviorScore, tBeh: t.layeredScores?.behaviorScore,
|
|
1055
|
-
cJudge: c.layeredScores?.judgeScore, tJudge: t.layeredScores?.judgeScore,
|
|
1056
|
-
cComp, tComp, delta,
|
|
1057
|
-
});
|
|
1058
|
-
}
|
|
1059
|
-
// Sort by |delta| desc so the most impactful rows surface first.
|
|
1060
|
-
rows.sort((a, b) => Math.abs(b.delta) - Math.abs(a.delta));
|
|
1061
|
-
let filtered = rows;
|
|
1062
|
-
if (regressionsOnly)
|
|
1063
|
-
filtered = filtered.filter((r) => r.delta < threshold);
|
|
1064
|
-
if (topN !== undefined)
|
|
1065
|
-
filtered = filtered.slice(0, topN);
|
|
1066
|
-
console.log(`\n Sample-level diff: ${treatment} vs ${control} (report ${reportId})`);
|
|
1067
|
-
if (regressionsOnly)
|
|
1068
|
-
console.log(` Filter: regressions only (Δ < ${threshold})`);
|
|
1069
|
-
console.log('');
|
|
1070
|
-
console.log(' sample_id Δ composite (c→t) fact (c→t) behavior (c→t) judge (c→t)');
|
|
1071
|
-
console.log(' ' + '-'.repeat(100));
|
|
1072
|
-
if (filtered.length === 0) {
|
|
1073
|
-
console.log(regressionsOnly ? ' (no regressions found)' : ' (no shared samples)');
|
|
1074
|
-
console.log('');
|
|
1075
|
-
return;
|
|
1076
|
-
}
|
|
1077
|
-
const fmt = (a, b) => {
|
|
1078
|
-
const av = typeof a === 'number' ? a.toFixed(2) : '—';
|
|
1079
|
-
const bv = typeof b === 'number' ? b.toFixed(2) : '—';
|
|
1080
|
-
return `${av} → ${bv}`.padEnd(15);
|
|
1081
|
-
};
|
|
1082
|
-
for (const r of filtered) {
|
|
1083
|
-
const sign = r.delta > 0 ? '+' : '';
|
|
1084
|
-
const idCol = r.id.slice(0, 18).padEnd(20);
|
|
1085
|
-
const deltaCol = `${sign}${r.delta.toFixed(2)}`.padEnd(7);
|
|
1086
|
-
const compCol = `${r.cComp.toFixed(2)} → ${r.tComp.toFixed(2)}`.padEnd(17);
|
|
1087
|
-
console.log(` ${idCol}${deltaCol}${compCol}${fmt(r.cFact, r.tFact)} ${fmt(r.cBeh, r.tBeh)} ${fmt(r.cJudge, r.tJudge)}`);
|
|
1088
|
-
}
|
|
1089
|
-
console.log('');
|
|
1090
|
-
console.log(` Showing ${filtered.length} of ${rows.length} samples · sorted by |Δ|`);
|
|
1091
|
-
if (regressionsOnly) {
|
|
1092
|
-
const total = rows.length;
|
|
1093
|
-
const reg = rows.filter((r) => r.delta < threshold).length;
|
|
1094
|
-
console.log(` Regression rate: ${reg}/${total} samples (${total > 0 ? ((reg / total) * 100).toFixed(0) : 0}%)`);
|
|
1095
|
-
}
|
|
1096
|
-
console.log('');
|
|
1097
|
-
}
|
|
1098
|
-
// ---------------------------------------------------------------------------
|
|
1099
|
-
// handleGold — gold dataset workflow (init / validate / compare)
|
|
1100
|
-
// ---------------------------------------------------------------------------
|
|
1101
|
-
async function handleGold(argv) {
|
|
1102
|
-
const lang = langFromArgv(argv);
|
|
1103
|
-
const sub = argv[0];
|
|
1104
|
-
const rest = argv.slice(1);
|
|
1105
|
-
if (!sub || sub === '--help' || sub === '-h') {
|
|
1106
|
-
console.log(tCli('cli.help.gold', lang));
|
|
1107
|
-
process.exit(sub ? 0 : 1);
|
|
1108
|
-
}
|
|
1109
|
-
if (sub === 'init') {
|
|
1110
|
-
const { values } = parseArgsStrictOrExit({
|
|
1111
|
-
args: rest,
|
|
1112
|
-
options: {
|
|
1113
|
-
...COMMON_OPTIONS,
|
|
1114
|
-
out: { type: 'string', default: './gold-dataset' },
|
|
1115
|
-
annotator: { type: 'string' },
|
|
1116
|
-
},
|
|
1117
|
-
});
|
|
1118
|
-
const { initGoldDataset } = await import('../grading/gold-cli.js');
|
|
1119
|
-
try {
|
|
1120
|
-
const written = initGoldDataset(values.out, {
|
|
1121
|
-
annotator: values.annotator,
|
|
1122
|
-
});
|
|
1123
|
-
console.log(tCli('cli.gold.created_files', lang, {
|
|
1124
|
-
n: written.length, dir: values.out,
|
|
1125
|
-
}));
|
|
1126
|
-
for (const p of written)
|
|
1127
|
-
console.log(` ${p}`);
|
|
1128
|
-
console.log(tCli('cli.gold.next_step_edit_annotations', lang));
|
|
1129
|
-
}
|
|
1130
|
-
catch (err) {
|
|
1131
|
-
console.error(err.message);
|
|
1132
|
-
process.exit(1);
|
|
1133
|
-
}
|
|
1134
|
-
return;
|
|
1135
|
-
}
|
|
1136
|
-
if (sub === 'validate') {
|
|
1137
|
-
// 走 helper 让 `omk bench gold validate <dir> --bogus` 走 unknown option 路径,
|
|
1138
|
-
// 而不是直接执行 validate 后再报 dataset 错。
|
|
1139
|
-
const { positionals } = parseArgsStrictOrExit({
|
|
1140
|
-
args: rest,
|
|
1141
|
-
allowPositionals: true,
|
|
1142
|
-
options: { ...COMMON_OPTIONS },
|
|
1143
|
-
});
|
|
1144
|
-
const dir = positionals[0];
|
|
1145
|
-
if (!dir) {
|
|
1146
|
-
console.error(tCli('cli.common.usage_gold_validate', lang));
|
|
1147
|
-
process.exit(1);
|
|
1148
|
-
}
|
|
1149
|
-
const { validateGoldDataset } = await import('../grading/gold-cli.js');
|
|
1150
|
-
const result = validateGoldDataset(dir);
|
|
1151
|
-
if (result.ok) {
|
|
1152
|
-
console.log(tCli('cli.gold.validate_ok', lang, { n: result.sampleCount }));
|
|
1153
|
-
return;
|
|
1154
21
|
}
|
|
1155
|
-
|
|
1156
|
-
for (const msg of result.issues)
|
|
1157
|
-
console.error(` - ${msg}`);
|
|
1158
|
-
process.exit(1);
|
|
1159
|
-
}
|
|
1160
|
-
if (sub === 'compare') {
|
|
1161
|
-
const reportId = rest[0];
|
|
1162
|
-
if (!reportId) {
|
|
1163
|
-
console.error('Usage: omk bench gold compare <reportId> --gold-dir <dir>');
|
|
1164
|
-
process.exit(1);
|
|
1165
|
-
}
|
|
1166
|
-
const { values } = parseArgsStrictOrExit({
|
|
1167
|
-
args: rest.slice(1),
|
|
1168
|
-
options: {
|
|
1169
|
-
...COMMON_OPTIONS,
|
|
1170
|
-
'gold-dir': { type: 'string' },
|
|
1171
|
-
variant: { type: 'string' },
|
|
1172
|
-
'reports-dir': { type: 'string', default: DEFAULT_REPORTS_DIR },
|
|
1173
|
-
'bootstrap-samples': { type: 'string', default: '1000' },
|
|
1174
|
-
seed: { type: 'string' },
|
|
1175
|
-
},
|
|
1176
|
-
});
|
|
1177
|
-
const goldDir = values['gold-dir'];
|
|
1178
|
-
if (!goldDir) {
|
|
1179
|
-
console.error('--gold-dir is required');
|
|
1180
|
-
process.exit(1);
|
|
1181
|
-
}
|
|
1182
|
-
const { loadGoldDataset } = await import('../grading/gold-dataset.js');
|
|
1183
|
-
const { compareGoldToReport, formatGoldCompare } = await import('../grading/gold-cli.js');
|
|
1184
|
-
const { createFileStore } = await import('../server/report-store.js');
|
|
1185
|
-
const { dataset, issues } = loadGoldDataset(goldDir);
|
|
1186
|
-
if (!dataset) {
|
|
1187
|
-
console.error('Cannot load gold dataset:');
|
|
1188
|
-
for (const i of issues)
|
|
1189
|
-
console.error(` - ${i.message}`);
|
|
1190
|
-
process.exit(1);
|
|
1191
|
-
}
|
|
1192
|
-
if (issues.length) {
|
|
1193
|
-
// Non-fatal issues (e.g. duplicate already filtered) — surface them.
|
|
1194
|
-
for (const i of issues)
|
|
1195
|
-
console.error(`warn: ${i.message}`);
|
|
1196
|
-
}
|
|
1197
|
-
const store = createFileStore(resolve(values['reports-dir']));
|
|
1198
|
-
const report = requireEvaluationReport(await store.get(reportId), reportId, lang);
|
|
1199
|
-
const samples = Math.max(100, Number(values['bootstrap-samples']) || 1000);
|
|
1200
|
-
const seedVal = values.seed != null ? Number(values.seed) : undefined;
|
|
1201
|
-
const result = compareGoldToReport({
|
|
1202
|
-
report,
|
|
1203
|
-
gold: dataset,
|
|
1204
|
-
variant: values.variant,
|
|
1205
|
-
samples,
|
|
1206
|
-
seed: Number.isFinite(seedVal) ? seedVal : undefined,
|
|
1207
|
-
});
|
|
1208
|
-
console.log(formatGoldCompare(result, dataset));
|
|
1209
|
-
return;
|
|
1210
|
-
}
|
|
1211
|
-
console.error(`Unknown subcommand: gold ${sub}. Use init / validate / compare.`);
|
|
1212
|
-
process.exit(1);
|
|
1213
|
-
}
|
|
1214
|
-
// ---------------------------------------------------------------------------
|
|
1215
|
-
// handleDebiasValidate — measure length-debias prompt sensitivity
|
|
1216
|
-
// ---------------------------------------------------------------------------
|
|
1217
|
-
async function handleDebiasValidate(argv) {
|
|
1218
|
-
const lang = langFromArgv(argv);
|
|
1219
|
-
const sub = argv[0];
|
|
1220
|
-
const rest = argv.slice(1);
|
|
1221
|
-
if (!sub || sub === '--help' || sub === '-h') {
|
|
1222
|
-
console.log(tCli('cli.help.debias_validate', lang));
|
|
1223
|
-
process.exit(sub ? 0 : 1);
|
|
1224
|
-
}
|
|
1225
|
-
if (sub !== 'length') {
|
|
1226
|
-
console.error(`Unknown debias-validate kind: ${sub}. Use "length".`);
|
|
1227
|
-
process.exit(1);
|
|
1228
|
-
}
|
|
1229
|
-
const reportId = rest[0];
|
|
1230
|
-
if (!reportId) {
|
|
1231
|
-
console.error('Usage: omk bench debias-validate length <reportId>');
|
|
1232
|
-
process.exit(1);
|
|
1233
|
-
}
|
|
1234
|
-
const { values } = parseArgsStrictOrExit({
|
|
1235
|
-
args: rest.slice(1),
|
|
1236
|
-
options: {
|
|
1237
|
-
...COMMON_OPTIONS,
|
|
1238
|
-
'reports-dir': { type: 'string', default: DEFAULT_REPORTS_DIR },
|
|
1239
|
-
samples: { type: 'string' },
|
|
1240
|
-
variant: { type: 'string' },
|
|
1241
|
-
'judge-models': { type: 'string' },
|
|
1242
|
-
'bootstrap-samples': { type: 'string', default: '1000' },
|
|
1243
|
-
seed: { type: 'string' },
|
|
1244
|
-
},
|
|
1245
|
-
});
|
|
1246
|
-
// Parse --judge-models 在 load report 之前 fail-fast。重复 entry / 缺 executor /
|
|
1247
|
-
// 空串等参数错误应立即给 friendly error: + exit 2,不要等到 store IO 完成才暴露。
|
|
1248
|
-
const { parseJudgeModelsArgOrExit: parseJudgesA } = await import('./parse-run-config.js');
|
|
1249
|
-
const cliJudgeModelsA = values['judge-models'] !== undefined
|
|
1250
|
-
? parseJudgesA(values['judge-models'])
|
|
1251
|
-
: undefined;
|
|
1252
|
-
if (cliJudgeModelsA && cliJudgeModelsA.length > 1) {
|
|
1253
|
-
console.error(tCli('cli.common.judge_models_single_only', lang, { cmd: 'debias-validate' }));
|
|
1254
|
-
process.exit(2);
|
|
1255
|
-
}
|
|
1256
|
-
const { createFileStore } = await import('../server/report-store.js');
|
|
1257
|
-
const store = createFileStore(resolve(values['reports-dir']));
|
|
1258
|
-
const report = requireEvaluationReport(await store.get(reportId), reportId, lang);
|
|
1259
|
-
// Resolve samples path: --samples overrides; otherwise read from report.meta.request.
|
|
1260
|
-
const samplesPath = values.samples
|
|
1261
|
-
?? report.meta?.request?.samplesPath;
|
|
1262
|
-
if (!samplesPath) {
|
|
1263
|
-
console.error('Cannot find samples path. Pass --samples <path> or ensure report has request.samplesPath.');
|
|
1264
|
-
process.exit(1);
|
|
1265
|
-
}
|
|
1266
|
-
const { loadSamples } = await import('../inputs/load-samples.js');
|
|
1267
|
-
const { samples } = loadSamples(samplesPath);
|
|
1268
|
-
const debiasJudges = cliJudgeModelsA
|
|
1269
|
-
?? (report.meta?.judgeModels?.[0]
|
|
1270
|
-
? [{ executor: report.meta.judgeModels[0].executor, model: report.meta.judgeModels[0].model }]
|
|
1271
|
-
: []);
|
|
1272
|
-
if (debiasJudges.length === 0) {
|
|
1273
|
-
console.error(tCli('cli.common.no_judge_model', lang));
|
|
1274
|
-
process.exit(1);
|
|
1275
|
-
}
|
|
1276
|
-
process.stderr.write(tCli('cli.debias.warn_cost_doubles', lang));
|
|
1277
|
-
const { createExecutor } = await import('../executors/index.js');
|
|
1278
|
-
const judgeExecutor = createExecutor(debiasJudges[0].executor);
|
|
1279
|
-
const judgeModel = debiasJudges[0].model;
|
|
1280
|
-
const { validateLengthDebias, formatDebiasValidate } = await import('../grading/debias-validate.js');
|
|
1281
|
-
const seedVal = values.seed != null ? Number(values.seed) : undefined;
|
|
1282
|
-
const bsRaw = Number(values['bootstrap-samples']) || 1000;
|
|
1283
|
-
const result = await validateLengthDebias({
|
|
1284
|
-
report,
|
|
1285
|
-
samples,
|
|
1286
|
-
judgeExecutor,
|
|
1287
|
-
judgeModel,
|
|
1288
|
-
variant: values.variant,
|
|
1289
|
-
bootstrapSamples: Math.max(100, bsRaw),
|
|
1290
|
-
seed: Number.isFinite(seedVal) ? seedVal : undefined,
|
|
1291
|
-
onProgress: ({ sample_id, completed, total }) => {
|
|
1292
|
-
process.stderr.write(` judging ${completed}/${total}: ${sample_id}\n`);
|
|
1293
|
-
},
|
|
1294
|
-
});
|
|
1295
|
-
console.log(formatDebiasValidate(result));
|
|
1296
|
-
}
|
|
1297
|
-
// ---------------------------------------------------------------------------
|
|
1298
|
-
// handleSaturation — re-compute saturation verdict from a finished report
|
|
1299
|
-
// ---------------------------------------------------------------------------
|
|
1300
|
-
async function handleSaturation(argv) {
|
|
1301
|
-
const lang = langFromArgv(argv);
|
|
1302
|
-
const reportId = argv[0];
|
|
1303
|
-
if (!reportId || reportId === '--help' || reportId === '-h') {
|
|
1304
|
-
console.log(tCli('cli.help.saturation', lang));
|
|
1305
|
-
process.exit(reportId ? 0 : 1);
|
|
1306
|
-
}
|
|
1307
|
-
const { values } = parseArgsStrictOrExit({
|
|
1308
|
-
args: argv.slice(1),
|
|
1309
|
-
options: {
|
|
1310
|
-
...COMMON_OPTIONS,
|
|
1311
|
-
'reports-dir': { type: 'string', default: DEFAULT_REPORTS_DIR },
|
|
1312
|
-
variant: { type: 'string' },
|
|
1313
|
-
},
|
|
1314
|
-
});
|
|
1315
|
-
const { createFileStore } = await import('../server/report-store.js');
|
|
1316
|
-
const store = createFileStore(resolve(values['reports-dir']));
|
|
1317
|
-
const report = requireEvaluationReport(await store.get(reportId), reportId, lang);
|
|
1318
|
-
const saturation = report.variance?.saturation;
|
|
1319
|
-
if (!saturation) {
|
|
1320
|
-
console.error(tCli('cli.saturation.no_data', lang));
|
|
1321
|
-
process.exit(1);
|
|
1322
|
-
}
|
|
1323
|
-
// Print the persisted verdict from the original run. The trace stores
|
|
1324
|
-
// (mean, ciLow, ciHigh) per checkpoint but not raw scores, so re-running
|
|
1325
|
-
// findSaturationPoint with different method/threshold is not possible
|
|
1326
|
-
// here — that would need raw scores, which would have to be persisted
|
|
1327
|
-
// by runMultiple. Future work: opt-in `--persist-saturation-raw` flag at
|
|
1328
|
-
// run time to enable post-hoc parameter sweeps.
|
|
1329
|
-
const variants = report.meta.variants ?? [];
|
|
1330
|
-
const targetVariants = values.variant ? [values.variant] : variants;
|
|
1331
|
-
console.log(tCli('cli.saturation.verdict_header', lang));
|
|
1332
|
-
for (const variant of targetVariants) {
|
|
1333
|
-
const trace = saturation.perVariant[variant];
|
|
1334
|
-
if (!trace || trace.length === 0) {
|
|
1335
|
-
console.log(tCli('cli.saturation.variant_no_trace', lang, { variant }));
|
|
22
|
+
if (token.startsWith('--lang='))
|
|
1336
23
|
continue;
|
|
1337
|
-
|
|
1338
|
-
|
|
1339
|
-
|
|
1340
|
-
n: trace.length, list: trace.map((p) => p.n).join(', '),
|
|
1341
|
-
}));
|
|
1342
|
-
const last = trace[trace.length - 1];
|
|
1343
|
-
console.log(tCli('cli.saturation.last_point', lang, {
|
|
1344
|
-
mean: last.mean.toFixed(3), lo: last.ciLow.toFixed(3), hi: last.ciHigh.toFixed(3),
|
|
1345
|
-
}));
|
|
1346
|
-
if (saturation.verdicts?.[variant]) {
|
|
1347
|
-
const v = saturation.verdicts[variant];
|
|
1348
|
-
const result = v.saturated
|
|
1349
|
-
? tCli('cli.saturation.persisted_verdict_saturated', lang, { n: v.atN ?? '?' })
|
|
1350
|
-
: tCli('cli.saturation.persisted_verdict_unsaturated', lang);
|
|
1351
|
-
console.log(tCli('cli.saturation.persisted_verdict', lang, {
|
|
1352
|
-
method: v.method, result, reason: v.reason,
|
|
1353
|
-
}));
|
|
1354
|
-
}
|
|
1355
|
-
else if (trace.length < 5) {
|
|
1356
|
-
console.log(tCli('cli.saturation.skipped_too_few_points', lang, { n: trace.length }));
|
|
1357
|
-
}
|
|
24
|
+
if (token.startsWith('-'))
|
|
25
|
+
continue;
|
|
26
|
+
return cmd.subHelp[token] ?? cmd.helpKey;
|
|
1358
27
|
}
|
|
1359
|
-
|
|
28
|
+
return cmd.helpKey;
|
|
1360
29
|
}
|
|
1361
|
-
|
|
1362
|
-
|
|
1363
|
-
|
|
1364
|
-
|
|
1365
|
-
const lang = langFromArgv(argv);
|
|
1366
|
-
const reportId = argv[0];
|
|
1367
|
-
if (!reportId || reportId === '--help' || reportId === '-h') {
|
|
1368
|
-
console.log(tCli('cli.help.verdict', lang));
|
|
1369
|
-
process.exit(reportId ? 0 : 1);
|
|
1370
|
-
}
|
|
1371
|
-
const { values } = parseArgsStrictOrExit({
|
|
1372
|
-
args: argv.slice(1),
|
|
1373
|
-
options: {
|
|
1374
|
-
...COMMON_OPTIONS,
|
|
1375
|
-
'reports-dir': { type: 'string', default: DEFAULT_REPORTS_DIR },
|
|
1376
|
-
threshold: { type: 'string' },
|
|
1377
|
-
'trivial-diff': { type: 'string' },
|
|
1378
|
-
verbose: { type: 'boolean', default: false },
|
|
1379
|
-
},
|
|
1380
|
-
});
|
|
1381
|
-
const { createFileStore } = await import('../server/report-store.js');
|
|
1382
|
-
const store = createFileStore(resolve(values['reports-dir']));
|
|
1383
|
-
const report = requireEvaluationReport(await store.get(reportId), reportId, lang);
|
|
1384
|
-
const { computeVerdict, formatVerdictText } = await import('../eval-core/verdict.js');
|
|
1385
|
-
const { reportComparabilityWarnings, formatComparabilityWarnings } = await import('../eval-core/comparability.js');
|
|
1386
|
-
const comparability = formatComparabilityWarnings(reportComparabilityWarnings(report), lang);
|
|
1387
|
-
if (comparability)
|
|
1388
|
-
process.stderr.write(`${comparability}\n`);
|
|
1389
|
-
const result = computeVerdict(report, {
|
|
1390
|
-
gateThreshold: values.threshold != null ? Number(values.threshold) : undefined,
|
|
1391
|
-
triviallySmallDiff: values['trivial-diff'] != null ? Number(values['trivial-diff']) : undefined,
|
|
1392
|
-
});
|
|
1393
|
-
console.log(formatVerdictText(result, { verbose: Boolean(values.verbose) }));
|
|
1394
|
-
// Exit code reflects ship recommendation: 0 only on PROGRESS / SOLO-pass.
|
|
1395
|
-
// NOISE / UNDERPOWERED / CAUTIOUS / REGRESS all exit 1 so this composes
|
|
1396
|
-
// with shell `&&` chains in CI.
|
|
1397
|
-
if (result.level === 'PROGRESS') {
|
|
1398
|
-
process.exit(0);
|
|
1399
|
-
}
|
|
1400
|
-
if (result.level === 'SOLO' && result.headline.includes('PASS')) {
|
|
1401
|
-
process.exit(0);
|
|
30
|
+
function dispatchOrPrintHelp(cmd, argv, lang) {
|
|
31
|
+
if (argv.includes('--help') || argv.includes('-h')) {
|
|
32
|
+
console.log(tCli(helpKeyFor(cmd, argv), lang).trim());
|
|
33
|
+
throw new CliExit(0);
|
|
1402
34
|
}
|
|
1403
|
-
|
|
35
|
+
return cmd.execute(argv);
|
|
1404
36
|
}
|
|
1405
|
-
|
|
1406
|
-
|
|
1407
|
-
|
|
1408
|
-
|
|
1409
|
-
|
|
1410
|
-
|
|
1411
|
-
|
|
1412
|
-
console.log(tCli('cli.help.diagnose', lang));
|
|
1413
|
-
process.exit(reportId ? 0 : 1);
|
|
1414
|
-
}
|
|
1415
|
-
const { values } = parseArgsStrictOrExit({
|
|
1416
|
-
args: argv.slice(1),
|
|
1417
|
-
options: {
|
|
1418
|
-
...COMMON_OPTIONS,
|
|
1419
|
-
'reports-dir': { type: 'string', default: DEFAULT_REPORTS_DIR },
|
|
1420
|
-
samples: { type: 'string' },
|
|
1421
|
-
top: { type: 'string', default: '10' },
|
|
1422
|
-
'duplicate-rouge': { type: 'string' },
|
|
1423
|
-
'ambiguous-stddev': { type: 'string' },
|
|
1424
|
-
'cost-k': { type: 'string' },
|
|
1425
|
-
'latency-k': { type: 'string' },
|
|
1426
|
-
flat: { type: 'string' },
|
|
1427
|
-
},
|
|
1428
|
-
});
|
|
1429
|
-
const { createFileStore } = await import('../server/report-store.js');
|
|
1430
|
-
const store = createFileStore(resolve(values['reports-dir']));
|
|
1431
|
-
const report = requireEvaluationReport(await store.get(reportId), reportId, lang);
|
|
1432
|
-
// Try to read the samples file for near-duplicate detection. Source order:
|
|
1433
|
-
// 1. --samples <path> override
|
|
1434
|
-
// 2. report.meta.request.samplesPath (recorded at run time)
|
|
1435
|
-
// If neither resolves to a readable file, skip near-duplicate gracefully.
|
|
1436
|
-
let samples;
|
|
1437
|
-
const samplesPath = values.samples ?? report.meta?.request?.samplesPath;
|
|
1438
|
-
if (samplesPath && existsSync(samplesPath)) {
|
|
1439
|
-
try {
|
|
1440
|
-
const { loadSamples } = await import('../inputs/load-samples.js');
|
|
1441
|
-
samples = loadSamples(samplesPath).samples;
|
|
1442
|
-
}
|
|
1443
|
-
catch (err) {
|
|
1444
|
-
process.stderr.write(tCli('cli.common.warn_load_samples_failed', lang, {
|
|
1445
|
-
path: samplesPath, message: err.message,
|
|
1446
|
-
}));
|
|
1447
|
-
}
|
|
37
|
+
async function main() {
|
|
38
|
+
const lang = getCliLang(parseLangFromArgv(process.argv));
|
|
39
|
+
checkUpdate(lang);
|
|
40
|
+
const [command, ...rest] = process.argv.slice(2);
|
|
41
|
+
if (!command || command === '--help' || command === '-h') {
|
|
42
|
+
console.log(tCli('cli.help.product_main', lang).trim());
|
|
43
|
+
throw new CliExit(0);
|
|
1448
44
|
}
|
|
1449
|
-
const
|
|
1450
|
-
|
|
1451
|
-
|
|
1452
|
-
|
|
1453
|
-
samples,
|
|
1454
|
-
duplicateRouge: values['duplicate-rouge'] != null ? Number(values['duplicate-rouge']) : undefined,
|
|
1455
|
-
ambiguousStddev: values['ambiguous-stddev'] != null ? Number(values['ambiguous-stddev']) : undefined,
|
|
1456
|
-
costOutlierK: values['cost-k'] != null ? Number(values['cost-k']) : undefined,
|
|
1457
|
-
latencyOutlierK: values['latency-k'] != null ? Number(values['latency-k']) : undefined,
|
|
1458
|
-
flatThreshold: values.flat != null ? Number(values.flat) : undefined,
|
|
1459
|
-
});
|
|
1460
|
-
console.log(formatSampleDiagnostics(diag, { topN, lang }));
|
|
1461
|
-
// Sample design science coverage block. Render after diagnose 主体,因为
|
|
1462
|
-
// coverage 是声明式元数据(capability/difficulty/construct/provenance)的整体分布,
|
|
1463
|
-
// 跟 issue list 是不同视角的两件事。优先从 samples (现场加载) 算,fallback 到
|
|
1464
|
-
// report.analysis.sampleQuality(报告里持久化的数据)。
|
|
1465
|
-
const { renderSampleDesignCoverage } = await import('./coverage-renderer.js');
|
|
1466
|
-
const coverageBlock = renderSampleDesignCoverage(samples, report.analysis?.sampleQuality, lang);
|
|
1467
|
-
if (coverageBlock)
|
|
1468
|
-
console.log(coverageBlock);
|
|
1469
|
-
// Exit code: 0 if health ≥ 70 and no errors; 1 otherwise. CI-friendly.
|
|
1470
|
-
if (diag.totals.errors === 0 && diag.healthScore >= 70) {
|
|
1471
|
-
process.exit(0);
|
|
45
|
+
const cmd = PRODUCT_COMMANDS[command];
|
|
46
|
+
if (!cmd) {
|
|
47
|
+
console.error(tCli('cli.common.unknown_domain', lang, { domain: command }));
|
|
48
|
+
throw new CliExit(1);
|
|
1472
49
|
}
|
|
1473
|
-
|
|
50
|
+
await dispatchOrPrintHelp(cmd, rest, lang);
|
|
1474
51
|
}
|
|
1475
|
-
|
|
1476
|
-
//
|
|
1477
|
-
//
|
|
1478
|
-
|
|
1479
|
-
|
|
1480
|
-
const reportId = argv[0];
|
|
1481
|
-
if (!reportId || reportId === '--help' || reportId === '-h') {
|
|
1482
|
-
console.log(tCli('cli.help.failures', lang));
|
|
1483
|
-
process.exit(reportId ? 0 : 1);
|
|
52
|
+
main().catch((err) => {
|
|
53
|
+
// CliExit = 命令显式终止(--help / 业务失败 / parse 错误等),透传 exit code。
|
|
54
|
+
// 其他 throw 是未处理的运行时错误,打印 stack 后 exit 1。
|
|
55
|
+
if (err instanceof CliExit) {
|
|
56
|
+
process.exit(err.code);
|
|
1484
57
|
}
|
|
1485
|
-
|
|
1486
|
-
|
|
1487
|
-
|
|
1488
|
-
...COMMON_OPTIONS,
|
|
1489
|
-
'reports-dir': { type: 'string', default: DEFAULT_REPORTS_DIR },
|
|
1490
|
-
'judge-models': { type: 'string' },
|
|
1491
|
-
'max-clusters': { type: 'string', default: '5' },
|
|
1492
|
-
threshold: { type: 'string', default: '3' },
|
|
1493
|
-
'max-feed': { type: 'string', default: '50' },
|
|
1494
|
-
},
|
|
1495
|
-
});
|
|
1496
|
-
// Parse --judge-models 在 load report 之前 fail-fast(同 debias-validate)。
|
|
1497
|
-
const { parseJudgeModelsArgOrExit: parseJudgesB } = await import('./parse-run-config.js');
|
|
1498
|
-
const cliJudgeModelsB = values['judge-models'] !== undefined
|
|
1499
|
-
? parseJudgesB(values['judge-models'])
|
|
1500
|
-
: undefined;
|
|
1501
|
-
if (cliJudgeModelsB && cliJudgeModelsB.length > 1) {
|
|
1502
|
-
console.error(tCli('cli.common.judge_models_single_only', lang, { cmd: 'failures' }));
|
|
1503
|
-
process.exit(2);
|
|
1504
|
-
}
|
|
1505
|
-
const { createFileStore } = await import('../server/report-store.js');
|
|
1506
|
-
const store = createFileStore(resolve(values['reports-dir']));
|
|
1507
|
-
const report = requireEvaluationReport(await store.get(reportId), reportId, lang);
|
|
1508
|
-
const failuresJudges = cliJudgeModelsB
|
|
1509
|
-
?? (report.meta?.judgeModels?.[0]
|
|
1510
|
-
? [{ executor: report.meta.judgeModels[0].executor, model: report.meta.judgeModels[0].model }]
|
|
1511
|
-
: []);
|
|
1512
|
-
if (failuresJudges.length === 0) {
|
|
1513
|
-
console.error(tCli('cli.common.no_judge_model', lang));
|
|
1514
|
-
process.exit(1);
|
|
1515
|
-
}
|
|
1516
|
-
const { createExecutor } = await import('../executors/index.js');
|
|
1517
|
-
const executor = createExecutor(failuresJudges[0].executor);
|
|
1518
|
-
const judgeModel = failuresJudges[0].model;
|
|
1519
|
-
const { clusterFailures, formatFailureClusterReport } = await import('../analysis/failure-clusterer.js');
|
|
1520
|
-
const out = await clusterFailures({
|
|
1521
|
-
report,
|
|
1522
|
-
executor,
|
|
1523
|
-
judgeModel,
|
|
1524
|
-
maxClusters: Number(values['max-clusters']) || 5,
|
|
1525
|
-
failureThreshold: Number(values.threshold) || 3,
|
|
1526
|
-
maxFailuresFed: Number(values['max-feed']) || 50,
|
|
1527
|
-
});
|
|
1528
|
-
console.log(formatFailureClusterReport(out));
|
|
1529
|
-
}
|
|
1530
|
-
// ---------------------------------------------------------------------------
|
|
1531
|
-
// Entry
|
|
1532
|
-
// ---------------------------------------------------------------------------
|
|
1533
|
-
main();
|
|
58
|
+
console.error(err);
|
|
59
|
+
process.exit(1);
|
|
60
|
+
});
|
|
1534
61
|
//# sourceMappingURL=index.js.map
|