oh-my-knowledge 0.22.0 → 0.24.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +63 -63
- package/README.zh.md +64 -62
- package/dist/src/analysis/report-diagnostics.d.ts +28 -3
- package/dist/src/analysis/report-diagnostics.d.ts.map +1 -1
- package/dist/src/analysis/report-diagnostics.js +201 -81
- package/dist/src/analysis/report-diagnostics.js.map +1 -1
- package/dist/src/analysis/sample-diagnostics.d.ts +9 -2
- package/dist/src/analysis/sample-diagnostics.d.ts.map +1 -1
- package/dist/src/analysis/sample-diagnostics.js +224 -25
- package/dist/src/analysis/sample-diagnostics.js.map +1 -1
- package/dist/src/authoring/evolver.d.ts +7 -2
- package/dist/src/authoring/evolver.d.ts.map +1 -1
- package/dist/src/authoring/evolver.js +43 -9
- package/dist/src/authoring/evolver.js.map +1 -1
- package/dist/src/authoring/generator.d.ts +24 -0
- package/dist/src/authoring/generator.d.ts.map +1 -1
- package/dist/src/authoring/generator.js +67 -5
- package/dist/src/authoring/generator.js.map +1 -1
- package/dist/src/cli/coverage-renderer.d.ts +15 -0
- package/dist/src/cli/coverage-renderer.d.ts.map +1 -0
- package/dist/src/cli/coverage-renderer.js +74 -0
- package/dist/src/cli/coverage-renderer.js.map +1 -0
- package/dist/src/cli/i18n-dict.d.ts +1 -1
- package/dist/src/cli/i18n-dict.d.ts.map +1 -1
- package/dist/src/cli/i18n-dict.js +62 -40
- package/dist/src/cli/i18n-dict.js.map +1 -1
- package/dist/src/cli/index.d.ts +3 -0
- package/dist/src/cli/index.d.ts.map +1 -0
- package/dist/src/{cli.js → cli/index.js} +123 -384
- package/dist/src/cli/index.js.map +1 -0
- package/dist/src/cli/parse-run-config.d.ts +56 -0
- package/dist/src/cli/parse-run-config.d.ts.map +1 -0
- package/dist/src/cli/parse-run-config.js +195 -0
- package/dist/src/cli/parse-run-config.js.map +1 -0
- package/dist/src/cli/progress.d.ts +25 -0
- package/dist/src/cli/progress.d.ts.map +1 -0
- package/dist/src/cli/progress.js +62 -0
- package/dist/src/cli/progress.js.map +1 -0
- package/dist/src/cli/update-check.d.ts +3 -0
- package/dist/src/cli/update-check.d.ts.map +1 -0
- package/dist/src/cli/update-check.js +37 -0
- package/dist/src/cli/update-check.js.map +1 -0
- package/dist/src/eval-core/cache.d.ts +7 -5
- package/dist/src/eval-core/cache.d.ts.map +1 -1
- package/dist/src/eval-core/cache.js +11 -7
- package/dist/src/eval-core/cache.js.map +1 -1
- package/dist/src/eval-core/comparability.d.ts +11 -0
- package/dist/src/eval-core/comparability.d.ts.map +1 -0
- package/dist/src/eval-core/comparability.js +271 -0
- package/dist/src/eval-core/comparability.js.map +1 -0
- package/dist/src/eval-core/dependency-checker.d.ts +1 -1
- package/dist/src/eval-core/dependency-checker.js +1 -1
- package/dist/src/eval-core/evaluation-execution.d.ts +5 -2
- package/dist/src/eval-core/evaluation-execution.d.ts.map +1 -1
- package/dist/src/eval-core/evaluation-execution.js +29 -21
- package/dist/src/eval-core/evaluation-execution.js.map +1 -1
- package/dist/src/eval-core/evaluation-job.d.ts +2 -2
- package/dist/src/eval-core/evaluation-job.d.ts.map +1 -1
- package/dist/src/eval-core/evaluation-job.js +2 -2
- package/dist/src/eval-core/evaluation-job.js.map +1 -1
- package/dist/src/eval-core/evaluation-reporting.d.ts +3 -1
- package/dist/src/eval-core/evaluation-reporting.d.ts.map +1 -1
- package/dist/src/eval-core/evaluation-reporting.js +65 -5
- package/dist/src/eval-core/evaluation-reporting.js.map +1 -1
- package/dist/src/eval-core/execution-strategy.js +6 -6
- package/dist/src/eval-core/execution-strategy.js.map +1 -1
- package/dist/src/eval-core/schema.d.ts.map +1 -1
- package/dist/src/eval-core/schema.js +14 -1
- package/dist/src/eval-core/schema.js.map +1 -1
- package/dist/src/eval-core/verdict.d.ts +3 -3
- package/dist/src/eval-core/verdict.js +3 -3
- package/dist/src/eval-workflows/batch-evaluation-workflow.d.ts +111 -0
- package/dist/src/eval-workflows/batch-evaluation-workflow.d.ts.map +1 -0
- package/dist/src/eval-workflows/batch-evaluation-workflow.js +215 -0
- package/dist/src/eval-workflows/batch-evaluation-workflow.js.map +1 -0
- package/dist/src/eval-workflows/evaluation-pipeline.d.ts +7 -5
- package/dist/src/eval-workflows/evaluation-pipeline.d.ts.map +1 -1
- package/dist/src/eval-workflows/evaluation-pipeline.js +13 -8
- package/dist/src/eval-workflows/evaluation-pipeline.js.map +1 -1
- package/dist/src/eval-workflows/evaluation-preparation.d.ts +10 -8
- package/dist/src/eval-workflows/evaluation-preparation.d.ts.map +1 -1
- package/dist/src/eval-workflows/evaluation-preparation.js +8 -4
- package/dist/src/eval-workflows/evaluation-preparation.js.map +1 -1
- package/dist/src/eval-workflows/run-evaluation.d.ts +23 -20
- package/dist/src/eval-workflows/run-evaluation.d.ts.map +1 -1
- package/dist/src/eval-workflows/run-evaluation.js +34 -25
- package/dist/src/eval-workflows/run-evaluation.js.map +1 -1
- package/dist/src/executors/claude-cli.d.ts.map +1 -1
- package/dist/src/executors/claude-cli.js +12 -7
- package/dist/src/executors/claude-cli.js.map +1 -1
- package/dist/src/executors/claude-sdk.d.ts +1 -1
- package/dist/src/executors/claude-sdk.js +1 -1
- package/dist/src/executors/codex-cli-trace.d.ts +10 -0
- package/dist/src/executors/codex-cli-trace.d.ts.map +1 -0
- package/dist/src/executors/codex-cli-trace.js +123 -0
- package/dist/src/executors/codex-cli-trace.js.map +1 -0
- package/dist/src/executors/codex-cli.d.ts +18 -0
- package/dist/src/executors/codex-cli.d.ts.map +1 -0
- package/dist/src/executors/codex-cli.js +254 -0
- package/dist/src/executors/codex-cli.js.map +1 -0
- package/dist/src/executors/codex-sdk.d.ts +18 -0
- package/dist/src/executors/codex-sdk.d.ts.map +1 -0
- package/dist/src/executors/codex-sdk.js +214 -0
- package/dist/src/executors/codex-sdk.js.map +1 -0
- package/dist/src/executors/gemini.d.ts.map +1 -1
- package/dist/src/executors/gemini.js +28 -24
- package/dist/src/executors/gemini.js.map +1 -1
- package/dist/src/executors/index.d.ts.map +1 -1
- package/dist/src/executors/index.js +7 -2
- package/dist/src/executors/index.js.map +1 -1
- package/dist/src/executors/runtime-fingerprint.d.ts +7 -0
- package/dist/src/executors/runtime-fingerprint.d.ts.map +1 -0
- package/dist/src/executors/runtime-fingerprint.js +277 -0
- package/dist/src/executors/runtime-fingerprint.js.map +1 -0
- package/dist/src/executors/script.d.ts.map +1 -1
- package/dist/src/executors/script.js +48 -56
- package/dist/src/executors/script.js.map +1 -1
- package/dist/src/executors/shared.d.ts +78 -1
- package/dist/src/executors/shared.d.ts.map +1 -1
- package/dist/src/executors/shared.js +203 -1
- package/dist/src/executors/shared.js.map +1 -1
- package/dist/src/grading/assertions.d.ts.map +1 -1
- package/dist/src/grading/assertions.js +21 -6
- package/dist/src/grading/assertions.js.map +1 -1
- package/dist/src/grading/gold-dataset.d.ts +1 -1
- package/dist/src/grading/gold-dataset.js +1 -1
- package/dist/src/grading/index.d.ts.map +1 -1
- package/dist/src/grading/index.js +11 -0
- package/dist/src/grading/index.js.map +1 -1
- package/dist/src/grading/judge.d.ts.map +1 -1
- package/dist/src/grading/judge.js +75 -6
- package/dist/src/grading/judge.js.map +1 -1
- package/dist/src/inputs/eval-config.js +2 -2
- package/dist/src/inputs/eval-config.js.map +1 -1
- package/dist/src/inputs/load-samples.d.ts.map +1 -1
- package/dist/src/inputs/load-samples.js +30 -4
- package/dist/src/inputs/load-samples.js.map +1 -1
- package/dist/src/inputs/skill-loader.d.ts +2 -2
- package/dist/src/inputs/skill-loader.d.ts.map +1 -1
- package/dist/src/inputs/skill-loader.js +2 -2
- package/dist/src/inputs/skill-loader.js.map +1 -1
- package/dist/src/renderer/html-renderer.d.ts +5 -4
- package/dist/src/renderer/html-renderer.d.ts.map +1 -1
- package/dist/src/renderer/html-renderer.js +217 -93
- package/dist/src/renderer/html-renderer.js.map +1 -1
- package/dist/src/renderer/layout.d.ts +2 -1
- package/dist/src/renderer/layout.d.ts.map +1 -1
- package/dist/src/renderer/layout.js +30 -42
- package/dist/src/renderer/layout.js.map +1 -1
- package/dist/src/renderer/summary.d.ts +3 -3
- package/dist/src/renderer/summary.d.ts.map +1 -1
- package/dist/src/renderer/summary.js +232 -55
- package/dist/src/renderer/summary.js.map +1 -1
- package/dist/src/renderer/trends.d.ts.map +1 -1
- package/dist/src/renderer/trends.js +5 -3
- package/dist/src/renderer/trends.js.map +1 -1
- package/dist/src/server/report-server.js +4 -4
- package/dist/src/server/report-server.js.map +1 -1
- package/dist/src/server/report-store.d.ts +7 -5
- package/dist/src/server/report-store.d.ts.map +1 -1
- package/dist/src/server/report-store.js +39 -11
- package/dist/src/server/report-store.js.map +1 -1
- package/dist/src/types/eval.d.ts +27 -5
- package/dist/src/types/eval.d.ts.map +1 -1
- package/dist/src/types/executor.d.ts +5 -0
- package/dist/src/types/executor.d.ts.map +1 -1
- package/dist/src/types/judge.d.ts +10 -0
- package/dist/src/types/judge.d.ts.map +1 -1
- package/dist/src/types/report.d.ts +151 -29
- package/dist/src/types/report.d.ts.map +1 -1
- package/dist/src/types/storage.d.ts +7 -7
- package/dist/src/types/storage.d.ts.map +1 -1
- package/package.json +14 -5
- package/dist/src/cli.d.ts +0 -3
- package/dist/src/cli.d.ts.map +0 -1
- package/dist/src/cli.js.map +0 -1
- package/dist/src/eval-workflows/each-evaluation-workflow.d.ts +0 -153
- package/dist/src/eval-workflows/each-evaluation-workflow.d.ts.map +0 -1
- package/dist/src/eval-workflows/each-evaluation-workflow.js +0 -178
- package/dist/src/eval-workflows/each-evaluation-workflow.js.map +0 -1
- package/dist/src/executors/openai-cli.d.ts +0 -3
- package/dist/src/executors/openai-cli.d.ts.map +0 -1
- package/dist/src/executors/openai-cli.js +0 -60
- package/dist/src/executors/openai-cli.js.map +0 -1
|
@@ -1,241 +1,24 @@
|
|
|
1
1
|
#!/usr/bin/env node
|
|
2
2
|
import { parseArgs } from 'node:util';
|
|
3
3
|
import { resolve } from 'node:path';
|
|
4
|
-
import { homedir } from 'node:os';
|
|
5
4
|
import { join } from 'node:path';
|
|
6
5
|
import { existsSync } from 'node:fs';
|
|
7
|
-
import { tCli, getCliLang, parseLangFromArgv, langFromArgv } from './
|
|
8
|
-
import {
|
|
9
|
-
import {
|
|
10
|
-
|
|
11
|
-
|
|
12
|
-
|
|
13
|
-
|
|
14
|
-
|
|
15
|
-
// Defaults are applied inside parseRunConfig (after config-file merge) so that
|
|
16
|
-
// CLI `undefined` can be reliably distinguished from "user passed the default value".
|
|
17
|
-
// Priority order resolved in parseRunConfig: CLI arg > --config file > hard-coded default.
|
|
18
|
-
/**
|
|
19
|
-
* 所有子命令都接受的通用 flag。新增 --lang 让 parseArgs strict:false 模式下
|
|
20
|
-
* 仍能把值类型化到 values.lang 上(否则未声明的 flag 会被丢弃)。
|
|
21
|
-
*/
|
|
22
|
-
const COMMON_OPTIONS = {
|
|
23
|
-
lang: { type: 'string' },
|
|
24
|
-
};
|
|
25
|
-
const RUN_OPTIONS = {
|
|
26
|
-
...COMMON_OPTIONS,
|
|
27
|
-
samples: { type: 'string' },
|
|
28
|
-
'skill-dir': { type: 'string' },
|
|
29
|
-
control: { type: 'string' },
|
|
30
|
-
treatment: { type: 'string' },
|
|
31
|
-
config: { type: 'string' },
|
|
32
|
-
model: { type: 'string' },
|
|
33
|
-
'judge-model': { type: 'string' },
|
|
34
|
-
'output-dir': { type: 'string' },
|
|
35
|
-
'no-judge': { type: 'boolean' },
|
|
36
|
-
'no-cache': { type: 'boolean' },
|
|
37
|
-
'dry-run': { type: 'boolean' },
|
|
38
|
-
concurrency: { type: 'string' },
|
|
39
|
-
timeout: { type: 'string' },
|
|
40
|
-
executor: { type: 'string' },
|
|
41
|
-
'judge-executor': { type: 'string' },
|
|
42
|
-
each: { type: 'boolean' },
|
|
43
|
-
'skip-preflight': { type: 'boolean' },
|
|
44
|
-
'mcp-config': { type: 'string' },
|
|
45
|
-
'no-serve': { type: 'boolean' },
|
|
46
|
-
verbose: { type: 'boolean' },
|
|
47
|
-
retry: { type: 'string' },
|
|
48
|
-
resume: { type: 'string' },
|
|
49
|
-
'layered-stats': { type: 'boolean' },
|
|
50
|
-
// v0.22 — strict-baseline default true. Declare both forms; reconcile in
|
|
51
|
-
// parseRunConfig (后者赢)。strict-baseline 没传 + no-strict-baseline 没传 = default true。
|
|
52
|
-
'strict-baseline': { type: 'boolean' },
|
|
53
|
-
'no-strict-baseline': { type: 'boolean' },
|
|
54
|
-
};
|
|
55
|
-
// ---------------------------------------------------------------------------
|
|
56
|
-
// parseRunConfig
|
|
57
|
-
// ---------------------------------------------------------------------------
|
|
58
|
-
function parseRunConfig(argv, extraOptions = {}) {
|
|
59
|
-
const { values } = parseArgs({
|
|
60
|
-
args: argv,
|
|
61
|
-
options: { ...RUN_OPTIONS, ...extraOptions },
|
|
62
|
-
strict: false,
|
|
63
|
-
});
|
|
64
|
-
if (values.variants !== undefined) {
|
|
65
|
-
throw new Error(`--variants 已在 v0.16 废除,请改用 --control <expr> 与 --treatment <v1,v2,...>\n`
|
|
66
|
-
+ ` 迁移示例:--variants baseline,my-skill → --control baseline --treatment my-skill\n`
|
|
67
|
-
+ ` 复杂场景可用 --config eval.yaml(参见 docs/terminology-spec.md)`);
|
|
68
|
-
}
|
|
69
|
-
// 1) Load --config (if provided). All subsequent fields fall back to it when CLI is silent.
|
|
70
|
-
const evalConfig = values.config
|
|
71
|
-
? loadEvalConfig(values.config)
|
|
72
|
-
: null;
|
|
73
|
-
// 2) Resolve samples path: CLI > config > auto-detect .json/.yaml/.yml in cwd.
|
|
74
|
-
const cliSamples = values.samples;
|
|
75
|
-
let samplesFile;
|
|
76
|
-
if (cliSamples) {
|
|
77
|
-
samplesFile = cliSamples;
|
|
78
|
-
}
|
|
79
|
-
else if (evalConfig?.samples) {
|
|
80
|
-
samplesFile = evalConfig.samples; // already resolved against config file dir
|
|
81
|
-
}
|
|
82
|
-
else {
|
|
83
|
-
samplesFile = 'eval-samples.json';
|
|
84
|
-
if (!existsSync(resolve(samplesFile))) {
|
|
85
|
-
if (existsSync(resolve('eval-samples.yaml')))
|
|
86
|
-
samplesFile = 'eval-samples.yaml';
|
|
87
|
-
else if (existsSync(resolve('eval-samples.yml')))
|
|
88
|
-
samplesFile = 'eval-samples.yml';
|
|
89
|
-
}
|
|
90
|
-
}
|
|
91
|
-
const skillDir = resolve(values['skill-dir'] ?? 'skills');
|
|
92
|
-
// 3) Resolve variantSpecs: CLI > config. If neither, error with a helpful hint.
|
|
93
|
-
const controlExpr = values.control;
|
|
94
|
-
const treatmentExprs = values.treatment
|
|
95
|
-
? values.treatment.split(',').map((v) => v.trim()).filter(Boolean)
|
|
96
|
-
: [];
|
|
97
|
-
let variantSpecs;
|
|
98
|
-
if (controlExpr || treatmentExprs.length > 0) {
|
|
99
|
-
// CLI roles present → CLI entirely replaces config.variants (no merging).
|
|
100
|
-
variantSpecs = [];
|
|
101
|
-
if (controlExpr) {
|
|
102
|
-
variantSpecs.push({ name: parseVariantCwd(controlExpr).name, role: 'control', expr: controlExpr });
|
|
103
|
-
}
|
|
104
|
-
for (const expr of treatmentExprs) {
|
|
105
|
-
variantSpecs.push({ name: parseVariantCwd(expr).name, role: 'treatment', expr });
|
|
106
|
-
}
|
|
107
|
-
}
|
|
108
|
-
else if (evalConfig) {
|
|
109
|
-
variantSpecs = configVariantsToSpecs(evalConfig.variants);
|
|
110
|
-
}
|
|
111
|
-
else if (values.each) {
|
|
112
|
-
// --each 模式自动用 baseline (control) vs 每个 skill (treatment),
|
|
113
|
-
// 不需要用户显式传 --control / --treatment,校验跳过。
|
|
114
|
-
variantSpecs = [];
|
|
115
|
-
}
|
|
116
|
-
else {
|
|
117
|
-
const discovered = discoverVariants(skillDir);
|
|
118
|
-
const hint = discovered.length > 0 ? `\n skill-dir (${skillDir}) 下发现的候选:${discovered.join(', ')}` : '';
|
|
119
|
-
throw new Error(`请通过 --control / --treatment 或 --config eval.yaml 声明 variant 角色。\n`
|
|
120
|
-
+ ` 示例:omk bench run --control baseline --treatment my-skill${hint}\n`
|
|
121
|
-
+ ` --each 模式下自动用 baseline vs 每个 skill,无需显式声明\n`
|
|
122
|
-
+ ` 术语见 docs/terminology-spec.md(v0.16 起废除 --variants,改用 experiment role 显式声明)`);
|
|
123
|
-
}
|
|
124
|
-
const seenNames = new Set();
|
|
125
|
-
for (const spec of variantSpecs) {
|
|
126
|
-
if (seenNames.has(spec.name)) {
|
|
127
|
-
throw new Error(`variant "${spec.name}" 重复出现——同一 variant 不能同时属于 --control 与 --treatment,也不能在 --treatment 中重复。`);
|
|
128
|
-
}
|
|
129
|
-
seenNames.add(spec.name);
|
|
130
|
-
}
|
|
131
|
-
// 4) Apply CLI > config > hard-coded default for all other fields.
|
|
132
|
-
const executorName = values.executor ?? evalConfig?.executor ?? 'claude';
|
|
133
|
-
const judgeExecutorName = values['judge-executor'] ?? evalConfig?.judgeExecutor ?? executorName;
|
|
134
|
-
const model = values.model ?? evalConfig?.model ?? 'sonnet';
|
|
135
|
-
const judgeModelRaw = values['judge-model'] !== undefined
|
|
136
|
-
? values['judge-model']
|
|
137
|
-
: evalConfig?.judgeModel ?? 'haiku';
|
|
138
|
-
const judgeModel = judgeModelRaw ?? 'haiku';
|
|
139
|
-
const outputDir = resolve(values['output-dir'] ?? DEFAULT_REPORTS_DIR);
|
|
140
|
-
const concurrencyRaw = values.concurrency !== undefined
|
|
141
|
-
? Number(values.concurrency)
|
|
142
|
-
: evalConfig?.concurrency ?? 1;
|
|
143
|
-
const concurrency = Math.max(1, Number(concurrencyRaw) || 1);
|
|
144
|
-
const timeoutSec = values.timeout !== undefined
|
|
145
|
-
? Number(values.timeout)
|
|
146
|
-
: evalConfig?.timeoutMs
|
|
147
|
-
? evalConfig.timeoutMs / 1000
|
|
148
|
-
: 120;
|
|
149
|
-
const timeoutMs = Math.max(1, Number(timeoutSec) || 120) * 1000;
|
|
150
|
-
const noJudge = values['no-judge'] ?? false;
|
|
151
|
-
const noCache = values['no-cache'] ?? evalConfig?.noCache ?? false;
|
|
152
|
-
const dryRun = values['dry-run'] ?? false;
|
|
153
|
-
const skipPreflight = values['skip-preflight'] ?? false;
|
|
154
|
-
const mcpConfig = values['mcp-config'] ?? evalConfig?.mcpConfig;
|
|
155
|
-
const verbose = values.verbose ?? false;
|
|
156
|
-
const retry = Math.max(0, Number(values.retry ?? 0) || 0);
|
|
157
|
-
const resume = values.resume;
|
|
158
|
-
const blind = values.blind ?? evalConfig?.blind ?? false;
|
|
159
|
-
const layeredStats = values['layered-stats'] ?? false;
|
|
160
|
-
// v0.22 — strict-baseline default true. Reconcile both flag forms.
|
|
161
|
-
// Priority: --no-strict-baseline > --strict-baseline > undefined(=true).
|
|
162
|
-
const noStrictFlag = values['no-strict-baseline'];
|
|
163
|
-
const strictFlag = values['strict-baseline'];
|
|
164
|
-
const strictBaseline = noStrictFlag === true ? false : (strictFlag ?? true);
|
|
165
|
-
// v0.22 — extract eval.yaml variant.allowedSkills overrides (per-variant). Always
|
|
166
|
-
// wins over strictBaseline default. Empty object when no eval.yaml or no overrides.
|
|
167
|
-
const variantAllowedSkills = {};
|
|
168
|
-
if (evalConfig?.variants) {
|
|
169
|
-
for (const v of evalConfig.variants) {
|
|
170
|
-
if (v.allowedSkills !== undefined) {
|
|
171
|
-
variantAllowedSkills[v.name] = v.allowedSkills;
|
|
172
|
-
}
|
|
173
|
-
}
|
|
6
|
+
import { tCli, getCliLang, parseLangFromArgv, langFromArgv } from './i18n.js';
|
|
7
|
+
import { parseRunConfig, DEFAULT_REPORTS_DIR, COMMON_OPTIONS, } from './parse-run-config.js';
|
|
8
|
+
import { makeOnProgress } from './progress.js';
|
|
9
|
+
import { checkUpdate } from './update-check.js';
|
|
10
|
+
function requireEvaluationReport(report, id, lang) {
|
|
11
|
+
if (!report) {
|
|
12
|
+
console.error(tCli('cli.common.report_not_found', lang, { id }));
|
|
13
|
+
process.exit(1);
|
|
174
14
|
}
|
|
175
|
-
|
|
176
|
-
|
|
177
|
-
|
|
178
|
-
|
|
179
|
-
|
|
180
|
-
variantSpecs,
|
|
181
|
-
model,
|
|
182
|
-
judgeModel,
|
|
183
|
-
outputDir,
|
|
184
|
-
noJudge,
|
|
185
|
-
noCache,
|
|
186
|
-
dryRun,
|
|
187
|
-
concurrency,
|
|
188
|
-
timeoutMs,
|
|
189
|
-
executorName,
|
|
190
|
-
judgeExecutorName,
|
|
191
|
-
skipPreflight,
|
|
192
|
-
mcpConfig,
|
|
193
|
-
verbose,
|
|
194
|
-
retry,
|
|
195
|
-
resume,
|
|
196
|
-
blind,
|
|
197
|
-
layeredStats,
|
|
198
|
-
budget: evalConfig?.budget,
|
|
199
|
-
strictBaseline,
|
|
200
|
-
...(Object.keys(variantAllowedSkills).length > 0 && { variantAllowedSkills }),
|
|
201
|
-
},
|
|
202
|
-
};
|
|
203
|
-
}
|
|
204
|
-
// ---------------------------------------------------------------------------
|
|
205
|
-
// Update check
|
|
206
|
-
// ---------------------------------------------------------------------------
|
|
207
|
-
async function checkUpdate(lang) {
|
|
208
|
-
try {
|
|
209
|
-
const { readFileSync } = await import('node:fs');
|
|
210
|
-
const { fileURLToPath } = await import('node:url');
|
|
211
|
-
const { dirname, join } = await import('node:path');
|
|
212
|
-
const __dirname = dirname(fileURLToPath(import.meta.url));
|
|
213
|
-
const findPackageJson = (startDir) => {
|
|
214
|
-
let dir = startDir;
|
|
215
|
-
for (let i = 0; i < 5; i++) {
|
|
216
|
-
const candidate = join(dir, 'package.json');
|
|
217
|
-
if (existsSync(candidate))
|
|
218
|
-
return candidate;
|
|
219
|
-
dir = dirname(dir);
|
|
220
|
-
}
|
|
221
|
-
return null;
|
|
222
|
-
};
|
|
223
|
-
const pkgPath = findPackageJson(__dirname);
|
|
224
|
-
if (!pkgPath)
|
|
225
|
-
return;
|
|
226
|
-
const pkg = JSON.parse(readFileSync(pkgPath, 'utf-8'));
|
|
227
|
-
const registry = pkg.publishConfig?.registry || 'https://registry.npmjs.org';
|
|
228
|
-
const res = await fetch(`${registry}/${pkg.name}/latest`, { signal: AbortSignal.timeout(3000) });
|
|
229
|
-
if (!res.ok)
|
|
230
|
-
return;
|
|
231
|
-
const data = await res.json();
|
|
232
|
-
if (data.version && data.version !== pkg.version) {
|
|
233
|
-
process.stderr.write(tCli('cli.update.new_version_available', lang, {
|
|
234
|
-
old: pkg.version, new: data.version, pkg: pkg.name,
|
|
235
|
-
}));
|
|
236
|
-
}
|
|
15
|
+
if (report.kind === 'batch-evaluation') {
|
|
16
|
+
console.error(lang === 'zh'
|
|
17
|
+
? `报告 ${id} 是 BatchEvaluationReport。该命令需要单次 EvaluationReport;请使用其中的 child reportId。`
|
|
18
|
+
: `Report ${id} is a BatchEvaluationReport. This command requires an EvaluationReport; use a child reportId from the batch.`);
|
|
19
|
+
process.exit(1);
|
|
237
20
|
}
|
|
238
|
-
|
|
21
|
+
return report;
|
|
239
22
|
}
|
|
240
23
|
// ---------------------------------------------------------------------------
|
|
241
24
|
// Main
|
|
@@ -313,64 +96,15 @@ async function main() {
|
|
|
313
96
|
* Factory: 闭住 lang, 返回 onProgress callback。evaluation engine 回调时不传
|
|
314
97
|
* 上下文, 所以 lang 必须在 handler 入口处通过 closure 传进来。
|
|
315
98
|
*/
|
|
316
|
-
function makeOnProgress(lang) {
|
|
317
|
-
return ({ phase, completed, total, sample_id, variant, durationMs, inputTokens, outputTokens, costUSD, score, outputPreview, judgePhase: _judgePhase, judgeDim, skipped, attempt, maxAttempts, error, }) => {
|
|
318
|
-
const ctx = { i: completed ?? '', n: total ?? '', sample: sample_id ?? '', variant: variant ?? '' };
|
|
319
|
-
if (phase === 'preflight') {
|
|
320
|
-
process.stderr.write(tCli('cli.progress.preflight_starting', lang));
|
|
321
|
-
return;
|
|
322
|
-
}
|
|
323
|
-
if (phase === 'retry') {
|
|
324
|
-
process.stderr.write(tCli('cli.progress.sample_retry', lang, {
|
|
325
|
-
...ctx, attempt: attempt ?? '', max: maxAttempts ?? '',
|
|
326
|
-
}));
|
|
327
|
-
return;
|
|
328
|
-
}
|
|
329
|
-
if (phase === 'error') {
|
|
330
|
-
process.stderr.write(tCli('cli.progress.sample_error', lang, { ...ctx, error: error ?? '' }));
|
|
331
|
-
return;
|
|
332
|
-
}
|
|
333
|
-
if (phase === 'start') {
|
|
334
|
-
process.stderr.write(tCli('cli.progress.sample_executing', lang, ctx));
|
|
335
|
-
}
|
|
336
|
-
else if (phase === 'exec_done') {
|
|
337
|
-
const cost = costUSD != null && costUSD > 0 ? ` $${costUSD.toFixed(4)}` : '';
|
|
338
|
-
process.stderr.write(tCli('cli.progress.sample_exec_done', lang, {
|
|
339
|
-
...ctx, ms: durationMs ?? '', input: inputTokens ?? '', output: outputTokens ?? '', cost,
|
|
340
|
-
}));
|
|
341
|
-
if (outputPreview) {
|
|
342
|
-
process.stderr.write(tCli('cli.progress.output_preview', lang, {
|
|
343
|
-
preview: outputPreview.slice(0, 150).replace(/\n/g, ' '),
|
|
344
|
-
}));
|
|
345
|
-
}
|
|
346
|
-
}
|
|
347
|
-
else if (phase === 'grading') {
|
|
348
|
-
const dim = judgeDim ? ` [${judgeDim}]` : '';
|
|
349
|
-
process.stderr.write(tCli('cli.progress.judging', lang, { ...ctx, dim }));
|
|
350
|
-
}
|
|
351
|
-
else if (phase === 'judge_done') {
|
|
352
|
-
const dim = judgeDim ? ` [${judgeDim}]` : '';
|
|
353
|
-
process.stderr.write(tCli('cli.progress.judged', lang, { ...ctx, dim, score: score ?? '' }));
|
|
354
|
-
}
|
|
355
|
-
else if (phase === 'done' && skipped) {
|
|
356
|
-
if (sample_id)
|
|
357
|
-
process.stderr.write(tCli('cli.progress.skipped', lang, ctx));
|
|
358
|
-
}
|
|
359
|
-
else {
|
|
360
|
-
const cost = costUSD != null && costUSD > 0 ? ` $${costUSD.toFixed(4)}` : '';
|
|
361
|
-
const scoreInfo = typeof score === 'number' ? ` score=${score}` : '';
|
|
362
|
-
process.stderr.write(tCli('cli.progress.sample_done', lang, {
|
|
363
|
-
...ctx, ms: durationMs ?? '', input: inputTokens ?? '', output: outputTokens ?? '',
|
|
364
|
-
cost, score: scoreInfo,
|
|
365
|
-
}));
|
|
366
|
-
}
|
|
367
|
-
};
|
|
368
|
-
}
|
|
369
99
|
// ---------------------------------------------------------------------------
|
|
370
100
|
// handleRun
|
|
371
101
|
// ---------------------------------------------------------------------------
|
|
372
102
|
async function handleRun(argv) {
|
|
373
103
|
const lang = langFromArgv(argv);
|
|
104
|
+
if (argv.includes('--help') || argv.includes('-h')) {
|
|
105
|
+
console.log(tCli('cli.help.main', lang).trim());
|
|
106
|
+
process.exit(0);
|
|
107
|
+
}
|
|
374
108
|
const { values, config } = parseRunConfig(argv, {
|
|
375
109
|
blind: { type: 'boolean' },
|
|
376
110
|
repeat: { type: 'string', default: '1' },
|
|
@@ -384,13 +118,13 @@ async function handleRun(argv) {
|
|
|
384
118
|
'budget-per-sample-usd': { type: 'string' },
|
|
385
119
|
'budget-per-sample-ms': { type: 'string' },
|
|
386
120
|
});
|
|
387
|
-
const { runEvaluation, runMultiple,
|
|
121
|
+
const { runEvaluation, runMultiple, runBatchEvaluation } = await import('../eval-workflows/run-evaluation.js');
|
|
388
122
|
if (values.blind !== undefined) {
|
|
389
123
|
config.blind = values.blind;
|
|
390
124
|
}
|
|
391
125
|
config.onProgress = makeOnProgress(lang);
|
|
392
126
|
// --repeat 输入校验: 非 ≥1 整数时提示并钳到 1, 不静默掩盖用户错字 / 极端输入。
|
|
393
|
-
// 提前到 --
|
|
127
|
+
// 提前到 --batch 分支之前, 保证 batch 模式也能读到 repeat。
|
|
394
128
|
const repeatRaw = values.repeat;
|
|
395
129
|
const parsedRepeat = repeatRaw !== undefined ? Number(repeatRaw) : 1;
|
|
396
130
|
if (repeatRaw !== undefined && (!Number.isFinite(parsedRepeat) || parsedRepeat < 1)) {
|
|
@@ -430,7 +164,7 @@ async function handleRun(argv) {
|
|
|
430
164
|
}
|
|
431
165
|
}
|
|
432
166
|
// --budget-usd / --budget-per-sample-usd / --budget-per-sample-ms:
|
|
433
|
-
//
|
|
167
|
+
// hard budget caps. CLI flags override config-file values. When the
|
|
434
168
|
// total-USD cap is exceeded mid-run, remaining tasks are skipped and a
|
|
435
169
|
// partial report is persisted with meta.budgetExhausted=true.
|
|
436
170
|
const budgetUSD = values['budget-usd'] != null ? Number(values['budget-usd']) : undefined;
|
|
@@ -465,9 +199,9 @@ async function handleRun(argv) {
|
|
|
465
199
|
config.bootstrapSamples = bsCount;
|
|
466
200
|
}
|
|
467
201
|
try {
|
|
468
|
-
// --
|
|
469
|
-
if (values.
|
|
470
|
-
const { report, filePath } = await
|
|
202
|
+
// --batch mode: evaluate each skill independently
|
|
203
|
+
if (values.batch) {
|
|
204
|
+
const { report, filePath } = await runBatchEvaluation({
|
|
471
205
|
...config,
|
|
472
206
|
repeat: repeatCount,
|
|
473
207
|
onSkillProgress({ phase, skill, current, total }) {
|
|
@@ -483,7 +217,7 @@ async function handleRun(argv) {
|
|
|
483
217
|
process.stderr.write(tCli('cli.run.batch_complete', lang));
|
|
484
218
|
process.stderr.write(tCli('cli.run.report_saved', lang, { path: filePath }));
|
|
485
219
|
if (!values['no-serve'] && process.stdout.isTTY) {
|
|
486
|
-
const { createReportServer } = await import('
|
|
220
|
+
const { createReportServer } = await import('../server/report-server.js');
|
|
487
221
|
const server = createReportServer({ reportsDir: config.outputDir });
|
|
488
222
|
const serverUrl = await server.start();
|
|
489
223
|
const reportUrl = `${serverUrl}/reports/${report.id}`;
|
|
@@ -523,7 +257,7 @@ async function handleRun(argv) {
|
|
|
523
257
|
// --gold-dir: compute α/κ/Pearson against gold annotations and re-persist.
|
|
524
258
|
const goldDir = values['gold-dir'];
|
|
525
259
|
if (goldDir && filePath) {
|
|
526
|
-
const { attachGoldAgreementToReport, formatGoldCompare } = await import('
|
|
260
|
+
const { attachGoldAgreementToReport, formatGoldCompare } = await import('../grading/gold-cli.js');
|
|
527
261
|
const out = attachGoldAgreementToReport({
|
|
528
262
|
report,
|
|
529
263
|
goldDir,
|
|
@@ -551,7 +285,7 @@ async function handleRun(argv) {
|
|
|
551
285
|
process.stderr.write(tCli('cli.run.report_saved', lang, { path: filePath }));
|
|
552
286
|
if (!values['no-serve'] && process.stdout.isTTY) {
|
|
553
287
|
// Auto-start report server
|
|
554
|
-
const { createReportServer } = await import('
|
|
288
|
+
const { createReportServer } = await import('../server/report-server.js');
|
|
555
289
|
const server = createReportServer({
|
|
556
290
|
reportsDir: config.outputDir,
|
|
557
291
|
});
|
|
@@ -582,6 +316,10 @@ async function handleRun(argv) {
|
|
|
582
316
|
// ---------------------------------------------------------------------------
|
|
583
317
|
async function handleReport(argv) {
|
|
584
318
|
const lang = langFromArgv(argv);
|
|
319
|
+
if (argv.includes('--help') || argv.includes('-h')) {
|
|
320
|
+
console.log(tCli('cli.help.main', lang).trim());
|
|
321
|
+
process.exit(0);
|
|
322
|
+
}
|
|
585
323
|
const { values } = parseArgs({
|
|
586
324
|
args: argv,
|
|
587
325
|
options: {
|
|
@@ -612,8 +350,8 @@ async function handleReport(argv) {
|
|
|
612
350
|
return;
|
|
613
351
|
}
|
|
614
352
|
if (values.export) {
|
|
615
|
-
const { createFileStore } = await import('
|
|
616
|
-
const {
|
|
353
|
+
const { createFileStore } = await import('../server/report-store.js');
|
|
354
|
+
const { renderReportDocumentDetail } = await import('../renderer/html-renderer.js');
|
|
617
355
|
const { writeFileSync } = await import('node:fs');
|
|
618
356
|
const store = createFileStore(resolve(values['reports-dir']));
|
|
619
357
|
const report = await store.get(values.export);
|
|
@@ -621,14 +359,14 @@ async function handleReport(argv) {
|
|
|
621
359
|
console.error(tCli('cli.common.report_not_found', lang, { id: values.export }));
|
|
622
360
|
process.exit(1);
|
|
623
361
|
}
|
|
624
|
-
const html =
|
|
362
|
+
const html = renderReportDocumentDetail(report);
|
|
625
363
|
const outPath = resolve(`${values.export}.html`);
|
|
626
364
|
writeFileSync(outPath, html);
|
|
627
365
|
console.log(`Exported to: ${outPath}`);
|
|
628
366
|
console.log('Open in browser, or Ctrl+P to save as PDF');
|
|
629
367
|
return;
|
|
630
368
|
}
|
|
631
|
-
const { createReportServer } = await import('
|
|
369
|
+
const { createReportServer } = await import('../server/report-server.js');
|
|
632
370
|
const server = createReportServer({
|
|
633
371
|
port: Number(values.port),
|
|
634
372
|
reportsDir: resolve(values['reports-dir']),
|
|
@@ -751,7 +489,7 @@ async function handleAnalyze(argv) {
|
|
|
751
489
|
const to = values.to;
|
|
752
490
|
const skills = values.skills ? values.skills.split(',').map((s) => s.trim()).filter(Boolean) : undefined;
|
|
753
491
|
console.log(`[omk] analyzing ${tracePath}...`);
|
|
754
|
-
const { computeSkillHealthReport } = await import('
|
|
492
|
+
const { computeSkillHealthReport } = await import('../observability/skill-health-analyzer.js');
|
|
755
493
|
const report = computeSkillHealthReport(tracePath, {
|
|
756
494
|
kbRoot: values.kb ? resolve(values.kb) : undefined,
|
|
757
495
|
from,
|
|
@@ -800,11 +538,15 @@ async function handleInit(argv) {
|
|
|
800
538
|
// ---------------------------------------------------------------------------
|
|
801
539
|
async function handleGenSamples(argv) {
|
|
802
540
|
const lang = langFromArgv(argv);
|
|
541
|
+
if (argv.includes('--help') || argv.includes('-h')) {
|
|
542
|
+
console.log(tCli('cli.help.main', lang).trim());
|
|
543
|
+
process.exit(0);
|
|
544
|
+
}
|
|
803
545
|
const { values } = parseArgs({
|
|
804
546
|
args: argv,
|
|
805
547
|
options: {
|
|
806
548
|
...COMMON_OPTIONS,
|
|
807
|
-
|
|
549
|
+
batch: { type: 'boolean', default: false },
|
|
808
550
|
count: { type: 'string', default: '5' },
|
|
809
551
|
model: { type: 'string', default: 'sonnet' },
|
|
810
552
|
'skill-dir': { type: 'string', default: 'skills' },
|
|
@@ -812,11 +554,11 @@ async function handleGenSamples(argv) {
|
|
|
812
554
|
strict: false,
|
|
813
555
|
allowPositionals: true,
|
|
814
556
|
});
|
|
815
|
-
const { generateSamples } = await import('
|
|
557
|
+
const { generateSamples } = await import('../authoring/generator.js');
|
|
816
558
|
const { readFileSync, writeFileSync } = await import('node:fs');
|
|
817
559
|
const count = Math.max(1, Number(values.count) || 5);
|
|
818
560
|
const model = values.model;
|
|
819
|
-
if (values.
|
|
561
|
+
if (values.batch) {
|
|
820
562
|
// Batch mode: generate for all skills missing eval-samples
|
|
821
563
|
const skillDir = resolve(values['skill-dir']);
|
|
822
564
|
if (!existsSync(skillDir)) {
|
|
@@ -944,7 +686,7 @@ async function handleEvolve(argv) {
|
|
|
944
686
|
else if (existsSync(resolve('eval-samples.yml')))
|
|
945
687
|
samplesFile = 'eval-samples.yml';
|
|
946
688
|
}
|
|
947
|
-
const { evolveSkill } = await import('
|
|
689
|
+
const { evolveSkill } = await import('../authoring/evolver.js');
|
|
948
690
|
process.stderr.write(tCli('cli.evolve.section_header', lang, { path: skillPath }));
|
|
949
691
|
try {
|
|
950
692
|
const result = await evolveSkill({
|
|
@@ -960,10 +702,13 @@ async function handleEvolve(argv) {
|
|
|
960
702
|
timeoutMs: Math.max(1, Number(values.timeout) || 120) * 1000,
|
|
961
703
|
skipPreflight: values['skip-preflight'],
|
|
962
704
|
onProgress: makeOnProgress(lang),
|
|
963
|
-
onRoundProgress({ round, totalRounds: _totalRounds, phase, score, delta, accepted, costUSD, error }) {
|
|
705
|
+
onRoundProgress({ round, totalRounds: _totalRounds, phase, score, delta, accepted, costUSD, costReported, error }) {
|
|
706
|
+
// costReported=false 时显示「—」而不是 $0.0000(executor 不报 cost,如 codex)。
|
|
707
|
+
// 缺位 / true 当 reported 走旧格式。
|
|
708
|
+
const fmtRoundCost = (c, r) => r ? `$${c.toFixed(4)}` : '—';
|
|
964
709
|
if (phase === 'baseline') {
|
|
965
710
|
process.stderr.write(tCli('cli.evolve.round_baseline', lang, {
|
|
966
|
-
score: score.toFixed(2), cost: costUSD
|
|
711
|
+
score: score.toFixed(2), cost: fmtRoundCost(costUSD, costReported !== false),
|
|
967
712
|
}));
|
|
968
713
|
}
|
|
969
714
|
else if (phase === 'error') {
|
|
@@ -975,7 +720,7 @@ async function handleEvolve(argv) {
|
|
|
975
720
|
const delta_ = delta >= 0 ? `+${delta.toFixed(2)}` : delta.toFixed(2);
|
|
976
721
|
const status = accepted ? '✓ ACCEPT' : '✗ REJECT';
|
|
977
722
|
process.stderr.write(tCli('cli.evolve.round_done', lang, {
|
|
978
|
-
round, score: score.toFixed(2), delta: delta_, status, cost: costUSD
|
|
723
|
+
round, score: score.toFixed(2), delta: delta_, status, cost: fmtRoundCost(costUSD, costReported !== false),
|
|
979
724
|
}));
|
|
980
725
|
}
|
|
981
726
|
},
|
|
@@ -983,9 +728,12 @@ async function handleEvolve(argv) {
|
|
|
983
728
|
const improvement = result.startScore > 0
|
|
984
729
|
? ((result.finalScore - result.startScore) / result.startScore * 100).toFixed(1)
|
|
985
730
|
: '0';
|
|
731
|
+
const totalCostStr = result.costReported === false
|
|
732
|
+
? '—' // 任一轮的 executor 不报 cost → totalCostUSD 是 lower-bound
|
|
733
|
+
: `$${result.totalCostUSD.toFixed(4)}`;
|
|
986
734
|
process.stderr.write(tCli('cli.evolve.summary', lang, {
|
|
987
735
|
start: result.startScore.toFixed(2), final: result.finalScore.toFixed(2),
|
|
988
|
-
percent: improvement, rounds: result.totalRounds, cost:
|
|
736
|
+
percent: improvement, rounds: result.totalRounds, cost: totalCostStr,
|
|
989
737
|
}));
|
|
990
738
|
process.stderr.write(tCli('cli.evolve.best_path', lang, {
|
|
991
739
|
best: result.bestSkillPath, target: resolve(skillPath),
|
|
@@ -1018,19 +766,20 @@ async function handleGate(argv) {
|
|
|
1018
766
|
threshold: { type: 'string', default: '3.5' },
|
|
1019
767
|
'trivial-diff': { type: 'string' },
|
|
1020
768
|
});
|
|
1021
|
-
const { runEvaluation } = await import('
|
|
769
|
+
const { runEvaluation } = await import('../eval-workflows/run-evaluation.js');
|
|
1022
770
|
config.onProgress = makeOnProgress(lang);
|
|
1023
771
|
try {
|
|
1024
|
-
const { report } = (await runEvaluation(config));
|
|
1025
|
-
if (
|
|
772
|
+
const { report: document } = (await runEvaluation(config));
|
|
773
|
+
if (document.dryRun) {
|
|
1026
774
|
console.log('Gate dry-run: no scores to check');
|
|
1027
775
|
process.exit(0);
|
|
1028
776
|
}
|
|
777
|
+
const report = requireEvaluationReport(document, 'current run', lang);
|
|
1029
778
|
// gate 内核 = run + verdict, 自动覆盖 omk 全部决策维度(三层 layer-gate /
|
|
1030
779
|
// bootstrap diff CI / saturation / Krippendorff α)。computeVerdict 是单一
|
|
1031
780
|
// 决策源, exit code 跟 verdict.level 走 — 数据 underpowered 直接 FAIL,
|
|
1032
781
|
// 堵住"过 PASS 就 deploy"的漏洞。
|
|
1033
|
-
const { computeVerdict, formatVerdictText } = await import('
|
|
782
|
+
const { computeVerdict, formatVerdictText } = await import('../eval-core/verdict.js');
|
|
1034
783
|
const result = computeVerdict(report, {
|
|
1035
784
|
gateThreshold: Number(values.threshold),
|
|
1036
785
|
triviallySmallDiff: values['trivial-diff'] != null ? Number(values['trivial-diff']) : undefined,
|
|
@@ -1058,7 +807,7 @@ async function handleGate(argv) {
|
|
|
1058
807
|
async function handleDiff(argv) {
|
|
1059
808
|
const lang = langFromArgv(argv);
|
|
1060
809
|
// Flag-aware split: separate positional report IDs from flags so we can support
|
|
1061
|
-
// omk bench diff <id> — within-report sample-level
|
|
810
|
+
// omk bench diff <id> — within-report sample-level
|
|
1062
811
|
// omk bench diff <id1> <id2> — cross-report variant-level (legacy)
|
|
1063
812
|
// both with optional --regressions-only / --threshold / --variant flags.
|
|
1064
813
|
const positional = [];
|
|
@@ -1092,23 +841,15 @@ async function handleDiff(argv) {
|
|
|
1092
841
|
},
|
|
1093
842
|
strict: false,
|
|
1094
843
|
});
|
|
1095
|
-
const { createFileStore } = await import('
|
|
844
|
+
const { createFileStore } = await import('../server/report-store.js');
|
|
1096
845
|
const store = createFileStore(resolve(DEFAULT_REPORTS_DIR));
|
|
1097
846
|
if (positional.length === 1) {
|
|
1098
847
|
await runSampleLevelDiff(positional[0], store, values, lang);
|
|
1099
848
|
return;
|
|
1100
849
|
}
|
|
1101
850
|
const [id1, id2] = positional;
|
|
1102
|
-
const r1 = await store.get(id1);
|
|
1103
|
-
const r2 = await store.get(id2);
|
|
1104
|
-
if (!r1) {
|
|
1105
|
-
console.error(tCli('cli.common.report_not_found', lang, { id: id1 }));
|
|
1106
|
-
process.exit(1);
|
|
1107
|
-
}
|
|
1108
|
-
if (!r2) {
|
|
1109
|
-
console.error(tCli('cli.common.report_not_found', lang, { id: id2 }));
|
|
1110
|
-
process.exit(1);
|
|
1111
|
-
}
|
|
851
|
+
const r1 = requireEvaluationReport(await store.get(id1), id1, lang);
|
|
852
|
+
const r2 = requireEvaluationReport(await store.get(id2), id2, lang);
|
|
1112
853
|
console.log(`\n Diff: ${id1} → ${id2}\n`);
|
|
1113
854
|
// Git info — r1/r2 are guaranteed non-null after process.exit() guards above
|
|
1114
855
|
const g1 = r1.meta?.gitInfo;
|
|
@@ -1116,6 +857,10 @@ async function handleDiff(argv) {
|
|
|
1116
857
|
if (g1 || g2) {
|
|
1117
858
|
console.log(` Git: ${g1?.commitShort || '?'}${g1?.dirty ? '*' : ''} (${g1?.branch || '?'}) → ${g2?.commitShort || '?'}${g2?.dirty ? '*' : ''} (${g2?.branch || '?'})`);
|
|
1118
859
|
}
|
|
860
|
+
const { crossReportComparabilityWarnings, formatComparabilityWarnings } = await import('../eval-core/comparability.js');
|
|
861
|
+
const comparability = formatComparabilityWarnings(crossReportComparabilityWarnings(r1, r2), lang);
|
|
862
|
+
if (comparability)
|
|
863
|
+
process.stderr.write(`\n${comparability}\n\n`);
|
|
1119
864
|
// Per-variant comparison
|
|
1120
865
|
const variants = [...new Set([...(r1.meta?.variants || []), ...(r2.meta?.variants || [])])];
|
|
1121
866
|
for (const v of variants) {
|
|
@@ -1144,8 +889,14 @@ async function handleDiff(argv) {
|
|
|
1144
889
|
}
|
|
1145
890
|
const cost1 = s1?.avgCostPerSample ?? 0;
|
|
1146
891
|
const cost2 = s2?.avgCostPerSample ?? 0;
|
|
1147
|
-
const
|
|
1148
|
-
|
|
892
|
+
const reported1 = s1?.execCostReported !== false;
|
|
893
|
+
const reported2 = s2?.execCostReported !== false;
|
|
894
|
+
const fmt = (c, r) => r ? `$${c.toFixed(4)}` : '—';
|
|
895
|
+
// 任一边 not reported 就不报增减百分比(没意义)
|
|
896
|
+
const costPct = (reported1 && reported2 && cost1 > 0)
|
|
897
|
+
? ` (${cost2 > cost1 ? '+' : ''}${(((cost2 - cost1) / cost1) * 100).toFixed(0)}%)`
|
|
898
|
+
: '';
|
|
899
|
+
console.log(` Cost: ${fmt(cost1, reported1)} → ${fmt(cost2, reported2)}${costPct}`);
|
|
1149
900
|
// Skill hash change
|
|
1150
901
|
const h1 = r1.meta?.artifactHashes?.[v];
|
|
1151
902
|
const h2 = r2.meta?.artifactHashes?.[v];
|
|
@@ -1156,18 +907,18 @@ async function handleDiff(argv) {
|
|
|
1156
907
|
console.log('');
|
|
1157
908
|
}
|
|
1158
909
|
/**
|
|
1159
|
-
* Within-report sample-level diff
|
|
910
|
+
* Within-report sample-level diff. Compares two variants' scores on
|
|
1160
911
|
* each shared sample and surfaces the worst regressions / biggest wins.
|
|
1161
912
|
*
|
|
1162
913
|
* Default focus is variants[0] (control) vs variants[1] (treatment), but
|
|
1163
914
|
* `--variant` overrides which variant is the "treatment" side.
|
|
1164
915
|
*/
|
|
1165
916
|
async function runSampleLevelDiff(reportId, store, flags, lang) {
|
|
1166
|
-
const report = await store.get(reportId);
|
|
1167
|
-
|
|
1168
|
-
|
|
1169
|
-
|
|
1170
|
-
|
|
917
|
+
const report = requireEvaluationReport(await store.get(reportId), reportId, lang);
|
|
918
|
+
const { reportComparabilityWarnings, formatComparabilityWarnings } = await import('../eval-core/comparability.js');
|
|
919
|
+
const comparability = formatComparabilityWarnings(reportComparabilityWarnings(report), lang);
|
|
920
|
+
if (comparability)
|
|
921
|
+
process.stderr.write(`\n${comparability}\n\n`);
|
|
1171
922
|
const variants = report.meta?.variants ?? [];
|
|
1172
923
|
if (variants.length < 2) {
|
|
1173
924
|
console.error('Sample-level diff needs at least 2 variants in the report.');
|
|
@@ -1259,7 +1010,7 @@ async function handleGold(argv) {
|
|
|
1259
1010
|
},
|
|
1260
1011
|
strict: false,
|
|
1261
1012
|
});
|
|
1262
|
-
const { initGoldDataset } = await import('
|
|
1013
|
+
const { initGoldDataset } = await import('../grading/gold-cli.js');
|
|
1263
1014
|
try {
|
|
1264
1015
|
const written = initGoldDataset(values.out, {
|
|
1265
1016
|
annotator: values.annotator,
|
|
@@ -1283,7 +1034,7 @@ async function handleGold(argv) {
|
|
|
1283
1034
|
console.error(tCli('cli.common.usage_gold_validate', lang));
|
|
1284
1035
|
process.exit(1);
|
|
1285
1036
|
}
|
|
1286
|
-
const { validateGoldDataset } = await import('
|
|
1037
|
+
const { validateGoldDataset } = await import('../grading/gold-cli.js');
|
|
1287
1038
|
const result = validateGoldDataset(dir);
|
|
1288
1039
|
if (result.ok) {
|
|
1289
1040
|
console.log(tCli('cli.gold.validate_ok', lang, { n: result.sampleCount }));
|
|
@@ -1317,9 +1068,9 @@ async function handleGold(argv) {
|
|
|
1317
1068
|
console.error('--gold-dir is required');
|
|
1318
1069
|
process.exit(1);
|
|
1319
1070
|
}
|
|
1320
|
-
const { loadGoldDataset } = await import('
|
|
1321
|
-
const { compareGoldToReport, formatGoldCompare } = await import('
|
|
1322
|
-
const { createFileStore } = await import('
|
|
1071
|
+
const { loadGoldDataset } = await import('../grading/gold-dataset.js');
|
|
1072
|
+
const { compareGoldToReport, formatGoldCompare } = await import('../grading/gold-cli.js');
|
|
1073
|
+
const { createFileStore } = await import('../server/report-store.js');
|
|
1323
1074
|
const { dataset, issues } = loadGoldDataset(goldDir);
|
|
1324
1075
|
if (!dataset) {
|
|
1325
1076
|
console.error('Cannot load gold dataset:');
|
|
@@ -1333,15 +1084,11 @@ async function handleGold(argv) {
|
|
|
1333
1084
|
console.error(`warn: ${i.message}`);
|
|
1334
1085
|
}
|
|
1335
1086
|
const store = createFileStore(resolve(values['reports-dir']));
|
|
1336
|
-
const report = await store.get(reportId);
|
|
1337
|
-
if (!report) {
|
|
1338
|
-
console.error(tCli('cli.common.report_not_found', lang, { id: reportId }));
|
|
1339
|
-
process.exit(1);
|
|
1340
|
-
}
|
|
1087
|
+
const report = requireEvaluationReport(await store.get(reportId), reportId, lang);
|
|
1341
1088
|
const samples = Math.max(100, Number(values['bootstrap-samples']) || 1000);
|
|
1342
1089
|
const seedVal = values.seed != null ? Number(values.seed) : undefined;
|
|
1343
1090
|
const result = compareGoldToReport({
|
|
1344
|
-
report
|
|
1091
|
+
report,
|
|
1345
1092
|
gold: dataset,
|
|
1346
1093
|
variant: values.variant,
|
|
1347
1094
|
samples,
|
|
@@ -1387,13 +1134,9 @@ async function handleDebiasValidate(argv) {
|
|
|
1387
1134
|
},
|
|
1388
1135
|
strict: false,
|
|
1389
1136
|
});
|
|
1390
|
-
const { createFileStore } = await import('
|
|
1137
|
+
const { createFileStore } = await import('../server/report-store.js');
|
|
1391
1138
|
const store = createFileStore(resolve(values['reports-dir']));
|
|
1392
|
-
const report = await store.get(reportId);
|
|
1393
|
-
if (!report) {
|
|
1394
|
-
console.error(tCli('cli.common.report_not_found', lang, { id: reportId }));
|
|
1395
|
-
process.exit(1);
|
|
1396
|
-
}
|
|
1139
|
+
const report = requireEvaluationReport(await store.get(reportId), reportId, lang);
|
|
1397
1140
|
// Resolve samples path: --samples overrides; otherwise read from report.meta.request.
|
|
1398
1141
|
const samplesPath = values.samples
|
|
1399
1142
|
?? report.meta?.request?.samplesPath;
|
|
@@ -1401,7 +1144,7 @@ async function handleDebiasValidate(argv) {
|
|
|
1401
1144
|
console.error('Cannot find samples path. Pass --samples <path> or ensure report has request.samplesPath.');
|
|
1402
1145
|
process.exit(1);
|
|
1403
1146
|
}
|
|
1404
|
-
const { loadSamples } = await import('
|
|
1147
|
+
const { loadSamples } = await import('../inputs/load-samples.js');
|
|
1405
1148
|
const { samples } = loadSamples(samplesPath);
|
|
1406
1149
|
const judgeModel = values['judge-model']
|
|
1407
1150
|
?? report.meta?.judgeModel;
|
|
@@ -1410,13 +1153,13 @@ async function handleDebiasValidate(argv) {
|
|
|
1410
1153
|
process.exit(1);
|
|
1411
1154
|
}
|
|
1412
1155
|
process.stderr.write(tCli('cli.debias.warn_cost_doubles', lang));
|
|
1413
|
-
const { createExecutor } = await import('
|
|
1156
|
+
const { createExecutor } = await import('../executors/index.js');
|
|
1414
1157
|
const judgeExecutor = createExecutor(values['judge-executor']);
|
|
1415
|
-
const { validateLengthDebias, formatDebiasValidate } = await import('
|
|
1158
|
+
const { validateLengthDebias, formatDebiasValidate } = await import('../grading/debias-validate.js');
|
|
1416
1159
|
const seedVal = values.seed != null ? Number(values.seed) : undefined;
|
|
1417
1160
|
const bsRaw = Number(values['bootstrap-samples']) || 1000;
|
|
1418
1161
|
const result = await validateLengthDebias({
|
|
1419
|
-
report
|
|
1162
|
+
report,
|
|
1420
1163
|
samples,
|
|
1421
1164
|
judgeExecutor,
|
|
1422
1165
|
judgeModel,
|
|
@@ -1448,13 +1191,9 @@ async function handleSaturation(argv) {
|
|
|
1448
1191
|
},
|
|
1449
1192
|
strict: false,
|
|
1450
1193
|
});
|
|
1451
|
-
const { createFileStore } = await import('
|
|
1194
|
+
const { createFileStore } = await import('../server/report-store.js');
|
|
1452
1195
|
const store = createFileStore(resolve(values['reports-dir']));
|
|
1453
|
-
const report = await store.get(reportId);
|
|
1454
|
-
if (!report) {
|
|
1455
|
-
console.error(tCli('cli.common.report_not_found', lang, { id: reportId }));
|
|
1456
|
-
process.exit(1);
|
|
1457
|
-
}
|
|
1196
|
+
const report = requireEvaluationReport(await store.get(reportId), reportId, lang);
|
|
1458
1197
|
const saturation = report.variance?.saturation;
|
|
1459
1198
|
if (!saturation) {
|
|
1460
1199
|
console.error(tCli('cli.saturation.no_data', lang));
|
|
@@ -1499,7 +1238,7 @@ async function handleSaturation(argv) {
|
|
|
1499
1238
|
console.log('');
|
|
1500
1239
|
}
|
|
1501
1240
|
// ---------------------------------------------------------------------------
|
|
1502
|
-
// handleVerdict — one-line ship/no-ship verdict
|
|
1241
|
+
// handleVerdict — one-line ship/no-ship verdict
|
|
1503
1242
|
// ---------------------------------------------------------------------------
|
|
1504
1243
|
async function handleVerdict(argv) {
|
|
1505
1244
|
const lang = langFromArgv(argv);
|
|
@@ -1519,14 +1258,14 @@ async function handleVerdict(argv) {
|
|
|
1519
1258
|
},
|
|
1520
1259
|
strict: false,
|
|
1521
1260
|
});
|
|
1522
|
-
const { createFileStore } = await import('
|
|
1261
|
+
const { createFileStore } = await import('../server/report-store.js');
|
|
1523
1262
|
const store = createFileStore(resolve(values['reports-dir']));
|
|
1524
|
-
const report = await store.get(reportId);
|
|
1525
|
-
|
|
1526
|
-
|
|
1527
|
-
|
|
1528
|
-
|
|
1529
|
-
|
|
1263
|
+
const report = requireEvaluationReport(await store.get(reportId), reportId, lang);
|
|
1264
|
+
const { computeVerdict, formatVerdictText } = await import('../eval-core/verdict.js');
|
|
1265
|
+
const { reportComparabilityWarnings, formatComparabilityWarnings } = await import('../eval-core/comparability.js');
|
|
1266
|
+
const comparability = formatComparabilityWarnings(reportComparabilityWarnings(report), lang);
|
|
1267
|
+
if (comparability)
|
|
1268
|
+
process.stderr.write(`${comparability}\n`);
|
|
1530
1269
|
const result = computeVerdict(report, {
|
|
1531
1270
|
gateThreshold: values.threshold != null ? Number(values.threshold) : undefined,
|
|
1532
1271
|
triviallySmallDiff: values['trivial-diff'] != null ? Number(values['trivial-diff']) : undefined,
|
|
@@ -1568,13 +1307,9 @@ async function handleDiagnose(argv) {
|
|
|
1568
1307
|
},
|
|
1569
1308
|
strict: false,
|
|
1570
1309
|
});
|
|
1571
|
-
const { createFileStore } = await import('
|
|
1310
|
+
const { createFileStore } = await import('../server/report-store.js');
|
|
1572
1311
|
const store = createFileStore(resolve(values['reports-dir']));
|
|
1573
|
-
const report = await store.get(reportId);
|
|
1574
|
-
if (!report) {
|
|
1575
|
-
console.error(tCli('cli.common.report_not_found', lang, { id: reportId }));
|
|
1576
|
-
process.exit(1);
|
|
1577
|
-
}
|
|
1312
|
+
const report = requireEvaluationReport(await store.get(reportId), reportId, lang);
|
|
1578
1313
|
// Try to read the samples file for near-duplicate detection. Source order:
|
|
1579
1314
|
// 1. --samples <path> override
|
|
1580
1315
|
// 2. report.meta.request.samplesPath (recorded at run time)
|
|
@@ -1583,7 +1318,7 @@ async function handleDiagnose(argv) {
|
|
|
1583
1318
|
const samplesPath = values.samples ?? report.meta?.request?.samplesPath;
|
|
1584
1319
|
if (samplesPath && existsSync(samplesPath)) {
|
|
1585
1320
|
try {
|
|
1586
|
-
const { loadSamples } = await import('
|
|
1321
|
+
const { loadSamples } = await import('../inputs/load-samples.js');
|
|
1587
1322
|
samples = loadSamples(samplesPath).samples;
|
|
1588
1323
|
}
|
|
1589
1324
|
catch (err) {
|
|
@@ -1594,7 +1329,7 @@ async function handleDiagnose(argv) {
|
|
|
1594
1329
|
}
|
|
1595
1330
|
const topRaw = Number(values.top);
|
|
1596
1331
|
const topN = Number.isFinite(topRaw) && topRaw > 0 ? topRaw : undefined;
|
|
1597
|
-
const { diagnoseSamples, formatSampleDiagnostics } = await import('
|
|
1332
|
+
const { diagnoseSamples, formatSampleDiagnostics } = await import('../analysis/sample-diagnostics.js');
|
|
1598
1333
|
const diag = diagnoseSamples(report, {
|
|
1599
1334
|
samples,
|
|
1600
1335
|
duplicateRouge: values['duplicate-rouge'] != null ? Number(values['duplicate-rouge']) : undefined,
|
|
@@ -1603,7 +1338,15 @@ async function handleDiagnose(argv) {
|
|
|
1603
1338
|
latencyOutlierK: values['latency-k'] != null ? Number(values['latency-k']) : undefined,
|
|
1604
1339
|
flatThreshold: values.flat != null ? Number(values.flat) : undefined,
|
|
1605
1340
|
});
|
|
1606
|
-
console.log(formatSampleDiagnostics(diag, { topN }));
|
|
1341
|
+
console.log(formatSampleDiagnostics(diag, { topN, lang }));
|
|
1342
|
+
// Sample design science coverage block. Render after diagnose 主体,因为
|
|
1343
|
+
// coverage 是声明式元数据(capability/difficulty/construct/provenance)的整体分布,
|
|
1344
|
+
// 跟 issue list 是不同视角的两件事。优先从 samples (现场加载) 算,fallback 到
|
|
1345
|
+
// report.analysis.sampleQuality(报告里持久化的数据)。
|
|
1346
|
+
const { renderSampleDesignCoverage } = await import('./coverage-renderer.js');
|
|
1347
|
+
const coverageBlock = renderSampleDesignCoverage(samples, report.analysis?.sampleQuality, lang);
|
|
1348
|
+
if (coverageBlock)
|
|
1349
|
+
console.log(coverageBlock);
|
|
1607
1350
|
// Exit code: 0 if health ≥ 70 and no errors; 1 otherwise. CI-friendly.
|
|
1608
1351
|
if (diag.totals.errors === 0 && diag.healthScore >= 70) {
|
|
1609
1352
|
process.exit(0);
|
|
@@ -1633,23 +1376,19 @@ async function handleFailures(argv) {
|
|
|
1633
1376
|
},
|
|
1634
1377
|
strict: false,
|
|
1635
1378
|
});
|
|
1636
|
-
const { createFileStore } = await import('
|
|
1379
|
+
const { createFileStore } = await import('../server/report-store.js');
|
|
1637
1380
|
const store = createFileStore(resolve(values['reports-dir']));
|
|
1638
|
-
const report = await store.get(reportId);
|
|
1639
|
-
if (!report) {
|
|
1640
|
-
console.error(tCli('cli.common.report_not_found', lang, { id: reportId }));
|
|
1641
|
-
process.exit(1);
|
|
1642
|
-
}
|
|
1381
|
+
const report = requireEvaluationReport(await store.get(reportId), reportId, lang);
|
|
1643
1382
|
const judgeModel = values['judge-model'] ?? report.meta?.judgeModel;
|
|
1644
1383
|
if (!judgeModel) {
|
|
1645
1384
|
console.error(tCli('cli.common.no_judge_model', lang));
|
|
1646
1385
|
process.exit(1);
|
|
1647
1386
|
}
|
|
1648
|
-
const { createExecutor } = await import('
|
|
1387
|
+
const { createExecutor } = await import('../executors/index.js');
|
|
1649
1388
|
const executor = createExecutor(values['judge-executor']);
|
|
1650
|
-
const { clusterFailures, formatFailureClusterReport } = await import('
|
|
1389
|
+
const { clusterFailures, formatFailureClusterReport } = await import('../analysis/failure-clusterer.js');
|
|
1651
1390
|
const out = await clusterFailures({
|
|
1652
|
-
report
|
|
1391
|
+
report,
|
|
1653
1392
|
executor,
|
|
1654
1393
|
judgeModel,
|
|
1655
1394
|
maxClusters: Number(values['max-clusters']) || 5,
|
|
@@ -1662,4 +1401,4 @@ async function handleFailures(argv) {
|
|
|
1662
1401
|
// Entry
|
|
1663
1402
|
// ---------------------------------------------------------------------------
|
|
1664
1403
|
main();
|
|
1665
|
-
//# sourceMappingURL=
|
|
1404
|
+
//# sourceMappingURL=index.js.map
|