oh-my-knowledge 0.22.0 → 0.23.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -0
- package/README.zh.md +1 -0
- package/dist/src/analysis/report-diagnostics.d.ts +26 -2
- package/dist/src/analysis/report-diagnostics.d.ts.map +1 -1
- package/dist/src/analysis/report-diagnostics.js +96 -3
- package/dist/src/analysis/report-diagnostics.js.map +1 -1
- package/dist/src/analysis/sample-diagnostics.d.ts +5 -1
- package/dist/src/analysis/sample-diagnostics.d.ts.map +1 -1
- package/dist/src/analysis/sample-diagnostics.js +105 -0
- package/dist/src/analysis/sample-diagnostics.js.map +1 -1
- package/dist/src/authoring/evolver.d.ts +2 -2
- package/dist/src/authoring/evolver.d.ts.map +1 -1
- package/dist/src/authoring/evolver.js +24 -7
- package/dist/src/authoring/evolver.js.map +1 -1
- package/dist/src/authoring/generator.d.ts +24 -0
- package/dist/src/authoring/generator.d.ts.map +1 -1
- package/dist/src/authoring/generator.js +67 -5
- package/dist/src/authoring/generator.js.map +1 -1
- package/dist/src/cli/coverage-renderer.d.ts +15 -0
- package/dist/src/cli/coverage-renderer.d.ts.map +1 -0
- package/dist/src/cli/coverage-renderer.js +74 -0
- package/dist/src/cli/coverage-renderer.js.map +1 -0
- package/dist/src/cli/i18n-dict.d.ts +1 -1
- package/dist/src/cli/i18n-dict.d.ts.map +1 -1
- package/dist/src/cli/i18n-dict.js +34 -16
- package/dist/src/cli/i18n-dict.js.map +1 -1
- package/dist/src/cli/index.d.ts +3 -0
- package/dist/src/cli/index.d.ts.map +1 -0
- package/dist/src/{cli.js → cli/index.js} +48 -323
- package/dist/src/cli/index.js.map +1 -0
- package/dist/src/cli/parse-run-config.d.ts +56 -0
- package/dist/src/cli/parse-run-config.d.ts.map +1 -0
- package/dist/src/cli/parse-run-config.js +195 -0
- package/dist/src/cli/parse-run-config.js.map +1 -0
- package/dist/src/cli/progress.d.ts +24 -0
- package/dist/src/cli/progress.d.ts.map +1 -0
- package/dist/src/cli/progress.js +55 -0
- package/dist/src/cli/progress.js.map +1 -0
- package/dist/src/cli/update-check.d.ts +3 -0
- package/dist/src/cli/update-check.d.ts.map +1 -0
- package/dist/src/cli/update-check.js +37 -0
- package/dist/src/cli/update-check.js.map +1 -0
- package/dist/src/eval-core/cache.d.ts +1 -1
- package/dist/src/eval-core/cache.js +1 -1
- package/dist/src/eval-core/dependency-checker.d.ts +1 -1
- package/dist/src/eval-core/dependency-checker.js +1 -1
- package/dist/src/eval-core/evaluation-execution.d.ts +1 -1
- package/dist/src/eval-core/evaluation-execution.js +4 -4
- package/dist/src/eval-core/evaluation-execution.js.map +1 -1
- package/dist/src/eval-core/evaluation-reporting.js +1 -1
- package/dist/src/eval-core/evaluation-reporting.js.map +1 -1
- package/dist/src/eval-core/execution-strategy.js +4 -4
- package/dist/src/eval-core/execution-strategy.js.map +1 -1
- package/dist/src/eval-core/schema.js +1 -1
- package/dist/src/eval-core/schema.js.map +1 -1
- package/dist/src/eval-core/verdict.d.ts +3 -3
- package/dist/src/eval-core/verdict.js +3 -3
- package/dist/src/eval-workflows/each-evaluation-workflow.d.ts +2 -2
- package/dist/src/eval-workflows/each-evaluation-workflow.js +1 -1
- package/dist/src/eval-workflows/each-evaluation-workflow.js.map +1 -1
- package/dist/src/eval-workflows/evaluation-pipeline.d.ts +2 -2
- package/dist/src/eval-workflows/evaluation-pipeline.d.ts.map +1 -1
- package/dist/src/eval-workflows/evaluation-pipeline.js +7 -3
- package/dist/src/eval-workflows/evaluation-pipeline.js.map +1 -1
- package/dist/src/eval-workflows/evaluation-preparation.d.ts +2 -2
- package/dist/src/eval-workflows/evaluation-preparation.d.ts.map +1 -1
- package/dist/src/eval-workflows/run-evaluation.d.ts +5 -5
- package/dist/src/eval-workflows/run-evaluation.d.ts.map +1 -1
- package/dist/src/eval-workflows/run-evaluation.js +9 -10
- package/dist/src/eval-workflows/run-evaluation.js.map +1 -1
- package/dist/src/executors/claude-cli.js +1 -1
- package/dist/src/executors/claude-cli.js.map +1 -1
- package/dist/src/executors/claude-sdk.d.ts +1 -1
- package/dist/src/executors/claude-sdk.js +1 -1
- package/dist/src/executors/script.js +1 -1
- package/dist/src/executors/script.js.map +1 -1
- package/dist/src/grading/assertions.js +3 -3
- package/dist/src/grading/assertions.js.map +1 -1
- package/dist/src/grading/gold-dataset.d.ts +1 -1
- package/dist/src/grading/gold-dataset.js +1 -1
- package/dist/src/inputs/eval-config.js +2 -2
- package/dist/src/inputs/eval-config.js.map +1 -1
- package/dist/src/inputs/load-samples.d.ts.map +1 -1
- package/dist/src/inputs/load-samples.js +30 -4
- package/dist/src/inputs/load-samples.js.map +1 -1
- package/dist/src/inputs/skill-loader.d.ts +1 -1
- package/dist/src/inputs/skill-loader.d.ts.map +1 -1
- package/dist/src/inputs/skill-loader.js +1 -1
- package/dist/src/inputs/skill-loader.js.map +1 -1
- package/dist/src/renderer/summary.d.ts.map +1 -1
- package/dist/src/renderer/summary.js +0 -13
- package/dist/src/renderer/summary.js.map +1 -1
- package/dist/src/types/eval.d.ts +23 -3
- package/dist/src/types/eval.d.ts.map +1 -1
- package/dist/src/types/report.d.ts +31 -5
- package/dist/src/types/report.d.ts.map +1 -1
- package/package.json +13 -5
- package/dist/src/cli.d.ts +0 -3
- package/dist/src/cli.d.ts.map +0 -1
- package/dist/src/cli.js.map +0 -1
|
@@ -1,242 +1,12 @@
|
|
|
1
1
|
#!/usr/bin/env node
|
|
2
2
|
import { parseArgs } from 'node:util';
|
|
3
3
|
import { resolve } from 'node:path';
|
|
4
|
-
import { homedir } from 'node:os';
|
|
5
4
|
import { join } from 'node:path';
|
|
6
5
|
import { existsSync } from 'node:fs';
|
|
7
|
-
import { tCli, getCliLang, parseLangFromArgv, langFromArgv } from './
|
|
8
|
-
import {
|
|
9
|
-
import {
|
|
10
|
-
|
|
11
|
-
// Constants
|
|
12
|
-
// ---------------------------------------------------------------------------
|
|
13
|
-
const DEFAULT_REPORTS_DIR = join(homedir(), '.oh-my-knowledge', 'reports');
|
|
14
|
-
// Shared CLI options for run/ci commands.
|
|
15
|
-
// Defaults are applied inside parseRunConfig (after config-file merge) so that
|
|
16
|
-
// CLI `undefined` can be reliably distinguished from "user passed the default value".
|
|
17
|
-
// Priority order resolved in parseRunConfig: CLI arg > --config file > hard-coded default.
|
|
18
|
-
/**
|
|
19
|
-
* 所有子命令都接受的通用 flag。新增 --lang 让 parseArgs strict:false 模式下
|
|
20
|
-
* 仍能把值类型化到 values.lang 上(否则未声明的 flag 会被丢弃)。
|
|
21
|
-
*/
|
|
22
|
-
const COMMON_OPTIONS = {
|
|
23
|
-
lang: { type: 'string' },
|
|
24
|
-
};
|
|
25
|
-
const RUN_OPTIONS = {
|
|
26
|
-
...COMMON_OPTIONS,
|
|
27
|
-
samples: { type: 'string' },
|
|
28
|
-
'skill-dir': { type: 'string' },
|
|
29
|
-
control: { type: 'string' },
|
|
30
|
-
treatment: { type: 'string' },
|
|
31
|
-
config: { type: 'string' },
|
|
32
|
-
model: { type: 'string' },
|
|
33
|
-
'judge-model': { type: 'string' },
|
|
34
|
-
'output-dir': { type: 'string' },
|
|
35
|
-
'no-judge': { type: 'boolean' },
|
|
36
|
-
'no-cache': { type: 'boolean' },
|
|
37
|
-
'dry-run': { type: 'boolean' },
|
|
38
|
-
concurrency: { type: 'string' },
|
|
39
|
-
timeout: { type: 'string' },
|
|
40
|
-
executor: { type: 'string' },
|
|
41
|
-
'judge-executor': { type: 'string' },
|
|
42
|
-
each: { type: 'boolean' },
|
|
43
|
-
'skip-preflight': { type: 'boolean' },
|
|
44
|
-
'mcp-config': { type: 'string' },
|
|
45
|
-
'no-serve': { type: 'boolean' },
|
|
46
|
-
verbose: { type: 'boolean' },
|
|
47
|
-
retry: { type: 'string' },
|
|
48
|
-
resume: { type: 'string' },
|
|
49
|
-
'layered-stats': { type: 'boolean' },
|
|
50
|
-
// v0.22 — strict-baseline default true. Declare both forms; reconcile in
|
|
51
|
-
// parseRunConfig (后者赢)。strict-baseline 没传 + no-strict-baseline 没传 = default true。
|
|
52
|
-
'strict-baseline': { type: 'boolean' },
|
|
53
|
-
'no-strict-baseline': { type: 'boolean' },
|
|
54
|
-
};
|
|
55
|
-
// ---------------------------------------------------------------------------
|
|
56
|
-
// parseRunConfig
|
|
57
|
-
// ---------------------------------------------------------------------------
|
|
58
|
-
function parseRunConfig(argv, extraOptions = {}) {
|
|
59
|
-
const { values } = parseArgs({
|
|
60
|
-
args: argv,
|
|
61
|
-
options: { ...RUN_OPTIONS, ...extraOptions },
|
|
62
|
-
strict: false,
|
|
63
|
-
});
|
|
64
|
-
if (values.variants !== undefined) {
|
|
65
|
-
throw new Error(`--variants 已在 v0.16 废除,请改用 --control <expr> 与 --treatment <v1,v2,...>\n`
|
|
66
|
-
+ ` 迁移示例:--variants baseline,my-skill → --control baseline --treatment my-skill\n`
|
|
67
|
-
+ ` 复杂场景可用 --config eval.yaml(参见 docs/terminology-spec.md)`);
|
|
68
|
-
}
|
|
69
|
-
// 1) Load --config (if provided). All subsequent fields fall back to it when CLI is silent.
|
|
70
|
-
const evalConfig = values.config
|
|
71
|
-
? loadEvalConfig(values.config)
|
|
72
|
-
: null;
|
|
73
|
-
// 2) Resolve samples path: CLI > config > auto-detect .json/.yaml/.yml in cwd.
|
|
74
|
-
const cliSamples = values.samples;
|
|
75
|
-
let samplesFile;
|
|
76
|
-
if (cliSamples) {
|
|
77
|
-
samplesFile = cliSamples;
|
|
78
|
-
}
|
|
79
|
-
else if (evalConfig?.samples) {
|
|
80
|
-
samplesFile = evalConfig.samples; // already resolved against config file dir
|
|
81
|
-
}
|
|
82
|
-
else {
|
|
83
|
-
samplesFile = 'eval-samples.json';
|
|
84
|
-
if (!existsSync(resolve(samplesFile))) {
|
|
85
|
-
if (existsSync(resolve('eval-samples.yaml')))
|
|
86
|
-
samplesFile = 'eval-samples.yaml';
|
|
87
|
-
else if (existsSync(resolve('eval-samples.yml')))
|
|
88
|
-
samplesFile = 'eval-samples.yml';
|
|
89
|
-
}
|
|
90
|
-
}
|
|
91
|
-
const skillDir = resolve(values['skill-dir'] ?? 'skills');
|
|
92
|
-
// 3) Resolve variantSpecs: CLI > config. If neither, error with a helpful hint.
|
|
93
|
-
const controlExpr = values.control;
|
|
94
|
-
const treatmentExprs = values.treatment
|
|
95
|
-
? values.treatment.split(',').map((v) => v.trim()).filter(Boolean)
|
|
96
|
-
: [];
|
|
97
|
-
let variantSpecs;
|
|
98
|
-
if (controlExpr || treatmentExprs.length > 0) {
|
|
99
|
-
// CLI roles present → CLI entirely replaces config.variants (no merging).
|
|
100
|
-
variantSpecs = [];
|
|
101
|
-
if (controlExpr) {
|
|
102
|
-
variantSpecs.push({ name: parseVariantCwd(controlExpr).name, role: 'control', expr: controlExpr });
|
|
103
|
-
}
|
|
104
|
-
for (const expr of treatmentExprs) {
|
|
105
|
-
variantSpecs.push({ name: parseVariantCwd(expr).name, role: 'treatment', expr });
|
|
106
|
-
}
|
|
107
|
-
}
|
|
108
|
-
else if (evalConfig) {
|
|
109
|
-
variantSpecs = configVariantsToSpecs(evalConfig.variants);
|
|
110
|
-
}
|
|
111
|
-
else if (values.each) {
|
|
112
|
-
// --each 模式自动用 baseline (control) vs 每个 skill (treatment),
|
|
113
|
-
// 不需要用户显式传 --control / --treatment,校验跳过。
|
|
114
|
-
variantSpecs = [];
|
|
115
|
-
}
|
|
116
|
-
else {
|
|
117
|
-
const discovered = discoverVariants(skillDir);
|
|
118
|
-
const hint = discovered.length > 0 ? `\n skill-dir (${skillDir}) 下发现的候选:${discovered.join(', ')}` : '';
|
|
119
|
-
throw new Error(`请通过 --control / --treatment 或 --config eval.yaml 声明 variant 角色。\n`
|
|
120
|
-
+ ` 示例:omk bench run --control baseline --treatment my-skill${hint}\n`
|
|
121
|
-
+ ` --each 模式下自动用 baseline vs 每个 skill,无需显式声明\n`
|
|
122
|
-
+ ` 术语见 docs/terminology-spec.md(v0.16 起废除 --variants,改用 experiment role 显式声明)`);
|
|
123
|
-
}
|
|
124
|
-
const seenNames = new Set();
|
|
125
|
-
for (const spec of variantSpecs) {
|
|
126
|
-
if (seenNames.has(spec.name)) {
|
|
127
|
-
throw new Error(`variant "${spec.name}" 重复出现——同一 variant 不能同时属于 --control 与 --treatment,也不能在 --treatment 中重复。`);
|
|
128
|
-
}
|
|
129
|
-
seenNames.add(spec.name);
|
|
130
|
-
}
|
|
131
|
-
// 4) Apply CLI > config > hard-coded default for all other fields.
|
|
132
|
-
const executorName = values.executor ?? evalConfig?.executor ?? 'claude';
|
|
133
|
-
const judgeExecutorName = values['judge-executor'] ?? evalConfig?.judgeExecutor ?? executorName;
|
|
134
|
-
const model = values.model ?? evalConfig?.model ?? 'sonnet';
|
|
135
|
-
const judgeModelRaw = values['judge-model'] !== undefined
|
|
136
|
-
? values['judge-model']
|
|
137
|
-
: evalConfig?.judgeModel ?? 'haiku';
|
|
138
|
-
const judgeModel = judgeModelRaw ?? 'haiku';
|
|
139
|
-
const outputDir = resolve(values['output-dir'] ?? DEFAULT_REPORTS_DIR);
|
|
140
|
-
const concurrencyRaw = values.concurrency !== undefined
|
|
141
|
-
? Number(values.concurrency)
|
|
142
|
-
: evalConfig?.concurrency ?? 1;
|
|
143
|
-
const concurrency = Math.max(1, Number(concurrencyRaw) || 1);
|
|
144
|
-
const timeoutSec = values.timeout !== undefined
|
|
145
|
-
? Number(values.timeout)
|
|
146
|
-
: evalConfig?.timeoutMs
|
|
147
|
-
? evalConfig.timeoutMs / 1000
|
|
148
|
-
: 120;
|
|
149
|
-
const timeoutMs = Math.max(1, Number(timeoutSec) || 120) * 1000;
|
|
150
|
-
const noJudge = values['no-judge'] ?? false;
|
|
151
|
-
const noCache = values['no-cache'] ?? evalConfig?.noCache ?? false;
|
|
152
|
-
const dryRun = values['dry-run'] ?? false;
|
|
153
|
-
const skipPreflight = values['skip-preflight'] ?? false;
|
|
154
|
-
const mcpConfig = values['mcp-config'] ?? evalConfig?.mcpConfig;
|
|
155
|
-
const verbose = values.verbose ?? false;
|
|
156
|
-
const retry = Math.max(0, Number(values.retry ?? 0) || 0);
|
|
157
|
-
const resume = values.resume;
|
|
158
|
-
const blind = values.blind ?? evalConfig?.blind ?? false;
|
|
159
|
-
const layeredStats = values['layered-stats'] ?? false;
|
|
160
|
-
// v0.22 — strict-baseline default true. Reconcile both flag forms.
|
|
161
|
-
// Priority: --no-strict-baseline > --strict-baseline > undefined(=true).
|
|
162
|
-
const noStrictFlag = values['no-strict-baseline'];
|
|
163
|
-
const strictFlag = values['strict-baseline'];
|
|
164
|
-
const strictBaseline = noStrictFlag === true ? false : (strictFlag ?? true);
|
|
165
|
-
// v0.22 — extract eval.yaml variant.allowedSkills overrides (per-variant). Always
|
|
166
|
-
// wins over strictBaseline default. Empty object when no eval.yaml or no overrides.
|
|
167
|
-
const variantAllowedSkills = {};
|
|
168
|
-
if (evalConfig?.variants) {
|
|
169
|
-
for (const v of evalConfig.variants) {
|
|
170
|
-
if (v.allowedSkills !== undefined) {
|
|
171
|
-
variantAllowedSkills[v.name] = v.allowedSkills;
|
|
172
|
-
}
|
|
173
|
-
}
|
|
174
|
-
}
|
|
175
|
-
return {
|
|
176
|
-
values,
|
|
177
|
-
config: {
|
|
178
|
-
samplesPath: resolve(samplesFile),
|
|
179
|
-
skillDir,
|
|
180
|
-
variantSpecs,
|
|
181
|
-
model,
|
|
182
|
-
judgeModel,
|
|
183
|
-
outputDir,
|
|
184
|
-
noJudge,
|
|
185
|
-
noCache,
|
|
186
|
-
dryRun,
|
|
187
|
-
concurrency,
|
|
188
|
-
timeoutMs,
|
|
189
|
-
executorName,
|
|
190
|
-
judgeExecutorName,
|
|
191
|
-
skipPreflight,
|
|
192
|
-
mcpConfig,
|
|
193
|
-
verbose,
|
|
194
|
-
retry,
|
|
195
|
-
resume,
|
|
196
|
-
blind,
|
|
197
|
-
layeredStats,
|
|
198
|
-
budget: evalConfig?.budget,
|
|
199
|
-
strictBaseline,
|
|
200
|
-
...(Object.keys(variantAllowedSkills).length > 0 && { variantAllowedSkills }),
|
|
201
|
-
},
|
|
202
|
-
};
|
|
203
|
-
}
|
|
204
|
-
// ---------------------------------------------------------------------------
|
|
205
|
-
// Update check
|
|
206
|
-
// ---------------------------------------------------------------------------
|
|
207
|
-
async function checkUpdate(lang) {
|
|
208
|
-
try {
|
|
209
|
-
const { readFileSync } = await import('node:fs');
|
|
210
|
-
const { fileURLToPath } = await import('node:url');
|
|
211
|
-
const { dirname, join } = await import('node:path');
|
|
212
|
-
const __dirname = dirname(fileURLToPath(import.meta.url));
|
|
213
|
-
const findPackageJson = (startDir) => {
|
|
214
|
-
let dir = startDir;
|
|
215
|
-
for (let i = 0; i < 5; i++) {
|
|
216
|
-
const candidate = join(dir, 'package.json');
|
|
217
|
-
if (existsSync(candidate))
|
|
218
|
-
return candidate;
|
|
219
|
-
dir = dirname(dir);
|
|
220
|
-
}
|
|
221
|
-
return null;
|
|
222
|
-
};
|
|
223
|
-
const pkgPath = findPackageJson(__dirname);
|
|
224
|
-
if (!pkgPath)
|
|
225
|
-
return;
|
|
226
|
-
const pkg = JSON.parse(readFileSync(pkgPath, 'utf-8'));
|
|
227
|
-
const registry = pkg.publishConfig?.registry || 'https://registry.npmjs.org';
|
|
228
|
-
const res = await fetch(`${registry}/${pkg.name}/latest`, { signal: AbortSignal.timeout(3000) });
|
|
229
|
-
if (!res.ok)
|
|
230
|
-
return;
|
|
231
|
-
const data = await res.json();
|
|
232
|
-
if (data.version && data.version !== pkg.version) {
|
|
233
|
-
process.stderr.write(tCli('cli.update.new_version_available', lang, {
|
|
234
|
-
old: pkg.version, new: data.version, pkg: pkg.name,
|
|
235
|
-
}));
|
|
236
|
-
}
|
|
237
|
-
}
|
|
238
|
-
catch { /* 静默失败,不影响正常使用 */ }
|
|
239
|
-
}
|
|
6
|
+
import { tCli, getCliLang, parseLangFromArgv, langFromArgv } from './i18n.js';
|
|
7
|
+
import { parseRunConfig, DEFAULT_REPORTS_DIR, COMMON_OPTIONS, } from './parse-run-config.js';
|
|
8
|
+
import { makeOnProgress } from './progress.js';
|
|
9
|
+
import { checkUpdate } from './update-check.js';
|
|
240
10
|
// ---------------------------------------------------------------------------
|
|
241
11
|
// Main
|
|
242
12
|
// ---------------------------------------------------------------------------
|
|
@@ -313,59 +83,6 @@ async function main() {
|
|
|
313
83
|
* Factory: 闭住 lang, 返回 onProgress callback。evaluation engine 回调时不传
|
|
314
84
|
* 上下文, 所以 lang 必须在 handler 入口处通过 closure 传进来。
|
|
315
85
|
*/
|
|
316
|
-
function makeOnProgress(lang) {
|
|
317
|
-
return ({ phase, completed, total, sample_id, variant, durationMs, inputTokens, outputTokens, costUSD, score, outputPreview, judgePhase: _judgePhase, judgeDim, skipped, attempt, maxAttempts, error, }) => {
|
|
318
|
-
const ctx = { i: completed ?? '', n: total ?? '', sample: sample_id ?? '', variant: variant ?? '' };
|
|
319
|
-
if (phase === 'preflight') {
|
|
320
|
-
process.stderr.write(tCli('cli.progress.preflight_starting', lang));
|
|
321
|
-
return;
|
|
322
|
-
}
|
|
323
|
-
if (phase === 'retry') {
|
|
324
|
-
process.stderr.write(tCli('cli.progress.sample_retry', lang, {
|
|
325
|
-
...ctx, attempt: attempt ?? '', max: maxAttempts ?? '',
|
|
326
|
-
}));
|
|
327
|
-
return;
|
|
328
|
-
}
|
|
329
|
-
if (phase === 'error') {
|
|
330
|
-
process.stderr.write(tCli('cli.progress.sample_error', lang, { ...ctx, error: error ?? '' }));
|
|
331
|
-
return;
|
|
332
|
-
}
|
|
333
|
-
if (phase === 'start') {
|
|
334
|
-
process.stderr.write(tCli('cli.progress.sample_executing', lang, ctx));
|
|
335
|
-
}
|
|
336
|
-
else if (phase === 'exec_done') {
|
|
337
|
-
const cost = costUSD != null && costUSD > 0 ? ` $${costUSD.toFixed(4)}` : '';
|
|
338
|
-
process.stderr.write(tCli('cli.progress.sample_exec_done', lang, {
|
|
339
|
-
...ctx, ms: durationMs ?? '', input: inputTokens ?? '', output: outputTokens ?? '', cost,
|
|
340
|
-
}));
|
|
341
|
-
if (outputPreview) {
|
|
342
|
-
process.stderr.write(tCli('cli.progress.output_preview', lang, {
|
|
343
|
-
preview: outputPreview.slice(0, 150).replace(/\n/g, ' '),
|
|
344
|
-
}));
|
|
345
|
-
}
|
|
346
|
-
}
|
|
347
|
-
else if (phase === 'grading') {
|
|
348
|
-
const dim = judgeDim ? ` [${judgeDim}]` : '';
|
|
349
|
-
process.stderr.write(tCli('cli.progress.judging', lang, { ...ctx, dim }));
|
|
350
|
-
}
|
|
351
|
-
else if (phase === 'judge_done') {
|
|
352
|
-
const dim = judgeDim ? ` [${judgeDim}]` : '';
|
|
353
|
-
process.stderr.write(tCli('cli.progress.judged', lang, { ...ctx, dim, score: score ?? '' }));
|
|
354
|
-
}
|
|
355
|
-
else if (phase === 'done' && skipped) {
|
|
356
|
-
if (sample_id)
|
|
357
|
-
process.stderr.write(tCli('cli.progress.skipped', lang, ctx));
|
|
358
|
-
}
|
|
359
|
-
else {
|
|
360
|
-
const cost = costUSD != null && costUSD > 0 ? ` $${costUSD.toFixed(4)}` : '';
|
|
361
|
-
const scoreInfo = typeof score === 'number' ? ` score=${score}` : '';
|
|
362
|
-
process.stderr.write(tCli('cli.progress.sample_done', lang, {
|
|
363
|
-
...ctx, ms: durationMs ?? '', input: inputTokens ?? '', output: outputTokens ?? '',
|
|
364
|
-
cost, score: scoreInfo,
|
|
365
|
-
}));
|
|
366
|
-
}
|
|
367
|
-
};
|
|
368
|
-
}
|
|
369
86
|
// ---------------------------------------------------------------------------
|
|
370
87
|
// handleRun
|
|
371
88
|
// ---------------------------------------------------------------------------
|
|
@@ -384,7 +101,7 @@ async function handleRun(argv) {
|
|
|
384
101
|
'budget-per-sample-usd': { type: 'string' },
|
|
385
102
|
'budget-per-sample-ms': { type: 'string' },
|
|
386
103
|
});
|
|
387
|
-
const { runEvaluation, runMultiple, runEachEvaluation } = await import('
|
|
104
|
+
const { runEvaluation, runMultiple, runEachEvaluation } = await import('../eval-workflows/run-evaluation.js');
|
|
388
105
|
if (values.blind !== undefined) {
|
|
389
106
|
config.blind = values.blind;
|
|
390
107
|
}
|
|
@@ -430,7 +147,7 @@ async function handleRun(argv) {
|
|
|
430
147
|
}
|
|
431
148
|
}
|
|
432
149
|
// --budget-usd / --budget-per-sample-usd / --budget-per-sample-ms:
|
|
433
|
-
//
|
|
150
|
+
// hard budget caps. CLI flags override config-file values. When the
|
|
434
151
|
// total-USD cap is exceeded mid-run, remaining tasks are skipped and a
|
|
435
152
|
// partial report is persisted with meta.budgetExhausted=true.
|
|
436
153
|
const budgetUSD = values['budget-usd'] != null ? Number(values['budget-usd']) : undefined;
|
|
@@ -483,7 +200,7 @@ async function handleRun(argv) {
|
|
|
483
200
|
process.stderr.write(tCli('cli.run.batch_complete', lang));
|
|
484
201
|
process.stderr.write(tCli('cli.run.report_saved', lang, { path: filePath }));
|
|
485
202
|
if (!values['no-serve'] && process.stdout.isTTY) {
|
|
486
|
-
const { createReportServer } = await import('
|
|
203
|
+
const { createReportServer } = await import('../server/report-server.js');
|
|
487
204
|
const server = createReportServer({ reportsDir: config.outputDir });
|
|
488
205
|
const serverUrl = await server.start();
|
|
489
206
|
const reportUrl = `${serverUrl}/reports/${report.id}`;
|
|
@@ -523,7 +240,7 @@ async function handleRun(argv) {
|
|
|
523
240
|
// --gold-dir: compute α/κ/Pearson against gold annotations and re-persist.
|
|
524
241
|
const goldDir = values['gold-dir'];
|
|
525
242
|
if (goldDir && filePath) {
|
|
526
|
-
const { attachGoldAgreementToReport, formatGoldCompare } = await import('
|
|
243
|
+
const { attachGoldAgreementToReport, formatGoldCompare } = await import('../grading/gold-cli.js');
|
|
527
244
|
const out = attachGoldAgreementToReport({
|
|
528
245
|
report,
|
|
529
246
|
goldDir,
|
|
@@ -551,7 +268,7 @@ async function handleRun(argv) {
|
|
|
551
268
|
process.stderr.write(tCli('cli.run.report_saved', lang, { path: filePath }));
|
|
552
269
|
if (!values['no-serve'] && process.stdout.isTTY) {
|
|
553
270
|
// Auto-start report server
|
|
554
|
-
const { createReportServer } = await import('
|
|
271
|
+
const { createReportServer } = await import('../server/report-server.js');
|
|
555
272
|
const server = createReportServer({
|
|
556
273
|
reportsDir: config.outputDir,
|
|
557
274
|
});
|
|
@@ -612,8 +329,8 @@ async function handleReport(argv) {
|
|
|
612
329
|
return;
|
|
613
330
|
}
|
|
614
331
|
if (values.export) {
|
|
615
|
-
const { createFileStore } = await import('
|
|
616
|
-
const { renderRunDetail, renderEachRunDetail } = await import('
|
|
332
|
+
const { createFileStore } = await import('../server/report-store.js');
|
|
333
|
+
const { renderRunDetail, renderEachRunDetail } = await import('../renderer/html-renderer.js');
|
|
617
334
|
const { writeFileSync } = await import('node:fs');
|
|
618
335
|
const store = createFileStore(resolve(values['reports-dir']));
|
|
619
336
|
const report = await store.get(values.export);
|
|
@@ -628,7 +345,7 @@ async function handleReport(argv) {
|
|
|
628
345
|
console.log('Open in browser, or Ctrl+P to save as PDF');
|
|
629
346
|
return;
|
|
630
347
|
}
|
|
631
|
-
const { createReportServer } = await import('
|
|
348
|
+
const { createReportServer } = await import('../server/report-server.js');
|
|
632
349
|
const server = createReportServer({
|
|
633
350
|
port: Number(values.port),
|
|
634
351
|
reportsDir: resolve(values['reports-dir']),
|
|
@@ -751,7 +468,7 @@ async function handleAnalyze(argv) {
|
|
|
751
468
|
const to = values.to;
|
|
752
469
|
const skills = values.skills ? values.skills.split(',').map((s) => s.trim()).filter(Boolean) : undefined;
|
|
753
470
|
console.log(`[omk] analyzing ${tracePath}...`);
|
|
754
|
-
const { computeSkillHealthReport } = await import('
|
|
471
|
+
const { computeSkillHealthReport } = await import('../observability/skill-health-analyzer.js');
|
|
755
472
|
const report = computeSkillHealthReport(tracePath, {
|
|
756
473
|
kbRoot: values.kb ? resolve(values.kb) : undefined,
|
|
757
474
|
from,
|
|
@@ -812,7 +529,7 @@ async function handleGenSamples(argv) {
|
|
|
812
529
|
strict: false,
|
|
813
530
|
allowPositionals: true,
|
|
814
531
|
});
|
|
815
|
-
const { generateSamples } = await import('
|
|
532
|
+
const { generateSamples } = await import('../authoring/generator.js');
|
|
816
533
|
const { readFileSync, writeFileSync } = await import('node:fs');
|
|
817
534
|
const count = Math.max(1, Number(values.count) || 5);
|
|
818
535
|
const model = values.model;
|
|
@@ -944,7 +661,7 @@ async function handleEvolve(argv) {
|
|
|
944
661
|
else if (existsSync(resolve('eval-samples.yml')))
|
|
945
662
|
samplesFile = 'eval-samples.yml';
|
|
946
663
|
}
|
|
947
|
-
const { evolveSkill } = await import('
|
|
664
|
+
const { evolveSkill } = await import('../authoring/evolver.js');
|
|
948
665
|
process.stderr.write(tCli('cli.evolve.section_header', lang, { path: skillPath }));
|
|
949
666
|
try {
|
|
950
667
|
const result = await evolveSkill({
|
|
@@ -1018,7 +735,7 @@ async function handleGate(argv) {
|
|
|
1018
735
|
threshold: { type: 'string', default: '3.5' },
|
|
1019
736
|
'trivial-diff': { type: 'string' },
|
|
1020
737
|
});
|
|
1021
|
-
const { runEvaluation } = await import('
|
|
738
|
+
const { runEvaluation } = await import('../eval-workflows/run-evaluation.js');
|
|
1022
739
|
config.onProgress = makeOnProgress(lang);
|
|
1023
740
|
try {
|
|
1024
741
|
const { report } = (await runEvaluation(config));
|
|
@@ -1030,7 +747,7 @@ async function handleGate(argv) {
|
|
|
1030
747
|
// bootstrap diff CI / saturation / Krippendorff α)。computeVerdict 是单一
|
|
1031
748
|
// 决策源, exit code 跟 verdict.level 走 — 数据 underpowered 直接 FAIL,
|
|
1032
749
|
// 堵住"过 PASS 就 deploy"的漏洞。
|
|
1033
|
-
const { computeVerdict, formatVerdictText } = await import('
|
|
750
|
+
const { computeVerdict, formatVerdictText } = await import('../eval-core/verdict.js');
|
|
1034
751
|
const result = computeVerdict(report, {
|
|
1035
752
|
gateThreshold: Number(values.threshold),
|
|
1036
753
|
triviallySmallDiff: values['trivial-diff'] != null ? Number(values['trivial-diff']) : undefined,
|
|
@@ -1058,7 +775,7 @@ async function handleGate(argv) {
|
|
|
1058
775
|
async function handleDiff(argv) {
|
|
1059
776
|
const lang = langFromArgv(argv);
|
|
1060
777
|
// Flag-aware split: separate positional report IDs from flags so we can support
|
|
1061
|
-
// omk bench diff <id> — within-report sample-level
|
|
778
|
+
// omk bench diff <id> — within-report sample-level
|
|
1062
779
|
// omk bench diff <id1> <id2> — cross-report variant-level (legacy)
|
|
1063
780
|
// both with optional --regressions-only / --threshold / --variant flags.
|
|
1064
781
|
const positional = [];
|
|
@@ -1092,7 +809,7 @@ async function handleDiff(argv) {
|
|
|
1092
809
|
},
|
|
1093
810
|
strict: false,
|
|
1094
811
|
});
|
|
1095
|
-
const { createFileStore } = await import('
|
|
812
|
+
const { createFileStore } = await import('../server/report-store.js');
|
|
1096
813
|
const store = createFileStore(resolve(DEFAULT_REPORTS_DIR));
|
|
1097
814
|
if (positional.length === 1) {
|
|
1098
815
|
await runSampleLevelDiff(positional[0], store, values, lang);
|
|
@@ -1156,7 +873,7 @@ async function handleDiff(argv) {
|
|
|
1156
873
|
console.log('');
|
|
1157
874
|
}
|
|
1158
875
|
/**
|
|
1159
|
-
* Within-report sample-level diff
|
|
876
|
+
* Within-report sample-level diff. Compares two variants' scores on
|
|
1160
877
|
* each shared sample and surfaces the worst regressions / biggest wins.
|
|
1161
878
|
*
|
|
1162
879
|
* Default focus is variants[0] (control) vs variants[1] (treatment), but
|
|
@@ -1259,7 +976,7 @@ async function handleGold(argv) {
|
|
|
1259
976
|
},
|
|
1260
977
|
strict: false,
|
|
1261
978
|
});
|
|
1262
|
-
const { initGoldDataset } = await import('
|
|
979
|
+
const { initGoldDataset } = await import('../grading/gold-cli.js');
|
|
1263
980
|
try {
|
|
1264
981
|
const written = initGoldDataset(values.out, {
|
|
1265
982
|
annotator: values.annotator,
|
|
@@ -1283,7 +1000,7 @@ async function handleGold(argv) {
|
|
|
1283
1000
|
console.error(tCli('cli.common.usage_gold_validate', lang));
|
|
1284
1001
|
process.exit(1);
|
|
1285
1002
|
}
|
|
1286
|
-
const { validateGoldDataset } = await import('
|
|
1003
|
+
const { validateGoldDataset } = await import('../grading/gold-cli.js');
|
|
1287
1004
|
const result = validateGoldDataset(dir);
|
|
1288
1005
|
if (result.ok) {
|
|
1289
1006
|
console.log(tCli('cli.gold.validate_ok', lang, { n: result.sampleCount }));
|
|
@@ -1317,9 +1034,9 @@ async function handleGold(argv) {
|
|
|
1317
1034
|
console.error('--gold-dir is required');
|
|
1318
1035
|
process.exit(1);
|
|
1319
1036
|
}
|
|
1320
|
-
const { loadGoldDataset } = await import('
|
|
1321
|
-
const { compareGoldToReport, formatGoldCompare } = await import('
|
|
1322
|
-
const { createFileStore } = await import('
|
|
1037
|
+
const { loadGoldDataset } = await import('../grading/gold-dataset.js');
|
|
1038
|
+
const { compareGoldToReport, formatGoldCompare } = await import('../grading/gold-cli.js');
|
|
1039
|
+
const { createFileStore } = await import('../server/report-store.js');
|
|
1323
1040
|
const { dataset, issues } = loadGoldDataset(goldDir);
|
|
1324
1041
|
if (!dataset) {
|
|
1325
1042
|
console.error('Cannot load gold dataset:');
|
|
@@ -1387,7 +1104,7 @@ async function handleDebiasValidate(argv) {
|
|
|
1387
1104
|
},
|
|
1388
1105
|
strict: false,
|
|
1389
1106
|
});
|
|
1390
|
-
const { createFileStore } = await import('
|
|
1107
|
+
const { createFileStore } = await import('../server/report-store.js');
|
|
1391
1108
|
const store = createFileStore(resolve(values['reports-dir']));
|
|
1392
1109
|
const report = await store.get(reportId);
|
|
1393
1110
|
if (!report) {
|
|
@@ -1401,7 +1118,7 @@ async function handleDebiasValidate(argv) {
|
|
|
1401
1118
|
console.error('Cannot find samples path. Pass --samples <path> or ensure report has request.samplesPath.');
|
|
1402
1119
|
process.exit(1);
|
|
1403
1120
|
}
|
|
1404
|
-
const { loadSamples } = await import('
|
|
1121
|
+
const { loadSamples } = await import('../inputs/load-samples.js');
|
|
1405
1122
|
const { samples } = loadSamples(samplesPath);
|
|
1406
1123
|
const judgeModel = values['judge-model']
|
|
1407
1124
|
?? report.meta?.judgeModel;
|
|
@@ -1410,9 +1127,9 @@ async function handleDebiasValidate(argv) {
|
|
|
1410
1127
|
process.exit(1);
|
|
1411
1128
|
}
|
|
1412
1129
|
process.stderr.write(tCli('cli.debias.warn_cost_doubles', lang));
|
|
1413
|
-
const { createExecutor } = await import('
|
|
1130
|
+
const { createExecutor } = await import('../executors/index.js');
|
|
1414
1131
|
const judgeExecutor = createExecutor(values['judge-executor']);
|
|
1415
|
-
const { validateLengthDebias, formatDebiasValidate } = await import('
|
|
1132
|
+
const { validateLengthDebias, formatDebiasValidate } = await import('../grading/debias-validate.js');
|
|
1416
1133
|
const seedVal = values.seed != null ? Number(values.seed) : undefined;
|
|
1417
1134
|
const bsRaw = Number(values['bootstrap-samples']) || 1000;
|
|
1418
1135
|
const result = await validateLengthDebias({
|
|
@@ -1448,7 +1165,7 @@ async function handleSaturation(argv) {
|
|
|
1448
1165
|
},
|
|
1449
1166
|
strict: false,
|
|
1450
1167
|
});
|
|
1451
|
-
const { createFileStore } = await import('
|
|
1168
|
+
const { createFileStore } = await import('../server/report-store.js');
|
|
1452
1169
|
const store = createFileStore(resolve(values['reports-dir']));
|
|
1453
1170
|
const report = await store.get(reportId);
|
|
1454
1171
|
if (!report) {
|
|
@@ -1499,7 +1216,7 @@ async function handleSaturation(argv) {
|
|
|
1499
1216
|
console.log('');
|
|
1500
1217
|
}
|
|
1501
1218
|
// ---------------------------------------------------------------------------
|
|
1502
|
-
// handleVerdict — one-line ship/no-ship verdict
|
|
1219
|
+
// handleVerdict — one-line ship/no-ship verdict
|
|
1503
1220
|
// ---------------------------------------------------------------------------
|
|
1504
1221
|
async function handleVerdict(argv) {
|
|
1505
1222
|
const lang = langFromArgv(argv);
|
|
@@ -1519,14 +1236,14 @@ async function handleVerdict(argv) {
|
|
|
1519
1236
|
},
|
|
1520
1237
|
strict: false,
|
|
1521
1238
|
});
|
|
1522
|
-
const { createFileStore } = await import('
|
|
1239
|
+
const { createFileStore } = await import('../server/report-store.js');
|
|
1523
1240
|
const store = createFileStore(resolve(values['reports-dir']));
|
|
1524
1241
|
const report = await store.get(reportId);
|
|
1525
1242
|
if (!report) {
|
|
1526
1243
|
console.error(tCli('cli.common.report_not_found', lang, { id: reportId }));
|
|
1527
1244
|
process.exit(1);
|
|
1528
1245
|
}
|
|
1529
|
-
const { computeVerdict, formatVerdictText } = await import('
|
|
1246
|
+
const { computeVerdict, formatVerdictText } = await import('../eval-core/verdict.js');
|
|
1530
1247
|
const result = computeVerdict(report, {
|
|
1531
1248
|
gateThreshold: values.threshold != null ? Number(values.threshold) : undefined,
|
|
1532
1249
|
triviallySmallDiff: values['trivial-diff'] != null ? Number(values['trivial-diff']) : undefined,
|
|
@@ -1568,7 +1285,7 @@ async function handleDiagnose(argv) {
|
|
|
1568
1285
|
},
|
|
1569
1286
|
strict: false,
|
|
1570
1287
|
});
|
|
1571
|
-
const { createFileStore } = await import('
|
|
1288
|
+
const { createFileStore } = await import('../server/report-store.js');
|
|
1572
1289
|
const store = createFileStore(resolve(values['reports-dir']));
|
|
1573
1290
|
const report = await store.get(reportId);
|
|
1574
1291
|
if (!report) {
|
|
@@ -1583,7 +1300,7 @@ async function handleDiagnose(argv) {
|
|
|
1583
1300
|
const samplesPath = values.samples ?? report.meta?.request?.samplesPath;
|
|
1584
1301
|
if (samplesPath && existsSync(samplesPath)) {
|
|
1585
1302
|
try {
|
|
1586
|
-
const { loadSamples } = await import('
|
|
1303
|
+
const { loadSamples } = await import('../inputs/load-samples.js');
|
|
1587
1304
|
samples = loadSamples(samplesPath).samples;
|
|
1588
1305
|
}
|
|
1589
1306
|
catch (err) {
|
|
@@ -1594,7 +1311,7 @@ async function handleDiagnose(argv) {
|
|
|
1594
1311
|
}
|
|
1595
1312
|
const topRaw = Number(values.top);
|
|
1596
1313
|
const topN = Number.isFinite(topRaw) && topRaw > 0 ? topRaw : undefined;
|
|
1597
|
-
const { diagnoseSamples, formatSampleDiagnostics } = await import('
|
|
1314
|
+
const { diagnoseSamples, formatSampleDiagnostics } = await import('../analysis/sample-diagnostics.js');
|
|
1598
1315
|
const diag = diagnoseSamples(report, {
|
|
1599
1316
|
samples,
|
|
1600
1317
|
duplicateRouge: values['duplicate-rouge'] != null ? Number(values['duplicate-rouge']) : undefined,
|
|
@@ -1604,6 +1321,14 @@ async function handleDiagnose(argv) {
|
|
|
1604
1321
|
flatThreshold: values.flat != null ? Number(values.flat) : undefined,
|
|
1605
1322
|
});
|
|
1606
1323
|
console.log(formatSampleDiagnostics(diag, { topN }));
|
|
1324
|
+
// Sample design science coverage block. Render after diagnose 主体,因为
|
|
1325
|
+
// coverage 是声明式元数据(capability/difficulty/construct/provenance)的整体分布,
|
|
1326
|
+
// 跟 issue list 是不同视角的两件事。优先从 samples (现场加载) 算,fallback 到
|
|
1327
|
+
// report.analysis.sampleQuality(报告里持久化的数据)。
|
|
1328
|
+
const { renderSampleDesignCoverage } = await import('./coverage-renderer.js');
|
|
1329
|
+
const coverageBlock = renderSampleDesignCoverage(samples, report.analysis?.sampleQuality, lang);
|
|
1330
|
+
if (coverageBlock)
|
|
1331
|
+
console.log(coverageBlock);
|
|
1607
1332
|
// Exit code: 0 if health ≥ 70 and no errors; 1 otherwise. CI-friendly.
|
|
1608
1333
|
if (diag.totals.errors === 0 && diag.healthScore >= 70) {
|
|
1609
1334
|
process.exit(0);
|
|
@@ -1633,7 +1358,7 @@ async function handleFailures(argv) {
|
|
|
1633
1358
|
},
|
|
1634
1359
|
strict: false,
|
|
1635
1360
|
});
|
|
1636
|
-
const { createFileStore } = await import('
|
|
1361
|
+
const { createFileStore } = await import('../server/report-store.js');
|
|
1637
1362
|
const store = createFileStore(resolve(values['reports-dir']));
|
|
1638
1363
|
const report = await store.get(reportId);
|
|
1639
1364
|
if (!report) {
|
|
@@ -1645,9 +1370,9 @@ async function handleFailures(argv) {
|
|
|
1645
1370
|
console.error(tCli('cli.common.no_judge_model', lang));
|
|
1646
1371
|
process.exit(1);
|
|
1647
1372
|
}
|
|
1648
|
-
const { createExecutor } = await import('
|
|
1373
|
+
const { createExecutor } = await import('../executors/index.js');
|
|
1649
1374
|
const executor = createExecutor(values['judge-executor']);
|
|
1650
|
-
const { clusterFailures, formatFailureClusterReport } = await import('
|
|
1375
|
+
const { clusterFailures, formatFailureClusterReport } = await import('../analysis/failure-clusterer.js');
|
|
1651
1376
|
const out = await clusterFailures({
|
|
1652
1377
|
report: report,
|
|
1653
1378
|
executor,
|
|
@@ -1662,4 +1387,4 @@ async function handleFailures(argv) {
|
|
|
1662
1387
|
// Entry
|
|
1663
1388
|
// ---------------------------------------------------------------------------
|
|
1664
1389
|
main();
|
|
1665
|
-
//# sourceMappingURL=
|
|
1390
|
+
//# sourceMappingURL=index.js.map
|