oh-my-knowledge 0.22.0 → 0.23.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (100) hide show
  1. package/README.md +1 -0
  2. package/README.zh.md +1 -0
  3. package/dist/src/analysis/report-diagnostics.d.ts +26 -2
  4. package/dist/src/analysis/report-diagnostics.d.ts.map +1 -1
  5. package/dist/src/analysis/report-diagnostics.js +96 -3
  6. package/dist/src/analysis/report-diagnostics.js.map +1 -1
  7. package/dist/src/analysis/sample-diagnostics.d.ts +5 -1
  8. package/dist/src/analysis/sample-diagnostics.d.ts.map +1 -1
  9. package/dist/src/analysis/sample-diagnostics.js +105 -0
  10. package/dist/src/analysis/sample-diagnostics.js.map +1 -1
  11. package/dist/src/authoring/evolver.d.ts +2 -2
  12. package/dist/src/authoring/evolver.d.ts.map +1 -1
  13. package/dist/src/authoring/evolver.js +24 -7
  14. package/dist/src/authoring/evolver.js.map +1 -1
  15. package/dist/src/authoring/generator.d.ts +24 -0
  16. package/dist/src/authoring/generator.d.ts.map +1 -1
  17. package/dist/src/authoring/generator.js +67 -5
  18. package/dist/src/authoring/generator.js.map +1 -1
  19. package/dist/src/cli/coverage-renderer.d.ts +15 -0
  20. package/dist/src/cli/coverage-renderer.d.ts.map +1 -0
  21. package/dist/src/cli/coverage-renderer.js +74 -0
  22. package/dist/src/cli/coverage-renderer.js.map +1 -0
  23. package/dist/src/cli/i18n-dict.d.ts +1 -1
  24. package/dist/src/cli/i18n-dict.d.ts.map +1 -1
  25. package/dist/src/cli/i18n-dict.js +34 -16
  26. package/dist/src/cli/i18n-dict.js.map +1 -1
  27. package/dist/src/cli/index.d.ts +3 -0
  28. package/dist/src/cli/index.d.ts.map +1 -0
  29. package/dist/src/{cli.js → cli/index.js} +48 -323
  30. package/dist/src/cli/index.js.map +1 -0
  31. package/dist/src/cli/parse-run-config.d.ts +56 -0
  32. package/dist/src/cli/parse-run-config.d.ts.map +1 -0
  33. package/dist/src/cli/parse-run-config.js +195 -0
  34. package/dist/src/cli/parse-run-config.js.map +1 -0
  35. package/dist/src/cli/progress.d.ts +24 -0
  36. package/dist/src/cli/progress.d.ts.map +1 -0
  37. package/dist/src/cli/progress.js +55 -0
  38. package/dist/src/cli/progress.js.map +1 -0
  39. package/dist/src/cli/update-check.d.ts +3 -0
  40. package/dist/src/cli/update-check.d.ts.map +1 -0
  41. package/dist/src/cli/update-check.js +37 -0
  42. package/dist/src/cli/update-check.js.map +1 -0
  43. package/dist/src/eval-core/cache.d.ts +1 -1
  44. package/dist/src/eval-core/cache.js +1 -1
  45. package/dist/src/eval-core/dependency-checker.d.ts +1 -1
  46. package/dist/src/eval-core/dependency-checker.js +1 -1
  47. package/dist/src/eval-core/evaluation-execution.d.ts +1 -1
  48. package/dist/src/eval-core/evaluation-execution.js +4 -4
  49. package/dist/src/eval-core/evaluation-execution.js.map +1 -1
  50. package/dist/src/eval-core/evaluation-reporting.js +1 -1
  51. package/dist/src/eval-core/evaluation-reporting.js.map +1 -1
  52. package/dist/src/eval-core/execution-strategy.js +4 -4
  53. package/dist/src/eval-core/execution-strategy.js.map +1 -1
  54. package/dist/src/eval-core/schema.js +1 -1
  55. package/dist/src/eval-core/schema.js.map +1 -1
  56. package/dist/src/eval-core/verdict.d.ts +3 -3
  57. package/dist/src/eval-core/verdict.js +3 -3
  58. package/dist/src/eval-workflows/each-evaluation-workflow.d.ts +2 -2
  59. package/dist/src/eval-workflows/each-evaluation-workflow.js +1 -1
  60. package/dist/src/eval-workflows/each-evaluation-workflow.js.map +1 -1
  61. package/dist/src/eval-workflows/evaluation-pipeline.d.ts +2 -2
  62. package/dist/src/eval-workflows/evaluation-pipeline.d.ts.map +1 -1
  63. package/dist/src/eval-workflows/evaluation-pipeline.js +7 -3
  64. package/dist/src/eval-workflows/evaluation-pipeline.js.map +1 -1
  65. package/dist/src/eval-workflows/evaluation-preparation.d.ts +2 -2
  66. package/dist/src/eval-workflows/evaluation-preparation.d.ts.map +1 -1
  67. package/dist/src/eval-workflows/run-evaluation.d.ts +5 -5
  68. package/dist/src/eval-workflows/run-evaluation.d.ts.map +1 -1
  69. package/dist/src/eval-workflows/run-evaluation.js +9 -10
  70. package/dist/src/eval-workflows/run-evaluation.js.map +1 -1
  71. package/dist/src/executors/claude-cli.js +1 -1
  72. package/dist/src/executors/claude-cli.js.map +1 -1
  73. package/dist/src/executors/claude-sdk.d.ts +1 -1
  74. package/dist/src/executors/claude-sdk.js +1 -1
  75. package/dist/src/executors/script.js +1 -1
  76. package/dist/src/executors/script.js.map +1 -1
  77. package/dist/src/grading/assertions.js +3 -3
  78. package/dist/src/grading/assertions.js.map +1 -1
  79. package/dist/src/grading/gold-dataset.d.ts +1 -1
  80. package/dist/src/grading/gold-dataset.js +1 -1
  81. package/dist/src/inputs/eval-config.js +2 -2
  82. package/dist/src/inputs/eval-config.js.map +1 -1
  83. package/dist/src/inputs/load-samples.d.ts.map +1 -1
  84. package/dist/src/inputs/load-samples.js +30 -4
  85. package/dist/src/inputs/load-samples.js.map +1 -1
  86. package/dist/src/inputs/skill-loader.d.ts +1 -1
  87. package/dist/src/inputs/skill-loader.d.ts.map +1 -1
  88. package/dist/src/inputs/skill-loader.js +1 -1
  89. package/dist/src/inputs/skill-loader.js.map +1 -1
  90. package/dist/src/renderer/summary.d.ts.map +1 -1
  91. package/dist/src/renderer/summary.js +0 -13
  92. package/dist/src/renderer/summary.js.map +1 -1
  93. package/dist/src/types/eval.d.ts +23 -3
  94. package/dist/src/types/eval.d.ts.map +1 -1
  95. package/dist/src/types/report.d.ts +31 -5
  96. package/dist/src/types/report.d.ts.map +1 -1
  97. package/package.json +13 -5
  98. package/dist/src/cli.d.ts +0 -3
  99. package/dist/src/cli.d.ts.map +0 -1
  100. package/dist/src/cli.js.map +0 -1
@@ -1,242 +1,12 @@
1
1
  #!/usr/bin/env node
2
2
  import { parseArgs } from 'node:util';
3
3
  import { resolve } from 'node:path';
4
- import { homedir } from 'node:os';
5
4
  import { join } from 'node:path';
6
5
  import { existsSync } from 'node:fs';
7
- import { tCli, getCliLang, parseLangFromArgv, langFromArgv } from './cli/i18n.js';
8
- import { discoverVariants, parseVariantCwd } from './inputs/skill-loader.js';
9
- import { loadEvalConfig, configVariantsToSpecs } from './inputs/eval-config.js';
10
- // ---------------------------------------------------------------------------
11
- // Constants
12
- // ---------------------------------------------------------------------------
13
- const DEFAULT_REPORTS_DIR = join(homedir(), '.oh-my-knowledge', 'reports');
14
- // Shared CLI options for run/ci commands.
15
- // Defaults are applied inside parseRunConfig (after config-file merge) so that
16
- // CLI `undefined` can be reliably distinguished from "user passed the default value".
17
- // Priority order resolved in parseRunConfig: CLI arg > --config file > hard-coded default.
18
- /**
19
- * 所有子命令都接受的通用 flag。新增 --lang 让 parseArgs strict:false 模式下
20
- * 仍能把值类型化到 values.lang 上(否则未声明的 flag 会被丢弃)。
21
- */
22
- const COMMON_OPTIONS = {
23
- lang: { type: 'string' },
24
- };
25
- const RUN_OPTIONS = {
26
- ...COMMON_OPTIONS,
27
- samples: { type: 'string' },
28
- 'skill-dir': { type: 'string' },
29
- control: { type: 'string' },
30
- treatment: { type: 'string' },
31
- config: { type: 'string' },
32
- model: { type: 'string' },
33
- 'judge-model': { type: 'string' },
34
- 'output-dir': { type: 'string' },
35
- 'no-judge': { type: 'boolean' },
36
- 'no-cache': { type: 'boolean' },
37
- 'dry-run': { type: 'boolean' },
38
- concurrency: { type: 'string' },
39
- timeout: { type: 'string' },
40
- executor: { type: 'string' },
41
- 'judge-executor': { type: 'string' },
42
- each: { type: 'boolean' },
43
- 'skip-preflight': { type: 'boolean' },
44
- 'mcp-config': { type: 'string' },
45
- 'no-serve': { type: 'boolean' },
46
- verbose: { type: 'boolean' },
47
- retry: { type: 'string' },
48
- resume: { type: 'string' },
49
- 'layered-stats': { type: 'boolean' },
50
- // v0.22 — strict-baseline default true. Declare both forms; reconcile in
51
- // parseRunConfig (后者赢)。strict-baseline 没传 + no-strict-baseline 没传 = default true。
52
- 'strict-baseline': { type: 'boolean' },
53
- 'no-strict-baseline': { type: 'boolean' },
54
- };
55
- // ---------------------------------------------------------------------------
56
- // parseRunConfig
57
- // ---------------------------------------------------------------------------
58
- function parseRunConfig(argv, extraOptions = {}) {
59
- const { values } = parseArgs({
60
- args: argv,
61
- options: { ...RUN_OPTIONS, ...extraOptions },
62
- strict: false,
63
- });
64
- if (values.variants !== undefined) {
65
- throw new Error(`--variants 已在 v0.16 废除,请改用 --control <expr> 与 --treatment <v1,v2,...>\n`
66
- + ` 迁移示例:--variants baseline,my-skill → --control baseline --treatment my-skill\n`
67
- + ` 复杂场景可用 --config eval.yaml(参见 docs/terminology-spec.md)`);
68
- }
69
- // 1) Load --config (if provided). All subsequent fields fall back to it when CLI is silent.
70
- const evalConfig = values.config
71
- ? loadEvalConfig(values.config)
72
- : null;
73
- // 2) Resolve samples path: CLI > config > auto-detect .json/.yaml/.yml in cwd.
74
- const cliSamples = values.samples;
75
- let samplesFile;
76
- if (cliSamples) {
77
- samplesFile = cliSamples;
78
- }
79
- else if (evalConfig?.samples) {
80
- samplesFile = evalConfig.samples; // already resolved against config file dir
81
- }
82
- else {
83
- samplesFile = 'eval-samples.json';
84
- if (!existsSync(resolve(samplesFile))) {
85
- if (existsSync(resolve('eval-samples.yaml')))
86
- samplesFile = 'eval-samples.yaml';
87
- else if (existsSync(resolve('eval-samples.yml')))
88
- samplesFile = 'eval-samples.yml';
89
- }
90
- }
91
- const skillDir = resolve(values['skill-dir'] ?? 'skills');
92
- // 3) Resolve variantSpecs: CLI > config. If neither, error with a helpful hint.
93
- const controlExpr = values.control;
94
- const treatmentExprs = values.treatment
95
- ? values.treatment.split(',').map((v) => v.trim()).filter(Boolean)
96
- : [];
97
- let variantSpecs;
98
- if (controlExpr || treatmentExprs.length > 0) {
99
- // CLI roles present → CLI entirely replaces config.variants (no merging).
100
- variantSpecs = [];
101
- if (controlExpr) {
102
- variantSpecs.push({ name: parseVariantCwd(controlExpr).name, role: 'control', expr: controlExpr });
103
- }
104
- for (const expr of treatmentExprs) {
105
- variantSpecs.push({ name: parseVariantCwd(expr).name, role: 'treatment', expr });
106
- }
107
- }
108
- else if (evalConfig) {
109
- variantSpecs = configVariantsToSpecs(evalConfig.variants);
110
- }
111
- else if (values.each) {
112
- // --each 模式自动用 baseline (control) vs 每个 skill (treatment),
113
- // 不需要用户显式传 --control / --treatment,校验跳过。
114
- variantSpecs = [];
115
- }
116
- else {
117
- const discovered = discoverVariants(skillDir);
118
- const hint = discovered.length > 0 ? `\n skill-dir (${skillDir}) 下发现的候选:${discovered.join(', ')}` : '';
119
- throw new Error(`请通过 --control / --treatment 或 --config eval.yaml 声明 variant 角色。\n`
120
- + ` 示例:omk bench run --control baseline --treatment my-skill${hint}\n`
121
- + ` --each 模式下自动用 baseline vs 每个 skill,无需显式声明\n`
122
- + ` 术语见 docs/terminology-spec.md(v0.16 起废除 --variants,改用 experiment role 显式声明)`);
123
- }
124
- const seenNames = new Set();
125
- for (const spec of variantSpecs) {
126
- if (seenNames.has(spec.name)) {
127
- throw new Error(`variant "${spec.name}" 重复出现——同一 variant 不能同时属于 --control 与 --treatment,也不能在 --treatment 中重复。`);
128
- }
129
- seenNames.add(spec.name);
130
- }
131
- // 4) Apply CLI > config > hard-coded default for all other fields.
132
- const executorName = values.executor ?? evalConfig?.executor ?? 'claude';
133
- const judgeExecutorName = values['judge-executor'] ?? evalConfig?.judgeExecutor ?? executorName;
134
- const model = values.model ?? evalConfig?.model ?? 'sonnet';
135
- const judgeModelRaw = values['judge-model'] !== undefined
136
- ? values['judge-model']
137
- : evalConfig?.judgeModel ?? 'haiku';
138
- const judgeModel = judgeModelRaw ?? 'haiku';
139
- const outputDir = resolve(values['output-dir'] ?? DEFAULT_REPORTS_DIR);
140
- const concurrencyRaw = values.concurrency !== undefined
141
- ? Number(values.concurrency)
142
- : evalConfig?.concurrency ?? 1;
143
- const concurrency = Math.max(1, Number(concurrencyRaw) || 1);
144
- const timeoutSec = values.timeout !== undefined
145
- ? Number(values.timeout)
146
- : evalConfig?.timeoutMs
147
- ? evalConfig.timeoutMs / 1000
148
- : 120;
149
- const timeoutMs = Math.max(1, Number(timeoutSec) || 120) * 1000;
150
- const noJudge = values['no-judge'] ?? false;
151
- const noCache = values['no-cache'] ?? evalConfig?.noCache ?? false;
152
- const dryRun = values['dry-run'] ?? false;
153
- const skipPreflight = values['skip-preflight'] ?? false;
154
- const mcpConfig = values['mcp-config'] ?? evalConfig?.mcpConfig;
155
- const verbose = values.verbose ?? false;
156
- const retry = Math.max(0, Number(values.retry ?? 0) || 0);
157
- const resume = values.resume;
158
- const blind = values.blind ?? evalConfig?.blind ?? false;
159
- const layeredStats = values['layered-stats'] ?? false;
160
- // v0.22 — strict-baseline default true. Reconcile both flag forms.
161
- // Priority: --no-strict-baseline > --strict-baseline > undefined(=true).
162
- const noStrictFlag = values['no-strict-baseline'];
163
- const strictFlag = values['strict-baseline'];
164
- const strictBaseline = noStrictFlag === true ? false : (strictFlag ?? true);
165
- // v0.22 — extract eval.yaml variant.allowedSkills overrides (per-variant). Always
166
- // wins over strictBaseline default. Empty object when no eval.yaml or no overrides.
167
- const variantAllowedSkills = {};
168
- if (evalConfig?.variants) {
169
- for (const v of evalConfig.variants) {
170
- if (v.allowedSkills !== undefined) {
171
- variantAllowedSkills[v.name] = v.allowedSkills;
172
- }
173
- }
174
- }
175
- return {
176
- values,
177
- config: {
178
- samplesPath: resolve(samplesFile),
179
- skillDir,
180
- variantSpecs,
181
- model,
182
- judgeModel,
183
- outputDir,
184
- noJudge,
185
- noCache,
186
- dryRun,
187
- concurrency,
188
- timeoutMs,
189
- executorName,
190
- judgeExecutorName,
191
- skipPreflight,
192
- mcpConfig,
193
- verbose,
194
- retry,
195
- resume,
196
- blind,
197
- layeredStats,
198
- budget: evalConfig?.budget,
199
- strictBaseline,
200
- ...(Object.keys(variantAllowedSkills).length > 0 && { variantAllowedSkills }),
201
- },
202
- };
203
- }
204
- // ---------------------------------------------------------------------------
205
- // Update check
206
- // ---------------------------------------------------------------------------
207
- async function checkUpdate(lang) {
208
- try {
209
- const { readFileSync } = await import('node:fs');
210
- const { fileURLToPath } = await import('node:url');
211
- const { dirname, join } = await import('node:path');
212
- const __dirname = dirname(fileURLToPath(import.meta.url));
213
- const findPackageJson = (startDir) => {
214
- let dir = startDir;
215
- for (let i = 0; i < 5; i++) {
216
- const candidate = join(dir, 'package.json');
217
- if (existsSync(candidate))
218
- return candidate;
219
- dir = dirname(dir);
220
- }
221
- return null;
222
- };
223
- const pkgPath = findPackageJson(__dirname);
224
- if (!pkgPath)
225
- return;
226
- const pkg = JSON.parse(readFileSync(pkgPath, 'utf-8'));
227
- const registry = pkg.publishConfig?.registry || 'https://registry.npmjs.org';
228
- const res = await fetch(`${registry}/${pkg.name}/latest`, { signal: AbortSignal.timeout(3000) });
229
- if (!res.ok)
230
- return;
231
- const data = await res.json();
232
- if (data.version && data.version !== pkg.version) {
233
- process.stderr.write(tCli('cli.update.new_version_available', lang, {
234
- old: pkg.version, new: data.version, pkg: pkg.name,
235
- }));
236
- }
237
- }
238
- catch { /* 静默失败,不影响正常使用 */ }
239
- }
6
+ import { tCli, getCliLang, parseLangFromArgv, langFromArgv } from './i18n.js';
7
+ import { parseRunConfig, DEFAULT_REPORTS_DIR, COMMON_OPTIONS, } from './parse-run-config.js';
8
+ import { makeOnProgress } from './progress.js';
9
+ import { checkUpdate } from './update-check.js';
240
10
  // ---------------------------------------------------------------------------
241
11
  // Main
242
12
  // ---------------------------------------------------------------------------
@@ -313,59 +83,6 @@ async function main() {
313
83
  * Factory: 闭住 lang, 返回 onProgress callback。evaluation engine 回调时不传
314
84
  * 上下文, 所以 lang 必须在 handler 入口处通过 closure 传进来。
315
85
  */
316
- function makeOnProgress(lang) {
317
- return ({ phase, completed, total, sample_id, variant, durationMs, inputTokens, outputTokens, costUSD, score, outputPreview, judgePhase: _judgePhase, judgeDim, skipped, attempt, maxAttempts, error, }) => {
318
- const ctx = { i: completed ?? '', n: total ?? '', sample: sample_id ?? '', variant: variant ?? '' };
319
- if (phase === 'preflight') {
320
- process.stderr.write(tCli('cli.progress.preflight_starting', lang));
321
- return;
322
- }
323
- if (phase === 'retry') {
324
- process.stderr.write(tCli('cli.progress.sample_retry', lang, {
325
- ...ctx, attempt: attempt ?? '', max: maxAttempts ?? '',
326
- }));
327
- return;
328
- }
329
- if (phase === 'error') {
330
- process.stderr.write(tCli('cli.progress.sample_error', lang, { ...ctx, error: error ?? '' }));
331
- return;
332
- }
333
- if (phase === 'start') {
334
- process.stderr.write(tCli('cli.progress.sample_executing', lang, ctx));
335
- }
336
- else if (phase === 'exec_done') {
337
- const cost = costUSD != null && costUSD > 0 ? ` $${costUSD.toFixed(4)}` : '';
338
- process.stderr.write(tCli('cli.progress.sample_exec_done', lang, {
339
- ...ctx, ms: durationMs ?? '', input: inputTokens ?? '', output: outputTokens ?? '', cost,
340
- }));
341
- if (outputPreview) {
342
- process.stderr.write(tCli('cli.progress.output_preview', lang, {
343
- preview: outputPreview.slice(0, 150).replace(/\n/g, ' '),
344
- }));
345
- }
346
- }
347
- else if (phase === 'grading') {
348
- const dim = judgeDim ? ` [${judgeDim}]` : '';
349
- process.stderr.write(tCli('cli.progress.judging', lang, { ...ctx, dim }));
350
- }
351
- else if (phase === 'judge_done') {
352
- const dim = judgeDim ? ` [${judgeDim}]` : '';
353
- process.stderr.write(tCli('cli.progress.judged', lang, { ...ctx, dim, score: score ?? '' }));
354
- }
355
- else if (phase === 'done' && skipped) {
356
- if (sample_id)
357
- process.stderr.write(tCli('cli.progress.skipped', lang, ctx));
358
- }
359
- else {
360
- const cost = costUSD != null && costUSD > 0 ? ` $${costUSD.toFixed(4)}` : '';
361
- const scoreInfo = typeof score === 'number' ? ` score=${score}` : '';
362
- process.stderr.write(tCli('cli.progress.sample_done', lang, {
363
- ...ctx, ms: durationMs ?? '', input: inputTokens ?? '', output: outputTokens ?? '',
364
- cost, score: scoreInfo,
365
- }));
366
- }
367
- };
368
- }
369
86
  // ---------------------------------------------------------------------------
370
87
  // handleRun
371
88
  // ---------------------------------------------------------------------------
@@ -384,7 +101,7 @@ async function handleRun(argv) {
384
101
  'budget-per-sample-usd': { type: 'string' },
385
102
  'budget-per-sample-ms': { type: 'string' },
386
103
  });
387
- const { runEvaluation, runMultiple, runEachEvaluation } = await import('./eval-workflows/run-evaluation.js');
104
+ const { runEvaluation, runMultiple, runEachEvaluation } = await import('../eval-workflows/run-evaluation.js');
388
105
  if (values.blind !== undefined) {
389
106
  config.blind = values.blind;
390
107
  }
@@ -430,7 +147,7 @@ async function handleRun(argv) {
430
147
  }
431
148
  }
432
149
  // --budget-usd / --budget-per-sample-usd / --budget-per-sample-ms:
433
- // v0.22 hard budget caps. CLI flags override config-file values. When the
150
+ // hard budget caps. CLI flags override config-file values. When the
434
151
  // total-USD cap is exceeded mid-run, remaining tasks are skipped and a
435
152
  // partial report is persisted with meta.budgetExhausted=true.
436
153
  const budgetUSD = values['budget-usd'] != null ? Number(values['budget-usd']) : undefined;
@@ -483,7 +200,7 @@ async function handleRun(argv) {
483
200
  process.stderr.write(tCli('cli.run.batch_complete', lang));
484
201
  process.stderr.write(tCli('cli.run.report_saved', lang, { path: filePath }));
485
202
  if (!values['no-serve'] && process.stdout.isTTY) {
486
- const { createReportServer } = await import('./server/report-server.js');
203
+ const { createReportServer } = await import('../server/report-server.js');
487
204
  const server = createReportServer({ reportsDir: config.outputDir });
488
205
  const serverUrl = await server.start();
489
206
  const reportUrl = `${serverUrl}/reports/${report.id}`;
@@ -523,7 +240,7 @@ async function handleRun(argv) {
523
240
  // --gold-dir: compute α/κ/Pearson against gold annotations and re-persist.
524
241
  const goldDir = values['gold-dir'];
525
242
  if (goldDir && filePath) {
526
- const { attachGoldAgreementToReport, formatGoldCompare } = await import('./grading/gold-cli.js');
243
+ const { attachGoldAgreementToReport, formatGoldCompare } = await import('../grading/gold-cli.js');
527
244
  const out = attachGoldAgreementToReport({
528
245
  report,
529
246
  goldDir,
@@ -551,7 +268,7 @@ async function handleRun(argv) {
551
268
  process.stderr.write(tCli('cli.run.report_saved', lang, { path: filePath }));
552
269
  if (!values['no-serve'] && process.stdout.isTTY) {
553
270
  // Auto-start report server
554
- const { createReportServer } = await import('./server/report-server.js');
271
+ const { createReportServer } = await import('../server/report-server.js');
555
272
  const server = createReportServer({
556
273
  reportsDir: config.outputDir,
557
274
  });
@@ -612,8 +329,8 @@ async function handleReport(argv) {
612
329
  return;
613
330
  }
614
331
  if (values.export) {
615
- const { createFileStore } = await import('./server/report-store.js');
616
- const { renderRunDetail, renderEachRunDetail } = await import('./renderer/html-renderer.js');
332
+ const { createFileStore } = await import('../server/report-store.js');
333
+ const { renderRunDetail, renderEachRunDetail } = await import('../renderer/html-renderer.js');
617
334
  const { writeFileSync } = await import('node:fs');
618
335
  const store = createFileStore(resolve(values['reports-dir']));
619
336
  const report = await store.get(values.export);
@@ -628,7 +345,7 @@ async function handleReport(argv) {
628
345
  console.log('Open in browser, or Ctrl+P to save as PDF');
629
346
  return;
630
347
  }
631
- const { createReportServer } = await import('./server/report-server.js');
348
+ const { createReportServer } = await import('../server/report-server.js');
632
349
  const server = createReportServer({
633
350
  port: Number(values.port),
634
351
  reportsDir: resolve(values['reports-dir']),
@@ -751,7 +468,7 @@ async function handleAnalyze(argv) {
751
468
  const to = values.to;
752
469
  const skills = values.skills ? values.skills.split(',').map((s) => s.trim()).filter(Boolean) : undefined;
753
470
  console.log(`[omk] analyzing ${tracePath}...`);
754
- const { computeSkillHealthReport } = await import('./observability/skill-health-analyzer.js');
471
+ const { computeSkillHealthReport } = await import('../observability/skill-health-analyzer.js');
755
472
  const report = computeSkillHealthReport(tracePath, {
756
473
  kbRoot: values.kb ? resolve(values.kb) : undefined,
757
474
  from,
@@ -812,7 +529,7 @@ async function handleGenSamples(argv) {
812
529
  strict: false,
813
530
  allowPositionals: true,
814
531
  });
815
- const { generateSamples } = await import('./authoring/generator.js');
532
+ const { generateSamples } = await import('../authoring/generator.js');
816
533
  const { readFileSync, writeFileSync } = await import('node:fs');
817
534
  const count = Math.max(1, Number(values.count) || 5);
818
535
  const model = values.model;
@@ -944,7 +661,7 @@ async function handleEvolve(argv) {
944
661
  else if (existsSync(resolve('eval-samples.yml')))
945
662
  samplesFile = 'eval-samples.yml';
946
663
  }
947
- const { evolveSkill } = await import('./authoring/evolver.js');
664
+ const { evolveSkill } = await import('../authoring/evolver.js');
948
665
  process.stderr.write(tCli('cli.evolve.section_header', lang, { path: skillPath }));
949
666
  try {
950
667
  const result = await evolveSkill({
@@ -1018,7 +735,7 @@ async function handleGate(argv) {
1018
735
  threshold: { type: 'string', default: '3.5' },
1019
736
  'trivial-diff': { type: 'string' },
1020
737
  });
1021
- const { runEvaluation } = await import('./eval-workflows/run-evaluation.js');
738
+ const { runEvaluation } = await import('../eval-workflows/run-evaluation.js');
1022
739
  config.onProgress = makeOnProgress(lang);
1023
740
  try {
1024
741
  const { report } = (await runEvaluation(config));
@@ -1030,7 +747,7 @@ async function handleGate(argv) {
1030
747
  // bootstrap diff CI / saturation / Krippendorff α)。computeVerdict 是单一
1031
748
  // 决策源, exit code 跟 verdict.level 走 — 数据 underpowered 直接 FAIL,
1032
749
  // 堵住"过 PASS 就 deploy"的漏洞。
1033
- const { computeVerdict, formatVerdictText } = await import('./eval-core/verdict.js');
750
+ const { computeVerdict, formatVerdictText } = await import('../eval-core/verdict.js');
1034
751
  const result = computeVerdict(report, {
1035
752
  gateThreshold: Number(values.threshold),
1036
753
  triviallySmallDiff: values['trivial-diff'] != null ? Number(values['trivial-diff']) : undefined,
@@ -1058,7 +775,7 @@ async function handleGate(argv) {
1058
775
  async function handleDiff(argv) {
1059
776
  const lang = langFromArgv(argv);
1060
777
  // Flag-aware split: separate positional report IDs from flags so we can support
1061
- // omk bench diff <id> — within-report sample-level (v0.22)
778
+ // omk bench diff <id> — within-report sample-level
1062
779
  // omk bench diff <id1> <id2> — cross-report variant-level (legacy)
1063
780
  // both with optional --regressions-only / --threshold / --variant flags.
1064
781
  const positional = [];
@@ -1092,7 +809,7 @@ async function handleDiff(argv) {
1092
809
  },
1093
810
  strict: false,
1094
811
  });
1095
- const { createFileStore } = await import('./server/report-store.js');
812
+ const { createFileStore } = await import('../server/report-store.js');
1096
813
  const store = createFileStore(resolve(DEFAULT_REPORTS_DIR));
1097
814
  if (positional.length === 1) {
1098
815
  await runSampleLevelDiff(positional[0], store, values, lang);
@@ -1156,7 +873,7 @@ async function handleDiff(argv) {
1156
873
  console.log('');
1157
874
  }
1158
875
  /**
1159
- * Within-report sample-level diff (v0.22). Compares two variants' scores on
876
+ * Within-report sample-level diff. Compares two variants' scores on
1160
877
  * each shared sample and surfaces the worst regressions / biggest wins.
1161
878
  *
1162
879
  * Default focus is variants[0] (control) vs variants[1] (treatment), but
@@ -1259,7 +976,7 @@ async function handleGold(argv) {
1259
976
  },
1260
977
  strict: false,
1261
978
  });
1262
- const { initGoldDataset } = await import('./grading/gold-cli.js');
979
+ const { initGoldDataset } = await import('../grading/gold-cli.js');
1263
980
  try {
1264
981
  const written = initGoldDataset(values.out, {
1265
982
  annotator: values.annotator,
@@ -1283,7 +1000,7 @@ async function handleGold(argv) {
1283
1000
  console.error(tCli('cli.common.usage_gold_validate', lang));
1284
1001
  process.exit(1);
1285
1002
  }
1286
- const { validateGoldDataset } = await import('./grading/gold-cli.js');
1003
+ const { validateGoldDataset } = await import('../grading/gold-cli.js');
1287
1004
  const result = validateGoldDataset(dir);
1288
1005
  if (result.ok) {
1289
1006
  console.log(tCli('cli.gold.validate_ok', lang, { n: result.sampleCount }));
@@ -1317,9 +1034,9 @@ async function handleGold(argv) {
1317
1034
  console.error('--gold-dir is required');
1318
1035
  process.exit(1);
1319
1036
  }
1320
- const { loadGoldDataset } = await import('./grading/gold-dataset.js');
1321
- const { compareGoldToReport, formatGoldCompare } = await import('./grading/gold-cli.js');
1322
- const { createFileStore } = await import('./server/report-store.js');
1037
+ const { loadGoldDataset } = await import('../grading/gold-dataset.js');
1038
+ const { compareGoldToReport, formatGoldCompare } = await import('../grading/gold-cli.js');
1039
+ const { createFileStore } = await import('../server/report-store.js');
1323
1040
  const { dataset, issues } = loadGoldDataset(goldDir);
1324
1041
  if (!dataset) {
1325
1042
  console.error('Cannot load gold dataset:');
@@ -1387,7 +1104,7 @@ async function handleDebiasValidate(argv) {
1387
1104
  },
1388
1105
  strict: false,
1389
1106
  });
1390
- const { createFileStore } = await import('./server/report-store.js');
1107
+ const { createFileStore } = await import('../server/report-store.js');
1391
1108
  const store = createFileStore(resolve(values['reports-dir']));
1392
1109
  const report = await store.get(reportId);
1393
1110
  if (!report) {
@@ -1401,7 +1118,7 @@ async function handleDebiasValidate(argv) {
1401
1118
  console.error('Cannot find samples path. Pass --samples <path> or ensure report has request.samplesPath.');
1402
1119
  process.exit(1);
1403
1120
  }
1404
- const { loadSamples } = await import('./inputs/load-samples.js');
1121
+ const { loadSamples } = await import('../inputs/load-samples.js');
1405
1122
  const { samples } = loadSamples(samplesPath);
1406
1123
  const judgeModel = values['judge-model']
1407
1124
  ?? report.meta?.judgeModel;
@@ -1410,9 +1127,9 @@ async function handleDebiasValidate(argv) {
1410
1127
  process.exit(1);
1411
1128
  }
1412
1129
  process.stderr.write(tCli('cli.debias.warn_cost_doubles', lang));
1413
- const { createExecutor } = await import('./executors/index.js');
1130
+ const { createExecutor } = await import('../executors/index.js');
1414
1131
  const judgeExecutor = createExecutor(values['judge-executor']);
1415
- const { validateLengthDebias, formatDebiasValidate } = await import('./grading/debias-validate.js');
1132
+ const { validateLengthDebias, formatDebiasValidate } = await import('../grading/debias-validate.js');
1416
1133
  const seedVal = values.seed != null ? Number(values.seed) : undefined;
1417
1134
  const bsRaw = Number(values['bootstrap-samples']) || 1000;
1418
1135
  const result = await validateLengthDebias({
@@ -1448,7 +1165,7 @@ async function handleSaturation(argv) {
1448
1165
  },
1449
1166
  strict: false,
1450
1167
  });
1451
- const { createFileStore } = await import('./server/report-store.js');
1168
+ const { createFileStore } = await import('../server/report-store.js');
1452
1169
  const store = createFileStore(resolve(values['reports-dir']));
1453
1170
  const report = await store.get(reportId);
1454
1171
  if (!report) {
@@ -1499,7 +1216,7 @@ async function handleSaturation(argv) {
1499
1216
  console.log('');
1500
1217
  }
1501
1218
  // ---------------------------------------------------------------------------
1502
- // handleVerdict — one-line ship/no-ship verdict (v0.22)
1219
+ // handleVerdict — one-line ship/no-ship verdict
1503
1220
  // ---------------------------------------------------------------------------
1504
1221
  async function handleVerdict(argv) {
1505
1222
  const lang = langFromArgv(argv);
@@ -1519,14 +1236,14 @@ async function handleVerdict(argv) {
1519
1236
  },
1520
1237
  strict: false,
1521
1238
  });
1522
- const { createFileStore } = await import('./server/report-store.js');
1239
+ const { createFileStore } = await import('../server/report-store.js');
1523
1240
  const store = createFileStore(resolve(values['reports-dir']));
1524
1241
  const report = await store.get(reportId);
1525
1242
  if (!report) {
1526
1243
  console.error(tCli('cli.common.report_not_found', lang, { id: reportId }));
1527
1244
  process.exit(1);
1528
1245
  }
1529
- const { computeVerdict, formatVerdictText } = await import('./eval-core/verdict.js');
1246
+ const { computeVerdict, formatVerdictText } = await import('../eval-core/verdict.js');
1530
1247
  const result = computeVerdict(report, {
1531
1248
  gateThreshold: values.threshold != null ? Number(values.threshold) : undefined,
1532
1249
  triviallySmallDiff: values['trivial-diff'] != null ? Number(values['trivial-diff']) : undefined,
@@ -1568,7 +1285,7 @@ async function handleDiagnose(argv) {
1568
1285
  },
1569
1286
  strict: false,
1570
1287
  });
1571
- const { createFileStore } = await import('./server/report-store.js');
1288
+ const { createFileStore } = await import('../server/report-store.js');
1572
1289
  const store = createFileStore(resolve(values['reports-dir']));
1573
1290
  const report = await store.get(reportId);
1574
1291
  if (!report) {
@@ -1583,7 +1300,7 @@ async function handleDiagnose(argv) {
1583
1300
  const samplesPath = values.samples ?? report.meta?.request?.samplesPath;
1584
1301
  if (samplesPath && existsSync(samplesPath)) {
1585
1302
  try {
1586
- const { loadSamples } = await import('./inputs/load-samples.js');
1303
+ const { loadSamples } = await import('../inputs/load-samples.js');
1587
1304
  samples = loadSamples(samplesPath).samples;
1588
1305
  }
1589
1306
  catch (err) {
@@ -1594,7 +1311,7 @@ async function handleDiagnose(argv) {
1594
1311
  }
1595
1312
  const topRaw = Number(values.top);
1596
1313
  const topN = Number.isFinite(topRaw) && topRaw > 0 ? topRaw : undefined;
1597
- const { diagnoseSamples, formatSampleDiagnostics } = await import('./analysis/sample-diagnostics.js');
1314
+ const { diagnoseSamples, formatSampleDiagnostics } = await import('../analysis/sample-diagnostics.js');
1598
1315
  const diag = diagnoseSamples(report, {
1599
1316
  samples,
1600
1317
  duplicateRouge: values['duplicate-rouge'] != null ? Number(values['duplicate-rouge']) : undefined,
@@ -1604,6 +1321,14 @@ async function handleDiagnose(argv) {
1604
1321
  flatThreshold: values.flat != null ? Number(values.flat) : undefined,
1605
1322
  });
1606
1323
  console.log(formatSampleDiagnostics(diag, { topN }));
1324
+ // Sample design science coverage block. Render after diagnose 主体,因为
1325
+ // coverage 是声明式元数据(capability/difficulty/construct/provenance)的整体分布,
1326
+ // 跟 issue list 是不同视角的两件事。优先从 samples (现场加载) 算,fallback 到
1327
+ // report.analysis.sampleQuality(报告里持久化的数据)。
1328
+ const { renderSampleDesignCoverage } = await import('./coverage-renderer.js');
1329
+ const coverageBlock = renderSampleDesignCoverage(samples, report.analysis?.sampleQuality, lang);
1330
+ if (coverageBlock)
1331
+ console.log(coverageBlock);
1607
1332
  // Exit code: 0 if health ≥ 70 and no errors; 1 otherwise. CI-friendly.
1608
1333
  if (diag.totals.errors === 0 && diag.healthScore >= 70) {
1609
1334
  process.exit(0);
@@ -1633,7 +1358,7 @@ async function handleFailures(argv) {
1633
1358
  },
1634
1359
  strict: false,
1635
1360
  });
1636
- const { createFileStore } = await import('./server/report-store.js');
1361
+ const { createFileStore } = await import('../server/report-store.js');
1637
1362
  const store = createFileStore(resolve(values['reports-dir']));
1638
1363
  const report = await store.get(reportId);
1639
1364
  if (!report) {
@@ -1645,9 +1370,9 @@ async function handleFailures(argv) {
1645
1370
  console.error(tCli('cli.common.no_judge_model', lang));
1646
1371
  process.exit(1);
1647
1372
  }
1648
- const { createExecutor } = await import('./executors/index.js');
1373
+ const { createExecutor } = await import('../executors/index.js');
1649
1374
  const executor = createExecutor(values['judge-executor']);
1650
- const { clusterFailures, formatFailureClusterReport } = await import('./analysis/failure-clusterer.js');
1375
+ const { clusterFailures, formatFailureClusterReport } = await import('../analysis/failure-clusterer.js');
1651
1376
  const out = await clusterFailures({
1652
1377
  report: report,
1653
1378
  executor,
@@ -1662,4 +1387,4 @@ async function handleFailures(argv) {
1662
1387
  // Entry
1663
1388
  // ---------------------------------------------------------------------------
1664
1389
  main();
1665
- //# sourceMappingURL=cli.js.map
1390
+ //# sourceMappingURL=index.js.map