oh-my-knowledge 0.23.0 → 0.25.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (196) hide show
  1. package/README.md +146 -84
  2. package/README.zh.md +144 -83
  3. package/dist/src/analysis/report-diagnostics.d.ts +3 -2
  4. package/dist/src/analysis/report-diagnostics.d.ts.map +1 -1
  5. package/dist/src/analysis/report-diagnostics.js +107 -80
  6. package/dist/src/analysis/report-diagnostics.js.map +1 -1
  7. package/dist/src/analysis/sample-diagnostics.d.ts +4 -1
  8. package/dist/src/analysis/sample-diagnostics.d.ts.map +1 -1
  9. package/dist/src/analysis/sample-diagnostics.js +123 -29
  10. package/dist/src/analysis/sample-diagnostics.js.map +1 -1
  11. package/dist/src/analysis/saturation.d.ts +2 -2
  12. package/dist/src/analysis/saturation.js +2 -2
  13. package/dist/src/authoring/evolver.d.ts +12 -4
  14. package/dist/src/authoring/evolver.d.ts.map +1 -1
  15. package/dist/src/authoring/evolver.js +35 -10
  16. package/dist/src/authoring/evolver.js.map +1 -1
  17. package/dist/src/cli/i18n-dict.d.ts +1 -1
  18. package/dist/src/cli/i18n-dict.d.ts.map +1 -1
  19. package/dist/src/cli/i18n-dict.js +241 -64
  20. package/dist/src/cli/i18n-dict.js.map +1 -1
  21. package/dist/src/cli/index.js +267 -152
  22. package/dist/src/cli/index.js.map +1 -1
  23. package/dist/src/cli/parse-run-config.d.ts +32 -9
  24. package/dist/src/cli/parse-run-config.d.ts.map +1 -1
  25. package/dist/src/cli/parse-run-config.js +80 -25
  26. package/dist/src/cli/parse-run-config.js.map +1 -1
  27. package/dist/src/cli/parse-strict.d.ts +20 -0
  28. package/dist/src/cli/parse-strict.d.ts.map +1 -0
  29. package/dist/src/cli/parse-strict.js +25 -0
  30. package/dist/src/cli/parse-strict.js.map +1 -0
  31. package/dist/src/cli/progress.d.ts +1 -0
  32. package/dist/src/cli/progress.d.ts.map +1 -1
  33. package/dist/src/cli/progress.js +8 -1
  34. package/dist/src/cli/progress.js.map +1 -1
  35. package/dist/src/doctor/index.d.ts +19 -0
  36. package/dist/src/doctor/index.d.ts.map +1 -0
  37. package/dist/src/doctor/index.js +182 -0
  38. package/dist/src/doctor/index.js.map +1 -0
  39. package/dist/src/doctor/preflight.d.ts +32 -0
  40. package/dist/src/doctor/preflight.d.ts.map +1 -0
  41. package/dist/src/doctor/preflight.js +32 -0
  42. package/dist/src/doctor/preflight.js.map +1 -0
  43. package/dist/src/doctor/renderer.d.ts +13 -0
  44. package/dist/src/doctor/renderer.d.ts.map +1 -0
  45. package/dist/src/doctor/renderer.js +69 -0
  46. package/dist/src/doctor/renderer.js.map +1 -0
  47. package/dist/src/doctor/rules.d.ts +30 -0
  48. package/dist/src/doctor/rules.d.ts.map +1 -0
  49. package/dist/src/doctor/rules.js +216 -0
  50. package/dist/src/doctor/rules.js.map +1 -0
  51. package/dist/src/eval-core/bootstrap.d.ts +1 -1
  52. package/dist/src/eval-core/bootstrap.js +1 -1
  53. package/dist/src/eval-core/cache.d.ts +7 -5
  54. package/dist/src/eval-core/cache.d.ts.map +1 -1
  55. package/dist/src/eval-core/cache.js +11 -7
  56. package/dist/src/eval-core/cache.js.map +1 -1
  57. package/dist/src/eval-core/comparability.d.ts +11 -0
  58. package/dist/src/eval-core/comparability.d.ts.map +1 -0
  59. package/dist/src/eval-core/comparability.js +296 -0
  60. package/dist/src/eval-core/comparability.js.map +1 -0
  61. package/dist/src/eval-core/dependency-checker.js +1 -1
  62. package/dist/src/eval-core/dependency-checker.js.map +1 -1
  63. package/dist/src/eval-core/evaluation-execution.d.ts +23 -7
  64. package/dist/src/eval-core/evaluation-execution.d.ts.map +1 -1
  65. package/dist/src/eval-core/evaluation-execution.js +53 -21
  66. package/dist/src/eval-core/evaluation-execution.js.map +1 -1
  67. package/dist/src/eval-core/evaluation-job.d.ts +3 -5
  68. package/dist/src/eval-core/evaluation-job.d.ts.map +1 -1
  69. package/dist/src/eval-core/evaluation-job.js +2 -4
  70. package/dist/src/eval-core/evaluation-job.js.map +1 -1
  71. package/dist/src/eval-core/evaluation-reporting.d.ts +3 -1
  72. package/dist/src/eval-core/evaluation-reporting.d.ts.map +1 -1
  73. package/dist/src/eval-core/evaluation-reporting.js +67 -9
  74. package/dist/src/eval-core/evaluation-reporting.js.map +1 -1
  75. package/dist/src/eval-core/execution-strategy.js +2 -2
  76. package/dist/src/eval-core/execution-strategy.js.map +1 -1
  77. package/dist/src/eval-core/schema.d.ts.map +1 -1
  78. package/dist/src/eval-core/schema.js +20 -2
  79. package/dist/src/eval-core/schema.js.map +1 -1
  80. package/dist/src/eval-workflows/batch-evaluation-workflow.d.ts +113 -0
  81. package/dist/src/eval-workflows/batch-evaluation-workflow.d.ts.map +1 -0
  82. package/dist/src/eval-workflows/batch-evaluation-workflow.js +217 -0
  83. package/dist/src/eval-workflows/batch-evaluation-workflow.js.map +1 -0
  84. package/dist/src/eval-workflows/evaluation-pipeline.d.ts +6 -4
  85. package/dist/src/eval-workflows/evaluation-pipeline.d.ts.map +1 -1
  86. package/dist/src/eval-workflows/evaluation-pipeline.js +44 -38
  87. package/dist/src/eval-workflows/evaluation-pipeline.js.map +1 -1
  88. package/dist/src/eval-workflows/evaluation-preparation.d.ts +4 -17
  89. package/dist/src/eval-workflows/evaluation-preparation.d.ts.map +1 -1
  90. package/dist/src/eval-workflows/evaluation-preparation.js +3 -18
  91. package/dist/src/eval-workflows/evaluation-preparation.js.map +1 -1
  92. package/dist/src/eval-workflows/run-evaluation.d.ts +27 -23
  93. package/dist/src/eval-workflows/run-evaluation.d.ts.map +1 -1
  94. package/dist/src/eval-workflows/run-evaluation.js +137 -20
  95. package/dist/src/eval-workflows/run-evaluation.js.map +1 -1
  96. package/dist/src/executors/claude-cli.d.ts.map +1 -1
  97. package/dist/src/executors/claude-cli.js +11 -6
  98. package/dist/src/executors/claude-cli.js.map +1 -1
  99. package/dist/src/executors/codex-cli-trace.d.ts +10 -0
  100. package/dist/src/executors/codex-cli-trace.d.ts.map +1 -0
  101. package/dist/src/executors/codex-cli-trace.js +123 -0
  102. package/dist/src/executors/codex-cli-trace.js.map +1 -0
  103. package/dist/src/executors/codex-cli.d.ts +18 -0
  104. package/dist/src/executors/codex-cli.d.ts.map +1 -0
  105. package/dist/src/executors/codex-cli.js +254 -0
  106. package/dist/src/executors/codex-cli.js.map +1 -0
  107. package/dist/src/executors/codex-sdk.d.ts +18 -0
  108. package/dist/src/executors/codex-sdk.d.ts.map +1 -0
  109. package/dist/src/executors/codex-sdk.js +214 -0
  110. package/dist/src/executors/codex-sdk.js.map +1 -0
  111. package/dist/src/executors/gemini.d.ts.map +1 -1
  112. package/dist/src/executors/gemini.js +28 -24
  113. package/dist/src/executors/gemini.js.map +1 -1
  114. package/dist/src/executors/index.d.ts.map +1 -1
  115. package/dist/src/executors/index.js +7 -2
  116. package/dist/src/executors/index.js.map +1 -1
  117. package/dist/src/executors/runtime-fingerprint.d.ts +7 -0
  118. package/dist/src/executors/runtime-fingerprint.d.ts.map +1 -0
  119. package/dist/src/executors/runtime-fingerprint.js +277 -0
  120. package/dist/src/executors/runtime-fingerprint.js.map +1 -0
  121. package/dist/src/executors/script.d.ts.map +1 -1
  122. package/dist/src/executors/script.js +47 -55
  123. package/dist/src/executors/script.js.map +1 -1
  124. package/dist/src/executors/shared.d.ts +78 -1
  125. package/dist/src/executors/shared.d.ts.map +1 -1
  126. package/dist/src/executors/shared.js +203 -1
  127. package/dist/src/executors/shared.js.map +1 -1
  128. package/dist/src/grading/assertions.d.ts.map +1 -1
  129. package/dist/src/grading/assertions.js +22 -7
  130. package/dist/src/grading/assertions.js.map +1 -1
  131. package/dist/src/grading/gold-cli.js +4 -4
  132. package/dist/src/grading/gold-cli.js.map +1 -1
  133. package/dist/src/grading/human-gold.d.ts +1 -1
  134. package/dist/src/grading/human-gold.js +1 -1
  135. package/dist/src/grading/index.d.ts +20 -15
  136. package/dist/src/grading/index.d.ts.map +1 -1
  137. package/dist/src/grading/index.js +40 -16
  138. package/dist/src/grading/index.js.map +1 -1
  139. package/dist/src/grading/judge.d.ts +1 -1
  140. package/dist/src/grading/judge.d.ts.map +1 -1
  141. package/dist/src/grading/judge.js +76 -7
  142. package/dist/src/grading/judge.js.map +1 -1
  143. package/dist/src/inputs/eval-config.js +65 -7
  144. package/dist/src/inputs/eval-config.js.map +1 -1
  145. package/dist/src/inputs/skill-loader.d.ts +1 -1
  146. package/dist/src/inputs/skill-loader.d.ts.map +1 -1
  147. package/dist/src/inputs/skill-loader.js +1 -1
  148. package/dist/src/inputs/skill-loader.js.map +1 -1
  149. package/dist/src/renderer/html-renderer.d.ts +5 -4
  150. package/dist/src/renderer/html-renderer.d.ts.map +1 -1
  151. package/dist/src/renderer/html-renderer.js +240 -96
  152. package/dist/src/renderer/html-renderer.js.map +1 -1
  153. package/dist/src/renderer/layout.d.ts +2 -1
  154. package/dist/src/renderer/layout.d.ts.map +1 -1
  155. package/dist/src/renderer/layout.js +30 -42
  156. package/dist/src/renderer/layout.js.map +1 -1
  157. package/dist/src/renderer/summary.d.ts +3 -3
  158. package/dist/src/renderer/summary.d.ts.map +1 -1
  159. package/dist/src/renderer/summary.js +233 -43
  160. package/dist/src/renderer/summary.js.map +1 -1
  161. package/dist/src/renderer/trends.d.ts.map +1 -1
  162. package/dist/src/renderer/trends.js +5 -3
  163. package/dist/src/renderer/trends.js.map +1 -1
  164. package/dist/src/server/report-server.js +4 -4
  165. package/dist/src/server/report-server.js.map +1 -1
  166. package/dist/src/server/report-store.d.ts +7 -5
  167. package/dist/src/server/report-store.d.ts.map +1 -1
  168. package/dist/src/server/report-store.js +39 -11
  169. package/dist/src/server/report-store.js.map +1 -1
  170. package/dist/src/types/doctor.d.ts +95 -0
  171. package/dist/src/types/doctor.d.ts.map +1 -0
  172. package/dist/src/types/doctor.js +2 -0
  173. package/dist/src/types/doctor.js.map +1 -0
  174. package/dist/src/types/eval.d.ts +40 -19
  175. package/dist/src/types/eval.d.ts.map +1 -1
  176. package/dist/src/types/executor.d.ts +38 -0
  177. package/dist/src/types/executor.d.ts.map +1 -1
  178. package/dist/src/types/index.d.ts +1 -0
  179. package/dist/src/types/index.d.ts.map +1 -1
  180. package/dist/src/types/index.js +1 -0
  181. package/dist/src/types/index.js.map +1 -1
  182. package/dist/src/types/judge.d.ts +21 -0
  183. package/dist/src/types/judge.d.ts.map +1 -1
  184. package/dist/src/types/report.d.ts +100 -35
  185. package/dist/src/types/report.d.ts.map +1 -1
  186. package/dist/src/types/storage.d.ts +7 -7
  187. package/dist/src/types/storage.d.ts.map +1 -1
  188. package/package.json +6 -5
  189. package/dist/src/eval-workflows/each-evaluation-workflow.d.ts +0 -153
  190. package/dist/src/eval-workflows/each-evaluation-workflow.d.ts.map +0 -1
  191. package/dist/src/eval-workflows/each-evaluation-workflow.js +0 -178
  192. package/dist/src/eval-workflows/each-evaluation-workflow.js.map +0 -1
  193. package/dist/src/executors/openai-cli.d.ts +0 -3
  194. package/dist/src/executors/openai-cli.d.ts.map +0 -1
  195. package/dist/src/executors/openai-cli.js +0 -60
  196. package/dist/src/executors/openai-cli.js.map +0 -1
@@ -1,5 +1,4 @@
1
1
  #!/usr/bin/env node
2
- import { parseArgs } from 'node:util';
3
2
  import { resolve } from 'node:path';
4
3
  import { join } from 'node:path';
5
4
  import { existsSync } from 'node:fs';
@@ -7,6 +6,20 @@ import { tCli, getCliLang, parseLangFromArgv, langFromArgv } from './i18n.js';
7
6
  import { parseRunConfig, DEFAULT_REPORTS_DIR, COMMON_OPTIONS, } from './parse-run-config.js';
8
7
  import { makeOnProgress } from './progress.js';
9
8
  import { checkUpdate } from './update-check.js';
9
+ import { parseArgsStrictOrExit } from './parse-strict.js';
10
+ function requireEvaluationReport(report, id, lang) {
11
+ if (!report) {
12
+ console.error(tCli('cli.common.report_not_found', lang, { id }));
13
+ process.exit(1);
14
+ }
15
+ if (report.kind === 'batch-evaluation') {
16
+ console.error(lang === 'zh'
17
+ ? `报告 ${id} 是 BatchEvaluationReport。该命令需要单次 EvaluationReport;请使用其中的 child reportId。`
18
+ : `Report ${id} is a BatchEvaluationReport. This command requires an EvaluationReport; use a child reportId from the batch.`);
19
+ process.exit(1);
20
+ }
21
+ return report;
22
+ }
10
23
  // ---------------------------------------------------------------------------
11
24
  // Main
12
25
  // ---------------------------------------------------------------------------
@@ -23,6 +36,11 @@ async function main() {
23
36
  await handleAnalyze(args);
24
37
  return;
25
38
  }
39
+ if (domain === 'doctor') {
40
+ const args = command ? [command, ...rest] : [];
41
+ await handleDoctor(args);
42
+ return;
43
+ }
26
44
  if (domain !== 'bench') {
27
45
  console.error(tCli('cli.common.unknown_domain', lang, { domain }));
28
46
  process.exit(1);
@@ -88,64 +106,46 @@ async function main() {
88
106
  // ---------------------------------------------------------------------------
89
107
  async function handleRun(argv) {
90
108
  const lang = langFromArgv(argv);
91
- const { values, config } = parseRunConfig(argv, {
109
+ if (argv.includes('--help') || argv.includes('-h')) {
110
+ console.log(tCli('cli.help.main', lang).trim());
111
+ process.exit(0);
112
+ }
113
+ // 注: 这里**不**给 parseArgs default 值, 否则 values.xxx 永远不为 undefined,
114
+ // CLI > eval.yaml > hardcoded-default 三级 fallback 区分不开 ("用户没传" vs "用户传了等于 default 值")。
115
+ // hardcoded default 在下面处理 undefined 时显式给。
116
+ const { values, config, evalConfig } = parseRunConfig(argv, {
92
117
  blind: { type: 'boolean' },
93
- repeat: { type: 'string', default: '1' },
94
- 'judge-repeat': { type: 'string', default: '1' },
95
- 'judge-models': { type: 'string' },
96
- bootstrap: { type: 'boolean', default: false },
97
- 'bootstrap-samples': { type: 'string', default: '1000' },
118
+ repeat: { type: 'string' },
119
+ 'judge-repeat': { type: 'string' },
120
+ bootstrap: { type: 'boolean' },
121
+ 'bootstrap-samples': { type: 'string' },
98
122
  'gold-dir': { type: 'string' },
99
- 'no-debias-length': { type: 'boolean', default: false },
123
+ 'no-debias-length': { type: 'boolean' },
100
124
  'budget-usd': { type: 'string' },
101
125
  'budget-per-sample-usd': { type: 'string' },
102
126
  'budget-per-sample-ms': { type: 'string' },
103
127
  });
104
- const { runEvaluation, runMultiple, runEachEvaluation } = await import('../eval-workflows/run-evaluation.js');
128
+ const { runEvaluation, runMultiple, runBatchEvaluation } = await import('../eval-workflows/run-evaluation.js');
105
129
  if (values.blind !== undefined) {
106
130
  config.blind = values.blind;
107
131
  }
108
132
  config.onProgress = makeOnProgress(lang);
109
- // --repeat 输入校验: 非 ≥1 整数时提示并钳到 1, 不静默掩盖用户错字 / 极端输入。
110
- // 提前到 --each 分支之前, 保证 each 模式也能读到 repeat (曾经 bug: --each 吞 --repeat)。
133
+ // --repeat: CLI > eval.yaml > 1. 非 ≥1 整数时提示并钳到 1。
111
134
  const repeatRaw = values.repeat;
112
- const parsedRepeat = repeatRaw !== undefined ? Number(repeatRaw) : 1;
135
+ const parsedRepeat = repeatRaw !== undefined ? Number(repeatRaw) : (evalConfig?.repeat ?? 1);
113
136
  if (repeatRaw !== undefined && (!Number.isFinite(parsedRepeat) || parsedRepeat < 1)) {
114
137
  process.stderr.write(tCli('cli.run.invalid_repeat', lang, { value: repeatRaw }));
115
138
  }
116
139
  const repeatCount = Math.max(1, Math.floor(parsedRepeat) || 1);
117
- // --judge-repeat 同样校验: 非 ≥1 整数时钳到 1
140
+ // --judge-repeat: CLI > eval.yaml > 1.
118
141
  const judgeRepeatRaw = values['judge-repeat'];
119
- const parsedJudgeRepeat = judgeRepeatRaw !== undefined ? Number(judgeRepeatRaw) : 1;
142
+ const parsedJudgeRepeat = judgeRepeatRaw !== undefined ? Number(judgeRepeatRaw) : (evalConfig?.judgeRepeat ?? 1);
120
143
  if (judgeRepeatRaw !== undefined && (!Number.isFinite(parsedJudgeRepeat) || parsedJudgeRepeat < 1)) {
121
144
  process.stderr.write(tCli('cli.run.invalid_judge_repeat', lang, { value: judgeRepeatRaw }));
122
145
  }
123
146
  const judgeRepeatCount = Math.max(1, Math.floor(parsedJudgeRepeat) || 1);
124
147
  if (judgeRepeatCount > 1)
125
148
  config.judgeRepeat = judgeRepeatCount;
126
- // --judge-models executor:model,executor:model,... -> JudgeConfig[]
127
- // 至少 2 个才进 ensemble 模式, 1 个等同于 --judge-model
128
- const judgeModelsRaw = values['judge-models'];
129
- if (judgeModelsRaw) {
130
- const parts = judgeModelsRaw.split(',').map((s) => s.trim()).filter(Boolean);
131
- const judges = parts.map((p) => {
132
- const [executor, ...modelParts] = p.split(':');
133
- const model = modelParts.join(':');
134
- if (!executor || !model) {
135
- throw new Error(tCli('cli.run.invalid_judge_models_format', lang, { part: p }));
136
- }
137
- return { executor, model };
138
- });
139
- if (judges.length >= 2) {
140
- config.judgeModels = judges;
141
- }
142
- else if (judges.length === 1) {
143
- // 单 judge 不走 ensemble, 但允许这样写, 等同于 --judge-model + --executor
144
- process.stderr.write(tCli('cli.run.judge_models_single_warning', lang, {
145
- executor: judges[0].executor, model: judges[0].model,
146
- }));
147
- }
148
- }
149
149
  // --budget-usd / --budget-per-sample-usd / --budget-per-sample-ms:
150
150
  // hard budget caps. CLI flags override config-file values. When the
151
151
  // total-USD cap is exceeded mid-run, remaining tasks are skipped and a
@@ -160,18 +160,22 @@ async function handleRun(argv) {
160
160
  ...(budgetPerSampleMs !== undefined && Number.isFinite(budgetPerSampleMs) && budgetPerSampleMs >= 0 ? { perSampleMs: budgetPerSampleMs } : {}),
161
161
  };
162
162
  }
163
- // --no-debias-length: opt out of v0.21 Phase 3a length-controlled prompt.
164
- // Default behavior is debias-on (judge prompt v3-cot-length); flag flips it
165
- // off so historical reports (judgePromptHash from v2-cot era) can be reproduced.
166
- if (values['no-debias-length']) {
163
+ // --no-debias-length / eval.yaml `lengthDebias: false`: opt out of length-controlled prompt。
164
+ // Default debias-on (judge prompt v3-cot-length); flip off only to reproduce historical reports。
165
+ // CLI 显式 --no-debias-length > eval.yaml lengthDebias > 默认 true。
166
+ const lengthDebiasOff = values['no-debias-length'] === true
167
+ || (values['no-debias-length'] === undefined && evalConfig?.lengthDebias === false);
168
+ if (lengthDebiasOff) {
167
169
  config.lengthDebias = false;
168
170
  process.stderr.write(tCli('cli.run.no_debias_length_active', lang));
169
171
  }
170
- // --bootstrap / --bootstrap-samples
171
- if (values.bootstrap) {
172
+ // --bootstrap / --bootstrap-samples: CLI > eval.yaml > default(off / 1000)。
173
+ const bootstrapEnabled = values.bootstrap === true
174
+ || (values.bootstrap === undefined && evalConfig?.bootstrap === true);
175
+ if (bootstrapEnabled) {
172
176
  config.bootstrap = true;
173
177
  const bsRaw = values['bootstrap-samples'];
174
- const parsedBs = bsRaw !== undefined ? Number(bsRaw) : 1000;
178
+ const parsedBs = bsRaw !== undefined ? Number(bsRaw) : (evalConfig?.bootstrapSamples ?? 1000);
175
179
  if (bsRaw !== undefined && (!Number.isFinite(parsedBs) || parsedBs < 100)) {
176
180
  process.stderr.write(tCli('cli.run.invalid_bootstrap_samples', lang, { value: bsRaw }));
177
181
  }
@@ -181,10 +185,15 @@ async function handleRun(argv) {
181
185
  }
182
186
  config.bootstrapSamples = bsCount;
183
187
  }
188
+ // 注入 lang 让 evaluation pipeline 能渲染 doctor 报告(失败时)。
189
+ config.lang = lang;
190
+ if (values['skip-connectivity']) {
191
+ process.stderr.write(tCli('cli.run.skip_connectivity_warning', lang) + '\n');
192
+ }
184
193
  try {
185
- // --each mode: evaluate each skill independently
186
- if (values.each) {
187
- const { report, filePath } = await runEachEvaluation({
194
+ // --batch mode: evaluate each skill independently
195
+ if (values.batch) {
196
+ const { report, filePath } = await runBatchEvaluation({
188
197
  ...config,
189
198
  repeat: repeatCount,
190
199
  onSkillProgress({ phase, skill, current, total }) {
@@ -237,8 +246,8 @@ async function handleRun(argv) {
237
246
  report = result.report;
238
247
  filePath = result.filePath;
239
248
  }
240
- // --gold-dir: compute α/κ/Pearson against gold annotations and re-persist.
241
- const goldDir = values['gold-dir'];
249
+ // --gold-dir / eval.yaml goldDir: compute α/κ/Pearson against gold annotations and re-persist.
250
+ const goldDir = values['gold-dir'] ?? evalConfig?.goldDir;
242
251
  if (goldDir && filePath) {
243
252
  const { attachGoldAgreementToReport, formatGoldCompare } = await import('../grading/gold-cli.js');
244
253
  const out = attachGoldAgreementToReport({
@@ -299,7 +308,11 @@ async function handleRun(argv) {
299
308
  // ---------------------------------------------------------------------------
300
309
  async function handleReport(argv) {
301
310
  const lang = langFromArgv(argv);
302
- const { values } = parseArgs({
311
+ if (argv.includes('--help') || argv.includes('-h')) {
312
+ console.log(tCli('cli.help.main', lang).trim());
313
+ process.exit(0);
314
+ }
315
+ const { values } = parseArgsStrictOrExit({
303
316
  args: argv,
304
317
  options: {
305
318
  ...COMMON_OPTIONS,
@@ -308,7 +321,6 @@ async function handleReport(argv) {
308
321
  export: { type: 'string' },
309
322
  dev: { type: 'boolean', default: false },
310
323
  },
311
- strict: false,
312
324
  });
313
325
  // Dev mode: restart server on file changes via node --watch
314
326
  if (values.dev && !process.env.__OMK_DEV_CHILD) {
@@ -330,7 +342,7 @@ async function handleReport(argv) {
330
342
  }
331
343
  if (values.export) {
332
344
  const { createFileStore } = await import('../server/report-store.js');
333
- const { renderRunDetail, renderEachRunDetail } = await import('../renderer/html-renderer.js');
345
+ const { renderReportDocumentDetail } = await import('../renderer/html-renderer.js');
334
346
  const { writeFileSync } = await import('node:fs');
335
347
  const store = createFileStore(resolve(values['reports-dir']));
336
348
  const report = await store.get(values.export);
@@ -338,7 +350,7 @@ async function handleReport(argv) {
338
350
  console.error(tCli('cli.common.report_not_found', lang, { id: values.export }));
339
351
  process.exit(1);
340
352
  }
341
- const html = report.each ? renderEachRunDetail(report) : renderRunDetail(report);
353
+ const html = renderReportDocumentDetail(report);
342
354
  const outPath = resolve(`${values.export}.html`);
343
355
  writeFileSync(outPath, html);
344
356
  console.log(`Exported to: ${outPath}`);
@@ -429,9 +441,76 @@ function parseLastWindow(spec) {
429
441
  const ms = unit === 'd' ? n * 86400_000 : unit === 'h' ? n * 3600_000 : n * 60_000;
430
442
  return new Date(Date.now() - ms).toISOString();
431
443
  }
444
+ async function handleDoctor(argv) {
445
+ const lang = langFromArgv(argv);
446
+ if (argv.includes('--help') || argv.includes('-h')) {
447
+ console.log(tCli('cli.help.doctor_usage', lang));
448
+ process.exit(0);
449
+ }
450
+ const { values, positionals } = parseArgsStrictOrExit({
451
+ args: argv,
452
+ allowPositionals: true,
453
+ options: {
454
+ ...COMMON_OPTIONS,
455
+ json: { type: 'boolean', default: false },
456
+ gate: { type: 'boolean', default: false },
457
+ executor: { type: 'string' },
458
+ model: { type: 'string' },
459
+ timeout: { type: 'string' },
460
+ },
461
+ });
462
+ const target = positionals[0] ?? null;
463
+ const executorName = values.executor ?? 'claude';
464
+ const model = values.model ?? 'sonnet';
465
+ const timeoutRaw = values.timeout;
466
+ const timeoutSec = timeoutRaw != null ? Number(timeoutRaw) : 8;
467
+ const timeoutMs = Math.max(1000, Math.floor((Number.isFinite(timeoutSec) ? timeoutSec : 8) * 1000));
468
+ const cwd = process.cwd();
469
+ const { runDoctor } = await import('../doctor/index.js');
470
+ const { renderDoctorReportText, renderDoctorReportJson } = await import('../doctor/renderer.js');
471
+ let report;
472
+ try {
473
+ report = await runDoctor({
474
+ target,
475
+ cwd,
476
+ executorName,
477
+ model,
478
+ timeoutMs,
479
+ lang,
480
+ });
481
+ }
482
+ catch (err) {
483
+ const msg = err instanceof Error ? err.message : String(err);
484
+ console.error(tCli('cli.doctor.no_skill_found', lang, { path: target ?? cwd }));
485
+ console.error(`(${msg})`);
486
+ process.exit(1);
487
+ }
488
+ if (report.skills.length === 0) {
489
+ console.error(tCli('cli.doctor.no_skill_found', lang, { path: target ?? cwd }));
490
+ process.exit(1);
491
+ }
492
+ const isJson = values.json;
493
+ const isGate = values.gate;
494
+ if (isJson) {
495
+ console.log(renderDoctorReportJson(report));
496
+ }
497
+ else if (isGate) {
498
+ // gate 模式: 静默 stdout, fail 时简短 stderr 摘要(供 CI 抓 exit code)
499
+ if (report.failed) {
500
+ const summary = lang === 'zh'
501
+ ? `doctor failed: ${report.totals.fail} 个 skill 未通过 (${report.totals.warn} warn / ${report.totals.pass} pass)`
502
+ : `doctor failed: ${report.totals.fail} skills did not pass (${report.totals.warn} warn / ${report.totals.pass} pass)`;
503
+ console.error(summary);
504
+ }
505
+ }
506
+ else {
507
+ renderDoctorReportText(report, lang);
508
+ }
509
+ process.exit(report.failed ? 1 : 0);
510
+ }
432
511
  async function handleAnalyze(argv) {
433
512
  const lang = langFromArgv(argv);
434
- const { values, positionals } = parseArgs({
513
+ const { values: rawValues, positionals } = parseArgsStrictOrExit({
435
514
  args: argv,
436
515
  allowPositionals: true,
437
516
  options: {
@@ -444,6 +523,8 @@ async function handleAnalyze(argv) {
444
523
  'output-dir': { type: 'string' },
445
524
  },
446
525
  });
526
+ // 该 handler options 全是 string-typed (无 boolean), 收紧 cast 让 caller 直接 use values.xxx 当 string 用。
527
+ const values = rawValues;
447
528
  const dir = positionals[0];
448
529
  if (!dir) {
449
530
  console.error(tCli('cli.help.analyze_usage', lang));
@@ -499,7 +580,14 @@ async function handleAnalyze(argv) {
499
580
  }
500
581
  async function handleInit(argv) {
501
582
  const lang = langFromArgv(argv);
502
- const targetDir = resolve(argv[0] || '.');
583
+ // 走 helper 让未知 option fail-fast (e.g. `omk bench init --bogus`),
584
+ // 否则 argv[0] 直接当目录名, --bogus / --lang 都会被当成 dir 写文件。
585
+ const { positionals } = parseArgsStrictOrExit({
586
+ args: argv,
587
+ allowPositionals: true,
588
+ options: { ...COMMON_OPTIONS },
589
+ });
590
+ const targetDir = resolve(positionals[0] || '.');
503
591
  const { writeFileSync, mkdirSync } = await import('node:fs');
504
592
  mkdirSync(join(targetDir, 'skills'), { recursive: true });
505
593
  writeFileSync(join(targetDir, 'eval-samples.json'), INIT_SAMPLES);
@@ -517,23 +605,26 @@ async function handleInit(argv) {
517
605
  // ---------------------------------------------------------------------------
518
606
  async function handleGenSamples(argv) {
519
607
  const lang = langFromArgv(argv);
520
- const { values } = parseArgs({
608
+ if (argv.includes('--help') || argv.includes('-h')) {
609
+ console.log(tCli('cli.help.main', lang).trim());
610
+ process.exit(0);
611
+ }
612
+ const { values } = parseArgsStrictOrExit({
521
613
  args: argv,
522
614
  options: {
523
615
  ...COMMON_OPTIONS,
524
- each: { type: 'boolean', default: false },
616
+ batch: { type: 'boolean', default: false },
525
617
  count: { type: 'string', default: '5' },
526
618
  model: { type: 'string', default: 'sonnet' },
527
619
  'skill-dir': { type: 'string', default: 'skills' },
528
620
  },
529
- strict: false,
530
621
  allowPositionals: true,
531
622
  });
532
623
  const { generateSamples } = await import('../authoring/generator.js');
533
624
  const { readFileSync, writeFileSync } = await import('node:fs');
534
625
  const count = Math.max(1, Number(values.count) || 5);
535
626
  const model = values.model;
536
- if (values.each) {
627
+ if (values.batch) {
537
628
  // Batch mode: generate for all skills missing eval-samples
538
629
  const skillDir = resolve(values['skill-dir']);
539
630
  if (!existsSync(skillDir)) {
@@ -631,7 +722,7 @@ async function handleGenSamples(argv) {
631
722
  // ---------------------------------------------------------------------------
632
723
  async function handleEvolve(argv) {
633
724
  const lang = langFromArgv(argv);
634
- const { values } = parseArgs({
725
+ const { values, positionals } = parseArgsStrictOrExit({
635
726
  args: argv,
636
727
  options: {
637
728
  ...COMMON_OPTIONS,
@@ -639,17 +730,18 @@ async function handleEvolve(argv) {
639
730
  target: { type: 'string' },
640
731
  samples: { type: 'string', default: 'eval-samples.json' },
641
732
  model: { type: 'string', default: 'sonnet' },
642
- 'judge-model': { type: 'string', default: 'haiku' },
733
+ 'judge-models': { type: 'string', default: 'claude:haiku' },
643
734
  'improve-model': { type: 'string', default: 'sonnet' },
644
735
  concurrency: { type: 'string', default: '1' },
645
736
  timeout: { type: 'string', default: '120' },
646
737
  executor: { type: 'string', default: 'claude' },
647
- 'skip-preflight': { type: 'boolean', default: false },
738
+ 'skip-connectivity': { type: 'boolean', default: false },
648
739
  },
649
- strict: false,
650
740
  allowPositionals: true,
651
741
  });
652
- const skillPath = argv.find((a) => !a.startsWith('-'));
742
+ // skill path 走 parseArgs 的 positionals (避免 raw argv.find 把 flag value
743
+ // 当成 path 误识别 — 例如 `evolve --judge-models openai-api:gpt-4o foo.md`)。
744
+ const skillPath = positionals[0];
653
745
  if (!skillPath) {
654
746
  console.error(tCli('cli.evolve.specify_skill_path', lang));
655
747
  process.exit(1);
@@ -662,6 +754,12 @@ async function handleEvolve(argv) {
662
754
  samplesFile = 'eval-samples.yml';
663
755
  }
664
756
  const { evolveSkill } = await import('../authoring/evolver.js');
757
+ const { parseJudgeModelsArgOrExit } = await import('./parse-run-config.js');
758
+ const evolveJudges = parseJudgeModelsArgOrExit(values['judge-models']);
759
+ if (evolveJudges.length > 1) {
760
+ console.error(tCli('cli.common.judge_models_single_only', lang, { cmd: 'evolve' }));
761
+ process.exit(2);
762
+ }
665
763
  process.stderr.write(tCli('cli.evolve.section_header', lang, { path: skillPath }));
666
764
  try {
667
765
  const result = await evolveSkill({
@@ -670,17 +768,20 @@ async function handleEvolve(argv) {
670
768
  rounds: Math.max(1, Number(values.rounds) || 5),
671
769
  target: values.target ? Number(values.target) : null,
672
770
  model: values.model,
673
- judgeModel: values['judge-model'],
771
+ judgeModels: evolveJudges,
674
772
  improveModel: values['improve-model'],
675
773
  executorName: values.executor,
676
774
  concurrency: Math.max(1, Number(values.concurrency) || 1),
677
775
  timeoutMs: Math.max(1, Number(values.timeout) || 120) * 1000,
678
- skipPreflight: values['skip-preflight'],
776
+ skipConnectivity: values['skip-connectivity'],
679
777
  onProgress: makeOnProgress(lang),
680
- onRoundProgress({ round, totalRounds: _totalRounds, phase, score, delta, accepted, costUSD, error }) {
778
+ onRoundProgress({ round, totalRounds: _totalRounds, phase, score, delta, accepted, costUSD, costReported, error }) {
779
+ // costReported=false 时显示「—」而不是 $0.0000(executor 不报 cost,如 codex)。
780
+ // 缺位 / true 当 reported 走旧格式。
781
+ const fmtRoundCost = (c, r) => r ? `$${c.toFixed(4)}` : '—';
681
782
  if (phase === 'baseline') {
682
783
  process.stderr.write(tCli('cli.evolve.round_baseline', lang, {
683
- score: score.toFixed(2), cost: costUSD.toFixed(4),
784
+ score: score.toFixed(2), cost: fmtRoundCost(costUSD, costReported !== false),
684
785
  }));
685
786
  }
686
787
  else if (phase === 'error') {
@@ -692,7 +793,7 @@ async function handleEvolve(argv) {
692
793
  const delta_ = delta >= 0 ? `+${delta.toFixed(2)}` : delta.toFixed(2);
693
794
  const status = accepted ? '✓ ACCEPT' : '✗ REJECT';
694
795
  process.stderr.write(tCli('cli.evolve.round_done', lang, {
695
- round, score: score.toFixed(2), delta: delta_, status, cost: costUSD.toFixed(4),
796
+ round, score: score.toFixed(2), delta: delta_, status, cost: fmtRoundCost(costUSD, costReported !== false),
696
797
  }));
697
798
  }
698
799
  },
@@ -700,9 +801,12 @@ async function handleEvolve(argv) {
700
801
  const improvement = result.startScore > 0
701
802
  ? ((result.finalScore - result.startScore) / result.startScore * 100).toFixed(1)
702
803
  : '0';
804
+ const totalCostStr = result.costReported === false
805
+ ? '—' // 任一轮的 executor 不报 cost → totalCostUSD 是 lower-bound
806
+ : `$${result.totalCostUSD.toFixed(4)}`;
703
807
  process.stderr.write(tCli('cli.evolve.summary', lang, {
704
808
  start: result.startScore.toFixed(2), final: result.finalScore.toFixed(2),
705
- percent: improvement, rounds: result.totalRounds, cost: result.totalCostUSD.toFixed(4),
809
+ percent: improvement, rounds: result.totalRounds, cost: totalCostStr,
706
810
  }));
707
811
  process.stderr.write(tCli('cli.evolve.best_path', lang, {
708
812
  best: result.bestSkillPath, target: resolve(skillPath),
@@ -737,12 +841,18 @@ async function handleGate(argv) {
737
841
  });
738
842
  const { runEvaluation } = await import('../eval-workflows/run-evaluation.js');
739
843
  config.onProgress = makeOnProgress(lang);
844
+ // 注入 lang + skip-connectivity warning(若 flag set);doctor 由 evaluation 强制调, 无 skip 选项。
845
+ config.lang = lang;
846
+ if (values['skip-connectivity']) {
847
+ process.stderr.write(tCli('cli.run.skip_connectivity_warning', lang) + '\n');
848
+ }
740
849
  try {
741
- const { report } = (await runEvaluation(config));
742
- if (report.dryRun) {
850
+ const { report: document } = (await runEvaluation(config));
851
+ if (document.dryRun) {
743
852
  console.log('Gate dry-run: no scores to check');
744
853
  process.exit(0);
745
854
  }
855
+ const report = requireEvaluationReport(document, 'current run', lang);
746
856
  // gate 内核 = run + verdict, 自动覆盖 omk 全部决策维度(三层 layer-gate /
747
857
  // bootstrap diff CI / saturation / Krippendorff α)。computeVerdict 是单一
748
858
  // 决策源, exit code 跟 verdict.level 走 — 数据 underpowered 直接 FAIL,
@@ -798,7 +908,7 @@ async function handleDiff(argv) {
798
908
  console.error(tCli('cli.help.diff_usage', lang));
799
909
  process.exit(positional.length === 0 ? 1 : 0);
800
910
  }
801
- const { values } = parseArgs({
911
+ const { values } = parseArgsStrictOrExit({
802
912
  args: flagArgs,
803
913
  options: {
804
914
  ...COMMON_OPTIONS,
@@ -807,7 +917,6 @@ async function handleDiff(argv) {
807
917
  variant: { type: 'string' },
808
918
  top: { type: 'string' },
809
919
  },
810
- strict: false,
811
920
  });
812
921
  const { createFileStore } = await import('../server/report-store.js');
813
922
  const store = createFileStore(resolve(DEFAULT_REPORTS_DIR));
@@ -816,16 +925,8 @@ async function handleDiff(argv) {
816
925
  return;
817
926
  }
818
927
  const [id1, id2] = positional;
819
- const r1 = await store.get(id1);
820
- const r2 = await store.get(id2);
821
- if (!r1) {
822
- console.error(tCli('cli.common.report_not_found', lang, { id: id1 }));
823
- process.exit(1);
824
- }
825
- if (!r2) {
826
- console.error(tCli('cli.common.report_not_found', lang, { id: id2 }));
827
- process.exit(1);
828
- }
928
+ const r1 = requireEvaluationReport(await store.get(id1), id1, lang);
929
+ const r2 = requireEvaluationReport(await store.get(id2), id2, lang);
829
930
  console.log(`\n Diff: ${id1} → ${id2}\n`);
830
931
  // Git info — r1/r2 are guaranteed non-null after process.exit() guards above
831
932
  const g1 = r1.meta?.gitInfo;
@@ -833,6 +934,10 @@ async function handleDiff(argv) {
833
934
  if (g1 || g2) {
834
935
  console.log(` Git: ${g1?.commitShort || '?'}${g1?.dirty ? '*' : ''} (${g1?.branch || '?'}) → ${g2?.commitShort || '?'}${g2?.dirty ? '*' : ''} (${g2?.branch || '?'})`);
835
936
  }
937
+ const { crossReportComparabilityWarnings, formatComparabilityWarnings } = await import('../eval-core/comparability.js');
938
+ const comparability = formatComparabilityWarnings(crossReportComparabilityWarnings(r1, r2), lang);
939
+ if (comparability)
940
+ process.stderr.write(`\n${comparability}\n\n`);
836
941
  // Per-variant comparison
837
942
  const variants = [...new Set([...(r1.meta?.variants || []), ...(r2.meta?.variants || [])])];
838
943
  for (const v of variants) {
@@ -861,8 +966,14 @@ async function handleDiff(argv) {
861
966
  }
862
967
  const cost1 = s1?.avgCostPerSample ?? 0;
863
968
  const cost2 = s2?.avgCostPerSample ?? 0;
864
- const costPct = cost1 > 0 ? ` (${cost2 > cost1 ? '+' : ''}${(((cost2 - cost1) / cost1) * 100).toFixed(0)}%)` : '';
865
- console.log(` Cost: $${cost1.toFixed(4)} → $${cost2.toFixed(4)}${costPct}`);
969
+ const reported1 = s1?.execCostReported !== false;
970
+ const reported2 = s2?.execCostReported !== false;
971
+ const fmt = (c, r) => r ? `$${c.toFixed(4)}` : '—';
972
+ // 任一边 not reported 就不报增减百分比(没意义)
973
+ const costPct = (reported1 && reported2 && cost1 > 0)
974
+ ? ` (${cost2 > cost1 ? '+' : ''}${(((cost2 - cost1) / cost1) * 100).toFixed(0)}%)`
975
+ : '';
976
+ console.log(` Cost: ${fmt(cost1, reported1)} → ${fmt(cost2, reported2)}${costPct}`);
866
977
  // Skill hash change
867
978
  const h1 = r1.meta?.artifactHashes?.[v];
868
979
  const h2 = r2.meta?.artifactHashes?.[v];
@@ -880,11 +991,11 @@ async function handleDiff(argv) {
880
991
  * `--variant` overrides which variant is the "treatment" side.
881
992
  */
882
993
  async function runSampleLevelDiff(reportId, store, flags, lang) {
883
- const report = await store.get(reportId);
884
- if (!report) {
885
- console.error(tCli('cli.common.report_not_found', lang, { id: reportId }));
886
- process.exit(1);
887
- }
994
+ const report = requireEvaluationReport(await store.get(reportId), reportId, lang);
995
+ const { reportComparabilityWarnings, formatComparabilityWarnings } = await import('../eval-core/comparability.js');
996
+ const comparability = formatComparabilityWarnings(reportComparabilityWarnings(report), lang);
997
+ if (comparability)
998
+ process.stderr.write(`\n${comparability}\n\n`);
888
999
  const variants = report.meta?.variants ?? [];
889
1000
  if (variants.length < 2) {
890
1001
  console.error('Sample-level diff needs at least 2 variants in the report.');
@@ -967,14 +1078,13 @@ async function handleGold(argv) {
967
1078
  process.exit(sub ? 0 : 1);
968
1079
  }
969
1080
  if (sub === 'init') {
970
- const { values } = parseArgs({
1081
+ const { values } = parseArgsStrictOrExit({
971
1082
  args: rest,
972
1083
  options: {
973
1084
  ...COMMON_OPTIONS,
974
1085
  out: { type: 'string', default: './gold-dataset' },
975
1086
  annotator: { type: 'string' },
976
1087
  },
977
- strict: false,
978
1088
  });
979
1089
  const { initGoldDataset } = await import('../grading/gold-cli.js');
980
1090
  try {
@@ -995,7 +1105,14 @@ async function handleGold(argv) {
995
1105
  return;
996
1106
  }
997
1107
  if (sub === 'validate') {
998
- const dir = rest[0];
1108
+ // 走 helper 让 `omk bench gold validate <dir> --bogus` 走 unknown option 路径,
1109
+ // 而不是直接执行 validate 后再报 dataset 错。
1110
+ const { positionals } = parseArgsStrictOrExit({
1111
+ args: rest,
1112
+ allowPositionals: true,
1113
+ options: { ...COMMON_OPTIONS },
1114
+ });
1115
+ const dir = positionals[0];
999
1116
  if (!dir) {
1000
1117
  console.error(tCli('cli.common.usage_gold_validate', lang));
1001
1118
  process.exit(1);
@@ -1017,7 +1134,7 @@ async function handleGold(argv) {
1017
1134
  console.error('Usage: omk bench gold compare <reportId> --gold-dir <dir>');
1018
1135
  process.exit(1);
1019
1136
  }
1020
- const { values } = parseArgs({
1137
+ const { values } = parseArgsStrictOrExit({
1021
1138
  args: rest.slice(1),
1022
1139
  options: {
1023
1140
  ...COMMON_OPTIONS,
@@ -1027,7 +1144,6 @@ async function handleGold(argv) {
1027
1144
  'bootstrap-samples': { type: 'string', default: '1000' },
1028
1145
  seed: { type: 'string' },
1029
1146
  },
1030
- strict: false,
1031
1147
  });
1032
1148
  const goldDir = values['gold-dir'];
1033
1149
  if (!goldDir) {
@@ -1050,15 +1166,11 @@ async function handleGold(argv) {
1050
1166
  console.error(`warn: ${i.message}`);
1051
1167
  }
1052
1168
  const store = createFileStore(resolve(values['reports-dir']));
1053
- const report = await store.get(reportId);
1054
- if (!report) {
1055
- console.error(tCli('cli.common.report_not_found', lang, { id: reportId }));
1056
- process.exit(1);
1057
- }
1169
+ const report = requireEvaluationReport(await store.get(reportId), reportId, lang);
1058
1170
  const samples = Math.max(100, Number(values['bootstrap-samples']) || 1000);
1059
1171
  const seedVal = values.seed != null ? Number(values.seed) : undefined;
1060
1172
  const result = compareGoldToReport({
1061
- report: report,
1173
+ report,
1062
1174
  gold: dataset,
1063
1175
  variant: values.variant,
1064
1176
  samples,
@@ -1071,7 +1183,7 @@ async function handleGold(argv) {
1071
1183
  process.exit(1);
1072
1184
  }
1073
1185
  // ---------------------------------------------------------------------------
1074
- // handleDebiasValidate — measure length-debias prompt sensitivity (Phase 3a)
1186
+ // handleDebiasValidate — measure length-debias prompt sensitivity
1075
1187
  // ---------------------------------------------------------------------------
1076
1188
  async function handleDebiasValidate(argv) {
1077
1189
  const lang = langFromArgv(argv);
@@ -1090,27 +1202,31 @@ async function handleDebiasValidate(argv) {
1090
1202
  console.error('Usage: omk bench debias-validate length <reportId>');
1091
1203
  process.exit(1);
1092
1204
  }
1093
- const { values } = parseArgs({
1205
+ const { values } = parseArgsStrictOrExit({
1094
1206
  args: rest.slice(1),
1095
1207
  options: {
1096
1208
  ...COMMON_OPTIONS,
1097
1209
  'reports-dir': { type: 'string', default: DEFAULT_REPORTS_DIR },
1098
1210
  samples: { type: 'string' },
1099
1211
  variant: { type: 'string' },
1100
- 'judge-executor': { type: 'string', default: 'claude' },
1101
- 'judge-model': { type: 'string' },
1212
+ 'judge-models': { type: 'string' },
1102
1213
  'bootstrap-samples': { type: 'string', default: '1000' },
1103
1214
  seed: { type: 'string' },
1104
1215
  },
1105
- strict: false,
1106
1216
  });
1217
+ // Parse --judge-models 在 load report 之前 fail-fast。重复 entry / 缺 executor /
1218
+ // 空串等参数错误应立即给 friendly error: + exit 2,不要等到 store IO 完成才暴露。
1219
+ const { parseJudgeModelsArgOrExit: parseJudgesA } = await import('./parse-run-config.js');
1220
+ const cliJudgeModelsA = values['judge-models'] !== undefined
1221
+ ? parseJudgesA(values['judge-models'])
1222
+ : undefined;
1223
+ if (cliJudgeModelsA && cliJudgeModelsA.length > 1) {
1224
+ console.error(tCli('cli.common.judge_models_single_only', lang, { cmd: 'debias-validate' }));
1225
+ process.exit(2);
1226
+ }
1107
1227
  const { createFileStore } = await import('../server/report-store.js');
1108
1228
  const store = createFileStore(resolve(values['reports-dir']));
1109
- const report = await store.get(reportId);
1110
- if (!report) {
1111
- console.error(tCli('cli.common.report_not_found', lang, { id: reportId }));
1112
- process.exit(1);
1113
- }
1229
+ const report = requireEvaluationReport(await store.get(reportId), reportId, lang);
1114
1230
  // Resolve samples path: --samples overrides; otherwise read from report.meta.request.
1115
1231
  const samplesPath = values.samples
1116
1232
  ?? report.meta?.request?.samplesPath;
@@ -1120,20 +1236,23 @@ async function handleDebiasValidate(argv) {
1120
1236
  }
1121
1237
  const { loadSamples } = await import('../inputs/load-samples.js');
1122
1238
  const { samples } = loadSamples(samplesPath);
1123
- const judgeModel = values['judge-model']
1124
- ?? report.meta?.judgeModel;
1125
- if (!judgeModel) {
1239
+ const debiasJudges = cliJudgeModelsA
1240
+ ?? (report.meta?.judgeModels?.[0]
1241
+ ? [{ executor: report.meta.judgeModels[0].executor, model: report.meta.judgeModels[0].model }]
1242
+ : []);
1243
+ if (debiasJudges.length === 0) {
1126
1244
  console.error(tCli('cli.common.no_judge_model', lang));
1127
1245
  process.exit(1);
1128
1246
  }
1129
1247
  process.stderr.write(tCli('cli.debias.warn_cost_doubles', lang));
1130
1248
  const { createExecutor } = await import('../executors/index.js');
1131
- const judgeExecutor = createExecutor(values['judge-executor']);
1249
+ const judgeExecutor = createExecutor(debiasJudges[0].executor);
1250
+ const judgeModel = debiasJudges[0].model;
1132
1251
  const { validateLengthDebias, formatDebiasValidate } = await import('../grading/debias-validate.js');
1133
1252
  const seedVal = values.seed != null ? Number(values.seed) : undefined;
1134
1253
  const bsRaw = Number(values['bootstrap-samples']) || 1000;
1135
1254
  const result = await validateLengthDebias({
1136
- report: report,
1255
+ report,
1137
1256
  samples,
1138
1257
  judgeExecutor,
1139
1258
  judgeModel,
@@ -1156,22 +1275,17 @@ async function handleSaturation(argv) {
1156
1275
  console.log(tCli('cli.help.saturation', lang));
1157
1276
  process.exit(reportId ? 0 : 1);
1158
1277
  }
1159
- const { values } = parseArgs({
1278
+ const { values } = parseArgsStrictOrExit({
1160
1279
  args: argv.slice(1),
1161
1280
  options: {
1162
1281
  ...COMMON_OPTIONS,
1163
1282
  'reports-dir': { type: 'string', default: DEFAULT_REPORTS_DIR },
1164
1283
  variant: { type: 'string' },
1165
1284
  },
1166
- strict: false,
1167
1285
  });
1168
1286
  const { createFileStore } = await import('../server/report-store.js');
1169
1287
  const store = createFileStore(resolve(values['reports-dir']));
1170
- const report = await store.get(reportId);
1171
- if (!report) {
1172
- console.error(tCli('cli.common.report_not_found', lang, { id: reportId }));
1173
- process.exit(1);
1174
- }
1288
+ const report = requireEvaluationReport(await store.get(reportId), reportId, lang);
1175
1289
  const saturation = report.variance?.saturation;
1176
1290
  if (!saturation) {
1177
1291
  console.error(tCli('cli.saturation.no_data', lang));
@@ -1225,7 +1339,7 @@ async function handleVerdict(argv) {
1225
1339
  console.log(tCli('cli.help.verdict', lang));
1226
1340
  process.exit(reportId ? 0 : 1);
1227
1341
  }
1228
- const { values } = parseArgs({
1342
+ const { values } = parseArgsStrictOrExit({
1229
1343
  args: argv.slice(1),
1230
1344
  options: {
1231
1345
  ...COMMON_OPTIONS,
@@ -1234,16 +1348,15 @@ async function handleVerdict(argv) {
1234
1348
  'trivial-diff': { type: 'string' },
1235
1349
  verbose: { type: 'boolean', default: false },
1236
1350
  },
1237
- strict: false,
1238
1351
  });
1239
1352
  const { createFileStore } = await import('../server/report-store.js');
1240
1353
  const store = createFileStore(resolve(values['reports-dir']));
1241
- const report = await store.get(reportId);
1242
- if (!report) {
1243
- console.error(tCli('cli.common.report_not_found', lang, { id: reportId }));
1244
- process.exit(1);
1245
- }
1354
+ const report = requireEvaluationReport(await store.get(reportId), reportId, lang);
1246
1355
  const { computeVerdict, formatVerdictText } = await import('../eval-core/verdict.js');
1356
+ const { reportComparabilityWarnings, formatComparabilityWarnings } = await import('../eval-core/comparability.js');
1357
+ const comparability = formatComparabilityWarnings(reportComparabilityWarnings(report), lang);
1358
+ if (comparability)
1359
+ process.stderr.write(`${comparability}\n`);
1247
1360
  const result = computeVerdict(report, {
1248
1361
  gateThreshold: values.threshold != null ? Number(values.threshold) : undefined,
1249
1362
  triviallySmallDiff: values['trivial-diff'] != null ? Number(values['trivial-diff']) : undefined,
@@ -1270,7 +1383,7 @@ async function handleDiagnose(argv) {
1270
1383
  console.log(tCli('cli.help.diagnose', lang));
1271
1384
  process.exit(reportId ? 0 : 1);
1272
1385
  }
1273
- const { values } = parseArgs({
1386
+ const { values } = parseArgsStrictOrExit({
1274
1387
  args: argv.slice(1),
1275
1388
  options: {
1276
1389
  ...COMMON_OPTIONS,
@@ -1283,15 +1396,10 @@ async function handleDiagnose(argv) {
1283
1396
  'latency-k': { type: 'string' },
1284
1397
  flat: { type: 'string' },
1285
1398
  },
1286
- strict: false,
1287
1399
  });
1288
1400
  const { createFileStore } = await import('../server/report-store.js');
1289
1401
  const store = createFileStore(resolve(values['reports-dir']));
1290
- const report = await store.get(reportId);
1291
- if (!report) {
1292
- console.error(tCli('cli.common.report_not_found', lang, { id: reportId }));
1293
- process.exit(1);
1294
- }
1402
+ const report = requireEvaluationReport(await store.get(reportId), reportId, lang);
1295
1403
  // Try to read the samples file for near-duplicate detection. Source order:
1296
1404
  // 1. --samples <path> override
1297
1405
  // 2. report.meta.request.samplesPath (recorded at run time)
@@ -1320,7 +1428,7 @@ async function handleDiagnose(argv) {
1320
1428
  latencyOutlierK: values['latency-k'] != null ? Number(values['latency-k']) : undefined,
1321
1429
  flatThreshold: values.flat != null ? Number(values.flat) : undefined,
1322
1430
  });
1323
- console.log(formatSampleDiagnostics(diag, { topN }));
1431
+ console.log(formatSampleDiagnostics(diag, { topN, lang }));
1324
1432
  // Sample design science coverage block. Render after diagnose 主体,因为
1325
1433
  // coverage 是声明式元数据(capability/difficulty/construct/provenance)的整体分布,
1326
1434
  // 跟 issue list 是不同视角的两件事。优先从 samples (现场加载) 算,fallback 到
@@ -1345,36 +1453,43 @@ async function handleFailures(argv) {
1345
1453
  console.log(tCli('cli.help.failures', lang));
1346
1454
  process.exit(reportId ? 0 : 1);
1347
1455
  }
1348
- const { values } = parseArgs({
1456
+ const { values } = parseArgsStrictOrExit({
1349
1457
  args: argv.slice(1),
1350
1458
  options: {
1351
1459
  ...COMMON_OPTIONS,
1352
1460
  'reports-dir': { type: 'string', default: DEFAULT_REPORTS_DIR },
1353
- 'judge-executor': { type: 'string', default: 'claude' },
1354
- 'judge-model': { type: 'string' },
1461
+ 'judge-models': { type: 'string' },
1355
1462
  'max-clusters': { type: 'string', default: '5' },
1356
1463
  threshold: { type: 'string', default: '3' },
1357
1464
  'max-feed': { type: 'string', default: '50' },
1358
1465
  },
1359
- strict: false,
1360
1466
  });
1467
+ // Parse --judge-models 在 load report 之前 fail-fast(同 debias-validate)。
1468
+ const { parseJudgeModelsArgOrExit: parseJudgesB } = await import('./parse-run-config.js');
1469
+ const cliJudgeModelsB = values['judge-models'] !== undefined
1470
+ ? parseJudgesB(values['judge-models'])
1471
+ : undefined;
1472
+ if (cliJudgeModelsB && cliJudgeModelsB.length > 1) {
1473
+ console.error(tCli('cli.common.judge_models_single_only', lang, { cmd: 'failures' }));
1474
+ process.exit(2);
1475
+ }
1361
1476
  const { createFileStore } = await import('../server/report-store.js');
1362
1477
  const store = createFileStore(resolve(values['reports-dir']));
1363
- const report = await store.get(reportId);
1364
- if (!report) {
1365
- console.error(tCli('cli.common.report_not_found', lang, { id: reportId }));
1366
- process.exit(1);
1367
- }
1368
- const judgeModel = values['judge-model'] ?? report.meta?.judgeModel;
1369
- if (!judgeModel) {
1478
+ const report = requireEvaluationReport(await store.get(reportId), reportId, lang);
1479
+ const failuresJudges = cliJudgeModelsB
1480
+ ?? (report.meta?.judgeModels?.[0]
1481
+ ? [{ executor: report.meta.judgeModels[0].executor, model: report.meta.judgeModels[0].model }]
1482
+ : []);
1483
+ if (failuresJudges.length === 0) {
1370
1484
  console.error(tCli('cli.common.no_judge_model', lang));
1371
1485
  process.exit(1);
1372
1486
  }
1373
1487
  const { createExecutor } = await import('../executors/index.js');
1374
- const executor = createExecutor(values['judge-executor']);
1488
+ const executor = createExecutor(failuresJudges[0].executor);
1489
+ const judgeModel = failuresJudges[0].model;
1375
1490
  const { clusterFailures, formatFailureClusterReport } = await import('../analysis/failure-clusterer.js');
1376
1491
  const out = await clusterFailures({
1377
- report: report,
1492
+ report,
1378
1493
  executor,
1379
1494
  judgeModel,
1380
1495
  maxClusters: Number(values['max-clusters']) || 5,