oh-my-knowledge 0.24.0 → 0.25.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (111) hide show
  1. package/README.md +84 -21
  2. package/README.zh.md +81 -21
  3. package/dist/src/analysis/saturation.d.ts +2 -2
  4. package/dist/src/analysis/saturation.js +2 -2
  5. package/dist/src/authoring/evolver.d.ts +7 -4
  6. package/dist/src/authoring/evolver.d.ts.map +1 -1
  7. package/dist/src/authoring/evolver.js +15 -7
  8. package/dist/src/authoring/evolver.js.map +1 -1
  9. package/dist/src/cli/i18n-dict.d.ts +1 -1
  10. package/dist/src/cli/i18n-dict.d.ts.map +1 -1
  11. package/dist/src/cli/i18n-dict.js +213 -40
  12. package/dist/src/cli/i18n-dict.js.map +1 -1
  13. package/dist/src/cli/index.js +188 -87
  14. package/dist/src/cli/index.js.map +1 -1
  15. package/dist/src/cli/parse-run-config.d.ts +32 -9
  16. package/dist/src/cli/parse-run-config.d.ts.map +1 -1
  17. package/dist/src/cli/parse-run-config.js +76 -21
  18. package/dist/src/cli/parse-run-config.js.map +1 -1
  19. package/dist/src/cli/parse-strict.d.ts +20 -0
  20. package/dist/src/cli/parse-strict.d.ts.map +1 -0
  21. package/dist/src/cli/parse-strict.js +25 -0
  22. package/dist/src/cli/parse-strict.js.map +1 -0
  23. package/dist/src/doctor/index.d.ts +19 -0
  24. package/dist/src/doctor/index.d.ts.map +1 -0
  25. package/dist/src/doctor/index.js +182 -0
  26. package/dist/src/doctor/index.js.map +1 -0
  27. package/dist/src/doctor/preflight.d.ts +32 -0
  28. package/dist/src/doctor/preflight.d.ts.map +1 -0
  29. package/dist/src/doctor/preflight.js +32 -0
  30. package/dist/src/doctor/preflight.js.map +1 -0
  31. package/dist/src/doctor/renderer.d.ts +13 -0
  32. package/dist/src/doctor/renderer.d.ts.map +1 -0
  33. package/dist/src/doctor/renderer.js +69 -0
  34. package/dist/src/doctor/renderer.js.map +1 -0
  35. package/dist/src/doctor/rules.d.ts +30 -0
  36. package/dist/src/doctor/rules.d.ts.map +1 -0
  37. package/dist/src/doctor/rules.js +216 -0
  38. package/dist/src/doctor/rules.js.map +1 -0
  39. package/dist/src/eval-core/bootstrap.d.ts +1 -1
  40. package/dist/src/eval-core/bootstrap.js +1 -1
  41. package/dist/src/eval-core/comparability.d.ts.map +1 -1
  42. package/dist/src/eval-core/comparability.js +82 -57
  43. package/dist/src/eval-core/comparability.js.map +1 -1
  44. package/dist/src/eval-core/dependency-checker.js +1 -1
  45. package/dist/src/eval-core/dependency-checker.js.map +1 -1
  46. package/dist/src/eval-core/evaluation-execution.d.ts +20 -7
  47. package/dist/src/eval-core/evaluation-execution.d.ts.map +1 -1
  48. package/dist/src/eval-core/evaluation-execution.js +29 -5
  49. package/dist/src/eval-core/evaluation-execution.js.map +1 -1
  50. package/dist/src/eval-core/evaluation-job.d.ts +2 -4
  51. package/dist/src/eval-core/evaluation-job.d.ts.map +1 -1
  52. package/dist/src/eval-core/evaluation-job.js +1 -3
  53. package/dist/src/eval-core/evaluation-job.js.map +1 -1
  54. package/dist/src/eval-core/evaluation-reporting.d.ts.map +1 -1
  55. package/dist/src/eval-core/evaluation-reporting.js +12 -14
  56. package/dist/src/eval-core/evaluation-reporting.js.map +1 -1
  57. package/dist/src/eval-core/schema.d.ts.map +1 -1
  58. package/dist/src/eval-core/schema.js +7 -2
  59. package/dist/src/eval-core/schema.js.map +1 -1
  60. package/dist/src/eval-workflows/batch-evaluation-workflow.d.ts +6 -4
  61. package/dist/src/eval-workflows/batch-evaluation-workflow.d.ts.map +1 -1
  62. package/dist/src/eval-workflows/batch-evaluation-workflow.js +14 -12
  63. package/dist/src/eval-workflows/batch-evaluation-workflow.js.map +1 -1
  64. package/dist/src/eval-workflows/evaluation-pipeline.d.ts +2 -2
  65. package/dist/src/eval-workflows/evaluation-pipeline.d.ts.map +1 -1
  66. package/dist/src/eval-workflows/evaluation-pipeline.js +39 -34
  67. package/dist/src/eval-workflows/evaluation-pipeline.js.map +1 -1
  68. package/dist/src/eval-workflows/evaluation-preparation.d.ts +5 -20
  69. package/dist/src/eval-workflows/evaluation-preparation.d.ts.map +1 -1
  70. package/dist/src/eval-workflows/evaluation-preparation.js +2 -21
  71. package/dist/src/eval-workflows/evaluation-preparation.js.map +1 -1
  72. package/dist/src/eval-workflows/run-evaluation.d.ts +11 -10
  73. package/dist/src/eval-workflows/run-evaluation.d.ts.map +1 -1
  74. package/dist/src/eval-workflows/run-evaluation.js +123 -16
  75. package/dist/src/eval-workflows/run-evaluation.js.map +1 -1
  76. package/dist/src/grading/assertions.js +4 -4
  77. package/dist/src/grading/assertions.js.map +1 -1
  78. package/dist/src/grading/gold-cli.js +4 -4
  79. package/dist/src/grading/gold-cli.js.map +1 -1
  80. package/dist/src/grading/human-gold.d.ts +1 -1
  81. package/dist/src/grading/human-gold.js +1 -1
  82. package/dist/src/grading/index.d.ts +20 -15
  83. package/dist/src/grading/index.d.ts.map +1 -1
  84. package/dist/src/grading/index.js +29 -16
  85. package/dist/src/grading/index.js.map +1 -1
  86. package/dist/src/grading/judge.d.ts +1 -1
  87. package/dist/src/grading/judge.js +1 -1
  88. package/dist/src/inputs/eval-config.js +65 -7
  89. package/dist/src/inputs/eval-config.js.map +1 -1
  90. package/dist/src/renderer/html-renderer.d.ts.map +1 -1
  91. package/dist/src/renderer/html-renderer.js +32 -12
  92. package/dist/src/renderer/html-renderer.js.map +1 -1
  93. package/dist/src/renderer/summary.js +1 -1
  94. package/dist/src/renderer/summary.js.map +1 -1
  95. package/dist/src/types/doctor.d.ts +95 -0
  96. package/dist/src/types/doctor.d.ts.map +1 -0
  97. package/dist/src/types/doctor.js +2 -0
  98. package/dist/src/types/doctor.js.map +1 -0
  99. package/dist/src/types/eval.d.ts +36 -17
  100. package/dist/src/types/eval.d.ts.map +1 -1
  101. package/dist/src/types/executor.d.ts +33 -0
  102. package/dist/src/types/executor.d.ts.map +1 -1
  103. package/dist/src/types/index.d.ts +1 -0
  104. package/dist/src/types/index.d.ts.map +1 -1
  105. package/dist/src/types/index.js +1 -0
  106. package/dist/src/types/index.js.map +1 -1
  107. package/dist/src/types/judge.d.ts +11 -0
  108. package/dist/src/types/judge.d.ts.map +1 -1
  109. package/dist/src/types/report.d.ts +21 -52
  110. package/dist/src/types/report.d.ts.map +1 -1
  111. package/package.json +5 -5
@@ -1,5 +1,4 @@
1
1
  #!/usr/bin/env node
2
- import { parseArgs } from 'node:util';
3
2
  import { resolve } from 'node:path';
4
3
  import { join } from 'node:path';
5
4
  import { existsSync } from 'node:fs';
@@ -7,6 +6,7 @@ import { tCli, getCliLang, parseLangFromArgv, langFromArgv } from './i18n.js';
7
6
  import { parseRunConfig, DEFAULT_REPORTS_DIR, COMMON_OPTIONS, } from './parse-run-config.js';
8
7
  import { makeOnProgress } from './progress.js';
9
8
  import { checkUpdate } from './update-check.js';
9
+ import { parseArgsStrictOrExit } from './parse-strict.js';
10
10
  function requireEvaluationReport(report, id, lang) {
11
11
  if (!report) {
12
12
  console.error(tCli('cli.common.report_not_found', lang, { id }));
@@ -36,6 +36,11 @@ async function main() {
36
36
  await handleAnalyze(args);
37
37
  return;
38
38
  }
39
+ if (domain === 'doctor') {
40
+ const args = command ? [command, ...rest] : [];
41
+ await handleDoctor(args);
42
+ return;
43
+ }
39
44
  if (domain !== 'bench') {
40
45
  console.error(tCli('cli.common.unknown_domain', lang, { domain }));
41
46
  process.exit(1);
@@ -105,15 +110,17 @@ async function handleRun(argv) {
105
110
  console.log(tCli('cli.help.main', lang).trim());
106
111
  process.exit(0);
107
112
  }
108
- const { values, config } = parseRunConfig(argv, {
113
+ // 注: 这里**不**给 parseArgs default 值, 否则 values.xxx 永远不为 undefined,
114
+ // CLI > eval.yaml > hardcoded-default 三级 fallback 区分不开 ("用户没传" vs "用户传了等于 default 值")。
115
+ // hardcoded default 在下面处理 undefined 时显式给。
116
+ const { values, config, evalConfig } = parseRunConfig(argv, {
109
117
  blind: { type: 'boolean' },
110
- repeat: { type: 'string', default: '1' },
111
- 'judge-repeat': { type: 'string', default: '1' },
112
- 'judge-models': { type: 'string' },
113
- bootstrap: { type: 'boolean', default: false },
114
- 'bootstrap-samples': { type: 'string', default: '1000' },
118
+ repeat: { type: 'string' },
119
+ 'judge-repeat': { type: 'string' },
120
+ bootstrap: { type: 'boolean' },
121
+ 'bootstrap-samples': { type: 'string' },
115
122
  'gold-dir': { type: 'string' },
116
- 'no-debias-length': { type: 'boolean', default: false },
123
+ 'no-debias-length': { type: 'boolean' },
117
124
  'budget-usd': { type: 'string' },
118
125
  'budget-per-sample-usd': { type: 'string' },
119
126
  'budget-per-sample-ms': { type: 'string' },
@@ -123,46 +130,22 @@ async function handleRun(argv) {
123
130
  config.blind = values.blind;
124
131
  }
125
132
  config.onProgress = makeOnProgress(lang);
126
- // --repeat 输入校验: 非 ≥1 整数时提示并钳到 1, 不静默掩盖用户错字 / 极端输入。
127
- // 提前到 --batch 分支之前, 保证 batch 模式也能读到 repeat。
133
+ // --repeat: CLI > eval.yaml > 1. 非 ≥1 整数时提示并钳到 1。
128
134
  const repeatRaw = values.repeat;
129
- const parsedRepeat = repeatRaw !== undefined ? Number(repeatRaw) : 1;
135
+ const parsedRepeat = repeatRaw !== undefined ? Number(repeatRaw) : (evalConfig?.repeat ?? 1);
130
136
  if (repeatRaw !== undefined && (!Number.isFinite(parsedRepeat) || parsedRepeat < 1)) {
131
137
  process.stderr.write(tCli('cli.run.invalid_repeat', lang, { value: repeatRaw }));
132
138
  }
133
139
  const repeatCount = Math.max(1, Math.floor(parsedRepeat) || 1);
134
- // --judge-repeat 同样校验: 非 ≥1 整数时钳到 1
140
+ // --judge-repeat: CLI > eval.yaml > 1.
135
141
  const judgeRepeatRaw = values['judge-repeat'];
136
- const parsedJudgeRepeat = judgeRepeatRaw !== undefined ? Number(judgeRepeatRaw) : 1;
142
+ const parsedJudgeRepeat = judgeRepeatRaw !== undefined ? Number(judgeRepeatRaw) : (evalConfig?.judgeRepeat ?? 1);
137
143
  if (judgeRepeatRaw !== undefined && (!Number.isFinite(parsedJudgeRepeat) || parsedJudgeRepeat < 1)) {
138
144
  process.stderr.write(tCli('cli.run.invalid_judge_repeat', lang, { value: judgeRepeatRaw }));
139
145
  }
140
146
  const judgeRepeatCount = Math.max(1, Math.floor(parsedJudgeRepeat) || 1);
141
147
  if (judgeRepeatCount > 1)
142
148
  config.judgeRepeat = judgeRepeatCount;
143
- // --judge-models executor:model,executor:model,... -> JudgeConfig[]
144
- // 至少 2 个才进 ensemble 模式, 1 个等同于 --judge-model
145
- const judgeModelsRaw = values['judge-models'];
146
- if (judgeModelsRaw) {
147
- const parts = judgeModelsRaw.split(',').map((s) => s.trim()).filter(Boolean);
148
- const judges = parts.map((p) => {
149
- const [executor, ...modelParts] = p.split(':');
150
- const model = modelParts.join(':');
151
- if (!executor || !model) {
152
- throw new Error(tCli('cli.run.invalid_judge_models_format', lang, { part: p }));
153
- }
154
- return { executor, model };
155
- });
156
- if (judges.length >= 2) {
157
- config.judgeModels = judges;
158
- }
159
- else if (judges.length === 1) {
160
- // 单 judge 不走 ensemble, 但允许这样写, 等同于 --judge-model + --executor
161
- process.stderr.write(tCli('cli.run.judge_models_single_warning', lang, {
162
- executor: judges[0].executor, model: judges[0].model,
163
- }));
164
- }
165
- }
166
149
  // --budget-usd / --budget-per-sample-usd / --budget-per-sample-ms:
167
150
  // hard budget caps. CLI flags override config-file values. When the
168
151
  // total-USD cap is exceeded mid-run, remaining tasks are skipped and a
@@ -177,18 +160,22 @@ async function handleRun(argv) {
177
160
  ...(budgetPerSampleMs !== undefined && Number.isFinite(budgetPerSampleMs) && budgetPerSampleMs >= 0 ? { perSampleMs: budgetPerSampleMs } : {}),
178
161
  };
179
162
  }
180
- // --no-debias-length: opt out of v0.21 Phase 3a length-controlled prompt.
181
- // Default behavior is debias-on (judge prompt v3-cot-length); flag flips it
182
- // off so historical reports (judgePromptHash from v2-cot era) can be reproduced.
183
- if (values['no-debias-length']) {
163
+ // --no-debias-length / eval.yaml `lengthDebias: false`: opt out of length-controlled prompt。
164
+ // Default debias-on (judge prompt v3-cot-length); flip off only to reproduce historical reports。
165
+ // CLI 显式 --no-debias-length > eval.yaml lengthDebias > 默认 true。
166
+ const lengthDebiasOff = values['no-debias-length'] === true
167
+ || (values['no-debias-length'] === undefined && evalConfig?.lengthDebias === false);
168
+ if (lengthDebiasOff) {
184
169
  config.lengthDebias = false;
185
170
  process.stderr.write(tCli('cli.run.no_debias_length_active', lang));
186
171
  }
187
- // --bootstrap / --bootstrap-samples
188
- if (values.bootstrap) {
172
+ // --bootstrap / --bootstrap-samples: CLI > eval.yaml > default(off / 1000)。
173
+ const bootstrapEnabled = values.bootstrap === true
174
+ || (values.bootstrap === undefined && evalConfig?.bootstrap === true);
175
+ if (bootstrapEnabled) {
189
176
  config.bootstrap = true;
190
177
  const bsRaw = values['bootstrap-samples'];
191
- const parsedBs = bsRaw !== undefined ? Number(bsRaw) : 1000;
178
+ const parsedBs = bsRaw !== undefined ? Number(bsRaw) : (evalConfig?.bootstrapSamples ?? 1000);
192
179
  if (bsRaw !== undefined && (!Number.isFinite(parsedBs) || parsedBs < 100)) {
193
180
  process.stderr.write(tCli('cli.run.invalid_bootstrap_samples', lang, { value: bsRaw }));
194
181
  }
@@ -198,6 +185,11 @@ async function handleRun(argv) {
198
185
  }
199
186
  config.bootstrapSamples = bsCount;
200
187
  }
188
+ // 注入 lang 让 evaluation pipeline 能渲染 doctor 报告(失败时)。
189
+ config.lang = lang;
190
+ if (values['skip-connectivity']) {
191
+ process.stderr.write(tCli('cli.run.skip_connectivity_warning', lang) + '\n');
192
+ }
201
193
  try {
202
194
  // --batch mode: evaluate each skill independently
203
195
  if (values.batch) {
@@ -254,8 +246,8 @@ async function handleRun(argv) {
254
246
  report = result.report;
255
247
  filePath = result.filePath;
256
248
  }
257
- // --gold-dir: compute α/κ/Pearson against gold annotations and re-persist.
258
- const goldDir = values['gold-dir'];
249
+ // --gold-dir / eval.yaml goldDir: compute α/κ/Pearson against gold annotations and re-persist.
250
+ const goldDir = values['gold-dir'] ?? evalConfig?.goldDir;
259
251
  if (goldDir && filePath) {
260
252
  const { attachGoldAgreementToReport, formatGoldCompare } = await import('../grading/gold-cli.js');
261
253
  const out = attachGoldAgreementToReport({
@@ -320,7 +312,7 @@ async function handleReport(argv) {
320
312
  console.log(tCli('cli.help.main', lang).trim());
321
313
  process.exit(0);
322
314
  }
323
- const { values } = parseArgs({
315
+ const { values } = parseArgsStrictOrExit({
324
316
  args: argv,
325
317
  options: {
326
318
  ...COMMON_OPTIONS,
@@ -329,7 +321,6 @@ async function handleReport(argv) {
329
321
  export: { type: 'string' },
330
322
  dev: { type: 'boolean', default: false },
331
323
  },
332
- strict: false,
333
324
  });
334
325
  // Dev mode: restart server on file changes via node --watch
335
326
  if (values.dev && !process.env.__OMK_DEV_CHILD) {
@@ -450,9 +441,76 @@ function parseLastWindow(spec) {
450
441
  const ms = unit === 'd' ? n * 86400_000 : unit === 'h' ? n * 3600_000 : n * 60_000;
451
442
  return new Date(Date.now() - ms).toISOString();
452
443
  }
444
+ async function handleDoctor(argv) {
445
+ const lang = langFromArgv(argv);
446
+ if (argv.includes('--help') || argv.includes('-h')) {
447
+ console.log(tCli('cli.help.doctor_usage', lang));
448
+ process.exit(0);
449
+ }
450
+ const { values, positionals } = parseArgsStrictOrExit({
451
+ args: argv,
452
+ allowPositionals: true,
453
+ options: {
454
+ ...COMMON_OPTIONS,
455
+ json: { type: 'boolean', default: false },
456
+ gate: { type: 'boolean', default: false },
457
+ executor: { type: 'string' },
458
+ model: { type: 'string' },
459
+ timeout: { type: 'string' },
460
+ },
461
+ });
462
+ const target = positionals[0] ?? null;
463
+ const executorName = values.executor ?? 'claude';
464
+ const model = values.model ?? 'sonnet';
465
+ const timeoutRaw = values.timeout;
466
+ const timeoutSec = timeoutRaw != null ? Number(timeoutRaw) : 8;
467
+ const timeoutMs = Math.max(1000, Math.floor((Number.isFinite(timeoutSec) ? timeoutSec : 8) * 1000));
468
+ const cwd = process.cwd();
469
+ const { runDoctor } = await import('../doctor/index.js');
470
+ const { renderDoctorReportText, renderDoctorReportJson } = await import('../doctor/renderer.js');
471
+ let report;
472
+ try {
473
+ report = await runDoctor({
474
+ target,
475
+ cwd,
476
+ executorName,
477
+ model,
478
+ timeoutMs,
479
+ lang,
480
+ });
481
+ }
482
+ catch (err) {
483
+ const msg = err instanceof Error ? err.message : String(err);
484
+ console.error(tCli('cli.doctor.no_skill_found', lang, { path: target ?? cwd }));
485
+ console.error(`(${msg})`);
486
+ process.exit(1);
487
+ }
488
+ if (report.skills.length === 0) {
489
+ console.error(tCli('cli.doctor.no_skill_found', lang, { path: target ?? cwd }));
490
+ process.exit(1);
491
+ }
492
+ const isJson = values.json;
493
+ const isGate = values.gate;
494
+ if (isJson) {
495
+ console.log(renderDoctorReportJson(report));
496
+ }
497
+ else if (isGate) {
498
+ // gate 模式: 静默 stdout, fail 时简短 stderr 摘要(供 CI 抓 exit code)
499
+ if (report.failed) {
500
+ const summary = lang === 'zh'
501
+ ? `doctor failed: ${report.totals.fail} 个 skill 未通过 (${report.totals.warn} warn / ${report.totals.pass} pass)`
502
+ : `doctor failed: ${report.totals.fail} skills did not pass (${report.totals.warn} warn / ${report.totals.pass} pass)`;
503
+ console.error(summary);
504
+ }
505
+ }
506
+ else {
507
+ renderDoctorReportText(report, lang);
508
+ }
509
+ process.exit(report.failed ? 1 : 0);
510
+ }
453
511
  async function handleAnalyze(argv) {
454
512
  const lang = langFromArgv(argv);
455
- const { values, positionals } = parseArgs({
513
+ const { values: rawValues, positionals } = parseArgsStrictOrExit({
456
514
  args: argv,
457
515
  allowPositionals: true,
458
516
  options: {
@@ -465,6 +523,8 @@ async function handleAnalyze(argv) {
465
523
  'output-dir': { type: 'string' },
466
524
  },
467
525
  });
526
+ // 该 handler options 全是 string-typed (无 boolean), 收紧 cast 让 caller 直接 use values.xxx 当 string 用。
527
+ const values = rawValues;
468
528
  const dir = positionals[0];
469
529
  if (!dir) {
470
530
  console.error(tCli('cli.help.analyze_usage', lang));
@@ -520,7 +580,14 @@ async function handleAnalyze(argv) {
520
580
  }
521
581
  async function handleInit(argv) {
522
582
  const lang = langFromArgv(argv);
523
- const targetDir = resolve(argv[0] || '.');
583
+ // 走 helper 让未知 option fail-fast (e.g. `omk bench init --bogus`),
584
+ // 否则 argv[0] 直接当目录名, --bogus / --lang 都会被当成 dir 写文件。
585
+ const { positionals } = parseArgsStrictOrExit({
586
+ args: argv,
587
+ allowPositionals: true,
588
+ options: { ...COMMON_OPTIONS },
589
+ });
590
+ const targetDir = resolve(positionals[0] || '.');
524
591
  const { writeFileSync, mkdirSync } = await import('node:fs');
525
592
  mkdirSync(join(targetDir, 'skills'), { recursive: true });
526
593
  writeFileSync(join(targetDir, 'eval-samples.json'), INIT_SAMPLES);
@@ -542,7 +609,7 @@ async function handleGenSamples(argv) {
542
609
  console.log(tCli('cli.help.main', lang).trim());
543
610
  process.exit(0);
544
611
  }
545
- const { values } = parseArgs({
612
+ const { values } = parseArgsStrictOrExit({
546
613
  args: argv,
547
614
  options: {
548
615
  ...COMMON_OPTIONS,
@@ -551,7 +618,6 @@ async function handleGenSamples(argv) {
551
618
  model: { type: 'string', default: 'sonnet' },
552
619
  'skill-dir': { type: 'string', default: 'skills' },
553
620
  },
554
- strict: false,
555
621
  allowPositionals: true,
556
622
  });
557
623
  const { generateSamples } = await import('../authoring/generator.js');
@@ -656,7 +722,7 @@ async function handleGenSamples(argv) {
656
722
  // ---------------------------------------------------------------------------
657
723
  async function handleEvolve(argv) {
658
724
  const lang = langFromArgv(argv);
659
- const { values } = parseArgs({
725
+ const { values, positionals } = parseArgsStrictOrExit({
660
726
  args: argv,
661
727
  options: {
662
728
  ...COMMON_OPTIONS,
@@ -664,17 +730,18 @@ async function handleEvolve(argv) {
664
730
  target: { type: 'string' },
665
731
  samples: { type: 'string', default: 'eval-samples.json' },
666
732
  model: { type: 'string', default: 'sonnet' },
667
- 'judge-model': { type: 'string', default: 'haiku' },
733
+ 'judge-models': { type: 'string', default: 'claude:haiku' },
668
734
  'improve-model': { type: 'string', default: 'sonnet' },
669
735
  concurrency: { type: 'string', default: '1' },
670
736
  timeout: { type: 'string', default: '120' },
671
737
  executor: { type: 'string', default: 'claude' },
672
- 'skip-preflight': { type: 'boolean', default: false },
738
+ 'skip-connectivity': { type: 'boolean', default: false },
673
739
  },
674
- strict: false,
675
740
  allowPositionals: true,
676
741
  });
677
- const skillPath = argv.find((a) => !a.startsWith('-'));
742
+ // skill path 走 parseArgs 的 positionals (避免 raw argv.find 把 flag value
743
+ // 当成 path 误识别 — 例如 `evolve --judge-models openai-api:gpt-4o foo.md`)。
744
+ const skillPath = positionals[0];
678
745
  if (!skillPath) {
679
746
  console.error(tCli('cli.evolve.specify_skill_path', lang));
680
747
  process.exit(1);
@@ -687,6 +754,12 @@ async function handleEvolve(argv) {
687
754
  samplesFile = 'eval-samples.yml';
688
755
  }
689
756
  const { evolveSkill } = await import('../authoring/evolver.js');
757
+ const { parseJudgeModelsArgOrExit } = await import('./parse-run-config.js');
758
+ const evolveJudges = parseJudgeModelsArgOrExit(values['judge-models']);
759
+ if (evolveJudges.length > 1) {
760
+ console.error(tCli('cli.common.judge_models_single_only', lang, { cmd: 'evolve' }));
761
+ process.exit(2);
762
+ }
690
763
  process.stderr.write(tCli('cli.evolve.section_header', lang, { path: skillPath }));
691
764
  try {
692
765
  const result = await evolveSkill({
@@ -695,12 +768,12 @@ async function handleEvolve(argv) {
695
768
  rounds: Math.max(1, Number(values.rounds) || 5),
696
769
  target: values.target ? Number(values.target) : null,
697
770
  model: values.model,
698
- judgeModel: values['judge-model'],
771
+ judgeModels: evolveJudges,
699
772
  improveModel: values['improve-model'],
700
773
  executorName: values.executor,
701
774
  concurrency: Math.max(1, Number(values.concurrency) || 1),
702
775
  timeoutMs: Math.max(1, Number(values.timeout) || 120) * 1000,
703
- skipPreflight: values['skip-preflight'],
776
+ skipConnectivity: values['skip-connectivity'],
704
777
  onProgress: makeOnProgress(lang),
705
778
  onRoundProgress({ round, totalRounds: _totalRounds, phase, score, delta, accepted, costUSD, costReported, error }) {
706
779
  // costReported=false 时显示「—」而不是 $0.0000(executor 不报 cost,如 codex)。
@@ -768,6 +841,11 @@ async function handleGate(argv) {
768
841
  });
769
842
  const { runEvaluation } = await import('../eval-workflows/run-evaluation.js');
770
843
  config.onProgress = makeOnProgress(lang);
844
+ // 注入 lang + skip-connectivity warning(若 flag set);doctor 由 evaluation 强制调, 无 skip 选项。
845
+ config.lang = lang;
846
+ if (values['skip-connectivity']) {
847
+ process.stderr.write(tCli('cli.run.skip_connectivity_warning', lang) + '\n');
848
+ }
771
849
  try {
772
850
  const { report: document } = (await runEvaluation(config));
773
851
  if (document.dryRun) {
@@ -830,7 +908,7 @@ async function handleDiff(argv) {
830
908
  console.error(tCli('cli.help.diff_usage', lang));
831
909
  process.exit(positional.length === 0 ? 1 : 0);
832
910
  }
833
- const { values } = parseArgs({
911
+ const { values } = parseArgsStrictOrExit({
834
912
  args: flagArgs,
835
913
  options: {
836
914
  ...COMMON_OPTIONS,
@@ -839,7 +917,6 @@ async function handleDiff(argv) {
839
917
  variant: { type: 'string' },
840
918
  top: { type: 'string' },
841
919
  },
842
- strict: false,
843
920
  });
844
921
  const { createFileStore } = await import('../server/report-store.js');
845
922
  const store = createFileStore(resolve(DEFAULT_REPORTS_DIR));
@@ -1001,14 +1078,13 @@ async function handleGold(argv) {
1001
1078
  process.exit(sub ? 0 : 1);
1002
1079
  }
1003
1080
  if (sub === 'init') {
1004
- const { values } = parseArgs({
1081
+ const { values } = parseArgsStrictOrExit({
1005
1082
  args: rest,
1006
1083
  options: {
1007
1084
  ...COMMON_OPTIONS,
1008
1085
  out: { type: 'string', default: './gold-dataset' },
1009
1086
  annotator: { type: 'string' },
1010
1087
  },
1011
- strict: false,
1012
1088
  });
1013
1089
  const { initGoldDataset } = await import('../grading/gold-cli.js');
1014
1090
  try {
@@ -1029,7 +1105,14 @@ async function handleGold(argv) {
1029
1105
  return;
1030
1106
  }
1031
1107
  if (sub === 'validate') {
1032
- const dir = rest[0];
1108
+ // 走 helper 让 `omk bench gold validate <dir> --bogus` 走 unknown option 路径,
1109
+ // 而不是直接执行 validate 后再报 dataset 错。
1110
+ const { positionals } = parseArgsStrictOrExit({
1111
+ args: rest,
1112
+ allowPositionals: true,
1113
+ options: { ...COMMON_OPTIONS },
1114
+ });
1115
+ const dir = positionals[0];
1033
1116
  if (!dir) {
1034
1117
  console.error(tCli('cli.common.usage_gold_validate', lang));
1035
1118
  process.exit(1);
@@ -1051,7 +1134,7 @@ async function handleGold(argv) {
1051
1134
  console.error('Usage: omk bench gold compare <reportId> --gold-dir <dir>');
1052
1135
  process.exit(1);
1053
1136
  }
1054
- const { values } = parseArgs({
1137
+ const { values } = parseArgsStrictOrExit({
1055
1138
  args: rest.slice(1),
1056
1139
  options: {
1057
1140
  ...COMMON_OPTIONS,
@@ -1061,7 +1144,6 @@ async function handleGold(argv) {
1061
1144
  'bootstrap-samples': { type: 'string', default: '1000' },
1062
1145
  seed: { type: 'string' },
1063
1146
  },
1064
- strict: false,
1065
1147
  });
1066
1148
  const goldDir = values['gold-dir'];
1067
1149
  if (!goldDir) {
@@ -1101,7 +1183,7 @@ async function handleGold(argv) {
1101
1183
  process.exit(1);
1102
1184
  }
1103
1185
  // ---------------------------------------------------------------------------
1104
- // handleDebiasValidate — measure length-debias prompt sensitivity (Phase 3a)
1186
+ // handleDebiasValidate — measure length-debias prompt sensitivity
1105
1187
  // ---------------------------------------------------------------------------
1106
1188
  async function handleDebiasValidate(argv) {
1107
1189
  const lang = langFromArgv(argv);
@@ -1120,20 +1202,28 @@ async function handleDebiasValidate(argv) {
1120
1202
  console.error('Usage: omk bench debias-validate length <reportId>');
1121
1203
  process.exit(1);
1122
1204
  }
1123
- const { values } = parseArgs({
1205
+ const { values } = parseArgsStrictOrExit({
1124
1206
  args: rest.slice(1),
1125
1207
  options: {
1126
1208
  ...COMMON_OPTIONS,
1127
1209
  'reports-dir': { type: 'string', default: DEFAULT_REPORTS_DIR },
1128
1210
  samples: { type: 'string' },
1129
1211
  variant: { type: 'string' },
1130
- 'judge-executor': { type: 'string', default: 'claude' },
1131
- 'judge-model': { type: 'string' },
1212
+ 'judge-models': { type: 'string' },
1132
1213
  'bootstrap-samples': { type: 'string', default: '1000' },
1133
1214
  seed: { type: 'string' },
1134
1215
  },
1135
- strict: false,
1136
1216
  });
1217
+ // Parse --judge-models 在 load report 之前 fail-fast。重复 entry / 缺 executor /
1218
+ // 空串等参数错误应立即给 friendly error: + exit 2,不要等到 store IO 完成才暴露。
1219
+ const { parseJudgeModelsArgOrExit: parseJudgesA } = await import('./parse-run-config.js');
1220
+ const cliJudgeModelsA = values['judge-models'] !== undefined
1221
+ ? parseJudgesA(values['judge-models'])
1222
+ : undefined;
1223
+ if (cliJudgeModelsA && cliJudgeModelsA.length > 1) {
1224
+ console.error(tCli('cli.common.judge_models_single_only', lang, { cmd: 'debias-validate' }));
1225
+ process.exit(2);
1226
+ }
1137
1227
  const { createFileStore } = await import('../server/report-store.js');
1138
1228
  const store = createFileStore(resolve(values['reports-dir']));
1139
1229
  const report = requireEvaluationReport(await store.get(reportId), reportId, lang);
@@ -1146,15 +1236,18 @@ async function handleDebiasValidate(argv) {
1146
1236
  }
1147
1237
  const { loadSamples } = await import('../inputs/load-samples.js');
1148
1238
  const { samples } = loadSamples(samplesPath);
1149
- const judgeModel = values['judge-model']
1150
- ?? report.meta?.judgeModel;
1151
- if (!judgeModel) {
1239
+ const debiasJudges = cliJudgeModelsA
1240
+ ?? (report.meta?.judgeModels?.[0]
1241
+ ? [{ executor: report.meta.judgeModels[0].executor, model: report.meta.judgeModels[0].model }]
1242
+ : []);
1243
+ if (debiasJudges.length === 0) {
1152
1244
  console.error(tCli('cli.common.no_judge_model', lang));
1153
1245
  process.exit(1);
1154
1246
  }
1155
1247
  process.stderr.write(tCli('cli.debias.warn_cost_doubles', lang));
1156
1248
  const { createExecutor } = await import('../executors/index.js');
1157
- const judgeExecutor = createExecutor(values['judge-executor']);
1249
+ const judgeExecutor = createExecutor(debiasJudges[0].executor);
1250
+ const judgeModel = debiasJudges[0].model;
1158
1251
  const { validateLengthDebias, formatDebiasValidate } = await import('../grading/debias-validate.js');
1159
1252
  const seedVal = values.seed != null ? Number(values.seed) : undefined;
1160
1253
  const bsRaw = Number(values['bootstrap-samples']) || 1000;
@@ -1182,14 +1275,13 @@ async function handleSaturation(argv) {
1182
1275
  console.log(tCli('cli.help.saturation', lang));
1183
1276
  process.exit(reportId ? 0 : 1);
1184
1277
  }
1185
- const { values } = parseArgs({
1278
+ const { values } = parseArgsStrictOrExit({
1186
1279
  args: argv.slice(1),
1187
1280
  options: {
1188
1281
  ...COMMON_OPTIONS,
1189
1282
  'reports-dir': { type: 'string', default: DEFAULT_REPORTS_DIR },
1190
1283
  variant: { type: 'string' },
1191
1284
  },
1192
- strict: false,
1193
1285
  });
1194
1286
  const { createFileStore } = await import('../server/report-store.js');
1195
1287
  const store = createFileStore(resolve(values['reports-dir']));
@@ -1247,7 +1339,7 @@ async function handleVerdict(argv) {
1247
1339
  console.log(tCli('cli.help.verdict', lang));
1248
1340
  process.exit(reportId ? 0 : 1);
1249
1341
  }
1250
- const { values } = parseArgs({
1342
+ const { values } = parseArgsStrictOrExit({
1251
1343
  args: argv.slice(1),
1252
1344
  options: {
1253
1345
  ...COMMON_OPTIONS,
@@ -1256,7 +1348,6 @@ async function handleVerdict(argv) {
1256
1348
  'trivial-diff': { type: 'string' },
1257
1349
  verbose: { type: 'boolean', default: false },
1258
1350
  },
1259
- strict: false,
1260
1351
  });
1261
1352
  const { createFileStore } = await import('../server/report-store.js');
1262
1353
  const store = createFileStore(resolve(values['reports-dir']));
@@ -1292,7 +1383,7 @@ async function handleDiagnose(argv) {
1292
1383
  console.log(tCli('cli.help.diagnose', lang));
1293
1384
  process.exit(reportId ? 0 : 1);
1294
1385
  }
1295
- const { values } = parseArgs({
1386
+ const { values } = parseArgsStrictOrExit({
1296
1387
  args: argv.slice(1),
1297
1388
  options: {
1298
1389
  ...COMMON_OPTIONS,
@@ -1305,7 +1396,6 @@ async function handleDiagnose(argv) {
1305
1396
  'latency-k': { type: 'string' },
1306
1397
  flat: { type: 'string' },
1307
1398
  },
1308
- strict: false,
1309
1399
  });
1310
1400
  const { createFileStore } = await import('../server/report-store.js');
1311
1401
  const store = createFileStore(resolve(values['reports-dir']));
@@ -1363,29 +1453,40 @@ async function handleFailures(argv) {
1363
1453
  console.log(tCli('cli.help.failures', lang));
1364
1454
  process.exit(reportId ? 0 : 1);
1365
1455
  }
1366
- const { values } = parseArgs({
1456
+ const { values } = parseArgsStrictOrExit({
1367
1457
  args: argv.slice(1),
1368
1458
  options: {
1369
1459
  ...COMMON_OPTIONS,
1370
1460
  'reports-dir': { type: 'string', default: DEFAULT_REPORTS_DIR },
1371
- 'judge-executor': { type: 'string', default: 'claude' },
1372
- 'judge-model': { type: 'string' },
1461
+ 'judge-models': { type: 'string' },
1373
1462
  'max-clusters': { type: 'string', default: '5' },
1374
1463
  threshold: { type: 'string', default: '3' },
1375
1464
  'max-feed': { type: 'string', default: '50' },
1376
1465
  },
1377
- strict: false,
1378
1466
  });
1467
+ // Parse --judge-models 在 load report 之前 fail-fast(同 debias-validate)。
1468
+ const { parseJudgeModelsArgOrExit: parseJudgesB } = await import('./parse-run-config.js');
1469
+ const cliJudgeModelsB = values['judge-models'] !== undefined
1470
+ ? parseJudgesB(values['judge-models'])
1471
+ : undefined;
1472
+ if (cliJudgeModelsB && cliJudgeModelsB.length > 1) {
1473
+ console.error(tCli('cli.common.judge_models_single_only', lang, { cmd: 'failures' }));
1474
+ process.exit(2);
1475
+ }
1379
1476
  const { createFileStore } = await import('../server/report-store.js');
1380
1477
  const store = createFileStore(resolve(values['reports-dir']));
1381
1478
  const report = requireEvaluationReport(await store.get(reportId), reportId, lang);
1382
- const judgeModel = values['judge-model'] ?? report.meta?.judgeModel;
1383
- if (!judgeModel) {
1479
+ const failuresJudges = cliJudgeModelsB
1480
+ ?? (report.meta?.judgeModels?.[0]
1481
+ ? [{ executor: report.meta.judgeModels[0].executor, model: report.meta.judgeModels[0].model }]
1482
+ : []);
1483
+ if (failuresJudges.length === 0) {
1384
1484
  console.error(tCli('cli.common.no_judge_model', lang));
1385
1485
  process.exit(1);
1386
1486
  }
1387
1487
  const { createExecutor } = await import('../executors/index.js');
1388
- const executor = createExecutor(values['judge-executor']);
1488
+ const executor = createExecutor(failuresJudges[0].executor);
1489
+ const judgeModel = failuresJudges[0].model;
1389
1490
  const { clusterFailures, formatFailureClusterReport } = await import('../analysis/failure-clusterer.js');
1390
1491
  const out = await clusterFailures({
1391
1492
  report,