oh-my-knowledge 0.19.0 → 0.20.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (96) hide show
  1. package/README.md +205 -8
  2. package/README.zh.md +200 -7
  3. package/dist/src/analysis/failure-clusterer.d.ts +96 -0
  4. package/dist/src/analysis/failure-clusterer.d.ts.map +1 -0
  5. package/dist/src/analysis/failure-clusterer.js +298 -0
  6. package/dist/src/analysis/failure-clusterer.js.map +1 -0
  7. package/dist/src/analysis/sample-diagnostics.d.ts +78 -0
  8. package/dist/src/analysis/sample-diagnostics.d.ts.map +1 -0
  9. package/dist/src/analysis/sample-diagnostics.js +259 -0
  10. package/dist/src/analysis/sample-diagnostics.js.map +1 -0
  11. package/dist/src/analysis/saturation.d.ts +85 -0
  12. package/dist/src/analysis/saturation.d.ts.map +1 -0
  13. package/dist/src/analysis/saturation.js +174 -0
  14. package/dist/src/analysis/saturation.js.map +1 -0
  15. package/dist/src/cli.js +751 -5
  16. package/dist/src/cli.js.map +1 -1
  17. package/dist/src/eval-core/bootstrap.d.ts +72 -0
  18. package/dist/src/eval-core/bootstrap.d.ts.map +1 -0
  19. package/dist/src/eval-core/bootstrap.js +174 -0
  20. package/dist/src/eval-core/bootstrap.js.map +1 -0
  21. package/dist/src/eval-core/evaluation-execution.d.ts +15 -1
  22. package/dist/src/eval-core/evaluation-execution.d.ts.map +1 -1
  23. package/dist/src/eval-core/evaluation-execution.js +37 -3
  24. package/dist/src/eval-core/evaluation-execution.js.map +1 -1
  25. package/dist/src/eval-core/evaluation-job.d.ts +8 -2
  26. package/dist/src/eval-core/evaluation-job.d.ts.map +1 -1
  27. package/dist/src/eval-core/evaluation-job.js +7 -1
  28. package/dist/src/eval-core/evaluation-job.js.map +1 -1
  29. package/dist/src/eval-core/evaluation-reporting.d.ts.map +1 -1
  30. package/dist/src/eval-core/evaluation-reporting.js +90 -0
  31. package/dist/src/eval-core/evaluation-reporting.js.map +1 -1
  32. package/dist/src/eval-core/schema.d.ts.map +1 -1
  33. package/dist/src/eval-core/schema.js +69 -0
  34. package/dist/src/eval-core/schema.js.map +1 -1
  35. package/dist/src/eval-core/verdict.d.ts +74 -0
  36. package/dist/src/eval-core/verdict.d.ts.map +1 -0
  37. package/dist/src/eval-core/verdict.js +283 -0
  38. package/dist/src/eval-core/verdict.js.map +1 -0
  39. package/dist/src/eval-workflows/each-evaluation-workflow.d.ts +10 -1
  40. package/dist/src/eval-workflows/each-evaluation-workflow.d.ts.map +1 -1
  41. package/dist/src/eval-workflows/each-evaluation-workflow.js +4 -1
  42. package/dist/src/eval-workflows/each-evaluation-workflow.js.map +1 -1
  43. package/dist/src/eval-workflows/evaluation-pipeline.d.ts +13 -1
  44. package/dist/src/eval-workflows/evaluation-pipeline.d.ts.map +1 -1
  45. package/dist/src/eval-workflows/evaluation-pipeline.js +41 -3
  46. package/dist/src/eval-workflows/evaluation-pipeline.js.map +1 -1
  47. package/dist/src/eval-workflows/run-evaluation.d.ts +17 -2
  48. package/dist/src/eval-workflows/run-evaluation.d.ts.map +1 -1
  49. package/dist/src/eval-workflows/run-evaluation.js +97 -5
  50. package/dist/src/eval-workflows/run-evaluation.js.map +1 -1
  51. package/dist/src/grading/assertions.d.ts +16 -0
  52. package/dist/src/grading/assertions.d.ts.map +1 -1
  53. package/dist/src/grading/assertions.js +385 -111
  54. package/dist/src/grading/assertions.js.map +1 -1
  55. package/dist/src/grading/debias-validate.d.ts +84 -0
  56. package/dist/src/grading/debias-validate.d.ts.map +1 -0
  57. package/dist/src/grading/debias-validate.js +173 -0
  58. package/dist/src/grading/debias-validate.js.map +1 -0
  59. package/dist/src/grading/gold-cli.d.ts +88 -0
  60. package/dist/src/grading/gold-cli.d.ts.map +1 -0
  61. package/dist/src/grading/gold-cli.js +251 -0
  62. package/dist/src/grading/gold-cli.js.map +1 -0
  63. package/dist/src/grading/gold-dataset.d.ts +73 -0
  64. package/dist/src/grading/gold-dataset.d.ts.map +1 -0
  65. package/dist/src/grading/gold-dataset.js +161 -0
  66. package/dist/src/grading/gold-dataset.js.map +1 -0
  67. package/dist/src/grading/human-gold.d.ts +102 -0
  68. package/dist/src/grading/human-gold.d.ts.map +1 -0
  69. package/dist/src/grading/human-gold.js +188 -0
  70. package/dist/src/grading/human-gold.js.map +1 -0
  71. package/dist/src/grading/index.d.ts +27 -2
  72. package/dist/src/grading/index.d.ts.map +1 -1
  73. package/dist/src/grading/index.js +36 -18
  74. package/dist/src/grading/index.js.map +1 -1
  75. package/dist/src/grading/judge.d.ts +65 -2
  76. package/dist/src/grading/judge.d.ts.map +1 -1
  77. package/dist/src/grading/judge.js +280 -23
  78. package/dist/src/grading/judge.js.map +1 -1
  79. package/dist/src/inputs/eval-config.js +19 -0
  80. package/dist/src/inputs/eval-config.js.map +1 -1
  81. package/dist/src/renderer/html-renderer.d.ts.map +1 -1
  82. package/dist/src/renderer/html-renderer.js +16 -4
  83. package/dist/src/renderer/html-renderer.js.map +1 -1
  84. package/dist/src/renderer/layout.d.ts.map +1 -1
  85. package/dist/src/renderer/layout.js +32 -4
  86. package/dist/src/renderer/layout.js.map +1 -1
  87. package/dist/src/renderer/summary.d.ts +28 -1
  88. package/dist/src/renderer/summary.d.ts.map +1 -1
  89. package/dist/src/renderer/summary.js +321 -7
  90. package/dist/src/renderer/summary.js.map +1 -1
  91. package/dist/src/renderer/table.d.ts.map +1 -1
  92. package/dist/src/renderer/table.js +63 -2
  93. package/dist/src/renderer/table.js.map +1 -1
  94. package/dist/src/types.d.ts +241 -0
  95. package/dist/src/types.d.ts.map +1 -1
  96. package/package.json +15 -4
package/dist/src/cli.js CHANGED
@@ -167,6 +167,7 @@ function parseRunConfig(argv, extraOptions = {}) {
167
167
  resume,
168
168
  blind,
169
169
  layeredStats,
170
+ budget: evalConfig?.budget,
170
171
  },
171
172
  };
172
173
  }
@@ -214,6 +215,22 @@ Options for "bench run":
214
215
  --concurrency <n> Number of parallel tasks (default: 1)
215
216
  --timeout <seconds> Executor timeout per task in seconds (default: 120)
216
217
  --repeat <n> Run evaluation N times for variance analysis (default: 1)
218
+ --judge-repeat <n> Call LLM judge N times per (sample × dimension) for self-
219
+ consistency (default: 1). High stddev across runs = the
220
+ judge is unstable on this rubric and the score is noisy.
221
+ --judge-models <list> Multi-judge ensemble. Comma-separated executor:model pairs,
222
+ e.g. claude:opus,openai:gpt-4o,gemini:pro. Each judge scores
223
+ every (sample × dimension); report includes per-judge break-
224
+ down + Pearson/MAD inter-judge agreement. Refutes "Claude
225
+ judge Claude same-modality bias" critique. Combines with
226
+ --judge-repeat. Cost ~ N_judges × N_repeat × N_samples.
227
+ --bootstrap Compute bootstrap confidence intervals (distribution-free,
228
+ preferred over t-interval for ordinal LLM scores). Adds
229
+ per-variant CI on the mean + pairwise CI on treatment-vs-
230
+ control difference (significant=0 outside CI). Reports both
231
+ t-interval and bootstrap so old tooling still works.
232
+ --bootstrap-samples <n> Number of bootstrap resamples (default 1000). N>10000
233
+ triggers a stderr warning about runtime cost.
217
234
  --retry <n> Retry failed tasks up to N times with exponential backoff (default: 0)
218
235
  --resume <report-id> Resume from a previous report, skipping completed tasks
219
236
  --executor <name> Executor: claude, openai, gemini, anthropic-api, openai-api,
@@ -360,8 +377,26 @@ async function main() {
360
377
  case 'diff':
361
378
  await handleDiff(rest);
362
379
  break;
380
+ case 'gold':
381
+ await handleGold(rest);
382
+ break;
383
+ case 'debias-validate':
384
+ await handleDebiasValidate(rest);
385
+ break;
386
+ case 'saturation':
387
+ await handleSaturation(rest);
388
+ break;
389
+ case 'verdict':
390
+ await handleVerdict(rest);
391
+ break;
392
+ case 'diagnose':
393
+ await handleDiagnose(rest);
394
+ break;
395
+ case 'failures':
396
+ await handleFailures(rest);
397
+ break;
363
398
  default:
364
- console.error(`Unknown command: bench ${command}. Use "run", "report", "ci", "init", "gen-samples", or "evolve".`);
399
+ console.error(`Unknown command: bench ${command}. Use "run", "report", "ci", "init", "gen-samples", "evolve", "diff", "gold", "debias-validate", "saturation", "verdict", "diagnose", or "failures".`);
365
400
  process.exit(1);
366
401
  }
367
402
  }
@@ -416,6 +451,15 @@ async function handleRun(argv) {
416
451
  const { values, config } = parseRunConfig(argv, {
417
452
  blind: { type: 'boolean', default: false },
418
453
  repeat: { type: 'string', default: '1' },
454
+ 'judge-repeat': { type: 'string', default: '1' },
455
+ 'judge-models': { type: 'string' },
456
+ bootstrap: { type: 'boolean', default: false },
457
+ 'bootstrap-samples': { type: 'string', default: '1000' },
458
+ 'gold-dir': { type: 'string' },
459
+ 'no-debias-length': { type: 'boolean', default: false },
460
+ 'budget-usd': { type: 'string' },
461
+ 'budget-per-sample-usd': { type: 'string' },
462
+ 'budget-per-sample-ms': { type: 'string' },
419
463
  });
420
464
  const { runEvaluation, runMultiple, runEachEvaluation } = await import('./eval-workflows/run-evaluation.js');
421
465
  config.blind = values.blind;
@@ -428,6 +472,71 @@ async function handleRun(argv) {
428
472
  process.stderr.write(`⚠ --repeat "${repeatRaw}" 无效(期望 ≥ 1 的整数),已按 1 次评测执行\n`);
429
473
  }
430
474
  const repeatCount = Math.max(1, Math.floor(parsedRepeat) || 1);
475
+ // --judge-repeat 同样的诚实校验:非 ≥1 整数时钳到 1
476
+ const judgeRepeatRaw = values['judge-repeat'];
477
+ const parsedJudgeRepeat = judgeRepeatRaw !== undefined ? Number(judgeRepeatRaw) : 1;
478
+ if (judgeRepeatRaw !== undefined && (!Number.isFinite(parsedJudgeRepeat) || parsedJudgeRepeat < 1)) {
479
+ process.stderr.write(`⚠ --judge-repeat "${judgeRepeatRaw}" 无效(期望 ≥ 1 的整数),已按 1 次 judge 执行\n`);
480
+ }
481
+ const judgeRepeatCount = Math.max(1, Math.floor(parsedJudgeRepeat) || 1);
482
+ if (judgeRepeatCount > 1)
483
+ config.judgeRepeat = judgeRepeatCount;
484
+ // --judge-models executor:model,executor:model,... -> JudgeConfig[]
485
+ // 至少 2 个才进 ensemble 模式,1 个等同于 --judge-model
486
+ const judgeModelsRaw = values['judge-models'];
487
+ if (judgeModelsRaw) {
488
+ const parts = judgeModelsRaw.split(',').map((s) => s.trim()).filter(Boolean);
489
+ const judges = parts.map((p) => {
490
+ const [executor, ...modelParts] = p.split(':');
491
+ const model = modelParts.join(':');
492
+ if (!executor || !model) {
493
+ throw new Error(`--judge-models 格式错误: "${p}",应为 "executor:model" (如 claude:opus)`);
494
+ }
495
+ return { executor, model };
496
+ });
497
+ if (judges.length >= 2) {
498
+ config.judgeModels = judges;
499
+ }
500
+ else if (judges.length === 1) {
501
+ // 单 judge 不走 ensemble,但允许这样写,等同于 --judge-model + --executor
502
+ process.stderr.write(`ℹ --judge-models 只指定 1 个 judge (${judges[0].executor}:${judges[0].model}),不触发 ensemble。如需 ensemble 至少给 2 个。\n`);
503
+ }
504
+ }
505
+ // --budget-usd / --budget-per-sample-usd / --budget-per-sample-ms:
506
+ // v0.22 hard budget caps. CLI flags override config-file values. When the
507
+ // total-USD cap is exceeded mid-run, remaining tasks are skipped and a
508
+ // partial report is persisted with meta.budgetExhausted=true.
509
+ const budgetUSD = values['budget-usd'] != null ? Number(values['budget-usd']) : undefined;
510
+ const budgetPerSampleUSD = values['budget-per-sample-usd'] != null ? Number(values['budget-per-sample-usd']) : undefined;
511
+ const budgetPerSampleMs = values['budget-per-sample-ms'] != null ? Number(values['budget-per-sample-ms']) : undefined;
512
+ if (budgetUSD !== undefined || budgetPerSampleUSD !== undefined || budgetPerSampleMs !== undefined) {
513
+ config.budget = {
514
+ ...(budgetUSD !== undefined && Number.isFinite(budgetUSD) && budgetUSD >= 0 ? { totalUSD: budgetUSD } : {}),
515
+ ...(budgetPerSampleUSD !== undefined && Number.isFinite(budgetPerSampleUSD) && budgetPerSampleUSD >= 0 ? { perSampleUSD: budgetPerSampleUSD } : {}),
516
+ ...(budgetPerSampleMs !== undefined && Number.isFinite(budgetPerSampleMs) && budgetPerSampleMs >= 0 ? { perSampleMs: budgetPerSampleMs } : {}),
517
+ };
518
+ }
519
+ // --no-debias-length: opt out of v0.21 Phase 3a length-controlled prompt.
520
+ // Default behavior is debias-on (judge prompt v3-cot-length); flag flips it
521
+ // off so historical reports (judgePromptHash from v2-cot era) can be reproduced.
522
+ if (values['no-debias-length']) {
523
+ config.lengthDebias = false;
524
+ process.stderr.write('ℹ --no-debias-length 已生效:judge prompt 退回 v2-cot,与 < v0.21 报告 hash 一致。\n');
525
+ }
526
+ // --bootstrap / --bootstrap-samples
527
+ if (values.bootstrap) {
528
+ config.bootstrap = true;
529
+ const bsRaw = values['bootstrap-samples'];
530
+ const parsedBs = bsRaw !== undefined ? Number(bsRaw) : 1000;
531
+ if (bsRaw !== undefined && (!Number.isFinite(parsedBs) || parsedBs < 100)) {
532
+ process.stderr.write(`⚠ --bootstrap-samples "${bsRaw}" 无效(期望 ≥ 100 的整数),已按 1000 执行\n`);
533
+ }
534
+ const bsCount = Math.max(100, Math.floor(parsedBs) || 1000);
535
+ if (bsCount > 10000) {
536
+ process.stderr.write(`⚠ --bootstrap-samples ${bsCount} 较大,可能耗时数秒。1000 是业内标准,通常已够用。\n`);
537
+ }
538
+ config.bootstrapSamples = bsCount;
539
+ }
431
540
  try {
432
541
  // --each mode: evaluate each skill independently
433
542
  if (values.each) {
@@ -482,6 +591,28 @@ async function handleRun(argv) {
482
591
  report = result.report;
483
592
  filePath = result.filePath;
484
593
  }
594
+ // --gold-dir: compute α/κ/Pearson against gold annotations and re-persist.
595
+ const goldDir = values['gold-dir'];
596
+ if (goldDir && filePath) {
597
+ const { attachGoldAgreementToReport, formatGoldCompare } = await import('./grading/gold-cli.js');
598
+ const out = attachGoldAgreementToReport({
599
+ report,
600
+ goldDir,
601
+ outputDir: config.outputDir,
602
+ samples: config.bootstrapSamples,
603
+ });
604
+ if (out.result && out.gold) {
605
+ process.stderr.write(formatGoldCompare(out.result, out.gold));
606
+ if (out.result.contaminationWarning) {
607
+ process.stderr.write(`\n⚠ ${out.result.contaminationWarning}\n`);
608
+ }
609
+ }
610
+ else {
611
+ process.stderr.write(`\n⚠ gold dataset 加载失败 (${goldDir}):\n`);
612
+ for (const m of out.loadIssues)
613
+ process.stderr.write(` - ${m}\n`);
614
+ }
615
+ }
485
616
  console.log(JSON.stringify(report, null, 2));
486
617
  if (filePath) {
487
618
  process.stderr.write('\n✅ 评测完成\n');
@@ -942,13 +1073,57 @@ async function handleCi(argv) {
942
1073
  // handleDiff
943
1074
  // ---------------------------------------------------------------------------
944
1075
  async function handleDiff(argv) {
945
- if (argv.length < 2) {
946
- console.error('Usage: omk bench diff <report-id-1> <report-id-2>');
947
- process.exit(1);
1076
+ // Flag-aware split: separate positional report IDs from flags so we can support
1077
+ // omk bench diff <id> — within-report sample-level (v0.22)
1078
+ // omk bench diff <id1> <id2> — cross-report variant-level (legacy)
1079
+ // both with optional --regressions-only / --threshold / --variant flags.
1080
+ const positional = [];
1081
+ const flagArgs = [];
1082
+ for (let i = 0; i < argv.length; i++) {
1083
+ const a = argv[i];
1084
+ if (a.startsWith('--')) {
1085
+ flagArgs.push(a);
1086
+ const next = argv[i + 1];
1087
+ if (next !== undefined && !next.startsWith('--')) {
1088
+ flagArgs.push(next);
1089
+ i++;
1090
+ }
1091
+ }
1092
+ else {
1093
+ positional.push(a);
1094
+ }
1095
+ }
1096
+ if (positional.length === 0) {
1097
+ console.error([
1098
+ 'Usage:',
1099
+ ' omk bench diff <reportId> within-report per-sample diff (v0.22)',
1100
+ ' omk bench diff <reportId1> <reportId2> cross-report variant-level diff',
1101
+ '',
1102
+ 'Options:',
1103
+ ' --regressions-only 只列 treatment < control 的样本',
1104
+ ' --threshold <num> regression 阈值 (default 0,即任一负 Δ 算回退)',
1105
+ ' --variant <name> within-report 模式下指定要钻取的 variant (default: variants[1])',
1106
+ ' --top <n> 只列差距最大的前 N 个样本',
1107
+ ].join('\n'));
1108
+ process.exit(positional.length === 0 ? 1 : 0);
948
1109
  }
1110
+ const { values } = parseArgs({
1111
+ args: flagArgs,
1112
+ options: {
1113
+ 'regressions-only': { type: 'boolean', default: false },
1114
+ threshold: { type: 'string' },
1115
+ variant: { type: 'string' },
1116
+ top: { type: 'string' },
1117
+ },
1118
+ strict: false,
1119
+ });
949
1120
  const { createFileStore } = await import('./server/report-store.js');
950
1121
  const store = createFileStore(resolve(DEFAULT_REPORTS_DIR));
951
- const [id1, id2] = argv;
1122
+ if (positional.length === 1) {
1123
+ await runSampleLevelDiff(positional[0], store, values);
1124
+ return;
1125
+ }
1126
+ const [id1, id2] = positional;
952
1127
  const r1 = await store.get(id1);
953
1128
  const r2 = await store.get(id2);
954
1129
  if (!r1) {
@@ -1005,6 +1180,577 @@ async function handleDiff(argv) {
1005
1180
  }
1006
1181
  console.log('');
1007
1182
  }
1183
+ /**
1184
+ * Within-report sample-level diff (v0.22). Compares two variants' scores on
1185
+ * each shared sample and surfaces the worst regressions / biggest wins.
1186
+ *
1187
+ * Default focus is variants[0] (control) vs variants[1] (treatment), but
1188
+ * `--variant` overrides which variant is the "treatment" side.
1189
+ */
1190
+ async function runSampleLevelDiff(reportId, store, flags) {
1191
+ const report = await store.get(reportId);
1192
+ if (!report) {
1193
+ console.error(`Report not found: ${reportId}`);
1194
+ process.exit(1);
1195
+ }
1196
+ const variants = report.meta?.variants ?? [];
1197
+ if (variants.length < 2) {
1198
+ console.error('Sample-level diff needs at least 2 variants in the report.');
1199
+ process.exit(1);
1200
+ }
1201
+ const control = variants[0];
1202
+ const treatment = flags.variant ?? variants[1];
1203
+ if (!variants.includes(treatment)) {
1204
+ console.error(`Variant "${treatment}" not in report. Available: ${variants.join(', ')}`);
1205
+ process.exit(1);
1206
+ }
1207
+ const threshold = flags.threshold != null ? Number(flags.threshold) : 0;
1208
+ const regressionsOnly = Boolean(flags['regressions-only']);
1209
+ const topN = flags.top != null ? Math.max(1, Number(flags.top) || 0) : undefined;
1210
+ const rows = [];
1211
+ for (const entry of report.results ?? []) {
1212
+ const c = entry.variants?.[control];
1213
+ const t = entry.variants?.[treatment];
1214
+ if (!c || !t)
1215
+ continue;
1216
+ const cComp = c.compositeScore ?? c.llmScore ?? 0;
1217
+ const tComp = t.compositeScore ?? t.llmScore ?? 0;
1218
+ const delta = Number((tComp - cComp).toFixed(3));
1219
+ rows.push({
1220
+ id: entry.sample_id,
1221
+ cFact: c.layeredScores?.factScore, tFact: t.layeredScores?.factScore,
1222
+ cBeh: c.layeredScores?.behaviorScore, tBeh: t.layeredScores?.behaviorScore,
1223
+ cJudge: c.layeredScores?.judgeScore, tJudge: t.layeredScores?.judgeScore,
1224
+ cComp, tComp, delta,
1225
+ });
1226
+ }
1227
+ // Sort by |delta| desc so the most impactful rows surface first.
1228
+ rows.sort((a, b) => Math.abs(b.delta) - Math.abs(a.delta));
1229
+ let filtered = rows;
1230
+ if (regressionsOnly)
1231
+ filtered = filtered.filter((r) => r.delta < threshold);
1232
+ if (topN !== undefined)
1233
+ filtered = filtered.slice(0, topN);
1234
+ console.log(`\n Sample-level diff: ${treatment} vs ${control} (report ${reportId})`);
1235
+ if (regressionsOnly)
1236
+ console.log(` Filter: regressions only (Δ < ${threshold})`);
1237
+ console.log('');
1238
+ console.log(' sample_id Δ composite (c→t) fact (c→t) behavior (c→t) judge (c→t)');
1239
+ console.log(' ' + '-'.repeat(100));
1240
+ if (filtered.length === 0) {
1241
+ console.log(regressionsOnly ? ' (no regressions found)' : ' (no shared samples)');
1242
+ console.log('');
1243
+ return;
1244
+ }
1245
+ const fmt = (a, b) => {
1246
+ const av = typeof a === 'number' ? a.toFixed(2) : '—';
1247
+ const bv = typeof b === 'number' ? b.toFixed(2) : '—';
1248
+ return `${av} → ${bv}`.padEnd(15);
1249
+ };
1250
+ for (const r of filtered) {
1251
+ const sign = r.delta > 0 ? '+' : '';
1252
+ const idCol = r.id.slice(0, 18).padEnd(20);
1253
+ const deltaCol = `${sign}${r.delta.toFixed(2)}`.padEnd(7);
1254
+ const compCol = `${r.cComp.toFixed(2)} → ${r.tComp.toFixed(2)}`.padEnd(17);
1255
+ console.log(` ${idCol}${deltaCol}${compCol}${fmt(r.cFact, r.tFact)} ${fmt(r.cBeh, r.tBeh)} ${fmt(r.cJudge, r.tJudge)}`);
1256
+ }
1257
+ console.log('');
1258
+ console.log(` Showing ${filtered.length} of ${rows.length} samples · sorted by |Δ|`);
1259
+ if (regressionsOnly) {
1260
+ const total = rows.length;
1261
+ const reg = rows.filter((r) => r.delta < threshold).length;
1262
+ console.log(` Regression rate: ${reg}/${total} samples (${total > 0 ? ((reg / total) * 100).toFixed(0) : 0}%)`);
1263
+ }
1264
+ console.log('');
1265
+ }
1266
+ // ---------------------------------------------------------------------------
1267
+ // handleGold — gold dataset workflow (init / validate / compare)
1268
+ // ---------------------------------------------------------------------------
1269
+ async function handleGold(argv) {
1270
+ const sub = argv[0];
1271
+ const rest = argv.slice(1);
1272
+ if (!sub || sub === '--help' || sub === '-h') {
1273
+ console.log([
1274
+ '',
1275
+ 'Usage: omk bench gold <subcommand>',
1276
+ '',
1277
+ 'Subcommands:',
1278
+ ' init [--out <dir>] [--annotator <id>] 生成空白 gold dataset 模板',
1279
+ ' validate <dir> 校验数据集结构',
1280
+ ' compare <reportId> --gold-dir <dir> 与已有 report 计算 α/κ/Pearson',
1281
+ ' [--variant <name>] [--reports-dir <d>]',
1282
+ ' [--bootstrap-samples N] [--seed N]',
1283
+ '',
1284
+ ].join('\n'));
1285
+ process.exit(sub ? 0 : 1);
1286
+ }
1287
+ if (sub === 'init') {
1288
+ const { values } = parseArgs({
1289
+ args: rest,
1290
+ options: {
1291
+ out: { type: 'string', default: './gold-dataset' },
1292
+ annotator: { type: 'string' },
1293
+ },
1294
+ strict: false,
1295
+ });
1296
+ const { initGoldDataset } = await import('./grading/gold-cli.js');
1297
+ try {
1298
+ const written = initGoldDataset(values.out, {
1299
+ annotator: values.annotator,
1300
+ });
1301
+ console.log(`Created ${written.length} files in ${values.out}:`);
1302
+ for (const p of written)
1303
+ console.log(` ${p}`);
1304
+ console.log('\n下一步: 编辑 annotations.yaml 加入真实标注 → 跑 omk bench gold validate');
1305
+ }
1306
+ catch (err) {
1307
+ console.error(err.message);
1308
+ process.exit(1);
1309
+ }
1310
+ return;
1311
+ }
1312
+ if (sub === 'validate') {
1313
+ const dir = rest[0];
1314
+ if (!dir) {
1315
+ console.error('Usage: omk bench gold validate <dir>');
1316
+ process.exit(1);
1317
+ }
1318
+ const { validateGoldDataset } = await import('./grading/gold-cli.js');
1319
+ const result = validateGoldDataset(dir);
1320
+ if (result.ok) {
1321
+ console.log(`✓ gold dataset OK — ${result.sampleCount} 条标注`);
1322
+ return;
1323
+ }
1324
+ console.error(`✗ gold dataset has ${result.issues.length} issue(s):`);
1325
+ for (const msg of result.issues)
1326
+ console.error(` - ${msg}`);
1327
+ process.exit(1);
1328
+ }
1329
+ if (sub === 'compare') {
1330
+ const reportId = rest[0];
1331
+ if (!reportId) {
1332
+ console.error('Usage: omk bench gold compare <reportId> --gold-dir <dir>');
1333
+ process.exit(1);
1334
+ }
1335
+ const { values } = parseArgs({
1336
+ args: rest.slice(1),
1337
+ options: {
1338
+ 'gold-dir': { type: 'string' },
1339
+ variant: { type: 'string' },
1340
+ 'reports-dir': { type: 'string', default: DEFAULT_REPORTS_DIR },
1341
+ 'bootstrap-samples': { type: 'string', default: '1000' },
1342
+ seed: { type: 'string' },
1343
+ },
1344
+ strict: false,
1345
+ });
1346
+ const goldDir = values['gold-dir'];
1347
+ if (!goldDir) {
1348
+ console.error('--gold-dir is required');
1349
+ process.exit(1);
1350
+ }
1351
+ const { loadGoldDataset } = await import('./grading/gold-dataset.js');
1352
+ const { compareGoldToReport, formatGoldCompare } = await import('./grading/gold-cli.js');
1353
+ const { createFileStore } = await import('./server/report-store.js');
1354
+ const { dataset, issues } = loadGoldDataset(goldDir);
1355
+ if (!dataset) {
1356
+ console.error('Cannot load gold dataset:');
1357
+ for (const i of issues)
1358
+ console.error(` - ${i.message}`);
1359
+ process.exit(1);
1360
+ }
1361
+ if (issues.length) {
1362
+ // Non-fatal issues (e.g. duplicate already filtered) — surface them.
1363
+ for (const i of issues)
1364
+ console.error(`warn: ${i.message}`);
1365
+ }
1366
+ const store = createFileStore(resolve(values['reports-dir']));
1367
+ const report = await store.get(reportId);
1368
+ if (!report) {
1369
+ console.error(`Report not found: ${reportId}`);
1370
+ process.exit(1);
1371
+ }
1372
+ const samples = Math.max(100, Number(values['bootstrap-samples']) || 1000);
1373
+ const seedVal = values.seed != null ? Number(values.seed) : undefined;
1374
+ const result = compareGoldToReport({
1375
+ report: report,
1376
+ gold: dataset,
1377
+ variant: values.variant,
1378
+ samples,
1379
+ seed: Number.isFinite(seedVal) ? seedVal : undefined,
1380
+ });
1381
+ console.log(formatGoldCompare(result, dataset));
1382
+ return;
1383
+ }
1384
+ console.error(`Unknown subcommand: gold ${sub}. Use init / validate / compare.`);
1385
+ process.exit(1);
1386
+ }
1387
+ // ---------------------------------------------------------------------------
1388
+ // handleDebiasValidate — measure length-debias prompt sensitivity (Phase 3a)
1389
+ // ---------------------------------------------------------------------------
1390
+ async function handleDebiasValidate(argv) {
1391
+ const sub = argv[0];
1392
+ const rest = argv.slice(1);
1393
+ if (!sub || sub === '--help' || sub === '-h') {
1394
+ console.log([
1395
+ '',
1396
+ 'Usage: omk bench debias-validate <kind> <reportId> [options]',
1397
+ '',
1398
+ 'Kinds:',
1399
+ ' length re-judge with the opposite length-debias setting and bootstrap CI',
1400
+ ' on the score diff. Cost ~doubles vs the original judge pass.',
1401
+ '',
1402
+ 'Options:',
1403
+ ' --reports-dir <dir> report store dir (default: ~/.oh-my-knowledge/reports)',
1404
+ ' --samples <path> override samples file (default: from report.meta.request)',
1405
+ ' --variant <name> which variant to validate (default: first)',
1406
+ ' --judge-executor <name> executor for judge calls (default: claude)',
1407
+ ' --judge-model <model> judge model id (default: from report)',
1408
+ ' --bootstrap-samples N bootstrap iterations (default 1000)',
1409
+ ' --seed N deterministic CI seed',
1410
+ '',
1411
+ ].join('\n'));
1412
+ process.exit(sub ? 0 : 1);
1413
+ }
1414
+ if (sub !== 'length') {
1415
+ console.error(`Unknown debias-validate kind: ${sub}. Use "length".`);
1416
+ process.exit(1);
1417
+ }
1418
+ const reportId = rest[0];
1419
+ if (!reportId) {
1420
+ console.error('Usage: omk bench debias-validate length <reportId>');
1421
+ process.exit(1);
1422
+ }
1423
+ const { values } = parseArgs({
1424
+ args: rest.slice(1),
1425
+ options: {
1426
+ 'reports-dir': { type: 'string', default: DEFAULT_REPORTS_DIR },
1427
+ samples: { type: 'string' },
1428
+ variant: { type: 'string' },
1429
+ 'judge-executor': { type: 'string', default: 'claude' },
1430
+ 'judge-model': { type: 'string' },
1431
+ 'bootstrap-samples': { type: 'string', default: '1000' },
1432
+ seed: { type: 'string' },
1433
+ },
1434
+ strict: false,
1435
+ });
1436
+ const { createFileStore } = await import('./server/report-store.js');
1437
+ const store = createFileStore(resolve(values['reports-dir']));
1438
+ const report = await store.get(reportId);
1439
+ if (!report) {
1440
+ console.error(`Report not found: ${reportId}`);
1441
+ process.exit(1);
1442
+ }
1443
+ // Resolve samples path: --samples overrides; otherwise read from report.meta.request.
1444
+ const samplesPath = values.samples
1445
+ ?? report.meta?.request?.samplesPath;
1446
+ if (!samplesPath) {
1447
+ console.error('Cannot find samples path. Pass --samples <path> or ensure report has request.samplesPath.');
1448
+ process.exit(1);
1449
+ }
1450
+ const { loadSamples } = await import('./inputs/load-samples.js');
1451
+ const { samples } = loadSamples(samplesPath);
1452
+ const judgeModel = values['judge-model']
1453
+ ?? report.meta?.judgeModel;
1454
+ if (!judgeModel) {
1455
+ console.error('No judge model. Pass --judge-model <id> or ensure report has meta.judgeModel.');
1456
+ process.exit(1);
1457
+ }
1458
+ process.stderr.write('\n⚠ debias-validate 会重判所有 (sample × variant),judge cost 大致翻倍。\n');
1459
+ const { createExecutor } = await import('./executors/index.js');
1460
+ const judgeExecutor = createExecutor(values['judge-executor']);
1461
+ const { validateLengthDebias, formatDebiasValidate } = await import('./grading/debias-validate.js');
1462
+ const seedVal = values.seed != null ? Number(values.seed) : undefined;
1463
+ const bsRaw = Number(values['bootstrap-samples']) || 1000;
1464
+ const result = await validateLengthDebias({
1465
+ report: report,
1466
+ samples,
1467
+ judgeExecutor,
1468
+ judgeModel,
1469
+ variant: values.variant,
1470
+ bootstrapSamples: Math.max(100, bsRaw),
1471
+ seed: Number.isFinite(seedVal) ? seedVal : undefined,
1472
+ onProgress: ({ sample_id, completed, total }) => {
1473
+ process.stderr.write(` judging ${completed}/${total}: ${sample_id}\n`);
1474
+ },
1475
+ });
1476
+ console.log(formatDebiasValidate(result));
1477
+ }
1478
+ // ---------------------------------------------------------------------------
1479
+ // handleSaturation — re-compute saturation verdict from a finished report
1480
+ // ---------------------------------------------------------------------------
1481
+ async function handleSaturation(argv) {
1482
+ const reportId = argv[0];
1483
+ if (!reportId || reportId === '--help' || reportId === '-h') {
1484
+ console.log([
1485
+ '',
1486
+ 'Usage: omk bench saturation <reportId> [options]',
1487
+ '',
1488
+ '回答"我跑够样本了吗?"。复述已有 report 中持久化的饱和判定。',
1489
+ '',
1490
+ '注:本命令读取 run 时跑出的 verdict(运行时已用 method=bootstrap-ci-width',
1491
+ '默认阈值 + 3 窗口 持续条件)。如要换 method/threshold 重新计算,需要重跑',
1492
+ '`omk bench run --repeat ≥ 5`(运行时持久化的 trace 不含原始分数,无法',
1493
+ '在事后用其他参数复算)。',
1494
+ '',
1495
+ 'Options:',
1496
+ ' --reports-dir <dir> report store dir (default: ~/.oh-my-knowledge/reports)',
1497
+ ' --variant <name> 只看一个 variant (default: all)',
1498
+ '',
1499
+ ].join('\n'));
1500
+ process.exit(reportId ? 0 : 1);
1501
+ }
1502
+ const { values } = parseArgs({
1503
+ args: argv.slice(1),
1504
+ options: {
1505
+ 'reports-dir': { type: 'string', default: DEFAULT_REPORTS_DIR },
1506
+ variant: { type: 'string' },
1507
+ },
1508
+ strict: false,
1509
+ });
1510
+ const { createFileStore } = await import('./server/report-store.js');
1511
+ const store = createFileStore(resolve(values['reports-dir']));
1512
+ const report = await store.get(reportId);
1513
+ if (!report) {
1514
+ console.error(`Report not found: ${reportId}`);
1515
+ process.exit(1);
1516
+ }
1517
+ const saturation = report.variance?.saturation;
1518
+ if (!saturation) {
1519
+ console.error('该 report 无 saturation 数据 (需要 --repeat ≥ 2 才会记录)。');
1520
+ process.exit(1);
1521
+ }
1522
+ // Print the persisted verdict from the original run. The trace stores
1523
+ // (mean, ciLow, ciHigh) per checkpoint but not raw scores, so re-running
1524
+ // findSaturationPoint with different method/threshold is not possible
1525
+ // here — that would need raw scores, which would have to be persisted
1526
+ // by runMultiple. Future work: opt-in `--persist-saturation-raw` flag at
1527
+ // run time to enable post-hoc parameter sweeps.
1528
+ const variants = report.meta.variants ?? [];
1529
+ const targetVariants = values.variant ? [values.variant] : variants;
1530
+ console.log(`\n Saturation verdict (复述持久化结果)\n`);
1531
+ for (const variant of targetVariants) {
1532
+ const trace = saturation.perVariant[variant];
1533
+ if (!trace || trace.length === 0) {
1534
+ console.log(` ${variant}: 无 trace 数据`);
1535
+ continue;
1536
+ }
1537
+ console.log(` ${variant}:`);
1538
+ console.log(` checkpoints: ${trace.length} (N=${trace.map((p) => p.n).join(', ')})`);
1539
+ console.log(` 最近一点 mean=${trace[trace.length - 1].mean.toFixed(3)}, CI=[${trace[trace.length - 1].ciLow.toFixed(3)}, ${trace[trace.length - 1].ciHigh.toFixed(3)}]`);
1540
+ if (saturation.verdicts?.[variant]) {
1541
+ const v = saturation.verdicts[variant];
1542
+ console.log(` 持久化判定 (${v.method}): ${v.saturated ? `已饱和@N=${v.atN}` : '未饱和'} - ${v.reason}`);
1543
+ }
1544
+ else if (trace.length < 5) {
1545
+ console.log(` 判定: 数据点 ${trace.length} < 5,跳过 (跑 --repeat 5 以上才输出)`);
1546
+ }
1547
+ }
1548
+ console.log('');
1549
+ }
1550
+ // ---------------------------------------------------------------------------
1551
+ // handleVerdict — one-line ship/no-ship verdict (v0.22)
1552
+ // ---------------------------------------------------------------------------
1553
+ async function handleVerdict(argv) {
1554
+ const reportId = argv[0];
1555
+ if (!reportId || reportId === '--help' || reportId === '-h') {
1556
+ console.log([
1557
+ '',
1558
+ 'Usage: omk bench verdict <reportId> [options]',
1559
+ '',
1560
+ '聚合 bootstrap CI / 三层 ci-gate / saturation / human α 给出一行结论。',
1561
+ '',
1562
+ 'Verdict 等级:',
1563
+ ' PROGRESS 显著改进 + 三层全过',
1564
+ ' CAUTIOUS 改进真实但有警告 (gate 破 / 幅度太小 / 控制组本身崩)',
1565
+ ' REGRESS 显著回退 — 不要 ship',
1566
+ ' NOISE CI 跨 0,无法判定',
1567
+ ' UNDERPOWERED 样本不足,需要扩 N 重测',
1568
+ ' SOLO 单变体报告,无对比对象',
1569
+ '',
1570
+ 'Options:',
1571
+ ' --reports-dir <dir> report store dir (default: ~/.oh-my-knowledge/reports)',
1572
+ ' --threshold <num> 三层 gate 阈值 (default 3.5,匹配 omk bench ci)',
1573
+ ' --trivial-diff <num> "幅度太小"阈值 (default 0.1)',
1574
+ ' --verbose 展开 per-pair 详情',
1575
+ '',
1576
+ ].join('\n'));
1577
+ process.exit(reportId ? 0 : 1);
1578
+ }
1579
+ const { values } = parseArgs({
1580
+ args: argv.slice(1),
1581
+ options: {
1582
+ 'reports-dir': { type: 'string', default: DEFAULT_REPORTS_DIR },
1583
+ threshold: { type: 'string' },
1584
+ 'trivial-diff': { type: 'string' },
1585
+ verbose: { type: 'boolean', default: false },
1586
+ },
1587
+ strict: false,
1588
+ });
1589
+ const { createFileStore } = await import('./server/report-store.js');
1590
+ const store = createFileStore(resolve(values['reports-dir']));
1591
+ const report = await store.get(reportId);
1592
+ if (!report) {
1593
+ console.error(`Report not found: ${reportId}`);
1594
+ process.exit(1);
1595
+ }
1596
+ const { computeVerdict, formatVerdictText } = await import('./eval-core/verdict.js');
1597
+ const result = computeVerdict(report, {
1598
+ ciThreshold: values.threshold != null ? Number(values.threshold) : undefined,
1599
+ triviallySmallDiff: values['trivial-diff'] != null ? Number(values['trivial-diff']) : undefined,
1600
+ });
1601
+ console.log(formatVerdictText(result, { verbose: Boolean(values.verbose) }));
1602
+ // Exit code reflects ship recommendation: 0 only on PROGRESS / SOLO-pass.
1603
+ // NOISE / UNDERPOWERED / CAUTIOUS / REGRESS all exit 1 so this composes
1604
+ // with shell `&&` chains in CI.
1605
+ if (result.level === 'PROGRESS') {
1606
+ process.exit(0);
1607
+ }
1608
+ if (result.level === 'SOLO' && result.headline.includes('PASS')) {
1609
+ process.exit(0);
1610
+ }
1611
+ process.exit(1);
1612
+ }
1613
+ // ---------------------------------------------------------------------------
1614
+ // handleDiagnose — per-sample quality diagnostics (v0.23 A)
1615
+ // ---------------------------------------------------------------------------
1616
+ async function handleDiagnose(argv) {
1617
+ const reportId = argv[0];
1618
+ if (!reportId || reportId === '--help' || reportId === '-h') {
1619
+ console.log([
1620
+ '',
1621
+ 'Usage: omk bench diagnose <reportId> [options]',
1622
+ '',
1623
+ '诊断样本集本身的质量问题:区分度低 / 重复 / 歧义 / 成本异常 / 全 fail。',
1624
+ '回答"测评结论是否被坏样本污染"——与 omk bench verdict 互补。',
1625
+ '',
1626
+ 'Options:',
1627
+ ' --reports-dir <dir> report store dir',
1628
+ ' --samples <path> 样本文件路径 (用于 near-duplicate 检测;默认从 report.meta.request 读)',
1629
+ ' --top <n> 每类只显示前 N 个 (默认 10,0=全部)',
1630
+ ' --duplicate-rouge <num> near-duplicate ROUGE-1 阈值 (默认 0.7)',
1631
+ ' --ambiguous-stddev <num> 歧义阈值,judge stddev (默认 1.0,需要 --judge-repeat ≥ 2 数据)',
1632
+ ' --cost-k <num> 成本异常倍数 vs median (默认 3)',
1633
+ ' --latency-k <num> 耗时异常倍数 vs median (默认 3)',
1634
+ ' --flat <num> flat_scores 分差阈值 (默认 0.5)',
1635
+ '',
1636
+ ].join('\n'));
1637
+ process.exit(reportId ? 0 : 1);
1638
+ }
1639
+ const { values } = parseArgs({
1640
+ args: argv.slice(1),
1641
+ options: {
1642
+ 'reports-dir': { type: 'string', default: DEFAULT_REPORTS_DIR },
1643
+ samples: { type: 'string' },
1644
+ top: { type: 'string', default: '10' },
1645
+ 'duplicate-rouge': { type: 'string' },
1646
+ 'ambiguous-stddev': { type: 'string' },
1647
+ 'cost-k': { type: 'string' },
1648
+ 'latency-k': { type: 'string' },
1649
+ flat: { type: 'string' },
1650
+ },
1651
+ strict: false,
1652
+ });
1653
+ const { createFileStore } = await import('./server/report-store.js');
1654
+ const store = createFileStore(resolve(values['reports-dir']));
1655
+ const report = await store.get(reportId);
1656
+ if (!report) {
1657
+ console.error(`Report not found: ${reportId}`);
1658
+ process.exit(1);
1659
+ }
1660
+ // Try to read the samples file for near-duplicate detection. Source order:
1661
+ // 1. --samples <path> override
1662
+ // 2. report.meta.request.samplesPath (recorded at run time)
1663
+ // If neither resolves to a readable file, skip near-duplicate gracefully.
1664
+ let samples;
1665
+ const samplesPath = values.samples ?? report.meta?.request?.samplesPath;
1666
+ if (samplesPath && existsSync(samplesPath)) {
1667
+ try {
1668
+ const { loadSamples } = await import('./inputs/load-samples.js');
1669
+ samples = loadSamples(samplesPath).samples;
1670
+ }
1671
+ catch (err) {
1672
+ process.stderr.write(`warn: 加载 samples 文件失败 (${samplesPath}): ${err.message}\n`);
1673
+ }
1674
+ }
1675
+ const topRaw = Number(values.top);
1676
+ const topN = Number.isFinite(topRaw) && topRaw > 0 ? topRaw : undefined;
1677
+ const { diagnoseSamples, formatSampleDiagnostics } = await import('./analysis/sample-diagnostics.js');
1678
+ const diag = diagnoseSamples(report, {
1679
+ samples,
1680
+ duplicateRouge: values['duplicate-rouge'] != null ? Number(values['duplicate-rouge']) : undefined,
1681
+ ambiguousStddev: values['ambiguous-stddev'] != null ? Number(values['ambiguous-stddev']) : undefined,
1682
+ costOutlierK: values['cost-k'] != null ? Number(values['cost-k']) : undefined,
1683
+ latencyOutlierK: values['latency-k'] != null ? Number(values['latency-k']) : undefined,
1684
+ flatThreshold: values.flat != null ? Number(values.flat) : undefined,
1685
+ });
1686
+ console.log(formatSampleDiagnostics(diag, { topN }));
1687
+ // Exit code: 0 if health ≥ 70 and no errors; 1 otherwise. CI-friendly.
1688
+ if (diag.totals.errors === 0 && diag.healthScore >= 70) {
1689
+ process.exit(0);
1690
+ }
1691
+ process.exit(1);
1692
+ }
1693
+ // ---------------------------------------------------------------------------
1694
+ // handleFailures — LLM-driven failure clustering (v0.23 B)
1695
+ // ---------------------------------------------------------------------------
1696
+ async function handleFailures(argv) {
1697
+ const reportId = argv[0];
1698
+ if (!reportId || reportId === '--help' || reportId === '-h') {
1699
+ console.log([
1700
+ '',
1701
+ 'Usage: omk bench failures <reportId> [options]',
1702
+ '',
1703
+ '把已有 report 的失败样本喂给一次 LLM 调用,自动聚类 + 给修复建议。',
1704
+ '失败定义:compositeScore < threshold 或 ok=false。',
1705
+ '',
1706
+ 'Options:',
1707
+ ' --reports-dir <dir> report store dir',
1708
+ ' --judge-executor <name> 执行器 (default: claude)',
1709
+ ' --judge-model <id> 聚类用的 model (default: 沿用 report.meta.judgeModel)',
1710
+ ' --max-clusters <n> 最多多少 cluster (default 5)',
1711
+ ' --threshold <num> compositeScore < threshold 算失败 (default 3)',
1712
+ ' --max-feed <n> 最多喂给 LLM 多少条 (default 50,超出取最差)',
1713
+ '',
1714
+ ].join('\n'));
1715
+ process.exit(reportId ? 0 : 1);
1716
+ }
1717
+ const { values } = parseArgs({
1718
+ args: argv.slice(1),
1719
+ options: {
1720
+ 'reports-dir': { type: 'string', default: DEFAULT_REPORTS_DIR },
1721
+ 'judge-executor': { type: 'string', default: 'claude' },
1722
+ 'judge-model': { type: 'string' },
1723
+ 'max-clusters': { type: 'string', default: '5' },
1724
+ threshold: { type: 'string', default: '3' },
1725
+ 'max-feed': { type: 'string', default: '50' },
1726
+ },
1727
+ strict: false,
1728
+ });
1729
+ const { createFileStore } = await import('./server/report-store.js');
1730
+ const store = createFileStore(resolve(values['reports-dir']));
1731
+ const report = await store.get(reportId);
1732
+ if (!report) {
1733
+ console.error(`Report not found: ${reportId}`);
1734
+ process.exit(1);
1735
+ }
1736
+ const judgeModel = values['judge-model'] ?? report.meta?.judgeModel;
1737
+ if (!judgeModel) {
1738
+ console.error('No judge model. Pass --judge-model <id> or ensure report has meta.judgeModel.');
1739
+ process.exit(1);
1740
+ }
1741
+ const { createExecutor } = await import('./executors/index.js');
1742
+ const executor = createExecutor(values['judge-executor']);
1743
+ const { clusterFailures, formatFailureClusterReport } = await import('./analysis/failure-clusterer.js');
1744
+ const out = await clusterFailures({
1745
+ report: report,
1746
+ executor,
1747
+ judgeModel,
1748
+ maxClusters: Number(values['max-clusters']) || 5,
1749
+ failureThreshold: Number(values.threshold) || 3,
1750
+ maxFailuresFed: Number(values['max-feed']) || 50,
1751
+ });
1752
+ console.log(formatFailureClusterReport(out));
1753
+ }
1008
1754
  // ---------------------------------------------------------------------------
1009
1755
  // Entry
1010
1756
  // ---------------------------------------------------------------------------