oh-my-knowledge 0.18.0 → 0.20.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (113) hide show
  1. package/README.md +596 -326
  2. package/README.zh.md +917 -0
  3. package/dist/src/analysis/failure-clusterer.d.ts +96 -0
  4. package/dist/src/analysis/failure-clusterer.d.ts.map +1 -0
  5. package/dist/src/analysis/failure-clusterer.js +298 -0
  6. package/dist/src/analysis/failure-clusterer.js.map +1 -0
  7. package/dist/src/analysis/sample-diagnostics.d.ts +78 -0
  8. package/dist/src/analysis/sample-diagnostics.d.ts.map +1 -0
  9. package/dist/src/analysis/sample-diagnostics.js +259 -0
  10. package/dist/src/analysis/sample-diagnostics.js.map +1 -0
  11. package/dist/src/analysis/saturation.d.ts +85 -0
  12. package/dist/src/analysis/saturation.d.ts.map +1 -0
  13. package/dist/src/analysis/saturation.js +174 -0
  14. package/dist/src/analysis/saturation.js.map +1 -0
  15. package/dist/src/cli.js +772 -18
  16. package/dist/src/cli.js.map +1 -1
  17. package/dist/src/eval-core/bootstrap.d.ts +72 -0
  18. package/dist/src/eval-core/bootstrap.d.ts.map +1 -0
  19. package/dist/src/eval-core/bootstrap.js +174 -0
  20. package/dist/src/eval-core/bootstrap.js.map +1 -0
  21. package/dist/src/eval-core/evaluation-execution.d.ts +15 -1
  22. package/dist/src/eval-core/evaluation-execution.d.ts.map +1 -1
  23. package/dist/src/eval-core/evaluation-execution.js +37 -3
  24. package/dist/src/eval-core/evaluation-execution.js.map +1 -1
  25. package/dist/src/eval-core/evaluation-job.d.ts +10 -2
  26. package/dist/src/eval-core/evaluation-job.d.ts.map +1 -1
  27. package/dist/src/eval-core/evaluation-job.js +9 -1
  28. package/dist/src/eval-core/evaluation-job.js.map +1 -1
  29. package/dist/src/eval-core/evaluation-reporting.d.ts.map +1 -1
  30. package/dist/src/eval-core/evaluation-reporting.js +90 -0
  31. package/dist/src/eval-core/evaluation-reporting.js.map +1 -1
  32. package/dist/src/eval-core/schema.d.ts.map +1 -1
  33. package/dist/src/eval-core/schema.js +69 -0
  34. package/dist/src/eval-core/schema.js.map +1 -1
  35. package/dist/src/eval-core/verdict.d.ts +74 -0
  36. package/dist/src/eval-core/verdict.d.ts.map +1 -0
  37. package/dist/src/eval-core/verdict.js +283 -0
  38. package/dist/src/eval-core/verdict.js.map +1 -0
  39. package/dist/src/eval-workflows/each-evaluation-workflow.d.ts +16 -3
  40. package/dist/src/eval-workflows/each-evaluation-workflow.d.ts.map +1 -1
  41. package/dist/src/eval-workflows/each-evaluation-workflow.js +10 -2
  42. package/dist/src/eval-workflows/each-evaluation-workflow.js.map +1 -1
  43. package/dist/src/eval-workflows/evaluation-pipeline.d.ts +17 -1
  44. package/dist/src/eval-workflows/evaluation-pipeline.d.ts.map +1 -1
  45. package/dist/src/eval-workflows/evaluation-pipeline.js +45 -3
  46. package/dist/src/eval-workflows/evaluation-pipeline.js.map +1 -1
  47. package/dist/src/eval-workflows/run-evaluation.d.ts +23 -2
  48. package/dist/src/eval-workflows/run-evaluation.d.ts.map +1 -1
  49. package/dist/src/eval-workflows/run-evaluation.js +104 -4
  50. package/dist/src/eval-workflows/run-evaluation.js.map +1 -1
  51. package/dist/src/grading/assertions.d.ts +16 -0
  52. package/dist/src/grading/assertions.d.ts.map +1 -1
  53. package/dist/src/grading/assertions.js +385 -111
  54. package/dist/src/grading/assertions.js.map +1 -1
  55. package/dist/src/grading/debias-validate.d.ts +84 -0
  56. package/dist/src/grading/debias-validate.d.ts.map +1 -0
  57. package/dist/src/grading/debias-validate.js +173 -0
  58. package/dist/src/grading/debias-validate.js.map +1 -0
  59. package/dist/src/grading/gold-cli.d.ts +88 -0
  60. package/dist/src/grading/gold-cli.d.ts.map +1 -0
  61. package/dist/src/grading/gold-cli.js +251 -0
  62. package/dist/src/grading/gold-cli.js.map +1 -0
  63. package/dist/src/grading/gold-dataset.d.ts +73 -0
  64. package/dist/src/grading/gold-dataset.d.ts.map +1 -0
  65. package/dist/src/grading/gold-dataset.js +161 -0
  66. package/dist/src/grading/gold-dataset.js.map +1 -0
  67. package/dist/src/grading/human-gold.d.ts +102 -0
  68. package/dist/src/grading/human-gold.d.ts.map +1 -0
  69. package/dist/src/grading/human-gold.js +188 -0
  70. package/dist/src/grading/human-gold.js.map +1 -0
  71. package/dist/src/grading/index.d.ts +27 -2
  72. package/dist/src/grading/index.d.ts.map +1 -1
  73. package/dist/src/grading/index.js +36 -18
  74. package/dist/src/grading/index.js.map +1 -1
  75. package/dist/src/grading/judge.d.ts +65 -2
  76. package/dist/src/grading/judge.d.ts.map +1 -1
  77. package/dist/src/grading/judge.js +280 -23
  78. package/dist/src/grading/judge.js.map +1 -1
  79. package/dist/src/inputs/eval-config.js +19 -0
  80. package/dist/src/inputs/eval-config.js.map +1 -1
  81. package/dist/src/observability/{production-analyzer.d.ts → skill-health-analyzer.d.ts} +24 -2
  82. package/dist/src/observability/skill-health-analyzer.d.ts.map +1 -0
  83. package/dist/src/observability/{production-analyzer.js → skill-health-analyzer.js} +61 -6
  84. package/dist/src/observability/skill-health-analyzer.js.map +1 -0
  85. package/dist/src/observability/trace-adapter.d.ts.map +1 -1
  86. package/dist/src/observability/trace-adapter.js +27 -1
  87. package/dist/src/observability/trace-adapter.js.map +1 -1
  88. package/dist/src/renderer/html-renderer.d.ts.map +1 -1
  89. package/dist/src/renderer/html-renderer.js +40 -6
  90. package/dist/src/renderer/html-renderer.js.map +1 -1
  91. package/dist/src/renderer/layout.d.ts.map +1 -1
  92. package/dist/src/renderer/layout.js +138 -4
  93. package/dist/src/renderer/layout.js.map +1 -1
  94. package/dist/src/renderer/skill-health-renderer.d.ts +2 -2
  95. package/dist/src/renderer/skill-health-renderer.d.ts.map +1 -1
  96. package/dist/src/renderer/skill-health-renderer.js +39 -4
  97. package/dist/src/renderer/skill-health-renderer.js.map +1 -1
  98. package/dist/src/renderer/summary.d.ts +28 -1
  99. package/dist/src/renderer/summary.d.ts.map +1 -1
  100. package/dist/src/renderer/summary.js +322 -8
  101. package/dist/src/renderer/summary.js.map +1 -1
  102. package/dist/src/renderer/table.d.ts.map +1 -1
  103. package/dist/src/renderer/table.js +63 -2
  104. package/dist/src/renderer/table.js.map +1 -1
  105. package/dist/src/server/report-server.d.ts +2 -1
  106. package/dist/src/server/report-server.d.ts.map +1 -1
  107. package/dist/src/server/report-server.js +397 -2
  108. package/dist/src/server/report-server.js.map +1 -1
  109. package/dist/src/types.d.ts +247 -0
  110. package/dist/src/types.d.ts.map +1 -1
  111. package/package.json +24 -6
  112. package/dist/src/observability/production-analyzer.d.ts.map +0 -1
  113. package/dist/src/observability/production-analyzer.js.map +0 -1
package/dist/src/cli.js CHANGED
@@ -95,11 +95,17 @@ function parseRunConfig(argv, extraOptions = {}) {
95
95
  else if (evalConfig) {
96
96
  variantSpecs = configVariantsToSpecs(evalConfig.variants);
97
97
  }
98
+ else if (values.each) {
99
+ // --each 模式自动用 baseline (control) vs 每个 skill (treatment),
100
+ // 不需要用户显式传 --control / --treatment,校验跳过。
101
+ variantSpecs = [];
102
+ }
98
103
  else {
99
104
  const discovered = discoverVariants(skillDir);
100
105
  const hint = discovered.length > 0 ? `\n skill-dir (${skillDir}) 下发现的候选:${discovered.join(', ')}` : '';
101
106
  throw new Error(`请通过 --control / --treatment 或 --config eval.yaml 声明 variant 角色。\n`
102
107
  + ` 示例:omk bench run --control baseline --treatment my-skill${hint}\n`
108
+ + ` --each 模式下自动用 baseline vs 每个 skill,无需显式声明\n`
103
109
  + ` 术语见 docs/terminology-spec.md(v0.16 起废除 --variants,改用 experiment role 显式声明)`);
104
110
  }
105
111
  const seenNames = new Set();
@@ -161,6 +167,7 @@ function parseRunConfig(argv, extraOptions = {}) {
161
167
  resume,
162
168
  blind,
163
169
  layeredStats,
170
+ budget: evalConfig?.budget,
164
171
  },
165
172
  };
166
173
  }
@@ -208,6 +215,22 @@ Options for "bench run":
208
215
  --concurrency <n> Number of parallel tasks (default: 1)
209
216
  --timeout <seconds> Executor timeout per task in seconds (default: 120)
210
217
  --repeat <n> Run evaluation N times for variance analysis (default: 1)
218
+ --judge-repeat <n> Call LLM judge N times per (sample × dimension) for self-
219
+ consistency (default: 1). High stddev across runs = the
220
+ judge is unstable on this rubric and the score is noisy.
221
+ --judge-models <list> Multi-judge ensemble. Comma-separated executor:model pairs,
222
+ e.g. claude:opus,openai:gpt-4o,gemini:pro. Each judge scores
223
+ every (sample × dimension); report includes per-judge break-
224
+ down + Pearson/MAD inter-judge agreement. Refutes "Claude
225
+ judge Claude same-modality bias" critique. Combines with
226
+ --judge-repeat. Cost ~ N_judges × N_repeat × N_samples.
227
+ --bootstrap Compute bootstrap confidence intervals (distribution-free,
228
+ preferred over t-interval for ordinal LLM scores). Adds
229
+ per-variant CI on the mean + pairwise CI on treatment-vs-
230
+ control difference (significant=0 outside CI). Reports both
231
+ t-interval and bootstrap so old tooling still works.
232
+ --bootstrap-samples <n> Number of bootstrap resamples (default 1000). N>10000
233
+ triggers a stderr warning about runtime cost.
211
234
  --retry <n> Retry failed tasks up to N times with exponential backoff (default: 0)
212
235
  --resume <report-id> Resume from a previous report, skipping completed tasks
213
236
  --executor <name> Executor: claude, openai, gemini, anthropic-api, openai-api,
@@ -354,8 +377,26 @@ async function main() {
354
377
  case 'diff':
355
378
  await handleDiff(rest);
356
379
  break;
380
+ case 'gold':
381
+ await handleGold(rest);
382
+ break;
383
+ case 'debias-validate':
384
+ await handleDebiasValidate(rest);
385
+ break;
386
+ case 'saturation':
387
+ await handleSaturation(rest);
388
+ break;
389
+ case 'verdict':
390
+ await handleVerdict(rest);
391
+ break;
392
+ case 'diagnose':
393
+ await handleDiagnose(rest);
394
+ break;
395
+ case 'failures':
396
+ await handleFailures(rest);
397
+ break;
357
398
  default:
358
- console.error(`Unknown command: bench ${command}. Use "run", "report", "ci", "init", "gen-samples", or "evolve".`);
399
+ console.error(`Unknown command: bench ${command}. Use "run", "report", "ci", "init", "gen-samples", "evolve", "diff", "gold", "debias-validate", "saturation", "verdict", "diagnose", or "failures".`);
359
400
  process.exit(1);
360
401
  }
361
402
  }
@@ -410,15 +451,98 @@ async function handleRun(argv) {
410
451
  const { values, config } = parseRunConfig(argv, {
411
452
  blind: { type: 'boolean', default: false },
412
453
  repeat: { type: 'string', default: '1' },
454
+ 'judge-repeat': { type: 'string', default: '1' },
455
+ 'judge-models': { type: 'string' },
456
+ bootstrap: { type: 'boolean', default: false },
457
+ 'bootstrap-samples': { type: 'string', default: '1000' },
458
+ 'gold-dir': { type: 'string' },
459
+ 'no-debias-length': { type: 'boolean', default: false },
460
+ 'budget-usd': { type: 'string' },
461
+ 'budget-per-sample-usd': { type: 'string' },
462
+ 'budget-per-sample-ms': { type: 'string' },
413
463
  });
414
464
  const { runEvaluation, runMultiple, runEachEvaluation } = await import('./eval-workflows/run-evaluation.js');
415
465
  config.blind = values.blind;
416
466
  config.onProgress = defaultOnProgress;
467
+ // --repeat 诚实输入校验:非 ≥1 整数时提示并钳到 1,不静默掩盖用户错字/极端输入
468
+ // 提前到 --each 分支之前,保证 each 模式也能读到 repeat (曾经 bug: --each 吞 --repeat)
469
+ const repeatRaw = values.repeat;
470
+ const parsedRepeat = repeatRaw !== undefined ? Number(repeatRaw) : 1;
471
+ if (repeatRaw !== undefined && (!Number.isFinite(parsedRepeat) || parsedRepeat < 1)) {
472
+ process.stderr.write(`⚠ --repeat "${repeatRaw}" 无效(期望 ≥ 1 的整数),已按 1 次评测执行\n`);
473
+ }
474
+ const repeatCount = Math.max(1, Math.floor(parsedRepeat) || 1);
475
+ // --judge-repeat 同样的诚实校验:非 ≥1 整数时钳到 1
476
+ const judgeRepeatRaw = values['judge-repeat'];
477
+ const parsedJudgeRepeat = judgeRepeatRaw !== undefined ? Number(judgeRepeatRaw) : 1;
478
+ if (judgeRepeatRaw !== undefined && (!Number.isFinite(parsedJudgeRepeat) || parsedJudgeRepeat < 1)) {
479
+ process.stderr.write(`⚠ --judge-repeat "${judgeRepeatRaw}" 无效(期望 ≥ 1 的整数),已按 1 次 judge 执行\n`);
480
+ }
481
+ const judgeRepeatCount = Math.max(1, Math.floor(parsedJudgeRepeat) || 1);
482
+ if (judgeRepeatCount > 1)
483
+ config.judgeRepeat = judgeRepeatCount;
484
+ // --judge-models executor:model,executor:model,... -> JudgeConfig[]
485
+ // 至少 2 个才进 ensemble 模式,1 个等同于 --judge-model
486
+ const judgeModelsRaw = values['judge-models'];
487
+ if (judgeModelsRaw) {
488
+ const parts = judgeModelsRaw.split(',').map((s) => s.trim()).filter(Boolean);
489
+ const judges = parts.map((p) => {
490
+ const [executor, ...modelParts] = p.split(':');
491
+ const model = modelParts.join(':');
492
+ if (!executor || !model) {
493
+ throw new Error(`--judge-models 格式错误: "${p}",应为 "executor:model" (如 claude:opus)`);
494
+ }
495
+ return { executor, model };
496
+ });
497
+ if (judges.length >= 2) {
498
+ config.judgeModels = judges;
499
+ }
500
+ else if (judges.length === 1) {
501
+ // 单 judge 不走 ensemble,但允许这样写,等同于 --judge-model + --executor
502
+ process.stderr.write(`ℹ --judge-models 只指定 1 个 judge (${judges[0].executor}:${judges[0].model}),不触发 ensemble。如需 ensemble 至少给 2 个。\n`);
503
+ }
504
+ }
505
+ // --budget-usd / --budget-per-sample-usd / --budget-per-sample-ms:
506
+ // v0.22 hard budget caps. CLI flags override config-file values. When the
507
+ // total-USD cap is exceeded mid-run, remaining tasks are skipped and a
508
+ // partial report is persisted with meta.budgetExhausted=true.
509
+ const budgetUSD = values['budget-usd'] != null ? Number(values['budget-usd']) : undefined;
510
+ const budgetPerSampleUSD = values['budget-per-sample-usd'] != null ? Number(values['budget-per-sample-usd']) : undefined;
511
+ const budgetPerSampleMs = values['budget-per-sample-ms'] != null ? Number(values['budget-per-sample-ms']) : undefined;
512
+ if (budgetUSD !== undefined || budgetPerSampleUSD !== undefined || budgetPerSampleMs !== undefined) {
513
+ config.budget = {
514
+ ...(budgetUSD !== undefined && Number.isFinite(budgetUSD) && budgetUSD >= 0 ? { totalUSD: budgetUSD } : {}),
515
+ ...(budgetPerSampleUSD !== undefined && Number.isFinite(budgetPerSampleUSD) && budgetPerSampleUSD >= 0 ? { perSampleUSD: budgetPerSampleUSD } : {}),
516
+ ...(budgetPerSampleMs !== undefined && Number.isFinite(budgetPerSampleMs) && budgetPerSampleMs >= 0 ? { perSampleMs: budgetPerSampleMs } : {}),
517
+ };
518
+ }
519
+ // --no-debias-length: opt out of v0.21 Phase 3a length-controlled prompt.
520
+ // Default behavior is debias-on (judge prompt v3-cot-length); flag flips it
521
+ // off so historical reports (judgePromptHash from v2-cot era) can be reproduced.
522
+ if (values['no-debias-length']) {
523
+ config.lengthDebias = false;
524
+ process.stderr.write('ℹ --no-debias-length 已生效:judge prompt 退回 v2-cot,与 < v0.21 报告 hash 一致。\n');
525
+ }
526
+ // --bootstrap / --bootstrap-samples
527
+ if (values.bootstrap) {
528
+ config.bootstrap = true;
529
+ const bsRaw = values['bootstrap-samples'];
530
+ const parsedBs = bsRaw !== undefined ? Number(bsRaw) : 1000;
531
+ if (bsRaw !== undefined && (!Number.isFinite(parsedBs) || parsedBs < 100)) {
532
+ process.stderr.write(`⚠ --bootstrap-samples "${bsRaw}" 无效(期望 ≥ 100 的整数),已按 1000 执行\n`);
533
+ }
534
+ const bsCount = Math.max(100, Math.floor(parsedBs) || 1000);
535
+ if (bsCount > 10000) {
536
+ process.stderr.write(`⚠ --bootstrap-samples ${bsCount} 较大,可能耗时数秒。1000 是业内标准,通常已够用。\n`);
537
+ }
538
+ config.bootstrapSamples = bsCount;
539
+ }
417
540
  try {
418
541
  // --each mode: evaluate each skill independently
419
542
  if (values.each) {
420
543
  const { report, filePath } = await runEachEvaluation({
421
544
  ...config,
545
+ repeat: repeatCount,
422
546
  onSkillProgress({ phase, skill, current, total }) {
423
547
  if (phase === 'start') {
424
548
  process.stderr.write(`\n=== [${current}/${total}] Skill: ${skill} ===\n`);
@@ -449,13 +573,6 @@ async function handleRun(argv) {
449
573
  }
450
574
  return;
451
575
  }
452
- // --repeat 诚实输入校验:非 ≥1 整数时提示并钳到 1,不静默掩盖用户错字/极端输入
453
- const repeatRaw = values.repeat;
454
- const parsedRepeat = repeatRaw !== undefined ? Number(repeatRaw) : 1;
455
- if (repeatRaw !== undefined && (!Number.isFinite(parsedRepeat) || parsedRepeat < 1)) {
456
- process.stderr.write(`⚠ --repeat "${repeatRaw}" 无效(期望 ≥ 1 的整数),已按 1 次评测执行\n`);
457
- }
458
- const repeatCount = Math.max(1, Math.floor(parsedRepeat) || 1);
459
576
  let report;
460
577
  let filePath;
461
578
  if (repeatCount > 1) {
@@ -474,6 +591,28 @@ async function handleRun(argv) {
474
591
  report = result.report;
475
592
  filePath = result.filePath;
476
593
  }
594
+ // --gold-dir: compute α/κ/Pearson against gold annotations and re-persist.
595
+ const goldDir = values['gold-dir'];
596
+ if (goldDir && filePath) {
597
+ const { attachGoldAgreementToReport, formatGoldCompare } = await import('./grading/gold-cli.js');
598
+ const out = attachGoldAgreementToReport({
599
+ report,
600
+ goldDir,
601
+ outputDir: config.outputDir,
602
+ samples: config.bootstrapSamples,
603
+ });
604
+ if (out.result && out.gold) {
605
+ process.stderr.write(formatGoldCompare(out.result, out.gold));
606
+ if (out.result.contaminationWarning) {
607
+ process.stderr.write(`\n⚠ ${out.result.contaminationWarning}\n`);
608
+ }
609
+ }
610
+ else {
611
+ process.stderr.write(`\n⚠ gold dataset 加载失败 (${goldDir}):\n`);
612
+ for (const m of out.loadIssues)
613
+ process.stderr.write(` - ${m}\n`);
614
+ }
615
+ }
477
616
  console.log(JSON.stringify(report, null, 2));
478
617
  if (filePath) {
479
618
  process.stderr.write('\n✅ 评测完成\n');
@@ -676,20 +815,19 @@ async function handleAnalyze(argv) {
676
815
  const to = values.to;
677
816
  const skills = values.skills ? values.skills.split(',').map((s) => s.trim()).filter(Boolean) : undefined;
678
817
  console.log(`[omk] analyzing ${tracePath}...`);
679
- const { computeSkillHealthReport } = await import('./observability/production-analyzer.js');
818
+ const { computeSkillHealthReport } = await import('./observability/skill-health-analyzer.js');
680
819
  const report = computeSkillHealthReport(tracePath, {
681
820
  kbRoot: values.kb ? resolve(values.kb) : undefined,
682
821
  from,
683
822
  to,
684
823
  skills,
685
824
  });
686
- const { renderSkillHealthReport } = await import('./renderer/skill-health-renderer.js');
687
- const html = renderSkillHealthReport(report);
825
+ // JSON 是主产物; HTML 由 report server 的 /analyses/:id 按需渲染 (和 bench run 一致)
688
826
  const outDir = resolve(values['output-dir'] || join(process.env.HOME || '.', '.oh-my-knowledge', 'analyses'));
689
827
  mkdirSync(outDir, { recursive: true });
690
828
  const timestamp = new Date().toISOString().replace(/[:.]/g, '-').slice(0, 19);
691
- const outPath = join(outDir, `${timestamp}-skill-health.html`);
692
- writeFileSync(outPath, html);
829
+ const jsonPath = join(outDir, `${timestamp}-skill-health.json`);
830
+ writeFileSync(jsonPath, JSON.stringify(report, null, 2));
693
831
  // 控制台摘要
694
832
  const { sessionCount, segmentCount, toolCallCount, toolFailureRate } = report.meta;
695
833
  console.log('');
@@ -703,7 +841,8 @@ async function handleAnalyze(argv) {
703
841
  console.log('top skills:');
704
842
  console.log(skillRows.join('\n'));
705
843
  console.log('');
706
- console.log(`report written to: ${outPath}`);
844
+ console.log(`report written to: ${jsonPath}`);
845
+ console.log(`view in browser: omk bench report # 打开后点首页的 "📊 Skill 健康度日报"`);
707
846
  }
708
847
  async function handleInit(argv) {
709
848
  const targetDir = resolve(argv[0] || '.');
@@ -934,13 +1073,57 @@ async function handleCi(argv) {
934
1073
  // handleDiff
935
1074
  // ---------------------------------------------------------------------------
936
1075
  async function handleDiff(argv) {
937
- if (argv.length < 2) {
938
- console.error('Usage: omk bench diff <report-id-1> <report-id-2>');
939
- process.exit(1);
1076
+ // Flag-aware split: separate positional report IDs from flags so we can support
1077
+ // omk bench diff <id> — within-report sample-level (v0.22)
1078
+ // omk bench diff <id1> <id2> — cross-report variant-level (legacy)
1079
+ // both with optional --regressions-only / --threshold / --variant flags.
1080
+ const positional = [];
1081
+ const flagArgs = [];
1082
+ for (let i = 0; i < argv.length; i++) {
1083
+ const a = argv[i];
1084
+ if (a.startsWith('--')) {
1085
+ flagArgs.push(a);
1086
+ const next = argv[i + 1];
1087
+ if (next !== undefined && !next.startsWith('--')) {
1088
+ flagArgs.push(next);
1089
+ i++;
1090
+ }
1091
+ }
1092
+ else {
1093
+ positional.push(a);
1094
+ }
1095
+ }
1096
+ if (positional.length === 0) {
1097
+ console.error([
1098
+ 'Usage:',
1099
+ ' omk bench diff <reportId> within-report per-sample diff (v0.22)',
1100
+ ' omk bench diff <reportId1> <reportId2> cross-report variant-level diff',
1101
+ '',
1102
+ 'Options:',
1103
+ ' --regressions-only 只列 treatment < control 的样本',
1104
+ ' --threshold <num> regression 阈值 (default 0,即任一负 Δ 算回退)',
1105
+ ' --variant <name> within-report 模式下指定要钻取的 variant (default: variants[1])',
1106
+ ' --top <n> 只列差距最大的前 N 个样本',
1107
+ ].join('\n'));
1108
+ process.exit(positional.length === 0 ? 1 : 0);
940
1109
  }
1110
+ const { values } = parseArgs({
1111
+ args: flagArgs,
1112
+ options: {
1113
+ 'regressions-only': { type: 'boolean', default: false },
1114
+ threshold: { type: 'string' },
1115
+ variant: { type: 'string' },
1116
+ top: { type: 'string' },
1117
+ },
1118
+ strict: false,
1119
+ });
941
1120
  const { createFileStore } = await import('./server/report-store.js');
942
1121
  const store = createFileStore(resolve(DEFAULT_REPORTS_DIR));
943
- const [id1, id2] = argv;
1122
+ if (positional.length === 1) {
1123
+ await runSampleLevelDiff(positional[0], store, values);
1124
+ return;
1125
+ }
1126
+ const [id1, id2] = positional;
944
1127
  const r1 = await store.get(id1);
945
1128
  const r2 = await store.get(id2);
946
1129
  if (!r1) {
@@ -997,6 +1180,577 @@ async function handleDiff(argv) {
997
1180
  }
998
1181
  console.log('');
999
1182
  }
1183
+ /**
1184
+ * Within-report sample-level diff (v0.22). Compares two variants' scores on
1185
+ * each shared sample and surfaces the worst regressions / biggest wins.
1186
+ *
1187
+ * Default focus is variants[0] (control) vs variants[1] (treatment), but
1188
+ * `--variant` overrides which variant is the "treatment" side.
1189
+ */
1190
+ async function runSampleLevelDiff(reportId, store, flags) {
1191
+ const report = await store.get(reportId);
1192
+ if (!report) {
1193
+ console.error(`Report not found: ${reportId}`);
1194
+ process.exit(1);
1195
+ }
1196
+ const variants = report.meta?.variants ?? [];
1197
+ if (variants.length < 2) {
1198
+ console.error('Sample-level diff needs at least 2 variants in the report.');
1199
+ process.exit(1);
1200
+ }
1201
+ const control = variants[0];
1202
+ const treatment = flags.variant ?? variants[1];
1203
+ if (!variants.includes(treatment)) {
1204
+ console.error(`Variant "${treatment}" not in report. Available: ${variants.join(', ')}`);
1205
+ process.exit(1);
1206
+ }
1207
+ const threshold = flags.threshold != null ? Number(flags.threshold) : 0;
1208
+ const regressionsOnly = Boolean(flags['regressions-only']);
1209
+ const topN = flags.top != null ? Math.max(1, Number(flags.top) || 0) : undefined;
1210
+ const rows = [];
1211
+ for (const entry of report.results ?? []) {
1212
+ const c = entry.variants?.[control];
1213
+ const t = entry.variants?.[treatment];
1214
+ if (!c || !t)
1215
+ continue;
1216
+ const cComp = c.compositeScore ?? c.llmScore ?? 0;
1217
+ const tComp = t.compositeScore ?? t.llmScore ?? 0;
1218
+ const delta = Number((tComp - cComp).toFixed(3));
1219
+ rows.push({
1220
+ id: entry.sample_id,
1221
+ cFact: c.layeredScores?.factScore, tFact: t.layeredScores?.factScore,
1222
+ cBeh: c.layeredScores?.behaviorScore, tBeh: t.layeredScores?.behaviorScore,
1223
+ cJudge: c.layeredScores?.judgeScore, tJudge: t.layeredScores?.judgeScore,
1224
+ cComp, tComp, delta,
1225
+ });
1226
+ }
1227
+ // Sort by |delta| desc so the most impactful rows surface first.
1228
+ rows.sort((a, b) => Math.abs(b.delta) - Math.abs(a.delta));
1229
+ let filtered = rows;
1230
+ if (regressionsOnly)
1231
+ filtered = filtered.filter((r) => r.delta < threshold);
1232
+ if (topN !== undefined)
1233
+ filtered = filtered.slice(0, topN);
1234
+ console.log(`\n Sample-level diff: ${treatment} vs ${control} (report ${reportId})`);
1235
+ if (regressionsOnly)
1236
+ console.log(` Filter: regressions only (Δ < ${threshold})`);
1237
+ console.log('');
1238
+ console.log(' sample_id Δ composite (c→t) fact (c→t) behavior (c→t) judge (c→t)');
1239
+ console.log(' ' + '-'.repeat(100));
1240
+ if (filtered.length === 0) {
1241
+ console.log(regressionsOnly ? ' (no regressions found)' : ' (no shared samples)');
1242
+ console.log('');
1243
+ return;
1244
+ }
1245
+ const fmt = (a, b) => {
1246
+ const av = typeof a === 'number' ? a.toFixed(2) : '—';
1247
+ const bv = typeof b === 'number' ? b.toFixed(2) : '—';
1248
+ return `${av} → ${bv}`.padEnd(15);
1249
+ };
1250
+ for (const r of filtered) {
1251
+ const sign = r.delta > 0 ? '+' : '';
1252
+ const idCol = r.id.slice(0, 18).padEnd(20);
1253
+ const deltaCol = `${sign}${r.delta.toFixed(2)}`.padEnd(7);
1254
+ const compCol = `${r.cComp.toFixed(2)} → ${r.tComp.toFixed(2)}`.padEnd(17);
1255
+ console.log(` ${idCol}${deltaCol}${compCol}${fmt(r.cFact, r.tFact)} ${fmt(r.cBeh, r.tBeh)} ${fmt(r.cJudge, r.tJudge)}`);
1256
+ }
1257
+ console.log('');
1258
+ console.log(` Showing ${filtered.length} of ${rows.length} samples · sorted by |Δ|`);
1259
+ if (regressionsOnly) {
1260
+ const total = rows.length;
1261
+ const reg = rows.filter((r) => r.delta < threshold).length;
1262
+ console.log(` Regression rate: ${reg}/${total} samples (${total > 0 ? ((reg / total) * 100).toFixed(0) : 0}%)`);
1263
+ }
1264
+ console.log('');
1265
+ }
1266
+ // ---------------------------------------------------------------------------
1267
+ // handleGold — gold dataset workflow (init / validate / compare)
1268
+ // ---------------------------------------------------------------------------
1269
+ async function handleGold(argv) {
1270
+ const sub = argv[0];
1271
+ const rest = argv.slice(1);
1272
+ if (!sub || sub === '--help' || sub === '-h') {
1273
+ console.log([
1274
+ '',
1275
+ 'Usage: omk bench gold <subcommand>',
1276
+ '',
1277
+ 'Subcommands:',
1278
+ ' init [--out <dir>] [--annotator <id>] 生成空白 gold dataset 模板',
1279
+ ' validate <dir> 校验数据集结构',
1280
+ ' compare <reportId> --gold-dir <dir> 与已有 report 计算 α/κ/Pearson',
1281
+ ' [--variant <name>] [--reports-dir <d>]',
1282
+ ' [--bootstrap-samples N] [--seed N]',
1283
+ '',
1284
+ ].join('\n'));
1285
+ process.exit(sub ? 0 : 1);
1286
+ }
1287
+ if (sub === 'init') {
1288
+ const { values } = parseArgs({
1289
+ args: rest,
1290
+ options: {
1291
+ out: { type: 'string', default: './gold-dataset' },
1292
+ annotator: { type: 'string' },
1293
+ },
1294
+ strict: false,
1295
+ });
1296
+ const { initGoldDataset } = await import('./grading/gold-cli.js');
1297
+ try {
1298
+ const written = initGoldDataset(values.out, {
1299
+ annotator: values.annotator,
1300
+ });
1301
+ console.log(`Created ${written.length} files in ${values.out}:`);
1302
+ for (const p of written)
1303
+ console.log(` ${p}`);
1304
+ console.log('\n下一步: 编辑 annotations.yaml 加入真实标注 → 跑 omk bench gold validate');
1305
+ }
1306
+ catch (err) {
1307
+ console.error(err.message);
1308
+ process.exit(1);
1309
+ }
1310
+ return;
1311
+ }
1312
+ if (sub === 'validate') {
1313
+ const dir = rest[0];
1314
+ if (!dir) {
1315
+ console.error('Usage: omk bench gold validate <dir>');
1316
+ process.exit(1);
1317
+ }
1318
+ const { validateGoldDataset } = await import('./grading/gold-cli.js');
1319
+ const result = validateGoldDataset(dir);
1320
+ if (result.ok) {
1321
+ console.log(`✓ gold dataset OK — ${result.sampleCount} 条标注`);
1322
+ return;
1323
+ }
1324
+ console.error(`✗ gold dataset has ${result.issues.length} issue(s):`);
1325
+ for (const msg of result.issues)
1326
+ console.error(` - ${msg}`);
1327
+ process.exit(1);
1328
+ }
1329
+ if (sub === 'compare') {
1330
+ const reportId = rest[0];
1331
+ if (!reportId) {
1332
+ console.error('Usage: omk bench gold compare <reportId> --gold-dir <dir>');
1333
+ process.exit(1);
1334
+ }
1335
+ const { values } = parseArgs({
1336
+ args: rest.slice(1),
1337
+ options: {
1338
+ 'gold-dir': { type: 'string' },
1339
+ variant: { type: 'string' },
1340
+ 'reports-dir': { type: 'string', default: DEFAULT_REPORTS_DIR },
1341
+ 'bootstrap-samples': { type: 'string', default: '1000' },
1342
+ seed: { type: 'string' },
1343
+ },
1344
+ strict: false,
1345
+ });
1346
+ const goldDir = values['gold-dir'];
1347
+ if (!goldDir) {
1348
+ console.error('--gold-dir is required');
1349
+ process.exit(1);
1350
+ }
1351
+ const { loadGoldDataset } = await import('./grading/gold-dataset.js');
1352
+ const { compareGoldToReport, formatGoldCompare } = await import('./grading/gold-cli.js');
1353
+ const { createFileStore } = await import('./server/report-store.js');
1354
+ const { dataset, issues } = loadGoldDataset(goldDir);
1355
+ if (!dataset) {
1356
+ console.error('Cannot load gold dataset:');
1357
+ for (const i of issues)
1358
+ console.error(` - ${i.message}`);
1359
+ process.exit(1);
1360
+ }
1361
+ if (issues.length) {
1362
+ // Non-fatal issues (e.g. duplicate already filtered) — surface them.
1363
+ for (const i of issues)
1364
+ console.error(`warn: ${i.message}`);
1365
+ }
1366
+ const store = createFileStore(resolve(values['reports-dir']));
1367
+ const report = await store.get(reportId);
1368
+ if (!report) {
1369
+ console.error(`Report not found: ${reportId}`);
1370
+ process.exit(1);
1371
+ }
1372
+ const samples = Math.max(100, Number(values['bootstrap-samples']) || 1000);
1373
+ const seedVal = values.seed != null ? Number(values.seed) : undefined;
1374
+ const result = compareGoldToReport({
1375
+ report: report,
1376
+ gold: dataset,
1377
+ variant: values.variant,
1378
+ samples,
1379
+ seed: Number.isFinite(seedVal) ? seedVal : undefined,
1380
+ });
1381
+ console.log(formatGoldCompare(result, dataset));
1382
+ return;
1383
+ }
1384
+ console.error(`Unknown subcommand: gold ${sub}. Use init / validate / compare.`);
1385
+ process.exit(1);
1386
+ }
1387
+ // ---------------------------------------------------------------------------
1388
+ // handleDebiasValidate — measure length-debias prompt sensitivity (Phase 3a)
1389
+ // ---------------------------------------------------------------------------
1390
+ async function handleDebiasValidate(argv) {
1391
+ const sub = argv[0];
1392
+ const rest = argv.slice(1);
1393
+ if (!sub || sub === '--help' || sub === '-h') {
1394
+ console.log([
1395
+ '',
1396
+ 'Usage: omk bench debias-validate <kind> <reportId> [options]',
1397
+ '',
1398
+ 'Kinds:',
1399
+ ' length re-judge with the opposite length-debias setting and bootstrap CI',
1400
+ ' on the score diff. Cost ~doubles vs the original judge pass.',
1401
+ '',
1402
+ 'Options:',
1403
+ ' --reports-dir <dir> report store dir (default: ~/.oh-my-knowledge/reports)',
1404
+ ' --samples <path> override samples file (default: from report.meta.request)',
1405
+ ' --variant <name> which variant to validate (default: first)',
1406
+ ' --judge-executor <name> executor for judge calls (default: claude)',
1407
+ ' --judge-model <model> judge model id (default: from report)',
1408
+ ' --bootstrap-samples N bootstrap iterations (default 1000)',
1409
+ ' --seed N deterministic CI seed',
1410
+ '',
1411
+ ].join('\n'));
1412
+ process.exit(sub ? 0 : 1);
1413
+ }
1414
+ if (sub !== 'length') {
1415
+ console.error(`Unknown debias-validate kind: ${sub}. Use "length".`);
1416
+ process.exit(1);
1417
+ }
1418
+ const reportId = rest[0];
1419
+ if (!reportId) {
1420
+ console.error('Usage: omk bench debias-validate length <reportId>');
1421
+ process.exit(1);
1422
+ }
1423
+ const { values } = parseArgs({
1424
+ args: rest.slice(1),
1425
+ options: {
1426
+ 'reports-dir': { type: 'string', default: DEFAULT_REPORTS_DIR },
1427
+ samples: { type: 'string' },
1428
+ variant: { type: 'string' },
1429
+ 'judge-executor': { type: 'string', default: 'claude' },
1430
+ 'judge-model': { type: 'string' },
1431
+ 'bootstrap-samples': { type: 'string', default: '1000' },
1432
+ seed: { type: 'string' },
1433
+ },
1434
+ strict: false,
1435
+ });
1436
+ const { createFileStore } = await import('./server/report-store.js');
1437
+ const store = createFileStore(resolve(values['reports-dir']));
1438
+ const report = await store.get(reportId);
1439
+ if (!report) {
1440
+ console.error(`Report not found: ${reportId}`);
1441
+ process.exit(1);
1442
+ }
1443
+ // Resolve samples path: --samples overrides; otherwise read from report.meta.request.
1444
+ const samplesPath = values.samples
1445
+ ?? report.meta?.request?.samplesPath;
1446
+ if (!samplesPath) {
1447
+ console.error('Cannot find samples path. Pass --samples <path> or ensure report has request.samplesPath.');
1448
+ process.exit(1);
1449
+ }
1450
+ const { loadSamples } = await import('./inputs/load-samples.js');
1451
+ const { samples } = loadSamples(samplesPath);
1452
+ const judgeModel = values['judge-model']
1453
+ ?? report.meta?.judgeModel;
1454
+ if (!judgeModel) {
1455
+ console.error('No judge model. Pass --judge-model <id> or ensure report has meta.judgeModel.');
1456
+ process.exit(1);
1457
+ }
1458
+ process.stderr.write('\n⚠ debias-validate 会重判所有 (sample × variant),judge cost 大致翻倍。\n');
1459
+ const { createExecutor } = await import('./executors/index.js');
1460
+ const judgeExecutor = createExecutor(values['judge-executor']);
1461
+ const { validateLengthDebias, formatDebiasValidate } = await import('./grading/debias-validate.js');
1462
+ const seedVal = values.seed != null ? Number(values.seed) : undefined;
1463
+ const bsRaw = Number(values['bootstrap-samples']) || 1000;
1464
+ const result = await validateLengthDebias({
1465
+ report: report,
1466
+ samples,
1467
+ judgeExecutor,
1468
+ judgeModel,
1469
+ variant: values.variant,
1470
+ bootstrapSamples: Math.max(100, bsRaw),
1471
+ seed: Number.isFinite(seedVal) ? seedVal : undefined,
1472
+ onProgress: ({ sample_id, completed, total }) => {
1473
+ process.stderr.write(` judging ${completed}/${total}: ${sample_id}\n`);
1474
+ },
1475
+ });
1476
+ console.log(formatDebiasValidate(result));
1477
+ }
1478
+ // ---------------------------------------------------------------------------
1479
+ // handleSaturation — re-compute saturation verdict from a finished report
1480
+ // ---------------------------------------------------------------------------
1481
+ async function handleSaturation(argv) {
1482
+ const reportId = argv[0];
1483
+ if (!reportId || reportId === '--help' || reportId === '-h') {
1484
+ console.log([
1485
+ '',
1486
+ 'Usage: omk bench saturation <reportId> [options]',
1487
+ '',
1488
+ '回答"我跑够样本了吗?"。复述已有 report 中持久化的饱和判定。',
1489
+ '',
1490
+ '注:本命令读取 run 时跑出的 verdict(运行时已用 method=bootstrap-ci-width',
1491
+ '默认阈值 + 3 窗口 持续条件)。如要换 method/threshold 重新计算,需要重跑',
1492
+ '`omk bench run --repeat ≥ 5`(运行时持久化的 trace 不含原始分数,无法',
1493
+ '在事后用其他参数复算)。',
1494
+ '',
1495
+ 'Options:',
1496
+ ' --reports-dir <dir> report store dir (default: ~/.oh-my-knowledge/reports)',
1497
+ ' --variant <name> 只看一个 variant (default: all)',
1498
+ '',
1499
+ ].join('\n'));
1500
+ process.exit(reportId ? 0 : 1);
1501
+ }
1502
+ const { values } = parseArgs({
1503
+ args: argv.slice(1),
1504
+ options: {
1505
+ 'reports-dir': { type: 'string', default: DEFAULT_REPORTS_DIR },
1506
+ variant: { type: 'string' },
1507
+ },
1508
+ strict: false,
1509
+ });
1510
+ const { createFileStore } = await import('./server/report-store.js');
1511
+ const store = createFileStore(resolve(values['reports-dir']));
1512
+ const report = await store.get(reportId);
1513
+ if (!report) {
1514
+ console.error(`Report not found: ${reportId}`);
1515
+ process.exit(1);
1516
+ }
1517
+ const saturation = report.variance?.saturation;
1518
+ if (!saturation) {
1519
+ console.error('该 report 无 saturation 数据 (需要 --repeat ≥ 2 才会记录)。');
1520
+ process.exit(1);
1521
+ }
1522
+ // Print the persisted verdict from the original run. The trace stores
1523
+ // (mean, ciLow, ciHigh) per checkpoint but not raw scores, so re-running
1524
+ // findSaturationPoint with different method/threshold is not possible
1525
+ // here — that would need raw scores, which would have to be persisted
1526
+ // by runMultiple. Future work: opt-in `--persist-saturation-raw` flag at
1527
+ // run time to enable post-hoc parameter sweeps.
1528
+ const variants = report.meta.variants ?? [];
1529
+ const targetVariants = values.variant ? [values.variant] : variants;
1530
+ console.log(`\n Saturation verdict (复述持久化结果)\n`);
1531
+ for (const variant of targetVariants) {
1532
+ const trace = saturation.perVariant[variant];
1533
+ if (!trace || trace.length === 0) {
1534
+ console.log(` ${variant}: 无 trace 数据`);
1535
+ continue;
1536
+ }
1537
+ console.log(` ${variant}:`);
1538
+ console.log(` checkpoints: ${trace.length} (N=${trace.map((p) => p.n).join(', ')})`);
1539
+ console.log(` 最近一点 mean=${trace[trace.length - 1].mean.toFixed(3)}, CI=[${trace[trace.length - 1].ciLow.toFixed(3)}, ${trace[trace.length - 1].ciHigh.toFixed(3)}]`);
1540
+ if (saturation.verdicts?.[variant]) {
1541
+ const v = saturation.verdicts[variant];
1542
+ console.log(` 持久化判定 (${v.method}): ${v.saturated ? `已饱和@N=${v.atN}` : '未饱和'} - ${v.reason}`);
1543
+ }
1544
+ else if (trace.length < 5) {
1545
+ console.log(` 判定: 数据点 ${trace.length} < 5,跳过 (跑 --repeat 5 以上才输出)`);
1546
+ }
1547
+ }
1548
+ console.log('');
1549
+ }
1550
+ // ---------------------------------------------------------------------------
1551
+ // handleVerdict — one-line ship/no-ship verdict (v0.22)
1552
+ // ---------------------------------------------------------------------------
1553
+ async function handleVerdict(argv) {
1554
+ const reportId = argv[0];
1555
+ if (!reportId || reportId === '--help' || reportId === '-h') {
1556
+ console.log([
1557
+ '',
1558
+ 'Usage: omk bench verdict <reportId> [options]',
1559
+ '',
1560
+ '聚合 bootstrap CI / 三层 ci-gate / saturation / human α 给出一行结论。',
1561
+ '',
1562
+ 'Verdict 等级:',
1563
+ ' PROGRESS 显著改进 + 三层全过',
1564
+ ' CAUTIOUS 改进真实但有警告 (gate 破 / 幅度太小 / 控制组本身崩)',
1565
+ ' REGRESS 显著回退 — 不要 ship',
1566
+ ' NOISE CI 跨 0,无法判定',
1567
+ ' UNDERPOWERED 样本不足,需要扩 N 重测',
1568
+ ' SOLO 单变体报告,无对比对象',
1569
+ '',
1570
+ 'Options:',
1571
+ ' --reports-dir <dir> report store dir (default: ~/.oh-my-knowledge/reports)',
1572
+ ' --threshold <num> 三层 gate 阈值 (default 3.5,匹配 omk bench ci)',
1573
+ ' --trivial-diff <num> "幅度太小"阈值 (default 0.1)',
1574
+ ' --verbose 展开 per-pair 详情',
1575
+ '',
1576
+ ].join('\n'));
1577
+ process.exit(reportId ? 0 : 1);
1578
+ }
1579
+ const { values } = parseArgs({
1580
+ args: argv.slice(1),
1581
+ options: {
1582
+ 'reports-dir': { type: 'string', default: DEFAULT_REPORTS_DIR },
1583
+ threshold: { type: 'string' },
1584
+ 'trivial-diff': { type: 'string' },
1585
+ verbose: { type: 'boolean', default: false },
1586
+ },
1587
+ strict: false,
1588
+ });
1589
+ const { createFileStore } = await import('./server/report-store.js');
1590
+ const store = createFileStore(resolve(values['reports-dir']));
1591
+ const report = await store.get(reportId);
1592
+ if (!report) {
1593
+ console.error(`Report not found: ${reportId}`);
1594
+ process.exit(1);
1595
+ }
1596
+ const { computeVerdict, formatVerdictText } = await import('./eval-core/verdict.js');
1597
+ const result = computeVerdict(report, {
1598
+ ciThreshold: values.threshold != null ? Number(values.threshold) : undefined,
1599
+ triviallySmallDiff: values['trivial-diff'] != null ? Number(values['trivial-diff']) : undefined,
1600
+ });
1601
+ console.log(formatVerdictText(result, { verbose: Boolean(values.verbose) }));
1602
+ // Exit code reflects ship recommendation: 0 only on PROGRESS / SOLO-pass.
1603
+ // NOISE / UNDERPOWERED / CAUTIOUS / REGRESS all exit 1 so this composes
1604
+ // with shell `&&` chains in CI.
1605
+ if (result.level === 'PROGRESS') {
1606
+ process.exit(0);
1607
+ }
1608
+ if (result.level === 'SOLO' && result.headline.includes('PASS')) {
1609
+ process.exit(0);
1610
+ }
1611
+ process.exit(1);
1612
+ }
1613
+ // ---------------------------------------------------------------------------
1614
+ // handleDiagnose — per-sample quality diagnostics (v0.23 A)
1615
+ // ---------------------------------------------------------------------------
1616
+ async function handleDiagnose(argv) {
1617
+ const reportId = argv[0];
1618
+ if (!reportId || reportId === '--help' || reportId === '-h') {
1619
+ console.log([
1620
+ '',
1621
+ 'Usage: omk bench diagnose <reportId> [options]',
1622
+ '',
1623
+ '诊断样本集本身的质量问题:区分度低 / 重复 / 歧义 / 成本异常 / 全 fail。',
1624
+ '回答"测评结论是否被坏样本污染"——与 omk bench verdict 互补。',
1625
+ '',
1626
+ 'Options:',
1627
+ ' --reports-dir <dir> report store dir',
1628
+ ' --samples <path> 样本文件路径 (用于 near-duplicate 检测;默认从 report.meta.request 读)',
1629
+ ' --top <n> 每类只显示前 N 个 (默认 10,0=全部)',
1630
+ ' --duplicate-rouge <num> near-duplicate ROUGE-1 阈值 (默认 0.7)',
1631
+ ' --ambiguous-stddev <num> 歧义阈值,judge stddev (默认 1.0,需要 --judge-repeat ≥ 2 数据)',
1632
+ ' --cost-k <num> 成本异常倍数 vs median (默认 3)',
1633
+ ' --latency-k <num> 耗时异常倍数 vs median (默认 3)',
1634
+ ' --flat <num> flat_scores 分差阈值 (默认 0.5)',
1635
+ '',
1636
+ ].join('\n'));
1637
+ process.exit(reportId ? 0 : 1);
1638
+ }
1639
+ const { values } = parseArgs({
1640
+ args: argv.slice(1),
1641
+ options: {
1642
+ 'reports-dir': { type: 'string', default: DEFAULT_REPORTS_DIR },
1643
+ samples: { type: 'string' },
1644
+ top: { type: 'string', default: '10' },
1645
+ 'duplicate-rouge': { type: 'string' },
1646
+ 'ambiguous-stddev': { type: 'string' },
1647
+ 'cost-k': { type: 'string' },
1648
+ 'latency-k': { type: 'string' },
1649
+ flat: { type: 'string' },
1650
+ },
1651
+ strict: false,
1652
+ });
1653
+ const { createFileStore } = await import('./server/report-store.js');
1654
+ const store = createFileStore(resolve(values['reports-dir']));
1655
+ const report = await store.get(reportId);
1656
+ if (!report) {
1657
+ console.error(`Report not found: ${reportId}`);
1658
+ process.exit(1);
1659
+ }
1660
+ // Try to read the samples file for near-duplicate detection. Source order:
1661
+ // 1. --samples <path> override
1662
+ // 2. report.meta.request.samplesPath (recorded at run time)
1663
+ // If neither resolves to a readable file, skip near-duplicate gracefully.
1664
+ let samples;
1665
+ const samplesPath = values.samples ?? report.meta?.request?.samplesPath;
1666
+ if (samplesPath && existsSync(samplesPath)) {
1667
+ try {
1668
+ const { loadSamples } = await import('./inputs/load-samples.js');
1669
+ samples = loadSamples(samplesPath).samples;
1670
+ }
1671
+ catch (err) {
1672
+ process.stderr.write(`warn: 加载 samples 文件失败 (${samplesPath}): ${err.message}\n`);
1673
+ }
1674
+ }
1675
+ const topRaw = Number(values.top);
1676
+ const topN = Number.isFinite(topRaw) && topRaw > 0 ? topRaw : undefined;
1677
+ const { diagnoseSamples, formatSampleDiagnostics } = await import('./analysis/sample-diagnostics.js');
1678
+ const diag = diagnoseSamples(report, {
1679
+ samples,
1680
+ duplicateRouge: values['duplicate-rouge'] != null ? Number(values['duplicate-rouge']) : undefined,
1681
+ ambiguousStddev: values['ambiguous-stddev'] != null ? Number(values['ambiguous-stddev']) : undefined,
1682
+ costOutlierK: values['cost-k'] != null ? Number(values['cost-k']) : undefined,
1683
+ latencyOutlierK: values['latency-k'] != null ? Number(values['latency-k']) : undefined,
1684
+ flatThreshold: values.flat != null ? Number(values.flat) : undefined,
1685
+ });
1686
+ console.log(formatSampleDiagnostics(diag, { topN }));
1687
+ // Exit code: 0 if health ≥ 70 and no errors; 1 otherwise. CI-friendly.
1688
+ if (diag.totals.errors === 0 && diag.healthScore >= 70) {
1689
+ process.exit(0);
1690
+ }
1691
+ process.exit(1);
1692
+ }
1693
+ // ---------------------------------------------------------------------------
1694
+ // handleFailures — LLM-driven failure clustering (v0.23 B)
1695
+ // ---------------------------------------------------------------------------
1696
+ async function handleFailures(argv) {
1697
+ const reportId = argv[0];
1698
+ if (!reportId || reportId === '--help' || reportId === '-h') {
1699
+ console.log([
1700
+ '',
1701
+ 'Usage: omk bench failures <reportId> [options]',
1702
+ '',
1703
+ '把已有 report 的失败样本喂给一次 LLM 调用,自动聚类 + 给修复建议。',
1704
+ '失败定义:compositeScore < threshold 或 ok=false。',
1705
+ '',
1706
+ 'Options:',
1707
+ ' --reports-dir <dir> report store dir',
1708
+ ' --judge-executor <name> 执行器 (default: claude)',
1709
+ ' --judge-model <id> 聚类用的 model (default: 沿用 report.meta.judgeModel)',
1710
+ ' --max-clusters <n> 最多多少 cluster (default 5)',
1711
+ ' --threshold <num> compositeScore < threshold 算失败 (default 3)',
1712
+ ' --max-feed <n> 最多喂给 LLM 多少条 (default 50,超出取最差)',
1713
+ '',
1714
+ ].join('\n'));
1715
+ process.exit(reportId ? 0 : 1);
1716
+ }
1717
+ const { values } = parseArgs({
1718
+ args: argv.slice(1),
1719
+ options: {
1720
+ 'reports-dir': { type: 'string', default: DEFAULT_REPORTS_DIR },
1721
+ 'judge-executor': { type: 'string', default: 'claude' },
1722
+ 'judge-model': { type: 'string' },
1723
+ 'max-clusters': { type: 'string', default: '5' },
1724
+ threshold: { type: 'string', default: '3' },
1725
+ 'max-feed': { type: 'string', default: '50' },
1726
+ },
1727
+ strict: false,
1728
+ });
1729
+ const { createFileStore } = await import('./server/report-store.js');
1730
+ const store = createFileStore(resolve(values['reports-dir']));
1731
+ const report = await store.get(reportId);
1732
+ if (!report) {
1733
+ console.error(`Report not found: ${reportId}`);
1734
+ process.exit(1);
1735
+ }
1736
+ const judgeModel = values['judge-model'] ?? report.meta?.judgeModel;
1737
+ if (!judgeModel) {
1738
+ console.error('No judge model. Pass --judge-model <id> or ensure report has meta.judgeModel.');
1739
+ process.exit(1);
1740
+ }
1741
+ const { createExecutor } = await import('./executors/index.js');
1742
+ const executor = createExecutor(values['judge-executor']);
1743
+ const { clusterFailures, formatFailureClusterReport } = await import('./analysis/failure-clusterer.js');
1744
+ const out = await clusterFailures({
1745
+ report: report,
1746
+ executor,
1747
+ judgeModel,
1748
+ maxClusters: Number(values['max-clusters']) || 5,
1749
+ failureThreshold: Number(values.threshold) || 3,
1750
+ maxFailuresFed: Number(values['max-feed']) || 50,
1751
+ });
1752
+ console.log(formatFailureClusterReport(out));
1753
+ }
1000
1754
  // ---------------------------------------------------------------------------
1001
1755
  // Entry
1002
1756
  // ---------------------------------------------------------------------------