oh-my-knowledge 0.19.0 → 0.20.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +205 -8
- package/README.zh.md +200 -7
- package/dist/src/analysis/failure-clusterer.d.ts +96 -0
- package/dist/src/analysis/failure-clusterer.d.ts.map +1 -0
- package/dist/src/analysis/failure-clusterer.js +298 -0
- package/dist/src/analysis/failure-clusterer.js.map +1 -0
- package/dist/src/analysis/sample-diagnostics.d.ts +78 -0
- package/dist/src/analysis/sample-diagnostics.d.ts.map +1 -0
- package/dist/src/analysis/sample-diagnostics.js +259 -0
- package/dist/src/analysis/sample-diagnostics.js.map +1 -0
- package/dist/src/analysis/saturation.d.ts +85 -0
- package/dist/src/analysis/saturation.d.ts.map +1 -0
- package/dist/src/analysis/saturation.js +174 -0
- package/dist/src/analysis/saturation.js.map +1 -0
- package/dist/src/cli.js +751 -5
- package/dist/src/cli.js.map +1 -1
- package/dist/src/eval-core/bootstrap.d.ts +72 -0
- package/dist/src/eval-core/bootstrap.d.ts.map +1 -0
- package/dist/src/eval-core/bootstrap.js +174 -0
- package/dist/src/eval-core/bootstrap.js.map +1 -0
- package/dist/src/eval-core/evaluation-execution.d.ts +15 -1
- package/dist/src/eval-core/evaluation-execution.d.ts.map +1 -1
- package/dist/src/eval-core/evaluation-execution.js +37 -3
- package/dist/src/eval-core/evaluation-execution.js.map +1 -1
- package/dist/src/eval-core/evaluation-job.d.ts +8 -2
- package/dist/src/eval-core/evaluation-job.d.ts.map +1 -1
- package/dist/src/eval-core/evaluation-job.js +7 -1
- package/dist/src/eval-core/evaluation-job.js.map +1 -1
- package/dist/src/eval-core/evaluation-reporting.d.ts.map +1 -1
- package/dist/src/eval-core/evaluation-reporting.js +90 -0
- package/dist/src/eval-core/evaluation-reporting.js.map +1 -1
- package/dist/src/eval-core/schema.d.ts.map +1 -1
- package/dist/src/eval-core/schema.js +69 -0
- package/dist/src/eval-core/schema.js.map +1 -1
- package/dist/src/eval-core/verdict.d.ts +74 -0
- package/dist/src/eval-core/verdict.d.ts.map +1 -0
- package/dist/src/eval-core/verdict.js +283 -0
- package/dist/src/eval-core/verdict.js.map +1 -0
- package/dist/src/eval-workflows/each-evaluation-workflow.d.ts +10 -1
- package/dist/src/eval-workflows/each-evaluation-workflow.d.ts.map +1 -1
- package/dist/src/eval-workflows/each-evaluation-workflow.js +4 -1
- package/dist/src/eval-workflows/each-evaluation-workflow.js.map +1 -1
- package/dist/src/eval-workflows/evaluation-pipeline.d.ts +13 -1
- package/dist/src/eval-workflows/evaluation-pipeline.d.ts.map +1 -1
- package/dist/src/eval-workflows/evaluation-pipeline.js +41 -3
- package/dist/src/eval-workflows/evaluation-pipeline.js.map +1 -1
- package/dist/src/eval-workflows/run-evaluation.d.ts +17 -2
- package/dist/src/eval-workflows/run-evaluation.d.ts.map +1 -1
- package/dist/src/eval-workflows/run-evaluation.js +97 -5
- package/dist/src/eval-workflows/run-evaluation.js.map +1 -1
- package/dist/src/grading/assertions.d.ts +16 -0
- package/dist/src/grading/assertions.d.ts.map +1 -1
- package/dist/src/grading/assertions.js +385 -111
- package/dist/src/grading/assertions.js.map +1 -1
- package/dist/src/grading/debias-validate.d.ts +84 -0
- package/dist/src/grading/debias-validate.d.ts.map +1 -0
- package/dist/src/grading/debias-validate.js +173 -0
- package/dist/src/grading/debias-validate.js.map +1 -0
- package/dist/src/grading/gold-cli.d.ts +88 -0
- package/dist/src/grading/gold-cli.d.ts.map +1 -0
- package/dist/src/grading/gold-cli.js +251 -0
- package/dist/src/grading/gold-cli.js.map +1 -0
- package/dist/src/grading/gold-dataset.d.ts +73 -0
- package/dist/src/grading/gold-dataset.d.ts.map +1 -0
- package/dist/src/grading/gold-dataset.js +161 -0
- package/dist/src/grading/gold-dataset.js.map +1 -0
- package/dist/src/grading/human-gold.d.ts +102 -0
- package/dist/src/grading/human-gold.d.ts.map +1 -0
- package/dist/src/grading/human-gold.js +188 -0
- package/dist/src/grading/human-gold.js.map +1 -0
- package/dist/src/grading/index.d.ts +27 -2
- package/dist/src/grading/index.d.ts.map +1 -1
- package/dist/src/grading/index.js +36 -18
- package/dist/src/grading/index.js.map +1 -1
- package/dist/src/grading/judge.d.ts +65 -2
- package/dist/src/grading/judge.d.ts.map +1 -1
- package/dist/src/grading/judge.js +280 -23
- package/dist/src/grading/judge.js.map +1 -1
- package/dist/src/inputs/eval-config.js +19 -0
- package/dist/src/inputs/eval-config.js.map +1 -1
- package/dist/src/renderer/html-renderer.d.ts.map +1 -1
- package/dist/src/renderer/html-renderer.js +16 -4
- package/dist/src/renderer/html-renderer.js.map +1 -1
- package/dist/src/renderer/layout.d.ts.map +1 -1
- package/dist/src/renderer/layout.js +32 -4
- package/dist/src/renderer/layout.js.map +1 -1
- package/dist/src/renderer/summary.d.ts +28 -1
- package/dist/src/renderer/summary.d.ts.map +1 -1
- package/dist/src/renderer/summary.js +321 -7
- package/dist/src/renderer/summary.js.map +1 -1
- package/dist/src/renderer/table.d.ts.map +1 -1
- package/dist/src/renderer/table.js +63 -2
- package/dist/src/renderer/table.js.map +1 -1
- package/dist/src/types.d.ts +241 -0
- package/dist/src/types.d.ts.map +1 -1
- package/package.json +15 -4
package/dist/src/cli.js
CHANGED
|
@@ -167,6 +167,7 @@ function parseRunConfig(argv, extraOptions = {}) {
|
|
|
167
167
|
resume,
|
|
168
168
|
blind,
|
|
169
169
|
layeredStats,
|
|
170
|
+
budget: evalConfig?.budget,
|
|
170
171
|
},
|
|
171
172
|
};
|
|
172
173
|
}
|
|
@@ -214,6 +215,22 @@ Options for "bench run":
|
|
|
214
215
|
--concurrency <n> Number of parallel tasks (default: 1)
|
|
215
216
|
--timeout <seconds> Executor timeout per task in seconds (default: 120)
|
|
216
217
|
--repeat <n> Run evaluation N times for variance analysis (default: 1)
|
|
218
|
+
--judge-repeat <n> Call LLM judge N times per (sample × dimension) for self-
|
|
219
|
+
consistency (default: 1). High stddev across runs = the
|
|
220
|
+
judge is unstable on this rubric and the score is noisy.
|
|
221
|
+
--judge-models <list> Multi-judge ensemble. Comma-separated executor:model pairs,
|
|
222
|
+
e.g. claude:opus,openai:gpt-4o,gemini:pro. Each judge scores
|
|
223
|
+
every (sample × dimension); report includes per-judge break-
|
|
224
|
+
down + Pearson/MAD inter-judge agreement. Refutes "Claude
|
|
225
|
+
judge Claude same-modality bias" critique. Combines with
|
|
226
|
+
--judge-repeat. Cost ~ N_judges × N_repeat × N_samples.
|
|
227
|
+
--bootstrap Compute bootstrap confidence intervals (distribution-free,
|
|
228
|
+
preferred over t-interval for ordinal LLM scores). Adds
|
|
229
|
+
per-variant CI on the mean + pairwise CI on treatment-vs-
|
|
230
|
+
control difference (significant=0 outside CI). Reports both
|
|
231
|
+
t-interval and bootstrap so old tooling still works.
|
|
232
|
+
--bootstrap-samples <n> Number of bootstrap resamples (default 1000). N>10000
|
|
233
|
+
triggers a stderr warning about runtime cost.
|
|
217
234
|
--retry <n> Retry failed tasks up to N times with exponential backoff (default: 0)
|
|
218
235
|
--resume <report-id> Resume from a previous report, skipping completed tasks
|
|
219
236
|
--executor <name> Executor: claude, openai, gemini, anthropic-api, openai-api,
|
|
@@ -360,8 +377,26 @@ async function main() {
|
|
|
360
377
|
case 'diff':
|
|
361
378
|
await handleDiff(rest);
|
|
362
379
|
break;
|
|
380
|
+
case 'gold':
|
|
381
|
+
await handleGold(rest);
|
|
382
|
+
break;
|
|
383
|
+
case 'debias-validate':
|
|
384
|
+
await handleDebiasValidate(rest);
|
|
385
|
+
break;
|
|
386
|
+
case 'saturation':
|
|
387
|
+
await handleSaturation(rest);
|
|
388
|
+
break;
|
|
389
|
+
case 'verdict':
|
|
390
|
+
await handleVerdict(rest);
|
|
391
|
+
break;
|
|
392
|
+
case 'diagnose':
|
|
393
|
+
await handleDiagnose(rest);
|
|
394
|
+
break;
|
|
395
|
+
case 'failures':
|
|
396
|
+
await handleFailures(rest);
|
|
397
|
+
break;
|
|
363
398
|
default:
|
|
364
|
-
console.error(`Unknown command: bench ${command}. Use "run", "report", "ci", "init", "gen-samples", or "
|
|
399
|
+
console.error(`Unknown command: bench ${command}. Use "run", "report", "ci", "init", "gen-samples", "evolve", "diff", "gold", "debias-validate", "saturation", "verdict", "diagnose", or "failures".`);
|
|
365
400
|
process.exit(1);
|
|
366
401
|
}
|
|
367
402
|
}
|
|
@@ -416,6 +451,15 @@ async function handleRun(argv) {
|
|
|
416
451
|
const { values, config } = parseRunConfig(argv, {
|
|
417
452
|
blind: { type: 'boolean', default: false },
|
|
418
453
|
repeat: { type: 'string', default: '1' },
|
|
454
|
+
'judge-repeat': { type: 'string', default: '1' },
|
|
455
|
+
'judge-models': { type: 'string' },
|
|
456
|
+
bootstrap: { type: 'boolean', default: false },
|
|
457
|
+
'bootstrap-samples': { type: 'string', default: '1000' },
|
|
458
|
+
'gold-dir': { type: 'string' },
|
|
459
|
+
'no-debias-length': { type: 'boolean', default: false },
|
|
460
|
+
'budget-usd': { type: 'string' },
|
|
461
|
+
'budget-per-sample-usd': { type: 'string' },
|
|
462
|
+
'budget-per-sample-ms': { type: 'string' },
|
|
419
463
|
});
|
|
420
464
|
const { runEvaluation, runMultiple, runEachEvaluation } = await import('./eval-workflows/run-evaluation.js');
|
|
421
465
|
config.blind = values.blind;
|
|
@@ -428,6 +472,71 @@ async function handleRun(argv) {
|
|
|
428
472
|
process.stderr.write(`⚠ --repeat "${repeatRaw}" 无效(期望 ≥ 1 的整数),已按 1 次评测执行\n`);
|
|
429
473
|
}
|
|
430
474
|
const repeatCount = Math.max(1, Math.floor(parsedRepeat) || 1);
|
|
475
|
+
// --judge-repeat 同样的诚实校验:非 ≥1 整数时钳到 1
|
|
476
|
+
const judgeRepeatRaw = values['judge-repeat'];
|
|
477
|
+
const parsedJudgeRepeat = judgeRepeatRaw !== undefined ? Number(judgeRepeatRaw) : 1;
|
|
478
|
+
if (judgeRepeatRaw !== undefined && (!Number.isFinite(parsedJudgeRepeat) || parsedJudgeRepeat < 1)) {
|
|
479
|
+
process.stderr.write(`⚠ --judge-repeat "${judgeRepeatRaw}" 无效(期望 ≥ 1 的整数),已按 1 次 judge 执行\n`);
|
|
480
|
+
}
|
|
481
|
+
const judgeRepeatCount = Math.max(1, Math.floor(parsedJudgeRepeat) || 1);
|
|
482
|
+
if (judgeRepeatCount > 1)
|
|
483
|
+
config.judgeRepeat = judgeRepeatCount;
|
|
484
|
+
// --judge-models executor:model,executor:model,... -> JudgeConfig[]
|
|
485
|
+
// 至少 2 个才进 ensemble 模式,1 个等同于 --judge-model
|
|
486
|
+
const judgeModelsRaw = values['judge-models'];
|
|
487
|
+
if (judgeModelsRaw) {
|
|
488
|
+
const parts = judgeModelsRaw.split(',').map((s) => s.trim()).filter(Boolean);
|
|
489
|
+
const judges = parts.map((p) => {
|
|
490
|
+
const [executor, ...modelParts] = p.split(':');
|
|
491
|
+
const model = modelParts.join(':');
|
|
492
|
+
if (!executor || !model) {
|
|
493
|
+
throw new Error(`--judge-models 格式错误: "${p}",应为 "executor:model" (如 claude:opus)`);
|
|
494
|
+
}
|
|
495
|
+
return { executor, model };
|
|
496
|
+
});
|
|
497
|
+
if (judges.length >= 2) {
|
|
498
|
+
config.judgeModels = judges;
|
|
499
|
+
}
|
|
500
|
+
else if (judges.length === 1) {
|
|
501
|
+
// 单 judge 不走 ensemble,但允许这样写,等同于 --judge-model + --executor
|
|
502
|
+
process.stderr.write(`ℹ --judge-models 只指定 1 个 judge (${judges[0].executor}:${judges[0].model}),不触发 ensemble。如需 ensemble 至少给 2 个。\n`);
|
|
503
|
+
}
|
|
504
|
+
}
|
|
505
|
+
// --budget-usd / --budget-per-sample-usd / --budget-per-sample-ms:
|
|
506
|
+
// v0.22 hard budget caps. CLI flags override config-file values. When the
|
|
507
|
+
// total-USD cap is exceeded mid-run, remaining tasks are skipped and a
|
|
508
|
+
// partial report is persisted with meta.budgetExhausted=true.
|
|
509
|
+
const budgetUSD = values['budget-usd'] != null ? Number(values['budget-usd']) : undefined;
|
|
510
|
+
const budgetPerSampleUSD = values['budget-per-sample-usd'] != null ? Number(values['budget-per-sample-usd']) : undefined;
|
|
511
|
+
const budgetPerSampleMs = values['budget-per-sample-ms'] != null ? Number(values['budget-per-sample-ms']) : undefined;
|
|
512
|
+
if (budgetUSD !== undefined || budgetPerSampleUSD !== undefined || budgetPerSampleMs !== undefined) {
|
|
513
|
+
config.budget = {
|
|
514
|
+
...(budgetUSD !== undefined && Number.isFinite(budgetUSD) && budgetUSD >= 0 ? { totalUSD: budgetUSD } : {}),
|
|
515
|
+
...(budgetPerSampleUSD !== undefined && Number.isFinite(budgetPerSampleUSD) && budgetPerSampleUSD >= 0 ? { perSampleUSD: budgetPerSampleUSD } : {}),
|
|
516
|
+
...(budgetPerSampleMs !== undefined && Number.isFinite(budgetPerSampleMs) && budgetPerSampleMs >= 0 ? { perSampleMs: budgetPerSampleMs } : {}),
|
|
517
|
+
};
|
|
518
|
+
}
|
|
519
|
+
// --no-debias-length: opt out of v0.21 Phase 3a length-controlled prompt.
|
|
520
|
+
// Default behavior is debias-on (judge prompt v3-cot-length); flag flips it
|
|
521
|
+
// off so historical reports (judgePromptHash from v2-cot era) can be reproduced.
|
|
522
|
+
if (values['no-debias-length']) {
|
|
523
|
+
config.lengthDebias = false;
|
|
524
|
+
process.stderr.write('ℹ --no-debias-length 已生效:judge prompt 退回 v2-cot,与 < v0.21 报告 hash 一致。\n');
|
|
525
|
+
}
|
|
526
|
+
// --bootstrap / --bootstrap-samples
|
|
527
|
+
if (values.bootstrap) {
|
|
528
|
+
config.bootstrap = true;
|
|
529
|
+
const bsRaw = values['bootstrap-samples'];
|
|
530
|
+
const parsedBs = bsRaw !== undefined ? Number(bsRaw) : 1000;
|
|
531
|
+
if (bsRaw !== undefined && (!Number.isFinite(parsedBs) || parsedBs < 100)) {
|
|
532
|
+
process.stderr.write(`⚠ --bootstrap-samples "${bsRaw}" 无效(期望 ≥ 100 的整数),已按 1000 执行\n`);
|
|
533
|
+
}
|
|
534
|
+
const bsCount = Math.max(100, Math.floor(parsedBs) || 1000);
|
|
535
|
+
if (bsCount > 10000) {
|
|
536
|
+
process.stderr.write(`⚠ --bootstrap-samples ${bsCount} 较大,可能耗时数秒。1000 是业内标准,通常已够用。\n`);
|
|
537
|
+
}
|
|
538
|
+
config.bootstrapSamples = bsCount;
|
|
539
|
+
}
|
|
431
540
|
try {
|
|
432
541
|
// --each mode: evaluate each skill independently
|
|
433
542
|
if (values.each) {
|
|
@@ -482,6 +591,28 @@ async function handleRun(argv) {
|
|
|
482
591
|
report = result.report;
|
|
483
592
|
filePath = result.filePath;
|
|
484
593
|
}
|
|
594
|
+
// --gold-dir: compute α/κ/Pearson against gold annotations and re-persist.
|
|
595
|
+
const goldDir = values['gold-dir'];
|
|
596
|
+
if (goldDir && filePath) {
|
|
597
|
+
const { attachGoldAgreementToReport, formatGoldCompare } = await import('./grading/gold-cli.js');
|
|
598
|
+
const out = attachGoldAgreementToReport({
|
|
599
|
+
report,
|
|
600
|
+
goldDir,
|
|
601
|
+
outputDir: config.outputDir,
|
|
602
|
+
samples: config.bootstrapSamples,
|
|
603
|
+
});
|
|
604
|
+
if (out.result && out.gold) {
|
|
605
|
+
process.stderr.write(formatGoldCompare(out.result, out.gold));
|
|
606
|
+
if (out.result.contaminationWarning) {
|
|
607
|
+
process.stderr.write(`\n⚠ ${out.result.contaminationWarning}\n`);
|
|
608
|
+
}
|
|
609
|
+
}
|
|
610
|
+
else {
|
|
611
|
+
process.stderr.write(`\n⚠ gold dataset 加载失败 (${goldDir}):\n`);
|
|
612
|
+
for (const m of out.loadIssues)
|
|
613
|
+
process.stderr.write(` - ${m}\n`);
|
|
614
|
+
}
|
|
615
|
+
}
|
|
485
616
|
console.log(JSON.stringify(report, null, 2));
|
|
486
617
|
if (filePath) {
|
|
487
618
|
process.stderr.write('\n✅ 评测完成\n');
|
|
@@ -942,13 +1073,57 @@ async function handleCi(argv) {
|
|
|
942
1073
|
// handleDiff
|
|
943
1074
|
// ---------------------------------------------------------------------------
|
|
944
1075
|
async function handleDiff(argv) {
|
|
945
|
-
|
|
946
|
-
|
|
947
|
-
|
|
1076
|
+
// Flag-aware split: separate positional report IDs from flags so we can support
|
|
1077
|
+
// omk bench diff <id> — within-report sample-level (v0.22)
|
|
1078
|
+
// omk bench diff <id1> <id2> — cross-report variant-level (legacy)
|
|
1079
|
+
// both with optional --regressions-only / --threshold / --variant flags.
|
|
1080
|
+
const positional = [];
|
|
1081
|
+
const flagArgs = [];
|
|
1082
|
+
for (let i = 0; i < argv.length; i++) {
|
|
1083
|
+
const a = argv[i];
|
|
1084
|
+
if (a.startsWith('--')) {
|
|
1085
|
+
flagArgs.push(a);
|
|
1086
|
+
const next = argv[i + 1];
|
|
1087
|
+
if (next !== undefined && !next.startsWith('--')) {
|
|
1088
|
+
flagArgs.push(next);
|
|
1089
|
+
i++;
|
|
1090
|
+
}
|
|
1091
|
+
}
|
|
1092
|
+
else {
|
|
1093
|
+
positional.push(a);
|
|
1094
|
+
}
|
|
1095
|
+
}
|
|
1096
|
+
if (positional.length === 0) {
|
|
1097
|
+
console.error([
|
|
1098
|
+
'Usage:',
|
|
1099
|
+
' omk bench diff <reportId> within-report per-sample diff (v0.22)',
|
|
1100
|
+
' omk bench diff <reportId1> <reportId2> cross-report variant-level diff',
|
|
1101
|
+
'',
|
|
1102
|
+
'Options:',
|
|
1103
|
+
' --regressions-only 只列 treatment < control 的样本',
|
|
1104
|
+
' --threshold <num> regression 阈值 (default 0,即任一负 Δ 算回退)',
|
|
1105
|
+
' --variant <name> within-report 模式下指定要钻取的 variant (default: variants[1])',
|
|
1106
|
+
' --top <n> 只列差距最大的前 N 个样本',
|
|
1107
|
+
].join('\n'));
|
|
1108
|
+
process.exit(positional.length === 0 ? 1 : 0);
|
|
948
1109
|
}
|
|
1110
|
+
const { values } = parseArgs({
|
|
1111
|
+
args: flagArgs,
|
|
1112
|
+
options: {
|
|
1113
|
+
'regressions-only': { type: 'boolean', default: false },
|
|
1114
|
+
threshold: { type: 'string' },
|
|
1115
|
+
variant: { type: 'string' },
|
|
1116
|
+
top: { type: 'string' },
|
|
1117
|
+
},
|
|
1118
|
+
strict: false,
|
|
1119
|
+
});
|
|
949
1120
|
const { createFileStore } = await import('./server/report-store.js');
|
|
950
1121
|
const store = createFileStore(resolve(DEFAULT_REPORTS_DIR));
|
|
951
|
-
|
|
1122
|
+
if (positional.length === 1) {
|
|
1123
|
+
await runSampleLevelDiff(positional[0], store, values);
|
|
1124
|
+
return;
|
|
1125
|
+
}
|
|
1126
|
+
const [id1, id2] = positional;
|
|
952
1127
|
const r1 = await store.get(id1);
|
|
953
1128
|
const r2 = await store.get(id2);
|
|
954
1129
|
if (!r1) {
|
|
@@ -1005,6 +1180,577 @@ async function handleDiff(argv) {
|
|
|
1005
1180
|
}
|
|
1006
1181
|
console.log('');
|
|
1007
1182
|
}
|
|
1183
|
+
/**
|
|
1184
|
+
* Within-report sample-level diff (v0.22). Compares two variants' scores on
|
|
1185
|
+
* each shared sample and surfaces the worst regressions / biggest wins.
|
|
1186
|
+
*
|
|
1187
|
+
* Default focus is variants[0] (control) vs variants[1] (treatment), but
|
|
1188
|
+
* `--variant` overrides which variant is the "treatment" side.
|
|
1189
|
+
*/
|
|
1190
|
+
async function runSampleLevelDiff(reportId, store, flags) {
|
|
1191
|
+
const report = await store.get(reportId);
|
|
1192
|
+
if (!report) {
|
|
1193
|
+
console.error(`Report not found: ${reportId}`);
|
|
1194
|
+
process.exit(1);
|
|
1195
|
+
}
|
|
1196
|
+
const variants = report.meta?.variants ?? [];
|
|
1197
|
+
if (variants.length < 2) {
|
|
1198
|
+
console.error('Sample-level diff needs at least 2 variants in the report.');
|
|
1199
|
+
process.exit(1);
|
|
1200
|
+
}
|
|
1201
|
+
const control = variants[0];
|
|
1202
|
+
const treatment = flags.variant ?? variants[1];
|
|
1203
|
+
if (!variants.includes(treatment)) {
|
|
1204
|
+
console.error(`Variant "${treatment}" not in report. Available: ${variants.join(', ')}`);
|
|
1205
|
+
process.exit(1);
|
|
1206
|
+
}
|
|
1207
|
+
const threshold = flags.threshold != null ? Number(flags.threshold) : 0;
|
|
1208
|
+
const regressionsOnly = Boolean(flags['regressions-only']);
|
|
1209
|
+
const topN = flags.top != null ? Math.max(1, Number(flags.top) || 0) : undefined;
|
|
1210
|
+
const rows = [];
|
|
1211
|
+
for (const entry of report.results ?? []) {
|
|
1212
|
+
const c = entry.variants?.[control];
|
|
1213
|
+
const t = entry.variants?.[treatment];
|
|
1214
|
+
if (!c || !t)
|
|
1215
|
+
continue;
|
|
1216
|
+
const cComp = c.compositeScore ?? c.llmScore ?? 0;
|
|
1217
|
+
const tComp = t.compositeScore ?? t.llmScore ?? 0;
|
|
1218
|
+
const delta = Number((tComp - cComp).toFixed(3));
|
|
1219
|
+
rows.push({
|
|
1220
|
+
id: entry.sample_id,
|
|
1221
|
+
cFact: c.layeredScores?.factScore, tFact: t.layeredScores?.factScore,
|
|
1222
|
+
cBeh: c.layeredScores?.behaviorScore, tBeh: t.layeredScores?.behaviorScore,
|
|
1223
|
+
cJudge: c.layeredScores?.judgeScore, tJudge: t.layeredScores?.judgeScore,
|
|
1224
|
+
cComp, tComp, delta,
|
|
1225
|
+
});
|
|
1226
|
+
}
|
|
1227
|
+
// Sort by |delta| desc so the most impactful rows surface first.
|
|
1228
|
+
rows.sort((a, b) => Math.abs(b.delta) - Math.abs(a.delta));
|
|
1229
|
+
let filtered = rows;
|
|
1230
|
+
if (regressionsOnly)
|
|
1231
|
+
filtered = filtered.filter((r) => r.delta < threshold);
|
|
1232
|
+
if (topN !== undefined)
|
|
1233
|
+
filtered = filtered.slice(0, topN);
|
|
1234
|
+
console.log(`\n Sample-level diff: ${treatment} vs ${control} (report ${reportId})`);
|
|
1235
|
+
if (regressionsOnly)
|
|
1236
|
+
console.log(` Filter: regressions only (Δ < ${threshold})`);
|
|
1237
|
+
console.log('');
|
|
1238
|
+
console.log(' sample_id Δ composite (c→t) fact (c→t) behavior (c→t) judge (c→t)');
|
|
1239
|
+
console.log(' ' + '-'.repeat(100));
|
|
1240
|
+
if (filtered.length === 0) {
|
|
1241
|
+
console.log(regressionsOnly ? ' (no regressions found)' : ' (no shared samples)');
|
|
1242
|
+
console.log('');
|
|
1243
|
+
return;
|
|
1244
|
+
}
|
|
1245
|
+
const fmt = (a, b) => {
|
|
1246
|
+
const av = typeof a === 'number' ? a.toFixed(2) : '—';
|
|
1247
|
+
const bv = typeof b === 'number' ? b.toFixed(2) : '—';
|
|
1248
|
+
return `${av} → ${bv}`.padEnd(15);
|
|
1249
|
+
};
|
|
1250
|
+
for (const r of filtered) {
|
|
1251
|
+
const sign = r.delta > 0 ? '+' : '';
|
|
1252
|
+
const idCol = r.id.slice(0, 18).padEnd(20);
|
|
1253
|
+
const deltaCol = `${sign}${r.delta.toFixed(2)}`.padEnd(7);
|
|
1254
|
+
const compCol = `${r.cComp.toFixed(2)} → ${r.tComp.toFixed(2)}`.padEnd(17);
|
|
1255
|
+
console.log(` ${idCol}${deltaCol}${compCol}${fmt(r.cFact, r.tFact)} ${fmt(r.cBeh, r.tBeh)} ${fmt(r.cJudge, r.tJudge)}`);
|
|
1256
|
+
}
|
|
1257
|
+
console.log('');
|
|
1258
|
+
console.log(` Showing ${filtered.length} of ${rows.length} samples · sorted by |Δ|`);
|
|
1259
|
+
if (regressionsOnly) {
|
|
1260
|
+
const total = rows.length;
|
|
1261
|
+
const reg = rows.filter((r) => r.delta < threshold).length;
|
|
1262
|
+
console.log(` Regression rate: ${reg}/${total} samples (${total > 0 ? ((reg / total) * 100).toFixed(0) : 0}%)`);
|
|
1263
|
+
}
|
|
1264
|
+
console.log('');
|
|
1265
|
+
}
|
|
1266
|
+
// ---------------------------------------------------------------------------
|
|
1267
|
+
// handleGold — gold dataset workflow (init / validate / compare)
|
|
1268
|
+
// ---------------------------------------------------------------------------
|
|
1269
|
+
async function handleGold(argv) {
|
|
1270
|
+
const sub = argv[0];
|
|
1271
|
+
const rest = argv.slice(1);
|
|
1272
|
+
if (!sub || sub === '--help' || sub === '-h') {
|
|
1273
|
+
console.log([
|
|
1274
|
+
'',
|
|
1275
|
+
'Usage: omk bench gold <subcommand>',
|
|
1276
|
+
'',
|
|
1277
|
+
'Subcommands:',
|
|
1278
|
+
' init [--out <dir>] [--annotator <id>] 生成空白 gold dataset 模板',
|
|
1279
|
+
' validate <dir> 校验数据集结构',
|
|
1280
|
+
' compare <reportId> --gold-dir <dir> 与已有 report 计算 α/κ/Pearson',
|
|
1281
|
+
' [--variant <name>] [--reports-dir <d>]',
|
|
1282
|
+
' [--bootstrap-samples N] [--seed N]',
|
|
1283
|
+
'',
|
|
1284
|
+
].join('\n'));
|
|
1285
|
+
process.exit(sub ? 0 : 1);
|
|
1286
|
+
}
|
|
1287
|
+
if (sub === 'init') {
|
|
1288
|
+
const { values } = parseArgs({
|
|
1289
|
+
args: rest,
|
|
1290
|
+
options: {
|
|
1291
|
+
out: { type: 'string', default: './gold-dataset' },
|
|
1292
|
+
annotator: { type: 'string' },
|
|
1293
|
+
},
|
|
1294
|
+
strict: false,
|
|
1295
|
+
});
|
|
1296
|
+
const { initGoldDataset } = await import('./grading/gold-cli.js');
|
|
1297
|
+
try {
|
|
1298
|
+
const written = initGoldDataset(values.out, {
|
|
1299
|
+
annotator: values.annotator,
|
|
1300
|
+
});
|
|
1301
|
+
console.log(`Created ${written.length} files in ${values.out}:`);
|
|
1302
|
+
for (const p of written)
|
|
1303
|
+
console.log(` ${p}`);
|
|
1304
|
+
console.log('\n下一步: 编辑 annotations.yaml 加入真实标注 → 跑 omk bench gold validate');
|
|
1305
|
+
}
|
|
1306
|
+
catch (err) {
|
|
1307
|
+
console.error(err.message);
|
|
1308
|
+
process.exit(1);
|
|
1309
|
+
}
|
|
1310
|
+
return;
|
|
1311
|
+
}
|
|
1312
|
+
if (sub === 'validate') {
|
|
1313
|
+
const dir = rest[0];
|
|
1314
|
+
if (!dir) {
|
|
1315
|
+
console.error('Usage: omk bench gold validate <dir>');
|
|
1316
|
+
process.exit(1);
|
|
1317
|
+
}
|
|
1318
|
+
const { validateGoldDataset } = await import('./grading/gold-cli.js');
|
|
1319
|
+
const result = validateGoldDataset(dir);
|
|
1320
|
+
if (result.ok) {
|
|
1321
|
+
console.log(`✓ gold dataset OK — ${result.sampleCount} 条标注`);
|
|
1322
|
+
return;
|
|
1323
|
+
}
|
|
1324
|
+
console.error(`✗ gold dataset has ${result.issues.length} issue(s):`);
|
|
1325
|
+
for (const msg of result.issues)
|
|
1326
|
+
console.error(` - ${msg}`);
|
|
1327
|
+
process.exit(1);
|
|
1328
|
+
}
|
|
1329
|
+
if (sub === 'compare') {
|
|
1330
|
+
const reportId = rest[0];
|
|
1331
|
+
if (!reportId) {
|
|
1332
|
+
console.error('Usage: omk bench gold compare <reportId> --gold-dir <dir>');
|
|
1333
|
+
process.exit(1);
|
|
1334
|
+
}
|
|
1335
|
+
const { values } = parseArgs({
|
|
1336
|
+
args: rest.slice(1),
|
|
1337
|
+
options: {
|
|
1338
|
+
'gold-dir': { type: 'string' },
|
|
1339
|
+
variant: { type: 'string' },
|
|
1340
|
+
'reports-dir': { type: 'string', default: DEFAULT_REPORTS_DIR },
|
|
1341
|
+
'bootstrap-samples': { type: 'string', default: '1000' },
|
|
1342
|
+
seed: { type: 'string' },
|
|
1343
|
+
},
|
|
1344
|
+
strict: false,
|
|
1345
|
+
});
|
|
1346
|
+
const goldDir = values['gold-dir'];
|
|
1347
|
+
if (!goldDir) {
|
|
1348
|
+
console.error('--gold-dir is required');
|
|
1349
|
+
process.exit(1);
|
|
1350
|
+
}
|
|
1351
|
+
const { loadGoldDataset } = await import('./grading/gold-dataset.js');
|
|
1352
|
+
const { compareGoldToReport, formatGoldCompare } = await import('./grading/gold-cli.js');
|
|
1353
|
+
const { createFileStore } = await import('./server/report-store.js');
|
|
1354
|
+
const { dataset, issues } = loadGoldDataset(goldDir);
|
|
1355
|
+
if (!dataset) {
|
|
1356
|
+
console.error('Cannot load gold dataset:');
|
|
1357
|
+
for (const i of issues)
|
|
1358
|
+
console.error(` - ${i.message}`);
|
|
1359
|
+
process.exit(1);
|
|
1360
|
+
}
|
|
1361
|
+
if (issues.length) {
|
|
1362
|
+
// Non-fatal issues (e.g. duplicate already filtered) — surface them.
|
|
1363
|
+
for (const i of issues)
|
|
1364
|
+
console.error(`warn: ${i.message}`);
|
|
1365
|
+
}
|
|
1366
|
+
const store = createFileStore(resolve(values['reports-dir']));
|
|
1367
|
+
const report = await store.get(reportId);
|
|
1368
|
+
if (!report) {
|
|
1369
|
+
console.error(`Report not found: ${reportId}`);
|
|
1370
|
+
process.exit(1);
|
|
1371
|
+
}
|
|
1372
|
+
const samples = Math.max(100, Number(values['bootstrap-samples']) || 1000);
|
|
1373
|
+
const seedVal = values.seed != null ? Number(values.seed) : undefined;
|
|
1374
|
+
const result = compareGoldToReport({
|
|
1375
|
+
report: report,
|
|
1376
|
+
gold: dataset,
|
|
1377
|
+
variant: values.variant,
|
|
1378
|
+
samples,
|
|
1379
|
+
seed: Number.isFinite(seedVal) ? seedVal : undefined,
|
|
1380
|
+
});
|
|
1381
|
+
console.log(formatGoldCompare(result, dataset));
|
|
1382
|
+
return;
|
|
1383
|
+
}
|
|
1384
|
+
console.error(`Unknown subcommand: gold ${sub}. Use init / validate / compare.`);
|
|
1385
|
+
process.exit(1);
|
|
1386
|
+
}
|
|
1387
|
+
// ---------------------------------------------------------------------------
|
|
1388
|
+
// handleDebiasValidate — measure length-debias prompt sensitivity (Phase 3a)
|
|
1389
|
+
// ---------------------------------------------------------------------------
|
|
1390
|
+
async function handleDebiasValidate(argv) {
|
|
1391
|
+
const sub = argv[0];
|
|
1392
|
+
const rest = argv.slice(1);
|
|
1393
|
+
if (!sub || sub === '--help' || sub === '-h') {
|
|
1394
|
+
console.log([
|
|
1395
|
+
'',
|
|
1396
|
+
'Usage: omk bench debias-validate <kind> <reportId> [options]',
|
|
1397
|
+
'',
|
|
1398
|
+
'Kinds:',
|
|
1399
|
+
' length re-judge with the opposite length-debias setting and bootstrap CI',
|
|
1400
|
+
' on the score diff. Cost ~doubles vs the original judge pass.',
|
|
1401
|
+
'',
|
|
1402
|
+
'Options:',
|
|
1403
|
+
' --reports-dir <dir> report store dir (default: ~/.oh-my-knowledge/reports)',
|
|
1404
|
+
' --samples <path> override samples file (default: from report.meta.request)',
|
|
1405
|
+
' --variant <name> which variant to validate (default: first)',
|
|
1406
|
+
' --judge-executor <name> executor for judge calls (default: claude)',
|
|
1407
|
+
' --judge-model <model> judge model id (default: from report)',
|
|
1408
|
+
' --bootstrap-samples N bootstrap iterations (default 1000)',
|
|
1409
|
+
' --seed N deterministic CI seed',
|
|
1410
|
+
'',
|
|
1411
|
+
].join('\n'));
|
|
1412
|
+
process.exit(sub ? 0 : 1);
|
|
1413
|
+
}
|
|
1414
|
+
if (sub !== 'length') {
|
|
1415
|
+
console.error(`Unknown debias-validate kind: ${sub}. Use "length".`);
|
|
1416
|
+
process.exit(1);
|
|
1417
|
+
}
|
|
1418
|
+
const reportId = rest[0];
|
|
1419
|
+
if (!reportId) {
|
|
1420
|
+
console.error('Usage: omk bench debias-validate length <reportId>');
|
|
1421
|
+
process.exit(1);
|
|
1422
|
+
}
|
|
1423
|
+
const { values } = parseArgs({
|
|
1424
|
+
args: rest.slice(1),
|
|
1425
|
+
options: {
|
|
1426
|
+
'reports-dir': { type: 'string', default: DEFAULT_REPORTS_DIR },
|
|
1427
|
+
samples: { type: 'string' },
|
|
1428
|
+
variant: { type: 'string' },
|
|
1429
|
+
'judge-executor': { type: 'string', default: 'claude' },
|
|
1430
|
+
'judge-model': { type: 'string' },
|
|
1431
|
+
'bootstrap-samples': { type: 'string', default: '1000' },
|
|
1432
|
+
seed: { type: 'string' },
|
|
1433
|
+
},
|
|
1434
|
+
strict: false,
|
|
1435
|
+
});
|
|
1436
|
+
const { createFileStore } = await import('./server/report-store.js');
|
|
1437
|
+
const store = createFileStore(resolve(values['reports-dir']));
|
|
1438
|
+
const report = await store.get(reportId);
|
|
1439
|
+
if (!report) {
|
|
1440
|
+
console.error(`Report not found: ${reportId}`);
|
|
1441
|
+
process.exit(1);
|
|
1442
|
+
}
|
|
1443
|
+
// Resolve samples path: --samples overrides; otherwise read from report.meta.request.
|
|
1444
|
+
const samplesPath = values.samples
|
|
1445
|
+
?? report.meta?.request?.samplesPath;
|
|
1446
|
+
if (!samplesPath) {
|
|
1447
|
+
console.error('Cannot find samples path. Pass --samples <path> or ensure report has request.samplesPath.');
|
|
1448
|
+
process.exit(1);
|
|
1449
|
+
}
|
|
1450
|
+
const { loadSamples } = await import('./inputs/load-samples.js');
|
|
1451
|
+
const { samples } = loadSamples(samplesPath);
|
|
1452
|
+
const judgeModel = values['judge-model']
|
|
1453
|
+
?? report.meta?.judgeModel;
|
|
1454
|
+
if (!judgeModel) {
|
|
1455
|
+
console.error('No judge model. Pass --judge-model <id> or ensure report has meta.judgeModel.');
|
|
1456
|
+
process.exit(1);
|
|
1457
|
+
}
|
|
1458
|
+
process.stderr.write('\n⚠ debias-validate 会重判所有 (sample × variant),judge cost 大致翻倍。\n');
|
|
1459
|
+
const { createExecutor } = await import('./executors/index.js');
|
|
1460
|
+
const judgeExecutor = createExecutor(values['judge-executor']);
|
|
1461
|
+
const { validateLengthDebias, formatDebiasValidate } = await import('./grading/debias-validate.js');
|
|
1462
|
+
const seedVal = values.seed != null ? Number(values.seed) : undefined;
|
|
1463
|
+
const bsRaw = Number(values['bootstrap-samples']) || 1000;
|
|
1464
|
+
const result = await validateLengthDebias({
|
|
1465
|
+
report: report,
|
|
1466
|
+
samples,
|
|
1467
|
+
judgeExecutor,
|
|
1468
|
+
judgeModel,
|
|
1469
|
+
variant: values.variant,
|
|
1470
|
+
bootstrapSamples: Math.max(100, bsRaw),
|
|
1471
|
+
seed: Number.isFinite(seedVal) ? seedVal : undefined,
|
|
1472
|
+
onProgress: ({ sample_id, completed, total }) => {
|
|
1473
|
+
process.stderr.write(` judging ${completed}/${total}: ${sample_id}\n`);
|
|
1474
|
+
},
|
|
1475
|
+
});
|
|
1476
|
+
console.log(formatDebiasValidate(result));
|
|
1477
|
+
}
|
|
1478
|
+
// ---------------------------------------------------------------------------
|
|
1479
|
+
// handleSaturation — re-compute saturation verdict from a finished report
|
|
1480
|
+
// ---------------------------------------------------------------------------
|
|
1481
|
+
async function handleSaturation(argv) {
|
|
1482
|
+
const reportId = argv[0];
|
|
1483
|
+
if (!reportId || reportId === '--help' || reportId === '-h') {
|
|
1484
|
+
console.log([
|
|
1485
|
+
'',
|
|
1486
|
+
'Usage: omk bench saturation <reportId> [options]',
|
|
1487
|
+
'',
|
|
1488
|
+
'回答"我跑够样本了吗?"。复述已有 report 中持久化的饱和判定。',
|
|
1489
|
+
'',
|
|
1490
|
+
'注:本命令读取 run 时跑出的 verdict(运行时已用 method=bootstrap-ci-width',
|
|
1491
|
+
'默认阈值 + 3 窗口 持续条件)。如要换 method/threshold 重新计算,需要重跑',
|
|
1492
|
+
'`omk bench run --repeat ≥ 5`(运行时持久化的 trace 不含原始分数,无法',
|
|
1493
|
+
'在事后用其他参数复算)。',
|
|
1494
|
+
'',
|
|
1495
|
+
'Options:',
|
|
1496
|
+
' --reports-dir <dir> report store dir (default: ~/.oh-my-knowledge/reports)',
|
|
1497
|
+
' --variant <name> 只看一个 variant (default: all)',
|
|
1498
|
+
'',
|
|
1499
|
+
].join('\n'));
|
|
1500
|
+
process.exit(reportId ? 0 : 1);
|
|
1501
|
+
}
|
|
1502
|
+
const { values } = parseArgs({
|
|
1503
|
+
args: argv.slice(1),
|
|
1504
|
+
options: {
|
|
1505
|
+
'reports-dir': { type: 'string', default: DEFAULT_REPORTS_DIR },
|
|
1506
|
+
variant: { type: 'string' },
|
|
1507
|
+
},
|
|
1508
|
+
strict: false,
|
|
1509
|
+
});
|
|
1510
|
+
const { createFileStore } = await import('./server/report-store.js');
|
|
1511
|
+
const store = createFileStore(resolve(values['reports-dir']));
|
|
1512
|
+
const report = await store.get(reportId);
|
|
1513
|
+
if (!report) {
|
|
1514
|
+
console.error(`Report not found: ${reportId}`);
|
|
1515
|
+
process.exit(1);
|
|
1516
|
+
}
|
|
1517
|
+
const saturation = report.variance?.saturation;
|
|
1518
|
+
if (!saturation) {
|
|
1519
|
+
console.error('该 report 无 saturation 数据 (需要 --repeat ≥ 2 才会记录)。');
|
|
1520
|
+
process.exit(1);
|
|
1521
|
+
}
|
|
1522
|
+
// Print the persisted verdict from the original run. The trace stores
|
|
1523
|
+
// (mean, ciLow, ciHigh) per checkpoint but not raw scores, so re-running
|
|
1524
|
+
// findSaturationPoint with different method/threshold is not possible
|
|
1525
|
+
// here — that would need raw scores, which would have to be persisted
|
|
1526
|
+
// by runMultiple. Future work: opt-in `--persist-saturation-raw` flag at
|
|
1527
|
+
// run time to enable post-hoc parameter sweeps.
|
|
1528
|
+
const variants = report.meta.variants ?? [];
|
|
1529
|
+
const targetVariants = values.variant ? [values.variant] : variants;
|
|
1530
|
+
console.log(`\n Saturation verdict (复述持久化结果)\n`);
|
|
1531
|
+
for (const variant of targetVariants) {
|
|
1532
|
+
const trace = saturation.perVariant[variant];
|
|
1533
|
+
if (!trace || trace.length === 0) {
|
|
1534
|
+
console.log(` ${variant}: 无 trace 数据`);
|
|
1535
|
+
continue;
|
|
1536
|
+
}
|
|
1537
|
+
console.log(` ${variant}:`);
|
|
1538
|
+
console.log(` checkpoints: ${trace.length} (N=${trace.map((p) => p.n).join(', ')})`);
|
|
1539
|
+
console.log(` 最近一点 mean=${trace[trace.length - 1].mean.toFixed(3)}, CI=[${trace[trace.length - 1].ciLow.toFixed(3)}, ${trace[trace.length - 1].ciHigh.toFixed(3)}]`);
|
|
1540
|
+
if (saturation.verdicts?.[variant]) {
|
|
1541
|
+
const v = saturation.verdicts[variant];
|
|
1542
|
+
console.log(` 持久化判定 (${v.method}): ${v.saturated ? `已饱和@N=${v.atN}` : '未饱和'} - ${v.reason}`);
|
|
1543
|
+
}
|
|
1544
|
+
else if (trace.length < 5) {
|
|
1545
|
+
console.log(` 判定: 数据点 ${trace.length} < 5,跳过 (跑 --repeat 5 以上才输出)`);
|
|
1546
|
+
}
|
|
1547
|
+
}
|
|
1548
|
+
console.log('');
|
|
1549
|
+
}
|
|
1550
|
+
// ---------------------------------------------------------------------------
|
|
1551
|
+
// handleVerdict — one-line ship/no-ship verdict (v0.22)
|
|
1552
|
+
// ---------------------------------------------------------------------------
|
|
1553
|
+
async function handleVerdict(argv) {
|
|
1554
|
+
const reportId = argv[0];
|
|
1555
|
+
if (!reportId || reportId === '--help' || reportId === '-h') {
|
|
1556
|
+
console.log([
|
|
1557
|
+
'',
|
|
1558
|
+
'Usage: omk bench verdict <reportId> [options]',
|
|
1559
|
+
'',
|
|
1560
|
+
'聚合 bootstrap CI / 三层 ci-gate / saturation / human α 给出一行结论。',
|
|
1561
|
+
'',
|
|
1562
|
+
'Verdict 等级:',
|
|
1563
|
+
' PROGRESS 显著改进 + 三层全过',
|
|
1564
|
+
' CAUTIOUS 改进真实但有警告 (gate 破 / 幅度太小 / 控制组本身崩)',
|
|
1565
|
+
' REGRESS 显著回退 — 不要 ship',
|
|
1566
|
+
' NOISE CI 跨 0,无法判定',
|
|
1567
|
+
' UNDERPOWERED 样本不足,需要扩 N 重测',
|
|
1568
|
+
' SOLO 单变体报告,无对比对象',
|
|
1569
|
+
'',
|
|
1570
|
+
'Options:',
|
|
1571
|
+
' --reports-dir <dir> report store dir (default: ~/.oh-my-knowledge/reports)',
|
|
1572
|
+
' --threshold <num> 三层 gate 阈值 (default 3.5,匹配 omk bench ci)',
|
|
1573
|
+
' --trivial-diff <num> "幅度太小"阈值 (default 0.1)',
|
|
1574
|
+
' --verbose 展开 per-pair 详情',
|
|
1575
|
+
'',
|
|
1576
|
+
].join('\n'));
|
|
1577
|
+
process.exit(reportId ? 0 : 1);
|
|
1578
|
+
}
|
|
1579
|
+
const { values } = parseArgs({
|
|
1580
|
+
args: argv.slice(1),
|
|
1581
|
+
options: {
|
|
1582
|
+
'reports-dir': { type: 'string', default: DEFAULT_REPORTS_DIR },
|
|
1583
|
+
threshold: { type: 'string' },
|
|
1584
|
+
'trivial-diff': { type: 'string' },
|
|
1585
|
+
verbose: { type: 'boolean', default: false },
|
|
1586
|
+
},
|
|
1587
|
+
strict: false,
|
|
1588
|
+
});
|
|
1589
|
+
const { createFileStore } = await import('./server/report-store.js');
|
|
1590
|
+
const store = createFileStore(resolve(values['reports-dir']));
|
|
1591
|
+
const report = await store.get(reportId);
|
|
1592
|
+
if (!report) {
|
|
1593
|
+
console.error(`Report not found: ${reportId}`);
|
|
1594
|
+
process.exit(1);
|
|
1595
|
+
}
|
|
1596
|
+
const { computeVerdict, formatVerdictText } = await import('./eval-core/verdict.js');
|
|
1597
|
+
const result = computeVerdict(report, {
|
|
1598
|
+
ciThreshold: values.threshold != null ? Number(values.threshold) : undefined,
|
|
1599
|
+
triviallySmallDiff: values['trivial-diff'] != null ? Number(values['trivial-diff']) : undefined,
|
|
1600
|
+
});
|
|
1601
|
+
console.log(formatVerdictText(result, { verbose: Boolean(values.verbose) }));
|
|
1602
|
+
// Exit code reflects ship recommendation: 0 only on PROGRESS / SOLO-pass.
|
|
1603
|
+
// NOISE / UNDERPOWERED / CAUTIOUS / REGRESS all exit 1 so this composes
|
|
1604
|
+
// with shell `&&` chains in CI.
|
|
1605
|
+
if (result.level === 'PROGRESS') {
|
|
1606
|
+
process.exit(0);
|
|
1607
|
+
}
|
|
1608
|
+
if (result.level === 'SOLO' && result.headline.includes('PASS')) {
|
|
1609
|
+
process.exit(0);
|
|
1610
|
+
}
|
|
1611
|
+
process.exit(1);
|
|
1612
|
+
}
|
|
1613
|
+
// ---------------------------------------------------------------------------
|
|
1614
|
+
// handleDiagnose — per-sample quality diagnostics (v0.23 A)
|
|
1615
|
+
// ---------------------------------------------------------------------------
|
|
1616
|
+
async function handleDiagnose(argv) {
|
|
1617
|
+
const reportId = argv[0];
|
|
1618
|
+
if (!reportId || reportId === '--help' || reportId === '-h') {
|
|
1619
|
+
console.log([
|
|
1620
|
+
'',
|
|
1621
|
+
'Usage: omk bench diagnose <reportId> [options]',
|
|
1622
|
+
'',
|
|
1623
|
+
'诊断样本集本身的质量问题:区分度低 / 重复 / 歧义 / 成本异常 / 全 fail。',
|
|
1624
|
+
'回答"测评结论是否被坏样本污染"——与 omk bench verdict 互补。',
|
|
1625
|
+
'',
|
|
1626
|
+
'Options:',
|
|
1627
|
+
' --reports-dir <dir> report store dir',
|
|
1628
|
+
' --samples <path> 样本文件路径 (用于 near-duplicate 检测;默认从 report.meta.request 读)',
|
|
1629
|
+
' --top <n> 每类只显示前 N 个 (默认 10,0=全部)',
|
|
1630
|
+
' --duplicate-rouge <num> near-duplicate ROUGE-1 阈值 (默认 0.7)',
|
|
1631
|
+
' --ambiguous-stddev <num> 歧义阈值,judge stddev (默认 1.0,需要 --judge-repeat ≥ 2 数据)',
|
|
1632
|
+
' --cost-k <num> 成本异常倍数 vs median (默认 3)',
|
|
1633
|
+
' --latency-k <num> 耗时异常倍数 vs median (默认 3)',
|
|
1634
|
+
' --flat <num> flat_scores 分差阈值 (默认 0.5)',
|
|
1635
|
+
'',
|
|
1636
|
+
].join('\n'));
|
|
1637
|
+
process.exit(reportId ? 0 : 1);
|
|
1638
|
+
}
|
|
1639
|
+
const { values } = parseArgs({
|
|
1640
|
+
args: argv.slice(1),
|
|
1641
|
+
options: {
|
|
1642
|
+
'reports-dir': { type: 'string', default: DEFAULT_REPORTS_DIR },
|
|
1643
|
+
samples: { type: 'string' },
|
|
1644
|
+
top: { type: 'string', default: '10' },
|
|
1645
|
+
'duplicate-rouge': { type: 'string' },
|
|
1646
|
+
'ambiguous-stddev': { type: 'string' },
|
|
1647
|
+
'cost-k': { type: 'string' },
|
|
1648
|
+
'latency-k': { type: 'string' },
|
|
1649
|
+
flat: { type: 'string' },
|
|
1650
|
+
},
|
|
1651
|
+
strict: false,
|
|
1652
|
+
});
|
|
1653
|
+
const { createFileStore } = await import('./server/report-store.js');
|
|
1654
|
+
const store = createFileStore(resolve(values['reports-dir']));
|
|
1655
|
+
const report = await store.get(reportId);
|
|
1656
|
+
if (!report) {
|
|
1657
|
+
console.error(`Report not found: ${reportId}`);
|
|
1658
|
+
process.exit(1);
|
|
1659
|
+
}
|
|
1660
|
+
// Try to read the samples file for near-duplicate detection. Source order:
|
|
1661
|
+
// 1. --samples <path> override
|
|
1662
|
+
// 2. report.meta.request.samplesPath (recorded at run time)
|
|
1663
|
+
// If neither resolves to a readable file, skip near-duplicate gracefully.
|
|
1664
|
+
let samples;
|
|
1665
|
+
const samplesPath = values.samples ?? report.meta?.request?.samplesPath;
|
|
1666
|
+
if (samplesPath && existsSync(samplesPath)) {
|
|
1667
|
+
try {
|
|
1668
|
+
const { loadSamples } = await import('./inputs/load-samples.js');
|
|
1669
|
+
samples = loadSamples(samplesPath).samples;
|
|
1670
|
+
}
|
|
1671
|
+
catch (err) {
|
|
1672
|
+
process.stderr.write(`warn: 加载 samples 文件失败 (${samplesPath}): ${err.message}\n`);
|
|
1673
|
+
}
|
|
1674
|
+
}
|
|
1675
|
+
const topRaw = Number(values.top);
|
|
1676
|
+
const topN = Number.isFinite(topRaw) && topRaw > 0 ? topRaw : undefined;
|
|
1677
|
+
const { diagnoseSamples, formatSampleDiagnostics } = await import('./analysis/sample-diagnostics.js');
|
|
1678
|
+
const diag = diagnoseSamples(report, {
|
|
1679
|
+
samples,
|
|
1680
|
+
duplicateRouge: values['duplicate-rouge'] != null ? Number(values['duplicate-rouge']) : undefined,
|
|
1681
|
+
ambiguousStddev: values['ambiguous-stddev'] != null ? Number(values['ambiguous-stddev']) : undefined,
|
|
1682
|
+
costOutlierK: values['cost-k'] != null ? Number(values['cost-k']) : undefined,
|
|
1683
|
+
latencyOutlierK: values['latency-k'] != null ? Number(values['latency-k']) : undefined,
|
|
1684
|
+
flatThreshold: values.flat != null ? Number(values.flat) : undefined,
|
|
1685
|
+
});
|
|
1686
|
+
console.log(formatSampleDiagnostics(diag, { topN }));
|
|
1687
|
+
// Exit code: 0 if health ≥ 70 and no errors; 1 otherwise. CI-friendly.
|
|
1688
|
+
if (diag.totals.errors === 0 && diag.healthScore >= 70) {
|
|
1689
|
+
process.exit(0);
|
|
1690
|
+
}
|
|
1691
|
+
process.exit(1);
|
|
1692
|
+
}
|
|
1693
|
+
// ---------------------------------------------------------------------------
|
|
1694
|
+
// handleFailures — LLM-driven failure clustering (v0.23 B)
|
|
1695
|
+
// ---------------------------------------------------------------------------
|
|
1696
|
+
async function handleFailures(argv) {
|
|
1697
|
+
const reportId = argv[0];
|
|
1698
|
+
if (!reportId || reportId === '--help' || reportId === '-h') {
|
|
1699
|
+
console.log([
|
|
1700
|
+
'',
|
|
1701
|
+
'Usage: omk bench failures <reportId> [options]',
|
|
1702
|
+
'',
|
|
1703
|
+
'把已有 report 的失败样本喂给一次 LLM 调用,自动聚类 + 给修复建议。',
|
|
1704
|
+
'失败定义:compositeScore < threshold 或 ok=false。',
|
|
1705
|
+
'',
|
|
1706
|
+
'Options:',
|
|
1707
|
+
' --reports-dir <dir> report store dir',
|
|
1708
|
+
' --judge-executor <name> 执行器 (default: claude)',
|
|
1709
|
+
' --judge-model <id> 聚类用的 model (default: 沿用 report.meta.judgeModel)',
|
|
1710
|
+
' --max-clusters <n> 最多多少 cluster (default 5)',
|
|
1711
|
+
' --threshold <num> compositeScore < threshold 算失败 (default 3)',
|
|
1712
|
+
' --max-feed <n> 最多喂给 LLM 多少条 (default 50,超出取最差)',
|
|
1713
|
+
'',
|
|
1714
|
+
].join('\n'));
|
|
1715
|
+
process.exit(reportId ? 0 : 1);
|
|
1716
|
+
}
|
|
1717
|
+
const { values } = parseArgs({
|
|
1718
|
+
args: argv.slice(1),
|
|
1719
|
+
options: {
|
|
1720
|
+
'reports-dir': { type: 'string', default: DEFAULT_REPORTS_DIR },
|
|
1721
|
+
'judge-executor': { type: 'string', default: 'claude' },
|
|
1722
|
+
'judge-model': { type: 'string' },
|
|
1723
|
+
'max-clusters': { type: 'string', default: '5' },
|
|
1724
|
+
threshold: { type: 'string', default: '3' },
|
|
1725
|
+
'max-feed': { type: 'string', default: '50' },
|
|
1726
|
+
},
|
|
1727
|
+
strict: false,
|
|
1728
|
+
});
|
|
1729
|
+
const { createFileStore } = await import('./server/report-store.js');
|
|
1730
|
+
const store = createFileStore(resolve(values['reports-dir']));
|
|
1731
|
+
const report = await store.get(reportId);
|
|
1732
|
+
if (!report) {
|
|
1733
|
+
console.error(`Report not found: ${reportId}`);
|
|
1734
|
+
process.exit(1);
|
|
1735
|
+
}
|
|
1736
|
+
const judgeModel = values['judge-model'] ?? report.meta?.judgeModel;
|
|
1737
|
+
if (!judgeModel) {
|
|
1738
|
+
console.error('No judge model. Pass --judge-model <id> or ensure report has meta.judgeModel.');
|
|
1739
|
+
process.exit(1);
|
|
1740
|
+
}
|
|
1741
|
+
const { createExecutor } = await import('./executors/index.js');
|
|
1742
|
+
const executor = createExecutor(values['judge-executor']);
|
|
1743
|
+
const { clusterFailures, formatFailureClusterReport } = await import('./analysis/failure-clusterer.js');
|
|
1744
|
+
const out = await clusterFailures({
|
|
1745
|
+
report: report,
|
|
1746
|
+
executor,
|
|
1747
|
+
judgeModel,
|
|
1748
|
+
maxClusters: Number(values['max-clusters']) || 5,
|
|
1749
|
+
failureThreshold: Number(values.threshold) || 3,
|
|
1750
|
+
maxFailuresFed: Number(values['max-feed']) || 50,
|
|
1751
|
+
});
|
|
1752
|
+
console.log(formatFailureClusterReport(out));
|
|
1753
|
+
}
|
|
1008
1754
|
// ---------------------------------------------------------------------------
|
|
1009
1755
|
// Entry
|
|
1010
1756
|
// ---------------------------------------------------------------------------
|