oh-my-knowledge 0.18.0 → 0.20.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +596 -326
- package/README.zh.md +917 -0
- package/dist/src/analysis/failure-clusterer.d.ts +96 -0
- package/dist/src/analysis/failure-clusterer.d.ts.map +1 -0
- package/dist/src/analysis/failure-clusterer.js +298 -0
- package/dist/src/analysis/failure-clusterer.js.map +1 -0
- package/dist/src/analysis/sample-diagnostics.d.ts +78 -0
- package/dist/src/analysis/sample-diagnostics.d.ts.map +1 -0
- package/dist/src/analysis/sample-diagnostics.js +259 -0
- package/dist/src/analysis/sample-diagnostics.js.map +1 -0
- package/dist/src/analysis/saturation.d.ts +85 -0
- package/dist/src/analysis/saturation.d.ts.map +1 -0
- package/dist/src/analysis/saturation.js +174 -0
- package/dist/src/analysis/saturation.js.map +1 -0
- package/dist/src/cli.js +772 -18
- package/dist/src/cli.js.map +1 -1
- package/dist/src/eval-core/bootstrap.d.ts +72 -0
- package/dist/src/eval-core/bootstrap.d.ts.map +1 -0
- package/dist/src/eval-core/bootstrap.js +174 -0
- package/dist/src/eval-core/bootstrap.js.map +1 -0
- package/dist/src/eval-core/evaluation-execution.d.ts +15 -1
- package/dist/src/eval-core/evaluation-execution.d.ts.map +1 -1
- package/dist/src/eval-core/evaluation-execution.js +37 -3
- package/dist/src/eval-core/evaluation-execution.js.map +1 -1
- package/dist/src/eval-core/evaluation-job.d.ts +10 -2
- package/dist/src/eval-core/evaluation-job.d.ts.map +1 -1
- package/dist/src/eval-core/evaluation-job.js +9 -1
- package/dist/src/eval-core/evaluation-job.js.map +1 -1
- package/dist/src/eval-core/evaluation-reporting.d.ts.map +1 -1
- package/dist/src/eval-core/evaluation-reporting.js +90 -0
- package/dist/src/eval-core/evaluation-reporting.js.map +1 -1
- package/dist/src/eval-core/schema.d.ts.map +1 -1
- package/dist/src/eval-core/schema.js +69 -0
- package/dist/src/eval-core/schema.js.map +1 -1
- package/dist/src/eval-core/verdict.d.ts +74 -0
- package/dist/src/eval-core/verdict.d.ts.map +1 -0
- package/dist/src/eval-core/verdict.js +283 -0
- package/dist/src/eval-core/verdict.js.map +1 -0
- package/dist/src/eval-workflows/each-evaluation-workflow.d.ts +16 -3
- package/dist/src/eval-workflows/each-evaluation-workflow.d.ts.map +1 -1
- package/dist/src/eval-workflows/each-evaluation-workflow.js +10 -2
- package/dist/src/eval-workflows/each-evaluation-workflow.js.map +1 -1
- package/dist/src/eval-workflows/evaluation-pipeline.d.ts +17 -1
- package/dist/src/eval-workflows/evaluation-pipeline.d.ts.map +1 -1
- package/dist/src/eval-workflows/evaluation-pipeline.js +45 -3
- package/dist/src/eval-workflows/evaluation-pipeline.js.map +1 -1
- package/dist/src/eval-workflows/run-evaluation.d.ts +23 -2
- package/dist/src/eval-workflows/run-evaluation.d.ts.map +1 -1
- package/dist/src/eval-workflows/run-evaluation.js +104 -4
- package/dist/src/eval-workflows/run-evaluation.js.map +1 -1
- package/dist/src/grading/assertions.d.ts +16 -0
- package/dist/src/grading/assertions.d.ts.map +1 -1
- package/dist/src/grading/assertions.js +385 -111
- package/dist/src/grading/assertions.js.map +1 -1
- package/dist/src/grading/debias-validate.d.ts +84 -0
- package/dist/src/grading/debias-validate.d.ts.map +1 -0
- package/dist/src/grading/debias-validate.js +173 -0
- package/dist/src/grading/debias-validate.js.map +1 -0
- package/dist/src/grading/gold-cli.d.ts +88 -0
- package/dist/src/grading/gold-cli.d.ts.map +1 -0
- package/dist/src/grading/gold-cli.js +251 -0
- package/dist/src/grading/gold-cli.js.map +1 -0
- package/dist/src/grading/gold-dataset.d.ts +73 -0
- package/dist/src/grading/gold-dataset.d.ts.map +1 -0
- package/dist/src/grading/gold-dataset.js +161 -0
- package/dist/src/grading/gold-dataset.js.map +1 -0
- package/dist/src/grading/human-gold.d.ts +102 -0
- package/dist/src/grading/human-gold.d.ts.map +1 -0
- package/dist/src/grading/human-gold.js +188 -0
- package/dist/src/grading/human-gold.js.map +1 -0
- package/dist/src/grading/index.d.ts +27 -2
- package/dist/src/grading/index.d.ts.map +1 -1
- package/dist/src/grading/index.js +36 -18
- package/dist/src/grading/index.js.map +1 -1
- package/dist/src/grading/judge.d.ts +65 -2
- package/dist/src/grading/judge.d.ts.map +1 -1
- package/dist/src/grading/judge.js +280 -23
- package/dist/src/grading/judge.js.map +1 -1
- package/dist/src/inputs/eval-config.js +19 -0
- package/dist/src/inputs/eval-config.js.map +1 -1
- package/dist/src/observability/{production-analyzer.d.ts → skill-health-analyzer.d.ts} +24 -2
- package/dist/src/observability/skill-health-analyzer.d.ts.map +1 -0
- package/dist/src/observability/{production-analyzer.js → skill-health-analyzer.js} +61 -6
- package/dist/src/observability/skill-health-analyzer.js.map +1 -0
- package/dist/src/observability/trace-adapter.d.ts.map +1 -1
- package/dist/src/observability/trace-adapter.js +27 -1
- package/dist/src/observability/trace-adapter.js.map +1 -1
- package/dist/src/renderer/html-renderer.d.ts.map +1 -1
- package/dist/src/renderer/html-renderer.js +40 -6
- package/dist/src/renderer/html-renderer.js.map +1 -1
- package/dist/src/renderer/layout.d.ts.map +1 -1
- package/dist/src/renderer/layout.js +138 -4
- package/dist/src/renderer/layout.js.map +1 -1
- package/dist/src/renderer/skill-health-renderer.d.ts +2 -2
- package/dist/src/renderer/skill-health-renderer.d.ts.map +1 -1
- package/dist/src/renderer/skill-health-renderer.js +39 -4
- package/dist/src/renderer/skill-health-renderer.js.map +1 -1
- package/dist/src/renderer/summary.d.ts +28 -1
- package/dist/src/renderer/summary.d.ts.map +1 -1
- package/dist/src/renderer/summary.js +322 -8
- package/dist/src/renderer/summary.js.map +1 -1
- package/dist/src/renderer/table.d.ts.map +1 -1
- package/dist/src/renderer/table.js +63 -2
- package/dist/src/renderer/table.js.map +1 -1
- package/dist/src/server/report-server.d.ts +2 -1
- package/dist/src/server/report-server.d.ts.map +1 -1
- package/dist/src/server/report-server.js +397 -2
- package/dist/src/server/report-server.js.map +1 -1
- package/dist/src/types.d.ts +247 -0
- package/dist/src/types.d.ts.map +1 -1
- package/package.json +24 -6
- package/dist/src/observability/production-analyzer.d.ts.map +0 -1
- package/dist/src/observability/production-analyzer.js.map +0 -1
package/dist/src/cli.js
CHANGED
|
@@ -95,11 +95,17 @@ function parseRunConfig(argv, extraOptions = {}) {
|
|
|
95
95
|
else if (evalConfig) {
|
|
96
96
|
variantSpecs = configVariantsToSpecs(evalConfig.variants);
|
|
97
97
|
}
|
|
98
|
+
else if (values.each) {
|
|
99
|
+
// --each 模式自动用 baseline (control) vs 每个 skill (treatment),
|
|
100
|
+
// 不需要用户显式传 --control / --treatment,校验跳过。
|
|
101
|
+
variantSpecs = [];
|
|
102
|
+
}
|
|
98
103
|
else {
|
|
99
104
|
const discovered = discoverVariants(skillDir);
|
|
100
105
|
const hint = discovered.length > 0 ? `\n skill-dir (${skillDir}) 下发现的候选:${discovered.join(', ')}` : '';
|
|
101
106
|
throw new Error(`请通过 --control / --treatment 或 --config eval.yaml 声明 variant 角色。\n`
|
|
102
107
|
+ ` 示例:omk bench run --control baseline --treatment my-skill${hint}\n`
|
|
108
|
+
+ ` --each 模式下自动用 baseline vs 每个 skill,无需显式声明\n`
|
|
103
109
|
+ ` 术语见 docs/terminology-spec.md(v0.16 起废除 --variants,改用 experiment role 显式声明)`);
|
|
104
110
|
}
|
|
105
111
|
const seenNames = new Set();
|
|
@@ -161,6 +167,7 @@ function parseRunConfig(argv, extraOptions = {}) {
|
|
|
161
167
|
resume,
|
|
162
168
|
blind,
|
|
163
169
|
layeredStats,
|
|
170
|
+
budget: evalConfig?.budget,
|
|
164
171
|
},
|
|
165
172
|
};
|
|
166
173
|
}
|
|
@@ -208,6 +215,22 @@ Options for "bench run":
|
|
|
208
215
|
--concurrency <n> Number of parallel tasks (default: 1)
|
|
209
216
|
--timeout <seconds> Executor timeout per task in seconds (default: 120)
|
|
210
217
|
--repeat <n> Run evaluation N times for variance analysis (default: 1)
|
|
218
|
+
--judge-repeat <n> Call LLM judge N times per (sample × dimension) for self-
|
|
219
|
+
consistency (default: 1). High stddev across runs = the
|
|
220
|
+
judge is unstable on this rubric and the score is noisy.
|
|
221
|
+
--judge-models <list> Multi-judge ensemble. Comma-separated executor:model pairs,
|
|
222
|
+
e.g. claude:opus,openai:gpt-4o,gemini:pro. Each judge scores
|
|
223
|
+
every (sample × dimension); report includes per-judge break-
|
|
224
|
+
down + Pearson/MAD inter-judge agreement. Refutes "Claude
|
|
225
|
+
judge Claude same-modality bias" critique. Combines with
|
|
226
|
+
--judge-repeat. Cost ~ N_judges × N_repeat × N_samples.
|
|
227
|
+
--bootstrap Compute bootstrap confidence intervals (distribution-free,
|
|
228
|
+
preferred over t-interval for ordinal LLM scores). Adds
|
|
229
|
+
per-variant CI on the mean + pairwise CI on treatment-vs-
|
|
230
|
+
control difference (significant=0 outside CI). Reports both
|
|
231
|
+
t-interval and bootstrap so old tooling still works.
|
|
232
|
+
--bootstrap-samples <n> Number of bootstrap resamples (default 1000). N>10000
|
|
233
|
+
triggers a stderr warning about runtime cost.
|
|
211
234
|
--retry <n> Retry failed tasks up to N times with exponential backoff (default: 0)
|
|
212
235
|
--resume <report-id> Resume from a previous report, skipping completed tasks
|
|
213
236
|
--executor <name> Executor: claude, openai, gemini, anthropic-api, openai-api,
|
|
@@ -354,8 +377,26 @@ async function main() {
|
|
|
354
377
|
case 'diff':
|
|
355
378
|
await handleDiff(rest);
|
|
356
379
|
break;
|
|
380
|
+
case 'gold':
|
|
381
|
+
await handleGold(rest);
|
|
382
|
+
break;
|
|
383
|
+
case 'debias-validate':
|
|
384
|
+
await handleDebiasValidate(rest);
|
|
385
|
+
break;
|
|
386
|
+
case 'saturation':
|
|
387
|
+
await handleSaturation(rest);
|
|
388
|
+
break;
|
|
389
|
+
case 'verdict':
|
|
390
|
+
await handleVerdict(rest);
|
|
391
|
+
break;
|
|
392
|
+
case 'diagnose':
|
|
393
|
+
await handleDiagnose(rest);
|
|
394
|
+
break;
|
|
395
|
+
case 'failures':
|
|
396
|
+
await handleFailures(rest);
|
|
397
|
+
break;
|
|
357
398
|
default:
|
|
358
|
-
console.error(`Unknown command: bench ${command}. Use "run", "report", "ci", "init", "gen-samples", or "
|
|
399
|
+
console.error(`Unknown command: bench ${command}. Use "run", "report", "ci", "init", "gen-samples", "evolve", "diff", "gold", "debias-validate", "saturation", "verdict", "diagnose", or "failures".`);
|
|
359
400
|
process.exit(1);
|
|
360
401
|
}
|
|
361
402
|
}
|
|
@@ -410,15 +451,98 @@ async function handleRun(argv) {
|
|
|
410
451
|
const { values, config } = parseRunConfig(argv, {
|
|
411
452
|
blind: { type: 'boolean', default: false },
|
|
412
453
|
repeat: { type: 'string', default: '1' },
|
|
454
|
+
'judge-repeat': { type: 'string', default: '1' },
|
|
455
|
+
'judge-models': { type: 'string' },
|
|
456
|
+
bootstrap: { type: 'boolean', default: false },
|
|
457
|
+
'bootstrap-samples': { type: 'string', default: '1000' },
|
|
458
|
+
'gold-dir': { type: 'string' },
|
|
459
|
+
'no-debias-length': { type: 'boolean', default: false },
|
|
460
|
+
'budget-usd': { type: 'string' },
|
|
461
|
+
'budget-per-sample-usd': { type: 'string' },
|
|
462
|
+
'budget-per-sample-ms': { type: 'string' },
|
|
413
463
|
});
|
|
414
464
|
const { runEvaluation, runMultiple, runEachEvaluation } = await import('./eval-workflows/run-evaluation.js');
|
|
415
465
|
config.blind = values.blind;
|
|
416
466
|
config.onProgress = defaultOnProgress;
|
|
467
|
+
// --repeat 诚实输入校验:非 ≥1 整数时提示并钳到 1,不静默掩盖用户错字/极端输入
|
|
468
|
+
// 提前到 --each 分支之前,保证 each 模式也能读到 repeat (曾经 bug: --each 吞 --repeat)
|
|
469
|
+
const repeatRaw = values.repeat;
|
|
470
|
+
const parsedRepeat = repeatRaw !== undefined ? Number(repeatRaw) : 1;
|
|
471
|
+
if (repeatRaw !== undefined && (!Number.isFinite(parsedRepeat) || parsedRepeat < 1)) {
|
|
472
|
+
process.stderr.write(`⚠ --repeat "${repeatRaw}" 无效(期望 ≥ 1 的整数),已按 1 次评测执行\n`);
|
|
473
|
+
}
|
|
474
|
+
const repeatCount = Math.max(1, Math.floor(parsedRepeat) || 1);
|
|
475
|
+
// --judge-repeat 同样的诚实校验:非 ≥1 整数时钳到 1
|
|
476
|
+
const judgeRepeatRaw = values['judge-repeat'];
|
|
477
|
+
const parsedJudgeRepeat = judgeRepeatRaw !== undefined ? Number(judgeRepeatRaw) : 1;
|
|
478
|
+
if (judgeRepeatRaw !== undefined && (!Number.isFinite(parsedJudgeRepeat) || parsedJudgeRepeat < 1)) {
|
|
479
|
+
process.stderr.write(`⚠ --judge-repeat "${judgeRepeatRaw}" 无效(期望 ≥ 1 的整数),已按 1 次 judge 执行\n`);
|
|
480
|
+
}
|
|
481
|
+
const judgeRepeatCount = Math.max(1, Math.floor(parsedJudgeRepeat) || 1);
|
|
482
|
+
if (judgeRepeatCount > 1)
|
|
483
|
+
config.judgeRepeat = judgeRepeatCount;
|
|
484
|
+
// --judge-models executor:model,executor:model,... -> JudgeConfig[]
|
|
485
|
+
// 至少 2 个才进 ensemble 模式,1 个等同于 --judge-model
|
|
486
|
+
const judgeModelsRaw = values['judge-models'];
|
|
487
|
+
if (judgeModelsRaw) {
|
|
488
|
+
const parts = judgeModelsRaw.split(',').map((s) => s.trim()).filter(Boolean);
|
|
489
|
+
const judges = parts.map((p) => {
|
|
490
|
+
const [executor, ...modelParts] = p.split(':');
|
|
491
|
+
const model = modelParts.join(':');
|
|
492
|
+
if (!executor || !model) {
|
|
493
|
+
throw new Error(`--judge-models 格式错误: "${p}",应为 "executor:model" (如 claude:opus)`);
|
|
494
|
+
}
|
|
495
|
+
return { executor, model };
|
|
496
|
+
});
|
|
497
|
+
if (judges.length >= 2) {
|
|
498
|
+
config.judgeModels = judges;
|
|
499
|
+
}
|
|
500
|
+
else if (judges.length === 1) {
|
|
501
|
+
// 单 judge 不走 ensemble,但允许这样写,等同于 --judge-model + --executor
|
|
502
|
+
process.stderr.write(`ℹ --judge-models 只指定 1 个 judge (${judges[0].executor}:${judges[0].model}),不触发 ensemble。如需 ensemble 至少给 2 个。\n`);
|
|
503
|
+
}
|
|
504
|
+
}
|
|
505
|
+
// --budget-usd / --budget-per-sample-usd / --budget-per-sample-ms:
|
|
506
|
+
// v0.22 hard budget caps. CLI flags override config-file values. When the
|
|
507
|
+
// total-USD cap is exceeded mid-run, remaining tasks are skipped and a
|
|
508
|
+
// partial report is persisted with meta.budgetExhausted=true.
|
|
509
|
+
const budgetUSD = values['budget-usd'] != null ? Number(values['budget-usd']) : undefined;
|
|
510
|
+
const budgetPerSampleUSD = values['budget-per-sample-usd'] != null ? Number(values['budget-per-sample-usd']) : undefined;
|
|
511
|
+
const budgetPerSampleMs = values['budget-per-sample-ms'] != null ? Number(values['budget-per-sample-ms']) : undefined;
|
|
512
|
+
if (budgetUSD !== undefined || budgetPerSampleUSD !== undefined || budgetPerSampleMs !== undefined) {
|
|
513
|
+
config.budget = {
|
|
514
|
+
...(budgetUSD !== undefined && Number.isFinite(budgetUSD) && budgetUSD >= 0 ? { totalUSD: budgetUSD } : {}),
|
|
515
|
+
...(budgetPerSampleUSD !== undefined && Number.isFinite(budgetPerSampleUSD) && budgetPerSampleUSD >= 0 ? { perSampleUSD: budgetPerSampleUSD } : {}),
|
|
516
|
+
...(budgetPerSampleMs !== undefined && Number.isFinite(budgetPerSampleMs) && budgetPerSampleMs >= 0 ? { perSampleMs: budgetPerSampleMs } : {}),
|
|
517
|
+
};
|
|
518
|
+
}
|
|
519
|
+
// --no-debias-length: opt out of v0.21 Phase 3a length-controlled prompt.
|
|
520
|
+
// Default behavior is debias-on (judge prompt v3-cot-length); flag flips it
|
|
521
|
+
// off so historical reports (judgePromptHash from v2-cot era) can be reproduced.
|
|
522
|
+
if (values['no-debias-length']) {
|
|
523
|
+
config.lengthDebias = false;
|
|
524
|
+
process.stderr.write('ℹ --no-debias-length 已生效:judge prompt 退回 v2-cot,与 < v0.21 报告 hash 一致。\n');
|
|
525
|
+
}
|
|
526
|
+
// --bootstrap / --bootstrap-samples
|
|
527
|
+
if (values.bootstrap) {
|
|
528
|
+
config.bootstrap = true;
|
|
529
|
+
const bsRaw = values['bootstrap-samples'];
|
|
530
|
+
const parsedBs = bsRaw !== undefined ? Number(bsRaw) : 1000;
|
|
531
|
+
if (bsRaw !== undefined && (!Number.isFinite(parsedBs) || parsedBs < 100)) {
|
|
532
|
+
process.stderr.write(`⚠ --bootstrap-samples "${bsRaw}" 无效(期望 ≥ 100 的整数),已按 1000 执行\n`);
|
|
533
|
+
}
|
|
534
|
+
const bsCount = Math.max(100, Math.floor(parsedBs) || 1000);
|
|
535
|
+
if (bsCount > 10000) {
|
|
536
|
+
process.stderr.write(`⚠ --bootstrap-samples ${bsCount} 较大,可能耗时数秒。1000 是业内标准,通常已够用。\n`);
|
|
537
|
+
}
|
|
538
|
+
config.bootstrapSamples = bsCount;
|
|
539
|
+
}
|
|
417
540
|
try {
|
|
418
541
|
// --each mode: evaluate each skill independently
|
|
419
542
|
if (values.each) {
|
|
420
543
|
const { report, filePath } = await runEachEvaluation({
|
|
421
544
|
...config,
|
|
545
|
+
repeat: repeatCount,
|
|
422
546
|
onSkillProgress({ phase, skill, current, total }) {
|
|
423
547
|
if (phase === 'start') {
|
|
424
548
|
process.stderr.write(`\n=== [${current}/${total}] Skill: ${skill} ===\n`);
|
|
@@ -449,13 +573,6 @@ async function handleRun(argv) {
|
|
|
449
573
|
}
|
|
450
574
|
return;
|
|
451
575
|
}
|
|
452
|
-
// --repeat 诚实输入校验:非 ≥1 整数时提示并钳到 1,不静默掩盖用户错字/极端输入
|
|
453
|
-
const repeatRaw = values.repeat;
|
|
454
|
-
const parsedRepeat = repeatRaw !== undefined ? Number(repeatRaw) : 1;
|
|
455
|
-
if (repeatRaw !== undefined && (!Number.isFinite(parsedRepeat) || parsedRepeat < 1)) {
|
|
456
|
-
process.stderr.write(`⚠ --repeat "${repeatRaw}" 无效(期望 ≥ 1 的整数),已按 1 次评测执行\n`);
|
|
457
|
-
}
|
|
458
|
-
const repeatCount = Math.max(1, Math.floor(parsedRepeat) || 1);
|
|
459
576
|
let report;
|
|
460
577
|
let filePath;
|
|
461
578
|
if (repeatCount > 1) {
|
|
@@ -474,6 +591,28 @@ async function handleRun(argv) {
|
|
|
474
591
|
report = result.report;
|
|
475
592
|
filePath = result.filePath;
|
|
476
593
|
}
|
|
594
|
+
// --gold-dir: compute α/κ/Pearson against gold annotations and re-persist.
|
|
595
|
+
const goldDir = values['gold-dir'];
|
|
596
|
+
if (goldDir && filePath) {
|
|
597
|
+
const { attachGoldAgreementToReport, formatGoldCompare } = await import('./grading/gold-cli.js');
|
|
598
|
+
const out = attachGoldAgreementToReport({
|
|
599
|
+
report,
|
|
600
|
+
goldDir,
|
|
601
|
+
outputDir: config.outputDir,
|
|
602
|
+
samples: config.bootstrapSamples,
|
|
603
|
+
});
|
|
604
|
+
if (out.result && out.gold) {
|
|
605
|
+
process.stderr.write(formatGoldCompare(out.result, out.gold));
|
|
606
|
+
if (out.result.contaminationWarning) {
|
|
607
|
+
process.stderr.write(`\n⚠ ${out.result.contaminationWarning}\n`);
|
|
608
|
+
}
|
|
609
|
+
}
|
|
610
|
+
else {
|
|
611
|
+
process.stderr.write(`\n⚠ gold dataset 加载失败 (${goldDir}):\n`);
|
|
612
|
+
for (const m of out.loadIssues)
|
|
613
|
+
process.stderr.write(` - ${m}\n`);
|
|
614
|
+
}
|
|
615
|
+
}
|
|
477
616
|
console.log(JSON.stringify(report, null, 2));
|
|
478
617
|
if (filePath) {
|
|
479
618
|
process.stderr.write('\n✅ 评测完成\n');
|
|
@@ -676,20 +815,19 @@ async function handleAnalyze(argv) {
|
|
|
676
815
|
const to = values.to;
|
|
677
816
|
const skills = values.skills ? values.skills.split(',').map((s) => s.trim()).filter(Boolean) : undefined;
|
|
678
817
|
console.log(`[omk] analyzing ${tracePath}...`);
|
|
679
|
-
const { computeSkillHealthReport } = await import('./observability/
|
|
818
|
+
const { computeSkillHealthReport } = await import('./observability/skill-health-analyzer.js');
|
|
680
819
|
const report = computeSkillHealthReport(tracePath, {
|
|
681
820
|
kbRoot: values.kb ? resolve(values.kb) : undefined,
|
|
682
821
|
from,
|
|
683
822
|
to,
|
|
684
823
|
skills,
|
|
685
824
|
});
|
|
686
|
-
|
|
687
|
-
const html = renderSkillHealthReport(report);
|
|
825
|
+
// JSON 是主产物; HTML 由 report server 的 /analyses/:id 按需渲染 (和 bench run 一致)
|
|
688
826
|
const outDir = resolve(values['output-dir'] || join(process.env.HOME || '.', '.oh-my-knowledge', 'analyses'));
|
|
689
827
|
mkdirSync(outDir, { recursive: true });
|
|
690
828
|
const timestamp = new Date().toISOString().replace(/[:.]/g, '-').slice(0, 19);
|
|
691
|
-
const
|
|
692
|
-
writeFileSync(
|
|
829
|
+
const jsonPath = join(outDir, `${timestamp}-skill-health.json`);
|
|
830
|
+
writeFileSync(jsonPath, JSON.stringify(report, null, 2));
|
|
693
831
|
// 控制台摘要
|
|
694
832
|
const { sessionCount, segmentCount, toolCallCount, toolFailureRate } = report.meta;
|
|
695
833
|
console.log('');
|
|
@@ -703,7 +841,8 @@ async function handleAnalyze(argv) {
|
|
|
703
841
|
console.log('top skills:');
|
|
704
842
|
console.log(skillRows.join('\n'));
|
|
705
843
|
console.log('');
|
|
706
|
-
console.log(`report written to: ${
|
|
844
|
+
console.log(`report written to: ${jsonPath}`);
|
|
845
|
+
console.log(`view in browser: omk bench report # 打开后点首页的 "📊 Skill 健康度日报"`);
|
|
707
846
|
}
|
|
708
847
|
async function handleInit(argv) {
|
|
709
848
|
const targetDir = resolve(argv[0] || '.');
|
|
@@ -934,13 +1073,57 @@ async function handleCi(argv) {
|
|
|
934
1073
|
// handleDiff
|
|
935
1074
|
// ---------------------------------------------------------------------------
|
|
936
1075
|
async function handleDiff(argv) {
|
|
937
|
-
|
|
938
|
-
|
|
939
|
-
|
|
1076
|
+
// Flag-aware split: separate positional report IDs from flags so we can support
|
|
1077
|
+
// omk bench diff <id> — within-report sample-level (v0.22)
|
|
1078
|
+
// omk bench diff <id1> <id2> — cross-report variant-level (legacy)
|
|
1079
|
+
// both with optional --regressions-only / --threshold / --variant flags.
|
|
1080
|
+
const positional = [];
|
|
1081
|
+
const flagArgs = [];
|
|
1082
|
+
for (let i = 0; i < argv.length; i++) {
|
|
1083
|
+
const a = argv[i];
|
|
1084
|
+
if (a.startsWith('--')) {
|
|
1085
|
+
flagArgs.push(a);
|
|
1086
|
+
const next = argv[i + 1];
|
|
1087
|
+
if (next !== undefined && !next.startsWith('--')) {
|
|
1088
|
+
flagArgs.push(next);
|
|
1089
|
+
i++;
|
|
1090
|
+
}
|
|
1091
|
+
}
|
|
1092
|
+
else {
|
|
1093
|
+
positional.push(a);
|
|
1094
|
+
}
|
|
1095
|
+
}
|
|
1096
|
+
if (positional.length === 0) {
|
|
1097
|
+
console.error([
|
|
1098
|
+
'Usage:',
|
|
1099
|
+
' omk bench diff <reportId> within-report per-sample diff (v0.22)',
|
|
1100
|
+
' omk bench diff <reportId1> <reportId2> cross-report variant-level diff',
|
|
1101
|
+
'',
|
|
1102
|
+
'Options:',
|
|
1103
|
+
' --regressions-only 只列 treatment < control 的样本',
|
|
1104
|
+
' --threshold <num> regression 阈值 (default 0,即任一负 Δ 算回退)',
|
|
1105
|
+
' --variant <name> within-report 模式下指定要钻取的 variant (default: variants[1])',
|
|
1106
|
+
' --top <n> 只列差距最大的前 N 个样本',
|
|
1107
|
+
].join('\n'));
|
|
1108
|
+
process.exit(positional.length === 0 ? 1 : 0);
|
|
940
1109
|
}
|
|
1110
|
+
const { values } = parseArgs({
|
|
1111
|
+
args: flagArgs,
|
|
1112
|
+
options: {
|
|
1113
|
+
'regressions-only': { type: 'boolean', default: false },
|
|
1114
|
+
threshold: { type: 'string' },
|
|
1115
|
+
variant: { type: 'string' },
|
|
1116
|
+
top: { type: 'string' },
|
|
1117
|
+
},
|
|
1118
|
+
strict: false,
|
|
1119
|
+
});
|
|
941
1120
|
const { createFileStore } = await import('./server/report-store.js');
|
|
942
1121
|
const store = createFileStore(resolve(DEFAULT_REPORTS_DIR));
|
|
943
|
-
|
|
1122
|
+
if (positional.length === 1) {
|
|
1123
|
+
await runSampleLevelDiff(positional[0], store, values);
|
|
1124
|
+
return;
|
|
1125
|
+
}
|
|
1126
|
+
const [id1, id2] = positional;
|
|
944
1127
|
const r1 = await store.get(id1);
|
|
945
1128
|
const r2 = await store.get(id2);
|
|
946
1129
|
if (!r1) {
|
|
@@ -997,6 +1180,577 @@ async function handleDiff(argv) {
|
|
|
997
1180
|
}
|
|
998
1181
|
console.log('');
|
|
999
1182
|
}
|
|
1183
|
+
/**
|
|
1184
|
+
* Within-report sample-level diff (v0.22). Compares two variants' scores on
|
|
1185
|
+
* each shared sample and surfaces the worst regressions / biggest wins.
|
|
1186
|
+
*
|
|
1187
|
+
* Default focus is variants[0] (control) vs variants[1] (treatment), but
|
|
1188
|
+
* `--variant` overrides which variant is the "treatment" side.
|
|
1189
|
+
*/
|
|
1190
|
+
async function runSampleLevelDiff(reportId, store, flags) {
|
|
1191
|
+
const report = await store.get(reportId);
|
|
1192
|
+
if (!report) {
|
|
1193
|
+
console.error(`Report not found: ${reportId}`);
|
|
1194
|
+
process.exit(1);
|
|
1195
|
+
}
|
|
1196
|
+
const variants = report.meta?.variants ?? [];
|
|
1197
|
+
if (variants.length < 2) {
|
|
1198
|
+
console.error('Sample-level diff needs at least 2 variants in the report.');
|
|
1199
|
+
process.exit(1);
|
|
1200
|
+
}
|
|
1201
|
+
const control = variants[0];
|
|
1202
|
+
const treatment = flags.variant ?? variants[1];
|
|
1203
|
+
if (!variants.includes(treatment)) {
|
|
1204
|
+
console.error(`Variant "${treatment}" not in report. Available: ${variants.join(', ')}`);
|
|
1205
|
+
process.exit(1);
|
|
1206
|
+
}
|
|
1207
|
+
const threshold = flags.threshold != null ? Number(flags.threshold) : 0;
|
|
1208
|
+
const regressionsOnly = Boolean(flags['regressions-only']);
|
|
1209
|
+
const topN = flags.top != null ? Math.max(1, Number(flags.top) || 0) : undefined;
|
|
1210
|
+
const rows = [];
|
|
1211
|
+
for (const entry of report.results ?? []) {
|
|
1212
|
+
const c = entry.variants?.[control];
|
|
1213
|
+
const t = entry.variants?.[treatment];
|
|
1214
|
+
if (!c || !t)
|
|
1215
|
+
continue;
|
|
1216
|
+
const cComp = c.compositeScore ?? c.llmScore ?? 0;
|
|
1217
|
+
const tComp = t.compositeScore ?? t.llmScore ?? 0;
|
|
1218
|
+
const delta = Number((tComp - cComp).toFixed(3));
|
|
1219
|
+
rows.push({
|
|
1220
|
+
id: entry.sample_id,
|
|
1221
|
+
cFact: c.layeredScores?.factScore, tFact: t.layeredScores?.factScore,
|
|
1222
|
+
cBeh: c.layeredScores?.behaviorScore, tBeh: t.layeredScores?.behaviorScore,
|
|
1223
|
+
cJudge: c.layeredScores?.judgeScore, tJudge: t.layeredScores?.judgeScore,
|
|
1224
|
+
cComp, tComp, delta,
|
|
1225
|
+
});
|
|
1226
|
+
}
|
|
1227
|
+
// Sort by |delta| desc so the most impactful rows surface first.
|
|
1228
|
+
rows.sort((a, b) => Math.abs(b.delta) - Math.abs(a.delta));
|
|
1229
|
+
let filtered = rows;
|
|
1230
|
+
if (regressionsOnly)
|
|
1231
|
+
filtered = filtered.filter((r) => r.delta < threshold);
|
|
1232
|
+
if (topN !== undefined)
|
|
1233
|
+
filtered = filtered.slice(0, topN);
|
|
1234
|
+
console.log(`\n Sample-level diff: ${treatment} vs ${control} (report ${reportId})`);
|
|
1235
|
+
if (regressionsOnly)
|
|
1236
|
+
console.log(` Filter: regressions only (Δ < ${threshold})`);
|
|
1237
|
+
console.log('');
|
|
1238
|
+
console.log(' sample_id Δ composite (c→t) fact (c→t) behavior (c→t) judge (c→t)');
|
|
1239
|
+
console.log(' ' + '-'.repeat(100));
|
|
1240
|
+
if (filtered.length === 0) {
|
|
1241
|
+
console.log(regressionsOnly ? ' (no regressions found)' : ' (no shared samples)');
|
|
1242
|
+
console.log('');
|
|
1243
|
+
return;
|
|
1244
|
+
}
|
|
1245
|
+
const fmt = (a, b) => {
|
|
1246
|
+
const av = typeof a === 'number' ? a.toFixed(2) : '—';
|
|
1247
|
+
const bv = typeof b === 'number' ? b.toFixed(2) : '—';
|
|
1248
|
+
return `${av} → ${bv}`.padEnd(15);
|
|
1249
|
+
};
|
|
1250
|
+
for (const r of filtered) {
|
|
1251
|
+
const sign = r.delta > 0 ? '+' : '';
|
|
1252
|
+
const idCol = r.id.slice(0, 18).padEnd(20);
|
|
1253
|
+
const deltaCol = `${sign}${r.delta.toFixed(2)}`.padEnd(7);
|
|
1254
|
+
const compCol = `${r.cComp.toFixed(2)} → ${r.tComp.toFixed(2)}`.padEnd(17);
|
|
1255
|
+
console.log(` ${idCol}${deltaCol}${compCol}${fmt(r.cFact, r.tFact)} ${fmt(r.cBeh, r.tBeh)} ${fmt(r.cJudge, r.tJudge)}`);
|
|
1256
|
+
}
|
|
1257
|
+
console.log('');
|
|
1258
|
+
console.log(` Showing ${filtered.length} of ${rows.length} samples · sorted by |Δ|`);
|
|
1259
|
+
if (regressionsOnly) {
|
|
1260
|
+
const total = rows.length;
|
|
1261
|
+
const reg = rows.filter((r) => r.delta < threshold).length;
|
|
1262
|
+
console.log(` Regression rate: ${reg}/${total} samples (${total > 0 ? ((reg / total) * 100).toFixed(0) : 0}%)`);
|
|
1263
|
+
}
|
|
1264
|
+
console.log('');
|
|
1265
|
+
}
|
|
1266
|
+
// ---------------------------------------------------------------------------
|
|
1267
|
+
// handleGold — gold dataset workflow (init / validate / compare)
|
|
1268
|
+
// ---------------------------------------------------------------------------
|
|
1269
|
+
async function handleGold(argv) {
|
|
1270
|
+
const sub = argv[0];
|
|
1271
|
+
const rest = argv.slice(1);
|
|
1272
|
+
if (!sub || sub === '--help' || sub === '-h') {
|
|
1273
|
+
console.log([
|
|
1274
|
+
'',
|
|
1275
|
+
'Usage: omk bench gold <subcommand>',
|
|
1276
|
+
'',
|
|
1277
|
+
'Subcommands:',
|
|
1278
|
+
' init [--out <dir>] [--annotator <id>] 生成空白 gold dataset 模板',
|
|
1279
|
+
' validate <dir> 校验数据集结构',
|
|
1280
|
+
' compare <reportId> --gold-dir <dir> 与已有 report 计算 α/κ/Pearson',
|
|
1281
|
+
' [--variant <name>] [--reports-dir <d>]',
|
|
1282
|
+
' [--bootstrap-samples N] [--seed N]',
|
|
1283
|
+
'',
|
|
1284
|
+
].join('\n'));
|
|
1285
|
+
process.exit(sub ? 0 : 1);
|
|
1286
|
+
}
|
|
1287
|
+
if (sub === 'init') {
|
|
1288
|
+
const { values } = parseArgs({
|
|
1289
|
+
args: rest,
|
|
1290
|
+
options: {
|
|
1291
|
+
out: { type: 'string', default: './gold-dataset' },
|
|
1292
|
+
annotator: { type: 'string' },
|
|
1293
|
+
},
|
|
1294
|
+
strict: false,
|
|
1295
|
+
});
|
|
1296
|
+
const { initGoldDataset } = await import('./grading/gold-cli.js');
|
|
1297
|
+
try {
|
|
1298
|
+
const written = initGoldDataset(values.out, {
|
|
1299
|
+
annotator: values.annotator,
|
|
1300
|
+
});
|
|
1301
|
+
console.log(`Created ${written.length} files in ${values.out}:`);
|
|
1302
|
+
for (const p of written)
|
|
1303
|
+
console.log(` ${p}`);
|
|
1304
|
+
console.log('\n下一步: 编辑 annotations.yaml 加入真实标注 → 跑 omk bench gold validate');
|
|
1305
|
+
}
|
|
1306
|
+
catch (err) {
|
|
1307
|
+
console.error(err.message);
|
|
1308
|
+
process.exit(1);
|
|
1309
|
+
}
|
|
1310
|
+
return;
|
|
1311
|
+
}
|
|
1312
|
+
if (sub === 'validate') {
|
|
1313
|
+
const dir = rest[0];
|
|
1314
|
+
if (!dir) {
|
|
1315
|
+
console.error('Usage: omk bench gold validate <dir>');
|
|
1316
|
+
process.exit(1);
|
|
1317
|
+
}
|
|
1318
|
+
const { validateGoldDataset } = await import('./grading/gold-cli.js');
|
|
1319
|
+
const result = validateGoldDataset(dir);
|
|
1320
|
+
if (result.ok) {
|
|
1321
|
+
console.log(`✓ gold dataset OK — ${result.sampleCount} 条标注`);
|
|
1322
|
+
return;
|
|
1323
|
+
}
|
|
1324
|
+
console.error(`✗ gold dataset has ${result.issues.length} issue(s):`);
|
|
1325
|
+
for (const msg of result.issues)
|
|
1326
|
+
console.error(` - ${msg}`);
|
|
1327
|
+
process.exit(1);
|
|
1328
|
+
}
|
|
1329
|
+
if (sub === 'compare') {
|
|
1330
|
+
const reportId = rest[0];
|
|
1331
|
+
if (!reportId) {
|
|
1332
|
+
console.error('Usage: omk bench gold compare <reportId> --gold-dir <dir>');
|
|
1333
|
+
process.exit(1);
|
|
1334
|
+
}
|
|
1335
|
+
const { values } = parseArgs({
|
|
1336
|
+
args: rest.slice(1),
|
|
1337
|
+
options: {
|
|
1338
|
+
'gold-dir': { type: 'string' },
|
|
1339
|
+
variant: { type: 'string' },
|
|
1340
|
+
'reports-dir': { type: 'string', default: DEFAULT_REPORTS_DIR },
|
|
1341
|
+
'bootstrap-samples': { type: 'string', default: '1000' },
|
|
1342
|
+
seed: { type: 'string' },
|
|
1343
|
+
},
|
|
1344
|
+
strict: false,
|
|
1345
|
+
});
|
|
1346
|
+
const goldDir = values['gold-dir'];
|
|
1347
|
+
if (!goldDir) {
|
|
1348
|
+
console.error('--gold-dir is required');
|
|
1349
|
+
process.exit(1);
|
|
1350
|
+
}
|
|
1351
|
+
const { loadGoldDataset } = await import('./grading/gold-dataset.js');
|
|
1352
|
+
const { compareGoldToReport, formatGoldCompare } = await import('./grading/gold-cli.js');
|
|
1353
|
+
const { createFileStore } = await import('./server/report-store.js');
|
|
1354
|
+
const { dataset, issues } = loadGoldDataset(goldDir);
|
|
1355
|
+
if (!dataset) {
|
|
1356
|
+
console.error('Cannot load gold dataset:');
|
|
1357
|
+
for (const i of issues)
|
|
1358
|
+
console.error(` - ${i.message}`);
|
|
1359
|
+
process.exit(1);
|
|
1360
|
+
}
|
|
1361
|
+
if (issues.length) {
|
|
1362
|
+
// Non-fatal issues (e.g. duplicate already filtered) — surface them.
|
|
1363
|
+
for (const i of issues)
|
|
1364
|
+
console.error(`warn: ${i.message}`);
|
|
1365
|
+
}
|
|
1366
|
+
const store = createFileStore(resolve(values['reports-dir']));
|
|
1367
|
+
const report = await store.get(reportId);
|
|
1368
|
+
if (!report) {
|
|
1369
|
+
console.error(`Report not found: ${reportId}`);
|
|
1370
|
+
process.exit(1);
|
|
1371
|
+
}
|
|
1372
|
+
const samples = Math.max(100, Number(values['bootstrap-samples']) || 1000);
|
|
1373
|
+
const seedVal = values.seed != null ? Number(values.seed) : undefined;
|
|
1374
|
+
const result = compareGoldToReport({
|
|
1375
|
+
report: report,
|
|
1376
|
+
gold: dataset,
|
|
1377
|
+
variant: values.variant,
|
|
1378
|
+
samples,
|
|
1379
|
+
seed: Number.isFinite(seedVal) ? seedVal : undefined,
|
|
1380
|
+
});
|
|
1381
|
+
console.log(formatGoldCompare(result, dataset));
|
|
1382
|
+
return;
|
|
1383
|
+
}
|
|
1384
|
+
console.error(`Unknown subcommand: gold ${sub}. Use init / validate / compare.`);
|
|
1385
|
+
process.exit(1);
|
|
1386
|
+
}
|
|
1387
|
+
// ---------------------------------------------------------------------------
|
|
1388
|
+
// handleDebiasValidate — measure length-debias prompt sensitivity (Phase 3a)
|
|
1389
|
+
// ---------------------------------------------------------------------------
|
|
1390
|
+
async function handleDebiasValidate(argv) {
|
|
1391
|
+
const sub = argv[0];
|
|
1392
|
+
const rest = argv.slice(1);
|
|
1393
|
+
if (!sub || sub === '--help' || sub === '-h') {
|
|
1394
|
+
console.log([
|
|
1395
|
+
'',
|
|
1396
|
+
'Usage: omk bench debias-validate <kind> <reportId> [options]',
|
|
1397
|
+
'',
|
|
1398
|
+
'Kinds:',
|
|
1399
|
+
' length re-judge with the opposite length-debias setting and bootstrap CI',
|
|
1400
|
+
' on the score diff. Cost ~doubles vs the original judge pass.',
|
|
1401
|
+
'',
|
|
1402
|
+
'Options:',
|
|
1403
|
+
' --reports-dir <dir> report store dir (default: ~/.oh-my-knowledge/reports)',
|
|
1404
|
+
' --samples <path> override samples file (default: from report.meta.request)',
|
|
1405
|
+
' --variant <name> which variant to validate (default: first)',
|
|
1406
|
+
' --judge-executor <name> executor for judge calls (default: claude)',
|
|
1407
|
+
' --judge-model <model> judge model id (default: from report)',
|
|
1408
|
+
' --bootstrap-samples N bootstrap iterations (default 1000)',
|
|
1409
|
+
' --seed N deterministic CI seed',
|
|
1410
|
+
'',
|
|
1411
|
+
].join('\n'));
|
|
1412
|
+
process.exit(sub ? 0 : 1);
|
|
1413
|
+
}
|
|
1414
|
+
if (sub !== 'length') {
|
|
1415
|
+
console.error(`Unknown debias-validate kind: ${sub}. Use "length".`);
|
|
1416
|
+
process.exit(1);
|
|
1417
|
+
}
|
|
1418
|
+
const reportId = rest[0];
|
|
1419
|
+
if (!reportId) {
|
|
1420
|
+
console.error('Usage: omk bench debias-validate length <reportId>');
|
|
1421
|
+
process.exit(1);
|
|
1422
|
+
}
|
|
1423
|
+
const { values } = parseArgs({
|
|
1424
|
+
args: rest.slice(1),
|
|
1425
|
+
options: {
|
|
1426
|
+
'reports-dir': { type: 'string', default: DEFAULT_REPORTS_DIR },
|
|
1427
|
+
samples: { type: 'string' },
|
|
1428
|
+
variant: { type: 'string' },
|
|
1429
|
+
'judge-executor': { type: 'string', default: 'claude' },
|
|
1430
|
+
'judge-model': { type: 'string' },
|
|
1431
|
+
'bootstrap-samples': { type: 'string', default: '1000' },
|
|
1432
|
+
seed: { type: 'string' },
|
|
1433
|
+
},
|
|
1434
|
+
strict: false,
|
|
1435
|
+
});
|
|
1436
|
+
const { createFileStore } = await import('./server/report-store.js');
|
|
1437
|
+
const store = createFileStore(resolve(values['reports-dir']));
|
|
1438
|
+
const report = await store.get(reportId);
|
|
1439
|
+
if (!report) {
|
|
1440
|
+
console.error(`Report not found: ${reportId}`);
|
|
1441
|
+
process.exit(1);
|
|
1442
|
+
}
|
|
1443
|
+
// Resolve samples path: --samples overrides; otherwise read from report.meta.request.
|
|
1444
|
+
const samplesPath = values.samples
|
|
1445
|
+
?? report.meta?.request?.samplesPath;
|
|
1446
|
+
if (!samplesPath) {
|
|
1447
|
+
console.error('Cannot find samples path. Pass --samples <path> or ensure report has request.samplesPath.');
|
|
1448
|
+
process.exit(1);
|
|
1449
|
+
}
|
|
1450
|
+
const { loadSamples } = await import('./inputs/load-samples.js');
|
|
1451
|
+
const { samples } = loadSamples(samplesPath);
|
|
1452
|
+
const judgeModel = values['judge-model']
|
|
1453
|
+
?? report.meta?.judgeModel;
|
|
1454
|
+
if (!judgeModel) {
|
|
1455
|
+
console.error('No judge model. Pass --judge-model <id> or ensure report has meta.judgeModel.');
|
|
1456
|
+
process.exit(1);
|
|
1457
|
+
}
|
|
1458
|
+
process.stderr.write('\n⚠ debias-validate 会重判所有 (sample × variant),judge cost 大致翻倍。\n');
|
|
1459
|
+
const { createExecutor } = await import('./executors/index.js');
|
|
1460
|
+
const judgeExecutor = createExecutor(values['judge-executor']);
|
|
1461
|
+
const { validateLengthDebias, formatDebiasValidate } = await import('./grading/debias-validate.js');
|
|
1462
|
+
const seedVal = values.seed != null ? Number(values.seed) : undefined;
|
|
1463
|
+
const bsRaw = Number(values['bootstrap-samples']) || 1000;
|
|
1464
|
+
const result = await validateLengthDebias({
|
|
1465
|
+
report: report,
|
|
1466
|
+
samples,
|
|
1467
|
+
judgeExecutor,
|
|
1468
|
+
judgeModel,
|
|
1469
|
+
variant: values.variant,
|
|
1470
|
+
bootstrapSamples: Math.max(100, bsRaw),
|
|
1471
|
+
seed: Number.isFinite(seedVal) ? seedVal : undefined,
|
|
1472
|
+
onProgress: ({ sample_id, completed, total }) => {
|
|
1473
|
+
process.stderr.write(` judging ${completed}/${total}: ${sample_id}\n`);
|
|
1474
|
+
},
|
|
1475
|
+
});
|
|
1476
|
+
console.log(formatDebiasValidate(result));
|
|
1477
|
+
}
|
|
1478
|
+
// ---------------------------------------------------------------------------
|
|
1479
|
+
// handleSaturation — re-compute saturation verdict from a finished report
|
|
1480
|
+
// ---------------------------------------------------------------------------
|
|
1481
|
+
async function handleSaturation(argv) {
|
|
1482
|
+
const reportId = argv[0];
|
|
1483
|
+
if (!reportId || reportId === '--help' || reportId === '-h') {
|
|
1484
|
+
console.log([
|
|
1485
|
+
'',
|
|
1486
|
+
'Usage: omk bench saturation <reportId> [options]',
|
|
1487
|
+
'',
|
|
1488
|
+
'回答"我跑够样本了吗?"。复述已有 report 中持久化的饱和判定。',
|
|
1489
|
+
'',
|
|
1490
|
+
'注:本命令读取 run 时跑出的 verdict(运行时已用 method=bootstrap-ci-width',
|
|
1491
|
+
'默认阈值 + 3 窗口 持续条件)。如要换 method/threshold 重新计算,需要重跑',
|
|
1492
|
+
'`omk bench run --repeat ≥ 5`(运行时持久化的 trace 不含原始分数,无法',
|
|
1493
|
+
'在事后用其他参数复算)。',
|
|
1494
|
+
'',
|
|
1495
|
+
'Options:',
|
|
1496
|
+
' --reports-dir <dir> report store dir (default: ~/.oh-my-knowledge/reports)',
|
|
1497
|
+
' --variant <name> 只看一个 variant (default: all)',
|
|
1498
|
+
'',
|
|
1499
|
+
].join('\n'));
|
|
1500
|
+
process.exit(reportId ? 0 : 1);
|
|
1501
|
+
}
|
|
1502
|
+
const { values } = parseArgs({
|
|
1503
|
+
args: argv.slice(1),
|
|
1504
|
+
options: {
|
|
1505
|
+
'reports-dir': { type: 'string', default: DEFAULT_REPORTS_DIR },
|
|
1506
|
+
variant: { type: 'string' },
|
|
1507
|
+
},
|
|
1508
|
+
strict: false,
|
|
1509
|
+
});
|
|
1510
|
+
const { createFileStore } = await import('./server/report-store.js');
|
|
1511
|
+
const store = createFileStore(resolve(values['reports-dir']));
|
|
1512
|
+
const report = await store.get(reportId);
|
|
1513
|
+
if (!report) {
|
|
1514
|
+
console.error(`Report not found: ${reportId}`);
|
|
1515
|
+
process.exit(1);
|
|
1516
|
+
}
|
|
1517
|
+
const saturation = report.variance?.saturation;
|
|
1518
|
+
if (!saturation) {
|
|
1519
|
+
console.error('该 report 无 saturation 数据 (需要 --repeat ≥ 2 才会记录)。');
|
|
1520
|
+
process.exit(1);
|
|
1521
|
+
}
|
|
1522
|
+
// Print the persisted verdict from the original run. The trace stores
|
|
1523
|
+
// (mean, ciLow, ciHigh) per checkpoint but not raw scores, so re-running
|
|
1524
|
+
// findSaturationPoint with different method/threshold is not possible
|
|
1525
|
+
// here — that would need raw scores, which would have to be persisted
|
|
1526
|
+
// by runMultiple. Future work: opt-in `--persist-saturation-raw` flag at
|
|
1527
|
+
// run time to enable post-hoc parameter sweeps.
|
|
1528
|
+
const variants = report.meta.variants ?? [];
|
|
1529
|
+
const targetVariants = values.variant ? [values.variant] : variants;
|
|
1530
|
+
console.log(`\n Saturation verdict (复述持久化结果)\n`);
|
|
1531
|
+
for (const variant of targetVariants) {
|
|
1532
|
+
const trace = saturation.perVariant[variant];
|
|
1533
|
+
if (!trace || trace.length === 0) {
|
|
1534
|
+
console.log(` ${variant}: 无 trace 数据`);
|
|
1535
|
+
continue;
|
|
1536
|
+
}
|
|
1537
|
+
console.log(` ${variant}:`);
|
|
1538
|
+
console.log(` checkpoints: ${trace.length} (N=${trace.map((p) => p.n).join(', ')})`);
|
|
1539
|
+
console.log(` 最近一点 mean=${trace[trace.length - 1].mean.toFixed(3)}, CI=[${trace[trace.length - 1].ciLow.toFixed(3)}, ${trace[trace.length - 1].ciHigh.toFixed(3)}]`);
|
|
1540
|
+
if (saturation.verdicts?.[variant]) {
|
|
1541
|
+
const v = saturation.verdicts[variant];
|
|
1542
|
+
console.log(` 持久化判定 (${v.method}): ${v.saturated ? `已饱和@N=${v.atN}` : '未饱和'} - ${v.reason}`);
|
|
1543
|
+
}
|
|
1544
|
+
else if (trace.length < 5) {
|
|
1545
|
+
console.log(` 判定: 数据点 ${trace.length} < 5,跳过 (跑 --repeat 5 以上才输出)`);
|
|
1546
|
+
}
|
|
1547
|
+
}
|
|
1548
|
+
console.log('');
|
|
1549
|
+
}
|
|
1550
|
+
// ---------------------------------------------------------------------------
|
|
1551
|
+
// handleVerdict — one-line ship/no-ship verdict (v0.22)
|
|
1552
|
+
// ---------------------------------------------------------------------------
|
|
1553
|
+
async function handleVerdict(argv) {
|
|
1554
|
+
const reportId = argv[0];
|
|
1555
|
+
if (!reportId || reportId === '--help' || reportId === '-h') {
|
|
1556
|
+
console.log([
|
|
1557
|
+
'',
|
|
1558
|
+
'Usage: omk bench verdict <reportId> [options]',
|
|
1559
|
+
'',
|
|
1560
|
+
'聚合 bootstrap CI / 三层 ci-gate / saturation / human α 给出一行结论。',
|
|
1561
|
+
'',
|
|
1562
|
+
'Verdict 等级:',
|
|
1563
|
+
' PROGRESS 显著改进 + 三层全过',
|
|
1564
|
+
' CAUTIOUS 改进真实但有警告 (gate 破 / 幅度太小 / 控制组本身崩)',
|
|
1565
|
+
' REGRESS 显著回退 — 不要 ship',
|
|
1566
|
+
' NOISE CI 跨 0,无法判定',
|
|
1567
|
+
' UNDERPOWERED 样本不足,需要扩 N 重测',
|
|
1568
|
+
' SOLO 单变体报告,无对比对象',
|
|
1569
|
+
'',
|
|
1570
|
+
'Options:',
|
|
1571
|
+
' --reports-dir <dir> report store dir (default: ~/.oh-my-knowledge/reports)',
|
|
1572
|
+
' --threshold <num> 三层 gate 阈值 (default 3.5,匹配 omk bench ci)',
|
|
1573
|
+
' --trivial-diff <num> "幅度太小"阈值 (default 0.1)',
|
|
1574
|
+
' --verbose 展开 per-pair 详情',
|
|
1575
|
+
'',
|
|
1576
|
+
].join('\n'));
|
|
1577
|
+
process.exit(reportId ? 0 : 1);
|
|
1578
|
+
}
|
|
1579
|
+
const { values } = parseArgs({
|
|
1580
|
+
args: argv.slice(1),
|
|
1581
|
+
options: {
|
|
1582
|
+
'reports-dir': { type: 'string', default: DEFAULT_REPORTS_DIR },
|
|
1583
|
+
threshold: { type: 'string' },
|
|
1584
|
+
'trivial-diff': { type: 'string' },
|
|
1585
|
+
verbose: { type: 'boolean', default: false },
|
|
1586
|
+
},
|
|
1587
|
+
strict: false,
|
|
1588
|
+
});
|
|
1589
|
+
const { createFileStore } = await import('./server/report-store.js');
|
|
1590
|
+
const store = createFileStore(resolve(values['reports-dir']));
|
|
1591
|
+
const report = await store.get(reportId);
|
|
1592
|
+
if (!report) {
|
|
1593
|
+
console.error(`Report not found: ${reportId}`);
|
|
1594
|
+
process.exit(1);
|
|
1595
|
+
}
|
|
1596
|
+
const { computeVerdict, formatVerdictText } = await import('./eval-core/verdict.js');
|
|
1597
|
+
const result = computeVerdict(report, {
|
|
1598
|
+
ciThreshold: values.threshold != null ? Number(values.threshold) : undefined,
|
|
1599
|
+
triviallySmallDiff: values['trivial-diff'] != null ? Number(values['trivial-diff']) : undefined,
|
|
1600
|
+
});
|
|
1601
|
+
console.log(formatVerdictText(result, { verbose: Boolean(values.verbose) }));
|
|
1602
|
+
// Exit code reflects ship recommendation: 0 only on PROGRESS / SOLO-pass.
|
|
1603
|
+
// NOISE / UNDERPOWERED / CAUTIOUS / REGRESS all exit 1 so this composes
|
|
1604
|
+
// with shell `&&` chains in CI.
|
|
1605
|
+
if (result.level === 'PROGRESS') {
|
|
1606
|
+
process.exit(0);
|
|
1607
|
+
}
|
|
1608
|
+
if (result.level === 'SOLO' && result.headline.includes('PASS')) {
|
|
1609
|
+
process.exit(0);
|
|
1610
|
+
}
|
|
1611
|
+
process.exit(1);
|
|
1612
|
+
}
|
|
1613
|
+
// ---------------------------------------------------------------------------
|
|
1614
|
+
// handleDiagnose — per-sample quality diagnostics (v0.23 A)
|
|
1615
|
+
// ---------------------------------------------------------------------------
|
|
1616
|
+
async function handleDiagnose(argv) {
|
|
1617
|
+
const reportId = argv[0];
|
|
1618
|
+
if (!reportId || reportId === '--help' || reportId === '-h') {
|
|
1619
|
+
console.log([
|
|
1620
|
+
'',
|
|
1621
|
+
'Usage: omk bench diagnose <reportId> [options]',
|
|
1622
|
+
'',
|
|
1623
|
+
'诊断样本集本身的质量问题:区分度低 / 重复 / 歧义 / 成本异常 / 全 fail。',
|
|
1624
|
+
'回答"测评结论是否被坏样本污染"——与 omk bench verdict 互补。',
|
|
1625
|
+
'',
|
|
1626
|
+
'Options:',
|
|
1627
|
+
' --reports-dir <dir> report store dir',
|
|
1628
|
+
' --samples <path> 样本文件路径 (用于 near-duplicate 检测;默认从 report.meta.request 读)',
|
|
1629
|
+
' --top <n> 每类只显示前 N 个 (默认 10,0=全部)',
|
|
1630
|
+
' --duplicate-rouge <num> near-duplicate ROUGE-1 阈值 (默认 0.7)',
|
|
1631
|
+
' --ambiguous-stddev <num> 歧义阈值,judge stddev (默认 1.0,需要 --judge-repeat ≥ 2 数据)',
|
|
1632
|
+
' --cost-k <num> 成本异常倍数 vs median (默认 3)',
|
|
1633
|
+
' --latency-k <num> 耗时异常倍数 vs median (默认 3)',
|
|
1634
|
+
' --flat <num> flat_scores 分差阈值 (默认 0.5)',
|
|
1635
|
+
'',
|
|
1636
|
+
].join('\n'));
|
|
1637
|
+
process.exit(reportId ? 0 : 1);
|
|
1638
|
+
}
|
|
1639
|
+
const { values } = parseArgs({
|
|
1640
|
+
args: argv.slice(1),
|
|
1641
|
+
options: {
|
|
1642
|
+
'reports-dir': { type: 'string', default: DEFAULT_REPORTS_DIR },
|
|
1643
|
+
samples: { type: 'string' },
|
|
1644
|
+
top: { type: 'string', default: '10' },
|
|
1645
|
+
'duplicate-rouge': { type: 'string' },
|
|
1646
|
+
'ambiguous-stddev': { type: 'string' },
|
|
1647
|
+
'cost-k': { type: 'string' },
|
|
1648
|
+
'latency-k': { type: 'string' },
|
|
1649
|
+
flat: { type: 'string' },
|
|
1650
|
+
},
|
|
1651
|
+
strict: false,
|
|
1652
|
+
});
|
|
1653
|
+
const { createFileStore } = await import('./server/report-store.js');
|
|
1654
|
+
const store = createFileStore(resolve(values['reports-dir']));
|
|
1655
|
+
const report = await store.get(reportId);
|
|
1656
|
+
if (!report) {
|
|
1657
|
+
console.error(`Report not found: ${reportId}`);
|
|
1658
|
+
process.exit(1);
|
|
1659
|
+
}
|
|
1660
|
+
// Try to read the samples file for near-duplicate detection. Source order:
|
|
1661
|
+
// 1. --samples <path> override
|
|
1662
|
+
// 2. report.meta.request.samplesPath (recorded at run time)
|
|
1663
|
+
// If neither resolves to a readable file, skip near-duplicate gracefully.
|
|
1664
|
+
let samples;
|
|
1665
|
+
const samplesPath = values.samples ?? report.meta?.request?.samplesPath;
|
|
1666
|
+
if (samplesPath && existsSync(samplesPath)) {
|
|
1667
|
+
try {
|
|
1668
|
+
const { loadSamples } = await import('./inputs/load-samples.js');
|
|
1669
|
+
samples = loadSamples(samplesPath).samples;
|
|
1670
|
+
}
|
|
1671
|
+
catch (err) {
|
|
1672
|
+
process.stderr.write(`warn: 加载 samples 文件失败 (${samplesPath}): ${err.message}\n`);
|
|
1673
|
+
}
|
|
1674
|
+
}
|
|
1675
|
+
const topRaw = Number(values.top);
|
|
1676
|
+
const topN = Number.isFinite(topRaw) && topRaw > 0 ? topRaw : undefined;
|
|
1677
|
+
const { diagnoseSamples, formatSampleDiagnostics } = await import('./analysis/sample-diagnostics.js');
|
|
1678
|
+
const diag = diagnoseSamples(report, {
|
|
1679
|
+
samples,
|
|
1680
|
+
duplicateRouge: values['duplicate-rouge'] != null ? Number(values['duplicate-rouge']) : undefined,
|
|
1681
|
+
ambiguousStddev: values['ambiguous-stddev'] != null ? Number(values['ambiguous-stddev']) : undefined,
|
|
1682
|
+
costOutlierK: values['cost-k'] != null ? Number(values['cost-k']) : undefined,
|
|
1683
|
+
latencyOutlierK: values['latency-k'] != null ? Number(values['latency-k']) : undefined,
|
|
1684
|
+
flatThreshold: values.flat != null ? Number(values.flat) : undefined,
|
|
1685
|
+
});
|
|
1686
|
+
console.log(formatSampleDiagnostics(diag, { topN }));
|
|
1687
|
+
// Exit code: 0 if health ≥ 70 and no errors; 1 otherwise. CI-friendly.
|
|
1688
|
+
if (diag.totals.errors === 0 && diag.healthScore >= 70) {
|
|
1689
|
+
process.exit(0);
|
|
1690
|
+
}
|
|
1691
|
+
process.exit(1);
|
|
1692
|
+
}
|
|
1693
|
+
// ---------------------------------------------------------------------------
|
|
1694
|
+
// handleFailures — LLM-driven failure clustering (v0.23 B)
|
|
1695
|
+
// ---------------------------------------------------------------------------
|
|
1696
|
+
async function handleFailures(argv) {
|
|
1697
|
+
const reportId = argv[0];
|
|
1698
|
+
if (!reportId || reportId === '--help' || reportId === '-h') {
|
|
1699
|
+
console.log([
|
|
1700
|
+
'',
|
|
1701
|
+
'Usage: omk bench failures <reportId> [options]',
|
|
1702
|
+
'',
|
|
1703
|
+
'把已有 report 的失败样本喂给一次 LLM 调用,自动聚类 + 给修复建议。',
|
|
1704
|
+
'失败定义:compositeScore < threshold 或 ok=false。',
|
|
1705
|
+
'',
|
|
1706
|
+
'Options:',
|
|
1707
|
+
' --reports-dir <dir> report store dir',
|
|
1708
|
+
' --judge-executor <name> 执行器 (default: claude)',
|
|
1709
|
+
' --judge-model <id> 聚类用的 model (default: 沿用 report.meta.judgeModel)',
|
|
1710
|
+
' --max-clusters <n> 最多多少 cluster (default 5)',
|
|
1711
|
+
' --threshold <num> compositeScore < threshold 算失败 (default 3)',
|
|
1712
|
+
' --max-feed <n> 最多喂给 LLM 多少条 (default 50,超出取最差)',
|
|
1713
|
+
'',
|
|
1714
|
+
].join('\n'));
|
|
1715
|
+
process.exit(reportId ? 0 : 1);
|
|
1716
|
+
}
|
|
1717
|
+
const { values } = parseArgs({
|
|
1718
|
+
args: argv.slice(1),
|
|
1719
|
+
options: {
|
|
1720
|
+
'reports-dir': { type: 'string', default: DEFAULT_REPORTS_DIR },
|
|
1721
|
+
'judge-executor': { type: 'string', default: 'claude' },
|
|
1722
|
+
'judge-model': { type: 'string' },
|
|
1723
|
+
'max-clusters': { type: 'string', default: '5' },
|
|
1724
|
+
threshold: { type: 'string', default: '3' },
|
|
1725
|
+
'max-feed': { type: 'string', default: '50' },
|
|
1726
|
+
},
|
|
1727
|
+
strict: false,
|
|
1728
|
+
});
|
|
1729
|
+
const { createFileStore } = await import('./server/report-store.js');
|
|
1730
|
+
const store = createFileStore(resolve(values['reports-dir']));
|
|
1731
|
+
const report = await store.get(reportId);
|
|
1732
|
+
if (!report) {
|
|
1733
|
+
console.error(`Report not found: ${reportId}`);
|
|
1734
|
+
process.exit(1);
|
|
1735
|
+
}
|
|
1736
|
+
const judgeModel = values['judge-model'] ?? report.meta?.judgeModel;
|
|
1737
|
+
if (!judgeModel) {
|
|
1738
|
+
console.error('No judge model. Pass --judge-model <id> or ensure report has meta.judgeModel.');
|
|
1739
|
+
process.exit(1);
|
|
1740
|
+
}
|
|
1741
|
+
const { createExecutor } = await import('./executors/index.js');
|
|
1742
|
+
const executor = createExecutor(values['judge-executor']);
|
|
1743
|
+
const { clusterFailures, formatFailureClusterReport } = await import('./analysis/failure-clusterer.js');
|
|
1744
|
+
const out = await clusterFailures({
|
|
1745
|
+
report: report,
|
|
1746
|
+
executor,
|
|
1747
|
+
judgeModel,
|
|
1748
|
+
maxClusters: Number(values['max-clusters']) || 5,
|
|
1749
|
+
failureThreshold: Number(values.threshold) || 3,
|
|
1750
|
+
maxFailuresFed: Number(values['max-feed']) || 50,
|
|
1751
|
+
});
|
|
1752
|
+
console.log(formatFailureClusterReport(out));
|
|
1753
|
+
}
|
|
1000
1754
|
// ---------------------------------------------------------------------------
|
|
1001
1755
|
// Entry
|
|
1002
1756
|
// ---------------------------------------------------------------------------
|