oh-my-knowledge 0.40.0 → 0.42.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (71) hide show
  1. package/README.md +8 -4
  2. package/README.zh.md +8 -4
  3. package/dist/analysis/report-diagnostics.d.ts +8 -1
  4. package/dist/analysis/report-diagnostics.js +109 -1
  5. package/dist/assets/agent-skills/omk/references/commands.md +2 -2
  6. package/dist/authoring/evolver.d.ts +3 -14
  7. package/dist/authoring/evolver.js +1 -52
  8. package/dist/authoring/generator.d.ts +24 -0
  9. package/dist/authoring/generator.js +36 -8
  10. package/dist/cli/commands/eval/index.d.ts +1 -1
  11. package/dist/cli/commands/eval/index.js +50 -15
  12. package/dist/cli/commands/init.js +10 -7
  13. package/dist/cli/lib/cmd-flags.d.ts +0 -1
  14. package/dist/cli/lib/i18n-dict/init.d.ts +1 -1
  15. package/dist/cli/lib/i18n-dict/init.js +14 -11
  16. package/dist/cli/lib/i18n-dict/run.d.ts +1 -1
  17. package/dist/cli/lib/i18n-dict/run.js +6 -2
  18. package/dist/cli/lib/parse-run-config.d.ts +3 -1
  19. package/dist/cli/lib/parse-run-config.js +0 -2
  20. package/dist/eval-core/evaluation-job.d.ts +2 -2
  21. package/dist/eval-core/evaluation-job.js +2 -2
  22. package/dist/eval-core/evaluation-reporting.d.ts +0 -1
  23. package/dist/eval-core/evaluation-reporting.js +9 -41
  24. package/dist/eval-core/execution-strategy.js +3 -2
  25. package/dist/eval-core/holdout.d.ts +66 -0
  26. package/dist/eval-core/holdout.js +118 -0
  27. package/dist/eval-core/judge-independence.d.ts +28 -0
  28. package/dist/eval-core/judge-independence.js +29 -0
  29. package/dist/eval-core/verdict.d.ts +53 -2
  30. package/dist/eval-core/verdict.js +216 -16
  31. package/dist/eval-workflows/batch-evaluation-workflow.d.ts +1 -1
  32. package/dist/eval-workflows/batch-evaluation-workflow.js +1 -2
  33. package/dist/eval-workflows/evaluation-pipeline/report-finalize.d.ts +1 -5
  34. package/dist/eval-workflows/evaluation-pipeline/report-finalize.js +19 -8
  35. package/dist/eval-workflows/evaluation-pipeline/run-state.d.ts +2 -2
  36. package/dist/eval-workflows/evaluation-pipeline/run-state.js +2 -2
  37. package/dist/eval-workflows/evaluation-pipeline.d.ts +3 -2
  38. package/dist/eval-workflows/evaluation-pipeline.js +3 -4
  39. package/dist/eval-workflows/run-evaluation.d.ts +6 -4
  40. package/dist/eval-workflows/run-evaluation.js +8 -6
  41. package/dist/executors/claude-cli.js +5 -6
  42. package/dist/executors/claude-sdk.d.ts +5 -2
  43. package/dist/executors/claude-sdk.js +13 -8
  44. package/dist/executors/codex-cli.js +3 -4
  45. package/dist/executors/shared.d.ts +2 -0
  46. package/dist/executors/shared.js +15 -0
  47. package/dist/grading/assertions.js +6 -122
  48. package/dist/grading/gold-cli.js +1 -1
  49. package/dist/grading/human-gold.d.ts +5 -3
  50. package/dist/grading/human-gold.js +5 -3
  51. package/dist/grading/index.d.ts +4 -4
  52. package/dist/grading/judge.d.ts +6 -14
  53. package/dist/grading/judge.js +5 -88
  54. package/dist/inputs/eval-config.js +12 -2
  55. package/dist/managed/evidence.js +1 -2
  56. package/dist/managed/version-scores.js +1 -1
  57. package/dist/renderer/html-renderer.js +0 -9
  58. package/dist/renderer/layout.js +4 -4
  59. package/dist/renderer/summary.js +59 -4
  60. package/dist/shared/llm-prompts/debias-instructions.d.ts +4 -0
  61. package/dist/shared/llm-prompts/debias-instructions.js +44 -0
  62. package/dist/shared/llm-prompts/judge-prompts.d.ts +30 -0
  63. package/dist/shared/llm-prompts/judge-prompts.js +205 -0
  64. package/dist/shared/llm-prompts/registry.d.ts +27 -0
  65. package/dist/shared/llm-prompts/registry.js +69 -0
  66. package/dist/types/eval.d.ts +15 -8
  67. package/dist/types/judge.d.ts +1 -1
  68. package/dist/types/report.d.ts +50 -3
  69. package/package.json +1 -1
  70. package/dist/grading/debias-validate.d.ts +0 -83
  71. package/dist/grading/debias-validate.js +0 -176
@@ -0,0 +1,118 @@
1
+ /**
2
+ * Holdout split + train/holdout composite breakdown.
3
+ *
4
+ * Deterministic (no RNG) train/holdout partitioning of a sample set, plus the
5
+ * subset-composite recompute that lets `omk eval --holdout-ratio` and `omk evolve`
6
+ * score a variant on a withheld slice using the *same* aggregation as the headline
7
+ * composite (`buildVariantSummary`). Lives in eval-core so both the eval pipeline
8
+ * and the authoring/evolve loop depend *down* into it (authoring → eval-core is the
9
+ * established direction).
10
+ *
11
+ * A large train − holdout composite gap is the generalization / sample-set-overfitting
12
+ * signal the verdict's overfitting gate reads (`src/eval-core/verdict.ts`).
13
+ */
14
+ import { buildVariantSummary } from './schema.js';
15
+ /** Below this many samples on any side, a split is too small to be meaningful —
16
+ * callers fall back to full-set scoring and mark the breakdown `disabled`. */
17
+ export const MIN_HOLDOUT_SUBSET = 3;
18
+ /** Pick `count` ids at an even stride across `ids` (deterministic, no RNG) so the
19
+ * picked subset is representative of the ordering and stable across rounds/runs. */
20
+ export function pickByStride(ids, count) {
21
+ const picked = new Set();
22
+ if (count <= 0)
23
+ return picked;
24
+ const stride = ids.length / count;
25
+ for (let k = 0; k < count; k++)
26
+ picked.add(ids[Math.floor(k * stride)]);
27
+ return picked;
28
+ }
29
+ /**
30
+ * Deterministically split sample ids into train / holdout by `ratio` (fraction
31
+ * held out). Holdout members are picked at an even stride so the partition is
32
+ * representative of the ordering, and the split is stable across rounds and runs
33
+ * (no RNG). Returns null when ratio ≤ 0 or either side would drop below
34
+ * MIN_HOLDOUT_SUBSET — the caller then scores on the full set.
35
+ */
36
+ export function splitHoldout(sampleIds, ratio) {
37
+ if (!(ratio > 0) || sampleIds.length === 0)
38
+ return null;
39
+ const holdoutCount = Math.round(sampleIds.length * ratio);
40
+ const trainCount = sampleIds.length - holdoutCount;
41
+ if (holdoutCount < MIN_HOLDOUT_SUBSET || trainCount < MIN_HOLDOUT_SUBSET)
42
+ return null;
43
+ const holdoutIds = pickByStride(sampleIds, holdoutCount);
44
+ const trainIds = new Set(sampleIds.filter((id) => !holdoutIds.has(id)));
45
+ return { trainIds, holdoutIds };
46
+ }
47
+ /**
48
+ * Mean composite over the subset of a report's results whose sample_id is in
49
+ * `ids`, using the same aggregation as the full-run summary
50
+ * (`buildVariantSummary`) so train / holdout scores stay comparable to the
51
+ * headline composite. Returns 0 when the subset has no scorable entries.
52
+ */
53
+ export function subsetCompositeScore(report, variantKey, ids) {
54
+ const entries = [];
55
+ for (const r of report.results) {
56
+ if (!ids.has(r.sample_id))
57
+ continue;
58
+ const v = r.variants[variantKey];
59
+ if (v)
60
+ entries.push(v);
61
+ }
62
+ if (entries.length === 0)
63
+ return 0;
64
+ return buildVariantSummary(entries).avgCompositeScore ?? 0;
65
+ }
66
+ /**
67
+ * How many subset results actually produced a usable composite (> 0) for a variant.
68
+ * `buildVariantSummary` averages only `compositeScore > 0` entries (schema.ts), so
69
+ * the mean can rest on far fewer samples than the authored split size when runs
70
+ * flake / partial-error / budget-abort. The overfitting gate must trust THIS count,
71
+ * not the authored `trainCount` / `holdoutCount`, or a 1-of-3 holdout gets dressed
72
+ * up as a 3-sample-backed conclusion.
73
+ */
74
+ export function subsetScorableCount(report, variantKey, ids) {
75
+ let n = 0;
76
+ for (const r of report.results) {
77
+ if (!ids.has(r.sample_id))
78
+ continue;
79
+ const v = r.variants[variantKey];
80
+ if (v && typeof v.compositeScore === 'number' && v.compositeScore > 0)
81
+ n++;
82
+ }
83
+ return n;
84
+ }
85
+ /**
86
+ * Train vs holdout composite breakdown per variant for `omk eval --holdout-ratio`.
87
+ * Post-hoc — never perturbs the headline aggregation or bootstrap CI.
88
+ *
89
+ * The split is taken over `sampleIdOrder` — the **stable authored sample order**
90
+ * (the loaded `samples` file order), NOT `report.results`, whose insertion order
91
+ * is the concurrent-completion order and drifts run-to-run. Binding the stride pick
92
+ * to the authored order is what makes the holdout (and the verdict overfitting gate
93
+ * it feeds) deterministic and reproducible. Subset scores are then read from
94
+ * `report.results` by id-set membership, which is order-independent.
95
+ *
96
+ * When the split is too small on either side (< MIN_HOLDOUT_SUBSET) it returns
97
+ * `{ disabled: true }` with an empty `perVariant`, so the verdict overfitting gate
98
+ * stays inert. The testSetHash watermark (gap-spec §7.1) is attached by the caller,
99
+ * shared with gapReports.
100
+ */
101
+ export function computeHoldoutBreakdown(report, variantNames, ratio, sampleIdOrder) {
102
+ const split = splitHoldout(sampleIdOrder, ratio);
103
+ if (!split) {
104
+ return { ratio, disabled: true, perVariant: {} };
105
+ }
106
+ const perVariant = {};
107
+ for (const v of variantNames) {
108
+ perVariant[v] = {
109
+ trainScore: Number(subsetCompositeScore(report, v, split.trainIds).toFixed(4)),
110
+ holdoutScore: Number(subsetCompositeScore(report, v, split.holdoutIds).toFixed(4)),
111
+ trainCount: split.trainIds.size,
112
+ holdoutCount: split.holdoutIds.size,
113
+ trainScorable: subsetScorableCount(report, v, split.trainIds),
114
+ holdoutScorable: subsetScorableCount(report, v, split.holdoutIds),
115
+ };
116
+ }
117
+ return { ratio, perVariant };
118
+ }
@@ -0,0 +1,28 @@
1
+ import type { Report } from '../types/index.js';
2
+ import { type ExecutorVendor } from '../executors/shared.js';
3
+ /**
4
+ * 评委独立性分析(单一来源,verdict caveat 与 analysis 诊断共用)。
5
+ *
6
+ * LLM 评委有自我偏好偏置:偏爱与自己同模型家族产出的输出。omk 默认评委(claude:haiku)与默认
7
+ * 执行器(claude:*)同属一家,敞口默认就开着。本 helper 只算客观事实(评委/输出各属哪家、有没有
8
+ * 跨厂商评委、是不是单厂商 ensemble、有没有挂 gold 校准),严重度与文案交给消费方(report-diagnostics
9
+ * / verdict)决定。
10
+ *
11
+ * 缓解阶梯(行业共识):换跨厂商评委 > 跨厂商陪审团 > 人工金标校准 > 警告。omk 无法强制换 key,
12
+ * 故只检测 + 警告并指向 `--judge-models <跨厂商>` 和 `omk eval gold compare`。
13
+ */
14
+ export interface JudgeIndependence {
15
+ /** 各评委的厂商家族(与 judgeModels 同序)。 */
16
+ judgeVendors: ExecutorVendor[];
17
+ /** 被测输出涉及的厂商家族(去重)。 */
18
+ outputVendors: ExecutorVendor[];
19
+ /** 至少有一个评委的厂商不在被测输出厂商集合里 —— 存在独立(跨厂商)评委。 */
20
+ crossVendorJudgePresent: boolean;
21
+ /** 有评委、厂商全可归类、且无任何跨厂商评委 —— 自我偏好敞口(J1)。 */
22
+ sameVendorJudge: boolean;
23
+ /** ≥ 2 评委且全部同一个(已归类)厂商 —— ensemble 一致性不反驳共有偏置(J2)。 */
24
+ singleVendorEnsemble: boolean;
25
+ /** 本次 run 挂了人工 gold 校准(report.meta.humanAgreement 存在)→ 敞口有背板。 */
26
+ goldCalibrated: boolean;
27
+ }
28
+ export declare function analyzeJudgeIndependence(report: Report): JudgeIndependence;
@@ -0,0 +1,29 @@
1
+ import { executorVendor } from '../executors/shared.js';
2
+ export function analyzeJudgeIndependence(report) {
3
+ const judges = report.meta?.judgeModels ?? [];
4
+ const judgeVendors = judges.map((j) => executorVendor(j.executor));
5
+ const runtimes = report.meta?.executorRuntimes;
6
+ const outputExecutors = runtimes && Object.keys(runtimes).length > 0
7
+ ? Object.values(runtimes).map((r) => r.executor).filter((e) => !!e)
8
+ : (report.meta?.executor ? [report.meta.executor] : []);
9
+ const outputVendors = [...new Set(outputExecutors.map(executorVendor))];
10
+ const goldCalibrated = report.meta?.humanAgreement != null;
11
+ const base = { judgeVendors, outputVendors, goldCalibrated };
12
+ // 不评估(不误报)的情形:
13
+ // - noJudge:judgeModels 为审计保留(非空),但评委根本没跑、composite 无评委层 → 自我偏好 moot;
14
+ // - 无评委 / 任一厂商无法归类(自定义 script)/ 拿不到被测厂商。
15
+ const noJudge = report.meta?.noJudge === true;
16
+ const anyUnknown = judgeVendors.includes('unknown') || outputVendors.includes('unknown');
17
+ if (judges.length === 0 || noJudge || anyUnknown || outputVendors.length === 0) {
18
+ return { ...base, crossVendorJudgePresent: false, sameVendorJudge: false, singleVendorEnsemble: false };
19
+ }
20
+ const outSet = new Set(outputVendors);
21
+ const crossVendorJudgePresent = judgeVendors.some((v) => !outSet.has(v));
22
+ const singleVendorEnsemble = judges.length >= 2 && new Set(judgeVendors).size === 1;
23
+ return {
24
+ ...base,
25
+ crossVendorJudgePresent,
26
+ sameVendorJudge: !crossVendorJudgePresent,
27
+ singleVendorEnsemble,
28
+ };
29
+ }
@@ -27,7 +27,7 @@
27
27
  * proven optimal. Each rule's source (NIST AI 800-3 / Krippendorff thresholds /
28
28
  * empirical) is documented inline so users can audit and override.
29
29
  */
30
- import type { Report } from '../types/index.js';
30
+ import type { Lang, Report } from '../types/index.js';
31
31
  /**
32
32
  * Below this sample count a non-significant diff is read as UNDERPOWERED
33
33
  * (only large effects are detectable) rather than NOISE. Matches the
@@ -55,6 +55,27 @@ export declare const ENSEMBLE_DISSENT_PEARSON = 0.4;
55
55
  * never gated by it — see `computeVerdict`.
56
56
  */
57
57
  export declare const STABILITY_UNSTABLE_CV = 0.15;
58
+ /**
59
+ * Per-layer pass/fail line for the three-layer gate, on the 1-5 scale.
60
+ * **Pragmatic default, not derived from an external standard**: 3.5 is a clear margin
61
+ * above the 3.0 scale midpoint ("basically acceptable"), so a layer must land
62
+ * comfortably in the upper half to pass. Overridable via `omk eval --threshold`.
63
+ * doc ↔ code parity guarded by `test/scripts/doc-constants-drift.test.ts`.
64
+ */
65
+ export declare const DEFAULT_GATE_THRESHOLD = 3.5;
66
+ /**
67
+ * Train − holdout composite gap (1-5 scale) above which `omk eval --holdout-ratio`
68
+ * is read as **sample-set overfitting**: the gain lives on the samples the skill
69
+ * was shaped around and does not carry to the held-out slice. Like the stability
70
+ * gate, a would-be PROGRESS is then downgraded to CAUTIOUS — a win that does not
71
+ * generalize is not shippable.
72
+ * **Pragmatic default, not from an external standard**: 0.5 is 10% of the 1-5 scale
73
+ * — small enough to catch a real generalization drop, wide enough to ignore the
74
+ * sampling noise of a small holdout slice. Only fires when a holdout split is
75
+ * present (opt-in), so it never moves a default report's verdict.
76
+ * doc ↔ code parity guarded by `test/scripts/doc-constants-drift.test.ts`.
77
+ */
78
+ export declare const OVERFITTING_GAP_THRESHOLD = 0.5;
58
79
  export type VerdictLevel = 'PROGRESS' | 'CAUTIOUS' | 'REGRESS' | 'NOISE' | 'UNDERPOWERED' | 'SOLO';
59
80
  export interface VerdictResult {
60
81
  level: VerdictLevel;
@@ -76,13 +97,42 @@ export interface VerdictResult {
76
97
  * 让用户感受到 single-run 的盲区,而不是默默不提。 */
77
98
  stability?: string;
78
99
  judgeAgreement?: string;
100
+ /** Overfitting (train vs holdout) caveat — only present under `--holdout-ratio`. */
101
+ overfitting?: string;
102
+ /** Knowledge-gap caveat — informational, watermarked, never gates (gap-spec §8). */
103
+ gapSignal?: string;
79
104
  shipRecommendation?: string;
80
105
  };
106
+ /** The pair the top-level verdict is about — the worst pair from the roll-up, NOT
107
+ * variants[1]. Surfaces let the HTML pill name the right treatment in a
108
+ * control-vs-many report instead of re-deriving from the first pair. Undefined for
109
+ * SOLO / pairless reports. */
110
+ representative?: {
111
+ control: string;
112
+ treatment: string;
113
+ };
114
+ /** Structured caveats (language-neutral) so HTML / other surfaces can i18n them
115
+ * instead of re-parsing the zh `rationale` strings. Present only when the caveat
116
+ * fires; mirrors `rationale.overfitting` / `rationale.gapSignal`. */
117
+ caveats?: {
118
+ overfitting?: {
119
+ variant: string;
120
+ trainScore: number;
121
+ holdoutScore: number;
122
+ gap: number;
123
+ };
124
+ gapSignal?: {
125
+ variant: string;
126
+ gapRatePct: number;
127
+ testSetPath?: string | null;
128
+ testSetHash?: string | null;
129
+ };
130
+ };
81
131
  /** Variants present in the report (best-vs-control framing). */
82
132
  variants: string[];
83
133
  }
84
134
  export interface VerdictOptions {
85
- /** Three-layer ci-gate threshold; defaults to 3.5 (matches `omk eval`). */
135
+ /** Three-layer ci-gate threshold; defaults to DEFAULT_GATE_THRESHOLD (matches `omk eval`). */
86
136
  gateThreshold?: number;
87
137
  /**
88
138
  * Magnitude (in raw score points) below which a "significant" diff is treated
@@ -121,4 +171,5 @@ export declare function medianStabilityCV(report: Report): {
121
171
  */
122
172
  export declare function formatVerdictText(result: VerdictResult, options?: {
123
173
  verbose?: boolean;
174
+ lang?: Lang;
124
175
  }): string;
@@ -29,6 +29,8 @@
29
29
  */
30
30
  import { evaluateLayerGates } from './layer-gates.js';
31
31
  import { ciLevelLabel } from './bootstrap.js';
32
+ import { analyzeJudgeIndependence } from './judge-independence.js';
33
+ import { MIN_HOLDOUT_SUBSET } from './holdout.js';
32
34
  /**
33
35
  * Below this sample count a non-significant diff is read as UNDERPOWERED
34
36
  * (only large effects are detectable) rather than NOISE. Matches the
@@ -56,11 +58,32 @@ export const ENSEMBLE_DISSENT_PEARSON = 0.4;
56
58
  * never gated by it — see `computeVerdict`.
57
59
  */
58
60
  export const STABILITY_UNSTABLE_CV = 0.15;
61
+ /**
62
+ * Per-layer pass/fail line for the three-layer gate, on the 1-5 scale.
63
+ * **Pragmatic default, not derived from an external standard**: 3.5 is a clear margin
64
+ * above the 3.0 scale midpoint ("basically acceptable"), so a layer must land
65
+ * comfortably in the upper half to pass. Overridable via `omk eval --threshold`.
66
+ * doc ↔ code parity guarded by `test/scripts/doc-constants-drift.test.ts`.
67
+ */
68
+ export const DEFAULT_GATE_THRESHOLD = 3.5;
69
+ /**
70
+ * Train − holdout composite gap (1-5 scale) above which `omk eval --holdout-ratio`
71
+ * is read as **sample-set overfitting**: the gain lives on the samples the skill
72
+ * was shaped around and does not carry to the held-out slice. Like the stability
73
+ * gate, a would-be PROGRESS is then downgraded to CAUTIOUS — a win that does not
74
+ * generalize is not shippable.
75
+ * **Pragmatic default, not from an external standard**: 0.5 is 10% of the 1-5 scale
76
+ * — small enough to catch a real generalization drop, wide enough to ignore the
77
+ * sampling noise of a small holdout slice. Only fires when a holdout split is
78
+ * present (opt-in), so it never moves a default report's verdict.
79
+ * doc ↔ code parity guarded by `test/scripts/doc-constants-drift.test.ts`.
80
+ */
81
+ export const OVERFITTING_GAP_THRESHOLD = 0.5;
59
82
  /**
60
83
  * Compute a verdict for a finished report. Pure function — no I/O.
61
84
  */
62
85
  export function computeVerdict(report, options = {}) {
63
- const { gateThreshold = 3.5, triviallySmallDiff = 0.1 } = options;
86
+ const { gateThreshold = DEFAULT_GATE_THRESHOLD, triviallySmallDiff = 0.1 } = options;
64
87
  const variants = report.meta?.variants ?? [];
65
88
  const summary = report.summary ?? {};
66
89
  const sampleCount = report.meta?.sampleCount ?? 0;
@@ -68,16 +91,30 @@ export function computeVerdict(report, options = {}) {
68
91
  // Single-variant — no comparison possible. Just report whether the variant
69
92
  // passes its own three-layer gate.
70
93
  const gate = evaluateLayerGates(summary, gateThreshold);
94
+ // SOLO 只有绝对分、无 A/B 差值可抵消自我偏好,故同厂商评委的 caveat 更该出。
95
+ const judgeInd = judgeIndependenceCaveat(report);
96
+ // 过拟合 / gap 在 SOLO 也有意义(单变体是否泛化 / 缺口多大),但 SOLO 无 PROGRESS 可降,只附提示不门控。
97
+ const overfit = overfittingCaveat(report);
98
+ const gap = gapSignalCaveat(report);
71
99
  return {
72
100
  level: 'SOLO',
73
- headline: gate.allPass
101
+ headline: (gate.allPass
74
102
  ? `SOLO · single variant, three-layer gate PASS @ threshold ${gateThreshold}`
75
- : `SOLO · single variant, three-layer gate FAIL — see ci output`,
103
+ : `SOLO · single variant, three-layer gate FAIL — see ci output`) + judgeInd.note + overfit.note + gap.note,
76
104
  rationale: {
77
105
  layerWinners: gate.lines.join('; '),
78
106
  sampleSize: `N=${sampleCount}`,
79
107
  stability: formatStability(report),
108
+ ...(judgeInd.rationale ? { judgeAgreement: judgeInd.rationale } : {}),
109
+ ...(overfit.rationale ? { overfitting: overfit.rationale } : {}),
110
+ ...(gap.rationale ? { gapSignal: gap.rationale } : {}),
80
111
  },
112
+ ...((overfit.data || gap.data) ? {
113
+ caveats: {
114
+ ...(overfit.data ? { overfitting: overfit.data } : {}),
115
+ ...(gap.data ? { gapSignal: gap.data } : {}),
116
+ },
117
+ } : {}),
81
118
  variants,
82
119
  };
83
120
  }
@@ -106,22 +143,29 @@ export function computeVerdict(report, options = {}) {
106
143
  // 不再加码,顺序与 worst-case roll-up 一致。
107
144
  const stab = medianStabilityCV(report);
108
145
  const stabilityGated = topLevel === 'PROGRESS' && stab !== null && stab.cv > STABILITY_UNSTABLE_CV;
109
- const level = stabilityGated ? 'CAUTIOUS' : topLevel;
146
+ // 过拟合门控:与稳定性门控同形——opt-in holdout 下 train/holdout 分差过大 → PROGRESS 降 CAUTIOUS。
147
+ // overfittingCaveat 首行短路无 holdout 的报告,故默认报告 level 与 headline 逐字节不变。
148
+ const overfit = overfittingCaveat(report);
149
+ const overfitGated = topLevel === 'PROGRESS' && overfit.gated;
150
+ const level = (stabilityGated || overfitGated) ? 'CAUTIOUS' : topLevel;
110
151
  const stabilityNote = stabilityGated && stab
111
152
  ? ` · 显著但 run-to-run 不稳(CV=${(stab.cv * 100).toFixed(1)}% > ${(STABILITY_UNSTABLE_CV * 100).toFixed(0)}%)`
112
153
  : '';
154
+ const judgeInd = judgeIndependenceCaveat(report);
155
+ // gap 软提示:不改 level(spec §8),低缺口 / 无 gapReports 时空串,headline 逐字节不变。
156
+ const gap = gapSignalCaveat(report);
113
157
  const significance = representative
114
158
  ? formatSignificance(representative)
115
159
  : 'no pairwise comparison available — was --bootstrap used?';
116
160
  const layerWinners = formatLayerWinners(summary, variants);
117
161
  const sampleSize = formatSampleSize(report);
118
162
  const stability = formatStability(report);
119
- const judgeAgreement = formatJudgeAgreement(report);
163
+ const judgeAgreement = [formatJudgeAgreement(report), judgeInd.rationale].filter(Boolean).join(' · ') || undefined;
120
164
  const shipRecommendation = recommendation(level, perPair);
121
165
  return {
122
166
  level,
123
167
  headline: representative
124
- ? `${level} · ${representative.treatment} vs ${representative.control}: ${representative.headline}${stabilityNote}`
168
+ ? `${level} · ${representative.treatment} vs ${representative.control}: ${representative.headline}${stabilityNote}${judgeInd.note}${overfit.note}${gap.note}`
125
169
  : `${level} · ${variants.length} variants`,
126
170
  perPair,
127
171
  rationale: {
@@ -130,8 +174,17 @@ export function computeVerdict(report, options = {}) {
130
174
  sampleSize,
131
175
  stability,
132
176
  judgeAgreement,
177
+ ...(overfit.rationale ? { overfitting: overfit.rationale } : {}),
178
+ ...(gap.rationale ? { gapSignal: gap.rationale } : {}),
133
179
  shipRecommendation,
134
180
  },
181
+ ...(representative ? { representative: { control: representative.control, treatment: representative.treatment } } : {}),
182
+ ...((overfit.data || gap.data) ? {
183
+ caveats: {
184
+ ...(overfit.data ? { overfitting: overfit.data } : {}),
185
+ ...(gap.data ? { gapSignal: gap.data } : {}),
186
+ },
187
+ } : {}),
135
188
  variants,
136
189
  };
137
190
  }
@@ -376,12 +429,150 @@ function formatJudgeAgreement(report) {
376
429
  : 'poor';
377
430
  return `α=${Number.isNaN(a.alpha) ? 'NaN' : a.alpha.toFixed(2)} (${verdict}) vs gold ${a.goldAnnotator}`;
378
431
  }
379
- function recommendation(level, _perPair) {
432
+ /**
433
+ * 评委独立性 caveat(自我偏好 J1 / 单厂商 ensemble J2)。返回 { note, rationale }:
434
+ * note —— 追到 headline 的短提示(仅未 gold 校准时出,提醒读者敞口);
435
+ * rationale —— 并进 rationale.judgeAgreement 的可执行说明(指向跨厂商评委 / gold)。
436
+ * **不改 verdict level**:omk 固定模型,自我偏好对 baseline / treatment 两臂同等加成、在 verdict
437
+ * 在意的 A/B 差值里大幅抵消,不该翻 ship/no-ship;真正受影响的是绝对分 / 版本曲线 / 跨模型比较。
438
+ */
439
+ function judgeIndependenceCaveat(report) {
440
+ const ind = analyzeJudgeIndependence(report);
441
+ const reasons = [];
442
+ if (ind.sameVendorJudge)
443
+ reasons.push(`评委与被测输出同厂商(${ind.outputVendors.join('/')})`);
444
+ if (ind.singleVendorEnsemble)
445
+ reasons.push(`${ind.judgeVendors.length} 个评委同厂商,ensemble 一致性不反驳同模型偏置`);
446
+ if (reasons.length === 0)
447
+ return { note: '' };
448
+ if (ind.goldCalibrated) {
449
+ // 有 gold 校准背板 → 不进 headline,只软提示。
450
+ return { note: '', rationale: `自我偏好敞口(${reasons.join(';')})已有 gold 校准背板` };
451
+ }
452
+ return {
453
+ note: ' · 评委自我偏好敞口未校准',
454
+ rationale: `${reasons.join(';')} —— 绝对分可能偏高;换跨厂商评委(--judge-models)或挂 gold(omk eval gold compare)校准`,
455
+ };
456
+ }
457
+ /**
458
+ * Overfitting caveat from the opt-in train/holdout breakdown (`--holdout-ratio`).
459
+ * `gated` drives a PROGRESS → CAUTIOUS downgrade (a win that does not carry to the
460
+ * held-out slice is not shippable), mirroring the stability gate. **First line
461
+ * short-circuits when there is no holdout split**, so default reports (no
462
+ * `analysis.holdout`) are byte-identical — the gate only ever fires when the user
463
+ * opted into a holdout, which is brand-new behaviour with no historical reports.
464
+ */
465
+ /**
466
+ * The treatment variants a report-level caveat should scan. control = variants[0];
467
+ * treatments = the rest. For a single-variant (SOLO) report there is no control,
468
+ * so the lone variant is itself the subject. Scanning **all** treatments (not just
469
+ * variants[1]) keeps the caveats aligned with the verdict's worst-case roll-up over
470
+ * `perPair` — a control-vs-many report must not miss the 2nd/3rd treatment.
471
+ */
472
+ function treatmentVariants(report) {
473
+ const variants = report.meta?.variants ?? [];
474
+ return variants.length >= 2 ? variants.slice(1) : variants.slice(0, 1);
475
+ }
476
+ function overfittingCaveat(report) {
477
+ const holdout = report.analysis?.holdout;
478
+ if (!holdout || holdout.disabled)
479
+ return { note: '', gated: false };
480
+ // Worst-case over all treatments — matches the verdict roll-up. A PROGRESS top-level
481
+ // means every pair passed, so a single overfitting treatment must still downgrade it.
482
+ let worst = null;
483
+ for (const t of treatmentVariants(report)) {
484
+ const pv = holdout.perVariant[t];
485
+ if (!pv)
486
+ continue;
487
+ // 门控绑「实际可评分条目数」,不是 authored 切分数:某侧 3 条里只有 1 条真出分(其余
488
+ // error / budget-abort)时,分差只有 1 个样本支撑,把它包装成「3 条结论」会误判过拟合。
489
+ // 两侧 scorable 都 ≥ MIN_HOLDOUT_SUBSET 才信。这也顺带挡掉 score=0(scorable=0)的测量假象。
490
+ if (pv.trainScorable < MIN_HOLDOUT_SUBSET || pv.holdoutScorable < MIN_HOLDOUT_SUBSET)
491
+ continue;
492
+ const gap = pv.trainScore - pv.holdoutScore;
493
+ if (gap <= OVERFITTING_GAP_THRESHOLD)
494
+ continue;
495
+ if (!worst || gap > worst.gap)
496
+ worst = { variant: t, train: pv.trainScore, holdout: pv.holdoutScore, gap };
497
+ }
498
+ if (!worst)
499
+ return { note: '', gated: false };
500
+ return {
501
+ note: ` · 过拟合敞口(${worst.variant}: train ${worst.train.toFixed(2)} − holdout ${worst.holdout.toFixed(2)} = ${worst.gap.toFixed(2)} > ${OVERFITTING_GAP_THRESHOLD})`,
502
+ rationale: `${worst.variant} train/holdout 综合分差 ${worst.gap.toFixed(2)} 超阈值 ${OVERFITTING_GAP_THRESHOLD}(holdout ratio ${holdout.ratio})—— 提升可能是对用例集过拟合、对 holdout 不泛化;扩充用例集或换独立外验集复核`,
503
+ gated: true,
504
+ data: { variant: worst.variant, trainScore: worst.train, holdoutScore: worst.holdout, gap: worst.gap },
505
+ };
506
+ }
507
+ /**
508
+ * Knowledge-gap rate above which the verdict appends an **informational** caveat.
509
+ * Internal-only, deliberately NOT exported: gap rate is informational, never a
510
+ * gate (knowledge-gap-signal-spec.md §8), so exposing this as a tunable constant
511
+ * would invite reading it as a pass/fail line.
512
+ */
513
+ const GAP_CAVEAT_BAND = 0.2;
514
+ /**
515
+ * Knowledge-gap caveat (`report.analysis.gapReports`). **Soft only — never changes
516
+ * the verdict level** (spec §8: a nudge, not a fail). Surfaces the treatment's gap
517
+ * rate with its mandatory test-set watermark (spec §7.1: a gap number without a
518
+ * watermark is invalid output) plus the "informational, not completeness" framing.
519
+ * Empty note below the band, so the common low-gap case keeps verdict headlines
520
+ * byte-identical.
521
+ */
522
+ function gapSignalCaveat(report) {
523
+ const gapReports = report.analysis?.gapReports;
524
+ if (!gapReports)
525
+ return { note: '' };
526
+ // Worst-case (highest gap rate) over all treatments — a control-vs-many report must
527
+ // surface the noisiest treatment, not just variants[1]. spec §7.1: gap 数必须带
528
+ // test-set 水印,否则视为无效输出 —— 无水印的条目不参与(而非吐裸缺口率)。生产报告
529
+ // 恒带水印(report-finalize 强制),此处只防手搓 / 退化报告。
530
+ let gr = null;
531
+ for (const t of treatmentVariants(report)) {
532
+ const cand = gapReports[t];
533
+ if (!cand || cand.gapRate < GAP_CAVEAT_BAND)
534
+ continue;
535
+ if (!cand.testSetPath && !cand.testSetHash)
536
+ continue;
537
+ if (!gr || cand.gapRate > gr.gapRate)
538
+ gr = cand;
539
+ }
540
+ if (!gr)
541
+ return { note: '' };
542
+ const pct = (gr.gapRate * 100).toFixed(0);
543
+ const shortHash = gr.testSetHash ? gr.testSetHash.slice(0, 8) : '';
544
+ const tag = shortHash || gr.testSetPath;
545
+ const watermark = gr.testSetPath
546
+ ? `${gr.testSetPath}${shortHash ? ` @ ${shortHash}` : ''}`
547
+ : shortHash;
548
+ return {
549
+ note: ` · 知识缺口率 ${pct}% @ ${tag}`,
550
+ rationale: `知识缺口率 ${pct}%(test set: ${watermark},N=${gr.sampleCount})—— informational,反映当前用例集与知识库的交互、非完备性度量;高缺口提示扩充知识库或复核未覆盖文件`,
551
+ data: { variant: gr.variant, gapRatePct: Number(pct), testSetPath: gr.testSetPath, testSetHash: gr.testSetHash },
552
+ };
553
+ }
554
+ function recommendation(level, _perPair, lang = 'en') {
555
+ if (lang === 'zh') {
556
+ switch (level) {
557
+ case 'PROGRESS':
558
+ return '可发布 —— 实验组显著更优,且通过所有分层门控。';
559
+ case 'CAUTIOUS':
560
+ return '需排查 —— 提升是真的,但至少触发了一条告警(门控破损 / 提升微不足道 / 仅部分恢复 / 评委分歧 / 跨轮不稳 / 训练-留出过拟合)。不要盲发。';
561
+ case 'REGRESS':
562
+ return '勿发布 —— 实验组退步。检查最差的那一层,修好再重跑。';
563
+ case 'NOISE':
564
+ return '不下结论 —— 差异置信区间跨过 0,当前 N 下分辨不出效果。';
565
+ case 'UNDERPOWERED':
566
+ return '数据不足 —— 增加用例数(建议 2× 当前)后重跑。';
567
+ case 'SOLO':
568
+ return '缺对照 —— 单变体报告。用 --control baseline --treatment <名字> 重跑。';
569
+ }
570
+ }
380
571
  switch (level) {
381
572
  case 'PROGRESS':
382
573
  return 'SHIP — treatment is significantly better and passes all layer gates.';
383
574
  case 'CAUTIOUS':
384
- return 'INVESTIGATE — the gain is real but at least one warning fired (broken gate, trivially small, partial recovery, judge dissent, or run-to-run unstable). Do not ship blind.';
575
+ return 'INVESTIGATE — the gain is real but at least one warning fired (broken gate, trivially small, partial recovery, judge dissent, run-to-run unstable, or train/holdout overfitting). Do not ship blind.';
385
576
  case 'REGRESS':
386
577
  return 'DO NOT SHIP — treatment regresses. Check the worst layer and re-run with the fix.';
387
578
  case 'NOISE':
@@ -397,22 +588,31 @@ function recommendation(level, _perPair) {
397
588
  * spec — one verdict + four rationale bullets + one ship recommendation.
398
589
  */
399
590
  export function formatVerdictText(result, options = {}) {
591
+ // lang 默认 'en':保留既有英文输出逐字节不变(verdict.test 与历史 CLI 行为)。zh 只本地化
592
+ // 标签与 ship 建议;headline 是 Δ/CI/N 统计记号 —— 跨语言中性、且会随 report 持久化,
593
+ // 不翻译(翻它=改可比性锚点)。recommendation 在 format 时按 lang 重新派生,不动 computeVerdict。
594
+ const zh = options.lang === 'zh';
400
595
  const lines = [];
401
- lines.push(`Verdict: ${result.level}`);
596
+ lines.push(zh ? `判定:${result.level}` : `Verdict: ${result.level}`);
402
597
  lines.push(` ${result.headline}`);
403
598
  if (result.rationale.layerWinners)
404
- lines.push(` Layer winners: ${result.rationale.layerWinners}`);
599
+ lines.push(zh ? ` 分层优胜:${result.rationale.layerWinners}` : ` Layer winners: ${result.rationale.layerWinners}`);
405
600
  if (result.rationale.sampleSize)
406
- lines.push(` Sample size: ${result.rationale.sampleSize}`);
601
+ lines.push(zh ? ` 用例规模:${result.rationale.sampleSize}` : ` Sample size: ${result.rationale.sampleSize}`);
407
602
  if (result.rationale.stability)
408
- lines.push(` Stability: ${result.rationale.stability}`);
603
+ lines.push(zh ? ` 跨轮稳定:${result.rationale.stability}` : ` Stability: ${result.rationale.stability}`);
409
604
  if (result.rationale.judgeAgreement)
410
- lines.push(` Judge α: ${result.rationale.judgeAgreement}`);
411
- if (result.rationale.shipRecommendation)
412
- lines.push(` ${result.rationale.shipRecommendation}`);
605
+ lines.push(zh ? ` 评委 α:${result.rationale.judgeAgreement}` : ` Judge α: ${result.rationale.judgeAgreement}`);
606
+ if (result.rationale.overfitting)
607
+ lines.push(zh ? ` 过拟合:${result.rationale.overfitting}` : ` Overfitting: ${result.rationale.overfitting}`);
608
+ if (result.rationale.gapSignal)
609
+ lines.push(zh ? ` 知识缺口:${result.rationale.gapSignal}` : ` Gap signal: ${result.rationale.gapSignal}`);
610
+ if (result.rationale.shipRecommendation) {
611
+ lines.push(` ${zh ? recommendation(result.level, [], 'zh') : result.rationale.shipRecommendation}`);
612
+ }
413
613
  if (options.verbose && result.perPair && result.perPair.length > 1) {
414
614
  lines.push('');
415
- lines.push(' Per-pair detail:');
615
+ lines.push(zh ? ' 逐对明细:' : ' Per-pair detail:');
416
616
  for (const p of result.perPair) {
417
617
  lines.push(` ${p.level}: ${p.treatment} vs ${p.control} — ${p.headline}`);
418
618
  }
@@ -52,7 +52,7 @@ interface CompletedBatchSkillRun {
52
52
  * 导致 #183 角色误绑各存一份)。
53
53
  * 两个 variant 的 allowedSkills 都从 eval.yaml variants[].allowedSkills 取:treatment 按 skill
54
54
  * 名 entry.name 查、baseline 按保留名 `baseline` 查,挂到对应 spec 上由 prepareEvaluationRun
55
- * 统一绑定。baseline 的显式声明(白名单或 `[]`)必须保留——eval.yaml variant.allowedSkills
55
+ * 统一绑定。baseline 的显式声明(`[]`)必须保留——eval.yaml variant.allowedSkills
56
56
  * 优先于 strictBaseline 默认,漏挂会让 `--batch --config` 的 baseline 隔离配置静默失效。 */
57
57
  export declare function buildBatchVariantSpecs(entry: {
58
58
  name: string;
@@ -10,7 +10,7 @@ import { DEFAULT_JOBS_DIR } from '../eval-core/default-dirs.js';
10
10
  * 导致 #183 角色误绑各存一份)。
11
11
  * 两个 variant 的 allowedSkills 都从 eval.yaml variants[].allowedSkills 取:treatment 按 skill
12
12
  * 名 entry.name 查、baseline 按保留名 `baseline` 查,挂到对应 spec 上由 prepareEvaluationRun
13
- * 统一绑定。baseline 的显式声明(白名单或 `[]`)必须保留——eval.yaml variant.allowedSkills
13
+ * 统一绑定。baseline 的显式声明(`[]`)必须保留——eval.yaml variant.allowedSkills
14
14
  * 优先于 strictBaseline 默认,漏挂会让 `--batch --config` 的 baseline 隔离配置静默失效。 */
15
15
  export function buildBatchVariantSpecs(entry, variantAllowedSkills) {
16
16
  const baselineAllowed = variantAllowedSkills?.baseline;
@@ -98,7 +98,6 @@ export function buildBatchEvaluationReport({ batchRunId, skillDir, skillEntries,
98
98
  timeoutMs,
99
99
  noCache,
100
100
  dryRun: false,
101
- blind: false,
102
101
  project,
103
102
  owner,
104
103
  tags,
@@ -9,20 +9,16 @@
9
9
  * - 可选 gapReports: 文本信号(markers / hedging)始终适用,与 tool trace 无关;
10
10
  * 带上 testSetHash 水印(spec §7.1 强制要求)
11
11
  *
12
- * 最后一步 `applyBlindMode` 是 blind 模式下的字段脱敏,放在所有 analysis 之后,
13
- * 避免脱敏后字段被分析逻辑读取。
14
- *
15
12
  * 仅被 orchestrator 调用;独立拆出主要为让 orchestrator 的 try-finally 主干
16
13
  * 看起来纯粹是「执行→收尾」时序。
17
14
  */
18
15
  import type { Artifact, Report, Sample, VariantResult } from '../../types/index.js';
19
16
  type EvaluationResults = Record<string, Record<string, VariantResult>>;
20
- export declare function finalizeEvaluationReport({ report, results, artifacts, variantNames, blind, samplesPath, samplesSourceFiles, samples, }: {
17
+ export declare function finalizeEvaluationReport({ report, results, artifacts, variantNames, samplesPath, samplesSourceFiles, samples, }: {
21
18
  report: Report;
22
19
  results: EvaluationResults;
23
20
  artifacts: Artifact[];
24
21
  variantNames: string[];
25
- blind: boolean;
26
22
  samplesPath: string;
27
23
  /** 目录模式下,bundle 内所有源文件;单文件模式下 [samplesPath]。computeTestSetHash 用。 */
28
24
  samplesSourceFiles?: string[];