oh-my-knowledge 0.38.0 → 0.40.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (55) hide show
  1. package/dist/assets/agent-skills/omk/references/commands.md +2 -0
  2. package/dist/authoring/evolver.js +9 -5
  3. package/dist/cli/commands/list.d.ts +4 -1
  4. package/dist/cli/commands/list.js +12 -4
  5. package/dist/cli/commands/observe/index.d.ts +9 -0
  6. package/dist/cli/commands/observe/index.js +74 -1
  7. package/dist/cli/commands/sample.d.ts +13 -0
  8. package/dist/cli/commands/sample.js +113 -16
  9. package/dist/cli/lib/cmd-flags.d.ts +1 -0
  10. package/dist/cli/lib/i18n-dict/common.d.ts +1 -1
  11. package/dist/cli/lib/i18n-dict/common.js +8 -0
  12. package/dist/cli/lib/i18n-dict/gen.d.ts +1 -1
  13. package/dist/cli/lib/i18n-dict/gen.js +8 -0
  14. package/dist/cli/lib/i18n-dict/list.d.ts +1 -1
  15. package/dist/cli/lib/i18n-dict/list.js +4 -0
  16. package/dist/cli/lib/record-evolve-outcome.js +19 -10
  17. package/dist/eval-core/bootstrap.d.ts +53 -2
  18. package/dist/eval-core/bootstrap.js +82 -5
  19. package/dist/eval-core/evaluation-reporting.js +37 -15
  20. package/dist/eval-core/execution-strategy.js +1 -0
  21. package/dist/eval-core/schema.js +5 -3
  22. package/dist/eval-core/verdict.d.ts +30 -0
  23. package/dist/eval-core/verdict.js +71 -21
  24. package/dist/grading/assertions.js +28 -0
  25. package/dist/grading/debias-validate.js +6 -2
  26. package/dist/grading/index.js +6 -1
  27. package/dist/grading/layered-scores.d.ts +15 -0
  28. package/dist/grading/layered-scores.js +64 -38
  29. package/dist/inputs/load-samples.d.ts +5 -0
  30. package/dist/inputs/load-samples.js +4 -2
  31. package/dist/inputs/skill-loader.d.ts +8 -0
  32. package/dist/inputs/skill-loader.js +23 -0
  33. package/dist/managed/evidence.d.ts +2 -1
  34. package/dist/managed/evidence.js +15 -2
  35. package/dist/managed/index.d.ts +2 -0
  36. package/dist/managed/index.js +2 -0
  37. package/dist/managed/list-view.d.ts +15 -1
  38. package/dist/managed/list-view.js +9 -1
  39. package/dist/managed/observe-feedback.d.ts +60 -0
  40. package/dist/managed/observe-feedback.js +49 -0
  41. package/dist/managed/store.d.ts +38 -1
  42. package/dist/managed/store.js +116 -3
  43. package/dist/managed/version-scores.d.ts +33 -0
  44. package/dist/managed/version-scores.js +70 -0
  45. package/dist/observability/skill-health-analyzer.d.ts +5 -0
  46. package/dist/observability/skill-health-analyzer.js +3 -2
  47. package/dist/renderer/managed-history-renderer.d.ts +2 -2
  48. package/dist/renderer/managed-history-renderer.js +189 -7
  49. package/dist/renderer/summary.js +23 -22
  50. package/dist/server/report-server.js +11 -2
  51. package/dist/types/eval.d.ts +2 -0
  52. package/dist/types/judge.d.ts +8 -0
  53. package/dist/types/managed.d.ts +40 -0
  54. package/dist/types/report.d.ts +10 -0
  55. package/package.json +3 -3
@@ -17,8 +17,14 @@
17
17
  * - bootstrapDiffCI: CI for the difference (B - A); 0 outside the CI = significant
18
18
  * - bootstrapWithMetric: generic interface so saturation analysis can reuse
19
19
  *
20
- * Reproducibility: pass a fixed `seed` to get deterministic CIs across runs.
21
- * Without a seed we use Math.random() — fine for production but not for tests.
20
+ * Reproducibility: CIs are **deterministic by default** — when no `seed` is passed,
21
+ * a fixed `DEFAULT_BOOTSTRAP_SEED` is used, so the same eval run twice yields
22
+ * byte-identical CIs (and a stable verdict near the significance boundary). This is
23
+ * a measurement-validity requirement: an unseeded `Math.random()` would let the
24
+ * `significant` flag flip between identical runs. Library callers (and specific paths
25
+ * such as `eval gold compare --seed`) may pass an explicit `seed` to vary the draw; the
26
+ * main `omk eval` deliberately exposes no seed knob — a fixed default also prevents
27
+ * seed-shopping for significance.
22
28
  */
23
29
  export interface BootstrapCI {
24
30
  /** Lower bound of the CI. */
@@ -42,6 +48,21 @@ export interface BootstrapDiffCI extends BootstrapCI {
42
48
  export declare const DEFAULT_BOOTSTRAP_SAMPLES = 1000;
43
49
  /** Default significance level; 0.05 → 95% CI. */
44
50
  export declare const DEFAULT_BOOTSTRAP_ALPHA = 0.05;
51
+ /**
52
+ * Fixed default bootstrap seed → CIs are **deterministic by default** (omk default-strict:
53
+ * reproducibility affects verdict validity, so it is on by default, not opt-in). The specific
54
+ * value is arbitrary — only that it is **fixed** matters; it is an implementation detail, not a
55
+ * user-facing constant, so unlike DEFAULT_BOOTSTRAP_SAMPLES / α it is neither cited in docs nor
56
+ * guarded by `doc-constants-drift.test.ts`. Callers wanting a different draw pass an explicit `seed`.
57
+ */
58
+ export declare const DEFAULT_BOOTSTRAP_SEED = 20260616;
59
+ /**
60
+ * Confidence-level label for display: `(1 − α)·100%`. Multiple-comparison (Bonferroni) correction
61
+ * shrinks a pairwise α to α/K and widens the CI accordingly — the label must track α so a corrected
62
+ * (wider) interval is never mislabeled "95%". No alpha (single comparison / classic A-B) ⇒ nominal 95%.
63
+ * Shared single source for the HTML renderer and the `omk eval` CLI verdict so both read one scale.
64
+ */
65
+ export declare function ciLevelLabel(alpha?: number): string;
45
66
  /**
46
67
  * Bootstrap confidence interval for the mean of a single sample.
47
68
  *
@@ -68,6 +89,36 @@ export declare function bootstrapMeanCI(scores: number[], alpha?: number, sample
68
89
  * @returns BootstrapDiffCI with low/high of (B - A) and significant flag.
69
90
  */
70
91
  export declare function bootstrapDiffCI(scoresA: number[], scoresB: number[], alpha?: number, samples?: number, seed?: number): BootstrapDiffCI;
92
+ /**
93
+ * Bootstrap CI for the *difference* of two means on **paired** observations — when A and B
94
+ * are two measurements of the **same unit** (e.g. control vs treatment on the same sample,
95
+ * or original vs alternate judge prompt on the same response). Resamples the **pair indices
96
+ * jointly** and averages each pair's `b - a`, so the within-pair correlation is preserved.
97
+ *
98
+ * Why paired (vs `bootstrapDiffCI`'s independent resampling): when A and B move together
99
+ * across units (the usual case — the same sample scored by two variants is positively
100
+ * correlated), much of each group's variance is shared and cancels in the per-pair diff.
101
+ * The independent (unpaired) bootstrap ignores that, over-states the diff's variance, and
102
+ * widens the CI — *conservative*, costing real power. Use paired whenever the design is
103
+ * paired; use `bootstrapDiffCI` only for genuinely independent groups (or where a deliberate
104
+ * conservative bias is wanted). The point estimate is identical (mean of per-pair diffs =
105
+ * difference of paired means); only the CI tightens.
106
+ *
107
+ * `significant` is derived from the **rounded** `low`/`high` (the persisted bounds), so the
108
+ * flag never contradicts what is stored / displayed: a CI that rounds to include 0 reads as
109
+ * not-significant. (Computing it on the unrounded bounds would let the JSON say `low: 0,
110
+ * significant: true` — a self-contradictory `CI=[0, …]` that downstream `computeVerdict` and
111
+ * external consumers cannot reconcile.) Matches `bootstrapDiffCI`.
112
+ *
113
+ * @param pairs Aligned observations; `a` = control/baseline, `b` = treatment. diff = b - a.
114
+ * @param alpha Significance level. Default 0.05.
115
+ * @param samples Bootstrap resamples. Default 1000.
116
+ * @param seed Optional seed (deterministic by default — see module header).
117
+ */
118
+ export declare function bootstrapPairedDiffCI(pairs: Array<{
119
+ a: number;
120
+ b: number;
121
+ }>, alpha?: number, samples?: number, seed?: number): BootstrapDiffCI;
71
122
  /**
72
123
  * Generic bootstrap CI for an arbitrary sample-level metric. Used by
73
124
  * saturation analysis to get CI on metrics like stddev or
@@ -17,8 +17,14 @@
17
17
  * - bootstrapDiffCI: CI for the difference (B - A); 0 outside the CI = significant
18
18
  * - bootstrapWithMetric: generic interface so saturation analysis can reuse
19
19
  *
20
- * Reproducibility: pass a fixed `seed` to get deterministic CIs across runs.
21
- * Without a seed we use Math.random() — fine for production but not for tests.
20
+ * Reproducibility: CIs are **deterministic by default** — when no `seed` is passed,
21
+ * a fixed `DEFAULT_BOOTSTRAP_SEED` is used, so the same eval run twice yields
22
+ * byte-identical CIs (and a stable verdict near the significance boundary). This is
23
+ * a measurement-validity requirement: an unseeded `Math.random()` would let the
24
+ * `significant` flag flip between identical runs. Library callers (and specific paths
25
+ * such as `eval gold compare --seed`) may pass an explicit `seed` to vary the draw; the
26
+ * main `omk eval` deliberately exposes no seed knob — a fixed default also prevents
27
+ * seed-shopping for significance.
22
28
  */
23
29
  /**
24
30
  * Default number of bootstrap resamples. Every eval path uses this unless
@@ -28,6 +34,24 @@
28
34
  export const DEFAULT_BOOTSTRAP_SAMPLES = 1000;
29
35
  /** Default significance level; 0.05 → 95% CI. */
30
36
  export const DEFAULT_BOOTSTRAP_ALPHA = 0.05;
37
+ /**
38
+ * Fixed default bootstrap seed → CIs are **deterministic by default** (omk default-strict:
39
+ * reproducibility affects verdict validity, so it is on by default, not opt-in). The specific
40
+ * value is arbitrary — only that it is **fixed** matters; it is an implementation detail, not a
41
+ * user-facing constant, so unlike DEFAULT_BOOTSTRAP_SAMPLES / α it is neither cited in docs nor
42
+ * guarded by `doc-constants-drift.test.ts`. Callers wanting a different draw pass an explicit `seed`.
43
+ */
44
+ export const DEFAULT_BOOTSTRAP_SEED = 20260616;
45
+ /**
46
+ * Confidence-level label for display: `(1 − α)·100%`. Multiple-comparison (Bonferroni) correction
47
+ * shrinks a pairwise α to α/K and widens the CI accordingly — the label must track α so a corrected
48
+ * (wider) interval is never mislabeled "95%". No alpha (single comparison / classic A-B) ⇒ nominal 95%.
49
+ * Shared single source for the HTML renderer and the `omk eval` CLI verdict so both read one scale.
50
+ */
51
+ export function ciLevelLabel(alpha) {
52
+ const pct = (1 - (alpha ?? DEFAULT_BOOTSTRAP_ALPHA)) * 100;
53
+ return `${Number.isInteger(pct) ? pct : Number(pct.toFixed(1))}%`;
54
+ }
31
55
  /** Mulberry32 PRNG — seedable, deterministic for tests. */
32
56
  function mulberry32(seed) {
33
57
  let s = seed >>> 0;
@@ -40,9 +64,9 @@ function mulberry32(seed) {
40
64
  };
41
65
  }
42
66
  function makeRng(seed) {
43
- if (seed == null)
44
- return Math.random;
45
- return mulberry32(seed);
67
+ // 默认确定性:无显式 seed 时退 DEFAULT_BOOTSTRAP_SEED(而非 Math.random)——否则同一 eval 两跑会得到
68
+ // 不同 CI,临界点 significant 翻转 → verdict 不可复现。见模块头 Reproducibility。
69
+ return mulberry32(seed ?? DEFAULT_BOOTSTRAP_SEED);
46
70
  }
47
71
  /** Sample n indices with replacement from [0, length) using the given PRNG. */
48
72
  function resampleIndices(length, n, rng) {
@@ -150,6 +174,59 @@ export function bootstrapDiffCI(scoresA, scoresB, alpha = DEFAULT_BOOTSTRAP_ALPH
150
174
  significant: !(low <= 0 && 0 <= high),
151
175
  };
152
176
  }
177
+ /**
178
+ * Bootstrap CI for the *difference* of two means on **paired** observations — when A and B
179
+ * are two measurements of the **same unit** (e.g. control vs treatment on the same sample,
180
+ * or original vs alternate judge prompt on the same response). Resamples the **pair indices
181
+ * jointly** and averages each pair's `b - a`, so the within-pair correlation is preserved.
182
+ *
183
+ * Why paired (vs `bootstrapDiffCI`'s independent resampling): when A and B move together
184
+ * across units (the usual case — the same sample scored by two variants is positively
185
+ * correlated), much of each group's variance is shared and cancels in the per-pair diff.
186
+ * The independent (unpaired) bootstrap ignores that, over-states the diff's variance, and
187
+ * widens the CI — *conservative*, costing real power. Use paired whenever the design is
188
+ * paired; use `bootstrapDiffCI` only for genuinely independent groups (or where a deliberate
189
+ * conservative bias is wanted). The point estimate is identical (mean of per-pair diffs =
190
+ * difference of paired means); only the CI tightens.
191
+ *
192
+ * `significant` is derived from the **rounded** `low`/`high` (the persisted bounds), so the
193
+ * flag never contradicts what is stored / displayed: a CI that rounds to include 0 reads as
194
+ * not-significant. (Computing it on the unrounded bounds would let the JSON say `low: 0,
195
+ * significant: true` — a self-contradictory `CI=[0, …]` that downstream `computeVerdict` and
196
+ * external consumers cannot reconcile.) Matches `bootstrapDiffCI`.
197
+ *
198
+ * @param pairs Aligned observations; `a` = control/baseline, `b` = treatment. diff = b - a.
199
+ * @param alpha Significance level. Default 0.05.
200
+ * @param samples Bootstrap resamples. Default 1000.
201
+ * @param seed Optional seed (deterministic by default — see module header).
202
+ */
203
+ export function bootstrapPairedDiffCI(pairs, alpha = DEFAULT_BOOTSTRAP_ALPHA, samples = DEFAULT_BOOTSTRAP_SAMPLES, seed) {
204
+ if (pairs.length === 0) {
205
+ return { low: 0, high: 0, estimate: 0, samples: 0, significant: false };
206
+ }
207
+ const n = pairs.length;
208
+ const diffs = pairs.map((p) => p.b - p.a);
209
+ const rng = makeRng(seed);
210
+ const resampleDiffMeans = new Array(samples);
211
+ for (let s = 0; s < samples; s++) {
212
+ const idx = resampleIndices(n, n, rng);
213
+ let sum = 0;
214
+ for (const i of idx)
215
+ sum += diffs[i];
216
+ resampleDiffMeans[s] = sum / n;
217
+ }
218
+ resampleDiffMeans.sort((a, b) => a - b);
219
+ const low = round4(sortedQuantile(resampleDiffMeans, alpha / 2));
220
+ const high = round4(sortedQuantile(resampleDiffMeans, 1 - alpha / 2));
221
+ return {
222
+ low,
223
+ high,
224
+ estimate: round4(mean(diffs)),
225
+ samples,
226
+ // significant 与持久化的(舍入)边界一致 —— 见函数头:绝不出现「low:0 但 significant:true」自相矛盾。
227
+ significant: !(low <= 0 && 0 <= high),
228
+ };
229
+ }
153
230
  /**
154
231
  * Generic bootstrap CI for an arbitrary sample-level metric. Used by
155
232
  * saturation analysis to get CI on metrics like stddev or
@@ -8,7 +8,7 @@ import { indexReportWrite } from './artifact-index.js';
8
8
  import { buildVariantSummary } from './schema.js';
9
9
  import { buildVariantConfig, resolveExecutionStrategy } from './execution-strategy.js';
10
10
  import { getJudgePromptHash } from '../grading/judge.js';
11
- import { bootstrapMeanCI, bootstrapDiffCI, DEFAULT_BOOTSTRAP_ALPHA, DEFAULT_BOOTSTRAP_SAMPLES, } from './bootstrap.js';
11
+ import { bootstrapMeanCI, bootstrapPairedDiffCI, DEFAULT_BOOTSTRAP_ALPHA, DEFAULT_BOOTSTRAP_SAMPLES, } from './bootstrap.js';
12
12
  import { getExecutorRuntimeFingerprint } from '../executors/runtime-fingerprint.js';
13
13
  const __dirname = dirname(fileURLToPath(import.meta.url));
14
14
  function findPackageJson(startDir) {
@@ -121,6 +121,9 @@ export function aggregateReport({ runId, variants, model, judgeModel, noJudge, e
121
121
  const bootstrapSamples = request?.bootstrapSamples ?? DEFAULT_BOOTSTRAP_SAMPLES;
122
122
  let pairComparisons;
123
123
  if (bootstrapEnabled) {
124
+ // 不变量(见 grading/layered-scores.ts):composite 在至少一层可测时恒 ≥ 1,`compositeScore === 0`
125
+ // 当且仅当该样本**无任何可测层**(真·缺测,如纯评委样本且评委失败)。故 `> 0` 过滤精确剔除非测量、
126
+ // 绝不丢"低分内容"(评委失败已在上游当缺测,不会以 0 进 composite)。下同(control / treatment)。
124
127
  for (const variant of variants) {
125
128
  const entries = Object.values(results).map((r) => r[variant]).filter(Boolean);
126
129
  const compositeScores = entries
@@ -135,23 +138,42 @@ export function aggregateReport({ runId, variants, model, judgeModel, noJudge, e
135
138
  if (variants.length >= 2) {
136
139
  pairComparisons = [];
137
140
  const controlName = variants[0];
138
- const controlEntries = Object.values(results).map((r) => r[controlName]).filter(Boolean);
139
- const controlScores = controlEntries
140
- .filter((e) => typeof e.compositeScore === 'number' && e.compositeScore > 0)
141
- .map((e) => e.compositeScore);
141
+ const sampleRecords = Object.values(results);
142
+ // **配对** diff CI:A/B 是同一批 sample 分别过 control / treatment(配对设计),按 sample 对齐 ——
143
+ // 同一 sample 上 control 与 treatment 都可测(composite > 0,见上不变量)才入对。同一 sample 两 variant
144
+ // 的分数正相关,配对 bootstrap 据此抵消共有方差、收紧 diff CI、更有功效;旧的独立(非配对)重采样高估
145
+ // 方差、CI 偏宽、保守失功效(见 bootstrapPairedDiffCI)。点估计不变,只收紧 CI。
146
+ // 先收齐每个 treatment 的配对数据;只有 ≥2 对(否则无 CI 可算)才算一个真正被检验的比较。
147
+ const eligible = [];
142
148
  for (let i = 1; i < variants.length; i++) {
143
149
  const treatmentName = variants[i];
144
- const treatmentEntries = Object.values(results).map((r) => r[treatmentName]).filter(Boolean);
145
- const treatmentScores = treatmentEntries
146
- .filter((e) => typeof e.compositeScore === 'number' && e.compositeScore > 0)
147
- .map((e) => e.compositeScore);
148
- if (controlScores.length >= 2 && treatmentScores.length >= 2) {
149
- pairComparisons.push({
150
- control: controlName,
151
- treatment: treatmentName,
152
- diffBootstrapCI: bootstrapDiffCI(controlScores, treatmentScores, DEFAULT_BOOTSTRAP_ALPHA, bootstrapSamples),
153
- });
150
+ const pairs = [];
151
+ for (const r of sampleRecords) {
152
+ const c = r[controlName];
153
+ const t = r[treatmentName];
154
+ const a = c && typeof c.compositeScore === 'number' && c.compositeScore > 0 ? c.compositeScore : undefined;
155
+ const b = t && typeof t.compositeScore === 'number' && t.compositeScore > 0 ? t.compositeScore : undefined;
156
+ if (a !== undefined && b !== undefined)
157
+ pairs.push({ a, b });
154
158
  }
159
+ if (pairs.length >= 2)
160
+ eligible.push({ treatment: treatmentName, pairs });
161
+ }
162
+ // 多重比较(Bonferroni)校正:同时检验 K 个 treatment-vs-control 假设时,family-wise 假阳性随 K 膨胀
163
+ // (computeVerdict 的 worst-case roll-up 取最差 —— 任一对假阳即拉高总判定)。每对 CI 用 α/K(K = 实际
164
+ // 产出 CI 的比较数)把 family-wise error 压回名义 α。K=1(单 treatment / 经典 A-B)即 α 不变、与历史单对
165
+ // 口径逐字节一致,此时不写 alpha 字段(渲染按名义 95% CI,既有报告 / 快照不动)。CI 与 significant 同在 α/K
166
+ // 下算,二者自洽(绝不出现 CI 含 0 却 significant 的矛盾)。注:K 很大时 α/K 落到极端分位,1000 重采样的
167
+ // 尾部分位偏粗,大 K 慎读 —— 不在本 PR 提采样数。
168
+ const familySize = eligible.length;
169
+ const perComparisonAlpha = familySize >= 1 ? DEFAULT_BOOTSTRAP_ALPHA / familySize : DEFAULT_BOOTSTRAP_ALPHA;
170
+ for (const { treatment, pairs } of eligible) {
171
+ pairComparisons.push({
172
+ control: controlName,
173
+ treatment,
174
+ diffBootstrapCI: bootstrapPairedDiffCI(pairs, perComparisonAlpha, bootstrapSamples),
175
+ ...(familySize >= 2 ? { alpha: perComparisonAlpha } : {}),
176
+ });
155
177
  }
156
178
  }
157
179
  }
@@ -66,6 +66,7 @@ export function buildVariantConfig(artifact) {
66
66
  cwd: artifact.cwd || null,
67
67
  locator: artifact.locator,
68
68
  ref: artifact.ref,
69
+ ...(artifact.resolvedCommit ? { resolvedCommit: artifact.resolvedCommit } : {}),
69
70
  // propagate skill-isolation declaration so report.meta.skillIsolation
70
71
  // 能在 evaluation-reporting 阶段从 variantConfigs 提取 (avoid re-resolving artifacts).
71
72
  ...(artifact.allowedSkills !== undefined && { allowedSkills: artifact.allowedSkills }),
@@ -88,7 +88,8 @@ export function buildVariantResult(execResult, gradeResult, options) {
88
88
  layeredScores.factScore = assertionFact != null
89
89
  ? Number(((assertionFact + hardScore) / 2).toFixed(2))
90
90
  : hardScore;
91
- // Recompute composite from updated layers (保留 0 分,仅过滤真正缺失)
91
+ // Recompute composite from updated layers. 仅过滤 null/undefined(缺测层);评委失败时 judgeScore
92
+ // 本就是 undefined(见 grading 的 score=0 修复:评委失败=缺测、不以 0 进层),不存在「0 分」要保留。
92
93
  const scores = [layeredScores.factScore, layeredScores.behaviorScore, layeredScores.judgeScore].filter((s) => s != null);
93
94
  compositeScore = scores.length > 0 ? Number((scores.reduce((a, b) => a + b, 0) / scores.length).toFixed(2)) : compositeScore;
94
95
  }
@@ -204,8 +205,9 @@ export function buildVariantSummary(entries) {
204
205
  };
205
206
  })(),
206
207
  ...(() => {
207
- // 保留 0 分用例(评委打"完全不合格"是合法低分,不是缺失)。
208
- // 仅 filter null / undefined(真正缺数据,如该 sample 无对应断言或未配 judge)。
208
+ // 各层分量为 null/undefined = 该层缺测(无对应断言 / 未配 judge / 评委失败)。评委失败已在 grading 层
209
+ // 当缺测、judgeScore 留 undefined(见 score=0 修复),不存在「0 分内容」要保留;评委有效分恒 1-5。
210
+ // filter != null 即精确剔除缺测层。
209
211
  const factScores = ok.map((e) => e.layeredScores?.factScore).filter((s) => s != null);
210
212
  const behaviorScores = ok.map((e) => e.layeredScores?.behaviorScore).filter((s) => s != null);
211
213
  const judgeScores = ok.map((e) => e.layeredScores?.judgeScore).filter((s) => s != null);
@@ -45,6 +45,16 @@ export declare const UNDERPOWERED_MIN_SAMPLES = 20;
45
45
  */
46
46
  export declare const ENSEMBLE_STRONG_PEARSON = 0.7;
47
47
  export declare const ENSEMBLE_DISSENT_PEARSON = 0.4;
48
+ /**
49
+ * Run-to-run instability threshold on the median coefficient of variation (CV = stddev/mean).
50
+ * Once stability is **actually measured** (`--repeat ≥ 2`) and the median CV exceeds this line,
51
+ * a would-be PROGRESS is downgraded to CAUTIOUS — a statistically significant but run-to-run
52
+ * irreproducible "gain" is not shippable. It is the upper bound of the 5/15% stability bands in
53
+ * `docs/specs/terminology-spec.md` §5; doc ↔ code parity is guarded by
54
+ * `test/scripts/doc-constants-drift.test.ts`. Single-run reports (stability not measured) are
55
+ * never gated by it — see `computeVerdict`.
56
+ */
57
+ export declare const STABILITY_UNSTABLE_CV = 0.15;
48
58
  export type VerdictLevel = 'PROGRESS' | 'CAUTIOUS' | 'REGRESS' | 'NOISE' | 'UNDERPOWERED' | 'SOLO';
49
59
  export interface VerdictResult {
50
60
  level: VerdictLevel;
@@ -85,6 +95,26 @@ export interface VerdictOptions {
85
95
  * Compute a verdict for a finished report. Pure function — no I/O.
86
96
  */
87
97
  export declare function computeVerdict(report: Report, options?: VerdictOptions): VerdictResult;
98
+ /**
99
+ * Stability rationale. 三种状态:
100
+ * - --repeat ≥ 2 + 有 variance 数据: 报告 CV (variation coefficient) 主指标
101
+ * - --repeat ≥ 2 但 variance 缺失: 异常,标 "—" 提示数据丢失
102
+ * - --repeat < 2: 显式说"未测量,需 --repeat ≥ 2",而不是默默不提
103
+ *
104
+ * 单轮场景关键:不是"稳定 = 100%"(常见误读),而是"测不到稳定性"。
105
+ * Verdict 必须诚实交代这个盲区,不能让用户以为没说就是 OK。
106
+ */
107
+ /**
108
+ * 跨轮稳定性的中位 CV(variation coefficient = stddev/mean)。仅在**已测**(runs≥2)且 variance 数据齐时
109
+ * 返回 { runs, cv },否则 null(单轮未测 / variance 缺失 / CV 全算不出 → 不参与门控)。formatStability 的
110
+ * 文字、computeVerdict 的稳定性门控、renderer 的 hero CV chip 共用这一处计算,口径一致、绝不漂移。
111
+ * **真·中位**:偶数个 variant 取中间两项的平均(不是上中位)—— 最常见的 A/B 报告恰好是两个 variant,取上中位
112
+ * 会退化成"较大的那个 CV",把门控口径从「中位」悄悄变成「max」、直接改变 ship/no-ship(复审 P2)。
113
+ */
114
+ export declare function medianStabilityCV(report: Report): {
115
+ runs: number;
116
+ cv: number;
117
+ } | null;
88
118
  /**
89
119
  * Plain-text formatter for the `omk eval` verdict. Stays under 6 lines per the
90
120
  * spec — one verdict + four rationale bullets + one ship recommendation.
@@ -28,6 +28,7 @@
28
28
  * empirical) is documented inline so users can audit and override.
29
29
  */
30
30
  import { evaluateLayerGates } from './layer-gates.js';
31
+ import { ciLevelLabel } from './bootstrap.js';
31
32
  /**
32
33
  * Below this sample count a non-significant diff is read as UNDERPOWERED
33
34
  * (only large effects are detectable) rather than NOISE. Matches the
@@ -45,6 +46,16 @@ export const UNDERPOWERED_MIN_SAMPLES = 20;
45
46
  */
46
47
  export const ENSEMBLE_STRONG_PEARSON = 0.7;
47
48
  export const ENSEMBLE_DISSENT_PEARSON = 0.4;
49
+ /**
50
+ * Run-to-run instability threshold on the median coefficient of variation (CV = stddev/mean).
51
+ * Once stability is **actually measured** (`--repeat ≥ 2`) and the median CV exceeds this line,
52
+ * a would-be PROGRESS is downgraded to CAUTIOUS — a statistically significant but run-to-run
53
+ * irreproducible "gain" is not shippable. It is the upper bound of the 5/15% stability bands in
54
+ * `docs/specs/terminology-spec.md` §5; doc ↔ code parity is guarded by
55
+ * `test/scripts/doc-constants-drift.test.ts`. Single-run reports (stability not measured) are
56
+ * never gated by it — see `computeVerdict`.
57
+ */
58
+ export const STABILITY_UNSTABLE_CV = 0.15;
48
59
  /**
49
60
  * Compute a verdict for a finished report. Pure function — no I/O.
50
61
  */
@@ -89,6 +100,16 @@ export function computeVerdict(report, options = {}) {
89
100
  }
90
101
  // Single representative pair for the top-level rationale (the worst one).
91
102
  const representative = perPair.find((p) => p.level === topLevel) ?? perPair[0];
103
+ // 稳定性门控(报告级,非 per-pair):仅当**已测**(runs≥2)且 run-to-run 不稳(median CV > STABILITY_UNSTABLE_CV)
104
+ // 时,把 PROGRESS 降为 CAUTIOUS —— 显著但跨轮不可复现的"进展"不可 ship。单轮(未测稳定性)不门控:rationale
105
+ // 已诚实标"未测量"(terminology-spec §5「诚实交代测不到的东西」),默认单轮全降级会过激。只压 PROGRESS:已是 CAUTIOUS/REGRESS 等
106
+ // 不再加码,顺序与 worst-case roll-up 一致。
107
+ const stab = medianStabilityCV(report);
108
+ const stabilityGated = topLevel === 'PROGRESS' && stab !== null && stab.cv > STABILITY_UNSTABLE_CV;
109
+ const level = stabilityGated ? 'CAUTIOUS' : topLevel;
110
+ const stabilityNote = stabilityGated && stab
111
+ ? ` · 显著但 run-to-run 不稳(CV=${(stab.cv * 100).toFixed(1)}% > ${(STABILITY_UNSTABLE_CV * 100).toFixed(0)}%)`
112
+ : '';
92
113
  const significance = representative
93
114
  ? formatSignificance(representative)
94
115
  : 'no pairwise comparison available — was --bootstrap used?';
@@ -96,12 +117,12 @@ export function computeVerdict(report, options = {}) {
96
117
  const sampleSize = formatSampleSize(report);
97
118
  const stability = formatStability(report);
98
119
  const judgeAgreement = formatJudgeAgreement(report);
99
- const shipRecommendation = recommendation(topLevel, perPair);
120
+ const shipRecommendation = recommendation(level, perPair);
100
121
  return {
101
- level: topLevel,
122
+ level,
102
123
  headline: representative
103
- ? `${topLevel} · ${representative.treatment} vs ${representative.control}: ${representative.headline}`
104
- : `${topLevel} · ${variants.length} variants`,
124
+ ? `${level} · ${representative.treatment} vs ${representative.control}: ${representative.headline}${stabilityNote}`
125
+ : `${level} · ${variants.length} variants`,
105
126
  perPair,
106
127
  rationale: {
107
128
  significance,
@@ -130,7 +151,10 @@ function verdictForPair(pair, summary, sampleCount, report, gateThreshold, trivi
130
151
  // Layer-gate check: did any layer fall below threshold for either variant?
131
152
  const cGate = evaluateLayerGates({ [control]: summary[control] }, gateThreshold);
132
153
  const tGate = evaluateLayerGates({ [treatment]: summary[treatment] }, gateThreshold);
133
- // No bootstrap CI available → fall back to point-estimate diff comparison.
154
+ // No bootstrap CI available → fall back to point-estimate diff comparison. 这是 `--no-bootstrap` 的**降级
155
+ // 模式**:bootstrap 默认开,正常路径永远有 CI,这里只在用户显式关掉时触达。降级路径刻意不对称——正 Δ 最多给
156
+ // CAUTIOUS(没 CI 不敢判 PROGRESS),负 Δ 直接 REGRESS(不做显著性检验)。这对"检测变差"是保守安全方向(宁可
157
+ // 误报回归也别漏掉),代价是把噪声级的负 Δ 也叫 REGRESS;要严谨结论就别关 bootstrap。
134
158
  if (!diff) {
135
159
  const cMean = avgComposite(summary[control]);
136
160
  const tMean = avgComposite(summary[treatment]);
@@ -146,7 +170,10 @@ function verdictForPair(pair, summary, sampleCount, report, gateThreshold, trivi
146
170
  headline,
147
171
  };
148
172
  }
149
- const headlineCore = `Δ=${diff.estimate >= 0 ? '+' : ''}${diff.estimate} CI=[${diff.low}, ${diff.high}]`;
173
+ // 多重比较把本对的 α 收到 α/K → CI 变宽,标签随 α 走(与 HTML pairwise 表同口径);K=1(无 alpha)不加标签,
174
+ // headline 与历史逐字节一致。让 omk eval CLI 也诚实显示真实置信水平,而非裸区间。
175
+ const ciLabel = pair.alpha != null ? `${ciLevelLabel(pair.alpha)} ` : '';
176
+ const headlineCore = `Δ=${diff.estimate >= 0 ? '+' : ''}${diff.estimate} ${ciLabel}CI=[${diff.low}, ${diff.high}]`;
150
177
  if (!diff.significant) {
151
178
  // Diff CI contains 0. Distinguish "underpowered (saturation says: more samples needed)"
152
179
  // from "noise (saturation says: we're saturated, the effect just isn't there)".
@@ -233,7 +260,10 @@ function avgComposite(s) {
233
260
  return 0;
234
261
  if (typeof s.avgCompositeScore === 'number')
235
262
  return s.avgCompositeScore;
236
- // Fallback: average of the three layers when composite isn't on the summary.
263
+ // 兜底:summary 无 avgCompositeScore 时,用三层均值近似。**有偏**:真 composite 是 per-sample
264
+ // (present-layers 均值)再跨 sample 平均;这里是「跨 sample 的层均值」再跨层平均,各层在不同 sample 上缺失
265
+ // 不均时两者不等(Jensen / 分母不一致)。仅 no-bootstrap 降级路径 + summary 缺 composite 的老报告才触达
266
+ // (默认开 bootstrap、新报告必带 avgCompositeScore,故极罕见),不值得回填 per-sample 重算 —— 标注保留近似。
237
267
  const layers = [s.avgFactScore, s.avgBehaviorScore, s.avgJudgeScore].filter((x) => typeof x === 'number');
238
268
  if (layers.length === 0)
239
269
  return 0;
@@ -289,16 +319,20 @@ function formatSampleSize(report) {
289
319
  * 单轮场景关键:不是"稳定 = 100%"(常见误读),而是"测不到稳定性"。
290
320
  * Verdict 必须诚实交代这个盲区,不能让用户以为没说就是 OK。
291
321
  */
292
- function formatStability(report) {
322
+ /**
323
+ * 跨轮稳定性的中位 CV(variation coefficient = stddev/mean)。仅在**已测**(runs≥2)且 variance 数据齐时
324
+ * 返回 { runs, cv },否则 null(单轮未测 / variance 缺失 / CV 全算不出 → 不参与门控)。formatStability 的
325
+ * 文字、computeVerdict 的稳定性门控、renderer 的 hero CV chip 共用这一处计算,口径一致、绝不漂移。
326
+ * **真·中位**:偶数个 variant 取中间两项的平均(不是上中位)—— 最常见的 A/B 报告恰好是两个 variant,取上中位
327
+ * 会退化成"较大的那个 CV",把门控口径从「中位」悄悄变成「max」、直接改变 ship/no-ship(复审 P2)。
328
+ */
329
+ export function medianStabilityCV(report) {
293
330
  const runs = report.variance?.runs ?? report.meta?.request?.repeat ?? 1;
294
- if (runs < 2) {
295
- return '稳定性未测量(单轮评测,需 --repeat ≥ 2 才能测 CV)';
296
- }
331
+ if (runs < 2)
332
+ return null;
297
333
  const variance = report.variance?.perVariant;
298
- if (!variance || Object.keys(variance).length === 0) {
299
- return `runs=${runs} 但 variance 数据缺失`;
300
- }
301
- // CV = stddev / mean,取所有 variant 的中位数(单 variant 易有 NaN/0,聚合更稳)
334
+ if (!variance || Object.keys(variance).length === 0)
335
+ return null;
302
336
  const cvs = [];
303
337
  for (const v of Object.values(variance)) {
304
338
  if (typeof v.stddev === 'number' && typeof v.mean === 'number' && v.mean > 0) {
@@ -306,12 +340,28 @@ function formatStability(report) {
306
340
  }
307
341
  }
308
342
  if (cvs.length === 0)
309
- return `runs=${runs}, CV 计算失败(stddev/mean 数据缺失)`;
343
+ return null;
310
344
  cvs.sort((a, b) => a - b);
311
- const median = cvs[Math.floor(cvs.length / 2)];
312
- const cvPct = (median * 100).toFixed(1);
313
- // 阈值参考 terminology-spec §5:<5% 稳 / 5-15% 中 / >15% 不稳
314
- const verdict = median < 0.05 ? '稳定' : median < 0.15 ? '中等' : '不稳';
345
+ const mid = Math.floor(cvs.length / 2);
346
+ const median = cvs.length % 2 === 0 ? (cvs[mid - 1] + cvs[mid]) / 2 : cvs[mid];
347
+ return { runs, cv: median };
348
+ }
349
+ function formatStability(report) {
350
+ const runs = report.variance?.runs ?? report.meta?.request?.repeat ?? 1;
351
+ if (runs < 2) {
352
+ return '稳定性未测量(单轮评测,需 --repeat ≥ 2 才能测 CV)';
353
+ }
354
+ const variance = report.variance?.perVariant;
355
+ if (!variance || Object.keys(variance).length === 0) {
356
+ return `runs=${runs} 但 variance 数据缺失`;
357
+ }
358
+ const stab = medianStabilityCV(report);
359
+ if (!stab)
360
+ return `runs=${runs}, CV 计算失败(stddev/mean 数据缺失)`;
361
+ const cvPct = (stab.cv * 100).toFixed(1);
362
+ // 阈值参考 terminology-spec §5:<5% 稳 / 5-15% 中 / >15% 不稳。不稳上界 = STABILITY_UNSTABLE_CV,
363
+ // 与门控同一根线:label 判"不稳" ⟺ 门控触发(cv > STABILITY_UNSTABLE_CV)。
364
+ const verdict = stab.cv < 0.05 ? '稳定' : stab.cv <= STABILITY_UNSTABLE_CV ? '中等' : '不稳';
315
365
  return `CV=${cvPct}% (${verdict}, runs=${runs}; 阈值 <5%=稳/5-15%=中/>15%=不稳)`;
316
366
  }
317
367
  function formatJudgeAgreement(report) {
@@ -331,7 +381,7 @@ function recommendation(level, _perPair) {
331
381
  case 'PROGRESS':
332
382
  return 'SHIP — treatment is significantly better and passes all layer gates.';
333
383
  case 'CAUTIOUS':
334
- return 'INVESTIGATE — the gain is real but at least one warning fired (broken gate, trivially small, or partial recovery). Do not ship blind.';
384
+ return 'INVESTIGATE — the gain is real but at least one warning fired (broken gate, trivially small, partial recovery, judge dissent, or run-to-run unstable). Do not ship blind.';
335
385
  case 'REGRESS':
336
386
  return 'DO NOT SHIP — treatment regresses. Check the worst layer and re-run with the fix.';
337
387
  case 'NOISE':
@@ -1,5 +1,6 @@
1
1
  import { resolve } from 'node:path';
2
2
  import _Ajv from 'ajv';
3
+ import { ASSERTION_LAYER } from './layered-scores.js';
3
4
  const Ajv = _Ajv.default ?? _Ajv;
4
5
  const ajv = new Ajv();
5
6
  const CUSTOM_ASSERTION_TIMEOUT_MS = 30_000;
@@ -266,6 +267,30 @@ function evalAssertion(output, assertion, ctx) {
266
267
  return false;
267
268
  }
268
269
  }
270
+ /**
271
+ * assert-set 是布尔组合器,不是叶子断言,没有静态的「层」。它在 runAssertions 里只产出一条聚合明细
272
+ * (`type: 'assert-set'`、aggregate pass/fail),computeLayeredScores 无法据此判 fact / behavior。
273
+ * 这里在 grading 期(能看到 children 树时)按**叶子子断言**的层解析:递归收集所有叶子(穿透嵌套 assert-set),
274
+ * - 同层(全 fact 或全 behavior)→ 返回该层:聚合 pass/fail 归这一层,与 any/all 模式无关(同层信号无歧义)。
275
+ * - 混层 / 空 / 含未分类叶子 → 返回 undefined:不归层(混层组合器塞进单层会引入口径偏差,诚实做法是不计入分层)。
276
+ * 解析结果落到 detail.layer,computeLayeredScores 优先读它。
277
+ */
278
+ function resolveAssertSetLayer(assertion) {
279
+ const layers = new Set();
280
+ const collect = (a) => {
281
+ if (a.type === 'assert-set') {
282
+ for (const c of a.children ?? [])
283
+ collect(c);
284
+ return;
285
+ }
286
+ layers.add(ASSERTION_LAYER[a.type] ?? 'unknown');
287
+ };
288
+ collect(assertion);
289
+ if (layers.size !== 1)
290
+ return undefined; // 0 / 混层 → 不归层
291
+ const [only] = [...layers];
292
+ return only === 'unknown' ? undefined : only;
293
+ }
269
294
  export function runAssertions(output, assertions, context = {}) {
270
295
  const outputLower = output.toLowerCase();
271
296
  const toolCalls = context.toolCalls || [];
@@ -276,11 +301,14 @@ export function runAssertions(output, assertions, context = {}) {
276
301
  const weight = assertion.weight ?? 1;
277
302
  const raw = evalAssertion(output, assertion, ctx);
278
303
  const passed = assertion.not ? !raw : raw;
304
+ // assert-set 组合器:在此(能看到 children)解析其层,供 computeLayeredScores 用;叶子断言按静态映射归层、不带 layer。
305
+ const layer = assertion.type === 'assert-set' ? resolveAssertSetLayer(assertion) : undefined;
279
306
  details.push({
280
307
  type: assertion.type,
281
308
  value: assertion.value ?? assertion.pattern ?? assertion.values?.join(', ') ?? '',
282
309
  weight,
283
310
  passed,
311
+ ...(layer ? { layer } : {}),
284
312
  });
285
313
  }
286
314
  const totalWeight = details.reduce((s, d) => s + d.weight, 0);
@@ -24,7 +24,7 @@
24
24
  * large/expensive evaluations can opt in deliberately.
25
25
  */
26
26
  import { llmJudge } from './judge.js';
27
- import { bootstrapDiffCI } from '../eval-core/bootstrap.js';
27
+ import { bootstrapPairedDiffCI } from '../eval-core/bootstrap.js';
28
28
  /**
29
29
  * Map a bootstrap diff CI to a verdict bucket. The ranges are deliberately
30
30
  * conservative: we only label "strong" when the CI fully sits >= |0.5| away
@@ -119,7 +119,11 @@ export async function validateLengthDebias(input) {
119
119
  }
120
120
  const meanOriginal = avg(pairs.map((p) => p.originalScore));
121
121
  const meanAlternate = avg(pairs.map((p) => p.alternateScore));
122
- const diffCI = bootstrapDiffCI(pairs.map((p) => p.originalScore), pairs.map((p) => p.alternateScore), 0.05, bootstrapSamples, seed);
122
+ // **配对** diff CI:每个 sample 同时有 original 与 alternate prompt 两个分数(同一回答、两个 judge prompt
123
+ // = 配对设计)。原先拆成两数组喂独立重采样,丢弃了配对、高估方差、CI 偏宽 —— 对一个**检测**长度偏置敏感性
124
+ // 的工具,保守方向恰好是错的(更难检出真实偏置)。改配对:重采样 sample 下标、按 (alternate − original) 算,
125
+ // 保留 within-sample 相关、收紧 CI,提升对偏置的检出力。diff = b − a = alternate − original(同原约定)。
126
+ const diffCI = bootstrapPairedDiffCI(pairs.map((p) => ({ a: p.originalScore, b: p.alternateScore })), 0.05, bootstrapSamples, seed);
123
127
  const verdict = classifyVerdict(diffCI);
124
128
  return {
125
129
  variant,
@@ -109,7 +109,12 @@ export async function grade({ output, sample, judgeModels, judgeExecutors, allow
109
109
  const judge = useEnsemble
110
110
  ? await llmJudgeEnsemble(rubricOptions, judgeModels, executorByName, judgeRepeat)
111
111
  : await llmJudgeRepeat(rubricOptions, judgeRepeat);
112
- results.llmScore = judge.score;
112
+ // judge.score <= 0 表示该样本所有判定尝试都失败(非 JSON / parse / executor 错——见 judge.ts 失败哨兵
113
+ // 268/275/298,prompt 只发 1-5),是**缺测**而非「0 分内容」。不写 llmScore(留 undefined)→ 让
114
+ // computeLayeredScores 当缺层排除,而不是把基础设施失败当内容分污染 composite(与多维度路径 filter(s>0)
115
+ // 及多轮路径 validSamples 已排除失败同口径)。llmReason / scoreSamples 仍记录,供诊断"为何失败"。
116
+ if (judge.score > 0)
117
+ results.llmScore = judge.score;
113
118
  results.llmReason = judge.reason;
114
119
  if (judge.reasoning)
115
120
  results.llmReasoning = judge.reasoning;
@@ -1,4 +1,19 @@
1
1
  import type { AssertionDetail, LayeredScores } from '../types/index.js';
2
+ /**
3
+ * 每个 assertion 类型归到事实层 / 行为层。**单一来源**:`computeLayeredScores` 据此把 assertion 明细拆进
4
+ * fact / behavior 两层算分。
5
+ * - **事实层(fact)**:测输出内容对不对 —— 命中 / 匹配 / 结构合法 / 与参考的文本相似度。
6
+ * - **行为层(behavior)**:测做事的方式 —— 长度 / 成本 / 轮次 / 工具调用 / 必经里程碑,不看内容本身。
7
+ *
8
+ * 不变量:**runner 支持的每个 assertion 类型都必须能被归层**,否则该类型的 pass/fail 会被 computeLayeredScores
9
+ * 从 fact 与 behavior 同时漏掉 —— 既不报错也不进 composite,静默丢分(曾漏掉七类:mock_hit / rouge_n_min /
10
+ * bleu_min / levenshtein_max + RAG 三件套 faithfulness / answer_relevancy / context_recall)。叶子断言在此静态
11
+ * 分类;组合器 `assert-set` 没有静态层,由 `assertions.ts` 的 resolveAssertSetLayer 在 grading 期按其叶子 children
12
+ * 解析(同层→归层、混层→不计),结果落 detail.layer。`test/grading/layered-scores-exhaustiveness.test.ts` 扫
13
+ * runner 源(evalAssertion 的 case ∪ `assertion.type ===` 组合器 ∪ ASYNC_ASSERTION_TYPES)守住:新增类型既不在本
14
+ * 映射、又不是已知组合器,即 CI 失败。
15
+ */
16
+ export declare const ASSERTION_LAYER: Record<string, 'fact' | 'behavior'>;
2
17
  interface CompositeInput {
3
18
  assertions?: {
4
19
  details?: AssertionDetail[];