oh-my-knowledge 0.38.0 → 0.40.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/assets/agent-skills/omk/references/commands.md +2 -0
- package/dist/authoring/evolver.js +9 -5
- package/dist/cli/commands/list.d.ts +4 -1
- package/dist/cli/commands/list.js +12 -4
- package/dist/cli/commands/observe/index.d.ts +9 -0
- package/dist/cli/commands/observe/index.js +74 -1
- package/dist/cli/commands/sample.d.ts +13 -0
- package/dist/cli/commands/sample.js +113 -16
- package/dist/cli/lib/cmd-flags.d.ts +1 -0
- package/dist/cli/lib/i18n-dict/common.d.ts +1 -1
- package/dist/cli/lib/i18n-dict/common.js +8 -0
- package/dist/cli/lib/i18n-dict/gen.d.ts +1 -1
- package/dist/cli/lib/i18n-dict/gen.js +8 -0
- package/dist/cli/lib/i18n-dict/list.d.ts +1 -1
- package/dist/cli/lib/i18n-dict/list.js +4 -0
- package/dist/cli/lib/record-evolve-outcome.js +19 -10
- package/dist/eval-core/bootstrap.d.ts +53 -2
- package/dist/eval-core/bootstrap.js +82 -5
- package/dist/eval-core/evaluation-reporting.js +37 -15
- package/dist/eval-core/execution-strategy.js +1 -0
- package/dist/eval-core/schema.js +5 -3
- package/dist/eval-core/verdict.d.ts +30 -0
- package/dist/eval-core/verdict.js +71 -21
- package/dist/grading/assertions.js +28 -0
- package/dist/grading/debias-validate.js +6 -2
- package/dist/grading/index.js +6 -1
- package/dist/grading/layered-scores.d.ts +15 -0
- package/dist/grading/layered-scores.js +64 -38
- package/dist/inputs/load-samples.d.ts +5 -0
- package/dist/inputs/load-samples.js +4 -2
- package/dist/inputs/skill-loader.d.ts +8 -0
- package/dist/inputs/skill-loader.js +23 -0
- package/dist/managed/evidence.d.ts +2 -1
- package/dist/managed/evidence.js +15 -2
- package/dist/managed/index.d.ts +2 -0
- package/dist/managed/index.js +2 -0
- package/dist/managed/list-view.d.ts +15 -1
- package/dist/managed/list-view.js +9 -1
- package/dist/managed/observe-feedback.d.ts +60 -0
- package/dist/managed/observe-feedback.js +49 -0
- package/dist/managed/store.d.ts +38 -1
- package/dist/managed/store.js +116 -3
- package/dist/managed/version-scores.d.ts +33 -0
- package/dist/managed/version-scores.js +70 -0
- package/dist/observability/skill-health-analyzer.d.ts +5 -0
- package/dist/observability/skill-health-analyzer.js +3 -2
- package/dist/renderer/managed-history-renderer.d.ts +2 -2
- package/dist/renderer/managed-history-renderer.js +189 -7
- package/dist/renderer/summary.js +23 -22
- package/dist/server/report-server.js +11 -2
- package/dist/types/eval.d.ts +2 -0
- package/dist/types/judge.d.ts +8 -0
- package/dist/types/managed.d.ts +40 -0
- package/dist/types/report.d.ts +10 -0
- package/package.json +3 -3
|
@@ -17,8 +17,14 @@
|
|
|
17
17
|
* - bootstrapDiffCI: CI for the difference (B - A); 0 outside the CI = significant
|
|
18
18
|
* - bootstrapWithMetric: generic interface so saturation analysis can reuse
|
|
19
19
|
*
|
|
20
|
-
* Reproducibility:
|
|
21
|
-
*
|
|
20
|
+
* Reproducibility: CIs are **deterministic by default** — when no `seed` is passed,
|
|
21
|
+
* a fixed `DEFAULT_BOOTSTRAP_SEED` is used, so the same eval run twice yields
|
|
22
|
+
* byte-identical CIs (and a stable verdict near the significance boundary). This is
|
|
23
|
+
* a measurement-validity requirement: an unseeded `Math.random()` would let the
|
|
24
|
+
* `significant` flag flip between identical runs. Library callers (and specific paths
|
|
25
|
+
* such as `eval gold compare --seed`) may pass an explicit `seed` to vary the draw; the
|
|
26
|
+
* main `omk eval` deliberately exposes no seed knob — a fixed default also prevents
|
|
27
|
+
* seed-shopping for significance.
|
|
22
28
|
*/
|
|
23
29
|
export interface BootstrapCI {
|
|
24
30
|
/** Lower bound of the CI. */
|
|
@@ -42,6 +48,21 @@ export interface BootstrapDiffCI extends BootstrapCI {
|
|
|
42
48
|
export declare const DEFAULT_BOOTSTRAP_SAMPLES = 1000;
|
|
43
49
|
/** Default significance level; 0.05 → 95% CI. */
|
|
44
50
|
export declare const DEFAULT_BOOTSTRAP_ALPHA = 0.05;
|
|
51
|
+
/**
|
|
52
|
+
* Fixed default bootstrap seed → CIs are **deterministic by default** (omk default-strict:
|
|
53
|
+
* reproducibility affects verdict validity, so it is on by default, not opt-in). The specific
|
|
54
|
+
* value is arbitrary — only that it is **fixed** matters; it is an implementation detail, not a
|
|
55
|
+
* user-facing constant, so unlike DEFAULT_BOOTSTRAP_SAMPLES / α it is neither cited in docs nor
|
|
56
|
+
* guarded by `doc-constants-drift.test.ts`. Callers wanting a different draw pass an explicit `seed`.
|
|
57
|
+
*/
|
|
58
|
+
export declare const DEFAULT_BOOTSTRAP_SEED = 20260616;
|
|
59
|
+
/**
|
|
60
|
+
* Confidence-level label for display: `(1 − α)·100%`. Multiple-comparison (Bonferroni) correction
|
|
61
|
+
* shrinks a pairwise α to α/K and widens the CI accordingly — the label must track α so a corrected
|
|
62
|
+
* (wider) interval is never mislabeled "95%". No alpha (single comparison / classic A-B) ⇒ nominal 95%.
|
|
63
|
+
* Shared single source for the HTML renderer and the `omk eval` CLI verdict so both read one scale.
|
|
64
|
+
*/
|
|
65
|
+
export declare function ciLevelLabel(alpha?: number): string;
|
|
45
66
|
/**
|
|
46
67
|
* Bootstrap confidence interval for the mean of a single sample.
|
|
47
68
|
*
|
|
@@ -68,6 +89,36 @@ export declare function bootstrapMeanCI(scores: number[], alpha?: number, sample
|
|
|
68
89
|
* @returns BootstrapDiffCI with low/high of (B - A) and significant flag.
|
|
69
90
|
*/
|
|
70
91
|
export declare function bootstrapDiffCI(scoresA: number[], scoresB: number[], alpha?: number, samples?: number, seed?: number): BootstrapDiffCI;
|
|
92
|
+
/**
|
|
93
|
+
* Bootstrap CI for the *difference* of two means on **paired** observations — when A and B
|
|
94
|
+
* are two measurements of the **same unit** (e.g. control vs treatment on the same sample,
|
|
95
|
+
* or original vs alternate judge prompt on the same response). Resamples the **pair indices
|
|
96
|
+
* jointly** and averages each pair's `b - a`, so the within-pair correlation is preserved.
|
|
97
|
+
*
|
|
98
|
+
* Why paired (vs `bootstrapDiffCI`'s independent resampling): when A and B move together
|
|
99
|
+
* across units (the usual case — the same sample scored by two variants is positively
|
|
100
|
+
* correlated), much of each group's variance is shared and cancels in the per-pair diff.
|
|
101
|
+
* The independent (unpaired) bootstrap ignores that, over-states the diff's variance, and
|
|
102
|
+
* widens the CI — *conservative*, costing real power. Use paired whenever the design is
|
|
103
|
+
* paired; use `bootstrapDiffCI` only for genuinely independent groups (or where a deliberate
|
|
104
|
+
* conservative bias is wanted). The point estimate is identical (mean of per-pair diffs =
|
|
105
|
+
* difference of paired means); only the CI tightens.
|
|
106
|
+
*
|
|
107
|
+
* `significant` is derived from the **rounded** `low`/`high` (the persisted bounds), so the
|
|
108
|
+
* flag never contradicts what is stored / displayed: a CI that rounds to include 0 reads as
|
|
109
|
+
* not-significant. (Computing it on the unrounded bounds would let the JSON say `low: 0,
|
|
110
|
+
* significant: true` — a self-contradictory `CI=[0, …]` that downstream `computeVerdict` and
|
|
111
|
+
* external consumers cannot reconcile.) Matches `bootstrapDiffCI`.
|
|
112
|
+
*
|
|
113
|
+
* @param pairs Aligned observations; `a` = control/baseline, `b` = treatment. diff = b - a.
|
|
114
|
+
* @param alpha Significance level. Default 0.05.
|
|
115
|
+
* @param samples Bootstrap resamples. Default 1000.
|
|
116
|
+
* @param seed Optional seed (deterministic by default — see module header).
|
|
117
|
+
*/
|
|
118
|
+
export declare function bootstrapPairedDiffCI(pairs: Array<{
|
|
119
|
+
a: number;
|
|
120
|
+
b: number;
|
|
121
|
+
}>, alpha?: number, samples?: number, seed?: number): BootstrapDiffCI;
|
|
71
122
|
/**
|
|
72
123
|
* Generic bootstrap CI for an arbitrary sample-level metric. Used by
|
|
73
124
|
* saturation analysis to get CI on metrics like stddev or
|
|
@@ -17,8 +17,14 @@
|
|
|
17
17
|
* - bootstrapDiffCI: CI for the difference (B - A); 0 outside the CI = significant
|
|
18
18
|
* - bootstrapWithMetric: generic interface so saturation analysis can reuse
|
|
19
19
|
*
|
|
20
|
-
* Reproducibility:
|
|
21
|
-
*
|
|
20
|
+
* Reproducibility: CIs are **deterministic by default** — when no `seed` is passed,
|
|
21
|
+
* a fixed `DEFAULT_BOOTSTRAP_SEED` is used, so the same eval run twice yields
|
|
22
|
+
* byte-identical CIs (and a stable verdict near the significance boundary). This is
|
|
23
|
+
* a measurement-validity requirement: an unseeded `Math.random()` would let the
|
|
24
|
+
* `significant` flag flip between identical runs. Library callers (and specific paths
|
|
25
|
+
* such as `eval gold compare --seed`) may pass an explicit `seed` to vary the draw; the
|
|
26
|
+
* main `omk eval` deliberately exposes no seed knob — a fixed default also prevents
|
|
27
|
+
* seed-shopping for significance.
|
|
22
28
|
*/
|
|
23
29
|
/**
|
|
24
30
|
* Default number of bootstrap resamples. Every eval path uses this unless
|
|
@@ -28,6 +34,24 @@
|
|
|
28
34
|
export const DEFAULT_BOOTSTRAP_SAMPLES = 1000;
|
|
29
35
|
/** Default significance level; 0.05 → 95% CI. */
|
|
30
36
|
export const DEFAULT_BOOTSTRAP_ALPHA = 0.05;
|
|
37
|
+
/**
|
|
38
|
+
* Fixed default bootstrap seed → CIs are **deterministic by default** (omk default-strict:
|
|
39
|
+
* reproducibility affects verdict validity, so it is on by default, not opt-in). The specific
|
|
40
|
+
* value is arbitrary — only that it is **fixed** matters; it is an implementation detail, not a
|
|
41
|
+
* user-facing constant, so unlike DEFAULT_BOOTSTRAP_SAMPLES / α it is neither cited in docs nor
|
|
42
|
+
* guarded by `doc-constants-drift.test.ts`. Callers wanting a different draw pass an explicit `seed`.
|
|
43
|
+
*/
|
|
44
|
+
export const DEFAULT_BOOTSTRAP_SEED = 20260616;
|
|
45
|
+
/**
|
|
46
|
+
* Confidence-level label for display: `(1 − α)·100%`. Multiple-comparison (Bonferroni) correction
|
|
47
|
+
* shrinks a pairwise α to α/K and widens the CI accordingly — the label must track α so a corrected
|
|
48
|
+
* (wider) interval is never mislabeled "95%". No alpha (single comparison / classic A-B) ⇒ nominal 95%.
|
|
49
|
+
* Shared single source for the HTML renderer and the `omk eval` CLI verdict so both read one scale.
|
|
50
|
+
*/
|
|
51
|
+
export function ciLevelLabel(alpha) {
|
|
52
|
+
const pct = (1 - (alpha ?? DEFAULT_BOOTSTRAP_ALPHA)) * 100;
|
|
53
|
+
return `${Number.isInteger(pct) ? pct : Number(pct.toFixed(1))}%`;
|
|
54
|
+
}
|
|
31
55
|
/** Mulberry32 PRNG — seedable, deterministic for tests. */
|
|
32
56
|
function mulberry32(seed) {
|
|
33
57
|
let s = seed >>> 0;
|
|
@@ -40,9 +64,9 @@ function mulberry32(seed) {
|
|
|
40
64
|
};
|
|
41
65
|
}
|
|
42
66
|
function makeRng(seed) {
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
return mulberry32(seed);
|
|
67
|
+
// 默认确定性:无显式 seed 时退 DEFAULT_BOOTSTRAP_SEED(而非 Math.random)——否则同一 eval 两跑会得到
|
|
68
|
+
// 不同 CI,临界点 significant 翻转 → verdict 不可复现。见模块头 Reproducibility。
|
|
69
|
+
return mulberry32(seed ?? DEFAULT_BOOTSTRAP_SEED);
|
|
46
70
|
}
|
|
47
71
|
/** Sample n indices with replacement from [0, length) using the given PRNG. */
|
|
48
72
|
function resampleIndices(length, n, rng) {
|
|
@@ -150,6 +174,59 @@ export function bootstrapDiffCI(scoresA, scoresB, alpha = DEFAULT_BOOTSTRAP_ALPH
|
|
|
150
174
|
significant: !(low <= 0 && 0 <= high),
|
|
151
175
|
};
|
|
152
176
|
}
|
|
177
|
+
/**
|
|
178
|
+
* Bootstrap CI for the *difference* of two means on **paired** observations — when A and B
|
|
179
|
+
* are two measurements of the **same unit** (e.g. control vs treatment on the same sample,
|
|
180
|
+
* or original vs alternate judge prompt on the same response). Resamples the **pair indices
|
|
181
|
+
* jointly** and averages each pair's `b - a`, so the within-pair correlation is preserved.
|
|
182
|
+
*
|
|
183
|
+
* Why paired (vs `bootstrapDiffCI`'s independent resampling): when A and B move together
|
|
184
|
+
* across units (the usual case — the same sample scored by two variants is positively
|
|
185
|
+
* correlated), much of each group's variance is shared and cancels in the per-pair diff.
|
|
186
|
+
* The independent (unpaired) bootstrap ignores that, over-states the diff's variance, and
|
|
187
|
+
* widens the CI — *conservative*, costing real power. Use paired whenever the design is
|
|
188
|
+
* paired; use `bootstrapDiffCI` only for genuinely independent groups (or where a deliberate
|
|
189
|
+
* conservative bias is wanted). The point estimate is identical (mean of per-pair diffs =
|
|
190
|
+
* difference of paired means); only the CI tightens.
|
|
191
|
+
*
|
|
192
|
+
* `significant` is derived from the **rounded** `low`/`high` (the persisted bounds), so the
|
|
193
|
+
* flag never contradicts what is stored / displayed: a CI that rounds to include 0 reads as
|
|
194
|
+
* not-significant. (Computing it on the unrounded bounds would let the JSON say `low: 0,
|
|
195
|
+
* significant: true` — a self-contradictory `CI=[0, …]` that downstream `computeVerdict` and
|
|
196
|
+
* external consumers cannot reconcile.) Matches `bootstrapDiffCI`.
|
|
197
|
+
*
|
|
198
|
+
* @param pairs Aligned observations; `a` = control/baseline, `b` = treatment. diff = b - a.
|
|
199
|
+
* @param alpha Significance level. Default 0.05.
|
|
200
|
+
* @param samples Bootstrap resamples. Default 1000.
|
|
201
|
+
* @param seed Optional seed (deterministic by default — see module header).
|
|
202
|
+
*/
|
|
203
|
+
export function bootstrapPairedDiffCI(pairs, alpha = DEFAULT_BOOTSTRAP_ALPHA, samples = DEFAULT_BOOTSTRAP_SAMPLES, seed) {
|
|
204
|
+
if (pairs.length === 0) {
|
|
205
|
+
return { low: 0, high: 0, estimate: 0, samples: 0, significant: false };
|
|
206
|
+
}
|
|
207
|
+
const n = pairs.length;
|
|
208
|
+
const diffs = pairs.map((p) => p.b - p.a);
|
|
209
|
+
const rng = makeRng(seed);
|
|
210
|
+
const resampleDiffMeans = new Array(samples);
|
|
211
|
+
for (let s = 0; s < samples; s++) {
|
|
212
|
+
const idx = resampleIndices(n, n, rng);
|
|
213
|
+
let sum = 0;
|
|
214
|
+
for (const i of idx)
|
|
215
|
+
sum += diffs[i];
|
|
216
|
+
resampleDiffMeans[s] = sum / n;
|
|
217
|
+
}
|
|
218
|
+
resampleDiffMeans.sort((a, b) => a - b);
|
|
219
|
+
const low = round4(sortedQuantile(resampleDiffMeans, alpha / 2));
|
|
220
|
+
const high = round4(sortedQuantile(resampleDiffMeans, 1 - alpha / 2));
|
|
221
|
+
return {
|
|
222
|
+
low,
|
|
223
|
+
high,
|
|
224
|
+
estimate: round4(mean(diffs)),
|
|
225
|
+
samples,
|
|
226
|
+
// significant 与持久化的(舍入)边界一致 —— 见函数头:绝不出现「low:0 但 significant:true」自相矛盾。
|
|
227
|
+
significant: !(low <= 0 && 0 <= high),
|
|
228
|
+
};
|
|
229
|
+
}
|
|
153
230
|
/**
|
|
154
231
|
* Generic bootstrap CI for an arbitrary sample-level metric. Used by
|
|
155
232
|
* saturation analysis to get CI on metrics like stddev or
|
|
@@ -8,7 +8,7 @@ import { indexReportWrite } from './artifact-index.js';
|
|
|
8
8
|
import { buildVariantSummary } from './schema.js';
|
|
9
9
|
import { buildVariantConfig, resolveExecutionStrategy } from './execution-strategy.js';
|
|
10
10
|
import { getJudgePromptHash } from '../grading/judge.js';
|
|
11
|
-
import { bootstrapMeanCI,
|
|
11
|
+
import { bootstrapMeanCI, bootstrapPairedDiffCI, DEFAULT_BOOTSTRAP_ALPHA, DEFAULT_BOOTSTRAP_SAMPLES, } from './bootstrap.js';
|
|
12
12
|
import { getExecutorRuntimeFingerprint } from '../executors/runtime-fingerprint.js';
|
|
13
13
|
const __dirname = dirname(fileURLToPath(import.meta.url));
|
|
14
14
|
function findPackageJson(startDir) {
|
|
@@ -121,6 +121,9 @@ export function aggregateReport({ runId, variants, model, judgeModel, noJudge, e
|
|
|
121
121
|
const bootstrapSamples = request?.bootstrapSamples ?? DEFAULT_BOOTSTRAP_SAMPLES;
|
|
122
122
|
let pairComparisons;
|
|
123
123
|
if (bootstrapEnabled) {
|
|
124
|
+
// 不变量(见 grading/layered-scores.ts):composite 在至少一层可测时恒 ≥ 1,`compositeScore === 0`
|
|
125
|
+
// 当且仅当该样本**无任何可测层**(真·缺测,如纯评委样本且评委失败)。故 `> 0` 过滤精确剔除非测量、
|
|
126
|
+
// 绝不丢"低分内容"(评委失败已在上游当缺测,不会以 0 进 composite)。下同(control / treatment)。
|
|
124
127
|
for (const variant of variants) {
|
|
125
128
|
const entries = Object.values(results).map((r) => r[variant]).filter(Boolean);
|
|
126
129
|
const compositeScores = entries
|
|
@@ -135,23 +138,42 @@ export function aggregateReport({ runId, variants, model, judgeModel, noJudge, e
|
|
|
135
138
|
if (variants.length >= 2) {
|
|
136
139
|
pairComparisons = [];
|
|
137
140
|
const controlName = variants[0];
|
|
138
|
-
const
|
|
139
|
-
|
|
140
|
-
|
|
141
|
-
|
|
141
|
+
const sampleRecords = Object.values(results);
|
|
142
|
+
// **配对** diff CI:A/B 是同一批 sample 分别过 control / treatment(配对设计),按 sample 对齐 ——
|
|
143
|
+
// 同一 sample 上 control 与 treatment 都可测(composite > 0,见上不变量)才入对。同一 sample 两 variant
|
|
144
|
+
// 的分数正相关,配对 bootstrap 据此抵消共有方差、收紧 diff CI、更有功效;旧的独立(非配对)重采样高估
|
|
145
|
+
// 方差、CI 偏宽、保守失功效(见 bootstrapPairedDiffCI)。点估计不变,只收紧 CI。
|
|
146
|
+
// 先收齐每个 treatment 的配对数据;只有 ≥2 对(否则无 CI 可算)才算一个真正被检验的比较。
|
|
147
|
+
const eligible = [];
|
|
142
148
|
for (let i = 1; i < variants.length; i++) {
|
|
143
149
|
const treatmentName = variants[i];
|
|
144
|
-
const
|
|
145
|
-
const
|
|
146
|
-
|
|
147
|
-
|
|
148
|
-
|
|
149
|
-
|
|
150
|
-
|
|
151
|
-
|
|
152
|
-
diffBootstrapCI: bootstrapDiffCI(controlScores, treatmentScores, DEFAULT_BOOTSTRAP_ALPHA, bootstrapSamples),
|
|
153
|
-
});
|
|
150
|
+
const pairs = [];
|
|
151
|
+
for (const r of sampleRecords) {
|
|
152
|
+
const c = r[controlName];
|
|
153
|
+
const t = r[treatmentName];
|
|
154
|
+
const a = c && typeof c.compositeScore === 'number' && c.compositeScore > 0 ? c.compositeScore : undefined;
|
|
155
|
+
const b = t && typeof t.compositeScore === 'number' && t.compositeScore > 0 ? t.compositeScore : undefined;
|
|
156
|
+
if (a !== undefined && b !== undefined)
|
|
157
|
+
pairs.push({ a, b });
|
|
154
158
|
}
|
|
159
|
+
if (pairs.length >= 2)
|
|
160
|
+
eligible.push({ treatment: treatmentName, pairs });
|
|
161
|
+
}
|
|
162
|
+
// 多重比较(Bonferroni)校正:同时检验 K 个 treatment-vs-control 假设时,family-wise 假阳性随 K 膨胀
|
|
163
|
+
// (computeVerdict 的 worst-case roll-up 取最差 —— 任一对假阳即拉高总判定)。每对 CI 用 α/K(K = 实际
|
|
164
|
+
// 产出 CI 的比较数)把 family-wise error 压回名义 α。K=1(单 treatment / 经典 A-B)即 α 不变、与历史单对
|
|
165
|
+
// 口径逐字节一致,此时不写 alpha 字段(渲染按名义 95% CI,既有报告 / 快照不动)。CI 与 significant 同在 α/K
|
|
166
|
+
// 下算,二者自洽(绝不出现 CI 含 0 却 significant 的矛盾)。注:K 很大时 α/K 落到极端分位,1000 重采样的
|
|
167
|
+
// 尾部分位偏粗,大 K 慎读 —— 不在本 PR 提采样数。
|
|
168
|
+
const familySize = eligible.length;
|
|
169
|
+
const perComparisonAlpha = familySize >= 1 ? DEFAULT_BOOTSTRAP_ALPHA / familySize : DEFAULT_BOOTSTRAP_ALPHA;
|
|
170
|
+
for (const { treatment, pairs } of eligible) {
|
|
171
|
+
pairComparisons.push({
|
|
172
|
+
control: controlName,
|
|
173
|
+
treatment,
|
|
174
|
+
diffBootstrapCI: bootstrapPairedDiffCI(pairs, perComparisonAlpha, bootstrapSamples),
|
|
175
|
+
...(familySize >= 2 ? { alpha: perComparisonAlpha } : {}),
|
|
176
|
+
});
|
|
155
177
|
}
|
|
156
178
|
}
|
|
157
179
|
}
|
|
@@ -66,6 +66,7 @@ export function buildVariantConfig(artifact) {
|
|
|
66
66
|
cwd: artifact.cwd || null,
|
|
67
67
|
locator: artifact.locator,
|
|
68
68
|
ref: artifact.ref,
|
|
69
|
+
...(artifact.resolvedCommit ? { resolvedCommit: artifact.resolvedCommit } : {}),
|
|
69
70
|
// propagate skill-isolation declaration so report.meta.skillIsolation
|
|
70
71
|
// 能在 evaluation-reporting 阶段从 variantConfigs 提取 (avoid re-resolving artifacts).
|
|
71
72
|
...(artifact.allowedSkills !== undefined && { allowedSkills: artifact.allowedSkills }),
|
package/dist/eval-core/schema.js
CHANGED
|
@@ -88,7 +88,8 @@ export function buildVariantResult(execResult, gradeResult, options) {
|
|
|
88
88
|
layeredScores.factScore = assertionFact != null
|
|
89
89
|
? Number(((assertionFact + hardScore) / 2).toFixed(2))
|
|
90
90
|
: hardScore;
|
|
91
|
-
// Recompute composite from updated layers (
|
|
91
|
+
// Recompute composite from updated layers. 仅过滤 null/undefined(缺测层);评委失败时 judgeScore
|
|
92
|
+
// 本就是 undefined(见 grading 的 score=0 修复:评委失败=缺测、不以 0 进层),不存在「0 分」要保留。
|
|
92
93
|
const scores = [layeredScores.factScore, layeredScores.behaviorScore, layeredScores.judgeScore].filter((s) => s != null);
|
|
93
94
|
compositeScore = scores.length > 0 ? Number((scores.reduce((a, b) => a + b, 0) / scores.length).toFixed(2)) : compositeScore;
|
|
94
95
|
}
|
|
@@ -204,8 +205,9 @@ export function buildVariantSummary(entries) {
|
|
|
204
205
|
};
|
|
205
206
|
})(),
|
|
206
207
|
...(() => {
|
|
207
|
-
//
|
|
208
|
-
//
|
|
208
|
+
// 各层分量为 null/undefined = 该层缺测(无对应断言 / 未配 judge / 评委失败)。评委失败已在 grading 层
|
|
209
|
+
// 当缺测、judgeScore 留 undefined(见 score=0 修复),不存在「0 分内容」要保留;评委有效分恒 1-5。
|
|
210
|
+
// filter != null 即精确剔除缺测层。
|
|
209
211
|
const factScores = ok.map((e) => e.layeredScores?.factScore).filter((s) => s != null);
|
|
210
212
|
const behaviorScores = ok.map((e) => e.layeredScores?.behaviorScore).filter((s) => s != null);
|
|
211
213
|
const judgeScores = ok.map((e) => e.layeredScores?.judgeScore).filter((s) => s != null);
|
|
@@ -45,6 +45,16 @@ export declare const UNDERPOWERED_MIN_SAMPLES = 20;
|
|
|
45
45
|
*/
|
|
46
46
|
export declare const ENSEMBLE_STRONG_PEARSON = 0.7;
|
|
47
47
|
export declare const ENSEMBLE_DISSENT_PEARSON = 0.4;
|
|
48
|
+
/**
|
|
49
|
+
* Run-to-run instability threshold on the median coefficient of variation (CV = stddev/mean).
|
|
50
|
+
* Once stability is **actually measured** (`--repeat ≥ 2`) and the median CV exceeds this line,
|
|
51
|
+
* a would-be PROGRESS is downgraded to CAUTIOUS — a statistically significant but run-to-run
|
|
52
|
+
* irreproducible "gain" is not shippable. It is the upper bound of the 5/15% stability bands in
|
|
53
|
+
* `docs/specs/terminology-spec.md` §5; doc ↔ code parity is guarded by
|
|
54
|
+
* `test/scripts/doc-constants-drift.test.ts`. Single-run reports (stability not measured) are
|
|
55
|
+
* never gated by it — see `computeVerdict`.
|
|
56
|
+
*/
|
|
57
|
+
export declare const STABILITY_UNSTABLE_CV = 0.15;
|
|
48
58
|
export type VerdictLevel = 'PROGRESS' | 'CAUTIOUS' | 'REGRESS' | 'NOISE' | 'UNDERPOWERED' | 'SOLO';
|
|
49
59
|
export interface VerdictResult {
|
|
50
60
|
level: VerdictLevel;
|
|
@@ -85,6 +95,26 @@ export interface VerdictOptions {
|
|
|
85
95
|
* Compute a verdict for a finished report. Pure function — no I/O.
|
|
86
96
|
*/
|
|
87
97
|
export declare function computeVerdict(report: Report, options?: VerdictOptions): VerdictResult;
|
|
98
|
+
/**
|
|
99
|
+
* Stability rationale. 三种状态:
|
|
100
|
+
* - --repeat ≥ 2 + 有 variance 数据: 报告 CV (variation coefficient) 主指标
|
|
101
|
+
* - --repeat ≥ 2 但 variance 缺失: 异常,标 "—" 提示数据丢失
|
|
102
|
+
* - --repeat < 2: 显式说"未测量,需 --repeat ≥ 2",而不是默默不提
|
|
103
|
+
*
|
|
104
|
+
* 单轮场景关键:不是"稳定 = 100%"(常见误读),而是"测不到稳定性"。
|
|
105
|
+
* Verdict 必须诚实交代这个盲区,不能让用户以为没说就是 OK。
|
|
106
|
+
*/
|
|
107
|
+
/**
|
|
108
|
+
* 跨轮稳定性的中位 CV(variation coefficient = stddev/mean)。仅在**已测**(runs≥2)且 variance 数据齐时
|
|
109
|
+
* 返回 { runs, cv },否则 null(单轮未测 / variance 缺失 / CV 全算不出 → 不参与门控)。formatStability 的
|
|
110
|
+
* 文字、computeVerdict 的稳定性门控、renderer 的 hero CV chip 共用这一处计算,口径一致、绝不漂移。
|
|
111
|
+
* **真·中位**:偶数个 variant 取中间两项的平均(不是上中位)—— 最常见的 A/B 报告恰好是两个 variant,取上中位
|
|
112
|
+
* 会退化成"较大的那个 CV",把门控口径从「中位」悄悄变成「max」、直接改变 ship/no-ship(复审 P2)。
|
|
113
|
+
*/
|
|
114
|
+
export declare function medianStabilityCV(report: Report): {
|
|
115
|
+
runs: number;
|
|
116
|
+
cv: number;
|
|
117
|
+
} | null;
|
|
88
118
|
/**
|
|
89
119
|
* Plain-text formatter for the `omk eval` verdict. Stays under 6 lines per the
|
|
90
120
|
* spec — one verdict + four rationale bullets + one ship recommendation.
|
|
@@ -28,6 +28,7 @@
|
|
|
28
28
|
* empirical) is documented inline so users can audit and override.
|
|
29
29
|
*/
|
|
30
30
|
import { evaluateLayerGates } from './layer-gates.js';
|
|
31
|
+
import { ciLevelLabel } from './bootstrap.js';
|
|
31
32
|
/**
|
|
32
33
|
* Below this sample count a non-significant diff is read as UNDERPOWERED
|
|
33
34
|
* (only large effects are detectable) rather than NOISE. Matches the
|
|
@@ -45,6 +46,16 @@ export const UNDERPOWERED_MIN_SAMPLES = 20;
|
|
|
45
46
|
*/
|
|
46
47
|
export const ENSEMBLE_STRONG_PEARSON = 0.7;
|
|
47
48
|
export const ENSEMBLE_DISSENT_PEARSON = 0.4;
|
|
49
|
+
/**
|
|
50
|
+
* Run-to-run instability threshold on the median coefficient of variation (CV = stddev/mean).
|
|
51
|
+
* Once stability is **actually measured** (`--repeat ≥ 2`) and the median CV exceeds this line,
|
|
52
|
+
* a would-be PROGRESS is downgraded to CAUTIOUS — a statistically significant but run-to-run
|
|
53
|
+
* irreproducible "gain" is not shippable. It is the upper bound of the 5/15% stability bands in
|
|
54
|
+
* `docs/specs/terminology-spec.md` §5; doc ↔ code parity is guarded by
|
|
55
|
+
* `test/scripts/doc-constants-drift.test.ts`. Single-run reports (stability not measured) are
|
|
56
|
+
* never gated by it — see `computeVerdict`.
|
|
57
|
+
*/
|
|
58
|
+
export const STABILITY_UNSTABLE_CV = 0.15;
|
|
48
59
|
/**
|
|
49
60
|
* Compute a verdict for a finished report. Pure function — no I/O.
|
|
50
61
|
*/
|
|
@@ -89,6 +100,16 @@ export function computeVerdict(report, options = {}) {
|
|
|
89
100
|
}
|
|
90
101
|
// Single representative pair for the top-level rationale (the worst one).
|
|
91
102
|
const representative = perPair.find((p) => p.level === topLevel) ?? perPair[0];
|
|
103
|
+
// 稳定性门控(报告级,非 per-pair):仅当**已测**(runs≥2)且 run-to-run 不稳(median CV > STABILITY_UNSTABLE_CV)
|
|
104
|
+
// 时,把 PROGRESS 降为 CAUTIOUS —— 显著但跨轮不可复现的"进展"不可 ship。单轮(未测稳定性)不门控:rationale
|
|
105
|
+
// 已诚实标"未测量"(terminology-spec §5「诚实交代测不到的东西」),默认单轮全降级会过激。只压 PROGRESS:已是 CAUTIOUS/REGRESS 等
|
|
106
|
+
// 不再加码,顺序与 worst-case roll-up 一致。
|
|
107
|
+
const stab = medianStabilityCV(report);
|
|
108
|
+
const stabilityGated = topLevel === 'PROGRESS' && stab !== null && stab.cv > STABILITY_UNSTABLE_CV;
|
|
109
|
+
const level = stabilityGated ? 'CAUTIOUS' : topLevel;
|
|
110
|
+
const stabilityNote = stabilityGated && stab
|
|
111
|
+
? ` · 显著但 run-to-run 不稳(CV=${(stab.cv * 100).toFixed(1)}% > ${(STABILITY_UNSTABLE_CV * 100).toFixed(0)}%)`
|
|
112
|
+
: '';
|
|
92
113
|
const significance = representative
|
|
93
114
|
? formatSignificance(representative)
|
|
94
115
|
: 'no pairwise comparison available — was --bootstrap used?';
|
|
@@ -96,12 +117,12 @@ export function computeVerdict(report, options = {}) {
|
|
|
96
117
|
const sampleSize = formatSampleSize(report);
|
|
97
118
|
const stability = formatStability(report);
|
|
98
119
|
const judgeAgreement = formatJudgeAgreement(report);
|
|
99
|
-
const shipRecommendation = recommendation(
|
|
120
|
+
const shipRecommendation = recommendation(level, perPair);
|
|
100
121
|
return {
|
|
101
|
-
level
|
|
122
|
+
level,
|
|
102
123
|
headline: representative
|
|
103
|
-
? `${
|
|
104
|
-
: `${
|
|
124
|
+
? `${level} · ${representative.treatment} vs ${representative.control}: ${representative.headline}${stabilityNote}`
|
|
125
|
+
: `${level} · ${variants.length} variants`,
|
|
105
126
|
perPair,
|
|
106
127
|
rationale: {
|
|
107
128
|
significance,
|
|
@@ -130,7 +151,10 @@ function verdictForPair(pair, summary, sampleCount, report, gateThreshold, trivi
|
|
|
130
151
|
// Layer-gate check: did any layer fall below threshold for either variant?
|
|
131
152
|
const cGate = evaluateLayerGates({ [control]: summary[control] }, gateThreshold);
|
|
132
153
|
const tGate = evaluateLayerGates({ [treatment]: summary[treatment] }, gateThreshold);
|
|
133
|
-
// No bootstrap CI available → fall back to point-estimate diff comparison.
|
|
154
|
+
// No bootstrap CI available → fall back to point-estimate diff comparison. 这是 `--no-bootstrap` 的**降级
|
|
155
|
+
// 模式**:bootstrap 默认开,正常路径永远有 CI,这里只在用户显式关掉时触达。降级路径刻意不对称——正 Δ 最多给
|
|
156
|
+
// CAUTIOUS(没 CI 不敢判 PROGRESS),负 Δ 直接 REGRESS(不做显著性检验)。这对"检测变差"是保守安全方向(宁可
|
|
157
|
+
// 误报回归也别漏掉),代价是把噪声级的负 Δ 也叫 REGRESS;要严谨结论就别关 bootstrap。
|
|
134
158
|
if (!diff) {
|
|
135
159
|
const cMean = avgComposite(summary[control]);
|
|
136
160
|
const tMean = avgComposite(summary[treatment]);
|
|
@@ -146,7 +170,10 @@ function verdictForPair(pair, summary, sampleCount, report, gateThreshold, trivi
|
|
|
146
170
|
headline,
|
|
147
171
|
};
|
|
148
172
|
}
|
|
149
|
-
|
|
173
|
+
// 多重比较把本对的 α 收到 α/K → CI 变宽,标签随 α 走(与 HTML pairwise 表同口径);K=1(无 alpha)不加标签,
|
|
174
|
+
// headline 与历史逐字节一致。让 omk eval CLI 也诚实显示真实置信水平,而非裸区间。
|
|
175
|
+
const ciLabel = pair.alpha != null ? `${ciLevelLabel(pair.alpha)} ` : '';
|
|
176
|
+
const headlineCore = `Δ=${diff.estimate >= 0 ? '+' : ''}${diff.estimate} ${ciLabel}CI=[${diff.low}, ${diff.high}]`;
|
|
150
177
|
if (!diff.significant) {
|
|
151
178
|
// Diff CI contains 0. Distinguish "underpowered (saturation says: more samples needed)"
|
|
152
179
|
// from "noise (saturation says: we're saturated, the effect just isn't there)".
|
|
@@ -233,7 +260,10 @@ function avgComposite(s) {
|
|
|
233
260
|
return 0;
|
|
234
261
|
if (typeof s.avgCompositeScore === 'number')
|
|
235
262
|
return s.avgCompositeScore;
|
|
236
|
-
//
|
|
263
|
+
// 兜底:summary 无 avgCompositeScore 时,用三层均值近似。**有偏**:真 composite 是 per-sample
|
|
264
|
+
// (present-layers 均值)再跨 sample 平均;这里是「跨 sample 的层均值」再跨层平均,各层在不同 sample 上缺失
|
|
265
|
+
// 不均时两者不等(Jensen / 分母不一致)。仅 no-bootstrap 降级路径 + summary 缺 composite 的老报告才触达
|
|
266
|
+
// (默认开 bootstrap、新报告必带 avgCompositeScore,故极罕见),不值得回填 per-sample 重算 —— 标注保留近似。
|
|
237
267
|
const layers = [s.avgFactScore, s.avgBehaviorScore, s.avgJudgeScore].filter((x) => typeof x === 'number');
|
|
238
268
|
if (layers.length === 0)
|
|
239
269
|
return 0;
|
|
@@ -289,16 +319,20 @@ function formatSampleSize(report) {
|
|
|
289
319
|
* 单轮场景关键:不是"稳定 = 100%"(常见误读),而是"测不到稳定性"。
|
|
290
320
|
* Verdict 必须诚实交代这个盲区,不能让用户以为没说就是 OK。
|
|
291
321
|
*/
|
|
292
|
-
|
|
322
|
+
/**
|
|
323
|
+
* 跨轮稳定性的中位 CV(variation coefficient = stddev/mean)。仅在**已测**(runs≥2)且 variance 数据齐时
|
|
324
|
+
* 返回 { runs, cv },否则 null(单轮未测 / variance 缺失 / CV 全算不出 → 不参与门控)。formatStability 的
|
|
325
|
+
* 文字、computeVerdict 的稳定性门控、renderer 的 hero CV chip 共用这一处计算,口径一致、绝不漂移。
|
|
326
|
+
* **真·中位**:偶数个 variant 取中间两项的平均(不是上中位)—— 最常见的 A/B 报告恰好是两个 variant,取上中位
|
|
327
|
+
* 会退化成"较大的那个 CV",把门控口径从「中位」悄悄变成「max」、直接改变 ship/no-ship(复审 P2)。
|
|
328
|
+
*/
|
|
329
|
+
export function medianStabilityCV(report) {
|
|
293
330
|
const runs = report.variance?.runs ?? report.meta?.request?.repeat ?? 1;
|
|
294
|
-
if (runs < 2)
|
|
295
|
-
return
|
|
296
|
-
}
|
|
331
|
+
if (runs < 2)
|
|
332
|
+
return null;
|
|
297
333
|
const variance = report.variance?.perVariant;
|
|
298
|
-
if (!variance || Object.keys(variance).length === 0)
|
|
299
|
-
return
|
|
300
|
-
}
|
|
301
|
-
// CV = stddev / mean,取所有 variant 的中位数(单 variant 易有 NaN/0,聚合更稳)
|
|
334
|
+
if (!variance || Object.keys(variance).length === 0)
|
|
335
|
+
return null;
|
|
302
336
|
const cvs = [];
|
|
303
337
|
for (const v of Object.values(variance)) {
|
|
304
338
|
if (typeof v.stddev === 'number' && typeof v.mean === 'number' && v.mean > 0) {
|
|
@@ -306,12 +340,28 @@ function formatStability(report) {
|
|
|
306
340
|
}
|
|
307
341
|
}
|
|
308
342
|
if (cvs.length === 0)
|
|
309
|
-
return
|
|
343
|
+
return null;
|
|
310
344
|
cvs.sort((a, b) => a - b);
|
|
311
|
-
const
|
|
312
|
-
const
|
|
313
|
-
|
|
314
|
-
|
|
345
|
+
const mid = Math.floor(cvs.length / 2);
|
|
346
|
+
const median = cvs.length % 2 === 0 ? (cvs[mid - 1] + cvs[mid]) / 2 : cvs[mid];
|
|
347
|
+
return { runs, cv: median };
|
|
348
|
+
}
|
|
349
|
+
function formatStability(report) {
|
|
350
|
+
const runs = report.variance?.runs ?? report.meta?.request?.repeat ?? 1;
|
|
351
|
+
if (runs < 2) {
|
|
352
|
+
return '稳定性未测量(单轮评测,需 --repeat ≥ 2 才能测 CV)';
|
|
353
|
+
}
|
|
354
|
+
const variance = report.variance?.perVariant;
|
|
355
|
+
if (!variance || Object.keys(variance).length === 0) {
|
|
356
|
+
return `runs=${runs} 但 variance 数据缺失`;
|
|
357
|
+
}
|
|
358
|
+
const stab = medianStabilityCV(report);
|
|
359
|
+
if (!stab)
|
|
360
|
+
return `runs=${runs}, CV 计算失败(stddev/mean 数据缺失)`;
|
|
361
|
+
const cvPct = (stab.cv * 100).toFixed(1);
|
|
362
|
+
// 阈值参考 terminology-spec §5:<5% 稳 / 5-15% 中 / >15% 不稳。不稳上界 = STABILITY_UNSTABLE_CV,
|
|
363
|
+
// 与门控同一根线:label 判"不稳" ⟺ 门控触发(cv > STABILITY_UNSTABLE_CV)。
|
|
364
|
+
const verdict = stab.cv < 0.05 ? '稳定' : stab.cv <= STABILITY_UNSTABLE_CV ? '中等' : '不稳';
|
|
315
365
|
return `CV=${cvPct}% (${verdict}, runs=${runs}; 阈值 <5%=稳/5-15%=中/>15%=不稳)`;
|
|
316
366
|
}
|
|
317
367
|
function formatJudgeAgreement(report) {
|
|
@@ -331,7 +381,7 @@ function recommendation(level, _perPair) {
|
|
|
331
381
|
case 'PROGRESS':
|
|
332
382
|
return 'SHIP — treatment is significantly better and passes all layer gates.';
|
|
333
383
|
case 'CAUTIOUS':
|
|
334
|
-
return 'INVESTIGATE — the gain is real but at least one warning fired (broken gate, trivially small, or
|
|
384
|
+
return 'INVESTIGATE — the gain is real but at least one warning fired (broken gate, trivially small, partial recovery, judge dissent, or run-to-run unstable). Do not ship blind.';
|
|
335
385
|
case 'REGRESS':
|
|
336
386
|
return 'DO NOT SHIP — treatment regresses. Check the worst layer and re-run with the fix.';
|
|
337
387
|
case 'NOISE':
|
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
import { resolve } from 'node:path';
|
|
2
2
|
import _Ajv from 'ajv';
|
|
3
|
+
import { ASSERTION_LAYER } from './layered-scores.js';
|
|
3
4
|
const Ajv = _Ajv.default ?? _Ajv;
|
|
4
5
|
const ajv = new Ajv();
|
|
5
6
|
const CUSTOM_ASSERTION_TIMEOUT_MS = 30_000;
|
|
@@ -266,6 +267,30 @@ function evalAssertion(output, assertion, ctx) {
|
|
|
266
267
|
return false;
|
|
267
268
|
}
|
|
268
269
|
}
|
|
270
|
+
/**
|
|
271
|
+
* assert-set 是布尔组合器,不是叶子断言,没有静态的「层」。它在 runAssertions 里只产出一条聚合明细
|
|
272
|
+
* (`type: 'assert-set'`、aggregate pass/fail),computeLayeredScores 无法据此判 fact / behavior。
|
|
273
|
+
* 这里在 grading 期(能看到 children 树时)按**叶子子断言**的层解析:递归收集所有叶子(穿透嵌套 assert-set),
|
|
274
|
+
* - 同层(全 fact 或全 behavior)→ 返回该层:聚合 pass/fail 归这一层,与 any/all 模式无关(同层信号无歧义)。
|
|
275
|
+
* - 混层 / 空 / 含未分类叶子 → 返回 undefined:不归层(混层组合器塞进单层会引入口径偏差,诚实做法是不计入分层)。
|
|
276
|
+
* 解析结果落到 detail.layer,computeLayeredScores 优先读它。
|
|
277
|
+
*/
|
|
278
|
+
function resolveAssertSetLayer(assertion) {
|
|
279
|
+
const layers = new Set();
|
|
280
|
+
const collect = (a) => {
|
|
281
|
+
if (a.type === 'assert-set') {
|
|
282
|
+
for (const c of a.children ?? [])
|
|
283
|
+
collect(c);
|
|
284
|
+
return;
|
|
285
|
+
}
|
|
286
|
+
layers.add(ASSERTION_LAYER[a.type] ?? 'unknown');
|
|
287
|
+
};
|
|
288
|
+
collect(assertion);
|
|
289
|
+
if (layers.size !== 1)
|
|
290
|
+
return undefined; // 0 / 混层 → 不归层
|
|
291
|
+
const [only] = [...layers];
|
|
292
|
+
return only === 'unknown' ? undefined : only;
|
|
293
|
+
}
|
|
269
294
|
export function runAssertions(output, assertions, context = {}) {
|
|
270
295
|
const outputLower = output.toLowerCase();
|
|
271
296
|
const toolCalls = context.toolCalls || [];
|
|
@@ -276,11 +301,14 @@ export function runAssertions(output, assertions, context = {}) {
|
|
|
276
301
|
const weight = assertion.weight ?? 1;
|
|
277
302
|
const raw = evalAssertion(output, assertion, ctx);
|
|
278
303
|
const passed = assertion.not ? !raw : raw;
|
|
304
|
+
// assert-set 组合器:在此(能看到 children)解析其层,供 computeLayeredScores 用;叶子断言按静态映射归层、不带 layer。
|
|
305
|
+
const layer = assertion.type === 'assert-set' ? resolveAssertSetLayer(assertion) : undefined;
|
|
279
306
|
details.push({
|
|
280
307
|
type: assertion.type,
|
|
281
308
|
value: assertion.value ?? assertion.pattern ?? assertion.values?.join(', ') ?? '',
|
|
282
309
|
weight,
|
|
283
310
|
passed,
|
|
311
|
+
...(layer ? { layer } : {}),
|
|
284
312
|
});
|
|
285
313
|
}
|
|
286
314
|
const totalWeight = details.reduce((s, d) => s + d.weight, 0);
|
|
@@ -24,7 +24,7 @@
|
|
|
24
24
|
* large/expensive evaluations can opt in deliberately.
|
|
25
25
|
*/
|
|
26
26
|
import { llmJudge } from './judge.js';
|
|
27
|
-
import {
|
|
27
|
+
import { bootstrapPairedDiffCI } from '../eval-core/bootstrap.js';
|
|
28
28
|
/**
|
|
29
29
|
* Map a bootstrap diff CI to a verdict bucket. The ranges are deliberately
|
|
30
30
|
* conservative: we only label "strong" when the CI fully sits >= |0.5| away
|
|
@@ -119,7 +119,11 @@ export async function validateLengthDebias(input) {
|
|
|
119
119
|
}
|
|
120
120
|
const meanOriginal = avg(pairs.map((p) => p.originalScore));
|
|
121
121
|
const meanAlternate = avg(pairs.map((p) => p.alternateScore));
|
|
122
|
-
|
|
122
|
+
// **配对** diff CI:每个 sample 同时有 original 与 alternate prompt 两个分数(同一回答、两个 judge prompt
|
|
123
|
+
// = 配对设计)。原先拆成两数组喂独立重采样,丢弃了配对、高估方差、CI 偏宽 —— 对一个**检测**长度偏置敏感性
|
|
124
|
+
// 的工具,保守方向恰好是错的(更难检出真实偏置)。改配对:重采样 sample 下标、按 (alternate − original) 算,
|
|
125
|
+
// 保留 within-sample 相关、收紧 CI,提升对偏置的检出力。diff = b − a = alternate − original(同原约定)。
|
|
126
|
+
const diffCI = bootstrapPairedDiffCI(pairs.map((p) => ({ a: p.originalScore, b: p.alternateScore })), 0.05, bootstrapSamples, seed);
|
|
123
127
|
const verdict = classifyVerdict(diffCI);
|
|
124
128
|
return {
|
|
125
129
|
variant,
|
package/dist/grading/index.js
CHANGED
|
@@ -109,7 +109,12 @@ export async function grade({ output, sample, judgeModels, judgeExecutors, allow
|
|
|
109
109
|
const judge = useEnsemble
|
|
110
110
|
? await llmJudgeEnsemble(rubricOptions, judgeModels, executorByName, judgeRepeat)
|
|
111
111
|
: await llmJudgeRepeat(rubricOptions, judgeRepeat);
|
|
112
|
-
|
|
112
|
+
// judge.score <= 0 表示该样本所有判定尝试都失败(非 JSON / parse / executor 错——见 judge.ts 失败哨兵
|
|
113
|
+
// 268/275/298,prompt 只发 1-5),是**缺测**而非「0 分内容」。不写 llmScore(留 undefined)→ 让
|
|
114
|
+
// computeLayeredScores 当缺层排除,而不是把基础设施失败当内容分污染 composite(与多维度路径 filter(s>0)
|
|
115
|
+
// 及多轮路径 validSamples 已排除失败同口径)。llmReason / scoreSamples 仍记录,供诊断"为何失败"。
|
|
116
|
+
if (judge.score > 0)
|
|
117
|
+
results.llmScore = judge.score;
|
|
113
118
|
results.llmReason = judge.reason;
|
|
114
119
|
if (judge.reasoning)
|
|
115
120
|
results.llmReasoning = judge.reasoning;
|
|
@@ -1,4 +1,19 @@
|
|
|
1
1
|
import type { AssertionDetail, LayeredScores } from '../types/index.js';
|
|
2
|
+
/**
|
|
3
|
+
* 每个 assertion 类型归到事实层 / 行为层。**单一来源**:`computeLayeredScores` 据此把 assertion 明细拆进
|
|
4
|
+
* fact / behavior 两层算分。
|
|
5
|
+
* - **事实层(fact)**:测输出内容对不对 —— 命中 / 匹配 / 结构合法 / 与参考的文本相似度。
|
|
6
|
+
* - **行为层(behavior)**:测做事的方式 —— 长度 / 成本 / 轮次 / 工具调用 / 必经里程碑,不看内容本身。
|
|
7
|
+
*
|
|
8
|
+
* 不变量:**runner 支持的每个 assertion 类型都必须能被归层**,否则该类型的 pass/fail 会被 computeLayeredScores
|
|
9
|
+
* 从 fact 与 behavior 同时漏掉 —— 既不报错也不进 composite,静默丢分(曾漏掉七类:mock_hit / rouge_n_min /
|
|
10
|
+
* bleu_min / levenshtein_max + RAG 三件套 faithfulness / answer_relevancy / context_recall)。叶子断言在此静态
|
|
11
|
+
* 分类;组合器 `assert-set` 没有静态层,由 `assertions.ts` 的 resolveAssertSetLayer 在 grading 期按其叶子 children
|
|
12
|
+
* 解析(同层→归层、混层→不计),结果落 detail.layer。`test/grading/layered-scores-exhaustiveness.test.ts` 扫
|
|
13
|
+
* runner 源(evalAssertion 的 case ∪ `assertion.type ===` 组合器 ∪ ASYNC_ASSERTION_TYPES)守住:新增类型既不在本
|
|
14
|
+
* 映射、又不是已知组合器,即 CI 失败。
|
|
15
|
+
*/
|
|
16
|
+
export declare const ASSERTION_LAYER: Record<string, 'fact' | 'behavior'>;
|
|
2
17
|
interface CompositeInput {
|
|
3
18
|
assertions?: {
|
|
4
19
|
details?: AssertionDetail[];
|