oh-my-knowledge 0.41.0 → 0.42.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +7 -2
- package/README.zh.md +7 -2
- package/dist/analysis/report-diagnostics.d.ts +8 -1
- package/dist/analysis/report-diagnostics.js +82 -1
- package/dist/assets/agent-skills/omk/references/commands.md +1 -0
- package/dist/authoring/evolver.d.ts +3 -14
- package/dist/authoring/evolver.js +1 -52
- package/dist/authoring/generator.d.ts +24 -0
- package/dist/authoring/generator.js +33 -6
- package/dist/cli/commands/eval/index.d.ts +1 -0
- package/dist/cli/commands/eval/index.js +40 -5
- package/dist/cli/commands/init.js +10 -7
- package/dist/cli/lib/i18n-dict/init.d.ts +1 -1
- package/dist/cli/lib/i18n-dict/init.js +14 -11
- package/dist/cli/lib/i18n-dict/run.d.ts +1 -1
- package/dist/cli/lib/i18n-dict/run.js +4 -0
- package/dist/cli/lib/parse-run-config.d.ts +3 -0
- package/dist/eval-core/evaluation-job.d.ts +2 -1
- package/dist/eval-core/evaluation-job.js +2 -1
- package/dist/eval-core/evaluation-reporting.js +7 -3
- package/dist/eval-core/holdout.d.ts +66 -0
- package/dist/eval-core/holdout.js +118 -0
- package/dist/eval-core/verdict.d.ts +44 -1
- package/dist/eval-core/verdict.js +175 -13
- package/dist/eval-workflows/evaluation-pipeline/report-finalize.js +19 -1
- package/dist/eval-workflows/evaluation-pipeline/run-state.d.ts +2 -1
- package/dist/eval-workflows/evaluation-pipeline/run-state.js +2 -1
- package/dist/eval-workflows/evaluation-pipeline.d.ts +3 -1
- package/dist/eval-workflows/evaluation-pipeline.js +2 -1
- package/dist/eval-workflows/run-evaluation.d.ts +5 -2
- package/dist/eval-workflows/run-evaluation.js +8 -5
- package/dist/inputs/eval-config.js +6 -0
- package/dist/renderer/summary.js +36 -3
- package/dist/types/eval.d.ts +7 -0
- package/dist/types/report.d.ts +49 -0
- package/package.json +1 -1
|
@@ -51,6 +51,10 @@ export const runDict = {
|
|
|
51
51
|
zh: '⚠ --judge-repeat "{value}" 无效 (期望 ≥ 1 的整数), 已按 1 次 judge 执行\n',
|
|
52
52
|
en: '⚠ --judge-repeat "{value}" is invalid (expected an integer ≥ 1), falling back to 1 judge call\n',
|
|
53
53
|
},
|
|
54
|
+
'cli.run.invalid_holdout_ratio': {
|
|
55
|
+
zh: '⚠ --holdout-ratio "{value}" 无效 (期望 0 到 1 之间的小数), 已忽略、不做 holdout 切分\n',
|
|
56
|
+
en: '⚠ --holdout-ratio "{value}" is invalid (expected a fraction in (0, 1)), ignored — no holdout split\n',
|
|
57
|
+
},
|
|
54
58
|
'cli.run.no_debias_length_active': {
|
|
55
59
|
zh: 'ℹ --no-debias-length 已生效:judge prompt 去掉长度去偏指令(debias-off 变体),hash 与默认开启时不同。\n',
|
|
56
60
|
en: 'ℹ --no-debias-length is active: the judge prompt drops the length-debias instruction (debias-off variant); its hash differs from the default.\n',
|
|
@@ -45,6 +45,9 @@ export interface RunConfig {
|
|
|
45
45
|
retry?: number;
|
|
46
46
|
resume?: string;
|
|
47
47
|
layeredStats?: boolean;
|
|
48
|
+
/** --holdout-ratio R (0 < R < 1). Hold out a deterministic sample slice; report-finalize
|
|
49
|
+
* computes train vs holdout composite (report.analysis.holdout) for the overfitting gate. */
|
|
50
|
+
holdoutRatio?: number;
|
|
48
51
|
/** --judge-repeat N. Calls LLM judge N times per (sample × dimension). Default 1. */
|
|
49
52
|
judgeRepeat?: number;
|
|
50
53
|
/** Unified judge config. Always non-empty; 1 entry = single judge, ≥ 2 = ensemble.
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import type { Artifact, EvaluationErrorCategory, EvaluationJob, EvaluationRequest, EvaluationRun, JudgeConfig } from '../types/index.js';
|
|
2
|
-
export declare function buildEvaluationRequest({ samplesPath, skillDir, artifacts, model, executor, noJudge, concurrency, timeoutMs, noCache, dryRun, project, owner, tags, repeat, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, effort, }: {
|
|
2
|
+
export declare function buildEvaluationRequest({ samplesPath, skillDir, artifacts, model, executor, noJudge, concurrency, timeoutMs, noCache, dryRun, project, owner, tags, repeat, holdoutRatio, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, effort, }: {
|
|
3
3
|
samplesPath: string;
|
|
4
4
|
skillDir: string;
|
|
5
5
|
artifacts: Artifact[];
|
|
@@ -14,6 +14,7 @@ export declare function buildEvaluationRequest({ samplesPath, skillDir, artifact
|
|
|
14
14
|
owner?: string;
|
|
15
15
|
tags?: string[];
|
|
16
16
|
repeat?: number;
|
|
17
|
+
holdoutRatio?: number;
|
|
17
18
|
batch?: boolean;
|
|
18
19
|
judgeRepeat?: number;
|
|
19
20
|
judgeModels: JudgeConfig[];
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
function nowIso() {
|
|
2
2
|
return new Date().toISOString();
|
|
3
3
|
}
|
|
4
|
-
export function buildEvaluationRequest({ samplesPath, skillDir, artifacts, model, executor, noJudge, concurrency, timeoutMs, noCache, dryRun, project, owner, tags, repeat, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, effort, }) {
|
|
4
|
+
export function buildEvaluationRequest({ samplesPath, skillDir, artifacts, model, executor, noJudge, concurrency, timeoutMs, noCache, dryRun, project, owner, tags, repeat, holdoutRatio, batch, judgeRepeat, judgeModels, bootstrap, bootstrapSamples, lengthDebias, budget, effort, }) {
|
|
5
5
|
return {
|
|
6
6
|
samplesPath,
|
|
7
7
|
skillDir,
|
|
@@ -17,6 +17,7 @@ export function buildEvaluationRequest({ samplesPath, skillDir, artifacts, model
|
|
|
17
17
|
owner,
|
|
18
18
|
tags,
|
|
19
19
|
repeat,
|
|
20
|
+
holdoutRatio,
|
|
20
21
|
batch,
|
|
21
22
|
judgeRepeat,
|
|
22
23
|
judgeModels,
|
|
@@ -61,10 +61,14 @@ export function getCliVersion() {
|
|
|
61
61
|
return PKG.version;
|
|
62
62
|
}
|
|
63
63
|
export function getGitInfo() {
|
|
64
|
+
// stdio 静默 stderr:在非 git 目录(如 omk init 出来的 demo)里 rev-parse 会打印
|
|
65
|
+
// `fatal: not a git repository` 到终端。catch 已把失败兜成 null(报告省略 git 信息),
|
|
66
|
+
// 这条 fatal 对用户是纯噪声,吞掉它。与 skill-loader 的 GIT_PROBE_STDIO 同口径。
|
|
67
|
+
const gitProbeStdio = ['ignore', 'pipe', 'ignore'];
|
|
64
68
|
try {
|
|
65
|
-
const commit = execFileSync('git', ['rev-parse', 'HEAD'], { encoding: 'utf-8' }).trim();
|
|
66
|
-
const branch = execFileSync('git', ['rev-parse', '--abbrev-ref', 'HEAD'], { encoding: 'utf-8' }).trim();
|
|
67
|
-
const dirty = execFileSync('git', ['status', '--porcelain'], { encoding: 'utf-8' }).trim().length > 0;
|
|
69
|
+
const commit = execFileSync('git', ['rev-parse', 'HEAD'], { encoding: 'utf-8', stdio: gitProbeStdio }).trim();
|
|
70
|
+
const branch = execFileSync('git', ['rev-parse', '--abbrev-ref', 'HEAD'], { encoding: 'utf-8', stdio: gitProbeStdio }).trim();
|
|
71
|
+
const dirty = execFileSync('git', ['status', '--porcelain'], { encoding: 'utf-8', stdio: gitProbeStdio }).trim().length > 0;
|
|
68
72
|
return { commit, commitShort: commit.slice(0, 7), branch, dirty };
|
|
69
73
|
}
|
|
70
74
|
catch {
|
|
@@ -0,0 +1,66 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Holdout split + train/holdout composite breakdown.
|
|
3
|
+
*
|
|
4
|
+
* Deterministic (no RNG) train/holdout partitioning of a sample set, plus the
|
|
5
|
+
* subset-composite recompute that lets `omk eval --holdout-ratio` and `omk evolve`
|
|
6
|
+
* score a variant on a withheld slice using the *same* aggregation as the headline
|
|
7
|
+
* composite (`buildVariantSummary`). Lives in eval-core so both the eval pipeline
|
|
8
|
+
* and the authoring/evolve loop depend *down* into it (authoring → eval-core is the
|
|
9
|
+
* established direction).
|
|
10
|
+
*
|
|
11
|
+
* A large train − holdout composite gap is the generalization / sample-set-overfitting
|
|
12
|
+
* signal the verdict's overfitting gate reads (`src/eval-core/verdict.ts`).
|
|
13
|
+
*/
|
|
14
|
+
import type { Report, HoldoutBreakdown } from '../types/index.js';
|
|
15
|
+
/** A train / holdout partition of a sample set. */
|
|
16
|
+
export interface HoldoutSplit {
|
|
17
|
+
trainIds: Set<string>;
|
|
18
|
+
holdoutIds: Set<string>;
|
|
19
|
+
}
|
|
20
|
+
/** Below this many samples on any side, a split is too small to be meaningful —
|
|
21
|
+
* callers fall back to full-set scoring and mark the breakdown `disabled`. */
|
|
22
|
+
export declare const MIN_HOLDOUT_SUBSET = 3;
|
|
23
|
+
/** Pick `count` ids at an even stride across `ids` (deterministic, no RNG) so the
|
|
24
|
+
* picked subset is representative of the ordering and stable across rounds/runs. */
|
|
25
|
+
export declare function pickByStride(ids: string[], count: number): Set<string>;
|
|
26
|
+
/**
|
|
27
|
+
* Deterministically split sample ids into train / holdout by `ratio` (fraction
|
|
28
|
+
* held out). Holdout members are picked at an even stride so the partition is
|
|
29
|
+
* representative of the ordering, and the split is stable across rounds and runs
|
|
30
|
+
* (no RNG). Returns null when ratio ≤ 0 or either side would drop below
|
|
31
|
+
* MIN_HOLDOUT_SUBSET — the caller then scores on the full set.
|
|
32
|
+
*/
|
|
33
|
+
export declare function splitHoldout(sampleIds: string[], ratio: number): HoldoutSplit | null;
|
|
34
|
+
/**
|
|
35
|
+
* Mean composite over the subset of a report's results whose sample_id is in
|
|
36
|
+
* `ids`, using the same aggregation as the full-run summary
|
|
37
|
+
* (`buildVariantSummary`) so train / holdout scores stay comparable to the
|
|
38
|
+
* headline composite. Returns 0 when the subset has no scorable entries.
|
|
39
|
+
*/
|
|
40
|
+
export declare function subsetCompositeScore(report: Report, variantKey: string, ids: Set<string>): number;
|
|
41
|
+
/**
|
|
42
|
+
* How many subset results actually produced a usable composite (> 0) for a variant.
|
|
43
|
+
* `buildVariantSummary` averages only `compositeScore > 0` entries (schema.ts), so
|
|
44
|
+
* the mean can rest on far fewer samples than the authored split size when runs
|
|
45
|
+
* flake / partial-error / budget-abort. The overfitting gate must trust THIS count,
|
|
46
|
+
* not the authored `trainCount` / `holdoutCount`, or a 1-of-3 holdout gets dressed
|
|
47
|
+
* up as a 3-sample-backed conclusion.
|
|
48
|
+
*/
|
|
49
|
+
export declare function subsetScorableCount(report: Report, variantKey: string, ids: Set<string>): number;
|
|
50
|
+
/**
|
|
51
|
+
* Train vs holdout composite breakdown per variant for `omk eval --holdout-ratio`.
|
|
52
|
+
* Post-hoc — never perturbs the headline aggregation or bootstrap CI.
|
|
53
|
+
*
|
|
54
|
+
* The split is taken over `sampleIdOrder` — the **stable authored sample order**
|
|
55
|
+
* (the loaded `samples` file order), NOT `report.results`, whose insertion order
|
|
56
|
+
* is the concurrent-completion order and drifts run-to-run. Binding the stride pick
|
|
57
|
+
* to the authored order is what makes the holdout (and the verdict overfitting gate
|
|
58
|
+
* it feeds) deterministic and reproducible. Subset scores are then read from
|
|
59
|
+
* `report.results` by id-set membership, which is order-independent.
|
|
60
|
+
*
|
|
61
|
+
* When the split is too small on either side (< MIN_HOLDOUT_SUBSET) it returns
|
|
62
|
+
* `{ disabled: true }` with an empty `perVariant`, so the verdict overfitting gate
|
|
63
|
+
* stays inert. The testSetHash watermark (gap-spec §7.1) is attached by the caller,
|
|
64
|
+
* shared with gapReports.
|
|
65
|
+
*/
|
|
66
|
+
export declare function computeHoldoutBreakdown(report: Report, variantNames: string[], ratio: number, sampleIdOrder: string[]): HoldoutBreakdown;
|
|
@@ -0,0 +1,118 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Holdout split + train/holdout composite breakdown.
|
|
3
|
+
*
|
|
4
|
+
* Deterministic (no RNG) train/holdout partitioning of a sample set, plus the
|
|
5
|
+
* subset-composite recompute that lets `omk eval --holdout-ratio` and `omk evolve`
|
|
6
|
+
* score a variant on a withheld slice using the *same* aggregation as the headline
|
|
7
|
+
* composite (`buildVariantSummary`). Lives in eval-core so both the eval pipeline
|
|
8
|
+
* and the authoring/evolve loop depend *down* into it (authoring → eval-core is the
|
|
9
|
+
* established direction).
|
|
10
|
+
*
|
|
11
|
+
* A large train − holdout composite gap is the generalization / sample-set-overfitting
|
|
12
|
+
* signal the verdict's overfitting gate reads (`src/eval-core/verdict.ts`).
|
|
13
|
+
*/
|
|
14
|
+
import { buildVariantSummary } from './schema.js';
|
|
15
|
+
/** Below this many samples on any side, a split is too small to be meaningful —
|
|
16
|
+
* callers fall back to full-set scoring and mark the breakdown `disabled`. */
|
|
17
|
+
export const MIN_HOLDOUT_SUBSET = 3;
|
|
18
|
+
/** Pick `count` ids at an even stride across `ids` (deterministic, no RNG) so the
|
|
19
|
+
* picked subset is representative of the ordering and stable across rounds/runs. */
|
|
20
|
+
export function pickByStride(ids, count) {
|
|
21
|
+
const picked = new Set();
|
|
22
|
+
if (count <= 0)
|
|
23
|
+
return picked;
|
|
24
|
+
const stride = ids.length / count;
|
|
25
|
+
for (let k = 0; k < count; k++)
|
|
26
|
+
picked.add(ids[Math.floor(k * stride)]);
|
|
27
|
+
return picked;
|
|
28
|
+
}
|
|
29
|
+
/**
|
|
30
|
+
* Deterministically split sample ids into train / holdout by `ratio` (fraction
|
|
31
|
+
* held out). Holdout members are picked at an even stride so the partition is
|
|
32
|
+
* representative of the ordering, and the split is stable across rounds and runs
|
|
33
|
+
* (no RNG). Returns null when ratio ≤ 0 or either side would drop below
|
|
34
|
+
* MIN_HOLDOUT_SUBSET — the caller then scores on the full set.
|
|
35
|
+
*/
|
|
36
|
+
export function splitHoldout(sampleIds, ratio) {
|
|
37
|
+
if (!(ratio > 0) || sampleIds.length === 0)
|
|
38
|
+
return null;
|
|
39
|
+
const holdoutCount = Math.round(sampleIds.length * ratio);
|
|
40
|
+
const trainCount = sampleIds.length - holdoutCount;
|
|
41
|
+
if (holdoutCount < MIN_HOLDOUT_SUBSET || trainCount < MIN_HOLDOUT_SUBSET)
|
|
42
|
+
return null;
|
|
43
|
+
const holdoutIds = pickByStride(sampleIds, holdoutCount);
|
|
44
|
+
const trainIds = new Set(sampleIds.filter((id) => !holdoutIds.has(id)));
|
|
45
|
+
return { trainIds, holdoutIds };
|
|
46
|
+
}
|
|
47
|
+
/**
|
|
48
|
+
* Mean composite over the subset of a report's results whose sample_id is in
|
|
49
|
+
* `ids`, using the same aggregation as the full-run summary
|
|
50
|
+
* (`buildVariantSummary`) so train / holdout scores stay comparable to the
|
|
51
|
+
* headline composite. Returns 0 when the subset has no scorable entries.
|
|
52
|
+
*/
|
|
53
|
+
export function subsetCompositeScore(report, variantKey, ids) {
|
|
54
|
+
const entries = [];
|
|
55
|
+
for (const r of report.results) {
|
|
56
|
+
if (!ids.has(r.sample_id))
|
|
57
|
+
continue;
|
|
58
|
+
const v = r.variants[variantKey];
|
|
59
|
+
if (v)
|
|
60
|
+
entries.push(v);
|
|
61
|
+
}
|
|
62
|
+
if (entries.length === 0)
|
|
63
|
+
return 0;
|
|
64
|
+
return buildVariantSummary(entries).avgCompositeScore ?? 0;
|
|
65
|
+
}
|
|
66
|
+
/**
|
|
67
|
+
* How many subset results actually produced a usable composite (> 0) for a variant.
|
|
68
|
+
* `buildVariantSummary` averages only `compositeScore > 0` entries (schema.ts), so
|
|
69
|
+
* the mean can rest on far fewer samples than the authored split size when runs
|
|
70
|
+
* flake / partial-error / budget-abort. The overfitting gate must trust THIS count,
|
|
71
|
+
* not the authored `trainCount` / `holdoutCount`, or a 1-of-3 holdout gets dressed
|
|
72
|
+
* up as a 3-sample-backed conclusion.
|
|
73
|
+
*/
|
|
74
|
+
export function subsetScorableCount(report, variantKey, ids) {
|
|
75
|
+
let n = 0;
|
|
76
|
+
for (const r of report.results) {
|
|
77
|
+
if (!ids.has(r.sample_id))
|
|
78
|
+
continue;
|
|
79
|
+
const v = r.variants[variantKey];
|
|
80
|
+
if (v && typeof v.compositeScore === 'number' && v.compositeScore > 0)
|
|
81
|
+
n++;
|
|
82
|
+
}
|
|
83
|
+
return n;
|
|
84
|
+
}
|
|
85
|
+
/**
|
|
86
|
+
* Train vs holdout composite breakdown per variant for `omk eval --holdout-ratio`.
|
|
87
|
+
* Post-hoc — never perturbs the headline aggregation or bootstrap CI.
|
|
88
|
+
*
|
|
89
|
+
* The split is taken over `sampleIdOrder` — the **stable authored sample order**
|
|
90
|
+
* (the loaded `samples` file order), NOT `report.results`, whose insertion order
|
|
91
|
+
* is the concurrent-completion order and drifts run-to-run. Binding the stride pick
|
|
92
|
+
* to the authored order is what makes the holdout (and the verdict overfitting gate
|
|
93
|
+
* it feeds) deterministic and reproducible. Subset scores are then read from
|
|
94
|
+
* `report.results` by id-set membership, which is order-independent.
|
|
95
|
+
*
|
|
96
|
+
* When the split is too small on either side (< MIN_HOLDOUT_SUBSET) it returns
|
|
97
|
+
* `{ disabled: true }` with an empty `perVariant`, so the verdict overfitting gate
|
|
98
|
+
* stays inert. The testSetHash watermark (gap-spec §7.1) is attached by the caller,
|
|
99
|
+
* shared with gapReports.
|
|
100
|
+
*/
|
|
101
|
+
export function computeHoldoutBreakdown(report, variantNames, ratio, sampleIdOrder) {
|
|
102
|
+
const split = splitHoldout(sampleIdOrder, ratio);
|
|
103
|
+
if (!split) {
|
|
104
|
+
return { ratio, disabled: true, perVariant: {} };
|
|
105
|
+
}
|
|
106
|
+
const perVariant = {};
|
|
107
|
+
for (const v of variantNames) {
|
|
108
|
+
perVariant[v] = {
|
|
109
|
+
trainScore: Number(subsetCompositeScore(report, v, split.trainIds).toFixed(4)),
|
|
110
|
+
holdoutScore: Number(subsetCompositeScore(report, v, split.holdoutIds).toFixed(4)),
|
|
111
|
+
trainCount: split.trainIds.size,
|
|
112
|
+
holdoutCount: split.holdoutIds.size,
|
|
113
|
+
trainScorable: subsetScorableCount(report, v, split.trainIds),
|
|
114
|
+
holdoutScorable: subsetScorableCount(report, v, split.holdoutIds),
|
|
115
|
+
};
|
|
116
|
+
}
|
|
117
|
+
return { ratio, perVariant };
|
|
118
|
+
}
|
|
@@ -27,7 +27,7 @@
|
|
|
27
27
|
* proven optimal. Each rule's source (NIST AI 800-3 / Krippendorff thresholds /
|
|
28
28
|
* empirical) is documented inline so users can audit and override.
|
|
29
29
|
*/
|
|
30
|
-
import type { Report } from '../types/index.js';
|
|
30
|
+
import type { Lang, Report } from '../types/index.js';
|
|
31
31
|
/**
|
|
32
32
|
* Below this sample count a non-significant diff is read as UNDERPOWERED
|
|
33
33
|
* (only large effects are detectable) rather than NOISE. Matches the
|
|
@@ -63,6 +63,19 @@ export declare const STABILITY_UNSTABLE_CV = 0.15;
|
|
|
63
63
|
* doc ↔ code parity guarded by `test/scripts/doc-constants-drift.test.ts`.
|
|
64
64
|
*/
|
|
65
65
|
export declare const DEFAULT_GATE_THRESHOLD = 3.5;
|
|
66
|
+
/**
|
|
67
|
+
* Train − holdout composite gap (1-5 scale) above which `omk eval --holdout-ratio`
|
|
68
|
+
* is read as **sample-set overfitting**: the gain lives on the samples the skill
|
|
69
|
+
* was shaped around and does not carry to the held-out slice. Like the stability
|
|
70
|
+
* gate, a would-be PROGRESS is then downgraded to CAUTIOUS — a win that does not
|
|
71
|
+
* generalize is not shippable.
|
|
72
|
+
* **Pragmatic default, not from an external standard**: 0.5 is 10% of the 1-5 scale
|
|
73
|
+
* — small enough to catch a real generalization drop, wide enough to ignore the
|
|
74
|
+
* sampling noise of a small holdout slice. Only fires when a holdout split is
|
|
75
|
+
* present (opt-in), so it never moves a default report's verdict.
|
|
76
|
+
* doc ↔ code parity guarded by `test/scripts/doc-constants-drift.test.ts`.
|
|
77
|
+
*/
|
|
78
|
+
export declare const OVERFITTING_GAP_THRESHOLD = 0.5;
|
|
66
79
|
export type VerdictLevel = 'PROGRESS' | 'CAUTIOUS' | 'REGRESS' | 'NOISE' | 'UNDERPOWERED' | 'SOLO';
|
|
67
80
|
export interface VerdictResult {
|
|
68
81
|
level: VerdictLevel;
|
|
@@ -84,8 +97,37 @@ export interface VerdictResult {
|
|
|
84
97
|
* 让用户感受到 single-run 的盲区,而不是默默不提。 */
|
|
85
98
|
stability?: string;
|
|
86
99
|
judgeAgreement?: string;
|
|
100
|
+
/** Overfitting (train vs holdout) caveat — only present under `--holdout-ratio`. */
|
|
101
|
+
overfitting?: string;
|
|
102
|
+
/** Knowledge-gap caveat — informational, watermarked, never gates (gap-spec §8). */
|
|
103
|
+
gapSignal?: string;
|
|
87
104
|
shipRecommendation?: string;
|
|
88
105
|
};
|
|
106
|
+
/** The pair the top-level verdict is about — the worst pair from the roll-up, NOT
|
|
107
|
+
* variants[1]. Surfaces let the HTML pill name the right treatment in a
|
|
108
|
+
* control-vs-many report instead of re-deriving from the first pair. Undefined for
|
|
109
|
+
* SOLO / pairless reports. */
|
|
110
|
+
representative?: {
|
|
111
|
+
control: string;
|
|
112
|
+
treatment: string;
|
|
113
|
+
};
|
|
114
|
+
/** Structured caveats (language-neutral) so HTML / other surfaces can i18n them
|
|
115
|
+
* instead of re-parsing the zh `rationale` strings. Present only when the caveat
|
|
116
|
+
* fires; mirrors `rationale.overfitting` / `rationale.gapSignal`. */
|
|
117
|
+
caveats?: {
|
|
118
|
+
overfitting?: {
|
|
119
|
+
variant: string;
|
|
120
|
+
trainScore: number;
|
|
121
|
+
holdoutScore: number;
|
|
122
|
+
gap: number;
|
|
123
|
+
};
|
|
124
|
+
gapSignal?: {
|
|
125
|
+
variant: string;
|
|
126
|
+
gapRatePct: number;
|
|
127
|
+
testSetPath?: string | null;
|
|
128
|
+
testSetHash?: string | null;
|
|
129
|
+
};
|
|
130
|
+
};
|
|
89
131
|
/** Variants present in the report (best-vs-control framing). */
|
|
90
132
|
variants: string[];
|
|
91
133
|
}
|
|
@@ -129,4 +171,5 @@ export declare function medianStabilityCV(report: Report): {
|
|
|
129
171
|
*/
|
|
130
172
|
export declare function formatVerdictText(result: VerdictResult, options?: {
|
|
131
173
|
verbose?: boolean;
|
|
174
|
+
lang?: Lang;
|
|
132
175
|
}): string;
|
|
@@ -30,6 +30,7 @@
|
|
|
30
30
|
import { evaluateLayerGates } from './layer-gates.js';
|
|
31
31
|
import { ciLevelLabel } from './bootstrap.js';
|
|
32
32
|
import { analyzeJudgeIndependence } from './judge-independence.js';
|
|
33
|
+
import { MIN_HOLDOUT_SUBSET } from './holdout.js';
|
|
33
34
|
/**
|
|
34
35
|
* Below this sample count a non-significant diff is read as UNDERPOWERED
|
|
35
36
|
* (only large effects are detectable) rather than NOISE. Matches the
|
|
@@ -65,6 +66,19 @@ export const STABILITY_UNSTABLE_CV = 0.15;
|
|
|
65
66
|
* doc ↔ code parity guarded by `test/scripts/doc-constants-drift.test.ts`.
|
|
66
67
|
*/
|
|
67
68
|
export const DEFAULT_GATE_THRESHOLD = 3.5;
|
|
69
|
+
/**
|
|
70
|
+
* Train − holdout composite gap (1-5 scale) above which `omk eval --holdout-ratio`
|
|
71
|
+
* is read as **sample-set overfitting**: the gain lives on the samples the skill
|
|
72
|
+
* was shaped around and does not carry to the held-out slice. Like the stability
|
|
73
|
+
* gate, a would-be PROGRESS is then downgraded to CAUTIOUS — a win that does not
|
|
74
|
+
* generalize is not shippable.
|
|
75
|
+
* **Pragmatic default, not from an external standard**: 0.5 is 10% of the 1-5 scale
|
|
76
|
+
* — small enough to catch a real generalization drop, wide enough to ignore the
|
|
77
|
+
* sampling noise of a small holdout slice. Only fires when a holdout split is
|
|
78
|
+
* present (opt-in), so it never moves a default report's verdict.
|
|
79
|
+
* doc ↔ code parity guarded by `test/scripts/doc-constants-drift.test.ts`.
|
|
80
|
+
*/
|
|
81
|
+
export const OVERFITTING_GAP_THRESHOLD = 0.5;
|
|
68
82
|
/**
|
|
69
83
|
* Compute a verdict for a finished report. Pure function — no I/O.
|
|
70
84
|
*/
|
|
@@ -79,17 +93,28 @@ export function computeVerdict(report, options = {}) {
|
|
|
79
93
|
const gate = evaluateLayerGates(summary, gateThreshold);
|
|
80
94
|
// SOLO 只有绝对分、无 A/B 差值可抵消自我偏好,故同厂商评委的 caveat 更该出。
|
|
81
95
|
const judgeInd = judgeIndependenceCaveat(report);
|
|
96
|
+
// 过拟合 / gap 在 SOLO 也有意义(单变体是否泛化 / 缺口多大),但 SOLO 无 PROGRESS 可降,只附提示不门控。
|
|
97
|
+
const overfit = overfittingCaveat(report);
|
|
98
|
+
const gap = gapSignalCaveat(report);
|
|
82
99
|
return {
|
|
83
100
|
level: 'SOLO',
|
|
84
101
|
headline: (gate.allPass
|
|
85
102
|
? `SOLO · single variant, three-layer gate PASS @ threshold ${gateThreshold}`
|
|
86
|
-
: `SOLO · single variant, three-layer gate FAIL — see ci output`) + judgeInd.note,
|
|
103
|
+
: `SOLO · single variant, three-layer gate FAIL — see ci output`) + judgeInd.note + overfit.note + gap.note,
|
|
87
104
|
rationale: {
|
|
88
105
|
layerWinners: gate.lines.join('; '),
|
|
89
106
|
sampleSize: `N=${sampleCount}`,
|
|
90
107
|
stability: formatStability(report),
|
|
91
108
|
...(judgeInd.rationale ? { judgeAgreement: judgeInd.rationale } : {}),
|
|
109
|
+
...(overfit.rationale ? { overfitting: overfit.rationale } : {}),
|
|
110
|
+
...(gap.rationale ? { gapSignal: gap.rationale } : {}),
|
|
92
111
|
},
|
|
112
|
+
...((overfit.data || gap.data) ? {
|
|
113
|
+
caveats: {
|
|
114
|
+
...(overfit.data ? { overfitting: overfit.data } : {}),
|
|
115
|
+
...(gap.data ? { gapSignal: gap.data } : {}),
|
|
116
|
+
},
|
|
117
|
+
} : {}),
|
|
93
118
|
variants,
|
|
94
119
|
};
|
|
95
120
|
}
|
|
@@ -118,11 +143,17 @@ export function computeVerdict(report, options = {}) {
|
|
|
118
143
|
// 不再加码,顺序与 worst-case roll-up 一致。
|
|
119
144
|
const stab = medianStabilityCV(report);
|
|
120
145
|
const stabilityGated = topLevel === 'PROGRESS' && stab !== null && stab.cv > STABILITY_UNSTABLE_CV;
|
|
121
|
-
|
|
146
|
+
// 过拟合门控:与稳定性门控同形——opt-in holdout 下 train/holdout 分差过大 → PROGRESS 降 CAUTIOUS。
|
|
147
|
+
// overfittingCaveat 首行短路无 holdout 的报告,故默认报告 level 与 headline 逐字节不变。
|
|
148
|
+
const overfit = overfittingCaveat(report);
|
|
149
|
+
const overfitGated = topLevel === 'PROGRESS' && overfit.gated;
|
|
150
|
+
const level = (stabilityGated || overfitGated) ? 'CAUTIOUS' : topLevel;
|
|
122
151
|
const stabilityNote = stabilityGated && stab
|
|
123
152
|
? ` · 显著但 run-to-run 不稳(CV=${(stab.cv * 100).toFixed(1)}% > ${(STABILITY_UNSTABLE_CV * 100).toFixed(0)}%)`
|
|
124
153
|
: '';
|
|
125
154
|
const judgeInd = judgeIndependenceCaveat(report);
|
|
155
|
+
// gap 软提示:不改 level(spec §8),低缺口 / 无 gapReports 时空串,headline 逐字节不变。
|
|
156
|
+
const gap = gapSignalCaveat(report);
|
|
126
157
|
const significance = representative
|
|
127
158
|
? formatSignificance(representative)
|
|
128
159
|
: 'no pairwise comparison available — was --bootstrap used?';
|
|
@@ -134,7 +165,7 @@ export function computeVerdict(report, options = {}) {
|
|
|
134
165
|
return {
|
|
135
166
|
level,
|
|
136
167
|
headline: representative
|
|
137
|
-
? `${level} · ${representative.treatment} vs ${representative.control}: ${representative.headline}${stabilityNote}${judgeInd.note}`
|
|
168
|
+
? `${level} · ${representative.treatment} vs ${representative.control}: ${representative.headline}${stabilityNote}${judgeInd.note}${overfit.note}${gap.note}`
|
|
138
169
|
: `${level} · ${variants.length} variants`,
|
|
139
170
|
perPair,
|
|
140
171
|
rationale: {
|
|
@@ -143,8 +174,17 @@ export function computeVerdict(report, options = {}) {
|
|
|
143
174
|
sampleSize,
|
|
144
175
|
stability,
|
|
145
176
|
judgeAgreement,
|
|
177
|
+
...(overfit.rationale ? { overfitting: overfit.rationale } : {}),
|
|
178
|
+
...(gap.rationale ? { gapSignal: gap.rationale } : {}),
|
|
146
179
|
shipRecommendation,
|
|
147
180
|
},
|
|
181
|
+
...(representative ? { representative: { control: representative.control, treatment: representative.treatment } } : {}),
|
|
182
|
+
...((overfit.data || gap.data) ? {
|
|
183
|
+
caveats: {
|
|
184
|
+
...(overfit.data ? { overfitting: overfit.data } : {}),
|
|
185
|
+
...(gap.data ? { gapSignal: gap.data } : {}),
|
|
186
|
+
},
|
|
187
|
+
} : {}),
|
|
148
188
|
variants,
|
|
149
189
|
};
|
|
150
190
|
}
|
|
@@ -414,12 +454,125 @@ function judgeIndependenceCaveat(report) {
|
|
|
414
454
|
rationale: `${reasons.join(';')} —— 绝对分可能偏高;换跨厂商评委(--judge-models)或挂 gold(omk eval gold compare)校准`,
|
|
415
455
|
};
|
|
416
456
|
}
|
|
417
|
-
|
|
457
|
+
/**
|
|
458
|
+
* Overfitting caveat from the opt-in train/holdout breakdown (`--holdout-ratio`).
|
|
459
|
+
* `gated` drives a PROGRESS → CAUTIOUS downgrade (a win that does not carry to the
|
|
460
|
+
* held-out slice is not shippable), mirroring the stability gate. **First line
|
|
461
|
+
* short-circuits when there is no holdout split**, so default reports (no
|
|
462
|
+
* `analysis.holdout`) are byte-identical — the gate only ever fires when the user
|
|
463
|
+
* opted into a holdout, which is brand-new behaviour with no historical reports.
|
|
464
|
+
*/
|
|
465
|
+
/**
|
|
466
|
+
* The treatment variants a report-level caveat should scan. control = variants[0];
|
|
467
|
+
* treatments = the rest. For a single-variant (SOLO) report there is no control,
|
|
468
|
+
* so the lone variant is itself the subject. Scanning **all** treatments (not just
|
|
469
|
+
* variants[1]) keeps the caveats aligned with the verdict's worst-case roll-up over
|
|
470
|
+
* `perPair` — a control-vs-many report must not miss the 2nd/3rd treatment.
|
|
471
|
+
*/
|
|
472
|
+
function treatmentVariants(report) {
|
|
473
|
+
const variants = report.meta?.variants ?? [];
|
|
474
|
+
return variants.length >= 2 ? variants.slice(1) : variants.slice(0, 1);
|
|
475
|
+
}
|
|
476
|
+
function overfittingCaveat(report) {
|
|
477
|
+
const holdout = report.analysis?.holdout;
|
|
478
|
+
if (!holdout || holdout.disabled)
|
|
479
|
+
return { note: '', gated: false };
|
|
480
|
+
// Worst-case over all treatments — matches the verdict roll-up. A PROGRESS top-level
|
|
481
|
+
// means every pair passed, so a single overfitting treatment must still downgrade it.
|
|
482
|
+
let worst = null;
|
|
483
|
+
for (const t of treatmentVariants(report)) {
|
|
484
|
+
const pv = holdout.perVariant[t];
|
|
485
|
+
if (!pv)
|
|
486
|
+
continue;
|
|
487
|
+
// 门控绑「实际可评分条目数」,不是 authored 切分数:某侧 3 条里只有 1 条真出分(其余
|
|
488
|
+
// error / budget-abort)时,分差只有 1 个样本支撑,把它包装成「3 条结论」会误判过拟合。
|
|
489
|
+
// 两侧 scorable 都 ≥ MIN_HOLDOUT_SUBSET 才信。这也顺带挡掉 score=0(scorable=0)的测量假象。
|
|
490
|
+
if (pv.trainScorable < MIN_HOLDOUT_SUBSET || pv.holdoutScorable < MIN_HOLDOUT_SUBSET)
|
|
491
|
+
continue;
|
|
492
|
+
const gap = pv.trainScore - pv.holdoutScore;
|
|
493
|
+
if (gap <= OVERFITTING_GAP_THRESHOLD)
|
|
494
|
+
continue;
|
|
495
|
+
if (!worst || gap > worst.gap)
|
|
496
|
+
worst = { variant: t, train: pv.trainScore, holdout: pv.holdoutScore, gap };
|
|
497
|
+
}
|
|
498
|
+
if (!worst)
|
|
499
|
+
return { note: '', gated: false };
|
|
500
|
+
return {
|
|
501
|
+
note: ` · 过拟合敞口(${worst.variant}: train ${worst.train.toFixed(2)} − holdout ${worst.holdout.toFixed(2)} = ${worst.gap.toFixed(2)} > ${OVERFITTING_GAP_THRESHOLD})`,
|
|
502
|
+
rationale: `${worst.variant} train/holdout 综合分差 ${worst.gap.toFixed(2)} 超阈值 ${OVERFITTING_GAP_THRESHOLD}(holdout ratio ${holdout.ratio})—— 提升可能是对用例集过拟合、对 holdout 不泛化;扩充用例集或换独立外验集复核`,
|
|
503
|
+
gated: true,
|
|
504
|
+
data: { variant: worst.variant, trainScore: worst.train, holdoutScore: worst.holdout, gap: worst.gap },
|
|
505
|
+
};
|
|
506
|
+
}
|
|
507
|
+
/**
|
|
508
|
+
* Knowledge-gap rate above which the verdict appends an **informational** caveat.
|
|
509
|
+
* Internal-only, deliberately NOT exported: gap rate is informational, never a
|
|
510
|
+
* gate (knowledge-gap-signal-spec.md §8), so exposing this as a tunable constant
|
|
511
|
+
* would invite reading it as a pass/fail line.
|
|
512
|
+
*/
|
|
513
|
+
const GAP_CAVEAT_BAND = 0.2;
|
|
514
|
+
/**
|
|
515
|
+
* Knowledge-gap caveat (`report.analysis.gapReports`). **Soft only — never changes
|
|
516
|
+
* the verdict level** (spec §8: a nudge, not a fail). Surfaces the treatment's gap
|
|
517
|
+
* rate with its mandatory test-set watermark (spec §7.1: a gap number without a
|
|
518
|
+
* watermark is invalid output) plus the "informational, not completeness" framing.
|
|
519
|
+
* Empty note below the band, so the common low-gap case keeps verdict headlines
|
|
520
|
+
* byte-identical.
|
|
521
|
+
*/
|
|
522
|
+
function gapSignalCaveat(report) {
|
|
523
|
+
const gapReports = report.analysis?.gapReports;
|
|
524
|
+
if (!gapReports)
|
|
525
|
+
return { note: '' };
|
|
526
|
+
// Worst-case (highest gap rate) over all treatments — a control-vs-many report must
|
|
527
|
+
// surface the noisiest treatment, not just variants[1]. spec §7.1: gap 数必须带
|
|
528
|
+
// test-set 水印,否则视为无效输出 —— 无水印的条目不参与(而非吐裸缺口率)。生产报告
|
|
529
|
+
// 恒带水印(report-finalize 强制),此处只防手搓 / 退化报告。
|
|
530
|
+
let gr = null;
|
|
531
|
+
for (const t of treatmentVariants(report)) {
|
|
532
|
+
const cand = gapReports[t];
|
|
533
|
+
if (!cand || cand.gapRate < GAP_CAVEAT_BAND)
|
|
534
|
+
continue;
|
|
535
|
+
if (!cand.testSetPath && !cand.testSetHash)
|
|
536
|
+
continue;
|
|
537
|
+
if (!gr || cand.gapRate > gr.gapRate)
|
|
538
|
+
gr = cand;
|
|
539
|
+
}
|
|
540
|
+
if (!gr)
|
|
541
|
+
return { note: '' };
|
|
542
|
+
const pct = (gr.gapRate * 100).toFixed(0);
|
|
543
|
+
const shortHash = gr.testSetHash ? gr.testSetHash.slice(0, 8) : '';
|
|
544
|
+
const tag = shortHash || gr.testSetPath;
|
|
545
|
+
const watermark = gr.testSetPath
|
|
546
|
+
? `${gr.testSetPath}${shortHash ? ` @ ${shortHash}` : ''}`
|
|
547
|
+
: shortHash;
|
|
548
|
+
return {
|
|
549
|
+
note: ` · 知识缺口率 ${pct}% @ ${tag}`,
|
|
550
|
+
rationale: `知识缺口率 ${pct}%(test set: ${watermark},N=${gr.sampleCount})—— informational,反映当前用例集与知识库的交互、非完备性度量;高缺口提示扩充知识库或复核未覆盖文件`,
|
|
551
|
+
data: { variant: gr.variant, gapRatePct: Number(pct), testSetPath: gr.testSetPath, testSetHash: gr.testSetHash },
|
|
552
|
+
};
|
|
553
|
+
}
|
|
554
|
+
function recommendation(level, _perPair, lang = 'en') {
|
|
555
|
+
if (lang === 'zh') {
|
|
556
|
+
switch (level) {
|
|
557
|
+
case 'PROGRESS':
|
|
558
|
+
return '可发布 —— 实验组显著更优,且通过所有分层门控。';
|
|
559
|
+
case 'CAUTIOUS':
|
|
560
|
+
return '需排查 —— 提升是真的,但至少触发了一条告警(门控破损 / 提升微不足道 / 仅部分恢复 / 评委分歧 / 跨轮不稳 / 训练-留出过拟合)。不要盲发。';
|
|
561
|
+
case 'REGRESS':
|
|
562
|
+
return '勿发布 —— 实验组退步。检查最差的那一层,修好再重跑。';
|
|
563
|
+
case 'NOISE':
|
|
564
|
+
return '不下结论 —— 差异置信区间跨过 0,当前 N 下分辨不出效果。';
|
|
565
|
+
case 'UNDERPOWERED':
|
|
566
|
+
return '数据不足 —— 增加用例数(建议 2× 当前)后重跑。';
|
|
567
|
+
case 'SOLO':
|
|
568
|
+
return '缺对照 —— 单变体报告。用 --control baseline --treatment <名字> 重跑。';
|
|
569
|
+
}
|
|
570
|
+
}
|
|
418
571
|
switch (level) {
|
|
419
572
|
case 'PROGRESS':
|
|
420
573
|
return 'SHIP — treatment is significantly better and passes all layer gates.';
|
|
421
574
|
case 'CAUTIOUS':
|
|
422
|
-
return 'INVESTIGATE — the gain is real but at least one warning fired (broken gate, trivially small, partial recovery, judge dissent,
|
|
575
|
+
return 'INVESTIGATE — the gain is real but at least one warning fired (broken gate, trivially small, partial recovery, judge dissent, run-to-run unstable, or train/holdout overfitting). Do not ship blind.';
|
|
423
576
|
case 'REGRESS':
|
|
424
577
|
return 'DO NOT SHIP — treatment regresses. Check the worst layer and re-run with the fix.';
|
|
425
578
|
case 'NOISE':
|
|
@@ -435,22 +588,31 @@ function recommendation(level, _perPair) {
|
|
|
435
588
|
* spec — one verdict + four rationale bullets + one ship recommendation.
|
|
436
589
|
*/
|
|
437
590
|
export function formatVerdictText(result, options = {}) {
|
|
591
|
+
// lang 默认 'en':保留既有英文输出逐字节不变(verdict.test 与历史 CLI 行为)。zh 只本地化
|
|
592
|
+
// 标签与 ship 建议;headline 是 Δ/CI/N 统计记号 —— 跨语言中性、且会随 report 持久化,
|
|
593
|
+
// 不翻译(翻它=改可比性锚点)。recommendation 在 format 时按 lang 重新派生,不动 computeVerdict。
|
|
594
|
+
const zh = options.lang === 'zh';
|
|
438
595
|
const lines = [];
|
|
439
|
-
lines.push(`Verdict: ${result.level}`);
|
|
596
|
+
lines.push(zh ? `判定:${result.level}` : `Verdict: ${result.level}`);
|
|
440
597
|
lines.push(` ${result.headline}`);
|
|
441
598
|
if (result.rationale.layerWinners)
|
|
442
|
-
lines.push(` Layer winners: ${result.rationale.layerWinners}`);
|
|
599
|
+
lines.push(zh ? ` 分层优胜:${result.rationale.layerWinners}` : ` Layer winners: ${result.rationale.layerWinners}`);
|
|
443
600
|
if (result.rationale.sampleSize)
|
|
444
|
-
lines.push(` Sample size: ${result.rationale.sampleSize}`);
|
|
601
|
+
lines.push(zh ? ` 用例规模:${result.rationale.sampleSize}` : ` Sample size: ${result.rationale.sampleSize}`);
|
|
445
602
|
if (result.rationale.stability)
|
|
446
|
-
lines.push(` Stability: ${result.rationale.stability}`);
|
|
603
|
+
lines.push(zh ? ` 跨轮稳定:${result.rationale.stability}` : ` Stability: ${result.rationale.stability}`);
|
|
447
604
|
if (result.rationale.judgeAgreement)
|
|
448
|
-
lines.push(` Judge α: ${result.rationale.judgeAgreement}`);
|
|
449
|
-
if (result.rationale.
|
|
450
|
-
lines.push(` ${result.rationale.
|
|
605
|
+
lines.push(zh ? ` 评委 α:${result.rationale.judgeAgreement}` : ` Judge α: ${result.rationale.judgeAgreement}`);
|
|
606
|
+
if (result.rationale.overfitting)
|
|
607
|
+
lines.push(zh ? ` 过拟合:${result.rationale.overfitting}` : ` Overfitting: ${result.rationale.overfitting}`);
|
|
608
|
+
if (result.rationale.gapSignal)
|
|
609
|
+
lines.push(zh ? ` 知识缺口:${result.rationale.gapSignal}` : ` Gap signal: ${result.rationale.gapSignal}`);
|
|
610
|
+
if (result.rationale.shipRecommendation) {
|
|
611
|
+
lines.push(` ${zh ? recommendation(result.level, [], 'zh') : result.rationale.shipRecommendation}`);
|
|
612
|
+
}
|
|
451
613
|
if (options.verbose && result.perPair && result.perPair.length > 1) {
|
|
452
614
|
lines.push('');
|
|
453
|
-
lines.push(' Per-pair detail:');
|
|
615
|
+
lines.push(zh ? ' 逐对明细:' : ' Per-pair detail:');
|
|
454
616
|
for (const p of result.perPair) {
|
|
455
617
|
lines.push(` ${p.level}: ${p.treatment} vs ${p.control} — ${p.headline}`);
|
|
456
618
|
}
|
|
@@ -15,6 +15,7 @@
|
|
|
15
15
|
import { analyzeResults } from '../../analysis/report-diagnostics.js';
|
|
16
16
|
import { computeReportCoverage } from '../../analysis/coverage-analyzer.js';
|
|
17
17
|
import { computeReportGapRates } from '../../analysis/gap-analyzer.js';
|
|
18
|
+
import { computeHoldoutBreakdown } from '../../eval-core/holdout.js';
|
|
18
19
|
import { computeTestSetHash } from './test-set-hash.js';
|
|
19
20
|
export function finalizeEvaluationReport({ report, results, artifacts, variantNames, samplesPath, samplesSourceFiles, samples, }) {
|
|
20
21
|
// pass samples so analyzeResults can populate analysis.sampleQuality
|
|
@@ -33,9 +34,13 @@ export function finalizeEvaluationReport({ report, results, artifacts, variantNa
|
|
|
33
34
|
// Gap rate computation runs on every successful report regardless of whether
|
|
34
35
|
// tool trace data is present — text-based signals (markers, hedging) still
|
|
35
36
|
// apply. The samples-file SHA is the mandatory watermark required by spec §7.1.
|
|
37
|
+
// The same hash watermarks the opt-in holdout breakdown below, so it is computed
|
|
38
|
+
// once and shared (both coverage-class numbers must carry the same test-set id).
|
|
39
|
+
const holdoutRatio = report.meta?.request?.holdoutRatio ?? 0;
|
|
36
40
|
const gapReports = computeReportGapRates(report.results, variantNames);
|
|
41
|
+
const needsWatermark = Object.keys(gapReports).length > 0 || holdoutRatio > 0;
|
|
42
|
+
const testSetHash = needsWatermark ? computeTestSetHash(samplesPath, samplesSourceFiles) : null;
|
|
37
43
|
if (Object.keys(gapReports).length > 0) {
|
|
38
|
-
const testSetHash = computeTestSetHash(samplesPath, samplesSourceFiles);
|
|
39
44
|
for (const variant of variantNames) {
|
|
40
45
|
const gr = gapReports[variant];
|
|
41
46
|
if (!gr)
|
|
@@ -45,5 +50,18 @@ export function finalizeEvaluationReport({ report, results, artifacts, variantNa
|
|
|
45
50
|
}
|
|
46
51
|
report.analysis.gapReports = gapReports;
|
|
47
52
|
}
|
|
53
|
+
// Opt-in train/holdout generalization breakdown (`omk eval --holdout-ratio`).
|
|
54
|
+
// Post-hoc over report.results — never perturbs the headline composite or the
|
|
55
|
+
// bootstrap CI. A large train − holdout gap is the overfitting signal the verdict
|
|
56
|
+
// overfitting gate reads (src/eval-core/verdict.ts); absent on default runs.
|
|
57
|
+
if (holdoutRatio > 0) {
|
|
58
|
+
// 切分按 samples 的稳定原始顺序(文件顺序),不依赖 report.results 的并发完成落盘顺序,
|
|
59
|
+
// 否则同批样本在不同并发/时序下 holdout 子集会漂、verdict 过拟合门控跟着漂。
|
|
60
|
+
const sampleIdOrder = samples.map((s) => s.sample_id);
|
|
61
|
+
const holdout = computeHoldoutBreakdown(report, variantNames, holdoutRatio, sampleIdOrder);
|
|
62
|
+
holdout.testSetPath = samplesPath;
|
|
63
|
+
holdout.testSetHash = testSetHash;
|
|
64
|
+
report.analysis.holdout = holdout;
|
|
65
|
+
}
|
|
48
66
|
return report;
|
|
49
67
|
}
|