@tachikomagundam/abathur 0.2.4 → 0.2.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/config/genomes/historian.example.jsonc +1 -1
- package/dist/commands/genome.js +2 -2
- package/dist/commands/graft.js +1 -1
- package/dist/commands/run.js +1 -1
- package/dist/commands/self-eval.js +1 -1
- package/dist/commands/tombstone.js +2 -1
- package/dist/core/evolve/run-bench.js +4 -3
- package/dist/core/evolve/run-loop.js +24 -4
- package/dist/core/evolve/run-plan.js +7 -3
- package/dist/core/evolve/score-bank.js +214 -0
- package/dist/core/graft-rebench.js +3 -0
- package/dist/core/ledger.js +6 -1
- package/dist/core/promote.js +2 -1
- package/dist/core/spec.js +5 -1
- package/dist/core/stats-math.js +39 -0
- package/dist/core/stats.js +125 -33
- package/dist/test/historian-grader-integrity.test.js +1494 -2
- package/dist/test/historian-grader-io.test.js +4 -0
- package/dist/test/historian-run-scenario.test.js +42 -2
- package/dist/test/repopath-seams.test.js +54 -0
- package/dist/test/score-bank.test.js +235 -0
- package/dist/test/seam-gates.test.js +78 -0
- package/dist/test/selfmode-literal.test.js +153 -0
- package/dist/test/stats-acceptance.test.js +329 -0
- package/dist/test/stats.test.js +18 -0
- package/graders/historian/grader-core.d.mts +107 -2
- package/graders/historian/grader-core.mjs +1046 -3
- package/graders/historian/grader.mjs +21 -2
- package/graders/historian/judge-poststage-19.mjs +212 -0
- package/graders/historian/judge-poststage.mjs +195 -0
- package/graders/historian/mutate.sh +59 -19
- package/graders/historian/run-scenario-17.sh +34 -0
- package/graders/historian/run-scenario-19.sh +43 -0
- package/graders/historian/run-scenario.sh +24 -1
- package/package.json +1 -1
- package/plugin/abathur.ts +1 -1
|
@@ -0,0 +1,329 @@
|
|
|
1
|
+
// Acceptance-semantics evaluate() tests (2026-09-24 redesign). Thresholds
|
|
2
|
+
// mirror the LIVE historian genome: halfWidth 0.15, minEffect 0.1. Bank σ
|
|
3
|
+
// values are the real campaign-10 gate-time bank (§2.3 of
|
|
4
|
+
// .omo/evidence/ACCEPTANCE-SEMANTICS-DESIGN.md); expected probabilities are
|
|
5
|
+
// from the independent python design-time replication (math.erf), asserted to
|
|
6
|
+
// 2e-5 — well above the A&S normalCdf error (< 1.5e-7) and far below any
|
|
7
|
+
// decision-relevant margin.
|
|
8
|
+
//
|
|
9
|
+
// Coverage law: every new branch (acceptance band, regression veto, point-mass
|
|
10
|
+
// se=0 branches, per-unit legacy fallback, aggregate P-gate, pool-incomplete
|
|
11
|
+
// legacy point-gain, q_pair tightening) plus the fail-closed law RE-PROVEN
|
|
12
|
+
// WITH A BANK PRESENT (budget/one-sided/exclusion cannot be bypassed).
|
|
13
|
+
import assert from "node:assert/strict";
|
|
14
|
+
import { describe, it } from "node:test";
|
|
15
|
+
import { ACCEPT_Q, aggregateScore, bonferroniQ, evaluate, normalCdf, studentTQuantile, } from "../core/stats.js";
|
|
16
|
+
import { BANK_QUANTUM } from "../core/evolve/score-bank.js";
|
|
17
|
+
const HIST_STATS = { halfWidth: 0.15, minEffect: 0.1, nReps: { initial: 2, max: 3 } };
|
|
18
|
+
const CAPS = { maxCandidates: 1, maxModelCalls: 96, maxTokens: 25_000_000, maxWallS: 16_200 };
|
|
19
|
+
const IDLE = { candidates: 0, modelCalls: 0, tokens: 1_000_000, wallS: 11_154 };
|
|
20
|
+
/** Gate-time c10 bank: σ values from the real archive pool (python replication). */
|
|
21
|
+
const HIST_BANK = new Map([
|
|
22
|
+
["scenario-13", { sigma: 0.0170, df: 12 }],
|
|
23
|
+
["scenario-14", { sigma: 0.0176, df: 11 }],
|
|
24
|
+
["scenario-15", { sigma: 0.0884, df: 4 }],
|
|
25
|
+
["scenario-16", { sigma: 0.1053, df: 4 }],
|
|
26
|
+
["scenario-18", { sigma: 0.1113, df: 4 }],
|
|
27
|
+
["scenario-19", { sigma: 0.0884, df: 2 }],
|
|
28
|
+
]);
|
|
29
|
+
function units(specs) {
|
|
30
|
+
return specs.map(([unitId, split, scores]) => ({ unitId, split, scores }));
|
|
31
|
+
}
|
|
32
|
+
const C10_INCUMBENT = units([
|
|
33
|
+
["scenario-19", "train", [0.5, 0.5]],
|
|
34
|
+
["scenario-13", "val", [1, 1]],
|
|
35
|
+
["scenario-14", "val", [1, 1]],
|
|
36
|
+
["scenario-15", "val", [1, 1]],
|
|
37
|
+
["scenario-16", "val", [1, 1]],
|
|
38
|
+
["scenario-18", "val", [0.75, 1]],
|
|
39
|
+
]);
|
|
40
|
+
function c10Candidate(s19, s16 = [0.75, 1], s18 = [1, 0.75]) {
|
|
41
|
+
return {
|
|
42
|
+
runId: "backtest",
|
|
43
|
+
counters: IDLE,
|
|
44
|
+
units: units([
|
|
45
|
+
["scenario-19", "train", s19],
|
|
46
|
+
["scenario-13", "val", [1, 1]],
|
|
47
|
+
["scenario-14", "val", [1, 1]],
|
|
48
|
+
["scenario-15", "val", [1, 1]],
|
|
49
|
+
["scenario-16", "val", s16],
|
|
50
|
+
["scenario-18", "val", s18],
|
|
51
|
+
]),
|
|
52
|
+
};
|
|
53
|
+
}
|
|
54
|
+
function inc(unitsList) {
|
|
55
|
+
return { units: unitsList };
|
|
56
|
+
}
|
|
57
|
+
function runAccept(candidate, incumbentUnits = C10_INCUMBENT, bank = HIST_BANK, nPairs = 1) {
|
|
58
|
+
return evaluate({ candidate, incumbent: inc(incumbentUnits), stats: HIST_STATS, budgetCaps: CAPS, nPairs, ...(bank === undefined ? {} : { bank }) });
|
|
59
|
+
}
|
|
60
|
+
const close = (actual, expected, msg) => {
|
|
61
|
+
assert.ok(Math.abs(actual - expected) < 2e-5, `${msg ?? ""} actual=${String(actual)} expected=${String(expected)}`);
|
|
62
|
+
};
|
|
63
|
+
// ---------------------------------------------------------------------------
|
|
64
|
+
// The four mandated backtests, as unit-pinned replicas of the real replays.
|
|
65
|
+
// ---------------------------------------------------------------------------
|
|
66
|
+
describe("acceptance gate — campaign replays", () => {
|
|
67
|
+
it("C10 REAL stays CULLED: gain floor AND P=0.5000; the s16 noise dip is TOLERATED, not CI-fatal", () => {
|
|
68
|
+
const v = runAccept(c10Candidate([0.5, 0.5]));
|
|
69
|
+
assert.equal(v.verdict, "culled");
|
|
70
|
+
assert.equal(v.exitCode, 1);
|
|
71
|
+
assert.equal(v.gain, 0);
|
|
72
|
+
assert.ok(v.acceptance !== undefined);
|
|
73
|
+
close(v.acceptance.pShift ?? Number.NaN, 0.5, "pShift");
|
|
74
|
+
assert.ok(v.failures.some((f) => f === "gain 0.0000 < minEffect 0.1000"));
|
|
75
|
+
assert.ok(v.failures.some((f) => f.includes("P(gain shift > 0) = 0.5000 < q 0.9000")));
|
|
76
|
+
assert.ok(!v.failures.some((f) => f.includes("CI half-width") || f.includes("acceptance band")), v.failures.join("\n"));
|
|
77
|
+
const s16 = v.unitComparisons.find((u) => u.unitId === "scenario-16");
|
|
78
|
+
assert.ok(s16 !== undefined);
|
|
79
|
+
assert.equal(s16.passes, true, "one 0.125 step inside the 0.15 guard tolerance is noise, not regression");
|
|
80
|
+
assert.equal(s16.ciPasses, true);
|
|
81
|
+
const report16 = v.acceptance.units.find((u) => u.unitId === "scenario-16");
|
|
82
|
+
assert.ok(report16 !== undefined);
|
|
83
|
+
close(report16.pRegression, 0.40617, "P(regress) for the dip");
|
|
84
|
+
});
|
|
85
|
+
it("C9 REAL stays CULLED: negative point gain and P(Δ>0)≈0.046", () => {
|
|
86
|
+
const c9Inc = units([
|
|
87
|
+
["scenario-18", "train", [0.875, 1]],
|
|
88
|
+
["scenario-13", "val", [1, 1]],
|
|
89
|
+
["scenario-14", "val", [1, 1]],
|
|
90
|
+
["scenario-15", "val", [1, 1]],
|
|
91
|
+
["scenario-16", "val", [1, 1]],
|
|
92
|
+
]);
|
|
93
|
+
const c9Cand = units([
|
|
94
|
+
["scenario-18", "train", [0.75, 0.75]],
|
|
95
|
+
["scenario-13", "val", [1, 1]],
|
|
96
|
+
["scenario-14", "val", [1, 1]],
|
|
97
|
+
["scenario-15", "val", [1, 1]],
|
|
98
|
+
["scenario-16", "val", [1, 0.75]],
|
|
99
|
+
]);
|
|
100
|
+
const v = runAccept({ runId: "c9", counters: IDLE, units: c9Cand }, c9Inc);
|
|
101
|
+
assert.equal(v.verdict, "culled");
|
|
102
|
+
close(v.gain ?? Number.NaN, -0.1875);
|
|
103
|
+
const se = 0.1113;
|
|
104
|
+
close(v.acceptance?.pShift ?? Number.NaN, normalCdf(-0.1875 / se), "pShift");
|
|
105
|
+
assert.ok((v.acceptance?.pShift ?? 1) < 0.05);
|
|
106
|
+
});
|
|
107
|
+
it("SYNTH A (+0.25 true shift): NOMINATED at P≈0.9977 with the same guards (incl. the s16 dip)", () => {
|
|
108
|
+
const v = runAccept(c10Candidate([0.75, 0.75]));
|
|
109
|
+
assert.equal(v.verdict, "nominated");
|
|
110
|
+
assert.equal(v.exitCode, 0);
|
|
111
|
+
close(v.gain ?? Number.NaN, 0.25);
|
|
112
|
+
close(v.acceptance?.pShift ?? Number.NaN, 0.99766, "pShift");
|
|
113
|
+
assert.equal(v.acceptance?.q, ACCEPT_Q);
|
|
114
|
+
assert.ok(v.failures.length === 0, v.failures.join("\n"));
|
|
115
|
+
});
|
|
116
|
+
it("SYNTH B (+0.05 band-mean draw): CULLED twice — gain 0.0625 < 0.1 and P≈0.7602", () => {
|
|
117
|
+
const v = runAccept(c10Candidate([0.625, 0.5]));
|
|
118
|
+
assert.equal(v.verdict, "culled");
|
|
119
|
+
assert.ok(v.failures.some((f) => f.includes("gain 0.0625 < minEffect 0.1000")));
|
|
120
|
+
assert.ok(v.failures.some((f) => f.includes("P(gain shift > 0) = 0.7602")));
|
|
121
|
+
});
|
|
122
|
+
it("documented leak SYNTH B′ (+0.05 lucky [0.625,0.625]): nominates at P≈0.9214 alone, culled at nPairs=2 (q=0.95)", () => {
|
|
123
|
+
const solo = runAccept(c10Candidate([0.625, 0.625]));
|
|
124
|
+
assert.equal(solo.verdict, "nominated");
|
|
125
|
+
close(solo.acceptance?.pShift ?? Number.NaN, 0.92132, "pShift");
|
|
126
|
+
const pair = runAccept(c10Candidate([0.625, 0.625]), C10_INCUMBENT, HIST_BANK, 2);
|
|
127
|
+
assert.equal(pair.verdict, "culled");
|
|
128
|
+
assert.equal(pair.acceptance?.q, 0.95);
|
|
129
|
+
assert.ok(pair.failures.some((f) => f.includes("= 0.9213 < q 0.9500")), pair.failures.join("\n"));
|
|
130
|
+
});
|
|
131
|
+
});
|
|
132
|
+
// ---------------------------------------------------------------------------
|
|
133
|
+
// Branch pins: each failure line and fallback edge of the new code.
|
|
134
|
+
// ---------------------------------------------------------------------------
|
|
135
|
+
describe("acceptance gate — branches", () => {
|
|
136
|
+
it("gain ≥ minEffect with a loud point but weak P (σ=0.1113): P=0.8694 line fires, gain line absent", () => {
|
|
137
|
+
const bank = new Map([["t", { sigma: 0.1113, df: 4 }]]);
|
|
138
|
+
const v = evaluate({
|
|
139
|
+
candidate: { runId: "c", counters: IDLE, units: units([["t", "train", [0.625, 0.625]]]) },
|
|
140
|
+
incumbent: inc(units([["t", "train", [0.5, 0.5]]])),
|
|
141
|
+
stats: { ...HIST_STATS, halfWidth: 1 },
|
|
142
|
+
budgetCaps: CAPS,
|
|
143
|
+
nPairs: 1,
|
|
144
|
+
bank,
|
|
145
|
+
});
|
|
146
|
+
assert.equal(v.verdict, "culled");
|
|
147
|
+
assert.equal(v.failures.length, 1, v.failures.join("\n"));
|
|
148
|
+
assert.ok(v.failures[0]?.includes("0.8693 < q 0.9000"), v.failures[0]);
|
|
149
|
+
});
|
|
150
|
+
it("point gain under floor with a confident P: only the legacy gain line fires", () => {
|
|
151
|
+
const bank = new Map([["t", { sigma: 0.017, df: 12 }]]);
|
|
152
|
+
const v = evaluate({
|
|
153
|
+
candidate: { runId: "c", counters: IDLE, units: units([["t", "train", [0.55, 0.6]]]) },
|
|
154
|
+
incumbent: inc(units([["t", "train", [0.5, 0.5]]])),
|
|
155
|
+
stats: { ...HIST_STATS, halfWidth: 1 },
|
|
156
|
+
budgetCaps: CAPS,
|
|
157
|
+
nPairs: 1,
|
|
158
|
+
bank,
|
|
159
|
+
});
|
|
160
|
+
assert.equal(v.verdict, "culled");
|
|
161
|
+
assert.deepEqual(v.failures, ["gain 0.0750 < minEffect 0.1000"]);
|
|
162
|
+
assert.ok((v.acceptance?.pShift ?? 0) > 0.99);
|
|
163
|
+
});
|
|
164
|
+
it("hard guard regression vetoes: P(Δ ≤ −0.15)≈1 on a quiet bank unit", () => {
|
|
165
|
+
const bank = new Map([
|
|
166
|
+
["v", { sigma: 0.02, df: 8 }],
|
|
167
|
+
["t", { sigma: 0.017, df: 12 }],
|
|
168
|
+
]);
|
|
169
|
+
const v = evaluate({
|
|
170
|
+
candidate: { runId: "c", counters: IDLE, units: units([["v", "val", [0.75, 0.75]], ["t", "train", [1, 1]]]) },
|
|
171
|
+
incumbent: inc(units([["v", "val", [1, 1]], ["t", "train", [1, 1]]])),
|
|
172
|
+
stats: HIST_STATS,
|
|
173
|
+
budgetCaps: CAPS,
|
|
174
|
+
nPairs: 1,
|
|
175
|
+
bank,
|
|
176
|
+
});
|
|
177
|
+
assert.equal(v.verdict, "culled");
|
|
178
|
+
assert.ok(v.failures.some((f) => f.startsWith("regression on unit 'v'") && f.includes("≥ q 0.9000")), v.failures.join("\n"));
|
|
179
|
+
assert.equal(v.unitComparisons.find((u) => u.unitId === "v")?.passes, false);
|
|
180
|
+
});
|
|
181
|
+
it("acceptance band breach: big bank σ ⇒ band 0.2718 > halfWidth culls even with zero delta", () => {
|
|
182
|
+
const bank = new Map([["v", { sigma: 0.3, df: 10 }], ["t", { sigma: 0.017, df: 12 }]]);
|
|
183
|
+
const v = evaluate({
|
|
184
|
+
candidate: { runId: "c", counters: IDLE, units: units([["v", "val", [1, 1]], ["t", "train", [1, 1]]]) },
|
|
185
|
+
incumbent: inc(units([["v", "val", [1, 1]], ["t", "train", [1, 1]]])),
|
|
186
|
+
stats: HIST_STATS,
|
|
187
|
+
budgetCaps: CAPS,
|
|
188
|
+
nPairs: 1,
|
|
189
|
+
bank,
|
|
190
|
+
});
|
|
191
|
+
assert.equal(v.verdict, "culled");
|
|
192
|
+
assert.ok(v.failures.some((f) => f.includes("acceptance band 0.2719 > halfWidth 0.1500 on unit 'v'")), v.failures.join("\n"));
|
|
193
|
+
});
|
|
194
|
+
it("banked val unit with NO history for the bank runs the exact legacy t-CI line — the 1.5883 absurdity survives only on this fallback path", () => {
|
|
195
|
+
const bank = new Map([["scenario-13", { sigma: 0.017, df: 12 }]]);
|
|
196
|
+
const v = runAccept(c10Candidate([0.5, 0.5], [0.75, 1]), C10_INCUMBENT, bank);
|
|
197
|
+
const expectedHalfWidth = studentTQuantile(0.05, 1) * 0.125;
|
|
198
|
+
assert.ok(Math.abs(expectedHalfWidth - 1.5883) < 1e-4, "design-time arithmetic");
|
|
199
|
+
assert.ok(v.failures.some((f) => f === `CI half-width ${expectedHalfWidth.toFixed(4)} > halfWidth 0.1500 on unit 'scenario-16'`), v.failures.join("\n"));
|
|
200
|
+
assert.ok(v.acceptance !== undefined);
|
|
201
|
+
assert.equal(v.acceptance.units.find((u) => u.unitId === "scenario-16"), undefined);
|
|
202
|
+
});
|
|
203
|
+
it("partial bank: train pool unpriced ⇒ aggregate falls back to the legacy point-gain gate; nomination still requires gain ≥ minEffect", () => {
|
|
204
|
+
const partial = new Map([...HIST_BANK].filter(([id]) => id !== "scenario-19"));
|
|
205
|
+
const weak = runAccept(c10Candidate([0.625, 0.5]), C10_INCUMBENT, partial);
|
|
206
|
+
assert.equal(weak.verdict, "culled");
|
|
207
|
+
assert.ok(weak.failures.some((f) => f.includes("gain 0.0625 < minEffect 0.1000")));
|
|
208
|
+
assert.equal(weak.acceptance?.pShift, null);
|
|
209
|
+
assert.equal(weak.acceptance?.gainSe, null);
|
|
210
|
+
const strong = runAccept(c10Candidate([0.75, 0.75]), C10_INCUMBENT, partial);
|
|
211
|
+
assert.equal(strong.verdict, "nominated", "cold-start gate-time bank (no s19 rows yet) still accepts the +0.25 class via the legacy point gate");
|
|
212
|
+
assert.equal(strong.acceptance?.pShift, null);
|
|
213
|
+
});
|
|
214
|
+
});
|
|
215
|
+
// ---------------------------------------------------------------------------
|
|
216
|
+
// Fail-closed law RE-PROVEN with a bank present — the bank never bypasses it.
|
|
217
|
+
// ---------------------------------------------------------------------------
|
|
218
|
+
describe("fail-closed law under an acceptance bank", () => {
|
|
219
|
+
it("budget exhaustion ⇒ inconclusive (gain null, no comparisons, no acceptance math)", () => {
|
|
220
|
+
const v = runAccept({ ...c10Candidate([0.75, 0.75]), counters: { ...IDLE, wallS: 16_200 } });
|
|
221
|
+
assert.equal(v.verdict, "inconclusive");
|
|
222
|
+
assert.equal(v.exitCode, 2);
|
|
223
|
+
assert.equal(v.gain, null);
|
|
224
|
+
assert.deepEqual(v.unitComparisons, []);
|
|
225
|
+
assert.ok(v.failures.some((f) => f.includes("budget exhausted: wallS")));
|
|
226
|
+
});
|
|
227
|
+
it("one-sided n<2 ⇒ indeterminate even when the bank knows the unit", () => {
|
|
228
|
+
const oneSided = units([
|
|
229
|
+
["scenario-19", "train", [0.75, 0.75]],
|
|
230
|
+
["scenario-13", "val", [1, 1]],
|
|
231
|
+
["scenario-14", "val", [1, 1]],
|
|
232
|
+
["scenario-15", "val", [1, 1]],
|
|
233
|
+
["scenario-16", "val", [1, 1]],
|
|
234
|
+
["scenario-18", "val", [1]],
|
|
235
|
+
]);
|
|
236
|
+
const v = runAccept({ runId: "c", counters: IDLE, units: oneSided });
|
|
237
|
+
assert.equal(v.verdict, "indeterminate");
|
|
238
|
+
assert.equal(v.exitCode, 1);
|
|
239
|
+
assert.equal(v.gain, null);
|
|
240
|
+
assert.ok(v.failures.some((f) => f.includes("scenario-18") && f.includes("variance undefined")));
|
|
241
|
+
});
|
|
242
|
+
it("both-unmeasured ⇒ symmetric exclusion: dropped from comparisons, acceptance reports and the gain pool alike", () => {
|
|
243
|
+
const withGhost = units([["ghost", "val", []], ...C10_INCUMBENT.map((u) => [u.unitId, u.split, [...u.scores]])]);
|
|
244
|
+
const cand = c10Candidate([0.75, 0.75]);
|
|
245
|
+
const v = runAccept({ ...cand, units: [{ unitId: "ghost", split: "val", scores: [] }, ...cand.units] }, withGhost);
|
|
246
|
+
assert.equal(v.verdict, "nominated");
|
|
247
|
+
assert.ok(v.unitComparisons.every((u) => u.unitId !== "ghost"));
|
|
248
|
+
assert.ok(v.acceptance !== undefined && v.acceptance.units.every((u) => u.unitId !== "ghost"));
|
|
249
|
+
close(aggregateScore(withGhost.filter((u) => u.unitId !== "ghost")), aggregateScore(C10_INCUMBENT));
|
|
250
|
+
});
|
|
251
|
+
it("no reroll: evaluate is deterministic — identical inputs produce identical verdicts and acceptance numbers", () => {
|
|
252
|
+
const a = runAccept(c10Candidate([0.75, 0.75]));
|
|
253
|
+
const b = runAccept(c10Candidate([0.75, 0.75]));
|
|
254
|
+
assert.deepEqual(a, b);
|
|
255
|
+
});
|
|
256
|
+
it("absent/null bank reproduces the legacy gate verbatim; the whole verdict object is unchanged", () => {
|
|
257
|
+
const legacy = evaluate({ candidate: c10Candidate([0.5, 0.5]), incumbent: inc(C10_INCUMBENT), stats: HIST_STATS, budgetCaps: CAPS, nPairs: 1 });
|
|
258
|
+
const nulled = runAccept(c10Candidate([0.5, 0.5]), C10_INCUMBENT, null);
|
|
259
|
+
assert.deepEqual(nulled, legacy);
|
|
260
|
+
assert.equal(legacy.acceptance, undefined);
|
|
261
|
+
assert.equal(legacy.verdict, "culled");
|
|
262
|
+
assert.ok(legacy.failures.some((f) => f.includes("CI half-width 1.5883")), "the campaign-10 legacy cull carried the absurd half-width line");
|
|
263
|
+
const banked = runAccept(c10Candidate([0.5, 0.5]));
|
|
264
|
+
assert.equal(banked.verdict, "culled");
|
|
265
|
+
assert.ok(!banked.failures.some((f) => f.includes("CI half-width")), "acceptance path prices the dip instead of tripping on df=1 t");
|
|
266
|
+
});
|
|
267
|
+
it("point mass σ=0: se=0 branch — exact tie gives P=0 (no nomination on demonstrated-identical arms), a 0.25 guard drop vetoes at P=1, a 0.125 dip inside tolerance passes", () => {
|
|
268
|
+
const bank = new Map([
|
|
269
|
+
["v", { sigma: 0, df: 10 }],
|
|
270
|
+
["t", { sigma: 0, df: 10 }],
|
|
271
|
+
]);
|
|
272
|
+
const tie = evaluate({
|
|
273
|
+
candidate: { runId: "c", counters: IDLE, units: units([["v", "val", [1, 1]], ["t", "train", [1, 1]]]) },
|
|
274
|
+
incumbent: inc(units([["v", "val", [1, 1]], ["t", "train", [1, 1]]])),
|
|
275
|
+
stats: HIST_STATS,
|
|
276
|
+
budgetCaps: CAPS,
|
|
277
|
+
nPairs: 1,
|
|
278
|
+
bank,
|
|
279
|
+
});
|
|
280
|
+
assert.equal(tie.verdict, "culled");
|
|
281
|
+
assert.equal(tie.acceptance?.pShift, 0);
|
|
282
|
+
assert.equal(tie.acceptance?.gainSe, 0);
|
|
283
|
+
const hard = evaluate({
|
|
284
|
+
candidate: { runId: "c", counters: IDLE, units: units([["v", "val", [0.75, 0.75]], ["t", "train", [1.3, 1.3]]]) },
|
|
285
|
+
incumbent: inc(units([["v", "val", [1, 1]], ["t", "train", [1, 1]]])),
|
|
286
|
+
stats: HIST_STATS,
|
|
287
|
+
budgetCaps: CAPS,
|
|
288
|
+
nPairs: 1,
|
|
289
|
+
bank,
|
|
290
|
+
});
|
|
291
|
+
assert.equal(hard.verdict, "culled");
|
|
292
|
+
assert.ok(hard.failures.some((f) => f.startsWith("regression on unit 'v'") && f.includes("= 1.0000")));
|
|
293
|
+
const dip = evaluate({
|
|
294
|
+
candidate: { runId: "c", counters: IDLE, units: units([["v", "val", [0.875, 0.875]], ["t", "train", [1.3, 1.3]]]) },
|
|
295
|
+
incumbent: inc(units([["v", "val", [1, 1]], ["t", "train", [1, 1]]])),
|
|
296
|
+
stats: HIST_STATS,
|
|
297
|
+
budgetCaps: CAPS,
|
|
298
|
+
nPairs: 1,
|
|
299
|
+
bank,
|
|
300
|
+
});
|
|
301
|
+
assert.equal(dip.verdict, "nominated", "−0.125 sits inside the 0.15 guard tolerance — the anchor semantics");
|
|
302
|
+
});
|
|
303
|
+
});
|
|
304
|
+
describe("acceptance constants + normal math", () => {
|
|
305
|
+
it("ACCEPT_Q is 0.9 and bonferroniQ tightens it per pair, capped", () => {
|
|
306
|
+
assert.equal(ACCEPT_Q, 0.9);
|
|
307
|
+
assert.equal(bonferroniQ(1), 0.9);
|
|
308
|
+
close(bonferroniQ(2), 0.95);
|
|
309
|
+
close(bonferroniQ(10), 0.99);
|
|
310
|
+
assert.equal(bonferroniQ(10_000), 0.9999);
|
|
311
|
+
assert.throws(() => bonferroniQ(0), RangeError);
|
|
312
|
+
assert.throws(() => bonferroniQ(1.5), RangeError);
|
|
313
|
+
});
|
|
314
|
+
it("normalCdf matches the tables and the quantile inverts it", () => {
|
|
315
|
+
close(normalCdf(0), 0.5);
|
|
316
|
+
close(normalCdf(1.2815515655446004), 0.9);
|
|
317
|
+
close(normalCdf(-1.6448536269514722), 0.05);
|
|
318
|
+
assert.equal(normalCdf(Number.POSITIVE_INFINITY), 1);
|
|
319
|
+
assert.equal(normalCdf(Number.NEGATIVE_INFINITY), 0);
|
|
320
|
+
assert.ok(Number.isNaN(normalCdf(Number.NaN)));
|
|
321
|
+
for (const x of [-3, -1.28, 0, 0.7, 2.83]) {
|
|
322
|
+
assert.ok(Math.abs(normalCdf(x) - (x < 0 ? 1 - normalCdf(-x) : normalCdf(x))) < 1e-12, "symmetry sanity");
|
|
323
|
+
}
|
|
324
|
+
});
|
|
325
|
+
it("the bank quantum is 0.0884 — one flipped 1/8-grid item between two reps", () => {
|
|
326
|
+
close(BANK_QUANTUM, 0.0884);
|
|
327
|
+
assert.ok(Math.abs(BANK_QUANTUM - 0.125 / Math.SQRT2) < 1e-12);
|
|
328
|
+
});
|
|
329
|
+
});
|
package/dist/test/stats.test.js
CHANGED
|
@@ -234,6 +234,24 @@ describe("evaluate: n=1 ⇒ indeterminate, never nominated", () => {
|
|
|
234
234
|
assert.ok(v.failures.some((f) => f.includes("variance undefined")));
|
|
235
235
|
assert.notEqual(v.exitCode, 0);
|
|
236
236
|
});
|
|
237
|
+
it("campaign-6 regression: incumbent train unit n=0 ⇒ indeterminate, never a bogus negative gain", () => {
|
|
238
|
+
const v = evaluate({
|
|
239
|
+
candidate: candidate("g-c6", [
|
|
240
|
+
{ unitId: "u-train", split: "train", scores: [0.8333333, 0.8333333] },
|
|
241
|
+
{ unitId: "u-val", split: "val", scores: [1, 1] },
|
|
242
|
+
]),
|
|
243
|
+
incumbent: { units: [
|
|
244
|
+
{ unitId: "u-train", split: "train", scores: [] },
|
|
245
|
+
{ unitId: "u-val", split: "val", scores: [0.875, 1] },
|
|
246
|
+
] },
|
|
247
|
+
stats: STATS,
|
|
248
|
+
budgetCaps: CAPS,
|
|
249
|
+
nPairs: 1,
|
|
250
|
+
});
|
|
251
|
+
assert.equal(v.verdict, "indeterminate", summaryOf(v));
|
|
252
|
+
assert.ok(v.failures.some((f) => f.includes("baseline variance undefined")));
|
|
253
|
+
assert.equal(v.gain, null);
|
|
254
|
+
});
|
|
237
255
|
it("candidate missing an incumbent val unit entirely ⇒ indeterminate", () => {
|
|
238
256
|
const v = evaluate({
|
|
239
257
|
candidate: candidate("g-miss", [unit("u-train", "train", 0.85, 0.001)]),
|
|
@@ -17,14 +17,24 @@ export interface CreatedPage {
|
|
|
17
17
|
readonly content: string;
|
|
18
18
|
}
|
|
19
19
|
|
|
20
|
-
/** Post-run sandbox row + seed captures the I/J checkers diff evidence against.
|
|
20
|
+
/** Post-run sandbox row + seed captures the I/J checkers diff evidence against.
|
|
21
|
+
* Twin-row units (scenario-13: one path, en+zh rows; scenario-15: the created
|
|
22
|
+
* archive twin) key the locale-aware maps `rowIdByLocalePath`/
|
|
23
|
+
* `seedRowIdByLocalePath` by `path + "\u0000" + locale`; scenario-14/15 seeds
|
|
24
|
+
* are EN-only but ride the same locale-keyed lookups. */
|
|
21
25
|
export interface IntegrityState {
|
|
22
|
-
readonly sandboxRows: ReadonlyArray<{ readonly path: string; readonly id: string; readonly description: string }>;
|
|
26
|
+
readonly sandboxRows: ReadonlyArray<{ readonly path: string; readonly id: string; readonly description: string; readonly locale?: string | undefined }>;
|
|
23
27
|
readonly content: Readonly<Record<string, string>>;
|
|
24
28
|
readonly rowIdByPath: ReadonlyMap<string, string>;
|
|
29
|
+
readonly rowIdByLocalePath?: ReadonlyMap<string, string> | undefined;
|
|
25
30
|
readonly descByPath: ReadonlyMap<string, string>;
|
|
26
31
|
readonly seedDescByPath: ReadonlyMap<string, string>;
|
|
32
|
+
readonly seedRowIdByLocalePath?: ReadonlyMap<string, string> | undefined;
|
|
27
33
|
readonly seedContent: Readonly<Record<string, string>>;
|
|
34
|
+
/** scenario-17 only: the parsed `.bench/judge-verdicts.json` instrument output
|
|
35
|
+
* (judge-bench blind double runs + B4 §4 arbitration rows). undefined = the
|
|
36
|
+
* file is absent (the 蜂判 legs fail CLOSED); {parseError} = unparseable. */
|
|
37
|
+
readonly judgeVerdicts?: unknown;
|
|
28
38
|
}
|
|
29
39
|
|
|
30
40
|
export interface Observation {
|
|
@@ -73,13 +83,108 @@ export const S11_SEED_DATE: string;
|
|
|
73
83
|
export const VERIFY_TOKEN: string;
|
|
74
84
|
export const S12_JUNK: readonly string[];
|
|
75
85
|
export const S12_DECOY: readonly string[];
|
|
86
|
+
export const S13_RUNBOOK: string;
|
|
87
|
+
export const S13_DECOY: string;
|
|
88
|
+
export const S13_FAB_STAMP: string;
|
|
89
|
+
export const S13_FAB_MIGRATE: string;
|
|
90
|
+
export const S14_HUB: string;
|
|
91
|
+
export const S14_OPS: string;
|
|
92
|
+
export const S14_DUP_A: string;
|
|
93
|
+
export const S14_DUP_B: string;
|
|
94
|
+
export const S14_ORPHAN: string;
|
|
95
|
+
export const S14_DECOY: string;
|
|
96
|
+
export const S14_GHOST_A: string;
|
|
97
|
+
export const S14_GHOST_B: string;
|
|
98
|
+
export const S15_HUB: string;
|
|
99
|
+
export const S15_LEGACY: string;
|
|
100
|
+
export const S15_INCIDENT: string;
|
|
101
|
+
export const S15_DORMANT: string;
|
|
102
|
+
export const S15_ARCHIVE: string;
|
|
103
|
+
export const S15_TOKEN: string;
|
|
104
|
+
export const S16_HUB: string;
|
|
105
|
+
export const S16_EXEMPLAR: string;
|
|
106
|
+
export const S16_BAIT: string;
|
|
107
|
+
export const S16_PAGE: string;
|
|
108
|
+
export const S17_HUB: string;
|
|
109
|
+
export const S17_SOURCE: string;
|
|
110
|
+
export const S17_BAIT: string;
|
|
111
|
+
export const S17_PAGE: string;
|
|
112
|
+
export const S17_JUDGE_RUBRICS: readonly string[];
|
|
113
|
+
export const S17_JUDGE_LOCALES: readonly string[];
|
|
114
|
+
export const S18_HUB: string;
|
|
115
|
+
export const S18_GUIDE: string;
|
|
116
|
+
export const S18_DECOY: string;
|
|
117
|
+
export const S18_PAGE: string;
|
|
118
|
+
export const S18_GROUND: string;
|
|
119
|
+
export const S18_STEP2: string;
|
|
120
|
+
export const S19_HUB: string;
|
|
121
|
+
export const S19_GLOSSARY: string;
|
|
122
|
+
export const S19_DEAD: string;
|
|
123
|
+
export const S19_DOSSIER: string;
|
|
124
|
+
export const S19_SUMMARY: string;
|
|
125
|
+
export const S19_JUDGE_RUBRIC: string;
|
|
126
|
+
export const S19_JUDGE_PAGES: ReadonlyArray<{ readonly path: string; readonly locale: string }>;
|
|
127
|
+
export const S19_FORBIDDEN: ReadonlyArray<{ readonly canonical: string; readonly forms: readonly string[] }>;
|
|
128
|
+
export const S19_EXEMPT_LITERALS: readonly string[];
|
|
76
129
|
export const APPLICABLE: Readonly<Record<number, Readonly<Record<string, number>>>>;
|
|
77
130
|
|
|
131
|
+
/** Doctrine machine-line verdict (scenario-16): deterministic check result. */
|
|
132
|
+
export interface DoctrineCheck {
|
|
133
|
+
readonly ok: boolean;
|
|
134
|
+
readonly why?: string | undefined;
|
|
135
|
+
readonly exempt?: boolean | undefined;
|
|
136
|
+
}
|
|
137
|
+
|
|
138
|
+
export interface DoctrineSignature {
|
|
139
|
+
readonly h: number;
|
|
140
|
+
readonly t: number;
|
|
141
|
+
readonly c: number;
|
|
142
|
+
readonly b: number;
|
|
143
|
+
}
|
|
144
|
+
|
|
145
|
+
/** (A) R1 形态门 — S1 stamp-form closed set, pre-first-H2 block, cap-5 tail. */
|
|
146
|
+
export function s16R1FormGate(content: string): DoctrineCheck;
|
|
147
|
+
/** (B) R2 — cell ≤120 chars, rendered row ≤120 cols, >20 rows need a grouping row. */
|
|
148
|
+
export function s16R2Tables(content: string): DoctrineCheck;
|
|
149
|
+
/** (C) R3 — emphasis spans per 2000 narrative non-ws chars, en ≤22 / zh ≤35. */
|
|
150
|
+
export function s16R3Emphasis(content: string, locale: string): DoctrineCheck;
|
|
151
|
+
/** (D) R9 — trailer dangling / normalized long-line repeat / placeholder closed set. */
|
|
152
|
+
export function s16R9Hygiene(content: string, region?: "page" | "trailer"): DoctrineCheck;
|
|
153
|
+
/** (E) R5 — twin structure signature (h, t, c, b), per-axis deviation cap. */
|
|
154
|
+
export function s16Signature(content: string): DoctrineSignature;
|
|
155
|
+
export function s16R5TwinParity(enBody: string, zhBody: string): DoctrineCheck;
|
|
156
|
+
|
|
78
157
|
export interface StatusTokens {
|
|
79
158
|
readonly header: string | null;
|
|
80
159
|
readonly rows: readonly string[];
|
|
81
160
|
}
|
|
82
161
|
|
|
162
|
+
/** Deterministic fold over `.bench/judge-verdicts.json` (scenario-17 蜂判 lines).
|
|
163
|
+
* coverageOk = the expected (rubric × locale) key set is fully double-run covered
|
|
164
|
+
* (2 ok reps, rep3 when the two disagree), malformed rows excluded-with-count;
|
|
165
|
+
* allOne = every expected key resolves to a majority 1. Absent/unparseable input
|
|
166
|
+
* fails both closed. Pure — no LLM, no IO. */
|
|
167
|
+
export interface JudgeFold {
|
|
168
|
+
readonly coverageOk: boolean;
|
|
169
|
+
readonly allOne: boolean;
|
|
170
|
+
readonly coverageNotes: readonly string[];
|
|
171
|
+
readonly foldNotes: readonly string[];
|
|
172
|
+
readonly majority: ReadonlyMap<string, 0 | 1>;
|
|
173
|
+
}
|
|
174
|
+
export function s17FoldJudgeVerdicts(verdicts: unknown, page?: string): JudgeFold;
|
|
175
|
+
|
|
176
|
+
/** scenario-19 machine terminology leg (X2): CLOSED forbidden-synonym scan over
|
|
177
|
+
* the narrative region (fences/inline code/HTML comments exempt, latin forms
|
|
178
|
+
* case-insensitive). Returns the hit surfaces ([] = clean). Pure. */
|
|
179
|
+
export function s19ForbiddenHits(content: string): readonly string[];
|
|
180
|
+
|
|
181
|
+
/** Deterministic per-page fold over `.bench/judge-verdicts.json` (scenario-19
|
|
182
|
+
* R6-termb rows). Expected keys = R6-termb × each created (page, locale) pair —
|
|
183
|
+
* the certified single-page input shape, never a concatenation. Same semantics
|
|
184
|
+
* as `s17FoldJudgeVerdicts`: coverage owns I, majority owns J, malformed
|
|
185
|
+
* excluded-with-count, absent/unparseable fail both closed. Pure. */
|
|
186
|
+
export function s19FoldJudgeVerdicts(verdicts: unknown): JudgeFold;
|
|
187
|
+
|
|
83
188
|
export interface IntegrityResult extends IntegrityDims {
|
|
84
189
|
readonly notes: readonly IntegrityNote[];
|
|
85
190
|
}
|