@tachikomagundam/abathur 0.2.5 → 0.2.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +3 -3
- package/dist/cli.js +4 -0
- package/dist/commands/genome.js +2 -2
- package/dist/commands/graft.js +1 -1
- package/dist/commands/readjudicate.js +28 -0
- package/dist/commands/retract.js +53 -0
- package/dist/commands/run.js +1 -1
- package/dist/commands/self-eval.js +1 -1
- package/dist/commands/tombstone.js +30 -16
- package/dist/core/evolve/run-loop.js +14 -0
- package/dist/core/evolve/run-rows.js +14 -0
- package/dist/core/evolve/score-bank.js +214 -0
- package/dist/core/graft-rebench.js +3 -0
- package/dist/core/ledger.js +8 -1
- package/dist/core/promote.js +63 -47
- package/dist/core/readjudicate.js +247 -0
- package/dist/core/retract.js +166 -0
- package/dist/core/stats-math.js +39 -0
- package/dist/core/stats.js +93 -19
- package/dist/core/worktree.js +1 -1
- package/dist/genomes/toy-smoke/init.mjs +1 -1
- package/dist/test/bundle.test.js +1 -1
- package/dist/test/fixture-loop.test.js +1 -1
- package/dist/test/fixtures-self.js +1 -1
- package/dist/test/fixtures-wt.js +1 -1
- package/dist/test/friction.test.js +3 -3
- package/dist/test/graft.test.js +1 -1
- package/dist/test/historian-grader-integrity.test.js +509 -2
- package/dist/test/promote.test.js +222 -5
- package/dist/test/readjudicate.test.js +212 -0
- package/dist/test/reflect.test.js +1 -1
- package/dist/test/repopath-seams.test.js +55 -0
- package/dist/test/score-bank.test.js +235 -0
- package/dist/test/seam-gates.test.js +84 -0
- package/dist/test/self-snapshot.test.js +2 -2
- package/dist/test/stats-acceptance.test.js +329 -0
- package/graders/historian/grader-core.d.mts +27 -0
- package/graders/historian/grader-core.mjs +311 -19
- package/graders/historian/grader.mjs +5 -5
- package/graders/historian/judge-poststage-19.mjs +212 -0
- package/graders/historian/run-scenario-19.sh +43 -0
- package/graders/historian/run-scenario.sh +16 -0
- package/package.json +1 -1
- package/plugin/abathur-command.md +3 -3
- package/plugin/abathur.ts +6 -6
|
@@ -0,0 +1,329 @@
|
|
|
1
|
+
// Acceptance-semantics evaluate() tests (2026-09-24 redesign). Thresholds
|
|
2
|
+
// mirror the LIVE historian genome: halfWidth 0.15, minEffect 0.1. Bank σ
|
|
3
|
+
// values are the real campaign-10 gate-time bank (§2.3 of
|
|
4
|
+
// .omo/evidence/ACCEPTANCE-SEMANTICS-DESIGN.md); expected probabilities are
|
|
5
|
+
// from the independent python design-time replication (math.erf), asserted to
|
|
6
|
+
// 2e-5 — well above the A&S normalCdf error (< 1.5e-7) and far below any
|
|
7
|
+
// decision-relevant margin.
|
|
8
|
+
//
|
|
9
|
+
// Coverage law: every new branch (acceptance band, regression veto, point-mass
|
|
10
|
+
// se=0 branches, per-unit legacy fallback, aggregate P-gate, pool-incomplete
|
|
11
|
+
// legacy point-gain, q_pair tightening) plus the fail-closed law RE-PROVEN
|
|
12
|
+
// WITH A BANK PRESENT (budget/one-sided/exclusion cannot be bypassed).
|
|
13
|
+
import assert from "node:assert/strict";
|
|
14
|
+
import { describe, it } from "node:test";
|
|
15
|
+
import { ACCEPT_Q, aggregateScore, bonferroniQ, evaluate, normalCdf, studentTQuantile, } from "../core/stats.js";
|
|
16
|
+
import { BANK_QUANTUM } from "../core/evolve/score-bank.js";
|
|
17
|
+
const HIST_STATS = { halfWidth: 0.15, minEffect: 0.1, nReps: { initial: 2, max: 3 } };
|
|
18
|
+
const CAPS = { maxCandidates: 1, maxModelCalls: 96, maxTokens: 25_000_000, maxWallS: 16_200 };
|
|
19
|
+
const IDLE = { candidates: 0, modelCalls: 0, tokens: 1_000_000, wallS: 11_154 };
|
|
20
|
+
/** Gate-time c10 bank: σ values from the real archive pool (python replication). */
|
|
21
|
+
const HIST_BANK = new Map([
|
|
22
|
+
["scenario-13", { sigma: 0.0170, df: 12 }],
|
|
23
|
+
["scenario-14", { sigma: 0.0176, df: 11 }],
|
|
24
|
+
["scenario-15", { sigma: 0.0884, df: 4 }],
|
|
25
|
+
["scenario-16", { sigma: 0.1053, df: 4 }],
|
|
26
|
+
["scenario-18", { sigma: 0.1113, df: 4 }],
|
|
27
|
+
["scenario-19", { sigma: 0.0884, df: 2 }],
|
|
28
|
+
]);
|
|
29
|
+
function units(specs) {
|
|
30
|
+
return specs.map(([unitId, split, scores]) => ({ unitId, split, scores }));
|
|
31
|
+
}
|
|
32
|
+
const C10_INCUMBENT = units([
|
|
33
|
+
["scenario-19", "train", [0.5, 0.5]],
|
|
34
|
+
["scenario-13", "val", [1, 1]],
|
|
35
|
+
["scenario-14", "val", [1, 1]],
|
|
36
|
+
["scenario-15", "val", [1, 1]],
|
|
37
|
+
["scenario-16", "val", [1, 1]],
|
|
38
|
+
["scenario-18", "val", [0.75, 1]],
|
|
39
|
+
]);
|
|
40
|
+
function c10Candidate(s19, s16 = [0.75, 1], s18 = [1, 0.75]) {
|
|
41
|
+
return {
|
|
42
|
+
runId: "backtest",
|
|
43
|
+
counters: IDLE,
|
|
44
|
+
units: units([
|
|
45
|
+
["scenario-19", "train", s19],
|
|
46
|
+
["scenario-13", "val", [1, 1]],
|
|
47
|
+
["scenario-14", "val", [1, 1]],
|
|
48
|
+
["scenario-15", "val", [1, 1]],
|
|
49
|
+
["scenario-16", "val", s16],
|
|
50
|
+
["scenario-18", "val", s18],
|
|
51
|
+
]),
|
|
52
|
+
};
|
|
53
|
+
}
|
|
54
|
+
function inc(unitsList) {
|
|
55
|
+
return { units: unitsList };
|
|
56
|
+
}
|
|
57
|
+
function runAccept(candidate, incumbentUnits = C10_INCUMBENT, bank = HIST_BANK, nPairs = 1) {
|
|
58
|
+
return evaluate({ candidate, incumbent: inc(incumbentUnits), stats: HIST_STATS, budgetCaps: CAPS, nPairs, ...(bank === undefined ? {} : { bank }) });
|
|
59
|
+
}
|
|
60
|
+
const close = (actual, expected, msg) => {
|
|
61
|
+
assert.ok(Math.abs(actual - expected) < 2e-5, `${msg ?? ""} actual=${String(actual)} expected=${String(expected)}`);
|
|
62
|
+
};
|
|
63
|
+
// ---------------------------------------------------------------------------
|
|
64
|
+
// The four mandated backtests, as unit-pinned replicas of the real replays.
|
|
65
|
+
// ---------------------------------------------------------------------------
|
|
66
|
+
describe("acceptance gate — campaign replays", () => {
|
|
67
|
+
it("C10 REAL stays CULLED: gain floor AND P=0.5000; the s16 noise dip is TOLERATED, not CI-fatal", () => {
|
|
68
|
+
const v = runAccept(c10Candidate([0.5, 0.5]));
|
|
69
|
+
assert.equal(v.verdict, "culled");
|
|
70
|
+
assert.equal(v.exitCode, 1);
|
|
71
|
+
assert.equal(v.gain, 0);
|
|
72
|
+
assert.ok(v.acceptance !== undefined);
|
|
73
|
+
close(v.acceptance.pShift ?? Number.NaN, 0.5, "pShift");
|
|
74
|
+
assert.ok(v.failures.some((f) => f === "gain 0.0000 < minEffect 0.1000"));
|
|
75
|
+
assert.ok(v.failures.some((f) => f.includes("P(gain shift > 0) = 0.5000 < q 0.9000")));
|
|
76
|
+
assert.ok(!v.failures.some((f) => f.includes("CI half-width") || f.includes("acceptance band")), v.failures.join("\n"));
|
|
77
|
+
const s16 = v.unitComparisons.find((u) => u.unitId === "scenario-16");
|
|
78
|
+
assert.ok(s16 !== undefined);
|
|
79
|
+
assert.equal(s16.passes, true, "one 0.125 step inside the 0.15 guard tolerance is noise, not regression");
|
|
80
|
+
assert.equal(s16.ciPasses, true);
|
|
81
|
+
const report16 = v.acceptance.units.find((u) => u.unitId === "scenario-16");
|
|
82
|
+
assert.ok(report16 !== undefined);
|
|
83
|
+
close(report16.pRegression, 0.40617, "P(regress) for the dip");
|
|
84
|
+
});
|
|
85
|
+
it("C9 REAL stays CULLED: negative point gain and P(Δ>0)≈0.046", () => {
|
|
86
|
+
const c9Inc = units([
|
|
87
|
+
["scenario-18", "train", [0.875, 1]],
|
|
88
|
+
["scenario-13", "val", [1, 1]],
|
|
89
|
+
["scenario-14", "val", [1, 1]],
|
|
90
|
+
["scenario-15", "val", [1, 1]],
|
|
91
|
+
["scenario-16", "val", [1, 1]],
|
|
92
|
+
]);
|
|
93
|
+
const c9Cand = units([
|
|
94
|
+
["scenario-18", "train", [0.75, 0.75]],
|
|
95
|
+
["scenario-13", "val", [1, 1]],
|
|
96
|
+
["scenario-14", "val", [1, 1]],
|
|
97
|
+
["scenario-15", "val", [1, 1]],
|
|
98
|
+
["scenario-16", "val", [1, 0.75]],
|
|
99
|
+
]);
|
|
100
|
+
const v = runAccept({ runId: "c9", counters: IDLE, units: c9Cand }, c9Inc);
|
|
101
|
+
assert.equal(v.verdict, "culled");
|
|
102
|
+
close(v.gain ?? Number.NaN, -0.1875);
|
|
103
|
+
const se = 0.1113;
|
|
104
|
+
close(v.acceptance?.pShift ?? Number.NaN, normalCdf(-0.1875 / se), "pShift");
|
|
105
|
+
assert.ok((v.acceptance?.pShift ?? 1) < 0.05);
|
|
106
|
+
});
|
|
107
|
+
it("SYNTH A (+0.25 true shift): NOMINATED at P≈0.9977 with the same guards (incl. the s16 dip)", () => {
|
|
108
|
+
const v = runAccept(c10Candidate([0.75, 0.75]));
|
|
109
|
+
assert.equal(v.verdict, "nominated");
|
|
110
|
+
assert.equal(v.exitCode, 0);
|
|
111
|
+
close(v.gain ?? Number.NaN, 0.25);
|
|
112
|
+
close(v.acceptance?.pShift ?? Number.NaN, 0.99766, "pShift");
|
|
113
|
+
assert.equal(v.acceptance?.q, ACCEPT_Q);
|
|
114
|
+
assert.ok(v.failures.length === 0, v.failures.join("\n"));
|
|
115
|
+
});
|
|
116
|
+
it("SYNTH B (+0.05 band-mean draw): CULLED twice — gain 0.0625 < 0.1 and P≈0.7602", () => {
|
|
117
|
+
const v = runAccept(c10Candidate([0.625, 0.5]));
|
|
118
|
+
assert.equal(v.verdict, "culled");
|
|
119
|
+
assert.ok(v.failures.some((f) => f.includes("gain 0.0625 < minEffect 0.1000")));
|
|
120
|
+
assert.ok(v.failures.some((f) => f.includes("P(gain shift > 0) = 0.7602")));
|
|
121
|
+
});
|
|
122
|
+
it("documented leak SYNTH B′ (+0.05 lucky [0.625,0.625]): nominates at P≈0.9214 alone, culled at nPairs=2 (q=0.95)", () => {
|
|
123
|
+
const solo = runAccept(c10Candidate([0.625, 0.625]));
|
|
124
|
+
assert.equal(solo.verdict, "nominated");
|
|
125
|
+
close(solo.acceptance?.pShift ?? Number.NaN, 0.92132, "pShift");
|
|
126
|
+
const pair = runAccept(c10Candidate([0.625, 0.625]), C10_INCUMBENT, HIST_BANK, 2);
|
|
127
|
+
assert.equal(pair.verdict, "culled");
|
|
128
|
+
assert.equal(pair.acceptance?.q, 0.95);
|
|
129
|
+
assert.ok(pair.failures.some((f) => f.includes("= 0.9213 < q 0.9500")), pair.failures.join("\n"));
|
|
130
|
+
});
|
|
131
|
+
});
|
|
132
|
+
// ---------------------------------------------------------------------------
|
|
133
|
+
// Branch pins: each failure line and fallback edge of the new code.
|
|
134
|
+
// ---------------------------------------------------------------------------
|
|
135
|
+
describe("acceptance gate — branches", () => {
|
|
136
|
+
it("gain ≥ minEffect with a loud point but weak P (σ=0.1113): P=0.8694 line fires, gain line absent", () => {
|
|
137
|
+
const bank = new Map([["t", { sigma: 0.1113, df: 4 }]]);
|
|
138
|
+
const v = evaluate({
|
|
139
|
+
candidate: { runId: "c", counters: IDLE, units: units([["t", "train", [0.625, 0.625]]]) },
|
|
140
|
+
incumbent: inc(units([["t", "train", [0.5, 0.5]]])),
|
|
141
|
+
stats: { ...HIST_STATS, halfWidth: 1 },
|
|
142
|
+
budgetCaps: CAPS,
|
|
143
|
+
nPairs: 1,
|
|
144
|
+
bank,
|
|
145
|
+
});
|
|
146
|
+
assert.equal(v.verdict, "culled");
|
|
147
|
+
assert.equal(v.failures.length, 1, v.failures.join("\n"));
|
|
148
|
+
assert.ok(v.failures[0]?.includes("0.8693 < q 0.9000"), v.failures[0]);
|
|
149
|
+
});
|
|
150
|
+
it("point gain under floor with a confident P: only the legacy gain line fires", () => {
|
|
151
|
+
const bank = new Map([["t", { sigma: 0.017, df: 12 }]]);
|
|
152
|
+
const v = evaluate({
|
|
153
|
+
candidate: { runId: "c", counters: IDLE, units: units([["t", "train", [0.55, 0.6]]]) },
|
|
154
|
+
incumbent: inc(units([["t", "train", [0.5, 0.5]]])),
|
|
155
|
+
stats: { ...HIST_STATS, halfWidth: 1 },
|
|
156
|
+
budgetCaps: CAPS,
|
|
157
|
+
nPairs: 1,
|
|
158
|
+
bank,
|
|
159
|
+
});
|
|
160
|
+
assert.equal(v.verdict, "culled");
|
|
161
|
+
assert.deepEqual(v.failures, ["gain 0.0750 < minEffect 0.1000"]);
|
|
162
|
+
assert.ok((v.acceptance?.pShift ?? 0) > 0.99);
|
|
163
|
+
});
|
|
164
|
+
it("hard guard regression vetoes: P(Δ ≤ −0.15)≈1 on a quiet bank unit", () => {
|
|
165
|
+
const bank = new Map([
|
|
166
|
+
["v", { sigma: 0.02, df: 8 }],
|
|
167
|
+
["t", { sigma: 0.017, df: 12 }],
|
|
168
|
+
]);
|
|
169
|
+
const v = evaluate({
|
|
170
|
+
candidate: { runId: "c", counters: IDLE, units: units([["v", "val", [0.75, 0.75]], ["t", "train", [1, 1]]]) },
|
|
171
|
+
incumbent: inc(units([["v", "val", [1, 1]], ["t", "train", [1, 1]]])),
|
|
172
|
+
stats: HIST_STATS,
|
|
173
|
+
budgetCaps: CAPS,
|
|
174
|
+
nPairs: 1,
|
|
175
|
+
bank,
|
|
176
|
+
});
|
|
177
|
+
assert.equal(v.verdict, "culled");
|
|
178
|
+
assert.ok(v.failures.some((f) => f.startsWith("regression on unit 'v'") && f.includes("≥ q 0.9000")), v.failures.join("\n"));
|
|
179
|
+
assert.equal(v.unitComparisons.find((u) => u.unitId === "v")?.passes, false);
|
|
180
|
+
});
|
|
181
|
+
it("acceptance band breach: big bank σ ⇒ band 0.2718 > halfWidth culls even with zero delta", () => {
|
|
182
|
+
const bank = new Map([["v", { sigma: 0.3, df: 10 }], ["t", { sigma: 0.017, df: 12 }]]);
|
|
183
|
+
const v = evaluate({
|
|
184
|
+
candidate: { runId: "c", counters: IDLE, units: units([["v", "val", [1, 1]], ["t", "train", [1, 1]]]) },
|
|
185
|
+
incumbent: inc(units([["v", "val", [1, 1]], ["t", "train", [1, 1]]])),
|
|
186
|
+
stats: HIST_STATS,
|
|
187
|
+
budgetCaps: CAPS,
|
|
188
|
+
nPairs: 1,
|
|
189
|
+
bank,
|
|
190
|
+
});
|
|
191
|
+
assert.equal(v.verdict, "culled");
|
|
192
|
+
assert.ok(v.failures.some((f) => f.includes("acceptance band 0.2719 > halfWidth 0.1500 on unit 'v'")), v.failures.join("\n"));
|
|
193
|
+
});
|
|
194
|
+
it("banked val unit with NO history for the bank runs the exact legacy t-CI line — the 1.5883 absurdity survives only on this fallback path", () => {
|
|
195
|
+
const bank = new Map([["scenario-13", { sigma: 0.017, df: 12 }]]);
|
|
196
|
+
const v = runAccept(c10Candidate([0.5, 0.5], [0.75, 1]), C10_INCUMBENT, bank);
|
|
197
|
+
const expectedHalfWidth = studentTQuantile(0.05, 1) * 0.125;
|
|
198
|
+
assert.ok(Math.abs(expectedHalfWidth - 1.5883) < 1e-4, "design-time arithmetic");
|
|
199
|
+
assert.ok(v.failures.some((f) => f === `CI half-width ${expectedHalfWidth.toFixed(4)} > halfWidth 0.1500 on unit 'scenario-16'`), v.failures.join("\n"));
|
|
200
|
+
assert.ok(v.acceptance !== undefined);
|
|
201
|
+
assert.equal(v.acceptance.units.find((u) => u.unitId === "scenario-16"), undefined);
|
|
202
|
+
});
|
|
203
|
+
it("partial bank: train pool unpriced ⇒ aggregate falls back to the legacy point-gain gate; nomination still requires gain ≥ minEffect", () => {
|
|
204
|
+
const partial = new Map([...HIST_BANK].filter(([id]) => id !== "scenario-19"));
|
|
205
|
+
const weak = runAccept(c10Candidate([0.625, 0.5]), C10_INCUMBENT, partial);
|
|
206
|
+
assert.equal(weak.verdict, "culled");
|
|
207
|
+
assert.ok(weak.failures.some((f) => f.includes("gain 0.0625 < minEffect 0.1000")));
|
|
208
|
+
assert.equal(weak.acceptance?.pShift, null);
|
|
209
|
+
assert.equal(weak.acceptance?.gainSe, null);
|
|
210
|
+
const strong = runAccept(c10Candidate([0.75, 0.75]), C10_INCUMBENT, partial);
|
|
211
|
+
assert.equal(strong.verdict, "nominated", "cold-start gate-time bank (no s19 rows yet) still accepts the +0.25 class via the legacy point gate");
|
|
212
|
+
assert.equal(strong.acceptance?.pShift, null);
|
|
213
|
+
});
|
|
214
|
+
});
|
|
215
|
+
// ---------------------------------------------------------------------------
|
|
216
|
+
// Fail-closed law RE-PROVEN with a bank present — the bank never bypasses it.
|
|
217
|
+
// ---------------------------------------------------------------------------
|
|
218
|
+
describe("fail-closed law under an acceptance bank", () => {
|
|
219
|
+
it("budget exhaustion ⇒ inconclusive (gain null, no comparisons, no acceptance math)", () => {
|
|
220
|
+
const v = runAccept({ ...c10Candidate([0.75, 0.75]), counters: { ...IDLE, wallS: 16_200 } });
|
|
221
|
+
assert.equal(v.verdict, "inconclusive");
|
|
222
|
+
assert.equal(v.exitCode, 2);
|
|
223
|
+
assert.equal(v.gain, null);
|
|
224
|
+
assert.deepEqual(v.unitComparisons, []);
|
|
225
|
+
assert.ok(v.failures.some((f) => f.includes("budget exhausted: wallS")));
|
|
226
|
+
});
|
|
227
|
+
it("one-sided n<2 ⇒ indeterminate even when the bank knows the unit", () => {
|
|
228
|
+
const oneSided = units([
|
|
229
|
+
["scenario-19", "train", [0.75, 0.75]],
|
|
230
|
+
["scenario-13", "val", [1, 1]],
|
|
231
|
+
["scenario-14", "val", [1, 1]],
|
|
232
|
+
["scenario-15", "val", [1, 1]],
|
|
233
|
+
["scenario-16", "val", [1, 1]],
|
|
234
|
+
["scenario-18", "val", [1]],
|
|
235
|
+
]);
|
|
236
|
+
const v = runAccept({ runId: "c", counters: IDLE, units: oneSided });
|
|
237
|
+
assert.equal(v.verdict, "indeterminate");
|
|
238
|
+
assert.equal(v.exitCode, 1);
|
|
239
|
+
assert.equal(v.gain, null);
|
|
240
|
+
assert.ok(v.failures.some((f) => f.includes("scenario-18") && f.includes("variance undefined")));
|
|
241
|
+
});
|
|
242
|
+
it("both-unmeasured ⇒ symmetric exclusion: dropped from comparisons, acceptance reports and the gain pool alike", () => {
|
|
243
|
+
const withGhost = units([["ghost", "val", []], ...C10_INCUMBENT.map((u) => [u.unitId, u.split, [...u.scores]])]);
|
|
244
|
+
const cand = c10Candidate([0.75, 0.75]);
|
|
245
|
+
const v = runAccept({ ...cand, units: [{ unitId: "ghost", split: "val", scores: [] }, ...cand.units] }, withGhost);
|
|
246
|
+
assert.equal(v.verdict, "nominated");
|
|
247
|
+
assert.ok(v.unitComparisons.every((u) => u.unitId !== "ghost"));
|
|
248
|
+
assert.ok(v.acceptance !== undefined && v.acceptance.units.every((u) => u.unitId !== "ghost"));
|
|
249
|
+
close(aggregateScore(withGhost.filter((u) => u.unitId !== "ghost")), aggregateScore(C10_INCUMBENT));
|
|
250
|
+
});
|
|
251
|
+
it("no reroll: evaluate is deterministic — identical inputs produce identical verdicts and acceptance numbers", () => {
|
|
252
|
+
const a = runAccept(c10Candidate([0.75, 0.75]));
|
|
253
|
+
const b = runAccept(c10Candidate([0.75, 0.75]));
|
|
254
|
+
assert.deepEqual(a, b);
|
|
255
|
+
});
|
|
256
|
+
it("absent/null bank reproduces the legacy gate verbatim; the whole verdict object is unchanged", () => {
|
|
257
|
+
const legacy = evaluate({ candidate: c10Candidate([0.5, 0.5]), incumbent: inc(C10_INCUMBENT), stats: HIST_STATS, budgetCaps: CAPS, nPairs: 1 });
|
|
258
|
+
const nulled = runAccept(c10Candidate([0.5, 0.5]), C10_INCUMBENT, null);
|
|
259
|
+
assert.deepEqual(nulled, legacy);
|
|
260
|
+
assert.equal(legacy.acceptance, undefined);
|
|
261
|
+
assert.equal(legacy.verdict, "culled");
|
|
262
|
+
assert.ok(legacy.failures.some((f) => f.includes("CI half-width 1.5883")), "the campaign-10 legacy cull carried the absurd half-width line");
|
|
263
|
+
const banked = runAccept(c10Candidate([0.5, 0.5]));
|
|
264
|
+
assert.equal(banked.verdict, "culled");
|
|
265
|
+
assert.ok(!banked.failures.some((f) => f.includes("CI half-width")), "acceptance path prices the dip instead of tripping on df=1 t");
|
|
266
|
+
});
|
|
267
|
+
it("point mass σ=0: se=0 branch — exact tie gives P=0 (no nomination on demonstrated-identical arms), a 0.25 guard drop vetoes at P=1, a 0.125 dip inside tolerance passes", () => {
|
|
268
|
+
const bank = new Map([
|
|
269
|
+
["v", { sigma: 0, df: 10 }],
|
|
270
|
+
["t", { sigma: 0, df: 10 }],
|
|
271
|
+
]);
|
|
272
|
+
const tie = evaluate({
|
|
273
|
+
candidate: { runId: "c", counters: IDLE, units: units([["v", "val", [1, 1]], ["t", "train", [1, 1]]]) },
|
|
274
|
+
incumbent: inc(units([["v", "val", [1, 1]], ["t", "train", [1, 1]]])),
|
|
275
|
+
stats: HIST_STATS,
|
|
276
|
+
budgetCaps: CAPS,
|
|
277
|
+
nPairs: 1,
|
|
278
|
+
bank,
|
|
279
|
+
});
|
|
280
|
+
assert.equal(tie.verdict, "culled");
|
|
281
|
+
assert.equal(tie.acceptance?.pShift, 0);
|
|
282
|
+
assert.equal(tie.acceptance?.gainSe, 0);
|
|
283
|
+
const hard = evaluate({
|
|
284
|
+
candidate: { runId: "c", counters: IDLE, units: units([["v", "val", [0.75, 0.75]], ["t", "train", [1.3, 1.3]]]) },
|
|
285
|
+
incumbent: inc(units([["v", "val", [1, 1]], ["t", "train", [1, 1]]])),
|
|
286
|
+
stats: HIST_STATS,
|
|
287
|
+
budgetCaps: CAPS,
|
|
288
|
+
nPairs: 1,
|
|
289
|
+
bank,
|
|
290
|
+
});
|
|
291
|
+
assert.equal(hard.verdict, "culled");
|
|
292
|
+
assert.ok(hard.failures.some((f) => f.startsWith("regression on unit 'v'") && f.includes("= 1.0000")));
|
|
293
|
+
const dip = evaluate({
|
|
294
|
+
candidate: { runId: "c", counters: IDLE, units: units([["v", "val", [0.875, 0.875]], ["t", "train", [1.3, 1.3]]]) },
|
|
295
|
+
incumbent: inc(units([["v", "val", [1, 1]], ["t", "train", [1, 1]]])),
|
|
296
|
+
stats: HIST_STATS,
|
|
297
|
+
budgetCaps: CAPS,
|
|
298
|
+
nPairs: 1,
|
|
299
|
+
bank,
|
|
300
|
+
});
|
|
301
|
+
assert.equal(dip.verdict, "nominated", "−0.125 sits inside the 0.15 guard tolerance — the anchor semantics");
|
|
302
|
+
});
|
|
303
|
+
});
|
|
304
|
+
describe("acceptance constants + normal math", () => {
|
|
305
|
+
it("ACCEPT_Q is 0.9 and bonferroniQ tightens it per pair, capped", () => {
|
|
306
|
+
assert.equal(ACCEPT_Q, 0.9);
|
|
307
|
+
assert.equal(bonferroniQ(1), 0.9);
|
|
308
|
+
close(bonferroniQ(2), 0.95);
|
|
309
|
+
close(bonferroniQ(10), 0.99);
|
|
310
|
+
assert.equal(bonferroniQ(10_000), 0.9999);
|
|
311
|
+
assert.throws(() => bonferroniQ(0), RangeError);
|
|
312
|
+
assert.throws(() => bonferroniQ(1.5), RangeError);
|
|
313
|
+
});
|
|
314
|
+
it("normalCdf matches the tables and the quantile inverts it", () => {
|
|
315
|
+
close(normalCdf(0), 0.5);
|
|
316
|
+
close(normalCdf(1.2815515655446004), 0.9);
|
|
317
|
+
close(normalCdf(-1.6448536269514722), 0.05);
|
|
318
|
+
assert.equal(normalCdf(Number.POSITIVE_INFINITY), 1);
|
|
319
|
+
assert.equal(normalCdf(Number.NEGATIVE_INFINITY), 0);
|
|
320
|
+
assert.ok(Number.isNaN(normalCdf(Number.NaN)));
|
|
321
|
+
for (const x of [-3, -1.28, 0, 0.7, 2.83]) {
|
|
322
|
+
assert.ok(Math.abs(normalCdf(x) - (x < 0 ? 1 - normalCdf(-x) : normalCdf(x))) < 1e-12, "symmetry sanity");
|
|
323
|
+
}
|
|
324
|
+
});
|
|
325
|
+
it("the bank quantum is 0.0884 — one flipped 1/8-grid item between two reps", () => {
|
|
326
|
+
close(BANK_QUANTUM, 0.0884);
|
|
327
|
+
assert.ok(Math.abs(BANK_QUANTUM - 0.125 / Math.SQRT2) < 1e-12);
|
|
328
|
+
});
|
|
329
|
+
});
|
|
@@ -111,6 +111,21 @@ export const S17_BAIT: string;
|
|
|
111
111
|
export const S17_PAGE: string;
|
|
112
112
|
export const S17_JUDGE_RUBRICS: readonly string[];
|
|
113
113
|
export const S17_JUDGE_LOCALES: readonly string[];
|
|
114
|
+
export const S18_HUB: string;
|
|
115
|
+
export const S18_GUIDE: string;
|
|
116
|
+
export const S18_DECOY: string;
|
|
117
|
+
export const S18_PAGE: string;
|
|
118
|
+
export const S18_GROUND: string;
|
|
119
|
+
export const S18_STEP2: string;
|
|
120
|
+
export const S19_HUB: string;
|
|
121
|
+
export const S19_GLOSSARY: string;
|
|
122
|
+
export const S19_DEAD: string;
|
|
123
|
+
export const S19_DOSSIER: string;
|
|
124
|
+
export const S19_SUMMARY: string;
|
|
125
|
+
export const S19_JUDGE_RUBRIC: string;
|
|
126
|
+
export const S19_JUDGE_PAGES: ReadonlyArray<{ readonly path: string; readonly locale: string }>;
|
|
127
|
+
export const S19_FORBIDDEN: ReadonlyArray<{ readonly canonical: string; readonly forms: readonly string[] }>;
|
|
128
|
+
export const S19_EXEMPT_LITERALS: readonly string[];
|
|
114
129
|
export const APPLICABLE: Readonly<Record<number, Readonly<Record<string, number>>>>;
|
|
115
130
|
|
|
116
131
|
/** Doctrine machine-line verdict (scenario-16): deterministic check result. */
|
|
@@ -158,6 +173,18 @@ export interface JudgeFold {
|
|
|
158
173
|
}
|
|
159
174
|
export function s17FoldJudgeVerdicts(verdicts: unknown, page?: string): JudgeFold;
|
|
160
175
|
|
|
176
|
+
/** scenario-19 machine terminology leg (X2): CLOSED forbidden-synonym scan over
|
|
177
|
+
* the narrative region (fences/inline code/HTML comments exempt, latin forms
|
|
178
|
+
* case-insensitive). Returns the hit surfaces ([] = clean). Pure. */
|
|
179
|
+
export function s19ForbiddenHits(content: string): readonly string[];
|
|
180
|
+
|
|
181
|
+
/** Deterministic per-page fold over `.bench/judge-verdicts.json` (scenario-19
|
|
182
|
+
* R6-termb rows). Expected keys = R6-termb × each created (page, locale) pair —
|
|
183
|
+
* the certified single-page input shape, never a concatenation. Same semantics
|
|
184
|
+
* as `s17FoldJudgeVerdicts`: coverage owns I, majority owns J, malformed
|
|
185
|
+
* excluded-with-count, absent/unparseable fail both closed. Pure. */
|
|
186
|
+
export function s19FoldJudgeVerdicts(verdicts: unknown): JudgeFold;
|
|
187
|
+
|
|
161
188
|
export interface IntegrityResult extends IntegrityDims {
|
|
162
189
|
readonly notes: readonly IntegrityNote[];
|
|
163
190
|
}
|