@tachikomagundam/abathur 0.2.5 → 0.2.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/commands/genome.js +2 -2
- package/dist/commands/graft.js +1 -1
- package/dist/commands/run.js +1 -1
- package/dist/commands/self-eval.js +1 -1
- package/dist/commands/tombstone.js +2 -1
- package/dist/core/evolve/run-loop.js +14 -0
- package/dist/core/evolve/score-bank.js +214 -0
- package/dist/core/graft-rebench.js +3 -0
- package/dist/core/ledger.js +6 -1
- package/dist/core/promote.js +2 -1
- package/dist/core/stats-math.js +39 -0
- package/dist/core/stats.js +93 -19
- package/dist/test/historian-grader-integrity.test.js +509 -2
- package/dist/test/historian-run-scenario.test.js +2 -2
- package/dist/test/repopath-seams.test.js +54 -0
- package/dist/test/score-bank.test.js +235 -0
- package/dist/test/seam-gates.test.js +78 -0
- package/dist/test/stats-acceptance.test.js +329 -0
- package/graders/historian/grader-core.d.mts +27 -0
- package/graders/historian/grader-core.mjs +311 -19
- package/graders/historian/grader.mjs +5 -5
- package/graders/historian/judge-poststage-19.mjs +212 -0
- package/graders/historian/judge-poststage.mjs +3 -8
- package/graders/historian/run-scenario-19.sh +43 -0
- package/graders/historian/run-scenario.sh +16 -0
- package/package.json +1 -1
- package/plugin/abathur.ts +1 -1
|
@@ -0,0 +1,235 @@
|
|
|
1
|
+
// Score-bank loader tests (acceptance-semantics redesign, 2026-09-24): the
|
|
2
|
+
// fail-closed activation law, exact shrinkage/quantum-floor arithmetic, row
|
|
3
|
+
// skipping with notices, corrupt-tail/backup exclusion and the gate-time
|
|
4
|
+
// candidate-tree exclusion hook the A9 backtest replays through.
|
|
5
|
+
import assert from "node:assert/strict";
|
|
6
|
+
import { existsSync, mkdirSync, mkdtempSync, rmSync, writeFileSync } from "node:fs";
|
|
7
|
+
import os from "node:os";
|
|
8
|
+
import path from "node:path";
|
|
9
|
+
import { describe, it } from "node:test";
|
|
10
|
+
import { BANK_QUANTUM, BANK_THIN_DF, loadScoreBank } from "../core/evolve/score-bank.js";
|
|
11
|
+
const STATE_DIR = path.join(".state", "abathur");
|
|
12
|
+
function generationLine(row) {
|
|
13
|
+
const data = {
|
|
14
|
+
source: row.source,
|
|
15
|
+
...(row.source === "candidate" ? { candidateId: `c-${row.tree}`, treeSha: row.tree, commitSha: row.tree ?? "c" } : {}),
|
|
16
|
+
headCommit: row.head,
|
|
17
|
+
complete: row.complete ?? true,
|
|
18
|
+
reps: 2,
|
|
19
|
+
units: row.units.map((u) => ({ ...u, runIds: u.scores.map((_, i) => `r${String(i)}`), failures: [] })),
|
|
20
|
+
counters: { candidates: row.source === "candidate" ? 1 : 0, modelCalls: 0, tokens: 1, wallS: 1 },
|
|
21
|
+
manifest: [],
|
|
22
|
+
benchProvenance: { benchType: "toy", versions: [] },
|
|
23
|
+
};
|
|
24
|
+
return `${JSON.stringify({ v: 1, ts: row.ts ?? "2026-09-24T00:00:00.000Z", kind: "generation_complete", genId: "g-1", runId: "a-1", data })}\n`;
|
|
25
|
+
}
|
|
26
|
+
function withRepo(write) {
|
|
27
|
+
const repo = mkdtempSync(path.join(os.tmpdir(), "abathur-bank-"));
|
|
28
|
+
try {
|
|
29
|
+
write(repo);
|
|
30
|
+
}
|
|
31
|
+
finally {
|
|
32
|
+
rmSync(repo, { recursive: true, force: true });
|
|
33
|
+
}
|
|
34
|
+
}
|
|
35
|
+
function stateDir(repo) {
|
|
36
|
+
const dir = path.join(repo, STATE_DIR);
|
|
37
|
+
mkdirSync(dir, { recursive: true });
|
|
38
|
+
return dir;
|
|
39
|
+
}
|
|
40
|
+
describe("loadScoreBank fail-closed activation", () => {
|
|
41
|
+
it("no .state dir at all ⇒ null — and the loader never creates one (peekPlanState discipline)", () => {
|
|
42
|
+
withRepo((repo) => {
|
|
43
|
+
assert.equal(loadScoreBank(repo), null);
|
|
44
|
+
assert.equal(existsSync(path.join(repo, STATE_DIR)), false);
|
|
45
|
+
});
|
|
46
|
+
});
|
|
47
|
+
it("live ledger only ⇒ null — rotation is the activation law", () => {
|
|
48
|
+
withRepo((repo) => {
|
|
49
|
+
const dir = stateDir(repo);
|
|
50
|
+
writeFileSync(path.join(dir, "ledger.jsonl"), generationLine({ source: "incumbent", head: "a".repeat(40), units: [{ unitId: "u", split: "train", scores: [0, 2, 1, 3] }] }));
|
|
51
|
+
assert.equal(loadScoreBank(repo), null);
|
|
52
|
+
});
|
|
53
|
+
});
|
|
54
|
+
it("ledger.corrupt-* is NOT a rotated archive; an empty dir ⇒ null", () => {
|
|
55
|
+
withRepo((repo) => {
|
|
56
|
+
const dir = stateDir(repo);
|
|
57
|
+
writeFileSync(path.join(dir, "ledger.corrupt-20260924-9.jsonl"), generationLine({ source: "incumbent", head: "a".repeat(40), units: [{ unitId: "u", split: "train", scores: [0, 2, 1, 3] }] }));
|
|
58
|
+
assert.equal(loadScoreBank(repo), null);
|
|
59
|
+
rmSync(path.join(dir, "ledger.corrupt-20260924-9.jsonl"));
|
|
60
|
+
writeFileSync(path.join(dir, "notes.txt"), "unrelated");
|
|
61
|
+
assert.equal(loadScoreBank(repo), null);
|
|
62
|
+
});
|
|
63
|
+
});
|
|
64
|
+
it("archive + live load; totalDf=0 (only single-rep groups ever) ⇒ null", () => {
|
|
65
|
+
withRepo((repo) => {
|
|
66
|
+
const dir = stateDir(repo);
|
|
67
|
+
writeFileSync(path.join(dir, "ledger.campaign1-x.jsonl"), generationLine({ source: "incumbent", head: "a".repeat(40), units: [{ unitId: "u", split: "train", scores: [1] }, { unitId: "v", split: "val", scores: [0.5] }] }));
|
|
68
|
+
writeFileSync(path.join(dir, "ledger.jsonl"), generationLine({ source: "incumbent", head: "b".repeat(40), units: [{ unitId: "u", split: "train", scores: [2] }] }));
|
|
69
|
+
assert.equal(loadScoreBank(repo), null);
|
|
70
|
+
});
|
|
71
|
+
});
|
|
72
|
+
});
|
|
73
|
+
describe("loadScoreBank arithmetic", () => {
|
|
74
|
+
const scatterRow = (head, tree, source, scores) => generationLine({ source, head, ...(tree === undefined ? {} : { tree }), units: [{ unitId: "u", split: "train", scores }] });
|
|
75
|
+
it("pools within-group SS across lineages; thin df floors σ at the quantum; df≥6 keeps the shrunk σ", () => {
|
|
76
|
+
withRepo((repo) => {
|
|
77
|
+
const dir = stateDir(repo);
|
|
78
|
+
// group g1 (campaign1 archive): [0,2] → mean 1, SS 2, df 1 — the ONLY scatter in the bank.
|
|
79
|
+
writeFileSync(path.join(dir, "ledger.campaign1-x.jsonl"), `${scatterRow("a".repeat(40), undefined, "incumbent", [0, 2])}` +
|
|
80
|
+
// group g2 (live): same unit, different lineage head, zero scatter
|
|
81
|
+
scatterRow("b".repeat(40), undefined, "incumbent", [1, 1]));
|
|
82
|
+
writeFileSync(path.join(dir, "ledger.jsonl"), generationLine({ source: "candidate", head: "b".repeat(40), tree: "t1", units: [{ unitId: "q", split: "train", scores: [1, 1, 1, 1] }] }));
|
|
83
|
+
const bank = loadScoreBank(repo);
|
|
84
|
+
assert.ok(bank !== null);
|
|
85
|
+
// totals: u groups df 1+1 (SS 2+0), q group df 3 (SS 0) ⇒ total df 5, prior σ² = 2/5.
|
|
86
|
+
assert.equal(bank.totalDf, 5);
|
|
87
|
+
assert.ok(Math.abs(bank.priorSigma - Math.sqrt(0.4)) < 1e-12);
|
|
88
|
+
// unit u: σ̂² = 2/2 = 1; shrunk = (2·1 + 2·0.4)/4 = 0.7 ⇒ σ = √0.7 (floor 0.0884 does not bind).
|
|
89
|
+
const u = bank.units.get("u");
|
|
90
|
+
assert.ok(u !== undefined);
|
|
91
|
+
assert.equal(u.df, 2);
|
|
92
|
+
assert.ok(Math.abs(u.sigma - Math.sqrt(0.7)) < 1e-12, `sigma ${String(u.sigma)}`);
|
|
93
|
+
// unit q: df 3, zero SS ⇒ shrunk toward the prior: (3·0 + 2·0.4)/5 = 0.16 ⇒ σ = 0.4 > floor.
|
|
94
|
+
const q = bank.units.get("q");
|
|
95
|
+
assert.ok(q !== undefined);
|
|
96
|
+
assert.ok(Math.abs(q.sigma - 0.4) < 1e-12);
|
|
97
|
+
});
|
|
98
|
+
});
|
|
99
|
+
it("zero-scatter bank: thin units sit exactly on BANK_QUANTUM, proven-quiet units (df≥6) on their shrunk σ", () => {
|
|
100
|
+
withRepo((repo) => {
|
|
101
|
+
const dir = stateDir(repo);
|
|
102
|
+
const thin = generationLine({ source: "incumbent", head: "a".repeat(40), units: [{ unitId: "thin", split: "train", scores: [0.5, 0.5] }] });
|
|
103
|
+
const wide = generationLine({ source: "incumbent", head: "b".repeat(40), units: [{ unitId: "wide", split: "train", scores: [1, 1, 1, 1, 1, 1] }] }) + generationLine({ source: "candidate", head: "b".repeat(40), tree: "x", units: [{ unitId: "wide", split: "train", scores: [1, 1, 1] }] });
|
|
104
|
+
writeFileSync(path.join(dir, "ledger.campaign1-x.jsonl"), thin + wide);
|
|
105
|
+
const bank = loadScoreBank(repo);
|
|
106
|
+
assert.ok(bank !== null);
|
|
107
|
+
assert.equal(bank.priorSigma, 0);
|
|
108
|
+
assert.equal(bank.units.get("thin")?.sigma, BANK_QUANTUM); // shrunk 0 → floored
|
|
109
|
+
assert.equal(bank.units.get("thin")?.df, 1);
|
|
110
|
+
assert.equal(bank.units.get("wide")?.sigma, 0); // df 4+1+… ≥ BANK_THIN_DF, zero prior ⇒ stays 0
|
|
111
|
+
assert.ok((bank.units.get("wide")?.df ?? 0) >= BANK_THIN_DF);
|
|
112
|
+
});
|
|
113
|
+
});
|
|
114
|
+
});
|
|
115
|
+
describe("loadScoreBank row hygiene", () => {
|
|
116
|
+
it("corrupt JSON lines and schema-invalid rows are skipped with notices; valid rows still price", () => {
|
|
117
|
+
withRepo((repo) => {
|
|
118
|
+
const dir = stateDir(repo);
|
|
119
|
+
const good = generationLine({ source: "incumbent", head: "a".repeat(40), units: [{ unitId: "u", split: "train", scores: [0, 2, 2, 0] }] });
|
|
120
|
+
writeFileSync(path.join(dir, "ledger.campaign1-x.jsonl"), `${good}{"v":1,"ts":"x\n${generationLine({ source: "incumbent", head: "a".repeat(40), units: [{ unitId: "bad", split: "nope", scores: [1, 2] }] })}`);
|
|
121
|
+
const bank = loadScoreBank(repo);
|
|
122
|
+
assert.ok(bank !== null);
|
|
123
|
+
assert.ok(bank.units.has("u"));
|
|
124
|
+
assert.equal(bank.units.has("bad"), false);
|
|
125
|
+
assert.equal(bank.notices.length, 2);
|
|
126
|
+
assert.match(bank.notices.join("\n"), /not JSON/);
|
|
127
|
+
assert.match(bank.notices.join("\n"), /unreadable/);
|
|
128
|
+
});
|
|
129
|
+
});
|
|
130
|
+
it("budget-truncated rows (complete:false) still contribute their real replicate scores", () => {
|
|
131
|
+
withRepo((repo) => {
|
|
132
|
+
const dir = stateDir(repo);
|
|
133
|
+
writeFileSync(path.join(dir, "ledger.campaign1-x.jsonl"), generationLine({ source: "candidate", head: "a".repeat(40), tree: "t", complete: false, units: [{ unitId: "u", split: "train", scores: [0, 2] }] }));
|
|
134
|
+
const bank = loadScoreBank(repo);
|
|
135
|
+
assert.equal(bank?.units.get("u")?.df, 1);
|
|
136
|
+
});
|
|
137
|
+
});
|
|
138
|
+
it("excludeCandidateTrees replays the gate-time world (rows written after evaluate)", () => {
|
|
139
|
+
withRepo((repo) => {
|
|
140
|
+
const dir = stateDir(repo);
|
|
141
|
+
writeFileSync(path.join(dir, "ledger.campaign1-x.jsonl"), generationLine({ source: "incumbent", head: "a".repeat(40), units: [{ unitId: "u", split: "train", scores: [0, 2] }] }) +
|
|
142
|
+
generationLine({ source: "candidate", head: "a".repeat(40), tree: "to-exclude", units: [{ unitId: "u", split: "train", scores: [4, 0] }] }));
|
|
143
|
+
const withIt = loadScoreBank(repo);
|
|
144
|
+
const without = loadScoreBank(repo, { excludeCandidateTrees: ["to-exclude"] });
|
|
145
|
+
assert.equal(withIt?.units.get("u")?.df, 2);
|
|
146
|
+
assert.equal(without?.units.get("u")?.df, 1);
|
|
147
|
+
assert.equal(withIt?.totalDf, 2);
|
|
148
|
+
assert.equal(without?.totalDf, 1);
|
|
149
|
+
});
|
|
150
|
+
});
|
|
151
|
+
it("excludeFiles hides a future campaign's ledger; null once no rotated archive remains", () => {
|
|
152
|
+
withRepo((repo) => {
|
|
153
|
+
const dir = stateDir(repo);
|
|
154
|
+
writeFileSync(path.join(dir, "ledger.campaign1-x.jsonl"), generationLine({ source: "incumbent", head: "a".repeat(40), units: [{ unitId: "u", split: "train", scores: [0, 2, 2, 0] }] }));
|
|
155
|
+
writeFileSync(path.join(dir, "ledger.campaign2-y.jsonl"), generationLine({ source: "incumbent", head: "b".repeat(40), units: [{ unitId: "u", split: "train", scores: [1, 5] }] }));
|
|
156
|
+
assert.equal(loadScoreBank(repo)?.units.get("u")?.df, 4);
|
|
157
|
+
const hidden = loadScoreBank(repo, { excludeFiles: ["ledger.campaign2-y.jsonl"] });
|
|
158
|
+
assert.equal(hidden?.units.get("u")?.df, 3);
|
|
159
|
+
assert.equal(hidden?.totalDf, 3);
|
|
160
|
+
assert.equal(loadScoreBank(repo, { excludeFiles: ["ledger.campaign1-x.jsonl", "ledger.campaign2-y.jsonl"] }), null);
|
|
161
|
+
});
|
|
162
|
+
});
|
|
163
|
+
});
|
|
164
|
+
describe("loadScoreBank bank-epoch lineage quarantine", () => {
|
|
165
|
+
const quietNew = {
|
|
166
|
+
source: "incumbent",
|
|
167
|
+
head: "h1",
|
|
168
|
+
ts: "2026-09-25T00:00:00.000Z",
|
|
169
|
+
units: [{ unitId: "scenario-9", split: "val", scores: [1, 1] }],
|
|
170
|
+
};
|
|
171
|
+
const noisyOld = {
|
|
172
|
+
source: "incumbent",
|
|
173
|
+
head: "h1",
|
|
174
|
+
ts: "2026-09-20T00:00:00.000Z",
|
|
175
|
+
units: [{ unitId: "scenario-9", split: "val", scores: [0, 1, 0, 1] }],
|
|
176
|
+
};
|
|
177
|
+
it("groups with ts before the epoch are quarantined: df drops, notice names the count and why", () => {
|
|
178
|
+
withRepo((repo) => {
|
|
179
|
+
const dir = stateDir(repo);
|
|
180
|
+
writeFileSync(path.join(dir, "ledger.campaign1-2026-09-20.jsonl"), generationLine(noisyOld) + generationLine(quietNew));
|
|
181
|
+
writeFileSync(path.join(dir, "bank-epoch.json"), JSON.stringify({ "scenario-9": { sinceIso: "2026-09-24T00:00:00Z", why: "pin era voided" } }));
|
|
182
|
+
const bank = loadScoreBank(repo);
|
|
183
|
+
assert.ok(bank !== null);
|
|
184
|
+
const stat = bank.units.get("scenario-9");
|
|
185
|
+
assert.ok(stat !== undefined);
|
|
186
|
+
assert.equal(stat.df, 1); // only the quiet new group counts
|
|
187
|
+
assert.ok(stat.sigma >= BANK_QUANTUM - 1e-12); // thin history floor holds the prior
|
|
188
|
+
assert.ok(bank.notices.some((n) => n.includes("bank-epoch 'scenario-9' since") && n.includes("1 group(s) quarantined") && n.includes("pin era voided")));
|
|
189
|
+
assert.ok(bank.notices.some((n) => n.includes("since 2026-09-24T00:00:00Z") && !n.includes("undefined")), "notice must print the ISO epoch, never undefined");
|
|
190
|
+
});
|
|
191
|
+
});
|
|
192
|
+
it("quarantined-only units leave zero pooled df ⇒ bank null ⇒ caller runs the legacy gate", () => {
|
|
193
|
+
withRepo((repo) => {
|
|
194
|
+
const dir = stateDir(repo);
|
|
195
|
+
writeFileSync(path.join(dir, "ledger.campaign1-2026-09-20.jsonl"), generationLine(noisyOld));
|
|
196
|
+
writeFileSync(path.join(dir, "bank-epoch.json"), JSON.stringify({ "scenario-9": { sinceIso: "2026-09-24T00:00:00Z" } }));
|
|
197
|
+
assert.equal(loadScoreBank(repo), null);
|
|
198
|
+
});
|
|
199
|
+
});
|
|
200
|
+
it("unreadable row ts fails closed for quarantined units but leaves unruled units untouched", () => {
|
|
201
|
+
withRepo((repo) => {
|
|
202
|
+
const dir = stateDir(repo);
|
|
203
|
+
const ruled = { ...noisyOld, ts: "not-a-date", units: [
|
|
204
|
+
{ unitId: "scenario-9", split: "val", scores: [0, 1] },
|
|
205
|
+
{ unitId: "scenario-8", split: "val", scores: [1, 0] },
|
|
206
|
+
] };
|
|
207
|
+
writeFileSync(path.join(dir, "ledger.campaign1-2026-09-20.jsonl"), generationLine(ruled));
|
|
208
|
+
writeFileSync(path.join(dir, "bank-epoch.json"), JSON.stringify({ "scenario-9": { sinceIso: "2026-09-24T00:00:00Z" } }));
|
|
209
|
+
const bank = loadScoreBank(repo);
|
|
210
|
+
assert.ok(bank !== null);
|
|
211
|
+
assert.equal(bank.units.has("scenario-9"), false); // provably-at-or-after failed ⇒ quarantined
|
|
212
|
+
assert.ok(bank.units.has("scenario-8")); // no rule ⇒ untouched
|
|
213
|
+
});
|
|
214
|
+
});
|
|
215
|
+
it("malformed quarantine sidecars disable the WHOLE bank (never a silently ignored quarantine)", () => {
|
|
216
|
+
const bad = ["{not json", JSON.stringify({ "scenario-9": { sinceIso: "not-a-date" } }), JSON.stringify({ "scenario-9": { sinceIso: "2026-09-24T00:00:00Z", extra: 1 } }), JSON.stringify(["scenario-9"])];
|
|
217
|
+
for (const body of bad) {
|
|
218
|
+
withRepo((repo) => {
|
|
219
|
+
const dir = stateDir(repo);
|
|
220
|
+
writeFileSync(path.join(dir, "ledger.campaign1-2026-09-20.jsonl"), generationLine(noisyOld) + generationLine(quietNew));
|
|
221
|
+
writeFileSync(path.join(dir, "bank-epoch.json"), body);
|
|
222
|
+
assert.equal(loadScoreBank(repo), null, `sidecar must disable bank: ${body.slice(0, 24)}`);
|
|
223
|
+
});
|
|
224
|
+
}
|
|
225
|
+
});
|
|
226
|
+
it("absent sidecar = no quarantine, byte-identical legacy behavior", () => {
|
|
227
|
+
withRepo((repo) => {
|
|
228
|
+
const dir = stateDir(repo);
|
|
229
|
+
writeFileSync(path.join(dir, "ledger.campaign1-2026-09-20.jsonl"), generationLine(noisyOld) + generationLine(quietNew));
|
|
230
|
+
const bank = loadScoreBank(repo);
|
|
231
|
+
assert.ok(bank !== null);
|
|
232
|
+
assert.equal(bank.units.get("scenario-9")?.df, 4); // 3 df old + 1 df new — nothing skipped
|
|
233
|
+
});
|
|
234
|
+
});
|
|
235
|
+
});
|
|
@@ -0,0 +1,78 @@
|
|
|
1
|
+
// Seam test (postmortem 863e6c2): the human-gate commands `promote` and
|
|
2
|
+
// `tombstone` must resolve an env-literal repoPath exactly like status/run
|
|
3
|
+
// (4997618 pattern). Pre-fix, promote.ts opened the ledger under the STORED
|
|
4
|
+
// UNRESOLVED `${VAR}` literal, read an empty ledger, and reported the
|
|
5
|
+
// misleading "no generation_complete ledger row" with the literal still in
|
|
6
|
+
// the hint — the same fail-open seam class as the status spawn-ENOENT.
|
|
7
|
+
// Pins, for both gates:
|
|
8
|
+
// (1) env exported ⇒ gate reads the REAL ledger at the resolved path:
|
|
9
|
+
// the refusal names the resolved repo, never the `${VAR}` literal;
|
|
10
|
+
// (2) env unset ⇒ clean exit 2 NAMING the variable;
|
|
11
|
+
// (3) stderr never contains the unresolved literal or a spawn-ENOENT.
|
|
12
|
+
import assert from "node:assert/strict";
|
|
13
|
+
import { spawnSync } from "node:child_process";
|
|
14
|
+
import { mkdirSync, readFileSync, writeFileSync } from "node:fs";
|
|
15
|
+
import { mkdtempSync, rmSync } from "node:fs";
|
|
16
|
+
import * as os from "node:os";
|
|
17
|
+
import path from "node:path";
|
|
18
|
+
import { fileURLToPath } from "node:url";
|
|
19
|
+
import { test } from "node:test";
|
|
20
|
+
import { registerGenome } from "../core/genome.js";
|
|
21
|
+
import { prepareToyGenome } from "../bench/toy.js";
|
|
22
|
+
const CLI = fileURLToPath(new URL("../../dist/cli.js", import.meta.url));
|
|
23
|
+
const LIT_VAR = "ABATHUR_SEAM_GATES_REPO";
|
|
24
|
+
async function fixture(t, label) {
|
|
25
|
+
const root = mkdtempSync(path.join(os.tmpdir(), "seam-gates-"));
|
|
26
|
+
t.after(() => rmSync(root, { recursive: true, force: true }));
|
|
27
|
+
const configDir = path.join(root, "config");
|
|
28
|
+
mkdirSync(configDir, { recursive: true });
|
|
29
|
+
writeFileSync(path.join(configDir, "config.jsonc"), '{ "opencodeBin": null }\n', "utf8");
|
|
30
|
+
const repo = await prepareToyGenome(path.join(root, "genome"));
|
|
31
|
+
const doc = JSON.parse(readFileSync(path.join(repo, "genome.jsonc"), "utf8"));
|
|
32
|
+
doc["label"] = label;
|
|
33
|
+
doc["repoPath"] = `\${${LIT_VAR}}`;
|
|
34
|
+
const specFile = path.join(root, "private.jsonc");
|
|
35
|
+
writeFileSync(specFile, `${JSON.stringify(doc, null, 2)}\n`, "utf8");
|
|
36
|
+
process.env[LIT_VAR] = repo;
|
|
37
|
+
try {
|
|
38
|
+
registerGenome(configDir, specFile);
|
|
39
|
+
}
|
|
40
|
+
finally {
|
|
41
|
+
delete process.env[LIT_VAR];
|
|
42
|
+
}
|
|
43
|
+
return { root, configDir, repo, label };
|
|
44
|
+
}
|
|
45
|
+
function cli(f, args, env) {
|
|
46
|
+
return spawnSync(process.execPath, [CLI, ...args], {
|
|
47
|
+
env: { ...process.env, ABATHUR_CONFIG: path.join(f.configDir, "config.jsonc"), ...env },
|
|
48
|
+
encoding: "utf8",
|
|
49
|
+
timeout: 60_000,
|
|
50
|
+
});
|
|
51
|
+
}
|
|
52
|
+
function assertNoLiteralOrEnoent(r) {
|
|
53
|
+
const blob = `${r.stdout}\n${r.stderr}`;
|
|
54
|
+
assert.ok(!blob.includes(`\${${LIT_VAR}}`), `unresolved literal leaked into output:\n${blob}`);
|
|
55
|
+
assert.ok(!blob.includes("ENOENT"), `misleading ENOENT leaked into output:\n${blob}`);
|
|
56
|
+
}
|
|
57
|
+
test("promote with env-literal repoPath: exported ⇒ blocked naming the RESOLVED repo path", async (t) => {
|
|
58
|
+
const f = await fixture(t, "seam-gate-promote");
|
|
59
|
+
const r = cli(f, ["promote", f.label, "g-nonexistent"], { [LIT_VAR]: f.repo });
|
|
60
|
+
assert.equal(r.status, 1, `stdout: ${r.stdout}\nstderr: ${r.stderr}`);
|
|
61
|
+
assertNoLiteralOrEnoent(r);
|
|
62
|
+
assert.ok(r.stderr.includes(f.repo), `refusal must name the resolved repo path so the operator can debug:\n${r.stderr}`);
|
|
63
|
+
});
|
|
64
|
+
test("tombstone with env-literal repoPath: exported ⇒ blocked, no literal leak", async (t) => {
|
|
65
|
+
const f = await fixture(t, "seam-gate-tombstone");
|
|
66
|
+
const r = cli(f, ["tombstone", f.label, "g-nonexistent", "--reason", "seam test"], { [LIT_VAR]: f.repo });
|
|
67
|
+
assert.equal(r.status, 1, `stdout: ${r.stdout}\nstderr: ${r.stderr}`);
|
|
68
|
+
assertNoLiteralOrEnoent(r);
|
|
69
|
+
});
|
|
70
|
+
test("promote with env-literal repoPath: unset ⇒ clean exit 2 naming the variable", async (t) => {
|
|
71
|
+
const f = await fixture(t, "seam-gate-promote-unset");
|
|
72
|
+
const env = {};
|
|
73
|
+
env[LIT_VAR] = undefined;
|
|
74
|
+
const r = cli(f, ["promote", f.label, "g-nonexistent"], env);
|
|
75
|
+
assert.equal(r.status, 2, `stdout: ${r.stdout}\nstderr: ${r.stderr}`);
|
|
76
|
+
assert.ok(r.stderr.includes(LIT_VAR), `exit-2 message must name the env var:\n${r.stderr}`);
|
|
77
|
+
assert.ok(!r.stderr.includes("ENOENT"), `misleading ENOENT leaked into output:\n${r.stderr}`);
|
|
78
|
+
});
|
|
@@ -0,0 +1,329 @@
|
|
|
1
|
+
// Acceptance-semantics evaluate() tests (2026-09-24 redesign). Thresholds
|
|
2
|
+
// mirror the LIVE historian genome: halfWidth 0.15, minEffect 0.1. Bank σ
|
|
3
|
+
// values are the real campaign-10 gate-time bank (§2.3 of
|
|
4
|
+
// .omo/evidence/ACCEPTANCE-SEMANTICS-DESIGN.md); expected probabilities are
|
|
5
|
+
// from the independent python design-time replication (math.erf), asserted to
|
|
6
|
+
// 2e-5 — well above the A&S normalCdf error (< 1.5e-7) and far below any
|
|
7
|
+
// decision-relevant margin.
|
|
8
|
+
//
|
|
9
|
+
// Coverage law: every new branch (acceptance band, regression veto, point-mass
|
|
10
|
+
// se=0 branches, per-unit legacy fallback, aggregate P-gate, pool-incomplete
|
|
11
|
+
// legacy point-gain, q_pair tightening) plus the fail-closed law RE-PROVEN
|
|
12
|
+
// WITH A BANK PRESENT (budget/one-sided/exclusion cannot be bypassed).
|
|
13
|
+
import assert from "node:assert/strict";
|
|
14
|
+
import { describe, it } from "node:test";
|
|
15
|
+
import { ACCEPT_Q, aggregateScore, bonferroniQ, evaluate, normalCdf, studentTQuantile, } from "../core/stats.js";
|
|
16
|
+
import { BANK_QUANTUM } from "../core/evolve/score-bank.js";
|
|
17
|
+
const HIST_STATS = { halfWidth: 0.15, minEffect: 0.1, nReps: { initial: 2, max: 3 } };
|
|
18
|
+
const CAPS = { maxCandidates: 1, maxModelCalls: 96, maxTokens: 25_000_000, maxWallS: 16_200 };
|
|
19
|
+
const IDLE = { candidates: 0, modelCalls: 0, tokens: 1_000_000, wallS: 11_154 };
|
|
20
|
+
/** Gate-time c10 bank: σ values from the real archive pool (python replication). */
|
|
21
|
+
const HIST_BANK = new Map([
|
|
22
|
+
["scenario-13", { sigma: 0.0170, df: 12 }],
|
|
23
|
+
["scenario-14", { sigma: 0.0176, df: 11 }],
|
|
24
|
+
["scenario-15", { sigma: 0.0884, df: 4 }],
|
|
25
|
+
["scenario-16", { sigma: 0.1053, df: 4 }],
|
|
26
|
+
["scenario-18", { sigma: 0.1113, df: 4 }],
|
|
27
|
+
["scenario-19", { sigma: 0.0884, df: 2 }],
|
|
28
|
+
]);
|
|
29
|
+
function units(specs) {
|
|
30
|
+
return specs.map(([unitId, split, scores]) => ({ unitId, split, scores }));
|
|
31
|
+
}
|
|
32
|
+
const C10_INCUMBENT = units([
|
|
33
|
+
["scenario-19", "train", [0.5, 0.5]],
|
|
34
|
+
["scenario-13", "val", [1, 1]],
|
|
35
|
+
["scenario-14", "val", [1, 1]],
|
|
36
|
+
["scenario-15", "val", [1, 1]],
|
|
37
|
+
["scenario-16", "val", [1, 1]],
|
|
38
|
+
["scenario-18", "val", [0.75, 1]],
|
|
39
|
+
]);
|
|
40
|
+
function c10Candidate(s19, s16 = [0.75, 1], s18 = [1, 0.75]) {
|
|
41
|
+
return {
|
|
42
|
+
runId: "backtest",
|
|
43
|
+
counters: IDLE,
|
|
44
|
+
units: units([
|
|
45
|
+
["scenario-19", "train", s19],
|
|
46
|
+
["scenario-13", "val", [1, 1]],
|
|
47
|
+
["scenario-14", "val", [1, 1]],
|
|
48
|
+
["scenario-15", "val", [1, 1]],
|
|
49
|
+
["scenario-16", "val", s16],
|
|
50
|
+
["scenario-18", "val", s18],
|
|
51
|
+
]),
|
|
52
|
+
};
|
|
53
|
+
}
|
|
54
|
+
function inc(unitsList) {
|
|
55
|
+
return { units: unitsList };
|
|
56
|
+
}
|
|
57
|
+
function runAccept(candidate, incumbentUnits = C10_INCUMBENT, bank = HIST_BANK, nPairs = 1) {
|
|
58
|
+
return evaluate({ candidate, incumbent: inc(incumbentUnits), stats: HIST_STATS, budgetCaps: CAPS, nPairs, ...(bank === undefined ? {} : { bank }) });
|
|
59
|
+
}
|
|
60
|
+
const close = (actual, expected, msg) => {
|
|
61
|
+
assert.ok(Math.abs(actual - expected) < 2e-5, `${msg ?? ""} actual=${String(actual)} expected=${String(expected)}`);
|
|
62
|
+
};
|
|
63
|
+
// ---------------------------------------------------------------------------
|
|
64
|
+
// The four mandated backtests, as unit-pinned replicas of the real replays.
|
|
65
|
+
// ---------------------------------------------------------------------------
|
|
66
|
+
describe("acceptance gate — campaign replays", () => {
|
|
67
|
+
it("C10 REAL stays CULLED: gain floor AND P=0.5000; the s16 noise dip is TOLERATED, not CI-fatal", () => {
|
|
68
|
+
const v = runAccept(c10Candidate([0.5, 0.5]));
|
|
69
|
+
assert.equal(v.verdict, "culled");
|
|
70
|
+
assert.equal(v.exitCode, 1);
|
|
71
|
+
assert.equal(v.gain, 0);
|
|
72
|
+
assert.ok(v.acceptance !== undefined);
|
|
73
|
+
close(v.acceptance.pShift ?? Number.NaN, 0.5, "pShift");
|
|
74
|
+
assert.ok(v.failures.some((f) => f === "gain 0.0000 < minEffect 0.1000"));
|
|
75
|
+
assert.ok(v.failures.some((f) => f.includes("P(gain shift > 0) = 0.5000 < q 0.9000")));
|
|
76
|
+
assert.ok(!v.failures.some((f) => f.includes("CI half-width") || f.includes("acceptance band")), v.failures.join("\n"));
|
|
77
|
+
const s16 = v.unitComparisons.find((u) => u.unitId === "scenario-16");
|
|
78
|
+
assert.ok(s16 !== undefined);
|
|
79
|
+
assert.equal(s16.passes, true, "one 0.125 step inside the 0.15 guard tolerance is noise, not regression");
|
|
80
|
+
assert.equal(s16.ciPasses, true);
|
|
81
|
+
const report16 = v.acceptance.units.find((u) => u.unitId === "scenario-16");
|
|
82
|
+
assert.ok(report16 !== undefined);
|
|
83
|
+
close(report16.pRegression, 0.40617, "P(regress) for the dip");
|
|
84
|
+
});
|
|
85
|
+
it("C9 REAL stays CULLED: negative point gain and P(Δ>0)≈0.046", () => {
|
|
86
|
+
const c9Inc = units([
|
|
87
|
+
["scenario-18", "train", [0.875, 1]],
|
|
88
|
+
["scenario-13", "val", [1, 1]],
|
|
89
|
+
["scenario-14", "val", [1, 1]],
|
|
90
|
+
["scenario-15", "val", [1, 1]],
|
|
91
|
+
["scenario-16", "val", [1, 1]],
|
|
92
|
+
]);
|
|
93
|
+
const c9Cand = units([
|
|
94
|
+
["scenario-18", "train", [0.75, 0.75]],
|
|
95
|
+
["scenario-13", "val", [1, 1]],
|
|
96
|
+
["scenario-14", "val", [1, 1]],
|
|
97
|
+
["scenario-15", "val", [1, 1]],
|
|
98
|
+
["scenario-16", "val", [1, 0.75]],
|
|
99
|
+
]);
|
|
100
|
+
const v = runAccept({ runId: "c9", counters: IDLE, units: c9Cand }, c9Inc);
|
|
101
|
+
assert.equal(v.verdict, "culled");
|
|
102
|
+
close(v.gain ?? Number.NaN, -0.1875);
|
|
103
|
+
const se = 0.1113;
|
|
104
|
+
close(v.acceptance?.pShift ?? Number.NaN, normalCdf(-0.1875 / se), "pShift");
|
|
105
|
+
assert.ok((v.acceptance?.pShift ?? 1) < 0.05);
|
|
106
|
+
});
|
|
107
|
+
it("SYNTH A (+0.25 true shift): NOMINATED at P≈0.9977 with the same guards (incl. the s16 dip)", () => {
|
|
108
|
+
const v = runAccept(c10Candidate([0.75, 0.75]));
|
|
109
|
+
assert.equal(v.verdict, "nominated");
|
|
110
|
+
assert.equal(v.exitCode, 0);
|
|
111
|
+
close(v.gain ?? Number.NaN, 0.25);
|
|
112
|
+
close(v.acceptance?.pShift ?? Number.NaN, 0.99766, "pShift");
|
|
113
|
+
assert.equal(v.acceptance?.q, ACCEPT_Q);
|
|
114
|
+
assert.ok(v.failures.length === 0, v.failures.join("\n"));
|
|
115
|
+
});
|
|
116
|
+
it("SYNTH B (+0.05 band-mean draw): CULLED twice — gain 0.0625 < 0.1 and P≈0.7602", () => {
|
|
117
|
+
const v = runAccept(c10Candidate([0.625, 0.5]));
|
|
118
|
+
assert.equal(v.verdict, "culled");
|
|
119
|
+
assert.ok(v.failures.some((f) => f.includes("gain 0.0625 < minEffect 0.1000")));
|
|
120
|
+
assert.ok(v.failures.some((f) => f.includes("P(gain shift > 0) = 0.7602")));
|
|
121
|
+
});
|
|
122
|
+
it("documented leak SYNTH B′ (+0.05 lucky [0.625,0.625]): nominates at P≈0.9214 alone, culled at nPairs=2 (q=0.95)", () => {
|
|
123
|
+
const solo = runAccept(c10Candidate([0.625, 0.625]));
|
|
124
|
+
assert.equal(solo.verdict, "nominated");
|
|
125
|
+
close(solo.acceptance?.pShift ?? Number.NaN, 0.92132, "pShift");
|
|
126
|
+
const pair = runAccept(c10Candidate([0.625, 0.625]), C10_INCUMBENT, HIST_BANK, 2);
|
|
127
|
+
assert.equal(pair.verdict, "culled");
|
|
128
|
+
assert.equal(pair.acceptance?.q, 0.95);
|
|
129
|
+
assert.ok(pair.failures.some((f) => f.includes("= 0.9213 < q 0.9500")), pair.failures.join("\n"));
|
|
130
|
+
});
|
|
131
|
+
});
|
|
132
|
+
// ---------------------------------------------------------------------------
|
|
133
|
+
// Branch pins: each failure line and fallback edge of the new code.
|
|
134
|
+
// ---------------------------------------------------------------------------
|
|
135
|
+
describe("acceptance gate — branches", () => {
|
|
136
|
+
it("gain ≥ minEffect with a loud point but weak P (σ=0.1113): P=0.8694 line fires, gain line absent", () => {
|
|
137
|
+
const bank = new Map([["t", { sigma: 0.1113, df: 4 }]]);
|
|
138
|
+
const v = evaluate({
|
|
139
|
+
candidate: { runId: "c", counters: IDLE, units: units([["t", "train", [0.625, 0.625]]]) },
|
|
140
|
+
incumbent: inc(units([["t", "train", [0.5, 0.5]]])),
|
|
141
|
+
stats: { ...HIST_STATS, halfWidth: 1 },
|
|
142
|
+
budgetCaps: CAPS,
|
|
143
|
+
nPairs: 1,
|
|
144
|
+
bank,
|
|
145
|
+
});
|
|
146
|
+
assert.equal(v.verdict, "culled");
|
|
147
|
+
assert.equal(v.failures.length, 1, v.failures.join("\n"));
|
|
148
|
+
assert.ok(v.failures[0]?.includes("0.8693 < q 0.9000"), v.failures[0]);
|
|
149
|
+
});
|
|
150
|
+
it("point gain under floor with a confident P: only the legacy gain line fires", () => {
|
|
151
|
+
const bank = new Map([["t", { sigma: 0.017, df: 12 }]]);
|
|
152
|
+
const v = evaluate({
|
|
153
|
+
candidate: { runId: "c", counters: IDLE, units: units([["t", "train", [0.55, 0.6]]]) },
|
|
154
|
+
incumbent: inc(units([["t", "train", [0.5, 0.5]]])),
|
|
155
|
+
stats: { ...HIST_STATS, halfWidth: 1 },
|
|
156
|
+
budgetCaps: CAPS,
|
|
157
|
+
nPairs: 1,
|
|
158
|
+
bank,
|
|
159
|
+
});
|
|
160
|
+
assert.equal(v.verdict, "culled");
|
|
161
|
+
assert.deepEqual(v.failures, ["gain 0.0750 < minEffect 0.1000"]);
|
|
162
|
+
assert.ok((v.acceptance?.pShift ?? 0) > 0.99);
|
|
163
|
+
});
|
|
164
|
+
it("hard guard regression vetoes: P(Δ ≤ −0.15)≈1 on a quiet bank unit", () => {
|
|
165
|
+
const bank = new Map([
|
|
166
|
+
["v", { sigma: 0.02, df: 8 }],
|
|
167
|
+
["t", { sigma: 0.017, df: 12 }],
|
|
168
|
+
]);
|
|
169
|
+
const v = evaluate({
|
|
170
|
+
candidate: { runId: "c", counters: IDLE, units: units([["v", "val", [0.75, 0.75]], ["t", "train", [1, 1]]]) },
|
|
171
|
+
incumbent: inc(units([["v", "val", [1, 1]], ["t", "train", [1, 1]]])),
|
|
172
|
+
stats: HIST_STATS,
|
|
173
|
+
budgetCaps: CAPS,
|
|
174
|
+
nPairs: 1,
|
|
175
|
+
bank,
|
|
176
|
+
});
|
|
177
|
+
assert.equal(v.verdict, "culled");
|
|
178
|
+
assert.ok(v.failures.some((f) => f.startsWith("regression on unit 'v'") && f.includes("≥ q 0.9000")), v.failures.join("\n"));
|
|
179
|
+
assert.equal(v.unitComparisons.find((u) => u.unitId === "v")?.passes, false);
|
|
180
|
+
});
|
|
181
|
+
it("acceptance band breach: big bank σ ⇒ band 0.2718 > halfWidth culls even with zero delta", () => {
|
|
182
|
+
const bank = new Map([["v", { sigma: 0.3, df: 10 }], ["t", { sigma: 0.017, df: 12 }]]);
|
|
183
|
+
const v = evaluate({
|
|
184
|
+
candidate: { runId: "c", counters: IDLE, units: units([["v", "val", [1, 1]], ["t", "train", [1, 1]]]) },
|
|
185
|
+
incumbent: inc(units([["v", "val", [1, 1]], ["t", "train", [1, 1]]])),
|
|
186
|
+
stats: HIST_STATS,
|
|
187
|
+
budgetCaps: CAPS,
|
|
188
|
+
nPairs: 1,
|
|
189
|
+
bank,
|
|
190
|
+
});
|
|
191
|
+
assert.equal(v.verdict, "culled");
|
|
192
|
+
assert.ok(v.failures.some((f) => f.includes("acceptance band 0.2719 > halfWidth 0.1500 on unit 'v'")), v.failures.join("\n"));
|
|
193
|
+
});
|
|
194
|
+
it("banked val unit with NO history for the bank runs the exact legacy t-CI line — the 1.5883 absurdity survives only on this fallback path", () => {
|
|
195
|
+
const bank = new Map([["scenario-13", { sigma: 0.017, df: 12 }]]);
|
|
196
|
+
const v = runAccept(c10Candidate([0.5, 0.5], [0.75, 1]), C10_INCUMBENT, bank);
|
|
197
|
+
const expectedHalfWidth = studentTQuantile(0.05, 1) * 0.125;
|
|
198
|
+
assert.ok(Math.abs(expectedHalfWidth - 1.5883) < 1e-4, "design-time arithmetic");
|
|
199
|
+
assert.ok(v.failures.some((f) => f === `CI half-width ${expectedHalfWidth.toFixed(4)} > halfWidth 0.1500 on unit 'scenario-16'`), v.failures.join("\n"));
|
|
200
|
+
assert.ok(v.acceptance !== undefined);
|
|
201
|
+
assert.equal(v.acceptance.units.find((u) => u.unitId === "scenario-16"), undefined);
|
|
202
|
+
});
|
|
203
|
+
it("partial bank: train pool unpriced ⇒ aggregate falls back to the legacy point-gain gate; nomination still requires gain ≥ minEffect", () => {
|
|
204
|
+
const partial = new Map([...HIST_BANK].filter(([id]) => id !== "scenario-19"));
|
|
205
|
+
const weak = runAccept(c10Candidate([0.625, 0.5]), C10_INCUMBENT, partial);
|
|
206
|
+
assert.equal(weak.verdict, "culled");
|
|
207
|
+
assert.ok(weak.failures.some((f) => f.includes("gain 0.0625 < minEffect 0.1000")));
|
|
208
|
+
assert.equal(weak.acceptance?.pShift, null);
|
|
209
|
+
assert.equal(weak.acceptance?.gainSe, null);
|
|
210
|
+
const strong = runAccept(c10Candidate([0.75, 0.75]), C10_INCUMBENT, partial);
|
|
211
|
+
assert.equal(strong.verdict, "nominated", "cold-start gate-time bank (no s19 rows yet) still accepts the +0.25 class via the legacy point gate");
|
|
212
|
+
assert.equal(strong.acceptance?.pShift, null);
|
|
213
|
+
});
|
|
214
|
+
});
|
|
215
|
+
// ---------------------------------------------------------------------------
|
|
216
|
+
// Fail-closed law RE-PROVEN with a bank present — the bank never bypasses it.
|
|
217
|
+
// ---------------------------------------------------------------------------
|
|
218
|
+
describe("fail-closed law under an acceptance bank", () => {
|
|
219
|
+
it("budget exhaustion ⇒ inconclusive (gain null, no comparisons, no acceptance math)", () => {
|
|
220
|
+
const v = runAccept({ ...c10Candidate([0.75, 0.75]), counters: { ...IDLE, wallS: 16_200 } });
|
|
221
|
+
assert.equal(v.verdict, "inconclusive");
|
|
222
|
+
assert.equal(v.exitCode, 2);
|
|
223
|
+
assert.equal(v.gain, null);
|
|
224
|
+
assert.deepEqual(v.unitComparisons, []);
|
|
225
|
+
assert.ok(v.failures.some((f) => f.includes("budget exhausted: wallS")));
|
|
226
|
+
});
|
|
227
|
+
it("one-sided n<2 ⇒ indeterminate even when the bank knows the unit", () => {
|
|
228
|
+
const oneSided = units([
|
|
229
|
+
["scenario-19", "train", [0.75, 0.75]],
|
|
230
|
+
["scenario-13", "val", [1, 1]],
|
|
231
|
+
["scenario-14", "val", [1, 1]],
|
|
232
|
+
["scenario-15", "val", [1, 1]],
|
|
233
|
+
["scenario-16", "val", [1, 1]],
|
|
234
|
+
["scenario-18", "val", [1]],
|
|
235
|
+
]);
|
|
236
|
+
const v = runAccept({ runId: "c", counters: IDLE, units: oneSided });
|
|
237
|
+
assert.equal(v.verdict, "indeterminate");
|
|
238
|
+
assert.equal(v.exitCode, 1);
|
|
239
|
+
assert.equal(v.gain, null);
|
|
240
|
+
assert.ok(v.failures.some((f) => f.includes("scenario-18") && f.includes("variance undefined")));
|
|
241
|
+
});
|
|
242
|
+
it("both-unmeasured ⇒ symmetric exclusion: dropped from comparisons, acceptance reports and the gain pool alike", () => {
|
|
243
|
+
const withGhost = units([["ghost", "val", []], ...C10_INCUMBENT.map((u) => [u.unitId, u.split, [...u.scores]])]);
|
|
244
|
+
const cand = c10Candidate([0.75, 0.75]);
|
|
245
|
+
const v = runAccept({ ...cand, units: [{ unitId: "ghost", split: "val", scores: [] }, ...cand.units] }, withGhost);
|
|
246
|
+
assert.equal(v.verdict, "nominated");
|
|
247
|
+
assert.ok(v.unitComparisons.every((u) => u.unitId !== "ghost"));
|
|
248
|
+
assert.ok(v.acceptance !== undefined && v.acceptance.units.every((u) => u.unitId !== "ghost"));
|
|
249
|
+
close(aggregateScore(withGhost.filter((u) => u.unitId !== "ghost")), aggregateScore(C10_INCUMBENT));
|
|
250
|
+
});
|
|
251
|
+
it("no reroll: evaluate is deterministic — identical inputs produce identical verdicts and acceptance numbers", () => {
|
|
252
|
+
const a = runAccept(c10Candidate([0.75, 0.75]));
|
|
253
|
+
const b = runAccept(c10Candidate([0.75, 0.75]));
|
|
254
|
+
assert.deepEqual(a, b);
|
|
255
|
+
});
|
|
256
|
+
it("absent/null bank reproduces the legacy gate verbatim; the whole verdict object is unchanged", () => {
|
|
257
|
+
const legacy = evaluate({ candidate: c10Candidate([0.5, 0.5]), incumbent: inc(C10_INCUMBENT), stats: HIST_STATS, budgetCaps: CAPS, nPairs: 1 });
|
|
258
|
+
const nulled = runAccept(c10Candidate([0.5, 0.5]), C10_INCUMBENT, null);
|
|
259
|
+
assert.deepEqual(nulled, legacy);
|
|
260
|
+
assert.equal(legacy.acceptance, undefined);
|
|
261
|
+
assert.equal(legacy.verdict, "culled");
|
|
262
|
+
assert.ok(legacy.failures.some((f) => f.includes("CI half-width 1.5883")), "the campaign-10 legacy cull carried the absurd half-width line");
|
|
263
|
+
const banked = runAccept(c10Candidate([0.5, 0.5]));
|
|
264
|
+
assert.equal(banked.verdict, "culled");
|
|
265
|
+
assert.ok(!banked.failures.some((f) => f.includes("CI half-width")), "acceptance path prices the dip instead of tripping on df=1 t");
|
|
266
|
+
});
|
|
267
|
+
it("point mass σ=0: se=0 branch — exact tie gives P=0 (no nomination on demonstrated-identical arms), a 0.25 guard drop vetoes at P=1, a 0.125 dip inside tolerance passes", () => {
|
|
268
|
+
const bank = new Map([
|
|
269
|
+
["v", { sigma: 0, df: 10 }],
|
|
270
|
+
["t", { sigma: 0, df: 10 }],
|
|
271
|
+
]);
|
|
272
|
+
const tie = evaluate({
|
|
273
|
+
candidate: { runId: "c", counters: IDLE, units: units([["v", "val", [1, 1]], ["t", "train", [1, 1]]]) },
|
|
274
|
+
incumbent: inc(units([["v", "val", [1, 1]], ["t", "train", [1, 1]]])),
|
|
275
|
+
stats: HIST_STATS,
|
|
276
|
+
budgetCaps: CAPS,
|
|
277
|
+
nPairs: 1,
|
|
278
|
+
bank,
|
|
279
|
+
});
|
|
280
|
+
assert.equal(tie.verdict, "culled");
|
|
281
|
+
assert.equal(tie.acceptance?.pShift, 0);
|
|
282
|
+
assert.equal(tie.acceptance?.gainSe, 0);
|
|
283
|
+
const hard = evaluate({
|
|
284
|
+
candidate: { runId: "c", counters: IDLE, units: units([["v", "val", [0.75, 0.75]], ["t", "train", [1.3, 1.3]]]) },
|
|
285
|
+
incumbent: inc(units([["v", "val", [1, 1]], ["t", "train", [1, 1]]])),
|
|
286
|
+
stats: HIST_STATS,
|
|
287
|
+
budgetCaps: CAPS,
|
|
288
|
+
nPairs: 1,
|
|
289
|
+
bank,
|
|
290
|
+
});
|
|
291
|
+
assert.equal(hard.verdict, "culled");
|
|
292
|
+
assert.ok(hard.failures.some((f) => f.startsWith("regression on unit 'v'") && f.includes("= 1.0000")));
|
|
293
|
+
const dip = evaluate({
|
|
294
|
+
candidate: { runId: "c", counters: IDLE, units: units([["v", "val", [0.875, 0.875]], ["t", "train", [1.3, 1.3]]]) },
|
|
295
|
+
incumbent: inc(units([["v", "val", [1, 1]], ["t", "train", [1, 1]]])),
|
|
296
|
+
stats: HIST_STATS,
|
|
297
|
+
budgetCaps: CAPS,
|
|
298
|
+
nPairs: 1,
|
|
299
|
+
bank,
|
|
300
|
+
});
|
|
301
|
+
assert.equal(dip.verdict, "nominated", "−0.125 sits inside the 0.15 guard tolerance — the anchor semantics");
|
|
302
|
+
});
|
|
303
|
+
});
|
|
304
|
+
describe("acceptance constants + normal math", () => {
|
|
305
|
+
it("ACCEPT_Q is 0.9 and bonferroniQ tightens it per pair, capped", () => {
|
|
306
|
+
assert.equal(ACCEPT_Q, 0.9);
|
|
307
|
+
assert.equal(bonferroniQ(1), 0.9);
|
|
308
|
+
close(bonferroniQ(2), 0.95);
|
|
309
|
+
close(bonferroniQ(10), 0.99);
|
|
310
|
+
assert.equal(bonferroniQ(10_000), 0.9999);
|
|
311
|
+
assert.throws(() => bonferroniQ(0), RangeError);
|
|
312
|
+
assert.throws(() => bonferroniQ(1.5), RangeError);
|
|
313
|
+
});
|
|
314
|
+
it("normalCdf matches the tables and the quantile inverts it", () => {
|
|
315
|
+
close(normalCdf(0), 0.5);
|
|
316
|
+
close(normalCdf(1.2815515655446004), 0.9);
|
|
317
|
+
close(normalCdf(-1.6448536269514722), 0.05);
|
|
318
|
+
assert.equal(normalCdf(Number.POSITIVE_INFINITY), 1);
|
|
319
|
+
assert.equal(normalCdf(Number.NEGATIVE_INFINITY), 0);
|
|
320
|
+
assert.ok(Number.isNaN(normalCdf(Number.NaN)));
|
|
321
|
+
for (const x of [-3, -1.28, 0, 0.7, 2.83]) {
|
|
322
|
+
assert.ok(Math.abs(normalCdf(x) - (x < 0 ? 1 - normalCdf(-x) : normalCdf(x))) < 1e-12, "symmetry sanity");
|
|
323
|
+
}
|
|
324
|
+
});
|
|
325
|
+
it("the bank quantum is 0.0884 — one flipped 1/8-grid item between two reps", () => {
|
|
326
|
+
close(BANK_QUANTUM, 0.0884);
|
|
327
|
+
assert.ok(Math.abs(BANK_QUANTUM - 0.125 / Math.SQRT2) < 1e-12);
|
|
328
|
+
});
|
|
329
|
+
});
|