@tachikomagundam/abathur 0.2.4 → 0.2.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/config/genomes/historian.example.jsonc +1 -1
- package/dist/commands/genome.js +2 -2
- package/dist/commands/graft.js +1 -1
- package/dist/commands/run.js +1 -1
- package/dist/commands/self-eval.js +1 -1
- package/dist/commands/tombstone.js +2 -1
- package/dist/core/evolve/run-bench.js +4 -3
- package/dist/core/evolve/run-loop.js +24 -4
- package/dist/core/evolve/run-plan.js +7 -3
- package/dist/core/evolve/score-bank.js +214 -0
- package/dist/core/graft-rebench.js +3 -0
- package/dist/core/ledger.js +6 -1
- package/dist/core/promote.js +2 -1
- package/dist/core/spec.js +5 -1
- package/dist/core/stats-math.js +39 -0
- package/dist/core/stats.js +125 -33
- package/dist/test/historian-grader-integrity.test.js +1494 -2
- package/dist/test/historian-grader-io.test.js +4 -0
- package/dist/test/historian-run-scenario.test.js +42 -2
- package/dist/test/repopath-seams.test.js +54 -0
- package/dist/test/score-bank.test.js +235 -0
- package/dist/test/seam-gates.test.js +78 -0
- package/dist/test/selfmode-literal.test.js +153 -0
- package/dist/test/stats-acceptance.test.js +329 -0
- package/dist/test/stats.test.js +18 -0
- package/graders/historian/grader-core.d.mts +107 -2
- package/graders/historian/grader-core.mjs +1046 -3
- package/graders/historian/grader.mjs +21 -2
- package/graders/historian/judge-poststage-19.mjs +212 -0
- package/graders/historian/judge-poststage.mjs +195 -0
- package/graders/historian/mutate.sh +59 -19
- package/graders/historian/run-scenario-17.sh +34 -0
- package/graders/historian/run-scenario-19.sh +43 -0
- package/graders/historian/run-scenario.sh +24 -1
- package/package.json +1 -1
- package/plugin/abathur.ts +1 -1
package/dist/core/stats.js
CHANGED
|
@@ -26,15 +26,39 @@
|
|
|
26
26
|
// - Pareto over (trainScore, tokens, wallS); budget-truncated generations
|
|
27
27
|
// (complete:false) never enter the set.
|
|
28
28
|
//
|
|
29
|
+
// ACCEPTANCE SEMANTICS (2026-09-24 redesign; design doc lives in the target
|
|
30
|
+
// genome repo's evidence folder, not shipped here): when EvaluateInput.bank
|
|
31
|
+
// carries the historical per-unit variance bank, precision/regression checks
|
|
32
|
+
// price NOISE
|
|
33
|
+
// with bank σ instead of per-arm t(df=n−1) CIs (the campaign-9/10 resolution
|
|
34
|
+
// floor: t(0.05,1)=12.706 turned a 2-point sample into a 1.5883 half-width).
|
|
35
|
+
// A val unit with bank σ clears when P(Δ ≤ −halfWidth) < q_pair AND the
|
|
36
|
+
// acceptance band Z(q)·σ/√n_candidate ≤ halfWidth; the aggregate additionally
|
|
37
|
+
// requires P(gain shift > 0) = Φ(gain/SE) ≥ q_pair, q_pair = 1−(1−ACCEPT_Q)/nPairs.
|
|
38
|
+
// EVERY fail-closed rule ABOVE the bank path is untouched and can never be
|
|
39
|
+
// bypassed by it: budget⇒inconclusive, one-sided n<2⇒indeterminate,
|
|
40
|
+
// both-unmeasured⇒symmetric exclusion. A unit the bank cannot price falls back
|
|
41
|
+
// to the exact legacy t-CI line; a null/absent bank reproduces the pre-2026-09-24
|
|
42
|
+
// gate verbatim. minEffect 0.1 / halfWidth 0.15 point semantics stand.
|
|
43
|
+
//
|
|
29
44
|
// The inverse Student-t lives in stats-math.ts and the Pareto ranking in
|
|
30
45
|
// stats-pareto.ts; both are re-exported here so the public surface of this
|
|
31
46
|
// module — the one todos 9/10/12 consume — is unchanged. Accuracy: |t − table|
|
|
32
47
|
// < 1e-3 for the t-distribution rows checked in stats.test.ts (df=1..30, α=0.05).
|
|
33
48
|
import { EXIT_BLOCKED, EXIT_CANNOT_ANSWER, EXIT_OK } from "../exit.js";
|
|
34
49
|
/** Per-unit replicate scores (train and val units alike). */
|
|
35
|
-
import { studentTQuantile } from "./stats-math.js";
|
|
50
|
+
import { studentTQuantile, normalCdf, normalQuantile } from "./stats-math.js";
|
|
36
51
|
import { seededRandom, paretoFrontier } from "./stats-pareto.js";
|
|
37
|
-
export { studentTQuantile, seededRandom, paretoFrontier };
|
|
52
|
+
export { studentTQuantile, normalCdf, normalQuantile, seededRandom, paretoFrontier };
|
|
53
|
+
/** Nomination acceptance probability (per finalist pair after Bonferroni). */
|
|
54
|
+
export const ACCEPT_Q = 0.9;
|
|
55
|
+
/** q_pair = 1 − (1 − ACCEPT_Q)/nPairs — Bonferroni on the acceptance ERROR rate. */
|
|
56
|
+
export function bonferroniQ(nPairs) {
|
|
57
|
+
if (!Number.isInteger(nPairs) || nPairs < 1) {
|
|
58
|
+
throw new RangeError(`bonferroniQ: nPairs=${String(nPairs)} must be an integer >= 1`);
|
|
59
|
+
}
|
|
60
|
+
return Math.min(0.9999, 1 - (1 - ACCEPT_Q) / nPairs);
|
|
61
|
+
}
|
|
38
62
|
/** Family-wise alpha across finalist pairs (bonferroniAlpha divides it). */
|
|
39
63
|
export const FAMILY_ALPHA = 0.05;
|
|
40
64
|
/** Verdict → process exit code. inconclusive = 2 (cannot answer), never 0. */
|
|
@@ -132,53 +156,121 @@ export function evaluate(input) {
|
|
|
132
156
|
const candByUnit = new Map(candidate.units.map((u) => [u.unitId, u]));
|
|
133
157
|
const incByUnit = new Map(incumbent.units.map((u) => [u.unitId, u]));
|
|
134
158
|
const failures = [];
|
|
135
|
-
// 2)
|
|
136
|
-
//
|
|
137
|
-
|
|
138
|
-
|
|
139
|
-
|
|
159
|
+
// 2) Cross-arm replicate rule. A unit measured cleanly (n>=2) on ONE arm but
|
|
160
|
+
// not on the other leaves that comparison's variance undefined ⇒
|
|
161
|
+
// indeterminate, never a silent cull and never a nomination (campaign-6:
|
|
162
|
+
// the candidate's train unit scored while the incumbent's had timed out;
|
|
163
|
+
// the old one-sided drop shifted the incumbent aggregate to the val
|
|
164
|
+
// fallback and minted a bogus negative gain). A unit unmeasured on BOTH
|
|
165
|
+
// arms is excluded symmetrically — every aggregate and comparison then
|
|
166
|
+
// runs on a pool both arms share replicate-for-replicate.
|
|
167
|
+
const excluded = new Set();
|
|
168
|
+
for (const unitId of new Set([...incByUnit.keys(), ...candByUnit.keys()])) {
|
|
169
|
+
const cn = candByUnit.get(unitId)?.scores.length ?? 0;
|
|
170
|
+
const inn = incByUnit.get(unitId)?.scores.length ?? 0;
|
|
171
|
+
if (cn >= 2 && inn < 2) {
|
|
172
|
+
failures.push(`n=${inn} on incumbent unit '${unitId}': baseline variance undefined — indeterminate, never nominated`);
|
|
140
173
|
}
|
|
141
|
-
|
|
142
|
-
|
|
143
|
-
|
|
144
|
-
|
|
174
|
+
else if (inn >= 2 && cn < 2) {
|
|
175
|
+
if (incByUnit.get(unitId)?.split === "val") {
|
|
176
|
+
failures.push(`no replicates for val unit '${unitId}' on candidate '${candidate.runId}' — variance undefined, never nominated`);
|
|
177
|
+
}
|
|
178
|
+
else {
|
|
179
|
+
failures.push(`n=${cn} on unit '${unitId}': variance undefined — indeterminate, never nominated`);
|
|
180
|
+
}
|
|
181
|
+
}
|
|
182
|
+
else if (cn < 2 && inn < 2) {
|
|
183
|
+
excluded.add(unitId);
|
|
145
184
|
}
|
|
146
185
|
}
|
|
147
186
|
if (failures.length > 0) {
|
|
148
|
-
return { ...base, verdict: "indeterminate", exitCode: needsExit("indeterminate"), gain:
|
|
187
|
+
return { ...base, verdict: "indeterminate", exitCode: needsExit("indeterminate"), gain: null, unitComparisons: [], failures };
|
|
149
188
|
}
|
|
150
|
-
|
|
189
|
+
const candKept = candidate.units.filter((u) => !excluded.has(u.unitId));
|
|
190
|
+
const incKept = incumbent.units.filter((u) => !excluded.has(u.unitId));
|
|
191
|
+
const candByUnitKept = new Map(candKept.map((u) => [u.unitId, u]));
|
|
192
|
+
// 3) Per-val-unit comparisons + the two precision gates. With a bank that
|
|
193
|
+
// knows the unit, acceptance semantics (see module header) replace the
|
|
194
|
+
// per-arm t-CI machinery FOR THAT UNIT ONLY; bank-unknown units keep the
|
|
195
|
+
// legacy lines verbatim. The fail-closed rules above run before and
|
|
196
|
+
// independently of any bank.
|
|
197
|
+
const bank = input.bank ?? null;
|
|
198
|
+
const qPair = bonferroniQ(nPairs);
|
|
199
|
+
const acceptanceUnits = [];
|
|
151
200
|
const unitComparisons = [];
|
|
152
201
|
for (const [unitId, incUnit] of incByUnit) {
|
|
153
|
-
if (incUnit.split !== "val")
|
|
202
|
+
if (incUnit.split !== "val" || excluded.has(unitId))
|
|
154
203
|
continue;
|
|
155
|
-
const candUnit =
|
|
204
|
+
const candUnit = candByUnitKept.get(unitId);
|
|
156
205
|
if (candUnit === undefined)
|
|
157
|
-
continue; //
|
|
158
|
-
const cs = summarizeUnit(candUnit.scores, alpha);
|
|
206
|
+
continue; // excluded/unreachable
|
|
159
207
|
const incumbentMean = summarizeUnit(incUnit.scores).mean;
|
|
208
|
+
const cs = summarizeUnit(candUnit.scores, alpha);
|
|
160
209
|
const delta = cs.mean - incumbentMean;
|
|
161
|
-
|
|
162
|
-
|
|
163
|
-
|
|
164
|
-
|
|
165
|
-
|
|
166
|
-
|
|
167
|
-
|
|
168
|
-
|
|
169
|
-
|
|
170
|
-
|
|
171
|
-
|
|
210
|
+
const banked = bank?.get(unitId) ?? null;
|
|
211
|
+
if (banked === null) {
|
|
212
|
+
unitComparisons.push({
|
|
213
|
+
unitId,
|
|
214
|
+
candidateMean: cs.mean,
|
|
215
|
+
incumbentMean,
|
|
216
|
+
ciHalfWidth: cs.ciHalfWidth,
|
|
217
|
+
delta,
|
|
218
|
+
ciPasses: cs.ciHalfWidth <= stats.halfWidth,
|
|
219
|
+
passes: delta >= -cs.ciHalfWidth, // ties allowed
|
|
220
|
+
});
|
|
221
|
+
if (cs.ciHalfWidth > stats.halfWidth) {
|
|
222
|
+
failures.push(`CI half-width ${fmt(cs.ciHalfWidth)} > halfWidth ${fmt(stats.halfWidth)} on unit '${unitId}'`);
|
|
223
|
+
}
|
|
224
|
+
if (delta < -cs.ciHalfWidth) {
|
|
225
|
+
failures.push(`regression on unit '${unitId}': delta ${fmt(delta)} < ${fmt(-cs.ciHalfWidth)} (mean ${fmt(cs.mean)} < ${fmt(incumbentMean)} - ${fmt(cs.ciHalfWidth)})`);
|
|
226
|
+
}
|
|
227
|
+
continue;
|
|
172
228
|
}
|
|
173
|
-
|
|
174
|
-
|
|
229
|
+
const se = banked.sigma * Math.sqrt(1 / candUnit.scores.length + 1 / incUnit.scores.length);
|
|
230
|
+
const band = (normalQuantile(qPair) * banked.sigma) / Math.sqrt(candUnit.scores.length);
|
|
231
|
+
const pRegression = se === 0 ? (delta <= -stats.halfWidth ? 1 : 0) : normalCdf((-stats.halfWidth - delta) / se);
|
|
232
|
+
const ciPasses = band <= stats.halfWidth;
|
|
233
|
+
const passes = pRegression < qPair; // ties at −halfWidth sit at Φ(0)=0.5 < q ⇒ still pass
|
|
234
|
+
unitComparisons.push({ unitId, candidateMean: cs.mean, incumbentMean, ciHalfWidth: band, delta, ciPasses, passes });
|
|
235
|
+
acceptanceUnits.push({ unitId, sigma: banked.sigma, df: banked.df, delta, se, band, pRegression });
|
|
236
|
+
if (!ciPasses) {
|
|
237
|
+
failures.push(`acceptance band ${fmt(band)} > halfWidth ${fmt(stats.halfWidth)} on unit '${unitId}' (bank sigma ${fmt(banked.sigma)}, df ${String(banked.df)})`);
|
|
238
|
+
}
|
|
239
|
+
else if (!passes) {
|
|
240
|
+
failures.push(`regression on unit '${unitId}': delta ${fmt(delta)} — P(Δ ≤ −${fmt(stats.halfWidth)}) = ${fmt(pRegression)} ≥ q ${fmt(qPair)} (bank sigma ${fmt(banked.sigma)})`);
|
|
175
241
|
}
|
|
176
242
|
}
|
|
177
|
-
// 4) Effect-size floor on the aggregate gain
|
|
178
|
-
|
|
243
|
+
// 4) Effect-size floor on the aggregate gain (shared, symmetric pool) —
|
|
244
|
+
// unchanged — plus the acceptance gate: when the bank can price EVERY
|
|
245
|
+
// unit of the gain pool, nomination additionally requires P(Δ>0) ≥ q.
|
|
246
|
+
// A bank-priced point mass exactly at 0 (se=0, gain=0) fails (pShift=0):
|
|
247
|
+
// demonstrated-identical arms are never nominated on a tie.
|
|
248
|
+
const gain = aggregateScore(candKept) - aggregateScore(incKept);
|
|
179
249
|
if (gain < stats.minEffect) {
|
|
180
250
|
failures.push(`gain ${fmt(gain)} < minEffect ${fmt(stats.minEffect)}`);
|
|
181
251
|
}
|
|
252
|
+
let gainSe = null;
|
|
253
|
+
let pShift = null;
|
|
254
|
+
if (bank !== null) {
|
|
255
|
+
const train = candKept.filter((u) => u.split === "train");
|
|
256
|
+
const pool = train.length > 0 ? train : candKept;
|
|
257
|
+
const incCounts = new Map(incKept.map((u) => [u.unitId, u.scores.length]));
|
|
258
|
+
const priced = pool.map((u) => bank.get(u.unitId) ?? null);
|
|
259
|
+
if (priced.every((s) => s !== null)) {
|
|
260
|
+
const k = pool.length;
|
|
261
|
+
let variance = 0;
|
|
262
|
+
pool.forEach((u, index) => {
|
|
263
|
+
const s = priced[index];
|
|
264
|
+
variance += ((s.sigma * s.sigma) * (1 / u.scores.length + 1 / (incCounts.get(u.unitId) ?? 0))) / (k * k);
|
|
265
|
+
});
|
|
266
|
+
gainSe = Math.sqrt(variance);
|
|
267
|
+
pShift = gainSe === 0 ? (gain > 0 ? 1 : 0) : normalCdf(gain / gainSe);
|
|
268
|
+
if (pShift < qPair) {
|
|
269
|
+
failures.push(`acceptance: P(gain shift > 0) = ${fmt(pShift)} < q ${fmt(qPair)} (se ${fmt(gainSe)}, gain ${fmt(gain)})`);
|
|
270
|
+
}
|
|
271
|
+
}
|
|
272
|
+
}
|
|
182
273
|
const verdict = failures.length === 0 ? "nominated" : "culled";
|
|
183
|
-
|
|
274
|
+
const acceptance = bank === null ? undefined : { q: qPair, gainSe, pShift, units: acceptanceUnits };
|
|
275
|
+
return { ...base, verdict, exitCode: needsExit(verdict), gain, unitComparisons, failures, ...(acceptance === undefined ? {} : { acceptance }) };
|
|
184
276
|
}
|