@tachikomagundam/abathur 0.2.4 → 0.2.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (36) hide show
  1. package/config/genomes/historian.example.jsonc +1 -1
  2. package/dist/commands/genome.js +2 -2
  3. package/dist/commands/graft.js +1 -1
  4. package/dist/commands/run.js +1 -1
  5. package/dist/commands/self-eval.js +1 -1
  6. package/dist/commands/tombstone.js +2 -1
  7. package/dist/core/evolve/run-bench.js +4 -3
  8. package/dist/core/evolve/run-loop.js +24 -4
  9. package/dist/core/evolve/run-plan.js +7 -3
  10. package/dist/core/evolve/score-bank.js +214 -0
  11. package/dist/core/graft-rebench.js +3 -0
  12. package/dist/core/ledger.js +6 -1
  13. package/dist/core/promote.js +2 -1
  14. package/dist/core/spec.js +5 -1
  15. package/dist/core/stats-math.js +39 -0
  16. package/dist/core/stats.js +125 -33
  17. package/dist/test/historian-grader-integrity.test.js +1494 -2
  18. package/dist/test/historian-grader-io.test.js +4 -0
  19. package/dist/test/historian-run-scenario.test.js +42 -2
  20. package/dist/test/repopath-seams.test.js +54 -0
  21. package/dist/test/score-bank.test.js +235 -0
  22. package/dist/test/seam-gates.test.js +78 -0
  23. package/dist/test/selfmode-literal.test.js +153 -0
  24. package/dist/test/stats-acceptance.test.js +329 -0
  25. package/dist/test/stats.test.js +18 -0
  26. package/graders/historian/grader-core.d.mts +107 -2
  27. package/graders/historian/grader-core.mjs +1046 -3
  28. package/graders/historian/grader.mjs +21 -2
  29. package/graders/historian/judge-poststage-19.mjs +212 -0
  30. package/graders/historian/judge-poststage.mjs +195 -0
  31. package/graders/historian/mutate.sh +59 -19
  32. package/graders/historian/run-scenario-17.sh +34 -0
  33. package/graders/historian/run-scenario-19.sh +43 -0
  34. package/graders/historian/run-scenario.sh +24 -1
  35. package/package.json +1 -1
  36. package/plugin/abathur.ts +1 -1
@@ -26,15 +26,39 @@
26
26
  // - Pareto over (trainScore, tokens, wallS); budget-truncated generations
27
27
  // (complete:false) never enter the set.
28
28
  //
29
+ // ACCEPTANCE SEMANTICS (2026-09-24 redesign; design doc lives in the target
30
+ // genome repo's evidence folder, not shipped here): when EvaluateInput.bank
31
+ // carries the historical per-unit variance bank, precision/regression checks
32
+ // price NOISE
33
+ // with bank σ instead of per-arm t(df=n−1) CIs (the campaign-9/10 resolution
34
+ // floor: t(0.05,1)=12.706 turned a 2-point sample into a 1.5883 half-width).
35
+ // A val unit with bank σ clears when P(Δ ≤ −halfWidth) < q_pair AND the
36
+ // acceptance band Z(q)·σ/√n_candidate ≤ halfWidth; the aggregate additionally
37
+ // requires P(gain shift > 0) = Φ(gain/SE) ≥ q_pair, q_pair = 1−(1−ACCEPT_Q)/nPairs.
38
+ // EVERY fail-closed rule ABOVE the bank path is untouched and can never be
39
+ // bypassed by it: budget⇒inconclusive, one-sided n<2⇒indeterminate,
40
+ // both-unmeasured⇒symmetric exclusion. A unit the bank cannot price falls back
41
+ // to the exact legacy t-CI line; a null/absent bank reproduces the pre-2026-09-24
42
+ // gate verbatim. minEffect 0.1 / halfWidth 0.15 point semantics stand.
43
+ //
29
44
  // The inverse Student-t lives in stats-math.ts and the Pareto ranking in
30
45
  // stats-pareto.ts; both are re-exported here so the public surface of this
31
46
  // module — the one todos 9/10/12 consume — is unchanged. Accuracy: |t − table|
32
47
  // < 1e-3 for the t-distribution rows checked in stats.test.ts (df=1..30, α=0.05).
33
48
  import { EXIT_BLOCKED, EXIT_CANNOT_ANSWER, EXIT_OK } from "../exit.js";
34
49
  /** Per-unit replicate scores (train and val units alike). */
35
- import { studentTQuantile } from "./stats-math.js";
50
+ import { studentTQuantile, normalCdf, normalQuantile } from "./stats-math.js";
36
51
  import { seededRandom, paretoFrontier } from "./stats-pareto.js";
37
- export { studentTQuantile, seededRandom, paretoFrontier };
52
+ export { studentTQuantile, normalCdf, normalQuantile, seededRandom, paretoFrontier };
53
+ /** Nomination acceptance probability (per finalist pair after Bonferroni). */
54
+ export const ACCEPT_Q = 0.9;
55
+ /** q_pair = 1 − (1 − ACCEPT_Q)/nPairs — Bonferroni on the acceptance ERROR rate. */
56
+ export function bonferroniQ(nPairs) {
57
+ if (!Number.isInteger(nPairs) || nPairs < 1) {
58
+ throw new RangeError(`bonferroniQ: nPairs=${String(nPairs)} must be an integer >= 1`);
59
+ }
60
+ return Math.min(0.9999, 1 - (1 - ACCEPT_Q) / nPairs);
61
+ }
38
62
  /** Family-wise alpha across finalist pairs (bonferroniAlpha divides it). */
39
63
  export const FAMILY_ALPHA = 0.05;
40
64
  /** Verdict → process exit code. inconclusive = 2 (cannot answer), never 0. */
@@ -132,53 +156,121 @@ export function evaluate(input) {
132
156
  const candByUnit = new Map(candidate.units.map((u) => [u.unitId, u]));
133
157
  const incByUnit = new Map(incumbent.units.map((u) => [u.unitId, u]));
134
158
  const failures = [];
135
- // 2) HARD RULE nReps ≥ 2: n<2 on ANY candidate unit (or a val unit missing
136
- // from the candidate ⇒ n=0) leaves the variance undefined.
137
- for (const unit of candidate.units) {
138
- if (unit.scores.length < 2) {
139
- failures.push(`n=${unit.scores.length} on unit '${unit.unitId}': variance undefined — indeterminate, never nominated`);
159
+ // 2) Cross-arm replicate rule. A unit measured cleanly (n>=2) on ONE arm but
160
+ // not on the other leaves that comparison's variance undefined ⇒
161
+ // indeterminate, never a silent cull and never a nomination (campaign-6:
162
+ // the candidate's train unit scored while the incumbent's had timed out;
163
+ // the old one-sided drop shifted the incumbent aggregate to the val
164
+ // fallback and minted a bogus negative gain). A unit unmeasured on BOTH
165
+ // arms is excluded symmetrically — every aggregate and comparison then
166
+ // runs on a pool both arms share replicate-for-replicate.
167
+ const excluded = new Set();
168
+ for (const unitId of new Set([...incByUnit.keys(), ...candByUnit.keys()])) {
169
+ const cn = candByUnit.get(unitId)?.scores.length ?? 0;
170
+ const inn = incByUnit.get(unitId)?.scores.length ?? 0;
171
+ if (cn >= 2 && inn < 2) {
172
+ failures.push(`n=${inn} on incumbent unit '${unitId}': baseline variance undefined — indeterminate, never nominated`);
140
173
  }
141
- }
142
- for (const [unitId, incUnit] of incByUnit) {
143
- if (incUnit.split === "val" && !candByUnit.has(unitId)) {
144
- failures.push(`no replicates for val unit '${unitId}' on candidate '${candidate.runId}' — variance undefined, never nominated`);
174
+ else if (inn >= 2 && cn < 2) {
175
+ if (incByUnit.get(unitId)?.split === "val") {
176
+ failures.push(`no replicates for val unit '${unitId}' on candidate '${candidate.runId}' — variance undefined, never nominated`);
177
+ }
178
+ else {
179
+ failures.push(`n=${cn} on unit '${unitId}': variance undefined — indeterminate, never nominated`);
180
+ }
181
+ }
182
+ else if (cn < 2 && inn < 2) {
183
+ excluded.add(unitId);
145
184
  }
146
185
  }
147
186
  if (failures.length > 0) {
148
- return { ...base, verdict: "indeterminate", exitCode: needsExit("indeterminate"), gain: aggregateScore(candidate.units) - aggregateScore(incumbent.units), unitComparisons: [], failures };
187
+ return { ...base, verdict: "indeterminate", exitCode: needsExit("indeterminate"), gain: null, unitComparisons: [], failures };
149
188
  }
150
- // 3) Per-val-unit comparisons + the two precision gates.
189
+ const candKept = candidate.units.filter((u) => !excluded.has(u.unitId));
190
+ const incKept = incumbent.units.filter((u) => !excluded.has(u.unitId));
191
+ const candByUnitKept = new Map(candKept.map((u) => [u.unitId, u]));
192
+ // 3) Per-val-unit comparisons + the two precision gates. With a bank that
193
+ // knows the unit, acceptance semantics (see module header) replace the
194
+ // per-arm t-CI machinery FOR THAT UNIT ONLY; bank-unknown units keep the
195
+ // legacy lines verbatim. The fail-closed rules above run before and
196
+ // independently of any bank.
197
+ const bank = input.bank ?? null;
198
+ const qPair = bonferroniQ(nPairs);
199
+ const acceptanceUnits = [];
151
200
  const unitComparisons = [];
152
201
  for (const [unitId, incUnit] of incByUnit) {
153
- if (incUnit.split !== "val")
202
+ if (incUnit.split !== "val" || excluded.has(unitId))
154
203
  continue;
155
- const candUnit = candByUnit.get(unitId);
204
+ const candUnit = candByUnitKept.get(unitId);
156
205
  if (candUnit === undefined)
157
- continue; // already ruled indeterminate above
158
- const cs = summarizeUnit(candUnit.scores, alpha);
206
+ continue; // excluded/unreachable
159
207
  const incumbentMean = summarizeUnit(incUnit.scores).mean;
208
+ const cs = summarizeUnit(candUnit.scores, alpha);
160
209
  const delta = cs.mean - incumbentMean;
161
- unitComparisons.push({
162
- unitId,
163
- candidateMean: cs.mean,
164
- incumbentMean,
165
- ciHalfWidth: cs.ciHalfWidth,
166
- delta,
167
- ciPasses: cs.ciHalfWidth <= stats.halfWidth,
168
- passes: delta >= -cs.ciHalfWidth, // ties allowed
169
- });
170
- if (cs.ciHalfWidth > stats.halfWidth) {
171
- failures.push(`CI half-width ${fmt(cs.ciHalfWidth)} > halfWidth ${fmt(stats.halfWidth)} on unit '${unitId}'`);
210
+ const banked = bank?.get(unitId) ?? null;
211
+ if (banked === null) {
212
+ unitComparisons.push({
213
+ unitId,
214
+ candidateMean: cs.mean,
215
+ incumbentMean,
216
+ ciHalfWidth: cs.ciHalfWidth,
217
+ delta,
218
+ ciPasses: cs.ciHalfWidth <= stats.halfWidth,
219
+ passes: delta >= -cs.ciHalfWidth, // ties allowed
220
+ });
221
+ if (cs.ciHalfWidth > stats.halfWidth) {
222
+ failures.push(`CI half-width ${fmt(cs.ciHalfWidth)} > halfWidth ${fmt(stats.halfWidth)} on unit '${unitId}'`);
223
+ }
224
+ if (delta < -cs.ciHalfWidth) {
225
+ failures.push(`regression on unit '${unitId}': delta ${fmt(delta)} < ${fmt(-cs.ciHalfWidth)} (mean ${fmt(cs.mean)} < ${fmt(incumbentMean)} - ${fmt(cs.ciHalfWidth)})`);
226
+ }
227
+ continue;
172
228
  }
173
- if (delta < -cs.ciHalfWidth) {
174
- failures.push(`regression on unit '${unitId}': delta ${fmt(delta)} < ${fmt(-cs.ciHalfWidth)} (mean ${fmt(cs.mean)} < ${fmt(incumbentMean)} - ${fmt(cs.ciHalfWidth)})`);
229
+ const se = banked.sigma * Math.sqrt(1 / candUnit.scores.length + 1 / incUnit.scores.length);
230
+ const band = (normalQuantile(qPair) * banked.sigma) / Math.sqrt(candUnit.scores.length);
231
+ const pRegression = se === 0 ? (delta <= -stats.halfWidth ? 1 : 0) : normalCdf((-stats.halfWidth - delta) / se);
232
+ const ciPasses = band <= stats.halfWidth;
233
+ const passes = pRegression < qPair; // ties at −halfWidth sit at Φ(0)=0.5 < q ⇒ still pass
234
+ unitComparisons.push({ unitId, candidateMean: cs.mean, incumbentMean, ciHalfWidth: band, delta, ciPasses, passes });
235
+ acceptanceUnits.push({ unitId, sigma: banked.sigma, df: banked.df, delta, se, band, pRegression });
236
+ if (!ciPasses) {
237
+ failures.push(`acceptance band ${fmt(band)} > halfWidth ${fmt(stats.halfWidth)} on unit '${unitId}' (bank sigma ${fmt(banked.sigma)}, df ${String(banked.df)})`);
238
+ }
239
+ else if (!passes) {
240
+ failures.push(`regression on unit '${unitId}': delta ${fmt(delta)} — P(Δ ≤ −${fmt(stats.halfWidth)}) = ${fmt(pRegression)} ≥ q ${fmt(qPair)} (bank sigma ${fmt(banked.sigma)})`);
175
241
  }
176
242
  }
177
- // 4) Effect-size floor on the aggregate gain.
178
- const gain = aggregateScore(candidate.units) - aggregateScore(incumbent.units);
243
+ // 4) Effect-size floor on the aggregate gain (shared, symmetric pool) —
244
+ // unchanged — plus the acceptance gate: when the bank can price EVERY
245
+ // unit of the gain pool, nomination additionally requires P(Δ>0) ≥ q.
246
+ // A bank-priced point mass exactly at 0 (se=0, gain=0) fails (pShift=0):
247
+ // demonstrated-identical arms are never nominated on a tie.
248
+ const gain = aggregateScore(candKept) - aggregateScore(incKept);
179
249
  if (gain < stats.minEffect) {
180
250
  failures.push(`gain ${fmt(gain)} < minEffect ${fmt(stats.minEffect)}`);
181
251
  }
252
+ let gainSe = null;
253
+ let pShift = null;
254
+ if (bank !== null) {
255
+ const train = candKept.filter((u) => u.split === "train");
256
+ const pool = train.length > 0 ? train : candKept;
257
+ const incCounts = new Map(incKept.map((u) => [u.unitId, u.scores.length]));
258
+ const priced = pool.map((u) => bank.get(u.unitId) ?? null);
259
+ if (priced.every((s) => s !== null)) {
260
+ const k = pool.length;
261
+ let variance = 0;
262
+ pool.forEach((u, index) => {
263
+ const s = priced[index];
264
+ variance += ((s.sigma * s.sigma) * (1 / u.scores.length + 1 / (incCounts.get(u.unitId) ?? 0))) / (k * k);
265
+ });
266
+ gainSe = Math.sqrt(variance);
267
+ pShift = gainSe === 0 ? (gain > 0 ? 1 : 0) : normalCdf(gain / gainSe);
268
+ if (pShift < qPair) {
269
+ failures.push(`acceptance: P(gain shift > 0) = ${fmt(pShift)} < q ${fmt(qPair)} (se ${fmt(gainSe)}, gain ${fmt(gain)})`);
270
+ }
271
+ }
272
+ }
182
273
  const verdict = failures.length === 0 ? "nominated" : "culled";
183
- return { ...base, verdict, exitCode: needsExit(verdict), gain, unitComparisons, failures };
274
+ const acceptance = bank === null ? undefined : { q: qPair, gainSe, pShift, units: acceptanceUnits };
275
+ return { ...base, verdict, exitCode: needsExit(verdict), gain, unitComparisons, failures, ...(acceptance === undefined ? {} : { acceptance }) };
184
276
  }