@tachikomagundam/abathur 0.2.4 → 0.2.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/config/genomes/historian.example.jsonc +1 -1
- package/dist/core/evolve/run-bench.js +4 -3
- package/dist/core/evolve/run-loop.js +10 -4
- package/dist/core/evolve/run-plan.js +7 -3
- package/dist/core/spec.js +5 -1
- package/dist/core/stats.js +33 -15
- package/dist/test/historian-grader-integrity.test.js +987 -2
- package/dist/test/historian-grader-io.test.js +4 -0
- package/dist/test/historian-run-scenario.test.js +42 -2
- package/dist/test/selfmode-literal.test.js +153 -0
- package/dist/test/stats.test.js +18 -0
- package/graders/historian/grader-core.d.mts +80 -2
- package/graders/historian/grader-core.mjs +754 -3
- package/graders/historian/grader.mjs +21 -2
- package/graders/historian/judge-poststage.mjs +200 -0
- package/graders/historian/mutate.sh +59 -19
- package/graders/historian/run-scenario-17.sh +34 -0
- package/graders/historian/run-scenario.sh +8 -1
- package/package.json +1 -1
- package/plugin/abathur.ts +1 -1
|
@@ -56,7 +56,7 @@
|
|
|
56
56
|
"resetCommand": "bash ${ABATHUR_REPO}/graders/historian/reset-sandbox.sh",
|
|
57
57
|
"runCommand": "bash ${ABATHUR_REPO}/graders/historian/run-scenario.sh {unit.id} {repoRoot}/{unit.path}",
|
|
58
58
|
"graderCommand": "node ${ABATHUR_REPO}/graders/historian/grader.mjs {unit.id} {repoRoot}/{unit.path} ${ABATHUR_WIKI_BASE}",
|
|
59
|
-
"agentModel": "
|
|
59
|
+
"agentModel": "replace-me/provider/model", // operator: substitute the real <provider>/<model> when copying
|
|
60
60
|
// Real agent runs, not unit tests: 1200s caps one stuck scenario without
|
|
61
61
|
// poisoning the generation (timeout => unscored/inconclusive, excluded by
|
|
62
62
|
// the stats layer). seed+reset hooks share a FIXED 60s adapter budget
|
|
@@ -16,10 +16,11 @@ export function cloneMatrixRow(u) {
|
|
|
16
16
|
export function copyProvenance(p) {
|
|
17
17
|
return { benchType: p.benchType, versions: p.versions.map((v) => ({ bin: v.bin, version: v.version })) };
|
|
18
18
|
}
|
|
19
|
-
//
|
|
20
|
-
//
|
|
19
|
+
// Invariant: unscored units (timeout/infra_failed/budget cut) stay in the matrix —
|
|
20
|
+
// dropping them asymmetrically shifts one arm's aggregate pool (campaign-6 artifact);
|
|
21
|
+
// evaluate() fail-closes any arm's n<2 as indeterminate before any aggregation.
|
|
21
22
|
export function asReplicates(rows) {
|
|
22
|
-
return rows.
|
|
23
|
+
return rows.map((u) => ({ unitId: u.unitId, split: u.split, scores: u.scores }));
|
|
23
24
|
}
|
|
24
25
|
export * from "./run-rows.js";
|
|
25
26
|
/** toy ⇒ stateless adapter; fixture ⇒ ctor acquires the fingerprint lock immediately. */
|
|
@@ -32,7 +32,7 @@ import { compactUtc, fingerprint, genId } from "../ids.js";
|
|
|
32
32
|
import { LEDGER_KIND_GENERATION_COMPLETE, acquireGenomeLock, Ledger } from "../ledger.js";
|
|
33
33
|
import { openGenome } from "../worktree.js";
|
|
34
34
|
import { clampReps, evaluate, needsExit } from "../stats.js";
|
|
35
|
-
import { effectiveRepoPath, isEnvRepoLiteral } from "../spec.js";
|
|
35
|
+
import { effectiveRepoPath, envRepoName, isEnvRepoLiteral } from "../spec.js";
|
|
36
36
|
import { ChildTracker, reapOrphans } from "./child-track.js";
|
|
37
37
|
import { buildBrief } from "./brief.js";
|
|
38
38
|
import { startRunFriction } from "./run-friction.js";
|
|
@@ -101,7 +101,13 @@ async function evolve(opts, mutatorCommand, ledger, caps, reps, env, now) {
|
|
|
101
101
|
const spec = opts.entry.spec;
|
|
102
102
|
const lines = [];
|
|
103
103
|
const genomeRepo = effectiveRepoPath(spec.repoPath);
|
|
104
|
-
|
|
104
|
+
// Engine-self ONLY: a `${VAR}` literal is a path-injection convenience any
|
|
105
|
+
// harness repo may use (campaign-1 misroute pin: selfmode-literal.test.ts).
|
|
106
|
+
const selfMode = envRepoName(spec.repoPath) === "ABATHUR_SELF_REPO";
|
|
107
|
+
// Env-literals resolve to concrete paths exactly once, here at the FS
|
|
108
|
+
// boundary; bench/reflect drivers below never see the raw literal.
|
|
109
|
+
// Concrete repoPaths flow through byte-identical to before.
|
|
110
|
+
const loopSpec = isEnvRepoLiteral(spec.repoPath) ? { ...spec, repoPath: genomeRepo } : spec;
|
|
105
111
|
const fric = opts.friction === undefined ? null : startRunFriction(opts.friction);
|
|
106
112
|
// Reap only under the genome lock: every remaining log entry belongs to a dead run.
|
|
107
113
|
const reap = reapOrphans(genomeRepo);
|
|
@@ -131,7 +137,7 @@ async function evolve(opts, mutatorCommand, ledger, caps, reps, env, now) {
|
|
|
131
137
|
else {
|
|
132
138
|
const gen = genId(fingerprint({ incumbent: opened.headCommit, at: now().getTime() }), now());
|
|
133
139
|
const out = await benchTarget({
|
|
134
|
-
spec,
|
|
140
|
+
spec: loopSpec,
|
|
135
141
|
genId: gen,
|
|
136
142
|
reps,
|
|
137
143
|
caps,
|
|
@@ -181,7 +187,7 @@ async function evolve(opts, mutatorCommand, ledger, caps, reps, env, now) {
|
|
|
181
187
|
const brief = buildBrief(spec, evidence, counters);
|
|
182
188
|
tracker.phase(`mutator-${invId}`, "mutator-session");
|
|
183
189
|
const session = await runMutatorSession({
|
|
184
|
-
spec:
|
|
190
|
+
spec: loopSpec,
|
|
185
191
|
brief,
|
|
186
192
|
mutatorCommand,
|
|
187
193
|
...(opts.opencodeBin === undefined ? {} : { opencodeBin: opts.opencodeBin }),
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
// --dry-run plan rendering + budget-cap resolution (todo 9). Presentation of
|
|
2
2
|
// the resolved startup state only — the dry-run gate in run-loop.ts calls this
|
|
3
3
|
// AFTER the kernel audit and BEFORE any ledger/lock/spawn touch.
|
|
4
|
-
import
|
|
4
|
+
import { effectiveRepoPath } from "../spec.js";
|
|
5
5
|
import { peekPlanState } from "./run-rows.js";
|
|
6
6
|
/** `--max-candidates` clamps DOWN to the genome's own cap, never above. */
|
|
7
7
|
export function effectiveCaps(spec, clamp) {
|
|
@@ -15,12 +15,16 @@ export function effectiveCaps(spec, clamp) {
|
|
|
15
15
|
}
|
|
16
16
|
export function planLines(req) {
|
|
17
17
|
const spec = req.entry.spec;
|
|
18
|
-
|
|
18
|
+
// `${VAR}` literals resolve through the same FS-boundary contract as every
|
|
19
|
+
// other consumer; the plan line + resume peek must never join the raw
|
|
20
|
+
// literal against cwd.
|
|
21
|
+
const genomeRepo = effectiveRepoPath(spec.repoPath);
|
|
22
|
+
const peek = peekPlanState(genomeRepo);
|
|
19
23
|
const specMax = spec.budget.maxCandidates;
|
|
20
24
|
const unitList = spec.bench.units.map((u) => `${u.id}(${u.split})`).join(" ");
|
|
21
25
|
return [
|
|
22
26
|
`run plan — genome '${spec.label}' (${req.entry.fingerprint})`,
|
|
23
|
-
` repo: ${
|
|
27
|
+
` repo: ${genomeRepo}`,
|
|
24
28
|
` bench: ${spec.bench.type} — timeoutS=${String(spec.bench.timeoutS)} stats minEffect=${String(spec.bench.stats.minEffect)} halfWidth=${String(spec.bench.stats.halfWidth)}`,
|
|
25
29
|
` units: ${unitList}`,
|
|
26
30
|
...(req.requiresProbed === undefined
|
package/dist/core/spec.js
CHANGED
|
@@ -126,7 +126,11 @@ export function errorText(cause) {
|
|
|
126
126
|
* resolved ONLY at FS boundaries, never inside the spec.
|
|
127
127
|
*/
|
|
128
128
|
const ENV_REPO_LITERAL = /^\$\{([A-Za-z_][A-Za-z0-9_]*)\}$/;
|
|
129
|
-
/**
|
|
129
|
+
/**
|
|
130
|
+
* True when the spec stores an unresolved `${VAR}` env literal — a path-
|
|
131
|
+
* injection convenience ANY genome may use, NOT an engine-self discriminator
|
|
132
|
+
* (for that compare envRepoName against ABATHUR_SELF_REPO).
|
|
133
|
+
*/
|
|
130
134
|
export function isEnvRepoLiteral(repoPath) {
|
|
131
135
|
return ENV_REPO_LITERAL.test(repoPath);
|
|
132
136
|
}
|
package/dist/core/stats.js
CHANGED
|
@@ -132,29 +132,47 @@ export function evaluate(input) {
|
|
|
132
132
|
const candByUnit = new Map(candidate.units.map((u) => [u.unitId, u]));
|
|
133
133
|
const incByUnit = new Map(incumbent.units.map((u) => [u.unitId, u]));
|
|
134
134
|
const failures = [];
|
|
135
|
-
// 2)
|
|
136
|
-
//
|
|
137
|
-
|
|
138
|
-
|
|
139
|
-
|
|
135
|
+
// 2) Cross-arm replicate rule. A unit measured cleanly (n>=2) on ONE arm but
|
|
136
|
+
// not on the other leaves that comparison's variance undefined ⇒
|
|
137
|
+
// indeterminate, never a silent cull and never a nomination (campaign-6:
|
|
138
|
+
// the candidate's train unit scored while the incumbent's had timed out;
|
|
139
|
+
// the old one-sided drop shifted the incumbent aggregate to the val
|
|
140
|
+
// fallback and minted a bogus negative gain). A unit unmeasured on BOTH
|
|
141
|
+
// arms is excluded symmetrically — every aggregate and comparison then
|
|
142
|
+
// runs on a pool both arms share replicate-for-replicate.
|
|
143
|
+
const excluded = new Set();
|
|
144
|
+
for (const unitId of new Set([...incByUnit.keys(), ...candByUnit.keys()])) {
|
|
145
|
+
const cn = candByUnit.get(unitId)?.scores.length ?? 0;
|
|
146
|
+
const inn = incByUnit.get(unitId)?.scores.length ?? 0;
|
|
147
|
+
if (cn >= 2 && inn < 2) {
|
|
148
|
+
failures.push(`n=${inn} on incumbent unit '${unitId}': baseline variance undefined — indeterminate, never nominated`);
|
|
140
149
|
}
|
|
141
|
-
|
|
142
|
-
|
|
143
|
-
|
|
144
|
-
|
|
150
|
+
else if (inn >= 2 && cn < 2) {
|
|
151
|
+
if (incByUnit.get(unitId)?.split === "val") {
|
|
152
|
+
failures.push(`no replicates for val unit '${unitId}' on candidate '${candidate.runId}' — variance undefined, never nominated`);
|
|
153
|
+
}
|
|
154
|
+
else {
|
|
155
|
+
failures.push(`n=${cn} on unit '${unitId}': variance undefined — indeterminate, never nominated`);
|
|
156
|
+
}
|
|
157
|
+
}
|
|
158
|
+
else if (cn < 2 && inn < 2) {
|
|
159
|
+
excluded.add(unitId);
|
|
145
160
|
}
|
|
146
161
|
}
|
|
147
162
|
if (failures.length > 0) {
|
|
148
|
-
return { ...base, verdict: "indeterminate", exitCode: needsExit("indeterminate"), gain:
|
|
163
|
+
return { ...base, verdict: "indeterminate", exitCode: needsExit("indeterminate"), gain: null, unitComparisons: [], failures };
|
|
149
164
|
}
|
|
165
|
+
const candKept = candidate.units.filter((u) => !excluded.has(u.unitId));
|
|
166
|
+
const incKept = incumbent.units.filter((u) => !excluded.has(u.unitId));
|
|
167
|
+
const candByUnitKept = new Map(candKept.map((u) => [u.unitId, u]));
|
|
150
168
|
// 3) Per-val-unit comparisons + the two precision gates.
|
|
151
169
|
const unitComparisons = [];
|
|
152
170
|
for (const [unitId, incUnit] of incByUnit) {
|
|
153
|
-
if (incUnit.split !== "val")
|
|
171
|
+
if (incUnit.split !== "val" || excluded.has(unitId))
|
|
154
172
|
continue;
|
|
155
|
-
const candUnit =
|
|
173
|
+
const candUnit = candByUnitKept.get(unitId);
|
|
156
174
|
if (candUnit === undefined)
|
|
157
|
-
continue; //
|
|
175
|
+
continue; // excluded/unreachable
|
|
158
176
|
const cs = summarizeUnit(candUnit.scores, alpha);
|
|
159
177
|
const incumbentMean = summarizeUnit(incUnit.scores).mean;
|
|
160
178
|
const delta = cs.mean - incumbentMean;
|
|
@@ -174,8 +192,8 @@ export function evaluate(input) {
|
|
|
174
192
|
failures.push(`regression on unit '${unitId}': delta ${fmt(delta)} < ${fmt(-cs.ciHalfWidth)} (mean ${fmt(cs.mean)} < ${fmt(incumbentMean)} - ${fmt(cs.ciHalfWidth)})`);
|
|
175
193
|
}
|
|
176
194
|
}
|
|
177
|
-
// 4) Effect-size floor on the aggregate gain.
|
|
178
|
-
const gain = aggregateScore(
|
|
195
|
+
// 4) Effect-size floor on the aggregate gain (shared, symmetric pool).
|
|
196
|
+
const gain = aggregateScore(candKept) - aggregateScore(incKept);
|
|
179
197
|
if (gain < stats.minEffect) {
|
|
180
198
|
failures.push(`gain ${fmt(gain)} < minEffect ${fmt(stats.minEffect)}`);
|
|
181
199
|
}
|