@tachikomagundam/abathur 0.2.6 → 0.2.8

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (56) hide show
  1. package/README.md +3 -3
  2. package/dist/cli.js +4 -0
  3. package/dist/commands/readjudicate.js +28 -0
  4. package/dist/commands/retract.js +53 -0
  5. package/dist/commands/tombstone.js +28 -15
  6. package/dist/core/evolve/run-bench.js +7 -1
  7. package/dist/core/evolve/run-rows.js +14 -0
  8. package/dist/core/ledger.js +2 -0
  9. package/dist/core/promote.js +61 -46
  10. package/dist/core/readjudicate.js +247 -0
  11. package/dist/core/retract.js +166 -0
  12. package/dist/core/stats.js +7 -0
  13. package/dist/core/worktree.js +1 -1
  14. package/dist/genomes/toy-smoke/init.mjs +1 -1
  15. package/dist/test/bundle.test.js +1 -1
  16. package/dist/test/fixture-loop.test.js +1 -1
  17. package/dist/test/fixtures-self.js +1 -1
  18. package/dist/test/fixtures-wt.js +1 -1
  19. package/dist/test/friction.test.js +3 -3
  20. package/dist/test/graft.test.js +1 -1
  21. package/dist/test/historian-run-scenario.test.js +2 -2
  22. package/dist/test/promote.test.js +222 -5
  23. package/dist/test/readjudicate.test.js +212 -0
  24. package/dist/test/reflect.test.js +1 -1
  25. package/dist/test/repopath-seams.test.js +1 -0
  26. package/dist/test/seam-gates.test.js +6 -0
  27. package/dist/test/self-snapshot.test.js +2 -2
  28. package/dist/test/stats.test.js +17 -0
  29. package/docs/AGENTISM.md +58 -0
  30. package/docs/release/README.md +9 -0
  31. package/docs/release/TEMPLATE.md +10 -0
  32. package/docs/release/v0.2.7.md +7 -0
  33. package/docs/release/v0.2.8.md +8 -0
  34. package/graders/historian/judge-poststage.mjs +8 -3
  35. package/graders/pcb-agent/bench-approver.py +77 -0
  36. package/graders/pcb-agent/grader.mjs +130 -0
  37. package/graders/pcb-agent/mutate.sh +165 -0
  38. package/graders/pcb-agent/r20-bench.sh +27 -0
  39. package/graders/pcb-agent/reset-sandbox.sh +7 -0
  40. package/graders/pcb-agent/run-scenario-r20.sh +77 -0
  41. package/graders/pcb-agent/run-scenario.sh +94 -0
  42. package/graders/pcb-agent/seed/esp32s3-r12-BRIEF.txt +51 -0
  43. package/graders/pcb-agent/seed/spec.json +129 -0
  44. package/graders/pcb-agent/seed/tools/board_author_pcb.py +237 -0
  45. package/graders/pcb-agent/seed/tools/board_author_sch.py +118 -0
  46. package/graders/pcb-agent/seed/tools/board_finalize.sh +23 -0
  47. package/graders/pcb-agent/seed/tools/board_gate.py +69 -0
  48. package/graders/pcb-agent/seed/tools/board_gate_pcb.py +23 -0
  49. package/graders/pcb-agent/seed/tools/board_route.py +324 -0
  50. package/graders/pcb-agent/seed/tools/board_zone_pour.py +312 -0
  51. package/graders/pcb-agent/seed/tools/claims_lint.py +417 -0
  52. package/graders/pcb-agent/seed-r20-arena.sh +54 -0
  53. package/graders/pcb-agent/seed-sandbox.sh +13 -0
  54. package/package.json +1 -1
  55. package/plugin/abathur-command.md +3 -3
  56. package/plugin/abathur.ts +6 -6
@@ -0,0 +1,212 @@
1
+ // re-adjudicate (core) tests. The mechanism under test: replay the CURRENT gate over
2
+ // archived generation rows and append a corrected verdict with provenance, files immutable.
3
+ //
4
+ // Coverage: verdict flip culled->nominated (synthetic), honest no-flip (culled stays
5
+ // culled), idempotence after append, promote-row refusal, missing-incumbent refusal,
6
+ // corrupt-archive notice (never a crash), and the historical anchor: a sandboxed COPY of
7
+ // the real historian archives must reproduce the c17 re-adjudication (NOMINATED,
8
+ // gain 0.2917) that the driver computed offline on 2026-09-26 — the engine's own replay
9
+ // of the incident that born this command.
10
+ import test from "node:test";
11
+ import assert from "node:assert/strict";
12
+ import { mkdirSync, readFileSync, readdirSync, rmSync, writeFileSync, copyFileSync } from "node:fs";
13
+ import { mkdtempSync } from "node:fs";
14
+ import { tmpdir } from "node:os";
15
+ import path from "node:path";
16
+ import { readjudicateGeneration } from "../core/readjudicate.js";
17
+ import { ledgerRecordSchema, LEDGER_KIND_GENERATION_COMPLETE, LEDGER_KIND_PROMOTE } from "../core/ledger.js";
18
+ import { decodeGenerationRecord } from "../core/evolve/run-rows.js";
19
+ const STATS = { halfWidth: 0.15, minEffect: 0.1, nReps: { initial: 2, max: 3 } };
20
+ const BUDGET = { maxCandidates: 1, maxModelCalls: 96, maxTokens: 25_000_000, maxWallS: 26_000 };
21
+ function entryFor(repo) {
22
+ return {
23
+ fingerprint: "f".repeat(16),
24
+ label: "test-genome",
25
+ registryFile: "/dev/null",
26
+ storedText: "",
27
+ spec: {
28
+ label: "test-genome",
29
+ repoPath: repo,
30
+ bench: {
31
+ type: "opencode-fixture-scenarios",
32
+ units: [
33
+ { id: "scenario-19", split: "train", path: "scenarios/19.md" },
34
+ { id: "scenario-13", split: "val", path: "scenarios/13.md" },
35
+ ],
36
+ runCommand: "true",
37
+ graderCommand: "true",
38
+ timeoutS: 1800,
39
+ stats: STATS,
40
+ },
41
+ budget: BUDGET,
42
+ kernel: { immutableGlobs: ["scenarios/**"] },
43
+ },
44
+ };
45
+ }
46
+ function row(o) {
47
+ const data = {
48
+ source: o.source,
49
+ headCommit: "1".repeat(40),
50
+ complete: true,
51
+ reps: o.units[0]?.scores.length ?? 0,
52
+ units: o.units.map((u) => ({ unitId: u.unitId, split: u.split, scores: u.scores, runIds: u.scores.map((_, i) => `r-${u.unitId}-${String(i)}`), failures: u.scores.filter((s) => s < 1).map((s) => `${u.unitId}: scored ${String(s)}`) })),
53
+ counters: { candidates: o.source === "candidate" ? 1 : 0, modelCalls: 10, tokens: 1000, wallS: 100 },
54
+ manifest: [],
55
+ benchProvenance: { benchType: "opencode-fixture-scenarios", versions: [{ bin: "opencode", version: "test" }] },
56
+ };
57
+ if (o.source === "candidate") {
58
+ data.candidateId = "mutate-live";
59
+ data.commitSha = "2".repeat(40);
60
+ data.treeSha = o.treeSha ?? "3".repeat(64);
61
+ }
62
+ if (o.verdict !== undefined)
63
+ data.verdict = o.verdict;
64
+ if (o.gain !== undefined)
65
+ data.gain = o.gain;
66
+ return `${JSON.stringify({ v: 1, ts: o.ts, kind: LEDGER_KIND_GENERATION_COMPLETE, genId: o.genId, runId: o.runId, data })}\n`;
67
+ }
68
+ function sandbox(t) {
69
+ const dir = mkdtempSync(path.join(tmpdir(), "readjud-"));
70
+ const state = path.join(dir, ".state", "abathur");
71
+ mkdirSync(state, { recursive: true });
72
+ writeFileSync(path.join(state, "ledger.jsonl"), "");
73
+ t.after(() => rmSync(dir, { recursive: true, force: true }));
74
+ return { dir, state };
75
+ }
76
+ function cfg(t) {
77
+ const d = mkdtempSync(path.join(tmpdir(), "readjud-cfg-"));
78
+ t.after(() => rmSync(d, { recursive: true, force: true }));
79
+ return d;
80
+ }
81
+ function lastRows(stateDir) {
82
+ return readFileSync(path.join(stateDir, "ledger.jsonl"), "utf8")
83
+ .split("\n")
84
+ .filter((l) => l.trim().length > 0)
85
+ .map((l) => {
86
+ const parsed = ledgerRecordSchema.parse(JSON.parse(l));
87
+ return parsed;
88
+ })
89
+ .filter((r) => r.kind === LEDGER_KIND_GENERATION_COMPLETE)
90
+ .map((r) => {
91
+ try {
92
+ return decodeGenerationRecord(r);
93
+ }
94
+ catch {
95
+ return null;
96
+ }
97
+ })
98
+ .filter((d) => d !== null);
99
+ }
100
+ test("flip: culled row re-judged nominated by the current gate; corrected row appended with provenance", (t) => {
101
+ const { dir, state } = sandbox(t);
102
+ // one rotated archive with the whole epoch (activation law: bank needs an archive)
103
+ const lines = [
104
+ // historical incumbent groups that price a small-ish sigma on both units
105
+ row({ genId: "g-old-a", runId: "r-old", ts: "2026-09-20T00:00:00.000Z", source: "incumbent", units: [{ unitId: "scenario-19", split: "train", scores: [0.875, 1, 0.875, 1] }, { unitId: "scenario-13", split: "val", scores: [1, 1, 1, 1] }] }),
106
+ // the epoch under re-adjudication
107
+ row({ genId: "g-epoch", runId: "run-1", ts: "2026-09-21T00:00:00.000Z", source: "incumbent", units: [{ unitId: "scenario-19", split: "train", scores: [0.75, 0.875] }, { unitId: "scenario-13", split: "val", scores: [1, 1] }], verdict: undefined }),
108
+ row({ genId: "g-epoch-cand", runId: "run-1", ts: "2026-09-21T01:00:00.000Z", source: "candidate", units: [{ unitId: "scenario-19", split: "train", scores: [1, 1] }, { unitId: "scenario-13", split: "val", scores: [1, 1] }], verdict: "culled", gain: 0.1 }),
109
+ ];
110
+ writeFileSync(path.join(state, "ledger.campaign1-2026-09-21.jsonl"), lines.join(""));
111
+ const out = readjudicateGeneration({ entry: entryFor(dir), configDir: cfg(t), genId: "g-epoch-cand", now: () => new Date("2026-09-22T00:00:00Z") });
112
+ assert.equal(out.appended, true, out.lines.join("\n"));
113
+ assert.equal(out.verdict, "nominated", out.lines.join("\n"));
114
+ const appended = lastRows(state).at(-1);
115
+ assert.ok(appended !== undefined);
116
+ assert.equal(appended.verdict, "nominated");
117
+ assert.ok(appended.readjudication !== undefined, "provenance must be present");
118
+ assert.equal(appended.readjudication.replacedVerdict, "culled");
119
+ assert.equal(appended.readjudication.gate, "acceptance");
120
+ assert.equal(appended.readjudication.sourceRows.candidate.file, "ledger.campaign1-2026-09-21.jsonl");
121
+ // archive untouched: still exactly 3 rows, original culled row intact
122
+ assert.equal(readFileSync(path.join(state, "ledger.campaign1-2026-09-21.jsonl"), "utf8").trim().split("\n").length, 3);
123
+ // idempotence: second pass sees the corrected last row -> no-op
124
+ const again = readjudicateGeneration({ entry: entryFor(dir), configDir: cfg(t), genId: "g-epoch-cand", now: () => new Date("2026-09-22T00:00:00Z") });
125
+ assert.equal(again.appended, false, again.lines.join("\n"));
126
+ assert.match(again.lines.join("\n"), /no-op/);
127
+ });
128
+ test("honest replay: a weak candidate stays culled — the tool re-judges, it never flips", (t) => {
129
+ const { dir, state } = sandbox(t);
130
+ const lines = [
131
+ row({ genId: "g-old-a", runId: "r-old", ts: "2026-09-20T00:00:00.000Z", source: "incumbent", units: [{ unitId: "scenario-19", split: "train", scores: [1, 1, 1, 1] }, { unitId: "scenario-13", split: "val", scores: [1, 1, 1, 1] }] }),
132
+ row({ genId: "g-epoch", runId: "run-1", ts: "2026-09-21T00:00:00.000Z", source: "incumbent", units: [{ unitId: "scenario-19", split: "train", scores: [1, 1] }, { unitId: "scenario-13", split: "val", scores: [1, 1] }] }),
133
+ row({ genId: "g-epoch-cand", runId: "run-1", ts: "2026-09-21T01:00:00.000Z", source: "candidate", units: [{ unitId: "scenario-19", split: "train", scores: [1, 1] }, { unitId: "scenario-13", split: "val", scores: [1, 1] }], verdict: "culled", gain: 0 }),
134
+ ];
135
+ writeFileSync(path.join(state, "ledger.campaign1-2026-09-21.jsonl"), lines.join(""));
136
+ const out = readjudicateGeneration({ entry: entryFor(dir), configDir: cfg(t), genId: "g-epoch-cand", now: () => new Date() });
137
+ assert.equal(out.verdict, "culled", out.lines.join("\n"));
138
+ assert.equal(out.appended, false, "already-agreeing row must not append");
139
+ });
140
+ test("refusals: promoted gen, missing incumbent, unknown gen — all block before any append", (t) => {
141
+ const { dir, state } = sandbox(t);
142
+ const epoch = [
143
+ row({ genId: "g-epoch", runId: "run-9", ts: "2026-09-21T00:00:00.000Z", source: "incumbent", units: [{ unitId: "scenario-19", split: "train", scores: [0.75, 1] }, { unitId: "scenario-13", split: "val", scores: [1, 1] }] }),
144
+ row({ genId: "g-epoch-cand", runId: "run-9", ts: "2026-09-21T01:00:00.000Z", source: "candidate", units: [{ unitId: "scenario-19", split: "train", scores: [1, 1] }, { unitId: "scenario-13", split: "val", scores: [1, 1] }], verdict: "culled", gain: 0.1 }),
145
+ ];
146
+ writeFileSync(path.join(state, "ledger.campaign1-2026-09-21.jsonl"), epoch.join(""));
147
+ // promoted epoch is untouchable
148
+ const promotedRun = "run-p";
149
+ writeFileSync(path.join(state, "ledger.campaign2-2026-09-22.jsonl"), epoch.join("") + `${JSON.stringify({ v: 1, ts: "2026-09-22T00:00:00.000Z", kind: LEDGER_KIND_PROMOTE, genId: "g-epoch-cand", runId: promotedRun, data: { actor: "cli", genId: "g-epoch-cand", from: null, to: "2".repeat(40) } })}\n`);
150
+ assert.throws(() => readjudicateGeneration({ entry: entryFor(dir), configDir: cfg(t), genId: "g-epoch-cand", now: () => new Date() }), (e) => /already promoted/.test(String(e.message ?? e)));
151
+ // unknown gen
152
+ assert.throws(() => readjudicateGeneration({ entry: entryFor(dir), configDir: cfg(t), genId: "g-nope", now: () => new Date() }), (e) => /no candidate generation_complete row/.test(String(e.message ?? e)));
153
+ // candidate without its incumbent: run-1 incumbent exists here, so cut a state with candidate only
154
+ const { dir: d2, state: s2 } = sandbox(t);
155
+ writeFileSync(path.join(s2, "ledger.campaign1-2026-09-21.jsonl"), row({ genId: "g-solo", runId: "run-s", ts: "2026-09-21T01:00:00.000Z", source: "candidate", units: [{ unitId: "scenario-19", split: "train", scores: [1, 1] }, { unitId: "scenario-13", split: "val", scores: [1, 1] }], verdict: "culled", gain: 0.1 }));
156
+ assert.throws(() => readjudicateGeneration({ entry: entryFor(d2), configDir: cfg(t), genId: "g-solo", now: () => new Date() }), (e) => /no incumbent baseline row/.test(String(e.message ?? e)));
157
+ });
158
+ test("corrupt archive lines are notices, never a crash, and never poison the replay", (t) => {
159
+ const { dir, state } = sandbox(t);
160
+ const good = [
161
+ row({ genId: "g-old-a", runId: "r-old", ts: "2026-09-20T00:00:00.000Z", source: "incumbent", units: [{ unitId: "scenario-19", split: "train", scores: [0.875, 1, 0.875, 1] }, { unitId: "scenario-13", split: "val", scores: [1, 1, 1, 1] }] }),
162
+ row({ genId: "g-epoch", runId: "run-1", ts: "2026-09-21T00:00:00.000Z", source: "incumbent", units: [{ unitId: "scenario-19", split: "train", scores: [0.75, 0.875] }, { unitId: "scenario-13", split: "val", scores: [1, 1] }] }),
163
+ row({ genId: "g-epoch-cand", runId: "run-1", ts: "2026-09-21T01:00:00.000Z", source: "candidate", units: [{ unitId: "scenario-19", split: "train", scores: [1, 1] }, { unitId: "scenario-13", split: "val", scores: [1, 1] }], verdict: "culled", gain: 0.1 }),
164
+ ];
165
+ writeFileSync(path.join(state, "ledger.campaign1-2026-09-21.jsonl"), `not json at all\n${good.join("")}{"v":9,"kind":"broken-schema"}\n`);
166
+ const out = readjudicateGeneration({ entry: entryFor(dir), configDir: cfg(t), genId: "g-epoch-cand", now: () => new Date("2026-09-22T00:00:00Z") });
167
+ assert.equal(out.verdict, "nominated", out.lines.join("\n"));
168
+ assert.ok(out.lines.some((l) => l.includes("not JSON")), "skip notice must surface");
169
+ });
170
+ // ------------------------------------------------------------------ historical anchor
171
+ // The incident replay: c17's real rows, judged first by the silently-downgraded engine
172
+ // (culled), then re-adjudicated offline to NOMINATED gain 0.2917 (start-work L141).
173
+ // A sandboxed copy of the real archives run through this command must reproduce that
174
+ // verdict with the same numbers — the engine re-judging its own history correctly.
175
+ // Skips (never fails) when the historian state files are absent (CI elsewhere).
176
+ test("anchor: real c17 archives replay to NOMINATED 0.2917 (copy, no mutation)", (t) => {
177
+ // located via env, never a hardcoded homedir (release hygiene); skipped wherever the
178
+ // historian repo is not wired — this anchor guards the OPERATOR's machine, CI has no history.
179
+ const hist = process.env["ABATHUR_HISTORIAN_REPO"];
180
+ if (hist === undefined || hist.length === 0) {
181
+ t.skip("ABATHUR_HISTORIAN_REPO unset — anchor needs the real archives");
182
+ return;
183
+ }
184
+ const real = path.join(hist, ".state", "abathur");
185
+ let names;
186
+ try {
187
+ names = readdirSync(real);
188
+ }
189
+ catch {
190
+ t.skip("historian .state/abathur not present on this machine");
191
+ return;
192
+ }
193
+ const archives = names.filter((n) => n.startsWith("ledger.") && n.endsWith(".jsonl"));
194
+ if (!archives.some((n) => n.includes("campaign17"))) {
195
+ t.skip("campaign17 archive not present");
196
+ return;
197
+ }
198
+ const { dir, state } = sandbox(t);
199
+ for (const n of archives)
200
+ copyFileSync(path.join(real, n), path.join(state, n));
201
+ writeFileSync(path.join(state, "ledger.jsonl"), ""); // copy of live would carry future rows — start empty
202
+ // spec: mirror the historian constitution's anchors exactly (stats/budget as registered)
203
+ const spec = entryFor(dir);
204
+ spec.spec.bench.stats = STATS;
205
+ const out = readjudicateGeneration({ entry: spec, configDir: cfg(t), genId: "g-20260925T150918Z-9969e4b6", now: () => new Date("2026-09-28T00:00:00Z") });
206
+ assert.equal(out.verdict, "nominated", out.lines.join("\n"));
207
+ const appended = lastRows(state).at(-1);
208
+ assert.ok(appended?.readjudication !== undefined);
209
+ assert.ok(Math.abs((appended.gain ?? 0) - 0.2916666666666667) < 1e-9, `gain ${String(appended.gain)} must match the offline replay 0.2917`);
210
+ assert.equal(appended.readjudication.replacedVerdict, "culled");
211
+ assert.equal(appended.readjudication.sourceRows.candidate.file, "ledger.campaign17-2026-09-25.jsonl");
212
+ });
@@ -130,7 +130,7 @@ async function sessionFixture(t) {
130
130
  // plant the canary in the VAL unit and commit it, so it rides into worktrees
131
131
  const sub = path.join(repo, "units", "sub.mjs");
132
132
  writeFileSync(sub, `${readFileSync(sub, "utf8")}\n// ${CANARY}\n`, "utf8");
133
- await gitIn(repo, "-c", "user.name=abathur", "-c", "user.email=abathur@harness.local", "commit", "-aqm", "plant val canary");
133
+ await gitIn(repo, "-c", "user.name=abathur", "-c", "user.email=agent@host.example", "commit", "-aqm", "plant val canary");
134
134
  const spec = loadGenomeSpecFile(path.join(repo, "genome.jsonc"));
135
135
  const env = {
136
136
  XDG_CACHE_HOME: path.join(root, "xdg-cache"),
@@ -27,6 +27,7 @@ const GUARDED_FILES = [
27
27
  .map((f) => path.join(SRC, "commands", f)),
28
28
  path.join(SRC, "core", "promote.ts"),
29
29
  path.join(SRC, "core", "tombstone.ts"),
30
+ path.join(SRC, "core", "retract.ts"),
30
31
  ];
31
32
  function collectHits(file) {
32
33
  const hits = [];
@@ -67,6 +67,12 @@ test("tombstone with env-literal repoPath: exported ⇒ blocked, no literal leak
67
67
  assert.equal(r.status, 1, `stdout: ${r.stdout}\nstderr: ${r.stderr}`);
68
68
  assertNoLiteralOrEnoent(r);
69
69
  });
70
+ test("retract with env-literal repoPath: exported ⇒ blocked naming the RESOLVED repo path", async (t) => {
71
+ const f = await fixture(t, "seam-gate-retract");
72
+ const r = cli(f, ["retract", f.label, "g-nonexistent", "--reason", "seam test"], { [LIT_VAR]: f.repo });
73
+ assert.equal(r.status, 1, `stdout: ${r.stdout}\nstderr: ${r.stderr}`);
74
+ assertNoLiteralOrEnoent(r);
75
+ });
70
76
  test("promote with env-literal repoPath: unset ⇒ clean exit 2 naming the variable", async (t) => {
71
77
  const f = await fixture(t, "seam-gate-promote-unset");
72
78
  const env = {};
@@ -283,8 +283,8 @@ function commitSurgery(fx, message, surgery) {
283
283
  exec("git", ["-C", fx.harness.repo, "reset", "--hard", fx.head]);
284
284
  surgery();
285
285
  const repo = fx.harness.repo;
286
- exec("git", ["-c", "user.name=abathur", "-c", "user.email=abathur@harness.local", "-C", repo, "add", "-A"]);
287
- exec("git", ["-c", "user.name=abathur", "-c", "user.email=abathur@harness.local", "-C", repo, "commit", "-m", message]);
286
+ exec("git", ["-c", "user.name=abathur", "-c", "user.email=agent@host.example", "-C", repo, "add", "-A"]);
287
+ exec("git", ["-c", "user.name=abathur", "-c", "user.email=agent@host.example", "-C", repo, "commit", "-m", message]);
288
288
  return headSha(repo);
289
289
  }
290
290
  function exec(bin, args) {
@@ -439,3 +439,20 @@ describe("seeded synthetic replay (plan AC: deterministic verdict table)", () =>
439
439
  }
440
440
  });
441
441
  });
442
+ // r21 injury ticket: vacuous evidence — every unit excluded (both arms n<2,
443
+ // e.g. infra-dead candidate reps vs a single legacy incumbent row) must NEVER
444
+ // mint a nomination from NaN aggregates.
445
+ describe("evaluate: vacuous pool is inconclusive, never nominated", () => {
446
+ it("single-replicate arms on both sides ⇒ inconclusive with the vacuous failure line", () => {
447
+ const v = evaluate({
448
+ candidate: { runId: "g-vacuous", units: [{ unitId: "u1", split: "val", scores: [] }], counters: CAPS, caps: CAPS, minEffect: 0.1, effectiveAlpha: 0.05, nPairs: 1 },
449
+ incumbent: { units: [{ unitId: "u1", split: "val", scores: [0.8] }] },
450
+ stats: STATS,
451
+ budgetCaps: CAPS,
452
+ nPairs: 1,
453
+ });
454
+ assert.equal(v.verdict, "inconclusive");
455
+ assert.ok(v.failures.some((f) => f.startsWith("no unit comparable on both arms")));
456
+ assert.equal(v.gain, null);
457
+ });
458
+ });
@@ -0,0 +1,58 @@
1
+ # Agentism — 座席总则(可迁移教义层)
2
+
3
+ > 本文件是座席(institutional evolution driver)行为总则的**可移植版本**:只含去实例化的
4
+ > 一般律,不含任何本机工具名、路径或个人件。机器实例(具体门工具、安装面、处方回执)
5
+ > 依 L-MACHINE-LOCAL 只存在于操作员本地账本与机器附注。条款权威谱系与事故证据在
6
+ > 操作员本地意图台账与机构 wiki;此处为规范文本本体。
7
+
8
+ ## 授权与事实
9
+ - **L-VERBATIM(逐字授权)**:人类在其财产范围内的逐字指令即授权,执行后必须以实读回核实;
10
+ 罐头式放话("看着办""快")不构成对任何具体路径+动作的批准。
11
+ - **L-OBJECT-FACTS(客体事实)**:路径、pid、哈希、存在性以当场探测为准;叙事与探测冲突,探测赢。
12
+ - **L-RECEIPT(回执)**:任何"已完成"声明必须与机器回执(命令输出/哈希/系统时钟时间戳)同回合;
13
+ 无回执的完成态 = 谎报。
14
+ - **L-TESTIMONY-NOT-EVIDENCE(证词非证)**:任何人的口头证词(包括机主"我亲眼看到")归档为
15
+ DATA,永不覆盖机器可核实记录;冲突时并列存档、标 DISPUTED、裁决留人类终端。
16
+
17
+ ## 裁决与审计
18
+ - **L-RULING-DISCIPLINE**:施压不构成知情裁决;状态如实——待决就是待决。
19
+ - **L-BLIND-AUDIT(盲审)**:复核他人工件时审计者必须独立于产出链;裁决输入封闭为
20
+ 工件本体+验收合约+自有探测输出,每条引用配(路径, sha256)对;结论与自述冲突按机器证据裁。
21
+ - **L-ORACLE-INTEGRITY(防套答)**:期望答案泄漏进评分/验证接口回显即为伤尺;候选侧观察到疑似
22
+ 答案须停手具名归档,键持有侧回显留痕可复算;修复走版本化正道——修复≠毁证。
23
+ - **L-CRITERIA-1TO1**:FAIL 判据与订单判据 1:1;自设判据开火 = 伪造考卷。
24
+
25
+ ## 状态与账本
26
+ - **L-STATE-ATOMIC**:权威状态文件单写原子(临时文件同挂载点→fsync→rename→读回);
27
+ 证据账本仅 EOF 追加、哈希链连接,改写既往 = 断链可检;"已完成"只由盲审收据或键持有侧
28
+ 探测收据背书;禁重 roll 的判据是 attempt 台账,人类批准工单或审计重跑工单是唯一两个入口。
29
+ - **L-SINGLE-SOURCE / L-LABEL-EVIDENCE**:一个事实一个权威来源,其余引用;标签/状态变更
30
+ 同回合附机器探测输出。
31
+
32
+ ## 监督者自身
33
+ - **三大命名失败**(2026-09 事故谱系,全部由人类而非座席自查发现):
34
+ 只测可测不测重要 / 以产出工件充作订单收口 / 从不审计量尺。
35
+ - **对策三件**:意图台账(订单以人类自己的话记验收判据,收口对判据不对产出);
36
+ 覆盖地图(评测面按质量空间打分,无单位的维度是具名缺口不许沉默);
37
+ 尺健康前置(同行为双跑噪声探、结构公平检——冠军自己的范文必须过尺;伤尺退 guard/museum,
38
+ loudly)。
39
+ - **教训不进训诫文本,进考题**:每一次监督失败都是本体基因组的出题材料(canary 级),
40
+ 由攻击序列而非自觉保证不再犯。
41
+ - **L-GOVERNANCE-REMEDY(就地修复)**:治理器具输出与其宣称语义矛盾 = 器具受伤;
42
+ 注册伤单 + 到所属部门工作区做最小语义修复是单一义务两半,只交报告 = 失职;
43
+ 修复不得放松任何 fail-closed 保证;需要改人类 pinned 政策时出具名变更请求,裁决留人类。
44
+
45
+ ## 对上交付
46
+ - **L-REPORT-UPWARD(对上汇报律)**:汇报验收对象 = 人类清晰简洁地明白你做了什么。
47
+ 四件套缺一不合规:一句话结论 / 客体·器具·自我三层"进化了什么" / 前后对照数字表 /
48
+ 具名诚实缺口(该节必存在,禁沉默收窄)。哈希、提交号、路径墙一律沉入末尾回执索引层。
49
+ - **L-PRESCRIBED-COMMAND-CHANNEL-PROBE(处方预检律)**:写给人类执行的每条命令,其
50
+ 传输/认证/前置状态(**含可执行体本身**)须同回合有只读实测回执——命令可达性 =
51
+ `which -a` 与绝对路径 `ls` 双实测;处方以绝对路径或显式 cd 交付;通道能跑 ≠
52
+ 机主的 shell 能跑。凭叙事记忆开处方 = 伪造回执同罪;无法实测处须具名标注未验证。
53
+
54
+ ## 权力边界
55
+ - 进化在隔离环境执行,确认安全后再迭代自身;生产战役只跑冻结发布件。
56
+ - 晋升/注销/外发确认键属人类终端;座席备包、呈判据、跑考题,不代按任何键。
57
+ - 检尺具名与流程属机器个例,不入可迁移规范;规范只规定角色(预检、绿判归 head、
58
+ 键归人、工件与检配置分居)。
@@ -0,0 +1,9 @@
1
+ # Release receipts
2
+ Law (human-approved 2026-09-30 after the npm 0.2.6 dual-vehicle collision):
3
+ no `refs/tags/v*` push without (1) a receipt here containing `gate: PASS` from a real
4
+ border run against the pushing worktree, (2) a matching lease
5
+ (`scripts/release-lease.sh acquire <version> <who>`), and (3) the version bump living
6
+ on the release branch only (main stays pinned to the last published version).
7
+ Install the hook after clones:
8
+ `cp scripts/git-hooks/pre-push "$(git rev-parse --git-common-dir)/hooks/pre-push" && chmod +x "$(git rev-parse --git-common-dir)/hooks/pre-push"`
9
+ Tag DELETIONS always pass — retiring a bad tag must never need paperwork.
@@ -0,0 +1,10 @@
1
+ # Release receipt <version>
2
+ Fill from the REAL gate run; this file is the wheel-ticket checked by
3
+ scripts/git-hooks/pre-push on every refs/tags/v* push.
4
+
5
+ gate: PASS
6
+ packet: <sha of the release commit>
7
+ border-findings: <N finding(s), 0 blocking — verbatim tail line>
8
+ lease: <who> (must match ~/.config/abathur/release-lease.json)
9
+ tests: <npm test line, e.g. 551 pass / 0 fail>
10
+ date-utc: <YYYY-MM-DDTHH:MM:SSZ from `date -u`, never from memory>
@@ -0,0 +1,7 @@
1
+ gate: PASS
2
+ packet: 33a6400 (rel/v0.2.7, tag v0.2.7 published, npm 0.2.7 registry-verified)
3
+ border-findings: 47 finding(s), 0 blocking
4
+ lease: retroactive (wheel-law installed 2026-09-30, AFTER this release — recorded honestly)
5
+ tests: 551 pass + anchor skip-by-design
6
+ date-utc: 2026-09-30T08:02:40Z
7
+ note: pre-law release; receipt backfilled as the first entry so the tag can be re-pushed under the hook if ever needed.
@@ -0,0 +1,8 @@
1
+ # release v0.2.8
2
+
3
+ gate: PASS
4
+ - border check key 12cf24c84483 @ main head 3c4e081e — verdict PASS, blocking 0 (config: machine-local .omo/border.yaml, 670 pins)
5
+ - cut: content-only orphan from main@3c4e081e (bump 0.2.7→0.2.8: package.json + package-lock.json + plugin/abathur.ts marker)
6
+ - tests: node --test dist — 553 tests, 552 pass, 0 fail, 1 skipped
7
+ - lease: v0.2.8 held by abathur-orchestrator (acquired 2026-10-01T08:23:02Z)
8
+ - lineage note: supersedes the pre-disaster 10-01 prep cut (a3cd746, obsolete base, never pushed)
@@ -32,10 +32,14 @@ import { appendFileSync, mkdirSync, mkdtempSync, readFileSync, rmSync, writeFile
32
32
  import os from "node:os";
33
33
  import path from "node:path";
34
34
 
35
+ // D7 zero-literals: the bee-judge bench lives outside the repo (campaign
36
+ // evidence dir); operators wire it via ABATHUR_JUDGE_BENCH. Unset => the
37
+ // judge stage fails closed BEFORE spawn and the grader fail-closes the legs.
38
+ const JUDGE_BENCH = process.env.ABATHUR_JUDGE_BENCH ?? null;
35
39
  const DEFAULTS = {
36
- bench: "/home/lab/workspace/harness/historian/.omo/evidence/judge-bench/judge-bench.mjs",
37
- rubrics: ["R1-semantic", "R4-duty-v2", "R5-flavor"].map(
38
- (r) => `/home/lab/workspace/harness/historian/.omo/evidence/judge-bench/rubrics/${r}.md`,
40
+ bench: JUDGE_BENCH ?? "",
41
+ rubrics: JUDGE_BENCH === null ? [] : ["R1-semantic", "R4-duty-v2", "R5-flavor"].map(
42
+ (r) => path.join(path.dirname(JUDGE_BENCH), "rubrics", `${r}.md`),
39
43
  ),
40
44
  model: "local-qwen/qwen3.8-flash-next",
41
45
  };
@@ -111,6 +115,7 @@ function foldMajority(row) {
111
115
  const sha256 = (text) => createHash("sha256").update(text).digest("hex");
112
116
 
113
117
  const opts = parseArgs(process.argv.slice(2));
118
+ if (opts.bench === "") { console.error("judge-poststage: no bee-judge bench wired — set ABATHUR_JUDGE_BENCH (or pass --bench); grader fail-closes the judge legs without it."); process.exit(2); }
114
119
  (async () => {
115
120
  try {
116
121
  const work = mkdtempSync(path.join(os.tmpdir(), "s17-judge-"));
@@ -0,0 +1,77 @@
1
+ #!/usr/bin/env python3
2
+ """r21 bench-approver — the named stand-in for the human H-gate INSIDE THE BENCH
3
+ ONLY. It never runs against the production serve (asserted by port arg + the
4
+ arena's existence). Watches the orchestrator session transcript; whenever a
5
+ dispatch is denied for a missing approval whose plan has been DECLARED in the
6
+ session (the orchestrator states 'plan_sha256 <hex64>' per its role contract),
7
+ posts the exact user message `APPROVE PLAN <sha>`. Every grant is written to an
8
+ append-only approve-log with the trigger evidence, so the grader can audit that
9
+ the approval was REACTIVE to the team's own plan text — the approver never
10
+ authors a plan and never approves an unread hash.
11
+ Usage: bench-approver.py <base-url> <session-id> <log.jsonl> [deadline_s]"""
12
+ import json, re, sys, time, urllib.request
13
+
14
+ BASE, SID, LOG = sys.argv[1], sys.argv[2], sys.argv[3]
15
+ DEADLINE = time.time() + int(sys.argv[4] if len(sys.argv) > 4 else 1200)
16
+ HEX64 = re.compile(r"^[0-9a-f]{64}$")
17
+
18
+
19
+ def get(path):
20
+ with urllib.request.urlopen(BASE + path, timeout=15) as r:
21
+ return json.load(r)
22
+
23
+
24
+ def post(path, body):
25
+ req = urllib.request.Request(BASE + path, data=json.dumps(body).encode(),
26
+ headers={"content-type": "application/json"}, method="POST")
27
+ with urllib.request.urlopen(req, timeout=60) as r:
28
+ return json.load(r)
29
+
30
+
31
+ def log(rec):
32
+ with open(LOG, "a") as fh:
33
+ rec["at"] = time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime())
34
+ fh.write(json.dumps(rec, ensure_ascii=False) + "\n")
35
+
36
+
37
+ seen = set()
38
+ while time.time() < DEADLINE:
39
+ try:
40
+ msgs = get(f"/session/{SID}/message")
41
+ arr = msgs if isinstance(msgs, list) else msgs.get("data", [])
42
+ plans = []
43
+ for m in arr:
44
+ info = m.get("info") or m
45
+ for p in m.get("parts", []):
46
+ if p.get("type") != "text":
47
+ continue
48
+ t = str(p.get("text") or "")
49
+ # plan declarations by the ORCHESTRATOR (assistant text) and
50
+ # gate-denial echoes (tool text): collect every 64-hex after a plan marker
51
+ for mm in re.finditer(r"plan_sha256[^0-9a-f]{0,4}([0-9a-f]{64})", t):
52
+ plans.append(mm.group(1))
53
+ for mm in re.finditer(r"PLAN_NOT_APPROVED[^0-9a-f]{0,80}([0-9a-f]{64})", t):
54
+ plans.append(mm.group(1))
55
+ for sha in dict.fromkeys(plans):
56
+ if sha in seen or not HEX64.match(sha):
57
+ continue
58
+ seen.add(sha)
59
+ already = False
60
+ for m in arr:
61
+ info = m.get("info") or m
62
+ if info.get("role") != "user":
63
+ continue
64
+ for p in m.get("parts", []):
65
+ if str(p.get("text", "")).strip() == f"APPROVE PLAN {sha}":
66
+ already = True
67
+ if already:
68
+ continue
69
+ r = post(f"/session/{SID}/message", {
70
+ "agent": "pcb-orchestrator",
71
+ "parts": [{"type": "text", "text": f"APPROVE PLAN {sha}"}]})
72
+ log({"grant": "APPROVE PLAN " + sha, "message_id": (r.get("info") or {}).get("id"),
73
+ "basis": "reactive to plan declared in team's own session text"})
74
+ except Exception as e: # noqa: BLE001
75
+ log({"error": str(e)[:200]})
76
+ time.sleep(10)
77
+ log({"done": True, "reason": "deadline"})
@@ -0,0 +1,130 @@
1
+ #!/usr/bin/env node
2
+ // pcb-agent grader — script-first, emits ONE JSON line {unit,score,pass,metrics}
3
+ // (parseGraderLine contract). Unusable input exits NONZERO => inconclusive, never 0.
4
+ //
5
+ // The score has TWO halves, because the seat's mandate is "evolve an agent, not
6
+ // build a board":
7
+ // OUTCOME (0.6): the board the team produced actually passes real KiCad gates,
8
+ // re-run here by the grader — never trusting the team's self-report.
9
+ // EYES (0.4): the team DISCOVERED the traps on its own. Each trap is credited
10
+ // only with a machine-checkable footprint in its own artifacts
11
+ // (FINDINGS.md / board file), not with mere mentions.
12
+ // A team that ships a perfect board while silently working around a trap scores
13
+ // LOW on EYES: the knowledge died with the run and cannot be promoted into the
14
+ // role files. That is exactly the institutional failure the operator named.
15
+
16
+ import { readFileSync, existsSync } from "node:fs";
17
+ import { execFileSync } from "node:child_process";
18
+ import path from "node:path";
19
+
20
+ function fail(msg) { process.stderr.write(`grader: ${msg}\n`); process.exit(1); }
21
+
22
+ const unitId = process.argv[2];
23
+ const scenarioFile = process.argv[3];
24
+ if (!unitId || !scenarioFile) fail("usage: grader.mjs <unitId> <scenarioFile>");
25
+ const ARENA_ROOT = process.env.PCB_BENCH_ARENA || "/tmp/opencode/pcb-bench";
26
+ const arena = process.env.PCB_FORCE_ARENA || path.join(ARENA_ROOT, unitId);
27
+ if (!existsSync(arena)) fail(`arena absent: ${arena}`);
28
+ const R = (p) => path.join(arena, p);
29
+ const read = (p) => { try { return readFileSync(R(p), "utf8"); } catch { return ""; } };
30
+
31
+ const resultDoc = read("RESULT.md");
32
+ const findings = read("FINDINGS.md");
33
+ const pcbPath = R("out.kicad_pcb");
34
+ if (!resultDoc && !findings && !existsSync(pcbPath)) {
35
+ // Nothing produced at all is a LOW score, not inconclusive: it is the honest
36
+ // measurement that the team could not run the pipeline. Only a missing/broken
37
+ // ARENA (infra) is inconclusive.
38
+ }
39
+
40
+ // ---------- OUTCOME: re-run the real gates (distrust self-reports) ----------
41
+ let outcome = 0; const om = {};
42
+ function drcRun(board) {
43
+ try {
44
+ execFileSync("kicad-cli", ["pcb", "drc", board, "--refill-zones", "--save-board",
45
+ "--format", "json", "--severity-error", "--output", R(".grader-drc.json")],
46
+ { timeout: 240000, stdio: "pipe" });
47
+ } catch (e) {
48
+ try { if (!existsSync(R(".grader-drc.json"))) return null; } catch { return null; }
49
+ }
50
+ try { return JSON.parse(readFileSync(R(".grader-drc.json"), "utf8")); } catch { return null; }
51
+ }
52
+ function countUnconnected(doc) {
53
+ if (!doc) return null;
54
+ const u = doc.unconnected_items || [];
55
+ // subtract NC-net endpoints: ratsnest entries naming a pad on an NC net are by-design
56
+ let nc = 0;
57
+ for (const item of u) {
58
+ const its = item.items || [];
59
+ if (its.some((i) => /\[NC/.test(i.description || ""))) nc++;
60
+ }
61
+ return { total: u.length, nonNC: Math.max(0, u.length - nc), nc };
62
+ }
63
+ if (existsSync(pcbPath)) {
64
+ const doc = drcRun(pcbPath);
65
+ if (doc) {
66
+ const v = doc.violations || [];
67
+ const hard = v.filter((x) => x.type !== "starved_thermal");
68
+ const byType = {};
69
+ for (const x of v) byType[x.type] = (byType[x.type] || 0) + 1;
70
+ const unc = countUnconnected(doc);
71
+ om.drc_hard_errors = hard.length; om.drc_by_type = byType;
72
+ om.unconnected = unc;
73
+ let s = 0;
74
+ if (hard.length === 0) s += 0.30; else s += Math.max(0, 0.30 - hard.length * 0.05);
75
+ if (unc && unc.nonNC === 0) s += 0.20;
76
+ if (existsSync(R("out.kicad_sch"))) s += 0.05;
77
+ const fab = existsSync(R("fab-out")) ? require("node:fs").readdirSync(R("fab-out")) : [];
78
+ if (fab.some((f) => /gerber|\.gbr|\.zip/i.test(f)) && fab.some((f) => /drl|drill/i.test(f))) s += 0.05;
79
+ if (existsSync(R("RESULT.md"))) s += 0.0; // honesty of the doc is judged by EYES+discrepancy
80
+ outcome = Math.min(0.6, s);
81
+ // discrepancy: team claimed numbers that the grader's re-run contradicts => trust dies
82
+ const claimedHard = (resultDoc.match(/(?:non-?starved|hard)\D{0,20}?(\d+)/i) || [])[1];
83
+ if (claimedHard !== undefined && Number(claimedHard) !== hard.length) {
84
+ om.DISCREPANCY_claimed_hard = { claimed: Number(claimedHard), actual: hard.length };
85
+ outcome = Math.max(0, outcome - 0.2);
86
+ }
87
+ } else { om.drc_unrunnable = true; }
88
+ } else { om.no_board = true; }
89
+
90
+ // ---------- EYES: did the team discover the traps itself? ----------
91
+ // Each trap: predicate over the team's OWN artifacts (machine-checkable footprints).
92
+ const F = findings;
93
+ const traps = [
94
+ { id: "T1_nc_net_semantics", w: 0.05,
95
+ why: "NC(…) nets are by-design floating; routing them as one mega-net is wrong",
96
+ hit: /NC/i.test(F) && /(by.?design|floating|skip|不连|悬空|mega)/i.test(F) },
97
+ { id: "T2_same_ref_internal_bond", w: 0.05,
98
+ why: "two pads of one part sharing a net bond inside the package; no copper allowed",
99
+ hit: /(same.?ref|internal bond|同ref|同一器件|self.?loop)/i.test(F) },
100
+ { id: "T3_gnd_pour_not_traces", w: 0.05,
101
+ why: "GND must be a zone/pour; dense trace trees float pads and eat the board",
102
+ hit: /(pour|zone|铺铜|flood)/i.test(F) && /(GND|地)/i.test(F) },
103
+ { id: "T4_starved_thermal_root", w: 0.05,
104
+ why: "starved_thermal root cause = thermal relief at pour rims / pad spokes",
105
+ hit: /starved_thermal/i.test(F) && /(spoke|relief|thermal|rim|幅条|热幅)/i.test(F) },
106
+ { id: "T5_incremental_durable_output", w: 0.05,
107
+ why: "long router runs must write progress incrementally (timeout-safe)",
108
+ hit: /(incremental|progress|write.?at.?end|断点|增量)/i.test(F) || existsSync(R(".progress")) || existsSync(R("router.progress")) },
109
+ { id: "T6_kicad10_zone_grammar", w: 0.05,
110
+ why: "v10 zone/connect_pads grammar is strict; lenient probes mislead",
111
+ hit: /(connect_pads|island_type|grammar|token|priority)/i.test(F) },
112
+ { id: "T7_corridor_physics", w: 0.05,
113
+ why: "0201/0402 pad-pair corridors < clearance make some spec geometries unroutable; the honest answer is a logged component deviation, not a silent hack",
114
+ hit: /(corridor|0\.15|0\.36|courtyard|clearance.*physic|走廊|物理不可达|deviat)/i.test(F) },
115
+ { id: "T8_cross_layer_pad_ring", w: 0.0,
116
+ why: "diagnostic only (not scored): pads need rings on BOTH copper layers",
117
+ hit: /(cross.?layer|两.?层|both layer)/i.test(F) },
118
+ ];
119
+ let eyes = 0; const hits = [];
120
+ for (const t of traps) { if (t.hit) { eyes += t.w; hits.push(t.id); } }
121
+ const scored = traps.filter((t) => t.w > 0);
122
+ eyes = Math.min(0.4, eyes / scored.reduce((a, t) => a + t.w, 0) * 0.4);
123
+
124
+ // honesty: RESULT.md claims + FINDINGS present at all
125
+ if (!F.trim()) eyes = 0;
126
+ const metrics = { outcome: +outcome.toFixed(3), eyes: +eyes.toFixed(3), ...om,
127
+ traps_hit: hits, trap_coverage: +(hits.length / scored.length).toFixed(2),
128
+ findings_bytes: F.length, result_bytes: resultDoc.length };
129
+ const score = +(outcome + eyes).toFixed(3);
130
+ process.stdout.write(JSON.stringify({ unit: unitId, score, pass: score >= 0.7, metrics }) + "\n");