@tachikomagundam/abathur 0.2.4 → 0.2.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (36) hide show
  1. package/config/genomes/historian.example.jsonc +1 -1
  2. package/dist/commands/genome.js +2 -2
  3. package/dist/commands/graft.js +1 -1
  4. package/dist/commands/run.js +1 -1
  5. package/dist/commands/self-eval.js +1 -1
  6. package/dist/commands/tombstone.js +2 -1
  7. package/dist/core/evolve/run-bench.js +4 -3
  8. package/dist/core/evolve/run-loop.js +24 -4
  9. package/dist/core/evolve/run-plan.js +7 -3
  10. package/dist/core/evolve/score-bank.js +214 -0
  11. package/dist/core/graft-rebench.js +3 -0
  12. package/dist/core/ledger.js +6 -1
  13. package/dist/core/promote.js +2 -1
  14. package/dist/core/spec.js +5 -1
  15. package/dist/core/stats-math.js +39 -0
  16. package/dist/core/stats.js +125 -33
  17. package/dist/test/historian-grader-integrity.test.js +1494 -2
  18. package/dist/test/historian-grader-io.test.js +4 -0
  19. package/dist/test/historian-run-scenario.test.js +42 -2
  20. package/dist/test/repopath-seams.test.js +54 -0
  21. package/dist/test/score-bank.test.js +235 -0
  22. package/dist/test/seam-gates.test.js +78 -0
  23. package/dist/test/selfmode-literal.test.js +153 -0
  24. package/dist/test/stats-acceptance.test.js +329 -0
  25. package/dist/test/stats.test.js +18 -0
  26. package/graders/historian/grader-core.d.mts +107 -2
  27. package/graders/historian/grader-core.mjs +1046 -3
  28. package/graders/historian/grader.mjs +21 -2
  29. package/graders/historian/judge-poststage-19.mjs +212 -0
  30. package/graders/historian/judge-poststage.mjs +195 -0
  31. package/graders/historian/mutate.sh +59 -19
  32. package/graders/historian/run-scenario-17.sh +34 -0
  33. package/graders/historian/run-scenario-19.sh +43 -0
  34. package/graders/historian/run-scenario.sh +24 -1
  35. package/package.json +1 -1
  36. package/plugin/abathur.ts +1 -1
@@ -138,16 +138,35 @@ if (APPLICABLE[scenarioNo] !== undefined) {
138
138
  }
139
139
  const sandboxRows = state.post
140
140
  .filter((r) => isSandboxPath(r.path))
141
- .map((r) => ({ path: r.path, id: String(r.id), description: String(r.description ?? "") }));
141
+ .map((r) => ({ path: r.path, locale: String(r.locale ?? "en"), id: String(r.id), description: String(r.description ?? "") }));
142
142
  obs.tools = scanToolEvents(transcriptText);
143
- obs.integrity = {
143
+ const integrity = {
144
144
  sandboxRows,
145
145
  content: state.content,
146
146
  rowIdByPath: new Map(sandboxRows.map((r) => [r.path, r.id])),
147
+ rowIdByLocalePath: new Map(sandboxRows.map((r) => [`${r.path}\u0000${r.locale}`, r.id])),
147
148
  descByPath: new Map(sandboxRows.map((r) => [r.path, r.description])),
148
149
  seedDescByPath: new Map(seed.rows.map((r) => [r.path, String(r.description ?? "")])),
150
+ seedRowIdByLocalePath: new Map(seed.rows.map((r) => [`${String(r.path)}\u0000${String(r.locale ?? "en")}`, String(r.id)])),
149
151
  seedContent: Object.fromEntries(seed.rows.map((r) => [String(r.id), seed.content[String(r.id)] ?? ""])),
150
152
  };
153
+ // s17/s19 (蜂判卷): the run-stage poststage's instrument output is OBSERVABLE
154
+ // STATE at .bench/judge-verdicts.json (same flow as .bench/transcripts).
155
+ // Absent ⇒ undefined (branches 17/19 fail the 蜂判 legs closed with an explicit
156
+ // note); unparseable ⇒ {parseError}. Never thrown here — scoring stays
157
+ // deterministic given inputs, and an honest 0 beats a vacuous crash-to-inconclusive.
158
+ if (scenarioNo === 17 || scenarioNo === 19) {
159
+ const verdictsPath = path.join(process.cwd(), ".bench", "judge-verdicts.json");
160
+ try {
161
+ integrity.judgeVerdicts = JSON.parse(readFileSync(verdictsPath, "utf8"));
162
+ } catch (cause) {
163
+ const code = cause && typeof cause === "object" ? String(cause.code ?? "") : "";
164
+ integrity.judgeVerdicts = code === "ENOENT"
165
+ ? undefined
166
+ : { parseError: cause instanceof Error ? cause.message : String(cause) };
167
+ }
168
+ }
169
+ obs.integrity = integrity;
151
170
  }
152
171
 
153
172
  const result = scoreUnit(obs);
@@ -0,0 +1,212 @@
1
+ #!/usr/bin/env node
2
+ /**
3
+ * judge-poststage-19.mjs — scenario-19 蜂判 run-stage poststage (deterministic
4
+ * plumbing, zero deps, argv-only spawns).
5
+ *
6
+ * Clone-and-parameterize of judge-poststage.mjs (which stays byte-frozen under
7
+ * its own tests): same certified instrument (judge-bench.mjs pair/matrix, blind
8
+ * double runs, B4 §4 arbitration, fail-closed exit-2 contract), two shape
9
+ * changes demanded by the s19 架构法:
10
+ * 1. PAGE SET — `--page <path>=<locale>` is REPEATABLE: s19 judges the two
11
+ * created pages (dossier en + summary zh), each as a STANDALONE page.
12
+ * The R6-termb certificate covers the single-page UNTRUSTED DATA input
13
+ * shape only, so bodies are staged as separate neutral files and judged
14
+ * row-per-page — never concatenated into one multi-page doc.
15
+ * 2. RUBRICS LIST — defaults to the certified rubrics/R6-termb.md alone
16
+ * (verbatim reuse; the instrument bytes are the certification object).
17
+ * Offline fixture runs: `--file <label>=<path.md>`, label = index into the
18
+ * ordered --page specs (label-first keeps the `=`-bearing page=locale spec from
19
+ * colliding with the file path on a naive split).
20
+ *
21
+ * Usage:
22
+ * node judge-poststage-19.mjs --page _sandbox/eval19/warm-pool-dossier=en \
23
+ * --page _sandbox/eval19/warm-pool-summary=zh \
24
+ * --wiki-base http://localhost:3000 [--out .bench/judge-verdicts.json] \
25
+ * [--ledger .bench/judge-ledger.jsonl] [--reps 2] [--sleep-s 2] \
26
+ * [--timeout-s 240] [--model local-qwen/qwen3.8-flash-next] \
27
+ * [--rubrics a.md,b.md] [--bench judge-bench.mjs]
28
+ * node judge-poststage-19.mjs --page P=en --page Q=zh --files 0=p.md,1=q.md
29
+ * (offline legs are labelled by page-spec index or by the wiki path)
30
+ */
31
+ import { createHash } from "node:crypto";
32
+ import { execFileSync } from "node:child_process";
33
+ import { mkdirSync, mkdtempSync, readFileSync, rmSync, writeFileSync } from "node:fs";
34
+ import os from "node:os";
35
+ import path from "node:path";
36
+
37
+ const DEFAULTS = {
38
+ bench: "/home/lab/workspace/harness/historian/.omo/evidence/judge-bench/judge-bench.mjs",
39
+ rubrics: ["/home/lab/workspace/harness/historian/.omo/evidence/judge-bench/rubrics/R6-termb.md"],
40
+ model: "local-qwen/qwen3.8-flash-next",
41
+ };
42
+
43
+ function parseArgs(argv) {
44
+ const o = { ...DEFAULTS, rubrics: [...DEFAULTS.rubrics], out: path.join(".bench", "judge-verdicts.json"), ledger: null, reps: 2, sleepS: 2, timeoutS: 240, specs: [], wikiBase: null, files: null, title: process.env.S19_JUDGE_TITLE || "s19-bee" }; // session batch-tag (B): every judge spawn carries this title for audit/cleanup
45
+ for (let i = 0; i < argv.length; i += 1) {
46
+ const a = argv[i];
47
+ const need = () => { i += 1; if (i >= argv.length) throw new Error(`${a} needs a value`); return argv[i]; };
48
+ if (a === "--page") {
49
+ const v = need();
50
+ const eq = v.lastIndexOf("=");
51
+ if (eq <= 0 || eq === v.length - 1) throw new Error(`--page wants <path>=<locale>, got: ${v}`);
52
+ o.specs.push({ path: v.slice(0, eq), locale: v.slice(eq + 1) });
53
+ }
54
+ else if (a === "--wiki-base") o.wikiBase = need();
55
+ else if (a === "--files") o.files = need();
56
+ else if (a === "--rubrics") o.rubrics = need().split(",").filter(Boolean);
57
+ else if (a === "--bench") o.bench = need();
58
+ else if (a === "--model") o.model = need();
59
+ else if (a === "--out") o.out = need();
60
+ else if (a === "--ledger") o.ledger = need();
61
+ else if (a === "--reps") o.reps = Number(need());
62
+ else if (a === "--sleep-s") o.sleepS = Number(need());
63
+ else if (a === "--timeout-s") o.timeoutS = Number(need());
64
+ else throw new Error(`unknown arg: ${a}`);
65
+ }
66
+ if (o.specs.length === 0) throw new Error("need at least one --page <path>=<locale>");
67
+ if (new Set(o.specs.map((s) => `${s.path}|${s.locale}`)).size !== o.specs.length) throw new Error("duplicate --page specs");
68
+ if (o.wikiBase === null && o.files === null) throw new Error("need --wiki-base <url> (live) or --files <label>=<f>,... (offline fixtures)");
69
+ if (Number.isNaN(o.reps) || o.reps < 2) throw new Error("--reps must be ≥2 (盲评双跑 is the certified mode)");
70
+ if (o.files !== null) {
71
+ const parts = o.files.split(",");
72
+ const dup = new Set(parts.map((p) => p.split("=")[0]));
73
+ if (dup.size !== parts.length) throw new Error("--files has duplicate labels");
74
+ if (parts.length > o.specs.length) throw new Error(`--files carries ${String(parts.length)} legs for ${String(o.specs.length)} pages`);
75
+ }
76
+ return o;
77
+ }
78
+
79
+ const t0 = Date.now();
80
+ const log = (m) => process.stderr.write(`[judge-poststage-19] ${m}\n`);
81
+
82
+ async function fetchLocaleBody(base, pagePath, locale) {
83
+ const token = readFileSync(path.join(process.env.HOME ?? "", ".wikijs-api-key"), "utf8").trim();
84
+ const gql = async (query) => {
85
+ const res = await fetch(`${base.replace(/\/$/, "")}/graphql`, {
86
+ method: "POST",
87
+ headers: { "Content-Type": "application/json", Authorization: `Bearer ${token}` },
88
+ body: JSON.stringify({ query }),
89
+ });
90
+ if (!res.ok) throw new Error(`graphql http ${String(res.status)}`);
91
+ const doc = await res.json();
92
+ if (doc.errors !== undefined) throw new Error(`graphql ${JSON.stringify(doc.errors).slice(0, 200)}`);
93
+ return doc.data;
94
+ };
95
+ const list = (await gql("{ pages { list { id path locale } } }")).pages.list;
96
+ const row = list.find((r) => r.path === pagePath && String(r.locale ?? "en") === locale);
97
+ if (row === undefined) return null;
98
+ const one = (await gql(`{ pages { single(id: ${String(row.id)}) { content } } }`)).pages.single;
99
+ return one === null ? null : String(one.content ?? "");
100
+ }
101
+
102
+ function runBench(bench, args) {
103
+ // argv-only spawn (no shell), same contract as judge-bench's own isolation.
104
+ return execFileSync(process.execPath, [bench, ...args], { encoding: "utf8", stdio: ["ignore", "pipe", "pipe"], maxBuffer: 64 * 1024 * 1024 });
105
+ }
106
+
107
+ function ledgerRows(ledgerPath) {
108
+ try {
109
+ return readFileSync(ledgerPath, "utf8").split("\n").filter((l) => l.trim().length > 0).map((l) => JSON.parse(l));
110
+ } catch {
111
+ return [];
112
+ }
113
+ }
114
+
115
+ function foldMajority(row) {
116
+ const okRep = (rep, repNo) => rep !== null && rep !== undefined && rep.rep === repNo && rep.status === "ok" && (rep.score === 0 || rep.score === 1);
117
+ if (!okRep(row.rep1, 1) || !okRep(row.rep2, 2)) return null;
118
+ if (row.rep1.score === row.rep2.score) return row.rep1.score;
119
+ if (!okRep(row.rep3, 3)) return null; // unresolved (B4 §4-c): never silently agrees
120
+ return row.rep1.score === row.rep3.score ? row.rep1.score : row.rep2.score;
121
+ }
122
+
123
+ const sha256 = (text) => createHash("sha256").update(text).digest("hex");
124
+
125
+ const opts = parseArgs(process.argv.slice(2));
126
+ (async () => {
127
+ try {
128
+ const work = mkdtempSync(path.join(os.tmpdir(), "s19-judge-"));
129
+ const fileLegs = opts.files === null ? [] : opts.files.split(",").map((s) => {
130
+ const eq = s.indexOf("=");
131
+ return eq < 0 ? [s, ""] : [s.slice(0, eq), s.slice(eq + 1)];
132
+ });
133
+ const staged = new Map(); // specIndex -> staged neutral file
134
+ for (const [i, spec] of opts.specs.entries()) {
135
+ let body = null;
136
+ if (opts.files !== null) {
137
+ const pair = fileLegs.find(([label]) => label === String(i) || label === spec.path);
138
+ if (pair === undefined || pair[1].length === 0) { log(`--files has no leg for ${spec.path}=${spec.locale} — skipped (coverage fails closed downstream)`); continue; }
139
+ body = readFileSync(path.resolve(pair[1]), "utf8");
140
+ } else {
141
+ body = await fetchLocaleBody(opts.wikiBase, spec.path, spec.locale);
142
+ if (body === null) { log(`live wiki has no (${spec.path}, ${spec.locale}) row — skipped (coverage fails closed downstream)`); continue; }
143
+ }
144
+ const f = path.join(work, `p${String(i + 1)}.md`); // neutral names: page labels never enter the judge channel
145
+ writeFileSync(f, body);
146
+ staged.set(i, f);
147
+ }
148
+ if (staged.size === 0) {
149
+ rmSync(work, { recursive: true, force: true });
150
+ throw new Error("nothing to judge — every page body missing; write no verdicts file (grader fail-closes the 蜂判 legs)");
151
+ }
152
+
153
+ const ledgerPath = opts.ledger ?? path.join(".bench", "judge-ledger.jsonl");
154
+ mkdirSync(path.dirname(ledgerPath), { recursive: true });
155
+ const files = [...staged.values()].join(",");
156
+ log(`matrix: ${opts.rubrics.length} rubric(s) × ${String(staged.size)} page file(s) × ${String(opts.reps)} reps = ${String(opts.rubrics.length * staged.size * opts.reps)} calls`);
157
+ runBench(opts.bench, ["matrix", "--title", opts.title, "--rubrics", opts.rubrics.join(","), "--pages", files, "--reps", String(opts.reps),
158
+ "--sleep-s", String(opts.sleepS), "--timeout-s", String(opts.timeoutS), "--model", opts.model, "--ledger", ledgerPath]);
159
+ const fileToLeg = new Map([...staged.entries()].map(([i, f]) => [f, opts.specs[i]]));
160
+ let rows = ledgerRows(ledgerPath).filter((r) => fileToLeg.has(r.page));
161
+ if (rows.length === 0) {
162
+ rmSync(work, { recursive: true, force: true });
163
+ throw new Error("judge-bench emitted no ledger rows for the staged pages (instrument failure, not a verdict)");
164
+ }
165
+
166
+ const arbLedger = `${ledgerPath}.arb`;
167
+ let arbitrations = 0;
168
+ for (const row of rows) {
169
+ if (row.agree === true) continue;
170
+ arbitrations += 1;
171
+ const leg = fileToLeg.get(row.page);
172
+ log(`arbitration (B4 §4-b): ${row.rubric} × ${leg === undefined ? row.page : `${leg.path}=${leg.locale}`} — fresh isolated rep3`);
173
+ try {
174
+ runBench(opts.bench, ["pair", "--title", opts.title, "--rubric", opts.rubrics.find((r) => path.basename(r, ".md") === row.rubric) ?? row.rubric,
175
+ "--page", row.page, "--reps", "1", "--sleep-s", String(opts.sleepS), "--timeout-s", String(opts.timeoutS),
176
+ "--model", opts.model, "--ledger", arbLedger]);
177
+ const arb = ledgerRows(arbLedger).filter((r) => r.rubric === row.rubric && r.page === row.page).pop();
178
+ if (arb?.rep1 !== undefined) row.rep3 = { ...arb.rep1, rep: 3 };
179
+ } catch (e) {
180
+ log(`arbitration call failed: ${String(e.message).slice(0, 160)} — row stays UNRESOLVED`);
181
+ }
182
+ row.majority = foldMajority(row);
183
+ }
184
+ for (const row of rows) {
185
+ row.majority = row.majority ?? foldMajority(row);
186
+ }
187
+
188
+ const doc = {
189
+ generated: new Date().toISOString(),
190
+ unit: "scenario-19",
191
+ pages: opts.specs.map((s) => ({ page: s.path, locale: s.locale })),
192
+ mode: opts.files !== null ? "offline-files" : "live-wiki",
193
+ model: opts.model,
194
+ instrument: { bench: opts.bench, title: opts.title, rubrics: opts.rubrics.map((r) => ({ file: r, name: path.basename(r, ".md"), sha256: sha256(readFileSync(r, "utf8")) })), reps: opts.reps, sleepS: opts.sleepS, timeoutS: opts.timeoutS, ledger: ledgerPath, arbitrations },
195
+ wall_ms: Date.now() - t0,
196
+ rows: rows.map((r) => ({ ...r, page: String(fileToLeg.get(r.page)?.path ?? r.page), locale: String(fileToLeg.get(r.page)?.locale ?? "en") })),
197
+ };
198
+
199
+ mkdirSync(path.dirname(opts.out), { recursive: true });
200
+ writeFileSync(opts.out, JSON.stringify(doc, null, 1) + "\n");
201
+ rmSync(work, { recursive: true, force: true });
202
+ const counts = { ok: 0, zero: 0, unresolved: 0 };
203
+ for (const r of doc.rows) { if (r.majority === 1) counts.ok += 1; else if (r.majority === 0) counts.zero += 1; else counts.unresolved += 1; }
204
+ log(`wrote ${opts.out}: ${String(doc.rows.length)} rows, majority 1×${String(counts.ok)} 0×${String(counts.zero)} unresolved×${String(counts.unresolved)}, wall ${String(Math.round(doc.wall_ms / 1000))}s, arbitrations ${String(arbitrations)}`);
205
+ process.stdout.write(JSON.stringify({ out: opts.out, rows: doc.rows.length, ...counts, arbitrations, wall_ms: doc.wall_ms }) + "\n");
206
+ } catch (e) {
207
+ // instrument failure ≠ a verdict: exit 2 (inconclusive plumbing), no file —
208
+ // the grader's fail-closed 蜂判 legs own the honest 0 with an explicit note.
209
+ process.stderr.write(`[judge-poststage-19] FATAL ${e instanceof Error ? e.message : String(e)}\n`);
210
+ process.exit(2);
211
+ }
212
+ })();
@@ -0,0 +1,195 @@
1
+ #!/usr/bin/env node
2
+ /**
3
+ * judge-poststage.mjs — scenario-17 蜂判 run-stage poststage (deterministic
4
+ * plumbing, zero deps, argv-only spawns).
5
+ *
6
+ * Doctrine binding (historian .omo/evidence/good-wiki-readability-doctrine-FINAL.md):
7
+ * 总则2 [蜂判] = 传感器:盲评双跑、分歧仲裁、一致率入法官健康台账。
8
+ * 总则4 grader 只算结构/状态/工具事件;语义一律蜂判 ⇒ the LLM calls live
9
+ * HERE (run stage), never inside the shipped deterministic grader.
10
+ * B4 §4 arbitration = fresh isolated third run on the SAME (rubric, page),
11
+ * appended as rep3; majority 2/3; unresolved never agrees.
12
+ *
13
+ * Flow: materialize the created dossier pair (live wiki fetch by (path, locale)
14
+ * — read-only — or --files locale=path for offline fixture runs) → invoke
15
+ * judge-bench.mjs matrix (the certified instrument, imported by CLI spawn, NOT
16
+ * re-implemented) → map ledger rows onto {page, locale} → for every agree=false
17
+ * row, one `pair --reps 1` arbitration run appended as rep3 → fold the majority
18
+ * per row → write .bench/judge-verdicts.json (grader consumes it as observable
19
+ * state; s17FoldJudgeVerdicts re-derives everything fail-closed).
20
+ *
21
+ * Usage:
22
+ * node judge-poststage.mjs --page _sandbox/eval17/gpu-warm-pool-dossier \
23
+ * --wiki-base http://localhost:3000 [--out .bench/judge-verdicts.json] \
24
+ * [--ledger .bench/judge-ledger.jsonl] [--reps 2] [--sleep-s 2] \
25
+ * [--timeout-s 240] [--model local-qwen/qwen3.8-flash-next] \
26
+ * [--rubrics a.md,b.md,c.md] [--bench judge-bench.mjs]
27
+ * node judge-poststage.mjs --page <label> --files en=g.md,zh=d.md [...]
28
+ */
29
+ import { createHash } from "node:crypto";
30
+ import { execFileSync } from "node:child_process";
31
+ import { appendFileSync, mkdirSync, mkdtempSync, readFileSync, rmSync, writeFileSync } from "node:fs";
32
+ import os from "node:os";
33
+ import path from "node:path";
34
+
35
+ const DEFAULTS = {
36
+ bench: "/home/lab/workspace/harness/historian/.omo/evidence/judge-bench/judge-bench.mjs",
37
+ rubrics: ["R1-semantic", "R4-duty-v2", "R5-flavor"].map(
38
+ (r) => `/home/lab/workspace/harness/historian/.omo/evidence/judge-bench/rubrics/${r}.md`,
39
+ ),
40
+ model: "local-qwen/qwen3.8-flash-next",
41
+ };
42
+
43
+ function parseArgs(argv) {
44
+ const o = { ...DEFAULTS, out: path.join(".bench", "judge-verdicts.json"), ledger: null, reps: 2, sleepS: 2, timeoutS: 240, page: null, wikiBase: null, files: null };
45
+ for (let i = 0; i < argv.length; i += 1) {
46
+ const a = argv[i];
47
+ const need = () => { i += 1; if (i >= argv.length) throw new Error(`${a} needs a value`); return argv[i]; };
48
+ if (a === "--page") o.page = need();
49
+ else if (a === "--wiki-base") o.wikiBase = need();
50
+ else if (a === "--files") o.files = need();
51
+ else if (a === "--rubrics") o.rubrics = need().split(",").filter(Boolean);
52
+ else if (a === "--bench") o.bench = need();
53
+ else if (a === "--model") o.model = need();
54
+ else if (a === "--out") o.out = need();
55
+ else if (a === "--ledger") o.ledger = need();
56
+ else if (a === "--reps") o.reps = Number(need());
57
+ else if (a === "--sleep-s") o.sleepS = Number(need());
58
+ else if (a === "--timeout-s") o.timeoutS = Number(need());
59
+ else throw new Error(`unknown arg: ${a}`);
60
+ }
61
+ if (o.page === null) throw new Error("--page <wiki path or label> required");
62
+ if (o.wikiBase === null && o.files === null) throw new Error("need --wiki-base <url> (live) or --files en=<f>,zh=<f> (offline fixtures)");
63
+ if (Number.isNaN(o.reps) || o.reps < 2) throw new Error("--reps must be ≥2 (盲评双跑 is the certified mode)");
64
+ return o;
65
+ }
66
+
67
+ const t0 = Date.now();
68
+ const log = (m) => process.stderr.write(`[judge-poststage] ${m}\n`);
69
+
70
+ async function fetchLocaleBody(base, pagePath, locale) {
71
+ const token = readFileSync(path.join(process.env.HOME ?? "", ".wikijs-api-key"), "utf8").trim();
72
+ const gql = async (query) => {
73
+ const res = await fetch(`${base.replace(/\/$/, "")}/graphql`, {
74
+ method: "POST",
75
+ headers: { "Content-Type": "application/json", Authorization: `Bearer ${token}` },
76
+ body: JSON.stringify({ query }),
77
+ });
78
+ if (!res.ok) throw new Error(`graphql http ${String(res.status)}`);
79
+ const doc = await res.json();
80
+ if (doc.errors !== undefined) throw new Error(`graphql ${JSON.stringify(doc.errors).slice(0, 200)}`);
81
+ return doc.data;
82
+ };
83
+ const list = (await gql("{ pages { list { id path locale } } }")).pages.list;
84
+ const row = list.find((r) => r.path === pagePath && String(r.locale ?? "en") === locale);
85
+ if (row === undefined) return null;
86
+ const one = (await gql(`{ pages { single(id: ${String(row.id)}) { content } } }`)).pages.single;
87
+ return one === null ? null : String(one.content ?? "");
88
+ }
89
+
90
+ function runBench(bench, args) {
91
+ // argv-only spawn (no shell), same contract as judge-bench's own isolation.
92
+ return execFileSync(process.execPath, [bench, ...args], { encoding: "utf8", stdio: ["ignore", "pipe", "pipe"], maxBuffer: 64 * 1024 * 1024 });
93
+ }
94
+
95
+ function ledgerRows(ledgerPath) {
96
+ try {
97
+ return readFileSync(ledgerPath, "utf8").split("\n").filter((l) => l.trim().length > 0).map((l) => JSON.parse(l));
98
+ } catch {
99
+ return [];
100
+ }
101
+ }
102
+
103
+ function foldMajority(row) {
104
+ const okRep = (rep, repNo) => rep !== null && rep !== undefined && rep.rep === repNo && rep.status === "ok" && (rep.score === 0 || rep.score === 1);
105
+ if (!okRep(row.rep1, 1) || !okRep(row.rep2, 2)) return null;
106
+ if (row.rep1.score === row.rep2.score) return row.rep1.score;
107
+ if (!okRep(row.rep3, 3)) return null; // unresolved (B4 §4-c): never silently agrees
108
+ return row.rep1.score === row.rep3.score ? row.rep1.score : row.rep2.score;
109
+ }
110
+
111
+ const sha256 = (text) => createHash("sha256").update(text).digest("hex");
112
+
113
+ const opts = parseArgs(process.argv.slice(2));
114
+ (async () => {
115
+ try {
116
+ const work = mkdtempSync(path.join(os.tmpdir(), "s17-judge-"));
117
+ const localeFile = new Map();
118
+ const targets = [["en", "en.md"], ["zh", "zh.md"]];
119
+ for (const [locale, fname] of targets) {
120
+ let body = null;
121
+ if (opts.files !== null) {
122
+ const pair = opts.files.split(",").map((s) => s.split("=")).find(([k]) => k === locale);
123
+ if (pair === undefined) { log(`--files has no ${locale} leg — skipped (coverage fails closed downstream)`); continue; }
124
+ body = readFileSync(path.resolve(pair[1]), "utf8");
125
+ } else {
126
+ body = await fetchLocaleBody(opts.wikiBase, opts.page, locale);
127
+ if (body === null) { log(`live wiki has no (${opts.page}, ${locale}) row — skipped (coverage fails closed downstream)`); continue; }
128
+ }
129
+ const f = path.join(work, fname); // neutral names: page label never enters the judge channel
130
+ writeFileSync(f, body);
131
+ localeFile.set(locale, f);
132
+ }
133
+ if (localeFile.size === 0) {
134
+ rmSync(work, { recursive: true, force: true });
135
+ throw new Error("nothing to judge — both locale bodies missing; write no verdicts file (grader fail-closes the 蜂判 legs)");
136
+ }
137
+
138
+ const ledgerPath = opts.ledger ?? path.join(".bench", "judge-ledger.jsonl");
139
+ mkdirSync(path.dirname(ledgerPath), { recursive: true });
140
+ const files = [...localeFile.values()].join(",");
141
+ log(`matrix: ${opts.rubrics.length} rubrics × ${String(localeFile.size)} locale file(s) × ${String(opts.reps)} reps = ${String(opts.rubrics.length * localeFile.size * opts.reps)} calls`);
142
+ runBench(opts.bench, ["matrix", "--rubrics", opts.rubrics.join(","), "--pages", files, "--reps", String(opts.reps),
143
+ "--sleep-s", String(opts.sleepS), "--timeout-s", String(opts.timeoutS), "--model", opts.model, "--ledger", ledgerPath]);
144
+ const fileToTag = new Map([...localeFile.entries()].map(([loc, f]) => [f, loc]));
145
+ let rows = ledgerRows(ledgerPath).filter((r) => fileToTag.has(r.page));
146
+ if (rows.length === 0) {
147
+ rmSync(work, { recursive: true, force: true });
148
+ throw new Error("judge-bench emitted no ledger rows for the staged pages (instrument failure, not a verdict)");
149
+ }
150
+
151
+ const arbLedger = `${ledgerPath}.arb`;
152
+ let arbitrations = 0;
153
+ for (const row of rows) {
154
+ if (row.agree === true) continue;
155
+ arbitrations += 1;
156
+ log(`arbitration (B4 §4-b): ${row.rubric} × ${fileToTag.get(row.page)} — fresh isolated rep3`);
157
+ try {
158
+ runBench(opts.bench, ["pair", "--rubric", opts.rubrics.find((r) => path.basename(r, ".md") === row.rubric) ?? row.rubric,
159
+ "--page", row.page, "--reps", "1", "--sleep-s", String(opts.sleepS), "--timeout-s", String(opts.timeoutS),
160
+ "--model", opts.model, "--ledger", arbLedger]);
161
+ const arb = ledgerRows(arbLedger).filter((r) => r.rubric === row.rubric && r.page === row.page).pop();
162
+ if (arb?.rep1 !== undefined) row.rep3 = { ...arb.rep1, rep: 3 };
163
+ } catch (e) {
164
+ log(`arbitration call failed: ${String(e.message).slice(0, 160)} — row stays UNRESOLVED`);
165
+ }
166
+ row.majority = foldMajority(row);
167
+ }
168
+ for (const row of rows) {
169
+ row.majority = row.majority ?? foldMajority(row);
170
+ }
171
+
172
+ const doc = {
173
+ generated: new Date().toISOString(),
174
+ unit: "scenario-17",
175
+ page: opts.page,
176
+ mode: opts.files !== null ? "offline-files" : "live-wiki",
177
+ model: opts.model,
178
+ instrument: { bench: opts.bench, rubrics: opts.rubrics.map((r) => ({ file: r, name: path.basename(r, ".md"), sha256: sha256(readFileSync(r, "utf8")) })), reps: opts.reps, sleepS: opts.sleepS, timeoutS: opts.timeoutS, ledger: ledgerPath, arbitrations },
179
+ wall_ms: Date.now() - t0,
180
+ rows: rows.map((r) => ({ ...r, page: opts.page, locale: fileToTag.get(r.page) ?? String(r.page) })),
181
+ };
182
+ mkdirSync(path.dirname(opts.out), { recursive: true });
183
+ writeFileSync(opts.out, JSON.stringify(doc, null, 1) + "\n");
184
+ rmSync(work, { recursive: true, force: true });
185
+ const counts = { ok: 0, zero: 0, unresolved: 0 };
186
+ for (const r of doc.rows) { if (r.majority === 1) counts.ok += 1; else if (r.majority === 0) counts.zero += 1; else counts.unresolved += 1; }
187
+ log(`wrote ${opts.out}: ${String(doc.rows.length)} rows, majority 1×${String(counts.ok)} 0×${String(counts.zero)} unresolved×${String(counts.unresolved)}, wall ${String(Math.round(doc.wall_ms / 1000))}s, arbitrations ${String(arbitrations)}`);
188
+ process.stdout.write(JSON.stringify({ out: opts.out, rows: doc.rows.length, ...counts, arbitrations, wall_ms: doc.wall_ms }) + "\n");
189
+ } catch (e) {
190
+ // instrument failure ≠ a verdict: exit 2 (inconclusive plumbing), no file —
191
+ // the grader's fail-closed 蜂判 legs own the honest 0 with an explicit note.
192
+ process.stderr.write(`[judge-poststage] FATAL ${e instanceof Error ? e.message : String(e)}\n`);
193
+ process.exit(2);
194
+ }
195
+ })();
@@ -36,10 +36,10 @@ raw="${ABATHUR_MUTATOR_RAW:-/tmp/abathur-mutate-raw.jsonl}"
36
36
  rc=$?
37
37
  echo "mutate: opencode rc=$rc raw=$raw" >&2
38
38
 
39
- python3 - "$raw" "$canonical" "$rc" <<'PY'
40
- import json, re, sys
39
+ python3 - "$raw" "$canonical" "$rc" "$worktree" <<'PY'
40
+ import json, os, re, sys
41
41
 
42
- raw, canonical_path, rc = sys.argv[1], sys.argv[2], int(sys.argv[3])
42
+ raw, canonical_path, rc, worktree = sys.argv[1], sys.argv[2], int(sys.argv[3]), sys.argv[4]
43
43
 
44
44
  def create_diff(path, text):
45
45
  lines = text.split("\n")
@@ -73,19 +73,55 @@ for text in reversed(last_text_events(raw)):
73
73
  fence = re.search(r"```(?:json)?\s*(\{.*\})\s*```", candidate, re.S)
74
74
  if fence:
75
75
  candidate = fence.group(1)
76
- start, end = candidate.find("{"), candidate.rfind("}")
77
- if start < 0 or end <= start:
76
+ end = candidate.rfind("}")
77
+ if end < 0:
78
78
  continue
79
- try:
80
- doc = json.loads(candidate[start : end + 1])
81
- except ValueError:
82
- continue
83
- if isinstance(doc, dict):
79
+ # prose preambles can contain set-notation braces like {症状/Symptoms, ...} —
80
+ # the first '{' is not necessarily JSON; try every open brace from the LAST
81
+ # backwards so the outermost-to-last-close object wins when one parses.
82
+ opens = [i for i, ch in enumerate(candidate) if ch == "{" and i < end]
83
+ doc = None
84
+ for start in reversed(opens):
85
+ try:
86
+ parsed = json.loads(candidate[start : end + 1])
87
+ except ValueError:
88
+ continue
89
+ if isinstance(parsed, dict):
90
+ doc = parsed
91
+ break
92
+ if doc is not None:
84
93
  proposal = doc
85
94
  break
86
95
 
87
96
  spec_text = open(canonical_path, encoding="utf-8").read()
88
- packaging = create_diff("genome.jsonc", spec_text if spec_text.endswith("\n") else spec_text + "\n")
97
+ if not spec_text.endswith("\n"):
98
+ spec_text += "\n"
99
+ # genome.jsonc in the base tree was written by this wrapper's own prior round,
100
+ # so a replace-all modify hunk always anchors deterministically at line 1.
101
+ # A CREATE here would break apply the moment the file lands in the base (udiff
102
+ # refuses create-over-existing and discards the WHOLE candidate).
103
+ gen_path = os.path.join(worktree, "genome.jsonc")
104
+ try:
105
+ with open(gen_path, encoding="utf-8") as fh:
106
+ base_text = fh.read()
107
+ except FileNotFoundError:
108
+ packaging = create_diff("genome.jsonc", spec_text)
109
+ except OSError as err:
110
+ sys.stderr.write("mutate: genome.jsonc unreadable: %s\n" % err)
111
+ sys.exit(3)
112
+ else:
113
+ if base_text == spec_text:
114
+ packaging = None
115
+ else:
116
+ base_body = base_text.split("\n")
117
+ if base_body and base_body[-1] == "":
118
+ base_body.pop()
119
+ old_h = "".join("-" + ln + "\n" for ln in base_body)
120
+ new_h = "".join("+" + ln + "\n" for ln in spec_text.split("\n")[:-1])
121
+ packaging = (
122
+ "--- a/genome.jsonc\n+++ b/genome.jsonc\n"
123
+ "@@ -1,%d +1,%d @@\n%s%s" % (len(base_body), len(spec_text.split("\n")) - 1, old_h, new_h)
124
+ )
89
125
 
90
126
  if isinstance(proposal, dict):
91
127
  path = proposal.get("path")
@@ -103,15 +139,19 @@ if isinstance(proposal, dict):
103
139
  content = content.replace("\r", "")
104
140
  if not content.endswith("\n"):
105
141
  content += "\n"
106
- candidate = {"id": "mutate-live", "rationale": rationale, "diffs": [create_diff(path, content), packaging]}
142
+ diffs = [create_diff(path, content)] + ([packaging] if packaging else [])
143
+ candidate = {"id": "mutate-live", "rationale": rationale, "diffs": diffs}
107
144
  print(json.dumps({"candidates": [candidate]}))
108
145
  sys.exit(0)
109
146
 
110
- print(json.dumps({
111
- "candidates": [{
112
- "id": "mutate-package-only",
113
- "rationale": "packaging-only candidate: model output unparseable or path rejected (opencode rc=%d); lineage genome.jsonc added, no skill mutation proposed" % rc,
114
- "diffs": [packaging],
115
- }]
116
- }))
147
+ if packaging:
148
+ print(json.dumps({
149
+ "candidates": [{
150
+ "id": "mutate-package-only",
151
+ "rationale": "packaging-only candidate: model output unparseable or path rejected (opencode rc=%d); lineage genome.jsonc added, no skill mutation proposed" % rc,
152
+ "diffs": [packaging],
153
+ }]
154
+ }))
155
+ else:
156
+ print(json.dumps({"candidates": []}))
117
157
  PY
@@ -0,0 +1,34 @@
1
+ #!/usr/bin/env bash
2
+ # scenario-17 run hook = run-scenario.sh (agent stage, c7 ≤8min band) + the 蜂判
3
+ # poststage (judge-poststage.mjs) against the ACTUAL pages the agent produced.
4
+ # Mirrors the .bench/transcripts flow: the shipped grader never calls a model;
5
+ # the LLM measurement is instrument output staged at run time into
6
+ # .bench/judge-verdicts.json, which grader.mjs reads as observable state
7
+ # (fail-closed when absent). The LAST stdout line stays run-scenario.sh's
8
+ # {unit,tokensEst,turns,opencodeExit} meta JSON — parseRunMeta's contract — so
9
+ # the engine-facing format is byte-compatible with units 01–16.
10
+ set -euo pipefail
11
+ unit_id="${1:?usage: run-scenario-17.sh <unitId> <scenario-file> [repoRoot]}"
12
+ scenario_file="${2:?usage: run-scenario-17.sh <unitId> <scenario-file> [repoRoot]}"
13
+ here="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
14
+ dossier="${S17_DOSSIER:-_sandbox/eval17/gpu-warm-pool-dossier}"
15
+ wiki_base="${ABATHUR_WIKI_BASE:-http://localhost:3000}"
16
+ meta=""
17
+ agent_s=0
18
+ judge_s=0
19
+ if [ "${S17_SKIP_AGENT:-0}" != "1" ]; then
20
+ t=$SECONDS
21
+ meta="$(bash "$here/run-scenario.sh" "$@")"
22
+ agent_s=$((SECONDS - t))
23
+ fi
24
+ t=$SECONDS
25
+ node "$here/judge-poststage.mjs" --page "$dossier" --wiki-base "$wiki_base" \
26
+ --out "${ABATHUR_JUDGE_VERDICTS:-.bench/judge-verdicts.json}" \
27
+ --ledger "${ABATHUR_JUDGE_LEDGER:-.bench/judge-ledger-${unit_id}.jsonl}" 1>&2 || echo "[run-scenario-17] judge stage FAILED (no verdicts file — grader fail-closes the 蜂判 legs)" >&2
28
+ judge_s=$((SECONDS - t))
29
+ echo "[run-scenario-17] stage walls: agent=${agent_s}s judge=${judge_s}s (unit=${unit_id})" >&2
30
+ if [ -n "$meta" ]; then
31
+ printf '%s\n' "$meta"
32
+ else
33
+ printf '%s\n' "{\"unit\": \"$unit_id\", \"tokensEst\": 0, \"turns\": 0, \"opencodeExit\": -1}"
34
+ fi
@@ -0,0 +1,43 @@
1
+ #!/usr/bin/env bash
2
+ # scenario-19 run hook = run-scenario.sh (agent stage, 1200s ceiling) + the R6-termb
3
+ # 蜂判 poststage (judge-poststage-19.mjs, per-page certified shape) against the
4
+ # ACTUAL pages the agent produced. Same data flow as run-scenario-17.sh (which is
5
+ # left untouched): the shipped grader never calls a model; the LLM measurement is
6
+ # instrument output staged at run time into .bench/judge-verdicts.json, which
7
+ # grader.mjs reads as observable state (fail-closed when absent). The matrix wall
8
+ # budget is pinned via JUDGE_BENCH_DEADLINE (+S19_JUDGE_BUDGET_S, default 600s):
9
+ # a truncated matrix leaves coverage incomplete ⇒ I fail-closes, never silently.
10
+ # The LAST stdout line stays run-scenario.sh's {unit,tokensEst,turns,opencodeExit}
11
+ # meta JSON — parseRunMeta's contract — so the engine-facing format is byte-
12
+ # compatible with units 01–18.
13
+ set -euo pipefail
14
+ unit_id="${1:?usage: run-scenario-19.sh <unitId> <scenario-file> [repoRoot]}"
15
+ scenario_file="${2:?usage: run-scenario-19.sh <unitId> <scenario-file> [repoRoot]}"
16
+ here="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
17
+ pages="${S19_PAGES:-_sandbox/eval19/warm-pool-dossier=en,_sandbox/eval19/warm-pool-summary=zh}"
18
+ wiki_base="${ABATHUR_WIKI_BASE:-http://localhost:3000}"
19
+ agent_timeout_s="${S19_AGENT_TIMEOUT_S:-1200}"
20
+ judge_budget_s="${S19_JUDGE_BUDGET_S:-600}"
21
+ meta=""
22
+ agent_s=0
23
+ judge_s=0
24
+ if [ "${S19_SKIP_AGENT:-0}" != "1" ]; then
25
+ t=$SECONDS
26
+ meta="$(timeout --foreground -k 30 "${agent_timeout_s}" bash "$here/run-scenario.sh" "$@" || true)"
27
+ agent_s=$((SECONDS - t))
28
+ fi
29
+ page_args=()
30
+ IFS=',' read -ra specs <<< "$pages"
31
+ for s in "${specs[@]}"; do page_args+=(--page "$s"); done
32
+ t=$SECONDS
33
+ export JUDGE_BENCH_DEADLINE="$(( ( $(date +%s) + judge_budget_s ) * 1000 ))"
34
+ node "$here/judge-poststage-19.mjs" "${page_args[@]}" --wiki-base "$wiki_base" \
35
+ --out "${ABATHUR_JUDGE_VERDICTS:-.bench/judge-verdicts.json}" \
36
+ --ledger "${ABATHUR_JUDGE_LEDGER:-.bench/judge-ledger-${unit_id}.jsonl}" 1>&2 || echo "[run-scenario-19] judge stage FAILED (no verdicts file — grader fail-closes the 蜂判 legs)" >&2
37
+ judge_s=$((SECONDS - t))
38
+ echo "[run-scenario-19] stage walls: agent=${agent_s}s (ceiling ${agent_timeout_s}s) judge=${judge_s}s (budget ${judge_budget_s}s) (unit=${unit_id})" >&2
39
+ if [ -n "$meta" ]; then
40
+ printf '%s\n' "$meta"
41
+ else
42
+ printf '%s\n' "{\"unit\": \"$unit_id\", \"tokensEst\": 0, \"turns\": 0, \"opencodeExit\": -1}"
43
+ fi