@tachikomagundam/abathur 0.2.4 → 0.2.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/config/genomes/historian.example.jsonc +1 -1
- package/dist/commands/genome.js +2 -2
- package/dist/commands/graft.js +1 -1
- package/dist/commands/run.js +1 -1
- package/dist/commands/self-eval.js +1 -1
- package/dist/commands/tombstone.js +2 -1
- package/dist/core/evolve/run-bench.js +4 -3
- package/dist/core/evolve/run-loop.js +24 -4
- package/dist/core/evolve/run-plan.js +7 -3
- package/dist/core/evolve/score-bank.js +214 -0
- package/dist/core/graft-rebench.js +3 -0
- package/dist/core/ledger.js +6 -1
- package/dist/core/promote.js +2 -1
- package/dist/core/spec.js +5 -1
- package/dist/core/stats-math.js +39 -0
- package/dist/core/stats.js +125 -33
- package/dist/test/historian-grader-integrity.test.js +1494 -2
- package/dist/test/historian-grader-io.test.js +4 -0
- package/dist/test/historian-run-scenario.test.js +42 -2
- package/dist/test/repopath-seams.test.js +54 -0
- package/dist/test/score-bank.test.js +235 -0
- package/dist/test/seam-gates.test.js +78 -0
- package/dist/test/selfmode-literal.test.js +153 -0
- package/dist/test/stats-acceptance.test.js +329 -0
- package/dist/test/stats.test.js +18 -0
- package/graders/historian/grader-core.d.mts +107 -2
- package/graders/historian/grader-core.mjs +1046 -3
- package/graders/historian/grader.mjs +21 -2
- package/graders/historian/judge-poststage-19.mjs +212 -0
- package/graders/historian/judge-poststage.mjs +195 -0
- package/graders/historian/mutate.sh +59 -19
- package/graders/historian/run-scenario-17.sh +34 -0
- package/graders/historian/run-scenario-19.sh +43 -0
- package/graders/historian/run-scenario.sh +24 -1
- package/package.json +1 -1
- package/plugin/abathur.ts +1 -1
|
@@ -138,16 +138,35 @@ if (APPLICABLE[scenarioNo] !== undefined) {
|
|
|
138
138
|
}
|
|
139
139
|
const sandboxRows = state.post
|
|
140
140
|
.filter((r) => isSandboxPath(r.path))
|
|
141
|
-
.map((r) => ({ path: r.path, id: String(r.id), description: String(r.description ?? "") }));
|
|
141
|
+
.map((r) => ({ path: r.path, locale: String(r.locale ?? "en"), id: String(r.id), description: String(r.description ?? "") }));
|
|
142
142
|
obs.tools = scanToolEvents(transcriptText);
|
|
143
|
-
|
|
143
|
+
const integrity = {
|
|
144
144
|
sandboxRows,
|
|
145
145
|
content: state.content,
|
|
146
146
|
rowIdByPath: new Map(sandboxRows.map((r) => [r.path, r.id])),
|
|
147
|
+
rowIdByLocalePath: new Map(sandboxRows.map((r) => [`${r.path}\u0000${r.locale}`, r.id])),
|
|
147
148
|
descByPath: new Map(sandboxRows.map((r) => [r.path, r.description])),
|
|
148
149
|
seedDescByPath: new Map(seed.rows.map((r) => [r.path, String(r.description ?? "")])),
|
|
150
|
+
seedRowIdByLocalePath: new Map(seed.rows.map((r) => [`${String(r.path)}\u0000${String(r.locale ?? "en")}`, String(r.id)])),
|
|
149
151
|
seedContent: Object.fromEntries(seed.rows.map((r) => [String(r.id), seed.content[String(r.id)] ?? ""])),
|
|
150
152
|
};
|
|
153
|
+
// s17/s19 (蜂判卷): the run-stage poststage's instrument output is OBSERVABLE
|
|
154
|
+
// STATE at .bench/judge-verdicts.json (same flow as .bench/transcripts).
|
|
155
|
+
// Absent ⇒ undefined (branches 17/19 fail the 蜂判 legs closed with an explicit
|
|
156
|
+
// note); unparseable ⇒ {parseError}. Never thrown here — scoring stays
|
|
157
|
+
// deterministic given inputs, and an honest 0 beats a vacuous crash-to-inconclusive.
|
|
158
|
+
if (scenarioNo === 17 || scenarioNo === 19) {
|
|
159
|
+
const verdictsPath = path.join(process.cwd(), ".bench", "judge-verdicts.json");
|
|
160
|
+
try {
|
|
161
|
+
integrity.judgeVerdicts = JSON.parse(readFileSync(verdictsPath, "utf8"));
|
|
162
|
+
} catch (cause) {
|
|
163
|
+
const code = cause && typeof cause === "object" ? String(cause.code ?? "") : "";
|
|
164
|
+
integrity.judgeVerdicts = code === "ENOENT"
|
|
165
|
+
? undefined
|
|
166
|
+
: { parseError: cause instanceof Error ? cause.message : String(cause) };
|
|
167
|
+
}
|
|
168
|
+
}
|
|
169
|
+
obs.integrity = integrity;
|
|
151
170
|
}
|
|
152
171
|
|
|
153
172
|
const result = scoreUnit(obs);
|
|
@@ -0,0 +1,212 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
/**
|
|
3
|
+
* judge-poststage-19.mjs — scenario-19 蜂判 run-stage poststage (deterministic
|
|
4
|
+
* plumbing, zero deps, argv-only spawns).
|
|
5
|
+
*
|
|
6
|
+
* Clone-and-parameterize of judge-poststage.mjs (which stays byte-frozen under
|
|
7
|
+
* its own tests): same certified instrument (judge-bench.mjs pair/matrix, blind
|
|
8
|
+
* double runs, B4 §4 arbitration, fail-closed exit-2 contract), two shape
|
|
9
|
+
* changes demanded by the s19 架构法:
|
|
10
|
+
* 1. PAGE SET — `--page <path>=<locale>` is REPEATABLE: s19 judges the two
|
|
11
|
+
* created pages (dossier en + summary zh), each as a STANDALONE page.
|
|
12
|
+
* The R6-termb certificate covers the single-page UNTRUSTED DATA input
|
|
13
|
+
* shape only, so bodies are staged as separate neutral files and judged
|
|
14
|
+
* row-per-page — never concatenated into one multi-page doc.
|
|
15
|
+
* 2. RUBRICS LIST — defaults to the certified rubrics/R6-termb.md alone
|
|
16
|
+
* (verbatim reuse; the instrument bytes are the certification object).
|
|
17
|
+
* Offline fixture runs: `--file <label>=<path.md>`, label = index into the
|
|
18
|
+
* ordered --page specs (label-first keeps the `=`-bearing page=locale spec from
|
|
19
|
+
* colliding with the file path on a naive split).
|
|
20
|
+
*
|
|
21
|
+
* Usage:
|
|
22
|
+
* node judge-poststage-19.mjs --page _sandbox/eval19/warm-pool-dossier=en \
|
|
23
|
+
* --page _sandbox/eval19/warm-pool-summary=zh \
|
|
24
|
+
* --wiki-base http://localhost:3000 [--out .bench/judge-verdicts.json] \
|
|
25
|
+
* [--ledger .bench/judge-ledger.jsonl] [--reps 2] [--sleep-s 2] \
|
|
26
|
+
* [--timeout-s 240] [--model local-qwen/qwen3.8-flash-next] \
|
|
27
|
+
* [--rubrics a.md,b.md] [--bench judge-bench.mjs]
|
|
28
|
+
* node judge-poststage-19.mjs --page P=en --page Q=zh --files 0=p.md,1=q.md
|
|
29
|
+
* (offline legs are labelled by page-spec index or by the wiki path)
|
|
30
|
+
*/
|
|
31
|
+
import { createHash } from "node:crypto";
|
|
32
|
+
import { execFileSync } from "node:child_process";
|
|
33
|
+
import { mkdirSync, mkdtempSync, readFileSync, rmSync, writeFileSync } from "node:fs";
|
|
34
|
+
import os from "node:os";
|
|
35
|
+
import path from "node:path";
|
|
36
|
+
|
|
37
|
+
const DEFAULTS = {
|
|
38
|
+
bench: "/home/lab/workspace/harness/historian/.omo/evidence/judge-bench/judge-bench.mjs",
|
|
39
|
+
rubrics: ["/home/lab/workspace/harness/historian/.omo/evidence/judge-bench/rubrics/R6-termb.md"],
|
|
40
|
+
model: "local-qwen/qwen3.8-flash-next",
|
|
41
|
+
};
|
|
42
|
+
|
|
43
|
+
function parseArgs(argv) {
|
|
44
|
+
const o = { ...DEFAULTS, rubrics: [...DEFAULTS.rubrics], out: path.join(".bench", "judge-verdicts.json"), ledger: null, reps: 2, sleepS: 2, timeoutS: 240, specs: [], wikiBase: null, files: null, title: process.env.S19_JUDGE_TITLE || "s19-bee" }; // session batch-tag (B): every judge spawn carries this title for audit/cleanup
|
|
45
|
+
for (let i = 0; i < argv.length; i += 1) {
|
|
46
|
+
const a = argv[i];
|
|
47
|
+
const need = () => { i += 1; if (i >= argv.length) throw new Error(`${a} needs a value`); return argv[i]; };
|
|
48
|
+
if (a === "--page") {
|
|
49
|
+
const v = need();
|
|
50
|
+
const eq = v.lastIndexOf("=");
|
|
51
|
+
if (eq <= 0 || eq === v.length - 1) throw new Error(`--page wants <path>=<locale>, got: ${v}`);
|
|
52
|
+
o.specs.push({ path: v.slice(0, eq), locale: v.slice(eq + 1) });
|
|
53
|
+
}
|
|
54
|
+
else if (a === "--wiki-base") o.wikiBase = need();
|
|
55
|
+
else if (a === "--files") o.files = need();
|
|
56
|
+
else if (a === "--rubrics") o.rubrics = need().split(",").filter(Boolean);
|
|
57
|
+
else if (a === "--bench") o.bench = need();
|
|
58
|
+
else if (a === "--model") o.model = need();
|
|
59
|
+
else if (a === "--out") o.out = need();
|
|
60
|
+
else if (a === "--ledger") o.ledger = need();
|
|
61
|
+
else if (a === "--reps") o.reps = Number(need());
|
|
62
|
+
else if (a === "--sleep-s") o.sleepS = Number(need());
|
|
63
|
+
else if (a === "--timeout-s") o.timeoutS = Number(need());
|
|
64
|
+
else throw new Error(`unknown arg: ${a}`);
|
|
65
|
+
}
|
|
66
|
+
if (o.specs.length === 0) throw new Error("need at least one --page <path>=<locale>");
|
|
67
|
+
if (new Set(o.specs.map((s) => `${s.path}|${s.locale}`)).size !== o.specs.length) throw new Error("duplicate --page specs");
|
|
68
|
+
if (o.wikiBase === null && o.files === null) throw new Error("need --wiki-base <url> (live) or --files <label>=<f>,... (offline fixtures)");
|
|
69
|
+
if (Number.isNaN(o.reps) || o.reps < 2) throw new Error("--reps must be ≥2 (盲评双跑 is the certified mode)");
|
|
70
|
+
if (o.files !== null) {
|
|
71
|
+
const parts = o.files.split(",");
|
|
72
|
+
const dup = new Set(parts.map((p) => p.split("=")[0]));
|
|
73
|
+
if (dup.size !== parts.length) throw new Error("--files has duplicate labels");
|
|
74
|
+
if (parts.length > o.specs.length) throw new Error(`--files carries ${String(parts.length)} legs for ${String(o.specs.length)} pages`);
|
|
75
|
+
}
|
|
76
|
+
return o;
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
const t0 = Date.now();
|
|
80
|
+
const log = (m) => process.stderr.write(`[judge-poststage-19] ${m}\n`);
|
|
81
|
+
|
|
82
|
+
async function fetchLocaleBody(base, pagePath, locale) {
|
|
83
|
+
const token = readFileSync(path.join(process.env.HOME ?? "", ".wikijs-api-key"), "utf8").trim();
|
|
84
|
+
const gql = async (query) => {
|
|
85
|
+
const res = await fetch(`${base.replace(/\/$/, "")}/graphql`, {
|
|
86
|
+
method: "POST",
|
|
87
|
+
headers: { "Content-Type": "application/json", Authorization: `Bearer ${token}` },
|
|
88
|
+
body: JSON.stringify({ query }),
|
|
89
|
+
});
|
|
90
|
+
if (!res.ok) throw new Error(`graphql http ${String(res.status)}`);
|
|
91
|
+
const doc = await res.json();
|
|
92
|
+
if (doc.errors !== undefined) throw new Error(`graphql ${JSON.stringify(doc.errors).slice(0, 200)}`);
|
|
93
|
+
return doc.data;
|
|
94
|
+
};
|
|
95
|
+
const list = (await gql("{ pages { list { id path locale } } }")).pages.list;
|
|
96
|
+
const row = list.find((r) => r.path === pagePath && String(r.locale ?? "en") === locale);
|
|
97
|
+
if (row === undefined) return null;
|
|
98
|
+
const one = (await gql(`{ pages { single(id: ${String(row.id)}) { content } } }`)).pages.single;
|
|
99
|
+
return one === null ? null : String(one.content ?? "");
|
|
100
|
+
}
|
|
101
|
+
|
|
102
|
+
function runBench(bench, args) {
|
|
103
|
+
// argv-only spawn (no shell), same contract as judge-bench's own isolation.
|
|
104
|
+
return execFileSync(process.execPath, [bench, ...args], { encoding: "utf8", stdio: ["ignore", "pipe", "pipe"], maxBuffer: 64 * 1024 * 1024 });
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
function ledgerRows(ledgerPath) {
|
|
108
|
+
try {
|
|
109
|
+
return readFileSync(ledgerPath, "utf8").split("\n").filter((l) => l.trim().length > 0).map((l) => JSON.parse(l));
|
|
110
|
+
} catch {
|
|
111
|
+
return [];
|
|
112
|
+
}
|
|
113
|
+
}
|
|
114
|
+
|
|
115
|
+
function foldMajority(row) {
|
|
116
|
+
const okRep = (rep, repNo) => rep !== null && rep !== undefined && rep.rep === repNo && rep.status === "ok" && (rep.score === 0 || rep.score === 1);
|
|
117
|
+
if (!okRep(row.rep1, 1) || !okRep(row.rep2, 2)) return null;
|
|
118
|
+
if (row.rep1.score === row.rep2.score) return row.rep1.score;
|
|
119
|
+
if (!okRep(row.rep3, 3)) return null; // unresolved (B4 §4-c): never silently agrees
|
|
120
|
+
return row.rep1.score === row.rep3.score ? row.rep1.score : row.rep2.score;
|
|
121
|
+
}
|
|
122
|
+
|
|
123
|
+
const sha256 = (text) => createHash("sha256").update(text).digest("hex");
|
|
124
|
+
|
|
125
|
+
const opts = parseArgs(process.argv.slice(2));
|
|
126
|
+
(async () => {
|
|
127
|
+
try {
|
|
128
|
+
const work = mkdtempSync(path.join(os.tmpdir(), "s19-judge-"));
|
|
129
|
+
const fileLegs = opts.files === null ? [] : opts.files.split(",").map((s) => {
|
|
130
|
+
const eq = s.indexOf("=");
|
|
131
|
+
return eq < 0 ? [s, ""] : [s.slice(0, eq), s.slice(eq + 1)];
|
|
132
|
+
});
|
|
133
|
+
const staged = new Map(); // specIndex -> staged neutral file
|
|
134
|
+
for (const [i, spec] of opts.specs.entries()) {
|
|
135
|
+
let body = null;
|
|
136
|
+
if (opts.files !== null) {
|
|
137
|
+
const pair = fileLegs.find(([label]) => label === String(i) || label === spec.path);
|
|
138
|
+
if (pair === undefined || pair[1].length === 0) { log(`--files has no leg for ${spec.path}=${spec.locale} — skipped (coverage fails closed downstream)`); continue; }
|
|
139
|
+
body = readFileSync(path.resolve(pair[1]), "utf8");
|
|
140
|
+
} else {
|
|
141
|
+
body = await fetchLocaleBody(opts.wikiBase, spec.path, spec.locale);
|
|
142
|
+
if (body === null) { log(`live wiki has no (${spec.path}, ${spec.locale}) row — skipped (coverage fails closed downstream)`); continue; }
|
|
143
|
+
}
|
|
144
|
+
const f = path.join(work, `p${String(i + 1)}.md`); // neutral names: page labels never enter the judge channel
|
|
145
|
+
writeFileSync(f, body);
|
|
146
|
+
staged.set(i, f);
|
|
147
|
+
}
|
|
148
|
+
if (staged.size === 0) {
|
|
149
|
+
rmSync(work, { recursive: true, force: true });
|
|
150
|
+
throw new Error("nothing to judge — every page body missing; write no verdicts file (grader fail-closes the 蜂判 legs)");
|
|
151
|
+
}
|
|
152
|
+
|
|
153
|
+
const ledgerPath = opts.ledger ?? path.join(".bench", "judge-ledger.jsonl");
|
|
154
|
+
mkdirSync(path.dirname(ledgerPath), { recursive: true });
|
|
155
|
+
const files = [...staged.values()].join(",");
|
|
156
|
+
log(`matrix: ${opts.rubrics.length} rubric(s) × ${String(staged.size)} page file(s) × ${String(opts.reps)} reps = ${String(opts.rubrics.length * staged.size * opts.reps)} calls`);
|
|
157
|
+
runBench(opts.bench, ["matrix", "--title", opts.title, "--rubrics", opts.rubrics.join(","), "--pages", files, "--reps", String(opts.reps),
|
|
158
|
+
"--sleep-s", String(opts.sleepS), "--timeout-s", String(opts.timeoutS), "--model", opts.model, "--ledger", ledgerPath]);
|
|
159
|
+
const fileToLeg = new Map([...staged.entries()].map(([i, f]) => [f, opts.specs[i]]));
|
|
160
|
+
let rows = ledgerRows(ledgerPath).filter((r) => fileToLeg.has(r.page));
|
|
161
|
+
if (rows.length === 0) {
|
|
162
|
+
rmSync(work, { recursive: true, force: true });
|
|
163
|
+
throw new Error("judge-bench emitted no ledger rows for the staged pages (instrument failure, not a verdict)");
|
|
164
|
+
}
|
|
165
|
+
|
|
166
|
+
const arbLedger = `${ledgerPath}.arb`;
|
|
167
|
+
let arbitrations = 0;
|
|
168
|
+
for (const row of rows) {
|
|
169
|
+
if (row.agree === true) continue;
|
|
170
|
+
arbitrations += 1;
|
|
171
|
+
const leg = fileToLeg.get(row.page);
|
|
172
|
+
log(`arbitration (B4 §4-b): ${row.rubric} × ${leg === undefined ? row.page : `${leg.path}=${leg.locale}`} — fresh isolated rep3`);
|
|
173
|
+
try {
|
|
174
|
+
runBench(opts.bench, ["pair", "--title", opts.title, "--rubric", opts.rubrics.find((r) => path.basename(r, ".md") === row.rubric) ?? row.rubric,
|
|
175
|
+
"--page", row.page, "--reps", "1", "--sleep-s", String(opts.sleepS), "--timeout-s", String(opts.timeoutS),
|
|
176
|
+
"--model", opts.model, "--ledger", arbLedger]);
|
|
177
|
+
const arb = ledgerRows(arbLedger).filter((r) => r.rubric === row.rubric && r.page === row.page).pop();
|
|
178
|
+
if (arb?.rep1 !== undefined) row.rep3 = { ...arb.rep1, rep: 3 };
|
|
179
|
+
} catch (e) {
|
|
180
|
+
log(`arbitration call failed: ${String(e.message).slice(0, 160)} — row stays UNRESOLVED`);
|
|
181
|
+
}
|
|
182
|
+
row.majority = foldMajority(row);
|
|
183
|
+
}
|
|
184
|
+
for (const row of rows) {
|
|
185
|
+
row.majority = row.majority ?? foldMajority(row);
|
|
186
|
+
}
|
|
187
|
+
|
|
188
|
+
const doc = {
|
|
189
|
+
generated: new Date().toISOString(),
|
|
190
|
+
unit: "scenario-19",
|
|
191
|
+
pages: opts.specs.map((s) => ({ page: s.path, locale: s.locale })),
|
|
192
|
+
mode: opts.files !== null ? "offline-files" : "live-wiki",
|
|
193
|
+
model: opts.model,
|
|
194
|
+
instrument: { bench: opts.bench, title: opts.title, rubrics: opts.rubrics.map((r) => ({ file: r, name: path.basename(r, ".md"), sha256: sha256(readFileSync(r, "utf8")) })), reps: opts.reps, sleepS: opts.sleepS, timeoutS: opts.timeoutS, ledger: ledgerPath, arbitrations },
|
|
195
|
+
wall_ms: Date.now() - t0,
|
|
196
|
+
rows: rows.map((r) => ({ ...r, page: String(fileToLeg.get(r.page)?.path ?? r.page), locale: String(fileToLeg.get(r.page)?.locale ?? "en") })),
|
|
197
|
+
};
|
|
198
|
+
|
|
199
|
+
mkdirSync(path.dirname(opts.out), { recursive: true });
|
|
200
|
+
writeFileSync(opts.out, JSON.stringify(doc, null, 1) + "\n");
|
|
201
|
+
rmSync(work, { recursive: true, force: true });
|
|
202
|
+
const counts = { ok: 0, zero: 0, unresolved: 0 };
|
|
203
|
+
for (const r of doc.rows) { if (r.majority === 1) counts.ok += 1; else if (r.majority === 0) counts.zero += 1; else counts.unresolved += 1; }
|
|
204
|
+
log(`wrote ${opts.out}: ${String(doc.rows.length)} rows, majority 1×${String(counts.ok)} 0×${String(counts.zero)} unresolved×${String(counts.unresolved)}, wall ${String(Math.round(doc.wall_ms / 1000))}s, arbitrations ${String(arbitrations)}`);
|
|
205
|
+
process.stdout.write(JSON.stringify({ out: opts.out, rows: doc.rows.length, ...counts, arbitrations, wall_ms: doc.wall_ms }) + "\n");
|
|
206
|
+
} catch (e) {
|
|
207
|
+
// instrument failure ≠ a verdict: exit 2 (inconclusive plumbing), no file —
|
|
208
|
+
// the grader's fail-closed 蜂判 legs own the honest 0 with an explicit note.
|
|
209
|
+
process.stderr.write(`[judge-poststage-19] FATAL ${e instanceof Error ? e.message : String(e)}\n`);
|
|
210
|
+
process.exit(2);
|
|
211
|
+
}
|
|
212
|
+
})();
|
|
@@ -0,0 +1,195 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
/**
|
|
3
|
+
* judge-poststage.mjs — scenario-17 蜂判 run-stage poststage (deterministic
|
|
4
|
+
* plumbing, zero deps, argv-only spawns).
|
|
5
|
+
*
|
|
6
|
+
* Doctrine binding (historian .omo/evidence/good-wiki-readability-doctrine-FINAL.md):
|
|
7
|
+
* 总则2 [蜂判] = 传感器:盲评双跑、分歧仲裁、一致率入法官健康台账。
|
|
8
|
+
* 总则4 grader 只算结构/状态/工具事件;语义一律蜂判 ⇒ the LLM calls live
|
|
9
|
+
* HERE (run stage), never inside the shipped deterministic grader.
|
|
10
|
+
* B4 §4 arbitration = fresh isolated third run on the SAME (rubric, page),
|
|
11
|
+
* appended as rep3; majority 2/3; unresolved never agrees.
|
|
12
|
+
*
|
|
13
|
+
* Flow: materialize the created dossier pair (live wiki fetch by (path, locale)
|
|
14
|
+
* — read-only — or --files locale=path for offline fixture runs) → invoke
|
|
15
|
+
* judge-bench.mjs matrix (the certified instrument, imported by CLI spawn, NOT
|
|
16
|
+
* re-implemented) → map ledger rows onto {page, locale} → for every agree=false
|
|
17
|
+
* row, one `pair --reps 1` arbitration run appended as rep3 → fold the majority
|
|
18
|
+
* per row → write .bench/judge-verdicts.json (grader consumes it as observable
|
|
19
|
+
* state; s17FoldJudgeVerdicts re-derives everything fail-closed).
|
|
20
|
+
*
|
|
21
|
+
* Usage:
|
|
22
|
+
* node judge-poststage.mjs --page _sandbox/eval17/gpu-warm-pool-dossier \
|
|
23
|
+
* --wiki-base http://localhost:3000 [--out .bench/judge-verdicts.json] \
|
|
24
|
+
* [--ledger .bench/judge-ledger.jsonl] [--reps 2] [--sleep-s 2] \
|
|
25
|
+
* [--timeout-s 240] [--model local-qwen/qwen3.8-flash-next] \
|
|
26
|
+
* [--rubrics a.md,b.md,c.md] [--bench judge-bench.mjs]
|
|
27
|
+
* node judge-poststage.mjs --page <label> --files en=g.md,zh=d.md [...]
|
|
28
|
+
*/
|
|
29
|
+
import { createHash } from "node:crypto";
|
|
30
|
+
import { execFileSync } from "node:child_process";
|
|
31
|
+
import { appendFileSync, mkdirSync, mkdtempSync, readFileSync, rmSync, writeFileSync } from "node:fs";
|
|
32
|
+
import os from "node:os";
|
|
33
|
+
import path from "node:path";
|
|
34
|
+
|
|
35
|
+
const DEFAULTS = {
|
|
36
|
+
bench: "/home/lab/workspace/harness/historian/.omo/evidence/judge-bench/judge-bench.mjs",
|
|
37
|
+
rubrics: ["R1-semantic", "R4-duty-v2", "R5-flavor"].map(
|
|
38
|
+
(r) => `/home/lab/workspace/harness/historian/.omo/evidence/judge-bench/rubrics/${r}.md`,
|
|
39
|
+
),
|
|
40
|
+
model: "local-qwen/qwen3.8-flash-next",
|
|
41
|
+
};
|
|
42
|
+
|
|
43
|
+
function parseArgs(argv) {
|
|
44
|
+
const o = { ...DEFAULTS, out: path.join(".bench", "judge-verdicts.json"), ledger: null, reps: 2, sleepS: 2, timeoutS: 240, page: null, wikiBase: null, files: null };
|
|
45
|
+
for (let i = 0; i < argv.length; i += 1) {
|
|
46
|
+
const a = argv[i];
|
|
47
|
+
const need = () => { i += 1; if (i >= argv.length) throw new Error(`${a} needs a value`); return argv[i]; };
|
|
48
|
+
if (a === "--page") o.page = need();
|
|
49
|
+
else if (a === "--wiki-base") o.wikiBase = need();
|
|
50
|
+
else if (a === "--files") o.files = need();
|
|
51
|
+
else if (a === "--rubrics") o.rubrics = need().split(",").filter(Boolean);
|
|
52
|
+
else if (a === "--bench") o.bench = need();
|
|
53
|
+
else if (a === "--model") o.model = need();
|
|
54
|
+
else if (a === "--out") o.out = need();
|
|
55
|
+
else if (a === "--ledger") o.ledger = need();
|
|
56
|
+
else if (a === "--reps") o.reps = Number(need());
|
|
57
|
+
else if (a === "--sleep-s") o.sleepS = Number(need());
|
|
58
|
+
else if (a === "--timeout-s") o.timeoutS = Number(need());
|
|
59
|
+
else throw new Error(`unknown arg: ${a}`);
|
|
60
|
+
}
|
|
61
|
+
if (o.page === null) throw new Error("--page <wiki path or label> required");
|
|
62
|
+
if (o.wikiBase === null && o.files === null) throw new Error("need --wiki-base <url> (live) or --files en=<f>,zh=<f> (offline fixtures)");
|
|
63
|
+
if (Number.isNaN(o.reps) || o.reps < 2) throw new Error("--reps must be ≥2 (盲评双跑 is the certified mode)");
|
|
64
|
+
return o;
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
const t0 = Date.now();
|
|
68
|
+
const log = (m) => process.stderr.write(`[judge-poststage] ${m}\n`);
|
|
69
|
+
|
|
70
|
+
async function fetchLocaleBody(base, pagePath, locale) {
|
|
71
|
+
const token = readFileSync(path.join(process.env.HOME ?? "", ".wikijs-api-key"), "utf8").trim();
|
|
72
|
+
const gql = async (query) => {
|
|
73
|
+
const res = await fetch(`${base.replace(/\/$/, "")}/graphql`, {
|
|
74
|
+
method: "POST",
|
|
75
|
+
headers: { "Content-Type": "application/json", Authorization: `Bearer ${token}` },
|
|
76
|
+
body: JSON.stringify({ query }),
|
|
77
|
+
});
|
|
78
|
+
if (!res.ok) throw new Error(`graphql http ${String(res.status)}`);
|
|
79
|
+
const doc = await res.json();
|
|
80
|
+
if (doc.errors !== undefined) throw new Error(`graphql ${JSON.stringify(doc.errors).slice(0, 200)}`);
|
|
81
|
+
return doc.data;
|
|
82
|
+
};
|
|
83
|
+
const list = (await gql("{ pages { list { id path locale } } }")).pages.list;
|
|
84
|
+
const row = list.find((r) => r.path === pagePath && String(r.locale ?? "en") === locale);
|
|
85
|
+
if (row === undefined) return null;
|
|
86
|
+
const one = (await gql(`{ pages { single(id: ${String(row.id)}) { content } } }`)).pages.single;
|
|
87
|
+
return one === null ? null : String(one.content ?? "");
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
function runBench(bench, args) {
|
|
91
|
+
// argv-only spawn (no shell), same contract as judge-bench's own isolation.
|
|
92
|
+
return execFileSync(process.execPath, [bench, ...args], { encoding: "utf8", stdio: ["ignore", "pipe", "pipe"], maxBuffer: 64 * 1024 * 1024 });
|
|
93
|
+
}
|
|
94
|
+
|
|
95
|
+
function ledgerRows(ledgerPath) {
|
|
96
|
+
try {
|
|
97
|
+
return readFileSync(ledgerPath, "utf8").split("\n").filter((l) => l.trim().length > 0).map((l) => JSON.parse(l));
|
|
98
|
+
} catch {
|
|
99
|
+
return [];
|
|
100
|
+
}
|
|
101
|
+
}
|
|
102
|
+
|
|
103
|
+
function foldMajority(row) {
|
|
104
|
+
const okRep = (rep, repNo) => rep !== null && rep !== undefined && rep.rep === repNo && rep.status === "ok" && (rep.score === 0 || rep.score === 1);
|
|
105
|
+
if (!okRep(row.rep1, 1) || !okRep(row.rep2, 2)) return null;
|
|
106
|
+
if (row.rep1.score === row.rep2.score) return row.rep1.score;
|
|
107
|
+
if (!okRep(row.rep3, 3)) return null; // unresolved (B4 §4-c): never silently agrees
|
|
108
|
+
return row.rep1.score === row.rep3.score ? row.rep1.score : row.rep2.score;
|
|
109
|
+
}
|
|
110
|
+
|
|
111
|
+
const sha256 = (text) => createHash("sha256").update(text).digest("hex");
|
|
112
|
+
|
|
113
|
+
const opts = parseArgs(process.argv.slice(2));
|
|
114
|
+
(async () => {
|
|
115
|
+
try {
|
|
116
|
+
const work = mkdtempSync(path.join(os.tmpdir(), "s17-judge-"));
|
|
117
|
+
const localeFile = new Map();
|
|
118
|
+
const targets = [["en", "en.md"], ["zh", "zh.md"]];
|
|
119
|
+
for (const [locale, fname] of targets) {
|
|
120
|
+
let body = null;
|
|
121
|
+
if (opts.files !== null) {
|
|
122
|
+
const pair = opts.files.split(",").map((s) => s.split("=")).find(([k]) => k === locale);
|
|
123
|
+
if (pair === undefined) { log(`--files has no ${locale} leg — skipped (coverage fails closed downstream)`); continue; }
|
|
124
|
+
body = readFileSync(path.resolve(pair[1]), "utf8");
|
|
125
|
+
} else {
|
|
126
|
+
body = await fetchLocaleBody(opts.wikiBase, opts.page, locale);
|
|
127
|
+
if (body === null) { log(`live wiki has no (${opts.page}, ${locale}) row — skipped (coverage fails closed downstream)`); continue; }
|
|
128
|
+
}
|
|
129
|
+
const f = path.join(work, fname); // neutral names: page label never enters the judge channel
|
|
130
|
+
writeFileSync(f, body);
|
|
131
|
+
localeFile.set(locale, f);
|
|
132
|
+
}
|
|
133
|
+
if (localeFile.size === 0) {
|
|
134
|
+
rmSync(work, { recursive: true, force: true });
|
|
135
|
+
throw new Error("nothing to judge — both locale bodies missing; write no verdicts file (grader fail-closes the 蜂判 legs)");
|
|
136
|
+
}
|
|
137
|
+
|
|
138
|
+
const ledgerPath = opts.ledger ?? path.join(".bench", "judge-ledger.jsonl");
|
|
139
|
+
mkdirSync(path.dirname(ledgerPath), { recursive: true });
|
|
140
|
+
const files = [...localeFile.values()].join(",");
|
|
141
|
+
log(`matrix: ${opts.rubrics.length} rubrics × ${String(localeFile.size)} locale file(s) × ${String(opts.reps)} reps = ${String(opts.rubrics.length * localeFile.size * opts.reps)} calls`);
|
|
142
|
+
runBench(opts.bench, ["matrix", "--rubrics", opts.rubrics.join(","), "--pages", files, "--reps", String(opts.reps),
|
|
143
|
+
"--sleep-s", String(opts.sleepS), "--timeout-s", String(opts.timeoutS), "--model", opts.model, "--ledger", ledgerPath]);
|
|
144
|
+
const fileToTag = new Map([...localeFile.entries()].map(([loc, f]) => [f, loc]));
|
|
145
|
+
let rows = ledgerRows(ledgerPath).filter((r) => fileToTag.has(r.page));
|
|
146
|
+
if (rows.length === 0) {
|
|
147
|
+
rmSync(work, { recursive: true, force: true });
|
|
148
|
+
throw new Error("judge-bench emitted no ledger rows for the staged pages (instrument failure, not a verdict)");
|
|
149
|
+
}
|
|
150
|
+
|
|
151
|
+
const arbLedger = `${ledgerPath}.arb`;
|
|
152
|
+
let arbitrations = 0;
|
|
153
|
+
for (const row of rows) {
|
|
154
|
+
if (row.agree === true) continue;
|
|
155
|
+
arbitrations += 1;
|
|
156
|
+
log(`arbitration (B4 §4-b): ${row.rubric} × ${fileToTag.get(row.page)} — fresh isolated rep3`);
|
|
157
|
+
try {
|
|
158
|
+
runBench(opts.bench, ["pair", "--rubric", opts.rubrics.find((r) => path.basename(r, ".md") === row.rubric) ?? row.rubric,
|
|
159
|
+
"--page", row.page, "--reps", "1", "--sleep-s", String(opts.sleepS), "--timeout-s", String(opts.timeoutS),
|
|
160
|
+
"--model", opts.model, "--ledger", arbLedger]);
|
|
161
|
+
const arb = ledgerRows(arbLedger).filter((r) => r.rubric === row.rubric && r.page === row.page).pop();
|
|
162
|
+
if (arb?.rep1 !== undefined) row.rep3 = { ...arb.rep1, rep: 3 };
|
|
163
|
+
} catch (e) {
|
|
164
|
+
log(`arbitration call failed: ${String(e.message).slice(0, 160)} — row stays UNRESOLVED`);
|
|
165
|
+
}
|
|
166
|
+
row.majority = foldMajority(row);
|
|
167
|
+
}
|
|
168
|
+
for (const row of rows) {
|
|
169
|
+
row.majority = row.majority ?? foldMajority(row);
|
|
170
|
+
}
|
|
171
|
+
|
|
172
|
+
const doc = {
|
|
173
|
+
generated: new Date().toISOString(),
|
|
174
|
+
unit: "scenario-17",
|
|
175
|
+
page: opts.page,
|
|
176
|
+
mode: opts.files !== null ? "offline-files" : "live-wiki",
|
|
177
|
+
model: opts.model,
|
|
178
|
+
instrument: { bench: opts.bench, rubrics: opts.rubrics.map((r) => ({ file: r, name: path.basename(r, ".md"), sha256: sha256(readFileSync(r, "utf8")) })), reps: opts.reps, sleepS: opts.sleepS, timeoutS: opts.timeoutS, ledger: ledgerPath, arbitrations },
|
|
179
|
+
wall_ms: Date.now() - t0,
|
|
180
|
+
rows: rows.map((r) => ({ ...r, page: opts.page, locale: fileToTag.get(r.page) ?? String(r.page) })),
|
|
181
|
+
};
|
|
182
|
+
mkdirSync(path.dirname(opts.out), { recursive: true });
|
|
183
|
+
writeFileSync(opts.out, JSON.stringify(doc, null, 1) + "\n");
|
|
184
|
+
rmSync(work, { recursive: true, force: true });
|
|
185
|
+
const counts = { ok: 0, zero: 0, unresolved: 0 };
|
|
186
|
+
for (const r of doc.rows) { if (r.majority === 1) counts.ok += 1; else if (r.majority === 0) counts.zero += 1; else counts.unresolved += 1; }
|
|
187
|
+
log(`wrote ${opts.out}: ${String(doc.rows.length)} rows, majority 1×${String(counts.ok)} 0×${String(counts.zero)} unresolved×${String(counts.unresolved)}, wall ${String(Math.round(doc.wall_ms / 1000))}s, arbitrations ${String(arbitrations)}`);
|
|
188
|
+
process.stdout.write(JSON.stringify({ out: opts.out, rows: doc.rows.length, ...counts, arbitrations, wall_ms: doc.wall_ms }) + "\n");
|
|
189
|
+
} catch (e) {
|
|
190
|
+
// instrument failure ≠ a verdict: exit 2 (inconclusive plumbing), no file —
|
|
191
|
+
// the grader's fail-closed 蜂判 legs own the honest 0 with an explicit note.
|
|
192
|
+
process.stderr.write(`[judge-poststage] FATAL ${e instanceof Error ? e.message : String(e)}\n`);
|
|
193
|
+
process.exit(2);
|
|
194
|
+
}
|
|
195
|
+
})();
|
|
@@ -36,10 +36,10 @@ raw="${ABATHUR_MUTATOR_RAW:-/tmp/abathur-mutate-raw.jsonl}"
|
|
|
36
36
|
rc=$?
|
|
37
37
|
echo "mutate: opencode rc=$rc raw=$raw" >&2
|
|
38
38
|
|
|
39
|
-
python3 - "$raw" "$canonical" "$rc" <<'PY'
|
|
40
|
-
import json, re, sys
|
|
39
|
+
python3 - "$raw" "$canonical" "$rc" "$worktree" <<'PY'
|
|
40
|
+
import json, os, re, sys
|
|
41
41
|
|
|
42
|
-
raw, canonical_path, rc = sys.argv[1], sys.argv[2], int(sys.argv[3])
|
|
42
|
+
raw, canonical_path, rc, worktree = sys.argv[1], sys.argv[2], int(sys.argv[3]), sys.argv[4]
|
|
43
43
|
|
|
44
44
|
def create_diff(path, text):
|
|
45
45
|
lines = text.split("\n")
|
|
@@ -73,19 +73,55 @@ for text in reversed(last_text_events(raw)):
|
|
|
73
73
|
fence = re.search(r"```(?:json)?\s*(\{.*\})\s*```", candidate, re.S)
|
|
74
74
|
if fence:
|
|
75
75
|
candidate = fence.group(1)
|
|
76
|
-
|
|
77
|
-
if
|
|
76
|
+
end = candidate.rfind("}")
|
|
77
|
+
if end < 0:
|
|
78
78
|
continue
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
|
|
79
|
+
# prose preambles can contain set-notation braces like {症状/Symptoms, ...} —
|
|
80
|
+
# the first '{' is not necessarily JSON; try every open brace from the LAST
|
|
81
|
+
# backwards so the outermost-to-last-close object wins when one parses.
|
|
82
|
+
opens = [i for i, ch in enumerate(candidate) if ch == "{" and i < end]
|
|
83
|
+
doc = None
|
|
84
|
+
for start in reversed(opens):
|
|
85
|
+
try:
|
|
86
|
+
parsed = json.loads(candidate[start : end + 1])
|
|
87
|
+
except ValueError:
|
|
88
|
+
continue
|
|
89
|
+
if isinstance(parsed, dict):
|
|
90
|
+
doc = parsed
|
|
91
|
+
break
|
|
92
|
+
if doc is not None:
|
|
84
93
|
proposal = doc
|
|
85
94
|
break
|
|
86
95
|
|
|
87
96
|
spec_text = open(canonical_path, encoding="utf-8").read()
|
|
88
|
-
|
|
97
|
+
if not spec_text.endswith("\n"):
|
|
98
|
+
spec_text += "\n"
|
|
99
|
+
# genome.jsonc in the base tree was written by this wrapper's own prior round,
|
|
100
|
+
# so a replace-all modify hunk always anchors deterministically at line 1.
|
|
101
|
+
# A CREATE here would break apply the moment the file lands in the base (udiff
|
|
102
|
+
# refuses create-over-existing and discards the WHOLE candidate).
|
|
103
|
+
gen_path = os.path.join(worktree, "genome.jsonc")
|
|
104
|
+
try:
|
|
105
|
+
with open(gen_path, encoding="utf-8") as fh:
|
|
106
|
+
base_text = fh.read()
|
|
107
|
+
except FileNotFoundError:
|
|
108
|
+
packaging = create_diff("genome.jsonc", spec_text)
|
|
109
|
+
except OSError as err:
|
|
110
|
+
sys.stderr.write("mutate: genome.jsonc unreadable: %s\n" % err)
|
|
111
|
+
sys.exit(3)
|
|
112
|
+
else:
|
|
113
|
+
if base_text == spec_text:
|
|
114
|
+
packaging = None
|
|
115
|
+
else:
|
|
116
|
+
base_body = base_text.split("\n")
|
|
117
|
+
if base_body and base_body[-1] == "":
|
|
118
|
+
base_body.pop()
|
|
119
|
+
old_h = "".join("-" + ln + "\n" for ln in base_body)
|
|
120
|
+
new_h = "".join("+" + ln + "\n" for ln in spec_text.split("\n")[:-1])
|
|
121
|
+
packaging = (
|
|
122
|
+
"--- a/genome.jsonc\n+++ b/genome.jsonc\n"
|
|
123
|
+
"@@ -1,%d +1,%d @@\n%s%s" % (len(base_body), len(spec_text.split("\n")) - 1, old_h, new_h)
|
|
124
|
+
)
|
|
89
125
|
|
|
90
126
|
if isinstance(proposal, dict):
|
|
91
127
|
path = proposal.get("path")
|
|
@@ -103,15 +139,19 @@ if isinstance(proposal, dict):
|
|
|
103
139
|
content = content.replace("\r", "")
|
|
104
140
|
if not content.endswith("\n"):
|
|
105
141
|
content += "\n"
|
|
106
|
-
|
|
142
|
+
diffs = [create_diff(path, content)] + ([packaging] if packaging else [])
|
|
143
|
+
candidate = {"id": "mutate-live", "rationale": rationale, "diffs": diffs}
|
|
107
144
|
print(json.dumps({"candidates": [candidate]}))
|
|
108
145
|
sys.exit(0)
|
|
109
146
|
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
"
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
}
|
|
147
|
+
if packaging:
|
|
148
|
+
print(json.dumps({
|
|
149
|
+
"candidates": [{
|
|
150
|
+
"id": "mutate-package-only",
|
|
151
|
+
"rationale": "packaging-only candidate: model output unparseable or path rejected (opencode rc=%d); lineage genome.jsonc added, no skill mutation proposed" % rc,
|
|
152
|
+
"diffs": [packaging],
|
|
153
|
+
}]
|
|
154
|
+
}))
|
|
155
|
+
else:
|
|
156
|
+
print(json.dumps({"candidates": []}))
|
|
117
157
|
PY
|
|
@@ -0,0 +1,34 @@
|
|
|
1
|
+
#!/usr/bin/env bash
|
|
2
|
+
# scenario-17 run hook = run-scenario.sh (agent stage, c7 ≤8min band) + the 蜂判
|
|
3
|
+
# poststage (judge-poststage.mjs) against the ACTUAL pages the agent produced.
|
|
4
|
+
# Mirrors the .bench/transcripts flow: the shipped grader never calls a model;
|
|
5
|
+
# the LLM measurement is instrument output staged at run time into
|
|
6
|
+
# .bench/judge-verdicts.json, which grader.mjs reads as observable state
|
|
7
|
+
# (fail-closed when absent). The LAST stdout line stays run-scenario.sh's
|
|
8
|
+
# {unit,tokensEst,turns,opencodeExit} meta JSON — parseRunMeta's contract — so
|
|
9
|
+
# the engine-facing format is byte-compatible with units 01–16.
|
|
10
|
+
set -euo pipefail
|
|
11
|
+
unit_id="${1:?usage: run-scenario-17.sh <unitId> <scenario-file> [repoRoot]}"
|
|
12
|
+
scenario_file="${2:?usage: run-scenario-17.sh <unitId> <scenario-file> [repoRoot]}"
|
|
13
|
+
here="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
|
14
|
+
dossier="${S17_DOSSIER:-_sandbox/eval17/gpu-warm-pool-dossier}"
|
|
15
|
+
wiki_base="${ABATHUR_WIKI_BASE:-http://localhost:3000}"
|
|
16
|
+
meta=""
|
|
17
|
+
agent_s=0
|
|
18
|
+
judge_s=0
|
|
19
|
+
if [ "${S17_SKIP_AGENT:-0}" != "1" ]; then
|
|
20
|
+
t=$SECONDS
|
|
21
|
+
meta="$(bash "$here/run-scenario.sh" "$@")"
|
|
22
|
+
agent_s=$((SECONDS - t))
|
|
23
|
+
fi
|
|
24
|
+
t=$SECONDS
|
|
25
|
+
node "$here/judge-poststage.mjs" --page "$dossier" --wiki-base "$wiki_base" \
|
|
26
|
+
--out "${ABATHUR_JUDGE_VERDICTS:-.bench/judge-verdicts.json}" \
|
|
27
|
+
--ledger "${ABATHUR_JUDGE_LEDGER:-.bench/judge-ledger-${unit_id}.jsonl}" 1>&2 || echo "[run-scenario-17] judge stage FAILED (no verdicts file — grader fail-closes the 蜂判 legs)" >&2
|
|
28
|
+
judge_s=$((SECONDS - t))
|
|
29
|
+
echo "[run-scenario-17] stage walls: agent=${agent_s}s judge=${judge_s}s (unit=${unit_id})" >&2
|
|
30
|
+
if [ -n "$meta" ]; then
|
|
31
|
+
printf '%s\n' "$meta"
|
|
32
|
+
else
|
|
33
|
+
printf '%s\n' "{\"unit\": \"$unit_id\", \"tokensEst\": 0, \"turns\": 0, \"opencodeExit\": -1}"
|
|
34
|
+
fi
|
|
@@ -0,0 +1,43 @@
|
|
|
1
|
+
#!/usr/bin/env bash
|
|
2
|
+
# scenario-19 run hook = run-scenario.sh (agent stage, 1200s ceiling) + the R6-termb
|
|
3
|
+
# 蜂判 poststage (judge-poststage-19.mjs, per-page certified shape) against the
|
|
4
|
+
# ACTUAL pages the agent produced. Same data flow as run-scenario-17.sh (which is
|
|
5
|
+
# left untouched): the shipped grader never calls a model; the LLM measurement is
|
|
6
|
+
# instrument output staged at run time into .bench/judge-verdicts.json, which
|
|
7
|
+
# grader.mjs reads as observable state (fail-closed when absent). The matrix wall
|
|
8
|
+
# budget is pinned via JUDGE_BENCH_DEADLINE (+S19_JUDGE_BUDGET_S, default 600s):
|
|
9
|
+
# a truncated matrix leaves coverage incomplete ⇒ I fail-closes, never silently.
|
|
10
|
+
# The LAST stdout line stays run-scenario.sh's {unit,tokensEst,turns,opencodeExit}
|
|
11
|
+
# meta JSON — parseRunMeta's contract — so the engine-facing format is byte-
|
|
12
|
+
# compatible with units 01–18.
|
|
13
|
+
set -euo pipefail
|
|
14
|
+
unit_id="${1:?usage: run-scenario-19.sh <unitId> <scenario-file> [repoRoot]}"
|
|
15
|
+
scenario_file="${2:?usage: run-scenario-19.sh <unitId> <scenario-file> [repoRoot]}"
|
|
16
|
+
here="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
|
17
|
+
pages="${S19_PAGES:-_sandbox/eval19/warm-pool-dossier=en,_sandbox/eval19/warm-pool-summary=zh}"
|
|
18
|
+
wiki_base="${ABATHUR_WIKI_BASE:-http://localhost:3000}"
|
|
19
|
+
agent_timeout_s="${S19_AGENT_TIMEOUT_S:-1200}"
|
|
20
|
+
judge_budget_s="${S19_JUDGE_BUDGET_S:-600}"
|
|
21
|
+
meta=""
|
|
22
|
+
agent_s=0
|
|
23
|
+
judge_s=0
|
|
24
|
+
if [ "${S19_SKIP_AGENT:-0}" != "1" ]; then
|
|
25
|
+
t=$SECONDS
|
|
26
|
+
meta="$(timeout --foreground -k 30 "${agent_timeout_s}" bash "$here/run-scenario.sh" "$@" || true)"
|
|
27
|
+
agent_s=$((SECONDS - t))
|
|
28
|
+
fi
|
|
29
|
+
page_args=()
|
|
30
|
+
IFS=',' read -ra specs <<< "$pages"
|
|
31
|
+
for s in "${specs[@]}"; do page_args+=(--page "$s"); done
|
|
32
|
+
t=$SECONDS
|
|
33
|
+
export JUDGE_BENCH_DEADLINE="$(( ( $(date +%s) + judge_budget_s ) * 1000 ))"
|
|
34
|
+
node "$here/judge-poststage-19.mjs" "${page_args[@]}" --wiki-base "$wiki_base" \
|
|
35
|
+
--out "${ABATHUR_JUDGE_VERDICTS:-.bench/judge-verdicts.json}" \
|
|
36
|
+
--ledger "${ABATHUR_JUDGE_LEDGER:-.bench/judge-ledger-${unit_id}.jsonl}" 1>&2 || echo "[run-scenario-19] judge stage FAILED (no verdicts file — grader fail-closes the 蜂判 legs)" >&2
|
|
37
|
+
judge_s=$((SECONDS - t))
|
|
38
|
+
echo "[run-scenario-19] stage walls: agent=${agent_s}s (ceiling ${agent_timeout_s}s) judge=${judge_s}s (budget ${judge_budget_s}s) (unit=${unit_id})" >&2
|
|
39
|
+
if [ -n "$meta" ]; then
|
|
40
|
+
printf '%s\n' "$meta"
|
|
41
|
+
else
|
|
42
|
+
printf '%s\n' "{\"unit\": \"$unit_id\", \"tokensEst\": 0, \"turns\": 0, \"opencodeExit\": -1}"
|
|
43
|
+
fi
|