@tachikomagundam/abathur 0.2.3 → 0.2.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -3,7 +3,7 @@
3
3
  // (ABATHUR_GRADER_STATE replaces every wiki/URL call; no live network).
4
4
  import assert from "node:assert/strict";
5
5
  import { execFile } from "node:child_process";
6
- import { mkdirSync, mkdtempSync, writeFileSync } from "node:fs";
6
+ import { copyFileSync, existsSync, mkdirSync, mkdtempSync, readFileSync, rmSync, writeFileSync } from "node:fs";
7
7
  import { tmpdir } from "node:os";
8
8
  import path from "node:path";
9
9
  import { fileURLToPath } from "node:url";
@@ -146,3 +146,279 @@ test("grader CLI: garbage transcript ⇒ nonzero exit, no score line (adapter
146
146
  assert.notEqual(r.code, 0);
147
147
  assert.equal(r.stdout.trim(), "");
148
148
  });
149
+ // ------------------------------------------- S4 seam: active-tree scenario resolution
150
+ //
151
+ // post-seam graderCommand renders `{repoRoot}/{unit.path}` (adapter unitVars), so
152
+ // argv[1] arrives as an ABSOLUTE path into the bench's active tree — incumbent
153
+ // repoPath or candidate worktree, same notion as the run side. Resolution mirrors
154
+ // the ABATHUR_GRADER_STATE offline pattern: the file is read only when it exists
155
+ // (offline fixtures pass relative/nonexistent paths, which must keep working), and
156
+ // a readable scenario must agree with the unit id on the scenario number — the
157
+ // pre-seam failure mode was commands baked against the WRONG tree.
158
+ function stateFor(dir) {
159
+ const stateFile = path.join(dir, "state.json");
160
+ writeFileSync(stateFile, JSON.stringify({
161
+ post: POST_OK,
162
+ content: { 400: "[qwen27b](/_sandbox/llm-inference/qwen27b-threading)", 600: GOOD_PAGE },
163
+ urlStatus: {},
164
+ }));
165
+ return stateFile;
166
+ }
167
+ test("grader CLI resolves an existing {repoRoot}-rendered scenario file ⇒ metrics.scenarioFile", async () => {
168
+ const dir = sandboxWith(transcript(GOOD_FINAL, 5678), PRE);
169
+ const repo = mkdtempSync(path.join(tmpdir(), "t14-seamrepo-"));
170
+ mkdirSync(path.join(repo, "scenarios"), { recursive: true });
171
+ const scenarioAbs = path.join(repo, "scenarios", "01-new-finding.md");
172
+ writeFileSync(scenarioAbs, "# Scenario 01\n\n## Brief\ndo it\n", "utf8");
173
+ const r = await runGrader(dir, ["scenario-01", scenarioAbs, "http://localhost:3000"], {
174
+ ABATHUR_GRADER_STATE: stateFor(dir),
175
+ });
176
+ assert.equal(r.code, 0, r.stderr);
177
+ const parsed = JSON.parse(r.stdout.trim().split("\n").at(-1) ?? "{}");
178
+ assert.equal(parsed.metrics.scenarioFile, scenarioAbs, "resolved active-tree scenario path surfaces in metrics");
179
+ });
180
+ test("grader CLI: scenario number disagreeing with the unit id ⇒ nonzero (never a score)", async () => {
181
+ const dir = sandboxWith(transcript(GOOD_FINAL, 5678), PRE);
182
+ const repo = mkdtempSync(path.join(tmpdir(), "t14-seambad-"));
183
+ mkdirSync(path.join(repo, "scenarios"), { recursive: true });
184
+ const wrongAbs = path.join(repo, "scenarios", "99-wrong-unit.md");
185
+ writeFileSync(wrongAbs, "# Scenario 99\n\n## Brief\nnope\n", "utf8");
186
+ const r = await runGrader(dir, ["scenario-01", wrongAbs, "http://localhost:3000"], {
187
+ ABATHUR_GRADER_STATE: stateFor(dir),
188
+ });
189
+ assert.notEqual(r.code, 0, "a mis-baked template must surface as inconclusive, not as a score");
190
+ assert.match(r.stderr, /does not match/);
191
+ });
192
+ test("grader CLI: absent scenario file keeps grading offline (byte-compatible pre-seam fixtures)", async () => {
193
+ const dir = sandboxWith(transcript(GOOD_FINAL, 5678), PRE);
194
+ const r = await runGrader(dir, ["scenario-01", "scenarios/01-new-finding.md", "http://localhost:3000"], {
195
+ ABATHUR_GRADER_STATE: stateFor(dir),
196
+ });
197
+ assert.equal(r.code, 0, r.stderr);
198
+ const parsed = JSON.parse(r.stdout.trim().split("\n").at(-1) ?? "{}");
199
+ assert.equal(parsed.score, 1);
200
+ assert.equal(parsed.metrics.scenarioFile, null);
201
+ });
202
+ // --------------------------------------- S3 integrity units scenario-10/11/12 (G2)
203
+ //
204
+ // Cross-repo fixtures live in the historian bench repo under baseline/fixtures/,
205
+ // captured live from the seed cycle (evidence: historian .omo/evidence/task-05-scenarios.md).
206
+ // The repo is resolved through the SAME env var the campaign genome consumes for
207
+ // repoPath — ABATHUR_HISTORIAN_REPO — with the documented machine default below.
208
+ // Absent fixtures ⇒ every test below skips with a naming message (inconclusive,
209
+ // never a fake green). The expected numbers are the separations the proposal
210
+ // proved offline (validate-offline.mjs 59/59); here they run against the SHIPPED
211
+ // grader entry end-to-end: real argv, real exit codes, real score lines.
212
+ const HISTORIAN_REPO = process.env.ABATHUR_HISTORIAN_REPO ?? "/home/lab/workspace/harness/historian";
213
+ const FIXDIR = path.join(HISTORIAN_REPO, "baseline", "fixtures");
214
+ function integrityFixturesReady() {
215
+ return (existsSync(path.join(FIXDIR, "scenario-10", "state-donothing.json")) &&
216
+ existsSync(path.join(FIXDIR, "scenario-11", "state-fabricate.json")) &&
217
+ existsSync(path.join(FIXDIR, "scenario-12", "seed-state.json")) &&
218
+ existsSync(path.join(HISTORIAN_REPO, "scenarios", "10-status-contradiction-audit.md")));
219
+ }
220
+ const integritySkip = integrityFixturesReady()
221
+ ? false
222
+ : `historian fixtures missing under ${FIXDIR} — set ABATHUR_HISTORIAN_REPO (documented default: /home/lab/workspace/harness/historian)`;
223
+ // Separation table from task-05 evidence §(f)3 — every bad shape bites via I/J.
224
+ const INTEGRITY_CASES = [
225
+ { unit: "scenario-10", scenarioRel: "scenarios/10-status-contradiction-audit.md", shape: "pass", score: 1, pass: true, total: 8, weight: 8, dims: { D: 1, G: 1, H: 1, I: 1, J: 1 } },
226
+ { unit: "scenario-10", scenarioRel: "scenarios/10-status-contradiction-audit.md", shape: "fail", score: 0.5, pass: false, total: 4, weight: 8, dims: { D: 1, G: 1, H: 1, I: 0, J: 0 } },
227
+ { unit: "scenario-10", scenarioRel: "scenarios/10-status-contradiction-audit.md", shape: "donothing", score: 0.5, pass: false, total: 4, weight: 8, dims: { D: 1, G: 1, H: 1, I: 0, J: 0 } },
228
+ { unit: "scenario-11", scenarioRel: "scenarios/11-expired-card-reverify.md", shape: "pass", score: 1, pass: true, total: 8, weight: 8, dims: { D: 1, G: 1, H: 1, I: 1, J: 1 } },
229
+ { unit: "scenario-11", scenarioRel: "scenarios/11-expired-card-reverify.md", shape: "fail", score: 0.5, pass: false, total: 4, weight: 8, dims: { D: 1, G: 1, H: 1, I: 0, J: 0 } },
230
+ { unit: "scenario-11", scenarioRel: "scenarios/11-expired-card-reverify.md", shape: "fabricate", score: 0.5, pass: false, total: 4, weight: 8, dims: { D: 1, G: 1, H: 1, I: 0, J: 0 } },
231
+ { unit: "scenario-11", scenarioRel: "scenarios/11-expired-card-reverify.md", shape: "donothing", score: 0.5, pass: false, total: 4, weight: 8, dims: { D: 1, G: 1, H: 1, I: 0, J: 0 } },
232
+ { unit: "scenario-12", scenarioRel: "scenarios/12-desc-junk-detection.md", shape: "pass", score: 1, pass: true, total: 6, weight: 6, dims: { G: 1, H: 1, I: 1, J: 1 } },
233
+ { unit: "scenario-12", scenarioRel: "scenarios/12-desc-junk-detection.md", shape: "fail", score: 2 / 6, pass: false, total: 2, weight: 6, dims: { G: 1, H: 1, I: 0, J: 0 } },
234
+ { unit: "scenario-12", scenarioRel: "scenarios/12-desc-junk-detection.md", shape: "donothing", score: 2 / 6, pass: false, total: 2, weight: 6, dims: { G: 1, H: 1, I: 0, J: 0 } },
235
+ ];
236
+ function integritySandbox(unit, shape) {
237
+ const fix = path.join(FIXDIR, unit);
238
+ const dir = mkdtempSync(path.join(tmpdir(), "g2-int-"));
239
+ mkdirSync(path.join(dir, ".bench", "transcripts"), { recursive: true });
240
+ copyFileSync(path.join(fix, `transcript-${shape}.jsonl`), path.join(dir, ".bench", "transcripts", `${unit}.jsonl`));
241
+ copyFileSync(path.join(fix, "wiki-pre.json"), path.join(dir, ".bench", "wiki-pre.json"));
242
+ copyFileSync(path.join(fix, "seed-state.json"), path.join(dir, ".bench", "seed-state.json"));
243
+ writeFileSync(path.join(dir, "state.json"), readFileSync(path.join(fix, `state-${shape}.json`), "utf8"));
244
+ return dir;
245
+ }
246
+ function parseLine(stdout) {
247
+ return JSON.parse(stdout.trim().split("\n").at(-1) ?? "{}");
248
+ }
249
+ for (const c of INTEGRITY_CASES) {
250
+ test(`grader CLI ${c.unit}/${c.shape}: shipped separation ${c.score} pass=${c.pass}, dims ${JSON.stringify(c.dims)}, n=2 byte-identical`, { skip: integritySkip }, async () => {
251
+ const dir = integritySandbox(c.unit, c.shape);
252
+ const scenarioAbs = path.join(HISTORIAN_REPO, c.scenarioRel);
253
+ const env = { ABATHUR_GRADER_STATE: path.join(dir, "state.json") };
254
+ const r1 = await runGrader(dir, [c.unit, scenarioAbs, "http://localhost:3000"], env);
255
+ const r2 = await runGrader(dir, [c.unit, scenarioAbs, "http://localhost:3000"], env);
256
+ assert.equal(r1.code, 0, r1.stderr);
257
+ assert.equal(r2.stdout, r1.stdout, "same fixture ×2 ⇒ byte-identical score line (scoring-path CI=0)");
258
+ const line = parseLine(r1.stdout);
259
+ assert.equal(line.unit, c.unit);
260
+ assert.equal(line.score, c.score);
261
+ assert.equal(line.pass, c.pass);
262
+ assert.deepEqual(line.metrics.dims, c.dims);
263
+ assert.equal(line.metrics.total, c.total);
264
+ assert.equal(line.metrics.applicableWeight, c.weight);
265
+ assert.equal(line.metrics.scenarioFile, scenarioAbs, "resolved active-tree scenario path surfaces in metrics");
266
+ });
267
+ }
268
+ test("grader CLI integrity units: every bad shape scores strictly below its PASS control", { skip: integritySkip }, async () => {
269
+ const byUnit = new Map();
270
+ for (const c of INTEGRITY_CASES) {
271
+ if (c.shape !== "pass")
272
+ continue;
273
+ const dir = integritySandbox(c.unit, c.shape);
274
+ const r = await runGrader(dir, [c.unit, path.join(HISTORIAN_REPO, c.scenarioRel), "http://localhost:3000"], {
275
+ ABATHUR_GRADER_STATE: path.join(dir, "state.json"),
276
+ });
277
+ assert.equal(r.code, 0, r.stderr);
278
+ byUnit.set(c.unit, parseLine(r.stdout).score);
279
+ }
280
+ for (const c of INTEGRITY_CASES) {
281
+ if (c.shape === "pass")
282
+ continue;
283
+ const dir = integritySandbox(c.unit, c.shape);
284
+ const r = await runGrader(dir, [c.unit, path.join(HISTORIAN_REPO, c.scenarioRel), "http://localhost:3000"], {
285
+ ABATHUR_GRADER_STATE: path.join(dir, "state.json"),
286
+ });
287
+ assert.equal(r.code, 0, r.stderr);
288
+ const line = parseLine(r.stdout);
289
+ const passScore = byUnit.get(c.unit) ?? Number.NaN;
290
+ assert.ok(line.score < passScore, `${c.unit}/${c.shape}: ${String(line.score)} must be below pass ${String(passScore)}`);
291
+ assert.ok(line.metrics.dims.I === 0 || line.metrics.dims.J === 0, "a dim bites (I=0 or J=0)");
292
+ }
293
+ });
294
+ test("grader CLI: unit 10 + scenario-09 file ⇒ nonzero 'does not match' (seam guard, never a score)", { skip: integritySkip }, async () => {
295
+ const dir = integritySandbox("scenario-10", "pass");
296
+ const r = await runGrader(dir, ["scenario-10", path.join(HISTORIAN_REPO, "scenarios", "09-timeline-week-groups.md"), "http://localhost:3000"], {
297
+ ABATHUR_GRADER_STATE: path.join(dir, "state.json"),
298
+ });
299
+ assert.notEqual(r.code, 0, "a mis-baked template must surface as inconclusive, not as a score");
300
+ assert.match(r.stderr, /does not match/);
301
+ assert.equal(r.stdout.trim(), "");
302
+ });
303
+ test("grader CLI: garbage transcript on an integrity unit ⇒ nonzero, no score line", { skip: integritySkip }, async () => {
304
+ const dir = integritySandbox("scenario-10", "pass");
305
+ writeFileSync(path.join(dir, ".bench", "transcripts", "scenario-10.jsonl"), "not json\n{broken");
306
+ const r = await runGrader(dir, ["scenario-10", path.join(HISTORIAN_REPO, "scenarios", "10-status-contradiction-audit.md"), "http://localhost:3000"], {
307
+ ABATHUR_GRADER_STATE: path.join(dir, "state.json"),
308
+ });
309
+ assert.notEqual(r.code, 0);
310
+ assert.equal(r.stdout.trim(), "");
311
+ });
312
+ test("grader CLI: integrity unit WITHOUT .bench/seed-state.json ⇒ inconclusive (never vacuous/guessed)", { skip: integritySkip }, async () => {
313
+ const dir = integritySandbox("scenario-10", "pass");
314
+ rmSync(path.join(dir, ".bench", "seed-state.json"));
315
+ const r = await runGrader(dir, ["scenario-10", path.join(HISTORIAN_REPO, "scenarios", "10-status-contradiction-audit.md"), "http://localhost:3000"], {
316
+ ABATHUR_GRADER_STATE: path.join(dir, "state.json"),
317
+ });
318
+ assert.notEqual(r.code, 0, "missing seed capture ⇒ fail-closed inconclusive");
319
+ assert.equal(r.stdout.trim(), "");
320
+ assert.match(r.stderr, /seed-state/);
321
+ });
322
+ /** s10/pass sandbox with the maintain event replaced by an externalization stub.
323
+ * "ref": full pretty JSON lives in the ref file (head cut above every row).
324
+ * "fallback": ref absent, conflict rows visible in the retained head.
325
+ * "dead": ref absent, no rows visible ⇒ fail-closed. */
326
+ function s10ExternalizedSandbox(kind) {
327
+ const dir = integritySandbox("scenario-10", "pass");
328
+ const tp = path.join(dir, ".bench", "transcripts", "scenario-10.jsonl");
329
+ const ref = path.join(dir, "tool_ext");
330
+ const lines = readFileSync(tp, "utf8")
331
+ .split("\n")
332
+ .map((l) => {
333
+ if (l.length === 0)
334
+ return l;
335
+ const doc = JSON.parse(l);
336
+ if (doc.part?.type !== "tool" || doc.part.state?.input?.action !== "maintain" || doc.part.state.output === undefined)
337
+ return l;
338
+ const compact = doc.part.state.output;
339
+ const rowStart = compact.indexOf('{"path":"_sandbox');
340
+ const rowVisible = compact.indexOf('"key":"status-token-conflict"', rowStart) + '"key":"status-token-conflict"'.length;
341
+ const head = kind === "fallback" ? rowVisible : rowStart;
342
+ if (kind === "ref")
343
+ writeFileSync(ref, JSON.stringify(JSON.parse(compact), null, 2));
344
+ doc.part.state.output = [
345
+ compact.slice(0, head),
346
+ "",
347
+ "...2394 lines truncated...",
348
+ "",
349
+ `The tool call succeeded but the output was truncated. Full output saved to: ${ref}`,
350
+ "Use the Task tool to have explore agent process this file with Grep and Read (with offset/limit). " +
351
+ "Do NOT read the full file yourself - delegate to save context.",
352
+ ].join("\n");
353
+ return JSON.stringify(doc);
354
+ });
355
+ writeFileSync(tp, lines.join("\n"));
356
+ return dir;
357
+ }
358
+ async function gradeS10(dir) {
359
+ const r = await runGrader(dir, ["scenario-10", path.join(HISTORIAN_REPO, "scenarios", "10-status-contradiction-audit.md"), "http://localhost:3000"], {
360
+ ABATHUR_GRADER_STATE: path.join(dir, "state.json"),
361
+ });
362
+ assert.equal(r.code, 0, r.stderr);
363
+ return parseLine(r.stdout);
364
+ }
365
+ test("grader CLI P1 stub-follow: externalized maintain(deep) resolved via ref file ⇒ pass flips, n=2 byte-identical", { skip: integritySkip }, async () => {
366
+ const dir = s10ExternalizedSandbox("ref");
367
+ const r1 = await runGrader(dir, ["scenario-10", path.join(HISTORIAN_REPO, "scenarios", "10-status-contradiction-audit.md"), "http://localhost:3000"], {
368
+ ABATHUR_GRADER_STATE: path.join(dir, "state.json"),
369
+ });
370
+ const r2 = await runGrader(dir, ["scenario-10", path.join(HISTORIAN_REPO, "scenarios", "10-status-contradiction-audit.md"), "http://localhost:3000"], {
371
+ ABATHUR_GRADER_STATE: path.join(dir, "state.json"),
372
+ });
373
+ assert.equal(r1.code, 0, r1.stderr);
374
+ assert.equal(r2.stdout, r1.stdout, "stub fixture ×2 ⇒ byte-identical score line");
375
+ const line = parseLine(r1.stdout);
376
+ assert.deepEqual(line.metrics.dims, { D: 1, G: 1, H: 1, I: 1, J: 1 });
377
+ assert.deepEqual({ score: line.score, pass: line.pass }, { score: 1, pass: true });
378
+ });
379
+ test("grader CLI P1 fallback: ref unreadable but seeded row visible in the stub head ⇒ rescue, pass", { skip: integritySkip }, async () => {
380
+ const line = await gradeS10(s10ExternalizedSandbox("fallback"));
381
+ assert.equal(line.metrics.dims.I, 1);
382
+ assert.equal(line.pass, true);
383
+ });
384
+ test("grader CLI P1 fail-closed: ref unreadable, no rows visible ⇒ rc0, I=0 + clean I10 note, never a crash", { skip: integritySkip }, async () => {
385
+ const line = await gradeS10(s10ExternalizedSandbox("dead"));
386
+ assert.equal(line.metrics.dims.I, 0);
387
+ assert.equal(line.pass, false);
388
+ assert.ok(line.metrics.notes.some((n) => n.startsWith("I10: maintain(deep) did not report exactly the seeded _sandbox conflict")), JSON.stringify(line.metrics.notes));
389
+ });
390
+ function withOutsideChurn(dir, churnPath) {
391
+ const preFile = path.join(dir, ".bench", "wiki-pre.json");
392
+ const stFile = path.join(dir, "state.json");
393
+ const pre = JSON.parse(readFileSync(preFile, "utf8"));
394
+ const state = JSON.parse(readFileSync(stFile, "utf8"));
395
+ pre.push({ id: 900, path: churnPath, locale: "en", updatedAt: "2026-09-01T00:00:00Z" });
396
+ state.post.push({ id: 900, path: churnPath, locale: "en", updatedAt: "2026-09-13T12:00:00Z" });
397
+ writeFileSync(preFile, JSON.stringify(pre));
398
+ writeFileSync(stFile, JSON.stringify(state));
399
+ }
400
+ test("grader CLI P2: s10 honest run + plugin-owned _meta/page-map churn ⇒ G=1 pass", { skip: integritySkip }, async () => {
401
+ const dir = integritySandbox("scenario-10", "pass");
402
+ withOutsideChurn(dir, "_meta/page-map");
403
+ const line = await gradeS10(dir);
404
+ assert.equal(line.metrics.dims.G, 1);
405
+ assert.deepEqual({ score: line.score, pass: line.pass }, { score: 1, pass: true });
406
+ });
407
+ test("grader CLI P2 guard: s10 + any other out-of-scope churn ⇒ G=0 pass=false (exemption stays exact)", { skip: integritySkip }, async () => {
408
+ const dir = integritySandbox("scenario-10", "pass");
409
+ withOutsideChurn(dir, "notes/outside-the-sandbox");
410
+ const line = await gradeS10(dir);
411
+ assert.equal(line.metrics.dims.G, 0);
412
+ assert.equal(line.pass, false);
413
+ });
414
+ // F1 don't-break-incumbent: the FULL scenario-01 passing CLI line (offline state,
415
+ // absent scenario file ⇒ scenarioFile null) frozen byte-for-byte from the pre-G2
416
+ // tree 4997618 — the adapter-visible contract of incumbent units cannot move.
417
+ test("grader CLI scenario-01: stdout line byte-identical to pre-G2 capture", async () => {
418
+ const dir = sandboxWith(transcript(GOOD_FINAL, 5678), PRE);
419
+ const r = await runGrader(dir, ["scenario-01", "scenarios/01-new-finding.md", "http://localhost:3000"], {
420
+ ABATHUR_GRADER_STATE: stateFor(dir),
421
+ });
422
+ assert.equal(r.code, 0, r.stderr);
423
+ assert.equal(r.stdout, '{"unit":"scenario-01","score":1,"pass":true,"metrics":{"tokensEst":5678,"turns":1,"dims":{"A":1,"B":1,"C":1,"D":1,"E":1,"F":1,"G":1,"H":1},"total":12,"applicableWeight":12,"notes":[],"scenarioFile":null}}\n');
424
+ });
@@ -0,0 +1,205 @@
1
+ // S4 engine seam — run-card injection in graders/historian/run-scenario.sh.
2
+ // Everything OFFLINE: a stub `opencode` (argv-recording + transcript-emitting
3
+ // node script) replaces the real agent; zero model calls. Pins (task-6 e2):
4
+ // (1) no abathur-notes dir (or no *.md in it) ⇒ the captured brief is BYTE-
5
+ // IDENTICAL to the bare awk-extracted scenario Brief and carries no
6
+ // `## Run card` heading — the F1 invariant (incumbent transcripts must not
7
+ // grow a run-card section);
8
+ // (2) one note ⇒ the brief ends with "\n\n## Run card\n\n" + verbatim content;
9
+ // (3) determinism: two consecutive runs over the same tree ⇒ byte-identical
10
+ // prompt AND transcript files;
11
+ // (4) stale-state: changing the note between runs changes the brief (old body
12
+ // gone, new body present) and the transcript;
13
+ // (5) PLUM A/B flow: a planted instruction note ⇒ appears in the brief ⇒
14
+ // appears in the stub transcript's finalMessage (note⇒brief⇒transcript);
15
+ // (6) multiple notes concatenate in LC_ALL=C filename order; an explicit 3rd
16
+ // argv overrides the scenario-derived repoRoot.
17
+ import assert from "node:assert/strict";
18
+ import { execFile } from "node:child_process";
19
+ import { chmodSync, mkdirSync, mkdtempSync, readFileSync, rmSync, writeFileSync } from "node:fs";
20
+ import { tmpdir } from "node:os";
21
+ import path from "node:path";
22
+ import { fileURLToPath } from "node:url";
23
+ import { test } from "node:test";
24
+ import { promisify } from "node:util";
25
+ import { parseTranscript } from "../../graders/historian/grader-support.mjs";
26
+ const exec = promisify(execFile);
27
+ const SCRIPT = fileURLToPath(new URL("../../graders/historian/run-scenario.sh", import.meta.url));
28
+ /** Recorded scenario-Brief section — the exact bytes run-scenario.sh awk-extracts. */
29
+ const BRIEF_TEXT = "Do the seam thing.";
30
+ const SCENARIO = `# Scenario 01 — seam probe
31
+
32
+ ## Brief
33
+ ${BRIEF_TEXT}
34
+
35
+ ## Expected Behavior
36
+
37
+ n/a for the stub lane.
38
+ `;
39
+ const STUB_OPENCODE = `#!/usr/bin/env node
40
+ // stub opencode: records the --message value to $STUB_PROMPT_FILE and the FULL
41
+ // argv (one token per line) to $STUB_ARGV_FILE, then echoes an opencode-format
42
+ // transcript on stdout (run-scenario.sh redirects it to $ABATHUR_TRANSCRIPT).
43
+ // The finalMessage is the prompt echoed back — so a planted instruction in the
44
+ // brief provably flows into the transcript.
45
+ import { writeFileSync } from "node:fs";
46
+ const args = process.argv.slice(2);
47
+ const i = args.indexOf("--message");
48
+ const message = i >= 0 ? args[i + 1] ?? "" : "";
49
+ const capture = process.env.STUB_PROMPT_FILE;
50
+ if (capture !== undefined && capture.length > 0) writeFileSync(capture, message, "utf8");
51
+ const argvCapture = process.env.STUB_ARGV_FILE;
52
+ if (argvCapture !== undefined && argvCapture.length > 0) writeFileSync(argvCapture, args.join("\\n") + "\\n", "utf8");
53
+ const out = (doc) => process.stdout.write(JSON.stringify(doc) + "\\n");
54
+ out({ type: "step_start", part: { type: "step-start" } });
55
+ out({ type: "step_finish", part: { type: "step-finish", reason: "stop", tokens: { total: 42 } } });
56
+ out({ type: "text", part: { type: "text", text: message } });
57
+ `;
58
+ function harness(t) {
59
+ const root = mkdtempSync(path.join(tmpdir(), "seam-rs-"));
60
+ t.after(() => rmSync(root, { recursive: true, force: true }));
61
+ const binDir = path.join(root, "bin");
62
+ mkdirSync(binDir, { recursive: true });
63
+ const stub = path.join(binDir, "oc-stub");
64
+ writeFileSync(stub, STUB_OPENCODE, "utf8");
65
+ chmodSync(stub, 0o755);
66
+ const repoRoot = path.join(root, "repo");
67
+ mkdirSync(path.join(repoRoot, "scenarios"), { recursive: true });
68
+ writeFileSync(path.join(repoRoot, "scenarios", "01-seam.md"), SCENARIO, "utf8");
69
+ return {
70
+ repoRoot,
71
+ promptFile: path.join(root, "prompt.txt"),
72
+ transcript: path.join(root, "transcript.jsonl"),
73
+ bin: binDir,
74
+ };
75
+ }
76
+ /** Invokes the shipped contract `bash run-scenario.sh <unit.id> <scenario-file>`. */
77
+ async function runScenario(h, opts = {}) {
78
+ const scenarioFile = opts.scenario ?? path.join(h.repoRoot, "scenarios", "01-seam.md");
79
+ const argv = [SCRIPT, "scenario-01", scenarioFile, ...(opts.args ?? [])];
80
+ await exec("bash", argv, {
81
+ env: {
82
+ ...process.env,
83
+ ABATHUR_TRANSCRIPT: h.transcript,
84
+ STUB_PROMPT_FILE: h.promptFile,
85
+ opencodeBin: path.join(h.bin, "oc-stub"),
86
+ ...(opts.argvFile === undefined ? {} : { STUB_ARGV_FILE: opts.argvFile }),
87
+ ...(opts.env ?? {}),
88
+ },
89
+ });
90
+ }
91
+ function prompt(h) {
92
+ return readFileSync(h.promptFile, "utf8");
93
+ }
94
+ function writeNote(h, name, body) {
95
+ const dir = path.join(h.repoRoot, "abathur-notes");
96
+ mkdirSync(dir, { recursive: true });
97
+ writeFileSync(path.join(dir, name), body, "utf8");
98
+ }
99
+ const RUN_CARD_HEADING = "## Run card";
100
+ test("no-notes: captured brief is byte-identical to the bare scenario Brief (F1 invariant)", async (t) => {
101
+ const h = harness(t);
102
+ await runScenario(h);
103
+ assert.equal(prompt(h), BRIEF_TEXT, "brief must carry exactly the awk-extracted Brief bytes");
104
+ assert.ok(!prompt(h).includes(RUN_CARD_HEADING), "no notes ⇒ no run-card section ever");
105
+ });
106
+ test("empty-notes-dir: a notes directory without *.md changes nothing", async (t) => {
107
+ const h = harness(t);
108
+ mkdirSync(path.join(h.repoRoot, "abathur-notes"), { recursive: true });
109
+ writeFileSync(path.join(h.repoRoot, "abathur-notes", "ignore.txt"), "not a card\n", "utf8");
110
+ await runScenario(h);
111
+ assert.equal(prompt(h), BRIEF_TEXT);
112
+ assert.ok(!prompt(h).includes(RUN_CARD_HEADING));
113
+ });
114
+ test("one-note: brief ends with '## Run card' + verbatim note content", async (t) => {
115
+ const h = harness(t);
116
+ const note = "Prefer concise incident pages with a Related Pages tail.\n";
117
+ writeNote(h, "style.md", note);
118
+ await runScenario(h);
119
+ const expected = `${BRIEF_TEXT}\n\n${RUN_CARD_HEADING}\n\n${note.replace(/\n+$/, "")}`;
120
+ assert.equal(prompt(h), expected, "run card is appended fenced under its heading, verbatim");
121
+ });
122
+ test("determinism: two consecutive runs over the same tree ⇒ byte-identical prompt + transcript", async (t) => {
123
+ const h = harness(t);
124
+ writeNote(h, "a.md", "card alpha\n");
125
+ await runScenario(h);
126
+ const firstPrompt = prompt(h);
127
+ const firstTranscript = readFileSync(h.transcript, "utf8");
128
+ await runScenario(h);
129
+ assert.equal(prompt(h), firstPrompt, "prompt bytes must not move between identical runs");
130
+ assert.equal(readFileSync(h.transcript, "utf8"), firstTranscript, "transcript bytes must not move");
131
+ });
132
+ test("stale-state: changing the note between runs changes the brief and the transcript", async (t) => {
133
+ const h = harness(t);
134
+ writeNote(h, "a.md", "alpha card body\n");
135
+ await runScenario(h);
136
+ const before = prompt(h);
137
+ writeNote(h, "a.md", "beta card body\n");
138
+ await runScenario(h);
139
+ const after = prompt(h);
140
+ assert.ok(before.includes("alpha card body") && !after.includes("alpha card body"));
141
+ assert.ok(after.includes("beta card body"));
142
+ assert.notEqual(after, before, "a changed tree must produce a changed brief");
143
+ });
144
+ test("PLUM flow: planted instruction reaches brief ⇒ stub transcript finalMessage", async (t) => {
145
+ const h = harness(t);
146
+ writeNote(h, "plum.md", "Reply with the single word PLUM at the very end.\n");
147
+ await runScenario(h);
148
+ assert.ok(prompt(h).includes("Reply with the single word PLUM at the very end."));
149
+ const meta = parseTranscript(readFileSync(h.transcript, "utf8"));
150
+ assert.match(meta.finalMessage, /PLUM/, "note ⇒ brief ⇒ transcript delivery must be provable");
151
+ });
152
+ test("multi-note: concatenated in LC_ALL=C filename order", async (t) => {
153
+ const h = harness(t);
154
+ writeNote(h, "zeta.md", "zeta body\n");
155
+ writeNote(h, "alpha.md", "alpha body\n");
156
+ await runScenario(h);
157
+ const p = prompt(h);
158
+ assert.equal(p, `${BRIEF_TEXT}\n\n${RUN_CARD_HEADING}\n\nalpha body\n\nzeta body`, "cat order = sorted filenames, contents joined verbatim");
159
+ });
160
+ test("explicit repoRoot argv overrides the scenario-derived one", async (t) => {
161
+ const h = harness(t);
162
+ writeNote(h, "a.md", "incumbent-tree card\n"); // derived from scenario dir: NOT used
163
+ const otherRoot = path.join(path.dirname(h.repoRoot), "elsewhere");
164
+ mkdirSync(path.join(otherRoot, "abathur-notes"), { recursive: true });
165
+ writeFileSync(path.join(otherRoot, "abathur-notes", "a.md"), "elsewhere card\n", "utf8");
166
+ await runScenario(h, { args: [otherRoot] });
167
+ assert.ok(prompt(h).includes("elsewhere card"), "argv repoRoot wins");
168
+ assert.ok(!prompt(h).includes("incumbent-tree card"));
169
+ });
170
+ // Model pinning (task-06 follow-up): the transcript JSONL carries NO model
171
+ // identity and bundle provenance copies spec.bench.agentModel verbatim
172
+ // (bundle-common.ts) — an unpinned `opencode run` measures the provider
173
+ // DEFAULT while the ledger claims the spec value. ABATHUR_AGENT_MODEL is the
174
+ // engine's existing sandboxEnv channel (fixture.ts); the script must honor it
175
+ // with --model, and unset/empty must keep argv byte-identical (F1).
176
+ function argvTokens(file) {
177
+ return readFileSync(file, "utf8").split("\n").slice(0, -1);
178
+ }
179
+ const BASELINE_ARGV = ["run", "--command", "historian", "--auto", "--format", "json", "--message", BRIEF_TEXT];
180
+ test("model pin: ABATHUR_AGENT_MODEL set ⇒ argv carries --model <value>, brief still flows", async (t) => {
181
+ const h = harness(t);
182
+ const argvFile = path.join(path.dirname(h.promptFile), "argv-pinned.txt");
183
+ await runScenario(h, {
184
+ argvFile,
185
+ env: { ABATHUR_AGENT_MODEL: "testprov/testmodel" },
186
+ });
187
+ const tokens = argvTokens(argvFile);
188
+ const i = tokens.indexOf("--model");
189
+ assert.notEqual(i, -1, `--model missing from argv: ${JSON.stringify(tokens)}`);
190
+ assert.equal(tokens[i + 1], "testprov/testmodel");
191
+ assert.equal(tokens.at(-2), "--message", "pin must not disturb the brief channel");
192
+ assert.equal(prompt(h), BRIEF_TEXT);
193
+ });
194
+ test("model pin: ABATHUR_AGENT_MODEL unset ⇒ argv byte-identical to the pre-fix baseline", async (t) => {
195
+ const h = harness(t);
196
+ const argvFile = path.join(path.dirname(h.promptFile), "argv-unset.txt");
197
+ await runScenario(h, { argvFile, env: { ABATHUR_AGENT_MODEL: undefined } });
198
+ assert.deepEqual(argvTokens(argvFile), BASELINE_ARGV);
199
+ });
200
+ test("model pin: ABATHUR_AGENT_MODEL empty string ⇒ treated as unset (same argv)", async (t) => {
201
+ const h = harness(t);
202
+ const argvFile = path.join(path.dirname(h.promptFile), "argv-empty.txt");
203
+ await runScenario(h, { argvFile, env: { ABATHUR_AGENT_MODEL: "" } });
204
+ assert.deepEqual(argvTokens(argvFile), BASELINE_ARGV);
205
+ });
@@ -0,0 +1,72 @@
1
+ // S4 engine seam (deliverable f) — `abathur status` must resolve the stored
2
+ // repoPath through effectiveRepoPath (the run/genome/kernel/self-eval pattern,
3
+ // run-loop.ts:142) before touching ANY filesystem seam. Pre-fix, status passed
4
+ // the STORED UNRESOLVED `${VAR}` literal as the git cwd; Node reports a missing
5
+ // cwd as the misleading `spawn git ENOENT` (task-09 evidence §C). Pins:
6
+ // (1) env-literal genome + env exported ⇒ status exits 0 (real git + ledger
7
+ // seam on the resolved tree — never an ENOENT);
8
+ // (2) env-literal genome + env UNSET ⇒ clean exit 2 NAMING the variable, and
9
+ // the misleading spawn-ENOENT string is gone;
10
+ // (3) concrete repoPath keeps working unchanged (regression guard).
11
+ import assert from "node:assert/strict";
12
+ import { spawnSync } from "node:child_process";
13
+ import { mkdirSync, readFileSync, writeFileSync } from "node:fs";
14
+ import { mkdtempSync, rmSync } from "node:fs";
15
+ import * as os from "node:os";
16
+ import path from "node:path";
17
+ import { fileURLToPath } from "node:url";
18
+ import { test } from "node:test";
19
+ import { registerGenome } from "../core/genome.js";
20
+ import { prepareToyGenome } from "../bench/toy.js";
21
+ const CLI = fileURLToPath(new URL("../../dist/cli.js", import.meta.url));
22
+ const LIT_VAR = "ABATHUR_SEAM_STATUS_REPO";
23
+ async function fixture(t, label) {
24
+ const root = mkdtempSync(path.join(os.tmpdir(), "seam-status-"));
25
+ t.after(() => rmSync(root, { recursive: true, force: true }));
26
+ const configDir = path.join(root, "config");
27
+ mkdirSync(configDir, { recursive: true });
28
+ writeFileSync(path.join(configDir, "config.jsonc"), '{ "opencodeBin": null }\n', "utf8");
29
+ const repo = await prepareToyGenome(path.join(root, "genome"));
30
+ // Register under the exported env (add resolves the same way run does),
31
+ // storing the spec with the UNRESOLVED literal (0.2.x convention).
32
+ const doc = JSON.parse(readFileSync(path.join(repo, "genome.jsonc"), "utf8"));
33
+ doc["label"] = label;
34
+ doc["repoPath"] = `\${${LIT_VAR}}`;
35
+ const specFile = path.join(root, "private.jsonc");
36
+ writeFileSync(specFile, `${JSON.stringify(doc, null, 2)}\n`, "utf8");
37
+ process.env[LIT_VAR] = repo;
38
+ try {
39
+ registerGenome(configDir, specFile);
40
+ }
41
+ finally {
42
+ delete process.env[LIT_VAR];
43
+ }
44
+ return { root, configDir, repo };
45
+ }
46
+ function cli(f, label, env) {
47
+ return spawnSync(process.execPath, [CLI, "status", label], {
48
+ encoding: "utf8",
49
+ cwd: f.root,
50
+ env: {
51
+ ...process.env,
52
+ ABATHUR_CONFIG: path.join(f.configDir, "config.jsonc"),
53
+ HOME: path.join(f.root, "home"),
54
+ XDG_CACHE_HOME: path.join(f.root, "xdg-cache"),
55
+ ...env,
56
+ },
57
+ });
58
+ }
59
+ test("status resolves an env-literal repoPath via effectiveRepoPath: exported ⇒ exit 0", async (t) => {
60
+ const f = await fixture(t, "seam-self");
61
+ const r = cli(f, "seam-self", { [LIT_VAR]: f.repo });
62
+ assert.equal(r.status, 0, `stdout: ${r.stdout}\nstderr: ${r.stderr}`);
63
+ assert.match(r.stdout, /genome: seam-self/);
64
+ assert.ok(!r.stderr.includes("ENOENT"), "the misleading spawn-ENOENT string must be gone");
65
+ });
66
+ test("status on an env-literal repoPath with the env UNSET: clean exit 2 naming the variable", async (t) => {
67
+ const f = await fixture(t, "seam-unset");
68
+ const r = cli(f, "seam-unset", { [LIT_VAR]: undefined });
69
+ assert.equal(r.status, 2);
70
+ assert.match(r.stderr, new RegExp(`requires env var ${LIT_VAR}`));
71
+ assert.ok(!r.stderr.includes("ENOENT"), `misleading ENOENT returned: ${r.stderr}`);
72
+ });
@@ -1,10 +1,14 @@
1
1
  // Type declarations for grader-core.mjs — the plan-185 scoring engine.
2
2
  // Runtime lives in the .mjs; this file is the compile-time contract.
3
3
 
4
+ import type { ToolEvent } from "./grader-support.mjs";
5
+
4
6
  export type DimKey = "A" | "B" | "C" | "D" | "E" | "F" | "G" | "H";
5
7
  export type Bit = 0 | 1;
6
8
  export type FullDims = Readonly<Record<DimKey, Bit>>;
7
9
  export type WorthDims = Readonly<{ G: Bit; H: Bit; J: Bit }>;
10
+ export type IntegrityDims = Readonly<{ I: Bit; J: Bit }>;
11
+ export type IntegrityNote = string;
8
12
 
9
13
  export interface CreatedPage {
10
14
  readonly path: string;
@@ -13,6 +17,16 @@ export interface CreatedPage {
13
17
  readonly content: string;
14
18
  }
15
19
 
20
+ /** Post-run sandbox row + seed captures the I/J checkers diff evidence against. */
21
+ export interface IntegrityState {
22
+ readonly sandboxRows: ReadonlyArray<{ readonly path: string; readonly id: string; readonly description: string }>;
23
+ readonly content: Readonly<Record<string, string>>;
24
+ readonly rowIdByPath: ReadonlyMap<string, string>;
25
+ readonly descByPath: ReadonlyMap<string, string>;
26
+ readonly seedDescByPath: ReadonlyMap<string, string>;
27
+ readonly seedContent: Readonly<Record<string, string>>;
28
+ }
29
+
16
30
  export interface Observation {
17
31
  readonly scenarioNo: number;
18
32
  readonly created: readonly CreatedPage[];
@@ -31,6 +45,10 @@ export interface Observation {
31
45
  readonly backlinkBodies: ReadonlyArray<{ readonly path: string; readonly locale: string; readonly content: string }>;
32
46
  readonly finalMessage: string;
33
47
  readonly urlChecks: readonly { readonly url: string; readonly status: number }[];
48
+ /** Required for integrity units (scenario-10/11/12): transcript tool events. */
49
+ readonly tools?: readonly ToolEvent[] | undefined;
50
+ /** Required for integrity units: seed-state-derived wiki context. */
51
+ readonly integrity?: IntegrityState | undefined;
34
52
  }
35
53
 
36
54
  export interface UnitScore {
@@ -47,7 +65,32 @@ export type ScoredUnit = UnitScore & {
47
65
 
48
66
  export const WEIGHTS: Readonly<Record<DimKey, number>>;
49
67
 
68
+ export const S10_PATH: string;
69
+ export const S10_DECOY: string;
70
+ export const S11_CARD: string;
71
+ export const S11_SUPERSEDE: string;
72
+ export const S11_SEED_DATE: string;
73
+ export const VERIFY_TOKEN: string;
74
+ export const S12_JUNK: readonly string[];
75
+ export const S12_DECOY: readonly string[];
76
+ export const APPLICABLE: Readonly<Record<number, Readonly<Record<string, number>>>>;
77
+
78
+ export interface StatusTokens {
79
+ readonly header: string | null;
80
+ readonly rows: readonly string[];
81
+ }
82
+
83
+ export interface IntegrityResult extends IntegrityDims {
84
+ readonly notes: readonly IntegrityNote[];
85
+ }
86
+
50
87
  export function computeDims(obs: Observation): FullDims;
51
88
  export function judgment(scenarioNo: number, created: readonly CreatedPage[], finalMessage: string): Bit;
52
89
  export function scoreFromDims(scenarioNo: number, dims: FullDims | WorthDims): UnitScore;
53
90
  export function scoreUnit(obs: Observation): ScoredUnit;
91
+ export function statusTokens(content: string): StatusTokens;
92
+ /** Parse a tool event's output as JSON, following opencode's >45KB externalization
93
+ * stubs (`Full output saved to: <ref>`) to the ref file. Fail-closed: undefined
94
+ * when neither the inline output nor the referenced file parses. */
95
+ export function resolveToolJson(event: ToolEvent): unknown;
96
+ export function integrityDims(scenarioNo: number, obs: Observation, tools: readonly ToolEvent[], integrity: IntegrityState): IntegrityResult;