@forwardimpact/libharness 1.4.0 → 3.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -17,12 +17,12 @@ traces they produce, and edits skill files under controlled conditions.
17
17
 
18
18
  | CLI | Purpose |
19
19
  | --------------- | ---------------------------------------------------------------------- |
20
- | `fit-harness` | Run agents in `run`/`supervise`/`facilitate`/`discuss` subcommands. |
21
- | `fit-trace` | Download, query, and analyze NDJSON traces produced by `fit-harness`. |
22
- | `fit-benchmark` | Run task families for N runs each and aggregate pass@k. |
23
- | `fit-selfedit` | Write stdin to `.claude/**` paths, gated by settings.json + branch. |
20
+ | `gemba-harness` | Run agents in `run`/`supervise`/`facilitate`/`discuss` subcommands. |
21
+ | `gemba-trace` | Download, query, and analyze NDJSON traces produced by `gemba-harness`. |
22
+ | `gemba-benchmark` | Run task families for N runs each and aggregate pass@k. |
23
+ | `gemba-selfedit` | Write stdin to `.claude/**` paths, gated by settings.json + branch. |
24
24
 
25
- `fit-harness`'s subcommands share one orchestration loop and one async tool
25
+ `gemba-harness`'s subcommands share one orchestration loop and one async tool
26
26
  surface, below. The `judge` role is a profile passed to `supervise`.
27
27
 
28
28
  ## Modes
@@ -147,9 +147,9 @@ Each line is `{ "source": "<participant|orchestrator>", "seq": N, "event":
147
147
  {…} }`. `seq` is monotonic across the whole trace; `orchestrator` emits
148
148
  `session_start`, `agent_start`, `protocol_violation`, `lead_turn_limit`,
149
149
  and `summary`. `event` is the SDK event verbatim or the orchestrator
150
- payload. `fit-trace` consumes this format.
150
+ payload. `gemba-trace` consumes this format.
151
151
 
152
- Redaction is on by default for `fit-harness run`/`supervise`/`facilitate`
152
+ Redaction is on by default for `gemba-harness run`/`supervise`/`facilitate`
153
153
  and composes two layers:
154
154
 
155
155
  - **Env-var allowlist** — `ANTHROPIC_API_KEY`, `GH_TOKEN`, `GITHUB_TOKEN`
@@ -178,7 +178,7 @@ downloadable through retention.
178
178
  | `trace-collector.js` / `trace-query.js` / `trace-github.js` | Trace ingestion / querying / GitHub-attachment helpers. |
179
179
  | `redaction.js` | Env-var allowlist + credential-shape pattern redaction. |
180
180
 
181
- ## fit-selfedit
181
+ ## gemba-selfedit
182
182
 
183
183
  A narrow, audited bypass for sessions where `Edit`/`Write` (and bash
184
184
  writes) are blocked against paths the project's own allowlist permits.
@@ -186,7 +186,7 @@ Reads stdin, writes the target, exits 0 / 2 (safeguard violation) / 1
186
186
  (I/O error).
187
187
 
188
188
  ```sh
189
- echo "<content>" | bunx fit-selfedit <path>
189
+ echo "<content>" | bunx gemba-selfedit <path>
190
190
  ```
191
191
 
192
192
  Two safeguards, checked in order:
@@ -221,7 +221,7 @@ lists the `Edit()` rules that were tried.
221
221
  — end-to-end workflow from dataset generation through evaluation to trace
222
222
  analysis, including multi-agent collaboration sessions.
223
223
  - [Analyze Traces](https://www.forwardimpact.team/docs/libraries/prove-changes/trace-analysis/index.md)
224
- — read the NDJSON traces produced by `fit-harness` with `fit-trace`.
224
+ — read the NDJSON traces produced by `gemba-harness` with `gemba-trace`.
225
225
  - [Agent Teams](https://www.forwardimpact.team/docs/products/agent-teams/index.md)
226
226
  — author the profiles consumed by `--agent-profile`, `--lead-profile`, and
227
227
  `--agent-profiles`.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@forwardimpact/libharness",
3
- "version": "1.4.0",
3
+ "version": "3.0.0",
4
4
  "description": "Autonomous agent team harness — coordinate a lead and participant agents in one async session, with eval, benchmark, and trace tooling to prove the changes worked.",
5
5
  "keywords": [
6
6
  "orchestration",
@@ -40,20 +40,22 @@
40
40
  "main": "./src/index.js",
41
41
  "exports": {
42
42
  ".": "./src/index.js",
43
- "./bin/fit-harness.js": "./bin/fit-harness.js",
44
- "./bin/fit-trace.js": "./bin/fit-trace.js",
45
- "./bin/fit-benchmark.js": "./bin/fit-benchmark.js",
46
- "./bin/fit-selfedit.js": "./bin/fit-selfedit.js"
47
- },
48
- "bin": {
49
- "fit-harness": "./bin/fit-harness.js",
50
- "fit-trace": "./bin/fit-trace.js",
51
- "fit-benchmark": "./bin/fit-benchmark.js",
52
- "fit-selfedit": "./bin/fit-selfedit.js"
43
+ "./commands/output.js": "./src/commands/output.js",
44
+ "./commands/tee.js": "./src/commands/tee.js",
45
+ "./commands/run.js": "./src/commands/run.js",
46
+ "./commands/supervise.js": "./src/commands/supervise.js",
47
+ "./commands/facilitate.js": "./src/commands/facilitate.js",
48
+ "./commands/discuss.js": "./src/commands/discuss.js",
49
+ "./commands/callback.js": "./src/commands/callback.js",
50
+ "./commands/scan-logs.js": "./src/commands/scan-logs.js",
51
+ "./commands/trace.js": "./src/commands/trace.js",
52
+ "./commands/assert.js": "./src/commands/assert.js",
53
+ "./commands/by-discussion.js": "./src/commands/by-discussion.js",
54
+ "./commands/benchmark-definition.js": "./src/commands/benchmark-definition.js",
55
+ "./commands/selfedit.js": "./src/commands/selfedit.js"
53
56
  },
54
57
  "files": [
55
58
  "src/**/*.js",
56
- "bin/**/*.js",
57
59
  "README.md"
58
60
  ],
59
61
  "scripts": {
@@ -1,7 +1,7 @@
1
1
  /**
2
2
  * AgentRunner — runs a single Claude Agent SDK session and emits raw
3
- * NDJSON events to an output stream. Building block for `fit-harness run`,
4
- * `fit-harness supervise`, `fit-harness facilitate`, and `fit-harness discuss`.
3
+ * NDJSON events to an output stream. Building block for `gemba-harness run`,
4
+ * `gemba-harness supervise`, `gemba-harness facilitate`, and `gemba-harness discuss`.
5
5
  *
6
6
  * Follows OO+DI: constructor injection, factory function, tests bypass factory.
7
7
  */
@@ -37,7 +37,7 @@ function modelDidWork(result) {
37
37
  return tokens > 0 || (cost ?? 0) > 0;
38
38
  }
39
39
 
40
- // fit-harness and kata-action run headless in CI/CD with no human to answer
40
+ // gemba-harness and kata-action run headless in CI/CD with no human to answer
41
41
  // permission prompts. The SDK is always launched in bypass mode — not
42
42
  // overridable — so a future caller can't accidentally reduce permissions.
43
43
  const PERMISSION_MODE = "bypassPermissions";
@@ -0,0 +1,222 @@
1
+ /**
2
+ * Grading derivation — the sole home of the check-row arithmetic.
3
+ *
4
+ * Check rows are the single authoritative grading channel. Every row is a
5
+ * check by default; a row declares its role with its own fields, checked in
6
+ * order:
7
+ *
8
+ * 1. Gate — `gate` is exactly `true`, `pass` is boolean, and no
9
+ * `weight` key is present. Any failing gate → `gatesPass`
10
+ * false.
11
+ * 2. Diagnostic — no `gate` key and `weight` is exactly `0`. Free-form;
12
+ * never graded.
13
+ * 3. Scored — no `gate` key, boolean `pass`, `weight` absent (defaults
14
+ * to 1) or finite > 0.
15
+ * 4. Malformed — everything else: any `gate`+`weight` co-occurrence (a
16
+ * stray weight must never silently disarm a gate), a
17
+ * non-boolean `gate`, a missing or non-boolean `pass` on a
18
+ * graded row, an invalid `weight`, an fd-3 line that failed
19
+ * to parse, a non-object row. Counts as a **failing scored
20
+ * check** — dropping a defect could mint full marks;
21
+ * failing the whole run would zero completed work.
22
+ *
23
+ * The producers' `source` stamp is display metadata, never a grading input.
24
+ */
25
+
26
+ /**
27
+ * @typedef {object} GradeResult
28
+ * @property {"pass" | "fail"} verdict - `healthy ∧ gatesPass ∧ fullMarks`.
29
+ * @property {boolean} gatesPass - Every gate row passes (vacuously true).
30
+ * @property {number | null} score - Weighted fraction of passing scored
31
+ * checks; `null` when the cell has zero scored checks (binary task).
32
+ * @property {boolean} fullMarks - Integer count predicate: no malformed rows
33
+ * and every scored check passes. Never a float comparison, so fractional
34
+ * weights carry no equality hazard. Vacuously true with zero scored checks.
35
+ * @property {number} malformed - Malformed row count.
36
+ */
37
+
38
+ /**
39
+ * Grade the merged check rows against grader health.
40
+ *
41
+ * `healthy` is the completion signal a crashed grader cannot fake: when it is
42
+ * false the verdict is `fail` whatever the rows say, so a hook that dies
43
+ * after emitting passing rows can never mint marks.
44
+ * @param {unknown[]} details - Merged check rows from both producers.
45
+ * @param {boolean} healthy - Invariants exited 0 AND the hidden-test engine
46
+ * did not throw.
47
+ * @returns {GradeResult}
48
+ */
49
+ export function gradeChecks(details, healthy) {
50
+ const tally = {
51
+ gatesPass: true,
52
+ malformed: 0,
53
+ scored: 0,
54
+ passing: 0,
55
+ weightAll: 0,
56
+ weightPassing: 0,
57
+ };
58
+ for (const row of details) tallyRow(tally, row);
59
+
60
+ const score =
61
+ tally.scored + tally.malformed === 0
62
+ ? null
63
+ : tally.weightPassing / tally.weightAll;
64
+ const fullMarks = tally.malformed === 0 && tally.passing === tally.scored;
65
+ const verdict = healthy && tally.gatesPass && fullMarks ? "pass" : "fail";
66
+ return {
67
+ verdict,
68
+ gatesPass: tally.gatesPass,
69
+ score,
70
+ fullMarks,
71
+ malformed: tally.malformed,
72
+ };
73
+ }
74
+
75
+ /**
76
+ * Fold one row into the running tally per its classified role.
77
+ * @param {{gatesPass: boolean, malformed: number, scored: number, passing: number, weightAll: number, weightPassing: number}} tally
78
+ * @param {unknown} row
79
+ */
80
+ function tallyRow(tally, row) {
81
+ const role = classifyRow(row);
82
+ if (role === "gate") {
83
+ if (!row.pass) tally.gatesPass = false;
84
+ } else if (role === "scored") {
85
+ const weight = row.weight ?? 1;
86
+ tally.scored++;
87
+ tally.weightAll += weight;
88
+ if (row.pass) {
89
+ tally.passing++;
90
+ tally.weightPassing += weight;
91
+ }
92
+ } else if (role === "malformed") {
93
+ tally.malformed++;
94
+ tally.weightAll += malformedWeight(row);
95
+ }
96
+ }
97
+
98
+ /**
99
+ * Run both check-row producers and grade the merged rows — the one
100
+ * composition shared by the runner and the `grade` subcommand. An engine
101
+ * throw is grader fault: its message lands on the returned `engineError`
102
+ * and health fails, so a crashed grader can never mint marks from rows it
103
+ * happened to emit first.
104
+ * @param {import("./task-family.js").Task} task
105
+ * @param {{cwd: string, port: number, runDir: string, familyDir?: string|null}} ctx
106
+ * @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
107
+ * @param {{runInvariants: Function, runHiddenTests: Function}} producers -
108
+ * The two producer functions (real implementations or test seams).
109
+ * @returns {Promise<{invariants: object, hiddenRows: object[], engineError: Error|null, rows: unknown[], healthy: boolean, grade: object}>}
110
+ */
111
+ export async function runProducersAndGrade(task, ctx, runtime, producers) {
112
+ const invariants = await producers.runInvariants(task, ctx, runtime);
113
+ let hiddenRows = [];
114
+ let engineError = null;
115
+ try {
116
+ const hidden = await producers.runHiddenTests(task, ctx, runtime);
117
+ hiddenRows = hidden.details;
118
+ } catch (e) {
119
+ engineError = e;
120
+ }
121
+ const rows = mergeRows(invariants.details, hiddenRows);
122
+ const healthy = invariants.exitCode === 0 && !engineError;
123
+ const grade = normalizeGrade(gradeChecks(rows, healthy));
124
+ return { invariants, hiddenRows, engineError, rows, healthy, grade };
125
+ }
126
+
127
+ /**
128
+ * Merge the two producers' rows (invariants first) and stamp each row's
129
+ * provenance. The stamp is display metadata, never a grading input, and
130
+ * non-object rows (malformed by contract) pass through verbatim.
131
+ * @param {unknown[]} invariantsDetails
132
+ * @param {unknown[]} hiddenDetails
133
+ * @returns {unknown[]}
134
+ */
135
+ export function mergeRows(invariantsDetails, hiddenDetails) {
136
+ return [
137
+ ...invariantsDetails.map((row) => stampSource(row, "invariants")),
138
+ ...hiddenDetails.map((row) => stampSource(row, "tests")),
139
+ ];
140
+ }
141
+
142
+ function stampSource(row, source) {
143
+ if (row === null || typeof row !== "object" || Array.isArray(row)) {
144
+ return row;
145
+ }
146
+ return { ...row, source };
147
+ }
148
+
149
+ /**
150
+ * Project the raw `gradeChecks` return onto the record schema: `fullMarks`
151
+ * is derivable and dropped, `score` is omitted on binary tasks (`null`),
152
+ * `malformed` is omitted when clean.
153
+ * @param {GradeResult} raw
154
+ * @returns {{verdict: "pass"|"fail", gatesPass: boolean, score?: number, malformed?: number}}
155
+ */
156
+ export function normalizeGrade({ verdict, gatesPass, score, malformed }) {
157
+ return {
158
+ verdict,
159
+ gatesPass,
160
+ ...(score !== null && { score }),
161
+ ...(malformed > 0 && { malformed }),
162
+ };
163
+ }
164
+
165
+ /**
166
+ * Classify one row per the role order in the module contract.
167
+ * @param {unknown} row
168
+ * @returns {"gate" | "diagnostic" | "scored" | "malformed"}
169
+ */
170
+ function classifyRow(row) {
171
+ if (row === null || typeof row !== "object" || Array.isArray(row)) {
172
+ return "malformed";
173
+ }
174
+ if ("gate" in row) return classifyGateRow(row);
175
+ if ("weight" in row) return classifyWeightedRow(row);
176
+ return typeof row.pass === "boolean" ? "scored" : "malformed";
177
+ }
178
+
179
+ /**
180
+ * A row carrying a `gate` key: valid only as `gate: true` with a boolean
181
+ * `pass` and no `weight` key — any co-occurring weight is malformed so a
182
+ * stray weight can never silently disarm a gate.
183
+ * @param {object} row
184
+ * @returns {"gate" | "malformed"}
185
+ */
186
+ function classifyGateRow(row) {
187
+ if ("weight" in row) return "malformed";
188
+ return row.gate === true && typeof row.pass === "boolean"
189
+ ? "gate"
190
+ : "malformed";
191
+ }
192
+
193
+ /**
194
+ * A gate-less row carrying a `weight` key: exactly 0 is a diagnostic, a
195
+ * finite positive weight with a boolean `pass` is scored, anything else is
196
+ * malformed.
197
+ * @param {object} row
198
+ * @returns {"diagnostic" | "scored" | "malformed"}
199
+ */
200
+ function classifyWeightedRow(row) {
201
+ if (row.weight === 0) return "diagnostic";
202
+ return isValidWeight(row.weight) && typeof row.pass === "boolean"
203
+ ? "scored"
204
+ : "malformed";
205
+ }
206
+
207
+ /**
208
+ * A malformed row fails at its own weight when it carries a valid positive
209
+ * one, else at unit weight 1.
210
+ * @param {unknown} row
211
+ * @returns {number}
212
+ */
213
+ function malformedWeight(row) {
214
+ if (row !== null && typeof row === "object" && isValidWeight(row.weight)) {
215
+ return row.weight;
216
+ }
217
+ return 1;
218
+ }
219
+
220
+ function isValidWeight(w) {
221
+ return typeof w === "number" && Number.isFinite(w) && w > 0;
222
+ }
@@ -0,0 +1,180 @@
1
+ /**
2
+ * Hidden-test engine — executes a task's `tests/` overlay against the
3
+ * post-run agent CWD: stage each file at its mirrored path, run each check
4
+ * with `node --test`, convert the exit status into one check row, and
5
+ * restore the tree so the judge sees the workdir exactly as the agent left
6
+ * it.
7
+ *
8
+ * Fault attribution is the engine's contract: a stage or spawn failure (the
9
+ * agent deleted the scaffold) is a *failing row* — agent fault; the engine
10
+ * itself throwing is grader fault, which the caller records as unhealthy so
11
+ * a crashed grader can never mint marks.
12
+ */
13
+
14
+ import { dirname, join } from "node:path";
15
+
16
+ import { buildHookEnv } from "./hook-env.js";
17
+
18
+ // Fixed per-check budget. A wedged test process runs outside the agent
19
+ // watchdog, so this bound is what keeps a hung hidden test from stalling the
20
+ // cell; the timeout row keeps the failure visible.
21
+ const CHECK_TIMEOUT_MS = 120_000;
22
+ const STDERR_TAIL_CHARS = 500;
23
+
24
+ /**
25
+ * Run the task's hidden test suite.
26
+ * @param {import("./task-family.js").Task} task
27
+ * @param {{cwd: string, port: number, runDir: string, familyDir?: string|null}} ctx
28
+ * @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
29
+ * @param {{timeoutMs?: number}} [opts] - Test seam for the per-check timeout.
30
+ * @returns {Promise<{details: object[]}>}
31
+ */
32
+ export async function runHiddenTests(task, ctx, runtime, opts = {}) {
33
+ if (!runtime) throw new Error("runtime is required");
34
+ if (!task.tests) return { details: [] };
35
+ const timeoutMs = opts.timeoutMs ?? CHECK_TIMEOUT_MS;
36
+ const fs = runtime.fs;
37
+ const details = [];
38
+ const supportStager = newStager();
39
+ try {
40
+ for (const file of task.tests.support) {
41
+ await stageFile(fs, ctx.cwd, supportStager, file);
42
+ }
43
+ for (const check of task.tests.checks) {
44
+ details.push(await runOneCheck(task, ctx, runtime, timeoutMs, check));
45
+ }
46
+ } finally {
47
+ await unstage(fs, supportStager);
48
+ }
49
+ return { details };
50
+ }
51
+
52
+ /**
53
+ * Stage one check, run it, and restore its staging — the check's own row is
54
+ * the only trace it leaves. A stage failure is the agent's fault (a deleted
55
+ * scaffold), so it becomes a failing row rather than a throw.
56
+ */
57
+ async function runOneCheck(task, ctx, runtime, timeoutMs, check) {
58
+ const stager = newStager();
59
+ try {
60
+ try {
61
+ await stageFile(runtime.fs, ctx.cwd, stager, check);
62
+ } catch (e) {
63
+ return checkRow(check, false, `stage failed: ${e.message}`);
64
+ }
65
+ return await spawnCheck(task, ctx, runtime, timeoutMs, check);
66
+ } finally {
67
+ await unstage(runtime.fs, stager);
68
+ }
69
+ }
70
+
71
+ /**
72
+ * Spawn `node --test <staged path>` from the agent CWD under the hook env
73
+ * and map the exit status onto one row. The clock timer SIGKILLs a child
74
+ * that outlives the per-check budget; the row fails with a timeout message.
75
+ */
76
+ async function spawnCheck(task, ctx, runtime, timeoutMs, check) {
77
+ const env = buildHookEnv(runtime.proc.env, {
78
+ cwd: ctx.cwd,
79
+ port: ctx.port,
80
+ taskId: task.id,
81
+ taskDir: task.paths.taskDir,
82
+ hooksDir: task.paths.hooks,
83
+ familyDir: ctx.familyDir,
84
+ });
85
+ // An inherited test-runner context makes the child `node --test` report
86
+ // exit 0 even when its tests fail — a failing check would mint a passing
87
+ // row whenever the harness itself runs under `node --test`.
88
+ delete env.NODE_TEST_CONTEXT;
89
+ const child = runtime.subprocess.spawn("node", ["--test", check.stagePath], {
90
+ cwd: ctx.cwd,
91
+ env,
92
+ stdio: ["ignore", "pipe", "pipe"],
93
+ });
94
+ let timedOut = false;
95
+ const timer = runtime.clock.setTimeout(() => {
96
+ timedOut = true;
97
+ child.kill("SIGKILL");
98
+ }, timeoutMs);
99
+ const drainStdout = (async () => {
100
+ for await (const _chunk of child.stdout) {
101
+ // discard
102
+ }
103
+ })();
104
+ let stderr = "";
105
+ for await (const chunk of child.stderr) stderr += chunk.toString();
106
+ await drainStdout;
107
+ const exit = await child.exitCode;
108
+ runtime.clock.clearTimeout(timer);
109
+
110
+ if (timedOut) {
111
+ return checkRow(check, false, `timed out after ${timeoutMs}ms`);
112
+ }
113
+ if (exit === 0) return checkRow(check, true);
114
+ const tail = stderr.trim().slice(-STDERR_TAIL_CHARS);
115
+ return checkRow(check, false, `exit ${exit}${tail ? `: ${tail}` : ""}`);
116
+ }
117
+
118
+ function checkRow(check, pass, message) {
119
+ return {
120
+ test: check.name,
121
+ pass,
122
+ ...(check.gate && { gate: true }),
123
+ ...(message && { message }),
124
+ };
125
+ }
126
+
127
+ function newStager() {
128
+ return { staged: [], backups: [], createdDirs: [] };
129
+ }
130
+
131
+ /**
132
+ * Copy the symlink-resolved source to its mirrored path under the agent CWD,
133
+ * backing up a collided file's bytes and tracking every directory created so
134
+ * `unstage` can put the tree back exactly.
135
+ */
136
+ async function stageFile(fs, cwd, stager, { sourcePath, stagePath }) {
137
+ const target = join(cwd, stagePath);
138
+ let collided = null;
139
+ try {
140
+ collided = await fs.readFile(target);
141
+ } catch {
142
+ // no collision
143
+ }
144
+ if (collided !== null) stager.backups.push({ target, bytes: collided });
145
+ await ensureParents(fs, cwd, stager, dirname(target));
146
+ const resolved = await fs.realpath(sourcePath);
147
+ await fs.copyFile(resolved, target);
148
+ stager.staged.push(target);
149
+ }
150
+
151
+ async function ensureParents(fs, cwd, stager, dir) {
152
+ if (dir === cwd) return;
153
+ try {
154
+ await fs.access(dir);
155
+ return;
156
+ } catch {
157
+ // missing — create below
158
+ }
159
+ await ensureParents(fs, cwd, stager, dirname(dir));
160
+ await fs.mkdir(dir);
161
+ stager.createdDirs.push(dir);
162
+ }
163
+
164
+ /**
165
+ * Reverse the staging: staged copies out, collided bytes back, created
166
+ * directories removed (deepest first — a check's own artifacts inside a
167
+ * created directory go with it, since that directory did not exist when the
168
+ * agent finished).
169
+ */
170
+ async function unstage(fs, stager) {
171
+ for (const target of stager.staged) {
172
+ await fs.rm(target, { force: true });
173
+ }
174
+ for (const backup of stager.backups) {
175
+ await fs.writeFile(backup.target, backup.bytes);
176
+ }
177
+ for (const dir of [...stager.createdDirs].reverse()) {
178
+ await fs.rm(dir, { recursive: true, force: true });
179
+ }
180
+ }
@@ -1,7 +1,10 @@
1
1
  /**
2
2
  * Invariants — runs `<task.paths.hooks>/invariants.sh` from the template path
3
- * against the post-run agent CWD. The exit code is authoritative for the
4
- * verdict; structured per-check rows arrive on fd 3 (`$RESULTS_FD=3`) as NDJSON.
3
+ * against the post-run agent CWD. A pure collector with no verdict of its
4
+ * own: structured per-check rows arrive on fd 3 (`$RESULTS_FD=3`) as NDJSON
5
+ * and grading happens downstream over the merged rows. The exit code is
6
+ * script health only — nonzero means the grader itself failed, never that a
7
+ * check failed.
5
8
  *
6
9
  * Subprocess access flows through `runtime.subprocess.spawn`; the fd-3 backing
7
10
  * store and the stderr log use the sync filesystem surface (`runtime.fsSync`) —
@@ -14,9 +17,9 @@ import { buildHookEnv } from "./hook-env.js";
14
17
 
15
18
  /**
16
19
  * @typedef {object} InvariantsResult
17
- * @property {"pass" | "fail"} verdict
18
20
  * @property {Array<object>} details
19
- * @property {number} exitCode
21
+ * @property {number} exitCode - Script health: nonzero means the hook itself
22
+ * failed, never that a check failed.
20
23
  * @property {string} [stderr] - Trimmed script stderr, present only when the
21
24
  * script wrote to stderr. Surfaces hook failures (e.g. a missing tool) that
22
25
  * leave `details` empty, so they read distinctly from a real invariant miss.
@@ -32,7 +35,7 @@ import { buildHookEnv } from "./hook-env.js";
32
35
  export async function runInvariants(task, ctx, runtime) {
33
36
  if (!runtime) throw new Error("runtime is required");
34
37
  if (!task.paths.invariants) {
35
- return { verdict: "pass", details: [], exitCode: 0 };
38
+ return { details: [], exitCode: 0 };
36
39
  }
37
40
  const fsSync = runtime.fsSync;
38
41
  const script = task.paths.invariants;
@@ -84,11 +87,7 @@ export async function runInvariants(task, ctx, runtime) {
84
87
  const details = [];
85
88
  parseFd3Buffer(raw, details);
86
89
  const exitCode = typeof code === "number" ? code : -1;
87
- const result = {
88
- verdict: exitCode === 0 ? "pass" : "fail",
89
- details,
90
- exitCode,
91
- };
90
+ const result = { details, exitCode };
92
91
  const trimmedStderr = stderr.trim();
93
92
  if (trimmedStderr) result.stderr = trimmedStderr;
94
93
  return result;
@@ -9,7 +9,7 @@
9
9
  * {{AGENT_INSTRUCTIONS}} — contents of agent.task.md
10
10
  * {{AGENT_PROFILE}} — agent profile body (empty string if none)
11
11
  * {{AGENT_TRACE_PATH}} — path to agent.ndjson
12
- * {{INVARIANTS_RESULT}} — JSON invariants object
12
+ * {{GRADE_RESULT}} — JSON grade object plus the merged check rows
13
13
  * {{SKILL_SET_HASH}} — SHA-256 from apm.lock.yaml
14
14
  * {{TASK_ID}} — task name (directory under tasks/)
15
15
  * {{TASK_DIR}} — agent working directory path
@@ -40,22 +40,25 @@ import { sumTraceCost } from "../cost.js";
40
40
  */
41
41
 
42
42
  /**
43
- * Run the judge over a completed task run.
43
+ * Run the judge over a completed task run. The judge is a binary gate over
44
+ * the grade's validity, never a grade itself: `gradeResult` reaches the
45
+ * template as evidence, and the verdict stays pass/fail.
44
46
  * @param {import("./task-family.js").Task} task
45
47
  * @param {import("./workdir.js").Workdir} workdir
46
- * @param {import("./invariants.js").InvariantsResult} invariants
48
+ * @param {{verdict: string, gatesPass: boolean, score?: number, malformed?: number, rows: unknown[]}} gradeResult -
49
+ * The normalized grade plus the merged, source-stamped check rows.
47
50
  * @param {{query: Function, model: string, judgeProfile?: string, profilesDir?: string, runtime: import("@forwardimpact/libutil/runtime").Runtime}} deps
48
51
  * @param {JudgeContext} [context]
49
52
  * @returns {Promise<JudgeVerdict>}
50
53
  */
51
- export async function runJudge(task, workdir, invariants, deps, context) {
54
+ export async function runJudge(task, workdir, gradeResult, deps, context) {
52
55
  const runtime = deps.runtime;
53
56
  if (!runtime) throw new Error("runtime is required");
54
57
  const fs = runtime.fs;
55
58
  const template = await fs.readFile(task.paths.judge, "utf8");
56
- const invariantsJson = JSON.stringify(invariants, null, 2);
59
+ const gradeJson = JSON.stringify(gradeResult, null, 2);
57
60
  const taskText = template
58
- .replaceAll("{{INVARIANTS_RESULT}}", invariantsJson)
61
+ .replaceAll("{{GRADE_RESULT}}", gradeJson)
59
62
  .replaceAll("{{AGENT_TRACE_PATH}}", workdir.agentTracePath)
60
63
  .replaceAll("{{AGENT_INSTRUCTIONS}}", context?.agentInstructions ?? "")
61
64
  .replaceAll("{{AGENT_PROFILE}}", context?.agentProfile ?? "")