@forwardimpact/libharness 1.4.0 → 2.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -4,8 +4,10 @@
4
4
  * Phases per (task, runIndex):
5
5
  * 1. WorkdirManager.start → seed CWD + run pre-flight probe
6
6
  * 2. Supervisor session (agent + supervisor) → produce traces + submission
7
- * 3. Invariants.runInvariants → exit-code-driven verdict via fd-3 NDJSON
8
- * 4. Judge.runJudge → Conclude-driven verdict mapped to pass/fail
7
+ * 3. Invariants collector + hidden-test engine → merged check rows,
8
+ * graded by `gradeChecks` (rows are authoritative; script exit is
9
+ * grader health only)
10
+ * 4. Judge.runJudge → Conclude-driven binary gate mapped to pass/fail
9
11
  * 5. WorkdirManager.teardown → process-group cleanup
10
12
  *
11
13
  * Cells run with bounded in-process concurrency (`CellScheduler`); `run()`
@@ -17,7 +19,6 @@
17
19
  * stream.
18
20
  */
19
21
 
20
- import { createInterface } from "node:readline";
21
22
  import { join, resolve as resolvePath } from "node:path";
22
23
 
23
24
  import { DEFAULT_ENV_ALLOWLIST, createRedactor } from "../redaction.js";
@@ -28,7 +29,10 @@ import { installNpm as defaultInstallNpm } from "./npm-installer.js";
28
29
  import { runJudge } from "./judge.js";
29
30
  import { validateResultRecord } from "./result.js";
30
31
  import { runInvariants } from "./invariants.js";
32
+ import { runHiddenTests } from "./hidden-tests.js";
33
+ import { runProducersAndGrade } from "./grade.js";
31
34
  import { assertJudgeProfileStaged, loadTaskFamily } from "./task-family.js";
35
+ import { splitAndSummarize } from "./trace-split.js";
32
36
  import { createWorkdirManager } from "./workdir.js";
33
37
  import { CellScheduler } from "./scheduler.js";
34
38
 
@@ -82,9 +86,13 @@ export class BenchmarkRunner {
82
86
  * threaded into the installers, workdir manager, invariants, and judge.
83
87
  * @param {Function} [opts.runInvariants] - Test seam: replaces `runInvariants`.
84
88
  * Same contract as `runInvariants(task, ctx, runtime)`. Internal testing only.
89
+ * @param {Function} [opts.runHiddenTests] - Test seam: replaces
90
+ * `runHiddenTests`. Same contract as `runHiddenTests(task, ctx, runtime)`.
91
+ * Internal testing only.
85
92
  * @param {Function} [opts.runJudge] - Test seam: replaces `runJudge`. Same
86
- * contract as `runJudge(task, workdir, invariants, deps)` (deps carries
87
- * `runtime`). Internal testing only.
93
+ * contract as `runJudge(task, workdir, gradeResult, deps)` where
94
+ * `gradeResult` is the normalized grade plus the merged, source-stamped
95
+ * check `rows` (deps carries `runtime`). Internal testing only.
88
96
  * @param {Function} [opts.installApm] - Test seam: replaces `installApm`.
89
97
  * Same contract as `installApm(family, outputDir, runtime)`. Lets tests
90
98
  * inject a fake subprocess (or skip the install entirely) so the suite
@@ -114,6 +122,7 @@ export class BenchmarkRunner {
114
122
  // Test seams — default to the real implementations.
115
123
  runAgent,
116
124
  runInvariants: runInvariantsHook,
125
+ runHiddenTests: runHiddenTestsHook,
117
126
  runJudge: runJudgeHook,
118
127
  installApm: installApmHook,
119
128
  installNpm: installNpmHook,
@@ -141,6 +150,7 @@ export class BenchmarkRunner {
141
150
  this.termGraceMs = termGraceMs;
142
151
  this._runAgentHook = runAgent ?? null;
143
152
  this._runInvariantsHook = runInvariantsHook ?? runInvariants;
153
+ this._runHiddenTestsHook = runHiddenTestsHook ?? runHiddenTests;
144
154
  this._runJudgeHook = runJudgeHook ?? runJudge;
145
155
  this._installApmHook = installApmHook ?? defaultInstallApm;
146
156
  this._installNpmHook = installNpmHook ?? defaultInstallNpm;
@@ -291,55 +301,37 @@ export class BenchmarkRunner {
291
301
  const agentRun = await this.#runAgentSafe(task, workdir);
292
302
  const { costUsd, turns, submission, agentError } = agentRun;
293
303
  const breakdown = agentRun.costBreakdown ?? { agent: 0, supervisor: 0 };
294
- const invariants = await this._runInvariantsHook(
304
+ const graded = await this.#gradeCell(family, task, workdir);
305
+ const { invariants, hiddenRows, engineError, rows, grade } = graded;
306
+ const { judgeVerdict, judgeCost } = await this.#judgeCell({
295
307
  task,
296
- {
297
- cwd: workdir.cwd,
298
- port: workdir.port,
299
- runDir: workdir.runDir,
300
- familyDir: family.rootPath,
301
- },
302
- this.runtime,
303
- );
304
- let judgeVerdict = null;
305
- let judgeCost = 0;
306
- if (task.paths.judge) {
307
- const judgeContext = await this.#buildJudgeContext(
308
- task,
309
- workdir,
310
- skillSetHash,
311
- );
312
- const judgeResult = await this._runJudgeHook(
313
- task,
314
- workdir,
315
- invariants,
316
- {
317
- query: this.query,
318
- model: this.judgeModel,
319
- judgeProfile: this.profiles.judge ?? undefined,
320
- profilesDir: judgeProfilesDir,
321
- runtime: this.runtime,
322
- },
323
- judgeContext,
324
- );
325
- judgeCost = judgeResult.costUsd ?? 0;
326
- // The record's judgeVerdict carries only the verdict + summary; the
327
- // judge's cost is folded into costUsd / costBreakdown instead.
328
- judgeVerdict = {
329
- verdict: judgeResult.verdict,
330
- summary: judgeResult.summary,
331
- };
332
- }
333
- const verdict =
334
- invariants.verdict === "pass" &&
335
- (judgeVerdict === null || judgeVerdict.verdict === "pass")
336
- ? "pass"
337
- : "fail";
308
+ workdir,
309
+ gradeResult: { ...grade, rows },
310
+ skillSetHash,
311
+ judgeProfilesDir,
312
+ });
313
+ const judgePass =
314
+ judgeVerdict === null || judgeVerdict.verdict === "pass";
315
+ const verdict = grade.verdict === "pass" && judgePass ? "pass" : "fail";
316
+ // Gates protect the score: an unhealthy grader, a failing gate row, or
317
+ // a failing judge zeroes the effective score. Full marks does not — a
318
+ // fractional score with verdict fail is the point.
319
+ const scoreValid = graded.healthy && grade.gatesPass && judgePass;
338
320
  const record = {
339
321
  taskId: task.id,
340
322
  runIndex,
341
323
  verdict,
342
324
  invariants,
325
+ grade,
326
+ ...(task.tests && {
327
+ hiddenTests: {
328
+ details: hiddenRows,
329
+ ...(engineError && { error: engineError.message }),
330
+ },
331
+ }),
332
+ ...(grade.score !== undefined && {
333
+ score: scoreValid ? grade.score : 0,
334
+ }),
343
335
  submission,
344
336
  ...(judgeVerdict && { judgeVerdict }),
345
337
  costUsd: costUsd + judgeCost,
@@ -371,6 +363,65 @@ export class BenchmarkRunner {
371
363
  }
372
364
  }
373
365
 
366
+ /**
367
+ * Run the judge (when the task ships a template) over the grade result.
368
+ * The record's judgeVerdict carries only the verdict + summary; the
369
+ * judge's cost is folded into costUsd / costBreakdown instead.
370
+ */
371
+ async #judgeCell({
372
+ task,
373
+ workdir,
374
+ gradeResult,
375
+ skillSetHash,
376
+ judgeProfilesDir,
377
+ }) {
378
+ if (!task.paths.judge) return { judgeVerdict: null, judgeCost: 0 };
379
+ const judgeContext = await this.#buildJudgeContext(
380
+ task,
381
+ workdir,
382
+ skillSetHash,
383
+ );
384
+ const judgeResult = await this._runJudgeHook(
385
+ task,
386
+ workdir,
387
+ gradeResult,
388
+ {
389
+ query: this.query,
390
+ model: this.judgeModel,
391
+ judgeProfile: this.profiles.judge ?? undefined,
392
+ profilesDir: judgeProfilesDir,
393
+ runtime: this.runtime,
394
+ },
395
+ judgeContext,
396
+ );
397
+ return {
398
+ judgeVerdict: {
399
+ verdict: judgeResult.verdict,
400
+ summary: judgeResult.summary,
401
+ },
402
+ judgeCost: judgeResult.costUsd ?? 0,
403
+ };
404
+ }
405
+
406
+ /**
407
+ * Run both check-row producers against the post-run CWD and grade the
408
+ * merged rows via the shared derivation. Restoration happens inside the
409
+ * engine, so the judge (which runs after) sees the workdir exactly as the
410
+ * agent left it.
411
+ */
412
+ #gradeCell(family, task, workdir) {
413
+ const ctx = {
414
+ cwd: workdir.cwd,
415
+ port: workdir.port,
416
+ runDir: workdir.runDir,
417
+ familyDir: family.rootPath,
418
+ };
419
+ return runProducersAndGrade(task, ctx, this.runtime, {
420
+ runInvariants: this._runInvariantsHook,
421
+ runHiddenTests: this._runHiddenTestsHook,
422
+ });
423
+ }
424
+
374
425
  /**
375
426
  * Dispatch to either the injected hook or the default `#runAgent`. Either
376
427
  * path can throw; catch here so a thrown error becomes an `agentError` on
@@ -614,67 +665,6 @@ async function writeRecord(stream, record) {
614
665
  });
615
666
  }
616
667
 
617
- /**
618
- * Split the combined supervisor trace into agent and supervisor files and
619
- * extract turn count and submission in a single pass. Agent-source events go
620
- * to `agentPath`; supervisor and orchestrator events go to `supervisorPath`.
621
- *
622
- * Cost is deliberately not summed here — the caller derives it from the same
623
- * combined trace via `sumTraceCost`, so there is one cost path across the
624
- * benchmark, callback, and `fit-trace cost` consumers.
625
- */
626
- // biome-ignore lint/complexity/noExcessiveCognitiveComplexity: stream-splitting state machine
627
- async function splitAndSummarize(
628
- runtime,
629
- combinedPath,
630
- agentPath,
631
- supervisorPath,
632
- ) {
633
- const fs = runtime.fs;
634
- const agentStream = fs.createWriteStream(agentPath);
635
- const supStream = fs.createWriteStream(supervisorPath);
636
- const rl = createInterface({
637
- input: fs.createReadStream(combinedPath),
638
- crlfDelay: Infinity,
639
- });
640
- let turns = 0;
641
- let submission = "";
642
- for await (const line of rl) {
643
- if (!line.trim()) continue;
644
- let event;
645
- try {
646
- event = JSON.parse(line);
647
- } catch {
648
- continue;
649
- }
650
- const target = event.source === "agent" ? agentStream : supStream;
651
- target.write(line + "\n");
652
- const inner = event.event;
653
- if (!inner) continue;
654
- if (event.source === "agent" && inner.type === "assistant") {
655
- const text = extractText(inner);
656
- if (text) submission = text;
657
- }
658
- if (event.source === "orchestrator" && inner.type === "summary") {
659
- turns = inner.turns ?? 0;
660
- }
661
- }
662
- await Promise.all([
663
- new Promise((r) => agentStream.end(r)),
664
- new Promise((r) => supStream.end(r)),
665
- ]);
666
- return { turns, submission };
667
- }
668
-
669
- function extractText(inner) {
670
- const content = inner.message?.content ?? inner.content;
671
- if (!Array.isArray(content)) return null;
672
- for (let i = content.length - 1; i >= 0; i--) {
673
- if (content[i].type === "text" && content[i].text) return content[i].text;
674
- }
675
- return null;
676
- }
677
-
678
668
  /**
679
669
  * Factory function — wires real dependencies.
680
670
  * @param {ConstructorParameters<typeof BenchmarkRunner>[0]} opts
@@ -10,9 +10,16 @@
10
10
  * hooks/ # harness-only; never copied to agent CWD
11
11
  * preflight.sh
12
12
  * invariants.sh
13
+ * tests/ # optional hidden test suite; harness-only overlay
13
14
  * specs/ # copied into agent CWD
14
15
  * workdir/ # copied into agent CWD
15
16
  *
17
+ * `tests/` is an overlay mirror of the agent CWD: a file's path under
18
+ * `tests/` is its staging path. Every `*.test.js` file is one check —
19
+ * `*.gate.test.js` marks a gate, any other `*.test.js` is scored — and every
20
+ * other file is support material, staged but never graded. The layout is
21
+ * validated eagerly here so authoring errors fail before any agent spend.
22
+ *
16
23
  * Local paths or git URLs are both accepted; git URLs are shallow-cloned into
17
24
  * a temp dir and `familyRevision` becomes `git:<sha>` of HEAD at clone time.
18
25
  * Local paths use the canonical-tree algorithm from design § Family revision
@@ -113,36 +120,124 @@ async function discoverTasks(runtime, rootPath) {
113
120
  }
114
121
  for (const entry of entries) {
115
122
  if (!entry.isDirectory()) continue;
116
- const taskDir = join(tasksRoot, entry.name);
117
- const supervisorPath = join(taskDir, "supervisor.task.md");
118
- const judgePath = join(taskDir, "judge.task.md");
119
- const preflightPath = join(taskDir, "hooks", "preflight.sh");
120
- const invariantsPath = join(taskDir, "hooks", "invariants.sh");
121
- tasks.push({
122
- id: entry.name,
123
- paths: {
124
- taskDir,
125
- instructions: join(taskDir, "agent.task.md"),
126
- supervisor: (await fileExists(fs, supervisorPath))
127
- ? supervisorPath
128
- : null,
129
- judge: (await fileExists(fs, judgePath)) ? judgePath : null,
130
- hooks: join(taskDir, "hooks"),
131
- preflight: (await fileExecutable(fs, preflightPath))
132
- ? preflightPath
133
- : null,
134
- invariants: (await fileExecutable(fs, invariantsPath))
135
- ? invariantsPath
136
- : null,
137
- specs: join(taskDir, "specs"),
138
- workdir: join(taskDir, "workdir"),
139
- },
140
- });
123
+ tasks.push(await loadTask(fs, join(tasksRoot, entry.name), entry.name));
141
124
  }
142
125
  tasks.sort((a, b) => (a.id < b.id ? -1 : a.id > b.id ? 1 : 0));
143
126
  return tasks;
144
127
  }
145
128
 
129
+ async function loadTask(fs, taskDir, id) {
130
+ const supervisorPath = join(taskDir, "supervisor.task.md");
131
+ const judgePath = join(taskDir, "judge.task.md");
132
+ const preflightPath = join(taskDir, "hooks", "preflight.sh");
133
+ const invariantsPath = join(taskDir, "hooks", "invariants.sh");
134
+ const suite = await discoverSuite(fs, taskDir);
135
+ return {
136
+ id,
137
+ paths: {
138
+ taskDir,
139
+ instructions: join(taskDir, "agent.task.md"),
140
+ supervisor: (await fileExists(fs, supervisorPath))
141
+ ? supervisorPath
142
+ : null,
143
+ judge: (await fileExists(fs, judgePath)) ? judgePath : null,
144
+ hooks: join(taskDir, "hooks"),
145
+ preflight: (await fileExecutable(fs, preflightPath))
146
+ ? preflightPath
147
+ : null,
148
+ invariants: (await fileExecutable(fs, invariantsPath))
149
+ ? invariantsPath
150
+ : null,
151
+ tests: suite ? join(taskDir, "tests") : null,
152
+ specs: join(taskDir, "specs"),
153
+ workdir: join(taskDir, "workdir"),
154
+ },
155
+ tests: suite,
156
+ };
157
+ }
158
+
159
+ const CHECK_SUFFIX = ".test.js";
160
+ const GATE_SUFFIX = ".gate.test.js";
161
+
162
+ /**
163
+ * Discover and validate a task's hidden test suite under `<taskDir>/tests/`.
164
+ * Returns null when the directory is absent; throws on an invalid layout
165
+ * (no check files, a dangling symlink, duplicate check names) so
166
+ * `loadTaskFamily` rejects broken suites before any agent spend.
167
+ * @param {object} fs - Async filesystem surface (`runtime.fs`).
168
+ * @param {string} taskDir
169
+ * @returns {Promise<HiddenSuite | null>}
170
+ */
171
+ async function discoverSuite(fs, taskDir) {
172
+ const testsRoot = join(taskDir, "tests");
173
+ try {
174
+ const st = await fs.lstat(testsRoot);
175
+ if (!st.isDirectory()) return null;
176
+ } catch {
177
+ return null;
178
+ }
179
+ const files = [];
180
+ await walkSuiteFiles(fs, testsRoot, testsRoot, files);
181
+ files.sort((a, b) =>
182
+ a.stagePath < b.stagePath ? -1 : a.stagePath > b.stagePath ? 1 : 0,
183
+ );
184
+
185
+ const checks = [];
186
+ const support = [];
187
+ for (const file of files) {
188
+ const base = file.stagePath.split(sep).at(-1);
189
+ if (!base.endsWith(CHECK_SUFFIX)) {
190
+ support.push(file);
191
+ continue;
192
+ }
193
+ const gate = base.endsWith(GATE_SUFFIX);
194
+ const name = base.slice(0, -(gate ? GATE_SUFFIX : CHECK_SUFFIX).length);
195
+ checks.push({ name, gate, ...file });
196
+ }
197
+
198
+ if (checks.length === 0) {
199
+ throw new Error(`hidden test suite has no check files: ${testsRoot}`);
200
+ }
201
+ const seen = new Set();
202
+ for (const check of checks) {
203
+ if (seen.has(check.name)) {
204
+ throw new Error(
205
+ `hidden test suite has duplicate check name '${check.name}': ${check.sourcePath}`,
206
+ );
207
+ }
208
+ seen.add(check.name);
209
+ }
210
+ return { checks, support };
211
+ }
212
+
213
+ /**
214
+ * Walk a suite tree collecting `{sourcePath, stagePath}` entries. Unlike
215
+ * `walkFiles` (which silently skips dangling symlinks for hashing), every
216
+ * entry here must be a regular file after symlink resolution — a dangling
217
+ * symlink or a link to a non-file target is an authoring error.
218
+ */
219
+ async function walkSuiteFiles(fs, root, dir, out) {
220
+ const entries = await fs.readdir(dir, { withFileTypes: true });
221
+ for (const entry of entries) {
222
+ const full = join(dir, entry.name);
223
+ if (entry.isDirectory()) {
224
+ await walkSuiteFiles(fs, root, full, out);
225
+ } else if (entry.isFile()) {
226
+ out.push({ sourcePath: full, stagePath: relative(root, full) });
227
+ } else if (entry.isSymbolicLink()) {
228
+ const resolved = await resolveSymlinkToFile(fs, full);
229
+ if (!resolved) {
230
+ throw new Error(
231
+ `hidden test suite entry is not a regular file: ${full}`,
232
+ );
233
+ }
234
+ out.push({ sourcePath: full, stagePath: relative(root, full) });
235
+ } else {
236
+ throw new Error(`hidden test suite entry is not a regular file: ${full}`);
237
+ }
238
+ }
239
+ }
240
+
146
241
  async function fileExists(fs, path) {
147
242
  try {
148
243
  await fs.access(path);
@@ -246,10 +341,28 @@ async function git(runtime, args) {
246
341
  return stdout;
247
342
  }
248
343
 
344
+ /**
345
+ * @typedef {object} HiddenCheck
346
+ * @property {string} name - Basename stem with the check suffix stripped.
347
+ * @property {boolean} gate - True iff the filename ends `.gate.test.js`.
348
+ * @property {string} sourcePath - Absolute path under `tests/`.
349
+ * @property {string} stagePath - Path relative to `tests/` — the overlay
350
+ * mirror of the staging path under the agent CWD.
351
+ */
352
+
353
+ /**
354
+ * @typedef {object} HiddenSuite
355
+ * @property {HiddenCheck[]} checks - In sorted stage-path order.
356
+ * @property {{sourcePath: string, stagePath: string}[]} support - Non-check
357
+ * files, staged for the whole pass but never graded.
358
+ */
359
+
249
360
  /**
250
361
  * @typedef {object} Task
251
362
  * @property {string} id - Task name (directory name under tasks/)
252
- * @property {{taskDir: string, instructions: string, supervisor: string|null, judge: string|null, hooks: string, preflight: string|null, invariants: string|null, specs: string, workdir: string}} paths
363
+ * @property {{taskDir: string, instructions: string, supervisor: string|null, judge: string|null, hooks: string, preflight: string|null, invariants: string|null, tests: string|null, specs: string, workdir: string}} paths
364
+ * @property {HiddenSuite | null} tests - Hidden test suite (null when the
365
+ * task ships no `tests/` directory)
253
366
  */
254
367
 
255
368
  /**
@@ -0,0 +1,73 @@
1
+ /**
2
+ * Combined-supervisor-trace splitting for the benchmark runner: one pass
3
+ * over the tagged NDJSON envelope stream separates agent events from
4
+ * supervisor/orchestrator events and extracts the run summary.
5
+ */
6
+
7
+ import { createInterface } from "node:readline";
8
+
9
+ /**
10
+ * Split the combined supervisor trace into agent and supervisor files and
11
+ * extract turn count and submission in a single pass. Agent-source events go
12
+ * to `agentPath`; supervisor and orchestrator events go to `supervisorPath`.
13
+ *
14
+ * Cost is deliberately not summed here — the caller derives it from the same
15
+ * combined trace via `sumTraceCost`, so there is one cost path across the
16
+ * benchmark, callback, and `fit-trace cost` consumers.
17
+ * @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
18
+ * @param {string} combinedPath
19
+ * @param {string} agentPath
20
+ * @param {string} supervisorPath
21
+ * @returns {Promise<{turns: number, submission: string}>}
22
+ */
23
+ // biome-ignore lint/complexity/noExcessiveCognitiveComplexity: stream-splitting state machine
24
+ export async function splitAndSummarize(
25
+ runtime,
26
+ combinedPath,
27
+ agentPath,
28
+ supervisorPath,
29
+ ) {
30
+ const fs = runtime.fs;
31
+ const agentStream = fs.createWriteStream(agentPath);
32
+ const supStream = fs.createWriteStream(supervisorPath);
33
+ const rl = createInterface({
34
+ input: fs.createReadStream(combinedPath),
35
+ crlfDelay: Infinity,
36
+ });
37
+ let turns = 0;
38
+ let submission = "";
39
+ for await (const line of rl) {
40
+ if (!line.trim()) continue;
41
+ let event;
42
+ try {
43
+ event = JSON.parse(line);
44
+ } catch {
45
+ continue;
46
+ }
47
+ const target = event.source === "agent" ? agentStream : supStream;
48
+ target.write(line + "\n");
49
+ const inner = event.event;
50
+ if (!inner) continue;
51
+ if (event.source === "agent" && inner.type === "assistant") {
52
+ const text = extractText(inner);
53
+ if (text) submission = text;
54
+ }
55
+ if (event.source === "orchestrator" && inner.type === "summary") {
56
+ turns = inner.turns ?? 0;
57
+ }
58
+ }
59
+ await Promise.all([
60
+ new Promise((r) => agentStream.end(r)),
61
+ new Promise((r) => supStream.end(r)),
62
+ ]);
63
+ return { turns, submission };
64
+ }
65
+
66
+ function extractText(inner) {
67
+ const content = inner.message?.content ?? inner.content;
68
+ if (!Array.isArray(content)) return null;
69
+ for (let i = content.length - 1; i >= 0; i--) {
70
+ if (content[i].type === "text" && content[i].text) return content[i].text;
71
+ }
72
+ return null;
73
+ }
@@ -268,7 +268,12 @@ async function runPreflight(runtime, script, cwd, port, vars) {
268
268
  };
269
269
  }
270
270
 
271
- function probeFreePort() {
271
+ /**
272
+ * Allocate a free TCP port by binding to 0 and releasing it. Shared with the
273
+ * `grade` subcommand, which needs a plausible `$PORT` for the hook env.
274
+ * @returns {Promise<number>}
275
+ */
276
+ export function probeFreePort() {
272
277
  return new Promise((res, rej) => {
273
278
  const server = createServer();
274
279
  server.unref();
@@ -62,12 +62,74 @@ export function evaluateAssertion(values, args, fsSync) {
62
62
 
63
63
  const output = { test: testName, pass: result.pass };
64
64
  if (result.message) output.message = result.message;
65
+ applyGradingFlags(values, output);
65
66
  return output;
66
67
  }
67
68
 
69
+ /**
70
+ * Attach the check-row grading role: `--gate` marks a gate check, `--weight`
71
+ * attaches a numeric weight (0 marks the row diagnostic). `--gate` with any
72
+ * `--weight` — 0 included — is invalid: a stray weight must never silently
73
+ * disarm a gate.
74
+ * @param {object} values
75
+ * @param {{test: string, pass: boolean, message?: string}} output - Mutated.
76
+ */
77
+ function applyGradingFlags(values, output) {
78
+ const hasWeight = values.weight !== undefined;
79
+ if (values.gate && hasWeight) {
80
+ throw new Error("assert: --gate cannot be combined with --weight");
81
+ }
82
+ if (values.gate) output.gate = true;
83
+ if (hasWeight) {
84
+ const weight = parseWeight(values.weight);
85
+ if (weight === null) {
86
+ throw new Error(
87
+ `assert: invalid --weight '${values.weight}' (expected a finite number ≥ 0)`,
88
+ );
89
+ }
90
+ output.weight = weight;
91
+ }
92
+ }
93
+
94
+ /**
95
+ * Parse a `--weight` value; null when invalid. A blank string is invalid —
96
+ * `Number("")` is 0, which would silently demote the check to a diagnostic.
97
+ * @param {string} raw
98
+ * @returns {number | null}
99
+ */
100
+ function parseWeight(raw) {
101
+ if (typeof raw === "string" && raw.trim() === "") return null;
102
+ const weight = Number(raw);
103
+ return Number.isFinite(weight) && weight >= 0 ? weight : null;
104
+ }
105
+
106
+ /**
107
+ * The grading role an emit-then-fail row keeps: a failing check must not
108
+ * lose its authored role — an errored gate that demoted to a scored row
109
+ * would let a broken scaffold earn partial credit instead of zeroing the
110
+ * score. Invalid or conflicting flags yield no role (the row fails as a
111
+ * unit-weight scored check).
112
+ * @param {object} values
113
+ * @returns {{gate?: true, weight?: number}}
114
+ */
115
+ function errorRowRole(values) {
116
+ const hasWeight = values.weight !== undefined;
117
+ if (values.gate && !hasWeight) return { gate: true };
118
+ if (!values.gate && hasWeight) {
119
+ const weight = parseWeight(values.weight);
120
+ if (weight !== null) return { weight };
121
+ }
122
+ return {};
123
+ }
124
+
68
125
  /**
69
126
  * Run an assertion, write JSON to stdout, and return a failure envelope when
70
127
  * the assertion does not pass.
128
+ *
129
+ * Emit-then-fail on every failure path: an invalid grading flag or an
130
+ * errored evaluation (e.g. `--grep` against a file the agent deleted) writes
131
+ * a failing row before the nonzero exit, so a typo or a vanished target
132
+ * shrinks the score, never the denominator.
71
133
  * @param {import("@forwardimpact/libcli").InvocationContext} ctx
72
134
  * @returns {Promise<{ok: true} | {ok: false, code: number, error: string}>}
73
135
  */
@@ -78,6 +140,16 @@ export async function runAssertCommand(ctx) {
78
140
  try {
79
141
  result = evaluateAssertion(ctx.options, args, runtime.fsSync);
80
142
  } catch (err) {
143
+ const reason = err.message.startsWith("assert: ")
144
+ ? err.message
145
+ : `assert: ${err.message}`;
146
+ const row = {
147
+ test: ctx.args["test-name"] ?? "(missing test name)",
148
+ pass: false,
149
+ ...errorRowRole(ctx.options),
150
+ message: reason,
151
+ };
152
+ runtime.proc.stdout.write(JSON.stringify(row) + "\n");
81
153
  return { ok: false, code: 1, error: err.message };
82
154
  }
83
155
  runtime.proc.stdout.write(JSON.stringify(result) + "\n");
@@ -5,7 +5,7 @@
5
5
  */
6
6
 
7
7
  import { runBenchmarkRunCommand } from "./benchmark-run.js";
8
- import { runBenchmarkInvariantsCommand } from "./benchmark-invariants.js";
8
+ import { runBenchmarkGradeCommand } from "./benchmark-grade.js";
9
9
  import { runBenchmarkReportCommand } from "./benchmark-report.js";
10
10
  import {
11
11
  BENCHMARK_AGENT_MODEL,
@@ -95,11 +95,11 @@ export const definition = {
95
95
  },
96
96
  },
97
97
  {
98
- name: "invariants",
98
+ name: "grade",
99
99
  args: [],
100
- handler: runBenchmarkInvariantsCommand,
100
+ handler: runBenchmarkGradeCommand,
101
101
  description:
102
- "Check a single task's invariants against a post-run workdir without invoking an agent.",
102
+ "Grade a single task against a post-run workdir without invoking an agent: run the hidden test suite and the invariants script, then derive the verdict from the check rows (the exit mirrors it).",
103
103
  options: {
104
104
  family: {
105
105
  type: "string",
@@ -112,7 +112,7 @@ export const definition = {
112
112
  "run-dir": {
113
113
  type: "string",
114
114
  description:
115
- "Post-run directory whose cwd/ subdir is the agent CWD; invariants run against that cwd — the path hooks receive as $AGENT_CWD",
115
+ "Post-run directory whose cwd/ subdir is the agent CWD; both producers run against that cwd — the path hooks receive as $AGENT_CWD",
116
116
  },
117
117
  output: {
118
118
  type: "string",
@@ -159,7 +159,7 @@ export const definition = {
159
159
  "fit-benchmark run --family=./families/coding --work-tracker=filesystem",
160
160
  "fit-benchmark run --family=./families/coding --skills-from=. --task=todo-api",
161
161
  `fit-benchmark run --family=./families/coding --runs=10 --agent-model=${BENCHMARK_AGENT_MODEL}`,
162
- "fit-benchmark invariants --family=./families/coding --task=todo-api --run-dir=./benchmark-runs/runs/todo-api/0",
162
+ "fit-benchmark grade --family=./families/coding --task=todo-api --run-dir=./benchmark-runs/runs/todo-api/0",
163
163
  "fit-benchmark report --format=text",
164
164
  "fit-benchmark report --input=./runs/today --k=1,3,5 --format=text",
165
165
  ],