@forwardimpact/libharness 3.0.0 → 3.0.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (69) hide show
  1. package/README.md +60 -57
  2. package/package.json +2 -2
  3. package/src/advisor.js +47 -41
  4. package/src/agent-runner.js +57 -47
  5. package/src/benchmark/apm-installer.js +28 -28
  6. package/src/benchmark/env-loader.js +24 -16
  7. package/src/benchmark/grade.js +44 -41
  8. package/src/benchmark/hidden-tests.js +71 -39
  9. package/src/benchmark/hook-env.js +11 -9
  10. package/src/benchmark/invariants.js +20 -17
  11. package/src/benchmark/judge.js +29 -28
  12. package/src/benchmark/npm-installer.js +9 -8
  13. package/src/benchmark/report.js +53 -50
  14. package/src/benchmark/result.js +24 -23
  15. package/src/benchmark/runner.js +75 -69
  16. package/src/benchmark/scheduler.js +17 -16
  17. package/src/benchmark/task-family.js +28 -26
  18. package/src/benchmark/trace-split.js +8 -7
  19. package/src/benchmark/workdir.js +27 -25
  20. package/src/claude-code-executable.js +11 -11
  21. package/src/commands/advisor-flags.js +8 -7
  22. package/src/commands/assert.js +16 -15
  23. package/src/commands/benchmark-definition.js +11 -11
  24. package/src/commands/benchmark-grade.js +12 -11
  25. package/src/commands/benchmark-report.js +5 -5
  26. package/src/commands/benchmark-run.js +31 -28
  27. package/src/commands/by-discussion.js +10 -10
  28. package/src/commands/callback.js +11 -11
  29. package/src/commands/discuss.js +8 -7
  30. package/src/commands/facilitate.js +15 -13
  31. package/src/commands/output.js +3 -2
  32. package/src/commands/run.js +14 -14
  33. package/src/commands/scan-logs.js +21 -19
  34. package/src/commands/selfedit.js +14 -14
  35. package/src/commands/supervise.js +11 -9
  36. package/src/commands/task-input.js +9 -9
  37. package/src/commands/tee.js +10 -9
  38. package/src/commands/trace.js +55 -42
  39. package/src/commands/work-tracker.js +4 -3
  40. package/src/cost.js +17 -17
  41. package/src/discuss-tools.js +16 -16
  42. package/src/discusser.js +39 -38
  43. package/src/events/github.js +54 -37
  44. package/src/facilitator.js +21 -21
  45. package/src/inbox-poller.js +4 -4
  46. package/src/judge.js +32 -30
  47. package/src/message-bus.js +12 -11
  48. package/src/orchestration-loop.js +35 -36
  49. package/src/orchestration-toolkit.js +58 -53
  50. package/src/orchestrator-helpers.js +2 -2
  51. package/src/profile-prompt.js +54 -53
  52. package/src/redaction.js +63 -57
  53. package/src/render/line-renderer.js +5 -5
  54. package/src/render/orchestrator-filter.js +3 -3
  55. package/src/render/palette.js +11 -9
  56. package/src/render/tool-hints.js +18 -15
  57. package/src/render/turn-renderer.js +4 -4
  58. package/src/reply-emitter.js +2 -2
  59. package/src/sequence-counter.js +4 -3
  60. package/src/signature-filter.js +7 -6
  61. package/src/supervisor.js +19 -18
  62. package/src/tee-writer.js +25 -25
  63. package/src/trace-collector.js +53 -48
  64. package/src/trace-github.js +53 -44
  65. package/src/trace-multi.js +15 -13
  66. package/src/trace-query.js +61 -52
  67. package/src/trace-render.js +18 -18
  68. package/src/trace-usage.js +31 -28
  69. package/src/transcript-recorder.js +24 -20
@@ -1,13 +1,13 @@
1
1
  /**
2
2
  * ApmInstaller — runs `apm install --target claude` in the family root to
3
- * materialise skills and agents, copies the resulting `.claude/` into a
4
- * staging directory, and computes the manifest fingerprint from the lockfile.
5
- * Per-task copy happens later in WorkdirManager.
3
+ * materialise skills and agents. It copies the resulting `.claude/` into a
4
+ * staging directory. It computes the manifest fingerprint from the lockfile.
5
+ * WorkdirManager makes the per-task copy later.
6
6
  *
7
- * Subprocess and filesystem access route through the injected `runtime` bag
8
- * (`runtime.subprocess.spawn` for the streaming `apm` child, `runtime.fs` for
9
- * the async staging copies). See `createApmInstaller` for the real-dependency
10
- * wiring; `installApm` is a thin free-function wrapper.
7
+ * Subprocess and filesystem access route through the injected `runtime` bag.
8
+ * The `apm` child streams through `runtime.subprocess.spawn`. The async
9
+ * staging copies use `runtime.fs`. See `createApmInstaller`, which wires the
10
+ * real dependencies. `installApm` is a thin free-function wrapper.
11
11
  */
12
12
 
13
13
  import { createHash } from "node:crypto";
@@ -18,7 +18,7 @@ export class ApmInstaller {
18
18
  /**
19
19
  * @param {object} deps
20
20
  * @param {import("@forwardimpact/libutil/runtime").Runtime} deps.runtime -
21
- * Ambient collaborators; uses `subprocess.spawn` and `fs`.
21
+ * Ambient collaborators. The installer uses `subprocess.spawn` and `fs`.
22
22
  */
23
23
  constructor({ runtime }) {
24
24
  if (!runtime) throw new Error("runtime is required");
@@ -30,8 +30,8 @@ export class ApmInstaller {
30
30
  * @param {string} outputDir - The benchmark run's output directory.
31
31
  * @param {object} [options]
32
32
  * @param {string|null} [options.skillsFrom] - Stage `.claude/` from this
33
- * directory instead of running apm install. The path is a root containing
34
- * a `.claude/` tree (e.g. a working tree), letting a run exercise local,
33
+ * directory and do not run apm install. The path is a root that contains
34
+ * a `.claude/` tree (e.g. a working tree). A run can then exercise local,
35
35
  * unpublished skills.
36
36
  * @returns {Promise<{stagingDir: string, skillSetHash: string, judgeProfilesDir: string}>}
37
37
  */
@@ -44,7 +44,7 @@ export class ApmInstaller {
44
44
  : join(family.rootPath, ".claude");
45
45
  const apmYml = join(family.rootPath, "apm.yml");
46
46
 
47
- // --skills-from takes precedence over apm install: the caller is supplying
47
+ // --skills-from takes precedence over apm install. The caller supplies
48
48
  // the skill tree explicitly, so no remote fetch runs.
49
49
  const hasApm =
50
50
  !skillsFrom &&
@@ -59,7 +59,7 @@ export class ApmInstaller {
59
59
  await fs.access(sourceClaude);
60
60
  } catch {
61
61
  throw new Error(
62
- `apm install did not produce .claude/ at ${sourceClaude}; check the family's apm.yml`,
62
+ `apm install did not produce .claude/ at ${sourceClaude}. Check the family's apm.yml`,
63
63
  );
64
64
  }
65
65
  }
@@ -78,18 +78,18 @@ export class ApmInstaller {
78
78
  await fs.mkdir(stagedClaude, { recursive: true });
79
79
  }
80
80
 
81
- // apm's claude target deploys a pack's skills/ into .claude/skills/ but
82
- // never its agents/ subtree (agent profiles + references). Stage that from
83
- // the installed apm_modules into .claude/agents/ so a skill that cites an
84
- // agent reference (e.g. the work-item tracker matrix) resolves in the
85
- // agent CWD. No-op when --skills-from supplied a tree or no apm_modules
86
- // exist.
81
+ // apm's claude target deploys a pack's skills/ into .claude/skills/. It
82
+ // never deploys that pack's agents/ subtree (agent profiles +
83
+ // references). Stage that subtree from the installed apm_modules into
84
+ // .claude/agents/. A skill that cites an agent reference (e.g. the
85
+ // work-item tracker matrix) then resolves in the agent CWD. This is a
86
+ // no-op when --skills-from supplied a tree or no apm_modules exist.
87
87
  if (!skillsFrom) {
88
88
  await this.#stageApmAgents(family.rootPath, stagedClaude);
89
89
  }
90
90
 
91
- // Stage the family-local judge profile outside .claude/ so it is available
92
- // to the judge but never copied into the agent-under-test's CWD.
91
+ // Stage the family-local judge profile outside .claude/, so the judge can
92
+ // reach it. Nothing copies it into the agent-under-test's CWD.
93
93
  const judgeSource = join(family.rootPath, "judge.md");
94
94
  const judgeProfilesDir = join(stagingDir, "judge-profiles");
95
95
  try {
@@ -106,7 +106,7 @@ export class ApmInstaller {
106
106
  "sha256:" +
107
107
  createHash("sha256").update(normalizeLf(lockBytes)).digest("hex");
108
108
  } catch {
109
- // No lockfile — family doesn't use skill packs.
109
+ // No lockfile. The family doesn't use skill packs.
110
110
  }
111
111
 
112
112
  return { stagingDir, skillSetHash, judgeProfilesDir };
@@ -115,8 +115,8 @@ export class ApmInstaller {
115
115
  /**
116
116
  * Merge each installed pack's `agents/` subtree (profiles + references) from
117
117
  * `apm_modules/<owner>/<pack>/agents/` into the staged `.claude/agents/`.
118
- * apm's claude target deploys `skills/` only, so without this an agent
119
- * reference a skill cites is absent from the agent CWD.
118
+ * apm's claude target deploys `skills/` only. Without this merge, an agent
119
+ * reference that a skill cites is absent from the agent CWD.
120
120
  * @param {string} familyRoot
121
121
  * @param {string} stagedClaude
122
122
  */
@@ -127,7 +127,7 @@ export class ApmInstaller {
127
127
  try {
128
128
  owners = await fs.readdir(modulesRoot, { withFileTypes: true });
129
129
  } catch {
130
- return; // no apm_modules — nothing to stage
130
+ return; // no apm_modules, so nothing to stage
131
131
  }
132
132
  const stagedAgents = join(stagedClaude, "agents");
133
133
  for (const owner of owners) {
@@ -159,8 +159,8 @@ export class ApmInstaller {
159
159
  ["install", "--target", "claude"],
160
160
  { cwd, stdio: ["ignore", "pipe", "pipe"] },
161
161
  );
162
- // Drain stdout concurrently so the child never blocks on backpressure;
163
- // capture stderr for the failure message.
162
+ // Drain stdout concurrently so the child never blocks on backpressure.
163
+ // Capture stderr for the failure message.
164
164
  let stderr = "";
165
165
  const drainStdout = (async () => {
166
166
  for await (const _chunk of child.stdout) {
@@ -199,8 +199,8 @@ export function createApmInstaller(deps) {
199
199
  * @param {import("./task-family.js").TaskFamily} family
200
200
  * @param {string} outputDir
201
201
  * @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
202
- * @param {object} [options] - Forwarded to `ApmInstaller.install` (e.g.
203
- * `{ skillsFrom }`).
202
+ * @param {object} [options] - This function forwards these to
203
+ * `ApmInstaller.install` (e.g. `{ skillsFrom }`).
204
204
  */
205
205
  export function installApm(family, outputDir, runtime, options = {}) {
206
206
  return new ApmInstaller({ runtime }).install(family, outputDir, options);
@@ -1,17 +1,20 @@
1
1
  /**
2
2
  * Env-loader — auto-discover `.env` / `.env.local` files in a task family
3
- * and its tasks, load them into `process.env`, and render the merged result
3
+ * and its tasks. Load them into `process.env`. Render the merged result
4
4
  * into each agent CWD.
5
5
  *
6
- * Discovery paths (loaded in this order, first value per key wins):
7
- * 1. process.env (CI secrets, shell env — never overwritten)
6
+ * The loader reads the discovery paths in this order. The first value per
7
+ * key wins:
8
+ * 1. process.env (CI secrets and shell env, which the loader never
9
+ * overwrites)
8
10
  * 2. <family>/.env.local
9
11
  * 3. <family>/.env
10
12
  * 4. tasks/<id>/.env.local
11
13
  * 5. tasks/<id>/.env
12
14
  *
13
- * Every discovered env file — family or task — is loaded into process.env
14
- * AND rendered (with resolved values) into the agent working directory.
15
+ * The loader loads every discovered env file, family or task, into
16
+ * process.env. It also renders the file (with resolved values) into the
17
+ * agent working directory.
15
18
  */
16
19
 
17
20
  import { join } from "node:path";
@@ -20,7 +23,8 @@ const ENV_FILES = [".env.local", ".env"];
20
23
 
21
24
  /**
22
25
  * Parse a `.env` file into an array of {key, value} pairs.
23
- * Handles KEY=VALUE, # comments, blank lines, and single/double-quoted values.
26
+ * It handles KEY=VALUE, # comments, blank lines, and single/double-quoted
27
+ * values.
24
28
  * @param {string} content
25
29
  * @returns {Array<{key: string, value: string}>}
26
30
  */
@@ -46,7 +50,7 @@ export function parseEnvFile(content) {
46
50
  }
47
51
 
48
52
  /**
49
- * Read and parse an env file, returning [] if the file does not exist.
53
+ * Read and parse an env file. Return [] if the file does not exist.
50
54
  * @param {object} fs - Async filesystem surface (`runtime.fs`).
51
55
  * @param {string} filePath
52
56
  * @returns {Promise<Array<{key: string, value: string}>>}
@@ -62,10 +66,11 @@ async function readEnvFile(fs, filePath) {
62
66
  }
63
67
 
64
68
  /**
65
- * Load entries into the process env map. Existing keys are never overwritten.
69
+ * Load entries into the process env map. This function never overwrites an
70
+ * existing key.
66
71
  * @param {Record<string, string|undefined>} env - The `runtime.proc.env` map.
67
72
  * @param {Array<{key: string, value: string}>} entries
68
- * @returns {string[]} var names that were loaded
73
+ * @returns {string[]} the var names it loaded
69
74
  */
70
75
  function applyToProcessEnv(env, entries) {
71
76
  const names = [];
@@ -79,7 +84,8 @@ function applyToProcessEnv(env, entries) {
79
84
  }
80
85
 
81
86
  /**
82
- * Load one env file: apply to the env map, record keys in the merged map.
87
+ * Load one env file. Apply it to the env map. Record the keys in the merged
88
+ * map.
83
89
  * @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
84
90
  * @param {string} dir
85
91
  * @param {string} file
@@ -100,7 +106,7 @@ async function loadOneEnvFile(runtime, dir, file, names, merged) {
100
106
  }
101
107
 
102
108
  /**
103
- * Scan directories for env files, load into the env map, and collect
109
+ * Scan directories for env files. Load them into the env map. Collect
104
110
  * a merged key manifest per filename.
105
111
  * @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
106
112
  * @param {string[]} dirs
@@ -118,7 +124,7 @@ async function collectEnvEntries(runtime, dirs) {
118
124
  }
119
125
 
120
126
  /**
121
- * Write resolved env files into the agent CWD and warn about empty values.
127
+ * Write resolved env files into the agent CWD. Warn about empty values.
122
128
  * @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
123
129
  * @param {Map<string, Map<string, true>>} merged
124
130
  * @param {string} agentCwd
@@ -142,14 +148,16 @@ async function renderEnvFiles(runtime, merged, agentCwd) {
142
148
  }
143
149
 
144
150
  /**
145
- * Discover `.env` / `.env.local` in one or more directories, load them
146
- * into the process env map, and render the resolved values into the agent CWD.
151
+ * Discover `.env` / `.env.local` in one or more directories. Load them
152
+ * into the process env map. Render the resolved values into the agent CWD.
147
153
  *
148
154
  * @param {string[]} dirs - Directories to scan (family root, task dir, etc.)
149
155
  * @param {string} agentCwd - Agent working directory to render into.
150
156
  * @param {import("@forwardimpact/libutil/runtime").Runtime} runtime - Ambient
151
- * collaborators; uses `fs` (async read/write), `proc.env`, `proc.stderr`.
152
- * @returns {Promise<string[]>} All var names discovered (for redaction).
157
+ * collaborators. It uses `fs` (async read/write), `proc.env`, and
158
+ * `proc.stderr`.
159
+ * @returns {Promise<string[]>} Every var name the loader discovered (for
160
+ * redaction).
153
161
  */
154
162
  export async function loadEnv(dirs, agentCwd, runtime) {
155
163
  const { names, merged } = await collectEnvEntries(runtime, dirs);
@@ -1,46 +1,50 @@
1
1
  /**
2
- * Grading derivation — the sole home of the check-row arithmetic.
2
+ * Grade derivation — the sole home of the check-row arithmetic.
3
3
  *
4
- * Check rows are the single authoritative grading channel. Every row is a
5
- * check by default; a row declares its role with its own fields, checked in
6
- * order:
4
+ * Check rows are the one authoritative input to the grade. Every row is a
5
+ * check by default. A row declares its role with its own fields. The
6
+ * classifier checks the roles in this order:
7
7
  *
8
8
  * 1. Gate — `gate` is exactly `true`, `pass` is boolean, and no
9
- * `weight` key is present. Any failing gate → `gatesPass`
10
- * false.
11
- * 2. Diagnostic — no `gate` key and `weight` is exactly `0`. Free-form;
12
- * never graded.
9
+ * `weight` key is present. A gate that fails sets
10
+ * `gatesPass` to false.
11
+ * 2. Diagnostic — no `gate` key and `weight` is exactly `0`. The row is
12
+ * free-form. The grader never scores it.
13
13
  * 3. Scored — no `gate` key, boolean `pass`, `weight` absent (defaults
14
14
  * to 1) or finite > 0.
15
- * 4. Malformed — everything else: any `gate`+`weight` co-occurrence (a
16
- * stray weight must never silently disarm a gate), a
15
+ * 4. Malformed — everything else: any `gate`+`weight` co-occurrence, a
17
16
  * non-boolean `gate`, a missing or non-boolean `pass` on a
18
17
  * graded row, an invalid `weight`, an fd-3 line that failed
19
- * to parse, a non-object row. Counts as a **failing scored
20
- * check** — dropping a defect could mint full marks;
21
- * failing the whole run would zero completed work.
18
+ * to parse, a non-object row. A stray weight must never
19
+ * silently disarm a gate. A malformed row counts as a
20
+ * **scored check that fails**. If the grader dropped the
21
+ * defect, the cell could mint full marks. If it failed the
22
+ * whole run, it would zero completed work.
22
23
  *
23
- * The producers' `source` stamp is display metadata, never a grading input.
24
+ * The producers' `source` stamp is display metadata. The grader never reads
25
+ * it.
24
26
  */
25
27
 
26
28
  /**
27
29
  * @typedef {object} GradeResult
28
30
  * @property {"pass" | "fail"} verdict - `healthy ∧ gatesPass ∧ fullMarks`.
29
31
  * @property {boolean} gatesPass - Every gate row passes (vacuously true).
30
- * @property {number | null} score - Weighted fraction of passing scored
31
- * checks; `null` when the cell has zero scored checks (binary task).
32
- * @property {boolean} fullMarks - Integer count predicate: no malformed rows
33
- * and every scored check passes. Never a float comparison, so fractional
34
- * weights carry no equality hazard. Vacuously true with zero scored checks.
32
+ * @property {number | null} score - Weighted fraction of the scored checks
33
+ * that pass. It is `null` when the cell has zero scored checks (binary
34
+ * task).
35
+ * @property {boolean} fullMarks - An integer count predicate. It is true when
36
+ * no row is malformed and every scored check passes. It is never a float
37
+ * comparison, so fractional weights carry no equality hazard. It is
38
+ * vacuously true with zero scored checks.
35
39
  * @property {number} malformed - Malformed row count.
36
40
  */
37
41
 
38
42
  /**
39
43
  * Grade the merged check rows against grader health.
40
44
  *
41
- * `healthy` is the completion signal a crashed grader cannot fake: when it is
42
- * false the verdict is `fail` whatever the rows say, so a hook that dies
43
- * after emitting passing rows can never mint marks.
45
+ * `healthy` is the completion signal a crashed grader cannot fake. When it is
46
+ * false, the verdict is `fail` whatever the rows say. A hook that emits rows
47
+ * that pass and then dies can never mint marks.
44
48
  * @param {unknown[]} details - Merged check rows from both producers.
45
49
  * @param {boolean} healthy - Invariants exited 0 AND the hidden-test engine
46
50
  * did not throw.
@@ -73,7 +77,7 @@ export function gradeChecks(details, healthy) {
73
77
  }
74
78
 
75
79
  /**
76
- * Fold one row into the running tally per its classified role.
80
+ * Fold one row into the tally per its classified role.
77
81
  * @param {{gatesPass: boolean, malformed: number, scored: number, passing: number, weightAll: number, weightPassing: number}} tally
78
82
  * @param {unknown} row
79
83
  */
@@ -96,11 +100,10 @@ function tallyRow(tally, row) {
96
100
  }
97
101
 
98
102
  /**
99
- * Run both check-row producers and grade the merged rows — the one
100
- * composition shared by the runner and the `grade` subcommand. An engine
101
- * throw is grader fault: its message lands on the returned `engineError`
102
- * and health fails, so a crashed grader can never mint marks from rows it
103
- * happened to emit first.
103
+ * Run both check-row producers and grade the merged rows. The runner and the
104
+ * `grade` subcommand share this one composition. An engine throw is grader
105
+ * fault. Its message lands on the returned `engineError` and health fails. A
106
+ * crashed grader can never mint marks from rows it emitted first.
104
107
  * @param {import("./task-family.js").Task} task
105
108
  * @param {{cwd: string, port: number, runDir: string, familyDir?: string|null}} ctx
106
109
  * @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
@@ -126,8 +129,8 @@ export async function runProducersAndGrade(task, ctx, runtime, producers) {
126
129
 
127
130
  /**
128
131
  * Merge the two producers' rows (invariants first) and stamp each row's
129
- * provenance. The stamp is display metadata, never a grading input, and
130
- * non-object rows (malformed by contract) pass through verbatim.
132
+ * provenance. The stamp is display metadata. The grader never reads it.
133
+ * Non-object rows (malformed by contract) pass through verbatim.
131
134
  * @param {unknown[]} invariantsDetails
132
135
  * @param {unknown[]} hiddenDetails
133
136
  * @returns {unknown[]}
@@ -147,9 +150,9 @@ function stampSource(row, source) {
147
150
  }
148
151
 
149
152
  /**
150
- * Project the raw `gradeChecks` return onto the record schema: `fullMarks`
151
- * is derivable and dropped, `score` is omitted on binary tasks (`null`),
152
- * `malformed` is omitted when clean.
153
+ * Project the raw `gradeChecks` return onto the record schema. `fullMarks` is
154
+ * derivable, so this function drops it. It omits `score` on binary tasks
155
+ * (`null`). It omits `malformed` when the rows are clean.
153
156
  * @param {GradeResult} raw
154
157
  * @returns {{verdict: "pass"|"fail", gatesPass: boolean, score?: number, malformed?: number}}
155
158
  */
@@ -177,9 +180,9 @@ function classifyRow(row) {
177
180
  }
178
181
 
179
182
  /**
180
- * A row carrying a `gate` key: valid only as `gate: true` with a boolean
181
- * `pass` and no `weight` key — any co-occurring weight is malformed so a
182
- * stray weight can never silently disarm a gate.
183
+ * Classify a row that has a `gate` key. The row is valid only as `gate: true`
184
+ * with a boolean `pass` and no `weight` key. Any weight that co-occurs makes
185
+ * the row malformed. A stray weight can never silently disarm a gate.
183
186
  * @param {object} row
184
187
  * @returns {"gate" | "malformed"}
185
188
  */
@@ -191,9 +194,9 @@ function classifyGateRow(row) {
191
194
  }
192
195
 
193
196
  /**
194
- * A gate-less row carrying a `weight` key: exactly 0 is a diagnostic, a
195
- * finite positive weight with a boolean `pass` is scored, anything else is
196
- * malformed.
197
+ * Classify a gate-less row that has a `weight` key. A weight of exactly 0 is
198
+ * a diagnostic. A finite positive weight with a boolean `pass` is scored.
199
+ * Anything else is malformed.
197
200
  * @param {object} row
198
201
  * @returns {"diagnostic" | "scored" | "malformed"}
199
202
  */
@@ -205,8 +208,8 @@ function classifyWeightedRow(row) {
205
208
  }
206
209
 
207
210
  /**
208
- * A malformed row fails at its own weight when it carries a valid positive
209
- * one, else at unit weight 1.
211
+ * A malformed row fails at its own weight when that weight is valid and
212
+ * positive. Otherwise it fails at unit weight 1.
210
213
  * @param {unknown} row
211
214
  * @returns {number}
212
215
  */
@@ -1,14 +1,14 @@
1
1
  /**
2
- * Hidden-test engine — executes a task's `tests/` overlay against the
3
- * post-run agent CWD: stage each file at its mirrored path, run each check
4
- * with `node --test`, convert the exit status into one check row, and
5
- * restore the tree so the judge sees the workdir exactly as the agent left
6
- * it.
2
+ * Hidden-test engine — runs a task's `tests/` overlay against the post-run
3
+ * agent CWD. The engine stages each file at its mirrored path. It runs each
4
+ * check with `bun test`. It converts the exit status into one check row.
5
+ * It restores the tree, so the judge sees the workdir exactly as the agent
6
+ * left it.
7
7
  *
8
- * Fault attribution is the engine's contract: a stage or spawn failure (the
9
- * agent deleted the scaffold) is a *failing row* — agent fault; the engine
10
- * itself throwing is grader fault, which the caller records as unhealthy so
11
- * a crashed grader can never mint marks.
8
+ * Fault attribution is the engine's contract. A stage or spawn failure (the
9
+ * agent deleted the scaffold) is agent fault, so the engine returns a *row
10
+ * that fails*. A throw from the engine itself is grader fault. The caller
11
+ * records that as unhealthy, so a crashed grader can never mint marks.
12
12
  */
13
13
 
14
14
  import { dirname, join } from "node:path";
@@ -16,8 +16,8 @@ import { dirname, join } from "node:path";
16
16
  import { buildHookEnv } from "./hook-env.js";
17
17
 
18
18
  // Fixed per-check budget. A wedged test process runs outside the agent
19
- // watchdog, so this bound is what keeps a hung hidden test from stalling the
20
- // cell; the timeout row keeps the failure visible.
19
+ // watchdog. Without this bound, a hung hidden test would stall the cell. The
20
+ // timeout row keeps the failure visible.
21
21
  const CHECK_TIMEOUT_MS = 120_000;
22
22
  const STDERR_TAIL_CHARS = 500;
23
23
 
@@ -50,9 +50,9 @@ export async function runHiddenTests(task, ctx, runtime, opts = {}) {
50
50
  }
51
51
 
52
52
  /**
53
- * Stage one check, run it, and restore its staging — the check's own row is
54
- * the only trace it leaves. A stage failure is the agent's fault (a deleted
55
- * scaffold), so it becomes a failing row rather than a throw.
53
+ * Stage one check, run it, then put the tree back. The check's own row is the
54
+ * only trace it leaves. A stage failure is the agent's fault (a deleted
55
+ * scaffold). The engine returns a row that fails. It does not throw.
56
56
  */
57
57
  async function runOneCheck(task, ctx, runtime, timeoutMs, check) {
58
58
  const stager = newStager();
@@ -69,9 +69,28 @@ async function runOneCheck(task, ctx, runtime, timeoutMs, check) {
69
69
  }
70
70
 
71
71
  /**
72
- * Spawn `node --test <staged path>` from the agent CWD under the hook env
72
+ * Reads a child pipe to a string. Returns what it read when the stream tears
73
+ * down mid-read, because a pipe that closes as the child exits is a race, not
74
+ * a check failure. The caller reads the exit code for the verdict.
75
+ *
76
+ * @param {import("node:stream").Readable} stream - Child stdout or stderr.
77
+ * @returns {Promise<string>} Everything read before the stream ended.
78
+ */
79
+ async function drainQuietly(stream) {
80
+ let out = "";
81
+ try {
82
+ for await (const chunk of stream) out += chunk.toString();
83
+ } catch {
84
+ // The stream closed under us. Keep what arrived.
85
+ }
86
+ return out;
87
+ }
88
+
89
+ /**
90
+ * Spawn `bun test <staged path>` from the agent CWD under the hook env
73
91
  * and map the exit status onto one row. The clock timer SIGKILLs a child
74
- * that outlives the per-check budget; the row fails with a timeout message.
92
+ * that outlives the per-check budget. The row then fails with a timeout
93
+ * message.
75
94
  */
76
95
  async function spawnCheck(task, ctx, runtime, timeoutMs, check) {
77
96
  const env = buildHookEnv(runtime.proc.env, {
@@ -82,27 +101,40 @@ async function spawnCheck(task, ctx, runtime, timeoutMs, check) {
82
101
  hooksDir: task.paths.hooks,
83
102
  familyDir: ctx.familyDir,
84
103
  });
85
- // An inherited test-runner context makes the child `node --test` report
86
- // exit 0 even when its tests fail — a failing check would mint a passing
87
- // row whenever the harness itself runs under `node --test`.
88
- delete env.NODE_TEST_CONTEXT;
89
- const child = runtime.subprocess.spawn("node", ["--test", check.stagePath], {
90
- cwd: ctx.cwd,
91
- env,
92
- stdio: ["ignore", "pipe", "pipe"],
93
- });
104
+ // `bun test` sets no test-context variable that a nested run inherits, so
105
+ // a child reports its own exit status even when the harness itself runs
106
+ // under a test runner. That removes the `NODE_TEST_CONTEXT` scrub the
107
+ // `node --test` engine needed to stop a failing check minting a passing
108
+ // row. The "fractional score" grade test is the standing guard: it fails
109
+ // if a failing check ever reports success again.
110
+ //
111
+ // Pass the ABSOLUTE staged path. `bun test` reads its argument as a
112
+ // substring filter over discovered paths, not as one file, so the relative
113
+ // `app/test/x.test.js` also matches an agent-authored `sub/app/test/
114
+ // x.test.js` and folds that file's result into this check's row. An
115
+ // absolute path cannot be a substring of a deeper path, so it selects
116
+ // exactly the staged file. One `*.test.js` stays one check.
117
+ const child = runtime.subprocess.spawn(
118
+ "bun",
119
+ ["test", join(ctx.cwd, check.stagePath)],
120
+ {
121
+ cwd: ctx.cwd,
122
+ env,
123
+ stdio: ["ignore", "pipe", "pipe"],
124
+ },
125
+ );
94
126
  let timedOut = false;
95
127
  const timer = runtime.clock.setTimeout(() => {
96
128
  timedOut = true;
97
129
  child.kill("SIGKILL");
98
130
  }, timeoutMs);
99
- const drainStdout = (async () => {
100
- for await (const _chunk of child.stdout) {
101
- // discard
102
- }
103
- })();
104
- let stderr = "";
105
- for await (const chunk of child.stderr) stderr += chunk.toString();
131
+ // Both pipes drain only so a chatty child never blocks on a full pipe.
132
+ // A child that exits while its pipe is still open makes the async iterator
133
+ // reject with ERR_STREAM_PREMATURE_CLOSE on some runtimes. That teardown
134
+ // race is not a check result, so it must not throw out of the check. The
135
+ // exit code below is the verdict.
136
+ const drainStdout = drainQuietly(child.stdout);
137
+ const stderr = await drainQuietly(child.stderr);
106
138
  await drainStdout;
107
139
  const exit = await child.exitCode;
108
140
  runtime.clock.clearTimeout(timer);
@@ -129,8 +161,8 @@ function newStager() {
129
161
  }
130
162
 
131
163
  /**
132
- * Copy the symlink-resolved source to its mirrored path under the agent CWD,
133
- * backing up a collided file's bytes and tracking every directory created so
164
+ * Copy the symlink-resolved source to its mirrored path under the agent CWD.
165
+ * Back up the bytes of a collided file. Track every directory created, so
134
166
  * `unstage` can put the tree back exactly.
135
167
  */
136
168
  async function stageFile(fs, cwd, stager, { sourcePath, stagePath }) {
@@ -154,7 +186,7 @@ async function ensureParents(fs, cwd, stager, dir) {
154
186
  await fs.access(dir);
155
187
  return;
156
188
  } catch {
157
- // missing — create below
189
+ // missing, so create it below
158
190
  }
159
191
  await ensureParents(fs, cwd, stager, dirname(dir));
160
192
  await fs.mkdir(dir);
@@ -162,10 +194,10 @@ async function ensureParents(fs, cwd, stager, dir) {
162
194
  }
163
195
 
164
196
  /**
165
- * Reverse the staging: staged copies out, collided bytes back, created
166
- * directories removed (deepest first — a check's own artifacts inside a
167
- * created directory go with it, since that directory did not exist when the
168
- * agent finished).
197
+ * Reverse the stage step. Remove the staged copies. Write the collided bytes
198
+ * back. Remove the created directories, deepest first. A check's own
199
+ * artifacts inside a created directory go with it, because that directory did
200
+ * not exist when the agent finished.
169
201
  */
170
202
  async function unstage(fs, stager) {
171
203
  for (const target of stager.staged) {
@@ -1,12 +1,13 @@
1
1
  /**
2
- * Shared environment builder for the benchmark hook scripts (`preflight.sh` and
3
- * `invariants.sh`). Keeping both spawns on one helper guarantees they expose the
4
- * same variable set, so hook authors never have to wonder which vars a given
5
- * hook receives.
2
+ * Shared environment builder for the benchmark hook scripts (`preflight.sh`
3
+ * and `invariants.sh`). One helper serves both spawns, so both expose the
4
+ * same variable set. Hook authors never have to wonder which vars a hook
5
+ * receives.
6
6
  *
7
7
  * Path vars (TASK_DIR, FAMILY_DIR, HOOKS_DIR) let hooks reference real
8
- * locations instead of reconstructing them from `$0`. They are paths, not
9
- * secrets, so they need no redaction allowlist entry.
8
+ * locations. Hooks do not have to rebuild the locations from `$0`. They are
9
+ * paths. They are not secrets, so they need no entry in the redaction
10
+ * allowlist.
10
11
  */
11
12
 
12
13
  /**
@@ -27,9 +28,10 @@ export function buildHookEnv(
27
28
  ) {
28
29
  return {
29
30
  ...baseEnv,
30
- // The agent CWD itself — hooks reference emitted files as `$AGENT_CWD/<path>`.
31
- // Distinct from the `invariants` CLI's `--run-dir` (the parent that
32
- // *contains* `cwd/`), so the two are never confused.
31
+ // The agent CWD itself. Hooks reference emitted files as
32
+ // `$AGENT_CWD/<path>`. This var is distinct from the `invariants` CLI's
33
+ // `--run-dir` (the parent that *contains* `cwd/`), so the two are never
34
+ // confused.
33
35
  AGENT_CWD: cwd,
34
36
  PORT: String(port),
35
37
  TASK_ID: taskId,