@forwardimpact/libharness 2.0.0 → 3.0.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (73) hide show
  1. package/README.md +68 -65
  2. package/package.json +15 -13
  3. package/src/advisor.js +47 -41
  4. package/src/agent-runner.js +58 -48
  5. package/src/benchmark/apm-installer.js +28 -28
  6. package/src/benchmark/env-loader.js +24 -16
  7. package/src/benchmark/grade.js +44 -41
  8. package/src/benchmark/hidden-tests.js +25 -24
  9. package/src/benchmark/hook-env.js +11 -9
  10. package/src/benchmark/invariants.js +20 -17
  11. package/src/benchmark/judge.js +29 -28
  12. package/src/benchmark/npm-installer.js +9 -8
  13. package/src/benchmark/report.js +53 -50
  14. package/src/benchmark/result.js +24 -23
  15. package/src/benchmark/runner.js +75 -69
  16. package/src/benchmark/scheduler.js +17 -16
  17. package/src/benchmark/task-family.js +29 -27
  18. package/src/benchmark/trace-split.js +9 -8
  19. package/src/benchmark/workdir.js +27 -25
  20. package/src/claude-code-executable.js +11 -11
  21. package/src/commands/advisor-flags.js +8 -7
  22. package/src/commands/assert.js +16 -15
  23. package/src/commands/benchmark-definition.js +20 -20
  24. package/src/commands/benchmark-grade.js +13 -12
  25. package/src/commands/benchmark-report.js +5 -5
  26. package/src/commands/benchmark-run.js +31 -28
  27. package/src/commands/by-discussion.js +11 -11
  28. package/src/commands/callback.js +11 -11
  29. package/src/commands/discuss.js +8 -7
  30. package/src/commands/facilitate.js +16 -14
  31. package/src/commands/output.js +4 -3
  32. package/src/commands/run.js +15 -15
  33. package/src/commands/scan-logs.js +22 -20
  34. package/src/commands/selfedit.js +124 -0
  35. package/src/commands/supervise.js +13 -11
  36. package/src/commands/task-input.js +9 -9
  37. package/src/commands/tee.js +11 -10
  38. package/src/commands/trace.js +55 -42
  39. package/src/commands/work-tracker.js +4 -3
  40. package/src/cost.js +17 -17
  41. package/src/discuss-tools.js +16 -16
  42. package/src/discusser.js +39 -38
  43. package/src/events/github.js +54 -37
  44. package/src/facilitator.js +21 -21
  45. package/src/inbox-poller.js +4 -4
  46. package/src/judge.js +32 -30
  47. package/src/message-bus.js +12 -11
  48. package/src/orchestration-loop.js +35 -36
  49. package/src/orchestration-toolkit.js +58 -53
  50. package/src/orchestrator-helpers.js +2 -2
  51. package/src/profile-prompt.js +54 -53
  52. package/src/redaction.js +63 -57
  53. package/src/render/line-renderer.js +5 -5
  54. package/src/render/orchestrator-filter.js +3 -3
  55. package/src/render/palette.js +11 -9
  56. package/src/render/tool-hints.js +18 -15
  57. package/src/render/turn-renderer.js +4 -4
  58. package/src/reply-emitter.js +2 -2
  59. package/src/sequence-counter.js +4 -3
  60. package/src/signature-filter.js +7 -6
  61. package/src/supervisor.js +19 -18
  62. package/src/tee-writer.js +25 -25
  63. package/src/trace-collector.js +53 -48
  64. package/src/trace-github.js +53 -44
  65. package/src/trace-multi.js +16 -14
  66. package/src/trace-query.js +61 -52
  67. package/src/trace-render.js +19 -19
  68. package/src/trace-usage.js +31 -28
  69. package/src/transcript-recorder.js +24 -20
  70. package/bin/fit-benchmark.js +0 -44
  71. package/bin/fit-harness.js +0 -412
  72. package/bin/fit-selfedit.js +0 -165
  73. package/bin/fit-trace.js +0 -520
@@ -1,14 +1,14 @@
1
1
  /**
2
- * Hidden-test engine — executes a task's `tests/` overlay against the
3
- * post-run agent CWD: stage each file at its mirrored path, run each check
4
- * with `node --test`, convert the exit status into one check row, and
5
- * restore the tree so the judge sees the workdir exactly as the agent left
6
- * it.
2
+ * Hidden-test engine — runs a task's `tests/` overlay against the post-run
3
+ * agent CWD. The engine stages each file at its mirrored path. It runs each
4
+ * check with `node --test`. It converts the exit status into one check row.
5
+ * It restores the tree, so the judge sees the workdir exactly as the agent
6
+ * left it.
7
7
  *
8
- * Fault attribution is the engine's contract: a stage or spawn failure (the
9
- * agent deleted the scaffold) is a *failing row* — agent fault; the engine
10
- * itself throwing is grader fault, which the caller records as unhealthy so
11
- * a crashed grader can never mint marks.
8
+ * Fault attribution is the engine's contract. A stage or spawn failure (the
9
+ * agent deleted the scaffold) is agent fault, so the engine returns a *row
10
+ * that fails*. A throw from the engine itself is grader fault. The caller
11
+ * records that as unhealthy, so a crashed grader can never mint marks.
12
12
  */
13
13
 
14
14
  import { dirname, join } from "node:path";
@@ -16,8 +16,8 @@ import { dirname, join } from "node:path";
16
16
  import { buildHookEnv } from "./hook-env.js";
17
17
 
18
18
  // Fixed per-check budget. A wedged test process runs outside the agent
19
- // watchdog, so this bound is what keeps a hung hidden test from stalling the
20
- // cell; the timeout row keeps the failure visible.
19
+ // watchdog. Without this bound, a hung hidden test would stall the cell. The
20
+ // timeout row keeps the failure visible.
21
21
  const CHECK_TIMEOUT_MS = 120_000;
22
22
  const STDERR_TAIL_CHARS = 500;
23
23
 
@@ -50,9 +50,9 @@ export async function runHiddenTests(task, ctx, runtime, opts = {}) {
50
50
  }
51
51
 
52
52
  /**
53
- * Stage one check, run it, and restore its staging — the check's own row is
54
- * the only trace it leaves. A stage failure is the agent's fault (a deleted
55
- * scaffold), so it becomes a failing row rather than a throw.
53
+ * Stage one check, run it, then put the tree back. The check's own row is the
54
+ * only trace it leaves. A stage failure is the agent's fault (a deleted
55
+ * scaffold). The engine returns a row that fails. It does not throw.
56
56
  */
57
57
  async function runOneCheck(task, ctx, runtime, timeoutMs, check) {
58
58
  const stager = newStager();
@@ -71,7 +71,8 @@ async function runOneCheck(task, ctx, runtime, timeoutMs, check) {
71
71
  /**
72
72
  * Spawn `node --test <staged path>` from the agent CWD under the hook env
73
73
  * and map the exit status onto one row. The clock timer SIGKILLs a child
74
- * that outlives the per-check budget; the row fails with a timeout message.
74
+ * that outlives the per-check budget. The row then fails with a timeout
75
+ * message.
75
76
  */
76
77
  async function spawnCheck(task, ctx, runtime, timeoutMs, check) {
77
78
  const env = buildHookEnv(runtime.proc.env, {
@@ -83,8 +84,8 @@ async function spawnCheck(task, ctx, runtime, timeoutMs, check) {
83
84
  familyDir: ctx.familyDir,
84
85
  });
85
86
  // An inherited test-runner context makes the child `node --test` report
86
- // exit 0 even when its tests fail — a failing check would mint a passing
87
- // row whenever the harness itself runs under `node --test`.
87
+ // exit 0 even when its tests fail. A check that fails would then mint a row
88
+ // that passes whenever the harness itself runs under `node --test`.
88
89
  delete env.NODE_TEST_CONTEXT;
89
90
  const child = runtime.subprocess.spawn("node", ["--test", check.stagePath], {
90
91
  cwd: ctx.cwd,
@@ -129,8 +130,8 @@ function newStager() {
129
130
  }
130
131
 
131
132
  /**
132
- * Copy the symlink-resolved source to its mirrored path under the agent CWD,
133
- * backing up a collided file's bytes and tracking every directory created so
133
+ * Copy the symlink-resolved source to its mirrored path under the agent CWD.
134
+ * Back up the bytes of a collided file. Track every directory created, so
134
135
  * `unstage` can put the tree back exactly.
135
136
  */
136
137
  async function stageFile(fs, cwd, stager, { sourcePath, stagePath }) {
@@ -154,7 +155,7 @@ async function ensureParents(fs, cwd, stager, dir) {
154
155
  await fs.access(dir);
155
156
  return;
156
157
  } catch {
157
- // missing — create below
158
+ // missing, so create it below
158
159
  }
159
160
  await ensureParents(fs, cwd, stager, dirname(dir));
160
161
  await fs.mkdir(dir);
@@ -162,10 +163,10 @@ async function ensureParents(fs, cwd, stager, dir) {
162
163
  }
163
164
 
164
165
  /**
165
- * Reverse the staging: staged copies out, collided bytes back, created
166
- * directories removed (deepest first — a check's own artifacts inside a
167
- * created directory go with it, since that directory did not exist when the
168
- * agent finished).
166
+ * Reverse the stage step. Remove the staged copies. Write the collided bytes
167
+ * back. Remove the created directories, deepest first. A check's own
168
+ * artifacts inside a created directory go with it, because that directory did
169
+ * not exist when the agent finished.
169
170
  */
170
171
  async function unstage(fs, stager) {
171
172
  for (const target of stager.staged) {
@@ -1,12 +1,13 @@
1
1
  /**
2
- * Shared environment builder for the benchmark hook scripts (`preflight.sh` and
3
- * `invariants.sh`). Keeping both spawns on one helper guarantees they expose the
4
- * same variable set, so hook authors never have to wonder which vars a given
5
- * hook receives.
2
+ * Shared environment builder for the benchmark hook scripts (`preflight.sh`
3
+ * and `invariants.sh`). One helper serves both spawns, so both expose the
4
+ * same variable set. Hook authors never have to wonder which vars a hook
5
+ * receives.
6
6
  *
7
7
  * Path vars (TASK_DIR, FAMILY_DIR, HOOKS_DIR) let hooks reference real
8
- * locations instead of reconstructing them from `$0`. They are paths, not
9
- * secrets, so they need no redaction allowlist entry.
8
+ * locations. Hooks do not have to rebuild the locations from `$0`. They are
9
+ * paths. They are not secrets, so they need no entry in the redaction
10
+ * allowlist.
10
11
  */
11
12
 
12
13
  /**
@@ -27,9 +28,10 @@ export function buildHookEnv(
27
28
  ) {
28
29
  return {
29
30
  ...baseEnv,
30
- // The agent CWD itself — hooks reference emitted files as `$AGENT_CWD/<path>`.
31
- // Distinct from the `invariants` CLI's `--run-dir` (the parent that
32
- // *contains* `cwd/`), so the two are never confused.
31
+ // The agent CWD itself. Hooks reference emitted files as
32
+ // `$AGENT_CWD/<path>`. This var is distinct from the `invariants` CLI's
33
+ // `--run-dir` (the parent that *contains* `cwd/`), so the two are never
34
+ // confused.
33
35
  AGENT_CWD: cwd,
34
36
  PORT: String(port),
35
37
  TASK_ID: taskId,
@@ -1,14 +1,15 @@
1
1
  /**
2
2
  * Invariants — runs `<task.paths.hooks>/invariants.sh` from the template path
3
- * against the post-run agent CWD. A pure collector with no verdict of its
4
- * own: structured per-check rows arrive on fd 3 (`$RESULTS_FD=3`) as NDJSON
5
- * and grading happens downstream over the merged rows. The exit code is
6
- * script health only — nonzero means the grader itself failed, never that a
7
- * check failed.
3
+ * against the post-run agent CWD. The module is a pure collector with no
4
+ * verdict of its own. Structured per-check rows arrive on fd 3
5
+ * (`$RESULTS_FD=3`) as NDJSON. Downstream code grades the merged rows. The
6
+ * exit code is script health only. A nonzero code means the grader itself
7
+ * failed. It never means that a check failed.
8
8
  *
9
- * Subprocess access flows through `runtime.subprocess.spawn`; the fd-3 backing
10
- * store and the stderr log use the sync filesystem surface (`runtime.fsSync`) —
11
- * the only surface this module touches, per design Decision 7.
9
+ * Subprocess access flows through `runtime.subprocess.spawn`. The fd-3
10
+ * backing store and the stderr log use the sync filesystem surface
11
+ * (`runtime.fsSync`). That surface is the only one this module touches, per
12
+ * design Decision 7.
12
13
  */
13
14
 
14
15
  import { join } from "node:path";
@@ -18,11 +19,12 @@ import { buildHookEnv } from "./hook-env.js";
18
19
  /**
19
20
  * @typedef {object} InvariantsResult
20
21
  * @property {Array<object>} details
21
- * @property {number} exitCode - Script health: nonzero means the hook itself
22
- * failed, never that a check failed.
22
+ * @property {number} exitCode - Script health. A nonzero code means the hook
23
+ * itself failed. It never means that a check failed.
23
24
  * @property {string} [stderr] - Trimmed script stderr, present only when the
24
- * script wrote to stderr. Surfaces hook failures (e.g. a missing tool) that
25
- * leave `details` empty, so they read distinctly from a real invariant miss.
25
+ * script wrote to stderr. This field surfaces hook failures (e.g. a missing
26
+ * tool) that leave `details` empty. A reader can then tell them apart from
27
+ * a real invariant miss.
26
28
  */
27
29
 
28
30
  /**
@@ -43,8 +45,8 @@ export async function runInvariants(task, ctx, runtime) {
43
45
 
44
46
  // Bun's child_process pipe setup for fd >= 3 is racy under load (it
45
47
  // creates a unix socket pair and the connect() can return ENOENT). Use
46
- // a temp file as the fd-3 backing store instead — the script still
47
- // writes via `$RESULTS_FD`, but we hand it a real file descriptor.
48
+ // a temp file as the fd-3 backing store instead. The script still writes
49
+ // through `$RESULTS_FD`, but we hand it a real file descriptor.
48
50
  const fd3Path = join(ctx.runDir, "invariants.fd3.ndjson");
49
51
  const fd3File = fsSync.openSync(fd3Path, "w+");
50
52
 
@@ -69,7 +71,8 @@ export async function runInvariants(task, ctx, runtime) {
69
71
  throw e;
70
72
  }
71
73
 
72
- // Drain stdout (do not require consumers to read it); capture stderr to log.
74
+ // Drain stdout (do not require consumers to read it). Capture stderr to
75
+ // the log.
73
76
  const drainStdout = (async () => {
74
77
  for await (const _chunk of child.stdout) {
75
78
  // discard
@@ -127,8 +130,8 @@ function readAndUnlink(fsSync, path) {
127
130
  }
128
131
 
129
132
  /**
130
- * Parse the fd-3 buffer (read from the temp-file backing) into one NDJSON
131
- * row per detail entry.
133
+ * Parse the fd-3 buffer (read from the temp-file backing store) into one
134
+ * NDJSON row per detail entry.
132
135
  */
133
136
  function parseFd3Buffer(buf, details) {
134
137
  if (!buf) return;
@@ -1,8 +1,9 @@
1
1
  /**
2
- * Benchmark adapter for the libharness `Judge`. Templates the family's
3
- * `judge.task.md` with structured context variables, runs the judge against
4
- * the post-run agent CWD, and returns the verdict in the benchmark's
5
- * `pass`/`fail` vocabulary (mapped from libharness's `success`/`failure`).
2
+ * Benchmark adapter for the libharness `Judge`. The adapter fills the
3
+ * family's `judge.task.md` template with structured context variables. It
4
+ * runs the judge against the post-run agent CWD. It returns the verdict in
5
+ * the benchmark's `pass`/`fail` vocabulary (mapped from libharness's
6
+ * `success`/`failure`).
6
7
  *
7
8
  * Template variables available in `judge.task.md`:
8
9
  *
@@ -12,13 +13,13 @@
12
13
  * {{GRADE_RESULT}} — JSON grade object plus the merged check rows
13
14
  * {{SKILL_SET_HASH}} — SHA-256 from apm.lock.yaml
14
15
  * {{TASK_ID}} — task name (directory under tasks/)
15
- * {{TASK_DIR}} — agent working directory path
16
+ * {{TASK_DIR}} — path to the agent working directory
16
17
  *
17
- * The judge verdict is captured from the orchestration context's
18
- * `concluded` flag directly — no trace parsing on the happy path.
19
- * `parseConcludeFromTrace` is preserved for offline analysis and as a
20
- * fallback when the runtime ctx isn't available (e.g. re-grading a
21
- * historical run from its judge.ndjson file).
18
+ * The adapter reads the judge verdict directly from the orchestration
19
+ * context's `concluded` flag. It does not parse the trace on the happy path.
20
+ * `parseConcludeFromTrace` stays for offline analysis. It is also a fallback
21
+ * when the runtime ctx is not available, for example when you re-grade a
22
+ * historical run from its judge.ndjson file.
22
23
  */
23
24
 
24
25
  import { createJudge } from "../judge.js";
@@ -41,8 +42,8 @@ import { sumTraceCost } from "../cost.js";
41
42
 
42
43
  /**
43
44
  * Run the judge over a completed task run. The judge is a binary gate over
44
- * the grade's validity, never a grade itself: `gradeResult` reaches the
45
- * template as evidence, and the verdict stays pass/fail.
45
+ * the grade's validity. The judge is never a grade itself. `gradeResult`
46
+ * reaches the template as evidence. The verdict stays pass/fail.
46
47
  * @param {import("./task-family.js").Task} task
47
48
  * @param {import("./workdir.js").Workdir} workdir
48
49
  * @param {{verdict: string, gatesPass: boolean, score?: number, malformed?: number, rows: unknown[]}} gradeResult -
@@ -86,9 +87,9 @@ export async function runJudge(task, workdir, gradeResult, deps, context) {
86
87
  await new Promise((r) => output.end(r));
87
88
  }
88
89
 
89
- // The judge is its own SDK session; its spend lands in the judge trace we
90
- // just wrote, not in the supervisor's combined trace. Read it back so the
91
- // benchmark record's cost includes the judge.
90
+ // The judge is its own SDK session. Its spend lands in the judge trace we
91
+ // just wrote. It does not land in the supervisor's combined trace. Read the
92
+ // trace back, so the benchmark record's cost includes the judge.
92
93
  const judgeTrace = await fs
93
94
  .readFile(workdir.judgeTracePath, "utf8")
94
95
  .catch(() => "");
@@ -109,10 +110,11 @@ export async function runJudge(task, workdir, gradeResult, deps, context) {
109
110
  }
110
111
 
111
112
  /**
112
- * Parse the last judge-source (or supervisor-source, for backward compat
113
- * with pre-Judge-class traces) `Conclude` tool call from an NDJSON trace
114
- * and map the verdict (`success → pass`, `failure → fail`). Preserved for
115
- * offline analysis; not used on the runtime happy path.
113
+ * Parse the last `Conclude` tool call from an NDJSON trace and map the
114
+ * verdict (`success → pass`, `failure → fail`). The call may carry a
115
+ * judge source or a supervisor source. The supervisor source keeps backward
116
+ * compatibility with pre-Judge-class traces. This function stays for offline
117
+ * analysis. The runtime happy path does not use it.
116
118
  * @param {string} tracePath
117
119
  * @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
118
120
  * @returns {Promise<JudgeVerdict | null>}
@@ -133,9 +135,9 @@ export async function parseConcludeFromTrace(tracePath, runtime) {
133
135
  }
134
136
 
135
137
  /**
136
- * Return the `Conclude` tool input if the line carries a judge-source or
137
- * supervisor-source assistant message ending in a `Conclude` tool_use
138
- * block; null otherwise.
138
+ * Return the `Conclude` tool input when the line carries a judge-source or
139
+ * supervisor-source assistant message that ends in a `Conclude` tool_use
140
+ * block. Return null otherwise.
139
141
  * @param {string} line
140
142
  * @returns {{verdict: string, summary?: string} | null}
141
143
  */
@@ -176,12 +178,11 @@ function extractConcludeInput(line) {
176
178
  }
177
179
 
178
180
  /**
179
- * The Claude Agent SDK reports MCP tool names as
180
- * `mcp__<server>__<tool>` when the model invokes them — the orchestration
181
- * `Conclude` arrives as `mcp__orchestration__Conclude`. Pre-baked
182
- * supervisor traces (and the libharness-internal envelopes) sometimes carry
183
- * the bare `Conclude` name. Accept both forms so the parser is robust to
184
- * trace source.
181
+ * The Claude Agent SDK reports MCP tool names as `mcp__<server>__<tool>` when
182
+ * the model invokes them. The orchestration `Conclude` arrives as
183
+ * `mcp__orchestration__Conclude`. Pre-baked supervisor traces (and the
184
+ * libharness-internal envelopes) sometimes carry the bare `Conclude` name.
185
+ * Accept both forms, so the parser is robust to the trace source.
185
186
  */
186
187
  function isConcludeToolName(name) {
187
188
  if (typeof name !== "string") return false;
@@ -1,10 +1,11 @@
1
1
  /**
2
- * NpmInstaller — runs `bun install` in the family root when a package.json
3
- * is present, then copies the resulting `node_modules/` into the staging
4
- * directory so WorkdirManager can seed each per-task CWD.
2
+ * NpmInstaller — runs `bun install` in the family root when a package.json is
3
+ * present. It then copies the `node_modules/` that `bun install` produced
4
+ * into the staging directory, so WorkdirManager can seed each per-task CWD.
5
5
  *
6
- * Symmetric to ApmInstaller: the subprocess and filesystem flow through the
7
- * injected `runtime` bag (`runtime.subprocess.spawn` + `runtime.fs`).
6
+ * NpmInstaller is symmetric to ApmInstaller. The subprocess and the
7
+ * filesystem flow through the injected `runtime` bag
8
+ * (`runtime.subprocess.spawn` + `runtime.fs`).
8
9
  */
9
10
 
10
11
  import { join } from "node:path";
@@ -14,7 +15,7 @@ export class NpmInstaller {
14
15
  /**
15
16
  * @param {object} deps
16
17
  * @param {import("@forwardimpact/libutil/runtime").Runtime} deps.runtime -
17
- * Ambient collaborators; uses `subprocess.spawn` and `fs`.
18
+ * Ambient collaborators. This class uses `subprocess.spawn` and `fs`.
18
19
  */
19
20
  constructor({ runtime }) {
20
21
  if (!runtime) throw new Error("runtime is required");
@@ -23,7 +24,7 @@ export class NpmInstaller {
23
24
 
24
25
  /**
25
26
  * @param {import("./task-family.js").TaskFamily} family
26
- * @param {string} stagingDir - The staging directory (created by ApmInstaller).
27
+ * @param {string} stagingDir - The staging directory ApmInstaller creates.
27
28
  * @returns {Promise<void>}
28
29
  */
29
30
  async install(family, stagingDir) {
@@ -42,7 +43,7 @@ export class NpmInstaller {
42
43
  await fs.access(sourceModules);
43
44
  } catch {
44
45
  throw new Error(
45
- `bun install did not produce node_modules/ at ${sourceModules}; check the family's package.json`,
46
+ `bun install did not produce node_modules/ at ${sourceModules}. Check the family's package.json`,
46
47
  );
47
48
  }
48
49
 
@@ -1,15 +1,15 @@
1
1
  /**
2
- * ReportAggregator — read a run-output directory's `results.jsonl`, group
3
- * records by `taskId`, and compute pass@k via the OpenAI HumanEval
2
+ * ReportAggregator — read a run-output directory's `results.jsonl`. Group
3
+ * the records by `taskId`. Compute pass@k with the OpenAI HumanEval
4
4
  * unbiased estimator: `1 - C(n-c, k) / C(n, k)`.
5
5
  *
6
6
  * When `includeRuns` is true, each task carries per-run detail (invariant
7
- * checks, judge commentary, cost, duration) and the text renderer produces
7
+ * checks, judge commentary, cost, duration). The text renderer then produces
8
8
  * a full markdown report instead of just the pass@k table.
9
9
  *
10
- * Records that fail schema validation are skipped with a stderr warning
11
- * (counted under `totals.skipped`) so a corrupt line cannot abort the
12
- * whole report.
10
+ * The loader skips records that fail schema validation. It writes a stderr
11
+ * warning and counts each skip under `totals.skipped`. A corrupt line cannot
12
+ * abort the whole report.
13
13
  */
14
14
 
15
15
  import { join } from "node:path";
@@ -37,7 +37,7 @@ import { mergeRows } from "./grade.js";
37
37
  * @typedef {object} TaskReport
38
38
  * @property {string} taskId
39
39
  * @property {number} n - Total runs.
40
- * @property {number} c - Passing runs.
40
+ * @property {number} c - Runs that passed.
41
41
  * @property {Record<string|number, number|null>} passAtK
42
42
  * @property {RunDetail[]} [runs] - Per-run detail (only when includeRuns).
43
43
  */
@@ -109,12 +109,12 @@ export async function aggregate({
109
109
 
110
110
  /**
111
111
  * Attach `meanScore` and `scoreAtK` to a scored task group. A group is
112
- * scored iff any record carries an effective score; a score-less record in
113
- * a scored group (a preflight failure never reached grading, or a binary
114
- * run) contributes its verdict as the degenerate score — skipping it would
115
- * inflate the mean exactly when the agent fails hardest. Binary groups gain
116
- * neither field.
117
- * @param {object} task - Mutated.
112
+ * scored iff any record carries an effective score. A score-less record in a
113
+ * scored group contributes its verdict as the degenerate score. Such a record
114
+ * comes from a preflight failure that never reached the grade step, or from a
115
+ * binary run. A skip would inflate the mean exactly when the agent fails
116
+ * hardest. Binary groups gain neither field.
117
+ * @param {object} task - The function mutates it.
118
118
  * @param {object[]} group
119
119
  * @param {number[]} kValues
120
120
  */
@@ -127,9 +127,9 @@ function applyScoreFields(task, group, kValues) {
127
127
  }
128
128
 
129
129
  /**
130
- * Build a normalized per-run detail object and accumulate duration/turn
131
- * samples for median calculation. Extracted from `aggregate` to keep its
132
- * cognitive complexity below the lint ceiling.
130
+ * Build a normalized per-run detail object. Accumulate duration/turn samples
131
+ * so the caller can calculate the median. It is separate from `aggregate` to
132
+ * keep that function's cognitive complexity below the lint ceiling.
133
133
  * @param {object} r - Raw record.
134
134
  * @param {{allDurations: number[], allTurns: number[]}} acc
135
135
  * @returns {RunDetail}
@@ -155,9 +155,9 @@ function buildRunDetail(r, acc) {
155
155
 
156
156
  /**
157
157
  * Render an aggregate report as markdown. When the report contains per-run
158
- * detail (from `includeRuns: true`), renders a full report with summary,
159
- * pass@k table, and per-task detail sections. Otherwise falls back to the
160
- * compact pass@k table.
158
+ * detail (from `includeRuns: true`), the renderer produces a full report.
159
+ * That report has a summary, a pass@k table, and per-task detail sections.
160
+ * Otherwise the renderer falls back to the compact pass@k table.
161
161
  * @param {Awaited<ReturnType<typeof aggregate>>} report
162
162
  * @param {number[]} kValues
163
163
  * @returns {string}
@@ -170,10 +170,10 @@ export function renderTextReport(report, kValues) {
170
170
  }
171
171
 
172
172
  // ---------------------------------------------------------------------------
173
- // Compact report — status line + pass@k table, no per-task detail. Selected by
174
- // `report --detail=compact` (aggregate without `includeRuns`); the per-shard
175
- // summary uses it so a sharded run stays short while the merge job renders the
176
- // full report over the combined ledger.
173
+ // Compact report — status line + pass@k table, no per-task detail.
174
+ // `report --detail=compact` selects it (aggregate without `includeRuns`).
175
+ // The per-shard summary uses it so a sharded run stays short. The merge job
176
+ // renders the full report over the combined ledger.
177
177
  // ---------------------------------------------------------------------------
178
178
 
179
179
  function renderCompactReport(report, kValues) {
@@ -500,17 +500,18 @@ function median(arr) {
500
500
  // Record loading
501
501
  // ---------------------------------------------------------------------------
502
502
 
503
- // Directories never worth descending for a `results.jsonl`.
503
+ // The walk never descends into these directories for a `results.jsonl`.
504
504
  const SKIP_DIRS = new Set([".git", "node_modules"]);
505
505
 
506
506
  /**
507
507
  * Load and union every `results.jsonl` found recursively under `inputDir`.
508
508
  *
509
- * A single non-sharded run has one root-level ledger — the trivial one-match
510
- * case of the same walk. A sharded run lays each shard's partial ledger in its
511
- * own subdirectory; merging them equals reporting a single run over the same
512
- * cells. An *existing* dir with no ledger yields the empty union (exit 0); a
513
- * *missing* dir lets `readdir`'s ENOENT propagate so `report` still errors.
509
+ * A single non-sharded run has one root-level ledger. That is the trivial
510
+ * one-match case of the same walk. A sharded run lays each shard's partial
511
+ * ledger in its own subdirectory. A merge of those ledgers equals a report of
512
+ * a single run over the same cells. An *existing* dir with no ledger yields
513
+ * the empty union (exit 0). A *missing* dir lets `readdir`'s ENOENT
514
+ * propagate, so `report` still errors.
514
515
  * @param {string} inputDir
515
516
  * @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
516
517
  * @returns {Promise<{records: object[], skipped: number}>}
@@ -520,10 +521,10 @@ async function loadRecords(inputDir, runtime) {
520
521
  try {
521
522
  files = await collectResultsFiles(inputDir, runtime);
522
523
  } catch (e) {
523
- // Re-throw with the stack collapsed to the message line so the CLI's
524
- // error rendering stays free of node-internal async `readdir` frames
525
- // (a missing --input dir surfaces its ENOENT as exit 1, matching the
526
- // pre-1370 stream-error shape the golden captured).
524
+ // Re-throw with the stack collapsed to the message line. The CLI error
525
+ // output then stays free of node-internal async `readdir` frames. A
526
+ // missing --input dir surfaces its ENOENT as exit 1, which matches the
527
+ // pre-1370 stream-error shape the golden captured.
527
528
  const err = new Error(e.message);
528
529
  if (e.code) err.code = e.code;
529
530
  err.stack = `Error: ${e.message}`;
@@ -540,11 +541,12 @@ async function loadRecords(inputDir, runtime) {
540
541
  }
541
542
 
542
543
  /**
543
- * Parse one ledger's JSONL into `records`, skipping malformed or schema-invalid
544
- * lines with a stderr warning. Returns the skipped count. Extracted from
545
- * `loadRecords` to keep its cognitive complexity under the lint ceiling.
544
+ * Parse one ledger's JSONL into `records`. Skip each malformed or
545
+ * schema-invalid line and write a stderr warning. Return the skipped count.
546
+ * It is separate from `loadRecords` to keep that function's cognitive
547
+ * complexity under the lint ceiling.
546
548
  * @param {string} content
547
- * @param {object[]} records - Accumulator, appended in place.
549
+ * @param {object[]} records - Accumulator. The function appends in place.
548
550
  * @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
549
551
  * @returns {number} Skipped line count.
550
552
  */
@@ -567,7 +569,7 @@ function parseLedgerInto(content, records, runtime) {
567
569
  validateResultRecord(record);
568
570
  } catch (e) {
569
571
  runtime.proc.stderr.write(
570
- `benchmark report: skipped record failing schema — ${describeError(e)}\n`,
572
+ `benchmark report: skipped schema-invalid record — ${describeError(e)}\n`,
571
573
  );
572
574
  skipped++;
573
575
  continue;
@@ -578,10 +580,10 @@ function parseLedgerInto(content, records, runtime) {
578
580
  }
579
581
 
580
582
  /**
581
- * Recursively collect paths of every file named `results.jsonl` under `dir`,
582
- * skipping `.git`/`node_modules` and never following symlinks. A purpose-built
583
- * `readdir` walk — `task-family.js`'s private `walkFiles` resolves symlinks and
584
- * is unexported, which is the wrong contract here.
583
+ * Recursively collect paths of every file named `results.jsonl` under `dir`.
584
+ * Skip `.git` and `node_modules`. Never follow a symlink. This is a
585
+ * purpose-built `readdir` walk. `task-family.js`'s private `walkFiles`
586
+ * resolves symlinks and is unexported, which is the wrong contract here.
585
587
  * @param {string} dir
586
588
  * @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
587
589
  * @returns {Promise<string[]>}
@@ -604,10 +606,10 @@ async function collectResultsFiles(dir, runtime) {
604
606
  }
605
607
 
606
608
  /**
607
- * Warn (do not silently merge) when a `(taskId, runIndex)` cell appears more
608
- * than once across shard ledgers. The shard partition guarantees uniqueness, so
609
- * a duplicate signals misconfiguration; both copies stay in the group so the
610
- * count is honest.
609
+ * Warn when a `(taskId, runIndex)` cell appears more than once across shard
610
+ * ledgers. Do not silently merge the copies. The shard partition guarantees
611
+ * uniqueness, so a duplicate signals misconfiguration. Both copies stay in
612
+ * the group so the count is honest.
611
613
  * @param {object[]} records
612
614
  * @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
613
615
  */
@@ -620,7 +622,7 @@ function warnOnDuplicateCells(records, runtime) {
620
622
  for (const [key, n] of counts) {
621
623
  if (n > 1)
622
624
  runtime.proc.stderr.write(
623
- `benchmark report: duplicate cell ${key} appears ${n} times across shard ledgers — the shard partition should make each cell unique\n`,
625
+ `benchmark report: duplicate cell ${key} appears ${n} times across shard ledgers. The shard partition should make each cell unique\n`,
624
626
  );
625
627
  }
626
628
  }
@@ -660,14 +662,15 @@ function passAtKValue(n, c, k) {
660
662
 
661
663
  /**
662
664
  * score@k — the expected **maximum** score over k runs drawn without
663
- * replacement from the n recorded scores; the continuous analog of pass@k.
664
- * With scores sorted ascending s₍₁₎…s₍ₙ₎:
665
+ * replacement from the n recorded scores. It is the continuous analog of
666
+ * pass@k. With scores sorted ascending s₍₁₎…s₍ₙ₎:
665
667
  *
666
668
  * score@k = Σ_{i=k..n} s₍ᵢ₎ · C(i−1, k−1) / C(n, k)
667
669
  *
668
670
  * Each term weights s₍ᵢ₎ by the probability it is the k-subset's maximum.
669
671
  * Binary scores reduce exactly to the pass@k estimator (same BigInt binomial
670
- * helper); `k > n` yields the same `{error}` value — one idiom.
672
+ * helper). `k > n` yields the same `{error}` value, so both functions use
673
+ * one idiom.
671
674
  * @param {number[]} scores - Effective per-record scores.
672
675
  * @param {number} k
673
676
  * @returns {number | {error: string}}