@forwardimpact/libharness 3.0.0 → 3.0.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (69) hide show
  1. package/README.md +60 -57
  2. package/package.json +2 -2
  3. package/src/advisor.js +47 -41
  4. package/src/agent-runner.js +57 -47
  5. package/src/benchmark/apm-installer.js +28 -28
  6. package/src/benchmark/env-loader.js +24 -16
  7. package/src/benchmark/grade.js +44 -41
  8. package/src/benchmark/hidden-tests.js +71 -39
  9. package/src/benchmark/hook-env.js +11 -9
  10. package/src/benchmark/invariants.js +20 -17
  11. package/src/benchmark/judge.js +29 -28
  12. package/src/benchmark/npm-installer.js +9 -8
  13. package/src/benchmark/report.js +53 -50
  14. package/src/benchmark/result.js +24 -23
  15. package/src/benchmark/runner.js +75 -69
  16. package/src/benchmark/scheduler.js +17 -16
  17. package/src/benchmark/task-family.js +28 -26
  18. package/src/benchmark/trace-split.js +8 -7
  19. package/src/benchmark/workdir.js +27 -25
  20. package/src/claude-code-executable.js +11 -11
  21. package/src/commands/advisor-flags.js +8 -7
  22. package/src/commands/assert.js +16 -15
  23. package/src/commands/benchmark-definition.js +11 -11
  24. package/src/commands/benchmark-grade.js +12 -11
  25. package/src/commands/benchmark-report.js +5 -5
  26. package/src/commands/benchmark-run.js +31 -28
  27. package/src/commands/by-discussion.js +10 -10
  28. package/src/commands/callback.js +11 -11
  29. package/src/commands/discuss.js +8 -7
  30. package/src/commands/facilitate.js +15 -13
  31. package/src/commands/output.js +3 -2
  32. package/src/commands/run.js +14 -14
  33. package/src/commands/scan-logs.js +21 -19
  34. package/src/commands/selfedit.js +14 -14
  35. package/src/commands/supervise.js +11 -9
  36. package/src/commands/task-input.js +9 -9
  37. package/src/commands/tee.js +10 -9
  38. package/src/commands/trace.js +55 -42
  39. package/src/commands/work-tracker.js +4 -3
  40. package/src/cost.js +17 -17
  41. package/src/discuss-tools.js +16 -16
  42. package/src/discusser.js +39 -38
  43. package/src/events/github.js +54 -37
  44. package/src/facilitator.js +21 -21
  45. package/src/inbox-poller.js +4 -4
  46. package/src/judge.js +32 -30
  47. package/src/message-bus.js +12 -11
  48. package/src/orchestration-loop.js +35 -36
  49. package/src/orchestration-toolkit.js +58 -53
  50. package/src/orchestrator-helpers.js +2 -2
  51. package/src/profile-prompt.js +54 -53
  52. package/src/redaction.js +63 -57
  53. package/src/render/line-renderer.js +5 -5
  54. package/src/render/orchestrator-filter.js +3 -3
  55. package/src/render/palette.js +11 -9
  56. package/src/render/tool-hints.js +18 -15
  57. package/src/render/turn-renderer.js +4 -4
  58. package/src/reply-emitter.js +2 -2
  59. package/src/sequence-counter.js +4 -3
  60. package/src/signature-filter.js +7 -6
  61. package/src/supervisor.js +19 -18
  62. package/src/tee-writer.js +25 -25
  63. package/src/trace-collector.js +53 -48
  64. package/src/trace-github.js +53 -44
  65. package/src/trace-multi.js +15 -13
  66. package/src/trace-query.js +61 -52
  67. package/src/trace-render.js +18 -18
  68. package/src/trace-usage.js +31 -28
  69. package/src/transcript-recorder.js +24 -20
@@ -1,14 +1,15 @@
1
1
  /**
2
2
  * Invariants — runs `<task.paths.hooks>/invariants.sh` from the template path
3
- * against the post-run agent CWD. A pure collector with no verdict of its
4
- * own: structured per-check rows arrive on fd 3 (`$RESULTS_FD=3`) as NDJSON
5
- * and grading happens downstream over the merged rows. The exit code is
6
- * script health only — nonzero means the grader itself failed, never that a
7
- * check failed.
3
+ * against the post-run agent CWD. The module is a pure collector with no
4
+ * verdict of its own. Structured per-check rows arrive on fd 3
5
+ * (`$RESULTS_FD=3`) as NDJSON. Downstream code grades the merged rows. The
6
+ * exit code is script health only. A nonzero code means the grader itself
7
+ * failed. It never means that a check failed.
8
8
  *
9
- * Subprocess access flows through `runtime.subprocess.spawn`; the fd-3 backing
10
- * store and the stderr log use the sync filesystem surface (`runtime.fsSync`) —
11
- * the only surface this module touches, per design Decision 7.
9
+ * Subprocess access flows through `runtime.subprocess.spawn`. The fd-3
10
+ * backing store and the stderr log use the sync filesystem surface
11
+ * (`runtime.fsSync`). That surface is the only one this module touches, per
12
+ * design Decision 7.
12
13
  */
13
14
 
14
15
  import { join } from "node:path";
@@ -18,11 +19,12 @@ import { buildHookEnv } from "./hook-env.js";
18
19
  /**
19
20
  * @typedef {object} InvariantsResult
20
21
  * @property {Array<object>} details
21
- * @property {number} exitCode - Script health: nonzero means the hook itself
22
- * failed, never that a check failed.
22
+ * @property {number} exitCode - Script health. A nonzero code means the hook
23
+ * itself failed. It never means that a check failed.
23
24
  * @property {string} [stderr] - Trimmed script stderr, present only when the
24
- * script wrote to stderr. Surfaces hook failures (e.g. a missing tool) that
25
- * leave `details` empty, so they read distinctly from a real invariant miss.
25
+ * script wrote to stderr. This field surfaces hook failures (e.g. a missing
26
+ * tool) that leave `details` empty. A reader can then tell them apart from
27
+ * a real invariant miss.
26
28
  */
27
29
 
28
30
  /**
@@ -43,8 +45,8 @@ export async function runInvariants(task, ctx, runtime) {
43
45
 
44
46
  // Bun's child_process pipe setup for fd >= 3 is racy under load (it
45
47
  // creates a unix socket pair and the connect() can return ENOENT). Use
46
- // a temp file as the fd-3 backing store instead — the script still
47
- // writes via `$RESULTS_FD`, but we hand it a real file descriptor.
48
+ // a temp file as the fd-3 backing store instead. The script still writes
49
+ // through `$RESULTS_FD`, but we hand it a real file descriptor.
48
50
  const fd3Path = join(ctx.runDir, "invariants.fd3.ndjson");
49
51
  const fd3File = fsSync.openSync(fd3Path, "w+");
50
52
 
@@ -69,7 +71,8 @@ export async function runInvariants(task, ctx, runtime) {
69
71
  throw e;
70
72
  }
71
73
 
72
- // Drain stdout (do not require consumers to read it); capture stderr to log.
74
+ // Drain stdout (do not require consumers to read it). Capture stderr to
75
+ // the log.
73
76
  const drainStdout = (async () => {
74
77
  for await (const _chunk of child.stdout) {
75
78
  // discard
@@ -127,8 +130,8 @@ function readAndUnlink(fsSync, path) {
127
130
  }
128
131
 
129
132
  /**
130
- * Parse the fd-3 buffer (read from the temp-file backing) into one NDJSON
131
- * row per detail entry.
133
+ * Parse the fd-3 buffer (read from the temp-file backing store) into one
134
+ * NDJSON row per detail entry.
132
135
  */
133
136
  function parseFd3Buffer(buf, details) {
134
137
  if (!buf) return;
@@ -1,8 +1,9 @@
1
1
  /**
2
- * Benchmark adapter for the libharness `Judge`. Templates the family's
3
- * `judge.task.md` with structured context variables, runs the judge against
4
- * the post-run agent CWD, and returns the verdict in the benchmark's
5
- * `pass`/`fail` vocabulary (mapped from libharness's `success`/`failure`).
2
+ * Benchmark adapter for the libharness `Judge`. The adapter fills the
3
+ * family's `judge.task.md` template with structured context variables. It
4
+ * runs the judge against the post-run agent CWD. It returns the verdict in
5
+ * the benchmark's `pass`/`fail` vocabulary (mapped from libharness's
6
+ * `success`/`failure`).
6
7
  *
7
8
  * Template variables available in `judge.task.md`:
8
9
  *
@@ -12,13 +13,13 @@
12
13
  * {{GRADE_RESULT}} — JSON grade object plus the merged check rows
13
14
  * {{SKILL_SET_HASH}} — SHA-256 from apm.lock.yaml
14
15
  * {{TASK_ID}} — task name (directory under tasks/)
15
- * {{TASK_DIR}} — agent working directory path
16
+ * {{TASK_DIR}} — path to the agent working directory
16
17
  *
17
- * The judge verdict is captured from the orchestration context's
18
- * `concluded` flag directly — no trace parsing on the happy path.
19
- * `parseConcludeFromTrace` is preserved for offline analysis and as a
20
- * fallback when the runtime ctx isn't available (e.g. re-grading a
21
- * historical run from its judge.ndjson file).
18
+ * The adapter reads the judge verdict directly from the orchestration
19
+ * context's `concluded` flag. It does not parse the trace on the happy path.
20
+ * `parseConcludeFromTrace` stays for offline analysis. It is also a fallback
21
+ * when the runtime ctx is not available, for example when you re-grade a
22
+ * historical run from its judge.ndjson file.
22
23
  */
23
24
 
24
25
  import { createJudge } from "../judge.js";
@@ -41,8 +42,8 @@ import { sumTraceCost } from "../cost.js";
41
42
 
42
43
  /**
43
44
  * Run the judge over a completed task run. The judge is a binary gate over
44
- * the grade's validity, never a grade itself: `gradeResult` reaches the
45
- * template as evidence, and the verdict stays pass/fail.
45
+ * the grade's validity. The judge is never a grade itself. `gradeResult`
46
+ * reaches the template as evidence. The verdict stays pass/fail.
46
47
  * @param {import("./task-family.js").Task} task
47
48
  * @param {import("./workdir.js").Workdir} workdir
48
49
  * @param {{verdict: string, gatesPass: boolean, score?: number, malformed?: number, rows: unknown[]}} gradeResult -
@@ -86,9 +87,9 @@ export async function runJudge(task, workdir, gradeResult, deps, context) {
86
87
  await new Promise((r) => output.end(r));
87
88
  }
88
89
 
89
- // The judge is its own SDK session; its spend lands in the judge trace we
90
- // just wrote, not in the supervisor's combined trace. Read it back so the
91
- // benchmark record's cost includes the judge.
90
+ // The judge is its own SDK session. Its spend lands in the judge trace we
91
+ // just wrote. It does not land in the supervisor's combined trace. Read the
92
+ // trace back, so the benchmark record's cost includes the judge.
92
93
  const judgeTrace = await fs
93
94
  .readFile(workdir.judgeTracePath, "utf8")
94
95
  .catch(() => "");
@@ -109,10 +110,11 @@ export async function runJudge(task, workdir, gradeResult, deps, context) {
109
110
  }
110
111
 
111
112
  /**
112
- * Parse the last judge-source (or supervisor-source, for backward compat
113
- * with pre-Judge-class traces) `Conclude` tool call from an NDJSON trace
114
- * and map the verdict (`success → pass`, `failure → fail`). Preserved for
115
- * offline analysis; not used on the runtime happy path.
113
+ * Parse the last `Conclude` tool call from an NDJSON trace and map the
114
+ * verdict (`success → pass`, `failure → fail`). The call may carry a
115
+ * judge source or a supervisor source. The supervisor source keeps backward
116
+ * compatibility with pre-Judge-class traces. This function stays for offline
117
+ * analysis. The runtime happy path does not use it.
116
118
  * @param {string} tracePath
117
119
  * @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
118
120
  * @returns {Promise<JudgeVerdict | null>}
@@ -133,9 +135,9 @@ export async function parseConcludeFromTrace(tracePath, runtime) {
133
135
  }
134
136
 
135
137
  /**
136
- * Return the `Conclude` tool input if the line carries a judge-source or
137
- * supervisor-source assistant message ending in a `Conclude` tool_use
138
- * block; null otherwise.
138
+ * Return the `Conclude` tool input when the line carries a judge-source or
139
+ * supervisor-source assistant message that ends in a `Conclude` tool_use
140
+ * block. Return null otherwise.
139
141
  * @param {string} line
140
142
  * @returns {{verdict: string, summary?: string} | null}
141
143
  */
@@ -176,12 +178,11 @@ function extractConcludeInput(line) {
176
178
  }
177
179
 
178
180
  /**
179
- * The Claude Agent SDK reports MCP tool names as
180
- * `mcp__<server>__<tool>` when the model invokes them — the orchestration
181
- * `Conclude` arrives as `mcp__orchestration__Conclude`. Pre-baked
182
- * supervisor traces (and the libharness-internal envelopes) sometimes carry
183
- * the bare `Conclude` name. Accept both forms so the parser is robust to
184
- * trace source.
181
+ * The Claude Agent SDK reports MCP tool names as `mcp__<server>__<tool>` when
182
+ * the model invokes them. The orchestration `Conclude` arrives as
183
+ * `mcp__orchestration__Conclude`. Pre-baked supervisor traces (and the
184
+ * libharness-internal envelopes) sometimes carry the bare `Conclude` name.
185
+ * Accept both forms, so the parser is robust to the trace source.
185
186
  */
186
187
  function isConcludeToolName(name) {
187
188
  if (typeof name !== "string") return false;
@@ -1,10 +1,11 @@
1
1
  /**
2
- * NpmInstaller — runs `bun install` in the family root when a package.json
3
- * is present, then copies the resulting `node_modules/` into the staging
4
- * directory so WorkdirManager can seed each per-task CWD.
2
+ * NpmInstaller — runs `bun install` in the family root when a package.json is
3
+ * present. It then copies the `node_modules/` that `bun install` produced
4
+ * into the staging directory, so WorkdirManager can seed each per-task CWD.
5
5
  *
6
- * Symmetric to ApmInstaller: the subprocess and filesystem flow through the
7
- * injected `runtime` bag (`runtime.subprocess.spawn` + `runtime.fs`).
6
+ * NpmInstaller is symmetric to ApmInstaller. The subprocess and the
7
+ * filesystem flow through the injected `runtime` bag
8
+ * (`runtime.subprocess.spawn` + `runtime.fs`).
8
9
  */
9
10
 
10
11
  import { join } from "node:path";
@@ -14,7 +15,7 @@ export class NpmInstaller {
14
15
  /**
15
16
  * @param {object} deps
16
17
  * @param {import("@forwardimpact/libutil/runtime").Runtime} deps.runtime -
17
- * Ambient collaborators; uses `subprocess.spawn` and `fs`.
18
+ * Ambient collaborators. This class uses `subprocess.spawn` and `fs`.
18
19
  */
19
20
  constructor({ runtime }) {
20
21
  if (!runtime) throw new Error("runtime is required");
@@ -23,7 +24,7 @@ export class NpmInstaller {
23
24
 
24
25
  /**
25
26
  * @param {import("./task-family.js").TaskFamily} family
26
- * @param {string} stagingDir - The staging directory (created by ApmInstaller).
27
+ * @param {string} stagingDir - The staging directory ApmInstaller creates.
27
28
  * @returns {Promise<void>}
28
29
  */
29
30
  async install(family, stagingDir) {
@@ -42,7 +43,7 @@ export class NpmInstaller {
42
43
  await fs.access(sourceModules);
43
44
  } catch {
44
45
  throw new Error(
45
- `bun install did not produce node_modules/ at ${sourceModules}; check the family's package.json`,
46
+ `bun install did not produce node_modules/ at ${sourceModules}. Check the family's package.json`,
46
47
  );
47
48
  }
48
49
 
@@ -1,15 +1,15 @@
1
1
  /**
2
- * ReportAggregator — read a run-output directory's `results.jsonl`, group
3
- * records by `taskId`, and compute pass@k via the OpenAI HumanEval
2
+ * ReportAggregator — read a run-output directory's `results.jsonl`. Group
3
+ * the records by `taskId`. Compute pass@k with the OpenAI HumanEval
4
4
  * unbiased estimator: `1 - C(n-c, k) / C(n, k)`.
5
5
  *
6
6
  * When `includeRuns` is true, each task carries per-run detail (invariant
7
- * checks, judge commentary, cost, duration) and the text renderer produces
7
+ * checks, judge commentary, cost, duration). The text renderer then produces
8
8
  * a full markdown report instead of just the pass@k table.
9
9
  *
10
- * Records that fail schema validation are skipped with a stderr warning
11
- * (counted under `totals.skipped`) so a corrupt line cannot abort the
12
- * whole report.
10
+ * The loader skips records that fail schema validation. It writes a stderr
11
+ * warning and counts each skip under `totals.skipped`. A corrupt line cannot
12
+ * abort the whole report.
13
13
  */
14
14
 
15
15
  import { join } from "node:path";
@@ -37,7 +37,7 @@ import { mergeRows } from "./grade.js";
37
37
  * @typedef {object} TaskReport
38
38
  * @property {string} taskId
39
39
  * @property {number} n - Total runs.
40
- * @property {number} c - Passing runs.
40
+ * @property {number} c - Runs that passed.
41
41
  * @property {Record<string|number, number|null>} passAtK
42
42
  * @property {RunDetail[]} [runs] - Per-run detail (only when includeRuns).
43
43
  */
@@ -109,12 +109,12 @@ export async function aggregate({
109
109
 
110
110
  /**
111
111
  * Attach `meanScore` and `scoreAtK` to a scored task group. A group is
112
- * scored iff any record carries an effective score; a score-less record in
113
- * a scored group (a preflight failure never reached grading, or a binary
114
- * run) contributes its verdict as the degenerate score — skipping it would
115
- * inflate the mean exactly when the agent fails hardest. Binary groups gain
116
- * neither field.
117
- * @param {object} task - Mutated.
112
+ * scored iff any record carries an effective score. A score-less record in a
113
+ * scored group contributes its verdict as the degenerate score. Such a record
114
+ * comes from a preflight failure that never reached the grade step, or from a
115
+ * binary run. A skip would inflate the mean exactly when the agent fails
116
+ * hardest. Binary groups gain neither field.
117
+ * @param {object} task - The function mutates it.
118
118
  * @param {object[]} group
119
119
  * @param {number[]} kValues
120
120
  */
@@ -127,9 +127,9 @@ function applyScoreFields(task, group, kValues) {
127
127
  }
128
128
 
129
129
  /**
130
- * Build a normalized per-run detail object and accumulate duration/turn
131
- * samples for median calculation. Extracted from `aggregate` to keep its
132
- * cognitive complexity below the lint ceiling.
130
+ * Build a normalized per-run detail object. Accumulate duration/turn samples
131
+ * so the caller can calculate the median. It is separate from `aggregate` to
132
+ * keep that function's cognitive complexity below the lint ceiling.
133
133
  * @param {object} r - Raw record.
134
134
  * @param {{allDurations: number[], allTurns: number[]}} acc
135
135
  * @returns {RunDetail}
@@ -155,9 +155,9 @@ function buildRunDetail(r, acc) {
155
155
 
156
156
  /**
157
157
  * Render an aggregate report as markdown. When the report contains per-run
158
- * detail (from `includeRuns: true`), renders a full report with summary,
159
- * pass@k table, and per-task detail sections. Otherwise falls back to the
160
- * compact pass@k table.
158
+ * detail (from `includeRuns: true`), the renderer produces a full report.
159
+ * That report has a summary, a pass@k table, and per-task detail sections.
160
+ * Otherwise the renderer falls back to the compact pass@k table.
161
161
  * @param {Awaited<ReturnType<typeof aggregate>>} report
162
162
  * @param {number[]} kValues
163
163
  * @returns {string}
@@ -170,10 +170,10 @@ export function renderTextReport(report, kValues) {
170
170
  }
171
171
 
172
172
  // ---------------------------------------------------------------------------
173
- // Compact report — status line + pass@k table, no per-task detail. Selected by
174
- // `report --detail=compact` (aggregate without `includeRuns`); the per-shard
175
- // summary uses it so a sharded run stays short while the merge job renders the
176
- // full report over the combined ledger.
173
+ // Compact report — status line + pass@k table, no per-task detail.
174
+ // `report --detail=compact` selects it (aggregate without `includeRuns`).
175
+ // The per-shard summary uses it so a sharded run stays short. The merge job
176
+ // renders the full report over the combined ledger.
177
177
  // ---------------------------------------------------------------------------
178
178
 
179
179
  function renderCompactReport(report, kValues) {
@@ -500,17 +500,18 @@ function median(arr) {
500
500
  // Record loading
501
501
  // ---------------------------------------------------------------------------
502
502
 
503
- // Directories never worth descending for a `results.jsonl`.
503
+ // The walk never descends into these directories for a `results.jsonl`.
504
504
  const SKIP_DIRS = new Set([".git", "node_modules"]);
505
505
 
506
506
  /**
507
507
  * Load and union every `results.jsonl` found recursively under `inputDir`.
508
508
  *
509
- * A single non-sharded run has one root-level ledger — the trivial one-match
510
- * case of the same walk. A sharded run lays each shard's partial ledger in its
511
- * own subdirectory; merging them equals reporting a single run over the same
512
- * cells. An *existing* dir with no ledger yields the empty union (exit 0); a
513
- * *missing* dir lets `readdir`'s ENOENT propagate so `report` still errors.
509
+ * A single non-sharded run has one root-level ledger. That is the trivial
510
+ * one-match case of the same walk. A sharded run lays each shard's partial
511
+ * ledger in its own subdirectory. A merge of those ledgers equals a report of
512
+ * a single run over the same cells. An *existing* dir with no ledger yields
513
+ * the empty union (exit 0). A *missing* dir lets `readdir`'s ENOENT
514
+ * propagate, so `report` still errors.
514
515
  * @param {string} inputDir
515
516
  * @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
516
517
  * @returns {Promise<{records: object[], skipped: number}>}
@@ -520,10 +521,10 @@ async function loadRecords(inputDir, runtime) {
520
521
  try {
521
522
  files = await collectResultsFiles(inputDir, runtime);
522
523
  } catch (e) {
523
- // Re-throw with the stack collapsed to the message line so the CLI's
524
- // error rendering stays free of node-internal async `readdir` frames
525
- // (a missing --input dir surfaces its ENOENT as exit 1, matching the
526
- // pre-1370 stream-error shape the golden captured).
524
+ // Re-throw with the stack collapsed to the message line. The CLI error
525
+ // output then stays free of node-internal async `readdir` frames. A
526
+ // missing --input dir surfaces its ENOENT as exit 1, which matches the
527
+ // pre-1370 stream-error shape the golden captured.
527
528
  const err = new Error(e.message);
528
529
  if (e.code) err.code = e.code;
529
530
  err.stack = `Error: ${e.message}`;
@@ -540,11 +541,12 @@ async function loadRecords(inputDir, runtime) {
540
541
  }
541
542
 
542
543
  /**
543
- * Parse one ledger's JSONL into `records`, skipping malformed or schema-invalid
544
- * lines with a stderr warning. Returns the skipped count. Extracted from
545
- * `loadRecords` to keep its cognitive complexity under the lint ceiling.
544
+ * Parse one ledger's JSONL into `records`. Skip each malformed or
545
+ * schema-invalid line and write a stderr warning. Return the skipped count.
546
+ * It is separate from `loadRecords` to keep that function's cognitive
547
+ * complexity under the lint ceiling.
546
548
  * @param {string} content
547
- * @param {object[]} records - Accumulator, appended in place.
549
+ * @param {object[]} records - Accumulator. The function appends in place.
548
550
  * @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
549
551
  * @returns {number} Skipped line count.
550
552
  */
@@ -567,7 +569,7 @@ function parseLedgerInto(content, records, runtime) {
567
569
  validateResultRecord(record);
568
570
  } catch (e) {
569
571
  runtime.proc.stderr.write(
570
- `benchmark report: skipped record failing schema — ${describeError(e)}\n`,
572
+ `benchmark report: skipped schema-invalid record — ${describeError(e)}\n`,
571
573
  );
572
574
  skipped++;
573
575
  continue;
@@ -578,10 +580,10 @@ function parseLedgerInto(content, records, runtime) {
578
580
  }
579
581
 
580
582
  /**
581
- * Recursively collect paths of every file named `results.jsonl` under `dir`,
582
- * skipping `.git`/`node_modules` and never following symlinks. A purpose-built
583
- * `readdir` walk — `task-family.js`'s private `walkFiles` resolves symlinks and
584
- * is unexported, which is the wrong contract here.
583
+ * Recursively collect paths of every file named `results.jsonl` under `dir`.
584
+ * Skip `.git` and `node_modules`. Never follow a symlink. This is a
585
+ * purpose-built `readdir` walk. `task-family.js`'s private `walkFiles`
586
+ * resolves symlinks and is unexported, which is the wrong contract here.
585
587
  * @param {string} dir
586
588
  * @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
587
589
  * @returns {Promise<string[]>}
@@ -604,10 +606,10 @@ async function collectResultsFiles(dir, runtime) {
604
606
  }
605
607
 
606
608
  /**
607
- * Warn (do not silently merge) when a `(taskId, runIndex)` cell appears more
608
- * than once across shard ledgers. The shard partition guarantees uniqueness, so
609
- * a duplicate signals misconfiguration; both copies stay in the group so the
610
- * count is honest.
609
+ * Warn when a `(taskId, runIndex)` cell appears more than once across shard
610
+ * ledgers. Do not silently merge the copies. The shard partition guarantees
611
+ * uniqueness, so a duplicate signals misconfiguration. Both copies stay in
612
+ * the group so the count is honest.
611
613
  * @param {object[]} records
612
614
  * @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
613
615
  */
@@ -620,7 +622,7 @@ function warnOnDuplicateCells(records, runtime) {
620
622
  for (const [key, n] of counts) {
621
623
  if (n > 1)
622
624
  runtime.proc.stderr.write(
623
- `benchmark report: duplicate cell ${key} appears ${n} times across shard ledgers — the shard partition should make each cell unique\n`,
625
+ `benchmark report: duplicate cell ${key} appears ${n} times across shard ledgers. The shard partition should make each cell unique\n`,
624
626
  );
625
627
  }
626
628
  }
@@ -660,14 +662,15 @@ function passAtKValue(n, c, k) {
660
662
 
661
663
  /**
662
664
  * score@k — the expected **maximum** score over k runs drawn without
663
- * replacement from the n recorded scores; the continuous analog of pass@k.
664
- * With scores sorted ascending s₍₁₎…s₍ₙ₎:
665
+ * replacement from the n recorded scores. It is the continuous analog of
666
+ * pass@k. With scores sorted ascending s₍₁₎…s₍ₙ₎:
665
667
  *
666
668
  * score@k = Σ_{i=k..n} s₍ᵢ₎ · C(i−1, k−1) / C(n, k)
667
669
  *
668
670
  * Each term weights s₍ᵢ₎ by the probability it is the k-subset's maximum.
669
671
  * Binary scores reduce exactly to the pass@k estimator (same BigInt binomial
670
- * helper); `k > n` yields the same `{error}` value — one idiom.
672
+ * helper). `k > n` yields the same `{error}` value, so both functions use
673
+ * one idiom.
671
674
  * @param {number[]} scores - Effective per-record scores.
672
675
  * @param {number} k
673
676
  * @returns {number | {error: string}}
@@ -3,16 +3,17 @@
3
3
  *
4
4
  * Two schemas live here:
5
5
  * - RESULT_RECORD_SCHEMA — one record per (task, runIndex) from a full
6
- * benchmark run. Has a happy branch (grade + collectors + judge present)
7
- * and a pre-flight-failure branch (grade/judgeVerdict/submission absent).
8
- * - GRADE_RECORD_SCHEMA — narrower output of `benchmark-grade`: ad-hoc
9
- * grading without a full lifecycle.
6
+ * benchmark run. It has a happy branch (grade + collectors + judge
7
+ * present) and a pre-flight-failure branch (grade/judgeVerdict/submission
8
+ * absent).
9
+ * - GRADE_RECORD_SCHEMA — narrower output of `benchmark-grade`, which
10
+ * grades ad hoc without a full lifecycle.
10
11
  *
11
- * The check rows are the authoritative grading channel: the happy branch
12
- * requires a `grade` object, so a pre-break record fails validation rather
13
- * than rendering under semantics it never carried.
12
+ * The check rows are the authoritative channel for grades. The happy branch
13
+ * requires a `grade` object, so a pre-break record fails validation. It does
14
+ * not render under semantics it never carried.
14
15
  *
15
- * Validation is throw-on-mismatch so the runner can wrap every JSONL append
16
+ * The validators throw on mismatch, so the runner can wrap every JSONL append
16
17
  * in a guard and reject schema drift at write time.
17
18
  */
18
19
 
@@ -27,8 +28,8 @@ const INVARIANTS_SHAPE = z.object({
27
28
  });
28
29
 
29
30
  /**
30
- * The normalized grading projection: `score` appears only on scored tasks,
31
- * `malformed` only when at least one row was malformed.
31
+ * The normalized projection of a grade. `score` appears only on scored
32
+ * tasks. `malformed` appears only when at least one row was malformed.
32
33
  */
33
34
  const GRADE_SHAPE = z.object({
34
35
  verdict: VERDICT_ENUM,
@@ -48,9 +49,9 @@ const JUDGE_VERDICT_SHAPE = z.object({
48
49
  });
49
50
 
50
51
  /**
51
- * Per-participant cost attribution. `costUsd` is the sum of these; the
52
+ * Per-participant cost attribution. `costUsd` is the sum of these. The
52
53
  * breakdown lets reports show where the spend went. The judge runs as its
53
- * own SDK session, so its cost is tracked separately from agent/supervisor.
54
+ * own SDK session, so its cost stays separate from agent/supervisor.
54
55
  */
55
56
  const COST_BREAKDOWN_SHAPE = z.object({
56
57
  agent: z.number(),
@@ -98,8 +99,8 @@ const HAPPY_RECORD = z.object({
98
99
  invariants: INVARIANTS_SHAPE,
99
100
  grade: GRADE_SHAPE,
100
101
  hiddenTests: HIDDEN_TESTS_SHAPE.optional(),
101
- // The effective, judge-zeroed score `report` aggregates — present only on
102
- // scored tasks.
102
+ // The effective, judge-zeroed score `report` aggregates. It is present
103
+ // only on scored tasks.
103
104
  score: z.number().min(0).max(1).optional(),
104
105
  submission: z.string(),
105
106
  judgeVerdict: JUDGE_VERDICT_SHAPE.optional(),
@@ -114,9 +115,9 @@ const PREFLIGHT_RECORD = z.object({
114
115
  ...COMMON_FIELDS,
115
116
  costUsd: z.literal(0),
116
117
  preflightError: PREFLIGHT_ERROR_SHAPE,
117
- // Trace paths are populated even on preflight failure (the runner allocates
118
- // them in WorkdirManager.start) so the record is uniform across branches
119
- // and downstream consumers can reference them without conditional fields.
118
+ // The runner allocates the trace paths in WorkdirManager.start, even on
119
+ // preflight failure. The record then stays uniform across branches, and
120
+ // downstream consumers can reference the paths without conditional fields.
120
121
  agentTracePath: z.string(),
121
122
  supervisorTracePath: z.string(),
122
123
  judgeTracePath: z.string(),
@@ -133,15 +134,15 @@ export const RESULT_RECORD_SCHEMA = z.union([HAPPY_RECORD, PREFLIGHT_RECORD]);
133
134
 
134
135
  export const GRADE_RECORD_SCHEMA = z.object({
135
136
  taskId: z.string().min(1),
136
- // Unlike the happy result record — where `grade.score` is the raw
137
- // weighted fraction and the effective (zeroed) value lives on the
138
- // top-level `score` — this record has no second score field, so its
139
- // `grade.score` carries the effective health/gate-zeroed value.
137
+ // In the happy result record, `grade.score` is the raw weighted fraction,
138
+ // and the effective (zeroed) value lives on the top-level `score`. This
139
+ // record has no second score field, so its `grade.score` carries the
140
+ // effective health/gate-zeroed value.
140
141
  grade: GRADE_SHAPE,
141
142
  invariants: INVARIANTS_SHAPE,
142
143
  hiddenTests: HIDDEN_TESTS_SHAPE.optional(),
143
- // Mirrors the invariants script's exit for diagnosis; the graded verdict
144
- // is what drives the command's process exit.
144
+ // This mirrors the invariants script's exit for diagnosis. The graded
145
+ // verdict drives the command's process exit.
145
146
  exitCode: z.number().int(),
146
147
  });
147
148