@forwardimpact/libharness 2.0.0 → 3.0.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (73) hide show
  1. package/README.md +68 -65
  2. package/package.json +15 -13
  3. package/src/advisor.js +47 -41
  4. package/src/agent-runner.js +58 -48
  5. package/src/benchmark/apm-installer.js +28 -28
  6. package/src/benchmark/env-loader.js +24 -16
  7. package/src/benchmark/grade.js +44 -41
  8. package/src/benchmark/hidden-tests.js +25 -24
  9. package/src/benchmark/hook-env.js +11 -9
  10. package/src/benchmark/invariants.js +20 -17
  11. package/src/benchmark/judge.js +29 -28
  12. package/src/benchmark/npm-installer.js +9 -8
  13. package/src/benchmark/report.js +53 -50
  14. package/src/benchmark/result.js +24 -23
  15. package/src/benchmark/runner.js +75 -69
  16. package/src/benchmark/scheduler.js +17 -16
  17. package/src/benchmark/task-family.js +29 -27
  18. package/src/benchmark/trace-split.js +9 -8
  19. package/src/benchmark/workdir.js +27 -25
  20. package/src/claude-code-executable.js +11 -11
  21. package/src/commands/advisor-flags.js +8 -7
  22. package/src/commands/assert.js +16 -15
  23. package/src/commands/benchmark-definition.js +20 -20
  24. package/src/commands/benchmark-grade.js +13 -12
  25. package/src/commands/benchmark-report.js +5 -5
  26. package/src/commands/benchmark-run.js +31 -28
  27. package/src/commands/by-discussion.js +11 -11
  28. package/src/commands/callback.js +11 -11
  29. package/src/commands/discuss.js +8 -7
  30. package/src/commands/facilitate.js +16 -14
  31. package/src/commands/output.js +4 -3
  32. package/src/commands/run.js +15 -15
  33. package/src/commands/scan-logs.js +22 -20
  34. package/src/commands/selfedit.js +124 -0
  35. package/src/commands/supervise.js +13 -11
  36. package/src/commands/task-input.js +9 -9
  37. package/src/commands/tee.js +11 -10
  38. package/src/commands/trace.js +55 -42
  39. package/src/commands/work-tracker.js +4 -3
  40. package/src/cost.js +17 -17
  41. package/src/discuss-tools.js +16 -16
  42. package/src/discusser.js +39 -38
  43. package/src/events/github.js +54 -37
  44. package/src/facilitator.js +21 -21
  45. package/src/inbox-poller.js +4 -4
  46. package/src/judge.js +32 -30
  47. package/src/message-bus.js +12 -11
  48. package/src/orchestration-loop.js +35 -36
  49. package/src/orchestration-toolkit.js +58 -53
  50. package/src/orchestrator-helpers.js +2 -2
  51. package/src/profile-prompt.js +54 -53
  52. package/src/redaction.js +63 -57
  53. package/src/render/line-renderer.js +5 -5
  54. package/src/render/orchestrator-filter.js +3 -3
  55. package/src/render/palette.js +11 -9
  56. package/src/render/tool-hints.js +18 -15
  57. package/src/render/turn-renderer.js +4 -4
  58. package/src/reply-emitter.js +2 -2
  59. package/src/sequence-counter.js +4 -3
  60. package/src/signature-filter.js +7 -6
  61. package/src/supervisor.js +19 -18
  62. package/src/tee-writer.js +25 -25
  63. package/src/trace-collector.js +53 -48
  64. package/src/trace-github.js +53 -44
  65. package/src/trace-multi.js +16 -14
  66. package/src/trace-query.js +61 -52
  67. package/src/trace-render.js +19 -19
  68. package/src/trace-usage.js +31 -28
  69. package/src/transcript-recorder.js +24 -20
  70. package/bin/fit-benchmark.js +0 -44
  71. package/bin/fit-harness.js +0 -412
  72. package/bin/fit-selfedit.js +0 -165
  73. package/bin/fit-trace.js +0 -520
@@ -3,16 +3,17 @@
3
3
  *
4
4
  * Two schemas live here:
5
5
  * - RESULT_RECORD_SCHEMA — one record per (task, runIndex) from a full
6
- * benchmark run. Has a happy branch (grade + collectors + judge present)
7
- * and a pre-flight-failure branch (grade/judgeVerdict/submission absent).
8
- * - GRADE_RECORD_SCHEMA — narrower output of `benchmark-grade`: ad-hoc
9
- * grading without a full lifecycle.
6
+ * benchmark run. It has a happy branch (grade + collectors + judge
7
+ * present) and a pre-flight-failure branch (grade/judgeVerdict/submission
8
+ * absent).
9
+ * - GRADE_RECORD_SCHEMA — narrower output of `benchmark-grade`, which
10
+ * grades ad hoc without a full lifecycle.
10
11
  *
11
- * The check rows are the authoritative grading channel: the happy branch
12
- * requires a `grade` object, so a pre-break record fails validation rather
13
- * than rendering under semantics it never carried.
12
+ * The check rows are the authoritative channel for grades. The happy branch
13
+ * requires a `grade` object, so a pre-break record fails validation. It does
14
+ * not render under semantics it never carried.
14
15
  *
15
- * Validation is throw-on-mismatch so the runner can wrap every JSONL append
16
+ * The validators throw on mismatch, so the runner can wrap every JSONL append
16
17
  * in a guard and reject schema drift at write time.
17
18
  */
18
19
 
@@ -27,8 +28,8 @@ const INVARIANTS_SHAPE = z.object({
27
28
  });
28
29
 
29
30
  /**
30
- * The normalized grading projection: `score` appears only on scored tasks,
31
- * `malformed` only when at least one row was malformed.
31
+ * The normalized projection of a grade. `score` appears only on scored
32
+ * tasks. `malformed` appears only when at least one row was malformed.
32
33
  */
33
34
  const GRADE_SHAPE = z.object({
34
35
  verdict: VERDICT_ENUM,
@@ -48,9 +49,9 @@ const JUDGE_VERDICT_SHAPE = z.object({
48
49
  });
49
50
 
50
51
  /**
51
- * Per-participant cost attribution. `costUsd` is the sum of these; the
52
+ * Per-participant cost attribution. `costUsd` is the sum of these. The
52
53
  * breakdown lets reports show where the spend went. The judge runs as its
53
- * own SDK session, so its cost is tracked separately from agent/supervisor.
54
+ * own SDK session, so its cost stays separate from agent/supervisor.
54
55
  */
55
56
  const COST_BREAKDOWN_SHAPE = z.object({
56
57
  agent: z.number(),
@@ -98,8 +99,8 @@ const HAPPY_RECORD = z.object({
98
99
  invariants: INVARIANTS_SHAPE,
99
100
  grade: GRADE_SHAPE,
100
101
  hiddenTests: HIDDEN_TESTS_SHAPE.optional(),
101
- // The effective, judge-zeroed score `report` aggregates — present only on
102
- // scored tasks.
102
+ // The effective, judge-zeroed score `report` aggregates. It is present
103
+ // only on scored tasks.
103
104
  score: z.number().min(0).max(1).optional(),
104
105
  submission: z.string(),
105
106
  judgeVerdict: JUDGE_VERDICT_SHAPE.optional(),
@@ -114,9 +115,9 @@ const PREFLIGHT_RECORD = z.object({
114
115
  ...COMMON_FIELDS,
115
116
  costUsd: z.literal(0),
116
117
  preflightError: PREFLIGHT_ERROR_SHAPE,
117
- // Trace paths are populated even on preflight failure (the runner allocates
118
- // them in WorkdirManager.start) so the record is uniform across branches
119
- // and downstream consumers can reference them without conditional fields.
118
+ // The runner allocates the trace paths in WorkdirManager.start, even on
119
+ // preflight failure. The record then stays uniform across branches, and
120
+ // downstream consumers can reference the paths without conditional fields.
120
121
  agentTracePath: z.string(),
121
122
  supervisorTracePath: z.string(),
122
123
  judgeTracePath: z.string(),
@@ -133,15 +134,15 @@ export const RESULT_RECORD_SCHEMA = z.union([HAPPY_RECORD, PREFLIGHT_RECORD]);
133
134
 
134
135
  export const GRADE_RECORD_SCHEMA = z.object({
135
136
  taskId: z.string().min(1),
136
- // Unlike the happy result record — where `grade.score` is the raw
137
- // weighted fraction and the effective (zeroed) value lives on the
138
- // top-level `score` — this record has no second score field, so its
139
- // `grade.score` carries the effective health/gate-zeroed value.
137
+ // In the happy result record, `grade.score` is the raw weighted fraction,
138
+ // and the effective (zeroed) value lives on the top-level `score`. This
139
+ // record has no second score field, so its `grade.score` carries the
140
+ // effective health/gate-zeroed value.
140
141
  grade: GRADE_SHAPE,
141
142
  invariants: INVARIANTS_SHAPE,
142
143
  hiddenTests: HIDDEN_TESTS_SHAPE.optional(),
143
- // Mirrors the invariants script's exit for diagnosis; the graded verdict
144
- // is what drives the command's process exit.
144
+ // This mirrors the invariants script's exit for diagnosis. The graded
145
+ // verdict drives the command's process exit.
145
146
  exitCode: z.number().int(),
146
147
  });
147
148
 
@@ -4,19 +4,19 @@
4
4
  * Phases per (task, runIndex):
5
5
  * 1. WorkdirManager.start → seed CWD + run pre-flight probe
6
6
  * 2. Supervisor session (agent + supervisor) → produce traces + submission
7
- * 3. Invariants collector + hidden-test engine → merged check rows,
8
- * graded by `gradeChecks` (rows are authoritative; script exit is
9
- * grader health only)
7
+ * 3. Invariants collector + hidden-test engine → merged check rows.
8
+ * `gradeChecks` grades them. The rows are authoritative. The script
9
+ * exit reports grader health only.
10
10
  * 4. Judge.runJudge → Conclude-driven binary gate mapped to pass/fail
11
11
  * 5. WorkdirManager.teardown → process-group cleanup
12
12
  *
13
- * Cells run with bounded in-process concurrency (`CellScheduler`); `run()`
14
- * yields records in **completion order**, not grid order. A single drain loop
15
- * is the sole writer of `<output>/results.jsonl`, appending each record the
16
- * moment its cell settles — that incremental append is the durability and
17
- * crash-safety mechanism, so a killed run keeps every completed cell and there
18
- * is no sidecar ledger. The iterator drives CLI stdout mirroring off the same
19
- * stream.
13
+ * Cells run with bounded in-process concurrency (`CellScheduler`). `run()`
14
+ * yields records in **completion order**. It does not yield them in grid
15
+ * order. A single drain loop is the sole writer of
16
+ * `<output>/results.jsonl`. It appends each record the moment its cell
17
+ * settles. That incremental append is the durability and crash-safety
18
+ * mechanism, so a killed run keeps every completed cell. There is no sidecar
19
+ * ledger. The iterator mirrors the same stream to CLI stdout.
20
20
  */
21
21
 
22
22
  import { join, resolve as resolvePath } from "node:path";
@@ -47,11 +47,11 @@ const BASE_TOOLS = [
47
47
  "TodoWrite",
48
48
  ];
49
49
 
50
- // Upper bound on a single supervised agent run. A run that produces no terminal
51
- // message within this window is treated as a stall and recorded as an
52
- // agentError, so the benchmark never hangs the event loop into a silent exit.
53
- // Overridable per-runner via `watchdogMs` so a test can force a stall to fire
54
- // without waiting the full 20 minutes.
50
+ // Upper bound on a single supervised agent run. The runner treats a run that
51
+ // produces no terminal message within this window as a stall. It records the
52
+ // stall as an agentError, so the benchmark never hangs the event loop into a
53
+ // silent exit. Set `watchdogMs` per runner to override this default. A test
54
+ // can then force a stall and skip the full 20-minute wait.
55
55
  const AGENT_WATCHDOG_MS = 20 * 60 * 1000;
56
56
 
57
57
  /** Sole orchestrator for a task-family benchmark run. */
@@ -65,25 +65,26 @@ export class BenchmarkRunner {
65
65
  * @param {string} opts.supervisorModel
66
66
  * @param {string} opts.judgeModel
67
67
  * @param {{agent?: string, judge?: string}} [opts.profiles]
68
- * @param {Function} opts.query - SDK query (injected for testability).
68
+ * @param {Function} opts.query - SDK query. A test injects its own.
69
69
  * @param {string[]} [opts.allowedTools] - Agent tool allowlist (default: BASE_TOOLS).
70
70
  * @param {number} [opts.maxTurns] - Agent-under-test turn budget.
71
71
  * @param {number} [opts.concurrency] - Max cells in flight (integer ≥ 1).
72
- * Defaults to 1 as a defensive floor; the CLI always passes a resolved value.
72
+ * Defaults to 1 as a defensive floor. The CLI always passes a resolved value.
73
73
  * @param {{index: number, total: number}} [opts.shard] - Run only the cells
74
74
  * assigned to shard `index` of `total` (1-based). Absent ≡ the whole grid
75
75
  * (identity `1/1`).
76
76
  * @param {number} [opts.watchdogMs] - Per-agent stall watchdog (ms). Defaults
77
- * to `AGENT_WATCHDOG_MS`; injectable so tests can force a stall in-test.
77
+ * to `AGENT_WATCHDOG_MS`. A test injects its own to force a stall in-test.
78
78
  * @param {number} [opts.termGraceMs] - SIGTERM→SIGKILL grace (ms) for the per-task process group.
79
79
  * @param {Function} [opts.runAgent] - Test seam: replaces the agent-under-test
80
80
  * session. Must return `{costUsd, turns, submission, agentError?}` and
81
81
  * write a valid NDJSON trace to `workdir.agentTracePath`. Default uses
82
82
  * `createAgentRunner` with the harness `BASE_TOOLS` allowlist. Internal
83
- * testing only — not part of the public API.
83
+ * testing only. It is not part of the public API.
84
84
  * @param {import("@forwardimpact/libutil/runtime").Runtime} opts.runtime -
85
- * Injected ambient collaborators (`fs`, `subprocess`, `clock`, `proc`),
86
- * threaded into the installers, workdir manager, invariants, and judge.
85
+ * The host injects these ambient collaborators (`fs`, `subprocess`,
86
+ * `clock`, `proc`). The runner threads them into the installers, the
87
+ * workdir manager, the invariants, and the judge.
87
88
  * @param {Function} [opts.runInvariants] - Test seam: replaces `runInvariants`.
88
89
  * Same contract as `runInvariants(task, ctx, runtime)`. Internal testing only.
89
90
  * @param {Function} [opts.runHiddenTests] - Test seam: replaces
@@ -203,9 +204,9 @@ export class BenchmarkRunner {
203
204
  });
204
205
 
205
206
  const allCells = enumerateCells(tasks, this.runs);
206
- // Sharding selects a deterministic subset of the grid; an unsharded run is
207
- // the identity 1/1. A high-index shard may select zero cells — a valid run
208
- // whose results.jsonl ends up empty.
207
+ // A shard selects a deterministic subset of the grid. An unsharded run is
208
+ // the identity 1/1. A high-index shard may select zero cells. That is a
209
+ // valid run whose results.jsonl ends up empty.
209
210
  const cells = this.shard
210
211
  ? selectShard(allCells, this.shard.index, this.shard.total)
211
212
  : allCells;
@@ -228,8 +229,8 @@ export class BenchmarkRunner {
228
229
  });
229
230
  // Single-writer drain: the scheduler runs up to `concurrency` cells at
230
231
  // once and pushes each settled record here in completion order. This loop
231
- // is the sole writer of `results.jsonl` — workers never touch the stream —
232
- // and the per-completion append is the crash-safety mechanism.
232
+ // is the sole writer of `results.jsonl`. Workers never touch the stream.
233
+ // The per-completion append is the crash-safety mechanism.
233
234
  try {
234
235
  for await (const record of scheduler.run(cells)) {
235
236
  await writeRecord(resultsStream, record);
@@ -255,11 +256,12 @@ export class BenchmarkRunner {
255
256
  t0,
256
257
  });
257
258
  } catch (e) {
258
- // `wm.start()` (port acquire + workdir/env seeding) is the one throw site
259
- // not caught inside `#executeCell`. Turn it into the runner's own fallback
260
- // record so `#runOne` never rejects — the scheduler's one-record-per-cell
261
- // contract depends on that. The fallback is schema-skipped by `report`,
262
- // the same as any other runner-side schema failure.
259
+ // `wm.start()` (port acquire + workdir/env seed) is the one throw site
260
+ // that `#executeCell` does not catch. Turn it into the runner's own
261
+ // fallback record so `#runOne` never rejects. The scheduler's
262
+ // one-record-per-cell contract depends on that. `report` skips the
263
+ // fallback because it fails the schema, the same as any other
264
+ // runner-side schema failure.
263
265
  return {
264
266
  taskId: task.id,
265
267
  runIndex,
@@ -273,9 +275,9 @@ export class BenchmarkRunner {
273
275
 
274
276
  /**
275
277
  * Run one cell's lifecycle against an already-started workdir: preflight
276
- * gate → supervised agent → invariants → judge → assembled record. Extracted
277
- * from `#runOne` so the start/teardown/error wrapper stays under the
278
- * complexity ceiling.
278
+ * gate → supervised agent → invariants → judge → assembled record. It is
279
+ * separate from `#runOne` so the start/teardown/error wrapper stays under
280
+ * the complexity ceiling.
279
281
  */
280
282
  async #executeCell({
281
283
  family,
@@ -313,9 +315,9 @@ export class BenchmarkRunner {
313
315
  const judgePass =
314
316
  judgeVerdict === null || judgeVerdict.verdict === "pass";
315
317
  const verdict = grade.verdict === "pass" && judgePass ? "pass" : "fail";
316
- // Gates protect the score: an unhealthy grader, a failing gate row, or
317
- // a failing judge zeroes the effective score. Full marks does not — a
318
- // fractional score with verdict fail is the point.
318
+ // Gates protect the score: an unhealthy grader, a gate row that fails,
319
+ // or a judge that fails zeroes the effective score. Full marks does not
320
+ // zero it. A fractional score with verdict fail is the point.
319
321
  const scoreValid = graded.healthy && grade.gatesPass && judgePass;
320
322
  const record = {
321
323
  taskId: task.id,
@@ -365,8 +367,8 @@ export class BenchmarkRunner {
365
367
 
366
368
  /**
367
369
  * Run the judge (when the task ships a template) over the grade result.
368
- * The record's judgeVerdict carries only the verdict + summary; the
369
- * judge's cost is folded into costUsd / costBreakdown instead.
370
+ * The record's judgeVerdict carries only the verdict + summary. The runner
371
+ * folds the judge's cost into costUsd / costBreakdown instead.
370
372
  */
371
373
  async #judgeCell({
372
374
  task,
@@ -404,10 +406,9 @@ export class BenchmarkRunner {
404
406
  }
405
407
 
406
408
  /**
407
- * Run both check-row producers against the post-run CWD and grade the
408
- * merged rows via the shared derivation. Restoration happens inside the
409
- * engine, so the judge (which runs after) sees the workdir exactly as the
410
- * agent left it.
409
+ * Run both check-row producers against the post-run CWD. Grade the merged
410
+ * rows through the shared derivation. The engine restores the workdir, so
411
+ * the judge (which runs after) sees it exactly as the agent left it.
411
412
  */
412
413
  #gradeCell(family, task, workdir) {
413
414
  const ctx = {
@@ -424,9 +425,9 @@ export class BenchmarkRunner {
424
425
 
425
426
  /**
426
427
  * Dispatch to either the injected hook or the default `#runAgent`. Either
427
- * path can throw; catch here so a thrown error becomes an `agentError` on
428
- * the record (spec criterion 1: records on agent failure) rather than
429
- * aborting the whole iterator.
428
+ * path can throw. Catch the error here so it becomes an `agentError` on the
429
+ * record (spec criterion 1: records on agent failure). The iterator then
430
+ * does not abort.
430
431
  */
431
432
  async #runAgentSafe(task, workdir) {
432
433
  try {
@@ -448,8 +449,9 @@ export class BenchmarkRunner {
448
449
 
449
450
  /**
450
451
  * Run the agent-under-test under a Supervisor. The supervisor writes
451
- * a combined tagged NDJSON trace; after the session we split it into
452
- * agent.ndjson and supervisor.ndjson and extract cost/turns/submission.
452
+ * a combined tagged NDJSON trace. After the session, this method splits
453
+ * the trace into agent.ndjson and supervisor.ndjson. It also extracts
454
+ * cost/turns/submission.
453
455
  */
454
456
  async #runAgent(task, workdir) {
455
457
  const fs = this.runtime.fs;
@@ -477,11 +479,12 @@ export class BenchmarkRunner {
477
479
  });
478
480
  const instructions = await fs.readFile(task.paths.instructions, "utf8");
479
481
  let agentError = null;
480
- // Watchdog: a supervised session can hang without settling (e.g. the agent
481
- // SDK subprocess exits without a terminal message), which would empty the
482
- // event loop and exit the process mid-run with zero records. Race the run
483
- // against a bounded timer so a stall becomes an `agentError` record instead
484
- // of a silent exit; the timer also keeps the loop alive until it fires.
482
+ // Watchdog: a supervised session can hang and never settle (e.g. the agent
483
+ // SDK subprocess exits without a terminal message). The hang would empty
484
+ // the event loop and exit the process mid-run with zero records. Race the
485
+ // run against a bounded timer so a stall becomes an `agentError` record
486
+ // instead of a silent exit. The timer also keeps the loop alive until it
487
+ // fires.
485
488
  let watchdog;
486
489
  try {
487
490
  const result = await Promise.race([
@@ -513,8 +516,9 @@ export class BenchmarkRunner {
513
516
  workdir.agentTracePath,
514
517
  workdir.supervisorTracePath,
515
518
  );
516
- // Cost is summed across every participant's result events from the one
517
- // combined trace, attributed per source. Read before unlinking.
519
+ // `sumTraceCost` sums the cost across every participant's result events
520
+ // from the one combined trace, and attributes it per source. Read the
521
+ // trace before you unlink it.
518
522
  const combined = await fs.readFile(combinedPath, "utf8");
519
523
  const { totalCostUsd, bySource } = sumTraceCost(combined.split("\n"));
520
524
  await fs.unlink(combinedPath).catch(() => {});
@@ -586,9 +590,9 @@ export class BenchmarkRunner {
586
590
  validateResultRecord(record);
587
591
  return record;
588
592
  } catch (e) {
589
- // The runner constructed the record — a schema failure is a real bug,
590
- // not bad family input. Emit a noisy fallback so the iterator stays
591
- // consumable and the agent budget isn't silently dropped.
593
+ // The runner constructed the record. A schema failure is a real bug.
594
+ // Bad family input is not the cause. Emit a noisy fallback so the
595
+ // iterator stays consumable and nothing silently drops the agent budget.
592
596
  return {
593
597
  taskId: record.taskId ?? key.taskId,
594
598
  runIndex: record.runIndex ?? key.runIndex,
@@ -601,9 +605,10 @@ export class BenchmarkRunner {
601
605
 
602
606
  /**
603
607
  * Flatten the grid into a stable ordered cell list, task-major /
604
- * runIndex-minor. Load-bearing ordering: Part 02's round-robin shard balance
605
- * depends on a task's runIndexes being adjacent in this list. Single source of
606
- * the cell list for both the scheduler and the shard selector.
608
+ * runIndex-minor. The order is load-bearing. A task's runIndexes are adjacent
609
+ * in this list, and Part 02's round-robin shard balance depends on that. This
610
+ * function is the single source of the cell list for both the scheduler and
611
+ * the shard selector.
607
612
  * @param {import("./task-family.js").Task[]} tasks
608
613
  * @param {number} runs
609
614
  * @returns {{task: import("./task-family.js").Task, runIndex: number}[]}
@@ -619,11 +624,11 @@ export function enumerateCells(tasks, runs) {
619
624
  /**
620
625
  * Round-robin partition of the enumerated cells: the cell at position `p` runs
621
626
  * iff `p % total === i - 1`. `i` is 1-based (Playwright-style). The union over
622
- * `i ∈ 1..total` is the exact grid, each cell once; when `total > cells.length`
623
- * the high-index shards select **zero** cells — a valid run. Because
624
- * `enumerateCells` is task-major, a task's run indexes are adjacent, so
625
- * round-robin spreads them across shards rather than handing one shard a slow
626
- * task's whole run block.
627
+ * `i ∈ 1..total` is the exact grid, each cell once. When
628
+ * `total > cells.length`, the high-index shards select **zero** cells. That is
629
+ * a valid run. `enumerateCells` is task-major, so a task's run indexes are
630
+ * adjacent. Round-robin then spreads them across shards. It does not hand one
631
+ * shard a slow task's whole run block.
627
632
  * @param {{task: object, runIndex: number}[]} cells
628
633
  * @param {number} i - 1-based shard index.
629
634
  * @param {number} total - Shard count.
@@ -634,8 +639,9 @@ export function selectShard(cells, i, total) {
634
639
  }
635
640
 
636
641
  /**
637
- * Validate the required BenchmarkRunner constructor arguments. Extracted from
638
- * the constructor to keep its cognitive complexity under the lint ceiling.
642
+ * Validate the required BenchmarkRunner constructor arguments. It is separate
643
+ * from the constructor to keep that function's cognitive complexity under the
644
+ * lint ceiling.
639
645
  */
640
646
  function validateRunnerArgs({
641
647
  family,
@@ -674,5 +680,5 @@ export function createBenchmarkRunner(opts) {
674
680
  return new BenchmarkRunner(opts);
675
681
  }
676
682
 
677
- // Internal exports used by tests.
683
+ // Tests use these internal exports.
678
684
  export const __BASE_TOOLS = BASE_TOOLS;
@@ -1,11 +1,12 @@
1
1
  /**
2
- * CellScheduler — bounded concurrent execution of benchmark cells.
2
+ * CellScheduler — runs benchmark cells concurrently under a bound.
3
3
  *
4
- * Keeps at most `concurrency` `runCell(cell)` calls in flight at once and
5
- * yields each settled record in **completion order** (not grid order). The
6
- * runner's drain loop consumes this async iterable as the sole writer of
7
- * `results.jsonl`, so concurrency lives here in execution while the ledger
8
- * stays single-writer — no write mutex on the hot path.
4
+ * The scheduler keeps at most `concurrency` `runCell(cell)` calls in flight
5
+ * at once. It yields each settled record in **completion order**. It does
6
+ * not yield them in grid order. The runner's drain loop consumes this async
7
+ * iterable as the sole writer of `results.jsonl`. Concurrency lives here in
8
+ * execution, and the ledger stays single-writer. The hot path needs no write
9
+ * mutex.
9
10
  */
10
11
 
11
12
  /** Bounded pool that streams settled cell records in completion order. */
@@ -14,10 +15,10 @@ export class CellScheduler {
14
15
  * @param {object} opts
15
16
  * @param {number} opts.concurrency - Max cells in flight (integer ≥ 1).
16
17
  * @param {(cell: {task: object, runIndex: number}) => Promise<object>} opts.runCell -
17
- * Runs one cell to a settled record. By contract `runCell` never rejects —
18
- * the runner's `#runOne` catches setup, agent, and schema failures and
19
- * returns a record rather than throwing — but a rejection is still guarded
20
- * so one bad cell cannot wedge the drain.
18
+ * Runs one cell to a settled record. By contract `runCell` never rejects.
19
+ * The runner's `#runOne` catches setup, agent, and schema failures. It
20
+ * returns a record instead of a throw. The scheduler still guards against
21
+ * a rejection, so one bad cell cannot wedge the drain.
21
22
  */
22
23
  constructor({ concurrency, runCell }) {
23
24
  if (!Number.isInteger(concurrency) || concurrency < 1)
@@ -29,7 +30,7 @@ export class CellScheduler {
29
30
  }
30
31
 
31
32
  /**
32
- * Run every cell with bounded concurrency, yielding each settled record the
33
+ * Run every cell with bounded concurrency. Yield each settled record the
33
34
  * moment its cell completes.
34
35
  * @param {{task: object, runIndex: number}[]} cells
35
36
  * @returns {AsyncGenerator<object>}
@@ -42,8 +43,8 @@ export class CellScheduler {
42
43
  const launch = () => {
43
44
  const cell = cells[next++];
44
45
  // The wrapper resolves to its own handle (for O(1) removal) plus the
45
- // settled record, and never rejects — a thrown runCell becomes a fail
46
- // record so the drain keeps consuming.
46
+ // settled record. It never rejects. A thrown runCell becomes a fail
47
+ // record, so the drain keeps consuming.
47
48
  const p = Promise.resolve()
48
49
  .then(() => this.runCell(cell))
49
50
  .then(
@@ -64,9 +65,9 @@ export class CellScheduler {
64
65
  }
65
66
 
66
67
  /**
67
- * Defensive fallback when `runCell` rejects (contract says it cannot). Keeps
68
- * the drain consumable; the record is intentionally minimal and will be
69
- * skipped by `report`'s schema validation, counted as skipped.
68
+ * Defensive fallback when `runCell` rejects (the contract says it cannot).
69
+ * The fallback keeps the drain consumable. The record is deliberately
70
+ * minimal. `report`'s schema validation skips it and counts it as skipped.
70
71
  */
71
72
  function schedulerFailRecord(cell, error) {
72
73
  return {
@@ -14,16 +14,18 @@
14
14
  * specs/ # copied into agent CWD
15
15
  * workdir/ # copied into agent CWD
16
16
  *
17
- * `tests/` is an overlay mirror of the agent CWD: a file's path under
18
- * `tests/` is its staging path. Every `*.test.js` file is one check —
19
- * `*.gate.test.js` marks a gate, any other `*.test.js` is scored — and every
20
- * other file is support material, staged but never graded. The layout is
21
- * validated eagerly here so authoring errors fail before any agent spend.
17
+ * `tests/` is an overlay mirror of the agent CWD. A file's path under
18
+ * `tests/` is its staging path. Every `*.test.js` file is one check.
19
+ * `*.gate.test.js` marks a gate. Any other `*.test.js` counts toward the
20
+ * score. Every other file is support material. The benchmark stages it but
21
+ * never grades it. This loader validates the layout eagerly, so authoring
22
+ * errors fail before any agent spend.
22
23
  *
23
- * Local paths or git URLs are both accepted; git URLs are shallow-cloned into
24
- * a temp dir and `familyRevision` becomes `git:<sha>` of HEAD at clone time.
25
- * Local paths use the canonical-tree algorithm from design § Family revision
26
- * algorithm so the result is stable across operating systems.
24
+ * The loader accepts a local path or a git URL. For a git URL it makes a
25
+ * shallow clone into a temp dir. `familyRevision` then becomes `git:<sha>` of
26
+ * HEAD at clone time. Local paths use the canonical-tree algorithm from
27
+ * design § Family revision algorithm, so the result is stable across
28
+ * operating systems.
27
29
  *
28
30
  * Filesystem and subprocess access route through the injected `runtime` bag
29
31
  * (`runtime.fs` async, `runtime.subprocess.run` one-shot, `tmpdir` derived
@@ -35,13 +37,13 @@ import { join, posix, relative, resolve, sep } from "node:path";
35
37
 
36
38
  const GIT_URL_RE = /^(git@|https?:\/\/|ssh:\/\/|git:\/\/)/;
37
39
  const SKIP_DIRS = new Set([".git", "node_modules"]);
38
- // POSIX `X_OK` (execute permission); node's fs honours the numeric mode, so we
39
- // avoid importing `node:fs`'s `constants` (which would light the fs smell).
40
+ // POSIX `X_OK` (execute permission). Node's fs honours the numeric mode, so we
41
+ // do not import `node:fs`'s `constants`, which would light the fs smell.
40
42
  const X_OK = 1;
41
43
 
42
44
  /**
43
- * Derive the system temp dir from the env (node's `os.tmpdir()` is itself an
44
- * env-respecting wrapper). The runtime bag has no `os` slot by design.
45
+ * Derive the system temp dir from the env. Node's `os.tmpdir()` also reads
46
+ * the env. The runtime bag has no `os` slot by design.
45
47
  * @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
46
48
  * @returns {string}
47
49
  */
@@ -62,7 +64,7 @@ export async function loadTaskFamily(rootPathOrGitUrl, runtime) {
62
64
  let familyRevision;
63
65
  if (isGit) {
64
66
  const dir = await runtime.fs.mkdtemp(
65
- join(tmpdir(runtime), "fit-benchmark-family-"),
67
+ join(tmpdir(runtime), "gemba-benchmark-family-"),
66
68
  );
67
69
  await gitClone(runtime, rootPathOrGitUrl, dir);
68
70
  rootPath = dir;
@@ -84,9 +86,9 @@ export async function loadTaskFamily(rootPathOrGitUrl, runtime) {
84
86
  }
85
87
 
86
88
  /**
87
- * Assert that `<judgeProfilesDir>/<judgeProfile>.md` exists. Called from
88
- * `BenchmarkRunner.run()` so a missing judge profile fails the family
89
- * install before any agent session starts.
89
+ * Assert that `<judgeProfilesDir>/<judgeProfile>.md` exists.
90
+ * `BenchmarkRunner.run()` calls it, so a missing judge profile fails the
91
+ * family install before any agent session starts.
90
92
  * @param {TaskFamily} _family
91
93
  * @param {string} judgeProfilesDir
92
94
  * @param {string} judgeProfile
@@ -161,8 +163,8 @@ const GATE_SUFFIX = ".gate.test.js";
161
163
 
162
164
  /**
163
165
  * Discover and validate a task's hidden test suite under `<taskDir>/tests/`.
164
- * Returns null when the directory is absent; throws on an invalid layout
165
- * (no check files, a dangling symlink, duplicate check names) so
166
+ * Returns null when the directory is absent. Throws on an invalid layout
167
+ * (no check files, a dangling symlink, duplicate check names), so
166
168
  * `loadTaskFamily` rejects broken suites before any agent spend.
167
169
  * @param {object} fs - Async filesystem surface (`runtime.fs`).
168
170
  * @param {string} taskDir
@@ -211,10 +213,10 @@ async function discoverSuite(fs, taskDir) {
211
213
  }
212
214
 
213
215
  /**
214
- * Walk a suite tree collecting `{sourcePath, stagePath}` entries. Unlike
215
- * `walkFiles` (which silently skips dangling symlinks for hashing), every
216
- * entry here must be a regular file after symlink resolution — a dangling
217
- * symlink or a link to a non-file target is an authoring error.
216
+ * Walk a suite tree and collect `{sourcePath, stagePath}` entries. Every
217
+ * entry here must be a regular file after symlink resolution. `walkFiles` is
218
+ * different. It silently skips a dangling symlink when it hashes the tree. A
219
+ * dangling symlink or a link to a non-file target is an authoring error.
218
220
  */
219
221
  async function walkSuiteFiles(fs, root, dir, out) {
220
222
  const entries = await fs.readdir(dir, { withFileTypes: true });
@@ -343,18 +345,18 @@ async function git(runtime, args) {
343
345
 
344
346
  /**
345
347
  * @typedef {object} HiddenCheck
346
- * @property {string} name - Basename stem with the check suffix stripped.
348
+ * @property {string} name - Basename stem without the check suffix.
347
349
  * @property {boolean} gate - True iff the filename ends `.gate.test.js`.
348
350
  * @property {string} sourcePath - Absolute path under `tests/`.
349
- * @property {string} stagePath - Path relative to `tests/` — the overlay
350
- * mirror of the staging path under the agent CWD.
351
+ * @property {string} stagePath - Path relative to `tests/`. It mirrors the
352
+ * staging path under the agent CWD.
351
353
  */
352
354
 
353
355
  /**
354
356
  * @typedef {object} HiddenSuite
355
357
  * @property {HiddenCheck[]} checks - In sorted stage-path order.
356
358
  * @property {{sourcePath: string, stagePath: string}[]} support - Non-check
357
- * files, staged for the whole pass but never graded.
359
+ * files. The benchmark stages them for the whole pass but never grades them.
358
360
  */
359
361
 
360
362
  /**
@@ -1,19 +1,20 @@
1
1
  /**
2
- * Combined-supervisor-trace splitting for the benchmark runner: one pass
2
+ * Split the combined supervisor trace for the benchmark runner. One pass
3
3
  * over the tagged NDJSON envelope stream separates agent events from
4
- * supervisor/orchestrator events and extracts the run summary.
4
+ * supervisor and orchestrator events. The same pass extracts the run summary.
5
5
  */
6
6
 
7
7
  import { createInterface } from "node:readline";
8
8
 
9
9
  /**
10
- * Split the combined supervisor trace into agent and supervisor files and
11
- * extract turn count and submission in a single pass. Agent-source events go
12
- * to `agentPath`; supervisor and orchestrator events go to `supervisorPath`.
10
+ * Split the combined supervisor trace into agent and supervisor files in a
11
+ * single pass. The same pass extracts the turn count and the submission.
12
+ * Agent-source events go to `agentPath`. Supervisor and orchestrator events
13
+ * go to `supervisorPath`.
13
14
  *
14
- * Cost is deliberately not summed here — the caller derives it from the same
15
- * combined trace via `sumTraceCost`, so there is one cost path across the
16
- * benchmark, callback, and `fit-trace cost` consumers.
15
+ * This function deliberately does not sum cost. The caller derives it from
16
+ * the same combined trace with `sumTraceCost`. One cost path then serves the
17
+ * benchmark, callback, and `gemba-trace cost` consumers.
17
18
  * @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
18
19
  * @param {string} combinedPath
19
20
  * @param {string} agentPath