@forwardimpact/libharness 2.0.0 → 3.0.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (73) hide show
  1. package/README.md +68 -65
  2. package/package.json +15 -13
  3. package/src/advisor.js +47 -41
  4. package/src/agent-runner.js +58 -48
  5. package/src/benchmark/apm-installer.js +28 -28
  6. package/src/benchmark/env-loader.js +24 -16
  7. package/src/benchmark/grade.js +44 -41
  8. package/src/benchmark/hidden-tests.js +25 -24
  9. package/src/benchmark/hook-env.js +11 -9
  10. package/src/benchmark/invariants.js +20 -17
  11. package/src/benchmark/judge.js +29 -28
  12. package/src/benchmark/npm-installer.js +9 -8
  13. package/src/benchmark/report.js +53 -50
  14. package/src/benchmark/result.js +24 -23
  15. package/src/benchmark/runner.js +75 -69
  16. package/src/benchmark/scheduler.js +17 -16
  17. package/src/benchmark/task-family.js +29 -27
  18. package/src/benchmark/trace-split.js +9 -8
  19. package/src/benchmark/workdir.js +27 -25
  20. package/src/claude-code-executable.js +11 -11
  21. package/src/commands/advisor-flags.js +8 -7
  22. package/src/commands/assert.js +16 -15
  23. package/src/commands/benchmark-definition.js +20 -20
  24. package/src/commands/benchmark-grade.js +13 -12
  25. package/src/commands/benchmark-report.js +5 -5
  26. package/src/commands/benchmark-run.js +31 -28
  27. package/src/commands/by-discussion.js +11 -11
  28. package/src/commands/callback.js +11 -11
  29. package/src/commands/discuss.js +8 -7
  30. package/src/commands/facilitate.js +16 -14
  31. package/src/commands/output.js +4 -3
  32. package/src/commands/run.js +15 -15
  33. package/src/commands/scan-logs.js +22 -20
  34. package/src/commands/selfedit.js +124 -0
  35. package/src/commands/supervise.js +13 -11
  36. package/src/commands/task-input.js +9 -9
  37. package/src/commands/tee.js +11 -10
  38. package/src/commands/trace.js +55 -42
  39. package/src/commands/work-tracker.js +4 -3
  40. package/src/cost.js +17 -17
  41. package/src/discuss-tools.js +16 -16
  42. package/src/discusser.js +39 -38
  43. package/src/events/github.js +54 -37
  44. package/src/facilitator.js +21 -21
  45. package/src/inbox-poller.js +4 -4
  46. package/src/judge.js +32 -30
  47. package/src/message-bus.js +12 -11
  48. package/src/orchestration-loop.js +35 -36
  49. package/src/orchestration-toolkit.js +58 -53
  50. package/src/orchestrator-helpers.js +2 -2
  51. package/src/profile-prompt.js +54 -53
  52. package/src/redaction.js +63 -57
  53. package/src/render/line-renderer.js +5 -5
  54. package/src/render/orchestrator-filter.js +3 -3
  55. package/src/render/palette.js +11 -9
  56. package/src/render/tool-hints.js +18 -15
  57. package/src/render/turn-renderer.js +4 -4
  58. package/src/reply-emitter.js +2 -2
  59. package/src/sequence-counter.js +4 -3
  60. package/src/signature-filter.js +7 -6
  61. package/src/supervisor.js +19 -18
  62. package/src/tee-writer.js +25 -25
  63. package/src/trace-collector.js +53 -48
  64. package/src/trace-github.js +53 -44
  65. package/src/trace-multi.js +16 -14
  66. package/src/trace-query.js +61 -52
  67. package/src/trace-render.js +19 -19
  68. package/src/trace-usage.js +31 -28
  69. package/src/transcript-recorder.js +24 -20
  70. package/bin/fit-benchmark.js +0 -44
  71. package/bin/fit-harness.js +0 -412
  72. package/bin/fit-selfedit.js +0 -165
  73. package/bin/fit-trace.js +0 -520
@@ -7,9 +7,9 @@
7
7
  * runAgent → invariants → judge → teardown.
8
8
  *
9
9
  * Filesystem, subprocess, clock, and process-signal access all route through
10
- * the injected `runtime` bag. Only raw TCP plumbing (`node:net`) stays direct —
11
- * it is not an ambient-dependency smell and the runtime bag models no socket
12
- * surface.
10
+ * the injected `runtime` bag. Only raw TCP plumbing (`node:net`) stays
11
+ * direct. It is not an ambient-dependency smell, and the runtime bag models
12
+ * no socket surface.
13
13
  */
14
14
 
15
15
  import { createServer } from "node:net";
@@ -27,7 +27,7 @@ const DEFAULT_TERM_GRACE_MS = 5_000;
27
27
  * @property {string} runDir - Parent of `cwd`; holds trace/log siblings.
28
28
  * @property {number} port - Allocated TCP port for the agent.
29
29
  * @property {number} pgid - Process-group id captured from the preflight child.
30
- * @property {*} scaffold - Reserved per design § Components; v1 sets null.
30
+ * @property {*} scaffold - Reserved per design § Components. v1 sets null.
31
31
  * @property {string} agentTracePath
32
32
  * @property {string} supervisorTracePath
33
33
  * @property {string} judgeTracePath
@@ -59,8 +59,8 @@ export class WorkdirManager {
59
59
  this.termGraceMs = termGraceMs ?? DEFAULT_TERM_GRACE_MS;
60
60
  this.familyRootPath = familyRootPath ?? null;
61
61
  this.runtime = runtime;
62
- // One registry per manager: hands out distinct, bindable ports under a lock
63
- // so concurrent cells can never be handed the same number.
62
+ // One registry per manager. It hands out distinct, bindable ports under a
63
+ // lock, so two concurrent cells never get the same number.
64
64
  this.ports = new PortRegistry();
65
65
  }
66
66
 
@@ -77,9 +77,10 @@ export class WorkdirManager {
77
77
  const cwd = join(runDir, "cwd");
78
78
  await fs.mkdir(cwd, { recursive: true });
79
79
 
80
- // Family-level shared fixtures: convention-over-configuration, copied if
81
- // present. They form the shared base; the per-task workdir/specs below
82
- // overlay on top (fs.cp defaults to force:true, so a per-task file wins).
80
+ // Family-level shared fixtures follow convention over configuration. The
81
+ // manager copies them if they are present. They form the shared base. The
82
+ // per-task workdir/specs below overlay on top (fs.cp defaults to
83
+ // force:true, so a per-task file wins).
83
84
  if (this.familyRootPath) {
84
85
  await fs
85
86
  .cp(join(this.familyRootPath, "workdir"), cwd, { recursive: true })
@@ -162,7 +163,7 @@ export class WorkdirManager {
162
163
  try {
163
164
  proc.kill(-workdir.pgid, "SIGTERM");
164
165
  } catch {
165
- // Process group already gone — fine.
166
+ // The process group is already gone. That is fine.
166
167
  }
167
168
  await clock.sleep(this.termGraceMs);
168
169
  try {
@@ -170,17 +171,17 @@ export class WorkdirManager {
170
171
  } catch {
171
172
  // Already exited.
172
173
  }
173
- // Poll briefly until the process group is empty — SIGKILL returns
174
- // before the kernel finishes reaping descendants.
174
+ // Poll briefly until the process group is empty. SIGKILL returns
175
+ // before the kernel reaps every descendant.
175
176
  await waitFor(
176
177
  this.runtime,
177
178
  async () => (await countDescendants(this.runtime, workdir.pgid)) === 0,
178
179
  2_000,
179
180
  );
180
181
  }
181
- // Release the reservation in a finally so a throwing probe cannot leak the
182
- // number from the in-use set; release after the port-free probe so a freed
183
- // number can be re-handed to a waiting cell.
182
+ // Release the reservation in a finally, so a probe that throws cannot leak
183
+ // the number from the in-use set. Release after the port-free probe, so the
184
+ // registry can hand a freed number to a cell that waits.
184
185
  try {
185
186
  const portFree = await isPortFree(workdir.port);
186
187
  const descendants = await countDescendants(this.runtime, workdir.pgid);
@@ -196,10 +197,10 @@ export class WorkdirManager {
196
197
  * close-then-return allocator whose allocate→bind window let two concurrent
197
198
  * cells receive the same number.
198
199
  *
199
- * The reservation is the *number*, not a held socket — a held socket could not
200
- * be bound by the agent later. `acquire` serializes through a one-slot promise
201
- * chain and re-probes if the OS hands back a number already in the live in-use
202
- * set, so no two in-flight cells share a port.
200
+ * The reservation is the *number*. The registry holds no socket, because the
201
+ * agent could not bind a held socket later. `acquire` serializes through a
202
+ * one-slot promise chain. It re-probes if the OS hands back a number already
203
+ * in the live in-use set, so no two in-flight cells share a port.
203
204
  */
204
205
  export class PortRegistry {
205
206
  #inUse = new Set();
@@ -216,7 +217,7 @@ export class PortRegistry {
216
217
  return port;
217
218
  });
218
219
  // Keep the chain alive even if one acquire rejects, so later acquires
219
- // still run; swallow here, surface the rejection on `next`.
220
+ // still run. Swallow the rejection here. Surface it on `next`.
220
221
  this.#tail = next.catch(() => {});
221
222
  return next;
222
223
  }
@@ -231,8 +232,8 @@ export class PortRegistry {
231
232
  * Spawn preflight. Stays detached so we can SIGTERM the whole process group.
232
233
  * @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
233
234
  * @param {string} script
234
- * @param {string} cwd - Agent CWD passed via $AGENT_CWD.
235
- * @param {number} port - Free TCP port passed via $PORT.
235
+ * @param {string} cwd - Agent CWD, passed in $AGENT_CWD.
236
+ * @param {number} port - Free TCP port, passed in $PORT.
236
237
  * @param {{taskId: string, taskDir: string, hooksDir: string, familyDir: string|null}} vars - Extra hook env vars.
237
238
  * @returns {Promise<{pgid: number, error?: {phase: string, message: string, exitCode: number}}>}
238
239
  */
@@ -269,8 +270,9 @@ async function runPreflight(runtime, script, cwd, port, vars) {
269
270
  }
270
271
 
271
272
  /**
272
- * Allocate a free TCP port by binding to 0 and releasing it. Shared with the
273
- * `grade` subcommand, which needs a plausible `$PORT` for the hook env.
273
+ * Allocate a free TCP port. Bind to port 0, then release it. The `grade`
274
+ * subcommand shares this function, because it needs a plausible `$PORT` for
275
+ * the hook env.
274
276
  * @returns {Promise<number>}
275
277
  */
276
278
  export function probeFreePort() {
@@ -340,7 +342,7 @@ async function waitFor(runtime, predicate, timeoutMs) {
340
342
  }
341
343
 
342
344
  /**
343
- * Factory function — wires real dependencies.
345
+ * Factory function. Wires the real dependencies.
344
346
  * @param {ConstructorParameters<typeof WorkdirManager>[0]} deps
345
347
  * @returns {WorkdirManager}
346
348
  */
@@ -3,26 +3,26 @@
3
3
  *
4
4
  * `query()` spawns a native `claude` binary that the SDK resolves from its own
5
5
  * platform-specific optional dependency (`@anthropic-ai/claude-agent-sdk-<platform>`).
6
- * `bun build --compile` bundles the SDK's JavaScript but not that separate
7
- * native package — it is not part of the import graph — so a compiled fit-*
8
- * binary can't self-resolve it and `query()` throws "Native CLI binary for
9
- * <platform> not found".
6
+ * `bun build --compile` bundles the SDK's JavaScript. It does not bundle that
7
+ * separate native package, which is not part of the import graph. So a
8
+ * compiled fit-* binary cannot self-resolve it, and `query()` throws "Native
9
+ * CLI binary for <platform> not found".
10
10
  *
11
- * In a compiled binary we point the SDK at the standalone `claude` on PATH,
12
- * installed beside fit-harness by the bootstrap action's `fit-install.sh`.
13
- * Running from source keeps `node_modules`, where the SDK resolves its own
14
- * version-matched binary, so there we return undefined and defer to the SDK.
11
+ * In a compiled binary we point the SDK at the standalone `claude` on PATH.
12
+ * The gemba-bootstrap action's `fit-install.sh` installs it beside gemba-harness. A
13
+ * run from source keeps `node_modules`, where the SDK resolves its own
14
+ * version-matched binary. There we return undefined and defer to the SDK.
15
15
  */
16
16
  import { LIBCLI_IS_COMPILED } from "@forwardimpact/libcli";
17
17
 
18
18
  /**
19
19
  * @param {object} [deps]
20
20
  * @param {(cmd: string) => string | null | undefined} [deps.which] -
21
- * PATH resolver (injected for testing).
21
+ * PATH resolver (tests inject it).
22
22
  * @param {boolean} [deps.isCompiled] -
23
- * Whether this is a `bun --compile` binary (injected for testing).
23
+ * Whether this is a `bun --compile` binary (tests inject it).
24
24
  * @returns {string | undefined} absolute path to `claude`, or undefined to
25
- * defer resolution to the SDK.
25
+ * let the SDK resolve it.
26
26
  */
27
27
  export function resolveClaudeCodeExecutable({
28
28
  which = defaultWhich,
@@ -1,15 +1,16 @@
1
1
  /**
2
- * Shared advisor-flag parsing for the four session-mode commands. The two
3
- * flags are identical everywhere: `--advisor-model` (no default — absent
4
- * means the Advisor tool is not offered) and `--advisor-max-uses`
5
- * (default 3), which is a usage error without the model flag.
2
+ * Shared advisor-flag parser for the four session-mode commands. The two
3
+ * flags are identical everywhere: `--advisor-model` and `--advisor-max-uses`.
4
+ * `--advisor-model` has no default. When it is absent, the commands do not
5
+ * offer the Advisor tool. `--advisor-max-uses` defaults to 3. It is a usage
6
+ * error without the model flag.
6
7
  */
7
8
 
8
9
  /**
9
10
  * Parse `--advisor-model` / `--advisor-max-uses` from parsed option values.
10
- * A malformed max-uses is a usage error, not a silent fallback: NaN would
11
- * make the budget check (`used >= maxUses`) permanently false and disable
12
- * the code-enforced cap the flag exists to guarantee.
11
+ * A malformed max-uses raises a usage error. It never falls back silently.
12
+ * NaN would make the budget check (`used >= maxUses`) permanently false. It
13
+ * would disable the code-enforced cap the flag exists to guarantee.
13
14
  * @param {object} values - Parsed option values from cli.parse()
14
15
  * @returns {{advisorModel: string|undefined, advisorMaxUses: number}}
15
16
  */
@@ -67,10 +67,10 @@ export function evaluateAssertion(values, args, fsSync) {
67
67
  }
68
68
 
69
69
  /**
70
- * Attach the check-row grading role: `--gate` marks a gate check, `--weight`
71
- * attaches a numeric weight (0 marks the row diagnostic). `--gate` with any
72
- * `--weight` — 0 included — is invalid: a stray weight must never silently
73
- * disarm a gate.
70
+ * Attach the grading role of the check row. `--gate` marks a gate check.
71
+ * `--weight` attaches a numeric weight, and 0 marks the row diagnostic.
72
+ * `--gate` with any `--weight` is invalid, and that includes 0. A stray
73
+ * weight must never silently disarm a gate.
74
74
  * @param {object} values
75
75
  * @param {{test: string, pass: boolean, message?: string}} output - Mutated.
76
76
  */
@@ -92,8 +92,9 @@ function applyGradingFlags(values, output) {
92
92
  }
93
93
 
94
94
  /**
95
- * Parse a `--weight` value; null when invalid. A blank string is invalid —
96
- * `Number("")` is 0, which would silently demote the check to a diagnostic.
95
+ * Parse a `--weight` value. Return null when it is invalid. A blank string is
96
+ * invalid, because `Number("")` is 0. That would silently demote the check to
97
+ * a diagnostic.
97
98
  * @param {string} raw
98
99
  * @returns {number | null}
99
100
  */
@@ -104,11 +105,11 @@ function parseWeight(raw) {
104
105
  }
105
106
 
106
107
  /**
107
- * The grading role an emit-then-fail row keeps: a failing check must not
108
- * lose its authored role — an errored gate that demoted to a scored row
109
- * would let a broken scaffold earn partial credit instead of zeroing the
110
- * score. Invalid or conflicting flags yield no role (the row fails as a
111
- * unit-weight scored check).
108
+ * The grading role an emit-then-fail row keeps. A check that fails must not
109
+ * lose its authored role. An errored gate that demoted to a scored row would
110
+ * let a broken scaffold earn partial credit. The score would not drop to
111
+ * zero. Invalid flags, or flags that conflict, yield no role. The row then
112
+ * fails as a unit-weight scored check.
112
113
  * @param {object} values
113
114
  * @returns {{gate?: true, weight?: number}}
114
115
  */
@@ -126,10 +127,10 @@ function errorRowRole(values) {
126
127
  * Run an assertion, write JSON to stdout, and return a failure envelope when
127
128
  * the assertion does not pass.
128
129
  *
129
- * Emit-then-fail on every failure path: an invalid grading flag or an
130
- * errored evaluation (e.g. `--grep` against a file the agent deleted) writes
131
- * a failing row before the nonzero exit, so a typo or a vanished target
132
- * shrinks the score, never the denominator.
130
+ * Emit-then-fail applies on every failure path. An invalid grading flag
131
+ * writes a failed row before the nonzero exit. An errored evaluation does the
132
+ * same, for example `--grep` against a file the agent deleted. So a typo or a
133
+ * vanished target shrinks the score. It never shrinks the denominator.
133
134
  * @param {import("@forwardimpact/libcli").InvocationContext} ctx
134
135
  * @returns {Promise<{ok: true} | {ok: false, code: number, error: string}>}
135
136
  */
@@ -1,7 +1,7 @@
1
1
  /**
2
- * `fit-benchmark` CLI definition. Lives in `src/` so the bin stays an
3
- * execute-on-import entry point — launcher packages import the bin to run
4
- * it — while tests import the definition without running the CLI.
2
+ * `gemba-benchmark` CLI definition. It lives in `src/` so the bin stays an
3
+ * execute-on-import entry point. Launcher packages import the bin to run it.
4
+ * Tests import the definition and never run the CLI.
5
5
  */
6
6
 
7
7
  import { runBenchmarkRunCommand } from "./benchmark-run.js";
@@ -13,7 +13,7 @@ import {
13
13
  } from "@forwardimpact/libutil/models";
14
14
 
15
15
  export const definition = {
16
- name: "fit-benchmark",
16
+ name: "gemba-benchmark",
17
17
  description:
18
18
  "Run coding-agent task families, grade hidden tests, and aggregate pass@k across runs.",
19
19
  commands: [
@@ -36,7 +36,7 @@ export const definition = {
36
36
  "skills-from": {
37
37
  type: "string",
38
38
  description:
39
- "Stage .claude/ from this directory (a root containing .claude/) instead of running apm install — exercise local, unpublished skills",
39
+ "Stage .claude/ from this directory (a root that holds .claude/) instead of an apm install, to exercise local, unpublished skills",
40
40
  },
41
41
  output: {
42
42
  type: "string",
@@ -85,7 +85,7 @@ export const definition = {
85
85
  shard: {
86
86
  type: "string",
87
87
  description:
88
- "Run only shard i of N as i/N (1-based; default: the whole family). Each shard writes a partial results.jsonl; report --input merges them.",
88
+ "Run only shard i of N as i/N (1-based; default: the whole family). Each shard writes a partial results.jsonl. report --input merges them.",
89
89
  },
90
90
  "allowed-tools": {
91
91
  type: "string",
@@ -99,7 +99,7 @@ export const definition = {
99
99
  args: [],
100
100
  handler: runBenchmarkGradeCommand,
101
101
  description:
102
- "Grade a single task against a post-run workdir without invoking an agent: run the hidden test suite and the invariants script, then derive the verdict from the check rows (the exit mirrors it).",
102
+ "Grade a single task against a post-run workdir with no agent. Run the hidden test suite and the invariants script. Then derive the verdict from the check rows (the exit mirrors it).",
103
103
  options: {
104
104
  family: {
105
105
  type: "string",
@@ -112,7 +112,7 @@ export const definition = {
112
112
  "run-dir": {
113
113
  type: "string",
114
114
  description:
115
- "Post-run directory whose cwd/ subdir is the agent CWD; both producers run against that cwd — the path hooks receive as $AGENT_CWD",
115
+ "Post-run directory whose cwd/ subdir is the agent CWD. Both producers run against that cwd. Hooks receive it as $AGENT_CWD",
116
116
  },
117
117
  output: {
118
118
  type: "string",
@@ -125,12 +125,12 @@ export const definition = {
125
125
  args: [],
126
126
  handler: runBenchmarkReportCommand,
127
127
  description:
128
- "Aggregate result records into pass@k via the OpenAI HumanEval estimator.",
128
+ "Aggregate result records into pass@k with the OpenAI HumanEval estimator.",
129
129
  options: {
130
130
  input: {
131
131
  type: "string",
132
132
  description:
133
- "Run-output directory containing results.jsonl (default: benchmark-runs)",
133
+ "Run-output directory that holds results.jsonl (default: benchmark-runs)",
134
134
  },
135
135
  k: {
136
136
  type: "string",
@@ -143,7 +143,7 @@ export const definition = {
143
143
  detail: {
144
144
  type: "string",
145
145
  description:
146
- "Text report verbosity (full|compact, default: full). compact omits per-task detail — useful for sharded run summaries.",
146
+ "Text report verbosity (full|compact, default: full). compact omits per-task detail, which helps with sharded run summaries.",
147
147
  },
148
148
  },
149
149
  },
@@ -154,14 +154,14 @@ export const definition = {
154
154
  json: { type: "boolean", description: "Output help as JSON" },
155
155
  },
156
156
  examples: [
157
- "fit-benchmark run --family=./families/coding",
158
- "fit-benchmark run --family=./families/coding --task=todo-api --runs=1",
159
- "fit-benchmark run --family=./families/coding --work-tracker=filesystem",
160
- "fit-benchmark run --family=./families/coding --skills-from=. --task=todo-api",
161
- `fit-benchmark run --family=./families/coding --runs=10 --agent-model=${BENCHMARK_AGENT_MODEL}`,
162
- "fit-benchmark grade --family=./families/coding --task=todo-api --run-dir=./benchmark-runs/runs/todo-api/0",
163
- "fit-benchmark report --format=text",
164
- "fit-benchmark report --input=./runs/today --k=1,3,5 --format=text",
157
+ "gemba-benchmark run --family=./families/coding",
158
+ "gemba-benchmark run --family=./families/coding --task=todo-api --runs=1",
159
+ "gemba-benchmark run --family=./families/coding --work-tracker=filesystem",
160
+ "gemba-benchmark run --family=./families/coding --skills-from=. --task=todo-api",
161
+ `gemba-benchmark run --family=./families/coding --runs=10 --agent-model=${BENCHMARK_AGENT_MODEL}`,
162
+ "gemba-benchmark grade --family=./families/coding --task=todo-api --run-dir=./benchmark-runs/runs/todo-api/0",
163
+ "gemba-benchmark report --format=text",
164
+ "gemba-benchmark report --input=./runs/today --k=1,3,5 --format=text",
165
165
  ],
166
166
  documentation: [
167
167
  {
@@ -174,7 +174,7 @@ export const definition = {
174
174
  title: "Automate with GitHub Actions",
175
175
  url: "https://www.forwardimpact.team/docs/libraries/prove-changes/run-benchmark/ci-workflow/index.md",
176
176
  description:
177
- "Run benchmarks in CI with the forwardimpact/benchmark action.",
177
+ "Run benchmarks in CI with the forwardimpact/gemba-benchmark action.",
178
178
  },
179
179
  ],
180
180
  };
@@ -1,9 +1,9 @@
1
1
  /**
2
- * `fit-benchmark grade` — run both check-row producers (the hidden test
3
- * suite and the invariants script) against a post-run workdir directory and
4
- * grade the merged rows with the same derivation the benchmark runner uses.
5
- * No agent and no judge run, so authors validate a task's grading material
6
- * against fixtures without paying for agent sessions; the process exit
2
+ * `gemba-benchmark grade` — run both check-row producers (the hidden test
3
+ * suite and the invariants script) against a post-run workdir directory.
4
+ * Grade the merged rows with the same derivation the benchmark runner uses.
5
+ * No agent runs and no judge runs. So an author validates a task's grading
6
+ * material against fixtures, and pays for no agent session. The process exit
7
7
  * mirrors the graded verdict.
8
8
  */
9
9
 
@@ -46,12 +46,12 @@ export async function runBenchmarkGradeCommand(ctx) {
46
46
  runInvariants,
47
47
  runHiddenTests,
48
48
  });
49
- // Same effective-score rule as the runner, minus the judge (none runs
50
- // here): an unhealthy grader or a failing gate zeroes the score, so a
51
- // crashed hook can never mint marks from the rows it emitted before dying.
52
- // Unlike a runner record — where `grade.score` stays the raw fraction and
53
- // the zeroing lands on the top-level `score` — this record has no second
54
- // field, so `grade.score` carries the effective value here.
49
+ // The effective-score rule matches the runner, minus the judge (none runs
50
+ // here). An unhealthy grader or a gate that fails zeroes the score. So a
51
+ // crashed hook can never mint marks from the rows it emitted before it
52
+ // died. A runner record keeps the raw fraction in `grade.score` and zeroes
53
+ // the top-level `score` instead. This record has no second field, so
54
+ // `grade.score` carries the effective value here.
55
55
  if (grade.score !== undefined && !(healthy && grade.gatesPass)) {
56
56
  grade.score = 0;
57
57
  }
@@ -65,7 +65,8 @@ export async function runBenchmarkGradeCommand(ctx) {
65
65
  ...(engineError && { error: engineError.message }),
66
66
  },
67
67
  }),
68
- // Mirrors the script for diagnosis; the graded verdict drives the exit.
68
+ // This mirrors the script for diagnosis. The graded verdict drives the
69
+ // exit.
69
70
  exitCode: invariants.exitCode,
70
71
  };
71
72
  validateGradeRecord(record);
@@ -1,9 +1,9 @@
1
1
  /**
2
- * `fit-benchmark report` — aggregate `results.jsonl` into pass@k via the
3
- * OpenAI HumanEval estimator. Output is JSON by default; pass --format=text
4
- * to render a markdown table. --detail=compact drops the per-task detail
5
- * sections so a sharded run's per-shard summary stays short (the merge job
6
- * renders the full report over the combined ledger).
2
+ * `gemba-benchmark report` — aggregate `results.jsonl` into pass@k with the
3
+ * OpenAI HumanEval estimator. The command writes JSON by default. Pass
4
+ * --format=text to render a markdown table. --detail=compact drops the
5
+ * per-task detail sections, so a sharded run's per-shard summary stays short.
6
+ * The merge job renders the full report over the combined ledger.
7
7
  */
8
8
 
9
9
  import { resolve } from "node:path";
@@ -1,7 +1,7 @@
1
1
  /**
2
- * `fit-benchmark run` — run every task in a family for N runs, stream each
3
- * ResultRecord to stdout (one JSON line per record), and append to the
4
- * canonical `<output>/results.jsonl` for the report subcommand.
2
+ * `gemba-benchmark run` — run every task in a family for N runs. Stream each
3
+ * ResultRecord to stdout (one JSON line per record). Append each record to
4
+ * the canonical `<output>/results.jsonl` for the report subcommand.
5
5
  */
6
6
 
7
7
  import { resolve } from "node:path";
@@ -30,15 +30,15 @@ export async function runBenchmarkRunCommand(ctx) {
30
30
  }
31
31
  const config = await createConfig("script", "benchmark");
32
32
  runtime.proc.env.ANTHROPIC_API_KEY = await config.anthropicToken();
33
- // The benchmark agent runs via createBenchmarkRunner, not the supervise
34
- // command, so the active-tracker env must land here before the runner
35
- // spawns the subprocess that inherits process.env.
33
+ // The benchmark agent runs through createBenchmarkRunner. The supervise
34
+ // command does not run it. So the active-tracker env must land here before
35
+ // the runner spawns the subprocess that inherits process.env.
36
36
  runtime.proc.env.LIBHARNESS_WORK_TRACKER = opts.workTracker;
37
37
 
38
38
  // The Claude Agent SDK spawns a `claude` subprocess that inherits
39
- // process.env. NODE_EXTRA_CA_CERTS causes undici (the HTTP client
40
- // inside that subprocess) to fail with UND_ERR_INVALID_ARG on
41
- // Node 22+, aborting every API call after 10 retries. Strip it
39
+ // process.env. NODE_EXTRA_CA_CERTS makes undici (the HTTP client
40
+ // inside that subprocess) fail with UND_ERR_INVALID_ARG on Node 22+.
41
+ // Undici then aborts every API call after 10 retries. Strip it
42
42
  // before the SDK loads so the subprocess gets a clean environment.
43
43
  delete runtime.proc.env.NODE_EXTRA_CA_CERTS;
44
44
 
@@ -57,13 +57,14 @@ export async function runBenchmarkRunCommand(ctx) {
57
57
  }
58
58
 
59
59
  /**
60
- * Decide the exit outcome when a run streamed zero records. A run that emits no
61
- * records normally did nothing (no tasks discovered, or the agent never
62
- * produced output) — a failure, surfaced loudly so CI does not go green on an
63
- * empty benchmark. The one exception is a deliberately-empty shard: a
64
- * high-index `--shard=i/N` with `N > cell count` legitimately selects zero
65
- * cells, so it exits 0 with a stderr note. Exported so the relaxed-guard branch
66
- * is testable without the full handler's config/SDK setup.
60
+ * Decide the exit outcome when a run streamed zero records. A run that emits
61
+ * no records normally did nothing. Either it discovered no tasks, or the
62
+ * agent never produced output. That is a failure. The command surfaces it
63
+ * loudly so CI does not go green on an empty benchmark. A deliberately-empty
64
+ * shard is the one exception. A high-index `--shard=i/N` with
65
+ * `N > cell count` legitimately selects zero cells, so it exits 0 with a
66
+ * stderr note. This function is exported so a test can reach the
67
+ * relaxed-guard branch without the full handler's config and SDK setup.
67
68
  * @param {{shard: {index: number, total: number} | null}} opts
68
69
  * @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
69
70
  * @returns {{ok: true} | {ok: false, code: number, error: string}}
@@ -79,16 +80,17 @@ export function resolveZeroRecordOutcome(opts, runtime) {
79
80
  ok: false,
80
81
  code: 1,
81
82
  error:
82
- "benchmark produced no result records — no task ran to completion; check the family's tasks/, apm install, and agent availability (ANTHROPIC_API_KEY / claude CLI / IS_SANDBOX)",
83
+ "benchmark produced no result records. No task ran to completion. Check the family's tasks/, apm install, and agent availability (ANTHROPIC_API_KEY / claude CLI / IS_SANDBOX)",
83
84
  };
84
85
  }
85
86
 
86
87
  /**
87
- * Parse and validate benchmark run options. Exported so tests can verify
88
- * defaults, including the resolved work tracker.
88
+ * Parse and validate benchmark run options. This function is exported so a
89
+ * test can verify the defaults and the resolved work tracker.
89
90
  * @param {Record<string, string|undefined>} values - Parsed option values
90
- * @param {Record<string, string|undefined>} [env] - Process environment, read
91
- * for the `LIBHARNESS_WORK_TRACKER` fallback when `--work-tracker` is absent.
91
+ * @param {Record<string, string|undefined>} [env] - Process environment. The
92
+ * parser reads it for the `LIBHARNESS_WORK_TRACKER` fallback when
93
+ * `--work-tracker` is absent.
92
94
  * @returns {object}
93
95
  */
94
96
  export function parseRunOptions(values, env = {}) {
@@ -132,7 +134,8 @@ function parseMaxTurns(raw) {
132
134
 
133
135
  /**
134
136
  * Parse a `--shard=<i>/<N>` selector into `{index, total}` (1-based), or `null`
135
- * for an unsharded run. Validates `1 ≤ index ≤ total` with integer parts.
137
+ * for an unsharded run. Validate that the parts are integers and that
138
+ * `1 ≤ index ≤ total`.
136
139
  * @param {string|undefined} raw
137
140
  * @returns {{index: number, total: number} | null}
138
141
  */
@@ -147,17 +150,17 @@ export function parseShard(raw) {
147
150
  return { index, total };
148
151
  }
149
152
 
150
- // Conservative because each cell spawns ~3 agent subprocesses (lead +
151
- // agent-under-test + judge); a low ceiling keeps a single runner from
152
- // thrashing. The bulk of the CI speedup comes from Layer-2 sharding across
153
- // machines, not from raising this in-job default.
153
+ // The ceiling is conservative because each cell spawns ~3 agent subprocesses
154
+ // (lead + agent-under-test + judge). A low ceiling makes sure that a single
155
+ // runner does not thrash. Most of the CI speedup comes from Layer-2 shards
156
+ // across machines. It does not come from a higher in-job default.
154
157
  const CONCURRENCY_CEILING = 4;
155
158
 
156
159
  /**
157
160
  * Resolve the cell concurrency: `--concurrency` flag > the
158
161
  * `LIBHARNESS_BENCHMARK_CONCURRENCY` env var > a CPU-aware default of
159
- * `min(CONCURRENCY_CEILING, max(2, ⌊cores/2⌋))`. The default is `> 1` so
160
- * concurrency is on transparently without any consumer opting in.
162
+ * `min(CONCURRENCY_CEILING, max(2, ⌊cores/2⌋))`. The default is `> 1`, so
163
+ * concurrency is on transparently. No consumer needs to opt in.
161
164
  * @param {Record<string, string|undefined>} values
162
165
  * @param {Record<string, string|undefined>} [env]
163
166
  * @returns {number}
@@ -4,11 +4,11 @@ const FIRST_LINE_CAP = 64 * 1024;
4
4
 
5
5
  /**
6
6
  * Read the first newline-terminated line of a file, bounded to the first
7
- * {@link FIRST_LINE_CAP} bytes. Trace `.ndjson` files can be many MB; the
8
- * Step 2.6 meta header is always small, so a bounded positional read avoids
9
- * loading whole files into memory just to inspect the header. The positional
10
- * `openSync`/`readSync`/`closeSync` trio is read off the injected
11
- * `runtime.fsSync` surface.
7
+ * {@link FIRST_LINE_CAP} bytes. Trace `.ndjson` files can be many MB. The
8
+ * Step 2.6 meta header is always small. So a bounded positional read does
9
+ * not load whole files into memory just to inspect the header. This function
10
+ * reads the positional `openSync`/`readSync`/`closeSync` trio off the
11
+ * injected `runtime.fsSync` surface.
12
12
  *
13
13
  * @param {object} fsSync - Sync filesystem surface (`runtime.fsSync`).
14
14
  * @param {string} path
@@ -30,9 +30,9 @@ function readFirstLine(fsSync, path) {
30
30
  /**
31
31
  * Scan a directory for `.ndjson` files whose meta header carries the
32
32
  * given discussion_id. The Step 2.6 first-line guarantee makes the
33
- * lookup cheap: we read only the first line per file. Files without a
34
- * meta header (e.g. legacy supervise/facilitate traces) are skipped
35
- * silently — not erroneous.
33
+ * lookup cheap. This function reads only the first line per file. It
34
+ * silently skips files without a meta header (e.g. legacy
35
+ * supervise/facilitate traces). Such a file is not an error.
36
36
  *
37
37
  * @param {string} dir
38
38
  * @param {string} discussionId
@@ -72,10 +72,10 @@ export function findTracesByDiscussion(dir, discussionId, fsSync) {
72
72
  }
73
73
 
74
74
  /**
75
- * `fit-trace by-discussion <discussion-id> [trace-dir]` — list trace
75
+ * `gemba-trace by-discussion <discussion-id> [trace-dir]` — list trace
76
76
  * files whose meta header carries the given discussion_id, one per
77
- * line, ordered by first-event timestamp (file mtime ascending). The
78
- * result is usable with `xargs cat` for a chronological merge.
77
+ * line. Order them by first-event timestamp (file mtime ascending).
78
+ * You can use the result with `xargs cat` for a chronological merge.
79
79
  *
80
80
  * @param {import("@forwardimpact/libcli").InvocationContext} ctx
81
81
  * @returns {Promise<{ok: true} | {ok: false, code: number, error: string}>}
@@ -3,12 +3,12 @@ import { sumTraceCost } from "../cost.js";
3
3
  /**
4
4
  * Scan an NDJSON trace and return the last orchestrator summary event,
5
5
  * the first `meta` event's `discussion_id`, and any structured replies
6
- * collected by the discusser. Skips malformed lines.
6
+ * the discusser collected. This function skips malformed lines.
7
7
  *
8
- * The runner is verdict-agnostic — verbatim passthrough of whatever the
9
- * trace carries ("success"/"failure" from supervise/facilitate; canonical
10
- * "adjourned"/"recessed"/"failed" from discuss). The bridge layer maps to
11
- * its channel semantics.
8
+ * The runner is verdict-agnostic. It passes through whatever the trace
9
+ * carries, verbatim ("success"/"failure" from supervise/facilitate;
10
+ * canonical "adjourned"/"recessed"/"failed" from discuss). The bridge
11
+ * layer maps to its channel semantics.
12
12
  *
13
13
  * @param {string} content - Raw NDJSON trace content.
14
14
  * @returns {{verdict: string, summary: string, replies: object[], trigger?: object, discussionId?: string} | null}
@@ -53,9 +53,9 @@ function readTraceSummary(content) {
53
53
  }
54
54
 
55
55
  /**
56
- * Callback command — read an NDJSON trace, extract the terminal
57
- * orchestrator summary, and POST a canonical callback body to the
58
- * configured URL. Used by `kata-dispatch.yml` to deliver the lead's
56
+ * Callback command — read an NDJSON trace and extract the terminal
57
+ * orchestrator summary. POST a canonical callback body to the configured
58
+ * URL. `kata-dispatch.yml` uses this command to deliver the lead's
59
59
  * conclusion to the bridge that dispatched the run.
60
60
  *
61
61
  * Wire shape (single shape across modes):
@@ -87,11 +87,11 @@ export async function runCallbackCommand(ctx) {
87
87
  const content = runtime.fsSync.readFileSync(traceFile, "utf8");
88
88
  const found = readTraceSummary(content) ?? {
89
89
  verdict: "failed",
90
- summary: "Run ended without producing a summary.",
90
+ summary: "The run ended and produced no summary.",
91
91
  replies: [],
92
92
  };
93
- // Total spend across every participant in the trace — the bridge surfaces
94
- // it alongside the verdict so a dispatched run reports what it cost.
93
+ // Total spend across every participant in the trace. The bridge surfaces
94
+ // it alongside the verdict, so a dispatched run reports what it cost.
95
95
  const { totalCostUsd } = sumTraceCost(content.split("\n"));
96
96
 
97
97
  const discussionId = found.discussionId ?? discussionIdOverride ?? null;