@forwardimpact/libharness 3.0.0 → 3.0.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (69) hide show
  1. package/README.md +60 -57
  2. package/package.json +2 -2
  3. package/src/advisor.js +47 -41
  4. package/src/agent-runner.js +57 -47
  5. package/src/benchmark/apm-installer.js +28 -28
  6. package/src/benchmark/env-loader.js +24 -16
  7. package/src/benchmark/grade.js +44 -41
  8. package/src/benchmark/hidden-tests.js +71 -39
  9. package/src/benchmark/hook-env.js +11 -9
  10. package/src/benchmark/invariants.js +20 -17
  11. package/src/benchmark/judge.js +29 -28
  12. package/src/benchmark/npm-installer.js +9 -8
  13. package/src/benchmark/report.js +53 -50
  14. package/src/benchmark/result.js +24 -23
  15. package/src/benchmark/runner.js +75 -69
  16. package/src/benchmark/scheduler.js +17 -16
  17. package/src/benchmark/task-family.js +28 -26
  18. package/src/benchmark/trace-split.js +8 -7
  19. package/src/benchmark/workdir.js +27 -25
  20. package/src/claude-code-executable.js +11 -11
  21. package/src/commands/advisor-flags.js +8 -7
  22. package/src/commands/assert.js +16 -15
  23. package/src/commands/benchmark-definition.js +11 -11
  24. package/src/commands/benchmark-grade.js +12 -11
  25. package/src/commands/benchmark-report.js +5 -5
  26. package/src/commands/benchmark-run.js +31 -28
  27. package/src/commands/by-discussion.js +10 -10
  28. package/src/commands/callback.js +11 -11
  29. package/src/commands/discuss.js +8 -7
  30. package/src/commands/facilitate.js +15 -13
  31. package/src/commands/output.js +3 -2
  32. package/src/commands/run.js +14 -14
  33. package/src/commands/scan-logs.js +21 -19
  34. package/src/commands/selfedit.js +14 -14
  35. package/src/commands/supervise.js +11 -9
  36. package/src/commands/task-input.js +9 -9
  37. package/src/commands/tee.js +10 -9
  38. package/src/commands/trace.js +55 -42
  39. package/src/commands/work-tracker.js +4 -3
  40. package/src/cost.js +17 -17
  41. package/src/discuss-tools.js +16 -16
  42. package/src/discusser.js +39 -38
  43. package/src/events/github.js +54 -37
  44. package/src/facilitator.js +21 -21
  45. package/src/inbox-poller.js +4 -4
  46. package/src/judge.js +32 -30
  47. package/src/message-bus.js +12 -11
  48. package/src/orchestration-loop.js +35 -36
  49. package/src/orchestration-toolkit.js +58 -53
  50. package/src/orchestrator-helpers.js +2 -2
  51. package/src/profile-prompt.js +54 -53
  52. package/src/redaction.js +63 -57
  53. package/src/render/line-renderer.js +5 -5
  54. package/src/render/orchestrator-filter.js +3 -3
  55. package/src/render/palette.js +11 -9
  56. package/src/render/tool-hints.js +18 -15
  57. package/src/render/turn-renderer.js +4 -4
  58. package/src/reply-emitter.js +2 -2
  59. package/src/sequence-counter.js +4 -3
  60. package/src/signature-filter.js +7 -6
  61. package/src/supervisor.js +19 -18
  62. package/src/tee-writer.js +25 -25
  63. package/src/trace-collector.js +53 -48
  64. package/src/trace-github.js +53 -44
  65. package/src/trace-multi.js +15 -13
  66. package/src/trace-query.js +61 -52
  67. package/src/trace-render.js +18 -18
  68. package/src/trace-usage.js +31 -28
  69. package/src/transcript-recorder.js +24 -20
@@ -3,26 +3,26 @@
3
3
  *
4
4
  * `query()` spawns a native `claude` binary that the SDK resolves from its own
5
5
  * platform-specific optional dependency (`@anthropic-ai/claude-agent-sdk-<platform>`).
6
- * `bun build --compile` bundles the SDK's JavaScript but not that separate
7
- * native package — it is not part of the import graph — so a compiled fit-*
8
- * binary can't self-resolve it and `query()` throws "Native CLI binary for
9
- * <platform> not found".
6
+ * `bun build --compile` bundles the SDK's JavaScript. It does not bundle that
7
+ * separate native package, which is not part of the import graph. So a
8
+ * compiled fit-* binary cannot self-resolve it, and `query()` throws "Native
9
+ * CLI binary for <platform> not found".
10
10
  *
11
- * In a compiled binary we point the SDK at the standalone `claude` on PATH,
12
- * installed beside gemba-harness by the bootstrap action's `fit-install.sh`.
13
- * Running from source keeps `node_modules`, where the SDK resolves its own
14
- * version-matched binary, so there we return undefined and defer to the SDK.
11
+ * In a compiled binary we point the SDK at the standalone `claude` on PATH.
12
+ * The gemba-bootstrap action's `fit-install.sh` installs it beside gemba-harness. A
13
+ * run from source keeps `node_modules`, where the SDK resolves its own
14
+ * version-matched binary. There we return undefined and defer to the SDK.
15
15
  */
16
16
  import { LIBCLI_IS_COMPILED } from "@forwardimpact/libcli";
17
17
 
18
18
  /**
19
19
  * @param {object} [deps]
20
20
  * @param {(cmd: string) => string | null | undefined} [deps.which] -
21
- * PATH resolver (injected for testing).
21
+ * PATH resolver (tests inject it).
22
22
  * @param {boolean} [deps.isCompiled] -
23
- * Whether this is a `bun --compile` binary (injected for testing).
23
+ * Whether this is a `bun --compile` binary (tests inject it).
24
24
  * @returns {string | undefined} absolute path to `claude`, or undefined to
25
- * defer resolution to the SDK.
25
+ * let the SDK resolve it.
26
26
  */
27
27
  export function resolveClaudeCodeExecutable({
28
28
  which = defaultWhich,
@@ -1,15 +1,16 @@
1
1
  /**
2
- * Shared advisor-flag parsing for the four session-mode commands. The two
3
- * flags are identical everywhere: `--advisor-model` (no default — absent
4
- * means the Advisor tool is not offered) and `--advisor-max-uses`
5
- * (default 3), which is a usage error without the model flag.
2
+ * Shared advisor-flag parser for the four session-mode commands. The two
3
+ * flags are identical everywhere: `--advisor-model` and `--advisor-max-uses`.
4
+ * `--advisor-model` has no default. When it is absent, the commands do not
5
+ * offer the Advisor tool. `--advisor-max-uses` defaults to 3. It is a usage
6
+ * error without the model flag.
6
7
  */
7
8
 
8
9
  /**
9
10
  * Parse `--advisor-model` / `--advisor-max-uses` from parsed option values.
10
- * A malformed max-uses is a usage error, not a silent fallback: NaN would
11
- * make the budget check (`used >= maxUses`) permanently false and disable
12
- * the code-enforced cap the flag exists to guarantee.
11
+ * A malformed max-uses raises a usage error. It never falls back silently.
12
+ * NaN would make the budget check (`used >= maxUses`) permanently false. It
13
+ * would disable the code-enforced cap the flag exists to guarantee.
13
14
  * @param {object} values - Parsed option values from cli.parse()
14
15
  * @returns {{advisorModel: string|undefined, advisorMaxUses: number}}
15
16
  */
@@ -67,10 +67,10 @@ export function evaluateAssertion(values, args, fsSync) {
67
67
  }
68
68
 
69
69
  /**
70
- * Attach the check-row grading role: `--gate` marks a gate check, `--weight`
71
- * attaches a numeric weight (0 marks the row diagnostic). `--gate` with any
72
- * `--weight` — 0 included — is invalid: a stray weight must never silently
73
- * disarm a gate.
70
+ * Attach the grading role of the check row. `--gate` marks a gate check.
71
+ * `--weight` attaches a numeric weight, and 0 marks the row diagnostic.
72
+ * `--gate` with any `--weight` is invalid, and that includes 0. A stray
73
+ * weight must never silently disarm a gate.
74
74
  * @param {object} values
75
75
  * @param {{test: string, pass: boolean, message?: string}} output - Mutated.
76
76
  */
@@ -92,8 +92,9 @@ function applyGradingFlags(values, output) {
92
92
  }
93
93
 
94
94
  /**
95
- * Parse a `--weight` value; null when invalid. A blank string is invalid —
96
- * `Number("")` is 0, which would silently demote the check to a diagnostic.
95
+ * Parse a `--weight` value. Return null when it is invalid. A blank string is
96
+ * invalid, because `Number("")` is 0. That would silently demote the check to
97
+ * a diagnostic.
97
98
  * @param {string} raw
98
99
  * @returns {number | null}
99
100
  */
@@ -104,11 +105,11 @@ function parseWeight(raw) {
104
105
  }
105
106
 
106
107
  /**
107
- * The grading role an emit-then-fail row keeps: a failing check must not
108
- * lose its authored role — an errored gate that demoted to a scored row
109
- * would let a broken scaffold earn partial credit instead of zeroing the
110
- * score. Invalid or conflicting flags yield no role (the row fails as a
111
- * unit-weight scored check).
108
+ * The grading role an emit-then-fail row keeps. A check that fails must not
109
+ * lose its authored role. An errored gate that demoted to a scored row would
110
+ * let a broken scaffold earn partial credit. The score would not drop to
111
+ * zero. Invalid flags, or flags that conflict, yield no role. The row then
112
+ * fails as a unit-weight scored check.
112
113
  * @param {object} values
113
114
  * @returns {{gate?: true, weight?: number}}
114
115
  */
@@ -126,10 +127,10 @@ function errorRowRole(values) {
126
127
  * Run an assertion, write JSON to stdout, and return a failure envelope when
127
128
  * the assertion does not pass.
128
129
  *
129
- * Emit-then-fail on every failure path: an invalid grading flag or an
130
- * errored evaluation (e.g. `--grep` against a file the agent deleted) writes
131
- * a failing row before the nonzero exit, so a typo or a vanished target
132
- * shrinks the score, never the denominator.
130
+ * Emit-then-fail applies on every failure path. An invalid grading flag
131
+ * writes a failed row before the nonzero exit. An errored evaluation does the
132
+ * same, for example `--grep` against a file the agent deleted. So a typo or a
133
+ * vanished target shrinks the score. It never shrinks the denominator.
133
134
  * @param {import("@forwardimpact/libcli").InvocationContext} ctx
134
135
  * @returns {Promise<{ok: true} | {ok: false, code: number, error: string}>}
135
136
  */
@@ -1,7 +1,7 @@
1
1
  /**
2
- * `gemba-benchmark` CLI definition. Lives in `src/` so the bin stays an
3
- * execute-on-import entry point — launcher packages import the bin to run
4
- * it — while tests import the definition without running the CLI.
2
+ * `gemba-benchmark` CLI definition. It lives in `src/` so the bin stays an
3
+ * execute-on-import entry point. Launcher packages import the bin to run it.
4
+ * Tests import the definition and never run the CLI.
5
5
  */
6
6
 
7
7
  import { runBenchmarkRunCommand } from "./benchmark-run.js";
@@ -36,7 +36,7 @@ export const definition = {
36
36
  "skills-from": {
37
37
  type: "string",
38
38
  description:
39
- "Stage .claude/ from this directory (a root containing .claude/) instead of running apm install — exercise local, unpublished skills",
39
+ "Stage .claude/ from this directory (a root that holds .claude/) instead of an apm install, to exercise local, unpublished skills",
40
40
  },
41
41
  output: {
42
42
  type: "string",
@@ -85,7 +85,7 @@ export const definition = {
85
85
  shard: {
86
86
  type: "string",
87
87
  description:
88
- "Run only shard i of N as i/N (1-based; default: the whole family). Each shard writes a partial results.jsonl; report --input merges them.",
88
+ "Run only shard i of N as i/N (1-based; default: the whole family). Each shard writes a partial results.jsonl. report --input merges them.",
89
89
  },
90
90
  "allowed-tools": {
91
91
  type: "string",
@@ -99,7 +99,7 @@ export const definition = {
99
99
  args: [],
100
100
  handler: runBenchmarkGradeCommand,
101
101
  description:
102
- "Grade a single task against a post-run workdir without invoking an agent: run the hidden test suite and the invariants script, then derive the verdict from the check rows (the exit mirrors it).",
102
+ "Grade a single task against a post-run workdir with no agent. Run the hidden test suite and the invariants script. Then derive the verdict from the check rows (the exit mirrors it).",
103
103
  options: {
104
104
  family: {
105
105
  type: "string",
@@ -112,7 +112,7 @@ export const definition = {
112
112
  "run-dir": {
113
113
  type: "string",
114
114
  description:
115
- "Post-run directory whose cwd/ subdir is the agent CWD; both producers run against that cwd — the path hooks receive as $AGENT_CWD",
115
+ "Post-run directory whose cwd/ subdir is the agent CWD. Both producers run against that cwd. Hooks receive it as $AGENT_CWD",
116
116
  },
117
117
  output: {
118
118
  type: "string",
@@ -125,12 +125,12 @@ export const definition = {
125
125
  args: [],
126
126
  handler: runBenchmarkReportCommand,
127
127
  description:
128
- "Aggregate result records into pass@k via the OpenAI HumanEval estimator.",
128
+ "Aggregate result records into pass@k with the OpenAI HumanEval estimator.",
129
129
  options: {
130
130
  input: {
131
131
  type: "string",
132
132
  description:
133
- "Run-output directory containing results.jsonl (default: benchmark-runs)",
133
+ "Run-output directory that holds results.jsonl (default: benchmark-runs)",
134
134
  },
135
135
  k: {
136
136
  type: "string",
@@ -143,7 +143,7 @@ export const definition = {
143
143
  detail: {
144
144
  type: "string",
145
145
  description:
146
- "Text report verbosity (full|compact, default: full). compact omits per-task detail — useful for sharded run summaries.",
146
+ "Text report verbosity (full|compact, default: full). compact omits per-task detail, which helps with sharded run summaries.",
147
147
  },
148
148
  },
149
149
  },
@@ -174,7 +174,7 @@ export const definition = {
174
174
  title: "Automate with GitHub Actions",
175
175
  url: "https://www.forwardimpact.team/docs/libraries/prove-changes/run-benchmark/ci-workflow/index.md",
176
176
  description:
177
- "Run benchmarks in CI with the forwardimpact/benchmark action.",
177
+ "Run benchmarks in CI with the forwardimpact/gemba-benchmark action.",
178
178
  },
179
179
  ],
180
180
  };
@@ -1,9 +1,9 @@
1
1
  /**
2
2
  * `gemba-benchmark grade` — run both check-row producers (the hidden test
3
- * suite and the invariants script) against a post-run workdir directory and
4
- * grade the merged rows with the same derivation the benchmark runner uses.
5
- * No agent and no judge run, so authors validate a task's grading material
6
- * against fixtures without paying for agent sessions; the process exit
3
+ * suite and the invariants script) against a post-run workdir directory.
4
+ * Grade the merged rows with the same derivation the benchmark runner uses.
5
+ * No agent runs and no judge runs. So an author validates a task's grading
6
+ * material against fixtures, and pays for no agent session. The process exit
7
7
  * mirrors the graded verdict.
8
8
  */
9
9
 
@@ -46,12 +46,12 @@ export async function runBenchmarkGradeCommand(ctx) {
46
46
  runInvariants,
47
47
  runHiddenTests,
48
48
  });
49
- // Same effective-score rule as the runner, minus the judge (none runs
50
- // here): an unhealthy grader or a failing gate zeroes the score, so a
51
- // crashed hook can never mint marks from the rows it emitted before dying.
52
- // Unlike a runner record — where `grade.score` stays the raw fraction and
53
- // the zeroing lands on the top-level `score` — this record has no second
54
- // field, so `grade.score` carries the effective value here.
49
+ // The effective-score rule matches the runner, minus the judge (none runs
50
+ // here). An unhealthy grader or a gate that fails zeroes the score. So a
51
+ // crashed hook can never mint marks from the rows it emitted before it
52
+ // died. A runner record keeps the raw fraction in `grade.score` and zeroes
53
+ // the top-level `score` instead. This record has no second field, so
54
+ // `grade.score` carries the effective value here.
55
55
  if (grade.score !== undefined && !(healthy && grade.gatesPass)) {
56
56
  grade.score = 0;
57
57
  }
@@ -65,7 +65,8 @@ export async function runBenchmarkGradeCommand(ctx) {
65
65
  ...(engineError && { error: engineError.message }),
66
66
  },
67
67
  }),
68
- // Mirrors the script for diagnosis; the graded verdict drives the exit.
68
+ // This mirrors the script for diagnosis. The graded verdict drives the
69
+ // exit.
69
70
  exitCode: invariants.exitCode,
70
71
  };
71
72
  validateGradeRecord(record);
@@ -1,9 +1,9 @@
1
1
  /**
2
- * `gemba-benchmark report` — aggregate `results.jsonl` into pass@k via the
3
- * OpenAI HumanEval estimator. Output is JSON by default; pass --format=text
4
- * to render a markdown table. --detail=compact drops the per-task detail
5
- * sections so a sharded run's per-shard summary stays short (the merge job
6
- * renders the full report over the combined ledger).
2
+ * `gemba-benchmark report` — aggregate `results.jsonl` into pass@k with the
3
+ * OpenAI HumanEval estimator. The command writes JSON by default. Pass
4
+ * --format=text to render a markdown table. --detail=compact drops the
5
+ * per-task detail sections, so a sharded run's per-shard summary stays short.
6
+ * The merge job renders the full report over the combined ledger.
7
7
  */
8
8
 
9
9
  import { resolve } from "node:path";
@@ -1,7 +1,7 @@
1
1
  /**
2
- * `gemba-benchmark run` — run every task in a family for N runs, stream each
3
- * ResultRecord to stdout (one JSON line per record), and append to the
4
- * canonical `<output>/results.jsonl` for the report subcommand.
2
+ * `gemba-benchmark run` — run every task in a family for N runs. Stream each
3
+ * ResultRecord to stdout (one JSON line per record). Append each record to
4
+ * the canonical `<output>/results.jsonl` for the report subcommand.
5
5
  */
6
6
 
7
7
  import { resolve } from "node:path";
@@ -30,15 +30,15 @@ export async function runBenchmarkRunCommand(ctx) {
30
30
  }
31
31
  const config = await createConfig("script", "benchmark");
32
32
  runtime.proc.env.ANTHROPIC_API_KEY = await config.anthropicToken();
33
- // The benchmark agent runs via createBenchmarkRunner, not the supervise
34
- // command, so the active-tracker env must land here before the runner
35
- // spawns the subprocess that inherits process.env.
33
+ // The benchmark agent runs through createBenchmarkRunner. The supervise
34
+ // command does not run it. So the active-tracker env must land here before
35
+ // the runner spawns the subprocess that inherits process.env.
36
36
  runtime.proc.env.LIBHARNESS_WORK_TRACKER = opts.workTracker;
37
37
 
38
38
  // The Claude Agent SDK spawns a `claude` subprocess that inherits
39
- // process.env. NODE_EXTRA_CA_CERTS causes undici (the HTTP client
40
- // inside that subprocess) to fail with UND_ERR_INVALID_ARG on
41
- // Node 22+, aborting every API call after 10 retries. Strip it
39
+ // process.env. NODE_EXTRA_CA_CERTS makes undici (the HTTP client
40
+ // inside that subprocess) fail with UND_ERR_INVALID_ARG on Node 22+.
41
+ // Undici then aborts every API call after 10 retries. Strip it
42
42
  // before the SDK loads so the subprocess gets a clean environment.
43
43
  delete runtime.proc.env.NODE_EXTRA_CA_CERTS;
44
44
 
@@ -57,13 +57,14 @@ export async function runBenchmarkRunCommand(ctx) {
57
57
  }
58
58
 
59
59
  /**
60
- * Decide the exit outcome when a run streamed zero records. A run that emits no
61
- * records normally did nothing (no tasks discovered, or the agent never
62
- * produced output) — a failure, surfaced loudly so CI does not go green on an
63
- * empty benchmark. The one exception is a deliberately-empty shard: a
64
- * high-index `--shard=i/N` with `N > cell count` legitimately selects zero
65
- * cells, so it exits 0 with a stderr note. Exported so the relaxed-guard branch
66
- * is testable without the full handler's config/SDK setup.
60
+ * Decide the exit outcome when a run streamed zero records. A run that emits
61
+ * no records normally did nothing. Either it discovered no tasks, or the
62
+ * agent never produced output. That is a failure. The command surfaces it
63
+ * loudly so CI does not go green on an empty benchmark. A deliberately-empty
64
+ * shard is the one exception. A high-index `--shard=i/N` with
65
+ * `N > cell count` legitimately selects zero cells, so it exits 0 with a
66
+ * stderr note. This function is exported so a test can reach the
67
+ * relaxed-guard branch without the full handler's config and SDK setup.
67
68
  * @param {{shard: {index: number, total: number} | null}} opts
68
69
  * @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
69
70
  * @returns {{ok: true} | {ok: false, code: number, error: string}}
@@ -79,16 +80,17 @@ export function resolveZeroRecordOutcome(opts, runtime) {
79
80
  ok: false,
80
81
  code: 1,
81
82
  error:
82
- "benchmark produced no result records — no task ran to completion; check the family's tasks/, apm install, and agent availability (ANTHROPIC_API_KEY / claude CLI / IS_SANDBOX)",
83
+ "benchmark produced no result records. No task ran to completion. Check the family's tasks/, apm install, and agent availability (ANTHROPIC_API_KEY / claude CLI / IS_SANDBOX)",
83
84
  };
84
85
  }
85
86
 
86
87
  /**
87
- * Parse and validate benchmark run options. Exported so tests can verify
88
- * defaults, including the resolved work tracker.
88
+ * Parse and validate benchmark run options. This function is exported so a
89
+ * test can verify the defaults and the resolved work tracker.
89
90
  * @param {Record<string, string|undefined>} values - Parsed option values
90
- * @param {Record<string, string|undefined>} [env] - Process environment, read
91
- * for the `LIBHARNESS_WORK_TRACKER` fallback when `--work-tracker` is absent.
91
+ * @param {Record<string, string|undefined>} [env] - Process environment. The
92
+ * parser reads it for the `LIBHARNESS_WORK_TRACKER` fallback when
93
+ * `--work-tracker` is absent.
92
94
  * @returns {object}
93
95
  */
94
96
  export function parseRunOptions(values, env = {}) {
@@ -132,7 +134,8 @@ function parseMaxTurns(raw) {
132
134
 
133
135
  /**
134
136
  * Parse a `--shard=<i>/<N>` selector into `{index, total}` (1-based), or `null`
135
- * for an unsharded run. Validates `1 ≤ index ≤ total` with integer parts.
137
+ * for an unsharded run. Validate that the parts are integers and that
138
+ * `1 ≤ index ≤ total`.
136
139
  * @param {string|undefined} raw
137
140
  * @returns {{index: number, total: number} | null}
138
141
  */
@@ -147,17 +150,17 @@ export function parseShard(raw) {
147
150
  return { index, total };
148
151
  }
149
152
 
150
- // Conservative because each cell spawns ~3 agent subprocesses (lead +
151
- // agent-under-test + judge); a low ceiling keeps a single runner from
152
- // thrashing. The bulk of the CI speedup comes from Layer-2 sharding across
153
- // machines, not from raising this in-job default.
153
+ // The ceiling is conservative because each cell spawns ~3 agent subprocesses
154
+ // (lead + agent-under-test + judge). A low ceiling makes sure that a single
155
+ // runner does not thrash. Most of the CI speedup comes from Layer-2 shards
156
+ // across machines. It does not come from a higher in-job default.
154
157
  const CONCURRENCY_CEILING = 4;
155
158
 
156
159
  /**
157
160
  * Resolve the cell concurrency: `--concurrency` flag > the
158
161
  * `LIBHARNESS_BENCHMARK_CONCURRENCY` env var > a CPU-aware default of
159
- * `min(CONCURRENCY_CEILING, max(2, ⌊cores/2⌋))`. The default is `> 1` so
160
- * concurrency is on transparently without any consumer opting in.
162
+ * `min(CONCURRENCY_CEILING, max(2, ⌊cores/2⌋))`. The default is `> 1`, so
163
+ * concurrency is on transparently. No consumer needs to opt in.
161
164
  * @param {Record<string, string|undefined>} values
162
165
  * @param {Record<string, string|undefined>} [env]
163
166
  * @returns {number}
@@ -4,11 +4,11 @@ const FIRST_LINE_CAP = 64 * 1024;
4
4
 
5
5
  /**
6
6
  * Read the first newline-terminated line of a file, bounded to the first
7
- * {@link FIRST_LINE_CAP} bytes. Trace `.ndjson` files can be many MB; the
8
- * Step 2.6 meta header is always small, so a bounded positional read avoids
9
- * loading whole files into memory just to inspect the header. The positional
10
- * `openSync`/`readSync`/`closeSync` trio is read off the injected
11
- * `runtime.fsSync` surface.
7
+ * {@link FIRST_LINE_CAP} bytes. Trace `.ndjson` files can be many MB. The
8
+ * Step 2.6 meta header is always small. So a bounded positional read does
9
+ * not load whole files into memory just to inspect the header. This function
10
+ * reads the positional `openSync`/`readSync`/`closeSync` trio off the
11
+ * injected `runtime.fsSync` surface.
12
12
  *
13
13
  * @param {object} fsSync - Sync filesystem surface (`runtime.fsSync`).
14
14
  * @param {string} path
@@ -30,9 +30,9 @@ function readFirstLine(fsSync, path) {
30
30
  /**
31
31
  * Scan a directory for `.ndjson` files whose meta header carries the
32
32
  * given discussion_id. The Step 2.6 first-line guarantee makes the
33
- * lookup cheap: we read only the first line per file. Files without a
34
- * meta header (e.g. legacy supervise/facilitate traces) are skipped
35
- * silently — not erroneous.
33
+ * lookup cheap. This function reads only the first line per file. It
34
+ * silently skips files without a meta header (e.g. legacy
35
+ * supervise/facilitate traces). Such a file is not an error.
36
36
  *
37
37
  * @param {string} dir
38
38
  * @param {string} discussionId
@@ -74,8 +74,8 @@ export function findTracesByDiscussion(dir, discussionId, fsSync) {
74
74
  /**
75
75
  * `gemba-trace by-discussion <discussion-id> [trace-dir]` — list trace
76
76
  * files whose meta header carries the given discussion_id, one per
77
- * line, ordered by first-event timestamp (file mtime ascending). The
78
- * result is usable with `xargs cat` for a chronological merge.
77
+ * line. Order them by first-event timestamp (file mtime ascending).
78
+ * You can use the result with `xargs cat` for a chronological merge.
79
79
  *
80
80
  * @param {import("@forwardimpact/libcli").InvocationContext} ctx
81
81
  * @returns {Promise<{ok: true} | {ok: false, code: number, error: string}>}
@@ -3,12 +3,12 @@ import { sumTraceCost } from "../cost.js";
3
3
  /**
4
4
  * Scan an NDJSON trace and return the last orchestrator summary event,
5
5
  * the first `meta` event's `discussion_id`, and any structured replies
6
- * collected by the discusser. Skips malformed lines.
6
+ * the discusser collected. This function skips malformed lines.
7
7
  *
8
- * The runner is verdict-agnostic — verbatim passthrough of whatever the
9
- * trace carries ("success"/"failure" from supervise/facilitate; canonical
10
- * "adjourned"/"recessed"/"failed" from discuss). The bridge layer maps to
11
- * its channel semantics.
8
+ * The runner is verdict-agnostic. It passes through whatever the trace
9
+ * carries, verbatim ("success"/"failure" from supervise/facilitate;
10
+ * canonical "adjourned"/"recessed"/"failed" from discuss). The bridge
11
+ * layer maps to its channel semantics.
12
12
  *
13
13
  * @param {string} content - Raw NDJSON trace content.
14
14
  * @returns {{verdict: string, summary: string, replies: object[], trigger?: object, discussionId?: string} | null}
@@ -53,9 +53,9 @@ function readTraceSummary(content) {
53
53
  }
54
54
 
55
55
  /**
56
- * Callback command — read an NDJSON trace, extract the terminal
57
- * orchestrator summary, and POST a canonical callback body to the
58
- * configured URL. Used by `kata-dispatch.yml` to deliver the lead's
56
+ * Callback command — read an NDJSON trace and extract the terminal
57
+ * orchestrator summary. POST a canonical callback body to the configured
58
+ * URL. `kata-dispatch.yml` uses this command to deliver the lead's
59
59
  * conclusion to the bridge that dispatched the run.
60
60
  *
61
61
  * Wire shape (single shape across modes):
@@ -87,11 +87,11 @@ export async function runCallbackCommand(ctx) {
87
87
  const content = runtime.fsSync.readFileSync(traceFile, "utf8");
88
88
  const found = readTraceSummary(content) ?? {
89
89
  verdict: "failed",
90
- summary: "Run ended without producing a summary.",
90
+ summary: "The run ended and produced no summary.",
91
91
  replies: [],
92
92
  };
93
- // Total spend across every participant in the trace — the bridge surfaces
94
- // it alongside the verdict so a dispatched run reports what it cost.
93
+ // Total spend across every participant in the trace. The bridge surfaces
94
+ // it alongside the verdict, so a dispatched run reports what it cost.
95
95
  const { totalCostUsd } = sumTraceCost(content.split("\n"));
96
96
 
97
97
  const discussionId = found.discussionId ?? discussionIdOverride ?? null;
@@ -17,8 +17,8 @@ function parseAgentProfiles(raw, cwd, maxTurns) {
17
17
  }
18
18
 
19
19
  /**
20
- * Parse and validate discuss command options. Exported so tests can verify
21
- * defaults and the legacy-flag clean break.
20
+ * Parse and validate discuss command options. This function is exported so a
21
+ * test can verify the defaults and the legacy-flag clean break.
22
22
  * @param {object} values - Parsed option values
23
23
  * @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
24
24
  * @returns {object}
@@ -30,8 +30,8 @@ export function parseDiscussOptions(values, runtime) {
30
30
  );
31
31
 
32
32
  const profilesRaw = values["agent-profiles"];
33
- // `||` (not `??`) so an empty-string flag from a CI forwarder falls back to
34
- // the default rather than overriding it with "".
33
+ // `||` (not `??`) so an empty-string flag from a CI forwarder falls back
34
+ // to the default. The empty string does not override the default.
35
35
  const agentCwd = resolve(values["agent-cwd"] || ".");
36
36
 
37
37
  const maxTurnsRaw = values["max-turns"] || "40";
@@ -74,8 +74,8 @@ export function parseDiscussOptions(values, runtime) {
74
74
 
75
75
  /**
76
76
  * Discuss command — run a discusser-led session with suspend/resume
77
- * semantics, threading `discussion_id` through the trace so multi-run
78
- * conversations are queryable as one.
77
+ * semantics. The session threads `discussion_id` through the trace, so
78
+ * you can query multi-run conversations as one.
79
79
  *
80
80
  * @param {import("@forwardimpact/libcli").InvocationContext} ctx
81
81
  * @returns {Promise<{ok: boolean, code?: number, error?: string}>}
@@ -102,7 +102,8 @@ export async function runDiscussCommand(ctx) {
102
102
  runtime.proc.env.LIBHARNESS_AGENT_PROFILE = opts.leadProfile;
103
103
  }
104
104
  // Unconditional so the default "github" is observable to the agent's
105
- // active-tracker resolution, mirroring --agent-profile's env write above.
105
+ // active-tracker resolution. This mirrors --agent-profile's env write
106
+ // above.
106
107
  runtime.proc.env.LIBHARNESS_WORK_TRACKER = opts.workTracker;
107
108
 
108
109
  const { query } = await import("@anthropic-ai/claude-agent-sdk");
@@ -22,9 +22,9 @@ function parseAgentProfiles(raw, cwd, maxTurns) {
22
22
  }
23
23
 
24
24
  /**
25
- * Parse and validate facilitate command options. Exported for test
26
- * coverage of the `--max-turns` → per-agent threading contract; not part
27
- * of the package's public API.
25
+ * Parse and validate facilitate command options. This function is exported
26
+ * so a test can cover the contract that threads `--max-turns` to each
27
+ * agent. It is not part of the package's public API.
28
28
  * @param {object} values - Parsed option values
29
29
  * @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
30
30
  * @returns {object} Parsed options
@@ -37,17 +37,18 @@ export function parseFacilitateOptions(values, runtime) {
37
37
 
38
38
  const profilesRaw = values["agent-profiles"];
39
39
  if (!profilesRaw) throw new Error("--agent-profiles is required");
40
- // `||` (not `??`) so an empty-string flag from a CI forwarder falls back to
41
- // the default rather than overriding it with "".
40
+ // `||` (not `??`) so an empty-string flag from a CI forwarder falls back
41
+ // to the default. The empty string does not override the default.
42
42
  const agentCwd = resolve(values["agent-cwd"] || ".");
43
43
 
44
44
  const maxTurnsRaw = values["max-turns"] || "20";
45
45
  const maxTurns = maxTurnsRaw === "0" ? 0 : parseInt(maxTurnsRaw, 10);
46
46
 
47
- // Thread --max-turns into each participant: without this, every facilitated
48
- // agent silently falls back to the 50-turn default in facilitator.js even
49
- // when the caller raises the budget. Observed in run 26078312414 where
50
- // staff-engineer terminated at 51 turns despite --max-turns=200.
47
+ // Thread --max-turns into each participant. Without this, every
48
+ // facilitated agent silently falls back to the 50-turn default in
49
+ // facilitator.js, even when the caller raises the budget. Run
50
+ // 26078312414 showed this. In it, staff-engineer terminated at 51
51
+ // turns despite --max-turns=200.
51
52
  const agentConfigs = parseAgentProfiles(profilesRaw, agentCwd, maxTurns);
52
53
 
53
54
  return {
@@ -77,9 +78,9 @@ export async function runFacilitateCommand(ctx) {
77
78
  const runtime = ctx.deps.runtime;
78
79
  const opts = parseFacilitateOptions(ctx.options, runtime);
79
80
 
80
- // Build the redactor as the first observable side-effect after option
81
- // parsing — the env snapshot must freeze BEFORE any in-process
82
- // env writes the command performs (e.g. LIBHARNESS_AGENT_PROFILE).
81
+ // Build the redactor as the first observable side-effect after the parser
82
+ // reads the options. The env snapshot must freeze BEFORE any in-process
83
+ // env write the command performs (e.g. LIBHARNESS_AGENT_PROFILE).
83
84
  const redactor = createRedactor({ runtime });
84
85
 
85
86
  const fileStream = opts.outputPath
@@ -98,7 +99,8 @@ export async function runFacilitateCommand(ctx) {
98
99
  runtime.proc.env.LIBHARNESS_AGENT_PROFILE = opts.facilitatorProfile;
99
100
  }
100
101
  // Unconditional so the default "github" is observable to the agent's
101
- // active-tracker resolution, mirroring --agent-profile's env write above.
102
+ // active-tracker resolution. This mirrors --agent-profile's env write
103
+ // above.
102
104
  runtime.proc.env.LIBHARNESS_WORK_TRACKER = opts.workTracker;
103
105
 
104
106
  const { query } = await import("@anthropic-ai/claude-agent-sdk");
@@ -21,8 +21,9 @@ export async function runOutputCommand(ctx) {
21
21
  now: () => isoTimestamp(runtime.clock.now()),
22
22
  });
23
23
 
24
- // `runtime.proc.stdin` is an AsyncIterable of UTF-8 lines (newline-split by
25
- // the runtime), so each yielded value is exactly one NDJSON record.
24
+ // `runtime.proc.stdin` is an AsyncIterable of UTF-8 lines. The runtime
25
+ // splits them on newlines. So each yielded value is exactly one NDJSON
26
+ // record.
26
27
  for await (const line of runtime.proc.stdin) {
27
28
  collector.addLine(line);
28
29
  }