@forwardimpact/libharness 3.0.0 → 3.0.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (69) hide show
  1. package/README.md +60 -57
  2. package/package.json +2 -2
  3. package/src/advisor.js +47 -41
  4. package/src/agent-runner.js +57 -47
  5. package/src/benchmark/apm-installer.js +28 -28
  6. package/src/benchmark/env-loader.js +24 -16
  7. package/src/benchmark/grade.js +44 -41
  8. package/src/benchmark/hidden-tests.js +25 -24
  9. package/src/benchmark/hook-env.js +11 -9
  10. package/src/benchmark/invariants.js +20 -17
  11. package/src/benchmark/judge.js +29 -28
  12. package/src/benchmark/npm-installer.js +9 -8
  13. package/src/benchmark/report.js +53 -50
  14. package/src/benchmark/result.js +24 -23
  15. package/src/benchmark/runner.js +75 -69
  16. package/src/benchmark/scheduler.js +17 -16
  17. package/src/benchmark/task-family.js +28 -26
  18. package/src/benchmark/trace-split.js +8 -7
  19. package/src/benchmark/workdir.js +27 -25
  20. package/src/claude-code-executable.js +11 -11
  21. package/src/commands/advisor-flags.js +8 -7
  22. package/src/commands/assert.js +16 -15
  23. package/src/commands/benchmark-definition.js +11 -11
  24. package/src/commands/benchmark-grade.js +12 -11
  25. package/src/commands/benchmark-report.js +5 -5
  26. package/src/commands/benchmark-run.js +31 -28
  27. package/src/commands/by-discussion.js +10 -10
  28. package/src/commands/callback.js +11 -11
  29. package/src/commands/discuss.js +8 -7
  30. package/src/commands/facilitate.js +15 -13
  31. package/src/commands/output.js +3 -2
  32. package/src/commands/run.js +14 -14
  33. package/src/commands/scan-logs.js +21 -19
  34. package/src/commands/selfedit.js +14 -14
  35. package/src/commands/supervise.js +11 -9
  36. package/src/commands/task-input.js +9 -9
  37. package/src/commands/tee.js +10 -9
  38. package/src/commands/trace.js +55 -42
  39. package/src/commands/work-tracker.js +4 -3
  40. package/src/cost.js +17 -17
  41. package/src/discuss-tools.js +16 -16
  42. package/src/discusser.js +39 -38
  43. package/src/events/github.js +54 -37
  44. package/src/facilitator.js +21 -21
  45. package/src/inbox-poller.js +4 -4
  46. package/src/judge.js +32 -30
  47. package/src/message-bus.js +12 -11
  48. package/src/orchestration-loop.js +35 -36
  49. package/src/orchestration-toolkit.js +58 -53
  50. package/src/orchestrator-helpers.js +2 -2
  51. package/src/profile-prompt.js +54 -53
  52. package/src/redaction.js +63 -57
  53. package/src/render/line-renderer.js +5 -5
  54. package/src/render/orchestrator-filter.js +3 -3
  55. package/src/render/palette.js +11 -9
  56. package/src/render/tool-hints.js +18 -15
  57. package/src/render/turn-renderer.js +4 -4
  58. package/src/reply-emitter.js +2 -2
  59. package/src/sequence-counter.js +4 -3
  60. package/src/signature-filter.js +7 -6
  61. package/src/supervisor.js +19 -18
  62. package/src/tee-writer.js +25 -25
  63. package/src/trace-collector.js +53 -48
  64. package/src/trace-github.js +53 -44
  65. package/src/trace-multi.js +15 -13
  66. package/src/trace-query.js +61 -52
  67. package/src/trace-render.js +18 -18
  68. package/src/trace-usage.js +31 -28
  69. package/src/transcript-recorder.js +24 -20
package/README.md CHANGED
@@ -8,22 +8,23 @@ worked.
8
8
 
9
9
  <!-- END:description -->
10
10
 
11
- `libharness` provides the runtime and tool surface for multi-LLM coordination —
12
- an agent talks to a supervisor, a facilitator chairs a meeting, a lead drives
13
- an asynchronous discussion — plus a CLI suite that runs evals, queries the
14
- traces they produce, and edits skill files under controlled conditions.
11
+ `libharness` provides the runtime and tool surface for multi-LLM coordination.
12
+ An agent talks to a supervisor. A facilitator chairs a meeting. A lead drives
13
+ an asynchronous discussion. `libharness` also provides a CLI suite. The suite
14
+ runs evals, queries the traces they produce, and edits skill files under
15
+ controlled conditions.
15
16
 
16
17
  ## CLIs
17
18
 
18
19
  | CLI | Purpose |
19
20
  | --------------- | ---------------------------------------------------------------------- |
20
21
  | `gemba-harness` | Run agents in `run`/`supervise`/`facilitate`/`discuss` subcommands. |
21
- | `gemba-trace` | Download, query, and analyze NDJSON traces produced by `gemba-harness`. |
22
+ | `gemba-trace` | Download, query, and analyze the NDJSON traces `gemba-harness` produces. |
22
23
  | `gemba-benchmark` | Run task families for N runs each and aggregate pass@k. |
23
- | `gemba-selfedit` | Write stdin to `.claude/**` paths, gated by settings.json + branch. |
24
+ | `gemba-selfedit` | Write stdin to `.claude/**` paths behind settings.json and branch gates. |
24
25
 
25
26
  `gemba-harness`'s subcommands share one orchestration loop and one async tool
26
- surface, below. The `judge` role is a profile passed to `supervise`.
27
+ surface, below. The `judge` role is a profile you pass to `supervise`.
27
28
 
28
29
  ## Modes
29
30
 
@@ -36,8 +37,8 @@ surface, below. The `judge` role is a profile passed to `supervise`.
36
37
  | `judge` | `judge` | (none) | `Conclude` |
37
38
 
38
39
  `run` and `judge` are one-shot. The other three share `OrchestrationLoop`
39
- plus an async Ask/Answer/Announce/RollCall tool surface; the loop fans
40
- messages out over an in-memory bus and emits a `{source, seq, event}`
40
+ plus an async Ask/Answer/Announce/RollCall tool surface. The loop fans
41
+ messages out over an in-memory bus. It emits a `{source, seq, event}`
41
42
  NDJSON envelope for every line.
42
43
 
43
44
  ## Async Ask / Answer / Announce
@@ -48,10 +49,10 @@ Answer({ message, askId? }) → routed to the asker
48
49
  Announce({ message }) → broadcast, no reply expected
49
50
  ```
50
51
 
51
- Every Ask returns immediately and registers a pending entry keyed by an
52
+ Every Ask returns immediately. It registers a pending entry under an
52
53
  `askId`. The reply arrives later on the asker's inbox as `[answer#N]
53
54
  <participant>: <text>`. Broadcast: omit `to` on a multi-participant
54
- lead. Answer's `askId` is optional — the handler is forgiving:
55
+ lead. Answer's `askId` is optional. The handler is forgiving:
55
56
 
56
57
  - **Provided + matches an ask owed by the caller** → routes to that asker.
57
58
  - **Provided but unknown or wrong addressee** → `isError` with a pointed
@@ -69,34 +70,35 @@ Inbox lines on resume:
69
70
  ```
70
71
 
71
72
  Async means the lead can issue Asks, end its turn, and plan in the gap
72
- while participants work in parallel — nothing blocks the LLM thread.
73
+ while participants work in parallel. Nothing blocks the LLM thread.
73
74
 
74
75
  ### Discuss-mode replies
75
76
 
76
- In discussion mode, Answer calls routed to the lead are streamed to
77
- the discussion thread as they are produced — each agent's Answer becomes
78
- a separate reply posted immediately, not batched at session end. The
79
- lead and agents can also call `Acknowledge` to post brief messages
80
- directly to the thread (status updates, human follow-up responses).
81
- The message bus intercepts answers and appends them to `ctx.replies[]`.
77
+ In discussion mode, Answer calls routed to the lead stream to the
78
+ discussion thread as the agents produce them. Each agent's Answer becomes
79
+ a separate reply. The thread receives it immediately. The session does not
80
+ batch the replies at its end. The lead and agents can also call
81
+ `Acknowledge` to post brief messages directly to the thread (status
82
+ updates, human follow-up responses). The message bus intercepts answers
83
+ and appends them to `ctx.replies[]`.
82
84
 
83
85
  `RequestForComment` is a separate coordination tool available on agent
84
86
  roles (facilitated agents and discuss agents). It queues an intent to
85
87
  open a new Discussion thread for long-horizon coordination on open
86
- questions; these are accumulated in `ctx.rfcs[]`, separate from the
87
- thread replies in `ctx.replies[]`.
88
+ questions. These intents accumulate in `ctx.rfcs[]`. They stay separate
89
+ from the thread replies in `ctx.replies[]`.
88
90
 
89
91
  ## Orchestration loop
90
92
 
91
- Each participant drains the bus (or waits), runs/resumes the LLM with
92
- drained messages as tagged lines, and on an unanswered owed Ask injects
93
- one synthetic reminder before emitting `protocol_violation` and
94
- unblocking the asker with a synthetic null answer.
93
+ Each participant drains the bus, or waits. It then runs or resumes the LLM
94
+ with the drained messages as tagged lines. On an unanswered owed Ask, the
95
+ participant injects one synthetic reminder. It then emits
96
+ `protocol_violation`. It unblocks the asker with a synthetic null answer.
95
97
 
96
98
  Termination uses two flags. `ctx.concluded` is explicit
97
- `Conclude`/`Adjourn`/`Recess` — also cancels in-flight Asks so askers
98
- see why their question won't be answered. `stopped` is broader: lead
99
- error, agent crash, abort path. Loops watch `stopped`; `ctx.concluded`
99
+ `Conclude`/`Adjourn`/`Recess`. It also cancels in-flight Asks, so askers
100
+ see why nobody will answer their question. `stopped` is broader: lead
101
+ error, agent crash, abort path. Loops watch `stopped`. `ctx.concluded`
100
102
  only feeds the summary's `success`/`verdict`.
101
103
 
102
104
  ## Tool surface, by role
@@ -113,7 +115,7 @@ only feeds the summary's `success`/`verdict`.
113
115
 
114
116
  Ask's `to` accepts a participant name on multi-participant roles
115
117
  (facilitator, discuss lead, all participants). The supervise pair has
116
- only one possible target so `to` is rejected there.
118
+ only one possible target, so it rejects `to`.
117
119
 
118
120
  ## Minimal example: two-participant facilitator
119
121
 
@@ -137,14 +139,14 @@ const result = await facilitator.run("Run a kata storyboard meeting.");
137
139
  // result.success / result.turns / NDJSON trace on process.stdout
138
140
  ```
139
141
 
140
- The facilitator gets `Ask`/`Answer`/`Announce`/`RollCall`/`Conclude`;
141
- each agent gets the same minus `Conclude`. Every tool call, bus
142
+ The facilitator gets `Ask`/`Answer`/`Announce`/`RollCall`/`Conclude`.
143
+ Each agent gets the same minus `Conclude`. Every tool call, bus
142
144
  message, and orchestrator event becomes one trace line.
143
145
 
144
146
  ## Trace format and redaction
145
147
 
146
148
  Each line is `{ "source": "<participant|orchestrator>", "seq": N, "event":
147
- {…} }`. `seq` is monotonic across the whole trace; `orchestrator` emits
149
+ {…} }`. `seq` is monotonic across the whole trace. `orchestrator` emits
148
150
  `session_start`, `agent_start`, `protocol_violation`, `lead_turn_limit`,
149
151
  and `summary`. `event` is the SDK event verbatim or the orchestrator
150
152
  payload. `gemba-trace` consumes this format.
@@ -153,61 +155,62 @@ Redaction is on by default for `gemba-harness run`/`supervise`/`facilitate`
153
155
  and composes two layers:
154
156
 
155
157
  - **Env-var allowlist** — `ANTHROPIC_API_KEY`, `GH_TOKEN`, `GITHUB_TOKEN`
156
- by default; override with `LIBHARNESS_REDACTION_ENV_VARS=NAME1,…`
157
- (replaces, not extends). Runtime values become `[REDACTED:env:NAME]`
158
- everywhere they appear.
158
+ by default. Override the list with
159
+ `LIBHARNESS_REDACTION_ENV_VARS=NAME1,…`. The variable replaces the
160
+ default list. It does not extend the list. Runtime values become
161
+ `[REDACTED:env:NAME]` everywhere they appear.
159
162
  - **Credential-shape patterns** — `sk-ant-`, `ghp_`, `ghs_`, `gho_`,
160
163
  `github_pat_`. Hits become `[REDACTED:pattern:KIND]`.
161
164
 
162
- Set `LIBHARNESS_REDACTION_DISABLED=1` to disable (one stderr warning per
163
- run). Never on CI for a public repo — workflow artifacts are
164
- downloadable through retention.
165
+ Set `LIBHARNESS_REDACTION_DISABLED=1` to disable it (one stderr warning
166
+ per run). Never disable it on CI for a public repo. Workflow artifacts
167
+ stay downloadable through retention.
165
168
 
166
169
  ## Module map
167
170
 
168
171
  | Module | Purpose |
169
172
  | ----------------------------------------------------------- | -------------------------------------------------------------------- |
170
- | `agent-runner.js` | One Claude Agent SDK session; emits NDJSON via the redactor. |
173
+ | `agent-runner.js` | One Claude Agent SDK session. Emits NDJSON through the redactor. |
171
174
  | `message-bus.js` | Per-participant queues + `waitForMessages` Promise wakeup. |
172
175
  | `orchestration-toolkit.js` | Shared Ask/Answer/Announce/Conclude/RollCall/RequestForComment handlers + builders. |
173
- | `orchestration-loop.js` | Unified lead+participant loop; reminder/violation handling. |
176
+ | `orchestration-loop.js` | Unified lead+participant loop. Handles reminders and violations. |
174
177
  | `facilitator.js` / `supervisor.js` / `discusser.js` / `judge.js` | Per-mode class + factory + system prompt. |
175
178
  | `discuss-tools.js` | Discuss-only `Recess`/`Adjourn`/`Acknowledge`. |
176
179
  | `reply-emitter.js` | Fire-and-forget POST of reply/ack events to the callback URL. |
177
180
  | `inbox-poller.js` | Long-poll the bridge inbox for injected human messages. |
178
- | `trace-collector.js` / `trace-query.js` / `trace-github.js` | Trace ingestion / querying / GitHub-attachment helpers. |
181
+ | `trace-collector.js` / `trace-query.js` / `trace-github.js` | Trace ingestion / query / GitHub-attachment helpers. |
179
182
  | `redaction.js` | Env-var allowlist + credential-shape pattern redaction. |
180
183
 
181
184
  ## gemba-selfedit
182
185
 
183
- A narrow, audited bypass for sessions where `Edit`/`Write` (and bash
184
- writes) are blocked against paths the project's own allowlist permits.
185
- Reads stdin, writes the target, exits 0 / 2 (safeguard violation) / 1
186
- (I/O error).
186
+ A narrow, audited bypass for sessions that block `Edit`/`Write` (and bash
187
+ writes) against paths the project's own allowlist permits. It reads stdin.
188
+ It writes the target. It exits 0, 2 (safeguard violation), or 1 (I/O
189
+ error).
187
190
 
188
191
  ```sh
189
192
  echo "<content>" | bunx gemba-selfedit <path>
190
193
  ```
191
194
 
192
- Two safeguards, checked in order:
195
+ The CLI checks two safeguards in order:
193
196
 
194
197
  1. **Settings-allow.** Walk upward from the target with
195
198
  [`Finder.findUpward`](../libutil/src/finder.js) to find the nearest
196
199
  `.claude/settings.json`. The target relative to its grandparent
197
200
  directory must match at least one `Edit(<glob>)` rule in
198
- `permissions.allow[]` (matched with
199
- [`minimatch`](https://github.com/isaacs/minimatch), `dot: true`).
200
- Settings.json is the single source of truth — widen the project
201
- allowlist and the CLI follows. Traversal like `.claude/../README.md`
202
- is rejected as a side effect: `path.resolve` collapses `..` first,
203
- then the resolved path tests against the rules.
201
+ `permissions.allow[]`. The CLI matches with
202
+ [`minimatch`](https://github.com/isaacs/minimatch) and `dot: true`.
203
+ Settings.json is the single source of truth. Widen the project
204
+ allowlist and the CLI follows. The CLI also rejects traversal like
205
+ `.claude/../README.md` as a side effect. `path.resolve` collapses
206
+ `..` first. The resolved path then tests against the rules.
204
207
 
205
208
  2. **Branch scope.** `git rev-parse --abbrev-ref HEAD` must not be
206
209
  `HEAD` (detached) or `main`. Edits ride a feature branch through
207
- whatever merge gates the project has configured.
210
+ whatever merge gates the project configured.
208
211
 
209
- Failure messages name the safeguard that rejected; safeguard 1 also
210
- lists the `Edit()` rules that were tried.
212
+ Failure messages name the safeguard that rejected the write. Safeguard 1
213
+ also lists the `Edit()` rules it tried.
211
214
 
212
215
  ## Documentation
213
216
 
@@ -216,10 +219,10 @@ lists the `Edit()` rules that were tried.
216
219
  facilitate / discuss) with Ask/Answer/Announce and a single NDJSON trace.
217
220
  - [Run an Eval](https://www.forwardimpact.team/docs/libraries/prove-changes/run-eval/index.md)
218
221
  — author a judge profile, run an eval locally, wire it into CI, and inspect
219
- the resulting trace.
222
+ the trace it produces.
220
223
  - [Prove Agent Changes](https://www.forwardimpact.team/docs/libraries/prove-changes/index.md)
221
- — end-to-end workflow from dataset generation through evaluation to trace
222
- analysis, including multi-agent collaboration sessions.
224
+ — the end-to-end workflow from dataset generation through evaluation to
225
+ trace analysis, with multi-agent collaboration sessions.
223
226
  - [Analyze Traces](https://www.forwardimpact.team/docs/libraries/prove-changes/trace-analysis/index.md)
224
227
  — read the NDJSON traces produced by `gemba-harness` with `gemba-trace`.
225
228
  - [Agent Teams](https://www.forwardimpact.team/docs/products/agent-teams/index.md)
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@forwardimpact/libharness",
3
- "version": "3.0.0",
3
+ "version": "3.0.1",
4
4
  "description": "Autonomous agent team harness — coordinate a lead and participant agents in one async session, with eval, benchmark, and trace tooling to prove the changes worked.",
5
5
  "keywords": [
6
6
  "orchestration",
@@ -69,7 +69,7 @@
69
69
  "@forwardimpact/libtelemetry": "^0.1.22",
70
70
  "@forwardimpact/libutil": "^0.1.0",
71
71
  "jmespath": "^0.16.0",
72
- "minimatch": "^10.0.0",
72
+ "minimatch": "^10.2.6",
73
73
  "zod": "^4.4.3"
74
74
  },
75
75
  "devDependencies": {
package/src/advisor.js CHANGED
@@ -1,15 +1,15 @@
1
1
  /**
2
- * Advisor — the judge's mid-loop sibling: a solo, tool-restricted, one-shot
3
- * `AgentRunner` session on a stronger model whose final text is the advice.
4
- * Each consult forwards the caller's whole recorded context (system prompt,
5
- * delivered prompts, transcript so far) plus a focused question; the advisor
6
- * can inspect files read-only but holds no write, execute, subagent, or
7
- * orchestration tools and never appears on the message bus.
2
+ * Advisor — the judge's mid-loop sibling. It is a solo, tool-restricted,
3
+ * one-shot `AgentRunner` session on a stronger model. Its final text is the
4
+ * advice. Each consult forwards the caller's whole recorded context (system
5
+ * prompt, delivered prompts, transcript so far) plus a focused question. The
6
+ * advisor can inspect files read-only. It holds no write, execute, subagent,
7
+ * or orchestration tools. It never appears on the message bus.
8
8
  *
9
- * Consults are stateless — one fresh session per call, re-reading the
10
- * caller's context as it stands — and fail-open: timeout, error, and abort
11
- * all resolve to an in-band `{unavailable}` result so the caller's session
12
- * never stalls or crashes on a consult.
9
+ * Consults are stateless. Each call runs one fresh session that reads the
10
+ * caller's context as it stands. Consults also fail open. A timeout, an
11
+ * error, and an abort all resolve to an in-band `{unavailable}` result, so
12
+ * the caller's session never stalls or crashes on a consult.
13
13
  *
14
14
  * Follows OO+DI: factory function, tests inject a fake `query`.
15
15
  */
@@ -20,39 +20,42 @@ import { createAgentRunner } from "./agent-runner.js";
20
20
  import { composeSystemPrompt } from "./profile-prompt.js";
21
21
 
22
22
  /**
23
- * System-prompt trailer for the advisor session. Fixes the response
24
- * contract (spec criterion "Advice is bounded"): assessment,
25
- * recommendation, unsolicited findings, with a stated length ceiling.
23
+ * System-prompt trailer for the advisor session. It fixes the response
24
+ * contract (spec criterion "Advice is bounded"). The contract is an
25
+ * assessment, a recommendation, and unsolicited findings, with a stated
26
+ * length ceiling.
26
27
  */
27
28
  export const ADVISOR_SYSTEM_PROMPT =
28
- "You are a consulted specialist, not a worker. " +
29
- "Another agent paused its work to ask you one question; its full session context and the question are in the task. " +
30
- "You may Read, Glob, and Grep the files the transcript names to ground your advice; never modify anything. " +
31
- "Respond in one turn of prose — your final text is delivered to the caller verbatim. " +
29
+ "You are a consulted specialist. You are not a worker. " +
30
+ "Another agent paused its work to ask you one question. Its full session context and the question are in the task. " +
31
+ "You may Read, Glob, and Grep the files the transcript names to ground your advice. You must never modify anything. " +
32
+ "Respond in one turn of prose. The caller receives your final text verbatim. " +
32
33
  "Structure the response as: assessment (what you see), recommendation (what to do and why), and unsolicited findings (anything important the caller did not ask about). " +
33
34
  "Keep the whole response to at most three short paragraphs. " +
34
- "Do not ask follow-up questions — the caller cannot reply.";
35
+ "Do not ask follow-up questions. The caller cannot reply.";
35
36
 
36
37
  /**
37
38
  * Consult-guidance fragment for caller system prompts, present only when
38
- * the session runs with an advisor model. Steers the caller's judgment; it
39
- * mandates nothing.
39
+ * the session runs with an advisor model. It steers the caller's judgment.
40
+ * It mandates nothing.
40
41
  * @param {number} maxUses - The session-wide consult budget.
41
42
  * @returns {string}
42
43
  */
43
44
  export function advisorGuidance(maxUses) {
44
45
  return (
45
- "An `Advisor` tool is available: one focused question per call, answered by a stronger model that sees your full session context. " +
46
- "A consult pays off at hard decision points — architectural forks, unclear root causes, trade-offs you cannot rank — and early, before work builds on an unvalidated assumption. " +
46
+ "An `Advisor` tool is available. It takes one focused question per call. A stronger model that sees your full session context answers it. " +
47
+ "A consult pays off at hard decision points, such as architectural forks, unclear root causes, and trade-offs you cannot rank. " +
48
+ "A consult also pays off early, before work builds on an unvalidated assumption. " +
47
49
  "It does not pay off for routine reads, writes, or searches. " +
48
- `The session-wide budget is ${maxUses} consult${maxUses === 1 ? "" : "s"}, shared across all participants. ` +
49
- "Consulting is your judgment, never mandatory."
50
+ `The session-wide budget is ${maxUses} consult${maxUses === 1 ? "" : "s"}, which all participants share. ` +
51
+ "A consult is your judgment. It is never mandatory."
50
52
  );
51
53
  }
52
54
 
53
55
  /**
54
- * Create the session-wide consult budget, shared by every caller's tool
55
- * handler. Enforced in code by the tool handler, not in the prompt.
56
+ * Create the session-wide consult budget. Every caller's tool handler
57
+ * shares it. The tool handler enforces the budget in code. The prompt does
58
+ * not enforce it.
56
59
  * @param {number} maxUses
57
60
  * @returns {{maxUses: number, used: number}}
58
61
  */
@@ -62,7 +65,7 @@ export function createAdvisorBudget(maxUses) {
62
65
 
63
66
  /**
64
67
  * Fold the consult guidance into an existing run-specific amendment when
65
- * the advisor is enabled (budget present); return the amendment unchanged
68
+ * the advisor is enabled (budget present). Return the amendment unchanged
66
69
  * otherwise, so advisor-off prompts stay byte-identical.
67
70
  * @param {string|undefined} amend - The existing amendment, if any.
68
71
  * @param {{maxUses: number}|null} budget
@@ -73,16 +76,19 @@ export function withAdvisorGuidance(amend, budget) {
73
76
  return [amend, advisorGuidance(budget.maxUses)].filter(Boolean).join("\n\n");
74
77
  }
75
78
 
76
- /** Consult timeout — generous for a read-a-few-files-and-answer session, and the universal guard in modes with no stop path. */
79
+ /**
80
+ * Consult timeout. It is generous for a read-a-few-files-and-answer
81
+ * session. It is also the universal guard in modes with no stop path.
82
+ */
77
83
  export const DEFAULT_CONSULT_TIMEOUT_MS = 300_000;
78
84
 
79
85
  const ADVISOR_ALLOWED_TOOLS = ["Read", "Glob", "Grep"];
80
86
 
81
87
  // Under the harness's always-on bypassPermissions, `allowedTools` alone is
82
- // not structural — `disallowedTools` is what removes tools from the model's
83
- // context, the same treatment the lead runners get. The advisor's contract
84
- // is stricter than the leads' (read-only inspection, nothing else), so the
85
- // list also removes the write-capable and non-inspection built-ins the
88
+ // not structural. `disallowedTools` removes tools from the model's context.
89
+ // The lead runners get the same treatment. The advisor's contract is
90
+ // stricter than the leads' (read-only inspection, nothing else). So the
91
+ // list also removes the write-capable and non-inspection built-ins that the
86
92
  // lead convention leaves in.
87
93
  const ADVISOR_DISALLOWED_TOOLS = [
88
94
  "Bash",
@@ -105,18 +111,18 @@ const devNull = new Writable({
105
111
  });
106
112
 
107
113
  /**
108
- * Create a per-caller advisor closed over that caller's transcript
114
+ * Create a per-caller advisor that closes over that caller's transcript
109
115
  * recorder.
110
116
  *
111
117
  * @param {object} deps
112
118
  * @param {string} deps.model - Advisor model id.
113
- * @param {string} deps.cwd - The caller's working directory, so read-only inspection sees the caller's files.
114
- * @param {function} deps.query - SDK query function (injected for testing).
119
+ * @param {string} deps.cwd - The caller's working directory, so the read-only tools see the caller's files.
120
+ * @param {function} deps.query - SDK query function (tests inject it).
115
121
  * @param {{render: () => string}} deps.recorder - The caller's transcript recorder.
116
122
  * @param {import("./redaction.js").Redactor} deps.redactor
117
123
  * @param {import("@forwardimpact/libutil/runtime").Runtime} deps.runtime - Clock surface for timeout and duration.
118
- * @param {function} deps.onLine - Re-emitter for the advisor session's NDJSON lines (tagged `source: "advisor"` by the caller).
119
- * @param {number} [deps.maxTurns] - Default 5 — single-digit per the spec criterion.
124
+ * @param {function} deps.onLine - Re-emitter for the advisor session's NDJSON lines (the caller tags them `source: "advisor"`).
125
+ * @param {number} [deps.maxTurns] - Default 5, which is single-digit per the spec criterion.
120
126
  * @param {number} [deps.timeoutMs] - Default `DEFAULT_CONSULT_TIMEOUT_MS`.
121
127
  * @returns {{consult: (question: string) => Promise<{advice?: string, unavailable?: boolean, reason?: string, durationMs: number}>, abort: () => void}}
122
128
  */
@@ -147,7 +153,7 @@ export function createAdvisor({
147
153
  return {
148
154
  /**
149
155
  * Run one fresh advisor session over the caller's context as it
150
- * stands plus the question. Never rejects — every failure shape
156
+ * stands plus the question. It never rejects. Every failure shape
151
157
  * resolves to `{unavailable, reason}` (fail-open).
152
158
  * @param {string} question
153
159
  */
@@ -207,9 +213,9 @@ export function createAdvisor({
207
213
  },
208
214
 
209
215
  /**
210
- * Abort the in-flight consult, if any. A consult is a blocking tool
211
- * call, so one caller cannot overlap its own consults; advisors are
212
- * per-caller, so at most one runner is ever tracked.
216
+ * Abort the in-flight consult, if any. A consult is a tool call that
217
+ * blocks, so one caller cannot overlap its own consults. Each advisor
218
+ * belongs to one caller, so the code tracks at most one runner.
213
219
  */
214
220
  abort() {
215
221
  currentRunner?.currentAbortController?.abort();
@@ -1,7 +1,8 @@
1
1
  /**
2
2
  * AgentRunner — runs a single Claude Agent SDK session and emits raw
3
- * NDJSON events to an output stream. Building block for `gemba-harness run`,
4
- * `gemba-harness supervise`, `gemba-harness facilitate`, and `gemba-harness discuss`.
3
+ * NDJSON events to an output stream. `gemba-harness run`,
4
+ * `gemba-harness supervise`, `gemba-harness facilitate`, and
5
+ * `gemba-harness discuss` build on it.
5
6
  *
6
7
  * Follows OO+DI: constructor injection, factory function, tests bypass factory.
7
8
  */
@@ -12,15 +13,16 @@ import { resolveClaudeCodeExecutable } from "./claude-code-executable.js";
12
13
  const DEFAULT_ALLOWED_TOOLS = ["Bash", "Read", "Glob", "Grep", "Write", "Edit"];
13
14
 
14
15
  /**
15
- * Did the session actually invoke the model? A genuine run always bills
16
+ * Report whether the session invoked the model. A genuine run always bills
16
17
  * tokens (the system prompt alone is thousands of input tokens) and costs
17
- * more than zero. A `result` message with `subtype: "success"` but zero
18
- * token usage and zero cost means the model was never reached — the
19
- * canonical signature of a Claude Code init/auth failure (e.g. an invalid
20
- * `ANTHROPIC_API_KEY`), which the SDK otherwise reports as a clean success.
18
+ * more than zero. A `result` message can carry `subtype: "success"` with
19
+ * zero token usage and zero cost. That combination means the run never
20
+ * reached the model. It is the canonical signature of a Claude Code init or
21
+ * auth failure (e.g. an invalid `ANTHROPIC_API_KEY`). The SDK otherwise
22
+ * reports that failure as a clean success.
21
23
  *
22
24
  * If the SDK gave us neither a `usage` object nor `total_cost_usd`, don't
23
- * second-guess the subtype — trust the reported success.
25
+ * second-guess the subtype. Trust the reported success.
24
26
  * @param {object|null} result - The SDK `result` message, or null.
25
27
  * @returns {boolean}
26
28
  */
@@ -38,8 +40,9 @@ function modelDidWork(result) {
38
40
  }
39
41
 
40
42
  // gemba-harness and kata-action run headless in CI/CD with no human to answer
41
- // permission prompts. The SDK is always launched in bypass mode — not
42
- // overridable — so a future caller can't accidentally reduce permissions.
43
+ // permission prompts. The runner always launches the SDK in bypass mode. No
44
+ // caller can override that mode, so a future caller can't accidentally
45
+ // reduce permissions.
43
46
  const PERMISSION_MODE = "bypassPermissions";
44
47
 
45
48
  /** Run a single Claude Agent SDK session and emit raw NDJSON events to an output stream. */
@@ -47,25 +50,27 @@ export class AgentRunner {
47
50
  /**
48
51
  * @param {object} deps
49
52
  * @param {string} deps.cwd - Agent working directory
50
- * @param {function} deps.query - SDK query function (injected for testing)
53
+ * @param {function} deps.query - SDK query function (tests inject it)
51
54
  * @param {import("stream").Writable} deps.output - Stream to emit NDJSON to
52
55
  * @param {string} [deps.model] - Claude model identifier
53
- * @param {number} [deps.maxTurns] - Maximum agentic turns; 0 means unlimited
56
+ * @param {number} [deps.maxTurns] - Maximum agentic turns. 0 means unlimited
54
57
  * @param {string[]} [deps.allowedTools] - Tools the agent may use
55
- * @param {function} [deps.onLine] - Callback invoked with each NDJSON line as it's produced
56
- * @param {function} [deps.onPrompt] - Callback invoked with the effective (amend-applied) prompt of each run/resume
58
+ * @param {function} [deps.onLine] - Callback that receives each NDJSON line as the runner produces it
59
+ * @param {function} [deps.onPrompt] - Callback that receives the effective (amend-applied) prompt of each run/resume
57
60
  * @param {string[]} [deps.settingSources] - SDK setting sources (e.g. ['project'] to load CLAUDE.md)
58
- * @param {string|object} [deps.systemPrompt] - SDK system prompt (string replaces default; {type:'preset', preset:'claude_code', append} appends)
61
+ * @param {string|object} [deps.systemPrompt] - SDK system prompt. A string replaces the default. The preset form {type:'preset', preset:'claude_code', append} appends
59
62
  * @param {string[]} [deps.disallowedTools] - Tools to explicitly remove from the model's context
60
63
  * @param {Record<string, object>} [deps.mcpServers] - MCP server configs to pass to the SDK query
61
64
  * @param {string} [deps.pathToClaudeCodeExecutable] - Absolute path to the
62
- * native `claude` CLI the SDK should spawn. Set for compiled fit-* binaries,
63
- * which can't self-resolve the SDK's platform optional dependency; omitted
64
- * from source runs so the SDK resolves its own version-matched binary.
65
+ * native `claude` CLI the SDK should spawn. Set it for compiled fit-*
66
+ * binaries, which can't self-resolve the SDK's platform optional
67
+ * dependency. Omit it from source runs so the SDK resolves its own
68
+ * version-matched binary.
65
69
  * @param {object} deps.redactor
66
70
  * @param {import("@forwardimpact/libutil/runtime").Runtime} [deps.runtime] -
67
- * Ambient collaborators. Only `proc.env` is read (to record Skill
68
- * invocations into `LIBHARNESS_SKILL`); when absent the write is skipped.
71
+ * Ambient collaborators. The runner reads only `proc.env`, to record Skill
72
+ * invocations into `LIBHARNESS_SKILL`. When `runtime` is absent, the
73
+ * runner skips the write.
69
74
  */
70
75
  constructor(deps) {
71
76
  if (!deps.cwd) throw new Error("cwd is required");
@@ -81,15 +86,17 @@ export class AgentRunner {
81
86
  this.maxTurns = deps.maxTurns ?? 50;
82
87
  this.allowedTools = deps.allowedTools ?? DEFAULT_ALLOWED_TOOLS;
83
88
  this.onLine = deps.onLine ?? null;
84
- // Optional; read only through a truthy guard in run()/resume(), so an
85
- // absent value stays undefined rather than needing a `?? null` default.
89
+ // Optional. The code reads it only through a truthy guard in
90
+ // run()/resume(), so an absent value stays undefined and needs no
91
+ // `?? null` default.
86
92
  this.onPrompt = deps.onPrompt;
87
93
  this.settingSources = deps.settingSources ?? [];
88
94
  this.systemPrompt = deps.systemPrompt ?? null;
89
95
  this.disallowedTools = deps.disallowedTools ?? [];
90
96
  this.mcpServers = deps.mcpServers ?? null;
91
- // Optional; read only through a truthy guard in #callOptions, so an absent
92
- // value stays undefined rather than needing a `?? null` default.
97
+ // Optional. The code reads it only through a truthy guard in
98
+ // #callOptions, so an absent value stays undefined and needs no
99
+ // `?? null` default.
93
100
  this.pathToClaudeCodeExecutable = deps.pathToClaudeCodeExecutable;
94
101
  this.taskAmend = deps.taskAmend ?? null;
95
102
  this.sessionId = null;
@@ -146,16 +153,17 @@ export class AgentRunner {
146
153
  }
147
154
 
148
155
  /**
149
- * Build the options passed to every SDK query() call. Shared by run()
150
- * and resume() so the agent's configuration — cwd, tools, prompt,
151
- * setting sources, turn budget — is identical across the session's
152
- * lifetime. Only resume() layers `resume: this.sessionId` on top.
156
+ * Build the options for every SDK query() call. run() and resume()
157
+ * share this method, so the agent's configuration stays identical
158
+ * across the session's lifetime. That configuration is the cwd, the
159
+ * tools, the prompt, the setting sources, and the turn budget. Only
160
+ * resume() layers `resume: this.sessionId` on top.
153
161
  *
154
- * SDK options are call-attached, not session-attached: the resumed
155
- * call loads the prior conversation but otherwise uses whatever
156
- * options this call passes. Omitting tool/prompt/setting options on
157
- * resume causes the agent to silently lose its restrictions and
158
- * persona between turns.
162
+ * SDK options attach to the call. They do not attach to the session.
163
+ * The resumed call loads the prior conversation. It otherwise uses
164
+ * whatever options this call passes. If you omit the tool, prompt, or
165
+ * setting options on resume, the agent silently loses its restrictions
166
+ * and persona between turns.
159
167
  */
160
168
  #callOptions(abortController) {
161
169
  return {
@@ -179,14 +187,14 @@ export class AgentRunner {
179
187
  }
180
188
 
181
189
  /**
182
- * Iterate the SDK query iterator, mirroring every message to the
183
- * output stream and the `onLine` callback. Captures `sessionId` from
184
- * the SDK's `system/init` message and tracks Skill invocations into
190
+ * Iterate the SDK query iterator. Mirror every message to the output
191
+ * stream and to the `onLine` callback. Capture `sessionId` from the
192
+ * SDK's `system/init` message. Track Skill invocations into
185
193
  * `LIBHARNESS_SKILL` for downstream metrics.
186
194
  *
187
195
  * If the iterator throws and we triggered the abort ourselves
188
196
  * (`currentAbortController.signal.aborted`), we report `aborted:
189
- * true`; otherwise the error propagates as `error`.
197
+ * true`. Otherwise the error propagates as `error`.
190
198
  */
191
199
  async #consumeQuery(iterator) {
192
200
  let text = "";
@@ -212,10 +220,10 @@ export class AgentRunner {
212
220
  }
213
221
  }
214
222
 
215
- // A "success" subtype is necessary but not sufficient: the SDK reports a
216
- // failed init (e.g. an invalid API key) as success with zero model work.
217
- // Require evidence the model actually ran, and surface a clear error when
218
- // it didn't, so the masked failure can't be reported as a green run.
223
+ // A "success" subtype is necessary. It is not sufficient. The SDK reports
224
+ // a failed init (e.g. an invalid API key) as success with zero model work.
225
+ // Require evidence that the model ran. Surface a clear error when it did
226
+ // not, so nobody reports the masked failure as a green run.
219
227
  const reportedSuccess = stopReason === "success";
220
228
  const success =
221
229
  reportedSuccess &&
@@ -223,7 +231,7 @@ export class AgentRunner {
223
231
  modelDidWork(resultMessage);
224
232
  if (reportedSuccess && !success && !error) {
225
233
  error = new Error(
226
- "agent reported success but performed no model work (zero token usage) — likely a Claude Code init or authentication failure",
234
+ "agent reported success but did no model work (zero token usage), which is likely a Claude Code init or authentication failure",
227
235
  );
228
236
  }
229
237
 
@@ -251,8 +259,9 @@ export class AgentRunner {
251
259
  #trackSkillInvocation(message) {
252
260
  const content = message.message?.content ?? message.content;
253
261
  if (!Array.isArray(content)) return;
254
- // Skill metric is recorded into the env map; without a runtime there is
255
- // no env surface to write to, so the side-effect is simply skipped.
262
+ // The runner records the Skill metric into the env map. Without a
263
+ // runtime there is no env surface to write to, so the code simply skips
264
+ // the side-effect.
256
265
  const env = this.runtime?.proc?.env ?? null;
257
266
  if (!env) return;
258
267
  for (const block of content) {
@@ -268,9 +277,10 @@ export class AgentRunner {
268
277
  }
269
278
 
270
279
  /**
271
- * Factory function — wires real dependencies. Resolves the native `claude`
272
- * executable for compiled fit-* binaries so the SDK doesn't fail to find its
273
- * own platform optional dependency; an explicit `deps` value overrides it.
280
+ * Factory function — wires real dependencies. It resolves the native
281
+ * `claude` executable for compiled fit-* binaries, so the SDK doesn't fail
282
+ * to find its own platform optional dependency. An explicit `deps` value
283
+ * overrides it.
274
284
  */
275
285
  export function createAgentRunner(deps) {
276
286
  return new AgentRunner({