@forwardimpact/libharness 3.0.0 → 3.0.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (69) hide show
  1. package/README.md +60 -57
  2. package/package.json +2 -2
  3. package/src/advisor.js +47 -41
  4. package/src/agent-runner.js +57 -47
  5. package/src/benchmark/apm-installer.js +28 -28
  6. package/src/benchmark/env-loader.js +24 -16
  7. package/src/benchmark/grade.js +44 -41
  8. package/src/benchmark/hidden-tests.js +25 -24
  9. package/src/benchmark/hook-env.js +11 -9
  10. package/src/benchmark/invariants.js +20 -17
  11. package/src/benchmark/judge.js +29 -28
  12. package/src/benchmark/npm-installer.js +9 -8
  13. package/src/benchmark/report.js +53 -50
  14. package/src/benchmark/result.js +24 -23
  15. package/src/benchmark/runner.js +75 -69
  16. package/src/benchmark/scheduler.js +17 -16
  17. package/src/benchmark/task-family.js +28 -26
  18. package/src/benchmark/trace-split.js +8 -7
  19. package/src/benchmark/workdir.js +27 -25
  20. package/src/claude-code-executable.js +11 -11
  21. package/src/commands/advisor-flags.js +8 -7
  22. package/src/commands/assert.js +16 -15
  23. package/src/commands/benchmark-definition.js +11 -11
  24. package/src/commands/benchmark-grade.js +12 -11
  25. package/src/commands/benchmark-report.js +5 -5
  26. package/src/commands/benchmark-run.js +31 -28
  27. package/src/commands/by-discussion.js +10 -10
  28. package/src/commands/callback.js +11 -11
  29. package/src/commands/discuss.js +8 -7
  30. package/src/commands/facilitate.js +15 -13
  31. package/src/commands/output.js +3 -2
  32. package/src/commands/run.js +14 -14
  33. package/src/commands/scan-logs.js +21 -19
  34. package/src/commands/selfedit.js +14 -14
  35. package/src/commands/supervise.js +11 -9
  36. package/src/commands/task-input.js +9 -9
  37. package/src/commands/tee.js +10 -9
  38. package/src/commands/trace.js +55 -42
  39. package/src/commands/work-tracker.js +4 -3
  40. package/src/cost.js +17 -17
  41. package/src/discuss-tools.js +16 -16
  42. package/src/discusser.js +39 -38
  43. package/src/events/github.js +54 -37
  44. package/src/facilitator.js +21 -21
  45. package/src/inbox-poller.js +4 -4
  46. package/src/judge.js +32 -30
  47. package/src/message-bus.js +12 -11
  48. package/src/orchestration-loop.js +35 -36
  49. package/src/orchestration-toolkit.js +58 -53
  50. package/src/orchestrator-helpers.js +2 -2
  51. package/src/profile-prompt.js +54 -53
  52. package/src/redaction.js +63 -57
  53. package/src/render/line-renderer.js +5 -5
  54. package/src/render/orchestrator-filter.js +3 -3
  55. package/src/render/palette.js +11 -9
  56. package/src/render/tool-hints.js +18 -15
  57. package/src/render/turn-renderer.js +4 -4
  58. package/src/reply-emitter.js +2 -2
  59. package/src/sequence-counter.js +4 -3
  60. package/src/signature-filter.js +7 -6
  61. package/src/supervisor.js +19 -18
  62. package/src/tee-writer.js +25 -25
  63. package/src/trace-collector.js +53 -48
  64. package/src/trace-github.js +53 -44
  65. package/src/trace-multi.js +15 -13
  66. package/src/trace-query.js +61 -52
  67. package/src/trace-render.js +18 -18
  68. package/src/trace-usage.js +31 -28
  69. package/src/transcript-recorder.js +24 -20
package/src/judge.js CHANGED
@@ -1,13 +1,13 @@
1
1
  /**
2
2
  * Judge — one agent session that inspects a completed agent's work and emits
3
- * a verdict via the orchestration `Conclude` tool. Parallel concept to
4
- * `Supervisor` and `Facilitator`, but post-hoc and solo: no peer agents,
5
- * no message bus, no orchestration loop. The judge reads the task, optionally
6
- * inspects the working directory and trace via read-only tools, and calls
7
- * Conclude exactly once.
3
+ * a verdict through the orchestration `Conclude` tool. It is a parallel
4
+ * concept to `Supervisor` and `Facilitator`. It runs post-hoc and solo: no
5
+ * peer agents, no message bus, no orchestration loop. The judge reads the
6
+ * task. It can inspect the working directory and the trace with read-only
7
+ * tools. It calls Conclude exactly once.
8
8
  *
9
- * Trace lines are tagged `source: "judge"` so consumers can distinguish
10
- * judge sessions from supervisor or facilitator sessions in a unified
9
+ * The judge tags trace lines with `source: "judge"`. Consumers can then tell
10
+ * judge sessions apart from supervisor or facilitator sessions in a unified
11
11
  * NDJSON envelope.
12
12
  *
13
13
  * Follows OO+DI: constructor injection, factory function, tests bypass factory.
@@ -25,17 +25,19 @@ import {
25
25
  } from "./orchestration-toolkit.js";
26
26
 
27
27
  /**
28
- * System-prompt trailer appended to the judge's main thread. Always applied,
29
- * even when a `judgeProfile` is supplied — the profile layers on top of the
30
- * trailer, the same way `SUPERVISOR_SYSTEM_PROMPT` and
31
- * `FACILITATOR_SYSTEM_PROMPT` work for their respective roles.
28
+ * System-prompt trailer for the judge's main thread. The factory always
29
+ * applies it, even when the caller supplies a `judgeProfile`. The profile
30
+ * layers on top of the trailer. `SUPERVISOR_SYSTEM_PROMPT` and
31
+ * `FACILITATOR_SYSTEM_PROMPT` work the same way for their roles.
32
32
  */
33
33
  export const JUDGE_SYSTEM_PROMPT =
34
34
  "You are a post-hoc judge for an agent task benchmark. " +
35
- "The agent has already completed its work and an objective invariants step has already run; your role is to confirm or override the verdict by inspecting the agent's working directory and trace. " +
36
- "You have read-only inspection tools — Read, Glob, Grep, Bash — to investigate; do not modify the working directory. " +
37
- "Conclude ends the session with a verdict ('success' or 'failure') and a one-paragraph summary; verdict='success' iff the agent's work meets the criteria stated in the task. " +
38
- "Call Conclude as your final action — do not deliberate across multiple turns.";
35
+ "The agent already completed its work. An objective invariants step already ran. " +
36
+ "Confirm or override the verdict. To do so, inspect the agent's working directory and trace. " +
37
+ "You have read-only inspection tools to investigate: Read, Glob, Grep, and Bash. Do not modify the working directory. " +
38
+ "Conclude ends the session with a verdict ('success' or 'failure') and a one-paragraph summary. " +
39
+ "Set verdict='success' exactly when the agent's work meets the criteria the task states. " +
40
+ "Call Conclude as your final action. Do not deliberate across multiple turns.";
39
41
 
40
42
  const DEFAULT_JUDGE_ALLOWED_TOOLS = ["Read", "Glob", "Grep", "Bash"];
41
43
 
@@ -45,7 +47,7 @@ const devNull = new Writable({
45
47
  },
46
48
  });
47
49
 
48
- /** Run a single post-hoc judge session and emit a verdict via Conclude. */
50
+ /** Run a single post-hoc judge session and emit a verdict with Conclude. */
49
51
  export class Judge {
50
52
  /**
51
53
  * @param {object} deps
@@ -53,7 +55,7 @@ export class Judge {
53
55
  * @param {import("stream").Writable} deps.output - Stream to emit tagged NDJSON to.
54
56
  * @param {object} deps.ctx - Orchestration context (the Conclude handler writes to it).
55
57
  * @param {import("./redaction.js").Redactor} deps.redactor
56
- * @param {string} [deps.taskAmend] - Opaque addendum appended to the task before delivery.
58
+ * @param {string} [deps.taskAmend] - Opaque addendum. The judge appends it to the task before delivery.
57
59
  */
58
60
  constructor({ runner, output, ctx, redactor, taskAmend }) {
59
61
  if (!runner) throw new Error("runner is required");
@@ -70,7 +72,7 @@ export class Judge {
70
72
 
71
73
  /**
72
74
  * Run the judge session.
73
- * @param {string} task - The judge prompt (with placeholders already substituted).
75
+ * @param {string} task - The judge prompt (the caller already substituted the placeholders).
74
76
  * @returns {Promise<{success: boolean, verdict: string|null, summary: string|null, turns: number}>}
75
77
  */
76
78
  async run(task) {
@@ -89,7 +91,7 @@ export class Judge {
89
91
  return outcome;
90
92
  }
91
93
 
92
- // The judge ended without calling Conclude. Surface that explicitly so
94
+ // The judge ended and never called Conclude. Surface that explicitly so
93
95
  // callers can distinguish "judge said fail" from "judge never voted."
94
96
  const outcome = {
95
97
  success: false,
@@ -103,9 +105,9 @@ export class Judge {
103
105
 
104
106
  /**
105
107
  * Tag a single NDJSON line with `source: "judge"` and emit it to the
106
- * judge's output stream. Wired into the underlying AgentRunner via the
107
- * `onLine` callback so the judge's stream is the single source of truth
108
- * for the session's trace.
108
+ * judge's output stream. The factory wires this into the underlying
109
+ * AgentRunner through the `onLine` callback. The judge's stream is then the
110
+ * single source of truth for the session's trace.
109
111
  * @param {string} line
110
112
  */
111
113
  emitLine(line) {
@@ -115,7 +117,7 @@ export class Judge {
115
117
  }
116
118
 
117
119
  /**
118
- * Emit a final orchestrator summary line, wrapped in the universal envelope.
120
+ * Emit a final orchestrator summary line in the universal envelope.
119
121
  * @param {{success: boolean, verdict?: string|null, summary?: string|null, turns: number}} result
120
122
  */
121
123
  emitSummary(result) {
@@ -138,20 +140,20 @@ export class Judge {
138
140
  }
139
141
 
140
142
  /**
141
- * Factory function — wires the AgentRunner with the judge orchestration server
143
+ * Factory function. Wires the AgentRunner with the judge orchestration server
142
144
  * and the JUDGE_SYSTEM_PROMPT trailer. A `judgeProfile` (when supplied) layers
143
- * on top of the trailer via `composeSystemPrompt`, matching the
144
- * supervisor/facilitator pattern.
145
+ * on top of the trailer through `composeSystemPrompt`. This matches the
146
+ * supervisor and facilitator pattern.
145
147
  *
146
148
  * @param {object} deps
147
149
  * @param {string} deps.cwd - Judge working directory. Defaults to the directory whose `.claude/agents` holds `judgeProfile`.
148
- * @param {function} deps.query - SDK query function (injected for testing).
150
+ * @param {function} deps.query - SDK query function (injected so tests can replace it).
149
151
  * @param {import("stream").Writable} deps.output - Trace output stream.
150
152
  * @param {import("./redaction.js").Redactor} deps.redactor
151
153
  * @param {string} [deps.model]
152
- * @param {number} [deps.maxTurns] - Default 5 (the judge is expected to act in turn 1; 5 leaves headroom for tool inspection).
153
- * @param {string[]} [deps.allowedTools] - Default `["Read","Glob","Grep","Bash"]` — read-only inspection.
154
- * @param {string} [deps.judgeProfile] - Profile name; resolved into the system prompt via `composeSystemPrompt`.
154
+ * @param {number} [deps.maxTurns] - Default 5. The judge should act in turn 1. The other turns leave headroom for tool inspection.
155
+ * @param {string[]} [deps.allowedTools] - Default `["Read","Glob","Grep","Bash"]` for read-only inspection.
156
+ * @param {string} [deps.judgeProfile] - Profile name. `composeSystemPrompt` resolves it into the system prompt.
155
157
  * @param {string} [deps.profilesDir] - Defaults to `<cwd>/.claude/agents`.
156
158
  * @param {string} [deps.taskAmend]
157
159
  * @returns {Judge}
@@ -1,17 +1,18 @@
1
1
  /**
2
2
  * MessageBus — in-memory per-participant message queues.
3
3
  *
4
- * Four message kinds, each pushed onto the addressee's queue:
4
+ * Four message kinds exist. The bus pushes each one onto the addressee's
5
+ * queue:
5
6
  *
6
- * - `ask(from, to, text, askId)` — direct question; the toolkit owns the
7
- * pending-ask state separately. Fan-out (broadcast Ask) happens at the
8
- * handler level by calling `ask()` once per addressee.
7
+ * - `ask(from, to, text, askId)` — direct question. The toolkit owns the
8
+ * pending-ask state separately. The handler level does the fan-out
9
+ * (broadcast Ask). It calls `ask()` once per addressee.
9
10
  * - `answer(from, to, text, askId)` — direct reply to the original asker.
10
11
  * The orchestrator may inject synthetic answers (`from === "@orchestrator"`)
11
12
  * when an Ask times out.
12
- * - `announce(from, text)` — broadcast, no reply expected; lands on every
13
- * participant's queue except the sender's.
14
- * - `synthetic(to, text)` — orchestrator-only reminder injection.
13
+ * - `announce(from, text)` — broadcast. It expects no reply. It lands on
14
+ * every participant's queue except the sender's.
15
+ * - `synthetic(to, text)` — the orchestrator alone injects a reminder.
15
16
  *
16
17
  * Follows OO+DI: constructor injection, factory function, tests bypass factory.
17
18
  */
@@ -40,9 +41,9 @@ export class MessageBus {
40
41
  }
41
42
 
42
43
  /**
43
- * Reply to a pending ask. `from === "@orchestrator"` is allowed for
44
- * synthetic null answers — the orchestrator is not a real participant
45
- * but it routes through the bus.
44
+ * Reply to a pending ask. The bus allows `from === "@orchestrator"` for
45
+ * synthetic null answers. The orchestrator is not a real participant. It
46
+ * still routes through the bus.
46
47
  */
47
48
  answer(from, to, text, askId) {
48
49
  this.#assertParticipant(to);
@@ -71,7 +72,7 @@ export class MessageBus {
71
72
  this.#resolveWaiter(to);
72
73
  }
73
74
 
74
- /** Check whether a participant has pending messages without draining them. */
75
+ /** Check whether a participant has pending messages. It does not drain them. */
75
76
  hasPending(participant) {
76
77
  this.#assertParticipant(participant);
77
78
  return this.queues.get(participant).length > 0;
@@ -1,22 +1,22 @@
1
1
  /**
2
- * OrchestrationLoop — N agent sessions coordinated by one lead LLM session.
2
+ * OrchestrationLoop — one lead LLM session coordinates N agent sessions.
3
3
  *
4
- * Ask is **async**: the tool returns immediately, the actual reply arrives
4
+ * Ask is **async**. The tool returns immediately. The actual reply arrives
5
5
  * on a later turn as `[answer#N] participant: …` on the asker's bus queue.
6
- * Pending state keys by `askId` (visible in the `[ask#N]` tag), so duplicate
7
- * Asks to the same addressee coexist without overwriting each other, and
8
- * the asker can map each reply unambiguously back to its question.
6
+ * Pending state keys by `askId` (visible in the `[ask#N]` tag). Duplicate
7
+ * Asks to the same addressee then coexist and never overwrite each other.
8
+ * The asker can map each reply unambiguously back to its question.
9
9
  *
10
- * Both lead and participants follow the same outer pattern: drain the bus
11
- * queue, run / resume the LLM with the drained messages, then settle any
10
+ * Both lead and participants follow the same outer pattern. Drain the bus
11
+ * queue. Run or resume the LLM with the drained messages. Then settle any
12
12
  * unanswered Asks the participant owes. They differ only in how the first
13
- * turn starts (the lead receives the task; participants wait for traffic).
13
+ * turn starts. The lead receives the task. Participants wait for traffic.
14
14
  *
15
15
  * Termination signals:
16
16
  * - `ctx.concluded` — explicit Conclude / Adjourn / Recess.
17
- * - `stopped` — broader: also true on lead error, agent crash, or any
18
- * other abort path. Loops watch `stopped`; `ctx.concluded` is only used
19
- * for the summary's success/verdict.
17
+ * - `stopped` — broader. It is also true on lead error, agent crash, or any
18
+ * other abort path. Loops watch `stopped`. The code uses `ctx.concluded`
19
+ * only for the summary's success and verdict.
20
20
  */
21
21
  import { SequenceCounter } from "./sequence-counter.js";
22
22
  import {
@@ -26,10 +26,10 @@ import {
26
26
  } from "./orchestration-toolkit.js";
27
27
  import { formatMessages } from "./orchestrator-helpers.js";
28
28
 
29
- /** Default per-session lead-turn budget — accommodates multi-round injected conversations. */
29
+ /** Default per-session lead-turn budget. It fits multi-round injected conversations. */
30
30
  const DEFAULT_MAX_LEAD_TURNS = 200;
31
31
 
32
- /** Orchestrate N agent sessions coordinated by a single lead LLM session. */
32
+ /** Coordinate N agent sessions from a single lead LLM session. */
33
33
  export class OrchestrationLoop {
34
34
  /**
35
35
  * @param {object} deps
@@ -42,7 +42,7 @@ export class OrchestrationLoop {
42
42
  * @param {object} deps.ctx - Orchestration context (from `createOrchestrationContext()`).
43
43
  * @param {object} deps.redactor
44
44
  * @param {number} [deps.maxLeadTurns] - Cap on lead resumes per session (default 200).
45
- * @param {string} [deps.taskAmend] - Appended to the task before delivery.
45
+ * @param {string} [deps.taskAmend] - The loop appends it to the task before delivery.
46
46
  * @param {import("./inbox-poller.js").InboxPoller} [deps.inboxPoller]
47
47
  * @param {AbortController} [deps.abortController]
48
48
  */
@@ -90,7 +90,7 @@ export class OrchestrationLoop {
90
90
  this.#signalDone = resolveDone;
91
91
  }
92
92
 
93
- /** Internal — resolved when `stopped` flips true so waiters unblock. */
93
+ /** Internal. Resolves when `stopped` flips true so waiters unblock. */
94
94
  #signalDone;
95
95
 
96
96
  /**
@@ -112,9 +112,9 @@ export class OrchestrationLoop {
112
112
  this.#stop();
113
113
  };
114
114
 
115
- // Start agent loops in parallel. Wrapped so a crash flips `stopped`
116
- // but the wrapper itself resolves — Promise.allSettled below never
117
- // sees an unhandled rejection.
115
+ // Start agent loops in parallel. The wrapper makes a crash flip `stopped`
116
+ // and still resolves itself. Promise.allSettled below then never sees an
117
+ // unhandled rejection.
118
118
  const agentPromises = this.agents.map((a) =>
119
119
  this.#runAgent(a).catch(abort),
120
120
  );
@@ -153,14 +153,14 @@ export class OrchestrationLoop {
153
153
  }
154
154
 
155
155
  /**
156
- * Lead loop. The lead's first turn carries the task; every subsequent
157
- * turn is a resume triggered by something landing on its inbox.
156
+ * Lead loop. The lead's first turn carries the task. Every later turn is
157
+ * a resume, and something that lands on its inbox triggers it.
158
158
  *
159
159
  * `messages.length === 0` from `#drainOrWait` means the session ended
160
- * before any message arrived — that's the natural exit. If
161
- * `drainOrWait` returned messages, deliver them even if the session
162
- * concluded in the microtask window between wake-up and this check;
163
- * the inbox already has them and they deserve to be seen.
160
+ * before any message arrived. That is the natural exit. If `drainOrWait`
161
+ * returned messages, deliver them even when the session concluded in the
162
+ * microtask window between wake-up and this check. The inbox already holds
163
+ * them, so the lead should see them.
164
164
  */
165
165
  async #runLead(initialTask) {
166
166
  this.leadTurns = 1;
@@ -190,8 +190,8 @@ export class OrchestrationLoop {
190
190
  }
191
191
 
192
192
  /**
193
- * Agent loop. The first message off the inbox triggers `run()`; every
194
- * subsequent batch triggers `resume()`. No turn budget — the agent
193
+ * Agent loop. The first message off the inbox triggers `run()`. Every
194
+ * later batch triggers `resume()`. The loop has no turn budget. The agent
195
195
  * runner's own `maxTurns` caps each SDK call.
196
196
  */
197
197
  async #runAgent({ name, runner }) {
@@ -235,10 +235,10 @@ export class OrchestrationLoop {
235
235
 
236
236
  /**
237
237
  * If `name` left a pending Ask unanswered, inject one synthetic reminder
238
- * and resume once more. If still unanswered after the reminder, emit a
239
- * `protocol_violation` event per outstanding ask and cancel them — the
240
- * asker's queue gets a synthetic `[no answer: …]` so it doesn't deadlock
241
- * on a participant that's silently ignoring its inbox.
238
+ * and resume once more. If it is still unanswered after the reminder, emit
239
+ * a `protocol_violation` event per outstanding ask and cancel them. The
240
+ * asker's queue then gets a synthetic `[no answer: …]`, so the asker does
241
+ * not deadlock on a participant that silently ignores its inbox.
242
242
  */
243
243
  async #settleOwedAsks(name, runner) {
244
244
  if (pendingAsksOwedBy(this.ctx, name).length === 0) return;
@@ -267,9 +267,9 @@ export class OrchestrationLoop {
267
267
  }
268
268
 
269
269
  /**
270
- * Emit one NDJSON line tagged with its source (participant name) and a
271
- * monotonic seq, wrapped in the universal `{source, seq, event}` envelope.
272
- * Called from each runner's `onLine` callback.
270
+ * Emit one NDJSON line in the universal `{source, seq, event}` envelope.
271
+ * Tag it with its source (the participant name) and a monotonic seq.
272
+ * Each runner's `onLine` callback calls this.
273
273
  * @param {string} source
274
274
  * @param {string} line - Raw NDJSON line from the SDK iterator.
275
275
  */
@@ -288,8 +288,7 @@ export class OrchestrationLoop {
288
288
 
289
289
  /**
290
290
  * Emit one orchestrator-source event (`session_start`, `agent_start`,
291
- * `protocol_violation`, `lead_turn_limit`) wrapped in the universal
292
- * envelope.
291
+ * `protocol_violation`, `lead_turn_limit`) in the universal envelope.
293
292
  * @param {object} event
294
293
  */
295
294
  emitOrchestratorEvent(event) {
@@ -306,7 +305,7 @@ export class OrchestrationLoop {
306
305
 
307
306
  /**
308
307
  * Emit the terminal summary line. `Discusser` emits its own discuss-
309
- * augmented summary after this one; trace consumers keep the last
308
+ * augmented summary after this one. Trace consumers keep the last
310
309
  * summary they see.
311
310
  * @param {{success: boolean, verdict?: string|null, turns: number, summary?: string|null}} result
312
311
  */
@@ -1,7 +1,7 @@
1
1
  /**
2
2
  * OrchestrationToolkit — tool schemas, per-role tool sets, and handler
3
3
  * factories for orchestration between leads (facilitator, supervisor,
4
- * discuss-lead) and their participating agents.
4
+ * discuss-lead) and the agents that take part.
5
5
  *
6
6
  * **Tool surface, by role:**
7
7
  *
@@ -15,16 +15,16 @@
15
15
  * | Discuss agt | ✓ | ✓ | ✓ | ✓ | | RFC |
16
16
  * | Judge | | | | | ✓ | |
17
17
  *
18
- * **Ask is async.** Ask returns `{askIds:[…]}` immediately and posts the
18
+ * **Ask is async.** Ask returns `{askIds:[…]}` immediately. It posts the
19
19
  * question to the addressee's bus queue. The reply arrives on the asker's
20
- * next turn as `[answer#N] <participant>: <text>`. Pending state keys by
21
- * `askId` (visible in `[ask#N]` tags), so duplicate Asks to the same
22
- * addressee coexist without overwriting.
20
+ * next turn as `[answer#N] <participant>: <text>`. The toolkit keys pending
21
+ * state by `askId` (visible in `[ask#N]` tags). Duplicate Asks to the same
22
+ * addressee then coexist and never overwrite each other.
23
23
  *
24
24
  * **Answer's `askId` is optional.** With a matching askId, the reply
25
- * routes to that specific asker. Without, the handler auto-picks if
26
- * exactly one ask is owed to the caller, otherwise routes the message
27
- * as an Announce so it still reaches everyone.
25
+ * routes to that specific asker. Without one, the handler auto-picks when
26
+ * exactly one ask is owed to the caller. Otherwise the handler routes the
27
+ * message as an Announce so it still reaches everyone.
28
28
  */
29
29
 
30
30
  import { createSdkMcpServer, tool } from "@anthropic-ai/claude-agent-sdk";
@@ -48,9 +48,9 @@ export function createOrchestrationContext() {
48
48
 
49
49
  /**
50
50
  * Guard for terminal tools (`Conclude`, `Adjourn`, `Recess`). Returns an
51
- * error result when the caller still has Asks in flight, telling them to
52
- * end the turn and wait for the auto-resume. Returns `null` when no Asks
53
- * are pending and the terminal tool is free to run.
51
+ * error result when the caller still has Asks in flight. That result tells
52
+ * the caller to end the turn and wait for the auto-resume. Returns `null`
53
+ * when no Asks are pending and the terminal tool is free to run.
54
54
  */
55
55
  export function requireNoPendingAsks(ctx) {
56
56
  if (ctx.pendingAsks.size === 0) return null;
@@ -62,18 +62,18 @@ export function requireNoPendingAsks(ctx) {
62
62
  /**
63
63
  * Guard for terminal tools in discuss mode (`Adjourn`, `Recess`). Returns
64
64
  * an error result when the lead's inbox has unprocessed messages from the
65
- * human, telling them to end the turn and wait for the auto-resume.
66
- * Returns `null` when no inbox messages are pending and the terminal tool
67
- * is free to run.
65
+ * human. That result tells the lead to end the turn and wait for the
66
+ * auto-resume. Returns `null` when no inbox messages are pending and the
67
+ * terminal tool is free to run.
68
68
  */
69
69
  export function requireNoUnprocessedInbox(ctx) {
70
70
  if (!ctx.messageBus?.hasPending?.("lead")) return null;
71
71
  return errorResult(
72
- "New messages from the human are waiting. End your turn. You will be resumed to process them.",
72
+ "New messages from the human are in your inbox. End your turn. You will be resumed to process them.",
73
73
  );
74
74
  }
75
75
 
76
- /** Mark the session as concluded; cancel any open Asks so askers see the synthetic null on their next turn. */
76
+ /** Mark the session as concluded. Cancel any open Asks so askers see the synthetic null on their next turn. */
77
77
  export function createConcludeHandler(ctx) {
78
78
  return async ({ verdict, summary }) => {
79
79
  const guard = requireNoPendingAsks(ctx);
@@ -84,11 +84,11 @@ export function createConcludeHandler(ctx) {
84
84
  }
85
85
 
86
86
  /**
87
- * Shared terminal-tool helper. Conclude / Adjourn / Recess all set the
88
- * same three context fields (`concluded`, `verdict`, `summary`) and
89
- * cancel any in-flight Asks for the same reason: nobody will ever
90
- * answer them now. Mode-specific handlers (Adjourn, Recess) layer
91
- * extra state on top before calling this.
87
+ * Shared terminal-tool helper. Conclude, Adjourn, and Recess all set the
88
+ * same three context fields (`concluded`, `verdict`, `summary`). All three
89
+ * also cancel any in-flight Asks for the same reason. Nobody will ever
90
+ * answer them now. Mode-specific handlers (Adjourn, Recess) layer extra
91
+ * state on top before they call this.
92
92
  */
93
93
  export function concludeSession(ctx, { verdict, summary, reason }) {
94
94
  ctx.concluded = true;
@@ -124,21 +124,24 @@ function registerPendingAsk(ctx, { from, addressee, question }) {
124
124
  }
125
125
 
126
126
  /**
127
- * Create an Ask handler. Registers a pending entry per addressee, posts
128
- * the ask on the bus, returns `{askIds:[…]}` immediately. The LLM uses
129
- * those ids to match the `[answer#N]` it sees on a later turn.
127
+ * Create an Ask handler. The handler registers a pending entry for each
128
+ * addressee. It posts the ask on the bus. It returns `{askIds:[…]}`
129
+ * immediately. The LLM uses those ids to match the `[answer#N]` it sees on
130
+ * a later turn.
130
131
  *
131
132
  * @param {object} ctx
132
133
  * @param {object} opts
133
134
  * @param {string} opts.from
134
135
  * @param {string|undefined} opts.defaultTo - `undefined` means "broadcast
135
- * to everyone else"; a participant name means "target that one when
136
+ * to everyone else". A participant name means "target that one when
136
137
  * `to` is omitted."
137
138
  */
138
139
  export function createAskHandler(ctx, { from, defaultTo }) {
139
140
  return async ({ question, to }) => {
140
141
  if (ctx.concluded) {
141
- return errorResult("Session is concluded; Ask was not delivered.");
142
+ return errorResult(
143
+ "The session is concluded. The handler did not deliver your Ask.",
144
+ );
142
145
  }
143
146
  const addressees = resolveAddressees(ctx, { from, to, defaultTo });
144
147
  if (addressees.length === 0) {
@@ -157,7 +160,8 @@ export function createAskHandler(ctx, { from, defaultTo }) {
157
160
  * - askId provided + matches a pending entry whose addressee is the caller →
158
161
  * route the reply to the asker's queue and clear the pending entry.
159
162
  * - askId provided but unknown or wrong addressee → `isError`. The caller
160
- * tried to specify; we tell them why it didn't match.
163
+ * tried to name an askId. The handler tells the caller why it did not
164
+ * match.
161
165
  * - askId omitted + exactly one ask owed by the caller → auto-pick it.
162
166
  * - askId omitted + 0 or many pending → broadcast as Announce so the
163
167
  * message still reaches every other participant.
@@ -180,9 +184,9 @@ export function createAnswerHandler(ctx, { from }) {
180
184
  ctx.messageBus.announce(from, message);
181
185
  const reason =
182
186
  owed.length === 0
183
- ? "no pending ask for you"
184
- : `${owed.length} pending asks (askId omitted is ambiguous)`;
185
- return textResult(`Answer routed as Announce — ${reason}.`);
187
+ ? "You have no pending ask."
188
+ : `You have ${owed.length} pending asks. An omitted askId is ambiguous.`;
189
+ return textResult(`Answer routed as Announce. ${reason}`);
186
190
  };
187
191
  }
188
192
 
@@ -191,7 +195,7 @@ function routeAnswerByAskId(ctx, { from, askId, message }) {
191
195
  if (!entry) return errorResult(`No pending ask with askId=${askId}.`);
192
196
  if (entry.addresseeName !== from) {
193
197
  return errorResult(
194
- `Ask #${askId} is addressed to ${entry.addresseeName}, not ${from}.`,
198
+ `Ask #${askId} is addressed to ${entry.addresseeName}. You are ${from}.`,
195
199
  );
196
200
  }
197
201
  ctx.pendingAsks.delete(askId);
@@ -208,9 +212,9 @@ export function createAnnounceHandler(ctx, { from }) {
208
212
  }
209
213
 
210
214
  /**
211
- * Cancel pending Asks and route a synthetic `[no answer: <reason>]` to
212
- * each asker's queue so callers never deadlock on a participant ignoring
213
- * its inbox.
215
+ * Cancel pending Asks. Route a synthetic `[no answer: <reason>]` to each
216
+ * asker's queue, so callers never deadlock when a participant ignores its
217
+ * inbox.
214
218
  *
215
219
  * @param {object} ctx
216
220
  * @param {string} reason - Surfaced inside `[no answer: <reason>]`.
@@ -234,8 +238,8 @@ export function pendingAsksOwedBy(ctx, addressee) {
234
238
  }
235
239
 
236
240
  /**
237
- * Inject a synthetic reminder onto the addressee's bus queue and mark
238
- * each owed ask as reminded. Returns true when a reminder fired.
241
+ * Inject a synthetic reminder onto the addressee's bus queue. Mark each
242
+ * owed ask as reminded. Returns true when a reminder fired.
239
243
  */
240
244
  export function remindOwedAsks(ctx, addressee) {
241
245
  const owed = pendingAsksOwedBy(ctx, addressee).filter((e) => !e.reminded);
@@ -252,13 +256,13 @@ export function remindOwedAsks(ctx, addressee) {
252
256
  // --- Tool descriptions (shared across roles) ---
253
257
 
254
258
  const ASK_DESC_BROADCAST =
255
- "Send a question to one named participant, or omit 'to' to broadcast to every other participant. Returns {askIds:[…]} immediately; the reply arrives on a later turn as `[answer#N] <from>: <text>` in your inbox.";
259
+ "Send a question to one named participant. Omit 'to' to broadcast to every other participant. Returns {askIds:[…]} immediately. The reply arrives on a later turn as `[answer#N] <from>: <text>` in your inbox.";
256
260
 
257
261
  const ASK_DESC_TARGETED = (target) =>
258
- `Send a question to ${target}. Returns {askIds:[N]} immediately; the reply arrives on a later turn as \`[answer#N] ${target}: <text>\` in your inbox.`;
262
+ `Send a question to ${target}. Returns {askIds:[N]} immediately. The reply arrives on a later turn as \`[answer#N] ${target}: <text>\` in your inbox.`;
259
263
 
260
264
  const ANSWER_DESC =
261
- "Reply to an ask addressed to you. Quote askId from the [ask#N] tag on the question; omit it and the handler auto-picks the only pending ask, or routes your message as an Announce when 0 or many are pending.";
265
+ "Reply to an ask addressed to you. Quote askId from the [ask#N] tag on the question. Omit askId and the handler auto-picks the only pending ask. When 0 or many asks are pending, the handler routes your message as an Announce.";
262
266
 
263
267
  const ANNOUNCE_DESC = "Broadcast a message with no reply expected.";
264
268
 
@@ -275,7 +279,7 @@ const ADJOURN_DESC =
275
279
  "End the discussion. Provide a verdict ('adjourned' or 'failed') and a summary. Cancels any unanswered Asks.";
276
280
 
277
281
  const RECESS_DESC =
278
- "End the run and schedule an out-of-session re-dispatch. Cancels any unanswered Asks. Use only when waiting on an external reply or duration. Do not use to wait on in-flight Asks.";
282
+ "End the run. Schedule an out-of-session re-dispatch. Cancels any unanswered Asks. Use only when you wait on an external reply or duration. Do not use to wait on in-flight Asks.";
279
283
 
280
284
  // --- Tool builders ---
281
285
 
@@ -283,7 +287,7 @@ const RECESS_DESC =
283
287
  function textResult(text) {
284
288
  return { content: [{ type: "text", text }] };
285
289
  }
286
- /** Build an MCP tool error result wrapping a single text message. */
290
+ /** Build an MCP tool error result that wraps a single text message. */
287
291
  function errorResult(text) {
288
292
  return { content: [{ type: "text", text }], isError: true };
289
293
  }
@@ -298,10 +302,10 @@ function jsonResult(obj) {
298
302
  * @param {object} ctx
299
303
  * @param {object} opts
300
304
  * @param {string} opts.from - Caller's canonical name.
301
- * @param {string|undefined} opts.defaultTo - Default Ask target; `undefined`
305
+ * @param {string|undefined} opts.defaultTo - Default Ask target. `undefined`
302
306
  * means "broadcast across everyone else when `to` is omitted."
303
307
  * @param {boolean} opts.broadcast - Whether Ask accepts a `to` field at all.
304
- * Leads with multiple participants set this true; supervise's
308
+ * Leads with multiple participants set this true. Supervise's
305
309
  * single-participant roles set it false.
306
310
  */
307
311
  function baseTools(ctx, { from, defaultTo, broadcast }) {
@@ -338,19 +342,20 @@ function concludeTool(ctx) {
338
342
  }
339
343
 
340
344
  const ADVISOR_DESC =
341
- "Consult a stronger model on one focused question. Your full session context (system prompt, prompts, transcript so far) is forwarded automatically — you cannot restrict it. The advice returns in the tool result. The consult budget is shared session-wide across all participants.";
345
+ "Consult a stronger model on one focused question. The tool forwards your full session context (system prompt, prompts, transcript so far) automatically. You cannot restrict it. The advice returns in the tool result. All participants share one session-wide consult budget.";
342
346
 
343
347
  /**
344
- * Build the `Advisor` consult tool for one caller. Mode-agnostic: loop
345
- * modes pass it into the agent tool-server factories via `extraTools`;
346
- * run mode gives it a dedicated server. No orchestration-context
347
- * dependency — the budget object and emit callback are injected.
348
+ * Build the `Advisor` consult tool for one caller. The tool is
349
+ * mode-agnostic. Loop modes pass it into the agent tool-server factories
350
+ * through `extraTools`. Run mode gives it a dedicated server. The tool has
351
+ * no orchestration-context dependency. The budget object and the emit
352
+ * callback arrive as injected dependencies.
348
353
  *
349
354
  * @param {object} deps
350
355
  * @param {string} deps.from - Caller's canonical name (event attribution).
351
356
  * @param {(question: string) => Promise<{advice?: string, unavailable?: boolean, reason?: string, durationMs: number}>} deps.consult
352
357
  * @param {(event: object) => void} deps.emit - Orchestrator-event emitter for the `advisor_consult` event.
353
- * @param {{maxUses: number, used: number}} deps.budget - Session-wide budget shared by every caller's handler.
358
+ * @param {{maxUses: number, used: number}} deps.budget - Session-wide budget that every caller's handler shares.
354
359
  * @param {string} deps.model - Advisor model id, carried on the consult event.
355
360
  */
356
361
  export function advisorTool({ from, consult, emit, budget, model }) {
@@ -378,7 +383,7 @@ export function advisorTool({ from, consult, emit, budget, model }) {
378
383
  remaining,
379
384
  });
380
385
  if (r.unavailable) {
381
- // Not isError: fail-open, the caller continues normally.
386
+ // Not isError. This fails open, so the caller continues normally.
382
387
  return textResult(
383
388
  `The advisor is unavailable (${r.reason}) — proceed with your best judgment.`,
384
389
  );
@@ -477,7 +482,7 @@ export function createRequestForCommentHandler(ctx) {
477
482
  function requestForCommentTool(ctx) {
478
483
  return tool(
479
484
  "RequestForComment",
480
- "Open a new Discussion thread for long-horizon coordination on an open question. The bridge creates the thread; replies arrive asynchronously on future runs.",
485
+ "Open a new Discussion thread for long-horizon coordination on an open question. The bridge creates the thread. Replies arrive asynchronously on future runs.",
481
486
  {
482
487
  channel: z.string(),
483
488
  body: z.string(),
@@ -487,8 +492,8 @@ function requestForCommentTool(ctx) {
487
492
  );
488
493
  }
489
494
 
490
- // Re-export the building blocks discuss-tools.js needs to assemble its
491
- // own lead tool surface (it has two extra terminal tools).
495
+ // Re-export the parts discuss-tools.js needs to assemble its own lead tool
496
+ // surface (it has two extra terminal tools).
492
497
  export {
493
498
  ADJOURN_DESC,
494
499
  baseTools,
@@ -1,8 +1,8 @@
1
1
  /**
2
2
  * Render a drained batch of bus messages as tagged text lines so the
3
3
  * LLM can read its inbox at a glance. Asks and answers include the
4
- * `askId` in the tag (`[ask#42] facilitator: …`, `[answer#42] agent: …`)
5
- * so the addressee can quote it back via Answer's `askId` field.
4
+ * `askId` in the tag (`[ask#42] facilitator: …`, `[answer#42] agent: …`).
5
+ * The addressee can then quote it back in Answer's `askId` field.
6
6
  *
7
7
  * @param {Array<{from: string, text: string, kind?: string, askId?: number}>} messages
8
8
  * @returns {string}