@forwardimpact/libharness 3.0.0 → 3.0.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +60 -57
- package/package.json +2 -2
- package/src/advisor.js +47 -41
- package/src/agent-runner.js +57 -47
- package/src/benchmark/apm-installer.js +28 -28
- package/src/benchmark/env-loader.js +24 -16
- package/src/benchmark/grade.js +44 -41
- package/src/benchmark/hidden-tests.js +71 -39
- package/src/benchmark/hook-env.js +11 -9
- package/src/benchmark/invariants.js +20 -17
- package/src/benchmark/judge.js +29 -28
- package/src/benchmark/npm-installer.js +9 -8
- package/src/benchmark/report.js +53 -50
- package/src/benchmark/result.js +24 -23
- package/src/benchmark/runner.js +75 -69
- package/src/benchmark/scheduler.js +17 -16
- package/src/benchmark/task-family.js +28 -26
- package/src/benchmark/trace-split.js +8 -7
- package/src/benchmark/workdir.js +27 -25
- package/src/claude-code-executable.js +11 -11
- package/src/commands/advisor-flags.js +8 -7
- package/src/commands/assert.js +16 -15
- package/src/commands/benchmark-definition.js +11 -11
- package/src/commands/benchmark-grade.js +12 -11
- package/src/commands/benchmark-report.js +5 -5
- package/src/commands/benchmark-run.js +31 -28
- package/src/commands/by-discussion.js +10 -10
- package/src/commands/callback.js +11 -11
- package/src/commands/discuss.js +8 -7
- package/src/commands/facilitate.js +15 -13
- package/src/commands/output.js +3 -2
- package/src/commands/run.js +14 -14
- package/src/commands/scan-logs.js +21 -19
- package/src/commands/selfedit.js +14 -14
- package/src/commands/supervise.js +11 -9
- package/src/commands/task-input.js +9 -9
- package/src/commands/tee.js +10 -9
- package/src/commands/trace.js +55 -42
- package/src/commands/work-tracker.js +4 -3
- package/src/cost.js +17 -17
- package/src/discuss-tools.js +16 -16
- package/src/discusser.js +39 -38
- package/src/events/github.js +54 -37
- package/src/facilitator.js +21 -21
- package/src/inbox-poller.js +4 -4
- package/src/judge.js +32 -30
- package/src/message-bus.js +12 -11
- package/src/orchestration-loop.js +35 -36
- package/src/orchestration-toolkit.js +58 -53
- package/src/orchestrator-helpers.js +2 -2
- package/src/profile-prompt.js +54 -53
- package/src/redaction.js +63 -57
- package/src/render/line-renderer.js +5 -5
- package/src/render/orchestrator-filter.js +3 -3
- package/src/render/palette.js +11 -9
- package/src/render/tool-hints.js +18 -15
- package/src/render/turn-renderer.js +4 -4
- package/src/reply-emitter.js +2 -2
- package/src/sequence-counter.js +4 -3
- package/src/signature-filter.js +7 -6
- package/src/supervisor.js +19 -18
- package/src/tee-writer.js +25 -25
- package/src/trace-collector.js +53 -48
- package/src/trace-github.js +53 -44
- package/src/trace-multi.js +15 -13
- package/src/trace-query.js +61 -52
- package/src/trace-render.js +18 -18
- package/src/trace-usage.js +31 -28
- package/src/transcript-recorder.js +24 -20
package/src/judge.js
CHANGED
|
@@ -1,13 +1,13 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* Judge — one agent session that inspects a completed agent's work and emits
|
|
3
|
-
* a verdict
|
|
4
|
-
* `Supervisor` and `Facilitator
|
|
5
|
-
* no message bus, no orchestration loop. The judge reads the
|
|
6
|
-
*
|
|
7
|
-
* Conclude exactly once.
|
|
3
|
+
* a verdict through the orchestration `Conclude` tool. It is a parallel
|
|
4
|
+
* concept to `Supervisor` and `Facilitator`. It runs post-hoc and solo: no
|
|
5
|
+
* peer agents, no message bus, no orchestration loop. The judge reads the
|
|
6
|
+
* task. It can inspect the working directory and the trace with read-only
|
|
7
|
+
* tools. It calls Conclude exactly once.
|
|
8
8
|
*
|
|
9
|
-
*
|
|
10
|
-
* judge sessions from supervisor or facilitator sessions in a unified
|
|
9
|
+
* The judge tags trace lines with `source: "judge"`. Consumers can then tell
|
|
10
|
+
* judge sessions apart from supervisor or facilitator sessions in a unified
|
|
11
11
|
* NDJSON envelope.
|
|
12
12
|
*
|
|
13
13
|
* Follows OO+DI: constructor injection, factory function, tests bypass factory.
|
|
@@ -25,17 +25,19 @@ import {
|
|
|
25
25
|
} from "./orchestration-toolkit.js";
|
|
26
26
|
|
|
27
27
|
/**
|
|
28
|
-
* System-prompt trailer
|
|
29
|
-
* even when a `judgeProfile
|
|
30
|
-
*
|
|
31
|
-
* `FACILITATOR_SYSTEM_PROMPT` work for their
|
|
28
|
+
* System-prompt trailer for the judge's main thread. The factory always
|
|
29
|
+
* applies it, even when the caller supplies a `judgeProfile`. The profile
|
|
30
|
+
* layers on top of the trailer. `SUPERVISOR_SYSTEM_PROMPT` and
|
|
31
|
+
* `FACILITATOR_SYSTEM_PROMPT` work the same way for their roles.
|
|
32
32
|
*/
|
|
33
33
|
export const JUDGE_SYSTEM_PROMPT =
|
|
34
34
|
"You are a post-hoc judge for an agent task benchmark. " +
|
|
35
|
-
"The agent
|
|
36
|
-
"
|
|
37
|
-
"
|
|
38
|
-
"
|
|
35
|
+
"The agent already completed its work. An objective invariants step already ran. " +
|
|
36
|
+
"Confirm or override the verdict. To do so, inspect the agent's working directory and trace. " +
|
|
37
|
+
"You have read-only inspection tools to investigate: Read, Glob, Grep, and Bash. Do not modify the working directory. " +
|
|
38
|
+
"Conclude ends the session with a verdict ('success' or 'failure') and a one-paragraph summary. " +
|
|
39
|
+
"Set verdict='success' exactly when the agent's work meets the criteria the task states. " +
|
|
40
|
+
"Call Conclude as your final action. Do not deliberate across multiple turns.";
|
|
39
41
|
|
|
40
42
|
const DEFAULT_JUDGE_ALLOWED_TOOLS = ["Read", "Glob", "Grep", "Bash"];
|
|
41
43
|
|
|
@@ -45,7 +47,7 @@ const devNull = new Writable({
|
|
|
45
47
|
},
|
|
46
48
|
});
|
|
47
49
|
|
|
48
|
-
/** Run a single post-hoc judge session and emit a verdict
|
|
50
|
+
/** Run a single post-hoc judge session and emit a verdict with Conclude. */
|
|
49
51
|
export class Judge {
|
|
50
52
|
/**
|
|
51
53
|
* @param {object} deps
|
|
@@ -53,7 +55,7 @@ export class Judge {
|
|
|
53
55
|
* @param {import("stream").Writable} deps.output - Stream to emit tagged NDJSON to.
|
|
54
56
|
* @param {object} deps.ctx - Orchestration context (the Conclude handler writes to it).
|
|
55
57
|
* @param {import("./redaction.js").Redactor} deps.redactor
|
|
56
|
-
* @param {string} [deps.taskAmend] - Opaque addendum
|
|
58
|
+
* @param {string} [deps.taskAmend] - Opaque addendum. The judge appends it to the task before delivery.
|
|
57
59
|
*/
|
|
58
60
|
constructor({ runner, output, ctx, redactor, taskAmend }) {
|
|
59
61
|
if (!runner) throw new Error("runner is required");
|
|
@@ -70,7 +72,7 @@ export class Judge {
|
|
|
70
72
|
|
|
71
73
|
/**
|
|
72
74
|
* Run the judge session.
|
|
73
|
-
* @param {string} task - The judge prompt (
|
|
75
|
+
* @param {string} task - The judge prompt (the caller already substituted the placeholders).
|
|
74
76
|
* @returns {Promise<{success: boolean, verdict: string|null, summary: string|null, turns: number}>}
|
|
75
77
|
*/
|
|
76
78
|
async run(task) {
|
|
@@ -89,7 +91,7 @@ export class Judge {
|
|
|
89
91
|
return outcome;
|
|
90
92
|
}
|
|
91
93
|
|
|
92
|
-
// The judge ended
|
|
94
|
+
// The judge ended and never called Conclude. Surface that explicitly so
|
|
93
95
|
// callers can distinguish "judge said fail" from "judge never voted."
|
|
94
96
|
const outcome = {
|
|
95
97
|
success: false,
|
|
@@ -103,9 +105,9 @@ export class Judge {
|
|
|
103
105
|
|
|
104
106
|
/**
|
|
105
107
|
* Tag a single NDJSON line with `source: "judge"` and emit it to the
|
|
106
|
-
* judge's output stream.
|
|
107
|
-
* `onLine` callback
|
|
108
|
-
* for the session's trace.
|
|
108
|
+
* judge's output stream. The factory wires this into the underlying
|
|
109
|
+
* AgentRunner through the `onLine` callback. The judge's stream is then the
|
|
110
|
+
* single source of truth for the session's trace.
|
|
109
111
|
* @param {string} line
|
|
110
112
|
*/
|
|
111
113
|
emitLine(line) {
|
|
@@ -115,7 +117,7 @@ export class Judge {
|
|
|
115
117
|
}
|
|
116
118
|
|
|
117
119
|
/**
|
|
118
|
-
* Emit a final orchestrator summary line
|
|
120
|
+
* Emit a final orchestrator summary line in the universal envelope.
|
|
119
121
|
* @param {{success: boolean, verdict?: string|null, summary?: string|null, turns: number}} result
|
|
120
122
|
*/
|
|
121
123
|
emitSummary(result) {
|
|
@@ -138,20 +140,20 @@ export class Judge {
|
|
|
138
140
|
}
|
|
139
141
|
|
|
140
142
|
/**
|
|
141
|
-
* Factory function
|
|
143
|
+
* Factory function. Wires the AgentRunner with the judge orchestration server
|
|
142
144
|
* and the JUDGE_SYSTEM_PROMPT trailer. A `judgeProfile` (when supplied) layers
|
|
143
|
-
* on top of the trailer
|
|
144
|
-
* supervisor
|
|
145
|
+
* on top of the trailer through `composeSystemPrompt`. This matches the
|
|
146
|
+
* supervisor and facilitator pattern.
|
|
145
147
|
*
|
|
146
148
|
* @param {object} deps
|
|
147
149
|
* @param {string} deps.cwd - Judge working directory. Defaults to the directory whose `.claude/agents` holds `judgeProfile`.
|
|
148
|
-
* @param {function} deps.query - SDK query function (injected
|
|
150
|
+
* @param {function} deps.query - SDK query function (injected so tests can replace it).
|
|
149
151
|
* @param {import("stream").Writable} deps.output - Trace output stream.
|
|
150
152
|
* @param {import("./redaction.js").Redactor} deps.redactor
|
|
151
153
|
* @param {string} [deps.model]
|
|
152
|
-
* @param {number} [deps.maxTurns] - Default 5
|
|
153
|
-
* @param {string[]} [deps.allowedTools] - Default `["Read","Glob","Grep","Bash"]`
|
|
154
|
-
* @param {string} [deps.judgeProfile] - Profile name
|
|
154
|
+
* @param {number} [deps.maxTurns] - Default 5. The judge should act in turn 1. The other turns leave headroom for tool inspection.
|
|
155
|
+
* @param {string[]} [deps.allowedTools] - Default `["Read","Glob","Grep","Bash"]` for read-only inspection.
|
|
156
|
+
* @param {string} [deps.judgeProfile] - Profile name. `composeSystemPrompt` resolves it into the system prompt.
|
|
155
157
|
* @param {string} [deps.profilesDir] - Defaults to `<cwd>/.claude/agents`.
|
|
156
158
|
* @param {string} [deps.taskAmend]
|
|
157
159
|
* @returns {Judge}
|
package/src/message-bus.js
CHANGED
|
@@ -1,17 +1,18 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* MessageBus — in-memory per-participant message queues.
|
|
3
3
|
*
|
|
4
|
-
* Four message kinds
|
|
4
|
+
* Four message kinds exist. The bus pushes each one onto the addressee's
|
|
5
|
+
* queue:
|
|
5
6
|
*
|
|
6
|
-
* - `ask(from, to, text, askId)` — direct question
|
|
7
|
-
* pending-ask state separately.
|
|
8
|
-
*
|
|
7
|
+
* - `ask(from, to, text, askId)` — direct question. The toolkit owns the
|
|
8
|
+
* pending-ask state separately. The handler level does the fan-out
|
|
9
|
+
* (broadcast Ask). It calls `ask()` once per addressee.
|
|
9
10
|
* - `answer(from, to, text, askId)` — direct reply to the original asker.
|
|
10
11
|
* The orchestrator may inject synthetic answers (`from === "@orchestrator"`)
|
|
11
12
|
* when an Ask times out.
|
|
12
|
-
* - `announce(from, text)` — broadcast
|
|
13
|
-
* participant's queue except the sender's.
|
|
14
|
-
* - `synthetic(to, text)` — orchestrator
|
|
13
|
+
* - `announce(from, text)` — broadcast. It expects no reply. It lands on
|
|
14
|
+
* every participant's queue except the sender's.
|
|
15
|
+
* - `synthetic(to, text)` — the orchestrator alone injects a reminder.
|
|
15
16
|
*
|
|
16
17
|
* Follows OO+DI: constructor injection, factory function, tests bypass factory.
|
|
17
18
|
*/
|
|
@@ -40,9 +41,9 @@ export class MessageBus {
|
|
|
40
41
|
}
|
|
41
42
|
|
|
42
43
|
/**
|
|
43
|
-
* Reply to a pending ask. `from === "@orchestrator"`
|
|
44
|
-
* synthetic null answers
|
|
45
|
-
*
|
|
44
|
+
* Reply to a pending ask. The bus allows `from === "@orchestrator"` for
|
|
45
|
+
* synthetic null answers. The orchestrator is not a real participant. It
|
|
46
|
+
* still routes through the bus.
|
|
46
47
|
*/
|
|
47
48
|
answer(from, to, text, askId) {
|
|
48
49
|
this.#assertParticipant(to);
|
|
@@ -71,7 +72,7 @@ export class MessageBus {
|
|
|
71
72
|
this.#resolveWaiter(to);
|
|
72
73
|
}
|
|
73
74
|
|
|
74
|
-
/** Check whether a participant has pending messages
|
|
75
|
+
/** Check whether a participant has pending messages. It does not drain them. */
|
|
75
76
|
hasPending(participant) {
|
|
76
77
|
this.#assertParticipant(participant);
|
|
77
78
|
return this.queues.get(participant).length > 0;
|
|
@@ -1,22 +1,22 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* OrchestrationLoop —
|
|
2
|
+
* OrchestrationLoop — one lead LLM session coordinates N agent sessions.
|
|
3
3
|
*
|
|
4
|
-
* Ask is **async
|
|
4
|
+
* Ask is **async**. The tool returns immediately. The actual reply arrives
|
|
5
5
|
* on a later turn as `[answer#N] participant: …` on the asker's bus queue.
|
|
6
|
-
* Pending state keys by `askId` (visible in the `[ask#N]` tag)
|
|
7
|
-
* Asks to the same addressee coexist
|
|
8
|
-
*
|
|
6
|
+
* Pending state keys by `askId` (visible in the `[ask#N]` tag). Duplicate
|
|
7
|
+
* Asks to the same addressee then coexist and never overwrite each other.
|
|
8
|
+
* The asker can map each reply unambiguously back to its question.
|
|
9
9
|
*
|
|
10
|
-
* Both lead and participants follow the same outer pattern
|
|
11
|
-
* queue
|
|
10
|
+
* Both lead and participants follow the same outer pattern. Drain the bus
|
|
11
|
+
* queue. Run or resume the LLM with the drained messages. Then settle any
|
|
12
12
|
* unanswered Asks the participant owes. They differ only in how the first
|
|
13
|
-
* turn starts
|
|
13
|
+
* turn starts. The lead receives the task. Participants wait for traffic.
|
|
14
14
|
*
|
|
15
15
|
* Termination signals:
|
|
16
16
|
* - `ctx.concluded` — explicit Conclude / Adjourn / Recess.
|
|
17
|
-
* - `stopped` — broader
|
|
18
|
-
* other abort path. Loops watch `stopped
|
|
19
|
-
* for the summary's success
|
|
17
|
+
* - `stopped` — broader. It is also true on lead error, agent crash, or any
|
|
18
|
+
* other abort path. Loops watch `stopped`. The code uses `ctx.concluded`
|
|
19
|
+
* only for the summary's success and verdict.
|
|
20
20
|
*/
|
|
21
21
|
import { SequenceCounter } from "./sequence-counter.js";
|
|
22
22
|
import {
|
|
@@ -26,10 +26,10 @@ import {
|
|
|
26
26
|
} from "./orchestration-toolkit.js";
|
|
27
27
|
import { formatMessages } from "./orchestrator-helpers.js";
|
|
28
28
|
|
|
29
|
-
/** Default per-session lead-turn budget
|
|
29
|
+
/** Default per-session lead-turn budget. It fits multi-round injected conversations. */
|
|
30
30
|
const DEFAULT_MAX_LEAD_TURNS = 200;
|
|
31
31
|
|
|
32
|
-
/**
|
|
32
|
+
/** Coordinate N agent sessions from a single lead LLM session. */
|
|
33
33
|
export class OrchestrationLoop {
|
|
34
34
|
/**
|
|
35
35
|
* @param {object} deps
|
|
@@ -42,7 +42,7 @@ export class OrchestrationLoop {
|
|
|
42
42
|
* @param {object} deps.ctx - Orchestration context (from `createOrchestrationContext()`).
|
|
43
43
|
* @param {object} deps.redactor
|
|
44
44
|
* @param {number} [deps.maxLeadTurns] - Cap on lead resumes per session (default 200).
|
|
45
|
-
* @param {string} [deps.taskAmend] -
|
|
45
|
+
* @param {string} [deps.taskAmend] - The loop appends it to the task before delivery.
|
|
46
46
|
* @param {import("./inbox-poller.js").InboxPoller} [deps.inboxPoller]
|
|
47
47
|
* @param {AbortController} [deps.abortController]
|
|
48
48
|
*/
|
|
@@ -90,7 +90,7 @@ export class OrchestrationLoop {
|
|
|
90
90
|
this.#signalDone = resolveDone;
|
|
91
91
|
}
|
|
92
92
|
|
|
93
|
-
/** Internal
|
|
93
|
+
/** Internal. Resolves when `stopped` flips true so waiters unblock. */
|
|
94
94
|
#signalDone;
|
|
95
95
|
|
|
96
96
|
/**
|
|
@@ -112,9 +112,9 @@ export class OrchestrationLoop {
|
|
|
112
112
|
this.#stop();
|
|
113
113
|
};
|
|
114
114
|
|
|
115
|
-
// Start agent loops in parallel.
|
|
116
|
-
//
|
|
117
|
-
//
|
|
115
|
+
// Start agent loops in parallel. The wrapper makes a crash flip `stopped`
|
|
116
|
+
// and still resolves itself. Promise.allSettled below then never sees an
|
|
117
|
+
// unhandled rejection.
|
|
118
118
|
const agentPromises = this.agents.map((a) =>
|
|
119
119
|
this.#runAgent(a).catch(abort),
|
|
120
120
|
);
|
|
@@ -153,14 +153,14 @@ export class OrchestrationLoop {
|
|
|
153
153
|
}
|
|
154
154
|
|
|
155
155
|
/**
|
|
156
|
-
* Lead loop. The lead's first turn carries the task
|
|
157
|
-
*
|
|
156
|
+
* Lead loop. The lead's first turn carries the task. Every later turn is
|
|
157
|
+
* a resume, and something that lands on its inbox triggers it.
|
|
158
158
|
*
|
|
159
159
|
* `messages.length === 0` from `#drainOrWait` means the session ended
|
|
160
|
-
* before any message arrived
|
|
161
|
-
*
|
|
162
|
-
*
|
|
163
|
-
*
|
|
160
|
+
* before any message arrived. That is the natural exit. If `drainOrWait`
|
|
161
|
+
* returned messages, deliver them even when the session concluded in the
|
|
162
|
+
* microtask window between wake-up and this check. The inbox already holds
|
|
163
|
+
* them, so the lead should see them.
|
|
164
164
|
*/
|
|
165
165
|
async #runLead(initialTask) {
|
|
166
166
|
this.leadTurns = 1;
|
|
@@ -190,8 +190,8 @@ export class OrchestrationLoop {
|
|
|
190
190
|
}
|
|
191
191
|
|
|
192
192
|
/**
|
|
193
|
-
* Agent loop. The first message off the inbox triggers `run()
|
|
194
|
-
*
|
|
193
|
+
* Agent loop. The first message off the inbox triggers `run()`. Every
|
|
194
|
+
* later batch triggers `resume()`. The loop has no turn budget. The agent
|
|
195
195
|
* runner's own `maxTurns` caps each SDK call.
|
|
196
196
|
*/
|
|
197
197
|
async #runAgent({ name, runner }) {
|
|
@@ -235,10 +235,10 @@ export class OrchestrationLoop {
|
|
|
235
235
|
|
|
236
236
|
/**
|
|
237
237
|
* If `name` left a pending Ask unanswered, inject one synthetic reminder
|
|
238
|
-
* and resume once more. If still unanswered after the reminder, emit
|
|
239
|
-
* `protocol_violation` event per outstanding ask and cancel them
|
|
240
|
-
* asker's queue gets a synthetic `[no answer: …]
|
|
241
|
-
* on a participant that
|
|
238
|
+
* and resume once more. If it is still unanswered after the reminder, emit
|
|
239
|
+
* a `protocol_violation` event per outstanding ask and cancel them. The
|
|
240
|
+
* asker's queue then gets a synthetic `[no answer: …]`, so the asker does
|
|
241
|
+
* not deadlock on a participant that silently ignores its inbox.
|
|
242
242
|
*/
|
|
243
243
|
async #settleOwedAsks(name, runner) {
|
|
244
244
|
if (pendingAsksOwedBy(this.ctx, name).length === 0) return;
|
|
@@ -267,9 +267,9 @@ export class OrchestrationLoop {
|
|
|
267
267
|
}
|
|
268
268
|
|
|
269
269
|
/**
|
|
270
|
-
* Emit one NDJSON line
|
|
271
|
-
*
|
|
272
|
-
*
|
|
270
|
+
* Emit one NDJSON line in the universal `{source, seq, event}` envelope.
|
|
271
|
+
* Tag it with its source (the participant name) and a monotonic seq.
|
|
272
|
+
* Each runner's `onLine` callback calls this.
|
|
273
273
|
* @param {string} source
|
|
274
274
|
* @param {string} line - Raw NDJSON line from the SDK iterator.
|
|
275
275
|
*/
|
|
@@ -288,8 +288,7 @@ export class OrchestrationLoop {
|
|
|
288
288
|
|
|
289
289
|
/**
|
|
290
290
|
* Emit one orchestrator-source event (`session_start`, `agent_start`,
|
|
291
|
-
* `protocol_violation`, `lead_turn_limit`)
|
|
292
|
-
* envelope.
|
|
291
|
+
* `protocol_violation`, `lead_turn_limit`) in the universal envelope.
|
|
293
292
|
* @param {object} event
|
|
294
293
|
*/
|
|
295
294
|
emitOrchestratorEvent(event) {
|
|
@@ -306,7 +305,7 @@ export class OrchestrationLoop {
|
|
|
306
305
|
|
|
307
306
|
/**
|
|
308
307
|
* Emit the terminal summary line. `Discusser` emits its own discuss-
|
|
309
|
-
* augmented summary after this one
|
|
308
|
+
* augmented summary after this one. Trace consumers keep the last
|
|
310
309
|
* summary they see.
|
|
311
310
|
* @param {{success: boolean, verdict?: string|null, turns: number, summary?: string|null}} result
|
|
312
311
|
*/
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* OrchestrationToolkit — tool schemas, per-role tool sets, and handler
|
|
3
3
|
* factories for orchestration between leads (facilitator, supervisor,
|
|
4
|
-
* discuss-lead) and
|
|
4
|
+
* discuss-lead) and the agents that take part.
|
|
5
5
|
*
|
|
6
6
|
* **Tool surface, by role:**
|
|
7
7
|
*
|
|
@@ -15,16 +15,16 @@
|
|
|
15
15
|
* | Discuss agt | ✓ | ✓ | ✓ | ✓ | | RFC |
|
|
16
16
|
* | Judge | | | | | ✓ | |
|
|
17
17
|
*
|
|
18
|
-
* **Ask is async.** Ask returns `{askIds:[…]}` immediately
|
|
18
|
+
* **Ask is async.** Ask returns `{askIds:[…]}` immediately. It posts the
|
|
19
19
|
* question to the addressee's bus queue. The reply arrives on the asker's
|
|
20
|
-
* next turn as `[answer#N] <participant>: <text>`.
|
|
21
|
-
* `askId` (visible in `[ask#N]` tags)
|
|
22
|
-
* addressee coexist
|
|
20
|
+
* next turn as `[answer#N] <participant>: <text>`. The toolkit keys pending
|
|
21
|
+
* state by `askId` (visible in `[ask#N]` tags). Duplicate Asks to the same
|
|
22
|
+
* addressee then coexist and never overwrite each other.
|
|
23
23
|
*
|
|
24
24
|
* **Answer's `askId` is optional.** With a matching askId, the reply
|
|
25
|
-
* routes to that specific asker. Without, the handler auto-picks
|
|
26
|
-
* exactly one ask is owed to the caller
|
|
27
|
-
* as an Announce so it still reaches everyone.
|
|
25
|
+
* routes to that specific asker. Without one, the handler auto-picks when
|
|
26
|
+
* exactly one ask is owed to the caller. Otherwise the handler routes the
|
|
27
|
+
* message as an Announce so it still reaches everyone.
|
|
28
28
|
*/
|
|
29
29
|
|
|
30
30
|
import { createSdkMcpServer, tool } from "@anthropic-ai/claude-agent-sdk";
|
|
@@ -48,9 +48,9 @@ export function createOrchestrationContext() {
|
|
|
48
48
|
|
|
49
49
|
/**
|
|
50
50
|
* Guard for terminal tools (`Conclude`, `Adjourn`, `Recess`). Returns an
|
|
51
|
-
* error result when the caller still has Asks in flight
|
|
52
|
-
* end the turn and wait for the auto-resume. Returns `null`
|
|
53
|
-
* are pending and the terminal tool is free to run.
|
|
51
|
+
* error result when the caller still has Asks in flight. That result tells
|
|
52
|
+
* the caller to end the turn and wait for the auto-resume. Returns `null`
|
|
53
|
+
* when no Asks are pending and the terminal tool is free to run.
|
|
54
54
|
*/
|
|
55
55
|
export function requireNoPendingAsks(ctx) {
|
|
56
56
|
if (ctx.pendingAsks.size === 0) return null;
|
|
@@ -62,18 +62,18 @@ export function requireNoPendingAsks(ctx) {
|
|
|
62
62
|
/**
|
|
63
63
|
* Guard for terminal tools in discuss mode (`Adjourn`, `Recess`). Returns
|
|
64
64
|
* an error result when the lead's inbox has unprocessed messages from the
|
|
65
|
-
* human
|
|
66
|
-
* Returns `null` when no inbox messages are pending and the
|
|
67
|
-
* is free to run.
|
|
65
|
+
* human. That result tells the lead to end the turn and wait for the
|
|
66
|
+
* auto-resume. Returns `null` when no inbox messages are pending and the
|
|
67
|
+
* terminal tool is free to run.
|
|
68
68
|
*/
|
|
69
69
|
export function requireNoUnprocessedInbox(ctx) {
|
|
70
70
|
if (!ctx.messageBus?.hasPending?.("lead")) return null;
|
|
71
71
|
return errorResult(
|
|
72
|
-
"New messages from the human are
|
|
72
|
+
"New messages from the human are in your inbox. End your turn. You will be resumed to process them.",
|
|
73
73
|
);
|
|
74
74
|
}
|
|
75
75
|
|
|
76
|
-
/** Mark the session as concluded
|
|
76
|
+
/** Mark the session as concluded. Cancel any open Asks so askers see the synthetic null on their next turn. */
|
|
77
77
|
export function createConcludeHandler(ctx) {
|
|
78
78
|
return async ({ verdict, summary }) => {
|
|
79
79
|
const guard = requireNoPendingAsks(ctx);
|
|
@@ -84,11 +84,11 @@ export function createConcludeHandler(ctx) {
|
|
|
84
84
|
}
|
|
85
85
|
|
|
86
86
|
/**
|
|
87
|
-
* Shared terminal-tool helper. Conclude
|
|
88
|
-
* same three context fields (`concluded`, `verdict`, `summary`)
|
|
89
|
-
* cancel any in-flight Asks for the same reason
|
|
90
|
-
* answer them now. Mode-specific handlers (Adjourn, Recess) layer
|
|
91
|
-
*
|
|
87
|
+
* Shared terminal-tool helper. Conclude, Adjourn, and Recess all set the
|
|
88
|
+
* same three context fields (`concluded`, `verdict`, `summary`). All three
|
|
89
|
+
* also cancel any in-flight Asks for the same reason. Nobody will ever
|
|
90
|
+
* answer them now. Mode-specific handlers (Adjourn, Recess) layer extra
|
|
91
|
+
* state on top before they call this.
|
|
92
92
|
*/
|
|
93
93
|
export function concludeSession(ctx, { verdict, summary, reason }) {
|
|
94
94
|
ctx.concluded = true;
|
|
@@ -124,21 +124,24 @@ function registerPendingAsk(ctx, { from, addressee, question }) {
|
|
|
124
124
|
}
|
|
125
125
|
|
|
126
126
|
/**
|
|
127
|
-
* Create an Ask handler.
|
|
128
|
-
* the ask on the bus
|
|
129
|
-
* those ids to match the `[answer#N]` it sees on
|
|
127
|
+
* Create an Ask handler. The handler registers a pending entry for each
|
|
128
|
+
* addressee. It posts the ask on the bus. It returns `{askIds:[…]}`
|
|
129
|
+
* immediately. The LLM uses those ids to match the `[answer#N]` it sees on
|
|
130
|
+
* a later turn.
|
|
130
131
|
*
|
|
131
132
|
* @param {object} ctx
|
|
132
133
|
* @param {object} opts
|
|
133
134
|
* @param {string} opts.from
|
|
134
135
|
* @param {string|undefined} opts.defaultTo - `undefined` means "broadcast
|
|
135
|
-
* to everyone else"
|
|
136
|
+
* to everyone else". A participant name means "target that one when
|
|
136
137
|
* `to` is omitted."
|
|
137
138
|
*/
|
|
138
139
|
export function createAskHandler(ctx, { from, defaultTo }) {
|
|
139
140
|
return async ({ question, to }) => {
|
|
140
141
|
if (ctx.concluded) {
|
|
141
|
-
return errorResult(
|
|
142
|
+
return errorResult(
|
|
143
|
+
"The session is concluded. The handler did not deliver your Ask.",
|
|
144
|
+
);
|
|
142
145
|
}
|
|
143
146
|
const addressees = resolveAddressees(ctx, { from, to, defaultTo });
|
|
144
147
|
if (addressees.length === 0) {
|
|
@@ -157,7 +160,8 @@ export function createAskHandler(ctx, { from, defaultTo }) {
|
|
|
157
160
|
* - askId provided + matches a pending entry whose addressee is the caller →
|
|
158
161
|
* route the reply to the asker's queue and clear the pending entry.
|
|
159
162
|
* - askId provided but unknown or wrong addressee → `isError`. The caller
|
|
160
|
-
* tried to
|
|
163
|
+
* tried to name an askId. The handler tells the caller why it did not
|
|
164
|
+
* match.
|
|
161
165
|
* - askId omitted + exactly one ask owed by the caller → auto-pick it.
|
|
162
166
|
* - askId omitted + 0 or many pending → broadcast as Announce so the
|
|
163
167
|
* message still reaches every other participant.
|
|
@@ -180,9 +184,9 @@ export function createAnswerHandler(ctx, { from }) {
|
|
|
180
184
|
ctx.messageBus.announce(from, message);
|
|
181
185
|
const reason =
|
|
182
186
|
owed.length === 0
|
|
183
|
-
? "no pending ask
|
|
184
|
-
:
|
|
185
|
-
return textResult(`Answer routed as Announce
|
|
187
|
+
? "You have no pending ask."
|
|
188
|
+
: `You have ${owed.length} pending asks. An omitted askId is ambiguous.`;
|
|
189
|
+
return textResult(`Answer routed as Announce. ${reason}`);
|
|
186
190
|
};
|
|
187
191
|
}
|
|
188
192
|
|
|
@@ -191,7 +195,7 @@ function routeAnswerByAskId(ctx, { from, askId, message }) {
|
|
|
191
195
|
if (!entry) return errorResult(`No pending ask with askId=${askId}.`);
|
|
192
196
|
if (entry.addresseeName !== from) {
|
|
193
197
|
return errorResult(
|
|
194
|
-
`Ask #${askId} is addressed to ${entry.addresseeName}
|
|
198
|
+
`Ask #${askId} is addressed to ${entry.addresseeName}. You are ${from}.`,
|
|
195
199
|
);
|
|
196
200
|
}
|
|
197
201
|
ctx.pendingAsks.delete(askId);
|
|
@@ -208,9 +212,9 @@ export function createAnnounceHandler(ctx, { from }) {
|
|
|
208
212
|
}
|
|
209
213
|
|
|
210
214
|
/**
|
|
211
|
-
* Cancel pending Asks
|
|
212
|
-
*
|
|
213
|
-
*
|
|
215
|
+
* Cancel pending Asks. Route a synthetic `[no answer: <reason>]` to each
|
|
216
|
+
* asker's queue, so callers never deadlock when a participant ignores its
|
|
217
|
+
* inbox.
|
|
214
218
|
*
|
|
215
219
|
* @param {object} ctx
|
|
216
220
|
* @param {string} reason - Surfaced inside `[no answer: <reason>]`.
|
|
@@ -234,8 +238,8 @@ export function pendingAsksOwedBy(ctx, addressee) {
|
|
|
234
238
|
}
|
|
235
239
|
|
|
236
240
|
/**
|
|
237
|
-
* Inject a synthetic reminder onto the addressee's bus queue
|
|
238
|
-
*
|
|
241
|
+
* Inject a synthetic reminder onto the addressee's bus queue. Mark each
|
|
242
|
+
* owed ask as reminded. Returns true when a reminder fired.
|
|
239
243
|
*/
|
|
240
244
|
export function remindOwedAsks(ctx, addressee) {
|
|
241
245
|
const owed = pendingAsksOwedBy(ctx, addressee).filter((e) => !e.reminded);
|
|
@@ -252,13 +256,13 @@ export function remindOwedAsks(ctx, addressee) {
|
|
|
252
256
|
// --- Tool descriptions (shared across roles) ---
|
|
253
257
|
|
|
254
258
|
const ASK_DESC_BROADCAST =
|
|
255
|
-
"Send a question to one named participant
|
|
259
|
+
"Send a question to one named participant. Omit 'to' to broadcast to every other participant. Returns {askIds:[…]} immediately. The reply arrives on a later turn as `[answer#N] <from>: <text>` in your inbox.";
|
|
256
260
|
|
|
257
261
|
const ASK_DESC_TARGETED = (target) =>
|
|
258
|
-
`Send a question to ${target}. Returns {askIds:[N]} immediately
|
|
262
|
+
`Send a question to ${target}. Returns {askIds:[N]} immediately. The reply arrives on a later turn as \`[answer#N] ${target}: <text>\` in your inbox.`;
|
|
259
263
|
|
|
260
264
|
const ANSWER_DESC =
|
|
261
|
-
"Reply to an ask addressed to you. Quote askId from the [ask#N] tag on the question
|
|
265
|
+
"Reply to an ask addressed to you. Quote askId from the [ask#N] tag on the question. Omit askId and the handler auto-picks the only pending ask. When 0 or many asks are pending, the handler routes your message as an Announce.";
|
|
262
266
|
|
|
263
267
|
const ANNOUNCE_DESC = "Broadcast a message with no reply expected.";
|
|
264
268
|
|
|
@@ -275,7 +279,7 @@ const ADJOURN_DESC =
|
|
|
275
279
|
"End the discussion. Provide a verdict ('adjourned' or 'failed') and a summary. Cancels any unanswered Asks.";
|
|
276
280
|
|
|
277
281
|
const RECESS_DESC =
|
|
278
|
-
"End the run
|
|
282
|
+
"End the run. Schedule an out-of-session re-dispatch. Cancels any unanswered Asks. Use only when you wait on an external reply or duration. Do not use to wait on in-flight Asks.";
|
|
279
283
|
|
|
280
284
|
// --- Tool builders ---
|
|
281
285
|
|
|
@@ -283,7 +287,7 @@ const RECESS_DESC =
|
|
|
283
287
|
function textResult(text) {
|
|
284
288
|
return { content: [{ type: "text", text }] };
|
|
285
289
|
}
|
|
286
|
-
/** Build an MCP tool error result
|
|
290
|
+
/** Build an MCP tool error result that wraps a single text message. */
|
|
287
291
|
function errorResult(text) {
|
|
288
292
|
return { content: [{ type: "text", text }], isError: true };
|
|
289
293
|
}
|
|
@@ -298,10 +302,10 @@ function jsonResult(obj) {
|
|
|
298
302
|
* @param {object} ctx
|
|
299
303
|
* @param {object} opts
|
|
300
304
|
* @param {string} opts.from - Caller's canonical name.
|
|
301
|
-
* @param {string|undefined} opts.defaultTo - Default Ask target
|
|
305
|
+
* @param {string|undefined} opts.defaultTo - Default Ask target. `undefined`
|
|
302
306
|
* means "broadcast across everyone else when `to` is omitted."
|
|
303
307
|
* @param {boolean} opts.broadcast - Whether Ask accepts a `to` field at all.
|
|
304
|
-
* Leads with multiple participants set this true
|
|
308
|
+
* Leads with multiple participants set this true. Supervise's
|
|
305
309
|
* single-participant roles set it false.
|
|
306
310
|
*/
|
|
307
311
|
function baseTools(ctx, { from, defaultTo, broadcast }) {
|
|
@@ -338,19 +342,20 @@ function concludeTool(ctx) {
|
|
|
338
342
|
}
|
|
339
343
|
|
|
340
344
|
const ADVISOR_DESC =
|
|
341
|
-
"Consult a stronger model on one focused question.
|
|
345
|
+
"Consult a stronger model on one focused question. The tool forwards your full session context (system prompt, prompts, transcript so far) automatically. You cannot restrict it. The advice returns in the tool result. All participants share one session-wide consult budget.";
|
|
342
346
|
|
|
343
347
|
/**
|
|
344
|
-
* Build the `Advisor` consult tool for one caller.
|
|
345
|
-
* modes pass it into the agent tool-server factories
|
|
346
|
-
*
|
|
347
|
-
* dependency
|
|
348
|
+
* Build the `Advisor` consult tool for one caller. The tool is
|
|
349
|
+
* mode-agnostic. Loop modes pass it into the agent tool-server factories
|
|
350
|
+
* through `extraTools`. Run mode gives it a dedicated server. The tool has
|
|
351
|
+
* no orchestration-context dependency. The budget object and the emit
|
|
352
|
+
* callback arrive as injected dependencies.
|
|
348
353
|
*
|
|
349
354
|
* @param {object} deps
|
|
350
355
|
* @param {string} deps.from - Caller's canonical name (event attribution).
|
|
351
356
|
* @param {(question: string) => Promise<{advice?: string, unavailable?: boolean, reason?: string, durationMs: number}>} deps.consult
|
|
352
357
|
* @param {(event: object) => void} deps.emit - Orchestrator-event emitter for the `advisor_consult` event.
|
|
353
|
-
* @param {{maxUses: number, used: number}} deps.budget - Session-wide budget
|
|
358
|
+
* @param {{maxUses: number, used: number}} deps.budget - Session-wide budget that every caller's handler shares.
|
|
354
359
|
* @param {string} deps.model - Advisor model id, carried on the consult event.
|
|
355
360
|
*/
|
|
356
361
|
export function advisorTool({ from, consult, emit, budget, model }) {
|
|
@@ -378,7 +383,7 @@ export function advisorTool({ from, consult, emit, budget, model }) {
|
|
|
378
383
|
remaining,
|
|
379
384
|
});
|
|
380
385
|
if (r.unavailable) {
|
|
381
|
-
// Not isError
|
|
386
|
+
// Not isError. This fails open, so the caller continues normally.
|
|
382
387
|
return textResult(
|
|
383
388
|
`The advisor is unavailable (${r.reason}) — proceed with your best judgment.`,
|
|
384
389
|
);
|
|
@@ -477,7 +482,7 @@ export function createRequestForCommentHandler(ctx) {
|
|
|
477
482
|
function requestForCommentTool(ctx) {
|
|
478
483
|
return tool(
|
|
479
484
|
"RequestForComment",
|
|
480
|
-
"Open a new Discussion thread for long-horizon coordination on an open question. The bridge creates the thread
|
|
485
|
+
"Open a new Discussion thread for long-horizon coordination on an open question. The bridge creates the thread. Replies arrive asynchronously on future runs.",
|
|
481
486
|
{
|
|
482
487
|
channel: z.string(),
|
|
483
488
|
body: z.string(),
|
|
@@ -487,8 +492,8 @@ function requestForCommentTool(ctx) {
|
|
|
487
492
|
);
|
|
488
493
|
}
|
|
489
494
|
|
|
490
|
-
// Re-export the
|
|
491
|
-
//
|
|
495
|
+
// Re-export the parts discuss-tools.js needs to assemble its own lead tool
|
|
496
|
+
// surface (it has two extra terminal tools).
|
|
492
497
|
export {
|
|
493
498
|
ADJOURN_DESC,
|
|
494
499
|
baseTools,
|
|
@@ -1,8 +1,8 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* Render a drained batch of bus messages as tagged text lines so the
|
|
3
3
|
* LLM can read its inbox at a glance. Asks and answers include the
|
|
4
|
-
* `askId` in the tag (`[ask#42] facilitator: …`, `[answer#42] agent: …`)
|
|
5
|
-
*
|
|
4
|
+
* `askId` in the tag (`[ask#42] facilitator: …`, `[answer#42] agent: …`).
|
|
5
|
+
* The addressee can then quote it back in Answer's `askId` field.
|
|
6
6
|
*
|
|
7
7
|
* @param {Array<{from: string, text: string, kind?: string, askId?: number}>} messages
|
|
8
8
|
* @returns {string}
|