@forwardimpact/libharness 0.1.22 → 1.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -201
- package/README.md +196 -80
- package/bin/fit-benchmark.js +44 -0
- package/bin/fit-harness.js +358 -0
- package/bin/fit-selfedit.js +165 -0
- package/bin/fit-trace.js +510 -0
- package/package.json +41 -11
- package/src/agent-runner.js +256 -0
- package/src/benchmark/apm-installer.js +207 -0
- package/src/benchmark/env-loader.js +158 -0
- package/src/benchmark/hook-env.js +40 -0
- package/src/benchmark/invariants.js +141 -0
- package/src/benchmark/judge.js +187 -0
- package/src/benchmark/npm-installer.js +87 -0
- package/src/benchmark/report.js +522 -0
- package/src/benchmark/result.js +127 -0
- package/src/benchmark/runner.js +583 -0
- package/src/benchmark/task-family.js +260 -0
- package/src/benchmark/workdir.js +298 -0
- package/src/commands/assert.js +153 -0
- package/src/commands/benchmark-definition.js +165 -0
- package/src/commands/benchmark-invariants.js +73 -0
- package/src/commands/benchmark-report.js +51 -0
- package/src/commands/benchmark-run.js +111 -0
- package/src/commands/by-discussion.js +94 -0
- package/src/commands/callback.js +119 -0
- package/src/commands/discuss.js +132 -0
- package/src/commands/facilitate.js +123 -0
- package/src/commands/output.js +36 -0
- package/src/commands/run.js +152 -0
- package/src/commands/supervise.js +136 -0
- package/src/commands/task-input.js +54 -0
- package/src/commands/tee.js +53 -0
- package/src/commands/trace.js +630 -0
- package/src/commands/work-tracker.js +35 -0
- package/src/cost.js +79 -0
- package/src/discuss-tools.js +173 -0
- package/src/discusser.js +394 -0
- package/src/events/github.js +161 -0
- package/src/facilitator.js +205 -0
- package/src/inbox-poller.js +81 -0
- package/src/index.js +72 -2
- package/src/judge.js +210 -0
- package/src/message-bus.js +118 -0
- package/src/orchestration-loop.js +330 -0
- package/src/orchestration-toolkit.js +441 -0
- package/src/orchestrator-helpers.js +23 -0
- package/src/profile-prompt.js +266 -0
- package/src/redaction.js +253 -0
- package/src/render/line-renderer.js +54 -0
- package/src/render/orchestrator-filter.js +19 -0
- package/src/render/palette.js +63 -0
- package/src/render/tool-hints.js +154 -0
- package/src/render/turn-renderer.js +96 -0
- package/src/reply-emitter.js +47 -0
- package/src/sequence-counter.js +21 -0
- package/src/signature-filter.js +27 -0
- package/src/supervisor.js +236 -0
- package/src/tee-writer.js +150 -0
- package/src/trace-collector.js +444 -0
- package/src/trace-github.js +473 -0
- package/src/trace-multi.js +101 -0
- package/src/trace-query.js +748 -0
- package/src/trace-render.js +211 -0
- package/src/trace-usage.js +249 -0
- package/src/fixture/assertions.js +0 -42
- package/src/fixture/cache.js +0 -50
- package/src/fixture/eval.js +0 -146
- package/src/fixture/index.js +0 -9
- package/src/fixture/pathway.js +0 -451
- package/src/fixture/services.js +0 -56
- package/src/mock/clients.js +0 -135
- package/src/mock/config.js +0 -45
- package/src/mock/data.js +0 -46
- package/src/mock/fs.js +0 -111
- package/src/mock/grpc.js +0 -94
- package/src/mock/http.js +0 -60
- package/src/mock/index.js +0 -36
- package/src/mock/infra.js +0 -219
- package/src/mock/logger.js +0 -42
- package/src/mock/observer.js +0 -74
- package/src/mock/resource-index.js +0 -95
- package/src/mock/service-callbacks.js +0 -39
- package/src/mock/services.js +0 -79
- package/src/mock/spy.js +0 -44
- package/src/mock/storage.js +0 -118
package/src/judge.js
ADDED
|
@@ -0,0 +1,210 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Judge — one agent session that inspects a completed agent's work and emits
|
|
3
|
+
* a verdict via the orchestration `Conclude` tool. Parallel concept to
|
|
4
|
+
* `Supervisor` and `Facilitator`, but post-hoc and solo: no peer agents,
|
|
5
|
+
* no message bus, no orchestration loop. The judge reads the task, optionally
|
|
6
|
+
* inspects the working directory and trace via read-only tools, and calls
|
|
7
|
+
* Conclude exactly once.
|
|
8
|
+
*
|
|
9
|
+
* Trace lines are tagged `source: "judge"` so consumers can distinguish
|
|
10
|
+
* judge sessions from supervisor or facilitator sessions in a unified
|
|
11
|
+
* NDJSON envelope.
|
|
12
|
+
*
|
|
13
|
+
* Follows OO+DI: constructor injection, factory function, tests bypass factory.
|
|
14
|
+
*/
|
|
15
|
+
|
|
16
|
+
import { resolve } from "node:path";
|
|
17
|
+
import { Writable } from "node:stream";
|
|
18
|
+
|
|
19
|
+
import { createAgentRunner } from "./agent-runner.js";
|
|
20
|
+
import { composeSystemPrompt } from "./profile-prompt.js";
|
|
21
|
+
import { SequenceCounter } from "./sequence-counter.js";
|
|
22
|
+
import {
|
|
23
|
+
createJudgeToolServer,
|
|
24
|
+
createOrchestrationContext,
|
|
25
|
+
} from "./orchestration-toolkit.js";
|
|
26
|
+
|
|
27
|
+
/**
|
|
28
|
+
* System-prompt trailer appended to the judge's main thread. Always applied,
|
|
29
|
+
* even when a `judgeProfile` is supplied — the profile layers on top of the
|
|
30
|
+
* trailer, the same way `SUPERVISOR_SYSTEM_PROMPT` and
|
|
31
|
+
* `FACILITATOR_SYSTEM_PROMPT` work for their respective roles.
|
|
32
|
+
*/
|
|
33
|
+
export const JUDGE_SYSTEM_PROMPT =
|
|
34
|
+
"You are a post-hoc judge for an agent task benchmark. " +
|
|
35
|
+
"The agent has already completed its work and an objective invariants step has already run; your role is to confirm or override the verdict by inspecting the agent's working directory and trace. " +
|
|
36
|
+
"You have read-only inspection tools — Read, Glob, Grep, Bash — to investigate; do not modify the working directory. " +
|
|
37
|
+
"Conclude ends the session with a verdict ('success' or 'failure') and a one-paragraph summary; verdict='success' iff the agent's work meets the criteria stated in the task. " +
|
|
38
|
+
"Call Conclude as your final action — do not deliberate across multiple turns.";
|
|
39
|
+
|
|
40
|
+
const DEFAULT_JUDGE_ALLOWED_TOOLS = ["Read", "Glob", "Grep", "Bash"];
|
|
41
|
+
|
|
42
|
+
const devNull = new Writable({
|
|
43
|
+
write(_chunk, _enc, cb) {
|
|
44
|
+
cb();
|
|
45
|
+
},
|
|
46
|
+
});
|
|
47
|
+
|
|
48
|
+
/** Run a single post-hoc judge session and emit a verdict via Conclude. */
|
|
49
|
+
export class Judge {
|
|
50
|
+
/**
|
|
51
|
+
* @param {object} deps
|
|
52
|
+
* @param {import("./agent-runner.js").AgentRunner} deps.runner - The judge's AgentRunner.
|
|
53
|
+
* @param {import("stream").Writable} deps.output - Stream to emit tagged NDJSON to.
|
|
54
|
+
* @param {object} deps.ctx - Orchestration context (the Conclude handler writes to it).
|
|
55
|
+
* @param {import("./redaction.js").Redactor} deps.redactor
|
|
56
|
+
* @param {string} [deps.taskAmend] - Opaque addendum appended to the task before delivery.
|
|
57
|
+
*/
|
|
58
|
+
constructor({ runner, output, ctx, redactor, taskAmend }) {
|
|
59
|
+
if (!runner) throw new Error("runner is required");
|
|
60
|
+
if (!output) throw new Error("output is required");
|
|
61
|
+
if (!ctx) throw new Error("ctx is required");
|
|
62
|
+
if (!redactor) throw new Error("redactor is required");
|
|
63
|
+
this.runner = runner;
|
|
64
|
+
this.output = output;
|
|
65
|
+
this.ctx = ctx;
|
|
66
|
+
this.redactor = redactor;
|
|
67
|
+
this.taskAmend = taskAmend ?? null;
|
|
68
|
+
this.counter = new SequenceCounter();
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
/**
|
|
72
|
+
* Run the judge session.
|
|
73
|
+
* @param {string} task - The judge prompt (with placeholders already substituted).
|
|
74
|
+
* @returns {Promise<{success: boolean, verdict: string|null, summary: string|null, turns: number}>}
|
|
75
|
+
*/
|
|
76
|
+
async run(task) {
|
|
77
|
+
const fullTask = this.taskAmend ? `${task}\n\n${this.taskAmend}` : task;
|
|
78
|
+
const result = await this.runner.run(fullTask);
|
|
79
|
+
|
|
80
|
+
if (this.ctx.concluded) {
|
|
81
|
+
const success = this.ctx.verdict === "success";
|
|
82
|
+
const outcome = {
|
|
83
|
+
success,
|
|
84
|
+
verdict: this.ctx.verdict,
|
|
85
|
+
summary: this.ctx.summary ?? null,
|
|
86
|
+
turns: 1,
|
|
87
|
+
};
|
|
88
|
+
this.emitSummary(outcome);
|
|
89
|
+
return outcome;
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
// The judge ended without calling Conclude. Surface that explicitly so
|
|
93
|
+
// callers can distinguish "judge said fail" from "judge never voted."
|
|
94
|
+
const outcome = {
|
|
95
|
+
success: false,
|
|
96
|
+
verdict: null,
|
|
97
|
+
summary: null,
|
|
98
|
+
turns: result.success ? 1 : 0,
|
|
99
|
+
};
|
|
100
|
+
this.emitSummary(outcome);
|
|
101
|
+
return outcome;
|
|
102
|
+
}
|
|
103
|
+
|
|
104
|
+
/**
|
|
105
|
+
* Tag a single NDJSON line with `source: "judge"` and emit it to the
|
|
106
|
+
* judge's output stream. Wired into the underlying AgentRunner via the
|
|
107
|
+
* `onLine` callback so the judge's stream is the single source of truth
|
|
108
|
+
* for the session's trace.
|
|
109
|
+
* @param {string} line
|
|
110
|
+
*/
|
|
111
|
+
emitLine(line) {
|
|
112
|
+
const event = JSON.parse(line);
|
|
113
|
+
const tagged = { source: "judge", seq: this.counter.next(), event };
|
|
114
|
+
this.output.write(JSON.stringify(this.redactor.redactValue(tagged)) + "\n");
|
|
115
|
+
}
|
|
116
|
+
|
|
117
|
+
/**
|
|
118
|
+
* Emit a final orchestrator summary line, wrapped in the universal envelope.
|
|
119
|
+
* @param {{success: boolean, verdict?: string|null, summary?: string|null, turns: number}} result
|
|
120
|
+
*/
|
|
121
|
+
emitSummary(result) {
|
|
122
|
+
this.output.write(
|
|
123
|
+
JSON.stringify(
|
|
124
|
+
this.redactor.redactValue({
|
|
125
|
+
source: "orchestrator",
|
|
126
|
+
seq: this.counter.next(),
|
|
127
|
+
event: {
|
|
128
|
+
type: "summary",
|
|
129
|
+
success: result.success,
|
|
130
|
+
...(result.verdict && { verdict: result.verdict }),
|
|
131
|
+
turns: result.turns,
|
|
132
|
+
...(result.summary && { summary: result.summary }),
|
|
133
|
+
},
|
|
134
|
+
}),
|
|
135
|
+
) + "\n",
|
|
136
|
+
);
|
|
137
|
+
}
|
|
138
|
+
}
|
|
139
|
+
|
|
140
|
+
/**
|
|
141
|
+
* Factory function — wires the AgentRunner with the judge orchestration server
|
|
142
|
+
* and the JUDGE_SYSTEM_PROMPT trailer. A `judgeProfile` (when supplied) layers
|
|
143
|
+
* on top of the trailer via `composeSystemPrompt`, matching the
|
|
144
|
+
* supervisor/facilitator pattern.
|
|
145
|
+
*
|
|
146
|
+
* @param {object} deps
|
|
147
|
+
* @param {string} deps.cwd - Judge working directory. Defaults to the directory whose `.claude/agents` holds `judgeProfile`.
|
|
148
|
+
* @param {function} deps.query - SDK query function (injected for testing).
|
|
149
|
+
* @param {import("stream").Writable} deps.output - Trace output stream.
|
|
150
|
+
* @param {import("./redaction.js").Redactor} deps.redactor
|
|
151
|
+
* @param {string} [deps.model]
|
|
152
|
+
* @param {number} [deps.maxTurns] - Default 5 (the judge is expected to act in turn 1; 5 leaves headroom for tool inspection).
|
|
153
|
+
* @param {string[]} [deps.allowedTools] - Default `["Read","Glob","Grep","Bash"]` — read-only inspection.
|
|
154
|
+
* @param {string} [deps.judgeProfile] - Profile name; resolved into the system prompt via `composeSystemPrompt`.
|
|
155
|
+
* @param {string} [deps.profilesDir] - Defaults to `<cwd>/.claude/agents`.
|
|
156
|
+
* @param {string} [deps.taskAmend]
|
|
157
|
+
* @returns {Judge}
|
|
158
|
+
*/
|
|
159
|
+
export function createJudge({
|
|
160
|
+
cwd,
|
|
161
|
+
query,
|
|
162
|
+
output,
|
|
163
|
+
redactor,
|
|
164
|
+
model,
|
|
165
|
+
maxTurns,
|
|
166
|
+
allowedTools,
|
|
167
|
+
judgeProfile,
|
|
168
|
+
profilesDir,
|
|
169
|
+
taskAmend,
|
|
170
|
+
runtime,
|
|
171
|
+
}) {
|
|
172
|
+
if (!cwd) throw new Error("cwd is required");
|
|
173
|
+
if (!query) throw new Error("query is required");
|
|
174
|
+
if (!output) throw new Error("output is required");
|
|
175
|
+
if (!redactor) throw new Error("redactor is required");
|
|
176
|
+
if (!runtime) throw new Error("runtime is required");
|
|
177
|
+
|
|
178
|
+
const resolvedProfilesDir = profilesDir ?? resolve(cwd, ".claude/agents");
|
|
179
|
+
const systemPrompt = composeSystemPrompt({
|
|
180
|
+
role: "agent",
|
|
181
|
+
profile: judgeProfile,
|
|
182
|
+
profilesDir: resolvedProfilesDir,
|
|
183
|
+
trailer: JUDGE_SYSTEM_PROMPT,
|
|
184
|
+
runtime,
|
|
185
|
+
});
|
|
186
|
+
|
|
187
|
+
const ctx = createOrchestrationContext();
|
|
188
|
+
ctx.participants = [{ name: "judge", role: "judge" }];
|
|
189
|
+
const judgeServer = createJudgeToolServer(ctx);
|
|
190
|
+
|
|
191
|
+
let judge;
|
|
192
|
+
const onLine = (line) => judge.emitLine(line);
|
|
193
|
+
|
|
194
|
+
const runner = createAgentRunner({
|
|
195
|
+
cwd,
|
|
196
|
+
query,
|
|
197
|
+
output: devNull,
|
|
198
|
+
model,
|
|
199
|
+
maxTurns: maxTurns ?? 5,
|
|
200
|
+
allowedTools: allowedTools ?? DEFAULT_JUDGE_ALLOWED_TOOLS,
|
|
201
|
+
onLine,
|
|
202
|
+
settingSources: ["project"],
|
|
203
|
+
systemPrompt,
|
|
204
|
+
mcpServers: { orchestration: judgeServer },
|
|
205
|
+
redactor,
|
|
206
|
+
});
|
|
207
|
+
|
|
208
|
+
judge = new Judge({ runner, output, ctx, redactor, taskAmend });
|
|
209
|
+
return judge;
|
|
210
|
+
}
|
|
@@ -0,0 +1,118 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* MessageBus — in-memory per-participant message queues.
|
|
3
|
+
*
|
|
4
|
+
* Four message kinds, each pushed onto the addressee's queue:
|
|
5
|
+
*
|
|
6
|
+
* - `ask(from, to, text, askId)` — direct question; the toolkit owns the
|
|
7
|
+
* pending-ask state separately. Fan-out (broadcast Ask) happens at the
|
|
8
|
+
* handler level by calling `ask()` once per addressee.
|
|
9
|
+
* - `answer(from, to, text, askId)` — direct reply to the original asker.
|
|
10
|
+
* The orchestrator may inject synthetic answers (`from === "@orchestrator"`)
|
|
11
|
+
* when an Ask times out.
|
|
12
|
+
* - `announce(from, text)` — broadcast, no reply expected; lands on every
|
|
13
|
+
* participant's queue except the sender's.
|
|
14
|
+
* - `synthetic(to, text)` — orchestrator-only reminder injection.
|
|
15
|
+
*
|
|
16
|
+
* Follows OO+DI: constructor injection, factory function, tests bypass factory.
|
|
17
|
+
*/
|
|
18
|
+
|
|
19
|
+
/** In-memory per-participant message queues. */
|
|
20
|
+
export class MessageBus {
|
|
21
|
+
/**
|
|
22
|
+
* @param {object} deps
|
|
23
|
+
* @param {string[]} deps.participants - Canonical participant names.
|
|
24
|
+
*/
|
|
25
|
+
constructor({ participants }) {
|
|
26
|
+
this.queues = new Map();
|
|
27
|
+
this.waiters = new Map();
|
|
28
|
+
for (const name of participants) {
|
|
29
|
+
this.queues.set(name, []);
|
|
30
|
+
this.waiters.set(name, null);
|
|
31
|
+
}
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
/** Send a question to one participant. */
|
|
35
|
+
ask(from, to, text, askId) {
|
|
36
|
+
this.#assertParticipant(from);
|
|
37
|
+
this.#assertParticipant(to);
|
|
38
|
+
this.queues.get(to).push({ from, text, kind: "ask", askId });
|
|
39
|
+
this.#resolveWaiter(to);
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
/**
|
|
43
|
+
* Reply to a pending ask. `from === "@orchestrator"` is allowed for
|
|
44
|
+
* synthetic null answers — the orchestrator is not a real participant
|
|
45
|
+
* but it routes through the bus.
|
|
46
|
+
*/
|
|
47
|
+
answer(from, to, text, askId) {
|
|
48
|
+
this.#assertParticipant(to);
|
|
49
|
+
if (from !== "@orchestrator") this.#assertParticipant(from);
|
|
50
|
+
this.queues.get(to).push({ from, text, kind: "answer", askId });
|
|
51
|
+
this.#resolveWaiter(to);
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
/** Broadcast a message to every participant except the sender. */
|
|
55
|
+
announce(from, text) {
|
|
56
|
+
this.#assertParticipant(from);
|
|
57
|
+
const msg = { from, text, kind: "announce" };
|
|
58
|
+
for (const [name, queue] of this.queues) {
|
|
59
|
+
if (name === from) continue;
|
|
60
|
+
queue.push(msg);
|
|
61
|
+
this.#resolveWaiter(name);
|
|
62
|
+
}
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
/** Inject an orchestrator-originated reminder onto one participant's queue. */
|
|
66
|
+
synthetic(to, text) {
|
|
67
|
+
this.#assertParticipant(to);
|
|
68
|
+
this.queues
|
|
69
|
+
.get(to)
|
|
70
|
+
.push({ from: "@orchestrator", text, kind: "synthetic" });
|
|
71
|
+
this.#resolveWaiter(to);
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
/** Check whether a participant has pending messages without draining them. */
|
|
75
|
+
hasPending(participant) {
|
|
76
|
+
this.#assertParticipant(participant);
|
|
77
|
+
return this.queues.get(participant).length > 0;
|
|
78
|
+
}
|
|
79
|
+
|
|
80
|
+
/** Return and clear pending messages for a participant. */
|
|
81
|
+
drain(participant) {
|
|
82
|
+
this.#assertParticipant(participant);
|
|
83
|
+
return this.queues.get(participant).splice(0);
|
|
84
|
+
}
|
|
85
|
+
|
|
86
|
+
/**
|
|
87
|
+
* Return a Promise that resolves when at least one message is pending.
|
|
88
|
+
* Resolves immediately if messages are already queued.
|
|
89
|
+
*/
|
|
90
|
+
waitForMessages(participant) {
|
|
91
|
+
this.#assertParticipant(participant);
|
|
92
|
+
if (this.queues.get(participant).length > 0) {
|
|
93
|
+
return Promise.resolve();
|
|
94
|
+
}
|
|
95
|
+
return new Promise((resolve) => {
|
|
96
|
+
this.waiters.set(participant, resolve);
|
|
97
|
+
});
|
|
98
|
+
}
|
|
99
|
+
|
|
100
|
+
#assertParticipant(name) {
|
|
101
|
+
if (!this.queues.has(name)) {
|
|
102
|
+
throw new Error(`Unknown participant: ${name}`);
|
|
103
|
+
}
|
|
104
|
+
}
|
|
105
|
+
|
|
106
|
+
#resolveWaiter(name) {
|
|
107
|
+
const waiter = this.waiters.get(name);
|
|
108
|
+
if (waiter) {
|
|
109
|
+
this.waiters.set(name, null);
|
|
110
|
+
waiter();
|
|
111
|
+
}
|
|
112
|
+
}
|
|
113
|
+
}
|
|
114
|
+
|
|
115
|
+
/** Factory function. */
|
|
116
|
+
export function createMessageBus(deps) {
|
|
117
|
+
return new MessageBus(deps);
|
|
118
|
+
}
|
|
@@ -0,0 +1,330 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* OrchestrationLoop — N agent sessions coordinated by one lead LLM session.
|
|
3
|
+
*
|
|
4
|
+
* Ask is **async**: the tool returns immediately, the actual reply arrives
|
|
5
|
+
* on a later turn as `[answer#N] participant: …` on the asker's bus queue.
|
|
6
|
+
* Pending state keys by `askId` (visible in the `[ask#N]` tag), so duplicate
|
|
7
|
+
* Asks to the same addressee coexist without overwriting each other, and
|
|
8
|
+
* the asker can map each reply unambiguously back to its question.
|
|
9
|
+
*
|
|
10
|
+
* Both lead and participants follow the same outer pattern: drain the bus
|
|
11
|
+
* queue, run / resume the LLM with the drained messages, then settle any
|
|
12
|
+
* unanswered Asks the participant owes. They differ only in how the first
|
|
13
|
+
* turn starts (the lead receives the task; participants wait for traffic).
|
|
14
|
+
*
|
|
15
|
+
* Termination signals:
|
|
16
|
+
* - `ctx.concluded` — explicit Conclude / Adjourn / Recess.
|
|
17
|
+
* - `stopped` — broader: also true on lead error, agent crash, or any
|
|
18
|
+
* other abort path. Loops watch `stopped`; `ctx.concluded` is only used
|
|
19
|
+
* for the summary's success/verdict.
|
|
20
|
+
*/
|
|
21
|
+
import { SequenceCounter } from "./sequence-counter.js";
|
|
22
|
+
import {
|
|
23
|
+
cancelPendingAsks,
|
|
24
|
+
pendingAsksOwedBy,
|
|
25
|
+
remindOwedAsks,
|
|
26
|
+
} from "./orchestration-toolkit.js";
|
|
27
|
+
import { formatMessages } from "./orchestrator-helpers.js";
|
|
28
|
+
|
|
29
|
+
/** Default per-session lead-turn budget — accommodates multi-round injected conversations. */
|
|
30
|
+
const DEFAULT_MAX_LEAD_TURNS = 200;
|
|
31
|
+
|
|
32
|
+
/** Orchestrate N agent sessions coordinated by a single lead LLM session. */
|
|
33
|
+
export class OrchestrationLoop {
|
|
34
|
+
/**
|
|
35
|
+
* @param {object} deps
|
|
36
|
+
* @param {import("./agent-runner.js").AgentRunner} deps.leadRunner
|
|
37
|
+
* @param {Array<{name: string, role: string, runner: import("./agent-runner.js").AgentRunner}>} deps.agents
|
|
38
|
+
* @param {import("./message-bus.js").MessageBus} deps.messageBus
|
|
39
|
+
* @param {import("stream").Writable} deps.output
|
|
40
|
+
* @param {string} deps.leadName - Canonical name of the lead participant on the bus.
|
|
41
|
+
* @param {"facilitated"|"discussion"|"supervised"} deps.mode - Carries through to `protocol_violation` events.
|
|
42
|
+
* @param {object} deps.ctx - Orchestration context (from `createOrchestrationContext()`).
|
|
43
|
+
* @param {object} deps.redactor
|
|
44
|
+
* @param {number} [deps.maxLeadTurns] - Cap on lead resumes per session (default 200).
|
|
45
|
+
* @param {string} [deps.taskAmend] - Appended to the task before delivery.
|
|
46
|
+
* @param {import("./inbox-poller.js").InboxPoller} [deps.inboxPoller]
|
|
47
|
+
* @param {AbortController} [deps.abortController]
|
|
48
|
+
*/
|
|
49
|
+
constructor({
|
|
50
|
+
leadRunner,
|
|
51
|
+
agents,
|
|
52
|
+
messageBus,
|
|
53
|
+
output,
|
|
54
|
+
leadName,
|
|
55
|
+
mode,
|
|
56
|
+
maxLeadTurns,
|
|
57
|
+
ctx,
|
|
58
|
+
taskAmend,
|
|
59
|
+
redactor,
|
|
60
|
+
inboxPoller,
|
|
61
|
+
abortController,
|
|
62
|
+
}) {
|
|
63
|
+
if (!leadRunner) throw new Error("leadRunner is required");
|
|
64
|
+
if (!agents) throw new Error("agents is required");
|
|
65
|
+
if (!messageBus) throw new Error("messageBus is required");
|
|
66
|
+
if (!output) throw new Error("output is required");
|
|
67
|
+
if (!leadName) throw new Error("leadName is required");
|
|
68
|
+
if (!mode) throw new Error("mode is required");
|
|
69
|
+
if (!ctx) throw new Error("ctx is required");
|
|
70
|
+
if (!redactor) throw new Error("redactor is required");
|
|
71
|
+
this.leadRunner = leadRunner;
|
|
72
|
+
this.agents = agents;
|
|
73
|
+
this.messageBus = messageBus;
|
|
74
|
+
this.output = output;
|
|
75
|
+
this.leadName = leadName;
|
|
76
|
+
this.mode = mode;
|
|
77
|
+
this.ctx = ctx;
|
|
78
|
+
this.redactor = redactor;
|
|
79
|
+
this.taskAmend = taskAmend ?? null;
|
|
80
|
+
this.maxLeadTurns = maxLeadTurns ?? DEFAULT_MAX_LEAD_TURNS;
|
|
81
|
+
this.inboxPoller = inboxPoller ?? null;
|
|
82
|
+
this.abortController = abortController ?? null;
|
|
83
|
+
this.counter = new SequenceCounter();
|
|
84
|
+
this.leadTurns = 0;
|
|
85
|
+
this.stopped = false;
|
|
86
|
+
let resolveDone;
|
|
87
|
+
this.donePromise = new Promise((r) => {
|
|
88
|
+
resolveDone = r;
|
|
89
|
+
});
|
|
90
|
+
this.#signalDone = resolveDone;
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
/** Internal — resolved when `stopped` flips true so waiters unblock. */
|
|
94
|
+
#signalDone;
|
|
95
|
+
|
|
96
|
+
/**
|
|
97
|
+
* Run the full orchestrated session.
|
|
98
|
+
* @param {string} task
|
|
99
|
+
* @returns {Promise<{success: boolean, turns: number}>}
|
|
100
|
+
*/
|
|
101
|
+
async run(task) {
|
|
102
|
+
this.emitOrchestratorEvent({ type: "session_start" });
|
|
103
|
+
const initialTask = this.taskAmend
|
|
104
|
+
? task
|
|
105
|
+
? `${task}\n\n${this.taskAmend}`
|
|
106
|
+
: this.taskAmend
|
|
107
|
+
: task;
|
|
108
|
+
|
|
109
|
+
let firstError = null;
|
|
110
|
+
const abort = (err) => {
|
|
111
|
+
if (err && !firstError) firstError = err;
|
|
112
|
+
this.#stop();
|
|
113
|
+
};
|
|
114
|
+
|
|
115
|
+
// Start agent loops in parallel. Wrapped so a crash flips `stopped`
|
|
116
|
+
// but the wrapper itself resolves — Promise.allSettled below never
|
|
117
|
+
// sees an unhandled rejection.
|
|
118
|
+
const agentPromises = this.agents.map((a) =>
|
|
119
|
+
this.#runAgent(a).catch(abort),
|
|
120
|
+
);
|
|
121
|
+
const pollerPromise = this.inboxPoller?.run().catch(() => {});
|
|
122
|
+
|
|
123
|
+
try {
|
|
124
|
+
await this.#runLead(initialTask);
|
|
125
|
+
} catch (err) {
|
|
126
|
+
abort(err);
|
|
127
|
+
} finally {
|
|
128
|
+
this.#stop();
|
|
129
|
+
}
|
|
130
|
+
|
|
131
|
+
await Promise.allSettled([...agentPromises, pollerPromise].filter(Boolean));
|
|
132
|
+
if (firstError) throw firstError;
|
|
133
|
+
|
|
134
|
+
const success = this.ctx.concluded && this.ctx.verdict === "success";
|
|
135
|
+
this.emitSummary({
|
|
136
|
+
success,
|
|
137
|
+
verdict: this.ctx.verdict,
|
|
138
|
+
turns: this.leadTurns,
|
|
139
|
+
summary: this.ctx.summary,
|
|
140
|
+
});
|
|
141
|
+
return { success, turns: this.leadTurns };
|
|
142
|
+
}
|
|
143
|
+
|
|
144
|
+
#stop() {
|
|
145
|
+
if (this.stopped) return;
|
|
146
|
+
this.stopped = true;
|
|
147
|
+
this.#signalDone();
|
|
148
|
+
this.abortController?.abort();
|
|
149
|
+
for (const agent of this.agents) {
|
|
150
|
+
agent.runner.currentAbortController?.abort();
|
|
151
|
+
}
|
|
152
|
+
this.leadRunner.currentAbortController?.abort();
|
|
153
|
+
}
|
|
154
|
+
|
|
155
|
+
/**
|
|
156
|
+
* Lead loop. The lead's first turn carries the task; every subsequent
|
|
157
|
+
* turn is a resume triggered by something landing on its inbox.
|
|
158
|
+
*
|
|
159
|
+
* `messages.length === 0` from `#drainOrWait` means the session ended
|
|
160
|
+
* before any message arrived — that's the natural exit. If
|
|
161
|
+
* `drainOrWait` returned messages, deliver them even if the session
|
|
162
|
+
* concluded in the microtask window between wake-up and this check;
|
|
163
|
+
* the inbox already has them and they deserve to be seen.
|
|
164
|
+
*/
|
|
165
|
+
async #runLead(initialTask) {
|
|
166
|
+
this.leadTurns = 1;
|
|
167
|
+
this.emitOrchestratorEvent({ type: "agent_start", agent: this.leadName });
|
|
168
|
+
await this.leadRunner.run(initialTask);
|
|
169
|
+
if (this.#exiting()) return;
|
|
170
|
+
await this.#settleOwedAsks(this.leadName, this.leadRunner);
|
|
171
|
+
|
|
172
|
+
while (!this.#exiting()) {
|
|
173
|
+
if (this.leadTurns >= this.maxLeadTurns) {
|
|
174
|
+
this.emitOrchestratorEvent({
|
|
175
|
+
type: "lead_turn_limit",
|
|
176
|
+
limit: this.maxLeadTurns,
|
|
177
|
+
});
|
|
178
|
+
return;
|
|
179
|
+
}
|
|
180
|
+
const messages = await this.#drainOrWait(this.leadName);
|
|
181
|
+
if (messages.length === 0) return;
|
|
182
|
+
|
|
183
|
+
this.leadTurns++;
|
|
184
|
+
const hasSynthetic = messages.some((m) => m.kind === "synthetic");
|
|
185
|
+
await this.leadRunner.resume(formatMessages(messages));
|
|
186
|
+
if (hasSynthetic) this.inboxPoller?.markActed();
|
|
187
|
+
if (this.#exiting()) return;
|
|
188
|
+
await this.#settleOwedAsks(this.leadName, this.leadRunner);
|
|
189
|
+
}
|
|
190
|
+
}
|
|
191
|
+
|
|
192
|
+
/**
|
|
193
|
+
* Agent loop. The first message off the inbox triggers `run()`; every
|
|
194
|
+
* subsequent batch triggers `resume()`. No turn budget — the agent
|
|
195
|
+
* runner's own `maxTurns` caps each SDK call.
|
|
196
|
+
*/
|
|
197
|
+
async #runAgent({ name, runner }) {
|
|
198
|
+
let started = false;
|
|
199
|
+
while (!this.#exiting()) {
|
|
200
|
+
const messages = await this.#drainOrWait(name);
|
|
201
|
+
if (messages.length === 0) return;
|
|
202
|
+
|
|
203
|
+
if (!started) {
|
|
204
|
+
started = true;
|
|
205
|
+
this.emitOrchestratorEvent({ type: "agent_start", agent: name });
|
|
206
|
+
await runner.run(formatMessages(messages));
|
|
207
|
+
} else {
|
|
208
|
+
await runner.resume(formatMessages(messages));
|
|
209
|
+
}
|
|
210
|
+
if (this.#exiting()) return;
|
|
211
|
+
await this.#settleOwedAsks(name, runner);
|
|
212
|
+
}
|
|
213
|
+
}
|
|
214
|
+
|
|
215
|
+
/** Either an explicit Conclude or any abort path. */
|
|
216
|
+
#exiting() {
|
|
217
|
+
return this.stopped || this.ctx.concluded;
|
|
218
|
+
}
|
|
219
|
+
|
|
220
|
+
/**
|
|
221
|
+
* Drain the queue, or wait for the first message to arrive. Returns an
|
|
222
|
+
* empty array when the session ended before any message landed.
|
|
223
|
+
*/
|
|
224
|
+
async #drainOrWait(name) {
|
|
225
|
+
let messages = this.messageBus.drain(name);
|
|
226
|
+
if (messages.length > 0) return messages;
|
|
227
|
+
await Promise.race([
|
|
228
|
+
this.messageBus.waitForMessages(name),
|
|
229
|
+
this.donePromise,
|
|
230
|
+
]);
|
|
231
|
+
if (this.stopped) return [];
|
|
232
|
+
messages = this.messageBus.drain(name);
|
|
233
|
+
return messages;
|
|
234
|
+
}
|
|
235
|
+
|
|
236
|
+
/**
|
|
237
|
+
* If `name` left a pending Ask unanswered, inject one synthetic reminder
|
|
238
|
+
* and resume once more. If still unanswered after the reminder, emit a
|
|
239
|
+
* `protocol_violation` event per outstanding ask and cancel them — the
|
|
240
|
+
* asker's queue gets a synthetic `[no answer: …]` so it doesn't deadlock
|
|
241
|
+
* on a participant that's silently ignoring its inbox.
|
|
242
|
+
*/
|
|
243
|
+
async #settleOwedAsks(name, runner) {
|
|
244
|
+
if (pendingAsksOwedBy(this.ctx, name).length === 0) return;
|
|
245
|
+
if (this.stopped) return;
|
|
246
|
+
|
|
247
|
+
const reminded = remindOwedAsks(this.ctx, name);
|
|
248
|
+
if (!reminded) return;
|
|
249
|
+
const reminders = this.messageBus.drain(name);
|
|
250
|
+
if (reminders.length === 0) return;
|
|
251
|
+
|
|
252
|
+
await runner.resume(formatMessages(reminders));
|
|
253
|
+
if (this.stopped) return;
|
|
254
|
+
|
|
255
|
+
const stillOwed = pendingAsksOwedBy(this.ctx, name);
|
|
256
|
+
if (stillOwed.length === 0) return;
|
|
257
|
+
|
|
258
|
+
for (const entry of stillOwed) {
|
|
259
|
+
this.emitOrchestratorEvent({
|
|
260
|
+
type: "protocol_violation",
|
|
261
|
+
agent: name,
|
|
262
|
+
askId: entry.askId,
|
|
263
|
+
mode: this.mode,
|
|
264
|
+
});
|
|
265
|
+
}
|
|
266
|
+
cancelPendingAsks(this.ctx, `${name} did not answer after reminder`, name);
|
|
267
|
+
}
|
|
268
|
+
|
|
269
|
+
/**
|
|
270
|
+
* Emit one NDJSON line tagged with its source (participant name) and a
|
|
271
|
+
* monotonic seq, wrapped in the universal `{source, seq, event}` envelope.
|
|
272
|
+
* Called from each runner's `onLine` callback.
|
|
273
|
+
* @param {string} source
|
|
274
|
+
* @param {string} line - Raw NDJSON line from the SDK iterator.
|
|
275
|
+
*/
|
|
276
|
+
emitLine(source, line) {
|
|
277
|
+
const event = JSON.parse(line);
|
|
278
|
+
this.output.write(
|
|
279
|
+
JSON.stringify(
|
|
280
|
+
this.redactor.redactValue({
|
|
281
|
+
source,
|
|
282
|
+
seq: this.counter.next(),
|
|
283
|
+
event,
|
|
284
|
+
}),
|
|
285
|
+
) + "\n",
|
|
286
|
+
);
|
|
287
|
+
}
|
|
288
|
+
|
|
289
|
+
/**
|
|
290
|
+
* Emit one orchestrator-source event (`session_start`, `agent_start`,
|
|
291
|
+
* `protocol_violation`, `lead_turn_limit`) wrapped in the universal
|
|
292
|
+
* envelope.
|
|
293
|
+
* @param {object} event
|
|
294
|
+
*/
|
|
295
|
+
emitOrchestratorEvent(event) {
|
|
296
|
+
this.output.write(
|
|
297
|
+
JSON.stringify(
|
|
298
|
+
this.redactor.redactValue({
|
|
299
|
+
source: "orchestrator",
|
|
300
|
+
seq: this.counter.next(),
|
|
301
|
+
event,
|
|
302
|
+
}),
|
|
303
|
+
) + "\n",
|
|
304
|
+
);
|
|
305
|
+
}
|
|
306
|
+
|
|
307
|
+
/**
|
|
308
|
+
* Emit the terminal summary line. `Discusser` emits its own discuss-
|
|
309
|
+
* augmented summary after this one; trace consumers keep the last
|
|
310
|
+
* summary they see.
|
|
311
|
+
* @param {{success: boolean, verdict?: string|null, turns: number, summary?: string|null}} result
|
|
312
|
+
*/
|
|
313
|
+
emitSummary(result) {
|
|
314
|
+
this.output.write(
|
|
315
|
+
JSON.stringify(
|
|
316
|
+
this.redactor.redactValue({
|
|
317
|
+
source: "orchestrator",
|
|
318
|
+
seq: this.counter.next(),
|
|
319
|
+
event: {
|
|
320
|
+
type: "summary",
|
|
321
|
+
success: result.success,
|
|
322
|
+
...(result.verdict && { verdict: result.verdict }),
|
|
323
|
+
turns: result.turns,
|
|
324
|
+
...(result.summary && { summary: result.summary }),
|
|
325
|
+
},
|
|
326
|
+
}),
|
|
327
|
+
) + "\n",
|
|
328
|
+
);
|
|
329
|
+
}
|
|
330
|
+
}
|