@forwardimpact/libharness 0.1.20 → 1.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -201
- package/README.md +196 -80
- package/bin/fit-benchmark.js +44 -0
- package/bin/fit-harness.js +358 -0
- package/bin/fit-selfedit.js +165 -0
- package/bin/fit-trace.js +510 -0
- package/package.json +42 -12
- package/src/agent-runner.js +256 -0
- package/src/benchmark/apm-installer.js +207 -0
- package/src/benchmark/env-loader.js +158 -0
- package/src/benchmark/hook-env.js +40 -0
- package/src/benchmark/invariants.js +141 -0
- package/src/benchmark/judge.js +187 -0
- package/src/benchmark/npm-installer.js +87 -0
- package/src/benchmark/report.js +522 -0
- package/src/benchmark/result.js +127 -0
- package/src/benchmark/runner.js +583 -0
- package/src/benchmark/task-family.js +260 -0
- package/src/benchmark/workdir.js +298 -0
- package/src/commands/assert.js +153 -0
- package/src/commands/benchmark-definition.js +165 -0
- package/src/commands/benchmark-invariants.js +73 -0
- package/src/commands/benchmark-report.js +51 -0
- package/src/commands/benchmark-run.js +111 -0
- package/src/commands/by-discussion.js +94 -0
- package/src/commands/callback.js +119 -0
- package/src/commands/discuss.js +132 -0
- package/src/commands/facilitate.js +123 -0
- package/src/commands/output.js +36 -0
- package/src/commands/run.js +152 -0
- package/src/commands/supervise.js +136 -0
- package/src/commands/task-input.js +54 -0
- package/src/commands/tee.js +53 -0
- package/src/commands/trace.js +630 -0
- package/src/commands/work-tracker.js +35 -0
- package/src/cost.js +79 -0
- package/src/discuss-tools.js +173 -0
- package/src/discusser.js +394 -0
- package/src/events/github.js +161 -0
- package/src/facilitator.js +205 -0
- package/src/inbox-poller.js +81 -0
- package/src/index.js +72 -2
- package/src/judge.js +210 -0
- package/src/message-bus.js +118 -0
- package/src/orchestration-loop.js +330 -0
- package/src/orchestration-toolkit.js +441 -0
- package/src/orchestrator-helpers.js +23 -0
- package/src/profile-prompt.js +266 -0
- package/src/redaction.js +253 -0
- package/src/render/line-renderer.js +54 -0
- package/src/render/orchestrator-filter.js +19 -0
- package/src/render/palette.js +63 -0
- package/src/render/tool-hints.js +154 -0
- package/src/render/turn-renderer.js +96 -0
- package/src/reply-emitter.js +47 -0
- package/src/sequence-counter.js +21 -0
- package/src/signature-filter.js +27 -0
- package/src/supervisor.js +236 -0
- package/src/tee-writer.js +150 -0
- package/src/trace-collector.js +444 -0
- package/src/trace-github.js +473 -0
- package/src/trace-multi.js +101 -0
- package/src/trace-query.js +748 -0
- package/src/trace-render.js +211 -0
- package/src/trace-usage.js +249 -0
- package/src/fixture/assertions.js +0 -42
- package/src/fixture/cache.js +0 -50
- package/src/fixture/eval.js +0 -146
- package/src/fixture/index.js +0 -9
- package/src/fixture/pathway.js +0 -451
- package/src/fixture/services.js +0 -56
- package/src/mock/clients.js +0 -135
- package/src/mock/config.js +0 -45
- package/src/mock/data.js +0 -46
- package/src/mock/fs.js +0 -111
- package/src/mock/grpc.js +0 -94
- package/src/mock/http.js +0 -60
- package/src/mock/index.js +0 -36
- package/src/mock/infra.js +0 -219
- package/src/mock/logger.js +0 -42
- package/src/mock/observer.js +0 -74
- package/src/mock/resource-index.js +0 -95
- package/src/mock/service-callbacks.js +0 -39
- package/src/mock/services.js +0 -79
- package/src/mock/spy.js +0 -44
- package/src/mock/storage.js +0 -118
package/src/cost.js
ADDED
|
@@ -0,0 +1,79 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Cost aggregation over Claude Code NDJSON traces — the single source of
|
|
3
|
+
* truth for "how much did this run cost, across every participant?".
|
|
4
|
+
*
|
|
5
|
+
* The SDK reports the cumulative session cost on each `result` event as
|
|
6
|
+
* `total_cost_usd`. Supervised, facilitated, and discuss sessions interleave
|
|
7
|
+
* one runner's events with another's in a single combined trace, wrapping
|
|
8
|
+
* each in a `{source, seq, event}` envelope; a plain `run` trace carries bare
|
|
9
|
+
* events with no envelope. A judge runs as its own session in a separate
|
|
10
|
+
* trace. In every case the rule is the same: sum the `total_cost_usd` of each
|
|
11
|
+
* `result` event, and keep a per-source breakdown so callers can attribute
|
|
12
|
+
* spend to the agent, supervisor, judge, or any named participant.
|
|
13
|
+
*
|
|
14
|
+
* This mirrors `TraceCollector.handleResult`, which accumulates the same
|
|
15
|
+
* figure for its summary footer — kept as a standalone pure helper so the
|
|
16
|
+
* benchmark runner, the callback command, and `fit-trace cost` share one
|
|
17
|
+
* implementation rather than each re-deriving it (and drifting).
|
|
18
|
+
*/
|
|
19
|
+
|
|
20
|
+
/** Bucket key for bare (un-enveloped) `run`-mode events: a lone agent session. */
|
|
21
|
+
export const UNSOURCED = "agent";
|
|
22
|
+
|
|
23
|
+
/**
|
|
24
|
+
* Sum `total_cost_usd` across every `result` event in an NDJSON trace.
|
|
25
|
+
*
|
|
26
|
+
* @param {Iterable<string>} lines - NDJSON lines (e.g. `content.split("\n")`).
|
|
27
|
+
* Blank and malformed lines are skipped.
|
|
28
|
+
* @returns {{totalCostUsd: number, bySource: Record<string, number>}}
|
|
29
|
+
* `totalCostUsd` is the sum across all participants; `bySource` maps each
|
|
30
|
+
* envelope `source` (or {@link UNSOURCED} for bare events) to its subtotal.
|
|
31
|
+
*/
|
|
32
|
+
export function sumTraceCost(lines) {
|
|
33
|
+
let totalCostUsd = 0;
|
|
34
|
+
/** @type {Record<string, number>} */
|
|
35
|
+
const bySource = {};
|
|
36
|
+
|
|
37
|
+
for (const line of lines) {
|
|
38
|
+
const parsed = parseCostLine(line);
|
|
39
|
+
if (!parsed) continue;
|
|
40
|
+
const { source, cost } = parsed;
|
|
41
|
+
totalCostUsd += cost;
|
|
42
|
+
bySource[source] = (bySource[source] ?? 0) + cost;
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
return { totalCostUsd, bySource };
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
/**
|
|
49
|
+
* Parse a single NDJSON line and return its `result`-event cost contribution,
|
|
50
|
+
* or null when the line is blank, malformed, not a result event, or carries
|
|
51
|
+
* no numeric `total_cost_usd`.
|
|
52
|
+
*
|
|
53
|
+
* @param {string} line
|
|
54
|
+
* @returns {{source: string, cost: number} | null}
|
|
55
|
+
*/
|
|
56
|
+
function parseCostLine(line) {
|
|
57
|
+
const trimmed = line.trim();
|
|
58
|
+
if (!trimmed) return null;
|
|
59
|
+
|
|
60
|
+
let event;
|
|
61
|
+
try {
|
|
62
|
+
event = JSON.parse(trimmed);
|
|
63
|
+
} catch {
|
|
64
|
+
return null;
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
// Unwrap the combined-trace envelope {source, seq, event}; bare events
|
|
68
|
+
// (plain `run` traces) have a `type` and no `source`.
|
|
69
|
+
let source = UNSOURCED;
|
|
70
|
+
if (event.event && !event.type && typeof event.source === "string") {
|
|
71
|
+
source = event.source;
|
|
72
|
+
event = event.event;
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
if (event.type !== "result") return null;
|
|
76
|
+
if (typeof event.total_cost_usd !== "number") return null;
|
|
77
|
+
|
|
78
|
+
return { source, cost: event.total_cost_usd };
|
|
79
|
+
}
|
|
@@ -0,0 +1,173 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* DiscussTools — discuss-mode tool servers. The lead's surface extends the
|
|
3
|
+
* base set with two discuss-only terminal tools:
|
|
4
|
+
*
|
|
5
|
+
* - `Recess` suspends the session with a resumption trigger.
|
|
6
|
+
* - `Adjourn` ends the discussion with a verdict.
|
|
7
|
+
*
|
|
8
|
+
* `Conclude` is absent — discuss mode ends via Adjourn or Recess.
|
|
9
|
+
*
|
|
10
|
+
* `RequestForComment` is an agent-level coordination tool — available on
|
|
11
|
+
* discuss agents and facilitated agents (not leads). It opens a new
|
|
12
|
+
* Discussion thread for long-horizon coordination on open questions.
|
|
13
|
+
*
|
|
14
|
+
* In discuss mode, each agent Answer routed to the lead is captured as a
|
|
15
|
+
* thread reply delivered via the bridge callback — no explicit reply tool
|
|
16
|
+
* is needed on the lead surface.
|
|
17
|
+
*/
|
|
18
|
+
|
|
19
|
+
import { tool } from "@anthropic-ai/claude-agent-sdk";
|
|
20
|
+
import { z } from "zod";
|
|
21
|
+
|
|
22
|
+
import {
|
|
23
|
+
ADJOURN_DESC,
|
|
24
|
+
baseTools,
|
|
25
|
+
concludeSession,
|
|
26
|
+
orchestrationServer,
|
|
27
|
+
RECESS_DESC,
|
|
28
|
+
requestForCommentTool,
|
|
29
|
+
requireNoPendingAsks,
|
|
30
|
+
requireNoUnprocessedInbox,
|
|
31
|
+
} from "./orchestration-toolkit.js";
|
|
32
|
+
|
|
33
|
+
/** System prompt for discuss-mode agent participants. L0 mechanics only per COALIGNED. */
|
|
34
|
+
export const DISCUSS_AGENT_SYSTEM_PROMPT =
|
|
35
|
+
"You are a participant in a discussion.\n" +
|
|
36
|
+
"Each question arrives as `[ask#N] <name>: <text>` in your inbox.\n" +
|
|
37
|
+
"Quote N as askId on your `Answer` to route the reply correctly.\n" +
|
|
38
|
+
"Your `Answer` is posted to the discussion thread as a separate reply.\n" +
|
|
39
|
+
"If the task already contains a completed response with no new human input after it, `Answer` that no further action is needed.\n" +
|
|
40
|
+
"Do not redo completed work.";
|
|
41
|
+
|
|
42
|
+
const RESUME_TRIGGER_SCHEMA = z.discriminatedUnion("kind", [
|
|
43
|
+
z
|
|
44
|
+
.object({
|
|
45
|
+
kind: z.literal("missing_input"),
|
|
46
|
+
replies: z.number().int().positive(),
|
|
47
|
+
})
|
|
48
|
+
.strict(),
|
|
49
|
+
z
|
|
50
|
+
.object({
|
|
51
|
+
kind: z.literal("escalation_needed"),
|
|
52
|
+
signal: z.string().min(1),
|
|
53
|
+
})
|
|
54
|
+
.strict(),
|
|
55
|
+
z
|
|
56
|
+
.object({
|
|
57
|
+
kind: z.literal("elapsed"),
|
|
58
|
+
elapsed: z.string().min(1),
|
|
59
|
+
})
|
|
60
|
+
.strict(),
|
|
61
|
+
]);
|
|
62
|
+
|
|
63
|
+
/** Discuss-mode lead tool server. */
|
|
64
|
+
export function createDiscussLeadToolServer(ctx) {
|
|
65
|
+
return orchestrationServer([
|
|
66
|
+
...baseTools(ctx, { from: "lead", defaultTo: undefined, broadcast: true }),
|
|
67
|
+
tool(
|
|
68
|
+
"Acknowledge",
|
|
69
|
+
"Post a brief message directly to the discussion thread. Use when responding to a human follow-up or providing a status update while participants are working.",
|
|
70
|
+
{
|
|
71
|
+
message: z.string().describe("Message to post on the thread"),
|
|
72
|
+
},
|
|
73
|
+
async ({ message }) => {
|
|
74
|
+
const seq =
|
|
75
|
+
ctx.emitter?.emit({ kind: "ack", body: message, agent: "lead" }) ??
|
|
76
|
+
-1;
|
|
77
|
+
ctx.replies.push({
|
|
78
|
+
body: message,
|
|
79
|
+
agent: "lead",
|
|
80
|
+
kind: "ack",
|
|
81
|
+
seq,
|
|
82
|
+
...(ctx.discussionId && { thread_id: ctx.discussionId }),
|
|
83
|
+
});
|
|
84
|
+
return { content: [{ type: "text", text: "Posted." }] };
|
|
85
|
+
},
|
|
86
|
+
),
|
|
87
|
+
tool(
|
|
88
|
+
"Recess",
|
|
89
|
+
RECESS_DESC,
|
|
90
|
+
{ reason: z.string(), trigger: RESUME_TRIGGER_SCHEMA },
|
|
91
|
+
createRecessHandler(ctx),
|
|
92
|
+
),
|
|
93
|
+
tool(
|
|
94
|
+
"Adjourn",
|
|
95
|
+
ADJOURN_DESC,
|
|
96
|
+
{
|
|
97
|
+
verdict: z.enum(["adjourned", "failed"]),
|
|
98
|
+
summary: z.string(),
|
|
99
|
+
outcome: z.string().optional(),
|
|
100
|
+
},
|
|
101
|
+
createAdjournHandler(ctx),
|
|
102
|
+
),
|
|
103
|
+
]);
|
|
104
|
+
}
|
|
105
|
+
|
|
106
|
+
const ACKNOWLEDGE_DESC =
|
|
107
|
+
"Acknowledge an Ask before starting work. Posts a visible comment on the thread. Does not discharge the Ask — you still owe an Answer.";
|
|
108
|
+
|
|
109
|
+
/** Discuss-mode agent tool server. */
|
|
110
|
+
export function createDiscussAgentToolServer(ctx, { from }) {
|
|
111
|
+
return orchestrationServer([
|
|
112
|
+
...baseTools(ctx, { from, defaultTo: "lead", broadcast: true }),
|
|
113
|
+
requestForCommentTool(ctx),
|
|
114
|
+
tool(
|
|
115
|
+
"Acknowledge",
|
|
116
|
+
ACKNOWLEDGE_DESC,
|
|
117
|
+
{
|
|
118
|
+
message: z
|
|
119
|
+
.string()
|
|
120
|
+
.describe("Brief acknowledgement to post on the thread"),
|
|
121
|
+
askId: z.number().optional().describe("The ask being acknowledged"),
|
|
122
|
+
},
|
|
123
|
+
async ({ message }) => {
|
|
124
|
+
const seq =
|
|
125
|
+
ctx.emitter?.emit({ kind: "ack", body: message, agent: from }) ?? -1;
|
|
126
|
+
ctx.replies.push({
|
|
127
|
+
body: message,
|
|
128
|
+
agent: from,
|
|
129
|
+
kind: "ack",
|
|
130
|
+
seq,
|
|
131
|
+
...(ctx.discussionId && { thread_id: ctx.discussionId }),
|
|
132
|
+
});
|
|
133
|
+
return { content: [{ type: "text", text: "Acknowledged." }] };
|
|
134
|
+
},
|
|
135
|
+
),
|
|
136
|
+
]);
|
|
137
|
+
}
|
|
138
|
+
|
|
139
|
+
/**
|
|
140
|
+
* Recess handler — ends the run with a structured pause + resumption
|
|
141
|
+
* trigger; cancels any open Asks so askers see a synthetic null answer.
|
|
142
|
+
* `concluded` flips true (same as Adjourn); the `recessed` verdict
|
|
143
|
+
* distinguishes them, and `recessTrigger` carries the resume shape for
|
|
144
|
+
* the bridge.
|
|
145
|
+
*/
|
|
146
|
+
export function createRecessHandler(ctx) {
|
|
147
|
+
return async ({ reason, trigger }) => {
|
|
148
|
+
const guard = requireNoPendingAsks(ctx) ?? requireNoUnprocessedInbox(ctx);
|
|
149
|
+
if (guard) return guard;
|
|
150
|
+
ctx.recessTrigger = trigger;
|
|
151
|
+
concludeSession(ctx, {
|
|
152
|
+
verdict: "recessed",
|
|
153
|
+
summary: reason,
|
|
154
|
+
reason: "session recessed",
|
|
155
|
+
});
|
|
156
|
+
return { content: [{ type: "text", text: "Recess queued." }] };
|
|
157
|
+
};
|
|
158
|
+
}
|
|
159
|
+
|
|
160
|
+
/** Adjourn handler — ends the discussion with a verdict. */
|
|
161
|
+
export function createAdjournHandler(ctx) {
|
|
162
|
+
return async ({ verdict, summary, outcome }) => {
|
|
163
|
+
const guard = requireNoPendingAsks(ctx) ?? requireNoUnprocessedInbox(ctx);
|
|
164
|
+
if (guard) return guard;
|
|
165
|
+
if (outcome !== undefined) ctx.outcome = outcome;
|
|
166
|
+
concludeSession(ctx, {
|
|
167
|
+
verdict,
|
|
168
|
+
summary,
|
|
169
|
+
reason: "session adjourned",
|
|
170
|
+
});
|
|
171
|
+
return { content: [{ type: "text", text: "Session adjourned." }] };
|
|
172
|
+
};
|
|
173
|
+
}
|
package/src/discusser.js
ADDED
|
@@ -0,0 +1,394 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Discusser — async, suspendable orchestration on top of a within-run
|
|
3
|
+
* `OrchestrationLoop`. The lead role uses `DiscussTools` (Adjourn / Recess)
|
|
4
|
+
* instead of the facilitator's Conclude.
|
|
5
|
+
*
|
|
6
|
+
* Discuss mode is a sibling of facilitate mode, not a subset of it. The
|
|
7
|
+
* within-run turn loop is shared via `OrchestrationLoop`, but the lead
|
|
8
|
+
* role, tool set, system prompts, and participant naming all stay
|
|
9
|
+
* mode-local.
|
|
10
|
+
*
|
|
11
|
+
* Each agent Answer routed to the lead is captured as a thread reply
|
|
12
|
+
* delivered via the bridge callback — no explicit reply tool is needed
|
|
13
|
+
* on the lead surface.
|
|
14
|
+
*/
|
|
15
|
+
|
|
16
|
+
import { Writable } from "node:stream";
|
|
17
|
+
import { resolve } from "node:path";
|
|
18
|
+
|
|
19
|
+
import { createAgentRunner } from "./agent-runner.js";
|
|
20
|
+
import { InboxPoller } from "./inbox-poller.js";
|
|
21
|
+
import { ReplyEmitter } from "./reply-emitter.js";
|
|
22
|
+
import { composeSystemPrompt } from "./profile-prompt.js";
|
|
23
|
+
import { SequenceCounter } from "./sequence-counter.js";
|
|
24
|
+
import { createMessageBus } from "./message-bus.js";
|
|
25
|
+
import { createOrchestrationContext } from "./orchestration-toolkit.js";
|
|
26
|
+
import {
|
|
27
|
+
createDiscussLeadToolServer,
|
|
28
|
+
createDiscussAgentToolServer,
|
|
29
|
+
DISCUSS_AGENT_SYSTEM_PROMPT,
|
|
30
|
+
} from "./discuss-tools.js";
|
|
31
|
+
import { OrchestrationLoop } from "./orchestration-loop.js";
|
|
32
|
+
import { AGENT_MODEL, LEAD_MODEL } from "@forwardimpact/libutil/models";
|
|
33
|
+
|
|
34
|
+
/** System prompt for the discuss-mode lead. L0 mechanics only per COALIGNED. */
|
|
35
|
+
export const DISCUSS_SYSTEM_PROMPT =
|
|
36
|
+
"You lead a discussion.\n" +
|
|
37
|
+
"You have no tools to perform work yourself.\n" +
|
|
38
|
+
"Use `RollCall` to list participants.\n" +
|
|
39
|
+
"Use `Ask` to delegate work to the best-suited participant.\n" +
|
|
40
|
+
"Participants are domain experts; state the task, not how to do it.\n" +
|
|
41
|
+
"Each participant's `Answer` is posted to the discussion thread as a separate reply.\n" +
|
|
42
|
+
"`Ask` is async and returns {askIds:[N,…]} immediately.\n" +
|
|
43
|
+
"Answers arrive on your next turn as `[answer#N] <participant>: <text>` in your inbox.\n" +
|
|
44
|
+
"End your turn while Asks are pending. The system resumes you when answers arrive.\n" +
|
|
45
|
+
"Multiple `Ask` calls in one turn run participants in parallel.\n" +
|
|
46
|
+
"Use `Acknowledge` to post a brief message directly to the discussion thread — use it to respond to human follow-ups or give status updates while participants are working.\n" +
|
|
47
|
+
"End the discussion by calling `Adjourn` with a verdict and summary, or `Recess` only to wait on an external reply or duration.";
|
|
48
|
+
|
|
49
|
+
/**
|
|
50
|
+
* Augment a base orchestration context with discuss-mode fields.
|
|
51
|
+
* @param {object} ctx
|
|
52
|
+
* @param {string|null} discussionId
|
|
53
|
+
* @returns {object}
|
|
54
|
+
*/
|
|
55
|
+
export function augmentContextForDiscuss(ctx, discussionId) {
|
|
56
|
+
ctx.discussionId = discussionId;
|
|
57
|
+
ctx.recessTrigger = null;
|
|
58
|
+
ctx.replies = [];
|
|
59
|
+
ctx.rfcs = [];
|
|
60
|
+
ctx.rfcCounter = 0;
|
|
61
|
+
ctx.outcome = null;
|
|
62
|
+
return ctx;
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
const devNull = new Writable({
|
|
66
|
+
write(_chunk, _enc, cb) {
|
|
67
|
+
cb();
|
|
68
|
+
},
|
|
69
|
+
});
|
|
70
|
+
|
|
71
|
+
/**
|
|
72
|
+
* Async orchestrator for the `discuss` mode. Composes an
|
|
73
|
+
* `OrchestrationLoop` for the within-run turns but owns the discussion id,
|
|
74
|
+
* the resumption trigger, and the discuss-augmented terminal summary.
|
|
75
|
+
*/
|
|
76
|
+
export class Discusser {
|
|
77
|
+
/**
|
|
78
|
+
* @param {object} deps
|
|
79
|
+
* @param {OrchestrationLoop} deps.loop
|
|
80
|
+
* @param {object} deps.ctx
|
|
81
|
+
* @param {import("stream").Writable} deps.output
|
|
82
|
+
* @param {object} deps.redactor
|
|
83
|
+
* @param {string|null} [deps.discussionId]
|
|
84
|
+
* @param {SequenceCounter} [deps.counter]
|
|
85
|
+
*/
|
|
86
|
+
constructor({
|
|
87
|
+
loop,
|
|
88
|
+
ctx,
|
|
89
|
+
output,
|
|
90
|
+
discussionId,
|
|
91
|
+
counter,
|
|
92
|
+
redactor,
|
|
93
|
+
inboxPoller,
|
|
94
|
+
}) {
|
|
95
|
+
if (!loop) throw new Error("loop is required");
|
|
96
|
+
if (!ctx) throw new Error("ctx is required");
|
|
97
|
+
if (!output) throw new Error("output is required");
|
|
98
|
+
if (!redactor) throw new Error("redactor is required");
|
|
99
|
+
this.loop = loop;
|
|
100
|
+
this.ctx = ctx;
|
|
101
|
+
this.output = output;
|
|
102
|
+
this.discussionId = discussionId ?? null;
|
|
103
|
+
this.counter = counter ?? new SequenceCounter();
|
|
104
|
+
this.redactor = redactor;
|
|
105
|
+
this.inboxPoller = inboxPoller ?? null;
|
|
106
|
+
}
|
|
107
|
+
|
|
108
|
+
/**
|
|
109
|
+
* Run the discussion. Emits the meta header first (when a discussion_id
|
|
110
|
+
* is set), delegates the within-run loop to `OrchestrationLoop`, then
|
|
111
|
+
* emits the discuss-augmented summary (overrides the loop's earlier
|
|
112
|
+
* summary; trace consumers keep the last summary they see).
|
|
113
|
+
*
|
|
114
|
+
* @param {string} task
|
|
115
|
+
* @returns {Promise<{success: boolean, verdict: string, turns: number, replies: object[], trigger: object|null}>}
|
|
116
|
+
*/
|
|
117
|
+
async run(task) {
|
|
118
|
+
this.#emitMeta();
|
|
119
|
+
|
|
120
|
+
// The loop owns within-run turns. Its emitSummary fires once before
|
|
121
|
+
// run() returns; ours replaces it as the last summary line.
|
|
122
|
+
await this.loop.run(task);
|
|
123
|
+
|
|
124
|
+
const verdict = this.ctx.verdict ?? "failed";
|
|
125
|
+
const success = verdict === "adjourned";
|
|
126
|
+
this.#emitDiscussSummary({
|
|
127
|
+
success,
|
|
128
|
+
verdict,
|
|
129
|
+
turns: this.loop.leadTurns,
|
|
130
|
+
});
|
|
131
|
+
|
|
132
|
+
return {
|
|
133
|
+
success,
|
|
134
|
+
verdict,
|
|
135
|
+
turns: this.loop.leadTurns,
|
|
136
|
+
replies: this.ctx.replies.slice(),
|
|
137
|
+
trigger: this.ctx.recessTrigger ?? null,
|
|
138
|
+
};
|
|
139
|
+
}
|
|
140
|
+
|
|
141
|
+
#emitMeta() {
|
|
142
|
+
if (!this.discussionId) return;
|
|
143
|
+
this.output.write(
|
|
144
|
+
JSON.stringify(
|
|
145
|
+
this.redactor.redactValue({
|
|
146
|
+
source: "orchestrator",
|
|
147
|
+
seq: this.counter.next(),
|
|
148
|
+
event: { type: "meta", discussion_id: this.discussionId },
|
|
149
|
+
}),
|
|
150
|
+
) + "\n",
|
|
151
|
+
);
|
|
152
|
+
}
|
|
153
|
+
|
|
154
|
+
#emitDiscussSummary({ success, verdict, turns }) {
|
|
155
|
+
const event = {
|
|
156
|
+
type: "summary",
|
|
157
|
+
success,
|
|
158
|
+
verdict,
|
|
159
|
+
turns,
|
|
160
|
+
...(this.ctx.summary && { summary: this.ctx.summary }),
|
|
161
|
+
...(this.ctx.outcome && { outcome: this.ctx.outcome }),
|
|
162
|
+
replies: this.ctx.replies,
|
|
163
|
+
...(this.ctx.rfcs?.length && { rfcs: this.ctx.rfcs }),
|
|
164
|
+
...(this.ctx.recessTrigger && { trigger: this.ctx.recessTrigger }),
|
|
165
|
+
...(this.discussionId && { discussion_id: this.discussionId }),
|
|
166
|
+
lastActedSeq: this.inboxPoller?.lastActedSeq ?? -1,
|
|
167
|
+
};
|
|
168
|
+
this.output.write(
|
|
169
|
+
JSON.stringify(
|
|
170
|
+
this.redactor.redactValue({
|
|
171
|
+
source: "orchestrator",
|
|
172
|
+
seq: this.counter.next(),
|
|
173
|
+
event,
|
|
174
|
+
}),
|
|
175
|
+
) + "\n",
|
|
176
|
+
);
|
|
177
|
+
}
|
|
178
|
+
}
|
|
179
|
+
|
|
180
|
+
/**
|
|
181
|
+
* Factory — wires the lead and agent runners with `DiscussTools`, builds
|
|
182
|
+
* the `OrchestrationLoop` (with `leadName: "lead"` and discuss-mode
|
|
183
|
+
* protocol tagging) and the wrapping `Discusser`.
|
|
184
|
+
*
|
|
185
|
+
* Resume semantics: Recess ends the run, cancels any open Asks via
|
|
186
|
+
* `cancelPendingAsks`, and emits a synthetic null answer per cancelled
|
|
187
|
+
* ask so nothing dangles in the trace. The bridge later re-dispatches
|
|
188
|
+
* the workflow against a fresh context; the human reads the trail of
|
|
189
|
+
* events to decide what to re-ask.
|
|
190
|
+
*
|
|
191
|
+
* @param {object} deps
|
|
192
|
+
* @param {string} [deps.leadProfile]
|
|
193
|
+
* @param {string} [deps.leadModel]
|
|
194
|
+
* @param {string} [deps.agentModel]
|
|
195
|
+
* @param {Array<object>} [deps.agentConfigs]
|
|
196
|
+
* @param {string|null} [deps.discussionId]
|
|
197
|
+
* @param {object|null} [deps.resumeContext]
|
|
198
|
+
* @param {function} deps.query
|
|
199
|
+
* @param {import("stream").Writable} deps.output
|
|
200
|
+
* @param {number} [deps.maxTurns]
|
|
201
|
+
* @param {number} [deps.maxLeadTurns]
|
|
202
|
+
* @param {string} [deps.leadCwd]
|
|
203
|
+
* @param {string} [deps.profilesDir]
|
|
204
|
+
* @param {string} [deps.taskAmend]
|
|
205
|
+
* @param {object} deps.redactor
|
|
206
|
+
* @param {string|null} [deps.callbackUrl]
|
|
207
|
+
* @param {string|null} [deps.inboxUrl]
|
|
208
|
+
* @param {string|null} [deps.correlationId]
|
|
209
|
+
* @returns {Discusser}
|
|
210
|
+
*/
|
|
211
|
+
// biome-ignore lint/complexity/noExcessiveCognitiveComplexity: factory wires N runners + resume hydration paths
|
|
212
|
+
export function createDiscusser({
|
|
213
|
+
leadProfile,
|
|
214
|
+
leadModel,
|
|
215
|
+
agentModel,
|
|
216
|
+
agentConfigs,
|
|
217
|
+
discussionId,
|
|
218
|
+
resumeContext,
|
|
219
|
+
query,
|
|
220
|
+
output,
|
|
221
|
+
maxTurns,
|
|
222
|
+
maxLeadTurns,
|
|
223
|
+
leadCwd,
|
|
224
|
+
profilesDir,
|
|
225
|
+
taskAmend,
|
|
226
|
+
redactor,
|
|
227
|
+
callbackUrl,
|
|
228
|
+
inboxUrl,
|
|
229
|
+
correlationId,
|
|
230
|
+
runtime,
|
|
231
|
+
}) {
|
|
232
|
+
if (!redactor) throw new Error("redactor is required");
|
|
233
|
+
if (!runtime) throw new Error("runtime is required");
|
|
234
|
+
const resolvedLeadCwd = resolve(leadCwd ?? ".");
|
|
235
|
+
const resolvedProfilesDir =
|
|
236
|
+
profilesDir ?? resolve(resolvedLeadCwd, ".claude/agents");
|
|
237
|
+
const resolvedConfigs = agentConfigs ?? [];
|
|
238
|
+
|
|
239
|
+
const ctx = augmentContextForDiscuss(
|
|
240
|
+
createOrchestrationContext(),
|
|
241
|
+
discussionId ?? null,
|
|
242
|
+
);
|
|
243
|
+
|
|
244
|
+
// Hydrate resume context — participants, replies, counters. `pendingAsks`
|
|
245
|
+
// is intentionally not restored: Recess cancelled every in-flight Ask
|
|
246
|
+
// with a synthetic null answer, so there's nothing meaningful to carry
|
|
247
|
+
// forward.
|
|
248
|
+
if (resumeContext) {
|
|
249
|
+
if (Array.isArray(resumeContext.participants))
|
|
250
|
+
ctx.participants = resumeContext.participants;
|
|
251
|
+
if (Array.isArray(resumeContext.replies))
|
|
252
|
+
ctx.replies = resumeContext.replies;
|
|
253
|
+
if (typeof resumeContext.askIdCounter === "number")
|
|
254
|
+
ctx.askIdCounter = resumeContext.askIdCounter;
|
|
255
|
+
if (typeof resumeContext.rfcCounter === "number")
|
|
256
|
+
ctx.rfcCounter = resumeContext.rfcCounter;
|
|
257
|
+
}
|
|
258
|
+
|
|
259
|
+
const messageBus = createMessageBus({
|
|
260
|
+
participants: ["lead", ...resolvedConfigs.map((a) => a.name)],
|
|
261
|
+
});
|
|
262
|
+
|
|
263
|
+
const loopCounter = new SequenceCounter();
|
|
264
|
+
const emitter = new ReplyEmitter({
|
|
265
|
+
callbackUrl: callbackUrl ?? null,
|
|
266
|
+
correlationId: correlationId ?? null,
|
|
267
|
+
counter: loopCounter,
|
|
268
|
+
});
|
|
269
|
+
ctx.emitter = emitter;
|
|
270
|
+
|
|
271
|
+
const abortController = new AbortController();
|
|
272
|
+
const inboxPoller = inboxUrl
|
|
273
|
+
? new InboxPoller({
|
|
274
|
+
inboxUrl,
|
|
275
|
+
messageBus,
|
|
276
|
+
leadName: "lead",
|
|
277
|
+
signal: abortController.signal,
|
|
278
|
+
runtime,
|
|
279
|
+
})
|
|
280
|
+
: null;
|
|
281
|
+
|
|
282
|
+
// Intercept answers routed to the lead — each becomes a discussion reply.
|
|
283
|
+
const originalAnswer = messageBus.answer.bind(messageBus);
|
|
284
|
+
messageBus.answer = (from, to, text, askId) => {
|
|
285
|
+
if (to === "lead" && from !== "@orchestrator") {
|
|
286
|
+
const seq = emitter.emit({ kind: "reply", body: text, agent: from });
|
|
287
|
+
ctx.replies.push({
|
|
288
|
+
body: text,
|
|
289
|
+
agent: from,
|
|
290
|
+
kind: "reply",
|
|
291
|
+
seq,
|
|
292
|
+
...(ctx.discussionId && { thread_id: ctx.discussionId }),
|
|
293
|
+
});
|
|
294
|
+
}
|
|
295
|
+
originalAnswer(from, to, text, askId);
|
|
296
|
+
};
|
|
297
|
+
|
|
298
|
+
ctx.messageBus = messageBus;
|
|
299
|
+
if (ctx.participants.length === 0) {
|
|
300
|
+
ctx.participants = [
|
|
301
|
+
{ name: "lead", role: "lead" },
|
|
302
|
+
...resolvedConfigs.map((a) => ({ name: a.name, role: a.role })),
|
|
303
|
+
];
|
|
304
|
+
}
|
|
305
|
+
|
|
306
|
+
let discusser;
|
|
307
|
+
const leadServer = createDiscussLeadToolServer(ctx);
|
|
308
|
+
|
|
309
|
+
const agents = resolvedConfigs.map((config) => {
|
|
310
|
+
const agentServer = createDiscussAgentToolServer(ctx, {
|
|
311
|
+
from: config.name,
|
|
312
|
+
});
|
|
313
|
+
|
|
314
|
+
const runner = createAgentRunner({
|
|
315
|
+
cwd: config.cwd ?? resolvedLeadCwd,
|
|
316
|
+
query,
|
|
317
|
+
output: devNull,
|
|
318
|
+
model: agentModel ?? AGENT_MODEL,
|
|
319
|
+
maxTurns: config.maxTurns ?? 50,
|
|
320
|
+
allowedTools: config.allowedTools,
|
|
321
|
+
onLine: (line) => discusser.loop.emitLine(config.name, line),
|
|
322
|
+
mcpServers: { orchestration: agentServer },
|
|
323
|
+
settingSources: ["project"],
|
|
324
|
+
systemPrompt: composeSystemPrompt({
|
|
325
|
+
role: "agent",
|
|
326
|
+
profile: config.agentProfile,
|
|
327
|
+
profilesDir: resolvedProfilesDir,
|
|
328
|
+
trailer: DISCUSS_AGENT_SYSTEM_PROMPT,
|
|
329
|
+
amend: config.systemPromptAmend,
|
|
330
|
+
runtime,
|
|
331
|
+
}),
|
|
332
|
+
redactor,
|
|
333
|
+
});
|
|
334
|
+
|
|
335
|
+
return { name: config.name, role: config.role, runner };
|
|
336
|
+
});
|
|
337
|
+
|
|
338
|
+
const defaultDisallowed = [
|
|
339
|
+
"Agent",
|
|
340
|
+
"Task",
|
|
341
|
+
"TaskOutput",
|
|
342
|
+
"TaskStop",
|
|
343
|
+
"Bash",
|
|
344
|
+
"Write",
|
|
345
|
+
"Edit",
|
|
346
|
+
];
|
|
347
|
+
const leadRunner = createAgentRunner({
|
|
348
|
+
cwd: resolvedLeadCwd,
|
|
349
|
+
query,
|
|
350
|
+
output: devNull,
|
|
351
|
+
model: leadModel ?? LEAD_MODEL,
|
|
352
|
+
maxTurns: maxTurns ?? 80,
|
|
353
|
+
allowedTools: ["Read", "Glob", "Grep"],
|
|
354
|
+
disallowedTools: defaultDisallowed,
|
|
355
|
+
onLine: (line) => discusser.loop.emitLine("lead", line),
|
|
356
|
+
mcpServers: { orchestration: leadServer },
|
|
357
|
+
settingSources: ["project"],
|
|
358
|
+
systemPrompt: composeSystemPrompt({
|
|
359
|
+
role: "lead",
|
|
360
|
+
profile: leadProfile,
|
|
361
|
+
profilesDir: resolvedProfilesDir,
|
|
362
|
+
trailer: DISCUSS_SYSTEM_PROMPT,
|
|
363
|
+
runtime,
|
|
364
|
+
}),
|
|
365
|
+
redactor,
|
|
366
|
+
});
|
|
367
|
+
|
|
368
|
+
const loop = new OrchestrationLoop({
|
|
369
|
+
leadRunner,
|
|
370
|
+
agents,
|
|
371
|
+
messageBus,
|
|
372
|
+
output,
|
|
373
|
+
leadName: "lead",
|
|
374
|
+
mode: "discussion",
|
|
375
|
+
maxLeadTurns: maxLeadTurns ?? undefined,
|
|
376
|
+
ctx,
|
|
377
|
+
taskAmend,
|
|
378
|
+
redactor,
|
|
379
|
+
inboxPoller,
|
|
380
|
+
abortController,
|
|
381
|
+
});
|
|
382
|
+
loop.counter = loopCounter;
|
|
383
|
+
|
|
384
|
+
discusser = new Discusser({
|
|
385
|
+
loop,
|
|
386
|
+
ctx,
|
|
387
|
+
output,
|
|
388
|
+
discussionId: discussionId ?? null,
|
|
389
|
+
redactor,
|
|
390
|
+
counter: loopCounter,
|
|
391
|
+
inboxPoller,
|
|
392
|
+
});
|
|
393
|
+
return discusser;
|
|
394
|
+
}
|