@forwardimpact/libharness 1.3.0 → 1.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bin/fit-harness.js +18 -0
- package/package.json +1 -1
- package/src/advisor.js +218 -0
- package/src/agent-runner.js +6 -0
- package/src/benchmark/report.js +9 -1
- package/src/commands/advisor-flags.js +28 -0
- package/src/commands/benchmark-definition.js +5 -0
- package/src/commands/benchmark-report.js +12 -2
- package/src/commands/discuss.js +4 -0
- package/src/commands/facilitate.js +4 -0
- package/src/commands/run.js +162 -67
- package/src/commands/supervise.js +4 -0
- package/src/discuss-tools.js +2 -1
- package/src/discusser.js +65 -10
- package/src/facilitator.js +64 -9
- package/src/index.js +9 -0
- package/src/orchestration-toolkit.js +65 -7
- package/src/supervisor.js +72 -10
- package/src/transcript-recorder.js +94 -0
package/src/supervisor.js
CHANGED
|
@@ -21,11 +21,18 @@ import { createAgentRunner } from "./agent-runner.js";
|
|
|
21
21
|
import { composeSystemPrompt } from "./profile-prompt.js";
|
|
22
22
|
import { createMessageBus } from "./message-bus.js";
|
|
23
23
|
import {
|
|
24
|
+
advisorTool,
|
|
24
25
|
createOrchestrationContext,
|
|
25
26
|
createSupervisedAgentToolServer,
|
|
26
27
|
createSupervisorToolServer,
|
|
27
28
|
} from "./orchestration-toolkit.js";
|
|
28
29
|
import { OrchestrationLoop } from "./orchestration-loop.js";
|
|
30
|
+
import {
|
|
31
|
+
createAdvisor,
|
|
32
|
+
createAdvisorBudget,
|
|
33
|
+
withAdvisorGuidance,
|
|
34
|
+
} from "./advisor.js";
|
|
35
|
+
import { createTranscriptRecorder } from "./transcript-recorder.js";
|
|
29
36
|
|
|
30
37
|
/** System prompt for the supervisor lead. L0 mechanics only per COALIGNED. */
|
|
31
38
|
export const SUPERVISOR_SYSTEM_PROMPT =
|
|
@@ -59,6 +66,7 @@ export class Supervisor extends OrchestrationLoop {
|
|
|
59
66
|
* @param {object} deps.ctx
|
|
60
67
|
* @param {object} deps.redactor
|
|
61
68
|
* @param {string} [deps.taskAmend]
|
|
69
|
+
* @param {AbortController} [deps.abortController]
|
|
62
70
|
*/
|
|
63
71
|
constructor({
|
|
64
72
|
supervisorRunner,
|
|
@@ -68,6 +76,7 @@ export class Supervisor extends OrchestrationLoop {
|
|
|
68
76
|
ctx,
|
|
69
77
|
taskAmend,
|
|
70
78
|
redactor,
|
|
79
|
+
abortController,
|
|
71
80
|
}) {
|
|
72
81
|
if (!agentRunner) throw new Error("agentRunner is required");
|
|
73
82
|
if (!supervisorRunner) throw new Error("supervisorRunner is required");
|
|
@@ -82,6 +91,7 @@ export class Supervisor extends OrchestrationLoop {
|
|
|
82
91
|
ctx,
|
|
83
92
|
taskAmend,
|
|
84
93
|
redactor,
|
|
94
|
+
abortController,
|
|
85
95
|
});
|
|
86
96
|
}
|
|
87
97
|
|
|
@@ -125,6 +135,8 @@ const devNull = new Writable({
|
|
|
125
135
|
* @param {string} [deps.profilesDir]
|
|
126
136
|
* @param {string} [deps.taskAmend]
|
|
127
137
|
* @param {Record<string, object>} [deps.agentMcpServers]
|
|
138
|
+
* @param {string} [deps.advisorModel] - Claude model for advisor consults; absent means no Advisor tool is offered.
|
|
139
|
+
* @param {number} [deps.advisorMaxUses] - Session-wide consult budget (default 3).
|
|
128
140
|
* @returns {Supervisor}
|
|
129
141
|
*/
|
|
130
142
|
export function createSupervisor({
|
|
@@ -147,6 +159,8 @@ export function createSupervisor({
|
|
|
147
159
|
agentMcpServers,
|
|
148
160
|
redactor,
|
|
149
161
|
runtime,
|
|
162
|
+
advisorModel,
|
|
163
|
+
advisorMaxUses,
|
|
150
164
|
}) {
|
|
151
165
|
if (!redactor) throw new Error("redactor is required");
|
|
152
166
|
if (!runtime) throw new Error("runtime is required");
|
|
@@ -165,10 +179,58 @@ export function createSupervisor({
|
|
|
165
179
|
|
|
166
180
|
let supervisor;
|
|
167
181
|
const perRunBudget = maxTurns ?? 200;
|
|
182
|
+
const abortController = new AbortController();
|
|
183
|
+
|
|
184
|
+
// Advisor wiring — everything below is gated on advisorModel being set;
|
|
185
|
+
// with it unset the composed prompt and tool surface are byte-identical
|
|
186
|
+
// to today's.
|
|
187
|
+
const budget = advisorModel ? createAdvisorBudget(advisorMaxUses ?? 3) : null;
|
|
188
|
+
const agentSystemPrompt = composeSystemPrompt({
|
|
189
|
+
role: "agent",
|
|
190
|
+
profile: agentProfile,
|
|
191
|
+
profilesDir: resolvedProfilesDir,
|
|
192
|
+
trailer: AGENT_SYSTEM_PROMPT,
|
|
193
|
+
amend: withAdvisorGuidance(agentSystemPromptAmend, budget),
|
|
194
|
+
runtime,
|
|
195
|
+
});
|
|
168
196
|
|
|
169
|
-
|
|
197
|
+
let recorder = null;
|
|
198
|
+
let extraTools;
|
|
199
|
+
if (advisorModel) {
|
|
200
|
+
recorder = createTranscriptRecorder({
|
|
201
|
+
systemPrompt: agentSystemPrompt,
|
|
202
|
+
redactor,
|
|
203
|
+
});
|
|
204
|
+
// Late-bound through the `let supervisor` closure — the instance does
|
|
205
|
+
// not exist yet when the advisor and tool are built.
|
|
206
|
+
const advisor = createAdvisor({
|
|
207
|
+
model: advisorModel,
|
|
208
|
+
cwd: agentCwd,
|
|
209
|
+
query,
|
|
210
|
+
recorder,
|
|
211
|
+
redactor,
|
|
212
|
+
runtime,
|
|
213
|
+
onLine: (line) => supervisor.emitLine("advisor", line),
|
|
214
|
+
});
|
|
215
|
+
abortController.signal.addEventListener("abort", () => advisor.abort());
|
|
216
|
+
extraTools = [
|
|
217
|
+
advisorTool({
|
|
218
|
+
from: "agent",
|
|
219
|
+
consult: (q) => advisor.consult(q),
|
|
220
|
+
emit: (e) => supervisor.emitOrchestratorEvent(e),
|
|
221
|
+
budget,
|
|
222
|
+
model: advisorModel,
|
|
223
|
+
}),
|
|
224
|
+
];
|
|
225
|
+
}
|
|
226
|
+
|
|
227
|
+
const agentServer = createSupervisedAgentToolServer(
|
|
228
|
+
ctx,
|
|
229
|
+
extraTools ? { extraTools } : {},
|
|
230
|
+
);
|
|
170
231
|
const supervisorServer = createSupervisorToolServer(ctx);
|
|
171
232
|
|
|
233
|
+
const emitAgentLine = (line) => supervisor.emitLine("agent", line);
|
|
172
234
|
const agentRunner = createAgentRunner({
|
|
173
235
|
cwd: agentCwd,
|
|
174
236
|
query,
|
|
@@ -176,16 +238,15 @@ export function createSupervisor({
|
|
|
176
238
|
model: agentModel ?? model,
|
|
177
239
|
maxTurns: perRunBudget,
|
|
178
240
|
allowedTools,
|
|
179
|
-
onLine:
|
|
241
|
+
onLine: recorder
|
|
242
|
+
? (line) => {
|
|
243
|
+
emitAgentLine(line);
|
|
244
|
+
recorder.recordMessage(line);
|
|
245
|
+
}
|
|
246
|
+
: emitAgentLine,
|
|
247
|
+
...(recorder && { onPrompt: (text) => recorder.recordPrompt(text) }),
|
|
180
248
|
settingSources: ["project"],
|
|
181
|
-
systemPrompt:
|
|
182
|
-
role: "agent",
|
|
183
|
-
profile: agentProfile,
|
|
184
|
-
profilesDir: resolvedProfilesDir,
|
|
185
|
-
trailer: AGENT_SYSTEM_PROMPT,
|
|
186
|
-
amend: agentSystemPromptAmend,
|
|
187
|
-
runtime,
|
|
188
|
-
}),
|
|
249
|
+
systemPrompt: agentSystemPrompt,
|
|
189
250
|
mcpServers: { orchestration: agentServer, ...agentMcpServers },
|
|
190
251
|
redactor,
|
|
191
252
|
});
|
|
@@ -231,6 +292,7 @@ export function createSupervisor({
|
|
|
231
292
|
ctx,
|
|
232
293
|
taskAmend,
|
|
233
294
|
redactor,
|
|
295
|
+
abortController,
|
|
234
296
|
});
|
|
235
297
|
return supervisor;
|
|
236
298
|
}
|
|
@@ -0,0 +1,94 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* TranscriptRecorder — per-participant in-memory record of the composed
|
|
3
|
+
* system prompt, delivered prompts, and session messages, rendered into the
|
|
4
|
+
* context text an advisor consult forwards. Constructed only when a session
|
|
5
|
+
* runs with an advisor model; the harness otherwise keeps no per-participant
|
|
6
|
+
* record (session lines go straight to the trace stream).
|
|
7
|
+
*
|
|
8
|
+
* Redaction split: the message tap arrives post-redaction (fed from
|
|
9
|
+
* `AgentRunner.#recordLine`), but the seeded system prompt and the prompt
|
|
10
|
+
* tap are raw, so the recorder redacts those itself via the injected
|
|
11
|
+
* redactor.
|
|
12
|
+
*/
|
|
13
|
+
|
|
14
|
+
/**
|
|
15
|
+
* Normalize whatever the harness composed as a system prompt into plain
|
|
16
|
+
* text. In practice always a `{type:"preset", preset:"claude_code", append}`
|
|
17
|
+
* object (every recorded participant is an agent; leads are spec-excluded);
|
|
18
|
+
* a plain string is tolerated and `undefined` accepted.
|
|
19
|
+
* @param {string|{type: string, preset?: string, append?: string}|undefined} systemPrompt
|
|
20
|
+
* @returns {string|undefined}
|
|
21
|
+
*/
|
|
22
|
+
function normalizeSystemPrompt(systemPrompt) {
|
|
23
|
+
if (!systemPrompt) return undefined;
|
|
24
|
+
if (typeof systemPrompt === "string") return systemPrompt;
|
|
25
|
+
if (systemPrompt.append) {
|
|
26
|
+
return `(claude_code preset)\n${systemPrompt.append}`;
|
|
27
|
+
}
|
|
28
|
+
return undefined;
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
/** Wrap content in a tagged section, each tag on its own line. */
|
|
32
|
+
function wrapSection(tag, content) {
|
|
33
|
+
return `<${tag}>\n${content}\n</${tag}>`;
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
/**
|
|
37
|
+
* Create a per-participant transcript recorder.
|
|
38
|
+
*
|
|
39
|
+
* @param {object} deps
|
|
40
|
+
* @param {string|object} [deps.systemPrompt] - The system prompt the harness
|
|
41
|
+
* composed for the participant, as passed to its runner. Raw — redacted at
|
|
42
|
+
* construction.
|
|
43
|
+
* @param {import("./redaction.js").Redactor} deps.redactor
|
|
44
|
+
* @returns {{recordPrompt: (text: string) => void, recordMessage: (line: string) => void, render: () => string}}
|
|
45
|
+
*/
|
|
46
|
+
export function createTranscriptRecorder({ systemPrompt, redactor }) {
|
|
47
|
+
if (!redactor) throw new Error("redactor is required");
|
|
48
|
+
const normalized = normalizeSystemPrompt(systemPrompt);
|
|
49
|
+
const seededPrompt = normalized
|
|
50
|
+
? redactor.redactValue(normalized)
|
|
51
|
+
: undefined;
|
|
52
|
+
/** @type {string[]} */
|
|
53
|
+
const prompts = [];
|
|
54
|
+
/** @type {string[]} */
|
|
55
|
+
const messages = [];
|
|
56
|
+
|
|
57
|
+
return {
|
|
58
|
+
/**
|
|
59
|
+
* Record a delivered (amend-applied) prompt. Raw — redacted here.
|
|
60
|
+
* @param {string} text
|
|
61
|
+
*/
|
|
62
|
+
recordPrompt(text) {
|
|
63
|
+
prompts.push(redactor.redactValue(text));
|
|
64
|
+
},
|
|
65
|
+
/**
|
|
66
|
+
* Record one NDJSON session line as-is (it arrives already redacted
|
|
67
|
+
* from the runner's line path).
|
|
68
|
+
* @param {string} line
|
|
69
|
+
*/
|
|
70
|
+
recordMessage(line) {
|
|
71
|
+
messages.push(line);
|
|
72
|
+
},
|
|
73
|
+
/**
|
|
74
|
+
* Render the record as the advisor's context text: three tagged
|
|
75
|
+
* sections joined by blank lines, each present only when non-empty.
|
|
76
|
+
* NDJSON lines are verbatim — the forwarded context is uncurated by
|
|
77
|
+
* construction (context-size curation is spec-excluded).
|
|
78
|
+
* @returns {string}
|
|
79
|
+
*/
|
|
80
|
+
render() {
|
|
81
|
+
const sections = [];
|
|
82
|
+
if (seededPrompt) {
|
|
83
|
+
sections.push(wrapSection("caller_system_prompt", seededPrompt));
|
|
84
|
+
}
|
|
85
|
+
if (prompts.length > 0) {
|
|
86
|
+
sections.push(wrapSection("caller_prompts", prompts.join("\n\n")));
|
|
87
|
+
}
|
|
88
|
+
if (messages.length > 0) {
|
|
89
|
+
sections.push(wrapSection("caller_transcript", messages.join("\n")));
|
|
90
|
+
}
|
|
91
|
+
return sections.join("\n\n");
|
|
92
|
+
},
|
|
93
|
+
};
|
|
94
|
+
}
|