@forwardimpact/libharness 1.3.0 → 2.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bin/fit-harness.js +18 -0
- package/bin/fit-trace.js +10 -0
- package/package.json +1 -1
- package/src/advisor.js +218 -0
- package/src/agent-runner.js +6 -0
- package/src/benchmark/grade.js +222 -0
- package/src/benchmark/hidden-tests.js +180 -0
- package/src/benchmark/invariants.js +9 -10
- package/src/benchmark/judge.js +9 -6
- package/src/benchmark/report.js +150 -56
- package/src/benchmark/result.js +43 -8
- package/src/benchmark/runner.js +99 -109
- package/src/benchmark/task-family.js +139 -26
- package/src/benchmark/trace-split.js +73 -0
- package/src/benchmark/workdir.js +6 -1
- package/src/commands/advisor-flags.js +28 -0
- package/src/commands/assert.js +72 -0
- package/src/commands/benchmark-definition.js +11 -6
- package/src/commands/benchmark-grade.js +82 -0
- package/src/commands/benchmark-report.js +12 -2
- package/src/commands/discuss.js +4 -0
- package/src/commands/facilitate.js +4 -0
- package/src/commands/run.js +162 -67
- package/src/commands/supervise.js +4 -0
- package/src/discuss-tools.js +2 -1
- package/src/discusser.js +65 -10
- package/src/facilitator.js +64 -9
- package/src/index.js +9 -0
- package/src/orchestration-toolkit.js +65 -7
- package/src/supervisor.js +72 -10
- package/src/transcript-recorder.js +94 -0
- package/src/commands/benchmark-invariants.js +0 -73
package/src/discusser.js
CHANGED
|
@@ -22,7 +22,16 @@ import { ReplyEmitter } from "./reply-emitter.js";
|
|
|
22
22
|
import { composeSystemPrompt } from "./profile-prompt.js";
|
|
23
23
|
import { SequenceCounter } from "./sequence-counter.js";
|
|
24
24
|
import { createMessageBus } from "./message-bus.js";
|
|
25
|
-
import {
|
|
25
|
+
import {
|
|
26
|
+
advisorTool,
|
|
27
|
+
createOrchestrationContext,
|
|
28
|
+
} from "./orchestration-toolkit.js";
|
|
29
|
+
import {
|
|
30
|
+
createAdvisor,
|
|
31
|
+
createAdvisorBudget,
|
|
32
|
+
withAdvisorGuidance,
|
|
33
|
+
} from "./advisor.js";
|
|
34
|
+
import { createTranscriptRecorder } from "./transcript-recorder.js";
|
|
26
35
|
import {
|
|
27
36
|
createDiscussLeadToolServer,
|
|
28
37
|
createDiscussAgentToolServer,
|
|
@@ -206,6 +215,8 @@ export class Discusser {
|
|
|
206
215
|
* @param {string|null} [deps.callbackUrl]
|
|
207
216
|
* @param {string|null} [deps.inboxUrl]
|
|
208
217
|
* @param {string|null} [deps.correlationId]
|
|
218
|
+
* @param {string} [deps.advisorModel] - Claude model for advisor consults; absent means no Advisor tool is offered.
|
|
219
|
+
* @param {number} [deps.advisorMaxUses] - Session-wide consult budget shared by all agent participants (default 3).
|
|
209
220
|
* @returns {Discusser}
|
|
210
221
|
*/
|
|
211
222
|
// biome-ignore lint/complexity/noExcessiveCognitiveComplexity: factory wires N runners + resume hydration paths
|
|
@@ -228,6 +239,8 @@ export function createDiscusser({
|
|
|
228
239
|
inboxUrl,
|
|
229
240
|
correlationId,
|
|
230
241
|
runtime,
|
|
242
|
+
advisorModel,
|
|
243
|
+
advisorMaxUses,
|
|
231
244
|
}) {
|
|
232
245
|
if (!redactor) throw new Error("redactor is required");
|
|
233
246
|
if (!runtime) throw new Error("runtime is required");
|
|
@@ -306,11 +319,54 @@ export function createDiscusser({
|
|
|
306
319
|
let discusser;
|
|
307
320
|
const leadServer = createDiscussLeadToolServer(ctx);
|
|
308
321
|
|
|
322
|
+
// One budget per session, shared by every agent's Advisor handler.
|
|
323
|
+
const budget = advisorModel ? createAdvisorBudget(advisorMaxUses ?? 3) : null;
|
|
324
|
+
|
|
309
325
|
const agents = resolvedConfigs.map((config) => {
|
|
326
|
+
// Everything advisor-shaped is gated on advisorModel; with it unset the
|
|
327
|
+
// composed prompt and tool surface are byte-identical to today's.
|
|
328
|
+
const systemPrompt = composeSystemPrompt({
|
|
329
|
+
role: "agent",
|
|
330
|
+
profile: config.agentProfile,
|
|
331
|
+
profilesDir: resolvedProfilesDir,
|
|
332
|
+
trailer: DISCUSS_AGENT_SYSTEM_PROMPT,
|
|
333
|
+
amend: withAdvisorGuidance(config.systemPromptAmend, budget),
|
|
334
|
+
runtime,
|
|
335
|
+
});
|
|
336
|
+
|
|
337
|
+
let recorder = null;
|
|
338
|
+
let extraTools;
|
|
339
|
+
if (advisorModel) {
|
|
340
|
+
recorder = createTranscriptRecorder({ systemPrompt, redactor });
|
|
341
|
+
// Late-bound through the `let discusser` closure — the instance does
|
|
342
|
+
// not exist yet when the advisor and tool are built.
|
|
343
|
+
const advisor = createAdvisor({
|
|
344
|
+
model: advisorModel,
|
|
345
|
+
cwd: config.cwd ?? resolvedLeadCwd,
|
|
346
|
+
query,
|
|
347
|
+
recorder,
|
|
348
|
+
redactor,
|
|
349
|
+
runtime,
|
|
350
|
+
onLine: (line) => discusser.loop.emitLine("advisor", line),
|
|
351
|
+
});
|
|
352
|
+
abortController.signal.addEventListener("abort", () => advisor.abort());
|
|
353
|
+
extraTools = [
|
|
354
|
+
advisorTool({
|
|
355
|
+
from: config.name,
|
|
356
|
+
consult: (q) => advisor.consult(q),
|
|
357
|
+
emit: (e) => discusser.loop.emitOrchestratorEvent(e),
|
|
358
|
+
budget,
|
|
359
|
+
model: advisorModel,
|
|
360
|
+
}),
|
|
361
|
+
];
|
|
362
|
+
}
|
|
363
|
+
|
|
310
364
|
const agentServer = createDiscussAgentToolServer(ctx, {
|
|
311
365
|
from: config.name,
|
|
366
|
+
...(extraTools && { extraTools }),
|
|
312
367
|
});
|
|
313
368
|
|
|
369
|
+
const emitAgentLine = (line) => discusser.loop.emitLine(config.name, line);
|
|
314
370
|
const runner = createAgentRunner({
|
|
315
371
|
cwd: config.cwd ?? resolvedLeadCwd,
|
|
316
372
|
query,
|
|
@@ -318,17 +374,16 @@ export function createDiscusser({
|
|
|
318
374
|
model: agentModel ?? AGENT_MODEL,
|
|
319
375
|
maxTurns: config.maxTurns ?? 50,
|
|
320
376
|
allowedTools: config.allowedTools,
|
|
321
|
-
onLine:
|
|
377
|
+
onLine: recorder
|
|
378
|
+
? (line) => {
|
|
379
|
+
emitAgentLine(line);
|
|
380
|
+
recorder.recordMessage(line);
|
|
381
|
+
}
|
|
382
|
+
: emitAgentLine,
|
|
383
|
+
...(recorder && { onPrompt: (text) => recorder.recordPrompt(text) }),
|
|
322
384
|
mcpServers: { orchestration: agentServer },
|
|
323
385
|
settingSources: ["project"],
|
|
324
|
-
systemPrompt
|
|
325
|
-
role: "agent",
|
|
326
|
-
profile: config.agentProfile,
|
|
327
|
-
profilesDir: resolvedProfilesDir,
|
|
328
|
-
trailer: DISCUSS_AGENT_SYSTEM_PROMPT,
|
|
329
|
-
amend: config.systemPromptAmend,
|
|
330
|
-
runtime,
|
|
331
|
-
}),
|
|
386
|
+
systemPrompt,
|
|
332
387
|
redactor,
|
|
333
388
|
});
|
|
334
389
|
|
package/src/facilitator.js
CHANGED
|
@@ -12,11 +12,18 @@ import { createAgentRunner } from "./agent-runner.js";
|
|
|
12
12
|
import { composeSystemPrompt } from "./profile-prompt.js";
|
|
13
13
|
import { createMessageBus } from "./message-bus.js";
|
|
14
14
|
import {
|
|
15
|
+
advisorTool,
|
|
15
16
|
createOrchestrationContext,
|
|
16
17
|
createFacilitatorToolServer,
|
|
17
18
|
createFacilitatedAgentToolServer,
|
|
18
19
|
} from "./orchestration-toolkit.js";
|
|
19
20
|
import { OrchestrationLoop } from "./orchestration-loop.js";
|
|
21
|
+
import {
|
|
22
|
+
createAdvisor,
|
|
23
|
+
createAdvisorBudget,
|
|
24
|
+
withAdvisorGuidance,
|
|
25
|
+
} from "./advisor.js";
|
|
26
|
+
import { createTranscriptRecorder } from "./transcript-recorder.js";
|
|
20
27
|
|
|
21
28
|
/** System prompt for the facilitator lead. L0 mechanics only per COALIGNED. */
|
|
22
29
|
export const FACILITATOR_SYSTEM_PROMPT =
|
|
@@ -92,6 +99,8 @@ const devNull = new Writable({
|
|
|
92
99
|
* @param {string} [deps.facilitatorProfile]
|
|
93
100
|
* @param {string} [deps.profilesDir]
|
|
94
101
|
* @param {string} [deps.taskAmend]
|
|
102
|
+
* @param {string} [deps.advisorModel] - Claude model for advisor consults; absent means no Advisor tool is offered.
|
|
103
|
+
* @param {number} [deps.advisorMaxUses] - Session-wide consult budget shared by all agent participants (default 3).
|
|
95
104
|
* @returns {Facilitator}
|
|
96
105
|
*/
|
|
97
106
|
export function createFacilitator({
|
|
@@ -110,6 +119,8 @@ export function createFacilitator({
|
|
|
110
119
|
taskAmend,
|
|
111
120
|
redactor,
|
|
112
121
|
runtime,
|
|
122
|
+
advisorModel,
|
|
123
|
+
advisorMaxUses,
|
|
113
124
|
}) {
|
|
114
125
|
if (!redactor) throw new Error("redactor is required");
|
|
115
126
|
if (!runtime) throw new Error("runtime is required");
|
|
@@ -129,11 +140,55 @@ export function createFacilitator({
|
|
|
129
140
|
|
|
130
141
|
const facilitatorServer = createFacilitatorToolServer(ctx);
|
|
131
142
|
|
|
143
|
+
const abortController = new AbortController();
|
|
144
|
+
// One budget per session, shared by every agent's Advisor handler.
|
|
145
|
+
const budget = advisorModel ? createAdvisorBudget(advisorMaxUses ?? 3) : null;
|
|
146
|
+
|
|
132
147
|
const agents = agentConfigs.map((config) => {
|
|
148
|
+
// Everything advisor-shaped is gated on advisorModel; with it unset the
|
|
149
|
+
// composed prompt and tool surface are byte-identical to today's.
|
|
150
|
+
const systemPrompt = composeSystemPrompt({
|
|
151
|
+
role: "agent",
|
|
152
|
+
profile: config.agentProfile,
|
|
153
|
+
profilesDir: resolvedProfilesDir,
|
|
154
|
+
trailer: FACILITATED_AGENT_SYSTEM_PROMPT,
|
|
155
|
+
amend: withAdvisorGuidance(config.systemPromptAmend, budget),
|
|
156
|
+
runtime,
|
|
157
|
+
});
|
|
158
|
+
|
|
159
|
+
let recorder = null;
|
|
160
|
+
let extraTools;
|
|
161
|
+
if (advisorModel) {
|
|
162
|
+
recorder = createTranscriptRecorder({ systemPrompt, redactor });
|
|
163
|
+
// Late-bound through the `let facilitator` closure — the instance
|
|
164
|
+
// does not exist yet when the advisor and tool are built.
|
|
165
|
+
const advisor = createAdvisor({
|
|
166
|
+
model: advisorModel,
|
|
167
|
+
cwd: config.cwd ?? facilitatorCwd,
|
|
168
|
+
query,
|
|
169
|
+
recorder,
|
|
170
|
+
redactor,
|
|
171
|
+
runtime,
|
|
172
|
+
onLine: (line) => facilitator.emitLine("advisor", line),
|
|
173
|
+
});
|
|
174
|
+
abortController.signal.addEventListener("abort", () => advisor.abort());
|
|
175
|
+
extraTools = [
|
|
176
|
+
advisorTool({
|
|
177
|
+
from: config.name,
|
|
178
|
+
consult: (q) => advisor.consult(q),
|
|
179
|
+
emit: (e) => facilitator.emitOrchestratorEvent(e),
|
|
180
|
+
budget,
|
|
181
|
+
model: advisorModel,
|
|
182
|
+
}),
|
|
183
|
+
];
|
|
184
|
+
}
|
|
185
|
+
|
|
133
186
|
const agentServer = createFacilitatedAgentToolServer(ctx, {
|
|
134
187
|
from: config.name,
|
|
188
|
+
...(extraTools && { extraTools }),
|
|
135
189
|
});
|
|
136
190
|
|
|
191
|
+
const emitAgentLine = (line) => facilitator.emitLine(config.name, line);
|
|
137
192
|
const runner = createAgentRunner({
|
|
138
193
|
cwd: config.cwd ?? facilitatorCwd,
|
|
139
194
|
query,
|
|
@@ -141,17 +196,16 @@ export function createFacilitator({
|
|
|
141
196
|
model: agentModel ?? model,
|
|
142
197
|
maxTurns: config.maxTurns ?? 50,
|
|
143
198
|
allowedTools: config.allowedTools,
|
|
144
|
-
onLine:
|
|
199
|
+
onLine: recorder
|
|
200
|
+
? (line) => {
|
|
201
|
+
emitAgentLine(line);
|
|
202
|
+
recorder.recordMessage(line);
|
|
203
|
+
}
|
|
204
|
+
: emitAgentLine,
|
|
205
|
+
...(recorder && { onPrompt: (text) => recorder.recordPrompt(text) }),
|
|
145
206
|
mcpServers: { orchestration: agentServer },
|
|
146
207
|
settingSources: ["project"],
|
|
147
|
-
systemPrompt
|
|
148
|
-
role: "agent",
|
|
149
|
-
profile: config.agentProfile,
|
|
150
|
-
profilesDir: resolvedProfilesDir,
|
|
151
|
-
trailer: FACILITATED_AGENT_SYSTEM_PROMPT,
|
|
152
|
-
amend: config.systemPromptAmend,
|
|
153
|
-
runtime,
|
|
154
|
-
}),
|
|
208
|
+
systemPrompt,
|
|
155
209
|
redactor,
|
|
156
210
|
});
|
|
157
211
|
|
|
@@ -200,6 +254,7 @@ export function createFacilitator({
|
|
|
200
254
|
ctx,
|
|
201
255
|
taskAmend,
|
|
202
256
|
redactor,
|
|
257
|
+
abortController,
|
|
203
258
|
});
|
|
204
259
|
return facilitator;
|
|
205
260
|
}
|
package/src/index.js
CHANGED
|
@@ -26,6 +26,7 @@ export {
|
|
|
26
26
|
export { TeeWriter, createTeeWriter } from "./tee-writer.js";
|
|
27
27
|
export { SequenceCounter, createSequenceCounter } from "./sequence-counter.js";
|
|
28
28
|
export {
|
|
29
|
+
advisorTool,
|
|
29
30
|
createOrchestrationContext,
|
|
30
31
|
createRequestForCommentHandler,
|
|
31
32
|
createSupervisorToolServer,
|
|
@@ -34,6 +35,14 @@ export {
|
|
|
34
35
|
createFacilitatedAgentToolServer,
|
|
35
36
|
createJudgeToolServer,
|
|
36
37
|
} from "./orchestration-toolkit.js";
|
|
38
|
+
export {
|
|
39
|
+
ADVISOR_SYSTEM_PROMPT,
|
|
40
|
+
advisorGuidance,
|
|
41
|
+
createAdvisor,
|
|
42
|
+
createAdvisorBudget,
|
|
43
|
+
DEFAULT_CONSULT_TIMEOUT_MS,
|
|
44
|
+
} from "./advisor.js";
|
|
45
|
+
export { createTranscriptRecorder } from "./transcript-recorder.js";
|
|
37
46
|
export { MessageBus, createMessageBus } from "./message-bus.js";
|
|
38
47
|
export { OrchestrationLoop } from "./orchestration-loop.js";
|
|
39
48
|
export {
|
|
@@ -337,6 +337,59 @@ function concludeTool(ctx) {
|
|
|
337
337
|
);
|
|
338
338
|
}
|
|
339
339
|
|
|
340
|
+
const ADVISOR_DESC =
|
|
341
|
+
"Consult a stronger model on one focused question. Your full session context (system prompt, prompts, transcript so far) is forwarded automatically — you cannot restrict it. The advice returns in the tool result. The consult budget is shared session-wide across all participants.";
|
|
342
|
+
|
|
343
|
+
/**
|
|
344
|
+
* Build the `Advisor` consult tool for one caller. Mode-agnostic: loop
|
|
345
|
+
* modes pass it into the agent tool-server factories via `extraTools`;
|
|
346
|
+
* run mode gives it a dedicated server. No orchestration-context
|
|
347
|
+
* dependency — the budget object and emit callback are injected.
|
|
348
|
+
*
|
|
349
|
+
* @param {object} deps
|
|
350
|
+
* @param {string} deps.from - Caller's canonical name (event attribution).
|
|
351
|
+
* @param {(question: string) => Promise<{advice?: string, unavailable?: boolean, reason?: string, durationMs: number}>} deps.consult
|
|
352
|
+
* @param {(event: object) => void} deps.emit - Orchestrator-event emitter for the `advisor_consult` event.
|
|
353
|
+
* @param {{maxUses: number, used: number}} deps.budget - Session-wide budget shared by every caller's handler.
|
|
354
|
+
* @param {string} deps.model - Advisor model id, carried on the consult event.
|
|
355
|
+
*/
|
|
356
|
+
export function advisorTool({ from, consult, emit, budget, model }) {
|
|
357
|
+
return tool(
|
|
358
|
+
"Advisor",
|
|
359
|
+
ADVISOR_DESC,
|
|
360
|
+
{ question: z.string() },
|
|
361
|
+
async ({ question }) => {
|
|
362
|
+
if (budget.used >= budget.maxUses) {
|
|
363
|
+
return textResult(
|
|
364
|
+
`Consult limit reached (${budget.maxUses}/${budget.maxUses} used) — proceed with your best judgment.`,
|
|
365
|
+
);
|
|
366
|
+
}
|
|
367
|
+
// Increment before the first await so two concurrent callers cannot
|
|
368
|
+
// both pass a last-slot check.
|
|
369
|
+
budget.used++;
|
|
370
|
+
const r = await consult(question);
|
|
371
|
+
const remaining = budget.maxUses - budget.used;
|
|
372
|
+
emit({
|
|
373
|
+
type: "advisor_consult",
|
|
374
|
+
caller: from,
|
|
375
|
+
question,
|
|
376
|
+
model,
|
|
377
|
+
durationMs: r.durationMs,
|
|
378
|
+
remaining,
|
|
379
|
+
});
|
|
380
|
+
if (r.unavailable) {
|
|
381
|
+
// Not isError: fail-open, the caller continues normally.
|
|
382
|
+
return textResult(
|
|
383
|
+
`The advisor is unavailable (${r.reason}) — proceed with your best judgment.`,
|
|
384
|
+
);
|
|
385
|
+
}
|
|
386
|
+
return textResult(
|
|
387
|
+
`${r.advice}\n\n[advisor consults remaining: ${remaining}]`,
|
|
388
|
+
);
|
|
389
|
+
},
|
|
390
|
+
);
|
|
391
|
+
}
|
|
392
|
+
|
|
340
393
|
const orchestrationServer = (tools) =>
|
|
341
394
|
createSdkMcpServer({ name: "orchestration", tools });
|
|
342
395
|
|
|
@@ -354,15 +407,16 @@ export function createSupervisorToolServer(ctx) {
|
|
|
354
407
|
]);
|
|
355
408
|
}
|
|
356
409
|
|
|
357
|
-
/** Supervised agent tools: Ask + Answer + Announce + RollCall. */
|
|
358
|
-
export function createSupervisedAgentToolServer(ctx) {
|
|
359
|
-
return orchestrationServer(
|
|
360
|
-
baseTools(ctx, {
|
|
410
|
+
/** Supervised agent tools: Ask + Answer + Announce + RollCall (+ extras). */
|
|
411
|
+
export function createSupervisedAgentToolServer(ctx, { extraTools = [] } = {}) {
|
|
412
|
+
return orchestrationServer([
|
|
413
|
+
...baseTools(ctx, {
|
|
361
414
|
from: "agent",
|
|
362
415
|
defaultTo: "supervisor",
|
|
363
416
|
broadcast: false,
|
|
364
417
|
}),
|
|
365
|
-
|
|
418
|
+
...extraTools,
|
|
419
|
+
]);
|
|
366
420
|
}
|
|
367
421
|
|
|
368
422
|
/** Facilitator tools: Ask + Answer + Announce + RollCall + Conclude. */
|
|
@@ -377,11 +431,15 @@ export function createFacilitatorToolServer(ctx) {
|
|
|
377
431
|
]);
|
|
378
432
|
}
|
|
379
433
|
|
|
380
|
-
/** Facilitated agent tools: Ask + Answer + Announce + RollCall + RequestForComment. */
|
|
381
|
-
export function createFacilitatedAgentToolServer(
|
|
434
|
+
/** Facilitated agent tools: Ask + Answer + Announce + RollCall + RequestForComment (+ extras). */
|
|
435
|
+
export function createFacilitatedAgentToolServer(
|
|
436
|
+
ctx,
|
|
437
|
+
{ from, extraTools = [] },
|
|
438
|
+
) {
|
|
382
439
|
return orchestrationServer([
|
|
383
440
|
...baseTools(ctx, { from, defaultTo: "facilitator", broadcast: true }),
|
|
384
441
|
requestForCommentTool(ctx),
|
|
442
|
+
...extraTools,
|
|
385
443
|
]);
|
|
386
444
|
}
|
|
387
445
|
|
package/src/supervisor.js
CHANGED
|
@@ -21,11 +21,18 @@ import { createAgentRunner } from "./agent-runner.js";
|
|
|
21
21
|
import { composeSystemPrompt } from "./profile-prompt.js";
|
|
22
22
|
import { createMessageBus } from "./message-bus.js";
|
|
23
23
|
import {
|
|
24
|
+
advisorTool,
|
|
24
25
|
createOrchestrationContext,
|
|
25
26
|
createSupervisedAgentToolServer,
|
|
26
27
|
createSupervisorToolServer,
|
|
27
28
|
} from "./orchestration-toolkit.js";
|
|
28
29
|
import { OrchestrationLoop } from "./orchestration-loop.js";
|
|
30
|
+
import {
|
|
31
|
+
createAdvisor,
|
|
32
|
+
createAdvisorBudget,
|
|
33
|
+
withAdvisorGuidance,
|
|
34
|
+
} from "./advisor.js";
|
|
35
|
+
import { createTranscriptRecorder } from "./transcript-recorder.js";
|
|
29
36
|
|
|
30
37
|
/** System prompt for the supervisor lead. L0 mechanics only per COALIGNED. */
|
|
31
38
|
export const SUPERVISOR_SYSTEM_PROMPT =
|
|
@@ -59,6 +66,7 @@ export class Supervisor extends OrchestrationLoop {
|
|
|
59
66
|
* @param {object} deps.ctx
|
|
60
67
|
* @param {object} deps.redactor
|
|
61
68
|
* @param {string} [deps.taskAmend]
|
|
69
|
+
* @param {AbortController} [deps.abortController]
|
|
62
70
|
*/
|
|
63
71
|
constructor({
|
|
64
72
|
supervisorRunner,
|
|
@@ -68,6 +76,7 @@ export class Supervisor extends OrchestrationLoop {
|
|
|
68
76
|
ctx,
|
|
69
77
|
taskAmend,
|
|
70
78
|
redactor,
|
|
79
|
+
abortController,
|
|
71
80
|
}) {
|
|
72
81
|
if (!agentRunner) throw new Error("agentRunner is required");
|
|
73
82
|
if (!supervisorRunner) throw new Error("supervisorRunner is required");
|
|
@@ -82,6 +91,7 @@ export class Supervisor extends OrchestrationLoop {
|
|
|
82
91
|
ctx,
|
|
83
92
|
taskAmend,
|
|
84
93
|
redactor,
|
|
94
|
+
abortController,
|
|
85
95
|
});
|
|
86
96
|
}
|
|
87
97
|
|
|
@@ -125,6 +135,8 @@ const devNull = new Writable({
|
|
|
125
135
|
* @param {string} [deps.profilesDir]
|
|
126
136
|
* @param {string} [deps.taskAmend]
|
|
127
137
|
* @param {Record<string, object>} [deps.agentMcpServers]
|
|
138
|
+
* @param {string} [deps.advisorModel] - Claude model for advisor consults; absent means no Advisor tool is offered.
|
|
139
|
+
* @param {number} [deps.advisorMaxUses] - Session-wide consult budget (default 3).
|
|
128
140
|
* @returns {Supervisor}
|
|
129
141
|
*/
|
|
130
142
|
export function createSupervisor({
|
|
@@ -147,6 +159,8 @@ export function createSupervisor({
|
|
|
147
159
|
agentMcpServers,
|
|
148
160
|
redactor,
|
|
149
161
|
runtime,
|
|
162
|
+
advisorModel,
|
|
163
|
+
advisorMaxUses,
|
|
150
164
|
}) {
|
|
151
165
|
if (!redactor) throw new Error("redactor is required");
|
|
152
166
|
if (!runtime) throw new Error("runtime is required");
|
|
@@ -165,10 +179,58 @@ export function createSupervisor({
|
|
|
165
179
|
|
|
166
180
|
let supervisor;
|
|
167
181
|
const perRunBudget = maxTurns ?? 200;
|
|
182
|
+
const abortController = new AbortController();
|
|
183
|
+
|
|
184
|
+
// Advisor wiring — everything below is gated on advisorModel being set;
|
|
185
|
+
// with it unset the composed prompt and tool surface are byte-identical
|
|
186
|
+
// to today's.
|
|
187
|
+
const budget = advisorModel ? createAdvisorBudget(advisorMaxUses ?? 3) : null;
|
|
188
|
+
const agentSystemPrompt = composeSystemPrompt({
|
|
189
|
+
role: "agent",
|
|
190
|
+
profile: agentProfile,
|
|
191
|
+
profilesDir: resolvedProfilesDir,
|
|
192
|
+
trailer: AGENT_SYSTEM_PROMPT,
|
|
193
|
+
amend: withAdvisorGuidance(agentSystemPromptAmend, budget),
|
|
194
|
+
runtime,
|
|
195
|
+
});
|
|
168
196
|
|
|
169
|
-
|
|
197
|
+
let recorder = null;
|
|
198
|
+
let extraTools;
|
|
199
|
+
if (advisorModel) {
|
|
200
|
+
recorder = createTranscriptRecorder({
|
|
201
|
+
systemPrompt: agentSystemPrompt,
|
|
202
|
+
redactor,
|
|
203
|
+
});
|
|
204
|
+
// Late-bound through the `let supervisor` closure — the instance does
|
|
205
|
+
// not exist yet when the advisor and tool are built.
|
|
206
|
+
const advisor = createAdvisor({
|
|
207
|
+
model: advisorModel,
|
|
208
|
+
cwd: agentCwd,
|
|
209
|
+
query,
|
|
210
|
+
recorder,
|
|
211
|
+
redactor,
|
|
212
|
+
runtime,
|
|
213
|
+
onLine: (line) => supervisor.emitLine("advisor", line),
|
|
214
|
+
});
|
|
215
|
+
abortController.signal.addEventListener("abort", () => advisor.abort());
|
|
216
|
+
extraTools = [
|
|
217
|
+
advisorTool({
|
|
218
|
+
from: "agent",
|
|
219
|
+
consult: (q) => advisor.consult(q),
|
|
220
|
+
emit: (e) => supervisor.emitOrchestratorEvent(e),
|
|
221
|
+
budget,
|
|
222
|
+
model: advisorModel,
|
|
223
|
+
}),
|
|
224
|
+
];
|
|
225
|
+
}
|
|
226
|
+
|
|
227
|
+
const agentServer = createSupervisedAgentToolServer(
|
|
228
|
+
ctx,
|
|
229
|
+
extraTools ? { extraTools } : {},
|
|
230
|
+
);
|
|
170
231
|
const supervisorServer = createSupervisorToolServer(ctx);
|
|
171
232
|
|
|
233
|
+
const emitAgentLine = (line) => supervisor.emitLine("agent", line);
|
|
172
234
|
const agentRunner = createAgentRunner({
|
|
173
235
|
cwd: agentCwd,
|
|
174
236
|
query,
|
|
@@ -176,16 +238,15 @@ export function createSupervisor({
|
|
|
176
238
|
model: agentModel ?? model,
|
|
177
239
|
maxTurns: perRunBudget,
|
|
178
240
|
allowedTools,
|
|
179
|
-
onLine:
|
|
241
|
+
onLine: recorder
|
|
242
|
+
? (line) => {
|
|
243
|
+
emitAgentLine(line);
|
|
244
|
+
recorder.recordMessage(line);
|
|
245
|
+
}
|
|
246
|
+
: emitAgentLine,
|
|
247
|
+
...(recorder && { onPrompt: (text) => recorder.recordPrompt(text) }),
|
|
180
248
|
settingSources: ["project"],
|
|
181
|
-
systemPrompt:
|
|
182
|
-
role: "agent",
|
|
183
|
-
profile: agentProfile,
|
|
184
|
-
profilesDir: resolvedProfilesDir,
|
|
185
|
-
trailer: AGENT_SYSTEM_PROMPT,
|
|
186
|
-
amend: agentSystemPromptAmend,
|
|
187
|
-
runtime,
|
|
188
|
-
}),
|
|
249
|
+
systemPrompt: agentSystemPrompt,
|
|
189
250
|
mcpServers: { orchestration: agentServer, ...agentMcpServers },
|
|
190
251
|
redactor,
|
|
191
252
|
});
|
|
@@ -231,6 +292,7 @@ export function createSupervisor({
|
|
|
231
292
|
ctx,
|
|
232
293
|
taskAmend,
|
|
233
294
|
redactor,
|
|
295
|
+
abortController,
|
|
234
296
|
});
|
|
235
297
|
return supervisor;
|
|
236
298
|
}
|
|
@@ -0,0 +1,94 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* TranscriptRecorder — per-participant in-memory record of the composed
|
|
3
|
+
* system prompt, delivered prompts, and session messages, rendered into the
|
|
4
|
+
* context text an advisor consult forwards. Constructed only when a session
|
|
5
|
+
* runs with an advisor model; the harness otherwise keeps no per-participant
|
|
6
|
+
* record (session lines go straight to the trace stream).
|
|
7
|
+
*
|
|
8
|
+
* Redaction split: the message tap arrives post-redaction (fed from
|
|
9
|
+
* `AgentRunner.#recordLine`), but the seeded system prompt and the prompt
|
|
10
|
+
* tap are raw, so the recorder redacts those itself via the injected
|
|
11
|
+
* redactor.
|
|
12
|
+
*/
|
|
13
|
+
|
|
14
|
+
/**
|
|
15
|
+
* Normalize whatever the harness composed as a system prompt into plain
|
|
16
|
+
* text. In practice always a `{type:"preset", preset:"claude_code", append}`
|
|
17
|
+
* object (every recorded participant is an agent; leads are spec-excluded);
|
|
18
|
+
* a plain string is tolerated and `undefined` accepted.
|
|
19
|
+
* @param {string|{type: string, preset?: string, append?: string}|undefined} systemPrompt
|
|
20
|
+
* @returns {string|undefined}
|
|
21
|
+
*/
|
|
22
|
+
function normalizeSystemPrompt(systemPrompt) {
|
|
23
|
+
if (!systemPrompt) return undefined;
|
|
24
|
+
if (typeof systemPrompt === "string") return systemPrompt;
|
|
25
|
+
if (systemPrompt.append) {
|
|
26
|
+
return `(claude_code preset)\n${systemPrompt.append}`;
|
|
27
|
+
}
|
|
28
|
+
return undefined;
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
/** Wrap content in a tagged section, each tag on its own line. */
|
|
32
|
+
function wrapSection(tag, content) {
|
|
33
|
+
return `<${tag}>\n${content}\n</${tag}>`;
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
/**
|
|
37
|
+
* Create a per-participant transcript recorder.
|
|
38
|
+
*
|
|
39
|
+
* @param {object} deps
|
|
40
|
+
* @param {string|object} [deps.systemPrompt] - The system prompt the harness
|
|
41
|
+
* composed for the participant, as passed to its runner. Raw — redacted at
|
|
42
|
+
* construction.
|
|
43
|
+
* @param {import("./redaction.js").Redactor} deps.redactor
|
|
44
|
+
* @returns {{recordPrompt: (text: string) => void, recordMessage: (line: string) => void, render: () => string}}
|
|
45
|
+
*/
|
|
46
|
+
export function createTranscriptRecorder({ systemPrompt, redactor }) {
|
|
47
|
+
if (!redactor) throw new Error("redactor is required");
|
|
48
|
+
const normalized = normalizeSystemPrompt(systemPrompt);
|
|
49
|
+
const seededPrompt = normalized
|
|
50
|
+
? redactor.redactValue(normalized)
|
|
51
|
+
: undefined;
|
|
52
|
+
/** @type {string[]} */
|
|
53
|
+
const prompts = [];
|
|
54
|
+
/** @type {string[]} */
|
|
55
|
+
const messages = [];
|
|
56
|
+
|
|
57
|
+
return {
|
|
58
|
+
/**
|
|
59
|
+
* Record a delivered (amend-applied) prompt. Raw — redacted here.
|
|
60
|
+
* @param {string} text
|
|
61
|
+
*/
|
|
62
|
+
recordPrompt(text) {
|
|
63
|
+
prompts.push(redactor.redactValue(text));
|
|
64
|
+
},
|
|
65
|
+
/**
|
|
66
|
+
* Record one NDJSON session line as-is (it arrives already redacted
|
|
67
|
+
* from the runner's line path).
|
|
68
|
+
* @param {string} line
|
|
69
|
+
*/
|
|
70
|
+
recordMessage(line) {
|
|
71
|
+
messages.push(line);
|
|
72
|
+
},
|
|
73
|
+
/**
|
|
74
|
+
* Render the record as the advisor's context text: three tagged
|
|
75
|
+
* sections joined by blank lines, each present only when non-empty.
|
|
76
|
+
* NDJSON lines are verbatim — the forwarded context is uncurated by
|
|
77
|
+
* construction (context-size curation is spec-excluded).
|
|
78
|
+
* @returns {string}
|
|
79
|
+
*/
|
|
80
|
+
render() {
|
|
81
|
+
const sections = [];
|
|
82
|
+
if (seededPrompt) {
|
|
83
|
+
sections.push(wrapSection("caller_system_prompt", seededPrompt));
|
|
84
|
+
}
|
|
85
|
+
if (prompts.length > 0) {
|
|
86
|
+
sections.push(wrapSection("caller_prompts", prompts.join("\n\n")));
|
|
87
|
+
}
|
|
88
|
+
if (messages.length > 0) {
|
|
89
|
+
sections.push(wrapSection("caller_transcript", messages.join("\n")));
|
|
90
|
+
}
|
|
91
|
+
return sections.join("\n\n");
|
|
92
|
+
},
|
|
93
|
+
};
|
|
94
|
+
}
|
|
@@ -1,73 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* `fit-benchmark invariants` — check a single task's invariants against a
|
|
3
|
-
* post-run workdir directory without invoking an agent. Useful for
|
|
4
|
-
* re-checking an agent's output against revised grading material.
|
|
5
|
-
*/
|
|
6
|
-
|
|
7
|
-
import { join, resolve } from "node:path";
|
|
8
|
-
import { createServer } from "node:net";
|
|
9
|
-
|
|
10
|
-
import { validateInvariantsRecord } from "../benchmark/result.js";
|
|
11
|
-
import { runInvariants } from "../benchmark/invariants.js";
|
|
12
|
-
import { loadTaskFamily } from "../benchmark/task-family.js";
|
|
13
|
-
|
|
14
|
-
/**
|
|
15
|
-
* @param {import("@forwardimpact/libcli").InvocationContext} ctx
|
|
16
|
-
* @returns {Promise<{ok: true} | {ok: false, code: number, error: string}>}
|
|
17
|
-
*/
|
|
18
|
-
export async function runBenchmarkInvariantsCommand(ctx) {
|
|
19
|
-
const values = ctx.options;
|
|
20
|
-
const runtime = ctx.deps.runtime;
|
|
21
|
-
const familyInput = values.family;
|
|
22
|
-
if (!familyInput)
|
|
23
|
-
return { ok: false, code: 1, error: "--family is required" };
|
|
24
|
-
const taskId = values.task;
|
|
25
|
-
if (!taskId) return { ok: false, code: 1, error: "--task is required" };
|
|
26
|
-
const runDirArg = values["run-dir"];
|
|
27
|
-
if (!runDirArg) return { ok: false, code: 1, error: "--run-dir is required" };
|
|
28
|
-
|
|
29
|
-
const family = await loadTaskFamily(familyInput, runtime);
|
|
30
|
-
const task = family.tasks().find((t) => t.id === taskId);
|
|
31
|
-
if (!task)
|
|
32
|
-
return { ok: false, code: 1, error: `task not found in family: ${taskId}` };
|
|
33
|
-
|
|
34
|
-
const runDir = resolve(runDirArg);
|
|
35
|
-
const cwd = join(runDir, "cwd");
|
|
36
|
-
const port = await allocatePort();
|
|
37
|
-
|
|
38
|
-
const invariants = await runInvariants(task, { cwd, port, runDir }, runtime);
|
|
39
|
-
const record = {
|
|
40
|
-
taskId: task.id,
|
|
41
|
-
invariants,
|
|
42
|
-
exitCode: invariants.exitCode,
|
|
43
|
-
};
|
|
44
|
-
validateInvariantsRecord(record);
|
|
45
|
-
|
|
46
|
-
const line = JSON.stringify(record) + "\n";
|
|
47
|
-
if (values.output) {
|
|
48
|
-
runtime.fsSync.writeFileSync(resolve(values.output), line);
|
|
49
|
-
} else {
|
|
50
|
-
runtime.proc.stdout.write(line);
|
|
51
|
-
}
|
|
52
|
-
return invariants.verdict === "pass"
|
|
53
|
-
? { ok: true }
|
|
54
|
-
: { ok: false, code: 1, error: "" };
|
|
55
|
-
}
|
|
56
|
-
|
|
57
|
-
function allocatePort() {
|
|
58
|
-
return new Promise((res, rej) => {
|
|
59
|
-
const server = createServer();
|
|
60
|
-
server.unref();
|
|
61
|
-
server.on("error", rej);
|
|
62
|
-
server.listen(0, "127.0.0.1", () => {
|
|
63
|
-
const addr = server.address();
|
|
64
|
-
if (!addr || typeof addr === "string") {
|
|
65
|
-
server.close();
|
|
66
|
-
rej(new Error("failed to allocate port"));
|
|
67
|
-
return;
|
|
68
|
-
}
|
|
69
|
-
const port = addr.port;
|
|
70
|
-
server.close(() => res(port));
|
|
71
|
-
});
|
|
72
|
-
});
|
|
73
|
-
}
|