@forwardimpact/libharness 0.1.20 → 1.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -201
- package/README.md +196 -80
- package/bin/fit-benchmark.js +44 -0
- package/bin/fit-harness.js +358 -0
- package/bin/fit-selfedit.js +165 -0
- package/bin/fit-trace.js +510 -0
- package/package.json +42 -12
- package/src/agent-runner.js +256 -0
- package/src/benchmark/apm-installer.js +207 -0
- package/src/benchmark/env-loader.js +158 -0
- package/src/benchmark/hook-env.js +40 -0
- package/src/benchmark/invariants.js +141 -0
- package/src/benchmark/judge.js +187 -0
- package/src/benchmark/npm-installer.js +87 -0
- package/src/benchmark/report.js +522 -0
- package/src/benchmark/result.js +127 -0
- package/src/benchmark/runner.js +583 -0
- package/src/benchmark/task-family.js +260 -0
- package/src/benchmark/workdir.js +298 -0
- package/src/commands/assert.js +153 -0
- package/src/commands/benchmark-definition.js +165 -0
- package/src/commands/benchmark-invariants.js +73 -0
- package/src/commands/benchmark-report.js +51 -0
- package/src/commands/benchmark-run.js +111 -0
- package/src/commands/by-discussion.js +94 -0
- package/src/commands/callback.js +119 -0
- package/src/commands/discuss.js +132 -0
- package/src/commands/facilitate.js +123 -0
- package/src/commands/output.js +36 -0
- package/src/commands/run.js +152 -0
- package/src/commands/supervise.js +136 -0
- package/src/commands/task-input.js +54 -0
- package/src/commands/tee.js +53 -0
- package/src/commands/trace.js +630 -0
- package/src/commands/work-tracker.js +35 -0
- package/src/cost.js +79 -0
- package/src/discuss-tools.js +173 -0
- package/src/discusser.js +394 -0
- package/src/events/github.js +161 -0
- package/src/facilitator.js +205 -0
- package/src/inbox-poller.js +81 -0
- package/src/index.js +72 -2
- package/src/judge.js +210 -0
- package/src/message-bus.js +118 -0
- package/src/orchestration-loop.js +330 -0
- package/src/orchestration-toolkit.js +441 -0
- package/src/orchestrator-helpers.js +23 -0
- package/src/profile-prompt.js +266 -0
- package/src/redaction.js +253 -0
- package/src/render/line-renderer.js +54 -0
- package/src/render/orchestrator-filter.js +19 -0
- package/src/render/palette.js +63 -0
- package/src/render/tool-hints.js +154 -0
- package/src/render/turn-renderer.js +96 -0
- package/src/reply-emitter.js +47 -0
- package/src/sequence-counter.js +21 -0
- package/src/signature-filter.js +27 -0
- package/src/supervisor.js +236 -0
- package/src/tee-writer.js +150 -0
- package/src/trace-collector.js +444 -0
- package/src/trace-github.js +473 -0
- package/src/trace-multi.js +101 -0
- package/src/trace-query.js +748 -0
- package/src/trace-render.js +211 -0
- package/src/trace-usage.js +249 -0
- package/src/fixture/assertions.js +0 -42
- package/src/fixture/cache.js +0 -50
- package/src/fixture/eval.js +0 -146
- package/src/fixture/index.js +0 -9
- package/src/fixture/pathway.js +0 -451
- package/src/fixture/services.js +0 -56
- package/src/mock/clients.js +0 -135
- package/src/mock/config.js +0 -45
- package/src/mock/data.js +0 -46
- package/src/mock/fs.js +0 -111
- package/src/mock/grpc.js +0 -94
- package/src/mock/http.js +0 -60
- package/src/mock/index.js +0 -36
- package/src/mock/infra.js +0 -219
- package/src/mock/logger.js +0 -42
- package/src/mock/observer.js +0 -74
- package/src/mock/resource-index.js +0 -95
- package/src/mock/service-callbacks.js +0 -39
- package/src/mock/services.js +0 -79
- package/src/mock/spy.js +0 -44
- package/src/mock/storage.js +0 -118
|
@@ -0,0 +1,154 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Tool hints — pure one-line formatters for tool-call arguments and
|
|
3
|
+
* tool-result previews.
|
|
4
|
+
*
|
|
5
|
+
* `hintForCall(name, input)` renders the human-meaningful field for each
|
|
6
|
+
* tool (file path, command, pattern, …) sanitized to strip JSON punctuation
|
|
7
|
+
* (`{`, `}`, `"`) and collapsed to a single line ≤ 80 chars.
|
|
8
|
+
*
|
|
9
|
+
* MCP-prefixed tools (`mcp__*`) are an intentional carve-out: their hint is
|
|
10
|
+
* the full input rendered as compact single-line JSON, so `{` and `"` do
|
|
11
|
+
* appear on those lines. Readers of GitHub workflow logs need the full MCP
|
|
12
|
+
* payload to know what was actually sent across the protocol.
|
|
13
|
+
*
|
|
14
|
+
* `previewForResult(content, isError)` collapses a tool result to a single
|
|
15
|
+
* line ≤ 80 chars and flags errors so the renderer can apply the reserved
|
|
16
|
+
* error color and the `Error:` label.
|
|
17
|
+
*/
|
|
18
|
+
|
|
19
|
+
const MAX_HINT_CHARS = 80;
|
|
20
|
+
|
|
21
|
+
/**
|
|
22
|
+
* Strip `{`, `}`, `"`, collapse whitespace, and truncate to MAX_HINT_CHARS.
|
|
23
|
+
* First line only — anything past a newline is dropped. Always returns a
|
|
24
|
+
* string, never null/undefined.
|
|
25
|
+
* @param {unknown} raw
|
|
26
|
+
* @returns {string}
|
|
27
|
+
*/
|
|
28
|
+
function sanitize(raw) {
|
|
29
|
+
if (raw === null || raw === undefined) return "";
|
|
30
|
+
const str = String(raw);
|
|
31
|
+
const firstLine = str.split(/\r?\n/)[0] ?? "";
|
|
32
|
+
const stripped = firstLine.replace(/[{}"]/g, "");
|
|
33
|
+
const collapsed = stripped.replace(/\s+/g, " ").trim();
|
|
34
|
+
if (collapsed.length <= MAX_HINT_CHARS) return collapsed;
|
|
35
|
+
return collapsed.slice(0, MAX_HINT_CHARS - 3) + "...";
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
/**
|
|
39
|
+
* Truncate an already-sanitized string to MAX_HINT_CHARS with a trailing
|
|
40
|
+
* ellipsis when it overflows. Shared by the few handlers that concatenate
|
|
41
|
+
* multiple sanitized pieces before deciding on truncation.
|
|
42
|
+
* @param {string} str
|
|
43
|
+
* @returns {string}
|
|
44
|
+
*/
|
|
45
|
+
function truncate(str) {
|
|
46
|
+
return str.length <= MAX_HINT_CHARS
|
|
47
|
+
? str
|
|
48
|
+
: str.slice(0, MAX_HINT_CHARS - 3) + "...";
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
/**
|
|
52
|
+
* Per-tool hint handlers. Each entry takes the sanitized input object
|
|
53
|
+
* (never null) and returns the hint string. Kept as a flat table so adding
|
|
54
|
+
* a new tool is one entry, not a new branch in a growing switch.
|
|
55
|
+
*/
|
|
56
|
+
const HINT_HANDLERS = {
|
|
57
|
+
Bash: (i) => sanitize(i.command),
|
|
58
|
+
Read: (i) => sanitize(i.file_path),
|
|
59
|
+
Write: (i) => sanitize(i.file_path),
|
|
60
|
+
Edit: (i) => {
|
|
61
|
+
const base = sanitize(i.file_path);
|
|
62
|
+
return i.replace_all
|
|
63
|
+
? (base + " (replace_all)").slice(0, MAX_HINT_CHARS)
|
|
64
|
+
: base;
|
|
65
|
+
},
|
|
66
|
+
Glob: (i) => sanitize(i.pattern),
|
|
67
|
+
Grep: (i) => {
|
|
68
|
+
const pattern = sanitize(i.pattern);
|
|
69
|
+
return i.path ? truncate(`${pattern} in ${sanitize(i.path)}`) : pattern;
|
|
70
|
+
},
|
|
71
|
+
WebFetch: (i) => sanitize(i.url),
|
|
72
|
+
WebSearch: (i) => sanitize(i.query),
|
|
73
|
+
ToolSearch: (i) => sanitize(i.query),
|
|
74
|
+
TodoWrite: (i) => {
|
|
75
|
+
const count = Array.isArray(i.todos) ? i.todos.length : 0;
|
|
76
|
+
return `${count} todos`;
|
|
77
|
+
},
|
|
78
|
+
NotebookEdit: (i) => sanitize(i.notebook_path),
|
|
79
|
+
Skill: (i) => sanitize(i.skill),
|
|
80
|
+
Agent: (i) => sanitize(i.prompt ?? i.description),
|
|
81
|
+
Task: (i) => sanitize(i.prompt ?? i.description),
|
|
82
|
+
};
|
|
83
|
+
|
|
84
|
+
/**
|
|
85
|
+
* Strip the `mcp__<server>__` prefix from MCP-namespaced tool names so logs
|
|
86
|
+
* show the bare method (e.g. `mcp__orchestration__Ask` → `Ask`). Non-MCP
|
|
87
|
+
* names and malformed inputs pass through unchanged.
|
|
88
|
+
* @param {string} name
|
|
89
|
+
* @returns {string}
|
|
90
|
+
*/
|
|
91
|
+
export function simplifyToolName(name) {
|
|
92
|
+
if (!name) return "";
|
|
93
|
+
if (!name.startsWith("mcp__")) return name;
|
|
94
|
+
const parts = name.split("__");
|
|
95
|
+
if (parts.length < 3) return name;
|
|
96
|
+
return parts.slice(2).join("__");
|
|
97
|
+
}
|
|
98
|
+
|
|
99
|
+
/**
|
|
100
|
+
* Map a tool name and input to a one-line human hint.
|
|
101
|
+
*
|
|
102
|
+
* Three branches, in priority order:
|
|
103
|
+
* - A built-in tool with an entry in `HINT_HANDLERS` → sanitized hint, no
|
|
104
|
+
* `{` / `"` from the input (built-in tool hints stay free of JSON
|
|
105
|
+
* punctuation so readers see clean one-liners).
|
|
106
|
+
* - An MCP-prefixed tool (`mcp__*`) → full input rendered as compact
|
|
107
|
+
* single-line JSON; `{` and `"` intentionally appear so readers see
|
|
108
|
+
* the actual MCP payload.
|
|
109
|
+
* - Anything else → "" (the caller still shows the bare tool name).
|
|
110
|
+
*
|
|
111
|
+
* @param {string} name - Tool name (e.g. "Bash", "Read", "mcp__orchestration__Ask")
|
|
112
|
+
* @param {object|null|undefined} input - Raw tool input object from the trace
|
|
113
|
+
* @returns {string} One-line hint, or "" when no rule matches
|
|
114
|
+
*/
|
|
115
|
+
export function hintForCall(name, input) {
|
|
116
|
+
if (!name) return "";
|
|
117
|
+
const safeInput = input && typeof input === "object" ? input : {};
|
|
118
|
+
|
|
119
|
+
const handler = HINT_HANDLERS[name];
|
|
120
|
+
if (handler) return handler(safeInput);
|
|
121
|
+
|
|
122
|
+
if (name.startsWith("mcp__")) return JSON.stringify(safeInput);
|
|
123
|
+
|
|
124
|
+
return "";
|
|
125
|
+
}
|
|
126
|
+
|
|
127
|
+
/**
|
|
128
|
+
* Render a tool result as a single preview line plus an `isError` flag.
|
|
129
|
+
* The flag lets the line-renderer pick the reserved error color without
|
|
130
|
+
* re-inspecting the content.
|
|
131
|
+
*
|
|
132
|
+
* @param {string|object|null|undefined} content - Tool result content
|
|
133
|
+
* @param {boolean} isError - Whether the tool call failed
|
|
134
|
+
* @returns {{text: string, isError: boolean}}
|
|
135
|
+
*/
|
|
136
|
+
export function previewForResult(content, isError) {
|
|
137
|
+
const normalized =
|
|
138
|
+
content === null || content === undefined
|
|
139
|
+
? ""
|
|
140
|
+
: typeof content === "string"
|
|
141
|
+
? content
|
|
142
|
+
: JSON.stringify(content);
|
|
143
|
+
const firstNonBlank =
|
|
144
|
+
normalized
|
|
145
|
+
.split(/\r?\n/)
|
|
146
|
+
.map((l) => l.trim())
|
|
147
|
+
.find((l) => l.length > 0) ?? "";
|
|
148
|
+
|
|
149
|
+
const fallback = isError ? "(no output)" : "(ok)";
|
|
150
|
+
return {
|
|
151
|
+
text: truncate(firstNonBlank || fallback),
|
|
152
|
+
isError,
|
|
153
|
+
};
|
|
154
|
+
}
|
|
@@ -0,0 +1,96 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Turn renderer — maps a structured turn into formatted text lines.
|
|
3
|
+
*
|
|
4
|
+
* Shared by `TeeWriter.flushTurns()` (live stream) and
|
|
5
|
+
* `TraceCollector.toText()` (offline replay) so both emit identical output.
|
|
6
|
+
*/
|
|
7
|
+
|
|
8
|
+
import {
|
|
9
|
+
renderTextLine,
|
|
10
|
+
renderToolCallLine,
|
|
11
|
+
renderToolResultLine,
|
|
12
|
+
} from "./line-renderer.js";
|
|
13
|
+
import {
|
|
14
|
+
hintForCall,
|
|
15
|
+
previewForResult,
|
|
16
|
+
simplifyToolName,
|
|
17
|
+
} from "./tool-hints.js";
|
|
18
|
+
|
|
19
|
+
/**
|
|
20
|
+
* Render a single turn to formatted text lines.
|
|
21
|
+
*
|
|
22
|
+
* @param {object} turn - Structured turn object
|
|
23
|
+
* @param {boolean} withPrefix - Whether to include source labels
|
|
24
|
+
* @returns {string[]} Array of rendered line strings
|
|
25
|
+
*/
|
|
26
|
+
export function renderTurnLines(turn, withPrefix) {
|
|
27
|
+
return TURN_RENDERERS[turn.role]?.(turn, withPrefix) ?? [];
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
/** @param {object} turn @param {boolean} withPrefix @returns {string[]} */
|
|
31
|
+
function renderAssistantTurn(turn, withPrefix) {
|
|
32
|
+
const lines = [];
|
|
33
|
+
for (const block of turn.content) {
|
|
34
|
+
if (block.type === "text") {
|
|
35
|
+
lines.push(
|
|
36
|
+
renderTextLine({ source: turn.source, text: block.text, withPrefix }),
|
|
37
|
+
);
|
|
38
|
+
} else if (block.type === "tool_use") {
|
|
39
|
+
lines.push(
|
|
40
|
+
renderToolCallLine({
|
|
41
|
+
source: turn.source,
|
|
42
|
+
toolName: simplifyToolName(block.name),
|
|
43
|
+
hint: hintForCall(block.name, block.input),
|
|
44
|
+
withPrefix,
|
|
45
|
+
}),
|
|
46
|
+
);
|
|
47
|
+
}
|
|
48
|
+
}
|
|
49
|
+
return lines;
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
/** @param {object} turn @param {boolean} withPrefix @returns {string[]} */
|
|
53
|
+
function renderToolResultTurn(turn, withPrefix) {
|
|
54
|
+
// Successful tool results emit no preview line — the trace document keeps
|
|
55
|
+
// the structured turn, but readers of the streamed log only see errors.
|
|
56
|
+
if (!turn.isError) return [];
|
|
57
|
+
return [
|
|
58
|
+
renderToolResultLine({
|
|
59
|
+
source: turn.source,
|
|
60
|
+
preview: previewForResult(turn.content, true),
|
|
61
|
+
withPrefix,
|
|
62
|
+
}),
|
|
63
|
+
];
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
/** @param {object} turn @param {boolean} withPrefix @returns {string[]} */
|
|
67
|
+
function renderSystemTurn(turn, withPrefix) {
|
|
68
|
+
const label = turn.subtype ?? "system";
|
|
69
|
+
return [
|
|
70
|
+
renderTextLine({ source: turn.source, text: `[${label}]`, withPrefix }),
|
|
71
|
+
];
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
/** @param {object} turn @param {boolean} withPrefix @returns {string[]} */
|
|
75
|
+
function renderUserTurn(turn, withPrefix) {
|
|
76
|
+
const lines = [];
|
|
77
|
+
for (const block of turn.content) {
|
|
78
|
+
if (block.type === "text") {
|
|
79
|
+
lines.push(
|
|
80
|
+
renderTextLine({
|
|
81
|
+
source: turn.source,
|
|
82
|
+
text: `[user] ${block.text}`,
|
|
83
|
+
withPrefix,
|
|
84
|
+
}),
|
|
85
|
+
);
|
|
86
|
+
}
|
|
87
|
+
}
|
|
88
|
+
return lines;
|
|
89
|
+
}
|
|
90
|
+
|
|
91
|
+
const TURN_RENDERERS = {
|
|
92
|
+
assistant: renderAssistantTurn,
|
|
93
|
+
tool_result: renderToolResultTurn,
|
|
94
|
+
system: renderSystemTurn,
|
|
95
|
+
user: renderUserTurn,
|
|
96
|
+
};
|
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* ReplyEmitter — POST reply/ack events to the callback URL as they
|
|
3
|
+
* happen. Each emission is fire-and-forget so the message bus is never
|
|
4
|
+
* blocked on network I/O.
|
|
5
|
+
*/
|
|
6
|
+
export class ReplyEmitter {
|
|
7
|
+
#callbackUrl;
|
|
8
|
+
#correlationId;
|
|
9
|
+
#counter;
|
|
10
|
+
|
|
11
|
+
/**
|
|
12
|
+
* @param {object} deps
|
|
13
|
+
* @param {string|null} deps.callbackUrl
|
|
14
|
+
* @param {string|null} deps.correlationId
|
|
15
|
+
* @param {import("./sequence-counter.js").SequenceCounter} deps.counter
|
|
16
|
+
*/
|
|
17
|
+
constructor({ callbackUrl, correlationId, counter }) {
|
|
18
|
+
this.#callbackUrl = callbackUrl;
|
|
19
|
+
this.#correlationId = correlationId;
|
|
20
|
+
this.#counter = counter;
|
|
21
|
+
}
|
|
22
|
+
|
|
23
|
+
/**
|
|
24
|
+
* @param {object} event
|
|
25
|
+
* @param {"reply"|"ack"} event.kind
|
|
26
|
+
* @param {string} event.body
|
|
27
|
+
* @param {string} event.agent
|
|
28
|
+
* @returns {number} The assigned seq number
|
|
29
|
+
*/
|
|
30
|
+
emit({ kind, body, agent }) {
|
|
31
|
+
const seq = this.#counter.next();
|
|
32
|
+
if (this.#callbackUrl) {
|
|
33
|
+
fetch(this.#callbackUrl, {
|
|
34
|
+
method: "POST",
|
|
35
|
+
headers: { "Content-Type": "application/json" },
|
|
36
|
+
body: JSON.stringify({
|
|
37
|
+
correlation_id: this.#correlationId,
|
|
38
|
+
kind,
|
|
39
|
+
seq,
|
|
40
|
+
body,
|
|
41
|
+
agent,
|
|
42
|
+
}),
|
|
43
|
+
}).catch(() => {});
|
|
44
|
+
}
|
|
45
|
+
return seq;
|
|
46
|
+
}
|
|
47
|
+
}
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* SequenceCounter — global monotonic counter shared across all participants
|
|
3
|
+
* in a session. Single-threaded JS means no synchronization needed.
|
|
4
|
+
*/
|
|
5
|
+
/** Monotonic counter that assigns globally ordered sequence numbers within a session. */
|
|
6
|
+
export class SequenceCounter {
|
|
7
|
+
/** Initialize the counter at zero. */
|
|
8
|
+
constructor() {
|
|
9
|
+
this.value = 0;
|
|
10
|
+
}
|
|
11
|
+
|
|
12
|
+
/** Return the current value and advance the counter by one. */
|
|
13
|
+
next() {
|
|
14
|
+
return this.value++;
|
|
15
|
+
}
|
|
16
|
+
}
|
|
17
|
+
|
|
18
|
+
/** Create a new SequenceCounter starting at zero. */
|
|
19
|
+
export function createSequenceCounter() {
|
|
20
|
+
return new SequenceCounter();
|
|
21
|
+
}
|
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Strip `thinking.signature` base64 blobs from a JSON-serializable value.
|
|
3
|
+
*
|
|
4
|
+
* Applied at the CLI output boundary — the stored structured trace keeps
|
|
5
|
+
* signatures intact (lossless storage), and the display filter drops them
|
|
6
|
+
* by default because they dominate output without helping analysis.
|
|
7
|
+
*
|
|
8
|
+
* Recursively walks the input. For any object whose `type === "thinking"`,
|
|
9
|
+
* the `signature` field is removed after copying. Signatures on objects of
|
|
10
|
+
* any other type are preserved.
|
|
11
|
+
*
|
|
12
|
+
* @param {*} value - Any JSON-serializable value
|
|
13
|
+
* @returns {*} A deep-copy with thinking signatures removed
|
|
14
|
+
*/
|
|
15
|
+
export function stripSignatures(value) {
|
|
16
|
+
if (value === null || typeof value !== "object") return value;
|
|
17
|
+
if (Array.isArray(value)) return value.map(stripSignatures);
|
|
18
|
+
|
|
19
|
+
const result = {};
|
|
20
|
+
for (const [key, val] of Object.entries(value)) {
|
|
21
|
+
result[key] = stripSignatures(val);
|
|
22
|
+
}
|
|
23
|
+
if (result.type === "thinking") {
|
|
24
|
+
delete result.signature;
|
|
25
|
+
}
|
|
26
|
+
return result;
|
|
27
|
+
}
|
|
@@ -0,0 +1,236 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Supervisor — supervise-mode wrapper around `OrchestrationLoop`. One
|
|
3
|
+
* named participant (`"agent"`) coordinated by a lead participant
|
|
4
|
+
* (`"supervisor"`). Structurally the same as `Facilitator` with a
|
|
5
|
+
* single agent; differs only in role names, prompts, and pass-through
|
|
6
|
+
* accessors.
|
|
7
|
+
*
|
|
8
|
+
* Ask is async (same contract as facilitate / discuss): returns
|
|
9
|
+
* `{askIds:[N]}` immediately; the agent's reply arrives on the
|
|
10
|
+
* supervisor's next turn as `[answer#N] agent: <text>`. The supervisor
|
|
11
|
+
* sees the agent at each Ask boundary, plans the next step, and
|
|
12
|
+
* eventually calls Conclude.
|
|
13
|
+
*
|
|
14
|
+
* For tighter feedback loops, size the agent's per-turn budget down
|
|
15
|
+
* (smaller `maxTurns` on the agent runner) so each Ask returns sooner.
|
|
16
|
+
*/
|
|
17
|
+
|
|
18
|
+
import { Writable } from "node:stream";
|
|
19
|
+
import { resolve } from "node:path";
|
|
20
|
+
import { createAgentRunner } from "./agent-runner.js";
|
|
21
|
+
import { composeSystemPrompt } from "./profile-prompt.js";
|
|
22
|
+
import { createMessageBus } from "./message-bus.js";
|
|
23
|
+
import {
|
|
24
|
+
createOrchestrationContext,
|
|
25
|
+
createSupervisedAgentToolServer,
|
|
26
|
+
createSupervisorToolServer,
|
|
27
|
+
} from "./orchestration-toolkit.js";
|
|
28
|
+
import { OrchestrationLoop } from "./orchestration-loop.js";
|
|
29
|
+
|
|
30
|
+
/** System prompt for the supervisor lead. L0 mechanics only per COALIGNED. */
|
|
31
|
+
export const SUPERVISOR_SYSTEM_PROMPT =
|
|
32
|
+
"You supervise one agent.\n" +
|
|
33
|
+
"Use `Ask` to delegate the agent's task to the agent.\n" +
|
|
34
|
+
"`Ask` is async and returns {askIds:[N]} immediately.\n" +
|
|
35
|
+
"The reply arrives on your next turn as `[answer#N] agent: <text>` in your inbox.\n" +
|
|
36
|
+
"End your turn while Asks are pending. The system resumes you when an answer arrives.\n" +
|
|
37
|
+
"If the agent goes off-track, send a corrective `Ask`.\n" +
|
|
38
|
+
"End every session by calling `Conclude` with a verdict and summary.";
|
|
39
|
+
|
|
40
|
+
/** System prompt for the supervised agent. L0 mechanics only per COALIGNED. */
|
|
41
|
+
export const AGENT_SYSTEM_PROMPT =
|
|
42
|
+
"A supervisor directs your work.\n" +
|
|
43
|
+
"Each question arrives as `[ask#N] supervisor: <text>` in your inbox.\n" +
|
|
44
|
+
"Quote N as askId on your `Answer` to route the reply correctly.\n" +
|
|
45
|
+
"If the task already contains a completed response with no new human input after it, `Answer` that no further action is needed.\n" +
|
|
46
|
+
"Do not redo completed work.";
|
|
47
|
+
|
|
48
|
+
/**
|
|
49
|
+
* Supervise-mode wrapper around `OrchestrationLoop`. The lead is
|
|
50
|
+
* `"supervisor"`, one participant is `"agent"`, mode tag is `"supervised"`.
|
|
51
|
+
*/
|
|
52
|
+
export class Supervisor extends OrchestrationLoop {
|
|
53
|
+
/**
|
|
54
|
+
* @param {object} deps
|
|
55
|
+
* @param {import("./agent-runner.js").AgentRunner} deps.supervisorRunner
|
|
56
|
+
* @param {import("./agent-runner.js").AgentRunner} deps.agentRunner
|
|
57
|
+
* @param {import("./message-bus.js").MessageBus} deps.messageBus
|
|
58
|
+
* @param {import("stream").Writable} deps.output
|
|
59
|
+
* @param {object} deps.ctx
|
|
60
|
+
* @param {object} deps.redactor
|
|
61
|
+
* @param {string} [deps.taskAmend]
|
|
62
|
+
*/
|
|
63
|
+
constructor({
|
|
64
|
+
supervisorRunner,
|
|
65
|
+
agentRunner,
|
|
66
|
+
messageBus,
|
|
67
|
+
output,
|
|
68
|
+
ctx,
|
|
69
|
+
taskAmend,
|
|
70
|
+
redactor,
|
|
71
|
+
}) {
|
|
72
|
+
if (!agentRunner) throw new Error("agentRunner is required");
|
|
73
|
+
if (!supervisorRunner) throw new Error("supervisorRunner is required");
|
|
74
|
+
if (!output) throw new Error("output is required");
|
|
75
|
+
super({
|
|
76
|
+
leadRunner: supervisorRunner,
|
|
77
|
+
agents: [{ name: "agent", role: "agent", runner: agentRunner }],
|
|
78
|
+
messageBus,
|
|
79
|
+
output,
|
|
80
|
+
leadName: "supervisor",
|
|
81
|
+
mode: "supervised",
|
|
82
|
+
ctx,
|
|
83
|
+
taskAmend,
|
|
84
|
+
redactor,
|
|
85
|
+
});
|
|
86
|
+
}
|
|
87
|
+
|
|
88
|
+
/** Readability shims for tests that read the runners by their domain names. */
|
|
89
|
+
/** Readability shim — exposes the lead runner under its mode-specific name. */
|
|
90
|
+
get supervisorRunner() {
|
|
91
|
+
return this.leadRunner;
|
|
92
|
+
}
|
|
93
|
+
|
|
94
|
+
/** Readability shim — exposes the single agent runner directly. */
|
|
95
|
+
get agentRunner() {
|
|
96
|
+
return this.agents[0].runner;
|
|
97
|
+
}
|
|
98
|
+
}
|
|
99
|
+
|
|
100
|
+
const devNull = new Writable({
|
|
101
|
+
write(_chunk, _enc, cb) {
|
|
102
|
+
cb();
|
|
103
|
+
},
|
|
104
|
+
});
|
|
105
|
+
|
|
106
|
+
/**
|
|
107
|
+
* Factory — wires the supervisor + agent runners and the orchestration
|
|
108
|
+
* context. Mirrors the facilitator factory in shape.
|
|
109
|
+
*
|
|
110
|
+
* @param {object} deps
|
|
111
|
+
* @param {string} deps.supervisorCwd
|
|
112
|
+
* @param {string} deps.agentCwd
|
|
113
|
+
* @param {function} deps.query
|
|
114
|
+
* @param {import("stream").Writable} deps.output
|
|
115
|
+
* @param {string} [deps.model]
|
|
116
|
+
* @param {string} [deps.agentModel]
|
|
117
|
+
* @param {string} [deps.supervisorModel]
|
|
118
|
+
* @param {number} [deps.maxTurns] - Per-runner SDK turn budget (default 200).
|
|
119
|
+
* @param {string[]} [deps.allowedTools]
|
|
120
|
+
* @param {string[]} [deps.supervisorAllowedTools]
|
|
121
|
+
* @param {string[]} [deps.supervisorDisallowedTools]
|
|
122
|
+
* @param {string} [deps.supervisorProfile]
|
|
123
|
+
* @param {string} [deps.agentProfile]
|
|
124
|
+
* @param {string} [deps.agentSystemPromptAmend] - Amendment folded into the agent's `<session_protocol>` section, after the protocol trailer.
|
|
125
|
+
* @param {string} [deps.profilesDir]
|
|
126
|
+
* @param {string} [deps.taskAmend]
|
|
127
|
+
* @param {Record<string, object>} [deps.agentMcpServers]
|
|
128
|
+
* @returns {Supervisor}
|
|
129
|
+
*/
|
|
130
|
+
export function createSupervisor({
|
|
131
|
+
supervisorCwd,
|
|
132
|
+
agentCwd,
|
|
133
|
+
query,
|
|
134
|
+
output,
|
|
135
|
+
model,
|
|
136
|
+
agentModel,
|
|
137
|
+
supervisorModel,
|
|
138
|
+
maxTurns,
|
|
139
|
+
allowedTools,
|
|
140
|
+
supervisorAllowedTools,
|
|
141
|
+
supervisorDisallowedTools,
|
|
142
|
+
supervisorProfile,
|
|
143
|
+
agentProfile,
|
|
144
|
+
agentSystemPromptAmend,
|
|
145
|
+
profilesDir,
|
|
146
|
+
taskAmend,
|
|
147
|
+
agentMcpServers,
|
|
148
|
+
redactor,
|
|
149
|
+
runtime,
|
|
150
|
+
}) {
|
|
151
|
+
if (!redactor) throw new Error("redactor is required");
|
|
152
|
+
if (!runtime) throw new Error("runtime is required");
|
|
153
|
+
const resolvedProfilesDir =
|
|
154
|
+
profilesDir ?? resolve(supervisorCwd, ".claude/agents");
|
|
155
|
+
|
|
156
|
+
const ctx = createOrchestrationContext();
|
|
157
|
+
const messageBus = createMessageBus({
|
|
158
|
+
participants: ["supervisor", "agent"],
|
|
159
|
+
});
|
|
160
|
+
ctx.messageBus = messageBus;
|
|
161
|
+
ctx.participants = [
|
|
162
|
+
{ name: "supervisor", role: "supervisor" },
|
|
163
|
+
{ name: "agent", role: "agent" },
|
|
164
|
+
];
|
|
165
|
+
|
|
166
|
+
let supervisor;
|
|
167
|
+
const perRunBudget = maxTurns ?? 200;
|
|
168
|
+
|
|
169
|
+
const agentServer = createSupervisedAgentToolServer(ctx);
|
|
170
|
+
const supervisorServer = createSupervisorToolServer(ctx);
|
|
171
|
+
|
|
172
|
+
const agentRunner = createAgentRunner({
|
|
173
|
+
cwd: agentCwd,
|
|
174
|
+
query,
|
|
175
|
+
output: devNull,
|
|
176
|
+
model: agentModel ?? model,
|
|
177
|
+
maxTurns: perRunBudget,
|
|
178
|
+
allowedTools,
|
|
179
|
+
onLine: (line) => supervisor.emitLine("agent", line),
|
|
180
|
+
settingSources: ["project"],
|
|
181
|
+
systemPrompt: composeSystemPrompt({
|
|
182
|
+
role: "agent",
|
|
183
|
+
profile: agentProfile,
|
|
184
|
+
profilesDir: resolvedProfilesDir,
|
|
185
|
+
trailer: AGENT_SYSTEM_PROMPT,
|
|
186
|
+
amend: agentSystemPromptAmend,
|
|
187
|
+
runtime,
|
|
188
|
+
}),
|
|
189
|
+
mcpServers: { orchestration: agentServer, ...agentMcpServers },
|
|
190
|
+
redactor,
|
|
191
|
+
});
|
|
192
|
+
|
|
193
|
+
const defaultDisallowed = [
|
|
194
|
+
"Agent",
|
|
195
|
+
"Task",
|
|
196
|
+
"TaskOutput",
|
|
197
|
+
"TaskStop",
|
|
198
|
+
"Write",
|
|
199
|
+
"Edit",
|
|
200
|
+
];
|
|
201
|
+
const disallowedTools = supervisorDisallowedTools
|
|
202
|
+
? [...new Set([...defaultDisallowed, ...supervisorDisallowedTools])]
|
|
203
|
+
: defaultDisallowed;
|
|
204
|
+
|
|
205
|
+
const supervisorRunner = createAgentRunner({
|
|
206
|
+
cwd: supervisorCwd,
|
|
207
|
+
query,
|
|
208
|
+
output: devNull,
|
|
209
|
+
model: supervisorModel ?? model,
|
|
210
|
+
maxTurns: perRunBudget,
|
|
211
|
+
allowedTools: supervisorAllowedTools ?? ["Read", "Glob", "Grep", "Bash"],
|
|
212
|
+
disallowedTools,
|
|
213
|
+
onLine: (line) => supervisor.emitLine("supervisor", line),
|
|
214
|
+
settingSources: ["project"],
|
|
215
|
+
systemPrompt: composeSystemPrompt({
|
|
216
|
+
role: "lead",
|
|
217
|
+
profile: supervisorProfile,
|
|
218
|
+
profilesDir: resolvedProfilesDir,
|
|
219
|
+
trailer: SUPERVISOR_SYSTEM_PROMPT,
|
|
220
|
+
runtime,
|
|
221
|
+
}),
|
|
222
|
+
mcpServers: { orchestration: supervisorServer },
|
|
223
|
+
redactor,
|
|
224
|
+
});
|
|
225
|
+
|
|
226
|
+
supervisor = new Supervisor({
|
|
227
|
+
supervisorRunner,
|
|
228
|
+
agentRunner,
|
|
229
|
+
messageBus,
|
|
230
|
+
output,
|
|
231
|
+
ctx,
|
|
232
|
+
taskAmend,
|
|
233
|
+
redactor,
|
|
234
|
+
});
|
|
235
|
+
return supervisor;
|
|
236
|
+
}
|