@forwardimpact/libharness 1.3.0 → 2.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -62,12 +62,74 @@ export function evaluateAssertion(values, args, fsSync) {
62
62
 
63
63
  const output = { test: testName, pass: result.pass };
64
64
  if (result.message) output.message = result.message;
65
+ applyGradingFlags(values, output);
65
66
  return output;
66
67
  }
67
68
 
69
+ /**
70
+ * Attach the check-row grading role: `--gate` marks a gate check, `--weight`
71
+ * attaches a numeric weight (0 marks the row diagnostic). `--gate` with any
72
+ * `--weight` — 0 included — is invalid: a stray weight must never silently
73
+ * disarm a gate.
74
+ * @param {object} values
75
+ * @param {{test: string, pass: boolean, message?: string}} output - Mutated.
76
+ */
77
+ function applyGradingFlags(values, output) {
78
+ const hasWeight = values.weight !== undefined;
79
+ if (values.gate && hasWeight) {
80
+ throw new Error("assert: --gate cannot be combined with --weight");
81
+ }
82
+ if (values.gate) output.gate = true;
83
+ if (hasWeight) {
84
+ const weight = parseWeight(values.weight);
85
+ if (weight === null) {
86
+ throw new Error(
87
+ `assert: invalid --weight '${values.weight}' (expected a finite number ≥ 0)`,
88
+ );
89
+ }
90
+ output.weight = weight;
91
+ }
92
+ }
93
+
94
+ /**
95
+ * Parse a `--weight` value; null when invalid. A blank string is invalid —
96
+ * `Number("")` is 0, which would silently demote the check to a diagnostic.
97
+ * @param {string} raw
98
+ * @returns {number | null}
99
+ */
100
+ function parseWeight(raw) {
101
+ if (typeof raw === "string" && raw.trim() === "") return null;
102
+ const weight = Number(raw);
103
+ return Number.isFinite(weight) && weight >= 0 ? weight : null;
104
+ }
105
+
106
+ /**
107
+ * The grading role an emit-then-fail row keeps: a failing check must not
108
+ * lose its authored role — an errored gate that demoted to a scored row
109
+ * would let a broken scaffold earn partial credit instead of zeroing the
110
+ * score. Invalid or conflicting flags yield no role (the row fails as a
111
+ * unit-weight scored check).
112
+ * @param {object} values
113
+ * @returns {{gate?: true, weight?: number}}
114
+ */
115
+ function errorRowRole(values) {
116
+ const hasWeight = values.weight !== undefined;
117
+ if (values.gate && !hasWeight) return { gate: true };
118
+ if (!values.gate && hasWeight) {
119
+ const weight = parseWeight(values.weight);
120
+ if (weight !== null) return { weight };
121
+ }
122
+ return {};
123
+ }
124
+
68
125
  /**
69
126
  * Run an assertion, write JSON to stdout, and return a failure envelope when
70
127
  * the assertion does not pass.
128
+ *
129
+ * Emit-then-fail on every failure path: an invalid grading flag or an
130
+ * errored evaluation (e.g. `--grep` against a file the agent deleted) writes
131
+ * a failing row before the nonzero exit, so a typo or a vanished target
132
+ * shrinks the score, never the denominator.
71
133
  * @param {import("@forwardimpact/libcli").InvocationContext} ctx
72
134
  * @returns {Promise<{ok: true} | {ok: false, code: number, error: string}>}
73
135
  */
@@ -78,6 +140,16 @@ export async function runAssertCommand(ctx) {
78
140
  try {
79
141
  result = evaluateAssertion(ctx.options, args, runtime.fsSync);
80
142
  } catch (err) {
143
+ const reason = err.message.startsWith("assert: ")
144
+ ? err.message
145
+ : `assert: ${err.message}`;
146
+ const row = {
147
+ test: ctx.args["test-name"] ?? "(missing test name)",
148
+ pass: false,
149
+ ...errorRowRole(ctx.options),
150
+ message: reason,
151
+ };
152
+ runtime.proc.stdout.write(JSON.stringify(row) + "\n");
81
153
  return { ok: false, code: 1, error: err.message };
82
154
  }
83
155
  runtime.proc.stdout.write(JSON.stringify(result) + "\n");
@@ -5,7 +5,7 @@
5
5
  */
6
6
 
7
7
  import { runBenchmarkRunCommand } from "./benchmark-run.js";
8
- import { runBenchmarkInvariantsCommand } from "./benchmark-invariants.js";
8
+ import { runBenchmarkGradeCommand } from "./benchmark-grade.js";
9
9
  import { runBenchmarkReportCommand } from "./benchmark-report.js";
10
10
  import {
11
11
  BENCHMARK_AGENT_MODEL,
@@ -95,11 +95,11 @@ export const definition = {
95
95
  },
96
96
  },
97
97
  {
98
- name: "invariants",
98
+ name: "grade",
99
99
  args: [],
100
- handler: runBenchmarkInvariantsCommand,
100
+ handler: runBenchmarkGradeCommand,
101
101
  description:
102
- "Check a single task's invariants against a post-run workdir without invoking an agent.",
102
+ "Grade a single task against a post-run workdir without invoking an agent: run the hidden test suite and the invariants script, then derive the verdict from the check rows (the exit mirrors it).",
103
103
  options: {
104
104
  family: {
105
105
  type: "string",
@@ -112,7 +112,7 @@ export const definition = {
112
112
  "run-dir": {
113
113
  type: "string",
114
114
  description:
115
- "Post-run directory whose cwd/ subdir is the agent CWD; invariants run against that cwd — the path hooks receive as $AGENT_CWD",
115
+ "Post-run directory whose cwd/ subdir is the agent CWD; both producers run against that cwd — the path hooks receive as $AGENT_CWD",
116
116
  },
117
117
  output: {
118
118
  type: "string",
@@ -140,6 +140,11 @@ export const definition = {
140
140
  type: "string",
141
141
  description: "Output format (json|text, default: json)",
142
142
  },
143
+ detail: {
144
+ type: "string",
145
+ description:
146
+ "Text report verbosity (full|compact, default: full). compact omits per-task detail — useful for sharded run summaries.",
147
+ },
143
148
  },
144
149
  },
145
150
  ],
@@ -154,7 +159,7 @@ export const definition = {
154
159
  "fit-benchmark run --family=./families/coding --work-tracker=filesystem",
155
160
  "fit-benchmark run --family=./families/coding --skills-from=. --task=todo-api",
156
161
  `fit-benchmark run --family=./families/coding --runs=10 --agent-model=${BENCHMARK_AGENT_MODEL}`,
157
- "fit-benchmark invariants --family=./families/coding --task=todo-api --run-dir=./benchmark-runs/runs/todo-api/0",
162
+ "fit-benchmark grade --family=./families/coding --task=todo-api --run-dir=./benchmark-runs/runs/todo-api/0",
158
163
  "fit-benchmark report --format=text",
159
164
  "fit-benchmark report --input=./runs/today --k=1,3,5 --format=text",
160
165
  ],
@@ -0,0 +1,82 @@
1
+ /**
2
+ * `fit-benchmark grade` — run both check-row producers (the hidden test
3
+ * suite and the invariants script) against a post-run workdir directory and
4
+ * grade the merged rows with the same derivation the benchmark runner uses.
5
+ * No agent and no judge run, so authors validate a task's grading material
6
+ * against fixtures without paying for agent sessions; the process exit
7
+ * mirrors the graded verdict.
8
+ */
9
+
10
+ import { join, resolve } from "node:path";
11
+
12
+ import { validateGradeRecord } from "../benchmark/result.js";
13
+ import { runInvariants } from "../benchmark/invariants.js";
14
+ import { runHiddenTests } from "../benchmark/hidden-tests.js";
15
+ import { runProducersAndGrade } from "../benchmark/grade.js";
16
+ import { loadTaskFamily } from "../benchmark/task-family.js";
17
+ import { probeFreePort } from "../benchmark/workdir.js";
18
+
19
+ /**
20
+ * @param {import("@forwardimpact/libcli").InvocationContext} ctx
21
+ * @returns {Promise<{ok: true} | {ok: false, code: number, error: string}>}
22
+ */
23
+ export async function runBenchmarkGradeCommand(ctx) {
24
+ const values = ctx.options;
25
+ const runtime = ctx.deps.runtime;
26
+ const familyInput = values.family;
27
+ if (!familyInput)
28
+ return { ok: false, code: 1, error: "--family is required" };
29
+ const taskId = values.task;
30
+ if (!taskId) return { ok: false, code: 1, error: "--task is required" };
31
+ const runDirArg = values["run-dir"];
32
+ if (!runDirArg) return { ok: false, code: 1, error: "--run-dir is required" };
33
+
34
+ const family = await loadTaskFamily(familyInput, runtime);
35
+ const task = family.tasks().find((t) => t.id === taskId);
36
+ if (!task)
37
+ return { ok: false, code: 1, error: `task not found in family: ${taskId}` };
38
+
39
+ const runDir = resolve(runDirArg);
40
+ const cwd = join(runDir, "cwd");
41
+ const port = await probeFreePort();
42
+ const cellCtx = { cwd, port, runDir, familyDir: family.rootPath };
43
+
44
+ const { invariants, hiddenRows, engineError, healthy, grade } =
45
+ await runProducersAndGrade(task, cellCtx, runtime, {
46
+ runInvariants,
47
+ runHiddenTests,
48
+ });
49
+ // Same effective-score rule as the runner, minus the judge (none runs
50
+ // here): an unhealthy grader or a failing gate zeroes the score, so a
51
+ // crashed hook can never mint marks from the rows it emitted before dying.
52
+ // Unlike a runner record — where `grade.score` stays the raw fraction and
53
+ // the zeroing lands on the top-level `score` — this record has no second
54
+ // field, so `grade.score` carries the effective value here.
55
+ if (grade.score !== undefined && !(healthy && grade.gatesPass)) {
56
+ grade.score = 0;
57
+ }
58
+ const record = {
59
+ taskId: task.id,
60
+ grade,
61
+ invariants,
62
+ ...(task.tests && {
63
+ hiddenTests: {
64
+ details: hiddenRows,
65
+ ...(engineError && { error: engineError.message }),
66
+ },
67
+ }),
68
+ // Mirrors the script for diagnosis; the graded verdict drives the exit.
69
+ exitCode: invariants.exitCode,
70
+ };
71
+ validateGradeRecord(record);
72
+
73
+ const line = JSON.stringify(record) + "\n";
74
+ if (values.output) {
75
+ runtime.fsSync.writeFileSync(resolve(values.output), line);
76
+ } else {
77
+ runtime.proc.stdout.write(line);
78
+ }
79
+ return grade.verdict === "pass"
80
+ ? { ok: true }
81
+ : { ok: false, code: 1, error: "" };
82
+ }
@@ -1,7 +1,9 @@
1
1
  /**
2
2
  * `fit-benchmark report` — aggregate `results.jsonl` into pass@k via the
3
3
  * OpenAI HumanEval estimator. Output is JSON by default; pass --format=text
4
- * to render a markdown table.
4
+ * to render a markdown table. --detail=compact drops the per-task detail
5
+ * sections so a sharded run's per-shard summary stays short (the merge job
6
+ * renders the full report over the combined ledger).
5
7
  */
6
8
 
7
9
  import { resolve } from "node:path";
@@ -35,11 +37,19 @@ export async function runBenchmarkReportCommand(ctx) {
35
37
  if (format !== "json" && format !== "text") {
36
38
  return { ok: false, code: 1, error: "--format must be 'json' or 'text'" };
37
39
  }
40
+ const detail = values.detail ?? "full";
41
+ if (detail !== "full" && detail !== "compact") {
42
+ return {
43
+ ok: false,
44
+ code: 1,
45
+ error: "--detail must be 'full' or 'compact'",
46
+ };
47
+ }
38
48
 
39
49
  const report = await aggregate({
40
50
  inputDir: resolve(inputDir),
41
51
  kValues,
42
- includeRuns: format === "text",
52
+ includeRuns: format === "text" && detail === "full",
43
53
  runtime,
44
54
  });
45
55
  if (format === "text") {
@@ -3,6 +3,7 @@ import { isoTimestamp } from "@forwardimpact/libutil";
3
3
  import { createDiscusser } from "../discusser.js";
4
4
  import { createRedactor } from "../redaction.js";
5
5
  import { createTeeWriter } from "../tee-writer.js";
6
+ import { parseAdvisorOptions } from "./advisor-flags.js";
6
7
  import { resolveTaskContent } from "./task-input.js";
7
8
  import { resolveWorkTracker } from "./work-tracker.js";
8
9
  import { AGENT_MODEL, LEAD_MODEL } from "@forwardimpact/libutil/models";
@@ -67,6 +68,7 @@ export function parseDiscussOptions(values, runtime) {
67
68
  callbackUrl: runtime.proc.env.CALLBACK_URL ?? null,
68
69
  inboxUrl: runtime.proc.env.INBOX_URL ?? null,
69
70
  correlationId: runtime.proc.env.CORRELATION_ID ?? null,
71
+ ...parseAdvisorOptions(values),
70
72
  };
71
73
  }
72
74
 
@@ -121,6 +123,8 @@ export async function runDiscussCommand(ctx) {
121
123
  inboxUrl: opts.inboxUrl,
122
124
  correlationId: opts.correlationId,
123
125
  runtime,
126
+ advisorModel: opts.advisorModel,
127
+ advisorMaxUses: opts.advisorMaxUses,
124
128
  });
125
129
 
126
130
  const result = await discusser.run(opts.taskContent);
@@ -3,6 +3,7 @@ import { isoTimestamp } from "@forwardimpact/libutil";
3
3
  import { createFacilitator } from "../facilitator.js";
4
4
  import { createRedactor } from "../redaction.js";
5
5
  import { createTeeWriter } from "../tee-writer.js";
6
+ import { parseAdvisorOptions } from "./advisor-flags.js";
6
7
  import { resolveTaskContent } from "./task-input.js";
7
8
  import { resolveWorkTracker } from "./work-tracker.js";
8
9
  import { AGENT_MODEL, LEAD_MODEL } from "@forwardimpact/libutil/models";
@@ -60,6 +61,7 @@ export function parseFacilitateOptions(values, runtime) {
60
61
  outputPath: values.output,
61
62
  facilitatorProfile: values["lead-profile"] || undefined,
62
63
  workTracker: resolveWorkTracker(values, runtime?.proc?.env),
64
+ ...parseAdvisorOptions(values),
63
65
  };
64
66
  }
65
67
 
@@ -112,6 +114,8 @@ export async function runFacilitateCommand(ctx) {
112
114
  taskAmend: opts.taskAmend,
113
115
  redactor,
114
116
  runtime,
117
+ advisorModel: opts.advisorModel,
118
+ advisorMaxUses: opts.advisorMaxUses,
115
119
  });
116
120
 
117
121
  const result = await facilitator.run(opts.taskContent);
@@ -1,11 +1,23 @@
1
1
  import { Writable } from "node:stream";
2
2
  import { resolve } from "node:path";
3
3
  import { isoTimestamp } from "@forwardimpact/libutil";
4
+ import { createSdkMcpServer } from "@anthropic-ai/claude-agent-sdk";
4
5
  import { createAgentRunner } from "../agent-runner.js";
5
- import { composeProfilePrompt } from "../profile-prompt.js";
6
+ import {
7
+ advisorGuidance,
8
+ createAdvisor,
9
+ createAdvisorBudget,
10
+ } from "../advisor.js";
11
+ import { advisorTool } from "../orchestration-toolkit.js";
12
+ import {
13
+ composeProfilePrompt,
14
+ composeSystemPrompt,
15
+ } from "../profile-prompt.js";
6
16
  import { createRedactor } from "../redaction.js";
7
17
  import { createTeeWriter } from "../tee-writer.js";
18
+ import { createTranscriptRecorder } from "../transcript-recorder.js";
8
19
  import { SequenceCounter } from "../sequence-counter.js";
20
+ import { parseAdvisorOptions } from "./advisor-flags.js";
9
21
  import { resolveWorkTracker } from "./work-tracker.js";
10
22
  import { resolveTaskContent } from "./task-input.js";
11
23
  import { createServiceConfig } from "@forwardimpact/libconfig";
@@ -15,7 +27,7 @@ import { AGENT_MODEL } from "@forwardimpact/libutil/models";
15
27
  * Parse and validate run command options from parsed values.
16
28
  * @param {object} values - Parsed option values from cli.parse()
17
29
  * @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
18
- * @returns {{ taskContent: string, cwd: string, model: string, maxTurns: number, outputPath: string|undefined, agentProfile: string|undefined, workTracker: string, allowedTools: string[] }}
30
+ * @returns {{ taskContent: string, taskAmend: string|undefined, cwd: string, agentModel: string, maxTurns: number, outputPath: string|undefined, agentProfile: string|undefined, workTracker: string, allowedTools: string[], mcpServer: string|undefined, advisorModel: string|undefined, advisorMaxUses: number }}
19
31
  */
20
32
  export function parseRunOptions(values, runtime) {
21
33
  const { task: taskContent, amend: taskAmend } = resolveTaskContent(
@@ -40,9 +52,148 @@ export function parseRunOptions(values, runtime) {
40
52
  "Bash,Read,Glob,Grep,Write,Edit,Agent,TodoWrite"
41
53
  ).split(","),
42
54
  mcpServer: values["mcp-server"] || undefined,
55
+ ...parseAdvisorOptions(values),
43
56
  };
44
57
  }
45
58
 
59
+ const devNull = new Writable({
60
+ write(_chunk, _enc, cb) {
61
+ cb();
62
+ },
63
+ });
64
+
65
+ /**
66
+ * Wire the run-mode agent session: external MCP entry, `LIBHARNESS_*` env
67
+ * writes, system-prompt composition, and — when an advisor model is set —
68
+ * the advisor wiring (budget, recorder, advisor session, dedicated MCP
69
+ * server holding only the `Advisor` tool). Extracted from `runRunCommand`
70
+ * so tests can inject a fake `query`.
71
+ *
72
+ * Run mode has no stop path (the command simply awaits the runner), so the
73
+ * consult timeout is deliberately the advisor's only guard.
74
+ *
75
+ * @param {object} deps
76
+ * @param {ReturnType<typeof parseRunOptions>} deps.opts
77
+ * @param {import("../redaction.js").Redactor} deps.redactor
78
+ * @param {import("stream").Writable} deps.output - Envelope NDJSON sink.
79
+ * @param {SequenceCounter} deps.counter
80
+ * @param {function} deps.query - SDK query function.
81
+ * @param {import("@forwardimpact/libutil/runtime").Runtime} deps.runtime
82
+ * @returns {Promise<{runner: import("../agent-runner.js").AgentRunner, advisor: object|null}>}
83
+ */
84
+ export async function wireRunSession({
85
+ opts,
86
+ redactor,
87
+ output,
88
+ counter,
89
+ query,
90
+ runtime,
91
+ }) {
92
+ const emitEnvelope = (source, event) => {
93
+ output.write(
94
+ JSON.stringify(
95
+ redactor.redactValue({ source, seq: counter.next(), event }),
96
+ ) + "\n",
97
+ );
98
+ };
99
+ const onLine = (line) => emitEnvelope("agent", JSON.parse(line));
100
+
101
+ let mcpServers = null;
102
+ const allowedTools = opts.allowedTools;
103
+ if (opts.mcpServer) {
104
+ const mcpConfig = await createServiceConfig("mcp");
105
+ mcpServers = {
106
+ [opts.mcpServer]: {
107
+ type: "http",
108
+ url: mcpConfig.url,
109
+ headers: { Authorization: `Bearer ${mcpConfig.mcpToken()}` },
110
+ },
111
+ };
112
+ allowedTools.push(`mcp__${opts.mcpServer}__*`);
113
+ }
114
+
115
+ if (opts.agentProfile) {
116
+ runtime.proc.env.LIBHARNESS_AGENT_PROFILE = opts.agentProfile;
117
+ }
118
+ // Unconditional so the default "github" is observable to the agent's
119
+ // active-tracker resolution, mirroring --agent-profile's env write above.
120
+ runtime.proc.env.LIBHARNESS_WORK_TRACKER = opts.workTracker;
121
+
122
+ // With a profile, the consult guidance rides the profile composer's
123
+ // amendment parameter; with no profile, a preset-append prompt carries
124
+ // the guidance as its only session-protocol fragment. Advisor off and no
125
+ // profile means no system prompt — today's behavior, unchanged.
126
+ let systemPrompt;
127
+ if (opts.agentProfile) {
128
+ systemPrompt = composeProfilePrompt(opts.agentProfile, {
129
+ profilesDir: resolve(opts.cwd, ".claude/agents"),
130
+ runtime,
131
+ ...(opts.advisorModel && {
132
+ amend: advisorGuidance(opts.advisorMaxUses),
133
+ }),
134
+ });
135
+ } else if (opts.advisorModel) {
136
+ systemPrompt = composeSystemPrompt({
137
+ role: "agent",
138
+ trailer: advisorGuidance(opts.advisorMaxUses),
139
+ runtime,
140
+ });
141
+ }
142
+
143
+ let advisor = null;
144
+ let recorder = null;
145
+ if (opts.advisorModel) {
146
+ const budget = createAdvisorBudget(opts.advisorMaxUses);
147
+ recorder = createTranscriptRecorder({ systemPrompt, redactor });
148
+ advisor = createAdvisor({
149
+ model: opts.advisorModel,
150
+ cwd: opts.cwd,
151
+ query,
152
+ recorder,
153
+ redactor,
154
+ runtime,
155
+ onLine: (line) => emitEnvelope("advisor", JSON.parse(line)),
156
+ });
157
+ const advTool = advisorTool({
158
+ from: "agent",
159
+ consult: (q) => advisor.consult(q),
160
+ emit: (event) => emitEnvelope("orchestrator", event),
161
+ budget,
162
+ model: opts.advisorModel,
163
+ });
164
+ // No allowlist push: in-process SDK MCP servers work under
165
+ // bypassPermissions without allowlist entries (loop-mode precedent).
166
+ mcpServers = {
167
+ ...mcpServers,
168
+ advisor: createSdkMcpServer({ name: "advisor", tools: [advTool] }),
169
+ };
170
+ }
171
+
172
+ const runner = createAgentRunner({
173
+ cwd: opts.cwd,
174
+ query,
175
+ output: devNull,
176
+ model: opts.agentModel,
177
+ maxTurns: opts.maxTurns,
178
+ allowedTools,
179
+ onLine: recorder
180
+ ? (line) => {
181
+ onLine(line);
182
+ recorder.recordMessage(line);
183
+ }
184
+ : onLine,
185
+ ...(recorder && { onPrompt: (text) => recorder.recordPrompt(text) }),
186
+ settingSources: ["project"],
187
+ systemPrompt,
188
+ taskAmend: opts.taskAmend,
189
+ mcpServers,
190
+ redactor,
191
+ runtime,
192
+ });
193
+
194
+ return { runner, advisor };
195
+ }
196
+
46
197
  /**
47
198
  * Run command — execute a single agent via the Claude Agent SDK.
48
199
  *
@@ -53,18 +204,7 @@ export function parseRunOptions(values, runtime) {
53
204
  */
54
205
  export async function runRunCommand(ctx) {
55
206
  const runtime = ctx.deps.runtime;
56
- const {
57
- taskContent,
58
- taskAmend,
59
- cwd,
60
- agentModel,
61
- maxTurns,
62
- outputPath,
63
- agentProfile,
64
- workTracker,
65
- allowedTools,
66
- mcpServer,
67
- } = parseRunOptions(ctx.options, runtime);
207
+ const opts = parseRunOptions(ctx.options, runtime);
68
208
 
69
209
  // Build the redactor as the first observable side-effect after option
70
210
  // parsing — the env snapshot must freeze BEFORE any in-process
@@ -73,8 +213,8 @@ export async function runRunCommand(ctx) {
73
213
 
74
214
  // When --output is specified, stream text to stdout while writing NDJSON to file.
75
215
  // Otherwise, write NDJSON directly to stdout (backwards-compatible).
76
- const fileStream = outputPath
77
- ? runtime.fs.createWriteStream(outputPath)
216
+ const fileStream = opts.outputPath
217
+ ? runtime.fs.createWriteStream(opts.outputPath)
78
218
  : null;
79
219
  const output = fileStream
80
220
  ? createTeeWriter({
@@ -86,62 +226,17 @@ export async function runRunCommand(ctx) {
86
226
  : runtime.proc.stdout;
87
227
 
88
228
  const counter = new SequenceCounter();
89
- const devNull = new Writable({
90
- write(_chunk, _enc, cb) {
91
- cb();
92
- },
93
- });
94
- const onLine = (line) => {
95
- const event = JSON.parse(line);
96
- const tagged = { source: "agent", seq: counter.next(), event };
97
- output.write(JSON.stringify(redactor.redactValue(tagged)) + "\n");
98
- };
99
-
100
- let mcpServers = null;
101
- if (mcpServer) {
102
- const mcpConfig = await createServiceConfig("mcp");
103
- mcpServers = {
104
- [mcpServer]: {
105
- type: "http",
106
- url: mcpConfig.url,
107
- headers: { Authorization: `Bearer ${mcpConfig.mcpToken()}` },
108
- },
109
- };
110
- allowedTools.push(`mcp__${mcpServer}__*`);
111
- }
112
-
113
- if (agentProfile) {
114
- runtime.proc.env.LIBHARNESS_AGENT_PROFILE = agentProfile;
115
- }
116
- // Unconditional so the default "github" is observable to the agent's
117
- // active-tracker resolution, mirroring --agent-profile's env write above.
118
- runtime.proc.env.LIBHARNESS_WORK_TRACKER = workTracker;
119
-
120
- const systemPrompt = agentProfile
121
- ? composeProfilePrompt(agentProfile, {
122
- profilesDir: resolve(cwd, ".claude/agents"),
123
- runtime,
124
- })
125
- : undefined;
126
-
127
229
  const { query } = await import("@anthropic-ai/claude-agent-sdk");
128
- const runner = createAgentRunner({
129
- cwd,
130
- query,
131
- output: devNull,
132
- model: agentModel,
133
- maxTurns,
134
- allowedTools,
135
- onLine,
136
- settingSources: ["project"],
137
- systemPrompt,
138
- taskAmend,
139
- mcpServers,
230
+ const { runner } = await wireRunSession({
231
+ opts,
140
232
  redactor,
233
+ output,
234
+ counter,
235
+ query,
141
236
  runtime,
142
237
  });
143
238
 
144
- const result = await runner.run(taskContent);
239
+ const result = await runner.run(opts.taskContent);
145
240
 
146
241
  if (fileStream) {
147
242
  await new Promise((r) => output.end(r));
@@ -3,6 +3,7 @@ import { isoTimestamp } from "@forwardimpact/libutil";
3
3
  import { createSupervisor } from "../supervisor.js";
4
4
  import { createRedactor } from "../redaction.js";
5
5
  import { createTeeWriter } from "../tee-writer.js";
6
+ import { parseAdvisorOptions } from "./advisor-flags.js";
6
7
  import { resolveTaskContent } from "./task-input.js";
7
8
  import { resolveWorkTracker } from "./work-tracker.js";
8
9
  import { createServiceConfig } from "@forwardimpact/libconfig";
@@ -52,6 +53,7 @@ export async function parseSuperviseOptions(values, runtime) {
52
53
  ? supervisorAllowedToolsRaw.split(",")
53
54
  : undefined,
54
55
  mcpServer: values["mcp-server"] || undefined,
56
+ ...parseAdvisorOptions(values),
55
57
  };
56
58
  }
57
59
 
@@ -125,6 +127,8 @@ export async function runSuperviseCommand(ctx) {
125
127
  agentMcpServers,
126
128
  redactor,
127
129
  runtime,
130
+ advisorModel: opts.advisorModel,
131
+ advisorMaxUses: opts.advisorMaxUses,
128
132
  });
129
133
 
130
134
  const result = await supervisor.run(opts.taskContent);
@@ -107,7 +107,7 @@ const ACKNOWLEDGE_DESC =
107
107
  "Acknowledge an Ask before starting work. Posts a visible comment on the thread. Does not discharge the Ask — you still owe an Answer.";
108
108
 
109
109
  /** Discuss-mode agent tool server. */
110
- export function createDiscussAgentToolServer(ctx, { from }) {
110
+ export function createDiscussAgentToolServer(ctx, { from, extraTools = [] }) {
111
111
  return orchestrationServer([
112
112
  ...baseTools(ctx, { from, defaultTo: "lead", broadcast: true }),
113
113
  requestForCommentTool(ctx),
@@ -133,6 +133,7 @@ export function createDiscussAgentToolServer(ctx, { from }) {
133
133
  return { content: [{ type: "text", text: "Acknowledged." }] };
134
134
  },
135
135
  ),
136
+ ...extraTools,
136
137
  ]);
137
138
  }
138
139