@forwardimpact/libharness 1.3.0 → 1.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -27,6 +27,20 @@ const LEAD_OPTIONS = {
27
27
  },
28
28
  };
29
29
 
30
+ // Advisor consult flags, shared by all four session modes.
31
+ const ADVISOR_OPTIONS = {
32
+ "advisor-model": {
33
+ type: "string",
34
+ description:
35
+ "Claude model for advisor consults; omitting the flag disables the Advisor tool (default: off)",
36
+ },
37
+ "advisor-max-uses": {
38
+ type: "string",
39
+ description:
40
+ "Session-wide consult budget shared by all participants (default: 3; requires --advisor-model)",
41
+ },
42
+ };
43
+
30
44
  // Shared task-input flags: --task-file (path), --task-text (inline), and
31
45
  // --task-event (path to native GitHub event JSON composed into a task via
32
46
  // libharness/src/events/github.js). Exactly one of the three is required.
@@ -94,6 +108,7 @@ const definition = {
94
108
  description:
95
109
  "Connect to the MCP service (e.g. --mcp-server=guide); adds mcp__<name>__* to allowed tools",
96
110
  },
111
+ ...ADVISOR_OPTIONS,
97
112
  },
98
113
  },
99
114
  {
@@ -143,6 +158,7 @@ const definition = {
143
158
  description:
144
159
  "Connect to the MCP service (e.g. --mcp-server=guide); adds mcp__<name>__* to allowed tools",
145
160
  },
161
+ ...ADVISOR_OPTIONS,
146
162
  },
147
163
  },
148
164
  {
@@ -185,6 +201,7 @@ const definition = {
185
201
  description:
186
202
  "Active work-item tracker (github|filesystem, default: github)",
187
203
  },
204
+ ...ADVISOR_OPTIONS,
188
205
  },
189
206
  },
190
207
  {
@@ -231,6 +248,7 @@ const definition = {
231
248
  description:
232
249
  "Active work-item tracker (github|filesystem, default: github)",
233
250
  },
251
+ ...ADVISOR_OPTIONS,
234
252
  },
235
253
  },
236
254
  {
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@forwardimpact/libharness",
3
- "version": "1.3.0",
3
+ "version": "1.4.0",
4
4
  "description": "Autonomous agent team harness — coordinate a lead and participant agents in one async session, with eval, benchmark, and trace tooling to prove the changes worked.",
5
5
  "keywords": [
6
6
  "orchestration",
package/src/advisor.js ADDED
@@ -0,0 +1,218 @@
1
+ /**
2
+ * Advisor — the judge's mid-loop sibling: a solo, tool-restricted, one-shot
3
+ * `AgentRunner` session on a stronger model whose final text is the advice.
4
+ * Each consult forwards the caller's whole recorded context (system prompt,
5
+ * delivered prompts, transcript so far) plus a focused question; the advisor
6
+ * can inspect files read-only but holds no write, execute, subagent, or
7
+ * orchestration tools and never appears on the message bus.
8
+ *
9
+ * Consults are stateless — one fresh session per call, re-reading the
10
+ * caller's context as it stands — and fail-open: timeout, error, and abort
11
+ * all resolve to an in-band `{unavailable}` result so the caller's session
12
+ * never stalls or crashes on a consult.
13
+ *
14
+ * Follows OO+DI: factory function, tests inject a fake `query`.
15
+ */
16
+
17
+ import { Writable } from "node:stream";
18
+
19
+ import { createAgentRunner } from "./agent-runner.js";
20
+ import { composeSystemPrompt } from "./profile-prompt.js";
21
+
22
+ /**
23
+ * System-prompt trailer for the advisor session. Fixes the response
24
+ * contract (spec criterion "Advice is bounded"): assessment,
25
+ * recommendation, unsolicited findings, with a stated length ceiling.
26
+ */
27
+ export const ADVISOR_SYSTEM_PROMPT =
28
+ "You are a consulted specialist, not a worker. " +
29
+ "Another agent paused its work to ask you one question; its full session context and the question are in the task. " +
30
+ "You may Read, Glob, and Grep the files the transcript names to ground your advice; never modify anything. " +
31
+ "Respond in one turn of prose — your final text is delivered to the caller verbatim. " +
32
+ "Structure the response as: assessment (what you see), recommendation (what to do and why), and unsolicited findings (anything important the caller did not ask about). " +
33
+ "Keep the whole response to at most three short paragraphs. " +
34
+ "Do not ask follow-up questions — the caller cannot reply.";
35
+
36
+ /**
37
+ * Consult-guidance fragment for caller system prompts, present only when
38
+ * the session runs with an advisor model. Steers the caller's judgment; it
39
+ * mandates nothing.
40
+ * @param {number} maxUses - The session-wide consult budget.
41
+ * @returns {string}
42
+ */
43
+ export function advisorGuidance(maxUses) {
44
+ return (
45
+ "An `Advisor` tool is available: one focused question per call, answered by a stronger model that sees your full session context. " +
46
+ "A consult pays off at hard decision points — architectural forks, unclear root causes, trade-offs you cannot rank — and early, before work builds on an unvalidated assumption. " +
47
+ "It does not pay off for routine reads, writes, or searches. " +
48
+ `The session-wide budget is ${maxUses} consult${maxUses === 1 ? "" : "s"}, shared across all participants. ` +
49
+ "Consulting is your judgment, never mandatory."
50
+ );
51
+ }
52
+
53
+ /**
54
+ * Create the session-wide consult budget, shared by every caller's tool
55
+ * handler. Enforced in code by the tool handler, not in the prompt.
56
+ * @param {number} maxUses
57
+ * @returns {{maxUses: number, used: number}}
58
+ */
59
+ export function createAdvisorBudget(maxUses) {
60
+ return { maxUses, used: 0 };
61
+ }
62
+
63
+ /**
64
+ * Fold the consult guidance into an existing run-specific amendment when
65
+ * the advisor is enabled (budget present); return the amendment unchanged
66
+ * otherwise, so advisor-off prompts stay byte-identical.
67
+ * @param {string|undefined} amend - The existing amendment, if any.
68
+ * @param {{maxUses: number}|null} budget
69
+ * @returns {string|undefined}
70
+ */
71
+ export function withAdvisorGuidance(amend, budget) {
72
+ if (!budget) return amend;
73
+ return [amend, advisorGuidance(budget.maxUses)].filter(Boolean).join("\n\n");
74
+ }
75
+
76
+ /** Consult timeout — generous for a read-a-few-files-and-answer session, and the universal guard in modes with no stop path. */
77
+ export const DEFAULT_CONSULT_TIMEOUT_MS = 300_000;
78
+
79
+ const ADVISOR_ALLOWED_TOOLS = ["Read", "Glob", "Grep"];
80
+
81
+ // Under the harness's always-on bypassPermissions, `allowedTools` alone is
82
+ // not structural — `disallowedTools` is what removes tools from the model's
83
+ // context, the same treatment the lead runners get. The advisor's contract
84
+ // is stricter than the leads' (read-only inspection, nothing else), so the
85
+ // list also removes the write-capable and non-inspection built-ins the
86
+ // lead convention leaves in.
87
+ const ADVISOR_DISALLOWED_TOOLS = [
88
+ "Bash",
89
+ "Write",
90
+ "Edit",
91
+ "NotebookEdit",
92
+ "Agent",
93
+ "Task",
94
+ "TaskOutput",
95
+ "TaskStop",
96
+ "TodoWrite",
97
+ "WebFetch",
98
+ "WebSearch",
99
+ ];
100
+
101
+ const devNull = new Writable({
102
+ write(_chunk, _enc, cb) {
103
+ cb();
104
+ },
105
+ });
106
+
107
+ /**
108
+ * Create a per-caller advisor closed over that caller's transcript
109
+ * recorder.
110
+ *
111
+ * @param {object} deps
112
+ * @param {string} deps.model - Advisor model id.
113
+ * @param {string} deps.cwd - The caller's working directory, so read-only inspection sees the caller's files.
114
+ * @param {function} deps.query - SDK query function (injected for testing).
115
+ * @param {{render: () => string}} deps.recorder - The caller's transcript recorder.
116
+ * @param {import("./redaction.js").Redactor} deps.redactor
117
+ * @param {import("@forwardimpact/libutil/runtime").Runtime} deps.runtime - Clock surface for timeout and duration.
118
+ * @param {function} deps.onLine - Re-emitter for the advisor session's NDJSON lines (tagged `source: "advisor"` by the caller).
119
+ * @param {number} [deps.maxTurns] - Default 5 — single-digit per the spec criterion.
120
+ * @param {number} [deps.timeoutMs] - Default `DEFAULT_CONSULT_TIMEOUT_MS`.
121
+ * @returns {{consult: (question: string) => Promise<{advice?: string, unavailable?: boolean, reason?: string, durationMs: number}>, abort: () => void}}
122
+ */
123
+ export function createAdvisor({
124
+ model,
125
+ cwd,
126
+ query,
127
+ recorder,
128
+ redactor,
129
+ runtime,
130
+ onLine,
131
+ maxTurns,
132
+ timeoutMs,
133
+ }) {
134
+ if (!model) throw new Error("model is required");
135
+ if (!cwd) throw new Error("cwd is required");
136
+ if (!query) throw new Error("query is required");
137
+ if (!recorder) throw new Error("recorder is required");
138
+ if (!redactor) throw new Error("redactor is required");
139
+ if (!runtime) throw new Error("runtime is required");
140
+ if (!onLine) throw new Error("onLine is required");
141
+ const resolvedMaxTurns = maxTurns ?? 5;
142
+ const resolvedTimeoutMs = timeoutMs ?? DEFAULT_CONSULT_TIMEOUT_MS;
143
+
144
+ /** @type {import("./agent-runner.js").AgentRunner|null} */
145
+ let currentRunner = null;
146
+
147
+ return {
148
+ /**
149
+ * Run one fresh advisor session over the caller's context as it
150
+ * stands plus the question. Never rejects — every failure shape
151
+ * resolves to `{unavailable, reason}` (fail-open).
152
+ * @param {string} question
153
+ */
154
+ async consult(question) {
155
+ const started = runtime.clock.now();
156
+ const runner = createAgentRunner({
157
+ cwd,
158
+ query,
159
+ output: devNull,
160
+ model,
161
+ maxTurns: resolvedMaxTurns,
162
+ allowedTools: ADVISOR_ALLOWED_TOOLS,
163
+ disallowedTools: ADVISOR_DISALLOWED_TOOLS,
164
+ onLine,
165
+ settingSources: ["project"],
166
+ systemPrompt: composeSystemPrompt({
167
+ role: "agent",
168
+ trailer: ADVISOR_SYSTEM_PROMPT,
169
+ runtime,
170
+ }),
171
+ redactor,
172
+ });
173
+ currentRunner = runner;
174
+ const task = `${recorder.render()}\n\n<consult_question>\n${question}\n</consult_question>`;
175
+ const timer = runtime.clock.setTimeout(
176
+ () => runner.currentAbortController?.abort(),
177
+ resolvedTimeoutMs,
178
+ );
179
+ try {
180
+ const result = await runner.run(task);
181
+ const durationMs = runtime.clock.now() - started;
182
+ if (result.aborted) {
183
+ return {
184
+ unavailable: true,
185
+ reason: "timed out or aborted",
186
+ durationMs,
187
+ };
188
+ }
189
+ if (!result.success) {
190
+ return {
191
+ unavailable: true,
192
+ reason: result.error?.message ?? "advisor session failed",
193
+ durationMs,
194
+ };
195
+ }
196
+ return { advice: result.text, durationMs };
197
+ } catch (err) {
198
+ return {
199
+ unavailable: true,
200
+ reason: err?.message ?? "advisor session failed",
201
+ durationMs: runtime.clock.now() - started,
202
+ };
203
+ } finally {
204
+ runtime.clock.clearTimeout(timer);
205
+ currentRunner = null;
206
+ }
207
+ },
208
+
209
+ /**
210
+ * Abort the in-flight consult, if any. A consult is a blocking tool
211
+ * call, so one caller cannot overlap its own consults; advisors are
212
+ * per-caller, so at most one runner is ever tracked.
213
+ */
214
+ abort() {
215
+ currentRunner?.currentAbortController?.abort();
216
+ },
217
+ };
218
+ }
@@ -53,6 +53,7 @@ export class AgentRunner {
53
53
  * @param {number} [deps.maxTurns] - Maximum agentic turns; 0 means unlimited
54
54
  * @param {string[]} [deps.allowedTools] - Tools the agent may use
55
55
  * @param {function} [deps.onLine] - Callback invoked with each NDJSON line as it's produced
56
+ * @param {function} [deps.onPrompt] - Callback invoked with the effective (amend-applied) prompt of each run/resume
56
57
  * @param {string[]} [deps.settingSources] - SDK setting sources (e.g. ['project'] to load CLAUDE.md)
57
58
  * @param {string|object} [deps.systemPrompt] - SDK system prompt (string replaces default; {type:'preset', preset:'claude_code', append} appends)
58
59
  * @param {string[]} [deps.disallowedTools] - Tools to explicitly remove from the model's context
@@ -80,6 +81,9 @@ export class AgentRunner {
80
81
  this.maxTurns = deps.maxTurns ?? 50;
81
82
  this.allowedTools = deps.allowedTools ?? DEFAULT_ALLOWED_TOOLS;
82
83
  this.onLine = deps.onLine ?? null;
84
+ // Optional; read only through a truthy guard in run()/resume(), so an
85
+ // absent value stays undefined rather than needing a `?? null` default.
86
+ this.onPrompt = deps.onPrompt;
83
87
  this.settingSources = deps.settingSources ?? [];
84
88
  this.systemPrompt = deps.systemPrompt ?? null;
85
89
  this.disallowedTools = deps.disallowedTools ?? [];
@@ -106,6 +110,7 @@ export class AgentRunner {
106
110
  ? `${task}\n\n${this.taskAmend}`
107
111
  : this.taskAmend
108
112
  : task;
113
+ if (this.onPrompt) this.onPrompt(effectiveTask);
109
114
  try {
110
115
  const iterator = this.query({
111
116
  prompt: effectiveTask,
@@ -125,6 +130,7 @@ export class AgentRunner {
125
130
  async resume(prompt) {
126
131
  const abortController = new AbortController();
127
132
  this.currentAbortController = abortController;
133
+ if (this.onPrompt) this.onPrompt(prompt);
128
134
  try {
129
135
  const iterator = this.query({
130
136
  prompt,
@@ -143,11 +143,19 @@ export function renderTextReport(report, kValues) {
143
143
  }
144
144
 
145
145
  // ---------------------------------------------------------------------------
146
- // Compact report (legacy path)
146
+ // Compact report — status line + pass@k table, no per-task detail. Selected by
147
+ // `report --detail=compact` (aggregate without `includeRuns`); the per-shard
148
+ // summary uses it so a sharded run stays short while the merge job renders the
149
+ // full report over the combined ledger.
147
150
  // ---------------------------------------------------------------------------
148
151
 
149
152
  function renderCompactReport(report, kValues) {
153
+ const { totals } = report;
154
+ const passing = report.tasks.filter((t) => t.c > 0 && t.c === t.n).length;
155
+ const icon = statusIcon(passing === totals.tasks);
150
156
  const lines = [
157
+ `${icon} **${passing}/${totals.tasks} tasks passing** | ${totals.runs} runs${totals.skipped ? ` | ${totals.skipped} skipped` : ""}`,
158
+ "",
151
159
  renderPassAtKTable(report, kValues),
152
160
  "",
153
161
  renderTotalsLine(report),
@@ -0,0 +1,28 @@
1
+ /**
2
+ * Shared advisor-flag parsing for the four session-mode commands. The two
3
+ * flags are identical everywhere: `--advisor-model` (no default — absent
4
+ * means the Advisor tool is not offered) and `--advisor-max-uses`
5
+ * (default 3), which is a usage error without the model flag.
6
+ */
7
+
8
+ /**
9
+ * Parse `--advisor-model` / `--advisor-max-uses` from parsed option values.
10
+ * A malformed max-uses is a usage error, not a silent fallback: NaN would
11
+ * make the budget check (`used >= maxUses`) permanently false and disable
12
+ * the code-enforced cap the flag exists to guarantee.
13
+ * @param {object} values - Parsed option values from cli.parse()
14
+ * @returns {{advisorModel: string|undefined, advisorMaxUses: number}}
15
+ */
16
+ export function parseAdvisorOptions(values) {
17
+ if (values["advisor-max-uses"] && !values["advisor-model"]) {
18
+ throw new Error("--advisor-max-uses requires --advisor-model");
19
+ }
20
+ const advisorMaxUses = parseInt(values["advisor-max-uses"] || "3", 10);
21
+ if (Number.isNaN(advisorMaxUses) || advisorMaxUses < 1) {
22
+ throw new Error("--advisor-max-uses must be a positive integer");
23
+ }
24
+ return {
25
+ advisorModel: values["advisor-model"] || undefined,
26
+ advisorMaxUses,
27
+ };
28
+ }
@@ -140,6 +140,11 @@ export const definition = {
140
140
  type: "string",
141
141
  description: "Output format (json|text, default: json)",
142
142
  },
143
+ detail: {
144
+ type: "string",
145
+ description:
146
+ "Text report verbosity (full|compact, default: full). compact omits per-task detail — useful for sharded run summaries.",
147
+ },
143
148
  },
144
149
  },
145
150
  ],
@@ -1,7 +1,9 @@
1
1
  /**
2
2
  * `fit-benchmark report` — aggregate `results.jsonl` into pass@k via the
3
3
  * OpenAI HumanEval estimator. Output is JSON by default; pass --format=text
4
- * to render a markdown table.
4
+ * to render a markdown table. --detail=compact drops the per-task detail
5
+ * sections so a sharded run's per-shard summary stays short (the merge job
6
+ * renders the full report over the combined ledger).
5
7
  */
6
8
 
7
9
  import { resolve } from "node:path";
@@ -35,11 +37,19 @@ export async function runBenchmarkReportCommand(ctx) {
35
37
  if (format !== "json" && format !== "text") {
36
38
  return { ok: false, code: 1, error: "--format must be 'json' or 'text'" };
37
39
  }
40
+ const detail = values.detail ?? "full";
41
+ if (detail !== "full" && detail !== "compact") {
42
+ return {
43
+ ok: false,
44
+ code: 1,
45
+ error: "--detail must be 'full' or 'compact'",
46
+ };
47
+ }
38
48
 
39
49
  const report = await aggregate({
40
50
  inputDir: resolve(inputDir),
41
51
  kValues,
42
- includeRuns: format === "text",
52
+ includeRuns: format === "text" && detail === "full",
43
53
  runtime,
44
54
  });
45
55
  if (format === "text") {
@@ -3,6 +3,7 @@ import { isoTimestamp } from "@forwardimpact/libutil";
3
3
  import { createDiscusser } from "../discusser.js";
4
4
  import { createRedactor } from "../redaction.js";
5
5
  import { createTeeWriter } from "../tee-writer.js";
6
+ import { parseAdvisorOptions } from "./advisor-flags.js";
6
7
  import { resolveTaskContent } from "./task-input.js";
7
8
  import { resolveWorkTracker } from "./work-tracker.js";
8
9
  import { AGENT_MODEL, LEAD_MODEL } from "@forwardimpact/libutil/models";
@@ -67,6 +68,7 @@ export function parseDiscussOptions(values, runtime) {
67
68
  callbackUrl: runtime.proc.env.CALLBACK_URL ?? null,
68
69
  inboxUrl: runtime.proc.env.INBOX_URL ?? null,
69
70
  correlationId: runtime.proc.env.CORRELATION_ID ?? null,
71
+ ...parseAdvisorOptions(values),
70
72
  };
71
73
  }
72
74
 
@@ -121,6 +123,8 @@ export async function runDiscussCommand(ctx) {
121
123
  inboxUrl: opts.inboxUrl,
122
124
  correlationId: opts.correlationId,
123
125
  runtime,
126
+ advisorModel: opts.advisorModel,
127
+ advisorMaxUses: opts.advisorMaxUses,
124
128
  });
125
129
 
126
130
  const result = await discusser.run(opts.taskContent);
@@ -3,6 +3,7 @@ import { isoTimestamp } from "@forwardimpact/libutil";
3
3
  import { createFacilitator } from "../facilitator.js";
4
4
  import { createRedactor } from "../redaction.js";
5
5
  import { createTeeWriter } from "../tee-writer.js";
6
+ import { parseAdvisorOptions } from "./advisor-flags.js";
6
7
  import { resolveTaskContent } from "./task-input.js";
7
8
  import { resolveWorkTracker } from "./work-tracker.js";
8
9
  import { AGENT_MODEL, LEAD_MODEL } from "@forwardimpact/libutil/models";
@@ -60,6 +61,7 @@ export function parseFacilitateOptions(values, runtime) {
60
61
  outputPath: values.output,
61
62
  facilitatorProfile: values["lead-profile"] || undefined,
62
63
  workTracker: resolveWorkTracker(values, runtime?.proc?.env),
64
+ ...parseAdvisorOptions(values),
63
65
  };
64
66
  }
65
67
 
@@ -112,6 +114,8 @@ export async function runFacilitateCommand(ctx) {
112
114
  taskAmend: opts.taskAmend,
113
115
  redactor,
114
116
  runtime,
117
+ advisorModel: opts.advisorModel,
118
+ advisorMaxUses: opts.advisorMaxUses,
115
119
  });
116
120
 
117
121
  const result = await facilitator.run(opts.taskContent);