@forwardimpact/libharness 1.1.0 → 1.2.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -2,8 +2,9 @@
2
2
 
3
3
  <!-- BEGIN:description — Do not edit. Generated from package.json. -->
4
4
 
5
- Agent evaluation framework — prove whether agent changes improved outcomes with
6
- reproducible evidence.
5
+ Autonomous agent team harness — coordinate a lead and participant agents in one
6
+ async session, with eval, benchmark, and trace tooling to prove the changes
7
+ worked.
7
8
 
8
9
  <!-- END:description -->
9
10
 
@@ -53,7 +54,8 @@ Every Ask returns immediately and registers a pending entry keyed by an
53
54
  lead. Answer's `askId` is optional — the handler is forgiving:
54
55
 
55
56
  - **Provided + matches an ask owed by the caller** → routes to that asker.
56
- - **Provided but unknown or wrong addressee** → `isError` with a pointed message.
57
+ - **Provided but unknown or wrong addressee** → `isError` with a pointed
58
+ message.
57
59
  - **Omitted + exactly one ask owed to the caller** → auto-picks it.
58
60
  - **Omitted + 0 or many asks owed** → broadcasts as Announce.
59
61
 
@@ -209,6 +211,17 @@ lists the `Edit()` rules that were tried.
209
211
 
210
212
  ## Documentation
211
213
 
212
- - [Agent Evaluations Guide](https://www.forwardimpact.team/docs/libraries/agent-evaluations/index.md) — how to run an eval and read its trace.
213
- - [Agent Collaboration Guide](https://www.forwardimpact.team/docs/libraries/agent-collaboration/index.md) — supervise / facilitate / discuss in depth.
214
- - [Trace Analysis Guide](https://www.forwardimpact.team/docs/libraries/trace-analysis/index.md) — analysing NDJSON traces with `fit-trace`.
214
+ - [Coordinate an Agent Team](https://www.forwardimpact.team/docs/libraries/coordinate-team/index.md)
215
+ — run a lead and N participant agents in one async session (supervise /
216
+ facilitate / discuss) with Ask/Answer/Announce and a single NDJSON trace.
217
+ - [Run an Eval](https://www.forwardimpact.team/docs/libraries/prove-changes/run-eval/index.md)
218
+ — author a judge profile, run an eval locally, wire it into CI, and inspect
219
+ the resulting trace.
220
+ - [Prove Agent Changes](https://www.forwardimpact.team/docs/libraries/prove-changes/index.md)
221
+ — end-to-end workflow from dataset generation through evaluation to trace
222
+ analysis, including multi-agent collaboration sessions.
223
+ - [Analyze Traces](https://www.forwardimpact.team/docs/libraries/prove-changes/trace-analysis/index.md)
224
+ — read the NDJSON traces produced by `fit-harness` with `fit-trace`.
225
+ - [Agent Teams](https://www.forwardimpact.team/docs/products/agent-teams/index.md)
226
+ — author the profiles consumed by `--agent-profile`, `--lead-profile`, and
227
+ `--agent-profiles`.
@@ -295,6 +295,12 @@ const definition = {
295
295
  "fit-harness output --format=text < trace.ndjson",
296
296
  ],
297
297
  documentation: [
298
+ {
299
+ title: "Coordinate an Agent Team",
300
+ url: "https://www.forwardimpact.team/docs/libraries/coordinate-team/index.md",
301
+ description:
302
+ "Run a lead and N participant agents in one async session — supervise, facilitate, or discuss — with Ask/Answer/Announce and a single NDJSON trace.",
303
+ },
298
304
  {
299
305
  title: "Run an Eval",
300
306
  url: "https://www.forwardimpact.team/docs/libraries/prove-changes/run-eval/index.md",
package/bin/fit-trace.js CHANGED
@@ -350,7 +350,7 @@ const definition = {
350
350
  options: {
351
351
  mode: {
352
352
  type: "string",
353
- description: "Execution mode: run, supervise, or facilitate",
353
+ description: "Execution mode: run, supervise, facilitate, or discuss",
354
354
  },
355
355
  case: {
356
356
  type: "string",
package/package.json CHANGED
@@ -1,13 +1,14 @@
1
1
  {
2
2
  "name": "@forwardimpact/libharness",
3
- "version": "1.1.0",
4
- "description": "Agent evaluation framework — prove whether agent changes improved outcomes with reproducible evidence.",
3
+ "version": "1.2.1",
4
+ "description": "Autonomous agent team harness — coordinate a lead and participant agents in one async session, with eval, benchmark, and trace tooling to prove the changes worked.",
5
5
  "keywords": [
6
+ "orchestration",
7
+ "supervisor",
8
+ "facilitator",
6
9
  "eval",
7
- "agent",
8
10
  "trace",
9
- "claude-code",
10
- "supervisor"
11
+ "agent"
11
12
  ],
12
13
  "homepage": "https://www.forwardimpact.team",
13
14
  "repository": {
@@ -18,12 +19,20 @@
18
19
  "license": "Apache-2.0",
19
20
  "author": "D. Olsson <hi@senzilla.io>",
20
21
  "jobs": [
22
+ {
23
+ "user": "Platform Builders",
24
+ "goal": "Coordinate an Agent Team",
25
+ "trigger": "Coordinating several agents in one session means hand-rolling message passing, turn-taking, and termination, and every orchestration script re-solves it.",
26
+ "bigHire": "coordinate a lead and participant agents over async messages in one session.",
27
+ "littleHire": "run a supervised pair, a facilitated meeting, or a multi-agent discussion without writing the message bus and turn loop.",
28
+ "competesWith": "manual orchestration scripts; hand-rolled message passing and turn-taking; sequential single-agent calls; running one agent at a time"
29
+ },
21
30
  {
22
31
  "user": "Platform Builders",
23
32
  "goal": "Prove Agent Changes",
24
33
  "trigger": "An eval passes locally but fails in CI and the only output is 'assertion failed.'",
25
34
  "bigHire": "prove whether agent changes improved outcomes with reproducible evidence.",
26
- "littleHire": "run an eval and get a trace that shows exactly what the agent did.",
35
+ "littleHire": "run an eval or benchmark and get a trace that shows exactly what the agent did.",
27
36
  "competesWith": "manual before/after comparison; trusting gut feeling over evidence; skipping evaluation entirely"
28
37
  }
29
38
  ],
@@ -7,6 +7,7 @@
7
7
  */
8
8
 
9
9
  import { AGENT_MODEL } from "@forwardimpact/libutil/models";
10
+ import { resolveClaudeCodeExecutable } from "./claude-code-executable.js";
10
11
 
11
12
  const DEFAULT_ALLOWED_TOOLS = ["Bash", "Read", "Glob", "Grep", "Write", "Edit"];
12
13
 
@@ -56,6 +57,10 @@ export class AgentRunner {
56
57
  * @param {string|object} [deps.systemPrompt] - SDK system prompt (string replaces default; {type:'preset', preset:'claude_code', append} appends)
57
58
  * @param {string[]} [deps.disallowedTools] - Tools to explicitly remove from the model's context
58
59
  * @param {Record<string, object>} [deps.mcpServers] - MCP server configs to pass to the SDK query
60
+ * @param {string} [deps.pathToClaudeCodeExecutable] - Absolute path to the
61
+ * native `claude` CLI the SDK should spawn. Set for compiled fit-* binaries,
62
+ * which can't self-resolve the SDK's platform optional dependency; omitted
63
+ * from source runs so the SDK resolves its own version-matched binary.
59
64
  * @param {object} deps.redactor
60
65
  * @param {import("@forwardimpact/libutil/runtime").Runtime} [deps.runtime] -
61
66
  * Ambient collaborators. Only `proc.env` is read (to record Skill
@@ -79,6 +84,9 @@ export class AgentRunner {
79
84
  this.systemPrompt = deps.systemPrompt ?? null;
80
85
  this.disallowedTools = deps.disallowedTools ?? [];
81
86
  this.mcpServers = deps.mcpServers ?? null;
87
+ // Optional; read only through a truthy guard in #callOptions, so an absent
88
+ // value stays undefined rather than needing a `?? null` default.
89
+ this.pathToClaudeCodeExecutable = deps.pathToClaudeCodeExecutable;
82
90
  this.taskAmend = deps.taskAmend ?? null;
83
91
  this.sessionId = null;
84
92
  /** @type {AbortController|null} */
@@ -158,6 +166,9 @@ export class AgentRunner {
158
166
  }),
159
167
  ...(this.systemPrompt && { systemPrompt: this.systemPrompt }),
160
168
  ...(this.mcpServers && { mcpServers: this.mcpServers }),
169
+ ...(this.pathToClaudeCodeExecutable && {
170
+ pathToClaudeCodeExecutable: this.pathToClaudeCodeExecutable,
171
+ }),
161
172
  };
162
173
  }
163
174
 
@@ -250,7 +261,14 @@ export class AgentRunner {
250
261
  }
251
262
  }
252
263
 
253
- /** Factory function — wires real dependencies. */
264
+ /**
265
+ * Factory function — wires real dependencies. Resolves the native `claude`
266
+ * executable for compiled fit-* binaries so the SDK doesn't fail to find its
267
+ * own platform optional dependency; an explicit `deps` value overrides it.
268
+ */
254
269
  export function createAgentRunner(deps) {
255
- return new AgentRunner(deps);
270
+ return new AgentRunner({
271
+ pathToClaudeCodeExecutable: resolveClaudeCodeExecutable(),
272
+ ...deps,
273
+ });
256
274
  }
@@ -0,0 +1,44 @@
1
+ /**
2
+ * Resolve the Claude Code CLI the Agent SDK should spawn.
3
+ *
4
+ * `query()` spawns a native `claude` binary that the SDK resolves from its own
5
+ * platform-specific optional dependency (`@anthropic-ai/claude-agent-sdk-<platform>`).
6
+ * `bun build --compile` bundles the SDK's JavaScript but not that separate
7
+ * native package — it is not part of the import graph — so a compiled fit-*
8
+ * binary can't self-resolve it and `query()` throws "Native CLI binary for
9
+ * <platform> not found".
10
+ *
11
+ * In a compiled binary we point the SDK at the standalone `claude` on PATH,
12
+ * installed beside fit-harness by the bootstrap action's `fit-install.sh`.
13
+ * Running from source keeps `node_modules`, where the SDK resolves its own
14
+ * version-matched binary, so there we return undefined and defer to the SDK.
15
+ */
16
+ import { LIBCLI_IS_COMPILED } from "@forwardimpact/libcli";
17
+
18
+ /**
19
+ * @param {object} [deps]
20
+ * @param {(cmd: string) => string | null | undefined} [deps.which] -
21
+ * PATH resolver (injected for testing).
22
+ * @param {boolean} [deps.isCompiled] -
23
+ * Whether this is a `bun --compile` binary (injected for testing).
24
+ * @returns {string | undefined} absolute path to `claude`, or undefined to
25
+ * defer resolution to the SDK.
26
+ */
27
+ export function resolveClaudeCodeExecutable({
28
+ which = defaultWhich,
29
+ isCompiled = LIBCLI_IS_COMPILED,
30
+ } = {}) {
31
+ if (!isCompiled) return undefined;
32
+ return which("claude") ?? undefined;
33
+ }
34
+
35
+ /**
36
+ * A compiled fit-* binary runs under the Bun runtime, which exposes a
37
+ * synchronous PATH resolver. The `typeof` guard keeps this safe under Node too.
38
+ */
39
+ function defaultWhich(cmd) {
40
+ if (typeof Bun !== "undefined" && typeof Bun.which === "function") {
41
+ return Bun.which(cmd);
42
+ }
43
+ return null;
44
+ }
@@ -169,7 +169,7 @@ export const definition = {
169
169
  title: "Automate with GitHub Actions",
170
170
  url: "https://www.forwardimpact.team/docs/libraries/prove-changes/run-benchmark/ci-workflow/index.md",
171
171
  description:
172
- "Run benchmarks in CI with the forwardimpact/fit-benchmark action.",
172
+ "Run benchmarks in CI with the forwardimpact/benchmark action.",
173
173
  },
174
174
  ],
175
175
  };
@@ -29,9 +29,11 @@ export function parseDiscussOptions(values, runtime) {
29
29
  );
30
30
 
31
31
  const profilesRaw = values["agent-profiles"];
32
- const agentCwd = resolve(values["agent-cwd"] ?? ".");
32
+ // `||` (not `??`) so an empty-string flag from a CI forwarder falls back to
33
+ // the default rather than overriding it with "".
34
+ const agentCwd = resolve(values["agent-cwd"] || ".");
33
35
 
34
- const maxTurnsRaw = values["max-turns"] ?? "40";
36
+ const maxTurnsRaw = values["max-turns"] || "40";
35
37
  const maxTurns = maxTurnsRaw === "0" ? 0 : parseInt(maxTurnsRaw, 10);
36
38
 
37
39
  const agentConfigs = parseAgentProfiles(profilesRaw, agentCwd, maxTurns);
@@ -46,21 +48,21 @@ export function parseDiscussOptions(values, runtime) {
46
48
  }
47
49
  }
48
50
 
49
- const maxLeadTurnsRaw = values["max-lead-turns"] ?? "200";
51
+ const maxLeadTurnsRaw = values["max-lead-turns"] || "200";
50
52
  const maxLeadTurns = parseInt(maxLeadTurnsRaw, 10);
51
53
 
52
54
  return {
53
55
  taskContent,
54
56
  taskAmend,
55
57
  agentConfigs,
56
- leadProfile: values["lead-profile"] ?? undefined,
58
+ leadProfile: values["lead-profile"] || undefined,
57
59
  leadModel: values["lead-model"] || LEAD_MODEL,
58
60
  agentModel: values["agent-model"] || AGENT_MODEL,
59
61
  maxTurns,
60
62
  maxLeadTurns,
61
63
  outputPath: values.output,
62
64
  workTracker: resolveWorkTracker(values, runtime?.proc?.env),
63
- discussionId: values["discussion-id"] ?? null,
65
+ discussionId: values["discussion-id"] || null,
64
66
  resumeContext,
65
67
  callbackUrl: runtime.proc.env.CALLBACK_URL ?? null,
66
68
  inboxUrl: runtime.proc.env.INBOX_URL ?? null,
@@ -36,9 +36,11 @@ export function parseFacilitateOptions(values, runtime) {
36
36
 
37
37
  const profilesRaw = values["agent-profiles"];
38
38
  if (!profilesRaw) throw new Error("--agent-profiles is required");
39
- const agentCwd = resolve(values["agent-cwd"] ?? ".");
39
+ // `||` (not `??`) so an empty-string flag from a CI forwarder falls back to
40
+ // the default rather than overriding it with "".
41
+ const agentCwd = resolve(values["agent-cwd"] || ".");
40
42
 
41
- const maxTurnsRaw = values["max-turns"] ?? "20";
43
+ const maxTurnsRaw = values["max-turns"] || "20";
42
44
  const maxTurns = maxTurnsRaw === "0" ? 0 : parseInt(maxTurnsRaw, 10);
43
45
 
44
46
  // Thread --max-turns into each participant: without this, every facilitated
@@ -51,12 +53,12 @@ export function parseFacilitateOptions(values, runtime) {
51
53
  taskContent,
52
54
  taskAmend,
53
55
  agentConfigs,
54
- facilitatorCwd: resolve(values["facilitator-cwd"] ?? "."),
56
+ facilitatorCwd: resolve(values["facilitator-cwd"] || "."),
55
57
  agentModel: values["agent-model"] || AGENT_MODEL,
56
58
  facilitatorModel: values["lead-model"] || LEAD_MODEL,
57
59
  maxTurns,
58
60
  outputPath: values.output,
59
- facilitatorProfile: values["lead-profile"] ?? undefined,
61
+ facilitatorProfile: values["lead-profile"] || undefined,
60
62
  workTracker: resolveWorkTracker(values, runtime?.proc?.env),
61
63
  };
62
64
  }
@@ -22,22 +22,24 @@ export function parseRunOptions(values, runtime) {
22
22
  values,
23
23
  runtime,
24
24
  );
25
- const maxTurnsRaw = values["max-turns"] ?? "50";
25
+ // `||` (not `??`) so an empty-string flag from a CI forwarder falls back to
26
+ // the default, rather than overriding it with "".
27
+ const maxTurnsRaw = values["max-turns"] || "50";
26
28
 
27
29
  return {
28
30
  taskContent,
29
31
  taskAmend,
30
- cwd: resolve(values.cwd ?? "."),
32
+ cwd: resolve(values.cwd || "."),
31
33
  agentModel: values["agent-model"] || AGENT_MODEL,
32
34
  maxTurns: maxTurnsRaw === "0" ? 0 : parseInt(maxTurnsRaw, 10),
33
35
  outputPath: values.output,
34
- agentProfile: values["agent-profile"] ?? undefined,
36
+ agentProfile: values["agent-profile"] || undefined,
35
37
  workTracker: resolveWorkTracker(values, runtime?.proc?.env),
36
38
  allowedTools: (
37
- values["allowed-tools"] ??
39
+ values["allowed-tools"] ||
38
40
  "Bash,Read,Glob,Grep,Write,Edit,Agent,TodoWrite"
39
41
  ).split(","),
40
- mcpServer: values["mcp-server"] ?? undefined,
42
+ mcpServer: values["mcp-server"] || undefined,
41
43
  };
42
44
  }
43
45
 
@@ -21,35 +21,37 @@ export async function parseSuperviseOptions(values, runtime) {
21
21
  );
22
22
  const supervisorAllowedToolsRaw = values["supervisor-allowed-tools"];
23
23
 
24
+ // `||` (not `??`) throughout so an empty-string flag from a CI forwarder
25
+ // falls back to the default rather than overriding it with "".
24
26
  const tmpRoot = runtime.proc.env.TMPDIR ?? "/tmp";
25
27
  const agentCwd = resolve(
26
- values["agent-cwd"] ??
28
+ values["agent-cwd"] ||
27
29
  (await runtime.fs.mkdtemp(join(tmpRoot, "fit-harness-agent-"))),
28
30
  );
29
31
 
30
32
  return {
31
33
  taskContent,
32
34
  taskAmend,
33
- supervisorCwd: resolve(values["supervisor-cwd"] ?? "."),
35
+ supervisorCwd: resolve(values["supervisor-cwd"] || "."),
34
36
  agentCwd,
35
37
  agentModel: values["agent-model"] || AGENT_MODEL,
36
38
  supervisorModel: values["lead-model"] || LEAD_MODEL,
37
39
  maxTurns: (() => {
38
- const raw = values["max-turns"] ?? "200";
40
+ const raw = values["max-turns"] || "200";
39
41
  return raw === "0" ? 0 : parseInt(raw, 10);
40
42
  })(),
41
43
  outputPath: values.output,
42
- supervisorProfile: values["lead-profile"] ?? undefined,
43
- agentProfile: values["agent-profile"] ?? undefined,
44
+ supervisorProfile: values["lead-profile"] || undefined,
45
+ agentProfile: values["agent-profile"] || undefined,
44
46
  workTracker: resolveWorkTracker(values, runtime?.proc?.env),
45
47
  allowedTools: (
46
- values["allowed-tools"] ??
48
+ values["allowed-tools"] ||
47
49
  "Bash,Read,Glob,Grep,Write,Edit,Agent,TodoWrite"
48
50
  ).split(","),
49
51
  supervisorAllowedTools: supervisorAllowedToolsRaw
50
52
  ? supervisorAllowedToolsRaw.split(",")
51
53
  : undefined,
52
- mcpServer: values["mcp-server"] ?? undefined,
54
+ mcpServer: values["mcp-server"] || undefined,
53
55
  };
54
56
  }
55
57
 
@@ -367,9 +367,12 @@ export async function runStatsCommand(ctx) {
367
367
  */
368
368
  export async function runCostCommand(ctx) {
369
369
  const { runtime } = ctx.deps;
370
- const cost = computeTraceCost(
371
- runtime.fsSync.readFileSync(ctx.args.file, "utf8"),
372
- );
370
+ // Tolerate a missing/empty trace: a CI step reports cost under `always()`,
371
+ // so the trace may not exist (the run failed before producing one). Print
372
+ // nothing and exit 0 rather than throwing — the caller needs no `if [ -f ]`.
373
+ const file = ctx.args.file;
374
+ if (!file || !runtime.fsSync.existsSync(file)) return { ok: true };
375
+ const cost = computeTraceCost(runtime.fsSync.readFileSync(file, "utf8"));
373
376
  if (ctx.options.markdown) {
374
377
  runtime.proc.stdout.write(renderCostMarkdown(cost));
375
378
  } else {
@@ -512,9 +515,12 @@ export async function runSplitCommand(ctx) {
512
515
  const file = ctx.args.file;
513
516
  if (!file) return { ok: false, code: 1, error: "split: missing input file" };
514
517
 
518
+ // `discuss` has the same lead + N-participants shape as `facilitate`, and the
519
+ // splitter buckets purely by envelope `source` (mode-independent), so it is
520
+ // accepted alongside the structural modes — the CLI owns this, not callers.
515
521
  const mode = ctx.options.mode;
516
522
  if (!mode) return { ok: false, code: 1, error: "split: --mode is required" };
517
- if (!["run", "supervise", "facilitate"].includes(mode)) {
523
+ if (!["run", "supervise", "facilitate", "discuss"].includes(mode)) {
518
524
  return { ok: false, code: 1, error: `split: invalid --mode "${mode}"` };
519
525
  }
520
526
 
package/src/index.js CHANGED
@@ -11,6 +11,7 @@ export {
11
11
  pickTraceArtifact,
12
12
  } from "./trace-github.js";
13
13
  export { AgentRunner, createAgentRunner } from "./agent-runner.js";
14
+ export { resolveClaudeCodeExecutable } from "./claude-code-executable.js";
14
15
  export {
15
16
  composeProfilePrompt,
16
17
  composeLeadPrompt,