@forwardimpact/libharness 1.1.0 → 1.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -2,8 +2,9 @@
2
2
 
3
3
  <!-- BEGIN:description — Do not edit. Generated from package.json. -->
4
4
 
5
- Agent evaluation framework — prove whether agent changes improved outcomes with
6
- reproducible evidence.
5
+ Autonomous agent team harness — coordinate a lead and participant agents in one
6
+ async session, with eval, benchmark, and trace tooling to prove the changes
7
+ worked.
7
8
 
8
9
  <!-- END:description -->
9
10
 
@@ -53,7 +54,8 @@ Every Ask returns immediately and registers a pending entry keyed by an
53
54
  lead. Answer's `askId` is optional — the handler is forgiving:
54
55
 
55
56
  - **Provided + matches an ask owed by the caller** → routes to that asker.
56
- - **Provided but unknown or wrong addressee** → `isError` with a pointed message.
57
+ - **Provided but unknown or wrong addressee** → `isError` with a pointed
58
+ message.
57
59
  - **Omitted + exactly one ask owed to the caller** → auto-picks it.
58
60
  - **Omitted + 0 or many asks owed** → broadcasts as Announce.
59
61
 
@@ -209,6 +211,17 @@ lists the `Edit()` rules that were tried.
209
211
 
210
212
  ## Documentation
211
213
 
212
- - [Agent Evaluations Guide](https://www.forwardimpact.team/docs/libraries/agent-evaluations/index.md) — how to run an eval and read its trace.
213
- - [Agent Collaboration Guide](https://www.forwardimpact.team/docs/libraries/agent-collaboration/index.md) — supervise / facilitate / discuss in depth.
214
- - [Trace Analysis Guide](https://www.forwardimpact.team/docs/libraries/trace-analysis/index.md) — analysing NDJSON traces with `fit-trace`.
214
+ - [Coordinate an Agent Team](https://www.forwardimpact.team/docs/libraries/coordinate-team/index.md)
215
+ — run a lead and N participant agents in one async session (supervise /
216
+ facilitate / discuss) with Ask/Answer/Announce and a single NDJSON trace.
217
+ - [Run an Eval](https://www.forwardimpact.team/docs/libraries/prove-changes/run-eval/index.md)
218
+ — author a judge profile, run an eval locally, wire it into CI, and inspect
219
+ the resulting trace.
220
+ - [Prove Agent Changes](https://www.forwardimpact.team/docs/libraries/prove-changes/index.md)
221
+ — end-to-end workflow from dataset generation through evaluation to trace
222
+ analysis, including multi-agent collaboration sessions.
223
+ - [Analyze Traces](https://www.forwardimpact.team/docs/libraries/prove-changes/trace-analysis/index.md)
224
+ — read the NDJSON traces produced by `fit-harness` with `fit-trace`.
225
+ - [Agent Teams](https://www.forwardimpact.team/docs/products/agent-teams/index.md)
226
+ — author the profiles consumed by `--agent-profile`, `--lead-profile`, and
227
+ `--agent-profiles`.
@@ -295,6 +295,12 @@ const definition = {
295
295
  "fit-harness output --format=text < trace.ndjson",
296
296
  ],
297
297
  documentation: [
298
+ {
299
+ title: "Coordinate an Agent Team",
300
+ url: "https://www.forwardimpact.team/docs/libraries/coordinate-team/index.md",
301
+ description:
302
+ "Run a lead and N participant agents in one async session — supervise, facilitate, or discuss — with Ask/Answer/Announce and a single NDJSON trace.",
303
+ },
298
304
  {
299
305
  title: "Run an Eval",
300
306
  url: "https://www.forwardimpact.team/docs/libraries/prove-changes/run-eval/index.md",
package/bin/fit-trace.js CHANGED
@@ -350,7 +350,7 @@ const definition = {
350
350
  options: {
351
351
  mode: {
352
352
  type: "string",
353
- description: "Execution mode: run, supervise, or facilitate",
353
+ description: "Execution mode: run, supervise, facilitate, or discuss",
354
354
  },
355
355
  case: {
356
356
  type: "string",
package/package.json CHANGED
@@ -1,13 +1,14 @@
1
1
  {
2
2
  "name": "@forwardimpact/libharness",
3
- "version": "1.1.0",
4
- "description": "Agent evaluation framework — prove whether agent changes improved outcomes with reproducible evidence.",
3
+ "version": "1.2.0",
4
+ "description": "Autonomous agent team harness — coordinate a lead and participant agents in one async session, with eval, benchmark, and trace tooling to prove the changes worked.",
5
5
  "keywords": [
6
+ "orchestration",
7
+ "supervisor",
8
+ "facilitator",
6
9
  "eval",
7
- "agent",
8
10
  "trace",
9
- "claude-code",
10
- "supervisor"
11
+ "agent"
11
12
  ],
12
13
  "homepage": "https://www.forwardimpact.team",
13
14
  "repository": {
@@ -18,12 +19,20 @@
18
19
  "license": "Apache-2.0",
19
20
  "author": "D. Olsson <hi@senzilla.io>",
20
21
  "jobs": [
22
+ {
23
+ "user": "Platform Builders",
24
+ "goal": "Coordinate an Agent Team",
25
+ "trigger": "Coordinating several agents in one session means hand-rolling message passing, turn-taking, and termination, and every orchestration script re-solves it.",
26
+ "bigHire": "coordinate a lead and participant agents over async messages in one session.",
27
+ "littleHire": "run a supervised pair, a facilitated meeting, or a multi-agent discussion without writing the message bus and turn loop.",
28
+ "competesWith": "manual orchestration scripts; hand-rolled message passing and turn-taking; sequential single-agent calls; running one agent at a time"
29
+ },
21
30
  {
22
31
  "user": "Platform Builders",
23
32
  "goal": "Prove Agent Changes",
24
33
  "trigger": "An eval passes locally but fails in CI and the only output is 'assertion failed.'",
25
34
  "bigHire": "prove whether agent changes improved outcomes with reproducible evidence.",
26
- "littleHire": "run an eval and get a trace that shows exactly what the agent did.",
35
+ "littleHire": "run an eval or benchmark and get a trace that shows exactly what the agent did.",
27
36
  "competesWith": "manual before/after comparison; trusting gut feeling over evidence; skipping evaluation entirely"
28
37
  }
29
38
  ],
@@ -169,7 +169,7 @@ export const definition = {
169
169
  title: "Automate with GitHub Actions",
170
170
  url: "https://www.forwardimpact.team/docs/libraries/prove-changes/run-benchmark/ci-workflow/index.md",
171
171
  description:
172
- "Run benchmarks in CI with the forwardimpact/fit-benchmark action.",
172
+ "Run benchmarks in CI with the forwardimpact/benchmark action.",
173
173
  },
174
174
  ],
175
175
  };
@@ -29,9 +29,11 @@ export function parseDiscussOptions(values, runtime) {
29
29
  );
30
30
 
31
31
  const profilesRaw = values["agent-profiles"];
32
- const agentCwd = resolve(values["agent-cwd"] ?? ".");
32
+ // `||` (not `??`) so an empty-string flag from a CI forwarder falls back to
33
+ // the default rather than overriding it with "".
34
+ const agentCwd = resolve(values["agent-cwd"] || ".");
33
35
 
34
- const maxTurnsRaw = values["max-turns"] ?? "40";
36
+ const maxTurnsRaw = values["max-turns"] || "40";
35
37
  const maxTurns = maxTurnsRaw === "0" ? 0 : parseInt(maxTurnsRaw, 10);
36
38
 
37
39
  const agentConfigs = parseAgentProfiles(profilesRaw, agentCwd, maxTurns);
@@ -46,21 +48,21 @@ export function parseDiscussOptions(values, runtime) {
46
48
  }
47
49
  }
48
50
 
49
- const maxLeadTurnsRaw = values["max-lead-turns"] ?? "200";
51
+ const maxLeadTurnsRaw = values["max-lead-turns"] || "200";
50
52
  const maxLeadTurns = parseInt(maxLeadTurnsRaw, 10);
51
53
 
52
54
  return {
53
55
  taskContent,
54
56
  taskAmend,
55
57
  agentConfigs,
56
- leadProfile: values["lead-profile"] ?? undefined,
58
+ leadProfile: values["lead-profile"] || undefined,
57
59
  leadModel: values["lead-model"] || LEAD_MODEL,
58
60
  agentModel: values["agent-model"] || AGENT_MODEL,
59
61
  maxTurns,
60
62
  maxLeadTurns,
61
63
  outputPath: values.output,
62
64
  workTracker: resolveWorkTracker(values, runtime?.proc?.env),
63
- discussionId: values["discussion-id"] ?? null,
65
+ discussionId: values["discussion-id"] || null,
64
66
  resumeContext,
65
67
  callbackUrl: runtime.proc.env.CALLBACK_URL ?? null,
66
68
  inboxUrl: runtime.proc.env.INBOX_URL ?? null,
@@ -36,9 +36,11 @@ export function parseFacilitateOptions(values, runtime) {
36
36
 
37
37
  const profilesRaw = values["agent-profiles"];
38
38
  if (!profilesRaw) throw new Error("--agent-profiles is required");
39
- const agentCwd = resolve(values["agent-cwd"] ?? ".");
39
+ // `||` (not `??`) so an empty-string flag from a CI forwarder falls back to
40
+ // the default rather than overriding it with "".
41
+ const agentCwd = resolve(values["agent-cwd"] || ".");
40
42
 
41
- const maxTurnsRaw = values["max-turns"] ?? "20";
43
+ const maxTurnsRaw = values["max-turns"] || "20";
42
44
  const maxTurns = maxTurnsRaw === "0" ? 0 : parseInt(maxTurnsRaw, 10);
43
45
 
44
46
  // Thread --max-turns into each participant: without this, every facilitated
@@ -51,12 +53,12 @@ export function parseFacilitateOptions(values, runtime) {
51
53
  taskContent,
52
54
  taskAmend,
53
55
  agentConfigs,
54
- facilitatorCwd: resolve(values["facilitator-cwd"] ?? "."),
56
+ facilitatorCwd: resolve(values["facilitator-cwd"] || "."),
55
57
  agentModel: values["agent-model"] || AGENT_MODEL,
56
58
  facilitatorModel: values["lead-model"] || LEAD_MODEL,
57
59
  maxTurns,
58
60
  outputPath: values.output,
59
- facilitatorProfile: values["lead-profile"] ?? undefined,
61
+ facilitatorProfile: values["lead-profile"] || undefined,
60
62
  workTracker: resolveWorkTracker(values, runtime?.proc?.env),
61
63
  };
62
64
  }
@@ -22,22 +22,24 @@ export function parseRunOptions(values, runtime) {
22
22
  values,
23
23
  runtime,
24
24
  );
25
- const maxTurnsRaw = values["max-turns"] ?? "50";
25
+ // `||` (not `??`) so an empty-string flag from a CI forwarder falls back to
26
+ // the default, rather than overriding it with "".
27
+ const maxTurnsRaw = values["max-turns"] || "50";
26
28
 
27
29
  return {
28
30
  taskContent,
29
31
  taskAmend,
30
- cwd: resolve(values.cwd ?? "."),
32
+ cwd: resolve(values.cwd || "."),
31
33
  agentModel: values["agent-model"] || AGENT_MODEL,
32
34
  maxTurns: maxTurnsRaw === "0" ? 0 : parseInt(maxTurnsRaw, 10),
33
35
  outputPath: values.output,
34
- agentProfile: values["agent-profile"] ?? undefined,
36
+ agentProfile: values["agent-profile"] || undefined,
35
37
  workTracker: resolveWorkTracker(values, runtime?.proc?.env),
36
38
  allowedTools: (
37
- values["allowed-tools"] ??
39
+ values["allowed-tools"] ||
38
40
  "Bash,Read,Glob,Grep,Write,Edit,Agent,TodoWrite"
39
41
  ).split(","),
40
- mcpServer: values["mcp-server"] ?? undefined,
42
+ mcpServer: values["mcp-server"] || undefined,
41
43
  };
42
44
  }
43
45
 
@@ -21,35 +21,37 @@ export async function parseSuperviseOptions(values, runtime) {
21
21
  );
22
22
  const supervisorAllowedToolsRaw = values["supervisor-allowed-tools"];
23
23
 
24
+ // `||` (not `??`) throughout so an empty-string flag from a CI forwarder
25
+ // falls back to the default rather than overriding it with "".
24
26
  const tmpRoot = runtime.proc.env.TMPDIR ?? "/tmp";
25
27
  const agentCwd = resolve(
26
- values["agent-cwd"] ??
28
+ values["agent-cwd"] ||
27
29
  (await runtime.fs.mkdtemp(join(tmpRoot, "fit-harness-agent-"))),
28
30
  );
29
31
 
30
32
  return {
31
33
  taskContent,
32
34
  taskAmend,
33
- supervisorCwd: resolve(values["supervisor-cwd"] ?? "."),
35
+ supervisorCwd: resolve(values["supervisor-cwd"] || "."),
34
36
  agentCwd,
35
37
  agentModel: values["agent-model"] || AGENT_MODEL,
36
38
  supervisorModel: values["lead-model"] || LEAD_MODEL,
37
39
  maxTurns: (() => {
38
- const raw = values["max-turns"] ?? "200";
40
+ const raw = values["max-turns"] || "200";
39
41
  return raw === "0" ? 0 : parseInt(raw, 10);
40
42
  })(),
41
43
  outputPath: values.output,
42
- supervisorProfile: values["lead-profile"] ?? undefined,
43
- agentProfile: values["agent-profile"] ?? undefined,
44
+ supervisorProfile: values["lead-profile"] || undefined,
45
+ agentProfile: values["agent-profile"] || undefined,
44
46
  workTracker: resolveWorkTracker(values, runtime?.proc?.env),
45
47
  allowedTools: (
46
- values["allowed-tools"] ??
48
+ values["allowed-tools"] ||
47
49
  "Bash,Read,Glob,Grep,Write,Edit,Agent,TodoWrite"
48
50
  ).split(","),
49
51
  supervisorAllowedTools: supervisorAllowedToolsRaw
50
52
  ? supervisorAllowedToolsRaw.split(",")
51
53
  : undefined,
52
- mcpServer: values["mcp-server"] ?? undefined,
54
+ mcpServer: values["mcp-server"] || undefined,
53
55
  };
54
56
  }
55
57
 
@@ -367,9 +367,12 @@ export async function runStatsCommand(ctx) {
367
367
  */
368
368
  export async function runCostCommand(ctx) {
369
369
  const { runtime } = ctx.deps;
370
- const cost = computeTraceCost(
371
- runtime.fsSync.readFileSync(ctx.args.file, "utf8"),
372
- );
370
+ // Tolerate a missing/empty trace: a CI step reports cost under `always()`,
371
+ // so the trace may not exist (the run failed before producing one). Print
372
+ // nothing and exit 0 rather than throwing — the caller needs no `if [ -f ]`.
373
+ const file = ctx.args.file;
374
+ if (!file || !runtime.fsSync.existsSync(file)) return { ok: true };
375
+ const cost = computeTraceCost(runtime.fsSync.readFileSync(file, "utf8"));
373
376
  if (ctx.options.markdown) {
374
377
  runtime.proc.stdout.write(renderCostMarkdown(cost));
375
378
  } else {
@@ -512,9 +515,12 @@ export async function runSplitCommand(ctx) {
512
515
  const file = ctx.args.file;
513
516
  if (!file) return { ok: false, code: 1, error: "split: missing input file" };
514
517
 
518
+ // `discuss` has the same lead + N-participants shape as `facilitate`, and the
519
+ // splitter buckets purely by envelope `source` (mode-independent), so it is
520
+ // accepted alongside the structural modes — the CLI owns this, not callers.
515
521
  const mode = ctx.options.mode;
516
522
  if (!mode) return { ok: false, code: 1, error: "split: --mode is required" };
517
- if (!["run", "supervise", "facilitate"].includes(mode)) {
523
+ if (!["run", "supervise", "facilitate", "discuss"].includes(mode)) {
518
524
  return { ok: false, code: 1, error: `split: invalid --mode "${mode}"` };
519
525
  }
520
526