@forwardimpact/libharness 1.1.0 → 1.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +19 -6
- package/bin/fit-harness.js +6 -0
- package/bin/fit-trace.js +1 -1
- package/package.json +15 -6
- package/src/commands/benchmark-definition.js +1 -1
- package/src/commands/discuss.js +7 -5
- package/src/commands/facilitate.js +6 -4
- package/src/commands/run.js +7 -5
- package/src/commands/supervise.js +9 -7
- package/src/commands/trace.js +10 -4
package/README.md
CHANGED
|
@@ -2,8 +2,9 @@
|
|
|
2
2
|
|
|
3
3
|
<!-- BEGIN:description — Do not edit. Generated from package.json. -->
|
|
4
4
|
|
|
5
|
-
|
|
6
|
-
|
|
5
|
+
Autonomous agent team harness — coordinate a lead and participant agents in one
|
|
6
|
+
async session, with eval, benchmark, and trace tooling to prove the changes
|
|
7
|
+
worked.
|
|
7
8
|
|
|
8
9
|
<!-- END:description -->
|
|
9
10
|
|
|
@@ -53,7 +54,8 @@ Every Ask returns immediately and registers a pending entry keyed by an
|
|
|
53
54
|
lead. Answer's `askId` is optional — the handler is forgiving:
|
|
54
55
|
|
|
55
56
|
- **Provided + matches an ask owed by the caller** → routes to that asker.
|
|
56
|
-
- **Provided but unknown or wrong addressee** → `isError` with a pointed
|
|
57
|
+
- **Provided but unknown or wrong addressee** → `isError` with a pointed
|
|
58
|
+
message.
|
|
57
59
|
- **Omitted + exactly one ask owed to the caller** → auto-picks it.
|
|
58
60
|
- **Omitted + 0 or many asks owed** → broadcasts as Announce.
|
|
59
61
|
|
|
@@ -209,6 +211,17 @@ lists the `Edit()` rules that were tried.
|
|
|
209
211
|
|
|
210
212
|
## Documentation
|
|
211
213
|
|
|
212
|
-
- [Agent
|
|
213
|
-
|
|
214
|
-
|
|
214
|
+
- [Coordinate an Agent Team](https://www.forwardimpact.team/docs/libraries/coordinate-team/index.md)
|
|
215
|
+
— run a lead and N participant agents in one async session (supervise /
|
|
216
|
+
facilitate / discuss) with Ask/Answer/Announce and a single NDJSON trace.
|
|
217
|
+
- [Run an Eval](https://www.forwardimpact.team/docs/libraries/prove-changes/run-eval/index.md)
|
|
218
|
+
— author a judge profile, run an eval locally, wire it into CI, and inspect
|
|
219
|
+
the resulting trace.
|
|
220
|
+
- [Prove Agent Changes](https://www.forwardimpact.team/docs/libraries/prove-changes/index.md)
|
|
221
|
+
— end-to-end workflow from dataset generation through evaluation to trace
|
|
222
|
+
analysis, including multi-agent collaboration sessions.
|
|
223
|
+
- [Analyze Traces](https://www.forwardimpact.team/docs/libraries/prove-changes/trace-analysis/index.md)
|
|
224
|
+
— read the NDJSON traces produced by `fit-harness` with `fit-trace`.
|
|
225
|
+
- [Agent Teams](https://www.forwardimpact.team/docs/products/agent-teams/index.md)
|
|
226
|
+
— author the profiles consumed by `--agent-profile`, `--lead-profile`, and
|
|
227
|
+
`--agent-profiles`.
|
package/bin/fit-harness.js
CHANGED
|
@@ -295,6 +295,12 @@ const definition = {
|
|
|
295
295
|
"fit-harness output --format=text < trace.ndjson",
|
|
296
296
|
],
|
|
297
297
|
documentation: [
|
|
298
|
+
{
|
|
299
|
+
title: "Coordinate an Agent Team",
|
|
300
|
+
url: "https://www.forwardimpact.team/docs/libraries/coordinate-team/index.md",
|
|
301
|
+
description:
|
|
302
|
+
"Run a lead and N participant agents in one async session — supervise, facilitate, or discuss — with Ask/Answer/Announce and a single NDJSON trace.",
|
|
303
|
+
},
|
|
298
304
|
{
|
|
299
305
|
title: "Run an Eval",
|
|
300
306
|
url: "https://www.forwardimpact.team/docs/libraries/prove-changes/run-eval/index.md",
|
package/bin/fit-trace.js
CHANGED
package/package.json
CHANGED
|
@@ -1,13 +1,14 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@forwardimpact/libharness",
|
|
3
|
-
"version": "1.
|
|
4
|
-
"description": "
|
|
3
|
+
"version": "1.2.0",
|
|
4
|
+
"description": "Autonomous agent team harness — coordinate a lead and participant agents in one async session, with eval, benchmark, and trace tooling to prove the changes worked.",
|
|
5
5
|
"keywords": [
|
|
6
|
+
"orchestration",
|
|
7
|
+
"supervisor",
|
|
8
|
+
"facilitator",
|
|
6
9
|
"eval",
|
|
7
|
-
"agent",
|
|
8
10
|
"trace",
|
|
9
|
-
"
|
|
10
|
-
"supervisor"
|
|
11
|
+
"agent"
|
|
11
12
|
],
|
|
12
13
|
"homepage": "https://www.forwardimpact.team",
|
|
13
14
|
"repository": {
|
|
@@ -18,12 +19,20 @@
|
|
|
18
19
|
"license": "Apache-2.0",
|
|
19
20
|
"author": "D. Olsson <hi@senzilla.io>",
|
|
20
21
|
"jobs": [
|
|
22
|
+
{
|
|
23
|
+
"user": "Platform Builders",
|
|
24
|
+
"goal": "Coordinate an Agent Team",
|
|
25
|
+
"trigger": "Coordinating several agents in one session means hand-rolling message passing, turn-taking, and termination, and every orchestration script re-solves it.",
|
|
26
|
+
"bigHire": "coordinate a lead and participant agents over async messages in one session.",
|
|
27
|
+
"littleHire": "run a supervised pair, a facilitated meeting, or a multi-agent discussion without writing the message bus and turn loop.",
|
|
28
|
+
"competesWith": "manual orchestration scripts; hand-rolled message passing and turn-taking; sequential single-agent calls; running one agent at a time"
|
|
29
|
+
},
|
|
21
30
|
{
|
|
22
31
|
"user": "Platform Builders",
|
|
23
32
|
"goal": "Prove Agent Changes",
|
|
24
33
|
"trigger": "An eval passes locally but fails in CI and the only output is 'assertion failed.'",
|
|
25
34
|
"bigHire": "prove whether agent changes improved outcomes with reproducible evidence.",
|
|
26
|
-
"littleHire": "run an eval and get a trace that shows exactly what the agent did.",
|
|
35
|
+
"littleHire": "run an eval or benchmark and get a trace that shows exactly what the agent did.",
|
|
27
36
|
"competesWith": "manual before/after comparison; trusting gut feeling over evidence; skipping evaluation entirely"
|
|
28
37
|
}
|
|
29
38
|
],
|
|
@@ -169,7 +169,7 @@ export const definition = {
|
|
|
169
169
|
title: "Automate with GitHub Actions",
|
|
170
170
|
url: "https://www.forwardimpact.team/docs/libraries/prove-changes/run-benchmark/ci-workflow/index.md",
|
|
171
171
|
description:
|
|
172
|
-
"Run benchmarks in CI with the forwardimpact/
|
|
172
|
+
"Run benchmarks in CI with the forwardimpact/benchmark action.",
|
|
173
173
|
},
|
|
174
174
|
],
|
|
175
175
|
};
|
package/src/commands/discuss.js
CHANGED
|
@@ -29,9 +29,11 @@ export function parseDiscussOptions(values, runtime) {
|
|
|
29
29
|
);
|
|
30
30
|
|
|
31
31
|
const profilesRaw = values["agent-profiles"];
|
|
32
|
-
|
|
32
|
+
// `||` (not `??`) so an empty-string flag from a CI forwarder falls back to
|
|
33
|
+
// the default rather than overriding it with "".
|
|
34
|
+
const agentCwd = resolve(values["agent-cwd"] || ".");
|
|
33
35
|
|
|
34
|
-
const maxTurnsRaw = values["max-turns"]
|
|
36
|
+
const maxTurnsRaw = values["max-turns"] || "40";
|
|
35
37
|
const maxTurns = maxTurnsRaw === "0" ? 0 : parseInt(maxTurnsRaw, 10);
|
|
36
38
|
|
|
37
39
|
const agentConfigs = parseAgentProfiles(profilesRaw, agentCwd, maxTurns);
|
|
@@ -46,21 +48,21 @@ export function parseDiscussOptions(values, runtime) {
|
|
|
46
48
|
}
|
|
47
49
|
}
|
|
48
50
|
|
|
49
|
-
const maxLeadTurnsRaw = values["max-lead-turns"]
|
|
51
|
+
const maxLeadTurnsRaw = values["max-lead-turns"] || "200";
|
|
50
52
|
const maxLeadTurns = parseInt(maxLeadTurnsRaw, 10);
|
|
51
53
|
|
|
52
54
|
return {
|
|
53
55
|
taskContent,
|
|
54
56
|
taskAmend,
|
|
55
57
|
agentConfigs,
|
|
56
|
-
leadProfile: values["lead-profile"]
|
|
58
|
+
leadProfile: values["lead-profile"] || undefined,
|
|
57
59
|
leadModel: values["lead-model"] || LEAD_MODEL,
|
|
58
60
|
agentModel: values["agent-model"] || AGENT_MODEL,
|
|
59
61
|
maxTurns,
|
|
60
62
|
maxLeadTurns,
|
|
61
63
|
outputPath: values.output,
|
|
62
64
|
workTracker: resolveWorkTracker(values, runtime?.proc?.env),
|
|
63
|
-
discussionId: values["discussion-id"]
|
|
65
|
+
discussionId: values["discussion-id"] || null,
|
|
64
66
|
resumeContext,
|
|
65
67
|
callbackUrl: runtime.proc.env.CALLBACK_URL ?? null,
|
|
66
68
|
inboxUrl: runtime.proc.env.INBOX_URL ?? null,
|
|
@@ -36,9 +36,11 @@ export function parseFacilitateOptions(values, runtime) {
|
|
|
36
36
|
|
|
37
37
|
const profilesRaw = values["agent-profiles"];
|
|
38
38
|
if (!profilesRaw) throw new Error("--agent-profiles is required");
|
|
39
|
-
|
|
39
|
+
// `||` (not `??`) so an empty-string flag from a CI forwarder falls back to
|
|
40
|
+
// the default rather than overriding it with "".
|
|
41
|
+
const agentCwd = resolve(values["agent-cwd"] || ".");
|
|
40
42
|
|
|
41
|
-
const maxTurnsRaw = values["max-turns"]
|
|
43
|
+
const maxTurnsRaw = values["max-turns"] || "20";
|
|
42
44
|
const maxTurns = maxTurnsRaw === "0" ? 0 : parseInt(maxTurnsRaw, 10);
|
|
43
45
|
|
|
44
46
|
// Thread --max-turns into each participant: without this, every facilitated
|
|
@@ -51,12 +53,12 @@ export function parseFacilitateOptions(values, runtime) {
|
|
|
51
53
|
taskContent,
|
|
52
54
|
taskAmend,
|
|
53
55
|
agentConfigs,
|
|
54
|
-
facilitatorCwd: resolve(values["facilitator-cwd"]
|
|
56
|
+
facilitatorCwd: resolve(values["facilitator-cwd"] || "."),
|
|
55
57
|
agentModel: values["agent-model"] || AGENT_MODEL,
|
|
56
58
|
facilitatorModel: values["lead-model"] || LEAD_MODEL,
|
|
57
59
|
maxTurns,
|
|
58
60
|
outputPath: values.output,
|
|
59
|
-
facilitatorProfile: values["lead-profile"]
|
|
61
|
+
facilitatorProfile: values["lead-profile"] || undefined,
|
|
60
62
|
workTracker: resolveWorkTracker(values, runtime?.proc?.env),
|
|
61
63
|
};
|
|
62
64
|
}
|
package/src/commands/run.js
CHANGED
|
@@ -22,22 +22,24 @@ export function parseRunOptions(values, runtime) {
|
|
|
22
22
|
values,
|
|
23
23
|
runtime,
|
|
24
24
|
);
|
|
25
|
-
|
|
25
|
+
// `||` (not `??`) so an empty-string flag from a CI forwarder falls back to
|
|
26
|
+
// the default, rather than overriding it with "".
|
|
27
|
+
const maxTurnsRaw = values["max-turns"] || "50";
|
|
26
28
|
|
|
27
29
|
return {
|
|
28
30
|
taskContent,
|
|
29
31
|
taskAmend,
|
|
30
|
-
cwd: resolve(values.cwd
|
|
32
|
+
cwd: resolve(values.cwd || "."),
|
|
31
33
|
agentModel: values["agent-model"] || AGENT_MODEL,
|
|
32
34
|
maxTurns: maxTurnsRaw === "0" ? 0 : parseInt(maxTurnsRaw, 10),
|
|
33
35
|
outputPath: values.output,
|
|
34
|
-
agentProfile: values["agent-profile"]
|
|
36
|
+
agentProfile: values["agent-profile"] || undefined,
|
|
35
37
|
workTracker: resolveWorkTracker(values, runtime?.proc?.env),
|
|
36
38
|
allowedTools: (
|
|
37
|
-
values["allowed-tools"]
|
|
39
|
+
values["allowed-tools"] ||
|
|
38
40
|
"Bash,Read,Glob,Grep,Write,Edit,Agent,TodoWrite"
|
|
39
41
|
).split(","),
|
|
40
|
-
mcpServer: values["mcp-server"]
|
|
42
|
+
mcpServer: values["mcp-server"] || undefined,
|
|
41
43
|
};
|
|
42
44
|
}
|
|
43
45
|
|
|
@@ -21,35 +21,37 @@ export async function parseSuperviseOptions(values, runtime) {
|
|
|
21
21
|
);
|
|
22
22
|
const supervisorAllowedToolsRaw = values["supervisor-allowed-tools"];
|
|
23
23
|
|
|
24
|
+
// `||` (not `??`) throughout so an empty-string flag from a CI forwarder
|
|
25
|
+
// falls back to the default rather than overriding it with "".
|
|
24
26
|
const tmpRoot = runtime.proc.env.TMPDIR ?? "/tmp";
|
|
25
27
|
const agentCwd = resolve(
|
|
26
|
-
values["agent-cwd"]
|
|
28
|
+
values["agent-cwd"] ||
|
|
27
29
|
(await runtime.fs.mkdtemp(join(tmpRoot, "fit-harness-agent-"))),
|
|
28
30
|
);
|
|
29
31
|
|
|
30
32
|
return {
|
|
31
33
|
taskContent,
|
|
32
34
|
taskAmend,
|
|
33
|
-
supervisorCwd: resolve(values["supervisor-cwd"]
|
|
35
|
+
supervisorCwd: resolve(values["supervisor-cwd"] || "."),
|
|
34
36
|
agentCwd,
|
|
35
37
|
agentModel: values["agent-model"] || AGENT_MODEL,
|
|
36
38
|
supervisorModel: values["lead-model"] || LEAD_MODEL,
|
|
37
39
|
maxTurns: (() => {
|
|
38
|
-
const raw = values["max-turns"]
|
|
40
|
+
const raw = values["max-turns"] || "200";
|
|
39
41
|
return raw === "0" ? 0 : parseInt(raw, 10);
|
|
40
42
|
})(),
|
|
41
43
|
outputPath: values.output,
|
|
42
|
-
supervisorProfile: values["lead-profile"]
|
|
43
|
-
agentProfile: values["agent-profile"]
|
|
44
|
+
supervisorProfile: values["lead-profile"] || undefined,
|
|
45
|
+
agentProfile: values["agent-profile"] || undefined,
|
|
44
46
|
workTracker: resolveWorkTracker(values, runtime?.proc?.env),
|
|
45
47
|
allowedTools: (
|
|
46
|
-
values["allowed-tools"]
|
|
48
|
+
values["allowed-tools"] ||
|
|
47
49
|
"Bash,Read,Glob,Grep,Write,Edit,Agent,TodoWrite"
|
|
48
50
|
).split(","),
|
|
49
51
|
supervisorAllowedTools: supervisorAllowedToolsRaw
|
|
50
52
|
? supervisorAllowedToolsRaw.split(",")
|
|
51
53
|
: undefined,
|
|
52
|
-
mcpServer: values["mcp-server"]
|
|
54
|
+
mcpServer: values["mcp-server"] || undefined,
|
|
53
55
|
};
|
|
54
56
|
}
|
|
55
57
|
|
package/src/commands/trace.js
CHANGED
|
@@ -367,9 +367,12 @@ export async function runStatsCommand(ctx) {
|
|
|
367
367
|
*/
|
|
368
368
|
export async function runCostCommand(ctx) {
|
|
369
369
|
const { runtime } = ctx.deps;
|
|
370
|
-
|
|
371
|
-
|
|
372
|
-
|
|
370
|
+
// Tolerate a missing/empty trace: a CI step reports cost under `always()`,
|
|
371
|
+
// so the trace may not exist (the run failed before producing one). Print
|
|
372
|
+
// nothing and exit 0 rather than throwing — the caller needs no `if [ -f ]`.
|
|
373
|
+
const file = ctx.args.file;
|
|
374
|
+
if (!file || !runtime.fsSync.existsSync(file)) return { ok: true };
|
|
375
|
+
const cost = computeTraceCost(runtime.fsSync.readFileSync(file, "utf8"));
|
|
373
376
|
if (ctx.options.markdown) {
|
|
374
377
|
runtime.proc.stdout.write(renderCostMarkdown(cost));
|
|
375
378
|
} else {
|
|
@@ -512,9 +515,12 @@ export async function runSplitCommand(ctx) {
|
|
|
512
515
|
const file = ctx.args.file;
|
|
513
516
|
if (!file) return { ok: false, code: 1, error: "split: missing input file" };
|
|
514
517
|
|
|
518
|
+
// `discuss` has the same lead + N-participants shape as `facilitate`, and the
|
|
519
|
+
// splitter buckets purely by envelope `source` (mode-independent), so it is
|
|
520
|
+
// accepted alongside the structural modes — the CLI owns this, not callers.
|
|
515
521
|
const mode = ctx.options.mode;
|
|
516
522
|
if (!mode) return { ok: false, code: 1, error: "split: --mode is required" };
|
|
517
|
-
if (!["run", "supervise", "facilitate"].includes(mode)) {
|
|
523
|
+
if (!["run", "supervise", "facilitate", "discuss"].includes(mode)) {
|
|
518
524
|
return { ok: false, code: 1, error: `split: invalid --mode "${mode}"` };
|
|
519
525
|
}
|
|
520
526
|
|