@forwardimpact/libharness 0.1.22 → 1.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (87) hide show
  1. package/LICENSE +21 -201
  2. package/README.md +196 -80
  3. package/bin/fit-benchmark.js +44 -0
  4. package/bin/fit-harness.js +358 -0
  5. package/bin/fit-selfedit.js +165 -0
  6. package/bin/fit-trace.js +510 -0
  7. package/package.json +41 -11
  8. package/src/agent-runner.js +256 -0
  9. package/src/benchmark/apm-installer.js +207 -0
  10. package/src/benchmark/env-loader.js +158 -0
  11. package/src/benchmark/hook-env.js +40 -0
  12. package/src/benchmark/invariants.js +141 -0
  13. package/src/benchmark/judge.js +187 -0
  14. package/src/benchmark/npm-installer.js +87 -0
  15. package/src/benchmark/report.js +604 -0
  16. package/src/benchmark/result.js +127 -0
  17. package/src/benchmark/runner.js +688 -0
  18. package/src/benchmark/scheduler.js +78 -0
  19. package/src/benchmark/task-family.js +260 -0
  20. package/src/benchmark/workdir.js +344 -0
  21. package/src/commands/assert.js +153 -0
  22. package/src/commands/benchmark-definition.js +175 -0
  23. package/src/commands/benchmark-invariants.js +73 -0
  24. package/src/commands/benchmark-report.js +51 -0
  25. package/src/commands/benchmark-run.js +175 -0
  26. package/src/commands/by-discussion.js +94 -0
  27. package/src/commands/callback.js +119 -0
  28. package/src/commands/discuss.js +132 -0
  29. package/src/commands/facilitate.js +123 -0
  30. package/src/commands/output.js +36 -0
  31. package/src/commands/run.js +152 -0
  32. package/src/commands/supervise.js +136 -0
  33. package/src/commands/task-input.js +54 -0
  34. package/src/commands/tee.js +53 -0
  35. package/src/commands/trace.js +630 -0
  36. package/src/commands/work-tracker.js +35 -0
  37. package/src/cost.js +79 -0
  38. package/src/discuss-tools.js +173 -0
  39. package/src/discusser.js +394 -0
  40. package/src/events/github.js +161 -0
  41. package/src/facilitator.js +205 -0
  42. package/src/inbox-poller.js +81 -0
  43. package/src/index.js +72 -2
  44. package/src/judge.js +210 -0
  45. package/src/message-bus.js +118 -0
  46. package/src/orchestration-loop.js +330 -0
  47. package/src/orchestration-toolkit.js +441 -0
  48. package/src/orchestrator-helpers.js +23 -0
  49. package/src/profile-prompt.js +266 -0
  50. package/src/redaction.js +253 -0
  51. package/src/render/line-renderer.js +54 -0
  52. package/src/render/orchestrator-filter.js +19 -0
  53. package/src/render/palette.js +63 -0
  54. package/src/render/tool-hints.js +154 -0
  55. package/src/render/turn-renderer.js +96 -0
  56. package/src/reply-emitter.js +47 -0
  57. package/src/sequence-counter.js +21 -0
  58. package/src/signature-filter.js +27 -0
  59. package/src/supervisor.js +236 -0
  60. package/src/tee-writer.js +150 -0
  61. package/src/trace-collector.js +444 -0
  62. package/src/trace-github.js +473 -0
  63. package/src/trace-multi.js +101 -0
  64. package/src/trace-query.js +748 -0
  65. package/src/trace-render.js +211 -0
  66. package/src/trace-usage.js +249 -0
  67. package/src/fixture/assertions.js +0 -42
  68. package/src/fixture/cache.js +0 -50
  69. package/src/fixture/eval.js +0 -146
  70. package/src/fixture/index.js +0 -9
  71. package/src/fixture/pathway.js +0 -451
  72. package/src/fixture/services.js +0 -56
  73. package/src/mock/clients.js +0 -135
  74. package/src/mock/config.js +0 -45
  75. package/src/mock/data.js +0 -46
  76. package/src/mock/fs.js +0 -111
  77. package/src/mock/grpc.js +0 -94
  78. package/src/mock/http.js +0 -60
  79. package/src/mock/index.js +0 -36
  80. package/src/mock/infra.js +0 -219
  81. package/src/mock/logger.js +0 -42
  82. package/src/mock/observer.js +0 -74
  83. package/src/mock/resource-index.js +0 -95
  84. package/src/mock/service-callbacks.js +0 -39
  85. package/src/mock/services.js +0 -79
  86. package/src/mock/spy.js +0 -44
  87. package/src/mock/storage.js +0 -118
@@ -0,0 +1,94 @@
1
+ import { join } from "node:path";
2
+
3
+ const FIRST_LINE_CAP = 64 * 1024;
4
+
5
+ /**
6
+ * Read the first newline-terminated line of a file, bounded to the first
7
+ * {@link FIRST_LINE_CAP} bytes. Trace `.ndjson` files can be many MB; the
8
+ * Step 2.6 meta header is always small, so a bounded positional read avoids
9
+ * loading whole files into memory just to inspect the header. The positional
10
+ * `openSync`/`readSync`/`closeSync` trio is read off the injected
11
+ * `runtime.fsSync` surface.
12
+ *
13
+ * @param {object} fsSync - Sync filesystem surface (`runtime.fsSync`).
14
+ * @param {string} path
15
+ * @returns {string}
16
+ */
17
+ function readFirstLine(fsSync, path) {
18
+ const fd = fsSync.openSync(path, "r");
19
+ try {
20
+ const buf = Buffer.alloc(FIRST_LINE_CAP);
21
+ const bytes = fsSync.readSync(fd, buf, 0, buf.length, 0);
22
+ const text = buf.toString("utf8", 0, bytes);
23
+ const nl = text.indexOf("\n");
24
+ return nl === -1 ? text : text.slice(0, nl);
25
+ } finally {
26
+ fsSync.closeSync(fd);
27
+ }
28
+ }
29
+
30
+ /**
31
+ * Scan a directory for `.ndjson` files whose meta header carries the
32
+ * given discussion_id. The Step 2.6 first-line guarantee makes the
33
+ * lookup cheap: we read only the first line per file. Files without a
34
+ * meta header (e.g. legacy supervise/facilitate traces) are skipped
35
+ * silently — not erroneous.
36
+ *
37
+ * @param {string} dir
38
+ * @param {string} discussionId
39
+ * @param {object} fsSync - Sync filesystem surface (`runtime.fsSync`).
40
+ * @returns {Array<{path: string, mtimeMs: number}>}
41
+ */
42
+ export function findTracesByDiscussion(dir, discussionId, fsSync) {
43
+ const matches = [];
44
+ let entries;
45
+ try {
46
+ entries = fsSync.readdirSync(dir);
47
+ } catch {
48
+ return [];
49
+ }
50
+ for (const entry of entries) {
51
+ if (!entry.endsWith(".ndjson")) continue;
52
+ const path = join(dir, entry);
53
+ let firstLine;
54
+ try {
55
+ firstLine = readFirstLine(fsSync, path);
56
+ } catch {
57
+ continue;
58
+ }
59
+ let parsed;
60
+ try {
61
+ parsed = JSON.parse(firstLine);
62
+ } catch {
63
+ continue;
64
+ }
65
+ const event = parsed.event ?? parsed;
66
+ if (event?.type !== "meta") continue;
67
+ if (event.discussion_id !== discussionId) continue;
68
+ matches.push({ path, mtimeMs: fsSync.statSync(path).mtimeMs });
69
+ }
70
+ matches.sort((a, b) => a.mtimeMs - b.mtimeMs);
71
+ return matches;
72
+ }
73
+
74
+ /**
75
+ * `fit-trace by-discussion <discussion-id> [trace-dir]` — list trace
76
+ * files whose meta header carries the given discussion_id, one per
77
+ * line, ordered by first-event timestamp (file mtime ascending). The
78
+ * result is usable with `xargs cat` for a chronological merge.
79
+ *
80
+ * @param {import("@forwardimpact/libcli").InvocationContext} ctx
81
+ * @returns {Promise<{ok: true} | {ok: false, code: number, error: string}>}
82
+ */
83
+ export async function runByDiscussionCommand(ctx) {
84
+ const runtime = ctx.deps.runtime;
85
+ const discussionId = ctx.args["discussion-id"];
86
+ if (!discussionId)
87
+ return { ok: false, code: 1, error: "<discussion-id> is required" };
88
+ const dir = ctx.args["trace-dir"] ?? ctx.options["trace-dir"] ?? "traces";
89
+ const matches = findTracesByDiscussion(dir, discussionId, runtime.fsSync);
90
+ for (const { path } of matches) {
91
+ runtime.proc.stdout.write(`${path}\n`);
92
+ }
93
+ return { ok: true };
94
+ }
@@ -0,0 +1,119 @@
1
+ import { sumTraceCost } from "../cost.js";
2
+
3
+ /**
4
+ * Scan an NDJSON trace and return the last orchestrator summary event,
5
+ * the first `meta` event's `discussion_id`, and any structured replies
6
+ * collected by the discusser. Skips malformed lines.
7
+ *
8
+ * The runner is verdict-agnostic — verbatim passthrough of whatever the
9
+ * trace carries ("success"/"failure" from supervise/facilitate; canonical
10
+ * "adjourned"/"recessed"/"failed" from discuss). The bridge layer maps to
11
+ * its channel semantics.
12
+ *
13
+ * @param {string} content - Raw NDJSON trace content.
14
+ * @returns {{verdict: string, summary: string, replies: object[], trigger?: object, discussionId?: string} | null}
15
+ */
16
+ // biome-ignore lint/complexity/noExcessiveCognitiveComplexity: NDJSON scan with malformed-line tolerance + meta/summary dual extraction
17
+ function readTraceSummary(content) {
18
+ let summary = null;
19
+ let metaDiscussionId = null;
20
+ for (const line of content.split("\n")) {
21
+ if (!line.trim()) continue;
22
+ let record;
23
+ try {
24
+ record = JSON.parse(line);
25
+ } catch {
26
+ continue;
27
+ }
28
+ if (record.source !== "orchestrator") continue;
29
+ if (record.event?.type === "meta" && !metaDiscussionId) {
30
+ metaDiscussionId = record.event.discussion_id ?? null;
31
+ }
32
+ if (record.event?.type === "summary") {
33
+ summary = {
34
+ verdict: record.event.verdict ?? "failed",
35
+ summary: record.event.summary ?? "",
36
+ replies: Array.isArray(record.event.replies)
37
+ ? record.event.replies
38
+ : [],
39
+ ...(record.event.trigger && { trigger: record.event.trigger }),
40
+ ...(record.event.discussion_id && {
41
+ discussionId: record.event.discussion_id,
42
+ }),
43
+ ...(typeof record.event.lastActedSeq === "number" && {
44
+ lastActedSeq: record.event.lastActedSeq,
45
+ }),
46
+ };
47
+ }
48
+ }
49
+ if (summary && !summary.discussionId && metaDiscussionId) {
50
+ summary.discussionId = metaDiscussionId;
51
+ }
52
+ return summary;
53
+ }
54
+
55
+ /**
56
+ * Callback command — read an NDJSON trace, extract the terminal
57
+ * orchestrator summary, and POST a canonical callback body to the
58
+ * configured URL. Used by `kata-dispatch.yml` to deliver the lead's
59
+ * conclusion to the bridge that dispatched the run.
60
+ *
61
+ * Wire shape (single shape across modes):
62
+ *
63
+ * ```
64
+ * {
65
+ * correlation_id, verdict, summary, run_url,
66
+ * discussion_id?, replies: [], trigger?
67
+ * }
68
+ * ```
69
+ *
70
+ * @param {import("@forwardimpact/libcli").InvocationContext} ctx
71
+ * @returns {Promise<{ok: true} | {ok: false, code: number, error: string}>}
72
+ */
73
+ export async function runCallbackCommand(ctx) {
74
+ const values = ctx.options;
75
+ const runtime = ctx.deps.runtime;
76
+ const traceFile = values["trace-file"];
77
+ const callbackUrl = values["callback-url"];
78
+ const correlationId = values["correlation-id"];
79
+ const runUrl = values["run-url"] ?? "";
80
+ const discussionIdOverride = values["discussion-id"] ?? null;
81
+
82
+ if (!traceFile)
83
+ return { ok: false, code: 1, error: "--trace-file is required" };
84
+ if (!callbackUrl)
85
+ return { ok: false, code: 1, error: "--callback-url is required" };
86
+
87
+ const content = runtime.fsSync.readFileSync(traceFile, "utf8");
88
+ const found = readTraceSummary(content) ?? {
89
+ verdict: "failed",
90
+ summary: "Run ended without producing a summary.",
91
+ replies: [],
92
+ };
93
+ // Total spend across every participant in the trace — the bridge surfaces
94
+ // it alongside the verdict so a dispatched run reports what it cost.
95
+ const { totalCostUsd } = sumTraceCost(content.split("\n"));
96
+
97
+ const discussionId = found.discussionId ?? discussionIdOverride ?? null;
98
+ const payload = {
99
+ correlation_id: correlationId,
100
+ kind: "terminal",
101
+ verdict: found.verdict,
102
+ summary: found.summary,
103
+ run_url: runUrl,
104
+ cost_usd: totalCostUsd,
105
+ replies: found.replies,
106
+ last_acted_seq: found.lastActedSeq ?? -1,
107
+ ...(discussionId && { discussion_id: discussionId }),
108
+ ...(found.trigger && { trigger: found.trigger }),
109
+ };
110
+ const res = await fetch(callbackUrl, {
111
+ method: "POST",
112
+ headers: { "Content-Type": "application/json" },
113
+ body: JSON.stringify(payload),
114
+ });
115
+ if (!res.ok) {
116
+ return { ok: false, code: 1, error: `Callback POST failed: ${res.status}` };
117
+ }
118
+ return { ok: true };
119
+ }
@@ -0,0 +1,132 @@
1
+ import { resolve } from "node:path";
2
+ import { isoTimestamp } from "@forwardimpact/libutil";
3
+ import { createDiscusser } from "../discusser.js";
4
+ import { createRedactor } from "../redaction.js";
5
+ import { createTeeWriter } from "../tee-writer.js";
6
+ import { resolveTaskContent } from "./task-input.js";
7
+ import { resolveWorkTracker } from "./work-tracker.js";
8
+ import { AGENT_MODEL, LEAD_MODEL } from "@forwardimpact/libutil/models";
9
+
10
+ function parseAgentProfiles(raw, cwd, maxTurns) {
11
+ if (!raw) return [];
12
+ return raw.split(",").map((entry) => {
13
+ const name = entry.trim();
14
+ return { name, role: name, cwd, agentProfile: name, maxTurns };
15
+ });
16
+ }
17
+
18
+ /**
19
+ * Parse and validate discuss command options. Exported so tests can verify
20
+ * defaults and the legacy-flag clean break.
21
+ * @param {object} values - Parsed option values
22
+ * @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
23
+ * @returns {object}
24
+ */
25
+ export function parseDiscussOptions(values, runtime) {
26
+ const { task: taskContent, amend: taskAmend } = resolveTaskContent(
27
+ values,
28
+ runtime,
29
+ );
30
+
31
+ const profilesRaw = values["agent-profiles"];
32
+ const agentCwd = resolve(values["agent-cwd"] ?? ".");
33
+
34
+ const maxTurnsRaw = values["max-turns"] ?? "40";
35
+ const maxTurns = maxTurnsRaw === "0" ? 0 : parseInt(maxTurnsRaw, 10);
36
+
37
+ const agentConfigs = parseAgentProfiles(profilesRaw, agentCwd, maxTurns);
38
+
39
+ const resumeContextRaw = values["resume-context"];
40
+ let resumeContext = null;
41
+ if (resumeContextRaw) {
42
+ try {
43
+ resumeContext = JSON.parse(resumeContextRaw);
44
+ } catch (err) {
45
+ throw new Error(`--resume-context is not valid JSON: ${err.message}`);
46
+ }
47
+ }
48
+
49
+ const maxLeadTurnsRaw = values["max-lead-turns"] ?? "200";
50
+ const maxLeadTurns = parseInt(maxLeadTurnsRaw, 10);
51
+
52
+ return {
53
+ taskContent,
54
+ taskAmend,
55
+ agentConfigs,
56
+ leadProfile: values["lead-profile"] ?? undefined,
57
+ leadModel: values["lead-model"] || LEAD_MODEL,
58
+ agentModel: values["agent-model"] || AGENT_MODEL,
59
+ maxTurns,
60
+ maxLeadTurns,
61
+ outputPath: values.output,
62
+ workTracker: resolveWorkTracker(values, runtime?.proc?.env),
63
+ discussionId: values["discussion-id"] ?? null,
64
+ resumeContext,
65
+ callbackUrl: runtime.proc.env.CALLBACK_URL ?? null,
66
+ inboxUrl: runtime.proc.env.INBOX_URL ?? null,
67
+ correlationId: runtime.proc.env.CORRELATION_ID ?? null,
68
+ };
69
+ }
70
+
71
+ /**
72
+ * Discuss command — run a discusser-led session with suspend/resume
73
+ * semantics, threading `discussion_id` through the trace so multi-run
74
+ * conversations are queryable as one.
75
+ *
76
+ * @param {import("@forwardimpact/libcli").InvocationContext} ctx
77
+ * @returns {Promise<{ok: boolean, code?: number, error?: string}>}
78
+ */
79
+ export async function runDiscussCommand(ctx) {
80
+ const runtime = ctx.deps.runtime;
81
+ const opts = parseDiscussOptions(ctx.options, runtime);
82
+
83
+ const redactor = createRedactor({ runtime });
84
+
85
+ const fileStream = opts.outputPath
86
+ ? runtime.fs.createWriteStream(opts.outputPath)
87
+ : null;
88
+ const output = fileStream
89
+ ? createTeeWriter({
90
+ fileStream,
91
+ textStream: runtime.proc.stdout,
92
+ mode: "supervised",
93
+ now: () => isoTimestamp(runtime.clock.now()),
94
+ })
95
+ : runtime.proc.stdout;
96
+
97
+ if (opts.leadProfile) {
98
+ runtime.proc.env.LIBHARNESS_AGENT_PROFILE = opts.leadProfile;
99
+ }
100
+ // Unconditional so the default "github" is observable to the agent's
101
+ // active-tracker resolution, mirroring --agent-profile's env write above.
102
+ runtime.proc.env.LIBHARNESS_WORK_TRACKER = opts.workTracker;
103
+
104
+ const { query } = await import("@anthropic-ai/claude-agent-sdk");
105
+ const discusser = createDiscusser({
106
+ leadProfile: opts.leadProfile,
107
+ leadModel: opts.leadModel,
108
+ agentModel: opts.agentModel,
109
+ agentConfigs: opts.agentConfigs,
110
+ discussionId: opts.discussionId,
111
+ resumeContext: opts.resumeContext,
112
+ query,
113
+ output,
114
+ maxTurns: opts.maxTurns,
115
+ maxLeadTurns: opts.maxLeadTurns,
116
+ taskAmend: opts.taskAmend,
117
+ redactor,
118
+ callbackUrl: opts.callbackUrl,
119
+ inboxUrl: opts.inboxUrl,
120
+ correlationId: opts.correlationId,
121
+ runtime,
122
+ });
123
+
124
+ const result = await discusser.run(opts.taskContent);
125
+
126
+ if (fileStream) {
127
+ await new Promise((r) => output.end(r));
128
+ await new Promise((r) => fileStream.end(r));
129
+ }
130
+
131
+ return result.success ? { ok: true } : { ok: false, code: 1, error: "" };
132
+ }
@@ -0,0 +1,123 @@
1
+ import { resolve } from "node:path";
2
+ import { isoTimestamp } from "@forwardimpact/libutil";
3
+ import { createFacilitator } from "../facilitator.js";
4
+ import { createRedactor } from "../redaction.js";
5
+ import { createTeeWriter } from "../tee-writer.js";
6
+ import { resolveTaskContent } from "./task-input.js";
7
+ import { resolveWorkTracker } from "./work-tracker.js";
8
+ import { AGENT_MODEL, LEAD_MODEL } from "@forwardimpact/libutil/models";
9
+
10
+ /**
11
+ * Parse comma-separated agent profile names into structured configs.
12
+ * @param {string} raw - Comma-separated profile names
13
+ * @param {string} cwd - Shared working directory for all agents
14
+ * @returns {Array<{name: string, role: string, cwd: string, agentProfile: string}>}
15
+ */
16
+ function parseAgentProfiles(raw, cwd, maxTurns) {
17
+ return raw.split(",").map((entry) => {
18
+ const name = entry.trim();
19
+ return { name, role: name, cwd, agentProfile: name, maxTurns };
20
+ });
21
+ }
22
+
23
+ /**
24
+ * Parse and validate facilitate command options. Exported for test
25
+ * coverage of the `--max-turns` → per-agent threading contract; not part
26
+ * of the package's public API.
27
+ * @param {object} values - Parsed option values
28
+ * @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
29
+ * @returns {object} Parsed options
30
+ */
31
+ export function parseFacilitateOptions(values, runtime) {
32
+ const { task: taskContent, amend: taskAmend } = resolveTaskContent(
33
+ values,
34
+ runtime,
35
+ );
36
+
37
+ const profilesRaw = values["agent-profiles"];
38
+ if (!profilesRaw) throw new Error("--agent-profiles is required");
39
+ const agentCwd = resolve(values["agent-cwd"] ?? ".");
40
+
41
+ const maxTurnsRaw = values["max-turns"] ?? "20";
42
+ const maxTurns = maxTurnsRaw === "0" ? 0 : parseInt(maxTurnsRaw, 10);
43
+
44
+ // Thread --max-turns into each participant: without this, every facilitated
45
+ // agent silently falls back to the 50-turn default in facilitator.js even
46
+ // when the caller raises the budget. Observed in run 26078312414 where
47
+ // staff-engineer terminated at 51 turns despite --max-turns=200.
48
+ const agentConfigs = parseAgentProfiles(profilesRaw, agentCwd, maxTurns);
49
+
50
+ return {
51
+ taskContent,
52
+ taskAmend,
53
+ agentConfigs,
54
+ facilitatorCwd: resolve(values["facilitator-cwd"] ?? "."),
55
+ agentModel: values["agent-model"] || AGENT_MODEL,
56
+ facilitatorModel: values["lead-model"] || LEAD_MODEL,
57
+ maxTurns,
58
+ outputPath: values.output,
59
+ facilitatorProfile: values["lead-profile"] ?? undefined,
60
+ workTracker: resolveWorkTracker(values, runtime?.proc?.env),
61
+ };
62
+ }
63
+
64
+ /**
65
+ * Facilitate command — run a facilitated multi-agent session.
66
+ *
67
+ * Usage: fit-harness facilitate [options]
68
+ *
69
+ * @param {import("@forwardimpact/libcli").InvocationContext} ctx
70
+ * @returns {Promise<{ok: boolean, code?: number, error?: string}>}
71
+ */
72
+ export async function runFacilitateCommand(ctx) {
73
+ const runtime = ctx.deps.runtime;
74
+ const opts = parseFacilitateOptions(ctx.options, runtime);
75
+
76
+ // Build the redactor as the first observable side-effect after option
77
+ // parsing — the env snapshot must freeze BEFORE any in-process
78
+ // env writes the command performs (e.g. LIBHARNESS_AGENT_PROFILE).
79
+ const redactor = createRedactor({ runtime });
80
+
81
+ const fileStream = opts.outputPath
82
+ ? runtime.fs.createWriteStream(opts.outputPath)
83
+ : null;
84
+ const output = fileStream
85
+ ? createTeeWriter({
86
+ fileStream,
87
+ textStream: runtime.proc.stdout,
88
+ mode: "supervised",
89
+ now: () => isoTimestamp(runtime.clock.now()),
90
+ })
91
+ : runtime.proc.stdout;
92
+
93
+ if (opts.facilitatorProfile) {
94
+ runtime.proc.env.LIBHARNESS_AGENT_PROFILE = opts.facilitatorProfile;
95
+ }
96
+ // Unconditional so the default "github" is observable to the agent's
97
+ // active-tracker resolution, mirroring --agent-profile's env write above.
98
+ runtime.proc.env.LIBHARNESS_WORK_TRACKER = opts.workTracker;
99
+
100
+ const { query } = await import("@anthropic-ai/claude-agent-sdk");
101
+ const facilitator = createFacilitator({
102
+ facilitatorCwd: opts.facilitatorCwd,
103
+ agentConfigs: opts.agentConfigs,
104
+ query,
105
+ output,
106
+ agentModel: opts.agentModel,
107
+ facilitatorModel: opts.facilitatorModel,
108
+ maxTurns: opts.maxTurns,
109
+ facilitatorProfile: opts.facilitatorProfile,
110
+ taskAmend: opts.taskAmend,
111
+ redactor,
112
+ runtime,
113
+ });
114
+
115
+ const result = await facilitator.run(opts.taskContent);
116
+
117
+ if (fileStream) {
118
+ await new Promise((r) => output.end(r));
119
+ await new Promise((r) => fileStream.end(r));
120
+ }
121
+
122
+ return result.success ? { ok: true } : { ok: false, code: 1, error: "" };
123
+ }
@@ -0,0 +1,36 @@
1
+ import { isoTimestamp } from "@forwardimpact/libutil";
2
+ import { createTraceCollector } from "@forwardimpact/libharness";
3
+
4
+ /**
5
+ * Output command — process a complete NDJSON trace from stdin and write
6
+ * formatted output to stdout.
7
+ *
8
+ * Usage: fit-harness output [--format=json|text] < trace.ndjson
9
+ *
10
+ * @param {import("@forwardimpact/libcli").InvocationContext} ctx
11
+ * @returns {Promise<{ok: true}>}
12
+ */
13
+ export async function runOutputCommand(ctx) {
14
+ const values = ctx.options;
15
+ const runtime = ctx.deps.runtime;
16
+ const format =
17
+ values.format === "text" || values.format === "json"
18
+ ? values.format
19
+ : "json";
20
+ const collector = createTraceCollector({
21
+ now: () => isoTimestamp(runtime.clock.now()),
22
+ });
23
+
24
+ // `runtime.proc.stdin` is an AsyncIterable of UTF-8 lines (newline-split by
25
+ // the runtime), so each yielded value is exactly one NDJSON record.
26
+ for await (const line of runtime.proc.stdin) {
27
+ collector.addLine(line);
28
+ }
29
+
30
+ if (format === "text") {
31
+ runtime.proc.stdout.write(collector.toText() + "\n");
32
+ } else {
33
+ runtime.proc.stdout.write(JSON.stringify(collector.toJSON()) + "\n");
34
+ }
35
+ return { ok: true };
36
+ }
@@ -0,0 +1,152 @@
1
+ import { Writable } from "node:stream";
2
+ import { resolve } from "node:path";
3
+ import { isoTimestamp } from "@forwardimpact/libutil";
4
+ import { createAgentRunner } from "../agent-runner.js";
5
+ import { composeProfilePrompt } from "../profile-prompt.js";
6
+ import { createRedactor } from "../redaction.js";
7
+ import { createTeeWriter } from "../tee-writer.js";
8
+ import { SequenceCounter } from "../sequence-counter.js";
9
+ import { resolveWorkTracker } from "./work-tracker.js";
10
+ import { resolveTaskContent } from "./task-input.js";
11
+ import { createServiceConfig } from "@forwardimpact/libconfig";
12
+ import { AGENT_MODEL } from "@forwardimpact/libutil/models";
13
+
14
+ /**
15
+ * Parse and validate run command options from parsed values.
16
+ * @param {object} values - Parsed option values from cli.parse()
17
+ * @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
18
+ * @returns {{ taskContent: string, cwd: string, model: string, maxTurns: number, outputPath: string|undefined, agentProfile: string|undefined, workTracker: string, allowedTools: string[] }}
19
+ */
20
+ export function parseRunOptions(values, runtime) {
21
+ const { task: taskContent, amend: taskAmend } = resolveTaskContent(
22
+ values,
23
+ runtime,
24
+ );
25
+ const maxTurnsRaw = values["max-turns"] ?? "50";
26
+
27
+ return {
28
+ taskContent,
29
+ taskAmend,
30
+ cwd: resolve(values.cwd ?? "."),
31
+ agentModel: values["agent-model"] || AGENT_MODEL,
32
+ maxTurns: maxTurnsRaw === "0" ? 0 : parseInt(maxTurnsRaw, 10),
33
+ outputPath: values.output,
34
+ agentProfile: values["agent-profile"] ?? undefined,
35
+ workTracker: resolveWorkTracker(values, runtime?.proc?.env),
36
+ allowedTools: (
37
+ values["allowed-tools"] ??
38
+ "Bash,Read,Glob,Grep,Write,Edit,Agent,TodoWrite"
39
+ ).split(","),
40
+ mcpServer: values["mcp-server"] ?? undefined,
41
+ };
42
+ }
43
+
44
+ /**
45
+ * Run command — execute a single agent via the Claude Agent SDK.
46
+ *
47
+ * Usage: fit-harness run [options]
48
+ *
49
+ * @param {import("@forwardimpact/libcli").InvocationContext} ctx
50
+ * @returns {Promise<{ok: boolean, code?: number, error?: string}>}
51
+ */
52
+ export async function runRunCommand(ctx) {
53
+ const runtime = ctx.deps.runtime;
54
+ const {
55
+ taskContent,
56
+ taskAmend,
57
+ cwd,
58
+ agentModel,
59
+ maxTurns,
60
+ outputPath,
61
+ agentProfile,
62
+ workTracker,
63
+ allowedTools,
64
+ mcpServer,
65
+ } = parseRunOptions(ctx.options, runtime);
66
+
67
+ // Build the redactor as the first observable side-effect after option
68
+ // parsing — the env snapshot must freeze BEFORE any in-process
69
+ // env writes the command performs (e.g. LIBHARNESS_AGENT_PROFILE).
70
+ const redactor = createRedactor({ runtime });
71
+
72
+ // When --output is specified, stream text to stdout while writing NDJSON to file.
73
+ // Otherwise, write NDJSON directly to stdout (backwards-compatible).
74
+ const fileStream = outputPath
75
+ ? runtime.fs.createWriteStream(outputPath)
76
+ : null;
77
+ const output = fileStream
78
+ ? createTeeWriter({
79
+ fileStream,
80
+ textStream: runtime.proc.stdout,
81
+ mode: "raw",
82
+ now: () => isoTimestamp(runtime.clock.now()),
83
+ })
84
+ : runtime.proc.stdout;
85
+
86
+ const counter = new SequenceCounter();
87
+ const devNull = new Writable({
88
+ write(_chunk, _enc, cb) {
89
+ cb();
90
+ },
91
+ });
92
+ const onLine = (line) => {
93
+ const event = JSON.parse(line);
94
+ const tagged = { source: "agent", seq: counter.next(), event };
95
+ output.write(JSON.stringify(redactor.redactValue(tagged)) + "\n");
96
+ };
97
+
98
+ let mcpServers = null;
99
+ if (mcpServer) {
100
+ const mcpConfig = await createServiceConfig("mcp");
101
+ mcpServers = {
102
+ [mcpServer]: {
103
+ type: "http",
104
+ url: mcpConfig.url,
105
+ headers: { Authorization: `Bearer ${mcpConfig.mcpToken()}` },
106
+ },
107
+ };
108
+ allowedTools.push(`mcp__${mcpServer}__*`);
109
+ }
110
+
111
+ if (agentProfile) {
112
+ runtime.proc.env.LIBHARNESS_AGENT_PROFILE = agentProfile;
113
+ }
114
+ // Unconditional so the default "github" is observable to the agent's
115
+ // active-tracker resolution, mirroring --agent-profile's env write above.
116
+ runtime.proc.env.LIBHARNESS_WORK_TRACKER = workTracker;
117
+
118
+ const systemPrompt = agentProfile
119
+ ? composeProfilePrompt(agentProfile, {
120
+ profilesDir: resolve(cwd, ".claude/agents"),
121
+ runtime,
122
+ })
123
+ : undefined;
124
+
125
+ const { query } = await import("@anthropic-ai/claude-agent-sdk");
126
+ const runner = createAgentRunner({
127
+ cwd,
128
+ query,
129
+ output: devNull,
130
+ model: agentModel,
131
+ maxTurns,
132
+ allowedTools,
133
+ onLine,
134
+ settingSources: ["project"],
135
+ systemPrompt,
136
+ taskAmend,
137
+ mcpServers,
138
+ redactor,
139
+ runtime,
140
+ });
141
+
142
+ const result = await runner.run(taskContent);
143
+
144
+ if (fileStream) {
145
+ await new Promise((r) => output.end(r));
146
+ await new Promise((r) => fileStream.end(r));
147
+ }
148
+
149
+ return result.success
150
+ ? { ok: true }
151
+ : { ok: false, code: 1, error: result.error?.message ?? "" };
152
+ }