@forwardimpact/libharness 0.1.22 → 1.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -201
- package/README.md +196 -80
- package/bin/fit-benchmark.js +44 -0
- package/bin/fit-harness.js +358 -0
- package/bin/fit-selfedit.js +165 -0
- package/bin/fit-trace.js +510 -0
- package/package.json +41 -11
- package/src/agent-runner.js +256 -0
- package/src/benchmark/apm-installer.js +207 -0
- package/src/benchmark/env-loader.js +158 -0
- package/src/benchmark/hook-env.js +40 -0
- package/src/benchmark/invariants.js +141 -0
- package/src/benchmark/judge.js +187 -0
- package/src/benchmark/npm-installer.js +87 -0
- package/src/benchmark/report.js +604 -0
- package/src/benchmark/result.js +127 -0
- package/src/benchmark/runner.js +688 -0
- package/src/benchmark/scheduler.js +78 -0
- package/src/benchmark/task-family.js +260 -0
- package/src/benchmark/workdir.js +344 -0
- package/src/commands/assert.js +153 -0
- package/src/commands/benchmark-definition.js +175 -0
- package/src/commands/benchmark-invariants.js +73 -0
- package/src/commands/benchmark-report.js +51 -0
- package/src/commands/benchmark-run.js +175 -0
- package/src/commands/by-discussion.js +94 -0
- package/src/commands/callback.js +119 -0
- package/src/commands/discuss.js +132 -0
- package/src/commands/facilitate.js +123 -0
- package/src/commands/output.js +36 -0
- package/src/commands/run.js +152 -0
- package/src/commands/supervise.js +136 -0
- package/src/commands/task-input.js +54 -0
- package/src/commands/tee.js +53 -0
- package/src/commands/trace.js +630 -0
- package/src/commands/work-tracker.js +35 -0
- package/src/cost.js +79 -0
- package/src/discuss-tools.js +173 -0
- package/src/discusser.js +394 -0
- package/src/events/github.js +161 -0
- package/src/facilitator.js +205 -0
- package/src/inbox-poller.js +81 -0
- package/src/index.js +72 -2
- package/src/judge.js +210 -0
- package/src/message-bus.js +118 -0
- package/src/orchestration-loop.js +330 -0
- package/src/orchestration-toolkit.js +441 -0
- package/src/orchestrator-helpers.js +23 -0
- package/src/profile-prompt.js +266 -0
- package/src/redaction.js +253 -0
- package/src/render/line-renderer.js +54 -0
- package/src/render/orchestrator-filter.js +19 -0
- package/src/render/palette.js +63 -0
- package/src/render/tool-hints.js +154 -0
- package/src/render/turn-renderer.js +96 -0
- package/src/reply-emitter.js +47 -0
- package/src/sequence-counter.js +21 -0
- package/src/signature-filter.js +27 -0
- package/src/supervisor.js +236 -0
- package/src/tee-writer.js +150 -0
- package/src/trace-collector.js +444 -0
- package/src/trace-github.js +473 -0
- package/src/trace-multi.js +101 -0
- package/src/trace-query.js +748 -0
- package/src/trace-render.js +211 -0
- package/src/trace-usage.js +249 -0
- package/src/fixture/assertions.js +0 -42
- package/src/fixture/cache.js +0 -50
- package/src/fixture/eval.js +0 -146
- package/src/fixture/index.js +0 -9
- package/src/fixture/pathway.js +0 -451
- package/src/fixture/services.js +0 -56
- package/src/mock/clients.js +0 -135
- package/src/mock/config.js +0 -45
- package/src/mock/data.js +0 -46
- package/src/mock/fs.js +0 -111
- package/src/mock/grpc.js +0 -94
- package/src/mock/http.js +0 -60
- package/src/mock/index.js +0 -36
- package/src/mock/infra.js +0 -219
- package/src/mock/logger.js +0 -42
- package/src/mock/observer.js +0 -74
- package/src/mock/resource-index.js +0 -95
- package/src/mock/service-callbacks.js +0 -39
- package/src/mock/services.js +0 -79
- package/src/mock/spy.js +0 -44
- package/src/mock/storage.js +0 -118
|
@@ -0,0 +1,94 @@
|
|
|
1
|
+
import { join } from "node:path";
|
|
2
|
+
|
|
3
|
+
const FIRST_LINE_CAP = 64 * 1024;
|
|
4
|
+
|
|
5
|
+
/**
|
|
6
|
+
* Read the first newline-terminated line of a file, bounded to the first
|
|
7
|
+
* {@link FIRST_LINE_CAP} bytes. Trace `.ndjson` files can be many MB; the
|
|
8
|
+
* Step 2.6 meta header is always small, so a bounded positional read avoids
|
|
9
|
+
* loading whole files into memory just to inspect the header. The positional
|
|
10
|
+
* `openSync`/`readSync`/`closeSync` trio is read off the injected
|
|
11
|
+
* `runtime.fsSync` surface.
|
|
12
|
+
*
|
|
13
|
+
* @param {object} fsSync - Sync filesystem surface (`runtime.fsSync`).
|
|
14
|
+
* @param {string} path
|
|
15
|
+
* @returns {string}
|
|
16
|
+
*/
|
|
17
|
+
function readFirstLine(fsSync, path) {
|
|
18
|
+
const fd = fsSync.openSync(path, "r");
|
|
19
|
+
try {
|
|
20
|
+
const buf = Buffer.alloc(FIRST_LINE_CAP);
|
|
21
|
+
const bytes = fsSync.readSync(fd, buf, 0, buf.length, 0);
|
|
22
|
+
const text = buf.toString("utf8", 0, bytes);
|
|
23
|
+
const nl = text.indexOf("\n");
|
|
24
|
+
return nl === -1 ? text : text.slice(0, nl);
|
|
25
|
+
} finally {
|
|
26
|
+
fsSync.closeSync(fd);
|
|
27
|
+
}
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
/**
|
|
31
|
+
* Scan a directory for `.ndjson` files whose meta header carries the
|
|
32
|
+
* given discussion_id. The Step 2.6 first-line guarantee makes the
|
|
33
|
+
* lookup cheap: we read only the first line per file. Files without a
|
|
34
|
+
* meta header (e.g. legacy supervise/facilitate traces) are skipped
|
|
35
|
+
* silently — not erroneous.
|
|
36
|
+
*
|
|
37
|
+
* @param {string} dir
|
|
38
|
+
* @param {string} discussionId
|
|
39
|
+
* @param {object} fsSync - Sync filesystem surface (`runtime.fsSync`).
|
|
40
|
+
* @returns {Array<{path: string, mtimeMs: number}>}
|
|
41
|
+
*/
|
|
42
|
+
export function findTracesByDiscussion(dir, discussionId, fsSync) {
|
|
43
|
+
const matches = [];
|
|
44
|
+
let entries;
|
|
45
|
+
try {
|
|
46
|
+
entries = fsSync.readdirSync(dir);
|
|
47
|
+
} catch {
|
|
48
|
+
return [];
|
|
49
|
+
}
|
|
50
|
+
for (const entry of entries) {
|
|
51
|
+
if (!entry.endsWith(".ndjson")) continue;
|
|
52
|
+
const path = join(dir, entry);
|
|
53
|
+
let firstLine;
|
|
54
|
+
try {
|
|
55
|
+
firstLine = readFirstLine(fsSync, path);
|
|
56
|
+
} catch {
|
|
57
|
+
continue;
|
|
58
|
+
}
|
|
59
|
+
let parsed;
|
|
60
|
+
try {
|
|
61
|
+
parsed = JSON.parse(firstLine);
|
|
62
|
+
} catch {
|
|
63
|
+
continue;
|
|
64
|
+
}
|
|
65
|
+
const event = parsed.event ?? parsed;
|
|
66
|
+
if (event?.type !== "meta") continue;
|
|
67
|
+
if (event.discussion_id !== discussionId) continue;
|
|
68
|
+
matches.push({ path, mtimeMs: fsSync.statSync(path).mtimeMs });
|
|
69
|
+
}
|
|
70
|
+
matches.sort((a, b) => a.mtimeMs - b.mtimeMs);
|
|
71
|
+
return matches;
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
/**
|
|
75
|
+
* `fit-trace by-discussion <discussion-id> [trace-dir]` — list trace
|
|
76
|
+
* files whose meta header carries the given discussion_id, one per
|
|
77
|
+
* line, ordered by first-event timestamp (file mtime ascending). The
|
|
78
|
+
* result is usable with `xargs cat` for a chronological merge.
|
|
79
|
+
*
|
|
80
|
+
* @param {import("@forwardimpact/libcli").InvocationContext} ctx
|
|
81
|
+
* @returns {Promise<{ok: true} | {ok: false, code: number, error: string}>}
|
|
82
|
+
*/
|
|
83
|
+
export async function runByDiscussionCommand(ctx) {
|
|
84
|
+
const runtime = ctx.deps.runtime;
|
|
85
|
+
const discussionId = ctx.args["discussion-id"];
|
|
86
|
+
if (!discussionId)
|
|
87
|
+
return { ok: false, code: 1, error: "<discussion-id> is required" };
|
|
88
|
+
const dir = ctx.args["trace-dir"] ?? ctx.options["trace-dir"] ?? "traces";
|
|
89
|
+
const matches = findTracesByDiscussion(dir, discussionId, runtime.fsSync);
|
|
90
|
+
for (const { path } of matches) {
|
|
91
|
+
runtime.proc.stdout.write(`${path}\n`);
|
|
92
|
+
}
|
|
93
|
+
return { ok: true };
|
|
94
|
+
}
|
|
@@ -0,0 +1,119 @@
|
|
|
1
|
+
import { sumTraceCost } from "../cost.js";
|
|
2
|
+
|
|
3
|
+
/**
|
|
4
|
+
* Scan an NDJSON trace and return the last orchestrator summary event,
|
|
5
|
+
* the first `meta` event's `discussion_id`, and any structured replies
|
|
6
|
+
* collected by the discusser. Skips malformed lines.
|
|
7
|
+
*
|
|
8
|
+
* The runner is verdict-agnostic — verbatim passthrough of whatever the
|
|
9
|
+
* trace carries ("success"/"failure" from supervise/facilitate; canonical
|
|
10
|
+
* "adjourned"/"recessed"/"failed" from discuss). The bridge layer maps to
|
|
11
|
+
* its channel semantics.
|
|
12
|
+
*
|
|
13
|
+
* @param {string} content - Raw NDJSON trace content.
|
|
14
|
+
* @returns {{verdict: string, summary: string, replies: object[], trigger?: object, discussionId?: string} | null}
|
|
15
|
+
*/
|
|
16
|
+
// biome-ignore lint/complexity/noExcessiveCognitiveComplexity: NDJSON scan with malformed-line tolerance + meta/summary dual extraction
|
|
17
|
+
function readTraceSummary(content) {
|
|
18
|
+
let summary = null;
|
|
19
|
+
let metaDiscussionId = null;
|
|
20
|
+
for (const line of content.split("\n")) {
|
|
21
|
+
if (!line.trim()) continue;
|
|
22
|
+
let record;
|
|
23
|
+
try {
|
|
24
|
+
record = JSON.parse(line);
|
|
25
|
+
} catch {
|
|
26
|
+
continue;
|
|
27
|
+
}
|
|
28
|
+
if (record.source !== "orchestrator") continue;
|
|
29
|
+
if (record.event?.type === "meta" && !metaDiscussionId) {
|
|
30
|
+
metaDiscussionId = record.event.discussion_id ?? null;
|
|
31
|
+
}
|
|
32
|
+
if (record.event?.type === "summary") {
|
|
33
|
+
summary = {
|
|
34
|
+
verdict: record.event.verdict ?? "failed",
|
|
35
|
+
summary: record.event.summary ?? "",
|
|
36
|
+
replies: Array.isArray(record.event.replies)
|
|
37
|
+
? record.event.replies
|
|
38
|
+
: [],
|
|
39
|
+
...(record.event.trigger && { trigger: record.event.trigger }),
|
|
40
|
+
...(record.event.discussion_id && {
|
|
41
|
+
discussionId: record.event.discussion_id,
|
|
42
|
+
}),
|
|
43
|
+
...(typeof record.event.lastActedSeq === "number" && {
|
|
44
|
+
lastActedSeq: record.event.lastActedSeq,
|
|
45
|
+
}),
|
|
46
|
+
};
|
|
47
|
+
}
|
|
48
|
+
}
|
|
49
|
+
if (summary && !summary.discussionId && metaDiscussionId) {
|
|
50
|
+
summary.discussionId = metaDiscussionId;
|
|
51
|
+
}
|
|
52
|
+
return summary;
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
/**
|
|
56
|
+
* Callback command — read an NDJSON trace, extract the terminal
|
|
57
|
+
* orchestrator summary, and POST a canonical callback body to the
|
|
58
|
+
* configured URL. Used by `kata-dispatch.yml` to deliver the lead's
|
|
59
|
+
* conclusion to the bridge that dispatched the run.
|
|
60
|
+
*
|
|
61
|
+
* Wire shape (single shape across modes):
|
|
62
|
+
*
|
|
63
|
+
* ```
|
|
64
|
+
* {
|
|
65
|
+
* correlation_id, verdict, summary, run_url,
|
|
66
|
+
* discussion_id?, replies: [], trigger?
|
|
67
|
+
* }
|
|
68
|
+
* ```
|
|
69
|
+
*
|
|
70
|
+
* @param {import("@forwardimpact/libcli").InvocationContext} ctx
|
|
71
|
+
* @returns {Promise<{ok: true} | {ok: false, code: number, error: string}>}
|
|
72
|
+
*/
|
|
73
|
+
export async function runCallbackCommand(ctx) {
|
|
74
|
+
const values = ctx.options;
|
|
75
|
+
const runtime = ctx.deps.runtime;
|
|
76
|
+
const traceFile = values["trace-file"];
|
|
77
|
+
const callbackUrl = values["callback-url"];
|
|
78
|
+
const correlationId = values["correlation-id"];
|
|
79
|
+
const runUrl = values["run-url"] ?? "";
|
|
80
|
+
const discussionIdOverride = values["discussion-id"] ?? null;
|
|
81
|
+
|
|
82
|
+
if (!traceFile)
|
|
83
|
+
return { ok: false, code: 1, error: "--trace-file is required" };
|
|
84
|
+
if (!callbackUrl)
|
|
85
|
+
return { ok: false, code: 1, error: "--callback-url is required" };
|
|
86
|
+
|
|
87
|
+
const content = runtime.fsSync.readFileSync(traceFile, "utf8");
|
|
88
|
+
const found = readTraceSummary(content) ?? {
|
|
89
|
+
verdict: "failed",
|
|
90
|
+
summary: "Run ended without producing a summary.",
|
|
91
|
+
replies: [],
|
|
92
|
+
};
|
|
93
|
+
// Total spend across every participant in the trace — the bridge surfaces
|
|
94
|
+
// it alongside the verdict so a dispatched run reports what it cost.
|
|
95
|
+
const { totalCostUsd } = sumTraceCost(content.split("\n"));
|
|
96
|
+
|
|
97
|
+
const discussionId = found.discussionId ?? discussionIdOverride ?? null;
|
|
98
|
+
const payload = {
|
|
99
|
+
correlation_id: correlationId,
|
|
100
|
+
kind: "terminal",
|
|
101
|
+
verdict: found.verdict,
|
|
102
|
+
summary: found.summary,
|
|
103
|
+
run_url: runUrl,
|
|
104
|
+
cost_usd: totalCostUsd,
|
|
105
|
+
replies: found.replies,
|
|
106
|
+
last_acted_seq: found.lastActedSeq ?? -1,
|
|
107
|
+
...(discussionId && { discussion_id: discussionId }),
|
|
108
|
+
...(found.trigger && { trigger: found.trigger }),
|
|
109
|
+
};
|
|
110
|
+
const res = await fetch(callbackUrl, {
|
|
111
|
+
method: "POST",
|
|
112
|
+
headers: { "Content-Type": "application/json" },
|
|
113
|
+
body: JSON.stringify(payload),
|
|
114
|
+
});
|
|
115
|
+
if (!res.ok) {
|
|
116
|
+
return { ok: false, code: 1, error: `Callback POST failed: ${res.status}` };
|
|
117
|
+
}
|
|
118
|
+
return { ok: true };
|
|
119
|
+
}
|
|
@@ -0,0 +1,132 @@
|
|
|
1
|
+
import { resolve } from "node:path";
|
|
2
|
+
import { isoTimestamp } from "@forwardimpact/libutil";
|
|
3
|
+
import { createDiscusser } from "../discusser.js";
|
|
4
|
+
import { createRedactor } from "../redaction.js";
|
|
5
|
+
import { createTeeWriter } from "../tee-writer.js";
|
|
6
|
+
import { resolveTaskContent } from "./task-input.js";
|
|
7
|
+
import { resolveWorkTracker } from "./work-tracker.js";
|
|
8
|
+
import { AGENT_MODEL, LEAD_MODEL } from "@forwardimpact/libutil/models";
|
|
9
|
+
|
|
10
|
+
function parseAgentProfiles(raw, cwd, maxTurns) {
|
|
11
|
+
if (!raw) return [];
|
|
12
|
+
return raw.split(",").map((entry) => {
|
|
13
|
+
const name = entry.trim();
|
|
14
|
+
return { name, role: name, cwd, agentProfile: name, maxTurns };
|
|
15
|
+
});
|
|
16
|
+
}
|
|
17
|
+
|
|
18
|
+
/**
|
|
19
|
+
* Parse and validate discuss command options. Exported so tests can verify
|
|
20
|
+
* defaults and the legacy-flag clean break.
|
|
21
|
+
* @param {object} values - Parsed option values
|
|
22
|
+
* @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
|
|
23
|
+
* @returns {object}
|
|
24
|
+
*/
|
|
25
|
+
export function parseDiscussOptions(values, runtime) {
|
|
26
|
+
const { task: taskContent, amend: taskAmend } = resolveTaskContent(
|
|
27
|
+
values,
|
|
28
|
+
runtime,
|
|
29
|
+
);
|
|
30
|
+
|
|
31
|
+
const profilesRaw = values["agent-profiles"];
|
|
32
|
+
const agentCwd = resolve(values["agent-cwd"] ?? ".");
|
|
33
|
+
|
|
34
|
+
const maxTurnsRaw = values["max-turns"] ?? "40";
|
|
35
|
+
const maxTurns = maxTurnsRaw === "0" ? 0 : parseInt(maxTurnsRaw, 10);
|
|
36
|
+
|
|
37
|
+
const agentConfigs = parseAgentProfiles(profilesRaw, agentCwd, maxTurns);
|
|
38
|
+
|
|
39
|
+
const resumeContextRaw = values["resume-context"];
|
|
40
|
+
let resumeContext = null;
|
|
41
|
+
if (resumeContextRaw) {
|
|
42
|
+
try {
|
|
43
|
+
resumeContext = JSON.parse(resumeContextRaw);
|
|
44
|
+
} catch (err) {
|
|
45
|
+
throw new Error(`--resume-context is not valid JSON: ${err.message}`);
|
|
46
|
+
}
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
const maxLeadTurnsRaw = values["max-lead-turns"] ?? "200";
|
|
50
|
+
const maxLeadTurns = parseInt(maxLeadTurnsRaw, 10);
|
|
51
|
+
|
|
52
|
+
return {
|
|
53
|
+
taskContent,
|
|
54
|
+
taskAmend,
|
|
55
|
+
agentConfigs,
|
|
56
|
+
leadProfile: values["lead-profile"] ?? undefined,
|
|
57
|
+
leadModel: values["lead-model"] || LEAD_MODEL,
|
|
58
|
+
agentModel: values["agent-model"] || AGENT_MODEL,
|
|
59
|
+
maxTurns,
|
|
60
|
+
maxLeadTurns,
|
|
61
|
+
outputPath: values.output,
|
|
62
|
+
workTracker: resolveWorkTracker(values, runtime?.proc?.env),
|
|
63
|
+
discussionId: values["discussion-id"] ?? null,
|
|
64
|
+
resumeContext,
|
|
65
|
+
callbackUrl: runtime.proc.env.CALLBACK_URL ?? null,
|
|
66
|
+
inboxUrl: runtime.proc.env.INBOX_URL ?? null,
|
|
67
|
+
correlationId: runtime.proc.env.CORRELATION_ID ?? null,
|
|
68
|
+
};
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
/**
|
|
72
|
+
* Discuss command — run a discusser-led session with suspend/resume
|
|
73
|
+
* semantics, threading `discussion_id` through the trace so multi-run
|
|
74
|
+
* conversations are queryable as one.
|
|
75
|
+
*
|
|
76
|
+
* @param {import("@forwardimpact/libcli").InvocationContext} ctx
|
|
77
|
+
* @returns {Promise<{ok: boolean, code?: number, error?: string}>}
|
|
78
|
+
*/
|
|
79
|
+
export async function runDiscussCommand(ctx) {
|
|
80
|
+
const runtime = ctx.deps.runtime;
|
|
81
|
+
const opts = parseDiscussOptions(ctx.options, runtime);
|
|
82
|
+
|
|
83
|
+
const redactor = createRedactor({ runtime });
|
|
84
|
+
|
|
85
|
+
const fileStream = opts.outputPath
|
|
86
|
+
? runtime.fs.createWriteStream(opts.outputPath)
|
|
87
|
+
: null;
|
|
88
|
+
const output = fileStream
|
|
89
|
+
? createTeeWriter({
|
|
90
|
+
fileStream,
|
|
91
|
+
textStream: runtime.proc.stdout,
|
|
92
|
+
mode: "supervised",
|
|
93
|
+
now: () => isoTimestamp(runtime.clock.now()),
|
|
94
|
+
})
|
|
95
|
+
: runtime.proc.stdout;
|
|
96
|
+
|
|
97
|
+
if (opts.leadProfile) {
|
|
98
|
+
runtime.proc.env.LIBHARNESS_AGENT_PROFILE = opts.leadProfile;
|
|
99
|
+
}
|
|
100
|
+
// Unconditional so the default "github" is observable to the agent's
|
|
101
|
+
// active-tracker resolution, mirroring --agent-profile's env write above.
|
|
102
|
+
runtime.proc.env.LIBHARNESS_WORK_TRACKER = opts.workTracker;
|
|
103
|
+
|
|
104
|
+
const { query } = await import("@anthropic-ai/claude-agent-sdk");
|
|
105
|
+
const discusser = createDiscusser({
|
|
106
|
+
leadProfile: opts.leadProfile,
|
|
107
|
+
leadModel: opts.leadModel,
|
|
108
|
+
agentModel: opts.agentModel,
|
|
109
|
+
agentConfigs: opts.agentConfigs,
|
|
110
|
+
discussionId: opts.discussionId,
|
|
111
|
+
resumeContext: opts.resumeContext,
|
|
112
|
+
query,
|
|
113
|
+
output,
|
|
114
|
+
maxTurns: opts.maxTurns,
|
|
115
|
+
maxLeadTurns: opts.maxLeadTurns,
|
|
116
|
+
taskAmend: opts.taskAmend,
|
|
117
|
+
redactor,
|
|
118
|
+
callbackUrl: opts.callbackUrl,
|
|
119
|
+
inboxUrl: opts.inboxUrl,
|
|
120
|
+
correlationId: opts.correlationId,
|
|
121
|
+
runtime,
|
|
122
|
+
});
|
|
123
|
+
|
|
124
|
+
const result = await discusser.run(opts.taskContent);
|
|
125
|
+
|
|
126
|
+
if (fileStream) {
|
|
127
|
+
await new Promise((r) => output.end(r));
|
|
128
|
+
await new Promise((r) => fileStream.end(r));
|
|
129
|
+
}
|
|
130
|
+
|
|
131
|
+
return result.success ? { ok: true } : { ok: false, code: 1, error: "" };
|
|
132
|
+
}
|
|
@@ -0,0 +1,123 @@
|
|
|
1
|
+
import { resolve } from "node:path";
|
|
2
|
+
import { isoTimestamp } from "@forwardimpact/libutil";
|
|
3
|
+
import { createFacilitator } from "../facilitator.js";
|
|
4
|
+
import { createRedactor } from "../redaction.js";
|
|
5
|
+
import { createTeeWriter } from "../tee-writer.js";
|
|
6
|
+
import { resolveTaskContent } from "./task-input.js";
|
|
7
|
+
import { resolveWorkTracker } from "./work-tracker.js";
|
|
8
|
+
import { AGENT_MODEL, LEAD_MODEL } from "@forwardimpact/libutil/models";
|
|
9
|
+
|
|
10
|
+
/**
|
|
11
|
+
* Parse comma-separated agent profile names into structured configs.
|
|
12
|
+
* @param {string} raw - Comma-separated profile names
|
|
13
|
+
* @param {string} cwd - Shared working directory for all agents
|
|
14
|
+
* @returns {Array<{name: string, role: string, cwd: string, agentProfile: string}>}
|
|
15
|
+
*/
|
|
16
|
+
function parseAgentProfiles(raw, cwd, maxTurns) {
|
|
17
|
+
return raw.split(",").map((entry) => {
|
|
18
|
+
const name = entry.trim();
|
|
19
|
+
return { name, role: name, cwd, agentProfile: name, maxTurns };
|
|
20
|
+
});
|
|
21
|
+
}
|
|
22
|
+
|
|
23
|
+
/**
|
|
24
|
+
* Parse and validate facilitate command options. Exported for test
|
|
25
|
+
* coverage of the `--max-turns` → per-agent threading contract; not part
|
|
26
|
+
* of the package's public API.
|
|
27
|
+
* @param {object} values - Parsed option values
|
|
28
|
+
* @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
|
|
29
|
+
* @returns {object} Parsed options
|
|
30
|
+
*/
|
|
31
|
+
export function parseFacilitateOptions(values, runtime) {
|
|
32
|
+
const { task: taskContent, amend: taskAmend } = resolveTaskContent(
|
|
33
|
+
values,
|
|
34
|
+
runtime,
|
|
35
|
+
);
|
|
36
|
+
|
|
37
|
+
const profilesRaw = values["agent-profiles"];
|
|
38
|
+
if (!profilesRaw) throw new Error("--agent-profiles is required");
|
|
39
|
+
const agentCwd = resolve(values["agent-cwd"] ?? ".");
|
|
40
|
+
|
|
41
|
+
const maxTurnsRaw = values["max-turns"] ?? "20";
|
|
42
|
+
const maxTurns = maxTurnsRaw === "0" ? 0 : parseInt(maxTurnsRaw, 10);
|
|
43
|
+
|
|
44
|
+
// Thread --max-turns into each participant: without this, every facilitated
|
|
45
|
+
// agent silently falls back to the 50-turn default in facilitator.js even
|
|
46
|
+
// when the caller raises the budget. Observed in run 26078312414 where
|
|
47
|
+
// staff-engineer terminated at 51 turns despite --max-turns=200.
|
|
48
|
+
const agentConfigs = parseAgentProfiles(profilesRaw, agentCwd, maxTurns);
|
|
49
|
+
|
|
50
|
+
return {
|
|
51
|
+
taskContent,
|
|
52
|
+
taskAmend,
|
|
53
|
+
agentConfigs,
|
|
54
|
+
facilitatorCwd: resolve(values["facilitator-cwd"] ?? "."),
|
|
55
|
+
agentModel: values["agent-model"] || AGENT_MODEL,
|
|
56
|
+
facilitatorModel: values["lead-model"] || LEAD_MODEL,
|
|
57
|
+
maxTurns,
|
|
58
|
+
outputPath: values.output,
|
|
59
|
+
facilitatorProfile: values["lead-profile"] ?? undefined,
|
|
60
|
+
workTracker: resolveWorkTracker(values, runtime?.proc?.env),
|
|
61
|
+
};
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
/**
|
|
65
|
+
* Facilitate command — run a facilitated multi-agent session.
|
|
66
|
+
*
|
|
67
|
+
* Usage: fit-harness facilitate [options]
|
|
68
|
+
*
|
|
69
|
+
* @param {import("@forwardimpact/libcli").InvocationContext} ctx
|
|
70
|
+
* @returns {Promise<{ok: boolean, code?: number, error?: string}>}
|
|
71
|
+
*/
|
|
72
|
+
export async function runFacilitateCommand(ctx) {
|
|
73
|
+
const runtime = ctx.deps.runtime;
|
|
74
|
+
const opts = parseFacilitateOptions(ctx.options, runtime);
|
|
75
|
+
|
|
76
|
+
// Build the redactor as the first observable side-effect after option
|
|
77
|
+
// parsing — the env snapshot must freeze BEFORE any in-process
|
|
78
|
+
// env writes the command performs (e.g. LIBHARNESS_AGENT_PROFILE).
|
|
79
|
+
const redactor = createRedactor({ runtime });
|
|
80
|
+
|
|
81
|
+
const fileStream = opts.outputPath
|
|
82
|
+
? runtime.fs.createWriteStream(opts.outputPath)
|
|
83
|
+
: null;
|
|
84
|
+
const output = fileStream
|
|
85
|
+
? createTeeWriter({
|
|
86
|
+
fileStream,
|
|
87
|
+
textStream: runtime.proc.stdout,
|
|
88
|
+
mode: "supervised",
|
|
89
|
+
now: () => isoTimestamp(runtime.clock.now()),
|
|
90
|
+
})
|
|
91
|
+
: runtime.proc.stdout;
|
|
92
|
+
|
|
93
|
+
if (opts.facilitatorProfile) {
|
|
94
|
+
runtime.proc.env.LIBHARNESS_AGENT_PROFILE = opts.facilitatorProfile;
|
|
95
|
+
}
|
|
96
|
+
// Unconditional so the default "github" is observable to the agent's
|
|
97
|
+
// active-tracker resolution, mirroring --agent-profile's env write above.
|
|
98
|
+
runtime.proc.env.LIBHARNESS_WORK_TRACKER = opts.workTracker;
|
|
99
|
+
|
|
100
|
+
const { query } = await import("@anthropic-ai/claude-agent-sdk");
|
|
101
|
+
const facilitator = createFacilitator({
|
|
102
|
+
facilitatorCwd: opts.facilitatorCwd,
|
|
103
|
+
agentConfigs: opts.agentConfigs,
|
|
104
|
+
query,
|
|
105
|
+
output,
|
|
106
|
+
agentModel: opts.agentModel,
|
|
107
|
+
facilitatorModel: opts.facilitatorModel,
|
|
108
|
+
maxTurns: opts.maxTurns,
|
|
109
|
+
facilitatorProfile: opts.facilitatorProfile,
|
|
110
|
+
taskAmend: opts.taskAmend,
|
|
111
|
+
redactor,
|
|
112
|
+
runtime,
|
|
113
|
+
});
|
|
114
|
+
|
|
115
|
+
const result = await facilitator.run(opts.taskContent);
|
|
116
|
+
|
|
117
|
+
if (fileStream) {
|
|
118
|
+
await new Promise((r) => output.end(r));
|
|
119
|
+
await new Promise((r) => fileStream.end(r));
|
|
120
|
+
}
|
|
121
|
+
|
|
122
|
+
return result.success ? { ok: true } : { ok: false, code: 1, error: "" };
|
|
123
|
+
}
|
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
import { isoTimestamp } from "@forwardimpact/libutil";
|
|
2
|
+
import { createTraceCollector } from "@forwardimpact/libharness";
|
|
3
|
+
|
|
4
|
+
/**
|
|
5
|
+
* Output command — process a complete NDJSON trace from stdin and write
|
|
6
|
+
* formatted output to stdout.
|
|
7
|
+
*
|
|
8
|
+
* Usage: fit-harness output [--format=json|text] < trace.ndjson
|
|
9
|
+
*
|
|
10
|
+
* @param {import("@forwardimpact/libcli").InvocationContext} ctx
|
|
11
|
+
* @returns {Promise<{ok: true}>}
|
|
12
|
+
*/
|
|
13
|
+
export async function runOutputCommand(ctx) {
|
|
14
|
+
const values = ctx.options;
|
|
15
|
+
const runtime = ctx.deps.runtime;
|
|
16
|
+
const format =
|
|
17
|
+
values.format === "text" || values.format === "json"
|
|
18
|
+
? values.format
|
|
19
|
+
: "json";
|
|
20
|
+
const collector = createTraceCollector({
|
|
21
|
+
now: () => isoTimestamp(runtime.clock.now()),
|
|
22
|
+
});
|
|
23
|
+
|
|
24
|
+
// `runtime.proc.stdin` is an AsyncIterable of UTF-8 lines (newline-split by
|
|
25
|
+
// the runtime), so each yielded value is exactly one NDJSON record.
|
|
26
|
+
for await (const line of runtime.proc.stdin) {
|
|
27
|
+
collector.addLine(line);
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
if (format === "text") {
|
|
31
|
+
runtime.proc.stdout.write(collector.toText() + "\n");
|
|
32
|
+
} else {
|
|
33
|
+
runtime.proc.stdout.write(JSON.stringify(collector.toJSON()) + "\n");
|
|
34
|
+
}
|
|
35
|
+
return { ok: true };
|
|
36
|
+
}
|
|
@@ -0,0 +1,152 @@
|
|
|
1
|
+
import { Writable } from "node:stream";
|
|
2
|
+
import { resolve } from "node:path";
|
|
3
|
+
import { isoTimestamp } from "@forwardimpact/libutil";
|
|
4
|
+
import { createAgentRunner } from "../agent-runner.js";
|
|
5
|
+
import { composeProfilePrompt } from "../profile-prompt.js";
|
|
6
|
+
import { createRedactor } from "../redaction.js";
|
|
7
|
+
import { createTeeWriter } from "../tee-writer.js";
|
|
8
|
+
import { SequenceCounter } from "../sequence-counter.js";
|
|
9
|
+
import { resolveWorkTracker } from "./work-tracker.js";
|
|
10
|
+
import { resolveTaskContent } from "./task-input.js";
|
|
11
|
+
import { createServiceConfig } from "@forwardimpact/libconfig";
|
|
12
|
+
import { AGENT_MODEL } from "@forwardimpact/libutil/models";
|
|
13
|
+
|
|
14
|
+
/**
|
|
15
|
+
* Parse and validate run command options from parsed values.
|
|
16
|
+
* @param {object} values - Parsed option values from cli.parse()
|
|
17
|
+
* @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
|
|
18
|
+
* @returns {{ taskContent: string, cwd: string, model: string, maxTurns: number, outputPath: string|undefined, agentProfile: string|undefined, workTracker: string, allowedTools: string[] }}
|
|
19
|
+
*/
|
|
20
|
+
export function parseRunOptions(values, runtime) {
|
|
21
|
+
const { task: taskContent, amend: taskAmend } = resolveTaskContent(
|
|
22
|
+
values,
|
|
23
|
+
runtime,
|
|
24
|
+
);
|
|
25
|
+
const maxTurnsRaw = values["max-turns"] ?? "50";
|
|
26
|
+
|
|
27
|
+
return {
|
|
28
|
+
taskContent,
|
|
29
|
+
taskAmend,
|
|
30
|
+
cwd: resolve(values.cwd ?? "."),
|
|
31
|
+
agentModel: values["agent-model"] || AGENT_MODEL,
|
|
32
|
+
maxTurns: maxTurnsRaw === "0" ? 0 : parseInt(maxTurnsRaw, 10),
|
|
33
|
+
outputPath: values.output,
|
|
34
|
+
agentProfile: values["agent-profile"] ?? undefined,
|
|
35
|
+
workTracker: resolveWorkTracker(values, runtime?.proc?.env),
|
|
36
|
+
allowedTools: (
|
|
37
|
+
values["allowed-tools"] ??
|
|
38
|
+
"Bash,Read,Glob,Grep,Write,Edit,Agent,TodoWrite"
|
|
39
|
+
).split(","),
|
|
40
|
+
mcpServer: values["mcp-server"] ?? undefined,
|
|
41
|
+
};
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
/**
|
|
45
|
+
* Run command — execute a single agent via the Claude Agent SDK.
|
|
46
|
+
*
|
|
47
|
+
* Usage: fit-harness run [options]
|
|
48
|
+
*
|
|
49
|
+
* @param {import("@forwardimpact/libcli").InvocationContext} ctx
|
|
50
|
+
* @returns {Promise<{ok: boolean, code?: number, error?: string}>}
|
|
51
|
+
*/
|
|
52
|
+
export async function runRunCommand(ctx) {
|
|
53
|
+
const runtime = ctx.deps.runtime;
|
|
54
|
+
const {
|
|
55
|
+
taskContent,
|
|
56
|
+
taskAmend,
|
|
57
|
+
cwd,
|
|
58
|
+
agentModel,
|
|
59
|
+
maxTurns,
|
|
60
|
+
outputPath,
|
|
61
|
+
agentProfile,
|
|
62
|
+
workTracker,
|
|
63
|
+
allowedTools,
|
|
64
|
+
mcpServer,
|
|
65
|
+
} = parseRunOptions(ctx.options, runtime);
|
|
66
|
+
|
|
67
|
+
// Build the redactor as the first observable side-effect after option
|
|
68
|
+
// parsing — the env snapshot must freeze BEFORE any in-process
|
|
69
|
+
// env writes the command performs (e.g. LIBHARNESS_AGENT_PROFILE).
|
|
70
|
+
const redactor = createRedactor({ runtime });
|
|
71
|
+
|
|
72
|
+
// When --output is specified, stream text to stdout while writing NDJSON to file.
|
|
73
|
+
// Otherwise, write NDJSON directly to stdout (backwards-compatible).
|
|
74
|
+
const fileStream = outputPath
|
|
75
|
+
? runtime.fs.createWriteStream(outputPath)
|
|
76
|
+
: null;
|
|
77
|
+
const output = fileStream
|
|
78
|
+
? createTeeWriter({
|
|
79
|
+
fileStream,
|
|
80
|
+
textStream: runtime.proc.stdout,
|
|
81
|
+
mode: "raw",
|
|
82
|
+
now: () => isoTimestamp(runtime.clock.now()),
|
|
83
|
+
})
|
|
84
|
+
: runtime.proc.stdout;
|
|
85
|
+
|
|
86
|
+
const counter = new SequenceCounter();
|
|
87
|
+
const devNull = new Writable({
|
|
88
|
+
write(_chunk, _enc, cb) {
|
|
89
|
+
cb();
|
|
90
|
+
},
|
|
91
|
+
});
|
|
92
|
+
const onLine = (line) => {
|
|
93
|
+
const event = JSON.parse(line);
|
|
94
|
+
const tagged = { source: "agent", seq: counter.next(), event };
|
|
95
|
+
output.write(JSON.stringify(redactor.redactValue(tagged)) + "\n");
|
|
96
|
+
};
|
|
97
|
+
|
|
98
|
+
let mcpServers = null;
|
|
99
|
+
if (mcpServer) {
|
|
100
|
+
const mcpConfig = await createServiceConfig("mcp");
|
|
101
|
+
mcpServers = {
|
|
102
|
+
[mcpServer]: {
|
|
103
|
+
type: "http",
|
|
104
|
+
url: mcpConfig.url,
|
|
105
|
+
headers: { Authorization: `Bearer ${mcpConfig.mcpToken()}` },
|
|
106
|
+
},
|
|
107
|
+
};
|
|
108
|
+
allowedTools.push(`mcp__${mcpServer}__*`);
|
|
109
|
+
}
|
|
110
|
+
|
|
111
|
+
if (agentProfile) {
|
|
112
|
+
runtime.proc.env.LIBHARNESS_AGENT_PROFILE = agentProfile;
|
|
113
|
+
}
|
|
114
|
+
// Unconditional so the default "github" is observable to the agent's
|
|
115
|
+
// active-tracker resolution, mirroring --agent-profile's env write above.
|
|
116
|
+
runtime.proc.env.LIBHARNESS_WORK_TRACKER = workTracker;
|
|
117
|
+
|
|
118
|
+
const systemPrompt = agentProfile
|
|
119
|
+
? composeProfilePrompt(agentProfile, {
|
|
120
|
+
profilesDir: resolve(cwd, ".claude/agents"),
|
|
121
|
+
runtime,
|
|
122
|
+
})
|
|
123
|
+
: undefined;
|
|
124
|
+
|
|
125
|
+
const { query } = await import("@anthropic-ai/claude-agent-sdk");
|
|
126
|
+
const runner = createAgentRunner({
|
|
127
|
+
cwd,
|
|
128
|
+
query,
|
|
129
|
+
output: devNull,
|
|
130
|
+
model: agentModel,
|
|
131
|
+
maxTurns,
|
|
132
|
+
allowedTools,
|
|
133
|
+
onLine,
|
|
134
|
+
settingSources: ["project"],
|
|
135
|
+
systemPrompt,
|
|
136
|
+
taskAmend,
|
|
137
|
+
mcpServers,
|
|
138
|
+
redactor,
|
|
139
|
+
runtime,
|
|
140
|
+
});
|
|
141
|
+
|
|
142
|
+
const result = await runner.run(taskContent);
|
|
143
|
+
|
|
144
|
+
if (fileStream) {
|
|
145
|
+
await new Promise((r) => output.end(r));
|
|
146
|
+
await new Promise((r) => fileStream.end(r));
|
|
147
|
+
}
|
|
148
|
+
|
|
149
|
+
return result.success
|
|
150
|
+
? { ok: true }
|
|
151
|
+
: { ok: false, code: 1, error: result.error?.message ?? "" };
|
|
152
|
+
}
|