@forwardimpact/libharness 0.1.20 → 1.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -201
- package/README.md +196 -80
- package/bin/fit-benchmark.js +44 -0
- package/bin/fit-harness.js +358 -0
- package/bin/fit-selfedit.js +165 -0
- package/bin/fit-trace.js +510 -0
- package/package.json +42 -12
- package/src/agent-runner.js +256 -0
- package/src/benchmark/apm-installer.js +207 -0
- package/src/benchmark/env-loader.js +158 -0
- package/src/benchmark/hook-env.js +40 -0
- package/src/benchmark/invariants.js +141 -0
- package/src/benchmark/judge.js +187 -0
- package/src/benchmark/npm-installer.js +87 -0
- package/src/benchmark/report.js +522 -0
- package/src/benchmark/result.js +127 -0
- package/src/benchmark/runner.js +583 -0
- package/src/benchmark/task-family.js +260 -0
- package/src/benchmark/workdir.js +298 -0
- package/src/commands/assert.js +153 -0
- package/src/commands/benchmark-definition.js +165 -0
- package/src/commands/benchmark-invariants.js +73 -0
- package/src/commands/benchmark-report.js +51 -0
- package/src/commands/benchmark-run.js +111 -0
- package/src/commands/by-discussion.js +94 -0
- package/src/commands/callback.js +119 -0
- package/src/commands/discuss.js +132 -0
- package/src/commands/facilitate.js +123 -0
- package/src/commands/output.js +36 -0
- package/src/commands/run.js +152 -0
- package/src/commands/supervise.js +136 -0
- package/src/commands/task-input.js +54 -0
- package/src/commands/tee.js +53 -0
- package/src/commands/trace.js +630 -0
- package/src/commands/work-tracker.js +35 -0
- package/src/cost.js +79 -0
- package/src/discuss-tools.js +173 -0
- package/src/discusser.js +394 -0
- package/src/events/github.js +161 -0
- package/src/facilitator.js +205 -0
- package/src/inbox-poller.js +81 -0
- package/src/index.js +72 -2
- package/src/judge.js +210 -0
- package/src/message-bus.js +118 -0
- package/src/orchestration-loop.js +330 -0
- package/src/orchestration-toolkit.js +441 -0
- package/src/orchestrator-helpers.js +23 -0
- package/src/profile-prompt.js +266 -0
- package/src/redaction.js +253 -0
- package/src/render/line-renderer.js +54 -0
- package/src/render/orchestrator-filter.js +19 -0
- package/src/render/palette.js +63 -0
- package/src/render/tool-hints.js +154 -0
- package/src/render/turn-renderer.js +96 -0
- package/src/reply-emitter.js +47 -0
- package/src/sequence-counter.js +21 -0
- package/src/signature-filter.js +27 -0
- package/src/supervisor.js +236 -0
- package/src/tee-writer.js +150 -0
- package/src/trace-collector.js +444 -0
- package/src/trace-github.js +473 -0
- package/src/trace-multi.js +101 -0
- package/src/trace-query.js +748 -0
- package/src/trace-render.js +211 -0
- package/src/trace-usage.js +249 -0
- package/src/fixture/assertions.js +0 -42
- package/src/fixture/cache.js +0 -50
- package/src/fixture/eval.js +0 -146
- package/src/fixture/index.js +0 -9
- package/src/fixture/pathway.js +0 -451
- package/src/fixture/services.js +0 -56
- package/src/mock/clients.js +0 -135
- package/src/mock/config.js +0 -45
- package/src/mock/data.js +0 -46
- package/src/mock/fs.js +0 -111
- package/src/mock/grpc.js +0 -94
- package/src/mock/http.js +0 -60
- package/src/mock/index.js +0 -36
- package/src/mock/infra.js +0 -219
- package/src/mock/logger.js +0 -42
- package/src/mock/observer.js +0 -74
- package/src/mock/resource-index.js +0 -95
- package/src/mock/service-callbacks.js +0 -39
- package/src/mock/services.js +0 -79
- package/src/mock/spy.js +0 -44
- package/src/mock/storage.js +0 -118
|
@@ -0,0 +1,132 @@
|
|
|
1
|
+
import { resolve } from "node:path";
|
|
2
|
+
import { isoTimestamp } from "@forwardimpact/libutil";
|
|
3
|
+
import { createDiscusser } from "../discusser.js";
|
|
4
|
+
import { createRedactor } from "../redaction.js";
|
|
5
|
+
import { createTeeWriter } from "../tee-writer.js";
|
|
6
|
+
import { resolveTaskContent } from "./task-input.js";
|
|
7
|
+
import { resolveWorkTracker } from "./work-tracker.js";
|
|
8
|
+
import { AGENT_MODEL, LEAD_MODEL } from "@forwardimpact/libutil/models";
|
|
9
|
+
|
|
10
|
+
function parseAgentProfiles(raw, cwd, maxTurns) {
|
|
11
|
+
if (!raw) return [];
|
|
12
|
+
return raw.split(",").map((entry) => {
|
|
13
|
+
const name = entry.trim();
|
|
14
|
+
return { name, role: name, cwd, agentProfile: name, maxTurns };
|
|
15
|
+
});
|
|
16
|
+
}
|
|
17
|
+
|
|
18
|
+
/**
|
|
19
|
+
* Parse and validate discuss command options. Exported so tests can verify
|
|
20
|
+
* defaults and the legacy-flag clean break.
|
|
21
|
+
* @param {object} values - Parsed option values
|
|
22
|
+
* @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
|
|
23
|
+
* @returns {object}
|
|
24
|
+
*/
|
|
25
|
+
export function parseDiscussOptions(values, runtime) {
|
|
26
|
+
const { task: taskContent, amend: taskAmend } = resolveTaskContent(
|
|
27
|
+
values,
|
|
28
|
+
runtime,
|
|
29
|
+
);
|
|
30
|
+
|
|
31
|
+
const profilesRaw = values["agent-profiles"];
|
|
32
|
+
const agentCwd = resolve(values["agent-cwd"] ?? ".");
|
|
33
|
+
|
|
34
|
+
const maxTurnsRaw = values["max-turns"] ?? "40";
|
|
35
|
+
const maxTurns = maxTurnsRaw === "0" ? 0 : parseInt(maxTurnsRaw, 10);
|
|
36
|
+
|
|
37
|
+
const agentConfigs = parseAgentProfiles(profilesRaw, agentCwd, maxTurns);
|
|
38
|
+
|
|
39
|
+
const resumeContextRaw = values["resume-context"];
|
|
40
|
+
let resumeContext = null;
|
|
41
|
+
if (resumeContextRaw) {
|
|
42
|
+
try {
|
|
43
|
+
resumeContext = JSON.parse(resumeContextRaw);
|
|
44
|
+
} catch (err) {
|
|
45
|
+
throw new Error(`--resume-context is not valid JSON: ${err.message}`);
|
|
46
|
+
}
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
const maxLeadTurnsRaw = values["max-lead-turns"] ?? "200";
|
|
50
|
+
const maxLeadTurns = parseInt(maxLeadTurnsRaw, 10);
|
|
51
|
+
|
|
52
|
+
return {
|
|
53
|
+
taskContent,
|
|
54
|
+
taskAmend,
|
|
55
|
+
agentConfigs,
|
|
56
|
+
leadProfile: values["lead-profile"] ?? undefined,
|
|
57
|
+
leadModel: values["lead-model"] || LEAD_MODEL,
|
|
58
|
+
agentModel: values["agent-model"] || AGENT_MODEL,
|
|
59
|
+
maxTurns,
|
|
60
|
+
maxLeadTurns,
|
|
61
|
+
outputPath: values.output,
|
|
62
|
+
workTracker: resolveWorkTracker(values, runtime?.proc?.env),
|
|
63
|
+
discussionId: values["discussion-id"] ?? null,
|
|
64
|
+
resumeContext,
|
|
65
|
+
callbackUrl: runtime.proc.env.CALLBACK_URL ?? null,
|
|
66
|
+
inboxUrl: runtime.proc.env.INBOX_URL ?? null,
|
|
67
|
+
correlationId: runtime.proc.env.CORRELATION_ID ?? null,
|
|
68
|
+
};
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
/**
|
|
72
|
+
* Discuss command — run a discusser-led session with suspend/resume
|
|
73
|
+
* semantics, threading `discussion_id` through the trace so multi-run
|
|
74
|
+
* conversations are queryable as one.
|
|
75
|
+
*
|
|
76
|
+
* @param {import("@forwardimpact/libcli").InvocationContext} ctx
|
|
77
|
+
* @returns {Promise<{ok: boolean, code?: number, error?: string}>}
|
|
78
|
+
*/
|
|
79
|
+
export async function runDiscussCommand(ctx) {
|
|
80
|
+
const runtime = ctx.deps.runtime;
|
|
81
|
+
const opts = parseDiscussOptions(ctx.options, runtime);
|
|
82
|
+
|
|
83
|
+
const redactor = createRedactor({ runtime });
|
|
84
|
+
|
|
85
|
+
const fileStream = opts.outputPath
|
|
86
|
+
? runtime.fs.createWriteStream(opts.outputPath)
|
|
87
|
+
: null;
|
|
88
|
+
const output = fileStream
|
|
89
|
+
? createTeeWriter({
|
|
90
|
+
fileStream,
|
|
91
|
+
textStream: runtime.proc.stdout,
|
|
92
|
+
mode: "supervised",
|
|
93
|
+
now: () => isoTimestamp(runtime.clock.now()),
|
|
94
|
+
})
|
|
95
|
+
: runtime.proc.stdout;
|
|
96
|
+
|
|
97
|
+
if (opts.leadProfile) {
|
|
98
|
+
runtime.proc.env.LIBHARNESS_AGENT_PROFILE = opts.leadProfile;
|
|
99
|
+
}
|
|
100
|
+
// Unconditional so the default "github" is observable to the agent's
|
|
101
|
+
// active-tracker resolution, mirroring --agent-profile's env write above.
|
|
102
|
+
runtime.proc.env.LIBHARNESS_WORK_TRACKER = opts.workTracker;
|
|
103
|
+
|
|
104
|
+
const { query } = await import("@anthropic-ai/claude-agent-sdk");
|
|
105
|
+
const discusser = createDiscusser({
|
|
106
|
+
leadProfile: opts.leadProfile,
|
|
107
|
+
leadModel: opts.leadModel,
|
|
108
|
+
agentModel: opts.agentModel,
|
|
109
|
+
agentConfigs: opts.agentConfigs,
|
|
110
|
+
discussionId: opts.discussionId,
|
|
111
|
+
resumeContext: opts.resumeContext,
|
|
112
|
+
query,
|
|
113
|
+
output,
|
|
114
|
+
maxTurns: opts.maxTurns,
|
|
115
|
+
maxLeadTurns: opts.maxLeadTurns,
|
|
116
|
+
taskAmend: opts.taskAmend,
|
|
117
|
+
redactor,
|
|
118
|
+
callbackUrl: opts.callbackUrl,
|
|
119
|
+
inboxUrl: opts.inboxUrl,
|
|
120
|
+
correlationId: opts.correlationId,
|
|
121
|
+
runtime,
|
|
122
|
+
});
|
|
123
|
+
|
|
124
|
+
const result = await discusser.run(opts.taskContent);
|
|
125
|
+
|
|
126
|
+
if (fileStream) {
|
|
127
|
+
await new Promise((r) => output.end(r));
|
|
128
|
+
await new Promise((r) => fileStream.end(r));
|
|
129
|
+
}
|
|
130
|
+
|
|
131
|
+
return result.success ? { ok: true } : { ok: false, code: 1, error: "" };
|
|
132
|
+
}
|
|
@@ -0,0 +1,123 @@
|
|
|
1
|
+
import { resolve } from "node:path";
|
|
2
|
+
import { isoTimestamp } from "@forwardimpact/libutil";
|
|
3
|
+
import { createFacilitator } from "../facilitator.js";
|
|
4
|
+
import { createRedactor } from "../redaction.js";
|
|
5
|
+
import { createTeeWriter } from "../tee-writer.js";
|
|
6
|
+
import { resolveTaskContent } from "./task-input.js";
|
|
7
|
+
import { resolveWorkTracker } from "./work-tracker.js";
|
|
8
|
+
import { AGENT_MODEL, LEAD_MODEL } from "@forwardimpact/libutil/models";
|
|
9
|
+
|
|
10
|
+
/**
|
|
11
|
+
* Parse comma-separated agent profile names into structured configs.
|
|
12
|
+
* @param {string} raw - Comma-separated profile names
|
|
13
|
+
* @param {string} cwd - Shared working directory for all agents
|
|
14
|
+
* @returns {Array<{name: string, role: string, cwd: string, agentProfile: string}>}
|
|
15
|
+
*/
|
|
16
|
+
function parseAgentProfiles(raw, cwd, maxTurns) {
|
|
17
|
+
return raw.split(",").map((entry) => {
|
|
18
|
+
const name = entry.trim();
|
|
19
|
+
return { name, role: name, cwd, agentProfile: name, maxTurns };
|
|
20
|
+
});
|
|
21
|
+
}
|
|
22
|
+
|
|
23
|
+
/**
|
|
24
|
+
* Parse and validate facilitate command options. Exported for test
|
|
25
|
+
* coverage of the `--max-turns` → per-agent threading contract; not part
|
|
26
|
+
* of the package's public API.
|
|
27
|
+
* @param {object} values - Parsed option values
|
|
28
|
+
* @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
|
|
29
|
+
* @returns {object} Parsed options
|
|
30
|
+
*/
|
|
31
|
+
export function parseFacilitateOptions(values, runtime) {
|
|
32
|
+
const { task: taskContent, amend: taskAmend } = resolveTaskContent(
|
|
33
|
+
values,
|
|
34
|
+
runtime,
|
|
35
|
+
);
|
|
36
|
+
|
|
37
|
+
const profilesRaw = values["agent-profiles"];
|
|
38
|
+
if (!profilesRaw) throw new Error("--agent-profiles is required");
|
|
39
|
+
const agentCwd = resolve(values["agent-cwd"] ?? ".");
|
|
40
|
+
|
|
41
|
+
const maxTurnsRaw = values["max-turns"] ?? "20";
|
|
42
|
+
const maxTurns = maxTurnsRaw === "0" ? 0 : parseInt(maxTurnsRaw, 10);
|
|
43
|
+
|
|
44
|
+
// Thread --max-turns into each participant: without this, every facilitated
|
|
45
|
+
// agent silently falls back to the 50-turn default in facilitator.js even
|
|
46
|
+
// when the caller raises the budget. Observed in run 26078312414 where
|
|
47
|
+
// staff-engineer terminated at 51 turns despite --max-turns=200.
|
|
48
|
+
const agentConfigs = parseAgentProfiles(profilesRaw, agentCwd, maxTurns);
|
|
49
|
+
|
|
50
|
+
return {
|
|
51
|
+
taskContent,
|
|
52
|
+
taskAmend,
|
|
53
|
+
agentConfigs,
|
|
54
|
+
facilitatorCwd: resolve(values["facilitator-cwd"] ?? "."),
|
|
55
|
+
agentModel: values["agent-model"] || AGENT_MODEL,
|
|
56
|
+
facilitatorModel: values["lead-model"] || LEAD_MODEL,
|
|
57
|
+
maxTurns,
|
|
58
|
+
outputPath: values.output,
|
|
59
|
+
facilitatorProfile: values["lead-profile"] ?? undefined,
|
|
60
|
+
workTracker: resolveWorkTracker(values, runtime?.proc?.env),
|
|
61
|
+
};
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
/**
|
|
65
|
+
* Facilitate command — run a facilitated multi-agent session.
|
|
66
|
+
*
|
|
67
|
+
* Usage: fit-harness facilitate [options]
|
|
68
|
+
*
|
|
69
|
+
* @param {import("@forwardimpact/libcli").InvocationContext} ctx
|
|
70
|
+
* @returns {Promise<{ok: boolean, code?: number, error?: string}>}
|
|
71
|
+
*/
|
|
72
|
+
export async function runFacilitateCommand(ctx) {
|
|
73
|
+
const runtime = ctx.deps.runtime;
|
|
74
|
+
const opts = parseFacilitateOptions(ctx.options, runtime);
|
|
75
|
+
|
|
76
|
+
// Build the redactor as the first observable side-effect after option
|
|
77
|
+
// parsing — the env snapshot must freeze BEFORE any in-process
|
|
78
|
+
// env writes the command performs (e.g. LIBHARNESS_AGENT_PROFILE).
|
|
79
|
+
const redactor = createRedactor({ runtime });
|
|
80
|
+
|
|
81
|
+
const fileStream = opts.outputPath
|
|
82
|
+
? runtime.fs.createWriteStream(opts.outputPath)
|
|
83
|
+
: null;
|
|
84
|
+
const output = fileStream
|
|
85
|
+
? createTeeWriter({
|
|
86
|
+
fileStream,
|
|
87
|
+
textStream: runtime.proc.stdout,
|
|
88
|
+
mode: "supervised",
|
|
89
|
+
now: () => isoTimestamp(runtime.clock.now()),
|
|
90
|
+
})
|
|
91
|
+
: runtime.proc.stdout;
|
|
92
|
+
|
|
93
|
+
if (opts.facilitatorProfile) {
|
|
94
|
+
runtime.proc.env.LIBHARNESS_AGENT_PROFILE = opts.facilitatorProfile;
|
|
95
|
+
}
|
|
96
|
+
// Unconditional so the default "github" is observable to the agent's
|
|
97
|
+
// active-tracker resolution, mirroring --agent-profile's env write above.
|
|
98
|
+
runtime.proc.env.LIBHARNESS_WORK_TRACKER = opts.workTracker;
|
|
99
|
+
|
|
100
|
+
const { query } = await import("@anthropic-ai/claude-agent-sdk");
|
|
101
|
+
const facilitator = createFacilitator({
|
|
102
|
+
facilitatorCwd: opts.facilitatorCwd,
|
|
103
|
+
agentConfigs: opts.agentConfigs,
|
|
104
|
+
query,
|
|
105
|
+
output,
|
|
106
|
+
agentModel: opts.agentModel,
|
|
107
|
+
facilitatorModel: opts.facilitatorModel,
|
|
108
|
+
maxTurns: opts.maxTurns,
|
|
109
|
+
facilitatorProfile: opts.facilitatorProfile,
|
|
110
|
+
taskAmend: opts.taskAmend,
|
|
111
|
+
redactor,
|
|
112
|
+
runtime,
|
|
113
|
+
});
|
|
114
|
+
|
|
115
|
+
const result = await facilitator.run(opts.taskContent);
|
|
116
|
+
|
|
117
|
+
if (fileStream) {
|
|
118
|
+
await new Promise((r) => output.end(r));
|
|
119
|
+
await new Promise((r) => fileStream.end(r));
|
|
120
|
+
}
|
|
121
|
+
|
|
122
|
+
return result.success ? { ok: true } : { ok: false, code: 1, error: "" };
|
|
123
|
+
}
|
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
import { isoTimestamp } from "@forwardimpact/libutil";
|
|
2
|
+
import { createTraceCollector } from "@forwardimpact/libharness";
|
|
3
|
+
|
|
4
|
+
/**
|
|
5
|
+
* Output command — process a complete NDJSON trace from stdin and write
|
|
6
|
+
* formatted output to stdout.
|
|
7
|
+
*
|
|
8
|
+
* Usage: fit-harness output [--format=json|text] < trace.ndjson
|
|
9
|
+
*
|
|
10
|
+
* @param {import("@forwardimpact/libcli").InvocationContext} ctx
|
|
11
|
+
* @returns {Promise<{ok: true}>}
|
|
12
|
+
*/
|
|
13
|
+
export async function runOutputCommand(ctx) {
|
|
14
|
+
const values = ctx.options;
|
|
15
|
+
const runtime = ctx.deps.runtime;
|
|
16
|
+
const format =
|
|
17
|
+
values.format === "text" || values.format === "json"
|
|
18
|
+
? values.format
|
|
19
|
+
: "json";
|
|
20
|
+
const collector = createTraceCollector({
|
|
21
|
+
now: () => isoTimestamp(runtime.clock.now()),
|
|
22
|
+
});
|
|
23
|
+
|
|
24
|
+
// `runtime.proc.stdin` is an AsyncIterable of UTF-8 lines (newline-split by
|
|
25
|
+
// the runtime), so each yielded value is exactly one NDJSON record.
|
|
26
|
+
for await (const line of runtime.proc.stdin) {
|
|
27
|
+
collector.addLine(line);
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
if (format === "text") {
|
|
31
|
+
runtime.proc.stdout.write(collector.toText() + "\n");
|
|
32
|
+
} else {
|
|
33
|
+
runtime.proc.stdout.write(JSON.stringify(collector.toJSON()) + "\n");
|
|
34
|
+
}
|
|
35
|
+
return { ok: true };
|
|
36
|
+
}
|
|
@@ -0,0 +1,152 @@
|
|
|
1
|
+
import { Writable } from "node:stream";
|
|
2
|
+
import { resolve } from "node:path";
|
|
3
|
+
import { isoTimestamp } from "@forwardimpact/libutil";
|
|
4
|
+
import { createAgentRunner } from "../agent-runner.js";
|
|
5
|
+
import { composeProfilePrompt } from "../profile-prompt.js";
|
|
6
|
+
import { createRedactor } from "../redaction.js";
|
|
7
|
+
import { createTeeWriter } from "../tee-writer.js";
|
|
8
|
+
import { SequenceCounter } from "../sequence-counter.js";
|
|
9
|
+
import { resolveWorkTracker } from "./work-tracker.js";
|
|
10
|
+
import { resolveTaskContent } from "./task-input.js";
|
|
11
|
+
import { createServiceConfig } from "@forwardimpact/libconfig";
|
|
12
|
+
import { AGENT_MODEL } from "@forwardimpact/libutil/models";
|
|
13
|
+
|
|
14
|
+
/**
|
|
15
|
+
* Parse and validate run command options from parsed values.
|
|
16
|
+
* @param {object} values - Parsed option values from cli.parse()
|
|
17
|
+
* @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
|
|
18
|
+
* @returns {{ taskContent: string, cwd: string, model: string, maxTurns: number, outputPath: string|undefined, agentProfile: string|undefined, workTracker: string, allowedTools: string[] }}
|
|
19
|
+
*/
|
|
20
|
+
export function parseRunOptions(values, runtime) {
|
|
21
|
+
const { task: taskContent, amend: taskAmend } = resolveTaskContent(
|
|
22
|
+
values,
|
|
23
|
+
runtime,
|
|
24
|
+
);
|
|
25
|
+
const maxTurnsRaw = values["max-turns"] ?? "50";
|
|
26
|
+
|
|
27
|
+
return {
|
|
28
|
+
taskContent,
|
|
29
|
+
taskAmend,
|
|
30
|
+
cwd: resolve(values.cwd ?? "."),
|
|
31
|
+
agentModel: values["agent-model"] || AGENT_MODEL,
|
|
32
|
+
maxTurns: maxTurnsRaw === "0" ? 0 : parseInt(maxTurnsRaw, 10),
|
|
33
|
+
outputPath: values.output,
|
|
34
|
+
agentProfile: values["agent-profile"] ?? undefined,
|
|
35
|
+
workTracker: resolveWorkTracker(values, runtime?.proc?.env),
|
|
36
|
+
allowedTools: (
|
|
37
|
+
values["allowed-tools"] ??
|
|
38
|
+
"Bash,Read,Glob,Grep,Write,Edit,Agent,TodoWrite"
|
|
39
|
+
).split(","),
|
|
40
|
+
mcpServer: values["mcp-server"] ?? undefined,
|
|
41
|
+
};
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
/**
|
|
45
|
+
* Run command — execute a single agent via the Claude Agent SDK.
|
|
46
|
+
*
|
|
47
|
+
* Usage: fit-harness run [options]
|
|
48
|
+
*
|
|
49
|
+
* @param {import("@forwardimpact/libcli").InvocationContext} ctx
|
|
50
|
+
* @returns {Promise<{ok: boolean, code?: number, error?: string}>}
|
|
51
|
+
*/
|
|
52
|
+
export async function runRunCommand(ctx) {
|
|
53
|
+
const runtime = ctx.deps.runtime;
|
|
54
|
+
const {
|
|
55
|
+
taskContent,
|
|
56
|
+
taskAmend,
|
|
57
|
+
cwd,
|
|
58
|
+
agentModel,
|
|
59
|
+
maxTurns,
|
|
60
|
+
outputPath,
|
|
61
|
+
agentProfile,
|
|
62
|
+
workTracker,
|
|
63
|
+
allowedTools,
|
|
64
|
+
mcpServer,
|
|
65
|
+
} = parseRunOptions(ctx.options, runtime);
|
|
66
|
+
|
|
67
|
+
// Build the redactor as the first observable side-effect after option
|
|
68
|
+
// parsing — the env snapshot must freeze BEFORE any in-process
|
|
69
|
+
// env writes the command performs (e.g. LIBHARNESS_AGENT_PROFILE).
|
|
70
|
+
const redactor = createRedactor({ runtime });
|
|
71
|
+
|
|
72
|
+
// When --output is specified, stream text to stdout while writing NDJSON to file.
|
|
73
|
+
// Otherwise, write NDJSON directly to stdout (backwards-compatible).
|
|
74
|
+
const fileStream = outputPath
|
|
75
|
+
? runtime.fs.createWriteStream(outputPath)
|
|
76
|
+
: null;
|
|
77
|
+
const output = fileStream
|
|
78
|
+
? createTeeWriter({
|
|
79
|
+
fileStream,
|
|
80
|
+
textStream: runtime.proc.stdout,
|
|
81
|
+
mode: "raw",
|
|
82
|
+
now: () => isoTimestamp(runtime.clock.now()),
|
|
83
|
+
})
|
|
84
|
+
: runtime.proc.stdout;
|
|
85
|
+
|
|
86
|
+
const counter = new SequenceCounter();
|
|
87
|
+
const devNull = new Writable({
|
|
88
|
+
write(_chunk, _enc, cb) {
|
|
89
|
+
cb();
|
|
90
|
+
},
|
|
91
|
+
});
|
|
92
|
+
const onLine = (line) => {
|
|
93
|
+
const event = JSON.parse(line);
|
|
94
|
+
const tagged = { source: "agent", seq: counter.next(), event };
|
|
95
|
+
output.write(JSON.stringify(redactor.redactValue(tagged)) + "\n");
|
|
96
|
+
};
|
|
97
|
+
|
|
98
|
+
let mcpServers = null;
|
|
99
|
+
if (mcpServer) {
|
|
100
|
+
const mcpConfig = await createServiceConfig("mcp");
|
|
101
|
+
mcpServers = {
|
|
102
|
+
[mcpServer]: {
|
|
103
|
+
type: "http",
|
|
104
|
+
url: mcpConfig.url,
|
|
105
|
+
headers: { Authorization: `Bearer ${mcpConfig.mcpToken()}` },
|
|
106
|
+
},
|
|
107
|
+
};
|
|
108
|
+
allowedTools.push(`mcp__${mcpServer}__*`);
|
|
109
|
+
}
|
|
110
|
+
|
|
111
|
+
if (agentProfile) {
|
|
112
|
+
runtime.proc.env.LIBHARNESS_AGENT_PROFILE = agentProfile;
|
|
113
|
+
}
|
|
114
|
+
// Unconditional so the default "github" is observable to the agent's
|
|
115
|
+
// active-tracker resolution, mirroring --agent-profile's env write above.
|
|
116
|
+
runtime.proc.env.LIBHARNESS_WORK_TRACKER = workTracker;
|
|
117
|
+
|
|
118
|
+
const systemPrompt = agentProfile
|
|
119
|
+
? composeProfilePrompt(agentProfile, {
|
|
120
|
+
profilesDir: resolve(cwd, ".claude/agents"),
|
|
121
|
+
runtime,
|
|
122
|
+
})
|
|
123
|
+
: undefined;
|
|
124
|
+
|
|
125
|
+
const { query } = await import("@anthropic-ai/claude-agent-sdk");
|
|
126
|
+
const runner = createAgentRunner({
|
|
127
|
+
cwd,
|
|
128
|
+
query,
|
|
129
|
+
output: devNull,
|
|
130
|
+
model: agentModel,
|
|
131
|
+
maxTurns,
|
|
132
|
+
allowedTools,
|
|
133
|
+
onLine,
|
|
134
|
+
settingSources: ["project"],
|
|
135
|
+
systemPrompt,
|
|
136
|
+
taskAmend,
|
|
137
|
+
mcpServers,
|
|
138
|
+
redactor,
|
|
139
|
+
runtime,
|
|
140
|
+
});
|
|
141
|
+
|
|
142
|
+
const result = await runner.run(taskContent);
|
|
143
|
+
|
|
144
|
+
if (fileStream) {
|
|
145
|
+
await new Promise((r) => output.end(r));
|
|
146
|
+
await new Promise((r) => fileStream.end(r));
|
|
147
|
+
}
|
|
148
|
+
|
|
149
|
+
return result.success
|
|
150
|
+
? { ok: true }
|
|
151
|
+
: { ok: false, code: 1, error: result.error?.message ?? "" };
|
|
152
|
+
}
|
|
@@ -0,0 +1,136 @@
|
|
|
1
|
+
import { resolve, join } from "node:path";
|
|
2
|
+
import { isoTimestamp } from "@forwardimpact/libutil";
|
|
3
|
+
import { createSupervisor } from "../supervisor.js";
|
|
4
|
+
import { createRedactor } from "../redaction.js";
|
|
5
|
+
import { createTeeWriter } from "../tee-writer.js";
|
|
6
|
+
import { resolveTaskContent } from "./task-input.js";
|
|
7
|
+
import { resolveWorkTracker } from "./work-tracker.js";
|
|
8
|
+
import { createServiceConfig } from "@forwardimpact/libconfig";
|
|
9
|
+
import { AGENT_MODEL, LEAD_MODEL } from "@forwardimpact/libutil/models";
|
|
10
|
+
|
|
11
|
+
/**
|
|
12
|
+
* Parse all supervise flags from parsed values into an options object.
|
|
13
|
+
* @param {object} values - Parsed option values from cli.parse()
|
|
14
|
+
* @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
|
|
15
|
+
* @returns {Promise<object>}
|
|
16
|
+
*/
|
|
17
|
+
export async function parseSuperviseOptions(values, runtime) {
|
|
18
|
+
const { task: taskContent, amend: taskAmend } = resolveTaskContent(
|
|
19
|
+
values,
|
|
20
|
+
runtime,
|
|
21
|
+
);
|
|
22
|
+
const supervisorAllowedToolsRaw = values["supervisor-allowed-tools"];
|
|
23
|
+
|
|
24
|
+
const tmpRoot = runtime.proc.env.TMPDIR ?? "/tmp";
|
|
25
|
+
const agentCwd = resolve(
|
|
26
|
+
values["agent-cwd"] ??
|
|
27
|
+
(await runtime.fs.mkdtemp(join(tmpRoot, "fit-harness-agent-"))),
|
|
28
|
+
);
|
|
29
|
+
|
|
30
|
+
return {
|
|
31
|
+
taskContent,
|
|
32
|
+
taskAmend,
|
|
33
|
+
supervisorCwd: resolve(values["supervisor-cwd"] ?? "."),
|
|
34
|
+
agentCwd,
|
|
35
|
+
agentModel: values["agent-model"] || AGENT_MODEL,
|
|
36
|
+
supervisorModel: values["lead-model"] || LEAD_MODEL,
|
|
37
|
+
maxTurns: (() => {
|
|
38
|
+
const raw = values["max-turns"] ?? "200";
|
|
39
|
+
return raw === "0" ? 0 : parseInt(raw, 10);
|
|
40
|
+
})(),
|
|
41
|
+
outputPath: values.output,
|
|
42
|
+
supervisorProfile: values["lead-profile"] ?? undefined,
|
|
43
|
+
agentProfile: values["agent-profile"] ?? undefined,
|
|
44
|
+
workTracker: resolveWorkTracker(values, runtime?.proc?.env),
|
|
45
|
+
allowedTools: (
|
|
46
|
+
values["allowed-tools"] ??
|
|
47
|
+
"Bash,Read,Glob,Grep,Write,Edit,Agent,TodoWrite"
|
|
48
|
+
).split(","),
|
|
49
|
+
supervisorAllowedTools: supervisorAllowedToolsRaw
|
|
50
|
+
? supervisorAllowedToolsRaw.split(",")
|
|
51
|
+
: undefined,
|
|
52
|
+
mcpServer: values["mcp-server"] ?? undefined,
|
|
53
|
+
};
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
/**
|
|
57
|
+
* Supervise command — run one agent under a supervisor via the
|
|
58
|
+
* orchestration loop. The supervisor delegates work through Ask, sees
|
|
59
|
+
* each reply on its next turn, and ends with Conclude.
|
|
60
|
+
*
|
|
61
|
+
* Usage: fit-harness supervise [options]
|
|
62
|
+
*
|
|
63
|
+
* @param {import("@forwardimpact/libcli").InvocationContext} ctx
|
|
64
|
+
* @returns {Promise<{ok: boolean, code?: number, error?: string}>}
|
|
65
|
+
*/
|
|
66
|
+
export async function runSuperviseCommand(ctx) {
|
|
67
|
+
const runtime = ctx.deps.runtime;
|
|
68
|
+
const opts = await parseSuperviseOptions(ctx.options, runtime);
|
|
69
|
+
|
|
70
|
+
// Build the redactor as the first observable side-effect after option
|
|
71
|
+
// parsing — the env snapshot must freeze BEFORE any in-process
|
|
72
|
+
// env writes the command performs (e.g. LIBHARNESS_AGENT_PROFILE).
|
|
73
|
+
const redactor = createRedactor({ runtime });
|
|
74
|
+
|
|
75
|
+
// When --output is specified, stream text to stdout while writing NDJSON to file.
|
|
76
|
+
// Otherwise, write NDJSON directly to stdout (backwards-compatible).
|
|
77
|
+
const fileStream = opts.outputPath
|
|
78
|
+
? runtime.fs.createWriteStream(opts.outputPath)
|
|
79
|
+
: null;
|
|
80
|
+
const output = fileStream
|
|
81
|
+
? createTeeWriter({
|
|
82
|
+
fileStream,
|
|
83
|
+
textStream: runtime.proc.stdout,
|
|
84
|
+
mode: "supervised",
|
|
85
|
+
now: () => isoTimestamp(runtime.clock.now()),
|
|
86
|
+
})
|
|
87
|
+
: runtime.proc.stdout;
|
|
88
|
+
|
|
89
|
+
let agentMcpServers = null;
|
|
90
|
+
if (opts.mcpServer) {
|
|
91
|
+
const mcpConfig = await createServiceConfig("mcp");
|
|
92
|
+
agentMcpServers = {
|
|
93
|
+
[opts.mcpServer]: {
|
|
94
|
+
type: "http",
|
|
95
|
+
url: mcpConfig.url,
|
|
96
|
+
headers: { Authorization: `Bearer ${mcpConfig.mcpToken()}` },
|
|
97
|
+
},
|
|
98
|
+
};
|
|
99
|
+
opts.allowedTools.push(`mcp__${opts.mcpServer}__*`);
|
|
100
|
+
}
|
|
101
|
+
|
|
102
|
+
if (opts.agentProfile) {
|
|
103
|
+
runtime.proc.env.LIBHARNESS_AGENT_PROFILE = opts.agentProfile;
|
|
104
|
+
}
|
|
105
|
+
// Unconditional so the default "github" is observable to the agent's
|
|
106
|
+
// active-tracker resolution, mirroring --agent-profile's env write above.
|
|
107
|
+
runtime.proc.env.LIBHARNESS_WORK_TRACKER = opts.workTracker;
|
|
108
|
+
|
|
109
|
+
const { query } = await import("@anthropic-ai/claude-agent-sdk");
|
|
110
|
+
const supervisor = createSupervisor({
|
|
111
|
+
supervisorCwd: opts.supervisorCwd,
|
|
112
|
+
agentCwd: opts.agentCwd,
|
|
113
|
+
query,
|
|
114
|
+
output,
|
|
115
|
+
agentModel: opts.agentModel,
|
|
116
|
+
supervisorModel: opts.supervisorModel,
|
|
117
|
+
maxTurns: opts.maxTurns,
|
|
118
|
+
allowedTools: opts.allowedTools,
|
|
119
|
+
supervisorAllowedTools: opts.supervisorAllowedTools,
|
|
120
|
+
supervisorProfile: opts.supervisorProfile,
|
|
121
|
+
agentProfile: opts.agentProfile,
|
|
122
|
+
taskAmend: opts.taskAmend,
|
|
123
|
+
agentMcpServers,
|
|
124
|
+
redactor,
|
|
125
|
+
runtime,
|
|
126
|
+
});
|
|
127
|
+
|
|
128
|
+
const result = await supervisor.run(opts.taskContent);
|
|
129
|
+
|
|
130
|
+
if (fileStream) {
|
|
131
|
+
await new Promise((r) => output.end(r));
|
|
132
|
+
await new Promise((r) => fileStream.end(r));
|
|
133
|
+
}
|
|
134
|
+
|
|
135
|
+
return result.success ? { ok: true } : { ok: false, code: 1, error: "" };
|
|
136
|
+
}
|
|
@@ -0,0 +1,54 @@
|
|
|
1
|
+
import { composeTaskFromGitHubEvent } from "../events/github.js";
|
|
2
|
+
|
|
3
|
+
/**
|
|
4
|
+
* Resolve `--task-file` / `--task-text` / `--task-event` into the task pair the
|
|
5
|
+
* runner consumes. Exactly one of the three must be set. For `--task-event`,
|
|
6
|
+
* libharness reads the event payload and extracts both the main task (from the
|
|
7
|
+
* template that matches `$GITHUB_EVENT_NAME` + `payload.action`) and the
|
|
8
|
+
* amendment (from `payload.inputs?.prompt`) — so the workflow doesn't need to
|
|
9
|
+
* wire `--task-amend` separately. For the other two modes, `--task-amend`
|
|
10
|
+
* works as before.
|
|
11
|
+
*
|
|
12
|
+
* @param {object} values - Parsed option values from cli.parse()
|
|
13
|
+
* @param {import("@forwardimpact/libutil/runtime").Runtime} runtime - Ambient
|
|
14
|
+
* collaborators; `fsSync.readFileSync` loads `--task-file`/`--task-event`
|
|
15
|
+
* and `proc.env` resolves `GITHUB_EVENT_NAME`.
|
|
16
|
+
* @returns {{ task: string, amend: string | undefined }}
|
|
17
|
+
*/
|
|
18
|
+
export function resolveTaskContent(values, runtime) {
|
|
19
|
+
const taskFile = values["task-file"];
|
|
20
|
+
const taskText = values["task-text"];
|
|
21
|
+
const taskEvent = values["task-event"];
|
|
22
|
+
|
|
23
|
+
const set = [taskFile, taskText, taskEvent].filter(Boolean).length;
|
|
24
|
+
if (set === 0) {
|
|
25
|
+
throw new Error(
|
|
26
|
+
"one of --task-file, --task-text, --task-event is required",
|
|
27
|
+
);
|
|
28
|
+
}
|
|
29
|
+
if (set > 1) {
|
|
30
|
+
throw new Error(
|
|
31
|
+
"--task-file, --task-text, --task-event are mutually exclusive",
|
|
32
|
+
);
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
const amendFlag = values["task-amend"] ?? undefined;
|
|
36
|
+
|
|
37
|
+
if (taskFile) {
|
|
38
|
+
return {
|
|
39
|
+
task: runtime.fsSync.readFileSync(taskFile, "utf8"),
|
|
40
|
+
amend: amendFlag,
|
|
41
|
+
};
|
|
42
|
+
}
|
|
43
|
+
if (taskText) {
|
|
44
|
+
return { task: taskText, amend: amendFlag };
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
const eventName = runtime.proc.env.GITHUB_EVENT_NAME;
|
|
48
|
+
if (!eventName) {
|
|
49
|
+
throw new Error("--task-event requires GITHUB_EVENT_NAME to be set");
|
|
50
|
+
}
|
|
51
|
+
const payload = JSON.parse(runtime.fsSync.readFileSync(taskEvent, "utf8"));
|
|
52
|
+
const composed = composeTaskFromGitHubEvent(payload, eventName);
|
|
53
|
+
return { task: composed.task, amend: amendFlag ?? composed.amend };
|
|
54
|
+
}
|
|
@@ -0,0 +1,53 @@
|
|
|
1
|
+
import { PassThrough } from "node:stream";
|
|
2
|
+
import { pipeline } from "node:stream/promises";
|
|
3
|
+
import { isoTimestamp } from "@forwardimpact/libutil";
|
|
4
|
+
import { createTeeWriter } from "../tee-writer.js";
|
|
5
|
+
|
|
6
|
+
/**
|
|
7
|
+
* Tee command — stream text output to stdout while optionally saving the raw
|
|
8
|
+
* NDJSON to a file. Reads stdin line-by-line through the injected runtime and
|
|
9
|
+
* re-delimits each record with a newline so the TeeWriter's line splitter sees
|
|
10
|
+
* the same framing the raw byte stream produced.
|
|
11
|
+
*
|
|
12
|
+
* Usage: fit-harness tee [output.ndjson] < trace.ndjson
|
|
13
|
+
*
|
|
14
|
+
* @param {import("@forwardimpact/libcli").InvocationContext} ctx
|
|
15
|
+
* @returns {Promise<{ok: boolean, code?: number, error?: string}>}
|
|
16
|
+
*/
|
|
17
|
+
export async function runTeeCommand(ctx) {
|
|
18
|
+
const runtime = ctx.deps.runtime;
|
|
19
|
+
const outputPath = ctx.args.output ?? null;
|
|
20
|
+
const fileStream = outputPath
|
|
21
|
+
? runtime.fs.createWriteStream(outputPath)
|
|
22
|
+
: null;
|
|
23
|
+
|
|
24
|
+
// TeeWriter requires a fileStream; when no output file is specified,
|
|
25
|
+
// use a PassThrough as a no-op sink (NDJSON is not saved).
|
|
26
|
+
const sink = fileStream ?? new PassThrough();
|
|
27
|
+
const tee = createTeeWriter({
|
|
28
|
+
fileStream: sink,
|
|
29
|
+
textStream: runtime.proc.stdout,
|
|
30
|
+
mode: "raw",
|
|
31
|
+
now: () => isoTimestamp(runtime.clock.now()),
|
|
32
|
+
});
|
|
33
|
+
|
|
34
|
+
try {
|
|
35
|
+
// `runtime.proc.stdin` yields newline-stripped lines; re-append `\n` so the
|
|
36
|
+
// TeeWriter's `_write` line splitter frames records exactly as it did when
|
|
37
|
+
// piped the raw byte stream.
|
|
38
|
+
const lines = (async function* () {
|
|
39
|
+
for await (const line of runtime.proc.stdin) yield `${line}\n`;
|
|
40
|
+
})();
|
|
41
|
+
await pipeline(lines, tee);
|
|
42
|
+
return { ok: true };
|
|
43
|
+
} catch (error) {
|
|
44
|
+
return { ok: false, code: 1, error: error.message };
|
|
45
|
+
} finally {
|
|
46
|
+
if (fileStream) {
|
|
47
|
+
await new Promise((resolve, reject) => {
|
|
48
|
+
fileStream.end(() => resolve());
|
|
49
|
+
fileStream.on("error", reject);
|
|
50
|
+
});
|
|
51
|
+
}
|
|
52
|
+
}
|
|
53
|
+
}
|