@forwardimpact/libharness 0.1.20 → 1.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -201
- package/README.md +196 -80
- package/bin/fit-benchmark.js +44 -0
- package/bin/fit-harness.js +358 -0
- package/bin/fit-selfedit.js +165 -0
- package/bin/fit-trace.js +510 -0
- package/package.json +42 -12
- package/src/agent-runner.js +256 -0
- package/src/benchmark/apm-installer.js +207 -0
- package/src/benchmark/env-loader.js +158 -0
- package/src/benchmark/hook-env.js +40 -0
- package/src/benchmark/invariants.js +141 -0
- package/src/benchmark/judge.js +187 -0
- package/src/benchmark/npm-installer.js +87 -0
- package/src/benchmark/report.js +522 -0
- package/src/benchmark/result.js +127 -0
- package/src/benchmark/runner.js +583 -0
- package/src/benchmark/task-family.js +260 -0
- package/src/benchmark/workdir.js +298 -0
- package/src/commands/assert.js +153 -0
- package/src/commands/benchmark-definition.js +165 -0
- package/src/commands/benchmark-invariants.js +73 -0
- package/src/commands/benchmark-report.js +51 -0
- package/src/commands/benchmark-run.js +111 -0
- package/src/commands/by-discussion.js +94 -0
- package/src/commands/callback.js +119 -0
- package/src/commands/discuss.js +132 -0
- package/src/commands/facilitate.js +123 -0
- package/src/commands/output.js +36 -0
- package/src/commands/run.js +152 -0
- package/src/commands/supervise.js +136 -0
- package/src/commands/task-input.js +54 -0
- package/src/commands/tee.js +53 -0
- package/src/commands/trace.js +630 -0
- package/src/commands/work-tracker.js +35 -0
- package/src/cost.js +79 -0
- package/src/discuss-tools.js +173 -0
- package/src/discusser.js +394 -0
- package/src/events/github.js +161 -0
- package/src/facilitator.js +205 -0
- package/src/inbox-poller.js +81 -0
- package/src/index.js +72 -2
- package/src/judge.js +210 -0
- package/src/message-bus.js +118 -0
- package/src/orchestration-loop.js +330 -0
- package/src/orchestration-toolkit.js +441 -0
- package/src/orchestrator-helpers.js +23 -0
- package/src/profile-prompt.js +266 -0
- package/src/redaction.js +253 -0
- package/src/render/line-renderer.js +54 -0
- package/src/render/orchestrator-filter.js +19 -0
- package/src/render/palette.js +63 -0
- package/src/render/tool-hints.js +154 -0
- package/src/render/turn-renderer.js +96 -0
- package/src/reply-emitter.js +47 -0
- package/src/sequence-counter.js +21 -0
- package/src/signature-filter.js +27 -0
- package/src/supervisor.js +236 -0
- package/src/tee-writer.js +150 -0
- package/src/trace-collector.js +444 -0
- package/src/trace-github.js +473 -0
- package/src/trace-multi.js +101 -0
- package/src/trace-query.js +748 -0
- package/src/trace-render.js +211 -0
- package/src/trace-usage.js +249 -0
- package/src/fixture/assertions.js +0 -42
- package/src/fixture/cache.js +0 -50
- package/src/fixture/eval.js +0 -146
- package/src/fixture/index.js +0 -9
- package/src/fixture/pathway.js +0 -451
- package/src/fixture/services.js +0 -56
- package/src/mock/clients.js +0 -135
- package/src/mock/config.js +0 -45
- package/src/mock/data.js +0 -46
- package/src/mock/fs.js +0 -111
- package/src/mock/grpc.js +0 -94
- package/src/mock/http.js +0 -60
- package/src/mock/index.js +0 -36
- package/src/mock/infra.js +0 -219
- package/src/mock/logger.js +0 -42
- package/src/mock/observer.js +0 -74
- package/src/mock/resource-index.js +0 -95
- package/src/mock/service-callbacks.js +0 -39
- package/src/mock/services.js +0 -79
- package/src/mock/spy.js +0 -44
- package/src/mock/storage.js +0 -118
|
@@ -0,0 +1,256 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* AgentRunner — runs a single Claude Agent SDK session and emits raw
|
|
3
|
+
* NDJSON events to an output stream. Building block for `fit-harness run`,
|
|
4
|
+
* `fit-harness supervise`, `fit-harness facilitate`, and `fit-harness discuss`.
|
|
5
|
+
*
|
|
6
|
+
* Follows OO+DI: constructor injection, factory function, tests bypass factory.
|
|
7
|
+
*/
|
|
8
|
+
|
|
9
|
+
import { AGENT_MODEL } from "@forwardimpact/libutil/models";
|
|
10
|
+
|
|
11
|
+
const DEFAULT_ALLOWED_TOOLS = ["Bash", "Read", "Glob", "Grep", "Write", "Edit"];
|
|
12
|
+
|
|
13
|
+
/**
|
|
14
|
+
* Did the session actually invoke the model? A genuine run always bills
|
|
15
|
+
* tokens (the system prompt alone is thousands of input tokens) and costs
|
|
16
|
+
* more than zero. A `result` message with `subtype: "success"` but zero
|
|
17
|
+
* token usage and zero cost means the model was never reached — the
|
|
18
|
+
* canonical signature of a Claude Code init/auth failure (e.g. an invalid
|
|
19
|
+
* `ANTHROPIC_API_KEY`), which the SDK otherwise reports as a clean success.
|
|
20
|
+
*
|
|
21
|
+
* If the SDK gave us neither a `usage` object nor `total_cost_usd`, don't
|
|
22
|
+
* second-guess the subtype — trust the reported success.
|
|
23
|
+
* @param {object|null} result - The SDK `result` message, or null.
|
|
24
|
+
* @returns {boolean}
|
|
25
|
+
*/
|
|
26
|
+
function modelDidWork(result) {
|
|
27
|
+
if (!result) return false;
|
|
28
|
+
const { usage, total_cost_usd: cost } = result;
|
|
29
|
+
if (usage == null && cost == null) return true;
|
|
30
|
+
const tokens = usage
|
|
31
|
+
? (usage.input_tokens ?? 0) +
|
|
32
|
+
(usage.output_tokens ?? 0) +
|
|
33
|
+
(usage.cache_creation_input_tokens ?? 0) +
|
|
34
|
+
(usage.cache_read_input_tokens ?? 0)
|
|
35
|
+
: 0;
|
|
36
|
+
return tokens > 0 || (cost ?? 0) > 0;
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
// fit-harness and kata-action run headless in CI/CD with no human to answer
|
|
40
|
+
// permission prompts. The SDK is always launched in bypass mode — not
|
|
41
|
+
// overridable — so a future caller can't accidentally reduce permissions.
|
|
42
|
+
const PERMISSION_MODE = "bypassPermissions";
|
|
43
|
+
|
|
44
|
+
/** Run a single Claude Agent SDK session and emit raw NDJSON events to an output stream. */
|
|
45
|
+
export class AgentRunner {
|
|
46
|
+
/**
|
|
47
|
+
* @param {object} deps
|
|
48
|
+
* @param {string} deps.cwd - Agent working directory
|
|
49
|
+
* @param {function} deps.query - SDK query function (injected for testing)
|
|
50
|
+
* @param {import("stream").Writable} deps.output - Stream to emit NDJSON to
|
|
51
|
+
* @param {string} [deps.model] - Claude model identifier
|
|
52
|
+
* @param {number} [deps.maxTurns] - Maximum agentic turns; 0 means unlimited
|
|
53
|
+
* @param {string[]} [deps.allowedTools] - Tools the agent may use
|
|
54
|
+
* @param {function} [deps.onLine] - Callback invoked with each NDJSON line as it's produced
|
|
55
|
+
* @param {string[]} [deps.settingSources] - SDK setting sources (e.g. ['project'] to load CLAUDE.md)
|
|
56
|
+
* @param {string|object} [deps.systemPrompt] - SDK system prompt (string replaces default; {type:'preset', preset:'claude_code', append} appends)
|
|
57
|
+
* @param {string[]} [deps.disallowedTools] - Tools to explicitly remove from the model's context
|
|
58
|
+
* @param {Record<string, object>} [deps.mcpServers] - MCP server configs to pass to the SDK query
|
|
59
|
+
* @param {object} deps.redactor
|
|
60
|
+
* @param {import("@forwardimpact/libutil/runtime").Runtime} [deps.runtime] -
|
|
61
|
+
* Ambient collaborators. Only `proc.env` is read (to record Skill
|
|
62
|
+
* invocations into `LIBHARNESS_SKILL`); when absent the write is skipped.
|
|
63
|
+
*/
|
|
64
|
+
constructor(deps) {
|
|
65
|
+
if (!deps.cwd) throw new Error("cwd is required");
|
|
66
|
+
if (!deps.query) throw new Error("query is required");
|
|
67
|
+
if (!deps.output) throw new Error("output is required");
|
|
68
|
+
if (!deps.redactor) throw new Error("redactor is required");
|
|
69
|
+
this.runtime = deps.runtime ?? null;
|
|
70
|
+
this.cwd = deps.cwd;
|
|
71
|
+
this.query = deps.query;
|
|
72
|
+
this.output = deps.output;
|
|
73
|
+
this.redactor = deps.redactor;
|
|
74
|
+
this.model = deps.model ?? AGENT_MODEL;
|
|
75
|
+
this.maxTurns = deps.maxTurns ?? 50;
|
|
76
|
+
this.allowedTools = deps.allowedTools ?? DEFAULT_ALLOWED_TOOLS;
|
|
77
|
+
this.onLine = deps.onLine ?? null;
|
|
78
|
+
this.settingSources = deps.settingSources ?? [];
|
|
79
|
+
this.systemPrompt = deps.systemPrompt ?? null;
|
|
80
|
+
this.disallowedTools = deps.disallowedTools ?? [];
|
|
81
|
+
this.mcpServers = deps.mcpServers ?? null;
|
|
82
|
+
this.taskAmend = deps.taskAmend ?? null;
|
|
83
|
+
this.sessionId = null;
|
|
84
|
+
/** @type {AbortController|null} */
|
|
85
|
+
this.currentAbortController = null;
|
|
86
|
+
}
|
|
87
|
+
|
|
88
|
+
/**
|
|
89
|
+
* Run a new agent session with the given task.
|
|
90
|
+
* @param {string} task
|
|
91
|
+
* @returns {Promise<{success: boolean, text: string, sessionId: string|null, error: Error|null, aborted: boolean}>}
|
|
92
|
+
*/
|
|
93
|
+
async run(task) {
|
|
94
|
+
const abortController = new AbortController();
|
|
95
|
+
this.currentAbortController = abortController;
|
|
96
|
+
const effectiveTask = this.taskAmend
|
|
97
|
+
? task
|
|
98
|
+
? `${task}\n\n${this.taskAmend}`
|
|
99
|
+
: this.taskAmend
|
|
100
|
+
: task;
|
|
101
|
+
try {
|
|
102
|
+
const iterator = this.query({
|
|
103
|
+
prompt: effectiveTask,
|
|
104
|
+
options: this.#callOptions(abortController),
|
|
105
|
+
});
|
|
106
|
+
return await this.#consumeQuery(iterator);
|
|
107
|
+
} finally {
|
|
108
|
+
this.currentAbortController = null;
|
|
109
|
+
}
|
|
110
|
+
}
|
|
111
|
+
|
|
112
|
+
/**
|
|
113
|
+
* Resume an existing session with a follow-up prompt.
|
|
114
|
+
* @param {string} prompt
|
|
115
|
+
* @returns {Promise<{success: boolean, text: string, sessionId: string|null, error: Error|null, aborted: boolean}>}
|
|
116
|
+
*/
|
|
117
|
+
async resume(prompt) {
|
|
118
|
+
const abortController = new AbortController();
|
|
119
|
+
this.currentAbortController = abortController;
|
|
120
|
+
try {
|
|
121
|
+
const iterator = this.query({
|
|
122
|
+
prompt,
|
|
123
|
+
options: {
|
|
124
|
+
...this.#callOptions(abortController),
|
|
125
|
+
resume: this.sessionId,
|
|
126
|
+
},
|
|
127
|
+
});
|
|
128
|
+
return await this.#consumeQuery(iterator);
|
|
129
|
+
} finally {
|
|
130
|
+
this.currentAbortController = null;
|
|
131
|
+
}
|
|
132
|
+
}
|
|
133
|
+
|
|
134
|
+
/**
|
|
135
|
+
* Build the options passed to every SDK query() call. Shared by run()
|
|
136
|
+
* and resume() so the agent's configuration — cwd, tools, prompt,
|
|
137
|
+
* setting sources, turn budget — is identical across the session's
|
|
138
|
+
* lifetime. Only resume() layers `resume: this.sessionId` on top.
|
|
139
|
+
*
|
|
140
|
+
* SDK options are call-attached, not session-attached: the resumed
|
|
141
|
+
* call loads the prior conversation but otherwise uses whatever
|
|
142
|
+
* options this call passes. Omitting tool/prompt/setting options on
|
|
143
|
+
* resume causes the agent to silently lose its restrictions and
|
|
144
|
+
* persona between turns.
|
|
145
|
+
*/
|
|
146
|
+
#callOptions(abortController) {
|
|
147
|
+
return {
|
|
148
|
+
cwd: this.cwd,
|
|
149
|
+
allowedTools: this.allowedTools,
|
|
150
|
+
maxTurns: this.maxTurns === 0 ? Number.MAX_SAFE_INTEGER : this.maxTurns,
|
|
151
|
+
model: this.model,
|
|
152
|
+
permissionMode: PERMISSION_MODE,
|
|
153
|
+
allowDangerouslySkipPermissions: true,
|
|
154
|
+
settingSources: this.settingSources,
|
|
155
|
+
abortController,
|
|
156
|
+
...(this.disallowedTools.length > 0 && {
|
|
157
|
+
disallowedTools: this.disallowedTools,
|
|
158
|
+
}),
|
|
159
|
+
...(this.systemPrompt && { systemPrompt: this.systemPrompt }),
|
|
160
|
+
...(this.mcpServers && { mcpServers: this.mcpServers }),
|
|
161
|
+
};
|
|
162
|
+
}
|
|
163
|
+
|
|
164
|
+
/**
|
|
165
|
+
* Iterate the SDK query iterator, mirroring every message to the
|
|
166
|
+
* output stream and the `onLine` callback. Captures `sessionId` from
|
|
167
|
+
* the SDK's `system/init` message and tracks Skill invocations into
|
|
168
|
+
* `LIBHARNESS_SKILL` for downstream metrics.
|
|
169
|
+
*
|
|
170
|
+
* If the iterator throws and we triggered the abort ourselves
|
|
171
|
+
* (`currentAbortController.signal.aborted`), we report `aborted:
|
|
172
|
+
* true`; otherwise the error propagates as `error`.
|
|
173
|
+
*/
|
|
174
|
+
async #consumeQuery(iterator) {
|
|
175
|
+
let text = "";
|
|
176
|
+
let stopReason = null;
|
|
177
|
+
let resultMessage = null;
|
|
178
|
+
let error = null;
|
|
179
|
+
let aborted = false;
|
|
180
|
+
|
|
181
|
+
try {
|
|
182
|
+
for await (const message of iterator) {
|
|
183
|
+
this.#recordLine(message);
|
|
184
|
+
if (message.type === "result") {
|
|
185
|
+
text = message.result ?? "";
|
|
186
|
+
stopReason = message.subtype;
|
|
187
|
+
resultMessage = message;
|
|
188
|
+
}
|
|
189
|
+
}
|
|
190
|
+
} catch (err) {
|
|
191
|
+
if (this.currentAbortController?.signal.aborted) {
|
|
192
|
+
aborted = true;
|
|
193
|
+
} else {
|
|
194
|
+
error = err;
|
|
195
|
+
}
|
|
196
|
+
}
|
|
197
|
+
|
|
198
|
+
// A "success" subtype is necessary but not sufficient: the SDK reports a
|
|
199
|
+
// failed init (e.g. an invalid API key) as success with zero model work.
|
|
200
|
+
// Require evidence the model actually ran, and surface a clear error when
|
|
201
|
+
// it didn't, so the masked failure can't be reported as a green run.
|
|
202
|
+
const reportedSuccess = stopReason === "success";
|
|
203
|
+
const success =
|
|
204
|
+
reportedSuccess &&
|
|
205
|
+
resultMessage?.is_error !== true &&
|
|
206
|
+
modelDidWork(resultMessage);
|
|
207
|
+
if (reportedSuccess && !success && !error) {
|
|
208
|
+
error = new Error(
|
|
209
|
+
"agent reported success but performed no model work (zero token usage) — likely a Claude Code init or authentication failure",
|
|
210
|
+
);
|
|
211
|
+
}
|
|
212
|
+
|
|
213
|
+
return {
|
|
214
|
+
success,
|
|
215
|
+
text,
|
|
216
|
+
sessionId: this.sessionId,
|
|
217
|
+
error,
|
|
218
|
+
aborted,
|
|
219
|
+
};
|
|
220
|
+
}
|
|
221
|
+
|
|
222
|
+
#recordLine(message) {
|
|
223
|
+
const redacted = this.redactor.redactValue(message);
|
|
224
|
+
const line = JSON.stringify(redacted);
|
|
225
|
+
this.output.write(line + "\n");
|
|
226
|
+
if (this.onLine) this.onLine(line);
|
|
227
|
+
|
|
228
|
+
if (message.type === "system" && message.subtype === "init") {
|
|
229
|
+
this.sessionId = message.session_id;
|
|
230
|
+
}
|
|
231
|
+
if (message.type === "assistant") this.#trackSkillInvocation(message);
|
|
232
|
+
}
|
|
233
|
+
|
|
234
|
+
#trackSkillInvocation(message) {
|
|
235
|
+
const content = message.message?.content ?? message.content;
|
|
236
|
+
if (!Array.isArray(content)) return;
|
|
237
|
+
// Skill metric is recorded into the env map; without a runtime there is
|
|
238
|
+
// no env surface to write to, so the side-effect is simply skipped.
|
|
239
|
+
const env = this.runtime?.proc?.env ?? null;
|
|
240
|
+
if (!env) return;
|
|
241
|
+
for (const block of content) {
|
|
242
|
+
if (
|
|
243
|
+
block.type === "tool_use" &&
|
|
244
|
+
block.name === "Skill" &&
|
|
245
|
+
block.input?.skill
|
|
246
|
+
) {
|
|
247
|
+
env.LIBHARNESS_SKILL = block.input.skill;
|
|
248
|
+
}
|
|
249
|
+
}
|
|
250
|
+
}
|
|
251
|
+
}
|
|
252
|
+
|
|
253
|
+
/** Factory function — wires real dependencies. */
|
|
254
|
+
export function createAgentRunner(deps) {
|
|
255
|
+
return new AgentRunner(deps);
|
|
256
|
+
}
|
|
@@ -0,0 +1,207 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* ApmInstaller — runs `apm install --target claude` in the family root to
|
|
3
|
+
* materialise skills and agents, copies the resulting `.claude/` into a
|
|
4
|
+
* staging directory, and computes the manifest fingerprint from the lockfile.
|
|
5
|
+
* Per-task copy happens later in WorkdirManager.
|
|
6
|
+
*
|
|
7
|
+
* Subprocess and filesystem access route through the injected `runtime` bag
|
|
8
|
+
* (`runtime.subprocess.spawn` for the streaming `apm` child, `runtime.fs` for
|
|
9
|
+
* the async staging copies). See `createApmInstaller` for the real-dependency
|
|
10
|
+
* wiring; `installApm` is a thin free-function wrapper.
|
|
11
|
+
*/
|
|
12
|
+
|
|
13
|
+
import { createHash } from "node:crypto";
|
|
14
|
+
import { join, resolve } from "node:path";
|
|
15
|
+
|
|
16
|
+
/** Installs apm and stages `.claude/` for a task family. */
|
|
17
|
+
export class ApmInstaller {
|
|
18
|
+
/**
|
|
19
|
+
* @param {object} deps
|
|
20
|
+
* @param {import("@forwardimpact/libutil/runtime").Runtime} deps.runtime -
|
|
21
|
+
* Ambient collaborators; uses `subprocess.spawn` and `fs`.
|
|
22
|
+
*/
|
|
23
|
+
constructor({ runtime }) {
|
|
24
|
+
if (!runtime) throw new Error("runtime is required");
|
|
25
|
+
this.runtime = runtime;
|
|
26
|
+
}
|
|
27
|
+
|
|
28
|
+
/**
|
|
29
|
+
* @param {import("./task-family.js").TaskFamily} family
|
|
30
|
+
* @param {string} outputDir - The benchmark run's output directory.
|
|
31
|
+
* @param {object} [options]
|
|
32
|
+
* @param {string|null} [options.skillsFrom] - Stage `.claude/` from this
|
|
33
|
+
* directory instead of running apm install. The path is a root containing
|
|
34
|
+
* a `.claude/` tree (e.g. a working tree), letting a run exercise local,
|
|
35
|
+
* unpublished skills.
|
|
36
|
+
* @returns {Promise<{stagingDir: string, skillSetHash: string, judgeProfilesDir: string}>}
|
|
37
|
+
*/
|
|
38
|
+
async install(family, outputDir, { skillsFrom } = {}) {
|
|
39
|
+
const fs = this.runtime.fs;
|
|
40
|
+
const stagingDir = join(outputDir, ".apm-staging");
|
|
41
|
+
const stagedClaude = join(stagingDir, ".claude");
|
|
42
|
+
const sourceClaude = skillsFrom
|
|
43
|
+
? join(resolve(skillsFrom), ".claude")
|
|
44
|
+
: join(family.rootPath, ".claude");
|
|
45
|
+
const apmYml = join(family.rootPath, "apm.yml");
|
|
46
|
+
|
|
47
|
+
// --skills-from takes precedence over apm install: the caller is supplying
|
|
48
|
+
// the skill tree explicitly, so no remote fetch runs.
|
|
49
|
+
const hasApm =
|
|
50
|
+
!skillsFrom &&
|
|
51
|
+
(await fs
|
|
52
|
+
.access(apmYml)
|
|
53
|
+
.then(() => true)
|
|
54
|
+
.catch(() => false));
|
|
55
|
+
|
|
56
|
+
if (hasApm) {
|
|
57
|
+
await this.#runApmInstall(family.rootPath);
|
|
58
|
+
try {
|
|
59
|
+
await fs.access(sourceClaude);
|
|
60
|
+
} catch {
|
|
61
|
+
throw new Error(
|
|
62
|
+
`apm install did not produce .claude/ at ${sourceClaude}; check the family's apm.yml`,
|
|
63
|
+
);
|
|
64
|
+
}
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
await fs.rm(stagingDir, { recursive: true, force: true });
|
|
68
|
+
const hasClaudeDir = await fs
|
|
69
|
+
.access(sourceClaude)
|
|
70
|
+
.then(() => true)
|
|
71
|
+
.catch(() => false);
|
|
72
|
+
if (skillsFrom && !hasClaudeDir) {
|
|
73
|
+
throw new Error(`--skills-from has no .claude/ tree at ${sourceClaude}`);
|
|
74
|
+
}
|
|
75
|
+
if (hasClaudeDir) {
|
|
76
|
+
await fs.cp(sourceClaude, stagedClaude, { recursive: true });
|
|
77
|
+
} else {
|
|
78
|
+
await fs.mkdir(stagedClaude, { recursive: true });
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
// apm's claude target deploys a pack's skills/ into .claude/skills/ but
|
|
82
|
+
// never its agents/ subtree (agent profiles + references). Stage that from
|
|
83
|
+
// the installed apm_modules into .claude/agents/ so a skill that cites an
|
|
84
|
+
// agent reference (e.g. the work-item tracker matrix) resolves in the
|
|
85
|
+
// agent CWD. No-op when --skills-from supplied a tree or no apm_modules
|
|
86
|
+
// exist.
|
|
87
|
+
if (!skillsFrom) {
|
|
88
|
+
await this.#stageApmAgents(family.rootPath, stagedClaude);
|
|
89
|
+
}
|
|
90
|
+
|
|
91
|
+
// Stage the family-local judge profile outside .claude/ so it is available
|
|
92
|
+
// to the judge but never copied into the agent-under-test's CWD.
|
|
93
|
+
const judgeSource = join(family.rootPath, "judge.md");
|
|
94
|
+
const judgeProfilesDir = join(stagingDir, "judge-profiles");
|
|
95
|
+
try {
|
|
96
|
+
await fs.access(judgeSource);
|
|
97
|
+
await fs.mkdir(judgeProfilesDir, { recursive: true });
|
|
98
|
+
await fs.cp(judgeSource, join(judgeProfilesDir, "judge.md"));
|
|
99
|
+
} catch {}
|
|
100
|
+
|
|
101
|
+
const lockPath = join(family.rootPath, "apm.lock.yaml");
|
|
102
|
+
let skillSetHash = "";
|
|
103
|
+
try {
|
|
104
|
+
const lockBytes = await fs.readFile(lockPath);
|
|
105
|
+
skillSetHash =
|
|
106
|
+
"sha256:" +
|
|
107
|
+
createHash("sha256").update(normalizeLf(lockBytes)).digest("hex");
|
|
108
|
+
} catch {
|
|
109
|
+
// No lockfile — family doesn't use skill packs.
|
|
110
|
+
}
|
|
111
|
+
|
|
112
|
+
return { stagingDir, skillSetHash, judgeProfilesDir };
|
|
113
|
+
}
|
|
114
|
+
|
|
115
|
+
/**
|
|
116
|
+
* Merge each installed pack's `agents/` subtree (profiles + references) from
|
|
117
|
+
* `apm_modules/<owner>/<pack>/agents/` into the staged `.claude/agents/`.
|
|
118
|
+
* apm's claude target deploys `skills/` only, so without this an agent
|
|
119
|
+
* reference a skill cites is absent from the agent CWD.
|
|
120
|
+
* @param {string} familyRoot
|
|
121
|
+
* @param {string} stagedClaude
|
|
122
|
+
*/
|
|
123
|
+
async #stageApmAgents(familyRoot, stagedClaude) {
|
|
124
|
+
const fs = this.runtime.fs;
|
|
125
|
+
const modulesRoot = join(familyRoot, "apm_modules");
|
|
126
|
+
let owners;
|
|
127
|
+
try {
|
|
128
|
+
owners = await fs.readdir(modulesRoot, { withFileTypes: true });
|
|
129
|
+
} catch {
|
|
130
|
+
return; // no apm_modules — nothing to stage
|
|
131
|
+
}
|
|
132
|
+
const stagedAgents = join(stagedClaude, "agents");
|
|
133
|
+
for (const owner of owners) {
|
|
134
|
+
if (!owner.isDirectory()) continue;
|
|
135
|
+
const ownerDir = join(modulesRoot, owner.name);
|
|
136
|
+
let packs;
|
|
137
|
+
try {
|
|
138
|
+
packs = await fs.readdir(ownerDir, { withFileTypes: true });
|
|
139
|
+
} catch {
|
|
140
|
+
continue;
|
|
141
|
+
}
|
|
142
|
+
for (const pack of packs) {
|
|
143
|
+
if (!pack.isDirectory()) continue;
|
|
144
|
+
const agentsDir = join(ownerDir, pack.name, "agents");
|
|
145
|
+
const hasAgents = await fs
|
|
146
|
+
.access(agentsDir)
|
|
147
|
+
.then(() => true)
|
|
148
|
+
.catch(() => false);
|
|
149
|
+
if (hasAgents) {
|
|
150
|
+
await fs.cp(agentsDir, stagedAgents, { recursive: true });
|
|
151
|
+
}
|
|
152
|
+
}
|
|
153
|
+
}
|
|
154
|
+
}
|
|
155
|
+
|
|
156
|
+
async #runApmInstall(cwd) {
|
|
157
|
+
const child = this.runtime.subprocess.spawn(
|
|
158
|
+
"apm",
|
|
159
|
+
["install", "--target", "claude"],
|
|
160
|
+
{ cwd, stdio: ["ignore", "pipe", "pipe"] },
|
|
161
|
+
);
|
|
162
|
+
// Drain stdout concurrently so the child never blocks on backpressure;
|
|
163
|
+
// capture stderr for the failure message.
|
|
164
|
+
let stderr = "";
|
|
165
|
+
const drainStdout = (async () => {
|
|
166
|
+
for await (const _chunk of child.stdout) {
|
|
167
|
+
// discard
|
|
168
|
+
}
|
|
169
|
+
})();
|
|
170
|
+
for await (const chunk of child.stderr) stderr += chunk.toString();
|
|
171
|
+
await drainStdout;
|
|
172
|
+
const code = await child.exitCode;
|
|
173
|
+
if (code !== 0) {
|
|
174
|
+
throw new Error(`apm install exited ${code}: ${stderr}`);
|
|
175
|
+
}
|
|
176
|
+
}
|
|
177
|
+
}
|
|
178
|
+
|
|
179
|
+
function normalizeLf(buf) {
|
|
180
|
+
const out = [];
|
|
181
|
+
for (let i = 0; i < buf.length; i++) {
|
|
182
|
+
if (buf[i] === 0x0d && i + 1 < buf.length && buf[i + 1] === 0x0a) continue;
|
|
183
|
+
out.push(buf[i]);
|
|
184
|
+
}
|
|
185
|
+
return Buffer.from(out);
|
|
186
|
+
}
|
|
187
|
+
|
|
188
|
+
/**
|
|
189
|
+
* Factory function — wires real dependencies.
|
|
190
|
+
* @param {ConstructorParameters<typeof ApmInstaller>[0]} deps
|
|
191
|
+
* @returns {ApmInstaller}
|
|
192
|
+
*/
|
|
193
|
+
export function createApmInstaller(deps) {
|
|
194
|
+
return new ApmInstaller(deps);
|
|
195
|
+
}
|
|
196
|
+
|
|
197
|
+
/**
|
|
198
|
+
* Free-function shorthand for callers that thread a runtime bag.
|
|
199
|
+
* @param {import("./task-family.js").TaskFamily} family
|
|
200
|
+
* @param {string} outputDir
|
|
201
|
+
* @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
|
|
202
|
+
* @param {object} [options] - Forwarded to `ApmInstaller.install` (e.g.
|
|
203
|
+
* `{ skillsFrom }`).
|
|
204
|
+
*/
|
|
205
|
+
export function installApm(family, outputDir, runtime, options = {}) {
|
|
206
|
+
return new ApmInstaller({ runtime }).install(family, outputDir, options);
|
|
207
|
+
}
|
|
@@ -0,0 +1,158 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Env-loader — auto-discover `.env` / `.env.local` files in a task family
|
|
3
|
+
* and its tasks, load them into `process.env`, and render the merged result
|
|
4
|
+
* into each agent CWD.
|
|
5
|
+
*
|
|
6
|
+
* Discovery paths (loaded in this order, first value per key wins):
|
|
7
|
+
* 1. process.env (CI secrets, shell env — never overwritten)
|
|
8
|
+
* 2. <family>/.env.local
|
|
9
|
+
* 3. <family>/.env
|
|
10
|
+
* 4. tasks/<id>/.env.local
|
|
11
|
+
* 5. tasks/<id>/.env
|
|
12
|
+
*
|
|
13
|
+
* Every discovered env file — family or task — is loaded into process.env
|
|
14
|
+
* AND rendered (with resolved values) into the agent working directory.
|
|
15
|
+
*/
|
|
16
|
+
|
|
17
|
+
import { join } from "node:path";
|
|
18
|
+
|
|
19
|
+
const ENV_FILES = [".env.local", ".env"];
|
|
20
|
+
|
|
21
|
+
/**
|
|
22
|
+
* Parse a `.env` file into an array of {key, value} pairs.
|
|
23
|
+
* Handles KEY=VALUE, # comments, blank lines, and single/double-quoted values.
|
|
24
|
+
* @param {string} content
|
|
25
|
+
* @returns {Array<{key: string, value: string}>}
|
|
26
|
+
*/
|
|
27
|
+
export function parseEnvFile(content) {
|
|
28
|
+
const entries = [];
|
|
29
|
+
for (const raw of content.split("\n")) {
|
|
30
|
+
const line = raw.trim();
|
|
31
|
+
if (!line || line.startsWith("#")) continue;
|
|
32
|
+
const eq = line.indexOf("=");
|
|
33
|
+
if (eq === -1) continue;
|
|
34
|
+
const key = line.slice(0, eq).trim();
|
|
35
|
+
if (!key) continue;
|
|
36
|
+
let value = line.slice(eq + 1).trim();
|
|
37
|
+
if (
|
|
38
|
+
(value.startsWith('"') && value.endsWith('"')) ||
|
|
39
|
+
(value.startsWith("'") && value.endsWith("'"))
|
|
40
|
+
) {
|
|
41
|
+
value = value.slice(1, -1);
|
|
42
|
+
}
|
|
43
|
+
entries.push({ key, value });
|
|
44
|
+
}
|
|
45
|
+
return entries;
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
/**
|
|
49
|
+
* Read and parse an env file, returning [] if the file does not exist.
|
|
50
|
+
* @param {object} fs - Async filesystem surface (`runtime.fs`).
|
|
51
|
+
* @param {string} filePath
|
|
52
|
+
* @returns {Promise<Array<{key: string, value: string}>>}
|
|
53
|
+
*/
|
|
54
|
+
async function readEnvFile(fs, filePath) {
|
|
55
|
+
try {
|
|
56
|
+
const content = await fs.readFile(filePath, "utf8");
|
|
57
|
+
return parseEnvFile(content);
|
|
58
|
+
} catch (e) {
|
|
59
|
+
if (e.code === "ENOENT") return [];
|
|
60
|
+
throw e;
|
|
61
|
+
}
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
/**
|
|
65
|
+
* Load entries into the process env map. Existing keys are never overwritten.
|
|
66
|
+
* @param {Record<string, string|undefined>} env - The `runtime.proc.env` map.
|
|
67
|
+
* @param {Array<{key: string, value: string}>} entries
|
|
68
|
+
* @returns {string[]} var names that were loaded
|
|
69
|
+
*/
|
|
70
|
+
function applyToProcessEnv(env, entries) {
|
|
71
|
+
const names = [];
|
|
72
|
+
for (const { key, value } of entries) {
|
|
73
|
+
names.push(key);
|
|
74
|
+
if (env[key] === undefined) {
|
|
75
|
+
env[key] = value;
|
|
76
|
+
}
|
|
77
|
+
}
|
|
78
|
+
return names;
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
/**
|
|
82
|
+
* Load one env file: apply to the env map, record keys in the merged map.
|
|
83
|
+
* @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
|
|
84
|
+
* @param {string} dir
|
|
85
|
+
* @param {string} file
|
|
86
|
+
* @param {Set<string>} names
|
|
87
|
+
* @param {Map<string, Map<string, true>>} merged
|
|
88
|
+
*/
|
|
89
|
+
async function loadOneEnvFile(runtime, dir, file, names, merged) {
|
|
90
|
+
const entries = await readEnvFile(runtime.fs, join(dir, file));
|
|
91
|
+
if (entries.length === 0) return;
|
|
92
|
+
for (const name of applyToProcessEnv(runtime.proc.env, entries)) {
|
|
93
|
+
names.add(name);
|
|
94
|
+
}
|
|
95
|
+
if (!merged.has(file)) merged.set(file, new Map());
|
|
96
|
+
const fileMap = merged.get(file);
|
|
97
|
+
for (const { key } of entries) {
|
|
98
|
+
if (!fileMap.has(key)) fileMap.set(key, true);
|
|
99
|
+
}
|
|
100
|
+
}
|
|
101
|
+
|
|
102
|
+
/**
|
|
103
|
+
* Scan directories for env files, load into the env map, and collect
|
|
104
|
+
* a merged key manifest per filename.
|
|
105
|
+
* @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
|
|
106
|
+
* @param {string[]} dirs
|
|
107
|
+
* @returns {Promise<{names: Set<string>, merged: Map<string, Map<string, true>>}>}
|
|
108
|
+
*/
|
|
109
|
+
async function collectEnvEntries(runtime, dirs) {
|
|
110
|
+
const names = new Set();
|
|
111
|
+
const merged = new Map();
|
|
112
|
+
for (const dir of dirs) {
|
|
113
|
+
for (const file of ENV_FILES) {
|
|
114
|
+
await loadOneEnvFile(runtime, dir, file, names, merged);
|
|
115
|
+
}
|
|
116
|
+
}
|
|
117
|
+
return { names, merged };
|
|
118
|
+
}
|
|
119
|
+
|
|
120
|
+
/**
|
|
121
|
+
* Write resolved env files into the agent CWD and warn about empty values.
|
|
122
|
+
* @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
|
|
123
|
+
* @param {Map<string, Map<string, true>>} merged
|
|
124
|
+
* @param {string} agentCwd
|
|
125
|
+
*/
|
|
126
|
+
async function renderEnvFiles(runtime, merged, agentCwd) {
|
|
127
|
+
const env = runtime.proc.env;
|
|
128
|
+
for (const [file, keyMap] of merged) {
|
|
129
|
+
const keys = [...keyMap.keys()];
|
|
130
|
+
const resolved = keys.map((key) => `${key}=${env[key] ?? ""}`);
|
|
131
|
+
await runtime.fs.writeFile(
|
|
132
|
+
join(agentCwd, file),
|
|
133
|
+
resolved.join("\n") + "\n",
|
|
134
|
+
);
|
|
135
|
+
const empty = keys.filter((key) => !env[key]);
|
|
136
|
+
if (empty.length > 0) {
|
|
137
|
+
runtime.proc.stderr.write(
|
|
138
|
+
`libharness: env warning: ${file} declares vars with no value: ${empty.join(", ")}\n`,
|
|
139
|
+
);
|
|
140
|
+
}
|
|
141
|
+
}
|
|
142
|
+
}
|
|
143
|
+
|
|
144
|
+
/**
|
|
145
|
+
* Discover `.env` / `.env.local` in one or more directories, load them
|
|
146
|
+
* into the process env map, and render the resolved values into the agent CWD.
|
|
147
|
+
*
|
|
148
|
+
* @param {string[]} dirs - Directories to scan (family root, task dir, etc.)
|
|
149
|
+
* @param {string} agentCwd - Agent working directory to render into.
|
|
150
|
+
* @param {import("@forwardimpact/libutil/runtime").Runtime} runtime - Ambient
|
|
151
|
+
* collaborators; uses `fs` (async read/write), `proc.env`, `proc.stderr`.
|
|
152
|
+
* @returns {Promise<string[]>} All var names discovered (for redaction).
|
|
153
|
+
*/
|
|
154
|
+
export async function loadEnv(dirs, agentCwd, runtime) {
|
|
155
|
+
const { names, merged } = await collectEnvEntries(runtime, dirs);
|
|
156
|
+
await renderEnvFiles(runtime, merged, agentCwd);
|
|
157
|
+
return [...names];
|
|
158
|
+
}
|
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Shared environment builder for the benchmark hook scripts (`preflight.sh` and
|
|
3
|
+
* `invariants.sh`). Keeping both spawns on one helper guarantees they expose the
|
|
4
|
+
* same variable set, so hook authors never have to wonder which vars a given
|
|
5
|
+
* hook receives.
|
|
6
|
+
*
|
|
7
|
+
* Path vars (TASK_DIR, FAMILY_DIR, HOOKS_DIR) let hooks reference real
|
|
8
|
+
* locations instead of reconstructing them from `$0`. They are paths, not
|
|
9
|
+
* secrets, so they need no redaction allowlist entry.
|
|
10
|
+
*/
|
|
11
|
+
|
|
12
|
+
/**
|
|
13
|
+
* @param {Record<string, string>} baseEnv - Inherited env (`runtime.proc.env`).
|
|
14
|
+
* @param {object} vars
|
|
15
|
+
* @param {string} vars.cwd - Agent CWD → `$AGENT_CWD`.
|
|
16
|
+
* @param {number} vars.port - Allocated TCP port → `$PORT`.
|
|
17
|
+
* @param {string} vars.taskId - Task id → `$TASK_ID`.
|
|
18
|
+
* @param {string} vars.taskDir - Task directory on host → `$TASK_DIR`.
|
|
19
|
+
* @param {string} vars.hooksDir - Task `hooks/` dir on host → `$HOOKS_DIR`.
|
|
20
|
+
* @param {string|null} vars.familyDir - Family root on host → `$FAMILY_DIR`
|
|
21
|
+
* (null when the family root is unknown, e.g. a standalone task).
|
|
22
|
+
* @returns {Record<string, string>}
|
|
23
|
+
*/
|
|
24
|
+
export function buildHookEnv(
|
|
25
|
+
baseEnv,
|
|
26
|
+
{ cwd, port, taskId, taskDir, hooksDir, familyDir },
|
|
27
|
+
) {
|
|
28
|
+
return {
|
|
29
|
+
...baseEnv,
|
|
30
|
+
// The agent CWD itself — hooks reference emitted files as `$AGENT_CWD/<path>`.
|
|
31
|
+
// Distinct from the `invariants` CLI's `--run-dir` (the parent that
|
|
32
|
+
// *contains* `cwd/`), so the two are never confused.
|
|
33
|
+
AGENT_CWD: cwd,
|
|
34
|
+
PORT: String(port),
|
|
35
|
+
TASK_ID: taskId,
|
|
36
|
+
TASK_DIR: taskDir,
|
|
37
|
+
HOOKS_DIR: hooksDir,
|
|
38
|
+
FAMILY_DIR: familyDir ?? "",
|
|
39
|
+
};
|
|
40
|
+
}
|