@forwardimpact/libharness 0.1.22 → 1.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (86) hide show
  1. package/LICENSE +21 -201
  2. package/README.md +196 -80
  3. package/bin/fit-benchmark.js +44 -0
  4. package/bin/fit-harness.js +358 -0
  5. package/bin/fit-selfedit.js +165 -0
  6. package/bin/fit-trace.js +510 -0
  7. package/package.json +41 -11
  8. package/src/agent-runner.js +256 -0
  9. package/src/benchmark/apm-installer.js +207 -0
  10. package/src/benchmark/env-loader.js +158 -0
  11. package/src/benchmark/hook-env.js +40 -0
  12. package/src/benchmark/invariants.js +141 -0
  13. package/src/benchmark/judge.js +187 -0
  14. package/src/benchmark/npm-installer.js +87 -0
  15. package/src/benchmark/report.js +522 -0
  16. package/src/benchmark/result.js +127 -0
  17. package/src/benchmark/runner.js +583 -0
  18. package/src/benchmark/task-family.js +260 -0
  19. package/src/benchmark/workdir.js +298 -0
  20. package/src/commands/assert.js +153 -0
  21. package/src/commands/benchmark-definition.js +165 -0
  22. package/src/commands/benchmark-invariants.js +73 -0
  23. package/src/commands/benchmark-report.js +51 -0
  24. package/src/commands/benchmark-run.js +111 -0
  25. package/src/commands/by-discussion.js +94 -0
  26. package/src/commands/callback.js +119 -0
  27. package/src/commands/discuss.js +132 -0
  28. package/src/commands/facilitate.js +123 -0
  29. package/src/commands/output.js +36 -0
  30. package/src/commands/run.js +152 -0
  31. package/src/commands/supervise.js +136 -0
  32. package/src/commands/task-input.js +54 -0
  33. package/src/commands/tee.js +53 -0
  34. package/src/commands/trace.js +630 -0
  35. package/src/commands/work-tracker.js +35 -0
  36. package/src/cost.js +79 -0
  37. package/src/discuss-tools.js +173 -0
  38. package/src/discusser.js +394 -0
  39. package/src/events/github.js +161 -0
  40. package/src/facilitator.js +205 -0
  41. package/src/inbox-poller.js +81 -0
  42. package/src/index.js +72 -2
  43. package/src/judge.js +210 -0
  44. package/src/message-bus.js +118 -0
  45. package/src/orchestration-loop.js +330 -0
  46. package/src/orchestration-toolkit.js +441 -0
  47. package/src/orchestrator-helpers.js +23 -0
  48. package/src/profile-prompt.js +266 -0
  49. package/src/redaction.js +253 -0
  50. package/src/render/line-renderer.js +54 -0
  51. package/src/render/orchestrator-filter.js +19 -0
  52. package/src/render/palette.js +63 -0
  53. package/src/render/tool-hints.js +154 -0
  54. package/src/render/turn-renderer.js +96 -0
  55. package/src/reply-emitter.js +47 -0
  56. package/src/sequence-counter.js +21 -0
  57. package/src/signature-filter.js +27 -0
  58. package/src/supervisor.js +236 -0
  59. package/src/tee-writer.js +150 -0
  60. package/src/trace-collector.js +444 -0
  61. package/src/trace-github.js +473 -0
  62. package/src/trace-multi.js +101 -0
  63. package/src/trace-query.js +748 -0
  64. package/src/trace-render.js +211 -0
  65. package/src/trace-usage.js +249 -0
  66. package/src/fixture/assertions.js +0 -42
  67. package/src/fixture/cache.js +0 -50
  68. package/src/fixture/eval.js +0 -146
  69. package/src/fixture/index.js +0 -9
  70. package/src/fixture/pathway.js +0 -451
  71. package/src/fixture/services.js +0 -56
  72. package/src/mock/clients.js +0 -135
  73. package/src/mock/config.js +0 -45
  74. package/src/mock/data.js +0 -46
  75. package/src/mock/fs.js +0 -111
  76. package/src/mock/grpc.js +0 -94
  77. package/src/mock/http.js +0 -60
  78. package/src/mock/index.js +0 -36
  79. package/src/mock/infra.js +0 -219
  80. package/src/mock/logger.js +0 -42
  81. package/src/mock/observer.js +0 -74
  82. package/src/mock/resource-index.js +0 -95
  83. package/src/mock/service-callbacks.js +0 -39
  84. package/src/mock/services.js +0 -79
  85. package/src/mock/spy.js +0 -44
  86. package/src/mock/storage.js +0 -118
@@ -0,0 +1,256 @@
1
+ /**
2
+ * AgentRunner — runs a single Claude Agent SDK session and emits raw
3
+ * NDJSON events to an output stream. Building block for `fit-harness run`,
4
+ * `fit-harness supervise`, `fit-harness facilitate`, and `fit-harness discuss`.
5
+ *
6
+ * Follows OO+DI: constructor injection, factory function, tests bypass factory.
7
+ */
8
+
9
+ import { AGENT_MODEL } from "@forwardimpact/libutil/models";
10
+
11
+ const DEFAULT_ALLOWED_TOOLS = ["Bash", "Read", "Glob", "Grep", "Write", "Edit"];
12
+
13
+ /**
14
+ * Did the session actually invoke the model? A genuine run always bills
15
+ * tokens (the system prompt alone is thousands of input tokens) and costs
16
+ * more than zero. A `result` message with `subtype: "success"` but zero
17
+ * token usage and zero cost means the model was never reached — the
18
+ * canonical signature of a Claude Code init/auth failure (e.g. an invalid
19
+ * `ANTHROPIC_API_KEY`), which the SDK otherwise reports as a clean success.
20
+ *
21
+ * If the SDK gave us neither a `usage` object nor `total_cost_usd`, don't
22
+ * second-guess the subtype — trust the reported success.
23
+ * @param {object|null} result - The SDK `result` message, or null.
24
+ * @returns {boolean}
25
+ */
26
+ function modelDidWork(result) {
27
+ if (!result) return false;
28
+ const { usage, total_cost_usd: cost } = result;
29
+ if (usage == null && cost == null) return true;
30
+ const tokens = usage
31
+ ? (usage.input_tokens ?? 0) +
32
+ (usage.output_tokens ?? 0) +
33
+ (usage.cache_creation_input_tokens ?? 0) +
34
+ (usage.cache_read_input_tokens ?? 0)
35
+ : 0;
36
+ return tokens > 0 || (cost ?? 0) > 0;
37
+ }
38
+
39
+ // fit-harness and kata-action run headless in CI/CD with no human to answer
40
+ // permission prompts. The SDK is always launched in bypass mode — not
41
+ // overridable — so a future caller can't accidentally reduce permissions.
42
+ const PERMISSION_MODE = "bypassPermissions";
43
+
44
+ /** Run a single Claude Agent SDK session and emit raw NDJSON events to an output stream. */
45
+ export class AgentRunner {
46
+ /**
47
+ * @param {object} deps
48
+ * @param {string} deps.cwd - Agent working directory
49
+ * @param {function} deps.query - SDK query function (injected for testing)
50
+ * @param {import("stream").Writable} deps.output - Stream to emit NDJSON to
51
+ * @param {string} [deps.model] - Claude model identifier
52
+ * @param {number} [deps.maxTurns] - Maximum agentic turns; 0 means unlimited
53
+ * @param {string[]} [deps.allowedTools] - Tools the agent may use
54
+ * @param {function} [deps.onLine] - Callback invoked with each NDJSON line as it's produced
55
+ * @param {string[]} [deps.settingSources] - SDK setting sources (e.g. ['project'] to load CLAUDE.md)
56
+ * @param {string|object} [deps.systemPrompt] - SDK system prompt (string replaces default; {type:'preset', preset:'claude_code', append} appends)
57
+ * @param {string[]} [deps.disallowedTools] - Tools to explicitly remove from the model's context
58
+ * @param {Record<string, object>} [deps.mcpServers] - MCP server configs to pass to the SDK query
59
+ * @param {object} deps.redactor
60
+ * @param {import("@forwardimpact/libutil/runtime").Runtime} [deps.runtime] -
61
+ * Ambient collaborators. Only `proc.env` is read (to record Skill
62
+ * invocations into `LIBHARNESS_SKILL`); when absent the write is skipped.
63
+ */
64
+ constructor(deps) {
65
+ if (!deps.cwd) throw new Error("cwd is required");
66
+ if (!deps.query) throw new Error("query is required");
67
+ if (!deps.output) throw new Error("output is required");
68
+ if (!deps.redactor) throw new Error("redactor is required");
69
+ this.runtime = deps.runtime ?? null;
70
+ this.cwd = deps.cwd;
71
+ this.query = deps.query;
72
+ this.output = deps.output;
73
+ this.redactor = deps.redactor;
74
+ this.model = deps.model ?? AGENT_MODEL;
75
+ this.maxTurns = deps.maxTurns ?? 50;
76
+ this.allowedTools = deps.allowedTools ?? DEFAULT_ALLOWED_TOOLS;
77
+ this.onLine = deps.onLine ?? null;
78
+ this.settingSources = deps.settingSources ?? [];
79
+ this.systemPrompt = deps.systemPrompt ?? null;
80
+ this.disallowedTools = deps.disallowedTools ?? [];
81
+ this.mcpServers = deps.mcpServers ?? null;
82
+ this.taskAmend = deps.taskAmend ?? null;
83
+ this.sessionId = null;
84
+ /** @type {AbortController|null} */
85
+ this.currentAbortController = null;
86
+ }
87
+
88
+ /**
89
+ * Run a new agent session with the given task.
90
+ * @param {string} task
91
+ * @returns {Promise<{success: boolean, text: string, sessionId: string|null, error: Error|null, aborted: boolean}>}
92
+ */
93
+ async run(task) {
94
+ const abortController = new AbortController();
95
+ this.currentAbortController = abortController;
96
+ const effectiveTask = this.taskAmend
97
+ ? task
98
+ ? `${task}\n\n${this.taskAmend}`
99
+ : this.taskAmend
100
+ : task;
101
+ try {
102
+ const iterator = this.query({
103
+ prompt: effectiveTask,
104
+ options: this.#callOptions(abortController),
105
+ });
106
+ return await this.#consumeQuery(iterator);
107
+ } finally {
108
+ this.currentAbortController = null;
109
+ }
110
+ }
111
+
112
+ /**
113
+ * Resume an existing session with a follow-up prompt.
114
+ * @param {string} prompt
115
+ * @returns {Promise<{success: boolean, text: string, sessionId: string|null, error: Error|null, aborted: boolean}>}
116
+ */
117
+ async resume(prompt) {
118
+ const abortController = new AbortController();
119
+ this.currentAbortController = abortController;
120
+ try {
121
+ const iterator = this.query({
122
+ prompt,
123
+ options: {
124
+ ...this.#callOptions(abortController),
125
+ resume: this.sessionId,
126
+ },
127
+ });
128
+ return await this.#consumeQuery(iterator);
129
+ } finally {
130
+ this.currentAbortController = null;
131
+ }
132
+ }
133
+
134
+ /**
135
+ * Build the options passed to every SDK query() call. Shared by run()
136
+ * and resume() so the agent's configuration — cwd, tools, prompt,
137
+ * setting sources, turn budget — is identical across the session's
138
+ * lifetime. Only resume() layers `resume: this.sessionId` on top.
139
+ *
140
+ * SDK options are call-attached, not session-attached: the resumed
141
+ * call loads the prior conversation but otherwise uses whatever
142
+ * options this call passes. Omitting tool/prompt/setting options on
143
+ * resume causes the agent to silently lose its restrictions and
144
+ * persona between turns.
145
+ */
146
+ #callOptions(abortController) {
147
+ return {
148
+ cwd: this.cwd,
149
+ allowedTools: this.allowedTools,
150
+ maxTurns: this.maxTurns === 0 ? Number.MAX_SAFE_INTEGER : this.maxTurns,
151
+ model: this.model,
152
+ permissionMode: PERMISSION_MODE,
153
+ allowDangerouslySkipPermissions: true,
154
+ settingSources: this.settingSources,
155
+ abortController,
156
+ ...(this.disallowedTools.length > 0 && {
157
+ disallowedTools: this.disallowedTools,
158
+ }),
159
+ ...(this.systemPrompt && { systemPrompt: this.systemPrompt }),
160
+ ...(this.mcpServers && { mcpServers: this.mcpServers }),
161
+ };
162
+ }
163
+
164
+ /**
165
+ * Iterate the SDK query iterator, mirroring every message to the
166
+ * output stream and the `onLine` callback. Captures `sessionId` from
167
+ * the SDK's `system/init` message and tracks Skill invocations into
168
+ * `LIBHARNESS_SKILL` for downstream metrics.
169
+ *
170
+ * If the iterator throws and we triggered the abort ourselves
171
+ * (`currentAbortController.signal.aborted`), we report `aborted:
172
+ * true`; otherwise the error propagates as `error`.
173
+ */
174
+ async #consumeQuery(iterator) {
175
+ let text = "";
176
+ let stopReason = null;
177
+ let resultMessage = null;
178
+ let error = null;
179
+ let aborted = false;
180
+
181
+ try {
182
+ for await (const message of iterator) {
183
+ this.#recordLine(message);
184
+ if (message.type === "result") {
185
+ text = message.result ?? "";
186
+ stopReason = message.subtype;
187
+ resultMessage = message;
188
+ }
189
+ }
190
+ } catch (err) {
191
+ if (this.currentAbortController?.signal.aborted) {
192
+ aborted = true;
193
+ } else {
194
+ error = err;
195
+ }
196
+ }
197
+
198
+ // A "success" subtype is necessary but not sufficient: the SDK reports a
199
+ // failed init (e.g. an invalid API key) as success with zero model work.
200
+ // Require evidence the model actually ran, and surface a clear error when
201
+ // it didn't, so the masked failure can't be reported as a green run.
202
+ const reportedSuccess = stopReason === "success";
203
+ const success =
204
+ reportedSuccess &&
205
+ resultMessage?.is_error !== true &&
206
+ modelDidWork(resultMessage);
207
+ if (reportedSuccess && !success && !error) {
208
+ error = new Error(
209
+ "agent reported success but performed no model work (zero token usage) — likely a Claude Code init or authentication failure",
210
+ );
211
+ }
212
+
213
+ return {
214
+ success,
215
+ text,
216
+ sessionId: this.sessionId,
217
+ error,
218
+ aborted,
219
+ };
220
+ }
221
+
222
+ #recordLine(message) {
223
+ const redacted = this.redactor.redactValue(message);
224
+ const line = JSON.stringify(redacted);
225
+ this.output.write(line + "\n");
226
+ if (this.onLine) this.onLine(line);
227
+
228
+ if (message.type === "system" && message.subtype === "init") {
229
+ this.sessionId = message.session_id;
230
+ }
231
+ if (message.type === "assistant") this.#trackSkillInvocation(message);
232
+ }
233
+
234
+ #trackSkillInvocation(message) {
235
+ const content = message.message?.content ?? message.content;
236
+ if (!Array.isArray(content)) return;
237
+ // Skill metric is recorded into the env map; without a runtime there is
238
+ // no env surface to write to, so the side-effect is simply skipped.
239
+ const env = this.runtime?.proc?.env ?? null;
240
+ if (!env) return;
241
+ for (const block of content) {
242
+ if (
243
+ block.type === "tool_use" &&
244
+ block.name === "Skill" &&
245
+ block.input?.skill
246
+ ) {
247
+ env.LIBHARNESS_SKILL = block.input.skill;
248
+ }
249
+ }
250
+ }
251
+ }
252
+
253
+ /** Factory function — wires real dependencies. */
254
+ export function createAgentRunner(deps) {
255
+ return new AgentRunner(deps);
256
+ }
@@ -0,0 +1,207 @@
1
+ /**
2
+ * ApmInstaller — runs `apm install --target claude` in the family root to
3
+ * materialise skills and agents, copies the resulting `.claude/` into a
4
+ * staging directory, and computes the manifest fingerprint from the lockfile.
5
+ * Per-task copy happens later in WorkdirManager.
6
+ *
7
+ * Subprocess and filesystem access route through the injected `runtime` bag
8
+ * (`runtime.subprocess.spawn` for the streaming `apm` child, `runtime.fs` for
9
+ * the async staging copies). See `createApmInstaller` for the real-dependency
10
+ * wiring; `installApm` is a thin free-function wrapper.
11
+ */
12
+
13
+ import { createHash } from "node:crypto";
14
+ import { join, resolve } from "node:path";
15
+
16
+ /** Installs apm and stages `.claude/` for a task family. */
17
+ export class ApmInstaller {
18
+ /**
19
+ * @param {object} deps
20
+ * @param {import("@forwardimpact/libutil/runtime").Runtime} deps.runtime -
21
+ * Ambient collaborators; uses `subprocess.spawn` and `fs`.
22
+ */
23
+ constructor({ runtime }) {
24
+ if (!runtime) throw new Error("runtime is required");
25
+ this.runtime = runtime;
26
+ }
27
+
28
+ /**
29
+ * @param {import("./task-family.js").TaskFamily} family
30
+ * @param {string} outputDir - The benchmark run's output directory.
31
+ * @param {object} [options]
32
+ * @param {string|null} [options.skillsFrom] - Stage `.claude/` from this
33
+ * directory instead of running apm install. The path is a root containing
34
+ * a `.claude/` tree (e.g. a working tree), letting a run exercise local,
35
+ * unpublished skills.
36
+ * @returns {Promise<{stagingDir: string, skillSetHash: string, judgeProfilesDir: string}>}
37
+ */
38
+ async install(family, outputDir, { skillsFrom } = {}) {
39
+ const fs = this.runtime.fs;
40
+ const stagingDir = join(outputDir, ".apm-staging");
41
+ const stagedClaude = join(stagingDir, ".claude");
42
+ const sourceClaude = skillsFrom
43
+ ? join(resolve(skillsFrom), ".claude")
44
+ : join(family.rootPath, ".claude");
45
+ const apmYml = join(family.rootPath, "apm.yml");
46
+
47
+ // --skills-from takes precedence over apm install: the caller is supplying
48
+ // the skill tree explicitly, so no remote fetch runs.
49
+ const hasApm =
50
+ !skillsFrom &&
51
+ (await fs
52
+ .access(apmYml)
53
+ .then(() => true)
54
+ .catch(() => false));
55
+
56
+ if (hasApm) {
57
+ await this.#runApmInstall(family.rootPath);
58
+ try {
59
+ await fs.access(sourceClaude);
60
+ } catch {
61
+ throw new Error(
62
+ `apm install did not produce .claude/ at ${sourceClaude}; check the family's apm.yml`,
63
+ );
64
+ }
65
+ }
66
+
67
+ await fs.rm(stagingDir, { recursive: true, force: true });
68
+ const hasClaudeDir = await fs
69
+ .access(sourceClaude)
70
+ .then(() => true)
71
+ .catch(() => false);
72
+ if (skillsFrom && !hasClaudeDir) {
73
+ throw new Error(`--skills-from has no .claude/ tree at ${sourceClaude}`);
74
+ }
75
+ if (hasClaudeDir) {
76
+ await fs.cp(sourceClaude, stagedClaude, { recursive: true });
77
+ } else {
78
+ await fs.mkdir(stagedClaude, { recursive: true });
79
+ }
80
+
81
+ // apm's claude target deploys a pack's skills/ into .claude/skills/ but
82
+ // never its agents/ subtree (agent profiles + references). Stage that from
83
+ // the installed apm_modules into .claude/agents/ so a skill that cites an
84
+ // agent reference (e.g. the work-item tracker matrix) resolves in the
85
+ // agent CWD. No-op when --skills-from supplied a tree or no apm_modules
86
+ // exist.
87
+ if (!skillsFrom) {
88
+ await this.#stageApmAgents(family.rootPath, stagedClaude);
89
+ }
90
+
91
+ // Stage the family-local judge profile outside .claude/ so it is available
92
+ // to the judge but never copied into the agent-under-test's CWD.
93
+ const judgeSource = join(family.rootPath, "judge.md");
94
+ const judgeProfilesDir = join(stagingDir, "judge-profiles");
95
+ try {
96
+ await fs.access(judgeSource);
97
+ await fs.mkdir(judgeProfilesDir, { recursive: true });
98
+ await fs.cp(judgeSource, join(judgeProfilesDir, "judge.md"));
99
+ } catch {}
100
+
101
+ const lockPath = join(family.rootPath, "apm.lock.yaml");
102
+ let skillSetHash = "";
103
+ try {
104
+ const lockBytes = await fs.readFile(lockPath);
105
+ skillSetHash =
106
+ "sha256:" +
107
+ createHash("sha256").update(normalizeLf(lockBytes)).digest("hex");
108
+ } catch {
109
+ // No lockfile — family doesn't use skill packs.
110
+ }
111
+
112
+ return { stagingDir, skillSetHash, judgeProfilesDir };
113
+ }
114
+
115
+ /**
116
+ * Merge each installed pack's `agents/` subtree (profiles + references) from
117
+ * `apm_modules/<owner>/<pack>/agents/` into the staged `.claude/agents/`.
118
+ * apm's claude target deploys `skills/` only, so without this an agent
119
+ * reference a skill cites is absent from the agent CWD.
120
+ * @param {string} familyRoot
121
+ * @param {string} stagedClaude
122
+ */
123
+ async #stageApmAgents(familyRoot, stagedClaude) {
124
+ const fs = this.runtime.fs;
125
+ const modulesRoot = join(familyRoot, "apm_modules");
126
+ let owners;
127
+ try {
128
+ owners = await fs.readdir(modulesRoot, { withFileTypes: true });
129
+ } catch {
130
+ return; // no apm_modules — nothing to stage
131
+ }
132
+ const stagedAgents = join(stagedClaude, "agents");
133
+ for (const owner of owners) {
134
+ if (!owner.isDirectory()) continue;
135
+ const ownerDir = join(modulesRoot, owner.name);
136
+ let packs;
137
+ try {
138
+ packs = await fs.readdir(ownerDir, { withFileTypes: true });
139
+ } catch {
140
+ continue;
141
+ }
142
+ for (const pack of packs) {
143
+ if (!pack.isDirectory()) continue;
144
+ const agentsDir = join(ownerDir, pack.name, "agents");
145
+ const hasAgents = await fs
146
+ .access(agentsDir)
147
+ .then(() => true)
148
+ .catch(() => false);
149
+ if (hasAgents) {
150
+ await fs.cp(agentsDir, stagedAgents, { recursive: true });
151
+ }
152
+ }
153
+ }
154
+ }
155
+
156
+ async #runApmInstall(cwd) {
157
+ const child = this.runtime.subprocess.spawn(
158
+ "apm",
159
+ ["install", "--target", "claude"],
160
+ { cwd, stdio: ["ignore", "pipe", "pipe"] },
161
+ );
162
+ // Drain stdout concurrently so the child never blocks on backpressure;
163
+ // capture stderr for the failure message.
164
+ let stderr = "";
165
+ const drainStdout = (async () => {
166
+ for await (const _chunk of child.stdout) {
167
+ // discard
168
+ }
169
+ })();
170
+ for await (const chunk of child.stderr) stderr += chunk.toString();
171
+ await drainStdout;
172
+ const code = await child.exitCode;
173
+ if (code !== 0) {
174
+ throw new Error(`apm install exited ${code}: ${stderr}`);
175
+ }
176
+ }
177
+ }
178
+
179
+ function normalizeLf(buf) {
180
+ const out = [];
181
+ for (let i = 0; i < buf.length; i++) {
182
+ if (buf[i] === 0x0d && i + 1 < buf.length && buf[i + 1] === 0x0a) continue;
183
+ out.push(buf[i]);
184
+ }
185
+ return Buffer.from(out);
186
+ }
187
+
188
+ /**
189
+ * Factory function — wires real dependencies.
190
+ * @param {ConstructorParameters<typeof ApmInstaller>[0]} deps
191
+ * @returns {ApmInstaller}
192
+ */
193
+ export function createApmInstaller(deps) {
194
+ return new ApmInstaller(deps);
195
+ }
196
+
197
+ /**
198
+ * Free-function shorthand for callers that thread a runtime bag.
199
+ * @param {import("./task-family.js").TaskFamily} family
200
+ * @param {string} outputDir
201
+ * @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
202
+ * @param {object} [options] - Forwarded to `ApmInstaller.install` (e.g.
203
+ * `{ skillsFrom }`).
204
+ */
205
+ export function installApm(family, outputDir, runtime, options = {}) {
206
+ return new ApmInstaller({ runtime }).install(family, outputDir, options);
207
+ }
@@ -0,0 +1,158 @@
1
+ /**
2
+ * Env-loader — auto-discover `.env` / `.env.local` files in a task family
3
+ * and its tasks, load them into `process.env`, and render the merged result
4
+ * into each agent CWD.
5
+ *
6
+ * Discovery paths (loaded in this order, first value per key wins):
7
+ * 1. process.env (CI secrets, shell env — never overwritten)
8
+ * 2. <family>/.env.local
9
+ * 3. <family>/.env
10
+ * 4. tasks/<id>/.env.local
11
+ * 5. tasks/<id>/.env
12
+ *
13
+ * Every discovered env file — family or task — is loaded into process.env
14
+ * AND rendered (with resolved values) into the agent working directory.
15
+ */
16
+
17
+ import { join } from "node:path";
18
+
19
+ const ENV_FILES = [".env.local", ".env"];
20
+
21
+ /**
22
+ * Parse a `.env` file into an array of {key, value} pairs.
23
+ * Handles KEY=VALUE, # comments, blank lines, and single/double-quoted values.
24
+ * @param {string} content
25
+ * @returns {Array<{key: string, value: string}>}
26
+ */
27
+ export function parseEnvFile(content) {
28
+ const entries = [];
29
+ for (const raw of content.split("\n")) {
30
+ const line = raw.trim();
31
+ if (!line || line.startsWith("#")) continue;
32
+ const eq = line.indexOf("=");
33
+ if (eq === -1) continue;
34
+ const key = line.slice(0, eq).trim();
35
+ if (!key) continue;
36
+ let value = line.slice(eq + 1).trim();
37
+ if (
38
+ (value.startsWith('"') && value.endsWith('"')) ||
39
+ (value.startsWith("'") && value.endsWith("'"))
40
+ ) {
41
+ value = value.slice(1, -1);
42
+ }
43
+ entries.push({ key, value });
44
+ }
45
+ return entries;
46
+ }
47
+
48
+ /**
49
+ * Read and parse an env file, returning [] if the file does not exist.
50
+ * @param {object} fs - Async filesystem surface (`runtime.fs`).
51
+ * @param {string} filePath
52
+ * @returns {Promise<Array<{key: string, value: string}>>}
53
+ */
54
+ async function readEnvFile(fs, filePath) {
55
+ try {
56
+ const content = await fs.readFile(filePath, "utf8");
57
+ return parseEnvFile(content);
58
+ } catch (e) {
59
+ if (e.code === "ENOENT") return [];
60
+ throw e;
61
+ }
62
+ }
63
+
64
+ /**
65
+ * Load entries into the process env map. Existing keys are never overwritten.
66
+ * @param {Record<string, string|undefined>} env - The `runtime.proc.env` map.
67
+ * @param {Array<{key: string, value: string}>} entries
68
+ * @returns {string[]} var names that were loaded
69
+ */
70
+ function applyToProcessEnv(env, entries) {
71
+ const names = [];
72
+ for (const { key, value } of entries) {
73
+ names.push(key);
74
+ if (env[key] === undefined) {
75
+ env[key] = value;
76
+ }
77
+ }
78
+ return names;
79
+ }
80
+
81
+ /**
82
+ * Load one env file: apply to the env map, record keys in the merged map.
83
+ * @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
84
+ * @param {string} dir
85
+ * @param {string} file
86
+ * @param {Set<string>} names
87
+ * @param {Map<string, Map<string, true>>} merged
88
+ */
89
+ async function loadOneEnvFile(runtime, dir, file, names, merged) {
90
+ const entries = await readEnvFile(runtime.fs, join(dir, file));
91
+ if (entries.length === 0) return;
92
+ for (const name of applyToProcessEnv(runtime.proc.env, entries)) {
93
+ names.add(name);
94
+ }
95
+ if (!merged.has(file)) merged.set(file, new Map());
96
+ const fileMap = merged.get(file);
97
+ for (const { key } of entries) {
98
+ if (!fileMap.has(key)) fileMap.set(key, true);
99
+ }
100
+ }
101
+
102
+ /**
103
+ * Scan directories for env files, load into the env map, and collect
104
+ * a merged key manifest per filename.
105
+ * @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
106
+ * @param {string[]} dirs
107
+ * @returns {Promise<{names: Set<string>, merged: Map<string, Map<string, true>>}>}
108
+ */
109
+ async function collectEnvEntries(runtime, dirs) {
110
+ const names = new Set();
111
+ const merged = new Map();
112
+ for (const dir of dirs) {
113
+ for (const file of ENV_FILES) {
114
+ await loadOneEnvFile(runtime, dir, file, names, merged);
115
+ }
116
+ }
117
+ return { names, merged };
118
+ }
119
+
120
+ /**
121
+ * Write resolved env files into the agent CWD and warn about empty values.
122
+ * @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
123
+ * @param {Map<string, Map<string, true>>} merged
124
+ * @param {string} agentCwd
125
+ */
126
+ async function renderEnvFiles(runtime, merged, agentCwd) {
127
+ const env = runtime.proc.env;
128
+ for (const [file, keyMap] of merged) {
129
+ const keys = [...keyMap.keys()];
130
+ const resolved = keys.map((key) => `${key}=${env[key] ?? ""}`);
131
+ await runtime.fs.writeFile(
132
+ join(agentCwd, file),
133
+ resolved.join("\n") + "\n",
134
+ );
135
+ const empty = keys.filter((key) => !env[key]);
136
+ if (empty.length > 0) {
137
+ runtime.proc.stderr.write(
138
+ `libharness: env warning: ${file} declares vars with no value: ${empty.join(", ")}\n`,
139
+ );
140
+ }
141
+ }
142
+ }
143
+
144
+ /**
145
+ * Discover `.env` / `.env.local` in one or more directories, load them
146
+ * into the process env map, and render the resolved values into the agent CWD.
147
+ *
148
+ * @param {string[]} dirs - Directories to scan (family root, task dir, etc.)
149
+ * @param {string} agentCwd - Agent working directory to render into.
150
+ * @param {import("@forwardimpact/libutil/runtime").Runtime} runtime - Ambient
151
+ * collaborators; uses `fs` (async read/write), `proc.env`, `proc.stderr`.
152
+ * @returns {Promise<string[]>} All var names discovered (for redaction).
153
+ */
154
+ export async function loadEnv(dirs, agentCwd, runtime) {
155
+ const { names, merged } = await collectEnvEntries(runtime, dirs);
156
+ await renderEnvFiles(runtime, merged, agentCwd);
157
+ return [...names];
158
+ }
@@ -0,0 +1,40 @@
1
+ /**
2
+ * Shared environment builder for the benchmark hook scripts (`preflight.sh` and
3
+ * `invariants.sh`). Keeping both spawns on one helper guarantees they expose the
4
+ * same variable set, so hook authors never have to wonder which vars a given
5
+ * hook receives.
6
+ *
7
+ * Path vars (TASK_DIR, FAMILY_DIR, HOOKS_DIR) let hooks reference real
8
+ * locations instead of reconstructing them from `$0`. They are paths, not
9
+ * secrets, so they need no redaction allowlist entry.
10
+ */
11
+
12
+ /**
13
+ * @param {Record<string, string>} baseEnv - Inherited env (`runtime.proc.env`).
14
+ * @param {object} vars
15
+ * @param {string} vars.cwd - Agent CWD → `$AGENT_CWD`.
16
+ * @param {number} vars.port - Allocated TCP port → `$PORT`.
17
+ * @param {string} vars.taskId - Task id → `$TASK_ID`.
18
+ * @param {string} vars.taskDir - Task directory on host → `$TASK_DIR`.
19
+ * @param {string} vars.hooksDir - Task `hooks/` dir on host → `$HOOKS_DIR`.
20
+ * @param {string|null} vars.familyDir - Family root on host → `$FAMILY_DIR`
21
+ * (null when the family root is unknown, e.g. a standalone task).
22
+ * @returns {Record<string, string>}
23
+ */
24
+ export function buildHookEnv(
25
+ baseEnv,
26
+ { cwd, port, taskId, taskDir, hooksDir, familyDir },
27
+ ) {
28
+ return {
29
+ ...baseEnv,
30
+ // The agent CWD itself — hooks reference emitted files as `$AGENT_CWD/<path>`.
31
+ // Distinct from the `invariants` CLI's `--run-dir` (the parent that
32
+ // *contains* `cwd/`), so the two are never confused.
33
+ AGENT_CWD: cwd,
34
+ PORT: String(port),
35
+ TASK_ID: taskId,
36
+ TASK_DIR: taskDir,
37
+ HOOKS_DIR: hooksDir,
38
+ FAMILY_DIR: familyDir ?? "",
39
+ };
40
+ }