@forwardimpact/libharness 1.2.0 → 1.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -13,6 +13,7 @@ import { runSuperviseCommand } from "../src/commands/supervise.js";
13
13
  import { runFacilitateCommand } from "../src/commands/facilitate.js";
14
14
  import { runDiscussCommand } from "../src/commands/discuss.js";
15
15
  import { runCallbackCommand } from "../src/commands/callback.js";
16
+ import { runScanLogsCommand } from "../src/commands/scan-logs.js";
16
17
  import { AGENT_MODEL, LEAD_MODEL } from "@forwardimpact/libutil/models";
17
18
 
18
19
  const LEAD_OPTIONS = {
@@ -279,6 +280,35 @@ const definition = {
279
280
  },
280
281
  },
281
282
  },
283
+ {
284
+ name: "scan-logs",
285
+ args: [],
286
+ argsUsage: "",
287
+ handler: runScanLogsCommand,
288
+ description:
289
+ "Scan a run's log archive for secret literals; exit non-zero on any hit, fail closed on an unreadable archive",
290
+ options: {
291
+ archive: {
292
+ type: "string",
293
+ description: "Path to an already-resolved log archive (.zip)",
294
+ },
295
+ "run-id": {
296
+ type: "string",
297
+ description:
298
+ "GitHub Actions run id to download the log archive for (with --repo)",
299
+ },
300
+ repo: {
301
+ type: "string",
302
+ description: "owner/repo for the --run-id download",
303
+ },
304
+ secret: {
305
+ type: "string",
306
+ multiple: true,
307
+ description:
308
+ "Repeatable label=literal; the literal (everything after the first =) is searched for in the logs",
309
+ },
310
+ },
311
+ },
282
312
  ],
283
313
  globalOptions: {
284
314
  format: { type: "string", description: "Output format (json|text)" },
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@forwardimpact/libharness",
3
- "version": "1.2.0",
3
+ "version": "1.3.0",
4
4
  "description": "Autonomous agent team harness — coordinate a lead and participant agents in one async session, with eval, benchmark, and trace tooling to prove the changes worked.",
5
5
  "keywords": [
6
6
  "orchestration",
@@ -7,6 +7,7 @@
7
7
  */
8
8
 
9
9
  import { AGENT_MODEL } from "@forwardimpact/libutil/models";
10
+ import { resolveClaudeCodeExecutable } from "./claude-code-executable.js";
10
11
 
11
12
  const DEFAULT_ALLOWED_TOOLS = ["Bash", "Read", "Glob", "Grep", "Write", "Edit"];
12
13
 
@@ -56,6 +57,10 @@ export class AgentRunner {
56
57
  * @param {string|object} [deps.systemPrompt] - SDK system prompt (string replaces default; {type:'preset', preset:'claude_code', append} appends)
57
58
  * @param {string[]} [deps.disallowedTools] - Tools to explicitly remove from the model's context
58
59
  * @param {Record<string, object>} [deps.mcpServers] - MCP server configs to pass to the SDK query
60
+ * @param {string} [deps.pathToClaudeCodeExecutable] - Absolute path to the
61
+ * native `claude` CLI the SDK should spawn. Set for compiled fit-* binaries,
62
+ * which can't self-resolve the SDK's platform optional dependency; omitted
63
+ * from source runs so the SDK resolves its own version-matched binary.
59
64
  * @param {object} deps.redactor
60
65
  * @param {import("@forwardimpact/libutil/runtime").Runtime} [deps.runtime] -
61
66
  * Ambient collaborators. Only `proc.env` is read (to record Skill
@@ -79,6 +84,9 @@ export class AgentRunner {
79
84
  this.systemPrompt = deps.systemPrompt ?? null;
80
85
  this.disallowedTools = deps.disallowedTools ?? [];
81
86
  this.mcpServers = deps.mcpServers ?? null;
87
+ // Optional; read only through a truthy guard in #callOptions, so an absent
88
+ // value stays undefined rather than needing a `?? null` default.
89
+ this.pathToClaudeCodeExecutable = deps.pathToClaudeCodeExecutable;
82
90
  this.taskAmend = deps.taskAmend ?? null;
83
91
  this.sessionId = null;
84
92
  /** @type {AbortController|null} */
@@ -158,6 +166,9 @@ export class AgentRunner {
158
166
  }),
159
167
  ...(this.systemPrompt && { systemPrompt: this.systemPrompt }),
160
168
  ...(this.mcpServers && { mcpServers: this.mcpServers }),
169
+ ...(this.pathToClaudeCodeExecutable && {
170
+ pathToClaudeCodeExecutable: this.pathToClaudeCodeExecutable,
171
+ }),
161
172
  };
162
173
  }
163
174
 
@@ -250,7 +261,14 @@ export class AgentRunner {
250
261
  }
251
262
  }
252
263
 
253
- /** Factory function — wires real dependencies. */
264
+ /**
265
+ * Factory function — wires real dependencies. Resolves the native `claude`
266
+ * executable for compiled fit-* binaries so the SDK doesn't fail to find its
267
+ * own platform optional dependency; an explicit `deps` value overrides it.
268
+ */
254
269
  export function createAgentRunner(deps) {
255
- return new AgentRunner(deps);
270
+ return new AgentRunner({
271
+ pathToClaudeCodeExecutable: resolveClaudeCodeExecutable(),
272
+ ...deps,
273
+ });
256
274
  }
@@ -0,0 +1,44 @@
1
+ /**
2
+ * Resolve the Claude Code CLI the Agent SDK should spawn.
3
+ *
4
+ * `query()` spawns a native `claude` binary that the SDK resolves from its own
5
+ * platform-specific optional dependency (`@anthropic-ai/claude-agent-sdk-<platform>`).
6
+ * `bun build --compile` bundles the SDK's JavaScript but not that separate
7
+ * native package — it is not part of the import graph — so a compiled fit-*
8
+ * binary can't self-resolve it and `query()` throws "Native CLI binary for
9
+ * <platform> not found".
10
+ *
11
+ * In a compiled binary we point the SDK at the standalone `claude` on PATH,
12
+ * installed beside fit-harness by the bootstrap action's `fit-install.sh`.
13
+ * Running from source keeps `node_modules`, where the SDK resolves its own
14
+ * version-matched binary, so there we return undefined and defer to the SDK.
15
+ */
16
+ import { LIBCLI_IS_COMPILED } from "@forwardimpact/libcli";
17
+
18
+ /**
19
+ * @param {object} [deps]
20
+ * @param {(cmd: string) => string | null | undefined} [deps.which] -
21
+ * PATH resolver (injected for testing).
22
+ * @param {boolean} [deps.isCompiled] -
23
+ * Whether this is a `bun --compile` binary (injected for testing).
24
+ * @returns {string | undefined} absolute path to `claude`, or undefined to
25
+ * defer resolution to the SDK.
26
+ */
27
+ export function resolveClaudeCodeExecutable({
28
+ which = defaultWhich,
29
+ isCompiled = LIBCLI_IS_COMPILED,
30
+ } = {}) {
31
+ if (!isCompiled) return undefined;
32
+ return which("claude") ?? undefined;
33
+ }
34
+
35
+ /**
36
+ * A compiled fit-* binary runs under the Bun runtime, which exposes a
37
+ * synchronous PATH resolver. The `typeof` guard keeps this safe under Node too.
38
+ */
39
+ function defaultWhich(cmd) {
40
+ if (typeof Bun !== "undefined" && typeof Bun.which === "function") {
41
+ return Bun.which(cmd);
42
+ }
43
+ return null;
44
+ }
@@ -0,0 +1,155 @@
1
+ /**
2
+ * `fit-harness scan-logs` — scan a run's log archive for secret literals and
3
+ * fail closed.
4
+ *
5
+ * A run-lifecycle concern (not an NDJSON trace, so it lives here rather than
6
+ * in `fit-trace`): after a CI run that handled secrets, download or accept the
7
+ * run's own log archive and assert none of a supplied set of literals leaked
8
+ * into it. Any hit exits non-zero; any download/extract failure also exits
9
+ * non-zero — the gate must never silently disarm.
10
+ *
11
+ * Log resolution:
12
+ * - `--archive <zip>` — an already-resolved archive (extracted locally).
13
+ * - `--run-id <id> --repo <owner/repo>` — download this run's archive via
14
+ * `gh` first, then extract.
15
+ *
16
+ * Secrets are `--secret <label>=<literal>`, repeatable. The literal is
17
+ * everything after the FIRST `=` (JWTs and base64 keys contain `=`); the label
18
+ * is only cosmetic, named in the `FAIL:` line.
19
+ */
20
+
21
+ import { join } from "node:path";
22
+
23
+ /**
24
+ * Parse repeatable `--secret label=literal` flags. libcli's `multiple: true`
25
+ * yields an array from node's parseArgs in every case; tolerate a bare string
26
+ * or undefined defensively. Split on the FIRST `=` only.
27
+ *
28
+ * @param {string[]|string|undefined} secretOpt
29
+ * @returns {{label: string, literal: string}[]}
30
+ */
31
+ export function parseSecrets(secretOpt) {
32
+ const arr = Array.isArray(secretOpt)
33
+ ? secretOpt
34
+ : secretOpt
35
+ ? [secretOpt]
36
+ : [];
37
+ return arr.map((s) => {
38
+ const idx = s.indexOf("=");
39
+ if (idx === -1) return { label: s, literal: "" };
40
+ return { label: s.slice(0, idx), literal: s.slice(idx + 1) };
41
+ });
42
+ }
43
+
44
+ /**
45
+ * Walk a directory tree and return every file path. Uses per-level readdir so
46
+ * it works against both node:fs and the libmock fs (no `recursive` reliance).
47
+ */
48
+ async function collectFiles(dir, runtime) {
49
+ const out = [];
50
+ const entries = await runtime.fs.readdir(dir, { withFileTypes: true });
51
+ for (const ent of entries) {
52
+ const full = join(dir, ent.name);
53
+ if (ent.isDirectory()) {
54
+ out.push(...(await collectFiles(full, runtime)));
55
+ } else if (ent.isFile()) {
56
+ out.push(full);
57
+ }
58
+ }
59
+ return out;
60
+ }
61
+
62
+ /**
63
+ * Scan every file under `dir` for each secret literal. Returns the labels of
64
+ * secrets whose non-empty literal appears in any file (empty literals are
65
+ * skipped — a secret the run never set cannot leak).
66
+ *
67
+ * @param {object} params
68
+ * @param {string} params.dir
69
+ * @param {{label: string, literal: string}[]} params.secrets
70
+ * @param {import('@forwardimpact/libutil/runtime').Runtime} params.runtime
71
+ * @returns {Promise<string[]>} labels that hit
72
+ */
73
+ export async function scanDirectory({ dir, secrets, runtime }) {
74
+ const files = await collectFiles(dir, runtime);
75
+ const contents = await Promise.all(
76
+ files.map((f) => runtime.fs.readFile(f, "utf8").catch(() => "")),
77
+ );
78
+ const failures = [];
79
+ for (const { label, literal } of secrets) {
80
+ if (!literal) continue;
81
+ if (contents.some((c) => c.includes(literal))) failures.push(label);
82
+ }
83
+ return failures;
84
+ }
85
+
86
+ /**
87
+ * Resolve a directory of extracted log files, downloading the archive first
88
+ * when given a run id. Throws (→ fail closed) on any download or extract
89
+ * failure or on missing/invalid inputs.
90
+ */
91
+ async function resolveLogsDir({ options, runtime }) {
92
+ const tmpRoot = runtime.proc.env.RUNNER_TEMP || "/tmp";
93
+ const dir = await runtime.fs.mkdtemp(join(tmpRoot, "scan-logs-"));
94
+ let zip = options.archive;
95
+
96
+ if (!zip) {
97
+ const runId = options["run-id"];
98
+ const repo = options.repo;
99
+ if (!runId || !repo) {
100
+ throw new Error("requires --archive, or --run-id and --repo");
101
+ }
102
+ if (!/^\d+$/.test(String(runId))) {
103
+ throw new Error("--run-id must be numeric");
104
+ }
105
+ if (!/^[\w.-]+\/[\w.-]+$/.test(repo)) {
106
+ throw new Error("--repo must be owner/name");
107
+ }
108
+ zip = join(dir, "run-logs.zip");
109
+ const dl = await runtime.subprocess.run("bash", [
110
+ "-c",
111
+ `gh api -H "Accept: application/vnd.github+json" ` +
112
+ `"/repos/${repo}/actions/runs/${runId}/logs" > "${zip}"`,
113
+ ]);
114
+ if (dl.exitCode !== 0) {
115
+ throw new Error(
116
+ `log archive download failed (gh exit ${dl.exitCode}): ${dl.stderr ?? ""}`,
117
+ );
118
+ }
119
+ }
120
+
121
+ const unz = await runtime.subprocess.run("unzip", ["-q", zip, "-d", dir]);
122
+ if (unz.exitCode !== 0) {
123
+ throw new Error(
124
+ `log archive empty/unreadable (unzip exit ${unz.exitCode})`,
125
+ );
126
+ }
127
+ return dir;
128
+ }
129
+
130
+ /**
131
+ * scan-logs command handler.
132
+ *
133
+ * @param {import("@forwardimpact/libcli").InvocationContext} ctx
134
+ * @returns {Promise<{ok: boolean, code: number, error?: string}>}
135
+ */
136
+ export async function runScanLogsCommand(ctx) {
137
+ const runtime = ctx.deps.runtime;
138
+ const options = ctx.options;
139
+ const secrets = parseSecrets(options.secret);
140
+
141
+ let dir;
142
+ try {
143
+ dir = await resolveLogsDir({ options, runtime });
144
+ } catch (err) {
145
+ // Fail closed: an unresolvable archive must not pass as "no leak". The
146
+ // dispatcher prints the returned `error`, so don't also write it here.
147
+ return { ok: false, code: 1, error: `scan-logs: ${err.message}` };
148
+ }
149
+
150
+ const failures = await scanDirectory({ dir, secrets, runtime });
151
+ for (const label of failures) {
152
+ runtime.proc.stderr.write(`FAIL: ${label} literal in run logs\n`);
153
+ }
154
+ return { ok: failures.length === 0, code: failures.length ? 1 : 0 };
155
+ }
package/src/index.js CHANGED
@@ -11,6 +11,7 @@ export {
11
11
  pickTraceArtifact,
12
12
  } from "./trace-github.js";
13
13
  export { AgentRunner, createAgentRunner } from "./agent-runner.js";
14
+ export { resolveClaudeCodeExecutable } from "./claude-code-executable.js";
14
15
  export {
15
16
  composeProfilePrompt,
16
17
  composeLeadPrompt,