@forwardimpact/libharness 1.2.0 → 1.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bin/fit-harness.js +30 -0
- package/package.json +1 -1
- package/src/agent-runner.js +20 -2
- package/src/claude-code-executable.js +44 -0
- package/src/commands/scan-logs.js +155 -0
- package/src/index.js +1 -0
package/bin/fit-harness.js
CHANGED
|
@@ -13,6 +13,7 @@ import { runSuperviseCommand } from "../src/commands/supervise.js";
|
|
|
13
13
|
import { runFacilitateCommand } from "../src/commands/facilitate.js";
|
|
14
14
|
import { runDiscussCommand } from "../src/commands/discuss.js";
|
|
15
15
|
import { runCallbackCommand } from "../src/commands/callback.js";
|
|
16
|
+
import { runScanLogsCommand } from "../src/commands/scan-logs.js";
|
|
16
17
|
import { AGENT_MODEL, LEAD_MODEL } from "@forwardimpact/libutil/models";
|
|
17
18
|
|
|
18
19
|
const LEAD_OPTIONS = {
|
|
@@ -279,6 +280,35 @@ const definition = {
|
|
|
279
280
|
},
|
|
280
281
|
},
|
|
281
282
|
},
|
|
283
|
+
{
|
|
284
|
+
name: "scan-logs",
|
|
285
|
+
args: [],
|
|
286
|
+
argsUsage: "",
|
|
287
|
+
handler: runScanLogsCommand,
|
|
288
|
+
description:
|
|
289
|
+
"Scan a run's log archive for secret literals; exit non-zero on any hit, fail closed on an unreadable archive",
|
|
290
|
+
options: {
|
|
291
|
+
archive: {
|
|
292
|
+
type: "string",
|
|
293
|
+
description: "Path to an already-resolved log archive (.zip)",
|
|
294
|
+
},
|
|
295
|
+
"run-id": {
|
|
296
|
+
type: "string",
|
|
297
|
+
description:
|
|
298
|
+
"GitHub Actions run id to download the log archive for (with --repo)",
|
|
299
|
+
},
|
|
300
|
+
repo: {
|
|
301
|
+
type: "string",
|
|
302
|
+
description: "owner/repo for the --run-id download",
|
|
303
|
+
},
|
|
304
|
+
secret: {
|
|
305
|
+
type: "string",
|
|
306
|
+
multiple: true,
|
|
307
|
+
description:
|
|
308
|
+
"Repeatable label=literal; the literal (everything after the first =) is searched for in the logs",
|
|
309
|
+
},
|
|
310
|
+
},
|
|
311
|
+
},
|
|
282
312
|
],
|
|
283
313
|
globalOptions: {
|
|
284
314
|
format: { type: "string", description: "Output format (json|text)" },
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@forwardimpact/libharness",
|
|
3
|
-
"version": "1.
|
|
3
|
+
"version": "1.3.0",
|
|
4
4
|
"description": "Autonomous agent team harness — coordinate a lead and participant agents in one async session, with eval, benchmark, and trace tooling to prove the changes worked.",
|
|
5
5
|
"keywords": [
|
|
6
6
|
"orchestration",
|
package/src/agent-runner.js
CHANGED
|
@@ -7,6 +7,7 @@
|
|
|
7
7
|
*/
|
|
8
8
|
|
|
9
9
|
import { AGENT_MODEL } from "@forwardimpact/libutil/models";
|
|
10
|
+
import { resolveClaudeCodeExecutable } from "./claude-code-executable.js";
|
|
10
11
|
|
|
11
12
|
const DEFAULT_ALLOWED_TOOLS = ["Bash", "Read", "Glob", "Grep", "Write", "Edit"];
|
|
12
13
|
|
|
@@ -56,6 +57,10 @@ export class AgentRunner {
|
|
|
56
57
|
* @param {string|object} [deps.systemPrompt] - SDK system prompt (string replaces default; {type:'preset', preset:'claude_code', append} appends)
|
|
57
58
|
* @param {string[]} [deps.disallowedTools] - Tools to explicitly remove from the model's context
|
|
58
59
|
* @param {Record<string, object>} [deps.mcpServers] - MCP server configs to pass to the SDK query
|
|
60
|
+
* @param {string} [deps.pathToClaudeCodeExecutable] - Absolute path to the
|
|
61
|
+
* native `claude` CLI the SDK should spawn. Set for compiled fit-* binaries,
|
|
62
|
+
* which can't self-resolve the SDK's platform optional dependency; omitted
|
|
63
|
+
* from source runs so the SDK resolves its own version-matched binary.
|
|
59
64
|
* @param {object} deps.redactor
|
|
60
65
|
* @param {import("@forwardimpact/libutil/runtime").Runtime} [deps.runtime] -
|
|
61
66
|
* Ambient collaborators. Only `proc.env` is read (to record Skill
|
|
@@ -79,6 +84,9 @@ export class AgentRunner {
|
|
|
79
84
|
this.systemPrompt = deps.systemPrompt ?? null;
|
|
80
85
|
this.disallowedTools = deps.disallowedTools ?? [];
|
|
81
86
|
this.mcpServers = deps.mcpServers ?? null;
|
|
87
|
+
// Optional; read only through a truthy guard in #callOptions, so an absent
|
|
88
|
+
// value stays undefined rather than needing a `?? null` default.
|
|
89
|
+
this.pathToClaudeCodeExecutable = deps.pathToClaudeCodeExecutable;
|
|
82
90
|
this.taskAmend = deps.taskAmend ?? null;
|
|
83
91
|
this.sessionId = null;
|
|
84
92
|
/** @type {AbortController|null} */
|
|
@@ -158,6 +166,9 @@ export class AgentRunner {
|
|
|
158
166
|
}),
|
|
159
167
|
...(this.systemPrompt && { systemPrompt: this.systemPrompt }),
|
|
160
168
|
...(this.mcpServers && { mcpServers: this.mcpServers }),
|
|
169
|
+
...(this.pathToClaudeCodeExecutable && {
|
|
170
|
+
pathToClaudeCodeExecutable: this.pathToClaudeCodeExecutable,
|
|
171
|
+
}),
|
|
161
172
|
};
|
|
162
173
|
}
|
|
163
174
|
|
|
@@ -250,7 +261,14 @@ export class AgentRunner {
|
|
|
250
261
|
}
|
|
251
262
|
}
|
|
252
263
|
|
|
253
|
-
/**
|
|
264
|
+
/**
|
|
265
|
+
* Factory function — wires real dependencies. Resolves the native `claude`
|
|
266
|
+
* executable for compiled fit-* binaries so the SDK doesn't fail to find its
|
|
267
|
+
* own platform optional dependency; an explicit `deps` value overrides it.
|
|
268
|
+
*/
|
|
254
269
|
export function createAgentRunner(deps) {
|
|
255
|
-
return new AgentRunner(
|
|
270
|
+
return new AgentRunner({
|
|
271
|
+
pathToClaudeCodeExecutable: resolveClaudeCodeExecutable(),
|
|
272
|
+
...deps,
|
|
273
|
+
});
|
|
256
274
|
}
|
|
@@ -0,0 +1,44 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Resolve the Claude Code CLI the Agent SDK should spawn.
|
|
3
|
+
*
|
|
4
|
+
* `query()` spawns a native `claude` binary that the SDK resolves from its own
|
|
5
|
+
* platform-specific optional dependency (`@anthropic-ai/claude-agent-sdk-<platform>`).
|
|
6
|
+
* `bun build --compile` bundles the SDK's JavaScript but not that separate
|
|
7
|
+
* native package — it is not part of the import graph — so a compiled fit-*
|
|
8
|
+
* binary can't self-resolve it and `query()` throws "Native CLI binary for
|
|
9
|
+
* <platform> not found".
|
|
10
|
+
*
|
|
11
|
+
* In a compiled binary we point the SDK at the standalone `claude` on PATH,
|
|
12
|
+
* installed beside fit-harness by the bootstrap action's `fit-install.sh`.
|
|
13
|
+
* Running from source keeps `node_modules`, where the SDK resolves its own
|
|
14
|
+
* version-matched binary, so there we return undefined and defer to the SDK.
|
|
15
|
+
*/
|
|
16
|
+
import { LIBCLI_IS_COMPILED } from "@forwardimpact/libcli";
|
|
17
|
+
|
|
18
|
+
/**
|
|
19
|
+
* @param {object} [deps]
|
|
20
|
+
* @param {(cmd: string) => string | null | undefined} [deps.which] -
|
|
21
|
+
* PATH resolver (injected for testing).
|
|
22
|
+
* @param {boolean} [deps.isCompiled] -
|
|
23
|
+
* Whether this is a `bun --compile` binary (injected for testing).
|
|
24
|
+
* @returns {string | undefined} absolute path to `claude`, or undefined to
|
|
25
|
+
* defer resolution to the SDK.
|
|
26
|
+
*/
|
|
27
|
+
export function resolveClaudeCodeExecutable({
|
|
28
|
+
which = defaultWhich,
|
|
29
|
+
isCompiled = LIBCLI_IS_COMPILED,
|
|
30
|
+
} = {}) {
|
|
31
|
+
if (!isCompiled) return undefined;
|
|
32
|
+
return which("claude") ?? undefined;
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
/**
|
|
36
|
+
* A compiled fit-* binary runs under the Bun runtime, which exposes a
|
|
37
|
+
* synchronous PATH resolver. The `typeof` guard keeps this safe under Node too.
|
|
38
|
+
*/
|
|
39
|
+
function defaultWhich(cmd) {
|
|
40
|
+
if (typeof Bun !== "undefined" && typeof Bun.which === "function") {
|
|
41
|
+
return Bun.which(cmd);
|
|
42
|
+
}
|
|
43
|
+
return null;
|
|
44
|
+
}
|
|
@@ -0,0 +1,155 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* `fit-harness scan-logs` — scan a run's log archive for secret literals and
|
|
3
|
+
* fail closed.
|
|
4
|
+
*
|
|
5
|
+
* A run-lifecycle concern (not an NDJSON trace, so it lives here rather than
|
|
6
|
+
* in `fit-trace`): after a CI run that handled secrets, download or accept the
|
|
7
|
+
* run's own log archive and assert none of a supplied set of literals leaked
|
|
8
|
+
* into it. Any hit exits non-zero; any download/extract failure also exits
|
|
9
|
+
* non-zero — the gate must never silently disarm.
|
|
10
|
+
*
|
|
11
|
+
* Log resolution:
|
|
12
|
+
* - `--archive <zip>` — an already-resolved archive (extracted locally).
|
|
13
|
+
* - `--run-id <id> --repo <owner/repo>` — download this run's archive via
|
|
14
|
+
* `gh` first, then extract.
|
|
15
|
+
*
|
|
16
|
+
* Secrets are `--secret <label>=<literal>`, repeatable. The literal is
|
|
17
|
+
* everything after the FIRST `=` (JWTs and base64 keys contain `=`); the label
|
|
18
|
+
* is only cosmetic, named in the `FAIL:` line.
|
|
19
|
+
*/
|
|
20
|
+
|
|
21
|
+
import { join } from "node:path";
|
|
22
|
+
|
|
23
|
+
/**
|
|
24
|
+
* Parse repeatable `--secret label=literal` flags. libcli's `multiple: true`
|
|
25
|
+
* yields an array from node's parseArgs in every case; tolerate a bare string
|
|
26
|
+
* or undefined defensively. Split on the FIRST `=` only.
|
|
27
|
+
*
|
|
28
|
+
* @param {string[]|string|undefined} secretOpt
|
|
29
|
+
* @returns {{label: string, literal: string}[]}
|
|
30
|
+
*/
|
|
31
|
+
export function parseSecrets(secretOpt) {
|
|
32
|
+
const arr = Array.isArray(secretOpt)
|
|
33
|
+
? secretOpt
|
|
34
|
+
: secretOpt
|
|
35
|
+
? [secretOpt]
|
|
36
|
+
: [];
|
|
37
|
+
return arr.map((s) => {
|
|
38
|
+
const idx = s.indexOf("=");
|
|
39
|
+
if (idx === -1) return { label: s, literal: "" };
|
|
40
|
+
return { label: s.slice(0, idx), literal: s.slice(idx + 1) };
|
|
41
|
+
});
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
/**
|
|
45
|
+
* Walk a directory tree and return every file path. Uses per-level readdir so
|
|
46
|
+
* it works against both node:fs and the libmock fs (no `recursive` reliance).
|
|
47
|
+
*/
|
|
48
|
+
async function collectFiles(dir, runtime) {
|
|
49
|
+
const out = [];
|
|
50
|
+
const entries = await runtime.fs.readdir(dir, { withFileTypes: true });
|
|
51
|
+
for (const ent of entries) {
|
|
52
|
+
const full = join(dir, ent.name);
|
|
53
|
+
if (ent.isDirectory()) {
|
|
54
|
+
out.push(...(await collectFiles(full, runtime)));
|
|
55
|
+
} else if (ent.isFile()) {
|
|
56
|
+
out.push(full);
|
|
57
|
+
}
|
|
58
|
+
}
|
|
59
|
+
return out;
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
/**
|
|
63
|
+
* Scan every file under `dir` for each secret literal. Returns the labels of
|
|
64
|
+
* secrets whose non-empty literal appears in any file (empty literals are
|
|
65
|
+
* skipped — a secret the run never set cannot leak).
|
|
66
|
+
*
|
|
67
|
+
* @param {object} params
|
|
68
|
+
* @param {string} params.dir
|
|
69
|
+
* @param {{label: string, literal: string}[]} params.secrets
|
|
70
|
+
* @param {import('@forwardimpact/libutil/runtime').Runtime} params.runtime
|
|
71
|
+
* @returns {Promise<string[]>} labels that hit
|
|
72
|
+
*/
|
|
73
|
+
export async function scanDirectory({ dir, secrets, runtime }) {
|
|
74
|
+
const files = await collectFiles(dir, runtime);
|
|
75
|
+
const contents = await Promise.all(
|
|
76
|
+
files.map((f) => runtime.fs.readFile(f, "utf8").catch(() => "")),
|
|
77
|
+
);
|
|
78
|
+
const failures = [];
|
|
79
|
+
for (const { label, literal } of secrets) {
|
|
80
|
+
if (!literal) continue;
|
|
81
|
+
if (contents.some((c) => c.includes(literal))) failures.push(label);
|
|
82
|
+
}
|
|
83
|
+
return failures;
|
|
84
|
+
}
|
|
85
|
+
|
|
86
|
+
/**
|
|
87
|
+
* Resolve a directory of extracted log files, downloading the archive first
|
|
88
|
+
* when given a run id. Throws (→ fail closed) on any download or extract
|
|
89
|
+
* failure or on missing/invalid inputs.
|
|
90
|
+
*/
|
|
91
|
+
async function resolveLogsDir({ options, runtime }) {
|
|
92
|
+
const tmpRoot = runtime.proc.env.RUNNER_TEMP || "/tmp";
|
|
93
|
+
const dir = await runtime.fs.mkdtemp(join(tmpRoot, "scan-logs-"));
|
|
94
|
+
let zip = options.archive;
|
|
95
|
+
|
|
96
|
+
if (!zip) {
|
|
97
|
+
const runId = options["run-id"];
|
|
98
|
+
const repo = options.repo;
|
|
99
|
+
if (!runId || !repo) {
|
|
100
|
+
throw new Error("requires --archive, or --run-id and --repo");
|
|
101
|
+
}
|
|
102
|
+
if (!/^\d+$/.test(String(runId))) {
|
|
103
|
+
throw new Error("--run-id must be numeric");
|
|
104
|
+
}
|
|
105
|
+
if (!/^[\w.-]+\/[\w.-]+$/.test(repo)) {
|
|
106
|
+
throw new Error("--repo must be owner/name");
|
|
107
|
+
}
|
|
108
|
+
zip = join(dir, "run-logs.zip");
|
|
109
|
+
const dl = await runtime.subprocess.run("bash", [
|
|
110
|
+
"-c",
|
|
111
|
+
`gh api -H "Accept: application/vnd.github+json" ` +
|
|
112
|
+
`"/repos/${repo}/actions/runs/${runId}/logs" > "${zip}"`,
|
|
113
|
+
]);
|
|
114
|
+
if (dl.exitCode !== 0) {
|
|
115
|
+
throw new Error(
|
|
116
|
+
`log archive download failed (gh exit ${dl.exitCode}): ${dl.stderr ?? ""}`,
|
|
117
|
+
);
|
|
118
|
+
}
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
const unz = await runtime.subprocess.run("unzip", ["-q", zip, "-d", dir]);
|
|
122
|
+
if (unz.exitCode !== 0) {
|
|
123
|
+
throw new Error(
|
|
124
|
+
`log archive empty/unreadable (unzip exit ${unz.exitCode})`,
|
|
125
|
+
);
|
|
126
|
+
}
|
|
127
|
+
return dir;
|
|
128
|
+
}
|
|
129
|
+
|
|
130
|
+
/**
|
|
131
|
+
* scan-logs command handler.
|
|
132
|
+
*
|
|
133
|
+
* @param {import("@forwardimpact/libcli").InvocationContext} ctx
|
|
134
|
+
* @returns {Promise<{ok: boolean, code: number, error?: string}>}
|
|
135
|
+
*/
|
|
136
|
+
export async function runScanLogsCommand(ctx) {
|
|
137
|
+
const runtime = ctx.deps.runtime;
|
|
138
|
+
const options = ctx.options;
|
|
139
|
+
const secrets = parseSecrets(options.secret);
|
|
140
|
+
|
|
141
|
+
let dir;
|
|
142
|
+
try {
|
|
143
|
+
dir = await resolveLogsDir({ options, runtime });
|
|
144
|
+
} catch (err) {
|
|
145
|
+
// Fail closed: an unresolvable archive must not pass as "no leak". The
|
|
146
|
+
// dispatcher prints the returned `error`, so don't also write it here.
|
|
147
|
+
return { ok: false, code: 1, error: `scan-logs: ${err.message}` };
|
|
148
|
+
}
|
|
149
|
+
|
|
150
|
+
const failures = await scanDirectory({ dir, secrets, runtime });
|
|
151
|
+
for (const label of failures) {
|
|
152
|
+
runtime.proc.stderr.write(`FAIL: ${label} literal in run logs\n`);
|
|
153
|
+
}
|
|
154
|
+
return { ok: failures.length === 0, code: failures.length ? 1 : 0 };
|
|
155
|
+
}
|
package/src/index.js
CHANGED
|
@@ -11,6 +11,7 @@ export {
|
|
|
11
11
|
pickTraceArtifact,
|
|
12
12
|
} from "./trace-github.js";
|
|
13
13
|
export { AgentRunner, createAgentRunner } from "./agent-runner.js";
|
|
14
|
+
export { resolveClaudeCodeExecutable } from "./claude-code-executable.js";
|
|
14
15
|
export {
|
|
15
16
|
composeProfilePrompt,
|
|
16
17
|
composeLeadPrompt,
|