@forwardimpact/libharness 2.0.0 → 3.0.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +68 -65
- package/package.json +15 -13
- package/src/advisor.js +47 -41
- package/src/agent-runner.js +58 -48
- package/src/benchmark/apm-installer.js +28 -28
- package/src/benchmark/env-loader.js +24 -16
- package/src/benchmark/grade.js +44 -41
- package/src/benchmark/hidden-tests.js +25 -24
- package/src/benchmark/hook-env.js +11 -9
- package/src/benchmark/invariants.js +20 -17
- package/src/benchmark/judge.js +29 -28
- package/src/benchmark/npm-installer.js +9 -8
- package/src/benchmark/report.js +53 -50
- package/src/benchmark/result.js +24 -23
- package/src/benchmark/runner.js +75 -69
- package/src/benchmark/scheduler.js +17 -16
- package/src/benchmark/task-family.js +29 -27
- package/src/benchmark/trace-split.js +9 -8
- package/src/benchmark/workdir.js +27 -25
- package/src/claude-code-executable.js +11 -11
- package/src/commands/advisor-flags.js +8 -7
- package/src/commands/assert.js +16 -15
- package/src/commands/benchmark-definition.js +20 -20
- package/src/commands/benchmark-grade.js +13 -12
- package/src/commands/benchmark-report.js +5 -5
- package/src/commands/benchmark-run.js +31 -28
- package/src/commands/by-discussion.js +11 -11
- package/src/commands/callback.js +11 -11
- package/src/commands/discuss.js +8 -7
- package/src/commands/facilitate.js +16 -14
- package/src/commands/output.js +4 -3
- package/src/commands/run.js +15 -15
- package/src/commands/scan-logs.js +22 -20
- package/src/commands/selfedit.js +124 -0
- package/src/commands/supervise.js +13 -11
- package/src/commands/task-input.js +9 -9
- package/src/commands/tee.js +11 -10
- package/src/commands/trace.js +55 -42
- package/src/commands/work-tracker.js +4 -3
- package/src/cost.js +17 -17
- package/src/discuss-tools.js +16 -16
- package/src/discusser.js +39 -38
- package/src/events/github.js +54 -37
- package/src/facilitator.js +21 -21
- package/src/inbox-poller.js +4 -4
- package/src/judge.js +32 -30
- package/src/message-bus.js +12 -11
- package/src/orchestration-loop.js +35 -36
- package/src/orchestration-toolkit.js +58 -53
- package/src/orchestrator-helpers.js +2 -2
- package/src/profile-prompt.js +54 -53
- package/src/redaction.js +63 -57
- package/src/render/line-renderer.js +5 -5
- package/src/render/orchestrator-filter.js +3 -3
- package/src/render/palette.js +11 -9
- package/src/render/tool-hints.js +18 -15
- package/src/render/turn-renderer.js +4 -4
- package/src/reply-emitter.js +2 -2
- package/src/sequence-counter.js +4 -3
- package/src/signature-filter.js +7 -6
- package/src/supervisor.js +19 -18
- package/src/tee-writer.js +25 -25
- package/src/trace-collector.js +53 -48
- package/src/trace-github.js +53 -44
- package/src/trace-multi.js +16 -14
- package/src/trace-query.js +61 -52
- package/src/trace-render.js +19 -19
- package/src/trace-usage.js +31 -28
- package/src/transcript-recorder.js +24 -20
- package/bin/fit-benchmark.js +0 -44
- package/bin/fit-harness.js +0 -412
- package/bin/fit-selfedit.js +0 -165
- package/bin/fit-trace.js +0 -520
package/src/commands/discuss.js
CHANGED
|
@@ -17,8 +17,8 @@ function parseAgentProfiles(raw, cwd, maxTurns) {
|
|
|
17
17
|
}
|
|
18
18
|
|
|
19
19
|
/**
|
|
20
|
-
* Parse and validate discuss command options.
|
|
21
|
-
* defaults and the legacy-flag clean break.
|
|
20
|
+
* Parse and validate discuss command options. This function is exported so a
|
|
21
|
+
* test can verify the defaults and the legacy-flag clean break.
|
|
22
22
|
* @param {object} values - Parsed option values
|
|
23
23
|
* @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
|
|
24
24
|
* @returns {object}
|
|
@@ -30,8 +30,8 @@ export function parseDiscussOptions(values, runtime) {
|
|
|
30
30
|
);
|
|
31
31
|
|
|
32
32
|
const profilesRaw = values["agent-profiles"];
|
|
33
|
-
// `||` (not `??`) so an empty-string flag from a CI forwarder falls back
|
|
34
|
-
// the default
|
|
33
|
+
// `||` (not `??`) so an empty-string flag from a CI forwarder falls back
|
|
34
|
+
// to the default. The empty string does not override the default.
|
|
35
35
|
const agentCwd = resolve(values["agent-cwd"] || ".");
|
|
36
36
|
|
|
37
37
|
const maxTurnsRaw = values["max-turns"] || "40";
|
|
@@ -74,8 +74,8 @@ export function parseDiscussOptions(values, runtime) {
|
|
|
74
74
|
|
|
75
75
|
/**
|
|
76
76
|
* Discuss command — run a discusser-led session with suspend/resume
|
|
77
|
-
* semantics
|
|
78
|
-
*
|
|
77
|
+
* semantics. The session threads `discussion_id` through the trace, so
|
|
78
|
+
* you can query multi-run conversations as one.
|
|
79
79
|
*
|
|
80
80
|
* @param {import("@forwardimpact/libcli").InvocationContext} ctx
|
|
81
81
|
* @returns {Promise<{ok: boolean, code?: number, error?: string}>}
|
|
@@ -102,7 +102,8 @@ export async function runDiscussCommand(ctx) {
|
|
|
102
102
|
runtime.proc.env.LIBHARNESS_AGENT_PROFILE = opts.leadProfile;
|
|
103
103
|
}
|
|
104
104
|
// Unconditional so the default "github" is observable to the agent's
|
|
105
|
-
// active-tracker resolution
|
|
105
|
+
// active-tracker resolution. This mirrors --agent-profile's env write
|
|
106
|
+
// above.
|
|
106
107
|
runtime.proc.env.LIBHARNESS_WORK_TRACKER = opts.workTracker;
|
|
107
108
|
|
|
108
109
|
const { query } = await import("@anthropic-ai/claude-agent-sdk");
|
|
@@ -22,9 +22,9 @@ function parseAgentProfiles(raw, cwd, maxTurns) {
|
|
|
22
22
|
}
|
|
23
23
|
|
|
24
24
|
/**
|
|
25
|
-
* Parse and validate facilitate command options.
|
|
26
|
-
*
|
|
27
|
-
* of the package's public API.
|
|
25
|
+
* Parse and validate facilitate command options. This function is exported
|
|
26
|
+
* so a test can cover the contract that threads `--max-turns` to each
|
|
27
|
+
* agent. It is not part of the package's public API.
|
|
28
28
|
* @param {object} values - Parsed option values
|
|
29
29
|
* @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
|
|
30
30
|
* @returns {object} Parsed options
|
|
@@ -37,17 +37,18 @@ export function parseFacilitateOptions(values, runtime) {
|
|
|
37
37
|
|
|
38
38
|
const profilesRaw = values["agent-profiles"];
|
|
39
39
|
if (!profilesRaw) throw new Error("--agent-profiles is required");
|
|
40
|
-
// `||` (not `??`) so an empty-string flag from a CI forwarder falls back
|
|
41
|
-
// the default
|
|
40
|
+
// `||` (not `??`) so an empty-string flag from a CI forwarder falls back
|
|
41
|
+
// to the default. The empty string does not override the default.
|
|
42
42
|
const agentCwd = resolve(values["agent-cwd"] || ".");
|
|
43
43
|
|
|
44
44
|
const maxTurnsRaw = values["max-turns"] || "20";
|
|
45
45
|
const maxTurns = maxTurnsRaw === "0" ? 0 : parseInt(maxTurnsRaw, 10);
|
|
46
46
|
|
|
47
|
-
// Thread --max-turns into each participant
|
|
48
|
-
// agent silently falls back to the 50-turn default in
|
|
49
|
-
// when the caller raises the budget.
|
|
50
|
-
// staff-engineer terminated at 51
|
|
47
|
+
// Thread --max-turns into each participant. Without this, every
|
|
48
|
+
// facilitated agent silently falls back to the 50-turn default in
|
|
49
|
+
// facilitator.js, even when the caller raises the budget. Run
|
|
50
|
+
// 26078312414 showed this. In it, staff-engineer terminated at 51
|
|
51
|
+
// turns despite --max-turns=200.
|
|
51
52
|
const agentConfigs = parseAgentProfiles(profilesRaw, agentCwd, maxTurns);
|
|
52
53
|
|
|
53
54
|
return {
|
|
@@ -68,7 +69,7 @@ export function parseFacilitateOptions(values, runtime) {
|
|
|
68
69
|
/**
|
|
69
70
|
* Facilitate command — run a facilitated multi-agent session.
|
|
70
71
|
*
|
|
71
|
-
* Usage:
|
|
72
|
+
* Usage: gemba-harness facilitate [options]
|
|
72
73
|
*
|
|
73
74
|
* @param {import("@forwardimpact/libcli").InvocationContext} ctx
|
|
74
75
|
* @returns {Promise<{ok: boolean, code?: number, error?: string}>}
|
|
@@ -77,9 +78,9 @@ export async function runFacilitateCommand(ctx) {
|
|
|
77
78
|
const runtime = ctx.deps.runtime;
|
|
78
79
|
const opts = parseFacilitateOptions(ctx.options, runtime);
|
|
79
80
|
|
|
80
|
-
// Build the redactor as the first observable side-effect after
|
|
81
|
-
//
|
|
82
|
-
// env
|
|
81
|
+
// Build the redactor as the first observable side-effect after the parser
|
|
82
|
+
// reads the options. The env snapshot must freeze BEFORE any in-process
|
|
83
|
+
// env write the command performs (e.g. LIBHARNESS_AGENT_PROFILE).
|
|
83
84
|
const redactor = createRedactor({ runtime });
|
|
84
85
|
|
|
85
86
|
const fileStream = opts.outputPath
|
|
@@ -98,7 +99,8 @@ export async function runFacilitateCommand(ctx) {
|
|
|
98
99
|
runtime.proc.env.LIBHARNESS_AGENT_PROFILE = opts.facilitatorProfile;
|
|
99
100
|
}
|
|
100
101
|
// Unconditional so the default "github" is observable to the agent's
|
|
101
|
-
// active-tracker resolution
|
|
102
|
+
// active-tracker resolution. This mirrors --agent-profile's env write
|
|
103
|
+
// above.
|
|
102
104
|
runtime.proc.env.LIBHARNESS_WORK_TRACKER = opts.workTracker;
|
|
103
105
|
|
|
104
106
|
const { query } = await import("@anthropic-ai/claude-agent-sdk");
|
package/src/commands/output.js
CHANGED
|
@@ -5,7 +5,7 @@ import { createTraceCollector } from "@forwardimpact/libharness";
|
|
|
5
5
|
* Output command — process a complete NDJSON trace from stdin and write
|
|
6
6
|
* formatted output to stdout.
|
|
7
7
|
*
|
|
8
|
-
* Usage:
|
|
8
|
+
* Usage: gemba-harness output [--format=json|text] < trace.ndjson
|
|
9
9
|
*
|
|
10
10
|
* @param {import("@forwardimpact/libcli").InvocationContext} ctx
|
|
11
11
|
* @returns {Promise<{ok: true}>}
|
|
@@ -21,8 +21,9 @@ export async function runOutputCommand(ctx) {
|
|
|
21
21
|
now: () => isoTimestamp(runtime.clock.now()),
|
|
22
22
|
});
|
|
23
23
|
|
|
24
|
-
// `runtime.proc.stdin` is an AsyncIterable of UTF-8 lines
|
|
25
|
-
//
|
|
24
|
+
// `runtime.proc.stdin` is an AsyncIterable of UTF-8 lines. The runtime
|
|
25
|
+
// splits them on newlines. So each yielded value is exactly one NDJSON
|
|
26
|
+
// record.
|
|
26
27
|
for await (const line of runtime.proc.stdin) {
|
|
27
28
|
collector.addLine(line);
|
|
28
29
|
}
|
package/src/commands/run.js
CHANGED
|
@@ -34,8 +34,8 @@ export function parseRunOptions(values, runtime) {
|
|
|
34
34
|
values,
|
|
35
35
|
runtime,
|
|
36
36
|
);
|
|
37
|
-
// `||` (not `??`) so an empty-string flag from a CI forwarder falls back
|
|
38
|
-
// the default
|
|
37
|
+
// `||` (not `??`) so an empty-string flag from a CI forwarder falls back
|
|
38
|
+
// to the default. The empty string does not override the default.
|
|
39
39
|
const maxTurnsRaw = values["max-turns"] || "50";
|
|
40
40
|
|
|
41
41
|
return {
|
|
@@ -64,10 +64,10 @@ const devNull = new Writable({
|
|
|
64
64
|
|
|
65
65
|
/**
|
|
66
66
|
* Wire the run-mode agent session: external MCP entry, `LIBHARNESS_*` env
|
|
67
|
-
* writes, system-prompt composition
|
|
68
|
-
* the advisor
|
|
69
|
-
* server
|
|
70
|
-
* so
|
|
67
|
+
* writes, and system-prompt composition. When an advisor model is set, also
|
|
68
|
+
* wire the advisor (budget, recorder, advisor session, and a dedicated MCP
|
|
69
|
+
* server that holds only the `Advisor` tool). This function is extracted
|
|
70
|
+
* from `runRunCommand` so a test can inject a fake `query`.
|
|
71
71
|
*
|
|
72
72
|
* Run mode has no stop path (the command simply awaits the runner), so the
|
|
73
73
|
* consult timeout is deliberately the advisor's only guard.
|
|
@@ -120,9 +120,9 @@ export async function wireRunSession({
|
|
|
120
120
|
runtime.proc.env.LIBHARNESS_WORK_TRACKER = opts.workTracker;
|
|
121
121
|
|
|
122
122
|
// With a profile, the consult guidance rides the profile composer's
|
|
123
|
-
// amendment parameter
|
|
123
|
+
// amendment parameter. With no profile, a preset-append prompt carries
|
|
124
124
|
// the guidance as its only session-protocol fragment. Advisor off and no
|
|
125
|
-
// profile means no system prompt
|
|
125
|
+
// profile means no system prompt. That is today's behavior, unchanged.
|
|
126
126
|
let systemPrompt;
|
|
127
127
|
if (opts.agentProfile) {
|
|
128
128
|
systemPrompt = composeProfilePrompt(opts.agentProfile, {
|
|
@@ -161,7 +161,7 @@ export async function wireRunSession({
|
|
|
161
161
|
budget,
|
|
162
162
|
model: opts.advisorModel,
|
|
163
163
|
});
|
|
164
|
-
// No allowlist push
|
|
164
|
+
// No allowlist push. In-process SDK MCP servers work under
|
|
165
165
|
// bypassPermissions without allowlist entries (loop-mode precedent).
|
|
166
166
|
mcpServers = {
|
|
167
167
|
...mcpServers,
|
|
@@ -195,9 +195,9 @@ export async function wireRunSession({
|
|
|
195
195
|
}
|
|
196
196
|
|
|
197
197
|
/**
|
|
198
|
-
* Run command — execute a single agent
|
|
198
|
+
* Run command — execute a single agent through the Claude Agent SDK.
|
|
199
199
|
*
|
|
200
|
-
* Usage:
|
|
200
|
+
* Usage: gemba-harness run [options]
|
|
201
201
|
*
|
|
202
202
|
* @param {import("@forwardimpact/libcli").InvocationContext} ctx
|
|
203
203
|
* @returns {Promise<{ok: boolean, code?: number, error?: string}>}
|
|
@@ -206,12 +206,12 @@ export async function runRunCommand(ctx) {
|
|
|
206
206
|
const runtime = ctx.deps.runtime;
|
|
207
207
|
const opts = parseRunOptions(ctx.options, runtime);
|
|
208
208
|
|
|
209
|
-
// Build the redactor as the first observable side-effect after
|
|
210
|
-
//
|
|
211
|
-
// env
|
|
209
|
+
// Build the redactor as the first observable side-effect after the parser
|
|
210
|
+
// reads the options. The env snapshot must freeze BEFORE any in-process
|
|
211
|
+
// env write the command performs (e.g. LIBHARNESS_AGENT_PROFILE).
|
|
212
212
|
const redactor = createRedactor({ runtime });
|
|
213
213
|
|
|
214
|
-
//
|
|
214
|
+
// With --output, stream text to stdout and write NDJSON to the file.
|
|
215
215
|
// Otherwise, write NDJSON directly to stdout (backwards-compatible).
|
|
216
216
|
const fileStream = opts.outputPath
|
|
217
217
|
? runtime.fs.createWriteStream(opts.outputPath)
|
|
@@ -1,28 +1,29 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* `
|
|
2
|
+
* `gemba-harness scan-logs` — scan a run's log archive for secret literals and
|
|
3
3
|
* fail closed.
|
|
4
4
|
*
|
|
5
|
-
*
|
|
6
|
-
* in `
|
|
7
|
-
* run's own log archive
|
|
8
|
-
* into it. Any hit exits non-zero
|
|
9
|
-
* non-zero
|
|
5
|
+
* This is a run-lifecycle concern. It is not an NDJSON trace, so it lives
|
|
6
|
+
* here rather than in `gemba-trace`. After a CI run that handled secrets,
|
|
7
|
+
* download or accept the run's own log archive. Then assert that none of a
|
|
8
|
+
* supplied set of literals leaked into it. Any hit exits non-zero. Any
|
|
9
|
+
* download or extract failure also exits non-zero. The gate must never
|
|
10
|
+
* silently disarm.
|
|
10
11
|
*
|
|
11
12
|
* Log resolution:
|
|
12
13
|
* - `--archive <zip>` — an already-resolved archive (extracted locally).
|
|
13
|
-
* - `--run-id <id> --repo <owner/repo>` — download this run's archive
|
|
14
|
+
* - `--run-id <id> --repo <owner/repo>` — download this run's archive with
|
|
14
15
|
* `gh` first, then extract.
|
|
15
16
|
*
|
|
16
17
|
* Secrets are `--secret <label>=<literal>`, repeatable. The literal is
|
|
17
|
-
* everything after the FIRST `=` (JWTs and base64 keys contain `=`)
|
|
18
|
-
* is only cosmetic
|
|
18
|
+
* everything after the FIRST `=` (JWTs and base64 keys contain `=`). The
|
|
19
|
+
* label is only cosmetic. The `FAIL:` line names it.
|
|
19
20
|
*/
|
|
20
21
|
|
|
21
22
|
import { join } from "node:path";
|
|
22
23
|
|
|
23
24
|
/**
|
|
24
25
|
* Parse repeatable `--secret label=literal` flags. libcli's `multiple: true`
|
|
25
|
-
* yields an array from node's parseArgs in every case
|
|
26
|
+
* yields an array from node's parseArgs in every case. Tolerate a bare string
|
|
26
27
|
* or undefined defensively. Split on the FIRST `=` only.
|
|
27
28
|
*
|
|
28
29
|
* @param {string[]|string|undefined} secretOpt
|
|
@@ -42,8 +43,9 @@ export function parseSecrets(secretOpt) {
|
|
|
42
43
|
}
|
|
43
44
|
|
|
44
45
|
/**
|
|
45
|
-
* Walk a directory tree and return every file path.
|
|
46
|
-
* it works against both node:fs and the libmock fs (no `recursive`
|
|
46
|
+
* Walk a directory tree and return every file path. It uses per-level readdir
|
|
47
|
+
* so it works against both node:fs and the libmock fs (no `recursive`
|
|
48
|
+
* reliance).
|
|
47
49
|
*/
|
|
48
50
|
async function collectFiles(dir, runtime) {
|
|
49
51
|
const out = [];
|
|
@@ -60,9 +62,9 @@ async function collectFiles(dir, runtime) {
|
|
|
60
62
|
}
|
|
61
63
|
|
|
62
64
|
/**
|
|
63
|
-
* Scan every file under `dir` for each secret literal.
|
|
64
|
-
* secrets whose non-empty literal appears in any file
|
|
65
|
-
*
|
|
65
|
+
* Scan every file under `dir` for each secret literal. Return the labels of
|
|
66
|
+
* secrets whose non-empty literal appears in any file. The scan skips empty
|
|
67
|
+
* literals, because a secret the run never set cannot leak.
|
|
66
68
|
*
|
|
67
69
|
* @param {object} params
|
|
68
70
|
* @param {string} params.dir
|
|
@@ -84,9 +86,9 @@ export async function scanDirectory({ dir, secrets, runtime }) {
|
|
|
84
86
|
}
|
|
85
87
|
|
|
86
88
|
/**
|
|
87
|
-
* Resolve a directory of extracted log files
|
|
88
|
-
*
|
|
89
|
-
* failure
|
|
89
|
+
* Resolve a directory of extracted log files. Download the archive first when
|
|
90
|
+
* the caller gives a run id. Throw (→ fail closed) on any download or extract
|
|
91
|
+
* failure, and on a missing or invalid input.
|
|
90
92
|
*/
|
|
91
93
|
async function resolveLogsDir({ options, runtime }) {
|
|
92
94
|
const tmpRoot = runtime.proc.env.RUNNER_TEMP || "/tmp";
|
|
@@ -142,8 +144,8 @@ export async function runScanLogsCommand(ctx) {
|
|
|
142
144
|
try {
|
|
143
145
|
dir = await resolveLogsDir({ options, runtime });
|
|
144
146
|
} catch (err) {
|
|
145
|
-
// Fail closed
|
|
146
|
-
// dispatcher prints the returned `error`, so
|
|
147
|
+
// Fail closed. An unresolvable archive must not pass as "no leak". The
|
|
148
|
+
// dispatcher prints the returned `error`, so do not also write it here.
|
|
147
149
|
return { ok: false, code: 1, error: `scan-logs: ${err.message}` };
|
|
148
150
|
}
|
|
149
151
|
|
|
@@ -0,0 +1,124 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Safeguard-and-write logic behind the `gemba-selfedit` bin. It writes
|
|
3
|
+
* content to a path that .claude/settings.json permits Edit on. The write
|
|
4
|
+
* happens only on a non-main git branch. See
|
|
5
|
+
* libraries/libharness/README.md § gemba-selfedit for the full rationale.
|
|
6
|
+
*/
|
|
7
|
+
|
|
8
|
+
import { resolve, relative, dirname } from "node:path";
|
|
9
|
+
|
|
10
|
+
import { minimatch } from "minimatch";
|
|
11
|
+
|
|
12
|
+
/** A safeguard violation. Callers map it to exit code 2. */
|
|
13
|
+
export class SelfeditError extends Error {
|
|
14
|
+
/** @param {string} message failure description */
|
|
15
|
+
constructor(message) {
|
|
16
|
+
super(message);
|
|
17
|
+
this.name = "SelfeditError";
|
|
18
|
+
}
|
|
19
|
+
}
|
|
20
|
+
|
|
21
|
+
/**
|
|
22
|
+
* Check every safeguard for a selfedit write, then perform it.
|
|
23
|
+
*
|
|
24
|
+
* The function checks these safeguards in order:
|
|
25
|
+
* 1. The nearest .claude/settings.json must contain an Edit(<glob>) rule in
|
|
26
|
+
* permissions.allow[] that resolves to the target path.
|
|
27
|
+
* 2. HEAD must not be detached and the current branch must not be 'main'.
|
|
28
|
+
* 3. The target's parent directory must exist.
|
|
29
|
+
*
|
|
30
|
+
* @param {string} targetArg target path from the command line
|
|
31
|
+
* @param {Buffer} content bytes to write
|
|
32
|
+
* @param {{ runtime: object }} deps runtime bag (fsSync, proc, subprocess,
|
|
33
|
+
* finder). targetArg resolves against `runtime.proc.cwd()`
|
|
34
|
+
* @returns {{ bytes: number, relativeTarget: string, matchedPattern: string,
|
|
35
|
+
* branch: string }} what the function wrote and which rule allowed it
|
|
36
|
+
* @throws {SelfeditError} on any safeguard violation
|
|
37
|
+
*/
|
|
38
|
+
export function runSelfeditCommand(targetArg, content, { runtime }) {
|
|
39
|
+
const { fsSync, proc, subprocess, finder } = runtime;
|
|
40
|
+
const cwd = proc.cwd();
|
|
41
|
+
const absoluteTarget = resolve(cwd, targetArg);
|
|
42
|
+
|
|
43
|
+
// Safeguard 1: settings.json must grant Edit() on this path. Resolve the
|
|
44
|
+
// finder off the runtime bag. Do not construct a Finder here.
|
|
45
|
+
const settingsPath = finder.findUpward(
|
|
46
|
+
dirname(absoluteTarget),
|
|
47
|
+
".claude/settings.json",
|
|
48
|
+
20,
|
|
49
|
+
);
|
|
50
|
+
if (!settingsPath) {
|
|
51
|
+
throw new SelfeditError(
|
|
52
|
+
`no .claude/settings.json found walking upward from ${dirname(absoluteTarget)}`,
|
|
53
|
+
);
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
const projectRoot = dirname(dirname(settingsPath));
|
|
57
|
+
const relativeTarget = relative(projectRoot, absoluteTarget);
|
|
58
|
+
|
|
59
|
+
let settings;
|
|
60
|
+
try {
|
|
61
|
+
settings = JSON.parse(fsSync.readFileSync(settingsPath, "utf8"));
|
|
62
|
+
} catch (err) {
|
|
63
|
+
throw new SelfeditError(`failed to parse ${settingsPath}: ${err.message}`);
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
const allowRules = settings?.permissions?.allow;
|
|
67
|
+
if (!Array.isArray(allowRules)) {
|
|
68
|
+
throw new SelfeditError(`${settingsPath} has no permissions.allow[] array`);
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
const editPatterns = allowRules
|
|
72
|
+
.filter((rule) => typeof rule === "string")
|
|
73
|
+
.map((rule) => rule.match(/^Edit\((.+)\)$/)?.[1])
|
|
74
|
+
.filter(Boolean);
|
|
75
|
+
|
|
76
|
+
if (editPatterns.length === 0) {
|
|
77
|
+
throw new SelfeditError(
|
|
78
|
+
`${settingsPath} has no Edit() rules in permissions.allow[]`,
|
|
79
|
+
);
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
const matchedPattern = editPatterns.find((pattern) =>
|
|
83
|
+
minimatch(relativeTarget, pattern, { dot: true }),
|
|
84
|
+
);
|
|
85
|
+
if (!matchedPattern) {
|
|
86
|
+
throw new SelfeditError(
|
|
87
|
+
`no Edit() rule in ${relative(projectRoot, settingsPath)} matches '${relativeTarget}' ` +
|
|
88
|
+
`(tried: ${editPatterns.map((p) => `Edit(${p})`).join(", ")})`,
|
|
89
|
+
);
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
// Safeguard 2: the branch must not be main and HEAD must not be detached.
|
|
93
|
+
const git = subprocess.runSync("git", ["rev-parse", "--abbrev-ref", "HEAD"], {
|
|
94
|
+
cwd,
|
|
95
|
+
});
|
|
96
|
+
if (git.exitCode !== 0) {
|
|
97
|
+
throw new SelfeditError(
|
|
98
|
+
"failed to read the current git branch (this may not be a git repository)",
|
|
99
|
+
);
|
|
100
|
+
}
|
|
101
|
+
const branch = git.stdout.trim();
|
|
102
|
+
|
|
103
|
+
if (branch === "HEAD") {
|
|
104
|
+
throw new SelfeditError(
|
|
105
|
+
"will not write while HEAD is detached. Check out a non-main branch first",
|
|
106
|
+
);
|
|
107
|
+
}
|
|
108
|
+
if (branch === "main") {
|
|
109
|
+
throw new SelfeditError(
|
|
110
|
+
"will not write while on branch 'main'. Switch to a feature branch",
|
|
111
|
+
);
|
|
112
|
+
}
|
|
113
|
+
|
|
114
|
+
const parent = dirname(absoluteTarget);
|
|
115
|
+
if (!fsSync.existsSync(parent)) {
|
|
116
|
+
throw new SelfeditError(
|
|
117
|
+
`parent directory '${relative(projectRoot, parent)}' does not exist`,
|
|
118
|
+
);
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
fsSync.writeFileSync(absoluteTarget, content);
|
|
122
|
+
|
|
123
|
+
return { bytes: content.length, relativeTarget, matchedPattern, branch };
|
|
124
|
+
}
|
|
@@ -23,11 +23,12 @@ export async function parseSuperviseOptions(values, runtime) {
|
|
|
23
23
|
const supervisorAllowedToolsRaw = values["supervisor-allowed-tools"];
|
|
24
24
|
|
|
25
25
|
// `||` (not `??`) throughout so an empty-string flag from a CI forwarder
|
|
26
|
-
// falls back to the default
|
|
26
|
+
// falls back to the default. The empty string does not override the
|
|
27
|
+
// default.
|
|
27
28
|
const tmpRoot = runtime.proc.env.TMPDIR ?? "/tmp";
|
|
28
29
|
const agentCwd = resolve(
|
|
29
30
|
values["agent-cwd"] ||
|
|
30
|
-
(await runtime.fs.mkdtemp(join(tmpRoot, "
|
|
31
|
+
(await runtime.fs.mkdtemp(join(tmpRoot, "gemba-harness-agent-"))),
|
|
31
32
|
);
|
|
32
33
|
|
|
33
34
|
return {
|
|
@@ -58,11 +59,11 @@ export async function parseSuperviseOptions(values, runtime) {
|
|
|
58
59
|
}
|
|
59
60
|
|
|
60
61
|
/**
|
|
61
|
-
* Supervise command — run one agent under a supervisor
|
|
62
|
-
* orchestration loop. The supervisor delegates work through Ask
|
|
63
|
-
* each reply on its next turn
|
|
62
|
+
* Supervise command — run one agent under a supervisor through the
|
|
63
|
+
* orchestration loop. The supervisor delegates work through Ask. It sees
|
|
64
|
+
* each reply on its next turn. It ends with Conclude.
|
|
64
65
|
*
|
|
65
|
-
* Usage:
|
|
66
|
+
* Usage: gemba-harness supervise [options]
|
|
66
67
|
*
|
|
67
68
|
* @param {import("@forwardimpact/libcli").InvocationContext} ctx
|
|
68
69
|
* @returns {Promise<{ok: boolean, code?: number, error?: string}>}
|
|
@@ -71,12 +72,12 @@ export async function runSuperviseCommand(ctx) {
|
|
|
71
72
|
const runtime = ctx.deps.runtime;
|
|
72
73
|
const opts = await parseSuperviseOptions(ctx.options, runtime);
|
|
73
74
|
|
|
74
|
-
// Build the redactor as the first observable side-effect after
|
|
75
|
-
//
|
|
76
|
-
// env
|
|
75
|
+
// Build the redactor as the first observable side-effect after the parser
|
|
76
|
+
// reads the options. The env snapshot must freeze BEFORE any in-process
|
|
77
|
+
// env write the command performs (e.g. LIBHARNESS_AGENT_PROFILE).
|
|
77
78
|
const redactor = createRedactor({ runtime });
|
|
78
79
|
|
|
79
|
-
//
|
|
80
|
+
// With --output, stream text to stdout and write NDJSON to the file.
|
|
80
81
|
// Otherwise, write NDJSON directly to stdout (backwards-compatible).
|
|
81
82
|
const fileStream = opts.outputPath
|
|
82
83
|
? runtime.fs.createWriteStream(opts.outputPath)
|
|
@@ -107,7 +108,8 @@ export async function runSuperviseCommand(ctx) {
|
|
|
107
108
|
runtime.proc.env.LIBHARNESS_AGENT_PROFILE = opts.agentProfile;
|
|
108
109
|
}
|
|
109
110
|
// Unconditional so the default "github" is observable to the agent's
|
|
110
|
-
// active-tracker resolution
|
|
111
|
+
// active-tracker resolution. This mirrors --agent-profile's env write
|
|
112
|
+
// above.
|
|
111
113
|
runtime.proc.env.LIBHARNESS_WORK_TRACKER = opts.workTracker;
|
|
112
114
|
|
|
113
115
|
const { query } = await import("@anthropic-ai/claude-agent-sdk");
|
|
@@ -1,18 +1,18 @@
|
|
|
1
1
|
import { composeTaskFromGitHubEvent } from "../events/github.js";
|
|
2
2
|
|
|
3
3
|
/**
|
|
4
|
-
* Resolve `--task-file` / `--task-text` / `--task-event` into the task pair
|
|
5
|
-
* runner consumes. Exactly one of the three must be set. For
|
|
6
|
-
* libharness reads the event payload
|
|
7
|
-
* template that matches `$GITHUB_EVENT_NAME` +
|
|
8
|
-
* amendment (from `payload.inputs?.prompt`)
|
|
9
|
-
* wire `--task-amend` separately. For the other two
|
|
10
|
-
* works as before.
|
|
4
|
+
* Resolve `--task-file` / `--task-text` / `--task-event` into the task pair
|
|
5
|
+
* the runner consumes. Exactly one of the three must be set. For
|
|
6
|
+
* `--task-event`, libharness reads the event payload. It extracts both the
|
|
7
|
+
* main task (from the template that matches `$GITHUB_EVENT_NAME` +
|
|
8
|
+
* `payload.action`) and the amendment (from `payload.inputs?.prompt`). So the
|
|
9
|
+
* workflow does not need to wire `--task-amend` separately. For the other two
|
|
10
|
+
* modes, `--task-amend` works as before.
|
|
11
11
|
*
|
|
12
12
|
* @param {object} values - Parsed option values from cli.parse()
|
|
13
13
|
* @param {import("@forwardimpact/libutil/runtime").Runtime} runtime - Ambient
|
|
14
|
-
* collaborators
|
|
15
|
-
*
|
|
14
|
+
* collaborators. `fsSync.readFileSync` loads `--task-file`/`--task-event`.
|
|
15
|
+
* `proc.env` resolves `GITHUB_EVENT_NAME`.
|
|
16
16
|
* @returns {{ task: string, amend: string | undefined }}
|
|
17
17
|
*/
|
|
18
18
|
export function resolveTaskContent(values, runtime) {
|
package/src/commands/tee.js
CHANGED
|
@@ -4,12 +4,13 @@ import { isoTimestamp } from "@forwardimpact/libutil";
|
|
|
4
4
|
import { createTeeWriter } from "../tee-writer.js";
|
|
5
5
|
|
|
6
6
|
/**
|
|
7
|
-
* Tee command — stream text output to stdout
|
|
8
|
-
*
|
|
9
|
-
* re-delimits each record with a newline
|
|
10
|
-
*
|
|
7
|
+
* Tee command — stream text output to stdout. Save the raw NDJSON to a file
|
|
8
|
+
* when the caller gives an output path. The command reads stdin line by line
|
|
9
|
+
* through the injected runtime. It re-delimits each record with a newline.
|
|
10
|
+
* The TeeWriter's line splitter then sees the same record boundaries that the
|
|
11
|
+
* raw byte stream produced.
|
|
11
12
|
*
|
|
12
|
-
* Usage:
|
|
13
|
+
* Usage: gemba-harness tee [output.ndjson] < trace.ndjson
|
|
13
14
|
*
|
|
14
15
|
* @param {import("@forwardimpact/libcli").InvocationContext} ctx
|
|
15
16
|
* @returns {Promise<{ok: boolean, code?: number, error?: string}>}
|
|
@@ -21,8 +22,8 @@ export async function runTeeCommand(ctx) {
|
|
|
21
22
|
? runtime.fs.createWriteStream(outputPath)
|
|
22
23
|
: null;
|
|
23
24
|
|
|
24
|
-
// TeeWriter requires a fileStream
|
|
25
|
-
// use a PassThrough as a no-op sink
|
|
25
|
+
// TeeWriter requires a fileStream. When the caller gives no output file,
|
|
26
|
+
// use a PassThrough as a no-op sink. The command then saves no NDJSON.
|
|
26
27
|
const sink = fileStream ?? new PassThrough();
|
|
27
28
|
const tee = createTeeWriter({
|
|
28
29
|
fileStream: sink,
|
|
@@ -32,9 +33,9 @@ export async function runTeeCommand(ctx) {
|
|
|
32
33
|
});
|
|
33
34
|
|
|
34
35
|
try {
|
|
35
|
-
// `runtime.proc.stdin` yields newline-stripped lines
|
|
36
|
-
// TeeWriter's `_write` line splitter frames records exactly as it did
|
|
37
|
-
// piped the raw byte stream.
|
|
36
|
+
// `runtime.proc.stdin` yields newline-stripped lines. Re-append `\n` so
|
|
37
|
+
// the TeeWriter's `_write` line splitter frames records exactly as it did
|
|
38
|
+
// when the caller piped the raw byte stream into it.
|
|
38
39
|
const lines = (async function* () {
|
|
39
40
|
for await (const line of runtime.proc.stdin) yield `${line}\n`;
|
|
40
41
|
})();
|