@forwardimpact/libharness 3.0.0 → 3.0.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +60 -57
- package/package.json +2 -2
- package/src/advisor.js +47 -41
- package/src/agent-runner.js +57 -47
- package/src/benchmark/apm-installer.js +28 -28
- package/src/benchmark/env-loader.js +24 -16
- package/src/benchmark/grade.js +44 -41
- package/src/benchmark/hidden-tests.js +25 -24
- package/src/benchmark/hook-env.js +11 -9
- package/src/benchmark/invariants.js +20 -17
- package/src/benchmark/judge.js +29 -28
- package/src/benchmark/npm-installer.js +9 -8
- package/src/benchmark/report.js +53 -50
- package/src/benchmark/result.js +24 -23
- package/src/benchmark/runner.js +75 -69
- package/src/benchmark/scheduler.js +17 -16
- package/src/benchmark/task-family.js +28 -26
- package/src/benchmark/trace-split.js +8 -7
- package/src/benchmark/workdir.js +27 -25
- package/src/claude-code-executable.js +11 -11
- package/src/commands/advisor-flags.js +8 -7
- package/src/commands/assert.js +16 -15
- package/src/commands/benchmark-definition.js +11 -11
- package/src/commands/benchmark-grade.js +12 -11
- package/src/commands/benchmark-report.js +5 -5
- package/src/commands/benchmark-run.js +31 -28
- package/src/commands/by-discussion.js +10 -10
- package/src/commands/callback.js +11 -11
- package/src/commands/discuss.js +8 -7
- package/src/commands/facilitate.js +15 -13
- package/src/commands/output.js +3 -2
- package/src/commands/run.js +14 -14
- package/src/commands/scan-logs.js +21 -19
- package/src/commands/selfedit.js +14 -14
- package/src/commands/supervise.js +11 -9
- package/src/commands/task-input.js +9 -9
- package/src/commands/tee.js +10 -9
- package/src/commands/trace.js +55 -42
- package/src/commands/work-tracker.js +4 -3
- package/src/cost.js +17 -17
- package/src/discuss-tools.js +16 -16
- package/src/discusser.js +39 -38
- package/src/events/github.js +54 -37
- package/src/facilitator.js +21 -21
- package/src/inbox-poller.js +4 -4
- package/src/judge.js +32 -30
- package/src/message-bus.js +12 -11
- package/src/orchestration-loop.js +35 -36
- package/src/orchestration-toolkit.js +58 -53
- package/src/orchestrator-helpers.js +2 -2
- package/src/profile-prompt.js +54 -53
- package/src/redaction.js +63 -57
- package/src/render/line-renderer.js +5 -5
- package/src/render/orchestrator-filter.js +3 -3
- package/src/render/palette.js +11 -9
- package/src/render/tool-hints.js +18 -15
- package/src/render/turn-renderer.js +4 -4
- package/src/reply-emitter.js +2 -2
- package/src/sequence-counter.js +4 -3
- package/src/signature-filter.js +7 -6
- package/src/supervisor.js +19 -18
- package/src/tee-writer.js +25 -25
- package/src/trace-collector.js +53 -48
- package/src/trace-github.js +53 -44
- package/src/trace-multi.js +15 -13
- package/src/trace-query.js +61 -52
- package/src/trace-render.js +18 -18
- package/src/trace-usage.js +31 -28
- package/src/transcript-recorder.js +24 -20
package/src/commands/run.js
CHANGED
|
@@ -34,8 +34,8 @@ export function parseRunOptions(values, runtime) {
|
|
|
34
34
|
values,
|
|
35
35
|
runtime,
|
|
36
36
|
);
|
|
37
|
-
// `||` (not `??`) so an empty-string flag from a CI forwarder falls back
|
|
38
|
-
// the default
|
|
37
|
+
// `||` (not `??`) so an empty-string flag from a CI forwarder falls back
|
|
38
|
+
// to the default. The empty string does not override the default.
|
|
39
39
|
const maxTurnsRaw = values["max-turns"] || "50";
|
|
40
40
|
|
|
41
41
|
return {
|
|
@@ -64,10 +64,10 @@ const devNull = new Writable({
|
|
|
64
64
|
|
|
65
65
|
/**
|
|
66
66
|
* Wire the run-mode agent session: external MCP entry, `LIBHARNESS_*` env
|
|
67
|
-
* writes, system-prompt composition
|
|
68
|
-
* the advisor
|
|
69
|
-
* server
|
|
70
|
-
* so
|
|
67
|
+
* writes, and system-prompt composition. When an advisor model is set, also
|
|
68
|
+
* wire the advisor (budget, recorder, advisor session, and a dedicated MCP
|
|
69
|
+
* server that holds only the `Advisor` tool). This function is extracted
|
|
70
|
+
* from `runRunCommand` so a test can inject a fake `query`.
|
|
71
71
|
*
|
|
72
72
|
* Run mode has no stop path (the command simply awaits the runner), so the
|
|
73
73
|
* consult timeout is deliberately the advisor's only guard.
|
|
@@ -120,9 +120,9 @@ export async function wireRunSession({
|
|
|
120
120
|
runtime.proc.env.LIBHARNESS_WORK_TRACKER = opts.workTracker;
|
|
121
121
|
|
|
122
122
|
// With a profile, the consult guidance rides the profile composer's
|
|
123
|
-
// amendment parameter
|
|
123
|
+
// amendment parameter. With no profile, a preset-append prompt carries
|
|
124
124
|
// the guidance as its only session-protocol fragment. Advisor off and no
|
|
125
|
-
// profile means no system prompt
|
|
125
|
+
// profile means no system prompt. That is today's behavior, unchanged.
|
|
126
126
|
let systemPrompt;
|
|
127
127
|
if (opts.agentProfile) {
|
|
128
128
|
systemPrompt = composeProfilePrompt(opts.agentProfile, {
|
|
@@ -161,7 +161,7 @@ export async function wireRunSession({
|
|
|
161
161
|
budget,
|
|
162
162
|
model: opts.advisorModel,
|
|
163
163
|
});
|
|
164
|
-
// No allowlist push
|
|
164
|
+
// No allowlist push. In-process SDK MCP servers work under
|
|
165
165
|
// bypassPermissions without allowlist entries (loop-mode precedent).
|
|
166
166
|
mcpServers = {
|
|
167
167
|
...mcpServers,
|
|
@@ -195,7 +195,7 @@ export async function wireRunSession({
|
|
|
195
195
|
}
|
|
196
196
|
|
|
197
197
|
/**
|
|
198
|
-
* Run command — execute a single agent
|
|
198
|
+
* Run command — execute a single agent through the Claude Agent SDK.
|
|
199
199
|
*
|
|
200
200
|
* Usage: gemba-harness run [options]
|
|
201
201
|
*
|
|
@@ -206,12 +206,12 @@ export async function runRunCommand(ctx) {
|
|
|
206
206
|
const runtime = ctx.deps.runtime;
|
|
207
207
|
const opts = parseRunOptions(ctx.options, runtime);
|
|
208
208
|
|
|
209
|
-
// Build the redactor as the first observable side-effect after
|
|
210
|
-
//
|
|
211
|
-
// env
|
|
209
|
+
// Build the redactor as the first observable side-effect after the parser
|
|
210
|
+
// reads the options. The env snapshot must freeze BEFORE any in-process
|
|
211
|
+
// env write the command performs (e.g. LIBHARNESS_AGENT_PROFILE).
|
|
212
212
|
const redactor = createRedactor({ runtime });
|
|
213
213
|
|
|
214
|
-
//
|
|
214
|
+
// With --output, stream text to stdout and write NDJSON to the file.
|
|
215
215
|
// Otherwise, write NDJSON directly to stdout (backwards-compatible).
|
|
216
216
|
const fileStream = opts.outputPath
|
|
217
217
|
? runtime.fs.createWriteStream(opts.outputPath)
|
|
@@ -2,27 +2,28 @@
|
|
|
2
2
|
* `gemba-harness scan-logs` — scan a run's log archive for secret literals and
|
|
3
3
|
* fail closed.
|
|
4
4
|
*
|
|
5
|
-
*
|
|
6
|
-
* in `gemba-trace
|
|
7
|
-
* run's own log archive
|
|
8
|
-
* into it. Any hit exits non-zero
|
|
9
|
-
* non-zero
|
|
5
|
+
* This is a run-lifecycle concern. It is not an NDJSON trace, so it lives
|
|
6
|
+
* here rather than in `gemba-trace`. After a CI run that handled secrets,
|
|
7
|
+
* download or accept the run's own log archive. Then assert that none of a
|
|
8
|
+
* supplied set of literals leaked into it. Any hit exits non-zero. Any
|
|
9
|
+
* download or extract failure also exits non-zero. The gate must never
|
|
10
|
+
* silently disarm.
|
|
10
11
|
*
|
|
11
12
|
* Log resolution:
|
|
12
13
|
* - `--archive <zip>` — an already-resolved archive (extracted locally).
|
|
13
|
-
* - `--run-id <id> --repo <owner/repo>` — download this run's archive
|
|
14
|
+
* - `--run-id <id> --repo <owner/repo>` — download this run's archive with
|
|
14
15
|
* `gh` first, then extract.
|
|
15
16
|
*
|
|
16
17
|
* Secrets are `--secret <label>=<literal>`, repeatable. The literal is
|
|
17
|
-
* everything after the FIRST `=` (JWTs and base64 keys contain `=`)
|
|
18
|
-
* is only cosmetic
|
|
18
|
+
* everything after the FIRST `=` (JWTs and base64 keys contain `=`). The
|
|
19
|
+
* label is only cosmetic. The `FAIL:` line names it.
|
|
19
20
|
*/
|
|
20
21
|
|
|
21
22
|
import { join } from "node:path";
|
|
22
23
|
|
|
23
24
|
/**
|
|
24
25
|
* Parse repeatable `--secret label=literal` flags. libcli's `multiple: true`
|
|
25
|
-
* yields an array from node's parseArgs in every case
|
|
26
|
+
* yields an array from node's parseArgs in every case. Tolerate a bare string
|
|
26
27
|
* or undefined defensively. Split on the FIRST `=` only.
|
|
27
28
|
*
|
|
28
29
|
* @param {string[]|string|undefined} secretOpt
|
|
@@ -42,8 +43,9 @@ export function parseSecrets(secretOpt) {
|
|
|
42
43
|
}
|
|
43
44
|
|
|
44
45
|
/**
|
|
45
|
-
* Walk a directory tree and return every file path.
|
|
46
|
-
* it works against both node:fs and the libmock fs (no `recursive`
|
|
46
|
+
* Walk a directory tree and return every file path. It uses per-level readdir
|
|
47
|
+
* so it works against both node:fs and the libmock fs (no `recursive`
|
|
48
|
+
* reliance).
|
|
47
49
|
*/
|
|
48
50
|
async function collectFiles(dir, runtime) {
|
|
49
51
|
const out = [];
|
|
@@ -60,9 +62,9 @@ async function collectFiles(dir, runtime) {
|
|
|
60
62
|
}
|
|
61
63
|
|
|
62
64
|
/**
|
|
63
|
-
* Scan every file under `dir` for each secret literal.
|
|
64
|
-
* secrets whose non-empty literal appears in any file
|
|
65
|
-
*
|
|
65
|
+
* Scan every file under `dir` for each secret literal. Return the labels of
|
|
66
|
+
* secrets whose non-empty literal appears in any file. The scan skips empty
|
|
67
|
+
* literals, because a secret the run never set cannot leak.
|
|
66
68
|
*
|
|
67
69
|
* @param {object} params
|
|
68
70
|
* @param {string} params.dir
|
|
@@ -84,9 +86,9 @@ export async function scanDirectory({ dir, secrets, runtime }) {
|
|
|
84
86
|
}
|
|
85
87
|
|
|
86
88
|
/**
|
|
87
|
-
* Resolve a directory of extracted log files
|
|
88
|
-
*
|
|
89
|
-
* failure
|
|
89
|
+
* Resolve a directory of extracted log files. Download the archive first when
|
|
90
|
+
* the caller gives a run id. Throw (→ fail closed) on any download or extract
|
|
91
|
+
* failure, and on a missing or invalid input.
|
|
90
92
|
*/
|
|
91
93
|
async function resolveLogsDir({ options, runtime }) {
|
|
92
94
|
const tmpRoot = runtime.proc.env.RUNNER_TEMP || "/tmp";
|
|
@@ -142,8 +144,8 @@ export async function runScanLogsCommand(ctx) {
|
|
|
142
144
|
try {
|
|
143
145
|
dir = await resolveLogsDir({ options, runtime });
|
|
144
146
|
} catch (err) {
|
|
145
|
-
// Fail closed
|
|
146
|
-
// dispatcher prints the returned `error`, so
|
|
147
|
+
// Fail closed. An unresolvable archive must not pass as "no leak". The
|
|
148
|
+
// dispatcher prints the returned `error`, so do not also write it here.
|
|
147
149
|
return { ok: false, code: 1, error: `scan-logs: ${err.message}` };
|
|
148
150
|
}
|
|
149
151
|
|
package/src/commands/selfedit.js
CHANGED
|
@@ -1,15 +1,15 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* Safeguard-and-write logic behind the `gemba-selfedit` bin
|
|
3
|
-
* to a path that .claude/settings.json permits Edit on
|
|
4
|
-
*
|
|
5
|
-
* full rationale.
|
|
2
|
+
* Safeguard-and-write logic behind the `gemba-selfedit` bin. It writes
|
|
3
|
+
* content to a path that .claude/settings.json permits Edit on. The write
|
|
4
|
+
* happens only on a non-main git branch. See
|
|
5
|
+
* libraries/libharness/README.md § gemba-selfedit for the full rationale.
|
|
6
6
|
*/
|
|
7
7
|
|
|
8
8
|
import { resolve, relative, dirname } from "node:path";
|
|
9
9
|
|
|
10
10
|
import { minimatch } from "minimatch";
|
|
11
11
|
|
|
12
|
-
/** A safeguard violation
|
|
12
|
+
/** A safeguard violation. Callers map it to exit code 2. */
|
|
13
13
|
export class SelfeditError extends Error {
|
|
14
14
|
/** @param {string} message failure description */
|
|
15
15
|
constructor(message) {
|
|
@@ -21,18 +21,18 @@ export class SelfeditError extends Error {
|
|
|
21
21
|
/**
|
|
22
22
|
* Check every safeguard for a selfedit write, then perform it.
|
|
23
23
|
*
|
|
24
|
-
*
|
|
24
|
+
* The function checks these safeguards in order:
|
|
25
25
|
* 1. The nearest .claude/settings.json must contain an Edit(<glob>) rule in
|
|
26
26
|
* permissions.allow[] that resolves to the target path.
|
|
27
27
|
* 2. HEAD must not be detached and the current branch must not be 'main'.
|
|
28
28
|
* 3. The target's parent directory must exist.
|
|
29
29
|
*
|
|
30
|
-
* @param {string} targetArg target path
|
|
30
|
+
* @param {string} targetArg target path from the command line
|
|
31
31
|
* @param {Buffer} content bytes to write
|
|
32
32
|
* @param {{ runtime: object }} deps runtime bag (fsSync, proc, subprocess,
|
|
33
|
-
* finder)
|
|
33
|
+
* finder). targetArg resolves against `runtime.proc.cwd()`
|
|
34
34
|
* @returns {{ bytes: number, relativeTarget: string, matchedPattern: string,
|
|
35
|
-
* branch: string }} what
|
|
35
|
+
* branch: string }} what the function wrote and which rule allowed it
|
|
36
36
|
* @throws {SelfeditError} on any safeguard violation
|
|
37
37
|
*/
|
|
38
38
|
export function runSelfeditCommand(targetArg, content, { runtime }) {
|
|
@@ -41,7 +41,7 @@ export function runSelfeditCommand(targetArg, content, { runtime }) {
|
|
|
41
41
|
const absoluteTarget = resolve(cwd, targetArg);
|
|
42
42
|
|
|
43
43
|
// Safeguard 1: settings.json must grant Edit() on this path. Resolve the
|
|
44
|
-
// finder off the runtime bag
|
|
44
|
+
// finder off the runtime bag. Do not construct a Finder here.
|
|
45
45
|
const settingsPath = finder.findUpward(
|
|
46
46
|
dirname(absoluteTarget),
|
|
47
47
|
".claude/settings.json",
|
|
@@ -89,25 +89,25 @@ export function runSelfeditCommand(targetArg, content, { runtime }) {
|
|
|
89
89
|
);
|
|
90
90
|
}
|
|
91
91
|
|
|
92
|
-
// Safeguard 2: branch must not be main and HEAD must not be detached.
|
|
92
|
+
// Safeguard 2: the branch must not be main and HEAD must not be detached.
|
|
93
93
|
const git = subprocess.runSync("git", ["rev-parse", "--abbrev-ref", "HEAD"], {
|
|
94
94
|
cwd,
|
|
95
95
|
});
|
|
96
96
|
if (git.exitCode !== 0) {
|
|
97
97
|
throw new SelfeditError(
|
|
98
|
-
"failed to read current git branch (not
|
|
98
|
+
"failed to read the current git branch (this may not be a git repository)",
|
|
99
99
|
);
|
|
100
100
|
}
|
|
101
101
|
const branch = git.stdout.trim();
|
|
102
102
|
|
|
103
103
|
if (branch === "HEAD") {
|
|
104
104
|
throw new SelfeditError(
|
|
105
|
-
"HEAD is detached
|
|
105
|
+
"will not write while HEAD is detached. Check out a non-main branch first",
|
|
106
106
|
);
|
|
107
107
|
}
|
|
108
108
|
if (branch === "main") {
|
|
109
109
|
throw new SelfeditError(
|
|
110
|
-
"
|
|
110
|
+
"will not write while on branch 'main'. Switch to a feature branch",
|
|
111
111
|
);
|
|
112
112
|
}
|
|
113
113
|
|
|
@@ -23,7 +23,8 @@ export async function parseSuperviseOptions(values, runtime) {
|
|
|
23
23
|
const supervisorAllowedToolsRaw = values["supervisor-allowed-tools"];
|
|
24
24
|
|
|
25
25
|
// `||` (not `??`) throughout so an empty-string flag from a CI forwarder
|
|
26
|
-
// falls back to the default
|
|
26
|
+
// falls back to the default. The empty string does not override the
|
|
27
|
+
// default.
|
|
27
28
|
const tmpRoot = runtime.proc.env.TMPDIR ?? "/tmp";
|
|
28
29
|
const agentCwd = resolve(
|
|
29
30
|
values["agent-cwd"] ||
|
|
@@ -58,9 +59,9 @@ export async function parseSuperviseOptions(values, runtime) {
|
|
|
58
59
|
}
|
|
59
60
|
|
|
60
61
|
/**
|
|
61
|
-
* Supervise command — run one agent under a supervisor
|
|
62
|
-
* orchestration loop. The supervisor delegates work through Ask
|
|
63
|
-
* each reply on its next turn
|
|
62
|
+
* Supervise command — run one agent under a supervisor through the
|
|
63
|
+
* orchestration loop. The supervisor delegates work through Ask. It sees
|
|
64
|
+
* each reply on its next turn. It ends with Conclude.
|
|
64
65
|
*
|
|
65
66
|
* Usage: gemba-harness supervise [options]
|
|
66
67
|
*
|
|
@@ -71,12 +72,12 @@ export async function runSuperviseCommand(ctx) {
|
|
|
71
72
|
const runtime = ctx.deps.runtime;
|
|
72
73
|
const opts = await parseSuperviseOptions(ctx.options, runtime);
|
|
73
74
|
|
|
74
|
-
// Build the redactor as the first observable side-effect after
|
|
75
|
-
//
|
|
76
|
-
// env
|
|
75
|
+
// Build the redactor as the first observable side-effect after the parser
|
|
76
|
+
// reads the options. The env snapshot must freeze BEFORE any in-process
|
|
77
|
+
// env write the command performs (e.g. LIBHARNESS_AGENT_PROFILE).
|
|
77
78
|
const redactor = createRedactor({ runtime });
|
|
78
79
|
|
|
79
|
-
//
|
|
80
|
+
// With --output, stream text to stdout and write NDJSON to the file.
|
|
80
81
|
// Otherwise, write NDJSON directly to stdout (backwards-compatible).
|
|
81
82
|
const fileStream = opts.outputPath
|
|
82
83
|
? runtime.fs.createWriteStream(opts.outputPath)
|
|
@@ -107,7 +108,8 @@ export async function runSuperviseCommand(ctx) {
|
|
|
107
108
|
runtime.proc.env.LIBHARNESS_AGENT_PROFILE = opts.agentProfile;
|
|
108
109
|
}
|
|
109
110
|
// Unconditional so the default "github" is observable to the agent's
|
|
110
|
-
// active-tracker resolution
|
|
111
|
+
// active-tracker resolution. This mirrors --agent-profile's env write
|
|
112
|
+
// above.
|
|
111
113
|
runtime.proc.env.LIBHARNESS_WORK_TRACKER = opts.workTracker;
|
|
112
114
|
|
|
113
115
|
const { query } = await import("@anthropic-ai/claude-agent-sdk");
|
|
@@ -1,18 +1,18 @@
|
|
|
1
1
|
import { composeTaskFromGitHubEvent } from "../events/github.js";
|
|
2
2
|
|
|
3
3
|
/**
|
|
4
|
-
* Resolve `--task-file` / `--task-text` / `--task-event` into the task pair
|
|
5
|
-
* runner consumes. Exactly one of the three must be set. For
|
|
6
|
-
* libharness reads the event payload
|
|
7
|
-
* template that matches `$GITHUB_EVENT_NAME` +
|
|
8
|
-
* amendment (from `payload.inputs?.prompt`)
|
|
9
|
-
* wire `--task-amend` separately. For the other two
|
|
10
|
-
* works as before.
|
|
4
|
+
* Resolve `--task-file` / `--task-text` / `--task-event` into the task pair
|
|
5
|
+
* the runner consumes. Exactly one of the three must be set. For
|
|
6
|
+
* `--task-event`, libharness reads the event payload. It extracts both the
|
|
7
|
+
* main task (from the template that matches `$GITHUB_EVENT_NAME` +
|
|
8
|
+
* `payload.action`) and the amendment (from `payload.inputs?.prompt`). So the
|
|
9
|
+
* workflow does not need to wire `--task-amend` separately. For the other two
|
|
10
|
+
* modes, `--task-amend` works as before.
|
|
11
11
|
*
|
|
12
12
|
* @param {object} values - Parsed option values from cli.parse()
|
|
13
13
|
* @param {import("@forwardimpact/libutil/runtime").Runtime} runtime - Ambient
|
|
14
|
-
* collaborators
|
|
15
|
-
*
|
|
14
|
+
* collaborators. `fsSync.readFileSync` loads `--task-file`/`--task-event`.
|
|
15
|
+
* `proc.env` resolves `GITHUB_EVENT_NAME`.
|
|
16
16
|
* @returns {{ task: string, amend: string | undefined }}
|
|
17
17
|
*/
|
|
18
18
|
export function resolveTaskContent(values, runtime) {
|
package/src/commands/tee.js
CHANGED
|
@@ -4,10 +4,11 @@ import { isoTimestamp } from "@forwardimpact/libutil";
|
|
|
4
4
|
import { createTeeWriter } from "../tee-writer.js";
|
|
5
5
|
|
|
6
6
|
/**
|
|
7
|
-
* Tee command — stream text output to stdout
|
|
8
|
-
*
|
|
9
|
-
* re-delimits each record with a newline
|
|
10
|
-
*
|
|
7
|
+
* Tee command — stream text output to stdout. Save the raw NDJSON to a file
|
|
8
|
+
* when the caller gives an output path. The command reads stdin line by line
|
|
9
|
+
* through the injected runtime. It re-delimits each record with a newline.
|
|
10
|
+
* The TeeWriter's line splitter then sees the same record boundaries that the
|
|
11
|
+
* raw byte stream produced.
|
|
11
12
|
*
|
|
12
13
|
* Usage: gemba-harness tee [output.ndjson] < trace.ndjson
|
|
13
14
|
*
|
|
@@ -21,8 +22,8 @@ export async function runTeeCommand(ctx) {
|
|
|
21
22
|
? runtime.fs.createWriteStream(outputPath)
|
|
22
23
|
: null;
|
|
23
24
|
|
|
24
|
-
// TeeWriter requires a fileStream
|
|
25
|
-
// use a PassThrough as a no-op sink
|
|
25
|
+
// TeeWriter requires a fileStream. When the caller gives no output file,
|
|
26
|
+
// use a PassThrough as a no-op sink. The command then saves no NDJSON.
|
|
26
27
|
const sink = fileStream ?? new PassThrough();
|
|
27
28
|
const tee = createTeeWriter({
|
|
28
29
|
fileStream: sink,
|
|
@@ -32,9 +33,9 @@ export async function runTeeCommand(ctx) {
|
|
|
32
33
|
});
|
|
33
34
|
|
|
34
35
|
try {
|
|
35
|
-
// `runtime.proc.stdin` yields newline-stripped lines
|
|
36
|
-
// TeeWriter's `_write` line splitter frames records exactly as it did
|
|
37
|
-
// piped the raw byte stream.
|
|
36
|
+
// `runtime.proc.stdin` yields newline-stripped lines. Re-append `\n` so
|
|
37
|
+
// the TeeWriter's `_write` line splitter frames records exactly as it did
|
|
38
|
+
// when the caller piped the raw byte stream into it.
|
|
38
39
|
const lines = (async function* () {
|
|
39
40
|
for await (const line of runtime.proc.stdin) yield `${line}\n`;
|
|
40
41
|
})();
|
package/src/commands/trace.js
CHANGED
|
@@ -20,18 +20,18 @@ import {
|
|
|
20
20
|
// ctx.options — parsed flag values (`cli.parse().values`)
|
|
21
21
|
// ctx.args — named positionals declared on the subcommand
|
|
22
22
|
// ctx.deps — host-injected collaborators: `{ runtime, config }`
|
|
23
|
-
// Handlers read
|
|
24
|
-
// `ctx.deps.runtime
|
|
23
|
+
// Handlers read and write the filesystem and stdout only through
|
|
24
|
+
// `ctx.deps.runtime`. They return `{ ok: true }` on success.
|
|
25
25
|
|
|
26
|
-
/**
|
|
26
|
+
/** These characters mark a `--file` value as a glob. */
|
|
27
27
|
const GLOB_CHARS = /[*?[\]{}]/;
|
|
28
28
|
|
|
29
29
|
/**
|
|
30
30
|
* Resolve the cross-trace `--file` option (`ctx.options.file`) into a sorted
|
|
31
|
-
* flat list of file paths. A literal path passes through
|
|
32
|
-
* glob metacharacters expands
|
|
33
|
-
* fast path means the common single-file and shell-pre-expanded
|
|
34
|
-
* touch `globSync`.
|
|
31
|
+
* flat list of file paths. A literal path passes through. A value that
|
|
32
|
+
* carries glob metacharacters expands through `runtime.fsSync.globSync`. The
|
|
33
|
+
* literal-path fast path means the common single-file and shell-pre-expanded
|
|
34
|
+
* cases never touch `globSync`.
|
|
35
35
|
* @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
|
|
36
36
|
* @param {import("@forwardimpact/libcli").InvocationContext} ctx
|
|
37
37
|
* @returns {string[]}
|
|
@@ -51,10 +51,11 @@ function resolveFiles(runtime, ctx) {
|
|
|
51
51
|
}
|
|
52
52
|
|
|
53
53
|
/**
|
|
54
|
-
* Emit a query result for a cross-trace verb
|
|
55
|
-
* JSON payload
|
|
56
|
-
* deep-equals today's output
|
|
57
|
-
*
|
|
54
|
+
* Emit a query result for a cross-trace verb. Under `--format json` this
|
|
55
|
+
* function writes the JSON payload. Single-object verbs unwrap when there is
|
|
56
|
+
* one file, so the envelope deep-equals today's output. Otherwise this
|
|
57
|
+
* function renders text to stdout. The renderer owns source attribution.
|
|
58
|
+
* `multi` gates it.
|
|
58
59
|
* @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
|
|
59
60
|
* @param {object|object[]} result
|
|
60
61
|
* @param {Function} renderer
|
|
@@ -78,7 +79,7 @@ function emit(runtime, result, renderer, ctx, multi, unwrap = false) {
|
|
|
78
79
|
// --- GitHub commands ---
|
|
79
80
|
|
|
80
81
|
/**
|
|
81
|
-
* List recent workflow runs
|
|
82
|
+
* List recent workflow runs that match a pattern.
|
|
82
83
|
* @param {import("@forwardimpact/libcli").InvocationContext} ctx
|
|
83
84
|
*/
|
|
84
85
|
export async function runRunsCommand(ctx) {
|
|
@@ -332,8 +333,8 @@ export async function runStatsCommand(ctx) {
|
|
|
332
333
|
if (files.length === 0) return noFiles("stats");
|
|
333
334
|
const multi = files.length > 1;
|
|
334
335
|
const query = statsQuery(ctx);
|
|
335
|
-
// stats results are per-file objects
|
|
336
|
-
//
|
|
336
|
+
// stats results are per-file objects. Each file gets one block and no
|
|
337
|
+
// cross-file sum. Each block carries a source tag only when multi-file.
|
|
337
338
|
const results = files.map((file) => ({
|
|
338
339
|
result: query(loadTrace(runtime, file)),
|
|
339
340
|
source: multi ? basename(file) : undefined,
|
|
@@ -356,20 +357,22 @@ export async function runStatsCommand(ctx) {
|
|
|
356
357
|
}
|
|
357
358
|
|
|
358
359
|
/**
|
|
359
|
-
*
|
|
360
|
-
* named profile)
|
|
361
|
-
* per source. The combined trace from a
|
|
362
|
-
*
|
|
363
|
-
* run's spend.
|
|
364
|
-
* emits a GitHub-flavored
|
|
360
|
+
* Report the total run cost across every participant (agent, supervisor,
|
|
361
|
+
* judge, and any named profile). The command sums each `result` event in the
|
|
362
|
+
* trace. It attributes the cost per source. The combined trace from a
|
|
363
|
+
* supervised, facilitated, or discuss session already interleaves all
|
|
364
|
+
* participants, so one file yields the whole run's spend. The default output
|
|
365
|
+
* is `{totalCostUsd, bySource}` JSON. `--markdown` emits a GitHub-flavored
|
|
366
|
+
* block to redirect into `$GITHUB_STEP_SUMMARY`.
|
|
365
367
|
*
|
|
366
368
|
* @param {import("@forwardimpact/libcli").InvocationContext} ctx
|
|
367
369
|
*/
|
|
368
370
|
export async function runCostCommand(ctx) {
|
|
369
371
|
const { runtime } = ctx.deps;
|
|
370
|
-
// Tolerate a missing
|
|
371
|
-
// so the trace may not exist
|
|
372
|
-
// nothing and exit 0
|
|
372
|
+
// Tolerate a missing or empty trace. A CI step reports cost under
|
|
373
|
+
// `always()`, so the trace may not exist. The run can fail before it
|
|
374
|
+
// produces one. Print nothing and exit 0. Do not throw. The caller then
|
|
375
|
+
// needs no `if [ -f ]`.
|
|
373
376
|
const file = ctx.args.file;
|
|
374
377
|
if (!file || !runtime.fsSync.existsSync(file)) return { ok: true };
|
|
375
378
|
const cost = computeTraceCost(runtime.fsSync.readFileSync(file, "utf8"));
|
|
@@ -383,7 +386,8 @@ export async function runCostCommand(ctx) {
|
|
|
383
386
|
|
|
384
387
|
/**
|
|
385
388
|
* Render a cost summary as a GitHub-flavored markdown block for a CI step
|
|
386
|
-
* summary
|
|
389
|
+
* summary. The block holds a headline total and a per-participant table. The
|
|
390
|
+
* table lists the highest cost first.
|
|
387
391
|
* @param {{totalCostUsd: number, bySource: Record<string, number>}} cost
|
|
388
392
|
* @returns {string}
|
|
389
393
|
*/
|
|
@@ -391,7 +395,7 @@ function renderCostMarkdown(cost) {
|
|
|
391
395
|
const lines = [
|
|
392
396
|
`### 💰 Run cost: $${cost.totalCostUsd.toFixed(4)}`,
|
|
393
397
|
"",
|
|
394
|
-
"
|
|
398
|
+
"This total covers every participant (agent, supervisor, judge, named profiles).",
|
|
395
399
|
];
|
|
396
400
|
const sources = Object.entries(cost.bySource).sort((a, b) => b[1] - a[1]);
|
|
397
401
|
if (sources.length > 0) {
|
|
@@ -492,21 +496,27 @@ export async function runCompareCommand(ctx) {
|
|
|
492
496
|
|
|
493
497
|
// --- Split command ---
|
|
494
498
|
|
|
495
|
-
/**
|
|
499
|
+
/**
|
|
500
|
+
* A valid source name starts with a lowercase letter. The rest uses lowercase
|
|
501
|
+
* alphanumeric characters or hyphens.
|
|
502
|
+
*/
|
|
496
503
|
const VALID_SOURCE_NAME = /^[a-z][a-z0-9-]*$/;
|
|
497
504
|
|
|
498
|
-
/**
|
|
505
|
+
/**
|
|
506
|
+
* Sources whose name is itself a structural role. The splitter classifies
|
|
507
|
+
* each one into the role it represents.
|
|
508
|
+
*/
|
|
499
509
|
const STRUCTURAL_ROLES = new Set(["agent", "supervisor", "facilitator"]);
|
|
500
510
|
|
|
501
511
|
/**
|
|
502
|
-
* Split a combined NDJSON trace into per-source files
|
|
503
|
-
* `trace--<case>--<participant>.<role>.ndjson` convention.
|
|
512
|
+
* Split a combined NDJSON trace into per-source files. The output names
|
|
513
|
+
* follow the `trace--<case>--<participant>.<role>.ndjson` convention.
|
|
504
514
|
*
|
|
505
515
|
* Each valid envelope source becomes one output file. Structural sources
|
|
506
|
-
* (`agent`, `supervisor`, `facilitator`) classify into the matching role
|
|
507
|
-
* use their own name as participant
|
|
516
|
+
* (`agent`, `supervisor`, `facilitator`) classify into the matching role.
|
|
517
|
+
* They use their own name as participant. Profile-named sources (e.g.
|
|
508
518
|
* `staff-engineer`) classify as agents with the profile in the participant
|
|
509
|
-
* slot.
|
|
519
|
+
* slot. The command drops orchestrator events and invalid source names.
|
|
510
520
|
*
|
|
511
521
|
* @param {import("@forwardimpact/libcli").InvocationContext} ctx
|
|
512
522
|
*/
|
|
@@ -515,9 +525,10 @@ export async function runSplitCommand(ctx) {
|
|
|
515
525
|
const file = ctx.args.file;
|
|
516
526
|
if (!file) return { ok: false, code: 1, error: "split: missing input file" };
|
|
517
527
|
|
|
518
|
-
// `discuss` has the same lead + N-participants shape as `facilitate
|
|
519
|
-
// splitter buckets purely by envelope `source
|
|
520
|
-
//
|
|
528
|
+
// `discuss` has the same lead + N-participants shape as `facilitate`. The
|
|
529
|
+
// splitter buckets purely by envelope `source`, which is mode-independent.
|
|
530
|
+
// So the CLI accepts `discuss` alongside the structural modes. The CLI owns
|
|
531
|
+
// this rule. Callers do not.
|
|
521
532
|
const mode = ctx.options.mode;
|
|
522
533
|
if (!mode) return { ok: false, code: 1, error: "split: --mode is required" };
|
|
523
534
|
if (!["run", "supervise", "facilitate", "discuss"].includes(mode)) {
|
|
@@ -578,8 +589,9 @@ function parseBuckets(content) {
|
|
|
578
589
|
|
|
579
590
|
/**
|
|
580
591
|
* Compute total + per-source cost from raw file content. A structured JSON
|
|
581
|
-
* trace (from `gemba-trace download`) carries its total in
|
|
582
|
-
* but no per-source split
|
|
592
|
+
* trace (from `gemba-trace download`) carries its total in
|
|
593
|
+
* `summary.totalCostUsd` but no per-source split. `sumTraceCost` sums raw
|
|
594
|
+
* NDJSON.
|
|
583
595
|
* @param {string} content - Raw file content (structured JSON or NDJSON).
|
|
584
596
|
* @returns {{totalCostUsd: number, bySource: Record<string, number>}}
|
|
585
597
|
*/
|
|
@@ -590,7 +602,7 @@ function computeTraceCost(content) {
|
|
|
590
602
|
return { totalCostUsd: parsed.summary.totalCostUsd, bySource: {} };
|
|
591
603
|
}
|
|
592
604
|
} catch {
|
|
593
|
-
// Not a single JSON object
|
|
605
|
+
// Not a single JSON object. Treat it as NDJSON below.
|
|
594
606
|
}
|
|
595
607
|
return sumTraceCost(content.split("\n"));
|
|
596
608
|
}
|
|
@@ -610,7 +622,7 @@ export function loadTrace(runtime, file) {
|
|
|
610
622
|
return createTraceQuery(parsed);
|
|
611
623
|
}
|
|
612
624
|
} catch {
|
|
613
|
-
// Not valid JSON
|
|
625
|
+
// Not valid JSON. Fall through to NDJSON.
|
|
614
626
|
}
|
|
615
627
|
|
|
616
628
|
const collector = createTraceCollector({
|
|
@@ -623,9 +635,10 @@ export function loadTrace(runtime, file) {
|
|
|
623
635
|
}
|
|
624
636
|
|
|
625
637
|
/**
|
|
626
|
-
* Write JSON output to stdout. By default strips
|
|
627
|
-
* base64 blobs from the payload so they
|
|
628
|
-
*
|
|
638
|
+
* Write JSON output to stdout. By default the function strips
|
|
639
|
+
* `thinking.signature` base64 blobs from the payload so they do not dominate
|
|
640
|
+
* terminal output. Pass `--signatures` (surfaced as `values.signatures`) to
|
|
641
|
+
* keep them.
|
|
629
642
|
* @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
|
|
630
643
|
* @param {*} data
|
|
631
644
|
* @param {object} [values]
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* The active work-item tracker selects which column of the work-trackers
|
|
3
3
|
* matrix realizes each coordination operation (see the agent reference
|
|
4
|
-
* `work-trackers.md`). `github` is the production binding
|
|
4
|
+
* `work-trackers.md`). `github` is the production binding. The offline
|
|
5
5
|
* coordination benchmark runs under `filesystem`.
|
|
6
6
|
*/
|
|
7
7
|
export const DEFAULT_WORK_TRACKER = "github";
|
|
@@ -14,10 +14,11 @@ export const KNOWN_WORK_TRACKERS = ["github", "filesystem"];
|
|
|
14
14
|
* flag, then an inherited `LIBHARNESS_WORK_TRACKER` on the environment (so a CI
|
|
15
15
|
* job or harness can select it without the flag), then the `github` default.
|
|
16
16
|
* The harness writes the result to `LIBHARNESS_WORK_TRACKER` on the agent
|
|
17
|
-
* environment
|
|
17
|
+
* environment. This mirrors `--agent-profile` → `LIBHARNESS_AGENT_PROFILE`.
|
|
18
18
|
* @param {Record<string, string|undefined>} values - Parsed option values
|
|
19
19
|
* @param {Record<string, string|undefined>} [env] - Process environment
|
|
20
|
-
* (e.g. `runtime.proc.env`)
|
|
20
|
+
* (e.g. `runtime.proc.env`). The function reads it for the
|
|
21
|
+
* `LIBHARNESS_WORK_TRACKER` fallback.
|
|
21
22
|
* @returns {string}
|
|
22
23
|
* @throws {Error} if the resolved tracker is unknown
|
|
23
24
|
*/
|