@forwardimpact/libharness 3.0.0 → 3.0.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +60 -57
- package/package.json +2 -2
- package/src/advisor.js +47 -41
- package/src/agent-runner.js +57 -47
- package/src/benchmark/apm-installer.js +28 -28
- package/src/benchmark/env-loader.js +24 -16
- package/src/benchmark/grade.js +44 -41
- package/src/benchmark/hidden-tests.js +25 -24
- package/src/benchmark/hook-env.js +11 -9
- package/src/benchmark/invariants.js +20 -17
- package/src/benchmark/judge.js +29 -28
- package/src/benchmark/npm-installer.js +9 -8
- package/src/benchmark/report.js +53 -50
- package/src/benchmark/result.js +24 -23
- package/src/benchmark/runner.js +75 -69
- package/src/benchmark/scheduler.js +17 -16
- package/src/benchmark/task-family.js +28 -26
- package/src/benchmark/trace-split.js +8 -7
- package/src/benchmark/workdir.js +27 -25
- package/src/claude-code-executable.js +11 -11
- package/src/commands/advisor-flags.js +8 -7
- package/src/commands/assert.js +16 -15
- package/src/commands/benchmark-definition.js +11 -11
- package/src/commands/benchmark-grade.js +12 -11
- package/src/commands/benchmark-report.js +5 -5
- package/src/commands/benchmark-run.js +31 -28
- package/src/commands/by-discussion.js +10 -10
- package/src/commands/callback.js +11 -11
- package/src/commands/discuss.js +8 -7
- package/src/commands/facilitate.js +15 -13
- package/src/commands/output.js +3 -2
- package/src/commands/run.js +14 -14
- package/src/commands/scan-logs.js +21 -19
- package/src/commands/selfedit.js +14 -14
- package/src/commands/supervise.js +11 -9
- package/src/commands/task-input.js +9 -9
- package/src/commands/tee.js +10 -9
- package/src/commands/trace.js +55 -42
- package/src/commands/work-tracker.js +4 -3
- package/src/cost.js +17 -17
- package/src/discuss-tools.js +16 -16
- package/src/discusser.js +39 -38
- package/src/events/github.js +54 -37
- package/src/facilitator.js +21 -21
- package/src/inbox-poller.js +4 -4
- package/src/judge.js +32 -30
- package/src/message-bus.js +12 -11
- package/src/orchestration-loop.js +35 -36
- package/src/orchestration-toolkit.js +58 -53
- package/src/orchestrator-helpers.js +2 -2
- package/src/profile-prompt.js +54 -53
- package/src/redaction.js +63 -57
- package/src/render/line-renderer.js +5 -5
- package/src/render/orchestrator-filter.js +3 -3
- package/src/render/palette.js +11 -9
- package/src/render/tool-hints.js +18 -15
- package/src/render/turn-renderer.js +4 -4
- package/src/reply-emitter.js +2 -2
- package/src/sequence-counter.js +4 -3
- package/src/signature-filter.js +7 -6
- package/src/supervisor.js +19 -18
- package/src/tee-writer.js +25 -25
- package/src/trace-collector.js +53 -48
- package/src/trace-github.js +53 -44
- package/src/trace-multi.js +15 -13
- package/src/trace-query.js +61 -52
- package/src/trace-render.js +18 -18
- package/src/trace-usage.js +31 -28
- package/src/transcript-recorder.js +24 -20
|
@@ -3,26 +3,26 @@
|
|
|
3
3
|
*
|
|
4
4
|
* `query()` spawns a native `claude` binary that the SDK resolves from its own
|
|
5
5
|
* platform-specific optional dependency (`@anthropic-ai/claude-agent-sdk-<platform>`).
|
|
6
|
-
* `bun build --compile` bundles the SDK's JavaScript
|
|
7
|
-
* native package
|
|
8
|
-
* binary
|
|
9
|
-
* <platform> not found".
|
|
6
|
+
* `bun build --compile` bundles the SDK's JavaScript. It does not bundle that
|
|
7
|
+
* separate native package, which is not part of the import graph. So a
|
|
8
|
+
* compiled fit-* binary cannot self-resolve it, and `query()` throws "Native
|
|
9
|
+
* CLI binary for <platform> not found".
|
|
10
10
|
*
|
|
11
|
-
* In a compiled binary we point the SDK at the standalone `claude` on PATH
|
|
12
|
-
*
|
|
13
|
-
*
|
|
14
|
-
* version-matched binary
|
|
11
|
+
* In a compiled binary we point the SDK at the standalone `claude` on PATH.
|
|
12
|
+
* The gemba-bootstrap action's `fit-install.sh` installs it beside gemba-harness. A
|
|
13
|
+
* run from source keeps `node_modules`, where the SDK resolves its own
|
|
14
|
+
* version-matched binary. There we return undefined and defer to the SDK.
|
|
15
15
|
*/
|
|
16
16
|
import { LIBCLI_IS_COMPILED } from "@forwardimpact/libcli";
|
|
17
17
|
|
|
18
18
|
/**
|
|
19
19
|
* @param {object} [deps]
|
|
20
20
|
* @param {(cmd: string) => string | null | undefined} [deps.which] -
|
|
21
|
-
* PATH resolver (
|
|
21
|
+
* PATH resolver (tests inject it).
|
|
22
22
|
* @param {boolean} [deps.isCompiled] -
|
|
23
|
-
* Whether this is a `bun --compile` binary (
|
|
23
|
+
* Whether this is a `bun --compile` binary (tests inject it).
|
|
24
24
|
* @returns {string | undefined} absolute path to `claude`, or undefined to
|
|
25
|
-
*
|
|
25
|
+
* let the SDK resolve it.
|
|
26
26
|
*/
|
|
27
27
|
export function resolveClaudeCodeExecutable({
|
|
28
28
|
which = defaultWhich,
|
|
@@ -1,15 +1,16 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* Shared advisor-flag
|
|
3
|
-
* flags are identical everywhere: `--advisor-model`
|
|
4
|
-
*
|
|
5
|
-
*
|
|
2
|
+
* Shared advisor-flag parser for the four session-mode commands. The two
|
|
3
|
+
* flags are identical everywhere: `--advisor-model` and `--advisor-max-uses`.
|
|
4
|
+
* `--advisor-model` has no default. When it is absent, the commands do not
|
|
5
|
+
* offer the Advisor tool. `--advisor-max-uses` defaults to 3. It is a usage
|
|
6
|
+
* error without the model flag.
|
|
6
7
|
*/
|
|
7
8
|
|
|
8
9
|
/**
|
|
9
10
|
* Parse `--advisor-model` / `--advisor-max-uses` from parsed option values.
|
|
10
|
-
* A malformed max-uses
|
|
11
|
-
* make the budget check (`used >= maxUses`) permanently false
|
|
12
|
-
* the code-enforced cap the flag exists to guarantee.
|
|
11
|
+
* A malformed max-uses raises a usage error. It never falls back silently.
|
|
12
|
+
* NaN would make the budget check (`used >= maxUses`) permanently false. It
|
|
13
|
+
* would disable the code-enforced cap the flag exists to guarantee.
|
|
13
14
|
* @param {object} values - Parsed option values from cli.parse()
|
|
14
15
|
* @returns {{advisorModel: string|undefined, advisorMaxUses: number}}
|
|
15
16
|
*/
|
package/src/commands/assert.js
CHANGED
|
@@ -67,10 +67,10 @@ export function evaluateAssertion(values, args, fsSync) {
|
|
|
67
67
|
}
|
|
68
68
|
|
|
69
69
|
/**
|
|
70
|
-
* Attach the
|
|
71
|
-
* attaches a numeric weight
|
|
72
|
-
* `--
|
|
73
|
-
* disarm a gate.
|
|
70
|
+
* Attach the grading role of the check row. `--gate` marks a gate check.
|
|
71
|
+
* `--weight` attaches a numeric weight, and 0 marks the row diagnostic.
|
|
72
|
+
* `--gate` with any `--weight` is invalid, and that includes 0. A stray
|
|
73
|
+
* weight must never silently disarm a gate.
|
|
74
74
|
* @param {object} values
|
|
75
75
|
* @param {{test: string, pass: boolean, message?: string}} output - Mutated.
|
|
76
76
|
*/
|
|
@@ -92,8 +92,9 @@ function applyGradingFlags(values, output) {
|
|
|
92
92
|
}
|
|
93
93
|
|
|
94
94
|
/**
|
|
95
|
-
* Parse a `--weight` value
|
|
96
|
-
* `Number("")` is 0
|
|
95
|
+
* Parse a `--weight` value. Return null when it is invalid. A blank string is
|
|
96
|
+
* invalid, because `Number("")` is 0. That would silently demote the check to
|
|
97
|
+
* a diagnostic.
|
|
97
98
|
* @param {string} raw
|
|
98
99
|
* @returns {number | null}
|
|
99
100
|
*/
|
|
@@ -104,11 +105,11 @@ function parseWeight(raw) {
|
|
|
104
105
|
}
|
|
105
106
|
|
|
106
107
|
/**
|
|
107
|
-
* The grading role an emit-then-fail row keeps
|
|
108
|
-
* lose its authored role
|
|
109
|
-
*
|
|
110
|
-
*
|
|
111
|
-
* unit-weight scored check
|
|
108
|
+
* The grading role an emit-then-fail row keeps. A check that fails must not
|
|
109
|
+
* lose its authored role. An errored gate that demoted to a scored row would
|
|
110
|
+
* let a broken scaffold earn partial credit. The score would not drop to
|
|
111
|
+
* zero. Invalid flags, or flags that conflict, yield no role. The row then
|
|
112
|
+
* fails as a unit-weight scored check.
|
|
112
113
|
* @param {object} values
|
|
113
114
|
* @returns {{gate?: true, weight?: number}}
|
|
114
115
|
*/
|
|
@@ -126,10 +127,10 @@ function errorRowRole(values) {
|
|
|
126
127
|
* Run an assertion, write JSON to stdout, and return a failure envelope when
|
|
127
128
|
* the assertion does not pass.
|
|
128
129
|
*
|
|
129
|
-
* Emit-then-fail on every failure path
|
|
130
|
-
*
|
|
131
|
-
*
|
|
132
|
-
* shrinks the score
|
|
130
|
+
* Emit-then-fail applies on every failure path. An invalid grading flag
|
|
131
|
+
* writes a failed row before the nonzero exit. An errored evaluation does the
|
|
132
|
+
* same, for example `--grep` against a file the agent deleted. So a typo or a
|
|
133
|
+
* vanished target shrinks the score. It never shrinks the denominator.
|
|
133
134
|
* @param {import("@forwardimpact/libcli").InvocationContext} ctx
|
|
134
135
|
* @returns {Promise<{ok: true} | {ok: false, code: number, error: string}>}
|
|
135
136
|
*/
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* `gemba-benchmark` CLI definition.
|
|
3
|
-
* execute-on-import entry point
|
|
4
|
-
*
|
|
2
|
+
* `gemba-benchmark` CLI definition. It lives in `src/` so the bin stays an
|
|
3
|
+
* execute-on-import entry point. Launcher packages import the bin to run it.
|
|
4
|
+
* Tests import the definition and never run the CLI.
|
|
5
5
|
*/
|
|
6
6
|
|
|
7
7
|
import { runBenchmarkRunCommand } from "./benchmark-run.js";
|
|
@@ -36,7 +36,7 @@ export const definition = {
|
|
|
36
36
|
"skills-from": {
|
|
37
37
|
type: "string",
|
|
38
38
|
description:
|
|
39
|
-
"Stage .claude/ from this directory (a root
|
|
39
|
+
"Stage .claude/ from this directory (a root that holds .claude/) instead of an apm install, to exercise local, unpublished skills",
|
|
40
40
|
},
|
|
41
41
|
output: {
|
|
42
42
|
type: "string",
|
|
@@ -85,7 +85,7 @@ export const definition = {
|
|
|
85
85
|
shard: {
|
|
86
86
|
type: "string",
|
|
87
87
|
description:
|
|
88
|
-
"Run only shard i of N as i/N (1-based; default: the whole family). Each shard writes a partial results.jsonl
|
|
88
|
+
"Run only shard i of N as i/N (1-based; default: the whole family). Each shard writes a partial results.jsonl. report --input merges them.",
|
|
89
89
|
},
|
|
90
90
|
"allowed-tools": {
|
|
91
91
|
type: "string",
|
|
@@ -99,7 +99,7 @@ export const definition = {
|
|
|
99
99
|
args: [],
|
|
100
100
|
handler: runBenchmarkGradeCommand,
|
|
101
101
|
description:
|
|
102
|
-
"Grade a single task against a post-run workdir
|
|
102
|
+
"Grade a single task against a post-run workdir with no agent. Run the hidden test suite and the invariants script. Then derive the verdict from the check rows (the exit mirrors it).",
|
|
103
103
|
options: {
|
|
104
104
|
family: {
|
|
105
105
|
type: "string",
|
|
@@ -112,7 +112,7 @@ export const definition = {
|
|
|
112
112
|
"run-dir": {
|
|
113
113
|
type: "string",
|
|
114
114
|
description:
|
|
115
|
-
"Post-run directory whose cwd/ subdir is the agent CWD
|
|
115
|
+
"Post-run directory whose cwd/ subdir is the agent CWD. Both producers run against that cwd. Hooks receive it as $AGENT_CWD",
|
|
116
116
|
},
|
|
117
117
|
output: {
|
|
118
118
|
type: "string",
|
|
@@ -125,12 +125,12 @@ export const definition = {
|
|
|
125
125
|
args: [],
|
|
126
126
|
handler: runBenchmarkReportCommand,
|
|
127
127
|
description:
|
|
128
|
-
"Aggregate result records into pass@k
|
|
128
|
+
"Aggregate result records into pass@k with the OpenAI HumanEval estimator.",
|
|
129
129
|
options: {
|
|
130
130
|
input: {
|
|
131
131
|
type: "string",
|
|
132
132
|
description:
|
|
133
|
-
"Run-output directory
|
|
133
|
+
"Run-output directory that holds results.jsonl (default: benchmark-runs)",
|
|
134
134
|
},
|
|
135
135
|
k: {
|
|
136
136
|
type: "string",
|
|
@@ -143,7 +143,7 @@ export const definition = {
|
|
|
143
143
|
detail: {
|
|
144
144
|
type: "string",
|
|
145
145
|
description:
|
|
146
|
-
"Text report verbosity (full|compact, default: full). compact omits per-task detail
|
|
146
|
+
"Text report verbosity (full|compact, default: full). compact omits per-task detail, which helps with sharded run summaries.",
|
|
147
147
|
},
|
|
148
148
|
},
|
|
149
149
|
},
|
|
@@ -174,7 +174,7 @@ export const definition = {
|
|
|
174
174
|
title: "Automate with GitHub Actions",
|
|
175
175
|
url: "https://www.forwardimpact.team/docs/libraries/prove-changes/run-benchmark/ci-workflow/index.md",
|
|
176
176
|
description:
|
|
177
|
-
"Run benchmarks in CI with the forwardimpact/benchmark action.",
|
|
177
|
+
"Run benchmarks in CI with the forwardimpact/gemba-benchmark action.",
|
|
178
178
|
},
|
|
179
179
|
],
|
|
180
180
|
};
|
|
@@ -1,9 +1,9 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* `gemba-benchmark grade` — run both check-row producers (the hidden test
|
|
3
|
-
* suite and the invariants script) against a post-run workdir directory
|
|
4
|
-
*
|
|
5
|
-
* No agent and no judge
|
|
6
|
-
* against fixtures
|
|
3
|
+
* suite and the invariants script) against a post-run workdir directory.
|
|
4
|
+
* Grade the merged rows with the same derivation the benchmark runner uses.
|
|
5
|
+
* No agent runs and no judge runs. So an author validates a task's grading
|
|
6
|
+
* material against fixtures, and pays for no agent session. The process exit
|
|
7
7
|
* mirrors the graded verdict.
|
|
8
8
|
*/
|
|
9
9
|
|
|
@@ -46,12 +46,12 @@ export async function runBenchmarkGradeCommand(ctx) {
|
|
|
46
46
|
runInvariants,
|
|
47
47
|
runHiddenTests,
|
|
48
48
|
});
|
|
49
|
-
//
|
|
50
|
-
// here)
|
|
51
|
-
// crashed hook can never mint marks from the rows it emitted before
|
|
52
|
-
//
|
|
53
|
-
// the
|
|
54
|
-
//
|
|
49
|
+
// The effective-score rule matches the runner, minus the judge (none runs
|
|
50
|
+
// here). An unhealthy grader or a gate that fails zeroes the score. So a
|
|
51
|
+
// crashed hook can never mint marks from the rows it emitted before it
|
|
52
|
+
// died. A runner record keeps the raw fraction in `grade.score` and zeroes
|
|
53
|
+
// the top-level `score` instead. This record has no second field, so
|
|
54
|
+
// `grade.score` carries the effective value here.
|
|
55
55
|
if (grade.score !== undefined && !(healthy && grade.gatesPass)) {
|
|
56
56
|
grade.score = 0;
|
|
57
57
|
}
|
|
@@ -65,7 +65,8 @@ export async function runBenchmarkGradeCommand(ctx) {
|
|
|
65
65
|
...(engineError && { error: engineError.message }),
|
|
66
66
|
},
|
|
67
67
|
}),
|
|
68
|
-
//
|
|
68
|
+
// This mirrors the script for diagnosis. The graded verdict drives the
|
|
69
|
+
// exit.
|
|
69
70
|
exitCode: invariants.exitCode,
|
|
70
71
|
};
|
|
71
72
|
validateGradeRecord(record);
|
|
@@ -1,9 +1,9 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* `gemba-benchmark report` — aggregate `results.jsonl` into pass@k
|
|
3
|
-
* OpenAI HumanEval estimator.
|
|
4
|
-
* to render a markdown table. --detail=compact drops the
|
|
5
|
-
* sections so a sharded run's per-shard summary stays short
|
|
6
|
-
* renders the full report over the combined ledger
|
|
2
|
+
* `gemba-benchmark report` — aggregate `results.jsonl` into pass@k with the
|
|
3
|
+
* OpenAI HumanEval estimator. The command writes JSON by default. Pass
|
|
4
|
+
* --format=text to render a markdown table. --detail=compact drops the
|
|
5
|
+
* per-task detail sections, so a sharded run's per-shard summary stays short.
|
|
6
|
+
* The merge job renders the full report over the combined ledger.
|
|
7
7
|
*/
|
|
8
8
|
|
|
9
9
|
import { resolve } from "node:path";
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* `gemba-benchmark run` — run every task in a family for N runs
|
|
3
|
-
* ResultRecord to stdout (one JSON line per record)
|
|
4
|
-
* canonical `<output>/results.jsonl` for the report subcommand.
|
|
2
|
+
* `gemba-benchmark run` — run every task in a family for N runs. Stream each
|
|
3
|
+
* ResultRecord to stdout (one JSON line per record). Append each record to
|
|
4
|
+
* the canonical `<output>/results.jsonl` for the report subcommand.
|
|
5
5
|
*/
|
|
6
6
|
|
|
7
7
|
import { resolve } from "node:path";
|
|
@@ -30,15 +30,15 @@ export async function runBenchmarkRunCommand(ctx) {
|
|
|
30
30
|
}
|
|
31
31
|
const config = await createConfig("script", "benchmark");
|
|
32
32
|
runtime.proc.env.ANTHROPIC_API_KEY = await config.anthropicToken();
|
|
33
|
-
// The benchmark agent runs
|
|
34
|
-
// command
|
|
35
|
-
// spawns the subprocess that inherits process.env.
|
|
33
|
+
// The benchmark agent runs through createBenchmarkRunner. The supervise
|
|
34
|
+
// command does not run it. So the active-tracker env must land here before
|
|
35
|
+
// the runner spawns the subprocess that inherits process.env.
|
|
36
36
|
runtime.proc.env.LIBHARNESS_WORK_TRACKER = opts.workTracker;
|
|
37
37
|
|
|
38
38
|
// The Claude Agent SDK spawns a `claude` subprocess that inherits
|
|
39
|
-
// process.env. NODE_EXTRA_CA_CERTS
|
|
40
|
-
// inside that subprocess)
|
|
41
|
-
//
|
|
39
|
+
// process.env. NODE_EXTRA_CA_CERTS makes undici (the HTTP client
|
|
40
|
+
// inside that subprocess) fail with UND_ERR_INVALID_ARG on Node 22+.
|
|
41
|
+
// Undici then aborts every API call after 10 retries. Strip it
|
|
42
42
|
// before the SDK loads so the subprocess gets a clean environment.
|
|
43
43
|
delete runtime.proc.env.NODE_EXTRA_CA_CERTS;
|
|
44
44
|
|
|
@@ -57,13 +57,14 @@ export async function runBenchmarkRunCommand(ctx) {
|
|
|
57
57
|
}
|
|
58
58
|
|
|
59
59
|
/**
|
|
60
|
-
* Decide the exit outcome when a run streamed zero records. A run that emits
|
|
61
|
-
* records normally did nothing
|
|
62
|
-
* produced output
|
|
63
|
-
*
|
|
64
|
-
* high-index `--shard=i/N` with
|
|
65
|
-
* cells, so it exits 0 with a
|
|
66
|
-
* is
|
|
60
|
+
* Decide the exit outcome when a run streamed zero records. A run that emits
|
|
61
|
+
* no records normally did nothing. Either it discovered no tasks, or the
|
|
62
|
+
* agent never produced output. That is a failure. The command surfaces it
|
|
63
|
+
* loudly so CI does not go green on an empty benchmark. A deliberately-empty
|
|
64
|
+
* shard is the one exception. A high-index `--shard=i/N` with
|
|
65
|
+
* `N > cell count` legitimately selects zero cells, so it exits 0 with a
|
|
66
|
+
* stderr note. This function is exported so a test can reach the
|
|
67
|
+
* relaxed-guard branch without the full handler's config and SDK setup.
|
|
67
68
|
* @param {{shard: {index: number, total: number} | null}} opts
|
|
68
69
|
* @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
|
|
69
70
|
* @returns {{ok: true} | {ok: false, code: number, error: string}}
|
|
@@ -79,16 +80,17 @@ export function resolveZeroRecordOutcome(opts, runtime) {
|
|
|
79
80
|
ok: false,
|
|
80
81
|
code: 1,
|
|
81
82
|
error:
|
|
82
|
-
"benchmark produced no result records
|
|
83
|
+
"benchmark produced no result records. No task ran to completion. Check the family's tasks/, apm install, and agent availability (ANTHROPIC_API_KEY / claude CLI / IS_SANDBOX)",
|
|
83
84
|
};
|
|
84
85
|
}
|
|
85
86
|
|
|
86
87
|
/**
|
|
87
|
-
* Parse and validate benchmark run options.
|
|
88
|
-
* defaults
|
|
88
|
+
* Parse and validate benchmark run options. This function is exported so a
|
|
89
|
+
* test can verify the defaults and the resolved work tracker.
|
|
89
90
|
* @param {Record<string, string|undefined>} values - Parsed option values
|
|
90
|
-
* @param {Record<string, string|undefined>} [env] - Process environment
|
|
91
|
-
* for the `LIBHARNESS_WORK_TRACKER` fallback when
|
|
91
|
+
* @param {Record<string, string|undefined>} [env] - Process environment. The
|
|
92
|
+
* parser reads it for the `LIBHARNESS_WORK_TRACKER` fallback when
|
|
93
|
+
* `--work-tracker` is absent.
|
|
92
94
|
* @returns {object}
|
|
93
95
|
*/
|
|
94
96
|
export function parseRunOptions(values, env = {}) {
|
|
@@ -132,7 +134,8 @@ function parseMaxTurns(raw) {
|
|
|
132
134
|
|
|
133
135
|
/**
|
|
134
136
|
* Parse a `--shard=<i>/<N>` selector into `{index, total}` (1-based), or `null`
|
|
135
|
-
* for an unsharded run.
|
|
137
|
+
* for an unsharded run. Validate that the parts are integers and that
|
|
138
|
+
* `1 ≤ index ≤ total`.
|
|
136
139
|
* @param {string|undefined} raw
|
|
137
140
|
* @returns {{index: number, total: number} | null}
|
|
138
141
|
*/
|
|
@@ -147,17 +150,17 @@ export function parseShard(raw) {
|
|
|
147
150
|
return { index, total };
|
|
148
151
|
}
|
|
149
152
|
|
|
150
|
-
//
|
|
151
|
-
// agent-under-test + judge)
|
|
152
|
-
//
|
|
153
|
-
// machines
|
|
153
|
+
// The ceiling is conservative because each cell spawns ~3 agent subprocesses
|
|
154
|
+
// (lead + agent-under-test + judge). A low ceiling makes sure that a single
|
|
155
|
+
// runner does not thrash. Most of the CI speedup comes from Layer-2 shards
|
|
156
|
+
// across machines. It does not come from a higher in-job default.
|
|
154
157
|
const CONCURRENCY_CEILING = 4;
|
|
155
158
|
|
|
156
159
|
/**
|
|
157
160
|
* Resolve the cell concurrency: `--concurrency` flag > the
|
|
158
161
|
* `LIBHARNESS_BENCHMARK_CONCURRENCY` env var > a CPU-aware default of
|
|
159
|
-
* `min(CONCURRENCY_CEILING, max(2, ⌊cores/2⌋))`. The default is `> 1
|
|
160
|
-
* concurrency is on transparently
|
|
162
|
+
* `min(CONCURRENCY_CEILING, max(2, ⌊cores/2⌋))`. The default is `> 1`, so
|
|
163
|
+
* concurrency is on transparently. No consumer needs to opt in.
|
|
161
164
|
* @param {Record<string, string|undefined>} values
|
|
162
165
|
* @param {Record<string, string|undefined>} [env]
|
|
163
166
|
* @returns {number}
|
|
@@ -4,11 +4,11 @@ const FIRST_LINE_CAP = 64 * 1024;
|
|
|
4
4
|
|
|
5
5
|
/**
|
|
6
6
|
* Read the first newline-terminated line of a file, bounded to the first
|
|
7
|
-
* {@link FIRST_LINE_CAP} bytes. Trace `.ndjson` files can be many MB
|
|
8
|
-
* Step 2.6 meta header is always small
|
|
9
|
-
*
|
|
10
|
-
* `openSync`/`readSync`/`closeSync` trio
|
|
11
|
-
* `runtime.fsSync` surface.
|
|
7
|
+
* {@link FIRST_LINE_CAP} bytes. Trace `.ndjson` files can be many MB. The
|
|
8
|
+
* Step 2.6 meta header is always small. So a bounded positional read does
|
|
9
|
+
* not load whole files into memory just to inspect the header. This function
|
|
10
|
+
* reads the positional `openSync`/`readSync`/`closeSync` trio off the
|
|
11
|
+
* injected `runtime.fsSync` surface.
|
|
12
12
|
*
|
|
13
13
|
* @param {object} fsSync - Sync filesystem surface (`runtime.fsSync`).
|
|
14
14
|
* @param {string} path
|
|
@@ -30,9 +30,9 @@ function readFirstLine(fsSync, path) {
|
|
|
30
30
|
/**
|
|
31
31
|
* Scan a directory for `.ndjson` files whose meta header carries the
|
|
32
32
|
* given discussion_id. The Step 2.6 first-line guarantee makes the
|
|
33
|
-
* lookup cheap
|
|
34
|
-
* meta header (e.g. legacy
|
|
35
|
-
*
|
|
33
|
+
* lookup cheap. This function reads only the first line per file. It
|
|
34
|
+
* silently skips files without a meta header (e.g. legacy
|
|
35
|
+
* supervise/facilitate traces). Such a file is not an error.
|
|
36
36
|
*
|
|
37
37
|
* @param {string} dir
|
|
38
38
|
* @param {string} discussionId
|
|
@@ -74,8 +74,8 @@ export function findTracesByDiscussion(dir, discussionId, fsSync) {
|
|
|
74
74
|
/**
|
|
75
75
|
* `gemba-trace by-discussion <discussion-id> [trace-dir]` — list trace
|
|
76
76
|
* files whose meta header carries the given discussion_id, one per
|
|
77
|
-
* line
|
|
78
|
-
*
|
|
77
|
+
* line. Order them by first-event timestamp (file mtime ascending).
|
|
78
|
+
* You can use the result with `xargs cat` for a chronological merge.
|
|
79
79
|
*
|
|
80
80
|
* @param {import("@forwardimpact/libcli").InvocationContext} ctx
|
|
81
81
|
* @returns {Promise<{ok: true} | {ok: false, code: number, error: string}>}
|
package/src/commands/callback.js
CHANGED
|
@@ -3,12 +3,12 @@ import { sumTraceCost } from "../cost.js";
|
|
|
3
3
|
/**
|
|
4
4
|
* Scan an NDJSON trace and return the last orchestrator summary event,
|
|
5
5
|
* the first `meta` event's `discussion_id`, and any structured replies
|
|
6
|
-
*
|
|
6
|
+
* the discusser collected. This function skips malformed lines.
|
|
7
7
|
*
|
|
8
|
-
* The runner is verdict-agnostic
|
|
9
|
-
*
|
|
10
|
-
* "adjourned"/"recessed"/"failed" from discuss). The bridge
|
|
11
|
-
* its channel semantics.
|
|
8
|
+
* The runner is verdict-agnostic. It passes through whatever the trace
|
|
9
|
+
* carries, verbatim ("success"/"failure" from supervise/facilitate;
|
|
10
|
+
* canonical "adjourned"/"recessed"/"failed" from discuss). The bridge
|
|
11
|
+
* layer maps to its channel semantics.
|
|
12
12
|
*
|
|
13
13
|
* @param {string} content - Raw NDJSON trace content.
|
|
14
14
|
* @returns {{verdict: string, summary: string, replies: object[], trigger?: object, discussionId?: string} | null}
|
|
@@ -53,9 +53,9 @@ function readTraceSummary(content) {
|
|
|
53
53
|
}
|
|
54
54
|
|
|
55
55
|
/**
|
|
56
|
-
* Callback command — read an NDJSON trace
|
|
57
|
-
* orchestrator summary
|
|
58
|
-
*
|
|
56
|
+
* Callback command — read an NDJSON trace and extract the terminal
|
|
57
|
+
* orchestrator summary. POST a canonical callback body to the configured
|
|
58
|
+
* URL. `kata-dispatch.yml` uses this command to deliver the lead's
|
|
59
59
|
* conclusion to the bridge that dispatched the run.
|
|
60
60
|
*
|
|
61
61
|
* Wire shape (single shape across modes):
|
|
@@ -87,11 +87,11 @@ export async function runCallbackCommand(ctx) {
|
|
|
87
87
|
const content = runtime.fsSync.readFileSync(traceFile, "utf8");
|
|
88
88
|
const found = readTraceSummary(content) ?? {
|
|
89
89
|
verdict: "failed",
|
|
90
|
-
summary: "
|
|
90
|
+
summary: "The run ended and produced no summary.",
|
|
91
91
|
replies: [],
|
|
92
92
|
};
|
|
93
|
-
// Total spend across every participant in the trace
|
|
94
|
-
// it alongside the verdict so a dispatched run reports what it cost.
|
|
93
|
+
// Total spend across every participant in the trace. The bridge surfaces
|
|
94
|
+
// it alongside the verdict, so a dispatched run reports what it cost.
|
|
95
95
|
const { totalCostUsd } = sumTraceCost(content.split("\n"));
|
|
96
96
|
|
|
97
97
|
const discussionId = found.discussionId ?? discussionIdOverride ?? null;
|
package/src/commands/discuss.js
CHANGED
|
@@ -17,8 +17,8 @@ function parseAgentProfiles(raw, cwd, maxTurns) {
|
|
|
17
17
|
}
|
|
18
18
|
|
|
19
19
|
/**
|
|
20
|
-
* Parse and validate discuss command options.
|
|
21
|
-
* defaults and the legacy-flag clean break.
|
|
20
|
+
* Parse and validate discuss command options. This function is exported so a
|
|
21
|
+
* test can verify the defaults and the legacy-flag clean break.
|
|
22
22
|
* @param {object} values - Parsed option values
|
|
23
23
|
* @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
|
|
24
24
|
* @returns {object}
|
|
@@ -30,8 +30,8 @@ export function parseDiscussOptions(values, runtime) {
|
|
|
30
30
|
);
|
|
31
31
|
|
|
32
32
|
const profilesRaw = values["agent-profiles"];
|
|
33
|
-
// `||` (not `??`) so an empty-string flag from a CI forwarder falls back
|
|
34
|
-
// the default
|
|
33
|
+
// `||` (not `??`) so an empty-string flag from a CI forwarder falls back
|
|
34
|
+
// to the default. The empty string does not override the default.
|
|
35
35
|
const agentCwd = resolve(values["agent-cwd"] || ".");
|
|
36
36
|
|
|
37
37
|
const maxTurnsRaw = values["max-turns"] || "40";
|
|
@@ -74,8 +74,8 @@ export function parseDiscussOptions(values, runtime) {
|
|
|
74
74
|
|
|
75
75
|
/**
|
|
76
76
|
* Discuss command — run a discusser-led session with suspend/resume
|
|
77
|
-
* semantics
|
|
78
|
-
*
|
|
77
|
+
* semantics. The session threads `discussion_id` through the trace, so
|
|
78
|
+
* you can query multi-run conversations as one.
|
|
79
79
|
*
|
|
80
80
|
* @param {import("@forwardimpact/libcli").InvocationContext} ctx
|
|
81
81
|
* @returns {Promise<{ok: boolean, code?: number, error?: string}>}
|
|
@@ -102,7 +102,8 @@ export async function runDiscussCommand(ctx) {
|
|
|
102
102
|
runtime.proc.env.LIBHARNESS_AGENT_PROFILE = opts.leadProfile;
|
|
103
103
|
}
|
|
104
104
|
// Unconditional so the default "github" is observable to the agent's
|
|
105
|
-
// active-tracker resolution
|
|
105
|
+
// active-tracker resolution. This mirrors --agent-profile's env write
|
|
106
|
+
// above.
|
|
106
107
|
runtime.proc.env.LIBHARNESS_WORK_TRACKER = opts.workTracker;
|
|
107
108
|
|
|
108
109
|
const { query } = await import("@anthropic-ai/claude-agent-sdk");
|
|
@@ -22,9 +22,9 @@ function parseAgentProfiles(raw, cwd, maxTurns) {
|
|
|
22
22
|
}
|
|
23
23
|
|
|
24
24
|
/**
|
|
25
|
-
* Parse and validate facilitate command options.
|
|
26
|
-
*
|
|
27
|
-
* of the package's public API.
|
|
25
|
+
* Parse and validate facilitate command options. This function is exported
|
|
26
|
+
* so a test can cover the contract that threads `--max-turns` to each
|
|
27
|
+
* agent. It is not part of the package's public API.
|
|
28
28
|
* @param {object} values - Parsed option values
|
|
29
29
|
* @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
|
|
30
30
|
* @returns {object} Parsed options
|
|
@@ -37,17 +37,18 @@ export function parseFacilitateOptions(values, runtime) {
|
|
|
37
37
|
|
|
38
38
|
const profilesRaw = values["agent-profiles"];
|
|
39
39
|
if (!profilesRaw) throw new Error("--agent-profiles is required");
|
|
40
|
-
// `||` (not `??`) so an empty-string flag from a CI forwarder falls back
|
|
41
|
-
// the default
|
|
40
|
+
// `||` (not `??`) so an empty-string flag from a CI forwarder falls back
|
|
41
|
+
// to the default. The empty string does not override the default.
|
|
42
42
|
const agentCwd = resolve(values["agent-cwd"] || ".");
|
|
43
43
|
|
|
44
44
|
const maxTurnsRaw = values["max-turns"] || "20";
|
|
45
45
|
const maxTurns = maxTurnsRaw === "0" ? 0 : parseInt(maxTurnsRaw, 10);
|
|
46
46
|
|
|
47
|
-
// Thread --max-turns into each participant
|
|
48
|
-
// agent silently falls back to the 50-turn default in
|
|
49
|
-
// when the caller raises the budget.
|
|
50
|
-
// staff-engineer terminated at 51
|
|
47
|
+
// Thread --max-turns into each participant. Without this, every
|
|
48
|
+
// facilitated agent silently falls back to the 50-turn default in
|
|
49
|
+
// facilitator.js, even when the caller raises the budget. Run
|
|
50
|
+
// 26078312414 showed this. In it, staff-engineer terminated at 51
|
|
51
|
+
// turns despite --max-turns=200.
|
|
51
52
|
const agentConfigs = parseAgentProfiles(profilesRaw, agentCwd, maxTurns);
|
|
52
53
|
|
|
53
54
|
return {
|
|
@@ -77,9 +78,9 @@ export async function runFacilitateCommand(ctx) {
|
|
|
77
78
|
const runtime = ctx.deps.runtime;
|
|
78
79
|
const opts = parseFacilitateOptions(ctx.options, runtime);
|
|
79
80
|
|
|
80
|
-
// Build the redactor as the first observable side-effect after
|
|
81
|
-
//
|
|
82
|
-
// env
|
|
81
|
+
// Build the redactor as the first observable side-effect after the parser
|
|
82
|
+
// reads the options. The env snapshot must freeze BEFORE any in-process
|
|
83
|
+
// env write the command performs (e.g. LIBHARNESS_AGENT_PROFILE).
|
|
83
84
|
const redactor = createRedactor({ runtime });
|
|
84
85
|
|
|
85
86
|
const fileStream = opts.outputPath
|
|
@@ -98,7 +99,8 @@ export async function runFacilitateCommand(ctx) {
|
|
|
98
99
|
runtime.proc.env.LIBHARNESS_AGENT_PROFILE = opts.facilitatorProfile;
|
|
99
100
|
}
|
|
100
101
|
// Unconditional so the default "github" is observable to the agent's
|
|
101
|
-
// active-tracker resolution
|
|
102
|
+
// active-tracker resolution. This mirrors --agent-profile's env write
|
|
103
|
+
// above.
|
|
102
104
|
runtime.proc.env.LIBHARNESS_WORK_TRACKER = opts.workTracker;
|
|
103
105
|
|
|
104
106
|
const { query } = await import("@anthropic-ai/claude-agent-sdk");
|
package/src/commands/output.js
CHANGED
|
@@ -21,8 +21,9 @@ export async function runOutputCommand(ctx) {
|
|
|
21
21
|
now: () => isoTimestamp(runtime.clock.now()),
|
|
22
22
|
});
|
|
23
23
|
|
|
24
|
-
// `runtime.proc.stdin` is an AsyncIterable of UTF-8 lines
|
|
25
|
-
//
|
|
24
|
+
// `runtime.proc.stdin` is an AsyncIterable of UTF-8 lines. The runtime
|
|
25
|
+
// splits them on newlines. So each yielded value is exactly one NDJSON
|
|
26
|
+
// record.
|
|
26
27
|
for await (const line of runtime.proc.stdin) {
|
|
27
28
|
collector.addLine(line);
|
|
28
29
|
}
|