@forwardimpact/libharness 2.0.0 → 3.0.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +68 -65
- package/package.json +15 -13
- package/src/advisor.js +47 -41
- package/src/agent-runner.js +58 -48
- package/src/benchmark/apm-installer.js +28 -28
- package/src/benchmark/env-loader.js +24 -16
- package/src/benchmark/grade.js +44 -41
- package/src/benchmark/hidden-tests.js +25 -24
- package/src/benchmark/hook-env.js +11 -9
- package/src/benchmark/invariants.js +20 -17
- package/src/benchmark/judge.js +29 -28
- package/src/benchmark/npm-installer.js +9 -8
- package/src/benchmark/report.js +53 -50
- package/src/benchmark/result.js +24 -23
- package/src/benchmark/runner.js +75 -69
- package/src/benchmark/scheduler.js +17 -16
- package/src/benchmark/task-family.js +29 -27
- package/src/benchmark/trace-split.js +9 -8
- package/src/benchmark/workdir.js +27 -25
- package/src/claude-code-executable.js +11 -11
- package/src/commands/advisor-flags.js +8 -7
- package/src/commands/assert.js +16 -15
- package/src/commands/benchmark-definition.js +20 -20
- package/src/commands/benchmark-grade.js +13 -12
- package/src/commands/benchmark-report.js +5 -5
- package/src/commands/benchmark-run.js +31 -28
- package/src/commands/by-discussion.js +11 -11
- package/src/commands/callback.js +11 -11
- package/src/commands/discuss.js +8 -7
- package/src/commands/facilitate.js +16 -14
- package/src/commands/output.js +4 -3
- package/src/commands/run.js +15 -15
- package/src/commands/scan-logs.js +22 -20
- package/src/commands/selfedit.js +124 -0
- package/src/commands/supervise.js +13 -11
- package/src/commands/task-input.js +9 -9
- package/src/commands/tee.js +11 -10
- package/src/commands/trace.js +55 -42
- package/src/commands/work-tracker.js +4 -3
- package/src/cost.js +17 -17
- package/src/discuss-tools.js +16 -16
- package/src/discusser.js +39 -38
- package/src/events/github.js +54 -37
- package/src/facilitator.js +21 -21
- package/src/inbox-poller.js +4 -4
- package/src/judge.js +32 -30
- package/src/message-bus.js +12 -11
- package/src/orchestration-loop.js +35 -36
- package/src/orchestration-toolkit.js +58 -53
- package/src/orchestrator-helpers.js +2 -2
- package/src/profile-prompt.js +54 -53
- package/src/redaction.js +63 -57
- package/src/render/line-renderer.js +5 -5
- package/src/render/orchestrator-filter.js +3 -3
- package/src/render/palette.js +11 -9
- package/src/render/tool-hints.js +18 -15
- package/src/render/turn-renderer.js +4 -4
- package/src/reply-emitter.js +2 -2
- package/src/sequence-counter.js +4 -3
- package/src/signature-filter.js +7 -6
- package/src/supervisor.js +19 -18
- package/src/tee-writer.js +25 -25
- package/src/trace-collector.js +53 -48
- package/src/trace-github.js +53 -44
- package/src/trace-multi.js +16 -14
- package/src/trace-query.js +61 -52
- package/src/trace-render.js +19 -19
- package/src/trace-usage.js +31 -28
- package/src/transcript-recorder.js +24 -20
- package/bin/fit-benchmark.js +0 -44
- package/bin/fit-harness.js +0 -412
- package/bin/fit-selfedit.js +0 -165
- package/bin/fit-trace.js +0 -520
package/src/benchmark/workdir.js
CHANGED
|
@@ -7,9 +7,9 @@
|
|
|
7
7
|
* runAgent → invariants → judge → teardown.
|
|
8
8
|
*
|
|
9
9
|
* Filesystem, subprocess, clock, and process-signal access all route through
|
|
10
|
-
* the injected `runtime` bag. Only raw TCP plumbing (`node:net`) stays
|
|
11
|
-
*
|
|
12
|
-
* surface.
|
|
10
|
+
* the injected `runtime` bag. Only raw TCP plumbing (`node:net`) stays
|
|
11
|
+
* direct. It is not an ambient-dependency smell, and the runtime bag models
|
|
12
|
+
* no socket surface.
|
|
13
13
|
*/
|
|
14
14
|
|
|
15
15
|
import { createServer } from "node:net";
|
|
@@ -27,7 +27,7 @@ const DEFAULT_TERM_GRACE_MS = 5_000;
|
|
|
27
27
|
* @property {string} runDir - Parent of `cwd`; holds trace/log siblings.
|
|
28
28
|
* @property {number} port - Allocated TCP port for the agent.
|
|
29
29
|
* @property {number} pgid - Process-group id captured from the preflight child.
|
|
30
|
-
* @property {*} scaffold - Reserved per design § Components
|
|
30
|
+
* @property {*} scaffold - Reserved per design § Components. v1 sets null.
|
|
31
31
|
* @property {string} agentTracePath
|
|
32
32
|
* @property {string} supervisorTracePath
|
|
33
33
|
* @property {string} judgeTracePath
|
|
@@ -59,8 +59,8 @@ export class WorkdirManager {
|
|
|
59
59
|
this.termGraceMs = termGraceMs ?? DEFAULT_TERM_GRACE_MS;
|
|
60
60
|
this.familyRootPath = familyRootPath ?? null;
|
|
61
61
|
this.runtime = runtime;
|
|
62
|
-
// One registry per manager
|
|
63
|
-
// so concurrent cells
|
|
62
|
+
// One registry per manager. It hands out distinct, bindable ports under a
|
|
63
|
+
// lock, so two concurrent cells never get the same number.
|
|
64
64
|
this.ports = new PortRegistry();
|
|
65
65
|
}
|
|
66
66
|
|
|
@@ -77,9 +77,10 @@ export class WorkdirManager {
|
|
|
77
77
|
const cwd = join(runDir, "cwd");
|
|
78
78
|
await fs.mkdir(cwd, { recursive: true });
|
|
79
79
|
|
|
80
|
-
// Family-level shared fixtures
|
|
81
|
-
// present. They form the shared base
|
|
82
|
-
// overlay on top (fs.cp defaults to
|
|
80
|
+
// Family-level shared fixtures follow convention over configuration. The
|
|
81
|
+
// manager copies them if they are present. They form the shared base. The
|
|
82
|
+
// per-task workdir/specs below overlay on top (fs.cp defaults to
|
|
83
|
+
// force:true, so a per-task file wins).
|
|
83
84
|
if (this.familyRootPath) {
|
|
84
85
|
await fs
|
|
85
86
|
.cp(join(this.familyRootPath, "workdir"), cwd, { recursive: true })
|
|
@@ -162,7 +163,7 @@ export class WorkdirManager {
|
|
|
162
163
|
try {
|
|
163
164
|
proc.kill(-workdir.pgid, "SIGTERM");
|
|
164
165
|
} catch {
|
|
165
|
-
//
|
|
166
|
+
// The process group is already gone. That is fine.
|
|
166
167
|
}
|
|
167
168
|
await clock.sleep(this.termGraceMs);
|
|
168
169
|
try {
|
|
@@ -170,17 +171,17 @@ export class WorkdirManager {
|
|
|
170
171
|
} catch {
|
|
171
172
|
// Already exited.
|
|
172
173
|
}
|
|
173
|
-
// Poll briefly until the process group is empty
|
|
174
|
-
// before the kernel
|
|
174
|
+
// Poll briefly until the process group is empty. SIGKILL returns
|
|
175
|
+
// before the kernel reaps every descendant.
|
|
175
176
|
await waitFor(
|
|
176
177
|
this.runtime,
|
|
177
178
|
async () => (await countDescendants(this.runtime, workdir.pgid)) === 0,
|
|
178
179
|
2_000,
|
|
179
180
|
);
|
|
180
181
|
}
|
|
181
|
-
// Release the reservation in a finally so a
|
|
182
|
-
// number from the in-use set
|
|
183
|
-
//
|
|
182
|
+
// Release the reservation in a finally, so a probe that throws cannot leak
|
|
183
|
+
// the number from the in-use set. Release after the port-free probe, so the
|
|
184
|
+
// registry can hand a freed number to a cell that waits.
|
|
184
185
|
try {
|
|
185
186
|
const portFree = await isPortFree(workdir.port);
|
|
186
187
|
const descendants = await countDescendants(this.runtime, workdir.pgid);
|
|
@@ -196,10 +197,10 @@ export class WorkdirManager {
|
|
|
196
197
|
* close-then-return allocator whose allocate→bind window let two concurrent
|
|
197
198
|
* cells receive the same number.
|
|
198
199
|
*
|
|
199
|
-
* The reservation is the *number
|
|
200
|
-
*
|
|
201
|
-
* chain
|
|
202
|
-
* set, so no two in-flight cells share a port.
|
|
200
|
+
* The reservation is the *number*. The registry holds no socket, because the
|
|
201
|
+
* agent could not bind a held socket later. `acquire` serializes through a
|
|
202
|
+
* one-slot promise chain. It re-probes if the OS hands back a number already
|
|
203
|
+
* in the live in-use set, so no two in-flight cells share a port.
|
|
203
204
|
*/
|
|
204
205
|
export class PortRegistry {
|
|
205
206
|
#inUse = new Set();
|
|
@@ -216,7 +217,7 @@ export class PortRegistry {
|
|
|
216
217
|
return port;
|
|
217
218
|
});
|
|
218
219
|
// Keep the chain alive even if one acquire rejects, so later acquires
|
|
219
|
-
// still run
|
|
220
|
+
// still run. Swallow the rejection here. Surface it on `next`.
|
|
220
221
|
this.#tail = next.catch(() => {});
|
|
221
222
|
return next;
|
|
222
223
|
}
|
|
@@ -231,8 +232,8 @@ export class PortRegistry {
|
|
|
231
232
|
* Spawn preflight. Stays detached so we can SIGTERM the whole process group.
|
|
232
233
|
* @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
|
|
233
234
|
* @param {string} script
|
|
234
|
-
* @param {string} cwd - Agent CWD passed
|
|
235
|
-
* @param {number} port - Free TCP port passed
|
|
235
|
+
* @param {string} cwd - Agent CWD, passed in $AGENT_CWD.
|
|
236
|
+
* @param {number} port - Free TCP port, passed in $PORT.
|
|
236
237
|
* @param {{taskId: string, taskDir: string, hooksDir: string, familyDir: string|null}} vars - Extra hook env vars.
|
|
237
238
|
* @returns {Promise<{pgid: number, error?: {phase: string, message: string, exitCode: number}}>}
|
|
238
239
|
*/
|
|
@@ -269,8 +270,9 @@ async function runPreflight(runtime, script, cwd, port, vars) {
|
|
|
269
270
|
}
|
|
270
271
|
|
|
271
272
|
/**
|
|
272
|
-
* Allocate a free TCP port
|
|
273
|
-
*
|
|
273
|
+
* Allocate a free TCP port. Bind to port 0, then release it. The `grade`
|
|
274
|
+
* subcommand shares this function, because it needs a plausible `$PORT` for
|
|
275
|
+
* the hook env.
|
|
274
276
|
* @returns {Promise<number>}
|
|
275
277
|
*/
|
|
276
278
|
export function probeFreePort() {
|
|
@@ -340,7 +342,7 @@ async function waitFor(runtime, predicate, timeoutMs) {
|
|
|
340
342
|
}
|
|
341
343
|
|
|
342
344
|
/**
|
|
343
|
-
* Factory function
|
|
345
|
+
* Factory function. Wires the real dependencies.
|
|
344
346
|
* @param {ConstructorParameters<typeof WorkdirManager>[0]} deps
|
|
345
347
|
* @returns {WorkdirManager}
|
|
346
348
|
*/
|
|
@@ -3,26 +3,26 @@
|
|
|
3
3
|
*
|
|
4
4
|
* `query()` spawns a native `claude` binary that the SDK resolves from its own
|
|
5
5
|
* platform-specific optional dependency (`@anthropic-ai/claude-agent-sdk-<platform>`).
|
|
6
|
-
* `bun build --compile` bundles the SDK's JavaScript
|
|
7
|
-
* native package
|
|
8
|
-
* binary
|
|
9
|
-
* <platform> not found".
|
|
6
|
+
* `bun build --compile` bundles the SDK's JavaScript. It does not bundle that
|
|
7
|
+
* separate native package, which is not part of the import graph. So a
|
|
8
|
+
* compiled fit-* binary cannot self-resolve it, and `query()` throws "Native
|
|
9
|
+
* CLI binary for <platform> not found".
|
|
10
10
|
*
|
|
11
|
-
* In a compiled binary we point the SDK at the standalone `claude` on PATH
|
|
12
|
-
*
|
|
13
|
-
*
|
|
14
|
-
* version-matched binary
|
|
11
|
+
* In a compiled binary we point the SDK at the standalone `claude` on PATH.
|
|
12
|
+
* The gemba-bootstrap action's `fit-install.sh` installs it beside gemba-harness. A
|
|
13
|
+
* run from source keeps `node_modules`, where the SDK resolves its own
|
|
14
|
+
* version-matched binary. There we return undefined and defer to the SDK.
|
|
15
15
|
*/
|
|
16
16
|
import { LIBCLI_IS_COMPILED } from "@forwardimpact/libcli";
|
|
17
17
|
|
|
18
18
|
/**
|
|
19
19
|
* @param {object} [deps]
|
|
20
20
|
* @param {(cmd: string) => string | null | undefined} [deps.which] -
|
|
21
|
-
* PATH resolver (
|
|
21
|
+
* PATH resolver (tests inject it).
|
|
22
22
|
* @param {boolean} [deps.isCompiled] -
|
|
23
|
-
* Whether this is a `bun --compile` binary (
|
|
23
|
+
* Whether this is a `bun --compile` binary (tests inject it).
|
|
24
24
|
* @returns {string | undefined} absolute path to `claude`, or undefined to
|
|
25
|
-
*
|
|
25
|
+
* let the SDK resolve it.
|
|
26
26
|
*/
|
|
27
27
|
export function resolveClaudeCodeExecutable({
|
|
28
28
|
which = defaultWhich,
|
|
@@ -1,15 +1,16 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* Shared advisor-flag
|
|
3
|
-
* flags are identical everywhere: `--advisor-model`
|
|
4
|
-
*
|
|
5
|
-
*
|
|
2
|
+
* Shared advisor-flag parser for the four session-mode commands. The two
|
|
3
|
+
* flags are identical everywhere: `--advisor-model` and `--advisor-max-uses`.
|
|
4
|
+
* `--advisor-model` has no default. When it is absent, the commands do not
|
|
5
|
+
* offer the Advisor tool. `--advisor-max-uses` defaults to 3. It is a usage
|
|
6
|
+
* error without the model flag.
|
|
6
7
|
*/
|
|
7
8
|
|
|
8
9
|
/**
|
|
9
10
|
* Parse `--advisor-model` / `--advisor-max-uses` from parsed option values.
|
|
10
|
-
* A malformed max-uses
|
|
11
|
-
* make the budget check (`used >= maxUses`) permanently false
|
|
12
|
-
* the code-enforced cap the flag exists to guarantee.
|
|
11
|
+
* A malformed max-uses raises a usage error. It never falls back silently.
|
|
12
|
+
* NaN would make the budget check (`used >= maxUses`) permanently false. It
|
|
13
|
+
* would disable the code-enforced cap the flag exists to guarantee.
|
|
13
14
|
* @param {object} values - Parsed option values from cli.parse()
|
|
14
15
|
* @returns {{advisorModel: string|undefined, advisorMaxUses: number}}
|
|
15
16
|
*/
|
package/src/commands/assert.js
CHANGED
|
@@ -67,10 +67,10 @@ export function evaluateAssertion(values, args, fsSync) {
|
|
|
67
67
|
}
|
|
68
68
|
|
|
69
69
|
/**
|
|
70
|
-
* Attach the
|
|
71
|
-
* attaches a numeric weight
|
|
72
|
-
* `--
|
|
73
|
-
* disarm a gate.
|
|
70
|
+
* Attach the grading role of the check row. `--gate` marks a gate check.
|
|
71
|
+
* `--weight` attaches a numeric weight, and 0 marks the row diagnostic.
|
|
72
|
+
* `--gate` with any `--weight` is invalid, and that includes 0. A stray
|
|
73
|
+
* weight must never silently disarm a gate.
|
|
74
74
|
* @param {object} values
|
|
75
75
|
* @param {{test: string, pass: boolean, message?: string}} output - Mutated.
|
|
76
76
|
*/
|
|
@@ -92,8 +92,9 @@ function applyGradingFlags(values, output) {
|
|
|
92
92
|
}
|
|
93
93
|
|
|
94
94
|
/**
|
|
95
|
-
* Parse a `--weight` value
|
|
96
|
-
* `Number("")` is 0
|
|
95
|
+
* Parse a `--weight` value. Return null when it is invalid. A blank string is
|
|
96
|
+
* invalid, because `Number("")` is 0. That would silently demote the check to
|
|
97
|
+
* a diagnostic.
|
|
97
98
|
* @param {string} raw
|
|
98
99
|
* @returns {number | null}
|
|
99
100
|
*/
|
|
@@ -104,11 +105,11 @@ function parseWeight(raw) {
|
|
|
104
105
|
}
|
|
105
106
|
|
|
106
107
|
/**
|
|
107
|
-
* The grading role an emit-then-fail row keeps
|
|
108
|
-
* lose its authored role
|
|
109
|
-
*
|
|
110
|
-
*
|
|
111
|
-
* unit-weight scored check
|
|
108
|
+
* The grading role an emit-then-fail row keeps. A check that fails must not
|
|
109
|
+
* lose its authored role. An errored gate that demoted to a scored row would
|
|
110
|
+
* let a broken scaffold earn partial credit. The score would not drop to
|
|
111
|
+
* zero. Invalid flags, or flags that conflict, yield no role. The row then
|
|
112
|
+
* fails as a unit-weight scored check.
|
|
112
113
|
* @param {object} values
|
|
113
114
|
* @returns {{gate?: true, weight?: number}}
|
|
114
115
|
*/
|
|
@@ -126,10 +127,10 @@ function errorRowRole(values) {
|
|
|
126
127
|
* Run an assertion, write JSON to stdout, and return a failure envelope when
|
|
127
128
|
* the assertion does not pass.
|
|
128
129
|
*
|
|
129
|
-
* Emit-then-fail on every failure path
|
|
130
|
-
*
|
|
131
|
-
*
|
|
132
|
-
* shrinks the score
|
|
130
|
+
* Emit-then-fail applies on every failure path. An invalid grading flag
|
|
131
|
+
* writes a failed row before the nonzero exit. An errored evaluation does the
|
|
132
|
+
* same, for example `--grep` against a file the agent deleted. So a typo or a
|
|
133
|
+
* vanished target shrinks the score. It never shrinks the denominator.
|
|
133
134
|
* @param {import("@forwardimpact/libcli").InvocationContext} ctx
|
|
134
135
|
* @returns {Promise<{ok: true} | {ok: false, code: number, error: string}>}
|
|
135
136
|
*/
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* `
|
|
3
|
-
* execute-on-import entry point
|
|
4
|
-
*
|
|
2
|
+
* `gemba-benchmark` CLI definition. It lives in `src/` so the bin stays an
|
|
3
|
+
* execute-on-import entry point. Launcher packages import the bin to run it.
|
|
4
|
+
* Tests import the definition and never run the CLI.
|
|
5
5
|
*/
|
|
6
6
|
|
|
7
7
|
import { runBenchmarkRunCommand } from "./benchmark-run.js";
|
|
@@ -13,7 +13,7 @@ import {
|
|
|
13
13
|
} from "@forwardimpact/libutil/models";
|
|
14
14
|
|
|
15
15
|
export const definition = {
|
|
16
|
-
name: "
|
|
16
|
+
name: "gemba-benchmark",
|
|
17
17
|
description:
|
|
18
18
|
"Run coding-agent task families, grade hidden tests, and aggregate pass@k across runs.",
|
|
19
19
|
commands: [
|
|
@@ -36,7 +36,7 @@ export const definition = {
|
|
|
36
36
|
"skills-from": {
|
|
37
37
|
type: "string",
|
|
38
38
|
description:
|
|
39
|
-
"Stage .claude/ from this directory (a root
|
|
39
|
+
"Stage .claude/ from this directory (a root that holds .claude/) instead of an apm install, to exercise local, unpublished skills",
|
|
40
40
|
},
|
|
41
41
|
output: {
|
|
42
42
|
type: "string",
|
|
@@ -85,7 +85,7 @@ export const definition = {
|
|
|
85
85
|
shard: {
|
|
86
86
|
type: "string",
|
|
87
87
|
description:
|
|
88
|
-
"Run only shard i of N as i/N (1-based; default: the whole family). Each shard writes a partial results.jsonl
|
|
88
|
+
"Run only shard i of N as i/N (1-based; default: the whole family). Each shard writes a partial results.jsonl. report --input merges them.",
|
|
89
89
|
},
|
|
90
90
|
"allowed-tools": {
|
|
91
91
|
type: "string",
|
|
@@ -99,7 +99,7 @@ export const definition = {
|
|
|
99
99
|
args: [],
|
|
100
100
|
handler: runBenchmarkGradeCommand,
|
|
101
101
|
description:
|
|
102
|
-
"Grade a single task against a post-run workdir
|
|
102
|
+
"Grade a single task against a post-run workdir with no agent. Run the hidden test suite and the invariants script. Then derive the verdict from the check rows (the exit mirrors it).",
|
|
103
103
|
options: {
|
|
104
104
|
family: {
|
|
105
105
|
type: "string",
|
|
@@ -112,7 +112,7 @@ export const definition = {
|
|
|
112
112
|
"run-dir": {
|
|
113
113
|
type: "string",
|
|
114
114
|
description:
|
|
115
|
-
"Post-run directory whose cwd/ subdir is the agent CWD
|
|
115
|
+
"Post-run directory whose cwd/ subdir is the agent CWD. Both producers run against that cwd. Hooks receive it as $AGENT_CWD",
|
|
116
116
|
},
|
|
117
117
|
output: {
|
|
118
118
|
type: "string",
|
|
@@ -125,12 +125,12 @@ export const definition = {
|
|
|
125
125
|
args: [],
|
|
126
126
|
handler: runBenchmarkReportCommand,
|
|
127
127
|
description:
|
|
128
|
-
"Aggregate result records into pass@k
|
|
128
|
+
"Aggregate result records into pass@k with the OpenAI HumanEval estimator.",
|
|
129
129
|
options: {
|
|
130
130
|
input: {
|
|
131
131
|
type: "string",
|
|
132
132
|
description:
|
|
133
|
-
"Run-output directory
|
|
133
|
+
"Run-output directory that holds results.jsonl (default: benchmark-runs)",
|
|
134
134
|
},
|
|
135
135
|
k: {
|
|
136
136
|
type: "string",
|
|
@@ -143,7 +143,7 @@ export const definition = {
|
|
|
143
143
|
detail: {
|
|
144
144
|
type: "string",
|
|
145
145
|
description:
|
|
146
|
-
"Text report verbosity (full|compact, default: full). compact omits per-task detail
|
|
146
|
+
"Text report verbosity (full|compact, default: full). compact omits per-task detail, which helps with sharded run summaries.",
|
|
147
147
|
},
|
|
148
148
|
},
|
|
149
149
|
},
|
|
@@ -154,14 +154,14 @@ export const definition = {
|
|
|
154
154
|
json: { type: "boolean", description: "Output help as JSON" },
|
|
155
155
|
},
|
|
156
156
|
examples: [
|
|
157
|
-
"
|
|
158
|
-
"
|
|
159
|
-
"
|
|
160
|
-
"
|
|
161
|
-
`
|
|
162
|
-
"
|
|
163
|
-
"
|
|
164
|
-
"
|
|
157
|
+
"gemba-benchmark run --family=./families/coding",
|
|
158
|
+
"gemba-benchmark run --family=./families/coding --task=todo-api --runs=1",
|
|
159
|
+
"gemba-benchmark run --family=./families/coding --work-tracker=filesystem",
|
|
160
|
+
"gemba-benchmark run --family=./families/coding --skills-from=. --task=todo-api",
|
|
161
|
+
`gemba-benchmark run --family=./families/coding --runs=10 --agent-model=${BENCHMARK_AGENT_MODEL}`,
|
|
162
|
+
"gemba-benchmark grade --family=./families/coding --task=todo-api --run-dir=./benchmark-runs/runs/todo-api/0",
|
|
163
|
+
"gemba-benchmark report --format=text",
|
|
164
|
+
"gemba-benchmark report --input=./runs/today --k=1,3,5 --format=text",
|
|
165
165
|
],
|
|
166
166
|
documentation: [
|
|
167
167
|
{
|
|
@@ -174,7 +174,7 @@ export const definition = {
|
|
|
174
174
|
title: "Automate with GitHub Actions",
|
|
175
175
|
url: "https://www.forwardimpact.team/docs/libraries/prove-changes/run-benchmark/ci-workflow/index.md",
|
|
176
176
|
description:
|
|
177
|
-
"Run benchmarks in CI with the forwardimpact/benchmark action.",
|
|
177
|
+
"Run benchmarks in CI with the forwardimpact/gemba-benchmark action.",
|
|
178
178
|
},
|
|
179
179
|
],
|
|
180
180
|
};
|
|
@@ -1,9 +1,9 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* `
|
|
3
|
-
* suite and the invariants script) against a post-run workdir directory
|
|
4
|
-
*
|
|
5
|
-
* No agent and no judge
|
|
6
|
-
* against fixtures
|
|
2
|
+
* `gemba-benchmark grade` — run both check-row producers (the hidden test
|
|
3
|
+
* suite and the invariants script) against a post-run workdir directory.
|
|
4
|
+
* Grade the merged rows with the same derivation the benchmark runner uses.
|
|
5
|
+
* No agent runs and no judge runs. So an author validates a task's grading
|
|
6
|
+
* material against fixtures, and pays for no agent session. The process exit
|
|
7
7
|
* mirrors the graded verdict.
|
|
8
8
|
*/
|
|
9
9
|
|
|
@@ -46,12 +46,12 @@ export async function runBenchmarkGradeCommand(ctx) {
|
|
|
46
46
|
runInvariants,
|
|
47
47
|
runHiddenTests,
|
|
48
48
|
});
|
|
49
|
-
//
|
|
50
|
-
// here)
|
|
51
|
-
// crashed hook can never mint marks from the rows it emitted before
|
|
52
|
-
//
|
|
53
|
-
// the
|
|
54
|
-
//
|
|
49
|
+
// The effective-score rule matches the runner, minus the judge (none runs
|
|
50
|
+
// here). An unhealthy grader or a gate that fails zeroes the score. So a
|
|
51
|
+
// crashed hook can never mint marks from the rows it emitted before it
|
|
52
|
+
// died. A runner record keeps the raw fraction in `grade.score` and zeroes
|
|
53
|
+
// the top-level `score` instead. This record has no second field, so
|
|
54
|
+
// `grade.score` carries the effective value here.
|
|
55
55
|
if (grade.score !== undefined && !(healthy && grade.gatesPass)) {
|
|
56
56
|
grade.score = 0;
|
|
57
57
|
}
|
|
@@ -65,7 +65,8 @@ export async function runBenchmarkGradeCommand(ctx) {
|
|
|
65
65
|
...(engineError && { error: engineError.message }),
|
|
66
66
|
},
|
|
67
67
|
}),
|
|
68
|
-
//
|
|
68
|
+
// This mirrors the script for diagnosis. The graded verdict drives the
|
|
69
|
+
// exit.
|
|
69
70
|
exitCode: invariants.exitCode,
|
|
70
71
|
};
|
|
71
72
|
validateGradeRecord(record);
|
|
@@ -1,9 +1,9 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* `
|
|
3
|
-
* OpenAI HumanEval estimator.
|
|
4
|
-
* to render a markdown table. --detail=compact drops the
|
|
5
|
-
* sections so a sharded run's per-shard summary stays short
|
|
6
|
-
* renders the full report over the combined ledger
|
|
2
|
+
* `gemba-benchmark report` — aggregate `results.jsonl` into pass@k with the
|
|
3
|
+
* OpenAI HumanEval estimator. The command writes JSON by default. Pass
|
|
4
|
+
* --format=text to render a markdown table. --detail=compact drops the
|
|
5
|
+
* per-task detail sections, so a sharded run's per-shard summary stays short.
|
|
6
|
+
* The merge job renders the full report over the combined ledger.
|
|
7
7
|
*/
|
|
8
8
|
|
|
9
9
|
import { resolve } from "node:path";
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* `
|
|
3
|
-
* ResultRecord to stdout (one JSON line per record)
|
|
4
|
-
* canonical `<output>/results.jsonl` for the report subcommand.
|
|
2
|
+
* `gemba-benchmark run` — run every task in a family for N runs. Stream each
|
|
3
|
+
* ResultRecord to stdout (one JSON line per record). Append each record to
|
|
4
|
+
* the canonical `<output>/results.jsonl` for the report subcommand.
|
|
5
5
|
*/
|
|
6
6
|
|
|
7
7
|
import { resolve } from "node:path";
|
|
@@ -30,15 +30,15 @@ export async function runBenchmarkRunCommand(ctx) {
|
|
|
30
30
|
}
|
|
31
31
|
const config = await createConfig("script", "benchmark");
|
|
32
32
|
runtime.proc.env.ANTHROPIC_API_KEY = await config.anthropicToken();
|
|
33
|
-
// The benchmark agent runs
|
|
34
|
-
// command
|
|
35
|
-
// spawns the subprocess that inherits process.env.
|
|
33
|
+
// The benchmark agent runs through createBenchmarkRunner. The supervise
|
|
34
|
+
// command does not run it. So the active-tracker env must land here before
|
|
35
|
+
// the runner spawns the subprocess that inherits process.env.
|
|
36
36
|
runtime.proc.env.LIBHARNESS_WORK_TRACKER = opts.workTracker;
|
|
37
37
|
|
|
38
38
|
// The Claude Agent SDK spawns a `claude` subprocess that inherits
|
|
39
|
-
// process.env. NODE_EXTRA_CA_CERTS
|
|
40
|
-
// inside that subprocess)
|
|
41
|
-
//
|
|
39
|
+
// process.env. NODE_EXTRA_CA_CERTS makes undici (the HTTP client
|
|
40
|
+
// inside that subprocess) fail with UND_ERR_INVALID_ARG on Node 22+.
|
|
41
|
+
// Undici then aborts every API call after 10 retries. Strip it
|
|
42
42
|
// before the SDK loads so the subprocess gets a clean environment.
|
|
43
43
|
delete runtime.proc.env.NODE_EXTRA_CA_CERTS;
|
|
44
44
|
|
|
@@ -57,13 +57,14 @@ export async function runBenchmarkRunCommand(ctx) {
|
|
|
57
57
|
}
|
|
58
58
|
|
|
59
59
|
/**
|
|
60
|
-
* Decide the exit outcome when a run streamed zero records. A run that emits
|
|
61
|
-
* records normally did nothing
|
|
62
|
-
* produced output
|
|
63
|
-
*
|
|
64
|
-
* high-index `--shard=i/N` with
|
|
65
|
-
* cells, so it exits 0 with a
|
|
66
|
-
* is
|
|
60
|
+
* Decide the exit outcome when a run streamed zero records. A run that emits
|
|
61
|
+
* no records normally did nothing. Either it discovered no tasks, or the
|
|
62
|
+
* agent never produced output. That is a failure. The command surfaces it
|
|
63
|
+
* loudly so CI does not go green on an empty benchmark. A deliberately-empty
|
|
64
|
+
* shard is the one exception. A high-index `--shard=i/N` with
|
|
65
|
+
* `N > cell count` legitimately selects zero cells, so it exits 0 with a
|
|
66
|
+
* stderr note. This function is exported so a test can reach the
|
|
67
|
+
* relaxed-guard branch without the full handler's config and SDK setup.
|
|
67
68
|
* @param {{shard: {index: number, total: number} | null}} opts
|
|
68
69
|
* @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
|
|
69
70
|
* @returns {{ok: true} | {ok: false, code: number, error: string}}
|
|
@@ -79,16 +80,17 @@ export function resolveZeroRecordOutcome(opts, runtime) {
|
|
|
79
80
|
ok: false,
|
|
80
81
|
code: 1,
|
|
81
82
|
error:
|
|
82
|
-
"benchmark produced no result records
|
|
83
|
+
"benchmark produced no result records. No task ran to completion. Check the family's tasks/, apm install, and agent availability (ANTHROPIC_API_KEY / claude CLI / IS_SANDBOX)",
|
|
83
84
|
};
|
|
84
85
|
}
|
|
85
86
|
|
|
86
87
|
/**
|
|
87
|
-
* Parse and validate benchmark run options.
|
|
88
|
-
* defaults
|
|
88
|
+
* Parse and validate benchmark run options. This function is exported so a
|
|
89
|
+
* test can verify the defaults and the resolved work tracker.
|
|
89
90
|
* @param {Record<string, string|undefined>} values - Parsed option values
|
|
90
|
-
* @param {Record<string, string|undefined>} [env] - Process environment
|
|
91
|
-
* for the `LIBHARNESS_WORK_TRACKER` fallback when
|
|
91
|
+
* @param {Record<string, string|undefined>} [env] - Process environment. The
|
|
92
|
+
* parser reads it for the `LIBHARNESS_WORK_TRACKER` fallback when
|
|
93
|
+
* `--work-tracker` is absent.
|
|
92
94
|
* @returns {object}
|
|
93
95
|
*/
|
|
94
96
|
export function parseRunOptions(values, env = {}) {
|
|
@@ -132,7 +134,8 @@ function parseMaxTurns(raw) {
|
|
|
132
134
|
|
|
133
135
|
/**
|
|
134
136
|
* Parse a `--shard=<i>/<N>` selector into `{index, total}` (1-based), or `null`
|
|
135
|
-
* for an unsharded run.
|
|
137
|
+
* for an unsharded run. Validate that the parts are integers and that
|
|
138
|
+
* `1 ≤ index ≤ total`.
|
|
136
139
|
* @param {string|undefined} raw
|
|
137
140
|
* @returns {{index: number, total: number} | null}
|
|
138
141
|
*/
|
|
@@ -147,17 +150,17 @@ export function parseShard(raw) {
|
|
|
147
150
|
return { index, total };
|
|
148
151
|
}
|
|
149
152
|
|
|
150
|
-
//
|
|
151
|
-
// agent-under-test + judge)
|
|
152
|
-
//
|
|
153
|
-
// machines
|
|
153
|
+
// The ceiling is conservative because each cell spawns ~3 agent subprocesses
|
|
154
|
+
// (lead + agent-under-test + judge). A low ceiling makes sure that a single
|
|
155
|
+
// runner does not thrash. Most of the CI speedup comes from Layer-2 shards
|
|
156
|
+
// across machines. It does not come from a higher in-job default.
|
|
154
157
|
const CONCURRENCY_CEILING = 4;
|
|
155
158
|
|
|
156
159
|
/**
|
|
157
160
|
* Resolve the cell concurrency: `--concurrency` flag > the
|
|
158
161
|
* `LIBHARNESS_BENCHMARK_CONCURRENCY` env var > a CPU-aware default of
|
|
159
|
-
* `min(CONCURRENCY_CEILING, max(2, ⌊cores/2⌋))`. The default is `> 1
|
|
160
|
-
* concurrency is on transparently
|
|
162
|
+
* `min(CONCURRENCY_CEILING, max(2, ⌊cores/2⌋))`. The default is `> 1`, so
|
|
163
|
+
* concurrency is on transparently. No consumer needs to opt in.
|
|
161
164
|
* @param {Record<string, string|undefined>} values
|
|
162
165
|
* @param {Record<string, string|undefined>} [env]
|
|
163
166
|
* @returns {number}
|
|
@@ -4,11 +4,11 @@ const FIRST_LINE_CAP = 64 * 1024;
|
|
|
4
4
|
|
|
5
5
|
/**
|
|
6
6
|
* Read the first newline-terminated line of a file, bounded to the first
|
|
7
|
-
* {@link FIRST_LINE_CAP} bytes. Trace `.ndjson` files can be many MB
|
|
8
|
-
* Step 2.6 meta header is always small
|
|
9
|
-
*
|
|
10
|
-
* `openSync`/`readSync`/`closeSync` trio
|
|
11
|
-
* `runtime.fsSync` surface.
|
|
7
|
+
* {@link FIRST_LINE_CAP} bytes. Trace `.ndjson` files can be many MB. The
|
|
8
|
+
* Step 2.6 meta header is always small. So a bounded positional read does
|
|
9
|
+
* not load whole files into memory just to inspect the header. This function
|
|
10
|
+
* reads the positional `openSync`/`readSync`/`closeSync` trio off the
|
|
11
|
+
* injected `runtime.fsSync` surface.
|
|
12
12
|
*
|
|
13
13
|
* @param {object} fsSync - Sync filesystem surface (`runtime.fsSync`).
|
|
14
14
|
* @param {string} path
|
|
@@ -30,9 +30,9 @@ function readFirstLine(fsSync, path) {
|
|
|
30
30
|
/**
|
|
31
31
|
* Scan a directory for `.ndjson` files whose meta header carries the
|
|
32
32
|
* given discussion_id. The Step 2.6 first-line guarantee makes the
|
|
33
|
-
* lookup cheap
|
|
34
|
-
* meta header (e.g. legacy
|
|
35
|
-
*
|
|
33
|
+
* lookup cheap. This function reads only the first line per file. It
|
|
34
|
+
* silently skips files without a meta header (e.g. legacy
|
|
35
|
+
* supervise/facilitate traces). Such a file is not an error.
|
|
36
36
|
*
|
|
37
37
|
* @param {string} dir
|
|
38
38
|
* @param {string} discussionId
|
|
@@ -72,10 +72,10 @@ export function findTracesByDiscussion(dir, discussionId, fsSync) {
|
|
|
72
72
|
}
|
|
73
73
|
|
|
74
74
|
/**
|
|
75
|
-
* `
|
|
75
|
+
* `gemba-trace by-discussion <discussion-id> [trace-dir]` — list trace
|
|
76
76
|
* files whose meta header carries the given discussion_id, one per
|
|
77
|
-
* line
|
|
78
|
-
*
|
|
77
|
+
* line. Order them by first-event timestamp (file mtime ascending).
|
|
78
|
+
* You can use the result with `xargs cat` for a chronological merge.
|
|
79
79
|
*
|
|
80
80
|
* @param {import("@forwardimpact/libcli").InvocationContext} ctx
|
|
81
81
|
* @returns {Promise<{ok: true} | {ok: false, code: number, error: string}>}
|
package/src/commands/callback.js
CHANGED
|
@@ -3,12 +3,12 @@ import { sumTraceCost } from "../cost.js";
|
|
|
3
3
|
/**
|
|
4
4
|
* Scan an NDJSON trace and return the last orchestrator summary event,
|
|
5
5
|
* the first `meta` event's `discussion_id`, and any structured replies
|
|
6
|
-
*
|
|
6
|
+
* the discusser collected. This function skips malformed lines.
|
|
7
7
|
*
|
|
8
|
-
* The runner is verdict-agnostic
|
|
9
|
-
*
|
|
10
|
-
* "adjourned"/"recessed"/"failed" from discuss). The bridge
|
|
11
|
-
* its channel semantics.
|
|
8
|
+
* The runner is verdict-agnostic. It passes through whatever the trace
|
|
9
|
+
* carries, verbatim ("success"/"failure" from supervise/facilitate;
|
|
10
|
+
* canonical "adjourned"/"recessed"/"failed" from discuss). The bridge
|
|
11
|
+
* layer maps to its channel semantics.
|
|
12
12
|
*
|
|
13
13
|
* @param {string} content - Raw NDJSON trace content.
|
|
14
14
|
* @returns {{verdict: string, summary: string, replies: object[], trigger?: object, discussionId?: string} | null}
|
|
@@ -53,9 +53,9 @@ function readTraceSummary(content) {
|
|
|
53
53
|
}
|
|
54
54
|
|
|
55
55
|
/**
|
|
56
|
-
* Callback command — read an NDJSON trace
|
|
57
|
-
* orchestrator summary
|
|
58
|
-
*
|
|
56
|
+
* Callback command — read an NDJSON trace and extract the terminal
|
|
57
|
+
* orchestrator summary. POST a canonical callback body to the configured
|
|
58
|
+
* URL. `kata-dispatch.yml` uses this command to deliver the lead's
|
|
59
59
|
* conclusion to the bridge that dispatched the run.
|
|
60
60
|
*
|
|
61
61
|
* Wire shape (single shape across modes):
|
|
@@ -87,11 +87,11 @@ export async function runCallbackCommand(ctx) {
|
|
|
87
87
|
const content = runtime.fsSync.readFileSync(traceFile, "utf8");
|
|
88
88
|
const found = readTraceSummary(content) ?? {
|
|
89
89
|
verdict: "failed",
|
|
90
|
-
summary: "
|
|
90
|
+
summary: "The run ended and produced no summary.",
|
|
91
91
|
replies: [],
|
|
92
92
|
};
|
|
93
|
-
// Total spend across every participant in the trace
|
|
94
|
-
// it alongside the verdict so a dispatched run reports what it cost.
|
|
93
|
+
// Total spend across every participant in the trace. The bridge surfaces
|
|
94
|
+
// it alongside the verdict, so a dispatched run reports what it cost.
|
|
95
95
|
const { totalCostUsd } = sumTraceCost(content.split("\n"));
|
|
96
96
|
|
|
97
97
|
const discussionId = found.discussionId ?? discussionIdOverride ?? null;
|