@forwardimpact/libharness 2.0.0 → 3.0.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +68 -65
- package/package.json +15 -13
- package/src/advisor.js +47 -41
- package/src/agent-runner.js +58 -48
- package/src/benchmark/apm-installer.js +28 -28
- package/src/benchmark/env-loader.js +24 -16
- package/src/benchmark/grade.js +44 -41
- package/src/benchmark/hidden-tests.js +25 -24
- package/src/benchmark/hook-env.js +11 -9
- package/src/benchmark/invariants.js +20 -17
- package/src/benchmark/judge.js +29 -28
- package/src/benchmark/npm-installer.js +9 -8
- package/src/benchmark/report.js +53 -50
- package/src/benchmark/result.js +24 -23
- package/src/benchmark/runner.js +75 -69
- package/src/benchmark/scheduler.js +17 -16
- package/src/benchmark/task-family.js +29 -27
- package/src/benchmark/trace-split.js +9 -8
- package/src/benchmark/workdir.js +27 -25
- package/src/claude-code-executable.js +11 -11
- package/src/commands/advisor-flags.js +8 -7
- package/src/commands/assert.js +16 -15
- package/src/commands/benchmark-definition.js +20 -20
- package/src/commands/benchmark-grade.js +13 -12
- package/src/commands/benchmark-report.js +5 -5
- package/src/commands/benchmark-run.js +31 -28
- package/src/commands/by-discussion.js +11 -11
- package/src/commands/callback.js +11 -11
- package/src/commands/discuss.js +8 -7
- package/src/commands/facilitate.js +16 -14
- package/src/commands/output.js +4 -3
- package/src/commands/run.js +15 -15
- package/src/commands/scan-logs.js +22 -20
- package/src/commands/selfedit.js +124 -0
- package/src/commands/supervise.js +13 -11
- package/src/commands/task-input.js +9 -9
- package/src/commands/tee.js +11 -10
- package/src/commands/trace.js +55 -42
- package/src/commands/work-tracker.js +4 -3
- package/src/cost.js +17 -17
- package/src/discuss-tools.js +16 -16
- package/src/discusser.js +39 -38
- package/src/events/github.js +54 -37
- package/src/facilitator.js +21 -21
- package/src/inbox-poller.js +4 -4
- package/src/judge.js +32 -30
- package/src/message-bus.js +12 -11
- package/src/orchestration-loop.js +35 -36
- package/src/orchestration-toolkit.js +58 -53
- package/src/orchestrator-helpers.js +2 -2
- package/src/profile-prompt.js +54 -53
- package/src/redaction.js +63 -57
- package/src/render/line-renderer.js +5 -5
- package/src/render/orchestrator-filter.js +3 -3
- package/src/render/palette.js +11 -9
- package/src/render/tool-hints.js +18 -15
- package/src/render/turn-renderer.js +4 -4
- package/src/reply-emitter.js +2 -2
- package/src/sequence-counter.js +4 -3
- package/src/signature-filter.js +7 -6
- package/src/supervisor.js +19 -18
- package/src/tee-writer.js +25 -25
- package/src/trace-collector.js +53 -48
- package/src/trace-github.js +53 -44
- package/src/trace-multi.js +16 -14
- package/src/trace-query.js +61 -52
- package/src/trace-render.js +19 -19
- package/src/trace-usage.js +31 -28
- package/src/transcript-recorder.js +24 -20
- package/bin/fit-benchmark.js +0 -44
- package/bin/fit-harness.js +0 -412
- package/bin/fit-selfedit.js +0 -165
- package/bin/fit-trace.js +0 -520
|
@@ -1,14 +1,14 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* Hidden-test engine —
|
|
3
|
-
*
|
|
4
|
-
* with `node --test
|
|
5
|
-
*
|
|
6
|
-
* it.
|
|
2
|
+
* Hidden-test engine — runs a task's `tests/` overlay against the post-run
|
|
3
|
+
* agent CWD. The engine stages each file at its mirrored path. It runs each
|
|
4
|
+
* check with `node --test`. It converts the exit status into one check row.
|
|
5
|
+
* It restores the tree, so the judge sees the workdir exactly as the agent
|
|
6
|
+
* left it.
|
|
7
7
|
*
|
|
8
|
-
* Fault attribution is the engine's contract
|
|
9
|
-
* agent deleted the scaffold) is
|
|
10
|
-
*
|
|
11
|
-
* a crashed grader can never mint marks.
|
|
8
|
+
* Fault attribution is the engine's contract. A stage or spawn failure (the
|
|
9
|
+
* agent deleted the scaffold) is agent fault, so the engine returns a *row
|
|
10
|
+
* that fails*. A throw from the engine itself is grader fault. The caller
|
|
11
|
+
* records that as unhealthy, so a crashed grader can never mint marks.
|
|
12
12
|
*/
|
|
13
13
|
|
|
14
14
|
import { dirname, join } from "node:path";
|
|
@@ -16,8 +16,8 @@ import { dirname, join } from "node:path";
|
|
|
16
16
|
import { buildHookEnv } from "./hook-env.js";
|
|
17
17
|
|
|
18
18
|
// Fixed per-check budget. A wedged test process runs outside the agent
|
|
19
|
-
// watchdog
|
|
20
|
-
//
|
|
19
|
+
// watchdog. Without this bound, a hung hidden test would stall the cell. The
|
|
20
|
+
// timeout row keeps the failure visible.
|
|
21
21
|
const CHECK_TIMEOUT_MS = 120_000;
|
|
22
22
|
const STDERR_TAIL_CHARS = 500;
|
|
23
23
|
|
|
@@ -50,9 +50,9 @@ export async function runHiddenTests(task, ctx, runtime, opts = {}) {
|
|
|
50
50
|
}
|
|
51
51
|
|
|
52
52
|
/**
|
|
53
|
-
* Stage one check, run it,
|
|
54
|
-
*
|
|
55
|
-
* scaffold)
|
|
53
|
+
* Stage one check, run it, then put the tree back. The check's own row is the
|
|
54
|
+
* only trace it leaves. A stage failure is the agent's fault (a deleted
|
|
55
|
+
* scaffold). The engine returns a row that fails. It does not throw.
|
|
56
56
|
*/
|
|
57
57
|
async function runOneCheck(task, ctx, runtime, timeoutMs, check) {
|
|
58
58
|
const stager = newStager();
|
|
@@ -71,7 +71,8 @@ async function runOneCheck(task, ctx, runtime, timeoutMs, check) {
|
|
|
71
71
|
/**
|
|
72
72
|
* Spawn `node --test <staged path>` from the agent CWD under the hook env
|
|
73
73
|
* and map the exit status onto one row. The clock timer SIGKILLs a child
|
|
74
|
-
* that outlives the per-check budget
|
|
74
|
+
* that outlives the per-check budget. The row then fails with a timeout
|
|
75
|
+
* message.
|
|
75
76
|
*/
|
|
76
77
|
async function spawnCheck(task, ctx, runtime, timeoutMs, check) {
|
|
77
78
|
const env = buildHookEnv(runtime.proc.env, {
|
|
@@ -83,8 +84,8 @@ async function spawnCheck(task, ctx, runtime, timeoutMs, check) {
|
|
|
83
84
|
familyDir: ctx.familyDir,
|
|
84
85
|
});
|
|
85
86
|
// An inherited test-runner context makes the child `node --test` report
|
|
86
|
-
// exit 0 even when its tests fail
|
|
87
|
-
//
|
|
87
|
+
// exit 0 even when its tests fail. A check that fails would then mint a row
|
|
88
|
+
// that passes whenever the harness itself runs under `node --test`.
|
|
88
89
|
delete env.NODE_TEST_CONTEXT;
|
|
89
90
|
const child = runtime.subprocess.spawn("node", ["--test", check.stagePath], {
|
|
90
91
|
cwd: ctx.cwd,
|
|
@@ -129,8 +130,8 @@ function newStager() {
|
|
|
129
130
|
}
|
|
130
131
|
|
|
131
132
|
/**
|
|
132
|
-
* Copy the symlink-resolved source to its mirrored path under the agent CWD
|
|
133
|
-
*
|
|
133
|
+
* Copy the symlink-resolved source to its mirrored path under the agent CWD.
|
|
134
|
+
* Back up the bytes of a collided file. Track every directory created, so
|
|
134
135
|
* `unstage` can put the tree back exactly.
|
|
135
136
|
*/
|
|
136
137
|
async function stageFile(fs, cwd, stager, { sourcePath, stagePath }) {
|
|
@@ -154,7 +155,7 @@ async function ensureParents(fs, cwd, stager, dir) {
|
|
|
154
155
|
await fs.access(dir);
|
|
155
156
|
return;
|
|
156
157
|
} catch {
|
|
157
|
-
// missing
|
|
158
|
+
// missing, so create it below
|
|
158
159
|
}
|
|
159
160
|
await ensureParents(fs, cwd, stager, dirname(dir));
|
|
160
161
|
await fs.mkdir(dir);
|
|
@@ -162,10 +163,10 @@ async function ensureParents(fs, cwd, stager, dir) {
|
|
|
162
163
|
}
|
|
163
164
|
|
|
164
165
|
/**
|
|
165
|
-
* Reverse the
|
|
166
|
-
* directories
|
|
167
|
-
* created directory go with it,
|
|
168
|
-
* agent finished
|
|
166
|
+
* Reverse the stage step. Remove the staged copies. Write the collided bytes
|
|
167
|
+
* back. Remove the created directories, deepest first. A check's own
|
|
168
|
+
* artifacts inside a created directory go with it, because that directory did
|
|
169
|
+
* not exist when the agent finished.
|
|
169
170
|
*/
|
|
170
171
|
async function unstage(fs, stager) {
|
|
171
172
|
for (const target of stager.staged) {
|
|
@@ -1,12 +1,13 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* Shared environment builder for the benchmark hook scripts (`preflight.sh`
|
|
3
|
-
* `invariants.sh`).
|
|
4
|
-
* same variable set
|
|
5
|
-
*
|
|
2
|
+
* Shared environment builder for the benchmark hook scripts (`preflight.sh`
|
|
3
|
+
* and `invariants.sh`). One helper serves both spawns, so both expose the
|
|
4
|
+
* same variable set. Hook authors never have to wonder which vars a hook
|
|
5
|
+
* receives.
|
|
6
6
|
*
|
|
7
7
|
* Path vars (TASK_DIR, FAMILY_DIR, HOOKS_DIR) let hooks reference real
|
|
8
|
-
* locations
|
|
9
|
-
* secrets, so they need no
|
|
8
|
+
* locations. Hooks do not have to rebuild the locations from `$0`. They are
|
|
9
|
+
* paths. They are not secrets, so they need no entry in the redaction
|
|
10
|
+
* allowlist.
|
|
10
11
|
*/
|
|
11
12
|
|
|
12
13
|
/**
|
|
@@ -27,9 +28,10 @@ export function buildHookEnv(
|
|
|
27
28
|
) {
|
|
28
29
|
return {
|
|
29
30
|
...baseEnv,
|
|
30
|
-
// The agent CWD itself
|
|
31
|
-
//
|
|
32
|
-
// *contains* `cwd/`), so the two are never
|
|
31
|
+
// The agent CWD itself. Hooks reference emitted files as
|
|
32
|
+
// `$AGENT_CWD/<path>`. This var is distinct from the `invariants` CLI's
|
|
33
|
+
// `--run-dir` (the parent that *contains* `cwd/`), so the two are never
|
|
34
|
+
// confused.
|
|
33
35
|
AGENT_CWD: cwd,
|
|
34
36
|
PORT: String(port),
|
|
35
37
|
TASK_ID: taskId,
|
|
@@ -1,14 +1,15 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* Invariants — runs `<task.paths.hooks>/invariants.sh` from the template path
|
|
3
|
-
* against the post-run agent CWD.
|
|
4
|
-
* own
|
|
5
|
-
*
|
|
6
|
-
* script health only
|
|
7
|
-
* check failed.
|
|
3
|
+
* against the post-run agent CWD. The module is a pure collector with no
|
|
4
|
+
* verdict of its own. Structured per-check rows arrive on fd 3
|
|
5
|
+
* (`$RESULTS_FD=3`) as NDJSON. Downstream code grades the merged rows. The
|
|
6
|
+
* exit code is script health only. A nonzero code means the grader itself
|
|
7
|
+
* failed. It never means that a check failed.
|
|
8
8
|
*
|
|
9
|
-
* Subprocess access flows through `runtime.subprocess.spawn
|
|
10
|
-
* store and the stderr log use the sync filesystem surface
|
|
11
|
-
* the only
|
|
9
|
+
* Subprocess access flows through `runtime.subprocess.spawn`. The fd-3
|
|
10
|
+
* backing store and the stderr log use the sync filesystem surface
|
|
11
|
+
* (`runtime.fsSync`). That surface is the only one this module touches, per
|
|
12
|
+
* design Decision 7.
|
|
12
13
|
*/
|
|
13
14
|
|
|
14
15
|
import { join } from "node:path";
|
|
@@ -18,11 +19,12 @@ import { buildHookEnv } from "./hook-env.js";
|
|
|
18
19
|
/**
|
|
19
20
|
* @typedef {object} InvariantsResult
|
|
20
21
|
* @property {Array<object>} details
|
|
21
|
-
* @property {number} exitCode - Script health
|
|
22
|
-
* failed
|
|
22
|
+
* @property {number} exitCode - Script health. A nonzero code means the hook
|
|
23
|
+
* itself failed. It never means that a check failed.
|
|
23
24
|
* @property {string} [stderr] - Trimmed script stderr, present only when the
|
|
24
|
-
* script wrote to stderr.
|
|
25
|
-
* leave `details` empty
|
|
25
|
+
* script wrote to stderr. This field surfaces hook failures (e.g. a missing
|
|
26
|
+
* tool) that leave `details` empty. A reader can then tell them apart from
|
|
27
|
+
* a real invariant miss.
|
|
26
28
|
*/
|
|
27
29
|
|
|
28
30
|
/**
|
|
@@ -43,8 +45,8 @@ export async function runInvariants(task, ctx, runtime) {
|
|
|
43
45
|
|
|
44
46
|
// Bun's child_process pipe setup for fd >= 3 is racy under load (it
|
|
45
47
|
// creates a unix socket pair and the connect() can return ENOENT). Use
|
|
46
|
-
// a temp file as the fd-3 backing store instead
|
|
47
|
-
//
|
|
48
|
+
// a temp file as the fd-3 backing store instead. The script still writes
|
|
49
|
+
// through `$RESULTS_FD`, but we hand it a real file descriptor.
|
|
48
50
|
const fd3Path = join(ctx.runDir, "invariants.fd3.ndjson");
|
|
49
51
|
const fd3File = fsSync.openSync(fd3Path, "w+");
|
|
50
52
|
|
|
@@ -69,7 +71,8 @@ export async function runInvariants(task, ctx, runtime) {
|
|
|
69
71
|
throw e;
|
|
70
72
|
}
|
|
71
73
|
|
|
72
|
-
// Drain stdout (do not require consumers to read it)
|
|
74
|
+
// Drain stdout (do not require consumers to read it). Capture stderr to
|
|
75
|
+
// the log.
|
|
73
76
|
const drainStdout = (async () => {
|
|
74
77
|
for await (const _chunk of child.stdout) {
|
|
75
78
|
// discard
|
|
@@ -127,8 +130,8 @@ function readAndUnlink(fsSync, path) {
|
|
|
127
130
|
}
|
|
128
131
|
|
|
129
132
|
/**
|
|
130
|
-
* Parse the fd-3 buffer (read from the temp-file backing) into one
|
|
131
|
-
* row per detail entry.
|
|
133
|
+
* Parse the fd-3 buffer (read from the temp-file backing store) into one
|
|
134
|
+
* NDJSON row per detail entry.
|
|
132
135
|
*/
|
|
133
136
|
function parseFd3Buffer(buf, details) {
|
|
134
137
|
if (!buf) return;
|
package/src/benchmark/judge.js
CHANGED
|
@@ -1,8 +1,9 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* Benchmark adapter for the libharness `Judge`.
|
|
3
|
-
* `judge.task.md` with structured context variables
|
|
4
|
-
* the post-run agent CWD
|
|
5
|
-
* `pass`/`fail` vocabulary (mapped from libharness's
|
|
2
|
+
* Benchmark adapter for the libharness `Judge`. The adapter fills the
|
|
3
|
+
* family's `judge.task.md` template with structured context variables. It
|
|
4
|
+
* runs the judge against the post-run agent CWD. It returns the verdict in
|
|
5
|
+
* the benchmark's `pass`/`fail` vocabulary (mapped from libharness's
|
|
6
|
+
* `success`/`failure`).
|
|
6
7
|
*
|
|
7
8
|
* Template variables available in `judge.task.md`:
|
|
8
9
|
*
|
|
@@ -12,13 +13,13 @@
|
|
|
12
13
|
* {{GRADE_RESULT}} — JSON grade object plus the merged check rows
|
|
13
14
|
* {{SKILL_SET_HASH}} — SHA-256 from apm.lock.yaml
|
|
14
15
|
* {{TASK_ID}} — task name (directory under tasks/)
|
|
15
|
-
* {{TASK_DIR}} — agent working directory
|
|
16
|
+
* {{TASK_DIR}} — path to the agent working directory
|
|
16
17
|
*
|
|
17
|
-
* The judge verdict
|
|
18
|
-
* `concluded` flag
|
|
19
|
-
* `parseConcludeFromTrace`
|
|
20
|
-
*
|
|
21
|
-
* historical run from its judge.ndjson file
|
|
18
|
+
* The adapter reads the judge verdict directly from the orchestration
|
|
19
|
+
* context's `concluded` flag. It does not parse the trace on the happy path.
|
|
20
|
+
* `parseConcludeFromTrace` stays for offline analysis. It is also a fallback
|
|
21
|
+
* when the runtime ctx is not available, for example when you re-grade a
|
|
22
|
+
* historical run from its judge.ndjson file.
|
|
22
23
|
*/
|
|
23
24
|
|
|
24
25
|
import { createJudge } from "../judge.js";
|
|
@@ -41,8 +42,8 @@ import { sumTraceCost } from "../cost.js";
|
|
|
41
42
|
|
|
42
43
|
/**
|
|
43
44
|
* Run the judge over a completed task run. The judge is a binary gate over
|
|
44
|
-
* the grade's validity
|
|
45
|
-
* template as evidence
|
|
45
|
+
* the grade's validity. The judge is never a grade itself. `gradeResult`
|
|
46
|
+
* reaches the template as evidence. The verdict stays pass/fail.
|
|
46
47
|
* @param {import("./task-family.js").Task} task
|
|
47
48
|
* @param {import("./workdir.js").Workdir} workdir
|
|
48
49
|
* @param {{verdict: string, gatesPass: boolean, score?: number, malformed?: number, rows: unknown[]}} gradeResult -
|
|
@@ -86,9 +87,9 @@ export async function runJudge(task, workdir, gradeResult, deps, context) {
|
|
|
86
87
|
await new Promise((r) => output.end(r));
|
|
87
88
|
}
|
|
88
89
|
|
|
89
|
-
// The judge is its own SDK session
|
|
90
|
-
// just wrote
|
|
91
|
-
// benchmark record's cost includes the judge.
|
|
90
|
+
// The judge is its own SDK session. Its spend lands in the judge trace we
|
|
91
|
+
// just wrote. It does not land in the supervisor's combined trace. Read the
|
|
92
|
+
// trace back, so the benchmark record's cost includes the judge.
|
|
92
93
|
const judgeTrace = await fs
|
|
93
94
|
.readFile(workdir.judgeTracePath, "utf8")
|
|
94
95
|
.catch(() => "");
|
|
@@ -109,10 +110,11 @@ export async function runJudge(task, workdir, gradeResult, deps, context) {
|
|
|
109
110
|
}
|
|
110
111
|
|
|
111
112
|
/**
|
|
112
|
-
* Parse the last
|
|
113
|
-
*
|
|
114
|
-
*
|
|
115
|
-
*
|
|
113
|
+
* Parse the last `Conclude` tool call from an NDJSON trace and map the
|
|
114
|
+
* verdict (`success → pass`, `failure → fail`). The call may carry a
|
|
115
|
+
* judge source or a supervisor source. The supervisor source keeps backward
|
|
116
|
+
* compatibility with pre-Judge-class traces. This function stays for offline
|
|
117
|
+
* analysis. The runtime happy path does not use it.
|
|
116
118
|
* @param {string} tracePath
|
|
117
119
|
* @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
|
|
118
120
|
* @returns {Promise<JudgeVerdict | null>}
|
|
@@ -133,9 +135,9 @@ export async function parseConcludeFromTrace(tracePath, runtime) {
|
|
|
133
135
|
}
|
|
134
136
|
|
|
135
137
|
/**
|
|
136
|
-
* Return the `Conclude` tool input
|
|
137
|
-
* supervisor-source assistant message
|
|
138
|
-
* block
|
|
138
|
+
* Return the `Conclude` tool input when the line carries a judge-source or
|
|
139
|
+
* supervisor-source assistant message that ends in a `Conclude` tool_use
|
|
140
|
+
* block. Return null otherwise.
|
|
139
141
|
* @param {string} line
|
|
140
142
|
* @returns {{verdict: string, summary?: string} | null}
|
|
141
143
|
*/
|
|
@@ -176,12 +178,11 @@ function extractConcludeInput(line) {
|
|
|
176
178
|
}
|
|
177
179
|
|
|
178
180
|
/**
|
|
179
|
-
* The Claude Agent SDK reports MCP tool names as
|
|
180
|
-
*
|
|
181
|
-
* `
|
|
182
|
-
*
|
|
183
|
-
*
|
|
184
|
-
* trace source.
|
|
181
|
+
* The Claude Agent SDK reports MCP tool names as `mcp__<server>__<tool>` when
|
|
182
|
+
* the model invokes them. The orchestration `Conclude` arrives as
|
|
183
|
+
* `mcp__orchestration__Conclude`. Pre-baked supervisor traces (and the
|
|
184
|
+
* libharness-internal envelopes) sometimes carry the bare `Conclude` name.
|
|
185
|
+
* Accept both forms, so the parser is robust to the trace source.
|
|
185
186
|
*/
|
|
186
187
|
function isConcludeToolName(name) {
|
|
187
188
|
if (typeof name !== "string") return false;
|
|
@@ -1,10 +1,11 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* NpmInstaller — runs `bun install` in the family root when a package.json
|
|
3
|
-
*
|
|
4
|
-
* directory so WorkdirManager can seed each per-task CWD.
|
|
2
|
+
* NpmInstaller — runs `bun install` in the family root when a package.json is
|
|
3
|
+
* present. It then copies the `node_modules/` that `bun install` produced
|
|
4
|
+
* into the staging directory, so WorkdirManager can seed each per-task CWD.
|
|
5
5
|
*
|
|
6
|
-
*
|
|
7
|
-
*
|
|
6
|
+
* NpmInstaller is symmetric to ApmInstaller. The subprocess and the
|
|
7
|
+
* filesystem flow through the injected `runtime` bag
|
|
8
|
+
* (`runtime.subprocess.spawn` + `runtime.fs`).
|
|
8
9
|
*/
|
|
9
10
|
|
|
10
11
|
import { join } from "node:path";
|
|
@@ -14,7 +15,7 @@ export class NpmInstaller {
|
|
|
14
15
|
/**
|
|
15
16
|
* @param {object} deps
|
|
16
17
|
* @param {import("@forwardimpact/libutil/runtime").Runtime} deps.runtime -
|
|
17
|
-
* Ambient collaborators
|
|
18
|
+
* Ambient collaborators. This class uses `subprocess.spawn` and `fs`.
|
|
18
19
|
*/
|
|
19
20
|
constructor({ runtime }) {
|
|
20
21
|
if (!runtime) throw new Error("runtime is required");
|
|
@@ -23,7 +24,7 @@ export class NpmInstaller {
|
|
|
23
24
|
|
|
24
25
|
/**
|
|
25
26
|
* @param {import("./task-family.js").TaskFamily} family
|
|
26
|
-
* @param {string} stagingDir - The staging directory
|
|
27
|
+
* @param {string} stagingDir - The staging directory ApmInstaller creates.
|
|
27
28
|
* @returns {Promise<void>}
|
|
28
29
|
*/
|
|
29
30
|
async install(family, stagingDir) {
|
|
@@ -42,7 +43,7 @@ export class NpmInstaller {
|
|
|
42
43
|
await fs.access(sourceModules);
|
|
43
44
|
} catch {
|
|
44
45
|
throw new Error(
|
|
45
|
-
`bun install did not produce node_modules/ at ${sourceModules}
|
|
46
|
+
`bun install did not produce node_modules/ at ${sourceModules}. Check the family's package.json`,
|
|
46
47
|
);
|
|
47
48
|
}
|
|
48
49
|
|
package/src/benchmark/report.js
CHANGED
|
@@ -1,15 +1,15 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* ReportAggregator — read a run-output directory's `results.jsonl
|
|
3
|
-
* records by `taskId
|
|
2
|
+
* ReportAggregator — read a run-output directory's `results.jsonl`. Group
|
|
3
|
+
* the records by `taskId`. Compute pass@k with the OpenAI HumanEval
|
|
4
4
|
* unbiased estimator: `1 - C(n-c, k) / C(n, k)`.
|
|
5
5
|
*
|
|
6
6
|
* When `includeRuns` is true, each task carries per-run detail (invariant
|
|
7
|
-
* checks, judge commentary, cost, duration)
|
|
7
|
+
* checks, judge commentary, cost, duration). The text renderer then produces
|
|
8
8
|
* a full markdown report instead of just the pass@k table.
|
|
9
9
|
*
|
|
10
|
-
*
|
|
11
|
-
*
|
|
12
|
-
* whole report.
|
|
10
|
+
* The loader skips records that fail schema validation. It writes a stderr
|
|
11
|
+
* warning and counts each skip under `totals.skipped`. A corrupt line cannot
|
|
12
|
+
* abort the whole report.
|
|
13
13
|
*/
|
|
14
14
|
|
|
15
15
|
import { join } from "node:path";
|
|
@@ -37,7 +37,7 @@ import { mergeRows } from "./grade.js";
|
|
|
37
37
|
* @typedef {object} TaskReport
|
|
38
38
|
* @property {string} taskId
|
|
39
39
|
* @property {number} n - Total runs.
|
|
40
|
-
* @property {number} c -
|
|
40
|
+
* @property {number} c - Runs that passed.
|
|
41
41
|
* @property {Record<string|number, number|null>} passAtK
|
|
42
42
|
* @property {RunDetail[]} [runs] - Per-run detail (only when includeRuns).
|
|
43
43
|
*/
|
|
@@ -109,12 +109,12 @@ export async function aggregate({
|
|
|
109
109
|
|
|
110
110
|
/**
|
|
111
111
|
* Attach `meanScore` and `scoreAtK` to a scored task group. A group is
|
|
112
|
-
* scored iff any record carries an effective score
|
|
113
|
-
*
|
|
114
|
-
*
|
|
115
|
-
* inflate the mean exactly when the agent fails
|
|
116
|
-
* neither field.
|
|
117
|
-
* @param {object} task -
|
|
112
|
+
* scored iff any record carries an effective score. A score-less record in a
|
|
113
|
+
* scored group contributes its verdict as the degenerate score. Such a record
|
|
114
|
+
* comes from a preflight failure that never reached the grade step, or from a
|
|
115
|
+
* binary run. A skip would inflate the mean exactly when the agent fails
|
|
116
|
+
* hardest. Binary groups gain neither field.
|
|
117
|
+
* @param {object} task - The function mutates it.
|
|
118
118
|
* @param {object[]} group
|
|
119
119
|
* @param {number[]} kValues
|
|
120
120
|
*/
|
|
@@ -127,9 +127,9 @@ function applyScoreFields(task, group, kValues) {
|
|
|
127
127
|
}
|
|
128
128
|
|
|
129
129
|
/**
|
|
130
|
-
* Build a normalized per-run detail object
|
|
131
|
-
*
|
|
132
|
-
* cognitive complexity below the lint ceiling.
|
|
130
|
+
* Build a normalized per-run detail object. Accumulate duration/turn samples
|
|
131
|
+
* so the caller can calculate the median. It is separate from `aggregate` to
|
|
132
|
+
* keep that function's cognitive complexity below the lint ceiling.
|
|
133
133
|
* @param {object} r - Raw record.
|
|
134
134
|
* @param {{allDurations: number[], allTurns: number[]}} acc
|
|
135
135
|
* @returns {RunDetail}
|
|
@@ -155,9 +155,9 @@ function buildRunDetail(r, acc) {
|
|
|
155
155
|
|
|
156
156
|
/**
|
|
157
157
|
* Render an aggregate report as markdown. When the report contains per-run
|
|
158
|
-
* detail (from `includeRuns: true`),
|
|
159
|
-
* pass@k table, and per-task detail sections.
|
|
160
|
-
* compact pass@k table.
|
|
158
|
+
* detail (from `includeRuns: true`), the renderer produces a full report.
|
|
159
|
+
* That report has a summary, a pass@k table, and per-task detail sections.
|
|
160
|
+
* Otherwise the renderer falls back to the compact pass@k table.
|
|
161
161
|
* @param {Awaited<ReturnType<typeof aggregate>>} report
|
|
162
162
|
* @param {number[]} kValues
|
|
163
163
|
* @returns {string}
|
|
@@ -170,10 +170,10 @@ export function renderTextReport(report, kValues) {
|
|
|
170
170
|
}
|
|
171
171
|
|
|
172
172
|
// ---------------------------------------------------------------------------
|
|
173
|
-
// Compact report — status line + pass@k table, no per-task detail.
|
|
174
|
-
// `report --detail=compact` (aggregate without `includeRuns`)
|
|
175
|
-
// summary uses it so a sharded run stays short
|
|
176
|
-
// full report over the combined ledger.
|
|
173
|
+
// Compact report — status line + pass@k table, no per-task detail.
|
|
174
|
+
// `report --detail=compact` selects it (aggregate without `includeRuns`).
|
|
175
|
+
// The per-shard summary uses it so a sharded run stays short. The merge job
|
|
176
|
+
// renders the full report over the combined ledger.
|
|
177
177
|
// ---------------------------------------------------------------------------
|
|
178
178
|
|
|
179
179
|
function renderCompactReport(report, kValues) {
|
|
@@ -500,17 +500,18 @@ function median(arr) {
|
|
|
500
500
|
// Record loading
|
|
501
501
|
// ---------------------------------------------------------------------------
|
|
502
502
|
|
|
503
|
-
//
|
|
503
|
+
// The walk never descends into these directories for a `results.jsonl`.
|
|
504
504
|
const SKIP_DIRS = new Set([".git", "node_modules"]);
|
|
505
505
|
|
|
506
506
|
/**
|
|
507
507
|
* Load and union every `results.jsonl` found recursively under `inputDir`.
|
|
508
508
|
*
|
|
509
|
-
* A single non-sharded run has one root-level ledger
|
|
510
|
-
* case of the same walk. A sharded run lays each shard's partial
|
|
511
|
-
* own subdirectory
|
|
512
|
-
* cells. An *existing* dir with no ledger yields
|
|
513
|
-
* *missing* dir lets `readdir`'s ENOENT
|
|
509
|
+
* A single non-sharded run has one root-level ledger. That is the trivial
|
|
510
|
+
* one-match case of the same walk. A sharded run lays each shard's partial
|
|
511
|
+
* ledger in its own subdirectory. A merge of those ledgers equals a report of
|
|
512
|
+
* a single run over the same cells. An *existing* dir with no ledger yields
|
|
513
|
+
* the empty union (exit 0). A *missing* dir lets `readdir`'s ENOENT
|
|
514
|
+
* propagate, so `report` still errors.
|
|
514
515
|
* @param {string} inputDir
|
|
515
516
|
* @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
|
|
516
517
|
* @returns {Promise<{records: object[], skipped: number}>}
|
|
@@ -520,10 +521,10 @@ async function loadRecords(inputDir, runtime) {
|
|
|
520
521
|
try {
|
|
521
522
|
files = await collectResultsFiles(inputDir, runtime);
|
|
522
523
|
} catch (e) {
|
|
523
|
-
// Re-throw with the stack collapsed to the message line
|
|
524
|
-
//
|
|
525
|
-
//
|
|
526
|
-
// pre-1370 stream-error shape the golden captured
|
|
524
|
+
// Re-throw with the stack collapsed to the message line. The CLI error
|
|
525
|
+
// output then stays free of node-internal async `readdir` frames. A
|
|
526
|
+
// missing --input dir surfaces its ENOENT as exit 1, which matches the
|
|
527
|
+
// pre-1370 stream-error shape the golden captured.
|
|
527
528
|
const err = new Error(e.message);
|
|
528
529
|
if (e.code) err.code = e.code;
|
|
529
530
|
err.stack = `Error: ${e.message}`;
|
|
@@ -540,11 +541,12 @@ async function loadRecords(inputDir, runtime) {
|
|
|
540
541
|
}
|
|
541
542
|
|
|
542
543
|
/**
|
|
543
|
-
* Parse one ledger's JSONL into `records
|
|
544
|
-
*
|
|
545
|
-
* `loadRecords` to keep
|
|
544
|
+
* Parse one ledger's JSONL into `records`. Skip each malformed or
|
|
545
|
+
* schema-invalid line and write a stderr warning. Return the skipped count.
|
|
546
|
+
* It is separate from `loadRecords` to keep that function's cognitive
|
|
547
|
+
* complexity under the lint ceiling.
|
|
546
548
|
* @param {string} content
|
|
547
|
-
* @param {object[]} records - Accumulator
|
|
549
|
+
* @param {object[]} records - Accumulator. The function appends in place.
|
|
548
550
|
* @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
|
|
549
551
|
* @returns {number} Skipped line count.
|
|
550
552
|
*/
|
|
@@ -567,7 +569,7 @@ function parseLedgerInto(content, records, runtime) {
|
|
|
567
569
|
validateResultRecord(record);
|
|
568
570
|
} catch (e) {
|
|
569
571
|
runtime.proc.stderr.write(
|
|
570
|
-
`benchmark report: skipped record
|
|
572
|
+
`benchmark report: skipped schema-invalid record — ${describeError(e)}\n`,
|
|
571
573
|
);
|
|
572
574
|
skipped++;
|
|
573
575
|
continue;
|
|
@@ -578,10 +580,10 @@ function parseLedgerInto(content, records, runtime) {
|
|
|
578
580
|
}
|
|
579
581
|
|
|
580
582
|
/**
|
|
581
|
-
* Recursively collect paths of every file named `results.jsonl` under `dir
|
|
582
|
-
*
|
|
583
|
-
* `readdir` walk
|
|
584
|
-
* is unexported, which is the wrong contract here.
|
|
583
|
+
* Recursively collect paths of every file named `results.jsonl` under `dir`.
|
|
584
|
+
* Skip `.git` and `node_modules`. Never follow a symlink. This is a
|
|
585
|
+
* purpose-built `readdir` walk. `task-family.js`'s private `walkFiles`
|
|
586
|
+
* resolves symlinks and is unexported, which is the wrong contract here.
|
|
585
587
|
* @param {string} dir
|
|
586
588
|
* @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
|
|
587
589
|
* @returns {Promise<string[]>}
|
|
@@ -604,10 +606,10 @@ async function collectResultsFiles(dir, runtime) {
|
|
|
604
606
|
}
|
|
605
607
|
|
|
606
608
|
/**
|
|
607
|
-
* Warn
|
|
608
|
-
*
|
|
609
|
-
* a duplicate signals misconfiguration
|
|
610
|
-
* count is honest.
|
|
609
|
+
* Warn when a `(taskId, runIndex)` cell appears more than once across shard
|
|
610
|
+
* ledgers. Do not silently merge the copies. The shard partition guarantees
|
|
611
|
+
* uniqueness, so a duplicate signals misconfiguration. Both copies stay in
|
|
612
|
+
* the group so the count is honest.
|
|
611
613
|
* @param {object[]} records
|
|
612
614
|
* @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
|
|
613
615
|
*/
|
|
@@ -620,7 +622,7 @@ function warnOnDuplicateCells(records, runtime) {
|
|
|
620
622
|
for (const [key, n] of counts) {
|
|
621
623
|
if (n > 1)
|
|
622
624
|
runtime.proc.stderr.write(
|
|
623
|
-
`benchmark report: duplicate cell ${key} appears ${n} times across shard ledgers
|
|
625
|
+
`benchmark report: duplicate cell ${key} appears ${n} times across shard ledgers. The shard partition should make each cell unique\n`,
|
|
624
626
|
);
|
|
625
627
|
}
|
|
626
628
|
}
|
|
@@ -660,14 +662,15 @@ function passAtKValue(n, c, k) {
|
|
|
660
662
|
|
|
661
663
|
/**
|
|
662
664
|
* score@k — the expected **maximum** score over k runs drawn without
|
|
663
|
-
* replacement from the n recorded scores
|
|
664
|
-
* With scores sorted ascending s₍₁₎…s₍ₙ₎:
|
|
665
|
+
* replacement from the n recorded scores. It is the continuous analog of
|
|
666
|
+
* pass@k. With scores sorted ascending s₍₁₎…s₍ₙ₎:
|
|
665
667
|
*
|
|
666
668
|
* score@k = Σ_{i=k..n} s₍ᵢ₎ · C(i−1, k−1) / C(n, k)
|
|
667
669
|
*
|
|
668
670
|
* Each term weights s₍ᵢ₎ by the probability it is the k-subset's maximum.
|
|
669
671
|
* Binary scores reduce exactly to the pass@k estimator (same BigInt binomial
|
|
670
|
-
* helper)
|
|
672
|
+
* helper). `k > n` yields the same `{error}` value, so both functions use
|
|
673
|
+
* one idiom.
|
|
671
674
|
* @param {number[]} scores - Effective per-record scores.
|
|
672
675
|
* @param {number} k
|
|
673
676
|
* @returns {number | {error: string}}
|