@forwardimpact/libharness 3.0.1 → 3.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +26 -4
- package/package.json +1 -1
- package/src/benchmark/hidden-tests.js +49 -18
- package/src/benchmark/judge.js +8 -6
- package/src/benchmark/raw-summary.js +79 -0
- package/src/benchmark/result.js +13 -7
- package/src/benchmark/runner.js +70 -67
- package/src/benchmark/task-family.js +9 -0
- package/src/benchmark/workdir.js +28 -6
- package/src/commands/benchmark-definition.js +2 -2
- package/src/commands/trace.js +31 -68
- package/src/index.js +1 -1
- package/src/trace-github.js +78 -60
- package/src/trace-identity.js +132 -0
- package/src/trace-multi.js +2 -17
- package/src/trace-split.js +103 -0
- package/src/benchmark/trace-split.js +0 -74
package/README.md
CHANGED
|
@@ -214,17 +214,39 @@ also lists the `Edit()` rules it tried.
|
|
|
214
214
|
|
|
215
215
|
## Documentation
|
|
216
216
|
|
|
217
|
-
- [Coordinate an Agent Team](https://www.
|
|
217
|
+
- [Coordinate an Agent Team](https://www.gemba.team/docs/coordinate-team/index.md)
|
|
218
218
|
— run a lead and N participant agents in one async session (supervise /
|
|
219
219
|
facilitate / discuss) with Ask/Answer/Announce and a single NDJSON trace.
|
|
220
|
-
- [Run an Eval](https://www.
|
|
220
|
+
- [Run an Eval](https://www.gemba.team/docs/prove-changes/run-eval/index.md)
|
|
221
221
|
— author a judge profile, run an eval locally, wire it into CI, and inspect
|
|
222
222
|
the trace it produces.
|
|
223
|
-
- [Prove Agent Changes](https://www.
|
|
223
|
+
- [Prove Agent Changes](https://www.gemba.team/docs/prove-changes/index.md)
|
|
224
224
|
— the end-to-end workflow from dataset generation through evaluation to
|
|
225
225
|
trace analysis, with multi-agent collaboration sessions.
|
|
226
|
-
- [Analyze Traces](https://www.
|
|
226
|
+
- [Analyze Traces](https://www.gemba.team/docs/prove-changes/trace-analysis/index.md)
|
|
227
227
|
— read the NDJSON traces produced by `gemba-harness` with `gemba-trace`.
|
|
228
228
|
- [Agent Teams](https://www.forwardimpact.team/docs/products/agent-teams/index.md)
|
|
229
229
|
— author the profiles consumed by `--agent-profile`, `--lead-profile`, and
|
|
230
230
|
`--agent-profiles`.
|
|
231
|
+
|
|
232
|
+
## Documentation home
|
|
233
|
+
|
|
234
|
+
libharness is an import-only library. It declares no `bin`. The
|
|
235
|
+
`gemba-harness`, `gemba-trace`, `gemba-benchmark`, and `gemba-selfedit`
|
|
236
|
+
commands ship with the Gemba product, which imports these modules. Run them
|
|
237
|
+
with `npx gemba-harness`, `npx gemba-trace`, or `npx gemba-benchmark`, or use
|
|
238
|
+
the installed `gemba-*` binaries. `gemba-selfedit` publishes no bare launcher.
|
|
239
|
+
Install `@forwardimpact/gemba` to get it.
|
|
240
|
+
|
|
241
|
+
The package publishes as `@forwardimpact/libharness` on the Forward Impact npm
|
|
242
|
+
scope. Install it with `npm install @forwardimpact/libharness`. Its task guides
|
|
243
|
+
live on the Gemba site at <https://www.gemba.team/>. The Forward Impact library
|
|
244
|
+
guide tree at <https://www.forwardimpact.team/docs/libraries/index.md> is not
|
|
245
|
+
this library's guide home.
|
|
246
|
+
|
|
247
|
+
**Decision (2026-08-26):** the split package scope and guide host are
|
|
248
|
+
deliberate. libharness stays a Gear npm package, so `package.json .homepage`
|
|
249
|
+
keeps <https://www.forwardimpact.team>. The agent-runtime guides moved to
|
|
250
|
+
gemba.team with the rest of the Gemba product. The `## Documentation` list
|
|
251
|
+
above carries the current URLs. Old `forwardimpact.team/docs/libraries/`
|
|
252
|
+
addresses forward to gemba.team.
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@forwardimpact/libharness",
|
|
3
|
-
"version": "3.0
|
|
3
|
+
"version": "3.1.0",
|
|
4
4
|
"description": "Autonomous agent team harness — coordinate a lead and participant agents in one async session, with eval, benchmark, and trace tooling to prove the changes worked.",
|
|
5
5
|
"keywords": [
|
|
6
6
|
"orchestration",
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* Hidden-test engine — runs a task's `tests/` overlay against the post-run
|
|
3
3
|
* agent CWD. The engine stages each file at its mirrored path. It runs each
|
|
4
|
-
* check with `
|
|
4
|
+
* check with `bun test`. It converts the exit status into one check row.
|
|
5
5
|
* It restores the tree, so the judge sees the workdir exactly as the agent
|
|
6
6
|
* left it.
|
|
7
7
|
*
|
|
@@ -69,7 +69,25 @@ async function runOneCheck(task, ctx, runtime, timeoutMs, check) {
|
|
|
69
69
|
}
|
|
70
70
|
|
|
71
71
|
/**
|
|
72
|
-
*
|
|
72
|
+
* Reads a child pipe to a string. Returns what it read when the stream tears
|
|
73
|
+
* down mid-read, because a pipe that closes as the child exits is a race, not
|
|
74
|
+
* a check failure. The caller reads the exit code for the verdict.
|
|
75
|
+
*
|
|
76
|
+
* @param {import("node:stream").Readable} stream - Child stdout or stderr.
|
|
77
|
+
* @returns {Promise<string>} Everything read before the stream ended.
|
|
78
|
+
*/
|
|
79
|
+
async function drainQuietly(stream) {
|
|
80
|
+
let out = "";
|
|
81
|
+
try {
|
|
82
|
+
for await (const chunk of stream) out += chunk.toString();
|
|
83
|
+
} catch {
|
|
84
|
+
// The stream closed under us. Keep what arrived.
|
|
85
|
+
}
|
|
86
|
+
return out;
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
/**
|
|
90
|
+
* Spawn `bun test <staged path>` from the agent CWD under the hook env
|
|
73
91
|
* and map the exit status onto one row. The clock timer SIGKILLs a child
|
|
74
92
|
* that outlives the per-check budget. The row then fails with a timeout
|
|
75
93
|
* message.
|
|
@@ -83,27 +101,40 @@ async function spawnCheck(task, ctx, runtime, timeoutMs, check) {
|
|
|
83
101
|
hooksDir: task.paths.hooks,
|
|
84
102
|
familyDir: ctx.familyDir,
|
|
85
103
|
});
|
|
86
|
-
//
|
|
87
|
-
//
|
|
88
|
-
//
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
104
|
+
// `bun test` sets no test-context variable that a nested run inherits, so
|
|
105
|
+
// a child reports its own exit status even when the harness itself runs
|
|
106
|
+
// under a test runner. That removes the `NODE_TEST_CONTEXT` scrub the
|
|
107
|
+
// `node --test` engine needed to stop a failing check minting a passing
|
|
108
|
+
// row. The "fractional score" grade test is the standing guard: it fails
|
|
109
|
+
// if a failing check ever reports success again.
|
|
110
|
+
//
|
|
111
|
+
// Pass the ABSOLUTE staged path. `bun test` reads its argument as a
|
|
112
|
+
// substring filter over discovered paths, not as one file, so the relative
|
|
113
|
+
// `app/test/x.test.js` also matches an agent-authored `sub/app/test/
|
|
114
|
+
// x.test.js` and folds that file's result into this check's row. An
|
|
115
|
+
// absolute path cannot be a substring of a deeper path, so it selects
|
|
116
|
+
// exactly the staged file. One `*.test.js` stays one check.
|
|
117
|
+
const child = runtime.subprocess.spawn(
|
|
118
|
+
"bun",
|
|
119
|
+
["test", join(ctx.cwd, check.stagePath)],
|
|
120
|
+
{
|
|
121
|
+
cwd: ctx.cwd,
|
|
122
|
+
env,
|
|
123
|
+
stdio: ["ignore", "pipe", "pipe"],
|
|
124
|
+
},
|
|
125
|
+
);
|
|
95
126
|
let timedOut = false;
|
|
96
127
|
const timer = runtime.clock.setTimeout(() => {
|
|
97
128
|
timedOut = true;
|
|
98
129
|
child.kill("SIGKILL");
|
|
99
130
|
}, timeoutMs);
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
|
|
131
|
+
// Both pipes drain only so a chatty child never blocks on a full pipe.
|
|
132
|
+
// A child that exits while its pipe is still open makes the async iterator
|
|
133
|
+
// reject with ERR_STREAM_PREMATURE_CLOSE on some runtimes. That teardown
|
|
134
|
+
// race is not a check result, so it must not throw out of the check. The
|
|
135
|
+
// exit code below is the verdict.
|
|
136
|
+
const drainStdout = drainQuietly(child.stdout);
|
|
137
|
+
const stderr = await drainQuietly(child.stderr);
|
|
107
138
|
await drainStdout;
|
|
108
139
|
const exit = await child.exitCode;
|
|
109
140
|
runtime.clock.clearTimeout(timer);
|
package/src/benchmark/judge.js
CHANGED
|
@@ -9,17 +9,19 @@
|
|
|
9
9
|
*
|
|
10
10
|
* {{AGENT_INSTRUCTIONS}} — contents of agent.task.md
|
|
11
11
|
* {{AGENT_PROFILE}} — agent profile body (empty string if none)
|
|
12
|
-
* {{AGENT_TRACE_PATH}} — path to agent
|
|
12
|
+
* {{AGENT_TRACE_PATH}} — absolute path to the cell's agent lane,
|
|
13
|
+
* trace--<case>--agent.agent.ndjson (materialized
|
|
14
|
+
* before any session runs)
|
|
13
15
|
* {{GRADE_RESULT}} — JSON grade object plus the merged check rows
|
|
14
16
|
* {{SKILL_SET_HASH}} — SHA-256 from apm.lock.yaml
|
|
15
17
|
* {{TASK_ID}} — task name (directory under tasks/)
|
|
16
18
|
* {{TASK_DIR}} — path to the agent working directory
|
|
17
19
|
*
|
|
18
|
-
* The
|
|
19
|
-
*
|
|
20
|
-
* `parseConcludeFromTrace`
|
|
21
|
-
* when the runtime ctx
|
|
22
|
-
* historical run from its judge
|
|
20
|
+
* The judge verdict is captured from the orchestration context's
|
|
21
|
+
* `concluded` flag directly — no trace parsing on the happy path.
|
|
22
|
+
* `parseConcludeFromTrace` is preserved for offline analysis and as a
|
|
23
|
+
* fallback when the runtime ctx isn't available (e.g. re-grading a
|
|
24
|
+
* historical run from its preserved judge lane file).
|
|
23
25
|
*/
|
|
24
26
|
|
|
25
27
|
import { createJudge } from "../judge.js";
|
|
@@ -0,0 +1,79 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Raw-trace summary for the benchmark runner: one post-session read of the
|
|
3
|
+
* preserved raw combined trace derives cost, turns, and submission. Named as
|
|
4
|
+
* its own module so summarization never re-entangles with splitting — the
|
|
5
|
+
* coupling that caused the original split-policy divergence.
|
|
6
|
+
*/
|
|
7
|
+
|
|
8
|
+
import { sumTraceCost } from "../cost.js";
|
|
9
|
+
import { parseEnvelopeLine } from "../trace-split.js";
|
|
10
|
+
|
|
11
|
+
/**
|
|
12
|
+
* One read of the preserved raw combined trace:
|
|
13
|
+
* cost — `sumTraceCost` over the lines (the one cost path),
|
|
14
|
+
* turns — last orchestrator-source `summary` event's `turns`,
|
|
15
|
+
* submission — last agent-source assistant text block.
|
|
16
|
+
*
|
|
17
|
+
* An empty (materialized-stub) raw file yields zeros and an empty
|
|
18
|
+
* submission; malformed and blank lines are tolerated.
|
|
19
|
+
*
|
|
20
|
+
* @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
|
|
21
|
+
* @param {string} rawTracePath
|
|
22
|
+
* @returns {Promise<{costUsd: number,
|
|
23
|
+
* costBreakdown: {agent: number, supervisor: number},
|
|
24
|
+
* turns: number, submission: string}>}
|
|
25
|
+
*/
|
|
26
|
+
export async function summarizeRawTrace(runtime, rawTracePath) {
|
|
27
|
+
const content = await runtime.fs.readFile(rawTracePath, "utf8");
|
|
28
|
+
const lines = content.split("\n");
|
|
29
|
+
const { totalCostUsd, bySource } = sumTraceCost(lines);
|
|
30
|
+
const { turns, submission } = deriveTurnsAndSubmission(lines);
|
|
31
|
+
|
|
32
|
+
return {
|
|
33
|
+
costUsd: totalCostUsd,
|
|
34
|
+
costBreakdown: {
|
|
35
|
+
agent: bySource.agent ?? 0,
|
|
36
|
+
supervisor: bySource.supervisor ?? 0,
|
|
37
|
+
},
|
|
38
|
+
turns,
|
|
39
|
+
submission,
|
|
40
|
+
};
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
/**
|
|
44
|
+
* One walk of the parsed envelope lines: the last orchestrator `summary`
|
|
45
|
+
* event's `turns` and the last agent assistant text block.
|
|
46
|
+
* @param {string[]} lines
|
|
47
|
+
* @returns {{turns: number, submission: string}}
|
|
48
|
+
*/
|
|
49
|
+
function deriveTurnsAndSubmission(lines) {
|
|
50
|
+
let turns = 0;
|
|
51
|
+
let submission = "";
|
|
52
|
+
for (const line of lines) {
|
|
53
|
+
const envelope = parseEnvelopeLine(line);
|
|
54
|
+
if (!envelope) continue;
|
|
55
|
+
const inner = envelope.event;
|
|
56
|
+
if (envelope.source === "agent" && inner.type === "assistant") {
|
|
57
|
+
const text = extractText(inner);
|
|
58
|
+
if (text) submission = text;
|
|
59
|
+
}
|
|
60
|
+
if (envelope.source === "orchestrator" && inner.type === "summary") {
|
|
61
|
+
turns = inner.turns ?? 0;
|
|
62
|
+
}
|
|
63
|
+
}
|
|
64
|
+
return { turns, submission };
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
/**
|
|
68
|
+
* Last text block of an assistant event's content, or null when none exists.
|
|
69
|
+
* @param {object} inner - Unwrapped event.
|
|
70
|
+
* @returns {string|null}
|
|
71
|
+
*/
|
|
72
|
+
function extractText(inner) {
|
|
73
|
+
const content = inner.message?.content ?? inner.content;
|
|
74
|
+
if (!Array.isArray(content)) return null;
|
|
75
|
+
for (let i = content.length - 1; i >= 0; i--) {
|
|
76
|
+
if (content[i].type === "text" && content[i].text) return content[i].text;
|
|
77
|
+
}
|
|
78
|
+
return null;
|
|
79
|
+
}
|
package/src/benchmark/result.js
CHANGED
|
@@ -104,9 +104,15 @@ const HAPPY_RECORD = z.object({
|
|
|
104
104
|
score: z.number().min(0).max(1).optional(),
|
|
105
105
|
submission: z.string(),
|
|
106
106
|
judgeVerdict: JUDGE_VERDICT_SHAPE.optional(),
|
|
107
|
+
// Trace paths are relative to the run output directory — valid on the
|
|
108
|
+
// runner and inside a downloaded trace artifact alike. Raw and
|
|
109
|
+
// agent/supervisor lanes are materialized at workdir allocation, so they
|
|
110
|
+
// are present on every executed cell; the judge lane exists only on
|
|
111
|
+
// judged cells.
|
|
112
|
+
rawTracePath: z.string(),
|
|
107
113
|
agentTracePath: z.string(),
|
|
108
114
|
supervisorTracePath: z.string(),
|
|
109
|
-
judgeTracePath: z.string(),
|
|
115
|
+
judgeTracePath: z.string().optional(),
|
|
110
116
|
agentError: AGENT_ERROR_SHAPE.optional(),
|
|
111
117
|
preflightError: z.undefined().optional(),
|
|
112
118
|
});
|
|
@@ -115,12 +121,12 @@ const PREFLIGHT_RECORD = z.object({
|
|
|
115
121
|
...COMMON_FIELDS,
|
|
116
122
|
costUsd: z.literal(0),
|
|
117
123
|
preflightError: PREFLIGHT_ERROR_SHAPE,
|
|
118
|
-
//
|
|
119
|
-
//
|
|
120
|
-
|
|
121
|
-
agentTracePath: z.
|
|
122
|
-
supervisorTracePath: z.
|
|
123
|
-
judgeTracePath: z.
|
|
124
|
+
// No trace-path fields: a preflight-failure record references only traces
|
|
125
|
+
// a session produced, even though the materialized stubs exist on disk.
|
|
126
|
+
rawTracePath: z.undefined().optional(),
|
|
127
|
+
agentTracePath: z.undefined().optional(),
|
|
128
|
+
supervisorTracePath: z.undefined().optional(),
|
|
129
|
+
judgeTracePath: z.undefined().optional(),
|
|
124
130
|
invariants: z.undefined().optional(),
|
|
125
131
|
grade: z.undefined().optional(),
|
|
126
132
|
hiddenTests: z.undefined().optional(),
|
package/src/benchmark/runner.js
CHANGED
|
@@ -19,11 +19,11 @@
|
|
|
19
19
|
* ledger. The iterator mirrors the same stream to CLI stdout.
|
|
20
20
|
*/
|
|
21
21
|
|
|
22
|
-
import { join, resolve as resolvePath } from "node:path";
|
|
22
|
+
import { join, relative, resolve as resolvePath } from "node:path";
|
|
23
23
|
|
|
24
24
|
import { DEFAULT_ENV_ALLOWLIST, createRedactor } from "../redaction.js";
|
|
25
|
-
import { sumTraceCost } from "../cost.js";
|
|
26
25
|
import { createSupervisor } from "../supervisor.js";
|
|
26
|
+
import { splitTrace } from "../trace-split.js";
|
|
27
27
|
import { installApm as defaultInstallApm } from "./apm-installer.js";
|
|
28
28
|
import { installNpm as defaultInstallNpm } from "./npm-installer.js";
|
|
29
29
|
import { runJudge } from "./judge.js";
|
|
@@ -31,8 +31,8 @@ import { validateResultRecord } from "./result.js";
|
|
|
31
31
|
import { runInvariants } from "./invariants.js";
|
|
32
32
|
import { runHiddenTests } from "./hidden-tests.js";
|
|
33
33
|
import { runProducersAndGrade } from "./grade.js";
|
|
34
|
+
import { summarizeRawTrace } from "./raw-summary.js";
|
|
34
35
|
import { assertJudgeProfileStaged, loadTaskFamily } from "./task-family.js";
|
|
35
|
-
import { splitAndSummarize } from "./trace-split.js";
|
|
36
36
|
import { createWorkdirManager } from "./workdir.js";
|
|
37
37
|
import { CellScheduler } from "./scheduler.js";
|
|
38
38
|
|
|
@@ -77,9 +77,10 @@ export class BenchmarkRunner {
|
|
|
77
77
|
* to `AGENT_WATCHDOG_MS`. A test injects its own to force a stall in-test.
|
|
78
78
|
* @param {number} [opts.termGraceMs] - SIGTERM→SIGKILL grace (ms) for the per-task process group.
|
|
79
79
|
* @param {Function} [opts.runAgent] - Test seam: replaces the agent-under-test
|
|
80
|
-
* session. Must
|
|
81
|
-
*
|
|
82
|
-
*
|
|
80
|
+
* session. Must run the session, stream `{source, seq, event}` envelopes
|
|
81
|
+
* to `workdir.rawTracePath`, and return `{agentError?}` — cost, turns,
|
|
82
|
+
* and submission are always derived from the raw file by the shared
|
|
83
|
+
* split/summary pipeline, so the seam exercises the real path. Internal
|
|
83
84
|
* testing only. It is not part of the public API.
|
|
84
85
|
* @param {import("@forwardimpact/libutil/runtime").Runtime} opts.runtime -
|
|
85
86
|
* The host injects these ambient collaborators (`fs`, `subprocess`,
|
|
@@ -256,12 +257,12 @@ export class BenchmarkRunner {
|
|
|
256
257
|
t0,
|
|
257
258
|
});
|
|
258
259
|
} catch (e) {
|
|
259
|
-
// `wm.start()` (port
|
|
260
|
-
//
|
|
261
|
-
// fallback record so `#runOne` never
|
|
262
|
-
// one-record-per-cell contract depends on
|
|
263
|
-
// fallback because it fails the schema, the
|
|
264
|
-
// runner-side schema failure.
|
|
260
|
+
// Catches the throw sites `#executeCell` does not: `wm.start()` (port
|
|
261
|
+
// acquire + workdir/env seed) and the shared split/summary pipeline.
|
|
262
|
+
// Turn either into the runner's own fallback record so `#runOne` never
|
|
263
|
+
// rejects. The scheduler's one-record-per-cell contract depends on
|
|
264
|
+
// that. `report` skips the fallback because it fails the schema, the
|
|
265
|
+
// same as any other runner-side schema failure.
|
|
265
266
|
return {
|
|
266
267
|
taskId: task.id,
|
|
267
268
|
runIndex,
|
|
@@ -301,8 +302,8 @@ export class BenchmarkRunner {
|
|
|
301
302
|
}
|
|
302
303
|
{
|
|
303
304
|
const agentRun = await this.#runAgentSafe(task, workdir);
|
|
304
|
-
const { costUsd, turns, submission, agentError } =
|
|
305
|
-
|
|
305
|
+
const { costUsd, costBreakdown, turns, submission, agentError } =
|
|
306
|
+
agentRun;
|
|
306
307
|
const graded = await this.#gradeCell(family, task, workdir);
|
|
307
308
|
const { invariants, hiddenRows, engineError, rows, grade } = graded;
|
|
308
309
|
const { judgeVerdict, judgeCost } = await this.#judgeCell({
|
|
@@ -319,6 +320,7 @@ export class BenchmarkRunner {
|
|
|
319
320
|
// or a judge that fails zeroes the effective score. Full marks does not
|
|
320
321
|
// zero it. A fractional score with verdict fail is the point.
|
|
321
322
|
const scoreValid = graded.healthy && grade.gatesPass && judgePass;
|
|
323
|
+
const tracePaths = await this.#traceRecordPaths(workdir);
|
|
322
324
|
const record = {
|
|
323
325
|
taskId: task.id,
|
|
324
326
|
runIndex,
|
|
@@ -337,15 +339,9 @@ export class BenchmarkRunner {
|
|
|
337
339
|
submission,
|
|
338
340
|
...(judgeVerdict && { judgeVerdict }),
|
|
339
341
|
costUsd: costUsd + judgeCost,
|
|
340
|
-
costBreakdown: {
|
|
341
|
-
agent: breakdown.agent ?? 0,
|
|
342
|
-
supervisor: breakdown.supervisor ?? 0,
|
|
343
|
-
judge: judgeCost,
|
|
344
|
-
},
|
|
342
|
+
costBreakdown: { ...costBreakdown, judge: judgeCost },
|
|
345
343
|
turns,
|
|
346
|
-
|
|
347
|
-
supervisorTracePath: workdir.supervisorTracePath,
|
|
348
|
-
judgeTracePath: workdir.judgeTracePath,
|
|
344
|
+
...tracePaths,
|
|
349
345
|
profiles: {
|
|
350
346
|
agent: this.profiles.agent,
|
|
351
347
|
supervisor: null,
|
|
@@ -424,39 +420,42 @@ export class BenchmarkRunner {
|
|
|
424
420
|
}
|
|
425
421
|
|
|
426
422
|
/**
|
|
427
|
-
* Dispatch to either the injected hook or the default `#runAgent
|
|
428
|
-
*
|
|
429
|
-
*
|
|
430
|
-
*
|
|
423
|
+
* Dispatch to either the injected hook or the default `#runAgent`, then run
|
|
424
|
+
* the shared pipeline once: split the preserved raw trace into lanes and
|
|
425
|
+
* derive cost/turns/submission from the same file. Either session path can
|
|
426
|
+
* throw; catch here so a session error becomes an `agentError` on the
|
|
427
|
+
* record rather than aborting the whole iterator — the pipeline still runs,
|
|
428
|
+
* so failed cells keep coherent (possibly empty) lanes and zeroed totals.
|
|
429
|
+
* A pipeline throw itself (the raw file is materialized at allocation, so
|
|
430
|
+
* only an fs-level fault) propagates to `#runOne`'s fallback record.
|
|
431
431
|
*/
|
|
432
432
|
async #runAgentSafe(task, workdir) {
|
|
433
|
+
let agentError = null;
|
|
433
434
|
try {
|
|
434
|
-
|
|
435
|
-
|
|
436
|
-
|
|
437
|
-
|
|
438
|
-
return await this.#runAgent(task, workdir);
|
|
435
|
+
const r = this._runAgentHook
|
|
436
|
+
? await this._runAgentHook(task, workdir, this)
|
|
437
|
+
: await this.#runAgent(task, workdir);
|
|
438
|
+
agentError = r?.agentError ?? null;
|
|
439
439
|
} catch (e) {
|
|
440
|
-
|
|
441
|
-
costUsd: 0,
|
|
442
|
-
costBreakdown: { agent: 0, supervisor: 0 },
|
|
443
|
-
turns: 0,
|
|
444
|
-
submission: "",
|
|
445
|
-
agentError: { message: e.message ?? String(e), aborted: false },
|
|
446
|
-
};
|
|
440
|
+
agentError = { message: e.message ?? String(e), aborted: false };
|
|
447
441
|
}
|
|
442
|
+
await splitTrace(this.runtime, workdir.rawTracePath, {
|
|
443
|
+
caseId: workdir.caseId,
|
|
444
|
+
outputDir: workdir.runDir,
|
|
445
|
+
});
|
|
446
|
+
const summary = await summarizeRawTrace(this.runtime, workdir.rawTracePath);
|
|
447
|
+
return { ...summary, agentError };
|
|
448
448
|
}
|
|
449
449
|
|
|
450
450
|
/**
|
|
451
|
-
* Run the agent-under-test under a Supervisor. The supervisor
|
|
452
|
-
*
|
|
453
|
-
* the
|
|
454
|
-
*
|
|
451
|
+
* Run the agent-under-test under a Supervisor. The supervisor streams the
|
|
452
|
+
* combined tagged NDJSON envelope trace to `workdir.rawTracePath`, which is
|
|
453
|
+
* preserved for the life of the run output; `#runAgentSafe` splits it into
|
|
454
|
+
* the convention-named lanes and summarizes it afterwards.
|
|
455
455
|
*/
|
|
456
456
|
async #runAgent(task, workdir) {
|
|
457
457
|
const fs = this.runtime.fs;
|
|
458
|
-
const
|
|
459
|
-
const combinedStream = fs.createWriteStream(combinedPath);
|
|
458
|
+
const combinedStream = fs.createWriteStream(workdir.rawTracePath);
|
|
460
459
|
const supervisorInstructions = task.paths.supervisor
|
|
461
460
|
? await fs.readFile(task.paths.supervisor, "utf8").catch(() => null)
|
|
462
461
|
: null;
|
|
@@ -510,27 +509,31 @@ export class BenchmarkRunner {
|
|
|
510
509
|
this.runtime.clock.clearTimeout(watchdog);
|
|
511
510
|
await new Promise((r) => combinedStream.end(r));
|
|
512
511
|
}
|
|
513
|
-
|
|
514
|
-
|
|
515
|
-
|
|
516
|
-
|
|
517
|
-
|
|
518
|
-
|
|
519
|
-
|
|
520
|
-
|
|
521
|
-
|
|
522
|
-
|
|
523
|
-
const
|
|
524
|
-
|
|
525
|
-
|
|
526
|
-
|
|
527
|
-
|
|
528
|
-
costBreakdown: {
|
|
529
|
-
agent: bySource.agent ?? 0,
|
|
530
|
-
supervisor: bySource.supervisor ?? 0,
|
|
531
|
-
},
|
|
532
|
-
agentError,
|
|
512
|
+
return { agentError };
|
|
513
|
+
}
|
|
514
|
+
|
|
515
|
+
/**
|
|
516
|
+
* Run-output-relative trace-path record fields, each present only when its
|
|
517
|
+
* file exists: raw and agent/supervisor lanes are materialized at workdir
|
|
518
|
+
* allocation (present on every executed cell); the judge lane exists only
|
|
519
|
+
* on judged cells. Relative paths stay valid inside a downloaded artifact.
|
|
520
|
+
*/
|
|
521
|
+
async #traceRecordPaths(workdir) {
|
|
522
|
+
const fields = {
|
|
523
|
+
rawTracePath: workdir.rawTracePath,
|
|
524
|
+
agentTracePath: workdir.agentTracePath,
|
|
525
|
+
supervisorTracePath: workdir.supervisorTracePath,
|
|
526
|
+
judgeTracePath: workdir.judgeTracePath,
|
|
533
527
|
};
|
|
528
|
+
const out = {};
|
|
529
|
+
for (const [field, absPath] of Object.entries(fields)) {
|
|
530
|
+
const exists = await this.runtime.fs
|
|
531
|
+
.access(absPath)
|
|
532
|
+
.then(() => true)
|
|
533
|
+
.catch(() => false);
|
|
534
|
+
if (exists) out[field] = relative(this.output, absPath);
|
|
535
|
+
}
|
|
536
|
+
return out;
|
|
534
537
|
}
|
|
535
538
|
|
|
536
539
|
async #buildJudgeContext(task, workdir, skillSetHash) {
|
|
@@ -579,9 +582,9 @@ export class BenchmarkRunner {
|
|
|
579
582
|
skillSetHash,
|
|
580
583
|
familyRevision,
|
|
581
584
|
durationMs,
|
|
582
|
-
|
|
583
|
-
|
|
584
|
-
|
|
585
|
+
// No trace-path fields: even though the materialized stubs exist on
|
|
586
|
+
// disk, a preflight-failure record references only traces a session
|
|
587
|
+
// produced.
|
|
585
588
|
};
|
|
586
589
|
}
|
|
587
590
|
|
|
@@ -35,6 +35,8 @@
|
|
|
35
35
|
import { createHash } from "node:crypto";
|
|
36
36
|
import { join, posix, relative, resolve, sep } from "node:path";
|
|
37
37
|
|
|
38
|
+
import { isValidTaskId } from "../trace-identity.js";
|
|
39
|
+
|
|
38
40
|
const GIT_URL_RE = /^(git@|https?:\/\/|ssh:\/\/|git:\/\/)/;
|
|
39
41
|
const SKIP_DIRS = new Set([".git", "node_modules"]);
|
|
40
42
|
// POSIX `X_OK` (execute permission). Node's fs honours the numeric mode, so we
|
|
@@ -129,6 +131,13 @@ async function discoverTasks(runtime, rootPath) {
|
|
|
129
131
|
}
|
|
130
132
|
|
|
131
133
|
async function loadTask(fs, taskDir, id) {
|
|
134
|
+
// The rule itself lives in the identity module; the loader only invokes
|
|
135
|
+
// the predicate so an unbuildable case id fails before any agent spend.
|
|
136
|
+
if (!isValidTaskId(id)) {
|
|
137
|
+
throw new Error(
|
|
138
|
+
`invalid task id '${id}': task directory names must not contain "--" or start/end with "-"`,
|
|
139
|
+
);
|
|
140
|
+
}
|
|
132
141
|
const supervisorPath = join(taskDir, "supervisor.task.md");
|
|
133
142
|
const judgePath = join(taskDir, "judge.task.md");
|
|
134
143
|
const preflightPath = join(taskDir, "hooks", "preflight.sh");
|
package/src/benchmark/workdir.js
CHANGED
|
@@ -16,6 +16,11 @@ import { createServer } from "node:net";
|
|
|
16
16
|
import { connect } from "node:net";
|
|
17
17
|
import { join } from "node:path";
|
|
18
18
|
|
|
19
|
+
import {
|
|
20
|
+
buildCaseId,
|
|
21
|
+
laneFilename,
|
|
22
|
+
rawTraceFilename,
|
|
23
|
+
} from "../trace-identity.js";
|
|
19
24
|
import { loadEnv } from "./env-loader.js";
|
|
20
25
|
import { buildHookEnv } from "./hook-env.js";
|
|
21
26
|
|
|
@@ -28,6 +33,8 @@ const DEFAULT_TERM_GRACE_MS = 5_000;
|
|
|
28
33
|
* @property {number} port - Allocated TCP port for the agent.
|
|
29
34
|
* @property {number} pgid - Process-group id captured from the preflight child.
|
|
30
35
|
* @property {*} scaffold - Reserved per design § Components. v1 sets null.
|
|
36
|
+
* @property {string} caseId - Grid-unique case identity `<taskId>-r<runIndex>`.
|
|
37
|
+
* @property {string} rawTracePath - Combined raw envelope trace (kept).
|
|
31
38
|
* @property {string} agentTracePath
|
|
32
39
|
* @property {string} supervisorTracePath
|
|
33
40
|
* @property {string} judgeTracePath
|
|
@@ -72,8 +79,7 @@ export class WorkdirManager {
|
|
|
72
79
|
*/
|
|
73
80
|
async start(task, runIndex) {
|
|
74
81
|
const fs = this.runtime.fs;
|
|
75
|
-
const
|
|
76
|
-
const runDir = join(this.runOutputDir, "runs", slug, String(runIndex));
|
|
82
|
+
const runDir = join(this.runOutputDir, "runs", task.id, String(runIndex));
|
|
77
83
|
const cwd = join(runDir, "cwd");
|
|
78
84
|
await fs.mkdir(cwd, { recursive: true });
|
|
79
85
|
|
|
@@ -124,11 +130,25 @@ export class WorkdirManager {
|
|
|
124
130
|
const envNames =
|
|
125
131
|
envDirs.length > 0 ? await loadEnv(envDirs, cwd, this.runtime) : [];
|
|
126
132
|
|
|
127
|
-
|
|
128
|
-
|
|
129
|
-
const
|
|
130
|
-
const
|
|
133
|
+
// Allocate identity and trace paths before acquiring the port so an
|
|
134
|
+
// invalid task id cannot leak a port reservation.
|
|
135
|
+
const caseId = buildCaseId(task.id, runIndex);
|
|
136
|
+
const rawTracePath = join(runDir, rawTraceFilename(caseId));
|
|
137
|
+
const agentTracePath = join(runDir, laneFilename(caseId, "agent", "agent"));
|
|
138
|
+
const supervisorTracePath = join(
|
|
139
|
+
runDir,
|
|
140
|
+
laneFilename(caseId, "supervisor", "supervisor"),
|
|
141
|
+
);
|
|
142
|
+
const judgeTracePath = join(runDir, laneFilename(caseId, "judge", "judge"));
|
|
143
|
+
// Materialize the raw and agent/supervisor lanes empty at allocation so
|
|
144
|
+
// every path that reaches the judge — including a pre-session agent
|
|
145
|
+
// failure — finds them on disk. The judge lane is written by the judge
|
|
146
|
+
// session itself and exists only on judged cells.
|
|
147
|
+
for (const p of [rawTracePath, agentTracePath, supervisorTracePath]) {
|
|
148
|
+
await fs.writeFile(p, "");
|
|
149
|
+
}
|
|
131
150
|
|
|
151
|
+
const port = await this.ports.acquire();
|
|
132
152
|
const preflight = task.paths.preflight
|
|
133
153
|
? await runPreflight(this.runtime, task.paths.preflight, cwd, port, {
|
|
134
154
|
taskId: task.id,
|
|
@@ -144,6 +164,8 @@ export class WorkdirManager {
|
|
|
144
164
|
port,
|
|
145
165
|
pgid: preflight.pgid,
|
|
146
166
|
scaffold: null,
|
|
167
|
+
caseId,
|
|
168
|
+
rawTracePath,
|
|
147
169
|
agentTracePath,
|
|
148
170
|
supervisorTracePath,
|
|
149
171
|
judgeTracePath,
|
|
@@ -166,13 +166,13 @@ export const definition = {
|
|
|
166
166
|
documentation: [
|
|
167
167
|
{
|
|
168
168
|
title: "Run a Benchmark",
|
|
169
|
-
url: "https://www.
|
|
169
|
+
url: "https://www.gemba.team/docs/prove-changes/run-benchmark/index.md",
|
|
170
170
|
description:
|
|
171
171
|
"Author a coding-task family, run a benchmark across multiple runs, and read the pass@k report.",
|
|
172
172
|
},
|
|
173
173
|
{
|
|
174
174
|
title: "Automate with GitHub Actions",
|
|
175
|
-
url: "https://www.
|
|
175
|
+
url: "https://www.gemba.team/docs/prove-changes/run-benchmark/ci-workflow/index.md",
|
|
176
176
|
description:
|
|
177
177
|
"Run benchmarks in CI with the forwardimpact/gemba-benchmark action.",
|
|
178
178
|
},
|