@forwardimpact/libharness 0.1.22 → 1.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -201
- package/README.md +196 -80
- package/bin/fit-benchmark.js +44 -0
- package/bin/fit-harness.js +358 -0
- package/bin/fit-selfedit.js +165 -0
- package/bin/fit-trace.js +510 -0
- package/package.json +41 -11
- package/src/agent-runner.js +256 -0
- package/src/benchmark/apm-installer.js +207 -0
- package/src/benchmark/env-loader.js +158 -0
- package/src/benchmark/hook-env.js +40 -0
- package/src/benchmark/invariants.js +141 -0
- package/src/benchmark/judge.js +187 -0
- package/src/benchmark/npm-installer.js +87 -0
- package/src/benchmark/report.js +604 -0
- package/src/benchmark/result.js +127 -0
- package/src/benchmark/runner.js +688 -0
- package/src/benchmark/scheduler.js +78 -0
- package/src/benchmark/task-family.js +260 -0
- package/src/benchmark/workdir.js +344 -0
- package/src/commands/assert.js +153 -0
- package/src/commands/benchmark-definition.js +175 -0
- package/src/commands/benchmark-invariants.js +73 -0
- package/src/commands/benchmark-report.js +51 -0
- package/src/commands/benchmark-run.js +175 -0
- package/src/commands/by-discussion.js +94 -0
- package/src/commands/callback.js +119 -0
- package/src/commands/discuss.js +132 -0
- package/src/commands/facilitate.js +123 -0
- package/src/commands/output.js +36 -0
- package/src/commands/run.js +152 -0
- package/src/commands/supervise.js +136 -0
- package/src/commands/task-input.js +54 -0
- package/src/commands/tee.js +53 -0
- package/src/commands/trace.js +630 -0
- package/src/commands/work-tracker.js +35 -0
- package/src/cost.js +79 -0
- package/src/discuss-tools.js +173 -0
- package/src/discusser.js +394 -0
- package/src/events/github.js +161 -0
- package/src/facilitator.js +205 -0
- package/src/inbox-poller.js +81 -0
- package/src/index.js +72 -2
- package/src/judge.js +210 -0
- package/src/message-bus.js +118 -0
- package/src/orchestration-loop.js +330 -0
- package/src/orchestration-toolkit.js +441 -0
- package/src/orchestrator-helpers.js +23 -0
- package/src/profile-prompt.js +266 -0
- package/src/redaction.js +253 -0
- package/src/render/line-renderer.js +54 -0
- package/src/render/orchestrator-filter.js +19 -0
- package/src/render/palette.js +63 -0
- package/src/render/tool-hints.js +154 -0
- package/src/render/turn-renderer.js +96 -0
- package/src/reply-emitter.js +47 -0
- package/src/sequence-counter.js +21 -0
- package/src/signature-filter.js +27 -0
- package/src/supervisor.js +236 -0
- package/src/tee-writer.js +150 -0
- package/src/trace-collector.js +444 -0
- package/src/trace-github.js +473 -0
- package/src/trace-multi.js +101 -0
- package/src/trace-query.js +748 -0
- package/src/trace-render.js +211 -0
- package/src/trace-usage.js +249 -0
- package/src/fixture/assertions.js +0 -42
- package/src/fixture/cache.js +0 -50
- package/src/fixture/eval.js +0 -146
- package/src/fixture/index.js +0 -9
- package/src/fixture/pathway.js +0 -451
- package/src/fixture/services.js +0 -56
- package/src/mock/clients.js +0 -135
- package/src/mock/config.js +0 -45
- package/src/mock/data.js +0 -46
- package/src/mock/fs.js +0 -111
- package/src/mock/grpc.js +0 -94
- package/src/mock/http.js +0 -60
- package/src/mock/index.js +0 -36
- package/src/mock/infra.js +0 -219
- package/src/mock/logger.js +0 -42
- package/src/mock/observer.js +0 -74
- package/src/mock/resource-index.js +0 -95
- package/src/mock/service-callbacks.js +0 -39
- package/src/mock/services.js +0 -79
- package/src/mock/spy.js +0 -44
- package/src/mock/storage.js +0 -118
|
@@ -0,0 +1,688 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* BenchmarkRunner — sole orchestrator for a task-family benchmark run.
|
|
3
|
+
*
|
|
4
|
+
* Phases per (task, runIndex):
|
|
5
|
+
* 1. WorkdirManager.start → seed CWD + run pre-flight probe
|
|
6
|
+
* 2. Supervisor session (agent + supervisor) → produce traces + submission
|
|
7
|
+
* 3. Invariants.runInvariants → exit-code-driven verdict via fd-3 NDJSON
|
|
8
|
+
* 4. Judge.runJudge → Conclude-driven verdict mapped to pass/fail
|
|
9
|
+
* 5. WorkdirManager.teardown → process-group cleanup
|
|
10
|
+
*
|
|
11
|
+
* Cells run with bounded in-process concurrency (`CellScheduler`); `run()`
|
|
12
|
+
* yields records in **completion order**, not grid order. A single drain loop
|
|
13
|
+
* is the sole writer of `<output>/results.jsonl`, appending each record the
|
|
14
|
+
* moment its cell settles — that incremental append is the durability and
|
|
15
|
+
* crash-safety mechanism, so a killed run keeps every completed cell and there
|
|
16
|
+
* is no sidecar ledger. The iterator drives CLI stdout mirroring off the same
|
|
17
|
+
* stream.
|
|
18
|
+
*/
|
|
19
|
+
|
|
20
|
+
import { createInterface } from "node:readline";
|
|
21
|
+
import { join, resolve as resolvePath } from "node:path";
|
|
22
|
+
|
|
23
|
+
import { DEFAULT_ENV_ALLOWLIST, createRedactor } from "../redaction.js";
|
|
24
|
+
import { sumTraceCost } from "../cost.js";
|
|
25
|
+
import { createSupervisor } from "../supervisor.js";
|
|
26
|
+
import { installApm as defaultInstallApm } from "./apm-installer.js";
|
|
27
|
+
import { installNpm as defaultInstallNpm } from "./npm-installer.js";
|
|
28
|
+
import { runJudge } from "./judge.js";
|
|
29
|
+
import { validateResultRecord } from "./result.js";
|
|
30
|
+
import { runInvariants } from "./invariants.js";
|
|
31
|
+
import { assertJudgeProfileStaged, loadTaskFamily } from "./task-family.js";
|
|
32
|
+
import { createWorkdirManager } from "./workdir.js";
|
|
33
|
+
import { CellScheduler } from "./scheduler.js";
|
|
34
|
+
|
|
35
|
+
const BASE_TOOLS = [
|
|
36
|
+
"Bash",
|
|
37
|
+
"Read",
|
|
38
|
+
"Glob",
|
|
39
|
+
"Grep",
|
|
40
|
+
"Write",
|
|
41
|
+
"Edit",
|
|
42
|
+
"Agent",
|
|
43
|
+
"TodoWrite",
|
|
44
|
+
];
|
|
45
|
+
|
|
46
|
+
// Upper bound on a single supervised agent run. A run that produces no terminal
|
|
47
|
+
// message within this window is treated as a stall and recorded as an
|
|
48
|
+
// agentError, so the benchmark never hangs the event loop into a silent exit.
|
|
49
|
+
// Overridable per-runner via `watchdogMs` so a test can force a stall to fire
|
|
50
|
+
// without waiting the full 20 minutes.
|
|
51
|
+
const AGENT_WATCHDOG_MS = 20 * 60 * 1000;
|
|
52
|
+
|
|
53
|
+
/** Sole orchestrator for a task-family benchmark run. */
|
|
54
|
+
export class BenchmarkRunner {
|
|
55
|
+
/**
|
|
56
|
+
* @param {object} opts
|
|
57
|
+
* @param {import("./task-family.js").TaskFamily | string} opts.family
|
|
58
|
+
* @param {number} opts.runs - Runs per task (≥ 1).
|
|
59
|
+
* @param {string} opts.output - Run-output directory.
|
|
60
|
+
* @param {string} opts.agentModel
|
|
61
|
+
* @param {string} opts.supervisorModel
|
|
62
|
+
* @param {string} opts.judgeModel
|
|
63
|
+
* @param {{agent?: string, judge?: string}} [opts.profiles]
|
|
64
|
+
* @param {Function} opts.query - SDK query (injected for testability).
|
|
65
|
+
* @param {string[]} [opts.allowedTools] - Agent tool allowlist (default: BASE_TOOLS).
|
|
66
|
+
* @param {number} [opts.maxTurns] - Agent-under-test turn budget.
|
|
67
|
+
* @param {number} [opts.concurrency] - Max cells in flight (integer ≥ 1).
|
|
68
|
+
* Defaults to 1 as a defensive floor; the CLI always passes a resolved value.
|
|
69
|
+
* @param {{index: number, total: number}} [opts.shard] - Run only the cells
|
|
70
|
+
* assigned to shard `index` of `total` (1-based). Absent ≡ the whole grid
|
|
71
|
+
* (identity `1/1`).
|
|
72
|
+
* @param {number} [opts.watchdogMs] - Per-agent stall watchdog (ms). Defaults
|
|
73
|
+
* to `AGENT_WATCHDOG_MS`; injectable so tests can force a stall in-test.
|
|
74
|
+
* @param {number} [opts.termGraceMs] - SIGTERM→SIGKILL grace (ms) for the per-task process group.
|
|
75
|
+
* @param {Function} [opts.runAgent] - Test seam: replaces the agent-under-test
|
|
76
|
+
* session. Must return `{costUsd, turns, submission, agentError?}` and
|
|
77
|
+
* write a valid NDJSON trace to `workdir.agentTracePath`. Default uses
|
|
78
|
+
* `createAgentRunner` with the harness `BASE_TOOLS` allowlist. Internal
|
|
79
|
+
* testing only — not part of the public API.
|
|
80
|
+
* @param {import("@forwardimpact/libutil/runtime").Runtime} opts.runtime -
|
|
81
|
+
* Injected ambient collaborators (`fs`, `subprocess`, `clock`, `proc`),
|
|
82
|
+
* threaded into the installers, workdir manager, invariants, and judge.
|
|
83
|
+
* @param {Function} [opts.runInvariants] - Test seam: replaces `runInvariants`.
|
|
84
|
+
* Same contract as `runInvariants(task, ctx, runtime)`. Internal testing only.
|
|
85
|
+
* @param {Function} [opts.runJudge] - Test seam: replaces `runJudge`. Same
|
|
86
|
+
* contract as `runJudge(task, workdir, invariants, deps)` (deps carries
|
|
87
|
+
* `runtime`). Internal testing only.
|
|
88
|
+
* @param {Function} [opts.installApm] - Test seam: replaces `installApm`.
|
|
89
|
+
* Same contract as `installApm(family, outputDir, runtime)`. Lets tests
|
|
90
|
+
* inject a fake subprocess (or skip the install entirely) so the suite
|
|
91
|
+
* never shells out to a real `apm` binary. Internal testing only.
|
|
92
|
+
* @param {Function} [opts.installNpm] - Test seam: replaces `installNpm`.
|
|
93
|
+
* Same contract as `installNpm(family, stagingDir, runtime)`. Internal
|
|
94
|
+
* testing only.
|
|
95
|
+
*/
|
|
96
|
+
constructor({
|
|
97
|
+
family,
|
|
98
|
+
runs,
|
|
99
|
+
output,
|
|
100
|
+
agentModel,
|
|
101
|
+
supervisorModel,
|
|
102
|
+
judgeModel,
|
|
103
|
+
profiles,
|
|
104
|
+
query,
|
|
105
|
+
allowedTools,
|
|
106
|
+
maxTurns,
|
|
107
|
+
concurrency,
|
|
108
|
+
watchdogMs,
|
|
109
|
+
shard,
|
|
110
|
+
task,
|
|
111
|
+
skillsFrom,
|
|
112
|
+
termGraceMs,
|
|
113
|
+
runtime,
|
|
114
|
+
// Test seams — default to the real implementations.
|
|
115
|
+
runAgent,
|
|
116
|
+
runInvariants: runInvariantsHook,
|
|
117
|
+
runJudge: runJudgeHook,
|
|
118
|
+
installApm: installApmHook,
|
|
119
|
+
installNpm: installNpmHook,
|
|
120
|
+
}) {
|
|
121
|
+
validateRunnerArgs({ family, runs, output, agentModel, query, runtime });
|
|
122
|
+
this.runtime = runtime;
|
|
123
|
+
this.familyInput = family;
|
|
124
|
+
this.runs = runs;
|
|
125
|
+
this.output = output;
|
|
126
|
+
this.agentModel = agentModel;
|
|
127
|
+
this.supervisorModel = supervisorModel;
|
|
128
|
+
this.judgeModel = judgeModel;
|
|
129
|
+
this.allowedTools = allowedTools ?? BASE_TOOLS;
|
|
130
|
+
this.profiles = {
|
|
131
|
+
agent: profiles?.agent ?? null,
|
|
132
|
+
judge: profiles?.judge ?? null,
|
|
133
|
+
};
|
|
134
|
+
this.query = query;
|
|
135
|
+
this.maxTurns = maxTurns;
|
|
136
|
+
this.concurrency = concurrency ?? 1;
|
|
137
|
+
this.watchdogMs = watchdogMs ?? AGENT_WATCHDOG_MS;
|
|
138
|
+
this.shard = shard ?? null;
|
|
139
|
+
this.taskFilter = task ?? null;
|
|
140
|
+
this.skillsFrom = skillsFrom ?? null;
|
|
141
|
+
this.termGraceMs = termGraceMs;
|
|
142
|
+
this._runAgentHook = runAgent ?? null;
|
|
143
|
+
this._runInvariantsHook = runInvariantsHook ?? runInvariants;
|
|
144
|
+
this._runJudgeHook = runJudgeHook ?? runJudge;
|
|
145
|
+
this._installApmHook = installApmHook ?? defaultInstallApm;
|
|
146
|
+
this._installNpmHook = installNpmHook ?? defaultInstallNpm;
|
|
147
|
+
}
|
|
148
|
+
|
|
149
|
+
/**
|
|
150
|
+
* Yield one ResultRecord per (task, runIndex).
|
|
151
|
+
* @returns {AsyncGenerator<object>}
|
|
152
|
+
*/
|
|
153
|
+
async *run() {
|
|
154
|
+
const runtime = this.runtime;
|
|
155
|
+
const family =
|
|
156
|
+
typeof this.familyInput === "string"
|
|
157
|
+
? await loadTaskFamily(this.familyInput, runtime)
|
|
158
|
+
: this.familyInput;
|
|
159
|
+
|
|
160
|
+
await runtime.fs.mkdir(this.output, { recursive: true });
|
|
161
|
+
const { stagingDir, skillSetHash, judgeProfilesDir } =
|
|
162
|
+
await this._installApmHook(family, this.output, runtime, {
|
|
163
|
+
skillsFrom: this.skillsFrom,
|
|
164
|
+
});
|
|
165
|
+
await this._installNpmHook(family, stagingDir, runtime);
|
|
166
|
+
|
|
167
|
+
let tasks = family.tasks();
|
|
168
|
+
if (this.taskFilter) {
|
|
169
|
+
const matched = tasks.filter((t) => t.id === this.taskFilter);
|
|
170
|
+
if (matched.length === 0) {
|
|
171
|
+
const available = tasks.map((t) => t.id).join(", ");
|
|
172
|
+
throw new Error(
|
|
173
|
+
`no task '${this.taskFilter}' in family; available: ${available}`,
|
|
174
|
+
);
|
|
175
|
+
}
|
|
176
|
+
tasks = matched;
|
|
177
|
+
}
|
|
178
|
+
if (this.profiles.judge) {
|
|
179
|
+
await assertJudgeProfileStaged(
|
|
180
|
+
family,
|
|
181
|
+
judgeProfilesDir,
|
|
182
|
+
this.profiles.judge,
|
|
183
|
+
runtime,
|
|
184
|
+
);
|
|
185
|
+
}
|
|
186
|
+
|
|
187
|
+
const wm = createWorkdirManager({
|
|
188
|
+
stagingDir,
|
|
189
|
+
runOutputDir: this.output,
|
|
190
|
+
termGraceMs: this.termGraceMs,
|
|
191
|
+
familyRootPath: family.rootPath,
|
|
192
|
+
runtime,
|
|
193
|
+
});
|
|
194
|
+
|
|
195
|
+
const allCells = enumerateCells(tasks, this.runs);
|
|
196
|
+
// Sharding selects a deterministic subset of the grid; an unsharded run is
|
|
197
|
+
// the identity 1/1. A high-index shard may select zero cells — a valid run
|
|
198
|
+
// whose results.jsonl ends up empty.
|
|
199
|
+
const cells = this.shard
|
|
200
|
+
? selectShard(allCells, this.shard.index, this.shard.total)
|
|
201
|
+
: allCells;
|
|
202
|
+
const scheduler = new CellScheduler({
|
|
203
|
+
concurrency: this.concurrency,
|
|
204
|
+
runCell: (cell) =>
|
|
205
|
+
this.#runOne(
|
|
206
|
+
family,
|
|
207
|
+
wm,
|
|
208
|
+
cell.task,
|
|
209
|
+
cell.runIndex,
|
|
210
|
+
skillSetHash,
|
|
211
|
+
judgeProfilesDir,
|
|
212
|
+
),
|
|
213
|
+
});
|
|
214
|
+
|
|
215
|
+
const resultsPath = join(this.output, "results.jsonl");
|
|
216
|
+
const resultsStream = runtime.fs.createWriteStream(resultsPath, {
|
|
217
|
+
flags: "a",
|
|
218
|
+
});
|
|
219
|
+
// Single-writer drain: the scheduler runs up to `concurrency` cells at
|
|
220
|
+
// once and pushes each settled record here in completion order. This loop
|
|
221
|
+
// is the sole writer of `results.jsonl` — workers never touch the stream —
|
|
222
|
+
// and the per-completion append is the crash-safety mechanism.
|
|
223
|
+
try {
|
|
224
|
+
for await (const record of scheduler.run(cells)) {
|
|
225
|
+
await writeRecord(resultsStream, record);
|
|
226
|
+
yield record;
|
|
227
|
+
}
|
|
228
|
+
} finally {
|
|
229
|
+
await new Promise((r) => resultsStream.end(r));
|
|
230
|
+
}
|
|
231
|
+
}
|
|
232
|
+
|
|
233
|
+
async #runOne(family, wm, task, runIndex, skillSetHash, judgeProfilesDir) {
|
|
234
|
+
const t0 = this.runtime.clock.now();
|
|
235
|
+
let workdir;
|
|
236
|
+
try {
|
|
237
|
+
workdir = await wm.start(task, runIndex);
|
|
238
|
+
return await this.#executeCell({
|
|
239
|
+
family,
|
|
240
|
+
workdir,
|
|
241
|
+
task,
|
|
242
|
+
runIndex,
|
|
243
|
+
skillSetHash,
|
|
244
|
+
judgeProfilesDir,
|
|
245
|
+
t0,
|
|
246
|
+
});
|
|
247
|
+
} catch (e) {
|
|
248
|
+
// `wm.start()` (port acquire + workdir/env seeding) is the one throw site
|
|
249
|
+
// not caught inside `#executeCell`. Turn it into the runner's own fallback
|
|
250
|
+
// record so `#runOne` never rejects — the scheduler's one-record-per-cell
|
|
251
|
+
// contract depends on that. The fallback is schema-skipped by `report`,
|
|
252
|
+
// the same as any other runner-side schema failure.
|
|
253
|
+
return {
|
|
254
|
+
taskId: task.id,
|
|
255
|
+
runIndex,
|
|
256
|
+
verdict: "fail",
|
|
257
|
+
schemaError: `cell setup failed: ${e.message ?? String(e)}`,
|
|
258
|
+
};
|
|
259
|
+
} finally {
|
|
260
|
+
if (workdir) await wm.teardown(workdir).catch(() => {});
|
|
261
|
+
}
|
|
262
|
+
}
|
|
263
|
+
|
|
264
|
+
/**
|
|
265
|
+
* Run one cell's lifecycle against an already-started workdir: preflight
|
|
266
|
+
* gate → supervised agent → invariants → judge → assembled record. Extracted
|
|
267
|
+
* from `#runOne` so the start/teardown/error wrapper stays under the
|
|
268
|
+
* complexity ceiling.
|
|
269
|
+
*/
|
|
270
|
+
async #executeCell({
|
|
271
|
+
family,
|
|
272
|
+
workdir,
|
|
273
|
+
task,
|
|
274
|
+
runIndex,
|
|
275
|
+
skillSetHash,
|
|
276
|
+
judgeProfilesDir,
|
|
277
|
+
t0,
|
|
278
|
+
}) {
|
|
279
|
+
if (workdir.preflightError) {
|
|
280
|
+
const record = this.#buildPreflightFailureRecord({
|
|
281
|
+
task,
|
|
282
|
+
runIndex,
|
|
283
|
+
workdir,
|
|
284
|
+
skillSetHash,
|
|
285
|
+
familyRevision: family.familyRevision,
|
|
286
|
+
durationMs: this.runtime.clock.now() - t0,
|
|
287
|
+
});
|
|
288
|
+
return this.#validateOrFallback(record, resultsRecordKey(task, runIndex));
|
|
289
|
+
}
|
|
290
|
+
{
|
|
291
|
+
const agentRun = await this.#runAgentSafe(task, workdir);
|
|
292
|
+
const { costUsd, turns, submission, agentError } = agentRun;
|
|
293
|
+
const breakdown = agentRun.costBreakdown ?? { agent: 0, supervisor: 0 };
|
|
294
|
+
const invariants = await this._runInvariantsHook(
|
|
295
|
+
task,
|
|
296
|
+
{
|
|
297
|
+
cwd: workdir.cwd,
|
|
298
|
+
port: workdir.port,
|
|
299
|
+
runDir: workdir.runDir,
|
|
300
|
+
familyDir: family.rootPath,
|
|
301
|
+
},
|
|
302
|
+
this.runtime,
|
|
303
|
+
);
|
|
304
|
+
let judgeVerdict = null;
|
|
305
|
+
let judgeCost = 0;
|
|
306
|
+
if (task.paths.judge) {
|
|
307
|
+
const judgeContext = await this.#buildJudgeContext(
|
|
308
|
+
task,
|
|
309
|
+
workdir,
|
|
310
|
+
skillSetHash,
|
|
311
|
+
);
|
|
312
|
+
const judgeResult = await this._runJudgeHook(
|
|
313
|
+
task,
|
|
314
|
+
workdir,
|
|
315
|
+
invariants,
|
|
316
|
+
{
|
|
317
|
+
query: this.query,
|
|
318
|
+
model: this.judgeModel,
|
|
319
|
+
judgeProfile: this.profiles.judge ?? undefined,
|
|
320
|
+
profilesDir: judgeProfilesDir,
|
|
321
|
+
runtime: this.runtime,
|
|
322
|
+
},
|
|
323
|
+
judgeContext,
|
|
324
|
+
);
|
|
325
|
+
judgeCost = judgeResult.costUsd ?? 0;
|
|
326
|
+
// The record's judgeVerdict carries only the verdict + summary; the
|
|
327
|
+
// judge's cost is folded into costUsd / costBreakdown instead.
|
|
328
|
+
judgeVerdict = {
|
|
329
|
+
verdict: judgeResult.verdict,
|
|
330
|
+
summary: judgeResult.summary,
|
|
331
|
+
};
|
|
332
|
+
}
|
|
333
|
+
const verdict =
|
|
334
|
+
invariants.verdict === "pass" &&
|
|
335
|
+
(judgeVerdict === null || judgeVerdict.verdict === "pass")
|
|
336
|
+
? "pass"
|
|
337
|
+
: "fail";
|
|
338
|
+
const record = {
|
|
339
|
+
taskId: task.id,
|
|
340
|
+
runIndex,
|
|
341
|
+
verdict,
|
|
342
|
+
invariants,
|
|
343
|
+
submission,
|
|
344
|
+
...(judgeVerdict && { judgeVerdict }),
|
|
345
|
+
costUsd: costUsd + judgeCost,
|
|
346
|
+
costBreakdown: {
|
|
347
|
+
agent: breakdown.agent ?? 0,
|
|
348
|
+
supervisor: breakdown.supervisor ?? 0,
|
|
349
|
+
judge: judgeCost,
|
|
350
|
+
},
|
|
351
|
+
turns,
|
|
352
|
+
agentTracePath: workdir.agentTracePath,
|
|
353
|
+
supervisorTracePath: workdir.supervisorTracePath,
|
|
354
|
+
judgeTracePath: workdir.judgeTracePath,
|
|
355
|
+
profiles: {
|
|
356
|
+
agent: this.profiles.agent,
|
|
357
|
+
supervisor: null,
|
|
358
|
+
judge: this.profiles.judge,
|
|
359
|
+
},
|
|
360
|
+
model: {
|
|
361
|
+
agent: this.agentModel,
|
|
362
|
+
supervisor: this.supervisorModel,
|
|
363
|
+
judge: this.judgeModel,
|
|
364
|
+
},
|
|
365
|
+
skillSetHash,
|
|
366
|
+
familyRevision: family.familyRevision,
|
|
367
|
+
durationMs: this.runtime.clock.now() - t0,
|
|
368
|
+
...(agentError && { agentError }),
|
|
369
|
+
};
|
|
370
|
+
return this.#validateOrFallback(record, resultsRecordKey(task, runIndex));
|
|
371
|
+
}
|
|
372
|
+
}
|
|
373
|
+
|
|
374
|
+
/**
|
|
375
|
+
* Dispatch to either the injected hook or the default `#runAgent`. Either
|
|
376
|
+
* path can throw; catch here so a thrown error becomes an `agentError` on
|
|
377
|
+
* the record (spec criterion 1: records on agent failure) rather than
|
|
378
|
+
* aborting the whole iterator.
|
|
379
|
+
*/
|
|
380
|
+
async #runAgentSafe(task, workdir) {
|
|
381
|
+
try {
|
|
382
|
+
if (this._runAgentHook) {
|
|
383
|
+
const r = await this._runAgentHook(task, workdir, this);
|
|
384
|
+
return { agentError: null, ...r };
|
|
385
|
+
}
|
|
386
|
+
return await this.#runAgent(task, workdir);
|
|
387
|
+
} catch (e) {
|
|
388
|
+
return {
|
|
389
|
+
costUsd: 0,
|
|
390
|
+
costBreakdown: { agent: 0, supervisor: 0 },
|
|
391
|
+
turns: 0,
|
|
392
|
+
submission: "",
|
|
393
|
+
agentError: { message: e.message ?? String(e), aborted: false },
|
|
394
|
+
};
|
|
395
|
+
}
|
|
396
|
+
}
|
|
397
|
+
|
|
398
|
+
/**
|
|
399
|
+
* Run the agent-under-test under a Supervisor. The supervisor writes
|
|
400
|
+
* a combined tagged NDJSON trace; after the session we split it into
|
|
401
|
+
* agent.ndjson and supervisor.ndjson and extract cost/turns/submission.
|
|
402
|
+
*/
|
|
403
|
+
async #runAgent(task, workdir) {
|
|
404
|
+
const fs = this.runtime.fs;
|
|
405
|
+
const combinedPath = join(workdir.runDir, ".combined.ndjson");
|
|
406
|
+
const combinedStream = fs.createWriteStream(combinedPath);
|
|
407
|
+
const supervisorInstructions = task.paths.supervisor
|
|
408
|
+
? await fs.readFile(task.paths.supervisor, "utf8").catch(() => null)
|
|
409
|
+
: null;
|
|
410
|
+
const supervisor = createSupervisor({
|
|
411
|
+
supervisorCwd: workdir.cwd,
|
|
412
|
+
agentCwd: workdir.cwd,
|
|
413
|
+
query: this.query,
|
|
414
|
+
output: combinedStream,
|
|
415
|
+
agentModel: this.agentModel,
|
|
416
|
+
supervisorModel: this.supervisorModel,
|
|
417
|
+
maxTurns: this.maxTurns ?? 50,
|
|
418
|
+
allowedTools: this.allowedTools,
|
|
419
|
+
...(this.profiles.agent && { agentProfile: this.profiles.agent }),
|
|
420
|
+
...(supervisorInstructions && { taskAmend: supervisorInstructions }),
|
|
421
|
+
redactor: createRedactor({
|
|
422
|
+
allowlist: [...DEFAULT_ENV_ALLOWLIST, ...(workdir.envNames ?? [])],
|
|
423
|
+
runtime: this.runtime,
|
|
424
|
+
}),
|
|
425
|
+
runtime: this.runtime,
|
|
426
|
+
});
|
|
427
|
+
const instructions = await fs.readFile(task.paths.instructions, "utf8");
|
|
428
|
+
let agentError = null;
|
|
429
|
+
// Watchdog: a supervised session can hang without settling (e.g. the agent
|
|
430
|
+
// SDK subprocess exits without a terminal message), which would empty the
|
|
431
|
+
// event loop and exit the process mid-run with zero records. Race the run
|
|
432
|
+
// against a bounded timer so a stall becomes an `agentError` record instead
|
|
433
|
+
// of a silent exit; the timer also keeps the loop alive until it fires.
|
|
434
|
+
let watchdog;
|
|
435
|
+
try {
|
|
436
|
+
const result = await Promise.race([
|
|
437
|
+
supervisor.run(instructions),
|
|
438
|
+
new Promise((_, reject) => {
|
|
439
|
+
watchdog = this.runtime.clock.setTimeout(
|
|
440
|
+
() =>
|
|
441
|
+
reject(
|
|
442
|
+
new Error(
|
|
443
|
+
`agent run produced no result within ${this.watchdogMs}ms (possible stall)`,
|
|
444
|
+
),
|
|
445
|
+
),
|
|
446
|
+
this.watchdogMs,
|
|
447
|
+
);
|
|
448
|
+
}),
|
|
449
|
+
]);
|
|
450
|
+
if (!result.success && !result.concluded) {
|
|
451
|
+
agentError = { message: "supervisor did not succeed", aborted: false };
|
|
452
|
+
}
|
|
453
|
+
} catch (e) {
|
|
454
|
+
agentError = { message: e.message ?? String(e), aborted: false };
|
|
455
|
+
} finally {
|
|
456
|
+
this.runtime.clock.clearTimeout(watchdog);
|
|
457
|
+
await new Promise((r) => combinedStream.end(r));
|
|
458
|
+
}
|
|
459
|
+
const summary = await splitAndSummarize(
|
|
460
|
+
this.runtime,
|
|
461
|
+
combinedPath,
|
|
462
|
+
workdir.agentTracePath,
|
|
463
|
+
workdir.supervisorTracePath,
|
|
464
|
+
);
|
|
465
|
+
// Cost is summed across every participant's result events from the one
|
|
466
|
+
// combined trace, attributed per source. Read before unlinking.
|
|
467
|
+
const combined = await fs.readFile(combinedPath, "utf8");
|
|
468
|
+
const { totalCostUsd, bySource } = sumTraceCost(combined.split("\n"));
|
|
469
|
+
await fs.unlink(combinedPath).catch(() => {});
|
|
470
|
+
return {
|
|
471
|
+
...summary,
|
|
472
|
+
costUsd: totalCostUsd,
|
|
473
|
+
costBreakdown: {
|
|
474
|
+
agent: bySource.agent ?? 0,
|
|
475
|
+
supervisor: bySource.supervisor ?? 0,
|
|
476
|
+
},
|
|
477
|
+
agentError,
|
|
478
|
+
};
|
|
479
|
+
}
|
|
480
|
+
|
|
481
|
+
async #buildJudgeContext(task, workdir, skillSetHash) {
|
|
482
|
+
const fs = this.runtime.fs;
|
|
483
|
+
const agentInstructions = await fs.readFile(
|
|
484
|
+
task.paths.instructions,
|
|
485
|
+
"utf8",
|
|
486
|
+
);
|
|
487
|
+
let agentProfile = "";
|
|
488
|
+
if (this.profiles.agent) {
|
|
489
|
+
const profilePath = resolvePath(
|
|
490
|
+
workdir.cwd,
|
|
491
|
+
".claude/agents",
|
|
492
|
+
`${this.profiles.agent}.md`,
|
|
493
|
+
);
|
|
494
|
+
agentProfile = await fs.readFile(profilePath, "utf8").catch(() => "");
|
|
495
|
+
}
|
|
496
|
+
return { agentInstructions, agentProfile, skillSetHash };
|
|
497
|
+
}
|
|
498
|
+
|
|
499
|
+
#buildPreflightFailureRecord({
|
|
500
|
+
task,
|
|
501
|
+
runIndex,
|
|
502
|
+
workdir,
|
|
503
|
+
skillSetHash,
|
|
504
|
+
familyRevision,
|
|
505
|
+
durationMs,
|
|
506
|
+
}) {
|
|
507
|
+
return {
|
|
508
|
+
taskId: task.id,
|
|
509
|
+
runIndex,
|
|
510
|
+
verdict: "fail",
|
|
511
|
+
costUsd: 0,
|
|
512
|
+
turns: 0,
|
|
513
|
+
preflightError: workdir.preflightError,
|
|
514
|
+
profiles: {
|
|
515
|
+
agent: this.profiles.agent,
|
|
516
|
+
supervisor: null,
|
|
517
|
+
judge: this.profiles.judge,
|
|
518
|
+
},
|
|
519
|
+
model: {
|
|
520
|
+
agent: this.agentModel,
|
|
521
|
+
supervisor: this.supervisorModel,
|
|
522
|
+
judge: this.judgeModel,
|
|
523
|
+
},
|
|
524
|
+
skillSetHash,
|
|
525
|
+
familyRevision,
|
|
526
|
+
durationMs,
|
|
527
|
+
agentTracePath: workdir.agentTracePath,
|
|
528
|
+
supervisorTracePath: workdir.supervisorTracePath,
|
|
529
|
+
judgeTracePath: workdir.judgeTracePath,
|
|
530
|
+
};
|
|
531
|
+
}
|
|
532
|
+
|
|
533
|
+
#validateOrFallback(record, key) {
|
|
534
|
+
try {
|
|
535
|
+
validateResultRecord(record);
|
|
536
|
+
return record;
|
|
537
|
+
} catch (e) {
|
|
538
|
+
// The runner constructed the record — a schema failure is a real bug,
|
|
539
|
+
// not bad family input. Emit a noisy fallback so the iterator stays
|
|
540
|
+
// consumable and the agent budget isn't silently dropped.
|
|
541
|
+
return {
|
|
542
|
+
taskId: record.taskId ?? key.taskId,
|
|
543
|
+
runIndex: record.runIndex ?? key.runIndex,
|
|
544
|
+
verdict: "fail",
|
|
545
|
+
schemaError: e.message ?? String(e),
|
|
546
|
+
};
|
|
547
|
+
}
|
|
548
|
+
}
|
|
549
|
+
}
|
|
550
|
+
|
|
551
|
+
/**
|
|
552
|
+
* Flatten the grid into a stable ordered cell list, task-major /
|
|
553
|
+
* runIndex-minor. Load-bearing ordering: Part 02's round-robin shard balance
|
|
554
|
+
* depends on a task's runIndexes being adjacent in this list. Single source of
|
|
555
|
+
* the cell list for both the scheduler and the shard selector.
|
|
556
|
+
* @param {import("./task-family.js").Task[]} tasks
|
|
557
|
+
* @param {number} runs
|
|
558
|
+
* @returns {{task: import("./task-family.js").Task, runIndex: number}[]}
|
|
559
|
+
*/
|
|
560
|
+
export function enumerateCells(tasks, runs) {
|
|
561
|
+
const cells = [];
|
|
562
|
+
for (const task of tasks)
|
|
563
|
+
for (let runIndex = 0; runIndex < runs; runIndex++)
|
|
564
|
+
cells.push({ task, runIndex });
|
|
565
|
+
return cells;
|
|
566
|
+
}
|
|
567
|
+
|
|
568
|
+
/**
|
|
569
|
+
* Round-robin partition of the enumerated cells: the cell at position `p` runs
|
|
570
|
+
* iff `p % total === i - 1`. `i` is 1-based (Playwright-style). The union over
|
|
571
|
+
* `i ∈ 1..total` is the exact grid, each cell once; when `total > cells.length`
|
|
572
|
+
* the high-index shards select **zero** cells — a valid run. Because
|
|
573
|
+
* `enumerateCells` is task-major, a task's run indexes are adjacent, so
|
|
574
|
+
* round-robin spreads them across shards rather than handing one shard a slow
|
|
575
|
+
* task's whole run block.
|
|
576
|
+
* @param {{task: object, runIndex: number}[]} cells
|
|
577
|
+
* @param {number} i - 1-based shard index.
|
|
578
|
+
* @param {number} total - Shard count.
|
|
579
|
+
* @returns {{task: object, runIndex: number}[]}
|
|
580
|
+
*/
|
|
581
|
+
export function selectShard(cells, i, total) {
|
|
582
|
+
return cells.filter((_, p) => p % total === i - 1);
|
|
583
|
+
}
|
|
584
|
+
|
|
585
|
+
/**
|
|
586
|
+
* Validate the required BenchmarkRunner constructor arguments. Extracted from
|
|
587
|
+
* the constructor to keep its cognitive complexity under the lint ceiling.
|
|
588
|
+
*/
|
|
589
|
+
function validateRunnerArgs({
|
|
590
|
+
family,
|
|
591
|
+
runs,
|
|
592
|
+
output,
|
|
593
|
+
agentModel,
|
|
594
|
+
query,
|
|
595
|
+
runtime,
|
|
596
|
+
}) {
|
|
597
|
+
if (!family) throw new Error("family is required");
|
|
598
|
+
if (!Number.isInteger(runs) || runs < 1)
|
|
599
|
+
throw new Error("runs must be an integer ≥ 1");
|
|
600
|
+
if (!output) throw new Error("output is required");
|
|
601
|
+
if (!agentModel) throw new Error("agentModel is required");
|
|
602
|
+
if (!query) throw new Error("query is required");
|
|
603
|
+
if (!runtime) throw new Error("runtime is required");
|
|
604
|
+
}
|
|
605
|
+
|
|
606
|
+
function resultsRecordKey(task, runIndex) {
|
|
607
|
+
return { taskId: task.id, runIndex };
|
|
608
|
+
}
|
|
609
|
+
|
|
610
|
+
async function writeRecord(stream, record) {
|
|
611
|
+
const line = JSON.stringify(record) + "\n";
|
|
612
|
+
await new Promise((res, rej) => {
|
|
613
|
+
stream.write(line, (err) => (err ? rej(err) : res()));
|
|
614
|
+
});
|
|
615
|
+
}
|
|
616
|
+
|
|
617
|
+
/**
|
|
618
|
+
* Split the combined supervisor trace into agent and supervisor files and
|
|
619
|
+
* extract turn count and submission in a single pass. Agent-source events go
|
|
620
|
+
* to `agentPath`; supervisor and orchestrator events go to `supervisorPath`.
|
|
621
|
+
*
|
|
622
|
+
* Cost is deliberately not summed here — the caller derives it from the same
|
|
623
|
+
* combined trace via `sumTraceCost`, so there is one cost path across the
|
|
624
|
+
* benchmark, callback, and `fit-trace cost` consumers.
|
|
625
|
+
*/
|
|
626
|
+
// biome-ignore lint/complexity/noExcessiveCognitiveComplexity: stream-splitting state machine
|
|
627
|
+
async function splitAndSummarize(
|
|
628
|
+
runtime,
|
|
629
|
+
combinedPath,
|
|
630
|
+
agentPath,
|
|
631
|
+
supervisorPath,
|
|
632
|
+
) {
|
|
633
|
+
const fs = runtime.fs;
|
|
634
|
+
const agentStream = fs.createWriteStream(agentPath);
|
|
635
|
+
const supStream = fs.createWriteStream(supervisorPath);
|
|
636
|
+
const rl = createInterface({
|
|
637
|
+
input: fs.createReadStream(combinedPath),
|
|
638
|
+
crlfDelay: Infinity,
|
|
639
|
+
});
|
|
640
|
+
let turns = 0;
|
|
641
|
+
let submission = "";
|
|
642
|
+
for await (const line of rl) {
|
|
643
|
+
if (!line.trim()) continue;
|
|
644
|
+
let event;
|
|
645
|
+
try {
|
|
646
|
+
event = JSON.parse(line);
|
|
647
|
+
} catch {
|
|
648
|
+
continue;
|
|
649
|
+
}
|
|
650
|
+
const target = event.source === "agent" ? agentStream : supStream;
|
|
651
|
+
target.write(line + "\n");
|
|
652
|
+
const inner = event.event;
|
|
653
|
+
if (!inner) continue;
|
|
654
|
+
if (event.source === "agent" && inner.type === "assistant") {
|
|
655
|
+
const text = extractText(inner);
|
|
656
|
+
if (text) submission = text;
|
|
657
|
+
}
|
|
658
|
+
if (event.source === "orchestrator" && inner.type === "summary") {
|
|
659
|
+
turns = inner.turns ?? 0;
|
|
660
|
+
}
|
|
661
|
+
}
|
|
662
|
+
await Promise.all([
|
|
663
|
+
new Promise((r) => agentStream.end(r)),
|
|
664
|
+
new Promise((r) => supStream.end(r)),
|
|
665
|
+
]);
|
|
666
|
+
return { turns, submission };
|
|
667
|
+
}
|
|
668
|
+
|
|
669
|
+
function extractText(inner) {
|
|
670
|
+
const content = inner.message?.content ?? inner.content;
|
|
671
|
+
if (!Array.isArray(content)) return null;
|
|
672
|
+
for (let i = content.length - 1; i >= 0; i--) {
|
|
673
|
+
if (content[i].type === "text" && content[i].text) return content[i].text;
|
|
674
|
+
}
|
|
675
|
+
return null;
|
|
676
|
+
}
|
|
677
|
+
|
|
678
|
+
/**
|
|
679
|
+
* Factory function — wires real dependencies.
|
|
680
|
+
* @param {ConstructorParameters<typeof BenchmarkRunner>[0]} opts
|
|
681
|
+
* @returns {BenchmarkRunner}
|
|
682
|
+
*/
|
|
683
|
+
export function createBenchmarkRunner(opts) {
|
|
684
|
+
return new BenchmarkRunner(opts);
|
|
685
|
+
}
|
|
686
|
+
|
|
687
|
+
// Internal exports used by tests.
|
|
688
|
+
export const __BASE_TOOLS = BASE_TOOLS;
|