@forwardimpact/libharness 0.1.22 → 1.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -201
- package/README.md +196 -80
- package/bin/fit-benchmark.js +44 -0
- package/bin/fit-harness.js +358 -0
- package/bin/fit-selfedit.js +165 -0
- package/bin/fit-trace.js +510 -0
- package/package.json +41 -11
- package/src/agent-runner.js +256 -0
- package/src/benchmark/apm-installer.js +207 -0
- package/src/benchmark/env-loader.js +158 -0
- package/src/benchmark/hook-env.js +40 -0
- package/src/benchmark/invariants.js +141 -0
- package/src/benchmark/judge.js +187 -0
- package/src/benchmark/npm-installer.js +87 -0
- package/src/benchmark/report.js +522 -0
- package/src/benchmark/result.js +127 -0
- package/src/benchmark/runner.js +583 -0
- package/src/benchmark/task-family.js +260 -0
- package/src/benchmark/workdir.js +298 -0
- package/src/commands/assert.js +153 -0
- package/src/commands/benchmark-definition.js +165 -0
- package/src/commands/benchmark-invariants.js +73 -0
- package/src/commands/benchmark-report.js +51 -0
- package/src/commands/benchmark-run.js +111 -0
- package/src/commands/by-discussion.js +94 -0
- package/src/commands/callback.js +119 -0
- package/src/commands/discuss.js +132 -0
- package/src/commands/facilitate.js +123 -0
- package/src/commands/output.js +36 -0
- package/src/commands/run.js +152 -0
- package/src/commands/supervise.js +136 -0
- package/src/commands/task-input.js +54 -0
- package/src/commands/tee.js +53 -0
- package/src/commands/trace.js +630 -0
- package/src/commands/work-tracker.js +35 -0
- package/src/cost.js +79 -0
- package/src/discuss-tools.js +173 -0
- package/src/discusser.js +394 -0
- package/src/events/github.js +161 -0
- package/src/facilitator.js +205 -0
- package/src/inbox-poller.js +81 -0
- package/src/index.js +72 -2
- package/src/judge.js +210 -0
- package/src/message-bus.js +118 -0
- package/src/orchestration-loop.js +330 -0
- package/src/orchestration-toolkit.js +441 -0
- package/src/orchestrator-helpers.js +23 -0
- package/src/profile-prompt.js +266 -0
- package/src/redaction.js +253 -0
- package/src/render/line-renderer.js +54 -0
- package/src/render/orchestrator-filter.js +19 -0
- package/src/render/palette.js +63 -0
- package/src/render/tool-hints.js +154 -0
- package/src/render/turn-renderer.js +96 -0
- package/src/reply-emitter.js +47 -0
- package/src/sequence-counter.js +21 -0
- package/src/signature-filter.js +27 -0
- package/src/supervisor.js +236 -0
- package/src/tee-writer.js +150 -0
- package/src/trace-collector.js +444 -0
- package/src/trace-github.js +473 -0
- package/src/trace-multi.js +101 -0
- package/src/trace-query.js +748 -0
- package/src/trace-render.js +211 -0
- package/src/trace-usage.js +249 -0
- package/src/fixture/assertions.js +0 -42
- package/src/fixture/cache.js +0 -50
- package/src/fixture/eval.js +0 -146
- package/src/fixture/index.js +0 -9
- package/src/fixture/pathway.js +0 -451
- package/src/fixture/services.js +0 -56
- package/src/mock/clients.js +0 -135
- package/src/mock/config.js +0 -45
- package/src/mock/data.js +0 -46
- package/src/mock/fs.js +0 -111
- package/src/mock/grpc.js +0 -94
- package/src/mock/http.js +0 -60
- package/src/mock/index.js +0 -36
- package/src/mock/infra.js +0 -219
- package/src/mock/logger.js +0 -42
- package/src/mock/observer.js +0 -74
- package/src/mock/resource-index.js +0 -95
- package/src/mock/service-callbacks.js +0 -39
- package/src/mock/services.js +0 -79
- package/src/mock/spy.js +0 -44
- package/src/mock/storage.js +0 -118
|
@@ -0,0 +1,583 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* BenchmarkRunner — sole orchestrator for a task-family benchmark run.
|
|
3
|
+
*
|
|
4
|
+
* Phases per (task, runIndex):
|
|
5
|
+
* 1. WorkdirManager.start → seed CWD + run pre-flight probe
|
|
6
|
+
* 2. Supervisor session (agent + supervisor) → produce traces + submission
|
|
7
|
+
* 3. Invariants.runInvariants → exit-code-driven verdict via fd-3 NDJSON
|
|
8
|
+
* 4. Judge.runJudge → Conclude-driven verdict mapped to pass/fail
|
|
9
|
+
* 5. WorkdirManager.teardown → process-group cleanup
|
|
10
|
+
*
|
|
11
|
+
* Results stream as an async iterable AND are appended to
|
|
12
|
+
* `<output>/results.jsonl` for durability. The two paths are different
|
|
13
|
+
* consumers of the same record — the iterator drives CLI stdout mirroring,
|
|
14
|
+
* the JSONL append is the system of record.
|
|
15
|
+
*/
|
|
16
|
+
|
|
17
|
+
import { createInterface } from "node:readline";
|
|
18
|
+
import { join, resolve as resolvePath } from "node:path";
|
|
19
|
+
|
|
20
|
+
import { DEFAULT_ENV_ALLOWLIST, createRedactor } from "../redaction.js";
|
|
21
|
+
import { sumTraceCost } from "../cost.js";
|
|
22
|
+
import { createSupervisor } from "../supervisor.js";
|
|
23
|
+
import { installApm as defaultInstallApm } from "./apm-installer.js";
|
|
24
|
+
import { installNpm as defaultInstallNpm } from "./npm-installer.js";
|
|
25
|
+
import { runJudge } from "./judge.js";
|
|
26
|
+
import { validateResultRecord } from "./result.js";
|
|
27
|
+
import { runInvariants } from "./invariants.js";
|
|
28
|
+
import { assertJudgeProfileStaged, loadTaskFamily } from "./task-family.js";
|
|
29
|
+
import { createWorkdirManager } from "./workdir.js";
|
|
30
|
+
|
|
31
|
+
const BASE_TOOLS = [
|
|
32
|
+
"Bash",
|
|
33
|
+
"Read",
|
|
34
|
+
"Glob",
|
|
35
|
+
"Grep",
|
|
36
|
+
"Write",
|
|
37
|
+
"Edit",
|
|
38
|
+
"Agent",
|
|
39
|
+
"TodoWrite",
|
|
40
|
+
];
|
|
41
|
+
|
|
42
|
+
// Upper bound on a single supervised agent run. A run that produces no terminal
|
|
43
|
+
// message within this window is treated as a stall and recorded as an
|
|
44
|
+
// agentError, so the benchmark never hangs the event loop into a silent exit.
|
|
45
|
+
const AGENT_WATCHDOG_MS = 20 * 60 * 1000;
|
|
46
|
+
|
|
47
|
+
/** Sole orchestrator for a task-family benchmark run. */
|
|
48
|
+
export class BenchmarkRunner {
|
|
49
|
+
/**
|
|
50
|
+
* @param {object} opts
|
|
51
|
+
* @param {import("./task-family.js").TaskFamily | string} opts.family
|
|
52
|
+
* @param {number} opts.runs - Runs per task (≥ 1).
|
|
53
|
+
* @param {string} opts.output - Run-output directory.
|
|
54
|
+
* @param {string} opts.agentModel
|
|
55
|
+
* @param {string} opts.supervisorModel
|
|
56
|
+
* @param {string} opts.judgeModel
|
|
57
|
+
* @param {{agent?: string, judge?: string}} [opts.profiles]
|
|
58
|
+
* @param {Function} opts.query - SDK query (injected for testability).
|
|
59
|
+
* @param {string[]} [opts.allowedTools] - Agent tool allowlist (default: BASE_TOOLS).
|
|
60
|
+
* @param {number} [opts.maxTurns] - Agent-under-test turn budget.
|
|
61
|
+
* @param {number} [opts.termGraceMs] - SIGTERM→SIGKILL grace (ms) for the per-task process group.
|
|
62
|
+
* @param {Function} [opts.runAgent] - Test seam: replaces the agent-under-test
|
|
63
|
+
* session. Must return `{costUsd, turns, submission, agentError?}` and
|
|
64
|
+
* write a valid NDJSON trace to `workdir.agentTracePath`. Default uses
|
|
65
|
+
* `createAgentRunner` with the harness `BASE_TOOLS` allowlist. Internal
|
|
66
|
+
* testing only — not part of the public API.
|
|
67
|
+
* @param {import("@forwardimpact/libutil/runtime").Runtime} opts.runtime -
|
|
68
|
+
* Injected ambient collaborators (`fs`, `subprocess`, `clock`, `proc`),
|
|
69
|
+
* threaded into the installers, workdir manager, invariants, and judge.
|
|
70
|
+
* @param {Function} [opts.runInvariants] - Test seam: replaces `runInvariants`.
|
|
71
|
+
* Same contract as `runInvariants(task, ctx, runtime)`. Internal testing only.
|
|
72
|
+
* @param {Function} [opts.runJudge] - Test seam: replaces `runJudge`. Same
|
|
73
|
+
* contract as `runJudge(task, workdir, invariants, deps)` (deps carries
|
|
74
|
+
* `runtime`). Internal testing only.
|
|
75
|
+
* @param {Function} [opts.installApm] - Test seam: replaces `installApm`.
|
|
76
|
+
* Same contract as `installApm(family, outputDir, runtime)`. Lets tests
|
|
77
|
+
* inject a fake subprocess (or skip the install entirely) so the suite
|
|
78
|
+
* never shells out to a real `apm` binary. Internal testing only.
|
|
79
|
+
* @param {Function} [opts.installNpm] - Test seam: replaces `installNpm`.
|
|
80
|
+
* Same contract as `installNpm(family, stagingDir, runtime)`. Internal
|
|
81
|
+
* testing only.
|
|
82
|
+
*/
|
|
83
|
+
constructor({
|
|
84
|
+
family,
|
|
85
|
+
runs,
|
|
86
|
+
output,
|
|
87
|
+
agentModel,
|
|
88
|
+
supervisorModel,
|
|
89
|
+
judgeModel,
|
|
90
|
+
profiles,
|
|
91
|
+
query,
|
|
92
|
+
allowedTools,
|
|
93
|
+
maxTurns,
|
|
94
|
+
task,
|
|
95
|
+
skillsFrom,
|
|
96
|
+
termGraceMs,
|
|
97
|
+
runtime,
|
|
98
|
+
// Test seams — default to the real implementations.
|
|
99
|
+
runAgent,
|
|
100
|
+
runInvariants: runInvariantsHook,
|
|
101
|
+
runJudge: runJudgeHook,
|
|
102
|
+
installApm: installApmHook,
|
|
103
|
+
installNpm: installNpmHook,
|
|
104
|
+
}) {
|
|
105
|
+
validateRunnerArgs({ family, runs, output, agentModel, query, runtime });
|
|
106
|
+
this.runtime = runtime;
|
|
107
|
+
this.familyInput = family;
|
|
108
|
+
this.runs = runs;
|
|
109
|
+
this.output = output;
|
|
110
|
+
this.agentModel = agentModel;
|
|
111
|
+
this.supervisorModel = supervisorModel;
|
|
112
|
+
this.judgeModel = judgeModel;
|
|
113
|
+
this.allowedTools = allowedTools ?? BASE_TOOLS;
|
|
114
|
+
this.profiles = {
|
|
115
|
+
agent: profiles?.agent ?? null,
|
|
116
|
+
judge: profiles?.judge ?? null,
|
|
117
|
+
};
|
|
118
|
+
this.query = query;
|
|
119
|
+
this.maxTurns = maxTurns;
|
|
120
|
+
this.taskFilter = task ?? null;
|
|
121
|
+
this.skillsFrom = skillsFrom ?? null;
|
|
122
|
+
this.termGraceMs = termGraceMs;
|
|
123
|
+
this._runAgentHook = runAgent ?? null;
|
|
124
|
+
this._runInvariantsHook = runInvariantsHook ?? runInvariants;
|
|
125
|
+
this._runJudgeHook = runJudgeHook ?? runJudge;
|
|
126
|
+
this._installApmHook = installApmHook ?? defaultInstallApm;
|
|
127
|
+
this._installNpmHook = installNpmHook ?? defaultInstallNpm;
|
|
128
|
+
}
|
|
129
|
+
|
|
130
|
+
/**
|
|
131
|
+
* Yield one ResultRecord per (task, runIndex).
|
|
132
|
+
* @returns {AsyncGenerator<object>}
|
|
133
|
+
*/
|
|
134
|
+
async *run() {
|
|
135
|
+
const runtime = this.runtime;
|
|
136
|
+
const family =
|
|
137
|
+
typeof this.familyInput === "string"
|
|
138
|
+
? await loadTaskFamily(this.familyInput, runtime)
|
|
139
|
+
: this.familyInput;
|
|
140
|
+
|
|
141
|
+
await runtime.fs.mkdir(this.output, { recursive: true });
|
|
142
|
+
const { stagingDir, skillSetHash, judgeProfilesDir } =
|
|
143
|
+
await this._installApmHook(family, this.output, runtime, {
|
|
144
|
+
skillsFrom: this.skillsFrom,
|
|
145
|
+
});
|
|
146
|
+
await this._installNpmHook(family, stagingDir, runtime);
|
|
147
|
+
|
|
148
|
+
let tasks = family.tasks();
|
|
149
|
+
if (this.taskFilter) {
|
|
150
|
+
const matched = tasks.filter((t) => t.id === this.taskFilter);
|
|
151
|
+
if (matched.length === 0) {
|
|
152
|
+
const available = tasks.map((t) => t.id).join(", ");
|
|
153
|
+
throw new Error(
|
|
154
|
+
`no task '${this.taskFilter}' in family; available: ${available}`,
|
|
155
|
+
);
|
|
156
|
+
}
|
|
157
|
+
tasks = matched;
|
|
158
|
+
}
|
|
159
|
+
if (this.profiles.judge) {
|
|
160
|
+
await assertJudgeProfileStaged(
|
|
161
|
+
family,
|
|
162
|
+
judgeProfilesDir,
|
|
163
|
+
this.profiles.judge,
|
|
164
|
+
runtime,
|
|
165
|
+
);
|
|
166
|
+
}
|
|
167
|
+
|
|
168
|
+
const wm = createWorkdirManager({
|
|
169
|
+
stagingDir,
|
|
170
|
+
runOutputDir: this.output,
|
|
171
|
+
termGraceMs: this.termGraceMs,
|
|
172
|
+
familyRootPath: family.rootPath,
|
|
173
|
+
runtime,
|
|
174
|
+
});
|
|
175
|
+
|
|
176
|
+
const resultsPath = join(this.output, "results.jsonl");
|
|
177
|
+
const resultsStream = runtime.fs.createWriteStream(resultsPath, {
|
|
178
|
+
flags: "a",
|
|
179
|
+
});
|
|
180
|
+
try {
|
|
181
|
+
for (const task of tasks) {
|
|
182
|
+
for (let runIndex = 0; runIndex < this.runs; runIndex++) {
|
|
183
|
+
const record = await this.#runOne(
|
|
184
|
+
family,
|
|
185
|
+
wm,
|
|
186
|
+
task,
|
|
187
|
+
runIndex,
|
|
188
|
+
skillSetHash,
|
|
189
|
+
judgeProfilesDir,
|
|
190
|
+
);
|
|
191
|
+
await writeRecord(resultsStream, record);
|
|
192
|
+
yield record;
|
|
193
|
+
}
|
|
194
|
+
}
|
|
195
|
+
} finally {
|
|
196
|
+
await new Promise((r) => resultsStream.end(r));
|
|
197
|
+
}
|
|
198
|
+
}
|
|
199
|
+
|
|
200
|
+
async #runOne(family, wm, task, runIndex, skillSetHash, judgeProfilesDir) {
|
|
201
|
+
const t0 = this.runtime.clock.now();
|
|
202
|
+
const workdir = await wm.start(task, runIndex);
|
|
203
|
+
try {
|
|
204
|
+
if (workdir.preflightError) {
|
|
205
|
+
const record = this.#buildPreflightFailureRecord({
|
|
206
|
+
task,
|
|
207
|
+
runIndex,
|
|
208
|
+
workdir,
|
|
209
|
+
skillSetHash,
|
|
210
|
+
familyRevision: family.familyRevision,
|
|
211
|
+
durationMs: this.runtime.clock.now() - t0,
|
|
212
|
+
});
|
|
213
|
+
return this.#validateOrFallback(
|
|
214
|
+
record,
|
|
215
|
+
resultsRecordKey(task, runIndex),
|
|
216
|
+
);
|
|
217
|
+
}
|
|
218
|
+
const agentRun = await this.#runAgentSafe(task, workdir);
|
|
219
|
+
const { costUsd, turns, submission, agentError } = agentRun;
|
|
220
|
+
const breakdown = agentRun.costBreakdown ?? { agent: 0, supervisor: 0 };
|
|
221
|
+
const invariants = await this._runInvariantsHook(
|
|
222
|
+
task,
|
|
223
|
+
{
|
|
224
|
+
cwd: workdir.cwd,
|
|
225
|
+
port: workdir.port,
|
|
226
|
+
runDir: workdir.runDir,
|
|
227
|
+
familyDir: family.rootPath,
|
|
228
|
+
},
|
|
229
|
+
this.runtime,
|
|
230
|
+
);
|
|
231
|
+
let judgeVerdict = null;
|
|
232
|
+
let judgeCost = 0;
|
|
233
|
+
if (task.paths.judge) {
|
|
234
|
+
const judgeContext = await this.#buildJudgeContext(
|
|
235
|
+
task,
|
|
236
|
+
workdir,
|
|
237
|
+
skillSetHash,
|
|
238
|
+
);
|
|
239
|
+
const judgeResult = await this._runJudgeHook(
|
|
240
|
+
task,
|
|
241
|
+
workdir,
|
|
242
|
+
invariants,
|
|
243
|
+
{
|
|
244
|
+
query: this.query,
|
|
245
|
+
model: this.judgeModel,
|
|
246
|
+
judgeProfile: this.profiles.judge ?? undefined,
|
|
247
|
+
profilesDir: judgeProfilesDir,
|
|
248
|
+
runtime: this.runtime,
|
|
249
|
+
},
|
|
250
|
+
judgeContext,
|
|
251
|
+
);
|
|
252
|
+
judgeCost = judgeResult.costUsd ?? 0;
|
|
253
|
+
// The record's judgeVerdict carries only the verdict + summary; the
|
|
254
|
+
// judge's cost is folded into costUsd / costBreakdown instead.
|
|
255
|
+
judgeVerdict = {
|
|
256
|
+
verdict: judgeResult.verdict,
|
|
257
|
+
summary: judgeResult.summary,
|
|
258
|
+
};
|
|
259
|
+
}
|
|
260
|
+
const verdict =
|
|
261
|
+
invariants.verdict === "pass" &&
|
|
262
|
+
(judgeVerdict === null || judgeVerdict.verdict === "pass")
|
|
263
|
+
? "pass"
|
|
264
|
+
: "fail";
|
|
265
|
+
const record = {
|
|
266
|
+
taskId: task.id,
|
|
267
|
+
runIndex,
|
|
268
|
+
verdict,
|
|
269
|
+
invariants,
|
|
270
|
+
submission,
|
|
271
|
+
...(judgeVerdict && { judgeVerdict }),
|
|
272
|
+
costUsd: costUsd + judgeCost,
|
|
273
|
+
costBreakdown: {
|
|
274
|
+
agent: breakdown.agent ?? 0,
|
|
275
|
+
supervisor: breakdown.supervisor ?? 0,
|
|
276
|
+
judge: judgeCost,
|
|
277
|
+
},
|
|
278
|
+
turns,
|
|
279
|
+
agentTracePath: workdir.agentTracePath,
|
|
280
|
+
supervisorTracePath: workdir.supervisorTracePath,
|
|
281
|
+
judgeTracePath: workdir.judgeTracePath,
|
|
282
|
+
profiles: {
|
|
283
|
+
agent: this.profiles.agent,
|
|
284
|
+
supervisor: null,
|
|
285
|
+
judge: this.profiles.judge,
|
|
286
|
+
},
|
|
287
|
+
model: {
|
|
288
|
+
agent: this.agentModel,
|
|
289
|
+
supervisor: this.supervisorModel,
|
|
290
|
+
judge: this.judgeModel,
|
|
291
|
+
},
|
|
292
|
+
skillSetHash,
|
|
293
|
+
familyRevision: family.familyRevision,
|
|
294
|
+
durationMs: this.runtime.clock.now() - t0,
|
|
295
|
+
...(agentError && { agentError }),
|
|
296
|
+
};
|
|
297
|
+
return this.#validateOrFallback(record, resultsRecordKey(task, runIndex));
|
|
298
|
+
} finally {
|
|
299
|
+
await wm.teardown(workdir).catch(() => {});
|
|
300
|
+
}
|
|
301
|
+
}
|
|
302
|
+
|
|
303
|
+
/**
|
|
304
|
+
* Dispatch to either the injected hook or the default `#runAgent`. Either
|
|
305
|
+
* path can throw; catch here so a thrown error becomes an `agentError` on
|
|
306
|
+
* the record (spec criterion 1: records on agent failure) rather than
|
|
307
|
+
* aborting the whole iterator.
|
|
308
|
+
*/
|
|
309
|
+
async #runAgentSafe(task, workdir) {
|
|
310
|
+
try {
|
|
311
|
+
if (this._runAgentHook) {
|
|
312
|
+
const r = await this._runAgentHook(task, workdir, this);
|
|
313
|
+
return { agentError: null, ...r };
|
|
314
|
+
}
|
|
315
|
+
return await this.#runAgent(task, workdir);
|
|
316
|
+
} catch (e) {
|
|
317
|
+
return {
|
|
318
|
+
costUsd: 0,
|
|
319
|
+
costBreakdown: { agent: 0, supervisor: 0 },
|
|
320
|
+
turns: 0,
|
|
321
|
+
submission: "",
|
|
322
|
+
agentError: { message: e.message ?? String(e), aborted: false },
|
|
323
|
+
};
|
|
324
|
+
}
|
|
325
|
+
}
|
|
326
|
+
|
|
327
|
+
/**
|
|
328
|
+
* Run the agent-under-test under a Supervisor. The supervisor writes
|
|
329
|
+
* a combined tagged NDJSON trace; after the session we split it into
|
|
330
|
+
* agent.ndjson and supervisor.ndjson and extract cost/turns/submission.
|
|
331
|
+
*/
|
|
332
|
+
async #runAgent(task, workdir) {
|
|
333
|
+
const fs = this.runtime.fs;
|
|
334
|
+
const combinedPath = join(workdir.runDir, ".combined.ndjson");
|
|
335
|
+
const combinedStream = fs.createWriteStream(combinedPath);
|
|
336
|
+
const supervisorInstructions = task.paths.supervisor
|
|
337
|
+
? await fs.readFile(task.paths.supervisor, "utf8").catch(() => null)
|
|
338
|
+
: null;
|
|
339
|
+
const supervisor = createSupervisor({
|
|
340
|
+
supervisorCwd: workdir.cwd,
|
|
341
|
+
agentCwd: workdir.cwd,
|
|
342
|
+
query: this.query,
|
|
343
|
+
output: combinedStream,
|
|
344
|
+
agentModel: this.agentModel,
|
|
345
|
+
supervisorModel: this.supervisorModel,
|
|
346
|
+
maxTurns: this.maxTurns ?? 50,
|
|
347
|
+
allowedTools: this.allowedTools,
|
|
348
|
+
...(this.profiles.agent && { agentProfile: this.profiles.agent }),
|
|
349
|
+
...(supervisorInstructions && { taskAmend: supervisorInstructions }),
|
|
350
|
+
redactor: createRedactor({
|
|
351
|
+
allowlist: [...DEFAULT_ENV_ALLOWLIST, ...(workdir.envNames ?? [])],
|
|
352
|
+
runtime: this.runtime,
|
|
353
|
+
}),
|
|
354
|
+
runtime: this.runtime,
|
|
355
|
+
});
|
|
356
|
+
const instructions = await fs.readFile(task.paths.instructions, "utf8");
|
|
357
|
+
let agentError = null;
|
|
358
|
+
// Watchdog: a supervised session can hang without settling (e.g. the agent
|
|
359
|
+
// SDK subprocess exits without a terminal message), which would empty the
|
|
360
|
+
// event loop and exit the process mid-run with zero records. Race the run
|
|
361
|
+
// against a bounded timer so a stall becomes an `agentError` record instead
|
|
362
|
+
// of a silent exit; the timer also keeps the loop alive until it fires.
|
|
363
|
+
let watchdog;
|
|
364
|
+
try {
|
|
365
|
+
const result = await Promise.race([
|
|
366
|
+
supervisor.run(instructions),
|
|
367
|
+
new Promise((_, reject) => {
|
|
368
|
+
watchdog = this.runtime.clock.setTimeout(
|
|
369
|
+
() =>
|
|
370
|
+
reject(
|
|
371
|
+
new Error(
|
|
372
|
+
`agent run produced no result within ${AGENT_WATCHDOG_MS}ms (possible stall)`,
|
|
373
|
+
),
|
|
374
|
+
),
|
|
375
|
+
AGENT_WATCHDOG_MS,
|
|
376
|
+
);
|
|
377
|
+
}),
|
|
378
|
+
]);
|
|
379
|
+
if (!result.success && !result.concluded) {
|
|
380
|
+
agentError = { message: "supervisor did not succeed", aborted: false };
|
|
381
|
+
}
|
|
382
|
+
} catch (e) {
|
|
383
|
+
agentError = { message: e.message ?? String(e), aborted: false };
|
|
384
|
+
} finally {
|
|
385
|
+
this.runtime.clock.clearTimeout(watchdog);
|
|
386
|
+
await new Promise((r) => combinedStream.end(r));
|
|
387
|
+
}
|
|
388
|
+
const summary = await splitAndSummarize(
|
|
389
|
+
this.runtime,
|
|
390
|
+
combinedPath,
|
|
391
|
+
workdir.agentTracePath,
|
|
392
|
+
workdir.supervisorTracePath,
|
|
393
|
+
);
|
|
394
|
+
// Cost is summed across every participant's result events from the one
|
|
395
|
+
// combined trace, attributed per source. Read before unlinking.
|
|
396
|
+
const combined = await fs.readFile(combinedPath, "utf8");
|
|
397
|
+
const { totalCostUsd, bySource } = sumTraceCost(combined.split("\n"));
|
|
398
|
+
await fs.unlink(combinedPath).catch(() => {});
|
|
399
|
+
return {
|
|
400
|
+
...summary,
|
|
401
|
+
costUsd: totalCostUsd,
|
|
402
|
+
costBreakdown: {
|
|
403
|
+
agent: bySource.agent ?? 0,
|
|
404
|
+
supervisor: bySource.supervisor ?? 0,
|
|
405
|
+
},
|
|
406
|
+
agentError,
|
|
407
|
+
};
|
|
408
|
+
}
|
|
409
|
+
|
|
410
|
+
async #buildJudgeContext(task, workdir, skillSetHash) {
|
|
411
|
+
const fs = this.runtime.fs;
|
|
412
|
+
const agentInstructions = await fs.readFile(
|
|
413
|
+
task.paths.instructions,
|
|
414
|
+
"utf8",
|
|
415
|
+
);
|
|
416
|
+
let agentProfile = "";
|
|
417
|
+
if (this.profiles.agent) {
|
|
418
|
+
const profilePath = resolvePath(
|
|
419
|
+
workdir.cwd,
|
|
420
|
+
".claude/agents",
|
|
421
|
+
`${this.profiles.agent}.md`,
|
|
422
|
+
);
|
|
423
|
+
agentProfile = await fs.readFile(profilePath, "utf8").catch(() => "");
|
|
424
|
+
}
|
|
425
|
+
return { agentInstructions, agentProfile, skillSetHash };
|
|
426
|
+
}
|
|
427
|
+
|
|
428
|
+
#buildPreflightFailureRecord({
|
|
429
|
+
task,
|
|
430
|
+
runIndex,
|
|
431
|
+
workdir,
|
|
432
|
+
skillSetHash,
|
|
433
|
+
familyRevision,
|
|
434
|
+
durationMs,
|
|
435
|
+
}) {
|
|
436
|
+
return {
|
|
437
|
+
taskId: task.id,
|
|
438
|
+
runIndex,
|
|
439
|
+
verdict: "fail",
|
|
440
|
+
costUsd: 0,
|
|
441
|
+
turns: 0,
|
|
442
|
+
preflightError: workdir.preflightError,
|
|
443
|
+
profiles: {
|
|
444
|
+
agent: this.profiles.agent,
|
|
445
|
+
supervisor: null,
|
|
446
|
+
judge: this.profiles.judge,
|
|
447
|
+
},
|
|
448
|
+
model: {
|
|
449
|
+
agent: this.agentModel,
|
|
450
|
+
supervisor: this.supervisorModel,
|
|
451
|
+
judge: this.judgeModel,
|
|
452
|
+
},
|
|
453
|
+
skillSetHash,
|
|
454
|
+
familyRevision,
|
|
455
|
+
durationMs,
|
|
456
|
+
agentTracePath: workdir.agentTracePath,
|
|
457
|
+
supervisorTracePath: workdir.supervisorTracePath,
|
|
458
|
+
judgeTracePath: workdir.judgeTracePath,
|
|
459
|
+
};
|
|
460
|
+
}
|
|
461
|
+
|
|
462
|
+
#validateOrFallback(record, key) {
|
|
463
|
+
try {
|
|
464
|
+
validateResultRecord(record);
|
|
465
|
+
return record;
|
|
466
|
+
} catch (e) {
|
|
467
|
+
// The runner constructed the record — a schema failure is a real bug,
|
|
468
|
+
// not bad family input. Emit a noisy fallback so the iterator stays
|
|
469
|
+
// consumable and the agent budget isn't silently dropped.
|
|
470
|
+
return {
|
|
471
|
+
taskId: record.taskId ?? key.taskId,
|
|
472
|
+
runIndex: record.runIndex ?? key.runIndex,
|
|
473
|
+
verdict: "fail",
|
|
474
|
+
schemaError: e.message ?? String(e),
|
|
475
|
+
};
|
|
476
|
+
}
|
|
477
|
+
}
|
|
478
|
+
}
|
|
479
|
+
|
|
480
|
+
/**
|
|
481
|
+
* Validate the required BenchmarkRunner constructor arguments. Extracted from
|
|
482
|
+
* the constructor to keep its cognitive complexity under the lint ceiling.
|
|
483
|
+
*/
|
|
484
|
+
function validateRunnerArgs({
|
|
485
|
+
family,
|
|
486
|
+
runs,
|
|
487
|
+
output,
|
|
488
|
+
agentModel,
|
|
489
|
+
query,
|
|
490
|
+
runtime,
|
|
491
|
+
}) {
|
|
492
|
+
if (!family) throw new Error("family is required");
|
|
493
|
+
if (!Number.isInteger(runs) || runs < 1)
|
|
494
|
+
throw new Error("runs must be an integer ≥ 1");
|
|
495
|
+
if (!output) throw new Error("output is required");
|
|
496
|
+
if (!agentModel) throw new Error("agentModel is required");
|
|
497
|
+
if (!query) throw new Error("query is required");
|
|
498
|
+
if (!runtime) throw new Error("runtime is required");
|
|
499
|
+
}
|
|
500
|
+
|
|
501
|
+
function resultsRecordKey(task, runIndex) {
|
|
502
|
+
return { taskId: task.id, runIndex };
|
|
503
|
+
}
|
|
504
|
+
|
|
505
|
+
async function writeRecord(stream, record) {
|
|
506
|
+
const line = JSON.stringify(record) + "\n";
|
|
507
|
+
await new Promise((res, rej) => {
|
|
508
|
+
stream.write(line, (err) => (err ? rej(err) : res()));
|
|
509
|
+
});
|
|
510
|
+
}
|
|
511
|
+
|
|
512
|
+
/**
|
|
513
|
+
* Split the combined supervisor trace into agent and supervisor files and
|
|
514
|
+
* extract turn count and submission in a single pass. Agent-source events go
|
|
515
|
+
* to `agentPath`; supervisor and orchestrator events go to `supervisorPath`.
|
|
516
|
+
*
|
|
517
|
+
* Cost is deliberately not summed here — the caller derives it from the same
|
|
518
|
+
* combined trace via `sumTraceCost`, so there is one cost path across the
|
|
519
|
+
* benchmark, callback, and `fit-trace cost` consumers.
|
|
520
|
+
*/
|
|
521
|
+
// biome-ignore lint/complexity/noExcessiveCognitiveComplexity: stream-splitting state machine
|
|
522
|
+
async function splitAndSummarize(
|
|
523
|
+
runtime,
|
|
524
|
+
combinedPath,
|
|
525
|
+
agentPath,
|
|
526
|
+
supervisorPath,
|
|
527
|
+
) {
|
|
528
|
+
const fs = runtime.fs;
|
|
529
|
+
const agentStream = fs.createWriteStream(agentPath);
|
|
530
|
+
const supStream = fs.createWriteStream(supervisorPath);
|
|
531
|
+
const rl = createInterface({
|
|
532
|
+
input: fs.createReadStream(combinedPath),
|
|
533
|
+
crlfDelay: Infinity,
|
|
534
|
+
});
|
|
535
|
+
let turns = 0;
|
|
536
|
+
let submission = "";
|
|
537
|
+
for await (const line of rl) {
|
|
538
|
+
if (!line.trim()) continue;
|
|
539
|
+
let event;
|
|
540
|
+
try {
|
|
541
|
+
event = JSON.parse(line);
|
|
542
|
+
} catch {
|
|
543
|
+
continue;
|
|
544
|
+
}
|
|
545
|
+
const target = event.source === "agent" ? agentStream : supStream;
|
|
546
|
+
target.write(line + "\n");
|
|
547
|
+
const inner = event.event;
|
|
548
|
+
if (!inner) continue;
|
|
549
|
+
if (event.source === "agent" && inner.type === "assistant") {
|
|
550
|
+
const text = extractText(inner);
|
|
551
|
+
if (text) submission = text;
|
|
552
|
+
}
|
|
553
|
+
if (event.source === "orchestrator" && inner.type === "summary") {
|
|
554
|
+
turns = inner.turns ?? 0;
|
|
555
|
+
}
|
|
556
|
+
}
|
|
557
|
+
await Promise.all([
|
|
558
|
+
new Promise((r) => agentStream.end(r)),
|
|
559
|
+
new Promise((r) => supStream.end(r)),
|
|
560
|
+
]);
|
|
561
|
+
return { turns, submission };
|
|
562
|
+
}
|
|
563
|
+
|
|
564
|
+
function extractText(inner) {
|
|
565
|
+
const content = inner.message?.content ?? inner.content;
|
|
566
|
+
if (!Array.isArray(content)) return null;
|
|
567
|
+
for (let i = content.length - 1; i >= 0; i--) {
|
|
568
|
+
if (content[i].type === "text" && content[i].text) return content[i].text;
|
|
569
|
+
}
|
|
570
|
+
return null;
|
|
571
|
+
}
|
|
572
|
+
|
|
573
|
+
/**
|
|
574
|
+
* Factory function — wires real dependencies.
|
|
575
|
+
* @param {ConstructorParameters<typeof BenchmarkRunner>[0]} opts
|
|
576
|
+
* @returns {BenchmarkRunner}
|
|
577
|
+
*/
|
|
578
|
+
export function createBenchmarkRunner(opts) {
|
|
579
|
+
return new BenchmarkRunner(opts);
|
|
580
|
+
}
|
|
581
|
+
|
|
582
|
+
// Internal exports used by tests.
|
|
583
|
+
export const __BASE_TOOLS = BASE_TOOLS;
|