@forwardimpact/libharness 0.1.22 → 1.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (87) hide show
  1. package/LICENSE +21 -201
  2. package/README.md +196 -80
  3. package/bin/fit-benchmark.js +44 -0
  4. package/bin/fit-harness.js +358 -0
  5. package/bin/fit-selfedit.js +165 -0
  6. package/bin/fit-trace.js +510 -0
  7. package/package.json +41 -11
  8. package/src/agent-runner.js +256 -0
  9. package/src/benchmark/apm-installer.js +207 -0
  10. package/src/benchmark/env-loader.js +158 -0
  11. package/src/benchmark/hook-env.js +40 -0
  12. package/src/benchmark/invariants.js +141 -0
  13. package/src/benchmark/judge.js +187 -0
  14. package/src/benchmark/npm-installer.js +87 -0
  15. package/src/benchmark/report.js +604 -0
  16. package/src/benchmark/result.js +127 -0
  17. package/src/benchmark/runner.js +688 -0
  18. package/src/benchmark/scheduler.js +78 -0
  19. package/src/benchmark/task-family.js +260 -0
  20. package/src/benchmark/workdir.js +344 -0
  21. package/src/commands/assert.js +153 -0
  22. package/src/commands/benchmark-definition.js +175 -0
  23. package/src/commands/benchmark-invariants.js +73 -0
  24. package/src/commands/benchmark-report.js +51 -0
  25. package/src/commands/benchmark-run.js +175 -0
  26. package/src/commands/by-discussion.js +94 -0
  27. package/src/commands/callback.js +119 -0
  28. package/src/commands/discuss.js +132 -0
  29. package/src/commands/facilitate.js +123 -0
  30. package/src/commands/output.js +36 -0
  31. package/src/commands/run.js +152 -0
  32. package/src/commands/supervise.js +136 -0
  33. package/src/commands/task-input.js +54 -0
  34. package/src/commands/tee.js +53 -0
  35. package/src/commands/trace.js +630 -0
  36. package/src/commands/work-tracker.js +35 -0
  37. package/src/cost.js +79 -0
  38. package/src/discuss-tools.js +173 -0
  39. package/src/discusser.js +394 -0
  40. package/src/events/github.js +161 -0
  41. package/src/facilitator.js +205 -0
  42. package/src/inbox-poller.js +81 -0
  43. package/src/index.js +72 -2
  44. package/src/judge.js +210 -0
  45. package/src/message-bus.js +118 -0
  46. package/src/orchestration-loop.js +330 -0
  47. package/src/orchestration-toolkit.js +441 -0
  48. package/src/orchestrator-helpers.js +23 -0
  49. package/src/profile-prompt.js +266 -0
  50. package/src/redaction.js +253 -0
  51. package/src/render/line-renderer.js +54 -0
  52. package/src/render/orchestrator-filter.js +19 -0
  53. package/src/render/palette.js +63 -0
  54. package/src/render/tool-hints.js +154 -0
  55. package/src/render/turn-renderer.js +96 -0
  56. package/src/reply-emitter.js +47 -0
  57. package/src/sequence-counter.js +21 -0
  58. package/src/signature-filter.js +27 -0
  59. package/src/supervisor.js +236 -0
  60. package/src/tee-writer.js +150 -0
  61. package/src/trace-collector.js +444 -0
  62. package/src/trace-github.js +473 -0
  63. package/src/trace-multi.js +101 -0
  64. package/src/trace-query.js +748 -0
  65. package/src/trace-render.js +211 -0
  66. package/src/trace-usage.js +249 -0
  67. package/src/fixture/assertions.js +0 -42
  68. package/src/fixture/cache.js +0 -50
  69. package/src/fixture/eval.js +0 -146
  70. package/src/fixture/index.js +0 -9
  71. package/src/fixture/pathway.js +0 -451
  72. package/src/fixture/services.js +0 -56
  73. package/src/mock/clients.js +0 -135
  74. package/src/mock/config.js +0 -45
  75. package/src/mock/data.js +0 -46
  76. package/src/mock/fs.js +0 -111
  77. package/src/mock/grpc.js +0 -94
  78. package/src/mock/http.js +0 -60
  79. package/src/mock/index.js +0 -36
  80. package/src/mock/infra.js +0 -219
  81. package/src/mock/logger.js +0 -42
  82. package/src/mock/observer.js +0 -74
  83. package/src/mock/resource-index.js +0 -95
  84. package/src/mock/service-callbacks.js +0 -39
  85. package/src/mock/services.js +0 -79
  86. package/src/mock/spy.js +0 -44
  87. package/src/mock/storage.js +0 -118
@@ -0,0 +1,688 @@
1
+ /**
2
+ * BenchmarkRunner — sole orchestrator for a task-family benchmark run.
3
+ *
4
+ * Phases per (task, runIndex):
5
+ * 1. WorkdirManager.start → seed CWD + run pre-flight probe
6
+ * 2. Supervisor session (agent + supervisor) → produce traces + submission
7
+ * 3. Invariants.runInvariants → exit-code-driven verdict via fd-3 NDJSON
8
+ * 4. Judge.runJudge → Conclude-driven verdict mapped to pass/fail
9
+ * 5. WorkdirManager.teardown → process-group cleanup
10
+ *
11
+ * Cells run with bounded in-process concurrency (`CellScheduler`); `run()`
12
+ * yields records in **completion order**, not grid order. A single drain loop
13
+ * is the sole writer of `<output>/results.jsonl`, appending each record the
14
+ * moment its cell settles — that incremental append is the durability and
15
+ * crash-safety mechanism, so a killed run keeps every completed cell and there
16
+ * is no sidecar ledger. The iterator drives CLI stdout mirroring off the same
17
+ * stream.
18
+ */
19
+
20
+ import { createInterface } from "node:readline";
21
+ import { join, resolve as resolvePath } from "node:path";
22
+
23
+ import { DEFAULT_ENV_ALLOWLIST, createRedactor } from "../redaction.js";
24
+ import { sumTraceCost } from "../cost.js";
25
+ import { createSupervisor } from "../supervisor.js";
26
+ import { installApm as defaultInstallApm } from "./apm-installer.js";
27
+ import { installNpm as defaultInstallNpm } from "./npm-installer.js";
28
+ import { runJudge } from "./judge.js";
29
+ import { validateResultRecord } from "./result.js";
30
+ import { runInvariants } from "./invariants.js";
31
+ import { assertJudgeProfileStaged, loadTaskFamily } from "./task-family.js";
32
+ import { createWorkdirManager } from "./workdir.js";
33
+ import { CellScheduler } from "./scheduler.js";
34
+
35
+ const BASE_TOOLS = [
36
+ "Bash",
37
+ "Read",
38
+ "Glob",
39
+ "Grep",
40
+ "Write",
41
+ "Edit",
42
+ "Agent",
43
+ "TodoWrite",
44
+ ];
45
+
46
+ // Upper bound on a single supervised agent run. A run that produces no terminal
47
+ // message within this window is treated as a stall and recorded as an
48
+ // agentError, so the benchmark never hangs the event loop into a silent exit.
49
+ // Overridable per-runner via `watchdogMs` so a test can force a stall to fire
50
+ // without waiting the full 20 minutes.
51
+ const AGENT_WATCHDOG_MS = 20 * 60 * 1000;
52
+
53
+ /** Sole orchestrator for a task-family benchmark run. */
54
+ export class BenchmarkRunner {
55
+ /**
56
+ * @param {object} opts
57
+ * @param {import("./task-family.js").TaskFamily | string} opts.family
58
+ * @param {number} opts.runs - Runs per task (≥ 1).
59
+ * @param {string} opts.output - Run-output directory.
60
+ * @param {string} opts.agentModel
61
+ * @param {string} opts.supervisorModel
62
+ * @param {string} opts.judgeModel
63
+ * @param {{agent?: string, judge?: string}} [opts.profiles]
64
+ * @param {Function} opts.query - SDK query (injected for testability).
65
+ * @param {string[]} [opts.allowedTools] - Agent tool allowlist (default: BASE_TOOLS).
66
+ * @param {number} [opts.maxTurns] - Agent-under-test turn budget.
67
+ * @param {number} [opts.concurrency] - Max cells in flight (integer ≥ 1).
68
+ * Defaults to 1 as a defensive floor; the CLI always passes a resolved value.
69
+ * @param {{index: number, total: number}} [opts.shard] - Run only the cells
70
+ * assigned to shard `index` of `total` (1-based). Absent ≡ the whole grid
71
+ * (identity `1/1`).
72
+ * @param {number} [opts.watchdogMs] - Per-agent stall watchdog (ms). Defaults
73
+ * to `AGENT_WATCHDOG_MS`; injectable so tests can force a stall in-test.
74
+ * @param {number} [opts.termGraceMs] - SIGTERM→SIGKILL grace (ms) for the per-task process group.
75
+ * @param {Function} [opts.runAgent] - Test seam: replaces the agent-under-test
76
+ * session. Must return `{costUsd, turns, submission, agentError?}` and
77
+ * write a valid NDJSON trace to `workdir.agentTracePath`. Default uses
78
+ * `createAgentRunner` with the harness `BASE_TOOLS` allowlist. Internal
79
+ * testing only — not part of the public API.
80
+ * @param {import("@forwardimpact/libutil/runtime").Runtime} opts.runtime -
81
+ * Injected ambient collaborators (`fs`, `subprocess`, `clock`, `proc`),
82
+ * threaded into the installers, workdir manager, invariants, and judge.
83
+ * @param {Function} [opts.runInvariants] - Test seam: replaces `runInvariants`.
84
+ * Same contract as `runInvariants(task, ctx, runtime)`. Internal testing only.
85
+ * @param {Function} [opts.runJudge] - Test seam: replaces `runJudge`. Same
86
+ * contract as `runJudge(task, workdir, invariants, deps)` (deps carries
87
+ * `runtime`). Internal testing only.
88
+ * @param {Function} [opts.installApm] - Test seam: replaces `installApm`.
89
+ * Same contract as `installApm(family, outputDir, runtime)`. Lets tests
90
+ * inject a fake subprocess (or skip the install entirely) so the suite
91
+ * never shells out to a real `apm` binary. Internal testing only.
92
+ * @param {Function} [opts.installNpm] - Test seam: replaces `installNpm`.
93
+ * Same contract as `installNpm(family, stagingDir, runtime)`. Internal
94
+ * testing only.
95
+ */
96
+ constructor({
97
+ family,
98
+ runs,
99
+ output,
100
+ agentModel,
101
+ supervisorModel,
102
+ judgeModel,
103
+ profiles,
104
+ query,
105
+ allowedTools,
106
+ maxTurns,
107
+ concurrency,
108
+ watchdogMs,
109
+ shard,
110
+ task,
111
+ skillsFrom,
112
+ termGraceMs,
113
+ runtime,
114
+ // Test seams — default to the real implementations.
115
+ runAgent,
116
+ runInvariants: runInvariantsHook,
117
+ runJudge: runJudgeHook,
118
+ installApm: installApmHook,
119
+ installNpm: installNpmHook,
120
+ }) {
121
+ validateRunnerArgs({ family, runs, output, agentModel, query, runtime });
122
+ this.runtime = runtime;
123
+ this.familyInput = family;
124
+ this.runs = runs;
125
+ this.output = output;
126
+ this.agentModel = agentModel;
127
+ this.supervisorModel = supervisorModel;
128
+ this.judgeModel = judgeModel;
129
+ this.allowedTools = allowedTools ?? BASE_TOOLS;
130
+ this.profiles = {
131
+ agent: profiles?.agent ?? null,
132
+ judge: profiles?.judge ?? null,
133
+ };
134
+ this.query = query;
135
+ this.maxTurns = maxTurns;
136
+ this.concurrency = concurrency ?? 1;
137
+ this.watchdogMs = watchdogMs ?? AGENT_WATCHDOG_MS;
138
+ this.shard = shard ?? null;
139
+ this.taskFilter = task ?? null;
140
+ this.skillsFrom = skillsFrom ?? null;
141
+ this.termGraceMs = termGraceMs;
142
+ this._runAgentHook = runAgent ?? null;
143
+ this._runInvariantsHook = runInvariantsHook ?? runInvariants;
144
+ this._runJudgeHook = runJudgeHook ?? runJudge;
145
+ this._installApmHook = installApmHook ?? defaultInstallApm;
146
+ this._installNpmHook = installNpmHook ?? defaultInstallNpm;
147
+ }
148
+
149
+ /**
150
+ * Yield one ResultRecord per (task, runIndex).
151
+ * @returns {AsyncGenerator<object>}
152
+ */
153
+ async *run() {
154
+ const runtime = this.runtime;
155
+ const family =
156
+ typeof this.familyInput === "string"
157
+ ? await loadTaskFamily(this.familyInput, runtime)
158
+ : this.familyInput;
159
+
160
+ await runtime.fs.mkdir(this.output, { recursive: true });
161
+ const { stagingDir, skillSetHash, judgeProfilesDir } =
162
+ await this._installApmHook(family, this.output, runtime, {
163
+ skillsFrom: this.skillsFrom,
164
+ });
165
+ await this._installNpmHook(family, stagingDir, runtime);
166
+
167
+ let tasks = family.tasks();
168
+ if (this.taskFilter) {
169
+ const matched = tasks.filter((t) => t.id === this.taskFilter);
170
+ if (matched.length === 0) {
171
+ const available = tasks.map((t) => t.id).join(", ");
172
+ throw new Error(
173
+ `no task '${this.taskFilter}' in family; available: ${available}`,
174
+ );
175
+ }
176
+ tasks = matched;
177
+ }
178
+ if (this.profiles.judge) {
179
+ await assertJudgeProfileStaged(
180
+ family,
181
+ judgeProfilesDir,
182
+ this.profiles.judge,
183
+ runtime,
184
+ );
185
+ }
186
+
187
+ const wm = createWorkdirManager({
188
+ stagingDir,
189
+ runOutputDir: this.output,
190
+ termGraceMs: this.termGraceMs,
191
+ familyRootPath: family.rootPath,
192
+ runtime,
193
+ });
194
+
195
+ const allCells = enumerateCells(tasks, this.runs);
196
+ // Sharding selects a deterministic subset of the grid; an unsharded run is
197
+ // the identity 1/1. A high-index shard may select zero cells — a valid run
198
+ // whose results.jsonl ends up empty.
199
+ const cells = this.shard
200
+ ? selectShard(allCells, this.shard.index, this.shard.total)
201
+ : allCells;
202
+ const scheduler = new CellScheduler({
203
+ concurrency: this.concurrency,
204
+ runCell: (cell) =>
205
+ this.#runOne(
206
+ family,
207
+ wm,
208
+ cell.task,
209
+ cell.runIndex,
210
+ skillSetHash,
211
+ judgeProfilesDir,
212
+ ),
213
+ });
214
+
215
+ const resultsPath = join(this.output, "results.jsonl");
216
+ const resultsStream = runtime.fs.createWriteStream(resultsPath, {
217
+ flags: "a",
218
+ });
219
+ // Single-writer drain: the scheduler runs up to `concurrency` cells at
220
+ // once and pushes each settled record here in completion order. This loop
221
+ // is the sole writer of `results.jsonl` — workers never touch the stream —
222
+ // and the per-completion append is the crash-safety mechanism.
223
+ try {
224
+ for await (const record of scheduler.run(cells)) {
225
+ await writeRecord(resultsStream, record);
226
+ yield record;
227
+ }
228
+ } finally {
229
+ await new Promise((r) => resultsStream.end(r));
230
+ }
231
+ }
232
+
233
+ async #runOne(family, wm, task, runIndex, skillSetHash, judgeProfilesDir) {
234
+ const t0 = this.runtime.clock.now();
235
+ let workdir;
236
+ try {
237
+ workdir = await wm.start(task, runIndex);
238
+ return await this.#executeCell({
239
+ family,
240
+ workdir,
241
+ task,
242
+ runIndex,
243
+ skillSetHash,
244
+ judgeProfilesDir,
245
+ t0,
246
+ });
247
+ } catch (e) {
248
+ // `wm.start()` (port acquire + workdir/env seeding) is the one throw site
249
+ // not caught inside `#executeCell`. Turn it into the runner's own fallback
250
+ // record so `#runOne` never rejects — the scheduler's one-record-per-cell
251
+ // contract depends on that. The fallback is schema-skipped by `report`,
252
+ // the same as any other runner-side schema failure.
253
+ return {
254
+ taskId: task.id,
255
+ runIndex,
256
+ verdict: "fail",
257
+ schemaError: `cell setup failed: ${e.message ?? String(e)}`,
258
+ };
259
+ } finally {
260
+ if (workdir) await wm.teardown(workdir).catch(() => {});
261
+ }
262
+ }
263
+
264
+ /**
265
+ * Run one cell's lifecycle against an already-started workdir: preflight
266
+ * gate → supervised agent → invariants → judge → assembled record. Extracted
267
+ * from `#runOne` so the start/teardown/error wrapper stays under the
268
+ * complexity ceiling.
269
+ */
270
+ async #executeCell({
271
+ family,
272
+ workdir,
273
+ task,
274
+ runIndex,
275
+ skillSetHash,
276
+ judgeProfilesDir,
277
+ t0,
278
+ }) {
279
+ if (workdir.preflightError) {
280
+ const record = this.#buildPreflightFailureRecord({
281
+ task,
282
+ runIndex,
283
+ workdir,
284
+ skillSetHash,
285
+ familyRevision: family.familyRevision,
286
+ durationMs: this.runtime.clock.now() - t0,
287
+ });
288
+ return this.#validateOrFallback(record, resultsRecordKey(task, runIndex));
289
+ }
290
+ {
291
+ const agentRun = await this.#runAgentSafe(task, workdir);
292
+ const { costUsd, turns, submission, agentError } = agentRun;
293
+ const breakdown = agentRun.costBreakdown ?? { agent: 0, supervisor: 0 };
294
+ const invariants = await this._runInvariantsHook(
295
+ task,
296
+ {
297
+ cwd: workdir.cwd,
298
+ port: workdir.port,
299
+ runDir: workdir.runDir,
300
+ familyDir: family.rootPath,
301
+ },
302
+ this.runtime,
303
+ );
304
+ let judgeVerdict = null;
305
+ let judgeCost = 0;
306
+ if (task.paths.judge) {
307
+ const judgeContext = await this.#buildJudgeContext(
308
+ task,
309
+ workdir,
310
+ skillSetHash,
311
+ );
312
+ const judgeResult = await this._runJudgeHook(
313
+ task,
314
+ workdir,
315
+ invariants,
316
+ {
317
+ query: this.query,
318
+ model: this.judgeModel,
319
+ judgeProfile: this.profiles.judge ?? undefined,
320
+ profilesDir: judgeProfilesDir,
321
+ runtime: this.runtime,
322
+ },
323
+ judgeContext,
324
+ );
325
+ judgeCost = judgeResult.costUsd ?? 0;
326
+ // The record's judgeVerdict carries only the verdict + summary; the
327
+ // judge's cost is folded into costUsd / costBreakdown instead.
328
+ judgeVerdict = {
329
+ verdict: judgeResult.verdict,
330
+ summary: judgeResult.summary,
331
+ };
332
+ }
333
+ const verdict =
334
+ invariants.verdict === "pass" &&
335
+ (judgeVerdict === null || judgeVerdict.verdict === "pass")
336
+ ? "pass"
337
+ : "fail";
338
+ const record = {
339
+ taskId: task.id,
340
+ runIndex,
341
+ verdict,
342
+ invariants,
343
+ submission,
344
+ ...(judgeVerdict && { judgeVerdict }),
345
+ costUsd: costUsd + judgeCost,
346
+ costBreakdown: {
347
+ agent: breakdown.agent ?? 0,
348
+ supervisor: breakdown.supervisor ?? 0,
349
+ judge: judgeCost,
350
+ },
351
+ turns,
352
+ agentTracePath: workdir.agentTracePath,
353
+ supervisorTracePath: workdir.supervisorTracePath,
354
+ judgeTracePath: workdir.judgeTracePath,
355
+ profiles: {
356
+ agent: this.profiles.agent,
357
+ supervisor: null,
358
+ judge: this.profiles.judge,
359
+ },
360
+ model: {
361
+ agent: this.agentModel,
362
+ supervisor: this.supervisorModel,
363
+ judge: this.judgeModel,
364
+ },
365
+ skillSetHash,
366
+ familyRevision: family.familyRevision,
367
+ durationMs: this.runtime.clock.now() - t0,
368
+ ...(agentError && { agentError }),
369
+ };
370
+ return this.#validateOrFallback(record, resultsRecordKey(task, runIndex));
371
+ }
372
+ }
373
+
374
+ /**
375
+ * Dispatch to either the injected hook or the default `#runAgent`. Either
376
+ * path can throw; catch here so a thrown error becomes an `agentError` on
377
+ * the record (spec criterion 1: records on agent failure) rather than
378
+ * aborting the whole iterator.
379
+ */
380
+ async #runAgentSafe(task, workdir) {
381
+ try {
382
+ if (this._runAgentHook) {
383
+ const r = await this._runAgentHook(task, workdir, this);
384
+ return { agentError: null, ...r };
385
+ }
386
+ return await this.#runAgent(task, workdir);
387
+ } catch (e) {
388
+ return {
389
+ costUsd: 0,
390
+ costBreakdown: { agent: 0, supervisor: 0 },
391
+ turns: 0,
392
+ submission: "",
393
+ agentError: { message: e.message ?? String(e), aborted: false },
394
+ };
395
+ }
396
+ }
397
+
398
+ /**
399
+ * Run the agent-under-test under a Supervisor. The supervisor writes
400
+ * a combined tagged NDJSON trace; after the session we split it into
401
+ * agent.ndjson and supervisor.ndjson and extract cost/turns/submission.
402
+ */
403
+ async #runAgent(task, workdir) {
404
+ const fs = this.runtime.fs;
405
+ const combinedPath = join(workdir.runDir, ".combined.ndjson");
406
+ const combinedStream = fs.createWriteStream(combinedPath);
407
+ const supervisorInstructions = task.paths.supervisor
408
+ ? await fs.readFile(task.paths.supervisor, "utf8").catch(() => null)
409
+ : null;
410
+ const supervisor = createSupervisor({
411
+ supervisorCwd: workdir.cwd,
412
+ agentCwd: workdir.cwd,
413
+ query: this.query,
414
+ output: combinedStream,
415
+ agentModel: this.agentModel,
416
+ supervisorModel: this.supervisorModel,
417
+ maxTurns: this.maxTurns ?? 50,
418
+ allowedTools: this.allowedTools,
419
+ ...(this.profiles.agent && { agentProfile: this.profiles.agent }),
420
+ ...(supervisorInstructions && { taskAmend: supervisorInstructions }),
421
+ redactor: createRedactor({
422
+ allowlist: [...DEFAULT_ENV_ALLOWLIST, ...(workdir.envNames ?? [])],
423
+ runtime: this.runtime,
424
+ }),
425
+ runtime: this.runtime,
426
+ });
427
+ const instructions = await fs.readFile(task.paths.instructions, "utf8");
428
+ let agentError = null;
429
+ // Watchdog: a supervised session can hang without settling (e.g. the agent
430
+ // SDK subprocess exits without a terminal message), which would empty the
431
+ // event loop and exit the process mid-run with zero records. Race the run
432
+ // against a bounded timer so a stall becomes an `agentError` record instead
433
+ // of a silent exit; the timer also keeps the loop alive until it fires.
434
+ let watchdog;
435
+ try {
436
+ const result = await Promise.race([
437
+ supervisor.run(instructions),
438
+ new Promise((_, reject) => {
439
+ watchdog = this.runtime.clock.setTimeout(
440
+ () =>
441
+ reject(
442
+ new Error(
443
+ `agent run produced no result within ${this.watchdogMs}ms (possible stall)`,
444
+ ),
445
+ ),
446
+ this.watchdogMs,
447
+ );
448
+ }),
449
+ ]);
450
+ if (!result.success && !result.concluded) {
451
+ agentError = { message: "supervisor did not succeed", aborted: false };
452
+ }
453
+ } catch (e) {
454
+ agentError = { message: e.message ?? String(e), aborted: false };
455
+ } finally {
456
+ this.runtime.clock.clearTimeout(watchdog);
457
+ await new Promise((r) => combinedStream.end(r));
458
+ }
459
+ const summary = await splitAndSummarize(
460
+ this.runtime,
461
+ combinedPath,
462
+ workdir.agentTracePath,
463
+ workdir.supervisorTracePath,
464
+ );
465
+ // Cost is summed across every participant's result events from the one
466
+ // combined trace, attributed per source. Read before unlinking.
467
+ const combined = await fs.readFile(combinedPath, "utf8");
468
+ const { totalCostUsd, bySource } = sumTraceCost(combined.split("\n"));
469
+ await fs.unlink(combinedPath).catch(() => {});
470
+ return {
471
+ ...summary,
472
+ costUsd: totalCostUsd,
473
+ costBreakdown: {
474
+ agent: bySource.agent ?? 0,
475
+ supervisor: bySource.supervisor ?? 0,
476
+ },
477
+ agentError,
478
+ };
479
+ }
480
+
481
+ async #buildJudgeContext(task, workdir, skillSetHash) {
482
+ const fs = this.runtime.fs;
483
+ const agentInstructions = await fs.readFile(
484
+ task.paths.instructions,
485
+ "utf8",
486
+ );
487
+ let agentProfile = "";
488
+ if (this.profiles.agent) {
489
+ const profilePath = resolvePath(
490
+ workdir.cwd,
491
+ ".claude/agents",
492
+ `${this.profiles.agent}.md`,
493
+ );
494
+ agentProfile = await fs.readFile(profilePath, "utf8").catch(() => "");
495
+ }
496
+ return { agentInstructions, agentProfile, skillSetHash };
497
+ }
498
+
499
+ #buildPreflightFailureRecord({
500
+ task,
501
+ runIndex,
502
+ workdir,
503
+ skillSetHash,
504
+ familyRevision,
505
+ durationMs,
506
+ }) {
507
+ return {
508
+ taskId: task.id,
509
+ runIndex,
510
+ verdict: "fail",
511
+ costUsd: 0,
512
+ turns: 0,
513
+ preflightError: workdir.preflightError,
514
+ profiles: {
515
+ agent: this.profiles.agent,
516
+ supervisor: null,
517
+ judge: this.profiles.judge,
518
+ },
519
+ model: {
520
+ agent: this.agentModel,
521
+ supervisor: this.supervisorModel,
522
+ judge: this.judgeModel,
523
+ },
524
+ skillSetHash,
525
+ familyRevision,
526
+ durationMs,
527
+ agentTracePath: workdir.agentTracePath,
528
+ supervisorTracePath: workdir.supervisorTracePath,
529
+ judgeTracePath: workdir.judgeTracePath,
530
+ };
531
+ }
532
+
533
+ #validateOrFallback(record, key) {
534
+ try {
535
+ validateResultRecord(record);
536
+ return record;
537
+ } catch (e) {
538
+ // The runner constructed the record — a schema failure is a real bug,
539
+ // not bad family input. Emit a noisy fallback so the iterator stays
540
+ // consumable and the agent budget isn't silently dropped.
541
+ return {
542
+ taskId: record.taskId ?? key.taskId,
543
+ runIndex: record.runIndex ?? key.runIndex,
544
+ verdict: "fail",
545
+ schemaError: e.message ?? String(e),
546
+ };
547
+ }
548
+ }
549
+ }
550
+
551
+ /**
552
+ * Flatten the grid into a stable ordered cell list, task-major /
553
+ * runIndex-minor. Load-bearing ordering: Part 02's round-robin shard balance
554
+ * depends on a task's runIndexes being adjacent in this list. Single source of
555
+ * the cell list for both the scheduler and the shard selector.
556
+ * @param {import("./task-family.js").Task[]} tasks
557
+ * @param {number} runs
558
+ * @returns {{task: import("./task-family.js").Task, runIndex: number}[]}
559
+ */
560
+ export function enumerateCells(tasks, runs) {
561
+ const cells = [];
562
+ for (const task of tasks)
563
+ for (let runIndex = 0; runIndex < runs; runIndex++)
564
+ cells.push({ task, runIndex });
565
+ return cells;
566
+ }
567
+
568
+ /**
569
+ * Round-robin partition of the enumerated cells: the cell at position `p` runs
570
+ * iff `p % total === i - 1`. `i` is 1-based (Playwright-style). The union over
571
+ * `i ∈ 1..total` is the exact grid, each cell once; when `total > cells.length`
572
+ * the high-index shards select **zero** cells — a valid run. Because
573
+ * `enumerateCells` is task-major, a task's run indexes are adjacent, so
574
+ * round-robin spreads them across shards rather than handing one shard a slow
575
+ * task's whole run block.
576
+ * @param {{task: object, runIndex: number}[]} cells
577
+ * @param {number} i - 1-based shard index.
578
+ * @param {number} total - Shard count.
579
+ * @returns {{task: object, runIndex: number}[]}
580
+ */
581
+ export function selectShard(cells, i, total) {
582
+ return cells.filter((_, p) => p % total === i - 1);
583
+ }
584
+
585
+ /**
586
+ * Validate the required BenchmarkRunner constructor arguments. Extracted from
587
+ * the constructor to keep its cognitive complexity under the lint ceiling.
588
+ */
589
+ function validateRunnerArgs({
590
+ family,
591
+ runs,
592
+ output,
593
+ agentModel,
594
+ query,
595
+ runtime,
596
+ }) {
597
+ if (!family) throw new Error("family is required");
598
+ if (!Number.isInteger(runs) || runs < 1)
599
+ throw new Error("runs must be an integer ≥ 1");
600
+ if (!output) throw new Error("output is required");
601
+ if (!agentModel) throw new Error("agentModel is required");
602
+ if (!query) throw new Error("query is required");
603
+ if (!runtime) throw new Error("runtime is required");
604
+ }
605
+
606
+ function resultsRecordKey(task, runIndex) {
607
+ return { taskId: task.id, runIndex };
608
+ }
609
+
610
+ async function writeRecord(stream, record) {
611
+ const line = JSON.stringify(record) + "\n";
612
+ await new Promise((res, rej) => {
613
+ stream.write(line, (err) => (err ? rej(err) : res()));
614
+ });
615
+ }
616
+
617
+ /**
618
+ * Split the combined supervisor trace into agent and supervisor files and
619
+ * extract turn count and submission in a single pass. Agent-source events go
620
+ * to `agentPath`; supervisor and orchestrator events go to `supervisorPath`.
621
+ *
622
+ * Cost is deliberately not summed here — the caller derives it from the same
623
+ * combined trace via `sumTraceCost`, so there is one cost path across the
624
+ * benchmark, callback, and `fit-trace cost` consumers.
625
+ */
626
+ // biome-ignore lint/complexity/noExcessiveCognitiveComplexity: stream-splitting state machine
627
+ async function splitAndSummarize(
628
+ runtime,
629
+ combinedPath,
630
+ agentPath,
631
+ supervisorPath,
632
+ ) {
633
+ const fs = runtime.fs;
634
+ const agentStream = fs.createWriteStream(agentPath);
635
+ const supStream = fs.createWriteStream(supervisorPath);
636
+ const rl = createInterface({
637
+ input: fs.createReadStream(combinedPath),
638
+ crlfDelay: Infinity,
639
+ });
640
+ let turns = 0;
641
+ let submission = "";
642
+ for await (const line of rl) {
643
+ if (!line.trim()) continue;
644
+ let event;
645
+ try {
646
+ event = JSON.parse(line);
647
+ } catch {
648
+ continue;
649
+ }
650
+ const target = event.source === "agent" ? agentStream : supStream;
651
+ target.write(line + "\n");
652
+ const inner = event.event;
653
+ if (!inner) continue;
654
+ if (event.source === "agent" && inner.type === "assistant") {
655
+ const text = extractText(inner);
656
+ if (text) submission = text;
657
+ }
658
+ if (event.source === "orchestrator" && inner.type === "summary") {
659
+ turns = inner.turns ?? 0;
660
+ }
661
+ }
662
+ await Promise.all([
663
+ new Promise((r) => agentStream.end(r)),
664
+ new Promise((r) => supStream.end(r)),
665
+ ]);
666
+ return { turns, submission };
667
+ }
668
+
669
+ function extractText(inner) {
670
+ const content = inner.message?.content ?? inner.content;
671
+ if (!Array.isArray(content)) return null;
672
+ for (let i = content.length - 1; i >= 0; i--) {
673
+ if (content[i].type === "text" && content[i].text) return content[i].text;
674
+ }
675
+ return null;
676
+ }
677
+
678
+ /**
679
+ * Factory function — wires real dependencies.
680
+ * @param {ConstructorParameters<typeof BenchmarkRunner>[0]} opts
681
+ * @returns {BenchmarkRunner}
682
+ */
683
+ export function createBenchmarkRunner(opts) {
684
+ return new BenchmarkRunner(opts);
685
+ }
686
+
687
+ // Internal exports used by tests.
688
+ export const __BASE_TOOLS = BASE_TOOLS;