@forwardimpact/libharness 0.1.22 → 1.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (86) hide show
  1. package/LICENSE +21 -201
  2. package/README.md +196 -80
  3. package/bin/fit-benchmark.js +44 -0
  4. package/bin/fit-harness.js +358 -0
  5. package/bin/fit-selfedit.js +165 -0
  6. package/bin/fit-trace.js +510 -0
  7. package/package.json +41 -11
  8. package/src/agent-runner.js +256 -0
  9. package/src/benchmark/apm-installer.js +207 -0
  10. package/src/benchmark/env-loader.js +158 -0
  11. package/src/benchmark/hook-env.js +40 -0
  12. package/src/benchmark/invariants.js +141 -0
  13. package/src/benchmark/judge.js +187 -0
  14. package/src/benchmark/npm-installer.js +87 -0
  15. package/src/benchmark/report.js +522 -0
  16. package/src/benchmark/result.js +127 -0
  17. package/src/benchmark/runner.js +583 -0
  18. package/src/benchmark/task-family.js +260 -0
  19. package/src/benchmark/workdir.js +298 -0
  20. package/src/commands/assert.js +153 -0
  21. package/src/commands/benchmark-definition.js +165 -0
  22. package/src/commands/benchmark-invariants.js +73 -0
  23. package/src/commands/benchmark-report.js +51 -0
  24. package/src/commands/benchmark-run.js +111 -0
  25. package/src/commands/by-discussion.js +94 -0
  26. package/src/commands/callback.js +119 -0
  27. package/src/commands/discuss.js +132 -0
  28. package/src/commands/facilitate.js +123 -0
  29. package/src/commands/output.js +36 -0
  30. package/src/commands/run.js +152 -0
  31. package/src/commands/supervise.js +136 -0
  32. package/src/commands/task-input.js +54 -0
  33. package/src/commands/tee.js +53 -0
  34. package/src/commands/trace.js +630 -0
  35. package/src/commands/work-tracker.js +35 -0
  36. package/src/cost.js +79 -0
  37. package/src/discuss-tools.js +173 -0
  38. package/src/discusser.js +394 -0
  39. package/src/events/github.js +161 -0
  40. package/src/facilitator.js +205 -0
  41. package/src/inbox-poller.js +81 -0
  42. package/src/index.js +72 -2
  43. package/src/judge.js +210 -0
  44. package/src/message-bus.js +118 -0
  45. package/src/orchestration-loop.js +330 -0
  46. package/src/orchestration-toolkit.js +441 -0
  47. package/src/orchestrator-helpers.js +23 -0
  48. package/src/profile-prompt.js +266 -0
  49. package/src/redaction.js +253 -0
  50. package/src/render/line-renderer.js +54 -0
  51. package/src/render/orchestrator-filter.js +19 -0
  52. package/src/render/palette.js +63 -0
  53. package/src/render/tool-hints.js +154 -0
  54. package/src/render/turn-renderer.js +96 -0
  55. package/src/reply-emitter.js +47 -0
  56. package/src/sequence-counter.js +21 -0
  57. package/src/signature-filter.js +27 -0
  58. package/src/supervisor.js +236 -0
  59. package/src/tee-writer.js +150 -0
  60. package/src/trace-collector.js +444 -0
  61. package/src/trace-github.js +473 -0
  62. package/src/trace-multi.js +101 -0
  63. package/src/trace-query.js +748 -0
  64. package/src/trace-render.js +211 -0
  65. package/src/trace-usage.js +249 -0
  66. package/src/fixture/assertions.js +0 -42
  67. package/src/fixture/cache.js +0 -50
  68. package/src/fixture/eval.js +0 -146
  69. package/src/fixture/index.js +0 -9
  70. package/src/fixture/pathway.js +0 -451
  71. package/src/fixture/services.js +0 -56
  72. package/src/mock/clients.js +0 -135
  73. package/src/mock/config.js +0 -45
  74. package/src/mock/data.js +0 -46
  75. package/src/mock/fs.js +0 -111
  76. package/src/mock/grpc.js +0 -94
  77. package/src/mock/http.js +0 -60
  78. package/src/mock/index.js +0 -36
  79. package/src/mock/infra.js +0 -219
  80. package/src/mock/logger.js +0 -42
  81. package/src/mock/observer.js +0 -74
  82. package/src/mock/resource-index.js +0 -95
  83. package/src/mock/service-callbacks.js +0 -39
  84. package/src/mock/services.js +0 -79
  85. package/src/mock/spy.js +0 -44
  86. package/src/mock/storage.js +0 -118
@@ -0,0 +1,583 @@
1
+ /**
2
+ * BenchmarkRunner — sole orchestrator for a task-family benchmark run.
3
+ *
4
+ * Phases per (task, runIndex):
5
+ * 1. WorkdirManager.start → seed CWD + run pre-flight probe
6
+ * 2. Supervisor session (agent + supervisor) → produce traces + submission
7
+ * 3. Invariants.runInvariants → exit-code-driven verdict via fd-3 NDJSON
8
+ * 4. Judge.runJudge → Conclude-driven verdict mapped to pass/fail
9
+ * 5. WorkdirManager.teardown → process-group cleanup
10
+ *
11
+ * Results stream as an async iterable AND are appended to
12
+ * `<output>/results.jsonl` for durability. The two paths are different
13
+ * consumers of the same record — the iterator drives CLI stdout mirroring,
14
+ * the JSONL append is the system of record.
15
+ */
16
+
17
+ import { createInterface } from "node:readline";
18
+ import { join, resolve as resolvePath } from "node:path";
19
+
20
+ import { DEFAULT_ENV_ALLOWLIST, createRedactor } from "../redaction.js";
21
+ import { sumTraceCost } from "../cost.js";
22
+ import { createSupervisor } from "../supervisor.js";
23
+ import { installApm as defaultInstallApm } from "./apm-installer.js";
24
+ import { installNpm as defaultInstallNpm } from "./npm-installer.js";
25
+ import { runJudge } from "./judge.js";
26
+ import { validateResultRecord } from "./result.js";
27
+ import { runInvariants } from "./invariants.js";
28
+ import { assertJudgeProfileStaged, loadTaskFamily } from "./task-family.js";
29
+ import { createWorkdirManager } from "./workdir.js";
30
+
31
+ const BASE_TOOLS = [
32
+ "Bash",
33
+ "Read",
34
+ "Glob",
35
+ "Grep",
36
+ "Write",
37
+ "Edit",
38
+ "Agent",
39
+ "TodoWrite",
40
+ ];
41
+
42
+ // Upper bound on a single supervised agent run. A run that produces no terminal
43
+ // message within this window is treated as a stall and recorded as an
44
+ // agentError, so the benchmark never hangs the event loop into a silent exit.
45
+ const AGENT_WATCHDOG_MS = 20 * 60 * 1000;
46
+
47
+ /** Sole orchestrator for a task-family benchmark run. */
48
+ export class BenchmarkRunner {
49
+ /**
50
+ * @param {object} opts
51
+ * @param {import("./task-family.js").TaskFamily | string} opts.family
52
+ * @param {number} opts.runs - Runs per task (≥ 1).
53
+ * @param {string} opts.output - Run-output directory.
54
+ * @param {string} opts.agentModel
55
+ * @param {string} opts.supervisorModel
56
+ * @param {string} opts.judgeModel
57
+ * @param {{agent?: string, judge?: string}} [opts.profiles]
58
+ * @param {Function} opts.query - SDK query (injected for testability).
59
+ * @param {string[]} [opts.allowedTools] - Agent tool allowlist (default: BASE_TOOLS).
60
+ * @param {number} [opts.maxTurns] - Agent-under-test turn budget.
61
+ * @param {number} [opts.termGraceMs] - SIGTERM→SIGKILL grace (ms) for the per-task process group.
62
+ * @param {Function} [opts.runAgent] - Test seam: replaces the agent-under-test
63
+ * session. Must return `{costUsd, turns, submission, agentError?}` and
64
+ * write a valid NDJSON trace to `workdir.agentTracePath`. Default uses
65
+ * `createAgentRunner` with the harness `BASE_TOOLS` allowlist. Internal
66
+ * testing only — not part of the public API.
67
+ * @param {import("@forwardimpact/libutil/runtime").Runtime} opts.runtime -
68
+ * Injected ambient collaborators (`fs`, `subprocess`, `clock`, `proc`),
69
+ * threaded into the installers, workdir manager, invariants, and judge.
70
+ * @param {Function} [opts.runInvariants] - Test seam: replaces `runInvariants`.
71
+ * Same contract as `runInvariants(task, ctx, runtime)`. Internal testing only.
72
+ * @param {Function} [opts.runJudge] - Test seam: replaces `runJudge`. Same
73
+ * contract as `runJudge(task, workdir, invariants, deps)` (deps carries
74
+ * `runtime`). Internal testing only.
75
+ * @param {Function} [opts.installApm] - Test seam: replaces `installApm`.
76
+ * Same contract as `installApm(family, outputDir, runtime)`. Lets tests
77
+ * inject a fake subprocess (or skip the install entirely) so the suite
78
+ * never shells out to a real `apm` binary. Internal testing only.
79
+ * @param {Function} [opts.installNpm] - Test seam: replaces `installNpm`.
80
+ * Same contract as `installNpm(family, stagingDir, runtime)`. Internal
81
+ * testing only.
82
+ */
83
+ constructor({
84
+ family,
85
+ runs,
86
+ output,
87
+ agentModel,
88
+ supervisorModel,
89
+ judgeModel,
90
+ profiles,
91
+ query,
92
+ allowedTools,
93
+ maxTurns,
94
+ task,
95
+ skillsFrom,
96
+ termGraceMs,
97
+ runtime,
98
+ // Test seams — default to the real implementations.
99
+ runAgent,
100
+ runInvariants: runInvariantsHook,
101
+ runJudge: runJudgeHook,
102
+ installApm: installApmHook,
103
+ installNpm: installNpmHook,
104
+ }) {
105
+ validateRunnerArgs({ family, runs, output, agentModel, query, runtime });
106
+ this.runtime = runtime;
107
+ this.familyInput = family;
108
+ this.runs = runs;
109
+ this.output = output;
110
+ this.agentModel = agentModel;
111
+ this.supervisorModel = supervisorModel;
112
+ this.judgeModel = judgeModel;
113
+ this.allowedTools = allowedTools ?? BASE_TOOLS;
114
+ this.profiles = {
115
+ agent: profiles?.agent ?? null,
116
+ judge: profiles?.judge ?? null,
117
+ };
118
+ this.query = query;
119
+ this.maxTurns = maxTurns;
120
+ this.taskFilter = task ?? null;
121
+ this.skillsFrom = skillsFrom ?? null;
122
+ this.termGraceMs = termGraceMs;
123
+ this._runAgentHook = runAgent ?? null;
124
+ this._runInvariantsHook = runInvariantsHook ?? runInvariants;
125
+ this._runJudgeHook = runJudgeHook ?? runJudge;
126
+ this._installApmHook = installApmHook ?? defaultInstallApm;
127
+ this._installNpmHook = installNpmHook ?? defaultInstallNpm;
128
+ }
129
+
130
+ /**
131
+ * Yield one ResultRecord per (task, runIndex).
132
+ * @returns {AsyncGenerator<object>}
133
+ */
134
+ async *run() {
135
+ const runtime = this.runtime;
136
+ const family =
137
+ typeof this.familyInput === "string"
138
+ ? await loadTaskFamily(this.familyInput, runtime)
139
+ : this.familyInput;
140
+
141
+ await runtime.fs.mkdir(this.output, { recursive: true });
142
+ const { stagingDir, skillSetHash, judgeProfilesDir } =
143
+ await this._installApmHook(family, this.output, runtime, {
144
+ skillsFrom: this.skillsFrom,
145
+ });
146
+ await this._installNpmHook(family, stagingDir, runtime);
147
+
148
+ let tasks = family.tasks();
149
+ if (this.taskFilter) {
150
+ const matched = tasks.filter((t) => t.id === this.taskFilter);
151
+ if (matched.length === 0) {
152
+ const available = tasks.map((t) => t.id).join(", ");
153
+ throw new Error(
154
+ `no task '${this.taskFilter}' in family; available: ${available}`,
155
+ );
156
+ }
157
+ tasks = matched;
158
+ }
159
+ if (this.profiles.judge) {
160
+ await assertJudgeProfileStaged(
161
+ family,
162
+ judgeProfilesDir,
163
+ this.profiles.judge,
164
+ runtime,
165
+ );
166
+ }
167
+
168
+ const wm = createWorkdirManager({
169
+ stagingDir,
170
+ runOutputDir: this.output,
171
+ termGraceMs: this.termGraceMs,
172
+ familyRootPath: family.rootPath,
173
+ runtime,
174
+ });
175
+
176
+ const resultsPath = join(this.output, "results.jsonl");
177
+ const resultsStream = runtime.fs.createWriteStream(resultsPath, {
178
+ flags: "a",
179
+ });
180
+ try {
181
+ for (const task of tasks) {
182
+ for (let runIndex = 0; runIndex < this.runs; runIndex++) {
183
+ const record = await this.#runOne(
184
+ family,
185
+ wm,
186
+ task,
187
+ runIndex,
188
+ skillSetHash,
189
+ judgeProfilesDir,
190
+ );
191
+ await writeRecord(resultsStream, record);
192
+ yield record;
193
+ }
194
+ }
195
+ } finally {
196
+ await new Promise((r) => resultsStream.end(r));
197
+ }
198
+ }
199
+
200
+ async #runOne(family, wm, task, runIndex, skillSetHash, judgeProfilesDir) {
201
+ const t0 = this.runtime.clock.now();
202
+ const workdir = await wm.start(task, runIndex);
203
+ try {
204
+ if (workdir.preflightError) {
205
+ const record = this.#buildPreflightFailureRecord({
206
+ task,
207
+ runIndex,
208
+ workdir,
209
+ skillSetHash,
210
+ familyRevision: family.familyRevision,
211
+ durationMs: this.runtime.clock.now() - t0,
212
+ });
213
+ return this.#validateOrFallback(
214
+ record,
215
+ resultsRecordKey(task, runIndex),
216
+ );
217
+ }
218
+ const agentRun = await this.#runAgentSafe(task, workdir);
219
+ const { costUsd, turns, submission, agentError } = agentRun;
220
+ const breakdown = agentRun.costBreakdown ?? { agent: 0, supervisor: 0 };
221
+ const invariants = await this._runInvariantsHook(
222
+ task,
223
+ {
224
+ cwd: workdir.cwd,
225
+ port: workdir.port,
226
+ runDir: workdir.runDir,
227
+ familyDir: family.rootPath,
228
+ },
229
+ this.runtime,
230
+ );
231
+ let judgeVerdict = null;
232
+ let judgeCost = 0;
233
+ if (task.paths.judge) {
234
+ const judgeContext = await this.#buildJudgeContext(
235
+ task,
236
+ workdir,
237
+ skillSetHash,
238
+ );
239
+ const judgeResult = await this._runJudgeHook(
240
+ task,
241
+ workdir,
242
+ invariants,
243
+ {
244
+ query: this.query,
245
+ model: this.judgeModel,
246
+ judgeProfile: this.profiles.judge ?? undefined,
247
+ profilesDir: judgeProfilesDir,
248
+ runtime: this.runtime,
249
+ },
250
+ judgeContext,
251
+ );
252
+ judgeCost = judgeResult.costUsd ?? 0;
253
+ // The record's judgeVerdict carries only the verdict + summary; the
254
+ // judge's cost is folded into costUsd / costBreakdown instead.
255
+ judgeVerdict = {
256
+ verdict: judgeResult.verdict,
257
+ summary: judgeResult.summary,
258
+ };
259
+ }
260
+ const verdict =
261
+ invariants.verdict === "pass" &&
262
+ (judgeVerdict === null || judgeVerdict.verdict === "pass")
263
+ ? "pass"
264
+ : "fail";
265
+ const record = {
266
+ taskId: task.id,
267
+ runIndex,
268
+ verdict,
269
+ invariants,
270
+ submission,
271
+ ...(judgeVerdict && { judgeVerdict }),
272
+ costUsd: costUsd + judgeCost,
273
+ costBreakdown: {
274
+ agent: breakdown.agent ?? 0,
275
+ supervisor: breakdown.supervisor ?? 0,
276
+ judge: judgeCost,
277
+ },
278
+ turns,
279
+ agentTracePath: workdir.agentTracePath,
280
+ supervisorTracePath: workdir.supervisorTracePath,
281
+ judgeTracePath: workdir.judgeTracePath,
282
+ profiles: {
283
+ agent: this.profiles.agent,
284
+ supervisor: null,
285
+ judge: this.profiles.judge,
286
+ },
287
+ model: {
288
+ agent: this.agentModel,
289
+ supervisor: this.supervisorModel,
290
+ judge: this.judgeModel,
291
+ },
292
+ skillSetHash,
293
+ familyRevision: family.familyRevision,
294
+ durationMs: this.runtime.clock.now() - t0,
295
+ ...(agentError && { agentError }),
296
+ };
297
+ return this.#validateOrFallback(record, resultsRecordKey(task, runIndex));
298
+ } finally {
299
+ await wm.teardown(workdir).catch(() => {});
300
+ }
301
+ }
302
+
303
+ /**
304
+ * Dispatch to either the injected hook or the default `#runAgent`. Either
305
+ * path can throw; catch here so a thrown error becomes an `agentError` on
306
+ * the record (spec criterion 1: records on agent failure) rather than
307
+ * aborting the whole iterator.
308
+ */
309
+ async #runAgentSafe(task, workdir) {
310
+ try {
311
+ if (this._runAgentHook) {
312
+ const r = await this._runAgentHook(task, workdir, this);
313
+ return { agentError: null, ...r };
314
+ }
315
+ return await this.#runAgent(task, workdir);
316
+ } catch (e) {
317
+ return {
318
+ costUsd: 0,
319
+ costBreakdown: { agent: 0, supervisor: 0 },
320
+ turns: 0,
321
+ submission: "",
322
+ agentError: { message: e.message ?? String(e), aborted: false },
323
+ };
324
+ }
325
+ }
326
+
327
+ /**
328
+ * Run the agent-under-test under a Supervisor. The supervisor writes
329
+ * a combined tagged NDJSON trace; after the session we split it into
330
+ * agent.ndjson and supervisor.ndjson and extract cost/turns/submission.
331
+ */
332
+ async #runAgent(task, workdir) {
333
+ const fs = this.runtime.fs;
334
+ const combinedPath = join(workdir.runDir, ".combined.ndjson");
335
+ const combinedStream = fs.createWriteStream(combinedPath);
336
+ const supervisorInstructions = task.paths.supervisor
337
+ ? await fs.readFile(task.paths.supervisor, "utf8").catch(() => null)
338
+ : null;
339
+ const supervisor = createSupervisor({
340
+ supervisorCwd: workdir.cwd,
341
+ agentCwd: workdir.cwd,
342
+ query: this.query,
343
+ output: combinedStream,
344
+ agentModel: this.agentModel,
345
+ supervisorModel: this.supervisorModel,
346
+ maxTurns: this.maxTurns ?? 50,
347
+ allowedTools: this.allowedTools,
348
+ ...(this.profiles.agent && { agentProfile: this.profiles.agent }),
349
+ ...(supervisorInstructions && { taskAmend: supervisorInstructions }),
350
+ redactor: createRedactor({
351
+ allowlist: [...DEFAULT_ENV_ALLOWLIST, ...(workdir.envNames ?? [])],
352
+ runtime: this.runtime,
353
+ }),
354
+ runtime: this.runtime,
355
+ });
356
+ const instructions = await fs.readFile(task.paths.instructions, "utf8");
357
+ let agentError = null;
358
+ // Watchdog: a supervised session can hang without settling (e.g. the agent
359
+ // SDK subprocess exits without a terminal message), which would empty the
360
+ // event loop and exit the process mid-run with zero records. Race the run
361
+ // against a bounded timer so a stall becomes an `agentError` record instead
362
+ // of a silent exit; the timer also keeps the loop alive until it fires.
363
+ let watchdog;
364
+ try {
365
+ const result = await Promise.race([
366
+ supervisor.run(instructions),
367
+ new Promise((_, reject) => {
368
+ watchdog = this.runtime.clock.setTimeout(
369
+ () =>
370
+ reject(
371
+ new Error(
372
+ `agent run produced no result within ${AGENT_WATCHDOG_MS}ms (possible stall)`,
373
+ ),
374
+ ),
375
+ AGENT_WATCHDOG_MS,
376
+ );
377
+ }),
378
+ ]);
379
+ if (!result.success && !result.concluded) {
380
+ agentError = { message: "supervisor did not succeed", aborted: false };
381
+ }
382
+ } catch (e) {
383
+ agentError = { message: e.message ?? String(e), aborted: false };
384
+ } finally {
385
+ this.runtime.clock.clearTimeout(watchdog);
386
+ await new Promise((r) => combinedStream.end(r));
387
+ }
388
+ const summary = await splitAndSummarize(
389
+ this.runtime,
390
+ combinedPath,
391
+ workdir.agentTracePath,
392
+ workdir.supervisorTracePath,
393
+ );
394
+ // Cost is summed across every participant's result events from the one
395
+ // combined trace, attributed per source. Read before unlinking.
396
+ const combined = await fs.readFile(combinedPath, "utf8");
397
+ const { totalCostUsd, bySource } = sumTraceCost(combined.split("\n"));
398
+ await fs.unlink(combinedPath).catch(() => {});
399
+ return {
400
+ ...summary,
401
+ costUsd: totalCostUsd,
402
+ costBreakdown: {
403
+ agent: bySource.agent ?? 0,
404
+ supervisor: bySource.supervisor ?? 0,
405
+ },
406
+ agentError,
407
+ };
408
+ }
409
+
410
+ async #buildJudgeContext(task, workdir, skillSetHash) {
411
+ const fs = this.runtime.fs;
412
+ const agentInstructions = await fs.readFile(
413
+ task.paths.instructions,
414
+ "utf8",
415
+ );
416
+ let agentProfile = "";
417
+ if (this.profiles.agent) {
418
+ const profilePath = resolvePath(
419
+ workdir.cwd,
420
+ ".claude/agents",
421
+ `${this.profiles.agent}.md`,
422
+ );
423
+ agentProfile = await fs.readFile(profilePath, "utf8").catch(() => "");
424
+ }
425
+ return { agentInstructions, agentProfile, skillSetHash };
426
+ }
427
+
428
+ #buildPreflightFailureRecord({
429
+ task,
430
+ runIndex,
431
+ workdir,
432
+ skillSetHash,
433
+ familyRevision,
434
+ durationMs,
435
+ }) {
436
+ return {
437
+ taskId: task.id,
438
+ runIndex,
439
+ verdict: "fail",
440
+ costUsd: 0,
441
+ turns: 0,
442
+ preflightError: workdir.preflightError,
443
+ profiles: {
444
+ agent: this.profiles.agent,
445
+ supervisor: null,
446
+ judge: this.profiles.judge,
447
+ },
448
+ model: {
449
+ agent: this.agentModel,
450
+ supervisor: this.supervisorModel,
451
+ judge: this.judgeModel,
452
+ },
453
+ skillSetHash,
454
+ familyRevision,
455
+ durationMs,
456
+ agentTracePath: workdir.agentTracePath,
457
+ supervisorTracePath: workdir.supervisorTracePath,
458
+ judgeTracePath: workdir.judgeTracePath,
459
+ };
460
+ }
461
+
462
+ #validateOrFallback(record, key) {
463
+ try {
464
+ validateResultRecord(record);
465
+ return record;
466
+ } catch (e) {
467
+ // The runner constructed the record — a schema failure is a real bug,
468
+ // not bad family input. Emit a noisy fallback so the iterator stays
469
+ // consumable and the agent budget isn't silently dropped.
470
+ return {
471
+ taskId: record.taskId ?? key.taskId,
472
+ runIndex: record.runIndex ?? key.runIndex,
473
+ verdict: "fail",
474
+ schemaError: e.message ?? String(e),
475
+ };
476
+ }
477
+ }
478
+ }
479
+
480
+ /**
481
+ * Validate the required BenchmarkRunner constructor arguments. Extracted from
482
+ * the constructor to keep its cognitive complexity under the lint ceiling.
483
+ */
484
+ function validateRunnerArgs({
485
+ family,
486
+ runs,
487
+ output,
488
+ agentModel,
489
+ query,
490
+ runtime,
491
+ }) {
492
+ if (!family) throw new Error("family is required");
493
+ if (!Number.isInteger(runs) || runs < 1)
494
+ throw new Error("runs must be an integer ≥ 1");
495
+ if (!output) throw new Error("output is required");
496
+ if (!agentModel) throw new Error("agentModel is required");
497
+ if (!query) throw new Error("query is required");
498
+ if (!runtime) throw new Error("runtime is required");
499
+ }
500
+
501
+ function resultsRecordKey(task, runIndex) {
502
+ return { taskId: task.id, runIndex };
503
+ }
504
+
505
+ async function writeRecord(stream, record) {
506
+ const line = JSON.stringify(record) + "\n";
507
+ await new Promise((res, rej) => {
508
+ stream.write(line, (err) => (err ? rej(err) : res()));
509
+ });
510
+ }
511
+
512
+ /**
513
+ * Split the combined supervisor trace into agent and supervisor files and
514
+ * extract turn count and submission in a single pass. Agent-source events go
515
+ * to `agentPath`; supervisor and orchestrator events go to `supervisorPath`.
516
+ *
517
+ * Cost is deliberately not summed here — the caller derives it from the same
518
+ * combined trace via `sumTraceCost`, so there is one cost path across the
519
+ * benchmark, callback, and `fit-trace cost` consumers.
520
+ */
521
+ // biome-ignore lint/complexity/noExcessiveCognitiveComplexity: stream-splitting state machine
522
+ async function splitAndSummarize(
523
+ runtime,
524
+ combinedPath,
525
+ agentPath,
526
+ supervisorPath,
527
+ ) {
528
+ const fs = runtime.fs;
529
+ const agentStream = fs.createWriteStream(agentPath);
530
+ const supStream = fs.createWriteStream(supervisorPath);
531
+ const rl = createInterface({
532
+ input: fs.createReadStream(combinedPath),
533
+ crlfDelay: Infinity,
534
+ });
535
+ let turns = 0;
536
+ let submission = "";
537
+ for await (const line of rl) {
538
+ if (!line.trim()) continue;
539
+ let event;
540
+ try {
541
+ event = JSON.parse(line);
542
+ } catch {
543
+ continue;
544
+ }
545
+ const target = event.source === "agent" ? agentStream : supStream;
546
+ target.write(line + "\n");
547
+ const inner = event.event;
548
+ if (!inner) continue;
549
+ if (event.source === "agent" && inner.type === "assistant") {
550
+ const text = extractText(inner);
551
+ if (text) submission = text;
552
+ }
553
+ if (event.source === "orchestrator" && inner.type === "summary") {
554
+ turns = inner.turns ?? 0;
555
+ }
556
+ }
557
+ await Promise.all([
558
+ new Promise((r) => agentStream.end(r)),
559
+ new Promise((r) => supStream.end(r)),
560
+ ]);
561
+ return { turns, submission };
562
+ }
563
+
564
+ function extractText(inner) {
565
+ const content = inner.message?.content ?? inner.content;
566
+ if (!Array.isArray(content)) return null;
567
+ for (let i = content.length - 1; i >= 0; i--) {
568
+ if (content[i].type === "text" && content[i].text) return content[i].text;
569
+ }
570
+ return null;
571
+ }
572
+
573
+ /**
574
+ * Factory function — wires real dependencies.
575
+ * @param {ConstructorParameters<typeof BenchmarkRunner>[0]} opts
576
+ * @returns {BenchmarkRunner}
577
+ */
578
+ export function createBenchmarkRunner(opts) {
579
+ return new BenchmarkRunner(opts);
580
+ }
581
+
582
+ // Internal exports used by tests.
583
+ export const __BASE_TOOLS = BASE_TOOLS;