@forwardimpact/libharness 0.1.22 → 1.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (86) hide show
  1. package/LICENSE +21 -201
  2. package/README.md +196 -80
  3. package/bin/fit-benchmark.js +44 -0
  4. package/bin/fit-harness.js +358 -0
  5. package/bin/fit-selfedit.js +165 -0
  6. package/bin/fit-trace.js +510 -0
  7. package/package.json +41 -11
  8. package/src/agent-runner.js +256 -0
  9. package/src/benchmark/apm-installer.js +207 -0
  10. package/src/benchmark/env-loader.js +158 -0
  11. package/src/benchmark/hook-env.js +40 -0
  12. package/src/benchmark/invariants.js +141 -0
  13. package/src/benchmark/judge.js +187 -0
  14. package/src/benchmark/npm-installer.js +87 -0
  15. package/src/benchmark/report.js +522 -0
  16. package/src/benchmark/result.js +127 -0
  17. package/src/benchmark/runner.js +583 -0
  18. package/src/benchmark/task-family.js +260 -0
  19. package/src/benchmark/workdir.js +298 -0
  20. package/src/commands/assert.js +153 -0
  21. package/src/commands/benchmark-definition.js +165 -0
  22. package/src/commands/benchmark-invariants.js +73 -0
  23. package/src/commands/benchmark-report.js +51 -0
  24. package/src/commands/benchmark-run.js +111 -0
  25. package/src/commands/by-discussion.js +94 -0
  26. package/src/commands/callback.js +119 -0
  27. package/src/commands/discuss.js +132 -0
  28. package/src/commands/facilitate.js +123 -0
  29. package/src/commands/output.js +36 -0
  30. package/src/commands/run.js +152 -0
  31. package/src/commands/supervise.js +136 -0
  32. package/src/commands/task-input.js +54 -0
  33. package/src/commands/tee.js +53 -0
  34. package/src/commands/trace.js +630 -0
  35. package/src/commands/work-tracker.js +35 -0
  36. package/src/cost.js +79 -0
  37. package/src/discuss-tools.js +173 -0
  38. package/src/discusser.js +394 -0
  39. package/src/events/github.js +161 -0
  40. package/src/facilitator.js +205 -0
  41. package/src/inbox-poller.js +81 -0
  42. package/src/index.js +72 -2
  43. package/src/judge.js +210 -0
  44. package/src/message-bus.js +118 -0
  45. package/src/orchestration-loop.js +330 -0
  46. package/src/orchestration-toolkit.js +441 -0
  47. package/src/orchestrator-helpers.js +23 -0
  48. package/src/profile-prompt.js +266 -0
  49. package/src/redaction.js +253 -0
  50. package/src/render/line-renderer.js +54 -0
  51. package/src/render/orchestrator-filter.js +19 -0
  52. package/src/render/palette.js +63 -0
  53. package/src/render/tool-hints.js +154 -0
  54. package/src/render/turn-renderer.js +96 -0
  55. package/src/reply-emitter.js +47 -0
  56. package/src/sequence-counter.js +21 -0
  57. package/src/signature-filter.js +27 -0
  58. package/src/supervisor.js +236 -0
  59. package/src/tee-writer.js +150 -0
  60. package/src/trace-collector.js +444 -0
  61. package/src/trace-github.js +473 -0
  62. package/src/trace-multi.js +101 -0
  63. package/src/trace-query.js +748 -0
  64. package/src/trace-render.js +211 -0
  65. package/src/trace-usage.js +249 -0
  66. package/src/fixture/assertions.js +0 -42
  67. package/src/fixture/cache.js +0 -50
  68. package/src/fixture/eval.js +0 -146
  69. package/src/fixture/index.js +0 -9
  70. package/src/fixture/pathway.js +0 -451
  71. package/src/fixture/services.js +0 -56
  72. package/src/mock/clients.js +0 -135
  73. package/src/mock/config.js +0 -45
  74. package/src/mock/data.js +0 -46
  75. package/src/mock/fs.js +0 -111
  76. package/src/mock/grpc.js +0 -94
  77. package/src/mock/http.js +0 -60
  78. package/src/mock/index.js +0 -36
  79. package/src/mock/infra.js +0 -219
  80. package/src/mock/logger.js +0 -42
  81. package/src/mock/observer.js +0 -74
  82. package/src/mock/resource-index.js +0 -95
  83. package/src/mock/service-callbacks.js +0 -39
  84. package/src/mock/services.js +0 -79
  85. package/src/mock/spy.js +0 -44
  86. package/src/mock/storage.js +0 -118
@@ -0,0 +1,522 @@
1
+ /**
2
+ * ReportAggregator — read a run-output directory's `results.jsonl`, group
3
+ * records by `taskId`, and compute pass@k via the OpenAI HumanEval
4
+ * unbiased estimator: `1 - C(n-c, k) / C(n, k)`.
5
+ *
6
+ * When `includeRuns` is true, each task carries per-run detail (invariant
7
+ * checks, judge commentary, cost, duration) and the text renderer produces
8
+ * a full markdown report instead of just the pass@k table.
9
+ *
10
+ * Records that fail schema validation are skipped with a stderr warning
11
+ * (counted under `totals.skipped`) so a corrupt line cannot abort the
12
+ * whole report.
13
+ */
14
+
15
+ import { join } from "node:path";
16
+
17
+ import { validateResultRecord } from "./result.js";
18
+
19
+ /**
20
+ * @typedef {object} RunDetail
21
+ * @property {number} runIndex
22
+ * @property {"pass"|"fail"} verdict
23
+ * @property {{verdict: string, details: unknown[], exitCode: number}} [invariants]
24
+ * @property {{verdict: string, summary: string}} [judgeVerdict]
25
+ * @property {number} costUsd
26
+ * @property {number} turns
27
+ * @property {number} durationMs
28
+ * @property {{message: string, aborted: boolean}} [agentError]
29
+ * @property {{phase: string, message: string, exitCode: number}} [preflightError]
30
+ */
31
+
32
+ /**
33
+ * @typedef {object} TaskReport
34
+ * @property {string} taskId
35
+ * @property {number} n - Total runs.
36
+ * @property {number} c - Passing runs.
37
+ * @property {Record<string|number, number|null>} passAtK
38
+ * @property {RunDetail[]} [runs] - Per-run detail (only when includeRuns).
39
+ */
40
+
41
+ /**
42
+ * @param {{inputDir: string, kValues: number[], includeRuns?: boolean, runtime: import("@forwardimpact/libutil/runtime").Runtime}} opts
43
+ * @returns {Promise<{tasks: TaskReport[], totals: object}>}
44
+ */
45
+ export async function aggregate({
46
+ inputDir,
47
+ kValues,
48
+ includeRuns = false,
49
+ runtime,
50
+ }) {
51
+ if (!runtime) throw new Error("runtime is required");
52
+ const records = await loadRecords(inputDir, runtime);
53
+ const grouped = groupByTask(records.records);
54
+ const tasks = [];
55
+ let totalRuns = 0;
56
+ let totalCost = 0;
57
+ const allDurations = [];
58
+ const allTurns = [];
59
+ let firstRecord = null;
60
+
61
+ for (const [taskId, group] of grouped) {
62
+ const n = group.length;
63
+ const c = group.filter((r) => r.verdict === "pass").length;
64
+ totalRuns += n;
65
+ const passAtK = {};
66
+ for (const k of kValues) passAtK[k] = passAtKValue(n, c, k);
67
+
68
+ const task = { taskId, n, c, passAtK };
69
+
70
+ if (includeRuns) {
71
+ if (!firstRecord) firstRecord = group[0];
72
+ const accumulators = { allDurations, allTurns };
73
+ task.runs = group
74
+ .map((r) => {
75
+ totalCost += r.costUsd ?? 0;
76
+ return buildRunDetail(r, accumulators);
77
+ })
78
+ .sort((a, b) => a.runIndex - b.runIndex);
79
+ }
80
+
81
+ tasks.push(task);
82
+ }
83
+ tasks.sort((a, b) =>
84
+ a.taskId < b.taskId ? -1 : a.taskId > b.taskId ? 1 : 0,
85
+ );
86
+
87
+ const totals = {
88
+ tasks: tasks.length,
89
+ runs: totalRuns,
90
+ skipped: records.skipped,
91
+ };
92
+
93
+ if (includeRuns) {
94
+ totals.costUsd = totalCost;
95
+ totals.medianDurationMs = median(allDurations);
96
+ totals.medianTurns = median(allTurns);
97
+ totals.model = firstRecord?.model ?? "";
98
+ totals.skillSetHash = firstRecord?.skillSetHash ?? "";
99
+ totals.familyRevision = firstRecord?.familyRevision ?? "";
100
+ }
101
+
102
+ return { tasks, totals };
103
+ }
104
+
105
+ /**
106
+ * Build a normalized per-run detail object and accumulate duration/turn
107
+ * samples for median calculation. Extracted from `aggregate` to keep its
108
+ * cognitive complexity below the lint ceiling.
109
+ * @param {object} r - Raw record.
110
+ * @param {{allDurations: number[], allTurns: number[]}} acc
111
+ * @returns {RunDetail}
112
+ */
113
+ function buildRunDetail(r, acc) {
114
+ if (r.durationMs != null) acc.allDurations.push(r.durationMs);
115
+ if (r.turns != null) acc.allTurns.push(r.turns);
116
+ return {
117
+ runIndex: r.runIndex,
118
+ verdict: r.verdict,
119
+ ...(r.invariants && { invariants: r.invariants }),
120
+ ...(r.judgeVerdict && { judgeVerdict: r.judgeVerdict }),
121
+ costUsd: r.costUsd ?? 0,
122
+ turns: r.turns ?? 0,
123
+ durationMs: r.durationMs ?? 0,
124
+ ...(r.agentError && { agentError: r.agentError }),
125
+ ...(r.preflightError && { preflightError: r.preflightError }),
126
+ };
127
+ }
128
+
129
+ /**
130
+ * Render an aggregate report as markdown. When the report contains per-run
131
+ * detail (from `includeRuns: true`), renders a full report with summary,
132
+ * pass@k table, and per-task detail sections. Otherwise falls back to the
133
+ * compact pass@k table.
134
+ * @param {Awaited<ReturnType<typeof aggregate>>} report
135
+ * @param {number[]} kValues
136
+ * @returns {string}
137
+ */
138
+ export function renderTextReport(report, kValues) {
139
+ if (report.tasks[0]?.runs) {
140
+ return renderFullReport(report, kValues);
141
+ }
142
+ return renderCompactReport(report, kValues);
143
+ }
144
+
145
+ // ---------------------------------------------------------------------------
146
+ // Compact report (legacy path)
147
+ // ---------------------------------------------------------------------------
148
+
149
+ function renderCompactReport(report, kValues) {
150
+ const lines = [
151
+ renderPassAtKTable(report, kValues),
152
+ "",
153
+ renderTotalsLine(report),
154
+ ];
155
+ return lines.join("\n");
156
+ }
157
+
158
+ // ---------------------------------------------------------------------------
159
+ // Full report
160
+ // ---------------------------------------------------------------------------
161
+
162
+ function renderFullReport(report, kValues) {
163
+ const sections = [
164
+ renderSummary(report),
165
+ "## Pass@k",
166
+ "",
167
+ renderPassAtKTable(report, kValues),
168
+ "",
169
+ renderTotalsLine(report),
170
+ "",
171
+ "## Task Details",
172
+ ];
173
+
174
+ for (const task of report.tasks) {
175
+ sections.push("");
176
+ sections.push(renderTaskDetail(task));
177
+ }
178
+
179
+ return sections.join("\n");
180
+ }
181
+
182
+ function renderSummary(report) {
183
+ const { totals } = report;
184
+ const passing = report.tasks.filter((t) => t.c > 0 && t.c === t.n).length;
185
+ const icon = statusIcon(passing === totals.tasks);
186
+ const lines = [
187
+ "# Benchmark Report",
188
+ "",
189
+ `${icon} **${passing}/${totals.tasks} tasks passing** | ${totals.runs} runs${totals.skipped ? ` | ${totals.skipped} skipped` : ""}`,
190
+ ];
191
+
192
+ const headers = [];
193
+ const values = [];
194
+ if (totals.costUsd != null) {
195
+ headers.push("Cost");
196
+ values.push(formatCost(totals.costUsd));
197
+ }
198
+ if (totals.medianDurationMs != null) {
199
+ headers.push("Median Duration");
200
+ values.push(formatDuration(totals.medianDurationMs));
201
+ }
202
+ if (totals.medianTurns != null) {
203
+ headers.push("Median Turns");
204
+ values.push(String(totals.medianTurns));
205
+ }
206
+ if (headers.length) {
207
+ lines.push("");
208
+ lines.push(`| ${headers.join(" | ")} |`);
209
+ lines.push(`| ${headers.map(() => "---").join(" | ")} |`);
210
+ lines.push(`| ${values.join(" | ")} |`);
211
+ }
212
+
213
+ const meta = [];
214
+ if (totals.model) {
215
+ meta.push(`Agent: \`${totals.model.agent}\``);
216
+ meta.push(`Supervisor: \`${totals.model.supervisor}\``);
217
+ meta.push(`Judge: \`${totals.model.judge}\``);
218
+ }
219
+ if (totals.skillSetHash) meta.push(`Skill set: \`${totals.skillSetHash}\``);
220
+ if (totals.familyRevision) meta.push(`Family: \`${totals.familyRevision}\``);
221
+ if (meta.length) {
222
+ lines.push("");
223
+ lines.push(meta.join(" | "));
224
+ }
225
+
226
+ lines.push("");
227
+ return lines.join("\n");
228
+ }
229
+
230
+ // ---------------------------------------------------------------------------
231
+ // Pass@k table (shared between compact and full)
232
+ // ---------------------------------------------------------------------------
233
+
234
+ function renderPassAtKTable(report, kValues) {
235
+ const header = ["taskId", "n", "c", ...kValues.map((k) => `pass@${k}`)];
236
+ const rows = [header, header.map(() => "---")];
237
+ for (const t of report.tasks) {
238
+ rows.push([
239
+ t.taskId,
240
+ String(t.n),
241
+ String(t.c),
242
+ ...kValues.map((k) => formatPassAt(t.passAtK[k])),
243
+ ]);
244
+ }
245
+ return rows.map((r) => `| ${r.join(" | ")} |`).join("\n");
246
+ }
247
+
248
+ function renderTotalsLine(report) {
249
+ return `Totals — tasks: ${report.totals.tasks}, runs: ${report.totals.runs}, skipped: ${report.totals.skipped}`;
250
+ }
251
+
252
+ // ---------------------------------------------------------------------------
253
+ // Per-task detail
254
+ // ---------------------------------------------------------------------------
255
+
256
+ function renderTaskDetail(task) {
257
+ const runs = task.runs ?? [];
258
+ const icon = statusIcon(task.c === task.n);
259
+ const singleRun = runs.length === 1;
260
+
261
+ const lines = [
262
+ `### ${task.taskId}`,
263
+ "",
264
+ `${icon} **${task.c}/${task.n} runs passed**`,
265
+ ];
266
+
267
+ lines.push("", renderRunsTable(runs));
268
+
269
+ const checks = renderInvariantChecks(runs, singleRun);
270
+ if (checks) lines.push("", checks);
271
+
272
+ const commentary = renderJudgeCommentary(runs, singleRun);
273
+ if (commentary) lines.push("", commentary);
274
+
275
+ const errors = renderErrors(runs);
276
+ if (errors) lines.push("", errors);
277
+
278
+ return lines.join("\n");
279
+ }
280
+
281
+ function renderRunsTable(runs) {
282
+ const header = [
283
+ "Run",
284
+ "Verdict",
285
+ "Invariants",
286
+ "Judge",
287
+ "Cost",
288
+ "Turns",
289
+ "Duration",
290
+ ];
291
+ const rows = [header, header.map(() => "---")];
292
+ for (const r of runs) {
293
+ const invariantsCell = r.preflightError
294
+ ? "preflight error"
295
+ : r.invariants
296
+ ? statusIcon(r.invariants.verdict === "pass")
297
+ : "—";
298
+ const judgeCell = r.preflightError
299
+ ? "—"
300
+ : r.judgeVerdict
301
+ ? statusIcon(r.judgeVerdict.verdict === "pass")
302
+ : "—";
303
+ rows.push([
304
+ String(r.runIndex),
305
+ statusIcon(r.verdict === "pass"),
306
+ invariantsCell,
307
+ judgeCell,
308
+ formatCost(r.costUsd),
309
+ String(r.turns),
310
+ formatDuration(r.durationMs),
311
+ ]);
312
+ }
313
+ return rows.map((r) => `| ${r.join(" | ")} |`).join("\n");
314
+ }
315
+
316
+ function renderInvariantChecks(runs, singleRun) {
317
+ const rows = collectInvariantRows(runs);
318
+ if (!rows.length) return null;
319
+
320
+ const header = singleRun
321
+ ? ["Check", "Result", "Message"]
322
+ : ["Run", "Check", "Result", "Message"];
323
+ const lines = [
324
+ "#### Invariant Checks",
325
+ "",
326
+ `| ${header.join(" | ")} |`,
327
+ `| ${header.map(() => "---").join(" | ")} |`,
328
+ ];
329
+ for (const row of rows) {
330
+ const cells = singleRun
331
+ ? [row.check, row.result, row.message]
332
+ : [String(row.run), row.check, row.result, row.message];
333
+ lines.push(`| ${cells.join(" | ")} |`);
334
+ }
335
+ return lines.join("\n");
336
+ }
337
+
338
+ function collectInvariantRows(runs) {
339
+ const rows = [];
340
+ for (const r of runs) {
341
+ if (!r.invariants?.details?.length) continue;
342
+ for (const d of r.invariants.details) {
343
+ rows.push({
344
+ run: r.runIndex,
345
+ check: escapeCell(String(d.test ?? "(unnamed)")),
346
+ result: statusIcon(d.pass),
347
+ message: escapeCell(String(d.message ?? "")),
348
+ });
349
+ }
350
+ }
351
+ return rows;
352
+ }
353
+
354
+ function renderJudgeCommentary(runs, singleRun) {
355
+ const entries = runs.filter((r) => r.judgeVerdict?.summary);
356
+ if (!entries.length) return null;
357
+
358
+ const lines = ["#### Judge Commentary", ""];
359
+ for (let i = 0; i < entries.length; i++) {
360
+ const r = entries[i];
361
+ const summary = r.judgeVerdict.summary.replace(/\n/g, "\n> ");
362
+ if (singleRun) {
363
+ lines.push(`> ${summary}`);
364
+ } else {
365
+ lines.push(`> **Run ${r.runIndex}:** ${summary}`);
366
+ }
367
+ if (i < entries.length - 1) lines.push(">");
368
+ }
369
+ return lines.join("\n");
370
+ }
371
+
372
+ function renderErrors(runs) {
373
+ const lines = [];
374
+ for (const r of runs) {
375
+ if (r.agentError) {
376
+ lines.push(
377
+ `- **Run ${r.runIndex}:** Agent error — "${escapeCell(r.agentError.message)}" (aborted: ${r.agentError.aborted})`,
378
+ );
379
+ }
380
+ if (r.preflightError) {
381
+ lines.push(
382
+ `- **Run ${r.runIndex}:** Preflight error — "${escapeCell(r.preflightError.message)}" (exit ${r.preflightError.exitCode})`,
383
+ );
384
+ }
385
+ }
386
+ if (!lines.length) return null;
387
+ return ["#### Errors", "", ...lines].join("\n");
388
+ }
389
+
390
+ // ---------------------------------------------------------------------------
391
+ // Formatting helpers
392
+ // ---------------------------------------------------------------------------
393
+
394
+ function statusIcon(pass) {
395
+ return pass ? "✅" : "❌";
396
+ }
397
+
398
+ function formatPassAt(v) {
399
+ if (v == null) return "—";
400
+ if (typeof v === "object" && "error" in v) return v.error;
401
+ return Number(v).toFixed(4);
402
+ }
403
+
404
+ function formatDuration(ms) {
405
+ if (ms == null || ms === 0) return "0s";
406
+ const totalSeconds = Math.round(ms / 1000);
407
+ if (totalSeconds < 60) return `${totalSeconds}s`;
408
+ const minutes = Math.floor(totalSeconds / 60);
409
+ const seconds = totalSeconds % 60;
410
+ return seconds > 0 ? `${minutes}m ${seconds}s` : `${minutes}m`;
411
+ }
412
+
413
+ function formatCost(usd) {
414
+ if (usd == null) return "$0.00";
415
+ return `$${usd.toFixed(2)}`;
416
+ }
417
+
418
+ function escapeCell(str) {
419
+ return str.replace(/\|/g, "\\|");
420
+ }
421
+
422
+ function median(arr) {
423
+ if (!arr.length) return 0;
424
+ const sorted = [...arr].sort((a, b) => a - b);
425
+ const mid = Math.floor(sorted.length / 2);
426
+ if (sorted.length % 2 === 0) {
427
+ return Math.round((sorted[mid - 1] + sorted[mid]) / 2);
428
+ }
429
+ return sorted[mid];
430
+ }
431
+
432
+ // ---------------------------------------------------------------------------
433
+ // Record loading
434
+ // ---------------------------------------------------------------------------
435
+
436
+ async function loadRecords(inputDir, runtime) {
437
+ const path = join(inputDir, "results.jsonl");
438
+ let content;
439
+ try {
440
+ content = await runtime.fs.readFile(path, "utf8");
441
+ } catch (e) {
442
+ // Re-throw with the stack collapsed to the message line so the CLI's
443
+ // error rendering stays free of node-internal async `readFile` frames
444
+ // (matching the pre-1370 stream-error shape the golden captured).
445
+ const err = new Error(e.message);
446
+ if (e.code) err.code = e.code;
447
+ err.stack = `Error: ${e.message}`;
448
+ throw err;
449
+ }
450
+ const records = [];
451
+ let skipped = 0;
452
+ for (const line of content.split("\n")) {
453
+ const trimmed = line.trim();
454
+ if (!trimmed) continue;
455
+ let record;
456
+ try {
457
+ record = JSON.parse(trimmed);
458
+ } catch (e) {
459
+ runtime.proc.stderr.write(
460
+ `benchmark report: skipped malformed JSON line — ${e.message}\n`,
461
+ );
462
+ skipped++;
463
+ continue;
464
+ }
465
+ try {
466
+ validateResultRecord(record);
467
+ } catch (e) {
468
+ runtime.proc.stderr.write(
469
+ `benchmark report: skipped record failing schema — ${describeError(e)}\n`,
470
+ );
471
+ skipped++;
472
+ continue;
473
+ }
474
+ records.push(record);
475
+ }
476
+ return { records, skipped };
477
+ }
478
+
479
+ function describeError(e) {
480
+ if (e && Array.isArray(e.issues)) {
481
+ return e.issues.map((i) => `${i.path.join(".")}: ${i.message}`).join("; ");
482
+ }
483
+ return e.message ?? String(e);
484
+ }
485
+
486
+ function groupByTask(records) {
487
+ const out = new Map();
488
+ for (const r of records) {
489
+ if (!out.has(r.taskId)) out.set(r.taskId, []);
490
+ out.get(r.taskId).push(r);
491
+ }
492
+ return out;
493
+ }
494
+
495
+ /**
496
+ * pass@k = 1 - C(n - c, k) / C(n, k). Compute with BigInt to avoid
497
+ * floating-point loss on large n.
498
+ * @param {number} n
499
+ * @param {number} c
500
+ * @param {number} k
501
+ * @returns {number | {error: string}}
502
+ */
503
+ function passAtKValue(n, c, k) {
504
+ if (k > n) return { error: "k > n" };
505
+ if (n - c < k) return 1;
506
+ const total = binomial(BigInt(n), BigInt(k));
507
+ const fail = binomial(BigInt(n - c), BigInt(k));
508
+ const passing = total - fail;
509
+ return Number(passing) / Number(total);
510
+ }
511
+
512
+ function binomial(n, k) {
513
+ if (k < 0n || k > n) return 0n;
514
+ if (k === 0n || k === n) return 1n;
515
+ let kk = k;
516
+ if (kk > n - kk) kk = n - kk;
517
+ let result = 1n;
518
+ for (let i = 0n; i < kk; i++) {
519
+ result = (result * (n - i)) / (i + 1n);
520
+ }
521
+ return result;
522
+ }
@@ -0,0 +1,127 @@
1
+ /**
2
+ * Result-record schemas and runtime validators.
3
+ *
4
+ * Two schemas live here:
5
+ * - RESULT_RECORD_SCHEMA — one record per (task, runIndex) from a full
6
+ * benchmark run. Has a happy branch (invariants + judge present) and a
7
+ * pre-flight-failure branch (invariants/judgeVerdict/submission absent).
8
+ * - INVARIANTS_RECORD_SCHEMA — narrower output of `benchmark-invariants`:
9
+ * ad-hoc grading without a full lifecycle.
10
+ *
11
+ * Validation is throw-on-mismatch so the runner can wrap every JSONL append
12
+ * in a guard and reject schema drift at write time.
13
+ */
14
+
15
+ import { z } from "zod";
16
+
17
+ const VERDICT_ENUM = z.enum(["pass", "fail"]);
18
+
19
+ const INVARIANTS_SHAPE = z.object({
20
+ verdict: VERDICT_ENUM,
21
+ details: z.array(z.unknown()),
22
+ exitCode: z.number().int(),
23
+ stderr: z.string().optional(),
24
+ });
25
+
26
+ const JUDGE_VERDICT_SHAPE = z.object({
27
+ verdict: VERDICT_ENUM,
28
+ summary: z.string(),
29
+ });
30
+
31
+ /**
32
+ * Per-participant cost attribution. `costUsd` is the sum of these; the
33
+ * breakdown lets reports show where the spend went. The judge runs as its
34
+ * own SDK session, so its cost is tracked separately from agent/supervisor.
35
+ */
36
+ const COST_BREAKDOWN_SHAPE = z.object({
37
+ agent: z.number(),
38
+ supervisor: z.number(),
39
+ judge: z.number(),
40
+ });
41
+
42
+ const PROFILES_SHAPE = z.object({
43
+ agent: z.union([z.string(), z.null()]),
44
+ supervisor: z.union([z.string(), z.null()]),
45
+ judge: z.union([z.string(), z.null()]),
46
+ });
47
+
48
+ const PREFLIGHT_ERROR_SHAPE = z.object({
49
+ phase: z.string(),
50
+ message: z.string(),
51
+ exitCode: z.number().int(),
52
+ });
53
+
54
+ const COMMON_FIELDS = {
55
+ taskId: z.string().min(1),
56
+ runIndex: z.number().int().min(0),
57
+ verdict: VERDICT_ENUM,
58
+ costUsd: z.number(),
59
+ costBreakdown: COST_BREAKDOWN_SHAPE.optional(),
60
+ turns: z.number().int().min(0),
61
+ profiles: PROFILES_SHAPE,
62
+ model: z.object({
63
+ agent: z.string(),
64
+ supervisor: z.string().optional(),
65
+ judge: z.string().optional(),
66
+ }),
67
+ skillSetHash: z.string(),
68
+ familyRevision: z.string(),
69
+ durationMs: z.number().int().min(0),
70
+ };
71
+
72
+ const AGENT_ERROR_SHAPE = z.object({
73
+ message: z.string(),
74
+ aborted: z.boolean(),
75
+ });
76
+
77
+ const HAPPY_RECORD = z.object({
78
+ ...COMMON_FIELDS,
79
+ invariants: INVARIANTS_SHAPE,
80
+ submission: z.string(),
81
+ judgeVerdict: JUDGE_VERDICT_SHAPE.optional(),
82
+ agentTracePath: z.string(),
83
+ supervisorTracePath: z.string(),
84
+ judgeTracePath: z.string(),
85
+ agentError: AGENT_ERROR_SHAPE.optional(),
86
+ preflightError: z.undefined().optional(),
87
+ });
88
+
89
+ const PREFLIGHT_RECORD = z.object({
90
+ ...COMMON_FIELDS,
91
+ costUsd: z.literal(0),
92
+ preflightError: PREFLIGHT_ERROR_SHAPE,
93
+ // Trace paths are populated even on preflight failure (the runner allocates
94
+ // them in WorkdirManager.start) so the record is uniform across branches
95
+ // and downstream consumers can reference them without conditional fields.
96
+ agentTracePath: z.string(),
97
+ supervisorTracePath: z.string(),
98
+ judgeTracePath: z.string(),
99
+ invariants: z.undefined().optional(),
100
+ submission: z.undefined().optional(),
101
+ judgeVerdict: z.undefined().optional(),
102
+ agentError: z.undefined().optional(),
103
+ });
104
+
105
+ export const RESULT_RECORD_SCHEMA = z.union([HAPPY_RECORD, PREFLIGHT_RECORD]);
106
+
107
+ export const INVARIANTS_RECORD_SCHEMA = z.object({
108
+ taskId: z.string().min(1),
109
+ invariants: INVARIANTS_SHAPE,
110
+ exitCode: z.number().int(),
111
+ });
112
+
113
+ /**
114
+ * Throw on schema mismatch.
115
+ * @param {object} record
116
+ */
117
+ export function validateResultRecord(record) {
118
+ RESULT_RECORD_SCHEMA.parse(record);
119
+ }
120
+
121
+ /**
122
+ * Throw on schema mismatch.
123
+ * @param {object} record
124
+ */
125
+ export function validateInvariantsRecord(record) {
126
+ INVARIANTS_RECORD_SCHEMA.parse(record);
127
+ }