@forwardimpact/libharness 0.1.22 → 1.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -201
- package/README.md +196 -80
- package/bin/fit-benchmark.js +44 -0
- package/bin/fit-harness.js +358 -0
- package/bin/fit-selfedit.js +165 -0
- package/bin/fit-trace.js +510 -0
- package/package.json +41 -11
- package/src/agent-runner.js +256 -0
- package/src/benchmark/apm-installer.js +207 -0
- package/src/benchmark/env-loader.js +158 -0
- package/src/benchmark/hook-env.js +40 -0
- package/src/benchmark/invariants.js +141 -0
- package/src/benchmark/judge.js +187 -0
- package/src/benchmark/npm-installer.js +87 -0
- package/src/benchmark/report.js +522 -0
- package/src/benchmark/result.js +127 -0
- package/src/benchmark/runner.js +583 -0
- package/src/benchmark/task-family.js +260 -0
- package/src/benchmark/workdir.js +298 -0
- package/src/commands/assert.js +153 -0
- package/src/commands/benchmark-definition.js +165 -0
- package/src/commands/benchmark-invariants.js +73 -0
- package/src/commands/benchmark-report.js +51 -0
- package/src/commands/benchmark-run.js +111 -0
- package/src/commands/by-discussion.js +94 -0
- package/src/commands/callback.js +119 -0
- package/src/commands/discuss.js +132 -0
- package/src/commands/facilitate.js +123 -0
- package/src/commands/output.js +36 -0
- package/src/commands/run.js +152 -0
- package/src/commands/supervise.js +136 -0
- package/src/commands/task-input.js +54 -0
- package/src/commands/tee.js +53 -0
- package/src/commands/trace.js +630 -0
- package/src/commands/work-tracker.js +35 -0
- package/src/cost.js +79 -0
- package/src/discuss-tools.js +173 -0
- package/src/discusser.js +394 -0
- package/src/events/github.js +161 -0
- package/src/facilitator.js +205 -0
- package/src/inbox-poller.js +81 -0
- package/src/index.js +72 -2
- package/src/judge.js +210 -0
- package/src/message-bus.js +118 -0
- package/src/orchestration-loop.js +330 -0
- package/src/orchestration-toolkit.js +441 -0
- package/src/orchestrator-helpers.js +23 -0
- package/src/profile-prompt.js +266 -0
- package/src/redaction.js +253 -0
- package/src/render/line-renderer.js +54 -0
- package/src/render/orchestrator-filter.js +19 -0
- package/src/render/palette.js +63 -0
- package/src/render/tool-hints.js +154 -0
- package/src/render/turn-renderer.js +96 -0
- package/src/reply-emitter.js +47 -0
- package/src/sequence-counter.js +21 -0
- package/src/signature-filter.js +27 -0
- package/src/supervisor.js +236 -0
- package/src/tee-writer.js +150 -0
- package/src/trace-collector.js +444 -0
- package/src/trace-github.js +473 -0
- package/src/trace-multi.js +101 -0
- package/src/trace-query.js +748 -0
- package/src/trace-render.js +211 -0
- package/src/trace-usage.js +249 -0
- package/src/fixture/assertions.js +0 -42
- package/src/fixture/cache.js +0 -50
- package/src/fixture/eval.js +0 -146
- package/src/fixture/index.js +0 -9
- package/src/fixture/pathway.js +0 -451
- package/src/fixture/services.js +0 -56
- package/src/mock/clients.js +0 -135
- package/src/mock/config.js +0 -45
- package/src/mock/data.js +0 -46
- package/src/mock/fs.js +0 -111
- package/src/mock/grpc.js +0 -94
- package/src/mock/http.js +0 -60
- package/src/mock/index.js +0 -36
- package/src/mock/infra.js +0 -219
- package/src/mock/logger.js +0 -42
- package/src/mock/observer.js +0 -74
- package/src/mock/resource-index.js +0 -95
- package/src/mock/service-callbacks.js +0 -39
- package/src/mock/services.js +0 -79
- package/src/mock/spy.js +0 -44
- package/src/mock/storage.js +0 -118
|
@@ -0,0 +1,522 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* ReportAggregator — read a run-output directory's `results.jsonl`, group
|
|
3
|
+
* records by `taskId`, and compute pass@k via the OpenAI HumanEval
|
|
4
|
+
* unbiased estimator: `1 - C(n-c, k) / C(n, k)`.
|
|
5
|
+
*
|
|
6
|
+
* When `includeRuns` is true, each task carries per-run detail (invariant
|
|
7
|
+
* checks, judge commentary, cost, duration) and the text renderer produces
|
|
8
|
+
* a full markdown report instead of just the pass@k table.
|
|
9
|
+
*
|
|
10
|
+
* Records that fail schema validation are skipped with a stderr warning
|
|
11
|
+
* (counted under `totals.skipped`) so a corrupt line cannot abort the
|
|
12
|
+
* whole report.
|
|
13
|
+
*/
|
|
14
|
+
|
|
15
|
+
import { join } from "node:path";
|
|
16
|
+
|
|
17
|
+
import { validateResultRecord } from "./result.js";
|
|
18
|
+
|
|
19
|
+
/**
|
|
20
|
+
* @typedef {object} RunDetail
|
|
21
|
+
* @property {number} runIndex
|
|
22
|
+
* @property {"pass"|"fail"} verdict
|
|
23
|
+
* @property {{verdict: string, details: unknown[], exitCode: number}} [invariants]
|
|
24
|
+
* @property {{verdict: string, summary: string}} [judgeVerdict]
|
|
25
|
+
* @property {number} costUsd
|
|
26
|
+
* @property {number} turns
|
|
27
|
+
* @property {number} durationMs
|
|
28
|
+
* @property {{message: string, aborted: boolean}} [agentError]
|
|
29
|
+
* @property {{phase: string, message: string, exitCode: number}} [preflightError]
|
|
30
|
+
*/
|
|
31
|
+
|
|
32
|
+
/**
|
|
33
|
+
* @typedef {object} TaskReport
|
|
34
|
+
* @property {string} taskId
|
|
35
|
+
* @property {number} n - Total runs.
|
|
36
|
+
* @property {number} c - Passing runs.
|
|
37
|
+
* @property {Record<string|number, number|null>} passAtK
|
|
38
|
+
* @property {RunDetail[]} [runs] - Per-run detail (only when includeRuns).
|
|
39
|
+
*/
|
|
40
|
+
|
|
41
|
+
/**
|
|
42
|
+
* @param {{inputDir: string, kValues: number[], includeRuns?: boolean, runtime: import("@forwardimpact/libutil/runtime").Runtime}} opts
|
|
43
|
+
* @returns {Promise<{tasks: TaskReport[], totals: object}>}
|
|
44
|
+
*/
|
|
45
|
+
export async function aggregate({
|
|
46
|
+
inputDir,
|
|
47
|
+
kValues,
|
|
48
|
+
includeRuns = false,
|
|
49
|
+
runtime,
|
|
50
|
+
}) {
|
|
51
|
+
if (!runtime) throw new Error("runtime is required");
|
|
52
|
+
const records = await loadRecords(inputDir, runtime);
|
|
53
|
+
const grouped = groupByTask(records.records);
|
|
54
|
+
const tasks = [];
|
|
55
|
+
let totalRuns = 0;
|
|
56
|
+
let totalCost = 0;
|
|
57
|
+
const allDurations = [];
|
|
58
|
+
const allTurns = [];
|
|
59
|
+
let firstRecord = null;
|
|
60
|
+
|
|
61
|
+
for (const [taskId, group] of grouped) {
|
|
62
|
+
const n = group.length;
|
|
63
|
+
const c = group.filter((r) => r.verdict === "pass").length;
|
|
64
|
+
totalRuns += n;
|
|
65
|
+
const passAtK = {};
|
|
66
|
+
for (const k of kValues) passAtK[k] = passAtKValue(n, c, k);
|
|
67
|
+
|
|
68
|
+
const task = { taskId, n, c, passAtK };
|
|
69
|
+
|
|
70
|
+
if (includeRuns) {
|
|
71
|
+
if (!firstRecord) firstRecord = group[0];
|
|
72
|
+
const accumulators = { allDurations, allTurns };
|
|
73
|
+
task.runs = group
|
|
74
|
+
.map((r) => {
|
|
75
|
+
totalCost += r.costUsd ?? 0;
|
|
76
|
+
return buildRunDetail(r, accumulators);
|
|
77
|
+
})
|
|
78
|
+
.sort((a, b) => a.runIndex - b.runIndex);
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
tasks.push(task);
|
|
82
|
+
}
|
|
83
|
+
tasks.sort((a, b) =>
|
|
84
|
+
a.taskId < b.taskId ? -1 : a.taskId > b.taskId ? 1 : 0,
|
|
85
|
+
);
|
|
86
|
+
|
|
87
|
+
const totals = {
|
|
88
|
+
tasks: tasks.length,
|
|
89
|
+
runs: totalRuns,
|
|
90
|
+
skipped: records.skipped,
|
|
91
|
+
};
|
|
92
|
+
|
|
93
|
+
if (includeRuns) {
|
|
94
|
+
totals.costUsd = totalCost;
|
|
95
|
+
totals.medianDurationMs = median(allDurations);
|
|
96
|
+
totals.medianTurns = median(allTurns);
|
|
97
|
+
totals.model = firstRecord?.model ?? "";
|
|
98
|
+
totals.skillSetHash = firstRecord?.skillSetHash ?? "";
|
|
99
|
+
totals.familyRevision = firstRecord?.familyRevision ?? "";
|
|
100
|
+
}
|
|
101
|
+
|
|
102
|
+
return { tasks, totals };
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
/**
|
|
106
|
+
* Build a normalized per-run detail object and accumulate duration/turn
|
|
107
|
+
* samples for median calculation. Extracted from `aggregate` to keep its
|
|
108
|
+
* cognitive complexity below the lint ceiling.
|
|
109
|
+
* @param {object} r - Raw record.
|
|
110
|
+
* @param {{allDurations: number[], allTurns: number[]}} acc
|
|
111
|
+
* @returns {RunDetail}
|
|
112
|
+
*/
|
|
113
|
+
function buildRunDetail(r, acc) {
|
|
114
|
+
if (r.durationMs != null) acc.allDurations.push(r.durationMs);
|
|
115
|
+
if (r.turns != null) acc.allTurns.push(r.turns);
|
|
116
|
+
return {
|
|
117
|
+
runIndex: r.runIndex,
|
|
118
|
+
verdict: r.verdict,
|
|
119
|
+
...(r.invariants && { invariants: r.invariants }),
|
|
120
|
+
...(r.judgeVerdict && { judgeVerdict: r.judgeVerdict }),
|
|
121
|
+
costUsd: r.costUsd ?? 0,
|
|
122
|
+
turns: r.turns ?? 0,
|
|
123
|
+
durationMs: r.durationMs ?? 0,
|
|
124
|
+
...(r.agentError && { agentError: r.agentError }),
|
|
125
|
+
...(r.preflightError && { preflightError: r.preflightError }),
|
|
126
|
+
};
|
|
127
|
+
}
|
|
128
|
+
|
|
129
|
+
/**
|
|
130
|
+
* Render an aggregate report as markdown. When the report contains per-run
|
|
131
|
+
* detail (from `includeRuns: true`), renders a full report with summary,
|
|
132
|
+
* pass@k table, and per-task detail sections. Otherwise falls back to the
|
|
133
|
+
* compact pass@k table.
|
|
134
|
+
* @param {Awaited<ReturnType<typeof aggregate>>} report
|
|
135
|
+
* @param {number[]} kValues
|
|
136
|
+
* @returns {string}
|
|
137
|
+
*/
|
|
138
|
+
export function renderTextReport(report, kValues) {
|
|
139
|
+
if (report.tasks[0]?.runs) {
|
|
140
|
+
return renderFullReport(report, kValues);
|
|
141
|
+
}
|
|
142
|
+
return renderCompactReport(report, kValues);
|
|
143
|
+
}
|
|
144
|
+
|
|
145
|
+
// ---------------------------------------------------------------------------
|
|
146
|
+
// Compact report (legacy path)
|
|
147
|
+
// ---------------------------------------------------------------------------
|
|
148
|
+
|
|
149
|
+
function renderCompactReport(report, kValues) {
|
|
150
|
+
const lines = [
|
|
151
|
+
renderPassAtKTable(report, kValues),
|
|
152
|
+
"",
|
|
153
|
+
renderTotalsLine(report),
|
|
154
|
+
];
|
|
155
|
+
return lines.join("\n");
|
|
156
|
+
}
|
|
157
|
+
|
|
158
|
+
// ---------------------------------------------------------------------------
|
|
159
|
+
// Full report
|
|
160
|
+
// ---------------------------------------------------------------------------
|
|
161
|
+
|
|
162
|
+
function renderFullReport(report, kValues) {
|
|
163
|
+
const sections = [
|
|
164
|
+
renderSummary(report),
|
|
165
|
+
"## Pass@k",
|
|
166
|
+
"",
|
|
167
|
+
renderPassAtKTable(report, kValues),
|
|
168
|
+
"",
|
|
169
|
+
renderTotalsLine(report),
|
|
170
|
+
"",
|
|
171
|
+
"## Task Details",
|
|
172
|
+
];
|
|
173
|
+
|
|
174
|
+
for (const task of report.tasks) {
|
|
175
|
+
sections.push("");
|
|
176
|
+
sections.push(renderTaskDetail(task));
|
|
177
|
+
}
|
|
178
|
+
|
|
179
|
+
return sections.join("\n");
|
|
180
|
+
}
|
|
181
|
+
|
|
182
|
+
function renderSummary(report) {
|
|
183
|
+
const { totals } = report;
|
|
184
|
+
const passing = report.tasks.filter((t) => t.c > 0 && t.c === t.n).length;
|
|
185
|
+
const icon = statusIcon(passing === totals.tasks);
|
|
186
|
+
const lines = [
|
|
187
|
+
"# Benchmark Report",
|
|
188
|
+
"",
|
|
189
|
+
`${icon} **${passing}/${totals.tasks} tasks passing** | ${totals.runs} runs${totals.skipped ? ` | ${totals.skipped} skipped` : ""}`,
|
|
190
|
+
];
|
|
191
|
+
|
|
192
|
+
const headers = [];
|
|
193
|
+
const values = [];
|
|
194
|
+
if (totals.costUsd != null) {
|
|
195
|
+
headers.push("Cost");
|
|
196
|
+
values.push(formatCost(totals.costUsd));
|
|
197
|
+
}
|
|
198
|
+
if (totals.medianDurationMs != null) {
|
|
199
|
+
headers.push("Median Duration");
|
|
200
|
+
values.push(formatDuration(totals.medianDurationMs));
|
|
201
|
+
}
|
|
202
|
+
if (totals.medianTurns != null) {
|
|
203
|
+
headers.push("Median Turns");
|
|
204
|
+
values.push(String(totals.medianTurns));
|
|
205
|
+
}
|
|
206
|
+
if (headers.length) {
|
|
207
|
+
lines.push("");
|
|
208
|
+
lines.push(`| ${headers.join(" | ")} |`);
|
|
209
|
+
lines.push(`| ${headers.map(() => "---").join(" | ")} |`);
|
|
210
|
+
lines.push(`| ${values.join(" | ")} |`);
|
|
211
|
+
}
|
|
212
|
+
|
|
213
|
+
const meta = [];
|
|
214
|
+
if (totals.model) {
|
|
215
|
+
meta.push(`Agent: \`${totals.model.agent}\``);
|
|
216
|
+
meta.push(`Supervisor: \`${totals.model.supervisor}\``);
|
|
217
|
+
meta.push(`Judge: \`${totals.model.judge}\``);
|
|
218
|
+
}
|
|
219
|
+
if (totals.skillSetHash) meta.push(`Skill set: \`${totals.skillSetHash}\``);
|
|
220
|
+
if (totals.familyRevision) meta.push(`Family: \`${totals.familyRevision}\``);
|
|
221
|
+
if (meta.length) {
|
|
222
|
+
lines.push("");
|
|
223
|
+
lines.push(meta.join(" | "));
|
|
224
|
+
}
|
|
225
|
+
|
|
226
|
+
lines.push("");
|
|
227
|
+
return lines.join("\n");
|
|
228
|
+
}
|
|
229
|
+
|
|
230
|
+
// ---------------------------------------------------------------------------
|
|
231
|
+
// Pass@k table (shared between compact and full)
|
|
232
|
+
// ---------------------------------------------------------------------------
|
|
233
|
+
|
|
234
|
+
function renderPassAtKTable(report, kValues) {
|
|
235
|
+
const header = ["taskId", "n", "c", ...kValues.map((k) => `pass@${k}`)];
|
|
236
|
+
const rows = [header, header.map(() => "---")];
|
|
237
|
+
for (const t of report.tasks) {
|
|
238
|
+
rows.push([
|
|
239
|
+
t.taskId,
|
|
240
|
+
String(t.n),
|
|
241
|
+
String(t.c),
|
|
242
|
+
...kValues.map((k) => formatPassAt(t.passAtK[k])),
|
|
243
|
+
]);
|
|
244
|
+
}
|
|
245
|
+
return rows.map((r) => `| ${r.join(" | ")} |`).join("\n");
|
|
246
|
+
}
|
|
247
|
+
|
|
248
|
+
function renderTotalsLine(report) {
|
|
249
|
+
return `Totals — tasks: ${report.totals.tasks}, runs: ${report.totals.runs}, skipped: ${report.totals.skipped}`;
|
|
250
|
+
}
|
|
251
|
+
|
|
252
|
+
// ---------------------------------------------------------------------------
|
|
253
|
+
// Per-task detail
|
|
254
|
+
// ---------------------------------------------------------------------------
|
|
255
|
+
|
|
256
|
+
function renderTaskDetail(task) {
|
|
257
|
+
const runs = task.runs ?? [];
|
|
258
|
+
const icon = statusIcon(task.c === task.n);
|
|
259
|
+
const singleRun = runs.length === 1;
|
|
260
|
+
|
|
261
|
+
const lines = [
|
|
262
|
+
`### ${task.taskId}`,
|
|
263
|
+
"",
|
|
264
|
+
`${icon} **${task.c}/${task.n} runs passed**`,
|
|
265
|
+
];
|
|
266
|
+
|
|
267
|
+
lines.push("", renderRunsTable(runs));
|
|
268
|
+
|
|
269
|
+
const checks = renderInvariantChecks(runs, singleRun);
|
|
270
|
+
if (checks) lines.push("", checks);
|
|
271
|
+
|
|
272
|
+
const commentary = renderJudgeCommentary(runs, singleRun);
|
|
273
|
+
if (commentary) lines.push("", commentary);
|
|
274
|
+
|
|
275
|
+
const errors = renderErrors(runs);
|
|
276
|
+
if (errors) lines.push("", errors);
|
|
277
|
+
|
|
278
|
+
return lines.join("\n");
|
|
279
|
+
}
|
|
280
|
+
|
|
281
|
+
function renderRunsTable(runs) {
|
|
282
|
+
const header = [
|
|
283
|
+
"Run",
|
|
284
|
+
"Verdict",
|
|
285
|
+
"Invariants",
|
|
286
|
+
"Judge",
|
|
287
|
+
"Cost",
|
|
288
|
+
"Turns",
|
|
289
|
+
"Duration",
|
|
290
|
+
];
|
|
291
|
+
const rows = [header, header.map(() => "---")];
|
|
292
|
+
for (const r of runs) {
|
|
293
|
+
const invariantsCell = r.preflightError
|
|
294
|
+
? "preflight error"
|
|
295
|
+
: r.invariants
|
|
296
|
+
? statusIcon(r.invariants.verdict === "pass")
|
|
297
|
+
: "—";
|
|
298
|
+
const judgeCell = r.preflightError
|
|
299
|
+
? "—"
|
|
300
|
+
: r.judgeVerdict
|
|
301
|
+
? statusIcon(r.judgeVerdict.verdict === "pass")
|
|
302
|
+
: "—";
|
|
303
|
+
rows.push([
|
|
304
|
+
String(r.runIndex),
|
|
305
|
+
statusIcon(r.verdict === "pass"),
|
|
306
|
+
invariantsCell,
|
|
307
|
+
judgeCell,
|
|
308
|
+
formatCost(r.costUsd),
|
|
309
|
+
String(r.turns),
|
|
310
|
+
formatDuration(r.durationMs),
|
|
311
|
+
]);
|
|
312
|
+
}
|
|
313
|
+
return rows.map((r) => `| ${r.join(" | ")} |`).join("\n");
|
|
314
|
+
}
|
|
315
|
+
|
|
316
|
+
function renderInvariantChecks(runs, singleRun) {
|
|
317
|
+
const rows = collectInvariantRows(runs);
|
|
318
|
+
if (!rows.length) return null;
|
|
319
|
+
|
|
320
|
+
const header = singleRun
|
|
321
|
+
? ["Check", "Result", "Message"]
|
|
322
|
+
: ["Run", "Check", "Result", "Message"];
|
|
323
|
+
const lines = [
|
|
324
|
+
"#### Invariant Checks",
|
|
325
|
+
"",
|
|
326
|
+
`| ${header.join(" | ")} |`,
|
|
327
|
+
`| ${header.map(() => "---").join(" | ")} |`,
|
|
328
|
+
];
|
|
329
|
+
for (const row of rows) {
|
|
330
|
+
const cells = singleRun
|
|
331
|
+
? [row.check, row.result, row.message]
|
|
332
|
+
: [String(row.run), row.check, row.result, row.message];
|
|
333
|
+
lines.push(`| ${cells.join(" | ")} |`);
|
|
334
|
+
}
|
|
335
|
+
return lines.join("\n");
|
|
336
|
+
}
|
|
337
|
+
|
|
338
|
+
function collectInvariantRows(runs) {
|
|
339
|
+
const rows = [];
|
|
340
|
+
for (const r of runs) {
|
|
341
|
+
if (!r.invariants?.details?.length) continue;
|
|
342
|
+
for (const d of r.invariants.details) {
|
|
343
|
+
rows.push({
|
|
344
|
+
run: r.runIndex,
|
|
345
|
+
check: escapeCell(String(d.test ?? "(unnamed)")),
|
|
346
|
+
result: statusIcon(d.pass),
|
|
347
|
+
message: escapeCell(String(d.message ?? "")),
|
|
348
|
+
});
|
|
349
|
+
}
|
|
350
|
+
}
|
|
351
|
+
return rows;
|
|
352
|
+
}
|
|
353
|
+
|
|
354
|
+
function renderJudgeCommentary(runs, singleRun) {
|
|
355
|
+
const entries = runs.filter((r) => r.judgeVerdict?.summary);
|
|
356
|
+
if (!entries.length) return null;
|
|
357
|
+
|
|
358
|
+
const lines = ["#### Judge Commentary", ""];
|
|
359
|
+
for (let i = 0; i < entries.length; i++) {
|
|
360
|
+
const r = entries[i];
|
|
361
|
+
const summary = r.judgeVerdict.summary.replace(/\n/g, "\n> ");
|
|
362
|
+
if (singleRun) {
|
|
363
|
+
lines.push(`> ${summary}`);
|
|
364
|
+
} else {
|
|
365
|
+
lines.push(`> **Run ${r.runIndex}:** ${summary}`);
|
|
366
|
+
}
|
|
367
|
+
if (i < entries.length - 1) lines.push(">");
|
|
368
|
+
}
|
|
369
|
+
return lines.join("\n");
|
|
370
|
+
}
|
|
371
|
+
|
|
372
|
+
function renderErrors(runs) {
|
|
373
|
+
const lines = [];
|
|
374
|
+
for (const r of runs) {
|
|
375
|
+
if (r.agentError) {
|
|
376
|
+
lines.push(
|
|
377
|
+
`- **Run ${r.runIndex}:** Agent error — "${escapeCell(r.agentError.message)}" (aborted: ${r.agentError.aborted})`,
|
|
378
|
+
);
|
|
379
|
+
}
|
|
380
|
+
if (r.preflightError) {
|
|
381
|
+
lines.push(
|
|
382
|
+
`- **Run ${r.runIndex}:** Preflight error — "${escapeCell(r.preflightError.message)}" (exit ${r.preflightError.exitCode})`,
|
|
383
|
+
);
|
|
384
|
+
}
|
|
385
|
+
}
|
|
386
|
+
if (!lines.length) return null;
|
|
387
|
+
return ["#### Errors", "", ...lines].join("\n");
|
|
388
|
+
}
|
|
389
|
+
|
|
390
|
+
// ---------------------------------------------------------------------------
|
|
391
|
+
// Formatting helpers
|
|
392
|
+
// ---------------------------------------------------------------------------
|
|
393
|
+
|
|
394
|
+
function statusIcon(pass) {
|
|
395
|
+
return pass ? "✅" : "❌";
|
|
396
|
+
}
|
|
397
|
+
|
|
398
|
+
function formatPassAt(v) {
|
|
399
|
+
if (v == null) return "—";
|
|
400
|
+
if (typeof v === "object" && "error" in v) return v.error;
|
|
401
|
+
return Number(v).toFixed(4);
|
|
402
|
+
}
|
|
403
|
+
|
|
404
|
+
function formatDuration(ms) {
|
|
405
|
+
if (ms == null || ms === 0) return "0s";
|
|
406
|
+
const totalSeconds = Math.round(ms / 1000);
|
|
407
|
+
if (totalSeconds < 60) return `${totalSeconds}s`;
|
|
408
|
+
const minutes = Math.floor(totalSeconds / 60);
|
|
409
|
+
const seconds = totalSeconds % 60;
|
|
410
|
+
return seconds > 0 ? `${minutes}m ${seconds}s` : `${minutes}m`;
|
|
411
|
+
}
|
|
412
|
+
|
|
413
|
+
function formatCost(usd) {
|
|
414
|
+
if (usd == null) return "$0.00";
|
|
415
|
+
return `$${usd.toFixed(2)}`;
|
|
416
|
+
}
|
|
417
|
+
|
|
418
|
+
function escapeCell(str) {
|
|
419
|
+
return str.replace(/\|/g, "\\|");
|
|
420
|
+
}
|
|
421
|
+
|
|
422
|
+
function median(arr) {
|
|
423
|
+
if (!arr.length) return 0;
|
|
424
|
+
const sorted = [...arr].sort((a, b) => a - b);
|
|
425
|
+
const mid = Math.floor(sorted.length / 2);
|
|
426
|
+
if (sorted.length % 2 === 0) {
|
|
427
|
+
return Math.round((sorted[mid - 1] + sorted[mid]) / 2);
|
|
428
|
+
}
|
|
429
|
+
return sorted[mid];
|
|
430
|
+
}
|
|
431
|
+
|
|
432
|
+
// ---------------------------------------------------------------------------
|
|
433
|
+
// Record loading
|
|
434
|
+
// ---------------------------------------------------------------------------
|
|
435
|
+
|
|
436
|
+
async function loadRecords(inputDir, runtime) {
|
|
437
|
+
const path = join(inputDir, "results.jsonl");
|
|
438
|
+
let content;
|
|
439
|
+
try {
|
|
440
|
+
content = await runtime.fs.readFile(path, "utf8");
|
|
441
|
+
} catch (e) {
|
|
442
|
+
// Re-throw with the stack collapsed to the message line so the CLI's
|
|
443
|
+
// error rendering stays free of node-internal async `readFile` frames
|
|
444
|
+
// (matching the pre-1370 stream-error shape the golden captured).
|
|
445
|
+
const err = new Error(e.message);
|
|
446
|
+
if (e.code) err.code = e.code;
|
|
447
|
+
err.stack = `Error: ${e.message}`;
|
|
448
|
+
throw err;
|
|
449
|
+
}
|
|
450
|
+
const records = [];
|
|
451
|
+
let skipped = 0;
|
|
452
|
+
for (const line of content.split("\n")) {
|
|
453
|
+
const trimmed = line.trim();
|
|
454
|
+
if (!trimmed) continue;
|
|
455
|
+
let record;
|
|
456
|
+
try {
|
|
457
|
+
record = JSON.parse(trimmed);
|
|
458
|
+
} catch (e) {
|
|
459
|
+
runtime.proc.stderr.write(
|
|
460
|
+
`benchmark report: skipped malformed JSON line — ${e.message}\n`,
|
|
461
|
+
);
|
|
462
|
+
skipped++;
|
|
463
|
+
continue;
|
|
464
|
+
}
|
|
465
|
+
try {
|
|
466
|
+
validateResultRecord(record);
|
|
467
|
+
} catch (e) {
|
|
468
|
+
runtime.proc.stderr.write(
|
|
469
|
+
`benchmark report: skipped record failing schema — ${describeError(e)}\n`,
|
|
470
|
+
);
|
|
471
|
+
skipped++;
|
|
472
|
+
continue;
|
|
473
|
+
}
|
|
474
|
+
records.push(record);
|
|
475
|
+
}
|
|
476
|
+
return { records, skipped };
|
|
477
|
+
}
|
|
478
|
+
|
|
479
|
+
function describeError(e) {
|
|
480
|
+
if (e && Array.isArray(e.issues)) {
|
|
481
|
+
return e.issues.map((i) => `${i.path.join(".")}: ${i.message}`).join("; ");
|
|
482
|
+
}
|
|
483
|
+
return e.message ?? String(e);
|
|
484
|
+
}
|
|
485
|
+
|
|
486
|
+
function groupByTask(records) {
|
|
487
|
+
const out = new Map();
|
|
488
|
+
for (const r of records) {
|
|
489
|
+
if (!out.has(r.taskId)) out.set(r.taskId, []);
|
|
490
|
+
out.get(r.taskId).push(r);
|
|
491
|
+
}
|
|
492
|
+
return out;
|
|
493
|
+
}
|
|
494
|
+
|
|
495
|
+
/**
|
|
496
|
+
* pass@k = 1 - C(n - c, k) / C(n, k). Compute with BigInt to avoid
|
|
497
|
+
* floating-point loss on large n.
|
|
498
|
+
* @param {number} n
|
|
499
|
+
* @param {number} c
|
|
500
|
+
* @param {number} k
|
|
501
|
+
* @returns {number | {error: string}}
|
|
502
|
+
*/
|
|
503
|
+
function passAtKValue(n, c, k) {
|
|
504
|
+
if (k > n) return { error: "k > n" };
|
|
505
|
+
if (n - c < k) return 1;
|
|
506
|
+
const total = binomial(BigInt(n), BigInt(k));
|
|
507
|
+
const fail = binomial(BigInt(n - c), BigInt(k));
|
|
508
|
+
const passing = total - fail;
|
|
509
|
+
return Number(passing) / Number(total);
|
|
510
|
+
}
|
|
511
|
+
|
|
512
|
+
function binomial(n, k) {
|
|
513
|
+
if (k < 0n || k > n) return 0n;
|
|
514
|
+
if (k === 0n || k === n) return 1n;
|
|
515
|
+
let kk = k;
|
|
516
|
+
if (kk > n - kk) kk = n - kk;
|
|
517
|
+
let result = 1n;
|
|
518
|
+
for (let i = 0n; i < kk; i++) {
|
|
519
|
+
result = (result * (n - i)) / (i + 1n);
|
|
520
|
+
}
|
|
521
|
+
return result;
|
|
522
|
+
}
|
|
@@ -0,0 +1,127 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Result-record schemas and runtime validators.
|
|
3
|
+
*
|
|
4
|
+
* Two schemas live here:
|
|
5
|
+
* - RESULT_RECORD_SCHEMA — one record per (task, runIndex) from a full
|
|
6
|
+
* benchmark run. Has a happy branch (invariants + judge present) and a
|
|
7
|
+
* pre-flight-failure branch (invariants/judgeVerdict/submission absent).
|
|
8
|
+
* - INVARIANTS_RECORD_SCHEMA — narrower output of `benchmark-invariants`:
|
|
9
|
+
* ad-hoc grading without a full lifecycle.
|
|
10
|
+
*
|
|
11
|
+
* Validation is throw-on-mismatch so the runner can wrap every JSONL append
|
|
12
|
+
* in a guard and reject schema drift at write time.
|
|
13
|
+
*/
|
|
14
|
+
|
|
15
|
+
import { z } from "zod";
|
|
16
|
+
|
|
17
|
+
const VERDICT_ENUM = z.enum(["pass", "fail"]);
|
|
18
|
+
|
|
19
|
+
const INVARIANTS_SHAPE = z.object({
|
|
20
|
+
verdict: VERDICT_ENUM,
|
|
21
|
+
details: z.array(z.unknown()),
|
|
22
|
+
exitCode: z.number().int(),
|
|
23
|
+
stderr: z.string().optional(),
|
|
24
|
+
});
|
|
25
|
+
|
|
26
|
+
const JUDGE_VERDICT_SHAPE = z.object({
|
|
27
|
+
verdict: VERDICT_ENUM,
|
|
28
|
+
summary: z.string(),
|
|
29
|
+
});
|
|
30
|
+
|
|
31
|
+
/**
|
|
32
|
+
* Per-participant cost attribution. `costUsd` is the sum of these; the
|
|
33
|
+
* breakdown lets reports show where the spend went. The judge runs as its
|
|
34
|
+
* own SDK session, so its cost is tracked separately from agent/supervisor.
|
|
35
|
+
*/
|
|
36
|
+
const COST_BREAKDOWN_SHAPE = z.object({
|
|
37
|
+
agent: z.number(),
|
|
38
|
+
supervisor: z.number(),
|
|
39
|
+
judge: z.number(),
|
|
40
|
+
});
|
|
41
|
+
|
|
42
|
+
const PROFILES_SHAPE = z.object({
|
|
43
|
+
agent: z.union([z.string(), z.null()]),
|
|
44
|
+
supervisor: z.union([z.string(), z.null()]),
|
|
45
|
+
judge: z.union([z.string(), z.null()]),
|
|
46
|
+
});
|
|
47
|
+
|
|
48
|
+
const PREFLIGHT_ERROR_SHAPE = z.object({
|
|
49
|
+
phase: z.string(),
|
|
50
|
+
message: z.string(),
|
|
51
|
+
exitCode: z.number().int(),
|
|
52
|
+
});
|
|
53
|
+
|
|
54
|
+
const COMMON_FIELDS = {
|
|
55
|
+
taskId: z.string().min(1),
|
|
56
|
+
runIndex: z.number().int().min(0),
|
|
57
|
+
verdict: VERDICT_ENUM,
|
|
58
|
+
costUsd: z.number(),
|
|
59
|
+
costBreakdown: COST_BREAKDOWN_SHAPE.optional(),
|
|
60
|
+
turns: z.number().int().min(0),
|
|
61
|
+
profiles: PROFILES_SHAPE,
|
|
62
|
+
model: z.object({
|
|
63
|
+
agent: z.string(),
|
|
64
|
+
supervisor: z.string().optional(),
|
|
65
|
+
judge: z.string().optional(),
|
|
66
|
+
}),
|
|
67
|
+
skillSetHash: z.string(),
|
|
68
|
+
familyRevision: z.string(),
|
|
69
|
+
durationMs: z.number().int().min(0),
|
|
70
|
+
};
|
|
71
|
+
|
|
72
|
+
const AGENT_ERROR_SHAPE = z.object({
|
|
73
|
+
message: z.string(),
|
|
74
|
+
aborted: z.boolean(),
|
|
75
|
+
});
|
|
76
|
+
|
|
77
|
+
const HAPPY_RECORD = z.object({
|
|
78
|
+
...COMMON_FIELDS,
|
|
79
|
+
invariants: INVARIANTS_SHAPE,
|
|
80
|
+
submission: z.string(),
|
|
81
|
+
judgeVerdict: JUDGE_VERDICT_SHAPE.optional(),
|
|
82
|
+
agentTracePath: z.string(),
|
|
83
|
+
supervisorTracePath: z.string(),
|
|
84
|
+
judgeTracePath: z.string(),
|
|
85
|
+
agentError: AGENT_ERROR_SHAPE.optional(),
|
|
86
|
+
preflightError: z.undefined().optional(),
|
|
87
|
+
});
|
|
88
|
+
|
|
89
|
+
const PREFLIGHT_RECORD = z.object({
|
|
90
|
+
...COMMON_FIELDS,
|
|
91
|
+
costUsd: z.literal(0),
|
|
92
|
+
preflightError: PREFLIGHT_ERROR_SHAPE,
|
|
93
|
+
// Trace paths are populated even on preflight failure (the runner allocates
|
|
94
|
+
// them in WorkdirManager.start) so the record is uniform across branches
|
|
95
|
+
// and downstream consumers can reference them without conditional fields.
|
|
96
|
+
agentTracePath: z.string(),
|
|
97
|
+
supervisorTracePath: z.string(),
|
|
98
|
+
judgeTracePath: z.string(),
|
|
99
|
+
invariants: z.undefined().optional(),
|
|
100
|
+
submission: z.undefined().optional(),
|
|
101
|
+
judgeVerdict: z.undefined().optional(),
|
|
102
|
+
agentError: z.undefined().optional(),
|
|
103
|
+
});
|
|
104
|
+
|
|
105
|
+
export const RESULT_RECORD_SCHEMA = z.union([HAPPY_RECORD, PREFLIGHT_RECORD]);
|
|
106
|
+
|
|
107
|
+
export const INVARIANTS_RECORD_SCHEMA = z.object({
|
|
108
|
+
taskId: z.string().min(1),
|
|
109
|
+
invariants: INVARIANTS_SHAPE,
|
|
110
|
+
exitCode: z.number().int(),
|
|
111
|
+
});
|
|
112
|
+
|
|
113
|
+
/**
|
|
114
|
+
* Throw on schema mismatch.
|
|
115
|
+
* @param {object} record
|
|
116
|
+
*/
|
|
117
|
+
export function validateResultRecord(record) {
|
|
118
|
+
RESULT_RECORD_SCHEMA.parse(record);
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
/**
|
|
122
|
+
* Throw on schema mismatch.
|
|
123
|
+
* @param {object} record
|
|
124
|
+
*/
|
|
125
|
+
export function validateInvariantsRecord(record) {
|
|
126
|
+
INVARIANTS_RECORD_SCHEMA.parse(record);
|
|
127
|
+
}
|