@forwardimpact/libharness 0.1.22 → 1.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (87) hide show
  1. package/LICENSE +21 -201
  2. package/README.md +196 -80
  3. package/bin/fit-benchmark.js +44 -0
  4. package/bin/fit-harness.js +358 -0
  5. package/bin/fit-selfedit.js +165 -0
  6. package/bin/fit-trace.js +510 -0
  7. package/package.json +41 -11
  8. package/src/agent-runner.js +256 -0
  9. package/src/benchmark/apm-installer.js +207 -0
  10. package/src/benchmark/env-loader.js +158 -0
  11. package/src/benchmark/hook-env.js +40 -0
  12. package/src/benchmark/invariants.js +141 -0
  13. package/src/benchmark/judge.js +187 -0
  14. package/src/benchmark/npm-installer.js +87 -0
  15. package/src/benchmark/report.js +604 -0
  16. package/src/benchmark/result.js +127 -0
  17. package/src/benchmark/runner.js +688 -0
  18. package/src/benchmark/scheduler.js +78 -0
  19. package/src/benchmark/task-family.js +260 -0
  20. package/src/benchmark/workdir.js +344 -0
  21. package/src/commands/assert.js +153 -0
  22. package/src/commands/benchmark-definition.js +175 -0
  23. package/src/commands/benchmark-invariants.js +73 -0
  24. package/src/commands/benchmark-report.js +51 -0
  25. package/src/commands/benchmark-run.js +175 -0
  26. package/src/commands/by-discussion.js +94 -0
  27. package/src/commands/callback.js +119 -0
  28. package/src/commands/discuss.js +132 -0
  29. package/src/commands/facilitate.js +123 -0
  30. package/src/commands/output.js +36 -0
  31. package/src/commands/run.js +152 -0
  32. package/src/commands/supervise.js +136 -0
  33. package/src/commands/task-input.js +54 -0
  34. package/src/commands/tee.js +53 -0
  35. package/src/commands/trace.js +630 -0
  36. package/src/commands/work-tracker.js +35 -0
  37. package/src/cost.js +79 -0
  38. package/src/discuss-tools.js +173 -0
  39. package/src/discusser.js +394 -0
  40. package/src/events/github.js +161 -0
  41. package/src/facilitator.js +205 -0
  42. package/src/inbox-poller.js +81 -0
  43. package/src/index.js +72 -2
  44. package/src/judge.js +210 -0
  45. package/src/message-bus.js +118 -0
  46. package/src/orchestration-loop.js +330 -0
  47. package/src/orchestration-toolkit.js +441 -0
  48. package/src/orchestrator-helpers.js +23 -0
  49. package/src/profile-prompt.js +266 -0
  50. package/src/redaction.js +253 -0
  51. package/src/render/line-renderer.js +54 -0
  52. package/src/render/orchestrator-filter.js +19 -0
  53. package/src/render/palette.js +63 -0
  54. package/src/render/tool-hints.js +154 -0
  55. package/src/render/turn-renderer.js +96 -0
  56. package/src/reply-emitter.js +47 -0
  57. package/src/sequence-counter.js +21 -0
  58. package/src/signature-filter.js +27 -0
  59. package/src/supervisor.js +236 -0
  60. package/src/tee-writer.js +150 -0
  61. package/src/trace-collector.js +444 -0
  62. package/src/trace-github.js +473 -0
  63. package/src/trace-multi.js +101 -0
  64. package/src/trace-query.js +748 -0
  65. package/src/trace-render.js +211 -0
  66. package/src/trace-usage.js +249 -0
  67. package/src/fixture/assertions.js +0 -42
  68. package/src/fixture/cache.js +0 -50
  69. package/src/fixture/eval.js +0 -146
  70. package/src/fixture/index.js +0 -9
  71. package/src/fixture/pathway.js +0 -451
  72. package/src/fixture/services.js +0 -56
  73. package/src/mock/clients.js +0 -135
  74. package/src/mock/config.js +0 -45
  75. package/src/mock/data.js +0 -46
  76. package/src/mock/fs.js +0 -111
  77. package/src/mock/grpc.js +0 -94
  78. package/src/mock/http.js +0 -60
  79. package/src/mock/index.js +0 -36
  80. package/src/mock/infra.js +0 -219
  81. package/src/mock/logger.js +0 -42
  82. package/src/mock/observer.js +0 -74
  83. package/src/mock/resource-index.js +0 -95
  84. package/src/mock/service-callbacks.js +0 -39
  85. package/src/mock/services.js +0 -79
  86. package/src/mock/spy.js +0 -44
  87. package/src/mock/storage.js +0 -118
@@ -0,0 +1,604 @@
1
+ /**
2
+ * ReportAggregator — read a run-output directory's `results.jsonl`, group
3
+ * records by `taskId`, and compute pass@k via the OpenAI HumanEval
4
+ * unbiased estimator: `1 - C(n-c, k) / C(n, k)`.
5
+ *
6
+ * When `includeRuns` is true, each task carries per-run detail (invariant
7
+ * checks, judge commentary, cost, duration) and the text renderer produces
8
+ * a full markdown report instead of just the pass@k table.
9
+ *
10
+ * Records that fail schema validation are skipped with a stderr warning
11
+ * (counted under `totals.skipped`) so a corrupt line cannot abort the
12
+ * whole report.
13
+ */
14
+
15
+ import { join } from "node:path";
16
+
17
+ import { validateResultRecord } from "./result.js";
18
+
19
+ /**
20
+ * @typedef {object} RunDetail
21
+ * @property {number} runIndex
22
+ * @property {"pass"|"fail"} verdict
23
+ * @property {{verdict: string, details: unknown[], exitCode: number}} [invariants]
24
+ * @property {{verdict: string, summary: string}} [judgeVerdict]
25
+ * @property {number} costUsd
26
+ * @property {number} turns
27
+ * @property {number} durationMs
28
+ * @property {{message: string, aborted: boolean}} [agentError]
29
+ * @property {{phase: string, message: string, exitCode: number}} [preflightError]
30
+ */
31
+
32
+ /**
33
+ * @typedef {object} TaskReport
34
+ * @property {string} taskId
35
+ * @property {number} n - Total runs.
36
+ * @property {number} c - Passing runs.
37
+ * @property {Record<string|number, number|null>} passAtK
38
+ * @property {RunDetail[]} [runs] - Per-run detail (only when includeRuns).
39
+ */
40
+
41
+ /**
42
+ * @param {{inputDir: string, kValues: number[], includeRuns?: boolean, runtime: import("@forwardimpact/libutil/runtime").Runtime}} opts
43
+ * @returns {Promise<{tasks: TaskReport[], totals: object}>}
44
+ */
45
+ export async function aggregate({
46
+ inputDir,
47
+ kValues,
48
+ includeRuns = false,
49
+ runtime,
50
+ }) {
51
+ if (!runtime) throw new Error("runtime is required");
52
+ const records = await loadRecords(inputDir, runtime);
53
+ const grouped = groupByTask(records.records);
54
+ const tasks = [];
55
+ let totalRuns = 0;
56
+ let totalCost = 0;
57
+ const allDurations = [];
58
+ const allTurns = [];
59
+ let firstRecord = null;
60
+
61
+ for (const [taskId, group] of grouped) {
62
+ const n = group.length;
63
+ const c = group.filter((r) => r.verdict === "pass").length;
64
+ totalRuns += n;
65
+ const passAtK = {};
66
+ for (const k of kValues) passAtK[k] = passAtKValue(n, c, k);
67
+
68
+ const task = { taskId, n, c, passAtK };
69
+
70
+ if (includeRuns) {
71
+ if (!firstRecord) firstRecord = group[0];
72
+ const accumulators = { allDurations, allTurns };
73
+ task.runs = group
74
+ .map((r) => {
75
+ totalCost += r.costUsd ?? 0;
76
+ return buildRunDetail(r, accumulators);
77
+ })
78
+ .sort((a, b) => a.runIndex - b.runIndex);
79
+ }
80
+
81
+ tasks.push(task);
82
+ }
83
+ tasks.sort((a, b) =>
84
+ a.taskId < b.taskId ? -1 : a.taskId > b.taskId ? 1 : 0,
85
+ );
86
+
87
+ const totals = {
88
+ tasks: tasks.length,
89
+ runs: totalRuns,
90
+ skipped: records.skipped,
91
+ };
92
+
93
+ if (includeRuns) {
94
+ totals.costUsd = totalCost;
95
+ totals.medianDurationMs = median(allDurations);
96
+ totals.medianTurns = median(allTurns);
97
+ totals.model = firstRecord?.model ?? "";
98
+ totals.skillSetHash = firstRecord?.skillSetHash ?? "";
99
+ totals.familyRevision = firstRecord?.familyRevision ?? "";
100
+ }
101
+
102
+ return { tasks, totals };
103
+ }
104
+
105
+ /**
106
+ * Build a normalized per-run detail object and accumulate duration/turn
107
+ * samples for median calculation. Extracted from `aggregate` to keep its
108
+ * cognitive complexity below the lint ceiling.
109
+ * @param {object} r - Raw record.
110
+ * @param {{allDurations: number[], allTurns: number[]}} acc
111
+ * @returns {RunDetail}
112
+ */
113
+ function buildRunDetail(r, acc) {
114
+ if (r.durationMs != null) acc.allDurations.push(r.durationMs);
115
+ if (r.turns != null) acc.allTurns.push(r.turns);
116
+ return {
117
+ runIndex: r.runIndex,
118
+ verdict: r.verdict,
119
+ ...(r.invariants && { invariants: r.invariants }),
120
+ ...(r.judgeVerdict && { judgeVerdict: r.judgeVerdict }),
121
+ costUsd: r.costUsd ?? 0,
122
+ turns: r.turns ?? 0,
123
+ durationMs: r.durationMs ?? 0,
124
+ ...(r.agentError && { agentError: r.agentError }),
125
+ ...(r.preflightError && { preflightError: r.preflightError }),
126
+ };
127
+ }
128
+
129
+ /**
130
+ * Render an aggregate report as markdown. When the report contains per-run
131
+ * detail (from `includeRuns: true`), renders a full report with summary,
132
+ * pass@k table, and per-task detail sections. Otherwise falls back to the
133
+ * compact pass@k table.
134
+ * @param {Awaited<ReturnType<typeof aggregate>>} report
135
+ * @param {number[]} kValues
136
+ * @returns {string}
137
+ */
138
+ export function renderTextReport(report, kValues) {
139
+ if (report.tasks[0]?.runs) {
140
+ return renderFullReport(report, kValues);
141
+ }
142
+ return renderCompactReport(report, kValues);
143
+ }
144
+
145
+ // ---------------------------------------------------------------------------
146
+ // Compact report (legacy path)
147
+ // ---------------------------------------------------------------------------
148
+
149
+ function renderCompactReport(report, kValues) {
150
+ const lines = [
151
+ renderPassAtKTable(report, kValues),
152
+ "",
153
+ renderTotalsLine(report),
154
+ ];
155
+ return lines.join("\n");
156
+ }
157
+
158
+ // ---------------------------------------------------------------------------
159
+ // Full report
160
+ // ---------------------------------------------------------------------------
161
+
162
+ function renderFullReport(report, kValues) {
163
+ const sections = [
164
+ renderSummary(report),
165
+ "## Pass@k",
166
+ "",
167
+ renderPassAtKTable(report, kValues),
168
+ "",
169
+ renderTotalsLine(report),
170
+ "",
171
+ "## Task Details",
172
+ ];
173
+
174
+ for (const task of report.tasks) {
175
+ sections.push("");
176
+ sections.push(renderTaskDetail(task));
177
+ }
178
+
179
+ return sections.join("\n");
180
+ }
181
+
182
+ function renderSummary(report) {
183
+ const { totals } = report;
184
+ const passing = report.tasks.filter((t) => t.c > 0 && t.c === t.n).length;
185
+ const icon = statusIcon(passing === totals.tasks);
186
+ const lines = [
187
+ "# Benchmark Report",
188
+ "",
189
+ `${icon} **${passing}/${totals.tasks} tasks passing** | ${totals.runs} runs${totals.skipped ? ` | ${totals.skipped} skipped` : ""}`,
190
+ ];
191
+
192
+ const headers = [];
193
+ const values = [];
194
+ if (totals.costUsd != null) {
195
+ headers.push("Cost");
196
+ values.push(formatCost(totals.costUsd));
197
+ }
198
+ if (totals.medianDurationMs != null) {
199
+ headers.push("Median Duration");
200
+ values.push(formatDuration(totals.medianDurationMs));
201
+ }
202
+ if (totals.medianTurns != null) {
203
+ headers.push("Median Turns");
204
+ values.push(String(totals.medianTurns));
205
+ }
206
+ if (headers.length) {
207
+ lines.push("");
208
+ lines.push(`| ${headers.join(" | ")} |`);
209
+ lines.push(`| ${headers.map(() => "---").join(" | ")} |`);
210
+ lines.push(`| ${values.join(" | ")} |`);
211
+ }
212
+
213
+ const meta = [];
214
+ if (totals.model) {
215
+ meta.push(`Agent: \`${totals.model.agent}\``);
216
+ meta.push(`Supervisor: \`${totals.model.supervisor}\``);
217
+ meta.push(`Judge: \`${totals.model.judge}\``);
218
+ }
219
+ if (totals.skillSetHash) meta.push(`Skill set: \`${totals.skillSetHash}\``);
220
+ if (totals.familyRevision) meta.push(`Family: \`${totals.familyRevision}\``);
221
+ if (meta.length) {
222
+ lines.push("");
223
+ lines.push(meta.join(" | "));
224
+ }
225
+
226
+ lines.push("");
227
+ return lines.join("\n");
228
+ }
229
+
230
+ // ---------------------------------------------------------------------------
231
+ // Pass@k table (shared between compact and full)
232
+ // ---------------------------------------------------------------------------
233
+
234
+ function renderPassAtKTable(report, kValues) {
235
+ const header = ["taskId", "n", "c", ...kValues.map((k) => `pass@${k}`)];
236
+ const rows = [header, header.map(() => "---")];
237
+ for (const t of report.tasks) {
238
+ rows.push([
239
+ t.taskId,
240
+ String(t.n),
241
+ String(t.c),
242
+ ...kValues.map((k) => formatPassAt(t.passAtK[k])),
243
+ ]);
244
+ }
245
+ return rows.map((r) => `| ${r.join(" | ")} |`).join("\n");
246
+ }
247
+
248
+ function renderTotalsLine(report) {
249
+ return `Totals — tasks: ${report.totals.tasks}, runs: ${report.totals.runs}, skipped: ${report.totals.skipped}`;
250
+ }
251
+
252
+ // ---------------------------------------------------------------------------
253
+ // Per-task detail
254
+ // ---------------------------------------------------------------------------
255
+
256
+ function renderTaskDetail(task) {
257
+ const runs = task.runs ?? [];
258
+ const icon = statusIcon(task.c === task.n);
259
+ const singleRun = runs.length === 1;
260
+
261
+ const lines = [
262
+ `### ${task.taskId}`,
263
+ "",
264
+ `${icon} **${task.c}/${task.n} runs passed**`,
265
+ ];
266
+
267
+ lines.push("", renderRunsTable(runs));
268
+
269
+ const checks = renderInvariantChecks(runs, singleRun);
270
+ if (checks) lines.push("", checks);
271
+
272
+ const commentary = renderJudgeCommentary(runs, singleRun);
273
+ if (commentary) lines.push("", commentary);
274
+
275
+ const errors = renderErrors(runs);
276
+ if (errors) lines.push("", errors);
277
+
278
+ return lines.join("\n");
279
+ }
280
+
281
+ function renderRunsTable(runs) {
282
+ const header = [
283
+ "Run",
284
+ "Verdict",
285
+ "Invariants",
286
+ "Judge",
287
+ "Cost",
288
+ "Turns",
289
+ "Duration",
290
+ ];
291
+ const rows = [header, header.map(() => "---")];
292
+ for (const r of runs) {
293
+ const invariantsCell = r.preflightError
294
+ ? "preflight error"
295
+ : r.invariants
296
+ ? statusIcon(r.invariants.verdict === "pass")
297
+ : "—";
298
+ const judgeCell = r.preflightError
299
+ ? "—"
300
+ : r.judgeVerdict
301
+ ? statusIcon(r.judgeVerdict.verdict === "pass")
302
+ : "—";
303
+ rows.push([
304
+ String(r.runIndex),
305
+ statusIcon(r.verdict === "pass"),
306
+ invariantsCell,
307
+ judgeCell,
308
+ formatCost(r.costUsd),
309
+ String(r.turns),
310
+ formatDuration(r.durationMs),
311
+ ]);
312
+ }
313
+ return rows.map((r) => `| ${r.join(" | ")} |`).join("\n");
314
+ }
315
+
316
+ function renderInvariantChecks(runs, singleRun) {
317
+ const rows = collectInvariantRows(runs);
318
+ if (!rows.length) return null;
319
+
320
+ const header = singleRun
321
+ ? ["Check", "Result", "Message"]
322
+ : ["Run", "Check", "Result", "Message"];
323
+ const lines = [
324
+ "#### Invariant Checks",
325
+ "",
326
+ `| ${header.join(" | ")} |`,
327
+ `| ${header.map(() => "---").join(" | ")} |`,
328
+ ];
329
+ for (const row of rows) {
330
+ const cells = singleRun
331
+ ? [row.check, row.result, row.message]
332
+ : [String(row.run), row.check, row.result, row.message];
333
+ lines.push(`| ${cells.join(" | ")} |`);
334
+ }
335
+ return lines.join("\n");
336
+ }
337
+
338
+ function collectInvariantRows(runs) {
339
+ const rows = [];
340
+ for (const r of runs) {
341
+ if (!r.invariants?.details?.length) continue;
342
+ for (const d of r.invariants.details) {
343
+ rows.push({
344
+ run: r.runIndex,
345
+ check: escapeCell(String(d.test ?? "(unnamed)")),
346
+ result: statusIcon(d.pass),
347
+ message: escapeCell(String(d.message ?? "")),
348
+ });
349
+ }
350
+ }
351
+ return rows;
352
+ }
353
+
354
+ function renderJudgeCommentary(runs, singleRun) {
355
+ const entries = runs.filter((r) => r.judgeVerdict?.summary);
356
+ if (!entries.length) return null;
357
+
358
+ const lines = ["#### Judge Commentary", ""];
359
+ for (let i = 0; i < entries.length; i++) {
360
+ const r = entries[i];
361
+ const summary = r.judgeVerdict.summary.replace(/\n/g, "\n> ");
362
+ if (singleRun) {
363
+ lines.push(`> ${summary}`);
364
+ } else {
365
+ lines.push(`> **Run ${r.runIndex}:** ${summary}`);
366
+ }
367
+ if (i < entries.length - 1) lines.push(">");
368
+ }
369
+ return lines.join("\n");
370
+ }
371
+
372
+ function renderErrors(runs) {
373
+ const lines = [];
374
+ for (const r of runs) {
375
+ if (r.agentError) {
376
+ lines.push(
377
+ `- **Run ${r.runIndex}:** Agent error — "${escapeCell(r.agentError.message)}" (aborted: ${r.agentError.aborted})`,
378
+ );
379
+ }
380
+ if (r.preflightError) {
381
+ lines.push(
382
+ `- **Run ${r.runIndex}:** Preflight error — "${escapeCell(r.preflightError.message)}" (exit ${r.preflightError.exitCode})`,
383
+ );
384
+ }
385
+ }
386
+ if (!lines.length) return null;
387
+ return ["#### Errors", "", ...lines].join("\n");
388
+ }
389
+
390
+ // ---------------------------------------------------------------------------
391
+ // Formatting helpers
392
+ // ---------------------------------------------------------------------------
393
+
394
+ function statusIcon(pass) {
395
+ return pass ? "✅" : "❌";
396
+ }
397
+
398
+ function formatPassAt(v) {
399
+ if (v == null) return "—";
400
+ if (typeof v === "object" && "error" in v) return v.error;
401
+ return Number(v).toFixed(4);
402
+ }
403
+
404
+ function formatDuration(ms) {
405
+ if (ms == null || ms === 0) return "0s";
406
+ const totalSeconds = Math.round(ms / 1000);
407
+ if (totalSeconds < 60) return `${totalSeconds}s`;
408
+ const minutes = Math.floor(totalSeconds / 60);
409
+ const seconds = totalSeconds % 60;
410
+ return seconds > 0 ? `${minutes}m ${seconds}s` : `${minutes}m`;
411
+ }
412
+
413
+ function formatCost(usd) {
414
+ if (usd == null) return "$0.00";
415
+ return `$${usd.toFixed(2)}`;
416
+ }
417
+
418
+ function escapeCell(str) {
419
+ return str.replace(/\|/g, "\\|");
420
+ }
421
+
422
+ function median(arr) {
423
+ if (!arr.length) return 0;
424
+ const sorted = [...arr].sort((a, b) => a - b);
425
+ const mid = Math.floor(sorted.length / 2);
426
+ if (sorted.length % 2 === 0) {
427
+ return Math.round((sorted[mid - 1] + sorted[mid]) / 2);
428
+ }
429
+ return sorted[mid];
430
+ }
431
+
432
+ // ---------------------------------------------------------------------------
433
+ // Record loading
434
+ // ---------------------------------------------------------------------------
435
+
436
+ // Directories never worth descending for a `results.jsonl`.
437
+ const SKIP_DIRS = new Set([".git", "node_modules"]);
438
+
439
+ /**
440
+ * Load and union every `results.jsonl` found recursively under `inputDir`.
441
+ *
442
+ * A single non-sharded run has one root-level ledger — the trivial one-match
443
+ * case of the same walk. A sharded run lays each shard's partial ledger in its
444
+ * own subdirectory; merging them equals reporting a single run over the same
445
+ * cells. An *existing* dir with no ledger yields the empty union (exit 0); a
446
+ * *missing* dir lets `readdir`'s ENOENT propagate so `report` still errors.
447
+ * @param {string} inputDir
448
+ * @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
449
+ * @returns {Promise<{records: object[], skipped: number}>}
450
+ */
451
+ async function loadRecords(inputDir, runtime) {
452
+ let files;
453
+ try {
454
+ files = await collectResultsFiles(inputDir, runtime);
455
+ } catch (e) {
456
+ // Re-throw with the stack collapsed to the message line so the CLI's
457
+ // error rendering stays free of node-internal async `readdir` frames
458
+ // (a missing --input dir surfaces its ENOENT as exit 1, matching the
459
+ // pre-1370 stream-error shape the golden captured).
460
+ const err = new Error(e.message);
461
+ if (e.code) err.code = e.code;
462
+ err.stack = `Error: ${e.message}`;
463
+ throw err;
464
+ }
465
+ const records = [];
466
+ let skipped = 0;
467
+ for (const file of files) {
468
+ const content = await runtime.fs.readFile(file, "utf8");
469
+ skipped += parseLedgerInto(content, records, runtime);
470
+ }
471
+ warnOnDuplicateCells(records, runtime);
472
+ return { records, skipped };
473
+ }
474
+
475
+ /**
476
+ * Parse one ledger's JSONL into `records`, skipping malformed or schema-invalid
477
+ * lines with a stderr warning. Returns the skipped count. Extracted from
478
+ * `loadRecords` to keep its cognitive complexity under the lint ceiling.
479
+ * @param {string} content
480
+ * @param {object[]} records - Accumulator, appended in place.
481
+ * @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
482
+ * @returns {number} Skipped line count.
483
+ */
484
+ function parseLedgerInto(content, records, runtime) {
485
+ let skipped = 0;
486
+ for (const line of content.split("\n")) {
487
+ const trimmed = line.trim();
488
+ if (!trimmed) continue;
489
+ let record;
490
+ try {
491
+ record = JSON.parse(trimmed);
492
+ } catch (e) {
493
+ runtime.proc.stderr.write(
494
+ `benchmark report: skipped malformed JSON line — ${e.message}\n`,
495
+ );
496
+ skipped++;
497
+ continue;
498
+ }
499
+ try {
500
+ validateResultRecord(record);
501
+ } catch (e) {
502
+ runtime.proc.stderr.write(
503
+ `benchmark report: skipped record failing schema — ${describeError(e)}\n`,
504
+ );
505
+ skipped++;
506
+ continue;
507
+ }
508
+ records.push(record);
509
+ }
510
+ return skipped;
511
+ }
512
+
513
+ /**
514
+ * Recursively collect paths of every file named `results.jsonl` under `dir`,
515
+ * skipping `.git`/`node_modules` and never following symlinks. A purpose-built
516
+ * `readdir` walk — `task-family.js`'s private `walkFiles` resolves symlinks and
517
+ * is unexported, which is the wrong contract here.
518
+ * @param {string} dir
519
+ * @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
520
+ * @returns {Promise<string[]>}
521
+ */
522
+ async function collectResultsFiles(dir, runtime) {
523
+ const entries = await runtime.fs.readdir(dir, { withFileTypes: true });
524
+ entries.sort((a, b) => (a.name < b.name ? -1 : a.name > b.name ? 1 : 0));
525
+ const out = [];
526
+ for (const entry of entries) {
527
+ if (entry.isSymbolicLink()) continue;
528
+ const full = join(dir, entry.name);
529
+ if (entry.isDirectory()) {
530
+ if (SKIP_DIRS.has(entry.name)) continue;
531
+ out.push(...(await collectResultsFiles(full, runtime)));
532
+ } else if (entry.isFile() && entry.name === "results.jsonl") {
533
+ out.push(full);
534
+ }
535
+ }
536
+ return out;
537
+ }
538
+
539
+ /**
540
+ * Warn (do not silently merge) when a `(taskId, runIndex)` cell appears more
541
+ * than once across shard ledgers. The shard partition guarantees uniqueness, so
542
+ * a duplicate signals misconfiguration; both copies stay in the group so the
543
+ * count is honest.
544
+ * @param {object[]} records
545
+ * @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
546
+ */
547
+ function warnOnDuplicateCells(records, runtime) {
548
+ const counts = new Map();
549
+ for (const r of records) {
550
+ const key = `${r.taskId}#${r.runIndex}`;
551
+ counts.set(key, (counts.get(key) ?? 0) + 1);
552
+ }
553
+ for (const [key, n] of counts) {
554
+ if (n > 1)
555
+ runtime.proc.stderr.write(
556
+ `benchmark report: duplicate cell ${key} appears ${n} times across shard ledgers — the shard partition should make each cell unique\n`,
557
+ );
558
+ }
559
+ }
560
+
561
+ function describeError(e) {
562
+ if (e && Array.isArray(e.issues)) {
563
+ return e.issues.map((i) => `${i.path.join(".")}: ${i.message}`).join("; ");
564
+ }
565
+ return e.message ?? String(e);
566
+ }
567
+
568
+ function groupByTask(records) {
569
+ const out = new Map();
570
+ for (const r of records) {
571
+ if (!out.has(r.taskId)) out.set(r.taskId, []);
572
+ out.get(r.taskId).push(r);
573
+ }
574
+ return out;
575
+ }
576
+
577
+ /**
578
+ * pass@k = 1 - C(n - c, k) / C(n, k). Compute with BigInt to avoid
579
+ * floating-point loss on large n.
580
+ * @param {number} n
581
+ * @param {number} c
582
+ * @param {number} k
583
+ * @returns {number | {error: string}}
584
+ */
585
+ function passAtKValue(n, c, k) {
586
+ if (k > n) return { error: "k > n" };
587
+ if (n - c < k) return 1;
588
+ const total = binomial(BigInt(n), BigInt(k));
589
+ const fail = binomial(BigInt(n - c), BigInt(k));
590
+ const passing = total - fail;
591
+ return Number(passing) / Number(total);
592
+ }
593
+
594
+ function binomial(n, k) {
595
+ if (k < 0n || k > n) return 0n;
596
+ if (k === 0n || k === n) return 1n;
597
+ let kk = k;
598
+ if (kk > n - kk) kk = n - kk;
599
+ let result = 1n;
600
+ for (let i = 0n; i < kk; i++) {
601
+ result = (result * (n - i)) / (i + 1n);
602
+ }
603
+ return result;
604
+ }