@forwardimpact/libharness 1.4.0 → 2.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -15,12 +15,16 @@
15
15
  import { join } from "node:path";
16
16
 
17
17
  import { validateResultRecord } from "./result.js";
18
+ import { mergeRows } from "./grade.js";
18
19
 
19
20
  /**
20
21
  * @typedef {object} RunDetail
21
22
  * @property {number} runIndex
22
23
  * @property {"pass"|"fail"} verdict
23
- * @property {{verdict: string, details: unknown[], exitCode: number}} [invariants]
24
+ * @property {{details: unknown[], exitCode: number}} [invariants]
25
+ * @property {{verdict: string, gatesPass: boolean, score?: number, malformed?: number}} [grade]
26
+ * @property {{details: unknown[], error?: string}} [hiddenTests]
27
+ * @property {number} [score] - Effective judge-zeroed score (scored tasks).
24
28
  * @property {{verdict: string, summary: string}} [judgeVerdict]
25
29
  * @property {number} costUsd
26
30
  * @property {number} turns
@@ -66,6 +70,7 @@ export async function aggregate({
66
70
  for (const k of kValues) passAtK[k] = passAtKValue(n, c, k);
67
71
 
68
72
  const task = { taskId, n, c, passAtK };
73
+ applyScoreFields(task, group, kValues);
69
74
 
70
75
  if (includeRuns) {
71
76
  if (!firstRecord) firstRecord = group[0];
@@ -102,6 +107,25 @@ export async function aggregate({
102
107
  return { tasks, totals };
103
108
  }
104
109
 
110
+ /**
111
+ * Attach `meanScore` and `scoreAtK` to a scored task group. A group is
112
+ * scored iff any record carries an effective score; a score-less record in
113
+ * a scored group (a preflight failure never reached grading, or a binary
114
+ * run) contributes its verdict as the degenerate score — skipping it would
115
+ * inflate the mean exactly when the agent fails hardest. Binary groups gain
116
+ * neither field.
117
+ * @param {object} task - Mutated.
118
+ * @param {object[]} group
119
+ * @param {number[]} kValues
120
+ */
121
+ function applyScoreFields(task, group, kValues) {
122
+ if (!group.some((r) => r.score !== undefined)) return;
123
+ const scores = group.map((r) => r.score ?? (r.verdict === "pass" ? 1 : 0));
124
+ task.meanScore = scores.reduce((a, b) => a + b, 0) / scores.length;
125
+ task.scoreAtK = {};
126
+ for (const k of kValues) task.scoreAtK[k] = scoreAtKValue(scores, k);
127
+ }
128
+
105
129
  /**
106
130
  * Build a normalized per-run detail object and accumulate duration/turn
107
131
  * samples for median calculation. Extracted from `aggregate` to keep its
@@ -117,6 +141,9 @@ function buildRunDetail(r, acc) {
117
141
  runIndex: r.runIndex,
118
142
  verdict: r.verdict,
119
143
  ...(r.invariants && { invariants: r.invariants }),
144
+ ...(r.grade && { grade: r.grade }),
145
+ ...(r.hiddenTests && { hiddenTests: r.hiddenTests }),
146
+ ...(r.score !== undefined && { score: r.score }),
120
147
  ...(r.judgeVerdict && { judgeVerdict: r.judgeVerdict }),
121
148
  costUsd: r.costUsd ?? 0,
122
149
  turns: r.turns ?? 0,
@@ -156,7 +183,7 @@ function renderCompactReport(report, kValues) {
156
183
  const lines = [
157
184
  `${icon} **${passing}/${totals.tasks} tasks passing** | ${totals.runs} runs${totals.skipped ? ` | ${totals.skipped} skipped` : ""}`,
158
185
  "",
159
- renderPassAtKTable(report, kValues),
186
+ renderPassAtKTable(report, kValues, hasScoredTask(report)),
160
187
  "",
161
188
  renderTotalsLine(report),
162
189
  ];
@@ -167,12 +194,18 @@ function renderCompactReport(report, kValues) {
167
194
  // Full report
168
195
  // ---------------------------------------------------------------------------
169
196
 
197
+ /** Score columns render only when the report has at least one scored task. */
198
+ function hasScoredTask(report) {
199
+ return report.tasks.some((t) => t.meanScore !== undefined);
200
+ }
201
+
170
202
  function renderFullReport(report, kValues) {
203
+ const scored = hasScoredTask(report);
171
204
  const sections = [
172
205
  renderSummary(report),
173
206
  "## Pass@k",
174
207
  "",
175
- renderPassAtKTable(report, kValues),
208
+ renderPassAtKTable(report, kValues, scored),
176
209
  "",
177
210
  renderTotalsLine(report),
178
211
  "",
@@ -181,7 +214,7 @@ function renderFullReport(report, kValues) {
181
214
 
182
215
  for (const task of report.tasks) {
183
216
  sections.push("");
184
- sections.push(renderTaskDetail(task));
217
+ sections.push(renderTaskDetail(task, scored));
185
218
  }
186
219
 
187
220
  return sections.join("\n");
@@ -239,16 +272,27 @@ function renderSummary(report) {
239
272
  // Pass@k table (shared between compact and full)
240
273
  // ---------------------------------------------------------------------------
241
274
 
242
- function renderPassAtKTable(report, kValues) {
275
+ function renderPassAtKTable(report, kValues, scored) {
243
276
  const header = ["taskId", "n", "c", ...kValues.map((k) => `pass@${k}`)];
277
+ if (scored) {
278
+ header.push("score", ...kValues.map((k) => `score@${k}`));
279
+ }
244
280
  const rows = [header, header.map(() => "---")];
245
281
  for (const t of report.tasks) {
246
- rows.push([
282
+ const row = [
247
283
  t.taskId,
248
284
  String(t.n),
249
285
  String(t.c),
250
286
  ...kValues.map((k) => formatPassAt(t.passAtK[k])),
251
- ]);
287
+ ];
288
+ if (scored) {
289
+ // Binary tasks render "—" in every score column.
290
+ row.push(
291
+ formatPassAt(t.meanScore ?? null),
292
+ ...kValues.map((k) => formatPassAt(t.scoreAtK?.[k] ?? null)),
293
+ );
294
+ }
295
+ rows.push(row);
252
296
  }
253
297
  return rows.map((r) => `| ${r.join(" | ")} |`).join("\n");
254
298
  }
@@ -261,7 +305,7 @@ function renderTotalsLine(report) {
261
305
  // Per-task detail
262
306
  // ---------------------------------------------------------------------------
263
307
 
264
- function renderTaskDetail(task) {
308
+ function renderTaskDetail(task, scored) {
265
309
  const runs = task.runs ?? [];
266
310
  const icon = statusIcon(task.c === task.n);
267
311
  const singleRun = runs.length === 1;
@@ -272,9 +316,9 @@ function renderTaskDetail(task) {
272
316
  `${icon} **${task.c}/${task.n} runs passed**`,
273
317
  ];
274
318
 
275
- lines.push("", renderRunsTable(runs));
319
+ lines.push("", renderRunsTable(runs, scored));
276
320
 
277
- const checks = renderInvariantChecks(runs, singleRun);
321
+ const checks = renderChecks(runs, singleRun);
278
322
  if (checks) lines.push("", checks);
279
323
 
280
324
  const commentary = renderJudgeCommentary(runs, singleRun);
@@ -286,33 +330,32 @@ function renderTaskDetail(task) {
286
330
  return lines.join("\n");
287
331
  }
288
332
 
289
- function renderRunsTable(runs) {
333
+ function renderRunsTable(runs, scored) {
290
334
  const header = [
291
335
  "Run",
292
336
  "Verdict",
293
- "Invariants",
337
+ "Checks",
294
338
  "Judge",
339
+ ...(scored ? ["Score"] : []),
295
340
  "Cost",
296
341
  "Turns",
297
342
  "Duration",
298
343
  ];
299
344
  const rows = [header, header.map(() => "---")];
300
345
  for (const r of runs) {
301
- const invariantsCell = r.preflightError
302
- ? "preflight error"
303
- : r.invariants
304
- ? statusIcon(r.invariants.verdict === "pass")
305
- : "—";
306
- const judgeCell = r.preflightError
307
- ? "—"
308
- : r.judgeVerdict
309
- ? statusIcon(r.judgeVerdict.verdict === "pass")
310
- : "—";
346
+ // A preflight-failure record is the one grade-less branch in the schema.
347
+ const checksCell = r.grade ? statusIcon(r.grade.verdict === "pass") : "—";
348
+ const judgeCell = r.judgeVerdict
349
+ ? statusIcon(r.judgeVerdict.verdict === "pass")
350
+ : "—";
311
351
  rows.push([
312
352
  String(r.runIndex),
313
353
  statusIcon(r.verdict === "pass"),
314
- invariantsCell,
354
+ checksCell,
315
355
  judgeCell,
356
+ ...(scored
357
+ ? [r.score !== undefined ? Number(r.score).toFixed(4) : "—"]
358
+ : []),
316
359
  formatCost(r.costUsd),
317
360
  String(r.turns),
318
361
  formatDuration(r.durationMs),
@@ -321,42 +364,45 @@ function renderRunsTable(runs) {
321
364
  return rows.map((r) => `| ${r.join(" | ")} |`).join("\n");
322
365
  }
323
366
 
324
- function renderInvariantChecks(runs, singleRun) {
325
- const rows = collectInvariantRows(runs);
367
+ function renderChecks(runs, singleRun) {
368
+ const { rows, hasHidden } = collectCheckRows(runs);
326
369
  if (!rows.length) return null;
327
370
 
328
- const header = singleRun
329
- ? ["Check", "Result", "Message"]
330
- : ["Run", "Check", "Result", "Message"];
371
+ const header = singleRun ? ["Check"] : ["Run", "Check"];
372
+ if (hasHidden) header.push("Source");
373
+ header.push("Result", "Message");
331
374
  const lines = [
332
- "#### Invariant Checks",
375
+ "#### Checks",
333
376
  "",
334
377
  `| ${header.join(" | ")} |`,
335
378
  `| ${header.map(() => "---").join(" | ")} |`,
336
379
  ];
337
380
  for (const row of rows) {
338
- const cells = singleRun
339
- ? [row.check, row.result, row.message]
340
- : [String(row.run), row.check, row.result, row.message];
381
+ const cells = singleRun ? [row.check] : [String(row.run), row.check];
382
+ if (hasHidden) cells.push(row.source);
383
+ cells.push(row.result, row.message);
341
384
  lines.push(`| ${cells.join(" | ")} |`);
342
385
  }
343
386
  return lines.join("\n");
344
387
  }
345
388
 
346
- function collectInvariantRows(runs) {
347
- const rows = [];
348
- for (const r of runs) {
349
- if (!r.invariants?.details?.length) continue;
350
- for (const d of r.invariants.details) {
351
- rows.push({
389
+ /**
390
+ * Merge both producers' check rows per run. The Source column appears only
391
+ * when at least one row came from a hidden test suite.
392
+ */
393
+ function collectCheckRows(runs) {
394
+ const rows = runs.flatMap((r) =>
395
+ mergeRows(r.invariants?.details ?? [], r.hiddenTests?.details ?? [])
396
+ .filter((d) => d !== null && typeof d === "object")
397
+ .map((d) => ({
352
398
  run: r.runIndex,
353
399
  check: escapeCell(String(d.test ?? "(unnamed)")),
400
+ source: d.source ?? "",
354
401
  result: statusIcon(d.pass),
355
402
  message: escapeCell(String(d.message ?? "")),
356
- });
357
- }
358
- }
359
- return rows;
403
+ })),
404
+ );
405
+ return { rows, hasHidden: rows.some((row) => row.source === "tests") };
360
406
  }
361
407
 
362
408
  function renderJudgeCommentary(runs, singleRun) {
@@ -378,23 +424,36 @@ function renderJudgeCommentary(runs, singleRun) {
378
424
  }
379
425
 
380
426
  function renderErrors(runs) {
381
- const lines = [];
382
- for (const r of runs) {
383
- if (r.agentError) {
384
- lines.push(
385
- `- **Run ${r.runIndex}:** Agent error — "${escapeCell(r.agentError.message)}" (aborted: ${r.agentError.aborted})`,
386
- );
387
- }
388
- if (r.preflightError) {
389
- lines.push(
390
- `- **Run ${r.runIndex}:** Preflight error — "${escapeCell(r.preflightError.message)}" (exit ${r.preflightError.exitCode})`,
391
- );
392
- }
393
- }
427
+ const lines = runs.flatMap(runErrorLines);
394
428
  if (!lines.length) return null;
395
429
  return ["#### Errors", "", ...lines].join("\n");
396
430
  }
397
431
 
432
+ function runErrorLines(r) {
433
+ const lines = [];
434
+ if (r.grade?.malformed) {
435
+ lines.push(
436
+ `- **Run ${r.runIndex}:** ⚠️ ${r.grade.malformed} malformed check row(s) — counted as failing`,
437
+ );
438
+ }
439
+ if (r.hiddenTests?.error) {
440
+ lines.push(
441
+ `- **Run ${r.runIndex}:** Hidden-test engine error — "${escapeCell(r.hiddenTests.error)}"`,
442
+ );
443
+ }
444
+ if (r.agentError) {
445
+ lines.push(
446
+ `- **Run ${r.runIndex}:** Agent error — "${escapeCell(r.agentError.message)}" (aborted: ${r.agentError.aborted})`,
447
+ );
448
+ }
449
+ if (r.preflightError) {
450
+ lines.push(
451
+ `- **Run ${r.runIndex}:** Preflight error — "${escapeCell(r.preflightError.message)}" (exit ${r.preflightError.exitCode})`,
452
+ );
453
+ }
454
+ return lines;
455
+ }
456
+
398
457
  // ---------------------------------------------------------------------------
399
458
  // Formatting helpers
400
459
  // ---------------------------------------------------------------------------
@@ -599,6 +658,33 @@ function passAtKValue(n, c, k) {
599
658
  return Number(passing) / Number(total);
600
659
  }
601
660
 
661
+ /**
662
+ * score@k — the expected **maximum** score over k runs drawn without
663
+ * replacement from the n recorded scores; the continuous analog of pass@k.
664
+ * With scores sorted ascending s₍₁₎…s₍ₙ₎:
665
+ *
666
+ * score@k = Σ_{i=k..n} s₍ᵢ₎ · C(i−1, k−1) / C(n, k)
667
+ *
668
+ * Each term weights s₍ᵢ₎ by the probability it is the k-subset's maximum.
669
+ * Binary scores reduce exactly to the pass@k estimator (same BigInt binomial
670
+ * helper); `k > n` yields the same `{error}` value — one idiom.
671
+ * @param {number[]} scores - Effective per-record scores.
672
+ * @param {number} k
673
+ * @returns {number | {error: string}}
674
+ */
675
+ function scoreAtKValue(scores, k) {
676
+ const n = scores.length;
677
+ if (k > n) return { error: "k > n" };
678
+ const sorted = [...scores].sort((a, b) => a - b);
679
+ const total = Number(binomial(BigInt(n), BigInt(k)));
680
+ let sum = 0;
681
+ for (let i = k; i <= n; i++) {
682
+ const weight = Number(binomial(BigInt(i - 1), BigInt(k - 1))) / total;
683
+ sum += sorted[i - 1] * weight;
684
+ }
685
+ return sum;
686
+ }
687
+
602
688
  function binomial(n, k) {
603
689
  if (k < 0n || k > n) return 0n;
604
690
  if (k === 0n || k === n) return 1n;
@@ -3,10 +3,14 @@
3
3
  *
4
4
  * Two schemas live here:
5
5
  * - RESULT_RECORD_SCHEMA — one record per (task, runIndex) from a full
6
- * benchmark run. Has a happy branch (invariants + judge present) and a
7
- * pre-flight-failure branch (invariants/judgeVerdict/submission absent).
8
- * - INVARIANTS_RECORD_SCHEMA — narrower output of `benchmark-invariants`:
9
- * ad-hoc grading without a full lifecycle.
6
+ * benchmark run. Has a happy branch (grade + collectors + judge present)
7
+ * and a pre-flight-failure branch (grade/judgeVerdict/submission absent).
8
+ * - GRADE_RECORD_SCHEMA — narrower output of `benchmark-grade`: ad-hoc
9
+ * grading without a full lifecycle.
10
+ *
11
+ * The check rows are the authoritative grading channel: the happy branch
12
+ * requires a `grade` object, so a pre-break record fails validation rather
13
+ * than rendering under semantics it never carried.
10
14
  *
11
15
  * Validation is throw-on-mismatch so the runner can wrap every JSONL append
12
16
  * in a guard and reject schema drift at write time.
@@ -17,12 +21,27 @@ import { z } from "zod";
17
21
  const VERDICT_ENUM = z.enum(["pass", "fail"]);
18
22
 
19
23
  const INVARIANTS_SHAPE = z.object({
20
- verdict: VERDICT_ENUM,
21
24
  details: z.array(z.unknown()),
22
25
  exitCode: z.number().int(),
23
26
  stderr: z.string().optional(),
24
27
  });
25
28
 
29
+ /**
30
+ * The normalized grading projection: `score` appears only on scored tasks,
31
+ * `malformed` only when at least one row was malformed.
32
+ */
33
+ const GRADE_SHAPE = z.object({
34
+ verdict: VERDICT_ENUM,
35
+ gatesPass: z.boolean(),
36
+ score: z.number().min(0).max(1).optional(),
37
+ malformed: z.number().int().min(1).optional(),
38
+ });
39
+
40
+ const HIDDEN_TESTS_SHAPE = z.object({
41
+ details: z.array(z.unknown()),
42
+ error: z.string().optional(),
43
+ });
44
+
26
45
  const JUDGE_VERDICT_SHAPE = z.object({
27
46
  verdict: VERDICT_ENUM,
28
47
  summary: z.string(),
@@ -77,6 +96,11 @@ const AGENT_ERROR_SHAPE = z.object({
77
96
  const HAPPY_RECORD = z.object({
78
97
  ...COMMON_FIELDS,
79
98
  invariants: INVARIANTS_SHAPE,
99
+ grade: GRADE_SHAPE,
100
+ hiddenTests: HIDDEN_TESTS_SHAPE.optional(),
101
+ // The effective, judge-zeroed score `report` aggregates — present only on
102
+ // scored tasks.
103
+ score: z.number().min(0).max(1).optional(),
80
104
  submission: z.string(),
81
105
  judgeVerdict: JUDGE_VERDICT_SHAPE.optional(),
82
106
  agentTracePath: z.string(),
@@ -97,6 +121,9 @@ const PREFLIGHT_RECORD = z.object({
97
121
  supervisorTracePath: z.string(),
98
122
  judgeTracePath: z.string(),
99
123
  invariants: z.undefined().optional(),
124
+ grade: z.undefined().optional(),
125
+ hiddenTests: z.undefined().optional(),
126
+ score: z.undefined().optional(),
100
127
  submission: z.undefined().optional(),
101
128
  judgeVerdict: z.undefined().optional(),
102
129
  agentError: z.undefined().optional(),
@@ -104,9 +131,17 @@ const PREFLIGHT_RECORD = z.object({
104
131
 
105
132
  export const RESULT_RECORD_SCHEMA = z.union([HAPPY_RECORD, PREFLIGHT_RECORD]);
106
133
 
107
- export const INVARIANTS_RECORD_SCHEMA = z.object({
134
+ export const GRADE_RECORD_SCHEMA = z.object({
108
135
  taskId: z.string().min(1),
136
+ // Unlike the happy result record — where `grade.score` is the raw
137
+ // weighted fraction and the effective (zeroed) value lives on the
138
+ // top-level `score` — this record has no second score field, so its
139
+ // `grade.score` carries the effective health/gate-zeroed value.
140
+ grade: GRADE_SHAPE,
109
141
  invariants: INVARIANTS_SHAPE,
142
+ hiddenTests: HIDDEN_TESTS_SHAPE.optional(),
143
+ // Mirrors the invariants script's exit for diagnosis; the graded verdict
144
+ // is what drives the command's process exit.
110
145
  exitCode: z.number().int(),
111
146
  });
112
147
 
@@ -122,6 +157,6 @@ export function validateResultRecord(record) {
122
157
  * Throw on schema mismatch.
123
158
  * @param {object} record
124
159
  */
125
- export function validateInvariantsRecord(record) {
126
- INVARIANTS_RECORD_SCHEMA.parse(record);
160
+ export function validateGradeRecord(record) {
161
+ GRADE_RECORD_SCHEMA.parse(record);
127
162
  }