@forwardimpact/libharness 1.3.0 → 2.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,180 @@
1
+ /**
2
+ * Hidden-test engine — executes a task's `tests/` overlay against the
3
+ * post-run agent CWD: stage each file at its mirrored path, run each check
4
+ * with `node --test`, convert the exit status into one check row, and
5
+ * restore the tree so the judge sees the workdir exactly as the agent left
6
+ * it.
7
+ *
8
+ * Fault attribution is the engine's contract: a stage or spawn failure (the
9
+ * agent deleted the scaffold) is a *failing row* — agent fault; the engine
10
+ * itself throwing is grader fault, which the caller records as unhealthy so
11
+ * a crashed grader can never mint marks.
12
+ */
13
+
14
+ import { dirname, join } from "node:path";
15
+
16
+ import { buildHookEnv } from "./hook-env.js";
17
+
18
+ // Fixed per-check budget. A wedged test process runs outside the agent
19
+ // watchdog, so this bound is what keeps a hung hidden test from stalling the
20
+ // cell; the timeout row keeps the failure visible.
21
+ const CHECK_TIMEOUT_MS = 120_000;
22
+ const STDERR_TAIL_CHARS = 500;
23
+
24
+ /**
25
+ * Run the task's hidden test suite.
26
+ * @param {import("./task-family.js").Task} task
27
+ * @param {{cwd: string, port: number, runDir: string, familyDir?: string|null}} ctx
28
+ * @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
29
+ * @param {{timeoutMs?: number}} [opts] - Test seam for the per-check timeout.
30
+ * @returns {Promise<{details: object[]}>}
31
+ */
32
+ export async function runHiddenTests(task, ctx, runtime, opts = {}) {
33
+ if (!runtime) throw new Error("runtime is required");
34
+ if (!task.tests) return { details: [] };
35
+ const timeoutMs = opts.timeoutMs ?? CHECK_TIMEOUT_MS;
36
+ const fs = runtime.fs;
37
+ const details = [];
38
+ const supportStager = newStager();
39
+ try {
40
+ for (const file of task.tests.support) {
41
+ await stageFile(fs, ctx.cwd, supportStager, file);
42
+ }
43
+ for (const check of task.tests.checks) {
44
+ details.push(await runOneCheck(task, ctx, runtime, timeoutMs, check));
45
+ }
46
+ } finally {
47
+ await unstage(fs, supportStager);
48
+ }
49
+ return { details };
50
+ }
51
+
52
+ /**
53
+ * Stage one check, run it, and restore its staging — the check's own row is
54
+ * the only trace it leaves. A stage failure is the agent's fault (a deleted
55
+ * scaffold), so it becomes a failing row rather than a throw.
56
+ */
57
+ async function runOneCheck(task, ctx, runtime, timeoutMs, check) {
58
+ const stager = newStager();
59
+ try {
60
+ try {
61
+ await stageFile(runtime.fs, ctx.cwd, stager, check);
62
+ } catch (e) {
63
+ return checkRow(check, false, `stage failed: ${e.message}`);
64
+ }
65
+ return await spawnCheck(task, ctx, runtime, timeoutMs, check);
66
+ } finally {
67
+ await unstage(runtime.fs, stager);
68
+ }
69
+ }
70
+
71
+ /**
72
+ * Spawn `node --test <staged path>` from the agent CWD under the hook env
73
+ * and map the exit status onto one row. The clock timer SIGKILLs a child
74
+ * that outlives the per-check budget; the row fails with a timeout message.
75
+ */
76
+ async function spawnCheck(task, ctx, runtime, timeoutMs, check) {
77
+ const env = buildHookEnv(runtime.proc.env, {
78
+ cwd: ctx.cwd,
79
+ port: ctx.port,
80
+ taskId: task.id,
81
+ taskDir: task.paths.taskDir,
82
+ hooksDir: task.paths.hooks,
83
+ familyDir: ctx.familyDir,
84
+ });
85
+ // An inherited test-runner context makes the child `node --test` report
86
+ // exit 0 even when its tests fail — a failing check would mint a passing
87
+ // row whenever the harness itself runs under `node --test`.
88
+ delete env.NODE_TEST_CONTEXT;
89
+ const child = runtime.subprocess.spawn("node", ["--test", check.stagePath], {
90
+ cwd: ctx.cwd,
91
+ env,
92
+ stdio: ["ignore", "pipe", "pipe"],
93
+ });
94
+ let timedOut = false;
95
+ const timer = runtime.clock.setTimeout(() => {
96
+ timedOut = true;
97
+ child.kill("SIGKILL");
98
+ }, timeoutMs);
99
+ const drainStdout = (async () => {
100
+ for await (const _chunk of child.stdout) {
101
+ // discard
102
+ }
103
+ })();
104
+ let stderr = "";
105
+ for await (const chunk of child.stderr) stderr += chunk.toString();
106
+ await drainStdout;
107
+ const exit = await child.exitCode;
108
+ runtime.clock.clearTimeout(timer);
109
+
110
+ if (timedOut) {
111
+ return checkRow(check, false, `timed out after ${timeoutMs}ms`);
112
+ }
113
+ if (exit === 0) return checkRow(check, true);
114
+ const tail = stderr.trim().slice(-STDERR_TAIL_CHARS);
115
+ return checkRow(check, false, `exit ${exit}${tail ? `: ${tail}` : ""}`);
116
+ }
117
+
118
+ function checkRow(check, pass, message) {
119
+ return {
120
+ test: check.name,
121
+ pass,
122
+ ...(check.gate && { gate: true }),
123
+ ...(message && { message }),
124
+ };
125
+ }
126
+
127
+ function newStager() {
128
+ return { staged: [], backups: [], createdDirs: [] };
129
+ }
130
+
131
+ /**
132
+ * Copy the symlink-resolved source to its mirrored path under the agent CWD,
133
+ * backing up a collided file's bytes and tracking every directory created so
134
+ * `unstage` can put the tree back exactly.
135
+ */
136
+ async function stageFile(fs, cwd, stager, { sourcePath, stagePath }) {
137
+ const target = join(cwd, stagePath);
138
+ let collided = null;
139
+ try {
140
+ collided = await fs.readFile(target);
141
+ } catch {
142
+ // no collision
143
+ }
144
+ if (collided !== null) stager.backups.push({ target, bytes: collided });
145
+ await ensureParents(fs, cwd, stager, dirname(target));
146
+ const resolved = await fs.realpath(sourcePath);
147
+ await fs.copyFile(resolved, target);
148
+ stager.staged.push(target);
149
+ }
150
+
151
+ async function ensureParents(fs, cwd, stager, dir) {
152
+ if (dir === cwd) return;
153
+ try {
154
+ await fs.access(dir);
155
+ return;
156
+ } catch {
157
+ // missing — create below
158
+ }
159
+ await ensureParents(fs, cwd, stager, dirname(dir));
160
+ await fs.mkdir(dir);
161
+ stager.createdDirs.push(dir);
162
+ }
163
+
164
+ /**
165
+ * Reverse the staging: staged copies out, collided bytes back, created
166
+ * directories removed (deepest first — a check's own artifacts inside a
167
+ * created directory go with it, since that directory did not exist when the
168
+ * agent finished).
169
+ */
170
+ async function unstage(fs, stager) {
171
+ for (const target of stager.staged) {
172
+ await fs.rm(target, { force: true });
173
+ }
174
+ for (const backup of stager.backups) {
175
+ await fs.writeFile(backup.target, backup.bytes);
176
+ }
177
+ for (const dir of [...stager.createdDirs].reverse()) {
178
+ await fs.rm(dir, { recursive: true, force: true });
179
+ }
180
+ }
@@ -1,7 +1,10 @@
1
1
  /**
2
2
  * Invariants — runs `<task.paths.hooks>/invariants.sh` from the template path
3
- * against the post-run agent CWD. The exit code is authoritative for the
4
- * verdict; structured per-check rows arrive on fd 3 (`$RESULTS_FD=3`) as NDJSON.
3
+ * against the post-run agent CWD. A pure collector with no verdict of its
4
+ * own: structured per-check rows arrive on fd 3 (`$RESULTS_FD=3`) as NDJSON
5
+ * and grading happens downstream over the merged rows. The exit code is
6
+ * script health only — nonzero means the grader itself failed, never that a
7
+ * check failed.
5
8
  *
6
9
  * Subprocess access flows through `runtime.subprocess.spawn`; the fd-3 backing
7
10
  * store and the stderr log use the sync filesystem surface (`runtime.fsSync`) —
@@ -14,9 +17,9 @@ import { buildHookEnv } from "./hook-env.js";
14
17
 
15
18
  /**
16
19
  * @typedef {object} InvariantsResult
17
- * @property {"pass" | "fail"} verdict
18
20
  * @property {Array<object>} details
19
- * @property {number} exitCode
21
+ * @property {number} exitCode - Script health: nonzero means the hook itself
22
+ * failed, never that a check failed.
20
23
  * @property {string} [stderr] - Trimmed script stderr, present only when the
21
24
  * script wrote to stderr. Surfaces hook failures (e.g. a missing tool) that
22
25
  * leave `details` empty, so they read distinctly from a real invariant miss.
@@ -32,7 +35,7 @@ import { buildHookEnv } from "./hook-env.js";
32
35
  export async function runInvariants(task, ctx, runtime) {
33
36
  if (!runtime) throw new Error("runtime is required");
34
37
  if (!task.paths.invariants) {
35
- return { verdict: "pass", details: [], exitCode: 0 };
38
+ return { details: [], exitCode: 0 };
36
39
  }
37
40
  const fsSync = runtime.fsSync;
38
41
  const script = task.paths.invariants;
@@ -84,11 +87,7 @@ export async function runInvariants(task, ctx, runtime) {
84
87
  const details = [];
85
88
  parseFd3Buffer(raw, details);
86
89
  const exitCode = typeof code === "number" ? code : -1;
87
- const result = {
88
- verdict: exitCode === 0 ? "pass" : "fail",
89
- details,
90
- exitCode,
91
- };
90
+ const result = { details, exitCode };
92
91
  const trimmedStderr = stderr.trim();
93
92
  if (trimmedStderr) result.stderr = trimmedStderr;
94
93
  return result;
@@ -9,7 +9,7 @@
9
9
  * {{AGENT_INSTRUCTIONS}} — contents of agent.task.md
10
10
  * {{AGENT_PROFILE}} — agent profile body (empty string if none)
11
11
  * {{AGENT_TRACE_PATH}} — path to agent.ndjson
12
- * {{INVARIANTS_RESULT}} — JSON invariants object
12
+ * {{GRADE_RESULT}} — JSON grade object plus the merged check rows
13
13
  * {{SKILL_SET_HASH}} — SHA-256 from apm.lock.yaml
14
14
  * {{TASK_ID}} — task name (directory under tasks/)
15
15
  * {{TASK_DIR}} — agent working directory path
@@ -40,22 +40,25 @@ import { sumTraceCost } from "../cost.js";
40
40
  */
41
41
 
42
42
  /**
43
- * Run the judge over a completed task run.
43
+ * Run the judge over a completed task run. The judge is a binary gate over
44
+ * the grade's validity, never a grade itself: `gradeResult` reaches the
45
+ * template as evidence, and the verdict stays pass/fail.
44
46
  * @param {import("./task-family.js").Task} task
45
47
  * @param {import("./workdir.js").Workdir} workdir
46
- * @param {import("./invariants.js").InvariantsResult} invariants
48
+ * @param {{verdict: string, gatesPass: boolean, score?: number, malformed?: number, rows: unknown[]}} gradeResult -
49
+ * The normalized grade plus the merged, source-stamped check rows.
47
50
  * @param {{query: Function, model: string, judgeProfile?: string, profilesDir?: string, runtime: import("@forwardimpact/libutil/runtime").Runtime}} deps
48
51
  * @param {JudgeContext} [context]
49
52
  * @returns {Promise<JudgeVerdict>}
50
53
  */
51
- export async function runJudge(task, workdir, invariants, deps, context) {
54
+ export async function runJudge(task, workdir, gradeResult, deps, context) {
52
55
  const runtime = deps.runtime;
53
56
  if (!runtime) throw new Error("runtime is required");
54
57
  const fs = runtime.fs;
55
58
  const template = await fs.readFile(task.paths.judge, "utf8");
56
- const invariantsJson = JSON.stringify(invariants, null, 2);
59
+ const gradeJson = JSON.stringify(gradeResult, null, 2);
57
60
  const taskText = template
58
- .replaceAll("{{INVARIANTS_RESULT}}", invariantsJson)
61
+ .replaceAll("{{GRADE_RESULT}}", gradeJson)
59
62
  .replaceAll("{{AGENT_TRACE_PATH}}", workdir.agentTracePath)
60
63
  .replaceAll("{{AGENT_INSTRUCTIONS}}", context?.agentInstructions ?? "")
61
64
  .replaceAll("{{AGENT_PROFILE}}", context?.agentProfile ?? "")
@@ -15,12 +15,16 @@
15
15
  import { join } from "node:path";
16
16
 
17
17
  import { validateResultRecord } from "./result.js";
18
+ import { mergeRows } from "./grade.js";
18
19
 
19
20
  /**
20
21
  * @typedef {object} RunDetail
21
22
  * @property {number} runIndex
22
23
  * @property {"pass"|"fail"} verdict
23
- * @property {{verdict: string, details: unknown[], exitCode: number}} [invariants]
24
+ * @property {{details: unknown[], exitCode: number}} [invariants]
25
+ * @property {{verdict: string, gatesPass: boolean, score?: number, malformed?: number}} [grade]
26
+ * @property {{details: unknown[], error?: string}} [hiddenTests]
27
+ * @property {number} [score] - Effective judge-zeroed score (scored tasks).
24
28
  * @property {{verdict: string, summary: string}} [judgeVerdict]
25
29
  * @property {number} costUsd
26
30
  * @property {number} turns
@@ -66,6 +70,7 @@ export async function aggregate({
66
70
  for (const k of kValues) passAtK[k] = passAtKValue(n, c, k);
67
71
 
68
72
  const task = { taskId, n, c, passAtK };
73
+ applyScoreFields(task, group, kValues);
69
74
 
70
75
  if (includeRuns) {
71
76
  if (!firstRecord) firstRecord = group[0];
@@ -102,6 +107,25 @@ export async function aggregate({
102
107
  return { tasks, totals };
103
108
  }
104
109
 
110
+ /**
111
+ * Attach `meanScore` and `scoreAtK` to a scored task group. A group is
112
+ * scored iff any record carries an effective score; a score-less record in
113
+ * a scored group (a preflight failure never reached grading, or a binary
114
+ * run) contributes its verdict as the degenerate score — skipping it would
115
+ * inflate the mean exactly when the agent fails hardest. Binary groups gain
116
+ * neither field.
117
+ * @param {object} task - Mutated.
118
+ * @param {object[]} group
119
+ * @param {number[]} kValues
120
+ */
121
+ function applyScoreFields(task, group, kValues) {
122
+ if (!group.some((r) => r.score !== undefined)) return;
123
+ const scores = group.map((r) => r.score ?? (r.verdict === "pass" ? 1 : 0));
124
+ task.meanScore = scores.reduce((a, b) => a + b, 0) / scores.length;
125
+ task.scoreAtK = {};
126
+ for (const k of kValues) task.scoreAtK[k] = scoreAtKValue(scores, k);
127
+ }
128
+
105
129
  /**
106
130
  * Build a normalized per-run detail object and accumulate duration/turn
107
131
  * samples for median calculation. Extracted from `aggregate` to keep its
@@ -117,6 +141,9 @@ function buildRunDetail(r, acc) {
117
141
  runIndex: r.runIndex,
118
142
  verdict: r.verdict,
119
143
  ...(r.invariants && { invariants: r.invariants }),
144
+ ...(r.grade && { grade: r.grade }),
145
+ ...(r.hiddenTests && { hiddenTests: r.hiddenTests }),
146
+ ...(r.score !== undefined && { score: r.score }),
120
147
  ...(r.judgeVerdict && { judgeVerdict: r.judgeVerdict }),
121
148
  costUsd: r.costUsd ?? 0,
122
149
  turns: r.turns ?? 0,
@@ -143,12 +170,20 @@ export function renderTextReport(report, kValues) {
143
170
  }
144
171
 
145
172
  // ---------------------------------------------------------------------------
146
- // Compact report (legacy path)
173
+ // Compact report — status line + pass@k table, no per-task detail. Selected by
174
+ // `report --detail=compact` (aggregate without `includeRuns`); the per-shard
175
+ // summary uses it so a sharded run stays short while the merge job renders the
176
+ // full report over the combined ledger.
147
177
  // ---------------------------------------------------------------------------
148
178
 
149
179
  function renderCompactReport(report, kValues) {
180
+ const { totals } = report;
181
+ const passing = report.tasks.filter((t) => t.c > 0 && t.c === t.n).length;
182
+ const icon = statusIcon(passing === totals.tasks);
150
183
  const lines = [
151
- renderPassAtKTable(report, kValues),
184
+ `${icon} **${passing}/${totals.tasks} tasks passing** | ${totals.runs} runs${totals.skipped ? ` | ${totals.skipped} skipped` : ""}`,
185
+ "",
186
+ renderPassAtKTable(report, kValues, hasScoredTask(report)),
152
187
  "",
153
188
  renderTotalsLine(report),
154
189
  ];
@@ -159,12 +194,18 @@ function renderCompactReport(report, kValues) {
159
194
  // Full report
160
195
  // ---------------------------------------------------------------------------
161
196
 
197
+ /** Score columns render only when the report has at least one scored task. */
198
+ function hasScoredTask(report) {
199
+ return report.tasks.some((t) => t.meanScore !== undefined);
200
+ }
201
+
162
202
  function renderFullReport(report, kValues) {
203
+ const scored = hasScoredTask(report);
163
204
  const sections = [
164
205
  renderSummary(report),
165
206
  "## Pass@k",
166
207
  "",
167
- renderPassAtKTable(report, kValues),
208
+ renderPassAtKTable(report, kValues, scored),
168
209
  "",
169
210
  renderTotalsLine(report),
170
211
  "",
@@ -173,7 +214,7 @@ function renderFullReport(report, kValues) {
173
214
 
174
215
  for (const task of report.tasks) {
175
216
  sections.push("");
176
- sections.push(renderTaskDetail(task));
217
+ sections.push(renderTaskDetail(task, scored));
177
218
  }
178
219
 
179
220
  return sections.join("\n");
@@ -231,16 +272,27 @@ function renderSummary(report) {
231
272
  // Pass@k table (shared between compact and full)
232
273
  // ---------------------------------------------------------------------------
233
274
 
234
- function renderPassAtKTable(report, kValues) {
275
+ function renderPassAtKTable(report, kValues, scored) {
235
276
  const header = ["taskId", "n", "c", ...kValues.map((k) => `pass@${k}`)];
277
+ if (scored) {
278
+ header.push("score", ...kValues.map((k) => `score@${k}`));
279
+ }
236
280
  const rows = [header, header.map(() => "---")];
237
281
  for (const t of report.tasks) {
238
- rows.push([
282
+ const row = [
239
283
  t.taskId,
240
284
  String(t.n),
241
285
  String(t.c),
242
286
  ...kValues.map((k) => formatPassAt(t.passAtK[k])),
243
- ]);
287
+ ];
288
+ if (scored) {
289
+ // Binary tasks render "—" in every score column.
290
+ row.push(
291
+ formatPassAt(t.meanScore ?? null),
292
+ ...kValues.map((k) => formatPassAt(t.scoreAtK?.[k] ?? null)),
293
+ );
294
+ }
295
+ rows.push(row);
244
296
  }
245
297
  return rows.map((r) => `| ${r.join(" | ")} |`).join("\n");
246
298
  }
@@ -253,7 +305,7 @@ function renderTotalsLine(report) {
253
305
  // Per-task detail
254
306
  // ---------------------------------------------------------------------------
255
307
 
256
- function renderTaskDetail(task) {
308
+ function renderTaskDetail(task, scored) {
257
309
  const runs = task.runs ?? [];
258
310
  const icon = statusIcon(task.c === task.n);
259
311
  const singleRun = runs.length === 1;
@@ -264,9 +316,9 @@ function renderTaskDetail(task) {
264
316
  `${icon} **${task.c}/${task.n} runs passed**`,
265
317
  ];
266
318
 
267
- lines.push("", renderRunsTable(runs));
319
+ lines.push("", renderRunsTable(runs, scored));
268
320
 
269
- const checks = renderInvariantChecks(runs, singleRun);
321
+ const checks = renderChecks(runs, singleRun);
270
322
  if (checks) lines.push("", checks);
271
323
 
272
324
  const commentary = renderJudgeCommentary(runs, singleRun);
@@ -278,33 +330,32 @@ function renderTaskDetail(task) {
278
330
  return lines.join("\n");
279
331
  }
280
332
 
281
- function renderRunsTable(runs) {
333
+ function renderRunsTable(runs, scored) {
282
334
  const header = [
283
335
  "Run",
284
336
  "Verdict",
285
- "Invariants",
337
+ "Checks",
286
338
  "Judge",
339
+ ...(scored ? ["Score"] : []),
287
340
  "Cost",
288
341
  "Turns",
289
342
  "Duration",
290
343
  ];
291
344
  const rows = [header, header.map(() => "---")];
292
345
  for (const r of runs) {
293
- const invariantsCell = r.preflightError
294
- ? "preflight error"
295
- : r.invariants
296
- ? statusIcon(r.invariants.verdict === "pass")
297
- : "—";
298
- const judgeCell = r.preflightError
299
- ? "—"
300
- : r.judgeVerdict
301
- ? statusIcon(r.judgeVerdict.verdict === "pass")
302
- : "—";
346
+ // A preflight-failure record is the one grade-less branch in the schema.
347
+ const checksCell = r.grade ? statusIcon(r.grade.verdict === "pass") : "—";
348
+ const judgeCell = r.judgeVerdict
349
+ ? statusIcon(r.judgeVerdict.verdict === "pass")
350
+ : "—";
303
351
  rows.push([
304
352
  String(r.runIndex),
305
353
  statusIcon(r.verdict === "pass"),
306
- invariantsCell,
354
+ checksCell,
307
355
  judgeCell,
356
+ ...(scored
357
+ ? [r.score !== undefined ? Number(r.score).toFixed(4) : "—"]
358
+ : []),
308
359
  formatCost(r.costUsd),
309
360
  String(r.turns),
310
361
  formatDuration(r.durationMs),
@@ -313,42 +364,45 @@ function renderRunsTable(runs) {
313
364
  return rows.map((r) => `| ${r.join(" | ")} |`).join("\n");
314
365
  }
315
366
 
316
- function renderInvariantChecks(runs, singleRun) {
317
- const rows = collectInvariantRows(runs);
367
+ function renderChecks(runs, singleRun) {
368
+ const { rows, hasHidden } = collectCheckRows(runs);
318
369
  if (!rows.length) return null;
319
370
 
320
- const header = singleRun
321
- ? ["Check", "Result", "Message"]
322
- : ["Run", "Check", "Result", "Message"];
371
+ const header = singleRun ? ["Check"] : ["Run", "Check"];
372
+ if (hasHidden) header.push("Source");
373
+ header.push("Result", "Message");
323
374
  const lines = [
324
- "#### Invariant Checks",
375
+ "#### Checks",
325
376
  "",
326
377
  `| ${header.join(" | ")} |`,
327
378
  `| ${header.map(() => "---").join(" | ")} |`,
328
379
  ];
329
380
  for (const row of rows) {
330
- const cells = singleRun
331
- ? [row.check, row.result, row.message]
332
- : [String(row.run), row.check, row.result, row.message];
381
+ const cells = singleRun ? [row.check] : [String(row.run), row.check];
382
+ if (hasHidden) cells.push(row.source);
383
+ cells.push(row.result, row.message);
333
384
  lines.push(`| ${cells.join(" | ")} |`);
334
385
  }
335
386
  return lines.join("\n");
336
387
  }
337
388
 
338
- function collectInvariantRows(runs) {
339
- const rows = [];
340
- for (const r of runs) {
341
- if (!r.invariants?.details?.length) continue;
342
- for (const d of r.invariants.details) {
343
- rows.push({
389
+ /**
390
+ * Merge both producers' check rows per run. The Source column appears only
391
+ * when at least one row came from a hidden test suite.
392
+ */
393
+ function collectCheckRows(runs) {
394
+ const rows = runs.flatMap((r) =>
395
+ mergeRows(r.invariants?.details ?? [], r.hiddenTests?.details ?? [])
396
+ .filter((d) => d !== null && typeof d === "object")
397
+ .map((d) => ({
344
398
  run: r.runIndex,
345
399
  check: escapeCell(String(d.test ?? "(unnamed)")),
400
+ source: d.source ?? "",
346
401
  result: statusIcon(d.pass),
347
402
  message: escapeCell(String(d.message ?? "")),
348
- });
349
- }
350
- }
351
- return rows;
403
+ })),
404
+ );
405
+ return { rows, hasHidden: rows.some((row) => row.source === "tests") };
352
406
  }
353
407
 
354
408
  function renderJudgeCommentary(runs, singleRun) {
@@ -370,23 +424,36 @@ function renderJudgeCommentary(runs, singleRun) {
370
424
  }
371
425
 
372
426
  function renderErrors(runs) {
373
- const lines = [];
374
- for (const r of runs) {
375
- if (r.agentError) {
376
- lines.push(
377
- `- **Run ${r.runIndex}:** Agent error — "${escapeCell(r.agentError.message)}" (aborted: ${r.agentError.aborted})`,
378
- );
379
- }
380
- if (r.preflightError) {
381
- lines.push(
382
- `- **Run ${r.runIndex}:** Preflight error — "${escapeCell(r.preflightError.message)}" (exit ${r.preflightError.exitCode})`,
383
- );
384
- }
385
- }
427
+ const lines = runs.flatMap(runErrorLines);
386
428
  if (!lines.length) return null;
387
429
  return ["#### Errors", "", ...lines].join("\n");
388
430
  }
389
431
 
432
+ function runErrorLines(r) {
433
+ const lines = [];
434
+ if (r.grade?.malformed) {
435
+ lines.push(
436
+ `- **Run ${r.runIndex}:** ⚠️ ${r.grade.malformed} malformed check row(s) — counted as failing`,
437
+ );
438
+ }
439
+ if (r.hiddenTests?.error) {
440
+ lines.push(
441
+ `- **Run ${r.runIndex}:** Hidden-test engine error — "${escapeCell(r.hiddenTests.error)}"`,
442
+ );
443
+ }
444
+ if (r.agentError) {
445
+ lines.push(
446
+ `- **Run ${r.runIndex}:** Agent error — "${escapeCell(r.agentError.message)}" (aborted: ${r.agentError.aborted})`,
447
+ );
448
+ }
449
+ if (r.preflightError) {
450
+ lines.push(
451
+ `- **Run ${r.runIndex}:** Preflight error — "${escapeCell(r.preflightError.message)}" (exit ${r.preflightError.exitCode})`,
452
+ );
453
+ }
454
+ return lines;
455
+ }
456
+
390
457
  // ---------------------------------------------------------------------------
391
458
  // Formatting helpers
392
459
  // ---------------------------------------------------------------------------
@@ -591,6 +658,33 @@ function passAtKValue(n, c, k) {
591
658
  return Number(passing) / Number(total);
592
659
  }
593
660
 
661
+ /**
662
+ * score@k — the expected **maximum** score over k runs drawn without
663
+ * replacement from the n recorded scores; the continuous analog of pass@k.
664
+ * With scores sorted ascending s₍₁₎…s₍ₙ₎:
665
+ *
666
+ * score@k = Σ_{i=k..n} s₍ᵢ₎ · C(i−1, k−1) / C(n, k)
667
+ *
668
+ * Each term weights s₍ᵢ₎ by the probability it is the k-subset's maximum.
669
+ * Binary scores reduce exactly to the pass@k estimator (same BigInt binomial
670
+ * helper); `k > n` yields the same `{error}` value — one idiom.
671
+ * @param {number[]} scores - Effective per-record scores.
672
+ * @param {number} k
673
+ * @returns {number | {error: string}}
674
+ */
675
+ function scoreAtKValue(scores, k) {
676
+ const n = scores.length;
677
+ if (k > n) return { error: "k > n" };
678
+ const sorted = [...scores].sort((a, b) => a - b);
679
+ const total = Number(binomial(BigInt(n), BigInt(k)));
680
+ let sum = 0;
681
+ for (let i = k; i <= n; i++) {
682
+ const weight = Number(binomial(BigInt(i - 1), BigInt(k - 1))) / total;
683
+ sum += sorted[i - 1] * weight;
684
+ }
685
+ return sum;
686
+ }
687
+
594
688
  function binomial(n, k) {
595
689
  if (k < 0n || k > n) return 0n;
596
690
  if (k === 0n || k === n) return 1n;