@forwardimpact/libharness 1.4.0 → 3.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +10 -10
- package/package.json +14 -12
- package/src/agent-runner.js +3 -3
- package/src/benchmark/grade.js +222 -0
- package/src/benchmark/hidden-tests.js +180 -0
- package/src/benchmark/invariants.js +9 -10
- package/src/benchmark/judge.js +9 -6
- package/src/benchmark/report.js +141 -55
- package/src/benchmark/result.js +43 -8
- package/src/benchmark/runner.js +99 -109
- package/src/benchmark/task-family.js +140 -27
- package/src/benchmark/trace-split.js +73 -0
- package/src/benchmark/workdir.js +6 -1
- package/src/claude-code-executable.js +1 -1
- package/src/commands/assert.js +72 -0
- package/src/commands/benchmark-definition.js +15 -15
- package/src/commands/benchmark-grade.js +82 -0
- package/src/commands/benchmark-report.js +1 -1
- package/src/commands/benchmark-run.js +1 -1
- package/src/commands/by-discussion.js +1 -1
- package/src/commands/facilitate.js +1 -1
- package/src/commands/output.js +1 -1
- package/src/commands/run.js +1 -1
- package/src/commands/scan-logs.js +2 -2
- package/src/commands/selfedit.js +124 -0
- package/src/commands/supervise.js +2 -2
- package/src/commands/tee.js +1 -1
- package/src/commands/trace.js +1 -1
- package/src/cost.js +1 -1
- package/src/trace-collector.js +1 -1
- package/src/trace-multi.js +1 -1
- package/src/trace-render.js +1 -1
- package/bin/fit-benchmark.js +0 -44
- package/bin/fit-harness.js +0 -412
- package/bin/fit-selfedit.js +0 -165
- package/bin/fit-trace.js +0 -510
- package/src/commands/benchmark-invariants.js +0 -73
package/src/benchmark/report.js
CHANGED
|
@@ -15,12 +15,16 @@
|
|
|
15
15
|
import { join } from "node:path";
|
|
16
16
|
|
|
17
17
|
import { validateResultRecord } from "./result.js";
|
|
18
|
+
import { mergeRows } from "./grade.js";
|
|
18
19
|
|
|
19
20
|
/**
|
|
20
21
|
* @typedef {object} RunDetail
|
|
21
22
|
* @property {number} runIndex
|
|
22
23
|
* @property {"pass"|"fail"} verdict
|
|
23
|
-
* @property {{
|
|
24
|
+
* @property {{details: unknown[], exitCode: number}} [invariants]
|
|
25
|
+
* @property {{verdict: string, gatesPass: boolean, score?: number, malformed?: number}} [grade]
|
|
26
|
+
* @property {{details: unknown[], error?: string}} [hiddenTests]
|
|
27
|
+
* @property {number} [score] - Effective judge-zeroed score (scored tasks).
|
|
24
28
|
* @property {{verdict: string, summary: string}} [judgeVerdict]
|
|
25
29
|
* @property {number} costUsd
|
|
26
30
|
* @property {number} turns
|
|
@@ -66,6 +70,7 @@ export async function aggregate({
|
|
|
66
70
|
for (const k of kValues) passAtK[k] = passAtKValue(n, c, k);
|
|
67
71
|
|
|
68
72
|
const task = { taskId, n, c, passAtK };
|
|
73
|
+
applyScoreFields(task, group, kValues);
|
|
69
74
|
|
|
70
75
|
if (includeRuns) {
|
|
71
76
|
if (!firstRecord) firstRecord = group[0];
|
|
@@ -102,6 +107,25 @@ export async function aggregate({
|
|
|
102
107
|
return { tasks, totals };
|
|
103
108
|
}
|
|
104
109
|
|
|
110
|
+
/**
|
|
111
|
+
* Attach `meanScore` and `scoreAtK` to a scored task group. A group is
|
|
112
|
+
* scored iff any record carries an effective score; a score-less record in
|
|
113
|
+
* a scored group (a preflight failure never reached grading, or a binary
|
|
114
|
+
* run) contributes its verdict as the degenerate score — skipping it would
|
|
115
|
+
* inflate the mean exactly when the agent fails hardest. Binary groups gain
|
|
116
|
+
* neither field.
|
|
117
|
+
* @param {object} task - Mutated.
|
|
118
|
+
* @param {object[]} group
|
|
119
|
+
* @param {number[]} kValues
|
|
120
|
+
*/
|
|
121
|
+
function applyScoreFields(task, group, kValues) {
|
|
122
|
+
if (!group.some((r) => r.score !== undefined)) return;
|
|
123
|
+
const scores = group.map((r) => r.score ?? (r.verdict === "pass" ? 1 : 0));
|
|
124
|
+
task.meanScore = scores.reduce((a, b) => a + b, 0) / scores.length;
|
|
125
|
+
task.scoreAtK = {};
|
|
126
|
+
for (const k of kValues) task.scoreAtK[k] = scoreAtKValue(scores, k);
|
|
127
|
+
}
|
|
128
|
+
|
|
105
129
|
/**
|
|
106
130
|
* Build a normalized per-run detail object and accumulate duration/turn
|
|
107
131
|
* samples for median calculation. Extracted from `aggregate` to keep its
|
|
@@ -117,6 +141,9 @@ function buildRunDetail(r, acc) {
|
|
|
117
141
|
runIndex: r.runIndex,
|
|
118
142
|
verdict: r.verdict,
|
|
119
143
|
...(r.invariants && { invariants: r.invariants }),
|
|
144
|
+
...(r.grade && { grade: r.grade }),
|
|
145
|
+
...(r.hiddenTests && { hiddenTests: r.hiddenTests }),
|
|
146
|
+
...(r.score !== undefined && { score: r.score }),
|
|
120
147
|
...(r.judgeVerdict && { judgeVerdict: r.judgeVerdict }),
|
|
121
148
|
costUsd: r.costUsd ?? 0,
|
|
122
149
|
turns: r.turns ?? 0,
|
|
@@ -156,7 +183,7 @@ function renderCompactReport(report, kValues) {
|
|
|
156
183
|
const lines = [
|
|
157
184
|
`${icon} **${passing}/${totals.tasks} tasks passing** | ${totals.runs} runs${totals.skipped ? ` | ${totals.skipped} skipped` : ""}`,
|
|
158
185
|
"",
|
|
159
|
-
renderPassAtKTable(report, kValues),
|
|
186
|
+
renderPassAtKTable(report, kValues, hasScoredTask(report)),
|
|
160
187
|
"",
|
|
161
188
|
renderTotalsLine(report),
|
|
162
189
|
];
|
|
@@ -167,12 +194,18 @@ function renderCompactReport(report, kValues) {
|
|
|
167
194
|
// Full report
|
|
168
195
|
// ---------------------------------------------------------------------------
|
|
169
196
|
|
|
197
|
+
/** Score columns render only when the report has at least one scored task. */
|
|
198
|
+
function hasScoredTask(report) {
|
|
199
|
+
return report.tasks.some((t) => t.meanScore !== undefined);
|
|
200
|
+
}
|
|
201
|
+
|
|
170
202
|
function renderFullReport(report, kValues) {
|
|
203
|
+
const scored = hasScoredTask(report);
|
|
171
204
|
const sections = [
|
|
172
205
|
renderSummary(report),
|
|
173
206
|
"## Pass@k",
|
|
174
207
|
"",
|
|
175
|
-
renderPassAtKTable(report, kValues),
|
|
208
|
+
renderPassAtKTable(report, kValues, scored),
|
|
176
209
|
"",
|
|
177
210
|
renderTotalsLine(report),
|
|
178
211
|
"",
|
|
@@ -181,7 +214,7 @@ function renderFullReport(report, kValues) {
|
|
|
181
214
|
|
|
182
215
|
for (const task of report.tasks) {
|
|
183
216
|
sections.push("");
|
|
184
|
-
sections.push(renderTaskDetail(task));
|
|
217
|
+
sections.push(renderTaskDetail(task, scored));
|
|
185
218
|
}
|
|
186
219
|
|
|
187
220
|
return sections.join("\n");
|
|
@@ -239,16 +272,27 @@ function renderSummary(report) {
|
|
|
239
272
|
// Pass@k table (shared between compact and full)
|
|
240
273
|
// ---------------------------------------------------------------------------
|
|
241
274
|
|
|
242
|
-
function renderPassAtKTable(report, kValues) {
|
|
275
|
+
function renderPassAtKTable(report, kValues, scored) {
|
|
243
276
|
const header = ["taskId", "n", "c", ...kValues.map((k) => `pass@${k}`)];
|
|
277
|
+
if (scored) {
|
|
278
|
+
header.push("score", ...kValues.map((k) => `score@${k}`));
|
|
279
|
+
}
|
|
244
280
|
const rows = [header, header.map(() => "---")];
|
|
245
281
|
for (const t of report.tasks) {
|
|
246
|
-
|
|
282
|
+
const row = [
|
|
247
283
|
t.taskId,
|
|
248
284
|
String(t.n),
|
|
249
285
|
String(t.c),
|
|
250
286
|
...kValues.map((k) => formatPassAt(t.passAtK[k])),
|
|
251
|
-
]
|
|
287
|
+
];
|
|
288
|
+
if (scored) {
|
|
289
|
+
// Binary tasks render "—" in every score column.
|
|
290
|
+
row.push(
|
|
291
|
+
formatPassAt(t.meanScore ?? null),
|
|
292
|
+
...kValues.map((k) => formatPassAt(t.scoreAtK?.[k] ?? null)),
|
|
293
|
+
);
|
|
294
|
+
}
|
|
295
|
+
rows.push(row);
|
|
252
296
|
}
|
|
253
297
|
return rows.map((r) => `| ${r.join(" | ")} |`).join("\n");
|
|
254
298
|
}
|
|
@@ -261,7 +305,7 @@ function renderTotalsLine(report) {
|
|
|
261
305
|
// Per-task detail
|
|
262
306
|
// ---------------------------------------------------------------------------
|
|
263
307
|
|
|
264
|
-
function renderTaskDetail(task) {
|
|
308
|
+
function renderTaskDetail(task, scored) {
|
|
265
309
|
const runs = task.runs ?? [];
|
|
266
310
|
const icon = statusIcon(task.c === task.n);
|
|
267
311
|
const singleRun = runs.length === 1;
|
|
@@ -272,9 +316,9 @@ function renderTaskDetail(task) {
|
|
|
272
316
|
`${icon} **${task.c}/${task.n} runs passed**`,
|
|
273
317
|
];
|
|
274
318
|
|
|
275
|
-
lines.push("", renderRunsTable(runs));
|
|
319
|
+
lines.push("", renderRunsTable(runs, scored));
|
|
276
320
|
|
|
277
|
-
const checks =
|
|
321
|
+
const checks = renderChecks(runs, singleRun);
|
|
278
322
|
if (checks) lines.push("", checks);
|
|
279
323
|
|
|
280
324
|
const commentary = renderJudgeCommentary(runs, singleRun);
|
|
@@ -286,33 +330,32 @@ function renderTaskDetail(task) {
|
|
|
286
330
|
return lines.join("\n");
|
|
287
331
|
}
|
|
288
332
|
|
|
289
|
-
function renderRunsTable(runs) {
|
|
333
|
+
function renderRunsTable(runs, scored) {
|
|
290
334
|
const header = [
|
|
291
335
|
"Run",
|
|
292
336
|
"Verdict",
|
|
293
|
-
"
|
|
337
|
+
"Checks",
|
|
294
338
|
"Judge",
|
|
339
|
+
...(scored ? ["Score"] : []),
|
|
295
340
|
"Cost",
|
|
296
341
|
"Turns",
|
|
297
342
|
"Duration",
|
|
298
343
|
];
|
|
299
344
|
const rows = [header, header.map(() => "---")];
|
|
300
345
|
for (const r of runs) {
|
|
301
|
-
|
|
302
|
-
|
|
303
|
-
|
|
304
|
-
|
|
305
|
-
|
|
306
|
-
const judgeCell = r.preflightError
|
|
307
|
-
? "—"
|
|
308
|
-
: r.judgeVerdict
|
|
309
|
-
? statusIcon(r.judgeVerdict.verdict === "pass")
|
|
310
|
-
: "—";
|
|
346
|
+
// A preflight-failure record is the one grade-less branch in the schema.
|
|
347
|
+
const checksCell = r.grade ? statusIcon(r.grade.verdict === "pass") : "—";
|
|
348
|
+
const judgeCell = r.judgeVerdict
|
|
349
|
+
? statusIcon(r.judgeVerdict.verdict === "pass")
|
|
350
|
+
: "—";
|
|
311
351
|
rows.push([
|
|
312
352
|
String(r.runIndex),
|
|
313
353
|
statusIcon(r.verdict === "pass"),
|
|
314
|
-
|
|
354
|
+
checksCell,
|
|
315
355
|
judgeCell,
|
|
356
|
+
...(scored
|
|
357
|
+
? [r.score !== undefined ? Number(r.score).toFixed(4) : "—"]
|
|
358
|
+
: []),
|
|
316
359
|
formatCost(r.costUsd),
|
|
317
360
|
String(r.turns),
|
|
318
361
|
formatDuration(r.durationMs),
|
|
@@ -321,42 +364,45 @@ function renderRunsTable(runs) {
|
|
|
321
364
|
return rows.map((r) => `| ${r.join(" | ")} |`).join("\n");
|
|
322
365
|
}
|
|
323
366
|
|
|
324
|
-
function
|
|
325
|
-
const rows =
|
|
367
|
+
function renderChecks(runs, singleRun) {
|
|
368
|
+
const { rows, hasHidden } = collectCheckRows(runs);
|
|
326
369
|
if (!rows.length) return null;
|
|
327
370
|
|
|
328
|
-
const header = singleRun
|
|
329
|
-
|
|
330
|
-
|
|
371
|
+
const header = singleRun ? ["Check"] : ["Run", "Check"];
|
|
372
|
+
if (hasHidden) header.push("Source");
|
|
373
|
+
header.push("Result", "Message");
|
|
331
374
|
const lines = [
|
|
332
|
-
"####
|
|
375
|
+
"#### Checks",
|
|
333
376
|
"",
|
|
334
377
|
`| ${header.join(" | ")} |`,
|
|
335
378
|
`| ${header.map(() => "---").join(" | ")} |`,
|
|
336
379
|
];
|
|
337
380
|
for (const row of rows) {
|
|
338
|
-
const cells = singleRun
|
|
339
|
-
|
|
340
|
-
|
|
381
|
+
const cells = singleRun ? [row.check] : [String(row.run), row.check];
|
|
382
|
+
if (hasHidden) cells.push(row.source);
|
|
383
|
+
cells.push(row.result, row.message);
|
|
341
384
|
lines.push(`| ${cells.join(" | ")} |`);
|
|
342
385
|
}
|
|
343
386
|
return lines.join("\n");
|
|
344
387
|
}
|
|
345
388
|
|
|
346
|
-
|
|
347
|
-
|
|
348
|
-
|
|
349
|
-
|
|
350
|
-
|
|
351
|
-
|
|
389
|
+
/**
|
|
390
|
+
* Merge both producers' check rows per run. The Source column appears only
|
|
391
|
+
* when at least one row came from a hidden test suite.
|
|
392
|
+
*/
|
|
393
|
+
function collectCheckRows(runs) {
|
|
394
|
+
const rows = runs.flatMap((r) =>
|
|
395
|
+
mergeRows(r.invariants?.details ?? [], r.hiddenTests?.details ?? [])
|
|
396
|
+
.filter((d) => d !== null && typeof d === "object")
|
|
397
|
+
.map((d) => ({
|
|
352
398
|
run: r.runIndex,
|
|
353
399
|
check: escapeCell(String(d.test ?? "(unnamed)")),
|
|
400
|
+
source: d.source ?? "",
|
|
354
401
|
result: statusIcon(d.pass),
|
|
355
402
|
message: escapeCell(String(d.message ?? "")),
|
|
356
|
-
})
|
|
357
|
-
|
|
358
|
-
}
|
|
359
|
-
return rows;
|
|
403
|
+
})),
|
|
404
|
+
);
|
|
405
|
+
return { rows, hasHidden: rows.some((row) => row.source === "tests") };
|
|
360
406
|
}
|
|
361
407
|
|
|
362
408
|
function renderJudgeCommentary(runs, singleRun) {
|
|
@@ -378,23 +424,36 @@ function renderJudgeCommentary(runs, singleRun) {
|
|
|
378
424
|
}
|
|
379
425
|
|
|
380
426
|
function renderErrors(runs) {
|
|
381
|
-
const lines =
|
|
382
|
-
for (const r of runs) {
|
|
383
|
-
if (r.agentError) {
|
|
384
|
-
lines.push(
|
|
385
|
-
`- **Run ${r.runIndex}:** Agent error — "${escapeCell(r.agentError.message)}" (aborted: ${r.agentError.aborted})`,
|
|
386
|
-
);
|
|
387
|
-
}
|
|
388
|
-
if (r.preflightError) {
|
|
389
|
-
lines.push(
|
|
390
|
-
`- **Run ${r.runIndex}:** Preflight error — "${escapeCell(r.preflightError.message)}" (exit ${r.preflightError.exitCode})`,
|
|
391
|
-
);
|
|
392
|
-
}
|
|
393
|
-
}
|
|
427
|
+
const lines = runs.flatMap(runErrorLines);
|
|
394
428
|
if (!lines.length) return null;
|
|
395
429
|
return ["#### Errors", "", ...lines].join("\n");
|
|
396
430
|
}
|
|
397
431
|
|
|
432
|
+
function runErrorLines(r) {
|
|
433
|
+
const lines = [];
|
|
434
|
+
if (r.grade?.malformed) {
|
|
435
|
+
lines.push(
|
|
436
|
+
`- **Run ${r.runIndex}:** ⚠️ ${r.grade.malformed} malformed check row(s) — counted as failing`,
|
|
437
|
+
);
|
|
438
|
+
}
|
|
439
|
+
if (r.hiddenTests?.error) {
|
|
440
|
+
lines.push(
|
|
441
|
+
`- **Run ${r.runIndex}:** Hidden-test engine error — "${escapeCell(r.hiddenTests.error)}"`,
|
|
442
|
+
);
|
|
443
|
+
}
|
|
444
|
+
if (r.agentError) {
|
|
445
|
+
lines.push(
|
|
446
|
+
`- **Run ${r.runIndex}:** Agent error — "${escapeCell(r.agentError.message)}" (aborted: ${r.agentError.aborted})`,
|
|
447
|
+
);
|
|
448
|
+
}
|
|
449
|
+
if (r.preflightError) {
|
|
450
|
+
lines.push(
|
|
451
|
+
`- **Run ${r.runIndex}:** Preflight error — "${escapeCell(r.preflightError.message)}" (exit ${r.preflightError.exitCode})`,
|
|
452
|
+
);
|
|
453
|
+
}
|
|
454
|
+
return lines;
|
|
455
|
+
}
|
|
456
|
+
|
|
398
457
|
// ---------------------------------------------------------------------------
|
|
399
458
|
// Formatting helpers
|
|
400
459
|
// ---------------------------------------------------------------------------
|
|
@@ -599,6 +658,33 @@ function passAtKValue(n, c, k) {
|
|
|
599
658
|
return Number(passing) / Number(total);
|
|
600
659
|
}
|
|
601
660
|
|
|
661
|
+
/**
|
|
662
|
+
* score@k — the expected **maximum** score over k runs drawn without
|
|
663
|
+
* replacement from the n recorded scores; the continuous analog of pass@k.
|
|
664
|
+
* With scores sorted ascending s₍₁₎…s₍ₙ₎:
|
|
665
|
+
*
|
|
666
|
+
* score@k = Σ_{i=k..n} s₍ᵢ₎ · C(i−1, k−1) / C(n, k)
|
|
667
|
+
*
|
|
668
|
+
* Each term weights s₍ᵢ₎ by the probability it is the k-subset's maximum.
|
|
669
|
+
* Binary scores reduce exactly to the pass@k estimator (same BigInt binomial
|
|
670
|
+
* helper); `k > n` yields the same `{error}` value — one idiom.
|
|
671
|
+
* @param {number[]} scores - Effective per-record scores.
|
|
672
|
+
* @param {number} k
|
|
673
|
+
* @returns {number | {error: string}}
|
|
674
|
+
*/
|
|
675
|
+
function scoreAtKValue(scores, k) {
|
|
676
|
+
const n = scores.length;
|
|
677
|
+
if (k > n) return { error: "k > n" };
|
|
678
|
+
const sorted = [...scores].sort((a, b) => a - b);
|
|
679
|
+
const total = Number(binomial(BigInt(n), BigInt(k)));
|
|
680
|
+
let sum = 0;
|
|
681
|
+
for (let i = k; i <= n; i++) {
|
|
682
|
+
const weight = Number(binomial(BigInt(i - 1), BigInt(k - 1))) / total;
|
|
683
|
+
sum += sorted[i - 1] * weight;
|
|
684
|
+
}
|
|
685
|
+
return sum;
|
|
686
|
+
}
|
|
687
|
+
|
|
602
688
|
function binomial(n, k) {
|
|
603
689
|
if (k < 0n || k > n) return 0n;
|
|
604
690
|
if (k === 0n || k === n) return 1n;
|
package/src/benchmark/result.js
CHANGED
|
@@ -3,10 +3,14 @@
|
|
|
3
3
|
*
|
|
4
4
|
* Two schemas live here:
|
|
5
5
|
* - RESULT_RECORD_SCHEMA — one record per (task, runIndex) from a full
|
|
6
|
-
* benchmark run. Has a happy branch (
|
|
7
|
-
* pre-flight-failure branch (
|
|
8
|
-
* -
|
|
9
|
-
*
|
|
6
|
+
* benchmark run. Has a happy branch (grade + collectors + judge present)
|
|
7
|
+
* and a pre-flight-failure branch (grade/judgeVerdict/submission absent).
|
|
8
|
+
* - GRADE_RECORD_SCHEMA — narrower output of `benchmark-grade`: ad-hoc
|
|
9
|
+
* grading without a full lifecycle.
|
|
10
|
+
*
|
|
11
|
+
* The check rows are the authoritative grading channel: the happy branch
|
|
12
|
+
* requires a `grade` object, so a pre-break record fails validation rather
|
|
13
|
+
* than rendering under semantics it never carried.
|
|
10
14
|
*
|
|
11
15
|
* Validation is throw-on-mismatch so the runner can wrap every JSONL append
|
|
12
16
|
* in a guard and reject schema drift at write time.
|
|
@@ -17,12 +21,27 @@ import { z } from "zod";
|
|
|
17
21
|
const VERDICT_ENUM = z.enum(["pass", "fail"]);
|
|
18
22
|
|
|
19
23
|
const INVARIANTS_SHAPE = z.object({
|
|
20
|
-
verdict: VERDICT_ENUM,
|
|
21
24
|
details: z.array(z.unknown()),
|
|
22
25
|
exitCode: z.number().int(),
|
|
23
26
|
stderr: z.string().optional(),
|
|
24
27
|
});
|
|
25
28
|
|
|
29
|
+
/**
|
|
30
|
+
* The normalized grading projection: `score` appears only on scored tasks,
|
|
31
|
+
* `malformed` only when at least one row was malformed.
|
|
32
|
+
*/
|
|
33
|
+
const GRADE_SHAPE = z.object({
|
|
34
|
+
verdict: VERDICT_ENUM,
|
|
35
|
+
gatesPass: z.boolean(),
|
|
36
|
+
score: z.number().min(0).max(1).optional(),
|
|
37
|
+
malformed: z.number().int().min(1).optional(),
|
|
38
|
+
});
|
|
39
|
+
|
|
40
|
+
const HIDDEN_TESTS_SHAPE = z.object({
|
|
41
|
+
details: z.array(z.unknown()),
|
|
42
|
+
error: z.string().optional(),
|
|
43
|
+
});
|
|
44
|
+
|
|
26
45
|
const JUDGE_VERDICT_SHAPE = z.object({
|
|
27
46
|
verdict: VERDICT_ENUM,
|
|
28
47
|
summary: z.string(),
|
|
@@ -77,6 +96,11 @@ const AGENT_ERROR_SHAPE = z.object({
|
|
|
77
96
|
const HAPPY_RECORD = z.object({
|
|
78
97
|
...COMMON_FIELDS,
|
|
79
98
|
invariants: INVARIANTS_SHAPE,
|
|
99
|
+
grade: GRADE_SHAPE,
|
|
100
|
+
hiddenTests: HIDDEN_TESTS_SHAPE.optional(),
|
|
101
|
+
// The effective, judge-zeroed score `report` aggregates — present only on
|
|
102
|
+
// scored tasks.
|
|
103
|
+
score: z.number().min(0).max(1).optional(),
|
|
80
104
|
submission: z.string(),
|
|
81
105
|
judgeVerdict: JUDGE_VERDICT_SHAPE.optional(),
|
|
82
106
|
agentTracePath: z.string(),
|
|
@@ -97,6 +121,9 @@ const PREFLIGHT_RECORD = z.object({
|
|
|
97
121
|
supervisorTracePath: z.string(),
|
|
98
122
|
judgeTracePath: z.string(),
|
|
99
123
|
invariants: z.undefined().optional(),
|
|
124
|
+
grade: z.undefined().optional(),
|
|
125
|
+
hiddenTests: z.undefined().optional(),
|
|
126
|
+
score: z.undefined().optional(),
|
|
100
127
|
submission: z.undefined().optional(),
|
|
101
128
|
judgeVerdict: z.undefined().optional(),
|
|
102
129
|
agentError: z.undefined().optional(),
|
|
@@ -104,9 +131,17 @@ const PREFLIGHT_RECORD = z.object({
|
|
|
104
131
|
|
|
105
132
|
export const RESULT_RECORD_SCHEMA = z.union([HAPPY_RECORD, PREFLIGHT_RECORD]);
|
|
106
133
|
|
|
107
|
-
export const
|
|
134
|
+
export const GRADE_RECORD_SCHEMA = z.object({
|
|
108
135
|
taskId: z.string().min(1),
|
|
136
|
+
// Unlike the happy result record — where `grade.score` is the raw
|
|
137
|
+
// weighted fraction and the effective (zeroed) value lives on the
|
|
138
|
+
// top-level `score` — this record has no second score field, so its
|
|
139
|
+
// `grade.score` carries the effective health/gate-zeroed value.
|
|
140
|
+
grade: GRADE_SHAPE,
|
|
109
141
|
invariants: INVARIANTS_SHAPE,
|
|
142
|
+
hiddenTests: HIDDEN_TESTS_SHAPE.optional(),
|
|
143
|
+
// Mirrors the invariants script's exit for diagnosis; the graded verdict
|
|
144
|
+
// is what drives the command's process exit.
|
|
110
145
|
exitCode: z.number().int(),
|
|
111
146
|
});
|
|
112
147
|
|
|
@@ -122,6 +157,6 @@ export function validateResultRecord(record) {
|
|
|
122
157
|
* Throw on schema mismatch.
|
|
123
158
|
* @param {object} record
|
|
124
159
|
*/
|
|
125
|
-
export function
|
|
126
|
-
|
|
160
|
+
export function validateGradeRecord(record) {
|
|
161
|
+
GRADE_RECORD_SCHEMA.parse(record);
|
|
127
162
|
}
|