@forwardimpact/libharness 1.3.0 → 2.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bin/fit-harness.js +18 -0
- package/bin/fit-trace.js +10 -0
- package/package.json +1 -1
- package/src/advisor.js +218 -0
- package/src/agent-runner.js +6 -0
- package/src/benchmark/grade.js +222 -0
- package/src/benchmark/hidden-tests.js +180 -0
- package/src/benchmark/invariants.js +9 -10
- package/src/benchmark/judge.js +9 -6
- package/src/benchmark/report.js +150 -56
- package/src/benchmark/result.js +43 -8
- package/src/benchmark/runner.js +99 -109
- package/src/benchmark/task-family.js +139 -26
- package/src/benchmark/trace-split.js +73 -0
- package/src/benchmark/workdir.js +6 -1
- package/src/commands/advisor-flags.js +28 -0
- package/src/commands/assert.js +72 -0
- package/src/commands/benchmark-definition.js +11 -6
- package/src/commands/benchmark-grade.js +82 -0
- package/src/commands/benchmark-report.js +12 -2
- package/src/commands/discuss.js +4 -0
- package/src/commands/facilitate.js +4 -0
- package/src/commands/run.js +162 -67
- package/src/commands/supervise.js +4 -0
- package/src/discuss-tools.js +2 -1
- package/src/discusser.js +65 -10
- package/src/facilitator.js +64 -9
- package/src/index.js +9 -0
- package/src/orchestration-toolkit.js +65 -7
- package/src/supervisor.js +72 -10
- package/src/transcript-recorder.js +94 -0
- package/src/commands/benchmark-invariants.js +0 -73
|
@@ -0,0 +1,180 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Hidden-test engine — executes a task's `tests/` overlay against the
|
|
3
|
+
* post-run agent CWD: stage each file at its mirrored path, run each check
|
|
4
|
+
* with `node --test`, convert the exit status into one check row, and
|
|
5
|
+
* restore the tree so the judge sees the workdir exactly as the agent left
|
|
6
|
+
* it.
|
|
7
|
+
*
|
|
8
|
+
* Fault attribution is the engine's contract: a stage or spawn failure (the
|
|
9
|
+
* agent deleted the scaffold) is a *failing row* — agent fault; the engine
|
|
10
|
+
* itself throwing is grader fault, which the caller records as unhealthy so
|
|
11
|
+
* a crashed grader can never mint marks.
|
|
12
|
+
*/
|
|
13
|
+
|
|
14
|
+
import { dirname, join } from "node:path";
|
|
15
|
+
|
|
16
|
+
import { buildHookEnv } from "./hook-env.js";
|
|
17
|
+
|
|
18
|
+
// Fixed per-check budget. A wedged test process runs outside the agent
|
|
19
|
+
// watchdog, so this bound is what keeps a hung hidden test from stalling the
|
|
20
|
+
// cell; the timeout row keeps the failure visible.
|
|
21
|
+
const CHECK_TIMEOUT_MS = 120_000;
|
|
22
|
+
const STDERR_TAIL_CHARS = 500;
|
|
23
|
+
|
|
24
|
+
/**
|
|
25
|
+
* Run the task's hidden test suite.
|
|
26
|
+
* @param {import("./task-family.js").Task} task
|
|
27
|
+
* @param {{cwd: string, port: number, runDir: string, familyDir?: string|null}} ctx
|
|
28
|
+
* @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
|
|
29
|
+
* @param {{timeoutMs?: number}} [opts] - Test seam for the per-check timeout.
|
|
30
|
+
* @returns {Promise<{details: object[]}>}
|
|
31
|
+
*/
|
|
32
|
+
export async function runHiddenTests(task, ctx, runtime, opts = {}) {
|
|
33
|
+
if (!runtime) throw new Error("runtime is required");
|
|
34
|
+
if (!task.tests) return { details: [] };
|
|
35
|
+
const timeoutMs = opts.timeoutMs ?? CHECK_TIMEOUT_MS;
|
|
36
|
+
const fs = runtime.fs;
|
|
37
|
+
const details = [];
|
|
38
|
+
const supportStager = newStager();
|
|
39
|
+
try {
|
|
40
|
+
for (const file of task.tests.support) {
|
|
41
|
+
await stageFile(fs, ctx.cwd, supportStager, file);
|
|
42
|
+
}
|
|
43
|
+
for (const check of task.tests.checks) {
|
|
44
|
+
details.push(await runOneCheck(task, ctx, runtime, timeoutMs, check));
|
|
45
|
+
}
|
|
46
|
+
} finally {
|
|
47
|
+
await unstage(fs, supportStager);
|
|
48
|
+
}
|
|
49
|
+
return { details };
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
/**
|
|
53
|
+
* Stage one check, run it, and restore its staging — the check's own row is
|
|
54
|
+
* the only trace it leaves. A stage failure is the agent's fault (a deleted
|
|
55
|
+
* scaffold), so it becomes a failing row rather than a throw.
|
|
56
|
+
*/
|
|
57
|
+
async function runOneCheck(task, ctx, runtime, timeoutMs, check) {
|
|
58
|
+
const stager = newStager();
|
|
59
|
+
try {
|
|
60
|
+
try {
|
|
61
|
+
await stageFile(runtime.fs, ctx.cwd, stager, check);
|
|
62
|
+
} catch (e) {
|
|
63
|
+
return checkRow(check, false, `stage failed: ${e.message}`);
|
|
64
|
+
}
|
|
65
|
+
return await spawnCheck(task, ctx, runtime, timeoutMs, check);
|
|
66
|
+
} finally {
|
|
67
|
+
await unstage(runtime.fs, stager);
|
|
68
|
+
}
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
/**
|
|
72
|
+
* Spawn `node --test <staged path>` from the agent CWD under the hook env
|
|
73
|
+
* and map the exit status onto one row. The clock timer SIGKILLs a child
|
|
74
|
+
* that outlives the per-check budget; the row fails with a timeout message.
|
|
75
|
+
*/
|
|
76
|
+
async function spawnCheck(task, ctx, runtime, timeoutMs, check) {
|
|
77
|
+
const env = buildHookEnv(runtime.proc.env, {
|
|
78
|
+
cwd: ctx.cwd,
|
|
79
|
+
port: ctx.port,
|
|
80
|
+
taskId: task.id,
|
|
81
|
+
taskDir: task.paths.taskDir,
|
|
82
|
+
hooksDir: task.paths.hooks,
|
|
83
|
+
familyDir: ctx.familyDir,
|
|
84
|
+
});
|
|
85
|
+
// An inherited test-runner context makes the child `node --test` report
|
|
86
|
+
// exit 0 even when its tests fail — a failing check would mint a passing
|
|
87
|
+
// row whenever the harness itself runs under `node --test`.
|
|
88
|
+
delete env.NODE_TEST_CONTEXT;
|
|
89
|
+
const child = runtime.subprocess.spawn("node", ["--test", check.stagePath], {
|
|
90
|
+
cwd: ctx.cwd,
|
|
91
|
+
env,
|
|
92
|
+
stdio: ["ignore", "pipe", "pipe"],
|
|
93
|
+
});
|
|
94
|
+
let timedOut = false;
|
|
95
|
+
const timer = runtime.clock.setTimeout(() => {
|
|
96
|
+
timedOut = true;
|
|
97
|
+
child.kill("SIGKILL");
|
|
98
|
+
}, timeoutMs);
|
|
99
|
+
const drainStdout = (async () => {
|
|
100
|
+
for await (const _chunk of child.stdout) {
|
|
101
|
+
// discard
|
|
102
|
+
}
|
|
103
|
+
})();
|
|
104
|
+
let stderr = "";
|
|
105
|
+
for await (const chunk of child.stderr) stderr += chunk.toString();
|
|
106
|
+
await drainStdout;
|
|
107
|
+
const exit = await child.exitCode;
|
|
108
|
+
runtime.clock.clearTimeout(timer);
|
|
109
|
+
|
|
110
|
+
if (timedOut) {
|
|
111
|
+
return checkRow(check, false, `timed out after ${timeoutMs}ms`);
|
|
112
|
+
}
|
|
113
|
+
if (exit === 0) return checkRow(check, true);
|
|
114
|
+
const tail = stderr.trim().slice(-STDERR_TAIL_CHARS);
|
|
115
|
+
return checkRow(check, false, `exit ${exit}${tail ? `: ${tail}` : ""}`);
|
|
116
|
+
}
|
|
117
|
+
|
|
118
|
+
function checkRow(check, pass, message) {
|
|
119
|
+
return {
|
|
120
|
+
test: check.name,
|
|
121
|
+
pass,
|
|
122
|
+
...(check.gate && { gate: true }),
|
|
123
|
+
...(message && { message }),
|
|
124
|
+
};
|
|
125
|
+
}
|
|
126
|
+
|
|
127
|
+
function newStager() {
|
|
128
|
+
return { staged: [], backups: [], createdDirs: [] };
|
|
129
|
+
}
|
|
130
|
+
|
|
131
|
+
/**
|
|
132
|
+
* Copy the symlink-resolved source to its mirrored path under the agent CWD,
|
|
133
|
+
* backing up a collided file's bytes and tracking every directory created so
|
|
134
|
+
* `unstage` can put the tree back exactly.
|
|
135
|
+
*/
|
|
136
|
+
async function stageFile(fs, cwd, stager, { sourcePath, stagePath }) {
|
|
137
|
+
const target = join(cwd, stagePath);
|
|
138
|
+
let collided = null;
|
|
139
|
+
try {
|
|
140
|
+
collided = await fs.readFile(target);
|
|
141
|
+
} catch {
|
|
142
|
+
// no collision
|
|
143
|
+
}
|
|
144
|
+
if (collided !== null) stager.backups.push({ target, bytes: collided });
|
|
145
|
+
await ensureParents(fs, cwd, stager, dirname(target));
|
|
146
|
+
const resolved = await fs.realpath(sourcePath);
|
|
147
|
+
await fs.copyFile(resolved, target);
|
|
148
|
+
stager.staged.push(target);
|
|
149
|
+
}
|
|
150
|
+
|
|
151
|
+
async function ensureParents(fs, cwd, stager, dir) {
|
|
152
|
+
if (dir === cwd) return;
|
|
153
|
+
try {
|
|
154
|
+
await fs.access(dir);
|
|
155
|
+
return;
|
|
156
|
+
} catch {
|
|
157
|
+
// missing — create below
|
|
158
|
+
}
|
|
159
|
+
await ensureParents(fs, cwd, stager, dirname(dir));
|
|
160
|
+
await fs.mkdir(dir);
|
|
161
|
+
stager.createdDirs.push(dir);
|
|
162
|
+
}
|
|
163
|
+
|
|
164
|
+
/**
|
|
165
|
+
* Reverse the staging: staged copies out, collided bytes back, created
|
|
166
|
+
* directories removed (deepest first — a check's own artifacts inside a
|
|
167
|
+
* created directory go with it, since that directory did not exist when the
|
|
168
|
+
* agent finished).
|
|
169
|
+
*/
|
|
170
|
+
async function unstage(fs, stager) {
|
|
171
|
+
for (const target of stager.staged) {
|
|
172
|
+
await fs.rm(target, { force: true });
|
|
173
|
+
}
|
|
174
|
+
for (const backup of stager.backups) {
|
|
175
|
+
await fs.writeFile(backup.target, backup.bytes);
|
|
176
|
+
}
|
|
177
|
+
for (const dir of [...stager.createdDirs].reverse()) {
|
|
178
|
+
await fs.rm(dir, { recursive: true, force: true });
|
|
179
|
+
}
|
|
180
|
+
}
|
|
@@ -1,7 +1,10 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* Invariants — runs `<task.paths.hooks>/invariants.sh` from the template path
|
|
3
|
-
* against the post-run agent CWD.
|
|
4
|
-
*
|
|
3
|
+
* against the post-run agent CWD. A pure collector with no verdict of its
|
|
4
|
+
* own: structured per-check rows arrive on fd 3 (`$RESULTS_FD=3`) as NDJSON
|
|
5
|
+
* and grading happens downstream over the merged rows. The exit code is
|
|
6
|
+
* script health only — nonzero means the grader itself failed, never that a
|
|
7
|
+
* check failed.
|
|
5
8
|
*
|
|
6
9
|
* Subprocess access flows through `runtime.subprocess.spawn`; the fd-3 backing
|
|
7
10
|
* store and the stderr log use the sync filesystem surface (`runtime.fsSync`) —
|
|
@@ -14,9 +17,9 @@ import { buildHookEnv } from "./hook-env.js";
|
|
|
14
17
|
|
|
15
18
|
/**
|
|
16
19
|
* @typedef {object} InvariantsResult
|
|
17
|
-
* @property {"pass" | "fail"} verdict
|
|
18
20
|
* @property {Array<object>} details
|
|
19
|
-
* @property {number} exitCode
|
|
21
|
+
* @property {number} exitCode - Script health: nonzero means the hook itself
|
|
22
|
+
* failed, never that a check failed.
|
|
20
23
|
* @property {string} [stderr] - Trimmed script stderr, present only when the
|
|
21
24
|
* script wrote to stderr. Surfaces hook failures (e.g. a missing tool) that
|
|
22
25
|
* leave `details` empty, so they read distinctly from a real invariant miss.
|
|
@@ -32,7 +35,7 @@ import { buildHookEnv } from "./hook-env.js";
|
|
|
32
35
|
export async function runInvariants(task, ctx, runtime) {
|
|
33
36
|
if (!runtime) throw new Error("runtime is required");
|
|
34
37
|
if (!task.paths.invariants) {
|
|
35
|
-
return {
|
|
38
|
+
return { details: [], exitCode: 0 };
|
|
36
39
|
}
|
|
37
40
|
const fsSync = runtime.fsSync;
|
|
38
41
|
const script = task.paths.invariants;
|
|
@@ -84,11 +87,7 @@ export async function runInvariants(task, ctx, runtime) {
|
|
|
84
87
|
const details = [];
|
|
85
88
|
parseFd3Buffer(raw, details);
|
|
86
89
|
const exitCode = typeof code === "number" ? code : -1;
|
|
87
|
-
const result = {
|
|
88
|
-
verdict: exitCode === 0 ? "pass" : "fail",
|
|
89
|
-
details,
|
|
90
|
-
exitCode,
|
|
91
|
-
};
|
|
90
|
+
const result = { details, exitCode };
|
|
92
91
|
const trimmedStderr = stderr.trim();
|
|
93
92
|
if (trimmedStderr) result.stderr = trimmedStderr;
|
|
94
93
|
return result;
|
package/src/benchmark/judge.js
CHANGED
|
@@ -9,7 +9,7 @@
|
|
|
9
9
|
* {{AGENT_INSTRUCTIONS}} — contents of agent.task.md
|
|
10
10
|
* {{AGENT_PROFILE}} — agent profile body (empty string if none)
|
|
11
11
|
* {{AGENT_TRACE_PATH}} — path to agent.ndjson
|
|
12
|
-
* {{
|
|
12
|
+
* {{GRADE_RESULT}} — JSON grade object plus the merged check rows
|
|
13
13
|
* {{SKILL_SET_HASH}} — SHA-256 from apm.lock.yaml
|
|
14
14
|
* {{TASK_ID}} — task name (directory under tasks/)
|
|
15
15
|
* {{TASK_DIR}} — agent working directory path
|
|
@@ -40,22 +40,25 @@ import { sumTraceCost } from "../cost.js";
|
|
|
40
40
|
*/
|
|
41
41
|
|
|
42
42
|
/**
|
|
43
|
-
* Run the judge over a completed task run.
|
|
43
|
+
* Run the judge over a completed task run. The judge is a binary gate over
|
|
44
|
+
* the grade's validity, never a grade itself: `gradeResult` reaches the
|
|
45
|
+
* template as evidence, and the verdict stays pass/fail.
|
|
44
46
|
* @param {import("./task-family.js").Task} task
|
|
45
47
|
* @param {import("./workdir.js").Workdir} workdir
|
|
46
|
-
* @param {
|
|
48
|
+
* @param {{verdict: string, gatesPass: boolean, score?: number, malformed?: number, rows: unknown[]}} gradeResult -
|
|
49
|
+
* The normalized grade plus the merged, source-stamped check rows.
|
|
47
50
|
* @param {{query: Function, model: string, judgeProfile?: string, profilesDir?: string, runtime: import("@forwardimpact/libutil/runtime").Runtime}} deps
|
|
48
51
|
* @param {JudgeContext} [context]
|
|
49
52
|
* @returns {Promise<JudgeVerdict>}
|
|
50
53
|
*/
|
|
51
|
-
export async function runJudge(task, workdir,
|
|
54
|
+
export async function runJudge(task, workdir, gradeResult, deps, context) {
|
|
52
55
|
const runtime = deps.runtime;
|
|
53
56
|
if (!runtime) throw new Error("runtime is required");
|
|
54
57
|
const fs = runtime.fs;
|
|
55
58
|
const template = await fs.readFile(task.paths.judge, "utf8");
|
|
56
|
-
const
|
|
59
|
+
const gradeJson = JSON.stringify(gradeResult, null, 2);
|
|
57
60
|
const taskText = template
|
|
58
|
-
.replaceAll("{{
|
|
61
|
+
.replaceAll("{{GRADE_RESULT}}", gradeJson)
|
|
59
62
|
.replaceAll("{{AGENT_TRACE_PATH}}", workdir.agentTracePath)
|
|
60
63
|
.replaceAll("{{AGENT_INSTRUCTIONS}}", context?.agentInstructions ?? "")
|
|
61
64
|
.replaceAll("{{AGENT_PROFILE}}", context?.agentProfile ?? "")
|
package/src/benchmark/report.js
CHANGED
|
@@ -15,12 +15,16 @@
|
|
|
15
15
|
import { join } from "node:path";
|
|
16
16
|
|
|
17
17
|
import { validateResultRecord } from "./result.js";
|
|
18
|
+
import { mergeRows } from "./grade.js";
|
|
18
19
|
|
|
19
20
|
/**
|
|
20
21
|
* @typedef {object} RunDetail
|
|
21
22
|
* @property {number} runIndex
|
|
22
23
|
* @property {"pass"|"fail"} verdict
|
|
23
|
-
* @property {{
|
|
24
|
+
* @property {{details: unknown[], exitCode: number}} [invariants]
|
|
25
|
+
* @property {{verdict: string, gatesPass: boolean, score?: number, malformed?: number}} [grade]
|
|
26
|
+
* @property {{details: unknown[], error?: string}} [hiddenTests]
|
|
27
|
+
* @property {number} [score] - Effective judge-zeroed score (scored tasks).
|
|
24
28
|
* @property {{verdict: string, summary: string}} [judgeVerdict]
|
|
25
29
|
* @property {number} costUsd
|
|
26
30
|
* @property {number} turns
|
|
@@ -66,6 +70,7 @@ export async function aggregate({
|
|
|
66
70
|
for (const k of kValues) passAtK[k] = passAtKValue(n, c, k);
|
|
67
71
|
|
|
68
72
|
const task = { taskId, n, c, passAtK };
|
|
73
|
+
applyScoreFields(task, group, kValues);
|
|
69
74
|
|
|
70
75
|
if (includeRuns) {
|
|
71
76
|
if (!firstRecord) firstRecord = group[0];
|
|
@@ -102,6 +107,25 @@ export async function aggregate({
|
|
|
102
107
|
return { tasks, totals };
|
|
103
108
|
}
|
|
104
109
|
|
|
110
|
+
/**
|
|
111
|
+
* Attach `meanScore` and `scoreAtK` to a scored task group. A group is
|
|
112
|
+
* scored iff any record carries an effective score; a score-less record in
|
|
113
|
+
* a scored group (a preflight failure never reached grading, or a binary
|
|
114
|
+
* run) contributes its verdict as the degenerate score — skipping it would
|
|
115
|
+
* inflate the mean exactly when the agent fails hardest. Binary groups gain
|
|
116
|
+
* neither field.
|
|
117
|
+
* @param {object} task - Mutated.
|
|
118
|
+
* @param {object[]} group
|
|
119
|
+
* @param {number[]} kValues
|
|
120
|
+
*/
|
|
121
|
+
function applyScoreFields(task, group, kValues) {
|
|
122
|
+
if (!group.some((r) => r.score !== undefined)) return;
|
|
123
|
+
const scores = group.map((r) => r.score ?? (r.verdict === "pass" ? 1 : 0));
|
|
124
|
+
task.meanScore = scores.reduce((a, b) => a + b, 0) / scores.length;
|
|
125
|
+
task.scoreAtK = {};
|
|
126
|
+
for (const k of kValues) task.scoreAtK[k] = scoreAtKValue(scores, k);
|
|
127
|
+
}
|
|
128
|
+
|
|
105
129
|
/**
|
|
106
130
|
* Build a normalized per-run detail object and accumulate duration/turn
|
|
107
131
|
* samples for median calculation. Extracted from `aggregate` to keep its
|
|
@@ -117,6 +141,9 @@ function buildRunDetail(r, acc) {
|
|
|
117
141
|
runIndex: r.runIndex,
|
|
118
142
|
verdict: r.verdict,
|
|
119
143
|
...(r.invariants && { invariants: r.invariants }),
|
|
144
|
+
...(r.grade && { grade: r.grade }),
|
|
145
|
+
...(r.hiddenTests && { hiddenTests: r.hiddenTests }),
|
|
146
|
+
...(r.score !== undefined && { score: r.score }),
|
|
120
147
|
...(r.judgeVerdict && { judgeVerdict: r.judgeVerdict }),
|
|
121
148
|
costUsd: r.costUsd ?? 0,
|
|
122
149
|
turns: r.turns ?? 0,
|
|
@@ -143,12 +170,20 @@ export function renderTextReport(report, kValues) {
|
|
|
143
170
|
}
|
|
144
171
|
|
|
145
172
|
// ---------------------------------------------------------------------------
|
|
146
|
-
// Compact report
|
|
173
|
+
// Compact report — status line + pass@k table, no per-task detail. Selected by
|
|
174
|
+
// `report --detail=compact` (aggregate without `includeRuns`); the per-shard
|
|
175
|
+
// summary uses it so a sharded run stays short while the merge job renders the
|
|
176
|
+
// full report over the combined ledger.
|
|
147
177
|
// ---------------------------------------------------------------------------
|
|
148
178
|
|
|
149
179
|
function renderCompactReport(report, kValues) {
|
|
180
|
+
const { totals } = report;
|
|
181
|
+
const passing = report.tasks.filter((t) => t.c > 0 && t.c === t.n).length;
|
|
182
|
+
const icon = statusIcon(passing === totals.tasks);
|
|
150
183
|
const lines = [
|
|
151
|
-
|
|
184
|
+
`${icon} **${passing}/${totals.tasks} tasks passing** | ${totals.runs} runs${totals.skipped ? ` | ${totals.skipped} skipped` : ""}`,
|
|
185
|
+
"",
|
|
186
|
+
renderPassAtKTable(report, kValues, hasScoredTask(report)),
|
|
152
187
|
"",
|
|
153
188
|
renderTotalsLine(report),
|
|
154
189
|
];
|
|
@@ -159,12 +194,18 @@ function renderCompactReport(report, kValues) {
|
|
|
159
194
|
// Full report
|
|
160
195
|
// ---------------------------------------------------------------------------
|
|
161
196
|
|
|
197
|
+
/** Score columns render only when the report has at least one scored task. */
|
|
198
|
+
function hasScoredTask(report) {
|
|
199
|
+
return report.tasks.some((t) => t.meanScore !== undefined);
|
|
200
|
+
}
|
|
201
|
+
|
|
162
202
|
function renderFullReport(report, kValues) {
|
|
203
|
+
const scored = hasScoredTask(report);
|
|
163
204
|
const sections = [
|
|
164
205
|
renderSummary(report),
|
|
165
206
|
"## Pass@k",
|
|
166
207
|
"",
|
|
167
|
-
renderPassAtKTable(report, kValues),
|
|
208
|
+
renderPassAtKTable(report, kValues, scored),
|
|
168
209
|
"",
|
|
169
210
|
renderTotalsLine(report),
|
|
170
211
|
"",
|
|
@@ -173,7 +214,7 @@ function renderFullReport(report, kValues) {
|
|
|
173
214
|
|
|
174
215
|
for (const task of report.tasks) {
|
|
175
216
|
sections.push("");
|
|
176
|
-
sections.push(renderTaskDetail(task));
|
|
217
|
+
sections.push(renderTaskDetail(task, scored));
|
|
177
218
|
}
|
|
178
219
|
|
|
179
220
|
return sections.join("\n");
|
|
@@ -231,16 +272,27 @@ function renderSummary(report) {
|
|
|
231
272
|
// Pass@k table (shared between compact and full)
|
|
232
273
|
// ---------------------------------------------------------------------------
|
|
233
274
|
|
|
234
|
-
function renderPassAtKTable(report, kValues) {
|
|
275
|
+
function renderPassAtKTable(report, kValues, scored) {
|
|
235
276
|
const header = ["taskId", "n", "c", ...kValues.map((k) => `pass@${k}`)];
|
|
277
|
+
if (scored) {
|
|
278
|
+
header.push("score", ...kValues.map((k) => `score@${k}`));
|
|
279
|
+
}
|
|
236
280
|
const rows = [header, header.map(() => "---")];
|
|
237
281
|
for (const t of report.tasks) {
|
|
238
|
-
|
|
282
|
+
const row = [
|
|
239
283
|
t.taskId,
|
|
240
284
|
String(t.n),
|
|
241
285
|
String(t.c),
|
|
242
286
|
...kValues.map((k) => formatPassAt(t.passAtK[k])),
|
|
243
|
-
]
|
|
287
|
+
];
|
|
288
|
+
if (scored) {
|
|
289
|
+
// Binary tasks render "—" in every score column.
|
|
290
|
+
row.push(
|
|
291
|
+
formatPassAt(t.meanScore ?? null),
|
|
292
|
+
...kValues.map((k) => formatPassAt(t.scoreAtK?.[k] ?? null)),
|
|
293
|
+
);
|
|
294
|
+
}
|
|
295
|
+
rows.push(row);
|
|
244
296
|
}
|
|
245
297
|
return rows.map((r) => `| ${r.join(" | ")} |`).join("\n");
|
|
246
298
|
}
|
|
@@ -253,7 +305,7 @@ function renderTotalsLine(report) {
|
|
|
253
305
|
// Per-task detail
|
|
254
306
|
// ---------------------------------------------------------------------------
|
|
255
307
|
|
|
256
|
-
function renderTaskDetail(task) {
|
|
308
|
+
function renderTaskDetail(task, scored) {
|
|
257
309
|
const runs = task.runs ?? [];
|
|
258
310
|
const icon = statusIcon(task.c === task.n);
|
|
259
311
|
const singleRun = runs.length === 1;
|
|
@@ -264,9 +316,9 @@ function renderTaskDetail(task) {
|
|
|
264
316
|
`${icon} **${task.c}/${task.n} runs passed**`,
|
|
265
317
|
];
|
|
266
318
|
|
|
267
|
-
lines.push("", renderRunsTable(runs));
|
|
319
|
+
lines.push("", renderRunsTable(runs, scored));
|
|
268
320
|
|
|
269
|
-
const checks =
|
|
321
|
+
const checks = renderChecks(runs, singleRun);
|
|
270
322
|
if (checks) lines.push("", checks);
|
|
271
323
|
|
|
272
324
|
const commentary = renderJudgeCommentary(runs, singleRun);
|
|
@@ -278,33 +330,32 @@ function renderTaskDetail(task) {
|
|
|
278
330
|
return lines.join("\n");
|
|
279
331
|
}
|
|
280
332
|
|
|
281
|
-
function renderRunsTable(runs) {
|
|
333
|
+
function renderRunsTable(runs, scored) {
|
|
282
334
|
const header = [
|
|
283
335
|
"Run",
|
|
284
336
|
"Verdict",
|
|
285
|
-
"
|
|
337
|
+
"Checks",
|
|
286
338
|
"Judge",
|
|
339
|
+
...(scored ? ["Score"] : []),
|
|
287
340
|
"Cost",
|
|
288
341
|
"Turns",
|
|
289
342
|
"Duration",
|
|
290
343
|
];
|
|
291
344
|
const rows = [header, header.map(() => "---")];
|
|
292
345
|
for (const r of runs) {
|
|
293
|
-
|
|
294
|
-
|
|
295
|
-
|
|
296
|
-
|
|
297
|
-
|
|
298
|
-
const judgeCell = r.preflightError
|
|
299
|
-
? "—"
|
|
300
|
-
: r.judgeVerdict
|
|
301
|
-
? statusIcon(r.judgeVerdict.verdict === "pass")
|
|
302
|
-
: "—";
|
|
346
|
+
// A preflight-failure record is the one grade-less branch in the schema.
|
|
347
|
+
const checksCell = r.grade ? statusIcon(r.grade.verdict === "pass") : "—";
|
|
348
|
+
const judgeCell = r.judgeVerdict
|
|
349
|
+
? statusIcon(r.judgeVerdict.verdict === "pass")
|
|
350
|
+
: "—";
|
|
303
351
|
rows.push([
|
|
304
352
|
String(r.runIndex),
|
|
305
353
|
statusIcon(r.verdict === "pass"),
|
|
306
|
-
|
|
354
|
+
checksCell,
|
|
307
355
|
judgeCell,
|
|
356
|
+
...(scored
|
|
357
|
+
? [r.score !== undefined ? Number(r.score).toFixed(4) : "—"]
|
|
358
|
+
: []),
|
|
308
359
|
formatCost(r.costUsd),
|
|
309
360
|
String(r.turns),
|
|
310
361
|
formatDuration(r.durationMs),
|
|
@@ -313,42 +364,45 @@ function renderRunsTable(runs) {
|
|
|
313
364
|
return rows.map((r) => `| ${r.join(" | ")} |`).join("\n");
|
|
314
365
|
}
|
|
315
366
|
|
|
316
|
-
function
|
|
317
|
-
const rows =
|
|
367
|
+
function renderChecks(runs, singleRun) {
|
|
368
|
+
const { rows, hasHidden } = collectCheckRows(runs);
|
|
318
369
|
if (!rows.length) return null;
|
|
319
370
|
|
|
320
|
-
const header = singleRun
|
|
321
|
-
|
|
322
|
-
|
|
371
|
+
const header = singleRun ? ["Check"] : ["Run", "Check"];
|
|
372
|
+
if (hasHidden) header.push("Source");
|
|
373
|
+
header.push("Result", "Message");
|
|
323
374
|
const lines = [
|
|
324
|
-
"####
|
|
375
|
+
"#### Checks",
|
|
325
376
|
"",
|
|
326
377
|
`| ${header.join(" | ")} |`,
|
|
327
378
|
`| ${header.map(() => "---").join(" | ")} |`,
|
|
328
379
|
];
|
|
329
380
|
for (const row of rows) {
|
|
330
|
-
const cells = singleRun
|
|
331
|
-
|
|
332
|
-
|
|
381
|
+
const cells = singleRun ? [row.check] : [String(row.run), row.check];
|
|
382
|
+
if (hasHidden) cells.push(row.source);
|
|
383
|
+
cells.push(row.result, row.message);
|
|
333
384
|
lines.push(`| ${cells.join(" | ")} |`);
|
|
334
385
|
}
|
|
335
386
|
return lines.join("\n");
|
|
336
387
|
}
|
|
337
388
|
|
|
338
|
-
|
|
339
|
-
|
|
340
|
-
|
|
341
|
-
|
|
342
|
-
|
|
343
|
-
|
|
389
|
+
/**
|
|
390
|
+
* Merge both producers' check rows per run. The Source column appears only
|
|
391
|
+
* when at least one row came from a hidden test suite.
|
|
392
|
+
*/
|
|
393
|
+
function collectCheckRows(runs) {
|
|
394
|
+
const rows = runs.flatMap((r) =>
|
|
395
|
+
mergeRows(r.invariants?.details ?? [], r.hiddenTests?.details ?? [])
|
|
396
|
+
.filter((d) => d !== null && typeof d === "object")
|
|
397
|
+
.map((d) => ({
|
|
344
398
|
run: r.runIndex,
|
|
345
399
|
check: escapeCell(String(d.test ?? "(unnamed)")),
|
|
400
|
+
source: d.source ?? "",
|
|
346
401
|
result: statusIcon(d.pass),
|
|
347
402
|
message: escapeCell(String(d.message ?? "")),
|
|
348
|
-
})
|
|
349
|
-
|
|
350
|
-
}
|
|
351
|
-
return rows;
|
|
403
|
+
})),
|
|
404
|
+
);
|
|
405
|
+
return { rows, hasHidden: rows.some((row) => row.source === "tests") };
|
|
352
406
|
}
|
|
353
407
|
|
|
354
408
|
function renderJudgeCommentary(runs, singleRun) {
|
|
@@ -370,23 +424,36 @@ function renderJudgeCommentary(runs, singleRun) {
|
|
|
370
424
|
}
|
|
371
425
|
|
|
372
426
|
function renderErrors(runs) {
|
|
373
|
-
const lines =
|
|
374
|
-
for (const r of runs) {
|
|
375
|
-
if (r.agentError) {
|
|
376
|
-
lines.push(
|
|
377
|
-
`- **Run ${r.runIndex}:** Agent error — "${escapeCell(r.agentError.message)}" (aborted: ${r.agentError.aborted})`,
|
|
378
|
-
);
|
|
379
|
-
}
|
|
380
|
-
if (r.preflightError) {
|
|
381
|
-
lines.push(
|
|
382
|
-
`- **Run ${r.runIndex}:** Preflight error — "${escapeCell(r.preflightError.message)}" (exit ${r.preflightError.exitCode})`,
|
|
383
|
-
);
|
|
384
|
-
}
|
|
385
|
-
}
|
|
427
|
+
const lines = runs.flatMap(runErrorLines);
|
|
386
428
|
if (!lines.length) return null;
|
|
387
429
|
return ["#### Errors", "", ...lines].join("\n");
|
|
388
430
|
}
|
|
389
431
|
|
|
432
|
+
function runErrorLines(r) {
|
|
433
|
+
const lines = [];
|
|
434
|
+
if (r.grade?.malformed) {
|
|
435
|
+
lines.push(
|
|
436
|
+
`- **Run ${r.runIndex}:** ⚠️ ${r.grade.malformed} malformed check row(s) — counted as failing`,
|
|
437
|
+
);
|
|
438
|
+
}
|
|
439
|
+
if (r.hiddenTests?.error) {
|
|
440
|
+
lines.push(
|
|
441
|
+
`- **Run ${r.runIndex}:** Hidden-test engine error — "${escapeCell(r.hiddenTests.error)}"`,
|
|
442
|
+
);
|
|
443
|
+
}
|
|
444
|
+
if (r.agentError) {
|
|
445
|
+
lines.push(
|
|
446
|
+
`- **Run ${r.runIndex}:** Agent error — "${escapeCell(r.agentError.message)}" (aborted: ${r.agentError.aborted})`,
|
|
447
|
+
);
|
|
448
|
+
}
|
|
449
|
+
if (r.preflightError) {
|
|
450
|
+
lines.push(
|
|
451
|
+
`- **Run ${r.runIndex}:** Preflight error — "${escapeCell(r.preflightError.message)}" (exit ${r.preflightError.exitCode})`,
|
|
452
|
+
);
|
|
453
|
+
}
|
|
454
|
+
return lines;
|
|
455
|
+
}
|
|
456
|
+
|
|
390
457
|
// ---------------------------------------------------------------------------
|
|
391
458
|
// Formatting helpers
|
|
392
459
|
// ---------------------------------------------------------------------------
|
|
@@ -591,6 +658,33 @@ function passAtKValue(n, c, k) {
|
|
|
591
658
|
return Number(passing) / Number(total);
|
|
592
659
|
}
|
|
593
660
|
|
|
661
|
+
/**
|
|
662
|
+
* score@k — the expected **maximum** score over k runs drawn without
|
|
663
|
+
* replacement from the n recorded scores; the continuous analog of pass@k.
|
|
664
|
+
* With scores sorted ascending s₍₁₎…s₍ₙ₎:
|
|
665
|
+
*
|
|
666
|
+
* score@k = Σ_{i=k..n} s₍ᵢ₎ · C(i−1, k−1) / C(n, k)
|
|
667
|
+
*
|
|
668
|
+
* Each term weights s₍ᵢ₎ by the probability it is the k-subset's maximum.
|
|
669
|
+
* Binary scores reduce exactly to the pass@k estimator (same BigInt binomial
|
|
670
|
+
* helper); `k > n` yields the same `{error}` value — one idiom.
|
|
671
|
+
* @param {number[]} scores - Effective per-record scores.
|
|
672
|
+
* @param {number} k
|
|
673
|
+
* @returns {number | {error: string}}
|
|
674
|
+
*/
|
|
675
|
+
function scoreAtKValue(scores, k) {
|
|
676
|
+
const n = scores.length;
|
|
677
|
+
if (k > n) return { error: "k > n" };
|
|
678
|
+
const sorted = [...scores].sort((a, b) => a - b);
|
|
679
|
+
const total = Number(binomial(BigInt(n), BigInt(k)));
|
|
680
|
+
let sum = 0;
|
|
681
|
+
for (let i = k; i <= n; i++) {
|
|
682
|
+
const weight = Number(binomial(BigInt(i - 1), BigInt(k - 1))) / total;
|
|
683
|
+
sum += sorted[i - 1] * weight;
|
|
684
|
+
}
|
|
685
|
+
return sum;
|
|
686
|
+
}
|
|
687
|
+
|
|
594
688
|
function binomial(n, k) {
|
|
595
689
|
if (k < 0n || k > n) return 0n;
|
|
596
690
|
if (k === 0n || k === n) return 1n;
|