@forwardimpact/libharness 1.4.0 → 2.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bin/fit-trace.js +10 -0
- package/package.json +1 -1
- package/src/benchmark/grade.js +222 -0
- package/src/benchmark/hidden-tests.js +180 -0
- package/src/benchmark/invariants.js +9 -10
- package/src/benchmark/judge.js +9 -6
- package/src/benchmark/report.js +141 -55
- package/src/benchmark/result.js +43 -8
- package/src/benchmark/runner.js +99 -109
- package/src/benchmark/task-family.js +139 -26
- package/src/benchmark/trace-split.js +73 -0
- package/src/benchmark/workdir.js +6 -1
- package/src/commands/assert.js +72 -0
- package/src/commands/benchmark-definition.js +6 -6
- package/src/commands/benchmark-grade.js +82 -0
- package/src/commands/benchmark-invariants.js +0 -73
package/src/benchmark/runner.js
CHANGED
|
@@ -4,8 +4,10 @@
|
|
|
4
4
|
* Phases per (task, runIndex):
|
|
5
5
|
* 1. WorkdirManager.start → seed CWD + run pre-flight probe
|
|
6
6
|
* 2. Supervisor session (agent + supervisor) → produce traces + submission
|
|
7
|
-
* 3. Invariants
|
|
8
|
-
*
|
|
7
|
+
* 3. Invariants collector + hidden-test engine → merged check rows,
|
|
8
|
+
* graded by `gradeChecks` (rows are authoritative; script exit is
|
|
9
|
+
* grader health only)
|
|
10
|
+
* 4. Judge.runJudge → Conclude-driven binary gate mapped to pass/fail
|
|
9
11
|
* 5. WorkdirManager.teardown → process-group cleanup
|
|
10
12
|
*
|
|
11
13
|
* Cells run with bounded in-process concurrency (`CellScheduler`); `run()`
|
|
@@ -17,7 +19,6 @@
|
|
|
17
19
|
* stream.
|
|
18
20
|
*/
|
|
19
21
|
|
|
20
|
-
import { createInterface } from "node:readline";
|
|
21
22
|
import { join, resolve as resolvePath } from "node:path";
|
|
22
23
|
|
|
23
24
|
import { DEFAULT_ENV_ALLOWLIST, createRedactor } from "../redaction.js";
|
|
@@ -28,7 +29,10 @@ import { installNpm as defaultInstallNpm } from "./npm-installer.js";
|
|
|
28
29
|
import { runJudge } from "./judge.js";
|
|
29
30
|
import { validateResultRecord } from "./result.js";
|
|
30
31
|
import { runInvariants } from "./invariants.js";
|
|
32
|
+
import { runHiddenTests } from "./hidden-tests.js";
|
|
33
|
+
import { runProducersAndGrade } from "./grade.js";
|
|
31
34
|
import { assertJudgeProfileStaged, loadTaskFamily } from "./task-family.js";
|
|
35
|
+
import { splitAndSummarize } from "./trace-split.js";
|
|
32
36
|
import { createWorkdirManager } from "./workdir.js";
|
|
33
37
|
import { CellScheduler } from "./scheduler.js";
|
|
34
38
|
|
|
@@ -82,9 +86,13 @@ export class BenchmarkRunner {
|
|
|
82
86
|
* threaded into the installers, workdir manager, invariants, and judge.
|
|
83
87
|
* @param {Function} [opts.runInvariants] - Test seam: replaces `runInvariants`.
|
|
84
88
|
* Same contract as `runInvariants(task, ctx, runtime)`. Internal testing only.
|
|
89
|
+
* @param {Function} [opts.runHiddenTests] - Test seam: replaces
|
|
90
|
+
* `runHiddenTests`. Same contract as `runHiddenTests(task, ctx, runtime)`.
|
|
91
|
+
* Internal testing only.
|
|
85
92
|
* @param {Function} [opts.runJudge] - Test seam: replaces `runJudge`. Same
|
|
86
|
-
* contract as `runJudge(task, workdir,
|
|
87
|
-
* `
|
|
93
|
+
* contract as `runJudge(task, workdir, gradeResult, deps)` where
|
|
94
|
+
* `gradeResult` is the normalized grade plus the merged, source-stamped
|
|
95
|
+
* check `rows` (deps carries `runtime`). Internal testing only.
|
|
88
96
|
* @param {Function} [opts.installApm] - Test seam: replaces `installApm`.
|
|
89
97
|
* Same contract as `installApm(family, outputDir, runtime)`. Lets tests
|
|
90
98
|
* inject a fake subprocess (or skip the install entirely) so the suite
|
|
@@ -114,6 +122,7 @@ export class BenchmarkRunner {
|
|
|
114
122
|
// Test seams — default to the real implementations.
|
|
115
123
|
runAgent,
|
|
116
124
|
runInvariants: runInvariantsHook,
|
|
125
|
+
runHiddenTests: runHiddenTestsHook,
|
|
117
126
|
runJudge: runJudgeHook,
|
|
118
127
|
installApm: installApmHook,
|
|
119
128
|
installNpm: installNpmHook,
|
|
@@ -141,6 +150,7 @@ export class BenchmarkRunner {
|
|
|
141
150
|
this.termGraceMs = termGraceMs;
|
|
142
151
|
this._runAgentHook = runAgent ?? null;
|
|
143
152
|
this._runInvariantsHook = runInvariantsHook ?? runInvariants;
|
|
153
|
+
this._runHiddenTestsHook = runHiddenTestsHook ?? runHiddenTests;
|
|
144
154
|
this._runJudgeHook = runJudgeHook ?? runJudge;
|
|
145
155
|
this._installApmHook = installApmHook ?? defaultInstallApm;
|
|
146
156
|
this._installNpmHook = installNpmHook ?? defaultInstallNpm;
|
|
@@ -291,55 +301,37 @@ export class BenchmarkRunner {
|
|
|
291
301
|
const agentRun = await this.#runAgentSafe(task, workdir);
|
|
292
302
|
const { costUsd, turns, submission, agentError } = agentRun;
|
|
293
303
|
const breakdown = agentRun.costBreakdown ?? { agent: 0, supervisor: 0 };
|
|
294
|
-
const
|
|
304
|
+
const graded = await this.#gradeCell(family, task, workdir);
|
|
305
|
+
const { invariants, hiddenRows, engineError, rows, grade } = graded;
|
|
306
|
+
const { judgeVerdict, judgeCost } = await this.#judgeCell({
|
|
295
307
|
task,
|
|
296
|
-
|
|
297
|
-
|
|
298
|
-
|
|
299
|
-
|
|
300
|
-
|
|
301
|
-
|
|
302
|
-
|
|
303
|
-
|
|
304
|
-
|
|
305
|
-
|
|
306
|
-
|
|
307
|
-
|
|
308
|
-
task,
|
|
309
|
-
workdir,
|
|
310
|
-
skillSetHash,
|
|
311
|
-
);
|
|
312
|
-
const judgeResult = await this._runJudgeHook(
|
|
313
|
-
task,
|
|
314
|
-
workdir,
|
|
315
|
-
invariants,
|
|
316
|
-
{
|
|
317
|
-
query: this.query,
|
|
318
|
-
model: this.judgeModel,
|
|
319
|
-
judgeProfile: this.profiles.judge ?? undefined,
|
|
320
|
-
profilesDir: judgeProfilesDir,
|
|
321
|
-
runtime: this.runtime,
|
|
322
|
-
},
|
|
323
|
-
judgeContext,
|
|
324
|
-
);
|
|
325
|
-
judgeCost = judgeResult.costUsd ?? 0;
|
|
326
|
-
// The record's judgeVerdict carries only the verdict + summary; the
|
|
327
|
-
// judge's cost is folded into costUsd / costBreakdown instead.
|
|
328
|
-
judgeVerdict = {
|
|
329
|
-
verdict: judgeResult.verdict,
|
|
330
|
-
summary: judgeResult.summary,
|
|
331
|
-
};
|
|
332
|
-
}
|
|
333
|
-
const verdict =
|
|
334
|
-
invariants.verdict === "pass" &&
|
|
335
|
-
(judgeVerdict === null || judgeVerdict.verdict === "pass")
|
|
336
|
-
? "pass"
|
|
337
|
-
: "fail";
|
|
308
|
+
workdir,
|
|
309
|
+
gradeResult: { ...grade, rows },
|
|
310
|
+
skillSetHash,
|
|
311
|
+
judgeProfilesDir,
|
|
312
|
+
});
|
|
313
|
+
const judgePass =
|
|
314
|
+
judgeVerdict === null || judgeVerdict.verdict === "pass";
|
|
315
|
+
const verdict = grade.verdict === "pass" && judgePass ? "pass" : "fail";
|
|
316
|
+
// Gates protect the score: an unhealthy grader, a failing gate row, or
|
|
317
|
+
// a failing judge zeroes the effective score. Full marks does not — a
|
|
318
|
+
// fractional score with verdict fail is the point.
|
|
319
|
+
const scoreValid = graded.healthy && grade.gatesPass && judgePass;
|
|
338
320
|
const record = {
|
|
339
321
|
taskId: task.id,
|
|
340
322
|
runIndex,
|
|
341
323
|
verdict,
|
|
342
324
|
invariants,
|
|
325
|
+
grade,
|
|
326
|
+
...(task.tests && {
|
|
327
|
+
hiddenTests: {
|
|
328
|
+
details: hiddenRows,
|
|
329
|
+
...(engineError && { error: engineError.message }),
|
|
330
|
+
},
|
|
331
|
+
}),
|
|
332
|
+
...(grade.score !== undefined && {
|
|
333
|
+
score: scoreValid ? grade.score : 0,
|
|
334
|
+
}),
|
|
343
335
|
submission,
|
|
344
336
|
...(judgeVerdict && { judgeVerdict }),
|
|
345
337
|
costUsd: costUsd + judgeCost,
|
|
@@ -371,6 +363,65 @@ export class BenchmarkRunner {
|
|
|
371
363
|
}
|
|
372
364
|
}
|
|
373
365
|
|
|
366
|
+
/**
|
|
367
|
+
* Run the judge (when the task ships a template) over the grade result.
|
|
368
|
+
* The record's judgeVerdict carries only the verdict + summary; the
|
|
369
|
+
* judge's cost is folded into costUsd / costBreakdown instead.
|
|
370
|
+
*/
|
|
371
|
+
async #judgeCell({
|
|
372
|
+
task,
|
|
373
|
+
workdir,
|
|
374
|
+
gradeResult,
|
|
375
|
+
skillSetHash,
|
|
376
|
+
judgeProfilesDir,
|
|
377
|
+
}) {
|
|
378
|
+
if (!task.paths.judge) return { judgeVerdict: null, judgeCost: 0 };
|
|
379
|
+
const judgeContext = await this.#buildJudgeContext(
|
|
380
|
+
task,
|
|
381
|
+
workdir,
|
|
382
|
+
skillSetHash,
|
|
383
|
+
);
|
|
384
|
+
const judgeResult = await this._runJudgeHook(
|
|
385
|
+
task,
|
|
386
|
+
workdir,
|
|
387
|
+
gradeResult,
|
|
388
|
+
{
|
|
389
|
+
query: this.query,
|
|
390
|
+
model: this.judgeModel,
|
|
391
|
+
judgeProfile: this.profiles.judge ?? undefined,
|
|
392
|
+
profilesDir: judgeProfilesDir,
|
|
393
|
+
runtime: this.runtime,
|
|
394
|
+
},
|
|
395
|
+
judgeContext,
|
|
396
|
+
);
|
|
397
|
+
return {
|
|
398
|
+
judgeVerdict: {
|
|
399
|
+
verdict: judgeResult.verdict,
|
|
400
|
+
summary: judgeResult.summary,
|
|
401
|
+
},
|
|
402
|
+
judgeCost: judgeResult.costUsd ?? 0,
|
|
403
|
+
};
|
|
404
|
+
}
|
|
405
|
+
|
|
406
|
+
/**
|
|
407
|
+
* Run both check-row producers against the post-run CWD and grade the
|
|
408
|
+
* merged rows via the shared derivation. Restoration happens inside the
|
|
409
|
+
* engine, so the judge (which runs after) sees the workdir exactly as the
|
|
410
|
+
* agent left it.
|
|
411
|
+
*/
|
|
412
|
+
#gradeCell(family, task, workdir) {
|
|
413
|
+
const ctx = {
|
|
414
|
+
cwd: workdir.cwd,
|
|
415
|
+
port: workdir.port,
|
|
416
|
+
runDir: workdir.runDir,
|
|
417
|
+
familyDir: family.rootPath,
|
|
418
|
+
};
|
|
419
|
+
return runProducersAndGrade(task, ctx, this.runtime, {
|
|
420
|
+
runInvariants: this._runInvariantsHook,
|
|
421
|
+
runHiddenTests: this._runHiddenTestsHook,
|
|
422
|
+
});
|
|
423
|
+
}
|
|
424
|
+
|
|
374
425
|
/**
|
|
375
426
|
* Dispatch to either the injected hook or the default `#runAgent`. Either
|
|
376
427
|
* path can throw; catch here so a thrown error becomes an `agentError` on
|
|
@@ -614,67 +665,6 @@ async function writeRecord(stream, record) {
|
|
|
614
665
|
});
|
|
615
666
|
}
|
|
616
667
|
|
|
617
|
-
/**
|
|
618
|
-
* Split the combined supervisor trace into agent and supervisor files and
|
|
619
|
-
* extract turn count and submission in a single pass. Agent-source events go
|
|
620
|
-
* to `agentPath`; supervisor and orchestrator events go to `supervisorPath`.
|
|
621
|
-
*
|
|
622
|
-
* Cost is deliberately not summed here — the caller derives it from the same
|
|
623
|
-
* combined trace via `sumTraceCost`, so there is one cost path across the
|
|
624
|
-
* benchmark, callback, and `fit-trace cost` consumers.
|
|
625
|
-
*/
|
|
626
|
-
// biome-ignore lint/complexity/noExcessiveCognitiveComplexity: stream-splitting state machine
|
|
627
|
-
async function splitAndSummarize(
|
|
628
|
-
runtime,
|
|
629
|
-
combinedPath,
|
|
630
|
-
agentPath,
|
|
631
|
-
supervisorPath,
|
|
632
|
-
) {
|
|
633
|
-
const fs = runtime.fs;
|
|
634
|
-
const agentStream = fs.createWriteStream(agentPath);
|
|
635
|
-
const supStream = fs.createWriteStream(supervisorPath);
|
|
636
|
-
const rl = createInterface({
|
|
637
|
-
input: fs.createReadStream(combinedPath),
|
|
638
|
-
crlfDelay: Infinity,
|
|
639
|
-
});
|
|
640
|
-
let turns = 0;
|
|
641
|
-
let submission = "";
|
|
642
|
-
for await (const line of rl) {
|
|
643
|
-
if (!line.trim()) continue;
|
|
644
|
-
let event;
|
|
645
|
-
try {
|
|
646
|
-
event = JSON.parse(line);
|
|
647
|
-
} catch {
|
|
648
|
-
continue;
|
|
649
|
-
}
|
|
650
|
-
const target = event.source === "agent" ? agentStream : supStream;
|
|
651
|
-
target.write(line + "\n");
|
|
652
|
-
const inner = event.event;
|
|
653
|
-
if (!inner) continue;
|
|
654
|
-
if (event.source === "agent" && inner.type === "assistant") {
|
|
655
|
-
const text = extractText(inner);
|
|
656
|
-
if (text) submission = text;
|
|
657
|
-
}
|
|
658
|
-
if (event.source === "orchestrator" && inner.type === "summary") {
|
|
659
|
-
turns = inner.turns ?? 0;
|
|
660
|
-
}
|
|
661
|
-
}
|
|
662
|
-
await Promise.all([
|
|
663
|
-
new Promise((r) => agentStream.end(r)),
|
|
664
|
-
new Promise((r) => supStream.end(r)),
|
|
665
|
-
]);
|
|
666
|
-
return { turns, submission };
|
|
667
|
-
}
|
|
668
|
-
|
|
669
|
-
function extractText(inner) {
|
|
670
|
-
const content = inner.message?.content ?? inner.content;
|
|
671
|
-
if (!Array.isArray(content)) return null;
|
|
672
|
-
for (let i = content.length - 1; i >= 0; i--) {
|
|
673
|
-
if (content[i].type === "text" && content[i].text) return content[i].text;
|
|
674
|
-
}
|
|
675
|
-
return null;
|
|
676
|
-
}
|
|
677
|
-
|
|
678
668
|
/**
|
|
679
669
|
* Factory function — wires real dependencies.
|
|
680
670
|
* @param {ConstructorParameters<typeof BenchmarkRunner>[0]} opts
|
|
@@ -10,9 +10,16 @@
|
|
|
10
10
|
* hooks/ # harness-only; never copied to agent CWD
|
|
11
11
|
* preflight.sh
|
|
12
12
|
* invariants.sh
|
|
13
|
+
* tests/ # optional hidden test suite; harness-only overlay
|
|
13
14
|
* specs/ # copied into agent CWD
|
|
14
15
|
* workdir/ # copied into agent CWD
|
|
15
16
|
*
|
|
17
|
+
* `tests/` is an overlay mirror of the agent CWD: a file's path under
|
|
18
|
+
* `tests/` is its staging path. Every `*.test.js` file is one check —
|
|
19
|
+
* `*.gate.test.js` marks a gate, any other `*.test.js` is scored — and every
|
|
20
|
+
* other file is support material, staged but never graded. The layout is
|
|
21
|
+
* validated eagerly here so authoring errors fail before any agent spend.
|
|
22
|
+
*
|
|
16
23
|
* Local paths or git URLs are both accepted; git URLs are shallow-cloned into
|
|
17
24
|
* a temp dir and `familyRevision` becomes `git:<sha>` of HEAD at clone time.
|
|
18
25
|
* Local paths use the canonical-tree algorithm from design § Family revision
|
|
@@ -113,36 +120,124 @@ async function discoverTasks(runtime, rootPath) {
|
|
|
113
120
|
}
|
|
114
121
|
for (const entry of entries) {
|
|
115
122
|
if (!entry.isDirectory()) continue;
|
|
116
|
-
|
|
117
|
-
const supervisorPath = join(taskDir, "supervisor.task.md");
|
|
118
|
-
const judgePath = join(taskDir, "judge.task.md");
|
|
119
|
-
const preflightPath = join(taskDir, "hooks", "preflight.sh");
|
|
120
|
-
const invariantsPath = join(taskDir, "hooks", "invariants.sh");
|
|
121
|
-
tasks.push({
|
|
122
|
-
id: entry.name,
|
|
123
|
-
paths: {
|
|
124
|
-
taskDir,
|
|
125
|
-
instructions: join(taskDir, "agent.task.md"),
|
|
126
|
-
supervisor: (await fileExists(fs, supervisorPath))
|
|
127
|
-
? supervisorPath
|
|
128
|
-
: null,
|
|
129
|
-
judge: (await fileExists(fs, judgePath)) ? judgePath : null,
|
|
130
|
-
hooks: join(taskDir, "hooks"),
|
|
131
|
-
preflight: (await fileExecutable(fs, preflightPath))
|
|
132
|
-
? preflightPath
|
|
133
|
-
: null,
|
|
134
|
-
invariants: (await fileExecutable(fs, invariantsPath))
|
|
135
|
-
? invariantsPath
|
|
136
|
-
: null,
|
|
137
|
-
specs: join(taskDir, "specs"),
|
|
138
|
-
workdir: join(taskDir, "workdir"),
|
|
139
|
-
},
|
|
140
|
-
});
|
|
123
|
+
tasks.push(await loadTask(fs, join(tasksRoot, entry.name), entry.name));
|
|
141
124
|
}
|
|
142
125
|
tasks.sort((a, b) => (a.id < b.id ? -1 : a.id > b.id ? 1 : 0));
|
|
143
126
|
return tasks;
|
|
144
127
|
}
|
|
145
128
|
|
|
129
|
+
async function loadTask(fs, taskDir, id) {
|
|
130
|
+
const supervisorPath = join(taskDir, "supervisor.task.md");
|
|
131
|
+
const judgePath = join(taskDir, "judge.task.md");
|
|
132
|
+
const preflightPath = join(taskDir, "hooks", "preflight.sh");
|
|
133
|
+
const invariantsPath = join(taskDir, "hooks", "invariants.sh");
|
|
134
|
+
const suite = await discoverSuite(fs, taskDir);
|
|
135
|
+
return {
|
|
136
|
+
id,
|
|
137
|
+
paths: {
|
|
138
|
+
taskDir,
|
|
139
|
+
instructions: join(taskDir, "agent.task.md"),
|
|
140
|
+
supervisor: (await fileExists(fs, supervisorPath))
|
|
141
|
+
? supervisorPath
|
|
142
|
+
: null,
|
|
143
|
+
judge: (await fileExists(fs, judgePath)) ? judgePath : null,
|
|
144
|
+
hooks: join(taskDir, "hooks"),
|
|
145
|
+
preflight: (await fileExecutable(fs, preflightPath))
|
|
146
|
+
? preflightPath
|
|
147
|
+
: null,
|
|
148
|
+
invariants: (await fileExecutable(fs, invariantsPath))
|
|
149
|
+
? invariantsPath
|
|
150
|
+
: null,
|
|
151
|
+
tests: suite ? join(taskDir, "tests") : null,
|
|
152
|
+
specs: join(taskDir, "specs"),
|
|
153
|
+
workdir: join(taskDir, "workdir"),
|
|
154
|
+
},
|
|
155
|
+
tests: suite,
|
|
156
|
+
};
|
|
157
|
+
}
|
|
158
|
+
|
|
159
|
+
const CHECK_SUFFIX = ".test.js";
|
|
160
|
+
const GATE_SUFFIX = ".gate.test.js";
|
|
161
|
+
|
|
162
|
+
/**
|
|
163
|
+
* Discover and validate a task's hidden test suite under `<taskDir>/tests/`.
|
|
164
|
+
* Returns null when the directory is absent; throws on an invalid layout
|
|
165
|
+
* (no check files, a dangling symlink, duplicate check names) so
|
|
166
|
+
* `loadTaskFamily` rejects broken suites before any agent spend.
|
|
167
|
+
* @param {object} fs - Async filesystem surface (`runtime.fs`).
|
|
168
|
+
* @param {string} taskDir
|
|
169
|
+
* @returns {Promise<HiddenSuite | null>}
|
|
170
|
+
*/
|
|
171
|
+
async function discoverSuite(fs, taskDir) {
|
|
172
|
+
const testsRoot = join(taskDir, "tests");
|
|
173
|
+
try {
|
|
174
|
+
const st = await fs.lstat(testsRoot);
|
|
175
|
+
if (!st.isDirectory()) return null;
|
|
176
|
+
} catch {
|
|
177
|
+
return null;
|
|
178
|
+
}
|
|
179
|
+
const files = [];
|
|
180
|
+
await walkSuiteFiles(fs, testsRoot, testsRoot, files);
|
|
181
|
+
files.sort((a, b) =>
|
|
182
|
+
a.stagePath < b.stagePath ? -1 : a.stagePath > b.stagePath ? 1 : 0,
|
|
183
|
+
);
|
|
184
|
+
|
|
185
|
+
const checks = [];
|
|
186
|
+
const support = [];
|
|
187
|
+
for (const file of files) {
|
|
188
|
+
const base = file.stagePath.split(sep).at(-1);
|
|
189
|
+
if (!base.endsWith(CHECK_SUFFIX)) {
|
|
190
|
+
support.push(file);
|
|
191
|
+
continue;
|
|
192
|
+
}
|
|
193
|
+
const gate = base.endsWith(GATE_SUFFIX);
|
|
194
|
+
const name = base.slice(0, -(gate ? GATE_SUFFIX : CHECK_SUFFIX).length);
|
|
195
|
+
checks.push({ name, gate, ...file });
|
|
196
|
+
}
|
|
197
|
+
|
|
198
|
+
if (checks.length === 0) {
|
|
199
|
+
throw new Error(`hidden test suite has no check files: ${testsRoot}`);
|
|
200
|
+
}
|
|
201
|
+
const seen = new Set();
|
|
202
|
+
for (const check of checks) {
|
|
203
|
+
if (seen.has(check.name)) {
|
|
204
|
+
throw new Error(
|
|
205
|
+
`hidden test suite has duplicate check name '${check.name}': ${check.sourcePath}`,
|
|
206
|
+
);
|
|
207
|
+
}
|
|
208
|
+
seen.add(check.name);
|
|
209
|
+
}
|
|
210
|
+
return { checks, support };
|
|
211
|
+
}
|
|
212
|
+
|
|
213
|
+
/**
|
|
214
|
+
* Walk a suite tree collecting `{sourcePath, stagePath}` entries. Unlike
|
|
215
|
+
* `walkFiles` (which silently skips dangling symlinks for hashing), every
|
|
216
|
+
* entry here must be a regular file after symlink resolution — a dangling
|
|
217
|
+
* symlink or a link to a non-file target is an authoring error.
|
|
218
|
+
*/
|
|
219
|
+
async function walkSuiteFiles(fs, root, dir, out) {
|
|
220
|
+
const entries = await fs.readdir(dir, { withFileTypes: true });
|
|
221
|
+
for (const entry of entries) {
|
|
222
|
+
const full = join(dir, entry.name);
|
|
223
|
+
if (entry.isDirectory()) {
|
|
224
|
+
await walkSuiteFiles(fs, root, full, out);
|
|
225
|
+
} else if (entry.isFile()) {
|
|
226
|
+
out.push({ sourcePath: full, stagePath: relative(root, full) });
|
|
227
|
+
} else if (entry.isSymbolicLink()) {
|
|
228
|
+
const resolved = await resolveSymlinkToFile(fs, full);
|
|
229
|
+
if (!resolved) {
|
|
230
|
+
throw new Error(
|
|
231
|
+
`hidden test suite entry is not a regular file: ${full}`,
|
|
232
|
+
);
|
|
233
|
+
}
|
|
234
|
+
out.push({ sourcePath: full, stagePath: relative(root, full) });
|
|
235
|
+
} else {
|
|
236
|
+
throw new Error(`hidden test suite entry is not a regular file: ${full}`);
|
|
237
|
+
}
|
|
238
|
+
}
|
|
239
|
+
}
|
|
240
|
+
|
|
146
241
|
async function fileExists(fs, path) {
|
|
147
242
|
try {
|
|
148
243
|
await fs.access(path);
|
|
@@ -246,10 +341,28 @@ async function git(runtime, args) {
|
|
|
246
341
|
return stdout;
|
|
247
342
|
}
|
|
248
343
|
|
|
344
|
+
/**
|
|
345
|
+
* @typedef {object} HiddenCheck
|
|
346
|
+
* @property {string} name - Basename stem with the check suffix stripped.
|
|
347
|
+
* @property {boolean} gate - True iff the filename ends `.gate.test.js`.
|
|
348
|
+
* @property {string} sourcePath - Absolute path under `tests/`.
|
|
349
|
+
* @property {string} stagePath - Path relative to `tests/` — the overlay
|
|
350
|
+
* mirror of the staging path under the agent CWD.
|
|
351
|
+
*/
|
|
352
|
+
|
|
353
|
+
/**
|
|
354
|
+
* @typedef {object} HiddenSuite
|
|
355
|
+
* @property {HiddenCheck[]} checks - In sorted stage-path order.
|
|
356
|
+
* @property {{sourcePath: string, stagePath: string}[]} support - Non-check
|
|
357
|
+
* files, staged for the whole pass but never graded.
|
|
358
|
+
*/
|
|
359
|
+
|
|
249
360
|
/**
|
|
250
361
|
* @typedef {object} Task
|
|
251
362
|
* @property {string} id - Task name (directory name under tasks/)
|
|
252
|
-
* @property {{taskDir: string, instructions: string, supervisor: string|null, judge: string|null, hooks: string, preflight: string|null, invariants: string|null, specs: string, workdir: string}} paths
|
|
363
|
+
* @property {{taskDir: string, instructions: string, supervisor: string|null, judge: string|null, hooks: string, preflight: string|null, invariants: string|null, tests: string|null, specs: string, workdir: string}} paths
|
|
364
|
+
* @property {HiddenSuite | null} tests - Hidden test suite (null when the
|
|
365
|
+
* task ships no `tests/` directory)
|
|
253
366
|
*/
|
|
254
367
|
|
|
255
368
|
/**
|
|
@@ -0,0 +1,73 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Combined-supervisor-trace splitting for the benchmark runner: one pass
|
|
3
|
+
* over the tagged NDJSON envelope stream separates agent events from
|
|
4
|
+
* supervisor/orchestrator events and extracts the run summary.
|
|
5
|
+
*/
|
|
6
|
+
|
|
7
|
+
import { createInterface } from "node:readline";
|
|
8
|
+
|
|
9
|
+
/**
|
|
10
|
+
* Split the combined supervisor trace into agent and supervisor files and
|
|
11
|
+
* extract turn count and submission in a single pass. Agent-source events go
|
|
12
|
+
* to `agentPath`; supervisor and orchestrator events go to `supervisorPath`.
|
|
13
|
+
*
|
|
14
|
+
* Cost is deliberately not summed here — the caller derives it from the same
|
|
15
|
+
* combined trace via `sumTraceCost`, so there is one cost path across the
|
|
16
|
+
* benchmark, callback, and `fit-trace cost` consumers.
|
|
17
|
+
* @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
|
|
18
|
+
* @param {string} combinedPath
|
|
19
|
+
* @param {string} agentPath
|
|
20
|
+
* @param {string} supervisorPath
|
|
21
|
+
* @returns {Promise<{turns: number, submission: string}>}
|
|
22
|
+
*/
|
|
23
|
+
// biome-ignore lint/complexity/noExcessiveCognitiveComplexity: stream-splitting state machine
|
|
24
|
+
export async function splitAndSummarize(
|
|
25
|
+
runtime,
|
|
26
|
+
combinedPath,
|
|
27
|
+
agentPath,
|
|
28
|
+
supervisorPath,
|
|
29
|
+
) {
|
|
30
|
+
const fs = runtime.fs;
|
|
31
|
+
const agentStream = fs.createWriteStream(agentPath);
|
|
32
|
+
const supStream = fs.createWriteStream(supervisorPath);
|
|
33
|
+
const rl = createInterface({
|
|
34
|
+
input: fs.createReadStream(combinedPath),
|
|
35
|
+
crlfDelay: Infinity,
|
|
36
|
+
});
|
|
37
|
+
let turns = 0;
|
|
38
|
+
let submission = "";
|
|
39
|
+
for await (const line of rl) {
|
|
40
|
+
if (!line.trim()) continue;
|
|
41
|
+
let event;
|
|
42
|
+
try {
|
|
43
|
+
event = JSON.parse(line);
|
|
44
|
+
} catch {
|
|
45
|
+
continue;
|
|
46
|
+
}
|
|
47
|
+
const target = event.source === "agent" ? agentStream : supStream;
|
|
48
|
+
target.write(line + "\n");
|
|
49
|
+
const inner = event.event;
|
|
50
|
+
if (!inner) continue;
|
|
51
|
+
if (event.source === "agent" && inner.type === "assistant") {
|
|
52
|
+
const text = extractText(inner);
|
|
53
|
+
if (text) submission = text;
|
|
54
|
+
}
|
|
55
|
+
if (event.source === "orchestrator" && inner.type === "summary") {
|
|
56
|
+
turns = inner.turns ?? 0;
|
|
57
|
+
}
|
|
58
|
+
}
|
|
59
|
+
await Promise.all([
|
|
60
|
+
new Promise((r) => agentStream.end(r)),
|
|
61
|
+
new Promise((r) => supStream.end(r)),
|
|
62
|
+
]);
|
|
63
|
+
return { turns, submission };
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
function extractText(inner) {
|
|
67
|
+
const content = inner.message?.content ?? inner.content;
|
|
68
|
+
if (!Array.isArray(content)) return null;
|
|
69
|
+
for (let i = content.length - 1; i >= 0; i--) {
|
|
70
|
+
if (content[i].type === "text" && content[i].text) return content[i].text;
|
|
71
|
+
}
|
|
72
|
+
return null;
|
|
73
|
+
}
|
package/src/benchmark/workdir.js
CHANGED
|
@@ -268,7 +268,12 @@ async function runPreflight(runtime, script, cwd, port, vars) {
|
|
|
268
268
|
};
|
|
269
269
|
}
|
|
270
270
|
|
|
271
|
-
|
|
271
|
+
/**
|
|
272
|
+
* Allocate a free TCP port by binding to 0 and releasing it. Shared with the
|
|
273
|
+
* `grade` subcommand, which needs a plausible `$PORT` for the hook env.
|
|
274
|
+
* @returns {Promise<number>}
|
|
275
|
+
*/
|
|
276
|
+
export function probeFreePort() {
|
|
272
277
|
return new Promise((res, rej) => {
|
|
273
278
|
const server = createServer();
|
|
274
279
|
server.unref();
|
package/src/commands/assert.js
CHANGED
|
@@ -62,12 +62,74 @@ export function evaluateAssertion(values, args, fsSync) {
|
|
|
62
62
|
|
|
63
63
|
const output = { test: testName, pass: result.pass };
|
|
64
64
|
if (result.message) output.message = result.message;
|
|
65
|
+
applyGradingFlags(values, output);
|
|
65
66
|
return output;
|
|
66
67
|
}
|
|
67
68
|
|
|
69
|
+
/**
|
|
70
|
+
* Attach the check-row grading role: `--gate` marks a gate check, `--weight`
|
|
71
|
+
* attaches a numeric weight (0 marks the row diagnostic). `--gate` with any
|
|
72
|
+
* `--weight` — 0 included — is invalid: a stray weight must never silently
|
|
73
|
+
* disarm a gate.
|
|
74
|
+
* @param {object} values
|
|
75
|
+
* @param {{test: string, pass: boolean, message?: string}} output - Mutated.
|
|
76
|
+
*/
|
|
77
|
+
function applyGradingFlags(values, output) {
|
|
78
|
+
const hasWeight = values.weight !== undefined;
|
|
79
|
+
if (values.gate && hasWeight) {
|
|
80
|
+
throw new Error("assert: --gate cannot be combined with --weight");
|
|
81
|
+
}
|
|
82
|
+
if (values.gate) output.gate = true;
|
|
83
|
+
if (hasWeight) {
|
|
84
|
+
const weight = parseWeight(values.weight);
|
|
85
|
+
if (weight === null) {
|
|
86
|
+
throw new Error(
|
|
87
|
+
`assert: invalid --weight '${values.weight}' (expected a finite number ≥ 0)`,
|
|
88
|
+
);
|
|
89
|
+
}
|
|
90
|
+
output.weight = weight;
|
|
91
|
+
}
|
|
92
|
+
}
|
|
93
|
+
|
|
94
|
+
/**
|
|
95
|
+
* Parse a `--weight` value; null when invalid. A blank string is invalid —
|
|
96
|
+
* `Number("")` is 0, which would silently demote the check to a diagnostic.
|
|
97
|
+
* @param {string} raw
|
|
98
|
+
* @returns {number | null}
|
|
99
|
+
*/
|
|
100
|
+
function parseWeight(raw) {
|
|
101
|
+
if (typeof raw === "string" && raw.trim() === "") return null;
|
|
102
|
+
const weight = Number(raw);
|
|
103
|
+
return Number.isFinite(weight) && weight >= 0 ? weight : null;
|
|
104
|
+
}
|
|
105
|
+
|
|
106
|
+
/**
|
|
107
|
+
* The grading role an emit-then-fail row keeps: a failing check must not
|
|
108
|
+
* lose its authored role — an errored gate that demoted to a scored row
|
|
109
|
+
* would let a broken scaffold earn partial credit instead of zeroing the
|
|
110
|
+
* score. Invalid or conflicting flags yield no role (the row fails as a
|
|
111
|
+
* unit-weight scored check).
|
|
112
|
+
* @param {object} values
|
|
113
|
+
* @returns {{gate?: true, weight?: number}}
|
|
114
|
+
*/
|
|
115
|
+
function errorRowRole(values) {
|
|
116
|
+
const hasWeight = values.weight !== undefined;
|
|
117
|
+
if (values.gate && !hasWeight) return { gate: true };
|
|
118
|
+
if (!values.gate && hasWeight) {
|
|
119
|
+
const weight = parseWeight(values.weight);
|
|
120
|
+
if (weight !== null) return { weight };
|
|
121
|
+
}
|
|
122
|
+
return {};
|
|
123
|
+
}
|
|
124
|
+
|
|
68
125
|
/**
|
|
69
126
|
* Run an assertion, write JSON to stdout, and return a failure envelope when
|
|
70
127
|
* the assertion does not pass.
|
|
128
|
+
*
|
|
129
|
+
* Emit-then-fail on every failure path: an invalid grading flag or an
|
|
130
|
+
* errored evaluation (e.g. `--grep` against a file the agent deleted) writes
|
|
131
|
+
* a failing row before the nonzero exit, so a typo or a vanished target
|
|
132
|
+
* shrinks the score, never the denominator.
|
|
71
133
|
* @param {import("@forwardimpact/libcli").InvocationContext} ctx
|
|
72
134
|
* @returns {Promise<{ok: true} | {ok: false, code: number, error: string}>}
|
|
73
135
|
*/
|
|
@@ -78,6 +140,16 @@ export async function runAssertCommand(ctx) {
|
|
|
78
140
|
try {
|
|
79
141
|
result = evaluateAssertion(ctx.options, args, runtime.fsSync);
|
|
80
142
|
} catch (err) {
|
|
143
|
+
const reason = err.message.startsWith("assert: ")
|
|
144
|
+
? err.message
|
|
145
|
+
: `assert: ${err.message}`;
|
|
146
|
+
const row = {
|
|
147
|
+
test: ctx.args["test-name"] ?? "(missing test name)",
|
|
148
|
+
pass: false,
|
|
149
|
+
...errorRowRole(ctx.options),
|
|
150
|
+
message: reason,
|
|
151
|
+
};
|
|
152
|
+
runtime.proc.stdout.write(JSON.stringify(row) + "\n");
|
|
81
153
|
return { ok: false, code: 1, error: err.message };
|
|
82
154
|
}
|
|
83
155
|
runtime.proc.stdout.write(JSON.stringify(result) + "\n");
|
|
@@ -5,7 +5,7 @@
|
|
|
5
5
|
*/
|
|
6
6
|
|
|
7
7
|
import { runBenchmarkRunCommand } from "./benchmark-run.js";
|
|
8
|
-
import {
|
|
8
|
+
import { runBenchmarkGradeCommand } from "./benchmark-grade.js";
|
|
9
9
|
import { runBenchmarkReportCommand } from "./benchmark-report.js";
|
|
10
10
|
import {
|
|
11
11
|
BENCHMARK_AGENT_MODEL,
|
|
@@ -95,11 +95,11 @@ export const definition = {
|
|
|
95
95
|
},
|
|
96
96
|
},
|
|
97
97
|
{
|
|
98
|
-
name: "
|
|
98
|
+
name: "grade",
|
|
99
99
|
args: [],
|
|
100
|
-
handler:
|
|
100
|
+
handler: runBenchmarkGradeCommand,
|
|
101
101
|
description:
|
|
102
|
-
"
|
|
102
|
+
"Grade a single task against a post-run workdir without invoking an agent: run the hidden test suite and the invariants script, then derive the verdict from the check rows (the exit mirrors it).",
|
|
103
103
|
options: {
|
|
104
104
|
family: {
|
|
105
105
|
type: "string",
|
|
@@ -112,7 +112,7 @@ export const definition = {
|
|
|
112
112
|
"run-dir": {
|
|
113
113
|
type: "string",
|
|
114
114
|
description:
|
|
115
|
-
"Post-run directory whose cwd/ subdir is the agent CWD;
|
|
115
|
+
"Post-run directory whose cwd/ subdir is the agent CWD; both producers run against that cwd — the path hooks receive as $AGENT_CWD",
|
|
116
116
|
},
|
|
117
117
|
output: {
|
|
118
118
|
type: "string",
|
|
@@ -159,7 +159,7 @@ export const definition = {
|
|
|
159
159
|
"fit-benchmark run --family=./families/coding --work-tracker=filesystem",
|
|
160
160
|
"fit-benchmark run --family=./families/coding --skills-from=. --task=todo-api",
|
|
161
161
|
`fit-benchmark run --family=./families/coding --runs=10 --agent-model=${BENCHMARK_AGENT_MODEL}`,
|
|
162
|
-
"fit-benchmark
|
|
162
|
+
"fit-benchmark grade --family=./families/coding --task=todo-api --run-dir=./benchmark-runs/runs/todo-api/0",
|
|
163
163
|
"fit-benchmark report --format=text",
|
|
164
164
|
"fit-benchmark report --input=./runs/today --k=1,3,5 --format=text",
|
|
165
165
|
],
|