@forwardimpact/libharness 1.3.0 → 2.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bin/fit-harness.js +18 -0
- package/bin/fit-trace.js +10 -0
- package/package.json +1 -1
- package/src/advisor.js +218 -0
- package/src/agent-runner.js +6 -0
- package/src/benchmark/grade.js +222 -0
- package/src/benchmark/hidden-tests.js +180 -0
- package/src/benchmark/invariants.js +9 -10
- package/src/benchmark/judge.js +9 -6
- package/src/benchmark/report.js +150 -56
- package/src/benchmark/result.js +43 -8
- package/src/benchmark/runner.js +99 -109
- package/src/benchmark/task-family.js +139 -26
- package/src/benchmark/trace-split.js +73 -0
- package/src/benchmark/workdir.js +6 -1
- package/src/commands/advisor-flags.js +28 -0
- package/src/commands/assert.js +72 -0
- package/src/commands/benchmark-definition.js +11 -6
- package/src/commands/benchmark-grade.js +82 -0
- package/src/commands/benchmark-report.js +12 -2
- package/src/commands/discuss.js +4 -0
- package/src/commands/facilitate.js +4 -0
- package/src/commands/run.js +162 -67
- package/src/commands/supervise.js +4 -0
- package/src/discuss-tools.js +2 -1
- package/src/discusser.js +65 -10
- package/src/facilitator.js +64 -9
- package/src/index.js +9 -0
- package/src/orchestration-toolkit.js +65 -7
- package/src/supervisor.js +72 -10
- package/src/transcript-recorder.js +94 -0
- package/src/commands/benchmark-invariants.js +0 -73
package/src/benchmark/result.js
CHANGED
|
@@ -3,10 +3,14 @@
|
|
|
3
3
|
*
|
|
4
4
|
* Two schemas live here:
|
|
5
5
|
* - RESULT_RECORD_SCHEMA — one record per (task, runIndex) from a full
|
|
6
|
-
* benchmark run. Has a happy branch (
|
|
7
|
-
* pre-flight-failure branch (
|
|
8
|
-
* -
|
|
9
|
-
*
|
|
6
|
+
* benchmark run. Has a happy branch (grade + collectors + judge present)
|
|
7
|
+
* and a pre-flight-failure branch (grade/judgeVerdict/submission absent).
|
|
8
|
+
* - GRADE_RECORD_SCHEMA — narrower output of `benchmark-grade`: ad-hoc
|
|
9
|
+
* grading without a full lifecycle.
|
|
10
|
+
*
|
|
11
|
+
* The check rows are the authoritative grading channel: the happy branch
|
|
12
|
+
* requires a `grade` object, so a pre-break record fails validation rather
|
|
13
|
+
* than rendering under semantics it never carried.
|
|
10
14
|
*
|
|
11
15
|
* Validation is throw-on-mismatch so the runner can wrap every JSONL append
|
|
12
16
|
* in a guard and reject schema drift at write time.
|
|
@@ -17,12 +21,27 @@ import { z } from "zod";
|
|
|
17
21
|
const VERDICT_ENUM = z.enum(["pass", "fail"]);
|
|
18
22
|
|
|
19
23
|
const INVARIANTS_SHAPE = z.object({
|
|
20
|
-
verdict: VERDICT_ENUM,
|
|
21
24
|
details: z.array(z.unknown()),
|
|
22
25
|
exitCode: z.number().int(),
|
|
23
26
|
stderr: z.string().optional(),
|
|
24
27
|
});
|
|
25
28
|
|
|
29
|
+
/**
|
|
30
|
+
* The normalized grading projection: `score` appears only on scored tasks,
|
|
31
|
+
* `malformed` only when at least one row was malformed.
|
|
32
|
+
*/
|
|
33
|
+
const GRADE_SHAPE = z.object({
|
|
34
|
+
verdict: VERDICT_ENUM,
|
|
35
|
+
gatesPass: z.boolean(),
|
|
36
|
+
score: z.number().min(0).max(1).optional(),
|
|
37
|
+
malformed: z.number().int().min(1).optional(),
|
|
38
|
+
});
|
|
39
|
+
|
|
40
|
+
const HIDDEN_TESTS_SHAPE = z.object({
|
|
41
|
+
details: z.array(z.unknown()),
|
|
42
|
+
error: z.string().optional(),
|
|
43
|
+
});
|
|
44
|
+
|
|
26
45
|
const JUDGE_VERDICT_SHAPE = z.object({
|
|
27
46
|
verdict: VERDICT_ENUM,
|
|
28
47
|
summary: z.string(),
|
|
@@ -77,6 +96,11 @@ const AGENT_ERROR_SHAPE = z.object({
|
|
|
77
96
|
const HAPPY_RECORD = z.object({
|
|
78
97
|
...COMMON_FIELDS,
|
|
79
98
|
invariants: INVARIANTS_SHAPE,
|
|
99
|
+
grade: GRADE_SHAPE,
|
|
100
|
+
hiddenTests: HIDDEN_TESTS_SHAPE.optional(),
|
|
101
|
+
// The effective, judge-zeroed score `report` aggregates — present only on
|
|
102
|
+
// scored tasks.
|
|
103
|
+
score: z.number().min(0).max(1).optional(),
|
|
80
104
|
submission: z.string(),
|
|
81
105
|
judgeVerdict: JUDGE_VERDICT_SHAPE.optional(),
|
|
82
106
|
agentTracePath: z.string(),
|
|
@@ -97,6 +121,9 @@ const PREFLIGHT_RECORD = z.object({
|
|
|
97
121
|
supervisorTracePath: z.string(),
|
|
98
122
|
judgeTracePath: z.string(),
|
|
99
123
|
invariants: z.undefined().optional(),
|
|
124
|
+
grade: z.undefined().optional(),
|
|
125
|
+
hiddenTests: z.undefined().optional(),
|
|
126
|
+
score: z.undefined().optional(),
|
|
100
127
|
submission: z.undefined().optional(),
|
|
101
128
|
judgeVerdict: z.undefined().optional(),
|
|
102
129
|
agentError: z.undefined().optional(),
|
|
@@ -104,9 +131,17 @@ const PREFLIGHT_RECORD = z.object({
|
|
|
104
131
|
|
|
105
132
|
export const RESULT_RECORD_SCHEMA = z.union([HAPPY_RECORD, PREFLIGHT_RECORD]);
|
|
106
133
|
|
|
107
|
-
export const
|
|
134
|
+
export const GRADE_RECORD_SCHEMA = z.object({
|
|
108
135
|
taskId: z.string().min(1),
|
|
136
|
+
// Unlike the happy result record — where `grade.score` is the raw
|
|
137
|
+
// weighted fraction and the effective (zeroed) value lives on the
|
|
138
|
+
// top-level `score` — this record has no second score field, so its
|
|
139
|
+
// `grade.score` carries the effective health/gate-zeroed value.
|
|
140
|
+
grade: GRADE_SHAPE,
|
|
109
141
|
invariants: INVARIANTS_SHAPE,
|
|
142
|
+
hiddenTests: HIDDEN_TESTS_SHAPE.optional(),
|
|
143
|
+
// Mirrors the invariants script's exit for diagnosis; the graded verdict
|
|
144
|
+
// is what drives the command's process exit.
|
|
110
145
|
exitCode: z.number().int(),
|
|
111
146
|
});
|
|
112
147
|
|
|
@@ -122,6 +157,6 @@ export function validateResultRecord(record) {
|
|
|
122
157
|
* Throw on schema mismatch.
|
|
123
158
|
* @param {object} record
|
|
124
159
|
*/
|
|
125
|
-
export function
|
|
126
|
-
|
|
160
|
+
export function validateGradeRecord(record) {
|
|
161
|
+
GRADE_RECORD_SCHEMA.parse(record);
|
|
127
162
|
}
|
package/src/benchmark/runner.js
CHANGED
|
@@ -4,8 +4,10 @@
|
|
|
4
4
|
* Phases per (task, runIndex):
|
|
5
5
|
* 1. WorkdirManager.start → seed CWD + run pre-flight probe
|
|
6
6
|
* 2. Supervisor session (agent + supervisor) → produce traces + submission
|
|
7
|
-
* 3. Invariants
|
|
8
|
-
*
|
|
7
|
+
* 3. Invariants collector + hidden-test engine → merged check rows,
|
|
8
|
+
* graded by `gradeChecks` (rows are authoritative; script exit is
|
|
9
|
+
* grader health only)
|
|
10
|
+
* 4. Judge.runJudge → Conclude-driven binary gate mapped to pass/fail
|
|
9
11
|
* 5. WorkdirManager.teardown → process-group cleanup
|
|
10
12
|
*
|
|
11
13
|
* Cells run with bounded in-process concurrency (`CellScheduler`); `run()`
|
|
@@ -17,7 +19,6 @@
|
|
|
17
19
|
* stream.
|
|
18
20
|
*/
|
|
19
21
|
|
|
20
|
-
import { createInterface } from "node:readline";
|
|
21
22
|
import { join, resolve as resolvePath } from "node:path";
|
|
22
23
|
|
|
23
24
|
import { DEFAULT_ENV_ALLOWLIST, createRedactor } from "../redaction.js";
|
|
@@ -28,7 +29,10 @@ import { installNpm as defaultInstallNpm } from "./npm-installer.js";
|
|
|
28
29
|
import { runJudge } from "./judge.js";
|
|
29
30
|
import { validateResultRecord } from "./result.js";
|
|
30
31
|
import { runInvariants } from "./invariants.js";
|
|
32
|
+
import { runHiddenTests } from "./hidden-tests.js";
|
|
33
|
+
import { runProducersAndGrade } from "./grade.js";
|
|
31
34
|
import { assertJudgeProfileStaged, loadTaskFamily } from "./task-family.js";
|
|
35
|
+
import { splitAndSummarize } from "./trace-split.js";
|
|
32
36
|
import { createWorkdirManager } from "./workdir.js";
|
|
33
37
|
import { CellScheduler } from "./scheduler.js";
|
|
34
38
|
|
|
@@ -82,9 +86,13 @@ export class BenchmarkRunner {
|
|
|
82
86
|
* threaded into the installers, workdir manager, invariants, and judge.
|
|
83
87
|
* @param {Function} [opts.runInvariants] - Test seam: replaces `runInvariants`.
|
|
84
88
|
* Same contract as `runInvariants(task, ctx, runtime)`. Internal testing only.
|
|
89
|
+
* @param {Function} [opts.runHiddenTests] - Test seam: replaces
|
|
90
|
+
* `runHiddenTests`. Same contract as `runHiddenTests(task, ctx, runtime)`.
|
|
91
|
+
* Internal testing only.
|
|
85
92
|
* @param {Function} [opts.runJudge] - Test seam: replaces `runJudge`. Same
|
|
86
|
-
* contract as `runJudge(task, workdir,
|
|
87
|
-
* `
|
|
93
|
+
* contract as `runJudge(task, workdir, gradeResult, deps)` where
|
|
94
|
+
* `gradeResult` is the normalized grade plus the merged, source-stamped
|
|
95
|
+
* check `rows` (deps carries `runtime`). Internal testing only.
|
|
88
96
|
* @param {Function} [opts.installApm] - Test seam: replaces `installApm`.
|
|
89
97
|
* Same contract as `installApm(family, outputDir, runtime)`. Lets tests
|
|
90
98
|
* inject a fake subprocess (or skip the install entirely) so the suite
|
|
@@ -114,6 +122,7 @@ export class BenchmarkRunner {
|
|
|
114
122
|
// Test seams — default to the real implementations.
|
|
115
123
|
runAgent,
|
|
116
124
|
runInvariants: runInvariantsHook,
|
|
125
|
+
runHiddenTests: runHiddenTestsHook,
|
|
117
126
|
runJudge: runJudgeHook,
|
|
118
127
|
installApm: installApmHook,
|
|
119
128
|
installNpm: installNpmHook,
|
|
@@ -141,6 +150,7 @@ export class BenchmarkRunner {
|
|
|
141
150
|
this.termGraceMs = termGraceMs;
|
|
142
151
|
this._runAgentHook = runAgent ?? null;
|
|
143
152
|
this._runInvariantsHook = runInvariantsHook ?? runInvariants;
|
|
153
|
+
this._runHiddenTestsHook = runHiddenTestsHook ?? runHiddenTests;
|
|
144
154
|
this._runJudgeHook = runJudgeHook ?? runJudge;
|
|
145
155
|
this._installApmHook = installApmHook ?? defaultInstallApm;
|
|
146
156
|
this._installNpmHook = installNpmHook ?? defaultInstallNpm;
|
|
@@ -291,55 +301,37 @@ export class BenchmarkRunner {
|
|
|
291
301
|
const agentRun = await this.#runAgentSafe(task, workdir);
|
|
292
302
|
const { costUsd, turns, submission, agentError } = agentRun;
|
|
293
303
|
const breakdown = agentRun.costBreakdown ?? { agent: 0, supervisor: 0 };
|
|
294
|
-
const
|
|
304
|
+
const graded = await this.#gradeCell(family, task, workdir);
|
|
305
|
+
const { invariants, hiddenRows, engineError, rows, grade } = graded;
|
|
306
|
+
const { judgeVerdict, judgeCost } = await this.#judgeCell({
|
|
295
307
|
task,
|
|
296
|
-
|
|
297
|
-
|
|
298
|
-
|
|
299
|
-
|
|
300
|
-
|
|
301
|
-
|
|
302
|
-
|
|
303
|
-
|
|
304
|
-
|
|
305
|
-
|
|
306
|
-
|
|
307
|
-
|
|
308
|
-
task,
|
|
309
|
-
workdir,
|
|
310
|
-
skillSetHash,
|
|
311
|
-
);
|
|
312
|
-
const judgeResult = await this._runJudgeHook(
|
|
313
|
-
task,
|
|
314
|
-
workdir,
|
|
315
|
-
invariants,
|
|
316
|
-
{
|
|
317
|
-
query: this.query,
|
|
318
|
-
model: this.judgeModel,
|
|
319
|
-
judgeProfile: this.profiles.judge ?? undefined,
|
|
320
|
-
profilesDir: judgeProfilesDir,
|
|
321
|
-
runtime: this.runtime,
|
|
322
|
-
},
|
|
323
|
-
judgeContext,
|
|
324
|
-
);
|
|
325
|
-
judgeCost = judgeResult.costUsd ?? 0;
|
|
326
|
-
// The record's judgeVerdict carries only the verdict + summary; the
|
|
327
|
-
// judge's cost is folded into costUsd / costBreakdown instead.
|
|
328
|
-
judgeVerdict = {
|
|
329
|
-
verdict: judgeResult.verdict,
|
|
330
|
-
summary: judgeResult.summary,
|
|
331
|
-
};
|
|
332
|
-
}
|
|
333
|
-
const verdict =
|
|
334
|
-
invariants.verdict === "pass" &&
|
|
335
|
-
(judgeVerdict === null || judgeVerdict.verdict === "pass")
|
|
336
|
-
? "pass"
|
|
337
|
-
: "fail";
|
|
308
|
+
workdir,
|
|
309
|
+
gradeResult: { ...grade, rows },
|
|
310
|
+
skillSetHash,
|
|
311
|
+
judgeProfilesDir,
|
|
312
|
+
});
|
|
313
|
+
const judgePass =
|
|
314
|
+
judgeVerdict === null || judgeVerdict.verdict === "pass";
|
|
315
|
+
const verdict = grade.verdict === "pass" && judgePass ? "pass" : "fail";
|
|
316
|
+
// Gates protect the score: an unhealthy grader, a failing gate row, or
|
|
317
|
+
// a failing judge zeroes the effective score. Full marks does not — a
|
|
318
|
+
// fractional score with verdict fail is the point.
|
|
319
|
+
const scoreValid = graded.healthy && grade.gatesPass && judgePass;
|
|
338
320
|
const record = {
|
|
339
321
|
taskId: task.id,
|
|
340
322
|
runIndex,
|
|
341
323
|
verdict,
|
|
342
324
|
invariants,
|
|
325
|
+
grade,
|
|
326
|
+
...(task.tests && {
|
|
327
|
+
hiddenTests: {
|
|
328
|
+
details: hiddenRows,
|
|
329
|
+
...(engineError && { error: engineError.message }),
|
|
330
|
+
},
|
|
331
|
+
}),
|
|
332
|
+
...(grade.score !== undefined && {
|
|
333
|
+
score: scoreValid ? grade.score : 0,
|
|
334
|
+
}),
|
|
343
335
|
submission,
|
|
344
336
|
...(judgeVerdict && { judgeVerdict }),
|
|
345
337
|
costUsd: costUsd + judgeCost,
|
|
@@ -371,6 +363,65 @@ export class BenchmarkRunner {
|
|
|
371
363
|
}
|
|
372
364
|
}
|
|
373
365
|
|
|
366
|
+
/**
|
|
367
|
+
* Run the judge (when the task ships a template) over the grade result.
|
|
368
|
+
* The record's judgeVerdict carries only the verdict + summary; the
|
|
369
|
+
* judge's cost is folded into costUsd / costBreakdown instead.
|
|
370
|
+
*/
|
|
371
|
+
async #judgeCell({
|
|
372
|
+
task,
|
|
373
|
+
workdir,
|
|
374
|
+
gradeResult,
|
|
375
|
+
skillSetHash,
|
|
376
|
+
judgeProfilesDir,
|
|
377
|
+
}) {
|
|
378
|
+
if (!task.paths.judge) return { judgeVerdict: null, judgeCost: 0 };
|
|
379
|
+
const judgeContext = await this.#buildJudgeContext(
|
|
380
|
+
task,
|
|
381
|
+
workdir,
|
|
382
|
+
skillSetHash,
|
|
383
|
+
);
|
|
384
|
+
const judgeResult = await this._runJudgeHook(
|
|
385
|
+
task,
|
|
386
|
+
workdir,
|
|
387
|
+
gradeResult,
|
|
388
|
+
{
|
|
389
|
+
query: this.query,
|
|
390
|
+
model: this.judgeModel,
|
|
391
|
+
judgeProfile: this.profiles.judge ?? undefined,
|
|
392
|
+
profilesDir: judgeProfilesDir,
|
|
393
|
+
runtime: this.runtime,
|
|
394
|
+
},
|
|
395
|
+
judgeContext,
|
|
396
|
+
);
|
|
397
|
+
return {
|
|
398
|
+
judgeVerdict: {
|
|
399
|
+
verdict: judgeResult.verdict,
|
|
400
|
+
summary: judgeResult.summary,
|
|
401
|
+
},
|
|
402
|
+
judgeCost: judgeResult.costUsd ?? 0,
|
|
403
|
+
};
|
|
404
|
+
}
|
|
405
|
+
|
|
406
|
+
/**
|
|
407
|
+
* Run both check-row producers against the post-run CWD and grade the
|
|
408
|
+
* merged rows via the shared derivation. Restoration happens inside the
|
|
409
|
+
* engine, so the judge (which runs after) sees the workdir exactly as the
|
|
410
|
+
* agent left it.
|
|
411
|
+
*/
|
|
412
|
+
#gradeCell(family, task, workdir) {
|
|
413
|
+
const ctx = {
|
|
414
|
+
cwd: workdir.cwd,
|
|
415
|
+
port: workdir.port,
|
|
416
|
+
runDir: workdir.runDir,
|
|
417
|
+
familyDir: family.rootPath,
|
|
418
|
+
};
|
|
419
|
+
return runProducersAndGrade(task, ctx, this.runtime, {
|
|
420
|
+
runInvariants: this._runInvariantsHook,
|
|
421
|
+
runHiddenTests: this._runHiddenTestsHook,
|
|
422
|
+
});
|
|
423
|
+
}
|
|
424
|
+
|
|
374
425
|
/**
|
|
375
426
|
* Dispatch to either the injected hook or the default `#runAgent`. Either
|
|
376
427
|
* path can throw; catch here so a thrown error becomes an `agentError` on
|
|
@@ -614,67 +665,6 @@ async function writeRecord(stream, record) {
|
|
|
614
665
|
});
|
|
615
666
|
}
|
|
616
667
|
|
|
617
|
-
/**
|
|
618
|
-
* Split the combined supervisor trace into agent and supervisor files and
|
|
619
|
-
* extract turn count and submission in a single pass. Agent-source events go
|
|
620
|
-
* to `agentPath`; supervisor and orchestrator events go to `supervisorPath`.
|
|
621
|
-
*
|
|
622
|
-
* Cost is deliberately not summed here — the caller derives it from the same
|
|
623
|
-
* combined trace via `sumTraceCost`, so there is one cost path across the
|
|
624
|
-
* benchmark, callback, and `fit-trace cost` consumers.
|
|
625
|
-
*/
|
|
626
|
-
// biome-ignore lint/complexity/noExcessiveCognitiveComplexity: stream-splitting state machine
|
|
627
|
-
async function splitAndSummarize(
|
|
628
|
-
runtime,
|
|
629
|
-
combinedPath,
|
|
630
|
-
agentPath,
|
|
631
|
-
supervisorPath,
|
|
632
|
-
) {
|
|
633
|
-
const fs = runtime.fs;
|
|
634
|
-
const agentStream = fs.createWriteStream(agentPath);
|
|
635
|
-
const supStream = fs.createWriteStream(supervisorPath);
|
|
636
|
-
const rl = createInterface({
|
|
637
|
-
input: fs.createReadStream(combinedPath),
|
|
638
|
-
crlfDelay: Infinity,
|
|
639
|
-
});
|
|
640
|
-
let turns = 0;
|
|
641
|
-
let submission = "";
|
|
642
|
-
for await (const line of rl) {
|
|
643
|
-
if (!line.trim()) continue;
|
|
644
|
-
let event;
|
|
645
|
-
try {
|
|
646
|
-
event = JSON.parse(line);
|
|
647
|
-
} catch {
|
|
648
|
-
continue;
|
|
649
|
-
}
|
|
650
|
-
const target = event.source === "agent" ? agentStream : supStream;
|
|
651
|
-
target.write(line + "\n");
|
|
652
|
-
const inner = event.event;
|
|
653
|
-
if (!inner) continue;
|
|
654
|
-
if (event.source === "agent" && inner.type === "assistant") {
|
|
655
|
-
const text = extractText(inner);
|
|
656
|
-
if (text) submission = text;
|
|
657
|
-
}
|
|
658
|
-
if (event.source === "orchestrator" && inner.type === "summary") {
|
|
659
|
-
turns = inner.turns ?? 0;
|
|
660
|
-
}
|
|
661
|
-
}
|
|
662
|
-
await Promise.all([
|
|
663
|
-
new Promise((r) => agentStream.end(r)),
|
|
664
|
-
new Promise((r) => supStream.end(r)),
|
|
665
|
-
]);
|
|
666
|
-
return { turns, submission };
|
|
667
|
-
}
|
|
668
|
-
|
|
669
|
-
function extractText(inner) {
|
|
670
|
-
const content = inner.message?.content ?? inner.content;
|
|
671
|
-
if (!Array.isArray(content)) return null;
|
|
672
|
-
for (let i = content.length - 1; i >= 0; i--) {
|
|
673
|
-
if (content[i].type === "text" && content[i].text) return content[i].text;
|
|
674
|
-
}
|
|
675
|
-
return null;
|
|
676
|
-
}
|
|
677
|
-
|
|
678
668
|
/**
|
|
679
669
|
* Factory function — wires real dependencies.
|
|
680
670
|
* @param {ConstructorParameters<typeof BenchmarkRunner>[0]} opts
|
|
@@ -10,9 +10,16 @@
|
|
|
10
10
|
* hooks/ # harness-only; never copied to agent CWD
|
|
11
11
|
* preflight.sh
|
|
12
12
|
* invariants.sh
|
|
13
|
+
* tests/ # optional hidden test suite; harness-only overlay
|
|
13
14
|
* specs/ # copied into agent CWD
|
|
14
15
|
* workdir/ # copied into agent CWD
|
|
15
16
|
*
|
|
17
|
+
* `tests/` is an overlay mirror of the agent CWD: a file's path under
|
|
18
|
+
* `tests/` is its staging path. Every `*.test.js` file is one check —
|
|
19
|
+
* `*.gate.test.js` marks a gate, any other `*.test.js` is scored — and every
|
|
20
|
+
* other file is support material, staged but never graded. The layout is
|
|
21
|
+
* validated eagerly here so authoring errors fail before any agent spend.
|
|
22
|
+
*
|
|
16
23
|
* Local paths or git URLs are both accepted; git URLs are shallow-cloned into
|
|
17
24
|
* a temp dir and `familyRevision` becomes `git:<sha>` of HEAD at clone time.
|
|
18
25
|
* Local paths use the canonical-tree algorithm from design § Family revision
|
|
@@ -113,36 +120,124 @@ async function discoverTasks(runtime, rootPath) {
|
|
|
113
120
|
}
|
|
114
121
|
for (const entry of entries) {
|
|
115
122
|
if (!entry.isDirectory()) continue;
|
|
116
|
-
|
|
117
|
-
const supervisorPath = join(taskDir, "supervisor.task.md");
|
|
118
|
-
const judgePath = join(taskDir, "judge.task.md");
|
|
119
|
-
const preflightPath = join(taskDir, "hooks", "preflight.sh");
|
|
120
|
-
const invariantsPath = join(taskDir, "hooks", "invariants.sh");
|
|
121
|
-
tasks.push({
|
|
122
|
-
id: entry.name,
|
|
123
|
-
paths: {
|
|
124
|
-
taskDir,
|
|
125
|
-
instructions: join(taskDir, "agent.task.md"),
|
|
126
|
-
supervisor: (await fileExists(fs, supervisorPath))
|
|
127
|
-
? supervisorPath
|
|
128
|
-
: null,
|
|
129
|
-
judge: (await fileExists(fs, judgePath)) ? judgePath : null,
|
|
130
|
-
hooks: join(taskDir, "hooks"),
|
|
131
|
-
preflight: (await fileExecutable(fs, preflightPath))
|
|
132
|
-
? preflightPath
|
|
133
|
-
: null,
|
|
134
|
-
invariants: (await fileExecutable(fs, invariantsPath))
|
|
135
|
-
? invariantsPath
|
|
136
|
-
: null,
|
|
137
|
-
specs: join(taskDir, "specs"),
|
|
138
|
-
workdir: join(taskDir, "workdir"),
|
|
139
|
-
},
|
|
140
|
-
});
|
|
123
|
+
tasks.push(await loadTask(fs, join(tasksRoot, entry.name), entry.name));
|
|
141
124
|
}
|
|
142
125
|
tasks.sort((a, b) => (a.id < b.id ? -1 : a.id > b.id ? 1 : 0));
|
|
143
126
|
return tasks;
|
|
144
127
|
}
|
|
145
128
|
|
|
129
|
+
async function loadTask(fs, taskDir, id) {
|
|
130
|
+
const supervisorPath = join(taskDir, "supervisor.task.md");
|
|
131
|
+
const judgePath = join(taskDir, "judge.task.md");
|
|
132
|
+
const preflightPath = join(taskDir, "hooks", "preflight.sh");
|
|
133
|
+
const invariantsPath = join(taskDir, "hooks", "invariants.sh");
|
|
134
|
+
const suite = await discoverSuite(fs, taskDir);
|
|
135
|
+
return {
|
|
136
|
+
id,
|
|
137
|
+
paths: {
|
|
138
|
+
taskDir,
|
|
139
|
+
instructions: join(taskDir, "agent.task.md"),
|
|
140
|
+
supervisor: (await fileExists(fs, supervisorPath))
|
|
141
|
+
? supervisorPath
|
|
142
|
+
: null,
|
|
143
|
+
judge: (await fileExists(fs, judgePath)) ? judgePath : null,
|
|
144
|
+
hooks: join(taskDir, "hooks"),
|
|
145
|
+
preflight: (await fileExecutable(fs, preflightPath))
|
|
146
|
+
? preflightPath
|
|
147
|
+
: null,
|
|
148
|
+
invariants: (await fileExecutable(fs, invariantsPath))
|
|
149
|
+
? invariantsPath
|
|
150
|
+
: null,
|
|
151
|
+
tests: suite ? join(taskDir, "tests") : null,
|
|
152
|
+
specs: join(taskDir, "specs"),
|
|
153
|
+
workdir: join(taskDir, "workdir"),
|
|
154
|
+
},
|
|
155
|
+
tests: suite,
|
|
156
|
+
};
|
|
157
|
+
}
|
|
158
|
+
|
|
159
|
+
const CHECK_SUFFIX = ".test.js";
|
|
160
|
+
const GATE_SUFFIX = ".gate.test.js";
|
|
161
|
+
|
|
162
|
+
/**
|
|
163
|
+
* Discover and validate a task's hidden test suite under `<taskDir>/tests/`.
|
|
164
|
+
* Returns null when the directory is absent; throws on an invalid layout
|
|
165
|
+
* (no check files, a dangling symlink, duplicate check names) so
|
|
166
|
+
* `loadTaskFamily` rejects broken suites before any agent spend.
|
|
167
|
+
* @param {object} fs - Async filesystem surface (`runtime.fs`).
|
|
168
|
+
* @param {string} taskDir
|
|
169
|
+
* @returns {Promise<HiddenSuite | null>}
|
|
170
|
+
*/
|
|
171
|
+
async function discoverSuite(fs, taskDir) {
|
|
172
|
+
const testsRoot = join(taskDir, "tests");
|
|
173
|
+
try {
|
|
174
|
+
const st = await fs.lstat(testsRoot);
|
|
175
|
+
if (!st.isDirectory()) return null;
|
|
176
|
+
} catch {
|
|
177
|
+
return null;
|
|
178
|
+
}
|
|
179
|
+
const files = [];
|
|
180
|
+
await walkSuiteFiles(fs, testsRoot, testsRoot, files);
|
|
181
|
+
files.sort((a, b) =>
|
|
182
|
+
a.stagePath < b.stagePath ? -1 : a.stagePath > b.stagePath ? 1 : 0,
|
|
183
|
+
);
|
|
184
|
+
|
|
185
|
+
const checks = [];
|
|
186
|
+
const support = [];
|
|
187
|
+
for (const file of files) {
|
|
188
|
+
const base = file.stagePath.split(sep).at(-1);
|
|
189
|
+
if (!base.endsWith(CHECK_SUFFIX)) {
|
|
190
|
+
support.push(file);
|
|
191
|
+
continue;
|
|
192
|
+
}
|
|
193
|
+
const gate = base.endsWith(GATE_SUFFIX);
|
|
194
|
+
const name = base.slice(0, -(gate ? GATE_SUFFIX : CHECK_SUFFIX).length);
|
|
195
|
+
checks.push({ name, gate, ...file });
|
|
196
|
+
}
|
|
197
|
+
|
|
198
|
+
if (checks.length === 0) {
|
|
199
|
+
throw new Error(`hidden test suite has no check files: ${testsRoot}`);
|
|
200
|
+
}
|
|
201
|
+
const seen = new Set();
|
|
202
|
+
for (const check of checks) {
|
|
203
|
+
if (seen.has(check.name)) {
|
|
204
|
+
throw new Error(
|
|
205
|
+
`hidden test suite has duplicate check name '${check.name}': ${check.sourcePath}`,
|
|
206
|
+
);
|
|
207
|
+
}
|
|
208
|
+
seen.add(check.name);
|
|
209
|
+
}
|
|
210
|
+
return { checks, support };
|
|
211
|
+
}
|
|
212
|
+
|
|
213
|
+
/**
|
|
214
|
+
* Walk a suite tree collecting `{sourcePath, stagePath}` entries. Unlike
|
|
215
|
+
* `walkFiles` (which silently skips dangling symlinks for hashing), every
|
|
216
|
+
* entry here must be a regular file after symlink resolution — a dangling
|
|
217
|
+
* symlink or a link to a non-file target is an authoring error.
|
|
218
|
+
*/
|
|
219
|
+
async function walkSuiteFiles(fs, root, dir, out) {
|
|
220
|
+
const entries = await fs.readdir(dir, { withFileTypes: true });
|
|
221
|
+
for (const entry of entries) {
|
|
222
|
+
const full = join(dir, entry.name);
|
|
223
|
+
if (entry.isDirectory()) {
|
|
224
|
+
await walkSuiteFiles(fs, root, full, out);
|
|
225
|
+
} else if (entry.isFile()) {
|
|
226
|
+
out.push({ sourcePath: full, stagePath: relative(root, full) });
|
|
227
|
+
} else if (entry.isSymbolicLink()) {
|
|
228
|
+
const resolved = await resolveSymlinkToFile(fs, full);
|
|
229
|
+
if (!resolved) {
|
|
230
|
+
throw new Error(
|
|
231
|
+
`hidden test suite entry is not a regular file: ${full}`,
|
|
232
|
+
);
|
|
233
|
+
}
|
|
234
|
+
out.push({ sourcePath: full, stagePath: relative(root, full) });
|
|
235
|
+
} else {
|
|
236
|
+
throw new Error(`hidden test suite entry is not a regular file: ${full}`);
|
|
237
|
+
}
|
|
238
|
+
}
|
|
239
|
+
}
|
|
240
|
+
|
|
146
241
|
async function fileExists(fs, path) {
|
|
147
242
|
try {
|
|
148
243
|
await fs.access(path);
|
|
@@ -246,10 +341,28 @@ async function git(runtime, args) {
|
|
|
246
341
|
return stdout;
|
|
247
342
|
}
|
|
248
343
|
|
|
344
|
+
/**
|
|
345
|
+
* @typedef {object} HiddenCheck
|
|
346
|
+
* @property {string} name - Basename stem with the check suffix stripped.
|
|
347
|
+
* @property {boolean} gate - True iff the filename ends `.gate.test.js`.
|
|
348
|
+
* @property {string} sourcePath - Absolute path under `tests/`.
|
|
349
|
+
* @property {string} stagePath - Path relative to `tests/` — the overlay
|
|
350
|
+
* mirror of the staging path under the agent CWD.
|
|
351
|
+
*/
|
|
352
|
+
|
|
353
|
+
/**
|
|
354
|
+
* @typedef {object} HiddenSuite
|
|
355
|
+
* @property {HiddenCheck[]} checks - In sorted stage-path order.
|
|
356
|
+
* @property {{sourcePath: string, stagePath: string}[]} support - Non-check
|
|
357
|
+
* files, staged for the whole pass but never graded.
|
|
358
|
+
*/
|
|
359
|
+
|
|
249
360
|
/**
|
|
250
361
|
* @typedef {object} Task
|
|
251
362
|
* @property {string} id - Task name (directory name under tasks/)
|
|
252
|
-
* @property {{taskDir: string, instructions: string, supervisor: string|null, judge: string|null, hooks: string, preflight: string|null, invariants: string|null, specs: string, workdir: string}} paths
|
|
363
|
+
* @property {{taskDir: string, instructions: string, supervisor: string|null, judge: string|null, hooks: string, preflight: string|null, invariants: string|null, tests: string|null, specs: string, workdir: string}} paths
|
|
364
|
+
* @property {HiddenSuite | null} tests - Hidden test suite (null when the
|
|
365
|
+
* task ships no `tests/` directory)
|
|
253
366
|
*/
|
|
254
367
|
|
|
255
368
|
/**
|
|
@@ -0,0 +1,73 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Combined-supervisor-trace splitting for the benchmark runner: one pass
|
|
3
|
+
* over the tagged NDJSON envelope stream separates agent events from
|
|
4
|
+
* supervisor/orchestrator events and extracts the run summary.
|
|
5
|
+
*/
|
|
6
|
+
|
|
7
|
+
import { createInterface } from "node:readline";
|
|
8
|
+
|
|
9
|
+
/**
|
|
10
|
+
* Split the combined supervisor trace into agent and supervisor files and
|
|
11
|
+
* extract turn count and submission in a single pass. Agent-source events go
|
|
12
|
+
* to `agentPath`; supervisor and orchestrator events go to `supervisorPath`.
|
|
13
|
+
*
|
|
14
|
+
* Cost is deliberately not summed here — the caller derives it from the same
|
|
15
|
+
* combined trace via `sumTraceCost`, so there is one cost path across the
|
|
16
|
+
* benchmark, callback, and `fit-trace cost` consumers.
|
|
17
|
+
* @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
|
|
18
|
+
* @param {string} combinedPath
|
|
19
|
+
* @param {string} agentPath
|
|
20
|
+
* @param {string} supervisorPath
|
|
21
|
+
* @returns {Promise<{turns: number, submission: string}>}
|
|
22
|
+
*/
|
|
23
|
+
// biome-ignore lint/complexity/noExcessiveCognitiveComplexity: stream-splitting state machine
|
|
24
|
+
export async function splitAndSummarize(
|
|
25
|
+
runtime,
|
|
26
|
+
combinedPath,
|
|
27
|
+
agentPath,
|
|
28
|
+
supervisorPath,
|
|
29
|
+
) {
|
|
30
|
+
const fs = runtime.fs;
|
|
31
|
+
const agentStream = fs.createWriteStream(agentPath);
|
|
32
|
+
const supStream = fs.createWriteStream(supervisorPath);
|
|
33
|
+
const rl = createInterface({
|
|
34
|
+
input: fs.createReadStream(combinedPath),
|
|
35
|
+
crlfDelay: Infinity,
|
|
36
|
+
});
|
|
37
|
+
let turns = 0;
|
|
38
|
+
let submission = "";
|
|
39
|
+
for await (const line of rl) {
|
|
40
|
+
if (!line.trim()) continue;
|
|
41
|
+
let event;
|
|
42
|
+
try {
|
|
43
|
+
event = JSON.parse(line);
|
|
44
|
+
} catch {
|
|
45
|
+
continue;
|
|
46
|
+
}
|
|
47
|
+
const target = event.source === "agent" ? agentStream : supStream;
|
|
48
|
+
target.write(line + "\n");
|
|
49
|
+
const inner = event.event;
|
|
50
|
+
if (!inner) continue;
|
|
51
|
+
if (event.source === "agent" && inner.type === "assistant") {
|
|
52
|
+
const text = extractText(inner);
|
|
53
|
+
if (text) submission = text;
|
|
54
|
+
}
|
|
55
|
+
if (event.source === "orchestrator" && inner.type === "summary") {
|
|
56
|
+
turns = inner.turns ?? 0;
|
|
57
|
+
}
|
|
58
|
+
}
|
|
59
|
+
await Promise.all([
|
|
60
|
+
new Promise((r) => agentStream.end(r)),
|
|
61
|
+
new Promise((r) => supStream.end(r)),
|
|
62
|
+
]);
|
|
63
|
+
return { turns, submission };
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
function extractText(inner) {
|
|
67
|
+
const content = inner.message?.content ?? inner.content;
|
|
68
|
+
if (!Array.isArray(content)) return null;
|
|
69
|
+
for (let i = content.length - 1; i >= 0; i--) {
|
|
70
|
+
if (content[i].type === "text" && content[i].text) return content[i].text;
|
|
71
|
+
}
|
|
72
|
+
return null;
|
|
73
|
+
}
|
package/src/benchmark/workdir.js
CHANGED
|
@@ -268,7 +268,12 @@ async function runPreflight(runtime, script, cwd, port, vars) {
|
|
|
268
268
|
};
|
|
269
269
|
}
|
|
270
270
|
|
|
271
|
-
|
|
271
|
+
/**
|
|
272
|
+
* Allocate a free TCP port by binding to 0 and releasing it. Shared with the
|
|
273
|
+
* `grade` subcommand, which needs a plausible `$PORT` for the hook env.
|
|
274
|
+
* @returns {Promise<number>}
|
|
275
|
+
*/
|
|
276
|
+
export function probeFreePort() {
|
|
272
277
|
return new Promise((res, rej) => {
|
|
273
278
|
const server = createServer();
|
|
274
279
|
server.unref();
|
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Shared advisor-flag parsing for the four session-mode commands. The two
|
|
3
|
+
* flags are identical everywhere: `--advisor-model` (no default — absent
|
|
4
|
+
* means the Advisor tool is not offered) and `--advisor-max-uses`
|
|
5
|
+
* (default 3), which is a usage error without the model flag.
|
|
6
|
+
*/
|
|
7
|
+
|
|
8
|
+
/**
|
|
9
|
+
* Parse `--advisor-model` / `--advisor-max-uses` from parsed option values.
|
|
10
|
+
* A malformed max-uses is a usage error, not a silent fallback: NaN would
|
|
11
|
+
* make the budget check (`used >= maxUses`) permanently false and disable
|
|
12
|
+
* the code-enforced cap the flag exists to guarantee.
|
|
13
|
+
* @param {object} values - Parsed option values from cli.parse()
|
|
14
|
+
* @returns {{advisorModel: string|undefined, advisorMaxUses: number}}
|
|
15
|
+
*/
|
|
16
|
+
export function parseAdvisorOptions(values) {
|
|
17
|
+
if (values["advisor-max-uses"] && !values["advisor-model"]) {
|
|
18
|
+
throw new Error("--advisor-max-uses requires --advisor-model");
|
|
19
|
+
}
|
|
20
|
+
const advisorMaxUses = parseInt(values["advisor-max-uses"] || "3", 10);
|
|
21
|
+
if (Number.isNaN(advisorMaxUses) || advisorMaxUses < 1) {
|
|
22
|
+
throw new Error("--advisor-max-uses must be a positive integer");
|
|
23
|
+
}
|
|
24
|
+
return {
|
|
25
|
+
advisorModel: values["advisor-model"] || undefined,
|
|
26
|
+
advisorMaxUses,
|
|
27
|
+
};
|
|
28
|
+
}
|