@forwardimpact/libharness 2.0.0 → 3.0.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +68 -65
- package/package.json +15 -13
- package/src/advisor.js +47 -41
- package/src/agent-runner.js +58 -48
- package/src/benchmark/apm-installer.js +28 -28
- package/src/benchmark/env-loader.js +24 -16
- package/src/benchmark/grade.js +44 -41
- package/src/benchmark/hidden-tests.js +25 -24
- package/src/benchmark/hook-env.js +11 -9
- package/src/benchmark/invariants.js +20 -17
- package/src/benchmark/judge.js +29 -28
- package/src/benchmark/npm-installer.js +9 -8
- package/src/benchmark/report.js +53 -50
- package/src/benchmark/result.js +24 -23
- package/src/benchmark/runner.js +75 -69
- package/src/benchmark/scheduler.js +17 -16
- package/src/benchmark/task-family.js +29 -27
- package/src/benchmark/trace-split.js +9 -8
- package/src/benchmark/workdir.js +27 -25
- package/src/claude-code-executable.js +11 -11
- package/src/commands/advisor-flags.js +8 -7
- package/src/commands/assert.js +16 -15
- package/src/commands/benchmark-definition.js +20 -20
- package/src/commands/benchmark-grade.js +13 -12
- package/src/commands/benchmark-report.js +5 -5
- package/src/commands/benchmark-run.js +31 -28
- package/src/commands/by-discussion.js +11 -11
- package/src/commands/callback.js +11 -11
- package/src/commands/discuss.js +8 -7
- package/src/commands/facilitate.js +16 -14
- package/src/commands/output.js +4 -3
- package/src/commands/run.js +15 -15
- package/src/commands/scan-logs.js +22 -20
- package/src/commands/selfedit.js +124 -0
- package/src/commands/supervise.js +13 -11
- package/src/commands/task-input.js +9 -9
- package/src/commands/tee.js +11 -10
- package/src/commands/trace.js +55 -42
- package/src/commands/work-tracker.js +4 -3
- package/src/cost.js +17 -17
- package/src/discuss-tools.js +16 -16
- package/src/discusser.js +39 -38
- package/src/events/github.js +54 -37
- package/src/facilitator.js +21 -21
- package/src/inbox-poller.js +4 -4
- package/src/judge.js +32 -30
- package/src/message-bus.js +12 -11
- package/src/orchestration-loop.js +35 -36
- package/src/orchestration-toolkit.js +58 -53
- package/src/orchestrator-helpers.js +2 -2
- package/src/profile-prompt.js +54 -53
- package/src/redaction.js +63 -57
- package/src/render/line-renderer.js +5 -5
- package/src/render/orchestrator-filter.js +3 -3
- package/src/render/palette.js +11 -9
- package/src/render/tool-hints.js +18 -15
- package/src/render/turn-renderer.js +4 -4
- package/src/reply-emitter.js +2 -2
- package/src/sequence-counter.js +4 -3
- package/src/signature-filter.js +7 -6
- package/src/supervisor.js +19 -18
- package/src/tee-writer.js +25 -25
- package/src/trace-collector.js +53 -48
- package/src/trace-github.js +53 -44
- package/src/trace-multi.js +16 -14
- package/src/trace-query.js +61 -52
- package/src/trace-render.js +19 -19
- package/src/trace-usage.js +31 -28
- package/src/transcript-recorder.js +24 -20
- package/bin/fit-benchmark.js +0 -44
- package/bin/fit-harness.js +0 -412
- package/bin/fit-selfedit.js +0 -165
- package/bin/fit-trace.js +0 -520
package/src/benchmark/result.js
CHANGED
|
@@ -3,16 +3,17 @@
|
|
|
3
3
|
*
|
|
4
4
|
* Two schemas live here:
|
|
5
5
|
* - RESULT_RECORD_SCHEMA — one record per (task, runIndex) from a full
|
|
6
|
-
* benchmark run.
|
|
7
|
-
* and a pre-flight-failure branch (grade/judgeVerdict/submission
|
|
8
|
-
*
|
|
9
|
-
*
|
|
6
|
+
* benchmark run. It has a happy branch (grade + collectors + judge
|
|
7
|
+
* present) and a pre-flight-failure branch (grade/judgeVerdict/submission
|
|
8
|
+
* absent).
|
|
9
|
+
* - GRADE_RECORD_SCHEMA — narrower output of `benchmark-grade`, which
|
|
10
|
+
* grades ad hoc without a full lifecycle.
|
|
10
11
|
*
|
|
11
|
-
* The check rows are the authoritative
|
|
12
|
-
* requires a `grade` object, so a pre-break record fails validation
|
|
13
|
-
*
|
|
12
|
+
* The check rows are the authoritative channel for grades. The happy branch
|
|
13
|
+
* requires a `grade` object, so a pre-break record fails validation. It does
|
|
14
|
+
* not render under semantics it never carried.
|
|
14
15
|
*
|
|
15
|
-
*
|
|
16
|
+
* The validators throw on mismatch, so the runner can wrap every JSONL append
|
|
16
17
|
* in a guard and reject schema drift at write time.
|
|
17
18
|
*/
|
|
18
19
|
|
|
@@ -27,8 +28,8 @@ const INVARIANTS_SHAPE = z.object({
|
|
|
27
28
|
});
|
|
28
29
|
|
|
29
30
|
/**
|
|
30
|
-
* The normalized
|
|
31
|
-
* `malformed` only when at least one row was malformed.
|
|
31
|
+
* The normalized projection of a grade. `score` appears only on scored
|
|
32
|
+
* tasks. `malformed` appears only when at least one row was malformed.
|
|
32
33
|
*/
|
|
33
34
|
const GRADE_SHAPE = z.object({
|
|
34
35
|
verdict: VERDICT_ENUM,
|
|
@@ -48,9 +49,9 @@ const JUDGE_VERDICT_SHAPE = z.object({
|
|
|
48
49
|
});
|
|
49
50
|
|
|
50
51
|
/**
|
|
51
|
-
* Per-participant cost attribution. `costUsd` is the sum of these
|
|
52
|
+
* Per-participant cost attribution. `costUsd` is the sum of these. The
|
|
52
53
|
* breakdown lets reports show where the spend went. The judge runs as its
|
|
53
|
-
* own SDK session, so its cost
|
|
54
|
+
* own SDK session, so its cost stays separate from agent/supervisor.
|
|
54
55
|
*/
|
|
55
56
|
const COST_BREAKDOWN_SHAPE = z.object({
|
|
56
57
|
agent: z.number(),
|
|
@@ -98,8 +99,8 @@ const HAPPY_RECORD = z.object({
|
|
|
98
99
|
invariants: INVARIANTS_SHAPE,
|
|
99
100
|
grade: GRADE_SHAPE,
|
|
100
101
|
hiddenTests: HIDDEN_TESTS_SHAPE.optional(),
|
|
101
|
-
// The effective, judge-zeroed score `report` aggregates
|
|
102
|
-
// scored tasks.
|
|
102
|
+
// The effective, judge-zeroed score `report` aggregates. It is present
|
|
103
|
+
// only on scored tasks.
|
|
103
104
|
score: z.number().min(0).max(1).optional(),
|
|
104
105
|
submission: z.string(),
|
|
105
106
|
judgeVerdict: JUDGE_VERDICT_SHAPE.optional(),
|
|
@@ -114,9 +115,9 @@ const PREFLIGHT_RECORD = z.object({
|
|
|
114
115
|
...COMMON_FIELDS,
|
|
115
116
|
costUsd: z.literal(0),
|
|
116
117
|
preflightError: PREFLIGHT_ERROR_SHAPE,
|
|
117
|
-
//
|
|
118
|
-
//
|
|
119
|
-
//
|
|
118
|
+
// The runner allocates the trace paths in WorkdirManager.start, even on
|
|
119
|
+
// preflight failure. The record then stays uniform across branches, and
|
|
120
|
+
// downstream consumers can reference the paths without conditional fields.
|
|
120
121
|
agentTracePath: z.string(),
|
|
121
122
|
supervisorTracePath: z.string(),
|
|
122
123
|
judgeTracePath: z.string(),
|
|
@@ -133,15 +134,15 @@ export const RESULT_RECORD_SCHEMA = z.union([HAPPY_RECORD, PREFLIGHT_RECORD]);
|
|
|
133
134
|
|
|
134
135
|
export const GRADE_RECORD_SCHEMA = z.object({
|
|
135
136
|
taskId: z.string().min(1),
|
|
136
|
-
//
|
|
137
|
-
//
|
|
138
|
-
//
|
|
139
|
-
//
|
|
137
|
+
// In the happy result record, `grade.score` is the raw weighted fraction,
|
|
138
|
+
// and the effective (zeroed) value lives on the top-level `score`. This
|
|
139
|
+
// record has no second score field, so its `grade.score` carries the
|
|
140
|
+
// effective health/gate-zeroed value.
|
|
140
141
|
grade: GRADE_SHAPE,
|
|
141
142
|
invariants: INVARIANTS_SHAPE,
|
|
142
143
|
hiddenTests: HIDDEN_TESTS_SHAPE.optional(),
|
|
143
|
-
//
|
|
144
|
-
//
|
|
144
|
+
// This mirrors the invariants script's exit for diagnosis. The graded
|
|
145
|
+
// verdict drives the command's process exit.
|
|
145
146
|
exitCode: z.number().int(),
|
|
146
147
|
});
|
|
147
148
|
|
package/src/benchmark/runner.js
CHANGED
|
@@ -4,19 +4,19 @@
|
|
|
4
4
|
* Phases per (task, runIndex):
|
|
5
5
|
* 1. WorkdirManager.start → seed CWD + run pre-flight probe
|
|
6
6
|
* 2. Supervisor session (agent + supervisor) → produce traces + submission
|
|
7
|
-
* 3. Invariants collector + hidden-test engine → merged check rows
|
|
8
|
-
*
|
|
9
|
-
* grader health only
|
|
7
|
+
* 3. Invariants collector + hidden-test engine → merged check rows.
|
|
8
|
+
* `gradeChecks` grades them. The rows are authoritative. The script
|
|
9
|
+
* exit reports grader health only.
|
|
10
10
|
* 4. Judge.runJudge → Conclude-driven binary gate mapped to pass/fail
|
|
11
11
|
* 5. WorkdirManager.teardown → process-group cleanup
|
|
12
12
|
*
|
|
13
|
-
* Cells run with bounded in-process concurrency (`CellScheduler`)
|
|
14
|
-
* yields records in **completion order
|
|
15
|
-
* is the sole writer of
|
|
16
|
-
*
|
|
17
|
-
*
|
|
18
|
-
*
|
|
19
|
-
* stream.
|
|
13
|
+
* Cells run with bounded in-process concurrency (`CellScheduler`). `run()`
|
|
14
|
+
* yields records in **completion order**. It does not yield them in grid
|
|
15
|
+
* order. A single drain loop is the sole writer of
|
|
16
|
+
* `<output>/results.jsonl`. It appends each record the moment its cell
|
|
17
|
+
* settles. That incremental append is the durability and crash-safety
|
|
18
|
+
* mechanism, so a killed run keeps every completed cell. There is no sidecar
|
|
19
|
+
* ledger. The iterator mirrors the same stream to CLI stdout.
|
|
20
20
|
*/
|
|
21
21
|
|
|
22
22
|
import { join, resolve as resolvePath } from "node:path";
|
|
@@ -47,11 +47,11 @@ const BASE_TOOLS = [
|
|
|
47
47
|
"TodoWrite",
|
|
48
48
|
];
|
|
49
49
|
|
|
50
|
-
// Upper bound on a single supervised agent run.
|
|
51
|
-
// message within this window
|
|
52
|
-
// agentError, so the benchmark never hangs the event loop into a
|
|
53
|
-
//
|
|
54
|
-
//
|
|
50
|
+
// Upper bound on a single supervised agent run. The runner treats a run that
|
|
51
|
+
// produces no terminal message within this window as a stall. It records the
|
|
52
|
+
// stall as an agentError, so the benchmark never hangs the event loop into a
|
|
53
|
+
// silent exit. Set `watchdogMs` per runner to override this default. A test
|
|
54
|
+
// can then force a stall and skip the full 20-minute wait.
|
|
55
55
|
const AGENT_WATCHDOG_MS = 20 * 60 * 1000;
|
|
56
56
|
|
|
57
57
|
/** Sole orchestrator for a task-family benchmark run. */
|
|
@@ -65,25 +65,26 @@ export class BenchmarkRunner {
|
|
|
65
65
|
* @param {string} opts.supervisorModel
|
|
66
66
|
* @param {string} opts.judgeModel
|
|
67
67
|
* @param {{agent?: string, judge?: string}} [opts.profiles]
|
|
68
|
-
* @param {Function} opts.query - SDK query
|
|
68
|
+
* @param {Function} opts.query - SDK query. A test injects its own.
|
|
69
69
|
* @param {string[]} [opts.allowedTools] - Agent tool allowlist (default: BASE_TOOLS).
|
|
70
70
|
* @param {number} [opts.maxTurns] - Agent-under-test turn budget.
|
|
71
71
|
* @param {number} [opts.concurrency] - Max cells in flight (integer ≥ 1).
|
|
72
|
-
* Defaults to 1 as a defensive floor
|
|
72
|
+
* Defaults to 1 as a defensive floor. The CLI always passes a resolved value.
|
|
73
73
|
* @param {{index: number, total: number}} [opts.shard] - Run only the cells
|
|
74
74
|
* assigned to shard `index` of `total` (1-based). Absent ≡ the whole grid
|
|
75
75
|
* (identity `1/1`).
|
|
76
76
|
* @param {number} [opts.watchdogMs] - Per-agent stall watchdog (ms). Defaults
|
|
77
|
-
* to `AGENT_WATCHDOG_MS
|
|
77
|
+
* to `AGENT_WATCHDOG_MS`. A test injects its own to force a stall in-test.
|
|
78
78
|
* @param {number} [opts.termGraceMs] - SIGTERM→SIGKILL grace (ms) for the per-task process group.
|
|
79
79
|
* @param {Function} [opts.runAgent] - Test seam: replaces the agent-under-test
|
|
80
80
|
* session. Must return `{costUsd, turns, submission, agentError?}` and
|
|
81
81
|
* write a valid NDJSON trace to `workdir.agentTracePath`. Default uses
|
|
82
82
|
* `createAgentRunner` with the harness `BASE_TOOLS` allowlist. Internal
|
|
83
|
-
* testing only
|
|
83
|
+
* testing only. It is not part of the public API.
|
|
84
84
|
* @param {import("@forwardimpact/libutil/runtime").Runtime} opts.runtime -
|
|
85
|
-
*
|
|
86
|
-
*
|
|
85
|
+
* The host injects these ambient collaborators (`fs`, `subprocess`,
|
|
86
|
+
* `clock`, `proc`). The runner threads them into the installers, the
|
|
87
|
+
* workdir manager, the invariants, and the judge.
|
|
87
88
|
* @param {Function} [opts.runInvariants] - Test seam: replaces `runInvariants`.
|
|
88
89
|
* Same contract as `runInvariants(task, ctx, runtime)`. Internal testing only.
|
|
89
90
|
* @param {Function} [opts.runHiddenTests] - Test seam: replaces
|
|
@@ -203,9 +204,9 @@ export class BenchmarkRunner {
|
|
|
203
204
|
});
|
|
204
205
|
|
|
205
206
|
const allCells = enumerateCells(tasks, this.runs);
|
|
206
|
-
//
|
|
207
|
-
// the identity 1/1. A high-index shard may select zero cells
|
|
208
|
-
// whose results.jsonl ends up empty.
|
|
207
|
+
// A shard selects a deterministic subset of the grid. An unsharded run is
|
|
208
|
+
// the identity 1/1. A high-index shard may select zero cells. That is a
|
|
209
|
+
// valid run whose results.jsonl ends up empty.
|
|
209
210
|
const cells = this.shard
|
|
210
211
|
? selectShard(allCells, this.shard.index, this.shard.total)
|
|
211
212
|
: allCells;
|
|
@@ -228,8 +229,8 @@ export class BenchmarkRunner {
|
|
|
228
229
|
});
|
|
229
230
|
// Single-writer drain: the scheduler runs up to `concurrency` cells at
|
|
230
231
|
// once and pushes each settled record here in completion order. This loop
|
|
231
|
-
// is the sole writer of `results.jsonl
|
|
232
|
-
//
|
|
232
|
+
// is the sole writer of `results.jsonl`. Workers never touch the stream.
|
|
233
|
+
// The per-completion append is the crash-safety mechanism.
|
|
233
234
|
try {
|
|
234
235
|
for await (const record of scheduler.run(cells)) {
|
|
235
236
|
await writeRecord(resultsStream, record);
|
|
@@ -255,11 +256,12 @@ export class BenchmarkRunner {
|
|
|
255
256
|
t0,
|
|
256
257
|
});
|
|
257
258
|
} catch (e) {
|
|
258
|
-
// `wm.start()` (port acquire + workdir/env
|
|
259
|
-
//
|
|
260
|
-
// record so `#runOne` never rejects
|
|
261
|
-
// contract depends on that.
|
|
262
|
-
// the same as any other
|
|
259
|
+
// `wm.start()` (port acquire + workdir/env seed) is the one throw site
|
|
260
|
+
// that `#executeCell` does not catch. Turn it into the runner's own
|
|
261
|
+
// fallback record so `#runOne` never rejects. The scheduler's
|
|
262
|
+
// one-record-per-cell contract depends on that. `report` skips the
|
|
263
|
+
// fallback because it fails the schema, the same as any other
|
|
264
|
+
// runner-side schema failure.
|
|
263
265
|
return {
|
|
264
266
|
taskId: task.id,
|
|
265
267
|
runIndex,
|
|
@@ -273,9 +275,9 @@ export class BenchmarkRunner {
|
|
|
273
275
|
|
|
274
276
|
/**
|
|
275
277
|
* Run one cell's lifecycle against an already-started workdir: preflight
|
|
276
|
-
* gate → supervised agent → invariants → judge → assembled record.
|
|
277
|
-
* from `#runOne` so the start/teardown/error wrapper stays under
|
|
278
|
-
* complexity ceiling.
|
|
278
|
+
* gate → supervised agent → invariants → judge → assembled record. It is
|
|
279
|
+
* separate from `#runOne` so the start/teardown/error wrapper stays under
|
|
280
|
+
* the complexity ceiling.
|
|
279
281
|
*/
|
|
280
282
|
async #executeCell({
|
|
281
283
|
family,
|
|
@@ -313,9 +315,9 @@ export class BenchmarkRunner {
|
|
|
313
315
|
const judgePass =
|
|
314
316
|
judgeVerdict === null || judgeVerdict.verdict === "pass";
|
|
315
317
|
const verdict = grade.verdict === "pass" && judgePass ? "pass" : "fail";
|
|
316
|
-
// Gates protect the score: an unhealthy grader, a
|
|
317
|
-
// a
|
|
318
|
-
// fractional score with verdict fail is the point.
|
|
318
|
+
// Gates protect the score: an unhealthy grader, a gate row that fails,
|
|
319
|
+
// or a judge that fails zeroes the effective score. Full marks does not
|
|
320
|
+
// zero it. A fractional score with verdict fail is the point.
|
|
319
321
|
const scoreValid = graded.healthy && grade.gatesPass && judgePass;
|
|
320
322
|
const record = {
|
|
321
323
|
taskId: task.id,
|
|
@@ -365,8 +367,8 @@ export class BenchmarkRunner {
|
|
|
365
367
|
|
|
366
368
|
/**
|
|
367
369
|
* Run the judge (when the task ships a template) over the grade result.
|
|
368
|
-
* The record's judgeVerdict carries only the verdict + summary
|
|
369
|
-
* judge's cost
|
|
370
|
+
* The record's judgeVerdict carries only the verdict + summary. The runner
|
|
371
|
+
* folds the judge's cost into costUsd / costBreakdown instead.
|
|
370
372
|
*/
|
|
371
373
|
async #judgeCell({
|
|
372
374
|
task,
|
|
@@ -404,10 +406,9 @@ export class BenchmarkRunner {
|
|
|
404
406
|
}
|
|
405
407
|
|
|
406
408
|
/**
|
|
407
|
-
* Run both check-row producers against the post-run CWD
|
|
408
|
-
*
|
|
409
|
-
*
|
|
410
|
-
* agent left it.
|
|
409
|
+
* Run both check-row producers against the post-run CWD. Grade the merged
|
|
410
|
+
* rows through the shared derivation. The engine restores the workdir, so
|
|
411
|
+
* the judge (which runs after) sees it exactly as the agent left it.
|
|
411
412
|
*/
|
|
412
413
|
#gradeCell(family, task, workdir) {
|
|
413
414
|
const ctx = {
|
|
@@ -424,9 +425,9 @@ export class BenchmarkRunner {
|
|
|
424
425
|
|
|
425
426
|
/**
|
|
426
427
|
* Dispatch to either the injected hook or the default `#runAgent`. Either
|
|
427
|
-
* path can throw
|
|
428
|
-
*
|
|
429
|
-
*
|
|
428
|
+
* path can throw. Catch the error here so it becomes an `agentError` on the
|
|
429
|
+
* record (spec criterion 1: records on agent failure). The iterator then
|
|
430
|
+
* does not abort.
|
|
430
431
|
*/
|
|
431
432
|
async #runAgentSafe(task, workdir) {
|
|
432
433
|
try {
|
|
@@ -448,8 +449,9 @@ export class BenchmarkRunner {
|
|
|
448
449
|
|
|
449
450
|
/**
|
|
450
451
|
* Run the agent-under-test under a Supervisor. The supervisor writes
|
|
451
|
-
* a combined tagged NDJSON trace
|
|
452
|
-
* agent.ndjson and supervisor.ndjson
|
|
452
|
+
* a combined tagged NDJSON trace. After the session, this method splits
|
|
453
|
+
* the trace into agent.ndjson and supervisor.ndjson. It also extracts
|
|
454
|
+
* cost/turns/submission.
|
|
453
455
|
*/
|
|
454
456
|
async #runAgent(task, workdir) {
|
|
455
457
|
const fs = this.runtime.fs;
|
|
@@ -477,11 +479,12 @@ export class BenchmarkRunner {
|
|
|
477
479
|
});
|
|
478
480
|
const instructions = await fs.readFile(task.paths.instructions, "utf8");
|
|
479
481
|
let agentError = null;
|
|
480
|
-
// Watchdog: a supervised session can hang
|
|
481
|
-
// SDK subprocess exits without a terminal message)
|
|
482
|
-
// event loop and exit the process mid-run with zero records. Race the
|
|
483
|
-
// against a bounded timer so a stall becomes an `agentError` record
|
|
484
|
-
// of a silent exit
|
|
482
|
+
// Watchdog: a supervised session can hang and never settle (e.g. the agent
|
|
483
|
+
// SDK subprocess exits without a terminal message). The hang would empty
|
|
484
|
+
// the event loop and exit the process mid-run with zero records. Race the
|
|
485
|
+
// run against a bounded timer so a stall becomes an `agentError` record
|
|
486
|
+
// instead of a silent exit. The timer also keeps the loop alive until it
|
|
487
|
+
// fires.
|
|
485
488
|
let watchdog;
|
|
486
489
|
try {
|
|
487
490
|
const result = await Promise.race([
|
|
@@ -513,8 +516,9 @@ export class BenchmarkRunner {
|
|
|
513
516
|
workdir.agentTracePath,
|
|
514
517
|
workdir.supervisorTracePath,
|
|
515
518
|
);
|
|
516
|
-
//
|
|
517
|
-
// combined trace,
|
|
519
|
+
// `sumTraceCost` sums the cost across every participant's result events
|
|
520
|
+
// from the one combined trace, and attributes it per source. Read the
|
|
521
|
+
// trace before you unlink it.
|
|
518
522
|
const combined = await fs.readFile(combinedPath, "utf8");
|
|
519
523
|
const { totalCostUsd, bySource } = sumTraceCost(combined.split("\n"));
|
|
520
524
|
await fs.unlink(combinedPath).catch(() => {});
|
|
@@ -586,9 +590,9 @@ export class BenchmarkRunner {
|
|
|
586
590
|
validateResultRecord(record);
|
|
587
591
|
return record;
|
|
588
592
|
} catch (e) {
|
|
589
|
-
// The runner constructed the record
|
|
590
|
-
//
|
|
591
|
-
// consumable and the agent budget
|
|
593
|
+
// The runner constructed the record. A schema failure is a real bug.
|
|
594
|
+
// Bad family input is not the cause. Emit a noisy fallback so the
|
|
595
|
+
// iterator stays consumable and nothing silently drops the agent budget.
|
|
592
596
|
return {
|
|
593
597
|
taskId: record.taskId ?? key.taskId,
|
|
594
598
|
runIndex: record.runIndex ?? key.runIndex,
|
|
@@ -601,9 +605,10 @@ export class BenchmarkRunner {
|
|
|
601
605
|
|
|
602
606
|
/**
|
|
603
607
|
* Flatten the grid into a stable ordered cell list, task-major /
|
|
604
|
-
* runIndex-minor.
|
|
605
|
-
*
|
|
606
|
-
* the cell list for both the scheduler and
|
|
608
|
+
* runIndex-minor. The order is load-bearing. A task's runIndexes are adjacent
|
|
609
|
+
* in this list, and Part 02's round-robin shard balance depends on that. This
|
|
610
|
+
* function is the single source of the cell list for both the scheduler and
|
|
611
|
+
* the shard selector.
|
|
607
612
|
* @param {import("./task-family.js").Task[]} tasks
|
|
608
613
|
* @param {number} runs
|
|
609
614
|
* @returns {{task: import("./task-family.js").Task, runIndex: number}[]}
|
|
@@ -619,11 +624,11 @@ export function enumerateCells(tasks, runs) {
|
|
|
619
624
|
/**
|
|
620
625
|
* Round-robin partition of the enumerated cells: the cell at position `p` runs
|
|
621
626
|
* iff `p % total === i - 1`. `i` is 1-based (Playwright-style). The union over
|
|
622
|
-
* `i ∈ 1..total` is the exact grid, each cell once
|
|
623
|
-
* the high-index shards select **zero** cells
|
|
624
|
-
* `enumerateCells` is task-major, a task's run indexes are
|
|
625
|
-
*
|
|
626
|
-
* task's whole run block.
|
|
627
|
+
* `i ∈ 1..total` is the exact grid, each cell once. When
|
|
628
|
+
* `total > cells.length`, the high-index shards select **zero** cells. That is
|
|
629
|
+
* a valid run. `enumerateCells` is task-major, so a task's run indexes are
|
|
630
|
+
* adjacent. Round-robin then spreads them across shards. It does not hand one
|
|
631
|
+
* shard a slow task's whole run block.
|
|
627
632
|
* @param {{task: object, runIndex: number}[]} cells
|
|
628
633
|
* @param {number} i - 1-based shard index.
|
|
629
634
|
* @param {number} total - Shard count.
|
|
@@ -634,8 +639,9 @@ export function selectShard(cells, i, total) {
|
|
|
634
639
|
}
|
|
635
640
|
|
|
636
641
|
/**
|
|
637
|
-
* Validate the required BenchmarkRunner constructor arguments.
|
|
638
|
-
* the constructor to keep
|
|
642
|
+
* Validate the required BenchmarkRunner constructor arguments. It is separate
|
|
643
|
+
* from the constructor to keep that function's cognitive complexity under the
|
|
644
|
+
* lint ceiling.
|
|
639
645
|
*/
|
|
640
646
|
function validateRunnerArgs({
|
|
641
647
|
family,
|
|
@@ -674,5 +680,5 @@ export function createBenchmarkRunner(opts) {
|
|
|
674
680
|
return new BenchmarkRunner(opts);
|
|
675
681
|
}
|
|
676
682
|
|
|
677
|
-
//
|
|
683
|
+
// Tests use these internal exports.
|
|
678
684
|
export const __BASE_TOOLS = BASE_TOOLS;
|
|
@@ -1,11 +1,12 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* CellScheduler —
|
|
2
|
+
* CellScheduler — runs benchmark cells concurrently under a bound.
|
|
3
3
|
*
|
|
4
|
-
*
|
|
5
|
-
* yields each settled record in **completion order
|
|
6
|
-
* runner's drain loop consumes this async
|
|
7
|
-
* `results.jsonl
|
|
8
|
-
* stays single-writer
|
|
4
|
+
* The scheduler keeps at most `concurrency` `runCell(cell)` calls in flight
|
|
5
|
+
* at once. It yields each settled record in **completion order**. It does
|
|
6
|
+
* not yield them in grid order. The runner's drain loop consumes this async
|
|
7
|
+
* iterable as the sole writer of `results.jsonl`. Concurrency lives here in
|
|
8
|
+
* execution, and the ledger stays single-writer. The hot path needs no write
|
|
9
|
+
* mutex.
|
|
9
10
|
*/
|
|
10
11
|
|
|
11
12
|
/** Bounded pool that streams settled cell records in completion order. */
|
|
@@ -14,10 +15,10 @@ export class CellScheduler {
|
|
|
14
15
|
* @param {object} opts
|
|
15
16
|
* @param {number} opts.concurrency - Max cells in flight (integer ≥ 1).
|
|
16
17
|
* @param {(cell: {task: object, runIndex: number}) => Promise<object>} opts.runCell -
|
|
17
|
-
* Runs one cell to a settled record. By contract `runCell` never rejects
|
|
18
|
-
*
|
|
19
|
-
* returns a record
|
|
20
|
-
* so one bad cell cannot wedge the drain.
|
|
18
|
+
* Runs one cell to a settled record. By contract `runCell` never rejects.
|
|
19
|
+
* The runner's `#runOne` catches setup, agent, and schema failures. It
|
|
20
|
+
* returns a record instead of a throw. The scheduler still guards against
|
|
21
|
+
* a rejection, so one bad cell cannot wedge the drain.
|
|
21
22
|
*/
|
|
22
23
|
constructor({ concurrency, runCell }) {
|
|
23
24
|
if (!Number.isInteger(concurrency) || concurrency < 1)
|
|
@@ -29,7 +30,7 @@ export class CellScheduler {
|
|
|
29
30
|
}
|
|
30
31
|
|
|
31
32
|
/**
|
|
32
|
-
* Run every cell with bounded concurrency
|
|
33
|
+
* Run every cell with bounded concurrency. Yield each settled record the
|
|
33
34
|
* moment its cell completes.
|
|
34
35
|
* @param {{task: object, runIndex: number}[]} cells
|
|
35
36
|
* @returns {AsyncGenerator<object>}
|
|
@@ -42,8 +43,8 @@ export class CellScheduler {
|
|
|
42
43
|
const launch = () => {
|
|
43
44
|
const cell = cells[next++];
|
|
44
45
|
// The wrapper resolves to its own handle (for O(1) removal) plus the
|
|
45
|
-
// settled record
|
|
46
|
-
// record so the drain keeps consuming.
|
|
46
|
+
// settled record. It never rejects. A thrown runCell becomes a fail
|
|
47
|
+
// record, so the drain keeps consuming.
|
|
47
48
|
const p = Promise.resolve()
|
|
48
49
|
.then(() => this.runCell(cell))
|
|
49
50
|
.then(
|
|
@@ -64,9 +65,9 @@ export class CellScheduler {
|
|
|
64
65
|
}
|
|
65
66
|
|
|
66
67
|
/**
|
|
67
|
-
* Defensive fallback when `runCell` rejects (contract says it cannot).
|
|
68
|
-
* the drain consumable
|
|
69
|
-
*
|
|
68
|
+
* Defensive fallback when `runCell` rejects (the contract says it cannot).
|
|
69
|
+
* The fallback keeps the drain consumable. The record is deliberately
|
|
70
|
+
* minimal. `report`'s schema validation skips it and counts it as skipped.
|
|
70
71
|
*/
|
|
71
72
|
function schedulerFailRecord(cell, error) {
|
|
72
73
|
return {
|
|
@@ -14,16 +14,18 @@
|
|
|
14
14
|
* specs/ # copied into agent CWD
|
|
15
15
|
* workdir/ # copied into agent CWD
|
|
16
16
|
*
|
|
17
|
-
* `tests/` is an overlay mirror of the agent CWD
|
|
18
|
-
* `tests/` is its staging path. Every `*.test.js` file is one check
|
|
19
|
-
* `*.gate.test.js` marks a gate
|
|
20
|
-
* other file is support material
|
|
21
|
-
*
|
|
17
|
+
* `tests/` is an overlay mirror of the agent CWD. A file's path under
|
|
18
|
+
* `tests/` is its staging path. Every `*.test.js` file is one check.
|
|
19
|
+
* `*.gate.test.js` marks a gate. Any other `*.test.js` counts toward the
|
|
20
|
+
* score. Every other file is support material. The benchmark stages it but
|
|
21
|
+
* never grades it. This loader validates the layout eagerly, so authoring
|
|
22
|
+
* errors fail before any agent spend.
|
|
22
23
|
*
|
|
23
|
-
*
|
|
24
|
-
* a temp dir
|
|
25
|
-
* Local paths use the canonical-tree algorithm from
|
|
26
|
-
* algorithm so the result is stable across
|
|
24
|
+
* The loader accepts a local path or a git URL. For a git URL it makes a
|
|
25
|
+
* shallow clone into a temp dir. `familyRevision` then becomes `git:<sha>` of
|
|
26
|
+
* HEAD at clone time. Local paths use the canonical-tree algorithm from
|
|
27
|
+
* design § Family revision algorithm, so the result is stable across
|
|
28
|
+
* operating systems.
|
|
27
29
|
*
|
|
28
30
|
* Filesystem and subprocess access route through the injected `runtime` bag
|
|
29
31
|
* (`runtime.fs` async, `runtime.subprocess.run` one-shot, `tmpdir` derived
|
|
@@ -35,13 +37,13 @@ import { join, posix, relative, resolve, sep } from "node:path";
|
|
|
35
37
|
|
|
36
38
|
const GIT_URL_RE = /^(git@|https?:\/\/|ssh:\/\/|git:\/\/)/;
|
|
37
39
|
const SKIP_DIRS = new Set([".git", "node_modules"]);
|
|
38
|
-
// POSIX `X_OK` (execute permission)
|
|
39
|
-
//
|
|
40
|
+
// POSIX `X_OK` (execute permission). Node's fs honours the numeric mode, so we
|
|
41
|
+
// do not import `node:fs`'s `constants`, which would light the fs smell.
|
|
40
42
|
const X_OK = 1;
|
|
41
43
|
|
|
42
44
|
/**
|
|
43
|
-
* Derive the system temp dir from the env
|
|
44
|
-
* env
|
|
45
|
+
* Derive the system temp dir from the env. Node's `os.tmpdir()` also reads
|
|
46
|
+
* the env. The runtime bag has no `os` slot by design.
|
|
45
47
|
* @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
|
|
46
48
|
* @returns {string}
|
|
47
49
|
*/
|
|
@@ -62,7 +64,7 @@ export async function loadTaskFamily(rootPathOrGitUrl, runtime) {
|
|
|
62
64
|
let familyRevision;
|
|
63
65
|
if (isGit) {
|
|
64
66
|
const dir = await runtime.fs.mkdtemp(
|
|
65
|
-
join(tmpdir(runtime), "
|
|
67
|
+
join(tmpdir(runtime), "gemba-benchmark-family-"),
|
|
66
68
|
);
|
|
67
69
|
await gitClone(runtime, rootPathOrGitUrl, dir);
|
|
68
70
|
rootPath = dir;
|
|
@@ -84,9 +86,9 @@ export async function loadTaskFamily(rootPathOrGitUrl, runtime) {
|
|
|
84
86
|
}
|
|
85
87
|
|
|
86
88
|
/**
|
|
87
|
-
* Assert that `<judgeProfilesDir>/<judgeProfile>.md` exists.
|
|
88
|
-
* `BenchmarkRunner.run()` so a missing judge profile fails the
|
|
89
|
-
* install before any agent session starts.
|
|
89
|
+
* Assert that `<judgeProfilesDir>/<judgeProfile>.md` exists.
|
|
90
|
+
* `BenchmarkRunner.run()` calls it, so a missing judge profile fails the
|
|
91
|
+
* family install before any agent session starts.
|
|
90
92
|
* @param {TaskFamily} _family
|
|
91
93
|
* @param {string} judgeProfilesDir
|
|
92
94
|
* @param {string} judgeProfile
|
|
@@ -161,8 +163,8 @@ const GATE_SUFFIX = ".gate.test.js";
|
|
|
161
163
|
|
|
162
164
|
/**
|
|
163
165
|
* Discover and validate a task's hidden test suite under `<taskDir>/tests/`.
|
|
164
|
-
* Returns null when the directory is absent
|
|
165
|
-
* (no check files, a dangling symlink, duplicate check names) so
|
|
166
|
+
* Returns null when the directory is absent. Throws on an invalid layout
|
|
167
|
+
* (no check files, a dangling symlink, duplicate check names), so
|
|
166
168
|
* `loadTaskFamily` rejects broken suites before any agent spend.
|
|
167
169
|
* @param {object} fs - Async filesystem surface (`runtime.fs`).
|
|
168
170
|
* @param {string} taskDir
|
|
@@ -211,10 +213,10 @@ async function discoverSuite(fs, taskDir) {
|
|
|
211
213
|
}
|
|
212
214
|
|
|
213
215
|
/**
|
|
214
|
-
* Walk a suite tree
|
|
215
|
-
*
|
|
216
|
-
*
|
|
217
|
-
* symlink or a link to a non-file target is an authoring error.
|
|
216
|
+
* Walk a suite tree and collect `{sourcePath, stagePath}` entries. Every
|
|
217
|
+
* entry here must be a regular file after symlink resolution. `walkFiles` is
|
|
218
|
+
* different. It silently skips a dangling symlink when it hashes the tree. A
|
|
219
|
+
* dangling symlink or a link to a non-file target is an authoring error.
|
|
218
220
|
*/
|
|
219
221
|
async function walkSuiteFiles(fs, root, dir, out) {
|
|
220
222
|
const entries = await fs.readdir(dir, { withFileTypes: true });
|
|
@@ -343,18 +345,18 @@ async function git(runtime, args) {
|
|
|
343
345
|
|
|
344
346
|
/**
|
|
345
347
|
* @typedef {object} HiddenCheck
|
|
346
|
-
* @property {string} name - Basename stem
|
|
348
|
+
* @property {string} name - Basename stem without the check suffix.
|
|
347
349
|
* @property {boolean} gate - True iff the filename ends `.gate.test.js`.
|
|
348
350
|
* @property {string} sourcePath - Absolute path under `tests/`.
|
|
349
|
-
* @property {string} stagePath - Path relative to `tests
|
|
350
|
-
*
|
|
351
|
+
* @property {string} stagePath - Path relative to `tests/`. It mirrors the
|
|
352
|
+
* staging path under the agent CWD.
|
|
351
353
|
*/
|
|
352
354
|
|
|
353
355
|
/**
|
|
354
356
|
* @typedef {object} HiddenSuite
|
|
355
357
|
* @property {HiddenCheck[]} checks - In sorted stage-path order.
|
|
356
358
|
* @property {{sourcePath: string, stagePath: string}[]} support - Non-check
|
|
357
|
-
* files
|
|
359
|
+
* files. The benchmark stages them for the whole pass but never grades them.
|
|
358
360
|
*/
|
|
359
361
|
|
|
360
362
|
/**
|
|
@@ -1,19 +1,20 @@
|
|
|
1
1
|
/**
|
|
2
|
-
*
|
|
2
|
+
* Split the combined supervisor trace for the benchmark runner. One pass
|
|
3
3
|
* over the tagged NDJSON envelope stream separates agent events from
|
|
4
|
-
* supervisor
|
|
4
|
+
* supervisor and orchestrator events. The same pass extracts the run summary.
|
|
5
5
|
*/
|
|
6
6
|
|
|
7
7
|
import { createInterface } from "node:readline";
|
|
8
8
|
|
|
9
9
|
/**
|
|
10
|
-
* Split the combined supervisor trace into agent and supervisor files
|
|
11
|
-
*
|
|
12
|
-
* to `agentPath
|
|
10
|
+
* Split the combined supervisor trace into agent and supervisor files in a
|
|
11
|
+
* single pass. The same pass extracts the turn count and the submission.
|
|
12
|
+
* Agent-source events go to `agentPath`. Supervisor and orchestrator events
|
|
13
|
+
* go to `supervisorPath`.
|
|
13
14
|
*
|
|
14
|
-
*
|
|
15
|
-
* combined trace
|
|
16
|
-
* benchmark, callback, and `
|
|
15
|
+
* This function deliberately does not sum cost. The caller derives it from
|
|
16
|
+
* the same combined trace with `sumTraceCost`. One cost path then serves the
|
|
17
|
+
* benchmark, callback, and `gemba-trace cost` consumers.
|
|
17
18
|
* @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
|
|
18
19
|
* @param {string} combinedPath
|
|
19
20
|
* @param {string} agentPath
|