@forwardimpact/libharness 1.0.0 → 1.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json
CHANGED
package/src/benchmark/report.js
CHANGED
|
@@ -433,21 +433,55 @@ function median(arr) {
|
|
|
433
433
|
// Record loading
|
|
434
434
|
// ---------------------------------------------------------------------------
|
|
435
435
|
|
|
436
|
+
// Directories never worth descending for a `results.jsonl`.
|
|
437
|
+
const SKIP_DIRS = new Set([".git", "node_modules"]);
|
|
438
|
+
|
|
439
|
+
/**
|
|
440
|
+
* Load and union every `results.jsonl` found recursively under `inputDir`.
|
|
441
|
+
*
|
|
442
|
+
* A single non-sharded run has one root-level ledger — the trivial one-match
|
|
443
|
+
* case of the same walk. A sharded run lays each shard's partial ledger in its
|
|
444
|
+
* own subdirectory; merging them equals reporting a single run over the same
|
|
445
|
+
* cells. An *existing* dir with no ledger yields the empty union (exit 0); a
|
|
446
|
+
* *missing* dir lets `readdir`'s ENOENT propagate so `report` still errors.
|
|
447
|
+
* @param {string} inputDir
|
|
448
|
+
* @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
|
|
449
|
+
* @returns {Promise<{records: object[], skipped: number}>}
|
|
450
|
+
*/
|
|
436
451
|
async function loadRecords(inputDir, runtime) {
|
|
437
|
-
|
|
438
|
-
let content;
|
|
452
|
+
let files;
|
|
439
453
|
try {
|
|
440
|
-
|
|
454
|
+
files = await collectResultsFiles(inputDir, runtime);
|
|
441
455
|
} catch (e) {
|
|
442
456
|
// Re-throw with the stack collapsed to the message line so the CLI's
|
|
443
|
-
// error rendering stays free of node-internal async `
|
|
444
|
-
// (
|
|
457
|
+
// error rendering stays free of node-internal async `readdir` frames
|
|
458
|
+
// (a missing --input dir surfaces its ENOENT as exit 1, matching the
|
|
459
|
+
// pre-1370 stream-error shape the golden captured).
|
|
445
460
|
const err = new Error(e.message);
|
|
446
461
|
if (e.code) err.code = e.code;
|
|
447
462
|
err.stack = `Error: ${e.message}`;
|
|
448
463
|
throw err;
|
|
449
464
|
}
|
|
450
465
|
const records = [];
|
|
466
|
+
let skipped = 0;
|
|
467
|
+
for (const file of files) {
|
|
468
|
+
const content = await runtime.fs.readFile(file, "utf8");
|
|
469
|
+
skipped += parseLedgerInto(content, records, runtime);
|
|
470
|
+
}
|
|
471
|
+
warnOnDuplicateCells(records, runtime);
|
|
472
|
+
return { records, skipped };
|
|
473
|
+
}
|
|
474
|
+
|
|
475
|
+
/**
|
|
476
|
+
* Parse one ledger's JSONL into `records`, skipping malformed or schema-invalid
|
|
477
|
+
* lines with a stderr warning. Returns the skipped count. Extracted from
|
|
478
|
+
* `loadRecords` to keep its cognitive complexity under the lint ceiling.
|
|
479
|
+
* @param {string} content
|
|
480
|
+
* @param {object[]} records - Accumulator, appended in place.
|
|
481
|
+
* @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
|
|
482
|
+
* @returns {number} Skipped line count.
|
|
483
|
+
*/
|
|
484
|
+
function parseLedgerInto(content, records, runtime) {
|
|
451
485
|
let skipped = 0;
|
|
452
486
|
for (const line of content.split("\n")) {
|
|
453
487
|
const trimmed = line.trim();
|
|
@@ -473,7 +507,55 @@ async function loadRecords(inputDir, runtime) {
|
|
|
473
507
|
}
|
|
474
508
|
records.push(record);
|
|
475
509
|
}
|
|
476
|
-
return
|
|
510
|
+
return skipped;
|
|
511
|
+
}
|
|
512
|
+
|
|
513
|
+
/**
|
|
514
|
+
* Recursively collect paths of every file named `results.jsonl` under `dir`,
|
|
515
|
+
* skipping `.git`/`node_modules` and never following symlinks. A purpose-built
|
|
516
|
+
* `readdir` walk — `task-family.js`'s private `walkFiles` resolves symlinks and
|
|
517
|
+
* is unexported, which is the wrong contract here.
|
|
518
|
+
* @param {string} dir
|
|
519
|
+
* @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
|
|
520
|
+
* @returns {Promise<string[]>}
|
|
521
|
+
*/
|
|
522
|
+
async function collectResultsFiles(dir, runtime) {
|
|
523
|
+
const entries = await runtime.fs.readdir(dir, { withFileTypes: true });
|
|
524
|
+
entries.sort((a, b) => (a.name < b.name ? -1 : a.name > b.name ? 1 : 0));
|
|
525
|
+
const out = [];
|
|
526
|
+
for (const entry of entries) {
|
|
527
|
+
if (entry.isSymbolicLink()) continue;
|
|
528
|
+
const full = join(dir, entry.name);
|
|
529
|
+
if (entry.isDirectory()) {
|
|
530
|
+
if (SKIP_DIRS.has(entry.name)) continue;
|
|
531
|
+
out.push(...(await collectResultsFiles(full, runtime)));
|
|
532
|
+
} else if (entry.isFile() && entry.name === "results.jsonl") {
|
|
533
|
+
out.push(full);
|
|
534
|
+
}
|
|
535
|
+
}
|
|
536
|
+
return out;
|
|
537
|
+
}
|
|
538
|
+
|
|
539
|
+
/**
|
|
540
|
+
* Warn (do not silently merge) when a `(taskId, runIndex)` cell appears more
|
|
541
|
+
* than once across shard ledgers. The shard partition guarantees uniqueness, so
|
|
542
|
+
* a duplicate signals misconfiguration; both copies stay in the group so the
|
|
543
|
+
* count is honest.
|
|
544
|
+
* @param {object[]} records
|
|
545
|
+
* @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
|
|
546
|
+
*/
|
|
547
|
+
function warnOnDuplicateCells(records, runtime) {
|
|
548
|
+
const counts = new Map();
|
|
549
|
+
for (const r of records) {
|
|
550
|
+
const key = `${r.taskId}#${r.runIndex}`;
|
|
551
|
+
counts.set(key, (counts.get(key) ?? 0) + 1);
|
|
552
|
+
}
|
|
553
|
+
for (const [key, n] of counts) {
|
|
554
|
+
if (n > 1)
|
|
555
|
+
runtime.proc.stderr.write(
|
|
556
|
+
`benchmark report: duplicate cell ${key} appears ${n} times across shard ledgers — the shard partition should make each cell unique\n`,
|
|
557
|
+
);
|
|
558
|
+
}
|
|
477
559
|
}
|
|
478
560
|
|
|
479
561
|
function describeError(e) {
|
package/src/benchmark/runner.js
CHANGED
|
@@ -8,10 +8,13 @@
|
|
|
8
8
|
* 4. Judge.runJudge → Conclude-driven verdict mapped to pass/fail
|
|
9
9
|
* 5. WorkdirManager.teardown → process-group cleanup
|
|
10
10
|
*
|
|
11
|
-
*
|
|
12
|
-
*
|
|
13
|
-
*
|
|
14
|
-
*
|
|
11
|
+
* Cells run with bounded in-process concurrency (`CellScheduler`); `run()`
|
|
12
|
+
* yields records in **completion order**, not grid order. A single drain loop
|
|
13
|
+
* is the sole writer of `<output>/results.jsonl`, appending each record the
|
|
14
|
+
* moment its cell settles — that incremental append is the durability and
|
|
15
|
+
* crash-safety mechanism, so a killed run keeps every completed cell and there
|
|
16
|
+
* is no sidecar ledger. The iterator drives CLI stdout mirroring off the same
|
|
17
|
+
* stream.
|
|
15
18
|
*/
|
|
16
19
|
|
|
17
20
|
import { createInterface } from "node:readline";
|
|
@@ -27,6 +30,7 @@ import { validateResultRecord } from "./result.js";
|
|
|
27
30
|
import { runInvariants } from "./invariants.js";
|
|
28
31
|
import { assertJudgeProfileStaged, loadTaskFamily } from "./task-family.js";
|
|
29
32
|
import { createWorkdirManager } from "./workdir.js";
|
|
33
|
+
import { CellScheduler } from "./scheduler.js";
|
|
30
34
|
|
|
31
35
|
const BASE_TOOLS = [
|
|
32
36
|
"Bash",
|
|
@@ -42,6 +46,8 @@ const BASE_TOOLS = [
|
|
|
42
46
|
// Upper bound on a single supervised agent run. A run that produces no terminal
|
|
43
47
|
// message within this window is treated as a stall and recorded as an
|
|
44
48
|
// agentError, so the benchmark never hangs the event loop into a silent exit.
|
|
49
|
+
// Overridable per-runner via `watchdogMs` so a test can force a stall to fire
|
|
50
|
+
// without waiting the full 20 minutes.
|
|
45
51
|
const AGENT_WATCHDOG_MS = 20 * 60 * 1000;
|
|
46
52
|
|
|
47
53
|
/** Sole orchestrator for a task-family benchmark run. */
|
|
@@ -58,6 +64,13 @@ export class BenchmarkRunner {
|
|
|
58
64
|
* @param {Function} opts.query - SDK query (injected for testability).
|
|
59
65
|
* @param {string[]} [opts.allowedTools] - Agent tool allowlist (default: BASE_TOOLS).
|
|
60
66
|
* @param {number} [opts.maxTurns] - Agent-under-test turn budget.
|
|
67
|
+
* @param {number} [opts.concurrency] - Max cells in flight (integer ≥ 1).
|
|
68
|
+
* Defaults to 1 as a defensive floor; the CLI always passes a resolved value.
|
|
69
|
+
* @param {{index: number, total: number}} [opts.shard] - Run only the cells
|
|
70
|
+
* assigned to shard `index` of `total` (1-based). Absent ≡ the whole grid
|
|
71
|
+
* (identity `1/1`).
|
|
72
|
+
* @param {number} [opts.watchdogMs] - Per-agent stall watchdog (ms). Defaults
|
|
73
|
+
* to `AGENT_WATCHDOG_MS`; injectable so tests can force a stall in-test.
|
|
61
74
|
* @param {number} [opts.termGraceMs] - SIGTERM→SIGKILL grace (ms) for the per-task process group.
|
|
62
75
|
* @param {Function} [opts.runAgent] - Test seam: replaces the agent-under-test
|
|
63
76
|
* session. Must return `{costUsd, turns, submission, agentError?}` and
|
|
@@ -91,6 +104,9 @@ export class BenchmarkRunner {
|
|
|
91
104
|
query,
|
|
92
105
|
allowedTools,
|
|
93
106
|
maxTurns,
|
|
107
|
+
concurrency,
|
|
108
|
+
watchdogMs,
|
|
109
|
+
shard,
|
|
94
110
|
task,
|
|
95
111
|
skillsFrom,
|
|
96
112
|
termGraceMs,
|
|
@@ -117,6 +133,9 @@ export class BenchmarkRunner {
|
|
|
117
133
|
};
|
|
118
134
|
this.query = query;
|
|
119
135
|
this.maxTurns = maxTurns;
|
|
136
|
+
this.concurrency = concurrency ?? 1;
|
|
137
|
+
this.watchdogMs = watchdogMs ?? AGENT_WATCHDOG_MS;
|
|
138
|
+
this.shard = shard ?? null;
|
|
120
139
|
this.taskFilter = task ?? null;
|
|
121
140
|
this.skillsFrom = skillsFrom ?? null;
|
|
122
141
|
this.termGraceMs = termGraceMs;
|
|
@@ -173,24 +192,38 @@ export class BenchmarkRunner {
|
|
|
173
192
|
runtime,
|
|
174
193
|
});
|
|
175
194
|
|
|
195
|
+
const allCells = enumerateCells(tasks, this.runs);
|
|
196
|
+
// Sharding selects a deterministic subset of the grid; an unsharded run is
|
|
197
|
+
// the identity 1/1. A high-index shard may select zero cells — a valid run
|
|
198
|
+
// whose results.jsonl ends up empty.
|
|
199
|
+
const cells = this.shard
|
|
200
|
+
? selectShard(allCells, this.shard.index, this.shard.total)
|
|
201
|
+
: allCells;
|
|
202
|
+
const scheduler = new CellScheduler({
|
|
203
|
+
concurrency: this.concurrency,
|
|
204
|
+
runCell: (cell) =>
|
|
205
|
+
this.#runOne(
|
|
206
|
+
family,
|
|
207
|
+
wm,
|
|
208
|
+
cell.task,
|
|
209
|
+
cell.runIndex,
|
|
210
|
+
skillSetHash,
|
|
211
|
+
judgeProfilesDir,
|
|
212
|
+
),
|
|
213
|
+
});
|
|
214
|
+
|
|
176
215
|
const resultsPath = join(this.output, "results.jsonl");
|
|
177
216
|
const resultsStream = runtime.fs.createWriteStream(resultsPath, {
|
|
178
217
|
flags: "a",
|
|
179
218
|
});
|
|
219
|
+
// Single-writer drain: the scheduler runs up to `concurrency` cells at
|
|
220
|
+
// once and pushes each settled record here in completion order. This loop
|
|
221
|
+
// is the sole writer of `results.jsonl` — workers never touch the stream —
|
|
222
|
+
// and the per-completion append is the crash-safety mechanism.
|
|
180
223
|
try {
|
|
181
|
-
for (const
|
|
182
|
-
|
|
183
|
-
|
|
184
|
-
family,
|
|
185
|
-
wm,
|
|
186
|
-
task,
|
|
187
|
-
runIndex,
|
|
188
|
-
skillSetHash,
|
|
189
|
-
judgeProfilesDir,
|
|
190
|
-
);
|
|
191
|
-
await writeRecord(resultsStream, record);
|
|
192
|
-
yield record;
|
|
193
|
-
}
|
|
224
|
+
for await (const record of scheduler.run(cells)) {
|
|
225
|
+
await writeRecord(resultsStream, record);
|
|
226
|
+
yield record;
|
|
194
227
|
}
|
|
195
228
|
} finally {
|
|
196
229
|
await new Promise((r) => resultsStream.end(r));
|
|
@@ -199,22 +232,62 @@ export class BenchmarkRunner {
|
|
|
199
232
|
|
|
200
233
|
async #runOne(family, wm, task, runIndex, skillSetHash, judgeProfilesDir) {
|
|
201
234
|
const t0 = this.runtime.clock.now();
|
|
202
|
-
|
|
235
|
+
let workdir;
|
|
203
236
|
try {
|
|
204
|
-
|
|
205
|
-
|
|
206
|
-
|
|
207
|
-
|
|
208
|
-
|
|
209
|
-
|
|
210
|
-
|
|
211
|
-
|
|
212
|
-
|
|
213
|
-
|
|
214
|
-
|
|
215
|
-
|
|
216
|
-
|
|
217
|
-
|
|
237
|
+
workdir = await wm.start(task, runIndex);
|
|
238
|
+
return await this.#executeCell({
|
|
239
|
+
family,
|
|
240
|
+
workdir,
|
|
241
|
+
task,
|
|
242
|
+
runIndex,
|
|
243
|
+
skillSetHash,
|
|
244
|
+
judgeProfilesDir,
|
|
245
|
+
t0,
|
|
246
|
+
});
|
|
247
|
+
} catch (e) {
|
|
248
|
+
// `wm.start()` (port acquire + workdir/env seeding) is the one throw site
|
|
249
|
+
// not caught inside `#executeCell`. Turn it into the runner's own fallback
|
|
250
|
+
// record so `#runOne` never rejects — the scheduler's one-record-per-cell
|
|
251
|
+
// contract depends on that. The fallback is schema-skipped by `report`,
|
|
252
|
+
// the same as any other runner-side schema failure.
|
|
253
|
+
return {
|
|
254
|
+
taskId: task.id,
|
|
255
|
+
runIndex,
|
|
256
|
+
verdict: "fail",
|
|
257
|
+
schemaError: `cell setup failed: ${e.message ?? String(e)}`,
|
|
258
|
+
};
|
|
259
|
+
} finally {
|
|
260
|
+
if (workdir) await wm.teardown(workdir).catch(() => {});
|
|
261
|
+
}
|
|
262
|
+
}
|
|
263
|
+
|
|
264
|
+
/**
|
|
265
|
+
* Run one cell's lifecycle against an already-started workdir: preflight
|
|
266
|
+
* gate → supervised agent → invariants → judge → assembled record. Extracted
|
|
267
|
+
* from `#runOne` so the start/teardown/error wrapper stays under the
|
|
268
|
+
* complexity ceiling.
|
|
269
|
+
*/
|
|
270
|
+
async #executeCell({
|
|
271
|
+
family,
|
|
272
|
+
workdir,
|
|
273
|
+
task,
|
|
274
|
+
runIndex,
|
|
275
|
+
skillSetHash,
|
|
276
|
+
judgeProfilesDir,
|
|
277
|
+
t0,
|
|
278
|
+
}) {
|
|
279
|
+
if (workdir.preflightError) {
|
|
280
|
+
const record = this.#buildPreflightFailureRecord({
|
|
281
|
+
task,
|
|
282
|
+
runIndex,
|
|
283
|
+
workdir,
|
|
284
|
+
skillSetHash,
|
|
285
|
+
familyRevision: family.familyRevision,
|
|
286
|
+
durationMs: this.runtime.clock.now() - t0,
|
|
287
|
+
});
|
|
288
|
+
return this.#validateOrFallback(record, resultsRecordKey(task, runIndex));
|
|
289
|
+
}
|
|
290
|
+
{
|
|
218
291
|
const agentRun = await this.#runAgentSafe(task, workdir);
|
|
219
292
|
const { costUsd, turns, submission, agentError } = agentRun;
|
|
220
293
|
const breakdown = agentRun.costBreakdown ?? { agent: 0, supervisor: 0 };
|
|
@@ -295,8 +368,6 @@ export class BenchmarkRunner {
|
|
|
295
368
|
...(agentError && { agentError }),
|
|
296
369
|
};
|
|
297
370
|
return this.#validateOrFallback(record, resultsRecordKey(task, runIndex));
|
|
298
|
-
} finally {
|
|
299
|
-
await wm.teardown(workdir).catch(() => {});
|
|
300
371
|
}
|
|
301
372
|
}
|
|
302
373
|
|
|
@@ -369,10 +440,10 @@ export class BenchmarkRunner {
|
|
|
369
440
|
() =>
|
|
370
441
|
reject(
|
|
371
442
|
new Error(
|
|
372
|
-
`agent run produced no result within ${
|
|
443
|
+
`agent run produced no result within ${this.watchdogMs}ms (possible stall)`,
|
|
373
444
|
),
|
|
374
445
|
),
|
|
375
|
-
|
|
446
|
+
this.watchdogMs,
|
|
376
447
|
);
|
|
377
448
|
}),
|
|
378
449
|
]);
|
|
@@ -477,6 +548,40 @@ export class BenchmarkRunner {
|
|
|
477
548
|
}
|
|
478
549
|
}
|
|
479
550
|
|
|
551
|
+
/**
|
|
552
|
+
* Flatten the grid into a stable ordered cell list, task-major /
|
|
553
|
+
* runIndex-minor. Load-bearing ordering: Part 02's round-robin shard balance
|
|
554
|
+
* depends on a task's runIndexes being adjacent in this list. Single source of
|
|
555
|
+
* the cell list for both the scheduler and the shard selector.
|
|
556
|
+
* @param {import("./task-family.js").Task[]} tasks
|
|
557
|
+
* @param {number} runs
|
|
558
|
+
* @returns {{task: import("./task-family.js").Task, runIndex: number}[]}
|
|
559
|
+
*/
|
|
560
|
+
export function enumerateCells(tasks, runs) {
|
|
561
|
+
const cells = [];
|
|
562
|
+
for (const task of tasks)
|
|
563
|
+
for (let runIndex = 0; runIndex < runs; runIndex++)
|
|
564
|
+
cells.push({ task, runIndex });
|
|
565
|
+
return cells;
|
|
566
|
+
}
|
|
567
|
+
|
|
568
|
+
/**
|
|
569
|
+
* Round-robin partition of the enumerated cells: the cell at position `p` runs
|
|
570
|
+
* iff `p % total === i - 1`. `i` is 1-based (Playwright-style). The union over
|
|
571
|
+
* `i ∈ 1..total` is the exact grid, each cell once; when `total > cells.length`
|
|
572
|
+
* the high-index shards select **zero** cells — a valid run. Because
|
|
573
|
+
* `enumerateCells` is task-major, a task's run indexes are adjacent, so
|
|
574
|
+
* round-robin spreads them across shards rather than handing one shard a slow
|
|
575
|
+
* task's whole run block.
|
|
576
|
+
* @param {{task: object, runIndex: number}[]} cells
|
|
577
|
+
* @param {number} i - 1-based shard index.
|
|
578
|
+
* @param {number} total - Shard count.
|
|
579
|
+
* @returns {{task: object, runIndex: number}[]}
|
|
580
|
+
*/
|
|
581
|
+
export function selectShard(cells, i, total) {
|
|
582
|
+
return cells.filter((_, p) => p % total === i - 1);
|
|
583
|
+
}
|
|
584
|
+
|
|
480
585
|
/**
|
|
481
586
|
* Validate the required BenchmarkRunner constructor arguments. Extracted from
|
|
482
587
|
* the constructor to keep its cognitive complexity under the lint ceiling.
|
|
@@ -0,0 +1,78 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* CellScheduler — bounded concurrent execution of benchmark cells.
|
|
3
|
+
*
|
|
4
|
+
* Keeps at most `concurrency` `runCell(cell)` calls in flight at once and
|
|
5
|
+
* yields each settled record in **completion order** (not grid order). The
|
|
6
|
+
* runner's drain loop consumes this async iterable as the sole writer of
|
|
7
|
+
* `results.jsonl`, so concurrency lives here in execution while the ledger
|
|
8
|
+
* stays single-writer — no write mutex on the hot path.
|
|
9
|
+
*/
|
|
10
|
+
|
|
11
|
+
/** Bounded pool that streams settled cell records in completion order. */
|
|
12
|
+
export class CellScheduler {
|
|
13
|
+
/**
|
|
14
|
+
* @param {object} opts
|
|
15
|
+
* @param {number} opts.concurrency - Max cells in flight (integer ≥ 1).
|
|
16
|
+
* @param {(cell: {task: object, runIndex: number}) => Promise<object>} opts.runCell -
|
|
17
|
+
* Runs one cell to a settled record. By contract `runCell` never rejects —
|
|
18
|
+
* the runner's `#runOne` catches setup, agent, and schema failures and
|
|
19
|
+
* returns a record rather than throwing — but a rejection is still guarded
|
|
20
|
+
* so one bad cell cannot wedge the drain.
|
|
21
|
+
*/
|
|
22
|
+
constructor({ concurrency, runCell }) {
|
|
23
|
+
if (!Number.isInteger(concurrency) || concurrency < 1)
|
|
24
|
+
throw new Error("concurrency must be an integer ≥ 1");
|
|
25
|
+
if (typeof runCell !== "function")
|
|
26
|
+
throw new Error("runCell must be a function");
|
|
27
|
+
this.concurrency = concurrency;
|
|
28
|
+
this.runCell = runCell;
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
/**
|
|
32
|
+
* Run every cell with bounded concurrency, yielding each settled record the
|
|
33
|
+
* moment its cell completes.
|
|
34
|
+
* @param {{task: object, runIndex: number}[]} cells
|
|
35
|
+
* @returns {AsyncGenerator<object>}
|
|
36
|
+
*/
|
|
37
|
+
async *run(cells) {
|
|
38
|
+
let next = 0;
|
|
39
|
+
/** @type {Set<Promise<{p: Promise<*>, record: object}>>} */
|
|
40
|
+
const inFlight = new Set();
|
|
41
|
+
|
|
42
|
+
const launch = () => {
|
|
43
|
+
const cell = cells[next++];
|
|
44
|
+
// The wrapper resolves to its own handle (for O(1) removal) plus the
|
|
45
|
+
// settled record, and never rejects — a thrown runCell becomes a fail
|
|
46
|
+
// record so the drain keeps consuming.
|
|
47
|
+
const p = Promise.resolve()
|
|
48
|
+
.then(() => this.runCell(cell))
|
|
49
|
+
.then(
|
|
50
|
+
(record) => ({ p, record }),
|
|
51
|
+
(error) => ({ p, record: schedulerFailRecord(cell, error) }),
|
|
52
|
+
);
|
|
53
|
+
inFlight.add(p);
|
|
54
|
+
};
|
|
55
|
+
|
|
56
|
+
while (next < cells.length && inFlight.size < this.concurrency) launch();
|
|
57
|
+
while (inFlight.size > 0) {
|
|
58
|
+
const { p, record } = await Promise.race(inFlight);
|
|
59
|
+
inFlight.delete(p);
|
|
60
|
+
yield record;
|
|
61
|
+
if (next < cells.length) launch();
|
|
62
|
+
}
|
|
63
|
+
}
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
/**
|
|
67
|
+
* Defensive fallback when `runCell` rejects (contract says it cannot). Keeps
|
|
68
|
+
* the drain consumable; the record is intentionally minimal and will be
|
|
69
|
+
* skipped by `report`'s schema validation, counted as skipped.
|
|
70
|
+
*/
|
|
71
|
+
function schedulerFailRecord(cell, error) {
|
|
72
|
+
return {
|
|
73
|
+
taskId: cell.task?.id,
|
|
74
|
+
runIndex: cell.runIndex,
|
|
75
|
+
verdict: "fail",
|
|
76
|
+
schedulerError: error?.message ?? String(error),
|
|
77
|
+
};
|
|
78
|
+
}
|
package/src/benchmark/workdir.js
CHANGED
|
@@ -59,6 +59,9 @@ export class WorkdirManager {
|
|
|
59
59
|
this.termGraceMs = termGraceMs ?? DEFAULT_TERM_GRACE_MS;
|
|
60
60
|
this.familyRootPath = familyRootPath ?? null;
|
|
61
61
|
this.runtime = runtime;
|
|
62
|
+
// One registry per manager: hands out distinct, bindable ports under a lock
|
|
63
|
+
// so concurrent cells can never be handed the same number.
|
|
64
|
+
this.ports = new PortRegistry();
|
|
62
65
|
}
|
|
63
66
|
|
|
64
67
|
/**
|
|
@@ -120,7 +123,7 @@ export class WorkdirManager {
|
|
|
120
123
|
const envNames =
|
|
121
124
|
envDirs.length > 0 ? await loadEnv(envDirs, cwd, this.runtime) : [];
|
|
122
125
|
|
|
123
|
-
const port = await
|
|
126
|
+
const port = await this.ports.acquire();
|
|
124
127
|
const agentTracePath = join(runDir, "agent.ndjson");
|
|
125
128
|
const supervisorTracePath = join(runDir, "supervisor.ndjson");
|
|
126
129
|
const judgeTracePath = join(runDir, "judge.ndjson");
|
|
@@ -175,9 +178,52 @@ export class WorkdirManager {
|
|
|
175
178
|
2_000,
|
|
176
179
|
);
|
|
177
180
|
}
|
|
178
|
-
|
|
179
|
-
|
|
180
|
-
|
|
181
|
+
// Release the reservation in a finally so a throwing probe cannot leak the
|
|
182
|
+
// number from the in-use set; release after the port-free probe so a freed
|
|
183
|
+
// number can be re-handed to a waiting cell.
|
|
184
|
+
try {
|
|
185
|
+
const portFree = await isPortFree(workdir.port);
|
|
186
|
+
const descendants = await countDescendants(this.runtime, workdir.pgid);
|
|
187
|
+
return { portFree, descendants };
|
|
188
|
+
} finally {
|
|
189
|
+
this.ports.release(workdir.port);
|
|
190
|
+
}
|
|
191
|
+
}
|
|
192
|
+
}
|
|
193
|
+
|
|
194
|
+
/**
|
|
195
|
+
* Hands out distinct, bindable TCP ports under a lock. Replaces the bare
|
|
196
|
+
* close-then-return allocator whose allocate→bind window let two concurrent
|
|
197
|
+
* cells receive the same number.
|
|
198
|
+
*
|
|
199
|
+
* The reservation is the *number*, not a held socket — a held socket could not
|
|
200
|
+
* be bound by the agent later. `acquire` serializes through a one-slot promise
|
|
201
|
+
* chain and re-probes if the OS hands back a number already in the live in-use
|
|
202
|
+
* set, so no two in-flight cells share a port.
|
|
203
|
+
*/
|
|
204
|
+
export class PortRegistry {
|
|
205
|
+
#inUse = new Set();
|
|
206
|
+
#tail = Promise.resolve();
|
|
207
|
+
|
|
208
|
+
/** @returns {Promise<number>} A distinct, currently-bindable port. */
|
|
209
|
+
acquire() {
|
|
210
|
+
const next = this.#tail.then(async () => {
|
|
211
|
+
let port;
|
|
212
|
+
do {
|
|
213
|
+
port = await probeFreePort();
|
|
214
|
+
} while (this.#inUse.has(port));
|
|
215
|
+
this.#inUse.add(port);
|
|
216
|
+
return port;
|
|
217
|
+
});
|
|
218
|
+
// Keep the chain alive even if one acquire rejects, so later acquires
|
|
219
|
+
// still run; swallow here, surface the rejection on `next`.
|
|
220
|
+
this.#tail = next.catch(() => {});
|
|
221
|
+
return next;
|
|
222
|
+
}
|
|
223
|
+
|
|
224
|
+
/** @param {number} port */
|
|
225
|
+
release(port) {
|
|
226
|
+
this.#inUse.delete(port);
|
|
181
227
|
}
|
|
182
228
|
}
|
|
183
229
|
|
|
@@ -222,7 +268,7 @@ async function runPreflight(runtime, script, cwd, port, vars) {
|
|
|
222
268
|
};
|
|
223
269
|
}
|
|
224
270
|
|
|
225
|
-
function
|
|
271
|
+
function probeFreePort() {
|
|
226
272
|
return new Promise((res, rej) => {
|
|
227
273
|
const server = createServer();
|
|
228
274
|
server.unref();
|
|
@@ -77,6 +77,16 @@ export const definition = {
|
|
|
77
77
|
description:
|
|
78
78
|
"Agent-under-test turn budget (default: 50, 0 = unlimited)",
|
|
79
79
|
},
|
|
80
|
+
concurrency: {
|
|
81
|
+
type: "string",
|
|
82
|
+
description:
|
|
83
|
+
"Max cells run concurrently (positive integer; default: CPU-aware min(4, max(2, cores/2)); env: LIBHARNESS_BENCHMARK_CONCURRENCY)",
|
|
84
|
+
},
|
|
85
|
+
shard: {
|
|
86
|
+
type: "string",
|
|
87
|
+
description:
|
|
88
|
+
"Run only shard i of N as i/N (1-based; default: the whole family). Each shard writes a partial results.jsonl; report --input merges them.",
|
|
89
|
+
},
|
|
80
90
|
"allowed-tools": {
|
|
81
91
|
type: "string",
|
|
82
92
|
description:
|
|
@@ -5,6 +5,7 @@
|
|
|
5
5
|
*/
|
|
6
6
|
|
|
7
7
|
import { resolve } from "node:path";
|
|
8
|
+
import { availableParallelism } from "node:os";
|
|
8
9
|
|
|
9
10
|
import { createConfig } from "@forwardimpact/libconfig";
|
|
10
11
|
import { createBenchmarkRunner } from "../benchmark/runner.js";
|
|
@@ -51,20 +52,37 @@ export async function runBenchmarkRunCommand(ctx) {
|
|
|
51
52
|
runtime.proc.stdout.write(JSON.stringify(record) + "\n");
|
|
52
53
|
if (record.verdict !== "pass") anyFail = true;
|
|
53
54
|
}
|
|
54
|
-
|
|
55
|
-
// agent never produced output). That is a failure, not a silent success —
|
|
56
|
-
// surface it loudly so CI does not go green on an empty benchmark.
|
|
57
|
-
if (count === 0) {
|
|
58
|
-
return {
|
|
59
|
-
ok: false,
|
|
60
|
-
code: 1,
|
|
61
|
-
error:
|
|
62
|
-
"benchmark produced no result records — no task ran to completion; check the family's tasks/, apm install, and agent availability (ANTHROPIC_API_KEY / claude CLI / IS_SANDBOX)",
|
|
63
|
-
};
|
|
64
|
-
}
|
|
55
|
+
if (count === 0) return resolveZeroRecordOutcome(opts, runtime);
|
|
65
56
|
return anyFail ? { ok: false, code: 1, error: "" } : { ok: true };
|
|
66
57
|
}
|
|
67
58
|
|
|
59
|
+
/**
|
|
60
|
+
* Decide the exit outcome when a run streamed zero records. A run that emits no
|
|
61
|
+
* records normally did nothing (no tasks discovered, or the agent never
|
|
62
|
+
* produced output) — a failure, surfaced loudly so CI does not go green on an
|
|
63
|
+
* empty benchmark. The one exception is a deliberately-empty shard: a
|
|
64
|
+
* high-index `--shard=i/N` with `N > cell count` legitimately selects zero
|
|
65
|
+
* cells, so it exits 0 with a stderr note. Exported so the relaxed-guard branch
|
|
66
|
+
* is testable without the full handler's config/SDK setup.
|
|
67
|
+
* @param {{shard: {index: number, total: number} | null}} opts
|
|
68
|
+
* @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
|
|
69
|
+
* @returns {{ok: true} | {ok: false, code: number, error: string}}
|
|
70
|
+
*/
|
|
71
|
+
export function resolveZeroRecordOutcome(opts, runtime) {
|
|
72
|
+
if (opts.shard) {
|
|
73
|
+
runtime.proc.stderr.write(
|
|
74
|
+
`shard ${opts.shard.index}/${opts.shard.total} selected no cells\n`,
|
|
75
|
+
);
|
|
76
|
+
return { ok: true };
|
|
77
|
+
}
|
|
78
|
+
return {
|
|
79
|
+
ok: false,
|
|
80
|
+
code: 1,
|
|
81
|
+
error:
|
|
82
|
+
"benchmark produced no result records — no task ran to completion; check the family's tasks/, apm install, and agent availability (ANTHROPIC_API_KEY / claude CLI / IS_SANDBOX)",
|
|
83
|
+
};
|
|
84
|
+
}
|
|
85
|
+
|
|
68
86
|
/**
|
|
69
87
|
* Parse and validate benchmark run options. Exported so tests can verify
|
|
70
88
|
* defaults, including the resolved work tracker.
|
|
@@ -95,6 +113,8 @@ export function parseRunOptions(values, env = {}) {
|
|
|
95
113
|
judge: values["judge-profile"] ?? null,
|
|
96
114
|
},
|
|
97
115
|
maxTurns: parseMaxTurns(values["max-turns"]),
|
|
116
|
+
concurrency: resolveConcurrency(values, env),
|
|
117
|
+
shard: parseShard(values.shard),
|
|
98
118
|
allowedTools: values["allowed-tools"]
|
|
99
119
|
? values["allowed-tools"]
|
|
100
120
|
.split(",")
|
|
@@ -109,3 +129,47 @@ function parseMaxTurns(raw) {
|
|
|
109
129
|
if (raw === "0") return 0;
|
|
110
130
|
return Number.parseInt(raw, 10);
|
|
111
131
|
}
|
|
132
|
+
|
|
133
|
+
/**
|
|
134
|
+
* Parse a `--shard=<i>/<N>` selector into `{index, total}` (1-based), or `null`
|
|
135
|
+
* for an unsharded run. Validates `1 ≤ index ≤ total` with integer parts.
|
|
136
|
+
* @param {string|undefined} raw
|
|
137
|
+
* @returns {{index: number, total: number} | null}
|
|
138
|
+
*/
|
|
139
|
+
export function parseShard(raw) {
|
|
140
|
+
if (raw == null || raw === "") return null;
|
|
141
|
+
const m = /^(\d+)\/(\d+)$/.exec(raw.trim());
|
|
142
|
+
if (!m) throw new Error("--shard must be in the form i/N (e.g. 1/4)");
|
|
143
|
+
const index = Number.parseInt(m[1], 10);
|
|
144
|
+
const total = Number.parseInt(m[2], 10);
|
|
145
|
+
if (total < 1 || index < 1 || index > total)
|
|
146
|
+
throw new Error("--shard requires 1 ≤ i ≤ N");
|
|
147
|
+
return { index, total };
|
|
148
|
+
}
|
|
149
|
+
|
|
150
|
+
// Conservative because each cell spawns ~3 agent subprocesses (lead +
|
|
151
|
+
// agent-under-test + judge); a low ceiling keeps a single runner from
|
|
152
|
+
// thrashing. The bulk of the CI speedup comes from Layer-2 sharding across
|
|
153
|
+
// machines, not from raising this in-job default.
|
|
154
|
+
const CONCURRENCY_CEILING = 4;
|
|
155
|
+
|
|
156
|
+
/**
|
|
157
|
+
* Resolve the cell concurrency: `--concurrency` flag > the
|
|
158
|
+
* `LIBHARNESS_BENCHMARK_CONCURRENCY` env var > a CPU-aware default of
|
|
159
|
+
* `min(CONCURRENCY_CEILING, max(2, ⌊cores/2⌋))`. The default is `> 1` so
|
|
160
|
+
* concurrency is on transparently without any consumer opting in.
|
|
161
|
+
* @param {Record<string, string|undefined>} values
|
|
162
|
+
* @param {Record<string, string|undefined>} [env]
|
|
163
|
+
* @returns {number}
|
|
164
|
+
*/
|
|
165
|
+
export function resolveConcurrency(values, env = {}) {
|
|
166
|
+
const raw = values.concurrency ?? env.LIBHARNESS_BENCHMARK_CONCURRENCY;
|
|
167
|
+
if (raw != null && raw !== "") {
|
|
168
|
+
const n = Number.parseInt(raw, 10);
|
|
169
|
+
if (!Number.isFinite(n) || n < 1)
|
|
170
|
+
throw new Error("--concurrency must be a positive integer");
|
|
171
|
+
return n;
|
|
172
|
+
}
|
|
173
|
+
const cores = availableParallelism();
|
|
174
|
+
return Math.min(CONCURRENCY_CEILING, Math.max(2, Math.floor(cores / 2)));
|
|
175
|
+
}
|