@forwardimpact/libharness 1.0.0 → 1.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@forwardimpact/libharness",
3
- "version": "1.0.0",
3
+ "version": "1.1.0",
4
4
  "description": "Agent evaluation framework — prove whether agent changes improved outcomes with reproducible evidence.",
5
5
  "keywords": [
6
6
  "eval",
@@ -433,21 +433,55 @@ function median(arr) {
433
433
  // Record loading
434
434
  // ---------------------------------------------------------------------------
435
435
 
436
+ // Directories never worth descending for a `results.jsonl`.
437
+ const SKIP_DIRS = new Set([".git", "node_modules"]);
438
+
439
+ /**
440
+ * Load and union every `results.jsonl` found recursively under `inputDir`.
441
+ *
442
+ * A single non-sharded run has one root-level ledger — the trivial one-match
443
+ * case of the same walk. A sharded run lays each shard's partial ledger in its
444
+ * own subdirectory; merging them equals reporting a single run over the same
445
+ * cells. An *existing* dir with no ledger yields the empty union (exit 0); a
446
+ * *missing* dir lets `readdir`'s ENOENT propagate so `report` still errors.
447
+ * @param {string} inputDir
448
+ * @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
449
+ * @returns {Promise<{records: object[], skipped: number}>}
450
+ */
436
451
  async function loadRecords(inputDir, runtime) {
437
- const path = join(inputDir, "results.jsonl");
438
- let content;
452
+ let files;
439
453
  try {
440
- content = await runtime.fs.readFile(path, "utf8");
454
+ files = await collectResultsFiles(inputDir, runtime);
441
455
  } catch (e) {
442
456
  // Re-throw with the stack collapsed to the message line so the CLI's
443
- // error rendering stays free of node-internal async `readFile` frames
444
- // (matching the pre-1370 stream-error shape the golden captured).
457
+ // error rendering stays free of node-internal async `readdir` frames
458
+ // (a missing --input dir surfaces its ENOENT as exit 1, matching the
459
+ // pre-1370 stream-error shape the golden captured).
445
460
  const err = new Error(e.message);
446
461
  if (e.code) err.code = e.code;
447
462
  err.stack = `Error: ${e.message}`;
448
463
  throw err;
449
464
  }
450
465
  const records = [];
466
+ let skipped = 0;
467
+ for (const file of files) {
468
+ const content = await runtime.fs.readFile(file, "utf8");
469
+ skipped += parseLedgerInto(content, records, runtime);
470
+ }
471
+ warnOnDuplicateCells(records, runtime);
472
+ return { records, skipped };
473
+ }
474
+
475
+ /**
476
+ * Parse one ledger's JSONL into `records`, skipping malformed or schema-invalid
477
+ * lines with a stderr warning. Returns the skipped count. Extracted from
478
+ * `loadRecords` to keep its cognitive complexity under the lint ceiling.
479
+ * @param {string} content
480
+ * @param {object[]} records - Accumulator, appended in place.
481
+ * @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
482
+ * @returns {number} Skipped line count.
483
+ */
484
+ function parseLedgerInto(content, records, runtime) {
451
485
  let skipped = 0;
452
486
  for (const line of content.split("\n")) {
453
487
  const trimmed = line.trim();
@@ -473,7 +507,55 @@ async function loadRecords(inputDir, runtime) {
473
507
  }
474
508
  records.push(record);
475
509
  }
476
- return { records, skipped };
510
+ return skipped;
511
+ }
512
+
513
+ /**
514
+ * Recursively collect paths of every file named `results.jsonl` under `dir`,
515
+ * skipping `.git`/`node_modules` and never following symlinks. A purpose-built
516
+ * `readdir` walk — `task-family.js`'s private `walkFiles` resolves symlinks and
517
+ * is unexported, which is the wrong contract here.
518
+ * @param {string} dir
519
+ * @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
520
+ * @returns {Promise<string[]>}
521
+ */
522
+ async function collectResultsFiles(dir, runtime) {
523
+ const entries = await runtime.fs.readdir(dir, { withFileTypes: true });
524
+ entries.sort((a, b) => (a.name < b.name ? -1 : a.name > b.name ? 1 : 0));
525
+ const out = [];
526
+ for (const entry of entries) {
527
+ if (entry.isSymbolicLink()) continue;
528
+ const full = join(dir, entry.name);
529
+ if (entry.isDirectory()) {
530
+ if (SKIP_DIRS.has(entry.name)) continue;
531
+ out.push(...(await collectResultsFiles(full, runtime)));
532
+ } else if (entry.isFile() && entry.name === "results.jsonl") {
533
+ out.push(full);
534
+ }
535
+ }
536
+ return out;
537
+ }
538
+
539
+ /**
540
+ * Warn (do not silently merge) when a `(taskId, runIndex)` cell appears more
541
+ * than once across shard ledgers. The shard partition guarantees uniqueness, so
542
+ * a duplicate signals misconfiguration; both copies stay in the group so the
543
+ * count is honest.
544
+ * @param {object[]} records
545
+ * @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
546
+ */
547
+ function warnOnDuplicateCells(records, runtime) {
548
+ const counts = new Map();
549
+ for (const r of records) {
550
+ const key = `${r.taskId}#${r.runIndex}`;
551
+ counts.set(key, (counts.get(key) ?? 0) + 1);
552
+ }
553
+ for (const [key, n] of counts) {
554
+ if (n > 1)
555
+ runtime.proc.stderr.write(
556
+ `benchmark report: duplicate cell ${key} appears ${n} times across shard ledgers — the shard partition should make each cell unique\n`,
557
+ );
558
+ }
477
559
  }
478
560
 
479
561
  function describeError(e) {
@@ -8,10 +8,13 @@
8
8
  * 4. Judge.runJudge → Conclude-driven verdict mapped to pass/fail
9
9
  * 5. WorkdirManager.teardown → process-group cleanup
10
10
  *
11
- * Results stream as an async iterable AND are appended to
12
- * `<output>/results.jsonl` for durability. The two paths are different
13
- * consumers of the same record — the iterator drives CLI stdout mirroring,
14
- * the JSONL append is the system of record.
11
+ * Cells run with bounded in-process concurrency (`CellScheduler`); `run()`
12
+ * yields records in **completion order**, not grid order. A single drain loop
13
+ * is the sole writer of `<output>/results.jsonl`, appending each record the
14
+ * moment its cell settles — that incremental append is the durability and
15
+ * crash-safety mechanism, so a killed run keeps every completed cell and there
16
+ * is no sidecar ledger. The iterator drives CLI stdout mirroring off the same
17
+ * stream.
15
18
  */
16
19
 
17
20
  import { createInterface } from "node:readline";
@@ -27,6 +30,7 @@ import { validateResultRecord } from "./result.js";
27
30
  import { runInvariants } from "./invariants.js";
28
31
  import { assertJudgeProfileStaged, loadTaskFamily } from "./task-family.js";
29
32
  import { createWorkdirManager } from "./workdir.js";
33
+ import { CellScheduler } from "./scheduler.js";
30
34
 
31
35
  const BASE_TOOLS = [
32
36
  "Bash",
@@ -42,6 +46,8 @@ const BASE_TOOLS = [
42
46
  // Upper bound on a single supervised agent run. A run that produces no terminal
43
47
  // message within this window is treated as a stall and recorded as an
44
48
  // agentError, so the benchmark never hangs the event loop into a silent exit.
49
+ // Overridable per-runner via `watchdogMs` so a test can force a stall to fire
50
+ // without waiting the full 20 minutes.
45
51
  const AGENT_WATCHDOG_MS = 20 * 60 * 1000;
46
52
 
47
53
  /** Sole orchestrator for a task-family benchmark run. */
@@ -58,6 +64,13 @@ export class BenchmarkRunner {
58
64
  * @param {Function} opts.query - SDK query (injected for testability).
59
65
  * @param {string[]} [opts.allowedTools] - Agent tool allowlist (default: BASE_TOOLS).
60
66
  * @param {number} [opts.maxTurns] - Agent-under-test turn budget.
67
+ * @param {number} [opts.concurrency] - Max cells in flight (integer ≥ 1).
68
+ * Defaults to 1 as a defensive floor; the CLI always passes a resolved value.
69
+ * @param {{index: number, total: number}} [opts.shard] - Run only the cells
70
+ * assigned to shard `index` of `total` (1-based). Absent ≡ the whole grid
71
+ * (identity `1/1`).
72
+ * @param {number} [opts.watchdogMs] - Per-agent stall watchdog (ms). Defaults
73
+ * to `AGENT_WATCHDOG_MS`; injectable so tests can force a stall in-test.
61
74
  * @param {number} [opts.termGraceMs] - SIGTERM→SIGKILL grace (ms) for the per-task process group.
62
75
  * @param {Function} [opts.runAgent] - Test seam: replaces the agent-under-test
63
76
  * session. Must return `{costUsd, turns, submission, agentError?}` and
@@ -91,6 +104,9 @@ export class BenchmarkRunner {
91
104
  query,
92
105
  allowedTools,
93
106
  maxTurns,
107
+ concurrency,
108
+ watchdogMs,
109
+ shard,
94
110
  task,
95
111
  skillsFrom,
96
112
  termGraceMs,
@@ -117,6 +133,9 @@ export class BenchmarkRunner {
117
133
  };
118
134
  this.query = query;
119
135
  this.maxTurns = maxTurns;
136
+ this.concurrency = concurrency ?? 1;
137
+ this.watchdogMs = watchdogMs ?? AGENT_WATCHDOG_MS;
138
+ this.shard = shard ?? null;
120
139
  this.taskFilter = task ?? null;
121
140
  this.skillsFrom = skillsFrom ?? null;
122
141
  this.termGraceMs = termGraceMs;
@@ -173,24 +192,38 @@ export class BenchmarkRunner {
173
192
  runtime,
174
193
  });
175
194
 
195
+ const allCells = enumerateCells(tasks, this.runs);
196
+ // Sharding selects a deterministic subset of the grid; an unsharded run is
197
+ // the identity 1/1. A high-index shard may select zero cells — a valid run
198
+ // whose results.jsonl ends up empty.
199
+ const cells = this.shard
200
+ ? selectShard(allCells, this.shard.index, this.shard.total)
201
+ : allCells;
202
+ const scheduler = new CellScheduler({
203
+ concurrency: this.concurrency,
204
+ runCell: (cell) =>
205
+ this.#runOne(
206
+ family,
207
+ wm,
208
+ cell.task,
209
+ cell.runIndex,
210
+ skillSetHash,
211
+ judgeProfilesDir,
212
+ ),
213
+ });
214
+
176
215
  const resultsPath = join(this.output, "results.jsonl");
177
216
  const resultsStream = runtime.fs.createWriteStream(resultsPath, {
178
217
  flags: "a",
179
218
  });
219
+ // Single-writer drain: the scheduler runs up to `concurrency` cells at
220
+ // once and pushes each settled record here in completion order. This loop
221
+ // is the sole writer of `results.jsonl` — workers never touch the stream —
222
+ // and the per-completion append is the crash-safety mechanism.
180
223
  try {
181
- for (const task of tasks) {
182
- for (let runIndex = 0; runIndex < this.runs; runIndex++) {
183
- const record = await this.#runOne(
184
- family,
185
- wm,
186
- task,
187
- runIndex,
188
- skillSetHash,
189
- judgeProfilesDir,
190
- );
191
- await writeRecord(resultsStream, record);
192
- yield record;
193
- }
224
+ for await (const record of scheduler.run(cells)) {
225
+ await writeRecord(resultsStream, record);
226
+ yield record;
194
227
  }
195
228
  } finally {
196
229
  await new Promise((r) => resultsStream.end(r));
@@ -199,22 +232,62 @@ export class BenchmarkRunner {
199
232
 
200
233
  async #runOne(family, wm, task, runIndex, skillSetHash, judgeProfilesDir) {
201
234
  const t0 = this.runtime.clock.now();
202
- const workdir = await wm.start(task, runIndex);
235
+ let workdir;
203
236
  try {
204
- if (workdir.preflightError) {
205
- const record = this.#buildPreflightFailureRecord({
206
- task,
207
- runIndex,
208
- workdir,
209
- skillSetHash,
210
- familyRevision: family.familyRevision,
211
- durationMs: this.runtime.clock.now() - t0,
212
- });
213
- return this.#validateOrFallback(
214
- record,
215
- resultsRecordKey(task, runIndex),
216
- );
217
- }
237
+ workdir = await wm.start(task, runIndex);
238
+ return await this.#executeCell({
239
+ family,
240
+ workdir,
241
+ task,
242
+ runIndex,
243
+ skillSetHash,
244
+ judgeProfilesDir,
245
+ t0,
246
+ });
247
+ } catch (e) {
248
+ // `wm.start()` (port acquire + workdir/env seeding) is the one throw site
249
+ // not caught inside `#executeCell`. Turn it into the runner's own fallback
250
+ // record so `#runOne` never rejects — the scheduler's one-record-per-cell
251
+ // contract depends on that. The fallback is schema-skipped by `report`,
252
+ // the same as any other runner-side schema failure.
253
+ return {
254
+ taskId: task.id,
255
+ runIndex,
256
+ verdict: "fail",
257
+ schemaError: `cell setup failed: ${e.message ?? String(e)}`,
258
+ };
259
+ } finally {
260
+ if (workdir) await wm.teardown(workdir).catch(() => {});
261
+ }
262
+ }
263
+
264
+ /**
265
+ * Run one cell's lifecycle against an already-started workdir: preflight
266
+ * gate → supervised agent → invariants → judge → assembled record. Extracted
267
+ * from `#runOne` so the start/teardown/error wrapper stays under the
268
+ * complexity ceiling.
269
+ */
270
+ async #executeCell({
271
+ family,
272
+ workdir,
273
+ task,
274
+ runIndex,
275
+ skillSetHash,
276
+ judgeProfilesDir,
277
+ t0,
278
+ }) {
279
+ if (workdir.preflightError) {
280
+ const record = this.#buildPreflightFailureRecord({
281
+ task,
282
+ runIndex,
283
+ workdir,
284
+ skillSetHash,
285
+ familyRevision: family.familyRevision,
286
+ durationMs: this.runtime.clock.now() - t0,
287
+ });
288
+ return this.#validateOrFallback(record, resultsRecordKey(task, runIndex));
289
+ }
290
+ {
218
291
  const agentRun = await this.#runAgentSafe(task, workdir);
219
292
  const { costUsd, turns, submission, agentError } = agentRun;
220
293
  const breakdown = agentRun.costBreakdown ?? { agent: 0, supervisor: 0 };
@@ -295,8 +368,6 @@ export class BenchmarkRunner {
295
368
  ...(agentError && { agentError }),
296
369
  };
297
370
  return this.#validateOrFallback(record, resultsRecordKey(task, runIndex));
298
- } finally {
299
- await wm.teardown(workdir).catch(() => {});
300
371
  }
301
372
  }
302
373
 
@@ -369,10 +440,10 @@ export class BenchmarkRunner {
369
440
  () =>
370
441
  reject(
371
442
  new Error(
372
- `agent run produced no result within ${AGENT_WATCHDOG_MS}ms (possible stall)`,
443
+ `agent run produced no result within ${this.watchdogMs}ms (possible stall)`,
373
444
  ),
374
445
  ),
375
- AGENT_WATCHDOG_MS,
446
+ this.watchdogMs,
376
447
  );
377
448
  }),
378
449
  ]);
@@ -477,6 +548,40 @@ export class BenchmarkRunner {
477
548
  }
478
549
  }
479
550
 
551
+ /**
552
+ * Flatten the grid into a stable ordered cell list, task-major /
553
+ * runIndex-minor. Load-bearing ordering: Part 02's round-robin shard balance
554
+ * depends on a task's runIndexes being adjacent in this list. Single source of
555
+ * the cell list for both the scheduler and the shard selector.
556
+ * @param {import("./task-family.js").Task[]} tasks
557
+ * @param {number} runs
558
+ * @returns {{task: import("./task-family.js").Task, runIndex: number}[]}
559
+ */
560
+ export function enumerateCells(tasks, runs) {
561
+ const cells = [];
562
+ for (const task of tasks)
563
+ for (let runIndex = 0; runIndex < runs; runIndex++)
564
+ cells.push({ task, runIndex });
565
+ return cells;
566
+ }
567
+
568
+ /**
569
+ * Round-robin partition of the enumerated cells: the cell at position `p` runs
570
+ * iff `p % total === i - 1`. `i` is 1-based (Playwright-style). The union over
571
+ * `i ∈ 1..total` is the exact grid, each cell once; when `total > cells.length`
572
+ * the high-index shards select **zero** cells — a valid run. Because
573
+ * `enumerateCells` is task-major, a task's run indexes are adjacent, so
574
+ * round-robin spreads them across shards rather than handing one shard a slow
575
+ * task's whole run block.
576
+ * @param {{task: object, runIndex: number}[]} cells
577
+ * @param {number} i - 1-based shard index.
578
+ * @param {number} total - Shard count.
579
+ * @returns {{task: object, runIndex: number}[]}
580
+ */
581
+ export function selectShard(cells, i, total) {
582
+ return cells.filter((_, p) => p % total === i - 1);
583
+ }
584
+
480
585
  /**
481
586
  * Validate the required BenchmarkRunner constructor arguments. Extracted from
482
587
  * the constructor to keep its cognitive complexity under the lint ceiling.
@@ -0,0 +1,78 @@
1
+ /**
2
+ * CellScheduler — bounded concurrent execution of benchmark cells.
3
+ *
4
+ * Keeps at most `concurrency` `runCell(cell)` calls in flight at once and
5
+ * yields each settled record in **completion order** (not grid order). The
6
+ * runner's drain loop consumes this async iterable as the sole writer of
7
+ * `results.jsonl`, so concurrency lives here in execution while the ledger
8
+ * stays single-writer — no write mutex on the hot path.
9
+ */
10
+
11
+ /** Bounded pool that streams settled cell records in completion order. */
12
+ export class CellScheduler {
13
+ /**
14
+ * @param {object} opts
15
+ * @param {number} opts.concurrency - Max cells in flight (integer ≥ 1).
16
+ * @param {(cell: {task: object, runIndex: number}) => Promise<object>} opts.runCell -
17
+ * Runs one cell to a settled record. By contract `runCell` never rejects —
18
+ * the runner's `#runOne` catches setup, agent, and schema failures and
19
+ * returns a record rather than throwing — but a rejection is still guarded
20
+ * so one bad cell cannot wedge the drain.
21
+ */
22
+ constructor({ concurrency, runCell }) {
23
+ if (!Number.isInteger(concurrency) || concurrency < 1)
24
+ throw new Error("concurrency must be an integer ≥ 1");
25
+ if (typeof runCell !== "function")
26
+ throw new Error("runCell must be a function");
27
+ this.concurrency = concurrency;
28
+ this.runCell = runCell;
29
+ }
30
+
31
+ /**
32
+ * Run every cell with bounded concurrency, yielding each settled record the
33
+ * moment its cell completes.
34
+ * @param {{task: object, runIndex: number}[]} cells
35
+ * @returns {AsyncGenerator<object>}
36
+ */
37
+ async *run(cells) {
38
+ let next = 0;
39
+ /** @type {Set<Promise<{p: Promise<*>, record: object}>>} */
40
+ const inFlight = new Set();
41
+
42
+ const launch = () => {
43
+ const cell = cells[next++];
44
+ // The wrapper resolves to its own handle (for O(1) removal) plus the
45
+ // settled record, and never rejects — a thrown runCell becomes a fail
46
+ // record so the drain keeps consuming.
47
+ const p = Promise.resolve()
48
+ .then(() => this.runCell(cell))
49
+ .then(
50
+ (record) => ({ p, record }),
51
+ (error) => ({ p, record: schedulerFailRecord(cell, error) }),
52
+ );
53
+ inFlight.add(p);
54
+ };
55
+
56
+ while (next < cells.length && inFlight.size < this.concurrency) launch();
57
+ while (inFlight.size > 0) {
58
+ const { p, record } = await Promise.race(inFlight);
59
+ inFlight.delete(p);
60
+ yield record;
61
+ if (next < cells.length) launch();
62
+ }
63
+ }
64
+ }
65
+
66
+ /**
67
+ * Defensive fallback when `runCell` rejects (contract says it cannot). Keeps
68
+ * the drain consumable; the record is intentionally minimal and will be
69
+ * skipped by `report`'s schema validation, counted as skipped.
70
+ */
71
+ function schedulerFailRecord(cell, error) {
72
+ return {
73
+ taskId: cell.task?.id,
74
+ runIndex: cell.runIndex,
75
+ verdict: "fail",
76
+ schedulerError: error?.message ?? String(error),
77
+ };
78
+ }
@@ -59,6 +59,9 @@ export class WorkdirManager {
59
59
  this.termGraceMs = termGraceMs ?? DEFAULT_TERM_GRACE_MS;
60
60
  this.familyRootPath = familyRootPath ?? null;
61
61
  this.runtime = runtime;
62
+ // One registry per manager: hands out distinct, bindable ports under a lock
63
+ // so concurrent cells can never be handed the same number.
64
+ this.ports = new PortRegistry();
62
65
  }
63
66
 
64
67
  /**
@@ -120,7 +123,7 @@ export class WorkdirManager {
120
123
  const envNames =
121
124
  envDirs.length > 0 ? await loadEnv(envDirs, cwd, this.runtime) : [];
122
125
 
123
- const port = await allocatePort();
126
+ const port = await this.ports.acquire();
124
127
  const agentTracePath = join(runDir, "agent.ndjson");
125
128
  const supervisorTracePath = join(runDir, "supervisor.ndjson");
126
129
  const judgeTracePath = join(runDir, "judge.ndjson");
@@ -175,9 +178,52 @@ export class WorkdirManager {
175
178
  2_000,
176
179
  );
177
180
  }
178
- const portFree = await isPortFree(workdir.port);
179
- const descendants = await countDescendants(this.runtime, workdir.pgid);
180
- return { portFree, descendants };
181
+ // Release the reservation in a finally so a throwing probe cannot leak the
182
+ // number from the in-use set; release after the port-free probe so a freed
183
+ // number can be re-handed to a waiting cell.
184
+ try {
185
+ const portFree = await isPortFree(workdir.port);
186
+ const descendants = await countDescendants(this.runtime, workdir.pgid);
187
+ return { portFree, descendants };
188
+ } finally {
189
+ this.ports.release(workdir.port);
190
+ }
191
+ }
192
+ }
193
+
194
+ /**
195
+ * Hands out distinct, bindable TCP ports under a lock. Replaces the bare
196
+ * close-then-return allocator whose allocate→bind window let two concurrent
197
+ * cells receive the same number.
198
+ *
199
+ * The reservation is the *number*, not a held socket — a held socket could not
200
+ * be bound by the agent later. `acquire` serializes through a one-slot promise
201
+ * chain and re-probes if the OS hands back a number already in the live in-use
202
+ * set, so no two in-flight cells share a port.
203
+ */
204
+ export class PortRegistry {
205
+ #inUse = new Set();
206
+ #tail = Promise.resolve();
207
+
208
+ /** @returns {Promise<number>} A distinct, currently-bindable port. */
209
+ acquire() {
210
+ const next = this.#tail.then(async () => {
211
+ let port;
212
+ do {
213
+ port = await probeFreePort();
214
+ } while (this.#inUse.has(port));
215
+ this.#inUse.add(port);
216
+ return port;
217
+ });
218
+ // Keep the chain alive even if one acquire rejects, so later acquires
219
+ // still run; swallow here, surface the rejection on `next`.
220
+ this.#tail = next.catch(() => {});
221
+ return next;
222
+ }
223
+
224
+ /** @param {number} port */
225
+ release(port) {
226
+ this.#inUse.delete(port);
181
227
  }
182
228
  }
183
229
 
@@ -222,7 +268,7 @@ async function runPreflight(runtime, script, cwd, port, vars) {
222
268
  };
223
269
  }
224
270
 
225
- function allocatePort() {
271
+ function probeFreePort() {
226
272
  return new Promise((res, rej) => {
227
273
  const server = createServer();
228
274
  server.unref();
@@ -77,6 +77,16 @@ export const definition = {
77
77
  description:
78
78
  "Agent-under-test turn budget (default: 50, 0 = unlimited)",
79
79
  },
80
+ concurrency: {
81
+ type: "string",
82
+ description:
83
+ "Max cells run concurrently (positive integer; default: CPU-aware min(4, max(2, cores/2)); env: LIBHARNESS_BENCHMARK_CONCURRENCY)",
84
+ },
85
+ shard: {
86
+ type: "string",
87
+ description:
88
+ "Run only shard i of N as i/N (1-based; default: the whole family). Each shard writes a partial results.jsonl; report --input merges them.",
89
+ },
80
90
  "allowed-tools": {
81
91
  type: "string",
82
92
  description:
@@ -5,6 +5,7 @@
5
5
  */
6
6
 
7
7
  import { resolve } from "node:path";
8
+ import { availableParallelism } from "node:os";
8
9
 
9
10
  import { createConfig } from "@forwardimpact/libconfig";
10
11
  import { createBenchmarkRunner } from "../benchmark/runner.js";
@@ -51,20 +52,37 @@ export async function runBenchmarkRunCommand(ctx) {
51
52
  runtime.proc.stdout.write(JSON.stringify(record) + "\n");
52
53
  if (record.verdict !== "pass") anyFail = true;
53
54
  }
54
- // A run that emits zero records did nothing (no tasks discovered, or the
55
- // agent never produced output). That is a failure, not a silent success —
56
- // surface it loudly so CI does not go green on an empty benchmark.
57
- if (count === 0) {
58
- return {
59
- ok: false,
60
- code: 1,
61
- error:
62
- "benchmark produced no result records — no task ran to completion; check the family's tasks/, apm install, and agent availability (ANTHROPIC_API_KEY / claude CLI / IS_SANDBOX)",
63
- };
64
- }
55
+ if (count === 0) return resolveZeroRecordOutcome(opts, runtime);
65
56
  return anyFail ? { ok: false, code: 1, error: "" } : { ok: true };
66
57
  }
67
58
 
59
+ /**
60
+ * Decide the exit outcome when a run streamed zero records. A run that emits no
61
+ * records normally did nothing (no tasks discovered, or the agent never
62
+ * produced output) — a failure, surfaced loudly so CI does not go green on an
63
+ * empty benchmark. The one exception is a deliberately-empty shard: a
64
+ * high-index `--shard=i/N` with `N > cell count` legitimately selects zero
65
+ * cells, so it exits 0 with a stderr note. Exported so the relaxed-guard branch
66
+ * is testable without the full handler's config/SDK setup.
67
+ * @param {{shard: {index: number, total: number} | null}} opts
68
+ * @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
69
+ * @returns {{ok: true} | {ok: false, code: number, error: string}}
70
+ */
71
+ export function resolveZeroRecordOutcome(opts, runtime) {
72
+ if (opts.shard) {
73
+ runtime.proc.stderr.write(
74
+ `shard ${opts.shard.index}/${opts.shard.total} selected no cells\n`,
75
+ );
76
+ return { ok: true };
77
+ }
78
+ return {
79
+ ok: false,
80
+ code: 1,
81
+ error:
82
+ "benchmark produced no result records — no task ran to completion; check the family's tasks/, apm install, and agent availability (ANTHROPIC_API_KEY / claude CLI / IS_SANDBOX)",
83
+ };
84
+ }
85
+
68
86
  /**
69
87
  * Parse and validate benchmark run options. Exported so tests can verify
70
88
  * defaults, including the resolved work tracker.
@@ -95,6 +113,8 @@ export function parseRunOptions(values, env = {}) {
95
113
  judge: values["judge-profile"] ?? null,
96
114
  },
97
115
  maxTurns: parseMaxTurns(values["max-turns"]),
116
+ concurrency: resolveConcurrency(values, env),
117
+ shard: parseShard(values.shard),
98
118
  allowedTools: values["allowed-tools"]
99
119
  ? values["allowed-tools"]
100
120
  .split(",")
@@ -109,3 +129,47 @@ function parseMaxTurns(raw) {
109
129
  if (raw === "0") return 0;
110
130
  return Number.parseInt(raw, 10);
111
131
  }
132
+
133
+ /**
134
+ * Parse a `--shard=<i>/<N>` selector into `{index, total}` (1-based), or `null`
135
+ * for an unsharded run. Validates `1 ≤ index ≤ total` with integer parts.
136
+ * @param {string|undefined} raw
137
+ * @returns {{index: number, total: number} | null}
138
+ */
139
+ export function parseShard(raw) {
140
+ if (raw == null || raw === "") return null;
141
+ const m = /^(\d+)\/(\d+)$/.exec(raw.trim());
142
+ if (!m) throw new Error("--shard must be in the form i/N (e.g. 1/4)");
143
+ const index = Number.parseInt(m[1], 10);
144
+ const total = Number.parseInt(m[2], 10);
145
+ if (total < 1 || index < 1 || index > total)
146
+ throw new Error("--shard requires 1 ≤ i ≤ N");
147
+ return { index, total };
148
+ }
149
+
150
+ // Conservative because each cell spawns ~3 agent subprocesses (lead +
151
+ // agent-under-test + judge); a low ceiling keeps a single runner from
152
+ // thrashing. The bulk of the CI speedup comes from Layer-2 sharding across
153
+ // machines, not from raising this in-job default.
154
+ const CONCURRENCY_CEILING = 4;
155
+
156
+ /**
157
+ * Resolve the cell concurrency: `--concurrency` flag > the
158
+ * `LIBHARNESS_BENCHMARK_CONCURRENCY` env var > a CPU-aware default of
159
+ * `min(CONCURRENCY_CEILING, max(2, ⌊cores/2⌋))`. The default is `> 1` so
160
+ * concurrency is on transparently without any consumer opting in.
161
+ * @param {Record<string, string|undefined>} values
162
+ * @param {Record<string, string|undefined>} [env]
163
+ * @returns {number}
164
+ */
165
+ export function resolveConcurrency(values, env = {}) {
166
+ const raw = values.concurrency ?? env.LIBHARNESS_BENCHMARK_CONCURRENCY;
167
+ if (raw != null && raw !== "") {
168
+ const n = Number.parseInt(raw, 10);
169
+ if (!Number.isFinite(n) || n < 1)
170
+ throw new Error("--concurrency must be a positive integer");
171
+ return n;
172
+ }
173
+ const cores = availableParallelism();
174
+ return Math.min(CONCURRENCY_CEILING, Math.max(2, Math.floor(cores / 2)));
175
+ }