@forwardimpact/libharness 0.1.22 → 1.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (87) hide show
  1. package/LICENSE +21 -201
  2. package/README.md +196 -80
  3. package/bin/fit-benchmark.js +44 -0
  4. package/bin/fit-harness.js +358 -0
  5. package/bin/fit-selfedit.js +165 -0
  6. package/bin/fit-trace.js +510 -0
  7. package/package.json +41 -11
  8. package/src/agent-runner.js +256 -0
  9. package/src/benchmark/apm-installer.js +207 -0
  10. package/src/benchmark/env-loader.js +158 -0
  11. package/src/benchmark/hook-env.js +40 -0
  12. package/src/benchmark/invariants.js +141 -0
  13. package/src/benchmark/judge.js +187 -0
  14. package/src/benchmark/npm-installer.js +87 -0
  15. package/src/benchmark/report.js +604 -0
  16. package/src/benchmark/result.js +127 -0
  17. package/src/benchmark/runner.js +688 -0
  18. package/src/benchmark/scheduler.js +78 -0
  19. package/src/benchmark/task-family.js +260 -0
  20. package/src/benchmark/workdir.js +344 -0
  21. package/src/commands/assert.js +153 -0
  22. package/src/commands/benchmark-definition.js +175 -0
  23. package/src/commands/benchmark-invariants.js +73 -0
  24. package/src/commands/benchmark-report.js +51 -0
  25. package/src/commands/benchmark-run.js +175 -0
  26. package/src/commands/by-discussion.js +94 -0
  27. package/src/commands/callback.js +119 -0
  28. package/src/commands/discuss.js +132 -0
  29. package/src/commands/facilitate.js +123 -0
  30. package/src/commands/output.js +36 -0
  31. package/src/commands/run.js +152 -0
  32. package/src/commands/supervise.js +136 -0
  33. package/src/commands/task-input.js +54 -0
  34. package/src/commands/tee.js +53 -0
  35. package/src/commands/trace.js +630 -0
  36. package/src/commands/work-tracker.js +35 -0
  37. package/src/cost.js +79 -0
  38. package/src/discuss-tools.js +173 -0
  39. package/src/discusser.js +394 -0
  40. package/src/events/github.js +161 -0
  41. package/src/facilitator.js +205 -0
  42. package/src/inbox-poller.js +81 -0
  43. package/src/index.js +72 -2
  44. package/src/judge.js +210 -0
  45. package/src/message-bus.js +118 -0
  46. package/src/orchestration-loop.js +330 -0
  47. package/src/orchestration-toolkit.js +441 -0
  48. package/src/orchestrator-helpers.js +23 -0
  49. package/src/profile-prompt.js +266 -0
  50. package/src/redaction.js +253 -0
  51. package/src/render/line-renderer.js +54 -0
  52. package/src/render/orchestrator-filter.js +19 -0
  53. package/src/render/palette.js +63 -0
  54. package/src/render/tool-hints.js +154 -0
  55. package/src/render/turn-renderer.js +96 -0
  56. package/src/reply-emitter.js +47 -0
  57. package/src/sequence-counter.js +21 -0
  58. package/src/signature-filter.js +27 -0
  59. package/src/supervisor.js +236 -0
  60. package/src/tee-writer.js +150 -0
  61. package/src/trace-collector.js +444 -0
  62. package/src/trace-github.js +473 -0
  63. package/src/trace-multi.js +101 -0
  64. package/src/trace-query.js +748 -0
  65. package/src/trace-render.js +211 -0
  66. package/src/trace-usage.js +249 -0
  67. package/src/fixture/assertions.js +0 -42
  68. package/src/fixture/cache.js +0 -50
  69. package/src/fixture/eval.js +0 -146
  70. package/src/fixture/index.js +0 -9
  71. package/src/fixture/pathway.js +0 -451
  72. package/src/fixture/services.js +0 -56
  73. package/src/mock/clients.js +0 -135
  74. package/src/mock/config.js +0 -45
  75. package/src/mock/data.js +0 -46
  76. package/src/mock/fs.js +0 -111
  77. package/src/mock/grpc.js +0 -94
  78. package/src/mock/http.js +0 -60
  79. package/src/mock/index.js +0 -36
  80. package/src/mock/infra.js +0 -219
  81. package/src/mock/logger.js +0 -42
  82. package/src/mock/observer.js +0 -74
  83. package/src/mock/resource-index.js +0 -95
  84. package/src/mock/service-callbacks.js +0 -39
  85. package/src/mock/services.js +0 -79
  86. package/src/mock/spy.js +0 -44
  87. package/src/mock/storage.js +0 -118
@@ -0,0 +1,153 @@
1
+ import { basename } from "node:path";
2
+ import jmespath from "jmespath";
3
+
4
+ /**
5
+ * Evaluate an assertion and return the structured result.
6
+ * @param {object} values - { grep?: string, query?: string, exists?: boolean, not?: boolean, message?: string }
7
+ * @param {string[]} args - [testName, file]
8
+ * @param {object} fsSync - Sync filesystem surface (`runtime.fsSync`): `existsSync`, `readFileSync`.
9
+ * @returns {{ test: string, pass: boolean, message?: string }}
10
+ */
11
+ // biome-ignore lint/complexity/noExcessiveCognitiveComplexity: assertion dispatch by type
12
+ export function evaluateAssertion(values, args, fsSync) {
13
+ const testName = args[0];
14
+ if (!testName) throw new Error("assert: missing test name");
15
+
16
+ const file = args[1];
17
+ const modes = [
18
+ values.grep,
19
+ values.query,
20
+ values.exists,
21
+ values["cites-job"],
22
+ ].filter((v) => v !== undefined && v !== false);
23
+ if (modes.length === 0) {
24
+ throw new Error(
25
+ "assert: specify one of --grep, --query, --exists, or --cites-job",
26
+ );
27
+ }
28
+ if (modes.length > 1) {
29
+ throw new Error(
30
+ "assert: specify only one of --grep, --query, --exists, or --cites-job",
31
+ );
32
+ }
33
+
34
+ let result;
35
+ if (values.exists) {
36
+ if (!file) throw new Error("assert: missing file argument");
37
+ result = assertExists(file, fsSync);
38
+ } else if (values.grep) {
39
+ if (!file) throw new Error("assert: missing file argument for --grep");
40
+ result = assertGrep(values.grep, file, fsSync);
41
+ } else if (values["cites-job"]) {
42
+ if (!file) throw new Error("assert: missing file argument for --cites-job");
43
+ result = assertCitesJob(values["cites-job"], file, fsSync);
44
+ } else {
45
+ if (!file) throw new Error("assert: missing file argument for --query");
46
+ result = assertQuery(values.query, file, fsSync);
47
+ }
48
+
49
+ if (values.not) {
50
+ result.pass = !result.pass;
51
+ if (result.pass) {
52
+ delete result.message;
53
+ } else {
54
+ result.message =
55
+ result.message ?? `inverted assertion failed for ${basename(file)}`;
56
+ }
57
+ }
58
+
59
+ if (!result.pass && values.message) {
60
+ result.message = values.message;
61
+ }
62
+
63
+ const output = { test: testName, pass: result.pass };
64
+ if (result.message) output.message = result.message;
65
+ return output;
66
+ }
67
+
68
+ /**
69
+ * Run an assertion, write JSON to stdout, and return a failure envelope when
70
+ * the assertion does not pass.
71
+ * @param {import("@forwardimpact/libcli").InvocationContext} ctx
72
+ * @returns {Promise<{ok: true} | {ok: false, code: number, error: string}>}
73
+ */
74
+ export async function runAssertCommand(ctx) {
75
+ const runtime = ctx.deps.runtime;
76
+ const args = [ctx.args["test-name"], ctx.args.file];
77
+ let result;
78
+ try {
79
+ result = evaluateAssertion(ctx.options, args, runtime.fsSync);
80
+ } catch (err) {
81
+ return { ok: false, code: 1, error: err.message };
82
+ }
83
+ runtime.proc.stdout.write(JSON.stringify(result) + "\n");
84
+ return result.pass ? { ok: true } : { ok: false, code: 1, error: "" };
85
+ }
86
+
87
+ function assertExists(file, fsSync) {
88
+ if (fsSync.existsSync(file)) return { pass: true };
89
+ return { pass: false, message: `${file} not found` };
90
+ }
91
+
92
+ function assertGrep(pattern, file, fsSync) {
93
+ const content = fsSync.readFileSync(file, "utf8");
94
+ const re = new RegExp(pattern, "im");
95
+ if (re.test(content)) return { pass: true };
96
+ return {
97
+ pass: false,
98
+ message: `pattern "${pattern}" not found in ${basename(file)}`,
99
+ };
100
+ }
101
+
102
+ function assertQuery(expression, file, fsSync) {
103
+ const content = fsSync.readFileSync(file, "utf8");
104
+ const data = parseJsonOrNdjson(content);
105
+ const result = jmespath.search(data, expression);
106
+ const truthy =
107
+ result !== null &&
108
+ result !== undefined &&
109
+ result !== false &&
110
+ (Array.isArray(result) ? result.length > 0 : true);
111
+ if (truthy) return { pass: true };
112
+ return {
113
+ pass: false,
114
+ message: `query returned ${JSON.stringify(result)}`,
115
+ };
116
+ }
117
+
118
+ const JOB_TAG_RE = /<job\s+user="([^"]*)"\s+goal="([^"]*)">/;
119
+
120
+ function assertCitesJob(jobFile, file, fsSync) {
121
+ const jobContent = fsSync.readFileSync(jobFile, "utf8");
122
+ const match = JOB_TAG_RE.exec(jobContent);
123
+ if (!match) {
124
+ return {
125
+ pass: false,
126
+ message: `no <job> tag found in ${basename(jobFile)}`,
127
+ };
128
+ }
129
+ const citation = `${match[1]}: ${match[2]}`;
130
+ const content = fsSync.readFileSync(file, "utf8");
131
+ if (content.includes(citation)) return { pass: true };
132
+ return { pass: false, message: `missing "${citation}"` };
133
+ }
134
+
135
+ function parseJsonOrNdjson(content) {
136
+ try {
137
+ return JSON.parse(content);
138
+ } catch {
139
+ // Fall through to NDJSON
140
+ }
141
+ const lines = [];
142
+ for (const raw of content.split("\n")) {
143
+ const trimmed = raw.trim();
144
+ if (!trimmed) continue;
145
+ try {
146
+ lines.push(JSON.parse(trimmed));
147
+ } catch {
148
+ // skip unparseable lines
149
+ }
150
+ }
151
+ if (lines.length === 0) throw new Error("assert: no valid JSON in file");
152
+ return lines;
153
+ }
@@ -0,0 +1,175 @@
1
+ /**
2
+ * `fit-benchmark` CLI definition. Lives in `src/` so the bin stays an
3
+ * execute-on-import entry point — launcher packages import the bin to run
4
+ * it — while tests import the definition without running the CLI.
5
+ */
6
+
7
+ import { runBenchmarkRunCommand } from "./benchmark-run.js";
8
+ import { runBenchmarkInvariantsCommand } from "./benchmark-invariants.js";
9
+ import { runBenchmarkReportCommand } from "./benchmark-report.js";
10
+ import {
11
+ BENCHMARK_AGENT_MODEL,
12
+ LEAD_MODEL,
13
+ } from "@forwardimpact/libutil/models";
14
+
15
+ export const definition = {
16
+ name: "fit-benchmark",
17
+ description:
18
+ "Run coding-agent task families, grade hidden tests, and aggregate pass@k across runs.",
19
+ commands: [
20
+ {
21
+ name: "run",
22
+ args: [],
23
+ handler: runBenchmarkRunCommand,
24
+ description:
25
+ "Run every task in a family for N runs and emit one result record per (task, runIndex).",
26
+ options: {
27
+ family: {
28
+ type: "string",
29
+ description: "Path or git URL to a task family",
30
+ },
31
+ task: {
32
+ type: "string",
33
+ description:
34
+ "Run only this task id (directory name under tasks/, default: every task)",
35
+ },
36
+ "skills-from": {
37
+ type: "string",
38
+ description:
39
+ "Stage .claude/ from this directory (a root containing .claude/) instead of running apm install — exercise local, unpublished skills",
40
+ },
41
+ output: {
42
+ type: "string",
43
+ description:
44
+ "Run-output directory (created if missing, default: benchmark-runs)",
45
+ },
46
+ runs: {
47
+ type: "string",
48
+ description: "Runs per task (integer ≥ 1, default: 5)",
49
+ },
50
+ "agent-model": {
51
+ type: "string",
52
+ description: `Claude model for the agent-under-test (default: ${BENCHMARK_AGENT_MODEL})`,
53
+ },
54
+ "lead-model": {
55
+ type: "string",
56
+ description: `Claude model for the lead role (default: ${LEAD_MODEL})`,
57
+ },
58
+ "judge-model": {
59
+ type: "string",
60
+ description: `Claude model for the judge (default: ${LEAD_MODEL})`,
61
+ },
62
+ "agent-profile": {
63
+ type: "string",
64
+ description: "Agent-under-test profile name",
65
+ },
66
+ "judge-profile": {
67
+ type: "string",
68
+ description: "Judge profile name",
69
+ },
70
+ "work-tracker": {
71
+ type: "string",
72
+ description:
73
+ "Active work-item tracker (github|filesystem, default: github)",
74
+ },
75
+ "max-turns": {
76
+ type: "string",
77
+ description:
78
+ "Agent-under-test turn budget (default: 50, 0 = unlimited)",
79
+ },
80
+ concurrency: {
81
+ type: "string",
82
+ description:
83
+ "Max cells run concurrently (positive integer; default: CPU-aware min(4, max(2, cores/2)); env: LIBHARNESS_BENCHMARK_CONCURRENCY)",
84
+ },
85
+ shard: {
86
+ type: "string",
87
+ description:
88
+ "Run only shard i of N as i/N (1-based; default: the whole family). Each shard writes a partial results.jsonl; report --input merges them.",
89
+ },
90
+ "allowed-tools": {
91
+ type: "string",
92
+ description:
93
+ "Comma-separated tool allowlist for the agent-under-test (default: Bash,Read,Glob,Grep,Write,Edit,Agent,TodoWrite)",
94
+ },
95
+ },
96
+ },
97
+ {
98
+ name: "invariants",
99
+ args: [],
100
+ handler: runBenchmarkInvariantsCommand,
101
+ description:
102
+ "Check a single task's invariants against a post-run workdir without invoking an agent.",
103
+ options: {
104
+ family: {
105
+ type: "string",
106
+ description: "Path or git URL to a task family",
107
+ },
108
+ task: {
109
+ type: "string",
110
+ description: "Task id (directory name under tasks/)",
111
+ },
112
+ "run-dir": {
113
+ type: "string",
114
+ description:
115
+ "Post-run directory whose cwd/ subdir is the agent CWD; invariants run against that cwd — the path hooks receive as $AGENT_CWD",
116
+ },
117
+ output: {
118
+ type: "string",
119
+ description: "Output file (defaults to stdout; one JSONL line)",
120
+ },
121
+ },
122
+ },
123
+ {
124
+ name: "report",
125
+ args: [],
126
+ handler: runBenchmarkReportCommand,
127
+ description:
128
+ "Aggregate result records into pass@k via the OpenAI HumanEval estimator.",
129
+ options: {
130
+ input: {
131
+ type: "string",
132
+ description:
133
+ "Run-output directory containing results.jsonl (default: benchmark-runs)",
134
+ },
135
+ k: {
136
+ type: "string",
137
+ description: "Comma-separated k values (default: 1,3,5)",
138
+ },
139
+ format: {
140
+ type: "string",
141
+ description: "Output format (json|text, default: json)",
142
+ },
143
+ },
144
+ },
145
+ ],
146
+ globalOptions: {
147
+ help: { type: "boolean", short: "h", description: "Show this help" },
148
+ version: { type: "boolean", description: "Show version" },
149
+ json: { type: "boolean", description: "Output help as JSON" },
150
+ },
151
+ examples: [
152
+ "fit-benchmark run --family=./families/coding",
153
+ "fit-benchmark run --family=./families/coding --task=todo-api --runs=1",
154
+ "fit-benchmark run --family=./families/coding --work-tracker=filesystem",
155
+ "fit-benchmark run --family=./families/coding --skills-from=. --task=todo-api",
156
+ `fit-benchmark run --family=./families/coding --runs=10 --agent-model=${BENCHMARK_AGENT_MODEL}`,
157
+ "fit-benchmark invariants --family=./families/coding --task=todo-api --run-dir=./benchmark-runs/runs/todo-api/0",
158
+ "fit-benchmark report --format=text",
159
+ "fit-benchmark report --input=./runs/today --k=1,3,5 --format=text",
160
+ ],
161
+ documentation: [
162
+ {
163
+ title: "Run a Benchmark",
164
+ url: "https://www.forwardimpact.team/docs/libraries/prove-changes/run-benchmark/index.md",
165
+ description:
166
+ "Author a coding-task family, run a benchmark across multiple runs, and read the pass@k report.",
167
+ },
168
+ {
169
+ title: "Automate with GitHub Actions",
170
+ url: "https://www.forwardimpact.team/docs/libraries/prove-changes/run-benchmark/ci-workflow/index.md",
171
+ description:
172
+ "Run benchmarks in CI with the forwardimpact/fit-benchmark action.",
173
+ },
174
+ ],
175
+ };
@@ -0,0 +1,73 @@
1
+ /**
2
+ * `fit-benchmark invariants` — check a single task's invariants against a
3
+ * post-run workdir directory without invoking an agent. Useful for
4
+ * re-checking an agent's output against revised grading material.
5
+ */
6
+
7
+ import { join, resolve } from "node:path";
8
+ import { createServer } from "node:net";
9
+
10
+ import { validateInvariantsRecord } from "../benchmark/result.js";
11
+ import { runInvariants } from "../benchmark/invariants.js";
12
+ import { loadTaskFamily } from "../benchmark/task-family.js";
13
+
14
+ /**
15
+ * @param {import("@forwardimpact/libcli").InvocationContext} ctx
16
+ * @returns {Promise<{ok: true} | {ok: false, code: number, error: string}>}
17
+ */
18
+ export async function runBenchmarkInvariantsCommand(ctx) {
19
+ const values = ctx.options;
20
+ const runtime = ctx.deps.runtime;
21
+ const familyInput = values.family;
22
+ if (!familyInput)
23
+ return { ok: false, code: 1, error: "--family is required" };
24
+ const taskId = values.task;
25
+ if (!taskId) return { ok: false, code: 1, error: "--task is required" };
26
+ const runDirArg = values["run-dir"];
27
+ if (!runDirArg) return { ok: false, code: 1, error: "--run-dir is required" };
28
+
29
+ const family = await loadTaskFamily(familyInput, runtime);
30
+ const task = family.tasks().find((t) => t.id === taskId);
31
+ if (!task)
32
+ return { ok: false, code: 1, error: `task not found in family: ${taskId}` };
33
+
34
+ const runDir = resolve(runDirArg);
35
+ const cwd = join(runDir, "cwd");
36
+ const port = await allocatePort();
37
+
38
+ const invariants = await runInvariants(task, { cwd, port, runDir }, runtime);
39
+ const record = {
40
+ taskId: task.id,
41
+ invariants,
42
+ exitCode: invariants.exitCode,
43
+ };
44
+ validateInvariantsRecord(record);
45
+
46
+ const line = JSON.stringify(record) + "\n";
47
+ if (values.output) {
48
+ runtime.fsSync.writeFileSync(resolve(values.output), line);
49
+ } else {
50
+ runtime.proc.stdout.write(line);
51
+ }
52
+ return invariants.verdict === "pass"
53
+ ? { ok: true }
54
+ : { ok: false, code: 1, error: "" };
55
+ }
56
+
57
+ function allocatePort() {
58
+ return new Promise((res, rej) => {
59
+ const server = createServer();
60
+ server.unref();
61
+ server.on("error", rej);
62
+ server.listen(0, "127.0.0.1", () => {
63
+ const addr = server.address();
64
+ if (!addr || typeof addr === "string") {
65
+ server.close();
66
+ rej(new Error("failed to allocate port"));
67
+ return;
68
+ }
69
+ const port = addr.port;
70
+ server.close(() => res(port));
71
+ });
72
+ });
73
+ }
@@ -0,0 +1,51 @@
1
+ /**
2
+ * `fit-benchmark report` — aggregate `results.jsonl` into pass@k via the
3
+ * OpenAI HumanEval estimator. Output is JSON by default; pass --format=text
4
+ * to render a markdown table.
5
+ */
6
+
7
+ import { resolve } from "node:path";
8
+
9
+ import { aggregate, renderTextReport } from "../benchmark/report.js";
10
+
11
+ /**
12
+ * @param {import("@forwardimpact/libcli").InvocationContext} ctx
13
+ * @returns {Promise<{ok: true} | {ok: false, code: number, error: string}>}
14
+ */
15
+ export async function runBenchmarkReportCommand(ctx) {
16
+ const values = ctx.options;
17
+ const runtime = ctx.deps.runtime;
18
+ const inputDir = values.input ?? "benchmark-runs";
19
+ const kRaw = values.k ?? "1,3,5";
20
+ let kValues;
21
+ try {
22
+ kValues = kRaw.split(",").map((t) => {
23
+ const n = Number.parseInt(t.trim(), 10);
24
+ if (!Number.isFinite(n) || n < 1) {
25
+ throw new Error(
26
+ "--k must be a comma-separated list of positive integers",
27
+ );
28
+ }
29
+ return n;
30
+ });
31
+ } catch (err) {
32
+ return { ok: false, code: 1, error: err.message };
33
+ }
34
+ const format = values.format ?? "json";
35
+ if (format !== "json" && format !== "text") {
36
+ return { ok: false, code: 1, error: "--format must be 'json' or 'text'" };
37
+ }
38
+
39
+ const report = await aggregate({
40
+ inputDir: resolve(inputDir),
41
+ kValues,
42
+ includeRuns: format === "text",
43
+ runtime,
44
+ });
45
+ if (format === "text") {
46
+ runtime.proc.stdout.write(renderTextReport(report, kValues) + "\n");
47
+ } else {
48
+ runtime.proc.stdout.write(JSON.stringify(report, null, 2) + "\n");
49
+ }
50
+ return { ok: true };
51
+ }
@@ -0,0 +1,175 @@
1
+ /**
2
+ * `fit-benchmark run` — run every task in a family for N runs, stream each
3
+ * ResultRecord to stdout (one JSON line per record), and append to the
4
+ * canonical `<output>/results.jsonl` for the report subcommand.
5
+ */
6
+
7
+ import { resolve } from "node:path";
8
+ import { availableParallelism } from "node:os";
9
+
10
+ import { createConfig } from "@forwardimpact/libconfig";
11
+ import { createBenchmarkRunner } from "../benchmark/runner.js";
12
+ import { resolveWorkTracker } from "./work-tracker.js";
13
+ import {
14
+ BENCHMARK_AGENT_MODEL,
15
+ LEAD_MODEL,
16
+ } from "@forwardimpact/libutil/models";
17
+
18
+ /**
19
+ * @param {import("@forwardimpact/libcli").InvocationContext} ctx
20
+ * @returns {Promise<{ok: true} | {ok: false, code: number, error: string}>}
21
+ */
22
+ export async function runBenchmarkRunCommand(ctx) {
23
+ const values = ctx.options;
24
+ const runtime = ctx.deps.runtime;
25
+ let opts;
26
+ try {
27
+ opts = parseRunOptions(values, runtime.proc.env);
28
+ } catch (err) {
29
+ return { ok: false, code: 1, error: err.message };
30
+ }
31
+ const config = await createConfig("script", "benchmark");
32
+ runtime.proc.env.ANTHROPIC_API_KEY = await config.anthropicToken();
33
+ // The benchmark agent runs via createBenchmarkRunner, not the supervise
34
+ // command, so the active-tracker env must land here before the runner
35
+ // spawns the subprocess that inherits process.env.
36
+ runtime.proc.env.LIBHARNESS_WORK_TRACKER = opts.workTracker;
37
+
38
+ // The Claude Agent SDK spawns a `claude` subprocess that inherits
39
+ // process.env. NODE_EXTRA_CA_CERTS causes undici (the HTTP client
40
+ // inside that subprocess) to fail with UND_ERR_INVALID_ARG on
41
+ // Node 22+, aborting every API call after 10 retries. Strip it
42
+ // before the SDK loads so the subprocess gets a clean environment.
43
+ delete runtime.proc.env.NODE_EXTRA_CA_CERTS;
44
+
45
+ const { query } = await import("@anthropic-ai/claude-agent-sdk");
46
+ const runner = createBenchmarkRunner({ ...opts, query, runtime });
47
+
48
+ let anyFail = false;
49
+ let count = 0;
50
+ for await (const record of runner.run()) {
51
+ count++;
52
+ runtime.proc.stdout.write(JSON.stringify(record) + "\n");
53
+ if (record.verdict !== "pass") anyFail = true;
54
+ }
55
+ if (count === 0) return resolveZeroRecordOutcome(opts, runtime);
56
+ return anyFail ? { ok: false, code: 1, error: "" } : { ok: true };
57
+ }
58
+
59
+ /**
60
+ * Decide the exit outcome when a run streamed zero records. A run that emits no
61
+ * records normally did nothing (no tasks discovered, or the agent never
62
+ * produced output) — a failure, surfaced loudly so CI does not go green on an
63
+ * empty benchmark. The one exception is a deliberately-empty shard: a
64
+ * high-index `--shard=i/N` with `N > cell count` legitimately selects zero
65
+ * cells, so it exits 0 with a stderr note. Exported so the relaxed-guard branch
66
+ * is testable without the full handler's config/SDK setup.
67
+ * @param {{shard: {index: number, total: number} | null}} opts
68
+ * @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
69
+ * @returns {{ok: true} | {ok: false, code: number, error: string}}
70
+ */
71
+ export function resolveZeroRecordOutcome(opts, runtime) {
72
+ if (opts.shard) {
73
+ runtime.proc.stderr.write(
74
+ `shard ${opts.shard.index}/${opts.shard.total} selected no cells\n`,
75
+ );
76
+ return { ok: true };
77
+ }
78
+ return {
79
+ ok: false,
80
+ code: 1,
81
+ error:
82
+ "benchmark produced no result records — no task ran to completion; check the family's tasks/, apm install, and agent availability (ANTHROPIC_API_KEY / claude CLI / IS_SANDBOX)",
83
+ };
84
+ }
85
+
86
+ /**
87
+ * Parse and validate benchmark run options. Exported so tests can verify
88
+ * defaults, including the resolved work tracker.
89
+ * @param {Record<string, string|undefined>} values - Parsed option values
90
+ * @param {Record<string, string|undefined>} [env] - Process environment, read
91
+ * for the `LIBHARNESS_WORK_TRACKER` fallback when `--work-tracker` is absent.
92
+ * @returns {object}
93
+ */
94
+ export function parseRunOptions(values, env = {}) {
95
+ const family = values.family;
96
+ if (!family) throw new Error("--family is required");
97
+ const output = values.output ?? "benchmark-runs";
98
+ const runs = Number.parseInt(values.runs ?? "5", 10);
99
+ if (!Number.isFinite(runs) || runs < 1)
100
+ throw new Error("--runs must be a positive integer");
101
+ return {
102
+ family,
103
+ runs,
104
+ task: values.task ?? null,
105
+ skillsFrom: values["skills-from"] ?? null,
106
+ output: resolve(output),
107
+ agentModel: values["agent-model"] || BENCHMARK_AGENT_MODEL,
108
+ supervisorModel: values["lead-model"] || LEAD_MODEL,
109
+ judgeModel: values["judge-model"] || LEAD_MODEL,
110
+ workTracker: resolveWorkTracker(values, env),
111
+ profiles: {
112
+ agent: values["agent-profile"] ?? null,
113
+ judge: values["judge-profile"] ?? null,
114
+ },
115
+ maxTurns: parseMaxTurns(values["max-turns"]),
116
+ concurrency: resolveConcurrency(values, env),
117
+ shard: parseShard(values.shard),
118
+ allowedTools: values["allowed-tools"]
119
+ ? values["allowed-tools"]
120
+ .split(",")
121
+ .map((s) => s.trim())
122
+ .filter(Boolean)
123
+ : undefined,
124
+ };
125
+ }
126
+
127
+ function parseMaxTurns(raw) {
128
+ if (raw === undefined) return undefined;
129
+ if (raw === "0") return 0;
130
+ return Number.parseInt(raw, 10);
131
+ }
132
+
133
+ /**
134
+ * Parse a `--shard=<i>/<N>` selector into `{index, total}` (1-based), or `null`
135
+ * for an unsharded run. Validates `1 ≤ index ≤ total` with integer parts.
136
+ * @param {string|undefined} raw
137
+ * @returns {{index: number, total: number} | null}
138
+ */
139
+ export function parseShard(raw) {
140
+ if (raw == null || raw === "") return null;
141
+ const m = /^(\d+)\/(\d+)$/.exec(raw.trim());
142
+ if (!m) throw new Error("--shard must be in the form i/N (e.g. 1/4)");
143
+ const index = Number.parseInt(m[1], 10);
144
+ const total = Number.parseInt(m[2], 10);
145
+ if (total < 1 || index < 1 || index > total)
146
+ throw new Error("--shard requires 1 ≤ i ≤ N");
147
+ return { index, total };
148
+ }
149
+
150
+ // Conservative because each cell spawns ~3 agent subprocesses (lead +
151
+ // agent-under-test + judge); a low ceiling keeps a single runner from
152
+ // thrashing. The bulk of the CI speedup comes from Layer-2 sharding across
153
+ // machines, not from raising this in-job default.
154
+ const CONCURRENCY_CEILING = 4;
155
+
156
+ /**
157
+ * Resolve the cell concurrency: `--concurrency` flag > the
158
+ * `LIBHARNESS_BENCHMARK_CONCURRENCY` env var > a CPU-aware default of
159
+ * `min(CONCURRENCY_CEILING, max(2, ⌊cores/2⌋))`. The default is `> 1` so
160
+ * concurrency is on transparently without any consumer opting in.
161
+ * @param {Record<string, string|undefined>} values
162
+ * @param {Record<string, string|undefined>} [env]
163
+ * @returns {number}
164
+ */
165
+ export function resolveConcurrency(values, env = {}) {
166
+ const raw = values.concurrency ?? env.LIBHARNESS_BENCHMARK_CONCURRENCY;
167
+ if (raw != null && raw !== "") {
168
+ const n = Number.parseInt(raw, 10);
169
+ if (!Number.isFinite(n) || n < 1)
170
+ throw new Error("--concurrency must be a positive integer");
171
+ return n;
172
+ }
173
+ const cores = availableParallelism();
174
+ return Math.min(CONCURRENCY_CEILING, Math.max(2, Math.floor(cores / 2)));
175
+ }