@forwardimpact/libharness 0.1.22 → 1.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -201
- package/README.md +196 -80
- package/bin/fit-benchmark.js +44 -0
- package/bin/fit-harness.js +358 -0
- package/bin/fit-selfedit.js +165 -0
- package/bin/fit-trace.js +510 -0
- package/package.json +41 -11
- package/src/agent-runner.js +256 -0
- package/src/benchmark/apm-installer.js +207 -0
- package/src/benchmark/env-loader.js +158 -0
- package/src/benchmark/hook-env.js +40 -0
- package/src/benchmark/invariants.js +141 -0
- package/src/benchmark/judge.js +187 -0
- package/src/benchmark/npm-installer.js +87 -0
- package/src/benchmark/report.js +604 -0
- package/src/benchmark/result.js +127 -0
- package/src/benchmark/runner.js +688 -0
- package/src/benchmark/scheduler.js +78 -0
- package/src/benchmark/task-family.js +260 -0
- package/src/benchmark/workdir.js +344 -0
- package/src/commands/assert.js +153 -0
- package/src/commands/benchmark-definition.js +175 -0
- package/src/commands/benchmark-invariants.js +73 -0
- package/src/commands/benchmark-report.js +51 -0
- package/src/commands/benchmark-run.js +175 -0
- package/src/commands/by-discussion.js +94 -0
- package/src/commands/callback.js +119 -0
- package/src/commands/discuss.js +132 -0
- package/src/commands/facilitate.js +123 -0
- package/src/commands/output.js +36 -0
- package/src/commands/run.js +152 -0
- package/src/commands/supervise.js +136 -0
- package/src/commands/task-input.js +54 -0
- package/src/commands/tee.js +53 -0
- package/src/commands/trace.js +630 -0
- package/src/commands/work-tracker.js +35 -0
- package/src/cost.js +79 -0
- package/src/discuss-tools.js +173 -0
- package/src/discusser.js +394 -0
- package/src/events/github.js +161 -0
- package/src/facilitator.js +205 -0
- package/src/inbox-poller.js +81 -0
- package/src/index.js +72 -2
- package/src/judge.js +210 -0
- package/src/message-bus.js +118 -0
- package/src/orchestration-loop.js +330 -0
- package/src/orchestration-toolkit.js +441 -0
- package/src/orchestrator-helpers.js +23 -0
- package/src/profile-prompt.js +266 -0
- package/src/redaction.js +253 -0
- package/src/render/line-renderer.js +54 -0
- package/src/render/orchestrator-filter.js +19 -0
- package/src/render/palette.js +63 -0
- package/src/render/tool-hints.js +154 -0
- package/src/render/turn-renderer.js +96 -0
- package/src/reply-emitter.js +47 -0
- package/src/sequence-counter.js +21 -0
- package/src/signature-filter.js +27 -0
- package/src/supervisor.js +236 -0
- package/src/tee-writer.js +150 -0
- package/src/trace-collector.js +444 -0
- package/src/trace-github.js +473 -0
- package/src/trace-multi.js +101 -0
- package/src/trace-query.js +748 -0
- package/src/trace-render.js +211 -0
- package/src/trace-usage.js +249 -0
- package/src/fixture/assertions.js +0 -42
- package/src/fixture/cache.js +0 -50
- package/src/fixture/eval.js +0 -146
- package/src/fixture/index.js +0 -9
- package/src/fixture/pathway.js +0 -451
- package/src/fixture/services.js +0 -56
- package/src/mock/clients.js +0 -135
- package/src/mock/config.js +0 -45
- package/src/mock/data.js +0 -46
- package/src/mock/fs.js +0 -111
- package/src/mock/grpc.js +0 -94
- package/src/mock/http.js +0 -60
- package/src/mock/index.js +0 -36
- package/src/mock/infra.js +0 -219
- package/src/mock/logger.js +0 -42
- package/src/mock/observer.js +0 -74
- package/src/mock/resource-index.js +0 -95
- package/src/mock/service-callbacks.js +0 -39
- package/src/mock/services.js +0 -79
- package/src/mock/spy.js +0 -44
- package/src/mock/storage.js +0 -118
|
@@ -0,0 +1,153 @@
|
|
|
1
|
+
import { basename } from "node:path";
|
|
2
|
+
import jmespath from "jmespath";
|
|
3
|
+
|
|
4
|
+
/**
|
|
5
|
+
* Evaluate an assertion and return the structured result.
|
|
6
|
+
* @param {object} values - { grep?: string, query?: string, exists?: boolean, not?: boolean, message?: string }
|
|
7
|
+
* @param {string[]} args - [testName, file]
|
|
8
|
+
* @param {object} fsSync - Sync filesystem surface (`runtime.fsSync`): `existsSync`, `readFileSync`.
|
|
9
|
+
* @returns {{ test: string, pass: boolean, message?: string }}
|
|
10
|
+
*/
|
|
11
|
+
// biome-ignore lint/complexity/noExcessiveCognitiveComplexity: assertion dispatch by type
|
|
12
|
+
export function evaluateAssertion(values, args, fsSync) {
|
|
13
|
+
const testName = args[0];
|
|
14
|
+
if (!testName) throw new Error("assert: missing test name");
|
|
15
|
+
|
|
16
|
+
const file = args[1];
|
|
17
|
+
const modes = [
|
|
18
|
+
values.grep,
|
|
19
|
+
values.query,
|
|
20
|
+
values.exists,
|
|
21
|
+
values["cites-job"],
|
|
22
|
+
].filter((v) => v !== undefined && v !== false);
|
|
23
|
+
if (modes.length === 0) {
|
|
24
|
+
throw new Error(
|
|
25
|
+
"assert: specify one of --grep, --query, --exists, or --cites-job",
|
|
26
|
+
);
|
|
27
|
+
}
|
|
28
|
+
if (modes.length > 1) {
|
|
29
|
+
throw new Error(
|
|
30
|
+
"assert: specify only one of --grep, --query, --exists, or --cites-job",
|
|
31
|
+
);
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
let result;
|
|
35
|
+
if (values.exists) {
|
|
36
|
+
if (!file) throw new Error("assert: missing file argument");
|
|
37
|
+
result = assertExists(file, fsSync);
|
|
38
|
+
} else if (values.grep) {
|
|
39
|
+
if (!file) throw new Error("assert: missing file argument for --grep");
|
|
40
|
+
result = assertGrep(values.grep, file, fsSync);
|
|
41
|
+
} else if (values["cites-job"]) {
|
|
42
|
+
if (!file) throw new Error("assert: missing file argument for --cites-job");
|
|
43
|
+
result = assertCitesJob(values["cites-job"], file, fsSync);
|
|
44
|
+
} else {
|
|
45
|
+
if (!file) throw new Error("assert: missing file argument for --query");
|
|
46
|
+
result = assertQuery(values.query, file, fsSync);
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
if (values.not) {
|
|
50
|
+
result.pass = !result.pass;
|
|
51
|
+
if (result.pass) {
|
|
52
|
+
delete result.message;
|
|
53
|
+
} else {
|
|
54
|
+
result.message =
|
|
55
|
+
result.message ?? `inverted assertion failed for ${basename(file)}`;
|
|
56
|
+
}
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
if (!result.pass && values.message) {
|
|
60
|
+
result.message = values.message;
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
const output = { test: testName, pass: result.pass };
|
|
64
|
+
if (result.message) output.message = result.message;
|
|
65
|
+
return output;
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
/**
|
|
69
|
+
* Run an assertion, write JSON to stdout, and return a failure envelope when
|
|
70
|
+
* the assertion does not pass.
|
|
71
|
+
* @param {import("@forwardimpact/libcli").InvocationContext} ctx
|
|
72
|
+
* @returns {Promise<{ok: true} | {ok: false, code: number, error: string}>}
|
|
73
|
+
*/
|
|
74
|
+
export async function runAssertCommand(ctx) {
|
|
75
|
+
const runtime = ctx.deps.runtime;
|
|
76
|
+
const args = [ctx.args["test-name"], ctx.args.file];
|
|
77
|
+
let result;
|
|
78
|
+
try {
|
|
79
|
+
result = evaluateAssertion(ctx.options, args, runtime.fsSync);
|
|
80
|
+
} catch (err) {
|
|
81
|
+
return { ok: false, code: 1, error: err.message };
|
|
82
|
+
}
|
|
83
|
+
runtime.proc.stdout.write(JSON.stringify(result) + "\n");
|
|
84
|
+
return result.pass ? { ok: true } : { ok: false, code: 1, error: "" };
|
|
85
|
+
}
|
|
86
|
+
|
|
87
|
+
function assertExists(file, fsSync) {
|
|
88
|
+
if (fsSync.existsSync(file)) return { pass: true };
|
|
89
|
+
return { pass: false, message: `${file} not found` };
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
function assertGrep(pattern, file, fsSync) {
|
|
93
|
+
const content = fsSync.readFileSync(file, "utf8");
|
|
94
|
+
const re = new RegExp(pattern, "im");
|
|
95
|
+
if (re.test(content)) return { pass: true };
|
|
96
|
+
return {
|
|
97
|
+
pass: false,
|
|
98
|
+
message: `pattern "${pattern}" not found in ${basename(file)}`,
|
|
99
|
+
};
|
|
100
|
+
}
|
|
101
|
+
|
|
102
|
+
function assertQuery(expression, file, fsSync) {
|
|
103
|
+
const content = fsSync.readFileSync(file, "utf8");
|
|
104
|
+
const data = parseJsonOrNdjson(content);
|
|
105
|
+
const result = jmespath.search(data, expression);
|
|
106
|
+
const truthy =
|
|
107
|
+
result !== null &&
|
|
108
|
+
result !== undefined &&
|
|
109
|
+
result !== false &&
|
|
110
|
+
(Array.isArray(result) ? result.length > 0 : true);
|
|
111
|
+
if (truthy) return { pass: true };
|
|
112
|
+
return {
|
|
113
|
+
pass: false,
|
|
114
|
+
message: `query returned ${JSON.stringify(result)}`,
|
|
115
|
+
};
|
|
116
|
+
}
|
|
117
|
+
|
|
118
|
+
const JOB_TAG_RE = /<job\s+user="([^"]*)"\s+goal="([^"]*)">/;
|
|
119
|
+
|
|
120
|
+
function assertCitesJob(jobFile, file, fsSync) {
|
|
121
|
+
const jobContent = fsSync.readFileSync(jobFile, "utf8");
|
|
122
|
+
const match = JOB_TAG_RE.exec(jobContent);
|
|
123
|
+
if (!match) {
|
|
124
|
+
return {
|
|
125
|
+
pass: false,
|
|
126
|
+
message: `no <job> tag found in ${basename(jobFile)}`,
|
|
127
|
+
};
|
|
128
|
+
}
|
|
129
|
+
const citation = `${match[1]}: ${match[2]}`;
|
|
130
|
+
const content = fsSync.readFileSync(file, "utf8");
|
|
131
|
+
if (content.includes(citation)) return { pass: true };
|
|
132
|
+
return { pass: false, message: `missing "${citation}"` };
|
|
133
|
+
}
|
|
134
|
+
|
|
135
|
+
function parseJsonOrNdjson(content) {
|
|
136
|
+
try {
|
|
137
|
+
return JSON.parse(content);
|
|
138
|
+
} catch {
|
|
139
|
+
// Fall through to NDJSON
|
|
140
|
+
}
|
|
141
|
+
const lines = [];
|
|
142
|
+
for (const raw of content.split("\n")) {
|
|
143
|
+
const trimmed = raw.trim();
|
|
144
|
+
if (!trimmed) continue;
|
|
145
|
+
try {
|
|
146
|
+
lines.push(JSON.parse(trimmed));
|
|
147
|
+
} catch {
|
|
148
|
+
// skip unparseable lines
|
|
149
|
+
}
|
|
150
|
+
}
|
|
151
|
+
if (lines.length === 0) throw new Error("assert: no valid JSON in file");
|
|
152
|
+
return lines;
|
|
153
|
+
}
|
|
@@ -0,0 +1,175 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* `fit-benchmark` CLI definition. Lives in `src/` so the bin stays an
|
|
3
|
+
* execute-on-import entry point — launcher packages import the bin to run
|
|
4
|
+
* it — while tests import the definition without running the CLI.
|
|
5
|
+
*/
|
|
6
|
+
|
|
7
|
+
import { runBenchmarkRunCommand } from "./benchmark-run.js";
|
|
8
|
+
import { runBenchmarkInvariantsCommand } from "./benchmark-invariants.js";
|
|
9
|
+
import { runBenchmarkReportCommand } from "./benchmark-report.js";
|
|
10
|
+
import {
|
|
11
|
+
BENCHMARK_AGENT_MODEL,
|
|
12
|
+
LEAD_MODEL,
|
|
13
|
+
} from "@forwardimpact/libutil/models";
|
|
14
|
+
|
|
15
|
+
export const definition = {
|
|
16
|
+
name: "fit-benchmark",
|
|
17
|
+
description:
|
|
18
|
+
"Run coding-agent task families, grade hidden tests, and aggregate pass@k across runs.",
|
|
19
|
+
commands: [
|
|
20
|
+
{
|
|
21
|
+
name: "run",
|
|
22
|
+
args: [],
|
|
23
|
+
handler: runBenchmarkRunCommand,
|
|
24
|
+
description:
|
|
25
|
+
"Run every task in a family for N runs and emit one result record per (task, runIndex).",
|
|
26
|
+
options: {
|
|
27
|
+
family: {
|
|
28
|
+
type: "string",
|
|
29
|
+
description: "Path or git URL to a task family",
|
|
30
|
+
},
|
|
31
|
+
task: {
|
|
32
|
+
type: "string",
|
|
33
|
+
description:
|
|
34
|
+
"Run only this task id (directory name under tasks/, default: every task)",
|
|
35
|
+
},
|
|
36
|
+
"skills-from": {
|
|
37
|
+
type: "string",
|
|
38
|
+
description:
|
|
39
|
+
"Stage .claude/ from this directory (a root containing .claude/) instead of running apm install — exercise local, unpublished skills",
|
|
40
|
+
},
|
|
41
|
+
output: {
|
|
42
|
+
type: "string",
|
|
43
|
+
description:
|
|
44
|
+
"Run-output directory (created if missing, default: benchmark-runs)",
|
|
45
|
+
},
|
|
46
|
+
runs: {
|
|
47
|
+
type: "string",
|
|
48
|
+
description: "Runs per task (integer ≥ 1, default: 5)",
|
|
49
|
+
},
|
|
50
|
+
"agent-model": {
|
|
51
|
+
type: "string",
|
|
52
|
+
description: `Claude model for the agent-under-test (default: ${BENCHMARK_AGENT_MODEL})`,
|
|
53
|
+
},
|
|
54
|
+
"lead-model": {
|
|
55
|
+
type: "string",
|
|
56
|
+
description: `Claude model for the lead role (default: ${LEAD_MODEL})`,
|
|
57
|
+
},
|
|
58
|
+
"judge-model": {
|
|
59
|
+
type: "string",
|
|
60
|
+
description: `Claude model for the judge (default: ${LEAD_MODEL})`,
|
|
61
|
+
},
|
|
62
|
+
"agent-profile": {
|
|
63
|
+
type: "string",
|
|
64
|
+
description: "Agent-under-test profile name",
|
|
65
|
+
},
|
|
66
|
+
"judge-profile": {
|
|
67
|
+
type: "string",
|
|
68
|
+
description: "Judge profile name",
|
|
69
|
+
},
|
|
70
|
+
"work-tracker": {
|
|
71
|
+
type: "string",
|
|
72
|
+
description:
|
|
73
|
+
"Active work-item tracker (github|filesystem, default: github)",
|
|
74
|
+
},
|
|
75
|
+
"max-turns": {
|
|
76
|
+
type: "string",
|
|
77
|
+
description:
|
|
78
|
+
"Agent-under-test turn budget (default: 50, 0 = unlimited)",
|
|
79
|
+
},
|
|
80
|
+
concurrency: {
|
|
81
|
+
type: "string",
|
|
82
|
+
description:
|
|
83
|
+
"Max cells run concurrently (positive integer; default: CPU-aware min(4, max(2, cores/2)); env: LIBHARNESS_BENCHMARK_CONCURRENCY)",
|
|
84
|
+
},
|
|
85
|
+
shard: {
|
|
86
|
+
type: "string",
|
|
87
|
+
description:
|
|
88
|
+
"Run only shard i of N as i/N (1-based; default: the whole family). Each shard writes a partial results.jsonl; report --input merges them.",
|
|
89
|
+
},
|
|
90
|
+
"allowed-tools": {
|
|
91
|
+
type: "string",
|
|
92
|
+
description:
|
|
93
|
+
"Comma-separated tool allowlist for the agent-under-test (default: Bash,Read,Glob,Grep,Write,Edit,Agent,TodoWrite)",
|
|
94
|
+
},
|
|
95
|
+
},
|
|
96
|
+
},
|
|
97
|
+
{
|
|
98
|
+
name: "invariants",
|
|
99
|
+
args: [],
|
|
100
|
+
handler: runBenchmarkInvariantsCommand,
|
|
101
|
+
description:
|
|
102
|
+
"Check a single task's invariants against a post-run workdir without invoking an agent.",
|
|
103
|
+
options: {
|
|
104
|
+
family: {
|
|
105
|
+
type: "string",
|
|
106
|
+
description: "Path or git URL to a task family",
|
|
107
|
+
},
|
|
108
|
+
task: {
|
|
109
|
+
type: "string",
|
|
110
|
+
description: "Task id (directory name under tasks/)",
|
|
111
|
+
},
|
|
112
|
+
"run-dir": {
|
|
113
|
+
type: "string",
|
|
114
|
+
description:
|
|
115
|
+
"Post-run directory whose cwd/ subdir is the agent CWD; invariants run against that cwd — the path hooks receive as $AGENT_CWD",
|
|
116
|
+
},
|
|
117
|
+
output: {
|
|
118
|
+
type: "string",
|
|
119
|
+
description: "Output file (defaults to stdout; one JSONL line)",
|
|
120
|
+
},
|
|
121
|
+
},
|
|
122
|
+
},
|
|
123
|
+
{
|
|
124
|
+
name: "report",
|
|
125
|
+
args: [],
|
|
126
|
+
handler: runBenchmarkReportCommand,
|
|
127
|
+
description:
|
|
128
|
+
"Aggregate result records into pass@k via the OpenAI HumanEval estimator.",
|
|
129
|
+
options: {
|
|
130
|
+
input: {
|
|
131
|
+
type: "string",
|
|
132
|
+
description:
|
|
133
|
+
"Run-output directory containing results.jsonl (default: benchmark-runs)",
|
|
134
|
+
},
|
|
135
|
+
k: {
|
|
136
|
+
type: "string",
|
|
137
|
+
description: "Comma-separated k values (default: 1,3,5)",
|
|
138
|
+
},
|
|
139
|
+
format: {
|
|
140
|
+
type: "string",
|
|
141
|
+
description: "Output format (json|text, default: json)",
|
|
142
|
+
},
|
|
143
|
+
},
|
|
144
|
+
},
|
|
145
|
+
],
|
|
146
|
+
globalOptions: {
|
|
147
|
+
help: { type: "boolean", short: "h", description: "Show this help" },
|
|
148
|
+
version: { type: "boolean", description: "Show version" },
|
|
149
|
+
json: { type: "boolean", description: "Output help as JSON" },
|
|
150
|
+
},
|
|
151
|
+
examples: [
|
|
152
|
+
"fit-benchmark run --family=./families/coding",
|
|
153
|
+
"fit-benchmark run --family=./families/coding --task=todo-api --runs=1",
|
|
154
|
+
"fit-benchmark run --family=./families/coding --work-tracker=filesystem",
|
|
155
|
+
"fit-benchmark run --family=./families/coding --skills-from=. --task=todo-api",
|
|
156
|
+
`fit-benchmark run --family=./families/coding --runs=10 --agent-model=${BENCHMARK_AGENT_MODEL}`,
|
|
157
|
+
"fit-benchmark invariants --family=./families/coding --task=todo-api --run-dir=./benchmark-runs/runs/todo-api/0",
|
|
158
|
+
"fit-benchmark report --format=text",
|
|
159
|
+
"fit-benchmark report --input=./runs/today --k=1,3,5 --format=text",
|
|
160
|
+
],
|
|
161
|
+
documentation: [
|
|
162
|
+
{
|
|
163
|
+
title: "Run a Benchmark",
|
|
164
|
+
url: "https://www.forwardimpact.team/docs/libraries/prove-changes/run-benchmark/index.md",
|
|
165
|
+
description:
|
|
166
|
+
"Author a coding-task family, run a benchmark across multiple runs, and read the pass@k report.",
|
|
167
|
+
},
|
|
168
|
+
{
|
|
169
|
+
title: "Automate with GitHub Actions",
|
|
170
|
+
url: "https://www.forwardimpact.team/docs/libraries/prove-changes/run-benchmark/ci-workflow/index.md",
|
|
171
|
+
description:
|
|
172
|
+
"Run benchmarks in CI with the forwardimpact/fit-benchmark action.",
|
|
173
|
+
},
|
|
174
|
+
],
|
|
175
|
+
};
|
|
@@ -0,0 +1,73 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* `fit-benchmark invariants` — check a single task's invariants against a
|
|
3
|
+
* post-run workdir directory without invoking an agent. Useful for
|
|
4
|
+
* re-checking an agent's output against revised grading material.
|
|
5
|
+
*/
|
|
6
|
+
|
|
7
|
+
import { join, resolve } from "node:path";
|
|
8
|
+
import { createServer } from "node:net";
|
|
9
|
+
|
|
10
|
+
import { validateInvariantsRecord } from "../benchmark/result.js";
|
|
11
|
+
import { runInvariants } from "../benchmark/invariants.js";
|
|
12
|
+
import { loadTaskFamily } from "../benchmark/task-family.js";
|
|
13
|
+
|
|
14
|
+
/**
|
|
15
|
+
* @param {import("@forwardimpact/libcli").InvocationContext} ctx
|
|
16
|
+
* @returns {Promise<{ok: true} | {ok: false, code: number, error: string}>}
|
|
17
|
+
*/
|
|
18
|
+
export async function runBenchmarkInvariantsCommand(ctx) {
|
|
19
|
+
const values = ctx.options;
|
|
20
|
+
const runtime = ctx.deps.runtime;
|
|
21
|
+
const familyInput = values.family;
|
|
22
|
+
if (!familyInput)
|
|
23
|
+
return { ok: false, code: 1, error: "--family is required" };
|
|
24
|
+
const taskId = values.task;
|
|
25
|
+
if (!taskId) return { ok: false, code: 1, error: "--task is required" };
|
|
26
|
+
const runDirArg = values["run-dir"];
|
|
27
|
+
if (!runDirArg) return { ok: false, code: 1, error: "--run-dir is required" };
|
|
28
|
+
|
|
29
|
+
const family = await loadTaskFamily(familyInput, runtime);
|
|
30
|
+
const task = family.tasks().find((t) => t.id === taskId);
|
|
31
|
+
if (!task)
|
|
32
|
+
return { ok: false, code: 1, error: `task not found in family: ${taskId}` };
|
|
33
|
+
|
|
34
|
+
const runDir = resolve(runDirArg);
|
|
35
|
+
const cwd = join(runDir, "cwd");
|
|
36
|
+
const port = await allocatePort();
|
|
37
|
+
|
|
38
|
+
const invariants = await runInvariants(task, { cwd, port, runDir }, runtime);
|
|
39
|
+
const record = {
|
|
40
|
+
taskId: task.id,
|
|
41
|
+
invariants,
|
|
42
|
+
exitCode: invariants.exitCode,
|
|
43
|
+
};
|
|
44
|
+
validateInvariantsRecord(record);
|
|
45
|
+
|
|
46
|
+
const line = JSON.stringify(record) + "\n";
|
|
47
|
+
if (values.output) {
|
|
48
|
+
runtime.fsSync.writeFileSync(resolve(values.output), line);
|
|
49
|
+
} else {
|
|
50
|
+
runtime.proc.stdout.write(line);
|
|
51
|
+
}
|
|
52
|
+
return invariants.verdict === "pass"
|
|
53
|
+
? { ok: true }
|
|
54
|
+
: { ok: false, code: 1, error: "" };
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
function allocatePort() {
|
|
58
|
+
return new Promise((res, rej) => {
|
|
59
|
+
const server = createServer();
|
|
60
|
+
server.unref();
|
|
61
|
+
server.on("error", rej);
|
|
62
|
+
server.listen(0, "127.0.0.1", () => {
|
|
63
|
+
const addr = server.address();
|
|
64
|
+
if (!addr || typeof addr === "string") {
|
|
65
|
+
server.close();
|
|
66
|
+
rej(new Error("failed to allocate port"));
|
|
67
|
+
return;
|
|
68
|
+
}
|
|
69
|
+
const port = addr.port;
|
|
70
|
+
server.close(() => res(port));
|
|
71
|
+
});
|
|
72
|
+
});
|
|
73
|
+
}
|
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* `fit-benchmark report` — aggregate `results.jsonl` into pass@k via the
|
|
3
|
+
* OpenAI HumanEval estimator. Output is JSON by default; pass --format=text
|
|
4
|
+
* to render a markdown table.
|
|
5
|
+
*/
|
|
6
|
+
|
|
7
|
+
import { resolve } from "node:path";
|
|
8
|
+
|
|
9
|
+
import { aggregate, renderTextReport } from "../benchmark/report.js";
|
|
10
|
+
|
|
11
|
+
/**
|
|
12
|
+
* @param {import("@forwardimpact/libcli").InvocationContext} ctx
|
|
13
|
+
* @returns {Promise<{ok: true} | {ok: false, code: number, error: string}>}
|
|
14
|
+
*/
|
|
15
|
+
export async function runBenchmarkReportCommand(ctx) {
|
|
16
|
+
const values = ctx.options;
|
|
17
|
+
const runtime = ctx.deps.runtime;
|
|
18
|
+
const inputDir = values.input ?? "benchmark-runs";
|
|
19
|
+
const kRaw = values.k ?? "1,3,5";
|
|
20
|
+
let kValues;
|
|
21
|
+
try {
|
|
22
|
+
kValues = kRaw.split(",").map((t) => {
|
|
23
|
+
const n = Number.parseInt(t.trim(), 10);
|
|
24
|
+
if (!Number.isFinite(n) || n < 1) {
|
|
25
|
+
throw new Error(
|
|
26
|
+
"--k must be a comma-separated list of positive integers",
|
|
27
|
+
);
|
|
28
|
+
}
|
|
29
|
+
return n;
|
|
30
|
+
});
|
|
31
|
+
} catch (err) {
|
|
32
|
+
return { ok: false, code: 1, error: err.message };
|
|
33
|
+
}
|
|
34
|
+
const format = values.format ?? "json";
|
|
35
|
+
if (format !== "json" && format !== "text") {
|
|
36
|
+
return { ok: false, code: 1, error: "--format must be 'json' or 'text'" };
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
const report = await aggregate({
|
|
40
|
+
inputDir: resolve(inputDir),
|
|
41
|
+
kValues,
|
|
42
|
+
includeRuns: format === "text",
|
|
43
|
+
runtime,
|
|
44
|
+
});
|
|
45
|
+
if (format === "text") {
|
|
46
|
+
runtime.proc.stdout.write(renderTextReport(report, kValues) + "\n");
|
|
47
|
+
} else {
|
|
48
|
+
runtime.proc.stdout.write(JSON.stringify(report, null, 2) + "\n");
|
|
49
|
+
}
|
|
50
|
+
return { ok: true };
|
|
51
|
+
}
|
|
@@ -0,0 +1,175 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* `fit-benchmark run` — run every task in a family for N runs, stream each
|
|
3
|
+
* ResultRecord to stdout (one JSON line per record), and append to the
|
|
4
|
+
* canonical `<output>/results.jsonl` for the report subcommand.
|
|
5
|
+
*/
|
|
6
|
+
|
|
7
|
+
import { resolve } from "node:path";
|
|
8
|
+
import { availableParallelism } from "node:os";
|
|
9
|
+
|
|
10
|
+
import { createConfig } from "@forwardimpact/libconfig";
|
|
11
|
+
import { createBenchmarkRunner } from "../benchmark/runner.js";
|
|
12
|
+
import { resolveWorkTracker } from "./work-tracker.js";
|
|
13
|
+
import {
|
|
14
|
+
BENCHMARK_AGENT_MODEL,
|
|
15
|
+
LEAD_MODEL,
|
|
16
|
+
} from "@forwardimpact/libutil/models";
|
|
17
|
+
|
|
18
|
+
/**
|
|
19
|
+
* @param {import("@forwardimpact/libcli").InvocationContext} ctx
|
|
20
|
+
* @returns {Promise<{ok: true} | {ok: false, code: number, error: string}>}
|
|
21
|
+
*/
|
|
22
|
+
export async function runBenchmarkRunCommand(ctx) {
|
|
23
|
+
const values = ctx.options;
|
|
24
|
+
const runtime = ctx.deps.runtime;
|
|
25
|
+
let opts;
|
|
26
|
+
try {
|
|
27
|
+
opts = parseRunOptions(values, runtime.proc.env);
|
|
28
|
+
} catch (err) {
|
|
29
|
+
return { ok: false, code: 1, error: err.message };
|
|
30
|
+
}
|
|
31
|
+
const config = await createConfig("script", "benchmark");
|
|
32
|
+
runtime.proc.env.ANTHROPIC_API_KEY = await config.anthropicToken();
|
|
33
|
+
// The benchmark agent runs via createBenchmarkRunner, not the supervise
|
|
34
|
+
// command, so the active-tracker env must land here before the runner
|
|
35
|
+
// spawns the subprocess that inherits process.env.
|
|
36
|
+
runtime.proc.env.LIBHARNESS_WORK_TRACKER = opts.workTracker;
|
|
37
|
+
|
|
38
|
+
// The Claude Agent SDK spawns a `claude` subprocess that inherits
|
|
39
|
+
// process.env. NODE_EXTRA_CA_CERTS causes undici (the HTTP client
|
|
40
|
+
// inside that subprocess) to fail with UND_ERR_INVALID_ARG on
|
|
41
|
+
// Node 22+, aborting every API call after 10 retries. Strip it
|
|
42
|
+
// before the SDK loads so the subprocess gets a clean environment.
|
|
43
|
+
delete runtime.proc.env.NODE_EXTRA_CA_CERTS;
|
|
44
|
+
|
|
45
|
+
const { query } = await import("@anthropic-ai/claude-agent-sdk");
|
|
46
|
+
const runner = createBenchmarkRunner({ ...opts, query, runtime });
|
|
47
|
+
|
|
48
|
+
let anyFail = false;
|
|
49
|
+
let count = 0;
|
|
50
|
+
for await (const record of runner.run()) {
|
|
51
|
+
count++;
|
|
52
|
+
runtime.proc.stdout.write(JSON.stringify(record) + "\n");
|
|
53
|
+
if (record.verdict !== "pass") anyFail = true;
|
|
54
|
+
}
|
|
55
|
+
if (count === 0) return resolveZeroRecordOutcome(opts, runtime);
|
|
56
|
+
return anyFail ? { ok: false, code: 1, error: "" } : { ok: true };
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
/**
|
|
60
|
+
* Decide the exit outcome when a run streamed zero records. A run that emits no
|
|
61
|
+
* records normally did nothing (no tasks discovered, or the agent never
|
|
62
|
+
* produced output) — a failure, surfaced loudly so CI does not go green on an
|
|
63
|
+
* empty benchmark. The one exception is a deliberately-empty shard: a
|
|
64
|
+
* high-index `--shard=i/N` with `N > cell count` legitimately selects zero
|
|
65
|
+
* cells, so it exits 0 with a stderr note. Exported so the relaxed-guard branch
|
|
66
|
+
* is testable without the full handler's config/SDK setup.
|
|
67
|
+
* @param {{shard: {index: number, total: number} | null}} opts
|
|
68
|
+
* @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
|
|
69
|
+
* @returns {{ok: true} | {ok: false, code: number, error: string}}
|
|
70
|
+
*/
|
|
71
|
+
export function resolveZeroRecordOutcome(opts, runtime) {
|
|
72
|
+
if (opts.shard) {
|
|
73
|
+
runtime.proc.stderr.write(
|
|
74
|
+
`shard ${opts.shard.index}/${opts.shard.total} selected no cells\n`,
|
|
75
|
+
);
|
|
76
|
+
return { ok: true };
|
|
77
|
+
}
|
|
78
|
+
return {
|
|
79
|
+
ok: false,
|
|
80
|
+
code: 1,
|
|
81
|
+
error:
|
|
82
|
+
"benchmark produced no result records — no task ran to completion; check the family's tasks/, apm install, and agent availability (ANTHROPIC_API_KEY / claude CLI / IS_SANDBOX)",
|
|
83
|
+
};
|
|
84
|
+
}
|
|
85
|
+
|
|
86
|
+
/**
|
|
87
|
+
* Parse and validate benchmark run options. Exported so tests can verify
|
|
88
|
+
* defaults, including the resolved work tracker.
|
|
89
|
+
* @param {Record<string, string|undefined>} values - Parsed option values
|
|
90
|
+
* @param {Record<string, string|undefined>} [env] - Process environment, read
|
|
91
|
+
* for the `LIBHARNESS_WORK_TRACKER` fallback when `--work-tracker` is absent.
|
|
92
|
+
* @returns {object}
|
|
93
|
+
*/
|
|
94
|
+
export function parseRunOptions(values, env = {}) {
|
|
95
|
+
const family = values.family;
|
|
96
|
+
if (!family) throw new Error("--family is required");
|
|
97
|
+
const output = values.output ?? "benchmark-runs";
|
|
98
|
+
const runs = Number.parseInt(values.runs ?? "5", 10);
|
|
99
|
+
if (!Number.isFinite(runs) || runs < 1)
|
|
100
|
+
throw new Error("--runs must be a positive integer");
|
|
101
|
+
return {
|
|
102
|
+
family,
|
|
103
|
+
runs,
|
|
104
|
+
task: values.task ?? null,
|
|
105
|
+
skillsFrom: values["skills-from"] ?? null,
|
|
106
|
+
output: resolve(output),
|
|
107
|
+
agentModel: values["agent-model"] || BENCHMARK_AGENT_MODEL,
|
|
108
|
+
supervisorModel: values["lead-model"] || LEAD_MODEL,
|
|
109
|
+
judgeModel: values["judge-model"] || LEAD_MODEL,
|
|
110
|
+
workTracker: resolveWorkTracker(values, env),
|
|
111
|
+
profiles: {
|
|
112
|
+
agent: values["agent-profile"] ?? null,
|
|
113
|
+
judge: values["judge-profile"] ?? null,
|
|
114
|
+
},
|
|
115
|
+
maxTurns: parseMaxTurns(values["max-turns"]),
|
|
116
|
+
concurrency: resolveConcurrency(values, env),
|
|
117
|
+
shard: parseShard(values.shard),
|
|
118
|
+
allowedTools: values["allowed-tools"]
|
|
119
|
+
? values["allowed-tools"]
|
|
120
|
+
.split(",")
|
|
121
|
+
.map((s) => s.trim())
|
|
122
|
+
.filter(Boolean)
|
|
123
|
+
: undefined,
|
|
124
|
+
};
|
|
125
|
+
}
|
|
126
|
+
|
|
127
|
+
function parseMaxTurns(raw) {
|
|
128
|
+
if (raw === undefined) return undefined;
|
|
129
|
+
if (raw === "0") return 0;
|
|
130
|
+
return Number.parseInt(raw, 10);
|
|
131
|
+
}
|
|
132
|
+
|
|
133
|
+
/**
|
|
134
|
+
* Parse a `--shard=<i>/<N>` selector into `{index, total}` (1-based), or `null`
|
|
135
|
+
* for an unsharded run. Validates `1 ≤ index ≤ total` with integer parts.
|
|
136
|
+
* @param {string|undefined} raw
|
|
137
|
+
* @returns {{index: number, total: number} | null}
|
|
138
|
+
*/
|
|
139
|
+
export function parseShard(raw) {
|
|
140
|
+
if (raw == null || raw === "") return null;
|
|
141
|
+
const m = /^(\d+)\/(\d+)$/.exec(raw.trim());
|
|
142
|
+
if (!m) throw new Error("--shard must be in the form i/N (e.g. 1/4)");
|
|
143
|
+
const index = Number.parseInt(m[1], 10);
|
|
144
|
+
const total = Number.parseInt(m[2], 10);
|
|
145
|
+
if (total < 1 || index < 1 || index > total)
|
|
146
|
+
throw new Error("--shard requires 1 ≤ i ≤ N");
|
|
147
|
+
return { index, total };
|
|
148
|
+
}
|
|
149
|
+
|
|
150
|
+
// Conservative because each cell spawns ~3 agent subprocesses (lead +
|
|
151
|
+
// agent-under-test + judge); a low ceiling keeps a single runner from
|
|
152
|
+
// thrashing. The bulk of the CI speedup comes from Layer-2 sharding across
|
|
153
|
+
// machines, not from raising this in-job default.
|
|
154
|
+
const CONCURRENCY_CEILING = 4;
|
|
155
|
+
|
|
156
|
+
/**
|
|
157
|
+
* Resolve the cell concurrency: `--concurrency` flag > the
|
|
158
|
+
* `LIBHARNESS_BENCHMARK_CONCURRENCY` env var > a CPU-aware default of
|
|
159
|
+
* `min(CONCURRENCY_CEILING, max(2, ⌊cores/2⌋))`. The default is `> 1` so
|
|
160
|
+
* concurrency is on transparently without any consumer opting in.
|
|
161
|
+
* @param {Record<string, string|undefined>} values
|
|
162
|
+
* @param {Record<string, string|undefined>} [env]
|
|
163
|
+
* @returns {number}
|
|
164
|
+
*/
|
|
165
|
+
export function resolveConcurrency(values, env = {}) {
|
|
166
|
+
const raw = values.concurrency ?? env.LIBHARNESS_BENCHMARK_CONCURRENCY;
|
|
167
|
+
if (raw != null && raw !== "") {
|
|
168
|
+
const n = Number.parseInt(raw, 10);
|
|
169
|
+
if (!Number.isFinite(n) || n < 1)
|
|
170
|
+
throw new Error("--concurrency must be a positive integer");
|
|
171
|
+
return n;
|
|
172
|
+
}
|
|
173
|
+
const cores = availableParallelism();
|
|
174
|
+
return Math.min(CONCURRENCY_CEILING, Math.max(2, Math.floor(cores / 2)));
|
|
175
|
+
}
|