@forwardimpact/libharness 0.1.20 → 1.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -201
- package/README.md +196 -80
- package/bin/fit-benchmark.js +44 -0
- package/bin/fit-harness.js +358 -0
- package/bin/fit-selfedit.js +165 -0
- package/bin/fit-trace.js +510 -0
- package/package.json +42 -12
- package/src/agent-runner.js +256 -0
- package/src/benchmark/apm-installer.js +207 -0
- package/src/benchmark/env-loader.js +158 -0
- package/src/benchmark/hook-env.js +40 -0
- package/src/benchmark/invariants.js +141 -0
- package/src/benchmark/judge.js +187 -0
- package/src/benchmark/npm-installer.js +87 -0
- package/src/benchmark/report.js +522 -0
- package/src/benchmark/result.js +127 -0
- package/src/benchmark/runner.js +583 -0
- package/src/benchmark/task-family.js +260 -0
- package/src/benchmark/workdir.js +298 -0
- package/src/commands/assert.js +153 -0
- package/src/commands/benchmark-definition.js +165 -0
- package/src/commands/benchmark-invariants.js +73 -0
- package/src/commands/benchmark-report.js +51 -0
- package/src/commands/benchmark-run.js +111 -0
- package/src/commands/by-discussion.js +94 -0
- package/src/commands/callback.js +119 -0
- package/src/commands/discuss.js +132 -0
- package/src/commands/facilitate.js +123 -0
- package/src/commands/output.js +36 -0
- package/src/commands/run.js +152 -0
- package/src/commands/supervise.js +136 -0
- package/src/commands/task-input.js +54 -0
- package/src/commands/tee.js +53 -0
- package/src/commands/trace.js +630 -0
- package/src/commands/work-tracker.js +35 -0
- package/src/cost.js +79 -0
- package/src/discuss-tools.js +173 -0
- package/src/discusser.js +394 -0
- package/src/events/github.js +161 -0
- package/src/facilitator.js +205 -0
- package/src/inbox-poller.js +81 -0
- package/src/index.js +72 -2
- package/src/judge.js +210 -0
- package/src/message-bus.js +118 -0
- package/src/orchestration-loop.js +330 -0
- package/src/orchestration-toolkit.js +441 -0
- package/src/orchestrator-helpers.js +23 -0
- package/src/profile-prompt.js +266 -0
- package/src/redaction.js +253 -0
- package/src/render/line-renderer.js +54 -0
- package/src/render/orchestrator-filter.js +19 -0
- package/src/render/palette.js +63 -0
- package/src/render/tool-hints.js +154 -0
- package/src/render/turn-renderer.js +96 -0
- package/src/reply-emitter.js +47 -0
- package/src/sequence-counter.js +21 -0
- package/src/signature-filter.js +27 -0
- package/src/supervisor.js +236 -0
- package/src/tee-writer.js +150 -0
- package/src/trace-collector.js +444 -0
- package/src/trace-github.js +473 -0
- package/src/trace-multi.js +101 -0
- package/src/trace-query.js +748 -0
- package/src/trace-render.js +211 -0
- package/src/trace-usage.js +249 -0
- package/src/fixture/assertions.js +0 -42
- package/src/fixture/cache.js +0 -50
- package/src/fixture/eval.js +0 -146
- package/src/fixture/index.js +0 -9
- package/src/fixture/pathway.js +0 -451
- package/src/fixture/services.js +0 -56
- package/src/mock/clients.js +0 -135
- package/src/mock/config.js +0 -45
- package/src/mock/data.js +0 -46
- package/src/mock/fs.js +0 -111
- package/src/mock/grpc.js +0 -94
- package/src/mock/http.js +0 -60
- package/src/mock/index.js +0 -36
- package/src/mock/infra.js +0 -219
- package/src/mock/logger.js +0 -42
- package/src/mock/observer.js +0 -74
- package/src/mock/resource-index.js +0 -95
- package/src/mock/service-callbacks.js +0 -39
- package/src/mock/services.js +0 -79
- package/src/mock/spy.js +0 -44
- package/src/mock/storage.js +0 -118
|
@@ -0,0 +1,165 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* `fit-benchmark` CLI definition. Lives in `src/` so the bin stays an
|
|
3
|
+
* execute-on-import entry point — launcher packages import the bin to run
|
|
4
|
+
* it — while tests import the definition without running the CLI.
|
|
5
|
+
*/
|
|
6
|
+
|
|
7
|
+
import { runBenchmarkRunCommand } from "./benchmark-run.js";
|
|
8
|
+
import { runBenchmarkInvariantsCommand } from "./benchmark-invariants.js";
|
|
9
|
+
import { runBenchmarkReportCommand } from "./benchmark-report.js";
|
|
10
|
+
import {
|
|
11
|
+
BENCHMARK_AGENT_MODEL,
|
|
12
|
+
LEAD_MODEL,
|
|
13
|
+
} from "@forwardimpact/libutil/models";
|
|
14
|
+
|
|
15
|
+
export const definition = {
|
|
16
|
+
name: "fit-benchmark",
|
|
17
|
+
description:
|
|
18
|
+
"Run coding-agent task families, grade hidden tests, and aggregate pass@k across runs.",
|
|
19
|
+
commands: [
|
|
20
|
+
{
|
|
21
|
+
name: "run",
|
|
22
|
+
args: [],
|
|
23
|
+
handler: runBenchmarkRunCommand,
|
|
24
|
+
description:
|
|
25
|
+
"Run every task in a family for N runs and emit one result record per (task, runIndex).",
|
|
26
|
+
options: {
|
|
27
|
+
family: {
|
|
28
|
+
type: "string",
|
|
29
|
+
description: "Path or git URL to a task family",
|
|
30
|
+
},
|
|
31
|
+
task: {
|
|
32
|
+
type: "string",
|
|
33
|
+
description:
|
|
34
|
+
"Run only this task id (directory name under tasks/, default: every task)",
|
|
35
|
+
},
|
|
36
|
+
"skills-from": {
|
|
37
|
+
type: "string",
|
|
38
|
+
description:
|
|
39
|
+
"Stage .claude/ from this directory (a root containing .claude/) instead of running apm install — exercise local, unpublished skills",
|
|
40
|
+
},
|
|
41
|
+
output: {
|
|
42
|
+
type: "string",
|
|
43
|
+
description:
|
|
44
|
+
"Run-output directory (created if missing, default: benchmark-runs)",
|
|
45
|
+
},
|
|
46
|
+
runs: {
|
|
47
|
+
type: "string",
|
|
48
|
+
description: "Runs per task (integer ≥ 1, default: 5)",
|
|
49
|
+
},
|
|
50
|
+
"agent-model": {
|
|
51
|
+
type: "string",
|
|
52
|
+
description: `Claude model for the agent-under-test (default: ${BENCHMARK_AGENT_MODEL})`,
|
|
53
|
+
},
|
|
54
|
+
"lead-model": {
|
|
55
|
+
type: "string",
|
|
56
|
+
description: `Claude model for the lead role (default: ${LEAD_MODEL})`,
|
|
57
|
+
},
|
|
58
|
+
"judge-model": {
|
|
59
|
+
type: "string",
|
|
60
|
+
description: `Claude model for the judge (default: ${LEAD_MODEL})`,
|
|
61
|
+
},
|
|
62
|
+
"agent-profile": {
|
|
63
|
+
type: "string",
|
|
64
|
+
description: "Agent-under-test profile name",
|
|
65
|
+
},
|
|
66
|
+
"judge-profile": {
|
|
67
|
+
type: "string",
|
|
68
|
+
description: "Judge profile name",
|
|
69
|
+
},
|
|
70
|
+
"work-tracker": {
|
|
71
|
+
type: "string",
|
|
72
|
+
description:
|
|
73
|
+
"Active work-item tracker (github|filesystem, default: github)",
|
|
74
|
+
},
|
|
75
|
+
"max-turns": {
|
|
76
|
+
type: "string",
|
|
77
|
+
description:
|
|
78
|
+
"Agent-under-test turn budget (default: 50, 0 = unlimited)",
|
|
79
|
+
},
|
|
80
|
+
"allowed-tools": {
|
|
81
|
+
type: "string",
|
|
82
|
+
description:
|
|
83
|
+
"Comma-separated tool allowlist for the agent-under-test (default: Bash,Read,Glob,Grep,Write,Edit,Agent,TodoWrite)",
|
|
84
|
+
},
|
|
85
|
+
},
|
|
86
|
+
},
|
|
87
|
+
{
|
|
88
|
+
name: "invariants",
|
|
89
|
+
args: [],
|
|
90
|
+
handler: runBenchmarkInvariantsCommand,
|
|
91
|
+
description:
|
|
92
|
+
"Check a single task's invariants against a post-run workdir without invoking an agent.",
|
|
93
|
+
options: {
|
|
94
|
+
family: {
|
|
95
|
+
type: "string",
|
|
96
|
+
description: "Path or git URL to a task family",
|
|
97
|
+
},
|
|
98
|
+
task: {
|
|
99
|
+
type: "string",
|
|
100
|
+
description: "Task id (directory name under tasks/)",
|
|
101
|
+
},
|
|
102
|
+
"run-dir": {
|
|
103
|
+
type: "string",
|
|
104
|
+
description:
|
|
105
|
+
"Post-run directory whose cwd/ subdir is the agent CWD; invariants run against that cwd — the path hooks receive as $AGENT_CWD",
|
|
106
|
+
},
|
|
107
|
+
output: {
|
|
108
|
+
type: "string",
|
|
109
|
+
description: "Output file (defaults to stdout; one JSONL line)",
|
|
110
|
+
},
|
|
111
|
+
},
|
|
112
|
+
},
|
|
113
|
+
{
|
|
114
|
+
name: "report",
|
|
115
|
+
args: [],
|
|
116
|
+
handler: runBenchmarkReportCommand,
|
|
117
|
+
description:
|
|
118
|
+
"Aggregate result records into pass@k via the OpenAI HumanEval estimator.",
|
|
119
|
+
options: {
|
|
120
|
+
input: {
|
|
121
|
+
type: "string",
|
|
122
|
+
description:
|
|
123
|
+
"Run-output directory containing results.jsonl (default: benchmark-runs)",
|
|
124
|
+
},
|
|
125
|
+
k: {
|
|
126
|
+
type: "string",
|
|
127
|
+
description: "Comma-separated k values (default: 1,3,5)",
|
|
128
|
+
},
|
|
129
|
+
format: {
|
|
130
|
+
type: "string",
|
|
131
|
+
description: "Output format (json|text, default: json)",
|
|
132
|
+
},
|
|
133
|
+
},
|
|
134
|
+
},
|
|
135
|
+
],
|
|
136
|
+
globalOptions: {
|
|
137
|
+
help: { type: "boolean", short: "h", description: "Show this help" },
|
|
138
|
+
version: { type: "boolean", description: "Show version" },
|
|
139
|
+
json: { type: "boolean", description: "Output help as JSON" },
|
|
140
|
+
},
|
|
141
|
+
examples: [
|
|
142
|
+
"fit-benchmark run --family=./families/coding",
|
|
143
|
+
"fit-benchmark run --family=./families/coding --task=todo-api --runs=1",
|
|
144
|
+
"fit-benchmark run --family=./families/coding --work-tracker=filesystem",
|
|
145
|
+
"fit-benchmark run --family=./families/coding --skills-from=. --task=todo-api",
|
|
146
|
+
`fit-benchmark run --family=./families/coding --runs=10 --agent-model=${BENCHMARK_AGENT_MODEL}`,
|
|
147
|
+
"fit-benchmark invariants --family=./families/coding --task=todo-api --run-dir=./benchmark-runs/runs/todo-api/0",
|
|
148
|
+
"fit-benchmark report --format=text",
|
|
149
|
+
"fit-benchmark report --input=./runs/today --k=1,3,5 --format=text",
|
|
150
|
+
],
|
|
151
|
+
documentation: [
|
|
152
|
+
{
|
|
153
|
+
title: "Run a Benchmark",
|
|
154
|
+
url: "https://www.forwardimpact.team/docs/libraries/prove-changes/run-benchmark/index.md",
|
|
155
|
+
description:
|
|
156
|
+
"Author a coding-task family, run a benchmark across multiple runs, and read the pass@k report.",
|
|
157
|
+
},
|
|
158
|
+
{
|
|
159
|
+
title: "Automate with GitHub Actions",
|
|
160
|
+
url: "https://www.forwardimpact.team/docs/libraries/prove-changes/run-benchmark/ci-workflow/index.md",
|
|
161
|
+
description:
|
|
162
|
+
"Run benchmarks in CI with the forwardimpact/fit-benchmark action.",
|
|
163
|
+
},
|
|
164
|
+
],
|
|
165
|
+
};
|
|
@@ -0,0 +1,73 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* `fit-benchmark invariants` — check a single task's invariants against a
|
|
3
|
+
* post-run workdir directory without invoking an agent. Useful for
|
|
4
|
+
* re-checking an agent's output against revised grading material.
|
|
5
|
+
*/
|
|
6
|
+
|
|
7
|
+
import { join, resolve } from "node:path";
|
|
8
|
+
import { createServer } from "node:net";
|
|
9
|
+
|
|
10
|
+
import { validateInvariantsRecord } from "../benchmark/result.js";
|
|
11
|
+
import { runInvariants } from "../benchmark/invariants.js";
|
|
12
|
+
import { loadTaskFamily } from "../benchmark/task-family.js";
|
|
13
|
+
|
|
14
|
+
/**
|
|
15
|
+
* @param {import("@forwardimpact/libcli").InvocationContext} ctx
|
|
16
|
+
* @returns {Promise<{ok: true} | {ok: false, code: number, error: string}>}
|
|
17
|
+
*/
|
|
18
|
+
export async function runBenchmarkInvariantsCommand(ctx) {
|
|
19
|
+
const values = ctx.options;
|
|
20
|
+
const runtime = ctx.deps.runtime;
|
|
21
|
+
const familyInput = values.family;
|
|
22
|
+
if (!familyInput)
|
|
23
|
+
return { ok: false, code: 1, error: "--family is required" };
|
|
24
|
+
const taskId = values.task;
|
|
25
|
+
if (!taskId) return { ok: false, code: 1, error: "--task is required" };
|
|
26
|
+
const runDirArg = values["run-dir"];
|
|
27
|
+
if (!runDirArg) return { ok: false, code: 1, error: "--run-dir is required" };
|
|
28
|
+
|
|
29
|
+
const family = await loadTaskFamily(familyInput, runtime);
|
|
30
|
+
const task = family.tasks().find((t) => t.id === taskId);
|
|
31
|
+
if (!task)
|
|
32
|
+
return { ok: false, code: 1, error: `task not found in family: ${taskId}` };
|
|
33
|
+
|
|
34
|
+
const runDir = resolve(runDirArg);
|
|
35
|
+
const cwd = join(runDir, "cwd");
|
|
36
|
+
const port = await allocatePort();
|
|
37
|
+
|
|
38
|
+
const invariants = await runInvariants(task, { cwd, port, runDir }, runtime);
|
|
39
|
+
const record = {
|
|
40
|
+
taskId: task.id,
|
|
41
|
+
invariants,
|
|
42
|
+
exitCode: invariants.exitCode,
|
|
43
|
+
};
|
|
44
|
+
validateInvariantsRecord(record);
|
|
45
|
+
|
|
46
|
+
const line = JSON.stringify(record) + "\n";
|
|
47
|
+
if (values.output) {
|
|
48
|
+
runtime.fsSync.writeFileSync(resolve(values.output), line);
|
|
49
|
+
} else {
|
|
50
|
+
runtime.proc.stdout.write(line);
|
|
51
|
+
}
|
|
52
|
+
return invariants.verdict === "pass"
|
|
53
|
+
? { ok: true }
|
|
54
|
+
: { ok: false, code: 1, error: "" };
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
function allocatePort() {
|
|
58
|
+
return new Promise((res, rej) => {
|
|
59
|
+
const server = createServer();
|
|
60
|
+
server.unref();
|
|
61
|
+
server.on("error", rej);
|
|
62
|
+
server.listen(0, "127.0.0.1", () => {
|
|
63
|
+
const addr = server.address();
|
|
64
|
+
if (!addr || typeof addr === "string") {
|
|
65
|
+
server.close();
|
|
66
|
+
rej(new Error("failed to allocate port"));
|
|
67
|
+
return;
|
|
68
|
+
}
|
|
69
|
+
const port = addr.port;
|
|
70
|
+
server.close(() => res(port));
|
|
71
|
+
});
|
|
72
|
+
});
|
|
73
|
+
}
|
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* `fit-benchmark report` — aggregate `results.jsonl` into pass@k via the
|
|
3
|
+
* OpenAI HumanEval estimator. Output is JSON by default; pass --format=text
|
|
4
|
+
* to render a markdown table.
|
|
5
|
+
*/
|
|
6
|
+
|
|
7
|
+
import { resolve } from "node:path";
|
|
8
|
+
|
|
9
|
+
import { aggregate, renderTextReport } from "../benchmark/report.js";
|
|
10
|
+
|
|
11
|
+
/**
|
|
12
|
+
* @param {import("@forwardimpact/libcli").InvocationContext} ctx
|
|
13
|
+
* @returns {Promise<{ok: true} | {ok: false, code: number, error: string}>}
|
|
14
|
+
*/
|
|
15
|
+
export async function runBenchmarkReportCommand(ctx) {
|
|
16
|
+
const values = ctx.options;
|
|
17
|
+
const runtime = ctx.deps.runtime;
|
|
18
|
+
const inputDir = values.input ?? "benchmark-runs";
|
|
19
|
+
const kRaw = values.k ?? "1,3,5";
|
|
20
|
+
let kValues;
|
|
21
|
+
try {
|
|
22
|
+
kValues = kRaw.split(",").map((t) => {
|
|
23
|
+
const n = Number.parseInt(t.trim(), 10);
|
|
24
|
+
if (!Number.isFinite(n) || n < 1) {
|
|
25
|
+
throw new Error(
|
|
26
|
+
"--k must be a comma-separated list of positive integers",
|
|
27
|
+
);
|
|
28
|
+
}
|
|
29
|
+
return n;
|
|
30
|
+
});
|
|
31
|
+
} catch (err) {
|
|
32
|
+
return { ok: false, code: 1, error: err.message };
|
|
33
|
+
}
|
|
34
|
+
const format = values.format ?? "json";
|
|
35
|
+
if (format !== "json" && format !== "text") {
|
|
36
|
+
return { ok: false, code: 1, error: "--format must be 'json' or 'text'" };
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
const report = await aggregate({
|
|
40
|
+
inputDir: resolve(inputDir),
|
|
41
|
+
kValues,
|
|
42
|
+
includeRuns: format === "text",
|
|
43
|
+
runtime,
|
|
44
|
+
});
|
|
45
|
+
if (format === "text") {
|
|
46
|
+
runtime.proc.stdout.write(renderTextReport(report, kValues) + "\n");
|
|
47
|
+
} else {
|
|
48
|
+
runtime.proc.stdout.write(JSON.stringify(report, null, 2) + "\n");
|
|
49
|
+
}
|
|
50
|
+
return { ok: true };
|
|
51
|
+
}
|
|
@@ -0,0 +1,111 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* `fit-benchmark run` — run every task in a family for N runs, stream each
|
|
3
|
+
* ResultRecord to stdout (one JSON line per record), and append to the
|
|
4
|
+
* canonical `<output>/results.jsonl` for the report subcommand.
|
|
5
|
+
*/
|
|
6
|
+
|
|
7
|
+
import { resolve } from "node:path";
|
|
8
|
+
|
|
9
|
+
import { createConfig } from "@forwardimpact/libconfig";
|
|
10
|
+
import { createBenchmarkRunner } from "../benchmark/runner.js";
|
|
11
|
+
import { resolveWorkTracker } from "./work-tracker.js";
|
|
12
|
+
import {
|
|
13
|
+
BENCHMARK_AGENT_MODEL,
|
|
14
|
+
LEAD_MODEL,
|
|
15
|
+
} from "@forwardimpact/libutil/models";
|
|
16
|
+
|
|
17
|
+
/**
|
|
18
|
+
* @param {import("@forwardimpact/libcli").InvocationContext} ctx
|
|
19
|
+
* @returns {Promise<{ok: true} | {ok: false, code: number, error: string}>}
|
|
20
|
+
*/
|
|
21
|
+
export async function runBenchmarkRunCommand(ctx) {
|
|
22
|
+
const values = ctx.options;
|
|
23
|
+
const runtime = ctx.deps.runtime;
|
|
24
|
+
let opts;
|
|
25
|
+
try {
|
|
26
|
+
opts = parseRunOptions(values, runtime.proc.env);
|
|
27
|
+
} catch (err) {
|
|
28
|
+
return { ok: false, code: 1, error: err.message };
|
|
29
|
+
}
|
|
30
|
+
const config = await createConfig("script", "benchmark");
|
|
31
|
+
runtime.proc.env.ANTHROPIC_API_KEY = await config.anthropicToken();
|
|
32
|
+
// The benchmark agent runs via createBenchmarkRunner, not the supervise
|
|
33
|
+
// command, so the active-tracker env must land here before the runner
|
|
34
|
+
// spawns the subprocess that inherits process.env.
|
|
35
|
+
runtime.proc.env.LIBHARNESS_WORK_TRACKER = opts.workTracker;
|
|
36
|
+
|
|
37
|
+
// The Claude Agent SDK spawns a `claude` subprocess that inherits
|
|
38
|
+
// process.env. NODE_EXTRA_CA_CERTS causes undici (the HTTP client
|
|
39
|
+
// inside that subprocess) to fail with UND_ERR_INVALID_ARG on
|
|
40
|
+
// Node 22+, aborting every API call after 10 retries. Strip it
|
|
41
|
+
// before the SDK loads so the subprocess gets a clean environment.
|
|
42
|
+
delete runtime.proc.env.NODE_EXTRA_CA_CERTS;
|
|
43
|
+
|
|
44
|
+
const { query } = await import("@anthropic-ai/claude-agent-sdk");
|
|
45
|
+
const runner = createBenchmarkRunner({ ...opts, query, runtime });
|
|
46
|
+
|
|
47
|
+
let anyFail = false;
|
|
48
|
+
let count = 0;
|
|
49
|
+
for await (const record of runner.run()) {
|
|
50
|
+
count++;
|
|
51
|
+
runtime.proc.stdout.write(JSON.stringify(record) + "\n");
|
|
52
|
+
if (record.verdict !== "pass") anyFail = true;
|
|
53
|
+
}
|
|
54
|
+
// A run that emits zero records did nothing (no tasks discovered, or the
|
|
55
|
+
// agent never produced output). That is a failure, not a silent success —
|
|
56
|
+
// surface it loudly so CI does not go green on an empty benchmark.
|
|
57
|
+
if (count === 0) {
|
|
58
|
+
return {
|
|
59
|
+
ok: false,
|
|
60
|
+
code: 1,
|
|
61
|
+
error:
|
|
62
|
+
"benchmark produced no result records — no task ran to completion; check the family's tasks/, apm install, and agent availability (ANTHROPIC_API_KEY / claude CLI / IS_SANDBOX)",
|
|
63
|
+
};
|
|
64
|
+
}
|
|
65
|
+
return anyFail ? { ok: false, code: 1, error: "" } : { ok: true };
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
/**
|
|
69
|
+
* Parse and validate benchmark run options. Exported so tests can verify
|
|
70
|
+
* defaults, including the resolved work tracker.
|
|
71
|
+
* @param {Record<string, string|undefined>} values - Parsed option values
|
|
72
|
+
* @param {Record<string, string|undefined>} [env] - Process environment, read
|
|
73
|
+
* for the `LIBHARNESS_WORK_TRACKER` fallback when `--work-tracker` is absent.
|
|
74
|
+
* @returns {object}
|
|
75
|
+
*/
|
|
76
|
+
export function parseRunOptions(values, env = {}) {
|
|
77
|
+
const family = values.family;
|
|
78
|
+
if (!family) throw new Error("--family is required");
|
|
79
|
+
const output = values.output ?? "benchmark-runs";
|
|
80
|
+
const runs = Number.parseInt(values.runs ?? "5", 10);
|
|
81
|
+
if (!Number.isFinite(runs) || runs < 1)
|
|
82
|
+
throw new Error("--runs must be a positive integer");
|
|
83
|
+
return {
|
|
84
|
+
family,
|
|
85
|
+
runs,
|
|
86
|
+
task: values.task ?? null,
|
|
87
|
+
skillsFrom: values["skills-from"] ?? null,
|
|
88
|
+
output: resolve(output),
|
|
89
|
+
agentModel: values["agent-model"] || BENCHMARK_AGENT_MODEL,
|
|
90
|
+
supervisorModel: values["lead-model"] || LEAD_MODEL,
|
|
91
|
+
judgeModel: values["judge-model"] || LEAD_MODEL,
|
|
92
|
+
workTracker: resolveWorkTracker(values, env),
|
|
93
|
+
profiles: {
|
|
94
|
+
agent: values["agent-profile"] ?? null,
|
|
95
|
+
judge: values["judge-profile"] ?? null,
|
|
96
|
+
},
|
|
97
|
+
maxTurns: parseMaxTurns(values["max-turns"]),
|
|
98
|
+
allowedTools: values["allowed-tools"]
|
|
99
|
+
? values["allowed-tools"]
|
|
100
|
+
.split(",")
|
|
101
|
+
.map((s) => s.trim())
|
|
102
|
+
.filter(Boolean)
|
|
103
|
+
: undefined,
|
|
104
|
+
};
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
function parseMaxTurns(raw) {
|
|
108
|
+
if (raw === undefined) return undefined;
|
|
109
|
+
if (raw === "0") return 0;
|
|
110
|
+
return Number.parseInt(raw, 10);
|
|
111
|
+
}
|
|
@@ -0,0 +1,94 @@
|
|
|
1
|
+
import { join } from "node:path";
|
|
2
|
+
|
|
3
|
+
const FIRST_LINE_CAP = 64 * 1024;
|
|
4
|
+
|
|
5
|
+
/**
|
|
6
|
+
* Read the first newline-terminated line of a file, bounded to the first
|
|
7
|
+
* {@link FIRST_LINE_CAP} bytes. Trace `.ndjson` files can be many MB; the
|
|
8
|
+
* Step 2.6 meta header is always small, so a bounded positional read avoids
|
|
9
|
+
* loading whole files into memory just to inspect the header. The positional
|
|
10
|
+
* `openSync`/`readSync`/`closeSync` trio is read off the injected
|
|
11
|
+
* `runtime.fsSync` surface.
|
|
12
|
+
*
|
|
13
|
+
* @param {object} fsSync - Sync filesystem surface (`runtime.fsSync`).
|
|
14
|
+
* @param {string} path
|
|
15
|
+
* @returns {string}
|
|
16
|
+
*/
|
|
17
|
+
function readFirstLine(fsSync, path) {
|
|
18
|
+
const fd = fsSync.openSync(path, "r");
|
|
19
|
+
try {
|
|
20
|
+
const buf = Buffer.alloc(FIRST_LINE_CAP);
|
|
21
|
+
const bytes = fsSync.readSync(fd, buf, 0, buf.length, 0);
|
|
22
|
+
const text = buf.toString("utf8", 0, bytes);
|
|
23
|
+
const nl = text.indexOf("\n");
|
|
24
|
+
return nl === -1 ? text : text.slice(0, nl);
|
|
25
|
+
} finally {
|
|
26
|
+
fsSync.closeSync(fd);
|
|
27
|
+
}
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
/**
|
|
31
|
+
* Scan a directory for `.ndjson` files whose meta header carries the
|
|
32
|
+
* given discussion_id. The Step 2.6 first-line guarantee makes the
|
|
33
|
+
* lookup cheap: we read only the first line per file. Files without a
|
|
34
|
+
* meta header (e.g. legacy supervise/facilitate traces) are skipped
|
|
35
|
+
* silently — not erroneous.
|
|
36
|
+
*
|
|
37
|
+
* @param {string} dir
|
|
38
|
+
* @param {string} discussionId
|
|
39
|
+
* @param {object} fsSync - Sync filesystem surface (`runtime.fsSync`).
|
|
40
|
+
* @returns {Array<{path: string, mtimeMs: number}>}
|
|
41
|
+
*/
|
|
42
|
+
export function findTracesByDiscussion(dir, discussionId, fsSync) {
|
|
43
|
+
const matches = [];
|
|
44
|
+
let entries;
|
|
45
|
+
try {
|
|
46
|
+
entries = fsSync.readdirSync(dir);
|
|
47
|
+
} catch {
|
|
48
|
+
return [];
|
|
49
|
+
}
|
|
50
|
+
for (const entry of entries) {
|
|
51
|
+
if (!entry.endsWith(".ndjson")) continue;
|
|
52
|
+
const path = join(dir, entry);
|
|
53
|
+
let firstLine;
|
|
54
|
+
try {
|
|
55
|
+
firstLine = readFirstLine(fsSync, path);
|
|
56
|
+
} catch {
|
|
57
|
+
continue;
|
|
58
|
+
}
|
|
59
|
+
let parsed;
|
|
60
|
+
try {
|
|
61
|
+
parsed = JSON.parse(firstLine);
|
|
62
|
+
} catch {
|
|
63
|
+
continue;
|
|
64
|
+
}
|
|
65
|
+
const event = parsed.event ?? parsed;
|
|
66
|
+
if (event?.type !== "meta") continue;
|
|
67
|
+
if (event.discussion_id !== discussionId) continue;
|
|
68
|
+
matches.push({ path, mtimeMs: fsSync.statSync(path).mtimeMs });
|
|
69
|
+
}
|
|
70
|
+
matches.sort((a, b) => a.mtimeMs - b.mtimeMs);
|
|
71
|
+
return matches;
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
/**
|
|
75
|
+
* `fit-trace by-discussion <discussion-id> [trace-dir]` — list trace
|
|
76
|
+
* files whose meta header carries the given discussion_id, one per
|
|
77
|
+
* line, ordered by first-event timestamp (file mtime ascending). The
|
|
78
|
+
* result is usable with `xargs cat` for a chronological merge.
|
|
79
|
+
*
|
|
80
|
+
* @param {import("@forwardimpact/libcli").InvocationContext} ctx
|
|
81
|
+
* @returns {Promise<{ok: true} | {ok: false, code: number, error: string}>}
|
|
82
|
+
*/
|
|
83
|
+
export async function runByDiscussionCommand(ctx) {
|
|
84
|
+
const runtime = ctx.deps.runtime;
|
|
85
|
+
const discussionId = ctx.args["discussion-id"];
|
|
86
|
+
if (!discussionId)
|
|
87
|
+
return { ok: false, code: 1, error: "<discussion-id> is required" };
|
|
88
|
+
const dir = ctx.args["trace-dir"] ?? ctx.options["trace-dir"] ?? "traces";
|
|
89
|
+
const matches = findTracesByDiscussion(dir, discussionId, runtime.fsSync);
|
|
90
|
+
for (const { path } of matches) {
|
|
91
|
+
runtime.proc.stdout.write(`${path}\n`);
|
|
92
|
+
}
|
|
93
|
+
return { ok: true };
|
|
94
|
+
}
|
|
@@ -0,0 +1,119 @@
|
|
|
1
|
+
import { sumTraceCost } from "../cost.js";
|
|
2
|
+
|
|
3
|
+
/**
|
|
4
|
+
* Scan an NDJSON trace and return the last orchestrator summary event,
|
|
5
|
+
* the first `meta` event's `discussion_id`, and any structured replies
|
|
6
|
+
* collected by the discusser. Skips malformed lines.
|
|
7
|
+
*
|
|
8
|
+
* The runner is verdict-agnostic — verbatim passthrough of whatever the
|
|
9
|
+
* trace carries ("success"/"failure" from supervise/facilitate; canonical
|
|
10
|
+
* "adjourned"/"recessed"/"failed" from discuss). The bridge layer maps to
|
|
11
|
+
* its channel semantics.
|
|
12
|
+
*
|
|
13
|
+
* @param {string} content - Raw NDJSON trace content.
|
|
14
|
+
* @returns {{verdict: string, summary: string, replies: object[], trigger?: object, discussionId?: string} | null}
|
|
15
|
+
*/
|
|
16
|
+
// biome-ignore lint/complexity/noExcessiveCognitiveComplexity: NDJSON scan with malformed-line tolerance + meta/summary dual extraction
|
|
17
|
+
function readTraceSummary(content) {
|
|
18
|
+
let summary = null;
|
|
19
|
+
let metaDiscussionId = null;
|
|
20
|
+
for (const line of content.split("\n")) {
|
|
21
|
+
if (!line.trim()) continue;
|
|
22
|
+
let record;
|
|
23
|
+
try {
|
|
24
|
+
record = JSON.parse(line);
|
|
25
|
+
} catch {
|
|
26
|
+
continue;
|
|
27
|
+
}
|
|
28
|
+
if (record.source !== "orchestrator") continue;
|
|
29
|
+
if (record.event?.type === "meta" && !metaDiscussionId) {
|
|
30
|
+
metaDiscussionId = record.event.discussion_id ?? null;
|
|
31
|
+
}
|
|
32
|
+
if (record.event?.type === "summary") {
|
|
33
|
+
summary = {
|
|
34
|
+
verdict: record.event.verdict ?? "failed",
|
|
35
|
+
summary: record.event.summary ?? "",
|
|
36
|
+
replies: Array.isArray(record.event.replies)
|
|
37
|
+
? record.event.replies
|
|
38
|
+
: [],
|
|
39
|
+
...(record.event.trigger && { trigger: record.event.trigger }),
|
|
40
|
+
...(record.event.discussion_id && {
|
|
41
|
+
discussionId: record.event.discussion_id,
|
|
42
|
+
}),
|
|
43
|
+
...(typeof record.event.lastActedSeq === "number" && {
|
|
44
|
+
lastActedSeq: record.event.lastActedSeq,
|
|
45
|
+
}),
|
|
46
|
+
};
|
|
47
|
+
}
|
|
48
|
+
}
|
|
49
|
+
if (summary && !summary.discussionId && metaDiscussionId) {
|
|
50
|
+
summary.discussionId = metaDiscussionId;
|
|
51
|
+
}
|
|
52
|
+
return summary;
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
/**
|
|
56
|
+
* Callback command — read an NDJSON trace, extract the terminal
|
|
57
|
+
* orchestrator summary, and POST a canonical callback body to the
|
|
58
|
+
* configured URL. Used by `kata-dispatch.yml` to deliver the lead's
|
|
59
|
+
* conclusion to the bridge that dispatched the run.
|
|
60
|
+
*
|
|
61
|
+
* Wire shape (single shape across modes):
|
|
62
|
+
*
|
|
63
|
+
* ```
|
|
64
|
+
* {
|
|
65
|
+
* correlation_id, verdict, summary, run_url,
|
|
66
|
+
* discussion_id?, replies: [], trigger?
|
|
67
|
+
* }
|
|
68
|
+
* ```
|
|
69
|
+
*
|
|
70
|
+
* @param {import("@forwardimpact/libcli").InvocationContext} ctx
|
|
71
|
+
* @returns {Promise<{ok: true} | {ok: false, code: number, error: string}>}
|
|
72
|
+
*/
|
|
73
|
+
export async function runCallbackCommand(ctx) {
|
|
74
|
+
const values = ctx.options;
|
|
75
|
+
const runtime = ctx.deps.runtime;
|
|
76
|
+
const traceFile = values["trace-file"];
|
|
77
|
+
const callbackUrl = values["callback-url"];
|
|
78
|
+
const correlationId = values["correlation-id"];
|
|
79
|
+
const runUrl = values["run-url"] ?? "";
|
|
80
|
+
const discussionIdOverride = values["discussion-id"] ?? null;
|
|
81
|
+
|
|
82
|
+
if (!traceFile)
|
|
83
|
+
return { ok: false, code: 1, error: "--trace-file is required" };
|
|
84
|
+
if (!callbackUrl)
|
|
85
|
+
return { ok: false, code: 1, error: "--callback-url is required" };
|
|
86
|
+
|
|
87
|
+
const content = runtime.fsSync.readFileSync(traceFile, "utf8");
|
|
88
|
+
const found = readTraceSummary(content) ?? {
|
|
89
|
+
verdict: "failed",
|
|
90
|
+
summary: "Run ended without producing a summary.",
|
|
91
|
+
replies: [],
|
|
92
|
+
};
|
|
93
|
+
// Total spend across every participant in the trace — the bridge surfaces
|
|
94
|
+
// it alongside the verdict so a dispatched run reports what it cost.
|
|
95
|
+
const { totalCostUsd } = sumTraceCost(content.split("\n"));
|
|
96
|
+
|
|
97
|
+
const discussionId = found.discussionId ?? discussionIdOverride ?? null;
|
|
98
|
+
const payload = {
|
|
99
|
+
correlation_id: correlationId,
|
|
100
|
+
kind: "terminal",
|
|
101
|
+
verdict: found.verdict,
|
|
102
|
+
summary: found.summary,
|
|
103
|
+
run_url: runUrl,
|
|
104
|
+
cost_usd: totalCostUsd,
|
|
105
|
+
replies: found.replies,
|
|
106
|
+
last_acted_seq: found.lastActedSeq ?? -1,
|
|
107
|
+
...(discussionId && { discussion_id: discussionId }),
|
|
108
|
+
...(found.trigger && { trigger: found.trigger }),
|
|
109
|
+
};
|
|
110
|
+
const res = await fetch(callbackUrl, {
|
|
111
|
+
method: "POST",
|
|
112
|
+
headers: { "Content-Type": "application/json" },
|
|
113
|
+
body: JSON.stringify(payload),
|
|
114
|
+
});
|
|
115
|
+
if (!res.ok) {
|
|
116
|
+
return { ok: false, code: 1, error: `Callback POST failed: ${res.status}` };
|
|
117
|
+
}
|
|
118
|
+
return { ok: true };
|
|
119
|
+
}
|