@forwardimpact/libharness 1.4.0 → 3.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +10 -10
- package/package.json +14 -12
- package/src/agent-runner.js +3 -3
- package/src/benchmark/grade.js +222 -0
- package/src/benchmark/hidden-tests.js +180 -0
- package/src/benchmark/invariants.js +9 -10
- package/src/benchmark/judge.js +9 -6
- package/src/benchmark/report.js +141 -55
- package/src/benchmark/result.js +43 -8
- package/src/benchmark/runner.js +99 -109
- package/src/benchmark/task-family.js +140 -27
- package/src/benchmark/trace-split.js +73 -0
- package/src/benchmark/workdir.js +6 -1
- package/src/claude-code-executable.js +1 -1
- package/src/commands/assert.js +72 -0
- package/src/commands/benchmark-definition.js +15 -15
- package/src/commands/benchmark-grade.js +82 -0
- package/src/commands/benchmark-report.js +1 -1
- package/src/commands/benchmark-run.js +1 -1
- package/src/commands/by-discussion.js +1 -1
- package/src/commands/facilitate.js +1 -1
- package/src/commands/output.js +1 -1
- package/src/commands/run.js +1 -1
- package/src/commands/scan-logs.js +2 -2
- package/src/commands/selfedit.js +124 -0
- package/src/commands/supervise.js +2 -2
- package/src/commands/tee.js +1 -1
- package/src/commands/trace.js +1 -1
- package/src/cost.js +1 -1
- package/src/trace-collector.js +1 -1
- package/src/trace-multi.js +1 -1
- package/src/trace-render.js +1 -1
- package/bin/fit-benchmark.js +0 -44
- package/bin/fit-harness.js +0 -412
- package/bin/fit-selfedit.js +0 -165
- package/bin/fit-trace.js +0 -510
- package/src/commands/benchmark-invariants.js +0 -73
|
@@ -1,11 +1,11 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* `
|
|
2
|
+
* `gemba-benchmark` CLI definition. Lives in `src/` so the bin stays an
|
|
3
3
|
* execute-on-import entry point — launcher packages import the bin to run
|
|
4
4
|
* it — while tests import the definition without running the CLI.
|
|
5
5
|
*/
|
|
6
6
|
|
|
7
7
|
import { runBenchmarkRunCommand } from "./benchmark-run.js";
|
|
8
|
-
import {
|
|
8
|
+
import { runBenchmarkGradeCommand } from "./benchmark-grade.js";
|
|
9
9
|
import { runBenchmarkReportCommand } from "./benchmark-report.js";
|
|
10
10
|
import {
|
|
11
11
|
BENCHMARK_AGENT_MODEL,
|
|
@@ -13,7 +13,7 @@ import {
|
|
|
13
13
|
} from "@forwardimpact/libutil/models";
|
|
14
14
|
|
|
15
15
|
export const definition = {
|
|
16
|
-
name: "
|
|
16
|
+
name: "gemba-benchmark",
|
|
17
17
|
description:
|
|
18
18
|
"Run coding-agent task families, grade hidden tests, and aggregate pass@k across runs.",
|
|
19
19
|
commands: [
|
|
@@ -95,11 +95,11 @@ export const definition = {
|
|
|
95
95
|
},
|
|
96
96
|
},
|
|
97
97
|
{
|
|
98
|
-
name: "
|
|
98
|
+
name: "grade",
|
|
99
99
|
args: [],
|
|
100
|
-
handler:
|
|
100
|
+
handler: runBenchmarkGradeCommand,
|
|
101
101
|
description:
|
|
102
|
-
"
|
|
102
|
+
"Grade a single task against a post-run workdir without invoking an agent: run the hidden test suite and the invariants script, then derive the verdict from the check rows (the exit mirrors it).",
|
|
103
103
|
options: {
|
|
104
104
|
family: {
|
|
105
105
|
type: "string",
|
|
@@ -112,7 +112,7 @@ export const definition = {
|
|
|
112
112
|
"run-dir": {
|
|
113
113
|
type: "string",
|
|
114
114
|
description:
|
|
115
|
-
"Post-run directory whose cwd/ subdir is the agent CWD;
|
|
115
|
+
"Post-run directory whose cwd/ subdir is the agent CWD; both producers run against that cwd — the path hooks receive as $AGENT_CWD",
|
|
116
116
|
},
|
|
117
117
|
output: {
|
|
118
118
|
type: "string",
|
|
@@ -154,14 +154,14 @@ export const definition = {
|
|
|
154
154
|
json: { type: "boolean", description: "Output help as JSON" },
|
|
155
155
|
},
|
|
156
156
|
examples: [
|
|
157
|
-
"
|
|
158
|
-
"
|
|
159
|
-
"
|
|
160
|
-
"
|
|
161
|
-
`
|
|
162
|
-
"
|
|
163
|
-
"
|
|
164
|
-
"
|
|
157
|
+
"gemba-benchmark run --family=./families/coding",
|
|
158
|
+
"gemba-benchmark run --family=./families/coding --task=todo-api --runs=1",
|
|
159
|
+
"gemba-benchmark run --family=./families/coding --work-tracker=filesystem",
|
|
160
|
+
"gemba-benchmark run --family=./families/coding --skills-from=. --task=todo-api",
|
|
161
|
+
`gemba-benchmark run --family=./families/coding --runs=10 --agent-model=${BENCHMARK_AGENT_MODEL}`,
|
|
162
|
+
"gemba-benchmark grade --family=./families/coding --task=todo-api --run-dir=./benchmark-runs/runs/todo-api/0",
|
|
163
|
+
"gemba-benchmark report --format=text",
|
|
164
|
+
"gemba-benchmark report --input=./runs/today --k=1,3,5 --format=text",
|
|
165
165
|
],
|
|
166
166
|
documentation: [
|
|
167
167
|
{
|
|
@@ -0,0 +1,82 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* `gemba-benchmark grade` — run both check-row producers (the hidden test
|
|
3
|
+
* suite and the invariants script) against a post-run workdir directory and
|
|
4
|
+
* grade the merged rows with the same derivation the benchmark runner uses.
|
|
5
|
+
* No agent and no judge run, so authors validate a task's grading material
|
|
6
|
+
* against fixtures without paying for agent sessions; the process exit
|
|
7
|
+
* mirrors the graded verdict.
|
|
8
|
+
*/
|
|
9
|
+
|
|
10
|
+
import { join, resolve } from "node:path";
|
|
11
|
+
|
|
12
|
+
import { validateGradeRecord } from "../benchmark/result.js";
|
|
13
|
+
import { runInvariants } from "../benchmark/invariants.js";
|
|
14
|
+
import { runHiddenTests } from "../benchmark/hidden-tests.js";
|
|
15
|
+
import { runProducersAndGrade } from "../benchmark/grade.js";
|
|
16
|
+
import { loadTaskFamily } from "../benchmark/task-family.js";
|
|
17
|
+
import { probeFreePort } from "../benchmark/workdir.js";
|
|
18
|
+
|
|
19
|
+
/**
|
|
20
|
+
* @param {import("@forwardimpact/libcli").InvocationContext} ctx
|
|
21
|
+
* @returns {Promise<{ok: true} | {ok: false, code: number, error: string}>}
|
|
22
|
+
*/
|
|
23
|
+
export async function runBenchmarkGradeCommand(ctx) {
|
|
24
|
+
const values = ctx.options;
|
|
25
|
+
const runtime = ctx.deps.runtime;
|
|
26
|
+
const familyInput = values.family;
|
|
27
|
+
if (!familyInput)
|
|
28
|
+
return { ok: false, code: 1, error: "--family is required" };
|
|
29
|
+
const taskId = values.task;
|
|
30
|
+
if (!taskId) return { ok: false, code: 1, error: "--task is required" };
|
|
31
|
+
const runDirArg = values["run-dir"];
|
|
32
|
+
if (!runDirArg) return { ok: false, code: 1, error: "--run-dir is required" };
|
|
33
|
+
|
|
34
|
+
const family = await loadTaskFamily(familyInput, runtime);
|
|
35
|
+
const task = family.tasks().find((t) => t.id === taskId);
|
|
36
|
+
if (!task)
|
|
37
|
+
return { ok: false, code: 1, error: `task not found in family: ${taskId}` };
|
|
38
|
+
|
|
39
|
+
const runDir = resolve(runDirArg);
|
|
40
|
+
const cwd = join(runDir, "cwd");
|
|
41
|
+
const port = await probeFreePort();
|
|
42
|
+
const cellCtx = { cwd, port, runDir, familyDir: family.rootPath };
|
|
43
|
+
|
|
44
|
+
const { invariants, hiddenRows, engineError, healthy, grade } =
|
|
45
|
+
await runProducersAndGrade(task, cellCtx, runtime, {
|
|
46
|
+
runInvariants,
|
|
47
|
+
runHiddenTests,
|
|
48
|
+
});
|
|
49
|
+
// Same effective-score rule as the runner, minus the judge (none runs
|
|
50
|
+
// here): an unhealthy grader or a failing gate zeroes the score, so a
|
|
51
|
+
// crashed hook can never mint marks from the rows it emitted before dying.
|
|
52
|
+
// Unlike a runner record — where `grade.score` stays the raw fraction and
|
|
53
|
+
// the zeroing lands on the top-level `score` — this record has no second
|
|
54
|
+
// field, so `grade.score` carries the effective value here.
|
|
55
|
+
if (grade.score !== undefined && !(healthy && grade.gatesPass)) {
|
|
56
|
+
grade.score = 0;
|
|
57
|
+
}
|
|
58
|
+
const record = {
|
|
59
|
+
taskId: task.id,
|
|
60
|
+
grade,
|
|
61
|
+
invariants,
|
|
62
|
+
...(task.tests && {
|
|
63
|
+
hiddenTests: {
|
|
64
|
+
details: hiddenRows,
|
|
65
|
+
...(engineError && { error: engineError.message }),
|
|
66
|
+
},
|
|
67
|
+
}),
|
|
68
|
+
// Mirrors the script for diagnosis; the graded verdict drives the exit.
|
|
69
|
+
exitCode: invariants.exitCode,
|
|
70
|
+
};
|
|
71
|
+
validateGradeRecord(record);
|
|
72
|
+
|
|
73
|
+
const line = JSON.stringify(record) + "\n";
|
|
74
|
+
if (values.output) {
|
|
75
|
+
runtime.fsSync.writeFileSync(resolve(values.output), line);
|
|
76
|
+
} else {
|
|
77
|
+
runtime.proc.stdout.write(line);
|
|
78
|
+
}
|
|
79
|
+
return grade.verdict === "pass"
|
|
80
|
+
? { ok: true }
|
|
81
|
+
: { ok: false, code: 1, error: "" };
|
|
82
|
+
}
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* `
|
|
2
|
+
* `gemba-benchmark report` — aggregate `results.jsonl` into pass@k via the
|
|
3
3
|
* OpenAI HumanEval estimator. Output is JSON by default; pass --format=text
|
|
4
4
|
* to render a markdown table. --detail=compact drops the per-task detail
|
|
5
5
|
* sections so a sharded run's per-shard summary stays short (the merge job
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* `
|
|
2
|
+
* `gemba-benchmark run` — run every task in a family for N runs, stream each
|
|
3
3
|
* ResultRecord to stdout (one JSON line per record), and append to the
|
|
4
4
|
* canonical `<output>/results.jsonl` for the report subcommand.
|
|
5
5
|
*/
|
|
@@ -72,7 +72,7 @@ export function findTracesByDiscussion(dir, discussionId, fsSync) {
|
|
|
72
72
|
}
|
|
73
73
|
|
|
74
74
|
/**
|
|
75
|
-
* `
|
|
75
|
+
* `gemba-trace by-discussion <discussion-id> [trace-dir]` — list trace
|
|
76
76
|
* files whose meta header carries the given discussion_id, one per
|
|
77
77
|
* line, ordered by first-event timestamp (file mtime ascending). The
|
|
78
78
|
* result is usable with `xargs cat` for a chronological merge.
|
|
@@ -68,7 +68,7 @@ export function parseFacilitateOptions(values, runtime) {
|
|
|
68
68
|
/**
|
|
69
69
|
* Facilitate command — run a facilitated multi-agent session.
|
|
70
70
|
*
|
|
71
|
-
* Usage:
|
|
71
|
+
* Usage: gemba-harness facilitate [options]
|
|
72
72
|
*
|
|
73
73
|
* @param {import("@forwardimpact/libcli").InvocationContext} ctx
|
|
74
74
|
* @returns {Promise<{ok: boolean, code?: number, error?: string}>}
|
package/src/commands/output.js
CHANGED
|
@@ -5,7 +5,7 @@ import { createTraceCollector } from "@forwardimpact/libharness";
|
|
|
5
5
|
* Output command — process a complete NDJSON trace from stdin and write
|
|
6
6
|
* formatted output to stdout.
|
|
7
7
|
*
|
|
8
|
-
* Usage:
|
|
8
|
+
* Usage: gemba-harness output [--format=json|text] < trace.ndjson
|
|
9
9
|
*
|
|
10
10
|
* @param {import("@forwardimpact/libcli").InvocationContext} ctx
|
|
11
11
|
* @returns {Promise<{ok: true}>}
|
package/src/commands/run.js
CHANGED
|
@@ -197,7 +197,7 @@ export async function wireRunSession({
|
|
|
197
197
|
/**
|
|
198
198
|
* Run command — execute a single agent via the Claude Agent SDK.
|
|
199
199
|
*
|
|
200
|
-
* Usage:
|
|
200
|
+
* Usage: gemba-harness run [options]
|
|
201
201
|
*
|
|
202
202
|
* @param {import("@forwardimpact/libcli").InvocationContext} ctx
|
|
203
203
|
* @returns {Promise<{ok: boolean, code?: number, error?: string}>}
|
|
@@ -1,9 +1,9 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* `
|
|
2
|
+
* `gemba-harness scan-logs` — scan a run's log archive for secret literals and
|
|
3
3
|
* fail closed.
|
|
4
4
|
*
|
|
5
5
|
* A run-lifecycle concern (not an NDJSON trace, so it lives here rather than
|
|
6
|
-
* in `
|
|
6
|
+
* in `gemba-trace`): after a CI run that handled secrets, download or accept the
|
|
7
7
|
* run's own log archive and assert none of a supplied set of literals leaked
|
|
8
8
|
* into it. Any hit exits non-zero; any download/extract failure also exits
|
|
9
9
|
* non-zero — the gate must never silently disarm.
|
|
@@ -0,0 +1,124 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Safeguard-and-write logic behind the `gemba-selfedit` bin: write content
|
|
3
|
+
* to a path that .claude/settings.json permits Edit on, while on a non-main
|
|
4
|
+
* git branch. See libraries/libharness/README.md § gemba-selfedit for the
|
|
5
|
+
* full rationale.
|
|
6
|
+
*/
|
|
7
|
+
|
|
8
|
+
import { resolve, relative, dirname } from "node:path";
|
|
9
|
+
|
|
10
|
+
import { minimatch } from "minimatch";
|
|
11
|
+
|
|
12
|
+
/** A safeguard violation — callers map it to exit code 2. */
|
|
13
|
+
export class SelfeditError extends Error {
|
|
14
|
+
/** @param {string} message failure description */
|
|
15
|
+
constructor(message) {
|
|
16
|
+
super(message);
|
|
17
|
+
this.name = "SelfeditError";
|
|
18
|
+
}
|
|
19
|
+
}
|
|
20
|
+
|
|
21
|
+
/**
|
|
22
|
+
* Check every safeguard for a selfedit write, then perform it.
|
|
23
|
+
*
|
|
24
|
+
* Safeguards (checked in order):
|
|
25
|
+
* 1. The nearest .claude/settings.json must contain an Edit(<glob>) rule in
|
|
26
|
+
* permissions.allow[] that resolves to the target path.
|
|
27
|
+
* 2. HEAD must not be detached and the current branch must not be 'main'.
|
|
28
|
+
* 3. The target's parent directory must exist.
|
|
29
|
+
*
|
|
30
|
+
* @param {string} targetArg target path as given on the command line
|
|
31
|
+
* @param {Buffer} content bytes to write
|
|
32
|
+
* @param {{ runtime: object }} deps runtime bag (fsSync, proc, subprocess,
|
|
33
|
+
* finder); targetArg resolves against `runtime.proc.cwd()`
|
|
34
|
+
* @returns {{ bytes: number, relativeTarget: string, matchedPattern: string,
|
|
35
|
+
* branch: string }} what was written and which rule allowed it
|
|
36
|
+
* @throws {SelfeditError} on any safeguard violation
|
|
37
|
+
*/
|
|
38
|
+
export function runSelfeditCommand(targetArg, content, { runtime }) {
|
|
39
|
+
const { fsSync, proc, subprocess, finder } = runtime;
|
|
40
|
+
const cwd = proc.cwd();
|
|
41
|
+
const absoluteTarget = resolve(cwd, targetArg);
|
|
42
|
+
|
|
43
|
+
// Safeguard 1: settings.json must grant Edit() on this path. Resolve the
|
|
44
|
+
// finder off the runtime bag rather than constructing a Finder here.
|
|
45
|
+
const settingsPath = finder.findUpward(
|
|
46
|
+
dirname(absoluteTarget),
|
|
47
|
+
".claude/settings.json",
|
|
48
|
+
20,
|
|
49
|
+
);
|
|
50
|
+
if (!settingsPath) {
|
|
51
|
+
throw new SelfeditError(
|
|
52
|
+
`no .claude/settings.json found walking upward from ${dirname(absoluteTarget)}`,
|
|
53
|
+
);
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
const projectRoot = dirname(dirname(settingsPath));
|
|
57
|
+
const relativeTarget = relative(projectRoot, absoluteTarget);
|
|
58
|
+
|
|
59
|
+
let settings;
|
|
60
|
+
try {
|
|
61
|
+
settings = JSON.parse(fsSync.readFileSync(settingsPath, "utf8"));
|
|
62
|
+
} catch (err) {
|
|
63
|
+
throw new SelfeditError(`failed to parse ${settingsPath}: ${err.message}`);
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
const allowRules = settings?.permissions?.allow;
|
|
67
|
+
if (!Array.isArray(allowRules)) {
|
|
68
|
+
throw new SelfeditError(`${settingsPath} has no permissions.allow[] array`);
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
const editPatterns = allowRules
|
|
72
|
+
.filter((rule) => typeof rule === "string")
|
|
73
|
+
.map((rule) => rule.match(/^Edit\((.+)\)$/)?.[1])
|
|
74
|
+
.filter(Boolean);
|
|
75
|
+
|
|
76
|
+
if (editPatterns.length === 0) {
|
|
77
|
+
throw new SelfeditError(
|
|
78
|
+
`${settingsPath} has no Edit() rules in permissions.allow[]`,
|
|
79
|
+
);
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
const matchedPattern = editPatterns.find((pattern) =>
|
|
83
|
+
minimatch(relativeTarget, pattern, { dot: true }),
|
|
84
|
+
);
|
|
85
|
+
if (!matchedPattern) {
|
|
86
|
+
throw new SelfeditError(
|
|
87
|
+
`no Edit() rule in ${relative(projectRoot, settingsPath)} matches '${relativeTarget}' ` +
|
|
88
|
+
`(tried: ${editPatterns.map((p) => `Edit(${p})`).join(", ")})`,
|
|
89
|
+
);
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
// Safeguard 2: branch must not be main and HEAD must not be detached.
|
|
93
|
+
const git = subprocess.runSync("git", ["rev-parse", "--abbrev-ref", "HEAD"], {
|
|
94
|
+
cwd,
|
|
95
|
+
});
|
|
96
|
+
if (git.exitCode !== 0) {
|
|
97
|
+
throw new SelfeditError(
|
|
98
|
+
"failed to read current git branch (not inside a git repository?)",
|
|
99
|
+
);
|
|
100
|
+
}
|
|
101
|
+
const branch = git.stdout.trim();
|
|
102
|
+
|
|
103
|
+
if (branch === "HEAD") {
|
|
104
|
+
throw new SelfeditError(
|
|
105
|
+
"HEAD is detached — refusing (check out a non-main branch first)",
|
|
106
|
+
);
|
|
107
|
+
}
|
|
108
|
+
if (branch === "main") {
|
|
109
|
+
throw new SelfeditError(
|
|
110
|
+
"refusing to write while on branch 'main' — switch to a feature branch",
|
|
111
|
+
);
|
|
112
|
+
}
|
|
113
|
+
|
|
114
|
+
const parent = dirname(absoluteTarget);
|
|
115
|
+
if (!fsSync.existsSync(parent)) {
|
|
116
|
+
throw new SelfeditError(
|
|
117
|
+
`parent directory '${relative(projectRoot, parent)}' does not exist`,
|
|
118
|
+
);
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
fsSync.writeFileSync(absoluteTarget, content);
|
|
122
|
+
|
|
123
|
+
return { bytes: content.length, relativeTarget, matchedPattern, branch };
|
|
124
|
+
}
|
|
@@ -27,7 +27,7 @@ export async function parseSuperviseOptions(values, runtime) {
|
|
|
27
27
|
const tmpRoot = runtime.proc.env.TMPDIR ?? "/tmp";
|
|
28
28
|
const agentCwd = resolve(
|
|
29
29
|
values["agent-cwd"] ||
|
|
30
|
-
(await runtime.fs.mkdtemp(join(tmpRoot, "
|
|
30
|
+
(await runtime.fs.mkdtemp(join(tmpRoot, "gemba-harness-agent-"))),
|
|
31
31
|
);
|
|
32
32
|
|
|
33
33
|
return {
|
|
@@ -62,7 +62,7 @@ export async function parseSuperviseOptions(values, runtime) {
|
|
|
62
62
|
* orchestration loop. The supervisor delegates work through Ask, sees
|
|
63
63
|
* each reply on its next turn, and ends with Conclude.
|
|
64
64
|
*
|
|
65
|
-
* Usage:
|
|
65
|
+
* Usage: gemba-harness supervise [options]
|
|
66
66
|
*
|
|
67
67
|
* @param {import("@forwardimpact/libcli").InvocationContext} ctx
|
|
68
68
|
* @returns {Promise<{ok: boolean, code?: number, error?: string}>}
|
package/src/commands/tee.js
CHANGED
|
@@ -9,7 +9,7 @@ import { createTeeWriter } from "../tee-writer.js";
|
|
|
9
9
|
* re-delimits each record with a newline so the TeeWriter's line splitter sees
|
|
10
10
|
* the same framing the raw byte stream produced.
|
|
11
11
|
*
|
|
12
|
-
* Usage:
|
|
12
|
+
* Usage: gemba-harness tee [output.ndjson] < trace.ndjson
|
|
13
13
|
*
|
|
14
14
|
* @param {import("@forwardimpact/libcli").InvocationContext} ctx
|
|
15
15
|
* @returns {Promise<{ok: boolean, code?: number, error?: string}>}
|
package/src/commands/trace.js
CHANGED
|
@@ -578,7 +578,7 @@ function parseBuckets(content) {
|
|
|
578
578
|
|
|
579
579
|
/**
|
|
580
580
|
* Compute total + per-source cost from raw file content. A structured JSON
|
|
581
|
-
* trace (from `
|
|
581
|
+
* trace (from `gemba-trace download`) carries its total in `summary.totalCostUsd`
|
|
582
582
|
* but no per-source split; raw NDJSON is summed via `sumTraceCost`.
|
|
583
583
|
* @param {string} content - Raw file content (structured JSON or NDJSON).
|
|
584
584
|
* @returns {{totalCostUsd: number, bySource: Record<string, number>}}
|
package/src/cost.js
CHANGED
|
@@ -13,7 +13,7 @@
|
|
|
13
13
|
*
|
|
14
14
|
* This mirrors `TraceCollector.handleResult`, which accumulates the same
|
|
15
15
|
* figure for its summary footer — kept as a standalone pure helper so the
|
|
16
|
-
* benchmark runner, the callback command, and `
|
|
16
|
+
* benchmark runner, the callback command, and `gemba-trace cost` share one
|
|
17
17
|
* implementation rather than each re-deriving it (and drifting).
|
|
18
18
|
*/
|
|
19
19
|
|
package/src/trace-collector.js
CHANGED
|
@@ -271,7 +271,7 @@ export class TraceCollector {
|
|
|
271
271
|
|
|
272
272
|
/**
|
|
273
273
|
* Render the accumulated turns as human-readable text — the same path the
|
|
274
|
-
* live `TeeWriter` stream uses, so `
|
|
274
|
+
* live `TeeWriter` stream uses, so `gemba-harness output --format=text` over a
|
|
275
275
|
* captured trace reproduces what the live workflow log showed.
|
|
276
276
|
*
|
|
277
277
|
* Source prefixes are emitted whenever at least one turn has a non-null
|
package/src/trace-multi.js
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* Multi-file orchestrator for cross-trace `
|
|
2
|
+
* Multi-file orchestrator for cross-trace `gemba-trace` verbs.
|
|
3
3
|
*
|
|
4
4
|
* Two functions centralise the load-tag-concat (`runOver`) and
|
|
5
5
|
* aggregate-and-sort (`aggregate`) policies so every cross-trace verb shares
|
package/src/trace-render.js
CHANGED
package/bin/fit-benchmark.js
DELETED
|
@@ -1,44 +0,0 @@
|
|
|
1
|
-
#!/usr/bin/env node
|
|
2
|
-
|
|
3
|
-
import "@forwardimpact/libpreflight/node22";
|
|
4
|
-
|
|
5
|
-
import { createCli } from "@forwardimpact/libcli";
|
|
6
|
-
import { createDefaultRuntime } from "@forwardimpact/libutil/runtime";
|
|
7
|
-
import { createLogger } from "@forwardimpact/libtelemetry";
|
|
8
|
-
|
|
9
|
-
import { definition } from "../src/commands/benchmark-definition.js";
|
|
10
|
-
|
|
11
|
-
const runtime = createDefaultRuntime();
|
|
12
|
-
const logger = createLogger("benchmark", runtime);
|
|
13
|
-
|
|
14
|
-
async function main() {
|
|
15
|
-
const cli = createCli(definition, {
|
|
16
|
-
runtime,
|
|
17
|
-
packageJsonUrl: new URL("../package.json", import.meta.url),
|
|
18
|
-
});
|
|
19
|
-
const parsed = cli.parse(runtime.proc.argv.slice(2));
|
|
20
|
-
if (!parsed) return runtime.proc.exit(0);
|
|
21
|
-
|
|
22
|
-
const { positionals } = parsed;
|
|
23
|
-
if (positionals.length === 0) {
|
|
24
|
-
cli.usageError("no command specified");
|
|
25
|
-
return runtime.proc.exit(2);
|
|
26
|
-
}
|
|
27
|
-
|
|
28
|
-
const command = positionals[0];
|
|
29
|
-
if (!definition.commands.some((c) => c.name === command)) {
|
|
30
|
-
cli.usageError(`unknown command "${command}"`);
|
|
31
|
-
return runtime.proc.exit(2);
|
|
32
|
-
}
|
|
33
|
-
|
|
34
|
-
const result = await cli.dispatch(parsed, { deps: { runtime } });
|
|
35
|
-
const envelope = result ?? { ok: true };
|
|
36
|
-
if (!envelope.ok && envelope.error) cli.error(envelope.error);
|
|
37
|
-
runtime.proc.exit(envelope.ok ? 0 : (envelope.code ?? 1));
|
|
38
|
-
}
|
|
39
|
-
|
|
40
|
-
main().catch((error) => {
|
|
41
|
-
logger.exception("main", error);
|
|
42
|
-
createCli(definition, { runtime }).error(error.message);
|
|
43
|
-
process.exit(1);
|
|
44
|
-
});
|