@forwardimpact/libharness 1.4.0 → 3.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,11 +1,11 @@
1
1
  /**
2
- * `fit-benchmark` CLI definition. Lives in `src/` so the bin stays an
2
+ * `gemba-benchmark` CLI definition. Lives in `src/` so the bin stays an
3
3
  * execute-on-import entry point — launcher packages import the bin to run
4
4
  * it — while tests import the definition without running the CLI.
5
5
  */
6
6
 
7
7
  import { runBenchmarkRunCommand } from "./benchmark-run.js";
8
- import { runBenchmarkInvariantsCommand } from "./benchmark-invariants.js";
8
+ import { runBenchmarkGradeCommand } from "./benchmark-grade.js";
9
9
  import { runBenchmarkReportCommand } from "./benchmark-report.js";
10
10
  import {
11
11
  BENCHMARK_AGENT_MODEL,
@@ -13,7 +13,7 @@ import {
13
13
  } from "@forwardimpact/libutil/models";
14
14
 
15
15
  export const definition = {
16
- name: "fit-benchmark",
16
+ name: "gemba-benchmark",
17
17
  description:
18
18
  "Run coding-agent task families, grade hidden tests, and aggregate pass@k across runs.",
19
19
  commands: [
@@ -95,11 +95,11 @@ export const definition = {
95
95
  },
96
96
  },
97
97
  {
98
- name: "invariants",
98
+ name: "grade",
99
99
  args: [],
100
- handler: runBenchmarkInvariantsCommand,
100
+ handler: runBenchmarkGradeCommand,
101
101
  description:
102
- "Check a single task's invariants against a post-run workdir without invoking an agent.",
102
+ "Grade a single task against a post-run workdir without invoking an agent: run the hidden test suite and the invariants script, then derive the verdict from the check rows (the exit mirrors it).",
103
103
  options: {
104
104
  family: {
105
105
  type: "string",
@@ -112,7 +112,7 @@ export const definition = {
112
112
  "run-dir": {
113
113
  type: "string",
114
114
  description:
115
- "Post-run directory whose cwd/ subdir is the agent CWD; invariants run against that cwd — the path hooks receive as $AGENT_CWD",
115
+ "Post-run directory whose cwd/ subdir is the agent CWD; both producers run against that cwd — the path hooks receive as $AGENT_CWD",
116
116
  },
117
117
  output: {
118
118
  type: "string",
@@ -154,14 +154,14 @@ export const definition = {
154
154
  json: { type: "boolean", description: "Output help as JSON" },
155
155
  },
156
156
  examples: [
157
- "fit-benchmark run --family=./families/coding",
158
- "fit-benchmark run --family=./families/coding --task=todo-api --runs=1",
159
- "fit-benchmark run --family=./families/coding --work-tracker=filesystem",
160
- "fit-benchmark run --family=./families/coding --skills-from=. --task=todo-api",
161
- `fit-benchmark run --family=./families/coding --runs=10 --agent-model=${BENCHMARK_AGENT_MODEL}`,
162
- "fit-benchmark invariants --family=./families/coding --task=todo-api --run-dir=./benchmark-runs/runs/todo-api/0",
163
- "fit-benchmark report --format=text",
164
- "fit-benchmark report --input=./runs/today --k=1,3,5 --format=text",
157
+ "gemba-benchmark run --family=./families/coding",
158
+ "gemba-benchmark run --family=./families/coding --task=todo-api --runs=1",
159
+ "gemba-benchmark run --family=./families/coding --work-tracker=filesystem",
160
+ "gemba-benchmark run --family=./families/coding --skills-from=. --task=todo-api",
161
+ `gemba-benchmark run --family=./families/coding --runs=10 --agent-model=${BENCHMARK_AGENT_MODEL}`,
162
+ "gemba-benchmark grade --family=./families/coding --task=todo-api --run-dir=./benchmark-runs/runs/todo-api/0",
163
+ "gemba-benchmark report --format=text",
164
+ "gemba-benchmark report --input=./runs/today --k=1,3,5 --format=text",
165
165
  ],
166
166
  documentation: [
167
167
  {
@@ -0,0 +1,82 @@
1
+ /**
2
+ * `gemba-benchmark grade` — run both check-row producers (the hidden test
3
+ * suite and the invariants script) against a post-run workdir directory and
4
+ * grade the merged rows with the same derivation the benchmark runner uses.
5
+ * No agent and no judge run, so authors validate a task's grading material
6
+ * against fixtures without paying for agent sessions; the process exit
7
+ * mirrors the graded verdict.
8
+ */
9
+
10
+ import { join, resolve } from "node:path";
11
+
12
+ import { validateGradeRecord } from "../benchmark/result.js";
13
+ import { runInvariants } from "../benchmark/invariants.js";
14
+ import { runHiddenTests } from "../benchmark/hidden-tests.js";
15
+ import { runProducersAndGrade } from "../benchmark/grade.js";
16
+ import { loadTaskFamily } from "../benchmark/task-family.js";
17
+ import { probeFreePort } from "../benchmark/workdir.js";
18
+
19
+ /**
20
+ * @param {import("@forwardimpact/libcli").InvocationContext} ctx
21
+ * @returns {Promise<{ok: true} | {ok: false, code: number, error: string}>}
22
+ */
23
+ export async function runBenchmarkGradeCommand(ctx) {
24
+ const values = ctx.options;
25
+ const runtime = ctx.deps.runtime;
26
+ const familyInput = values.family;
27
+ if (!familyInput)
28
+ return { ok: false, code: 1, error: "--family is required" };
29
+ const taskId = values.task;
30
+ if (!taskId) return { ok: false, code: 1, error: "--task is required" };
31
+ const runDirArg = values["run-dir"];
32
+ if (!runDirArg) return { ok: false, code: 1, error: "--run-dir is required" };
33
+
34
+ const family = await loadTaskFamily(familyInput, runtime);
35
+ const task = family.tasks().find((t) => t.id === taskId);
36
+ if (!task)
37
+ return { ok: false, code: 1, error: `task not found in family: ${taskId}` };
38
+
39
+ const runDir = resolve(runDirArg);
40
+ const cwd = join(runDir, "cwd");
41
+ const port = await probeFreePort();
42
+ const cellCtx = { cwd, port, runDir, familyDir: family.rootPath };
43
+
44
+ const { invariants, hiddenRows, engineError, healthy, grade } =
45
+ await runProducersAndGrade(task, cellCtx, runtime, {
46
+ runInvariants,
47
+ runHiddenTests,
48
+ });
49
+ // Same effective-score rule as the runner, minus the judge (none runs
50
+ // here): an unhealthy grader or a failing gate zeroes the score, so a
51
+ // crashed hook can never mint marks from the rows it emitted before dying.
52
+ // Unlike a runner record — where `grade.score` stays the raw fraction and
53
+ // the zeroing lands on the top-level `score` — this record has no second
54
+ // field, so `grade.score` carries the effective value here.
55
+ if (grade.score !== undefined && !(healthy && grade.gatesPass)) {
56
+ grade.score = 0;
57
+ }
58
+ const record = {
59
+ taskId: task.id,
60
+ grade,
61
+ invariants,
62
+ ...(task.tests && {
63
+ hiddenTests: {
64
+ details: hiddenRows,
65
+ ...(engineError && { error: engineError.message }),
66
+ },
67
+ }),
68
+ // Mirrors the script for diagnosis; the graded verdict drives the exit.
69
+ exitCode: invariants.exitCode,
70
+ };
71
+ validateGradeRecord(record);
72
+
73
+ const line = JSON.stringify(record) + "\n";
74
+ if (values.output) {
75
+ runtime.fsSync.writeFileSync(resolve(values.output), line);
76
+ } else {
77
+ runtime.proc.stdout.write(line);
78
+ }
79
+ return grade.verdict === "pass"
80
+ ? { ok: true }
81
+ : { ok: false, code: 1, error: "" };
82
+ }
@@ -1,5 +1,5 @@
1
1
  /**
2
- * `fit-benchmark report` — aggregate `results.jsonl` into pass@k via the
2
+ * `gemba-benchmark report` — aggregate `results.jsonl` into pass@k via the
3
3
  * OpenAI HumanEval estimator. Output is JSON by default; pass --format=text
4
4
  * to render a markdown table. --detail=compact drops the per-task detail
5
5
  * sections so a sharded run's per-shard summary stays short (the merge job
@@ -1,5 +1,5 @@
1
1
  /**
2
- * `fit-benchmark run` — run every task in a family for N runs, stream each
2
+ * `gemba-benchmark run` — run every task in a family for N runs, stream each
3
3
  * ResultRecord to stdout (one JSON line per record), and append to the
4
4
  * canonical `<output>/results.jsonl` for the report subcommand.
5
5
  */
@@ -72,7 +72,7 @@ export function findTracesByDiscussion(dir, discussionId, fsSync) {
72
72
  }
73
73
 
74
74
  /**
75
- * `fit-trace by-discussion <discussion-id> [trace-dir]` — list trace
75
+ * `gemba-trace by-discussion <discussion-id> [trace-dir]` — list trace
76
76
  * files whose meta header carries the given discussion_id, one per
77
77
  * line, ordered by first-event timestamp (file mtime ascending). The
78
78
  * result is usable with `xargs cat` for a chronological merge.
@@ -68,7 +68,7 @@ export function parseFacilitateOptions(values, runtime) {
68
68
  /**
69
69
  * Facilitate command — run a facilitated multi-agent session.
70
70
  *
71
- * Usage: fit-harness facilitate [options]
71
+ * Usage: gemba-harness facilitate [options]
72
72
  *
73
73
  * @param {import("@forwardimpact/libcli").InvocationContext} ctx
74
74
  * @returns {Promise<{ok: boolean, code?: number, error?: string}>}
@@ -5,7 +5,7 @@ import { createTraceCollector } from "@forwardimpact/libharness";
5
5
  * Output command — process a complete NDJSON trace from stdin and write
6
6
  * formatted output to stdout.
7
7
  *
8
- * Usage: fit-harness output [--format=json|text] < trace.ndjson
8
+ * Usage: gemba-harness output [--format=json|text] < trace.ndjson
9
9
  *
10
10
  * @param {import("@forwardimpact/libcli").InvocationContext} ctx
11
11
  * @returns {Promise<{ok: true}>}
@@ -197,7 +197,7 @@ export async function wireRunSession({
197
197
  /**
198
198
  * Run command — execute a single agent via the Claude Agent SDK.
199
199
  *
200
- * Usage: fit-harness run [options]
200
+ * Usage: gemba-harness run [options]
201
201
  *
202
202
  * @param {import("@forwardimpact/libcli").InvocationContext} ctx
203
203
  * @returns {Promise<{ok: boolean, code?: number, error?: string}>}
@@ -1,9 +1,9 @@
1
1
  /**
2
- * `fit-harness scan-logs` — scan a run's log archive for secret literals and
2
+ * `gemba-harness scan-logs` — scan a run's log archive for secret literals and
3
3
  * fail closed.
4
4
  *
5
5
  * A run-lifecycle concern (not an NDJSON trace, so it lives here rather than
6
- * in `fit-trace`): after a CI run that handled secrets, download or accept the
6
+ * in `gemba-trace`): after a CI run that handled secrets, download or accept the
7
7
  * run's own log archive and assert none of a supplied set of literals leaked
8
8
  * into it. Any hit exits non-zero; any download/extract failure also exits
9
9
  * non-zero — the gate must never silently disarm.
@@ -0,0 +1,124 @@
1
+ /**
2
+ * Safeguard-and-write logic behind the `gemba-selfedit` bin: write content
3
+ * to a path that .claude/settings.json permits Edit on, while on a non-main
4
+ * git branch. See libraries/libharness/README.md § gemba-selfedit for the
5
+ * full rationale.
6
+ */
7
+
8
+ import { resolve, relative, dirname } from "node:path";
9
+
10
+ import { minimatch } from "minimatch";
11
+
12
+ /** A safeguard violation — callers map it to exit code 2. */
13
+ export class SelfeditError extends Error {
14
+ /** @param {string} message failure description */
15
+ constructor(message) {
16
+ super(message);
17
+ this.name = "SelfeditError";
18
+ }
19
+ }
20
+
21
+ /**
22
+ * Check every safeguard for a selfedit write, then perform it.
23
+ *
24
+ * Safeguards (checked in order):
25
+ * 1. The nearest .claude/settings.json must contain an Edit(<glob>) rule in
26
+ * permissions.allow[] that resolves to the target path.
27
+ * 2. HEAD must not be detached and the current branch must not be 'main'.
28
+ * 3. The target's parent directory must exist.
29
+ *
30
+ * @param {string} targetArg target path as given on the command line
31
+ * @param {Buffer} content bytes to write
32
+ * @param {{ runtime: object }} deps runtime bag (fsSync, proc, subprocess,
33
+ * finder); targetArg resolves against `runtime.proc.cwd()`
34
+ * @returns {{ bytes: number, relativeTarget: string, matchedPattern: string,
35
+ * branch: string }} what was written and which rule allowed it
36
+ * @throws {SelfeditError} on any safeguard violation
37
+ */
38
+ export function runSelfeditCommand(targetArg, content, { runtime }) {
39
+ const { fsSync, proc, subprocess, finder } = runtime;
40
+ const cwd = proc.cwd();
41
+ const absoluteTarget = resolve(cwd, targetArg);
42
+
43
+ // Safeguard 1: settings.json must grant Edit() on this path. Resolve the
44
+ // finder off the runtime bag rather than constructing a Finder here.
45
+ const settingsPath = finder.findUpward(
46
+ dirname(absoluteTarget),
47
+ ".claude/settings.json",
48
+ 20,
49
+ );
50
+ if (!settingsPath) {
51
+ throw new SelfeditError(
52
+ `no .claude/settings.json found walking upward from ${dirname(absoluteTarget)}`,
53
+ );
54
+ }
55
+
56
+ const projectRoot = dirname(dirname(settingsPath));
57
+ const relativeTarget = relative(projectRoot, absoluteTarget);
58
+
59
+ let settings;
60
+ try {
61
+ settings = JSON.parse(fsSync.readFileSync(settingsPath, "utf8"));
62
+ } catch (err) {
63
+ throw new SelfeditError(`failed to parse ${settingsPath}: ${err.message}`);
64
+ }
65
+
66
+ const allowRules = settings?.permissions?.allow;
67
+ if (!Array.isArray(allowRules)) {
68
+ throw new SelfeditError(`${settingsPath} has no permissions.allow[] array`);
69
+ }
70
+
71
+ const editPatterns = allowRules
72
+ .filter((rule) => typeof rule === "string")
73
+ .map((rule) => rule.match(/^Edit\((.+)\)$/)?.[1])
74
+ .filter(Boolean);
75
+
76
+ if (editPatterns.length === 0) {
77
+ throw new SelfeditError(
78
+ `${settingsPath} has no Edit() rules in permissions.allow[]`,
79
+ );
80
+ }
81
+
82
+ const matchedPattern = editPatterns.find((pattern) =>
83
+ minimatch(relativeTarget, pattern, { dot: true }),
84
+ );
85
+ if (!matchedPattern) {
86
+ throw new SelfeditError(
87
+ `no Edit() rule in ${relative(projectRoot, settingsPath)} matches '${relativeTarget}' ` +
88
+ `(tried: ${editPatterns.map((p) => `Edit(${p})`).join(", ")})`,
89
+ );
90
+ }
91
+
92
+ // Safeguard 2: branch must not be main and HEAD must not be detached.
93
+ const git = subprocess.runSync("git", ["rev-parse", "--abbrev-ref", "HEAD"], {
94
+ cwd,
95
+ });
96
+ if (git.exitCode !== 0) {
97
+ throw new SelfeditError(
98
+ "failed to read current git branch (not inside a git repository?)",
99
+ );
100
+ }
101
+ const branch = git.stdout.trim();
102
+
103
+ if (branch === "HEAD") {
104
+ throw new SelfeditError(
105
+ "HEAD is detached — refusing (check out a non-main branch first)",
106
+ );
107
+ }
108
+ if (branch === "main") {
109
+ throw new SelfeditError(
110
+ "refusing to write while on branch 'main' — switch to a feature branch",
111
+ );
112
+ }
113
+
114
+ const parent = dirname(absoluteTarget);
115
+ if (!fsSync.existsSync(parent)) {
116
+ throw new SelfeditError(
117
+ `parent directory '${relative(projectRoot, parent)}' does not exist`,
118
+ );
119
+ }
120
+
121
+ fsSync.writeFileSync(absoluteTarget, content);
122
+
123
+ return { bytes: content.length, relativeTarget, matchedPattern, branch };
124
+ }
@@ -27,7 +27,7 @@ export async function parseSuperviseOptions(values, runtime) {
27
27
  const tmpRoot = runtime.proc.env.TMPDIR ?? "/tmp";
28
28
  const agentCwd = resolve(
29
29
  values["agent-cwd"] ||
30
- (await runtime.fs.mkdtemp(join(tmpRoot, "fit-harness-agent-"))),
30
+ (await runtime.fs.mkdtemp(join(tmpRoot, "gemba-harness-agent-"))),
31
31
  );
32
32
 
33
33
  return {
@@ -62,7 +62,7 @@ export async function parseSuperviseOptions(values, runtime) {
62
62
  * orchestration loop. The supervisor delegates work through Ask, sees
63
63
  * each reply on its next turn, and ends with Conclude.
64
64
  *
65
- * Usage: fit-harness supervise [options]
65
+ * Usage: gemba-harness supervise [options]
66
66
  *
67
67
  * @param {import("@forwardimpact/libcli").InvocationContext} ctx
68
68
  * @returns {Promise<{ok: boolean, code?: number, error?: string}>}
@@ -9,7 +9,7 @@ import { createTeeWriter } from "../tee-writer.js";
9
9
  * re-delimits each record with a newline so the TeeWriter's line splitter sees
10
10
  * the same framing the raw byte stream produced.
11
11
  *
12
- * Usage: fit-harness tee [output.ndjson] < trace.ndjson
12
+ * Usage: gemba-harness tee [output.ndjson] < trace.ndjson
13
13
  *
14
14
  * @param {import("@forwardimpact/libcli").InvocationContext} ctx
15
15
  * @returns {Promise<{ok: boolean, code?: number, error?: string}>}
@@ -578,7 +578,7 @@ function parseBuckets(content) {
578
578
 
579
579
  /**
580
580
  * Compute total + per-source cost from raw file content. A structured JSON
581
- * trace (from `fit-trace download`) carries its total in `summary.totalCostUsd`
581
+ * trace (from `gemba-trace download`) carries its total in `summary.totalCostUsd`
582
582
  * but no per-source split; raw NDJSON is summed via `sumTraceCost`.
583
583
  * @param {string} content - Raw file content (structured JSON or NDJSON).
584
584
  * @returns {{totalCostUsd: number, bySource: Record<string, number>}}
package/src/cost.js CHANGED
@@ -13,7 +13,7 @@
13
13
  *
14
14
  * This mirrors `TraceCollector.handleResult`, which accumulates the same
15
15
  * figure for its summary footer — kept as a standalone pure helper so the
16
- * benchmark runner, the callback command, and `fit-trace cost` share one
16
+ * benchmark runner, the callback command, and `gemba-trace cost` share one
17
17
  * implementation rather than each re-deriving it (and drifting).
18
18
  */
19
19
 
@@ -271,7 +271,7 @@ export class TraceCollector {
271
271
 
272
272
  /**
273
273
  * Render the accumulated turns as human-readable text — the same path the
274
- * live `TeeWriter` stream uses, so `fit-harness output --format=text` over a
274
+ * live `TeeWriter` stream uses, so `gemba-harness output --format=text` over a
275
275
  * captured trace reproduces what the live workflow log showed.
276
276
  *
277
277
  * Source prefixes are emitted whenever at least one turn has a non-null
@@ -1,5 +1,5 @@
1
1
  /**
2
- * Multi-file orchestrator for cross-trace `fit-trace` verbs.
2
+ * Multi-file orchestrator for cross-trace `gemba-trace` verbs.
3
3
  *
4
4
  * Two functions centralise the load-tag-concat (`runOver`) and
5
5
  * aggregate-and-sort (`aggregate`) policies so every cross-trace verb shares
@@ -1,5 +1,5 @@
1
1
  /**
2
- * Text renderers for `fit-trace` query output.
2
+ * Text renderers for `gemba-trace` query output.
3
3
  *
4
4
  * One named export per renderable verb. Each renderer accepts the query result
5
5
  * plus `{multi, signatures}` and returns a string. `multi` controls
@@ -1,44 +0,0 @@
1
- #!/usr/bin/env node
2
-
3
- import "@forwardimpact/libpreflight/node22";
4
-
5
- import { createCli } from "@forwardimpact/libcli";
6
- import { createDefaultRuntime } from "@forwardimpact/libutil/runtime";
7
- import { createLogger } from "@forwardimpact/libtelemetry";
8
-
9
- import { definition } from "../src/commands/benchmark-definition.js";
10
-
11
- const runtime = createDefaultRuntime();
12
- const logger = createLogger("benchmark", runtime);
13
-
14
- async function main() {
15
- const cli = createCli(definition, {
16
- runtime,
17
- packageJsonUrl: new URL("../package.json", import.meta.url),
18
- });
19
- const parsed = cli.parse(runtime.proc.argv.slice(2));
20
- if (!parsed) return runtime.proc.exit(0);
21
-
22
- const { positionals } = parsed;
23
- if (positionals.length === 0) {
24
- cli.usageError("no command specified");
25
- return runtime.proc.exit(2);
26
- }
27
-
28
- const command = positionals[0];
29
- if (!definition.commands.some((c) => c.name === command)) {
30
- cli.usageError(`unknown command "${command}"`);
31
- return runtime.proc.exit(2);
32
- }
33
-
34
- const result = await cli.dispatch(parsed, { deps: { runtime } });
35
- const envelope = result ?? { ok: true };
36
- if (!envelope.ok && envelope.error) cli.error(envelope.error);
37
- runtime.proc.exit(envelope.ok ? 0 : (envelope.code ?? 1));
38
- }
39
-
40
- main().catch((error) => {
41
- logger.exception("main", error);
42
- createCli(definition, { runtime }).error(error.message);
43
- process.exit(1);
44
- });