@forwardimpact/libharness 2.0.0 → 3.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +10 -10
- package/package.json +14 -12
- package/src/agent-runner.js +3 -3
- package/src/benchmark/task-family.js +1 -1
- package/src/benchmark/trace-split.js +1 -1
- package/src/claude-code-executable.js +1 -1
- package/src/commands/benchmark-definition.js +10 -10
- package/src/commands/benchmark-grade.js +1 -1
- package/src/commands/benchmark-report.js +1 -1
- package/src/commands/benchmark-run.js +1 -1
- package/src/commands/by-discussion.js +1 -1
- package/src/commands/facilitate.js +1 -1
- package/src/commands/output.js +1 -1
- package/src/commands/run.js +1 -1
- package/src/commands/scan-logs.js +2 -2
- package/src/commands/selfedit.js +124 -0
- package/src/commands/supervise.js +2 -2
- package/src/commands/tee.js +1 -1
- package/src/commands/trace.js +1 -1
- package/src/cost.js +1 -1
- package/src/trace-collector.js +1 -1
- package/src/trace-multi.js +1 -1
- package/src/trace-render.js +1 -1
- package/bin/fit-benchmark.js +0 -44
- package/bin/fit-harness.js +0 -412
- package/bin/fit-selfedit.js +0 -165
- package/bin/fit-trace.js +0 -520
package/README.md
CHANGED
|
@@ -17,12 +17,12 @@ traces they produce, and edits skill files under controlled conditions.
|
|
|
17
17
|
|
|
18
18
|
| CLI | Purpose |
|
|
19
19
|
| --------------- | ---------------------------------------------------------------------- |
|
|
20
|
-
| `
|
|
21
|
-
| `
|
|
22
|
-
| `
|
|
23
|
-
| `
|
|
20
|
+
| `gemba-harness` | Run agents in `run`/`supervise`/`facilitate`/`discuss` subcommands. |
|
|
21
|
+
| `gemba-trace` | Download, query, and analyze NDJSON traces produced by `gemba-harness`. |
|
|
22
|
+
| `gemba-benchmark` | Run task families for N runs each and aggregate pass@k. |
|
|
23
|
+
| `gemba-selfedit` | Write stdin to `.claude/**` paths, gated by settings.json + branch. |
|
|
24
24
|
|
|
25
|
-
`
|
|
25
|
+
`gemba-harness`'s subcommands share one orchestration loop and one async tool
|
|
26
26
|
surface, below. The `judge` role is a profile passed to `supervise`.
|
|
27
27
|
|
|
28
28
|
## Modes
|
|
@@ -147,9 +147,9 @@ Each line is `{ "source": "<participant|orchestrator>", "seq": N, "event":
|
|
|
147
147
|
{…} }`. `seq` is monotonic across the whole trace; `orchestrator` emits
|
|
148
148
|
`session_start`, `agent_start`, `protocol_violation`, `lead_turn_limit`,
|
|
149
149
|
and `summary`. `event` is the SDK event verbatim or the orchestrator
|
|
150
|
-
payload. `
|
|
150
|
+
payload. `gemba-trace` consumes this format.
|
|
151
151
|
|
|
152
|
-
Redaction is on by default for `
|
|
152
|
+
Redaction is on by default for `gemba-harness run`/`supervise`/`facilitate`
|
|
153
153
|
and composes two layers:
|
|
154
154
|
|
|
155
155
|
- **Env-var allowlist** — `ANTHROPIC_API_KEY`, `GH_TOKEN`, `GITHUB_TOKEN`
|
|
@@ -178,7 +178,7 @@ downloadable through retention.
|
|
|
178
178
|
| `trace-collector.js` / `trace-query.js` / `trace-github.js` | Trace ingestion / querying / GitHub-attachment helpers. |
|
|
179
179
|
| `redaction.js` | Env-var allowlist + credential-shape pattern redaction. |
|
|
180
180
|
|
|
181
|
-
##
|
|
181
|
+
## gemba-selfedit
|
|
182
182
|
|
|
183
183
|
A narrow, audited bypass for sessions where `Edit`/`Write` (and bash
|
|
184
184
|
writes) are blocked against paths the project's own allowlist permits.
|
|
@@ -186,7 +186,7 @@ Reads stdin, writes the target, exits 0 / 2 (safeguard violation) / 1
|
|
|
186
186
|
(I/O error).
|
|
187
187
|
|
|
188
188
|
```sh
|
|
189
|
-
echo "<content>" | bunx
|
|
189
|
+
echo "<content>" | bunx gemba-selfedit <path>
|
|
190
190
|
```
|
|
191
191
|
|
|
192
192
|
Two safeguards, checked in order:
|
|
@@ -221,7 +221,7 @@ lists the `Edit()` rules that were tried.
|
|
|
221
221
|
— end-to-end workflow from dataset generation through evaluation to trace
|
|
222
222
|
analysis, including multi-agent collaboration sessions.
|
|
223
223
|
- [Analyze Traces](https://www.forwardimpact.team/docs/libraries/prove-changes/trace-analysis/index.md)
|
|
224
|
-
— read the NDJSON traces produced by `
|
|
224
|
+
— read the NDJSON traces produced by `gemba-harness` with `gemba-trace`.
|
|
225
225
|
- [Agent Teams](https://www.forwardimpact.team/docs/products/agent-teams/index.md)
|
|
226
226
|
— author the profiles consumed by `--agent-profile`, `--lead-profile`, and
|
|
227
227
|
`--agent-profiles`.
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@forwardimpact/libharness",
|
|
3
|
-
"version": "
|
|
3
|
+
"version": "3.0.0",
|
|
4
4
|
"description": "Autonomous agent team harness — coordinate a lead and participant agents in one async session, with eval, benchmark, and trace tooling to prove the changes worked.",
|
|
5
5
|
"keywords": [
|
|
6
6
|
"orchestration",
|
|
@@ -40,20 +40,22 @@
|
|
|
40
40
|
"main": "./src/index.js",
|
|
41
41
|
"exports": {
|
|
42
42
|
".": "./src/index.js",
|
|
43
|
-
"./
|
|
44
|
-
"./
|
|
45
|
-
"./
|
|
46
|
-
"./
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
"
|
|
50
|
-
"
|
|
51
|
-
"
|
|
52
|
-
"
|
|
43
|
+
"./commands/output.js": "./src/commands/output.js",
|
|
44
|
+
"./commands/tee.js": "./src/commands/tee.js",
|
|
45
|
+
"./commands/run.js": "./src/commands/run.js",
|
|
46
|
+
"./commands/supervise.js": "./src/commands/supervise.js",
|
|
47
|
+
"./commands/facilitate.js": "./src/commands/facilitate.js",
|
|
48
|
+
"./commands/discuss.js": "./src/commands/discuss.js",
|
|
49
|
+
"./commands/callback.js": "./src/commands/callback.js",
|
|
50
|
+
"./commands/scan-logs.js": "./src/commands/scan-logs.js",
|
|
51
|
+
"./commands/trace.js": "./src/commands/trace.js",
|
|
52
|
+
"./commands/assert.js": "./src/commands/assert.js",
|
|
53
|
+
"./commands/by-discussion.js": "./src/commands/by-discussion.js",
|
|
54
|
+
"./commands/benchmark-definition.js": "./src/commands/benchmark-definition.js",
|
|
55
|
+
"./commands/selfedit.js": "./src/commands/selfedit.js"
|
|
53
56
|
},
|
|
54
57
|
"files": [
|
|
55
58
|
"src/**/*.js",
|
|
56
|
-
"bin/**/*.js",
|
|
57
59
|
"README.md"
|
|
58
60
|
],
|
|
59
61
|
"scripts": {
|
package/src/agent-runner.js
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* AgentRunner — runs a single Claude Agent SDK session and emits raw
|
|
3
|
-
* NDJSON events to an output stream. Building block for `
|
|
4
|
-
* `
|
|
3
|
+
* NDJSON events to an output stream. Building block for `gemba-harness run`,
|
|
4
|
+
* `gemba-harness supervise`, `gemba-harness facilitate`, and `gemba-harness discuss`.
|
|
5
5
|
*
|
|
6
6
|
* Follows OO+DI: constructor injection, factory function, tests bypass factory.
|
|
7
7
|
*/
|
|
@@ -37,7 +37,7 @@ function modelDidWork(result) {
|
|
|
37
37
|
return tokens > 0 || (cost ?? 0) > 0;
|
|
38
38
|
}
|
|
39
39
|
|
|
40
|
-
//
|
|
40
|
+
// gemba-harness and kata-action run headless in CI/CD with no human to answer
|
|
41
41
|
// permission prompts. The SDK is always launched in bypass mode — not
|
|
42
42
|
// overridable — so a future caller can't accidentally reduce permissions.
|
|
43
43
|
const PERMISSION_MODE = "bypassPermissions";
|
|
@@ -62,7 +62,7 @@ export async function loadTaskFamily(rootPathOrGitUrl, runtime) {
|
|
|
62
62
|
let familyRevision;
|
|
63
63
|
if (isGit) {
|
|
64
64
|
const dir = await runtime.fs.mkdtemp(
|
|
65
|
-
join(tmpdir(runtime), "
|
|
65
|
+
join(tmpdir(runtime), "gemba-benchmark-family-"),
|
|
66
66
|
);
|
|
67
67
|
await gitClone(runtime, rootPathOrGitUrl, dir);
|
|
68
68
|
rootPath = dir;
|
|
@@ -13,7 +13,7 @@ import { createInterface } from "node:readline";
|
|
|
13
13
|
*
|
|
14
14
|
* Cost is deliberately not summed here — the caller derives it from the same
|
|
15
15
|
* combined trace via `sumTraceCost`, so there is one cost path across the
|
|
16
|
-
* benchmark, callback, and `
|
|
16
|
+
* benchmark, callback, and `gemba-trace cost` consumers.
|
|
17
17
|
* @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
|
|
18
18
|
* @param {string} combinedPath
|
|
19
19
|
* @param {string} agentPath
|
|
@@ -9,7 +9,7 @@
|
|
|
9
9
|
* <platform> not found".
|
|
10
10
|
*
|
|
11
11
|
* In a compiled binary we point the SDK at the standalone `claude` on PATH,
|
|
12
|
-
* installed beside
|
|
12
|
+
* installed beside gemba-harness by the bootstrap action's `fit-install.sh`.
|
|
13
13
|
* Running from source keeps `node_modules`, where the SDK resolves its own
|
|
14
14
|
* version-matched binary, so there we return undefined and defer to the SDK.
|
|
15
15
|
*/
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* `
|
|
2
|
+
* `gemba-benchmark` CLI definition. Lives in `src/` so the bin stays an
|
|
3
3
|
* execute-on-import entry point — launcher packages import the bin to run
|
|
4
4
|
* it — while tests import the definition without running the CLI.
|
|
5
5
|
*/
|
|
@@ -13,7 +13,7 @@ import {
|
|
|
13
13
|
} from "@forwardimpact/libutil/models";
|
|
14
14
|
|
|
15
15
|
export const definition = {
|
|
16
|
-
name: "
|
|
16
|
+
name: "gemba-benchmark",
|
|
17
17
|
description:
|
|
18
18
|
"Run coding-agent task families, grade hidden tests, and aggregate pass@k across runs.",
|
|
19
19
|
commands: [
|
|
@@ -154,14 +154,14 @@ export const definition = {
|
|
|
154
154
|
json: { type: "boolean", description: "Output help as JSON" },
|
|
155
155
|
},
|
|
156
156
|
examples: [
|
|
157
|
-
"
|
|
158
|
-
"
|
|
159
|
-
"
|
|
160
|
-
"
|
|
161
|
-
`
|
|
162
|
-
"
|
|
163
|
-
"
|
|
164
|
-
"
|
|
157
|
+
"gemba-benchmark run --family=./families/coding",
|
|
158
|
+
"gemba-benchmark run --family=./families/coding --task=todo-api --runs=1",
|
|
159
|
+
"gemba-benchmark run --family=./families/coding --work-tracker=filesystem",
|
|
160
|
+
"gemba-benchmark run --family=./families/coding --skills-from=. --task=todo-api",
|
|
161
|
+
`gemba-benchmark run --family=./families/coding --runs=10 --agent-model=${BENCHMARK_AGENT_MODEL}`,
|
|
162
|
+
"gemba-benchmark grade --family=./families/coding --task=todo-api --run-dir=./benchmark-runs/runs/todo-api/0",
|
|
163
|
+
"gemba-benchmark report --format=text",
|
|
164
|
+
"gemba-benchmark report --input=./runs/today --k=1,3,5 --format=text",
|
|
165
165
|
],
|
|
166
166
|
documentation: [
|
|
167
167
|
{
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* `
|
|
2
|
+
* `gemba-benchmark grade` — run both check-row producers (the hidden test
|
|
3
3
|
* suite and the invariants script) against a post-run workdir directory and
|
|
4
4
|
* grade the merged rows with the same derivation the benchmark runner uses.
|
|
5
5
|
* No agent and no judge run, so authors validate a task's grading material
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* `
|
|
2
|
+
* `gemba-benchmark report` — aggregate `results.jsonl` into pass@k via the
|
|
3
3
|
* OpenAI HumanEval estimator. Output is JSON by default; pass --format=text
|
|
4
4
|
* to render a markdown table. --detail=compact drops the per-task detail
|
|
5
5
|
* sections so a sharded run's per-shard summary stays short (the merge job
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* `
|
|
2
|
+
* `gemba-benchmark run` — run every task in a family for N runs, stream each
|
|
3
3
|
* ResultRecord to stdout (one JSON line per record), and append to the
|
|
4
4
|
* canonical `<output>/results.jsonl` for the report subcommand.
|
|
5
5
|
*/
|
|
@@ -72,7 +72,7 @@ export function findTracesByDiscussion(dir, discussionId, fsSync) {
|
|
|
72
72
|
}
|
|
73
73
|
|
|
74
74
|
/**
|
|
75
|
-
* `
|
|
75
|
+
* `gemba-trace by-discussion <discussion-id> [trace-dir]` — list trace
|
|
76
76
|
* files whose meta header carries the given discussion_id, one per
|
|
77
77
|
* line, ordered by first-event timestamp (file mtime ascending). The
|
|
78
78
|
* result is usable with `xargs cat` for a chronological merge.
|
|
@@ -68,7 +68,7 @@ export function parseFacilitateOptions(values, runtime) {
|
|
|
68
68
|
/**
|
|
69
69
|
* Facilitate command — run a facilitated multi-agent session.
|
|
70
70
|
*
|
|
71
|
-
* Usage:
|
|
71
|
+
* Usage: gemba-harness facilitate [options]
|
|
72
72
|
*
|
|
73
73
|
* @param {import("@forwardimpact/libcli").InvocationContext} ctx
|
|
74
74
|
* @returns {Promise<{ok: boolean, code?: number, error?: string}>}
|
package/src/commands/output.js
CHANGED
|
@@ -5,7 +5,7 @@ import { createTraceCollector } from "@forwardimpact/libharness";
|
|
|
5
5
|
* Output command — process a complete NDJSON trace from stdin and write
|
|
6
6
|
* formatted output to stdout.
|
|
7
7
|
*
|
|
8
|
-
* Usage:
|
|
8
|
+
* Usage: gemba-harness output [--format=json|text] < trace.ndjson
|
|
9
9
|
*
|
|
10
10
|
* @param {import("@forwardimpact/libcli").InvocationContext} ctx
|
|
11
11
|
* @returns {Promise<{ok: true}>}
|
package/src/commands/run.js
CHANGED
|
@@ -197,7 +197,7 @@ export async function wireRunSession({
|
|
|
197
197
|
/**
|
|
198
198
|
* Run command — execute a single agent via the Claude Agent SDK.
|
|
199
199
|
*
|
|
200
|
-
* Usage:
|
|
200
|
+
* Usage: gemba-harness run [options]
|
|
201
201
|
*
|
|
202
202
|
* @param {import("@forwardimpact/libcli").InvocationContext} ctx
|
|
203
203
|
* @returns {Promise<{ok: boolean, code?: number, error?: string}>}
|
|
@@ -1,9 +1,9 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* `
|
|
2
|
+
* `gemba-harness scan-logs` — scan a run's log archive for secret literals and
|
|
3
3
|
* fail closed.
|
|
4
4
|
*
|
|
5
5
|
* A run-lifecycle concern (not an NDJSON trace, so it lives here rather than
|
|
6
|
-
* in `
|
|
6
|
+
* in `gemba-trace`): after a CI run that handled secrets, download or accept the
|
|
7
7
|
* run's own log archive and assert none of a supplied set of literals leaked
|
|
8
8
|
* into it. Any hit exits non-zero; any download/extract failure also exits
|
|
9
9
|
* non-zero — the gate must never silently disarm.
|
|
@@ -0,0 +1,124 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Safeguard-and-write logic behind the `gemba-selfedit` bin: write content
|
|
3
|
+
* to a path that .claude/settings.json permits Edit on, while on a non-main
|
|
4
|
+
* git branch. See libraries/libharness/README.md § gemba-selfedit for the
|
|
5
|
+
* full rationale.
|
|
6
|
+
*/
|
|
7
|
+
|
|
8
|
+
import { resolve, relative, dirname } from "node:path";
|
|
9
|
+
|
|
10
|
+
import { minimatch } from "minimatch";
|
|
11
|
+
|
|
12
|
+
/** A safeguard violation — callers map it to exit code 2. */
|
|
13
|
+
export class SelfeditError extends Error {
|
|
14
|
+
/** @param {string} message failure description */
|
|
15
|
+
constructor(message) {
|
|
16
|
+
super(message);
|
|
17
|
+
this.name = "SelfeditError";
|
|
18
|
+
}
|
|
19
|
+
}
|
|
20
|
+
|
|
21
|
+
/**
|
|
22
|
+
* Check every safeguard for a selfedit write, then perform it.
|
|
23
|
+
*
|
|
24
|
+
* Safeguards (checked in order):
|
|
25
|
+
* 1. The nearest .claude/settings.json must contain an Edit(<glob>) rule in
|
|
26
|
+
* permissions.allow[] that resolves to the target path.
|
|
27
|
+
* 2. HEAD must not be detached and the current branch must not be 'main'.
|
|
28
|
+
* 3. The target's parent directory must exist.
|
|
29
|
+
*
|
|
30
|
+
* @param {string} targetArg target path as given on the command line
|
|
31
|
+
* @param {Buffer} content bytes to write
|
|
32
|
+
* @param {{ runtime: object }} deps runtime bag (fsSync, proc, subprocess,
|
|
33
|
+
* finder); targetArg resolves against `runtime.proc.cwd()`
|
|
34
|
+
* @returns {{ bytes: number, relativeTarget: string, matchedPattern: string,
|
|
35
|
+
* branch: string }} what was written and which rule allowed it
|
|
36
|
+
* @throws {SelfeditError} on any safeguard violation
|
|
37
|
+
*/
|
|
38
|
+
export function runSelfeditCommand(targetArg, content, { runtime }) {
|
|
39
|
+
const { fsSync, proc, subprocess, finder } = runtime;
|
|
40
|
+
const cwd = proc.cwd();
|
|
41
|
+
const absoluteTarget = resolve(cwd, targetArg);
|
|
42
|
+
|
|
43
|
+
// Safeguard 1: settings.json must grant Edit() on this path. Resolve the
|
|
44
|
+
// finder off the runtime bag rather than constructing a Finder here.
|
|
45
|
+
const settingsPath = finder.findUpward(
|
|
46
|
+
dirname(absoluteTarget),
|
|
47
|
+
".claude/settings.json",
|
|
48
|
+
20,
|
|
49
|
+
);
|
|
50
|
+
if (!settingsPath) {
|
|
51
|
+
throw new SelfeditError(
|
|
52
|
+
`no .claude/settings.json found walking upward from ${dirname(absoluteTarget)}`,
|
|
53
|
+
);
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
const projectRoot = dirname(dirname(settingsPath));
|
|
57
|
+
const relativeTarget = relative(projectRoot, absoluteTarget);
|
|
58
|
+
|
|
59
|
+
let settings;
|
|
60
|
+
try {
|
|
61
|
+
settings = JSON.parse(fsSync.readFileSync(settingsPath, "utf8"));
|
|
62
|
+
} catch (err) {
|
|
63
|
+
throw new SelfeditError(`failed to parse ${settingsPath}: ${err.message}`);
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
const allowRules = settings?.permissions?.allow;
|
|
67
|
+
if (!Array.isArray(allowRules)) {
|
|
68
|
+
throw new SelfeditError(`${settingsPath} has no permissions.allow[] array`);
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
const editPatterns = allowRules
|
|
72
|
+
.filter((rule) => typeof rule === "string")
|
|
73
|
+
.map((rule) => rule.match(/^Edit\((.+)\)$/)?.[1])
|
|
74
|
+
.filter(Boolean);
|
|
75
|
+
|
|
76
|
+
if (editPatterns.length === 0) {
|
|
77
|
+
throw new SelfeditError(
|
|
78
|
+
`${settingsPath} has no Edit() rules in permissions.allow[]`,
|
|
79
|
+
);
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
const matchedPattern = editPatterns.find((pattern) =>
|
|
83
|
+
minimatch(relativeTarget, pattern, { dot: true }),
|
|
84
|
+
);
|
|
85
|
+
if (!matchedPattern) {
|
|
86
|
+
throw new SelfeditError(
|
|
87
|
+
`no Edit() rule in ${relative(projectRoot, settingsPath)} matches '${relativeTarget}' ` +
|
|
88
|
+
`(tried: ${editPatterns.map((p) => `Edit(${p})`).join(", ")})`,
|
|
89
|
+
);
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
// Safeguard 2: branch must not be main and HEAD must not be detached.
|
|
93
|
+
const git = subprocess.runSync("git", ["rev-parse", "--abbrev-ref", "HEAD"], {
|
|
94
|
+
cwd,
|
|
95
|
+
});
|
|
96
|
+
if (git.exitCode !== 0) {
|
|
97
|
+
throw new SelfeditError(
|
|
98
|
+
"failed to read current git branch (not inside a git repository?)",
|
|
99
|
+
);
|
|
100
|
+
}
|
|
101
|
+
const branch = git.stdout.trim();
|
|
102
|
+
|
|
103
|
+
if (branch === "HEAD") {
|
|
104
|
+
throw new SelfeditError(
|
|
105
|
+
"HEAD is detached — refusing (check out a non-main branch first)",
|
|
106
|
+
);
|
|
107
|
+
}
|
|
108
|
+
if (branch === "main") {
|
|
109
|
+
throw new SelfeditError(
|
|
110
|
+
"refusing to write while on branch 'main' — switch to a feature branch",
|
|
111
|
+
);
|
|
112
|
+
}
|
|
113
|
+
|
|
114
|
+
const parent = dirname(absoluteTarget);
|
|
115
|
+
if (!fsSync.existsSync(parent)) {
|
|
116
|
+
throw new SelfeditError(
|
|
117
|
+
`parent directory '${relative(projectRoot, parent)}' does not exist`,
|
|
118
|
+
);
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
fsSync.writeFileSync(absoluteTarget, content);
|
|
122
|
+
|
|
123
|
+
return { bytes: content.length, relativeTarget, matchedPattern, branch };
|
|
124
|
+
}
|
|
@@ -27,7 +27,7 @@ export async function parseSuperviseOptions(values, runtime) {
|
|
|
27
27
|
const tmpRoot = runtime.proc.env.TMPDIR ?? "/tmp";
|
|
28
28
|
const agentCwd = resolve(
|
|
29
29
|
values["agent-cwd"] ||
|
|
30
|
-
(await runtime.fs.mkdtemp(join(tmpRoot, "
|
|
30
|
+
(await runtime.fs.mkdtemp(join(tmpRoot, "gemba-harness-agent-"))),
|
|
31
31
|
);
|
|
32
32
|
|
|
33
33
|
return {
|
|
@@ -62,7 +62,7 @@ export async function parseSuperviseOptions(values, runtime) {
|
|
|
62
62
|
* orchestration loop. The supervisor delegates work through Ask, sees
|
|
63
63
|
* each reply on its next turn, and ends with Conclude.
|
|
64
64
|
*
|
|
65
|
-
* Usage:
|
|
65
|
+
* Usage: gemba-harness supervise [options]
|
|
66
66
|
*
|
|
67
67
|
* @param {import("@forwardimpact/libcli").InvocationContext} ctx
|
|
68
68
|
* @returns {Promise<{ok: boolean, code?: number, error?: string}>}
|
package/src/commands/tee.js
CHANGED
|
@@ -9,7 +9,7 @@ import { createTeeWriter } from "../tee-writer.js";
|
|
|
9
9
|
* re-delimits each record with a newline so the TeeWriter's line splitter sees
|
|
10
10
|
* the same framing the raw byte stream produced.
|
|
11
11
|
*
|
|
12
|
-
* Usage:
|
|
12
|
+
* Usage: gemba-harness tee [output.ndjson] < trace.ndjson
|
|
13
13
|
*
|
|
14
14
|
* @param {import("@forwardimpact/libcli").InvocationContext} ctx
|
|
15
15
|
* @returns {Promise<{ok: boolean, code?: number, error?: string}>}
|
package/src/commands/trace.js
CHANGED
|
@@ -578,7 +578,7 @@ function parseBuckets(content) {
|
|
|
578
578
|
|
|
579
579
|
/**
|
|
580
580
|
* Compute total + per-source cost from raw file content. A structured JSON
|
|
581
|
-
* trace (from `
|
|
581
|
+
* trace (from `gemba-trace download`) carries its total in `summary.totalCostUsd`
|
|
582
582
|
* but no per-source split; raw NDJSON is summed via `sumTraceCost`.
|
|
583
583
|
* @param {string} content - Raw file content (structured JSON or NDJSON).
|
|
584
584
|
* @returns {{totalCostUsd: number, bySource: Record<string, number>}}
|
package/src/cost.js
CHANGED
|
@@ -13,7 +13,7 @@
|
|
|
13
13
|
*
|
|
14
14
|
* This mirrors `TraceCollector.handleResult`, which accumulates the same
|
|
15
15
|
* figure for its summary footer — kept as a standalone pure helper so the
|
|
16
|
-
* benchmark runner, the callback command, and `
|
|
16
|
+
* benchmark runner, the callback command, and `gemba-trace cost` share one
|
|
17
17
|
* implementation rather than each re-deriving it (and drifting).
|
|
18
18
|
*/
|
|
19
19
|
|
package/src/trace-collector.js
CHANGED
|
@@ -271,7 +271,7 @@ export class TraceCollector {
|
|
|
271
271
|
|
|
272
272
|
/**
|
|
273
273
|
* Render the accumulated turns as human-readable text — the same path the
|
|
274
|
-
* live `TeeWriter` stream uses, so `
|
|
274
|
+
* live `TeeWriter` stream uses, so `gemba-harness output --format=text` over a
|
|
275
275
|
* captured trace reproduces what the live workflow log showed.
|
|
276
276
|
*
|
|
277
277
|
* Source prefixes are emitted whenever at least one turn has a non-null
|
package/src/trace-multi.js
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* Multi-file orchestrator for cross-trace `
|
|
2
|
+
* Multi-file orchestrator for cross-trace `gemba-trace` verbs.
|
|
3
3
|
*
|
|
4
4
|
* Two functions centralise the load-tag-concat (`runOver`) and
|
|
5
5
|
* aggregate-and-sort (`aggregate`) policies so every cross-trace verb shares
|
package/src/trace-render.js
CHANGED
package/bin/fit-benchmark.js
DELETED
|
@@ -1,44 +0,0 @@
|
|
|
1
|
-
#!/usr/bin/env node
|
|
2
|
-
|
|
3
|
-
import "@forwardimpact/libpreflight/node22";
|
|
4
|
-
|
|
5
|
-
import { createCli } from "@forwardimpact/libcli";
|
|
6
|
-
import { createDefaultRuntime } from "@forwardimpact/libutil/runtime";
|
|
7
|
-
import { createLogger } from "@forwardimpact/libtelemetry";
|
|
8
|
-
|
|
9
|
-
import { definition } from "../src/commands/benchmark-definition.js";
|
|
10
|
-
|
|
11
|
-
const runtime = createDefaultRuntime();
|
|
12
|
-
const logger = createLogger("benchmark", runtime);
|
|
13
|
-
|
|
14
|
-
async function main() {
|
|
15
|
-
const cli = createCli(definition, {
|
|
16
|
-
runtime,
|
|
17
|
-
packageJsonUrl: new URL("../package.json", import.meta.url),
|
|
18
|
-
});
|
|
19
|
-
const parsed = cli.parse(runtime.proc.argv.slice(2));
|
|
20
|
-
if (!parsed) return runtime.proc.exit(0);
|
|
21
|
-
|
|
22
|
-
const { positionals } = parsed;
|
|
23
|
-
if (positionals.length === 0) {
|
|
24
|
-
cli.usageError("no command specified");
|
|
25
|
-
return runtime.proc.exit(2);
|
|
26
|
-
}
|
|
27
|
-
|
|
28
|
-
const command = positionals[0];
|
|
29
|
-
if (!definition.commands.some((c) => c.name === command)) {
|
|
30
|
-
cli.usageError(`unknown command "${command}"`);
|
|
31
|
-
return runtime.proc.exit(2);
|
|
32
|
-
}
|
|
33
|
-
|
|
34
|
-
const result = await cli.dispatch(parsed, { deps: { runtime } });
|
|
35
|
-
const envelope = result ?? { ok: true };
|
|
36
|
-
if (!envelope.ok && envelope.error) cli.error(envelope.error);
|
|
37
|
-
runtime.proc.exit(envelope.ok ? 0 : (envelope.code ?? 1));
|
|
38
|
-
}
|
|
39
|
-
|
|
40
|
-
main().catch((error) => {
|
|
41
|
-
logger.exception("main", error);
|
|
42
|
-
createCli(definition, { runtime }).error(error.message);
|
|
43
|
-
process.exit(1);
|
|
44
|
-
});
|