@forwardimpact/libharness 1.4.0 → 3.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +10 -10
- package/package.json +14 -12
- package/src/agent-runner.js +3 -3
- package/src/benchmark/grade.js +222 -0
- package/src/benchmark/hidden-tests.js +180 -0
- package/src/benchmark/invariants.js +9 -10
- package/src/benchmark/judge.js +9 -6
- package/src/benchmark/report.js +141 -55
- package/src/benchmark/result.js +43 -8
- package/src/benchmark/runner.js +99 -109
- package/src/benchmark/task-family.js +140 -27
- package/src/benchmark/trace-split.js +73 -0
- package/src/benchmark/workdir.js +6 -1
- package/src/claude-code-executable.js +1 -1
- package/src/commands/assert.js +72 -0
- package/src/commands/benchmark-definition.js +15 -15
- package/src/commands/benchmark-grade.js +82 -0
- package/src/commands/benchmark-report.js +1 -1
- package/src/commands/benchmark-run.js +1 -1
- package/src/commands/by-discussion.js +1 -1
- package/src/commands/facilitate.js +1 -1
- package/src/commands/output.js +1 -1
- package/src/commands/run.js +1 -1
- package/src/commands/scan-logs.js +2 -2
- package/src/commands/selfedit.js +124 -0
- package/src/commands/supervise.js +2 -2
- package/src/commands/tee.js +1 -1
- package/src/commands/trace.js +1 -1
- package/src/cost.js +1 -1
- package/src/trace-collector.js +1 -1
- package/src/trace-multi.js +1 -1
- package/src/trace-render.js +1 -1
- package/bin/fit-benchmark.js +0 -44
- package/bin/fit-harness.js +0 -412
- package/bin/fit-selfedit.js +0 -165
- package/bin/fit-trace.js +0 -510
- package/src/commands/benchmark-invariants.js +0 -73
package/README.md
CHANGED
|
@@ -17,12 +17,12 @@ traces they produce, and edits skill files under controlled conditions.
|
|
|
17
17
|
|
|
18
18
|
| CLI | Purpose |
|
|
19
19
|
| --------------- | ---------------------------------------------------------------------- |
|
|
20
|
-
| `
|
|
21
|
-
| `
|
|
22
|
-
| `
|
|
23
|
-
| `
|
|
20
|
+
| `gemba-harness` | Run agents in `run`/`supervise`/`facilitate`/`discuss` subcommands. |
|
|
21
|
+
| `gemba-trace` | Download, query, and analyze NDJSON traces produced by `gemba-harness`. |
|
|
22
|
+
| `gemba-benchmark` | Run task families for N runs each and aggregate pass@k. |
|
|
23
|
+
| `gemba-selfedit` | Write stdin to `.claude/**` paths, gated by settings.json + branch. |
|
|
24
24
|
|
|
25
|
-
`
|
|
25
|
+
`gemba-harness`'s subcommands share one orchestration loop and one async tool
|
|
26
26
|
surface, below. The `judge` role is a profile passed to `supervise`.
|
|
27
27
|
|
|
28
28
|
## Modes
|
|
@@ -147,9 +147,9 @@ Each line is `{ "source": "<participant|orchestrator>", "seq": N, "event":
|
|
|
147
147
|
{…} }`. `seq` is monotonic across the whole trace; `orchestrator` emits
|
|
148
148
|
`session_start`, `agent_start`, `protocol_violation`, `lead_turn_limit`,
|
|
149
149
|
and `summary`. `event` is the SDK event verbatim or the orchestrator
|
|
150
|
-
payload. `
|
|
150
|
+
payload. `gemba-trace` consumes this format.
|
|
151
151
|
|
|
152
|
-
Redaction is on by default for `
|
|
152
|
+
Redaction is on by default for `gemba-harness run`/`supervise`/`facilitate`
|
|
153
153
|
and composes two layers:
|
|
154
154
|
|
|
155
155
|
- **Env-var allowlist** — `ANTHROPIC_API_KEY`, `GH_TOKEN`, `GITHUB_TOKEN`
|
|
@@ -178,7 +178,7 @@ downloadable through retention.
|
|
|
178
178
|
| `trace-collector.js` / `trace-query.js` / `trace-github.js` | Trace ingestion / querying / GitHub-attachment helpers. |
|
|
179
179
|
| `redaction.js` | Env-var allowlist + credential-shape pattern redaction. |
|
|
180
180
|
|
|
181
|
-
##
|
|
181
|
+
## gemba-selfedit
|
|
182
182
|
|
|
183
183
|
A narrow, audited bypass for sessions where `Edit`/`Write` (and bash
|
|
184
184
|
writes) are blocked against paths the project's own allowlist permits.
|
|
@@ -186,7 +186,7 @@ Reads stdin, writes the target, exits 0 / 2 (safeguard violation) / 1
|
|
|
186
186
|
(I/O error).
|
|
187
187
|
|
|
188
188
|
```sh
|
|
189
|
-
echo "<content>" | bunx
|
|
189
|
+
echo "<content>" | bunx gemba-selfedit <path>
|
|
190
190
|
```
|
|
191
191
|
|
|
192
192
|
Two safeguards, checked in order:
|
|
@@ -221,7 +221,7 @@ lists the `Edit()` rules that were tried.
|
|
|
221
221
|
— end-to-end workflow from dataset generation through evaluation to trace
|
|
222
222
|
analysis, including multi-agent collaboration sessions.
|
|
223
223
|
- [Analyze Traces](https://www.forwardimpact.team/docs/libraries/prove-changes/trace-analysis/index.md)
|
|
224
|
-
— read the NDJSON traces produced by `
|
|
224
|
+
— read the NDJSON traces produced by `gemba-harness` with `gemba-trace`.
|
|
225
225
|
- [Agent Teams](https://www.forwardimpact.team/docs/products/agent-teams/index.md)
|
|
226
226
|
— author the profiles consumed by `--agent-profile`, `--lead-profile`, and
|
|
227
227
|
`--agent-profiles`.
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@forwardimpact/libharness",
|
|
3
|
-
"version": "
|
|
3
|
+
"version": "3.0.0",
|
|
4
4
|
"description": "Autonomous agent team harness — coordinate a lead and participant agents in one async session, with eval, benchmark, and trace tooling to prove the changes worked.",
|
|
5
5
|
"keywords": [
|
|
6
6
|
"orchestration",
|
|
@@ -40,20 +40,22 @@
|
|
|
40
40
|
"main": "./src/index.js",
|
|
41
41
|
"exports": {
|
|
42
42
|
".": "./src/index.js",
|
|
43
|
-
"./
|
|
44
|
-
"./
|
|
45
|
-
"./
|
|
46
|
-
"./
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
"
|
|
50
|
-
"
|
|
51
|
-
"
|
|
52
|
-
"
|
|
43
|
+
"./commands/output.js": "./src/commands/output.js",
|
|
44
|
+
"./commands/tee.js": "./src/commands/tee.js",
|
|
45
|
+
"./commands/run.js": "./src/commands/run.js",
|
|
46
|
+
"./commands/supervise.js": "./src/commands/supervise.js",
|
|
47
|
+
"./commands/facilitate.js": "./src/commands/facilitate.js",
|
|
48
|
+
"./commands/discuss.js": "./src/commands/discuss.js",
|
|
49
|
+
"./commands/callback.js": "./src/commands/callback.js",
|
|
50
|
+
"./commands/scan-logs.js": "./src/commands/scan-logs.js",
|
|
51
|
+
"./commands/trace.js": "./src/commands/trace.js",
|
|
52
|
+
"./commands/assert.js": "./src/commands/assert.js",
|
|
53
|
+
"./commands/by-discussion.js": "./src/commands/by-discussion.js",
|
|
54
|
+
"./commands/benchmark-definition.js": "./src/commands/benchmark-definition.js",
|
|
55
|
+
"./commands/selfedit.js": "./src/commands/selfedit.js"
|
|
53
56
|
},
|
|
54
57
|
"files": [
|
|
55
58
|
"src/**/*.js",
|
|
56
|
-
"bin/**/*.js",
|
|
57
59
|
"README.md"
|
|
58
60
|
],
|
|
59
61
|
"scripts": {
|
package/src/agent-runner.js
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* AgentRunner — runs a single Claude Agent SDK session and emits raw
|
|
3
|
-
* NDJSON events to an output stream. Building block for `
|
|
4
|
-
* `
|
|
3
|
+
* NDJSON events to an output stream. Building block for `gemba-harness run`,
|
|
4
|
+
* `gemba-harness supervise`, `gemba-harness facilitate`, and `gemba-harness discuss`.
|
|
5
5
|
*
|
|
6
6
|
* Follows OO+DI: constructor injection, factory function, tests bypass factory.
|
|
7
7
|
*/
|
|
@@ -37,7 +37,7 @@ function modelDidWork(result) {
|
|
|
37
37
|
return tokens > 0 || (cost ?? 0) > 0;
|
|
38
38
|
}
|
|
39
39
|
|
|
40
|
-
//
|
|
40
|
+
// gemba-harness and kata-action run headless in CI/CD with no human to answer
|
|
41
41
|
// permission prompts. The SDK is always launched in bypass mode — not
|
|
42
42
|
// overridable — so a future caller can't accidentally reduce permissions.
|
|
43
43
|
const PERMISSION_MODE = "bypassPermissions";
|
|
@@ -0,0 +1,222 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Grading derivation — the sole home of the check-row arithmetic.
|
|
3
|
+
*
|
|
4
|
+
* Check rows are the single authoritative grading channel. Every row is a
|
|
5
|
+
* check by default; a row declares its role with its own fields, checked in
|
|
6
|
+
* order:
|
|
7
|
+
*
|
|
8
|
+
* 1. Gate — `gate` is exactly `true`, `pass` is boolean, and no
|
|
9
|
+
* `weight` key is present. Any failing gate → `gatesPass`
|
|
10
|
+
* false.
|
|
11
|
+
* 2. Diagnostic — no `gate` key and `weight` is exactly `0`. Free-form;
|
|
12
|
+
* never graded.
|
|
13
|
+
* 3. Scored — no `gate` key, boolean `pass`, `weight` absent (defaults
|
|
14
|
+
* to 1) or finite > 0.
|
|
15
|
+
* 4. Malformed — everything else: any `gate`+`weight` co-occurrence (a
|
|
16
|
+
* stray weight must never silently disarm a gate), a
|
|
17
|
+
* non-boolean `gate`, a missing or non-boolean `pass` on a
|
|
18
|
+
* graded row, an invalid `weight`, an fd-3 line that failed
|
|
19
|
+
* to parse, a non-object row. Counts as a **failing scored
|
|
20
|
+
* check** — dropping a defect could mint full marks;
|
|
21
|
+
* failing the whole run would zero completed work.
|
|
22
|
+
*
|
|
23
|
+
* The producers' `source` stamp is display metadata, never a grading input.
|
|
24
|
+
*/
|
|
25
|
+
|
|
26
|
+
/**
|
|
27
|
+
* @typedef {object} GradeResult
|
|
28
|
+
* @property {"pass" | "fail"} verdict - `healthy ∧ gatesPass ∧ fullMarks`.
|
|
29
|
+
* @property {boolean} gatesPass - Every gate row passes (vacuously true).
|
|
30
|
+
* @property {number | null} score - Weighted fraction of passing scored
|
|
31
|
+
* checks; `null` when the cell has zero scored checks (binary task).
|
|
32
|
+
* @property {boolean} fullMarks - Integer count predicate: no malformed rows
|
|
33
|
+
* and every scored check passes. Never a float comparison, so fractional
|
|
34
|
+
* weights carry no equality hazard. Vacuously true with zero scored checks.
|
|
35
|
+
* @property {number} malformed - Malformed row count.
|
|
36
|
+
*/
|
|
37
|
+
|
|
38
|
+
/**
|
|
39
|
+
* Grade the merged check rows against grader health.
|
|
40
|
+
*
|
|
41
|
+
* `healthy` is the completion signal a crashed grader cannot fake: when it is
|
|
42
|
+
* false the verdict is `fail` whatever the rows say, so a hook that dies
|
|
43
|
+
* after emitting passing rows can never mint marks.
|
|
44
|
+
* @param {unknown[]} details - Merged check rows from both producers.
|
|
45
|
+
* @param {boolean} healthy - Invariants exited 0 AND the hidden-test engine
|
|
46
|
+
* did not throw.
|
|
47
|
+
* @returns {GradeResult}
|
|
48
|
+
*/
|
|
49
|
+
export function gradeChecks(details, healthy) {
|
|
50
|
+
const tally = {
|
|
51
|
+
gatesPass: true,
|
|
52
|
+
malformed: 0,
|
|
53
|
+
scored: 0,
|
|
54
|
+
passing: 0,
|
|
55
|
+
weightAll: 0,
|
|
56
|
+
weightPassing: 0,
|
|
57
|
+
};
|
|
58
|
+
for (const row of details) tallyRow(tally, row);
|
|
59
|
+
|
|
60
|
+
const score =
|
|
61
|
+
tally.scored + tally.malformed === 0
|
|
62
|
+
? null
|
|
63
|
+
: tally.weightPassing / tally.weightAll;
|
|
64
|
+
const fullMarks = tally.malformed === 0 && tally.passing === tally.scored;
|
|
65
|
+
const verdict = healthy && tally.gatesPass && fullMarks ? "pass" : "fail";
|
|
66
|
+
return {
|
|
67
|
+
verdict,
|
|
68
|
+
gatesPass: tally.gatesPass,
|
|
69
|
+
score,
|
|
70
|
+
fullMarks,
|
|
71
|
+
malformed: tally.malformed,
|
|
72
|
+
};
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
/**
|
|
76
|
+
* Fold one row into the running tally per its classified role.
|
|
77
|
+
* @param {{gatesPass: boolean, malformed: number, scored: number, passing: number, weightAll: number, weightPassing: number}} tally
|
|
78
|
+
* @param {unknown} row
|
|
79
|
+
*/
|
|
80
|
+
function tallyRow(tally, row) {
|
|
81
|
+
const role = classifyRow(row);
|
|
82
|
+
if (role === "gate") {
|
|
83
|
+
if (!row.pass) tally.gatesPass = false;
|
|
84
|
+
} else if (role === "scored") {
|
|
85
|
+
const weight = row.weight ?? 1;
|
|
86
|
+
tally.scored++;
|
|
87
|
+
tally.weightAll += weight;
|
|
88
|
+
if (row.pass) {
|
|
89
|
+
tally.passing++;
|
|
90
|
+
tally.weightPassing += weight;
|
|
91
|
+
}
|
|
92
|
+
} else if (role === "malformed") {
|
|
93
|
+
tally.malformed++;
|
|
94
|
+
tally.weightAll += malformedWeight(row);
|
|
95
|
+
}
|
|
96
|
+
}
|
|
97
|
+
|
|
98
|
+
/**
|
|
99
|
+
* Run both check-row producers and grade the merged rows — the one
|
|
100
|
+
* composition shared by the runner and the `grade` subcommand. An engine
|
|
101
|
+
* throw is grader fault: its message lands on the returned `engineError`
|
|
102
|
+
* and health fails, so a crashed grader can never mint marks from rows it
|
|
103
|
+
* happened to emit first.
|
|
104
|
+
* @param {import("./task-family.js").Task} task
|
|
105
|
+
* @param {{cwd: string, port: number, runDir: string, familyDir?: string|null}} ctx
|
|
106
|
+
* @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
|
|
107
|
+
* @param {{runInvariants: Function, runHiddenTests: Function}} producers -
|
|
108
|
+
* The two producer functions (real implementations or test seams).
|
|
109
|
+
* @returns {Promise<{invariants: object, hiddenRows: object[], engineError: Error|null, rows: unknown[], healthy: boolean, grade: object}>}
|
|
110
|
+
*/
|
|
111
|
+
export async function runProducersAndGrade(task, ctx, runtime, producers) {
|
|
112
|
+
const invariants = await producers.runInvariants(task, ctx, runtime);
|
|
113
|
+
let hiddenRows = [];
|
|
114
|
+
let engineError = null;
|
|
115
|
+
try {
|
|
116
|
+
const hidden = await producers.runHiddenTests(task, ctx, runtime);
|
|
117
|
+
hiddenRows = hidden.details;
|
|
118
|
+
} catch (e) {
|
|
119
|
+
engineError = e;
|
|
120
|
+
}
|
|
121
|
+
const rows = mergeRows(invariants.details, hiddenRows);
|
|
122
|
+
const healthy = invariants.exitCode === 0 && !engineError;
|
|
123
|
+
const grade = normalizeGrade(gradeChecks(rows, healthy));
|
|
124
|
+
return { invariants, hiddenRows, engineError, rows, healthy, grade };
|
|
125
|
+
}
|
|
126
|
+
|
|
127
|
+
/**
|
|
128
|
+
* Merge the two producers' rows (invariants first) and stamp each row's
|
|
129
|
+
* provenance. The stamp is display metadata, never a grading input, and
|
|
130
|
+
* non-object rows (malformed by contract) pass through verbatim.
|
|
131
|
+
* @param {unknown[]} invariantsDetails
|
|
132
|
+
* @param {unknown[]} hiddenDetails
|
|
133
|
+
* @returns {unknown[]}
|
|
134
|
+
*/
|
|
135
|
+
export function mergeRows(invariantsDetails, hiddenDetails) {
|
|
136
|
+
return [
|
|
137
|
+
...invariantsDetails.map((row) => stampSource(row, "invariants")),
|
|
138
|
+
...hiddenDetails.map((row) => stampSource(row, "tests")),
|
|
139
|
+
];
|
|
140
|
+
}
|
|
141
|
+
|
|
142
|
+
function stampSource(row, source) {
|
|
143
|
+
if (row === null || typeof row !== "object" || Array.isArray(row)) {
|
|
144
|
+
return row;
|
|
145
|
+
}
|
|
146
|
+
return { ...row, source };
|
|
147
|
+
}
|
|
148
|
+
|
|
149
|
+
/**
|
|
150
|
+
* Project the raw `gradeChecks` return onto the record schema: `fullMarks`
|
|
151
|
+
* is derivable and dropped, `score` is omitted on binary tasks (`null`),
|
|
152
|
+
* `malformed` is omitted when clean.
|
|
153
|
+
* @param {GradeResult} raw
|
|
154
|
+
* @returns {{verdict: "pass"|"fail", gatesPass: boolean, score?: number, malformed?: number}}
|
|
155
|
+
*/
|
|
156
|
+
export function normalizeGrade({ verdict, gatesPass, score, malformed }) {
|
|
157
|
+
return {
|
|
158
|
+
verdict,
|
|
159
|
+
gatesPass,
|
|
160
|
+
...(score !== null && { score }),
|
|
161
|
+
...(malformed > 0 && { malformed }),
|
|
162
|
+
};
|
|
163
|
+
}
|
|
164
|
+
|
|
165
|
+
/**
|
|
166
|
+
* Classify one row per the role order in the module contract.
|
|
167
|
+
* @param {unknown} row
|
|
168
|
+
* @returns {"gate" | "diagnostic" | "scored" | "malformed"}
|
|
169
|
+
*/
|
|
170
|
+
function classifyRow(row) {
|
|
171
|
+
if (row === null || typeof row !== "object" || Array.isArray(row)) {
|
|
172
|
+
return "malformed";
|
|
173
|
+
}
|
|
174
|
+
if ("gate" in row) return classifyGateRow(row);
|
|
175
|
+
if ("weight" in row) return classifyWeightedRow(row);
|
|
176
|
+
return typeof row.pass === "boolean" ? "scored" : "malformed";
|
|
177
|
+
}
|
|
178
|
+
|
|
179
|
+
/**
|
|
180
|
+
* A row carrying a `gate` key: valid only as `gate: true` with a boolean
|
|
181
|
+
* `pass` and no `weight` key — any co-occurring weight is malformed so a
|
|
182
|
+
* stray weight can never silently disarm a gate.
|
|
183
|
+
* @param {object} row
|
|
184
|
+
* @returns {"gate" | "malformed"}
|
|
185
|
+
*/
|
|
186
|
+
function classifyGateRow(row) {
|
|
187
|
+
if ("weight" in row) return "malformed";
|
|
188
|
+
return row.gate === true && typeof row.pass === "boolean"
|
|
189
|
+
? "gate"
|
|
190
|
+
: "malformed";
|
|
191
|
+
}
|
|
192
|
+
|
|
193
|
+
/**
|
|
194
|
+
* A gate-less row carrying a `weight` key: exactly 0 is a diagnostic, a
|
|
195
|
+
* finite positive weight with a boolean `pass` is scored, anything else is
|
|
196
|
+
* malformed.
|
|
197
|
+
* @param {object} row
|
|
198
|
+
* @returns {"diagnostic" | "scored" | "malformed"}
|
|
199
|
+
*/
|
|
200
|
+
function classifyWeightedRow(row) {
|
|
201
|
+
if (row.weight === 0) return "diagnostic";
|
|
202
|
+
return isValidWeight(row.weight) && typeof row.pass === "boolean"
|
|
203
|
+
? "scored"
|
|
204
|
+
: "malformed";
|
|
205
|
+
}
|
|
206
|
+
|
|
207
|
+
/**
|
|
208
|
+
* A malformed row fails at its own weight when it carries a valid positive
|
|
209
|
+
* one, else at unit weight 1.
|
|
210
|
+
* @param {unknown} row
|
|
211
|
+
* @returns {number}
|
|
212
|
+
*/
|
|
213
|
+
function malformedWeight(row) {
|
|
214
|
+
if (row !== null && typeof row === "object" && isValidWeight(row.weight)) {
|
|
215
|
+
return row.weight;
|
|
216
|
+
}
|
|
217
|
+
return 1;
|
|
218
|
+
}
|
|
219
|
+
|
|
220
|
+
function isValidWeight(w) {
|
|
221
|
+
return typeof w === "number" && Number.isFinite(w) && w > 0;
|
|
222
|
+
}
|
|
@@ -0,0 +1,180 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Hidden-test engine — executes a task's `tests/` overlay against the
|
|
3
|
+
* post-run agent CWD: stage each file at its mirrored path, run each check
|
|
4
|
+
* with `node --test`, convert the exit status into one check row, and
|
|
5
|
+
* restore the tree so the judge sees the workdir exactly as the agent left
|
|
6
|
+
* it.
|
|
7
|
+
*
|
|
8
|
+
* Fault attribution is the engine's contract: a stage or spawn failure (the
|
|
9
|
+
* agent deleted the scaffold) is a *failing row* — agent fault; the engine
|
|
10
|
+
* itself throwing is grader fault, which the caller records as unhealthy so
|
|
11
|
+
* a crashed grader can never mint marks.
|
|
12
|
+
*/
|
|
13
|
+
|
|
14
|
+
import { dirname, join } from "node:path";
|
|
15
|
+
|
|
16
|
+
import { buildHookEnv } from "./hook-env.js";
|
|
17
|
+
|
|
18
|
+
// Fixed per-check budget. A wedged test process runs outside the agent
|
|
19
|
+
// watchdog, so this bound is what keeps a hung hidden test from stalling the
|
|
20
|
+
// cell; the timeout row keeps the failure visible.
|
|
21
|
+
const CHECK_TIMEOUT_MS = 120_000;
|
|
22
|
+
const STDERR_TAIL_CHARS = 500;
|
|
23
|
+
|
|
24
|
+
/**
|
|
25
|
+
* Run the task's hidden test suite.
|
|
26
|
+
* @param {import("./task-family.js").Task} task
|
|
27
|
+
* @param {{cwd: string, port: number, runDir: string, familyDir?: string|null}} ctx
|
|
28
|
+
* @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
|
|
29
|
+
* @param {{timeoutMs?: number}} [opts] - Test seam for the per-check timeout.
|
|
30
|
+
* @returns {Promise<{details: object[]}>}
|
|
31
|
+
*/
|
|
32
|
+
export async function runHiddenTests(task, ctx, runtime, opts = {}) {
|
|
33
|
+
if (!runtime) throw new Error("runtime is required");
|
|
34
|
+
if (!task.tests) return { details: [] };
|
|
35
|
+
const timeoutMs = opts.timeoutMs ?? CHECK_TIMEOUT_MS;
|
|
36
|
+
const fs = runtime.fs;
|
|
37
|
+
const details = [];
|
|
38
|
+
const supportStager = newStager();
|
|
39
|
+
try {
|
|
40
|
+
for (const file of task.tests.support) {
|
|
41
|
+
await stageFile(fs, ctx.cwd, supportStager, file);
|
|
42
|
+
}
|
|
43
|
+
for (const check of task.tests.checks) {
|
|
44
|
+
details.push(await runOneCheck(task, ctx, runtime, timeoutMs, check));
|
|
45
|
+
}
|
|
46
|
+
} finally {
|
|
47
|
+
await unstage(fs, supportStager);
|
|
48
|
+
}
|
|
49
|
+
return { details };
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
/**
|
|
53
|
+
* Stage one check, run it, and restore its staging — the check's own row is
|
|
54
|
+
* the only trace it leaves. A stage failure is the agent's fault (a deleted
|
|
55
|
+
* scaffold), so it becomes a failing row rather than a throw.
|
|
56
|
+
*/
|
|
57
|
+
async function runOneCheck(task, ctx, runtime, timeoutMs, check) {
|
|
58
|
+
const stager = newStager();
|
|
59
|
+
try {
|
|
60
|
+
try {
|
|
61
|
+
await stageFile(runtime.fs, ctx.cwd, stager, check);
|
|
62
|
+
} catch (e) {
|
|
63
|
+
return checkRow(check, false, `stage failed: ${e.message}`);
|
|
64
|
+
}
|
|
65
|
+
return await spawnCheck(task, ctx, runtime, timeoutMs, check);
|
|
66
|
+
} finally {
|
|
67
|
+
await unstage(runtime.fs, stager);
|
|
68
|
+
}
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
/**
|
|
72
|
+
* Spawn `node --test <staged path>` from the agent CWD under the hook env
|
|
73
|
+
* and map the exit status onto one row. The clock timer SIGKILLs a child
|
|
74
|
+
* that outlives the per-check budget; the row fails with a timeout message.
|
|
75
|
+
*/
|
|
76
|
+
async function spawnCheck(task, ctx, runtime, timeoutMs, check) {
|
|
77
|
+
const env = buildHookEnv(runtime.proc.env, {
|
|
78
|
+
cwd: ctx.cwd,
|
|
79
|
+
port: ctx.port,
|
|
80
|
+
taskId: task.id,
|
|
81
|
+
taskDir: task.paths.taskDir,
|
|
82
|
+
hooksDir: task.paths.hooks,
|
|
83
|
+
familyDir: ctx.familyDir,
|
|
84
|
+
});
|
|
85
|
+
// An inherited test-runner context makes the child `node --test` report
|
|
86
|
+
// exit 0 even when its tests fail — a failing check would mint a passing
|
|
87
|
+
// row whenever the harness itself runs under `node --test`.
|
|
88
|
+
delete env.NODE_TEST_CONTEXT;
|
|
89
|
+
const child = runtime.subprocess.spawn("node", ["--test", check.stagePath], {
|
|
90
|
+
cwd: ctx.cwd,
|
|
91
|
+
env,
|
|
92
|
+
stdio: ["ignore", "pipe", "pipe"],
|
|
93
|
+
});
|
|
94
|
+
let timedOut = false;
|
|
95
|
+
const timer = runtime.clock.setTimeout(() => {
|
|
96
|
+
timedOut = true;
|
|
97
|
+
child.kill("SIGKILL");
|
|
98
|
+
}, timeoutMs);
|
|
99
|
+
const drainStdout = (async () => {
|
|
100
|
+
for await (const _chunk of child.stdout) {
|
|
101
|
+
// discard
|
|
102
|
+
}
|
|
103
|
+
})();
|
|
104
|
+
let stderr = "";
|
|
105
|
+
for await (const chunk of child.stderr) stderr += chunk.toString();
|
|
106
|
+
await drainStdout;
|
|
107
|
+
const exit = await child.exitCode;
|
|
108
|
+
runtime.clock.clearTimeout(timer);
|
|
109
|
+
|
|
110
|
+
if (timedOut) {
|
|
111
|
+
return checkRow(check, false, `timed out after ${timeoutMs}ms`);
|
|
112
|
+
}
|
|
113
|
+
if (exit === 0) return checkRow(check, true);
|
|
114
|
+
const tail = stderr.trim().slice(-STDERR_TAIL_CHARS);
|
|
115
|
+
return checkRow(check, false, `exit ${exit}${tail ? `: ${tail}` : ""}`);
|
|
116
|
+
}
|
|
117
|
+
|
|
118
|
+
function checkRow(check, pass, message) {
|
|
119
|
+
return {
|
|
120
|
+
test: check.name,
|
|
121
|
+
pass,
|
|
122
|
+
...(check.gate && { gate: true }),
|
|
123
|
+
...(message && { message }),
|
|
124
|
+
};
|
|
125
|
+
}
|
|
126
|
+
|
|
127
|
+
function newStager() {
|
|
128
|
+
return { staged: [], backups: [], createdDirs: [] };
|
|
129
|
+
}
|
|
130
|
+
|
|
131
|
+
/**
|
|
132
|
+
* Copy the symlink-resolved source to its mirrored path under the agent CWD,
|
|
133
|
+
* backing up a collided file's bytes and tracking every directory created so
|
|
134
|
+
* `unstage` can put the tree back exactly.
|
|
135
|
+
*/
|
|
136
|
+
async function stageFile(fs, cwd, stager, { sourcePath, stagePath }) {
|
|
137
|
+
const target = join(cwd, stagePath);
|
|
138
|
+
let collided = null;
|
|
139
|
+
try {
|
|
140
|
+
collided = await fs.readFile(target);
|
|
141
|
+
} catch {
|
|
142
|
+
// no collision
|
|
143
|
+
}
|
|
144
|
+
if (collided !== null) stager.backups.push({ target, bytes: collided });
|
|
145
|
+
await ensureParents(fs, cwd, stager, dirname(target));
|
|
146
|
+
const resolved = await fs.realpath(sourcePath);
|
|
147
|
+
await fs.copyFile(resolved, target);
|
|
148
|
+
stager.staged.push(target);
|
|
149
|
+
}
|
|
150
|
+
|
|
151
|
+
async function ensureParents(fs, cwd, stager, dir) {
|
|
152
|
+
if (dir === cwd) return;
|
|
153
|
+
try {
|
|
154
|
+
await fs.access(dir);
|
|
155
|
+
return;
|
|
156
|
+
} catch {
|
|
157
|
+
// missing — create below
|
|
158
|
+
}
|
|
159
|
+
await ensureParents(fs, cwd, stager, dirname(dir));
|
|
160
|
+
await fs.mkdir(dir);
|
|
161
|
+
stager.createdDirs.push(dir);
|
|
162
|
+
}
|
|
163
|
+
|
|
164
|
+
/**
|
|
165
|
+
* Reverse the staging: staged copies out, collided bytes back, created
|
|
166
|
+
* directories removed (deepest first — a check's own artifacts inside a
|
|
167
|
+
* created directory go with it, since that directory did not exist when the
|
|
168
|
+
* agent finished).
|
|
169
|
+
*/
|
|
170
|
+
async function unstage(fs, stager) {
|
|
171
|
+
for (const target of stager.staged) {
|
|
172
|
+
await fs.rm(target, { force: true });
|
|
173
|
+
}
|
|
174
|
+
for (const backup of stager.backups) {
|
|
175
|
+
await fs.writeFile(backup.target, backup.bytes);
|
|
176
|
+
}
|
|
177
|
+
for (const dir of [...stager.createdDirs].reverse()) {
|
|
178
|
+
await fs.rm(dir, { recursive: true, force: true });
|
|
179
|
+
}
|
|
180
|
+
}
|
|
@@ -1,7 +1,10 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* Invariants — runs `<task.paths.hooks>/invariants.sh` from the template path
|
|
3
|
-
* against the post-run agent CWD.
|
|
4
|
-
*
|
|
3
|
+
* against the post-run agent CWD. A pure collector with no verdict of its
|
|
4
|
+
* own: structured per-check rows arrive on fd 3 (`$RESULTS_FD=3`) as NDJSON
|
|
5
|
+
* and grading happens downstream over the merged rows. The exit code is
|
|
6
|
+
* script health only — nonzero means the grader itself failed, never that a
|
|
7
|
+
* check failed.
|
|
5
8
|
*
|
|
6
9
|
* Subprocess access flows through `runtime.subprocess.spawn`; the fd-3 backing
|
|
7
10
|
* store and the stderr log use the sync filesystem surface (`runtime.fsSync`) —
|
|
@@ -14,9 +17,9 @@ import { buildHookEnv } from "./hook-env.js";
|
|
|
14
17
|
|
|
15
18
|
/**
|
|
16
19
|
* @typedef {object} InvariantsResult
|
|
17
|
-
* @property {"pass" | "fail"} verdict
|
|
18
20
|
* @property {Array<object>} details
|
|
19
|
-
* @property {number} exitCode
|
|
21
|
+
* @property {number} exitCode - Script health: nonzero means the hook itself
|
|
22
|
+
* failed, never that a check failed.
|
|
20
23
|
* @property {string} [stderr] - Trimmed script stderr, present only when the
|
|
21
24
|
* script wrote to stderr. Surfaces hook failures (e.g. a missing tool) that
|
|
22
25
|
* leave `details` empty, so they read distinctly from a real invariant miss.
|
|
@@ -32,7 +35,7 @@ import { buildHookEnv } from "./hook-env.js";
|
|
|
32
35
|
export async function runInvariants(task, ctx, runtime) {
|
|
33
36
|
if (!runtime) throw new Error("runtime is required");
|
|
34
37
|
if (!task.paths.invariants) {
|
|
35
|
-
return {
|
|
38
|
+
return { details: [], exitCode: 0 };
|
|
36
39
|
}
|
|
37
40
|
const fsSync = runtime.fsSync;
|
|
38
41
|
const script = task.paths.invariants;
|
|
@@ -84,11 +87,7 @@ export async function runInvariants(task, ctx, runtime) {
|
|
|
84
87
|
const details = [];
|
|
85
88
|
parseFd3Buffer(raw, details);
|
|
86
89
|
const exitCode = typeof code === "number" ? code : -1;
|
|
87
|
-
const result = {
|
|
88
|
-
verdict: exitCode === 0 ? "pass" : "fail",
|
|
89
|
-
details,
|
|
90
|
-
exitCode,
|
|
91
|
-
};
|
|
90
|
+
const result = { details, exitCode };
|
|
92
91
|
const trimmedStderr = stderr.trim();
|
|
93
92
|
if (trimmedStderr) result.stderr = trimmedStderr;
|
|
94
93
|
return result;
|
package/src/benchmark/judge.js
CHANGED
|
@@ -9,7 +9,7 @@
|
|
|
9
9
|
* {{AGENT_INSTRUCTIONS}} — contents of agent.task.md
|
|
10
10
|
* {{AGENT_PROFILE}} — agent profile body (empty string if none)
|
|
11
11
|
* {{AGENT_TRACE_PATH}} — path to agent.ndjson
|
|
12
|
-
* {{
|
|
12
|
+
* {{GRADE_RESULT}} — JSON grade object plus the merged check rows
|
|
13
13
|
* {{SKILL_SET_HASH}} — SHA-256 from apm.lock.yaml
|
|
14
14
|
* {{TASK_ID}} — task name (directory under tasks/)
|
|
15
15
|
* {{TASK_DIR}} — agent working directory path
|
|
@@ -40,22 +40,25 @@ import { sumTraceCost } from "../cost.js";
|
|
|
40
40
|
*/
|
|
41
41
|
|
|
42
42
|
/**
|
|
43
|
-
* Run the judge over a completed task run.
|
|
43
|
+
* Run the judge over a completed task run. The judge is a binary gate over
|
|
44
|
+
* the grade's validity, never a grade itself: `gradeResult` reaches the
|
|
45
|
+
* template as evidence, and the verdict stays pass/fail.
|
|
44
46
|
* @param {import("./task-family.js").Task} task
|
|
45
47
|
* @param {import("./workdir.js").Workdir} workdir
|
|
46
|
-
* @param {
|
|
48
|
+
* @param {{verdict: string, gatesPass: boolean, score?: number, malformed?: number, rows: unknown[]}} gradeResult -
|
|
49
|
+
* The normalized grade plus the merged, source-stamped check rows.
|
|
47
50
|
* @param {{query: Function, model: string, judgeProfile?: string, profilesDir?: string, runtime: import("@forwardimpact/libutil/runtime").Runtime}} deps
|
|
48
51
|
* @param {JudgeContext} [context]
|
|
49
52
|
* @returns {Promise<JudgeVerdict>}
|
|
50
53
|
*/
|
|
51
|
-
export async function runJudge(task, workdir,
|
|
54
|
+
export async function runJudge(task, workdir, gradeResult, deps, context) {
|
|
52
55
|
const runtime = deps.runtime;
|
|
53
56
|
if (!runtime) throw new Error("runtime is required");
|
|
54
57
|
const fs = runtime.fs;
|
|
55
58
|
const template = await fs.readFile(task.paths.judge, "utf8");
|
|
56
|
-
const
|
|
59
|
+
const gradeJson = JSON.stringify(gradeResult, null, 2);
|
|
57
60
|
const taskText = template
|
|
58
|
-
.replaceAll("{{
|
|
61
|
+
.replaceAll("{{GRADE_RESULT}}", gradeJson)
|
|
59
62
|
.replaceAll("{{AGENT_TRACE_PATH}}", workdir.agentTracePath)
|
|
60
63
|
.replaceAll("{{AGENT_INSTRUCTIONS}}", context?.agentInstructions ?? "")
|
|
61
64
|
.replaceAll("{{AGENT_PROFILE}}", context?.agentProfile ?? "")
|