@forwardimpact/libharness 2.0.0 → 3.0.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +68 -65
- package/package.json +15 -13
- package/src/advisor.js +47 -41
- package/src/agent-runner.js +58 -48
- package/src/benchmark/apm-installer.js +28 -28
- package/src/benchmark/env-loader.js +24 -16
- package/src/benchmark/grade.js +44 -41
- package/src/benchmark/hidden-tests.js +25 -24
- package/src/benchmark/hook-env.js +11 -9
- package/src/benchmark/invariants.js +20 -17
- package/src/benchmark/judge.js +29 -28
- package/src/benchmark/npm-installer.js +9 -8
- package/src/benchmark/report.js +53 -50
- package/src/benchmark/result.js +24 -23
- package/src/benchmark/runner.js +75 -69
- package/src/benchmark/scheduler.js +17 -16
- package/src/benchmark/task-family.js +29 -27
- package/src/benchmark/trace-split.js +9 -8
- package/src/benchmark/workdir.js +27 -25
- package/src/claude-code-executable.js +11 -11
- package/src/commands/advisor-flags.js +8 -7
- package/src/commands/assert.js +16 -15
- package/src/commands/benchmark-definition.js +20 -20
- package/src/commands/benchmark-grade.js +13 -12
- package/src/commands/benchmark-report.js +5 -5
- package/src/commands/benchmark-run.js +31 -28
- package/src/commands/by-discussion.js +11 -11
- package/src/commands/callback.js +11 -11
- package/src/commands/discuss.js +8 -7
- package/src/commands/facilitate.js +16 -14
- package/src/commands/output.js +4 -3
- package/src/commands/run.js +15 -15
- package/src/commands/scan-logs.js +22 -20
- package/src/commands/selfedit.js +124 -0
- package/src/commands/supervise.js +13 -11
- package/src/commands/task-input.js +9 -9
- package/src/commands/tee.js +11 -10
- package/src/commands/trace.js +55 -42
- package/src/commands/work-tracker.js +4 -3
- package/src/cost.js +17 -17
- package/src/discuss-tools.js +16 -16
- package/src/discusser.js +39 -38
- package/src/events/github.js +54 -37
- package/src/facilitator.js +21 -21
- package/src/inbox-poller.js +4 -4
- package/src/judge.js +32 -30
- package/src/message-bus.js +12 -11
- package/src/orchestration-loop.js +35 -36
- package/src/orchestration-toolkit.js +58 -53
- package/src/orchestrator-helpers.js +2 -2
- package/src/profile-prompt.js +54 -53
- package/src/redaction.js +63 -57
- package/src/render/line-renderer.js +5 -5
- package/src/render/orchestrator-filter.js +3 -3
- package/src/render/palette.js +11 -9
- package/src/render/tool-hints.js +18 -15
- package/src/render/turn-renderer.js +4 -4
- package/src/reply-emitter.js +2 -2
- package/src/sequence-counter.js +4 -3
- package/src/signature-filter.js +7 -6
- package/src/supervisor.js +19 -18
- package/src/tee-writer.js +25 -25
- package/src/trace-collector.js +53 -48
- package/src/trace-github.js +53 -44
- package/src/trace-multi.js +16 -14
- package/src/trace-query.js +61 -52
- package/src/trace-render.js +19 -19
- package/src/trace-usage.js +31 -28
- package/src/transcript-recorder.js +24 -20
- package/bin/fit-benchmark.js +0 -44
- package/bin/fit-harness.js +0 -412
- package/bin/fit-selfedit.js +0 -165
- package/bin/fit-trace.js +0 -520
package/src/agent-runner.js
CHANGED
|
@@ -1,7 +1,8 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* AgentRunner — runs a single Claude Agent SDK session and emits raw
|
|
3
|
-
* NDJSON events to an output stream.
|
|
4
|
-
* `
|
|
3
|
+
* NDJSON events to an output stream. `gemba-harness run`,
|
|
4
|
+
* `gemba-harness supervise`, `gemba-harness facilitate`, and
|
|
5
|
+
* `gemba-harness discuss` build on it.
|
|
5
6
|
*
|
|
6
7
|
* Follows OO+DI: constructor injection, factory function, tests bypass factory.
|
|
7
8
|
*/
|
|
@@ -12,15 +13,16 @@ import { resolveClaudeCodeExecutable } from "./claude-code-executable.js";
|
|
|
12
13
|
const DEFAULT_ALLOWED_TOOLS = ["Bash", "Read", "Glob", "Grep", "Write", "Edit"];
|
|
13
14
|
|
|
14
15
|
/**
|
|
15
|
-
*
|
|
16
|
+
* Report whether the session invoked the model. A genuine run always bills
|
|
16
17
|
* tokens (the system prompt alone is thousands of input tokens) and costs
|
|
17
|
-
* more than zero. A `result` message
|
|
18
|
-
* token usage and zero cost means the
|
|
19
|
-
* canonical signature of a Claude Code init
|
|
20
|
-
* `ANTHROPIC_API_KEY`)
|
|
18
|
+
* more than zero. A `result` message can carry `subtype: "success"` with
|
|
19
|
+
* zero token usage and zero cost. That combination means the run never
|
|
20
|
+
* reached the model. It is the canonical signature of a Claude Code init or
|
|
21
|
+
* auth failure (e.g. an invalid `ANTHROPIC_API_KEY`). The SDK otherwise
|
|
22
|
+
* reports that failure as a clean success.
|
|
21
23
|
*
|
|
22
24
|
* If the SDK gave us neither a `usage` object nor `total_cost_usd`, don't
|
|
23
|
-
* second-guess the subtype
|
|
25
|
+
* second-guess the subtype. Trust the reported success.
|
|
24
26
|
* @param {object|null} result - The SDK `result` message, or null.
|
|
25
27
|
* @returns {boolean}
|
|
26
28
|
*/
|
|
@@ -37,9 +39,10 @@ function modelDidWork(result) {
|
|
|
37
39
|
return tokens > 0 || (cost ?? 0) > 0;
|
|
38
40
|
}
|
|
39
41
|
|
|
40
|
-
//
|
|
41
|
-
// permission prompts. The
|
|
42
|
-
//
|
|
42
|
+
// gemba-harness and kata-action run headless in CI/CD with no human to answer
|
|
43
|
+
// permission prompts. The runner always launches the SDK in bypass mode. No
|
|
44
|
+
// caller can override that mode, so a future caller can't accidentally
|
|
45
|
+
// reduce permissions.
|
|
43
46
|
const PERMISSION_MODE = "bypassPermissions";
|
|
44
47
|
|
|
45
48
|
/** Run a single Claude Agent SDK session and emit raw NDJSON events to an output stream. */
|
|
@@ -47,25 +50,27 @@ export class AgentRunner {
|
|
|
47
50
|
/**
|
|
48
51
|
* @param {object} deps
|
|
49
52
|
* @param {string} deps.cwd - Agent working directory
|
|
50
|
-
* @param {function} deps.query - SDK query function (
|
|
53
|
+
* @param {function} deps.query - SDK query function (tests inject it)
|
|
51
54
|
* @param {import("stream").Writable} deps.output - Stream to emit NDJSON to
|
|
52
55
|
* @param {string} [deps.model] - Claude model identifier
|
|
53
|
-
* @param {number} [deps.maxTurns] - Maximum agentic turns
|
|
56
|
+
* @param {number} [deps.maxTurns] - Maximum agentic turns. 0 means unlimited
|
|
54
57
|
* @param {string[]} [deps.allowedTools] - Tools the agent may use
|
|
55
|
-
* @param {function} [deps.onLine] - Callback
|
|
56
|
-
* @param {function} [deps.onPrompt] - Callback
|
|
58
|
+
* @param {function} [deps.onLine] - Callback that receives each NDJSON line as the runner produces it
|
|
59
|
+
* @param {function} [deps.onPrompt] - Callback that receives the effective (amend-applied) prompt of each run/resume
|
|
57
60
|
* @param {string[]} [deps.settingSources] - SDK setting sources (e.g. ['project'] to load CLAUDE.md)
|
|
58
|
-
* @param {string|object} [deps.systemPrompt] - SDK system prompt
|
|
61
|
+
* @param {string|object} [deps.systemPrompt] - SDK system prompt. A string replaces the default. The preset form {type:'preset', preset:'claude_code', append} appends
|
|
59
62
|
* @param {string[]} [deps.disallowedTools] - Tools to explicitly remove from the model's context
|
|
60
63
|
* @param {Record<string, object>} [deps.mcpServers] - MCP server configs to pass to the SDK query
|
|
61
64
|
* @param {string} [deps.pathToClaudeCodeExecutable] - Absolute path to the
|
|
62
|
-
* native `claude` CLI the SDK should spawn. Set for compiled fit-*
|
|
63
|
-
* which can't self-resolve the SDK's platform optional
|
|
64
|
-
* from source runs so the SDK resolves its own
|
|
65
|
+
* native `claude` CLI the SDK should spawn. Set it for compiled fit-*
|
|
66
|
+
* binaries, which can't self-resolve the SDK's platform optional
|
|
67
|
+
* dependency. Omit it from source runs so the SDK resolves its own
|
|
68
|
+
* version-matched binary.
|
|
65
69
|
* @param {object} deps.redactor
|
|
66
70
|
* @param {import("@forwardimpact/libutil/runtime").Runtime} [deps.runtime] -
|
|
67
|
-
* Ambient collaborators.
|
|
68
|
-
* invocations into `LIBHARNESS_SKILL`
|
|
71
|
+
* Ambient collaborators. The runner reads only `proc.env`, to record Skill
|
|
72
|
+
* invocations into `LIBHARNESS_SKILL`. When `runtime` is absent, the
|
|
73
|
+
* runner skips the write.
|
|
69
74
|
*/
|
|
70
75
|
constructor(deps) {
|
|
71
76
|
if (!deps.cwd) throw new Error("cwd is required");
|
|
@@ -81,15 +86,17 @@ export class AgentRunner {
|
|
|
81
86
|
this.maxTurns = deps.maxTurns ?? 50;
|
|
82
87
|
this.allowedTools = deps.allowedTools ?? DEFAULT_ALLOWED_TOOLS;
|
|
83
88
|
this.onLine = deps.onLine ?? null;
|
|
84
|
-
// Optional
|
|
85
|
-
// absent value stays undefined
|
|
89
|
+
// Optional. The code reads it only through a truthy guard in
|
|
90
|
+
// run()/resume(), so an absent value stays undefined and needs no
|
|
91
|
+
// `?? null` default.
|
|
86
92
|
this.onPrompt = deps.onPrompt;
|
|
87
93
|
this.settingSources = deps.settingSources ?? [];
|
|
88
94
|
this.systemPrompt = deps.systemPrompt ?? null;
|
|
89
95
|
this.disallowedTools = deps.disallowedTools ?? [];
|
|
90
96
|
this.mcpServers = deps.mcpServers ?? null;
|
|
91
|
-
// Optional
|
|
92
|
-
//
|
|
97
|
+
// Optional. The code reads it only through a truthy guard in
|
|
98
|
+
// #callOptions, so an absent value stays undefined and needs no
|
|
99
|
+
// `?? null` default.
|
|
93
100
|
this.pathToClaudeCodeExecutable = deps.pathToClaudeCodeExecutable;
|
|
94
101
|
this.taskAmend = deps.taskAmend ?? null;
|
|
95
102
|
this.sessionId = null;
|
|
@@ -146,16 +153,17 @@ export class AgentRunner {
|
|
|
146
153
|
}
|
|
147
154
|
|
|
148
155
|
/**
|
|
149
|
-
* Build the options
|
|
150
|
-
*
|
|
151
|
-
*
|
|
152
|
-
*
|
|
156
|
+
* Build the options for every SDK query() call. run() and resume()
|
|
157
|
+
* share this method, so the agent's configuration stays identical
|
|
158
|
+
* across the session's lifetime. That configuration is the cwd, the
|
|
159
|
+
* tools, the prompt, the setting sources, and the turn budget. Only
|
|
160
|
+
* resume() layers `resume: this.sessionId` on top.
|
|
153
161
|
*
|
|
154
|
-
* SDK options
|
|
155
|
-
* call loads the prior conversation
|
|
156
|
-
* options this call passes.
|
|
157
|
-
* resume
|
|
158
|
-
* persona between turns.
|
|
162
|
+
* SDK options attach to the call. They do not attach to the session.
|
|
163
|
+
* The resumed call loads the prior conversation. It otherwise uses
|
|
164
|
+
* whatever options this call passes. If you omit the tool, prompt, or
|
|
165
|
+
* setting options on resume, the agent silently loses its restrictions
|
|
166
|
+
* and persona between turns.
|
|
159
167
|
*/
|
|
160
168
|
#callOptions(abortController) {
|
|
161
169
|
return {
|
|
@@ -179,14 +187,14 @@ export class AgentRunner {
|
|
|
179
187
|
}
|
|
180
188
|
|
|
181
189
|
/**
|
|
182
|
-
* Iterate the SDK query iterator
|
|
183
|
-
*
|
|
184
|
-
*
|
|
190
|
+
* Iterate the SDK query iterator. Mirror every message to the output
|
|
191
|
+
* stream and to the `onLine` callback. Capture `sessionId` from the
|
|
192
|
+
* SDK's `system/init` message. Track Skill invocations into
|
|
185
193
|
* `LIBHARNESS_SKILL` for downstream metrics.
|
|
186
194
|
*
|
|
187
195
|
* If the iterator throws and we triggered the abort ourselves
|
|
188
196
|
* (`currentAbortController.signal.aborted`), we report `aborted:
|
|
189
|
-
* true
|
|
197
|
+
* true`. Otherwise the error propagates as `error`.
|
|
190
198
|
*/
|
|
191
199
|
async #consumeQuery(iterator) {
|
|
192
200
|
let text = "";
|
|
@@ -212,10 +220,10 @@ export class AgentRunner {
|
|
|
212
220
|
}
|
|
213
221
|
}
|
|
214
222
|
|
|
215
|
-
// A "success" subtype is necessary
|
|
216
|
-
// failed init (e.g. an invalid API key) as success with zero model work.
|
|
217
|
-
// Require evidence the model
|
|
218
|
-
//
|
|
223
|
+
// A "success" subtype is necessary. It is not sufficient. The SDK reports
|
|
224
|
+
// a failed init (e.g. an invalid API key) as success with zero model work.
|
|
225
|
+
// Require evidence that the model ran. Surface a clear error when it did
|
|
226
|
+
// not, so nobody reports the masked failure as a green run.
|
|
219
227
|
const reportedSuccess = stopReason === "success";
|
|
220
228
|
const success =
|
|
221
229
|
reportedSuccess &&
|
|
@@ -223,7 +231,7 @@ export class AgentRunner {
|
|
|
223
231
|
modelDidWork(resultMessage);
|
|
224
232
|
if (reportedSuccess && !success && !error) {
|
|
225
233
|
error = new Error(
|
|
226
|
-
"agent reported success but
|
|
234
|
+
"agent reported success but did no model work (zero token usage), which is likely a Claude Code init or authentication failure",
|
|
227
235
|
);
|
|
228
236
|
}
|
|
229
237
|
|
|
@@ -251,8 +259,9 @@ export class AgentRunner {
|
|
|
251
259
|
#trackSkillInvocation(message) {
|
|
252
260
|
const content = message.message?.content ?? message.content;
|
|
253
261
|
if (!Array.isArray(content)) return;
|
|
254
|
-
// Skill metric
|
|
255
|
-
// no env surface to write to, so the
|
|
262
|
+
// The runner records the Skill metric into the env map. Without a
|
|
263
|
+
// runtime there is no env surface to write to, so the code simply skips
|
|
264
|
+
// the side-effect.
|
|
256
265
|
const env = this.runtime?.proc?.env ?? null;
|
|
257
266
|
if (!env) return;
|
|
258
267
|
for (const block of content) {
|
|
@@ -268,9 +277,10 @@ export class AgentRunner {
|
|
|
268
277
|
}
|
|
269
278
|
|
|
270
279
|
/**
|
|
271
|
-
* Factory function — wires real dependencies.
|
|
272
|
-
* executable for compiled fit-* binaries so the SDK doesn't fail
|
|
273
|
-
* own platform optional dependency
|
|
280
|
+
* Factory function — wires real dependencies. It resolves the native
|
|
281
|
+
* `claude` executable for compiled fit-* binaries, so the SDK doesn't fail
|
|
282
|
+
* to find its own platform optional dependency. An explicit `deps` value
|
|
283
|
+
* overrides it.
|
|
274
284
|
*/
|
|
275
285
|
export function createAgentRunner(deps) {
|
|
276
286
|
return new AgentRunner({
|
|
@@ -1,13 +1,13 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* ApmInstaller — runs `apm install --target claude` in the family root to
|
|
3
|
-
* materialise skills and agents
|
|
4
|
-
* staging directory
|
|
5
|
-
*
|
|
3
|
+
* materialise skills and agents. It copies the resulting `.claude/` into a
|
|
4
|
+
* staging directory. It computes the manifest fingerprint from the lockfile.
|
|
5
|
+
* WorkdirManager makes the per-task copy later.
|
|
6
6
|
*
|
|
7
|
-
* Subprocess and filesystem access route through the injected `runtime` bag
|
|
8
|
-
*
|
|
9
|
-
*
|
|
10
|
-
*
|
|
7
|
+
* Subprocess and filesystem access route through the injected `runtime` bag.
|
|
8
|
+
* The `apm` child streams through `runtime.subprocess.spawn`. The async
|
|
9
|
+
* staging copies use `runtime.fs`. See `createApmInstaller`, which wires the
|
|
10
|
+
* real dependencies. `installApm` is a thin free-function wrapper.
|
|
11
11
|
*/
|
|
12
12
|
|
|
13
13
|
import { createHash } from "node:crypto";
|
|
@@ -18,7 +18,7 @@ export class ApmInstaller {
|
|
|
18
18
|
/**
|
|
19
19
|
* @param {object} deps
|
|
20
20
|
* @param {import("@forwardimpact/libutil/runtime").Runtime} deps.runtime -
|
|
21
|
-
* Ambient collaborators
|
|
21
|
+
* Ambient collaborators. The installer uses `subprocess.spawn` and `fs`.
|
|
22
22
|
*/
|
|
23
23
|
constructor({ runtime }) {
|
|
24
24
|
if (!runtime) throw new Error("runtime is required");
|
|
@@ -30,8 +30,8 @@ export class ApmInstaller {
|
|
|
30
30
|
* @param {string} outputDir - The benchmark run's output directory.
|
|
31
31
|
* @param {object} [options]
|
|
32
32
|
* @param {string|null} [options.skillsFrom] - Stage `.claude/` from this
|
|
33
|
-
* directory
|
|
34
|
-
* a `.claude/` tree (e.g. a working tree)
|
|
33
|
+
* directory and do not run apm install. The path is a root that contains
|
|
34
|
+
* a `.claude/` tree (e.g. a working tree). A run can then exercise local,
|
|
35
35
|
* unpublished skills.
|
|
36
36
|
* @returns {Promise<{stagingDir: string, skillSetHash: string, judgeProfilesDir: string}>}
|
|
37
37
|
*/
|
|
@@ -44,7 +44,7 @@ export class ApmInstaller {
|
|
|
44
44
|
: join(family.rootPath, ".claude");
|
|
45
45
|
const apmYml = join(family.rootPath, "apm.yml");
|
|
46
46
|
|
|
47
|
-
// --skills-from takes precedence over apm install
|
|
47
|
+
// --skills-from takes precedence over apm install. The caller supplies
|
|
48
48
|
// the skill tree explicitly, so no remote fetch runs.
|
|
49
49
|
const hasApm =
|
|
50
50
|
!skillsFrom &&
|
|
@@ -59,7 +59,7 @@ export class ApmInstaller {
|
|
|
59
59
|
await fs.access(sourceClaude);
|
|
60
60
|
} catch {
|
|
61
61
|
throw new Error(
|
|
62
|
-
`apm install did not produce .claude/ at ${sourceClaude}
|
|
62
|
+
`apm install did not produce .claude/ at ${sourceClaude}. Check the family's apm.yml`,
|
|
63
63
|
);
|
|
64
64
|
}
|
|
65
65
|
}
|
|
@@ -78,18 +78,18 @@ export class ApmInstaller {
|
|
|
78
78
|
await fs.mkdir(stagedClaude, { recursive: true });
|
|
79
79
|
}
|
|
80
80
|
|
|
81
|
-
// apm's claude target deploys a pack's skills/ into .claude/skills
|
|
82
|
-
// never
|
|
83
|
-
//
|
|
84
|
-
// agent reference (e.g. the
|
|
85
|
-
//
|
|
86
|
-
// exist.
|
|
81
|
+
// apm's claude target deploys a pack's skills/ into .claude/skills/. It
|
|
82
|
+
// never deploys that pack's agents/ subtree (agent profiles +
|
|
83
|
+
// references). Stage that subtree from the installed apm_modules into
|
|
84
|
+
// .claude/agents/. A skill that cites an agent reference (e.g. the
|
|
85
|
+
// work-item tracker matrix) then resolves in the agent CWD. This is a
|
|
86
|
+
// no-op when --skills-from supplied a tree or no apm_modules exist.
|
|
87
87
|
if (!skillsFrom) {
|
|
88
88
|
await this.#stageApmAgents(family.rootPath, stagedClaude);
|
|
89
89
|
}
|
|
90
90
|
|
|
91
|
-
// Stage the family-local judge profile outside .claude
|
|
92
|
-
//
|
|
91
|
+
// Stage the family-local judge profile outside .claude/, so the judge can
|
|
92
|
+
// reach it. Nothing copies it into the agent-under-test's CWD.
|
|
93
93
|
const judgeSource = join(family.rootPath, "judge.md");
|
|
94
94
|
const judgeProfilesDir = join(stagingDir, "judge-profiles");
|
|
95
95
|
try {
|
|
@@ -106,7 +106,7 @@ export class ApmInstaller {
|
|
|
106
106
|
"sha256:" +
|
|
107
107
|
createHash("sha256").update(normalizeLf(lockBytes)).digest("hex");
|
|
108
108
|
} catch {
|
|
109
|
-
// No lockfile
|
|
109
|
+
// No lockfile. The family doesn't use skill packs.
|
|
110
110
|
}
|
|
111
111
|
|
|
112
112
|
return { stagingDir, skillSetHash, judgeProfilesDir };
|
|
@@ -115,8 +115,8 @@ export class ApmInstaller {
|
|
|
115
115
|
/**
|
|
116
116
|
* Merge each installed pack's `agents/` subtree (profiles + references) from
|
|
117
117
|
* `apm_modules/<owner>/<pack>/agents/` into the staged `.claude/agents/`.
|
|
118
|
-
* apm's claude target deploys `skills/` only
|
|
119
|
-
* reference a skill cites is absent from the agent CWD.
|
|
118
|
+
* apm's claude target deploys `skills/` only. Without this merge, an agent
|
|
119
|
+
* reference that a skill cites is absent from the agent CWD.
|
|
120
120
|
* @param {string} familyRoot
|
|
121
121
|
* @param {string} stagedClaude
|
|
122
122
|
*/
|
|
@@ -127,7 +127,7 @@ export class ApmInstaller {
|
|
|
127
127
|
try {
|
|
128
128
|
owners = await fs.readdir(modulesRoot, { withFileTypes: true });
|
|
129
129
|
} catch {
|
|
130
|
-
return; // no apm_modules
|
|
130
|
+
return; // no apm_modules, so nothing to stage
|
|
131
131
|
}
|
|
132
132
|
const stagedAgents = join(stagedClaude, "agents");
|
|
133
133
|
for (const owner of owners) {
|
|
@@ -159,8 +159,8 @@ export class ApmInstaller {
|
|
|
159
159
|
["install", "--target", "claude"],
|
|
160
160
|
{ cwd, stdio: ["ignore", "pipe", "pipe"] },
|
|
161
161
|
);
|
|
162
|
-
// Drain stdout concurrently so the child never blocks on backpressure
|
|
163
|
-
//
|
|
162
|
+
// Drain stdout concurrently so the child never blocks on backpressure.
|
|
163
|
+
// Capture stderr for the failure message.
|
|
164
164
|
let stderr = "";
|
|
165
165
|
const drainStdout = (async () => {
|
|
166
166
|
for await (const _chunk of child.stdout) {
|
|
@@ -199,8 +199,8 @@ export function createApmInstaller(deps) {
|
|
|
199
199
|
* @param {import("./task-family.js").TaskFamily} family
|
|
200
200
|
* @param {string} outputDir
|
|
201
201
|
* @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
|
|
202
|
-
* @param {object} [options] -
|
|
203
|
-
* `{ skillsFrom }`).
|
|
202
|
+
* @param {object} [options] - This function forwards these to
|
|
203
|
+
* `ApmInstaller.install` (e.g. `{ skillsFrom }`).
|
|
204
204
|
*/
|
|
205
205
|
export function installApm(family, outputDir, runtime, options = {}) {
|
|
206
206
|
return new ApmInstaller({ runtime }).install(family, outputDir, options);
|
|
@@ -1,17 +1,20 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* Env-loader — auto-discover `.env` / `.env.local` files in a task family
|
|
3
|
-
* and its tasks
|
|
3
|
+
* and its tasks. Load them into `process.env`. Render the merged result
|
|
4
4
|
* into each agent CWD.
|
|
5
5
|
*
|
|
6
|
-
*
|
|
7
|
-
*
|
|
6
|
+
* The loader reads the discovery paths in this order. The first value per
|
|
7
|
+
* key wins:
|
|
8
|
+
* 1. process.env (CI secrets and shell env, which the loader never
|
|
9
|
+
* overwrites)
|
|
8
10
|
* 2. <family>/.env.local
|
|
9
11
|
* 3. <family>/.env
|
|
10
12
|
* 4. tasks/<id>/.env.local
|
|
11
13
|
* 5. tasks/<id>/.env
|
|
12
14
|
*
|
|
13
|
-
*
|
|
14
|
-
*
|
|
15
|
+
* The loader loads every discovered env file, family or task, into
|
|
16
|
+
* process.env. It also renders the file (with resolved values) into the
|
|
17
|
+
* agent working directory.
|
|
15
18
|
*/
|
|
16
19
|
|
|
17
20
|
import { join } from "node:path";
|
|
@@ -20,7 +23,8 @@ const ENV_FILES = [".env.local", ".env"];
|
|
|
20
23
|
|
|
21
24
|
/**
|
|
22
25
|
* Parse a `.env` file into an array of {key, value} pairs.
|
|
23
|
-
*
|
|
26
|
+
* It handles KEY=VALUE, # comments, blank lines, and single/double-quoted
|
|
27
|
+
* values.
|
|
24
28
|
* @param {string} content
|
|
25
29
|
* @returns {Array<{key: string, value: string}>}
|
|
26
30
|
*/
|
|
@@ -46,7 +50,7 @@ export function parseEnvFile(content) {
|
|
|
46
50
|
}
|
|
47
51
|
|
|
48
52
|
/**
|
|
49
|
-
* Read and parse an env file
|
|
53
|
+
* Read and parse an env file. Return [] if the file does not exist.
|
|
50
54
|
* @param {object} fs - Async filesystem surface (`runtime.fs`).
|
|
51
55
|
* @param {string} filePath
|
|
52
56
|
* @returns {Promise<Array<{key: string, value: string}>>}
|
|
@@ -62,10 +66,11 @@ async function readEnvFile(fs, filePath) {
|
|
|
62
66
|
}
|
|
63
67
|
|
|
64
68
|
/**
|
|
65
|
-
* Load entries into the process env map.
|
|
69
|
+
* Load entries into the process env map. This function never overwrites an
|
|
70
|
+
* existing key.
|
|
66
71
|
* @param {Record<string, string|undefined>} env - The `runtime.proc.env` map.
|
|
67
72
|
* @param {Array<{key: string, value: string}>} entries
|
|
68
|
-
* @returns {string[]} var names
|
|
73
|
+
* @returns {string[]} the var names it loaded
|
|
69
74
|
*/
|
|
70
75
|
function applyToProcessEnv(env, entries) {
|
|
71
76
|
const names = [];
|
|
@@ -79,7 +84,8 @@ function applyToProcessEnv(env, entries) {
|
|
|
79
84
|
}
|
|
80
85
|
|
|
81
86
|
/**
|
|
82
|
-
* Load one env file
|
|
87
|
+
* Load one env file. Apply it to the env map. Record the keys in the merged
|
|
88
|
+
* map.
|
|
83
89
|
* @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
|
|
84
90
|
* @param {string} dir
|
|
85
91
|
* @param {string} file
|
|
@@ -100,7 +106,7 @@ async function loadOneEnvFile(runtime, dir, file, names, merged) {
|
|
|
100
106
|
}
|
|
101
107
|
|
|
102
108
|
/**
|
|
103
|
-
* Scan directories for env files
|
|
109
|
+
* Scan directories for env files. Load them into the env map. Collect
|
|
104
110
|
* a merged key manifest per filename.
|
|
105
111
|
* @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
|
|
106
112
|
* @param {string[]} dirs
|
|
@@ -118,7 +124,7 @@ async function collectEnvEntries(runtime, dirs) {
|
|
|
118
124
|
}
|
|
119
125
|
|
|
120
126
|
/**
|
|
121
|
-
* Write resolved env files into the agent CWD
|
|
127
|
+
* Write resolved env files into the agent CWD. Warn about empty values.
|
|
122
128
|
* @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
|
|
123
129
|
* @param {Map<string, Map<string, true>>} merged
|
|
124
130
|
* @param {string} agentCwd
|
|
@@ -142,14 +148,16 @@ async function renderEnvFiles(runtime, merged, agentCwd) {
|
|
|
142
148
|
}
|
|
143
149
|
|
|
144
150
|
/**
|
|
145
|
-
* Discover `.env` / `.env.local` in one or more directories
|
|
146
|
-
* into the process env map
|
|
151
|
+
* Discover `.env` / `.env.local` in one or more directories. Load them
|
|
152
|
+
* into the process env map. Render the resolved values into the agent CWD.
|
|
147
153
|
*
|
|
148
154
|
* @param {string[]} dirs - Directories to scan (family root, task dir, etc.)
|
|
149
155
|
* @param {string} agentCwd - Agent working directory to render into.
|
|
150
156
|
* @param {import("@forwardimpact/libutil/runtime").Runtime} runtime - Ambient
|
|
151
|
-
* collaborators
|
|
152
|
-
*
|
|
157
|
+
* collaborators. It uses `fs` (async read/write), `proc.env`, and
|
|
158
|
+
* `proc.stderr`.
|
|
159
|
+
* @returns {Promise<string[]>} Every var name the loader discovered (for
|
|
160
|
+
* redaction).
|
|
153
161
|
*/
|
|
154
162
|
export async function loadEnv(dirs, agentCwd, runtime) {
|
|
155
163
|
const { names, merged } = await collectEnvEntries(runtime, dirs);
|
package/src/benchmark/grade.js
CHANGED
|
@@ -1,46 +1,50 @@
|
|
|
1
1
|
/**
|
|
2
|
-
*
|
|
2
|
+
* Grade derivation — the sole home of the check-row arithmetic.
|
|
3
3
|
*
|
|
4
|
-
* Check rows are the
|
|
5
|
-
* check by default
|
|
6
|
-
* order:
|
|
4
|
+
* Check rows are the one authoritative input to the grade. Every row is a
|
|
5
|
+
* check by default. A row declares its role with its own fields. The
|
|
6
|
+
* classifier checks the roles in this order:
|
|
7
7
|
*
|
|
8
8
|
* 1. Gate — `gate` is exactly `true`, `pass` is boolean, and no
|
|
9
|
-
* `weight` key is present.
|
|
10
|
-
* false.
|
|
11
|
-
* 2. Diagnostic — no `gate` key and `weight` is exactly `0`.
|
|
12
|
-
* never
|
|
9
|
+
* `weight` key is present. A gate that fails sets
|
|
10
|
+
* `gatesPass` to false.
|
|
11
|
+
* 2. Diagnostic — no `gate` key and `weight` is exactly `0`. The row is
|
|
12
|
+
* free-form. The grader never scores it.
|
|
13
13
|
* 3. Scored — no `gate` key, boolean `pass`, `weight` absent (defaults
|
|
14
14
|
* to 1) or finite > 0.
|
|
15
|
-
* 4. Malformed — everything else: any `gate`+`weight` co-occurrence
|
|
16
|
-
* stray weight must never silently disarm a gate), a
|
|
15
|
+
* 4. Malformed — everything else: any `gate`+`weight` co-occurrence, a
|
|
17
16
|
* non-boolean `gate`, a missing or non-boolean `pass` on a
|
|
18
17
|
* graded row, an invalid `weight`, an fd-3 line that failed
|
|
19
|
-
* to parse, a non-object row.
|
|
20
|
-
*
|
|
21
|
-
*
|
|
18
|
+
* to parse, a non-object row. A stray weight must never
|
|
19
|
+
* silently disarm a gate. A malformed row counts as a
|
|
20
|
+
* **scored check that fails**. If the grader dropped the
|
|
21
|
+
* defect, the cell could mint full marks. If it failed the
|
|
22
|
+
* whole run, it would zero completed work.
|
|
22
23
|
*
|
|
23
|
-
* The producers' `source` stamp is display metadata
|
|
24
|
+
* The producers' `source` stamp is display metadata. The grader never reads
|
|
25
|
+
* it.
|
|
24
26
|
*/
|
|
25
27
|
|
|
26
28
|
/**
|
|
27
29
|
* @typedef {object} GradeResult
|
|
28
30
|
* @property {"pass" | "fail"} verdict - `healthy ∧ gatesPass ∧ fullMarks`.
|
|
29
31
|
* @property {boolean} gatesPass - Every gate row passes (vacuously true).
|
|
30
|
-
* @property {number | null} score - Weighted fraction of
|
|
31
|
-
*
|
|
32
|
-
*
|
|
33
|
-
*
|
|
34
|
-
*
|
|
32
|
+
* @property {number | null} score - Weighted fraction of the scored checks
|
|
33
|
+
* that pass. It is `null` when the cell has zero scored checks (binary
|
|
34
|
+
* task).
|
|
35
|
+
* @property {boolean} fullMarks - An integer count predicate. It is true when
|
|
36
|
+
* no row is malformed and every scored check passes. It is never a float
|
|
37
|
+
* comparison, so fractional weights carry no equality hazard. It is
|
|
38
|
+
* vacuously true with zero scored checks.
|
|
35
39
|
* @property {number} malformed - Malformed row count.
|
|
36
40
|
*/
|
|
37
41
|
|
|
38
42
|
/**
|
|
39
43
|
* Grade the merged check rows against grader health.
|
|
40
44
|
*
|
|
41
|
-
* `healthy` is the completion signal a crashed grader cannot fake
|
|
42
|
-
* false the verdict is `fail` whatever the rows say
|
|
43
|
-
*
|
|
45
|
+
* `healthy` is the completion signal a crashed grader cannot fake. When it is
|
|
46
|
+
* false, the verdict is `fail` whatever the rows say. A hook that emits rows
|
|
47
|
+
* that pass and then dies can never mint marks.
|
|
44
48
|
* @param {unknown[]} details - Merged check rows from both producers.
|
|
45
49
|
* @param {boolean} healthy - Invariants exited 0 AND the hidden-test engine
|
|
46
50
|
* did not throw.
|
|
@@ -73,7 +77,7 @@ export function gradeChecks(details, healthy) {
|
|
|
73
77
|
}
|
|
74
78
|
|
|
75
79
|
/**
|
|
76
|
-
* Fold one row into the
|
|
80
|
+
* Fold one row into the tally per its classified role.
|
|
77
81
|
* @param {{gatesPass: boolean, malformed: number, scored: number, passing: number, weightAll: number, weightPassing: number}} tally
|
|
78
82
|
* @param {unknown} row
|
|
79
83
|
*/
|
|
@@ -96,11 +100,10 @@ function tallyRow(tally, row) {
|
|
|
96
100
|
}
|
|
97
101
|
|
|
98
102
|
/**
|
|
99
|
-
* Run both check-row producers and grade the merged rows
|
|
100
|
-
*
|
|
101
|
-
*
|
|
102
|
-
*
|
|
103
|
-
* happened to emit first.
|
|
103
|
+
* Run both check-row producers and grade the merged rows. The runner and the
|
|
104
|
+
* `grade` subcommand share this one composition. An engine throw is grader
|
|
105
|
+
* fault. Its message lands on the returned `engineError` and health fails. A
|
|
106
|
+
* crashed grader can never mint marks from rows it emitted first.
|
|
104
107
|
* @param {import("./task-family.js").Task} task
|
|
105
108
|
* @param {{cwd: string, port: number, runDir: string, familyDir?: string|null}} ctx
|
|
106
109
|
* @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
|
|
@@ -126,8 +129,8 @@ export async function runProducersAndGrade(task, ctx, runtime, producers) {
|
|
|
126
129
|
|
|
127
130
|
/**
|
|
128
131
|
* Merge the two producers' rows (invariants first) and stamp each row's
|
|
129
|
-
* provenance. The stamp is display metadata
|
|
130
|
-
*
|
|
132
|
+
* provenance. The stamp is display metadata. The grader never reads it.
|
|
133
|
+
* Non-object rows (malformed by contract) pass through verbatim.
|
|
131
134
|
* @param {unknown[]} invariantsDetails
|
|
132
135
|
* @param {unknown[]} hiddenDetails
|
|
133
136
|
* @returns {unknown[]}
|
|
@@ -147,9 +150,9 @@ function stampSource(row, source) {
|
|
|
147
150
|
}
|
|
148
151
|
|
|
149
152
|
/**
|
|
150
|
-
* Project the raw `gradeChecks` return onto the record schema
|
|
151
|
-
*
|
|
152
|
-
* `
|
|
153
|
+
* Project the raw `gradeChecks` return onto the record schema. `fullMarks` is
|
|
154
|
+
* derivable, so this function drops it. It omits `score` on binary tasks
|
|
155
|
+
* (`null`). It omits `malformed` when the rows are clean.
|
|
153
156
|
* @param {GradeResult} raw
|
|
154
157
|
* @returns {{verdict: "pass"|"fail", gatesPass: boolean, score?: number, malformed?: number}}
|
|
155
158
|
*/
|
|
@@ -177,9 +180,9 @@ function classifyRow(row) {
|
|
|
177
180
|
}
|
|
178
181
|
|
|
179
182
|
/**
|
|
180
|
-
*
|
|
181
|
-
* `pass` and no `weight` key
|
|
182
|
-
* stray weight can never silently disarm a gate.
|
|
183
|
+
* Classify a row that has a `gate` key. The row is valid only as `gate: true`
|
|
184
|
+
* with a boolean `pass` and no `weight` key. Any weight that co-occurs makes
|
|
185
|
+
* the row malformed. A stray weight can never silently disarm a gate.
|
|
183
186
|
* @param {object} row
|
|
184
187
|
* @returns {"gate" | "malformed"}
|
|
185
188
|
*/
|
|
@@ -191,9 +194,9 @@ function classifyGateRow(row) {
|
|
|
191
194
|
}
|
|
192
195
|
|
|
193
196
|
/**
|
|
194
|
-
*
|
|
195
|
-
* finite positive weight with a boolean `pass` is scored
|
|
196
|
-
* malformed.
|
|
197
|
+
* Classify a gate-less row that has a `weight` key. A weight of exactly 0 is
|
|
198
|
+
* a diagnostic. A finite positive weight with a boolean `pass` is scored.
|
|
199
|
+
* Anything else is malformed.
|
|
197
200
|
* @param {object} row
|
|
198
201
|
* @returns {"diagnostic" | "scored" | "malformed"}
|
|
199
202
|
*/
|
|
@@ -205,8 +208,8 @@ function classifyWeightedRow(row) {
|
|
|
205
208
|
}
|
|
206
209
|
|
|
207
210
|
/**
|
|
208
|
-
* A malformed row fails at its own weight when
|
|
209
|
-
*
|
|
211
|
+
* A malformed row fails at its own weight when that weight is valid and
|
|
212
|
+
* positive. Otherwise it fails at unit weight 1.
|
|
210
213
|
* @param {unknown} row
|
|
211
214
|
* @returns {number}
|
|
212
215
|
*/
|