@forwardimpact/libharness 2.0.0 → 3.0.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (73) hide show
  1. package/README.md +68 -65
  2. package/package.json +15 -13
  3. package/src/advisor.js +47 -41
  4. package/src/agent-runner.js +58 -48
  5. package/src/benchmark/apm-installer.js +28 -28
  6. package/src/benchmark/env-loader.js +24 -16
  7. package/src/benchmark/grade.js +44 -41
  8. package/src/benchmark/hidden-tests.js +25 -24
  9. package/src/benchmark/hook-env.js +11 -9
  10. package/src/benchmark/invariants.js +20 -17
  11. package/src/benchmark/judge.js +29 -28
  12. package/src/benchmark/npm-installer.js +9 -8
  13. package/src/benchmark/report.js +53 -50
  14. package/src/benchmark/result.js +24 -23
  15. package/src/benchmark/runner.js +75 -69
  16. package/src/benchmark/scheduler.js +17 -16
  17. package/src/benchmark/task-family.js +29 -27
  18. package/src/benchmark/trace-split.js +9 -8
  19. package/src/benchmark/workdir.js +27 -25
  20. package/src/claude-code-executable.js +11 -11
  21. package/src/commands/advisor-flags.js +8 -7
  22. package/src/commands/assert.js +16 -15
  23. package/src/commands/benchmark-definition.js +20 -20
  24. package/src/commands/benchmark-grade.js +13 -12
  25. package/src/commands/benchmark-report.js +5 -5
  26. package/src/commands/benchmark-run.js +31 -28
  27. package/src/commands/by-discussion.js +11 -11
  28. package/src/commands/callback.js +11 -11
  29. package/src/commands/discuss.js +8 -7
  30. package/src/commands/facilitate.js +16 -14
  31. package/src/commands/output.js +4 -3
  32. package/src/commands/run.js +15 -15
  33. package/src/commands/scan-logs.js +22 -20
  34. package/src/commands/selfedit.js +124 -0
  35. package/src/commands/supervise.js +13 -11
  36. package/src/commands/task-input.js +9 -9
  37. package/src/commands/tee.js +11 -10
  38. package/src/commands/trace.js +55 -42
  39. package/src/commands/work-tracker.js +4 -3
  40. package/src/cost.js +17 -17
  41. package/src/discuss-tools.js +16 -16
  42. package/src/discusser.js +39 -38
  43. package/src/events/github.js +54 -37
  44. package/src/facilitator.js +21 -21
  45. package/src/inbox-poller.js +4 -4
  46. package/src/judge.js +32 -30
  47. package/src/message-bus.js +12 -11
  48. package/src/orchestration-loop.js +35 -36
  49. package/src/orchestration-toolkit.js +58 -53
  50. package/src/orchestrator-helpers.js +2 -2
  51. package/src/profile-prompt.js +54 -53
  52. package/src/redaction.js +63 -57
  53. package/src/render/line-renderer.js +5 -5
  54. package/src/render/orchestrator-filter.js +3 -3
  55. package/src/render/palette.js +11 -9
  56. package/src/render/tool-hints.js +18 -15
  57. package/src/render/turn-renderer.js +4 -4
  58. package/src/reply-emitter.js +2 -2
  59. package/src/sequence-counter.js +4 -3
  60. package/src/signature-filter.js +7 -6
  61. package/src/supervisor.js +19 -18
  62. package/src/tee-writer.js +25 -25
  63. package/src/trace-collector.js +53 -48
  64. package/src/trace-github.js +53 -44
  65. package/src/trace-multi.js +16 -14
  66. package/src/trace-query.js +61 -52
  67. package/src/trace-render.js +19 -19
  68. package/src/trace-usage.js +31 -28
  69. package/src/transcript-recorder.js +24 -20
  70. package/bin/fit-benchmark.js +0 -44
  71. package/bin/fit-harness.js +0 -412
  72. package/bin/fit-selfedit.js +0 -165
  73. package/bin/fit-trace.js +0 -520
@@ -1,7 +1,8 @@
1
1
  /**
2
2
  * AgentRunner — runs a single Claude Agent SDK session and emits raw
3
- * NDJSON events to an output stream. Building block for `fit-harness run`,
4
- * `fit-harness supervise`, `fit-harness facilitate`, and `fit-harness discuss`.
3
+ * NDJSON events to an output stream. `gemba-harness run`,
4
+ * `gemba-harness supervise`, `gemba-harness facilitate`, and
5
+ * `gemba-harness discuss` build on it.
5
6
  *
6
7
  * Follows OO+DI: constructor injection, factory function, tests bypass factory.
7
8
  */
@@ -12,15 +13,16 @@ import { resolveClaudeCodeExecutable } from "./claude-code-executable.js";
12
13
  const DEFAULT_ALLOWED_TOOLS = ["Bash", "Read", "Glob", "Grep", "Write", "Edit"];
13
14
 
14
15
  /**
15
- * Did the session actually invoke the model? A genuine run always bills
16
+ * Report whether the session invoked the model. A genuine run always bills
16
17
  * tokens (the system prompt alone is thousands of input tokens) and costs
17
- * more than zero. A `result` message with `subtype: "success"` but zero
18
- * token usage and zero cost means the model was never reached — the
19
- * canonical signature of a Claude Code init/auth failure (e.g. an invalid
20
- * `ANTHROPIC_API_KEY`), which the SDK otherwise reports as a clean success.
18
+ * more than zero. A `result` message can carry `subtype: "success"` with
19
+ * zero token usage and zero cost. That combination means the run never
20
+ * reached the model. It is the canonical signature of a Claude Code init or
21
+ * auth failure (e.g. an invalid `ANTHROPIC_API_KEY`). The SDK otherwise
22
+ * reports that failure as a clean success.
21
23
  *
22
24
  * If the SDK gave us neither a `usage` object nor `total_cost_usd`, don't
23
- * second-guess the subtype — trust the reported success.
25
+ * second-guess the subtype. Trust the reported success.
24
26
  * @param {object|null} result - The SDK `result` message, or null.
25
27
  * @returns {boolean}
26
28
  */
@@ -37,9 +39,10 @@ function modelDidWork(result) {
37
39
  return tokens > 0 || (cost ?? 0) > 0;
38
40
  }
39
41
 
40
- // fit-harness and kata-action run headless in CI/CD with no human to answer
41
- // permission prompts. The SDK is always launched in bypass mode — not
42
- // overridable — so a future caller can't accidentally reduce permissions.
42
+ // gemba-harness and kata-action run headless in CI/CD with no human to answer
43
+ // permission prompts. The runner always launches the SDK in bypass mode. No
44
+ // caller can override that mode, so a future caller can't accidentally
45
+ // reduce permissions.
43
46
  const PERMISSION_MODE = "bypassPermissions";
44
47
 
45
48
  /** Run a single Claude Agent SDK session and emit raw NDJSON events to an output stream. */
@@ -47,25 +50,27 @@ export class AgentRunner {
47
50
  /**
48
51
  * @param {object} deps
49
52
  * @param {string} deps.cwd - Agent working directory
50
- * @param {function} deps.query - SDK query function (injected for testing)
53
+ * @param {function} deps.query - SDK query function (tests inject it)
51
54
  * @param {import("stream").Writable} deps.output - Stream to emit NDJSON to
52
55
  * @param {string} [deps.model] - Claude model identifier
53
- * @param {number} [deps.maxTurns] - Maximum agentic turns; 0 means unlimited
56
+ * @param {number} [deps.maxTurns] - Maximum agentic turns. 0 means unlimited
54
57
  * @param {string[]} [deps.allowedTools] - Tools the agent may use
55
- * @param {function} [deps.onLine] - Callback invoked with each NDJSON line as it's produced
56
- * @param {function} [deps.onPrompt] - Callback invoked with the effective (amend-applied) prompt of each run/resume
58
+ * @param {function} [deps.onLine] - Callback that receives each NDJSON line as the runner produces it
59
+ * @param {function} [deps.onPrompt] - Callback that receives the effective (amend-applied) prompt of each run/resume
57
60
  * @param {string[]} [deps.settingSources] - SDK setting sources (e.g. ['project'] to load CLAUDE.md)
58
- * @param {string|object} [deps.systemPrompt] - SDK system prompt (string replaces default; {type:'preset', preset:'claude_code', append} appends)
61
+ * @param {string|object} [deps.systemPrompt] - SDK system prompt. A string replaces the default. The preset form {type:'preset', preset:'claude_code', append} appends
59
62
  * @param {string[]} [deps.disallowedTools] - Tools to explicitly remove from the model's context
60
63
  * @param {Record<string, object>} [deps.mcpServers] - MCP server configs to pass to the SDK query
61
64
  * @param {string} [deps.pathToClaudeCodeExecutable] - Absolute path to the
62
- * native `claude` CLI the SDK should spawn. Set for compiled fit-* binaries,
63
- * which can't self-resolve the SDK's platform optional dependency; omitted
64
- * from source runs so the SDK resolves its own version-matched binary.
65
+ * native `claude` CLI the SDK should spawn. Set it for compiled fit-*
66
+ * binaries, which can't self-resolve the SDK's platform optional
67
+ * dependency. Omit it from source runs so the SDK resolves its own
68
+ * version-matched binary.
65
69
  * @param {object} deps.redactor
66
70
  * @param {import("@forwardimpact/libutil/runtime").Runtime} [deps.runtime] -
67
- * Ambient collaborators. Only `proc.env` is read (to record Skill
68
- * invocations into `LIBHARNESS_SKILL`); when absent the write is skipped.
71
+ * Ambient collaborators. The runner reads only `proc.env`, to record Skill
72
+ * invocations into `LIBHARNESS_SKILL`. When `runtime` is absent, the
73
+ * runner skips the write.
69
74
  */
70
75
  constructor(deps) {
71
76
  if (!deps.cwd) throw new Error("cwd is required");
@@ -81,15 +86,17 @@ export class AgentRunner {
81
86
  this.maxTurns = deps.maxTurns ?? 50;
82
87
  this.allowedTools = deps.allowedTools ?? DEFAULT_ALLOWED_TOOLS;
83
88
  this.onLine = deps.onLine ?? null;
84
- // Optional; read only through a truthy guard in run()/resume(), so an
85
- // absent value stays undefined rather than needing a `?? null` default.
89
+ // Optional. The code reads it only through a truthy guard in
90
+ // run()/resume(), so an absent value stays undefined and needs no
91
+ // `?? null` default.
86
92
  this.onPrompt = deps.onPrompt;
87
93
  this.settingSources = deps.settingSources ?? [];
88
94
  this.systemPrompt = deps.systemPrompt ?? null;
89
95
  this.disallowedTools = deps.disallowedTools ?? [];
90
96
  this.mcpServers = deps.mcpServers ?? null;
91
- // Optional; read only through a truthy guard in #callOptions, so an absent
92
- // value stays undefined rather than needing a `?? null` default.
97
+ // Optional. The code reads it only through a truthy guard in
98
+ // #callOptions, so an absent value stays undefined and needs no
99
+ // `?? null` default.
93
100
  this.pathToClaudeCodeExecutable = deps.pathToClaudeCodeExecutable;
94
101
  this.taskAmend = deps.taskAmend ?? null;
95
102
  this.sessionId = null;
@@ -146,16 +153,17 @@ export class AgentRunner {
146
153
  }
147
154
 
148
155
  /**
149
- * Build the options passed to every SDK query() call. Shared by run()
150
- * and resume() so the agent's configuration — cwd, tools, prompt,
151
- * setting sources, turn budget — is identical across the session's
152
- * lifetime. Only resume() layers `resume: this.sessionId` on top.
156
+ * Build the options for every SDK query() call. run() and resume()
157
+ * share this method, so the agent's configuration stays identical
158
+ * across the session's lifetime. That configuration is the cwd, the
159
+ * tools, the prompt, the setting sources, and the turn budget. Only
160
+ * resume() layers `resume: this.sessionId` on top.
153
161
  *
154
- * SDK options are call-attached, not session-attached: the resumed
155
- * call loads the prior conversation but otherwise uses whatever
156
- * options this call passes. Omitting tool/prompt/setting options on
157
- * resume causes the agent to silently lose its restrictions and
158
- * persona between turns.
162
+ * SDK options attach to the call. They do not attach to the session.
163
+ * The resumed call loads the prior conversation. It otherwise uses
164
+ * whatever options this call passes. If you omit the tool, prompt, or
165
+ * setting options on resume, the agent silently loses its restrictions
166
+ * and persona between turns.
159
167
  */
160
168
  #callOptions(abortController) {
161
169
  return {
@@ -179,14 +187,14 @@ export class AgentRunner {
179
187
  }
180
188
 
181
189
  /**
182
- * Iterate the SDK query iterator, mirroring every message to the
183
- * output stream and the `onLine` callback. Captures `sessionId` from
184
- * the SDK's `system/init` message and tracks Skill invocations into
190
+ * Iterate the SDK query iterator. Mirror every message to the output
191
+ * stream and to the `onLine` callback. Capture `sessionId` from the
192
+ * SDK's `system/init` message. Track Skill invocations into
185
193
  * `LIBHARNESS_SKILL` for downstream metrics.
186
194
  *
187
195
  * If the iterator throws and we triggered the abort ourselves
188
196
  * (`currentAbortController.signal.aborted`), we report `aborted:
189
- * true`; otherwise the error propagates as `error`.
197
+ * true`. Otherwise the error propagates as `error`.
190
198
  */
191
199
  async #consumeQuery(iterator) {
192
200
  let text = "";
@@ -212,10 +220,10 @@ export class AgentRunner {
212
220
  }
213
221
  }
214
222
 
215
- // A "success" subtype is necessary but not sufficient: the SDK reports a
216
- // failed init (e.g. an invalid API key) as success with zero model work.
217
- // Require evidence the model actually ran, and surface a clear error when
218
- // it didn't, so the masked failure can't be reported as a green run.
223
+ // A "success" subtype is necessary. It is not sufficient. The SDK reports
224
+ // a failed init (e.g. an invalid API key) as success with zero model work.
225
+ // Require evidence that the model ran. Surface a clear error when it did
226
+ // not, so nobody reports the masked failure as a green run.
219
227
  const reportedSuccess = stopReason === "success";
220
228
  const success =
221
229
  reportedSuccess &&
@@ -223,7 +231,7 @@ export class AgentRunner {
223
231
  modelDidWork(resultMessage);
224
232
  if (reportedSuccess && !success && !error) {
225
233
  error = new Error(
226
- "agent reported success but performed no model work (zero token usage) — likely a Claude Code init or authentication failure",
234
+ "agent reported success but did no model work (zero token usage), which is likely a Claude Code init or authentication failure",
227
235
  );
228
236
  }
229
237
 
@@ -251,8 +259,9 @@ export class AgentRunner {
251
259
  #trackSkillInvocation(message) {
252
260
  const content = message.message?.content ?? message.content;
253
261
  if (!Array.isArray(content)) return;
254
- // Skill metric is recorded into the env map; without a runtime there is
255
- // no env surface to write to, so the side-effect is simply skipped.
262
+ // The runner records the Skill metric into the env map. Without a
263
+ // runtime there is no env surface to write to, so the code simply skips
264
+ // the side-effect.
256
265
  const env = this.runtime?.proc?.env ?? null;
257
266
  if (!env) return;
258
267
  for (const block of content) {
@@ -268,9 +277,10 @@ export class AgentRunner {
268
277
  }
269
278
 
270
279
  /**
271
- * Factory function — wires real dependencies. Resolves the native `claude`
272
- * executable for compiled fit-* binaries so the SDK doesn't fail to find its
273
- * own platform optional dependency; an explicit `deps` value overrides it.
280
+ * Factory function — wires real dependencies. It resolves the native
281
+ * `claude` executable for compiled fit-* binaries, so the SDK doesn't fail
282
+ * to find its own platform optional dependency. An explicit `deps` value
283
+ * overrides it.
274
284
  */
275
285
  export function createAgentRunner(deps) {
276
286
  return new AgentRunner({
@@ -1,13 +1,13 @@
1
1
  /**
2
2
  * ApmInstaller — runs `apm install --target claude` in the family root to
3
- * materialise skills and agents, copies the resulting `.claude/` into a
4
- * staging directory, and computes the manifest fingerprint from the lockfile.
5
- * Per-task copy happens later in WorkdirManager.
3
+ * materialise skills and agents. It copies the resulting `.claude/` into a
4
+ * staging directory. It computes the manifest fingerprint from the lockfile.
5
+ * WorkdirManager makes the per-task copy later.
6
6
  *
7
- * Subprocess and filesystem access route through the injected `runtime` bag
8
- * (`runtime.subprocess.spawn` for the streaming `apm` child, `runtime.fs` for
9
- * the async staging copies). See `createApmInstaller` for the real-dependency
10
- * wiring; `installApm` is a thin free-function wrapper.
7
+ * Subprocess and filesystem access route through the injected `runtime` bag.
8
+ * The `apm` child streams through `runtime.subprocess.spawn`. The async
9
+ * staging copies use `runtime.fs`. See `createApmInstaller`, which wires the
10
+ * real dependencies. `installApm` is a thin free-function wrapper.
11
11
  */
12
12
 
13
13
  import { createHash } from "node:crypto";
@@ -18,7 +18,7 @@ export class ApmInstaller {
18
18
  /**
19
19
  * @param {object} deps
20
20
  * @param {import("@forwardimpact/libutil/runtime").Runtime} deps.runtime -
21
- * Ambient collaborators; uses `subprocess.spawn` and `fs`.
21
+ * Ambient collaborators. The installer uses `subprocess.spawn` and `fs`.
22
22
  */
23
23
  constructor({ runtime }) {
24
24
  if (!runtime) throw new Error("runtime is required");
@@ -30,8 +30,8 @@ export class ApmInstaller {
30
30
  * @param {string} outputDir - The benchmark run's output directory.
31
31
  * @param {object} [options]
32
32
  * @param {string|null} [options.skillsFrom] - Stage `.claude/` from this
33
- * directory instead of running apm install. The path is a root containing
34
- * a `.claude/` tree (e.g. a working tree), letting a run exercise local,
33
+ * directory and do not run apm install. The path is a root that contains
34
+ * a `.claude/` tree (e.g. a working tree). A run can then exercise local,
35
35
  * unpublished skills.
36
36
  * @returns {Promise<{stagingDir: string, skillSetHash: string, judgeProfilesDir: string}>}
37
37
  */
@@ -44,7 +44,7 @@ export class ApmInstaller {
44
44
  : join(family.rootPath, ".claude");
45
45
  const apmYml = join(family.rootPath, "apm.yml");
46
46
 
47
- // --skills-from takes precedence over apm install: the caller is supplying
47
+ // --skills-from takes precedence over apm install. The caller supplies
48
48
  // the skill tree explicitly, so no remote fetch runs.
49
49
  const hasApm =
50
50
  !skillsFrom &&
@@ -59,7 +59,7 @@ export class ApmInstaller {
59
59
  await fs.access(sourceClaude);
60
60
  } catch {
61
61
  throw new Error(
62
- `apm install did not produce .claude/ at ${sourceClaude}; check the family's apm.yml`,
62
+ `apm install did not produce .claude/ at ${sourceClaude}. Check the family's apm.yml`,
63
63
  );
64
64
  }
65
65
  }
@@ -78,18 +78,18 @@ export class ApmInstaller {
78
78
  await fs.mkdir(stagedClaude, { recursive: true });
79
79
  }
80
80
 
81
- // apm's claude target deploys a pack's skills/ into .claude/skills/ but
82
- // never its agents/ subtree (agent profiles + references). Stage that from
83
- // the installed apm_modules into .claude/agents/ so a skill that cites an
84
- // agent reference (e.g. the work-item tracker matrix) resolves in the
85
- // agent CWD. No-op when --skills-from supplied a tree or no apm_modules
86
- // exist.
81
+ // apm's claude target deploys a pack's skills/ into .claude/skills/. It
82
+ // never deploys that pack's agents/ subtree (agent profiles +
83
+ // references). Stage that subtree from the installed apm_modules into
84
+ // .claude/agents/. A skill that cites an agent reference (e.g. the
85
+ // work-item tracker matrix) then resolves in the agent CWD. This is a
86
+ // no-op when --skills-from supplied a tree or no apm_modules exist.
87
87
  if (!skillsFrom) {
88
88
  await this.#stageApmAgents(family.rootPath, stagedClaude);
89
89
  }
90
90
 
91
- // Stage the family-local judge profile outside .claude/ so it is available
92
- // to the judge but never copied into the agent-under-test's CWD.
91
+ // Stage the family-local judge profile outside .claude/, so the judge can
92
+ // reach it. Nothing copies it into the agent-under-test's CWD.
93
93
  const judgeSource = join(family.rootPath, "judge.md");
94
94
  const judgeProfilesDir = join(stagingDir, "judge-profiles");
95
95
  try {
@@ -106,7 +106,7 @@ export class ApmInstaller {
106
106
  "sha256:" +
107
107
  createHash("sha256").update(normalizeLf(lockBytes)).digest("hex");
108
108
  } catch {
109
- // No lockfile — family doesn't use skill packs.
109
+ // No lockfile. The family doesn't use skill packs.
110
110
  }
111
111
 
112
112
  return { stagingDir, skillSetHash, judgeProfilesDir };
@@ -115,8 +115,8 @@ export class ApmInstaller {
115
115
  /**
116
116
  * Merge each installed pack's `agents/` subtree (profiles + references) from
117
117
  * `apm_modules/<owner>/<pack>/agents/` into the staged `.claude/agents/`.
118
- * apm's claude target deploys `skills/` only, so without this an agent
119
- * reference a skill cites is absent from the agent CWD.
118
+ * apm's claude target deploys `skills/` only. Without this merge, an agent
119
+ * reference that a skill cites is absent from the agent CWD.
120
120
  * @param {string} familyRoot
121
121
  * @param {string} stagedClaude
122
122
  */
@@ -127,7 +127,7 @@ export class ApmInstaller {
127
127
  try {
128
128
  owners = await fs.readdir(modulesRoot, { withFileTypes: true });
129
129
  } catch {
130
- return; // no apm_modules — nothing to stage
130
+ return; // no apm_modules, so nothing to stage
131
131
  }
132
132
  const stagedAgents = join(stagedClaude, "agents");
133
133
  for (const owner of owners) {
@@ -159,8 +159,8 @@ export class ApmInstaller {
159
159
  ["install", "--target", "claude"],
160
160
  { cwd, stdio: ["ignore", "pipe", "pipe"] },
161
161
  );
162
- // Drain stdout concurrently so the child never blocks on backpressure;
163
- // capture stderr for the failure message.
162
+ // Drain stdout concurrently so the child never blocks on backpressure.
163
+ // Capture stderr for the failure message.
164
164
  let stderr = "";
165
165
  const drainStdout = (async () => {
166
166
  for await (const _chunk of child.stdout) {
@@ -199,8 +199,8 @@ export function createApmInstaller(deps) {
199
199
  * @param {import("./task-family.js").TaskFamily} family
200
200
  * @param {string} outputDir
201
201
  * @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
202
- * @param {object} [options] - Forwarded to `ApmInstaller.install` (e.g.
203
- * `{ skillsFrom }`).
202
+ * @param {object} [options] - This function forwards these to
203
+ * `ApmInstaller.install` (e.g. `{ skillsFrom }`).
204
204
  */
205
205
  export function installApm(family, outputDir, runtime, options = {}) {
206
206
  return new ApmInstaller({ runtime }).install(family, outputDir, options);
@@ -1,17 +1,20 @@
1
1
  /**
2
2
  * Env-loader — auto-discover `.env` / `.env.local` files in a task family
3
- * and its tasks, load them into `process.env`, and render the merged result
3
+ * and its tasks. Load them into `process.env`. Render the merged result
4
4
  * into each agent CWD.
5
5
  *
6
- * Discovery paths (loaded in this order, first value per key wins):
7
- * 1. process.env (CI secrets, shell env — never overwritten)
6
+ * The loader reads the discovery paths in this order. The first value per
7
+ * key wins:
8
+ * 1. process.env (CI secrets and shell env, which the loader never
9
+ * overwrites)
8
10
  * 2. <family>/.env.local
9
11
  * 3. <family>/.env
10
12
  * 4. tasks/<id>/.env.local
11
13
  * 5. tasks/<id>/.env
12
14
  *
13
- * Every discovered env file — family or task — is loaded into process.env
14
- * AND rendered (with resolved values) into the agent working directory.
15
+ * The loader loads every discovered env file, family or task, into
16
+ * process.env. It also renders the file (with resolved values) into the
17
+ * agent working directory.
15
18
  */
16
19
 
17
20
  import { join } from "node:path";
@@ -20,7 +23,8 @@ const ENV_FILES = [".env.local", ".env"];
20
23
 
21
24
  /**
22
25
  * Parse a `.env` file into an array of {key, value} pairs.
23
- * Handles KEY=VALUE, # comments, blank lines, and single/double-quoted values.
26
+ * It handles KEY=VALUE, # comments, blank lines, and single/double-quoted
27
+ * values.
24
28
  * @param {string} content
25
29
  * @returns {Array<{key: string, value: string}>}
26
30
  */
@@ -46,7 +50,7 @@ export function parseEnvFile(content) {
46
50
  }
47
51
 
48
52
  /**
49
- * Read and parse an env file, returning [] if the file does not exist.
53
+ * Read and parse an env file. Return [] if the file does not exist.
50
54
  * @param {object} fs - Async filesystem surface (`runtime.fs`).
51
55
  * @param {string} filePath
52
56
  * @returns {Promise<Array<{key: string, value: string}>>}
@@ -62,10 +66,11 @@ async function readEnvFile(fs, filePath) {
62
66
  }
63
67
 
64
68
  /**
65
- * Load entries into the process env map. Existing keys are never overwritten.
69
+ * Load entries into the process env map. This function never overwrites an
70
+ * existing key.
66
71
  * @param {Record<string, string|undefined>} env - The `runtime.proc.env` map.
67
72
  * @param {Array<{key: string, value: string}>} entries
68
- * @returns {string[]} var names that were loaded
73
+ * @returns {string[]} the var names it loaded
69
74
  */
70
75
  function applyToProcessEnv(env, entries) {
71
76
  const names = [];
@@ -79,7 +84,8 @@ function applyToProcessEnv(env, entries) {
79
84
  }
80
85
 
81
86
  /**
82
- * Load one env file: apply to the env map, record keys in the merged map.
87
+ * Load one env file. Apply it to the env map. Record the keys in the merged
88
+ * map.
83
89
  * @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
84
90
  * @param {string} dir
85
91
  * @param {string} file
@@ -100,7 +106,7 @@ async function loadOneEnvFile(runtime, dir, file, names, merged) {
100
106
  }
101
107
 
102
108
  /**
103
- * Scan directories for env files, load into the env map, and collect
109
+ * Scan directories for env files. Load them into the env map. Collect
104
110
  * a merged key manifest per filename.
105
111
  * @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
106
112
  * @param {string[]} dirs
@@ -118,7 +124,7 @@ async function collectEnvEntries(runtime, dirs) {
118
124
  }
119
125
 
120
126
  /**
121
- * Write resolved env files into the agent CWD and warn about empty values.
127
+ * Write resolved env files into the agent CWD. Warn about empty values.
122
128
  * @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
123
129
  * @param {Map<string, Map<string, true>>} merged
124
130
  * @param {string} agentCwd
@@ -142,14 +148,16 @@ async function renderEnvFiles(runtime, merged, agentCwd) {
142
148
  }
143
149
 
144
150
  /**
145
- * Discover `.env` / `.env.local` in one or more directories, load them
146
- * into the process env map, and render the resolved values into the agent CWD.
151
+ * Discover `.env` / `.env.local` in one or more directories. Load them
152
+ * into the process env map. Render the resolved values into the agent CWD.
147
153
  *
148
154
  * @param {string[]} dirs - Directories to scan (family root, task dir, etc.)
149
155
  * @param {string} agentCwd - Agent working directory to render into.
150
156
  * @param {import("@forwardimpact/libutil/runtime").Runtime} runtime - Ambient
151
- * collaborators; uses `fs` (async read/write), `proc.env`, `proc.stderr`.
152
- * @returns {Promise<string[]>} All var names discovered (for redaction).
157
+ * collaborators. It uses `fs` (async read/write), `proc.env`, and
158
+ * `proc.stderr`.
159
+ * @returns {Promise<string[]>} Every var name the loader discovered (for
160
+ * redaction).
153
161
  */
154
162
  export async function loadEnv(dirs, agentCwd, runtime) {
155
163
  const { names, merged } = await collectEnvEntries(runtime, dirs);
@@ -1,46 +1,50 @@
1
1
  /**
2
- * Grading derivation — the sole home of the check-row arithmetic.
2
+ * Grade derivation — the sole home of the check-row arithmetic.
3
3
  *
4
- * Check rows are the single authoritative grading channel. Every row is a
5
- * check by default; a row declares its role with its own fields, checked in
6
- * order:
4
+ * Check rows are the one authoritative input to the grade. Every row is a
5
+ * check by default. A row declares its role with its own fields. The
6
+ * classifier checks the roles in this order:
7
7
  *
8
8
  * 1. Gate — `gate` is exactly `true`, `pass` is boolean, and no
9
- * `weight` key is present. Any failing gate → `gatesPass`
10
- * false.
11
- * 2. Diagnostic — no `gate` key and `weight` is exactly `0`. Free-form;
12
- * never graded.
9
+ * `weight` key is present. A gate that fails sets
10
+ * `gatesPass` to false.
11
+ * 2. Diagnostic — no `gate` key and `weight` is exactly `0`. The row is
12
+ * free-form. The grader never scores it.
13
13
  * 3. Scored — no `gate` key, boolean `pass`, `weight` absent (defaults
14
14
  * to 1) or finite > 0.
15
- * 4. Malformed — everything else: any `gate`+`weight` co-occurrence (a
16
- * stray weight must never silently disarm a gate), a
15
+ * 4. Malformed — everything else: any `gate`+`weight` co-occurrence, a
17
16
  * non-boolean `gate`, a missing or non-boolean `pass` on a
18
17
  * graded row, an invalid `weight`, an fd-3 line that failed
19
- * to parse, a non-object row. Counts as a **failing scored
20
- * check** — dropping a defect could mint full marks;
21
- * failing the whole run would zero completed work.
18
+ * to parse, a non-object row. A stray weight must never
19
+ * silently disarm a gate. A malformed row counts as a
20
+ * **scored check that fails**. If the grader dropped the
21
+ * defect, the cell could mint full marks. If it failed the
22
+ * whole run, it would zero completed work.
22
23
  *
23
- * The producers' `source` stamp is display metadata, never a grading input.
24
+ * The producers' `source` stamp is display metadata. The grader never reads
25
+ * it.
24
26
  */
25
27
 
26
28
  /**
27
29
  * @typedef {object} GradeResult
28
30
  * @property {"pass" | "fail"} verdict - `healthy ∧ gatesPass ∧ fullMarks`.
29
31
  * @property {boolean} gatesPass - Every gate row passes (vacuously true).
30
- * @property {number | null} score - Weighted fraction of passing scored
31
- * checks; `null` when the cell has zero scored checks (binary task).
32
- * @property {boolean} fullMarks - Integer count predicate: no malformed rows
33
- * and every scored check passes. Never a float comparison, so fractional
34
- * weights carry no equality hazard. Vacuously true with zero scored checks.
32
+ * @property {number | null} score - Weighted fraction of the scored checks
33
+ * that pass. It is `null` when the cell has zero scored checks (binary
34
+ * task).
35
+ * @property {boolean} fullMarks - An integer count predicate. It is true when
36
+ * no row is malformed and every scored check passes. It is never a float
37
+ * comparison, so fractional weights carry no equality hazard. It is
38
+ * vacuously true with zero scored checks.
35
39
  * @property {number} malformed - Malformed row count.
36
40
  */
37
41
 
38
42
  /**
39
43
  * Grade the merged check rows against grader health.
40
44
  *
41
- * `healthy` is the completion signal a crashed grader cannot fake: when it is
42
- * false the verdict is `fail` whatever the rows say, so a hook that dies
43
- * after emitting passing rows can never mint marks.
45
+ * `healthy` is the completion signal a crashed grader cannot fake. When it is
46
+ * false, the verdict is `fail` whatever the rows say. A hook that emits rows
47
+ * that pass and then dies can never mint marks.
44
48
  * @param {unknown[]} details - Merged check rows from both producers.
45
49
  * @param {boolean} healthy - Invariants exited 0 AND the hidden-test engine
46
50
  * did not throw.
@@ -73,7 +77,7 @@ export function gradeChecks(details, healthy) {
73
77
  }
74
78
 
75
79
  /**
76
- * Fold one row into the running tally per its classified role.
80
+ * Fold one row into the tally per its classified role.
77
81
  * @param {{gatesPass: boolean, malformed: number, scored: number, passing: number, weightAll: number, weightPassing: number}} tally
78
82
  * @param {unknown} row
79
83
  */
@@ -96,11 +100,10 @@ function tallyRow(tally, row) {
96
100
  }
97
101
 
98
102
  /**
99
- * Run both check-row producers and grade the merged rows — the one
100
- * composition shared by the runner and the `grade` subcommand. An engine
101
- * throw is grader fault: its message lands on the returned `engineError`
102
- * and health fails, so a crashed grader can never mint marks from rows it
103
- * happened to emit first.
103
+ * Run both check-row producers and grade the merged rows. The runner and the
104
+ * `grade` subcommand share this one composition. An engine throw is grader
105
+ * fault. Its message lands on the returned `engineError` and health fails. A
106
+ * crashed grader can never mint marks from rows it emitted first.
104
107
  * @param {import("./task-family.js").Task} task
105
108
  * @param {{cwd: string, port: number, runDir: string, familyDir?: string|null}} ctx
106
109
  * @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
@@ -126,8 +129,8 @@ export async function runProducersAndGrade(task, ctx, runtime, producers) {
126
129
 
127
130
  /**
128
131
  * Merge the two producers' rows (invariants first) and stamp each row's
129
- * provenance. The stamp is display metadata, never a grading input, and
130
- * non-object rows (malformed by contract) pass through verbatim.
132
+ * provenance. The stamp is display metadata. The grader never reads it.
133
+ * Non-object rows (malformed by contract) pass through verbatim.
131
134
  * @param {unknown[]} invariantsDetails
132
135
  * @param {unknown[]} hiddenDetails
133
136
  * @returns {unknown[]}
@@ -147,9 +150,9 @@ function stampSource(row, source) {
147
150
  }
148
151
 
149
152
  /**
150
- * Project the raw `gradeChecks` return onto the record schema: `fullMarks`
151
- * is derivable and dropped, `score` is omitted on binary tasks (`null`),
152
- * `malformed` is omitted when clean.
153
+ * Project the raw `gradeChecks` return onto the record schema. `fullMarks` is
154
+ * derivable, so this function drops it. It omits `score` on binary tasks
155
+ * (`null`). It omits `malformed` when the rows are clean.
153
156
  * @param {GradeResult} raw
154
157
  * @returns {{verdict: "pass"|"fail", gatesPass: boolean, score?: number, malformed?: number}}
155
158
  */
@@ -177,9 +180,9 @@ function classifyRow(row) {
177
180
  }
178
181
 
179
182
  /**
180
- * A row carrying a `gate` key: valid only as `gate: true` with a boolean
181
- * `pass` and no `weight` key — any co-occurring weight is malformed so a
182
- * stray weight can never silently disarm a gate.
183
+ * Classify a row that has a `gate` key. The row is valid only as `gate: true`
184
+ * with a boolean `pass` and no `weight` key. Any weight that co-occurs makes
185
+ * the row malformed. A stray weight can never silently disarm a gate.
183
186
  * @param {object} row
184
187
  * @returns {"gate" | "malformed"}
185
188
  */
@@ -191,9 +194,9 @@ function classifyGateRow(row) {
191
194
  }
192
195
 
193
196
  /**
194
- * A gate-less row carrying a `weight` key: exactly 0 is a diagnostic, a
195
- * finite positive weight with a boolean `pass` is scored, anything else is
196
- * malformed.
197
+ * Classify a gate-less row that has a `weight` key. A weight of exactly 0 is
198
+ * a diagnostic. A finite positive weight with a boolean `pass` is scored.
199
+ * Anything else is malformed.
197
200
  * @param {object} row
198
201
  * @returns {"diagnostic" | "scored" | "malformed"}
199
202
  */
@@ -205,8 +208,8 @@ function classifyWeightedRow(row) {
205
208
  }
206
209
 
207
210
  /**
208
- * A malformed row fails at its own weight when it carries a valid positive
209
- * one, else at unit weight 1.
211
+ * A malformed row fails at its own weight when that weight is valid and
212
+ * positive. Otherwise it fails at unit weight 1.
210
213
  * @param {unknown} row
211
214
  * @returns {number}
212
215
  */