@forwardimpact/libharness 3.0.1 → 3.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -214,17 +214,39 @@ also lists the `Edit()` rules it tried.
214
214
 
215
215
  ## Documentation
216
216
 
217
- - [Coordinate an Agent Team](https://www.forwardimpact.team/docs/libraries/coordinate-team/index.md)
217
+ - [Coordinate an Agent Team](https://www.gemba.team/docs/coordinate-team/index.md)
218
218
  — run a lead and N participant agents in one async session (supervise /
219
219
  facilitate / discuss) with Ask/Answer/Announce and a single NDJSON trace.
220
- - [Run an Eval](https://www.forwardimpact.team/docs/libraries/prove-changes/run-eval/index.md)
220
+ - [Run an Eval](https://www.gemba.team/docs/prove-changes/run-eval/index.md)
221
221
  — author a judge profile, run an eval locally, wire it into CI, and inspect
222
222
  the trace it produces.
223
- - [Prove Agent Changes](https://www.forwardimpact.team/docs/libraries/prove-changes/index.md)
223
+ - [Prove Agent Changes](https://www.gemba.team/docs/prove-changes/index.md)
224
224
  — the end-to-end workflow from dataset generation through evaluation to
225
225
  trace analysis, with multi-agent collaboration sessions.
226
- - [Analyze Traces](https://www.forwardimpact.team/docs/libraries/prove-changes/trace-analysis/index.md)
226
+ - [Analyze Traces](https://www.gemba.team/docs/prove-changes/trace-analysis/index.md)
227
227
  — read the NDJSON traces produced by `gemba-harness` with `gemba-trace`.
228
228
  - [Agent Teams](https://www.forwardimpact.team/docs/products/agent-teams/index.md)
229
229
  — author the profiles consumed by `--agent-profile`, `--lead-profile`, and
230
230
  `--agent-profiles`.
231
+
232
+ ## Documentation home
233
+
234
+ libharness is an import-only library. It declares no `bin`. The
235
+ `gemba-harness`, `gemba-trace`, `gemba-benchmark`, and `gemba-selfedit`
236
+ commands ship with the Gemba product, which imports these modules. Run them
237
+ with `npx gemba-harness`, `npx gemba-trace`, or `npx gemba-benchmark`, or use
238
+ the installed `gemba-*` binaries. `gemba-selfedit` publishes no bare launcher.
239
+ Install `@forwardimpact/gemba` to get it.
240
+
241
+ The package publishes as `@forwardimpact/libharness` on the Forward Impact npm
242
+ scope. Install it with `npm install @forwardimpact/libharness`. Its task guides
243
+ live on the Gemba site at <https://www.gemba.team/>. The Forward Impact library
244
+ guide tree at <https://www.forwardimpact.team/docs/libraries/index.md> is not
245
+ this library's guide home.
246
+
247
+ **Decision (2026-08-26):** the split package scope and guide host are
248
+ deliberate. libharness stays a Gear npm package, so `package.json .homepage`
249
+ keeps <https://www.forwardimpact.team>. The agent-runtime guides moved to
250
+ gemba.team with the rest of the Gemba product. The `## Documentation` list
251
+ above carries the current URLs. Old `forwardimpact.team/docs/libraries/`
252
+ addresses forward to gemba.team.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@forwardimpact/libharness",
3
- "version": "3.0.1",
3
+ "version": "3.1.0",
4
4
  "description": "Autonomous agent team harness — coordinate a lead and participant agents in one async session, with eval, benchmark, and trace tooling to prove the changes worked.",
5
5
  "keywords": [
6
6
  "orchestration",
@@ -1,7 +1,7 @@
1
1
  /**
2
2
  * Hidden-test engine — runs a task's `tests/` overlay against the post-run
3
3
  * agent CWD. The engine stages each file at its mirrored path. It runs each
4
- * check with `node --test`. It converts the exit status into one check row.
4
+ * check with `bun test`. It converts the exit status into one check row.
5
5
  * It restores the tree, so the judge sees the workdir exactly as the agent
6
6
  * left it.
7
7
  *
@@ -69,7 +69,25 @@ async function runOneCheck(task, ctx, runtime, timeoutMs, check) {
69
69
  }
70
70
 
71
71
  /**
72
- * Spawn `node --test <staged path>` from the agent CWD under the hook env
72
+ * Reads a child pipe to a string. Returns what it read when the stream tears
73
+ * down mid-read, because a pipe that closes as the child exits is a race, not
74
+ * a check failure. The caller reads the exit code for the verdict.
75
+ *
76
+ * @param {import("node:stream").Readable} stream - Child stdout or stderr.
77
+ * @returns {Promise<string>} Everything read before the stream ended.
78
+ */
79
+ async function drainQuietly(stream) {
80
+ let out = "";
81
+ try {
82
+ for await (const chunk of stream) out += chunk.toString();
83
+ } catch {
84
+ // The stream closed under us. Keep what arrived.
85
+ }
86
+ return out;
87
+ }
88
+
89
+ /**
90
+ * Spawn `bun test <staged path>` from the agent CWD under the hook env
73
91
  * and map the exit status onto one row. The clock timer SIGKILLs a child
74
92
  * that outlives the per-check budget. The row then fails with a timeout
75
93
  * message.
@@ -83,27 +101,40 @@ async function spawnCheck(task, ctx, runtime, timeoutMs, check) {
83
101
  hooksDir: task.paths.hooks,
84
102
  familyDir: ctx.familyDir,
85
103
  });
86
- // An inherited test-runner context makes the child `node --test` report
87
- // exit 0 even when its tests fail. A check that fails would then mint a row
88
- // that passes whenever the harness itself runs under `node --test`.
89
- delete env.NODE_TEST_CONTEXT;
90
- const child = runtime.subprocess.spawn("node", ["--test", check.stagePath], {
91
- cwd: ctx.cwd,
92
- env,
93
- stdio: ["ignore", "pipe", "pipe"],
94
- });
104
+ // `bun test` sets no test-context variable that a nested run inherits, so
105
+ // a child reports its own exit status even when the harness itself runs
106
+ // under a test runner. That removes the `NODE_TEST_CONTEXT` scrub the
107
+ // `node --test` engine needed to stop a failing check minting a passing
108
+ // row. The "fractional score" grade test is the standing guard: it fails
109
+ // if a failing check ever reports success again.
110
+ //
111
+ // Pass the ABSOLUTE staged path. `bun test` reads its argument as a
112
+ // substring filter over discovered paths, not as one file, so the relative
113
+ // `app/test/x.test.js` also matches an agent-authored `sub/app/test/
114
+ // x.test.js` and folds that file's result into this check's row. An
115
+ // absolute path cannot be a substring of a deeper path, so it selects
116
+ // exactly the staged file. One `*.test.js` stays one check.
117
+ const child = runtime.subprocess.spawn(
118
+ "bun",
119
+ ["test", join(ctx.cwd, check.stagePath)],
120
+ {
121
+ cwd: ctx.cwd,
122
+ env,
123
+ stdio: ["ignore", "pipe", "pipe"],
124
+ },
125
+ );
95
126
  let timedOut = false;
96
127
  const timer = runtime.clock.setTimeout(() => {
97
128
  timedOut = true;
98
129
  child.kill("SIGKILL");
99
130
  }, timeoutMs);
100
- const drainStdout = (async () => {
101
- for await (const _chunk of child.stdout) {
102
- // discard
103
- }
104
- })();
105
- let stderr = "";
106
- for await (const chunk of child.stderr) stderr += chunk.toString();
131
+ // Both pipes drain only so a chatty child never blocks on a full pipe.
132
+ // A child that exits while its pipe is still open makes the async iterator
133
+ // reject with ERR_STREAM_PREMATURE_CLOSE on some runtimes. That teardown
134
+ // race is not a check result, so it must not throw out of the check. The
135
+ // exit code below is the verdict.
136
+ const drainStdout = drainQuietly(child.stdout);
137
+ const stderr = await drainQuietly(child.stderr);
107
138
  await drainStdout;
108
139
  const exit = await child.exitCode;
109
140
  runtime.clock.clearTimeout(timer);
@@ -9,17 +9,19 @@
9
9
  *
10
10
  * {{AGENT_INSTRUCTIONS}} — contents of agent.task.md
11
11
  * {{AGENT_PROFILE}} — agent profile body (empty string if none)
12
- * {{AGENT_TRACE_PATH}} — path to agent.ndjson
12
+ * {{AGENT_TRACE_PATH}} — absolute path to the cell's agent lane,
13
+ * trace--<case>--agent.agent.ndjson (materialized
14
+ * before any session runs)
13
15
  * {{GRADE_RESULT}} — JSON grade object plus the merged check rows
14
16
  * {{SKILL_SET_HASH}} — SHA-256 from apm.lock.yaml
15
17
  * {{TASK_ID}} — task name (directory under tasks/)
16
18
  * {{TASK_DIR}} — path to the agent working directory
17
19
  *
18
- * The adapter reads the judge verdict directly from the orchestration
19
- * context's `concluded` flag. It does not parse the trace on the happy path.
20
- * `parseConcludeFromTrace` stays for offline analysis. It is also a fallback
21
- * when the runtime ctx is not available, for example when you re-grade a
22
- * historical run from its judge.ndjson file.
20
+ * The judge verdict is captured from the orchestration context's
21
+ * `concluded` flag directly — no trace parsing on the happy path.
22
+ * `parseConcludeFromTrace` is preserved for offline analysis and as a
23
+ * fallback when the runtime ctx isn't available (e.g. re-grading a
24
+ * historical run from its preserved judge lane file).
23
25
  */
24
26
 
25
27
  import { createJudge } from "../judge.js";
@@ -0,0 +1,79 @@
1
+ /**
2
+ * Raw-trace summary for the benchmark runner: one post-session read of the
3
+ * preserved raw combined trace derives cost, turns, and submission. Named as
4
+ * its own module so summarization never re-entangles with splitting — the
5
+ * coupling that caused the original split-policy divergence.
6
+ */
7
+
8
+ import { sumTraceCost } from "../cost.js";
9
+ import { parseEnvelopeLine } from "../trace-split.js";
10
+
11
+ /**
12
+ * One read of the preserved raw combined trace:
13
+ * cost — `sumTraceCost` over the lines (the one cost path),
14
+ * turns — last orchestrator-source `summary` event's `turns`,
15
+ * submission — last agent-source assistant text block.
16
+ *
17
+ * An empty (materialized-stub) raw file yields zeros and an empty
18
+ * submission; malformed and blank lines are tolerated.
19
+ *
20
+ * @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
21
+ * @param {string} rawTracePath
22
+ * @returns {Promise<{costUsd: number,
23
+ * costBreakdown: {agent: number, supervisor: number},
24
+ * turns: number, submission: string}>}
25
+ */
26
+ export async function summarizeRawTrace(runtime, rawTracePath) {
27
+ const content = await runtime.fs.readFile(rawTracePath, "utf8");
28
+ const lines = content.split("\n");
29
+ const { totalCostUsd, bySource } = sumTraceCost(lines);
30
+ const { turns, submission } = deriveTurnsAndSubmission(lines);
31
+
32
+ return {
33
+ costUsd: totalCostUsd,
34
+ costBreakdown: {
35
+ agent: bySource.agent ?? 0,
36
+ supervisor: bySource.supervisor ?? 0,
37
+ },
38
+ turns,
39
+ submission,
40
+ };
41
+ }
42
+
43
+ /**
44
+ * One walk of the parsed envelope lines: the last orchestrator `summary`
45
+ * event's `turns` and the last agent assistant text block.
46
+ * @param {string[]} lines
47
+ * @returns {{turns: number, submission: string}}
48
+ */
49
+ function deriveTurnsAndSubmission(lines) {
50
+ let turns = 0;
51
+ let submission = "";
52
+ for (const line of lines) {
53
+ const envelope = parseEnvelopeLine(line);
54
+ if (!envelope) continue;
55
+ const inner = envelope.event;
56
+ if (envelope.source === "agent" && inner.type === "assistant") {
57
+ const text = extractText(inner);
58
+ if (text) submission = text;
59
+ }
60
+ if (envelope.source === "orchestrator" && inner.type === "summary") {
61
+ turns = inner.turns ?? 0;
62
+ }
63
+ }
64
+ return { turns, submission };
65
+ }
66
+
67
+ /**
68
+ * Last text block of an assistant event's content, or null when none exists.
69
+ * @param {object} inner - Unwrapped event.
70
+ * @returns {string|null}
71
+ */
72
+ function extractText(inner) {
73
+ const content = inner.message?.content ?? inner.content;
74
+ if (!Array.isArray(content)) return null;
75
+ for (let i = content.length - 1; i >= 0; i--) {
76
+ if (content[i].type === "text" && content[i].text) return content[i].text;
77
+ }
78
+ return null;
79
+ }
@@ -104,9 +104,15 @@ const HAPPY_RECORD = z.object({
104
104
  score: z.number().min(0).max(1).optional(),
105
105
  submission: z.string(),
106
106
  judgeVerdict: JUDGE_VERDICT_SHAPE.optional(),
107
+ // Trace paths are relative to the run output directory — valid on the
108
+ // runner and inside a downloaded trace artifact alike. Raw and
109
+ // agent/supervisor lanes are materialized at workdir allocation, so they
110
+ // are present on every executed cell; the judge lane exists only on
111
+ // judged cells.
112
+ rawTracePath: z.string(),
107
113
  agentTracePath: z.string(),
108
114
  supervisorTracePath: z.string(),
109
- judgeTracePath: z.string(),
115
+ judgeTracePath: z.string().optional(),
110
116
  agentError: AGENT_ERROR_SHAPE.optional(),
111
117
  preflightError: z.undefined().optional(),
112
118
  });
@@ -115,12 +121,12 @@ const PREFLIGHT_RECORD = z.object({
115
121
  ...COMMON_FIELDS,
116
122
  costUsd: z.literal(0),
117
123
  preflightError: PREFLIGHT_ERROR_SHAPE,
118
- // The runner allocates the trace paths in WorkdirManager.start, even on
119
- // preflight failure. The record then stays uniform across branches, and
120
- // downstream consumers can reference the paths without conditional fields.
121
- agentTracePath: z.string(),
122
- supervisorTracePath: z.string(),
123
- judgeTracePath: z.string(),
124
+ // No trace-path fields: a preflight-failure record references only traces
125
+ // a session produced, even though the materialized stubs exist on disk.
126
+ rawTracePath: z.undefined().optional(),
127
+ agentTracePath: z.undefined().optional(),
128
+ supervisorTracePath: z.undefined().optional(),
129
+ judgeTracePath: z.undefined().optional(),
124
130
  invariants: z.undefined().optional(),
125
131
  grade: z.undefined().optional(),
126
132
  hiddenTests: z.undefined().optional(),
@@ -19,11 +19,11 @@
19
19
  * ledger. The iterator mirrors the same stream to CLI stdout.
20
20
  */
21
21
 
22
- import { join, resolve as resolvePath } from "node:path";
22
+ import { join, relative, resolve as resolvePath } from "node:path";
23
23
 
24
24
  import { DEFAULT_ENV_ALLOWLIST, createRedactor } from "../redaction.js";
25
- import { sumTraceCost } from "../cost.js";
26
25
  import { createSupervisor } from "../supervisor.js";
26
+ import { splitTrace } from "../trace-split.js";
27
27
  import { installApm as defaultInstallApm } from "./apm-installer.js";
28
28
  import { installNpm as defaultInstallNpm } from "./npm-installer.js";
29
29
  import { runJudge } from "./judge.js";
@@ -31,8 +31,8 @@ import { validateResultRecord } from "./result.js";
31
31
  import { runInvariants } from "./invariants.js";
32
32
  import { runHiddenTests } from "./hidden-tests.js";
33
33
  import { runProducersAndGrade } from "./grade.js";
34
+ import { summarizeRawTrace } from "./raw-summary.js";
34
35
  import { assertJudgeProfileStaged, loadTaskFamily } from "./task-family.js";
35
- import { splitAndSummarize } from "./trace-split.js";
36
36
  import { createWorkdirManager } from "./workdir.js";
37
37
  import { CellScheduler } from "./scheduler.js";
38
38
 
@@ -77,9 +77,10 @@ export class BenchmarkRunner {
77
77
  * to `AGENT_WATCHDOG_MS`. A test injects its own to force a stall in-test.
78
78
  * @param {number} [opts.termGraceMs] - SIGTERM→SIGKILL grace (ms) for the per-task process group.
79
79
  * @param {Function} [opts.runAgent] - Test seam: replaces the agent-under-test
80
- * session. Must return `{costUsd, turns, submission, agentError?}` and
81
- * write a valid NDJSON trace to `workdir.agentTracePath`. Default uses
82
- * `createAgentRunner` with the harness `BASE_TOOLS` allowlist. Internal
80
+ * session. Must run the session, stream `{source, seq, event}` envelopes
81
+ * to `workdir.rawTracePath`, and return `{agentError?}` — cost, turns,
82
+ * and submission are always derived from the raw file by the shared
83
+ * split/summary pipeline, so the seam exercises the real path. Internal
83
84
  * testing only. It is not part of the public API.
84
85
  * @param {import("@forwardimpact/libutil/runtime").Runtime} opts.runtime -
85
86
  * The host injects these ambient collaborators (`fs`, `subprocess`,
@@ -256,12 +257,12 @@ export class BenchmarkRunner {
256
257
  t0,
257
258
  });
258
259
  } catch (e) {
259
- // `wm.start()` (port acquire + workdir/env seed) is the one throw site
260
- // that `#executeCell` does not catch. Turn it into the runner's own
261
- // fallback record so `#runOne` never rejects. The scheduler's
262
- // one-record-per-cell contract depends on that. `report` skips the
263
- // fallback because it fails the schema, the same as any other
264
- // runner-side schema failure.
260
+ // Catches the throw sites `#executeCell` does not: `wm.start()` (port
261
+ // acquire + workdir/env seed) and the shared split/summary pipeline.
262
+ // Turn either into the runner's own fallback record so `#runOne` never
263
+ // rejects. The scheduler's one-record-per-cell contract depends on
264
+ // that. `report` skips the fallback because it fails the schema, the
265
+ // same as any other runner-side schema failure.
265
266
  return {
266
267
  taskId: task.id,
267
268
  runIndex,
@@ -301,8 +302,8 @@ export class BenchmarkRunner {
301
302
  }
302
303
  {
303
304
  const agentRun = await this.#runAgentSafe(task, workdir);
304
- const { costUsd, turns, submission, agentError } = agentRun;
305
- const breakdown = agentRun.costBreakdown ?? { agent: 0, supervisor: 0 };
305
+ const { costUsd, costBreakdown, turns, submission, agentError } =
306
+ agentRun;
306
307
  const graded = await this.#gradeCell(family, task, workdir);
307
308
  const { invariants, hiddenRows, engineError, rows, grade } = graded;
308
309
  const { judgeVerdict, judgeCost } = await this.#judgeCell({
@@ -319,6 +320,7 @@ export class BenchmarkRunner {
319
320
  // or a judge that fails zeroes the effective score. Full marks does not
320
321
  // zero it. A fractional score with verdict fail is the point.
321
322
  const scoreValid = graded.healthy && grade.gatesPass && judgePass;
323
+ const tracePaths = await this.#traceRecordPaths(workdir);
322
324
  const record = {
323
325
  taskId: task.id,
324
326
  runIndex,
@@ -337,15 +339,9 @@ export class BenchmarkRunner {
337
339
  submission,
338
340
  ...(judgeVerdict && { judgeVerdict }),
339
341
  costUsd: costUsd + judgeCost,
340
- costBreakdown: {
341
- agent: breakdown.agent ?? 0,
342
- supervisor: breakdown.supervisor ?? 0,
343
- judge: judgeCost,
344
- },
342
+ costBreakdown: { ...costBreakdown, judge: judgeCost },
345
343
  turns,
346
- agentTracePath: workdir.agentTracePath,
347
- supervisorTracePath: workdir.supervisorTracePath,
348
- judgeTracePath: workdir.judgeTracePath,
344
+ ...tracePaths,
349
345
  profiles: {
350
346
  agent: this.profiles.agent,
351
347
  supervisor: null,
@@ -424,39 +420,42 @@ export class BenchmarkRunner {
424
420
  }
425
421
 
426
422
  /**
427
- * Dispatch to either the injected hook or the default `#runAgent`. Either
428
- * path can throw. Catch the error here so it becomes an `agentError` on the
429
- * record (spec criterion 1: records on agent failure). The iterator then
430
- * does not abort.
423
+ * Dispatch to either the injected hook or the default `#runAgent`, then run
424
+ * the shared pipeline once: split the preserved raw trace into lanes and
425
+ * derive cost/turns/submission from the same file. Either session path can
426
+ * throw; catch here so a session error becomes an `agentError` on the
427
+ * record rather than aborting the whole iterator — the pipeline still runs,
428
+ * so failed cells keep coherent (possibly empty) lanes and zeroed totals.
429
+ * A pipeline throw itself (the raw file is materialized at allocation, so
430
+ * only an fs-level fault) propagates to `#runOne`'s fallback record.
431
431
  */
432
432
  async #runAgentSafe(task, workdir) {
433
+ let agentError = null;
433
434
  try {
434
- if (this._runAgentHook) {
435
- const r = await this._runAgentHook(task, workdir, this);
436
- return { agentError: null, ...r };
437
- }
438
- return await this.#runAgent(task, workdir);
435
+ const r = this._runAgentHook
436
+ ? await this._runAgentHook(task, workdir, this)
437
+ : await this.#runAgent(task, workdir);
438
+ agentError = r?.agentError ?? null;
439
439
  } catch (e) {
440
- return {
441
- costUsd: 0,
442
- costBreakdown: { agent: 0, supervisor: 0 },
443
- turns: 0,
444
- submission: "",
445
- agentError: { message: e.message ?? String(e), aborted: false },
446
- };
440
+ agentError = { message: e.message ?? String(e), aborted: false };
447
441
  }
442
+ await splitTrace(this.runtime, workdir.rawTracePath, {
443
+ caseId: workdir.caseId,
444
+ outputDir: workdir.runDir,
445
+ });
446
+ const summary = await summarizeRawTrace(this.runtime, workdir.rawTracePath);
447
+ return { ...summary, agentError };
448
448
  }
449
449
 
450
450
  /**
451
- * Run the agent-under-test under a Supervisor. The supervisor writes
452
- * a combined tagged NDJSON trace. After the session, this method splits
453
- * the trace into agent.ndjson and supervisor.ndjson. It also extracts
454
- * cost/turns/submission.
451
+ * Run the agent-under-test under a Supervisor. The supervisor streams the
452
+ * combined tagged NDJSON envelope trace to `workdir.rawTracePath`, which is
453
+ * preserved for the life of the run output; `#runAgentSafe` splits it into
454
+ * the convention-named lanes and summarizes it afterwards.
455
455
  */
456
456
  async #runAgent(task, workdir) {
457
457
  const fs = this.runtime.fs;
458
- const combinedPath = join(workdir.runDir, ".combined.ndjson");
459
- const combinedStream = fs.createWriteStream(combinedPath);
458
+ const combinedStream = fs.createWriteStream(workdir.rawTracePath);
460
459
  const supervisorInstructions = task.paths.supervisor
461
460
  ? await fs.readFile(task.paths.supervisor, "utf8").catch(() => null)
462
461
  : null;
@@ -510,27 +509,31 @@ export class BenchmarkRunner {
510
509
  this.runtime.clock.clearTimeout(watchdog);
511
510
  await new Promise((r) => combinedStream.end(r));
512
511
  }
513
- const summary = await splitAndSummarize(
514
- this.runtime,
515
- combinedPath,
516
- workdir.agentTracePath,
517
- workdir.supervisorTracePath,
518
- );
519
- // `sumTraceCost` sums the cost across every participant's result events
520
- // from the one combined trace, and attributes it per source. Read the
521
- // trace before you unlink it.
522
- const combined = await fs.readFile(combinedPath, "utf8");
523
- const { totalCostUsd, bySource } = sumTraceCost(combined.split("\n"));
524
- await fs.unlink(combinedPath).catch(() => {});
525
- return {
526
- ...summary,
527
- costUsd: totalCostUsd,
528
- costBreakdown: {
529
- agent: bySource.agent ?? 0,
530
- supervisor: bySource.supervisor ?? 0,
531
- },
532
- agentError,
512
+ return { agentError };
513
+ }
514
+
515
+ /**
516
+ * Run-output-relative trace-path record fields, each present only when its
517
+ * file exists: raw and agent/supervisor lanes are materialized at workdir
518
+ * allocation (present on every executed cell); the judge lane exists only
519
+ * on judged cells. Relative paths stay valid inside a downloaded artifact.
520
+ */
521
+ async #traceRecordPaths(workdir) {
522
+ const fields = {
523
+ rawTracePath: workdir.rawTracePath,
524
+ agentTracePath: workdir.agentTracePath,
525
+ supervisorTracePath: workdir.supervisorTracePath,
526
+ judgeTracePath: workdir.judgeTracePath,
533
527
  };
528
+ const out = {};
529
+ for (const [field, absPath] of Object.entries(fields)) {
530
+ const exists = await this.runtime.fs
531
+ .access(absPath)
532
+ .then(() => true)
533
+ .catch(() => false);
534
+ if (exists) out[field] = relative(this.output, absPath);
535
+ }
536
+ return out;
534
537
  }
535
538
 
536
539
  async #buildJudgeContext(task, workdir, skillSetHash) {
@@ -579,9 +582,9 @@ export class BenchmarkRunner {
579
582
  skillSetHash,
580
583
  familyRevision,
581
584
  durationMs,
582
- agentTracePath: workdir.agentTracePath,
583
- supervisorTracePath: workdir.supervisorTracePath,
584
- judgeTracePath: workdir.judgeTracePath,
585
+ // No trace-path fields: even though the materialized stubs exist on
586
+ // disk, a preflight-failure record references only traces a session
587
+ // produced.
585
588
  };
586
589
  }
587
590
 
@@ -35,6 +35,8 @@
35
35
  import { createHash } from "node:crypto";
36
36
  import { join, posix, relative, resolve, sep } from "node:path";
37
37
 
38
+ import { isValidTaskId } from "../trace-identity.js";
39
+
38
40
  const GIT_URL_RE = /^(git@|https?:\/\/|ssh:\/\/|git:\/\/)/;
39
41
  const SKIP_DIRS = new Set([".git", "node_modules"]);
40
42
  // POSIX `X_OK` (execute permission). Node's fs honours the numeric mode, so we
@@ -129,6 +131,13 @@ async function discoverTasks(runtime, rootPath) {
129
131
  }
130
132
 
131
133
  async function loadTask(fs, taskDir, id) {
134
+ // The rule itself lives in the identity module; the loader only invokes
135
+ // the predicate so an unbuildable case id fails before any agent spend.
136
+ if (!isValidTaskId(id)) {
137
+ throw new Error(
138
+ `invalid task id '${id}': task directory names must not contain "--" or start/end with "-"`,
139
+ );
140
+ }
132
141
  const supervisorPath = join(taskDir, "supervisor.task.md");
133
142
  const judgePath = join(taskDir, "judge.task.md");
134
143
  const preflightPath = join(taskDir, "hooks", "preflight.sh");
@@ -16,6 +16,11 @@ import { createServer } from "node:net";
16
16
  import { connect } from "node:net";
17
17
  import { join } from "node:path";
18
18
 
19
+ import {
20
+ buildCaseId,
21
+ laneFilename,
22
+ rawTraceFilename,
23
+ } from "../trace-identity.js";
19
24
  import { loadEnv } from "./env-loader.js";
20
25
  import { buildHookEnv } from "./hook-env.js";
21
26
 
@@ -28,6 +33,8 @@ const DEFAULT_TERM_GRACE_MS = 5_000;
28
33
  * @property {number} port - Allocated TCP port for the agent.
29
34
  * @property {number} pgid - Process-group id captured from the preflight child.
30
35
  * @property {*} scaffold - Reserved per design § Components. v1 sets null.
36
+ * @property {string} caseId - Grid-unique case identity `<taskId>-r<runIndex>`.
37
+ * @property {string} rawTracePath - Combined raw envelope trace (kept).
31
38
  * @property {string} agentTracePath
32
39
  * @property {string} supervisorTracePath
33
40
  * @property {string} judgeTracePath
@@ -72,8 +79,7 @@ export class WorkdirManager {
72
79
  */
73
80
  async start(task, runIndex) {
74
81
  const fs = this.runtime.fs;
75
- const slug = task.id.replace("/", "__");
76
- const runDir = join(this.runOutputDir, "runs", slug, String(runIndex));
82
+ const runDir = join(this.runOutputDir, "runs", task.id, String(runIndex));
77
83
  const cwd = join(runDir, "cwd");
78
84
  await fs.mkdir(cwd, { recursive: true });
79
85
 
@@ -124,11 +130,25 @@ export class WorkdirManager {
124
130
  const envNames =
125
131
  envDirs.length > 0 ? await loadEnv(envDirs, cwd, this.runtime) : [];
126
132
 
127
- const port = await this.ports.acquire();
128
- const agentTracePath = join(runDir, "agent.ndjson");
129
- const supervisorTracePath = join(runDir, "supervisor.ndjson");
130
- const judgeTracePath = join(runDir, "judge.ndjson");
133
+ // Allocate identity and trace paths before acquiring the port so an
134
+ // invalid task id cannot leak a port reservation.
135
+ const caseId = buildCaseId(task.id, runIndex);
136
+ const rawTracePath = join(runDir, rawTraceFilename(caseId));
137
+ const agentTracePath = join(runDir, laneFilename(caseId, "agent", "agent"));
138
+ const supervisorTracePath = join(
139
+ runDir,
140
+ laneFilename(caseId, "supervisor", "supervisor"),
141
+ );
142
+ const judgeTracePath = join(runDir, laneFilename(caseId, "judge", "judge"));
143
+ // Materialize the raw and agent/supervisor lanes empty at allocation so
144
+ // every path that reaches the judge — including a pre-session agent
145
+ // failure — finds them on disk. The judge lane is written by the judge
146
+ // session itself and exists only on judged cells.
147
+ for (const p of [rawTracePath, agentTracePath, supervisorTracePath]) {
148
+ await fs.writeFile(p, "");
149
+ }
131
150
 
151
+ const port = await this.ports.acquire();
132
152
  const preflight = task.paths.preflight
133
153
  ? await runPreflight(this.runtime, task.paths.preflight, cwd, port, {
134
154
  taskId: task.id,
@@ -144,6 +164,8 @@ export class WorkdirManager {
144
164
  port,
145
165
  pgid: preflight.pgid,
146
166
  scaffold: null,
167
+ caseId,
168
+ rawTracePath,
147
169
  agentTracePath,
148
170
  supervisorTracePath,
149
171
  judgeTracePath,
@@ -166,13 +166,13 @@ export const definition = {
166
166
  documentation: [
167
167
  {
168
168
  title: "Run a Benchmark",
169
- url: "https://www.forwardimpact.team/docs/libraries/prove-changes/run-benchmark/index.md",
169
+ url: "https://www.gemba.team/docs/prove-changes/run-benchmark/index.md",
170
170
  description:
171
171
  "Author a coding-task family, run a benchmark across multiple runs, and read the pass@k report.",
172
172
  },
173
173
  {
174
174
  title: "Automate with GitHub Actions",
175
- url: "https://www.forwardimpact.team/docs/libraries/prove-changes/run-benchmark/ci-workflow/index.md",
175
+ url: "https://www.gemba.team/docs/prove-changes/run-benchmark/ci-workflow/index.md",
176
176
  description:
177
177
  "Run benchmarks in CI with the forwardimpact/gemba-benchmark action.",
178
178
  },