@forwardimpact/libharness 3.0.2 → 3.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -214,17 +214,39 @@ also lists the `Edit()` rules it tried.
214
214
 
215
215
  ## Documentation
216
216
 
217
- - [Coordinate an Agent Team](https://www.forwardimpact.team/docs/libraries/coordinate-team/index.md)
217
+ - [Coordinate an Agent Team](https://www.gemba.team/docs/coordinate-team/index.md)
218
218
  — run a lead and N participant agents in one async session (supervise /
219
219
  facilitate / discuss) with Ask/Answer/Announce and a single NDJSON trace.
220
- - [Run an Eval](https://www.forwardimpact.team/docs/libraries/prove-changes/run-eval/index.md)
220
+ - [Run an Eval](https://www.gemba.team/docs/prove-changes/run-eval/index.md)
221
221
  — author a judge profile, run an eval locally, wire it into CI, and inspect
222
222
  the trace it produces.
223
- - [Prove Agent Changes](https://www.forwardimpact.team/docs/libraries/prove-changes/index.md)
223
+ - [Prove Agent Changes](https://www.gemba.team/docs/prove-changes/index.md)
224
224
  — the end-to-end workflow from dataset generation through evaluation to
225
225
  trace analysis, with multi-agent collaboration sessions.
226
- - [Analyze Traces](https://www.forwardimpact.team/docs/libraries/prove-changes/trace-analysis/index.md)
226
+ - [Analyze Traces](https://www.gemba.team/docs/prove-changes/trace-analysis/index.md)
227
227
  — read the NDJSON traces produced by `gemba-harness` with `gemba-trace`.
228
228
  - [Agent Teams](https://www.forwardimpact.team/docs/products/agent-teams/index.md)
229
229
  — author the profiles consumed by `--agent-profile`, `--lead-profile`, and
230
230
  `--agent-profiles`.
231
+
232
+ ## Documentation home
233
+
234
+ libharness is an import-only library. It declares no `bin`. The
235
+ `gemba-harness`, `gemba-trace`, `gemba-benchmark`, and `gemba-selfedit`
236
+ commands ship with the Gemba product, which imports these modules. Run them
237
+ with `npx gemba-harness`, `npx gemba-trace`, or `npx gemba-benchmark`, or use
238
+ the installed `gemba-*` binaries. `gemba-selfedit` publishes no bare launcher.
239
+ Install `@forwardimpact/gemba` to get it.
240
+
241
+ The package publishes as `@forwardimpact/libharness` on the Forward Impact npm
242
+ scope. Install it with `npm install @forwardimpact/libharness`. Its task guides
243
+ live on the Gemba site at <https://www.gemba.team/>. The Forward Impact library
244
+ guide tree at <https://www.forwardimpact.team/docs/libraries/index.md> is not
245
+ this library's guide home.
246
+
247
+ **Decision (2026-08-26):** the split package scope and guide host are
248
+ deliberate. libharness stays a Gear npm package, so `package.json .homepage`
249
+ keeps <https://www.forwardimpact.team>. The agent-runtime guides moved to
250
+ gemba.team with the rest of the Gemba product. The `## Documentation` list
251
+ above carries the current URLs. Old `forwardimpact.team/docs/libraries/`
252
+ addresses forward to gemba.team.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@forwardimpact/libharness",
3
- "version": "3.0.2",
3
+ "version": "3.1.0",
4
4
  "description": "Autonomous agent team harness — coordinate a lead and participant agents in one async session, with eval, benchmark, and trace tooling to prove the changes worked.",
5
5
  "keywords": [
6
6
  "orchestration",
@@ -9,17 +9,19 @@
9
9
  *
10
10
  * {{AGENT_INSTRUCTIONS}} — contents of agent.task.md
11
11
  * {{AGENT_PROFILE}} — agent profile body (empty string if none)
12
- * {{AGENT_TRACE_PATH}} — path to agent.ndjson
12
+ * {{AGENT_TRACE_PATH}} — absolute path to the cell's agent lane,
13
+ * trace--<case>--agent.agent.ndjson (materialized
14
+ * before any session runs)
13
15
  * {{GRADE_RESULT}} — JSON grade object plus the merged check rows
14
16
  * {{SKILL_SET_HASH}} — SHA-256 from apm.lock.yaml
15
17
  * {{TASK_ID}} — task name (directory under tasks/)
16
18
  * {{TASK_DIR}} — path to the agent working directory
17
19
  *
18
- * The adapter reads the judge verdict directly from the orchestration
19
- * context's `concluded` flag. It does not parse the trace on the happy path.
20
- * `parseConcludeFromTrace` stays for offline analysis. It is also a fallback
21
- * when the runtime ctx is not available, for example when you re-grade a
22
- * historical run from its judge.ndjson file.
20
+ * The judge verdict is captured from the orchestration context's
21
+ * `concluded` flag directly — no trace parsing on the happy path.
22
+ * `parseConcludeFromTrace` is preserved for offline analysis and as a
23
+ * fallback when the runtime ctx isn't available (e.g. re-grading a
24
+ * historical run from its preserved judge lane file).
23
25
  */
24
26
 
25
27
  import { createJudge } from "../judge.js";
@@ -0,0 +1,79 @@
1
+ /**
2
+ * Raw-trace summary for the benchmark runner: one post-session read of the
3
+ * preserved raw combined trace derives cost, turns, and submission. Named as
4
+ * its own module so summarization never re-entangles with splitting — the
5
+ * coupling that caused the original split-policy divergence.
6
+ */
7
+
8
+ import { sumTraceCost } from "../cost.js";
9
+ import { parseEnvelopeLine } from "../trace-split.js";
10
+
11
+ /**
12
+ * One read of the preserved raw combined trace:
13
+ * cost — `sumTraceCost` over the lines (the one cost path),
14
+ * turns — last orchestrator-source `summary` event's `turns`,
15
+ * submission — last agent-source assistant text block.
16
+ *
17
+ * An empty (materialized-stub) raw file yields zeros and an empty
18
+ * submission; malformed and blank lines are tolerated.
19
+ *
20
+ * @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
21
+ * @param {string} rawTracePath
22
+ * @returns {Promise<{costUsd: number,
23
+ * costBreakdown: {agent: number, supervisor: number},
24
+ * turns: number, submission: string}>}
25
+ */
26
+ export async function summarizeRawTrace(runtime, rawTracePath) {
27
+ const content = await runtime.fs.readFile(rawTracePath, "utf8");
28
+ const lines = content.split("\n");
29
+ const { totalCostUsd, bySource } = sumTraceCost(lines);
30
+ const { turns, submission } = deriveTurnsAndSubmission(lines);
31
+
32
+ return {
33
+ costUsd: totalCostUsd,
34
+ costBreakdown: {
35
+ agent: bySource.agent ?? 0,
36
+ supervisor: bySource.supervisor ?? 0,
37
+ },
38
+ turns,
39
+ submission,
40
+ };
41
+ }
42
+
43
+ /**
44
+ * One walk of the parsed envelope lines: the last orchestrator `summary`
45
+ * event's `turns` and the last agent assistant text block.
46
+ * @param {string[]} lines
47
+ * @returns {{turns: number, submission: string}}
48
+ */
49
+ function deriveTurnsAndSubmission(lines) {
50
+ let turns = 0;
51
+ let submission = "";
52
+ for (const line of lines) {
53
+ const envelope = parseEnvelopeLine(line);
54
+ if (!envelope) continue;
55
+ const inner = envelope.event;
56
+ if (envelope.source === "agent" && inner.type === "assistant") {
57
+ const text = extractText(inner);
58
+ if (text) submission = text;
59
+ }
60
+ if (envelope.source === "orchestrator" && inner.type === "summary") {
61
+ turns = inner.turns ?? 0;
62
+ }
63
+ }
64
+ return { turns, submission };
65
+ }
66
+
67
+ /**
68
+ * Last text block of an assistant event's content, or null when none exists.
69
+ * @param {object} inner - Unwrapped event.
70
+ * @returns {string|null}
71
+ */
72
+ function extractText(inner) {
73
+ const content = inner.message?.content ?? inner.content;
74
+ if (!Array.isArray(content)) return null;
75
+ for (let i = content.length - 1; i >= 0; i--) {
76
+ if (content[i].type === "text" && content[i].text) return content[i].text;
77
+ }
78
+ return null;
79
+ }
@@ -104,9 +104,15 @@ const HAPPY_RECORD = z.object({
104
104
  score: z.number().min(0).max(1).optional(),
105
105
  submission: z.string(),
106
106
  judgeVerdict: JUDGE_VERDICT_SHAPE.optional(),
107
+ // Trace paths are relative to the run output directory — valid on the
108
+ // runner and inside a downloaded trace artifact alike. Raw and
109
+ // agent/supervisor lanes are materialized at workdir allocation, so they
110
+ // are present on every executed cell; the judge lane exists only on
111
+ // judged cells.
112
+ rawTracePath: z.string(),
107
113
  agentTracePath: z.string(),
108
114
  supervisorTracePath: z.string(),
109
- judgeTracePath: z.string(),
115
+ judgeTracePath: z.string().optional(),
110
116
  agentError: AGENT_ERROR_SHAPE.optional(),
111
117
  preflightError: z.undefined().optional(),
112
118
  });
@@ -115,12 +121,12 @@ const PREFLIGHT_RECORD = z.object({
115
121
  ...COMMON_FIELDS,
116
122
  costUsd: z.literal(0),
117
123
  preflightError: PREFLIGHT_ERROR_SHAPE,
118
- // The runner allocates the trace paths in WorkdirManager.start, even on
119
- // preflight failure. The record then stays uniform across branches, and
120
- // downstream consumers can reference the paths without conditional fields.
121
- agentTracePath: z.string(),
122
- supervisorTracePath: z.string(),
123
- judgeTracePath: z.string(),
124
+ // No trace-path fields: a preflight-failure record references only traces
125
+ // a session produced, even though the materialized stubs exist on disk.
126
+ rawTracePath: z.undefined().optional(),
127
+ agentTracePath: z.undefined().optional(),
128
+ supervisorTracePath: z.undefined().optional(),
129
+ judgeTracePath: z.undefined().optional(),
124
130
  invariants: z.undefined().optional(),
125
131
  grade: z.undefined().optional(),
126
132
  hiddenTests: z.undefined().optional(),
@@ -19,11 +19,11 @@
19
19
  * ledger. The iterator mirrors the same stream to CLI stdout.
20
20
  */
21
21
 
22
- import { join, resolve as resolvePath } from "node:path";
22
+ import { join, relative, resolve as resolvePath } from "node:path";
23
23
 
24
24
  import { DEFAULT_ENV_ALLOWLIST, createRedactor } from "../redaction.js";
25
- import { sumTraceCost } from "../cost.js";
26
25
  import { createSupervisor } from "../supervisor.js";
26
+ import { splitTrace } from "../trace-split.js";
27
27
  import { installApm as defaultInstallApm } from "./apm-installer.js";
28
28
  import { installNpm as defaultInstallNpm } from "./npm-installer.js";
29
29
  import { runJudge } from "./judge.js";
@@ -31,8 +31,8 @@ import { validateResultRecord } from "./result.js";
31
31
  import { runInvariants } from "./invariants.js";
32
32
  import { runHiddenTests } from "./hidden-tests.js";
33
33
  import { runProducersAndGrade } from "./grade.js";
34
+ import { summarizeRawTrace } from "./raw-summary.js";
34
35
  import { assertJudgeProfileStaged, loadTaskFamily } from "./task-family.js";
35
- import { splitAndSummarize } from "./trace-split.js";
36
36
  import { createWorkdirManager } from "./workdir.js";
37
37
  import { CellScheduler } from "./scheduler.js";
38
38
 
@@ -77,9 +77,10 @@ export class BenchmarkRunner {
77
77
  * to `AGENT_WATCHDOG_MS`. A test injects its own to force a stall in-test.
78
78
  * @param {number} [opts.termGraceMs] - SIGTERM→SIGKILL grace (ms) for the per-task process group.
79
79
  * @param {Function} [opts.runAgent] - Test seam: replaces the agent-under-test
80
- * session. Must return `{costUsd, turns, submission, agentError?}` and
81
- * write a valid NDJSON trace to `workdir.agentTracePath`. Default uses
82
- * `createAgentRunner` with the harness `BASE_TOOLS` allowlist. Internal
80
+ * session. Must run the session, stream `{source, seq, event}` envelopes
81
+ * to `workdir.rawTracePath`, and return `{agentError?}` — cost, turns,
82
+ * and submission are always derived from the raw file by the shared
83
+ * split/summary pipeline, so the seam exercises the real path. Internal
83
84
  * testing only. It is not part of the public API.
84
85
  * @param {import("@forwardimpact/libutil/runtime").Runtime} opts.runtime -
85
86
  * The host injects these ambient collaborators (`fs`, `subprocess`,
@@ -256,12 +257,12 @@ export class BenchmarkRunner {
256
257
  t0,
257
258
  });
258
259
  } catch (e) {
259
- // `wm.start()` (port acquire + workdir/env seed) is the one throw site
260
- // that `#executeCell` does not catch. Turn it into the runner's own
261
- // fallback record so `#runOne` never rejects. The scheduler's
262
- // one-record-per-cell contract depends on that. `report` skips the
263
- // fallback because it fails the schema, the same as any other
264
- // runner-side schema failure.
260
+ // Catches the throw sites `#executeCell` does not: `wm.start()` (port
261
+ // acquire + workdir/env seed) and the shared split/summary pipeline.
262
+ // Turn either into the runner's own fallback record so `#runOne` never
263
+ // rejects. The scheduler's one-record-per-cell contract depends on
264
+ // that. `report` skips the fallback because it fails the schema, the
265
+ // same as any other runner-side schema failure.
265
266
  return {
266
267
  taskId: task.id,
267
268
  runIndex,
@@ -301,8 +302,8 @@ export class BenchmarkRunner {
301
302
  }
302
303
  {
303
304
  const agentRun = await this.#runAgentSafe(task, workdir);
304
- const { costUsd, turns, submission, agentError } = agentRun;
305
- const breakdown = agentRun.costBreakdown ?? { agent: 0, supervisor: 0 };
305
+ const { costUsd, costBreakdown, turns, submission, agentError } =
306
+ agentRun;
306
307
  const graded = await this.#gradeCell(family, task, workdir);
307
308
  const { invariants, hiddenRows, engineError, rows, grade } = graded;
308
309
  const { judgeVerdict, judgeCost } = await this.#judgeCell({
@@ -319,6 +320,7 @@ export class BenchmarkRunner {
319
320
  // or a judge that fails zeroes the effective score. Full marks does not
320
321
  // zero it. A fractional score with verdict fail is the point.
321
322
  const scoreValid = graded.healthy && grade.gatesPass && judgePass;
323
+ const tracePaths = await this.#traceRecordPaths(workdir);
322
324
  const record = {
323
325
  taskId: task.id,
324
326
  runIndex,
@@ -337,15 +339,9 @@ export class BenchmarkRunner {
337
339
  submission,
338
340
  ...(judgeVerdict && { judgeVerdict }),
339
341
  costUsd: costUsd + judgeCost,
340
- costBreakdown: {
341
- agent: breakdown.agent ?? 0,
342
- supervisor: breakdown.supervisor ?? 0,
343
- judge: judgeCost,
344
- },
342
+ costBreakdown: { ...costBreakdown, judge: judgeCost },
345
343
  turns,
346
- agentTracePath: workdir.agentTracePath,
347
- supervisorTracePath: workdir.supervisorTracePath,
348
- judgeTracePath: workdir.judgeTracePath,
344
+ ...tracePaths,
349
345
  profiles: {
350
346
  agent: this.profiles.agent,
351
347
  supervisor: null,
@@ -424,39 +420,42 @@ export class BenchmarkRunner {
424
420
  }
425
421
 
426
422
  /**
427
- * Dispatch to either the injected hook or the default `#runAgent`. Either
428
- * path can throw. Catch the error here so it becomes an `agentError` on the
429
- * record (spec criterion 1: records on agent failure). The iterator then
430
- * does not abort.
423
+ * Dispatch to either the injected hook or the default `#runAgent`, then run
424
+ * the shared pipeline once: split the preserved raw trace into lanes and
425
+ * derive cost/turns/submission from the same file. Either session path can
426
+ * throw; catch here so a session error becomes an `agentError` on the
427
+ * record rather than aborting the whole iterator — the pipeline still runs,
428
+ * so failed cells keep coherent (possibly empty) lanes and zeroed totals.
429
+ * A pipeline throw itself (the raw file is materialized at allocation, so
430
+ * only an fs-level fault) propagates to `#runOne`'s fallback record.
431
431
  */
432
432
  async #runAgentSafe(task, workdir) {
433
+ let agentError = null;
433
434
  try {
434
- if (this._runAgentHook) {
435
- const r = await this._runAgentHook(task, workdir, this);
436
- return { agentError: null, ...r };
437
- }
438
- return await this.#runAgent(task, workdir);
435
+ const r = this._runAgentHook
436
+ ? await this._runAgentHook(task, workdir, this)
437
+ : await this.#runAgent(task, workdir);
438
+ agentError = r?.agentError ?? null;
439
439
  } catch (e) {
440
- return {
441
- costUsd: 0,
442
- costBreakdown: { agent: 0, supervisor: 0 },
443
- turns: 0,
444
- submission: "",
445
- agentError: { message: e.message ?? String(e), aborted: false },
446
- };
440
+ agentError = { message: e.message ?? String(e), aborted: false };
447
441
  }
442
+ await splitTrace(this.runtime, workdir.rawTracePath, {
443
+ caseId: workdir.caseId,
444
+ outputDir: workdir.runDir,
445
+ });
446
+ const summary = await summarizeRawTrace(this.runtime, workdir.rawTracePath);
447
+ return { ...summary, agentError };
448
448
  }
449
449
 
450
450
  /**
451
- * Run the agent-under-test under a Supervisor. The supervisor writes
452
- * a combined tagged NDJSON trace. After the session, this method splits
453
- * the trace into agent.ndjson and supervisor.ndjson. It also extracts
454
- * cost/turns/submission.
451
+ * Run the agent-under-test under a Supervisor. The supervisor streams the
452
+ * combined tagged NDJSON envelope trace to `workdir.rawTracePath`, which is
453
+ * preserved for the life of the run output; `#runAgentSafe` splits it into
454
+ * the convention-named lanes and summarizes it afterwards.
455
455
  */
456
456
  async #runAgent(task, workdir) {
457
457
  const fs = this.runtime.fs;
458
- const combinedPath = join(workdir.runDir, ".combined.ndjson");
459
- const combinedStream = fs.createWriteStream(combinedPath);
458
+ const combinedStream = fs.createWriteStream(workdir.rawTracePath);
460
459
  const supervisorInstructions = task.paths.supervisor
461
460
  ? await fs.readFile(task.paths.supervisor, "utf8").catch(() => null)
462
461
  : null;
@@ -510,27 +509,31 @@ export class BenchmarkRunner {
510
509
  this.runtime.clock.clearTimeout(watchdog);
511
510
  await new Promise((r) => combinedStream.end(r));
512
511
  }
513
- const summary = await splitAndSummarize(
514
- this.runtime,
515
- combinedPath,
516
- workdir.agentTracePath,
517
- workdir.supervisorTracePath,
518
- );
519
- // `sumTraceCost` sums the cost across every participant's result events
520
- // from the one combined trace, and attributes it per source. Read the
521
- // trace before you unlink it.
522
- const combined = await fs.readFile(combinedPath, "utf8");
523
- const { totalCostUsd, bySource } = sumTraceCost(combined.split("\n"));
524
- await fs.unlink(combinedPath).catch(() => {});
525
- return {
526
- ...summary,
527
- costUsd: totalCostUsd,
528
- costBreakdown: {
529
- agent: bySource.agent ?? 0,
530
- supervisor: bySource.supervisor ?? 0,
531
- },
532
- agentError,
512
+ return { agentError };
513
+ }
514
+
515
+ /**
516
+ * Run-output-relative trace-path record fields, each present only when its
517
+ * file exists: raw and agent/supervisor lanes are materialized at workdir
518
+ * allocation (present on every executed cell); the judge lane exists only
519
+ * on judged cells. Relative paths stay valid inside a downloaded artifact.
520
+ */
521
+ async #traceRecordPaths(workdir) {
522
+ const fields = {
523
+ rawTracePath: workdir.rawTracePath,
524
+ agentTracePath: workdir.agentTracePath,
525
+ supervisorTracePath: workdir.supervisorTracePath,
526
+ judgeTracePath: workdir.judgeTracePath,
533
527
  };
528
+ const out = {};
529
+ for (const [field, absPath] of Object.entries(fields)) {
530
+ const exists = await this.runtime.fs
531
+ .access(absPath)
532
+ .then(() => true)
533
+ .catch(() => false);
534
+ if (exists) out[field] = relative(this.output, absPath);
535
+ }
536
+ return out;
534
537
  }
535
538
 
536
539
  async #buildJudgeContext(task, workdir, skillSetHash) {
@@ -579,9 +582,9 @@ export class BenchmarkRunner {
579
582
  skillSetHash,
580
583
  familyRevision,
581
584
  durationMs,
582
- agentTracePath: workdir.agentTracePath,
583
- supervisorTracePath: workdir.supervisorTracePath,
584
- judgeTracePath: workdir.judgeTracePath,
585
+ // No trace-path fields: even though the materialized stubs exist on
586
+ // disk, a preflight-failure record references only traces a session
587
+ // produced.
585
588
  };
586
589
  }
587
590
 
@@ -35,6 +35,8 @@
35
35
  import { createHash } from "node:crypto";
36
36
  import { join, posix, relative, resolve, sep } from "node:path";
37
37
 
38
+ import { isValidTaskId } from "../trace-identity.js";
39
+
38
40
  const GIT_URL_RE = /^(git@|https?:\/\/|ssh:\/\/|git:\/\/)/;
39
41
  const SKIP_DIRS = new Set([".git", "node_modules"]);
40
42
  // POSIX `X_OK` (execute permission). Node's fs honours the numeric mode, so we
@@ -129,6 +131,13 @@ async function discoverTasks(runtime, rootPath) {
129
131
  }
130
132
 
131
133
  async function loadTask(fs, taskDir, id) {
134
+ // The rule itself lives in the identity module; the loader only invokes
135
+ // the predicate so an unbuildable case id fails before any agent spend.
136
+ if (!isValidTaskId(id)) {
137
+ throw new Error(
138
+ `invalid task id '${id}': task directory names must not contain "--" or start/end with "-"`,
139
+ );
140
+ }
132
141
  const supervisorPath = join(taskDir, "supervisor.task.md");
133
142
  const judgePath = join(taskDir, "judge.task.md");
134
143
  const preflightPath = join(taskDir, "hooks", "preflight.sh");
@@ -16,6 +16,11 @@ import { createServer } from "node:net";
16
16
  import { connect } from "node:net";
17
17
  import { join } from "node:path";
18
18
 
19
+ import {
20
+ buildCaseId,
21
+ laneFilename,
22
+ rawTraceFilename,
23
+ } from "../trace-identity.js";
19
24
  import { loadEnv } from "./env-loader.js";
20
25
  import { buildHookEnv } from "./hook-env.js";
21
26
 
@@ -28,6 +33,8 @@ const DEFAULT_TERM_GRACE_MS = 5_000;
28
33
  * @property {number} port - Allocated TCP port for the agent.
29
34
  * @property {number} pgid - Process-group id captured from the preflight child.
30
35
  * @property {*} scaffold - Reserved per design § Components. v1 sets null.
36
+ * @property {string} caseId - Grid-unique case identity `<taskId>-r<runIndex>`.
37
+ * @property {string} rawTracePath - Combined raw envelope trace (kept).
31
38
  * @property {string} agentTracePath
32
39
  * @property {string} supervisorTracePath
33
40
  * @property {string} judgeTracePath
@@ -72,8 +79,7 @@ export class WorkdirManager {
72
79
  */
73
80
  async start(task, runIndex) {
74
81
  const fs = this.runtime.fs;
75
- const slug = task.id.replace("/", "__");
76
- const runDir = join(this.runOutputDir, "runs", slug, String(runIndex));
82
+ const runDir = join(this.runOutputDir, "runs", task.id, String(runIndex));
77
83
  const cwd = join(runDir, "cwd");
78
84
  await fs.mkdir(cwd, { recursive: true });
79
85
 
@@ -124,11 +130,25 @@ export class WorkdirManager {
124
130
  const envNames =
125
131
  envDirs.length > 0 ? await loadEnv(envDirs, cwd, this.runtime) : [];
126
132
 
127
- const port = await this.ports.acquire();
128
- const agentTracePath = join(runDir, "agent.ndjson");
129
- const supervisorTracePath = join(runDir, "supervisor.ndjson");
130
- const judgeTracePath = join(runDir, "judge.ndjson");
133
+ // Allocate identity and trace paths before acquiring the port so an
134
+ // invalid task id cannot leak a port reservation.
135
+ const caseId = buildCaseId(task.id, runIndex);
136
+ const rawTracePath = join(runDir, rawTraceFilename(caseId));
137
+ const agentTracePath = join(runDir, laneFilename(caseId, "agent", "agent"));
138
+ const supervisorTracePath = join(
139
+ runDir,
140
+ laneFilename(caseId, "supervisor", "supervisor"),
141
+ );
142
+ const judgeTracePath = join(runDir, laneFilename(caseId, "judge", "judge"));
143
+ // Materialize the raw and agent/supervisor lanes empty at allocation so
144
+ // every path that reaches the judge — including a pre-session agent
145
+ // failure — finds them on disk. The judge lane is written by the judge
146
+ // session itself and exists only on judged cells.
147
+ for (const p of [rawTracePath, agentTracePath, supervisorTracePath]) {
148
+ await fs.writeFile(p, "");
149
+ }
131
150
 
151
+ const port = await this.ports.acquire();
132
152
  const preflight = task.paths.preflight
133
153
  ? await runPreflight(this.runtime, task.paths.preflight, cwd, port, {
134
154
  taskId: task.id,
@@ -144,6 +164,8 @@ export class WorkdirManager {
144
164
  port,
145
165
  pgid: preflight.pgid,
146
166
  scaffold: null,
167
+ caseId,
168
+ rawTracePath,
147
169
  agentTracePath,
148
170
  supervisorTracePath,
149
171
  judgeTracePath,
@@ -166,13 +166,13 @@ export const definition = {
166
166
  documentation: [
167
167
  {
168
168
  title: "Run a Benchmark",
169
- url: "https://www.forwardimpact.team/docs/libraries/prove-changes/run-benchmark/index.md",
169
+ url: "https://www.gemba.team/docs/prove-changes/run-benchmark/index.md",
170
170
  description:
171
171
  "Author a coding-task family, run a benchmark across multiple runs, and read the pass@k report.",
172
172
  },
173
173
  {
174
174
  title: "Automate with GitHub Actions",
175
- url: "https://www.forwardimpact.team/docs/libraries/prove-changes/run-benchmark/ci-workflow/index.md",
175
+ url: "https://www.gemba.team/docs/prove-changes/run-benchmark/ci-workflow/index.md",
176
176
  description:
177
177
  "Run benchmarks in CI with the forwardimpact/gemba-benchmark action.",
178
178
  },
@@ -3,6 +3,7 @@ import { isoTimestamp } from "@forwardimpact/libutil";
3
3
  import { createTraceCollector, sumTraceCost } from "@forwardimpact/libharness";
4
4
  import { createTraceQuery } from "../trace-query.js";
5
5
  import { createTraceGitHub } from "../trace-github.js";
6
+ import { splitTrace } from "../trace-split.js";
6
7
  import { stripSignatures } from "../signature-filter.js";
7
8
  import { runOver, aggregate, compareTwo } from "../trace-multi.js";
8
9
  import {
@@ -100,7 +101,9 @@ export async function runRunsCommand(ctx) {
100
101
  }
101
102
 
102
103
  /**
103
- * Resolve a participant's lane trace for a known run id in one keyed lookup.
104
+ * Resolve a trace lane for a known run id in one keyed lookup. The key may
105
+ * be an exact member filename, a case id, or a participant name; ambiguous
106
+ * keys error with the matching candidates.
104
107
  * @param {import("@forwardimpact/libcli").InvocationContext} ctx
105
108
  */
106
109
  export async function runFindCommand(ctx) {
@@ -110,7 +113,7 @@ export async function runFindCommand(ctx) {
110
113
  repo: ctx.options.repo,
111
114
  runtime,
112
115
  });
113
- const result = await gh.findByKey(ctx.args["run-id"], ctx.args.participant, {
116
+ const result = await gh.findByKey(ctx.args["run-id"], ctx.args.key, {
114
117
  dir: ctx.options.dir,
115
118
  });
116
119
  writeJSON(runtime, result, ctx.options);
@@ -118,7 +121,22 @@ export async function runFindCommand(ctx) {
118
121
  }
119
122
 
120
123
  /**
121
- * Download a trace artifact and auto-convert to structured JSON.
124
+ * The single `.ndjson` member to auto-convert to structured JSON, or null
125
+ * when the artifact carries zero or several. Multi-member bundles (kata
126
+ * dispatch, harness matrix, eval shards) get no `structured.json` — the
127
+ * prior first-member conversion picked an arbitrary lane, which was actively
128
+ * misleading; the analysis verbs read the `.ndjson` members directly.
129
+ * @param {string[]} files - Extracted member paths, relative to the artifact dir.
130
+ * @returns {string|null}
131
+ */
132
+ export function structuredConvertTarget(files) {
133
+ const ndjson = files.filter((f) => f.endsWith(".ndjson"));
134
+ return ndjson.length === 1 ? ndjson[0] : null;
135
+ }
136
+
137
+ /**
138
+ * Download a trace artifact; auto-convert to structured JSON only when the
139
+ * artifact carries exactly one `.ndjson` member.
122
140
  * @param {import("@forwardimpact/libcli").InvocationContext} ctx
123
141
  */
124
142
  export async function runDownloadCommand(ctx) {
@@ -133,7 +151,7 @@ export async function runDownloadCommand(ctx) {
133
151
  name: ctx.options.artifact,
134
152
  });
135
153
 
136
- const ndjsonFile = result.files.find((f) => f.endsWith(".ndjson"));
154
+ const ndjsonFile = structuredConvertTarget(result.files);
137
155
  if (ndjsonFile) {
138
156
  const ndjsonPath = join(result.dir, ndjsonFile);
139
157
  const collector = createTraceCollector({
@@ -496,27 +514,13 @@ export async function runCompareCommand(ctx) {
496
514
 
497
515
  // --- Split command ---
498
516
 
499
- /**
500
- * A valid source name starts with a lowercase letter. The rest uses lowercase
501
- * alphanumeric characters or hyphens.
502
- */
503
- const VALID_SOURCE_NAME = /^[a-z][a-z0-9-]*$/;
504
-
505
- /**
506
- * Sources whose name is itself a structural role. The splitter classifies
507
- * each one into the role it represents.
508
- */
509
- const STRUCTURAL_ROLES = new Set(["agent", "supervisor", "facilitator"]);
510
-
511
517
  /**
512
518
  * Split a combined NDJSON trace into per-source files. The output names
513
519
  * follow the `trace--<case>--<participant>.<role>.ndjson` convention.
514
520
  *
515
- * Each valid envelope source becomes one output file. Structural sources
516
- * (`agent`, `supervisor`, `facilitator`) classify into the matching role.
517
- * They use their own name as participant. Profile-named sources (e.g.
518
- * `staff-engineer`) classify as agents with the profile in the participant
519
- * slot. The command drops orchestrator events and invalid source names.
521
+ * The command owns the CLI concerns only: input and `--mode` validation,
522
+ * defaults, and output-dir creation. The shared `splitTrace` implementation
523
+ * owns the split itself, including source-to-role classification.
520
524
  *
521
525
  * @param {import("@forwardimpact/libcli").InvocationContext} ctx
522
526
  */
@@ -525,10 +529,11 @@ export async function runSplitCommand(ctx) {
525
529
  const file = ctx.args.file;
526
530
  if (!file) return { ok: false, code: 1, error: "split: missing input file" };
527
531
 
528
- // `discuss` has the same lead + N-participants shape as `facilitate`. The
529
- // splitter buckets purely by envelope `source`, which is mode-independent.
530
- // So the CLI accepts `discuss` alongside the structural modes. The CLI owns
531
- // this rule. Callers do not.
532
+ // `discuss` has the same lead + N-participants shape as `facilitate`, and the
533
+ // splitter buckets purely by envelope `source` (mode-independent), so it is
534
+ // accepted alongside the structural modes. The CLI owns this, not callers.
535
+ // `--mode` stays required-but-inert: the harness action passes it and that
536
+ // surface is out of scope for the shared-split extraction.
532
537
  const mode = ctx.options.mode;
533
538
  if (!mode) return { ok: false, code: 1, error: "split: --mode is required" };
534
539
  if (!["run", "supervise", "facilitate", "discuss"].includes(mode)) {
@@ -539,52 +544,10 @@ export async function runSplitCommand(ctx) {
539
544
  const outputDir = ctx.options["output-dir"] || dirname(file);
540
545
  runtime.fsSync.mkdirSync(outputDir, { recursive: true });
541
546
 
542
- const buckets = parseBuckets(runtime.fsSync.readFileSync(file, "utf8"));
543
-
544
- for (const [source, lines] of buckets.entries()) {
545
- if (!VALID_SOURCE_NAME.test(source)) continue;
546
- const role = STRUCTURAL_ROLES.has(source) ? source : "agent";
547
- const outPath = join(
548
- outputDir,
549
- `trace--${caseId}--${source}.${role}.ndjson`,
550
- );
551
- runtime.fsSync.writeFileSync(outPath, lines.join("\n") + "\n");
552
- }
547
+ await splitTrace(runtime, file, { caseId, outputDir });
553
548
  return { ok: true };
554
549
  }
555
550
 
556
- /**
557
- * Parse NDJSON content into per-source buckets of unwrapped event lines.
558
- * Skips empty lines, malformed JSON, non-envelope lines, and orchestrator events.
559
- * @param {string} content - Raw NDJSON file content
560
- * @returns {Map<string, string[]>} source name -> array of unwrapped JSON lines
561
- */
562
- function parseBuckets(content) {
563
- const buckets = new Map();
564
-
565
- for (const raw of content.split("\n")) {
566
- const trimmed = raw.trim();
567
- if (!trimmed) continue;
568
-
569
- let envelope;
570
- try {
571
- envelope = JSON.parse(trimmed);
572
- } catch {
573
- continue;
574
- }
575
-
576
- if (!envelope.event || typeof envelope.source !== "string") continue;
577
- if (envelope.source === "orchestrator") continue;
578
-
579
- if (!buckets.has(envelope.source)) {
580
- buckets.set(envelope.source, []);
581
- }
582
- buckets.get(envelope.source).push(JSON.stringify(envelope.event));
583
- }
584
-
585
- return buckets;
586
- }
587
-
588
551
  // --- Shared helpers ---
589
552
 
590
553
  /**
package/src/index.js CHANGED
@@ -7,9 +7,9 @@ export {
7
7
  createTraceGitHub,
8
8
  detectRepoSlug,
9
9
  parseGitRemote,
10
- participantInNames,
11
10
  pickTraceArtifact,
12
11
  } from "./trace-github.js";
12
+ export { participantInNames } from "./trace-identity.js";
13
13
  export { AgentRunner, createAgentRunner } from "./agent-runner.js";
14
14
  export { resolveClaudeCodeExecutable } from "./claude-code-executable.js";
15
15
  export {
@@ -4,6 +4,8 @@ import { Readable } from "node:stream";
4
4
 
5
5
  import { isoTimestamp } from "@forwardimpact/libutil";
6
6
 
7
+ import { nameMatchesKey, participantInNames } from "./trace-identity.js";
8
+
7
9
  const API = "https://api.github.com";
8
10
 
9
11
  /**
@@ -48,14 +50,18 @@ export class TraceGitHub {
48
50
  * trace content.
49
51
  *
50
52
  * @param {object} [opts]
51
- * @param {string} [opts.pattern] - Case-insensitive regex to match workflow name (default: "kata|agent", which covers `Kata: Shift`, `Kata: Dispatch`, and any `agent`-named workflow)
53
+ * @param {string} [opts.pattern] - Case-insensitive regex to match workflow name (default: "kata|agent|eval|benchmark" — covers `Kata: Shift`, `Kata: Dispatch`, benchmark-driven eval workflows, and any `agent`-named workflow)
52
54
  * @param {number} [opts.limit=50] - Max runs to return from GitHub API
53
55
  * @param {string} [opts.lookback="7d"] - How far back to search (e.g. "7d", "24h", "2w")
54
56
  * @param {string} [opts.participant] - Participant name. When set, the method filters and annotates runs by trace lane
55
57
  * @returns {Promise<object[]>} Array of {workflow, runId, status, conclusion, createdAt, branch, url[, match]}
56
58
  */
57
59
  async listRuns(opts = {}) {
58
- const { pattern = "kata|agent", limit = 50, lookback = "7d" } = opts;
60
+ const {
61
+ pattern = "kata|agent|eval|benchmark",
62
+ limit = 50,
63
+ lookback = "7d",
64
+ } = opts;
59
65
  const cutoff = parseLookback(lookback, this.runtime.clock.now());
60
66
 
61
67
  const params = new URLSearchParams({
@@ -140,33 +146,41 @@ export class TraceGitHub {
140
146
  }
141
147
 
142
148
  // Dispatch host: one shared artifact whose members name the participant.
143
- // Download and list member filenames (names only).
149
+ // Download and list member filenames (names only). Members are nested
150
+ // relative paths (`runs/<taskId>/<idx>/trace--*` on eval artifacts), so
151
+ // match on basenames — the `trace--` prefix check never matches a nested
152
+ // path directly.
144
153
  for (const artifact of traceArtifacts) {
145
154
  const { files } = await this.downloadTrace(runId, {
146
155
  name: artifact.name,
147
156
  });
148
- if (participantInNames(files, participant)) return "confirmed";
157
+ const basenames = files.map((f) => path.basename(f));
158
+ if (participantInNames(basenames, participant)) return "confirmed";
149
159
  }
150
160
  return "omit";
151
161
  }
152
162
 
153
163
  /**
154
- * Resolve a participant's lane trace path for a known run in one keyed
155
- * lookup. The method does not enumerate runs. It does not inspect trace
156
- * content.
164
+ * Resolve a trace lane path for a known run in one keyed lookup. The method
165
+ * does not enumerate runs. It does not inspect trace content. The key is an
166
+ * exact member filename, a case id, or a participant name.
157
167
  *
158
- * Matrix host: the artifact name carries the participant (no download).
159
- * Dispatch host: download the shared `trace--*` artifact. Return the
160
- * extracted member file whose name carries the participant.
168
+ * Matrix host: the artifact name carries the key, so no download happens.
169
+ * Dispatch host: download every `trace--*` artifact and match member
170
+ * basenames against the key. Exactly one match resolves. Several matches
171
+ * throw an error that lists the candidates, so the caller narrows the key.
172
+ * This replaces a silent first-match, which returned an arbitrary cell's
173
+ * lane on eval runs, because every cell emits the same participants.
161
174
  *
162
175
  * @param {number|string} runId
163
- * @param {string} participant
176
+ * @param {string} key - Exact member filename, case id, or participant name.
164
177
  * @param {object} [opts]
165
178
  * @param {string} [opts.dir] - Output directory for a downloaded dispatch artifact
166
- * @returns {Promise<{runId: (number|string), participant: string, host: "matrix"|"dispatch", artifact: string, path: string}>}
167
- * @throws {Error} when the run has no trace artifacts, or none carries the participant's lane.
179
+ * @returns {Promise<{runId: (number|string), key: string, host: "matrix"|"dispatch", artifact: string, path: string}>}
180
+ * @throws {Error} when the run has no trace artifacts, no member matches
181
+ * the key, or several members match.
168
182
  */
169
- async findByKey(runId, participant, opts = {}) {
183
+ async findByKey(runId, key, opts = {}) {
170
184
  const url = `${API}/repos/${this.owner}/${this.repo}/actions/runs/${runId}/artifacts`;
171
185
  const data = await this.#get(url);
172
186
  const artifacts = data.artifacts ?? [];
@@ -177,41 +191,57 @@ export class TraceGitHub {
177
191
  throw new Error(`No trace artifacts for run ${runId}`);
178
192
  }
179
193
 
180
- // Matrix host: the artifact name carries the participant. No download.
194
+ // Matrix host: the artifact name carries the key. No download.
181
195
  const matrix = traceArtifacts.find((a) =>
182
- participantInNames([a.name], participant),
196
+ participantInNames([a.name], key),
183
197
  );
184
198
  if (matrix) {
185
199
  return {
186
200
  runId,
187
- participant,
201
+ key,
188
202
  host: "matrix",
189
203
  artifact: matrix.name,
190
204
  path: matrix.name,
191
205
  };
192
206
  }
193
207
 
194
- // Dispatch host: download the shared artifact and match a member filename.
208
+ // Dispatch host: download every shared artifact and collect the members
209
+ // whose basename matches the key (members are nested relative paths).
210
+ // Each artifact extracts into its own subdirectory — a shared extract
211
+ // dir would re-list earlier artifacts' members on every iteration, so a
212
+ // uniquely-matching key on a multi-artifact (sharded) run would throw a
213
+ // spurious ambiguity error.
214
+ const baseDir = opts.dir ?? `/tmp/trace-${runId}`;
215
+ const matches = [];
195
216
  for (const artifact of traceArtifacts) {
196
217
  const { dir, files } = await this.downloadTrace(runId, {
197
218
  name: artifact.name,
198
- dir: opts.dir,
219
+ dir: path.join(baseDir, artifact.name),
199
220
  });
200
- const member = files.find((f) => participantInNames([f], participant));
201
- if (member) {
202
- return {
203
- runId,
204
- participant,
205
- host: "dispatch",
206
- artifact: artifact.name,
207
- path: path.join(dir, member),
208
- };
221
+ for (const member of files) {
222
+ if (nameMatchesKey(path.basename(member), key)) {
223
+ matches.push({ artifact: artifact.name, dir, member });
224
+ }
209
225
  }
210
226
  }
211
227
 
212
- throw new Error(
213
- `No trace lane for participant "${participant}" in run ${runId}`,
214
- );
228
+ if (matches.length === 1) {
229
+ const m = matches[0];
230
+ return {
231
+ runId,
232
+ key,
233
+ host: "dispatch",
234
+ artifact: m.artifact,
235
+ path: path.join(m.dir, m.member),
236
+ };
237
+ }
238
+ if (matches.length > 1) {
239
+ const names = matches.map((m) => m.member).join(", ");
240
+ throw new Error(
241
+ `Ambiguous key "${key}" for run ${runId}: matches ${names}. Narrow the key to a case id or exact filename.`,
242
+ );
243
+ }
244
+ throw new Error(`No trace lane for key "${key}" in run ${runId}`);
215
245
  }
216
246
 
217
247
  /**
@@ -271,9 +301,9 @@ export class TraceGitHub {
271
301
  );
272
302
  }
273
303
 
274
- // List extracted files.
275
- const entries = await fs.readdir(dir);
276
- const files = entries.filter((f) => !f.endsWith(".zip"));
304
+ // List extracted files — recursively, since eval artifacts carry nested
305
+ // members (`runs/<taskId>/<idx>/trace--*`).
306
+ const files = await listExtractedFiles(this.runtime, dir);
277
307
 
278
308
  return { dir, artifact: artifact.name, files };
279
309
  }
@@ -301,34 +331,22 @@ export class TraceGitHub {
301
331
  }
302
332
 
303
333
  /**
304
- * Test whether a participant's trace lane is present in a list of names.
305
- *
306
- * This function matches the two trace-name shapes by *name* only (never by
307
- * content):
308
- * - matrix artifact name: `trace--<participant>`
309
- * - dispatch member filename: `trace--<case>--<participant>.<role>.ndjson`
310
- *
311
- * The `--` separator delimits the participant segment. The segment ends at
312
- * the next `--`, at a `.`, or at the end of the string. So a substring like
313
- * `release` does not match `release-engineer` and vice versa.
314
- *
315
- * @param {string[]} names - Artifact names or extracted member filenames.
316
- * @param {string} participant - Participant name to look for.
317
- * @returns {boolean}
334
+ * List every regular file under `dir` recursively, as paths relative to
335
+ * `dir`, excluding `*.zip` (the downloaded archive itself). Sorted for a
336
+ * deterministic member order.
337
+ * @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
338
+ * @param {string} dir
339
+ * @returns {Promise<string[]>}
318
340
  */
319
- export function participantInNames(names, participant) {
320
- return names.some((name) => {
321
- if (!name.startsWith("trace--")) return false;
322
- const rest = name.slice("trace--".length);
323
- // Matrix: `<participant>` is the whole remainder (artifact name).
324
- if (rest === participant) return true;
325
- // Dispatch: `<case>--<participant>.<role>.ndjson`.
326
- const sep = rest.indexOf("--");
327
- if (sep === -1) return false;
328
- const afterCase = rest.slice(sep + 2);
329
- const participantSegment = afterCase.split(".")[0];
330
- return participantSegment === participant;
341
+ export async function listExtractedFiles(runtime, dir) {
342
+ const entries = await runtime.fs.readdir(dir, {
343
+ recursive: true,
344
+ withFileTypes: true,
331
345
  });
346
+ return entries
347
+ .filter((e) => e.isFile() && !e.name.endsWith(".zip"))
348
+ .map((e) => path.relative(dir, path.join(e.parentPath ?? e.path, e.name)))
349
+ .sort();
332
350
  }
333
351
 
334
352
  /**
@@ -0,0 +1,132 @@
1
+ /**
2
+ * Trace identity grammar — the single owner of case ids and lane filenames.
3
+ *
4
+ * Builds case ids (`<taskId>-r<runIndex>`), builds raw/lane filenames under
5
+ * the shared `trace--` convention, validates task ids, and parses names back
6
+ * into identity. Workdir allocation, task-family loading, the shared split
7
+ * module, and GitHub discovery all invoke this module — files agreeing by
8
+ * convention is the drifted-copies pattern this module retires.
9
+ */
10
+
11
+ import { basename } from "node:path";
12
+
13
+ /**
14
+ * Task ids must not contain "--" or start/end with "-": the "--" delimiter
15
+ * and the terminal "-r<digits>" suffix then parse unambiguously.
16
+ * @param {string} id
17
+ * @returns {boolean}
18
+ */
19
+ export function isValidTaskId(id) {
20
+ if (typeof id !== "string" || id.length === 0) return false;
21
+ if (id.includes("--")) return false;
22
+ if (id.startsWith("-") || id.endsWith("-")) return false;
23
+ return true;
24
+ }
25
+
26
+ /**
27
+ * Build the grid-unique case id `<taskId>-r<runIndex>`. Shards partition one
28
+ * grid, so (task, runIndex) is already grid- and shard-unique.
29
+ * @param {string} taskId
30
+ * @param {number} runIndex
31
+ * @returns {string}
32
+ * @throws {Error} when `isValidTaskId(taskId)` is false or `runIndex` is not
33
+ * a non-negative integer.
34
+ */
35
+ export function buildCaseId(taskId, runIndex) {
36
+ if (!isValidTaskId(taskId)) {
37
+ throw new Error(
38
+ `invalid task id '${taskId}': task ids must not contain "--" or start/end with "-"`,
39
+ );
40
+ }
41
+ if (!Number.isInteger(runIndex) || runIndex < 0) {
42
+ throw new Error(`invalid run index '${runIndex}': must be an integer ≥ 0`);
43
+ }
44
+ return `${taskId}-r${runIndex}`;
45
+ }
46
+
47
+ /**
48
+ * Filename of the combined raw envelope trace: `trace--<caseId>.raw.ndjson`.
49
+ * @param {string} caseId
50
+ * @returns {string}
51
+ */
52
+ export function rawTraceFilename(caseId) {
53
+ return `trace--${caseId}.raw.ndjson`;
54
+ }
55
+
56
+ /**
57
+ * Filename of a per-participant lane:
58
+ * `trace--<caseId>--<participant>.<role>.ndjson`.
59
+ * @param {string} caseId
60
+ * @param {string} participant
61
+ * @param {string} role
62
+ * @returns {string}
63
+ */
64
+ export function laneFilename(caseId, participant, role) {
65
+ return `trace--${caseId}--${participant}.${role}.ndjson`;
66
+ }
67
+
68
+ /**
69
+ * Parse `trace--<case>--<participant>.<role>.ndjson` into `{caseName,
70
+ * participant}`. On no match, `caseName` is the basename minus its final
71
+ * `.ndjson` extension only and `participant` is null.
72
+ * @param {string} file
73
+ * @returns {{caseName: string, participant: string|null}}
74
+ */
75
+ export function parseIdentity(file) {
76
+ const name = basename(file);
77
+ const match = name.match(/^trace--(.+?)--(.+?)\.[^.]+\.ndjson$/);
78
+ if (match) {
79
+ return { caseName: match[1], participant: match[2] };
80
+ }
81
+ return { caseName: name.replace(/\.ndjson$/, ""), participant: null };
82
+ }
83
+
84
+ /**
85
+ * Test whether a participant's trace lane is present in a list of names.
86
+ *
87
+ * Matches the two trace-naming shapes by *name* only (never by content):
88
+ * - matrix artifact name: `trace--<participant>`
89
+ * - dispatch member filename: `trace--<case>--<participant>.<role>.ndjson`
90
+ *
91
+ * The participant segment is delimited by `--` and ends at the next `--`, `.`,
92
+ * or end-of-string, so a substring like `release` does not match
93
+ * `release-engineer` and vice versa.
94
+ *
95
+ * Kept as a distinct shape from {@link parseIdentity} deliberately: this
96
+ * matcher also accepts bare artifact names with no extension, which the
97
+ * filename regex cannot.
98
+ *
99
+ * @param {string[]} names - Artifact names or extracted member filenames.
100
+ * @param {string} participant - Participant name to look for.
101
+ * @returns {boolean}
102
+ */
103
+ export function participantInNames(names, participant) {
104
+ return names.some((name) => {
105
+ if (!name.startsWith("trace--")) return false;
106
+ const rest = name.slice("trace--".length);
107
+ // Matrix: `<participant>` is the whole remainder (artifact name).
108
+ if (rest === participant) return true;
109
+ // Dispatch: `<case>--<participant>.<role>.ndjson`.
110
+ const sep = rest.indexOf("--");
111
+ if (sep === -1) return false;
112
+ const afterCase = rest.slice(sep + 2);
113
+ const participantSegment = afterCase.split(".")[0];
114
+ return participantSegment === participant;
115
+ });
116
+ }
117
+
118
+ /**
119
+ * Keyed-lookup rule for one name: true when `key` equals the exact basename,
120
+ * the parsed case segment, or the parsed participant segment. Derives case
121
+ * and participant via {@link parseIdentity} and reuses the
122
+ * {@link participantInNames} single-name check — no second grammar.
123
+ * @param {string} name - A member basename or artifact name.
124
+ * @param {string} key - Exact filename, case id, or participant name.
125
+ * @returns {boolean}
126
+ */
127
+ export function nameMatchesKey(name, key) {
128
+ if (name === key) return true;
129
+ const identity = parseIdentity(name);
130
+ if (identity.participant !== null && identity.caseName === key) return true;
131
+ return participantInNames([name], key);
132
+ }
@@ -12,6 +12,8 @@
12
12
  */
13
13
  import { basename } from "node:path";
14
14
 
15
+ import { parseIdentity } from "./trace-identity.js";
16
+
15
17
  /**
16
18
  * Load each file → `TraceQuery`. Run `query(tq)`. Tag each emitted record
17
19
  * with `source: <basename>` only when the caller supplies more than one file.
@@ -84,20 +86,3 @@ export function compareTwo(a, b, load) {
84
86
  bIdentity: parseIdentity(b),
85
87
  });
86
88
  }
87
-
88
- /**
89
- * Parse `trace--<case>--<participant>.<role>.ndjson` into `{caseName,
90
- * participant}`. On no match, `caseName` is the basename without its final
91
- * `.ndjson` extension. The function removes that one extension only.
92
- * `participant` is then null.
93
- * @param {string} file
94
- * @returns {{caseName: string, participant: string|null}}
95
- */
96
- export function parseIdentity(file) {
97
- const name = basename(file);
98
- const match = name.match(/^trace--(.+?)--(.+?)\.[^.]+\.ndjson$/);
99
- if (match) {
100
- return { caseName: match[1], participant: match[2] };
101
- }
102
- return { caseName: name.replace(/\.ndjson$/, ""), participant: null };
103
- }
@@ -0,0 +1,103 @@
1
+ /**
2
+ * Shared trace-split implementation — the single owner of source-to-role
3
+ * classification. Both the `gemba-trace split` command and the benchmark
4
+ * runner drive this module, so exactly one split policy exists (the same
5
+ * treatment the one-cost-path rule gives `sumTraceCost`).
6
+ */
7
+
8
+ import { join } from "node:path";
9
+ import { createInterface } from "node:readline";
10
+
11
+ import { laneFilename } from "./trace-identity.js";
12
+
13
+ /** Valid source name pattern: lowercase letter, then lowercase alphanumeric or hyphen. */
14
+ const VALID_SOURCE_NAME = /^[a-z][a-z0-9-]*$/;
15
+
16
+ /**
17
+ * Sources whose name is itself a structural role; classified into the role
18
+ * they represent. `judge` is structural so the judge lane classifies under
19
+ * one rule — no current producer feeds judge-source envelopes through split
20
+ * (the judge is its own session), so kata split output is unchanged.
21
+ */
22
+ const STRUCTURAL_ROLES = new Set([
23
+ "agent",
24
+ "supervisor",
25
+ "facilitator",
26
+ "judge",
27
+ ]);
28
+
29
+ /**
30
+ * Split a combined `{source, seq, event}` NDJSON trace into per-source lane
31
+ * files named by `laneFilename(caseId, source, role)`.
32
+ *
33
+ * Classification: sources in the structural-role set ("agent", "supervisor",
34
+ * "facilitator", "judge") take their own name as role; any other valid source
35
+ * name classifies as role "agent" with the source as participant. Skips
36
+ * empty/malformed/non-envelope lines and orchestrator events; drops sources
37
+ * failing `/^[a-z][a-z0-9-]*$/`. Lane files carry unwrapped event JSON, one
38
+ * per line.
39
+ *
40
+ * @param {import("@forwardimpact/libutil/runtime").Runtime} runtime -
41
+ * Ambient collaborators; streams via `runtime.fs`.
42
+ * @param {string} inputPath - Combined NDJSON trace to split.
43
+ * @param {object} opts
44
+ * @param {string} opts.caseId - Case identity embedded in lane filenames.
45
+ * @param {string} opts.outputDir - Directory the lane files are written to.
46
+ * @returns {Promise<string[]>} Paths written, resolved against `outputDir`
47
+ * (absolute iff `outputDir` is absolute).
48
+ */
49
+ export async function splitTrace(runtime, inputPath, { caseId, outputDir }) {
50
+ const fs = runtime.fs;
51
+ const rl = createInterface({
52
+ input: fs.createReadStream(inputPath),
53
+ crlfDelay: Infinity,
54
+ });
55
+ const streams = new Map();
56
+ const paths = [];
57
+ for await (const line of rl) {
58
+ const envelope = parseEnvelopeLine(line);
59
+ if (!envelope) continue;
60
+ if (envelope.source === "orchestrator") continue;
61
+ if (!VALID_SOURCE_NAME.test(envelope.source)) continue;
62
+
63
+ let stream = streams.get(envelope.source);
64
+ if (!stream) {
65
+ const role = STRUCTURAL_ROLES.has(envelope.source)
66
+ ? envelope.source
67
+ : "agent";
68
+ const outPath = join(
69
+ outputDir,
70
+ laneFilename(caseId, envelope.source, role),
71
+ );
72
+ stream = fs.createWriteStream(outPath);
73
+ streams.set(envelope.source, stream);
74
+ paths.push(outPath);
75
+ }
76
+ stream.write(JSON.stringify(envelope.event) + "\n");
77
+ }
78
+ await Promise.all(
79
+ [...streams.values()].map((s) => new Promise((r) => s.end(r))),
80
+ );
81
+ return paths;
82
+ }
83
+
84
+ /**
85
+ * Parse one NDJSON line into a `{source, seq, event}` envelope, or null when
86
+ * the line is blank, malformed, or not an envelope. Shared with the
87
+ * raw-summary reader so exactly one envelope-line parser exists; callers add
88
+ * their own source filters.
89
+ * @param {string} line
90
+ * @returns {{source: string, event: object}|null}
91
+ */
92
+ export function parseEnvelopeLine(line) {
93
+ const trimmed = line.trim();
94
+ if (!trimmed) return null;
95
+ let envelope;
96
+ try {
97
+ envelope = JSON.parse(trimmed);
98
+ } catch {
99
+ return null;
100
+ }
101
+ if (!envelope.event || typeof envelope.source !== "string") return null;
102
+ return envelope;
103
+ }
@@ -1,74 +0,0 @@
1
- /**
2
- * Split the combined supervisor trace for the benchmark runner. One pass
3
- * over the tagged NDJSON envelope stream separates agent events from
4
- * supervisor and orchestrator events. The same pass extracts the run summary.
5
- */
6
-
7
- import { createInterface } from "node:readline";
8
-
9
- /**
10
- * Split the combined supervisor trace into agent and supervisor files in a
11
- * single pass. The same pass extracts the turn count and the submission.
12
- * Agent-source events go to `agentPath`. Supervisor and orchestrator events
13
- * go to `supervisorPath`.
14
- *
15
- * This function deliberately does not sum cost. The caller derives it from
16
- * the same combined trace with `sumTraceCost`. One cost path then serves the
17
- * benchmark, callback, and `gemba-trace cost` consumers.
18
- * @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
19
- * @param {string} combinedPath
20
- * @param {string} agentPath
21
- * @param {string} supervisorPath
22
- * @returns {Promise<{turns: number, submission: string}>}
23
- */
24
- // biome-ignore lint/complexity/noExcessiveCognitiveComplexity: stream-splitting state machine
25
- export async function splitAndSummarize(
26
- runtime,
27
- combinedPath,
28
- agentPath,
29
- supervisorPath,
30
- ) {
31
- const fs = runtime.fs;
32
- const agentStream = fs.createWriteStream(agentPath);
33
- const supStream = fs.createWriteStream(supervisorPath);
34
- const rl = createInterface({
35
- input: fs.createReadStream(combinedPath),
36
- crlfDelay: Infinity,
37
- });
38
- let turns = 0;
39
- let submission = "";
40
- for await (const line of rl) {
41
- if (!line.trim()) continue;
42
- let event;
43
- try {
44
- event = JSON.parse(line);
45
- } catch {
46
- continue;
47
- }
48
- const target = event.source === "agent" ? agentStream : supStream;
49
- target.write(line + "\n");
50
- const inner = event.event;
51
- if (!inner) continue;
52
- if (event.source === "agent" && inner.type === "assistant") {
53
- const text = extractText(inner);
54
- if (text) submission = text;
55
- }
56
- if (event.source === "orchestrator" && inner.type === "summary") {
57
- turns = inner.turns ?? 0;
58
- }
59
- }
60
- await Promise.all([
61
- new Promise((r) => agentStream.end(r)),
62
- new Promise((r) => supStream.end(r)),
63
- ]);
64
- return { turns, submission };
65
- }
66
-
67
- function extractText(inner) {
68
- const content = inner.message?.content ?? inner.content;
69
- if (!Array.isArray(content)) return null;
70
- for (let i = content.length - 1; i >= 0; i--) {
71
- if (content[i].type === "text" && content[i].text) return content[i].text;
72
- }
73
- return null;
74
- }