@forwardimpact/libharness 3.0.2 → 3.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +26 -4
- package/package.json +1 -1
- package/src/benchmark/judge.js +8 -6
- package/src/benchmark/raw-summary.js +79 -0
- package/src/benchmark/result.js +13 -7
- package/src/benchmark/runner.js +70 -67
- package/src/benchmark/task-family.js +9 -0
- package/src/benchmark/workdir.js +28 -6
- package/src/commands/benchmark-definition.js +2 -2
- package/src/commands/trace.js +31 -68
- package/src/index.js +1 -1
- package/src/trace-github.js +78 -60
- package/src/trace-identity.js +132 -0
- package/src/trace-multi.js +2 -17
- package/src/trace-split.js +103 -0
- package/src/benchmark/trace-split.js +0 -74
package/README.md
CHANGED
|
@@ -214,17 +214,39 @@ also lists the `Edit()` rules it tried.
|
|
|
214
214
|
|
|
215
215
|
## Documentation
|
|
216
216
|
|
|
217
|
-
- [Coordinate an Agent Team](https://www.
|
|
217
|
+
- [Coordinate an Agent Team](https://www.gemba.team/docs/coordinate-team/index.md)
|
|
218
218
|
— run a lead and N participant agents in one async session (supervise /
|
|
219
219
|
facilitate / discuss) with Ask/Answer/Announce and a single NDJSON trace.
|
|
220
|
-
- [Run an Eval](https://www.
|
|
220
|
+
- [Run an Eval](https://www.gemba.team/docs/prove-changes/run-eval/index.md)
|
|
221
221
|
— author a judge profile, run an eval locally, wire it into CI, and inspect
|
|
222
222
|
the trace it produces.
|
|
223
|
-
- [Prove Agent Changes](https://www.
|
|
223
|
+
- [Prove Agent Changes](https://www.gemba.team/docs/prove-changes/index.md)
|
|
224
224
|
— the end-to-end workflow from dataset generation through evaluation to
|
|
225
225
|
trace analysis, with multi-agent collaboration sessions.
|
|
226
|
-
- [Analyze Traces](https://www.
|
|
226
|
+
- [Analyze Traces](https://www.gemba.team/docs/prove-changes/trace-analysis/index.md)
|
|
227
227
|
— read the NDJSON traces produced by `gemba-harness` with `gemba-trace`.
|
|
228
228
|
- [Agent Teams](https://www.forwardimpact.team/docs/products/agent-teams/index.md)
|
|
229
229
|
— author the profiles consumed by `--agent-profile`, `--lead-profile`, and
|
|
230
230
|
`--agent-profiles`.
|
|
231
|
+
|
|
232
|
+
## Documentation home
|
|
233
|
+
|
|
234
|
+
libharness is an import-only library. It declares no `bin`. The
|
|
235
|
+
`gemba-harness`, `gemba-trace`, `gemba-benchmark`, and `gemba-selfedit`
|
|
236
|
+
commands ship with the Gemba product, which imports these modules. Run them
|
|
237
|
+
with `npx gemba-harness`, `npx gemba-trace`, or `npx gemba-benchmark`, or use
|
|
238
|
+
the installed `gemba-*` binaries. `gemba-selfedit` publishes no bare launcher.
|
|
239
|
+
Install `@forwardimpact/gemba` to get it.
|
|
240
|
+
|
|
241
|
+
The package publishes as `@forwardimpact/libharness` on the Forward Impact npm
|
|
242
|
+
scope. Install it with `npm install @forwardimpact/libharness`. Its task guides
|
|
243
|
+
live on the Gemba site at <https://www.gemba.team/>. The Forward Impact library
|
|
244
|
+
guide tree at <https://www.forwardimpact.team/docs/libraries/index.md> is not
|
|
245
|
+
this library's guide home.
|
|
246
|
+
|
|
247
|
+
**Decision (2026-08-26):** the split package scope and guide host are
|
|
248
|
+
deliberate. libharness stays a Gear npm package, so `package.json .homepage`
|
|
249
|
+
keeps <https://www.forwardimpact.team>. The agent-runtime guides moved to
|
|
250
|
+
gemba.team with the rest of the Gemba product. The `## Documentation` list
|
|
251
|
+
above carries the current URLs. Old `forwardimpact.team/docs/libraries/`
|
|
252
|
+
addresses forward to gemba.team.
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@forwardimpact/libharness",
|
|
3
|
-
"version": "3.0
|
|
3
|
+
"version": "3.1.0",
|
|
4
4
|
"description": "Autonomous agent team harness — coordinate a lead and participant agents in one async session, with eval, benchmark, and trace tooling to prove the changes worked.",
|
|
5
5
|
"keywords": [
|
|
6
6
|
"orchestration",
|
package/src/benchmark/judge.js
CHANGED
|
@@ -9,17 +9,19 @@
|
|
|
9
9
|
*
|
|
10
10
|
* {{AGENT_INSTRUCTIONS}} — contents of agent.task.md
|
|
11
11
|
* {{AGENT_PROFILE}} — agent profile body (empty string if none)
|
|
12
|
-
* {{AGENT_TRACE_PATH}} — path to agent
|
|
12
|
+
* {{AGENT_TRACE_PATH}} — absolute path to the cell's agent lane,
|
|
13
|
+
* trace--<case>--agent.agent.ndjson (materialized
|
|
14
|
+
* before any session runs)
|
|
13
15
|
* {{GRADE_RESULT}} — JSON grade object plus the merged check rows
|
|
14
16
|
* {{SKILL_SET_HASH}} — SHA-256 from apm.lock.yaml
|
|
15
17
|
* {{TASK_ID}} — task name (directory under tasks/)
|
|
16
18
|
* {{TASK_DIR}} — path to the agent working directory
|
|
17
19
|
*
|
|
18
|
-
* The
|
|
19
|
-
*
|
|
20
|
-
* `parseConcludeFromTrace`
|
|
21
|
-
* when the runtime ctx
|
|
22
|
-
* historical run from its judge
|
|
20
|
+
* The judge verdict is captured from the orchestration context's
|
|
21
|
+
* `concluded` flag directly — no trace parsing on the happy path.
|
|
22
|
+
* `parseConcludeFromTrace` is preserved for offline analysis and as a
|
|
23
|
+
* fallback when the runtime ctx isn't available (e.g. re-grading a
|
|
24
|
+
* historical run from its preserved judge lane file).
|
|
23
25
|
*/
|
|
24
26
|
|
|
25
27
|
import { createJudge } from "../judge.js";
|
|
@@ -0,0 +1,79 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Raw-trace summary for the benchmark runner: one post-session read of the
|
|
3
|
+
* preserved raw combined trace derives cost, turns, and submission. Named as
|
|
4
|
+
* its own module so summarization never re-entangles with splitting — the
|
|
5
|
+
* coupling that caused the original split-policy divergence.
|
|
6
|
+
*/
|
|
7
|
+
|
|
8
|
+
import { sumTraceCost } from "../cost.js";
|
|
9
|
+
import { parseEnvelopeLine } from "../trace-split.js";
|
|
10
|
+
|
|
11
|
+
/**
|
|
12
|
+
* One read of the preserved raw combined trace:
|
|
13
|
+
* cost — `sumTraceCost` over the lines (the one cost path),
|
|
14
|
+
* turns — last orchestrator-source `summary` event's `turns`,
|
|
15
|
+
* submission — last agent-source assistant text block.
|
|
16
|
+
*
|
|
17
|
+
* An empty (materialized-stub) raw file yields zeros and an empty
|
|
18
|
+
* submission; malformed and blank lines are tolerated.
|
|
19
|
+
*
|
|
20
|
+
* @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
|
|
21
|
+
* @param {string} rawTracePath
|
|
22
|
+
* @returns {Promise<{costUsd: number,
|
|
23
|
+
* costBreakdown: {agent: number, supervisor: number},
|
|
24
|
+
* turns: number, submission: string}>}
|
|
25
|
+
*/
|
|
26
|
+
export async function summarizeRawTrace(runtime, rawTracePath) {
|
|
27
|
+
const content = await runtime.fs.readFile(rawTracePath, "utf8");
|
|
28
|
+
const lines = content.split("\n");
|
|
29
|
+
const { totalCostUsd, bySource } = sumTraceCost(lines);
|
|
30
|
+
const { turns, submission } = deriveTurnsAndSubmission(lines);
|
|
31
|
+
|
|
32
|
+
return {
|
|
33
|
+
costUsd: totalCostUsd,
|
|
34
|
+
costBreakdown: {
|
|
35
|
+
agent: bySource.agent ?? 0,
|
|
36
|
+
supervisor: bySource.supervisor ?? 0,
|
|
37
|
+
},
|
|
38
|
+
turns,
|
|
39
|
+
submission,
|
|
40
|
+
};
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
/**
|
|
44
|
+
* One walk of the parsed envelope lines: the last orchestrator `summary`
|
|
45
|
+
* event's `turns` and the last agent assistant text block.
|
|
46
|
+
* @param {string[]} lines
|
|
47
|
+
* @returns {{turns: number, submission: string}}
|
|
48
|
+
*/
|
|
49
|
+
function deriveTurnsAndSubmission(lines) {
|
|
50
|
+
let turns = 0;
|
|
51
|
+
let submission = "";
|
|
52
|
+
for (const line of lines) {
|
|
53
|
+
const envelope = parseEnvelopeLine(line);
|
|
54
|
+
if (!envelope) continue;
|
|
55
|
+
const inner = envelope.event;
|
|
56
|
+
if (envelope.source === "agent" && inner.type === "assistant") {
|
|
57
|
+
const text = extractText(inner);
|
|
58
|
+
if (text) submission = text;
|
|
59
|
+
}
|
|
60
|
+
if (envelope.source === "orchestrator" && inner.type === "summary") {
|
|
61
|
+
turns = inner.turns ?? 0;
|
|
62
|
+
}
|
|
63
|
+
}
|
|
64
|
+
return { turns, submission };
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
/**
|
|
68
|
+
* Last text block of an assistant event's content, or null when none exists.
|
|
69
|
+
* @param {object} inner - Unwrapped event.
|
|
70
|
+
* @returns {string|null}
|
|
71
|
+
*/
|
|
72
|
+
function extractText(inner) {
|
|
73
|
+
const content = inner.message?.content ?? inner.content;
|
|
74
|
+
if (!Array.isArray(content)) return null;
|
|
75
|
+
for (let i = content.length - 1; i >= 0; i--) {
|
|
76
|
+
if (content[i].type === "text" && content[i].text) return content[i].text;
|
|
77
|
+
}
|
|
78
|
+
return null;
|
|
79
|
+
}
|
package/src/benchmark/result.js
CHANGED
|
@@ -104,9 +104,15 @@ const HAPPY_RECORD = z.object({
|
|
|
104
104
|
score: z.number().min(0).max(1).optional(),
|
|
105
105
|
submission: z.string(),
|
|
106
106
|
judgeVerdict: JUDGE_VERDICT_SHAPE.optional(),
|
|
107
|
+
// Trace paths are relative to the run output directory — valid on the
|
|
108
|
+
// runner and inside a downloaded trace artifact alike. Raw and
|
|
109
|
+
// agent/supervisor lanes are materialized at workdir allocation, so they
|
|
110
|
+
// are present on every executed cell; the judge lane exists only on
|
|
111
|
+
// judged cells.
|
|
112
|
+
rawTracePath: z.string(),
|
|
107
113
|
agentTracePath: z.string(),
|
|
108
114
|
supervisorTracePath: z.string(),
|
|
109
|
-
judgeTracePath: z.string(),
|
|
115
|
+
judgeTracePath: z.string().optional(),
|
|
110
116
|
agentError: AGENT_ERROR_SHAPE.optional(),
|
|
111
117
|
preflightError: z.undefined().optional(),
|
|
112
118
|
});
|
|
@@ -115,12 +121,12 @@ const PREFLIGHT_RECORD = z.object({
|
|
|
115
121
|
...COMMON_FIELDS,
|
|
116
122
|
costUsd: z.literal(0),
|
|
117
123
|
preflightError: PREFLIGHT_ERROR_SHAPE,
|
|
118
|
-
//
|
|
119
|
-
//
|
|
120
|
-
|
|
121
|
-
agentTracePath: z.
|
|
122
|
-
supervisorTracePath: z.
|
|
123
|
-
judgeTracePath: z.
|
|
124
|
+
// No trace-path fields: a preflight-failure record references only traces
|
|
125
|
+
// a session produced, even though the materialized stubs exist on disk.
|
|
126
|
+
rawTracePath: z.undefined().optional(),
|
|
127
|
+
agentTracePath: z.undefined().optional(),
|
|
128
|
+
supervisorTracePath: z.undefined().optional(),
|
|
129
|
+
judgeTracePath: z.undefined().optional(),
|
|
124
130
|
invariants: z.undefined().optional(),
|
|
125
131
|
grade: z.undefined().optional(),
|
|
126
132
|
hiddenTests: z.undefined().optional(),
|
package/src/benchmark/runner.js
CHANGED
|
@@ -19,11 +19,11 @@
|
|
|
19
19
|
* ledger. The iterator mirrors the same stream to CLI stdout.
|
|
20
20
|
*/
|
|
21
21
|
|
|
22
|
-
import { join, resolve as resolvePath } from "node:path";
|
|
22
|
+
import { join, relative, resolve as resolvePath } from "node:path";
|
|
23
23
|
|
|
24
24
|
import { DEFAULT_ENV_ALLOWLIST, createRedactor } from "../redaction.js";
|
|
25
|
-
import { sumTraceCost } from "../cost.js";
|
|
26
25
|
import { createSupervisor } from "../supervisor.js";
|
|
26
|
+
import { splitTrace } from "../trace-split.js";
|
|
27
27
|
import { installApm as defaultInstallApm } from "./apm-installer.js";
|
|
28
28
|
import { installNpm as defaultInstallNpm } from "./npm-installer.js";
|
|
29
29
|
import { runJudge } from "./judge.js";
|
|
@@ -31,8 +31,8 @@ import { validateResultRecord } from "./result.js";
|
|
|
31
31
|
import { runInvariants } from "./invariants.js";
|
|
32
32
|
import { runHiddenTests } from "./hidden-tests.js";
|
|
33
33
|
import { runProducersAndGrade } from "./grade.js";
|
|
34
|
+
import { summarizeRawTrace } from "./raw-summary.js";
|
|
34
35
|
import { assertJudgeProfileStaged, loadTaskFamily } from "./task-family.js";
|
|
35
|
-
import { splitAndSummarize } from "./trace-split.js";
|
|
36
36
|
import { createWorkdirManager } from "./workdir.js";
|
|
37
37
|
import { CellScheduler } from "./scheduler.js";
|
|
38
38
|
|
|
@@ -77,9 +77,10 @@ export class BenchmarkRunner {
|
|
|
77
77
|
* to `AGENT_WATCHDOG_MS`. A test injects its own to force a stall in-test.
|
|
78
78
|
* @param {number} [opts.termGraceMs] - SIGTERM→SIGKILL grace (ms) for the per-task process group.
|
|
79
79
|
* @param {Function} [opts.runAgent] - Test seam: replaces the agent-under-test
|
|
80
|
-
* session. Must
|
|
81
|
-
*
|
|
82
|
-
*
|
|
80
|
+
* session. Must run the session, stream `{source, seq, event}` envelopes
|
|
81
|
+
* to `workdir.rawTracePath`, and return `{agentError?}` — cost, turns,
|
|
82
|
+
* and submission are always derived from the raw file by the shared
|
|
83
|
+
* split/summary pipeline, so the seam exercises the real path. Internal
|
|
83
84
|
* testing only. It is not part of the public API.
|
|
84
85
|
* @param {import("@forwardimpact/libutil/runtime").Runtime} opts.runtime -
|
|
85
86
|
* The host injects these ambient collaborators (`fs`, `subprocess`,
|
|
@@ -256,12 +257,12 @@ export class BenchmarkRunner {
|
|
|
256
257
|
t0,
|
|
257
258
|
});
|
|
258
259
|
} catch (e) {
|
|
259
|
-
// `wm.start()` (port
|
|
260
|
-
//
|
|
261
|
-
// fallback record so `#runOne` never
|
|
262
|
-
// one-record-per-cell contract depends on
|
|
263
|
-
// fallback because it fails the schema, the
|
|
264
|
-
// runner-side schema failure.
|
|
260
|
+
// Catches the throw sites `#executeCell` does not: `wm.start()` (port
|
|
261
|
+
// acquire + workdir/env seed) and the shared split/summary pipeline.
|
|
262
|
+
// Turn either into the runner's own fallback record so `#runOne` never
|
|
263
|
+
// rejects. The scheduler's one-record-per-cell contract depends on
|
|
264
|
+
// that. `report` skips the fallback because it fails the schema, the
|
|
265
|
+
// same as any other runner-side schema failure.
|
|
265
266
|
return {
|
|
266
267
|
taskId: task.id,
|
|
267
268
|
runIndex,
|
|
@@ -301,8 +302,8 @@ export class BenchmarkRunner {
|
|
|
301
302
|
}
|
|
302
303
|
{
|
|
303
304
|
const agentRun = await this.#runAgentSafe(task, workdir);
|
|
304
|
-
const { costUsd, turns, submission, agentError } =
|
|
305
|
-
|
|
305
|
+
const { costUsd, costBreakdown, turns, submission, agentError } =
|
|
306
|
+
agentRun;
|
|
306
307
|
const graded = await this.#gradeCell(family, task, workdir);
|
|
307
308
|
const { invariants, hiddenRows, engineError, rows, grade } = graded;
|
|
308
309
|
const { judgeVerdict, judgeCost } = await this.#judgeCell({
|
|
@@ -319,6 +320,7 @@ export class BenchmarkRunner {
|
|
|
319
320
|
// or a judge that fails zeroes the effective score. Full marks does not
|
|
320
321
|
// zero it. A fractional score with verdict fail is the point.
|
|
321
322
|
const scoreValid = graded.healthy && grade.gatesPass && judgePass;
|
|
323
|
+
const tracePaths = await this.#traceRecordPaths(workdir);
|
|
322
324
|
const record = {
|
|
323
325
|
taskId: task.id,
|
|
324
326
|
runIndex,
|
|
@@ -337,15 +339,9 @@ export class BenchmarkRunner {
|
|
|
337
339
|
submission,
|
|
338
340
|
...(judgeVerdict && { judgeVerdict }),
|
|
339
341
|
costUsd: costUsd + judgeCost,
|
|
340
|
-
costBreakdown: {
|
|
341
|
-
agent: breakdown.agent ?? 0,
|
|
342
|
-
supervisor: breakdown.supervisor ?? 0,
|
|
343
|
-
judge: judgeCost,
|
|
344
|
-
},
|
|
342
|
+
costBreakdown: { ...costBreakdown, judge: judgeCost },
|
|
345
343
|
turns,
|
|
346
|
-
|
|
347
|
-
supervisorTracePath: workdir.supervisorTracePath,
|
|
348
|
-
judgeTracePath: workdir.judgeTracePath,
|
|
344
|
+
...tracePaths,
|
|
349
345
|
profiles: {
|
|
350
346
|
agent: this.profiles.agent,
|
|
351
347
|
supervisor: null,
|
|
@@ -424,39 +420,42 @@ export class BenchmarkRunner {
|
|
|
424
420
|
}
|
|
425
421
|
|
|
426
422
|
/**
|
|
427
|
-
* Dispatch to either the injected hook or the default `#runAgent
|
|
428
|
-
*
|
|
429
|
-
*
|
|
430
|
-
*
|
|
423
|
+
* Dispatch to either the injected hook or the default `#runAgent`, then run
|
|
424
|
+
* the shared pipeline once: split the preserved raw trace into lanes and
|
|
425
|
+
* derive cost/turns/submission from the same file. Either session path can
|
|
426
|
+
* throw; catch here so a session error becomes an `agentError` on the
|
|
427
|
+
* record rather than aborting the whole iterator — the pipeline still runs,
|
|
428
|
+
* so failed cells keep coherent (possibly empty) lanes and zeroed totals.
|
|
429
|
+
* A pipeline throw itself (the raw file is materialized at allocation, so
|
|
430
|
+
* only an fs-level fault) propagates to `#runOne`'s fallback record.
|
|
431
431
|
*/
|
|
432
432
|
async #runAgentSafe(task, workdir) {
|
|
433
|
+
let agentError = null;
|
|
433
434
|
try {
|
|
434
|
-
|
|
435
|
-
|
|
436
|
-
|
|
437
|
-
|
|
438
|
-
return await this.#runAgent(task, workdir);
|
|
435
|
+
const r = this._runAgentHook
|
|
436
|
+
? await this._runAgentHook(task, workdir, this)
|
|
437
|
+
: await this.#runAgent(task, workdir);
|
|
438
|
+
agentError = r?.agentError ?? null;
|
|
439
439
|
} catch (e) {
|
|
440
|
-
|
|
441
|
-
costUsd: 0,
|
|
442
|
-
costBreakdown: { agent: 0, supervisor: 0 },
|
|
443
|
-
turns: 0,
|
|
444
|
-
submission: "",
|
|
445
|
-
agentError: { message: e.message ?? String(e), aborted: false },
|
|
446
|
-
};
|
|
440
|
+
agentError = { message: e.message ?? String(e), aborted: false };
|
|
447
441
|
}
|
|
442
|
+
await splitTrace(this.runtime, workdir.rawTracePath, {
|
|
443
|
+
caseId: workdir.caseId,
|
|
444
|
+
outputDir: workdir.runDir,
|
|
445
|
+
});
|
|
446
|
+
const summary = await summarizeRawTrace(this.runtime, workdir.rawTracePath);
|
|
447
|
+
return { ...summary, agentError };
|
|
448
448
|
}
|
|
449
449
|
|
|
450
450
|
/**
|
|
451
|
-
* Run the agent-under-test under a Supervisor. The supervisor
|
|
452
|
-
*
|
|
453
|
-
* the
|
|
454
|
-
*
|
|
451
|
+
* Run the agent-under-test under a Supervisor. The supervisor streams the
|
|
452
|
+
* combined tagged NDJSON envelope trace to `workdir.rawTracePath`, which is
|
|
453
|
+
* preserved for the life of the run output; `#runAgentSafe` splits it into
|
|
454
|
+
* the convention-named lanes and summarizes it afterwards.
|
|
455
455
|
*/
|
|
456
456
|
async #runAgent(task, workdir) {
|
|
457
457
|
const fs = this.runtime.fs;
|
|
458
|
-
const
|
|
459
|
-
const combinedStream = fs.createWriteStream(combinedPath);
|
|
458
|
+
const combinedStream = fs.createWriteStream(workdir.rawTracePath);
|
|
460
459
|
const supervisorInstructions = task.paths.supervisor
|
|
461
460
|
? await fs.readFile(task.paths.supervisor, "utf8").catch(() => null)
|
|
462
461
|
: null;
|
|
@@ -510,27 +509,31 @@ export class BenchmarkRunner {
|
|
|
510
509
|
this.runtime.clock.clearTimeout(watchdog);
|
|
511
510
|
await new Promise((r) => combinedStream.end(r));
|
|
512
511
|
}
|
|
513
|
-
|
|
514
|
-
|
|
515
|
-
|
|
516
|
-
|
|
517
|
-
|
|
518
|
-
|
|
519
|
-
|
|
520
|
-
|
|
521
|
-
|
|
522
|
-
|
|
523
|
-
const
|
|
524
|
-
|
|
525
|
-
|
|
526
|
-
|
|
527
|
-
|
|
528
|
-
costBreakdown: {
|
|
529
|
-
agent: bySource.agent ?? 0,
|
|
530
|
-
supervisor: bySource.supervisor ?? 0,
|
|
531
|
-
},
|
|
532
|
-
agentError,
|
|
512
|
+
return { agentError };
|
|
513
|
+
}
|
|
514
|
+
|
|
515
|
+
/**
|
|
516
|
+
* Run-output-relative trace-path record fields, each present only when its
|
|
517
|
+
* file exists: raw and agent/supervisor lanes are materialized at workdir
|
|
518
|
+
* allocation (present on every executed cell); the judge lane exists only
|
|
519
|
+
* on judged cells. Relative paths stay valid inside a downloaded artifact.
|
|
520
|
+
*/
|
|
521
|
+
async #traceRecordPaths(workdir) {
|
|
522
|
+
const fields = {
|
|
523
|
+
rawTracePath: workdir.rawTracePath,
|
|
524
|
+
agentTracePath: workdir.agentTracePath,
|
|
525
|
+
supervisorTracePath: workdir.supervisorTracePath,
|
|
526
|
+
judgeTracePath: workdir.judgeTracePath,
|
|
533
527
|
};
|
|
528
|
+
const out = {};
|
|
529
|
+
for (const [field, absPath] of Object.entries(fields)) {
|
|
530
|
+
const exists = await this.runtime.fs
|
|
531
|
+
.access(absPath)
|
|
532
|
+
.then(() => true)
|
|
533
|
+
.catch(() => false);
|
|
534
|
+
if (exists) out[field] = relative(this.output, absPath);
|
|
535
|
+
}
|
|
536
|
+
return out;
|
|
534
537
|
}
|
|
535
538
|
|
|
536
539
|
async #buildJudgeContext(task, workdir, skillSetHash) {
|
|
@@ -579,9 +582,9 @@ export class BenchmarkRunner {
|
|
|
579
582
|
skillSetHash,
|
|
580
583
|
familyRevision,
|
|
581
584
|
durationMs,
|
|
582
|
-
|
|
583
|
-
|
|
584
|
-
|
|
585
|
+
// No trace-path fields: even though the materialized stubs exist on
|
|
586
|
+
// disk, a preflight-failure record references only traces a session
|
|
587
|
+
// produced.
|
|
585
588
|
};
|
|
586
589
|
}
|
|
587
590
|
|
|
@@ -35,6 +35,8 @@
|
|
|
35
35
|
import { createHash } from "node:crypto";
|
|
36
36
|
import { join, posix, relative, resolve, sep } from "node:path";
|
|
37
37
|
|
|
38
|
+
import { isValidTaskId } from "../trace-identity.js";
|
|
39
|
+
|
|
38
40
|
const GIT_URL_RE = /^(git@|https?:\/\/|ssh:\/\/|git:\/\/)/;
|
|
39
41
|
const SKIP_DIRS = new Set([".git", "node_modules"]);
|
|
40
42
|
// POSIX `X_OK` (execute permission). Node's fs honours the numeric mode, so we
|
|
@@ -129,6 +131,13 @@ async function discoverTasks(runtime, rootPath) {
|
|
|
129
131
|
}
|
|
130
132
|
|
|
131
133
|
async function loadTask(fs, taskDir, id) {
|
|
134
|
+
// The rule itself lives in the identity module; the loader only invokes
|
|
135
|
+
// the predicate so an unbuildable case id fails before any agent spend.
|
|
136
|
+
if (!isValidTaskId(id)) {
|
|
137
|
+
throw new Error(
|
|
138
|
+
`invalid task id '${id}': task directory names must not contain "--" or start/end with "-"`,
|
|
139
|
+
);
|
|
140
|
+
}
|
|
132
141
|
const supervisorPath = join(taskDir, "supervisor.task.md");
|
|
133
142
|
const judgePath = join(taskDir, "judge.task.md");
|
|
134
143
|
const preflightPath = join(taskDir, "hooks", "preflight.sh");
|
package/src/benchmark/workdir.js
CHANGED
|
@@ -16,6 +16,11 @@ import { createServer } from "node:net";
|
|
|
16
16
|
import { connect } from "node:net";
|
|
17
17
|
import { join } from "node:path";
|
|
18
18
|
|
|
19
|
+
import {
|
|
20
|
+
buildCaseId,
|
|
21
|
+
laneFilename,
|
|
22
|
+
rawTraceFilename,
|
|
23
|
+
} from "../trace-identity.js";
|
|
19
24
|
import { loadEnv } from "./env-loader.js";
|
|
20
25
|
import { buildHookEnv } from "./hook-env.js";
|
|
21
26
|
|
|
@@ -28,6 +33,8 @@ const DEFAULT_TERM_GRACE_MS = 5_000;
|
|
|
28
33
|
* @property {number} port - Allocated TCP port for the agent.
|
|
29
34
|
* @property {number} pgid - Process-group id captured from the preflight child.
|
|
30
35
|
* @property {*} scaffold - Reserved per design § Components. v1 sets null.
|
|
36
|
+
* @property {string} caseId - Grid-unique case identity `<taskId>-r<runIndex>`.
|
|
37
|
+
* @property {string} rawTracePath - Combined raw envelope trace (kept).
|
|
31
38
|
* @property {string} agentTracePath
|
|
32
39
|
* @property {string} supervisorTracePath
|
|
33
40
|
* @property {string} judgeTracePath
|
|
@@ -72,8 +79,7 @@ export class WorkdirManager {
|
|
|
72
79
|
*/
|
|
73
80
|
async start(task, runIndex) {
|
|
74
81
|
const fs = this.runtime.fs;
|
|
75
|
-
const
|
|
76
|
-
const runDir = join(this.runOutputDir, "runs", slug, String(runIndex));
|
|
82
|
+
const runDir = join(this.runOutputDir, "runs", task.id, String(runIndex));
|
|
77
83
|
const cwd = join(runDir, "cwd");
|
|
78
84
|
await fs.mkdir(cwd, { recursive: true });
|
|
79
85
|
|
|
@@ -124,11 +130,25 @@ export class WorkdirManager {
|
|
|
124
130
|
const envNames =
|
|
125
131
|
envDirs.length > 0 ? await loadEnv(envDirs, cwd, this.runtime) : [];
|
|
126
132
|
|
|
127
|
-
|
|
128
|
-
|
|
129
|
-
const
|
|
130
|
-
const
|
|
133
|
+
// Allocate identity and trace paths before acquiring the port so an
|
|
134
|
+
// invalid task id cannot leak a port reservation.
|
|
135
|
+
const caseId = buildCaseId(task.id, runIndex);
|
|
136
|
+
const rawTracePath = join(runDir, rawTraceFilename(caseId));
|
|
137
|
+
const agentTracePath = join(runDir, laneFilename(caseId, "agent", "agent"));
|
|
138
|
+
const supervisorTracePath = join(
|
|
139
|
+
runDir,
|
|
140
|
+
laneFilename(caseId, "supervisor", "supervisor"),
|
|
141
|
+
);
|
|
142
|
+
const judgeTracePath = join(runDir, laneFilename(caseId, "judge", "judge"));
|
|
143
|
+
// Materialize the raw and agent/supervisor lanes empty at allocation so
|
|
144
|
+
// every path that reaches the judge — including a pre-session agent
|
|
145
|
+
// failure — finds them on disk. The judge lane is written by the judge
|
|
146
|
+
// session itself and exists only on judged cells.
|
|
147
|
+
for (const p of [rawTracePath, agentTracePath, supervisorTracePath]) {
|
|
148
|
+
await fs.writeFile(p, "");
|
|
149
|
+
}
|
|
131
150
|
|
|
151
|
+
const port = await this.ports.acquire();
|
|
132
152
|
const preflight = task.paths.preflight
|
|
133
153
|
? await runPreflight(this.runtime, task.paths.preflight, cwd, port, {
|
|
134
154
|
taskId: task.id,
|
|
@@ -144,6 +164,8 @@ export class WorkdirManager {
|
|
|
144
164
|
port,
|
|
145
165
|
pgid: preflight.pgid,
|
|
146
166
|
scaffold: null,
|
|
167
|
+
caseId,
|
|
168
|
+
rawTracePath,
|
|
147
169
|
agentTracePath,
|
|
148
170
|
supervisorTracePath,
|
|
149
171
|
judgeTracePath,
|
|
@@ -166,13 +166,13 @@ export const definition = {
|
|
|
166
166
|
documentation: [
|
|
167
167
|
{
|
|
168
168
|
title: "Run a Benchmark",
|
|
169
|
-
url: "https://www.
|
|
169
|
+
url: "https://www.gemba.team/docs/prove-changes/run-benchmark/index.md",
|
|
170
170
|
description:
|
|
171
171
|
"Author a coding-task family, run a benchmark across multiple runs, and read the pass@k report.",
|
|
172
172
|
},
|
|
173
173
|
{
|
|
174
174
|
title: "Automate with GitHub Actions",
|
|
175
|
-
url: "https://www.
|
|
175
|
+
url: "https://www.gemba.team/docs/prove-changes/run-benchmark/ci-workflow/index.md",
|
|
176
176
|
description:
|
|
177
177
|
"Run benchmarks in CI with the forwardimpact/gemba-benchmark action.",
|
|
178
178
|
},
|
package/src/commands/trace.js
CHANGED
|
@@ -3,6 +3,7 @@ import { isoTimestamp } from "@forwardimpact/libutil";
|
|
|
3
3
|
import { createTraceCollector, sumTraceCost } from "@forwardimpact/libharness";
|
|
4
4
|
import { createTraceQuery } from "../trace-query.js";
|
|
5
5
|
import { createTraceGitHub } from "../trace-github.js";
|
|
6
|
+
import { splitTrace } from "../trace-split.js";
|
|
6
7
|
import { stripSignatures } from "../signature-filter.js";
|
|
7
8
|
import { runOver, aggregate, compareTwo } from "../trace-multi.js";
|
|
8
9
|
import {
|
|
@@ -100,7 +101,9 @@ export async function runRunsCommand(ctx) {
|
|
|
100
101
|
}
|
|
101
102
|
|
|
102
103
|
/**
|
|
103
|
-
* Resolve a
|
|
104
|
+
* Resolve a trace lane for a known run id in one keyed lookup. The key may
|
|
105
|
+
* be an exact member filename, a case id, or a participant name; ambiguous
|
|
106
|
+
* keys error with the matching candidates.
|
|
104
107
|
* @param {import("@forwardimpact/libcli").InvocationContext} ctx
|
|
105
108
|
*/
|
|
106
109
|
export async function runFindCommand(ctx) {
|
|
@@ -110,7 +113,7 @@ export async function runFindCommand(ctx) {
|
|
|
110
113
|
repo: ctx.options.repo,
|
|
111
114
|
runtime,
|
|
112
115
|
});
|
|
113
|
-
const result = await gh.findByKey(ctx.args["run-id"], ctx.args.
|
|
116
|
+
const result = await gh.findByKey(ctx.args["run-id"], ctx.args.key, {
|
|
114
117
|
dir: ctx.options.dir,
|
|
115
118
|
});
|
|
116
119
|
writeJSON(runtime, result, ctx.options);
|
|
@@ -118,7 +121,22 @@ export async function runFindCommand(ctx) {
|
|
|
118
121
|
}
|
|
119
122
|
|
|
120
123
|
/**
|
|
121
|
-
*
|
|
124
|
+
* The single `.ndjson` member to auto-convert to structured JSON, or null
|
|
125
|
+
* when the artifact carries zero or several. Multi-member bundles (kata
|
|
126
|
+
* dispatch, harness matrix, eval shards) get no `structured.json` — the
|
|
127
|
+
* prior first-member conversion picked an arbitrary lane, which was actively
|
|
128
|
+
* misleading; the analysis verbs read the `.ndjson` members directly.
|
|
129
|
+
* @param {string[]} files - Extracted member paths, relative to the artifact dir.
|
|
130
|
+
* @returns {string|null}
|
|
131
|
+
*/
|
|
132
|
+
export function structuredConvertTarget(files) {
|
|
133
|
+
const ndjson = files.filter((f) => f.endsWith(".ndjson"));
|
|
134
|
+
return ndjson.length === 1 ? ndjson[0] : null;
|
|
135
|
+
}
|
|
136
|
+
|
|
137
|
+
/**
|
|
138
|
+
* Download a trace artifact; auto-convert to structured JSON only when the
|
|
139
|
+
* artifact carries exactly one `.ndjson` member.
|
|
122
140
|
* @param {import("@forwardimpact/libcli").InvocationContext} ctx
|
|
123
141
|
*/
|
|
124
142
|
export async function runDownloadCommand(ctx) {
|
|
@@ -133,7 +151,7 @@ export async function runDownloadCommand(ctx) {
|
|
|
133
151
|
name: ctx.options.artifact,
|
|
134
152
|
});
|
|
135
153
|
|
|
136
|
-
const ndjsonFile = result.files
|
|
154
|
+
const ndjsonFile = structuredConvertTarget(result.files);
|
|
137
155
|
if (ndjsonFile) {
|
|
138
156
|
const ndjsonPath = join(result.dir, ndjsonFile);
|
|
139
157
|
const collector = createTraceCollector({
|
|
@@ -496,27 +514,13 @@ export async function runCompareCommand(ctx) {
|
|
|
496
514
|
|
|
497
515
|
// --- Split command ---
|
|
498
516
|
|
|
499
|
-
/**
|
|
500
|
-
* A valid source name starts with a lowercase letter. The rest uses lowercase
|
|
501
|
-
* alphanumeric characters or hyphens.
|
|
502
|
-
*/
|
|
503
|
-
const VALID_SOURCE_NAME = /^[a-z][a-z0-9-]*$/;
|
|
504
|
-
|
|
505
|
-
/**
|
|
506
|
-
* Sources whose name is itself a structural role. The splitter classifies
|
|
507
|
-
* each one into the role it represents.
|
|
508
|
-
*/
|
|
509
|
-
const STRUCTURAL_ROLES = new Set(["agent", "supervisor", "facilitator"]);
|
|
510
|
-
|
|
511
517
|
/**
|
|
512
518
|
* Split a combined NDJSON trace into per-source files. The output names
|
|
513
519
|
* follow the `trace--<case>--<participant>.<role>.ndjson` convention.
|
|
514
520
|
*
|
|
515
|
-
*
|
|
516
|
-
*
|
|
517
|
-
*
|
|
518
|
-
* `staff-engineer`) classify as agents with the profile in the participant
|
|
519
|
-
* slot. The command drops orchestrator events and invalid source names.
|
|
521
|
+
* The command owns the CLI concerns only: input and `--mode` validation,
|
|
522
|
+
* defaults, and output-dir creation. The shared `splitTrace` implementation
|
|
523
|
+
* owns the split itself, including source-to-role classification.
|
|
520
524
|
*
|
|
521
525
|
* @param {import("@forwardimpact/libcli").InvocationContext} ctx
|
|
522
526
|
*/
|
|
@@ -525,10 +529,11 @@ export async function runSplitCommand(ctx) {
|
|
|
525
529
|
const file = ctx.args.file;
|
|
526
530
|
if (!file) return { ok: false, code: 1, error: "split: missing input file" };
|
|
527
531
|
|
|
528
|
-
// `discuss` has the same lead + N-participants shape as `facilitate
|
|
529
|
-
// splitter buckets purely by envelope `source
|
|
530
|
-
//
|
|
531
|
-
//
|
|
532
|
+
// `discuss` has the same lead + N-participants shape as `facilitate`, and the
|
|
533
|
+
// splitter buckets purely by envelope `source` (mode-independent), so it is
|
|
534
|
+
// accepted alongside the structural modes. The CLI owns this, not callers.
|
|
535
|
+
// `--mode` stays required-but-inert: the harness action passes it and that
|
|
536
|
+
// surface is out of scope for the shared-split extraction.
|
|
532
537
|
const mode = ctx.options.mode;
|
|
533
538
|
if (!mode) return { ok: false, code: 1, error: "split: --mode is required" };
|
|
534
539
|
if (!["run", "supervise", "facilitate", "discuss"].includes(mode)) {
|
|
@@ -539,52 +544,10 @@ export async function runSplitCommand(ctx) {
|
|
|
539
544
|
const outputDir = ctx.options["output-dir"] || dirname(file);
|
|
540
545
|
runtime.fsSync.mkdirSync(outputDir, { recursive: true });
|
|
541
546
|
|
|
542
|
-
|
|
543
|
-
|
|
544
|
-
for (const [source, lines] of buckets.entries()) {
|
|
545
|
-
if (!VALID_SOURCE_NAME.test(source)) continue;
|
|
546
|
-
const role = STRUCTURAL_ROLES.has(source) ? source : "agent";
|
|
547
|
-
const outPath = join(
|
|
548
|
-
outputDir,
|
|
549
|
-
`trace--${caseId}--${source}.${role}.ndjson`,
|
|
550
|
-
);
|
|
551
|
-
runtime.fsSync.writeFileSync(outPath, lines.join("\n") + "\n");
|
|
552
|
-
}
|
|
547
|
+
await splitTrace(runtime, file, { caseId, outputDir });
|
|
553
548
|
return { ok: true };
|
|
554
549
|
}
|
|
555
550
|
|
|
556
|
-
/**
|
|
557
|
-
* Parse NDJSON content into per-source buckets of unwrapped event lines.
|
|
558
|
-
* Skips empty lines, malformed JSON, non-envelope lines, and orchestrator events.
|
|
559
|
-
* @param {string} content - Raw NDJSON file content
|
|
560
|
-
* @returns {Map<string, string[]>} source name -> array of unwrapped JSON lines
|
|
561
|
-
*/
|
|
562
|
-
function parseBuckets(content) {
|
|
563
|
-
const buckets = new Map();
|
|
564
|
-
|
|
565
|
-
for (const raw of content.split("\n")) {
|
|
566
|
-
const trimmed = raw.trim();
|
|
567
|
-
if (!trimmed) continue;
|
|
568
|
-
|
|
569
|
-
let envelope;
|
|
570
|
-
try {
|
|
571
|
-
envelope = JSON.parse(trimmed);
|
|
572
|
-
} catch {
|
|
573
|
-
continue;
|
|
574
|
-
}
|
|
575
|
-
|
|
576
|
-
if (!envelope.event || typeof envelope.source !== "string") continue;
|
|
577
|
-
if (envelope.source === "orchestrator") continue;
|
|
578
|
-
|
|
579
|
-
if (!buckets.has(envelope.source)) {
|
|
580
|
-
buckets.set(envelope.source, []);
|
|
581
|
-
}
|
|
582
|
-
buckets.get(envelope.source).push(JSON.stringify(envelope.event));
|
|
583
|
-
}
|
|
584
|
-
|
|
585
|
-
return buckets;
|
|
586
|
-
}
|
|
587
|
-
|
|
588
551
|
// --- Shared helpers ---
|
|
589
552
|
|
|
590
553
|
/**
|
package/src/index.js
CHANGED
|
@@ -7,9 +7,9 @@ export {
|
|
|
7
7
|
createTraceGitHub,
|
|
8
8
|
detectRepoSlug,
|
|
9
9
|
parseGitRemote,
|
|
10
|
-
participantInNames,
|
|
11
10
|
pickTraceArtifact,
|
|
12
11
|
} from "./trace-github.js";
|
|
12
|
+
export { participantInNames } from "./trace-identity.js";
|
|
13
13
|
export { AgentRunner, createAgentRunner } from "./agent-runner.js";
|
|
14
14
|
export { resolveClaudeCodeExecutable } from "./claude-code-executable.js";
|
|
15
15
|
export {
|
package/src/trace-github.js
CHANGED
|
@@ -4,6 +4,8 @@ import { Readable } from "node:stream";
|
|
|
4
4
|
|
|
5
5
|
import { isoTimestamp } from "@forwardimpact/libutil";
|
|
6
6
|
|
|
7
|
+
import { nameMatchesKey, participantInNames } from "./trace-identity.js";
|
|
8
|
+
|
|
7
9
|
const API = "https://api.github.com";
|
|
8
10
|
|
|
9
11
|
/**
|
|
@@ -48,14 +50,18 @@ export class TraceGitHub {
|
|
|
48
50
|
* trace content.
|
|
49
51
|
*
|
|
50
52
|
* @param {object} [opts]
|
|
51
|
-
* @param {string} [opts.pattern] - Case-insensitive regex to match workflow name (default: "kata|agent"
|
|
53
|
+
* @param {string} [opts.pattern] - Case-insensitive regex to match workflow name (default: "kata|agent|eval|benchmark" — covers `Kata: Shift`, `Kata: Dispatch`, benchmark-driven eval workflows, and any `agent`-named workflow)
|
|
52
54
|
* @param {number} [opts.limit=50] - Max runs to return from GitHub API
|
|
53
55
|
* @param {string} [opts.lookback="7d"] - How far back to search (e.g. "7d", "24h", "2w")
|
|
54
56
|
* @param {string} [opts.participant] - Participant name. When set, the method filters and annotates runs by trace lane
|
|
55
57
|
* @returns {Promise<object[]>} Array of {workflow, runId, status, conclusion, createdAt, branch, url[, match]}
|
|
56
58
|
*/
|
|
57
59
|
async listRuns(opts = {}) {
|
|
58
|
-
const {
|
|
60
|
+
const {
|
|
61
|
+
pattern = "kata|agent|eval|benchmark",
|
|
62
|
+
limit = 50,
|
|
63
|
+
lookback = "7d",
|
|
64
|
+
} = opts;
|
|
59
65
|
const cutoff = parseLookback(lookback, this.runtime.clock.now());
|
|
60
66
|
|
|
61
67
|
const params = new URLSearchParams({
|
|
@@ -140,33 +146,41 @@ export class TraceGitHub {
|
|
|
140
146
|
}
|
|
141
147
|
|
|
142
148
|
// Dispatch host: one shared artifact whose members name the participant.
|
|
143
|
-
// Download and list member filenames (names only).
|
|
149
|
+
// Download and list member filenames (names only). Members are nested
|
|
150
|
+
// relative paths (`runs/<taskId>/<idx>/trace--*` on eval artifacts), so
|
|
151
|
+
// match on basenames — the `trace--` prefix check never matches a nested
|
|
152
|
+
// path directly.
|
|
144
153
|
for (const artifact of traceArtifacts) {
|
|
145
154
|
const { files } = await this.downloadTrace(runId, {
|
|
146
155
|
name: artifact.name,
|
|
147
156
|
});
|
|
148
|
-
|
|
157
|
+
const basenames = files.map((f) => path.basename(f));
|
|
158
|
+
if (participantInNames(basenames, participant)) return "confirmed";
|
|
149
159
|
}
|
|
150
160
|
return "omit";
|
|
151
161
|
}
|
|
152
162
|
|
|
153
163
|
/**
|
|
154
|
-
* Resolve a
|
|
155
|
-
*
|
|
156
|
-
*
|
|
164
|
+
* Resolve a trace lane path for a known run in one keyed lookup. The method
|
|
165
|
+
* does not enumerate runs. It does not inspect trace content. The key is an
|
|
166
|
+
* exact member filename, a case id, or a participant name.
|
|
157
167
|
*
|
|
158
|
-
* Matrix host: the artifact name carries the
|
|
159
|
-
* Dispatch host: download
|
|
160
|
-
*
|
|
168
|
+
* Matrix host: the artifact name carries the key, so no download happens.
|
|
169
|
+
* Dispatch host: download every `trace--*` artifact and match member
|
|
170
|
+
* basenames against the key. Exactly one match resolves. Several matches
|
|
171
|
+
* throw an error that lists the candidates, so the caller narrows the key.
|
|
172
|
+
* This replaces a silent first-match, which returned an arbitrary cell's
|
|
173
|
+
* lane on eval runs, because every cell emits the same participants.
|
|
161
174
|
*
|
|
162
175
|
* @param {number|string} runId
|
|
163
|
-
* @param {string} participant
|
|
176
|
+
* @param {string} key - Exact member filename, case id, or participant name.
|
|
164
177
|
* @param {object} [opts]
|
|
165
178
|
* @param {string} [opts.dir] - Output directory for a downloaded dispatch artifact
|
|
166
|
-
* @returns {Promise<{runId: (number|string),
|
|
167
|
-
* @throws {Error} when the run has no trace artifacts,
|
|
179
|
+
* @returns {Promise<{runId: (number|string), key: string, host: "matrix"|"dispatch", artifact: string, path: string}>}
|
|
180
|
+
* @throws {Error} when the run has no trace artifacts, no member matches
|
|
181
|
+
* the key, or several members match.
|
|
168
182
|
*/
|
|
169
|
-
async findByKey(runId,
|
|
183
|
+
async findByKey(runId, key, opts = {}) {
|
|
170
184
|
const url = `${API}/repos/${this.owner}/${this.repo}/actions/runs/${runId}/artifacts`;
|
|
171
185
|
const data = await this.#get(url);
|
|
172
186
|
const artifacts = data.artifacts ?? [];
|
|
@@ -177,41 +191,57 @@ export class TraceGitHub {
|
|
|
177
191
|
throw new Error(`No trace artifacts for run ${runId}`);
|
|
178
192
|
}
|
|
179
193
|
|
|
180
|
-
// Matrix host: the artifact name carries the
|
|
194
|
+
// Matrix host: the artifact name carries the key. No download.
|
|
181
195
|
const matrix = traceArtifacts.find((a) =>
|
|
182
|
-
participantInNames([a.name],
|
|
196
|
+
participantInNames([a.name], key),
|
|
183
197
|
);
|
|
184
198
|
if (matrix) {
|
|
185
199
|
return {
|
|
186
200
|
runId,
|
|
187
|
-
|
|
201
|
+
key,
|
|
188
202
|
host: "matrix",
|
|
189
203
|
artifact: matrix.name,
|
|
190
204
|
path: matrix.name,
|
|
191
205
|
};
|
|
192
206
|
}
|
|
193
207
|
|
|
194
|
-
// Dispatch host: download
|
|
208
|
+
// Dispatch host: download every shared artifact and collect the members
|
|
209
|
+
// whose basename matches the key (members are nested relative paths).
|
|
210
|
+
// Each artifact extracts into its own subdirectory — a shared extract
|
|
211
|
+
// dir would re-list earlier artifacts' members on every iteration, so a
|
|
212
|
+
// uniquely-matching key on a multi-artifact (sharded) run would throw a
|
|
213
|
+
// spurious ambiguity error.
|
|
214
|
+
const baseDir = opts.dir ?? `/tmp/trace-${runId}`;
|
|
215
|
+
const matches = [];
|
|
195
216
|
for (const artifact of traceArtifacts) {
|
|
196
217
|
const { dir, files } = await this.downloadTrace(runId, {
|
|
197
218
|
name: artifact.name,
|
|
198
|
-
dir:
|
|
219
|
+
dir: path.join(baseDir, artifact.name),
|
|
199
220
|
});
|
|
200
|
-
const member
|
|
201
|
-
|
|
202
|
-
|
|
203
|
-
|
|
204
|
-
participant,
|
|
205
|
-
host: "dispatch",
|
|
206
|
-
artifact: artifact.name,
|
|
207
|
-
path: path.join(dir, member),
|
|
208
|
-
};
|
|
221
|
+
for (const member of files) {
|
|
222
|
+
if (nameMatchesKey(path.basename(member), key)) {
|
|
223
|
+
matches.push({ artifact: artifact.name, dir, member });
|
|
224
|
+
}
|
|
209
225
|
}
|
|
210
226
|
}
|
|
211
227
|
|
|
212
|
-
|
|
213
|
-
|
|
214
|
-
|
|
228
|
+
if (matches.length === 1) {
|
|
229
|
+
const m = matches[0];
|
|
230
|
+
return {
|
|
231
|
+
runId,
|
|
232
|
+
key,
|
|
233
|
+
host: "dispatch",
|
|
234
|
+
artifact: m.artifact,
|
|
235
|
+
path: path.join(m.dir, m.member),
|
|
236
|
+
};
|
|
237
|
+
}
|
|
238
|
+
if (matches.length > 1) {
|
|
239
|
+
const names = matches.map((m) => m.member).join(", ");
|
|
240
|
+
throw new Error(
|
|
241
|
+
`Ambiguous key "${key}" for run ${runId}: matches ${names}. Narrow the key to a case id or exact filename.`,
|
|
242
|
+
);
|
|
243
|
+
}
|
|
244
|
+
throw new Error(`No trace lane for key "${key}" in run ${runId}`);
|
|
215
245
|
}
|
|
216
246
|
|
|
217
247
|
/**
|
|
@@ -271,9 +301,9 @@ export class TraceGitHub {
|
|
|
271
301
|
);
|
|
272
302
|
}
|
|
273
303
|
|
|
274
|
-
// List extracted files
|
|
275
|
-
|
|
276
|
-
const files =
|
|
304
|
+
// List extracted files — recursively, since eval artifacts carry nested
|
|
305
|
+
// members (`runs/<taskId>/<idx>/trace--*`).
|
|
306
|
+
const files = await listExtractedFiles(this.runtime, dir);
|
|
277
307
|
|
|
278
308
|
return { dir, artifact: artifact.name, files };
|
|
279
309
|
}
|
|
@@ -301,34 +331,22 @@ export class TraceGitHub {
|
|
|
301
331
|
}
|
|
302
332
|
|
|
303
333
|
/**
|
|
304
|
-
*
|
|
305
|
-
*
|
|
306
|
-
*
|
|
307
|
-
*
|
|
308
|
-
*
|
|
309
|
-
*
|
|
310
|
-
*
|
|
311
|
-
* The `--` separator delimits the participant segment. The segment ends at
|
|
312
|
-
* the next `--`, at a `.`, or at the end of the string. So a substring like
|
|
313
|
-
* `release` does not match `release-engineer` and vice versa.
|
|
314
|
-
*
|
|
315
|
-
* @param {string[]} names - Artifact names or extracted member filenames.
|
|
316
|
-
* @param {string} participant - Participant name to look for.
|
|
317
|
-
* @returns {boolean}
|
|
334
|
+
* List every regular file under `dir` recursively, as paths relative to
|
|
335
|
+
* `dir`, excluding `*.zip` (the downloaded archive itself). Sorted for a
|
|
336
|
+
* deterministic member order.
|
|
337
|
+
* @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
|
|
338
|
+
* @param {string} dir
|
|
339
|
+
* @returns {Promise<string[]>}
|
|
318
340
|
*/
|
|
319
|
-
export function
|
|
320
|
-
|
|
321
|
-
|
|
322
|
-
|
|
323
|
-
// Matrix: `<participant>` is the whole remainder (artifact name).
|
|
324
|
-
if (rest === participant) return true;
|
|
325
|
-
// Dispatch: `<case>--<participant>.<role>.ndjson`.
|
|
326
|
-
const sep = rest.indexOf("--");
|
|
327
|
-
if (sep === -1) return false;
|
|
328
|
-
const afterCase = rest.slice(sep + 2);
|
|
329
|
-
const participantSegment = afterCase.split(".")[0];
|
|
330
|
-
return participantSegment === participant;
|
|
341
|
+
export async function listExtractedFiles(runtime, dir) {
|
|
342
|
+
const entries = await runtime.fs.readdir(dir, {
|
|
343
|
+
recursive: true,
|
|
344
|
+
withFileTypes: true,
|
|
331
345
|
});
|
|
346
|
+
return entries
|
|
347
|
+
.filter((e) => e.isFile() && !e.name.endsWith(".zip"))
|
|
348
|
+
.map((e) => path.relative(dir, path.join(e.parentPath ?? e.path, e.name)))
|
|
349
|
+
.sort();
|
|
332
350
|
}
|
|
333
351
|
|
|
334
352
|
/**
|
|
@@ -0,0 +1,132 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Trace identity grammar — the single owner of case ids and lane filenames.
|
|
3
|
+
*
|
|
4
|
+
* Builds case ids (`<taskId>-r<runIndex>`), builds raw/lane filenames under
|
|
5
|
+
* the shared `trace--` convention, validates task ids, and parses names back
|
|
6
|
+
* into identity. Workdir allocation, task-family loading, the shared split
|
|
7
|
+
* module, and GitHub discovery all invoke this module — files agreeing by
|
|
8
|
+
* convention is the drifted-copies pattern this module retires.
|
|
9
|
+
*/
|
|
10
|
+
|
|
11
|
+
import { basename } from "node:path";
|
|
12
|
+
|
|
13
|
+
/**
|
|
14
|
+
* Task ids must not contain "--" or start/end with "-": the "--" delimiter
|
|
15
|
+
* and the terminal "-r<digits>" suffix then parse unambiguously.
|
|
16
|
+
* @param {string} id
|
|
17
|
+
* @returns {boolean}
|
|
18
|
+
*/
|
|
19
|
+
export function isValidTaskId(id) {
|
|
20
|
+
if (typeof id !== "string" || id.length === 0) return false;
|
|
21
|
+
if (id.includes("--")) return false;
|
|
22
|
+
if (id.startsWith("-") || id.endsWith("-")) return false;
|
|
23
|
+
return true;
|
|
24
|
+
}
|
|
25
|
+
|
|
26
|
+
/**
|
|
27
|
+
* Build the grid-unique case id `<taskId>-r<runIndex>`. Shards partition one
|
|
28
|
+
* grid, so (task, runIndex) is already grid- and shard-unique.
|
|
29
|
+
* @param {string} taskId
|
|
30
|
+
* @param {number} runIndex
|
|
31
|
+
* @returns {string}
|
|
32
|
+
* @throws {Error} when `isValidTaskId(taskId)` is false or `runIndex` is not
|
|
33
|
+
* a non-negative integer.
|
|
34
|
+
*/
|
|
35
|
+
export function buildCaseId(taskId, runIndex) {
|
|
36
|
+
if (!isValidTaskId(taskId)) {
|
|
37
|
+
throw new Error(
|
|
38
|
+
`invalid task id '${taskId}': task ids must not contain "--" or start/end with "-"`,
|
|
39
|
+
);
|
|
40
|
+
}
|
|
41
|
+
if (!Number.isInteger(runIndex) || runIndex < 0) {
|
|
42
|
+
throw new Error(`invalid run index '${runIndex}': must be an integer ≥ 0`);
|
|
43
|
+
}
|
|
44
|
+
return `${taskId}-r${runIndex}`;
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
/**
|
|
48
|
+
* Filename of the combined raw envelope trace: `trace--<caseId>.raw.ndjson`.
|
|
49
|
+
* @param {string} caseId
|
|
50
|
+
* @returns {string}
|
|
51
|
+
*/
|
|
52
|
+
export function rawTraceFilename(caseId) {
|
|
53
|
+
return `trace--${caseId}.raw.ndjson`;
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
/**
|
|
57
|
+
* Filename of a per-participant lane:
|
|
58
|
+
* `trace--<caseId>--<participant>.<role>.ndjson`.
|
|
59
|
+
* @param {string} caseId
|
|
60
|
+
* @param {string} participant
|
|
61
|
+
* @param {string} role
|
|
62
|
+
* @returns {string}
|
|
63
|
+
*/
|
|
64
|
+
export function laneFilename(caseId, participant, role) {
|
|
65
|
+
return `trace--${caseId}--${participant}.${role}.ndjson`;
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
/**
|
|
69
|
+
* Parse `trace--<case>--<participant>.<role>.ndjson` into `{caseName,
|
|
70
|
+
* participant}`. On no match, `caseName` is the basename minus its final
|
|
71
|
+
* `.ndjson` extension only and `participant` is null.
|
|
72
|
+
* @param {string} file
|
|
73
|
+
* @returns {{caseName: string, participant: string|null}}
|
|
74
|
+
*/
|
|
75
|
+
export function parseIdentity(file) {
|
|
76
|
+
const name = basename(file);
|
|
77
|
+
const match = name.match(/^trace--(.+?)--(.+?)\.[^.]+\.ndjson$/);
|
|
78
|
+
if (match) {
|
|
79
|
+
return { caseName: match[1], participant: match[2] };
|
|
80
|
+
}
|
|
81
|
+
return { caseName: name.replace(/\.ndjson$/, ""), participant: null };
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
/**
|
|
85
|
+
* Test whether a participant's trace lane is present in a list of names.
|
|
86
|
+
*
|
|
87
|
+
* Matches the two trace-naming shapes by *name* only (never by content):
|
|
88
|
+
* - matrix artifact name: `trace--<participant>`
|
|
89
|
+
* - dispatch member filename: `trace--<case>--<participant>.<role>.ndjson`
|
|
90
|
+
*
|
|
91
|
+
* The participant segment is delimited by `--` and ends at the next `--`, `.`,
|
|
92
|
+
* or end-of-string, so a substring like `release` does not match
|
|
93
|
+
* `release-engineer` and vice versa.
|
|
94
|
+
*
|
|
95
|
+
* Kept as a distinct shape from {@link parseIdentity} deliberately: this
|
|
96
|
+
* matcher also accepts bare artifact names with no extension, which the
|
|
97
|
+
* filename regex cannot.
|
|
98
|
+
*
|
|
99
|
+
* @param {string[]} names - Artifact names or extracted member filenames.
|
|
100
|
+
* @param {string} participant - Participant name to look for.
|
|
101
|
+
* @returns {boolean}
|
|
102
|
+
*/
|
|
103
|
+
export function participantInNames(names, participant) {
|
|
104
|
+
return names.some((name) => {
|
|
105
|
+
if (!name.startsWith("trace--")) return false;
|
|
106
|
+
const rest = name.slice("trace--".length);
|
|
107
|
+
// Matrix: `<participant>` is the whole remainder (artifact name).
|
|
108
|
+
if (rest === participant) return true;
|
|
109
|
+
// Dispatch: `<case>--<participant>.<role>.ndjson`.
|
|
110
|
+
const sep = rest.indexOf("--");
|
|
111
|
+
if (sep === -1) return false;
|
|
112
|
+
const afterCase = rest.slice(sep + 2);
|
|
113
|
+
const participantSegment = afterCase.split(".")[0];
|
|
114
|
+
return participantSegment === participant;
|
|
115
|
+
});
|
|
116
|
+
}
|
|
117
|
+
|
|
118
|
+
/**
|
|
119
|
+
* Keyed-lookup rule for one name: true when `key` equals the exact basename,
|
|
120
|
+
* the parsed case segment, or the parsed participant segment. Derives case
|
|
121
|
+
* and participant via {@link parseIdentity} and reuses the
|
|
122
|
+
* {@link participantInNames} single-name check — no second grammar.
|
|
123
|
+
* @param {string} name - A member basename or artifact name.
|
|
124
|
+
* @param {string} key - Exact filename, case id, or participant name.
|
|
125
|
+
* @returns {boolean}
|
|
126
|
+
*/
|
|
127
|
+
export function nameMatchesKey(name, key) {
|
|
128
|
+
if (name === key) return true;
|
|
129
|
+
const identity = parseIdentity(name);
|
|
130
|
+
if (identity.participant !== null && identity.caseName === key) return true;
|
|
131
|
+
return participantInNames([name], key);
|
|
132
|
+
}
|
package/src/trace-multi.js
CHANGED
|
@@ -12,6 +12,8 @@
|
|
|
12
12
|
*/
|
|
13
13
|
import { basename } from "node:path";
|
|
14
14
|
|
|
15
|
+
import { parseIdentity } from "./trace-identity.js";
|
|
16
|
+
|
|
15
17
|
/**
|
|
16
18
|
* Load each file → `TraceQuery`. Run `query(tq)`. Tag each emitted record
|
|
17
19
|
* with `source: <basename>` only when the caller supplies more than one file.
|
|
@@ -84,20 +86,3 @@ export function compareTwo(a, b, load) {
|
|
|
84
86
|
bIdentity: parseIdentity(b),
|
|
85
87
|
});
|
|
86
88
|
}
|
|
87
|
-
|
|
88
|
-
/**
|
|
89
|
-
* Parse `trace--<case>--<participant>.<role>.ndjson` into `{caseName,
|
|
90
|
-
* participant}`. On no match, `caseName` is the basename without its final
|
|
91
|
-
* `.ndjson` extension. The function removes that one extension only.
|
|
92
|
-
* `participant` is then null.
|
|
93
|
-
* @param {string} file
|
|
94
|
-
* @returns {{caseName: string, participant: string|null}}
|
|
95
|
-
*/
|
|
96
|
-
export function parseIdentity(file) {
|
|
97
|
-
const name = basename(file);
|
|
98
|
-
const match = name.match(/^trace--(.+?)--(.+?)\.[^.]+\.ndjson$/);
|
|
99
|
-
if (match) {
|
|
100
|
-
return { caseName: match[1], participant: match[2] };
|
|
101
|
-
}
|
|
102
|
-
return { caseName: name.replace(/\.ndjson$/, ""), participant: null };
|
|
103
|
-
}
|
|
@@ -0,0 +1,103 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Shared trace-split implementation — the single owner of source-to-role
|
|
3
|
+
* classification. Both the `gemba-trace split` command and the benchmark
|
|
4
|
+
* runner drive this module, so exactly one split policy exists (the same
|
|
5
|
+
* treatment the one-cost-path rule gives `sumTraceCost`).
|
|
6
|
+
*/
|
|
7
|
+
|
|
8
|
+
import { join } from "node:path";
|
|
9
|
+
import { createInterface } from "node:readline";
|
|
10
|
+
|
|
11
|
+
import { laneFilename } from "./trace-identity.js";
|
|
12
|
+
|
|
13
|
+
/** Valid source name pattern: lowercase letter, then lowercase alphanumeric or hyphen. */
|
|
14
|
+
const VALID_SOURCE_NAME = /^[a-z][a-z0-9-]*$/;
|
|
15
|
+
|
|
16
|
+
/**
|
|
17
|
+
* Sources whose name is itself a structural role; classified into the role
|
|
18
|
+
* they represent. `judge` is structural so the judge lane classifies under
|
|
19
|
+
* one rule — no current producer feeds judge-source envelopes through split
|
|
20
|
+
* (the judge is its own session), so kata split output is unchanged.
|
|
21
|
+
*/
|
|
22
|
+
const STRUCTURAL_ROLES = new Set([
|
|
23
|
+
"agent",
|
|
24
|
+
"supervisor",
|
|
25
|
+
"facilitator",
|
|
26
|
+
"judge",
|
|
27
|
+
]);
|
|
28
|
+
|
|
29
|
+
/**
|
|
30
|
+
* Split a combined `{source, seq, event}` NDJSON trace into per-source lane
|
|
31
|
+
* files named by `laneFilename(caseId, source, role)`.
|
|
32
|
+
*
|
|
33
|
+
* Classification: sources in the structural-role set ("agent", "supervisor",
|
|
34
|
+
* "facilitator", "judge") take their own name as role; any other valid source
|
|
35
|
+
* name classifies as role "agent" with the source as participant. Skips
|
|
36
|
+
* empty/malformed/non-envelope lines and orchestrator events; drops sources
|
|
37
|
+
* failing `/^[a-z][a-z0-9-]*$/`. Lane files carry unwrapped event JSON, one
|
|
38
|
+
* per line.
|
|
39
|
+
*
|
|
40
|
+
* @param {import("@forwardimpact/libutil/runtime").Runtime} runtime -
|
|
41
|
+
* Ambient collaborators; streams via `runtime.fs`.
|
|
42
|
+
* @param {string} inputPath - Combined NDJSON trace to split.
|
|
43
|
+
* @param {object} opts
|
|
44
|
+
* @param {string} opts.caseId - Case identity embedded in lane filenames.
|
|
45
|
+
* @param {string} opts.outputDir - Directory the lane files are written to.
|
|
46
|
+
* @returns {Promise<string[]>} Paths written, resolved against `outputDir`
|
|
47
|
+
* (absolute iff `outputDir` is absolute).
|
|
48
|
+
*/
|
|
49
|
+
export async function splitTrace(runtime, inputPath, { caseId, outputDir }) {
|
|
50
|
+
const fs = runtime.fs;
|
|
51
|
+
const rl = createInterface({
|
|
52
|
+
input: fs.createReadStream(inputPath),
|
|
53
|
+
crlfDelay: Infinity,
|
|
54
|
+
});
|
|
55
|
+
const streams = new Map();
|
|
56
|
+
const paths = [];
|
|
57
|
+
for await (const line of rl) {
|
|
58
|
+
const envelope = parseEnvelopeLine(line);
|
|
59
|
+
if (!envelope) continue;
|
|
60
|
+
if (envelope.source === "orchestrator") continue;
|
|
61
|
+
if (!VALID_SOURCE_NAME.test(envelope.source)) continue;
|
|
62
|
+
|
|
63
|
+
let stream = streams.get(envelope.source);
|
|
64
|
+
if (!stream) {
|
|
65
|
+
const role = STRUCTURAL_ROLES.has(envelope.source)
|
|
66
|
+
? envelope.source
|
|
67
|
+
: "agent";
|
|
68
|
+
const outPath = join(
|
|
69
|
+
outputDir,
|
|
70
|
+
laneFilename(caseId, envelope.source, role),
|
|
71
|
+
);
|
|
72
|
+
stream = fs.createWriteStream(outPath);
|
|
73
|
+
streams.set(envelope.source, stream);
|
|
74
|
+
paths.push(outPath);
|
|
75
|
+
}
|
|
76
|
+
stream.write(JSON.stringify(envelope.event) + "\n");
|
|
77
|
+
}
|
|
78
|
+
await Promise.all(
|
|
79
|
+
[...streams.values()].map((s) => new Promise((r) => s.end(r))),
|
|
80
|
+
);
|
|
81
|
+
return paths;
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
/**
|
|
85
|
+
* Parse one NDJSON line into a `{source, seq, event}` envelope, or null when
|
|
86
|
+
* the line is blank, malformed, or not an envelope. Shared with the
|
|
87
|
+
* raw-summary reader so exactly one envelope-line parser exists; callers add
|
|
88
|
+
* their own source filters.
|
|
89
|
+
* @param {string} line
|
|
90
|
+
* @returns {{source: string, event: object}|null}
|
|
91
|
+
*/
|
|
92
|
+
export function parseEnvelopeLine(line) {
|
|
93
|
+
const trimmed = line.trim();
|
|
94
|
+
if (!trimmed) return null;
|
|
95
|
+
let envelope;
|
|
96
|
+
try {
|
|
97
|
+
envelope = JSON.parse(trimmed);
|
|
98
|
+
} catch {
|
|
99
|
+
return null;
|
|
100
|
+
}
|
|
101
|
+
if (!envelope.event || typeof envelope.source !== "string") return null;
|
|
102
|
+
return envelope;
|
|
103
|
+
}
|
|
@@ -1,74 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* Split the combined supervisor trace for the benchmark runner. One pass
|
|
3
|
-
* over the tagged NDJSON envelope stream separates agent events from
|
|
4
|
-
* supervisor and orchestrator events. The same pass extracts the run summary.
|
|
5
|
-
*/
|
|
6
|
-
|
|
7
|
-
import { createInterface } from "node:readline";
|
|
8
|
-
|
|
9
|
-
/**
|
|
10
|
-
* Split the combined supervisor trace into agent and supervisor files in a
|
|
11
|
-
* single pass. The same pass extracts the turn count and the submission.
|
|
12
|
-
* Agent-source events go to `agentPath`. Supervisor and orchestrator events
|
|
13
|
-
* go to `supervisorPath`.
|
|
14
|
-
*
|
|
15
|
-
* This function deliberately does not sum cost. The caller derives it from
|
|
16
|
-
* the same combined trace with `sumTraceCost`. One cost path then serves the
|
|
17
|
-
* benchmark, callback, and `gemba-trace cost` consumers.
|
|
18
|
-
* @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
|
|
19
|
-
* @param {string} combinedPath
|
|
20
|
-
* @param {string} agentPath
|
|
21
|
-
* @param {string} supervisorPath
|
|
22
|
-
* @returns {Promise<{turns: number, submission: string}>}
|
|
23
|
-
*/
|
|
24
|
-
// biome-ignore lint/complexity/noExcessiveCognitiveComplexity: stream-splitting state machine
|
|
25
|
-
export async function splitAndSummarize(
|
|
26
|
-
runtime,
|
|
27
|
-
combinedPath,
|
|
28
|
-
agentPath,
|
|
29
|
-
supervisorPath,
|
|
30
|
-
) {
|
|
31
|
-
const fs = runtime.fs;
|
|
32
|
-
const agentStream = fs.createWriteStream(agentPath);
|
|
33
|
-
const supStream = fs.createWriteStream(supervisorPath);
|
|
34
|
-
const rl = createInterface({
|
|
35
|
-
input: fs.createReadStream(combinedPath),
|
|
36
|
-
crlfDelay: Infinity,
|
|
37
|
-
});
|
|
38
|
-
let turns = 0;
|
|
39
|
-
let submission = "";
|
|
40
|
-
for await (const line of rl) {
|
|
41
|
-
if (!line.trim()) continue;
|
|
42
|
-
let event;
|
|
43
|
-
try {
|
|
44
|
-
event = JSON.parse(line);
|
|
45
|
-
} catch {
|
|
46
|
-
continue;
|
|
47
|
-
}
|
|
48
|
-
const target = event.source === "agent" ? agentStream : supStream;
|
|
49
|
-
target.write(line + "\n");
|
|
50
|
-
const inner = event.event;
|
|
51
|
-
if (!inner) continue;
|
|
52
|
-
if (event.source === "agent" && inner.type === "assistant") {
|
|
53
|
-
const text = extractText(inner);
|
|
54
|
-
if (text) submission = text;
|
|
55
|
-
}
|
|
56
|
-
if (event.source === "orchestrator" && inner.type === "summary") {
|
|
57
|
-
turns = inner.turns ?? 0;
|
|
58
|
-
}
|
|
59
|
-
}
|
|
60
|
-
await Promise.all([
|
|
61
|
-
new Promise((r) => agentStream.end(r)),
|
|
62
|
-
new Promise((r) => supStream.end(r)),
|
|
63
|
-
]);
|
|
64
|
-
return { turns, submission };
|
|
65
|
-
}
|
|
66
|
-
|
|
67
|
-
function extractText(inner) {
|
|
68
|
-
const content = inner.message?.content ?? inner.content;
|
|
69
|
-
if (!Array.isArray(content)) return null;
|
|
70
|
-
for (let i = content.length - 1; i >= 0; i--) {
|
|
71
|
-
if (content[i].type === "text" && content[i].text) return content[i].text;
|
|
72
|
-
}
|
|
73
|
-
return null;
|
|
74
|
-
}
|