@alignfirst/openclaw-test 0.20.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (53) hide show
  1. package/Dockerfile.base +45 -0
  2. package/LICENSE +21 -0
  3. package/README.md +132 -0
  4. package/bin/cli.mjs +7 -0
  5. package/dist/bus.d.ts +1 -0
  6. package/dist/bus.js +20 -0
  7. package/dist/cell-result.d.ts +25 -0
  8. package/dist/cell-result.js +32 -0
  9. package/dist/cli.d.ts +5 -0
  10. package/dist/cli.js +110 -0
  11. package/dist/context.d.ts +220 -0
  12. package/dist/context.js +479 -0
  13. package/dist/cost.d.ts +2 -0
  14. package/dist/cost.js +19 -0
  15. package/dist/env-cli.d.ts +5 -0
  16. package/dist/env-cli.js +645 -0
  17. package/dist/exec-rpc.d.ts +13 -0
  18. package/dist/exec-rpc.js +46 -0
  19. package/dist/index.d.ts +3 -0
  20. package/dist/index.js +1 -0
  21. package/dist/judge.d.ts +47 -0
  22. package/dist/judge.js +183 -0
  23. package/dist/loop.d.ts +62 -0
  24. package/dist/loop.js +316 -0
  25. package/dist/mock-cli-server.d.ts +40 -0
  26. package/dist/mock-cli-server.js +194 -0
  27. package/dist/mock-cli-shim.d.ts +7 -0
  28. package/dist/mock-cli-shim.js +89 -0
  29. package/dist/models.d.ts +23 -0
  30. package/dist/models.js +78 -0
  31. package/dist/parse-tagged-json.d.ts +1 -0
  32. package/dist/parse-tagged-json.js +16 -0
  33. package/dist/report.d.ts +215 -0
  34. package/dist/report.js +19 -0
  35. package/dist/runner-args.d.ts +11 -0
  36. package/dist/runner-args.js +84 -0
  37. package/dist/runner.d.ts +2 -0
  38. package/dist/runner.js +371 -0
  39. package/dist/summary.d.ts +3 -0
  40. package/dist/summary.js +64 -0
  41. package/dist/transcript-dump.d.ts +1 -0
  42. package/dist/transcript-dump.js +50 -0
  43. package/dist/transcript-log.d.ts +67 -0
  44. package/dist/transcript-log.js +199 -0
  45. package/dist/transcript-store.d.ts +6 -0
  46. package/dist/transcript-store.js +69 -0
  47. package/docker-compose.yml +83 -0
  48. package/exec-watcher.mjs +155 -0
  49. package/package.json +60 -0
  50. package/templates/.env.local.example +28 -0
  51. package/templates/Dockerfile +47 -0
  52. package/templates/docker-compose.yml +10 -0
  53. package/templates/openclaw.json +40 -0
@@ -0,0 +1,11 @@
1
+ export interface RunnerArgs {
2
+ scenario: string;
3
+ channel: string;
4
+ modelId: string;
5
+ modelRef: string;
6
+ iterationIndex: number;
7
+ iterationWidth: number;
8
+ baseStamp: string;
9
+ resultsDir: string;
10
+ }
11
+ export declare function parseArgs(argv: string[]): RunnerArgs;
@@ -0,0 +1,84 @@
1
+ export function parseArgs(argv) {
2
+ let scenario;
3
+ let channel;
4
+ let modelId;
5
+ let modelRef;
6
+ let iterationIndex;
7
+ let iterationWidth;
8
+ let baseStamp;
9
+ let resultsDir;
10
+ for (let i = 0; i < argv.length; ++i) {
11
+ const a = argv[i];
12
+ const eat = (flag) => {
13
+ if (a === flag)
14
+ return argv[++i] ?? "";
15
+ return a.slice(`${flag}=`.length);
16
+ };
17
+ if (a === "--scenario" || a.startsWith("--scenario=")) {
18
+ scenario = eat("--scenario");
19
+ }
20
+ else if (a === "--channel" || a.startsWith("--channel=")) {
21
+ channel = eat("--channel");
22
+ }
23
+ else if (a === "--model-id" || a.startsWith("--model-id=")) {
24
+ modelId = eat("--model-id");
25
+ }
26
+ else if (a === "--model-ref" || a.startsWith("--model-ref=")) {
27
+ modelRef = eat("--model-ref");
28
+ }
29
+ else if (a === "--iteration-index" || a.startsWith("--iteration-index=")) {
30
+ iterationIndex = parseNonNegativeInt(eat("--iteration-index"), "--iteration-index", 1);
31
+ }
32
+ else if (a === "--iteration-width" || a.startsWith("--iteration-width=")) {
33
+ iterationWidth = parseNonNegativeInt(eat("--iteration-width"), "--iteration-width", 0);
34
+ }
35
+ else if (a === "--base-stamp" || a.startsWith("--base-stamp=")) {
36
+ baseStamp = eat("--base-stamp");
37
+ }
38
+ else if (a === "--results-dir" || a.startsWith("--results-dir=")) {
39
+ resultsDir = eat("--results-dir");
40
+ }
41
+ else {
42
+ throw new Error(`runner: unknown argument: ${a}`);
43
+ }
44
+ }
45
+ if (!scenario)
46
+ throw new Error("runner: --scenario <id> is required");
47
+ if (!channel)
48
+ throw new Error("runner: --channel <id> is required");
49
+ if (!modelId)
50
+ throw new Error("runner: --model-id <id> is required");
51
+ if (!modelRef)
52
+ throw new Error("runner: --model-ref <ref> is required");
53
+ if (iterationIndex === undefined)
54
+ throw new Error("runner: --iteration-index <n> is required");
55
+ if (iterationWidth === undefined)
56
+ throw new Error("runner: --iteration-width <w> is required");
57
+ if (!baseStamp)
58
+ throw new Error("runner: --base-stamp <iso> is required");
59
+ if (!resultsDir)
60
+ throw new Error("runner: --results-dir <path> is required");
61
+ return {
62
+ scenario,
63
+ channel,
64
+ modelId,
65
+ modelRef,
66
+ iterationIndex,
67
+ iterationWidth,
68
+ baseStamp,
69
+ resultsDir,
70
+ };
71
+ }
72
+ function parseNonNegativeInt(raw, flag, min) {
73
+ if (raw === undefined || raw === "") {
74
+ throw new Error(`runner: ${flag} expects an integer >= ${min}`);
75
+ }
76
+ if (!/^-?\d+$/.test(raw)) {
77
+ throw new Error(`runner: ${flag} expects an integer, got ${JSON.stringify(raw)}`);
78
+ }
79
+ const n = Number(raw);
80
+ if (!Number.isInteger(n) || n < min) {
81
+ throw new Error(`runner: ${flag} expects an integer >= ${min}, got ${JSON.stringify(raw)}`);
82
+ }
83
+ return n;
84
+ }
@@ -0,0 +1,2 @@
1
+ export declare function main(argv?: string[]): Promise<void>;
2
+ export declare function leadingCli(input: unknown): string | undefined;
package/dist/runner.js ADDED
@@ -0,0 +1,371 @@
1
+ import { pollQaBus } from "@alignfirst/openclaw-channel-mock-core";
2
+ import { randomBytes } from "node:crypto";
3
+ import { createWriteStream, mkdirSync, renameSync, writeFileSync } from "node:fs";
4
+ import { basename, join } from "node:path";
5
+ import { cellLeafName, writeCellResult } from "./cell-result.js";
6
+ import { createContext, } from "./context.js";
7
+ import { judgeCostUsd } from "./cost.js";
8
+ import { aggregateAgentToolCalls, fetchTranscriptSnapshot, readTranscriptCost, waitForTranscriptQuiescence, } from "./transcript-log.js";
9
+ import { startMockCliServer } from "./mock-cli-server.js";
10
+ import { parseArgs } from "./runner-args.js";
11
+ const ARTIFACTS_ROOT = process.env.OPENCLAW_TEST_ARTIFACTS_DIR ?? "/opt/openclaw-test/artifacts";
12
+ const SCENARIOS_ROOT = process.env.OPENCLAW_TEST_SCENARIOS_DIR ?? "/opt/openclaw-test/src/scenarios";
13
+ const BUS_URL = process.env.OPENCLAW_TEST_BUS_URL ?? "http://bus:43123";
14
+ export async function main(argv = process.argv.slice(2)) {
15
+ const args = parseArgs(argv);
16
+ console.log(`runner: scenario=${args.scenario} channel=${args.channel} iter=${args.iterationIndex}/${args.iterationWidth}`);
17
+ const exitCode = await runCell(args);
18
+ process.exit(exitCode);
19
+ }
20
+ async function runCell(args) {
21
+ const mockCliServer = startMockCliServer();
22
+ try {
23
+ const setup = setupRun(args);
24
+ const { ctx, internals, outDir, conversationId, accountId, startedAtIso, startedAtMs, logStream, } = setup;
25
+ mockCliServer.bind({
26
+ conversationId,
27
+ handlers: internals.getMockHandlers(),
28
+ emitCliMock: internals.emitCliMock,
29
+ isScenarioEnded: () => internals.isScenarioEnded(),
30
+ });
31
+ const initialCursor = await ctx.getCursor();
32
+ const subscription = startOutboundSubscription({
33
+ accountId,
34
+ conversationId,
35
+ initialCursor,
36
+ onMessage: internals.emitOutboundReceived,
37
+ });
38
+ const { failure } = await executeScenario(args.scenario, ctx);
39
+ await subscription.stop();
40
+ await mockCliServer.release();
41
+ const finishedAtMs = Date.now();
42
+ const durationMs = finishedAtMs - startedAtMs;
43
+ const finishedAtIso = new Date(finishedAtMs).toISOString();
44
+ const { entries, judgeUsages, result } = internals.finalize({ failure });
45
+ await waitForTranscriptQuiescence({ conversationId, startedAtIso });
46
+ const snapshot = await fetchTranscriptSnapshot({ conversationId, startedAtIso });
47
+ const { cost: agentCostUsd, turns: agentTurns } = readTranscriptCost(snapshot);
48
+ const judgeUsd = judgeUsages.reduce((sum, u) => sum + judgeCostUsd(u), 0);
49
+ // The per-session store dies with the stack recreation, so keep the raw
50
+ // transcripts as a cell artifact for post-mortems.
51
+ writeFileSync(join(outDir, "transcripts.json"), JSON.stringify(snapshot.sessions));
52
+ const agentCalls = aggregateAgentToolCalls(snapshot.sessions);
53
+ pairAgentCallsWithCliMocks(agentCalls, entries);
54
+ if (agentCalls.length === 0 && (snapshot.error !== undefined || snapshot.databases === 0)) {
55
+ entries.push({
56
+ entrySeq: entries.length,
57
+ ts: finishedAtIso,
58
+ kind: "scenarioLog",
59
+ message: snapshot.error !== undefined
60
+ ? `agentToolCall parsing failed: ${snapshot.error}`
61
+ : "agentToolCall parsing skipped: no agent session store found in the gateway",
62
+ });
63
+ }
64
+ const agentEntries = buildAgentToolCallEntries(agentCalls, entries.length, finishedAtIso);
65
+ appendAgentCallsToLog(logStream, agentEntries);
66
+ await closeStream(logStream);
67
+ const merged = mergeTimeline(entries, agentEntries);
68
+ const report = {
69
+ schemaVersion: 4,
70
+ scenario: args.scenario,
71
+ channel: args.channel,
72
+ model: args.modelId,
73
+ conversationId,
74
+ accountId,
75
+ startedAt: startedAtIso,
76
+ finishedAt: finishedAtIso,
77
+ durationMs,
78
+ cost: {
79
+ agentUsd: agentCostUsd,
80
+ judgeUsd,
81
+ totalUsd: agentCostUsd + judgeUsd,
82
+ agentTurns,
83
+ },
84
+ result,
85
+ entries: merged.map(prepareEntryForReport),
86
+ };
87
+ // Write the cell record BEFORE the artifact-dir rename, to a stable sibling path.
88
+ const leafBase = basename(outDir);
89
+ const resultsPath = join(args.resultsDir, `${leafBase}.json`);
90
+ mkdirSync(args.resultsDir, { recursive: true });
91
+ const finalOutDir = writeReportArtifacts(outDir, result.verdict, report);
92
+ writeCellResult(resultsPath, {
93
+ schemaVersion: 3,
94
+ scenarioId: args.scenario,
95
+ channel: args.channel,
96
+ model: args.modelId,
97
+ iterationIndex: args.iterationIndex,
98
+ verdict: result.verdict,
99
+ durationMs,
100
+ conversationId,
101
+ artifactDirName: basename(finalOutDir),
102
+ agentCostUsd,
103
+ agentTurns,
104
+ judgeUsd,
105
+ judgeUsages,
106
+ });
107
+ console.log(`[${args.channel}] ${args.scenario} ${result.verdict} in ${durationMs}ms — ${finalOutDir}`);
108
+ return result.verdict === "pass" ? 0 : 1;
109
+ }
110
+ finally {
111
+ await mockCliServer.close();
112
+ }
113
+ }
114
+ function setupRun(args) {
115
+ const conversationId = `${args.scenario}-${args.channel}-${shortRand()}`;
116
+ const accountId = args.channel;
117
+ const leaf = cellLeafName({
118
+ scenarioId: args.scenario,
119
+ modelId: args.modelId,
120
+ channel: args.channel,
121
+ iterationIndex: args.iterationIndex,
122
+ iterationWidth: args.iterationWidth,
123
+ });
124
+ const outDir = join(ARTIFACTS_ROOT, args.baseStamp, leaf);
125
+ mkdirSync(outDir, { recursive: true });
126
+ const startedAtIso = new Date().toISOString();
127
+ const startedAtMs = Date.now();
128
+ const logStream = createWriteStream(join(outDir, "scenario-log.jsonl"), { flags: "a" });
129
+ const emitSink = (event) => {
130
+ if (event.type === "entry") {
131
+ logStream.write(`${JSON.stringify(event.entry)}\n`);
132
+ }
133
+ else {
134
+ logStream.write(`${JSON.stringify({ entrySeq: event.entrySeq, augment: event.patch })}\n`);
135
+ }
136
+ };
137
+ const { ctx, internals } = createContext({
138
+ channel: args.channel,
139
+ conversationId,
140
+ startedAtIso,
141
+ emitSink,
142
+ });
143
+ return {
144
+ ctx,
145
+ internals,
146
+ outDir,
147
+ conversationId,
148
+ accountId,
149
+ startedAtIso,
150
+ startedAtMs,
151
+ logStream,
152
+ };
153
+ }
154
+ function startOutboundSubscription(params) {
155
+ const abort = new AbortController();
156
+ const done = (async () => {
157
+ let cursor = params.initialCursor;
158
+ while (!abort.signal.aborted) {
159
+ let result;
160
+ try {
161
+ result = await pollQaBus({
162
+ baseUrl: BUS_URL,
163
+ accountId: params.accountId,
164
+ cursor,
165
+ timeoutMs: 1000,
166
+ });
167
+ }
168
+ catch {
169
+ if (abort.signal.aborted)
170
+ return;
171
+ await new Promise((r) => setTimeout(r, 200));
172
+ continue;
173
+ }
174
+ cursor = result.cursor;
175
+ for (const e of result.events) {
176
+ if (e.kind !== "outbound-message" && e.kind !== "message-edited")
177
+ continue;
178
+ if (e.message.conversation.id !== params.conversationId)
179
+ continue;
180
+ params.onMessage(e.message);
181
+ }
182
+ }
183
+ })();
184
+ return {
185
+ stop: async () => {
186
+ abort.abort();
187
+ await done.catch(() => { });
188
+ },
189
+ };
190
+ }
191
+ async function executeScenario(scenarioId, ctx) {
192
+ try {
193
+ const mod = await import(`${SCENARIOS_ROOT}/${scenarioId}.ts`);
194
+ const fn = mod.default;
195
+ if (typeof fn !== "function") {
196
+ throw new Error(`scenario ${scenarioId} has no default export function`);
197
+ }
198
+ await fn(ctx);
199
+ return { failure: undefined };
200
+ }
201
+ catch (err) {
202
+ const e = err;
203
+ const isAssertion = e?.name === "AssertionError";
204
+ return {
205
+ failure: {
206
+ name: e?.name ?? "Error",
207
+ message: e?.message ?? String(err),
208
+ stack: e?.stack,
209
+ source: isAssertion ? "assertion" : "scenarioThrow",
210
+ },
211
+ };
212
+ }
213
+ }
214
+ function pairAgentCallsWithCliMocks(agentCalls, entries) {
215
+ const cliMockQueues = new Map();
216
+ for (const e of entries) {
217
+ if (e.kind !== "cliMock")
218
+ continue;
219
+ const q = cliMockQueues.get(e.call.cli) ?? [];
220
+ q.push(e.ts);
221
+ cliMockQueues.set(e.call.cli, q);
222
+ }
223
+ for (const call of agentCalls) {
224
+ const cli = leadingCli(call.input);
225
+ if (!cli)
226
+ continue;
227
+ const q = cliMockQueues.get(cli);
228
+ if (!q || q.length === 0)
229
+ continue;
230
+ call.inferredStartedAt = q.shift();
231
+ }
232
+ }
233
+ export function leadingCli(input) {
234
+ if (!input || typeof input !== "object")
235
+ return;
236
+ const cmd = input.command;
237
+ if (typeof cmd !== "string")
238
+ return;
239
+ const stripped = cmd.replace(/^(?:\s*cd\s+[^&;|]+(?:&&|;|\|\|)\s*)+/, "");
240
+ const m = stripped.match(/^\s*(?:[A-Za-z_][A-Za-z0-9_]*=\S+\s+)*([A-Za-z0-9_./-]+)/);
241
+ if (!m)
242
+ return;
243
+ // Reject bare env-assignments (e.g. `FOO=bar` alone): the capture class
244
+ // stops at `=`, so we'd otherwise return "FOO" as if it were a command.
245
+ if (stripped[m[0].length] === "=")
246
+ return;
247
+ return m[1].split("/").pop();
248
+ }
249
+ /**
250
+ * Build agentToolCall entries with stable `entrySeq` values continuing the live
251
+ * jsonl emission order. These values are shared by both artifact files —
252
+ * `scenario-log.jsonl` (appended at the tail in ts order) and `report.json`
253
+ * (interleaved with live entries by ts).
254
+ */
255
+ function buildAgentToolCallEntries(agentCalls, liveEntryCount, fallbackTs) {
256
+ const sorted = [...agentCalls].sort(compareStartedAt);
257
+ return sorted.map((call, i) => ({
258
+ entrySeq: liveEntryCount + i,
259
+ ts: call.startedAt ?? fallbackTs,
260
+ kind: "agentToolCall",
261
+ call,
262
+ }));
263
+ }
264
+ // A call whose message carried no timestamp sorts to the end, never the front.
265
+ function compareStartedAt(a, b) {
266
+ if (a.startedAt === undefined)
267
+ return b.startedAt === undefined ? 0 : 1;
268
+ if (b.startedAt === undefined)
269
+ return -1;
270
+ return a.startedAt < b.startedAt ? -1 : 1;
271
+ }
272
+ /**
273
+ * Append parsed agentToolCall entries to the tail of `scenario-log.jsonl` in
274
+ * ts order. Their `entrySeq` values continue past the live entries' and are the
275
+ * same values used in `report.json` — readers can cross-reference by `entrySeq`.
276
+ */
277
+ function appendAgentCallsToLog(logStream, agentEntries) {
278
+ for (const entry of agentEntries) {
279
+ logStream.write(`${JSON.stringify(entry)}\n`);
280
+ }
281
+ }
282
+ /**
283
+ * Interleave live entries with agentToolCall entries by `ts` for `report.json`.
284
+ * Entries keep their original `entrySeq`; the array order is ts, so iterating
285
+ * gives the unified timeline while `entrySeq` remains a cross-file identifier.
286
+ */
287
+ function mergeTimeline(entries, agentEntries) {
288
+ return [...entries, ...agentEntries].sort((a, b) => {
289
+ if (a.ts !== b.ts)
290
+ return a.ts < b.ts ? -1 : 1;
291
+ return kindOrder(a.kind) - kindOrder(b.kind);
292
+ });
293
+ }
294
+ function kindOrder(k) {
295
+ return k === "agentToolCall" ? 0 : 1;
296
+ }
297
+ const REPORT_CONTENT_TRUNCATE_AT = 60;
298
+ /**
299
+ * Replace bulky `content` with `truncatedContent` for `report.json` only. The
300
+ * original entry (already streamed to `scenario-log.jsonl`) keeps the full
301
+ * content; this returns a shallow-cloned `agentToolCall` entry with the
302
+ * truncated `result`.
303
+ */
304
+ function prepareEntryForReport(entry) {
305
+ if (entry.kind !== "agentToolCall" || !entry.call.result)
306
+ return entry;
307
+ const text = truncatableResultText(entry.call.toolName, entry.call.result.content);
308
+ if (text === undefined || text.length <= REPORT_CONTENT_TRUNCATE_AT)
309
+ return entry;
310
+ const truncatedContent = `${text.slice(0, REPORT_CONTENT_TRUNCATE_AT).replace(/\s+$/, "")}…`;
311
+ return {
312
+ ...entry,
313
+ call: {
314
+ ...entry.call,
315
+ result: { isError: entry.call.result.isError, truncatedContent },
316
+ },
317
+ };
318
+ }
319
+ /**
320
+ * The string to truncate for `report.json`, or `undefined` to keep `content`
321
+ * as-is. The `read` tool returns its file content as text blocks rather than a
322
+ * string, so its text is joined — the file is identified by `input`, so the
323
+ * report needs no more than a preview.
324
+ */
325
+ function truncatableResultText(toolName, content) {
326
+ if (typeof content === "string")
327
+ return content;
328
+ if (toolName === "read")
329
+ return readBlocksText(content);
330
+ return;
331
+ }
332
+ function readBlocksText(content) {
333
+ if (!Array.isArray(content))
334
+ return;
335
+ let text = "";
336
+ for (const block of content) {
337
+ if (isRecord(block) && block.type === "text" && typeof block.text === "string") {
338
+ text += block.text;
339
+ }
340
+ }
341
+ return text.length > 0 ? text : undefined;
342
+ }
343
+ function isRecord(value) {
344
+ return typeof value === "object" && value !== null;
345
+ }
346
+ function writeReportArtifacts(outDir, status, report) {
347
+ writeFileSync(join(outDir, "report.json"), JSON.stringify(report, null, 2));
348
+ const renamed = `${outDir}-${status.toUpperCase()}`;
349
+ try {
350
+ renameSync(outDir, renamed);
351
+ return renamed;
352
+ }
353
+ catch (err) {
354
+ console.warn(`runner: failed to rename ${outDir} -> ${renamed}:`, err);
355
+ return outDir;
356
+ }
357
+ }
358
+ function closeStream(stream) {
359
+ return new Promise((resolve, reject) => {
360
+ stream.end((err) => (err ? reject(err) : resolve()));
361
+ });
362
+ }
363
+ function shortRand() {
364
+ return randomBytes(3).toString("hex");
365
+ }
366
+ if (import.meta.url === `file://${process.argv[1]}`) {
367
+ main().catch((err) => {
368
+ console.error("runner crash:", err);
369
+ process.exit(1);
370
+ });
371
+ }
@@ -0,0 +1,3 @@
1
+ import type { CellResult } from "./cell-result.js";
2
+ export declare function printSummary(results: CellResult[], baseStamp: string): void;
3
+ export declare function printTotalCost(results: CellResult[]): void;
@@ -0,0 +1,64 @@
1
+ import { join } from "node:path";
2
+ import { judgeCostUsd } from "./cost.js";
3
+ function artifactsRoot() {
4
+ return process.env.OPENCLAW_TEST_ARTIFACTS_DIR ?? "/opt/openclaw-test/artifacts";
5
+ }
6
+ function groupByPair(results) {
7
+ const order = [];
8
+ const map = new Map();
9
+ for (const r of results) {
10
+ const key = `${r.scenarioId}|${r.channel}|${r.model}`;
11
+ let agg = map.get(key);
12
+ if (!agg) {
13
+ agg = {
14
+ scenarioId: r.scenarioId,
15
+ channel: r.channel,
16
+ model: r.model,
17
+ runCount: 0,
18
+ passCount: 0,
19
+ durationSumMs: 0,
20
+ };
21
+ map.set(key, agg);
22
+ order.push(key);
23
+ }
24
+ agg.runCount += 1;
25
+ agg.durationSumMs += r.durationMs;
26
+ if (r.verdict === "pass")
27
+ agg.passCount += 1;
28
+ }
29
+ return order.map((k) => map.get(k));
30
+ }
31
+ export function printSummary(results, baseStamp) {
32
+ const aggregates = groupByPair(results);
33
+ console.log("");
34
+ console.log("Summary:");
35
+ for (const a of aggregates) {
36
+ const passed = a.passCount === a.runCount;
37
+ const verdict = (passed ? "PASS" : "FAIL").padEnd(4, " ");
38
+ const counts = `${a.passCount}/${a.runCount}`.padStart(7, " ");
39
+ console.log(` ${verdict} ${a.channel.padEnd(12, " ")} ${a.model.padEnd(20, " ")} ${a.scenarioId.padEnd(40, " ")} ${counts} in ${a.durationSumMs}ms`);
40
+ }
41
+ console.log("");
42
+ console.log("Artifacts:");
43
+ console.log(` ${join(artifactsRoot(), baseStamp)}`);
44
+ }
45
+ export function printTotalCost(results) {
46
+ console.log("");
47
+ console.log(costLine(results));
48
+ const models = [...new Set(results.map((r) => r.model))];
49
+ if (models.length > 1) {
50
+ console.log("Per-model cost:");
51
+ for (const model of models) {
52
+ console.log(` ${model.padEnd(20, " ")} ${costLine(results.filter((r) => r.model === model))}`);
53
+ }
54
+ }
55
+ }
56
+ function costLine(results) {
57
+ const judgeCost = results
58
+ .flatMap((r) => r.judgeUsages)
59
+ .reduce((sum, u) => sum + judgeCostUsd(u), 0);
60
+ const agentCost = results.reduce((sum, r) => sum + r.agentCostUsd, 0);
61
+ const agentTurns = results.reduce((sum, r) => sum + r.agentTurns, 0);
62
+ const totalCost = agentCost + judgeCost;
63
+ return `Total LLM cost: $${totalCost.toFixed(4)} (agent: $${agentCost.toFixed(4)} over ${agentTurns} turns, judge: $${judgeCost.toFixed(4)})`;
64
+ }
@@ -0,0 +1 @@
1
+ export {};
@@ -0,0 +1,50 @@
1
+ // Gateway-side transcript dump, invoked by the runner through the exec-watcher
2
+ // RPC: `node transcript-dump.js <sinceIso> <conversationId> <outPath> [threadId...]`.
3
+ // OpenClaw 2026.8+ persists each session's transcript as SQLite rows in the
4
+ // per-agent store. Unlike the trajectory diagnostics (whose payloads are
5
+ // node-capped and redacted), the transcript is the full-fidelity record the
6
+ // gateway itself replays, appended per message — tool calls become visible as
7
+ // they happen, not at turn end. This script extracts the transcripts of a
8
+ // conversation's sessions and writes them as JSON to `outPath` (stdout would
9
+ // hit the watcher's 1 MiB cap).
10
+ import { readdirSync, renameSync, writeFileSync } from "node:fs";
11
+ import { homedir } from "node:os";
12
+ import { join } from "node:path";
13
+ import { DatabaseSync } from "node:sqlite";
14
+ import { readConversationSessions } from "./transcript-store.js";
15
+ main();
16
+ function main() {
17
+ const [sinceIso, conversationId, outPath, ...threadIds] = process.argv.slice(2);
18
+ if (sinceIso === undefined || conversationId === undefined || outPath === undefined) {
19
+ console.error("usage: transcript-dump.js <sinceIso> <conversationId> <outPath> [threadId...]");
20
+ process.exit(2);
21
+ }
22
+ const sinceMs = Date.parse(sinceIso);
23
+ const databases = findAgentDatabases();
24
+ const sessions = databases.flatMap((dbPath) => readConversationSessions(dbPath, sinceMs, conversationId, threadIds));
25
+ const tmpPath = `${outPath}.tmp`;
26
+ writeFileSync(tmpPath, JSON.stringify({ databases: databases.length, sessions }));
27
+ renameSync(tmpPath, outPath);
28
+ }
29
+ function findAgentDatabases() {
30
+ const agentsDir = join(homedir(), ".openclaw", "agents");
31
+ let agentIds;
32
+ try {
33
+ agentIds = readdirSync(agentsDir);
34
+ }
35
+ catch {
36
+ return [];
37
+ }
38
+ return agentIds
39
+ .map((id) => join(agentsDir, id, "agent", "openclaw-agent.sqlite"))
40
+ .filter(canOpenDatabase);
41
+ }
42
+ function canOpenDatabase(path) {
43
+ try {
44
+ new DatabaseSync(path, { readOnly: true }).close();
45
+ return true;
46
+ }
47
+ catch {
48
+ return false;
49
+ }
50
+ }
@@ -0,0 +1,67 @@
1
+ import type { AgentToolCall } from "./report.js";
2
+ export interface TranscriptSnapshot {
3
+ /** Agent stores found in the gateway. 0 means no store yet — or none at all. */
4
+ databases: number;
5
+ sessions: TranscriptSession[];
6
+ /**
7
+ * Set when the dump itself failed (non-zero exit, timeout, unreadable
8
+ * output). `databases` and `sessions` are then placeholders, not
9
+ * observations — report the failure, not "no store found".
10
+ */
11
+ error?: string;
12
+ }
13
+ export interface TranscriptSession {
14
+ sessionKey: string;
15
+ sessionId: string;
16
+ /** OpenClaw's neutral message shape: assistant `toolCall` blocks, `toolResult` messages. */
17
+ messages: unknown[];
18
+ }
19
+ /** The transcripts of the conversation and its bus-owned thread sessions. */
20
+ export declare function fetchTranscriptSnapshot(opts: {
21
+ conversationId: string;
22
+ startedAtIso: string;
23
+ }): Promise<TranscriptSnapshot>;
24
+ /**
25
+ * Polls the transcripts until they are *quiescent* for this conversation — no
26
+ * new message for `settleMs` AND no session's turn is demonstrably open — or
27
+ * the budget expires. The transcript is appended per message, so a settle
28
+ * window alone would return between two tool calls of one turn (a single
29
+ * model round-trip routinely exceeds it); the open-turn gate holds until the
30
+ * turn-final assistant message — the one carrying `usage.cost.total` — has
31
+ * landed. A conversation spans multiple OpenClaw sessions (e.g. Discord's
32
+ * channel session plus a per-thread session).
33
+ */
34
+ export declare function waitForTranscriptQuiescence(opts: {
35
+ conversationId: string;
36
+ startedAtIso: string;
37
+ maxWaitMs?: number;
38
+ pollMs?: number;
39
+ settleMs?: number;
40
+ }): Promise<void>;
41
+ /**
42
+ * A turn is open when a session's last message is a tool result or an
43
+ * assistant stop for tool use — more of the turn is coming. A trailing user
44
+ * message does not count as open: some sessions of a conversation never get a
45
+ * reply (the agent answers in a sibling session), and treating them as open
46
+ * would burn the whole budget on every cell.
47
+ */
48
+ export declare function hasOpenTurn(sessions: TranscriptSession[]): boolean;
49
+ /**
50
+ * Cost lives per assistant message, as `usage.cost.total`. Sum across every
51
+ * session of the conversation; `turns` counts the cost-bearing messages.
52
+ */
53
+ export declare function readTranscriptCost(snapshot: TranscriptSnapshot): {
54
+ cost: number;
55
+ turns: number;
56
+ };
57
+ /** One-shot fetch + aggregation of the conversation's agent tool calls. */
58
+ export declare function parseAgentToolCalls(opts: {
59
+ conversationId: string;
60
+ startedAtIso: string;
61
+ }): Promise<AgentToolCall[]>;
62
+ /**
63
+ * Walk each session's messages collecting assistant `toolCall` blocks matched
64
+ * with `toolResult` messages, then union across sessions deduped by
65
+ * `toolUseId`.
66
+ */
67
+ export declare function aggregateAgentToolCalls(sessions: TranscriptSession[]): AgentToolCall[];