@alignfirst/openclaw-test 0.20.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/Dockerfile.base +45 -0
- package/LICENSE +21 -0
- package/README.md +132 -0
- package/bin/cli.mjs +7 -0
- package/dist/bus.d.ts +1 -0
- package/dist/bus.js +20 -0
- package/dist/cell-result.d.ts +25 -0
- package/dist/cell-result.js +32 -0
- package/dist/cli.d.ts +5 -0
- package/dist/cli.js +110 -0
- package/dist/context.d.ts +220 -0
- package/dist/context.js +479 -0
- package/dist/cost.d.ts +2 -0
- package/dist/cost.js +19 -0
- package/dist/env-cli.d.ts +5 -0
- package/dist/env-cli.js +645 -0
- package/dist/exec-rpc.d.ts +13 -0
- package/dist/exec-rpc.js +46 -0
- package/dist/index.d.ts +3 -0
- package/dist/index.js +1 -0
- package/dist/judge.d.ts +47 -0
- package/dist/judge.js +183 -0
- package/dist/loop.d.ts +62 -0
- package/dist/loop.js +316 -0
- package/dist/mock-cli-server.d.ts +40 -0
- package/dist/mock-cli-server.js +194 -0
- package/dist/mock-cli-shim.d.ts +7 -0
- package/dist/mock-cli-shim.js +89 -0
- package/dist/models.d.ts +23 -0
- package/dist/models.js +78 -0
- package/dist/parse-tagged-json.d.ts +1 -0
- package/dist/parse-tagged-json.js +16 -0
- package/dist/report.d.ts +215 -0
- package/dist/report.js +19 -0
- package/dist/runner-args.d.ts +11 -0
- package/dist/runner-args.js +84 -0
- package/dist/runner.d.ts +2 -0
- package/dist/runner.js +371 -0
- package/dist/summary.d.ts +3 -0
- package/dist/summary.js +64 -0
- package/dist/transcript-dump.d.ts +1 -0
- package/dist/transcript-dump.js +50 -0
- package/dist/transcript-log.d.ts +67 -0
- package/dist/transcript-log.js +199 -0
- package/dist/transcript-store.d.ts +6 -0
- package/dist/transcript-store.js +69 -0
- package/docker-compose.yml +83 -0
- package/exec-watcher.mjs +155 -0
- package/package.json +60 -0
- package/templates/.env.local.example +28 -0
- package/templates/Dockerfile +47 -0
- package/templates/docker-compose.yml +10 -0
- package/templates/openclaw.json +40 -0
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
export interface RunnerArgs {
|
|
2
|
+
scenario: string;
|
|
3
|
+
channel: string;
|
|
4
|
+
modelId: string;
|
|
5
|
+
modelRef: string;
|
|
6
|
+
iterationIndex: number;
|
|
7
|
+
iterationWidth: number;
|
|
8
|
+
baseStamp: string;
|
|
9
|
+
resultsDir: string;
|
|
10
|
+
}
|
|
11
|
+
export declare function parseArgs(argv: string[]): RunnerArgs;
|
|
@@ -0,0 +1,84 @@
|
|
|
1
|
+
export function parseArgs(argv) {
|
|
2
|
+
let scenario;
|
|
3
|
+
let channel;
|
|
4
|
+
let modelId;
|
|
5
|
+
let modelRef;
|
|
6
|
+
let iterationIndex;
|
|
7
|
+
let iterationWidth;
|
|
8
|
+
let baseStamp;
|
|
9
|
+
let resultsDir;
|
|
10
|
+
for (let i = 0; i < argv.length; ++i) {
|
|
11
|
+
const a = argv[i];
|
|
12
|
+
const eat = (flag) => {
|
|
13
|
+
if (a === flag)
|
|
14
|
+
return argv[++i] ?? "";
|
|
15
|
+
return a.slice(`${flag}=`.length);
|
|
16
|
+
};
|
|
17
|
+
if (a === "--scenario" || a.startsWith("--scenario=")) {
|
|
18
|
+
scenario = eat("--scenario");
|
|
19
|
+
}
|
|
20
|
+
else if (a === "--channel" || a.startsWith("--channel=")) {
|
|
21
|
+
channel = eat("--channel");
|
|
22
|
+
}
|
|
23
|
+
else if (a === "--model-id" || a.startsWith("--model-id=")) {
|
|
24
|
+
modelId = eat("--model-id");
|
|
25
|
+
}
|
|
26
|
+
else if (a === "--model-ref" || a.startsWith("--model-ref=")) {
|
|
27
|
+
modelRef = eat("--model-ref");
|
|
28
|
+
}
|
|
29
|
+
else if (a === "--iteration-index" || a.startsWith("--iteration-index=")) {
|
|
30
|
+
iterationIndex = parseNonNegativeInt(eat("--iteration-index"), "--iteration-index", 1);
|
|
31
|
+
}
|
|
32
|
+
else if (a === "--iteration-width" || a.startsWith("--iteration-width=")) {
|
|
33
|
+
iterationWidth = parseNonNegativeInt(eat("--iteration-width"), "--iteration-width", 0);
|
|
34
|
+
}
|
|
35
|
+
else if (a === "--base-stamp" || a.startsWith("--base-stamp=")) {
|
|
36
|
+
baseStamp = eat("--base-stamp");
|
|
37
|
+
}
|
|
38
|
+
else if (a === "--results-dir" || a.startsWith("--results-dir=")) {
|
|
39
|
+
resultsDir = eat("--results-dir");
|
|
40
|
+
}
|
|
41
|
+
else {
|
|
42
|
+
throw new Error(`runner: unknown argument: ${a}`);
|
|
43
|
+
}
|
|
44
|
+
}
|
|
45
|
+
if (!scenario)
|
|
46
|
+
throw new Error("runner: --scenario <id> is required");
|
|
47
|
+
if (!channel)
|
|
48
|
+
throw new Error("runner: --channel <id> is required");
|
|
49
|
+
if (!modelId)
|
|
50
|
+
throw new Error("runner: --model-id <id> is required");
|
|
51
|
+
if (!modelRef)
|
|
52
|
+
throw new Error("runner: --model-ref <ref> is required");
|
|
53
|
+
if (iterationIndex === undefined)
|
|
54
|
+
throw new Error("runner: --iteration-index <n> is required");
|
|
55
|
+
if (iterationWidth === undefined)
|
|
56
|
+
throw new Error("runner: --iteration-width <w> is required");
|
|
57
|
+
if (!baseStamp)
|
|
58
|
+
throw new Error("runner: --base-stamp <iso> is required");
|
|
59
|
+
if (!resultsDir)
|
|
60
|
+
throw new Error("runner: --results-dir <path> is required");
|
|
61
|
+
return {
|
|
62
|
+
scenario,
|
|
63
|
+
channel,
|
|
64
|
+
modelId,
|
|
65
|
+
modelRef,
|
|
66
|
+
iterationIndex,
|
|
67
|
+
iterationWidth,
|
|
68
|
+
baseStamp,
|
|
69
|
+
resultsDir,
|
|
70
|
+
};
|
|
71
|
+
}
|
|
72
|
+
function parseNonNegativeInt(raw, flag, min) {
|
|
73
|
+
if (raw === undefined || raw === "") {
|
|
74
|
+
throw new Error(`runner: ${flag} expects an integer >= ${min}`);
|
|
75
|
+
}
|
|
76
|
+
if (!/^-?\d+$/.test(raw)) {
|
|
77
|
+
throw new Error(`runner: ${flag} expects an integer, got ${JSON.stringify(raw)}`);
|
|
78
|
+
}
|
|
79
|
+
const n = Number(raw);
|
|
80
|
+
if (!Number.isInteger(n) || n < min) {
|
|
81
|
+
throw new Error(`runner: ${flag} expects an integer >= ${min}, got ${JSON.stringify(raw)}`);
|
|
82
|
+
}
|
|
83
|
+
return n;
|
|
84
|
+
}
|
package/dist/runner.d.ts
ADDED
package/dist/runner.js
ADDED
|
@@ -0,0 +1,371 @@
|
|
|
1
|
+
import { pollQaBus } from "@alignfirst/openclaw-channel-mock-core";
|
|
2
|
+
import { randomBytes } from "node:crypto";
|
|
3
|
+
import { createWriteStream, mkdirSync, renameSync, writeFileSync } from "node:fs";
|
|
4
|
+
import { basename, join } from "node:path";
|
|
5
|
+
import { cellLeafName, writeCellResult } from "./cell-result.js";
|
|
6
|
+
import { createContext, } from "./context.js";
|
|
7
|
+
import { judgeCostUsd } from "./cost.js";
|
|
8
|
+
import { aggregateAgentToolCalls, fetchTranscriptSnapshot, readTranscriptCost, waitForTranscriptQuiescence, } from "./transcript-log.js";
|
|
9
|
+
import { startMockCliServer } from "./mock-cli-server.js";
|
|
10
|
+
import { parseArgs } from "./runner-args.js";
|
|
11
|
+
const ARTIFACTS_ROOT = process.env.OPENCLAW_TEST_ARTIFACTS_DIR ?? "/opt/openclaw-test/artifacts";
|
|
12
|
+
const SCENARIOS_ROOT = process.env.OPENCLAW_TEST_SCENARIOS_DIR ?? "/opt/openclaw-test/src/scenarios";
|
|
13
|
+
const BUS_URL = process.env.OPENCLAW_TEST_BUS_URL ?? "http://bus:43123";
|
|
14
|
+
export async function main(argv = process.argv.slice(2)) {
|
|
15
|
+
const args = parseArgs(argv);
|
|
16
|
+
console.log(`runner: scenario=${args.scenario} channel=${args.channel} iter=${args.iterationIndex}/${args.iterationWidth}`);
|
|
17
|
+
const exitCode = await runCell(args);
|
|
18
|
+
process.exit(exitCode);
|
|
19
|
+
}
|
|
20
|
+
async function runCell(args) {
|
|
21
|
+
const mockCliServer = startMockCliServer();
|
|
22
|
+
try {
|
|
23
|
+
const setup = setupRun(args);
|
|
24
|
+
const { ctx, internals, outDir, conversationId, accountId, startedAtIso, startedAtMs, logStream, } = setup;
|
|
25
|
+
mockCliServer.bind({
|
|
26
|
+
conversationId,
|
|
27
|
+
handlers: internals.getMockHandlers(),
|
|
28
|
+
emitCliMock: internals.emitCliMock,
|
|
29
|
+
isScenarioEnded: () => internals.isScenarioEnded(),
|
|
30
|
+
});
|
|
31
|
+
const initialCursor = await ctx.getCursor();
|
|
32
|
+
const subscription = startOutboundSubscription({
|
|
33
|
+
accountId,
|
|
34
|
+
conversationId,
|
|
35
|
+
initialCursor,
|
|
36
|
+
onMessage: internals.emitOutboundReceived,
|
|
37
|
+
});
|
|
38
|
+
const { failure } = await executeScenario(args.scenario, ctx);
|
|
39
|
+
await subscription.stop();
|
|
40
|
+
await mockCliServer.release();
|
|
41
|
+
const finishedAtMs = Date.now();
|
|
42
|
+
const durationMs = finishedAtMs - startedAtMs;
|
|
43
|
+
const finishedAtIso = new Date(finishedAtMs).toISOString();
|
|
44
|
+
const { entries, judgeUsages, result } = internals.finalize({ failure });
|
|
45
|
+
await waitForTranscriptQuiescence({ conversationId, startedAtIso });
|
|
46
|
+
const snapshot = await fetchTranscriptSnapshot({ conversationId, startedAtIso });
|
|
47
|
+
const { cost: agentCostUsd, turns: agentTurns } = readTranscriptCost(snapshot);
|
|
48
|
+
const judgeUsd = judgeUsages.reduce((sum, u) => sum + judgeCostUsd(u), 0);
|
|
49
|
+
// The per-session store dies with the stack recreation, so keep the raw
|
|
50
|
+
// transcripts as a cell artifact for post-mortems.
|
|
51
|
+
writeFileSync(join(outDir, "transcripts.json"), JSON.stringify(snapshot.sessions));
|
|
52
|
+
const agentCalls = aggregateAgentToolCalls(snapshot.sessions);
|
|
53
|
+
pairAgentCallsWithCliMocks(agentCalls, entries);
|
|
54
|
+
if (agentCalls.length === 0 && (snapshot.error !== undefined || snapshot.databases === 0)) {
|
|
55
|
+
entries.push({
|
|
56
|
+
entrySeq: entries.length,
|
|
57
|
+
ts: finishedAtIso,
|
|
58
|
+
kind: "scenarioLog",
|
|
59
|
+
message: snapshot.error !== undefined
|
|
60
|
+
? `agentToolCall parsing failed: ${snapshot.error}`
|
|
61
|
+
: "agentToolCall parsing skipped: no agent session store found in the gateway",
|
|
62
|
+
});
|
|
63
|
+
}
|
|
64
|
+
const agentEntries = buildAgentToolCallEntries(agentCalls, entries.length, finishedAtIso);
|
|
65
|
+
appendAgentCallsToLog(logStream, agentEntries);
|
|
66
|
+
await closeStream(logStream);
|
|
67
|
+
const merged = mergeTimeline(entries, agentEntries);
|
|
68
|
+
const report = {
|
|
69
|
+
schemaVersion: 4,
|
|
70
|
+
scenario: args.scenario,
|
|
71
|
+
channel: args.channel,
|
|
72
|
+
model: args.modelId,
|
|
73
|
+
conversationId,
|
|
74
|
+
accountId,
|
|
75
|
+
startedAt: startedAtIso,
|
|
76
|
+
finishedAt: finishedAtIso,
|
|
77
|
+
durationMs,
|
|
78
|
+
cost: {
|
|
79
|
+
agentUsd: agentCostUsd,
|
|
80
|
+
judgeUsd,
|
|
81
|
+
totalUsd: agentCostUsd + judgeUsd,
|
|
82
|
+
agentTurns,
|
|
83
|
+
},
|
|
84
|
+
result,
|
|
85
|
+
entries: merged.map(prepareEntryForReport),
|
|
86
|
+
};
|
|
87
|
+
// Write the cell record BEFORE the artifact-dir rename, to a stable sibling path.
|
|
88
|
+
const leafBase = basename(outDir);
|
|
89
|
+
const resultsPath = join(args.resultsDir, `${leafBase}.json`);
|
|
90
|
+
mkdirSync(args.resultsDir, { recursive: true });
|
|
91
|
+
const finalOutDir = writeReportArtifacts(outDir, result.verdict, report);
|
|
92
|
+
writeCellResult(resultsPath, {
|
|
93
|
+
schemaVersion: 3,
|
|
94
|
+
scenarioId: args.scenario,
|
|
95
|
+
channel: args.channel,
|
|
96
|
+
model: args.modelId,
|
|
97
|
+
iterationIndex: args.iterationIndex,
|
|
98
|
+
verdict: result.verdict,
|
|
99
|
+
durationMs,
|
|
100
|
+
conversationId,
|
|
101
|
+
artifactDirName: basename(finalOutDir),
|
|
102
|
+
agentCostUsd,
|
|
103
|
+
agentTurns,
|
|
104
|
+
judgeUsd,
|
|
105
|
+
judgeUsages,
|
|
106
|
+
});
|
|
107
|
+
console.log(`[${args.channel}] ${args.scenario} ${result.verdict} in ${durationMs}ms — ${finalOutDir}`);
|
|
108
|
+
return result.verdict === "pass" ? 0 : 1;
|
|
109
|
+
}
|
|
110
|
+
finally {
|
|
111
|
+
await mockCliServer.close();
|
|
112
|
+
}
|
|
113
|
+
}
|
|
114
|
+
function setupRun(args) {
|
|
115
|
+
const conversationId = `${args.scenario}-${args.channel}-${shortRand()}`;
|
|
116
|
+
const accountId = args.channel;
|
|
117
|
+
const leaf = cellLeafName({
|
|
118
|
+
scenarioId: args.scenario,
|
|
119
|
+
modelId: args.modelId,
|
|
120
|
+
channel: args.channel,
|
|
121
|
+
iterationIndex: args.iterationIndex,
|
|
122
|
+
iterationWidth: args.iterationWidth,
|
|
123
|
+
});
|
|
124
|
+
const outDir = join(ARTIFACTS_ROOT, args.baseStamp, leaf);
|
|
125
|
+
mkdirSync(outDir, { recursive: true });
|
|
126
|
+
const startedAtIso = new Date().toISOString();
|
|
127
|
+
const startedAtMs = Date.now();
|
|
128
|
+
const logStream = createWriteStream(join(outDir, "scenario-log.jsonl"), { flags: "a" });
|
|
129
|
+
const emitSink = (event) => {
|
|
130
|
+
if (event.type === "entry") {
|
|
131
|
+
logStream.write(`${JSON.stringify(event.entry)}\n`);
|
|
132
|
+
}
|
|
133
|
+
else {
|
|
134
|
+
logStream.write(`${JSON.stringify({ entrySeq: event.entrySeq, augment: event.patch })}\n`);
|
|
135
|
+
}
|
|
136
|
+
};
|
|
137
|
+
const { ctx, internals } = createContext({
|
|
138
|
+
channel: args.channel,
|
|
139
|
+
conversationId,
|
|
140
|
+
startedAtIso,
|
|
141
|
+
emitSink,
|
|
142
|
+
});
|
|
143
|
+
return {
|
|
144
|
+
ctx,
|
|
145
|
+
internals,
|
|
146
|
+
outDir,
|
|
147
|
+
conversationId,
|
|
148
|
+
accountId,
|
|
149
|
+
startedAtIso,
|
|
150
|
+
startedAtMs,
|
|
151
|
+
logStream,
|
|
152
|
+
};
|
|
153
|
+
}
|
|
154
|
+
function startOutboundSubscription(params) {
|
|
155
|
+
const abort = new AbortController();
|
|
156
|
+
const done = (async () => {
|
|
157
|
+
let cursor = params.initialCursor;
|
|
158
|
+
while (!abort.signal.aborted) {
|
|
159
|
+
let result;
|
|
160
|
+
try {
|
|
161
|
+
result = await pollQaBus({
|
|
162
|
+
baseUrl: BUS_URL,
|
|
163
|
+
accountId: params.accountId,
|
|
164
|
+
cursor,
|
|
165
|
+
timeoutMs: 1000,
|
|
166
|
+
});
|
|
167
|
+
}
|
|
168
|
+
catch {
|
|
169
|
+
if (abort.signal.aborted)
|
|
170
|
+
return;
|
|
171
|
+
await new Promise((r) => setTimeout(r, 200));
|
|
172
|
+
continue;
|
|
173
|
+
}
|
|
174
|
+
cursor = result.cursor;
|
|
175
|
+
for (const e of result.events) {
|
|
176
|
+
if (e.kind !== "outbound-message" && e.kind !== "message-edited")
|
|
177
|
+
continue;
|
|
178
|
+
if (e.message.conversation.id !== params.conversationId)
|
|
179
|
+
continue;
|
|
180
|
+
params.onMessage(e.message);
|
|
181
|
+
}
|
|
182
|
+
}
|
|
183
|
+
})();
|
|
184
|
+
return {
|
|
185
|
+
stop: async () => {
|
|
186
|
+
abort.abort();
|
|
187
|
+
await done.catch(() => { });
|
|
188
|
+
},
|
|
189
|
+
};
|
|
190
|
+
}
|
|
191
|
+
async function executeScenario(scenarioId, ctx) {
|
|
192
|
+
try {
|
|
193
|
+
const mod = await import(`${SCENARIOS_ROOT}/${scenarioId}.ts`);
|
|
194
|
+
const fn = mod.default;
|
|
195
|
+
if (typeof fn !== "function") {
|
|
196
|
+
throw new Error(`scenario ${scenarioId} has no default export function`);
|
|
197
|
+
}
|
|
198
|
+
await fn(ctx);
|
|
199
|
+
return { failure: undefined };
|
|
200
|
+
}
|
|
201
|
+
catch (err) {
|
|
202
|
+
const e = err;
|
|
203
|
+
const isAssertion = e?.name === "AssertionError";
|
|
204
|
+
return {
|
|
205
|
+
failure: {
|
|
206
|
+
name: e?.name ?? "Error",
|
|
207
|
+
message: e?.message ?? String(err),
|
|
208
|
+
stack: e?.stack,
|
|
209
|
+
source: isAssertion ? "assertion" : "scenarioThrow",
|
|
210
|
+
},
|
|
211
|
+
};
|
|
212
|
+
}
|
|
213
|
+
}
|
|
214
|
+
function pairAgentCallsWithCliMocks(agentCalls, entries) {
|
|
215
|
+
const cliMockQueues = new Map();
|
|
216
|
+
for (const e of entries) {
|
|
217
|
+
if (e.kind !== "cliMock")
|
|
218
|
+
continue;
|
|
219
|
+
const q = cliMockQueues.get(e.call.cli) ?? [];
|
|
220
|
+
q.push(e.ts);
|
|
221
|
+
cliMockQueues.set(e.call.cli, q);
|
|
222
|
+
}
|
|
223
|
+
for (const call of agentCalls) {
|
|
224
|
+
const cli = leadingCli(call.input);
|
|
225
|
+
if (!cli)
|
|
226
|
+
continue;
|
|
227
|
+
const q = cliMockQueues.get(cli);
|
|
228
|
+
if (!q || q.length === 0)
|
|
229
|
+
continue;
|
|
230
|
+
call.inferredStartedAt = q.shift();
|
|
231
|
+
}
|
|
232
|
+
}
|
|
233
|
+
export function leadingCli(input) {
|
|
234
|
+
if (!input || typeof input !== "object")
|
|
235
|
+
return;
|
|
236
|
+
const cmd = input.command;
|
|
237
|
+
if (typeof cmd !== "string")
|
|
238
|
+
return;
|
|
239
|
+
const stripped = cmd.replace(/^(?:\s*cd\s+[^&;|]+(?:&&|;|\|\|)\s*)+/, "");
|
|
240
|
+
const m = stripped.match(/^\s*(?:[A-Za-z_][A-Za-z0-9_]*=\S+\s+)*([A-Za-z0-9_./-]+)/);
|
|
241
|
+
if (!m)
|
|
242
|
+
return;
|
|
243
|
+
// Reject bare env-assignments (e.g. `FOO=bar` alone): the capture class
|
|
244
|
+
// stops at `=`, so we'd otherwise return "FOO" as if it were a command.
|
|
245
|
+
if (stripped[m[0].length] === "=")
|
|
246
|
+
return;
|
|
247
|
+
return m[1].split("/").pop();
|
|
248
|
+
}
|
|
249
|
+
/**
|
|
250
|
+
* Build agentToolCall entries with stable `entrySeq` values continuing the live
|
|
251
|
+
* jsonl emission order. These values are shared by both artifact files —
|
|
252
|
+
* `scenario-log.jsonl` (appended at the tail in ts order) and `report.json`
|
|
253
|
+
* (interleaved with live entries by ts).
|
|
254
|
+
*/
|
|
255
|
+
function buildAgentToolCallEntries(agentCalls, liveEntryCount, fallbackTs) {
|
|
256
|
+
const sorted = [...agentCalls].sort(compareStartedAt);
|
|
257
|
+
return sorted.map((call, i) => ({
|
|
258
|
+
entrySeq: liveEntryCount + i,
|
|
259
|
+
ts: call.startedAt ?? fallbackTs,
|
|
260
|
+
kind: "agentToolCall",
|
|
261
|
+
call,
|
|
262
|
+
}));
|
|
263
|
+
}
|
|
264
|
+
// A call whose message carried no timestamp sorts to the end, never the front.
|
|
265
|
+
function compareStartedAt(a, b) {
|
|
266
|
+
if (a.startedAt === undefined)
|
|
267
|
+
return b.startedAt === undefined ? 0 : 1;
|
|
268
|
+
if (b.startedAt === undefined)
|
|
269
|
+
return -1;
|
|
270
|
+
return a.startedAt < b.startedAt ? -1 : 1;
|
|
271
|
+
}
|
|
272
|
+
/**
|
|
273
|
+
* Append parsed agentToolCall entries to the tail of `scenario-log.jsonl` in
|
|
274
|
+
* ts order. Their `entrySeq` values continue past the live entries' and are the
|
|
275
|
+
* same values used in `report.json` — readers can cross-reference by `entrySeq`.
|
|
276
|
+
*/
|
|
277
|
+
function appendAgentCallsToLog(logStream, agentEntries) {
|
|
278
|
+
for (const entry of agentEntries) {
|
|
279
|
+
logStream.write(`${JSON.stringify(entry)}\n`);
|
|
280
|
+
}
|
|
281
|
+
}
|
|
282
|
+
/**
|
|
283
|
+
* Interleave live entries with agentToolCall entries by `ts` for `report.json`.
|
|
284
|
+
* Entries keep their original `entrySeq`; the array order is ts, so iterating
|
|
285
|
+
* gives the unified timeline while `entrySeq` remains a cross-file identifier.
|
|
286
|
+
*/
|
|
287
|
+
function mergeTimeline(entries, agentEntries) {
|
|
288
|
+
return [...entries, ...agentEntries].sort((a, b) => {
|
|
289
|
+
if (a.ts !== b.ts)
|
|
290
|
+
return a.ts < b.ts ? -1 : 1;
|
|
291
|
+
return kindOrder(a.kind) - kindOrder(b.kind);
|
|
292
|
+
});
|
|
293
|
+
}
|
|
294
|
+
function kindOrder(k) {
|
|
295
|
+
return k === "agentToolCall" ? 0 : 1;
|
|
296
|
+
}
|
|
297
|
+
const REPORT_CONTENT_TRUNCATE_AT = 60;
|
|
298
|
+
/**
|
|
299
|
+
* Replace bulky `content` with `truncatedContent` for `report.json` only. The
|
|
300
|
+
* original entry (already streamed to `scenario-log.jsonl`) keeps the full
|
|
301
|
+
* content; this returns a shallow-cloned `agentToolCall` entry with the
|
|
302
|
+
* truncated `result`.
|
|
303
|
+
*/
|
|
304
|
+
function prepareEntryForReport(entry) {
|
|
305
|
+
if (entry.kind !== "agentToolCall" || !entry.call.result)
|
|
306
|
+
return entry;
|
|
307
|
+
const text = truncatableResultText(entry.call.toolName, entry.call.result.content);
|
|
308
|
+
if (text === undefined || text.length <= REPORT_CONTENT_TRUNCATE_AT)
|
|
309
|
+
return entry;
|
|
310
|
+
const truncatedContent = `${text.slice(0, REPORT_CONTENT_TRUNCATE_AT).replace(/\s+$/, "")}…`;
|
|
311
|
+
return {
|
|
312
|
+
...entry,
|
|
313
|
+
call: {
|
|
314
|
+
...entry.call,
|
|
315
|
+
result: { isError: entry.call.result.isError, truncatedContent },
|
|
316
|
+
},
|
|
317
|
+
};
|
|
318
|
+
}
|
|
319
|
+
/**
|
|
320
|
+
* The string to truncate for `report.json`, or `undefined` to keep `content`
|
|
321
|
+
* as-is. The `read` tool returns its file content as text blocks rather than a
|
|
322
|
+
* string, so its text is joined — the file is identified by `input`, so the
|
|
323
|
+
* report needs no more than a preview.
|
|
324
|
+
*/
|
|
325
|
+
function truncatableResultText(toolName, content) {
|
|
326
|
+
if (typeof content === "string")
|
|
327
|
+
return content;
|
|
328
|
+
if (toolName === "read")
|
|
329
|
+
return readBlocksText(content);
|
|
330
|
+
return;
|
|
331
|
+
}
|
|
332
|
+
function readBlocksText(content) {
|
|
333
|
+
if (!Array.isArray(content))
|
|
334
|
+
return;
|
|
335
|
+
let text = "";
|
|
336
|
+
for (const block of content) {
|
|
337
|
+
if (isRecord(block) && block.type === "text" && typeof block.text === "string") {
|
|
338
|
+
text += block.text;
|
|
339
|
+
}
|
|
340
|
+
}
|
|
341
|
+
return text.length > 0 ? text : undefined;
|
|
342
|
+
}
|
|
343
|
+
function isRecord(value) {
|
|
344
|
+
return typeof value === "object" && value !== null;
|
|
345
|
+
}
|
|
346
|
+
function writeReportArtifacts(outDir, status, report) {
|
|
347
|
+
writeFileSync(join(outDir, "report.json"), JSON.stringify(report, null, 2));
|
|
348
|
+
const renamed = `${outDir}-${status.toUpperCase()}`;
|
|
349
|
+
try {
|
|
350
|
+
renameSync(outDir, renamed);
|
|
351
|
+
return renamed;
|
|
352
|
+
}
|
|
353
|
+
catch (err) {
|
|
354
|
+
console.warn(`runner: failed to rename ${outDir} -> ${renamed}:`, err);
|
|
355
|
+
return outDir;
|
|
356
|
+
}
|
|
357
|
+
}
|
|
358
|
+
function closeStream(stream) {
|
|
359
|
+
return new Promise((resolve, reject) => {
|
|
360
|
+
stream.end((err) => (err ? reject(err) : resolve()));
|
|
361
|
+
});
|
|
362
|
+
}
|
|
363
|
+
function shortRand() {
|
|
364
|
+
return randomBytes(3).toString("hex");
|
|
365
|
+
}
|
|
366
|
+
if (import.meta.url === `file://${process.argv[1]}`) {
|
|
367
|
+
main().catch((err) => {
|
|
368
|
+
console.error("runner crash:", err);
|
|
369
|
+
process.exit(1);
|
|
370
|
+
});
|
|
371
|
+
}
|
package/dist/summary.js
ADDED
|
@@ -0,0 +1,64 @@
|
|
|
1
|
+
import { join } from "node:path";
|
|
2
|
+
import { judgeCostUsd } from "./cost.js";
|
|
3
|
+
function artifactsRoot() {
|
|
4
|
+
return process.env.OPENCLAW_TEST_ARTIFACTS_DIR ?? "/opt/openclaw-test/artifacts";
|
|
5
|
+
}
|
|
6
|
+
function groupByPair(results) {
|
|
7
|
+
const order = [];
|
|
8
|
+
const map = new Map();
|
|
9
|
+
for (const r of results) {
|
|
10
|
+
const key = `${r.scenarioId}|${r.channel}|${r.model}`;
|
|
11
|
+
let agg = map.get(key);
|
|
12
|
+
if (!agg) {
|
|
13
|
+
agg = {
|
|
14
|
+
scenarioId: r.scenarioId,
|
|
15
|
+
channel: r.channel,
|
|
16
|
+
model: r.model,
|
|
17
|
+
runCount: 0,
|
|
18
|
+
passCount: 0,
|
|
19
|
+
durationSumMs: 0,
|
|
20
|
+
};
|
|
21
|
+
map.set(key, agg);
|
|
22
|
+
order.push(key);
|
|
23
|
+
}
|
|
24
|
+
agg.runCount += 1;
|
|
25
|
+
agg.durationSumMs += r.durationMs;
|
|
26
|
+
if (r.verdict === "pass")
|
|
27
|
+
agg.passCount += 1;
|
|
28
|
+
}
|
|
29
|
+
return order.map((k) => map.get(k));
|
|
30
|
+
}
|
|
31
|
+
export function printSummary(results, baseStamp) {
|
|
32
|
+
const aggregates = groupByPair(results);
|
|
33
|
+
console.log("");
|
|
34
|
+
console.log("Summary:");
|
|
35
|
+
for (const a of aggregates) {
|
|
36
|
+
const passed = a.passCount === a.runCount;
|
|
37
|
+
const verdict = (passed ? "PASS" : "FAIL").padEnd(4, " ");
|
|
38
|
+
const counts = `${a.passCount}/${a.runCount}`.padStart(7, " ");
|
|
39
|
+
console.log(` ${verdict} ${a.channel.padEnd(12, " ")} ${a.model.padEnd(20, " ")} ${a.scenarioId.padEnd(40, " ")} ${counts} in ${a.durationSumMs}ms`);
|
|
40
|
+
}
|
|
41
|
+
console.log("");
|
|
42
|
+
console.log("Artifacts:");
|
|
43
|
+
console.log(` ${join(artifactsRoot(), baseStamp)}`);
|
|
44
|
+
}
|
|
45
|
+
export function printTotalCost(results) {
|
|
46
|
+
console.log("");
|
|
47
|
+
console.log(costLine(results));
|
|
48
|
+
const models = [...new Set(results.map((r) => r.model))];
|
|
49
|
+
if (models.length > 1) {
|
|
50
|
+
console.log("Per-model cost:");
|
|
51
|
+
for (const model of models) {
|
|
52
|
+
console.log(` ${model.padEnd(20, " ")} ${costLine(results.filter((r) => r.model === model))}`);
|
|
53
|
+
}
|
|
54
|
+
}
|
|
55
|
+
}
|
|
56
|
+
function costLine(results) {
|
|
57
|
+
const judgeCost = results
|
|
58
|
+
.flatMap((r) => r.judgeUsages)
|
|
59
|
+
.reduce((sum, u) => sum + judgeCostUsd(u), 0);
|
|
60
|
+
const agentCost = results.reduce((sum, r) => sum + r.agentCostUsd, 0);
|
|
61
|
+
const agentTurns = results.reduce((sum, r) => sum + r.agentTurns, 0);
|
|
62
|
+
const totalCost = agentCost + judgeCost;
|
|
63
|
+
return `Total LLM cost: $${totalCost.toFixed(4)} (agent: $${agentCost.toFixed(4)} over ${agentTurns} turns, judge: $${judgeCost.toFixed(4)})`;
|
|
64
|
+
}
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
export {};
|
|
@@ -0,0 +1,50 @@
|
|
|
1
|
+
// Gateway-side transcript dump, invoked by the runner through the exec-watcher
|
|
2
|
+
// RPC: `node transcript-dump.js <sinceIso> <conversationId> <outPath> [threadId...]`.
|
|
3
|
+
// OpenClaw 2026.8+ persists each session's transcript as SQLite rows in the
|
|
4
|
+
// per-agent store. Unlike the trajectory diagnostics (whose payloads are
|
|
5
|
+
// node-capped and redacted), the transcript is the full-fidelity record the
|
|
6
|
+
// gateway itself replays, appended per message — tool calls become visible as
|
|
7
|
+
// they happen, not at turn end. This script extracts the transcripts of a
|
|
8
|
+
// conversation's sessions and writes them as JSON to `outPath` (stdout would
|
|
9
|
+
// hit the watcher's 1 MiB cap).
|
|
10
|
+
import { readdirSync, renameSync, writeFileSync } from "node:fs";
|
|
11
|
+
import { homedir } from "node:os";
|
|
12
|
+
import { join } from "node:path";
|
|
13
|
+
import { DatabaseSync } from "node:sqlite";
|
|
14
|
+
import { readConversationSessions } from "./transcript-store.js";
|
|
15
|
+
main();
|
|
16
|
+
function main() {
|
|
17
|
+
const [sinceIso, conversationId, outPath, ...threadIds] = process.argv.slice(2);
|
|
18
|
+
if (sinceIso === undefined || conversationId === undefined || outPath === undefined) {
|
|
19
|
+
console.error("usage: transcript-dump.js <sinceIso> <conversationId> <outPath> [threadId...]");
|
|
20
|
+
process.exit(2);
|
|
21
|
+
}
|
|
22
|
+
const sinceMs = Date.parse(sinceIso);
|
|
23
|
+
const databases = findAgentDatabases();
|
|
24
|
+
const sessions = databases.flatMap((dbPath) => readConversationSessions(dbPath, sinceMs, conversationId, threadIds));
|
|
25
|
+
const tmpPath = `${outPath}.tmp`;
|
|
26
|
+
writeFileSync(tmpPath, JSON.stringify({ databases: databases.length, sessions }));
|
|
27
|
+
renameSync(tmpPath, outPath);
|
|
28
|
+
}
|
|
29
|
+
function findAgentDatabases() {
|
|
30
|
+
const agentsDir = join(homedir(), ".openclaw", "agents");
|
|
31
|
+
let agentIds;
|
|
32
|
+
try {
|
|
33
|
+
agentIds = readdirSync(agentsDir);
|
|
34
|
+
}
|
|
35
|
+
catch {
|
|
36
|
+
return [];
|
|
37
|
+
}
|
|
38
|
+
return agentIds
|
|
39
|
+
.map((id) => join(agentsDir, id, "agent", "openclaw-agent.sqlite"))
|
|
40
|
+
.filter(canOpenDatabase);
|
|
41
|
+
}
|
|
42
|
+
function canOpenDatabase(path) {
|
|
43
|
+
try {
|
|
44
|
+
new DatabaseSync(path, { readOnly: true }).close();
|
|
45
|
+
return true;
|
|
46
|
+
}
|
|
47
|
+
catch {
|
|
48
|
+
return false;
|
|
49
|
+
}
|
|
50
|
+
}
|
|
@@ -0,0 +1,67 @@
|
|
|
1
|
+
import type { AgentToolCall } from "./report.js";
|
|
2
|
+
export interface TranscriptSnapshot {
|
|
3
|
+
/** Agent stores found in the gateway. 0 means no store yet — or none at all. */
|
|
4
|
+
databases: number;
|
|
5
|
+
sessions: TranscriptSession[];
|
|
6
|
+
/**
|
|
7
|
+
* Set when the dump itself failed (non-zero exit, timeout, unreadable
|
|
8
|
+
* output). `databases` and `sessions` are then placeholders, not
|
|
9
|
+
* observations — report the failure, not "no store found".
|
|
10
|
+
*/
|
|
11
|
+
error?: string;
|
|
12
|
+
}
|
|
13
|
+
export interface TranscriptSession {
|
|
14
|
+
sessionKey: string;
|
|
15
|
+
sessionId: string;
|
|
16
|
+
/** OpenClaw's neutral message shape: assistant `toolCall` blocks, `toolResult` messages. */
|
|
17
|
+
messages: unknown[];
|
|
18
|
+
}
|
|
19
|
+
/** The transcripts of the conversation and its bus-owned thread sessions. */
|
|
20
|
+
export declare function fetchTranscriptSnapshot(opts: {
|
|
21
|
+
conversationId: string;
|
|
22
|
+
startedAtIso: string;
|
|
23
|
+
}): Promise<TranscriptSnapshot>;
|
|
24
|
+
/**
|
|
25
|
+
* Polls the transcripts until they are *quiescent* for this conversation — no
|
|
26
|
+
* new message for `settleMs` AND no session's turn is demonstrably open — or
|
|
27
|
+
* the budget expires. The transcript is appended per message, so a settle
|
|
28
|
+
* window alone would return between two tool calls of one turn (a single
|
|
29
|
+
* model round-trip routinely exceeds it); the open-turn gate holds until the
|
|
30
|
+
* turn-final assistant message — the one carrying `usage.cost.total` — has
|
|
31
|
+
* landed. A conversation spans multiple OpenClaw sessions (e.g. Discord's
|
|
32
|
+
* channel session plus a per-thread session).
|
|
33
|
+
*/
|
|
34
|
+
export declare function waitForTranscriptQuiescence(opts: {
|
|
35
|
+
conversationId: string;
|
|
36
|
+
startedAtIso: string;
|
|
37
|
+
maxWaitMs?: number;
|
|
38
|
+
pollMs?: number;
|
|
39
|
+
settleMs?: number;
|
|
40
|
+
}): Promise<void>;
|
|
41
|
+
/**
|
|
42
|
+
* A turn is open when a session's last message is a tool result or an
|
|
43
|
+
* assistant stop for tool use — more of the turn is coming. A trailing user
|
|
44
|
+
* message does not count as open: some sessions of a conversation never get a
|
|
45
|
+
* reply (the agent answers in a sibling session), and treating them as open
|
|
46
|
+
* would burn the whole budget on every cell.
|
|
47
|
+
*/
|
|
48
|
+
export declare function hasOpenTurn(sessions: TranscriptSession[]): boolean;
|
|
49
|
+
/**
|
|
50
|
+
* Cost lives per assistant message, as `usage.cost.total`. Sum across every
|
|
51
|
+
* session of the conversation; `turns` counts the cost-bearing messages.
|
|
52
|
+
*/
|
|
53
|
+
export declare function readTranscriptCost(snapshot: TranscriptSnapshot): {
|
|
54
|
+
cost: number;
|
|
55
|
+
turns: number;
|
|
56
|
+
};
|
|
57
|
+
/** One-shot fetch + aggregation of the conversation's agent tool calls. */
|
|
58
|
+
export declare function parseAgentToolCalls(opts: {
|
|
59
|
+
conversationId: string;
|
|
60
|
+
startedAtIso: string;
|
|
61
|
+
}): Promise<AgentToolCall[]>;
|
|
62
|
+
/**
|
|
63
|
+
* Walk each session's messages collecting assistant `toolCall` blocks matched
|
|
64
|
+
* with `toolResult` messages, then union across sessions deduped by
|
|
65
|
+
* `toolUseId`.
|
|
66
|
+
*/
|
|
67
|
+
export declare function aggregateAgentToolCalls(sessions: TranscriptSession[]): AgentToolCall[];
|