@alignfirst/openclaw-test 0.20.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (53) hide show
  1. package/Dockerfile.base +45 -0
  2. package/LICENSE +21 -0
  3. package/README.md +132 -0
  4. package/bin/cli.mjs +7 -0
  5. package/dist/bus.d.ts +1 -0
  6. package/dist/bus.js +20 -0
  7. package/dist/cell-result.d.ts +25 -0
  8. package/dist/cell-result.js +32 -0
  9. package/dist/cli.d.ts +5 -0
  10. package/dist/cli.js +110 -0
  11. package/dist/context.d.ts +220 -0
  12. package/dist/context.js +479 -0
  13. package/dist/cost.d.ts +2 -0
  14. package/dist/cost.js +19 -0
  15. package/dist/env-cli.d.ts +5 -0
  16. package/dist/env-cli.js +645 -0
  17. package/dist/exec-rpc.d.ts +13 -0
  18. package/dist/exec-rpc.js +46 -0
  19. package/dist/index.d.ts +3 -0
  20. package/dist/index.js +1 -0
  21. package/dist/judge.d.ts +47 -0
  22. package/dist/judge.js +183 -0
  23. package/dist/loop.d.ts +62 -0
  24. package/dist/loop.js +316 -0
  25. package/dist/mock-cli-server.d.ts +40 -0
  26. package/dist/mock-cli-server.js +194 -0
  27. package/dist/mock-cli-shim.d.ts +7 -0
  28. package/dist/mock-cli-shim.js +89 -0
  29. package/dist/models.d.ts +23 -0
  30. package/dist/models.js +78 -0
  31. package/dist/parse-tagged-json.d.ts +1 -0
  32. package/dist/parse-tagged-json.js +16 -0
  33. package/dist/report.d.ts +215 -0
  34. package/dist/report.js +19 -0
  35. package/dist/runner-args.d.ts +11 -0
  36. package/dist/runner-args.js +84 -0
  37. package/dist/runner.d.ts +2 -0
  38. package/dist/runner.js +371 -0
  39. package/dist/summary.d.ts +3 -0
  40. package/dist/summary.js +64 -0
  41. package/dist/transcript-dump.d.ts +1 -0
  42. package/dist/transcript-dump.js +50 -0
  43. package/dist/transcript-log.d.ts +67 -0
  44. package/dist/transcript-log.js +199 -0
  45. package/dist/transcript-store.d.ts +6 -0
  46. package/dist/transcript-store.js +69 -0
  47. package/docker-compose.yml +83 -0
  48. package/exec-watcher.mjs +155 -0
  49. package/package.json +60 -0
  50. package/templates/.env.local.example +28 -0
  51. package/templates/Dockerfile +47 -0
  52. package/templates/docker-compose.yml +10 -0
  53. package/templates/openclaw.json +40 -0
@@ -0,0 +1,40 @@
1
+ /**
2
+ * Runner-side HTTP endpoint that the gateway-side shim calls.
3
+ *
4
+ * POST /mock-cli/invoke
5
+ * { cli, argv, cwd, stdin }
6
+ * → { stdout, stderr, exitCode }
7
+ *
8
+ * A single in-flight registry is bound by the runner for the lifetime of one
9
+ * scenario. Per-conversation handler registries are populated by `ctx.mockCli`.
10
+ */
11
+ import type { CliMockCall, CliMockHandler } from "./report.js";
12
+ export declare const MOCK_CLI_PORT = 43124;
13
+ export interface ConversationRegistry {
14
+ conversationId: string;
15
+ handlers: Map<string, CliMockHandler>;
16
+ /** Emit a `cliMock` event into the scenario's event stream. */
17
+ emitCliMock: (call: CliMockCall) => void;
18
+ /**
19
+ * `true` once the scenario has called `ctx.markScenarioAsEnded`. After that,
20
+ * the server bypasses handler dispatch: every incoming call is answered with
21
+ * the scenario-ended stub message so a lingering agent — still running while
22
+ * we move on — can read it and stop.
23
+ */
24
+ isScenarioEnded: () => boolean;
25
+ }
26
+ export interface MockCliServer {
27
+ /** Bind the in-flight conversation registry for the lifetime of one scenario. */
28
+ bind(reg: ConversationRegistry): void;
29
+ /**
30
+ * Release the binding. Waits for the gateway side to go quiet (no incoming
31
+ * call for `RELEASE_QUIET_MS`) before clearing, so any straggler from the
32
+ * just-ended scenario lands on its own registry and gets the scenario-ended
33
+ * stub — instead of leaking into the next scenario's bind window as a
34
+ * spurious `UnexpectedCall`. Bounded by `RELEASE_HARD_TIMEOUT_MS`.
35
+ */
36
+ release(): Promise<void>;
37
+ /** Stop the HTTP listener. */
38
+ close(): Promise<void>;
39
+ }
40
+ export declare function startMockCliServer(): MockCliServer;
@@ -0,0 +1,194 @@
1
+ /**
2
+ * Runner-side HTTP endpoint that the gateway-side shim calls.
3
+ *
4
+ * POST /mock-cli/invoke
5
+ * { cli, argv, cwd, stdin }
6
+ * → { stdout, stderr, exitCode }
7
+ *
8
+ * A single in-flight registry is bound by the runner for the lifetime of one
9
+ * scenario. Per-conversation handler registries are populated by `ctx.mockCli`.
10
+ */
11
+ import { createServer } from "node:http";
12
+ import { Readable, Writable } from "node:stream";
13
+ export const MOCK_CLI_PORT = 43124;
14
+ const SCENARIO_ENDED_STUB_MESSAGE = "Stubbed call. If you see this, it means that you are in a test scenario. You should stop and acknowledge to the user.\n";
15
+ const RELEASE_QUIET_MS = 500;
16
+ const RELEASE_HARD_TIMEOUT_MS = 5_000;
17
+ const RELEASE_POLL_MS = 100;
18
+ export function startMockCliServer() {
19
+ let current;
20
+ let lastCallAtMs = 0;
21
+ const server = createServer(async (req, res) => {
22
+ if (req.method !== "POST" || req.url !== "/mock-cli/invoke") {
23
+ res.statusCode = 404;
24
+ res.end("not found");
25
+ return;
26
+ }
27
+ lastCallAtMs = Date.now();
28
+ await handleInvoke(req, res, current);
29
+ });
30
+ server.listen(MOCK_CLI_PORT, "0.0.0.0", () => {
31
+ console.log(`mock-cli server listening on 0.0.0.0:${MOCK_CLI_PORT}`);
32
+ });
33
+ return {
34
+ bind: (reg) => {
35
+ current = reg;
36
+ },
37
+ release: async () => {
38
+ const deadline = Date.now() + RELEASE_HARD_TIMEOUT_MS;
39
+ while (lastCallAtMs > 0 && Date.now() - lastCallAtMs < RELEASE_QUIET_MS) {
40
+ if (Date.now() >= deadline)
41
+ break;
42
+ await new Promise((r) => setTimeout(r, RELEASE_POLL_MS));
43
+ }
44
+ current = undefined;
45
+ lastCallAtMs = 0;
46
+ },
47
+ close: () => new Promise((resolve) => {
48
+ server.close(() => resolve());
49
+ // Force-destroy lingering keep-alive sockets. A mocked CLI is invoked through the
50
+ // gateway shim over HTTP keep-alive; after the call the socket stays idle-open, and a
51
+ // bare `server.close()` waits for it indefinitely — the runner then hangs past the
52
+ // verdict instead of exiting. `closeAllConnections` lets `close()` complete at once.
53
+ server.closeAllConnections();
54
+ }),
55
+ };
56
+ }
57
+ async function handleInvoke(req, res, reg) {
58
+ let body;
59
+ try {
60
+ body = await readJsonBody(req);
61
+ }
62
+ catch (err) {
63
+ res.statusCode = 400;
64
+ res.end(`bad json: ${err.message}`);
65
+ return;
66
+ }
67
+ const invoke = normalizeInvoke(body);
68
+ if (!invoke) {
69
+ res.statusCode = 400;
70
+ res.end("missing cli");
71
+ return;
72
+ }
73
+ if (!reg) {
74
+ respondMissingScenario(res);
75
+ return;
76
+ }
77
+ const startedAt = new Date().toISOString();
78
+ const startedAtMs = Date.now();
79
+ if (reg.isScenarioEnded()) {
80
+ // Verdict is already decided. Don't dispatch to handlers and don't surface
81
+ // failures — just give the (still-running) agent a clear stop signal.
82
+ emitCallEvent(reg, invoke, { stdout: SCENARIO_ENDED_STUB_MESSAGE, stderr: "", exitCode: 0 }, startedAt, startedAtMs);
83
+ respondJson(res, { stdout: SCENARIO_ENDED_STUB_MESSAGE, stderr: "", exitCode: 0 });
84
+ return;
85
+ }
86
+ const handler = reg.handlers.get(invoke.cli);
87
+ if (!handler) {
88
+ const stderr = `mock-cli: unexpected call to ${invoke.cli}\n`;
89
+ emitCallEvent(reg, invoke, {
90
+ stdout: "",
91
+ stderr,
92
+ exitCode: 1,
93
+ handlerError: { name: "UnexpectedCall", message: `unexpected call to ${invoke.cli}` },
94
+ }, startedAt, startedAtMs);
95
+ respondJson(res, { stdout: "", stderr, exitCode: 1, missing: true });
96
+ return;
97
+ }
98
+ const outcome = await runHandler(handler, invoke);
99
+ emitCallEvent(reg, invoke, outcome, startedAt, startedAtMs);
100
+ respondJson(res, { stdout: outcome.stdout, stderr: outcome.stderr, exitCode: outcome.exitCode });
101
+ }
102
+ function normalizeInvoke(body) {
103
+ const cli = String(body.cli ?? "");
104
+ if (!cli)
105
+ return;
106
+ return {
107
+ cli,
108
+ argv: Array.isArray(body.argv) ? body.argv.map(String) : [],
109
+ cwd: String(body.cwd ?? ""),
110
+ stdin: String(body.stdin ?? ""),
111
+ };
112
+ }
113
+ async function runHandler(handler, invoke) {
114
+ const stdinStream = Readable.from([Buffer.from(invoke.stdin, "utf8")]);
115
+ const outCap = collectStream();
116
+ const errCap = collectStream();
117
+ const args = {
118
+ argv: invoke.argv,
119
+ cwd: invoke.cwd,
120
+ stdin: stdinStream,
121
+ stdout: outCap.writable,
122
+ stderr: errCap.writable,
123
+ };
124
+ let exitCode = 0;
125
+ let handlerError;
126
+ try {
127
+ const result = await handler(args);
128
+ exitCode = typeof result === "number" ? result : 0;
129
+ }
130
+ catch (err) {
131
+ const e = err;
132
+ handlerError = {
133
+ name: e?.name ?? "Error",
134
+ message: e?.message ?? String(err),
135
+ stack: e?.stack,
136
+ };
137
+ exitCode = 1;
138
+ }
139
+ outCap.writable.end();
140
+ errCap.writable.end();
141
+ return { stdout: outCap.read(), stderr: errCap.read(), exitCode, handlerError };
142
+ }
143
+ function emitCallEvent(reg, invoke, outcome, startedAt, startedAtMs) {
144
+ const call = {
145
+ cli: invoke.cli,
146
+ argv: invoke.argv,
147
+ cwd: invoke.cwd,
148
+ stdin: invoke.stdin,
149
+ stdout: outcome.stdout,
150
+ stderr: outcome.stderr,
151
+ exitCode: outcome.exitCode,
152
+ startedAt,
153
+ durationMs: Date.now() - startedAtMs,
154
+ ...(outcome.handlerError ? { handlerError: outcome.handlerError } : {}),
155
+ };
156
+ reg.emitCliMock(call);
157
+ }
158
+ function respondMissingScenario(res) {
159
+ const stderr = "mock-cli: no scenario currently owns this gateway\n";
160
+ respondJson(res, { stdout: "", stderr, exitCode: 1, missing: true });
161
+ }
162
+ function respondJson(res, body) {
163
+ res.setHeader("content-type", "application/json");
164
+ res.end(JSON.stringify(body));
165
+ }
166
+ function readJsonBody(req) {
167
+ return new Promise((resolve, reject) => {
168
+ const chunks = [];
169
+ req.on("data", (c) => chunks.push(c));
170
+ req.on("end", () => {
171
+ const raw = Buffer.concat(chunks).toString("utf8");
172
+ try {
173
+ resolve(raw ? JSON.parse(raw) : {});
174
+ }
175
+ catch (err) {
176
+ reject(err);
177
+ }
178
+ });
179
+ req.on("error", reject);
180
+ });
181
+ }
182
+ function collectStream() {
183
+ const chunks = [];
184
+ const writable = new Writable({
185
+ write(chunk, _enc, cb) {
186
+ chunks.push(typeof chunk === "string" ? Buffer.from(chunk) : chunk);
187
+ cb();
188
+ },
189
+ });
190
+ return {
191
+ writable,
192
+ read: () => Buffer.concat(chunks).toString("utf8"),
193
+ };
194
+ }
@@ -0,0 +1,7 @@
1
+ /**
2
+ * Mock-CLI shim. Consumers symlink it under `/opt/openclaw-test/mocks/bin/` for
3
+ * whichever commands they want to intercept (typically `claude`, `gh`, `glab`,
4
+ * etc.). Determines its invoked name from argv[1] basename, POSTs the call to
5
+ * the runner's /mock-cli/invoke endpoint, and replays the response locally.
6
+ */
7
+ export {};
@@ -0,0 +1,89 @@
1
+ /**
2
+ * Mock-CLI shim. Consumers symlink it under `/opt/openclaw-test/mocks/bin/` for
3
+ * whichever commands they want to intercept (typically `claude`, `gh`, `glab`,
4
+ * etc.). Determines its invoked name from argv[1] basename, POSTs the call to
5
+ * the runner's /mock-cli/invoke endpoint, and replays the response locally.
6
+ */
7
+ import { request as httpRequest } from "node:http";
8
+ import { basename } from "node:path";
9
+ import { URL } from "node:url";
10
+ function die(msg, code = 127) {
11
+ process.stderr.write(`mock-cli shim: ${msg}\n`);
12
+ process.exit(code);
13
+ }
14
+ async function readStdin() {
15
+ if (process.stdin.isTTY)
16
+ return "";
17
+ const chunks = [];
18
+ for await (const chunk of process.stdin) {
19
+ chunks.push(typeof chunk === "string" ? Buffer.from(chunk) : chunk);
20
+ }
21
+ return Buffer.concat(chunks).toString("utf8");
22
+ }
23
+ function post(urlStr, body) {
24
+ const url = new URL(urlStr);
25
+ return new Promise((resolve, reject) => {
26
+ const req = httpRequest({
27
+ method: "POST",
28
+ hostname: url.hostname,
29
+ port: url.port || 80,
30
+ path: url.pathname + url.search,
31
+ headers: {
32
+ "content-type": "application/json",
33
+ "content-length": Buffer.byteLength(body),
34
+ },
35
+ }, (res) => {
36
+ const chunks = [];
37
+ res.on("data", (c) => chunks.push(c));
38
+ res.on("end", () => {
39
+ const raw = Buffer.concat(chunks).toString("utf8");
40
+ if (res.statusCode !== 200) {
41
+ reject(new Error(`runner returned HTTP ${res.statusCode}: ${raw}`));
42
+ return;
43
+ }
44
+ try {
45
+ resolve(JSON.parse(raw));
46
+ }
47
+ catch (err) {
48
+ reject(new Error(`invalid JSON response: ${err.message}`));
49
+ }
50
+ });
51
+ res.on("error", reject);
52
+ });
53
+ req.on("error", reject);
54
+ req.write(body);
55
+ req.end();
56
+ });
57
+ }
58
+ async function main() {
59
+ // The sh wrapper at /opt/openclaw-test/mocks/bin/mock-cli-shim invokes us as
60
+ // exec node mock-cli-shim.js "$0" "$@"
61
+ // so argv[2] is the symlink path used to call us (e.g. /opt/openclaw-test/mocks/bin/git)
62
+ // and argv.slice(3) is the original argv tail.
63
+ const invokedAs = basename(process.argv[2] ?? "");
64
+ if (!invokedAs)
65
+ die("could not determine invoked binary name from argv[2]");
66
+ const runnerUrl = process.env.OPENCLAW_TEST_RUNNER_URL;
67
+ if (!runnerUrl)
68
+ die("OPENCLAW_TEST_RUNNER_URL is not set");
69
+ const stdin = await readStdin();
70
+ const body = JSON.stringify({
71
+ cli: invokedAs,
72
+ argv: process.argv.slice(3),
73
+ cwd: process.cwd(),
74
+ stdin,
75
+ });
76
+ let res;
77
+ try {
78
+ res = await post(`${runnerUrl}/mock-cli/invoke`, body);
79
+ }
80
+ catch (err) {
81
+ die(`POST ${runnerUrl}/mock-cli/invoke failed: ${err.message}`);
82
+ }
83
+ if (res.stdout)
84
+ process.stdout.write(res.stdout);
85
+ if (res.stderr)
86
+ process.stderr.write(res.stderr);
87
+ process.exit(typeof res.exitCode === "number" ? res.exitCode : 1);
88
+ }
89
+ main().catch((err) => die(err.message ?? String(err)));
@@ -0,0 +1,23 @@
1
+ export interface SelectedModel {
2
+ /** Bare id: the suffix after the last `/` of the full ref. */
3
+ id: string;
4
+ /** Full LiteLLM `provider/model` ref, as written in `OPENCLAW_TEST_MODELS`. */
5
+ ref: string;
6
+ }
7
+ /**
8
+ * Resolve the `--model` value against the configured model catalog.
9
+ *
10
+ * `OPENCLAW_TEST_MODELS` is a comma list of full `provider/model` refs — the only
11
+ * place the `provider/` prefix appears. The CLI value, `OPENCLAW_DEFAULT_TEST_MODEL`,
12
+ * and the recorded `model` are all bare ids (the suffix after the last `/`). A bare
13
+ * id resolves to its ref by suffix-match; zero or many matches is a hard error.
14
+ *
15
+ * The selection is `all` (whole catalog, sorted by id), a single bare id, or a comma
16
+ * list of bare ids (deduped, CLI order preserved); `undefined` falls back to
17
+ * `OPENCLAW_DEFAULT_TEST_MODEL`.
18
+ */
19
+ export declare function resolveSelectedModels(params: {
20
+ selection: string | undefined;
21
+ modelsEnv: string | undefined;
22
+ defaultEnv: string | undefined;
23
+ }): SelectedModel[];
package/dist/models.js ADDED
@@ -0,0 +1,78 @@
1
+ /**
2
+ * Resolve the `--model` value against the configured model catalog.
3
+ *
4
+ * `OPENCLAW_TEST_MODELS` is a comma list of full `provider/model` refs — the only
5
+ * place the `provider/` prefix appears. The CLI value, `OPENCLAW_DEFAULT_TEST_MODEL`,
6
+ * and the recorded `model` are all bare ids (the suffix after the last `/`). A bare
7
+ * id resolves to its ref by suffix-match; zero or many matches is a hard error.
8
+ *
9
+ * The selection is `all` (whole catalog, sorted by id), a single bare id, or a comma
10
+ * list of bare ids (deduped, CLI order preserved); `undefined` falls back to
11
+ * `OPENCLAW_DEFAULT_TEST_MODEL`.
12
+ */
13
+ export function resolveSelectedModels(params) {
14
+ const catalog = parseCatalog(params.modelsEnv);
15
+ if (params.selection === "all")
16
+ return [...catalog].sort((a, b) => a.id.localeCompare(b.id));
17
+ if (params.selection !== undefined)
18
+ return resolveIdList(catalog, params.selection);
19
+ if (!params.defaultEnv) {
20
+ throw new Error("run: no --model given and OPENCLAW_DEFAULT_TEST_MODEL is unset; pass --model <id|id,id,…|all> or set the default");
21
+ }
22
+ return [matchById(catalog, params.defaultEnv)];
23
+ }
24
+ function resolveIdList(catalog, selection) {
25
+ const ids = selection
26
+ .split(",")
27
+ .map((s) => s.trim())
28
+ .filter((s) => s.length > 0);
29
+ if (ids.length === 0) {
30
+ throw new Error(`run: --model expects a non-empty id list, got ${JSON.stringify(selection)}`);
31
+ }
32
+ const seen = new Set();
33
+ const selected = [];
34
+ for (const id of ids) {
35
+ if (seen.has(id))
36
+ continue;
37
+ seen.add(id);
38
+ selected.push(matchById(catalog, id));
39
+ }
40
+ return selected;
41
+ }
42
+ function parseCatalog(modelsEnv) {
43
+ const refs = (modelsEnv ?? "")
44
+ .split(",")
45
+ .map((s) => s.trim())
46
+ .filter((s) => s.length > 0);
47
+ if (refs.length === 0) {
48
+ throw new Error("run: OPENCLAW_TEST_MODELS is empty; set a comma list of provider/model refs");
49
+ }
50
+ const catalog = refs.map((ref) => ({ id: bareId(ref), ref }));
51
+ assertUniqueIds(catalog);
52
+ return catalog;
53
+ }
54
+ function bareId(ref) {
55
+ const id = ref.split("/").pop();
56
+ if (!id)
57
+ throw new Error(`run: invalid model ref ${JSON.stringify(ref)} in OPENCLAW_TEST_MODELS`);
58
+ return id;
59
+ }
60
+ function assertUniqueIds(catalog) {
61
+ const seen = new Set();
62
+ for (const { id } of catalog) {
63
+ if (seen.has(id)) {
64
+ throw new Error(`run: OPENCLAW_TEST_MODELS has two entries with bare id ${JSON.stringify(id)}`);
65
+ }
66
+ seen.add(id);
67
+ }
68
+ }
69
+ function matchById(catalog, id) {
70
+ const matches = catalog.filter((m) => m.id === id);
71
+ if (matches.length === 1)
72
+ return matches[0];
73
+ const known = catalog.map((m) => m.id).join(", ");
74
+ if (matches.length === 0) {
75
+ throw new Error(`run: model ${JSON.stringify(id)} not found in OPENCLAW_TEST_MODELS — known: ${known}`);
76
+ }
77
+ throw new Error(`run: model ${JSON.stringify(id)} is ambiguous in OPENCLAW_TEST_MODELS — known: ${known}`);
78
+ }
@@ -0,0 +1 @@
1
+ export declare function extractTaggedBlock(raw: string, tagName: string): string;
@@ -0,0 +1,16 @@
1
+ export function extractTaggedBlock(raw, tagName) {
2
+ const escaped = escapeRegex(tagName);
3
+ const open = new RegExp(`<${escaped}>`).exec(raw);
4
+ if (!open) {
5
+ throw new Error(`extractTaggedBlock: opening <${tagName}> not found. raw=${JSON.stringify(raw.slice(0, 200))}`);
6
+ }
7
+ const re = new RegExp(`<${escaped}>([\\s\\S]*?)</${escaped}>`);
8
+ const match = re.exec(raw);
9
+ if (!match) {
10
+ throw new Error(`extractTaggedBlock: closing </${tagName}> not found after opening tag. raw=${JSON.stringify(raw.slice(0, 200))}`);
11
+ }
12
+ return match[1].trim();
13
+ }
14
+ function escapeRegex(value) {
15
+ return value.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
16
+ }
@@ -0,0 +1,215 @@
1
+ /**
2
+ * Scenario report typing for openclaw-test.
3
+ *
4
+ * Two artifacts per scenario run, under `<OPENCLAW_TEST_ARTIFACTS_DIR>/<baseStamp>/<scenario>-<channel>[-iter<n>]/`:
5
+ *
6
+ * - `scenario-log.jsonl` — live append, one `ReportEntry` per line, written as
7
+ * things happen. Survives runner crash / hang.
8
+ * - `report.json` — final `ScenarioReport`, written once at end.
9
+ *
10
+ * `report.json` carries the same entries as `scenario-log.jsonl`, plus terminal
11
+ * data only computable at the end (status, durationMs, finishedAt, parsed
12
+ * agentToolCalls, cost).
13
+ *
14
+ * Each agent or scenario action is one entry. Assertions, scenario-log notes,
15
+ * and failures about that action live as nested fields on the entry, not as
16
+ * separate entries. Free-standing `scenarioLog` entries are the fallback for
17
+ * `ctx.log(...)` calls that are not bound to an action.
18
+ */
19
+ import type { Readable, Writable } from "node:stream";
20
+ export interface ScenarioReport {
21
+ schemaVersion: 4;
22
+ scenario: string;
23
+ channel: ChannelId;
24
+ model: string;
25
+ conversationId: string;
26
+ accountId: string;
27
+ startedAt: string;
28
+ finishedAt: string;
29
+ durationMs: number;
30
+ cost: CostBreakdown;
31
+ result: ScenarioResult;
32
+ entries: ReportEntry[];
33
+ }
34
+ export type ScenarioResult = PassScenarioResult | FailScenarioResult;
35
+ export type FailScenarioResult = FailedEntryScenarioResult | ErrorScenarioResult;
36
+ export interface PassScenarioResult {
37
+ verdict: "pass";
38
+ }
39
+ export interface FailedEntryScenarioResult {
40
+ verdict: "fail";
41
+ cause: "failedEntry";
42
+ /** Stable entry identifier shared by `scenario-log.jsonl` and `report.json`. */
43
+ entrySeq: number;
44
+ message: string;
45
+ }
46
+ export interface ErrorScenarioResult {
47
+ verdict: "fail";
48
+ cause: "error";
49
+ source: "assertion" | "judge" | "cliMock" | "timeout" | "scenarioThrow" | "runner";
50
+ errorName: string;
51
+ message: string;
52
+ stack?: string;
53
+ }
54
+ export type ChannelId = string;
55
+ export type ReportEntry = ScenarioLogEntry | ActionEntry;
56
+ export type ActionEntry = InboundSentEntry | OutboundReceivedEntry | CliMockEntry | AgentToolCallEntry;
57
+ export interface ReportEntryBase {
58
+ ts: string;
59
+ /**
60
+ * Stable entry id shared across `scenario-log.jsonl` and `report.json`. In the
61
+ * jsonl it matches the line's emission order; in `report.json` entries are
62
+ * sorted by `ts` so `entrySeq` is not array position — it stays a cross-file
63
+ * reference.
64
+ */
65
+ entrySeq: number;
66
+ }
67
+ export interface ActionEntryBase extends ReportEntryBase {
68
+ scenarioLog?: ScenarioLogNote;
69
+ assertions?: AssertionRecord[];
70
+ failure?: ScenarioFailure;
71
+ }
72
+ export interface ScenarioLogNote {
73
+ label?: string;
74
+ ts: string;
75
+ extra?: unknown;
76
+ }
77
+ export interface ScenarioLogEntry extends ReportEntryBase {
78
+ kind: "scenarioLog";
79
+ message: string;
80
+ }
81
+ export interface InboundSentEntry extends ActionEntryBase {
82
+ kind: "inboundSent";
83
+ messageId: string;
84
+ text: string;
85
+ senderId: string;
86
+ senderName?: string;
87
+ threadId?: string;
88
+ }
89
+ export interface OutboundReceivedEntry extends ActionEntryBase {
90
+ kind: "outboundReceived";
91
+ messageId: string;
92
+ text: string;
93
+ threadId?: string;
94
+ }
95
+ export interface CliMockEntry extends ActionEntryBase {
96
+ kind: "cliMock";
97
+ call: CliMockCall;
98
+ }
99
+ export interface AgentToolCallEntry extends ActionEntryBase {
100
+ kind: "agentToolCall";
101
+ call: AgentToolCall;
102
+ }
103
+ export type AssertionRecord = {
104
+ label: string;
105
+ ok: true;
106
+ extra?: unknown;
107
+ } | {
108
+ label: string;
109
+ ok: false;
110
+ detail: string;
111
+ extra?: unknown;
112
+ };
113
+ export interface CliMockCall {
114
+ cli: "git" | "npm" | "pnpm" | "yarn" | "claude" | (string & {});
115
+ argv: string[];
116
+ cwd: string;
117
+ stdin: string;
118
+ stdout: string;
119
+ stderr: string;
120
+ exitCode: number;
121
+ startedAt: string;
122
+ durationMs: number;
123
+ handlerError?: {
124
+ name: string;
125
+ message: string;
126
+ stack?: string;
127
+ };
128
+ }
129
+ export interface AgentToolCall {
130
+ toolName: string;
131
+ toolUseId: string;
132
+ /**
133
+ * `sessionKey` of the session transcript the call was collected from — the
134
+ * OpenClaw session that made it (e.g. channel vs per-thread session).
135
+ */
136
+ sessionKey?: string;
137
+ input: unknown;
138
+ result?: {
139
+ isError: boolean;
140
+ /**
141
+ * Full tool result. Always present in `scenario-log.jsonl`. In `report.json`,
142
+ * replaced by `truncatedContent` when truncatable: a string longer than 60
143
+ * chars, or a `read` call's text content blocks (the file is identified by
144
+ * `input` and kept in full in the jsonl).
145
+ */
146
+ content?: unknown;
147
+ /** Only in `report.json`: rtrimmed first 60 chars + `…` of the truncatable text. */
148
+ truncatedContent?: string;
149
+ };
150
+ /**
151
+ * Timestamp of the transcript's assistant message that issued the call —
152
+ * per-message, so ordering `entries` on it is meaningful. Absent when the
153
+ * message carried no timestamp; such calls sort to the end of the timeline.
154
+ */
155
+ startedAt?: string;
156
+ /**
157
+ * Best-effort estimate of when the tool call actually executed, inferred by
158
+ * matching the call's leading CLI against an in-order `cliMock` entry from
159
+ * the same scenario. `startedAt` stamps the issuing assistant message, not
160
+ * the execution. Not used for `entries` ordering.
161
+ */
162
+ inferredStartedAt?: string;
163
+ turn?: number;
164
+ }
165
+ export interface ScenarioFailure {
166
+ name: string;
167
+ message: string;
168
+ stack?: string;
169
+ source: "assertion" | "judge" | "cliMock" | "timeout" | "scenarioThrow" | "runner";
170
+ }
171
+ export interface CostBreakdown {
172
+ agentUsd: number;
173
+ judgeUsd: number;
174
+ totalUsd: number;
175
+ agentTurns: number;
176
+ }
177
+ export interface CliMockHandlerArgs {
178
+ argv: string[];
179
+ cwd: string;
180
+ stdin: Readable;
181
+ stdout: Writable;
182
+ stderr: Writable;
183
+ }
184
+ export type CliMockHandler = (args: CliMockHandlerArgs) => number | undefined | Promise<number | undefined>;
185
+ /**
186
+ * Patch describing the augmentation of an already-emitted entry. Carries only
187
+ * the new nested field so the live `scenario-log.jsonl` need not re-serialize
188
+ * the entry's full text/metadata on every assertion or annotation.
189
+ */
190
+ export type AugmentPatch = {
191
+ kind: "scenarioLog";
192
+ scenarioLog: ScenarioLogNote;
193
+ } | {
194
+ kind: "assertion";
195
+ assertion: AssertionRecord;
196
+ } | {
197
+ kind: "failure";
198
+ failure: ScenarioFailure;
199
+ };
200
+ export type SinkEvent = {
201
+ type: "entry";
202
+ entry: ReportEntry;
203
+ } | {
204
+ type: "augment";
205
+ entrySeq: number;
206
+ patch: AugmentPatch;
207
+ };
208
+ /**
209
+ * Callback the runner passes into `createContext` so live records can be
210
+ * appended to `scenario-log.jsonl` as they happen. First emission of an entry
211
+ * is `type: "entry"`. Subsequent nested-field additions (assertions, failure,
212
+ * scenarioLog) are `type: "augment"` with only the patch — readers reconstruct
213
+ * full state by folding augments onto entries by `seq`.
214
+ */
215
+ export type EmitSink = (event: SinkEvent) => void;
package/dist/report.js ADDED
@@ -0,0 +1,19 @@
1
+ /**
2
+ * Scenario report typing for openclaw-test.
3
+ *
4
+ * Two artifacts per scenario run, under `<OPENCLAW_TEST_ARTIFACTS_DIR>/<baseStamp>/<scenario>-<channel>[-iter<n>]/`:
5
+ *
6
+ * - `scenario-log.jsonl` — live append, one `ReportEntry` per line, written as
7
+ * things happen. Survives runner crash / hang.
8
+ * - `report.json` — final `ScenarioReport`, written once at end.
9
+ *
10
+ * `report.json` carries the same entries as `scenario-log.jsonl`, plus terminal
11
+ * data only computable at the end (status, durationMs, finishedAt, parsed
12
+ * agentToolCalls, cost).
13
+ *
14
+ * Each agent or scenario action is one entry. Assertions, scenario-log notes,
15
+ * and failures about that action live as nested fields on the entry, not as
16
+ * separate entries. Free-standing `scenarioLog` entries are the fallback for
17
+ * `ctx.log(...)` calls that are not bound to an action.
18
+ */
19
+ export {};