@alignfirst/openclaw-test 0.20.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (53) hide show
  1. package/Dockerfile.base +45 -0
  2. package/LICENSE +21 -0
  3. package/README.md +132 -0
  4. package/bin/cli.mjs +7 -0
  5. package/dist/bus.d.ts +1 -0
  6. package/dist/bus.js +20 -0
  7. package/dist/cell-result.d.ts +25 -0
  8. package/dist/cell-result.js +32 -0
  9. package/dist/cli.d.ts +5 -0
  10. package/dist/cli.js +110 -0
  11. package/dist/context.d.ts +220 -0
  12. package/dist/context.js +479 -0
  13. package/dist/cost.d.ts +2 -0
  14. package/dist/cost.js +19 -0
  15. package/dist/env-cli.d.ts +5 -0
  16. package/dist/env-cli.js +645 -0
  17. package/dist/exec-rpc.d.ts +13 -0
  18. package/dist/exec-rpc.js +46 -0
  19. package/dist/index.d.ts +3 -0
  20. package/dist/index.js +1 -0
  21. package/dist/judge.d.ts +47 -0
  22. package/dist/judge.js +183 -0
  23. package/dist/loop.d.ts +62 -0
  24. package/dist/loop.js +316 -0
  25. package/dist/mock-cli-server.d.ts +40 -0
  26. package/dist/mock-cli-server.js +194 -0
  27. package/dist/mock-cli-shim.d.ts +7 -0
  28. package/dist/mock-cli-shim.js +89 -0
  29. package/dist/models.d.ts +23 -0
  30. package/dist/models.js +78 -0
  31. package/dist/parse-tagged-json.d.ts +1 -0
  32. package/dist/parse-tagged-json.js +16 -0
  33. package/dist/report.d.ts +215 -0
  34. package/dist/report.js +19 -0
  35. package/dist/runner-args.d.ts +11 -0
  36. package/dist/runner-args.js +84 -0
  37. package/dist/runner.d.ts +2 -0
  38. package/dist/runner.js +371 -0
  39. package/dist/summary.d.ts +3 -0
  40. package/dist/summary.js +64 -0
  41. package/dist/transcript-dump.d.ts +1 -0
  42. package/dist/transcript-dump.js +50 -0
  43. package/dist/transcript-log.d.ts +67 -0
  44. package/dist/transcript-log.js +199 -0
  45. package/dist/transcript-store.d.ts +6 -0
  46. package/dist/transcript-store.js +69 -0
  47. package/docker-compose.yml +83 -0
  48. package/exec-watcher.mjs +155 -0
  49. package/package.json +60 -0
  50. package/templates/.env.local.example +28 -0
  51. package/templates/Dockerfile +47 -0
  52. package/templates/docker-compose.yml +10 -0
  53. package/templates/openclaw.json +40 -0
@@ -0,0 +1,199 @@
1
+ import { randomUUID } from "node:crypto";
2
+ import { readFileSync, rmSync } from "node:fs";
3
+ import { getQaBusState } from "@alignfirst/openclaw-channel-mock-core";
4
+ import { execInGateway, IPC_DIR } from "./exec-rpc.js";
5
+ // OpenClaw persists each session's transcript as SQLite rows in the gateway's
6
+ // per-agent store — the full-fidelity record the gateway itself replays,
7
+ // appended per message. That store lives outside the shared mounts, so the
8
+ // runner extracts a conversation's session transcripts through the
9
+ // exec-watcher RPC: the dump script (shipped in this package's dist, mounted
10
+ // into the gateway) writes them as JSON into the shared IPC volume.
11
+ const DUMP_SCRIPT = "/opt/openclaw-test/src/dist/transcript-dump.js";
12
+ const BUS_URL = process.env.OPENCLAW_TEST_BUS_URL ?? "http://bus:43123";
13
+ /** The transcripts of the conversation and its bus-owned thread sessions. */
14
+ export async function fetchTranscriptSnapshot(opts) {
15
+ const outPath = `${IPC_DIR}/${randomUUID()}.transcript.json`;
16
+ try {
17
+ const { threads } = await getQaBusState(BUS_URL);
18
+ const conversationId = opts.conversationId.toLowerCase();
19
+ const threadIds = threads
20
+ .filter((thread) => thread.conversationId.toLowerCase() === conversationId)
21
+ .map((thread) => thread.id);
22
+ const result = await execInGateway(["node", DUMP_SCRIPT, opts.startedAtIso, opts.conversationId, outPath, ...threadIds],
23
+ // A whole conversation can take a while to serialize; the default 30s
24
+ // exec timeout is too tight for long cells.
25
+ { timeoutMs: 60_000 });
26
+ if (result.exitCode !== 0) {
27
+ const error = `transcript dump failed (exit ${result.exitCode}): ${result.stderr}`;
28
+ console.warn(`openclaw-test: ${error}`);
29
+ return { databases: 0, sessions: [], error };
30
+ }
31
+ const raw = readFileSync(outPath, "utf8");
32
+ return JSON.parse(raw);
33
+ }
34
+ catch (err) {
35
+ const error = `transcript dump failed: ${String(err)}`;
36
+ console.warn(`openclaw-test: ${error}`);
37
+ return { databases: 0, sessions: [], error };
38
+ }
39
+ finally {
40
+ rmSync(outPath, { force: true });
41
+ }
42
+ }
43
+ /**
44
+ * Polls the transcripts until they are *quiescent* for this conversation — no
45
+ * new message for `settleMs` AND no session's turn is demonstrably open — or
46
+ * the budget expires. The transcript is appended per message, so a settle
47
+ * window alone would return between two tool calls of one turn (a single
48
+ * model round-trip routinely exceeds it); the open-turn gate holds until the
49
+ * turn-final assistant message — the one carrying `usage.cost.total` — has
50
+ * landed. A conversation spans multiple OpenClaw sessions (e.g. Discord's
51
+ * channel session plus a per-thread session).
52
+ */
53
+ export async function waitForTranscriptQuiescence(opts) {
54
+ const maxWaitMs = opts.maxWaitMs ?? 60_000;
55
+ const pollMs = opts.pollMs ?? 1_000;
56
+ const settleMs = opts.settleMs ?? 4_000;
57
+ const deadline = Date.now() + maxWaitMs;
58
+ let lastCount = 0;
59
+ let lastChangeAt = Date.now();
60
+ let seenAny = false;
61
+ while (Date.now() < deadline) {
62
+ const { sessions } = await fetchTranscriptSnapshot(opts);
63
+ const count = sessions.reduce((sum, s) => sum + s.messages.length, 0);
64
+ if (count > 0)
65
+ seenAny = true;
66
+ if (count !== lastCount) {
67
+ lastCount = count;
68
+ lastChangeAt = Date.now();
69
+ }
70
+ else if (seenAny && !hasOpenTurn(sessions) && Date.now() - lastChangeAt >= settleMs) {
71
+ return;
72
+ }
73
+ await new Promise((r) => setTimeout(r, pollMs));
74
+ }
75
+ }
76
+ /**
77
+ * A turn is open when a session's last message is a tool result or an
78
+ * assistant stop for tool use — more of the turn is coming. A trailing user
79
+ * message does not count as open: some sessions of a conversation never get a
80
+ * reply (the agent answers in a sibling session), and treating them as open
81
+ * would burn the whole budget on every cell.
82
+ */
83
+ export function hasOpenTurn(sessions) {
84
+ return sessions.some((session) => {
85
+ const last = session.messages.at(-1);
86
+ if (!isRecord(last))
87
+ return false;
88
+ if (last.role === "toolResult")
89
+ return true;
90
+ return last.role === "assistant" && last.stopReason === "toolUse";
91
+ });
92
+ }
93
+ /**
94
+ * Cost lives per assistant message, as `usage.cost.total`. Sum across every
95
+ * session of the conversation; `turns` counts the cost-bearing messages.
96
+ */
97
+ export function readTranscriptCost(snapshot) {
98
+ let cost = 0;
99
+ let turns = 0;
100
+ for (const session of snapshot.sessions) {
101
+ for (const msg of session.messages) {
102
+ if (!isRecord(msg) || msg.role !== "assistant")
103
+ continue;
104
+ const total = assistantCostTotal(msg);
105
+ if (typeof total !== "number")
106
+ continue;
107
+ cost += total;
108
+ turns += 1;
109
+ }
110
+ }
111
+ return { cost, turns };
112
+ }
113
+ function assistantCostTotal(msg) {
114
+ const usage = msg.usage;
115
+ if (!isRecord(usage))
116
+ return;
117
+ const cost = usage.cost;
118
+ if (!isRecord(cost))
119
+ return;
120
+ return typeof cost.total === "number" ? cost.total : undefined;
121
+ }
122
+ /** One-shot fetch + aggregation of the conversation's agent tool calls. */
123
+ export async function parseAgentToolCalls(opts) {
124
+ const { sessions } = await fetchTranscriptSnapshot(opts);
125
+ return aggregateAgentToolCalls(sessions);
126
+ }
127
+ /**
128
+ * Walk each session's messages collecting assistant `toolCall` blocks matched
129
+ * with `toolResult` messages, then union across sessions deduped by
130
+ * `toolUseId`.
131
+ */
132
+ export function aggregateAgentToolCalls(sessions) {
133
+ const calls = [];
134
+ const seen = new Set();
135
+ for (const session of sessions) {
136
+ const results = collectToolResults(session.messages);
137
+ for (const call of collectToolUses(session.messages, results, session.sessionKey)) {
138
+ if (seen.has(call.toolUseId))
139
+ continue;
140
+ seen.add(call.toolUseId);
141
+ calls.push(call);
142
+ }
143
+ }
144
+ return calls;
145
+ }
146
+ function collectToolResults(messages) {
147
+ const results = new Map();
148
+ for (const msg of messages) {
149
+ if (!isRecord(msg) || msg.role !== "toolResult")
150
+ continue;
151
+ const toolCallId = msg.toolCallId;
152
+ if (typeof toolCallId !== "string")
153
+ continue;
154
+ results.set(toolCallId, { isError: msg.isError === true, content: msg.content ?? null });
155
+ }
156
+ return results;
157
+ }
158
+ function collectToolUses(messages, results, sessionKey) {
159
+ const calls = [];
160
+ let turn = 0;
161
+ for (const msg of messages) {
162
+ if (!isRecord(msg) || msg.role !== "assistant" || !Array.isArray(msg.content))
163
+ continue;
164
+ turn += 1;
165
+ for (const block of msg.content) {
166
+ if (!isRecord(block) || block.type !== "toolCall")
167
+ continue;
168
+ const toolUseId = typeof block.id === "string" ? block.id : "";
169
+ const toolName = typeof block.name === "string" ? block.name : "";
170
+ if (!toolUseId || !toolName)
171
+ continue;
172
+ const startedAt = messageTimestampIso(msg);
173
+ const call = {
174
+ toolName,
175
+ toolUseId,
176
+ ...(sessionKey !== undefined ? { sessionKey } : {}),
177
+ input: block.arguments ?? null,
178
+ ...(startedAt !== undefined ? { startedAt } : {}),
179
+ turn,
180
+ };
181
+ const result = results.get(toolUseId);
182
+ if (result)
183
+ call.result = result;
184
+ calls.push(call);
185
+ }
186
+ }
187
+ return calls;
188
+ }
189
+ function messageTimestampIso(msg) {
190
+ const ts = msg.timestamp;
191
+ if (typeof ts === "number")
192
+ return new Date(ts).toISOString();
193
+ if (typeof ts === "string")
194
+ return ts;
195
+ return undefined;
196
+ }
197
+ function isRecord(value) {
198
+ return typeof value === "object" && value !== null;
199
+ }
@@ -0,0 +1,6 @@
1
+ export interface DumpedSession {
2
+ sessionKey: string;
3
+ sessionId: string;
4
+ messages: unknown[];
5
+ }
6
+ export declare function readConversationSessions(dbPath: string, sinceMs: number, conversationId: string, threadIds: string[]): DumpedSession[];
@@ -0,0 +1,69 @@
1
+ import { DatabaseSync } from "node:sqlite";
2
+ export function readConversationSessions(dbPath, sinceMs, conversationId, threadIds) {
3
+ let db;
4
+ try {
5
+ db = new DatabaseSync(dbPath, { readOnly: true });
6
+ }
7
+ catch {
8
+ return [];
9
+ }
10
+ try {
11
+ db.exec("PRAGMA busy_timeout = 2000;");
12
+ const wanted = conversationId.toLowerCase();
13
+ const nodes = db.prepare("SELECT session_key, current_session_id FROM session_nodes").all();
14
+ const sessions = [];
15
+ for (const node of nodes) {
16
+ const segments = node.session_key.toLowerCase().split(":");
17
+ if (!segments.includes(wanted) && !threadIds.some((id) => segments.includes(id)))
18
+ continue;
19
+ sessions.push({
20
+ sessionKey: node.session_key,
21
+ sessionId: node.current_session_id,
22
+ messages: readNodeMessages(db, node, sinceMs),
23
+ });
24
+ }
25
+ return sessions;
26
+ }
27
+ catch {
28
+ // Fresh store without the tables yet, or a transient lock: the runner's
29
+ // next poll retries.
30
+ return [];
31
+ }
32
+ finally {
33
+ db.close();
34
+ }
35
+ }
36
+ /**
37
+ * A session key can span several session windows: compaction, reset, or
38
+ * recovery mints a successor id that `current_session_id` then advances to,
39
+ * and `transcript_events` rows stay keyed to the window that wrote them.
40
+ * Union every window of the key (in creation order) so earlier tool calls and
41
+ * costs survive a mid-run rollover.
42
+ */
43
+ function readNodeMessages(db, node, sinceMs) {
44
+ const windows = db
45
+ .prepare("SELECT session_id FROM session_windows WHERE session_key = ? ORDER BY created_at")
46
+ .all(node.session_key);
47
+ const sessionIds = windows.map((w) => w.session_id);
48
+ if (!sessionIds.includes(node.current_session_id))
49
+ sessionIds.push(node.current_session_id);
50
+ return sessionIds.flatMap((id) => readSessionMessages(db, id, sinceMs));
51
+ }
52
+ function readSessionMessages(db, sessionId, sinceMs) {
53
+ const rows = db
54
+ .prepare("SELECT event_json FROM transcript_events WHERE session_id = ? AND created_at >= ? ORDER BY seq")
55
+ .all(sessionId, sinceMs);
56
+ const messages = [];
57
+ for (const row of rows) {
58
+ let event;
59
+ try {
60
+ event = JSON.parse(row.event_json);
61
+ }
62
+ catch {
63
+ continue;
64
+ }
65
+ if (event.type === "message" && event.message !== undefined)
66
+ messages.push(event.message);
67
+ }
68
+ return messages;
69
+ }
@@ -0,0 +1,83 @@
1
+ x-common-build: &common-build
2
+ context: ${OPENCLAW_TEST_PROJECT_DIR}
3
+ dockerfile: Dockerfile
4
+ args:
5
+ OPENCLAW_TEST_BASE_TAG: ${OPENCLAW_TEST_BASE_TAG}
6
+
7
+ # Single source of truth for service mounts. `openclaw-test-ipc` is only used by gateway
8
+ # (writer) and runner (reader), but mounting it on `bus` too is harmless and
9
+ # lets all three services share one anchor.
10
+ x-common-volumes: &common-volumes
11
+ - ${OPENCLAW_WORKSPACE_DIR}:/home/assistant/.openclaw/workspace
12
+ - ${OPENCLAW_CONFIG_PATH}:/home/assistant/.openclaw/openclaw.json
13
+ - ${OPENCLAW_TEST_PACKAGE_DIR}/dist/:/opt/openclaw-test/src/dist
14
+ - ${OPENCLAW_TEST_SCENARIOS_DIR}:/opt/openclaw-test/src/scenarios
15
+ - ${OPENCLAW_TEST_ARTIFACTS_DIR}:/opt/openclaw-test/artifacts
16
+ - ${OPENCLAW_TEST_GATEWAY_LOGS_DIR}:/home/assistant/.openclaw/logs
17
+ - openclaw-test-ipc:/var/run/openclaw-test-ipc
18
+
19
+ volumes:
20
+ openclaw-test-ipc:
21
+
22
+ # All three services run the same image. The explicit shared `image:` keeps worker
23
+ # Compose projects (`<base>-w<i>`) from each building/looking up their own
24
+ # project-named image; the CLI injects OPENCLAW_TEST_CONSUMER_IMAGE.
25
+ services:
26
+ bus:
27
+ build: *common-build
28
+ image: ${OPENCLAW_TEST_CONSUMER_IMAGE}
29
+ user: assistant
30
+ working_dir: /opt/openclaw-test/src
31
+ command: ["node", "/opt/openclaw-test/src/dist/bus.js"]
32
+ volumes: *common-volumes
33
+ healthcheck:
34
+ test: ["CMD", "wget", "-qO-", "http://127.0.0.1:43123/v1/state"]
35
+ interval: 2s
36
+ timeout: 2s
37
+ retries: 10
38
+
39
+ gateway:
40
+ build: *common-build
41
+ image: ${OPENCLAW_TEST_CONSUMER_IMAGE}
42
+ user: assistant
43
+ working_dir: /opt/openclaw-test/src
44
+ environment:
45
+ OPENCLAW_CONFIG_PATH: /home/assistant/.openclaw/openclaw.json
46
+ OPENCLAW_CONFIG_READONLY: "1"
47
+ ANTHROPIC_API_KEY: ${ANTHROPIC_API_KEY}
48
+ OPENROUTER_API_KEY: ${OPENROUTER_API_KEY:-}
49
+ OPENCLAW_RAW_STREAM: ${OPENCLAW_RAW_STREAM:-}
50
+ OPENCLAW_TEST_RUNNER_URL: "http://runner:43124"
51
+ PATH: "/opt/openclaw-test/mocks/bin:/usr/local/sbin:/usr/local/bin:/usr/sbin:/usr/bin:/sbin:/bin"
52
+ # Watcher in the background, then exec openclaw as PID 1's foreground process.
53
+ command:
54
+ ["sh", "-c",
55
+ "/usr/local/bin/exec-watcher & exec npx openclaw gateway run --port 18789 --bind loopback --auth none"]
56
+ volumes: *common-volumes
57
+ depends_on:
58
+ bus:
59
+ condition: service_healthy
60
+ healthcheck:
61
+ test: ["CMD-SHELL", "node -e \"require('net').connect(18789, '127.0.0.1').on('connect', () => process.exit(0)).on('error', () => process.exit(1))\""]
62
+ interval: 5s
63
+ timeout: 5s
64
+ retries: 20
65
+ start_period: 30s
66
+
67
+ runner:
68
+ build: *common-build
69
+ image: ${OPENCLAW_TEST_CONSUMER_IMAGE}
70
+ user: assistant
71
+ working_dir: /opt/openclaw-test/src
72
+ environment:
73
+ OPENCLAW_TEST_BUS_URL: http://bus:43123
74
+ OPENCLAW_CONFIG_PATH: /home/assistant/.openclaw/openclaw.json
75
+ ANTHROPIC_API_KEY: ${ANTHROPIC_API_KEY}
76
+ OPENROUTER_API_KEY: ${OPENROUTER_API_KEY:-}
77
+ entrypoint: ["node", "/opt/openclaw-test/src/dist/runner.js"]
78
+ volumes: *common-volumes
79
+ depends_on:
80
+ bus:
81
+ condition: service_healthy
82
+ gateway:
83
+ condition: service_healthy
@@ -0,0 +1,155 @@
1
+ #!/usr/bin/env node
2
+ import { spawn } from "node:child_process";
3
+ import { mkdirSync, readdirSync, readFileSync, renameSync, rmSync, writeFileSync } from "node:fs";
4
+
5
+ const IPC_DIR = "/var/run/openclaw-test-ipc";
6
+ const MAX_OUTPUT_BYTES = 1_048_576;
7
+ const POLL_INTERVAL_MS = 100;
8
+ const STALE_SUFFIXES = [
9
+ ".req.json",
10
+ ".req.json.processing",
11
+ ".res.json",
12
+ ".res.json.tmp",
13
+ ".transcript.json",
14
+ ".transcript.json.tmp",
15
+ ];
16
+
17
+ mkdirSync(IPC_DIR, { recursive: true });
18
+
19
+ // Sweep stale IPC artifacts left by a killed predecessor in the (named-volume,
20
+ // persistent) IPC dir, so the next runner cycle starts clean.
21
+ try {
22
+ for (const f of readdirSync(IPC_DIR)) {
23
+ if (STALE_SUFFIXES.some((suffix) => f.endsWith(suffix))) {
24
+ try {
25
+ rmSync(`${IPC_DIR}/${f}`, { force: true });
26
+ } catch (err) {
27
+ console.error(`exec-watcher: failed to clean stale IPC file ${f}:`, err);
28
+ }
29
+ }
30
+ }
31
+ } catch (err) {
32
+ console.error(`exec-watcher: failed to scan IPC dir ${IPC_DIR}:`, err);
33
+ }
34
+
35
+ async function main() {
36
+ for (;;) {
37
+ let entries;
38
+ try {
39
+ entries = readdirSync(IPC_DIR);
40
+ } catch {
41
+ entries = [];
42
+ }
43
+ const reqFiles = entries.filter((f) => f.endsWith(".req.json"));
44
+ for (const f of reqFiles) {
45
+ const reqPath = `${IPC_DIR}/${f}`;
46
+ const claimed = `${reqPath}.processing`;
47
+ try {
48
+ renameSync(reqPath, claimed);
49
+ } catch {
50
+ continue;
51
+ }
52
+ processRequest(claimed).catch((err) => {
53
+ console.error("exec-watcher: internal error", err);
54
+ });
55
+ }
56
+ await new Promise((r) => setTimeout(r, POLL_INTERVAL_MS));
57
+ }
58
+ }
59
+
60
+ async function processRequest(claimedPath) {
61
+ let id = "unknown";
62
+ try {
63
+ const raw = readFileSync(claimedPath, "utf8");
64
+ const req = JSON.parse(raw);
65
+ id = req.id;
66
+ const result = await runChild(req);
67
+ writeResult(id, result);
68
+ } catch (err) {
69
+ const message = err instanceof Error ? err.message : String(err);
70
+ writeResult(id, { exitCode: 255, stdout: "", stderr: `watcher error: ${message}` });
71
+ } finally {
72
+ rmSync(claimedPath, { force: true });
73
+ }
74
+ }
75
+
76
+ function runChild(req) {
77
+ const { argv, cwd, env, stdin, timeoutMs = 30_000 } = req;
78
+ return new Promise((resolve) => {
79
+ const child = spawn(argv[0], argv.slice(1), {
80
+ cwd,
81
+ env: { ...process.env, ...(env ?? {}) },
82
+ stdio: ["pipe", "pipe", "pipe"],
83
+ });
84
+
85
+ let stdout = "";
86
+ let stderr = "";
87
+ let stdoutDropped = 0;
88
+ let stderrDropped = 0;
89
+ let killedByTimeout = false;
90
+
91
+ child.stdout.on("data", (chunk) => {
92
+ const s = chunk.toString();
93
+ if (stdout.length + s.length > MAX_OUTPUT_BYTES) {
94
+ const room = Math.max(0, MAX_OUTPUT_BYTES - stdout.length);
95
+ stdout += s.slice(0, room);
96
+ stdoutDropped += s.length - room;
97
+ } else {
98
+ stdout += s;
99
+ }
100
+ });
101
+ child.stderr.on("data", (chunk) => {
102
+ const s = chunk.toString();
103
+ if (stderr.length + s.length > MAX_OUTPUT_BYTES) {
104
+ const room = Math.max(0, MAX_OUTPUT_BYTES - stderr.length);
105
+ stderr += s.slice(0, room);
106
+ stderrDropped += s.length - room;
107
+ } else {
108
+ stderr += s;
109
+ }
110
+ });
111
+
112
+ if (stdin !== undefined) {
113
+ child.stdin.write(stdin);
114
+ }
115
+ child.stdin.end();
116
+
117
+ const timer = setTimeout(() => {
118
+ killedByTimeout = true;
119
+ child.kill("SIGKILL");
120
+ }, timeoutMs);
121
+
122
+ child.on("error", (err) => {
123
+ clearTimeout(timer);
124
+ resolve({
125
+ exitCode: 255,
126
+ stdout,
127
+ stderr: `${stderr}\nspawn error: ${err.message}`,
128
+ });
129
+ });
130
+ child.on("exit", (code, signal) => {
131
+ clearTimeout(timer);
132
+ if (stdoutDropped > 0) stdout += `\n…[truncated ${stdoutDropped} bytes]`;
133
+ if (stderrDropped > 0) stderr += `\n…[truncated ${stderrDropped} bytes]`;
134
+ if (killedByTimeout) {
135
+ resolve({
136
+ exitCode: 124,
137
+ stdout,
138
+ stderr: `${stderr}\n…[killed by watcher timeout after ${timeoutMs}ms]`,
139
+ });
140
+ return;
141
+ }
142
+ const exitCode = code ?? (signal ? 128 : 1);
143
+ resolve({ exitCode, stdout, stderr });
144
+ });
145
+ });
146
+ }
147
+
148
+ function writeResult(id, result) {
149
+ const finalPath = `${IPC_DIR}/${id}.res.json`;
150
+ const tmp = `${finalPath}.tmp`;
151
+ writeFileSync(tmp, JSON.stringify(result));
152
+ renameSync(tmp, finalPath);
153
+ }
154
+
155
+ void main();
package/package.json ADDED
@@ -0,0 +1,60 @@
1
+ {
2
+ "name": "@alignfirst/openclaw-test",
3
+ "version": "0.20.0",
4
+ "description": "Dockerised regression-test framework for OpenClaw workspaces: bus, scenario driver, judge, Compose stack.",
5
+ "keywords": [
6
+ "openclaw",
7
+ "test",
8
+ "testing",
9
+ "harness",
10
+ "scenario",
11
+ "docker"
12
+ ],
13
+ "license": "MIT",
14
+ "author": "Thomas MUR",
15
+ "homepage": "https://alignfirst.paroi.tech/",
16
+ "repository": {
17
+ "type": "git",
18
+ "url": "git+https://github.com/paleo/alignfirst.git",
19
+ "directory": "packages/openclaw-test"
20
+ },
21
+ "engines": {
22
+ "node": ">=24.16.0 <25 || >=26.1.0"
23
+ },
24
+ "packageManager": "npm@11.19.0",
25
+ "type": "module",
26
+ "main": "./dist/index.js",
27
+ "bin": {
28
+ "openclaw-test": "./bin/cli.mjs"
29
+ },
30
+ "exports": "./dist/index.js",
31
+ "files": [
32
+ "dist/",
33
+ "bin/",
34
+ "templates/",
35
+ "docker-compose.yml",
36
+ "Dockerfile.base",
37
+ "exec-watcher.mjs",
38
+ "LICENSE"
39
+ ],
40
+ "publishConfig": {
41
+ "access": "public"
42
+ },
43
+ "peerDependencies": {
44
+ "openclaw": "*"
45
+ },
46
+ "dependencies": {
47
+ "@alignfirst/openclaw-channel-mock-core": "0.9.0",
48
+ "@alignfirst/openclaw-discord-mock": "0.5.0",
49
+ "@alignfirst/openclaw-slack-mock": "0.5.0",
50
+ "@anthropic-ai/sdk": "~0.122.0",
51
+ "openai": "~7.8.0"
52
+ },
53
+ "devDependencies": {
54
+ "@types/node": "~26.5.0",
55
+ "openclaw": "~2026.9.3",
56
+ "rimraf": "~6.1.3",
57
+ "typescript": "~7.0.2",
58
+ "vitest": "~4.1.11"
59
+ }
60
+ }
@@ -0,0 +1,28 @@
1
+ ANTHROPIC_API_KEY=
2
+
3
+ # Required only when running a Qwen model (OpenRouter); harmlessly empty otherwise.
4
+ OPENROUTER_API_KEY=
5
+
6
+ # Agent model catalog. Comma list of full LiteLLM provider/model refs — the only
7
+ # place the provider/ prefix appears. `run --model <id>` and OPENCLAW_DEFAULT_TEST_MODEL
8
+ # use the bare id (the suffix after the last "/"). `--model all` runs every entry and
9
+ # needs the API key for each referenced provider.
10
+ OPENCLAW_TEST_MODELS=anthropic/claude-sonnet-4-6,custom-openrouter/qwen/qwen3.6-plus
11
+ OPENCLAW_DEFAULT_TEST_MODEL=claude-sonnet-4-6
12
+
13
+ # Required: host path to the OpenClaw workspace (mounted into the gateway).
14
+ OPENCLAW_WORKSPACE_DIR=
15
+
16
+ # Optional: run N matrix cells concurrently, each on its own worker stack
17
+ # (`run --parallel` wins over this).
18
+ # OPENCLAW_TEST_PARALLEL=4
19
+
20
+ # Optional overrides (defaults relative to the project dir):
21
+ # OPENCLAW_CONFIG_PATH=./openclaw.json
22
+ # OPENCLAW_TEST_SCENARIOS_DIR=./scenarios
23
+ # OPENCLAW_TEST_ARTIFACTS_DIR=./artifacts
24
+ # OPENCLAW_TEST_GATEWAY_LOGS_DIR=./.gateway-logs
25
+
26
+ # Opt-in: also write raw-stream.jsonl alongside the always-on provider-neutral
27
+ # trajectory log (one <sessionId>.jsonl per session under .gateway-logs/trajectory/).
28
+ # OPENCLAW_RAW_STREAM=1
@@ -0,0 +1,47 @@
1
+ # Consumer-side test runner image. Extends the consumer-agnostic base built locally
2
+ # by `openclaw-test env build` and tagged `paleo/openclaw-test-base:<pkg-version>`.
3
+ # The CLI passes the matching tag via the OPENCLAW_TEST_BASE_TAG build arg.
4
+ #
5
+ # Build context is the project dir.
6
+
7
+ ARG OPENCLAW_TEST_BASE_TAG=__SET_VIA_OPENCLAW_TEST_CLI__
8
+ FROM paleo/openclaw-test-base:${OPENCLAW_TEST_BASE_TAG}
9
+
10
+ COPY --chown=assistant:assistant package.json package-lock.json /opt/openclaw-test/src/
11
+ COPY --chown=assistant:assistant openclaw.json /home/assistant/.openclaw/openclaw.json
12
+
13
+ RUN npm ci --include=dev && \
14
+ OPENCLAW_CONFIG_PATH=/home/assistant/.openclaw/openclaw.json npx openclaw plugins registry --refresh
15
+
16
+ # Consumer customizations below. Add RUN/COPY/ENV as needed. The base image
17
+ # is deliberately minimal — add only what your fixtures and scenarios need.
18
+ # Common patterns:
19
+ #
20
+ # * System tools the fixtures invoke (under USER root):
21
+ # RUN apk add --no-cache git
22
+ # RUN npm install --global corepack@0.36.0 && \
23
+ # corepack enable && corepack prepare pnpm@latest --activate
24
+ #
25
+ # * Per-command mock-CLI symlinks (the base ships only the bare shim;
26
+ # consumers pick which commands to intercept):
27
+ # USER root
28
+ # RUN for name in claude gh; do \
29
+ # ln -sf mock-cli-shim "/opt/openclaw-test/mocks/bin/$name"; \
30
+ # done
31
+ # USER assistant
32
+ #
33
+ # * Helper scripts invoked from scenarios via `ctx.execInGateway(...)`:
34
+ # USER root
35
+ # COPY --chown=root:root scripts/ /opt/openclaw-test/scripts/
36
+ # RUN chmod +x /opt/openclaw-test/scripts/*.mjs
37
+ # USER assistant
38
+ #
39
+ # * Project fixtures baked into the image (typical with a named-volume
40
+ # /home/assistant/projects, then reset/seeded per scenario):
41
+ # COPY --chown=assistant:assistant projects-fixture/<name>/ /opt/<consumer>/fixtures/<name>/
42
+ # RUN cd /opt/<consumer>/fixtures/<name> && pnpm install --frozen-lockfile
43
+ #
44
+ # * Skills under /home/assistant/.agents/skills/ for the assistant to discover
45
+ # (e.g. installed with the `skills` CLI from any repo or registry):
46
+ # RUN npx -y skills add <skills-source> --global --yes \
47
+ # --agent claude-code --skill <name> [--skill <name> …]
@@ -0,0 +1,10 @@
1
+ # Thin consumer overlay. Pulls in the parameterized base from
2
+ # @alignfirst/openclaw-test. Requires Docker Compose v2.20+.
3
+ #
4
+ # `.env.local` only needs ANTHROPIC_API_KEY and OPENCLAW_WORKSPACE_DIR.
5
+ # All other paths default to subdirs of the project dir; override in
6
+ # .env.local if you need a different layout. OPENCLAW_TEST_PROJECT_DIR, OPENCLAW_TEST_PACKAGE_DIR,
7
+ # ASSISTANT_UID, ASSISTANT_GID are set automatically by `openclaw-test env|run`.
8
+
9
+ include:
10
+ - ./node_modules/@alignfirst/openclaw-test/docker-compose.yml