@alignfirst/openclaw-test 0.20.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/Dockerfile.base +45 -0
- package/LICENSE +21 -0
- package/README.md +132 -0
- package/bin/cli.mjs +7 -0
- package/dist/bus.d.ts +1 -0
- package/dist/bus.js +20 -0
- package/dist/cell-result.d.ts +25 -0
- package/dist/cell-result.js +32 -0
- package/dist/cli.d.ts +5 -0
- package/dist/cli.js +110 -0
- package/dist/context.d.ts +220 -0
- package/dist/context.js +479 -0
- package/dist/cost.d.ts +2 -0
- package/dist/cost.js +19 -0
- package/dist/env-cli.d.ts +5 -0
- package/dist/env-cli.js +645 -0
- package/dist/exec-rpc.d.ts +13 -0
- package/dist/exec-rpc.js +46 -0
- package/dist/index.d.ts +3 -0
- package/dist/index.js +1 -0
- package/dist/judge.d.ts +47 -0
- package/dist/judge.js +183 -0
- package/dist/loop.d.ts +62 -0
- package/dist/loop.js +316 -0
- package/dist/mock-cli-server.d.ts +40 -0
- package/dist/mock-cli-server.js +194 -0
- package/dist/mock-cli-shim.d.ts +7 -0
- package/dist/mock-cli-shim.js +89 -0
- package/dist/models.d.ts +23 -0
- package/dist/models.js +78 -0
- package/dist/parse-tagged-json.d.ts +1 -0
- package/dist/parse-tagged-json.js +16 -0
- package/dist/report.d.ts +215 -0
- package/dist/report.js +19 -0
- package/dist/runner-args.d.ts +11 -0
- package/dist/runner-args.js +84 -0
- package/dist/runner.d.ts +2 -0
- package/dist/runner.js +371 -0
- package/dist/summary.d.ts +3 -0
- package/dist/summary.js +64 -0
- package/dist/transcript-dump.d.ts +1 -0
- package/dist/transcript-dump.js +50 -0
- package/dist/transcript-log.d.ts +67 -0
- package/dist/transcript-log.js +199 -0
- package/dist/transcript-store.d.ts +6 -0
- package/dist/transcript-store.js +69 -0
- package/docker-compose.yml +83 -0
- package/exec-watcher.mjs +155 -0
- package/package.json +60 -0
- package/templates/.env.local.example +28 -0
- package/templates/Dockerfile +47 -0
- package/templates/docker-compose.yml +10 -0
- package/templates/openclaw.json +40 -0
|
@@ -0,0 +1,199 @@
|
|
|
1
|
+
import { randomUUID } from "node:crypto";
|
|
2
|
+
import { readFileSync, rmSync } from "node:fs";
|
|
3
|
+
import { getQaBusState } from "@alignfirst/openclaw-channel-mock-core";
|
|
4
|
+
import { execInGateway, IPC_DIR } from "./exec-rpc.js";
|
|
5
|
+
// OpenClaw persists each session's transcript as SQLite rows in the gateway's
|
|
6
|
+
// per-agent store — the full-fidelity record the gateway itself replays,
|
|
7
|
+
// appended per message. That store lives outside the shared mounts, so the
|
|
8
|
+
// runner extracts a conversation's session transcripts through the
|
|
9
|
+
// exec-watcher RPC: the dump script (shipped in this package's dist, mounted
|
|
10
|
+
// into the gateway) writes them as JSON into the shared IPC volume.
|
|
11
|
+
const DUMP_SCRIPT = "/opt/openclaw-test/src/dist/transcript-dump.js";
|
|
12
|
+
const BUS_URL = process.env.OPENCLAW_TEST_BUS_URL ?? "http://bus:43123";
|
|
13
|
+
/** The transcripts of the conversation and its bus-owned thread sessions. */
|
|
14
|
+
export async function fetchTranscriptSnapshot(opts) {
|
|
15
|
+
const outPath = `${IPC_DIR}/${randomUUID()}.transcript.json`;
|
|
16
|
+
try {
|
|
17
|
+
const { threads } = await getQaBusState(BUS_URL);
|
|
18
|
+
const conversationId = opts.conversationId.toLowerCase();
|
|
19
|
+
const threadIds = threads
|
|
20
|
+
.filter((thread) => thread.conversationId.toLowerCase() === conversationId)
|
|
21
|
+
.map((thread) => thread.id);
|
|
22
|
+
const result = await execInGateway(["node", DUMP_SCRIPT, opts.startedAtIso, opts.conversationId, outPath, ...threadIds],
|
|
23
|
+
// A whole conversation can take a while to serialize; the default 30s
|
|
24
|
+
// exec timeout is too tight for long cells.
|
|
25
|
+
{ timeoutMs: 60_000 });
|
|
26
|
+
if (result.exitCode !== 0) {
|
|
27
|
+
const error = `transcript dump failed (exit ${result.exitCode}): ${result.stderr}`;
|
|
28
|
+
console.warn(`openclaw-test: ${error}`);
|
|
29
|
+
return { databases: 0, sessions: [], error };
|
|
30
|
+
}
|
|
31
|
+
const raw = readFileSync(outPath, "utf8");
|
|
32
|
+
return JSON.parse(raw);
|
|
33
|
+
}
|
|
34
|
+
catch (err) {
|
|
35
|
+
const error = `transcript dump failed: ${String(err)}`;
|
|
36
|
+
console.warn(`openclaw-test: ${error}`);
|
|
37
|
+
return { databases: 0, sessions: [], error };
|
|
38
|
+
}
|
|
39
|
+
finally {
|
|
40
|
+
rmSync(outPath, { force: true });
|
|
41
|
+
}
|
|
42
|
+
}
|
|
43
|
+
/**
|
|
44
|
+
* Polls the transcripts until they are *quiescent* for this conversation — no
|
|
45
|
+
* new message for `settleMs` AND no session's turn is demonstrably open — or
|
|
46
|
+
* the budget expires. The transcript is appended per message, so a settle
|
|
47
|
+
* window alone would return between two tool calls of one turn (a single
|
|
48
|
+
* model round-trip routinely exceeds it); the open-turn gate holds until the
|
|
49
|
+
* turn-final assistant message — the one carrying `usage.cost.total` — has
|
|
50
|
+
* landed. A conversation spans multiple OpenClaw sessions (e.g. Discord's
|
|
51
|
+
* channel session plus a per-thread session).
|
|
52
|
+
*/
|
|
53
|
+
export async function waitForTranscriptQuiescence(opts) {
|
|
54
|
+
const maxWaitMs = opts.maxWaitMs ?? 60_000;
|
|
55
|
+
const pollMs = opts.pollMs ?? 1_000;
|
|
56
|
+
const settleMs = opts.settleMs ?? 4_000;
|
|
57
|
+
const deadline = Date.now() + maxWaitMs;
|
|
58
|
+
let lastCount = 0;
|
|
59
|
+
let lastChangeAt = Date.now();
|
|
60
|
+
let seenAny = false;
|
|
61
|
+
while (Date.now() < deadline) {
|
|
62
|
+
const { sessions } = await fetchTranscriptSnapshot(opts);
|
|
63
|
+
const count = sessions.reduce((sum, s) => sum + s.messages.length, 0);
|
|
64
|
+
if (count > 0)
|
|
65
|
+
seenAny = true;
|
|
66
|
+
if (count !== lastCount) {
|
|
67
|
+
lastCount = count;
|
|
68
|
+
lastChangeAt = Date.now();
|
|
69
|
+
}
|
|
70
|
+
else if (seenAny && !hasOpenTurn(sessions) && Date.now() - lastChangeAt >= settleMs) {
|
|
71
|
+
return;
|
|
72
|
+
}
|
|
73
|
+
await new Promise((r) => setTimeout(r, pollMs));
|
|
74
|
+
}
|
|
75
|
+
}
|
|
76
|
+
/**
|
|
77
|
+
* A turn is open when a session's last message is a tool result or an
|
|
78
|
+
* assistant stop for tool use — more of the turn is coming. A trailing user
|
|
79
|
+
* message does not count as open: some sessions of a conversation never get a
|
|
80
|
+
* reply (the agent answers in a sibling session), and treating them as open
|
|
81
|
+
* would burn the whole budget on every cell.
|
|
82
|
+
*/
|
|
83
|
+
export function hasOpenTurn(sessions) {
|
|
84
|
+
return sessions.some((session) => {
|
|
85
|
+
const last = session.messages.at(-1);
|
|
86
|
+
if (!isRecord(last))
|
|
87
|
+
return false;
|
|
88
|
+
if (last.role === "toolResult")
|
|
89
|
+
return true;
|
|
90
|
+
return last.role === "assistant" && last.stopReason === "toolUse";
|
|
91
|
+
});
|
|
92
|
+
}
|
|
93
|
+
/**
|
|
94
|
+
* Cost lives per assistant message, as `usage.cost.total`. Sum across every
|
|
95
|
+
* session of the conversation; `turns` counts the cost-bearing messages.
|
|
96
|
+
*/
|
|
97
|
+
export function readTranscriptCost(snapshot) {
|
|
98
|
+
let cost = 0;
|
|
99
|
+
let turns = 0;
|
|
100
|
+
for (const session of snapshot.sessions) {
|
|
101
|
+
for (const msg of session.messages) {
|
|
102
|
+
if (!isRecord(msg) || msg.role !== "assistant")
|
|
103
|
+
continue;
|
|
104
|
+
const total = assistantCostTotal(msg);
|
|
105
|
+
if (typeof total !== "number")
|
|
106
|
+
continue;
|
|
107
|
+
cost += total;
|
|
108
|
+
turns += 1;
|
|
109
|
+
}
|
|
110
|
+
}
|
|
111
|
+
return { cost, turns };
|
|
112
|
+
}
|
|
113
|
+
function assistantCostTotal(msg) {
|
|
114
|
+
const usage = msg.usage;
|
|
115
|
+
if (!isRecord(usage))
|
|
116
|
+
return;
|
|
117
|
+
const cost = usage.cost;
|
|
118
|
+
if (!isRecord(cost))
|
|
119
|
+
return;
|
|
120
|
+
return typeof cost.total === "number" ? cost.total : undefined;
|
|
121
|
+
}
|
|
122
|
+
/** One-shot fetch + aggregation of the conversation's agent tool calls. */
|
|
123
|
+
export async function parseAgentToolCalls(opts) {
|
|
124
|
+
const { sessions } = await fetchTranscriptSnapshot(opts);
|
|
125
|
+
return aggregateAgentToolCalls(sessions);
|
|
126
|
+
}
|
|
127
|
+
/**
|
|
128
|
+
* Walk each session's messages collecting assistant `toolCall` blocks matched
|
|
129
|
+
* with `toolResult` messages, then union across sessions deduped by
|
|
130
|
+
* `toolUseId`.
|
|
131
|
+
*/
|
|
132
|
+
export function aggregateAgentToolCalls(sessions) {
|
|
133
|
+
const calls = [];
|
|
134
|
+
const seen = new Set();
|
|
135
|
+
for (const session of sessions) {
|
|
136
|
+
const results = collectToolResults(session.messages);
|
|
137
|
+
for (const call of collectToolUses(session.messages, results, session.sessionKey)) {
|
|
138
|
+
if (seen.has(call.toolUseId))
|
|
139
|
+
continue;
|
|
140
|
+
seen.add(call.toolUseId);
|
|
141
|
+
calls.push(call);
|
|
142
|
+
}
|
|
143
|
+
}
|
|
144
|
+
return calls;
|
|
145
|
+
}
|
|
146
|
+
function collectToolResults(messages) {
|
|
147
|
+
const results = new Map();
|
|
148
|
+
for (const msg of messages) {
|
|
149
|
+
if (!isRecord(msg) || msg.role !== "toolResult")
|
|
150
|
+
continue;
|
|
151
|
+
const toolCallId = msg.toolCallId;
|
|
152
|
+
if (typeof toolCallId !== "string")
|
|
153
|
+
continue;
|
|
154
|
+
results.set(toolCallId, { isError: msg.isError === true, content: msg.content ?? null });
|
|
155
|
+
}
|
|
156
|
+
return results;
|
|
157
|
+
}
|
|
158
|
+
function collectToolUses(messages, results, sessionKey) {
|
|
159
|
+
const calls = [];
|
|
160
|
+
let turn = 0;
|
|
161
|
+
for (const msg of messages) {
|
|
162
|
+
if (!isRecord(msg) || msg.role !== "assistant" || !Array.isArray(msg.content))
|
|
163
|
+
continue;
|
|
164
|
+
turn += 1;
|
|
165
|
+
for (const block of msg.content) {
|
|
166
|
+
if (!isRecord(block) || block.type !== "toolCall")
|
|
167
|
+
continue;
|
|
168
|
+
const toolUseId = typeof block.id === "string" ? block.id : "";
|
|
169
|
+
const toolName = typeof block.name === "string" ? block.name : "";
|
|
170
|
+
if (!toolUseId || !toolName)
|
|
171
|
+
continue;
|
|
172
|
+
const startedAt = messageTimestampIso(msg);
|
|
173
|
+
const call = {
|
|
174
|
+
toolName,
|
|
175
|
+
toolUseId,
|
|
176
|
+
...(sessionKey !== undefined ? { sessionKey } : {}),
|
|
177
|
+
input: block.arguments ?? null,
|
|
178
|
+
...(startedAt !== undefined ? { startedAt } : {}),
|
|
179
|
+
turn,
|
|
180
|
+
};
|
|
181
|
+
const result = results.get(toolUseId);
|
|
182
|
+
if (result)
|
|
183
|
+
call.result = result;
|
|
184
|
+
calls.push(call);
|
|
185
|
+
}
|
|
186
|
+
}
|
|
187
|
+
return calls;
|
|
188
|
+
}
|
|
189
|
+
function messageTimestampIso(msg) {
|
|
190
|
+
const ts = msg.timestamp;
|
|
191
|
+
if (typeof ts === "number")
|
|
192
|
+
return new Date(ts).toISOString();
|
|
193
|
+
if (typeof ts === "string")
|
|
194
|
+
return ts;
|
|
195
|
+
return undefined;
|
|
196
|
+
}
|
|
197
|
+
function isRecord(value) {
|
|
198
|
+
return typeof value === "object" && value !== null;
|
|
199
|
+
}
|
|
@@ -0,0 +1,69 @@
|
|
|
1
|
+
import { DatabaseSync } from "node:sqlite";
|
|
2
|
+
export function readConversationSessions(dbPath, sinceMs, conversationId, threadIds) {
|
|
3
|
+
let db;
|
|
4
|
+
try {
|
|
5
|
+
db = new DatabaseSync(dbPath, { readOnly: true });
|
|
6
|
+
}
|
|
7
|
+
catch {
|
|
8
|
+
return [];
|
|
9
|
+
}
|
|
10
|
+
try {
|
|
11
|
+
db.exec("PRAGMA busy_timeout = 2000;");
|
|
12
|
+
const wanted = conversationId.toLowerCase();
|
|
13
|
+
const nodes = db.prepare("SELECT session_key, current_session_id FROM session_nodes").all();
|
|
14
|
+
const sessions = [];
|
|
15
|
+
for (const node of nodes) {
|
|
16
|
+
const segments = node.session_key.toLowerCase().split(":");
|
|
17
|
+
if (!segments.includes(wanted) && !threadIds.some((id) => segments.includes(id)))
|
|
18
|
+
continue;
|
|
19
|
+
sessions.push({
|
|
20
|
+
sessionKey: node.session_key,
|
|
21
|
+
sessionId: node.current_session_id,
|
|
22
|
+
messages: readNodeMessages(db, node, sinceMs),
|
|
23
|
+
});
|
|
24
|
+
}
|
|
25
|
+
return sessions;
|
|
26
|
+
}
|
|
27
|
+
catch {
|
|
28
|
+
// Fresh store without the tables yet, or a transient lock: the runner's
|
|
29
|
+
// next poll retries.
|
|
30
|
+
return [];
|
|
31
|
+
}
|
|
32
|
+
finally {
|
|
33
|
+
db.close();
|
|
34
|
+
}
|
|
35
|
+
}
|
|
36
|
+
/**
|
|
37
|
+
* A session key can span several session windows: compaction, reset, or
|
|
38
|
+
* recovery mints a successor id that `current_session_id` then advances to,
|
|
39
|
+
* and `transcript_events` rows stay keyed to the window that wrote them.
|
|
40
|
+
* Union every window of the key (in creation order) so earlier tool calls and
|
|
41
|
+
* costs survive a mid-run rollover.
|
|
42
|
+
*/
|
|
43
|
+
function readNodeMessages(db, node, sinceMs) {
|
|
44
|
+
const windows = db
|
|
45
|
+
.prepare("SELECT session_id FROM session_windows WHERE session_key = ? ORDER BY created_at")
|
|
46
|
+
.all(node.session_key);
|
|
47
|
+
const sessionIds = windows.map((w) => w.session_id);
|
|
48
|
+
if (!sessionIds.includes(node.current_session_id))
|
|
49
|
+
sessionIds.push(node.current_session_id);
|
|
50
|
+
return sessionIds.flatMap((id) => readSessionMessages(db, id, sinceMs));
|
|
51
|
+
}
|
|
52
|
+
function readSessionMessages(db, sessionId, sinceMs) {
|
|
53
|
+
const rows = db
|
|
54
|
+
.prepare("SELECT event_json FROM transcript_events WHERE session_id = ? AND created_at >= ? ORDER BY seq")
|
|
55
|
+
.all(sessionId, sinceMs);
|
|
56
|
+
const messages = [];
|
|
57
|
+
for (const row of rows) {
|
|
58
|
+
let event;
|
|
59
|
+
try {
|
|
60
|
+
event = JSON.parse(row.event_json);
|
|
61
|
+
}
|
|
62
|
+
catch {
|
|
63
|
+
continue;
|
|
64
|
+
}
|
|
65
|
+
if (event.type === "message" && event.message !== undefined)
|
|
66
|
+
messages.push(event.message);
|
|
67
|
+
}
|
|
68
|
+
return messages;
|
|
69
|
+
}
|
|
@@ -0,0 +1,83 @@
|
|
|
1
|
+
x-common-build: &common-build
|
|
2
|
+
context: ${OPENCLAW_TEST_PROJECT_DIR}
|
|
3
|
+
dockerfile: Dockerfile
|
|
4
|
+
args:
|
|
5
|
+
OPENCLAW_TEST_BASE_TAG: ${OPENCLAW_TEST_BASE_TAG}
|
|
6
|
+
|
|
7
|
+
# Single source of truth for service mounts. `openclaw-test-ipc` is only used by gateway
|
|
8
|
+
# (writer) and runner (reader), but mounting it on `bus` too is harmless and
|
|
9
|
+
# lets all three services share one anchor.
|
|
10
|
+
x-common-volumes: &common-volumes
|
|
11
|
+
- ${OPENCLAW_WORKSPACE_DIR}:/home/assistant/.openclaw/workspace
|
|
12
|
+
- ${OPENCLAW_CONFIG_PATH}:/home/assistant/.openclaw/openclaw.json
|
|
13
|
+
- ${OPENCLAW_TEST_PACKAGE_DIR}/dist/:/opt/openclaw-test/src/dist
|
|
14
|
+
- ${OPENCLAW_TEST_SCENARIOS_DIR}:/opt/openclaw-test/src/scenarios
|
|
15
|
+
- ${OPENCLAW_TEST_ARTIFACTS_DIR}:/opt/openclaw-test/artifacts
|
|
16
|
+
- ${OPENCLAW_TEST_GATEWAY_LOGS_DIR}:/home/assistant/.openclaw/logs
|
|
17
|
+
- openclaw-test-ipc:/var/run/openclaw-test-ipc
|
|
18
|
+
|
|
19
|
+
volumes:
|
|
20
|
+
openclaw-test-ipc:
|
|
21
|
+
|
|
22
|
+
# All three services run the same image. The explicit shared `image:` keeps worker
|
|
23
|
+
# Compose projects (`<base>-w<i>`) from each building/looking up their own
|
|
24
|
+
# project-named image; the CLI injects OPENCLAW_TEST_CONSUMER_IMAGE.
|
|
25
|
+
services:
|
|
26
|
+
bus:
|
|
27
|
+
build: *common-build
|
|
28
|
+
image: ${OPENCLAW_TEST_CONSUMER_IMAGE}
|
|
29
|
+
user: assistant
|
|
30
|
+
working_dir: /opt/openclaw-test/src
|
|
31
|
+
command: ["node", "/opt/openclaw-test/src/dist/bus.js"]
|
|
32
|
+
volumes: *common-volumes
|
|
33
|
+
healthcheck:
|
|
34
|
+
test: ["CMD", "wget", "-qO-", "http://127.0.0.1:43123/v1/state"]
|
|
35
|
+
interval: 2s
|
|
36
|
+
timeout: 2s
|
|
37
|
+
retries: 10
|
|
38
|
+
|
|
39
|
+
gateway:
|
|
40
|
+
build: *common-build
|
|
41
|
+
image: ${OPENCLAW_TEST_CONSUMER_IMAGE}
|
|
42
|
+
user: assistant
|
|
43
|
+
working_dir: /opt/openclaw-test/src
|
|
44
|
+
environment:
|
|
45
|
+
OPENCLAW_CONFIG_PATH: /home/assistant/.openclaw/openclaw.json
|
|
46
|
+
OPENCLAW_CONFIG_READONLY: "1"
|
|
47
|
+
ANTHROPIC_API_KEY: ${ANTHROPIC_API_KEY}
|
|
48
|
+
OPENROUTER_API_KEY: ${OPENROUTER_API_KEY:-}
|
|
49
|
+
OPENCLAW_RAW_STREAM: ${OPENCLAW_RAW_STREAM:-}
|
|
50
|
+
OPENCLAW_TEST_RUNNER_URL: "http://runner:43124"
|
|
51
|
+
PATH: "/opt/openclaw-test/mocks/bin:/usr/local/sbin:/usr/local/bin:/usr/sbin:/usr/bin:/sbin:/bin"
|
|
52
|
+
# Watcher in the background, then exec openclaw as PID 1's foreground process.
|
|
53
|
+
command:
|
|
54
|
+
["sh", "-c",
|
|
55
|
+
"/usr/local/bin/exec-watcher & exec npx openclaw gateway run --port 18789 --bind loopback --auth none"]
|
|
56
|
+
volumes: *common-volumes
|
|
57
|
+
depends_on:
|
|
58
|
+
bus:
|
|
59
|
+
condition: service_healthy
|
|
60
|
+
healthcheck:
|
|
61
|
+
test: ["CMD-SHELL", "node -e \"require('net').connect(18789, '127.0.0.1').on('connect', () => process.exit(0)).on('error', () => process.exit(1))\""]
|
|
62
|
+
interval: 5s
|
|
63
|
+
timeout: 5s
|
|
64
|
+
retries: 20
|
|
65
|
+
start_period: 30s
|
|
66
|
+
|
|
67
|
+
runner:
|
|
68
|
+
build: *common-build
|
|
69
|
+
image: ${OPENCLAW_TEST_CONSUMER_IMAGE}
|
|
70
|
+
user: assistant
|
|
71
|
+
working_dir: /opt/openclaw-test/src
|
|
72
|
+
environment:
|
|
73
|
+
OPENCLAW_TEST_BUS_URL: http://bus:43123
|
|
74
|
+
OPENCLAW_CONFIG_PATH: /home/assistant/.openclaw/openclaw.json
|
|
75
|
+
ANTHROPIC_API_KEY: ${ANTHROPIC_API_KEY}
|
|
76
|
+
OPENROUTER_API_KEY: ${OPENROUTER_API_KEY:-}
|
|
77
|
+
entrypoint: ["node", "/opt/openclaw-test/src/dist/runner.js"]
|
|
78
|
+
volumes: *common-volumes
|
|
79
|
+
depends_on:
|
|
80
|
+
bus:
|
|
81
|
+
condition: service_healthy
|
|
82
|
+
gateway:
|
|
83
|
+
condition: service_healthy
|
package/exec-watcher.mjs
ADDED
|
@@ -0,0 +1,155 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
import { spawn } from "node:child_process";
|
|
3
|
+
import { mkdirSync, readdirSync, readFileSync, renameSync, rmSync, writeFileSync } from "node:fs";
|
|
4
|
+
|
|
5
|
+
const IPC_DIR = "/var/run/openclaw-test-ipc";
|
|
6
|
+
const MAX_OUTPUT_BYTES = 1_048_576;
|
|
7
|
+
const POLL_INTERVAL_MS = 100;
|
|
8
|
+
const STALE_SUFFIXES = [
|
|
9
|
+
".req.json",
|
|
10
|
+
".req.json.processing",
|
|
11
|
+
".res.json",
|
|
12
|
+
".res.json.tmp",
|
|
13
|
+
".transcript.json",
|
|
14
|
+
".transcript.json.tmp",
|
|
15
|
+
];
|
|
16
|
+
|
|
17
|
+
mkdirSync(IPC_DIR, { recursive: true });
|
|
18
|
+
|
|
19
|
+
// Sweep stale IPC artifacts left by a killed predecessor in the (named-volume,
|
|
20
|
+
// persistent) IPC dir, so the next runner cycle starts clean.
|
|
21
|
+
try {
|
|
22
|
+
for (const f of readdirSync(IPC_DIR)) {
|
|
23
|
+
if (STALE_SUFFIXES.some((suffix) => f.endsWith(suffix))) {
|
|
24
|
+
try {
|
|
25
|
+
rmSync(`${IPC_DIR}/${f}`, { force: true });
|
|
26
|
+
} catch (err) {
|
|
27
|
+
console.error(`exec-watcher: failed to clean stale IPC file ${f}:`, err);
|
|
28
|
+
}
|
|
29
|
+
}
|
|
30
|
+
}
|
|
31
|
+
} catch (err) {
|
|
32
|
+
console.error(`exec-watcher: failed to scan IPC dir ${IPC_DIR}:`, err);
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
async function main() {
|
|
36
|
+
for (;;) {
|
|
37
|
+
let entries;
|
|
38
|
+
try {
|
|
39
|
+
entries = readdirSync(IPC_DIR);
|
|
40
|
+
} catch {
|
|
41
|
+
entries = [];
|
|
42
|
+
}
|
|
43
|
+
const reqFiles = entries.filter((f) => f.endsWith(".req.json"));
|
|
44
|
+
for (const f of reqFiles) {
|
|
45
|
+
const reqPath = `${IPC_DIR}/${f}`;
|
|
46
|
+
const claimed = `${reqPath}.processing`;
|
|
47
|
+
try {
|
|
48
|
+
renameSync(reqPath, claimed);
|
|
49
|
+
} catch {
|
|
50
|
+
continue;
|
|
51
|
+
}
|
|
52
|
+
processRequest(claimed).catch((err) => {
|
|
53
|
+
console.error("exec-watcher: internal error", err);
|
|
54
|
+
});
|
|
55
|
+
}
|
|
56
|
+
await new Promise((r) => setTimeout(r, POLL_INTERVAL_MS));
|
|
57
|
+
}
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
async function processRequest(claimedPath) {
|
|
61
|
+
let id = "unknown";
|
|
62
|
+
try {
|
|
63
|
+
const raw = readFileSync(claimedPath, "utf8");
|
|
64
|
+
const req = JSON.parse(raw);
|
|
65
|
+
id = req.id;
|
|
66
|
+
const result = await runChild(req);
|
|
67
|
+
writeResult(id, result);
|
|
68
|
+
} catch (err) {
|
|
69
|
+
const message = err instanceof Error ? err.message : String(err);
|
|
70
|
+
writeResult(id, { exitCode: 255, stdout: "", stderr: `watcher error: ${message}` });
|
|
71
|
+
} finally {
|
|
72
|
+
rmSync(claimedPath, { force: true });
|
|
73
|
+
}
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
function runChild(req) {
|
|
77
|
+
const { argv, cwd, env, stdin, timeoutMs = 30_000 } = req;
|
|
78
|
+
return new Promise((resolve) => {
|
|
79
|
+
const child = spawn(argv[0], argv.slice(1), {
|
|
80
|
+
cwd,
|
|
81
|
+
env: { ...process.env, ...(env ?? {}) },
|
|
82
|
+
stdio: ["pipe", "pipe", "pipe"],
|
|
83
|
+
});
|
|
84
|
+
|
|
85
|
+
let stdout = "";
|
|
86
|
+
let stderr = "";
|
|
87
|
+
let stdoutDropped = 0;
|
|
88
|
+
let stderrDropped = 0;
|
|
89
|
+
let killedByTimeout = false;
|
|
90
|
+
|
|
91
|
+
child.stdout.on("data", (chunk) => {
|
|
92
|
+
const s = chunk.toString();
|
|
93
|
+
if (stdout.length + s.length > MAX_OUTPUT_BYTES) {
|
|
94
|
+
const room = Math.max(0, MAX_OUTPUT_BYTES - stdout.length);
|
|
95
|
+
stdout += s.slice(0, room);
|
|
96
|
+
stdoutDropped += s.length - room;
|
|
97
|
+
} else {
|
|
98
|
+
stdout += s;
|
|
99
|
+
}
|
|
100
|
+
});
|
|
101
|
+
child.stderr.on("data", (chunk) => {
|
|
102
|
+
const s = chunk.toString();
|
|
103
|
+
if (stderr.length + s.length > MAX_OUTPUT_BYTES) {
|
|
104
|
+
const room = Math.max(0, MAX_OUTPUT_BYTES - stderr.length);
|
|
105
|
+
stderr += s.slice(0, room);
|
|
106
|
+
stderrDropped += s.length - room;
|
|
107
|
+
} else {
|
|
108
|
+
stderr += s;
|
|
109
|
+
}
|
|
110
|
+
});
|
|
111
|
+
|
|
112
|
+
if (stdin !== undefined) {
|
|
113
|
+
child.stdin.write(stdin);
|
|
114
|
+
}
|
|
115
|
+
child.stdin.end();
|
|
116
|
+
|
|
117
|
+
const timer = setTimeout(() => {
|
|
118
|
+
killedByTimeout = true;
|
|
119
|
+
child.kill("SIGKILL");
|
|
120
|
+
}, timeoutMs);
|
|
121
|
+
|
|
122
|
+
child.on("error", (err) => {
|
|
123
|
+
clearTimeout(timer);
|
|
124
|
+
resolve({
|
|
125
|
+
exitCode: 255,
|
|
126
|
+
stdout,
|
|
127
|
+
stderr: `${stderr}\nspawn error: ${err.message}`,
|
|
128
|
+
});
|
|
129
|
+
});
|
|
130
|
+
child.on("exit", (code, signal) => {
|
|
131
|
+
clearTimeout(timer);
|
|
132
|
+
if (stdoutDropped > 0) stdout += `\n…[truncated ${stdoutDropped} bytes]`;
|
|
133
|
+
if (stderrDropped > 0) stderr += `\n…[truncated ${stderrDropped} bytes]`;
|
|
134
|
+
if (killedByTimeout) {
|
|
135
|
+
resolve({
|
|
136
|
+
exitCode: 124,
|
|
137
|
+
stdout,
|
|
138
|
+
stderr: `${stderr}\n…[killed by watcher timeout after ${timeoutMs}ms]`,
|
|
139
|
+
});
|
|
140
|
+
return;
|
|
141
|
+
}
|
|
142
|
+
const exitCode = code ?? (signal ? 128 : 1);
|
|
143
|
+
resolve({ exitCode, stdout, stderr });
|
|
144
|
+
});
|
|
145
|
+
});
|
|
146
|
+
}
|
|
147
|
+
|
|
148
|
+
function writeResult(id, result) {
|
|
149
|
+
const finalPath = `${IPC_DIR}/${id}.res.json`;
|
|
150
|
+
const tmp = `${finalPath}.tmp`;
|
|
151
|
+
writeFileSync(tmp, JSON.stringify(result));
|
|
152
|
+
renameSync(tmp, finalPath);
|
|
153
|
+
}
|
|
154
|
+
|
|
155
|
+
void main();
|
package/package.json
ADDED
|
@@ -0,0 +1,60 @@
|
|
|
1
|
+
{
|
|
2
|
+
"name": "@alignfirst/openclaw-test",
|
|
3
|
+
"version": "0.20.0",
|
|
4
|
+
"description": "Dockerised regression-test framework for OpenClaw workspaces: bus, scenario driver, judge, Compose stack.",
|
|
5
|
+
"keywords": [
|
|
6
|
+
"openclaw",
|
|
7
|
+
"test",
|
|
8
|
+
"testing",
|
|
9
|
+
"harness",
|
|
10
|
+
"scenario",
|
|
11
|
+
"docker"
|
|
12
|
+
],
|
|
13
|
+
"license": "MIT",
|
|
14
|
+
"author": "Thomas MUR",
|
|
15
|
+
"homepage": "https://alignfirst.paroi.tech/",
|
|
16
|
+
"repository": {
|
|
17
|
+
"type": "git",
|
|
18
|
+
"url": "git+https://github.com/paleo/alignfirst.git",
|
|
19
|
+
"directory": "packages/openclaw-test"
|
|
20
|
+
},
|
|
21
|
+
"engines": {
|
|
22
|
+
"node": ">=24.16.0 <25 || >=26.1.0"
|
|
23
|
+
},
|
|
24
|
+
"packageManager": "npm@11.19.0",
|
|
25
|
+
"type": "module",
|
|
26
|
+
"main": "./dist/index.js",
|
|
27
|
+
"bin": {
|
|
28
|
+
"openclaw-test": "./bin/cli.mjs"
|
|
29
|
+
},
|
|
30
|
+
"exports": "./dist/index.js",
|
|
31
|
+
"files": [
|
|
32
|
+
"dist/",
|
|
33
|
+
"bin/",
|
|
34
|
+
"templates/",
|
|
35
|
+
"docker-compose.yml",
|
|
36
|
+
"Dockerfile.base",
|
|
37
|
+
"exec-watcher.mjs",
|
|
38
|
+
"LICENSE"
|
|
39
|
+
],
|
|
40
|
+
"publishConfig": {
|
|
41
|
+
"access": "public"
|
|
42
|
+
},
|
|
43
|
+
"peerDependencies": {
|
|
44
|
+
"openclaw": "*"
|
|
45
|
+
},
|
|
46
|
+
"dependencies": {
|
|
47
|
+
"@alignfirst/openclaw-channel-mock-core": "0.9.0",
|
|
48
|
+
"@alignfirst/openclaw-discord-mock": "0.5.0",
|
|
49
|
+
"@alignfirst/openclaw-slack-mock": "0.5.0",
|
|
50
|
+
"@anthropic-ai/sdk": "~0.122.0",
|
|
51
|
+
"openai": "~7.8.0"
|
|
52
|
+
},
|
|
53
|
+
"devDependencies": {
|
|
54
|
+
"@types/node": "~26.5.0",
|
|
55
|
+
"openclaw": "~2026.9.3",
|
|
56
|
+
"rimraf": "~6.1.3",
|
|
57
|
+
"typescript": "~7.0.2",
|
|
58
|
+
"vitest": "~4.1.11"
|
|
59
|
+
}
|
|
60
|
+
}
|
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
ANTHROPIC_API_KEY=
|
|
2
|
+
|
|
3
|
+
# Required only when running a Qwen model (OpenRouter); harmlessly empty otherwise.
|
|
4
|
+
OPENROUTER_API_KEY=
|
|
5
|
+
|
|
6
|
+
# Agent model catalog. Comma list of full LiteLLM provider/model refs — the only
|
|
7
|
+
# place the provider/ prefix appears. `run --model <id>` and OPENCLAW_DEFAULT_TEST_MODEL
|
|
8
|
+
# use the bare id (the suffix after the last "/"). `--model all` runs every entry and
|
|
9
|
+
# needs the API key for each referenced provider.
|
|
10
|
+
OPENCLAW_TEST_MODELS=anthropic/claude-sonnet-4-6,custom-openrouter/qwen/qwen3.6-plus
|
|
11
|
+
OPENCLAW_DEFAULT_TEST_MODEL=claude-sonnet-4-6
|
|
12
|
+
|
|
13
|
+
# Required: host path to the OpenClaw workspace (mounted into the gateway).
|
|
14
|
+
OPENCLAW_WORKSPACE_DIR=
|
|
15
|
+
|
|
16
|
+
# Optional: run N matrix cells concurrently, each on its own worker stack
|
|
17
|
+
# (`run --parallel` wins over this).
|
|
18
|
+
# OPENCLAW_TEST_PARALLEL=4
|
|
19
|
+
|
|
20
|
+
# Optional overrides (defaults relative to the project dir):
|
|
21
|
+
# OPENCLAW_CONFIG_PATH=./openclaw.json
|
|
22
|
+
# OPENCLAW_TEST_SCENARIOS_DIR=./scenarios
|
|
23
|
+
# OPENCLAW_TEST_ARTIFACTS_DIR=./artifacts
|
|
24
|
+
# OPENCLAW_TEST_GATEWAY_LOGS_DIR=./.gateway-logs
|
|
25
|
+
|
|
26
|
+
# Opt-in: also write raw-stream.jsonl alongside the always-on provider-neutral
|
|
27
|
+
# trajectory log (one <sessionId>.jsonl per session under .gateway-logs/trajectory/).
|
|
28
|
+
# OPENCLAW_RAW_STREAM=1
|
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
# Consumer-side test runner image. Extends the consumer-agnostic base built locally
|
|
2
|
+
# by `openclaw-test env build` and tagged `paleo/openclaw-test-base:<pkg-version>`.
|
|
3
|
+
# The CLI passes the matching tag via the OPENCLAW_TEST_BASE_TAG build arg.
|
|
4
|
+
#
|
|
5
|
+
# Build context is the project dir.
|
|
6
|
+
|
|
7
|
+
ARG OPENCLAW_TEST_BASE_TAG=__SET_VIA_OPENCLAW_TEST_CLI__
|
|
8
|
+
FROM paleo/openclaw-test-base:${OPENCLAW_TEST_BASE_TAG}
|
|
9
|
+
|
|
10
|
+
COPY --chown=assistant:assistant package.json package-lock.json /opt/openclaw-test/src/
|
|
11
|
+
COPY --chown=assistant:assistant openclaw.json /home/assistant/.openclaw/openclaw.json
|
|
12
|
+
|
|
13
|
+
RUN npm ci --include=dev && \
|
|
14
|
+
OPENCLAW_CONFIG_PATH=/home/assistant/.openclaw/openclaw.json npx openclaw plugins registry --refresh
|
|
15
|
+
|
|
16
|
+
# Consumer customizations below. Add RUN/COPY/ENV as needed. The base image
|
|
17
|
+
# is deliberately minimal — add only what your fixtures and scenarios need.
|
|
18
|
+
# Common patterns:
|
|
19
|
+
#
|
|
20
|
+
# * System tools the fixtures invoke (under USER root):
|
|
21
|
+
# RUN apk add --no-cache git
|
|
22
|
+
# RUN npm install --global corepack@0.36.0 && \
|
|
23
|
+
# corepack enable && corepack prepare pnpm@latest --activate
|
|
24
|
+
#
|
|
25
|
+
# * Per-command mock-CLI symlinks (the base ships only the bare shim;
|
|
26
|
+
# consumers pick which commands to intercept):
|
|
27
|
+
# USER root
|
|
28
|
+
# RUN for name in claude gh; do \
|
|
29
|
+
# ln -sf mock-cli-shim "/opt/openclaw-test/mocks/bin/$name"; \
|
|
30
|
+
# done
|
|
31
|
+
# USER assistant
|
|
32
|
+
#
|
|
33
|
+
# * Helper scripts invoked from scenarios via `ctx.execInGateway(...)`:
|
|
34
|
+
# USER root
|
|
35
|
+
# COPY --chown=root:root scripts/ /opt/openclaw-test/scripts/
|
|
36
|
+
# RUN chmod +x /opt/openclaw-test/scripts/*.mjs
|
|
37
|
+
# USER assistant
|
|
38
|
+
#
|
|
39
|
+
# * Project fixtures baked into the image (typical with a named-volume
|
|
40
|
+
# /home/assistant/projects, then reset/seeded per scenario):
|
|
41
|
+
# COPY --chown=assistant:assistant projects-fixture/<name>/ /opt/<consumer>/fixtures/<name>/
|
|
42
|
+
# RUN cd /opt/<consumer>/fixtures/<name> && pnpm install --frozen-lockfile
|
|
43
|
+
#
|
|
44
|
+
# * Skills under /home/assistant/.agents/skills/ for the assistant to discover
|
|
45
|
+
# (e.g. installed with the `skills` CLI from any repo or registry):
|
|
46
|
+
# RUN npx -y skills add <skills-source> --global --yes \
|
|
47
|
+
# --agent claude-code --skill <name> [--skill <name> …]
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
# Thin consumer overlay. Pulls in the parameterized base from
|
|
2
|
+
# @alignfirst/openclaw-test. Requires Docker Compose v2.20+.
|
|
3
|
+
#
|
|
4
|
+
# `.env.local` only needs ANTHROPIC_API_KEY and OPENCLAW_WORKSPACE_DIR.
|
|
5
|
+
# All other paths default to subdirs of the project dir; override in
|
|
6
|
+
# .env.local if you need a different layout. OPENCLAW_TEST_PROJECT_DIR, OPENCLAW_TEST_PACKAGE_DIR,
|
|
7
|
+
# ASSISTANT_UID, ASSISTANT_GID are set automatically by `openclaw-test env|run`.
|
|
8
|
+
|
|
9
|
+
include:
|
|
10
|
+
- ./node_modules/@alignfirst/openclaw-test/docker-compose.yml
|