@alignfirst/openclaw-test 0.20.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (53) hide show
  1. package/Dockerfile.base +45 -0
  2. package/LICENSE +21 -0
  3. package/README.md +132 -0
  4. package/bin/cli.mjs +7 -0
  5. package/dist/bus.d.ts +1 -0
  6. package/dist/bus.js +20 -0
  7. package/dist/cell-result.d.ts +25 -0
  8. package/dist/cell-result.js +32 -0
  9. package/dist/cli.d.ts +5 -0
  10. package/dist/cli.js +110 -0
  11. package/dist/context.d.ts +220 -0
  12. package/dist/context.js +479 -0
  13. package/dist/cost.d.ts +2 -0
  14. package/dist/cost.js +19 -0
  15. package/dist/env-cli.d.ts +5 -0
  16. package/dist/env-cli.js +645 -0
  17. package/dist/exec-rpc.d.ts +13 -0
  18. package/dist/exec-rpc.js +46 -0
  19. package/dist/index.d.ts +3 -0
  20. package/dist/index.js +1 -0
  21. package/dist/judge.d.ts +47 -0
  22. package/dist/judge.js +183 -0
  23. package/dist/loop.d.ts +62 -0
  24. package/dist/loop.js +316 -0
  25. package/dist/mock-cli-server.d.ts +40 -0
  26. package/dist/mock-cli-server.js +194 -0
  27. package/dist/mock-cli-shim.d.ts +7 -0
  28. package/dist/mock-cli-shim.js +89 -0
  29. package/dist/models.d.ts +23 -0
  30. package/dist/models.js +78 -0
  31. package/dist/parse-tagged-json.d.ts +1 -0
  32. package/dist/parse-tagged-json.js +16 -0
  33. package/dist/report.d.ts +215 -0
  34. package/dist/report.js +19 -0
  35. package/dist/runner-args.d.ts +11 -0
  36. package/dist/runner-args.js +84 -0
  37. package/dist/runner.d.ts +2 -0
  38. package/dist/runner.js +371 -0
  39. package/dist/summary.d.ts +3 -0
  40. package/dist/summary.js +64 -0
  41. package/dist/transcript-dump.d.ts +1 -0
  42. package/dist/transcript-dump.js +50 -0
  43. package/dist/transcript-log.d.ts +67 -0
  44. package/dist/transcript-log.js +199 -0
  45. package/dist/transcript-store.d.ts +6 -0
  46. package/dist/transcript-store.js +69 -0
  47. package/docker-compose.yml +83 -0
  48. package/exec-watcher.mjs +155 -0
  49. package/package.json +60 -0
  50. package/templates/.env.local.example +28 -0
  51. package/templates/Dockerfile +47 -0
  52. package/templates/docker-compose.yml +10 -0
  53. package/templates/openclaw.json +40 -0
@@ -0,0 +1,45 @@
1
+ # Consumer-agnostic base for the openclaw-test runner stack. Built locally by the CLI
2
+ # (`openclaw-test env build`) and tagged `paleo/openclaw-test-base:<pkg-version>`.
3
+ # Ships: Node + `shadow` (for the UID/GID dance), the mock-cli shim binary (no
4
+ # per-command symlinks — consumers add their own), and the exec RPC watcher.
5
+ # The consumer's project `Dockerfile` does `FROM paleo/openclaw-test-base:${OPENCLAW_TEST_BASE_TAG}`
6
+ # and adds whatever else the fixture needs (e.g. git, pnpm via corepack, mock symlinks).
7
+ # Build context: this package's directory (so `dist/mock-cli-shim.js` is COPY-able).
8
+
9
+ FROM node:26-alpine
10
+
11
+ RUN apk add --no-cache shadow
12
+
13
+ ARG ASSISTANT_UID=1000
14
+ ARG ASSISTANT_GID=1000
15
+ RUN (getent passwd ${ASSISTANT_UID} && userdel -r "$(getent passwd ${ASSISTANT_UID} | cut -d: -f1)" || true) && \
16
+ (getent group ${ASSISTANT_GID} && groupdel "$(getent group ${ASSISTANT_GID} | cut -d: -f1)" || true) && \
17
+ groupadd --gid ${ASSISTANT_GID} assistant && \
18
+ useradd --create-home --shell /bin/sh --uid ${ASSISTANT_UID} --gid ${ASSISTANT_GID} assistant && \
19
+ mkdir -p /home/assistant/.openclaw /opt/openclaw-test/src /opt/openclaw-test/artifacts && \
20
+ chown -R assistant:assistant /home/assistant /opt/openclaw-test/src /opt/openclaw-test/artifacts
21
+
22
+ # Mock-CLI shim under /opt/openclaw-test/mocks. OpenClaw's exec tool runs commands via
23
+ # `/bin/sh -lc`, which sources /etc/profile; Alpine's default resets PATH to a
24
+ # "safe" set that excludes /opt/openclaw-test/mocks/bin, so we override it. Consumers
25
+ # create the per-command symlinks (e.g. claude, gh) in their own Dockerfile.
26
+ COPY --chown=assistant:assistant dist/mock-cli-shim.js /opt/openclaw-test/mocks/dist/mock-cli-shim.js
27
+ RUN mkdir -p /opt/openclaw-test/mocks/bin && \
28
+ printf '#!/bin/sh\nexec node /opt/openclaw-test/mocks/dist/mock-cli-shim.js "$0" "$@"\n' > /opt/openclaw-test/mocks/bin/mock-cli-shim && \
29
+ chmod +x /opt/openclaw-test/mocks/bin/mock-cli-shim && \
30
+ chown -R assistant:assistant /opt/openclaw-test/mocks && \
31
+ printf 'export PATH="/opt/openclaw-test/mocks/bin:/usr/local/sbin:/usr/local/bin:/usr/sbin:/usr/bin:/sbin:/bin"\n' > /etc/profile
32
+
33
+ # Exec RPC watcher. The gateway service spawns it in the background; the runner
34
+ # pokes requests onto the shared `openclaw-test-ipc` volume via `ctx.execInGateway(...)`.
35
+ # Pre-create the mountpoint with `assistant` ownership so a fresh named volume
36
+ # inherits it — otherwise Docker creates it as root:root and the assistant-uid
37
+ # processes get EACCES on the first write.
38
+ COPY --chown=root:root exec-watcher.mjs /usr/local/bin/exec-watcher
39
+ RUN chmod +x /usr/local/bin/exec-watcher && \
40
+ mkdir -p /var/run/openclaw-test-ipc && chown assistant:assistant /var/run/openclaw-test-ipc
41
+
42
+ USER assistant
43
+ WORKDIR /opt/openclaw-test/src
44
+
45
+ CMD ["sleep", "infinity"]
package/LICENSE ADDED
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Paleo
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
package/README.md ADDED
@@ -0,0 +1,132 @@
1
+ # @alignfirst/openclaw-test
2
+
3
+ Dockerised regression-test harness for OpenClaw workspaces. Drives the agent through two synthetic channels (`discord-mock`, `slack-mock`) and asserts the results.
4
+
5
+ Pair with [`@alignfirst/openclaw-channel-mock-core`](https://www.npmjs.com/package/@alignfirst/openclaw-channel-mock-core), [`@alignfirst/openclaw-discord-mock`](https://www.npmjs.com/package/@alignfirst/openclaw-discord-mock), [`@alignfirst/openclaw-slack-mock`](https://www.npmjs.com/package/@alignfirst/openclaw-slack-mock).
6
+
7
+ For internals (topology, Dockerfile pair, mocked-CLI shim, channel plugin mechanics, OpenClaw quirks), see [openclaw-test-architecture.md](https://github.com/paleo/alignfirst/blob/main/docs/alignfirst-dev-kit/openclaw-test-architecture.md).
8
+
9
+ ## Install
10
+
11
+ ```sh
12
+ npm i -D @alignfirst/openclaw-test @alignfirst/openclaw-channel-mock-core @alignfirst/openclaw-discord-mock @alignfirst/openclaw-slack-mock openclaw
13
+ ```
14
+
15
+ Requires Docker Compose.
16
+
17
+ ## Init
18
+
19
+ ```sh
20
+ npx @alignfirst/openclaw-test init <project-dir>
21
+ ```
22
+
23
+ Adds the four `package.json` scripts (`env:build`, `env:up`, `env:down`, `e2e`) if missing, and drops four files:
24
+
25
+ - `openclaw.json` — gateway config (mode `local`, both channel plugins enabled, main agent placeholder).
26
+ - `.env.local.example` — copy to `.env.local` and fill in; its comments document every variable.
27
+ - `docker-compose.yml` — thin overlay that `include:`s the base stack from `node_modules/`.
28
+ - `Dockerfile` — consumer-owned; its comments document the common customizations (system tools, mock-CLI symlinks, fixtures, reset scripts).
29
+
30
+ ## Configure
31
+
32
+ Edit `openclaw.json`:
33
+
34
+ - `agents.entries.main.model` — default `provider/model` ref; `run --model` overrides it per run.
35
+ - `agents.entries.main.workspace` — host path to your OpenClaw workspace. Field name is **`workspace`**, not `workspaceDir`.
36
+ - `channels.slack-mock.replyToMode` — `"all"` (default) routes eligible roots and replies through
37
+ one thread session; `"off"` leaves root turns in the channel session and threads only explicit
38
+ replies. `blockStreaming: true` keeps streamed replies as one bus message.
39
+
40
+ ## Env vars (`.env.local`)
41
+
42
+ ```sh
43
+ ANTHROPIC_API_KEY=sk-ant-… # Required for an Anthropic agent or judge.
44
+ OPENROUTER_API_KEY=sk-or-… # Required for an OpenRouter agent or judge.
45
+ OPENCLAW_WORKSPACE_DIR=/path/to/your/openclaw-workspace
46
+
47
+ # Model catalog: full LiteLLM refs. `run --model` picks by bare id (suffix after the last "/").
48
+ OPENCLAW_TEST_MODELS=anthropic/claude-sonnet-4-6,custom-openrouter/qwen/qwen3.6-plus
49
+ OPENCLAW_DEFAULT_TEST_MODEL=claude-sonnet-4-6
50
+
51
+ ```
52
+
53
+ See `.env.local.example` for the optional overrides (paths, raw stream log).
54
+
55
+ ## Scenarios
56
+
57
+ Drop scenarios under `scenarios/<id>.ts`: each default-exports `async (ctx: ScenarioContext) => void`. They are loaded by Node's built-in TypeScript stripping: no `enum`, `namespace`, decorators, ctor parameter properties. Shared helpers must go under subdirectories (e.g. `scenarios/_lib/`).
58
+
59
+ Project fixtures and their reset logic are consumer concerns — ship a reset script in your consumer image and invoke it via `ctx.execInGateway(...)`.
60
+
61
+ `ScenarioContext` primitives (authoritative types: `src/context.ts`):
62
+
63
+ - `channel`, `conversationId`, `accountId` — per-task isolation; never hard-code a conversation id.
64
+ - `busUrl` — the bus the gateway talks to, for direct bus calls such as `failNextQaBusOperation`.
65
+ - `sendInbound(input)` — push an inbound message on the bus.
66
+ - `createThread(input)` — seed a thread the bot did not open, as a surface user creating one; returns its id for `sendInbound`.
67
+ - `waitForOutbound(predicate, opts)` — await a matching outbound; fails fast on unmatched outbounds or mock-CLI silence.
68
+ - `poll`, `expectNoOutbound`, `getCursor` — bus consumers.
69
+ - `assertRegex`, `assertEqual`, `assertLength` — structural assertions.
70
+ - `judgeLLM({ message, rubric, label, attachTo? })` — LLM judgement, bound to an action entry.
71
+ - `mockCli(name, handler)` — intercept the gateway's CLI calls (`git`, `claude`, …); unregistered calls fail the scenario.
72
+ - `execInGateway(argv, opts)` — run a command inside the gateway container.
73
+ - `log(...)` — scenario log entry, free-standing or attached to an action.
74
+
75
+ Prefer structural assertions over `judgeLLM`; reserve the judge for free-form content claims.
76
+
77
+ Examples: [alignfirst-dev-kit-tests/scenarios](https://github.com/paleo/alignfirst/tree/main/alignfirst-dev-kit-tests/scenarios).
78
+
79
+ ## Run
80
+
81
+ ```sh
82
+ npm run env:build # build base + consumer image
83
+ npm run env:up # (optional) keep bus + gateway warm across iterative runs
84
+ npm run e2e -- --channel all <scenario> # one scenario, both channels
85
+ npm run e2e -- --channel all --all # every scenario, both channels
86
+ npm run e2e -- --channel discord-mock <scenario> # restrict to one channel
87
+ npm run e2e -- --channel all --model qwen3.6-plus <scenario> # pick a model by bare id
88
+ npm run e2e -- --channel all --model claude-sonnet-4-6,qwen3.6-plus <s> # a comma list of bare ids
89
+ npm run e2e -- --channel all --model all <scenario> # run every model in OPENCLAW_TEST_MODELS
90
+ npm run e2e -- --channel all --iterations 5 <scenario> # repeat each (scenario, channel) pair 5×
91
+ npm run e2e -- --channel all --iterations 5 --max-failures 1 <s> # abort a pair after >1 failure
92
+ npm run e2e -- --channel all --iterations 10 --parallel 4 <s> # run up to 4 cells concurrently
93
+ npm run e2e -- --channel discord-mock --reuse-stack <s> # skip per-cell bus+gateway recreation
94
+ npm run env:down # tear down all worker stacks
95
+ ```
96
+
97
+ `run` auto-starts the worker stacks it needs and tears down the ones it started; an explicit `env:up` beforehand keeps them warm across runs.
98
+
99
+ Rebuild (`npm run env:build`) after editing `openclaw.json` or the `Dockerfile`, or after bumping any `@alignfirst/openclaw-*` dependency.
100
+
101
+ Cells run serially per worker stack. Exit 0 iff every pair passes. Artifacts land under `artifacts/<runStamp>/` — see the [architecture doc](https://github.com/paleo/alignfirst/blob/main/docs/alignfirst-dev-kit/openclaw-test-architecture.md).
102
+
103
+ ### Parallel runs
104
+
105
+ `--parallel K` (default `OPENCLAW_TEST_PARALLEL` from `.env.local`, fallback 1; the flag wins) runs up to K cells concurrently, each on its own worker Compose stack — project `<project>-w<i>`, all sharing one image. Iterations of one pair parallelize too, so flakiness measurements (`--iterations N`) are the primary win. With K > 1, per-cell output is captured to `artifacts/<runStamp>/cells/<leaf>.log` and the console shows one compact line per cell event.
106
+
107
+ Each worker gets its own gateway logs dir (`.gateway-logs/w<i>/`) and a private workspace copy under `.workers/`, refreshed from `OPENCLAW_WORKSPACE_DIR` before every cell. Add `.workers/` to your `.gitignore` and `.dockerignore`.
108
+
109
+ `env up -- --parallel K` pre-warms K workers. `env down` takes no flag: it discovers every `<project>-w<N>` stack and tears them all down, including orphans from a crashed run.
110
+
111
+ **Upgrading from a pre-parallel version:** stacks now run under per-worker Compose project names, so the old un-suffixed project is orphaned. Tear it down once with `docker compose down` from the project dir.
112
+
113
+ ## Channels
114
+
115
+ - `discord-mock` — full Discord-shaped surface; no auto-thread.
116
+ - `slack-mock` — Slack-shaped surface with `send`, `react`, `read`, `edit`, `delete`, `reactions`,
117
+ and `search`. It supports `replyToMode: "off" | "all"`; fake thread creation/rename actions stay
118
+ disabled.
119
+
120
+ Assert on `conversation.id` / `threadId`, not envelope formatting.
121
+
122
+ ## Judge model
123
+
124
+ Defaults to `anthropic/claude-haiku-4-5`. Override it via `OPENCLAW_TEST_JUDGE_MODEL` on
125
+ the `runner` service (set in your consumer overlay). Direct Anthropic refs use
126
+ `anthropic/<model>`; OpenRouter refs use `openrouter/<model>`, for example
127
+ `openrouter/anthropic/claude-haiku-4.5`. The judge is **not** an OpenClaw agent — don't
128
+ configure it in `openclaw.json`.
129
+
130
+ ## Attribution
131
+
132
+ The runner package contains no upstream-adapted code. See sibling packages' `NOTICE.md` for OpenClaw attribution covering the channel plugins.
package/bin/cli.mjs ADDED
@@ -0,0 +1,7 @@
1
+ #!/usr/bin/env node
2
+ import { dispatch } from "../dist/cli.js";
3
+
4
+ dispatch(process.argv.slice(2)).catch((err) => {
5
+ console.error("openclaw-test crash:", err);
6
+ process.exit(1);
7
+ });
package/dist/bus.d.ts ADDED
@@ -0,0 +1 @@
1
+ export declare function startBus(): void;
package/dist/bus.js ADDED
@@ -0,0 +1,20 @@
1
+ import { createServer } from "node:http";
2
+ import { createBus } from "@alignfirst/openclaw-channel-mock-core";
3
+ const PORT = 43123;
4
+ const HOST = "0.0.0.0";
5
+ export function startBus() {
6
+ const { handler } = createBus();
7
+ const server = createServer(async (req, res) => {
8
+ const handled = await handler(req, res);
9
+ if (!handled) {
10
+ res.statusCode = 404;
11
+ res.end("not found");
12
+ }
13
+ });
14
+ server.listen(PORT, HOST, () => {
15
+ console.log(`channel-mock bus listening on ${HOST}:${PORT}`);
16
+ });
17
+ }
18
+ if (import.meta.url === `file://${process.argv[1]}`) {
19
+ startBus();
20
+ }
@@ -0,0 +1,25 @@
1
+ import type { JudgeUsage } from "./judge.js";
2
+ export interface CellResult {
3
+ schemaVersion: 3;
4
+ scenarioId: string;
5
+ channel: string;
6
+ model: string;
7
+ iterationIndex: number;
8
+ verdict: "pass" | "fail";
9
+ durationMs: number;
10
+ conversationId: string;
11
+ artifactDirName: string;
12
+ agentCostUsd: number;
13
+ agentTurns: number;
14
+ judgeUsd: number;
15
+ judgeUsages: JudgeUsage[];
16
+ }
17
+ export declare function cellLeafName(parts: {
18
+ scenarioId: string;
19
+ modelId: string;
20
+ channel: string;
21
+ iterationIndex: number;
22
+ iterationWidth: number;
23
+ }): string;
24
+ export declare function readCellResult(path: string): CellResult | undefined;
25
+ export declare function writeCellResult(path: string, r: CellResult): void;
@@ -0,0 +1,32 @@
1
+ import { readFileSync, writeFileSync } from "node:fs";
2
+ export function cellLeafName(parts) {
3
+ const iterSuffix = parts.iterationWidth > 0
4
+ ? `-#${String(parts.iterationIndex).padStart(parts.iterationWidth, "0")}`
5
+ : "";
6
+ return `${parts.modelId}-${parts.scenarioId}-${parts.channel}${iterSuffix}`;
7
+ }
8
+ export function readCellResult(path) {
9
+ let raw;
10
+ try {
11
+ raw = readFileSync(path, "utf8");
12
+ }
13
+ catch {
14
+ return undefined;
15
+ }
16
+ let parsed;
17
+ try {
18
+ parsed = JSON.parse(raw);
19
+ }
20
+ catch {
21
+ return undefined;
22
+ }
23
+ if (!parsed || typeof parsed !== "object")
24
+ return undefined;
25
+ const r = parsed;
26
+ if (r.schemaVersion !== 3)
27
+ return undefined;
28
+ return parsed;
29
+ }
30
+ export function writeCellResult(path, r) {
31
+ writeFileSync(path, JSON.stringify(r, null, 2));
32
+ }
package/dist/cli.d.ts ADDED
@@ -0,0 +1,5 @@
1
+ export declare function dispatch(argv: string[]): Promise<void>;
2
+ export interface PackageJsonScripts {
3
+ scripts?: Record<string, string>;
4
+ }
5
+ export declare function addInitScripts(pkg: PackageJsonScripts): string[];
package/dist/cli.js ADDED
@@ -0,0 +1,110 @@
1
+ import { copyFile, mkdir, readdir, readFile, writeFile } from "node:fs/promises";
2
+ import { dirname, join, resolve } from "node:path";
3
+ import { fileURLToPath } from "node:url";
4
+ import { startBus } from "./bus.js";
5
+ import { envCommand, runCommand } from "./env-cli.js";
6
+ import { main as runnerMain } from "./runner.js";
7
+ // `dist/cli.js` ships under `<package>/dist/`; one level up from its dir is the package root.
8
+ const PACKAGE_DIR = resolve(dirname(fileURLToPath(import.meta.url)), "..");
9
+ const INIT_SCRIPTS = {
10
+ "env:build": "openclaw-test env build",
11
+ "env:up": "openclaw-test env up",
12
+ "env:down": "openclaw-test env down",
13
+ e2e: "openclaw-test run",
14
+ };
15
+ export async function dispatch(argv) {
16
+ const [cmd, ...rest] = argv;
17
+ switch (cmd) {
18
+ case "init":
19
+ await initCommand(rest[0]);
20
+ return;
21
+ case "env":
22
+ await envCommand(PACKAGE_DIR, rest);
23
+ return;
24
+ case "run":
25
+ await runCommand(PACKAGE_DIR, rest);
26
+ return;
27
+ case "bus":
28
+ startBus();
29
+ return;
30
+ case "runner":
31
+ await runnerMain(rest);
32
+ return;
33
+ default:
34
+ usage();
35
+ }
36
+ }
37
+ async function initCommand(targetDirRaw) {
38
+ if (!targetDirRaw)
39
+ usage();
40
+ const target = resolve(process.cwd(), targetDirRaw);
41
+ const templatesDir = join(PACKAGE_DIR, "templates");
42
+ await mkdir(target, { recursive: true });
43
+ const entries = await readdir(templatesDir);
44
+ for (const name of entries) {
45
+ await copyFile(join(templatesDir, name), join(target, name));
46
+ console.log(`copied ${name} -> ${join(target, name)}`);
47
+ }
48
+ await addScriptsToPackageJson(target);
49
+ printNextSteps();
50
+ }
51
+ async function addScriptsToPackageJson(target) {
52
+ const pkgPath = join(target, "package.json");
53
+ let raw;
54
+ try {
55
+ raw = await readFile(pkgPath, "utf8");
56
+ }
57
+ catch {
58
+ console.log("no package.json found — run `npm init`, then add the scripts manually (see README)");
59
+ return;
60
+ }
61
+ let pkg;
62
+ try {
63
+ pkg = JSON.parse(raw);
64
+ }
65
+ catch {
66
+ console.error(`${pkgPath} is not valid JSON — fix it, then add the scripts manually (see README)`);
67
+ return;
68
+ }
69
+ const added = addInitScripts(pkg);
70
+ if (added.length === 0) {
71
+ console.log("package.json scripts already present");
72
+ return;
73
+ }
74
+ await writeFile(pkgPath, `${JSON.stringify(pkg, undefined, 2)}\n`);
75
+ console.log(`added scripts to package.json: ${added.join(", ")}`);
76
+ }
77
+ export function addInitScripts(pkg) {
78
+ pkg.scripts ??= {};
79
+ const scripts = pkg.scripts;
80
+ const added = [];
81
+ for (const [name, command] of Object.entries(INIT_SCRIPTS)) {
82
+ if (scripts[name] !== undefined)
83
+ continue;
84
+ scripts[name] = command;
85
+ added.push(name);
86
+ }
87
+ return added;
88
+ }
89
+ function printNextSteps() {
90
+ console.log("\nNext steps:\n" +
91
+ " 1. npm i -D @alignfirst/openclaw-test @alignfirst/openclaw-channel-mock-core" +
92
+ " @alignfirst/openclaw-discord-mock @alignfirst/openclaw-slack-mock openclaw\n" +
93
+ " 2. cp .env.local.example .env.local # then fill it in\n" +
94
+ " 3. npm run env:build");
95
+ }
96
+ function usage() {
97
+ console.error("usage: openclaw-test <init|env|run|bus|runner> [args]\n\n" +
98
+ " init <target-dir> copy templates into target dir\n" +
99
+ " env <build|up|down> drive the Compose stack (host-side)\n" +
100
+ " run [flags] [...] run scenarios against the stack (host-side)\n" +
101
+ " bus start the bus HTTP server (inside container)\n" +
102
+ " runner [flags] [...] execute scenarios (inside container)\n");
103
+ process.exit(1);
104
+ }
105
+ if (import.meta.url === `file://${process.argv[1]}`) {
106
+ dispatch(process.argv.slice(2)).catch((err) => {
107
+ console.error("openclaw-test crash:", err);
108
+ process.exit(1);
109
+ });
110
+ }
@@ -0,0 +1,220 @@
1
+ import { type QaBusConversation, type QaBusMessage } from "@alignfirst/openclaw-channel-mock-core";
2
+ import { type ExecInGatewayOptions, type ExecInGatewayResult } from "./exec-rpc.js";
3
+ import { type JudgeUsage, type JudgeVerdict, type JudgeVerdictJson, type JudgeVerdictRaw } from "./judge.js";
4
+ import type { ActionEntry, AgentToolCall, ChannelId, CliMockCall, CliMockEntry, CliMockHandler, EmitSink, InboundSentEntry, OutboundReceivedEntry, ReportEntry, ScenarioFailure, ScenarioResult } from "./report.js";
5
+ export type { ChannelId } from "./report.js";
6
+ export type Conversation = QaBusConversation;
7
+ export type BusMessage = QaBusMessage;
8
+ export interface PollResult {
9
+ messages: BusMessage[];
10
+ nextCursor: number;
11
+ }
12
+ export interface SendInboundResult {
13
+ message: BusMessage;
14
+ entry: InboundSentEntry;
15
+ }
16
+ export interface WaitForOutboundResult {
17
+ match: BusMessage;
18
+ entry: OutboundReceivedEntry;
19
+ nextCursor: number;
20
+ }
21
+ export type MockCliRegisterMode = "register" | "replace" | "ifAbsent";
22
+ export interface MockCliRegisterOptions {
23
+ /**
24
+ * - `"register"` (default): throws if a handler is already registered for `name`.
25
+ * - `"replace"`: overwrites the existing handler; throws if none was registered.
26
+ * - `"ifAbsent"`: no-op if a handler is already registered.
27
+ */
28
+ mode?: MockCliRegisterMode;
29
+ }
30
+ export interface ScenarioContext {
31
+ channel: ChannelId;
32
+ conversationId: string;
33
+ accountId: ChannelId;
34
+ /** The bus the gateway's channel plugins talk to; for direct bus calls such as fault injection. */
35
+ busUrl: string;
36
+ /**
37
+ * The most recent agent-action entry (`outboundReceived` / `cliMock` /
38
+ * `agentToolCall`). Capture this synchronously after `await` resolves to
39
+ * pin a handle to the action before any further awaits could overwrite it.
40
+ * `inboundSent` is scenario-emitted and does not update this property.
41
+ */
42
+ readonly currentEntry: ActionEntry | undefined;
43
+ /**
44
+ * `true` once the scenario has called `markScenarioAsEnded`. After that, the
45
+ * mock-cli server stops dispatching to registered handlers — any incoming
46
+ * call (typically a lingering agent action firing after the verdict is in)
47
+ * is answered with a "scenario ended" stub message and recorded as a
48
+ * post-end cliMock entry that does **not** affect the result.
49
+ */
50
+ readonly isScenarioEnded: boolean;
51
+ /**
52
+ * Declare the scenario's verdict signals are all in. From this point on
53
+ * the runner stops attributing tool-call failures to this scenario.
54
+ * Optional `reason` is recorded in the scenarioLog (e.g. `"PASS"`).
55
+ */
56
+ markScenarioAsEnded(reason?: string): void;
57
+ log(message: string): void;
58
+ log(opts: {
59
+ attachTo: ActionEntry;
60
+ label?: string;
61
+ extra?: unknown;
62
+ }): void;
63
+ sendInbound(input: SendInboundInput): Promise<SendInboundResult>;
64
+ /**
65
+ * Seed a thread the bot did not open, as a human creating one on the surface. Returns its id,
66
+ * to be passed as `sendInbound`'s `threadId`.
67
+ */
68
+ createThread(input: CreateThreadInput): Promise<string>;
69
+ poll(opts: {
70
+ sinceCursor: number;
71
+ timeoutMs?: number;
72
+ }): Promise<PollResult>;
73
+ waitForOutbound(predicate: (m: BusMessage) => boolean, opts: WaitForOutboundOptions): Promise<WaitForOutboundResult>;
74
+ expectNoOutbound(predicate: (m: BusMessage) => boolean, opts: {
75
+ withinMs: number;
76
+ sinceCursor: number;
77
+ }): Promise<{
78
+ nextCursor: number;
79
+ }>;
80
+ assertRegex(actual: string, pattern: RegExp, label: string): void;
81
+ assertEqual<T>(actual: T, expected: T, label: string): void;
82
+ assertLength(value: {
83
+ length: number;
84
+ } | string | unknown[], expected: number, label: string): void;
85
+ /**
86
+ * Anthropic-direct judgement. The verdict is recorded as an `AssertionRecord`
87
+ * on the `attachTo` action entry — usually the entry returned by
88
+ * `waitForOutbound` / `sendInbound`, or a snapshot of `ctx.currentEntry`
89
+ * taken right after the relevant `await` resolves.
90
+ */
91
+ judgeLLM(p: {
92
+ attachTo: ActionEntry;
93
+ message: string;
94
+ rubric: string;
95
+ label: string;
96
+ maxTokens?: number;
97
+ }): Promise<JudgeVerdict>;
98
+ judgeLLMJson<T>(p: {
99
+ message: string;
100
+ prompt: string;
101
+ returnType: string;
102
+ label: string;
103
+ maxTokens?: number;
104
+ }): Promise<JudgeVerdictJson<T>>;
105
+ judgeLLMRaw(p: {
106
+ prompt: string;
107
+ label: string;
108
+ maxTokens?: number;
109
+ }): Promise<JudgeVerdictRaw>;
110
+ getCursor(): Promise<number>;
111
+ mockCli(name: string, handler: CliMockHandler, opts?: MockCliRegisterOptions): void;
112
+ /**
113
+ * Execute an arbitrary command inside the gateway container via the exec
114
+ * watcher RPC. Always resolves once the wrapped command finishes (with its
115
+ * exit code, stdout, and stderr — non-zero exits do NOT throw). Throws only
116
+ * on transport failure or hard timeout (`timeoutMs + 5s` headroom for the
117
+ * watcher to record a kill before this side gives up).
118
+ */
119
+ execInGateway(argv: string[], opts?: ExecInGatewayOptions): Promise<ExecInGatewayResult>;
120
+ /**
121
+ * Poll the session transcripts until an agent tool call matches `predicate`,
122
+ * then record a passing `AssertionRecord` on the current entry and return the
123
+ * call. On timeout, record a failing assertion and throw (hard-fail) — use it
124
+ * to assert the agent took a specific action (read a file, ran a command).
125
+ * Aggregates across all the conversation's sessions, so it sees thread and
126
+ * subagent tool calls; the transcript is appended per message, so calls from
127
+ * a turn still in flight are already visible.
128
+ */
129
+ waitForAgentToolCall(predicate: (call: AgentToolCall) => boolean, opts: WaitForAgentToolCallOptions): Promise<AgentToolCall>;
130
+ /**
131
+ * One-shot parse of the conversation's agent tool calls from the session
132
+ * transcripts — no waiting, no assertion recorded.
133
+ */
134
+ getAgentToolCalls(): Promise<AgentToolCall[]>;
135
+ }
136
+ export interface WaitForAgentToolCallOptions {
137
+ /** Assertion label recorded on the current entry. */
138
+ label: string;
139
+ /** Default 30_000. */
140
+ timeoutMs?: number;
141
+ /** Default 500. */
142
+ pollMs?: number;
143
+ }
144
+ export interface WaitForOutboundOptions {
145
+ sinceCursor: number;
146
+ timeoutMs?: number;
147
+ failFastUnmatchedOutbounds?: number | false;
148
+ /**
149
+ * Liveness fail-fast: throw when a mocked CLI invoked *during this wait* is
150
+ * followed by neither a CLI call nor any outbound for this long (default
151
+ * 60s). CLI calls from before the wait never arm it; any outbound disarms it
152
+ * until the next CLI call. `false` disables it — for waits where long
153
+ * non-CLI agent work (exec calls, delegation) legitimately follows a CLI
154
+ * call with no post in between.
155
+ */
156
+ failFastCliMockGraceMs?: number | false;
157
+ }
158
+ export interface ScenarioInternals {
159
+ finalize(opts?: {
160
+ failure?: ScenarioFailure;
161
+ }): {
162
+ entries: ReportEntry[];
163
+ judgeUsages: JudgeUsage[];
164
+ result: ScenarioResult;
165
+ };
166
+ emitOutboundReceived(m: BusMessage): void;
167
+ emitCliMock(call: CliMockCall): void;
168
+ getMockHandlers(): Map<string, CliMockHandler>;
169
+ peekEntries(): {
170
+ entries: ReportEntry[];
171
+ };
172
+ isScenarioEnded(): boolean;
173
+ }
174
+ export interface SendInboundInput {
175
+ senderId: string;
176
+ senderName?: string;
177
+ text: string;
178
+ threadId?: string;
179
+ /** Thread name carried by the message, as a surface reports it. */
180
+ threadTitle?: string;
181
+ conversation?: Conversation;
182
+ }
183
+ export interface CreateThreadInput {
184
+ title: string;
185
+ /** The surface user who opened it. Defaults to the bot, so a human thread must name one. */
186
+ createdBy?: string;
187
+ }
188
+ export declare class AssertionError extends Error {
189
+ constructor(msg: string);
190
+ }
191
+ export declare function createContext(params: {
192
+ channel: ChannelId;
193
+ conversationId: string;
194
+ /** Scenario start time; bounds the transcript window for `waitForAgentToolCall`. */
195
+ startedAtIso?: string;
196
+ emitSink?: EmitSink;
197
+ }): {
198
+ ctx: ScenarioContext;
199
+ internals: ScenarioInternals;
200
+ };
201
+ export interface WaitForOutboundDeps {
202
+ accountId: ChannelId;
203
+ awaitEntry: (id: string) => Promise<OutboundReceivedEntry>;
204
+ getLastCliMock: () => {
205
+ atMs: number;
206
+ entry: CliMockEntry;
207
+ } | undefined;
208
+ }
209
+ export interface WaitForOutboundOpts {
210
+ sinceCursor: number;
211
+ timeoutMs?: number;
212
+ failFastUnmatchedOutbounds?: number | false;
213
+ failFastCliMockGraceMs?: number | false;
214
+ pollImpl?: (accountId: ChannelId, opts: {
215
+ sinceCursor: number;
216
+ timeoutMs?: number;
217
+ }) => Promise<PollResult>;
218
+ nowImpl?: () => number;
219
+ }
220
+ export declare function waitForOutbound(deps: WaitForOutboundDeps, predicate: (m: BusMessage) => boolean, opts: WaitForOutboundOpts): Promise<WaitForOutboundResult>;