@alignfirst/openclaw-test 0.21.1 → 0.22.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +8 -8
- package/dist/env-cli.js +1 -1
- package/dist/judge.js +1 -1
- package/dist/transcript-store.js +13 -3
- package/package.json +4 -4
- package/templates/.env.local.example +2 -2
package/README.md
CHANGED
|
@@ -45,8 +45,8 @@ OPENROUTER_API_KEY=sk-or-… # Required for an OpenRouter agent or judge.
|
|
|
45
45
|
OPENCLAW_WORKSPACE_DIR=/path/to/your/openclaw-workspace
|
|
46
46
|
|
|
47
47
|
# Model catalog: full LiteLLM refs. `run --model` picks by bare id (suffix after the last "/").
|
|
48
|
-
OPENCLAW_TEST_MODELS=anthropic/claude-sonnet-
|
|
49
|
-
OPENCLAW_DEFAULT_TEST_MODEL=claude-sonnet-
|
|
48
|
+
OPENCLAW_TEST_MODELS=anthropic/claude-sonnet-5,custom-openrouter/qwen/qwen3.8-flash
|
|
49
|
+
OPENCLAW_DEFAULT_TEST_MODEL=claude-sonnet-5
|
|
50
50
|
|
|
51
51
|
```
|
|
52
52
|
|
|
@@ -84,8 +84,8 @@ npm run env:up # (optional)
|
|
|
84
84
|
npm run e2e -- --channel all <scenario> # one scenario, both channels
|
|
85
85
|
npm run e2e -- --channel all --all # every scenario, both channels
|
|
86
86
|
npm run e2e -- --channel discord-mock <scenario> # restrict to one channel
|
|
87
|
-
npm run e2e -- --channel all --model qwen3.
|
|
88
|
-
npm run e2e -- --channel all --model claude-sonnet-
|
|
87
|
+
npm run e2e -- --channel all --model qwen3.8-flash <scenario> # pick a model by bare id
|
|
88
|
+
npm run e2e -- --channel all --model claude-sonnet-5,qwen3.8-flash <s> # a comma list of bare ids
|
|
89
89
|
npm run e2e -- --channel all --model all <scenario> # run every model in OPENCLAW_TEST_MODELS
|
|
90
90
|
npm run e2e -- --channel all --iterations 5 <scenario> # repeat each (scenario, channel) pair 5×
|
|
91
91
|
npm run e2e -- --channel all --iterations 5 --max-failures 1 <s> # abort a pair after >1 failure
|
|
@@ -121,10 +121,10 @@ Assert on `conversation.id` / `threadId`, not envelope formatting.
|
|
|
121
121
|
|
|
122
122
|
## Judge model
|
|
123
123
|
|
|
124
|
-
Defaults to `anthropic/claude-haiku-4
|
|
125
|
-
the `runner` service (set in your consumer overlay).
|
|
126
|
-
`
|
|
127
|
-
`
|
|
124
|
+
Defaults to `openrouter/anthropic/claude-haiku-4.5`, billed to `OPENROUTER_API_KEY`. Override it
|
|
125
|
+
via `OPENCLAW_TEST_JUDGE_MODEL` on the `runner` service (set in your consumer overlay). OpenRouter
|
|
126
|
+
refs use `openrouter/<model>`; direct Anthropic refs use `anthropic/<model>`, for example
|
|
127
|
+
`anthropic/claude-haiku-4-5`. The judge is **not** an OpenClaw agent — don't
|
|
128
128
|
configure it in `openclaw.json`.
|
|
129
129
|
|
|
130
130
|
## Attribution
|
package/dist/env-cli.js
CHANGED
|
@@ -20,7 +20,7 @@ const RUN_USAGE = `usage: openclaw-test run --channel <id|id,id,…|all> [<scena
|
|
|
20
20
|
|
|
21
21
|
Scenario selection is required: either a positional list or --all (mutually exclusive).
|
|
22
22
|
--model <id|id,id,…|all>
|
|
23
|
-
select the agent model(s): a bare id (e.g. claude-sonnet-
|
|
23
|
+
select the agent model(s): a bare id (e.g. claude-sonnet-5), a
|
|
24
24
|
comma list of bare ids, or "all". Defaults to OPENCLAW_DEFAULT_TEST_MODEL.
|
|
25
25
|
The catalog is OPENCLAW_TEST_MODELS (.env.local), a comma list of full
|
|
26
26
|
provider/model refs; the bare id is the suffix after the last "/".
|
package/dist/judge.js
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
import Anthropic from "@anthropic-ai/sdk";
|
|
2
2
|
import OpenAI from "openai";
|
|
3
3
|
import { extractTaggedBlock } from "./parse-tagged-json.js";
|
|
4
|
-
const DEFAULT_JUDGE_MODEL = "anthropic/claude-haiku-4
|
|
4
|
+
const DEFAULT_JUDGE_MODEL = "openrouter/anthropic/claude-haiku-4.5";
|
|
5
5
|
const DEFAULT_MAX_TOKENS = 1024;
|
|
6
6
|
const RESULT_TAG = "result-json";
|
|
7
7
|
let cached;
|
package/dist/transcript-store.js
CHANGED
|
@@ -1,4 +1,5 @@
|
|
|
1
1
|
import { DatabaseSync } from "node:sqlite";
|
|
2
|
+
import { zstdDecompressSync } from "node:zlib";
|
|
2
3
|
export function readConversationSessions(dbPath, sinceMs, conversationId, threadIds) {
|
|
3
4
|
let db;
|
|
4
5
|
try {
|
|
@@ -51,19 +52,28 @@ function readNodeMessages(db, node, sinceMs) {
|
|
|
51
52
|
}
|
|
52
53
|
function readSessionMessages(db, sessionId, sinceMs) {
|
|
53
54
|
const rows = db
|
|
54
|
-
.prepare("SELECT event_json FROM transcript_events
|
|
55
|
+
.prepare("SELECT event_json, event_zstd FROM transcript_events" +
|
|
56
|
+
" WHERE session_id = ? AND created_at >= ? ORDER BY seq")
|
|
55
57
|
.all(sessionId, sinceMs);
|
|
56
58
|
const messages = [];
|
|
57
59
|
for (const row of rows) {
|
|
58
60
|
let event;
|
|
59
61
|
try {
|
|
60
|
-
event = JSON.parse(row
|
|
62
|
+
event = JSON.parse(decodeTranscriptEvent(row));
|
|
61
63
|
}
|
|
62
64
|
catch {
|
|
63
65
|
continue;
|
|
64
66
|
}
|
|
65
|
-
if (event
|
|
67
|
+
if (event?.type === "message" && event.message !== undefined)
|
|
66
68
|
messages.push(event.message);
|
|
67
69
|
}
|
|
68
70
|
return messages;
|
|
69
71
|
}
|
|
72
|
+
/** Since OpenClaw 2026.9.6, an event of 1 KiB or more is stored zstd-compressed, `event_json` NULL. */
|
|
73
|
+
function decodeTranscriptEvent(row) {
|
|
74
|
+
if (row.event_json !== null)
|
|
75
|
+
return row.event_json;
|
|
76
|
+
if (row.event_zstd === null)
|
|
77
|
+
throw new Error("transcript event without a payload");
|
|
78
|
+
return zstdDecompressSync(row.event_zstd).toString("utf8");
|
|
79
|
+
}
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@alignfirst/openclaw-test",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.22.0",
|
|
4
4
|
"description": "Dockerised regression-test framework for OpenClaw workspaces: bus, scenario driver, judge, Compose stack.",
|
|
5
5
|
"keywords": [
|
|
6
6
|
"openclaw",
|
|
@@ -44,9 +44,9 @@
|
|
|
44
44
|
"openclaw": "*"
|
|
45
45
|
},
|
|
46
46
|
"dependencies": {
|
|
47
|
-
"@alignfirst/openclaw-channel-mock-core": "0.10.
|
|
48
|
-
"@alignfirst/openclaw-discord-mock": "0.6.
|
|
49
|
-
"@alignfirst/openclaw-slack-mock": "0.6.
|
|
47
|
+
"@alignfirst/openclaw-channel-mock-core": "0.10.2",
|
|
48
|
+
"@alignfirst/openclaw-discord-mock": "0.6.2",
|
|
49
|
+
"@alignfirst/openclaw-slack-mock": "0.6.2",
|
|
50
50
|
"@anthropic-ai/sdk": "~0.127.0",
|
|
51
51
|
"openai": "~7.8.0"
|
|
52
52
|
},
|
|
@@ -7,8 +7,8 @@ OPENROUTER_API_KEY=
|
|
|
7
7
|
# place the provider/ prefix appears. `run --model <id>` and OPENCLAW_DEFAULT_TEST_MODEL
|
|
8
8
|
# use the bare id (the suffix after the last "/"). `--model all` runs every entry and
|
|
9
9
|
# needs the API key for each referenced provider.
|
|
10
|
-
OPENCLAW_TEST_MODELS=anthropic/claude-sonnet-
|
|
11
|
-
OPENCLAW_DEFAULT_TEST_MODEL=claude-sonnet-
|
|
10
|
+
OPENCLAW_TEST_MODELS=anthropic/claude-sonnet-5,custom-openrouter/qwen/qwen3.8-flash
|
|
11
|
+
OPENCLAW_DEFAULT_TEST_MODEL=claude-sonnet-5
|
|
12
12
|
|
|
13
13
|
# Required: host path to the OpenClaw workspace (mounted into the gateway).
|
|
14
14
|
OPENCLAW_WORKSPACE_DIR=
|