@alignfirst/openclaw-test 0.20.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/Dockerfile.base +45 -0
- package/LICENSE +21 -0
- package/README.md +132 -0
- package/bin/cli.mjs +7 -0
- package/dist/bus.d.ts +1 -0
- package/dist/bus.js +20 -0
- package/dist/cell-result.d.ts +25 -0
- package/dist/cell-result.js +32 -0
- package/dist/cli.d.ts +5 -0
- package/dist/cli.js +110 -0
- package/dist/context.d.ts +220 -0
- package/dist/context.js +479 -0
- package/dist/cost.d.ts +2 -0
- package/dist/cost.js +19 -0
- package/dist/env-cli.d.ts +5 -0
- package/dist/env-cli.js +645 -0
- package/dist/exec-rpc.d.ts +13 -0
- package/dist/exec-rpc.js +46 -0
- package/dist/index.d.ts +3 -0
- package/dist/index.js +1 -0
- package/dist/judge.d.ts +47 -0
- package/dist/judge.js +183 -0
- package/dist/loop.d.ts +62 -0
- package/dist/loop.js +316 -0
- package/dist/mock-cli-server.d.ts +40 -0
- package/dist/mock-cli-server.js +194 -0
- package/dist/mock-cli-shim.d.ts +7 -0
- package/dist/mock-cli-shim.js +89 -0
- package/dist/models.d.ts +23 -0
- package/dist/models.js +78 -0
- package/dist/parse-tagged-json.d.ts +1 -0
- package/dist/parse-tagged-json.js +16 -0
- package/dist/report.d.ts +215 -0
- package/dist/report.js +19 -0
- package/dist/runner-args.d.ts +11 -0
- package/dist/runner-args.js +84 -0
- package/dist/runner.d.ts +2 -0
- package/dist/runner.js +371 -0
- package/dist/summary.d.ts +3 -0
- package/dist/summary.js +64 -0
- package/dist/transcript-dump.d.ts +1 -0
- package/dist/transcript-dump.js +50 -0
- package/dist/transcript-log.d.ts +67 -0
- package/dist/transcript-log.js +199 -0
- package/dist/transcript-store.d.ts +6 -0
- package/dist/transcript-store.js +69 -0
- package/docker-compose.yml +83 -0
- package/exec-watcher.mjs +155 -0
- package/package.json +60 -0
- package/templates/.env.local.example +28 -0
- package/templates/Dockerfile +47 -0
- package/templates/docker-compose.yml +10 -0
- package/templates/openclaw.json +40 -0
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Runner-side HTTP endpoint that the gateway-side shim calls.
|
|
3
|
+
*
|
|
4
|
+
* POST /mock-cli/invoke
|
|
5
|
+
* { cli, argv, cwd, stdin }
|
|
6
|
+
* → { stdout, stderr, exitCode }
|
|
7
|
+
*
|
|
8
|
+
* A single in-flight registry is bound by the runner for the lifetime of one
|
|
9
|
+
* scenario. Per-conversation handler registries are populated by `ctx.mockCli`.
|
|
10
|
+
*/
|
|
11
|
+
import type { CliMockCall, CliMockHandler } from "./report.js";
|
|
12
|
+
export declare const MOCK_CLI_PORT = 43124;
|
|
13
|
+
export interface ConversationRegistry {
|
|
14
|
+
conversationId: string;
|
|
15
|
+
handlers: Map<string, CliMockHandler>;
|
|
16
|
+
/** Emit a `cliMock` event into the scenario's event stream. */
|
|
17
|
+
emitCliMock: (call: CliMockCall) => void;
|
|
18
|
+
/**
|
|
19
|
+
* `true` once the scenario has called `ctx.markScenarioAsEnded`. After that,
|
|
20
|
+
* the server bypasses handler dispatch: every incoming call is answered with
|
|
21
|
+
* the scenario-ended stub message so a lingering agent — still running while
|
|
22
|
+
* we move on — can read it and stop.
|
|
23
|
+
*/
|
|
24
|
+
isScenarioEnded: () => boolean;
|
|
25
|
+
}
|
|
26
|
+
export interface MockCliServer {
|
|
27
|
+
/** Bind the in-flight conversation registry for the lifetime of one scenario. */
|
|
28
|
+
bind(reg: ConversationRegistry): void;
|
|
29
|
+
/**
|
|
30
|
+
* Release the binding. Waits for the gateway side to go quiet (no incoming
|
|
31
|
+
* call for `RELEASE_QUIET_MS`) before clearing, so any straggler from the
|
|
32
|
+
* just-ended scenario lands on its own registry and gets the scenario-ended
|
|
33
|
+
* stub — instead of leaking into the next scenario's bind window as a
|
|
34
|
+
* spurious `UnexpectedCall`. Bounded by `RELEASE_HARD_TIMEOUT_MS`.
|
|
35
|
+
*/
|
|
36
|
+
release(): Promise<void>;
|
|
37
|
+
/** Stop the HTTP listener. */
|
|
38
|
+
close(): Promise<void>;
|
|
39
|
+
}
|
|
40
|
+
export declare function startMockCliServer(): MockCliServer;
|
|
@@ -0,0 +1,194 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Runner-side HTTP endpoint that the gateway-side shim calls.
|
|
3
|
+
*
|
|
4
|
+
* POST /mock-cli/invoke
|
|
5
|
+
* { cli, argv, cwd, stdin }
|
|
6
|
+
* → { stdout, stderr, exitCode }
|
|
7
|
+
*
|
|
8
|
+
* A single in-flight registry is bound by the runner for the lifetime of one
|
|
9
|
+
* scenario. Per-conversation handler registries are populated by `ctx.mockCli`.
|
|
10
|
+
*/
|
|
11
|
+
import { createServer } from "node:http";
|
|
12
|
+
import { Readable, Writable } from "node:stream";
|
|
13
|
+
export const MOCK_CLI_PORT = 43124;
|
|
14
|
+
const SCENARIO_ENDED_STUB_MESSAGE = "Stubbed call. If you see this, it means that you are in a test scenario. You should stop and acknowledge to the user.\n";
|
|
15
|
+
const RELEASE_QUIET_MS = 500;
|
|
16
|
+
const RELEASE_HARD_TIMEOUT_MS = 5_000;
|
|
17
|
+
const RELEASE_POLL_MS = 100;
|
|
18
|
+
export function startMockCliServer() {
|
|
19
|
+
let current;
|
|
20
|
+
let lastCallAtMs = 0;
|
|
21
|
+
const server = createServer(async (req, res) => {
|
|
22
|
+
if (req.method !== "POST" || req.url !== "/mock-cli/invoke") {
|
|
23
|
+
res.statusCode = 404;
|
|
24
|
+
res.end("not found");
|
|
25
|
+
return;
|
|
26
|
+
}
|
|
27
|
+
lastCallAtMs = Date.now();
|
|
28
|
+
await handleInvoke(req, res, current);
|
|
29
|
+
});
|
|
30
|
+
server.listen(MOCK_CLI_PORT, "0.0.0.0", () => {
|
|
31
|
+
console.log(`mock-cli server listening on 0.0.0.0:${MOCK_CLI_PORT}`);
|
|
32
|
+
});
|
|
33
|
+
return {
|
|
34
|
+
bind: (reg) => {
|
|
35
|
+
current = reg;
|
|
36
|
+
},
|
|
37
|
+
release: async () => {
|
|
38
|
+
const deadline = Date.now() + RELEASE_HARD_TIMEOUT_MS;
|
|
39
|
+
while (lastCallAtMs > 0 && Date.now() - lastCallAtMs < RELEASE_QUIET_MS) {
|
|
40
|
+
if (Date.now() >= deadline)
|
|
41
|
+
break;
|
|
42
|
+
await new Promise((r) => setTimeout(r, RELEASE_POLL_MS));
|
|
43
|
+
}
|
|
44
|
+
current = undefined;
|
|
45
|
+
lastCallAtMs = 0;
|
|
46
|
+
},
|
|
47
|
+
close: () => new Promise((resolve) => {
|
|
48
|
+
server.close(() => resolve());
|
|
49
|
+
// Force-destroy lingering keep-alive sockets. A mocked CLI is invoked through the
|
|
50
|
+
// gateway shim over HTTP keep-alive; after the call the socket stays idle-open, and a
|
|
51
|
+
// bare `server.close()` waits for it indefinitely — the runner then hangs past the
|
|
52
|
+
// verdict instead of exiting. `closeAllConnections` lets `close()` complete at once.
|
|
53
|
+
server.closeAllConnections();
|
|
54
|
+
}),
|
|
55
|
+
};
|
|
56
|
+
}
|
|
57
|
+
async function handleInvoke(req, res, reg) {
|
|
58
|
+
let body;
|
|
59
|
+
try {
|
|
60
|
+
body = await readJsonBody(req);
|
|
61
|
+
}
|
|
62
|
+
catch (err) {
|
|
63
|
+
res.statusCode = 400;
|
|
64
|
+
res.end(`bad json: ${err.message}`);
|
|
65
|
+
return;
|
|
66
|
+
}
|
|
67
|
+
const invoke = normalizeInvoke(body);
|
|
68
|
+
if (!invoke) {
|
|
69
|
+
res.statusCode = 400;
|
|
70
|
+
res.end("missing cli");
|
|
71
|
+
return;
|
|
72
|
+
}
|
|
73
|
+
if (!reg) {
|
|
74
|
+
respondMissingScenario(res);
|
|
75
|
+
return;
|
|
76
|
+
}
|
|
77
|
+
const startedAt = new Date().toISOString();
|
|
78
|
+
const startedAtMs = Date.now();
|
|
79
|
+
if (reg.isScenarioEnded()) {
|
|
80
|
+
// Verdict is already decided. Don't dispatch to handlers and don't surface
|
|
81
|
+
// failures — just give the (still-running) agent a clear stop signal.
|
|
82
|
+
emitCallEvent(reg, invoke, { stdout: SCENARIO_ENDED_STUB_MESSAGE, stderr: "", exitCode: 0 }, startedAt, startedAtMs);
|
|
83
|
+
respondJson(res, { stdout: SCENARIO_ENDED_STUB_MESSAGE, stderr: "", exitCode: 0 });
|
|
84
|
+
return;
|
|
85
|
+
}
|
|
86
|
+
const handler = reg.handlers.get(invoke.cli);
|
|
87
|
+
if (!handler) {
|
|
88
|
+
const stderr = `mock-cli: unexpected call to ${invoke.cli}\n`;
|
|
89
|
+
emitCallEvent(reg, invoke, {
|
|
90
|
+
stdout: "",
|
|
91
|
+
stderr,
|
|
92
|
+
exitCode: 1,
|
|
93
|
+
handlerError: { name: "UnexpectedCall", message: `unexpected call to ${invoke.cli}` },
|
|
94
|
+
}, startedAt, startedAtMs);
|
|
95
|
+
respondJson(res, { stdout: "", stderr, exitCode: 1, missing: true });
|
|
96
|
+
return;
|
|
97
|
+
}
|
|
98
|
+
const outcome = await runHandler(handler, invoke);
|
|
99
|
+
emitCallEvent(reg, invoke, outcome, startedAt, startedAtMs);
|
|
100
|
+
respondJson(res, { stdout: outcome.stdout, stderr: outcome.stderr, exitCode: outcome.exitCode });
|
|
101
|
+
}
|
|
102
|
+
function normalizeInvoke(body) {
|
|
103
|
+
const cli = String(body.cli ?? "");
|
|
104
|
+
if (!cli)
|
|
105
|
+
return;
|
|
106
|
+
return {
|
|
107
|
+
cli,
|
|
108
|
+
argv: Array.isArray(body.argv) ? body.argv.map(String) : [],
|
|
109
|
+
cwd: String(body.cwd ?? ""),
|
|
110
|
+
stdin: String(body.stdin ?? ""),
|
|
111
|
+
};
|
|
112
|
+
}
|
|
113
|
+
async function runHandler(handler, invoke) {
|
|
114
|
+
const stdinStream = Readable.from([Buffer.from(invoke.stdin, "utf8")]);
|
|
115
|
+
const outCap = collectStream();
|
|
116
|
+
const errCap = collectStream();
|
|
117
|
+
const args = {
|
|
118
|
+
argv: invoke.argv,
|
|
119
|
+
cwd: invoke.cwd,
|
|
120
|
+
stdin: stdinStream,
|
|
121
|
+
stdout: outCap.writable,
|
|
122
|
+
stderr: errCap.writable,
|
|
123
|
+
};
|
|
124
|
+
let exitCode = 0;
|
|
125
|
+
let handlerError;
|
|
126
|
+
try {
|
|
127
|
+
const result = await handler(args);
|
|
128
|
+
exitCode = typeof result === "number" ? result : 0;
|
|
129
|
+
}
|
|
130
|
+
catch (err) {
|
|
131
|
+
const e = err;
|
|
132
|
+
handlerError = {
|
|
133
|
+
name: e?.name ?? "Error",
|
|
134
|
+
message: e?.message ?? String(err),
|
|
135
|
+
stack: e?.stack,
|
|
136
|
+
};
|
|
137
|
+
exitCode = 1;
|
|
138
|
+
}
|
|
139
|
+
outCap.writable.end();
|
|
140
|
+
errCap.writable.end();
|
|
141
|
+
return { stdout: outCap.read(), stderr: errCap.read(), exitCode, handlerError };
|
|
142
|
+
}
|
|
143
|
+
function emitCallEvent(reg, invoke, outcome, startedAt, startedAtMs) {
|
|
144
|
+
const call = {
|
|
145
|
+
cli: invoke.cli,
|
|
146
|
+
argv: invoke.argv,
|
|
147
|
+
cwd: invoke.cwd,
|
|
148
|
+
stdin: invoke.stdin,
|
|
149
|
+
stdout: outcome.stdout,
|
|
150
|
+
stderr: outcome.stderr,
|
|
151
|
+
exitCode: outcome.exitCode,
|
|
152
|
+
startedAt,
|
|
153
|
+
durationMs: Date.now() - startedAtMs,
|
|
154
|
+
...(outcome.handlerError ? { handlerError: outcome.handlerError } : {}),
|
|
155
|
+
};
|
|
156
|
+
reg.emitCliMock(call);
|
|
157
|
+
}
|
|
158
|
+
function respondMissingScenario(res) {
|
|
159
|
+
const stderr = "mock-cli: no scenario currently owns this gateway\n";
|
|
160
|
+
respondJson(res, { stdout: "", stderr, exitCode: 1, missing: true });
|
|
161
|
+
}
|
|
162
|
+
function respondJson(res, body) {
|
|
163
|
+
res.setHeader("content-type", "application/json");
|
|
164
|
+
res.end(JSON.stringify(body));
|
|
165
|
+
}
|
|
166
|
+
function readJsonBody(req) {
|
|
167
|
+
return new Promise((resolve, reject) => {
|
|
168
|
+
const chunks = [];
|
|
169
|
+
req.on("data", (c) => chunks.push(c));
|
|
170
|
+
req.on("end", () => {
|
|
171
|
+
const raw = Buffer.concat(chunks).toString("utf8");
|
|
172
|
+
try {
|
|
173
|
+
resolve(raw ? JSON.parse(raw) : {});
|
|
174
|
+
}
|
|
175
|
+
catch (err) {
|
|
176
|
+
reject(err);
|
|
177
|
+
}
|
|
178
|
+
});
|
|
179
|
+
req.on("error", reject);
|
|
180
|
+
});
|
|
181
|
+
}
|
|
182
|
+
function collectStream() {
|
|
183
|
+
const chunks = [];
|
|
184
|
+
const writable = new Writable({
|
|
185
|
+
write(chunk, _enc, cb) {
|
|
186
|
+
chunks.push(typeof chunk === "string" ? Buffer.from(chunk) : chunk);
|
|
187
|
+
cb();
|
|
188
|
+
},
|
|
189
|
+
});
|
|
190
|
+
return {
|
|
191
|
+
writable,
|
|
192
|
+
read: () => Buffer.concat(chunks).toString("utf8"),
|
|
193
|
+
};
|
|
194
|
+
}
|
|
@@ -0,0 +1,7 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Mock-CLI shim. Consumers symlink it under `/opt/openclaw-test/mocks/bin/` for
|
|
3
|
+
* whichever commands they want to intercept (typically `claude`, `gh`, `glab`,
|
|
4
|
+
* etc.). Determines its invoked name from argv[1] basename, POSTs the call to
|
|
5
|
+
* the runner's /mock-cli/invoke endpoint, and replays the response locally.
|
|
6
|
+
*/
|
|
7
|
+
export {};
|
|
@@ -0,0 +1,89 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Mock-CLI shim. Consumers symlink it under `/opt/openclaw-test/mocks/bin/` for
|
|
3
|
+
* whichever commands they want to intercept (typically `claude`, `gh`, `glab`,
|
|
4
|
+
* etc.). Determines its invoked name from argv[1] basename, POSTs the call to
|
|
5
|
+
* the runner's /mock-cli/invoke endpoint, and replays the response locally.
|
|
6
|
+
*/
|
|
7
|
+
import { request as httpRequest } from "node:http";
|
|
8
|
+
import { basename } from "node:path";
|
|
9
|
+
import { URL } from "node:url";
|
|
10
|
+
function die(msg, code = 127) {
|
|
11
|
+
process.stderr.write(`mock-cli shim: ${msg}\n`);
|
|
12
|
+
process.exit(code);
|
|
13
|
+
}
|
|
14
|
+
async function readStdin() {
|
|
15
|
+
if (process.stdin.isTTY)
|
|
16
|
+
return "";
|
|
17
|
+
const chunks = [];
|
|
18
|
+
for await (const chunk of process.stdin) {
|
|
19
|
+
chunks.push(typeof chunk === "string" ? Buffer.from(chunk) : chunk);
|
|
20
|
+
}
|
|
21
|
+
return Buffer.concat(chunks).toString("utf8");
|
|
22
|
+
}
|
|
23
|
+
function post(urlStr, body) {
|
|
24
|
+
const url = new URL(urlStr);
|
|
25
|
+
return new Promise((resolve, reject) => {
|
|
26
|
+
const req = httpRequest({
|
|
27
|
+
method: "POST",
|
|
28
|
+
hostname: url.hostname,
|
|
29
|
+
port: url.port || 80,
|
|
30
|
+
path: url.pathname + url.search,
|
|
31
|
+
headers: {
|
|
32
|
+
"content-type": "application/json",
|
|
33
|
+
"content-length": Buffer.byteLength(body),
|
|
34
|
+
},
|
|
35
|
+
}, (res) => {
|
|
36
|
+
const chunks = [];
|
|
37
|
+
res.on("data", (c) => chunks.push(c));
|
|
38
|
+
res.on("end", () => {
|
|
39
|
+
const raw = Buffer.concat(chunks).toString("utf8");
|
|
40
|
+
if (res.statusCode !== 200) {
|
|
41
|
+
reject(new Error(`runner returned HTTP ${res.statusCode}: ${raw}`));
|
|
42
|
+
return;
|
|
43
|
+
}
|
|
44
|
+
try {
|
|
45
|
+
resolve(JSON.parse(raw));
|
|
46
|
+
}
|
|
47
|
+
catch (err) {
|
|
48
|
+
reject(new Error(`invalid JSON response: ${err.message}`));
|
|
49
|
+
}
|
|
50
|
+
});
|
|
51
|
+
res.on("error", reject);
|
|
52
|
+
});
|
|
53
|
+
req.on("error", reject);
|
|
54
|
+
req.write(body);
|
|
55
|
+
req.end();
|
|
56
|
+
});
|
|
57
|
+
}
|
|
58
|
+
async function main() {
|
|
59
|
+
// The sh wrapper at /opt/openclaw-test/mocks/bin/mock-cli-shim invokes us as
|
|
60
|
+
// exec node mock-cli-shim.js "$0" "$@"
|
|
61
|
+
// so argv[2] is the symlink path used to call us (e.g. /opt/openclaw-test/mocks/bin/git)
|
|
62
|
+
// and argv.slice(3) is the original argv tail.
|
|
63
|
+
const invokedAs = basename(process.argv[2] ?? "");
|
|
64
|
+
if (!invokedAs)
|
|
65
|
+
die("could not determine invoked binary name from argv[2]");
|
|
66
|
+
const runnerUrl = process.env.OPENCLAW_TEST_RUNNER_URL;
|
|
67
|
+
if (!runnerUrl)
|
|
68
|
+
die("OPENCLAW_TEST_RUNNER_URL is not set");
|
|
69
|
+
const stdin = await readStdin();
|
|
70
|
+
const body = JSON.stringify({
|
|
71
|
+
cli: invokedAs,
|
|
72
|
+
argv: process.argv.slice(3),
|
|
73
|
+
cwd: process.cwd(),
|
|
74
|
+
stdin,
|
|
75
|
+
});
|
|
76
|
+
let res;
|
|
77
|
+
try {
|
|
78
|
+
res = await post(`${runnerUrl}/mock-cli/invoke`, body);
|
|
79
|
+
}
|
|
80
|
+
catch (err) {
|
|
81
|
+
die(`POST ${runnerUrl}/mock-cli/invoke failed: ${err.message}`);
|
|
82
|
+
}
|
|
83
|
+
if (res.stdout)
|
|
84
|
+
process.stdout.write(res.stdout);
|
|
85
|
+
if (res.stderr)
|
|
86
|
+
process.stderr.write(res.stderr);
|
|
87
|
+
process.exit(typeof res.exitCode === "number" ? res.exitCode : 1);
|
|
88
|
+
}
|
|
89
|
+
main().catch((err) => die(err.message ?? String(err)));
|
package/dist/models.d.ts
ADDED
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
export interface SelectedModel {
|
|
2
|
+
/** Bare id: the suffix after the last `/` of the full ref. */
|
|
3
|
+
id: string;
|
|
4
|
+
/** Full LiteLLM `provider/model` ref, as written in `OPENCLAW_TEST_MODELS`. */
|
|
5
|
+
ref: string;
|
|
6
|
+
}
|
|
7
|
+
/**
|
|
8
|
+
* Resolve the `--model` value against the configured model catalog.
|
|
9
|
+
*
|
|
10
|
+
* `OPENCLAW_TEST_MODELS` is a comma list of full `provider/model` refs — the only
|
|
11
|
+
* place the `provider/` prefix appears. The CLI value, `OPENCLAW_DEFAULT_TEST_MODEL`,
|
|
12
|
+
* and the recorded `model` are all bare ids (the suffix after the last `/`). A bare
|
|
13
|
+
* id resolves to its ref by suffix-match; zero or many matches is a hard error.
|
|
14
|
+
*
|
|
15
|
+
* The selection is `all` (whole catalog, sorted by id), a single bare id, or a comma
|
|
16
|
+
* list of bare ids (deduped, CLI order preserved); `undefined` falls back to
|
|
17
|
+
* `OPENCLAW_DEFAULT_TEST_MODEL`.
|
|
18
|
+
*/
|
|
19
|
+
export declare function resolveSelectedModels(params: {
|
|
20
|
+
selection: string | undefined;
|
|
21
|
+
modelsEnv: string | undefined;
|
|
22
|
+
defaultEnv: string | undefined;
|
|
23
|
+
}): SelectedModel[];
|
package/dist/models.js
ADDED
|
@@ -0,0 +1,78 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Resolve the `--model` value against the configured model catalog.
|
|
3
|
+
*
|
|
4
|
+
* `OPENCLAW_TEST_MODELS` is a comma list of full `provider/model` refs — the only
|
|
5
|
+
* place the `provider/` prefix appears. The CLI value, `OPENCLAW_DEFAULT_TEST_MODEL`,
|
|
6
|
+
* and the recorded `model` are all bare ids (the suffix after the last `/`). A bare
|
|
7
|
+
* id resolves to its ref by suffix-match; zero or many matches is a hard error.
|
|
8
|
+
*
|
|
9
|
+
* The selection is `all` (whole catalog, sorted by id), a single bare id, or a comma
|
|
10
|
+
* list of bare ids (deduped, CLI order preserved); `undefined` falls back to
|
|
11
|
+
* `OPENCLAW_DEFAULT_TEST_MODEL`.
|
|
12
|
+
*/
|
|
13
|
+
export function resolveSelectedModels(params) {
|
|
14
|
+
const catalog = parseCatalog(params.modelsEnv);
|
|
15
|
+
if (params.selection === "all")
|
|
16
|
+
return [...catalog].sort((a, b) => a.id.localeCompare(b.id));
|
|
17
|
+
if (params.selection !== undefined)
|
|
18
|
+
return resolveIdList(catalog, params.selection);
|
|
19
|
+
if (!params.defaultEnv) {
|
|
20
|
+
throw new Error("run: no --model given and OPENCLAW_DEFAULT_TEST_MODEL is unset; pass --model <id|id,id,…|all> or set the default");
|
|
21
|
+
}
|
|
22
|
+
return [matchById(catalog, params.defaultEnv)];
|
|
23
|
+
}
|
|
24
|
+
function resolveIdList(catalog, selection) {
|
|
25
|
+
const ids = selection
|
|
26
|
+
.split(",")
|
|
27
|
+
.map((s) => s.trim())
|
|
28
|
+
.filter((s) => s.length > 0);
|
|
29
|
+
if (ids.length === 0) {
|
|
30
|
+
throw new Error(`run: --model expects a non-empty id list, got ${JSON.stringify(selection)}`);
|
|
31
|
+
}
|
|
32
|
+
const seen = new Set();
|
|
33
|
+
const selected = [];
|
|
34
|
+
for (const id of ids) {
|
|
35
|
+
if (seen.has(id))
|
|
36
|
+
continue;
|
|
37
|
+
seen.add(id);
|
|
38
|
+
selected.push(matchById(catalog, id));
|
|
39
|
+
}
|
|
40
|
+
return selected;
|
|
41
|
+
}
|
|
42
|
+
function parseCatalog(modelsEnv) {
|
|
43
|
+
const refs = (modelsEnv ?? "")
|
|
44
|
+
.split(",")
|
|
45
|
+
.map((s) => s.trim())
|
|
46
|
+
.filter((s) => s.length > 0);
|
|
47
|
+
if (refs.length === 0) {
|
|
48
|
+
throw new Error("run: OPENCLAW_TEST_MODELS is empty; set a comma list of provider/model refs");
|
|
49
|
+
}
|
|
50
|
+
const catalog = refs.map((ref) => ({ id: bareId(ref), ref }));
|
|
51
|
+
assertUniqueIds(catalog);
|
|
52
|
+
return catalog;
|
|
53
|
+
}
|
|
54
|
+
function bareId(ref) {
|
|
55
|
+
const id = ref.split("/").pop();
|
|
56
|
+
if (!id)
|
|
57
|
+
throw new Error(`run: invalid model ref ${JSON.stringify(ref)} in OPENCLAW_TEST_MODELS`);
|
|
58
|
+
return id;
|
|
59
|
+
}
|
|
60
|
+
function assertUniqueIds(catalog) {
|
|
61
|
+
const seen = new Set();
|
|
62
|
+
for (const { id } of catalog) {
|
|
63
|
+
if (seen.has(id)) {
|
|
64
|
+
throw new Error(`run: OPENCLAW_TEST_MODELS has two entries with bare id ${JSON.stringify(id)}`);
|
|
65
|
+
}
|
|
66
|
+
seen.add(id);
|
|
67
|
+
}
|
|
68
|
+
}
|
|
69
|
+
function matchById(catalog, id) {
|
|
70
|
+
const matches = catalog.filter((m) => m.id === id);
|
|
71
|
+
if (matches.length === 1)
|
|
72
|
+
return matches[0];
|
|
73
|
+
const known = catalog.map((m) => m.id).join(", ");
|
|
74
|
+
if (matches.length === 0) {
|
|
75
|
+
throw new Error(`run: model ${JSON.stringify(id)} not found in OPENCLAW_TEST_MODELS — known: ${known}`);
|
|
76
|
+
}
|
|
77
|
+
throw new Error(`run: model ${JSON.stringify(id)} is ambiguous in OPENCLAW_TEST_MODELS — known: ${known}`);
|
|
78
|
+
}
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
export declare function extractTaggedBlock(raw: string, tagName: string): string;
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
export function extractTaggedBlock(raw, tagName) {
|
|
2
|
+
const escaped = escapeRegex(tagName);
|
|
3
|
+
const open = new RegExp(`<${escaped}>`).exec(raw);
|
|
4
|
+
if (!open) {
|
|
5
|
+
throw new Error(`extractTaggedBlock: opening <${tagName}> not found. raw=${JSON.stringify(raw.slice(0, 200))}`);
|
|
6
|
+
}
|
|
7
|
+
const re = new RegExp(`<${escaped}>([\\s\\S]*?)</${escaped}>`);
|
|
8
|
+
const match = re.exec(raw);
|
|
9
|
+
if (!match) {
|
|
10
|
+
throw new Error(`extractTaggedBlock: closing </${tagName}> not found after opening tag. raw=${JSON.stringify(raw.slice(0, 200))}`);
|
|
11
|
+
}
|
|
12
|
+
return match[1].trim();
|
|
13
|
+
}
|
|
14
|
+
function escapeRegex(value) {
|
|
15
|
+
return value.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
|
|
16
|
+
}
|
package/dist/report.d.ts
ADDED
|
@@ -0,0 +1,215 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Scenario report typing for openclaw-test.
|
|
3
|
+
*
|
|
4
|
+
* Two artifacts per scenario run, under `<OPENCLAW_TEST_ARTIFACTS_DIR>/<baseStamp>/<scenario>-<channel>[-iter<n>]/`:
|
|
5
|
+
*
|
|
6
|
+
* - `scenario-log.jsonl` — live append, one `ReportEntry` per line, written as
|
|
7
|
+
* things happen. Survives runner crash / hang.
|
|
8
|
+
* - `report.json` — final `ScenarioReport`, written once at end.
|
|
9
|
+
*
|
|
10
|
+
* `report.json` carries the same entries as `scenario-log.jsonl`, plus terminal
|
|
11
|
+
* data only computable at the end (status, durationMs, finishedAt, parsed
|
|
12
|
+
* agentToolCalls, cost).
|
|
13
|
+
*
|
|
14
|
+
* Each agent or scenario action is one entry. Assertions, scenario-log notes,
|
|
15
|
+
* and failures about that action live as nested fields on the entry, not as
|
|
16
|
+
* separate entries. Free-standing `scenarioLog` entries are the fallback for
|
|
17
|
+
* `ctx.log(...)` calls that are not bound to an action.
|
|
18
|
+
*/
|
|
19
|
+
import type { Readable, Writable } from "node:stream";
|
|
20
|
+
export interface ScenarioReport {
|
|
21
|
+
schemaVersion: 4;
|
|
22
|
+
scenario: string;
|
|
23
|
+
channel: ChannelId;
|
|
24
|
+
model: string;
|
|
25
|
+
conversationId: string;
|
|
26
|
+
accountId: string;
|
|
27
|
+
startedAt: string;
|
|
28
|
+
finishedAt: string;
|
|
29
|
+
durationMs: number;
|
|
30
|
+
cost: CostBreakdown;
|
|
31
|
+
result: ScenarioResult;
|
|
32
|
+
entries: ReportEntry[];
|
|
33
|
+
}
|
|
34
|
+
export type ScenarioResult = PassScenarioResult | FailScenarioResult;
|
|
35
|
+
export type FailScenarioResult = FailedEntryScenarioResult | ErrorScenarioResult;
|
|
36
|
+
export interface PassScenarioResult {
|
|
37
|
+
verdict: "pass";
|
|
38
|
+
}
|
|
39
|
+
export interface FailedEntryScenarioResult {
|
|
40
|
+
verdict: "fail";
|
|
41
|
+
cause: "failedEntry";
|
|
42
|
+
/** Stable entry identifier shared by `scenario-log.jsonl` and `report.json`. */
|
|
43
|
+
entrySeq: number;
|
|
44
|
+
message: string;
|
|
45
|
+
}
|
|
46
|
+
export interface ErrorScenarioResult {
|
|
47
|
+
verdict: "fail";
|
|
48
|
+
cause: "error";
|
|
49
|
+
source: "assertion" | "judge" | "cliMock" | "timeout" | "scenarioThrow" | "runner";
|
|
50
|
+
errorName: string;
|
|
51
|
+
message: string;
|
|
52
|
+
stack?: string;
|
|
53
|
+
}
|
|
54
|
+
export type ChannelId = string;
|
|
55
|
+
export type ReportEntry = ScenarioLogEntry | ActionEntry;
|
|
56
|
+
export type ActionEntry = InboundSentEntry | OutboundReceivedEntry | CliMockEntry | AgentToolCallEntry;
|
|
57
|
+
export interface ReportEntryBase {
|
|
58
|
+
ts: string;
|
|
59
|
+
/**
|
|
60
|
+
* Stable entry id shared across `scenario-log.jsonl` and `report.json`. In the
|
|
61
|
+
* jsonl it matches the line's emission order; in `report.json` entries are
|
|
62
|
+
* sorted by `ts` so `entrySeq` is not array position — it stays a cross-file
|
|
63
|
+
* reference.
|
|
64
|
+
*/
|
|
65
|
+
entrySeq: number;
|
|
66
|
+
}
|
|
67
|
+
export interface ActionEntryBase extends ReportEntryBase {
|
|
68
|
+
scenarioLog?: ScenarioLogNote;
|
|
69
|
+
assertions?: AssertionRecord[];
|
|
70
|
+
failure?: ScenarioFailure;
|
|
71
|
+
}
|
|
72
|
+
export interface ScenarioLogNote {
|
|
73
|
+
label?: string;
|
|
74
|
+
ts: string;
|
|
75
|
+
extra?: unknown;
|
|
76
|
+
}
|
|
77
|
+
export interface ScenarioLogEntry extends ReportEntryBase {
|
|
78
|
+
kind: "scenarioLog";
|
|
79
|
+
message: string;
|
|
80
|
+
}
|
|
81
|
+
export interface InboundSentEntry extends ActionEntryBase {
|
|
82
|
+
kind: "inboundSent";
|
|
83
|
+
messageId: string;
|
|
84
|
+
text: string;
|
|
85
|
+
senderId: string;
|
|
86
|
+
senderName?: string;
|
|
87
|
+
threadId?: string;
|
|
88
|
+
}
|
|
89
|
+
export interface OutboundReceivedEntry extends ActionEntryBase {
|
|
90
|
+
kind: "outboundReceived";
|
|
91
|
+
messageId: string;
|
|
92
|
+
text: string;
|
|
93
|
+
threadId?: string;
|
|
94
|
+
}
|
|
95
|
+
export interface CliMockEntry extends ActionEntryBase {
|
|
96
|
+
kind: "cliMock";
|
|
97
|
+
call: CliMockCall;
|
|
98
|
+
}
|
|
99
|
+
export interface AgentToolCallEntry extends ActionEntryBase {
|
|
100
|
+
kind: "agentToolCall";
|
|
101
|
+
call: AgentToolCall;
|
|
102
|
+
}
|
|
103
|
+
export type AssertionRecord = {
|
|
104
|
+
label: string;
|
|
105
|
+
ok: true;
|
|
106
|
+
extra?: unknown;
|
|
107
|
+
} | {
|
|
108
|
+
label: string;
|
|
109
|
+
ok: false;
|
|
110
|
+
detail: string;
|
|
111
|
+
extra?: unknown;
|
|
112
|
+
};
|
|
113
|
+
export interface CliMockCall {
|
|
114
|
+
cli: "git" | "npm" | "pnpm" | "yarn" | "claude" | (string & {});
|
|
115
|
+
argv: string[];
|
|
116
|
+
cwd: string;
|
|
117
|
+
stdin: string;
|
|
118
|
+
stdout: string;
|
|
119
|
+
stderr: string;
|
|
120
|
+
exitCode: number;
|
|
121
|
+
startedAt: string;
|
|
122
|
+
durationMs: number;
|
|
123
|
+
handlerError?: {
|
|
124
|
+
name: string;
|
|
125
|
+
message: string;
|
|
126
|
+
stack?: string;
|
|
127
|
+
};
|
|
128
|
+
}
|
|
129
|
+
export interface AgentToolCall {
|
|
130
|
+
toolName: string;
|
|
131
|
+
toolUseId: string;
|
|
132
|
+
/**
|
|
133
|
+
* `sessionKey` of the session transcript the call was collected from — the
|
|
134
|
+
* OpenClaw session that made it (e.g. channel vs per-thread session).
|
|
135
|
+
*/
|
|
136
|
+
sessionKey?: string;
|
|
137
|
+
input: unknown;
|
|
138
|
+
result?: {
|
|
139
|
+
isError: boolean;
|
|
140
|
+
/**
|
|
141
|
+
* Full tool result. Always present in `scenario-log.jsonl`. In `report.json`,
|
|
142
|
+
* replaced by `truncatedContent` when truncatable: a string longer than 60
|
|
143
|
+
* chars, or a `read` call's text content blocks (the file is identified by
|
|
144
|
+
* `input` and kept in full in the jsonl).
|
|
145
|
+
*/
|
|
146
|
+
content?: unknown;
|
|
147
|
+
/** Only in `report.json`: rtrimmed first 60 chars + `…` of the truncatable text. */
|
|
148
|
+
truncatedContent?: string;
|
|
149
|
+
};
|
|
150
|
+
/**
|
|
151
|
+
* Timestamp of the transcript's assistant message that issued the call —
|
|
152
|
+
* per-message, so ordering `entries` on it is meaningful. Absent when the
|
|
153
|
+
* message carried no timestamp; such calls sort to the end of the timeline.
|
|
154
|
+
*/
|
|
155
|
+
startedAt?: string;
|
|
156
|
+
/**
|
|
157
|
+
* Best-effort estimate of when the tool call actually executed, inferred by
|
|
158
|
+
* matching the call's leading CLI against an in-order `cliMock` entry from
|
|
159
|
+
* the same scenario. `startedAt` stamps the issuing assistant message, not
|
|
160
|
+
* the execution. Not used for `entries` ordering.
|
|
161
|
+
*/
|
|
162
|
+
inferredStartedAt?: string;
|
|
163
|
+
turn?: number;
|
|
164
|
+
}
|
|
165
|
+
export interface ScenarioFailure {
|
|
166
|
+
name: string;
|
|
167
|
+
message: string;
|
|
168
|
+
stack?: string;
|
|
169
|
+
source: "assertion" | "judge" | "cliMock" | "timeout" | "scenarioThrow" | "runner";
|
|
170
|
+
}
|
|
171
|
+
export interface CostBreakdown {
|
|
172
|
+
agentUsd: number;
|
|
173
|
+
judgeUsd: number;
|
|
174
|
+
totalUsd: number;
|
|
175
|
+
agentTurns: number;
|
|
176
|
+
}
|
|
177
|
+
export interface CliMockHandlerArgs {
|
|
178
|
+
argv: string[];
|
|
179
|
+
cwd: string;
|
|
180
|
+
stdin: Readable;
|
|
181
|
+
stdout: Writable;
|
|
182
|
+
stderr: Writable;
|
|
183
|
+
}
|
|
184
|
+
export type CliMockHandler = (args: CliMockHandlerArgs) => number | undefined | Promise<number | undefined>;
|
|
185
|
+
/**
|
|
186
|
+
* Patch describing the augmentation of an already-emitted entry. Carries only
|
|
187
|
+
* the new nested field so the live `scenario-log.jsonl` need not re-serialize
|
|
188
|
+
* the entry's full text/metadata on every assertion or annotation.
|
|
189
|
+
*/
|
|
190
|
+
export type AugmentPatch = {
|
|
191
|
+
kind: "scenarioLog";
|
|
192
|
+
scenarioLog: ScenarioLogNote;
|
|
193
|
+
} | {
|
|
194
|
+
kind: "assertion";
|
|
195
|
+
assertion: AssertionRecord;
|
|
196
|
+
} | {
|
|
197
|
+
kind: "failure";
|
|
198
|
+
failure: ScenarioFailure;
|
|
199
|
+
};
|
|
200
|
+
export type SinkEvent = {
|
|
201
|
+
type: "entry";
|
|
202
|
+
entry: ReportEntry;
|
|
203
|
+
} | {
|
|
204
|
+
type: "augment";
|
|
205
|
+
entrySeq: number;
|
|
206
|
+
patch: AugmentPatch;
|
|
207
|
+
};
|
|
208
|
+
/**
|
|
209
|
+
* Callback the runner passes into `createContext` so live records can be
|
|
210
|
+
* appended to `scenario-log.jsonl` as they happen. First emission of an entry
|
|
211
|
+
* is `type: "entry"`. Subsequent nested-field additions (assertions, failure,
|
|
212
|
+
* scenarioLog) are `type: "augment"` with only the patch — readers reconstruct
|
|
213
|
+
* full state by folding augments onto entries by `seq`.
|
|
214
|
+
*/
|
|
215
|
+
export type EmitSink = (event: SinkEvent) => void;
|
package/dist/report.js
ADDED
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Scenario report typing for openclaw-test.
|
|
3
|
+
*
|
|
4
|
+
* Two artifacts per scenario run, under `<OPENCLAW_TEST_ARTIFACTS_DIR>/<baseStamp>/<scenario>-<channel>[-iter<n>]/`:
|
|
5
|
+
*
|
|
6
|
+
* - `scenario-log.jsonl` — live append, one `ReportEntry` per line, written as
|
|
7
|
+
* things happen. Survives runner crash / hang.
|
|
8
|
+
* - `report.json` — final `ScenarioReport`, written once at end.
|
|
9
|
+
*
|
|
10
|
+
* `report.json` carries the same entries as `scenario-log.jsonl`, plus terminal
|
|
11
|
+
* data only computable at the end (status, durationMs, finishedAt, parsed
|
|
12
|
+
* agentToolCalls, cost).
|
|
13
|
+
*
|
|
14
|
+
* Each agent or scenario action is one entry. Assertions, scenario-log notes,
|
|
15
|
+
* and failures about that action live as nested fields on the entry, not as
|
|
16
|
+
* separate entries. Free-standing `scenarioLog` entries are the fallback for
|
|
17
|
+
* `ctx.log(...)` calls that are not bound to an action.
|
|
18
|
+
*/
|
|
19
|
+
export {};
|