@alignfirst/openclaw-test 0.20.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/Dockerfile.base +45 -0
- package/LICENSE +21 -0
- package/README.md +132 -0
- package/bin/cli.mjs +7 -0
- package/dist/bus.d.ts +1 -0
- package/dist/bus.js +20 -0
- package/dist/cell-result.d.ts +25 -0
- package/dist/cell-result.js +32 -0
- package/dist/cli.d.ts +5 -0
- package/dist/cli.js +110 -0
- package/dist/context.d.ts +220 -0
- package/dist/context.js +479 -0
- package/dist/cost.d.ts +2 -0
- package/dist/cost.js +19 -0
- package/dist/env-cli.d.ts +5 -0
- package/dist/env-cli.js +645 -0
- package/dist/exec-rpc.d.ts +13 -0
- package/dist/exec-rpc.js +46 -0
- package/dist/index.d.ts +3 -0
- package/dist/index.js +1 -0
- package/dist/judge.d.ts +47 -0
- package/dist/judge.js +183 -0
- package/dist/loop.d.ts +62 -0
- package/dist/loop.js +316 -0
- package/dist/mock-cli-server.d.ts +40 -0
- package/dist/mock-cli-server.js +194 -0
- package/dist/mock-cli-shim.d.ts +7 -0
- package/dist/mock-cli-shim.js +89 -0
- package/dist/models.d.ts +23 -0
- package/dist/models.js +78 -0
- package/dist/parse-tagged-json.d.ts +1 -0
- package/dist/parse-tagged-json.js +16 -0
- package/dist/report.d.ts +215 -0
- package/dist/report.js +19 -0
- package/dist/runner-args.d.ts +11 -0
- package/dist/runner-args.js +84 -0
- package/dist/runner.d.ts +2 -0
- package/dist/runner.js +371 -0
- package/dist/summary.d.ts +3 -0
- package/dist/summary.js +64 -0
- package/dist/transcript-dump.d.ts +1 -0
- package/dist/transcript-dump.js +50 -0
- package/dist/transcript-log.d.ts +67 -0
- package/dist/transcript-log.js +199 -0
- package/dist/transcript-store.d.ts +6 -0
- package/dist/transcript-store.js +69 -0
- package/docker-compose.yml +83 -0
- package/exec-watcher.mjs +155 -0
- package/package.json +60 -0
- package/templates/.env.local.example +28 -0
- package/templates/Dockerfile +47 -0
- package/templates/docker-compose.yml +10 -0
- package/templates/openclaw.json +40 -0
package/dist/context.js
ADDED
|
@@ -0,0 +1,479 @@
|
|
|
1
|
+
import { createQaBusThread, getQaBusState, injectQaBusInboundMessage, pollQaBus, } from "@alignfirst/openclaw-channel-mock-core";
|
|
2
|
+
import { judgeCostUsd } from "./cost.js";
|
|
3
|
+
import { execInGateway } from "./exec-rpc.js";
|
|
4
|
+
import { judgeLLM, judgeLLMJson, judgeLLMRaw, } from "./judge.js";
|
|
5
|
+
import { parseAgentToolCalls } from "./transcript-log.js";
|
|
6
|
+
const BUS_URL = process.env.OPENCLAW_TEST_BUS_URL ?? "http://bus:43123";
|
|
7
|
+
export class AssertionError extends Error {
|
|
8
|
+
constructor(msg) {
|
|
9
|
+
super(msg);
|
|
10
|
+
this.name = "AssertionError";
|
|
11
|
+
}
|
|
12
|
+
}
|
|
13
|
+
export function createContext(params) {
|
|
14
|
+
const { channel, conversationId, emitSink } = params;
|
|
15
|
+
const startedAtIso = params.startedAtIso ?? new Date().toISOString();
|
|
16
|
+
const accountId = channel;
|
|
17
|
+
const entries = [];
|
|
18
|
+
const judgeUsages = [];
|
|
19
|
+
const mockHandlers = new Map();
|
|
20
|
+
const outboundWaiters = new Map();
|
|
21
|
+
const outboundByMessageId = new Map();
|
|
22
|
+
let nextEntrySeq = 0;
|
|
23
|
+
let currentEntry;
|
|
24
|
+
let lastCliMock;
|
|
25
|
+
let scenarioEnded = false;
|
|
26
|
+
const nextEntrySeqTs = () => ({
|
|
27
|
+
entrySeq: nextEntrySeq++,
|
|
28
|
+
ts: new Date().toISOString(),
|
|
29
|
+
});
|
|
30
|
+
const emit = (entry) => {
|
|
31
|
+
entries.push(entry);
|
|
32
|
+
emitSink?.({ type: "entry", entry });
|
|
33
|
+
return entry;
|
|
34
|
+
};
|
|
35
|
+
const emitAugment = (entrySeq, patch) => {
|
|
36
|
+
emitSink?.({ type: "augment", entrySeq, patch });
|
|
37
|
+
};
|
|
38
|
+
const setCurrentEntry = (entry) => {
|
|
39
|
+
currentEntry = entry;
|
|
40
|
+
if (entry.kind === "cliMock") {
|
|
41
|
+
lastCliMock = { atMs: Date.now(), entry: entry };
|
|
42
|
+
}
|
|
43
|
+
return entry;
|
|
44
|
+
};
|
|
45
|
+
const emitOutboundReceived = (m) => {
|
|
46
|
+
const entry = {
|
|
47
|
+
...nextEntrySeqTs(),
|
|
48
|
+
kind: "outboundReceived",
|
|
49
|
+
messageId: m.id,
|
|
50
|
+
text: m.text,
|
|
51
|
+
...(m.threadId !== undefined ? { threadId: m.threadId } : {}),
|
|
52
|
+
};
|
|
53
|
+
emit(entry);
|
|
54
|
+
setCurrentEntry(entry);
|
|
55
|
+
outboundByMessageId.set(m.id, entry);
|
|
56
|
+
const waiter = outboundWaiters.get(m.id);
|
|
57
|
+
if (waiter) {
|
|
58
|
+
outboundWaiters.delete(m.id);
|
|
59
|
+
waiter(entry);
|
|
60
|
+
}
|
|
61
|
+
};
|
|
62
|
+
const emitCliMock = (call) => {
|
|
63
|
+
const entry = { ...nextEntrySeqTs(), kind: "cliMock", call };
|
|
64
|
+
emit(entry);
|
|
65
|
+
setCurrentEntry(entry);
|
|
66
|
+
if (call.handlerError) {
|
|
67
|
+
const failure = {
|
|
68
|
+
name: call.handlerError.name,
|
|
69
|
+
message: call.handlerError.message,
|
|
70
|
+
...(call.handlerError.stack ? { stack: call.handlerError.stack } : {}),
|
|
71
|
+
source: "cliMock",
|
|
72
|
+
};
|
|
73
|
+
entry.failure = failure;
|
|
74
|
+
emitAugment(entry.entrySeq, { kind: "failure", failure });
|
|
75
|
+
}
|
|
76
|
+
};
|
|
77
|
+
const ctx = {
|
|
78
|
+
channel,
|
|
79
|
+
conversationId,
|
|
80
|
+
accountId,
|
|
81
|
+
busUrl: BUS_URL,
|
|
82
|
+
get currentEntry() {
|
|
83
|
+
return currentEntry;
|
|
84
|
+
},
|
|
85
|
+
get isScenarioEnded() {
|
|
86
|
+
return scenarioEnded;
|
|
87
|
+
},
|
|
88
|
+
markScenarioAsEnded: (reason) => {
|
|
89
|
+
if (scenarioEnded)
|
|
90
|
+
return;
|
|
91
|
+
scenarioEnded = true;
|
|
92
|
+
const message = reason !== undefined && reason.length > 0 ? `scenario ended: ${reason}` : "scenario ended";
|
|
93
|
+
emit({ ...nextEntrySeqTs(), kind: "scenarioLog", message });
|
|
94
|
+
},
|
|
95
|
+
log: ((arg) => {
|
|
96
|
+
if (typeof arg === "string") {
|
|
97
|
+
emit({ ...nextEntrySeqTs(), kind: "scenarioLog", message: arg });
|
|
98
|
+
return;
|
|
99
|
+
}
|
|
100
|
+
const note = {
|
|
101
|
+
ts: new Date().toISOString(),
|
|
102
|
+
label: arg.label,
|
|
103
|
+
extra: arg.extra,
|
|
104
|
+
};
|
|
105
|
+
arg.attachTo.scenarioLog = note;
|
|
106
|
+
emitAugment(arg.attachTo.entrySeq, { kind: "scenarioLog", scenarioLog: note });
|
|
107
|
+
}),
|
|
108
|
+
sendInbound: (input) => sendInbound({ emit, nextEntrySeqTs }, accountId, conversationId, input),
|
|
109
|
+
createThread: (input) => createThread(accountId, conversationId, input),
|
|
110
|
+
poll: (opts) => poll(accountId, opts),
|
|
111
|
+
waitForOutbound: (predicate, opts) => waitForOutbound({
|
|
112
|
+
accountId,
|
|
113
|
+
awaitEntry: (id) => awaitOutboundEntry(id),
|
|
114
|
+
getLastCliMock: () => lastCliMock,
|
|
115
|
+
}, predicate, opts),
|
|
116
|
+
expectNoOutbound: (predicate, opts) => expectNoOutbound(accountId, predicate, opts),
|
|
117
|
+
assertRegex: (actual, pattern, label) => assertRegex({ getCurrentEntry: () => currentEntry, emitAugment }, actual, pattern, label),
|
|
118
|
+
assertEqual: (actual, expected, label) => assertEqual({ getCurrentEntry: () => currentEntry, emitAugment }, actual, expected, label),
|
|
119
|
+
assertLength: (value, expected, label) => assertLength({ getCurrentEntry: () => currentEntry, emitAugment }, value, expected, label),
|
|
120
|
+
judgeLLM: (p) => callJudgeVerdict({ emitAugment, judgeUsages }, p),
|
|
121
|
+
judgeLLMJson: (p) => callJudgeJson(judgeUsages, p),
|
|
122
|
+
judgeLLMRaw: (p) => callJudgeRaw(judgeUsages, p),
|
|
123
|
+
getCursor,
|
|
124
|
+
mockCli: (name, handler, opts) => registerMockCli(mockHandlers, name, handler, opts),
|
|
125
|
+
execInGateway: (argv, opts) => execInGateway(argv, opts),
|
|
126
|
+
waitForAgentToolCall: (predicate, opts) => waitForAgentToolCall({ conversationId, startedAtIso, getCurrentEntry: () => currentEntry, emitAugment }, predicate, opts),
|
|
127
|
+
getAgentToolCalls: () => parseAgentToolCalls({ conversationId, startedAtIso }),
|
|
128
|
+
};
|
|
129
|
+
const internals = {
|
|
130
|
+
finalize: (opts) => {
|
|
131
|
+
const result = computeResult(entries, currentEntry, emitAugment, opts?.failure);
|
|
132
|
+
return { entries, judgeUsages, result };
|
|
133
|
+
},
|
|
134
|
+
emitOutboundReceived,
|
|
135
|
+
emitCliMock,
|
|
136
|
+
getMockHandlers: () => mockHandlers,
|
|
137
|
+
peekEntries: () => ({ entries }),
|
|
138
|
+
isScenarioEnded: () => scenarioEnded,
|
|
139
|
+
};
|
|
140
|
+
function awaitOutboundEntry(messageId) {
|
|
141
|
+
const existing = outboundByMessageId.get(messageId);
|
|
142
|
+
if (existing)
|
|
143
|
+
return Promise.resolve(existing);
|
|
144
|
+
return new Promise((resolve) => {
|
|
145
|
+
outboundWaiters.set(messageId, resolve);
|
|
146
|
+
});
|
|
147
|
+
}
|
|
148
|
+
return { ctx, internals };
|
|
149
|
+
}
|
|
150
|
+
function computeResult(entries, currentEntry, emitAugment, scenarioFailure) {
|
|
151
|
+
if (scenarioFailure && currentEntry && !currentEntry.failure) {
|
|
152
|
+
currentEntry.failure = scenarioFailure;
|
|
153
|
+
emitAugment(currentEntry.entrySeq, { kind: "failure", failure: scenarioFailure });
|
|
154
|
+
}
|
|
155
|
+
const failedEntry = findFailedEntry(entries);
|
|
156
|
+
if (failedEntry) {
|
|
157
|
+
return {
|
|
158
|
+
verdict: "fail",
|
|
159
|
+
cause: "failedEntry",
|
|
160
|
+
entrySeq: failedEntry.entrySeq,
|
|
161
|
+
message: `[entry #${failedEntry.entrySeq}] ${failedEntry.failure?.message ?? ""}`,
|
|
162
|
+
};
|
|
163
|
+
}
|
|
164
|
+
if (!scenarioFailure)
|
|
165
|
+
return { verdict: "pass" };
|
|
166
|
+
const result = {
|
|
167
|
+
verdict: "fail",
|
|
168
|
+
cause: "error",
|
|
169
|
+
source: scenarioFailure.source,
|
|
170
|
+
errorName: scenarioFailure.name,
|
|
171
|
+
message: scenarioFailure.message,
|
|
172
|
+
};
|
|
173
|
+
if (scenarioFailure.stack)
|
|
174
|
+
result.stack = scenarioFailure.stack;
|
|
175
|
+
return result;
|
|
176
|
+
}
|
|
177
|
+
function findFailedEntry(entries) {
|
|
178
|
+
for (const e of entries) {
|
|
179
|
+
if (e.kind === "scenarioLog")
|
|
180
|
+
continue;
|
|
181
|
+
if (e.failure)
|
|
182
|
+
return e;
|
|
183
|
+
}
|
|
184
|
+
return;
|
|
185
|
+
}
|
|
186
|
+
function pushAssertion(deps, target, record) {
|
|
187
|
+
if (!target)
|
|
188
|
+
return;
|
|
189
|
+
if (!target.assertions)
|
|
190
|
+
target.assertions = [];
|
|
191
|
+
target.assertions.push(record);
|
|
192
|
+
deps.emitAugment(target.entrySeq, { kind: "assertion", assertion: record });
|
|
193
|
+
}
|
|
194
|
+
function registerMockCli(handlers, name, handler, opts) {
|
|
195
|
+
const mode = opts?.mode ?? "register";
|
|
196
|
+
const exists = handlers.has(name);
|
|
197
|
+
if (mode === "register") {
|
|
198
|
+
if (exists)
|
|
199
|
+
throw new Error(`mockCli: handler for "${name}" already registered`);
|
|
200
|
+
handlers.set(name, handler);
|
|
201
|
+
return;
|
|
202
|
+
}
|
|
203
|
+
if (mode === "replace") {
|
|
204
|
+
if (!exists)
|
|
205
|
+
throw new Error(`mockCli: no handler for "${name}" to replace`);
|
|
206
|
+
handlers.set(name, handler);
|
|
207
|
+
return;
|
|
208
|
+
}
|
|
209
|
+
// "ifAbsent"
|
|
210
|
+
if (!exists)
|
|
211
|
+
handlers.set(name, handler);
|
|
212
|
+
}
|
|
213
|
+
async function sendInbound(deps, accountId, conversationId, input) {
|
|
214
|
+
const conversation = input.conversation ?? {
|
|
215
|
+
kind: "channel",
|
|
216
|
+
id: conversationId,
|
|
217
|
+
title: conversationId,
|
|
218
|
+
};
|
|
219
|
+
const r = await injectQaBusInboundMessage({
|
|
220
|
+
baseUrl: BUS_URL,
|
|
221
|
+
input: {
|
|
222
|
+
accountId,
|
|
223
|
+
conversation,
|
|
224
|
+
senderId: input.senderId,
|
|
225
|
+
senderName: input.senderName,
|
|
226
|
+
text: input.text,
|
|
227
|
+
threadId: input.threadId,
|
|
228
|
+
threadTitle: input.threadTitle,
|
|
229
|
+
},
|
|
230
|
+
});
|
|
231
|
+
const entry = {
|
|
232
|
+
...deps.nextEntrySeqTs(),
|
|
233
|
+
kind: "inboundSent",
|
|
234
|
+
messageId: r.message.id,
|
|
235
|
+
text: input.text,
|
|
236
|
+
senderId: input.senderId,
|
|
237
|
+
...(input.senderName !== undefined ? { senderName: input.senderName } : {}),
|
|
238
|
+
...(input.threadId !== undefined ? { threadId: input.threadId } : {}),
|
|
239
|
+
};
|
|
240
|
+
deps.emit(entry);
|
|
241
|
+
return { message: r.message, entry };
|
|
242
|
+
}
|
|
243
|
+
async function createThread(accountId, conversationId, input) {
|
|
244
|
+
const r = await createQaBusThread({
|
|
245
|
+
baseUrl: BUS_URL,
|
|
246
|
+
accountId,
|
|
247
|
+
conversationId,
|
|
248
|
+
title: input.title,
|
|
249
|
+
createdBy: input.createdBy,
|
|
250
|
+
});
|
|
251
|
+
return r.thread.id;
|
|
252
|
+
}
|
|
253
|
+
async function poll(accountId, opts) {
|
|
254
|
+
const timeoutMs = opts.timeoutMs ?? 1000;
|
|
255
|
+
// Client-side hard stop. The bus holds the long-poll for `timeoutMs` then answers, but a wedged
|
|
256
|
+
// bus or a half-open connection would hang this `fetch` forever — and callers loop on their own
|
|
257
|
+
// deadline (`waitForOutbound`), so a hung poll silently defeats that deadline and the runner never
|
|
258
|
+
// exits (the intermittent post-verdict hang). Abort a bit past the server hold so a stall surfaces
|
|
259
|
+
// as an empty poll and the caller's loop keeps ticking toward its real timeout.
|
|
260
|
+
const controller = new AbortController();
|
|
261
|
+
const abortTimer = setTimeout(() => controller.abort(), timeoutMs + 2_000);
|
|
262
|
+
let r;
|
|
263
|
+
try {
|
|
264
|
+
r = await pollQaBus({
|
|
265
|
+
baseUrl: BUS_URL,
|
|
266
|
+
accountId,
|
|
267
|
+
cursor: opts.sinceCursor,
|
|
268
|
+
timeoutMs,
|
|
269
|
+
signal: controller.signal,
|
|
270
|
+
});
|
|
271
|
+
}
|
|
272
|
+
catch (err) {
|
|
273
|
+
if (controller.signal.aborted)
|
|
274
|
+
return { messages: [], nextCursor: opts.sinceCursor };
|
|
275
|
+
throw err;
|
|
276
|
+
}
|
|
277
|
+
finally {
|
|
278
|
+
clearTimeout(abortTimer);
|
|
279
|
+
}
|
|
280
|
+
const messages = r.events
|
|
281
|
+
.filter((e) => e.kind === "outbound-message" || e.kind === "message-edited")
|
|
282
|
+
.map((e) => e.message);
|
|
283
|
+
return { messages, nextCursor: r.cursor };
|
|
284
|
+
}
|
|
285
|
+
export async function waitForOutbound(deps, predicate, opts) {
|
|
286
|
+
const timeoutMs = opts.timeoutMs ?? 30_000;
|
|
287
|
+
const maxUnmatched = opts.failFastUnmatchedOutbounds ?? 3;
|
|
288
|
+
// 60s: OpenClaw 2026.8 turns run noticeably longer between a CLI call and the
|
|
289
|
+
// turn-end post, and Slack sessions post nothing before turn end.
|
|
290
|
+
const cliMockGraceMs = opts.failFastCliMockGraceMs ?? 60_000;
|
|
291
|
+
const pollFn = opts.pollImpl ?? poll;
|
|
292
|
+
const now = opts.nowImpl ?? Date.now;
|
|
293
|
+
const deadline = now() + timeoutMs;
|
|
294
|
+
// The grace fail-fast is a liveness heuristic: it fires only when the newest
|
|
295
|
+
// activity this wait has observed is a mocked-CLI call that then went quiet.
|
|
296
|
+
// A CLI call from before the wait started never arms it, and any outbound
|
|
297
|
+
// observed after the call disarms it (the unmatched fail-fast polices those).
|
|
298
|
+
const baselineCliSeq = deps.getLastCliMock()?.entry.entrySeq ?? -1;
|
|
299
|
+
let lastOutboundAtMs;
|
|
300
|
+
let cursor = opts.sinceCursor;
|
|
301
|
+
let unmatched = 0;
|
|
302
|
+
let lastUnmatched;
|
|
303
|
+
while (now() < deadline) {
|
|
304
|
+
const remaining = Math.max(0, deadline - now());
|
|
305
|
+
const { messages, nextCursor } = await pollFn(deps.accountId, {
|
|
306
|
+
sinceCursor: cursor,
|
|
307
|
+
timeoutMs: Math.min(5000, remaining),
|
|
308
|
+
});
|
|
309
|
+
cursor = nextCursor;
|
|
310
|
+
const match = messages.find(predicate);
|
|
311
|
+
if (match) {
|
|
312
|
+
const entry = await deps.awaitEntry(match.id);
|
|
313
|
+
return { match, entry, nextCursor };
|
|
314
|
+
}
|
|
315
|
+
if (messages.length > 0) {
|
|
316
|
+
lastOutboundAtMs = now();
|
|
317
|
+
unmatched += messages.length;
|
|
318
|
+
lastUnmatched = messages[messages.length - 1];
|
|
319
|
+
if (maxUnmatched !== false && unmatched >= maxUnmatched) {
|
|
320
|
+
throw await unmatchedFastFailError(deps, unmatched, lastUnmatched);
|
|
321
|
+
}
|
|
322
|
+
}
|
|
323
|
+
if (cliMockGraceMs !== false) {
|
|
324
|
+
const last = deps.getLastCliMock();
|
|
325
|
+
const armed = last !== undefined &&
|
|
326
|
+
last.entry.entrySeq > baselineCliSeq &&
|
|
327
|
+
(lastOutboundAtMs === undefined || lastOutboundAtMs < last.atMs);
|
|
328
|
+
if (armed && now() - last.atMs >= cliMockGraceMs) {
|
|
329
|
+
throw cliMockGraceFastFailError(cliMockGraceMs, last.entry);
|
|
330
|
+
}
|
|
331
|
+
}
|
|
332
|
+
}
|
|
333
|
+
throw new AssertionError(`waitForOutbound timed out after ${timeoutMs}ms`);
|
|
334
|
+
}
|
|
335
|
+
function truncate(text, max = 80) {
|
|
336
|
+
const flat = text.replace(/\s+/g, " ");
|
|
337
|
+
if (flat.length <= max)
|
|
338
|
+
return flat;
|
|
339
|
+
return `${flat.slice(0, max - 1)}…`;
|
|
340
|
+
}
|
|
341
|
+
async function unmatchedFastFailError(deps, count, msg) {
|
|
342
|
+
const entry = await raceTimeout(deps.awaitEntry(msg.id), 50);
|
|
343
|
+
const seqOrId = entry ? `entrySeq=${entry.entrySeq}` : `msg=${msg.id}`;
|
|
344
|
+
const thread = msg.threadId !== undefined ? `threadId=${msg.threadId}` : "no threadId";
|
|
345
|
+
const text = truncate(msg.text);
|
|
346
|
+
return new AssertionError(`waitForOutbound: agent posted ${count} outbounds but none matched the predicate\n` +
|
|
347
|
+
` observed: outbound ${seqOrId} (${thread}, text=${JSON.stringify(text)})`);
|
|
348
|
+
}
|
|
349
|
+
function cliMockGraceFastFailError(graceMs, entry) {
|
|
350
|
+
const cli = entry.call.cli;
|
|
351
|
+
const argvHead = truncate(entry.call.argv.join(" "));
|
|
352
|
+
return new AssertionError(`waitForOutbound: agent invoked a mocked CLI during this wait and posted no outbound for ${graceMs}ms\n` +
|
|
353
|
+
` observed: cliMock ${cli} (argv head: ${JSON.stringify(argvHead)}), then silence`);
|
|
354
|
+
}
|
|
355
|
+
function raceTimeout(p, ms) {
|
|
356
|
+
return Promise.race([
|
|
357
|
+
p,
|
|
358
|
+
new Promise((resolve) => setTimeout(() => resolve(undefined), ms)),
|
|
359
|
+
]);
|
|
360
|
+
}
|
|
361
|
+
async function expectNoOutbound(accountId, predicate, opts) {
|
|
362
|
+
const deadline = Date.now() + opts.withinMs;
|
|
363
|
+
let cursor = opts.sinceCursor;
|
|
364
|
+
while (Date.now() < deadline) {
|
|
365
|
+
const remaining = Math.max(0, deadline - Date.now());
|
|
366
|
+
const { messages, nextCursor } = await poll(accountId, {
|
|
367
|
+
sinceCursor: cursor,
|
|
368
|
+
timeoutMs: Math.min(remaining, 1000),
|
|
369
|
+
});
|
|
370
|
+
cursor = nextCursor;
|
|
371
|
+
const offender = messages.find(predicate);
|
|
372
|
+
if (offender) {
|
|
373
|
+
throw new AssertionError(`expectNoOutbound: forbidden message arrived: ${JSON.stringify({
|
|
374
|
+
id: offender.id,
|
|
375
|
+
text: offender.text,
|
|
376
|
+
threadId: offender.threadId,
|
|
377
|
+
})}`);
|
|
378
|
+
}
|
|
379
|
+
}
|
|
380
|
+
return { nextCursor: cursor };
|
|
381
|
+
}
|
|
382
|
+
async function waitForAgentToolCall(deps, predicate, opts) {
|
|
383
|
+
const timeoutMs = opts.timeoutMs ?? 30_000;
|
|
384
|
+
// Each poll spawns a full transcript dump inside the gateway container —
|
|
385
|
+
// keep the cadence low enough not to load the system under test.
|
|
386
|
+
const pollMs = opts.pollMs ?? 2_000;
|
|
387
|
+
const deadline = Date.now() + timeoutMs;
|
|
388
|
+
let calls = [];
|
|
389
|
+
while (Date.now() < deadline) {
|
|
390
|
+
calls = await parseAgentToolCalls({
|
|
391
|
+
conversationId: deps.conversationId,
|
|
392
|
+
startedAtIso: deps.startedAtIso,
|
|
393
|
+
});
|
|
394
|
+
const match = calls.find(predicate);
|
|
395
|
+
if (match) {
|
|
396
|
+
pushAssertion(deps, deps.getCurrentEntry(), { label: opts.label, ok: true });
|
|
397
|
+
return match;
|
|
398
|
+
}
|
|
399
|
+
await new Promise((r) => setTimeout(r, pollMs));
|
|
400
|
+
}
|
|
401
|
+
const detail = `no agent tool call matched within ${timeoutMs}ms; observed: ${summarizeToolCalls(calls)}`;
|
|
402
|
+
pushAssertion(deps, deps.getCurrentEntry(), { label: opts.label, ok: false, detail });
|
|
403
|
+
throw new AssertionError(`${opts.label}: ${detail}`);
|
|
404
|
+
}
|
|
405
|
+
function summarizeToolCalls(calls) {
|
|
406
|
+
if (calls.length === 0)
|
|
407
|
+
return "(none)";
|
|
408
|
+
return calls.map((c) => `${c.toolName}${toolCallHint(c.input)}`).join("; ");
|
|
409
|
+
}
|
|
410
|
+
function toolCallHint(input) {
|
|
411
|
+
if (!isRecord(input))
|
|
412
|
+
return "";
|
|
413
|
+
if (typeof input.path === "string")
|
|
414
|
+
return ` ${input.path}`;
|
|
415
|
+
if (typeof input.command === "string")
|
|
416
|
+
return ` ${input.command.slice(0, 60)}`;
|
|
417
|
+
return "";
|
|
418
|
+
}
|
|
419
|
+
function isRecord(value) {
|
|
420
|
+
return typeof value === "object" && value !== null;
|
|
421
|
+
}
|
|
422
|
+
function assertRegex(deps, actual, pattern, label) {
|
|
423
|
+
failingAssert(deps, pattern.test(actual), label, `value=${JSON.stringify(actual)} pattern=${pattern.source}`);
|
|
424
|
+
}
|
|
425
|
+
function assertEqual(deps, actual, expected, label) {
|
|
426
|
+
failingAssert(deps, actual === expected, label, `expected=${JSON.stringify(expected)} actual=${JSON.stringify(actual)}`);
|
|
427
|
+
}
|
|
428
|
+
function assertLength(deps, value, expected, label) {
|
|
429
|
+
failingAssert(deps, value.length === expected, label, `expected length=${expected} actual=${value.length}`);
|
|
430
|
+
}
|
|
431
|
+
function failingAssert(deps, ok, label, detail) {
|
|
432
|
+
if (ok)
|
|
433
|
+
return;
|
|
434
|
+
pushAssertion(deps, deps.getCurrentEntry(), { label, ok: false, detail });
|
|
435
|
+
throw new AssertionError(`${label}: ${detail}`);
|
|
436
|
+
}
|
|
437
|
+
async function callJudgeVerdict(deps, p) {
|
|
438
|
+
const verdict = await judgeLLM({
|
|
439
|
+
message: p.message,
|
|
440
|
+
rubric: p.rubric,
|
|
441
|
+
maxTokens: p.maxTokens,
|
|
442
|
+
});
|
|
443
|
+
deps.judgeUsages.push(verdict.usage);
|
|
444
|
+
const extra = {
|
|
445
|
+
judge: {
|
|
446
|
+
model: verdict.usage.model,
|
|
447
|
+
usage: {
|
|
448
|
+
inputTokens: verdict.usage.inputTokens,
|
|
449
|
+
outputTokens: verdict.usage.outputTokens,
|
|
450
|
+
},
|
|
451
|
+
costUsd: judgeCostUsd(verdict.usage),
|
|
452
|
+
},
|
|
453
|
+
};
|
|
454
|
+
if (verdict.verdict === "pass") {
|
|
455
|
+
pushAssertion(deps, p.attachTo, { label: p.label, ok: true, extra });
|
|
456
|
+
return verdict;
|
|
457
|
+
}
|
|
458
|
+
pushAssertion(deps, p.attachTo, { label: p.label, ok: false, detail: verdict.reasoning, extra });
|
|
459
|
+
throw new AssertionError(`${p.label}: ${verdict.reasoning}`);
|
|
460
|
+
}
|
|
461
|
+
async function callJudgeJson(judgeUsages, p) {
|
|
462
|
+
const result = await judgeLLMJson({
|
|
463
|
+
message: p.message,
|
|
464
|
+
prompt: p.prompt,
|
|
465
|
+
returnType: p.returnType,
|
|
466
|
+
maxTokens: p.maxTokens,
|
|
467
|
+
});
|
|
468
|
+
judgeUsages.push(result.usage);
|
|
469
|
+
return result;
|
|
470
|
+
}
|
|
471
|
+
async function callJudgeRaw(judgeUsages, p) {
|
|
472
|
+
const result = await judgeLLMRaw(p.prompt, { maxTokens: p.maxTokens });
|
|
473
|
+
judgeUsages.push(result.usage);
|
|
474
|
+
return result;
|
|
475
|
+
}
|
|
476
|
+
async function getCursor() {
|
|
477
|
+
const snap = await getQaBusState(BUS_URL);
|
|
478
|
+
return snap.cursor;
|
|
479
|
+
}
|
package/dist/cost.d.ts
ADDED
package/dist/cost.js
ADDED
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
// USD per million tokens. Judge never uses prompt caching. Keys are LiteLLM-style
|
|
2
|
+
// "provider/model" refs. Add models as needed.
|
|
3
|
+
const JUDGE_PRICING = {
|
|
4
|
+
"anthropic/claude-haiku-4-5": { input: 1.0, output: 5.0 },
|
|
5
|
+
"openrouter/anthropic/claude-haiku-4.5": { input: 1.0, output: 5.0 },
|
|
6
|
+
};
|
|
7
|
+
const warnedUnknownModels = new Set();
|
|
8
|
+
export function judgeCostUsd(usage) {
|
|
9
|
+
const key = usage.model.replace(/-\d{8}$/, "");
|
|
10
|
+
const price = JUDGE_PRICING[key];
|
|
11
|
+
if (!price) {
|
|
12
|
+
if (!warnedUnknownModels.has(key)) {
|
|
13
|
+
warnedUnknownModels.add(key);
|
|
14
|
+
console.warn(`cost: no JUDGE_PRICING entry for ${JSON.stringify(key)}; judge cost will report 0. Add it to src/cost.ts.`);
|
|
15
|
+
}
|
|
16
|
+
return 0;
|
|
17
|
+
}
|
|
18
|
+
return ((usage.inputTokens * price.input) / 1_000_000 + (usage.outputTokens * price.output) / 1_000_000);
|
|
19
|
+
}
|
|
@@ -0,0 +1,5 @@
|
|
|
1
|
+
export declare function envCommand(packageDir: string, argv: string[]): Promise<never>;
|
|
2
|
+
export declare function buildConsumerImageArgs(composeArgs: string[]): string[];
|
|
3
|
+
export declare function runCommand(packageDir: string, argv: string[]): Promise<never>;
|
|
4
|
+
/** Flag wins over the env var; fallback 1. Both must parse to an integer ≥ 1. */
|
|
5
|
+
export declare function resolveParallel(flag: string | undefined, envValue: string | undefined): number;
|