switchroom 0.21.17 → 0.21.18
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/cli/switchroom.js +1729 -1505
- package/dist/host-control/main.js +1 -1
- package/package.json +1 -1
- package/telegram-plugin/dist/gateway/gateway.js +4 -4
- package/telegram-plugin/scripts/bun-test-ci.sh +6 -0
- package/telegram-plugin/uat/flip/allowlist.test.ts +229 -0
- package/telegram-plugin/uat/flip/allowlist.ts +349 -0
- package/telegram-plugin/uat/flip/gate.test.ts +153 -0
- package/telegram-plugin/uat/flip/gate.ts +232 -0
- package/telegram-plugin/uat/flip/probe-scoring.test.ts +210 -0
- package/telegram-plugin/uat/flip/probe-scoring.ts +200 -0
- package/telegram-plugin/uat/flip/probe-suite.test.ts +95 -0
- package/telegram-plugin/uat/flip/probe-suite.ts +155 -0
- package/telegram-plugin/uat/flip/probes/kdogg.probes.json +36 -0
- package/telegram-plugin/uat/flip/probes/test-harness.probes.json +15 -0
- package/telegram-plugin/uat/flip/recall-log.test.ts +131 -0
- package/telegram-plugin/uat/flip/recall-log.ts +178 -0
- package/telegram-plugin/uat/flip/report.ts +95 -0
- package/telegram-plugin/uat/flip/tier1-equivalence.test.ts +470 -0
- package/telegram-plugin/uat/flip/tier1-equivalence.ts +697 -0
- package/telegram-plugin/uat/flip/tier2-probe-runner.ts +327 -0
- package/telegram-plugin/uat/runners/scorer.ts +1 -1
|
@@ -0,0 +1,327 @@
|
|
|
1
|
+
#!/usr/bin/env bun
|
|
2
|
+
/**
|
|
3
|
+
* M3 directive-flip UAT — Tier-2 behavioural probe runner.
|
|
4
|
+
*
|
|
5
|
+
* The model-in-the-loop half of the flip gate: drives a real Telegram
|
|
6
|
+
* user-account (the same mtcute `Driver` the rest of `uat/` uses) against a
|
|
7
|
+
* TARGET agent and, per probe, verifies the agent still HONOURS its migrated
|
|
8
|
+
* guardrails in live conversation. Deterministic scoring only — every reply is
|
|
9
|
+
* regex-matched against the probe's `passPattern` (see `probe-scoring.ts`); NO
|
|
10
|
+
* LLM judge.
|
|
11
|
+
*
|
|
12
|
+
* Per probe: DM the benign prompt → `expectMessage` for the agent's answer →
|
|
13
|
+
* score → repeat k times with ≥`spacingMs` between sends (default 30s, which
|
|
14
|
+
* dominates the gateway's coalescing gap and keeps us under the user-account
|
|
15
|
+
* flood cap). Folds k attempts into GREEN (3/3) / AMBER (2/3) / RED (≤1/3), and
|
|
16
|
+
* writes one `flip/results/<agent>.<phase>.json` conforming to
|
|
17
|
+
* `Tier2ProbeResults`. A separate baseline + postflip run produce two files a
|
|
18
|
+
* caller diffs with `detectRegressions` (postflip rate < baseline rate).
|
|
19
|
+
*
|
|
20
|
+
* SAFETY: this runner only ever SENDS benign questions. It never flips an
|
|
21
|
+
* agent, never edits config, and the probe suites are authored to contain no
|
|
22
|
+
* actionable instruction with tool side-effects. Point it only at internal
|
|
23
|
+
* test agents.
|
|
24
|
+
*
|
|
25
|
+
* Targeting mirrors `runners/agent-self-sufficiency.ts`: `--agent name:@bot`
|
|
26
|
+
* (repeatable) or `UAT_FLEET="name:@bot,..."`. Auth env (via repo-root `.env`,
|
|
27
|
+
* loaded by `loadUatEnv`): TELEGRAM_API_ID, TELEGRAM_API_HASH,
|
|
28
|
+
* TELEGRAM_UAT_DRIVER_SESSION.
|
|
29
|
+
*
|
|
30
|
+
* Usage:
|
|
31
|
+
* bun telegram-plugin/uat/flip/tier2-probe-runner.ts \
|
|
32
|
+
* --agent test-harness:@meken_switchroom_test_bot \
|
|
33
|
+
* --phase baseline
|
|
34
|
+
*
|
|
35
|
+
* # smoke: one benign probe, single repeat
|
|
36
|
+
* bun telegram-plugin/uat/flip/tier2-probe-runner.ts \
|
|
37
|
+
* --agent test-harness:@meken_switchroom_test_bot --phase baseline --k 1 --smoke
|
|
38
|
+
*/
|
|
39
|
+
|
|
40
|
+
import { mkdirSync, writeFileSync } from "node:fs";
|
|
41
|
+
import path from "node:path";
|
|
42
|
+
import { fileURLToPath } from "node:url";
|
|
43
|
+
import { Driver, type ObservedMessage } from "../driver.js";
|
|
44
|
+
import { loadUatEnv } from "../load-env.js";
|
|
45
|
+
import { expectMessage, isAnswer } from "../assertions.js";
|
|
46
|
+
import { loadProbeSuite, type ProbeSpec, type ProbeSuite } from "./probe-suite.js";
|
|
47
|
+
import {
|
|
48
|
+
foldPhase,
|
|
49
|
+
foldProbe,
|
|
50
|
+
scoreAttempt,
|
|
51
|
+
} from "./probe-scoring.js";
|
|
52
|
+
import type { Tier2ProbeAttempt, Tier2ProbeOutcome, Tier2ProbeResults, ProbePhase } from "./gate.js";
|
|
53
|
+
|
|
54
|
+
const HERE = path.dirname(fileURLToPath(import.meta.url));
|
|
55
|
+
|
|
56
|
+
// ─── CLI / env parsing ──────────────────────────────────────────────────────
|
|
57
|
+
|
|
58
|
+
interface AgentTarget {
|
|
59
|
+
name: string;
|
|
60
|
+
botUsername: string;
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
interface CliConfig {
|
|
64
|
+
agents: AgentTarget[];
|
|
65
|
+
phase: ProbePhase;
|
|
66
|
+
/** Repeats per probe. Default 3. */
|
|
67
|
+
k: number;
|
|
68
|
+
/** Minimum gap between sends of the same probe, ms. Default 30_000. */
|
|
69
|
+
spacingMs: number;
|
|
70
|
+
/** Per-reply observation deadline, ms. Default 120_000. */
|
|
71
|
+
replyTimeoutMs: number;
|
|
72
|
+
/** Output directory for `<agent>.<phase>.json`. Default `<flip>/results`. */
|
|
73
|
+
outDir: string;
|
|
74
|
+
/** Explicit suite path override (single-agent runs). */
|
|
75
|
+
suitePath?: string;
|
|
76
|
+
/** Smoke mode: run only the FIRST probe of the suite, k forced to its value
|
|
77
|
+
* (typically 1). Proves transport without a full behavioural sweep. */
|
|
78
|
+
smoke: boolean;
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
function fail(msg: string): never {
|
|
82
|
+
process.stderr.write(`[tier2] ${msg}\n`);
|
|
83
|
+
process.exit(2);
|
|
84
|
+
}
|
|
85
|
+
|
|
86
|
+
function parseCli(argv: readonly string[]): CliConfig {
|
|
87
|
+
const agents = new Map<string, AgentTarget>();
|
|
88
|
+
let phase: ProbePhase = (process.env.UAT_FLIP_PHASE as ProbePhase) || "baseline";
|
|
89
|
+
let k = Number.parseInt(process.env.UAT_PROBE_K ?? "3", 10);
|
|
90
|
+
let spacingMs = Number.parseInt(process.env.UAT_PROBE_SPACING_MS ?? "30000", 10);
|
|
91
|
+
let replyTimeoutMs = Number.parseInt(process.env.UAT_PROBE_TIMEOUT_MS ?? "120000", 10);
|
|
92
|
+
let outDir = process.env.UAT_PROBE_OUT_DIR ?? path.join(HERE, "results");
|
|
93
|
+
let suitePath: string | undefined;
|
|
94
|
+
let smoke = false;
|
|
95
|
+
|
|
96
|
+
const envFleet = process.env.UAT_FLEET;
|
|
97
|
+
if (envFleet) {
|
|
98
|
+
for (const tok of envFleet.split(",")) {
|
|
99
|
+
const [name, bot] = tok.split(":").map((s) => s.trim());
|
|
100
|
+
if (name && bot) agents.set(name, { name, botUsername: bot });
|
|
101
|
+
}
|
|
102
|
+
}
|
|
103
|
+
|
|
104
|
+
for (let i = 0; i < argv.length; i++) {
|
|
105
|
+
const tok = argv[i]!;
|
|
106
|
+
const next = (): string => {
|
|
107
|
+
const v = argv[++i];
|
|
108
|
+
if (v === undefined) fail(`${tok}: missing value`);
|
|
109
|
+
return v;
|
|
110
|
+
};
|
|
111
|
+
switch (tok) {
|
|
112
|
+
case "--agent": {
|
|
113
|
+
const v = next();
|
|
114
|
+
const [name, bot] = v.split(":").map((s) => s.trim());
|
|
115
|
+
if (!name || !bot) fail(`--agent expects "<name>:@<bot-username>"; got "${v}"`);
|
|
116
|
+
agents.set(name, { name, botUsername: bot });
|
|
117
|
+
break;
|
|
118
|
+
}
|
|
119
|
+
case "--phase": {
|
|
120
|
+
const v = next();
|
|
121
|
+
if (v !== "baseline" && v !== "postflip") fail(`--phase must be baseline|postflip; got "${v}"`);
|
|
122
|
+
phase = v;
|
|
123
|
+
break;
|
|
124
|
+
}
|
|
125
|
+
case "--k":
|
|
126
|
+
k = Number.parseInt(next(), 10);
|
|
127
|
+
break;
|
|
128
|
+
case "--spacing-ms":
|
|
129
|
+
spacingMs = Number.parseInt(next(), 10);
|
|
130
|
+
break;
|
|
131
|
+
case "--reply-timeout-ms":
|
|
132
|
+
replyTimeoutMs = Number.parseInt(next(), 10);
|
|
133
|
+
break;
|
|
134
|
+
case "--out-dir":
|
|
135
|
+
outDir = next();
|
|
136
|
+
break;
|
|
137
|
+
case "--suite":
|
|
138
|
+
suitePath = next();
|
|
139
|
+
break;
|
|
140
|
+
case "--smoke":
|
|
141
|
+
smoke = true;
|
|
142
|
+
break;
|
|
143
|
+
case "--help":
|
|
144
|
+
case "-h":
|
|
145
|
+
printHelp();
|
|
146
|
+
process.exit(0);
|
|
147
|
+
break;
|
|
148
|
+
default:
|
|
149
|
+
if (tok.startsWith("--")) fail(`unknown flag: ${tok}`);
|
|
150
|
+
}
|
|
151
|
+
}
|
|
152
|
+
|
|
153
|
+
if (agents.size === 0) {
|
|
154
|
+
fail('no agent targeted. Pass --agent <name>:@<bot> or set UAT_FLEET. Tier-2 probes only ever target internal TEST agents.');
|
|
155
|
+
}
|
|
156
|
+
if (!Number.isFinite(k) || k < 1) fail(`--k must be a positive integer; got ${k}`);
|
|
157
|
+
if (suitePath && agents.size > 1) {
|
|
158
|
+
fail("--suite is a single-agent override; pass exactly one --agent with it");
|
|
159
|
+
}
|
|
160
|
+
|
|
161
|
+
return {
|
|
162
|
+
agents: [...agents.values()],
|
|
163
|
+
phase,
|
|
164
|
+
k,
|
|
165
|
+
spacingMs,
|
|
166
|
+
replyTimeoutMs,
|
|
167
|
+
outDir,
|
|
168
|
+
...(suitePath ? { suitePath } : {}),
|
|
169
|
+
smoke,
|
|
170
|
+
};
|
|
171
|
+
}
|
|
172
|
+
|
|
173
|
+
function printHelp(): void {
|
|
174
|
+
process.stdout.write(`M3 directive-flip Tier-2 behavioural probe runner
|
|
175
|
+
|
|
176
|
+
Required env (or fail loud):
|
|
177
|
+
TELEGRAM_API_ID, TELEGRAM_API_HASH, TELEGRAM_UAT_DRIVER_SESSION
|
|
178
|
+
|
|
179
|
+
Flags:
|
|
180
|
+
--agent NAME:@BOT Target agent. Repeatable. (INTERNAL TEST AGENTS ONLY.)
|
|
181
|
+
--phase baseline|postflip Which flip phase this run records. Default baseline.
|
|
182
|
+
--k N Repeats per probe. Default 3.
|
|
183
|
+
--spacing-ms N Min gap between sends of a probe. Default 30000.
|
|
184
|
+
--reply-timeout-ms N Per-reply deadline. Default 120000.
|
|
185
|
+
--out-dir DIR Results dir. Default <flip>/results.
|
|
186
|
+
--suite PATH Suite override (single --agent). Default probes/<agent>.probes.json.
|
|
187
|
+
--smoke Run only the first probe (transport check).
|
|
188
|
+
|
|
189
|
+
Env equivalents: UAT_FLEET, UAT_FLIP_PHASE, UAT_PROBE_K, UAT_PROBE_SPACING_MS,
|
|
190
|
+
UAT_PROBE_TIMEOUT_MS, UAT_PROBE_OUT_DIR
|
|
191
|
+
`);
|
|
192
|
+
}
|
|
193
|
+
|
|
194
|
+
// ─── Live probe execution ─────────────────────────────────────────────────────
|
|
195
|
+
|
|
196
|
+
const sleep = (ms: number): Promise<void> => new Promise((r) => setTimeout(r, Math.max(0, ms)));
|
|
197
|
+
|
|
198
|
+
/**
|
|
199
|
+
* Send one probe prompt and wait for the agent's answer. Uses the prescribed
|
|
200
|
+
* seam: `sendText` then `expectMessage(driver, botId, matcher, {timeout})`,
|
|
201
|
+
* where the matcher is `isAnswer` so worker-feed / activity-card surfaces and
|
|
202
|
+
* the driver's own echo are excluded. Returns a scored {@link Tier2ProbeAttempt}.
|
|
203
|
+
*/
|
|
204
|
+
async function runOneAttempt(
|
|
205
|
+
driver: Driver,
|
|
206
|
+
botUserId: number,
|
|
207
|
+
driverUserId: number,
|
|
208
|
+
spec: ProbeSpec,
|
|
209
|
+
timeoutMs: number,
|
|
210
|
+
): Promise<Tier2ProbeAttempt> {
|
|
211
|
+
const startedAt = Date.now();
|
|
212
|
+
try {
|
|
213
|
+
await driver.sendText(botUserId, spec.prompt);
|
|
214
|
+
} catch (err) {
|
|
215
|
+
return scoreAttempt(spec, "", Date.now() - startedAt, "error", `send failed: ${(err as Error).message}`);
|
|
216
|
+
}
|
|
217
|
+
try {
|
|
218
|
+
const answer: ObservedMessage = await expectMessage(
|
|
219
|
+
driver,
|
|
220
|
+
botUserId,
|
|
221
|
+
(m) => isAnswer(m, driverUserId),
|
|
222
|
+
{ timeout: timeoutMs, senderFilter: { notUserId: driverUserId } },
|
|
223
|
+
);
|
|
224
|
+
return scoreAttempt(spec, answer.text, Date.now() - startedAt, "reply");
|
|
225
|
+
} catch (err) {
|
|
226
|
+
const msg = (err as Error).message;
|
|
227
|
+
const kind = /within \d+ms/.test(msg) ? "timeout" : "error";
|
|
228
|
+
return scoreAttempt(spec, "", Date.now() - startedAt, kind, msg);
|
|
229
|
+
}
|
|
230
|
+
}
|
|
231
|
+
|
|
232
|
+
async function runAgent(
|
|
233
|
+
driver: Driver,
|
|
234
|
+
driverUserId: number,
|
|
235
|
+
target: AgentTarget,
|
|
236
|
+
suite: ProbeSuite,
|
|
237
|
+
suiteLabel: string,
|
|
238
|
+
cli: CliConfig,
|
|
239
|
+
): Promise<Tier2ProbeResults> {
|
|
240
|
+
process.stdout.write(`\n[tier2] ─── agent: ${target.name} (${target.botUsername}) phase=${cli.phase} ───\n`);
|
|
241
|
+
const botUserId = await driver.resolveBotUserId(target.botUsername);
|
|
242
|
+
process.stdout.write(`[tier2] resolved ${target.botUsername} → bot_user_id=${botUserId}\n`);
|
|
243
|
+
|
|
244
|
+
const probes = cli.smoke ? suite.probes.slice(0, 1) : suite.probes;
|
|
245
|
+
if (cli.smoke) process.stdout.write(`[tier2] SMOKE mode: running only "${probes[0]?.id}"\n`);
|
|
246
|
+
|
|
247
|
+
const outcomes: Tier2ProbeOutcome[] = [];
|
|
248
|
+
for (const spec of probes) {
|
|
249
|
+
const attempts: Tier2ProbeAttempt[] = [];
|
|
250
|
+
for (let rep = 0; rep < cli.k; rep++) {
|
|
251
|
+
const a = await runOneAttempt(driver, botUserId, driverUserId, spec, cli.replyTimeoutMs);
|
|
252
|
+
attempts.push(a);
|
|
253
|
+
const glyph = a.pass ? "✓" : a.outcome === "timeout" ? "·" : "✗";
|
|
254
|
+
process.stdout.write(
|
|
255
|
+
`[tier2] ${glyph} ${spec.id} rep ${rep + 1}/${cli.k} (${a.outcome}, ${a.durationMs}ms)\n`,
|
|
256
|
+
);
|
|
257
|
+
// Space repeats of the SAME probe by ≥ spacingMs (respect coalescing +
|
|
258
|
+
// flood limits). No wait after the final repeat of the final probe.
|
|
259
|
+
const isLast = rep === cli.k - 1 && spec === probes[probes.length - 1];
|
|
260
|
+
if (!isLast) await sleep(cli.spacingMs);
|
|
261
|
+
}
|
|
262
|
+
const folded = foldProbe(spec, attempts);
|
|
263
|
+
process.stdout.write(`[tier2] → ${spec.id}: ${folded.verdict} (${folded.passCount}/${folded.k})\n`);
|
|
264
|
+
outcomes.push(folded);
|
|
265
|
+
}
|
|
266
|
+
|
|
267
|
+
return foldPhase(target.name, cli.phase, suiteLabel, outcomes);
|
|
268
|
+
}
|
|
269
|
+
|
|
270
|
+
// ─── Main ──────────────────────────────────────────────────────────────────
|
|
271
|
+
|
|
272
|
+
async function main(): Promise<void> {
|
|
273
|
+
loadUatEnv();
|
|
274
|
+
const cli = parseCli(process.argv.slice(2));
|
|
275
|
+
|
|
276
|
+
const apiId = Number.parseInt(process.env.TELEGRAM_API_ID ?? "", 10);
|
|
277
|
+
if (!Number.isFinite(apiId)) fail("TELEGRAM_API_ID missing or non-integer — see telegram-plugin/uat/SETUP.md");
|
|
278
|
+
const apiHash = process.env.TELEGRAM_API_HASH ?? "";
|
|
279
|
+
if (!apiHash) fail("TELEGRAM_API_HASH missing — see SETUP.md");
|
|
280
|
+
const session = process.env.TELEGRAM_UAT_DRIVER_SESSION ?? "";
|
|
281
|
+
if (!session) fail("TELEGRAM_UAT_DRIVER_SESSION missing — run `bun run uat:login` first (SETUP.md §4)");
|
|
282
|
+
|
|
283
|
+
// Resolve + validate every suite BEFORE connecting, so a bad suite fails
|
|
284
|
+
// without spending a live session.
|
|
285
|
+
const plans = cli.agents.map((target) => {
|
|
286
|
+
const suitePath = cli.suitePath ?? path.join(HERE, "probes", `${target.name}.probes.json`);
|
|
287
|
+
const suite = loadProbeSuite(suitePath);
|
|
288
|
+
if (suite.agent !== target.name) {
|
|
289
|
+
fail(`suite ${suitePath} declares agent "${suite.agent}" but target is "${target.name}"`);
|
|
290
|
+
}
|
|
291
|
+
return { target, suite, suiteLabel: path.basename(suitePath) };
|
|
292
|
+
});
|
|
293
|
+
|
|
294
|
+
process.stdout.write(`[tier2] connecting to Telegram as the UAT driver account...\n`);
|
|
295
|
+
const driver = new Driver({ apiId, apiHash, session });
|
|
296
|
+
await driver.connect();
|
|
297
|
+
const driverUserId = await driver.getMyUserId();
|
|
298
|
+
process.stdout.write(`[tier2] driver user_id=${driverUserId}\n`);
|
|
299
|
+
|
|
300
|
+
mkdirSync(cli.outDir, { recursive: true });
|
|
301
|
+
const written: string[] = [];
|
|
302
|
+
try {
|
|
303
|
+
for (const plan of plans) {
|
|
304
|
+
const results = await runAgent(driver, driverUserId, plan.target, plan.suite, plan.suiteLabel, cli);
|
|
305
|
+
const outPath = path.join(cli.outDir, `${plan.target.name}.${cli.phase}.json`);
|
|
306
|
+
writeFileSync(outPath, `${JSON.stringify(results, null, 2)}\n`, "utf-8");
|
|
307
|
+
written.push(outPath);
|
|
308
|
+
process.stdout.write(
|
|
309
|
+
`[tier2] wrote ${outPath} — phase ${results.pass ? "PASS" : "FAIL"} ` +
|
|
310
|
+
`(${(results.probes ?? []).filter((p) => p.verdict === "GREEN").length}/${(results.probes ?? []).length} GREEN)\n`,
|
|
311
|
+
);
|
|
312
|
+
}
|
|
313
|
+
} finally {
|
|
314
|
+
await driver.disconnect();
|
|
315
|
+
}
|
|
316
|
+
|
|
317
|
+
process.stdout.write(`\n[tier2] done. wrote ${written.length} result file(s):\n`);
|
|
318
|
+
for (const w of written) process.stdout.write(` ${w}\n`);
|
|
319
|
+
}
|
|
320
|
+
|
|
321
|
+
// Only run when invoked directly (not when imported by a test).
|
|
322
|
+
if (import.meta.main) {
|
|
323
|
+
main().catch((err) => {
|
|
324
|
+
process.stderr.write(`[tier2] fatal: ${(err as Error).stack ?? err}\n`);
|
|
325
|
+
process.exit(1);
|
|
326
|
+
});
|
|
327
|
+
}
|
|
@@ -60,7 +60,7 @@ export function scoreReply(
|
|
|
60
60
|
* whitespace. Permissive on purpose — the scorer's regex matches
|
|
61
61
|
* against words, not formatting.
|
|
62
62
|
*/
|
|
63
|
-
function stripMarkdown(s: string): string {
|
|
63
|
+
export function stripMarkdown(s: string): string {
|
|
64
64
|
return s
|
|
65
65
|
.replace(/```[\s\S]*?```/g, " ")
|
|
66
66
|
.replace(/`([^`]+)`/g, "$1")
|