@el4cteo/rbx-studio-mcp 0.4.6 → 0.5.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,529 @@
1
+ /**
2
+ * Checks the console panel's command line and the agents it starts.
3
+ *
4
+ * Everything here is a pure function or a call against a real Bridge with fake
5
+ * Studio sessions -- no sockets, no Studio, and above all no agent processes.
6
+ * That last one is the constraint that shapes the file: the interesting code is
7
+ * "start a coding agent and stream it back", and a test that actually did so
8
+ * would cost money, need an API key, and take a minute. So the seam is
9
+ * `harness.read`, which turns one line of a harness's output into console rows
10
+ * and is where every adapter's real work lives.
11
+ *
12
+ * The event shapes below are recorded from live runs of each CLI, not invented.
13
+ * A test built on a guessed envelope passes forever and proves nothing.
14
+ */
15
+ import assert from "node:assert/strict";
16
+ import { readFileSync } from "node:fs";
17
+ import { Bridge } from "../dist/bridge/rpc.js";
18
+ import { LocalBridge } from "../dist/bridge/api.js";
19
+ import { frame, handleConsole } from "../dist/bridge/console.js";
20
+ import { find, installed, matchesClient } from "../dist/bridge/harness.js";
21
+
22
+ let checks = 0;
23
+ const ok = (condition, what) => {
24
+ assert.ok(condition, what);
25
+ checks += 1;
26
+ };
27
+
28
+ /** Every row a harness produces for one line of its output. */
29
+ const readAll = (id, lines) => {
30
+ const harness = find(id);
31
+ const rows = [];
32
+ let session = null;
33
+ for (const line of lines) {
34
+ const reading = harness.read(line);
35
+ if (reading.session !== undefined) session = reading.session;
36
+ rows.push(...reading.lines);
37
+ }
38
+ return { rows, session };
39
+ };
40
+
41
+ // --- Claude Code -----------------------------------------------------------
42
+ {
43
+ const { rows, session } = readAll("claude", [
44
+ JSON.stringify({ type: "system", subtype: "init", session_id: "abc-123" }),
45
+ JSON.stringify({
46
+ type: "assistant",
47
+ message: {
48
+ content: [
49
+ { type: "text", text: "Looking at the place." },
50
+ {
51
+ type: "tool_use",
52
+ name: "mcp__rbx-studio__create",
53
+ input: { instances: [], parent: "Workspace" },
54
+ },
55
+ ],
56
+ },
57
+ }),
58
+ "not json at all",
59
+ JSON.stringify({ type: "result", duration_ms: 5600, total_cost_usd: 0.1282 }),
60
+ ]);
61
+
62
+ ok(session === "abc-123", "claude: session id is learned from the init event");
63
+ // Four rows, not three: the "not json at all" line is SHOWN. This assertion
64
+ // used to require it be dropped, which is the same instinct that made
65
+ // opencode print nothing -- a line we cannot parse is still evidence, and the
66
+ // panel is the only place the user can see it.
67
+ ok(rows.length === 4, "claude: init contributes no row, but junk is surfaced");
68
+ ok(
69
+ rows.some((row) => row.level === "dim" && row.message === "not json at all"),
70
+ "claude: an unparseable line is shown dim rather than swallowed",
71
+ );
72
+ ok(rows[0].level === "reply" && rows[0].message === "Looking at the place.", "claude: prose");
73
+ ok(rows[1].level === "call" && rows[1].message === "create", "claude: server prefix is stripped");
74
+ ok(rows[3].level === "ok" && rows[3].message === "agent done", "claude: result row");
75
+ ok(rows[3].detail === "5.6s $0.1282", "claude: duration and cost ride the detail column");
76
+
77
+ // A failure must not be reported as a completion. This is the one row a user
78
+ // reads to decide whether to trust what just happened to their place.
79
+ const failed = readAll("claude", [
80
+ JSON.stringify({ type: "result", is_error: true, duration_ms: 100 }),
81
+ ]);
82
+ ok(failed.rows[0].level === "error", "claude: is_error becomes an error row");
83
+ }
84
+
85
+ // --- Tool arguments --------------------------------------------------------
86
+ {
87
+ // Regression: a Luau snippet passed to execute_luau is multi-line, and 32
88
+ // characters of it used to carry a newline into a log whose rows are lines.
89
+ // The visible symptom was a stray "m" sitting at column 0 under the entry.
90
+ const { rows } = readAll("claude", [
91
+ JSON.stringify({
92
+ type: "assistant",
93
+ message: {
94
+ content: [
95
+ {
96
+ type: "tool_use",
97
+ name: "mcp__rbx-studio__execute_luau",
98
+ input: { source: "local m = workspace.SmallHouse\nm.Parent = nil\nprint(m)" },
99
+ },
100
+ ],
101
+ },
102
+ }),
103
+ ]);
104
+ ok(!rows[0].detail.includes("\n"), "tool detail never contains a newline");
105
+ ok(rows[0].message === "execute_luau", "tool name keeps its own underscores");
106
+ }
107
+
108
+ // --- every adapter, against its real envelope -------------------------------
109
+ //
110
+ // This block used to assert INVENTED event names -- `session.created` for
111
+ // codex, `message.part.updated` for opencode -- while the file's own header
112
+ // claimed the shapes were recorded from live runs. They were not, and the tests
113
+ // passed anyway, because a test written against the same wrong envelope as the
114
+ // code agrees with it perfectly.
115
+ //
116
+ // What that cost: a prompt sent to opencode printed NOTHING. Every line fell
117
+ // through to no rows, the panel logged a blank run and went idle, and the
118
+ // agent's answer was thrown away. The lines below are copied from real runs
119
+ // (opencode 1.18.30, claude 2.1.266) or from the vendors' own documented
120
+ // schemas for the CLIs not installed here.
121
+ {
122
+ // opencode 1.18.30, verbatim from `opencode run --format json`.
123
+ const oc = readAll("opencode", [
124
+ JSON.stringify({ type: "step_start", sessionID: "ses_abc", part: { type: "step-start" } }),
125
+ JSON.stringify({
126
+ type: "tool_use",
127
+ sessionID: "ses_abc",
128
+ part: { type: "tool", tool: "glob", state: { input: { pattern: "*.ts" } } },
129
+ }),
130
+ JSON.stringify({
131
+ type: "text",
132
+ sessionID: "ses_abc",
133
+ part: { type: "text", text: "Hi there, how's it going?" },
134
+ }),
135
+ JSON.stringify({ type: "step_finish", sessionID: "ses_abc", part: { type: "step-finish" } }),
136
+ ]);
137
+ ok(oc.session === "ses_abc", "opencode: session id rides on every event");
138
+ ok(
139
+ oc.rows.some((row) => row.level === "reply" && row.message === "Hi there, how's it going?"),
140
+ "opencode: the answer is printed -- the bug was that it never was",
141
+ );
142
+ ok(oc.rows.some((row) => row.message === "glob"), "opencode: tool calls are printed");
143
+ ok(
144
+ !oc.rows.some((row) => row.message === "agent done"),
145
+ "opencode: step_finish is per step, not the end of the run",
146
+ );
147
+
148
+ // Codex, from the documented exec --json protocol.
149
+ const codex = readAll("codex", [
150
+ JSON.stringify({ type: "thread.started", thread_id: "019cec77-af02" }),
151
+ JSON.stringify({ type: "turn.started" }),
152
+ JSON.stringify({
153
+ type: "item.started",
154
+ item: { id: "i1", type: "mcp_tool_call", server: "rbx", tool: "create", arguments: {} },
155
+ }),
156
+ JSON.stringify({
157
+ type: "item.completed",
158
+ item: { id: "i1", type: "mcp_tool_call", server: "rbx", tool: "create" },
159
+ }),
160
+ JSON.stringify({ type: "item.completed", item: { id: "i2", type: "agent_message", text: "Done." } }),
161
+ JSON.stringify({ type: "turn.completed", usage: {} }),
162
+ ]);
163
+ ok(codex.session === "019cec77-af02", "codex: session comes from thread.started/thread_id");
164
+ ok(
165
+ codex.rows.some((row) => row.level === "reply" && row.message === "Done."),
166
+ "codex: prose",
167
+ );
168
+ ok(
169
+ codex.rows.filter((row) => row.message === "create").length === 1,
170
+ "codex: a tool reported started AND completed is logged once",
171
+ );
172
+ ok(codex.rows.some((row) => row.message === "agent done"), "codex: turn.completed ends the run");
173
+
174
+ // Codex global flags must precede the `resume` subcommand or it refuses them.
175
+ const resumed = find("codex").argv("hello", "thread-1");
176
+ ok(
177
+ resumed.indexOf("--skip-git-repo-check") < resumed.indexOf("resume"),
178
+ "codex: global flags come before the resume subcommand",
179
+ );
180
+ ok(resumed[resumed.length - 1] === "hello", "codex: the prompt stays last");
181
+
182
+ // Gemini, from the documented headless stream-json events.
183
+ const gem = readAll("gemini", [
184
+ JSON.stringify({ type: "init", session_id: "gem-1", model: "gemini" }),
185
+ JSON.stringify({ type: "message", role: "user", content: "what did I ask" }),
186
+ JSON.stringify({ type: "message", role: "assistant", content: "Built it." }),
187
+ JSON.stringify({ type: "result" }),
188
+ ]);
189
+ ok(gem.session === "gem-1", "gemini: session id");
190
+ ok(gem.rows.some((row) => row.message === "Built it."), "gemini: the assistant half is printed");
191
+ ok(
192
+ !gem.rows.some((row) => row.message === "what did I ask"),
193
+ "gemini: the user half is not echoed back at them",
194
+ );
195
+
196
+ // Cursor, from the documented stream-json envelope.
197
+ const cur = readAll("cursor", [
198
+ JSON.stringify({ type: "system", subtype: "init", session_id: "cur-1" }),
199
+ JSON.stringify({
200
+ type: "tool_call",
201
+ subtype: "started",
202
+ session_id: "cur-1",
203
+ tool_call: { readToolCall: { args: { path: "a.ts" } } },
204
+ }),
205
+ JSON.stringify({
206
+ type: "assistant",
207
+ session_id: "cur-1",
208
+ message: { role: "assistant", content: [{ type: "text", text: "Read it." }] },
209
+ }),
210
+ JSON.stringify({ type: "result", subtype: "success", is_error: false, session_id: "cur-1" }),
211
+ ]);
212
+ ok(cur.session === "cur-1", "cursor: session id rides on every event");
213
+ ok(cur.rows.some((row) => row.message === "Read it."), "cursor: text is nested in message.content");
214
+ ok(cur.rows.some((row) => row.message === "readToolCall"), "cursor: the tool key names the call");
215
+ ok(cur.rows.some((row) => row.message === "agent done"), "cursor: result ends the run");
216
+
217
+ // Crush has no event stream, but it does have its own verb -- the generic
218
+ // adapter guessed `-p`, which crush rejects outright as an unknown flag.
219
+ const crush = find("crush").argv("hello", null);
220
+ ok(crush[0] === "run", "crush: uses its `run` verb, not a guessed -p flag");
221
+ ok(
222
+ readAll("crush", ["Placed the model."]).rows[0].message === "Placed the model.",
223
+ "crush: plain text still reaches the log",
224
+ );
225
+ }
226
+
227
+ // --- DeepSeek Harness ------------------------------------------------------
228
+ {
229
+ const dsh = find("dsh");
230
+ const argv = dsh.argv("add a spawn point", null);
231
+
232
+ // `dsh [options] [command] [args...]`: --patch is a launcher option and the
233
+ // prompt is a positional argument, so the flag has to come first. Getting
234
+ // this backwards is silent -- dsh reads the prompt as the patch path.
235
+ const patchAt = argv.indexOf("--patch");
236
+ ok(patchAt !== -1, "dsh: the overlay is passed");
237
+ ok(
238
+ patchAt < argv.indexOf("add a spawn point"),
239
+ "dsh: --patch comes before the prompt, or dsh reads the prompt as a path",
240
+ );
241
+ ok(argv[argv.length - 1] === "add a spawn point", "dsh: the prompt is last");
242
+ ok(dsh.mcpFlag === undefined, "dsh: builds its own argv rather than appending flags");
243
+
244
+ const { rows } = readAll("dsh", ["The spawn point is placed.", " ", ""]);
245
+ ok(rows.length === 1, "dsh: blank lines are not rows");
246
+ ok(rows[0].level === "reply", "dsh: headless prints prose, not events");
247
+ }
248
+
249
+ // --- The registry ----------------------------------------------------------
250
+ {
251
+ ok(find("claude") !== undefined && find("nonesuch") === undefined, "registry: lookup by id");
252
+ ok(
253
+ installed().every((entry) => typeof entry.id === "string" && entry.id.length > 0),
254
+ "registry: every detected harness is named",
255
+ );
256
+ }
257
+
258
+ // --- Panel-started agents are not other people's clients -------------------
259
+ {
260
+ const bridge = new Bridge();
261
+ const announcements = [];
262
+ bridge.watchClients((count) => announcements.push(count));
263
+
264
+ const mine = new LocalBridge(bridge);
265
+ ok(bridge.clientCount() === 1, "an ordinary client counts");
266
+
267
+ // What a spawned agent's own server reports when it says hello. Counting it
268
+ // flashed the badge to 2 and logged an arrival and a departure around every
269
+ // single prompt -- around output the user was trying to read.
270
+ bridge.noteClient("spawned-agent", { name: "claude-code", pid: 42, spawned: true });
271
+ ok(bridge.clientCount() === 1, "a panel-started agent is not counted");
272
+ ok(
273
+ bridge.clientList().every((client) => client.name !== "claude-code"),
274
+ "a panel-started agent is not in the roster",
275
+ );
276
+ ok(
277
+ announcements.every((count) => count <= 1),
278
+ "a panel-started agent never announces an arrival",
279
+ );
280
+
281
+ mine.goodbye();
282
+ }
283
+
284
+ // --- Command routing -------------------------------------------------------
285
+ {
286
+ const bridge = new Bridge();
287
+ const identity = (studioId, placeId, placeName) => ({
288
+ studioId,
289
+ placeName,
290
+ placeId,
291
+ pluginVersion: "test",
292
+ buildId: "test",
293
+ protocolVersion: 1,
294
+ transport: "poll",
295
+ context: "edit",
296
+ });
297
+ bridge.attach(identity("studio-a", 111, "Alpha"), null);
298
+ bridge.attach(identity("studio-b", 222, "Beta"), null);
299
+
300
+ const run = (command, args = []) =>
301
+ handleConsole(bridge, 44755, { studioId: "studio-a", command, args, line: command });
302
+
303
+ const studios = await run("studios");
304
+ ok(studios.length === 3, "studios: a heading and one row per Studio");
305
+ ok(
306
+ studios.some((row) => row.message.includes("(this panel)")),
307
+ "studios: the asking panel is marked, so two rows with one place name are told apart",
308
+ );
309
+
310
+ // `use` takes the number printed by `studios`, because requiring the id would
311
+ // mean reading a hex string off one line to type it into the next.
312
+ const used = await run("use", ["2"]);
313
+ ok(used[0].level === "ok" && used[0].message.includes("Beta"), "use: resolves a list number");
314
+ ok(bridge.activeId("nobody-in-particular") === "studio-b", "use: applies to clients too");
315
+
316
+ const bad = await run("use", ["nope"]);
317
+ ok(bad[0].level === "error", "use: an unknown target is an error, not a silent no-op");
318
+ ok(bridge.activeId("nobody") === "studio-b", "use: a failed switch changes nothing");
319
+
320
+ const noArg = await run("use");
321
+ ok(noArg[0].message.startsWith("usage:"), "use: says how to use it");
322
+
323
+ const stopped = await run("stop");
324
+ ok(stopped[0].message === "nothing is running", "stop: honest when idle");
325
+
326
+ const agents = await run("agent");
327
+ ok(agents.length > 0, "agent: always answers, installed or not");
328
+
329
+ const unknown = await run("wat");
330
+ ok(unknown[0].level === "error", "an unknown command is reported, not guessed at");
331
+ ok(
332
+ unknown[0].detail.includes("different builds"),
333
+ "an unknown command names the likely cause, since the plugin filters first",
334
+ );
335
+ }
336
+
337
+ // --- doctor ----------------------------------------------------------------
338
+ {
339
+ const bridge = new Bridge();
340
+ // A port nothing is listening on: doctor must still answer, because "why is
341
+ // nothing working" is exactly when it is run.
342
+ const rows = await handleConsole(bridge, 45999, {
343
+ studioId: "studio-a",
344
+ command: "doctor",
345
+ args: [],
346
+ line: "doctor",
347
+ });
348
+ ok(rows.length > 1, "doctor: reports against a dead port rather than failing");
349
+ ok(
350
+ rows.every((row) => typeof row.message === "string" && !row.message.includes("\n")),
351
+ "doctor: every row is one line",
352
+ );
353
+ const summary = rows[rows.length - 1];
354
+ ok(/passed.*warning.*failure/.test(summary.message), "doctor: ends with a tally");
355
+ }
356
+
357
+ // The framing preamble --------------------------------------------------------
358
+ //
359
+ // The shipped bug: "create a simple script, then edit it" sent the agent to the
360
+ // filesystem, because that is where a coding agent spawned in a repo assumes a
361
+ // script lives. It tried Bash, was refused, and reported the test impossible.
362
+ {
363
+ const framed = frame("make the door open");
364
+ ok(framed.endsWith("make the door open"), "frame: the user's words come last and unaltered");
365
+ ok(/Roblox Studio/.test(framed), "frame: says where the prompt came from");
366
+ ok(/script_create|script_edit/.test(framed), "frame: names the tools that reach the place");
367
+ ok(!framed.includes(String.fromCharCode(13)), "frame: no stray carriage returns");
368
+ // A framing that swallows an empty prompt would send the agent a wall of
369
+ // instructions and no request.
370
+ ok(frame("").trim().length > 0, "frame: survives an empty prompt");
371
+ }
372
+
373
+ // The spawned marker reaches the server the agent starts ----------------------
374
+ //
375
+ // The shipped bug: the agent inherited RBX_STUDIO_MCP_SPAWNED, but the agent is
376
+ // not what connects -- it launches its own copy of this server, and Claude did
377
+ // not pass its environment down. So the spawned server announced itself as a
378
+ // stranger: "2 MCP clients connected" on every prompt, and a stopped agent sat
379
+ // in `clients` until the stale timeout swept it.
380
+ {
381
+ const claude = find("claude");
382
+ ok(claude !== undefined, "registry: claude is registered");
383
+ const flag = claude.mcpFlag();
384
+ ok(flag[0] === "--mcp-config", "claude: passes an mcp config file");
385
+ const written = JSON.parse(readFileSync(flag[1], "utf8"));
386
+ const server = written.mcpServers["rbx-studio"];
387
+ ok(server !== undefined, "claude config: names this server");
388
+ ok(
389
+ server.env?.RBX_STUDIO_MCP_SPAWNED === "1",
390
+ "claude config: marks the server it starts as spawned by the panel",
391
+ );
392
+
393
+ const dsh = find("dsh");
394
+ const argv = dsh.argv("hello");
395
+ const patch = argv[argv.indexOf("--patch") + 1];
396
+ ok(
397
+ readFileSync(patch, "utf8").includes("RBX_STUDIO_MCP_SPAWNED: '1'"),
398
+ "dsh overlay: carries the same marker",
399
+ );
400
+ }
401
+
402
+ // Choosing which agent answers a prompt --------------------------------------
403
+ //
404
+ // The shipped bug: someone with opencode open typed a prompt into the panel and
405
+ // Claude Code answered, because the choice was "first one installed" and
406
+ // `claude` sorts first in the registry. The answer was fine and came from the
407
+ // wrong program.
408
+ {
409
+ const claude = find("claude");
410
+ const opencode = find("opencode");
411
+
412
+ ok(matchesClient(claude, "claude-code"), "claude matches the name its client reports");
413
+ ok(matchesClient(opencode, "opencode"), "opencode matches its own client name");
414
+ ok(!matchesClient(claude, "opencode"), "and does not match a different agent");
415
+ ok(!matchesClient(opencode, "claude-code"), "in either direction");
416
+ ok(!matchesClient(claude, ""), "a nameless client matches nothing");
417
+
418
+ // The registry marks `agent` rows and picks the prompt's target from the same
419
+ // function, so a listing can never say "in use" about an agent the prompt
420
+ // would not use. Exercised through the real bridge: what makes an agent a
421
+ // candidate is that it is CONNECTED, which only the bridge knows.
422
+ const bridge = new Bridge();
423
+ bridge.noteClient("one", { name: "opencode", version: "1", pid: 1 });
424
+ bridge.noteClient("two", { name: "claude-code", version: "2", pid: 2 });
425
+
426
+ const rows = await handleConsole(bridge, 44755, {
427
+ studioId: "studio-agents",
428
+ command: "agent",
429
+ args: [],
430
+ line: "agent",
431
+ });
432
+ const text = rows.map((row) => `${row.message} ${row.detail ?? ""}`).join(" | ");
433
+
434
+ // Only meaningful when both are actually installed on the machine running the
435
+ // suite; otherwise there is nothing to be ambiguous between.
436
+ const both = installed().filter((entry) => entry.id === "claude" || entry.id === "opencode");
437
+ if (both.length === 2) {
438
+ ok(
439
+ rows.some((row) => row.level === "warn" && /agent use <id>/.test(row.message)),
440
+ "two connected agents are not silently resolved to whichever sorts first",
441
+ );
442
+ ok(!/in use/.test(text), "and none is marked as the one in use");
443
+ ok(/connected/.test(text), "the ones that are attached are named as attached");
444
+ } else {
445
+ ok(true, "skipped: both agents are not installed here");
446
+ }
447
+ }
448
+
449
+ // --- the harnesses added after the opencode failure -------------------------
450
+ //
451
+ // Same rule as the block above: every shape here comes from the vendor's own
452
+ // documentation, not from a guess. The point of the rule is that a guessed
453
+ // envelope produces silence, and silence is the failure mode this whole file
454
+ // exists to catch.
455
+ {
456
+ // Amp says outright that it speaks Claude Code's protocol, so it is read by
457
+ // the same function -- and this asserts that, rather than trusting it.
458
+ const amp = readAll("amp", [
459
+ JSON.stringify({ type: "system", subtype: "init", session_id: "T-1", cwd: "/x" }),
460
+ JSON.stringify({
461
+ type: "assistant",
462
+ session_id: "T-1",
463
+ message: {
464
+ content: [
465
+ { type: "text", text: "Built the wall." },
466
+ { type: "tool_use", name: "create", input: { className: "Part" } },
467
+ ],
468
+ },
469
+ }),
470
+ JSON.stringify({ type: "result", session_id: "T-1", is_error: false }),
471
+ ]);
472
+ ok(amp.session === "T-1", "amp: session id");
473
+ ok(amp.rows.some((row) => row.message === "Built the wall."), "amp: prose");
474
+ ok(amp.rows.some((row) => row.message === "create"), "amp: tool calls");
475
+ ok(amp.rows.some((row) => row.message === "agent done"), "amp: result ends the run");
476
+
477
+ // Continuing is a different command, not a flag: `amp threads continue <id>`.
478
+ const ampResume = find("amp").argv("go on", "T-1");
479
+ ok(ampResume[0] === "threads" && ampResume[1] === "continue", "amp: resumes with its own verb");
480
+ ok(ampResume[2] === "T-1", "amp: the thread id follows the verb");
481
+ ok(find("amp").argv("hi", null)[0] === "-x", "amp: a fresh run uses the execute flag");
482
+
483
+ // Qwen Code is a Gemini CLI fork and kept its headless envelope.
484
+ const qwen = readAll("qwen", [
485
+ JSON.stringify({ type: "init", session_id: "q-1" }),
486
+ JSON.stringify({ type: "message", role: "assistant", content: "Done." }),
487
+ ]);
488
+ ok(qwen.session === "q-1", "qwen: session id, read as gemini");
489
+ ok(qwen.rows.some((row) => row.message === "Done."), "qwen: prose");
490
+
491
+ // Droid answers with one object at the end rather than a stream.
492
+ const droid = readAll("droid", [JSON.stringify({ session_id: "d-1", result: "Placed it." })]);
493
+ ok(droid.session === "d-1", "droid: session id");
494
+ ok(droid.rows.some((row) => row.message === "Placed it."), "droid: the final answer");
495
+ const droidArgv = find("droid").argv("hi", null);
496
+ ok(droidArgv[0] === "exec", "droid: uses its exec verb");
497
+ ok(droidArgv.includes("--auto"), "droid: sets an autonomy level, or it blocks on approval");
498
+
499
+ // goose names sessions instead of numbering them.
500
+ const goose = readAll("goose", [
501
+ JSON.stringify({ type: "message", text: "Ready.", session_id: "g-1" }),
502
+ ]);
503
+ ok(goose.rows.some((row) => row.message === "Ready."), "goose: prose");
504
+ const gooseResume = find("goose").argv("go on", "g-1");
505
+ ok(
506
+ gooseResume.includes("--resume") && gooseResume[gooseResume.indexOf("-n") + 1] === "g-1",
507
+ "goose: resumes a session by name",
508
+ );
509
+
510
+ // Copilot has no structured output, but it does block without --no-ask-user.
511
+ const copilotArgv = find("copilot").argv("hi", null);
512
+ ok(copilotArgv.includes("--no-ask-user"), "copilot: never waits for a human that is not there");
513
+ ok(
514
+ readAll("copilot", ["Explained it."]).rows[0].message === "Explained it.",
515
+ "copilot: plain text reaches the log",
516
+ );
517
+
518
+ // Aider blocks on confirmations unless told not to.
519
+ ok(find("aider").argv("hi", null).includes("--yes"), "aider: answers its own confirmations");
520
+
521
+ // Every harness must produce SOMETHING from a plain line. A reader that
522
+ // silently drops unknown input is exactly how opencode printed nothing.
523
+ for (const harness of ["amp", "qwen", "droid", "goose", "copilot", "aider", "crush"]) {
524
+ const rows = readAll(harness, ["some unstructured output"]).rows;
525
+ ok(rows.length > 0, harness + ": unrecognised output is shown, not swallowed");
526
+ }
527
+ }
528
+
529
+ process.stdout.write(`console: ${checks} checks pass\n`);
@@ -133,4 +133,51 @@ function handshake(port, studioId) {
133
133
  await owner.close();
134
134
  }
135
135
 
136
+ // POST /hello carries the panel's spawned marker all the way to the roster.
137
+ //
138
+ // The shipped bug: every layer marked the agent the panel starts -- the spawn's
139
+ // environment, the MCP config handed to the agent, the hello body -- and this
140
+ // route typed the body with three fields and dropped the fourth. So the panel
141
+ // said "2 MCP clients connected" on every prompt and kept a stopped agent in
142
+ // `clients` until the stale sweep. Tested at the route, because the route is
143
+ // the layer that was wrong while the bridge underneath it was right.
144
+ {
145
+ const owner = await startBridgeServer({ port: PORT });
146
+
147
+ const hello = async (id, body) => {
148
+ const sent = await fetch(`http://127.0.0.1:${PORT}/hello`, {
149
+ method: "POST",
150
+ headers: {
151
+ [CLIENT_HEADER]: "test",
152
+ [PEER_HEADER]: id,
153
+ "Content-Type": "application/json",
154
+ },
155
+ body: JSON.stringify(body),
156
+ });
157
+ return (await sent.json()).clients;
158
+ };
159
+
160
+ const alone = await hello("a-stranger", { name: "codex", version: "1", pid: 1 });
161
+ const withAgent = await hello("panel-agent", {
162
+ name: "claude-code",
163
+ version: "1",
164
+ pid: 2,
165
+ spawned: true,
166
+ });
167
+ assert.equal(withAgent, alone, "an agent the panel started is not another client");
168
+ // A second stranger does count, or the filter would be hiding everyone.
169
+ const withStranger = await hello("another-stranger", { name: "opencode", pid: 3 });
170
+ assert.equal(withStranger, alone + 1, "a client nobody asked for still counts");
171
+
172
+ // A keepalive that says nothing must not erase what the first hello said, and
173
+ // must not re-announce the client as if it had just arrived.
174
+ assert.equal(
175
+ await hello("a-stranger", { pid: 1 }),
176
+ withStranger,
177
+ "a nameless keepalive is not a new client",
178
+ );
179
+
180
+ await owner.close();
181
+ }
182
+
136
183
  process.stdout.write("failover: ok\n");