scenescout 1.1.0 → 1.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/installer.js CHANGED
@@ -9,16 +9,23 @@
9
9
  import { spawnSync } from "node:child_process";
10
10
  import fs from "node:fs";
11
11
  import path from "node:path";
12
+ import { engineOf } from "./browsers.js";
12
13
  export const SKILL_NAME = "scenescout";
13
14
  export const MCP_NAME = "scenescout";
14
15
  /** Names this tool's skill and MCP server had before renames; cleaned up on install so the old slash command and a duplicate tool set do not linger. */
15
16
  const LEGACY_SKILL_NAMES = ["frontend-tester", "scenecraft"];
16
17
  const LEGACY_MCP_NAMES = ["scenecraft"];
17
18
  /** Real runner. `missing` separates "the binary is not installed" from "it ran and failed". */
18
- export const spawnRunner = (command, args) => {
19
- const r = spawnSync(command, args, { encoding: "utf8", timeout: 60_000 });
20
- const missing = r.error?.code === "ENOENT";
21
- return { status: r.status, stdout: r.stdout ?? "", stderr: r.stderr ?? (r.error ? String(r.error.message) : ""), missing };
19
+ export const spawnRunner = (command, args, opts) => {
20
+ const r = spawnSync(command, args, { encoding: "utf8", timeout: 60_000, cwd: opts?.cwd });
21
+ const code = r.error?.code;
22
+ // On Windows node refuses to start a .cmd or .bat file directly (EINVAL). For
23
+ // the caller that is the same situation as a missing binary: nothing ran, and
24
+ // the command has to be handed to the person instead.
25
+ const missing = code === "ENOENT" || (process.platform === "win32" && code === "EINVAL");
26
+ // With `encoding` set, a command that never ran still yields "" for stderr: the error is the only account of it.
27
+ const stderr = r.stderr || (code === "ETIMEDOUT" ? "npm did not finish within 60 s" : r.error ? String(r.error.message) : "");
28
+ return { status: r.status, stdout: r.stdout ?? "", stderr, missing };
22
29
  };
23
30
  /** Written into a copy-mode install so a later install can tell its own copy from a user's directory. */
24
31
  const OWNERSHIP_MARKER = ".installed-by-scenescout";
@@ -243,16 +250,103 @@ export function parseRegistration(listing) {
243
250
  const field = (name) => new RegExp(`^[ \\t]*${name}:[ \\t]*(\\S.*?)[ \\t]*$`, "m").exec(listing)?.[1] ?? null;
244
251
  return { command: field("Command"), serverPath: field("Args") };
245
252
  }
253
+ /** The command the package's `bin` entry provides. */
254
+ export const CLI_NAME = "scenescout";
255
+ const isCheckoutRoot = (packageRoot) => fs.existsSync(path.join(packageRoot, "tsconfig.json")) && fs.existsSync(path.join(packageRoot, "src"));
256
+ /**
257
+ * Where a command resolves in the user's own shell, or null.
258
+ *
259
+ * npm and npx put `node_modules/.bin` directories on PATH for the length of a
260
+ * run. Under `npx scenescout install` that makes the command appear installed
261
+ * when it will be gone the moment the run ends, so those entries are skipped.
262
+ */
263
+ export function findOnUserPath(opts) {
264
+ const exists = opts.exists ?? fs.existsSync;
265
+ for (const dir of opts.pathValue.split(opts.delimiter ?? path.delimiter).filter(Boolean)) {
266
+ if (/[\\/]node_modules[\\/]\.bin[\\/]?$/.test(dir))
267
+ continue;
268
+ for (const name of opts.names) {
269
+ const candidate = path.join(dir, name);
270
+ if (exists(candidate))
271
+ return candidate;
272
+ }
273
+ }
274
+ return null;
275
+ }
276
+ /**
277
+ * What it takes for `scenescout` to work as a command in a terminal.
278
+ *
279
+ * The MCP registration never needed that: it stores an absolute launcher. But
280
+ * `scenescout status` and `scenescout watch` are typed by a person, and both a
281
+ * source checkout and an npx run leave nothing on PATH, so the commands the
282
+ * tool itself recommends answered "command not found".
283
+ *
284
+ * A checkout is linked, so the command always runs what was last built. Any
285
+ * other install gets the same version installed globally. A checkout takes the
286
+ * name over from another copy, the way it takes over the MCP registration; a
287
+ * packaged install leaves an existing command alone.
288
+ */
289
+ export function planCommand(opts) {
290
+ const checkout = isCheckoutRoot(opts.packageRoot);
291
+ const manual = checkout ? `npm link (run in ${opts.packageRoot})` : `npm install -g ${CLI_NAME}@${opts.version}`;
292
+ if (opts.resolved !== null) {
293
+ const mine = samePath(opts.resolved, path.join(opts.packageRoot, "dist", "cli.js"));
294
+ if (mine || !checkout)
295
+ return { action: "present", at: opts.resolved };
296
+ }
297
+ // npm on Windows is a .cmd shim, which node cannot start directly.
298
+ if (opts.platform === "win32")
299
+ return { action: "manual", manual, why: "this step cannot start npm on Windows" };
300
+ const beside = path.join(path.dirname(opts.nodePath), "npm");
301
+ // The npm beside the running node installs into that node's prefix, which is
302
+ // the one whose bin directory the shell that started us already has on PATH.
303
+ const npm = fs.existsSync(beside) ? beside : "npm";
304
+ return checkout
305
+ ? { action: "run", how: "link", command: npm, args: ["link"], cwd: opts.packageRoot, manual, replaces: opts.resolved }
306
+ : { action: "run", how: "global", command: npm, args: ["install", "-g", `${CLI_NAME}@${opts.version}`], manual, replaces: null };
307
+ }
308
+ export function ensureCommand(plan, run) {
309
+ if (plan.action === "present")
310
+ return { status: "present", at: plan.at };
311
+ if (plan.action === "manual")
312
+ return { status: "failed", manual: plan.manual, detail: plan.why };
313
+ const r = run(plan.command, plan.args, { cwd: plan.cwd });
314
+ if (r.missing)
315
+ return { status: "failed", manual: plan.manual, detail: "npm was not found" };
316
+ if (r.status !== 0) {
317
+ // A system-wide node owns its prefix as root; that is the usual reason, and
318
+ // the last line of npm's output names it. With no output at all (a timeout,
319
+ // a kill) the exit is all there is to say.
320
+ const lines = (r.stderr || r.stdout).trim().split("\n");
321
+ return {
322
+ status: "failed",
323
+ manual: plan.manual,
324
+ detail: lines.find((l) => /EACCES|EPERM|ERR!/.test(l))?.trim() || lines[lines.length - 1] || `npm exited ${r.status ?? "without finishing"}`,
325
+ };
326
+ }
327
+ return { status: "installed", how: plan.how, replaced: plan.replaces };
328
+ }
246
329
  /**
247
330
  * The command that repairs a setup, for THIS kind of install. A source checkout
248
331
  * has `npm run setup`; someone who installed from npm has no such script, and
249
332
  * telling them to run it sends them looking for a package.json they never had.
250
333
  */
251
334
  export function repairCommands(packageRoot) {
252
- const isCheckout = fs.existsSync(path.join(packageRoot, "tsconfig.json")) && fs.existsSync(path.join(packageRoot, "src"));
335
+ const isCheckout = isCheckoutRoot(packageRoot);
336
+ // Installing "chromium" brings the headless shell with it, so the plain
337
+ // setup command already repairs either Chromium build.
338
+ const browserFlags = (target) => (engineOf(target) === "chromium" ? "" : ` --browser-only --browsers ${target}`);
253
339
  return isCheckout
254
- ? { setup: "npm run setup", build: "npm run build" }
255
- : { setup: "npx -y scenescout install", build: "npx -y scenescout@latest install (the installed package is incomplete; fetch it again)" };
340
+ ? {
341
+ setup: "npm run setup",
342
+ build: "npm run build",
343
+ browser: (target) => (browserFlags(target) ? `node dist/cli.js install${browserFlags(target)}` : "npm run setup"),
344
+ }
345
+ : {
346
+ setup: "npx -y scenescout install",
347
+ build: "npx -y scenescout@latest install (the installed package is incomplete; fetch it again)",
348
+ browser: (target) => `npx -y scenescout install${browserFlags(target)}`,
349
+ };
256
350
  }
257
351
  /** Everything a working setup needs, each with the command that repairs it. */
258
352
  export function diagnose(opts) {
@@ -262,12 +356,12 @@ export function diagnose(opts) {
262
356
  checks.push({ name: "node >= 20", ok: major >= 20, detail: opts.nodeVersion, fix: "install Node 20 or newer" });
263
357
  const server = path.join(opts.packageRoot, "dist", "mcp-server.js");
264
358
  checks.push({ name: "engine built", ok: fs.existsSync(server), detail: server, fix: repair.build });
265
- const chromiumOk = !!opts.chromiumPath && fs.existsSync(opts.chromiumPath);
359
+ const browser = opts.defaultBrowser;
266
360
  checks.push({
267
- name: "chromium downloaded",
268
- ok: chromiumOk,
269
- detail: opts.chromiumPath ?? "playwright could not name a browser path",
270
- fix: `${repair.setup} (or: npx playwright install chromium)`,
361
+ name: `browser downloaded (${browser.target})`,
362
+ ok: browser.path !== null,
363
+ detail: browser.path ?? (browser.expected ? `not found at ${browser.expected}` : "playwright could not name a browser path"),
364
+ fix: `${repair.browser(browser.target)} (or: npx playwright install ${browser.target})`,
271
365
  });
272
366
  if (opts.scope === "engine")
273
367
  return checks;
@@ -21,19 +21,25 @@
21
21
  * - Self-healing: orphaned browser processes from crashed runs are reaped at
22
22
  * startup and on launch failure; attach retries once after reaping.
23
23
  * - Observable: .scenescout/status.json in the tested project always shows
24
- * what each session is doing right now (`scenescout status <project>`).
24
+ * what EVERY session is doing right now (`scenescout status <project>`), and
25
+ * a loopback-only live view shows what each one is looking at
26
+ * (`scenescout watch <project>`, engine/live.ts, ADR 7).
25
27
  */
26
28
  import fs from "node:fs";
27
29
  import path from "node:path";
30
+ import { fileURLToPath } from "node:url";
28
31
  import { McpServer } from "@modelcontextprotocol/sdk/server/mcp.js";
29
32
  import { StdioServerTransport } from "@modelcontextprotocol/sdk/server/stdio.js";
33
+ import { ErrorCode, GetPromptRequestSchema, ListPromptsRequestSchema, McpError } from "@modelcontextprotocol/sdk/types.js";
30
34
  import { z } from "zod";
31
35
  import { BrowserEngine } from "./engine/browser.js";
32
36
  import { reapOrphanBrowsers } from "./engine/reaper.js";
33
37
  import { MemoryStore, redactSecrets } from "./engine/memory.js";
34
38
  import { SessionQueue, withWatchdog } from "./engine/dispatch.js";
35
39
  import { FIXTURE_KINDS } from "./engine/fixtures.js";
40
+ import { feedForSession, LIVE_ENV, writeStatusFile, LIVE_TOKEN_FILE, LiveServer, StatusBoard } from "./engine/live.js";
36
41
  import { computeGaps, formatRouteCoverage, generateReport } from "./engine/report.js";
42
+ import { EXPLORE_PROMPT_ARGUMENTS, explorePrompt, loadPlaybook, PLAYBOOK_PROMPT, PLAYBOOK_TOOL, SERVER_INSTRUCTIONS } from "./playbook.js";
37
43
  import { formatScan, scanProject } from "./scan.js";
38
44
  /** Live sessions: each name owns an independent BrowserEngine (browser + auth). */
39
45
  const engines = new Map();
@@ -71,7 +77,8 @@ const PKG_VERSION = (() => {
71
77
  return "0.0.0";
72
78
  }
73
79
  })();
74
- const server = new McpServer({ name: "scenescout", version: PKG_VERSION });
80
+ const PACKAGE_ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), "..");
81
+ const server = new McpServer({ name: "scenescout", version: PKG_VERSION }, { instructions: SERVER_INSTRUCTIONS });
75
82
  function text(t, session) {
76
83
  const eng = engines.get(session);
77
84
  const prefix = engines.size > 1 ? `[session ${session}${eng ? ` · ${eng.role}` : ""}]\n` : "";
@@ -83,32 +90,170 @@ function errorText(err) {
83
90
  isError: true,
84
91
  };
85
92
  }
93
+ /** One entry per live session: what status.json and the live view both read. */
94
+ const board = new StatusBoard();
95
+ /** The session whose call wrote status last. The top-level fields of status.json describe it, as they always have. */
96
+ let lastWriter = null;
97
+ let live = null;
98
+ let liveAddress = null;
99
+ /** Set while the server is starting and kept afterwards, so every caller awaits the same start. */
100
+ let liveStart = null;
101
+ /** Project directories holding this process's token file, so shutdown can take it back. */
102
+ const liveDirs = new Set();
103
+ /** Token writes in flight, so shutdown waits for them instead of racing a file that appears after the rm. */
104
+ const liveTokenWrites = new Map();
105
+ let liveTokenWarned = false;
106
+ /** Why the live view could not start, when it could not: told to the agent and written to status.json. */
107
+ let liveError = null;
108
+ const liveProvider = {
109
+ snapshot: () => ({ pid: process.pid, version: PKG_VERSION, at: new Date().toISOString(), sessions: board.list() }),
110
+ // The engine holds no reasoning — it never sees one — so the feed is what the
111
+ // session DID: the action log, which is the same trail a finding's repro uses.
112
+ activity: (session, limit) => feedForSession(engines.get(session)?.memory?.actionLog ?? [], session, limit, redactSecrets),
113
+ // The same document scout_report writes at the end, rendered now and not
114
+ // written: what the run has found so far, its scores and its gap ledger.
115
+ // Findings and coverage are project-wide; the route, audit and mode figures
116
+ // are the session's that wrote status last, so in a multi-session run they
117
+ // can shift between polls.
118
+ report: () => {
119
+ const eng = (lastWriter && engines.get(lastWriter.session)) ?? engines.values().next().value;
120
+ if (!eng?.memory)
121
+ return null;
122
+ const { markdown } = generateReport(eng.memory, eng.oracleLog.all, reportExtras(eng), { write: false });
123
+ return { markdown: redactSecrets(markdown), at: new Date().toISOString() };
124
+ },
125
+ // Both go around the session queue on purpose: a viewer must never wait
126
+ // behind the agent's calls, and a session that is stuck is the one most
127
+ // worth looking at.
128
+ screenshot: async (session) => (await engines.get(session)?.liveShot()) ?? null,
129
+ startStream: async (session, onFrame, onEnd) => (await engines.get(session)?.startScreencast(onFrame, onEnd)) ?? null,
130
+ };
131
+ /** What the report needs to know beyond memory: the one place it is built, so the live view and scout_report cannot drift apart. */
132
+ function reportExtras(eng) {
133
+ const unvisited = eng.unvisitedKnownRoutes();
134
+ const all = eng.allKnownRoutes();
135
+ return {
136
+ routesVisited: all.length - unvisited.length,
137
+ routesTotal: all.length,
138
+ designAudits: eng.designAuditCount,
139
+ createdResources: eng.createdResources,
140
+ unvisitedRoutes: unvisited,
141
+ mode: eng.mode,
142
+ policyAttributed: eng.oracleLog.policyAttributed,
143
+ };
144
+ }
145
+ /** Hand the live view's token to `scenescout watch` through a file only the owner can read. */
146
+ function publishLiveToken(dir) {
147
+ if (!liveAddress || liveDirs.has(dir) || liveTokenWrites.has(dir))
148
+ return;
149
+ const file = path.join(dir, LIVE_TOKEN_FILE);
150
+ const write = fs.promises
151
+ .writeFile(file, liveAddress.token, { mode: 0o600 })
152
+ // `mode` applies only when the file is created; a leftover one keeps its old bits.
153
+ .then(() => fs.promises.chmod(file, 0o600))
154
+ .then(() => {
155
+ liveDirs.add(dir);
156
+ })
157
+ .catch((err) => {
158
+ // A file with the wrong bits, or none: either way nothing to take back later.
159
+ void fs.promises.rm(file, { force: true }).catch(() => { });
160
+ if (!liveTokenWarned)
161
+ console.error(`[scenescout] could not write the live view's token file in ${dir}: ${err instanceof Error ? err.message : String(err)}`);
162
+ liveTokenWarned = true;
163
+ })
164
+ .finally(() => liveTokenWrites.delete(dir));
165
+ liveTokenWrites.set(dir, write);
166
+ }
167
+ /**
168
+ * Started with the first attach rather than at boot: a server nobody attaches
169
+ * to should not open a port. Never rejects — observability is best-effort, and
170
+ * a port that will not open must not cost the run anything.
171
+ */
172
+ async function ensureLive(dir) {
173
+ if (process.env[LIVE_ENV] === "off")
174
+ return;
175
+ if (liveAddress)
176
+ return publishLiveToken(dir);
177
+ liveStart ??= startLiveServer();
178
+ await liveStart;
179
+ if (!liveAddress)
180
+ return;
181
+ publishLiveToken(dir);
182
+ // The port is new, so a reader polling status.json needs it rewritten.
183
+ flushStatus(dir);
184
+ }
185
+ function startLiveServer() {
186
+ const starting = new LiveServer(liveProvider);
187
+ return starting
188
+ .start()
189
+ .then((address) => {
190
+ live = starting;
191
+ liveAddress = address;
192
+ liveError = null;
193
+ })
194
+ .catch((err) => {
195
+ // Say why, once, and let a later attach try again.
196
+ const reason = err instanceof Error ? err.message : String(err);
197
+ if (liveError !== reason)
198
+ console.error(`[scenescout] the live view could not start: ${reason}`);
199
+ liveError = reason;
200
+ liveStart = null;
201
+ });
202
+ }
203
+ /**
204
+ * The line that hands the live view to the person running the agent. The
205
+ * address holds the token, and a tool result lands in the client's transcript;
206
+ * that is accepted (ADR 7) because the address answers on this machine only.
207
+ */
208
+ function liveLine() {
209
+ if (!liveAddress)
210
+ return liveError ? `\nLive view unavailable: ${liveError}` : "";
211
+ return (`\nLive view: http://127.0.0.1:${liveAddress.port}/${liveAddress.token}/ — give this address to the user so they can watch every session ` +
212
+ `(current tool, page thumbnail, optional live stream). It opens on this machine only and cannot act on the run.`);
213
+ }
214
+ function flushStatus(dir) {
215
+ // Fire-and-forget: status is best-effort observability on every tool call's
216
+ // hot path and must never add blocking filesystem latency. The writer
217
+ // queues writes per directory and lands each by rename, so a reader never
218
+ // sees a torn file.
219
+ void writeStatusFile(dir, JSON.stringify({
220
+ pid: process.pid,
221
+ phase: lastWriter?.phase ?? "idle",
222
+ tool: lastWriter?.tool ?? "",
223
+ session: lastWriter?.session ?? "",
224
+ role: lastWriter?.role ?? "anonymous",
225
+ sessions: [...engines.keys()],
226
+ url: lastWriter?.url ?? "",
227
+ at: new Date().toISOString(),
228
+ // Everything above describes one session. This is all of them.
229
+ detail: board.list(),
230
+ ...(liveAddress ? { live: { port: liveAddress.port } } : liveError ? { live: { error: liveError } } : {}),
231
+ }, null, 2));
232
+ }
86
233
  /**
87
234
  * Live status for the tested project (`scenescout status <project>` or any
88
- * supervising layer reads this): which session/tool is running right now.
235
+ * supervising layer reads this): what every session is doing right now.
89
236
  * Best-effort — observability must never break the tool call itself.
90
237
  */
91
- function writeStatus(session, phase, tool) {
92
- const dir = engines.get(session)?.memory?.dir;
93
- if (!dir)
238
+ function writeStatus(session, phase, tool, budgetMs) {
239
+ const eng = engines.get(session);
240
+ const dir = eng?.memory?.dir;
241
+ if (!eng || !dir)
94
242
  return;
95
- // Fire-and-forget async write: status is best-effort observability and runs
96
- // on every tool call's hot path — it must never add blocking filesystem
97
- // latency. Two sessions writing concurrently is a benign last-write-wins on
98
- // this one project-level file; each session's OWN status still reaches disk.
99
- void fs.promises
100
- .writeFile(path.join(dir, "status.json"), JSON.stringify({
101
- pid: process.pid,
243
+ // status.json is a poll target that gets pasted into bug reports.
244
+ const { task, objective, ...described } = eng.liveDescription;
245
+ lastWriter = board.update(session, {
246
+ role: eng.role,
102
247
  phase,
103
248
  tool,
104
- session,
105
- role: engines.get(session)?.role ?? "anonymous",
106
- sessions: [...engines.keys()],
107
- // status.json is a poll target that gets pasted into bug reports.
108
- url: redactSecrets(engines.get(session)?.currentUrl ?? ""),
109
- at: new Date().toISOString(),
110
- }, null, 2))
111
- .catch(() => { });
249
+ url: redactSecrets(eng.currentUrl),
250
+ ...(budgetMs ? { budgetMs } : {}),
251
+ ...described,
252
+ ...(task ? { task: redactSecrets(task) } : {}),
253
+ ...(objective ? { objective: redactSecrets(objective) } : {}),
254
+ });
255
+ void ensureLive(dir);
256
+ flushStatus(dir);
112
257
  }
113
258
  /** The watchdog's timeout answer — a diagnosable result, not a hang. */
114
259
  function watchdogTimeout(label, ms) {
@@ -129,7 +274,7 @@ function serializedPerSession(label, fn, timeoutMs = 60_000) {
129
274
  return (args) => {
130
275
  const session = args.session ?? activeName;
131
276
  const exec = async () => {
132
- writeStatus(session, "running", label);
277
+ writeStatus(session, "running", label, timeoutMs);
133
278
  try {
134
279
  const out = await withWatchdog(label, fn(args, session), timeoutMs, watchdogTimeout);
135
280
  // `activeName` is process-global and every scout_attach moves it. With
@@ -167,6 +312,48 @@ const sessionParam = z
167
312
  .max(40)
168
313
  .optional()
169
314
  .describe("Target this session directly instead of the active one — pass it explicitly when dispatching to MULTIPLE sessions in one turn (e.g. two scout_click calls with different `session`), which then run CONCURRENTLY rather than queueing. Omit for single-session sequential use.");
315
+ // The method, for every client that has no skill loader. It is read per call,
316
+ // not cached: a source checkout's skill file can change under a running server.
317
+ server.registerTool(PLAYBOOK_TOOL, {
318
+ description: "Return the SceneScout testing method: setup order, write modes, how to explore, what counts as done, how to report. " +
319
+ "Call this ONCE before the first scout_attach in a conversation, then follow it. " +
320
+ "If this client offers a SceneScout skill, load that instead — it is the same text, so never read both. Takes no input and touches no browser.",
321
+ // No inputSchema on purpose: with one, a call that carries no `arguments` field is rejected as invalid.
322
+ }, async () => {
323
+ try {
324
+ return { content: [{ type: "text", text: loadPlaybook(PACKAGE_ROOT) }] };
325
+ }
326
+ catch (err) {
327
+ return errorText(err);
328
+ }
329
+ });
330
+ // The same method as a prompt, for clients that list server prompts as commands.
331
+ // Registered on the protocol server directly: the SDK's prompt helper rejects a
332
+ // request that carries no `arguments` object, which is exactly what a client
333
+ // sends when the person typed none, and every argument here is optional.
334
+ server.server.registerCapabilities({ prompts: {} });
335
+ server.server.setRequestHandler(ListPromptsRequestSchema, () => ({
336
+ prompts: [
337
+ {
338
+ name: PLAYBOOK_PROMPT,
339
+ title: "Explore a web app with SceneScout",
340
+ description: "Start an exploratory test session: loads the SceneScout method and states the target.",
341
+ arguments: EXPLORE_PROMPT_ARGUMENTS,
342
+ },
343
+ ],
344
+ }));
345
+ server.server.setRequestHandler(GetPromptRequestSchema, (request) => {
346
+ if (request.params.name !== PLAYBOOK_PROMPT)
347
+ throw new McpError(ErrorCode.InvalidParams, `Unknown prompt: ${request.params.name}`);
348
+ let message;
349
+ try {
350
+ message = explorePrompt(loadPlaybook(PACKAGE_ROOT), request.params.arguments);
351
+ }
352
+ catch (err) {
353
+ throw new McpError(ErrorCode.InvalidParams, err instanceof Error ? err.message : String(err));
354
+ }
355
+ return { messages: [{ role: "user", content: { type: "text", text: message } }] };
356
+ });
170
357
  server.registerTool("scout_scan", {
171
358
  description: "Scan a project directory to discover the frontend workspace, framework, routes, dev command, Playwright auth storage states, and testid conventions. Run this first.",
172
359
  inputSchema: { projectPath: z.string().describe("Absolute path to the project root") },
@@ -179,7 +366,7 @@ server.registerTool("scout_scan", {
179
366
  }
180
367
  }));
181
368
  server.registerTool("scout_attach", {
182
- description: "Launch a browser and attach to a running web app. Write policy is enforced at the NETWORK layer: mode='observe' blocks EVERY request that is not a GET (login and token refresh excepted) — choose it for a target that holds real data, where even an ordinary form submission would create a record; mode='read-only' (default) blocks destructive-labeled elements AND all PUT/PATCH/DELETE + destructive POSTs, but lets ordinary form POSTs through; mode='safe-write' allows creating data and permits updates/deletes ONLY on resources this session created (use when the user wants create/edit flows tested); mode='destructive' allows everything — ONLY when the user explicitly confirmed a disposable/seeded environment. Pass a Playwright storage-state JSON to explore as an authenticated role. Pass `session` to keep MULTIPLE roles alive at once (one browser each, genuinely concurrent) for collaboration testing — target each directly with every tool's `session` param, or use scout_session to set which one is the default; coverage and findings merge into one project memory.",
369
+ description: "Launch a browser and attach to a running web app. First attach in this conversation and you have read neither the SceneScout skill nor scout_playbook? Call scout_playbook before this. Write policy is enforced at the NETWORK layer: mode='observe' blocks EVERY request that is not a GET (login and token refresh excepted) — choose it for a target that holds real data, where even an ordinary form submission would create a record; mode='read-only' (default) blocks destructive-labeled elements AND all PUT/PATCH/DELETE + destructive POSTs, but lets ordinary form POSTs through; mode='safe-write' allows creating data and permits updates/deletes ONLY on resources this session created (use when the user wants create/edit flows tested); mode='destructive' allows everything — ONLY when the user explicitly confirmed a disposable/seeded environment. Pass a Playwright storage-state JSON to explore as an authenticated role. Pass `session` to keep MULTIPLE roles alive at once (one browser each, genuinely concurrent) for collaboration testing — target each directly with every tool's `session` param, or use scout_session to set which one is the default; coverage and findings merge into one project memory.",
183
370
  inputSchema: {
184
371
  url: z.string().describe("Base URL of the running app, e.g. http://localhost:3000"),
185
372
  projectPath: z.string().describe("Absolute path to the project (memory + report live in .scenescout/ here)"),
@@ -189,15 +376,24 @@ server.registerTool("scout_attach", {
189
376
  .default("read-only")
190
377
  .describe("Write policy (see tool description). Never choose 'destructive' yourself — user opt-in only."),
191
378
  headed: z.boolean().default(false).describe("Show the browser window"),
379
+ browser: z
380
+ .enum(["chromium", "firefox", "webkit"])
381
+ .optional()
382
+ .describe("Browser to drive. Default: the SCENESCOUT_BROWSER environment variable, else chromium. firefox and webkit must be downloaded first (scenescout install --browser-only --browsers firefox). Use them for a cross-browser pass; stay on chromium otherwise."),
192
383
  viewportWidth: z.number().int().min(320).max(3840).optional().describe("Viewport width (default 1280); use e.g. 390 for a mobile pass"),
193
384
  viewportHeight: z.number().int().min(480).max(2400).optional().describe("Viewport height (default 900)"),
385
+ task: z
386
+ .string()
387
+ .max(300)
388
+ .optional()
389
+ .describe("What this session is for, in one sentence (e.g. 'Approve and reject orders as a manager'). Shown to the person watching the live view, next to the goal of whatever scout_journey is active. Worth setting whenever more than one session is running."),
194
390
  session: z
195
391
  .string()
196
392
  .max(40)
197
393
  .optional()
198
394
  .describe("Session name for multi-role runs (e.g. 'admin', 'qa'). Creates/replaces that session's browser and makes it the default. Default: 'default'."),
199
395
  },
200
- }, serializedControl(async ({ url, projectPath, storageStatePath, mode, headed, viewportWidth, viewportHeight, session, }) => {
396
+ }, serializedControl(async ({ url, projectPath, storageStatePath, mode, headed, browser, viewportWidth, viewportHeight, task, session, }) => {
201
397
  try {
202
398
  const target = session ?? activeName;
203
399
  if (session) {
@@ -258,9 +454,16 @@ server.registerTool("scout_attach", {
258
454
  /* conflict detection is best-effort */
259
455
  }
260
456
  const viewport = viewportWidth && viewportHeight ? { width: viewportWidth, height: viewportHeight } : undefined;
261
- const out = await eng.attach({ url, projectDir: projectPath, storageStatePath, mode, headed, viewport, memoryStore: store });
457
+ const out = await eng.attach({ url, projectDir: projectPath, storageStatePath, mode, headed, browser, viewport, task, memoryStore: store });
262
458
  eng.role = storageStatePath ? path.basename(storageStatePath).replace(/\.json$/i, "") : "anonymous";
263
- return text(out + conflictNote + (engines.size > 1 ? `\n${sessionLines()}` : ""), target);
459
+ // Put the session on the board now, so the live view shows it before its
460
+ // first tool call. liveLine() needs the port, so the server is awaited
461
+ // here rather than started in the background by writeStatus.
462
+ if (eng.memory?.dir) {
463
+ await ensureLive(eng.memory.dir);
464
+ writeStatus(target, "idle", "scout_attach");
465
+ }
466
+ return text(out + conflictNote + (engines.size > 1 ? `\n${sessionLines()}` : "") + liveLine(), target);
264
467
  }
265
468
  catch (err) {
266
469
  return errorText(err);
@@ -286,7 +489,7 @@ server.registerTool("scout_session", {
286
489
  try {
287
490
  name = name ?? session;
288
491
  if (!name)
289
- return text(sessionLines(), activeName);
492
+ return text(sessionLines() + liveLine(), activeName);
290
493
  if (!engines.has(name)) {
291
494
  return text(`No session named "${name}" yet — create it with scout_attach { session: "${name}", … }.\n${sessionLines()}`, activeName);
292
495
  }
@@ -769,19 +972,35 @@ server.registerTool("scout_close", {
769
972
  const names = [...engines.keys()];
770
973
  // Closes are independent per-browser — run them in parallel so N wedged
771
974
  // sessions cost one 8s teardown cap total, not N of them.
975
+ const dirs = new Set();
976
+ for (const e of engines.values())
977
+ if (e.memory?.dir)
978
+ dirs.add(e.memory.dir);
979
+ for (const name of engines.keys())
980
+ live?.dropSession(name);
772
981
  await Promise.allSettled([...engines.values()].map((e) => e.close()));
773
982
  engines.clear();
774
983
  sessionQueue.clear();
984
+ board.clear();
985
+ lastWriter = null;
986
+ for (const dir of dirs)
987
+ flushStatus(dir);
775
988
  return text(`All sessions closed (${names.join(", ") || "none were live"}). Memory and reports remain in .scenescout/.`, activeName);
776
989
  }
777
990
  const name = session ?? activeName;
778
991
  const eng = engines.get(name);
779
992
  if (!eng)
780
993
  return text(`No live session "${name}".`, name);
994
+ live?.dropSession(name);
781
995
  await eng.close();
782
996
  const saveError = eng.memory?.lastSaveError;
783
997
  engines.delete(name);
784
998
  sessionQueue.forget(name);
999
+ board.remove(name);
1000
+ if (lastWriter?.session === name)
1001
+ lastWriter = null;
1002
+ if (eng.memory?.dir)
1003
+ flushStatus(eng.memory.dir);
785
1004
  if (activeName === name)
786
1005
  activeName = engines.keys().next().value ?? "default";
787
1006
  return text(`Session "${name}" closed. Memory and report remain in .scenescout/.` +
@@ -795,13 +1014,34 @@ server.registerTool("scout_close", {
795
1014
  async function main() {
796
1015
  const transport = new StdioServerTransport();
797
1016
  await server.connect(transport);
1017
+ // A client that exits by closing the pipe sends no signal. Without this the
1018
+ // process, its port, its browsers and its token file all outlived the run.
1019
+ const clientGone = () => {
1020
+ void shutdown().finally(() => process.exit(0));
1021
+ };
1022
+ transport.onclose = clientGone;
1023
+ // The transport reports a closed pipe on some platforms and not others, and
1024
+ // on Windows the SIGTERM a client sends next is a plain kill that runs no
1025
+ // handler. stdin ending is the one signal every platform gives.
1026
+ process.stdin.once("end", clientGone);
1027
+ process.stdin.once("close", clientGone);
798
1028
  // Self-heal across restarts: browsers whose parent crashed/was killed can
799
1029
  // linger and have been observed to wedge fresh launches. After connect —
800
1030
  // the stdio handshake must not wait on a full process-table scan.
801
1031
  setImmediate(() => reapOrphanBrowsers());
802
1032
  }
803
1033
  async function shutdown() {
804
- await Promise.allSettled([...engines.values()].map((e) => e.close()));
1034
+ // The token outlives nothing: a file left behind would name a port some other process may get next.
1035
+ await Promise.allSettled(liveTokenWrites.values());
1036
+ for (const dir of liveDirs) {
1037
+ try {
1038
+ fs.rmSync(path.join(dir, LIVE_TOKEN_FILE), { force: true });
1039
+ }
1040
+ catch (err) {
1041
+ console.error(`[scenescout] could not remove the live view's token file in ${dir}: ${err instanceof Error ? err.message : String(err)}`);
1042
+ }
1043
+ }
1044
+ await Promise.allSettled([live?.stop(), ...[...engines.values()].map((e) => e.close())]);
805
1045
  }
806
1046
  process.on("SIGINT", () => {
807
1047
  void shutdown().finally(() => process.exit(0));
@@ -0,0 +1,83 @@
1
+ /**
2
+ * The testing method, served by the MCP server itself.
3
+ *
4
+ * The method lives in skills/scenescout/SKILL.md. Claude Code loads that file
5
+ * as a skill; no other client does, so an agent there gets the tools and none
6
+ * of the method: the setup order, the write modes, what counts as done. The
7
+ * server hands the same text to any client through a tool, a prompt and a short
8
+ * pointer in its instructions. One file, so the two can never disagree.
9
+ */
10
+ import fs from "node:fs";
11
+ import path from "node:path";
12
+ export const PLAYBOOK_TOOL = "scout_playbook";
13
+ export const PLAYBOOK_PROMPT = "explore";
14
+ /** Where the method is kept, relative to the package root. It is in the package's `files`. */
15
+ export const PLAYBOOK_RELATIVE_PATH = path.join("skills", "scenescout", "SKILL.md");
16
+ /**
17
+ * Sent to every client when it connects. Short on purpose: clients cut long
18
+ * instructions, and one that is cut in the middle is worse than a pointer.
19
+ */
20
+ export const SERVER_INSTRUCTIONS = `SceneScout explores a running web app in a real browser and reports bugs, UX problems and coverage. ` +
21
+ `The scout_* tools are deterministic; the method for using them well is a separate text. ` +
22
+ `If this client offers a SceneScout skill, load that. Otherwise call ${PLAYBOOK_TOOL} once, before the first scout_attach in a conversation, and follow what it returns. ` +
23
+ `They are the same text, so never read both. ` +
24
+ `Never choose mode="destructive" yourself: that needs the user's explicit opt-in.`;
25
+ export const LEVELS = ["minimal", "medium", "extensive"];
26
+ /** What the `explore` prompt accepts, as MCP lists it. Every argument is optional. */
27
+ export const EXPLORE_PROMPT_ARGUMENTS = [
28
+ { name: "url", description: "URL of the running app, e.g. http://localhost:3000", required: false },
29
+ { name: "level", description: `How far to go: ${LEVELS.join(", ")}`, required: false },
30
+ { name: "focus", description: "An area or flow to concentrate on", required: false },
31
+ ];
32
+ /** The skill file without its YAML front matter, which only a skill loader reads. */
33
+ export function stripFrontMatter(markdown) {
34
+ // An editor may save the file with a byte-order mark; it would hide the opening fence.
35
+ const text = markdown.replace(/^\uFEFF/, "");
36
+ // The fences may hold nothing between them, and the closing one may be the last line of the file.
37
+ const m = /^---[ \t]*\r?\n(?:[\s\S]*?\r?\n)?---[ \t]*(?:\r?\n|$)/.exec(text);
38
+ // Leading blank LINES go; indentation on the first line of the body stays.
39
+ return (m ? text.slice(m[0].length) : text).replace(/^(?:[ \t]*\r?\n)+/, "");
40
+ }
41
+ /**
42
+ * The method text. Throws when the file is not where the package puts it: an
43
+ * agent told "here is the method" and handed an empty string would proceed
44
+ * without one and never say so.
45
+ */
46
+ export function loadPlaybook(packageRoot) {
47
+ const file = path.join(packageRoot, PLAYBOOK_RELATIVE_PATH);
48
+ let raw;
49
+ try {
50
+ raw = fs.readFileSync(file, "utf8");
51
+ }
52
+ catch (err) {
53
+ throw new Error(`The SceneScout playbook is missing from this install (${file}): ${err instanceof Error ? err.message : String(err)}. Reinstall the package.`);
54
+ }
55
+ const body = stripFrontMatter(raw);
56
+ if (body.trim().length === 0)
57
+ throw new Error(`The SceneScout playbook at ${file} is empty. Reinstall the package.`);
58
+ // Front matter that was not recognised would be served to the agent as if it were the method.
59
+ if (/^---[ \t]*(\r?\n|$)/.test(body))
60
+ throw new Error(`The SceneScout playbook at ${file} has front matter that could not be read. Reinstall the package.`);
61
+ return body;
62
+ }
63
+ /**
64
+ * The opening message of the `explore` prompt: the method, then what the person
65
+ * asked for. `args` is whatever the client sent, which may be nothing at all.
66
+ * A level the method does not know is refused, not passed on for the agent to guess at.
67
+ */
68
+ export function explorePrompt(playbook, args) {
69
+ const given = (name) => {
70
+ const v = args?.[name];
71
+ return typeof v === "string" && v.trim() ? v.trim() : undefined;
72
+ };
73
+ const level = given("level");
74
+ if (level !== undefined && !LEVELS.includes(level)) {
75
+ throw new Error(`level "${level}" is not one the method knows. Use one of: ${LEVELS.join(", ")}.`);
76
+ }
77
+ const asks = [
78
+ given("url") ? `Target: ${given("url")}` : "Target: ask me for the URL of the running app, or find it from the project.",
79
+ level ? `Level: ${level}` : "",
80
+ given("focus") ? `Focus: ${given("focus")}` : "",
81
+ ].filter(Boolean);
82
+ return `${playbook}\n\n---\n\nRun an exploratory test session following the method above.\n${asks.join("\n")}`;
83
+ }