scenescout 1.1.0 → 1.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +29 -0
- package/README.md +126 -27
- package/dist/browsers.js +196 -0
- package/dist/cli.js +308 -70
- package/dist/clients.js +213 -0
- package/dist/engine/browser.js +148 -20
- package/dist/engine/hover.js +42 -0
- package/dist/engine/launch.js +12 -5
- package/dist/engine/live-page.js +644 -0
- package/dist/engine/live.js +549 -0
- package/dist/engine/memory.js +3 -0
- package/dist/engine/probes.js +4 -1
- package/dist/engine/report.js +8 -2
- package/dist/installer.js +106 -12
- package/dist/mcp-server.js +268 -28
- package/dist/playbook.js +83 -0
- package/package.json +17 -5
- package/skills/scenescout/SKILL.md +5 -4
package/dist/installer.js
CHANGED
|
@@ -9,16 +9,23 @@
|
|
|
9
9
|
import { spawnSync } from "node:child_process";
|
|
10
10
|
import fs from "node:fs";
|
|
11
11
|
import path from "node:path";
|
|
12
|
+
import { engineOf } from "./browsers.js";
|
|
12
13
|
export const SKILL_NAME = "scenescout";
|
|
13
14
|
export const MCP_NAME = "scenescout";
|
|
14
15
|
/** Names this tool's skill and MCP server had before renames; cleaned up on install so the old slash command and a duplicate tool set do not linger. */
|
|
15
16
|
const LEGACY_SKILL_NAMES = ["frontend-tester", "scenecraft"];
|
|
16
17
|
const LEGACY_MCP_NAMES = ["scenecraft"];
|
|
17
18
|
/** Real runner. `missing` separates "the binary is not installed" from "it ran and failed". */
|
|
18
|
-
export const spawnRunner = (command, args) => {
|
|
19
|
-
const r = spawnSync(command, args, { encoding: "utf8", timeout: 60_000 });
|
|
20
|
-
const
|
|
21
|
-
|
|
19
|
+
export const spawnRunner = (command, args, opts) => {
|
|
20
|
+
const r = spawnSync(command, args, { encoding: "utf8", timeout: 60_000, cwd: opts?.cwd });
|
|
21
|
+
const code = r.error?.code;
|
|
22
|
+
// On Windows node refuses to start a .cmd or .bat file directly (EINVAL). For
|
|
23
|
+
// the caller that is the same situation as a missing binary: nothing ran, and
|
|
24
|
+
// the command has to be handed to the person instead.
|
|
25
|
+
const missing = code === "ENOENT" || (process.platform === "win32" && code === "EINVAL");
|
|
26
|
+
// With `encoding` set, a command that never ran still yields "" for stderr: the error is the only account of it.
|
|
27
|
+
const stderr = r.stderr || (code === "ETIMEDOUT" ? "npm did not finish within 60 s" : r.error ? String(r.error.message) : "");
|
|
28
|
+
return { status: r.status, stdout: r.stdout ?? "", stderr, missing };
|
|
22
29
|
};
|
|
23
30
|
/** Written into a copy-mode install so a later install can tell its own copy from a user's directory. */
|
|
24
31
|
const OWNERSHIP_MARKER = ".installed-by-scenescout";
|
|
@@ -243,16 +250,103 @@ export function parseRegistration(listing) {
|
|
|
243
250
|
const field = (name) => new RegExp(`^[ \\t]*${name}:[ \\t]*(\\S.*?)[ \\t]*$`, "m").exec(listing)?.[1] ?? null;
|
|
244
251
|
return { command: field("Command"), serverPath: field("Args") };
|
|
245
252
|
}
|
|
253
|
+
/** The command the package's `bin` entry provides. */
|
|
254
|
+
export const CLI_NAME = "scenescout";
|
|
255
|
+
const isCheckoutRoot = (packageRoot) => fs.existsSync(path.join(packageRoot, "tsconfig.json")) && fs.existsSync(path.join(packageRoot, "src"));
|
|
256
|
+
/**
|
|
257
|
+
* Where a command resolves in the user's own shell, or null.
|
|
258
|
+
*
|
|
259
|
+
* npm and npx put `node_modules/.bin` directories on PATH for the length of a
|
|
260
|
+
* run. Under `npx scenescout install` that makes the command appear installed
|
|
261
|
+
* when it will be gone the moment the run ends, so those entries are skipped.
|
|
262
|
+
*/
|
|
263
|
+
export function findOnUserPath(opts) {
|
|
264
|
+
const exists = opts.exists ?? fs.existsSync;
|
|
265
|
+
for (const dir of opts.pathValue.split(opts.delimiter ?? path.delimiter).filter(Boolean)) {
|
|
266
|
+
if (/[\\/]node_modules[\\/]\.bin[\\/]?$/.test(dir))
|
|
267
|
+
continue;
|
|
268
|
+
for (const name of opts.names) {
|
|
269
|
+
const candidate = path.join(dir, name);
|
|
270
|
+
if (exists(candidate))
|
|
271
|
+
return candidate;
|
|
272
|
+
}
|
|
273
|
+
}
|
|
274
|
+
return null;
|
|
275
|
+
}
|
|
276
|
+
/**
|
|
277
|
+
* What it takes for `scenescout` to work as a command in a terminal.
|
|
278
|
+
*
|
|
279
|
+
* The MCP registration never needed that: it stores an absolute launcher. But
|
|
280
|
+
* `scenescout status` and `scenescout watch` are typed by a person, and both a
|
|
281
|
+
* source checkout and an npx run leave nothing on PATH, so the commands the
|
|
282
|
+
* tool itself recommends answered "command not found".
|
|
283
|
+
*
|
|
284
|
+
* A checkout is linked, so the command always runs what was last built. Any
|
|
285
|
+
* other install gets the same version installed globally. A checkout takes the
|
|
286
|
+
* name over from another copy, the way it takes over the MCP registration; a
|
|
287
|
+
* packaged install leaves an existing command alone.
|
|
288
|
+
*/
|
|
289
|
+
export function planCommand(opts) {
|
|
290
|
+
const checkout = isCheckoutRoot(opts.packageRoot);
|
|
291
|
+
const manual = checkout ? `npm link (run in ${opts.packageRoot})` : `npm install -g ${CLI_NAME}@${opts.version}`;
|
|
292
|
+
if (opts.resolved !== null) {
|
|
293
|
+
const mine = samePath(opts.resolved, path.join(opts.packageRoot, "dist", "cli.js"));
|
|
294
|
+
if (mine || !checkout)
|
|
295
|
+
return { action: "present", at: opts.resolved };
|
|
296
|
+
}
|
|
297
|
+
// npm on Windows is a .cmd shim, which node cannot start directly.
|
|
298
|
+
if (opts.platform === "win32")
|
|
299
|
+
return { action: "manual", manual, why: "this step cannot start npm on Windows" };
|
|
300
|
+
const beside = path.join(path.dirname(opts.nodePath), "npm");
|
|
301
|
+
// The npm beside the running node installs into that node's prefix, which is
|
|
302
|
+
// the one whose bin directory the shell that started us already has on PATH.
|
|
303
|
+
const npm = fs.existsSync(beside) ? beside : "npm";
|
|
304
|
+
return checkout
|
|
305
|
+
? { action: "run", how: "link", command: npm, args: ["link"], cwd: opts.packageRoot, manual, replaces: opts.resolved }
|
|
306
|
+
: { action: "run", how: "global", command: npm, args: ["install", "-g", `${CLI_NAME}@${opts.version}`], manual, replaces: null };
|
|
307
|
+
}
|
|
308
|
+
export function ensureCommand(plan, run) {
|
|
309
|
+
if (plan.action === "present")
|
|
310
|
+
return { status: "present", at: plan.at };
|
|
311
|
+
if (plan.action === "manual")
|
|
312
|
+
return { status: "failed", manual: plan.manual, detail: plan.why };
|
|
313
|
+
const r = run(plan.command, plan.args, { cwd: plan.cwd });
|
|
314
|
+
if (r.missing)
|
|
315
|
+
return { status: "failed", manual: plan.manual, detail: "npm was not found" };
|
|
316
|
+
if (r.status !== 0) {
|
|
317
|
+
// A system-wide node owns its prefix as root; that is the usual reason, and
|
|
318
|
+
// the last line of npm's output names it. With no output at all (a timeout,
|
|
319
|
+
// a kill) the exit is all there is to say.
|
|
320
|
+
const lines = (r.stderr || r.stdout).trim().split("\n");
|
|
321
|
+
return {
|
|
322
|
+
status: "failed",
|
|
323
|
+
manual: plan.manual,
|
|
324
|
+
detail: lines.find((l) => /EACCES|EPERM|ERR!/.test(l))?.trim() || lines[lines.length - 1] || `npm exited ${r.status ?? "without finishing"}`,
|
|
325
|
+
};
|
|
326
|
+
}
|
|
327
|
+
return { status: "installed", how: plan.how, replaced: plan.replaces };
|
|
328
|
+
}
|
|
246
329
|
/**
|
|
247
330
|
* The command that repairs a setup, for THIS kind of install. A source checkout
|
|
248
331
|
* has `npm run setup`; someone who installed from npm has no such script, and
|
|
249
332
|
* telling them to run it sends them looking for a package.json they never had.
|
|
250
333
|
*/
|
|
251
334
|
export function repairCommands(packageRoot) {
|
|
252
|
-
const isCheckout =
|
|
335
|
+
const isCheckout = isCheckoutRoot(packageRoot);
|
|
336
|
+
// Installing "chromium" brings the headless shell with it, so the plain
|
|
337
|
+
// setup command already repairs either Chromium build.
|
|
338
|
+
const browserFlags = (target) => (engineOf(target) === "chromium" ? "" : ` --browser-only --browsers ${target}`);
|
|
253
339
|
return isCheckout
|
|
254
|
-
? {
|
|
255
|
-
|
|
340
|
+
? {
|
|
341
|
+
setup: "npm run setup",
|
|
342
|
+
build: "npm run build",
|
|
343
|
+
browser: (target) => (browserFlags(target) ? `node dist/cli.js install${browserFlags(target)}` : "npm run setup"),
|
|
344
|
+
}
|
|
345
|
+
: {
|
|
346
|
+
setup: "npx -y scenescout install",
|
|
347
|
+
build: "npx -y scenescout@latest install (the installed package is incomplete; fetch it again)",
|
|
348
|
+
browser: (target) => `npx -y scenescout install${browserFlags(target)}`,
|
|
349
|
+
};
|
|
256
350
|
}
|
|
257
351
|
/** Everything a working setup needs, each with the command that repairs it. */
|
|
258
352
|
export function diagnose(opts) {
|
|
@@ -262,12 +356,12 @@ export function diagnose(opts) {
|
|
|
262
356
|
checks.push({ name: "node >= 20", ok: major >= 20, detail: opts.nodeVersion, fix: "install Node 20 or newer" });
|
|
263
357
|
const server = path.join(opts.packageRoot, "dist", "mcp-server.js");
|
|
264
358
|
checks.push({ name: "engine built", ok: fs.existsSync(server), detail: server, fix: repair.build });
|
|
265
|
-
const
|
|
359
|
+
const browser = opts.defaultBrowser;
|
|
266
360
|
checks.push({
|
|
267
|
-
name:
|
|
268
|
-
ok:
|
|
269
|
-
detail:
|
|
270
|
-
fix: `${repair.
|
|
361
|
+
name: `browser downloaded (${browser.target})`,
|
|
362
|
+
ok: browser.path !== null,
|
|
363
|
+
detail: browser.path ?? (browser.expected ? `not found at ${browser.expected}` : "playwright could not name a browser path"),
|
|
364
|
+
fix: `${repair.browser(browser.target)} (or: npx playwright install ${browser.target})`,
|
|
271
365
|
});
|
|
272
366
|
if (opts.scope === "engine")
|
|
273
367
|
return checks;
|
package/dist/mcp-server.js
CHANGED
|
@@ -21,19 +21,25 @@
|
|
|
21
21
|
* - Self-healing: orphaned browser processes from crashed runs are reaped at
|
|
22
22
|
* startup and on launch failure; attach retries once after reaping.
|
|
23
23
|
* - Observable: .scenescout/status.json in the tested project always shows
|
|
24
|
-
* what
|
|
24
|
+
* what EVERY session is doing right now (`scenescout status <project>`), and
|
|
25
|
+
* a loopback-only live view shows what each one is looking at
|
|
26
|
+
* (`scenescout watch <project>`, engine/live.ts, ADR 7).
|
|
25
27
|
*/
|
|
26
28
|
import fs from "node:fs";
|
|
27
29
|
import path from "node:path";
|
|
30
|
+
import { fileURLToPath } from "node:url";
|
|
28
31
|
import { McpServer } from "@modelcontextprotocol/sdk/server/mcp.js";
|
|
29
32
|
import { StdioServerTransport } from "@modelcontextprotocol/sdk/server/stdio.js";
|
|
33
|
+
import { ErrorCode, GetPromptRequestSchema, ListPromptsRequestSchema, McpError } from "@modelcontextprotocol/sdk/types.js";
|
|
30
34
|
import { z } from "zod";
|
|
31
35
|
import { BrowserEngine } from "./engine/browser.js";
|
|
32
36
|
import { reapOrphanBrowsers } from "./engine/reaper.js";
|
|
33
37
|
import { MemoryStore, redactSecrets } from "./engine/memory.js";
|
|
34
38
|
import { SessionQueue, withWatchdog } from "./engine/dispatch.js";
|
|
35
39
|
import { FIXTURE_KINDS } from "./engine/fixtures.js";
|
|
40
|
+
import { feedForSession, LIVE_ENV, writeStatusFile, LIVE_TOKEN_FILE, LiveServer, StatusBoard } from "./engine/live.js";
|
|
36
41
|
import { computeGaps, formatRouteCoverage, generateReport } from "./engine/report.js";
|
|
42
|
+
import { EXPLORE_PROMPT_ARGUMENTS, explorePrompt, loadPlaybook, PLAYBOOK_PROMPT, PLAYBOOK_TOOL, SERVER_INSTRUCTIONS } from "./playbook.js";
|
|
37
43
|
import { formatScan, scanProject } from "./scan.js";
|
|
38
44
|
/** Live sessions: each name owns an independent BrowserEngine (browser + auth). */
|
|
39
45
|
const engines = new Map();
|
|
@@ -71,7 +77,8 @@ const PKG_VERSION = (() => {
|
|
|
71
77
|
return "0.0.0";
|
|
72
78
|
}
|
|
73
79
|
})();
|
|
74
|
-
const
|
|
80
|
+
const PACKAGE_ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), "..");
|
|
81
|
+
const server = new McpServer({ name: "scenescout", version: PKG_VERSION }, { instructions: SERVER_INSTRUCTIONS });
|
|
75
82
|
function text(t, session) {
|
|
76
83
|
const eng = engines.get(session);
|
|
77
84
|
const prefix = engines.size > 1 ? `[session ${session}${eng ? ` · ${eng.role}` : ""}]\n` : "";
|
|
@@ -83,32 +90,170 @@ function errorText(err) {
|
|
|
83
90
|
isError: true,
|
|
84
91
|
};
|
|
85
92
|
}
|
|
93
|
+
/** One entry per live session: what status.json and the live view both read. */
|
|
94
|
+
const board = new StatusBoard();
|
|
95
|
+
/** The session whose call wrote status last. The top-level fields of status.json describe it, as they always have. */
|
|
96
|
+
let lastWriter = null;
|
|
97
|
+
let live = null;
|
|
98
|
+
let liveAddress = null;
|
|
99
|
+
/** Set while the server is starting and kept afterwards, so every caller awaits the same start. */
|
|
100
|
+
let liveStart = null;
|
|
101
|
+
/** Project directories holding this process's token file, so shutdown can take it back. */
|
|
102
|
+
const liveDirs = new Set();
|
|
103
|
+
/** Token writes in flight, so shutdown waits for them instead of racing a file that appears after the rm. */
|
|
104
|
+
const liveTokenWrites = new Map();
|
|
105
|
+
let liveTokenWarned = false;
|
|
106
|
+
/** Why the live view could not start, when it could not: told to the agent and written to status.json. */
|
|
107
|
+
let liveError = null;
|
|
108
|
+
const liveProvider = {
|
|
109
|
+
snapshot: () => ({ pid: process.pid, version: PKG_VERSION, at: new Date().toISOString(), sessions: board.list() }),
|
|
110
|
+
// The engine holds no reasoning — it never sees one — so the feed is what the
|
|
111
|
+
// session DID: the action log, which is the same trail a finding's repro uses.
|
|
112
|
+
activity: (session, limit) => feedForSession(engines.get(session)?.memory?.actionLog ?? [], session, limit, redactSecrets),
|
|
113
|
+
// The same document scout_report writes at the end, rendered now and not
|
|
114
|
+
// written: what the run has found so far, its scores and its gap ledger.
|
|
115
|
+
// Findings and coverage are project-wide; the route, audit and mode figures
|
|
116
|
+
// are the session's that wrote status last, so in a multi-session run they
|
|
117
|
+
// can shift between polls.
|
|
118
|
+
report: () => {
|
|
119
|
+
const eng = (lastWriter && engines.get(lastWriter.session)) ?? engines.values().next().value;
|
|
120
|
+
if (!eng?.memory)
|
|
121
|
+
return null;
|
|
122
|
+
const { markdown } = generateReport(eng.memory, eng.oracleLog.all, reportExtras(eng), { write: false });
|
|
123
|
+
return { markdown: redactSecrets(markdown), at: new Date().toISOString() };
|
|
124
|
+
},
|
|
125
|
+
// Both go around the session queue on purpose: a viewer must never wait
|
|
126
|
+
// behind the agent's calls, and a session that is stuck is the one most
|
|
127
|
+
// worth looking at.
|
|
128
|
+
screenshot: async (session) => (await engines.get(session)?.liveShot()) ?? null,
|
|
129
|
+
startStream: async (session, onFrame, onEnd) => (await engines.get(session)?.startScreencast(onFrame, onEnd)) ?? null,
|
|
130
|
+
};
|
|
131
|
+
/** What the report needs to know beyond memory: the one place it is built, so the live view and scout_report cannot drift apart. */
|
|
132
|
+
function reportExtras(eng) {
|
|
133
|
+
const unvisited = eng.unvisitedKnownRoutes();
|
|
134
|
+
const all = eng.allKnownRoutes();
|
|
135
|
+
return {
|
|
136
|
+
routesVisited: all.length - unvisited.length,
|
|
137
|
+
routesTotal: all.length,
|
|
138
|
+
designAudits: eng.designAuditCount,
|
|
139
|
+
createdResources: eng.createdResources,
|
|
140
|
+
unvisitedRoutes: unvisited,
|
|
141
|
+
mode: eng.mode,
|
|
142
|
+
policyAttributed: eng.oracleLog.policyAttributed,
|
|
143
|
+
};
|
|
144
|
+
}
|
|
145
|
+
/** Hand the live view's token to `scenescout watch` through a file only the owner can read. */
|
|
146
|
+
function publishLiveToken(dir) {
|
|
147
|
+
if (!liveAddress || liveDirs.has(dir) || liveTokenWrites.has(dir))
|
|
148
|
+
return;
|
|
149
|
+
const file = path.join(dir, LIVE_TOKEN_FILE);
|
|
150
|
+
const write = fs.promises
|
|
151
|
+
.writeFile(file, liveAddress.token, { mode: 0o600 })
|
|
152
|
+
// `mode` applies only when the file is created; a leftover one keeps its old bits.
|
|
153
|
+
.then(() => fs.promises.chmod(file, 0o600))
|
|
154
|
+
.then(() => {
|
|
155
|
+
liveDirs.add(dir);
|
|
156
|
+
})
|
|
157
|
+
.catch((err) => {
|
|
158
|
+
// A file with the wrong bits, or none: either way nothing to take back later.
|
|
159
|
+
void fs.promises.rm(file, { force: true }).catch(() => { });
|
|
160
|
+
if (!liveTokenWarned)
|
|
161
|
+
console.error(`[scenescout] could not write the live view's token file in ${dir}: ${err instanceof Error ? err.message : String(err)}`);
|
|
162
|
+
liveTokenWarned = true;
|
|
163
|
+
})
|
|
164
|
+
.finally(() => liveTokenWrites.delete(dir));
|
|
165
|
+
liveTokenWrites.set(dir, write);
|
|
166
|
+
}
|
|
167
|
+
/**
|
|
168
|
+
* Started with the first attach rather than at boot: a server nobody attaches
|
|
169
|
+
* to should not open a port. Never rejects — observability is best-effort, and
|
|
170
|
+
* a port that will not open must not cost the run anything.
|
|
171
|
+
*/
|
|
172
|
+
async function ensureLive(dir) {
|
|
173
|
+
if (process.env[LIVE_ENV] === "off")
|
|
174
|
+
return;
|
|
175
|
+
if (liveAddress)
|
|
176
|
+
return publishLiveToken(dir);
|
|
177
|
+
liveStart ??= startLiveServer();
|
|
178
|
+
await liveStart;
|
|
179
|
+
if (!liveAddress)
|
|
180
|
+
return;
|
|
181
|
+
publishLiveToken(dir);
|
|
182
|
+
// The port is new, so a reader polling status.json needs it rewritten.
|
|
183
|
+
flushStatus(dir);
|
|
184
|
+
}
|
|
185
|
+
function startLiveServer() {
|
|
186
|
+
const starting = new LiveServer(liveProvider);
|
|
187
|
+
return starting
|
|
188
|
+
.start()
|
|
189
|
+
.then((address) => {
|
|
190
|
+
live = starting;
|
|
191
|
+
liveAddress = address;
|
|
192
|
+
liveError = null;
|
|
193
|
+
})
|
|
194
|
+
.catch((err) => {
|
|
195
|
+
// Say why, once, and let a later attach try again.
|
|
196
|
+
const reason = err instanceof Error ? err.message : String(err);
|
|
197
|
+
if (liveError !== reason)
|
|
198
|
+
console.error(`[scenescout] the live view could not start: ${reason}`);
|
|
199
|
+
liveError = reason;
|
|
200
|
+
liveStart = null;
|
|
201
|
+
});
|
|
202
|
+
}
|
|
203
|
+
/**
|
|
204
|
+
* The line that hands the live view to the person running the agent. The
|
|
205
|
+
* address holds the token, and a tool result lands in the client's transcript;
|
|
206
|
+
* that is accepted (ADR 7) because the address answers on this machine only.
|
|
207
|
+
*/
|
|
208
|
+
function liveLine() {
|
|
209
|
+
if (!liveAddress)
|
|
210
|
+
return liveError ? `\nLive view unavailable: ${liveError}` : "";
|
|
211
|
+
return (`\nLive view: http://127.0.0.1:${liveAddress.port}/${liveAddress.token}/ — give this address to the user so they can watch every session ` +
|
|
212
|
+
`(current tool, page thumbnail, optional live stream). It opens on this machine only and cannot act on the run.`);
|
|
213
|
+
}
|
|
214
|
+
function flushStatus(dir) {
|
|
215
|
+
// Fire-and-forget: status is best-effort observability on every tool call's
|
|
216
|
+
// hot path and must never add blocking filesystem latency. The writer
|
|
217
|
+
// queues writes per directory and lands each by rename, so a reader never
|
|
218
|
+
// sees a torn file.
|
|
219
|
+
void writeStatusFile(dir, JSON.stringify({
|
|
220
|
+
pid: process.pid,
|
|
221
|
+
phase: lastWriter?.phase ?? "idle",
|
|
222
|
+
tool: lastWriter?.tool ?? "",
|
|
223
|
+
session: lastWriter?.session ?? "",
|
|
224
|
+
role: lastWriter?.role ?? "anonymous",
|
|
225
|
+
sessions: [...engines.keys()],
|
|
226
|
+
url: lastWriter?.url ?? "",
|
|
227
|
+
at: new Date().toISOString(),
|
|
228
|
+
// Everything above describes one session. This is all of them.
|
|
229
|
+
detail: board.list(),
|
|
230
|
+
...(liveAddress ? { live: { port: liveAddress.port } } : liveError ? { live: { error: liveError } } : {}),
|
|
231
|
+
}, null, 2));
|
|
232
|
+
}
|
|
86
233
|
/**
|
|
87
234
|
* Live status for the tested project (`scenescout status <project>` or any
|
|
88
|
-
* supervising layer reads this):
|
|
235
|
+
* supervising layer reads this): what every session is doing right now.
|
|
89
236
|
* Best-effort — observability must never break the tool call itself.
|
|
90
237
|
*/
|
|
91
|
-
function writeStatus(session, phase, tool) {
|
|
92
|
-
const
|
|
93
|
-
|
|
238
|
+
function writeStatus(session, phase, tool, budgetMs) {
|
|
239
|
+
const eng = engines.get(session);
|
|
240
|
+
const dir = eng?.memory?.dir;
|
|
241
|
+
if (!eng || !dir)
|
|
94
242
|
return;
|
|
95
|
-
//
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
void fs.promises
|
|
100
|
-
.writeFile(path.join(dir, "status.json"), JSON.stringify({
|
|
101
|
-
pid: process.pid,
|
|
243
|
+
// status.json is a poll target that gets pasted into bug reports.
|
|
244
|
+
const { task, objective, ...described } = eng.liveDescription;
|
|
245
|
+
lastWriter = board.update(session, {
|
|
246
|
+
role: eng.role,
|
|
102
247
|
phase,
|
|
103
248
|
tool,
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
249
|
+
url: redactSecrets(eng.currentUrl),
|
|
250
|
+
...(budgetMs ? { budgetMs } : {}),
|
|
251
|
+
...described,
|
|
252
|
+
...(task ? { task: redactSecrets(task) } : {}),
|
|
253
|
+
...(objective ? { objective: redactSecrets(objective) } : {}),
|
|
254
|
+
});
|
|
255
|
+
void ensureLive(dir);
|
|
256
|
+
flushStatus(dir);
|
|
112
257
|
}
|
|
113
258
|
/** The watchdog's timeout answer — a diagnosable result, not a hang. */
|
|
114
259
|
function watchdogTimeout(label, ms) {
|
|
@@ -129,7 +274,7 @@ function serializedPerSession(label, fn, timeoutMs = 60_000) {
|
|
|
129
274
|
return (args) => {
|
|
130
275
|
const session = args.session ?? activeName;
|
|
131
276
|
const exec = async () => {
|
|
132
|
-
writeStatus(session, "running", label);
|
|
277
|
+
writeStatus(session, "running", label, timeoutMs);
|
|
133
278
|
try {
|
|
134
279
|
const out = await withWatchdog(label, fn(args, session), timeoutMs, watchdogTimeout);
|
|
135
280
|
// `activeName` is process-global and every scout_attach moves it. With
|
|
@@ -167,6 +312,48 @@ const sessionParam = z
|
|
|
167
312
|
.max(40)
|
|
168
313
|
.optional()
|
|
169
314
|
.describe("Target this session directly instead of the active one — pass it explicitly when dispatching to MULTIPLE sessions in one turn (e.g. two scout_click calls with different `session`), which then run CONCURRENTLY rather than queueing. Omit for single-session sequential use.");
|
|
315
|
+
// The method, for every client that has no skill loader. It is read per call,
|
|
316
|
+
// not cached: a source checkout's skill file can change under a running server.
|
|
317
|
+
server.registerTool(PLAYBOOK_TOOL, {
|
|
318
|
+
description: "Return the SceneScout testing method: setup order, write modes, how to explore, what counts as done, how to report. " +
|
|
319
|
+
"Call this ONCE before the first scout_attach in a conversation, then follow it. " +
|
|
320
|
+
"If this client offers a SceneScout skill, load that instead — it is the same text, so never read both. Takes no input and touches no browser.",
|
|
321
|
+
// No inputSchema on purpose: with one, a call that carries no `arguments` field is rejected as invalid.
|
|
322
|
+
}, async () => {
|
|
323
|
+
try {
|
|
324
|
+
return { content: [{ type: "text", text: loadPlaybook(PACKAGE_ROOT) }] };
|
|
325
|
+
}
|
|
326
|
+
catch (err) {
|
|
327
|
+
return errorText(err);
|
|
328
|
+
}
|
|
329
|
+
});
|
|
330
|
+
// The same method as a prompt, for clients that list server prompts as commands.
|
|
331
|
+
// Registered on the protocol server directly: the SDK's prompt helper rejects a
|
|
332
|
+
// request that carries no `arguments` object, which is exactly what a client
|
|
333
|
+
// sends when the person typed none, and every argument here is optional.
|
|
334
|
+
server.server.registerCapabilities({ prompts: {} });
|
|
335
|
+
server.server.setRequestHandler(ListPromptsRequestSchema, () => ({
|
|
336
|
+
prompts: [
|
|
337
|
+
{
|
|
338
|
+
name: PLAYBOOK_PROMPT,
|
|
339
|
+
title: "Explore a web app with SceneScout",
|
|
340
|
+
description: "Start an exploratory test session: loads the SceneScout method and states the target.",
|
|
341
|
+
arguments: EXPLORE_PROMPT_ARGUMENTS,
|
|
342
|
+
},
|
|
343
|
+
],
|
|
344
|
+
}));
|
|
345
|
+
server.server.setRequestHandler(GetPromptRequestSchema, (request) => {
|
|
346
|
+
if (request.params.name !== PLAYBOOK_PROMPT)
|
|
347
|
+
throw new McpError(ErrorCode.InvalidParams, `Unknown prompt: ${request.params.name}`);
|
|
348
|
+
let message;
|
|
349
|
+
try {
|
|
350
|
+
message = explorePrompt(loadPlaybook(PACKAGE_ROOT), request.params.arguments);
|
|
351
|
+
}
|
|
352
|
+
catch (err) {
|
|
353
|
+
throw new McpError(ErrorCode.InvalidParams, err instanceof Error ? err.message : String(err));
|
|
354
|
+
}
|
|
355
|
+
return { messages: [{ role: "user", content: { type: "text", text: message } }] };
|
|
356
|
+
});
|
|
170
357
|
server.registerTool("scout_scan", {
|
|
171
358
|
description: "Scan a project directory to discover the frontend workspace, framework, routes, dev command, Playwright auth storage states, and testid conventions. Run this first.",
|
|
172
359
|
inputSchema: { projectPath: z.string().describe("Absolute path to the project root") },
|
|
@@ -179,7 +366,7 @@ server.registerTool("scout_scan", {
|
|
|
179
366
|
}
|
|
180
367
|
}));
|
|
181
368
|
server.registerTool("scout_attach", {
|
|
182
|
-
description: "Launch a browser and attach to a running web app. Write policy is enforced at the NETWORK layer: mode='observe' blocks EVERY request that is not a GET (login and token refresh excepted) — choose it for a target that holds real data, where even an ordinary form submission would create a record; mode='read-only' (default) blocks destructive-labeled elements AND all PUT/PATCH/DELETE + destructive POSTs, but lets ordinary form POSTs through; mode='safe-write' allows creating data and permits updates/deletes ONLY on resources this session created (use when the user wants create/edit flows tested); mode='destructive' allows everything — ONLY when the user explicitly confirmed a disposable/seeded environment. Pass a Playwright storage-state JSON to explore as an authenticated role. Pass `session` to keep MULTIPLE roles alive at once (one browser each, genuinely concurrent) for collaboration testing — target each directly with every tool's `session` param, or use scout_session to set which one is the default; coverage and findings merge into one project memory.",
|
|
369
|
+
description: "Launch a browser and attach to a running web app. First attach in this conversation and you have read neither the SceneScout skill nor scout_playbook? Call scout_playbook before this. Write policy is enforced at the NETWORK layer: mode='observe' blocks EVERY request that is not a GET (login and token refresh excepted) — choose it for a target that holds real data, where even an ordinary form submission would create a record; mode='read-only' (default) blocks destructive-labeled elements AND all PUT/PATCH/DELETE + destructive POSTs, but lets ordinary form POSTs through; mode='safe-write' allows creating data and permits updates/deletes ONLY on resources this session created (use when the user wants create/edit flows tested); mode='destructive' allows everything — ONLY when the user explicitly confirmed a disposable/seeded environment. Pass a Playwright storage-state JSON to explore as an authenticated role. Pass `session` to keep MULTIPLE roles alive at once (one browser each, genuinely concurrent) for collaboration testing — target each directly with every tool's `session` param, or use scout_session to set which one is the default; coverage and findings merge into one project memory.",
|
|
183
370
|
inputSchema: {
|
|
184
371
|
url: z.string().describe("Base URL of the running app, e.g. http://localhost:3000"),
|
|
185
372
|
projectPath: z.string().describe("Absolute path to the project (memory + report live in .scenescout/ here)"),
|
|
@@ -189,15 +376,24 @@ server.registerTool("scout_attach", {
|
|
|
189
376
|
.default("read-only")
|
|
190
377
|
.describe("Write policy (see tool description). Never choose 'destructive' yourself — user opt-in only."),
|
|
191
378
|
headed: z.boolean().default(false).describe("Show the browser window"),
|
|
379
|
+
browser: z
|
|
380
|
+
.enum(["chromium", "firefox", "webkit"])
|
|
381
|
+
.optional()
|
|
382
|
+
.describe("Browser to drive. Default: the SCENESCOUT_BROWSER environment variable, else chromium. firefox and webkit must be downloaded first (scenescout install --browser-only --browsers firefox). Use them for a cross-browser pass; stay on chromium otherwise."),
|
|
192
383
|
viewportWidth: z.number().int().min(320).max(3840).optional().describe("Viewport width (default 1280); use e.g. 390 for a mobile pass"),
|
|
193
384
|
viewportHeight: z.number().int().min(480).max(2400).optional().describe("Viewport height (default 900)"),
|
|
385
|
+
task: z
|
|
386
|
+
.string()
|
|
387
|
+
.max(300)
|
|
388
|
+
.optional()
|
|
389
|
+
.describe("What this session is for, in one sentence (e.g. 'Approve and reject orders as a manager'). Shown to the person watching the live view, next to the goal of whatever scout_journey is active. Worth setting whenever more than one session is running."),
|
|
194
390
|
session: z
|
|
195
391
|
.string()
|
|
196
392
|
.max(40)
|
|
197
393
|
.optional()
|
|
198
394
|
.describe("Session name for multi-role runs (e.g. 'admin', 'qa'). Creates/replaces that session's browser and makes it the default. Default: 'default'."),
|
|
199
395
|
},
|
|
200
|
-
}, serializedControl(async ({ url, projectPath, storageStatePath, mode, headed, viewportWidth, viewportHeight, session, }) => {
|
|
396
|
+
}, serializedControl(async ({ url, projectPath, storageStatePath, mode, headed, browser, viewportWidth, viewportHeight, task, session, }) => {
|
|
201
397
|
try {
|
|
202
398
|
const target = session ?? activeName;
|
|
203
399
|
if (session) {
|
|
@@ -258,9 +454,16 @@ server.registerTool("scout_attach", {
|
|
|
258
454
|
/* conflict detection is best-effort */
|
|
259
455
|
}
|
|
260
456
|
const viewport = viewportWidth && viewportHeight ? { width: viewportWidth, height: viewportHeight } : undefined;
|
|
261
|
-
const out = await eng.attach({ url, projectDir: projectPath, storageStatePath, mode, headed, viewport, memoryStore: store });
|
|
457
|
+
const out = await eng.attach({ url, projectDir: projectPath, storageStatePath, mode, headed, browser, viewport, task, memoryStore: store });
|
|
262
458
|
eng.role = storageStatePath ? path.basename(storageStatePath).replace(/\.json$/i, "") : "anonymous";
|
|
263
|
-
|
|
459
|
+
// Put the session on the board now, so the live view shows it before its
|
|
460
|
+
// first tool call. liveLine() needs the port, so the server is awaited
|
|
461
|
+
// here rather than started in the background by writeStatus.
|
|
462
|
+
if (eng.memory?.dir) {
|
|
463
|
+
await ensureLive(eng.memory.dir);
|
|
464
|
+
writeStatus(target, "idle", "scout_attach");
|
|
465
|
+
}
|
|
466
|
+
return text(out + conflictNote + (engines.size > 1 ? `\n${sessionLines()}` : "") + liveLine(), target);
|
|
264
467
|
}
|
|
265
468
|
catch (err) {
|
|
266
469
|
return errorText(err);
|
|
@@ -286,7 +489,7 @@ server.registerTool("scout_session", {
|
|
|
286
489
|
try {
|
|
287
490
|
name = name ?? session;
|
|
288
491
|
if (!name)
|
|
289
|
-
return text(sessionLines(), activeName);
|
|
492
|
+
return text(sessionLines() + liveLine(), activeName);
|
|
290
493
|
if (!engines.has(name)) {
|
|
291
494
|
return text(`No session named "${name}" yet — create it with scout_attach { session: "${name}", … }.\n${sessionLines()}`, activeName);
|
|
292
495
|
}
|
|
@@ -769,19 +972,35 @@ server.registerTool("scout_close", {
|
|
|
769
972
|
const names = [...engines.keys()];
|
|
770
973
|
// Closes are independent per-browser — run them in parallel so N wedged
|
|
771
974
|
// sessions cost one 8s teardown cap total, not N of them.
|
|
975
|
+
const dirs = new Set();
|
|
976
|
+
for (const e of engines.values())
|
|
977
|
+
if (e.memory?.dir)
|
|
978
|
+
dirs.add(e.memory.dir);
|
|
979
|
+
for (const name of engines.keys())
|
|
980
|
+
live?.dropSession(name);
|
|
772
981
|
await Promise.allSettled([...engines.values()].map((e) => e.close()));
|
|
773
982
|
engines.clear();
|
|
774
983
|
sessionQueue.clear();
|
|
984
|
+
board.clear();
|
|
985
|
+
lastWriter = null;
|
|
986
|
+
for (const dir of dirs)
|
|
987
|
+
flushStatus(dir);
|
|
775
988
|
return text(`All sessions closed (${names.join(", ") || "none were live"}). Memory and reports remain in .scenescout/.`, activeName);
|
|
776
989
|
}
|
|
777
990
|
const name = session ?? activeName;
|
|
778
991
|
const eng = engines.get(name);
|
|
779
992
|
if (!eng)
|
|
780
993
|
return text(`No live session "${name}".`, name);
|
|
994
|
+
live?.dropSession(name);
|
|
781
995
|
await eng.close();
|
|
782
996
|
const saveError = eng.memory?.lastSaveError;
|
|
783
997
|
engines.delete(name);
|
|
784
998
|
sessionQueue.forget(name);
|
|
999
|
+
board.remove(name);
|
|
1000
|
+
if (lastWriter?.session === name)
|
|
1001
|
+
lastWriter = null;
|
|
1002
|
+
if (eng.memory?.dir)
|
|
1003
|
+
flushStatus(eng.memory.dir);
|
|
785
1004
|
if (activeName === name)
|
|
786
1005
|
activeName = engines.keys().next().value ?? "default";
|
|
787
1006
|
return text(`Session "${name}" closed. Memory and report remain in .scenescout/.` +
|
|
@@ -795,13 +1014,34 @@ server.registerTool("scout_close", {
|
|
|
795
1014
|
async function main() {
|
|
796
1015
|
const transport = new StdioServerTransport();
|
|
797
1016
|
await server.connect(transport);
|
|
1017
|
+
// A client that exits by closing the pipe sends no signal. Without this the
|
|
1018
|
+
// process, its port, its browsers and its token file all outlived the run.
|
|
1019
|
+
const clientGone = () => {
|
|
1020
|
+
void shutdown().finally(() => process.exit(0));
|
|
1021
|
+
};
|
|
1022
|
+
transport.onclose = clientGone;
|
|
1023
|
+
// The transport reports a closed pipe on some platforms and not others, and
|
|
1024
|
+
// on Windows the SIGTERM a client sends next is a plain kill that runs no
|
|
1025
|
+
// handler. stdin ending is the one signal every platform gives.
|
|
1026
|
+
process.stdin.once("end", clientGone);
|
|
1027
|
+
process.stdin.once("close", clientGone);
|
|
798
1028
|
// Self-heal across restarts: browsers whose parent crashed/was killed can
|
|
799
1029
|
// linger and have been observed to wedge fresh launches. After connect —
|
|
800
1030
|
// the stdio handshake must not wait on a full process-table scan.
|
|
801
1031
|
setImmediate(() => reapOrphanBrowsers());
|
|
802
1032
|
}
|
|
803
1033
|
async function shutdown() {
|
|
804
|
-
|
|
1034
|
+
// The token outlives nothing: a file left behind would name a port some other process may get next.
|
|
1035
|
+
await Promise.allSettled(liveTokenWrites.values());
|
|
1036
|
+
for (const dir of liveDirs) {
|
|
1037
|
+
try {
|
|
1038
|
+
fs.rmSync(path.join(dir, LIVE_TOKEN_FILE), { force: true });
|
|
1039
|
+
}
|
|
1040
|
+
catch (err) {
|
|
1041
|
+
console.error(`[scenescout] could not remove the live view's token file in ${dir}: ${err instanceof Error ? err.message : String(err)}`);
|
|
1042
|
+
}
|
|
1043
|
+
}
|
|
1044
|
+
await Promise.allSettled([live?.stop(), ...[...engines.values()].map((e) => e.close())]);
|
|
805
1045
|
}
|
|
806
1046
|
process.on("SIGINT", () => {
|
|
807
1047
|
void shutdown().finally(() => process.exit(0));
|
package/dist/playbook.js
ADDED
|
@@ -0,0 +1,83 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The testing method, served by the MCP server itself.
|
|
3
|
+
*
|
|
4
|
+
* The method lives in skills/scenescout/SKILL.md. Claude Code loads that file
|
|
5
|
+
* as a skill; no other client does, so an agent there gets the tools and none
|
|
6
|
+
* of the method: the setup order, the write modes, what counts as done. The
|
|
7
|
+
* server hands the same text to any client through a tool, a prompt and a short
|
|
8
|
+
* pointer in its instructions. One file, so the two can never disagree.
|
|
9
|
+
*/
|
|
10
|
+
import fs from "node:fs";
|
|
11
|
+
import path from "node:path";
|
|
12
|
+
export const PLAYBOOK_TOOL = "scout_playbook";
|
|
13
|
+
export const PLAYBOOK_PROMPT = "explore";
|
|
14
|
+
/** Where the method is kept, relative to the package root. It is in the package's `files`. */
|
|
15
|
+
export const PLAYBOOK_RELATIVE_PATH = path.join("skills", "scenescout", "SKILL.md");
|
|
16
|
+
/**
|
|
17
|
+
* Sent to every client when it connects. Short on purpose: clients cut long
|
|
18
|
+
* instructions, and one that is cut in the middle is worse than a pointer.
|
|
19
|
+
*/
|
|
20
|
+
export const SERVER_INSTRUCTIONS = `SceneScout explores a running web app in a real browser and reports bugs, UX problems and coverage. ` +
|
|
21
|
+
`The scout_* tools are deterministic; the method for using them well is a separate text. ` +
|
|
22
|
+
`If this client offers a SceneScout skill, load that. Otherwise call ${PLAYBOOK_TOOL} once, before the first scout_attach in a conversation, and follow what it returns. ` +
|
|
23
|
+
`They are the same text, so never read both. ` +
|
|
24
|
+
`Never choose mode="destructive" yourself: that needs the user's explicit opt-in.`;
|
|
25
|
+
export const LEVELS = ["minimal", "medium", "extensive"];
|
|
26
|
+
/** What the `explore` prompt accepts, as MCP lists it. Every argument is optional. */
|
|
27
|
+
export const EXPLORE_PROMPT_ARGUMENTS = [
|
|
28
|
+
{ name: "url", description: "URL of the running app, e.g. http://localhost:3000", required: false },
|
|
29
|
+
{ name: "level", description: `How far to go: ${LEVELS.join(", ")}`, required: false },
|
|
30
|
+
{ name: "focus", description: "An area or flow to concentrate on", required: false },
|
|
31
|
+
];
|
|
32
|
+
/** The skill file without its YAML front matter, which only a skill loader reads. */
|
|
33
|
+
export function stripFrontMatter(markdown) {
|
|
34
|
+
// An editor may save the file with a byte-order mark; it would hide the opening fence.
|
|
35
|
+
const text = markdown.replace(/^\uFEFF/, "");
|
|
36
|
+
// The fences may hold nothing between them, and the closing one may be the last line of the file.
|
|
37
|
+
const m = /^---[ \t]*\r?\n(?:[\s\S]*?\r?\n)?---[ \t]*(?:\r?\n|$)/.exec(text);
|
|
38
|
+
// Leading blank LINES go; indentation on the first line of the body stays.
|
|
39
|
+
return (m ? text.slice(m[0].length) : text).replace(/^(?:[ \t]*\r?\n)+/, "");
|
|
40
|
+
}
|
|
41
|
+
/**
|
|
42
|
+
* The method text. Throws when the file is not where the package puts it: an
|
|
43
|
+
* agent told "here is the method" and handed an empty string would proceed
|
|
44
|
+
* without one and never say so.
|
|
45
|
+
*/
|
|
46
|
+
export function loadPlaybook(packageRoot) {
|
|
47
|
+
const file = path.join(packageRoot, PLAYBOOK_RELATIVE_PATH);
|
|
48
|
+
let raw;
|
|
49
|
+
try {
|
|
50
|
+
raw = fs.readFileSync(file, "utf8");
|
|
51
|
+
}
|
|
52
|
+
catch (err) {
|
|
53
|
+
throw new Error(`The SceneScout playbook is missing from this install (${file}): ${err instanceof Error ? err.message : String(err)}. Reinstall the package.`);
|
|
54
|
+
}
|
|
55
|
+
const body = stripFrontMatter(raw);
|
|
56
|
+
if (body.trim().length === 0)
|
|
57
|
+
throw new Error(`The SceneScout playbook at ${file} is empty. Reinstall the package.`);
|
|
58
|
+
// Front matter that was not recognised would be served to the agent as if it were the method.
|
|
59
|
+
if (/^---[ \t]*(\r?\n|$)/.test(body))
|
|
60
|
+
throw new Error(`The SceneScout playbook at ${file} has front matter that could not be read. Reinstall the package.`);
|
|
61
|
+
return body;
|
|
62
|
+
}
|
|
63
|
+
/**
|
|
64
|
+
* The opening message of the `explore` prompt: the method, then what the person
|
|
65
|
+
* asked for. `args` is whatever the client sent, which may be nothing at all.
|
|
66
|
+
* A level the method does not know is refused, not passed on for the agent to guess at.
|
|
67
|
+
*/
|
|
68
|
+
export function explorePrompt(playbook, args) {
|
|
69
|
+
const given = (name) => {
|
|
70
|
+
const v = args?.[name];
|
|
71
|
+
return typeof v === "string" && v.trim() ? v.trim() : undefined;
|
|
72
|
+
};
|
|
73
|
+
const level = given("level");
|
|
74
|
+
if (level !== undefined && !LEVELS.includes(level)) {
|
|
75
|
+
throw new Error(`level "${level}" is not one the method knows. Use one of: ${LEVELS.join(", ")}.`);
|
|
76
|
+
}
|
|
77
|
+
const asks = [
|
|
78
|
+
given("url") ? `Target: ${given("url")}` : "Target: ask me for the URL of the running app, or find it from the project.",
|
|
79
|
+
level ? `Level: ${level}` : "",
|
|
80
|
+
given("focus") ? `Focus: ${given("focus")}` : "",
|
|
81
|
+
].filter(Boolean);
|
|
82
|
+
return `${playbook}\n\n---\n\nRun an exploratory test session following the method above.\n${asks.join("\n")}`;
|
|
83
|
+
}
|