scenescout 3.7.0 → 3.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  import fs from "node:fs";
2
2
  import path from "node:path";
3
- import { SHARED_CHROME_ROUTE } from "./memory.js";
3
+ import { SHARED_CHROME_ROUTE, isEmbedKey } from "./memory.js";
4
4
  import { sayVerification } from "./verify.js";
5
5
  import { feedForSession } from "./live.js";
6
6
  import { buildReplayHtml, evidenceFor } from "./replay.js";
@@ -59,6 +59,38 @@ function violationRollup(oracleLog) {
59
59
  ``,
60
60
  ];
61
61
  }
62
+ /**
63
+ * What came from other sites' frames: their controls, and the violations
64
+ * they caused, by origin. Kept apart from the app's coverage and rollup —
65
+ * an embed's failing request is the embed's behaviour, and its controls are
66
+ * not the app's to cover — but listed, since the app chose to embed them.
67
+ */
68
+ export function embedSection(violations, coverage) {
69
+ if (violations.length === 0 && coverage.total === 0)
70
+ return [];
71
+ const lines = [`## Embeds (other sites' frames)`, ``];
72
+ if (coverage.total > 0) {
73
+ lines.push(`${coverage.exercised}/${coverage.total} of their controls exercised; not counted in the app's coverage or its gap ledger.`, ``);
74
+ }
75
+ if (violations.length > 0) {
76
+ const byOrigin = new Map();
77
+ for (const v of violations) {
78
+ const origin = v.embed ?? "";
79
+ const sig = `${v.kind}: ${v.detail.replace(/\b\d+\b/g, ":n").slice(0, 120)}`;
80
+ const sigs = byOrigin.get(origin) ?? new Map();
81
+ sigs.set(sig, (sigs.get(sig) ?? 0) + 1);
82
+ byOrigin.set(origin, sigs);
83
+ }
84
+ lines.push(`Violations inside them (${violations.length}), reported as the embed's behaviour, not the app's:`, ``, `| Embed | Count | Signature |`, `|---|---|---|`);
85
+ for (const [origin, sigs] of byOrigin) {
86
+ for (const [sig, count] of [...sigs.entries()].sort((a, b) => b[1] - a[1]).slice(0, 8)) {
87
+ lines.push(`| \`${escapeTableCell(origin)}\` | ${count} | \`${escapeTableCell(sig)}\` |`);
88
+ }
89
+ }
90
+ lines.push(``);
91
+ }
92
+ return lines;
93
+ }
62
94
  const SEVERITY_ORDER = { high: 0, medium: 1, low: 2 };
63
95
  const SEVERITY_ICON = { high: "🔴", medium: "🟠", low: "🟡" };
64
96
  /**
@@ -253,13 +285,15 @@ export function classifyFilledStates(memory, facts) {
253
285
  const unsubmitted = new Set();
254
286
  const noSubmitControl = new Set();
255
287
  for (const st of Object.values(memory.states)) {
256
- const keys = Object.keys(st.elements);
288
+ // Another site's frame is not the app's form: typing into an embed's chat leaves no app form unsubmitted.
289
+ const own = Object.fromEntries(Object.entries(st.elements).filter(([key]) => !isEmbedKey(key)));
290
+ const keys = Object.keys(own);
257
291
  const filledForReal = keys.some((key) => st.elements[key].exercised && /^(type|select|upload|plan:(type|select|upload))/.test(st.elements[key].lastAction ?? "") && !isFilterKey(key));
258
292
  if (!filledForReal)
259
293
  continue;
260
294
  if (facts[st.route]?.mutated || mutatedSiblingStep(st.route, facts))
261
295
  continue;
262
- if (offersSubmit(st.elements) || keys.length >= COLLECTOR_CAP)
296
+ if (offersSubmit(own) || keys.length >= COLLECTOR_CAP)
263
297
  unsubmitted.add(st.route);
264
298
  else
265
299
  noSubmitControl.add(st.route);
@@ -431,7 +465,8 @@ export function generateReport(memory, oracleLog, extras, opts = {}) {
431
465
  lines.push(`| States explored | ${cov.states} |`);
432
466
  if (extras)
433
467
  lines.push(`| Design audits this run (all sessions) | ${extras.designAudits} |`);
434
- lines.push(`| Oracle violations this session | ${oracleLog.length} |`);
468
+ const fromEmbeds = oracleLog.filter((v) => v.embed).length;
469
+ lines.push(`| Oracle violations this session | ${oracleLog.length - fromEmbeds}${fromEmbeds > 0 ? ` (plus ${fromEmbeds} inside other sites' frames)` : ""} |`);
435
470
  if (extras?.policyAttributed) {
436
471
  lines.push(`| Errors caused by the tester's own write-policy blocks (not counted above) | ${extras.policyAttributed} |`);
437
472
  }
@@ -614,7 +649,21 @@ export function generateReport(memory, oracleLog, extras, opts = {}) {
614
649
  lines.push(`- \`${r}\``);
615
650
  lines.push(``);
616
651
  }
617
- lines.push(...violationRollup(oracleLog));
652
+ if (extras?.trustedEmbeds && extras.trustedEmbeds.length > 0) {
653
+ lines.push(`## Trusted embeds`);
654
+ lines.push(``);
655
+ lines.push(extras.mode === "safe-write"
656
+ ? `Writes that frames of these origins sent outside the app were allowed, as the user asked; anything they created lives with that provider, and SceneScout cannot list it:`
657
+ : extras.mode === "destructive"
658
+ ? `Named as trusted, and not needed: destructive mode let every embed's writes out:`
659
+ : `Named as trusted, but not applied: trust only counts in safe-write mode, and this run was ${extras.mode ?? "read-only"}:`);
660
+ lines.push(``);
661
+ for (const o of extras.trustedEmbeds)
662
+ lines.push(`- \`${o}\``);
663
+ lines.push(``);
664
+ }
665
+ lines.push(...violationRollup(oracleLog.filter((v) => !v.embed)));
666
+ lines.push(...embedSection(oracleLog.filter((v) => v.embed), cov.embeds));
618
667
  if (cov.unexercised.length > 0) {
619
668
  lines.push(`## Unexplored surface (for the next run)`);
620
669
  lines.push(``);
@@ -243,6 +243,7 @@ function reportExtras(eng) {
243
243
  createdResources: eng.createdResources,
244
244
  unvisitedRoutes: unvisited,
245
245
  mode: eng.mode,
246
+ trustedEmbeds: [...eng.trustedEmbeds],
246
247
  policyAttributed: eng.oracleLog.policyAttributed,
247
248
  version: PKG_VERSION,
248
249
  attachedSessions: [...engines.keys()],
@@ -628,6 +629,12 @@ server.registerTool("scout_attach", {
628
629
  .max(60000)
629
630
  .optional()
630
631
  .describe("A floor between actions, in milliseconds, for when a person is watching and needs to keep up — following a flow, taking notes, demonstrating. Default 0: as fast as the page allows, which is what a run wants otherwise. Changeable mid-run with scout_session {paceMs}."),
632
+ trustedEmbeds: z
633
+ .array(z.string().max(200))
634
+ .max(10)
635
+ .optional()
636
+ .describe('Origins of embedded frames (e.g. "https://pay.example.com") whose writes out of the app may go out — ONLY when the user named them, typically a provider in test mode, and only in safe-write mode. ' +
637
+ "Never add one yourself. Hostile input, repeated-click probes and uploads stay refused in them."),
631
638
  record: z
632
639
  .boolean()
633
640
  .default(false)
@@ -639,7 +646,7 @@ server.registerTool("scout_attach", {
639
646
  .optional()
640
647
  .describe("Session name for multi-role runs (e.g. 'admin', 'qa'). Creates/replaces that session's browser and makes it the default. Default: 'default'."),
641
648
  },
642
- }, serializedControl(async ({ url, projectPath, storageStatePath, mode, headed, browser, viewportWidth, viewportHeight, objective, task, record, paceMs, session, }) => {
649
+ }, serializedControl(async ({ url, projectPath, storageStatePath, mode, headed, browser, viewportWidth, viewportHeight, objective, task, record, trustedEmbeds, paceMs, session, }) => {
643
650
  try {
644
651
  const target = session ?? activeName;
645
652
  if (session) {
@@ -722,6 +729,7 @@ server.registerTool("scout_attach", {
722
729
  paceMs,
723
730
  task: objective ? task : undefined,
724
731
  record,
732
+ trustedEmbeds,
725
733
  memoryStore: store,
726
734
  });
727
735
  eng.role = storageStatePath ? path.basename(storageStatePath).replace(/\.json$/i, "") : "anonymous";
@@ -1169,7 +1177,7 @@ server.registerTool("scout_coverage", {
1169
1177
  `⚠ MEMORY WRITE FAILING: ${eng.memory.lastSaveError} — coverage/findings since the last successful write are NOT persisted to disk. If this doesn't clear on its own, check the project directory still exists and is writable.`,
1170
1178
  ]
1171
1179
  : []),
1172
- `States known: ${cov.states} · Elements exercised: ${cov.elementsExercised}/${cov.elementsTotal}`,
1180
+ `States known: ${cov.states} · Elements exercised: ${cov.elementsExercised}/${cov.elementsTotal}${cov.embeds.total > 0 ? ` (plus ${cov.embeds.exercised}/${cov.embeds.total} inside other sites' frames, not counted)` : ""}`,
1173
1181
  formatRouteCoverage(eng.allKnownRoutes(), unvisited),
1174
1182
  `Unexercised elements by route:`,
1175
1183
  ...cov.unexercised.slice(0, 25).map((u) => ` ${u.state}: ${u.keys.slice(0, 6).join(", ")}${u.keys.length > 6 ? ` … +${u.keys.length - 6}` : ""}`),
@@ -1254,6 +1262,7 @@ server.registerTool("scout_report", {
1254
1262
  createdResources: eng.createdResources,
1255
1263
  unvisitedRoutes: unvisited,
1256
1264
  mode: eng.mode,
1265
+ trustedEmbeds: [...eng.trustedEmbeds],
1257
1266
  policyAttributed: eng.oracleLog.policyAttributed,
1258
1267
  // Which sessions are still open decides whether a quiet one is holding a browser, and how long its trailing idle runs.
1259
1268
  attachedSessions: [...engines.keys()],
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "scenescout",
3
- "version": "3.7.0",
3
+ "version": "3.9.0",
4
4
  "description": "SceneScout — exploratory UI testing for AI coding agents. An MCP server that gives any agent (Claude Code, Cursor, VS Code Copilot, Codex, Gemini CLI and others) a structured view of a running web app, always-on oracles, a network-level write policy, memory across runs and a gap-checked report.",
5
5
  "license": "MIT",
6
6
  "author": "brunoboto96",
@@ -34,7 +34,7 @@ You are the brain of an exploratory UI tester. The SceneScout MCP server gives y
34
34
  1. **`scout_crawl` first, always.** One call visits every known route (pass `paths` to sweep a specific subset instead), records coverage, and returns per-route health. This is the whole breadth pass — do not visit routes one-by-one with navigate+snapshot.
35
35
  2. **Investigate what the crawl flagged.** For each problem route (violations, dead-ends, auth-redirects): navigate there, `scout_snapshot`, reproduce, then `scout_finding`.
36
36
  3. **Run journeys with `scout_run_plan {steps}`.** Mechanical sequences (fill form → submit → check) go in ONE plan call — `steps` is an ordered list of `{action, target, value}` — with `testid=`/`text=`/`label=` targets — not one LLM turn per click. The plan aborts at the first violation and tells you where; that's your cue to investigate interactively.
37
- 4. **Snapshot economics:** `scout_snapshot` after landing somewhere new; re-snapshots of the same route return *diffs* with stable refs — "No element changes" costs you almost nothing. `scout_screenshot` ONLY for suspected pixel-native issues (a canvas, a rendering glitch); geometry problems (overlap, off-screen, a covered control) are already in the snapshot as GEOMETRY issues, and images that failed to load are listed under BROKEN IMAGES — file those, quoting the line.
37
+ 4. **Snapshot economics:** `scout_snapshot` after landing somewhere new; re-snapshots of the same route return *diffs* with stable refs — "No element changes" costs you almost nothing. `scout_screenshot` ONLY for suspected pixel-native issues (a canvas, a rendering glitch); geometry problems (overlap, off-screen, a covered control) are already in the snapshot as GEOMETRY issues, and images that failed to load are listed under BROKEN IMAGES — file those, quoting the line. Controls inside embeds (`<iframe>`) are listed with the page's, each marked `⟨in … frame⟩`, and act by ref like any other. A frame of the app's own origin is the app: test it fully. A frame of another site (`cross-origin`) is someone else's system that the user has not authorised you to test: click and type there as a user would, to see the embed render and respond, but never send it hostile input — the engine refuses markup, values over 200 characters, control characters, repeated-click probes and uploads there, and you must not try SQL, template or other injection shapes either — and file what you find there as the embed's behaviour, naming its origin, not as the app's bug. A failing request such a frame sent outside the app says so (`in an embed of …`) and is kept at medium; console errors cannot be told apart by frame and stay the app's; its controls are counted apart and never enter the app's coverage or gap ledger. Its container text is masked, and the writes it sends outside the app are refused in every mode but destructive — unless the user names that origin as a trusted embed (a provider in test mode, say): pass it in `scout_attach {trustedEmbeds: ["https://…"]}` and, in safe-write only, its writes go out. Never add an origin the user did not name. The FRAMES line names frames that were not read; say in your summary that their contents were not explored.
38
38
  5. **Native-user behaviours.** `scout_type {ref, textValue}` (or its alias `value`, matching `scout_select` and a plan step) APPENDS when a field already has content (menu clicks often insert @-mention chips or commands into composers — appending preserves them; the result reports what was already there); pass `replace=true` only to deliberately clear, and `pressEnter=true` to submit from the field the way a user would. Before concluding a badge, icon, or "N errors" indicator *does nothing*, `scout_hover` it — tooltips and hover cards are invisible to snapshots and clicks, and hover output includes what appeared. In HEADED mode (`scout_attach {headed:true}`, which the user asks for when they want to watch) the user's physical mouse competes with the synthetic pointer: if a hover reveals nothing and the finding matters, ask the user to move their mouse off the browser window and retry before filing. **Scroll long pages with `scout_scroll`** — the design audit and snapshot measure at the current scroll position, so judge deep sections by scrolling then re-auditing; it refuses to scroll where a real user couldn't and reports SCROLL LOCKED (the leaked modal scroll-lock that silently amputates everything below the fold — snapshots also flag it passively as an OVERLAY line), and scrolling triggers lazy-loaded content whose failures surface as fresh oracle violations. Elements fully clipped inside an overflow-hidden container are flagged UNREACHABLE in GEOMETRY issues — no amount of scrolling reveals them; that's a high-value layout bug, distinct from merely below-the-fold content. **A page can hold SEVERAL independent scroll regions** and plain `scout_scroll` moves the largest one, so a sidebar nav beside a taller main pane never budges: pass `scout_scroll {target:"testid=…"}` to scroll one region. Never report a nav item, tab or list row as missing/truncated until you have scrolled ITS container — content scrolled out of a secondary pane looks exactly like content that was cut off.
39
39
  6. **The rest of the input vocabulary.** `scout_select` sets a `<select>` option by value or visible label — use it rather than clicking a native dropdown open, which does not render as page DOM. `scout_press` sends a real key to the focused element (`Escape` to dismiss a modal, `Tab` to walk focus order, `Enter` to submit from a field); it is also how the keyboard-only pass at `extensive` is performed, and it vets the focused control first so a destructive action cannot be triggered blind in read-only mode. **`scout_upload {ref}` attaches a file the way a user does** — `ref` is a visible `<input type=file>` (snapshots list these with role `file`; `scout_type` on one redirects here) OR the button/label/dropzone that opens the file chooser (the chooser is intercepted and answered — that is how the hidden input behind a styled "Choose file" control is reached); omit `ref` when the page has exactly one file input, hidden or not (snapshots disclose hidden ones on a FILE INPUTS line). Nothing needs to exist on disk: a small VALID fixture is generated in memory, its kind inferred from the input's `accept` attribute or chosen with `fixture` (`pdf`, `png`, `txt`, `csv`, `json`); `filePath` uploads a real file but must live inside the attached project (fenced, like navigation is fenced to the origin); `name` overrides the filename. The result flags a file that violates `accept` (a mismatch the app then ACCEPTS is a validation finding), warns when the app cleared the input after selection, and says whether a state-changing request fired on selection — if none did, either click the form's submit or read the next snapshot for a client-side rejection. Plans take `{action:"upload", target, value:"pdf"}` steps (`target` required). When the input or its trigger was addressed by `ref`, the gap ledger counts an attached-but-unsent file as filled-never-submitted; the ref-less path has no listed element to mark.
40
40
  7. **Say what you are doing: `task` is required before a tool acts.** A session shows two lines to whoever is watching. Its **objective** is the whole remit you were given, set once at `scout_attach {objective}` ("Admin lane: §2 registers, §7 plan gating", "Approve and reject orders as a manager"). Its **task** is what you are doing *right now*, and every tool that changes the app or the page — `scout_navigate`, `scout_back`, `scout_click`, `scout_type`, `scout_select`, `scout_press`, `scout_upload`, `scout_run_plan` — takes it: a few words for the batch in front of you ("Filtering the documents register by status", "Filling the deviation form with invalid dates", "Signing in as QA_Team"). Say what you are DOING, not what you are checking — "§2.4 filtering narrows the set and is reflected in the URL" is the acceptance criteria, which is the result you will judge, not the batch you are running; naming the item is fine ("§2.4: filtering the documents register"). The task STAYS SET until you pass a different one, so a batch costs a few words, not one per call — pass a fresh one whenever you move on. Acting with none standing is refused: the person watching would otherwise see a session clicking through their app with nothing to say why. `scout_journey {action:"start", goal:…}` sets the task too while it runs — use a journey when you are MEASURING a whole user task, the parameter for everything else.