scenescout 3.11.1 → 3.12.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  import fs from "node:fs";
2
2
  import path from "node:path";
3
- import { SHARED_CHROME_ROUTE, isEmbedKey } from "./memory.js";
3
+ import { SHARED_CHROME_ROUTE, isEmbedKey, isWorthALook } from "./memory.js";
4
4
  import { sayVerification } from "./verify.js";
5
5
  import { feedForSession } from "./live.js";
6
6
  import { buildReplayHtml, evidenceFor } from "./replay.js";
@@ -103,6 +103,34 @@ function projectName(dir) {
103
103
  const parent = path.dirname(dir);
104
104
  return path.basename(parent) || path.basename(dir) || "project";
105
105
  }
106
+ /**
107
+ * The "Worth a look" section: observations that are real and are defects only
108
+ * under a convention of the project the run cannot see. Each says what was
109
+ * seen and what would confirm it. Listed below the findings and never in their
110
+ * totals, because SceneScout does not decide a project's conventions; nothing
111
+ * when there are none. The first bullet names the id, as a finding's does, so
112
+ * the live view hangs the frames recorded around it under the entry.
113
+ */
114
+ export function formatWorthALook(items, sessionStart) {
115
+ if (items.length === 0)
116
+ return [];
117
+ const lines = [
118
+ `## Worth a look (${items.length})`,
119
+ ``,
120
+ `Real observations that are defects only under a convention of your project that the run cannot see. They are not counted as findings above; each says what would make it one.`,
121
+ ``,
122
+ ];
123
+ for (const f of items) {
124
+ lines.push(`### ${f.title}`, ``);
125
+ lines.push(`- **Id:** \`${f.id}\` · **Category:** ${f.category}${f.foundAt >= sessionStart ? "" : " · seen in an earlier run"}`);
126
+ lines.push(`- **A defect only if** your project uses ${f.convention ?? "a convention the finding does not name"}`);
127
+ if (f.evidence)
128
+ lines.push(`- **Seen:** \`${f.evidence}\``);
129
+ lines.push(`- **Where:** \`${f.state}\` (${f.url})`);
130
+ lines.push(``, f.detail, ``);
131
+ }
132
+ return lines;
133
+ }
106
134
  /**
107
135
  * How long ago a finding was last seen, for the historical index. A finding
108
136
  * nobody has re-confirmed in four months is a different thing from one seen
@@ -447,7 +475,10 @@ export function generateReport(memory, oracleLog, extras, opts = {}) {
447
475
  const findings = [...memory.findings].sort((a, b) => SEVERITY_ORDER[a.severity] - SEVERITY_ORDER[b.severity]);
448
476
  const lines = [];
449
477
  const resolved = findings.filter((f) => f.status === "resolved");
450
- const open = findings.filter((f) => f.status !== "resolved");
478
+ // Worth a look is listed below the findings and counted nowhere a defect is:
479
+ // "open", "current" and "historical" hold defects only.
480
+ const worthALook = findings.filter((f) => f.status !== "resolved" && isWorthALook(f));
481
+ const open = findings.filter((f) => f.status !== "resolved" && !isWorthALook(f));
451
482
  const now = Date.now();
452
483
  const current = open.filter((f) => f.foundAt >= memory.sessionStart);
453
484
  const historical = open.filter((f) => f.foundAt < memory.sessionStart);
@@ -460,6 +491,8 @@ export function generateReport(memory, oracleLog, extras, opts = {}) {
460
491
  lines.push(`| Metric | Value |`);
461
492
  lines.push(`|---|---|`);
462
493
  lines.push(`| Open findings | ${open.length} (${open.filter((f) => f.severity === "high").length} high) — ${current.length} seen this session, ${historical.length} historical${resolved.length ? `, ${resolved.length} resolved (listed at the bottom)` : ""} |`);
494
+ if (worthALook.length > 0)
495
+ lines.push(`| Worth a look (not counted as defects) | ${worthALook.length} |`);
463
496
  if (extras && extras.routesTotal > 0)
464
497
  lines.push(`| Route coverage | ${extras.routesVisited}/${extras.routesTotal} |`);
465
498
  lines.push(`| States explored | ${cov.states} |`);
@@ -620,6 +653,7 @@ export function generateReport(memory, oracleLog, extras, opts = {}) {
620
653
  lines.push(``);
621
654
  }
622
655
  }
656
+ lines.push(...formatWorthALook(worthALook, memory.sessionStart));
623
657
  if (resolved.length > 0) {
624
658
  lines.push(`## ✅ Resolved (${resolved.length})`);
625
659
  lines.push(``);
@@ -750,6 +784,7 @@ export function generateReport(memory, oracleLog, extras, opts = {}) {
750
784
  `Top open findings:`,
751
785
  ...open.slice(0, 10).map((f) => ` ${SEVERITY_ICON[f.severity]} [${f.severity}] ${f.title} (${f.id})`),
752
786
  ...(open.length > 10 ? [` … +${open.length - 10} more in the report`] : []),
787
+ ...(worthALook.length > 0 ? [`Worth a look (defects only under a project convention; not counted above): ${worthALook.length}`] : []),
753
788
  ``,
754
789
  `Gap ledger${gaps.length === 0 ? ": EMPTY — nothing known left untested" : ` (${gaps.length}):`}`,
755
790
  ...gaps.map((g) => ` ⚠ ${g}`),
@@ -23,7 +23,7 @@
23
23
  * Pure, so the ordering and the wording can be table-tested.
24
24
  */
25
25
  import { normalizePath } from "./fingerprint.js";
26
- import { failingSignatures } from "./memory.js";
26
+ import { failingSignatures, isWorthALook } from "./memory.js";
27
27
  /** What a re-test found. */
28
28
  export const VERDICTS = ["gone", "present", "changed"];
29
29
  export const VERDICT_MEANING = {
@@ -230,10 +230,11 @@ export function retestResults(candidates, pages) {
230
230
  /**
231
231
  * The findings a check re-tests: the open ones it can reproduce by loading a
232
232
  * page, in the campaign's order and within its cap, plus how many open
233
- * findings there are in all, so the report can say how many it left to a run.
233
+ * findings (defects, not worth-a-look) there are in all, so the report can say how many it left to a run.
234
234
  */
235
235
  export function checkRetestPlan(findings) {
236
- const open = findings.filter((f) => (f.status ?? "open") !== "resolved");
236
+ // A worth-a-look is not a defect, so a check never re-tests it and it can never gate.
237
+ const open = findings.filter((f) => (f.status ?? "open") !== "resolved" && !isWorthALook(f));
237
238
  const eligible = open.filter((f) => loadRetest(item(f)) !== null);
238
239
  const candidates = verifyWorklist(eligible)
239
240
  .map(loadRetest)
@@ -34,8 +34,8 @@ import { ErrorCode, GetPromptRequestSchema, ListPromptsRequestSchema, McpError }
34
34
  import { z } from "zod";
35
35
  import { BrowserEngine } from "./engine/browser.js";
36
36
  import { reapOrphanBrowsers } from "./engine/reaper.js";
37
- import { FINDING_CATEGORIES, MemoryStore, mergeableCategories, redactSecrets } from "./engine/memory.js";
38
- import { decodedEntitiesNote, LANE_NAME_MAX, LaneLedger, laneCloseGuard, laneReportInstruction, parseLaneReport, summarizeLaneReport } from "./engine/lane.js";
37
+ import { FINDING_CATEGORIES, isWorthALook, MemoryStore, mergeableCategories, redactSecrets } from "./engine/memory.js";
38
+ import { decodedEntitiesNote, ignoredConventionsNote, LANE_CONVENTION_MAX, LANE_NAME_MAX, LaneLedger, laneCloseGuard, laneReportInstruction, parseLaneReport, summarizeLaneReport, } from "./engine/lane.js";
39
39
  import { MAX_UNFILED_NAMED, unfiledDefects } from "./engine/calibration.js";
40
40
  import { SessionQueue, withWatchdog } from "./engine/dispatch.js";
41
41
  import { FIXTURE_KINDS } from "./engine/fixtures.js";
@@ -562,6 +562,9 @@ server.registerTool("scout_lane_report", {
562
562
  const kept = owner
563
563
  ? owner.addLaneDecisions(lane, parsed.report.decisions.map((d) => ({ ...d, lane, at })))
564
564
  : 0;
565
+ // And the routes it covered, so the benchmark can tell a lane's remark
566
+ // about another lane's page from a verdict on its own.
567
+ owner?.addLaneRoutes(lane, parsed.report.routes);
565
568
  // Say when nothing was kept. Every lane closing its session before the
566
569
  // planner folds its report is the order the method describes, and it
567
570
  // leaves no memory to write to — reporting a bare "accepted" while the
@@ -586,7 +589,7 @@ server.registerTool("scout_lane_report", {
586
589
  : "";
587
590
  laneLedger.fold(lane, engines.get(lane)?.attached === true);
588
591
  const around = parsed.aroundIgnored ? `\n(The text around the report's JSON block was discarded unread.)` : "";
589
- const decoded = decodedEntitiesNote(parsed.entitiesDecoded);
592
+ const decoded = decodedEntitiesNote(parsed.entitiesDecoded) + ignoredConventionsNote(parsed.conventionsIgnored);
590
593
  return {
591
594
  content: [{ type: "text", text: `Lane report accepted — ${summarizeLaneReport(parsed.report)}${note}${around}${decoded}${followUp}` }],
592
595
  };
@@ -1147,28 +1150,39 @@ server.registerTool("scout_finding", {
1147
1150
  .string()
1148
1151
  .optional()
1149
1152
  .describe("Canonical machine signature for dedup, e.g. 'GET /api/reports/dashboard 403' or 'widget dashboard-summary-widget shows 0'. Same bug re-found later should produce the same string."),
1153
+ convention: z
1154
+ .string()
1155
+ .min(1)
1156
+ .max(LANE_CONVENTION_MAX)
1157
+ .optional()
1158
+ .describe("Only for a WORTH-A-LOOK finding: the observation is real, and it is a defect only under a convention of this project you cannot see. Name that convention, e.g. 'a 4px spacing scale' or 'test ids on every control'. The report lists it under \"Worth a look\", apart from the defects, and does not count it as one. Not for \"I could not tell\": leave that unfiled or look closer. Omit for a defect."),
1150
1159
  session: sessionParam,
1151
1160
  },
1152
- }, serializedPerSession("scout_finding", async ({ severity, category, title, detail, evidence, }, session) => {
1161
+ }, serializedPerSession("scout_finding", async ({ severity, category, title, detail, evidence, convention, }, session) => {
1153
1162
  try {
1154
1163
  const eng = engineFor(session);
1155
1164
  if (!eng.memory)
1156
1165
  throw new Error("Not attached — findings need an active session.");
1157
- const [finding, isNew] = eng.memory.addFinding({
1166
+ const [finding, isNew, promoted] = eng.memory.addFinding({
1158
1167
  severity,
1159
1168
  category: category,
1160
1169
  title,
1161
1170
  detail,
1162
1171
  evidence,
1172
+ ...(convention ? { tier: "worth_a_look", convention } : {}),
1163
1173
  url: eng.currentUrl,
1164
1174
  state: eng.currentState || "(unknown)",
1165
1175
  session: eng.sessionKey,
1166
1176
  });
1167
1177
  return text(isNew
1168
- ? `Finding recorded: [${finding.severity}] ${finding.title} (id ${finding.id})`
1169
- : finding.regressedAt
1170
- ? `⟳ REOPENED as a REGRESSION: finding ${finding.id} was previously resolved but the evidence reproduces again (seen in ${finding.runs} runs). Worth calling out to the user.`
1171
- : `Not recorded as new: merged into existing finding ${finding.id} — [${finding.severity}] ${finding.title}${finding.evidence ? ` (evidence: ${finding.evidence.slice(0, 160)})` : " (no evidence)"}, filed as ${finding.category}, seen in ${finding.runs} runs. If yours is a different bug, file it again: under the category that says what is wrong if it is another kind of defect (a finding filed as ${category} merges only with one filed as ${mergeableCategories(category).join(" or ")}), or with evidence naming the request that failed for you (method and path) — two findings are kept apart when both name requests and none is shared.`, session);
1178
+ ? isWorthALook(finding)
1179
+ ? `Recorded as worth a look (not a defect in the report's totals): ${finding.title} (id ${finding.id}) — a defect only if your project uses ${finding.convention}`
1180
+ : `Finding recorded: [${finding.severity}] ${finding.title} (id ${finding.id})`
1181
+ : promoted
1182
+ ? `Merged into finding ${finding.id}, which was worth a look, and promoted to a defect: [${finding.severity}] ${finding.title}. It now counts among the report's findings.`
1183
+ : finding.regressedAt
1184
+ ? `⟳ REOPENED as a REGRESSION: finding ${finding.id} was previously resolved but the evidence reproduces again (seen in ${finding.runs} runs). Worth calling out to the user.`
1185
+ : `Not recorded as new: merged into existing finding ${finding.id} — [${finding.severity}] ${finding.title}${finding.evidence ? ` (evidence: ${finding.evidence.slice(0, 160)})` : " (no evidence)"}, filed as ${finding.category}, seen in ${finding.runs} runs. If yours is a different bug, file it again: under the category that says what is wrong if it is another kind of defect (a finding filed as ${category} merges only with one filed as ${mergeableCategories(category).join(" or ")}), or with evidence naming the request that failed for you (method and path) — two findings are kept apart when both name requests and none is shared.`, session);
1172
1186
  }
1173
1187
  catch (err) {
1174
1188
  return errorText(err);
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "scenescout",
3
- "version": "3.11.1",
3
+ "version": "3.12.0",
4
4
  "description": "SceneScout — exploratory UI testing for AI coding agents. An MCP server that gives any agent (Claude Code, Cursor, VS Code Copilot, Codex, Gemini CLI and others) a structured view of a running web app, always-on oracles, a network-level write policy, memory across runs and a gap-checked report.",
5
5
  "license": "MIT",
6
6
  "author": "brunoboto96",
@@ -39,7 +39,7 @@ You are the brain of an exploratory UI tester. The SceneScout MCP server gives y
39
39
  6. **The rest of the input vocabulary.** `scout_select` sets a `<select>` option by value or visible label — use it rather than clicking a native dropdown open, which does not render as page DOM. `scout_press` sends a real key to the focused element (`Escape` to dismiss a modal, `Tab` to walk focus order, `Enter` to submit from a field); it is also how the keyboard-only pass at `extensive` is performed, and it vets the focused control first so a destructive action cannot be triggered blind in read-only mode. **`scout_upload {ref}` attaches a file the way a user does** — `ref` is a visible `<input type=file>` (snapshots list these with role `file`; `scout_type` on one redirects here) OR the button/label/dropzone that opens the file chooser (the chooser is intercepted and answered — that is how the hidden input behind a styled "Choose file" control is reached); omit `ref` when the page has exactly one file input, hidden or not (snapshots disclose hidden ones on a FILE INPUTS line). Nothing needs to exist on disk: a small VALID fixture is generated in memory, its kind inferred from the input's `accept` attribute or chosen with `fixture` (`pdf`, `png`, `txt`, `csv`, `json`); `filePath` uploads a real file but must live inside the attached project (fenced, like navigation is fenced to the origin); `name` overrides the filename. The result flags a file that violates `accept` (a mismatch the app then ACCEPTS is a validation finding), warns when the app cleared the input after selection, and says whether a state-changing request fired on selection — if none did, either click the form's submit or read the next snapshot for a client-side rejection. Plans take `{action:"upload", target, value:"pdf"}` steps (`target` required). When the input or its trigger was addressed by `ref`, the gap ledger counts an attached-but-unsent file as filled-never-submitted; the ref-less path has no listed element to mark.
40
40
  7. **Say what you are doing: `task` is required before a tool acts.** A session shows two lines to whoever is watching. Its **objective** is the whole remit you were given, set once at `scout_attach {objective}` ("Admin lane: §2 registers, §7 plan gating", "Approve and reject orders as a manager"). Its **task** is what you are doing *right now*, and every tool that changes the app or the page — `scout_navigate`, `scout_back`, `scout_click`, `scout_type`, `scout_select`, `scout_press`, `scout_upload`, `scout_run_plan` — takes it: a few words for the batch in front of you ("Filtering the documents register by status", "Filling the deviation form with invalid dates", "Signing in as QA_Team"). Say what you are DOING, not what you are checking — "§2.4 filtering narrows the set and is reflected in the URL" is the acceptance criteria, which is the result you will judge, not the batch you are running; naming the item is fine ("§2.4: filtering the documents register"). The task STAYS SET until you pass a different one, so a batch costs a few words, not one per call — pass a fresh one whenever you move on. Acting with none standing is refused: the person watching would otherwise see a session clicking through their app with nothing to say why. `scout_journey {action:"start", goal:…}` sets the task too while it runs — use a journey when you are MEASURING a whole user task, the parameter for everything else.
41
41
 
42
- 8. **Design-connoisseur pass without pixels: `scout_design_audit`.** Run it once per representative page (dashboard, a form, a detail view, a data table). Its output has two tiers: **⚠ measurable defects** (WCAG contrast, tiny targets, clipped text, aspect-distorted images, horizontal overflow, keyboard tab stops with no visible focus indicator — sampled with real Tab presses) and **→ craft suggestions** (line measure and line-height rhythm, spacing-grid adherence, typography entropy, gray census and accent-hue count, pure-#000 body text, elevation/control consistency, heading structure, indistinguishable links, AI-slop tells like gradient text/glassmorphism/side-stripe borders/identical card grids), closing with a SYSTEM SUMMARY of design-system coherence. Judge every line with product context (dense tables legitimately have small targets; a chart page legitimately uses many hues). File ⚠ defects as `visual`/`a11y`, and genuine → opportunities as `ux-polish` findings **quoting the concrete numbers** — "~142 characters per line (65–75 ideal)" beats "text feels wide". Every audit ends with a **PAGE SCORE** (0–100 overall + a11y/craft/consistency/task-clarity subscores) persisted per route — the report ranks pages worst-first, so re-runs show whether pages got better or worse. Separately, every `scout_snapshot` runs an **overlay/modal probe** automatically: an empty dialog over a grayed page, a backdrop with no dialog, a far-off-centre dialog leaving a blank band, or a dialog extending unreachably below the viewport appear as OVERLAY lines in GEOMETRY issues — treat these as high-value findings (the user is visually stuck). This is where "how could this page be better" gets answered, not just "is it broken".
42
+ 8. **Design-connoisseur pass without pixels: `scout_design_audit`.** Run it once per representative page (dashboard, a form, a detail view, a data table). Its output has two tiers: **⚠ measurable defects** (WCAG contrast, tiny targets, clipped text, aspect-distorted images, horizontal overflow, keyboard tab stops with no visible focus indicator — sampled with real Tab presses) and **→ craft suggestions** (line measure and line-height rhythm, spacing-grid adherence, typography entropy, gray census and accent-hue count, pure-#000 body text, elevation/control consistency, heading structure, indistinguishable links, AI-slop tells like gradient text/glassmorphism/side-stripe borders/identical card grids), closing with a SYSTEM SUMMARY of design-system coherence. Judge every line with product context (dense tables legitimately have small targets; a chart page legitimately uses many hues), and where the answer depends on a convention of the project you cannot see — a spacing scale, link styling in navigation, touch-target size on a desktop-only app — file it as worth a look (below) rather than deciding the convention for the project. File ⚠ defects as `visual`/`a11y`, and genuine → opportunities as `ux-polish` findings **quoting the concrete numbers** — "~142 characters per line (65–75 ideal)" beats "text feels wide". Every audit ends with a **PAGE SCORE** (0–100 overall + a11y/craft/consistency/task-clarity subscores) persisted per route — the report ranks pages worst-first, so re-runs show whether pages got better or worse. Separately, every `scout_snapshot` runs an **overlay/modal probe** automatically: an empty dialog over a grayed page, a backdrop with no dialog, a far-off-centre dialog leaving a blank band, or a dialog extending unreachably below the viewport appear as OVERLAY lines in GEOMETRY issues — treat these as high-value findings (the user is visually stuck). This is where "how could this page be better" gets answered, not just "is it broken".
43
43
  9. **Measure task EASE with `scout_journey`, not just correctness.** Wrap each module's primary task (`{action:"start", goal:"Create an order"}` → do it → `{action:"end", completed:…}`). Navigate by CLICKING like a first-time user — typing a known deep URL shortcuts the very thing being measured (a route you can only reach by editing the address bar is itself a finding). The result gives interaction count, distinct screens, the path taken, and BACKTRACKS — returning to a screen already left is the clearest evidence the next step wasn't discoverable. An abandoned journey (`completed:false`) is a high-severity finding: the task is blocked or undiscoverable, which no passing e2e suite would ever reveal.
44
44
  10. **Walk the auth surface too — anonymously.** Attach a second session WITHOUT a storage-state file (a fresh logged-out profile) and exercise signup, login failure states, and forgot/reset-password **as far as they physically go**. The mailbox wall is expected — reaching "check your email" IS the success condition; everything before it is what you're testing: does submit actually fire (a dead signup button is a high finding), are errors specific and actionable, can the user resend or recover from a typo, does the flow dead-end. Use plausible synthetic identities only (invent `qa-<runid>@example.com`-style addresses, never a real person's), submit each form valid AND invalid, and judge the feedback. Two classic findings live here: a forgot-password that answers "no account with that email" is an **account-enumeration leak** (file as security; "if an account exists, we sent a link" is the correct shape), and a signup that accepts the form then lands on a blank or logged-out page with no guidance is a **journey dead-end**. Signup creates a record, so what this pass may do depends on the mode. In `observe`, fill and submit the auth forms for their CLIENT-SIDE behaviour only: the engine blocks signup, password change and reset, and lets only a login itself go out. Disclose the server-side half as a gap. Actually creating an account needs the user's explicit okay and safe-write mode; the engine tracks the created account like any other creation.
45
45
  11. **`scout_coverage` decides what's next** — it lists unvisited routes, unexercised elements, and the options of each dropdown you used that no session has chosen this run (a filter counts as exercised after one choice, and the option you skipped can be the one whose request fails), and each `<form>` seen this run that no session has submitted with every text field blank (submit it once empty: a submit that silently does nothing — no request, no message — is the commonest defect there, and filling a form in first never finds it). Trust it over your memory. Prefer reaching routes by clicking real navigation; fall back to direct URLs for coverage completeness and re-verification, and say which you used when it affects the finding (see the provenance rule below).
@@ -74,7 +74,7 @@ Some flows need a TEAM — a document one role submits and another approves, a r
74
74
  - `scout_close {all: true}` at the end of a multi-role run; `scout_close {session}` to drop one role early.
75
75
  - **Several agents in parallel** (subagents or a workflow, each driving its own session): each agent attaches its session when it STARTS, and the PLANNER closes it by name once it has folded that lane's report. A lane must NOT close its own session before reporting: `scout_lane_report` keeps its decisions against that session's project memory, so a lane that closes first hands back decisions with nowhere to write — the tool says so rather than accepting silently, and the run's calibration section ends up with nothing to measure. Fold, then close. `scout_close` enforces it: a session that `scout_lane_brief` or `scout_lane_report` named as a lane is not closed until a report from it has been ACCEPTED (a refused one does not count, so re-send the corrected object before closing), and `scout_close {all: true}` names every such lane and closes nothing. Pass `force: true` only when a lane will never report, knowing its decisions are lost. Keep the planner's own session open until the last lane is folded: closing every session ends the run, and with it the record of which sessions are lanes. Never open sessions ahead for agents that have not started, and never hand an open session from one agent to the next: an agent waiting for its turn should hold no browser. Give each session an `objective` when you attach it (`scout_attach {session, objective:"Approve and reject orders as a manager"}`) and wrap each goal in `scout_journey`: the live view shows that objective and the task it is on beside the session's feed, which is how the person watching knows what every agent is for. Keep that goal TRUE: one journey per goal, one goal per thing you are checking ("Save a settings change as the auditor", not "Check every page"), ended the moment it is decided and the next one started before you move on. A journey that outlives its goal shows the viewer an objective the session left behind minutes ago. Run roughly as many agents at once as the machine has cores, less two, since each drives a real browser; beyond that they only queue. Exploring one area is well within a mid-tier model, so run these agents on one (Sonnet or its equivalent in your client) unless the user names a model; keep the larger model for the agent that plans the split and writes the report. No agent may call `scout_close {all: true}` while others run — only the last step, once every agent has finished.
76
76
  - **Let the engine divide the app: `scout_lane_brief {lanes, goal}`.** Call it after the first crawl, when route knowledge is complete. It splits the known routes into whole modules — everything under `/orders` goes to one lane, so that lane carries state between its own steps instead of re-learning the app on every route — deals the modules out so the lanes come out within a route or two of each other, and returns each lane's session name, the `objective` to attach it with, the routes it owns, the route it lands on (its own first route: lanes that all attached on the home page all met its defects first, and several filed the same one), and the rules each lane must follow — file each defect as it is judged, check a list's request status before calling it empty or stuck, submit one markup value on a create form and open where it is listed, probe a withheld control's endpoint, check coverage, and leave the session open for you to fold. Pass `routes` to split a subset instead. Hand each brief to its agent verbatim. Dividing by hand fails in two ways that a finished run cannot tell apart from success: two lanes audit the same register while a third module is never opened, and route coverage reads complete either way; and lanes launch without an objective, so the person watching sees browsers clicking through their app with nothing to say why.
77
- - **Lanes report in typed decisions, not prose.** Each lane files its findings with `scout_finding` as it goes, so the report already has them; what the lane hands BACK to the planner is one JSON object, the lane report, and nothing else. Get the paragraph to put in a lane's prompt from `scout_lane_report {lane}`: it names the shape (`status`, one decision per observation with `verdict`, `severity`, `category`, a calibrated `confidence` and a machine-signature `evidence`, the `routes` covered, `blocked_by`), every value's closed set (the categories are `scout_finding`'s) and every length limit, because a limit a lane is not told refuses good replies. When the lane hands back, pass its text to `scout_lane_report {lane, reply}`: it returns the one-line fold (defects, highs, unsure, mean confidence, routes, what blocked it) or the reason the reply was refused, and keeps each decision so the confidence the lane stated can be checked against what the run went on to file — the report's calibration section is built from that, and it only appears once enough decisions exist to mean something. **So the confidence is worth stating honestly**: a lane that writes 0.95 on everything makes the section say so. The fold also lists every defect the lane judged that no finding on the project matches yet: have the lane file each with the same `evidence` (a finding filed without evidence cannot be matched), or name the finding that covers it, before you close its session, because a defect that is only in a lane report never reaches the run's report. Prose around ONE fenced JSON block is discarded unread and the report accepted; anything less clear-cut is refused. Relay a refusal to the lane once and ask for the corrected object; never re-judge its prose yourself, and tell every lane that the object IS its final report, since a lane that writes the object and then hands back a summary of it has failed. The planner then folds lanes by counting, not by reading: a lane's reply is a few hundred tokens whatever it found, and a value the schema refuses is caught at the boundary instead of becoming a severity like "Low-Medium" in the report. Where the client lets you set a lane's reasoning effort, use **medium**: in this project's benchmark it judged as well as high in three-quarters of the time and under half of xhigh, low inflated severities, and max took three minutes per lane for no better agreement.
77
+ - **Lanes report in typed decisions, not prose.** Each lane files its findings with `scout_finding` as it goes, so the report already has them; what the lane hands BACK to the planner is one JSON object, the lane report, and nothing else. Get the paragraph to put in a lane's prompt from `scout_lane_report {lane}`: it names the shape (`status`, one decision per observation with `verdict`, `severity`, `category`, a calibrated `confidence` and a machine-signature `evidence`, the `routes` covered, `blocked_by`; the verdict `worth_a_look` carries a `convention` naming what would decide it, and is never scored for calibration), every value's closed set (the categories are `scout_finding`'s) and every length limit, because a limit a lane is not told refuses good replies. When the lane hands back, pass its text to `scout_lane_report {lane, reply}`: it returns the one-line fold (defects, highs, unsure, mean confidence, routes, what blocked it) or the reason the reply was refused, and keeps each decision so the confidence the lane stated can be checked against what the run went on to file — the report's calibration section is built from that, and it only appears once enough decisions exist to mean something. **So the confidence is worth stating honestly**: a lane that writes 0.95 on everything makes the section say so. The fold also lists every defect the lane judged that no finding on the project matches yet: have the lane file each with the same `evidence` (a finding filed without evidence cannot be matched), or name the finding that covers it, before you close its session, because a defect that is only in a lane report never reaches the run's report. Prose around ONE fenced JSON block is discarded unread and the report accepted; anything less clear-cut is refused. Relay a refusal to the lane once and ask for the corrected object; never re-judge its prose yourself, and tell every lane that the object IS its final report, since a lane that writes the object and then hands back a summary of it has failed. The planner then folds lanes by counting, not by reading: a lane's reply is a few hundred tokens whatever it found, and a value the schema refuses is caught at the boundary instead of becoming a severity like "Low-Medium" in the report. Where the client lets you set a lane's reasoning effort, use **medium**: in this project's benchmark it judged as well as high in three-quarters of the time and under half of xhigh, low inflated severities, and max took three minutes per lane for no better agreement.
78
78
 
79
79
  ## The impatient-user pass (extensive)
80
80
 
@@ -99,8 +99,9 @@ The engine is self-healing (orphaned browsers reaped, wedged calls time out with
99
99
  - **Re-testing what earlier runs left open: `scout_verify`.** Against an app this project has tested before, run it right after the crawl, and run it again whenever the user says a wave of fixes has landed. Called bare it returns the open findings in the order to re-test them — worst route first, grouped so you walk a route once instead of once per finding — each with the evidence that identifies it and the steps that produced it; `scout_verify {ids}` narrows it to specific ones, and names any that are not open rather than quietly shortening the list. Re-test a finding, then record what you saw with `scout_verify {id, verdict, note}`: `"gone"` resolves it, `"present"` stamps it confirmed so the report dates the confirmation instead of calling it unverified, `"changed"` keeps it open and says the behaviour differs — and if it is now a different bug, file that as its own finding. A campaign is far cheaper than re-deriving the same list from the report by hand, and it is the only way the historical section stops being a pile nobody trusts.
100
100
  - **A `SHARED CHROME` block in a design audit is ONE finding, not one per page.** Those elements (sidebar, header, breadcrumb bar) are the app shell; the audit already excludes them from the page's score and reports them once. File a single finding for the shell and move on — filing per-page produced five separate tickets for two CSS declarations in a real run.
101
101
  - **`scout_finding` takes `severity`, `category`, `title`, `detail` and `evidence`.** `category` is the finding's kind — `console-error`, `page-error`, `http-error`, `network`, `dead-end`, `ux-confusing`, `ux-polish`, `visual`, `a11y`, `permission-leak`, `data-inconsistency`, `stale-state`, `data-loss`, `performance`, `security`, `missing-testid`, `other` — and is what groups the report.
102
+ - **Worth a look: real, and a defect only under a convention you cannot see.** SceneScout is used on any app, so it does not decide a project's conventions. When an observation is certain but whether it matters depends on one — paddings off a 4px grid (a spacing scale), links styled like body text in navigation, a control with no `data-testid` (test ids on every control), touch-target size (above WCAG 2.2's 24px minimum, which is a published standard and a plain defect) on an app that may be desktop-only, no dominant action on a list page, a list order the page never promises — file it with `scout_finding {…, convention}`, naming the convention in a few words ("a 4px spacing scale"). The report lists it under **Worth a look**, below the findings, as "a defect only if your project uses …", and does not count it as a defect. When the convention IS visible — `scout_scan` found the project's test-id convention, ASSUMPTIONS.md records a spacing scale, the page's own design system states it — the observation is a plain defect or not one: judge it. It is **not** for "I could not tell whether this is broken": look closer, or in a lane report say `unsure`. Filing the same thing later as a defect makes it one; a defect is never turned back into a worth-a-look.
102
103
  - **Always pass `evidence`** to `scout_finding` — a canonical machine signature like `GET /api/reports/dashboard 403` or `widget dashboard-summary-widget shows 0`. It's what deduplicates the same bug across runs when titles get rephrased.
103
- - File judgment findings too: confusing flows, no-feedback actions, state lost on refresh, permission leaks (low-privilege role reaching admin surface), `missing-testid` (low), unnamed interactables (a11y, low).
104
+ - File judgment findings too: confusing flows, no-feedback actions, state lost on refresh, permission leaks (low-privilege role reaching admin surface), `missing-testid` (low, where the project's test-id convention is known; worth a look otherwise), unnamed interactables (a11y, low).
104
105
  - Respect refusals — never retry or route around a policy refusal; note it and move on.
105
106
  - Duplicates are fine; `scout_finding` dedups across runs.
106
107