scenescout 3.21.3 → 3.23.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -3,6 +3,7 @@ import crypto from "node:crypto";
3
3
  import path from "node:path";
4
4
  import { laneRoutePaths, normalizePath, shortHash, stripRouteQuery } from "./fingerprint.js";
5
5
  import { isFormBookkeeping } from "./forms.js";
6
+ import { readFromRunNotes, readRunRecords, recordFromJson, unionFromRunNotes, unionRecords } from "./from-run.js";
6
7
  import { addReading, addVerdict, MAX_TICKETS_KEPT, mergeTicketData } from "./tickets.js";
7
8
  /**
8
9
  * The element-name rule current keys are made under. 2: a link or button with
@@ -469,6 +470,17 @@ export function mergeMemory(mine, theirs) {
469
470
  }
470
471
  if (Object.keys(out.laneRoutes).length === 0)
471
472
  delete out.laneRoutes;
473
+ // A ci run notes what it started from in its own process, beside the server's writes: a union, as for decisions.
474
+ const records = unionRecords(readRunRecords(theirs), readRunRecords(mine));
475
+ if (records.length > 0)
476
+ out.runRecords = records;
477
+ else
478
+ delete out.runRecords;
479
+ const notes = unionFromRunNotes(readFromRunNotes(theirs), readFromRunNotes(mine));
480
+ if (notes.length > 0)
481
+ out.fromRuns = notes;
482
+ else
483
+ delete out.fromRuns;
472
484
  // Lanes record criterion verdicts in their own processes too.
473
485
  const ticketData = mergeTicketData(mine, theirs);
474
486
  if (ticketData.tickets)
@@ -1064,6 +1076,13 @@ export class MemoryStore {
1064
1076
  this.sessionStates.clear();
1065
1077
  this.runStates.clear();
1066
1078
  this.runId = newRunId();
1079
+ this.runStartedAt = new Date().toISOString();
1080
+ }
1081
+ /** When this run began: the store opening, or the last endRun. */
1082
+ runStartedAt = new Date().toISOString();
1083
+ /** This run's entries in the action log, in order: what a run record's steps are built from. */
1084
+ actionsThisRun() {
1085
+ return this.actionLog.filter((a) => a.at >= this.runStartedAt);
1067
1086
  }
1068
1087
  /**
1069
1088
  * This run's identity. A finding remembers the last run that filed it, so
@@ -1184,10 +1203,16 @@ export class MemoryStore {
1184
1203
  */
1185
1204
  recordFormSubmit(fingerprint, key, empty) {
1186
1205
  const entry = this.formEntry(fingerprint, key, false);
1206
+ if (entry)
1207
+ entry.submitted = true;
1187
1208
  if (entry && empty)
1188
1209
  entry.triedEmpty = true;
1189
1210
  return entry !== null;
1190
1211
  }
1212
+ /** Forms seen this run that no session submitted at all, in the order they were first seen. */
1213
+ formsNeverSubmitted() {
1214
+ return [...this.emptySubmits.values()].filter((f) => !f.submitted).map(({ route, key }) => ({ route, key }));
1215
+ }
1191
1216
  formEntry(fingerprint, key, create = true) {
1192
1217
  // Another site's frame is not the app's form to probe.
1193
1218
  if (isEmbedKey(key))
@@ -1626,6 +1651,40 @@ export class MemoryStore {
1626
1651
  const judged = new Set(verdicts.map((v) => v.ticket));
1627
1652
  return { tickets: this.tickets.filter((t) => t.loadedAt >= this.sessionStart || judged.has(t.id)), verdicts };
1628
1653
  }
1654
+ /** What earlier runs left when they wrote their reports, oldest first. Empty on a project none has reported on. */
1655
+ get runRecords() {
1656
+ return readRunRecords(this.data);
1657
+ }
1658
+ /** Keep what this run left (from-run.ts buildRunRecord), written when its report is. */
1659
+ addRunRecord(record) {
1660
+ this.data.runRecords = unionRecords(this.runRecords, [record]);
1661
+ this.flush();
1662
+ }
1663
+ /** The runs that started from an earlier one's record since this store opened, for the report. */
1664
+ fromRunsThisRun() {
1665
+ return readFromRunNotes(this.data).filter((n) => n.at >= this.sessionStart);
1666
+ }
1667
+ /** Note that this run started from an earlier one's record. */
1668
+ noteFromRun(note) {
1669
+ this.data.fromRuns = unionFromRunNotes(readFromRunNotes(this.data), [note]);
1670
+ this.flush();
1671
+ }
1672
+ /**
1673
+ * Take in the notes another process wrote in memory.json since this store
1674
+ * last read or wrote it, and nothing else. A ci run notes what it started
1675
+ * from in its own process; the report reads it through this. The next flush
1676
+ * merges the whole document as usual.
1677
+ */
1678
+ foldInFromRuns() {
1679
+ if (!this.changedUnderUs())
1680
+ return;
1681
+ const theirs = this.readForMerge();
1682
+ if (!theirs)
1683
+ return;
1684
+ const merged = unionFromRunNotes(readFromRunNotes(this.data), readFromRunNotes(theirs));
1685
+ if (merged.length > 0)
1686
+ this.data.fromRuns = merged;
1687
+ }
1629
1688
  /** The routes each lane's report said it covered. Empty on a project that has never run one. */
1630
1689
  get laneRoutes() {
1631
1690
  return this.data.laneRoutes ?? {};
@@ -2355,3 +2414,67 @@ export class MemoryStore {
2355
2414
  return (key) => routeCount >= CHROME_MIN_ROUTES && (routesPerKey.get(key) ?? 0) >= chromeMinRoutes;
2356
2415
  }
2357
2416
  }
2417
+ /**
2418
+ * Note in a project's memory.json, from outside the server, that a run started
2419
+ * from an earlier run's record: `scenescout ci` reads the record in its own
2420
+ * process while its server holds the store. Read, add, and replace atomically
2421
+ * under a name of this process's own; the server merges the file on its next
2422
+ * write (mergeMemory unions the notes), so neither side's are lost. A memory
2423
+ * not written yet is started with this note; one that does not parse is not
2424
+ * replaced, and the caller is told.
2425
+ */
2426
+ export function noteFromRunOnDisk(projectDir, note) {
2427
+ const file = path.join(projectDir, MEMORY_DIRNAME, "memory.json");
2428
+ const exists = fs.existsSync(file);
2429
+ if (!exists)
2430
+ fs.mkdirSync(path.dirname(file), { recursive: true });
2431
+ const doc = exists ? JSON.parse(fs.readFileSync(file, "utf8")) : { version: 1, states: {}, findings: [] };
2432
+ if (!doc || doc.version !== 1)
2433
+ throw new Error(`${file} is not a memory file this version reads`);
2434
+ doc.fromRuns = unionFromRunNotes(readFromRunNotes(doc), [note]);
2435
+ const tmp = `${file}.${process.pid}.from-run.tmp`;
2436
+ try {
2437
+ fs.writeFileSync(tmp, JSON.stringify(doc));
2438
+ fs.renameSync(tmp, file);
2439
+ }
2440
+ catch (err) {
2441
+ fs.rmSync(tmp, { force: true });
2442
+ throw err;
2443
+ }
2444
+ }
2445
+ /** The records in a project's memory.json, read without opening a store. None when there is no memory yet. */
2446
+ export function readRunRecordsOnDisk(projectDir) {
2447
+ const file = path.join(projectDir, MEMORY_DIRNAME, "memory.json");
2448
+ if (!fs.existsSync(file))
2449
+ return [];
2450
+ return readRunRecords(JSON.parse(fs.readFileSync(file, "utf8")));
2451
+ }
2452
+ /**
2453
+ * The record a run starts from (--from-run): a ci.json, a project directory
2454
+ * (its memory's records folded into one), a memory.json, or a directory holding
2455
+ * a ci.json. Throws, saying what was looked for, when there is none: a run
2456
+ * asked to continue must not quietly start fresh.
2457
+ */
2458
+ export function loadRunRecord(given) {
2459
+ if (!fs.existsSync(given))
2460
+ throw new Error(`${given} does not exist: give a ci.json or a project directory`);
2461
+ let file = given;
2462
+ if (fs.statSync(given).isDirectory()) {
2463
+ const candidates = [path.join(given, MEMORY_DIRNAME, "memory.json"), path.join(given, "ci.json"), path.join(given, "memory.json")];
2464
+ const found = candidates.find((c) => fs.existsSync(c));
2465
+ if (!found)
2466
+ throw new Error(`${given} holds no ${MEMORY_DIRNAME}/memory.json, ci.json or memory.json`);
2467
+ file = found;
2468
+ }
2469
+ let raw;
2470
+ try {
2471
+ raw = JSON.parse(fs.readFileSync(file, "utf8"));
2472
+ }
2473
+ catch (err) {
2474
+ throw new Error(`${file} could not be read as JSON: ${err instanceof Error ? err.message : String(err)}`);
2475
+ }
2476
+ const read = recordFromJson(raw, file);
2477
+ if (!read.ok)
2478
+ throw new Error(read.error);
2479
+ return read.record;
2480
+ }
@@ -11,6 +11,7 @@ import { formatNeverSubmittedEmpty } from "./forms.js";
11
11
  import { DEFAULT_REPORT_AUDIENCE, formatPlainSection, pictureOf } from "./plain.js";
12
12
  import { formatTicketsPlain, formatTicketsTechnical, ticketSummaryLine } from "./tickets.js";
13
13
  import { COLLECTOR_CAP } from "./collector.js";
14
+ import { buildRunRecord, fromRunLine } from "./from-run.js";
14
15
  function playwrightSkeleton(f) {
15
16
  const routeClass = f.state.split("#")[0].split("?")[0];
16
17
  let gotoPath = routeClass;
@@ -632,6 +633,28 @@ export function withAudience(technical, audience, plain) {
632
633
  ...body,
633
634
  ].join("\n");
634
635
  }
636
+ /**
637
+ * What this run left, as a record a later run continues or replays from
638
+ * (from-run.ts): its steps from the action log, and, on the routes it worked
639
+ * on, the controls never exercised, the forms never submitted, a form filled
640
+ * and never submitted and the options never chosen, with its gap ledger.
641
+ */
642
+ export function runRecordOf(memory, knownRoutes, gaps) {
643
+ const cov = memory.coverage();
644
+ const followed = memory.fromRunsThisRun().at(-1);
645
+ return buildRunRecord({
646
+ runId: memory.runId,
647
+ at: new Date().toISOString(),
648
+ knownRoutes,
649
+ steps: memory.actionsThisRun(),
650
+ unexercised: cov.unexercised.map((u) => ({ route: u.state.split("#")[0], keys: u.keys.filter((k) => !isEmbedKey(k)) })),
651
+ forms: memory.formsNeverSubmitted(),
652
+ filled: classifyFilledStates(memory, memory.routeFacts).unsubmitted,
653
+ unchosen: memory.unchosenOptions(),
654
+ gaps,
655
+ ...(followed ? { followed: { mode: followed.mode, runId: followed.runId } } : {}),
656
+ });
657
+ }
635
658
  /**
636
659
  * The report as the run stands now. `write` is what scout_report does at the
637
660
  * end; the live view renders the same document on request without touching
@@ -674,6 +697,9 @@ export function generateReport(memory, oracleLog, extras, opts = {}) {
674
697
  if (judged)
675
698
  lines.push(`| Finding dedup | ${judged.replace(/\|/g, "/")} |`);
676
699
  lines.push(`| Elements exercised (informational — denominator grows with every state) | ${cov.elementsExercised}/${cov.elementsTotal} |`);
700
+ // A run started from an earlier one's record says how and which, so it can be followed back (from-run.ts).
701
+ for (const n of memory.fromRunsThisRun())
702
+ lines.push(`| Started from | ${escapeTableCell(fromRunLine(n))} |`);
677
703
  lines.push(``);
678
704
  // ---- The tickets the run was given, criterion by criterion. ----
679
705
  const ticketRun = memory.ticketsThisRun();
@@ -55,13 +55,15 @@ import { FIXTURE_KINDS } from "./engine/fixtures.js";
55
55
  import { feedForSession, LIVE_ENV, writeStatusFile, LIVE_TOKEN_FILE, liveEngines, liveTokenFileName, pidAlive, statusFileName, LiveServer, StatusBoard, } from "./engine/live.js";
56
56
  import { liveViewUrl, MCP_APP_MIME, paneData, paneText, STATUS_PANE_URI, STATUS_POLL_TOOL, STATUS_TOOL, } from "./engine/status-pane.js";
57
57
  import { statusPanePage } from "./engine/status-pane-page.js";
58
- import { formatBriefs, MAX_LANES, planLanes } from "./engine/brief.js";
58
+ import { formatBriefs, MAX_LANES, planLanes, replayBriefs } from "./engine/brief.js";
59
59
  import { decideOpen, OPEN_CHOICES, OPEN_ENV, openChoiceFromEnv, openInBrowser } from "./engine/open.js";
60
60
  import { DEFAULT_EXPIRY_MARGIN_MINUTES, DEFAULT_RUN_MINUTES, judgeProfileFile } from "./engine/expiry.js";
61
61
  import { parseLoginArgs } from "./engine/profiles.js";
62
+ import { continueLines, continuePlan, FROM_RUN_ENV, FROM_RUN_MODE_ENV, FROM_RUN_MODES, fromRunLine, replayLines, replayPlan, resolveFromRun, } from "./engine/from-run.js";
63
+ import { loadRunRecord } from "./engine/memory.js";
62
64
  import { LOGIN_WINDOW_MAX_MS, savedLine, startLoginWindow } from "./login-run.js";
63
65
  import { LoginWindows, WAIT_SAYS } from "./engine/signed-in.js";
64
- import { computeGaps, coverageView, formatRouteCoverage, generateReport, replayDocument, reportEvidence } from "./engine/report.js";
66
+ import { computeGaps, coverageView, formatRouteCoverage, generateReport, replayDocument, reportEvidence, runRecordOf, } from "./engine/report.js";
65
67
  import { DEFAULT_REPORT_AUDIENCE, REPORT_AUDIENCES } from "./engine/plain.js";
66
68
  import { describeVerdict, formatWorklist, unknownIds, VERDICTS, verifyWorklist } from "./engine/verify.js";
67
69
  import { CRITERION_VERDICTS, findCriterion, formatCriteriaForLanes, formatReading, isTicketFileName, judgeCriterion, MAX_CRITERION_FINDINGS, MAX_REASON, MAX_TICKET_FILE_BYTES, MAX_TICKET_TEXT, MAX_TICKETS, NOT_TESTED_REASONS, parseTickets, TICKET_FILE_EXTENSIONS, } from "./engine/tickets.js";
@@ -661,9 +663,15 @@ server.registerTool("scout_lane_brief", {
661
663
  .max(240)
662
664
  .optional()
663
665
  .describe(`How long past the run a saved role's login must still last, in minutes (default ${DEFAULT_EXPIRY_MARGIN_MINUTES}).`),
666
+ fromRun: z
667
+ .string()
668
+ .max(1024)
669
+ .optional()
670
+ .describe(`Start from an earlier run's record: a ci.json, or a project directory (relative to this project). With fromRunMode continue, the lanes take the routes it never worked on first, then the ones it left work on with what to do first, then the rest; with replay, one lane per session it had, following its routes and steps in order. Default ${FROM_RUN_ENV}, else none (the stable split).`),
671
+ fromRunMode: z.enum(FROM_RUN_MODES).optional().describe(`With fromRun: continue (default) or replay. Default ${FROM_RUN_MODE_ENV}, else continue.`),
664
672
  session: sessionParam,
665
673
  },
666
- }, serializedPerSession("scout_lane_brief", async ({ lanes, goal, routes, runMinutes, expiryMarginMinutes, }, session) => {
674
+ }, serializedPerSession("scout_lane_brief", async ({ lanes, goal, routes, runMinutes, expiryMarginMinutes, fromRun, fromRunMode, }, session) => {
667
675
  try {
668
676
  const eng = engineFor(session);
669
677
  // Lanes that attach by a saved role all sign in from one file: check it
@@ -684,12 +692,53 @@ server.registerTool("scout_lane_brief", {
684
692
  expiryNote = `⚠ ${verdict.message}\n\n`;
685
693
  }
686
694
  const all = routes && routes.length > 0 ? routes : eng.allKnownRoutes();
687
- const briefs = planLanes(all, lanes, { goal, mode: eng.mode, role: eng.role });
695
+ // Started from an earlier run only when asked, by the input or the server's environment: otherwise the split is the stable one.
696
+ const asked = resolveFromRun({ path: fromRun, mode: fromRunMode }, process.env, { path: "fromRun", mode: "fromRunMode" });
697
+ if (!asked.ok)
698
+ return errorText(new Error(asked.error));
699
+ let briefs;
700
+ let fromRunNote;
701
+ let laneLines;
702
+ if (asked.fromRun) {
703
+ const projectDir = eng.memory ? path.dirname(eng.memory.dir) : process.cwd();
704
+ const source = path.resolve(projectDir, asked.fromRun.path);
705
+ const record = loadRunRecord(source);
706
+ // Named as it was given, so the brief and the report carry no local directory.
707
+ const note = { at: new Date().toISOString(), mode: asked.fromRun.mode, source: asked.fromRun.path, runId: record.runId, recordAt: record.at };
708
+ fromRunNote = fromRunLine(note);
709
+ if (asked.fromRun.mode === "continue") {
710
+ const items = continuePlan(record, all);
711
+ // Every route it orders is split, the record's own included, so none is left out of every lane.
712
+ const order = items.map((i) => i.route);
713
+ briefs = planLanes(order, lanes, { goal, mode: eng.mode, role: eng.role, order });
714
+ laneLines = (b) => continueLines(items, { from: note.source, only: b.routes });
715
+ }
716
+ else {
717
+ // A replay has one lane per session the recorded run had, whatever number was asked for.
718
+ const replay = replayPlan(record);
719
+ if (replay.length === 0)
720
+ return errorText(new Error(`The run recorded in ${note.source} took no steps, so there is nothing to replay.`));
721
+ briefs = replayBriefs(replay, goal);
722
+ const byLane = new Map(briefs.map((b, i) => [b.lane, replay[i]]));
723
+ laneLines = (b) => replayLines(byLane.get(b.lane), { from: note.source });
724
+ }
725
+ eng.memory?.noteFromRun(note);
726
+ }
727
+ else {
728
+ briefs = planLanes(all, lanes, { goal, mode: eng.mode, role: eng.role });
729
+ }
688
730
  laneLedger.nameBriefed(briefs.map((b) => b.lane), (s) => engines.has(s));
689
731
  // A run given tickets tells every lane which criteria it answers.
690
732
  const criteria = eng.memory ? formatCriteriaForLanes(eng.memory.ticketsThisRun().tickets) : "";
691
733
  return text(expiryNote +
692
- formatBriefs(briefs, { goal, mode: eng.mode, role: eng.role, roleProfile: eng.auth.kind === "role" }) +
734
+ formatBriefs(briefs, {
735
+ goal,
736
+ mode: eng.mode,
737
+ role: eng.role,
738
+ roleProfile: eng.auth.kind === "role",
739
+ ...(fromRunNote ? { fromRunNote } : {}),
740
+ ...(laneLines ? { laneLines } : {}),
741
+ }) +
693
742
  (criteria ? `\n${criteria}` : ""), session);
694
743
  }
695
744
  catch (err) {
@@ -1940,6 +1989,8 @@ server.registerTool("scout_report", {
1940
1989
  const eng = engineFor(session);
1941
1990
  if (!eng.memory)
1942
1991
  throw new Error("Not attached.");
1992
+ // A ci run notes the earlier run it started from in its own process: the report names it.
1993
+ eng.memory.foldInFromRuns();
1943
1994
  const unvisited = eng.unvisitedKnownRoutes();
1944
1995
  const gates = [];
1945
1996
  if (unvisited.length > 0) {
@@ -2004,6 +2055,13 @@ server.registerTool("scout_report", {
2004
2055
  attachedSessions: [...engines.keys()],
2005
2056
  });
2006
2057
  void p;
2058
+ // What this run left, for a later run to continue or replay from (--from-run). Never fatal: the report is written.
2059
+ try {
2060
+ eng.memory.addRunRecord(runRecordOf(eng.memory, all, gapList));
2061
+ }
2062
+ catch (err) {
2063
+ logLine(`the run's record could not be kept, so a later run cannot continue from it: ${err instanceof Error ? err.message : String(err)}`);
2064
+ }
2007
2065
  const openNote = html && openDecisions.get(session)?.report ? openForUser("the report", html) : "";
2008
2066
  return text(summary + openNote, session);
2009
2067
  }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "scenescout",
3
- "version": "3.21.3",
3
+ "version": "3.23.0",
4
4
  "description": "SceneScout — exploratory UI testing for AI coding agents. An MCP server that gives any agent (Claude Code, Cursor, VS Code Copilot, Codex, Gemini CLI and others) a structured view of a running web app, always-on oracles, a network-level write policy, memory across runs and a gap-checked report.",
5
5
  "license": "MIT",
6
6
  "author": "brunoboto96",
@@ -73,7 +73,7 @@
73
73
  "mcp-check": "npm run build && npm run mcp-check:run",
74
74
  "mcp-check:run": "tsx scripts/mcp-check.ts",
75
75
  "test": "npm run build && npm run test:unit && npm run smoke:run && npm run mcp-check:run",
76
- "test:unit": "npm run scan-test && npm run oracle-test && npm run policy-test && npm run fixture-test && npm run dispatch-test && npm run design-test && npm run check-test && npm run ci-test && npm run qa-test && npm run export-test && npm run contract-test && npm run pace-test && npm run limits-test && npm run request-test && npm run settle-test && npm run claims-test && npm run verify-test && npm run brief-test && npm run tickets-test && npm run lane-test && npm run calibration-test && npm run bench-test && npm run memory-test && npm run install-test && npm run profiles-test && npm run refresh-test && npm run scripted-login-test && npm run live-test && npm run demo-test && npm run holdout-test && npm run hygiene-test && npm run guide-test",
76
+ "test:unit": "npm run scan-test && npm run oracle-test && npm run policy-test && npm run fixture-test && npm run dispatch-test && npm run design-test && npm run check-test && npm run report-test && npm run ci-test && npm run qa-test && npm run export-test && npm run contract-test && npm run pace-test && npm run limits-test && npm run request-test && npm run settle-test && npm run claims-test && npm run verify-test && npm run brief-test && npm run tickets-test && npm run lane-test && npm run calibration-test && npm run bench-test && npm run memory-test && npm run install-test && npm run profiles-test && npm run refresh-test && npm run scripted-login-test && npm run live-test && npm run demo-test && npm run holdout-test && npm run hygiene-test && npm run guide-test",
77
77
  "scan-test": "tsx scripts/scan-test.ts",
78
78
  "oracle-test": "tsx --test scripts/oracle-test.ts",
79
79
  "policy-test": "tsx --test scripts/policy-test.ts",
@@ -81,6 +81,7 @@
81
81
  "dispatch-test": "tsx --test scripts/dispatch-test.ts",
82
82
  "design-test": "tsx --test scripts/design-test.ts",
83
83
  "check-test": "tsx --test scripts/check-test.ts",
84
+ "report-test": "tsx --test scripts/report-test.ts",
84
85
  "ci-test": "tsx --test scripts/ci-test.ts",
85
86
  "qa-test": "tsx --test scripts/qa-test.ts",
86
87
  "export-test": "tsx --test scripts/export-test.ts",
@@ -62,7 +62,7 @@ Never ask the person to choose a mode, a role name or a level, and never use the
62
62
  8. **Design-connoisseur pass without pixels: `scout_design_audit`.** Run it once per representative page (dashboard, a form, a detail view, a data table). Its output has two tiers: **⚠ measurable defects** (WCAG contrast, tiny targets, clipped text, aspect-distorted images, horizontal overflow, controls with no accessible name and fields labelled only by a placeholder, keyboard tab stops with no visible focus indicator — sampled with real Tab presses) and **→ craft suggestions** (line measure and line-height rhythm, spacing-grid adherence, typography entropy, gray census and accent-hue count, pure-#000 body text, elevation/control consistency, heading structure, indistinguishable links, AI-slop tells like gradient text/glassmorphism/side-stripe borders/identical card grids), closing with a SYSTEM SUMMARY of design-system coherence. Judge every line with product context (dense tables legitimately have small targets; a chart page legitimately uses many hues), and where the answer depends on a convention of the project you cannot see — a spacing scale, link styling in navigation, touch-target size on a desktop-only app — file it as worth a look (below) rather than deciding the convention for the project. File ⚠ defects as `visual`/`a11y`, and genuine → opportunities as `ux-polish` findings **quoting the concrete numbers** — "~142 characters per line (65–75 ideal)" beats "text feels wide". Every audit ends with a **PAGE SCORE** (0–100 overall + a11y/craft/consistency/task-clarity subscores) persisted per route — the report ranks pages worst-first, so re-runs show whether pages got better or worse. Separately, every `scout_snapshot` runs an **overlay/modal probe** automatically: an empty dialog over a grayed page, a backdrop with no dialog, a far-off-centre dialog leaving a blank band, or a dialog extending unreachably below the viewport appear as OVERLAY lines in GEOMETRY issues — treat these as high-value findings (the user is visually stuck). This is where "how could this page be better" gets answered, not just "is it broken".
63
63
  9. **Measure task EASE with `scout_journey`, not just correctness.** Wrap each module's primary task (`{action:"start", goal:"Create an order"}` → do it → `{action:"end", completed:…}`). Navigate by CLICKING like a first-time user — typing a known deep URL shortcuts the very thing being measured (a route you can only reach by editing the address bar is itself a finding). The result gives interaction count, distinct screens, the path taken, and BACKTRACKS — returning to a screen already left is the clearest evidence the next step wasn't discoverable. Its time is ACTIVE time: a pause over 30s between steps (you thinking between turns) is left out and the result says how many were. An abandoned journey (`completed:false`) is a high-severity finding: the task is blocked or undiscoverable, which no passing e2e suite would ever reveal.
64
64
  10. **Walk the auth surface too — anonymously.** Attach a second session WITHOUT a storage-state file (a fresh logged-out profile) and exercise signup, login failure states, and forgot/reset-password **as far as they physically go**. The mailbox wall is expected — reaching "check your email" IS the success condition; everything before it is what you're testing: does submit actually fire (a dead signup button is a high finding), are errors specific and actionable, can the user resend or recover from a typo, does the flow dead-end. Use plausible synthetic identities only (invent `qa-<runid>@example.com`-style addresses, never a real person's), submit each form valid AND invalid, and judge the feedback. Two classic findings live here: a forgot-password that answers "no account with that email" is an **account-enumeration leak** (file as security; "if an account exists, we sent a link" is the correct shape), and a signup that accepts the form then lands on a blank or logged-out page with no guidance is a **journey dead-end**. Signup creates a record, so what this pass may do depends on the mode. In `observe`, fill and submit the auth forms for their CLIENT-SIDE behaviour only: the engine blocks signup, password change and reset, and lets only a login itself go out. Disclose the server-side half as a gap. Actually creating an account needs the user's explicit okay and safe-write mode; the engine tracks the created account like any other creation.
65
- 11. **`scout_coverage` decides what's next** — it lists unvisited routes, unexercised elements, and the options of each dropdown you used that no session has chosen this run (a filter counts as exercised after one choice, and the option you skipped can be the one whose request fails), and each `<form>` seen this run that no session has submitted with every text field blank (submit it once empty: a submit that silently does nothing — no request, no message — is the commonest defect there, and filling a form in first never finds it). In a parallel run it shows THIS session's own work by default — the routes it reached this run and the forms it saw — so another lane's gaps never read as yours; `scope:"project"` shows every session's, tagged with who saw each. A path you crawl by name that answers as a page joins the route list, and a crawled path that landed elsewhere says `REDIRECTED → <route>`. Trust it over your memory. Prefer reaching routes by clicking real navigation; fall back to direct URLs for coverage completeness and re-verification, and say which you used when it affects the finding (see the provenance rule below).
65
+ 11. **`scout_coverage` decides what's next** — it lists unvisited routes, unexercised elements, and the options of each dropdown you used that no session has chosen this run (a filter counts as exercised after one choice, and the option you skipped can be the one whose request fails), and each `<form>` seen this run that no session has submitted with every text field blank (submit it once empty: a submit that silently does nothing — no request, no message — is the commonest defect there, and filling a form in first never finds it). In a parallel run it shows THIS session's own work by default — the routes it reached this run and the forms it saw — so another lane's gaps never read as yours; `scope:"project"` shows every session's, tagged with who saw each. A path you crawl by name that answers as a page joins the route list, and a crawled path that landed elsewhere says `REDIRECTED → <route>`. Trust it over your memory. Prefer reaching routes by clicking real navigation; fall back to direct URLs for coverage completeness and re-verification, and say which you used when it affects the finding (see the provenance rule below). **A run that continues an earlier one** (your first message, or your lane's brief, says so and lists routes in tiers) is there to start where that run left off: take the routes never worked on first, then on each route with work left do exactly what it lists (submit the forms, choose the options, exercise the controls) before anything else on that page, and leave the covered routes for last. Where it says how the earlier run reached a page (a path of steps from another page), take that path to get there. **A replay** lists the earlier run's steps: take them in order, one at a time, say in a note where a step's control is gone or the page differs, and file what you find as usual.
66
66
 
67
67
  ## Answering tickets
68
68
 
@@ -104,7 +104,7 @@ Some flows need a TEAM — a document one role submits and another approves, a r
104
104
  - **Infer the PERSONA behind each role, and write it down.** From what a role can see and do (its nav, its dashboard, the capability matrix in the report), state what this person is FOR: "qa = reviewer — approves orders, assigns reviewers, no admin" / "user = front-line user — reads documents, completes reviews, raises orders". Record it with `scout_note {section:'roles'}`. Then test the persona's WORLD, not just the permissions: does the operator's landing page serve an operator? Is anything they need N clicks deep? The capability matrix's divergent rows are questions, not verdicts — each is either a correct boundary or a gap ("should this role be able to do this?"); say which you believe it is and why.
105
105
  - `scout_close {all: true}` at the end of a multi-role run; `scout_close {session}` to drop one role early.
106
106
  - **Several agents in parallel** (subagents or a workflow, each driving its own session): each agent attaches its session when it STARTS, and the PLANNER closes it by name once it has folded that lane's report. A lane must NOT close its own session before reporting: `scout_lane_report` keeps its decisions against that session's project memory, so a lane that closes first hands back decisions with nowhere to write — the tool says so rather than accepting silently, and the run's calibration section ends up with nothing to measure. Fold, then close. `scout_close` enforces it: a session that `scout_lane_brief` or `scout_lane_report` named as a lane is not closed until a report from it has been ACCEPTED (a refused one does not count, so re-send the corrected object before closing), and `scout_close {all: true}` names every such lane and closes nothing. Pass `force: true` only when a lane will never report, knowing its decisions are lost. Keep the planner's own session open until the last lane is folded: closing every session ends the run, and with it the record of which sessions are lanes. Never open sessions ahead for agents that have not started, and never hand an open session from one agent to the next: an agent waiting for its turn should hold no browser. Give each session an `objective` when you attach it (`scout_attach {session, objective:"Approve and reject orders as a manager"}`) and wrap each goal in `scout_journey`: the live view shows that objective and the task it is on beside the session's feed, which is how the person watching knows what every agent is for. Keep that goal TRUE: one journey per goal, one goal per thing you are checking ("Save a settings change as the auditor", not "Check every page"), ended the moment it is decided and the next one started before you move on. A journey that outlives its goal shows the viewer an objective the session left behind minutes ago. Run roughly as many agents at once as the machine has cores, less two, since each drives a real browser; beyond that they only queue. Exploring one area is well within a mid-tier model, so run these agents on one (Sonnet or its equivalent in your client) unless the user names a model; keep the larger model for the agent that plans the split and writes the report. No agent may call `scout_close {all: true}` while others run — only the last step, once every agent has finished.
107
- - **Let the engine divide the app: `scout_lane_brief {lanes, goal}`.** Call it after the first crawl, when route knowledge is complete. It splits the known routes into whole modules — everything under `/orders` goes to one lane, so that lane carries state between its own steps instead of re-learning the app on every route — deals the modules out so the lanes come out within a route or two of each other, and returns each lane's session name, the `objective` to attach it with, the routes it owns, the route it lands on (its own first route: lanes that all attached on the home page all met its defects first, and several filed the same one), and the rules each lane must follow — file each defect as it is judged, check a list's request status before calling it empty or stuck, submit one markup value on a create form and open where it is listed, probe a withheld control's endpoint, check coverage, and leave the session open for you to fold. Pass `routes` to split a subset instead. When your session attached by a saved `role`, the lanes will all sign in from that one login, so pass `runMinutes` (how long the lanes will run; default 60): the brief is refused if the login will not last that long plus `expiryMarginMinutes` (default 10), and the refusal names the `scenescout login` command for the user to run again — ask them, then call `scout_lane_brief` again. A login the engine cannot be sure about (a session cookie, a token with no expiry, a cookie for another host) is a warning at the top of the brief, not a refusal. Hand each brief to its agent verbatim. Dividing by hand fails in two ways that a finished run cannot tell apart from success: two lanes audit the same register while a third module is never opened, and route coverage reads complete either way; and lanes launch without an objective, so the person watching sees browsers clicking through their app with nothing to say why.
107
+ - **Let the engine divide the app: `scout_lane_brief {lanes, goal}`.** Call it after the first crawl, when route knowledge is complete. It splits the known routes into whole modules — everything under `/orders` goes to one lane, so that lane carries state between its own steps instead of re-learning the app on every route — deals the modules out so the lanes come out within a route or two of each other, and returns each lane's session name, the `objective` to attach it with, the routes it owns, the route it lands on (its own first route: lanes that all attached on the home page all met its defects first, and several filed the same one), and the rules each lane must follow — file each defect as it is judged, check a list's request status before calling it empty or stuck, submit one markup value on a create form and open where it is listed, probe a withheld control's endpoint, check coverage, and leave the session open for you to fold. Pass `routes` to split a subset instead. When the user asks to pick up where an earlier run left off, pass `fromRun` (its `ci.json`, or a project directory): each lane gets its routes in that order with what to do first on each. When they ask to reproduce a run or check a fix, pass `fromRun` with `fromRunMode: "replay"`: one lane per session the earlier run had, each with its steps in order. When your session attached by a saved `role`, the lanes will all sign in from that one login, so pass `runMinutes` (how long the lanes will run; default 60): the brief is refused if the login will not last that long plus `expiryMarginMinutes` (default 10), and the refusal names the `scenescout login` command for the user to run again — ask them, then call `scout_lane_brief` again. A login the engine cannot be sure about (a session cookie, a token with no expiry, a cookie for another host) is a warning at the top of the brief, not a refusal. Hand each brief to its agent verbatim. Dividing by hand fails in two ways that a finished run cannot tell apart from success: two lanes audit the same register while a third module is never opened, and route coverage reads complete either way; and lanes launch without an objective, so the person watching sees browsers clicking through their app with nothing to say why.
108
108
  - **Lanes report in typed decisions, not prose.** Each lane files its findings with `scout_finding` as it goes, so the report already has them; what the lane hands BACK to the planner is one JSON object, the lane report, and nothing else. Get the paragraph to put in a lane's prompt from `scout_lane_report {lane}`: it names the shape (`status`, one decision per observation with `verdict`, `severity`, `category`, a calibrated `confidence` and a machine-signature `evidence`, the `routes` covered, `blocked_by`; the verdict `worth_a_look` carries a `convention` naming what would decide it, and is never scored for calibration; `finding` is the id `scout_finding` returned for that observation), every value's closed set (the categories are `scout_finding`'s) and every length limit, because a limit a lane is not told refuses good replies. When the lane hands back, pass its text to `scout_lane_report {lane, reply}`: it returns the one-line fold (defects, highs, unsure, mean confidence, routes, what blocked it) or the reason the reply was refused, and keeps each decision so the confidence the lane stated can be checked against what the run went on to file — the report's calibration section is built from that, and it only appears once enough decisions exist to mean something. **So the confidence is worth stating honestly**: a lane that writes 0.95 on everything makes the section say so. The fold also lists every defect the lane judged that no finding on the project matches yet: have the lane name each one's finding id in `finding`, which matches exactly, or file it with the same `evidence` (a finding filed without evidence cannot be matched), before you close its session, because a defect that is only in a lane report never reaches the run's report. Prose around ONE fenced JSON block is discarded unread and the report accepted; anything less clear-cut is refused. Relay a refusal to the lane once and ask for the corrected object; never re-judge its prose yourself, and tell every lane that the object IS its final report, since a lane that writes the object and then hands back a summary of it has failed. The planner then folds lanes by counting, not by reading: a lane's reply is a few hundred tokens whatever it found, and a value the schema refuses is caught at the boundary instead of becoming a severity like "Low-Medium" in the report. Where the client lets you set a lane's reasoning effort, use **medium**: in this project's benchmark it judged as well as high in three-quarters of the time and under half of xhigh, low inflated severities, and max took three minutes per lane for no better agreement.
109
109
 
110
110
  ## The impatient-user pass (extensive)