scenescout 3.15.0 → 3.17.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (59) hide show
  1. package/CHANGELOG.md +87 -0
  2. package/README.md +70 -18
  3. package/dist/browsers.js +28 -0
  4. package/dist/check-run.js +191 -14
  5. package/dist/ci-run.js +268 -52
  6. package/dist/cli.js +107 -47
  7. package/dist/commands.js +3 -2
  8. package/dist/engine/baseline.js +377 -0
  9. package/dist/engine/brief.js +16 -7
  10. package/dist/engine/browser.js +1147 -286
  11. package/dist/engine/calibration.js +61 -30
  12. package/dist/engine/capture.js +164 -0
  13. package/dist/engine/check.js +244 -42
  14. package/dist/engine/ci-lanes.js +215 -0
  15. package/dist/engine/ci.js +136 -18
  16. package/dist/engine/claims.js +159 -3
  17. package/dist/engine/collector.js +561 -30
  18. package/dist/engine/crawl.js +49 -0
  19. package/dist/engine/design.js +281 -38
  20. package/dist/engine/export.js +877 -0
  21. package/dist/engine/fingerprint.js +92 -4
  22. package/dist/engine/flow.js +18 -6
  23. package/dist/engine/forms.js +181 -18
  24. package/dist/engine/journey.js +29 -1
  25. package/dist/engine/lane.js +13 -3
  26. package/dist/engine/launch.js +45 -6
  27. package/dist/engine/limits.js +7 -0
  28. package/dist/engine/live-page.js +49 -2
  29. package/dist/engine/live.js +4 -1
  30. package/dist/engine/memory.js +501 -47
  31. package/dist/engine/open.js +118 -0
  32. package/dist/engine/oracles.js +41 -1
  33. package/dist/engine/plain.js +268 -0
  34. package/dist/engine/png.js +127 -0
  35. package/dist/engine/policy.js +379 -9
  36. package/dist/engine/probes.js +3 -2
  37. package/dist/engine/profiles.js +45 -9
  38. package/dist/engine/project-folder.js +191 -0
  39. package/dist/engine/refresh.js +68 -3
  40. package/dist/engine/replay.js +63 -10
  41. package/dist/engine/report.js +241 -40
  42. package/dist/engine/request.js +317 -23
  43. package/dist/engine/sarif.js +120 -0
  44. package/dist/engine/settle.js +67 -0
  45. package/dist/engine/signed-in.js +256 -0
  46. package/dist/engine/status-pane-page.js +441 -0
  47. package/dist/engine/status-pane.js +128 -0
  48. package/dist/engine/tickets.js +671 -0
  49. package/dist/engine/unload.js +3 -2
  50. package/dist/export-run.js +633 -0
  51. package/dist/first-run.js +5 -0
  52. package/dist/installer.js +378 -8
  53. package/dist/intake.js +104 -0
  54. package/dist/login-run.js +250 -36
  55. package/dist/mcp-server.js +660 -65
  56. package/dist/playbook.js +5 -0
  57. package/dist/prompts.js +106 -0
  58. package/package.json +8 -5
  59. package/skills/scenescout/SKILL.md +49 -16
@@ -29,7 +29,10 @@
29
29
  * a loopback-only live view shows what each one is looking at
30
30
  * (`scenescout watch <project>`, engine/live.ts, ADR 7).
31
31
  */
32
+ import { AsyncLocalStorage } from "node:async_hooks";
33
+ import { spawn } from "node:child_process";
32
34
  import fs from "node:fs";
35
+ import os from "node:os";
33
36
  import path from "node:path";
34
37
  import { fileURLToPath } from "node:url";
35
38
  import { McpServer } from "@modelcontextprotocol/sdk/server/mcp.js";
@@ -37,8 +40,11 @@ import { StdioServerTransport } from "@modelcontextprotocol/sdk/server/stdio.js"
37
40
  import { ErrorCode, GetPromptRequestSchema, ListPromptsRequestSchema, McpError } from "@modelcontextprotocol/sdk/types.js";
38
41
  import { z } from "zod";
39
42
  import { BrowserEngine } from "./engine/browser.js";
43
+ import { readyBrowser } from "./engine/launch.js";
44
+ import { defaultEngine } from "./browsers.js";
45
+ import { downloadBrowsers, presentBrowsers } from "./installer.js";
40
46
  import { reapOrphanBrowsers } from "./engine/reaper.js";
41
- import { FINDING_CATEGORIES, isWorthALook, MemoryStore, mergeableCategories, redactSecrets } from "./engine/memory.js";
47
+ import { describeMerge, FINDING_CATEGORIES, isWorthALook, MemoryStore, mergeableCategories, redactSecrets } from "./engine/memory.js";
42
48
  import { DEDUP_ENV, DEDUP_MODES, redactKeys, secretValues } from "./engine/ci.js";
43
49
  import { DedupJudge, planDedup, samplingAsk } from "./engine/dedup.js";
44
50
  import { httpJudgeAsk } from "./ci-run.js";
@@ -47,19 +53,29 @@ import { MAX_UNFILED_NAMED, unfiledDefects } from "./engine/calibration.js";
47
53
  import { SessionQueue, withWatchdog } from "./engine/dispatch.js";
48
54
  import { FIXTURE_KINDS } from "./engine/fixtures.js";
49
55
  import { feedForSession, LIVE_ENV, writeStatusFile, LIVE_TOKEN_FILE, liveEngines, liveTokenFileName, pidAlive, statusFileName, LiveServer, StatusBoard, } from "./engine/live.js";
56
+ import { liveViewUrl, MCP_APP_MIME, paneData, paneText, STATUS_PANE_URI, STATUS_POLL_TOOL, STATUS_TOOL, } from "./engine/status-pane.js";
57
+ import { statusPanePage } from "./engine/status-pane-page.js";
50
58
  import { formatBriefs, MAX_LANES, planLanes } from "./engine/brief.js";
59
+ import { decideOpen, OPEN_CHOICES, OPEN_ENV, openChoiceFromEnv, openInBrowser } from "./engine/open.js";
51
60
  import { DEFAULT_EXPIRY_MARGIN_MINUTES, DEFAULT_RUN_MINUTES, judgeProfileFile } from "./engine/expiry.js";
52
- import { loginCommand } from "./engine/profiles.js";
53
- import { formatNeverSubmittedEmpty } from "./engine/forms.js";
54
- import { computeGaps, formatRouteCoverage, formatUnchosenOptions, generateReport, replayDocument, reportEvidence } from "./engine/report.js";
61
+ import { parseLoginArgs } from "./engine/profiles.js";
62
+ import { LOGIN_WINDOW_MAX_MS, savedLine, startLoginWindow } from "./login-run.js";
63
+ import { LoginWindows, WAIT_SAYS } from "./engine/signed-in.js";
64
+ import { computeGaps, coverageView, formatRouteCoverage, generateReport, replayDocument, reportEvidence } from "./engine/report.js";
65
+ import { DEFAULT_REPORT_AUDIENCE, REPORT_AUDIENCES } from "./engine/plain.js";
55
66
  import { describeVerdict, formatWorklist, unknownIds, VERDICTS, verifyWorklist } from "./engine/verify.js";
67
+ import { CRITERION_VERDICTS, findCriterion, formatCriteriaForLanes, formatReading, isTicketFileName, judgeCriterion, MAX_CRITERION_FINDINGS, MAX_REASON, MAX_TICKET_FILE_BYTES, MAX_TICKET_TEXT, MAX_TICKETS, NOT_TESTED_REASONS, parseTickets, TICKET_FILE_EXTENSIONS, } from "./engine/tickets.js";
56
68
  import { ACTION_TIMEOUT_ENV, DEFAULT_ACTION_TIMEOUT_MS, DEFAULT_CRAWL_NAV_TIMEOUT_MS, DEFAULT_NAV_TIMEOUT_MS, LIMIT_BOUNDS, NAV_TIMEOUT_ENV, watchdogFor, } from "./engine/limits.js";
69
+ import { MAX_READ_POSTS, READ_POSTS_ENV } from "./engine/policy.js";
70
+ import { chooseProjectFolder, PROJECTS_DIR_ENV, workspaceFromRoots } from "./engine/project-folder.js";
57
71
  import { RECORD_MAX_FRAMES, resolveFrame } from "./engine/replay.js";
58
72
  import { describePace, normalizePace } from "./engine/settle.js";
59
73
  import { needsTask, taskRefusal, TASK_MAX } from "./engine/task.js";
60
74
  import { EXPLORE_PROMPT_ARGUMENTS, explorePrompt, loadPlaybook, PLAYBOOK_PROMPT, PLAYBOOK_TOOL, SERVER_INSTRUCTIONS } from "./playbook.js";
75
+ import { livePrompt, loginPrompt, LOGIN_PROMPT, LIVE_PROMPT, LOGIN_PROMPT_ARGUMENTS, LIVE_PROMPT_ARGUMENTS } from "./prompts.js";
61
76
  import { formatScan, scanProject } from "./scan.js";
62
- import { CAPTURE_MARGIN, CAPTURES_DIRNAME, captureFileName, captureResultText, MAX_CAPTURE_MARGIN } from "./engine/capture.js";
77
+ import { CAPTURE_MARGIN, CAPTURES_DIRNAME, captureFileName, captureResultText, describePicture, EVIDENCE_ENV, EVIDENCE_LIMITS, EVIDENCE_MARGIN, EVIDENCE_MODES, evidenceFrame, evidenceSettings, findingPicturePath, MAX_CAPTURE_MARGIN, RECORD_ENV, recordChoice, returnsInline, } from "./engine/capture.js";
78
+ import { decodePng, fitPicture } from "./engine/png.js";
63
79
  /** Live sessions: each name owns an independent BrowserEngine (browser + auth). */
64
80
  const engines = new Map();
65
81
  /**
@@ -254,8 +270,10 @@ function reportExtras(eng) {
254
270
  designAudits: eng.memory?.auditsThisRun ?? eng.designAuditCount,
255
271
  createdResources: eng.createdResources,
256
272
  unvisitedRoutes: unvisited,
273
+ knownRoutes: all,
257
274
  mode: eng.mode,
258
275
  trustedEmbeds: [...eng.trustedEmbeds],
276
+ readPosts: eng.readPosts.map((e) => e.entry),
259
277
  policyAttributed: eng.oracleLog.policyAttributed,
260
278
  version: PKG_VERSION,
261
279
  attachedSessions: [...engines.keys()],
@@ -331,9 +349,31 @@ function startLiveServer() {
331
349
  function liveLine() {
332
350
  if (!liveAddress)
333
351
  return liveError ? `\nLive view unavailable: ${liveError}` : "";
334
- return (`\nLive view: http://127.0.0.1:${liveAddress.port}/${liveAddress.token}/ — give this address to the user so they can watch every session ` +
352
+ return (`\nLive view: ${liveViewUrl(liveAddress.port, liveAddress.token)} — give this address to the user so they can watch every session ` +
335
353
  `(current tool, page thumbnail, optional live stream). It opens on this machine only and cannot act on the run.`);
336
354
  }
355
+ /** What each session's attach decided to open (engine/open.ts). scout_report reads it for the report. */
356
+ const openDecisions = new Map();
357
+ /** The live view is one address for every session, so it is opened once per server. */
358
+ let liveOpened = false;
359
+ /** Whether a result has named the setting that turns opening off; named once. */
360
+ let openSettingNamed = false;
361
+ /** Whether an attach has said why the live view was not opened; said once. */
362
+ let openWhySaid = false;
363
+ /** Open a target in the default browser and say so in a line for the tool result; a failure is said, never thrown. */
364
+ function openForUser(what, target) {
365
+ const outcome = openInBrowser(target, {
366
+ platform: process.platform,
367
+ spawn: (command, args, options) => spawn(command, args, options),
368
+ onError: (why) => logLine(`could not open ${what}: ${why}`),
369
+ });
370
+ if (!outcome.ok)
371
+ return `\nCould not open ${what}: ${outcome.why}.`;
372
+ if (openSettingNamed)
373
+ return `\nOpened ${what} in the default browser.`;
374
+ openSettingNamed = true;
375
+ return `\nOpened ${what} in the default browser (${OPEN_ENV}=none, or scout_attach {open: "none"}, turns this off).`;
376
+ }
337
377
  function flushStatus(dir) {
338
378
  // Not awaited on a tool call's hot path: status is best-effort observability
339
379
  // and must never add blocking filesystem latency there. The writer queues
@@ -451,11 +491,17 @@ function serializedPerSession(label, fn, timeoutMs = 60_000) {
451
491
  return sessionQueue.run(session, exec);
452
492
  };
453
493
  }
494
+ const requestExtra = new AsyncLocalStorage();
495
+ function asRequestExtra(value) {
496
+ return value && typeof value.sendNotification === "function" ? value : undefined;
497
+ }
454
498
  /** Control-plane tools (scout_scan/scout_session/scout_close-all) don't target one browser — their own tiny chain keeps them off session queues without racing each other. */
455
499
  let controlChain = Promise.resolve();
456
500
  function serializedControl(fn) {
457
501
  return (...args) => {
458
- const run = controlChain.then(() => fn(...args), () => fn(...args));
502
+ // The SDK passes the request's extra (progress token, notifications) after the arguments; it is kept for the call, as requestExtra.
503
+ const call = () => requestExtra.run(asRequestExtra(args[1]), () => fn(...args));
504
+ const run = controlChain.then(call, call);
459
505
  controlChain = run.catch(() => { });
460
506
  return run;
461
507
  };
@@ -497,10 +543,11 @@ server.registerTool(PLAYBOOK_TOOL, {
497
543
  return errorText(err);
498
544
  }
499
545
  });
500
- // The same method as a prompt, for clients that list server prompts as commands.
501
- // Registered on the protocol server directly: the SDK's prompt helper rejects a
502
- // request that carries no `arguments` object, which is exactly what a client
503
- // sends when the person typed none, and every argument here is optional.
546
+ // Prompts, for clients that list server prompts as commands. Registered on the
547
+ // protocol server directly: the SDK's prompt helper rejects a request that
548
+ // carries no `arguments` object, which is exactly what a client sends when the
549
+ // person typed none. `explore` and `live` accept that; `login` then says the
550
+ // role is missing. None of them takes a password.
504
551
  server.server.registerCapabilities({ prompts: {} });
505
552
  server.server.setRequestHandler(ListPromptsRequestSchema, () => ({
506
553
  prompts: [
@@ -510,16 +557,36 @@ server.server.setRequestHandler(ListPromptsRequestSchema, () => ({
510
557
  description: "Start an exploratory test session: loads the SceneScout method and states the target.",
511
558
  arguments: EXPLORE_PROMPT_ARGUMENTS,
512
559
  },
560
+ {
561
+ name: LIVE_PROMPT,
562
+ title: "Watch the live view",
563
+ description: "Return the loopback live-view URL for the current session. Takes no arguments and no password.",
564
+ arguments: LIVE_PROMPT_ARGUMENTS,
565
+ },
566
+ {
567
+ name: LOGIN_PROMPT,
568
+ title: "Sign in as a role",
569
+ description: "Call scout_login for a role and wait the way that tool waits. The person signs in in the window it opens. Takes the role, and the app URL when you have it. Never a password.",
570
+ arguments: LOGIN_PROMPT_ARGUMENTS,
571
+ },
513
572
  ],
514
573
  }));
515
574
  server.server.setRequestHandler(GetPromptRequestSchema, (request) => {
516
- if (request.params.name !== PLAYBOOK_PROMPT)
517
- throw new McpError(ErrorCode.InvalidParams, `Unknown prompt: ${request.params.name}`);
575
+ const name = request.params.name;
518
576
  let message;
519
577
  try {
520
- message = explorePrompt(loadPlaybook(PACKAGE_ROOT), request.params.arguments);
578
+ if (name === PLAYBOOK_PROMPT)
579
+ message = explorePrompt(loadPlaybook(PACKAGE_ROOT), request.params.arguments);
580
+ else if (name === LIVE_PROMPT)
581
+ message = livePrompt(request.params.arguments);
582
+ else if (name === LOGIN_PROMPT)
583
+ message = loginPrompt(request.params.arguments);
584
+ else
585
+ throw new McpError(ErrorCode.InvalidParams, `Unknown prompt: ${name}`);
521
586
  }
522
587
  catch (err) {
588
+ if (err instanceof McpError)
589
+ throw err;
523
590
  throw new McpError(ErrorCode.InvalidParams, err instanceof Error ? err.message : String(err));
524
591
  }
525
592
  return { messages: [{ role: "user", content: { type: "text", text: message } }] };
@@ -568,7 +635,7 @@ server.registerTool("scout_lane_brief", {
568
635
  runMs: (runMinutes ?? DEFAULT_RUN_MINUTES) * 60_000,
569
636
  marginMs: (expiryMarginMinutes ?? DEFAULT_EXPIRY_MARGIN_MINUTES) * 60_000,
570
637
  role: eng.auth.role,
571
- rerun: loginCommand(eng.auth.role, eng.baseUrl),
638
+ rerun: eng.reloginCommand(eng.auth.role),
572
639
  });
573
640
  if (verdict.kind === "refuse")
574
641
  return errorText(new Error(verdict.message));
@@ -578,7 +645,11 @@ server.registerTool("scout_lane_brief", {
578
645
  const all = routes && routes.length > 0 ? routes : eng.allKnownRoutes();
579
646
  const briefs = planLanes(all, lanes, { goal, mode: eng.mode, role: eng.role });
580
647
  laneLedger.nameBriefed(briefs.map((b) => b.lane), (s) => engines.has(s));
581
- return text(expiryNote + formatBriefs(briefs, { goal, mode: eng.mode, role: eng.role, roleProfile: eng.auth.kind === "role" }), session);
648
+ // A run given tickets tells every lane which criteria it answers.
649
+ const criteria = eng.memory ? formatCriteriaForLanes(eng.memory.ticketsThisRun().tickets) : "";
650
+ return text(expiryNote +
651
+ formatBriefs(briefs, { goal, mode: eng.mode, role: eng.role, roleProfile: eng.auth.kind === "role" }) +
652
+ (criteria ? `\n${criteria}` : ""), session);
582
653
  }
583
654
  catch (err) {
584
655
  return errorText(err);
@@ -645,7 +716,7 @@ server.registerTool("scout_lane_report", {
645
716
  .map((u) => ` · ${u}`)
646
717
  .join("\n") +
647
718
  (unfiled.length > MAX_UNFILED_NAMED ? `\n … +${unfiled.length - MAX_UNFILED_NAMED} more` : "") +
648
- `\nFile each with scout_finding (the same evidence), or confirm which finding already covers it, before closing the lane's session. A judged defect that is never filed is not in the report.`
719
+ `\nFile each with scout_finding (the same evidence), or have the lane name the finding's id in the decision's "finding", before closing the lane's session. A judged defect that is never filed is not in the report.`
649
720
  : "";
650
721
  laneLedger.fold(lane, engines.get(lane)?.attached === true);
651
722
  // A lane that lost its sign-in and re-attached from its role's profile
@@ -681,10 +752,14 @@ server.registerTool("scout_scan", {
681
752
  }
682
753
  }));
683
754
  server.registerTool("scout_attach", {
684
- description: "Launch a browser and attach to a running web app. First attach in this conversation and you have read neither the SceneScout skill nor scout_playbook? Call scout_playbook before this. Write policy is enforced at the NETWORK layer: mode='observe' blocks EVERY request that is not a GET (login and token refresh excepted) — choose it for a target that holds real data, where even an ordinary form submission would create a record; mode='read-only' (default) blocks destructive-labeled elements AND all PUT/PATCH/DELETE + destructive POSTs, but lets ordinary form POSTs through; mode='safe-write' allows creating data and permits updates/deletes ONLY on resources this session created (use when the user wants create/edit flows tested); mode='destructive' allows everything — ONLY when the user explicitly confirmed a disposable/seeded environment. Pass `role` to sign in with a login the user saved by `scenescout login <url> --role <name>`, or a Playwright storage-state JSON as storageStatePath. Pass `session` to keep MULTIPLE roles alive at once (one browser each, genuinely concurrent) for collaboration testing — target each directly with every tool's `session` param, or use scout_session to set which one is the default; coverage and findings merge into one project memory.",
755
+ description: "Launch a browser and attach to a running web app. First attach in this conversation and you have read neither the SceneScout skill nor scout_playbook? Call scout_playbook before this. Write policy is enforced at the NETWORK layer: mode='observe' blocks EVERY request that is not a GET (login and token refresh excepted, and POSTs the user named in readPosts) — choose it for a target that holds real data, where even an ordinary form submission would create a record; mode='read-only' (default) blocks destructive-labeled elements AND all PUT/PATCH/DELETE + destructive POSTs, but lets ordinary form POSTs through; mode='safe-write' allows creating data and permits updates/deletes ONLY on resources this session created (use when the user wants create/edit flows tested); mode='destructive' allows everything — ONLY when the user explicitly confirmed a disposable/seeded environment. Pass `role` to sign in with a login the user saved by `scenescout login <url> --role <name>`, or a Playwright storage-state JSON as storageStatePath. Pass `session` to keep MULTIPLE roles alive at once (one browser each, genuinely concurrent) for collaboration testing — target each directly with every tool's `session` param, or use scout_session to set which one is the default; coverage and findings merge into one project memory.",
685
756
  inputSchema: {
686
757
  url: z.string().describe("Base URL of the running app, e.g. http://localhost:3000"),
687
- projectPath: z.string().describe("Absolute path to the project (memory + report live in .scenescout/ here)"),
758
+ projectPath: z
759
+ .string()
760
+ .optional()
761
+ .describe("Absolute path to the project (memory + report live in .scenescout/ here). Pass it whenever you have a project or working folder. " +
762
+ `Omitted: the client's workspace folder, else a folder per tested site under the user's documents folder (Documents/SceneScout/<host>/, or ${PROJECTS_DIR_ENV}), which the result names — tell the user where it is.`),
688
763
  storageStatePath: z.string().optional().describe("Optional Playwright storage-state JSON path for authenticated exploration. Not with `role`."),
689
764
  role: z
690
765
  .string()
@@ -701,7 +776,7 @@ server.registerTool("scout_attach", {
701
776
  browser: z
702
777
  .enum(["chromium", "firefox", "webkit"])
703
778
  .optional()
704
- .describe("Browser to drive. Default: the SCENESCOUT_BROWSER environment variable, else chromium. firefox and webkit must be downloaded first (scenescout install --browser-only --browsers firefox). Use them for a cross-browser pass; stay on chromium otherwise."),
779
+ .describe("Browser to drive. Default: the SCENESCOUT_BROWSER environment variable, else chromium. A build that is not on disk is downloaded on this attach, once, except in CI or with SCENESCOUT_BROWSER_DOWNLOAD=off, where the attach names the command to run. Use firefox or webkit for a cross-browser pass; stay on chromium otherwise."),
705
780
  viewportWidth: z.number().int().min(320).max(3840).optional().describe("Viewport width (default 1280); use e.g. 390 for a mobile pass"),
706
781
  viewportHeight: z.number().int().min(480).max(2400).optional().describe("Viewport height (default 900)"),
707
782
  objective: z
@@ -729,11 +804,22 @@ server.registerTool("scout_attach", {
729
804
  .optional()
730
805
  .describe('Origins of embedded frames (e.g. "https://pay.example.com") whose writes out of the app may go out — ONLY when the user named them, typically a provider in test mode, and only in safe-write mode. ' +
731
806
  "Never add one yourself. Hostile input, repeated-click probes and uploads stay refused in them."),
807
+ readPosts: z
808
+ .array(z.string().max(300))
809
+ .max(MAX_READ_POSTS)
810
+ .optional()
811
+ .describe(`POST endpoints that only read, e.g. ["POST /api/search", "POST https://api.example.com/reports/query"], which observe mode then lets out — ONLY when the user named them. Never add one yourself, even when the gap ledger lists a refused POST: ask the user. ` +
812
+ `Exact paths; * stands for one path segment. Still refused when the path or body looks destructive or the body is a GraphQL mutation. Observe mode only. Default: the ${READ_POSTS_ENV} environment variable, else none.`),
732
813
  record: z
733
814
  .boolean()
734
- .default(false)
815
+ .optional()
735
816
  .describe("Keep a frame of the page after every action, under .scenescout/recordings/, and show it beside that step in report.html. " +
736
- "Off by default: a recording is pictures of the app under test sitting in the project folder. Turn it on for QA work, where the run is evidence and not only a report."),
817
+ `Default: the ${RECORD_ENV} environment variable (on or off), else off: a recording is pictures of the app under test sitting in the project folder. Turn it on for QA work, where the run is evidence and not only a report.`),
818
+ evidence: z
819
+ .enum(EVIDENCE_MODES)
820
+ .optional()
821
+ .describe("What happens to the picture each scout_finding takes of what it is about. 'inline': kept under .scenescout/recordings/, shown in report.html, and returned in the scout_finding result so the conversation shows it. 'file': kept and shown in the report only. 'off': none taken. " +
822
+ `Default: the ${EVIDENCE_ENV} environment variable, else 'inline', or 'file' in a CI job. Pictures are bounded in size and in how many one session returns (see the configuration reference).`),
737
823
  actionTimeoutMs: z
738
824
  .number()
739
825
  .int()
@@ -760,9 +846,20 @@ server.registerTool("scout_attach", {
760
846
  "'judge': the rule first, then, for a filing the rule keeps apart from everything recorded, a model is asked whether it is one of the open findings on its page, and merges it when it says so. " +
761
847
  "It needs ANTHROPIC_API_KEY or OPENAI_API_KEY in the server's environment, and sends each pair's titles, categories and evidence, and the page's path, to that provider — ONLY when the user asked for it. " +
762
848
  "Applies to every session of the project until the run ends."),
849
+ open: z
850
+ .enum(OPEN_CHOICES)
851
+ .optional()
852
+ .describe(`Open the live view (on this attach) and/or report.html (when scout_report writes it) in the user's default browser: 'live', 'report', 'both' or 'none'. ` +
853
+ `Default: the ${OPEN_ENV} environment variable, else 'both' on a local desktop session (headed or headless) and 'none' in CI, over SSH, or with no display. ` +
854
+ "Pass it only when the user asked for something other than the default."),
763
855
  },
764
- }, serializedControl(async ({ url, projectPath, storageStatePath, role, mode, headed, browser, viewportWidth, viewportHeight, objective, task, record, trustedEmbeds, paceMs, actionTimeoutMs, navTimeoutMs, session, dedup, }) => {
856
+ }, serializedControl(async ({ url, projectPath, storageStatePath, role, mode, headed, browser, viewportWidth, viewportHeight, objective, task, record, evidence, trustedEmbeds, readPosts, paceMs, actionTimeoutMs, navTimeoutMs, session, dedup, open, }) => {
765
857
  try {
858
+ // Settled first: a refusal leaves the default session as it was.
859
+ const folder = await projectFolderFor(url, projectPath);
860
+ if ("refused" in folder)
861
+ return errorText(new Error(folder.refused));
862
+ projectPath = folder.dir;
766
863
  const target = session ?? activeName;
767
864
  if (session) {
768
865
  activeName = session;
@@ -824,15 +921,25 @@ server.registerTool("scout_attach", {
824
921
  // Moving this session to another project leaves its old one; if it was
825
922
  // the last session there, that run is over. Re-attaching to the SAME
826
923
  // project is the same run, and keeps what the run has learned.
924
+ // Before anything changes: a SCENESCOUT_OPEN the server cannot use refuses the attach.
925
+ const opening = decideOpen(open ?? openChoiceFromEnv(process.env), {
926
+ env: process.env,
927
+ platform: process.platform,
928
+ });
827
929
  const previous = eng.memory;
828
930
  if (previous && previous !== store && ![...engines.values()].some((e) => e !== eng && e.memory === previous))
829
931
  previous.endRun();
830
932
  // Before the browser starts: a value it cannot use refuses the attach rather than being replaced.
831
933
  const dedupNote = configureDedup(store, dedup);
934
+ const recording = recordChoice(record, process.env);
935
+ const pictures = evidenceSettings(evidence, process.env);
832
936
  const viewport = viewportWidth && viewportHeight ? { width: viewportWidth, height: viewportHeight } : undefined;
937
+ // The first attach on a machine without the browser downloads it here, once, rather than failing with a command to run.
938
+ const browserNote = await readyBrowserFor({ engine: browser ?? defaultEngine(process.env), headed: headed ?? false }, requestExtra.getStore());
833
939
  const out = await eng.attach({
834
940
  url,
835
941
  projectDir: projectPath,
942
+ projectChosen: folder.source === "default",
836
943
  storageStatePath,
837
944
  role,
838
945
  mode,
@@ -846,8 +953,9 @@ server.registerTool("scout_attach", {
846
953
  objective: objective ?? task,
847
954
  paceMs,
848
955
  task: objective ? task : undefined,
849
- record,
956
+ record: recording,
850
957
  trustedEmbeds,
958
+ readPosts,
851
959
  actionTimeoutMs,
852
960
  navTimeoutMs,
853
961
  memoryStore: store,
@@ -859,18 +967,109 @@ server.registerTool("scout_attach", {
859
967
  await ensureLive(eng.memory.dir);
860
968
  writeStatus(target, "idle", "scout_attach");
861
969
  }
970
+ // A new attach is a new count of pictures returned.
971
+ evidenceFor.set(eng, { ...pictures, shown: 0 });
972
+ openDecisions.set(target, opening);
973
+ // The address holds the token, and goes only to this machine's opener: the live view's loopback and token rules are unchanged.
974
+ let openNote = "";
975
+ if (opening.live && liveAddress && !liveOpened) {
976
+ liveOpened = true;
977
+ openNote = openForUser("the live view", liveViewUrl(liveAddress.port, liveAddress.token));
978
+ }
979
+ else if (!opening.live && liveAddress && !openWhySaid) {
980
+ // Once per server, so a person wondering why no window appeared is told how to change it.
981
+ openWhySaid = true;
982
+ openNote = `\nNot opened in a browser (${opening.why}); scout_attach {open} or ${OPEN_ENV} changes that.`;
983
+ }
862
984
  // Recording writes pictures of the app under test into the project, so
863
985
  // a run doing it says where they go rather than leaving the person to
864
986
  // find a folder of screenshots later.
865
- const recordNote = record && eng.memory?.dir
987
+ const recordNote = recording && eng.memory?.dir
866
988
  ? `\n\n📸 RECORDING: a frame of the page after each action, under ${path.join(eng.memory.dir, "recordings", target)}/ (at most ${RECORD_MAX_FRAMES}). scout_report writes them into report.html beside report.md.`
867
989
  : "";
868
- return text(out + conflictNote + recordNote + dedupNote + describePace(eng.pace) + (engines.size > 1 ? `\n${sessionLines()}` : "") + liveLine(), target);
990
+ const picturesNote = pictures.mode !== "off" && eng.memory?.dir
991
+ ? `\n📷 FINDING PICTURES: ${pictures.mode} (${pictures.source}): each scout_finding keeps a picture of what it names under ${path.join(eng.memory.dir, "recordings")}/ for report.html${pictures.mode === "inline" ? `, and returns the first ${pictures.inlineMax} in its result` : ""}.`
992
+ : "";
993
+ return text(browserNote +
994
+ out +
995
+ (folder.note ? `\n\n${folder.note}` : "") +
996
+ conflictNote +
997
+ recordNote +
998
+ picturesNote +
999
+ dedupNote +
1000
+ describePace(eng.pace) +
1001
+ (engines.size > 1 ? `\n${sessionLines()}` : "") +
1002
+ liveLine() +
1003
+ openNote, target);
869
1004
  }
870
1005
  catch (err) {
871
1006
  return errorText(err);
872
1007
  }
873
1008
  }));
1009
+ /**
1010
+ * readyBrowser with the real download, returning the line that opens the
1011
+ * attach's answer when it downloaded (empty when it did not). Each line goes
1012
+ * to stderr and, when the client asked for progress, as a progress
1013
+ * notification; the last one is repeated while the download runs, so a client
1014
+ * that extends its timeout on progress keeps waiting.
1015
+ */
1016
+ async function readyBrowserFor(need, extra) {
1017
+ const token = extra?._meta?.progressToken;
1018
+ let progress = 0;
1019
+ let last = "";
1020
+ const notify = (message) => {
1021
+ if (token === undefined || !extra)
1022
+ return;
1023
+ extra.sendNotification({ method: "notifications/progress", params: { progressToken: token, progress: ++progress, message } }).catch((err) => {
1024
+ logLine(`progress notification failed: ${err instanceof Error ? err.message : String(err)}`);
1025
+ });
1026
+ };
1027
+ const say = (line) => {
1028
+ last = line;
1029
+ logLine(line);
1030
+ notify(line);
1031
+ };
1032
+ const note = await readyBrowser(need, {
1033
+ env: process.env,
1034
+ present: presentBrowsers,
1035
+ say,
1036
+ download: async (targets) => {
1037
+ const heartbeat = setInterval(() => notify(last), 10_000);
1038
+ try {
1039
+ return await downloadBrowsers(targets, "stderr");
1040
+ }
1041
+ finally {
1042
+ clearInterval(heartbeat);
1043
+ }
1044
+ },
1045
+ });
1046
+ return note ? `${note}\n\n` : "";
1047
+ }
1048
+ /**
1049
+ * The folder an attach keeps its files in (engine/project-folder.ts): the
1050
+ * projectPath given, else the client's workspace folder, else a folder per
1051
+ * tested site. Only a client that offers roots is asked for them.
1052
+ */
1053
+ async function projectFolderFor(url, given) {
1054
+ let workspace = null;
1055
+ if (given === undefined && server.server.getClientCapabilities()?.roots) {
1056
+ try {
1057
+ workspace = workspaceFromRoots((await server.server.listRoots(undefined, { timeout: 3000 })).roots);
1058
+ }
1059
+ catch (err) {
1060
+ logLine(`the client offers a workspace but did not list it (${err.message}); using the default folder`);
1061
+ }
1062
+ }
1063
+ const userDirsFile = path.join(os.homedir(), ".config", "user-dirs.dirs");
1064
+ const userDirs = process.platform === "linux" && fs.existsSync(userDirsFile) ? fs.readFileSync(userDirsFile, "utf8") : undefined;
1065
+ return chooseProjectFolder({
1066
+ given,
1067
+ workspace,
1068
+ url,
1069
+ home: { platform: process.platform, homedir: os.homedir(), env: process.env, userDirs },
1070
+ exists: fs.existsSync,
1071
+ });
1072
+ }
874
1073
  /** Keys in this process's environment, and anything shaped like one, taken out of a line before it is shown. */
875
1074
  const withoutKeys = (text) => redactKeys(text, secretValues(process.env));
876
1075
  /** A line for the operator, on stderr. */
@@ -904,6 +1103,61 @@ function configureDedup(store, asked) {
904
1103
  store.dedupOff = undefined;
905
1104
  return plan.note;
906
1105
  }
1106
+ /** Each session's finding-picture settings from its last attach, and how many pictures its results have carried since. */
1107
+ const evidenceFor = new WeakMap();
1108
+ /**
1109
+ * Take, bound and keep a filed finding's picture (capture.ts decides whether
1110
+ * and of what; png.ts fits it). Returns the line the result adds and, when the
1111
+ * result carries it, the picture. A picture that cannot be taken or kept never
1112
+ * fails the filing, which is already recorded: the line says why there is none.
1113
+ */
1114
+ async function findingPicture(eng, filed, ref) {
1115
+ const settings = evidenceFor.get(eng);
1116
+ const dir = eng.memory?.dir;
1117
+ if (!settings || !dir)
1118
+ return { line: "" };
1119
+ const { finding, isNew } = filed;
1120
+ const plan = evidenceFrame({
1121
+ mode: settings.mode,
1122
+ ref,
1123
+ pageOpen: eng.alive,
1124
+ isNew,
1125
+ hasPicture: !!finding.picture,
1126
+ regressed: !isNew && !!finding.regressedAt && finding.regressedAt === finding.foundAt,
1127
+ });
1128
+ if (!plan.take)
1129
+ return { line: settings.mode === "off" ? "" : `\nNo picture taken: ${plan.why}.` };
1130
+ try {
1131
+ const shot = await eng.captureEvidence(plan.frame === "element" ? plan.ref : undefined, EVIDENCE_MARGIN);
1132
+ const fitted = fitPicture(decodePng(shot.png), settings.maxPx, settings.maxBytes);
1133
+ if (!fitted)
1134
+ return { line: `\nNo picture kept: none small enough to be readable fits in ${Math.round(settings.maxBytes / 1024)} KB.` };
1135
+ const rel = findingPicturePath(eng.sessionKey, finding.id);
1136
+ const file = path.join(dir, rel);
1137
+ fs.mkdirSync(path.dirname(file), { recursive: true });
1138
+ fs.writeFileSync(file, fitted.png);
1139
+ const picture = {
1140
+ width: fitted.width,
1141
+ height: fitted.height,
1142
+ frame: shot.frame,
1143
+ ...(shot.label ? { label: shot.label } : {}),
1144
+ at: new Date().toISOString(),
1145
+ };
1146
+ eng.memory?.setPicture(finding.id, rel, picture);
1147
+ const said = `\n📷 Picture (${describePicture(picture)}${fitted.shrunk ? ", shrunk to fit" : ""}): ${file}${shot.note ? ` — ${shot.note}` : ""}`;
1148
+ if (!returnsInline(settings.mode, settings.shown, settings.inlineMax)) {
1149
+ const capped = settings.mode === "inline"
1150
+ ? ` Not shown here: this session has returned its ${settings.inlineMax} (${EVIDENCE_LIMITS.inline.env} sets how many); it is in report.html.`
1151
+ : "";
1152
+ return { line: said + capped };
1153
+ }
1154
+ settings.shown += 1;
1155
+ return { line: said, image: { type: "image", data: fitted.png.toString("base64"), mimeType: "image/png" } };
1156
+ }
1157
+ catch (err) {
1158
+ return { line: `\nNo picture kept: ${err instanceof Error ? err.message : String(err)}` };
1159
+ }
1160
+ }
907
1161
  /** What scout_finding tells the agent about what filing did. */
908
1162
  function filedText(filed, category) {
909
1163
  const { finding, isNew, promoted } = filed;
@@ -918,7 +1172,7 @@ function filedText(filed, category) {
918
1172
  return `Merged into finding ${finding.id}, which was worth a look, and promoted to a defect: [${finding.severity}] ${finding.title}. It now counts among the report's findings.`;
919
1173
  if (finding.regressedAt)
920
1174
  return `⟳ REOPENED as a REGRESSION: finding ${finding.id} was previously resolved but the evidence reproduces again (seen in ${finding.runs} runs). Worth calling out to the user.`;
921
- return `Not recorded as new: merged into existing finding ${finding.id} — [${finding.severity}] ${finding.title}${finding.evidence ? ` (evidence: ${finding.evidence.slice(0, 160)})` : " (no evidence)"}, filed as ${finding.category}, seen in ${finding.runs} runs. If yours is a different bug, file it again: under the category that says what is wrong if it is another kind of defect (a finding filed as ${category} merges only with one filed as ${mergeableCategories(category).join(" or ")}), or with evidence naming the request that failed for you (method and path) — two findings are kept apart when both name requests and none is shared.`;
1175
+ return `Not recorded as new: merged into existing finding ${finding.id} — [${finding.severity}] ${finding.title}${finding.evidence ? ` (evidence: ${finding.evidence.slice(0, 160)})` : " (no evidence)"}, filed as ${finding.category}, seen in ${finding.runs} run${finding.runs === 1 ? "" : "s"}${filed.merge?.sameRun ? " (already filed this run, so the count did not change)" : ""}.${filed.merge ? describeMerge(filed.merge, finding.convention) : ""} If yours is a different bug, file it again: under the category that says what is wrong if it is another kind of defect (a finding filed as ${category} merges only with one filed as ${mergeableCategories(category).join(" or ")}), or with evidence naming the request that failed for you (method and path) — two findings are kept apart when both name requests and none is shared.`;
922
1176
  }
923
1177
  function sessionLines() {
924
1178
  const lines = ["Live sessions:"];
@@ -927,6 +1181,114 @@ function sessionLines() {
927
1181
  }
928
1182
  return lines.join("\n");
929
1183
  }
1184
+ // Signing in from the conversation: a window the person signs in in, watched
1185
+ // in the background, so a call need not last as long as the person takes and
1186
+ // a second call picks up the same window. One per project and role.
1187
+ const pendingLogins = new LoginWindows(LOGIN_WINDOW_MAX_MS);
1188
+ /** How long one scout_login call waits for the person, by default and at most, in seconds. */
1189
+ const LOGIN_WAIT_DEFAULT_S = 120;
1190
+ const LOGIN_WAIT_MAX_S = 600;
1191
+ /** How often a waiting call tells a client that asked for progress that it is still going, in ms. */
1192
+ const LOGIN_PROGRESS_EVERY_MS = 10_000;
1193
+ server.registerTool("scout_login", {
1194
+ description: "Open a visible browser window for the USER to sign in to the app as a role, and save that sign-in for scout_attach { role }. " +
1195
+ "Use it when an attach is refused because no sign-in is saved for the role, or the saved one has expired. " +
1196
+ 'Tell the user first, in plain words: "A browser window is opening. Sign in there as you normally would; it closes by itself once you are in." ' +
1197
+ "The window saves once the user is back on the app with a new session (a round trip through a single sign-on provider is followed, not taken for the end) and closes. " +
1198
+ "Never type credentials into it yourself. Returns once signed in and saved, or after waitSeconds with the window still open: then call scout_login again with the same role to keep waiting. " +
1199
+ "Closing the window saves nothing. Needs a desktop: on a machine with no display, ask the user to run `scenescout login <url> --role <name>` where they can see the window.",
1200
+ inputSchema: {
1201
+ url: z.string().describe("Where to sign in: the app's address or its sign-in page, e.g. http://localhost:3000/login"),
1202
+ role: z.string().max(40).describe("The name to save the sign-in under, e.g. admin; scout_attach { role } signs in with it"),
1203
+ projectPath: z
1204
+ .string()
1205
+ .optional()
1206
+ .describe("Absolute path to the project (the sign-in is saved in .scenescout/auth/ here), as for scout_attach. Omitted: the same folder an attach with no projectPath uses for this site, which the result names."),
1207
+ browser: z
1208
+ .enum(["chromium", "firefox", "webkit"])
1209
+ .optional()
1210
+ .describe("Browser to open. Default: the SCENESCOUT_BROWSER environment variable, else chromium"),
1211
+ successUrl: z
1212
+ .string()
1213
+ .max(500)
1214
+ .optional()
1215
+ .describe("Only when the user says how to tell: signed in once the URL's path contains this, or the URL starts with it (an absolute URL), instead of when a new session appears"),
1216
+ waitSeconds: z
1217
+ .number()
1218
+ .int()
1219
+ .min(1)
1220
+ .max(LOGIN_WAIT_MAX_S)
1221
+ .optional()
1222
+ .describe(`How long this call waits for the user before returning with the window still open (default ${LOGIN_WAIT_DEFAULT_S})`),
1223
+ },
1224
+ }, async (args, extra) => {
1225
+ try {
1226
+ // The folder an attach with no projectPath would use, so the attach after this finds the sign-in.
1227
+ const folder = await projectFolderFor(args.url, args.projectPath);
1228
+ if ("refused" in folder)
1229
+ return errorText(new Error(folder.refused));
1230
+ const projectDir = path.resolve(folder.dir);
1231
+ const where = folder.note ? `\n\n${folder.note}` : "";
1232
+ const parsed = parseLoginArgs([args.url, "--role", args.role, ...(args.browser ? ["--browser", args.browser] : []), ...(args.successUrl ? ["--success-url", args.successUrl] : [])], projectDir);
1233
+ if (!parsed.ok)
1234
+ return errorText(new Error(parsed.error));
1235
+ const options = { ...parsed.options, projectDir };
1236
+ const key = `${options.projectDir}\0${options.role}`;
1237
+ const { window: pending, resumed } = await pendingLogins.get(key, () => startLoginWindow(options));
1238
+ const waitMs = (args.waitSeconds ?? LOGIN_WAIT_DEFAULT_S) * 1000;
1239
+ const progressToken = extra._meta?.progressToken;
1240
+ let ticks = 0;
1241
+ const ticker = progressToken !== undefined
1242
+ ? setInterval(() => {
1243
+ ticks += 1;
1244
+ const p = pending.progress();
1245
+ void extra
1246
+ .sendNotification({
1247
+ method: "notifications/progress",
1248
+ params: {
1249
+ progressToken,
1250
+ progress: ticks,
1251
+ message: `Waiting for the sign-in as "${options.role}": ${p.reason === "starting" ? "the window is opening" : WAIT_SAYS[p.reason]}`,
1252
+ },
1253
+ })
1254
+ .catch((err) => console.error(`[scenescout] scout_login progress: ${err instanceof Error ? err.message : String(err)}`));
1255
+ }, LOGIN_PROGRESS_EVERY_MS)
1256
+ : undefined;
1257
+ let timer;
1258
+ let onAbort;
1259
+ const outcome = await Promise.race([
1260
+ pending.done,
1261
+ new Promise((resolve) => {
1262
+ timer = setTimeout(() => resolve("waiting"), waitMs);
1263
+ }),
1264
+ new Promise((resolve) => {
1265
+ onAbort = () => resolve("cancelled");
1266
+ extra.signal.addEventListener("abort", onAbort, { once: true });
1267
+ }),
1268
+ ]).finally(() => {
1269
+ clearTimeout(timer);
1270
+ if (ticker)
1271
+ clearInterval(ticker);
1272
+ if (onAbort)
1273
+ extra.signal.removeEventListener("abort", onAbort);
1274
+ });
1275
+ const differs = resumed && pending.url !== options.url ? ` (the window was opened at ${pending.url} by an earlier call; that one is still the one being watched)` : "";
1276
+ if (outcome === "waiting" || outcome === "cancelled") {
1277
+ const p = pending.progress();
1278
+ return text(`Still waiting for the user to sign in as "${options.role}"${differs}: ${p.reason === "starting" ? "the window is opening" : WAIT_SAYS[p.reason]}. ` +
1279
+ `The window stays open for up to ${LOGIN_WINDOW_MAX_MS / 60_000} minutes from when it opened. Call scout_login again with the same role to keep waiting, once the user says they are done or to check.` +
1280
+ where, activeName);
1281
+ }
1282
+ // This call reports the outcome; the next call for the role opens a new window.
1283
+ pendingLogins.reported(key, pending);
1284
+ if (!outcome.ok)
1285
+ return errorText(new Error(`nothing was saved for role "${options.role}": ${outcome.error}`));
1286
+ return text(`${outcome.detected}\n${savedLine(options, outcome.saved)}${where}`, activeName);
1287
+ }
1288
+ catch (err) {
1289
+ return errorText(err);
1290
+ }
1291
+ });
930
1292
  server.registerTool("scout_session", {
931
1293
  description: "List live sessions, or set which one is the DEFAULT (used by any tool call that omits `session`). Prefer passing `session` directly on each tool call for multi-role work — that's what lets concurrent dispatch happen; scout_session is for sequential convenience (skip repeating `session` on every call) and for checking what's live. Both browsers stay live and authenticated regardless of which is default — re-snapshot a session after a break to see what changed while it was away.",
932
1294
  inputSchema: {
@@ -976,7 +1338,7 @@ server.registerTool("scout_session", {
976
1338
  }
977
1339
  }));
978
1340
  server.registerTool("scout_snapshot", {
979
- description: "Capture the current page state: URL, state fingerprint, a one-line summary of the main area's heading and text, interactable elements with refs (e1, e2, …) and their state (pressed, selected, checked, expanded, current), what the page announces (alert and status regions, by their text), geometry issues, coverage, and oracle violations since the last action. Re-snapshotting the same route returns a DIFF (refs stay stable). Cheap — prefer this over screenshots.",
1341
+ description: "Capture the current page state: URL, state fingerprint, a one-line summary of the main area's heading and text, interactable elements with refs (e1, e2, …) and their state (pressed, selected, checked, expanded, current), what the page announces (alert and status regions, by their text), geometry issues, coverage, and oracle violations since the last action. Re-snapshotting a route returns a DIFF against its last snapshot, even after visiting elsewhere, and another tab of the same screen diffs against that screen's last tab; refs stay stable, including across a search or filter that rewrites only the query string. On a dense page, says what past the element cap was cut. Cheap — prefer this over screenshots.",
980
1342
  inputSchema: {
981
1343
  full: z.boolean().default(false).describe("Force a full element list instead of a diff"),
982
1344
  session: sessionParam,
@@ -990,9 +1352,13 @@ server.registerTool("scout_snapshot", {
990
1352
  }
991
1353
  }));
992
1354
  server.registerTool("scout_crawl", {
993
- description: "Engine-side route sweep in ONE call: visits each path (default: all known routes not yet visited), records states into coverage memory, and returns a per-route health summary (HTTP status, element count, oracle violations, dead-ends, auth-redirects). Navigation-only — safe in read-only mode. Use this FIRST for broad coverage; explore interactively only where it flags problems or where journeys matter.",
1355
+ description: "Engine-side route sweep in ONE call: visits each path (default: all known routes not yet visited), records states into coverage memory, and returns a per-route health summary (HTTP status, element count, what the main area holds, oracle violations, dead-ends, auth-redirects, and ERROR-VIEW or STILL-LOADING for a main area showing only an alert or a loading placeholder). Navigation-only — safe in read-only mode. Use this FIRST for broad coverage; explore interactively only where it flags problems or where journeys matter.",
994
1356
  inputSchema: {
995
- paths: z.array(z.string()).max(150).optional().describe("Paths to visit, e.g. ['/orders','/settings']. Omit to crawl all unvisited known routes."),
1357
+ paths: z
1358
+ .array(z.string())
1359
+ .max(150)
1360
+ .optional()
1361
+ .describe("Paths to visit, e.g. ['/orders','/settings'], or full URLs on the attached origin. A path resolves from the origin's root, whatever page the session attached on. Omit to crawl all unvisited known routes."),
996
1362
  session: sessionParam,
997
1363
  },
998
1364
  }, serializedPerSession("scout_crawl", async ({ paths }, session) => {
@@ -1004,7 +1370,7 @@ server.registerTool("scout_crawl", {
1004
1370
  }
1005
1371
  }, 600_000));
1006
1372
  server.registerTool("scout_run_plan", {
1007
- description: "Execute up to 20 actions in ONE call — use for mechanical sequences (fill a form, walk a wizard) so each step doesn't cost a round-trip. Targets resolve at execution time by semantic locator: 'testid=…', 'text=…', 'label=…' or 'role=button[name=Save]' (never snapshot refs). The same steps, saved to .scenescout/flows/<name>.json with expect-text / expect-url / expect-request steps added, are replayed by `scenescout check` on every pull request. An `upload` step attaches a file as scout_upload does (target required — the file input or the control that opens its chooser; value = a fixture kind or a project-relative path). The plan ABORTS at the first NEW oracle violation, policy refusal, or failed step, returning a transcript of how far it got; repeats of already-reported violations do not abort (they stay logged for the report).",
1373
+ description: "Execute up to 20 actions in ONE call — use for mechanical sequences (fill a form, walk a wizard) so each step doesn't cost a round-trip. Targets resolve at execution time by semantic locator: 'testid=…', 'text=…', 'label=…' or 'role=button[name=Save]' (never snapshot refs). The same steps, saved to .scenescout/flows/<name>.json with expect-text / expect-url / expect-request steps added, are replayed by `scenescout check` on every pull request. An `upload` step attaches a file as scout_upload does (target required — the file input or the control that opens its chooser; value = a fixture kind or a project-relative path). The plan ABORTS at the first NEW oracle violation, policy refusal, or failed step, returning a transcript of how far it got; repeats of already-reported violations do not abort (they stay logged for the report). For a sweep of independent steps (tabs, filters, pages) pass onViolation \"continue\": a new error status or its console echo is listed on its step's line and the plan goes on; a failed step, a policy refusal or any other violation still stops it. A select step whose value names no option fails at once, listing the options.",
1008
1374
  inputSchema: {
1009
1375
  steps: z
1010
1376
  .array(z.object({
@@ -1022,30 +1388,39 @@ server.registerTool("scout_run_plan", {
1022
1388
  }))
1023
1389
  .min(1)
1024
1390
  .max(20),
1391
+ onViolation: z
1392
+ .enum(["stop", "continue"])
1393
+ .default("stop")
1394
+ .describe("stop (default): end the plan at the first new oracle violation, right for a form flow whose steps depend on each other. continue: list a new http_error or console_error on its step's line and run the next step, for a sweep of independent steps"),
1025
1395
  task: taskParam,
1026
1396
  objective: legacyObjectiveParam,
1027
1397
  session: sessionParam,
1028
1398
  },
1029
- }, serializedPerSession("scout_run_plan", async ({ steps }, session) => {
1399
+ }, serializedPerSession("scout_run_plan", async ({ steps, onViolation }, session) => {
1030
1400
  try {
1031
- return text(await engineFor(session).runPlan(steps), session);
1401
+ return text(await engineFor(session).runPlan(steps, onViolation ?? "stop"), session);
1032
1402
  }
1033
1403
  catch (err) {
1034
1404
  return errorText(err);
1035
1405
  }
1036
1406
  }, 240_000));
1407
+ const leaveParam = z
1408
+ .boolean()
1409
+ .optional()
1410
+ .describe("How to answer if the page asks to confirm leaving (a beforeunload prompt over unsent input): true leaves and discards that input, false stays. Omitted, observe and read-only stay and other modes leave. The result says when the page asked.");
1037
1411
  server.registerTool("scout_click", {
1038
1412
  description: "Click an element by its ref from the latest scout_snapshot. Returns the outcome plus any oracle violations triggered. clicks=2 (or 3) probes IMPATIENT-USER behaviour: a rapid multi-click that fires the same state-changing request twice means the control is not guarded against double submission (button stays enabled, endpoint not idempotent) — use it on every important submit/create button once; the result says explicitly whether duplicates fired.",
1039
1413
  inputSchema: {
1040
1414
  ref: z.string().describe("Element ref, e.g. e12"),
1041
1415
  clicks: z.number().int().min(1).max(3).default(1).describe("1 = normal; 2-3 = rapid repeated clicks (double-submit probe)"),
1416
+ leave: leaveParam,
1042
1417
  task: taskParam,
1043
1418
  objective: legacyObjectiveParam,
1044
1419
  session: sessionParam,
1045
1420
  },
1046
- }, serializedPerSession("scout_click", async ({ ref, clicks }, session) => {
1421
+ }, serializedPerSession("scout_click", async ({ ref, clicks, leave }, session) => {
1047
1422
  try {
1048
- return text(await engineFor(session).click(ref, clicks ?? 1), session);
1423
+ return text(await engineFor(session).click(ref, clicks ?? 1, leave), session);
1049
1424
  }
1050
1425
  catch (err) {
1051
1426
  return errorText(err);
@@ -1121,10 +1496,10 @@ server.registerTool("scout_hover", {
1121
1496
  }
1122
1497
  }));
1123
1498
  server.registerTool("scout_select", {
1124
- description: "Select an option in a <select> by ref.",
1499
+ description: "Select an option in a <select> by ref. The value is matched against the options before anything is picked: an exact value, an exact label, either ignoring case, then a label it starts with. A value matching no option, or several, is refused at once with the options listed.",
1125
1500
  inputSchema: {
1126
1501
  ref: z.string(),
1127
- value: z.string().describe("Option value or label"),
1502
+ value: z.string().describe("Option value or label (or the start of a label, when only one option has it)"),
1128
1503
  task: taskParam,
1129
1504
  objective: legacyObjectiveParam,
1130
1505
  session: sessionParam,
@@ -1138,23 +1513,24 @@ server.registerTool("scout_select", {
1138
1513
  }
1139
1514
  }));
1140
1515
  server.registerTool("scout_navigate", {
1141
- description: "Navigate to a URL or a path relative to the attached base URL (e.g. '/orders'). Also supports 'back' via scout_back.",
1516
+ description: "Navigate to a path on the attached origin (e.g. '/orders') or a full URL on it. A path resolves from the origin's root, whatever page the session attached on. Also supports 'back' via scout_back.",
1142
1517
  inputSchema: {
1143
1518
  target: z.string().describe("Absolute URL or path like /settings"),
1519
+ leave: leaveParam,
1144
1520
  task: taskParam,
1145
1521
  objective: legacyObjectiveParam,
1146
1522
  session: sessionParam,
1147
1523
  },
1148
- }, serializedPerSession("scout_navigate", async ({ target }, session) => {
1524
+ }, serializedPerSession("scout_navigate", async ({ target, leave }, session) => {
1149
1525
  try {
1150
- return text(await engineFor(session).navigate(target), session);
1526
+ return text(await engineFor(session).navigate(target, leave), session);
1151
1527
  }
1152
1528
  catch (err) {
1153
1529
  return errorText(err);
1154
1530
  }
1155
1531
  }));
1156
1532
  server.registerTool("scout_request", {
1157
- description: "Call the app's own API as this session, with the UI bypassed — the check that turns a hidden or disabled control into a proven refusal. A button that is not shown proves nothing; the same action refused by the server does. The fetch runs IN the page, so it carries the session's cookies and replays the Authorization header the app itself last sent, and it passes through the same interception the write policy is enforced on: in safe-write a mutation on a record this session did not create is refused here exactly as it would be for a click, and that refusal is the engine's safety net, not a finding. Returns the status line, the timing, the headers that decide whether two responses are truly identical (content-type, location, www-authenticate, retry-after, cache-control), and the body. Unlike a shell call, every request is recorded in the run's trail and its signature is what a finding should quote. Paths are fenced to the attached origin: use another session to reach another host.",
1533
+ description: "Call the app's own API as this session, with the UI bypassed — the check that turns a hidden or disabled control into a proven refusal. A button that is not shown proves nothing; the same action refused by the server does. The fetch runs IN the page, so it carries the session's cookies and replays the Authorization header the app itself last sent, and it passes through the same interception the write policy is enforced on: in safe-write a mutation on a record this session did not create is refused here exactly as it would be for a click, and that refusal is the engine's safety net, not a finding. Returns the status line, the timing, the headers that decide whether two responses are truly identical (content-type, location, www-authenticate, retry-after, cache-control), and the body — its first 2000 characters, or the part named by select (one JSON value by path) or offset/limit (a window of characters). Unlike a shell call, every request is recorded in the run's trail and its signature is what a finding should quote. Paths are fenced to the attached origin: use another session to reach another host.",
1158
1534
  inputSchema: {
1159
1535
  path: z.string().min(1).max(2000).describe("Path on the attached origin, e.g. /api/things/12, or a full URL on that same origin"),
1160
1536
  method: z.enum(["GET", "HEAD", "POST", "PUT", "PATCH", "DELETE", "OPTIONS"]).optional().describe("Default GET"),
@@ -1163,13 +1539,48 @@ server.registerTool("scout_request", {
1163
1539
  .record(z.string().max(2000))
1164
1540
  .optional()
1165
1541
  .describe("Extra headers. One given here wins over the app's own, which is how a session tests a different or absent credential."),
1542
+ select: z
1543
+ .string()
1544
+ .max(500)
1545
+ .optional()
1546
+ .describe('Return one value of a JSON response body by its dotted path, e.g. "stats.open" or "items.0.name", pretty-printed and up to 8000 characters. A path that is not there says which keys are.'),
1547
+ offset: z
1548
+ .number()
1549
+ .int()
1550
+ .min(0)
1551
+ .max(1_000_000)
1552
+ .optional()
1553
+ .describe("Return the body (or the selected value) from this character on. The body is cut at 2000 characters by default; the result names the next offset."),
1554
+ limit: z
1555
+ .number()
1556
+ .int()
1557
+ .min(1)
1558
+ .max(8000)
1559
+ .optional()
1560
+ .describe("How many characters to return with offset or select. Default 2000, or 8000 for a select with no offset."),
1166
1561
  task: taskParam,
1167
1562
  objective: legacyObjectiveParam,
1168
1563
  session: sessionParam,
1169
1564
  },
1170
1565
  }, serializedPerSession("scout_request", async (args, session) => {
1171
1566
  try {
1172
- return text(await engineFor(session).apiRequest({ method: args.method, path: args.path, body: args.body, headers: args.headers }), session);
1567
+ const view = { select: args.select, offset: args.offset, limit: args.limit };
1568
+ return text(await engineFor(session).apiRequest({ method: args.method, path: args.path, body: args.body, headers: args.headers, view }), session);
1569
+ }
1570
+ catch (err) {
1571
+ return errorText(err);
1572
+ }
1573
+ }));
1574
+ server.registerTool("scout_network", {
1575
+ description: "List the data requests (fetch and XHR) the current page made since its document loaded: method, path, status and time, oldest first, each marked with the route it was sent from when a client-side route change moved the page since. Use it when the page and the server seem to disagree — an empty list where the API has data, a stale value after a save — to tell a request that failed, one still pending and one that never ran apart. Read-only: it lists what the browser already saw and sends nothing. Credentials in query strings are redacted. A full page load starts a new list; scout_request's own calls are marked.",
1576
+ inputSchema: {
1577
+ contains: z.string().max(200).optional().describe("Only requests whose path contains this text, e.g. /api/things"),
1578
+ limit: z.number().int().min(1).max(200).optional().describe("How many of the newest to list. Default 40."),
1579
+ session: sessionParam,
1580
+ },
1581
+ }, serializedPerSession("scout_network", async (args, session) => {
1582
+ try {
1583
+ return text(engineFor(session).listPageRequests({ contains: args.contains, limit: args.limit }), session);
1173
1584
  }
1174
1585
  catch (err) {
1175
1586
  return errorText(err);
@@ -1177,10 +1588,10 @@ server.registerTool("scout_request", {
1177
1588
  }));
1178
1589
  server.registerTool("scout_back", {
1179
1590
  description: "Go back in browser history (tests back-button resilience).",
1180
- inputSchema: { task: taskParam, objective: legacyObjectiveParam, session: sessionParam },
1181
- }, serializedPerSession("scout_back", async (_args, session) => {
1591
+ inputSchema: { leave: leaveParam, task: taskParam, objective: legacyObjectiveParam, session: sessionParam },
1592
+ }, serializedPerSession("scout_back", async ({ leave }, session) => {
1182
1593
  try {
1183
- return text(await engineFor(session).goBack(), session);
1594
+ return text(await engineFor(session).goBack(leave), session);
1184
1595
  }
1185
1596
  catch (err) {
1186
1597
  return errorText(err);
@@ -1322,7 +1733,7 @@ server.registerTool("scout_capture", {
1322
1733
  }
1323
1734
  }));
1324
1735
  server.registerTool("scout_finding", {
1325
- description: "Record a structured finding (bug, UX issue, or improvement). Deduplicates across runs; automatically captures the recent action trace as the repro. Use for anything worth reporting: crashes, oracle violations you confirmed, dead ends, confusing UX, permission leaks, missing testids — and design-audit improvement opportunities (ux-polish) with their concrete measurements.",
1736
+ description: "Record a structured finding (bug, UX issue, or improvement). Deduplicates across runs; automatically captures the recent action trace as the repro, and a picture of what it is about (the element `ref` names, else the viewport), kept for report.html and returned in this result as an image. Use for anything worth reporting: crashes, oracle violations you confirmed, dead ends, confusing UX, permission leaks, missing testids — and design-audit improvement opportunities (ux-polish) with their concrete measurements.",
1326
1737
  inputSchema: {
1327
1738
  severity: z.enum(["high", "medium", "low"]),
1328
1739
  category: z.enum(FINDING_CATEGORIES).describe("Pick the closest — use 'other' only when nothing fits"),
@@ -1338,9 +1749,14 @@ server.registerTool("scout_finding", {
1338
1749
  .max(LANE_CONVENTION_MAX)
1339
1750
  .optional()
1340
1751
  .describe("Only for a WORTH-A-LOOK finding: the observation is real, and it is a defect only under a convention of this project you cannot see. Name that convention, e.g. 'a 4px spacing scale' or 'test ids on every control'. The report lists it under \"Worth a look\", apart from the defects, and does not count it as one. Not for \"I could not tell\": leave that unfiled or look closer. Omit for a defect."),
1752
+ ref: z
1753
+ .string()
1754
+ .max(20)
1755
+ .optional()
1756
+ .describe("The ref, from the latest scout_snapshot, of the element the finding is about. Its picture (the element plus a margin) is kept with the finding and shown in report.html. Omit it and the picture is the viewport."),
1341
1757
  session: sessionParam,
1342
1758
  },
1343
- }, serializedPerSession("scout_finding", async ({ severity, category, title, detail, evidence, convention, }, session) => {
1759
+ }, serializedPerSession("scout_finding", async ({ severity, category, title, detail, evidence, convention, ref, }, session) => {
1344
1760
  try {
1345
1761
  const eng = engineFor(session);
1346
1762
  if (!eng.memory)
@@ -1358,34 +1774,101 @@ server.registerTool("scout_finding", {
1358
1774
  });
1359
1775
  if (filed.judgeError)
1360
1776
  logLine(`dedup judge: ${filed.judgeError}; the rule decided`);
1361
- return text(filedText(filed, category), session);
1777
+ const picture = await findingPicture(eng, filed, ref);
1778
+ const result = text(filedText(filed, category) + picture.line, session);
1779
+ if (picture.image)
1780
+ result.content.push(picture.image);
1781
+ return result;
1782
+ }
1783
+ catch (err) {
1784
+ return errorText(err);
1785
+ }
1786
+ }));
1787
+ // ---- The run-status pane: an MCP App (SEP-1865, `io.modelcontextprotocol/ui`, spec 2026-01-26). ----
1788
+ // scout_status links the pane through _meta.ui.resourceUri; the pane polls the
1789
+ // app-only tool. Neither goes through a session queue: like the live view, a
1790
+ // watcher must never wait behind the agent's calls, and both only read.
1791
+ function statusPaneData() {
1792
+ const eng = [lastWriter ? engines.get(lastWriter.session) : undefined, ...engines.values()].find((e) => !!e?.memory);
1793
+ const memory = eng?.memory ?? null;
1794
+ let coverage = null;
1795
+ if (eng && memory) {
1796
+ const all = eng.allKnownRoutes();
1797
+ const cov = memory.coverage();
1798
+ coverage = {
1799
+ routesVisited: all.length - eng.unvisitedKnownRoutes().length,
1800
+ routesTotal: all.length,
1801
+ states: cov.states,
1802
+ elementsExercised: cov.elementsExercised,
1803
+ elementsTotal: cov.elementsTotal,
1804
+ };
1805
+ }
1806
+ return paneData({
1807
+ nowMs: Date.now(),
1808
+ version: PKG_VERSION,
1809
+ live: liveAddress,
1810
+ liveError,
1811
+ liveOff: process.env[LIVE_ENV] === "off",
1812
+ sessions: board.list(),
1813
+ findings: memory ? memory.findings : null,
1814
+ runStart: memory?.sessionStart,
1815
+ coverage,
1816
+ });
1817
+ }
1818
+ function statusResult() {
1819
+ try {
1820
+ const data = statusPaneData();
1821
+ return { content: [{ type: "text", text: paneText(data) }], structuredContent: { ...data } };
1362
1822
  }
1363
1823
  catch (err) {
1364
1824
  return errorText(err);
1365
1825
  }
1826
+ }
1827
+ server.registerResource("scenescout-status", STATUS_PANE_URI, {
1828
+ title: "SceneScout run status",
1829
+ description: "The run's sessions, open findings by severity, coverage and the live view's address, refreshed every few seconds.",
1830
+ mimeType: MCP_APP_MIME,
1831
+ // No outside origin: the page is one self-contained document, so the host's restrictive default CSP applies.
1832
+ _meta: { ui: { csp: {}, prefersBorder: true } },
1833
+ }, () => ({
1834
+ contents: [{ uri: STATUS_PANE_URI, mimeType: MCP_APP_MIME, text: statusPanePage(PKG_VERSION), _meta: { ui: { csp: {}, prefersBorder: true } } }],
1366
1835
  }));
1836
+ server.registerTool(STATUS_TOOL, {
1837
+ title: "Run status",
1838
+ description: "Show the person running you how the run stands: each session's objective and current task, open findings by severity, coverage, and the live view's address. " +
1839
+ "Call it when the user wants to watch or asks how the run is going. A host that renders MCP Apps shows a pane that keeps itself up to date; every other host gets the same as text, and you pass the `Live view:` address on. Takes no input and touches no browser.",
1840
+ // The legacy flat key alongside _meta.ui.resourceUri, as the ext-apps SDK's registerAppTool sets both for hosts that read only the old one.
1841
+ _meta: { ui: { resourceUri: STATUS_PANE_URI }, "ui/resourceUri": STATUS_PANE_URI },
1842
+ }, async () => statusResult());
1843
+ server.registerTool(STATUS_POLL_TOOL, {
1844
+ title: "Run status (for the pane)",
1845
+ description: `Called by the ${STATUS_TOOL} pane every few seconds to refresh itself. Not for the agent: call ${STATUS_TOOL} instead.`,
1846
+ _meta: { ui: { visibility: ["app"] } },
1847
+ }, async () => statusResult());
1367
1848
  server.registerTool("scout_coverage", {
1368
- description: "Show exploration coverage: states visited across all runs, which elements remain unexercised, which options of a dropdown used this run no session has chosen yet, and which forms seen this run no session has submitted with every text field blank. Use to decide where to explore next and when the level's budget is satisfied.",
1369
- inputSchema: { session: sessionParam },
1370
- }, serializedPerSession("scout_coverage", async (_args, session) => {
1849
+ description: "Show exploration coverage: states visited, which elements remain unexercised, which options of a dropdown used this run no session has chosen yet, and which forms seen this run no session has submitted with every text field blank. Use to decide where to explore next and when the level's budget is satisfied. In a parallel run it shows this session's own work by default — the routes it reached this run and the forms it saw — so one lane is not handed another's gaps; scope:\"project\" shows every session's, each form and route tagged with the sessions that saw it.",
1850
+ inputSchema: {
1851
+ scope: z
1852
+ .enum(["session", "project"])
1853
+ .optional()
1854
+ .describe("'session': only the routes this session reached this run and the forms it saw. 'project': every route in the memory, across runs and sessions. Default: 'session' when other sessions share this project, else 'project'."),
1855
+ session: sessionParam,
1856
+ },
1857
+ }, serializedPerSession("scout_coverage", async (args, session) => {
1371
1858
  try {
1372
1859
  const eng = engineFor(session);
1373
- if (!eng.memory)
1860
+ const memory = eng.memory;
1861
+ if (!memory)
1374
1862
  throw new Error("Not attached.");
1375
- const cov = eng.memory.coverage();
1863
+ const shared = [...engines.values()].some((e) => e !== eng && e.memory === memory);
1376
1864
  const unvisited = eng.unvisitedKnownRoutes();
1377
1865
  const lines = [
1378
- ...(eng.memory.lastSaveError
1866
+ ...(memory.lastSaveError
1379
1867
  ? [
1380
- `⚠ MEMORY WRITE FAILING: ${eng.memory.lastSaveError} — coverage/findings since the last successful write are NOT persisted to disk. If this doesn't clear on its own, check the project directory still exists and is writable.`,
1868
+ `⚠ MEMORY WRITE FAILING: ${memory.lastSaveError} — coverage/findings since the last successful write are NOT persisted to disk. If this doesn't clear on its own, check the project directory still exists and is writable.`,
1381
1869
  ]
1382
1870
  : []),
1383
- `States known: ${cov.states} · Elements exercised: ${cov.elementsExercised}/${cov.elementsTotal}${cov.embeds.total > 0 ? ` (plus ${cov.embeds.exercised}/${cov.embeds.total} inside other sites' frames, not counted)` : ""}`,
1384
- formatRouteCoverage(eng.allKnownRoutes(), unvisited),
1385
- `Unexercised elements by route:`,
1386
- ...cov.unexercised.slice(0, 25).map((u) => ` ${u.state}: ${u.keys.slice(0, 6).join(", ")}${u.keys.length > 6 ? ` … +${u.keys.length - 6}` : ""}`),
1387
- ...formatUnchosenOptions(eng.memory.unchosenOptions()),
1388
- ...formatNeverSubmittedEmpty(eng.memory.formsNeverSubmittedEmpty()),
1871
+ ...coverageView(memory, eng.sessionKey, args.scope ?? (shared ? "session" : "project"), formatRouteCoverage(eng.allKnownRoutes(), unvisited)),
1389
1872
  ];
1390
1873
  return text(lines.join("\n"), session);
1391
1874
  }
@@ -1405,9 +1888,13 @@ server.registerTool("scout_report", {
1405
1888
  .enum(["index", "full"])
1406
1889
  .default("index")
1407
1890
  .describe("How much of the history to print. 'index' lists findings from earlier runs, and resolved ones, as a row each: id, severity, age, title. 'full' prints every one in full as before — on one project that was 1.75 MB against 113 KB, nearly half of it findings already fixed. Use 'full' when handing the document to someone who has no access to the memory."),
1891
+ report: z
1892
+ .enum(REPORT_AUDIENCES)
1893
+ .default(DEFAULT_REPORT_AUDIENCE)
1894
+ .describe("Which parts the report carries. 'both' (default): a plain-language section first — a short summary, then each problem with numbered steps, what was expected, what happened, its picture and its impact (blocks users, annoying, cosmetic), each with its technical detail folded beneath — followed by the technical report. 'qa': the plain section alone, for a tester or anyone not technical. 'dev': the technical report alone, as before the plain section existed. It applies to the files this call writes; the live view always shows both."),
1408
1895
  session: sessionParam,
1409
1896
  },
1410
- }, serializedPerSession("scout_report", async ({ force, level, history }, session) => {
1897
+ }, serializedPerSession("scout_report", async ({ force, level, history, report }, session) => {
1411
1898
  try {
1412
1899
  const eng = engineFor(session);
1413
1900
  if (!eng.memory)
@@ -1444,6 +1931,7 @@ server.registerTool("scout_report", {
1444
1931
  routesTotal: all.length,
1445
1932
  designAudits: auditsThisRun,
1446
1933
  unvisitedRoutes: unvisited,
1934
+ knownRoutes: all,
1447
1935
  mode: eng.mode,
1448
1936
  });
1449
1937
  if (lvl === "extensive" && gapList.length > 0) {
@@ -1458,21 +1946,25 @@ server.registerTool("scout_report", {
1458
1946
  return text(`NOT GENERATED — the '${lvl}' completion contract is unmet:\n\n${gates.join("\n\n")}\n\n` +
1459
1947
  `Then call scout_report again. Pass force=true ONLY if the user explicitly capped the budget.`, session);
1460
1948
  }
1461
- const { path: p, summary } = generateReport(eng.memory, eng.oracleLog.all, {
1949
+ const { path: p, summary, html, } = generateReport(eng.memory, eng.oracleLog.all, {
1462
1950
  history,
1951
+ report,
1463
1952
  routesVisited: all.length - unvisited.length,
1464
1953
  routesTotal: all.length,
1465
1954
  designAudits: auditsThisRun,
1466
1955
  createdResources: eng.createdResources,
1467
1956
  unvisitedRoutes: unvisited,
1957
+ knownRoutes: all,
1468
1958
  mode: eng.mode,
1469
1959
  trustedEmbeds: [...eng.trustedEmbeds],
1960
+ readPosts: eng.readPosts.map((e) => e.entry),
1470
1961
  policyAttributed: eng.oracleLog.policyAttributed,
1471
1962
  // Which sessions are still open decides whether a quiet one is holding a browser, and how long its trailing idle runs.
1472
1963
  attachedSessions: [...engines.keys()],
1473
1964
  });
1474
1965
  void p;
1475
- return text(summary, session);
1966
+ const openNote = html && openDecisions.get(session)?.report ? openForUser("the report", html) : "";
1967
+ return text(summary + openNote, session);
1476
1968
  }
1477
1969
  catch (err) {
1478
1970
  return errorText(err);
@@ -1536,6 +2028,108 @@ server.registerTool("scout_verify", {
1536
2028
  return errorText(err);
1537
2029
  }
1538
2030
  }));
2031
+ // Answering the tickets: read their acceptance criteria, then record a verdict on each.
2032
+ server.registerTool("scout_tickets", {
2033
+ description: "Read the tickets or acceptance criteria the person gave this run, pasted (`text`) or from a file (`path`), and keep them so the report answers each criterion: passed, failed or not tested. " +
2034
+ 'Recognises Given/When/Then scenarios, checklists, numbered and "AC1:" criteria, and lists under an "Acceptance criteria" heading; several tickets may be given at once. ' +
2035
+ "A ticket with no recognisable criteria is reported as such, never guessed at. Returns each criterion's id (AC1, AC2, …) to judge it by with scout_criterion. Touches no browser.",
2036
+ inputSchema: {
2037
+ text: z.string().max(MAX_TICKET_TEXT).optional().describe("The tickets as pasted. Pass this or `path`."),
2038
+ path: z
2039
+ .string()
2040
+ .optional()
2041
+ .describe(`A ticket file to read (${TICKET_FILE_EXTENSIONS.join(", ")}), absolute or relative to the project folder. Pass this or \`text\`.`),
2042
+ session: sessionParam,
2043
+ },
2044
+ }, serializedPerSession("scout_tickets", async ({ text: pasted, path: file }, session) => {
2045
+ try {
2046
+ const eng = engineFor(session);
2047
+ if (!eng.memory)
2048
+ throw new Error("Not attached — attach first, so the tickets are kept with the project.");
2049
+ if ((pasted === undefined) === (file === undefined))
2050
+ return text("Pass the tickets as `text`, or a file as `path`: one of the two.", session);
2051
+ let body = pasted ?? "";
2052
+ let source = "pasted text";
2053
+ if (file !== undefined) {
2054
+ // The file a link points at is what is read, so its name is what is checked: a "notes.md" link to a key file is refused.
2055
+ const full = fs.realpathSync(path.isAbsolute(file) ? file : path.resolve(path.dirname(eng.memory.dir), file));
2056
+ if (!isTicketFileName(full))
2057
+ return text(`Not read: a ticket file is one of ${TICKET_FILE_EXTENSIONS.join(", ")}. Paste anything else as \`text\`.`, session);
2058
+ const stat = fs.statSync(full);
2059
+ if (!stat.isFile())
2060
+ return text(`Not read: ${full} is not a file.`, session);
2061
+ if (stat.size > MAX_TICKET_FILE_BYTES)
2062
+ return text(`Not read: ${full} is larger than ${MAX_TICKET_FILE_BYTES} bytes.`, session);
2063
+ body = fs.readFileSync(full, "utf8");
2064
+ source = path.basename(full);
2065
+ }
2066
+ if (!body.trim())
2067
+ return text("Nothing to read: the tickets are empty.", session);
2068
+ const parsed = parseTickets(body, source);
2069
+ const kept = eng.memory.addTickets(parsed);
2070
+ // Say what a bound left out, so a long backlog is never answered in part without a word.
2071
+ const cuts = [
2072
+ ...(body.length > MAX_TICKET_TEXT ? [`only the first ${MAX_TICKET_TEXT} characters were read`] : []),
2073
+ ...(parsed.length >= MAX_TICKETS ? [`at most ${MAX_TICKETS} tickets are read at once`] : []),
2074
+ ];
2075
+ return text(formatReading(kept) + (cuts.length ? `\n\n⚠ Not everything was read: ${cuts.join("; ")}. Read the rest in another call.` : ""), session);
2076
+ }
2077
+ catch (err) {
2078
+ return errorText(err);
2079
+ }
2080
+ }));
2081
+ server.registerTool("scout_criterion", {
2082
+ description: "Record whether one acceptance criterion of a ticket read with scout_tickets passed, failed or was not tested, with how sure you are. " +
2083
+ "The link from a criterion to the findings that show it is YOUR judgement, stated with a confidence — never matched on words. " +
2084
+ 'A "fail" names the findings that show it (file them with scout_finding first); "not-tested" says why in untestedBecause. Recording the same criterion again from the same session replaces your earlier verdict. Touches no browser.',
2085
+ inputSchema: {
2086
+ ticket: z.string().min(1).describe('The ticket\'s id as scout_tickets gave it, e.g. "PROJ-12" or "T1"'),
2087
+ criterion: z.string().min(1).describe('The criterion\'s id, e.g. "AC2" (or just "2")'),
2088
+ verdict: z.enum(CRITERION_VERDICTS).describe('"pass", "fail" or "not-tested"'),
2089
+ findings: z
2090
+ .array(z.string())
2091
+ .max(MAX_CRITERION_FINDINGS)
2092
+ .optional()
2093
+ .describe("Ids of the findings that show this criterion failing (required for a fail; may be given for a pass, none for not-tested)"),
2094
+ confidence: z.number().min(0).max(1).describe("How sure you are of this verdict and of the findings linked to it, from 0 to 1. State it honestly"),
2095
+ reason: z.string().min(1).max(MAX_REASON).describe("What you saw, or why it could not be tried, in a sentence"),
2096
+ untestedBecause: z
2097
+ .enum(NOT_TESTED_REASONS)
2098
+ .optional()
2099
+ .describe('Only with verdict "not-tested": "no-access" (the role this run used could not reach it), "observe-blocked" (it needs a change sent and this session is in observe mode), "out-of-scope" (it is outside what this run could check, such as an email or another system)'),
2100
+ session: sessionParam,
2101
+ },
2102
+ }, serializedPerSession("scout_criterion", async (args, session) => {
2103
+ try {
2104
+ const eng = engineFor(session);
2105
+ const memory = eng.memory;
2106
+ if (!memory)
2107
+ throw new Error("Not attached.");
2108
+ const judged = judgeCriterion(args, {
2109
+ tickets: memory.tickets,
2110
+ findings: memory.findings,
2111
+ mode: eng.mode,
2112
+ session: eng.sessionKey,
2113
+ at: new Date().toISOString(),
2114
+ });
2115
+ if (!judged.ok)
2116
+ return text(`Not recorded: ${judged.reason}.`, session);
2117
+ memory.addCriterionVerdict(judged.record);
2118
+ const r = judged.record;
2119
+ const ticket = memory.tickets.find((t) => t.id === r.ticket);
2120
+ const criterion = ticket ? findCriterion(ticket, r.criterion) : undefined;
2121
+ const linked = r.findings.map((id) => memory.findings.find((f) => f.id === id)).filter((f) => f !== undefined);
2122
+ const said = r.verdict === "pass" ? "passes" : r.verdict === "fail" ? "fails" : `was not tested (${r.untestedBecause})`;
2123
+ return text([
2124
+ `Recorded: ${r.ticket} ${r.criterion} ${said}, confidence ${r.confidence.toFixed(2)}.`,
2125
+ ...(criterion ? [` ${criterion.text}`] : []),
2126
+ ...linked.map((f) => ` linked: [${f.severity}] ${f.title} (${f.id})`),
2127
+ ].join("\n"), session);
2128
+ }
2129
+ catch (err) {
2130
+ return errorText(err);
2131
+ }
2132
+ }));
1539
2133
  server.registerTool("scout_close", {
1540
2134
  description: "Close a session's browser (memory persists on disk). Default: the DEFAULT session. Pass session to close a specific one, or all=true to close every live session at the end of a multi-role run. " +
1541
2135
  "A session that is a lane of a parallel run (named by scout_lane_brief or scout_lane_report) is not closed until its lane report has been accepted by scout_lane_report, since folding needs the session attached; the refusal names every such lane.",
@@ -1588,6 +2182,7 @@ server.registerTool("scout_close", {
1588
2182
  await eng.close();
1589
2183
  const saveError = eng.memory?.lastSaveError;
1590
2184
  engines.delete(name);
2185
+ openDecisions.delete(name);
1591
2186
  // The last session on this project ends its run.
1592
2187
  if (eng.memory && ![...engines.values()].some((e) => e.memory === eng.memory))
1593
2188
  eng.memory.endRun();
@@ -1648,7 +2243,7 @@ async function shutdown() {
1648
2243
  console.error(`[scenescout] could not remove the live view's token file in ${dir}: ${err instanceof Error ? err.message : String(err)}`);
1649
2244
  }
1650
2245
  }
1651
- await Promise.allSettled([live?.stop(), ...[...engines.values()].map((e) => e.close())]);
2246
+ await Promise.allSettled([live?.stop(), ...[...engines.values()].map((e) => e.close()), ...pendingLogins.all().map((p) => p.cancel())]);
1652
2247
  }
1653
2248
  process.on("SIGINT", () => {
1654
2249
  void shutdown().finally(() => process.exit(0));