scenescout 3.2.0 → 3.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,97 @@
1
+ /**
2
+ * How long to wait after an action before reading the page.
3
+ *
4
+ * This used to be a flat 400 ms sleep on every action, for a good reason:
5
+ * `waitForLoadState("networkidle")` latches once reached and then resolves
6
+ * instantly forever after, so it cannot be used to wait out the request an
7
+ * action just fired, and draining the oracles before that request lands would
8
+ * misattribute its violations to the next step.
9
+ *
10
+ * The floor was still the wrong shape. Measured against a demo app, a snapshot
11
+ * cost 740 ms of which 400 ms was the sleep — 54% of the wall time — and a run
12
+ * of 203 actions spent 81 seconds asleep. Meanwhile a page slower than 400 ms
13
+ * got cut off exactly when it mattered.
14
+ *
15
+ * The engine already intercepts every request, so it knows what is in flight.
16
+ * Waiting on THAT, with a quiet window after the last one starts, is both
17
+ * faster on a page that answers quickly and more patient with one that does
18
+ * not. The old constant survives as a ceiling rather than a floor.
19
+ *
20
+ * The same rule carries the deliberate slow-down: a run being watched by a
21
+ * person taking notes wants a fixed pace, not the fastest one. `paceMs` is a
22
+ * floor a session asks for; unset, a session goes as fast as its page allows.
23
+ */
24
+ /** No request has started for this long and none is in flight: the page has settled. */
25
+ export const QUIET_MS = 120;
26
+ /** Longest to wait for requests to drain, however busy the page is. A page that never goes quiet must not stall the run. */
27
+ export const SETTLE_CAP_MS = 2000;
28
+ /** How often the wait re-checks. */
29
+ export const SETTLE_TICK_MS = 20;
30
+ /**
31
+ * Whether to keep waiting. Split out from the browser so the rule can be
32
+ * table-tested: it is the difference between a run that is fast and one that
33
+ * reads a page before it has finished changing.
34
+ */
35
+ export function shouldKeepWaiting(state) {
36
+ // A pace a session asked for is a floor on the whole wait, and it is the
37
+ // only reason to wait once everything else says the page is ready.
38
+ if (state.elapsedMs < state.paceMs)
39
+ return true;
40
+ // Never longer than the cap, whatever the page is doing.
41
+ if (state.elapsedMs >= SETTLE_CAP_MS)
42
+ return false;
43
+ // Something is still out: wait for it, up to the cap.
44
+ if (state.inFlight > 0)
45
+ return true;
46
+ // Nothing in flight and nothing recent. The action itself is the last event
47
+ // that could still produce a request, so the quiet window runs from whichever
48
+ // of the two is later. Without this, an action whose request comes back
49
+ // through a hop — a click that posts to a service worker, which then fetches —
50
+ // is read while the page is between the two, and the request lands during the
51
+ // NEXT action, which is then blamed for it.
52
+ return Math.min(state.sinceLastStartMs, state.elapsedMs) < QUIET_MS;
53
+ }
54
+ /** A pace a session asked for, clamped. Negative or absurd values are a typo, not an instruction. */
55
+ export const PACE_MAX_MS = 60_000;
56
+ export function normalizePace(paceMs) {
57
+ if (typeof paceMs !== "number" || !Number.isFinite(paceMs) || paceMs <= 0)
58
+ return 0;
59
+ return Math.min(Math.round(paceMs), PACE_MAX_MS);
60
+ }
61
+ /** What the attach result says about a deliberate pace, or nothing when the session runs at full speed. */
62
+ export function describePace(paceMs) {
63
+ if (paceMs <= 0)
64
+ return "";
65
+ return `\n⏱ PACE: at least ${paceMs} ms between actions, so a person can follow along. Unset it with paceMs: 0 to go as fast as the page allows.`;
66
+ }
67
+ /**
68
+ * Waiting for a client-side redirect that has not happened yet.
69
+ *
70
+ * A guard that redirects on a timer after hydration issues no request until it
71
+ * fires, so there is nothing for the request-based settle above to wait on:
72
+ * it goes quiet, the URL is read, and the page reports the route it was asked
73
+ * for rather than the login page it actually bounced to. The old flat 400 ms
74
+ * sleep covered this by accident, and removing it made the bounce invisible
75
+ * whenever a loaded machine delayed the timer past the quiet window.
76
+ *
77
+ * So where a BOUNCE VERDICT is about to be made — attach judging a storage
78
+ * state, navigate judging coverage — the URL is watched until it has held
79
+ * still, rather than read once. This is paid per navigation, not per action.
80
+ */
81
+ /**
82
+ * How long the URL must hold still before a bounce verdict is believed.
83
+ *
84
+ * This is a window, not a guarantee: a guard slower than it still lands after
85
+ * the verdict. What it buys is the realistic case — a guard written to fire
86
+ * promptly, delayed by a loaded machine — which is the one that made this
87
+ * project's own CI fail intermittently on a 40 ms timer.
88
+ */
89
+ export const URL_QUIET_MS = 400;
90
+ /** Longest to watch. A page redirecting in a loop must not hold up the run. */
91
+ export const URL_CAP_MS = 3000;
92
+ /** Whether to keep watching the URL before judging where the page landed. */
93
+ export function keepWatchingUrl(w) {
94
+ if (w.elapsedMs >= URL_CAP_MS)
95
+ return false;
96
+ return w.sinceChangeMs < URL_QUIET_MS;
97
+ }
@@ -0,0 +1,142 @@
1
+ /**
2
+ * Re-testing findings a previous run left open.
3
+ *
4
+ * The report has always carried two kinds of finding and been honest that they
5
+ * are not the same thing: what this run saw, and what some earlier run saw.
6
+ * The second kind is labelled historical and unverified, which is accurate and
7
+ * almost useless — a reader cannot tell a bug fixed three weeks ago from one
8
+ * that is still costing users money today, and neither can the next run.
9
+ *
10
+ * Closing that gap by hand is possible and nobody does it. It means opening
11
+ * the report, copying each finding's route and evidence, re-walking them one
12
+ * at a time, and calling `scout_resolve` on the ones that are gone — per
13
+ * finding, for as many as the history holds. One project's history held over
14
+ * three hundred.
15
+ *
16
+ * So the campaign is a tool. Asked for a worklist, it hands back the open
17
+ * findings in the order a run should re-test them: by severity, grouped so a
18
+ * route is walked once rather than once per finding, each with the evidence
19
+ * that identifies it and the steps that produced it. Asked to record a
20
+ * verdict, it stamps what was seen and when, so the next report can say
21
+ * "confirmed still present" with a date instead of "unverified".
22
+ *
23
+ * Pure, so the ordering and the wording can be table-tested.
24
+ */
25
+ import { normalizePath } from "./fingerprint.js";
26
+ /** What a re-test found. */
27
+ export const VERDICTS = ["gone", "present", "changed"];
28
+ export const VERDICT_MEANING = {
29
+ gone: "re-tested and the evidence no longer reproduces — resolved",
30
+ present: "re-tested and still reproduces exactly as filed",
31
+ changed: "re-tested and something is different — still broken, but not as described",
32
+ };
33
+ /** Most findings a single campaign hands over at once. A worklist longer than this is a plan nobody follows. */
34
+ export const MAX_WORKLIST = 25;
35
+ const SEVERITY_ORDER = { high: 0, medium: 1, low: 2 };
36
+ function item(f) {
37
+ return {
38
+ id: f.id,
39
+ severity: f.severity,
40
+ category: f.category,
41
+ title: f.title,
42
+ route: normalizePath(f.url),
43
+ url: f.url,
44
+ evidence: f.evidence ?? "",
45
+ repro: f.repro ?? [],
46
+ foundAt: f.foundAt,
47
+ runs: f.runs ?? 1,
48
+ ...(f.verdict ? { lastVerdict: f.verdict } : {}),
49
+ ...(f.verifiedAt ? { lastVerifiedAt: f.verifiedAt } : {}),
50
+ };
51
+ }
52
+ /**
53
+ * The findings to re-test, in the order to do it.
54
+ *
55
+ * Severity decides what matters; route groups what is cheap, because walking
56
+ * one route and checking four findings on it costs a fraction of walking four
57
+ * routes. Within a route, a finding never verified comes before one verified
58
+ * before — re-checking the same finding twice while another has never been
59
+ * looked at is the failure mode this ordering exists to prevent.
60
+ */
61
+ export function verifyWorklist(findings, ids) {
62
+ const wanted = ids && ids.length > 0 ? new Set(ids) : null;
63
+ const open = findings.filter((f) => (f.status ?? "open") !== "resolved").filter((f) => (wanted ? wanted.has(f.id) : true));
64
+ // The severity of a route is its worst finding: a route carrying a high is
65
+ // walked before one carrying three mediums.
66
+ const worst = new Map();
67
+ for (const f of open) {
68
+ const route = normalizePath(f.url);
69
+ const rank = SEVERITY_ORDER[f.severity] ?? 3;
70
+ worst.set(route, Math.min(worst.get(route) ?? 9, rank));
71
+ }
72
+ return open
73
+ .map(item)
74
+ .sort((a, b) => (worst.get(a.route) ?? 9) - (worst.get(b.route) ?? 9) ||
75
+ a.route.localeCompare(b.route) ||
76
+ (SEVERITY_ORDER[a.severity] ?? 3) - (SEVERITY_ORDER[b.severity] ?? 3) ||
77
+ Number(Boolean(a.lastVerifiedAt)) - Number(Boolean(b.lastVerifiedAt)) ||
78
+ (a.lastVerifiedAt ?? "").localeCompare(b.lastVerifiedAt ?? "") ||
79
+ a.id.localeCompare(b.id))
80
+ .slice(0, MAX_WORKLIST);
81
+ }
82
+ /** Ids the caller asked for that no open finding matches — a silently short worklist is worse than a named miss. */
83
+ export function unknownIds(findings, ids) {
84
+ const open = new Set(findings.filter((f) => (f.status ?? "open") !== "resolved").map((f) => f.id));
85
+ return ids.filter((id) => !open.has(id));
86
+ }
87
+ /** The campaign brief the agent works down. */
88
+ export function formatWorklist(items, missing = []) {
89
+ if (items.length === 0) {
90
+ return missing.length > 0
91
+ ? `Nothing to verify. No OPEN finding matches: ${missing.join(", ")}. They may already be resolved, or belong to another project's memory.`
92
+ : "Nothing to verify — no open findings in this project's memory.";
93
+ }
94
+ const lines = [
95
+ `VERIFY CAMPAIGN — ${items.length} open finding(s) to re-test, worst route first.`,
96
+ `Re-test each one, then record what you saw: scout_verify { id, verdict: "gone" | "present" | "changed", note }.`,
97
+ `"gone" resolves it. "present" stamps it confirmed, so the report stops calling it unverified. "changed" keeps it open and says what differs.`,
98
+ ``,
99
+ ];
100
+ let route = "";
101
+ for (const it of items) {
102
+ if (it.route !== route) {
103
+ route = it.route;
104
+ lines.push(`── ${route} ──`);
105
+ }
106
+ const seen = it.lastVerifiedAt ? ` · last re-tested ${it.lastVerifiedAt.slice(0, 10)} (${it.lastVerdict})` : " · never re-tested";
107
+ lines.push(`[${it.severity}] ${it.id} ${it.title}`);
108
+ lines.push(` ${it.category} · found ${it.foundAt.slice(0, 10)} · ${it.runs} run(s)${seen}`);
109
+ if (it.evidence)
110
+ lines.push(` evidence: ${it.evidence}`);
111
+ if (it.repro.length > 0)
112
+ lines.push(` repro: ${it.repro.slice(0, 6).join(" → ")}`);
113
+ lines.push(` at: ${it.url}`);
114
+ }
115
+ if (missing.length > 0)
116
+ lines.push(``, `No OPEN finding matches: ${missing.join(", ")}.`);
117
+ return lines.join("\n");
118
+ }
119
+ /** Whether a verdict resolves the finding it is recorded against. */
120
+ export function resolvesFinding(verdict) {
121
+ return verdict === "gone";
122
+ }
123
+ /** What the agent reads back after recording one. */
124
+ export function describeVerdict(f, verdict, note) {
125
+ const tail = note ? `\n ${note}` : "";
126
+ if (verdict === "gone")
127
+ return `Resolved by re-test: [${f.severity}] ${f.title} (${f.id}).${tail}`;
128
+ if (verdict === "present")
129
+ return `Confirmed still present: [${f.severity}] ${f.title} (${f.id}). It stays open, and the report now dates the confirmation instead of calling it unverified.${tail}`;
130
+ return `Still open and changed: [${f.severity}] ${f.title} (${f.id}). Re-file the difference as its own finding if the behaviour is now a different bug.${tail}`;
131
+ }
132
+ /** How a verified finding reads in the report, or nothing when it was never re-tested. */
133
+ export function sayVerification(f) {
134
+ if (!f.verdict || !f.verifiedAt)
135
+ return "";
136
+ const on = f.verifiedAt.slice(0, 10);
137
+ if (f.verdict === "present")
138
+ return ` · confirmed still present on ${on}`;
139
+ if (f.verdict === "changed")
140
+ return ` · re-tested ${on}, behaviour has changed since it was filed`;
141
+ return ` · re-tested ${on}`;
142
+ }
@@ -38,9 +38,12 @@ import { FINDING_CATEGORIES, MemoryStore, redactSecrets } from "./engine/memory.
38
38
  import { LANE_NAME_MAX, laneReportInstruction, parseLaneReport, summarizeLaneReport } from "./engine/lane.js";
39
39
  import { SessionQueue, withWatchdog } from "./engine/dispatch.js";
40
40
  import { FIXTURE_KINDS } from "./engine/fixtures.js";
41
- import { feedForSession, LIVE_ENV, writeStatusFile, LIVE_TOKEN_FILE, LiveServer, StatusBoard, } from "./engine/live.js";
41
+ import { feedForSession, LIVE_ENV, writeStatusFile, LIVE_TOKEN_FILE, liveEngines, liveTokenFileName, pidAlive, statusFileName, LiveServer, StatusBoard, } from "./engine/live.js";
42
+ import { formatBriefs, MAX_LANES, planLanes } from "./engine/brief.js";
42
43
  import { computeGaps, formatRouteCoverage, generateReport, replayDocument, reportEvidence } from "./engine/report.js";
44
+ import { describeVerdict, formatWorklist, unknownIds, VERDICTS, verifyWorklist } from "./engine/verify.js";
43
45
  import { RECORD_MAX_FRAMES, resolveFrame } from "./engine/replay.js";
46
+ import { describePace, normalizePace } from "./engine/settle.js";
44
47
  import { needsTask, taskRefusal, TASK_MAX } from "./engine/task.js";
45
48
  import { EXPLORE_PROMPT_ARGUMENTS, explorePrompt, loadPlaybook, PLAYBOOK_PROMPT, PLAYBOOK_TOOL, SERVER_INSTRUCTIONS } from "./playbook.js";
46
49
  import { formatScan, scanProject } from "./scan.js";
@@ -241,15 +244,20 @@ function reportExtras(eng) {
241
244
  mode: eng.mode,
242
245
  policyAttributed: eng.oracleLog.policyAttributed,
243
246
  version: PKG_VERSION,
247
+ attachedSessions: [...engines.keys()],
244
248
  };
245
249
  }
246
250
  /** Hand the live view's token to `scenescout watch` through a file only the owner can read. */
247
251
  function publishLiveToken(dir) {
248
252
  if (!liveAddress || liveDirs.has(dir) || liveTokenWrites.has(dir))
249
253
  return;
250
- const file = path.join(dir, LIVE_TOKEN_FILE);
254
+ // Named for this process: two engines on one project would otherwise hand
255
+ // `watch` one token for two ports, and whichever wrote last would win.
256
+ const file = path.join(dir, liveTokenFileName(process.pid));
251
257
  const write = fs.promises
252
258
  .writeFile(file, liveAddress.token, { mode: 0o600 })
259
+ // The shared name stays too, for a `watch` from before per-pid files.
260
+ .then(() => fs.promises.writeFile(path.join(dir, LIVE_TOKEN_FILE), liveAddress.token, { mode: 0o600 }))
253
261
  // `mode` applies only when the file is created; a leftover one keeps its old bits.
254
262
  .then(() => fs.promises.chmod(file, 0o600))
255
263
  .then(() => {
@@ -487,6 +495,27 @@ server.server.setRequestHandler(GetPromptRequestSchema, (request) => {
487
495
  // The lane report: how a parallel agent hands its results back to the planner
488
496
  // as typed decisions. One tool for both halves, so the instruction a lane is
489
497
  // given and the parser its reply meets are the same code.
498
+ // Splitting the app between lanes: the other half of the parallel protocol.
499
+ // scout_lane_report is how a lane hands its answers back; this is what the
500
+ // planner hands it in the first place.
501
+ server.registerTool("scout_lane_brief", {
502
+ description: "For a run split across parallel agents (lanes). Divides the app's known routes between N lanes and returns each lane's session name, the `objective` to attach it with, and the routes it owns — whole modules per lane, balanced by route count, so no two lanes audit the same area and none is left unopened. Call it after the first crawl, when route knowledge is complete. Touches no browser; pass the briefs to your lane agents, then use scout_lane_report for what they hand back.",
503
+ inputSchema: {
504
+ lanes: z.number().int().min(1).max(MAX_LANES).describe(`How many lanes to split across (1–${MAX_LANES})`),
505
+ goal: z.string().max(200).optional().describe("What the whole run is for; each lane's objective is written against it"),
506
+ routes: z.array(z.string()).max(500).optional().describe("Routes to split. Omit to split every route this project knows about."),
507
+ session: sessionParam,
508
+ },
509
+ }, serializedPerSession("scout_lane_brief", async ({ lanes, goal, routes }, session) => {
510
+ try {
511
+ const eng = engineFor(session);
512
+ const all = routes && routes.length > 0 ? routes : eng.allKnownRoutes();
513
+ return text(formatBriefs(planLanes(all, lanes, { goal, mode: eng.mode, role: eng.role }), { goal, mode: eng.mode, role: eng.role }), session);
514
+ }
515
+ catch (err) {
516
+ return errorText(err);
517
+ }
518
+ }));
490
519
  server.registerTool("scout_lane_report", {
491
520
  description: "For a run split across parallel agents (lanes). Without `reply`: returns the paragraph to put in a lane's prompt, telling it to hand back ONE typed JSON object (verdicts, severities and categories from closed sets, a calibrated confidence per decision, routes covered, what blocked it). " +
492
521
  "With `reply`: parses what the lane handed back and returns the one-line fold (defects, highs, unsure, mean confidence, routes) or the reason it was refused, to relay to the lane once. Touches no browser.",
@@ -499,10 +528,35 @@ server.registerTool("scout_lane_report", {
499
528
  if (reply === undefined)
500
529
  return { content: [{ type: "text", text: laneReportInstruction(lane) }] };
501
530
  const parsed = parseLaneReport(reply, lane);
502
- const out = parsed.ok
503
- ? `Lane report accepted — ${summarizeLaneReport(parsed.report)}`
504
- : `Lane report REFUSED: ${parsed.reason}. Ask the lane once for the corrected object; do not re-judge its prose.`;
505
- return { content: [{ type: "text", text: out }] };
531
+ if (!parsed.ok) {
532
+ return {
533
+ content: [
534
+ { type: "text", text: `Lane report REFUSED: ${parsed.reason}. Ask the lane once for the corrected object; do not re-judge its prose.` },
535
+ ],
536
+ };
537
+ }
538
+ // Keep what the lane decided, so the confidence it stated can be checked
539
+ // against what the run goes on to file. Best-effort: a lane report is
540
+ // still accepted if this project has no memory open yet, because the
541
+ // planner's fold must not depend on where the report was written.
542
+ // ONLY the lane's own session. Falling back to any engine with memory
543
+ // open put one project's decisions into another project's store whenever
544
+ // two sessions were attached to different apps — and the lane having
545
+ // already closed makes that the ordinary case, not an edge one.
546
+ const owner = engines.get(lane)?.memory;
547
+ const at = new Date().toISOString();
548
+ const kept = owner
549
+ ? owner.addLaneDecisions(lane, parsed.report.decisions.map((d) => ({ ...d, lane, at })))
550
+ : 0;
551
+ // Say when nothing was kept. Every lane closing its session before the
552
+ // planner folds its report is the order the method describes, and it
553
+ // leaves no memory to write to — reporting a bare "accepted" while the
554
+ // skill promises the decisions are kept is the kind of silence that
555
+ // makes a later calibration section look wrong rather than absent.
556
+ const note = kept > 0
557
+ ? ` (${kept} decision(s) kept for calibration)`
558
+ : ` (nothing kept for calibration — session ${JSON.stringify(lane)} is not attached here, so there is no project memory to write to. Fold a lane report before closing that lane's session.)`;
559
+ return { content: [{ type: "text", text: `Lane report accepted — ${summarizeLaneReport(parsed.report)}${note}` }] };
506
560
  }
507
561
  catch (err) {
508
562
  return errorText(err);
@@ -543,7 +597,18 @@ server.registerTool("scout_attach", {
543
597
  .describe('This session\'s objective: the whole remit you were given, in one sentence ("Admin lane: §2 registers, §7 plan gating", ' +
544
598
  '"Approve and reject orders as a manager"). It sits above the task, which is what the session is doing at any moment. ' +
545
599
  "Shown to whoever is watching the run; worth setting whenever more than one session is live."),
546
- task: z.string().max(300).optional().describe("Old name for `objective` (2.0). Prefer `objective`."),
600
+ task: z
601
+ .string()
602
+ .max(300)
603
+ .optional()
604
+ .describe("What this session is doing right now, shown under its objective from the moment it appears, e.g. Signing in and taking stock. Passed alone it is read as the 2.0 spelling of `objective`. Defaults to a placeholder so a fresh card never reads as idle."),
605
+ paceMs: z
606
+ .number()
607
+ .int()
608
+ .min(0)
609
+ .max(60000)
610
+ .optional()
611
+ .describe("A floor between actions, in milliseconds, for when a person is watching and needs to keep up — following a flow, taking notes, demonstrating. Default 0: as fast as the page allows, which is what a run wants otherwise. Changeable mid-run with scout_session {paceMs}."),
547
612
  record: z
548
613
  .boolean()
549
614
  .default(false)
@@ -555,7 +620,7 @@ server.registerTool("scout_attach", {
555
620
  .optional()
556
621
  .describe("Session name for multi-role runs (e.g. 'admin', 'qa'). Creates/replaces that session's browser and makes it the default. Default: 'default'."),
557
622
  },
558
- }, serializedControl(async ({ url, projectPath, storageStatePath, mode, headed, browser, viewportWidth, viewportHeight, objective, task, record, session, }) => {
623
+ }, serializedControl(async ({ url, projectPath, storageStatePath, mode, headed, browser, viewportWidth, viewportHeight, objective, task, record, paceMs, session, }) => {
559
624
  try {
560
625
  const target = session ?? activeName;
561
626
  if (session) {
@@ -624,8 +689,13 @@ server.registerTool("scout_attach", {
624
689
  headed,
625
690
  browser,
626
691
  viewport,
627
- // `task` is what this was called in 2.0; it named the session's whole remit, which is the objective.
692
+ // `task` is what this was called in 2.0, where it named the session's
693
+ // whole remit. Alone it still means that. Given BESIDE an objective it
694
+ // means what it means everywhere else — what this session is doing
695
+ // right now — so the card says something from the moment it appears.
628
696
  objective: objective ?? task,
697
+ paceMs,
698
+ task: objective ? task : undefined,
629
699
  record,
630
700
  memoryStore: store,
631
701
  });
@@ -643,7 +713,7 @@ server.registerTool("scout_attach", {
643
713
  const recordNote = record && eng.memory?.dir
644
714
  ? `\n\n📸 RECORDING: a frame of the page after each action, under ${path.join(eng.memory.dir, "recordings", target)}/ (at most ${RECORD_MAX_FRAMES}). scout_report writes them into report.html beside report.md.`
645
715
  : "";
646
- return text(out + conflictNote + recordNote + (engines.size > 1 ? `\n${sessionLines()}` : "") + liveLine(), target);
716
+ return text(out + conflictNote + recordNote + describePace(eng.pace) + (engines.size > 1 ? `\n${sessionLines()}` : "") + liveLine(), target);
647
717
  }
648
718
  catch (err) {
649
719
  return errorText(err);
@@ -664,10 +734,32 @@ server.registerTool("scout_session", {
664
734
  // costs nothing and removes a guaranteed first-try rejection, since all
665
735
  // schemas are additionalProperties:false and reject the near-miss hard.
666
736
  session: z.string().max(40).optional().describe("Alias for `name`."),
737
+ paceMs: z
738
+ .number()
739
+ .int()
740
+ .min(0)
741
+ .max(60000)
742
+ .optional()
743
+ .describe("Change how fast this session acts, mid-run: a floor between actions in milliseconds, for when a person is watching and needs to keep up. 0 restores full speed. With `name`, applies to that session; without, to every live session — which is what 'slow everything down so I can follow' means."),
667
744
  },
668
- }, serializedControl(async ({ name, session }) => {
745
+ }, serializedControl(async ({ name, session, paceMs }) => {
669
746
  try {
670
747
  name = name ?? session;
748
+ // A pace with no session named is meant for the whole run: somebody is
749
+ // watching and wants to keep up with all of it, not one lane.
750
+ if (paceMs !== undefined && !name) {
751
+ const applied = [...engines.values()].map((e) => e.setPace(paceMs));
752
+ const at = applied[0] ?? normalizePace(paceMs);
753
+ return text((at > 0
754
+ ? `Every live session now waits at least ${at} ms between actions, so a person can follow along.`
755
+ : `Every live session is back to full speed: as fast as its page allows.`) + `\n${sessionLines()}`, activeName);
756
+ }
757
+ if (paceMs !== undefined && name) {
758
+ if (!engines.has(name))
759
+ return text(`No session named "${name}" yet — create it with scout_attach { session: "${name}", … }.\n${sessionLines()}`, activeName);
760
+ const at = engines.get(name).setPace(paceMs);
761
+ return text((at > 0 ? `${name} now waits at least ${at} ms between actions.` : `${name} is back to full speed.`) + `\n${sessionLines()}`, activeName);
762
+ }
671
763
  if (!name)
672
764
  return text(sessionLines() + liveLine(), activeName);
673
765
  if (!engines.has(name)) {
@@ -860,6 +952,28 @@ server.registerTool("scout_navigate", {
860
952
  return errorText(err);
861
953
  }
862
954
  }));
955
+ server.registerTool("scout_request", {
956
+ description: "Call the app's own API as this session, with the UI bypassed — the check that turns a hidden or disabled control into a proven refusal. A button that is not shown proves nothing; the same action refused by the server does. The fetch runs IN the page, so it carries the session's cookies and replays the Authorization header the app itself last sent, and it passes through the same interception the write policy is enforced on: in safe-write a mutation on a record this session did not create is refused here exactly as it would be for a click, and that refusal is the engine's safety net, not a finding. Returns the status line, the timing, the headers that decide whether two responses are truly identical (content-type, location, www-authenticate, retry-after, cache-control), and the body. Unlike a shell call, every request is recorded in the run's trail and its signature is what a finding should quote. Paths are fenced to the attached origin: use another session to reach another host.",
957
+ inputSchema: {
958
+ path: z.string().min(1).max(2000).describe("Path on the attached origin, e.g. /api/things/12, or a full URL on that same origin"),
959
+ method: z.enum(["GET", "HEAD", "POST", "PUT", "PATCH", "DELETE", "OPTIONS"]).optional().describe("Default GET"),
960
+ body: z.string().max(20000).optional().describe("Request body, sent as application/json unless a content-type header is given"),
961
+ headers: z
962
+ .record(z.string().max(2000))
963
+ .optional()
964
+ .describe("Extra headers. One given here wins over the app's own, which is how a session tests a different or absent credential."),
965
+ task: taskParam,
966
+ objective: legacyObjectiveParam,
967
+ session: sessionParam,
968
+ },
969
+ }, serializedPerSession("scout_request", async (args, session) => {
970
+ try {
971
+ return text(await engineFor(session).apiRequest({ method: args.method, path: args.path, body: args.body, headers: args.headers }), session);
972
+ }
973
+ catch (err) {
974
+ return errorText(err);
975
+ }
976
+ }));
863
977
  server.registerTool("scout_back", {
864
978
  description: "Go back in browser history (tests back-button resilience).",
865
979
  inputSchema: { task: taskParam, objective: legacyObjectiveParam, session: sessionParam },
@@ -936,11 +1050,11 @@ server.registerTool("scout_journey", {
936
1050
  }
937
1051
  }));
938
1052
  server.registerTool("scout_note", {
939
- description: "Cumulative WRITTEN knowledge about the tested app — .scenescout/ASSUMPTIONS.md, in prose a human can read and correct. memory.json stores coverage; this stores UNDERSTANDING, so every run starts smarter than the last. READ it at the start of every session ({action:'read'}). ADD durable learnings as you go ({action:'add', section, note}): what the app is for (app-model), who each role is and what they're FOR — infer the persona from what the role can see and do, e.g. 'qa-role = reviewer: approves orders, cannot administer' (roles), UI patterns the app follows (conventions), rules discovered the hard way like 'an order can only ship once approved' (constraints), fragile areas worth re-testing every run (risks), domain terms (glossary). Notes are dated, attributed to the acting role, and deduplicated. Do NOT record session-specific facts (ids, counts) — only durable knowledge.",
1053
+ description: "Cumulative WRITTEN knowledge about the tested app — .scenescout/ASSUMPTIONS.md, in prose a human can read and correct. memory.json stores coverage; this stores UNDERSTANDING, so every run starts smarter than the last. READ it at the start of every session ({action:'read'}). ADD durable learnings as you go ({action:'add', section, note}): what the app is for (app-model), who each role is and what they're FOR — infer the persona from what the role can see and do, e.g. 'qa-role = reviewer: approves orders, cannot administer' (roles), UI patterns the app follows (conventions), rules discovered the hard way like 'an order can only ship once approved' (constraints), fragile areas worth re-testing every run (risks), domain terms (glossary), and how to get the app testable at all — the command that regenerates an expired login state, what has to be running (setup), which the engine reads back to you the next time a storage state has expired. Notes are dated, attributed to the acting role, and deduplicated. Do NOT record session-specific facts (ids, counts) — only durable knowledge.",
940
1054
  inputSchema: {
941
1055
  action: z.enum(["read", "add"]).describe("'read' the accumulated knowledge, or 'add' one durable learning"),
942
1056
  section: z
943
- .enum(["app-model", "roles", "conventions", "constraints", "risks", "glossary"])
1057
+ .enum(["app-model", "roles", "conventions", "constraints", "risks", "glossary", "setup"])
944
1058
  .optional()
945
1059
  .describe("For add: which knowledge section this belongs to"),
946
1060
  note: z.string().max(500).optional().describe("For add: the learning, one or two sentences, written for a future reader with no context"),
@@ -1049,9 +1163,13 @@ server.registerTool("scout_report", {
1049
1163
  .enum(["minimal", "medium", "extensive"])
1050
1164
  .default("medium")
1051
1165
  .describe("Which completion contract to enforce — match the level the run was asked for"),
1166
+ history: z
1167
+ .enum(["index", "full"])
1168
+ .default("index")
1169
+ .describe("How much of the history to print. 'index' lists findings from earlier runs, and resolved ones, as a row each: id, severity, age, title. 'full' prints every one in full as before — on one project that was 1.75 MB against 113 KB, nearly half of it findings already fixed. Use 'full' when handing the document to someone who has no access to the memory."),
1052
1170
  session: sessionParam,
1053
1171
  },
1054
- }, serializedPerSession("scout_report", async ({ force, level }, session) => {
1172
+ }, serializedPerSession("scout_report", async ({ force, level, history }, session) => {
1055
1173
  try {
1056
1174
  const eng = engineFor(session);
1057
1175
  if (!eng.memory)
@@ -1100,6 +1218,7 @@ server.registerTool("scout_report", {
1100
1218
  `Then call scout_report again. Pass force=true ONLY if the user explicitly capped the budget.`, session);
1101
1219
  }
1102
1220
  const { path: p, summary } = generateReport(eng.memory, eng.oracleLog.all, {
1221
+ history,
1103
1222
  routesVisited: all.length - unvisited.length,
1104
1223
  routesTotal: all.length,
1105
1224
  designAudits: eng.designAuditCount,
@@ -1140,6 +1259,39 @@ server.registerTool("scout_resolve", {
1140
1259
  return errorText(err);
1141
1260
  }
1142
1261
  }));
1262
+ server.registerTool("scout_verify", {
1263
+ description: 'Re-test findings earlier runs left open. With no arguments, returns the open findings in the order to re-test them — worst route first, grouped so a route is walked once — each with its evidence and repro steps. Pass ids to narrow it to specific findings. After re-testing one, call again with id and verdict to record what you saw: "gone" resolves it, "present" stamps it confirmed so the report stops calling it unverified, "changed" keeps it open and says the behaviour differs. Use after a fix wave, or at the start of a run against an app this project has tested before.',
1264
+ inputSchema: {
1265
+ id: z.string().optional().describe("The finding being verified. Omit to get the worklist."),
1266
+ verdict: z.enum(VERDICTS).optional().describe('What the re-test found: "gone", "present" or "changed". Requires id.'),
1267
+ note: z.string().max(500).optional().describe("What you saw, in a sentence. Shown in the report beside the verdict."),
1268
+ ids: z.array(z.string()).max(50).optional().describe("Narrow the worklist to these finding ids."),
1269
+ session: sessionParam,
1270
+ },
1271
+ }, serializedPerSession("scout_verify", async ({ id, verdict, note, ids }, session) => {
1272
+ try {
1273
+ const eng = engineFor(session);
1274
+ if (!eng.memory)
1275
+ throw new Error("Not attached.");
1276
+ if (verdict && !id)
1277
+ return text(`Pass the finding the verdict is about: scout_verify { id: "a1b2c3d4e5", verdict: "${verdict}" }.`, session);
1278
+ if (id && !verdict) {
1279
+ return text(`Pass what the re-test found: scout_verify { id: "${id}", verdict: "gone" | "present" | "changed" }.`, session);
1280
+ }
1281
+ if (id && verdict) {
1282
+ const f = eng.memory.verifyFinding(id, verdict, note);
1283
+ if (!f)
1284
+ return text(`No finding with id ${id}.`, session);
1285
+ return text(describeVerdict(f, verdict, note), session);
1286
+ }
1287
+ const findings = eng.memory.findings;
1288
+ const missing = ids && ids.length > 0 ? unknownIds(findings, ids) : [];
1289
+ return text(formatWorklist(verifyWorklist(findings, ids), missing), session);
1290
+ }
1291
+ catch (err) {
1292
+ return errorText(err);
1293
+ }
1294
+ }));
1143
1295
  server.registerTool("scout_close", {
1144
1296
  description: "Close a session's browser (memory persists on disk). Default: the DEFAULT session. Pass session to close a specific one, or all=true to close every live session at the end of a multi-role run.",
1145
1297
  inputSchema: {
@@ -1218,7 +1370,13 @@ async function shutdown() {
1218
1370
  await Promise.allSettled(liveTokenWrites.values());
1219
1371
  for (const dir of liveDirs) {
1220
1372
  try {
1221
- fs.rmSync(path.join(dir, LIVE_TOKEN_FILE), { force: true });
1373
+ fs.rmSync(path.join(dir, liveTokenFileName(process.pid)), { force: true });
1374
+ fs.rmSync(path.join(dir, statusFileName(process.pid)), { force: true });
1375
+ // The shared names belong to whichever engine is still running, so they
1376
+ // are only removed when this process is the last one holding them.
1377
+ if (liveEngines(dir, pidAlive).filter((e) => e.pid !== process.pid).length === 0) {
1378
+ fs.rmSync(path.join(dir, LIVE_TOKEN_FILE), { force: true });
1379
+ }
1222
1380
  }
1223
1381
  catch (err) {
1224
1382
  console.error(`[scenescout] could not remove the live view's token file in ${dir}: ${err instanceof Error ? err.message : String(err)}`);
package/dist/scan.js CHANGED
@@ -340,6 +340,49 @@ export function scanProject(projectDir) {
340
340
  notes,
341
341
  };
342
342
  }
343
+ /**
344
+ * Whether a saved login is still good, read from the expiry inside its own
345
+ * token. A stale storage state is otherwise discovered only by attaching and
346
+ * being told AUTH FAILED, after a role has been chosen and a browser started —
347
+ * and in one project every one of 168 saved logins had expired.
348
+ *
349
+ * Best-effort by design: a state whose token cannot be read is reported as
350
+ * nothing rather than as a problem, because plenty of apps do not use JWTs.
351
+ */
352
+ export function describeAuthAge(file, nowMs = Date.now()) {
353
+ let raw;
354
+ try {
355
+ raw = fs.readFileSync(file, "utf8");
356
+ }
357
+ catch {
358
+ return "";
359
+ }
360
+ const exp = jwtExpiry(raw);
361
+ if (exp === null)
362
+ return "";
363
+ const left = Math.round((exp - nowMs) / 60_000);
364
+ if (left <= 0)
365
+ return " (EXPIRED)";
366
+ return left < 30 ? ` (${left}m left)` : "";
367
+ }
368
+ /** The soonest `exp` of any JWT-looking value in the file, as milliseconds. */
369
+ function jwtExpiry(raw) {
370
+ let soonest = null;
371
+ for (const match of raw.matchAll(/eyJ[A-Za-z0-9_-]{4,}\.([A-Za-z0-9_-]{4,})\.[A-Za-z0-9_-]{4,}/g)) {
372
+ try {
373
+ const body = Buffer.from(match[1].replace(/-/g, "+").replace(/_/g, "/"), "base64").toString("utf8");
374
+ const exp = JSON.parse(body).exp;
375
+ if (typeof exp !== "number")
376
+ continue;
377
+ const ms = exp * 1000;
378
+ soonest = soonest === null ? ms : Math.min(soonest, ms);
379
+ }
380
+ catch {
381
+ // Not a token we can read; the next match may be.
382
+ }
383
+ }
384
+ return soonest;
385
+ }
343
386
  export function formatScan(result) {
344
387
  const lines = [
345
388
  `Project: ${result.projectDir}`,
@@ -348,8 +391,11 @@ export function formatScan(result) {
348
391
  `Playwright config: ${result.hasPlaywright ? "yes" : "no"} · data-testid convention: ${result.usesTestids ? "yes" : "not detected"}`,
349
392
  `Auth storage states (${result.authStates.length}): ${result.authStates
350
393
  .slice(0, 8)
351
- .map((p) => path.basename(p))
394
+ .map((p) => `${path.basename(p)}${describeAuthAge(p)}`)
352
395
  .join(", ") || "none"}`,
396
+ ...(result.authStates.length > 0 && result.authStates.every((p) => describeAuthAge(p) === " (EXPIRED)")
397
+ ? [` ⚠ Every saved login listed here has expired. scout_attach would land on a login page; regenerate them before attaching.`]
398
+ : []),
353
399
  `Routes (${result.routes.length}):`,
354
400
  ...result.routes.slice(0, 60).map((r) => ` ${r}`),
355
401
  ...(result.routes.length > 60 ? [` … and ${result.routes.length - 60} more`] : []),
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "scenescout",
3
- "version": "3.2.0",
3
+ "version": "3.4.0",
4
4
  "description": "SceneScout — exploratory UI testing for AI coding agents. An MCP server that gives any agent (Claude Code, Cursor, VS Code Copilot, Codex, Gemini CLI and others) a structured view of a running web app, always-on oracles, a network-level write policy, memory across runs and a gap-checked report.",
5
5
  "license": "MIT",
6
6
  "author": "brunoboto96",
@@ -71,7 +71,7 @@
71
71
  "mcp-check": "npm run build && npm run mcp-check:run",
72
72
  "mcp-check:run": "tsx scripts/mcp-check.ts",
73
73
  "test": "npm run build && npm run test:unit && npm run smoke:run && npm run mcp-check:run",
74
- "test:unit": "npm run scan-test && npm run oracle-test && npm run policy-test && npm run fixture-test && npm run dispatch-test && npm run design-test && npm run contract-test && npm run lane-test && npm run memory-test && npm run install-test && npm run live-test && npm run demo-test && npm run hygiene-test",
74
+ "test:unit": "npm run scan-test && npm run oracle-test && npm run policy-test && npm run fixture-test && npm run dispatch-test && npm run design-test && npm run contract-test && npm run pace-test && npm run request-test && npm run settle-test && npm run claims-test && npm run verify-test && npm run brief-test && npm run lane-test && npm run calibration-test && npm run memory-test && npm run install-test && npm run live-test && npm run demo-test && npm run hygiene-test",
75
75
  "scan-test": "tsx scripts/scan-test.ts",
76
76
  "oracle-test": "tsx --test scripts/oracle-test.ts",
77
77
  "policy-test": "tsx --test scripts/policy-test.ts",
@@ -84,7 +84,14 @@
84
84
  "install-test": "tsx --test scripts/install-test.ts",
85
85
  "live-test": "tsx --test scripts/live-test.ts",
86
86
  "demo-test": "tsx --test scripts/demo-test.ts",
87
- "hygiene-test": "tsx --test scripts/hygiene-test.ts"
87
+ "hygiene-test": "tsx --test scripts/hygiene-test.ts",
88
+ "pace-test": "tsx --test scripts/pace-test.ts",
89
+ "request-test": "tsx --test scripts/request-test.ts",
90
+ "settle-test": "tsx --test scripts/settle-test.ts",
91
+ "claims-test": "tsx --test scripts/claims-test.ts",
92
+ "verify-test": "tsx --test scripts/verify-test.ts",
93
+ "brief-test": "tsx --test scripts/brief-test.ts",
94
+ "calibration-test": "tsx --test scripts/calibration-test.ts"
88
95
  },
89
96
  "dependencies": {
90
97
  "@modelcontextprotocol/sdk": "^1.12.0",