scenescout 3.2.0 → 3.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -25,6 +25,47 @@ export const FINDING_CATEGORIES = [
25
25
  "missing-testid",
26
26
  "other",
27
27
  ];
28
+ /**
29
+ * Most states a route may keep. A route accumulates one state per distinct
30
+ * element set, so a register with filters, tabs and paging produces dozens;
31
+ * one project reached 6,075 states across 48 routes, a 36 MB history parsed
32
+ * and re-serialised on every save. Coverage is asked per ROUTE, so keeping the
33
+ * most recent states of each route preserves every answer the gap ledger
34
+ * needs while dropping the long tail nothing will ask about again.
35
+ */
36
+ export const MAX_STATES_PER_ROUTE = 40;
37
+ /**
38
+ * The states to keep. Never drops one a finding points at — a finding's repro
39
+ * trace and its route identity are read back from it — and never drops the
40
+ * newest of a route, so a route that was visited stays visited.
41
+ *
42
+ * Returns the pruned map rather than mutating, so the rule can be table-tested.
43
+ */
44
+ export function pruneStates(states, findings, perRoute = MAX_STATES_PER_ROUTE) {
45
+ const pinned = new Set(findings.map((f) => f.state));
46
+ const byRoute = new Map();
47
+ for (const entry of Object.entries(states)) {
48
+ const route = entry[1].route;
49
+ const list = byRoute.get(route);
50
+ if (list)
51
+ list.push(entry);
52
+ else
53
+ byRoute.set(route, [entry]);
54
+ }
55
+ const kept = {};
56
+ let dropped = 0;
57
+ for (const list of byRoute.values()) {
58
+ // Newest first, so the survivors are the ones a next run will meet again.
59
+ list.sort((a, b) => (b[1].lastSeen ?? b[1].firstSeen).localeCompare(a[1].lastSeen ?? a[1].firstSeen));
60
+ list.forEach(([fp, rec], index) => {
61
+ if (index < perRoute || pinned.has(fp))
62
+ kept[fp] = rec;
63
+ else
64
+ dropped += 1;
65
+ });
66
+ }
67
+ return { kept, dropped };
68
+ }
28
69
  /** An element class must appear on this many routes at minimum before it can count as shared chrome. */
29
70
  const CHROME_MIN_ROUTES = 4;
30
71
  /** …and on at least this share of all visited routes (a majority — "it's on every page"). */
@@ -209,8 +250,39 @@ export function mergeMemory(mine, theirs) {
209
250
  }
210
251
  if (Object.keys(out.roleAccess).length === 0)
211
252
  delete out.roleAccess;
253
+ // Lanes commonly run in SEPARATE processes — that is the point of a lane —
254
+ // so without this the spread at the top would keep one process's decisions
255
+ // and drop every other lane's, which is the exact bug this merge exists to
256
+ // prevent for findings. Keyed so merging the same foreign document twice is
257
+ // idempotent.
258
+ const byDecision = new Map();
259
+ for (const d of theirs.laneDecisions ?? [])
260
+ byDecision.set(decisionKey(d), d);
261
+ for (const d of mine.laneDecisions ?? [])
262
+ byDecision.set(decisionKey(d), d);
263
+ out.laneDecisions = [...byDecision.values()].sort((a, b) => a.at.localeCompare(b.at)).slice(-MAX_LANE_DECISIONS);
264
+ if (out.laneDecisions.length === 0)
265
+ delete out.laneDecisions;
212
266
  return out;
213
267
  }
268
+ /**
269
+ * What makes two stored decisions the same judgement.
270
+ *
271
+ * Deliberately NOT the timestamp. `at` is stamped when the planner folds the
272
+ * reply, not when the lane judged, so relaying one reply twice — which the
273
+ * protocol invites, since a refused report is asked for again — wrote every
274
+ * decision a second time and doubled the lane's weight in the calibration.
275
+ * Identity is what was decided, so re-folding the same reply is a no-op.
276
+ */
277
+ function decisionKey(d) {
278
+ return [d.lane, d.observation, d.verdict, d.severity ?? "", d.category ?? "", d.confidence, d.evidence ?? ""].join("|");
279
+ }
280
+ /**
281
+ * Most lane decisions kept. Calibration wants a few dozen; a long-lived
282
+ * project would otherwise accumulate every decision ever made and re-serialise
283
+ * them on each save, which is what made an old history slow to open.
284
+ */
285
+ export const MAX_LANE_DECISIONS = 1000;
214
286
  const MAX_DISCOVERED_ROUTES = 300;
215
287
  /** Shared finding-similarity helpers (used by live dedup and retro-merge). */
216
288
  function findingTokens(s) {
@@ -254,7 +326,22 @@ function findingLiterals(...texts) {
254
326
  * (`/api/users/7` and `/api/users/9` are one endpoint) so a per-record repro
255
327
  * does not read as a per-record bug.
256
328
  */
257
- function endpointSignatures(evidence) {
329
+ /**
330
+ * The signatures that identify a BUG: only those carrying a failure status.
331
+ *
332
+ * A bare `POST /api/orders` says which endpoint was involved, not what went
333
+ * wrong — a double submit and an accepted negative quantity both name it, and
334
+ * are two bugs. Exported so calibration applies the store's own rule instead
335
+ * of a second copy of this regex, which is what it had.
336
+ */
337
+ export function failingSignatures(evidence) {
338
+ return new Set([...endpointSignatures(evidence)].filter((sig) => /\s[45]\d{2}$/.test(sig)));
339
+ }
340
+ /**
341
+ * The endpoint signatures a piece of evidence names: method + normalised path,
342
+ * with the failure status when one follows.
343
+ */
344
+ export function endpointSignatures(evidence) {
258
345
  const out = new Set();
259
346
  if (!evidence)
260
347
  return out;
@@ -297,11 +384,10 @@ function sharesEndpointSignature(a, b) {
297
384
  // `POST /api/orders` (the endpoint answered 2xx, or no status was named)
298
385
  // says which endpoint was involved, not what went wrong: a double submit
299
386
  // and an accepted negative quantity both name it, and are two bugs.
300
- const failing = (evidence) => new Set([...endpointSignatures(evidence)].filter((sig) => /\s[45]\d{2}$/.test(sig)));
301
- const aSigs = failing(a.evidence);
387
+ const aSigs = failingSignatures(a.evidence);
302
388
  if (aSigs.size === 0)
303
389
  return false;
304
- for (const sig of failing(b.evidence))
390
+ for (const sig of failingSignatures(b.evidence))
305
391
  if (aSigs.has(sig))
306
392
  return true;
307
393
  return false;
@@ -684,9 +770,34 @@ export class MemoryStore {
684
770
  constraints: "Constraints — rules discovered the hard way (gates, preconditions, limits)",
685
771
  risks: "Risks & watchpoints — fragile areas worth re-testing every run",
686
772
  glossary: "Glossary — domain terms and what they mean here",
773
+ setup: "Setup — how to get this app into a testable state (how a login state is regenerated, what has to be running)",
687
774
  };
688
775
  return `## ${titles[section] ?? section}`;
689
776
  }
777
+ /**
778
+ * What earlier runs recorded about getting this app testable, if anything.
779
+ *
780
+ * Read back by the auth-failure message. A storage state expires on a timer
781
+ * nobody remembers, and "regenerate it" is advice the reader already had;
782
+ * the command that worked last time is the part worth keeping, and it is
783
+ * exactly the kind of thing a run pays to find out and then forgets.
784
+ */
785
+ setupRecipe() {
786
+ if (!fs.existsSync(this.assumptionsPath))
787
+ return [];
788
+ const text = fs.readFileSync(this.assumptionsPath, "utf8");
789
+ const heading = MemoryStore.sectionHeading("setup");
790
+ const start = text.indexOf(heading);
791
+ if (start === -1)
792
+ return [];
793
+ const rest = text.slice(start + heading.length);
794
+ const end = rest.indexOf("\n## ");
795
+ return (end === -1 ? rest : rest.slice(0, end))
796
+ .split("\n")
797
+ .map((l) => l.trim())
798
+ .filter((l) => l.startsWith("- "))
799
+ .map((l) => l.slice(2).trim());
800
+ }
690
801
  readAssumptions() {
691
802
  if (!fs.existsSync(this.assumptionsPath))
692
803
  return "(no ASSUMPTIONS.md yet — record what you learn with scout_note as you explore)";
@@ -716,6 +827,45 @@ export class MemoryStore {
716
827
  fs.writeFileSync(this.assumptionsPath, content);
717
828
  return true;
718
829
  }
830
+ /** What the lanes decided, oldest first. Empty on a project that has never run one. */
831
+ get laneDecisions() {
832
+ return this.data.laneDecisions ?? [];
833
+ }
834
+ /**
835
+ * Record what a lane decided. Called once per accepted lane report, so the
836
+ * confidence it stated can be checked later against what the run filed.
837
+ * Free text from the lane is redacted like every other stored string: an
838
+ * observation is written by a model reading the app under test.
839
+ */
840
+ addLaneDecisions(lane, decisions) {
841
+ if (decisions.length === 0)
842
+ return 0;
843
+ const list = this.data.laneDecisions ?? [];
844
+ const seen = new Set(list.map(decisionKey));
845
+ let added = 0;
846
+ for (const d of decisions) {
847
+ const record = {
848
+ ...d,
849
+ lane,
850
+ observation: redactSecrets(d.observation).slice(0, 200),
851
+ evidence: d.evidence === null ? null : redactSecrets(d.evidence).slice(0, 200),
852
+ };
853
+ if (seen.has(decisionKey(record)))
854
+ continue;
855
+ seen.add(decisionKey(record));
856
+ list.push(record);
857
+ added += 1;
858
+ }
859
+ this.data.laneDecisions = list.slice(-MAX_LANE_DECISIONS);
860
+ if (added > 0)
861
+ this.flush();
862
+ // What survives the cap. Appends go to the tail and the cap keeps the
863
+ // tail, so all of `added` survives unless the call itself exceeds the cap.
864
+ // Measuring it as growth instead looked right and was not: `list` aliases
865
+ // the stored array, so the "before" length was read after the appends and
866
+ // every call after the first reported nothing kept — while storing fine.
867
+ return Math.min(added, MAX_LANE_DECISIONS);
868
+ }
719
869
  /** Mark a finding resolved; returns it or null. */
720
870
  resolveFinding(id) {
721
871
  const f = this.data.findings.find((x) => x.id === id);
@@ -725,13 +875,43 @@ export class MemoryStore {
725
875
  this.flush();
726
876
  return f;
727
877
  }
878
+ /**
879
+ * Record what a re-test of this finding found. "gone" resolves it; the other
880
+ * two leave it open and stamp the confirmation, which is what lets the report
881
+ * stop describing a finding somebody checked yesterday as unverified.
882
+ */
883
+ verifyFinding(id, verdict, note) {
884
+ const f = this.data.findings.find((x) => x.id === id);
885
+ if (!f)
886
+ return null;
887
+ f.verdict = verdict;
888
+ f.verifiedAt = new Date().toISOString();
889
+ if (note)
890
+ f.verifyNote = redactSecrets(note).slice(0, 500);
891
+ else
892
+ delete f.verifyNote;
893
+ if (verdict === "gone")
894
+ f.status = "resolved";
895
+ this.flush();
896
+ return f;
897
+ }
728
898
  load() {
729
899
  if (!fs.existsSync(this.memoryPath))
730
900
  return structuredClone(EMPTY);
731
901
  try {
732
902
  const raw = JSON.parse(fs.readFileSync(this.memoryPath, "utf8"));
733
- if (raw.version === 1)
903
+ if (raw.version === 1) {
904
+ // Prune once, on open: the long tail of per-route states is what makes
905
+ // an old history slow to parse and re-serialise, and it answers no
906
+ // question the gap ledger asks. Findings and the newest states of
907
+ // every route are kept, so coverage does not regress.
908
+ const { kept, dropped } = pruneStates(raw.states ?? {}, raw.findings ?? []);
909
+ if (dropped > 0) {
910
+ raw.states = kept;
911
+ this.prunedStates = dropped;
912
+ }
734
913
  return raw;
914
+ }
735
915
  this.loadWarning = `memory.json has unknown version ${String(raw.version)} — starting fresh.`;
736
916
  }
737
917
  catch (err) {
@@ -783,6 +963,8 @@ export class MemoryStore {
783
963
  // Deliberately NOT unref'd: a pending coverage write briefly holds the
784
964
  // process open so an exit without scout_close still lands the last save.
785
965
  }
966
+ /** How many states the last open pruned. Reported once, so a shrinking history is never silent. */
967
+ prunedStates = 0;
786
968
  /** Set when a debounced background write failed — cleared on the next successful write. Surfaced by scout_coverage/scout_close so a broken persistence path is never silently invisible. */
787
969
  lastSaveError = null;
788
970
  /**
@@ -907,6 +1089,7 @@ export class MemoryStore {
907
1089
  this.data.states[fingerprint] = rec;
908
1090
  }
909
1091
  rec.visits += 1;
1092
+ rec.lastSeen = new Date().toISOString();
910
1093
  const present = new Set(elementKeys);
911
1094
  for (const key of elementKeys) {
912
1095
  if (!rec.elements[key])
@@ -137,6 +137,15 @@ export class OracleMonitor {
137
137
  noteInjection(detail, url) {
138
138
  this.record({ kind: "dom_injection", severity: "high", detail, url });
139
139
  }
140
+ /**
141
+ * The page contradicted a request that was refused during the same action:
142
+ * a refused list rendered as an empty state, or a refused save reported as a
143
+ * success. Found by the engine's DOM scan (claims.ts), like injections, so
144
+ * it is reported through here rather than by a page event.
145
+ */
146
+ noteContradiction(c, url) {
147
+ this.record({ kind: c.kind, severity: "high", detail: c.detail, url });
148
+ }
140
149
  record(v) {
141
150
  if (isPolicyInduced(v, this.lastPolicyBlockAt === null ? null : Date.now() - this.lastPolicyBlockAt)) {
142
151
  this.policyAttributed += 1;
@@ -0,0 +1,112 @@
1
+ /**
2
+ * Log entries that are not actions: a stated task, an attach, the note that a
3
+ * resource was created. They carry no browser work and take no time, so
4
+ * counting them inflates the action count and drags the median gap toward
5
+ * zero. One real run logged 325 entries of which 107 were stated tasks.
6
+ */
7
+ const MARKER_ACTIONS = new Set(["task", "attach", "created-resource", "journey:start", "journey:end", "record:full", "record:failed"]);
8
+ /** Whether a log entry represents work the browser actually did. */
9
+ export function isActing(action) {
10
+ return !MARKER_ACTIONS.has(action);
11
+ }
12
+ /** A gap longer than this is the agent thinking, not the browser working. */
13
+ export const IDLE_GAP_MS = 30_000;
14
+ /** A session with no action for this long is probably forgotten, and is holding a browser for nothing. */
15
+ export const STALE_SESSION_MS = 5 * 60_000;
16
+ function median(sorted) {
17
+ if (sorted.length === 0)
18
+ return 0;
19
+ const mid = Math.floor(sorted.length / 2);
20
+ return sorted.length % 2 === 1 ? sorted[mid] : Math.round((sorted[mid - 1] + sorted[mid]) / 2);
21
+ }
22
+ /**
23
+ * The pace of each session in the log, and of the run as a whole. `nowMs` is
24
+ * passed rather than read so the numbers are the same every time they are
25
+ * computed from the same log.
26
+ */
27
+ export function measurePace(log, nowMs, attached = []) {
28
+ const stillOpen = new Set(attached);
29
+ const bySession = new Map();
30
+ for (const entry of log) {
31
+ if (!isActing(entry.action))
32
+ continue;
33
+ const name = entry.session ?? "default";
34
+ const list = bySession.get(name);
35
+ if (list)
36
+ list.push(entry);
37
+ else
38
+ bySession.set(name, [entry]);
39
+ }
40
+ const sessions = [];
41
+ let first = Number.POSITIVE_INFINITY;
42
+ let last = 0;
43
+ for (const [session, entries] of bySession) {
44
+ const times = entries.map((e) => Date.parse(e.at)).filter((t) => Number.isFinite(t));
45
+ if (times.length === 0)
46
+ continue;
47
+ times.sort((a, b) => a - b);
48
+ first = Math.min(first, times[0]);
49
+ last = Math.max(last, times[times.length - 1]);
50
+ const gaps = [];
51
+ for (let i = 1; i < times.length; i += 1)
52
+ gaps.push(times[i] - times[i - 1]);
53
+ const spanMs = times[times.length - 1] - times[0];
54
+ const idleMs = gaps.filter((g) => g > IDLE_GAP_MS).reduce((sum, g) => sum + g, 0);
55
+ sessions.push({
56
+ session,
57
+ actions: entries.length,
58
+ spanMs,
59
+ medianGapMs: median([...gaps].sort((a, b) => a - b)),
60
+ maxGapMs: gaps.length > 0 ? Math.max(...gaps) : 0,
61
+ idleShare: spanMs > 0 ? idleMs / spanMs : 0,
62
+ quietMs: Math.max(0, nowMs - times[times.length - 1]),
63
+ framed: entries.filter((e) => e.frame).length,
64
+ });
65
+ }
66
+ sessions.sort((a, b) => b.actions - a.actions);
67
+ return {
68
+ sessions,
69
+ spanMs: sessions.length > 0 ? last - first : 0,
70
+ actions: sessions.reduce((sum, x) => sum + x.actions, 0),
71
+ // Only a session that is STILL ATTACHED can be holding a browser. A lane
72
+ // that finished and closed is quiet because it is gone, and warning about
73
+ // it told the reader to close something that no longer exists — which is
74
+ // what the first run of this report did for six of its eleven sessions.
75
+ quiet: sessions.filter((s) => s.quietMs > STALE_SESSION_MS && stillOpen.has(s.session)).map((s) => s.session),
76
+ };
77
+ }
78
+ /** Milliseconds as a person says them: 45s, 4m12s, 1h03m. */
79
+ export function sayDuration(ms) {
80
+ const total = Math.max(0, Math.round(ms / 1000));
81
+ if (total < 60)
82
+ return `${total}s`;
83
+ const minutes = Math.floor(total / 60);
84
+ if (minutes < 60)
85
+ return `${minutes}m${String(total % 60).padStart(2, "0")}s`;
86
+ return `${Math.floor(minutes / 60)}h${String(minutes % 60).padStart(2, "0")}m`;
87
+ }
88
+ /**
89
+ * The run's pace as report lines. Says what the numbers mean, because "idle
90
+ * 84%" reads as a fault in the engine when it is a description of how the
91
+ * agent paced itself.
92
+ */
93
+ export function formatPace(pace) {
94
+ if (pace.sessions.length === 0)
95
+ return [];
96
+ const lines = [
97
+ `## How the run was paced`,
98
+ ``,
99
+ `${pace.actions} action(s) over ${sayDuration(pace.spanMs)}. Stated tasks and attaches are not counted: they take no time. Idle share is time the browser stood still waiting for the agent, not time the engine spent working.`,
100
+ ``,
101
+ `| Session | Actions | Span | Median gap | Longest gap | Idle | Frames |`,
102
+ `|---|---:|---:|---:|---:|---:|---:|`,
103
+ ];
104
+ for (const s of pace.sessions) {
105
+ lines.push(`| ${s.session} | ${s.actions} | ${sayDuration(s.spanMs)} | ${sayDuration(s.medianGapMs)} | ${sayDuration(s.maxGapMs)} | ${Math.round(s.idleShare * 100)}% | ${s.framed} |`);
106
+ }
107
+ lines.push(``);
108
+ if (pace.quiet.length > 0) {
109
+ lines.push(`⚠ Held a browser with nothing to do for over ${sayDuration(STALE_SESSION_MS)}: ${pace.quiet.join(", ")}. A session waiting for its turn should be closed and re-attached when it is needed.`, ``);
110
+ }
111
+ return lines;
112
+ }
@@ -1,8 +1,11 @@
1
1
  import fs from "node:fs";
2
2
  import path from "node:path";
3
3
  import { SHARED_CHROME_ROUTE } from "./memory.js";
4
+ import { sayVerification } from "./verify.js";
4
5
  import { feedForSession } from "./live.js";
5
6
  import { buildReplayHtml, evidenceFor } from "./replay.js";
7
+ import { calibrate, formatCalibration } from "./calibration.js";
8
+ import { formatPace, measurePace } from "./pace.js";
6
9
  function playwrightSkeleton(f) {
7
10
  const routeClass = f.state.split("#")[0].split("?")[0];
8
11
  let gotoPath = routeClass;
@@ -68,6 +71,24 @@ function projectName(dir) {
68
71
  const parent = path.dirname(dir);
69
72
  return path.basename(parent) || path.basename(dir) || "project";
70
73
  }
74
+ /**
75
+ * How long ago a finding was last seen, for the historical index. A finding
76
+ * nobody has re-confirmed in four months is a different thing from one seen
77
+ * last week, and the report should not make a reader open both to find out.
78
+ */
79
+ export function describeAge(foundAt, nowMs) {
80
+ const seen = Date.parse(foundAt);
81
+ if (!Number.isFinite(seen))
82
+ return "?";
83
+ const days = Math.floor((nowMs - seen) / 86_400_000);
84
+ if (days <= 0)
85
+ return "today";
86
+ if (days === 1)
87
+ return "1 day";
88
+ if (days < 60)
89
+ return `${days} days`;
90
+ return `${Math.floor(days / 30)} months`;
91
+ }
71
92
  /** Every session that did anything, with its steps in order — what the HTML replays. */
72
93
  function replaySessions(memory) {
73
94
  const names = [...new Set(memory.actionLog.map((e) => e.session ?? "default"))];
@@ -378,6 +399,7 @@ export function generateReport(memory, oracleLog, extras, opts = {}) {
378
399
  const lines = [];
379
400
  const resolved = findings.filter((f) => f.status === "resolved");
380
401
  const open = findings.filter((f) => f.status !== "resolved");
402
+ const now = Date.now();
381
403
  const current = open.filter((f) => f.foundAt >= memory.sessionStart);
382
404
  const historical = open.filter((f) => f.foundAt < memory.sessionStart);
383
405
  lines.push(`# SceneScout Report`);
@@ -496,6 +518,11 @@ export function generateReport(memory, oracleLog, extras, opts = {}) {
496
518
  lines.push(`- **Evidence:** \`${f.evidence}\``);
497
519
  lines.push(`- **Where:** \`${f.state}\` (${f.url})`);
498
520
  lines.push(`- **Seen in runs:** ${f.runs}`);
521
+ // Only printed once somebody has re-tested it. A finding nobody has looked
522
+ // at again says nothing here, which is the honest thing for it to say.
523
+ if (f.verdict && f.verifiedAt) {
524
+ lines.push(`- **Re-tested:**${sayVerification(f).replace(/^ · /, " ")}${f.verifyNote ? ` — ${f.verifyNote}` : ""}`);
525
+ }
499
526
  lines.push(``);
500
527
  lines.push(f.detail);
501
528
  lines.push(``);
@@ -521,18 +548,47 @@ export function generateReport(memory, oracleLog, extras, opts = {}) {
521
548
  if (historical.length > 0) {
522
549
  lines.push(`## Historical findings — not re-verified this session (${historical.length})`);
523
550
  lines.push(``);
524
- lines.push(`Recorded in earlier runs and not re-confirmed. Re-test before acting; resolve fixed ones with \`scout_resolve <id>\`.`);
525
- lines.push(``);
526
- for (const f of historical)
527
- renderFinding(f);
551
+ if (extras?.history === "full") {
552
+ lines.push(`Recorded in earlier runs and not re-confirmed. Re-test before acting; \`scout_verify\` hands them back in re-test order and records what each re-test found.`);
553
+ lines.push(``);
554
+ for (const f of historical)
555
+ renderFinding(f);
556
+ }
557
+ else {
558
+ // An index, not the findings themselves. One project reached 412
559
+ // historical findings and printing each in full made the report 1.75 MB
560
+ // — a document nobody opens, in which the eleven findings the run
561
+ // actually made were buried. Each row carries what decides whether to
562
+ // re-test it; the detail is one scout_report {history:"full"} away.
563
+ lines.push(`Recorded in earlier runs and NOT re-confirmed by this one, so none of it is evidence about the build under test. Listed as an index: ` +
564
+ `work it down with \`scout_verify\`, which hands back these findings in re-test order and records what each re-test found, ` +
565
+ `and pass \`history: "full"\` to scout_report for the full text.`, ``, `| Sev | Id | Age | Runs | Re-tested | Title |`, `|---|---|---:|---:|---|---|`);
566
+ for (const f of historical) {
567
+ const retested = f.verdict && f.verifiedAt ? `${f.verdict} ${f.verifiedAt.slice(0, 10)}` : "never";
568
+ lines.push(`| ${SEVERITY_ICON[f.severity]} | \`${f.id}\` | ${describeAge(f.foundAt, now)} | ${f.runs} | ${retested} | ${escapeTableCell(f.title)} |`);
569
+ }
570
+ lines.push(``);
571
+ }
528
572
  }
529
573
  if (resolved.length > 0) {
530
574
  lines.push(`## ✅ Resolved (${resolved.length})`);
531
575
  lines.push(``);
532
576
  lines.push(`Fixed and verified (or confirmed no longer reproducing). A resolved finding that is re-found reopens automatically and is flagged as a regression above.`);
533
577
  lines.push(``);
534
- for (const f of resolved)
535
- renderFinding(f, true);
578
+ if (extras?.history === "full") {
579
+ for (const f of resolved)
580
+ renderFinding(f, true);
581
+ }
582
+ else {
583
+ // The least actionable content in the document: these are fixed. On one
584
+ // project they were 314 findings and 783 KB — nearly half the report,
585
+ // none of it anything to do. The index keeps the record without the bulk.
586
+ lines.push(`| Sev | Id | Fixed | Title |`, `|---|---|---:|---|`);
587
+ for (const f of resolved) {
588
+ lines.push(`| ${SEVERITY_ICON[f.severity]} | \`${f.id}\` | ${describeAge(f.foundAt, now)} ago | ${escapeTableCell(f.title)} |`);
589
+ }
590
+ lines.push(``);
591
+ }
536
592
  }
537
593
  if (extras?.createdResources && extras.createdResources.length > 0) {
538
594
  lines.push(`## Data created by this session (cleanup list)`);
@@ -552,6 +608,18 @@ export function generateReport(memory, oracleLog, extras, opts = {}) {
552
608
  }
553
609
  lines.push(``);
554
610
  }
611
+ // How the run was paced. A reader who sees a session that did four actions
612
+ // in an hour learns more from that than from another coverage percentage.
613
+ // Omitted where the report is compared byte for byte, since it is timing.
614
+ // The sessions still attached, so the stale-browser warning names only the
615
+ // ones that could act on it. A lane that finished and closed is quiet
616
+ // because it is gone.
617
+ if (extras?.pace !== false) {
618
+ lines.push(...formatPace(measurePace(memory.actionLog, Date.now(), extras?.attachedSessions ?? [])));
619
+ }
620
+ // Only on a run that used lanes, and only once enough of them have been
621
+ // judged for the number to mean anything; formatCalibration decides both.
622
+ lines.push(...formatCalibration(calibrate(memory.laneDecisions, memory.findings)));
555
623
  const markdown = lines.join("\n");
556
624
  const outPath = path.join(memory.dir, "report.md");
557
625
  const htmlPath = path.join(memory.dir, "report.html");
@@ -0,0 +1,144 @@
1
+ /**
2
+ * Calling the app's own API with the UI bypassed.
3
+ *
4
+ * A refusal shown by hiding or disabling a button is not a refusal. Confirming
5
+ * that the server refuses the same action is the single most valuable check a
6
+ * permission pass makes, and until now it could only be done outside the tool
7
+ * — in a shell, with curl and a hand-extracted token — so none of that
8
+ * evidence reached the report. A whole validation run's permission matrices
9
+ * lived in shell history and vanished with it.
10
+ *
11
+ * The request is made by the PAGE, not beside it. That matters twice over:
12
+ * it goes through the same interception the write policy is enforced on, so a
13
+ * safe-write session cannot delete a record it does not own by calling the
14
+ * endpoint instead of clicking; and it carries the session's own credentials,
15
+ * because it is the same origin with the same cookies.
16
+ *
17
+ * Bearer schemes are handled by replaying the Authorization header the app
18
+ * itself last sent, which the engine already sees on every intercepted
19
+ * request. Nothing here knows what a token looks like or where an app keeps
20
+ * one, so nothing here is tuned to any app.
21
+ *
22
+ * Everything in this file is pure so it can be table-tested; the one
23
+ * `page.evaluate` lives in browser.ts.
24
+ */
25
+ /** Methods a session may replay. Anything else is refused before it reaches the page. */
26
+ export const REPLAYABLE_METHODS = ["GET", "HEAD", "POST", "PUT", "PATCH", "DELETE", "OPTIONS"];
27
+ /** Most of a response body that reaches the agent. A JSON list can be megabytes; the signature is in the first lines. */
28
+ export const BODY_MAX = 2000;
29
+ /** Response headers worth reporting. "Identical response" means status, body AND headers, so the ones that commonly differ are kept. */
30
+ export const REPORTED_HEADERS = ["content-type", "content-length", "location", "www-authenticate", "retry-after", "x-request-id", "cache-control"];
31
+ /**
32
+ * The path to call, resolved against the attached origin and fenced to it.
33
+ * Navigation is fenced the same way: a run attached to one app must not be
34
+ * able to make its browser talk to another host just because a path was
35
+ * spelled as a full URL.
36
+ */
37
+ export function resolveRequestUrl(baseUrl, path) {
38
+ const trimmed = path.trim();
39
+ if (!trimmed)
40
+ return { problem: "No path given. Pass a path such as /api/things, or a full URL on the attached origin." };
41
+ let target;
42
+ let base;
43
+ try {
44
+ base = new URL(baseUrl);
45
+ target = new URL(trimmed, base);
46
+ }
47
+ catch {
48
+ return { problem: `Could not read ${JSON.stringify(trimmed)} as a path or a URL.` };
49
+ }
50
+ if (target.protocol !== "http:" && target.protocol !== "https:") {
51
+ return { problem: `Only http and https can be requested; ${target.protocol} cannot.` };
52
+ }
53
+ if (target.origin !== base.origin) {
54
+ return {
55
+ problem: `${target.origin} is not the origin this session is attached to (${base.origin}). A session talks to its own app only; attach another session to test another host.`,
56
+ };
57
+ }
58
+ return { url: target.toString() };
59
+ }
60
+ /** The method, upper-cased, or the reason it cannot be replayed. */
61
+ export function resolveMethod(method) {
62
+ const upper = (method ?? "GET").trim().toUpperCase();
63
+ if (!REPLAYABLE_METHODS.includes(upper)) {
64
+ return { problem: `${upper} is not a method this can replay. Use one of: ${REPLAYABLE_METHODS.join(", ")}.` };
65
+ }
66
+ return { method: upper };
67
+ }
68
+ /**
69
+ * The script the page runs. Built as one expression so it can be evaluated
70
+ * directly, with every value passed through JSON rather than interpolated as
71
+ * code: a header value or a body is data from the agent, and a quote in it
72
+ * must not be able to end the string it sits in.
73
+ *
74
+ * `credentials: "include"` so the session's cookies go with it, and the
75
+ * app's own Authorization header is replayed when one has been seen.
76
+ */
77
+ export function buildRequestScript(input) {
78
+ const { url, method, body, headers } = input;
79
+ const init = { method, credentials: "include", headers };
80
+ if (body !== undefined && method !== "GET" && method !== "HEAD")
81
+ init.body = body;
82
+ return (`(async () => { const started = Date.now();` +
83
+ ` const res = await fetch(${JSON.stringify(url)}, ${JSON.stringify(init)});` +
84
+ ` const text = await res.text();` +
85
+ ` const headers = {}; res.headers.forEach((v, k) => { headers[k] = v; });` +
86
+ ` return { status: res.status, statusText: res.statusText, headers, body: text.slice(0, ${BODY_MAX * 2}), full: text.length, ms: Date.now() - started, url: res.url }; })()`);
87
+ }
88
+ /** The headers to send: what the caller asked for, plus the app's own auth and a JSON content type when a body is present. */
89
+ export function requestHeaders(input) {
90
+ const out = {};
91
+ // The app's own header first, so an explicit one from the caller wins — that
92
+ // is how a session tests what happens with a different or absent credential.
93
+ if (input.auth)
94
+ out["authorization"] = input.auth;
95
+ if (input.body !== undefined)
96
+ out["content-type"] = "application/json";
97
+ for (const [name, value] of Object.entries(input.given ?? {}))
98
+ out[name.toLowerCase()] = value;
99
+ return out;
100
+ }
101
+ /** The raw result of the in-page fetch, reduced to what the agent and the report need. */
102
+ export function toReplayResult(raw) {
103
+ const kept = {};
104
+ for (const name of REPORTED_HEADERS) {
105
+ const value = raw.headers[name];
106
+ if (value !== undefined)
107
+ kept[name] = value;
108
+ }
109
+ return {
110
+ status: raw.status,
111
+ statusText: raw.statusText,
112
+ headers: kept,
113
+ body: raw.body.slice(0, BODY_MAX),
114
+ truncated: raw.full > BODY_MAX,
115
+ ms: raw.ms,
116
+ url: raw.url,
117
+ };
118
+ }
119
+ /** The signature a finding carries for this call: the line that dedups the same refusal across runs. */
120
+ export function replaySignature(method, url, status) {
121
+ let path = url;
122
+ try {
123
+ const parsed = new URL(url);
124
+ path = parsed.pathname + parsed.search;
125
+ }
126
+ catch {
127
+ // Not a URL we can shorten; the whole string is the signature.
128
+ }
129
+ return `${method} ${path} ${status}`;
130
+ }
131
+ /** What the agent reads back. Leads with the signature, because that is what a finding quotes. */
132
+ export function formatReplay(method, result) {
133
+ const lines = [replaySignature(method, result.url, result.status) + (result.statusText ? ` ${result.statusText}` : ""), `took ${result.ms} ms`];
134
+ const headers = Object.entries(result.headers);
135
+ if (headers.length > 0)
136
+ lines.push(headers.map(([k, v]) => `${k}: ${v}`).join(" · "));
137
+ if (result.body.length > 0) {
138
+ lines.push("", result.body + (result.truncated ? `\n… truncated at ${BODY_MAX} characters` : ""));
139
+ }
140
+ else {
141
+ lines.push("", "(empty body)");
142
+ }
143
+ return lines.join("\n");
144
+ }