scenescout 3.2.0 → 3.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +91 -0
- package/README.md +3 -2
- package/dist/cli.js +39 -23
- package/dist/engine/brief.js +134 -0
- package/dist/engine/browser.js +262 -19
- package/dist/engine/calibration.js +217 -0
- package/dist/engine/claims.js +194 -0
- package/dist/engine/lane.js +20 -6
- package/dist/engine/live-page.js +101 -2
- package/dist/engine/live.js +60 -0
- package/dist/engine/memory.js +188 -5
- package/dist/engine/oracles.js +9 -0
- package/dist/engine/pace.js +112 -0
- package/dist/engine/report.js +74 -6
- package/dist/engine/request.js +144 -0
- package/dist/engine/settle.js +97 -0
- package/dist/engine/verify.js +142 -0
- package/dist/mcp-server.js +173 -15
- package/dist/scan.js +47 -1
- package/package.json +10 -3
- package/skills/scenescout/SKILL.md +10 -2
package/dist/engine/memory.js
CHANGED
|
@@ -25,6 +25,47 @@ export const FINDING_CATEGORIES = [
|
|
|
25
25
|
"missing-testid",
|
|
26
26
|
"other",
|
|
27
27
|
];
|
|
28
|
+
/**
|
|
29
|
+
* Most states a route may keep. A route accumulates one state per distinct
|
|
30
|
+
* element set, so a register with filters, tabs and paging produces dozens;
|
|
31
|
+
* one project reached 6,075 states across 48 routes, a 36 MB history parsed
|
|
32
|
+
* and re-serialised on every save. Coverage is asked per ROUTE, so keeping the
|
|
33
|
+
* most recent states of each route preserves every answer the gap ledger
|
|
34
|
+
* needs while dropping the long tail nothing will ask about again.
|
|
35
|
+
*/
|
|
36
|
+
export const MAX_STATES_PER_ROUTE = 40;
|
|
37
|
+
/**
|
|
38
|
+
* The states to keep. Never drops one a finding points at — a finding's repro
|
|
39
|
+
* trace and its route identity are read back from it — and never drops the
|
|
40
|
+
* newest of a route, so a route that was visited stays visited.
|
|
41
|
+
*
|
|
42
|
+
* Returns the pruned map rather than mutating, so the rule can be table-tested.
|
|
43
|
+
*/
|
|
44
|
+
export function pruneStates(states, findings, perRoute = MAX_STATES_PER_ROUTE) {
|
|
45
|
+
const pinned = new Set(findings.map((f) => f.state));
|
|
46
|
+
const byRoute = new Map();
|
|
47
|
+
for (const entry of Object.entries(states)) {
|
|
48
|
+
const route = entry[1].route;
|
|
49
|
+
const list = byRoute.get(route);
|
|
50
|
+
if (list)
|
|
51
|
+
list.push(entry);
|
|
52
|
+
else
|
|
53
|
+
byRoute.set(route, [entry]);
|
|
54
|
+
}
|
|
55
|
+
const kept = {};
|
|
56
|
+
let dropped = 0;
|
|
57
|
+
for (const list of byRoute.values()) {
|
|
58
|
+
// Newest first, so the survivors are the ones a next run will meet again.
|
|
59
|
+
list.sort((a, b) => (b[1].lastSeen ?? b[1].firstSeen).localeCompare(a[1].lastSeen ?? a[1].firstSeen));
|
|
60
|
+
list.forEach(([fp, rec], index) => {
|
|
61
|
+
if (index < perRoute || pinned.has(fp))
|
|
62
|
+
kept[fp] = rec;
|
|
63
|
+
else
|
|
64
|
+
dropped += 1;
|
|
65
|
+
});
|
|
66
|
+
}
|
|
67
|
+
return { kept, dropped };
|
|
68
|
+
}
|
|
28
69
|
/** An element class must appear on this many routes at minimum before it can count as shared chrome. */
|
|
29
70
|
const CHROME_MIN_ROUTES = 4;
|
|
30
71
|
/** …and on at least this share of all visited routes (a majority — "it's on every page"). */
|
|
@@ -209,8 +250,39 @@ export function mergeMemory(mine, theirs) {
|
|
|
209
250
|
}
|
|
210
251
|
if (Object.keys(out.roleAccess).length === 0)
|
|
211
252
|
delete out.roleAccess;
|
|
253
|
+
// Lanes commonly run in SEPARATE processes — that is the point of a lane —
|
|
254
|
+
// so without this the spread at the top would keep one process's decisions
|
|
255
|
+
// and drop every other lane's, which is the exact bug this merge exists to
|
|
256
|
+
// prevent for findings. Keyed so merging the same foreign document twice is
|
|
257
|
+
// idempotent.
|
|
258
|
+
const byDecision = new Map();
|
|
259
|
+
for (const d of theirs.laneDecisions ?? [])
|
|
260
|
+
byDecision.set(decisionKey(d), d);
|
|
261
|
+
for (const d of mine.laneDecisions ?? [])
|
|
262
|
+
byDecision.set(decisionKey(d), d);
|
|
263
|
+
out.laneDecisions = [...byDecision.values()].sort((a, b) => a.at.localeCompare(b.at)).slice(-MAX_LANE_DECISIONS);
|
|
264
|
+
if (out.laneDecisions.length === 0)
|
|
265
|
+
delete out.laneDecisions;
|
|
212
266
|
return out;
|
|
213
267
|
}
|
|
268
|
+
/**
|
|
269
|
+
* What makes two stored decisions the same judgement.
|
|
270
|
+
*
|
|
271
|
+
* Deliberately NOT the timestamp. `at` is stamped when the planner folds the
|
|
272
|
+
* reply, not when the lane judged, so relaying one reply twice — which the
|
|
273
|
+
* protocol invites, since a refused report is asked for again — wrote every
|
|
274
|
+
* decision a second time and doubled the lane's weight in the calibration.
|
|
275
|
+
* Identity is what was decided, so re-folding the same reply is a no-op.
|
|
276
|
+
*/
|
|
277
|
+
function decisionKey(d) {
|
|
278
|
+
return [d.lane, d.observation, d.verdict, d.severity ?? "", d.category ?? "", d.confidence, d.evidence ?? ""].join("|");
|
|
279
|
+
}
|
|
280
|
+
/**
|
|
281
|
+
* Most lane decisions kept. Calibration wants a few dozen; a long-lived
|
|
282
|
+
* project would otherwise accumulate every decision ever made and re-serialise
|
|
283
|
+
* them on each save, which is what made an old history slow to open.
|
|
284
|
+
*/
|
|
285
|
+
export const MAX_LANE_DECISIONS = 1000;
|
|
214
286
|
const MAX_DISCOVERED_ROUTES = 300;
|
|
215
287
|
/** Shared finding-similarity helpers (used by live dedup and retro-merge). */
|
|
216
288
|
function findingTokens(s) {
|
|
@@ -254,7 +326,22 @@ function findingLiterals(...texts) {
|
|
|
254
326
|
* (`/api/users/7` and `/api/users/9` are one endpoint) so a per-record repro
|
|
255
327
|
* does not read as a per-record bug.
|
|
256
328
|
*/
|
|
257
|
-
|
|
329
|
+
/**
|
|
330
|
+
* The signatures that identify a BUG: only those carrying a failure status.
|
|
331
|
+
*
|
|
332
|
+
* A bare `POST /api/orders` says which endpoint was involved, not what went
|
|
333
|
+
* wrong — a double submit and an accepted negative quantity both name it, and
|
|
334
|
+
* are two bugs. Exported so calibration applies the store's own rule instead
|
|
335
|
+
* of a second copy of this regex, which is what it had.
|
|
336
|
+
*/
|
|
337
|
+
export function failingSignatures(evidence) {
|
|
338
|
+
return new Set([...endpointSignatures(evidence)].filter((sig) => /\s[45]\d{2}$/.test(sig)));
|
|
339
|
+
}
|
|
340
|
+
/**
|
|
341
|
+
* The endpoint signatures a piece of evidence names: method + normalised path,
|
|
342
|
+
* with the failure status when one follows.
|
|
343
|
+
*/
|
|
344
|
+
export function endpointSignatures(evidence) {
|
|
258
345
|
const out = new Set();
|
|
259
346
|
if (!evidence)
|
|
260
347
|
return out;
|
|
@@ -297,11 +384,10 @@ function sharesEndpointSignature(a, b) {
|
|
|
297
384
|
// `POST /api/orders` (the endpoint answered 2xx, or no status was named)
|
|
298
385
|
// says which endpoint was involved, not what went wrong: a double submit
|
|
299
386
|
// and an accepted negative quantity both name it, and are two bugs.
|
|
300
|
-
const
|
|
301
|
-
const aSigs = failing(a.evidence);
|
|
387
|
+
const aSigs = failingSignatures(a.evidence);
|
|
302
388
|
if (aSigs.size === 0)
|
|
303
389
|
return false;
|
|
304
|
-
for (const sig of
|
|
390
|
+
for (const sig of failingSignatures(b.evidence))
|
|
305
391
|
if (aSigs.has(sig))
|
|
306
392
|
return true;
|
|
307
393
|
return false;
|
|
@@ -684,9 +770,34 @@ export class MemoryStore {
|
|
|
684
770
|
constraints: "Constraints — rules discovered the hard way (gates, preconditions, limits)",
|
|
685
771
|
risks: "Risks & watchpoints — fragile areas worth re-testing every run",
|
|
686
772
|
glossary: "Glossary — domain terms and what they mean here",
|
|
773
|
+
setup: "Setup — how to get this app into a testable state (how a login state is regenerated, what has to be running)",
|
|
687
774
|
};
|
|
688
775
|
return `## ${titles[section] ?? section}`;
|
|
689
776
|
}
|
|
777
|
+
/**
|
|
778
|
+
* What earlier runs recorded about getting this app testable, if anything.
|
|
779
|
+
*
|
|
780
|
+
* Read back by the auth-failure message. A storage state expires on a timer
|
|
781
|
+
* nobody remembers, and "regenerate it" is advice the reader already had;
|
|
782
|
+
* the command that worked last time is the part worth keeping, and it is
|
|
783
|
+
* exactly the kind of thing a run pays to find out and then forgets.
|
|
784
|
+
*/
|
|
785
|
+
setupRecipe() {
|
|
786
|
+
if (!fs.existsSync(this.assumptionsPath))
|
|
787
|
+
return [];
|
|
788
|
+
const text = fs.readFileSync(this.assumptionsPath, "utf8");
|
|
789
|
+
const heading = MemoryStore.sectionHeading("setup");
|
|
790
|
+
const start = text.indexOf(heading);
|
|
791
|
+
if (start === -1)
|
|
792
|
+
return [];
|
|
793
|
+
const rest = text.slice(start + heading.length);
|
|
794
|
+
const end = rest.indexOf("\n## ");
|
|
795
|
+
return (end === -1 ? rest : rest.slice(0, end))
|
|
796
|
+
.split("\n")
|
|
797
|
+
.map((l) => l.trim())
|
|
798
|
+
.filter((l) => l.startsWith("- "))
|
|
799
|
+
.map((l) => l.slice(2).trim());
|
|
800
|
+
}
|
|
690
801
|
readAssumptions() {
|
|
691
802
|
if (!fs.existsSync(this.assumptionsPath))
|
|
692
803
|
return "(no ASSUMPTIONS.md yet — record what you learn with scout_note as you explore)";
|
|
@@ -716,6 +827,45 @@ export class MemoryStore {
|
|
|
716
827
|
fs.writeFileSync(this.assumptionsPath, content);
|
|
717
828
|
return true;
|
|
718
829
|
}
|
|
830
|
+
/** What the lanes decided, oldest first. Empty on a project that has never run one. */
|
|
831
|
+
get laneDecisions() {
|
|
832
|
+
return this.data.laneDecisions ?? [];
|
|
833
|
+
}
|
|
834
|
+
/**
|
|
835
|
+
* Record what a lane decided. Called once per accepted lane report, so the
|
|
836
|
+
* confidence it stated can be checked later against what the run filed.
|
|
837
|
+
* Free text from the lane is redacted like every other stored string: an
|
|
838
|
+
* observation is written by a model reading the app under test.
|
|
839
|
+
*/
|
|
840
|
+
addLaneDecisions(lane, decisions) {
|
|
841
|
+
if (decisions.length === 0)
|
|
842
|
+
return 0;
|
|
843
|
+
const list = this.data.laneDecisions ?? [];
|
|
844
|
+
const seen = new Set(list.map(decisionKey));
|
|
845
|
+
let added = 0;
|
|
846
|
+
for (const d of decisions) {
|
|
847
|
+
const record = {
|
|
848
|
+
...d,
|
|
849
|
+
lane,
|
|
850
|
+
observation: redactSecrets(d.observation).slice(0, 200),
|
|
851
|
+
evidence: d.evidence === null ? null : redactSecrets(d.evidence).slice(0, 200),
|
|
852
|
+
};
|
|
853
|
+
if (seen.has(decisionKey(record)))
|
|
854
|
+
continue;
|
|
855
|
+
seen.add(decisionKey(record));
|
|
856
|
+
list.push(record);
|
|
857
|
+
added += 1;
|
|
858
|
+
}
|
|
859
|
+
this.data.laneDecisions = list.slice(-MAX_LANE_DECISIONS);
|
|
860
|
+
if (added > 0)
|
|
861
|
+
this.flush();
|
|
862
|
+
// What survives the cap. Appends go to the tail and the cap keeps the
|
|
863
|
+
// tail, so all of `added` survives unless the call itself exceeds the cap.
|
|
864
|
+
// Measuring it as growth instead looked right and was not: `list` aliases
|
|
865
|
+
// the stored array, so the "before" length was read after the appends and
|
|
866
|
+
// every call after the first reported nothing kept — while storing fine.
|
|
867
|
+
return Math.min(added, MAX_LANE_DECISIONS);
|
|
868
|
+
}
|
|
719
869
|
/** Mark a finding resolved; returns it or null. */
|
|
720
870
|
resolveFinding(id) {
|
|
721
871
|
const f = this.data.findings.find((x) => x.id === id);
|
|
@@ -725,13 +875,43 @@ export class MemoryStore {
|
|
|
725
875
|
this.flush();
|
|
726
876
|
return f;
|
|
727
877
|
}
|
|
878
|
+
/**
|
|
879
|
+
* Record what a re-test of this finding found. "gone" resolves it; the other
|
|
880
|
+
* two leave it open and stamp the confirmation, which is what lets the report
|
|
881
|
+
* stop describing a finding somebody checked yesterday as unverified.
|
|
882
|
+
*/
|
|
883
|
+
verifyFinding(id, verdict, note) {
|
|
884
|
+
const f = this.data.findings.find((x) => x.id === id);
|
|
885
|
+
if (!f)
|
|
886
|
+
return null;
|
|
887
|
+
f.verdict = verdict;
|
|
888
|
+
f.verifiedAt = new Date().toISOString();
|
|
889
|
+
if (note)
|
|
890
|
+
f.verifyNote = redactSecrets(note).slice(0, 500);
|
|
891
|
+
else
|
|
892
|
+
delete f.verifyNote;
|
|
893
|
+
if (verdict === "gone")
|
|
894
|
+
f.status = "resolved";
|
|
895
|
+
this.flush();
|
|
896
|
+
return f;
|
|
897
|
+
}
|
|
728
898
|
load() {
|
|
729
899
|
if (!fs.existsSync(this.memoryPath))
|
|
730
900
|
return structuredClone(EMPTY);
|
|
731
901
|
try {
|
|
732
902
|
const raw = JSON.parse(fs.readFileSync(this.memoryPath, "utf8"));
|
|
733
|
-
if (raw.version === 1)
|
|
903
|
+
if (raw.version === 1) {
|
|
904
|
+
// Prune once, on open: the long tail of per-route states is what makes
|
|
905
|
+
// an old history slow to parse and re-serialise, and it answers no
|
|
906
|
+
// question the gap ledger asks. Findings and the newest states of
|
|
907
|
+
// every route are kept, so coverage does not regress.
|
|
908
|
+
const { kept, dropped } = pruneStates(raw.states ?? {}, raw.findings ?? []);
|
|
909
|
+
if (dropped > 0) {
|
|
910
|
+
raw.states = kept;
|
|
911
|
+
this.prunedStates = dropped;
|
|
912
|
+
}
|
|
734
913
|
return raw;
|
|
914
|
+
}
|
|
735
915
|
this.loadWarning = `memory.json has unknown version ${String(raw.version)} — starting fresh.`;
|
|
736
916
|
}
|
|
737
917
|
catch (err) {
|
|
@@ -783,6 +963,8 @@ export class MemoryStore {
|
|
|
783
963
|
// Deliberately NOT unref'd: a pending coverage write briefly holds the
|
|
784
964
|
// process open so an exit without scout_close still lands the last save.
|
|
785
965
|
}
|
|
966
|
+
/** How many states the last open pruned. Reported once, so a shrinking history is never silent. */
|
|
967
|
+
prunedStates = 0;
|
|
786
968
|
/** Set when a debounced background write failed — cleared on the next successful write. Surfaced by scout_coverage/scout_close so a broken persistence path is never silently invisible. */
|
|
787
969
|
lastSaveError = null;
|
|
788
970
|
/**
|
|
@@ -907,6 +1089,7 @@ export class MemoryStore {
|
|
|
907
1089
|
this.data.states[fingerprint] = rec;
|
|
908
1090
|
}
|
|
909
1091
|
rec.visits += 1;
|
|
1092
|
+
rec.lastSeen = new Date().toISOString();
|
|
910
1093
|
const present = new Set(elementKeys);
|
|
911
1094
|
for (const key of elementKeys) {
|
|
912
1095
|
if (!rec.elements[key])
|
package/dist/engine/oracles.js
CHANGED
|
@@ -137,6 +137,15 @@ export class OracleMonitor {
|
|
|
137
137
|
noteInjection(detail, url) {
|
|
138
138
|
this.record({ kind: "dom_injection", severity: "high", detail, url });
|
|
139
139
|
}
|
|
140
|
+
/**
|
|
141
|
+
* The page contradicted a request that was refused during the same action:
|
|
142
|
+
* a refused list rendered as an empty state, or a refused save reported as a
|
|
143
|
+
* success. Found by the engine's DOM scan (claims.ts), like injections, so
|
|
144
|
+
* it is reported through here rather than by a page event.
|
|
145
|
+
*/
|
|
146
|
+
noteContradiction(c, url) {
|
|
147
|
+
this.record({ kind: c.kind, severity: "high", detail: c.detail, url });
|
|
148
|
+
}
|
|
140
149
|
record(v) {
|
|
141
150
|
if (isPolicyInduced(v, this.lastPolicyBlockAt === null ? null : Date.now() - this.lastPolicyBlockAt)) {
|
|
142
151
|
this.policyAttributed += 1;
|
|
@@ -0,0 +1,112 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Log entries that are not actions: a stated task, an attach, the note that a
|
|
3
|
+
* resource was created. They carry no browser work and take no time, so
|
|
4
|
+
* counting them inflates the action count and drags the median gap toward
|
|
5
|
+
* zero. One real run logged 325 entries of which 107 were stated tasks.
|
|
6
|
+
*/
|
|
7
|
+
const MARKER_ACTIONS = new Set(["task", "attach", "created-resource", "journey:start", "journey:end", "record:full", "record:failed"]);
|
|
8
|
+
/** Whether a log entry represents work the browser actually did. */
|
|
9
|
+
export function isActing(action) {
|
|
10
|
+
return !MARKER_ACTIONS.has(action);
|
|
11
|
+
}
|
|
12
|
+
/** A gap longer than this is the agent thinking, not the browser working. */
|
|
13
|
+
export const IDLE_GAP_MS = 30_000;
|
|
14
|
+
/** A session with no action for this long is probably forgotten, and is holding a browser for nothing. */
|
|
15
|
+
export const STALE_SESSION_MS = 5 * 60_000;
|
|
16
|
+
function median(sorted) {
|
|
17
|
+
if (sorted.length === 0)
|
|
18
|
+
return 0;
|
|
19
|
+
const mid = Math.floor(sorted.length / 2);
|
|
20
|
+
return sorted.length % 2 === 1 ? sorted[mid] : Math.round((sorted[mid - 1] + sorted[mid]) / 2);
|
|
21
|
+
}
|
|
22
|
+
/**
|
|
23
|
+
* The pace of each session in the log, and of the run as a whole. `nowMs` is
|
|
24
|
+
* passed rather than read so the numbers are the same every time they are
|
|
25
|
+
* computed from the same log.
|
|
26
|
+
*/
|
|
27
|
+
export function measurePace(log, nowMs, attached = []) {
|
|
28
|
+
const stillOpen = new Set(attached);
|
|
29
|
+
const bySession = new Map();
|
|
30
|
+
for (const entry of log) {
|
|
31
|
+
if (!isActing(entry.action))
|
|
32
|
+
continue;
|
|
33
|
+
const name = entry.session ?? "default";
|
|
34
|
+
const list = bySession.get(name);
|
|
35
|
+
if (list)
|
|
36
|
+
list.push(entry);
|
|
37
|
+
else
|
|
38
|
+
bySession.set(name, [entry]);
|
|
39
|
+
}
|
|
40
|
+
const sessions = [];
|
|
41
|
+
let first = Number.POSITIVE_INFINITY;
|
|
42
|
+
let last = 0;
|
|
43
|
+
for (const [session, entries] of bySession) {
|
|
44
|
+
const times = entries.map((e) => Date.parse(e.at)).filter((t) => Number.isFinite(t));
|
|
45
|
+
if (times.length === 0)
|
|
46
|
+
continue;
|
|
47
|
+
times.sort((a, b) => a - b);
|
|
48
|
+
first = Math.min(first, times[0]);
|
|
49
|
+
last = Math.max(last, times[times.length - 1]);
|
|
50
|
+
const gaps = [];
|
|
51
|
+
for (let i = 1; i < times.length; i += 1)
|
|
52
|
+
gaps.push(times[i] - times[i - 1]);
|
|
53
|
+
const spanMs = times[times.length - 1] - times[0];
|
|
54
|
+
const idleMs = gaps.filter((g) => g > IDLE_GAP_MS).reduce((sum, g) => sum + g, 0);
|
|
55
|
+
sessions.push({
|
|
56
|
+
session,
|
|
57
|
+
actions: entries.length,
|
|
58
|
+
spanMs,
|
|
59
|
+
medianGapMs: median([...gaps].sort((a, b) => a - b)),
|
|
60
|
+
maxGapMs: gaps.length > 0 ? Math.max(...gaps) : 0,
|
|
61
|
+
idleShare: spanMs > 0 ? idleMs / spanMs : 0,
|
|
62
|
+
quietMs: Math.max(0, nowMs - times[times.length - 1]),
|
|
63
|
+
framed: entries.filter((e) => e.frame).length,
|
|
64
|
+
});
|
|
65
|
+
}
|
|
66
|
+
sessions.sort((a, b) => b.actions - a.actions);
|
|
67
|
+
return {
|
|
68
|
+
sessions,
|
|
69
|
+
spanMs: sessions.length > 0 ? last - first : 0,
|
|
70
|
+
actions: sessions.reduce((sum, x) => sum + x.actions, 0),
|
|
71
|
+
// Only a session that is STILL ATTACHED can be holding a browser. A lane
|
|
72
|
+
// that finished and closed is quiet because it is gone, and warning about
|
|
73
|
+
// it told the reader to close something that no longer exists — which is
|
|
74
|
+
// what the first run of this report did for six of its eleven sessions.
|
|
75
|
+
quiet: sessions.filter((s) => s.quietMs > STALE_SESSION_MS && stillOpen.has(s.session)).map((s) => s.session),
|
|
76
|
+
};
|
|
77
|
+
}
|
|
78
|
+
/** Milliseconds as a person says them: 45s, 4m12s, 1h03m. */
|
|
79
|
+
export function sayDuration(ms) {
|
|
80
|
+
const total = Math.max(0, Math.round(ms / 1000));
|
|
81
|
+
if (total < 60)
|
|
82
|
+
return `${total}s`;
|
|
83
|
+
const minutes = Math.floor(total / 60);
|
|
84
|
+
if (minutes < 60)
|
|
85
|
+
return `${minutes}m${String(total % 60).padStart(2, "0")}s`;
|
|
86
|
+
return `${Math.floor(minutes / 60)}h${String(minutes % 60).padStart(2, "0")}m`;
|
|
87
|
+
}
|
|
88
|
+
/**
|
|
89
|
+
* The run's pace as report lines. Says what the numbers mean, because "idle
|
|
90
|
+
* 84%" reads as a fault in the engine when it is a description of how the
|
|
91
|
+
* agent paced itself.
|
|
92
|
+
*/
|
|
93
|
+
export function formatPace(pace) {
|
|
94
|
+
if (pace.sessions.length === 0)
|
|
95
|
+
return [];
|
|
96
|
+
const lines = [
|
|
97
|
+
`## How the run was paced`,
|
|
98
|
+
``,
|
|
99
|
+
`${pace.actions} action(s) over ${sayDuration(pace.spanMs)}. Stated tasks and attaches are not counted: they take no time. Idle share is time the browser stood still waiting for the agent, not time the engine spent working.`,
|
|
100
|
+
``,
|
|
101
|
+
`| Session | Actions | Span | Median gap | Longest gap | Idle | Frames |`,
|
|
102
|
+
`|---|---:|---:|---:|---:|---:|---:|`,
|
|
103
|
+
];
|
|
104
|
+
for (const s of pace.sessions) {
|
|
105
|
+
lines.push(`| ${s.session} | ${s.actions} | ${sayDuration(s.spanMs)} | ${sayDuration(s.medianGapMs)} | ${sayDuration(s.maxGapMs)} | ${Math.round(s.idleShare * 100)}% | ${s.framed} |`);
|
|
106
|
+
}
|
|
107
|
+
lines.push(``);
|
|
108
|
+
if (pace.quiet.length > 0) {
|
|
109
|
+
lines.push(`⚠ Held a browser with nothing to do for over ${sayDuration(STALE_SESSION_MS)}: ${pace.quiet.join(", ")}. A session waiting for its turn should be closed and re-attached when it is needed.`, ``);
|
|
110
|
+
}
|
|
111
|
+
return lines;
|
|
112
|
+
}
|
package/dist/engine/report.js
CHANGED
|
@@ -1,8 +1,11 @@
|
|
|
1
1
|
import fs from "node:fs";
|
|
2
2
|
import path from "node:path";
|
|
3
3
|
import { SHARED_CHROME_ROUTE } from "./memory.js";
|
|
4
|
+
import { sayVerification } from "./verify.js";
|
|
4
5
|
import { feedForSession } from "./live.js";
|
|
5
6
|
import { buildReplayHtml, evidenceFor } from "./replay.js";
|
|
7
|
+
import { calibrate, formatCalibration } from "./calibration.js";
|
|
8
|
+
import { formatPace, measurePace } from "./pace.js";
|
|
6
9
|
function playwrightSkeleton(f) {
|
|
7
10
|
const routeClass = f.state.split("#")[0].split("?")[0];
|
|
8
11
|
let gotoPath = routeClass;
|
|
@@ -68,6 +71,24 @@ function projectName(dir) {
|
|
|
68
71
|
const parent = path.dirname(dir);
|
|
69
72
|
return path.basename(parent) || path.basename(dir) || "project";
|
|
70
73
|
}
|
|
74
|
+
/**
|
|
75
|
+
* How long ago a finding was last seen, for the historical index. A finding
|
|
76
|
+
* nobody has re-confirmed in four months is a different thing from one seen
|
|
77
|
+
* last week, and the report should not make a reader open both to find out.
|
|
78
|
+
*/
|
|
79
|
+
export function describeAge(foundAt, nowMs) {
|
|
80
|
+
const seen = Date.parse(foundAt);
|
|
81
|
+
if (!Number.isFinite(seen))
|
|
82
|
+
return "?";
|
|
83
|
+
const days = Math.floor((nowMs - seen) / 86_400_000);
|
|
84
|
+
if (days <= 0)
|
|
85
|
+
return "today";
|
|
86
|
+
if (days === 1)
|
|
87
|
+
return "1 day";
|
|
88
|
+
if (days < 60)
|
|
89
|
+
return `${days} days`;
|
|
90
|
+
return `${Math.floor(days / 30)} months`;
|
|
91
|
+
}
|
|
71
92
|
/** Every session that did anything, with its steps in order — what the HTML replays. */
|
|
72
93
|
function replaySessions(memory) {
|
|
73
94
|
const names = [...new Set(memory.actionLog.map((e) => e.session ?? "default"))];
|
|
@@ -378,6 +399,7 @@ export function generateReport(memory, oracleLog, extras, opts = {}) {
|
|
|
378
399
|
const lines = [];
|
|
379
400
|
const resolved = findings.filter((f) => f.status === "resolved");
|
|
380
401
|
const open = findings.filter((f) => f.status !== "resolved");
|
|
402
|
+
const now = Date.now();
|
|
381
403
|
const current = open.filter((f) => f.foundAt >= memory.sessionStart);
|
|
382
404
|
const historical = open.filter((f) => f.foundAt < memory.sessionStart);
|
|
383
405
|
lines.push(`# SceneScout Report`);
|
|
@@ -496,6 +518,11 @@ export function generateReport(memory, oracleLog, extras, opts = {}) {
|
|
|
496
518
|
lines.push(`- **Evidence:** \`${f.evidence}\``);
|
|
497
519
|
lines.push(`- **Where:** \`${f.state}\` (${f.url})`);
|
|
498
520
|
lines.push(`- **Seen in runs:** ${f.runs}`);
|
|
521
|
+
// Only printed once somebody has re-tested it. A finding nobody has looked
|
|
522
|
+
// at again says nothing here, which is the honest thing for it to say.
|
|
523
|
+
if (f.verdict && f.verifiedAt) {
|
|
524
|
+
lines.push(`- **Re-tested:**${sayVerification(f).replace(/^ · /, " ")}${f.verifyNote ? ` — ${f.verifyNote}` : ""}`);
|
|
525
|
+
}
|
|
499
526
|
lines.push(``);
|
|
500
527
|
lines.push(f.detail);
|
|
501
528
|
lines.push(``);
|
|
@@ -521,18 +548,47 @@ export function generateReport(memory, oracleLog, extras, opts = {}) {
|
|
|
521
548
|
if (historical.length > 0) {
|
|
522
549
|
lines.push(`## Historical findings — not re-verified this session (${historical.length})`);
|
|
523
550
|
lines.push(``);
|
|
524
|
-
|
|
525
|
-
|
|
526
|
-
|
|
527
|
-
|
|
551
|
+
if (extras?.history === "full") {
|
|
552
|
+
lines.push(`Recorded in earlier runs and not re-confirmed. Re-test before acting; \`scout_verify\` hands them back in re-test order and records what each re-test found.`);
|
|
553
|
+
lines.push(``);
|
|
554
|
+
for (const f of historical)
|
|
555
|
+
renderFinding(f);
|
|
556
|
+
}
|
|
557
|
+
else {
|
|
558
|
+
// An index, not the findings themselves. One project reached 412
|
|
559
|
+
// historical findings and printing each in full made the report 1.75 MB
|
|
560
|
+
// — a document nobody opens, in which the eleven findings the run
|
|
561
|
+
// actually made were buried. Each row carries what decides whether to
|
|
562
|
+
// re-test it; the detail is one scout_report {history:"full"} away.
|
|
563
|
+
lines.push(`Recorded in earlier runs and NOT re-confirmed by this one, so none of it is evidence about the build under test. Listed as an index: ` +
|
|
564
|
+
`work it down with \`scout_verify\`, which hands back these findings in re-test order and records what each re-test found, ` +
|
|
565
|
+
`and pass \`history: "full"\` to scout_report for the full text.`, ``, `| Sev | Id | Age | Runs | Re-tested | Title |`, `|---|---|---:|---:|---|---|`);
|
|
566
|
+
for (const f of historical) {
|
|
567
|
+
const retested = f.verdict && f.verifiedAt ? `${f.verdict} ${f.verifiedAt.slice(0, 10)}` : "never";
|
|
568
|
+
lines.push(`| ${SEVERITY_ICON[f.severity]} | \`${f.id}\` | ${describeAge(f.foundAt, now)} | ${f.runs} | ${retested} | ${escapeTableCell(f.title)} |`);
|
|
569
|
+
}
|
|
570
|
+
lines.push(``);
|
|
571
|
+
}
|
|
528
572
|
}
|
|
529
573
|
if (resolved.length > 0) {
|
|
530
574
|
lines.push(`## ✅ Resolved (${resolved.length})`);
|
|
531
575
|
lines.push(``);
|
|
532
576
|
lines.push(`Fixed and verified (or confirmed no longer reproducing). A resolved finding that is re-found reopens automatically and is flagged as a regression above.`);
|
|
533
577
|
lines.push(``);
|
|
534
|
-
|
|
535
|
-
|
|
578
|
+
if (extras?.history === "full") {
|
|
579
|
+
for (const f of resolved)
|
|
580
|
+
renderFinding(f, true);
|
|
581
|
+
}
|
|
582
|
+
else {
|
|
583
|
+
// The least actionable content in the document: these are fixed. On one
|
|
584
|
+
// project they were 314 findings and 783 KB — nearly half the report,
|
|
585
|
+
// none of it anything to do. The index keeps the record without the bulk.
|
|
586
|
+
lines.push(`| Sev | Id | Fixed | Title |`, `|---|---|---:|---|`);
|
|
587
|
+
for (const f of resolved) {
|
|
588
|
+
lines.push(`| ${SEVERITY_ICON[f.severity]} | \`${f.id}\` | ${describeAge(f.foundAt, now)} ago | ${escapeTableCell(f.title)} |`);
|
|
589
|
+
}
|
|
590
|
+
lines.push(``);
|
|
591
|
+
}
|
|
536
592
|
}
|
|
537
593
|
if (extras?.createdResources && extras.createdResources.length > 0) {
|
|
538
594
|
lines.push(`## Data created by this session (cleanup list)`);
|
|
@@ -552,6 +608,18 @@ export function generateReport(memory, oracleLog, extras, opts = {}) {
|
|
|
552
608
|
}
|
|
553
609
|
lines.push(``);
|
|
554
610
|
}
|
|
611
|
+
// How the run was paced. A reader who sees a session that did four actions
|
|
612
|
+
// in an hour learns more from that than from another coverage percentage.
|
|
613
|
+
// Omitted where the report is compared byte for byte, since it is timing.
|
|
614
|
+
// The sessions still attached, so the stale-browser warning names only the
|
|
615
|
+
// ones that could act on it. A lane that finished and closed is quiet
|
|
616
|
+
// because it is gone.
|
|
617
|
+
if (extras?.pace !== false) {
|
|
618
|
+
lines.push(...formatPace(measurePace(memory.actionLog, Date.now(), extras?.attachedSessions ?? [])));
|
|
619
|
+
}
|
|
620
|
+
// Only on a run that used lanes, and only once enough of them have been
|
|
621
|
+
// judged for the number to mean anything; formatCalibration decides both.
|
|
622
|
+
lines.push(...formatCalibration(calibrate(memory.laneDecisions, memory.findings)));
|
|
555
623
|
const markdown = lines.join("\n");
|
|
556
624
|
const outPath = path.join(memory.dir, "report.md");
|
|
557
625
|
const htmlPath = path.join(memory.dir, "report.html");
|
|
@@ -0,0 +1,144 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Calling the app's own API with the UI bypassed.
|
|
3
|
+
*
|
|
4
|
+
* A refusal shown by hiding or disabling a button is not a refusal. Confirming
|
|
5
|
+
* that the server refuses the same action is the single most valuable check a
|
|
6
|
+
* permission pass makes, and until now it could only be done outside the tool
|
|
7
|
+
* — in a shell, with curl and a hand-extracted token — so none of that
|
|
8
|
+
* evidence reached the report. A whole validation run's permission matrices
|
|
9
|
+
* lived in shell history and vanished with it.
|
|
10
|
+
*
|
|
11
|
+
* The request is made by the PAGE, not beside it. That matters twice over:
|
|
12
|
+
* it goes through the same interception the write policy is enforced on, so a
|
|
13
|
+
* safe-write session cannot delete a record it does not own by calling the
|
|
14
|
+
* endpoint instead of clicking; and it carries the session's own credentials,
|
|
15
|
+
* because it is the same origin with the same cookies.
|
|
16
|
+
*
|
|
17
|
+
* Bearer schemes are handled by replaying the Authorization header the app
|
|
18
|
+
* itself last sent, which the engine already sees on every intercepted
|
|
19
|
+
* request. Nothing here knows what a token looks like or where an app keeps
|
|
20
|
+
* one, so nothing here is tuned to any app.
|
|
21
|
+
*
|
|
22
|
+
* Everything in this file is pure so it can be table-tested; the one
|
|
23
|
+
* `page.evaluate` lives in browser.ts.
|
|
24
|
+
*/
|
|
25
|
+
/** Methods a session may replay. Anything else is refused before it reaches the page. */
|
|
26
|
+
export const REPLAYABLE_METHODS = ["GET", "HEAD", "POST", "PUT", "PATCH", "DELETE", "OPTIONS"];
|
|
27
|
+
/** Most of a response body that reaches the agent. A JSON list can be megabytes; the signature is in the first lines. */
|
|
28
|
+
export const BODY_MAX = 2000;
|
|
29
|
+
/** Response headers worth reporting. "Identical response" means status, body AND headers, so the ones that commonly differ are kept. */
|
|
30
|
+
export const REPORTED_HEADERS = ["content-type", "content-length", "location", "www-authenticate", "retry-after", "x-request-id", "cache-control"];
|
|
31
|
+
/**
|
|
32
|
+
* The path to call, resolved against the attached origin and fenced to it.
|
|
33
|
+
* Navigation is fenced the same way: a run attached to one app must not be
|
|
34
|
+
* able to make its browser talk to another host just because a path was
|
|
35
|
+
* spelled as a full URL.
|
|
36
|
+
*/
|
|
37
|
+
export function resolveRequestUrl(baseUrl, path) {
|
|
38
|
+
const trimmed = path.trim();
|
|
39
|
+
if (!trimmed)
|
|
40
|
+
return { problem: "No path given. Pass a path such as /api/things, or a full URL on the attached origin." };
|
|
41
|
+
let target;
|
|
42
|
+
let base;
|
|
43
|
+
try {
|
|
44
|
+
base = new URL(baseUrl);
|
|
45
|
+
target = new URL(trimmed, base);
|
|
46
|
+
}
|
|
47
|
+
catch {
|
|
48
|
+
return { problem: `Could not read ${JSON.stringify(trimmed)} as a path or a URL.` };
|
|
49
|
+
}
|
|
50
|
+
if (target.protocol !== "http:" && target.protocol !== "https:") {
|
|
51
|
+
return { problem: `Only http and https can be requested; ${target.protocol} cannot.` };
|
|
52
|
+
}
|
|
53
|
+
if (target.origin !== base.origin) {
|
|
54
|
+
return {
|
|
55
|
+
problem: `${target.origin} is not the origin this session is attached to (${base.origin}). A session talks to its own app only; attach another session to test another host.`,
|
|
56
|
+
};
|
|
57
|
+
}
|
|
58
|
+
return { url: target.toString() };
|
|
59
|
+
}
|
|
60
|
+
/** The method, upper-cased, or the reason it cannot be replayed. */
|
|
61
|
+
export function resolveMethod(method) {
|
|
62
|
+
const upper = (method ?? "GET").trim().toUpperCase();
|
|
63
|
+
if (!REPLAYABLE_METHODS.includes(upper)) {
|
|
64
|
+
return { problem: `${upper} is not a method this can replay. Use one of: ${REPLAYABLE_METHODS.join(", ")}.` };
|
|
65
|
+
}
|
|
66
|
+
return { method: upper };
|
|
67
|
+
}
|
|
68
|
+
/**
|
|
69
|
+
* The script the page runs. Built as one expression so it can be evaluated
|
|
70
|
+
* directly, with every value passed through JSON rather than interpolated as
|
|
71
|
+
* code: a header value or a body is data from the agent, and a quote in it
|
|
72
|
+
* must not be able to end the string it sits in.
|
|
73
|
+
*
|
|
74
|
+
* `credentials: "include"` so the session's cookies go with it, and the
|
|
75
|
+
* app's own Authorization header is replayed when one has been seen.
|
|
76
|
+
*/
|
|
77
|
+
export function buildRequestScript(input) {
|
|
78
|
+
const { url, method, body, headers } = input;
|
|
79
|
+
const init = { method, credentials: "include", headers };
|
|
80
|
+
if (body !== undefined && method !== "GET" && method !== "HEAD")
|
|
81
|
+
init.body = body;
|
|
82
|
+
return (`(async () => { const started = Date.now();` +
|
|
83
|
+
` const res = await fetch(${JSON.stringify(url)}, ${JSON.stringify(init)});` +
|
|
84
|
+
` const text = await res.text();` +
|
|
85
|
+
` const headers = {}; res.headers.forEach((v, k) => { headers[k] = v; });` +
|
|
86
|
+
` return { status: res.status, statusText: res.statusText, headers, body: text.slice(0, ${BODY_MAX * 2}), full: text.length, ms: Date.now() - started, url: res.url }; })()`);
|
|
87
|
+
}
|
|
88
|
+
/** The headers to send: what the caller asked for, plus the app's own auth and a JSON content type when a body is present. */
|
|
89
|
+
export function requestHeaders(input) {
|
|
90
|
+
const out = {};
|
|
91
|
+
// The app's own header first, so an explicit one from the caller wins — that
|
|
92
|
+
// is how a session tests what happens with a different or absent credential.
|
|
93
|
+
if (input.auth)
|
|
94
|
+
out["authorization"] = input.auth;
|
|
95
|
+
if (input.body !== undefined)
|
|
96
|
+
out["content-type"] = "application/json";
|
|
97
|
+
for (const [name, value] of Object.entries(input.given ?? {}))
|
|
98
|
+
out[name.toLowerCase()] = value;
|
|
99
|
+
return out;
|
|
100
|
+
}
|
|
101
|
+
/** The raw result of the in-page fetch, reduced to what the agent and the report need. */
|
|
102
|
+
export function toReplayResult(raw) {
|
|
103
|
+
const kept = {};
|
|
104
|
+
for (const name of REPORTED_HEADERS) {
|
|
105
|
+
const value = raw.headers[name];
|
|
106
|
+
if (value !== undefined)
|
|
107
|
+
kept[name] = value;
|
|
108
|
+
}
|
|
109
|
+
return {
|
|
110
|
+
status: raw.status,
|
|
111
|
+
statusText: raw.statusText,
|
|
112
|
+
headers: kept,
|
|
113
|
+
body: raw.body.slice(0, BODY_MAX),
|
|
114
|
+
truncated: raw.full > BODY_MAX,
|
|
115
|
+
ms: raw.ms,
|
|
116
|
+
url: raw.url,
|
|
117
|
+
};
|
|
118
|
+
}
|
|
119
|
+
/** The signature a finding carries for this call: the line that dedups the same refusal across runs. */
|
|
120
|
+
export function replaySignature(method, url, status) {
|
|
121
|
+
let path = url;
|
|
122
|
+
try {
|
|
123
|
+
const parsed = new URL(url);
|
|
124
|
+
path = parsed.pathname + parsed.search;
|
|
125
|
+
}
|
|
126
|
+
catch {
|
|
127
|
+
// Not a URL we can shorten; the whole string is the signature.
|
|
128
|
+
}
|
|
129
|
+
return `${method} ${path} ${status}`;
|
|
130
|
+
}
|
|
131
|
+
/** What the agent reads back. Leads with the signature, because that is what a finding quotes. */
|
|
132
|
+
export function formatReplay(method, result) {
|
|
133
|
+
const lines = [replaySignature(method, result.url, result.status) + (result.statusText ? ` ${result.statusText}` : ""), `took ${result.ms} ms`];
|
|
134
|
+
const headers = Object.entries(result.headers);
|
|
135
|
+
if (headers.length > 0)
|
|
136
|
+
lines.push(headers.map(([k, v]) => `${k}: ${v}`).join(" · "));
|
|
137
|
+
if (result.body.length > 0) {
|
|
138
|
+
lines.push("", result.body + (result.truncated ? `\n… truncated at ${BODY_MAX} characters` : ""));
|
|
139
|
+
}
|
|
140
|
+
else {
|
|
141
|
+
lines.push("", "(empty body)");
|
|
142
|
+
}
|
|
143
|
+
return lines.join("\n");
|
|
144
|
+
}
|