scenescout 3.11.1 → 3.13.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +31 -0
- package/README.md +29 -3
- package/dist/check-run.js +4 -2
- package/dist/ci-run.js +366 -0
- package/dist/cli.js +69 -0
- package/dist/engine/bench.js +147 -14
- package/dist/engine/browser.js +63 -34
- package/dist/engine/calibration.js +23 -6
- package/dist/engine/check.js +89 -15
- package/dist/engine/ci.js +633 -0
- package/dist/engine/dedup.js +237 -0
- package/dist/engine/design.js +15 -1
- package/dist/engine/fingerprint.js +57 -0
- package/dist/engine/lane.js +57 -3
- package/dist/engine/memory.js +115 -8
- package/dist/engine/policy.js +42 -0
- package/dist/engine/provider.js +193 -0
- package/dist/engine/report.js +37 -2
- package/dist/engine/verify.js +4 -3
- package/dist/mcp-server.js +23 -9
- package/package.json +6 -3
- package/skills/scenescout/SKILL.md +4 -3
|
@@ -21,7 +21,7 @@
|
|
|
21
21
|
* Pure, so every rule here is table-tested.
|
|
22
22
|
*/
|
|
23
23
|
import { requestsPair, statedRequests, templatedPathsMatch } from "./lane.js";
|
|
24
|
-
import { failingSignatures } from "./memory.js";
|
|
24
|
+
import { failingSignatures, isWorthALook } from "./memory.js";
|
|
25
25
|
/**
|
|
26
26
|
* Upper edge of each confidence bucket. Five is enough to see a shape and few
|
|
27
27
|
* enough that each holds a usable count on a run of a few dozen decisions;
|
|
@@ -87,7 +87,8 @@ function usableConfidence(value) {
|
|
|
87
87
|
*/
|
|
88
88
|
export function calibrate(decisions, findings) {
|
|
89
89
|
const filedKeys = new Map();
|
|
90
|
-
|
|
90
|
+
// Filed as worth a look is not filed as a defect: a lane that called it a defect was not agreed with.
|
|
91
|
+
for (const f of findings.filter((x) => !isWorthALook(x))) {
|
|
91
92
|
if (!f.evidence)
|
|
92
93
|
continue;
|
|
93
94
|
// A finding verified as still present is the most informative match, so it
|
|
@@ -102,8 +103,11 @@ export function calibrate(decisions, findings) {
|
|
|
102
103
|
const claims = decisions.filter((d) => d.verdict === "defect" && d.evidence !== null);
|
|
103
104
|
const checkable = claims.filter((d) => usableConfidence(d.confidence) !== null && joinKeys(d.evidence).size > 0);
|
|
104
105
|
const unjoinable = claims.length - checkable.length;
|
|
106
|
+
const worthALook = decisions.filter((d) => d.verdict === "worth_a_look").length;
|
|
105
107
|
if (checkable.length === 0)
|
|
106
|
-
return unjoinable > 0
|
|
108
|
+
return unjoinable > 0 || worthALook > 0
|
|
109
|
+
? { checkable: 0, filed: 0, stated: 0, buckets: [], ece: 0, verified: { present: 0, gone: 0, changed: 0 }, unjoinable, worthALook }
|
|
110
|
+
: null;
|
|
107
111
|
const buckets = BUCKET_EDGES.map(() => ({ n: 0, conf: 0, hits: 0 }));
|
|
108
112
|
const verified = { present: 0, gone: 0, changed: 0 };
|
|
109
113
|
const matched = new Map();
|
|
@@ -147,7 +151,7 @@ export function calibrate(decisions, findings) {
|
|
|
147
151
|
ece += (b.n / n) * Math.abs(meanConf - rate);
|
|
148
152
|
out.push({ label: bucketLabel(i), decisions: b.n, stated: meanConf, filed: rate });
|
|
149
153
|
});
|
|
150
|
-
return { checkable: n, filed, stated: stated / n, buckets: out, ece, verified, unjoinable };
|
|
154
|
+
return { checkable: n, filed, stated: stated / n, buckets: out, ece, verified, unjoinable, worthALook };
|
|
151
155
|
}
|
|
152
156
|
const pct = (x) => `${Math.round(x * 100)}%`;
|
|
153
157
|
/**
|
|
@@ -163,7 +167,7 @@ export function formatCalibration(c) {
|
|
|
163
167
|
// and a run that recorded none look identical when the section simply
|
|
164
168
|
// vanishes, and the reader concludes the feature is broken — which is this
|
|
165
169
|
// project's own definition of a silent path.
|
|
166
|
-
const had = c.checkable + c.unjoinable;
|
|
170
|
+
const had = c.checkable + c.unjoinable + c.worthALook;
|
|
167
171
|
if (had === 0)
|
|
168
172
|
return [];
|
|
169
173
|
return [
|
|
@@ -171,6 +175,7 @@ export function formatCalibration(c) {
|
|
|
171
175
|
``,
|
|
172
176
|
`Not enough to say yet: ${c.checkable} lane decision(s) could be checked${c.unjoinable > 0 ? ` (and ${c.unjoinable} could not be looked up at all)` : ""}, ` +
|
|
173
177
|
`and ${MIN_FOR_A_VERDICT} are needed before a calibration figure survives one of them changing.`,
|
|
178
|
+
...worthALookNote(c),
|
|
174
179
|
``,
|
|
175
180
|
];
|
|
176
181
|
}
|
|
@@ -194,6 +199,7 @@ export function formatCalibration(c) {
|
|
|
194
199
|
if (c.unjoinable > 0) {
|
|
195
200
|
lines.push(``, `${c.unjoinable} further decision(s) called a defect but named no failing endpoint, so nothing could be looked up for them. They are excluded above rather than counted as wrong.`);
|
|
196
201
|
}
|
|
202
|
+
lines.push(...worthALookNote(c));
|
|
197
203
|
const seen = c.verified.present + c.verified.gone + c.verified.changed;
|
|
198
204
|
if (seen > 0) {
|
|
199
205
|
lines.push(``, `${seen} of the findings those decisions matched ${seen === 1 ? "has" : "have"} since been re-tested with \`scout_verify\`: ${c.verified.present} still present, ${c.verified.gone} gone, ${c.verified.changed} changed. ` +
|
|
@@ -202,6 +208,16 @@ export function formatCalibration(c) {
|
|
|
202
208
|
lines.push(``);
|
|
203
209
|
return lines;
|
|
204
210
|
}
|
|
211
|
+
/** The line saying how many "worth a look" decisions were left out, and why; nothing when there were none. */
|
|
212
|
+
function worthALookNote(c) {
|
|
213
|
+
if (c.worthALook === 0)
|
|
214
|
+
return [];
|
|
215
|
+
return [
|
|
216
|
+
``,
|
|
217
|
+
`${c.worthALook} decision(s) were marked worth a look: real, and a defect only under a convention of the project the run cannot see. ` +
|
|
218
|
+
`Whether one was filed says nothing about whether the lane was right, so they are not scored.`,
|
|
219
|
+
];
|
|
220
|
+
}
|
|
205
221
|
/** Below this, the number swings on a single decision and is worse than no number. */
|
|
206
222
|
export const MIN_FOR_A_VERDICT = 8;
|
|
207
223
|
/**
|
|
@@ -259,7 +275,8 @@ export const MAX_UNFILED_NAMED = 10;
|
|
|
259
275
|
export function unfiledDefects(decisions, findings) {
|
|
260
276
|
const keys = [];
|
|
261
277
|
const filed = [];
|
|
262
|
-
|
|
278
|
+
// A defect filed only as worth a look is not in the report's findings, so it is still unfiled.
|
|
279
|
+
for (const f of findings.filter((x) => !isWorthALook(x))) {
|
|
263
280
|
const text = `${f.evidence ?? ""} ${f.title}`;
|
|
264
281
|
const requests = statedRequests(f.evidence ?? "");
|
|
265
282
|
// The store's signatures, and the finding's own failing requests as
|
package/dist/engine/check.js
CHANGED
|
@@ -81,7 +81,31 @@ export const CHECK_RULES = {
|
|
|
81
81
|
help: "A step of a flow saved in .scenescout/flows could not be done, or what it expected was not there. The evidence names the flow, the step and what happened instead.",
|
|
82
82
|
},
|
|
83
83
|
};
|
|
84
|
-
|
|
84
|
+
/**
|
|
85
|
+
* Rules whose measurement is exact and whose meaning depends on a convention
|
|
86
|
+
* of the project the check cannot see. SceneScout is used against any app, so
|
|
87
|
+
* it does not decide those conventions: these are listed as "worth a look",
|
|
88
|
+
* each naming the convention that would make it a defect, and never counted,
|
|
89
|
+
* given a severity or gated on, at any --fail-on. SARIF reports them at level
|
|
90
|
+
* "note". `convention` finishes the sentence "a defect only if your project
|
|
91
|
+
* uses …". --ignore takes them like any other rule.
|
|
92
|
+
*/
|
|
93
|
+
export const WORTH_A_LOOK_RULES = {
|
|
94
|
+
"off-grid-spacing": {
|
|
95
|
+
title: "Spacing off a 4px grid",
|
|
96
|
+
help: "More than a fifth of the page's paddings or vertical margins are not multiples of 4px. That matters where a project keeps a 4px spacing scale, and not where it uses another scale or none.",
|
|
97
|
+
convention: "a 4px spacing scale",
|
|
98
|
+
},
|
|
99
|
+
"indistinct-link": {
|
|
100
|
+
title: "Link styled like body text",
|
|
101
|
+
help: "Links with no underline, in the same colour as the page's body text. In running text a reader cannot tell them from the text around them; in navigation this styling is common, and the check cannot tell the two apart.",
|
|
102
|
+
convention: "a visible link style (an underline or a distinct colour) wherever links appear, navigation included",
|
|
103
|
+
},
|
|
104
|
+
};
|
|
105
|
+
export const CHECK_RULE_IDS = [...Object.keys(CHECK_RULES), ...Object.keys(WORTH_A_LOOK_RULES)];
|
|
106
|
+
export function isWorthALookRule(rule) {
|
|
107
|
+
return Object.prototype.hasOwnProperty.call(WORTH_A_LOOK_RULES, rule);
|
|
108
|
+
}
|
|
85
109
|
const SEVERITY_RANK = { high: 0, medium: 1, low: 2 };
|
|
86
110
|
/** Snapshot refs (`e12`) are numbered per run; evidence carrying them would never match itself twice. */
|
|
87
111
|
function stripRefs(line) {
|
|
@@ -145,27 +169,36 @@ export function geometryRule(line) {
|
|
|
145
169
|
* oracles caught while it ran. The second kind goes through the same rules as
|
|
146
170
|
* a crawled page's, so a request that fails on load and again inside a flow is
|
|
147
171
|
* one issue, not two.
|
|
172
|
+
*
|
|
173
|
+
* A fact under a worth-a-look rule goes to `worthALook` instead, deduplicated
|
|
174
|
+
* the same way; the two lists never share an entry.
|
|
148
175
|
*/
|
|
149
|
-
export function
|
|
176
|
+
export function checkFindings(routes, origin, ignore = [], flows = []) {
|
|
150
177
|
const byKey = new Map();
|
|
178
|
+
const looks = new Map();
|
|
151
179
|
const add = (rule, evidence, route, opts = {}) => {
|
|
152
180
|
if (ignore.includes(rule))
|
|
153
181
|
return;
|
|
154
182
|
const clean = redactSecrets(withoutOrigin(evidence, origin)).slice(0, 300);
|
|
155
183
|
const key = `${rule}\u0000${clean}`;
|
|
156
|
-
const found = byKey.get(key);
|
|
184
|
+
const found = byKey.get(key) ?? looks.get(key);
|
|
157
185
|
if (found) {
|
|
158
186
|
if (!found.routes.includes(route))
|
|
159
187
|
found.routes.push(route);
|
|
160
188
|
return;
|
|
161
189
|
}
|
|
190
|
+
const fingerprint = createHash("sha256").update(key).digest("hex").slice(0, 32);
|
|
191
|
+
if (isWorthALookRule(rule)) {
|
|
192
|
+
looks.set(key, { rule, evidence: clean, routes: [route], convention: WORTH_A_LOOK_RULES[rule].convention, fingerprint });
|
|
193
|
+
return;
|
|
194
|
+
}
|
|
162
195
|
byKey.set(key, {
|
|
163
196
|
rule,
|
|
164
197
|
severity: opts.embed ? "medium" : (opts.severity ?? CHECK_RULES[rule].severity),
|
|
165
198
|
evidence: clean,
|
|
166
199
|
routes: [route],
|
|
167
200
|
...(opts.embed ? { embed: opts.embed } : {}),
|
|
168
|
-
fingerprint
|
|
201
|
+
fingerprint,
|
|
169
202
|
});
|
|
170
203
|
};
|
|
171
204
|
for (const r of routes) {
|
|
@@ -206,7 +239,14 @@ export function issuesFromRoutes(routes, origin, ignore = [], flows = []) {
|
|
|
206
239
|
if (f.outcome.status === "failed")
|
|
207
240
|
add("flow-step-failed", flowStepEvidence(f), f.outcome.path);
|
|
208
241
|
}
|
|
209
|
-
return
|
|
242
|
+
return {
|
|
243
|
+
issues: [...byKey.values()].sort((a, b) => SEVERITY_RANK[a.severity] - SEVERITY_RANK[b.severity] || a.rule.localeCompare(b.rule)),
|
|
244
|
+
worthALook: [...looks.values()].sort((a, b) => a.rule.localeCompare(b.rule)),
|
|
245
|
+
};
|
|
246
|
+
}
|
|
247
|
+
/** The defect tier of `checkFindings`: what counts, and what the gate reads. */
|
|
248
|
+
export function issuesFromRoutes(routes, origin, ignore = [], flows = []) {
|
|
249
|
+
return checkFindings(routes, origin, ignore, flows).issues;
|
|
210
250
|
}
|
|
211
251
|
/**
|
|
212
252
|
* Why a check has nothing to give a verdict on, or null when it measured the
|
|
@@ -383,7 +423,7 @@ export function parseCheckArgs(args, cwd) {
|
|
|
383
423
|
if (paths?.some((p) => !p.startsWith("/")))
|
|
384
424
|
return { ok: false, error: "--paths are paths on the app, each starting with /" };
|
|
385
425
|
const ignore = list(flags.get("ignore")) ?? [];
|
|
386
|
-
const unknownRules = ignore.filter((r) => !(r
|
|
426
|
+
const unknownRules = ignore.filter((r) => !CHECK_RULE_IDS.includes(r));
|
|
387
427
|
if (unknownRules.length > 0)
|
|
388
428
|
return { ok: false, error: `unknown rule(s) in --ignore: ${unknownRules.join(", ")}. Rules: ${CHECK_RULE_IDS.join(", ")}` };
|
|
389
429
|
const retest = flags.get("retest") ?? "on";
|
|
@@ -489,8 +529,17 @@ const RETEST_SARIF_RULE = {
|
|
|
489
529
|
help: { text: "A finding an earlier run filed and left open failed the same way when its page loaded again. It gates according to --gate-retests." },
|
|
490
530
|
defaultConfiguration: { level: "error" },
|
|
491
531
|
};
|
|
532
|
+
function sarifLocation(route) {
|
|
533
|
+
return route === "(shared chrome)"
|
|
534
|
+
? {
|
|
535
|
+
physicalLocation: { artifactLocation: { uri: "", uriBaseId: "APP" } },
|
|
536
|
+
message: { text: "the app's shared shell, on every page that renders it" },
|
|
537
|
+
}
|
|
538
|
+
: { physicalLocation: { artifactLocation: { uri: route.replace(/^\//, ""), uriBaseId: "APP" } } };
|
|
539
|
+
}
|
|
492
540
|
export function toSarif(result, toolVersion) {
|
|
493
541
|
const used = [...new Set(result.issues.map((i) => i.rule))];
|
|
542
|
+
const lookRules = [...new Set(result.worthALook.map((o) => o.rule))];
|
|
494
543
|
const base = new URL(result.url);
|
|
495
544
|
const reproducing = retestGateFailures(result);
|
|
496
545
|
const refused = result.flows.filter((f) => f.outcome.status === "refused");
|
|
@@ -500,14 +549,20 @@ export function toSarif(result, toolVersion) {
|
|
|
500
549
|
message: {
|
|
501
550
|
text: `${CHECK_RULES[i.rule].title}: ${i.evidence}${i.routes.length > 1 ? ` (on ${i.routes.length} routes)` : ""}${i.embed ? ` (in an embed of ${i.embed})` : ""}`,
|
|
502
551
|
},
|
|
503
|
-
locations: i.routes.slice(0, 10).map(
|
|
504
|
-
? {
|
|
505
|
-
physicalLocation: { artifactLocation: { uri: "", uriBaseId: "APP" } },
|
|
506
|
-
message: { text: "the app's shared shell, on every page that renders it" },
|
|
507
|
-
}
|
|
508
|
-
: { physicalLocation: { artifactLocation: { uri: r.replace(/^\//, ""), uriBaseId: "APP" } } }),
|
|
552
|
+
locations: i.routes.slice(0, 10).map(sarifLocation),
|
|
509
553
|
partialFingerprints: { "scenescoutCheck/v1": i.fingerprint },
|
|
510
554
|
}));
|
|
555
|
+
// Always "note", whatever --fail-on says: a result a code-scanning dashboard shows, never one that reads as an error.
|
|
556
|
+
const lookResults = result.worthALook.map((o) => ({
|
|
557
|
+
ruleId: o.rule,
|
|
558
|
+
level: "note",
|
|
559
|
+
message: {
|
|
560
|
+
text: `Worth a look — ${WORTH_A_LOOK_RULES[o.rule].title}: ${o.evidence}${o.routes.length > 1 ? ` (on ${o.routes.length} routes)` : ""}. A defect only if your project uses ${o.convention}.`,
|
|
561
|
+
},
|
|
562
|
+
locations: o.routes.slice(0, 10).map(sarifLocation),
|
|
563
|
+
partialFingerprints: { "scenescoutCheck/v1": o.fingerprint },
|
|
564
|
+
properties: { tier: "worth-a-look", convention: o.convention },
|
|
565
|
+
}));
|
|
511
566
|
// Only the re-tests that fail the gate are results: a reviewer reading code scanning sees what failed it.
|
|
512
567
|
const retestResults = reproducing.map((r) => ({
|
|
513
568
|
ruleId: RETEST_SARIF_RULE.id,
|
|
@@ -537,6 +592,14 @@ export function toSarif(result, toolVersion) {
|
|
|
537
592
|
help: { text: CHECK_RULES[id].help },
|
|
538
593
|
defaultConfiguration: { level: SARIF_LEVEL[CHECK_RULES[id].severity] },
|
|
539
594
|
})),
|
|
595
|
+
...lookRules.map((id) => ({
|
|
596
|
+
id,
|
|
597
|
+
name: WORTH_A_LOOK_RULES[id].title,
|
|
598
|
+
shortDescription: { text: WORTH_A_LOOK_RULES[id].title },
|
|
599
|
+
help: { text: `${WORTH_A_LOOK_RULES[id].help} Worth a look: a defect only if your project uses ${WORTH_A_LOOK_RULES[id].convention}.` },
|
|
600
|
+
defaultConfiguration: { level: "note" },
|
|
601
|
+
properties: { tags: ["worth-a-look"] },
|
|
602
|
+
})),
|
|
540
603
|
...(reproducing.length > 0 ? [RETEST_SARIF_RULE] : []),
|
|
541
604
|
],
|
|
542
605
|
},
|
|
@@ -553,7 +616,7 @@ export function toSarif(result, toolVersion) {
|
|
|
553
616
|
},
|
|
554
617
|
],
|
|
555
618
|
originalUriBaseIds: { APP: { uri: `${base.origin}/` } },
|
|
556
|
-
results: [...issueResults, ...retestResults],
|
|
619
|
+
results: [...issueResults, ...lookResults, ...retestResults],
|
|
557
620
|
},
|
|
558
621
|
],
|
|
559
622
|
};
|
|
@@ -567,8 +630,10 @@ function cell(text) {
|
|
|
567
630
|
return text
|
|
568
631
|
.replace(/\\/g, "\\\\")
|
|
569
632
|
.replace(/\|/g, "\\|")
|
|
570
|
-
.replace(/\s
|
|
633
|
+
.replace(/\s*[\r\n]\s*/g, " ");
|
|
571
634
|
}
|
|
635
|
+
/** The same escaping, for other markdown tables: the CI run's summary. */
|
|
636
|
+
export const markdownCell = cell;
|
|
572
637
|
/** A code span that the text cannot close: its fence is one backtick longer than any run of backticks inside it. */
|
|
573
638
|
function code(text) {
|
|
574
639
|
const longest = Math.max(0, ...(text.match(/`+/g) ?? []).map((run) => run.length));
|
|
@@ -596,7 +661,8 @@ export function formatCheck(result) {
|
|
|
596
661
|
(passed
|
|
597
662
|
? `${couldNotRun > 0 ? "passed" : "**PASSED**"} — ${counts.high} high · ${counts.medium} medium · ${counts.low} low`
|
|
598
663
|
: `${couldNotRun > 0 ? "failed" : "**FAILED**"} — ${failingText} · ${counts.high} high · ${counts.medium} medium · ${counts.low} low`) +
|
|
599
|
-
(unaudited > 0 ? ` · design not measured on ${unaudited} route(s)` : "")
|
|
664
|
+
(unaudited > 0 ? ` · design not measured on ${unaudited} route(s)` : "") +
|
|
665
|
+
(result.worthALook.length > 0 ? ` · ${result.worthALook.length} worth a look, never gated` : ""));
|
|
600
666
|
// Right under the verdict: what a green check was allowed to do is part of what it means.
|
|
601
667
|
lines.push("", `Settings — ${describeSettings(result)}`);
|
|
602
668
|
for (const sev of CHECK_SEVERITIES) {
|
|
@@ -608,6 +674,12 @@ export function formatCheck(result) {
|
|
|
608
674
|
lines.push(`- **${CHECK_RULES[i.rule].title}** \`${i.rule}\`: ${cell(i.evidence)}${i.embed ? ` _(in an embed of ${i.embed}: its behaviour, not the app's)_` : ""} — ${routeList(i.routes)}`);
|
|
609
675
|
}
|
|
610
676
|
}
|
|
677
|
+
if (result.worthALook.length > 0) {
|
|
678
|
+
lines.push("", `## Worth a look (${result.worthALook.length})`, "", "Measured exactly, and defects only under a convention of your project that the check cannot see. They are not counted above and never fail the gate, at any --fail-on.", "");
|
|
679
|
+
for (const o of result.worthALook) {
|
|
680
|
+
lines.push(`- **${WORTH_A_LOOK_RULES[o.rule].title}** \`${o.rule}\`: ${cell(o.evidence)} — a defect only if your project uses ${o.convention} — ${routeList(o.routes)}`);
|
|
681
|
+
}
|
|
682
|
+
}
|
|
611
683
|
lines.push("", "## Routes", "", "| Route | Status | Controls | Issues |", "|---|---|---|---|");
|
|
612
684
|
for (const r of result.routes) {
|
|
613
685
|
const n = result.issues.filter((i) => i.routes.includes(r.path)).length;
|
|
@@ -701,5 +773,7 @@ export function toSummaryJson(result, toolVersion) {
|
|
|
701
773
|
skippedFlows: result.skippedFlows,
|
|
702
774
|
retest: result.retest,
|
|
703
775
|
issues: result.issues,
|
|
776
|
+
// Apart from `issues` and `counts`, which the gate reads: none of these is counted or gated.
|
|
777
|
+
worthALook: result.worthALook,
|
|
704
778
|
};
|
|
705
779
|
}
|