scenescout 3.14.0 → 3.15.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  /**
2
- * Measuring the finding-dedup decision against the answer keys, and an
3
- * optional model-backed judge to measure beside it.
2
+ * Measuring the finding-dedup decision against the answer keys, and the
3
+ * model-backed judge the store can ask beside its rule.
4
4
  *
5
5
  * The store's dedup rule (memory.ts, isDuplicateFinding) decides whether a
6
6
  * newly filed finding is one it already records. Its mistakes are costly both
@@ -12,14 +12,19 @@
12
12
  *
13
13
  * The model judge asks for a verdict (same / different / unsure) and a
14
14
  * probability through one tool call, so its answer is structured, not parsed
15
- * from prose. It is measured here and not used by the store: the rule stays
16
- * the only thing that dedups until the judge's Brier beats it on these pairs.
17
- * Everything here is pure; the network is the caller's, through `ask`.
15
+ * from prose. Measured on these pairs it beat the rule on both apps
16
+ * (docs/benchmark.md), and every pair it won was a merge the rule missed, so
17
+ * the store asks it only about filings the rule keeps apart (DedupJudge,
18
+ * below): a `scenescout ci` run by default, the MCP server when told to.
19
+ * Nothing here touches the network; the caller's `ask` does.
18
20
  */
19
21
  import { createHash } from "node:crypto";
20
22
  import { archiveApp, classify, findingText } from "./bench.js";
21
23
  import { MIN_FOR_A_VERDICT } from "./calibration.js";
22
- import { findingId, isDuplicateFinding } from "./memory.js";
24
+ import { addUsage, dedupModeFromEnv, judgeKeyConfig, NO_USAGE } from "./ci.js";
25
+ import { withWatchdog } from "./dispatch.js";
26
+ import { findingId, isDuplicateFinding, titleSimilarity } from "./memory.js";
27
+ import { obj } from "./provider.js";
23
28
  const pagesOf = (e) => [e.route, ...(e.alsoOn ?? [])];
24
29
  const norm = (s) => (s ?? "").toLowerCase().replace(/\s+/g, " ").trim();
25
30
  /**
@@ -178,10 +183,12 @@ export function parseJudgement(turn) {
178
183
  * Decide one pair. With no judge the rule decides, as the store does. With a
179
184
  * judge, the model decides when it answers same or different; when it fails
180
185
  * for any reason, or is unsure, the rule decides and the note says why, so a
181
- * run never silently becomes a run of the rule.
186
+ * run never silently becomes a run of the rule. `ruleVerdict` is the rule's
187
+ * verdict when the caller already has it: the store asks only about pairs its
188
+ * rule keeps apart, and its rule reads more of a finding than a pair carries.
182
189
  */
183
- export async function decideDuplicate(p, ask) {
184
- const rule = () => ruleJudgement(p).verdict === "same";
190
+ export async function decideDuplicate(p, ask, ruleVerdict) {
191
+ const rule = ruleVerdict ?? (() => ruleJudgement(p).verdict === "same");
185
192
  if (!ask)
186
193
  return { duplicate: rule(), by: "rule" };
187
194
  let parsed;
@@ -203,6 +210,319 @@ export async function decideDuplicate(p, ask) {
203
210
  return { duplicate: rule(), by: "rule", judgement: parsed.judgement, note: "the model judge was unsure; the current rule decided" };
204
211
  return { duplicate: parsed.judgement.verdict === "same", by: "model", judgement: parsed.judgement };
205
212
  }
213
+ // ── the judge the store asks ────────────────────────────────────────────────
214
+ /** Calls made about one filing at most: its most alike open findings on the page. */
215
+ export const JUDGE_MAX_CALLS = 3;
216
+ /** The longest one judge call may take. */
217
+ export const JUDGE_CALL_MS = 15_000;
218
+ /**
219
+ * The longest the judge may spend on one filing, every call together. Under
220
+ * scout_finding's 60-second watchdog, so a slow provider costs a filing its
221
+ * judge and never the filing.
222
+ */
223
+ export const JUDGE_FILING_MS = 40_000;
224
+ /** Failed calls in a row after which the judge is switched off for the rest of the run. */
225
+ export const JUDGE_MAX_FAILURES = 3;
226
+ /** The most output one judge call may produce: the answer is one short tool call (24 to 26 tokens when measured). */
227
+ export const JUDGE_MAX_OUTPUT_TOKENS = 2_000;
228
+ /**
229
+ * The most of a title, and of evidence, the store's judge shows the model. A
230
+ * filing with a pasted log for evidence still makes a question of bounded
231
+ * size, which a client answering the judge accepts (MAX_JUDGE_KICKOFF_CHARS).
232
+ */
233
+ export const JUDGE_TITLE_CHARS = 500;
234
+ export const JUDGE_EVIDENCE_CHARS = 2_000;
235
+ const routeOf = (state) => state.split("#")[0];
236
+ /** A judge time limit as a log line says it: "15s", or "20ms" for a test's. */
237
+ export const durationText = (ms) => (ms < 1000 ? `${ms}ms` : `${Math.round(ms / 1000)}s`);
238
+ /**
239
+ * The model judge as the store asks it (memory.ts DuplicateJudge). It is
240
+ * asked only about a filing the rule keeps apart from everything stored, and
241
+ * only against the open findings on the filing's own page: the most alike
242
+ * title first, at most `maxCalls` of them, within `filingMs`. "Same" merges;
243
+ * anything else (different, unsure, a contradiction, a failed or slow call)
244
+ * leaves the rule's decision, which is to keep the filing apart. So the judge
245
+ * only ever adds merges the rule missed, which is where every pair it won in
246
+ * the measurement was (docs/benchmark.md).
247
+ *
248
+ * Each kind of fall-back is logged once, the cap once per page, and
249
+ * JUDGE_MAX_FAILURES failed calls in a row switch the judge off for the rest
250
+ * of the run. `log` reaches the operator; the caller redacts keys from it.
251
+ */
252
+ export class DedupJudge {
253
+ ask;
254
+ o;
255
+ counts = { calls: 0, same: 0, different: 0, fellBack: 0, capped: 0, usage: { ...NO_USAGE }, ms: 0 };
256
+ failuresInARow = 0;
257
+ offReason = null;
258
+ logged = new Set();
259
+ maxCalls;
260
+ callMs;
261
+ filingMs;
262
+ now;
263
+ redact;
264
+ constructor(ask, o) {
265
+ this.ask = ask;
266
+ this.o = o;
267
+ this.maxCalls = o.maxCalls ?? JUDGE_MAX_CALLS;
268
+ this.callMs = o.callMs ?? JUDGE_CALL_MS;
269
+ this.filingMs = o.filingMs ?? JUDGE_FILING_MS;
270
+ this.now = o.now ?? Date.now;
271
+ this.redact = o.redact ?? ((text) => text);
272
+ }
273
+ /** What the judge has done this run. */
274
+ get tally() {
275
+ return this.counts;
276
+ }
277
+ /** Why the judge is no longer asked, or null while it is. */
278
+ get off() {
279
+ return this.offReason;
280
+ }
281
+ async judge(incoming, stored) {
282
+ try {
283
+ return await this.decide(incoming, stored);
284
+ }
285
+ catch (err) {
286
+ // decideDuplicate catches whatever a call throws, so this is a fault in this
287
+ // class. It would recur on every filing, so the judge stops here, and says why;
288
+ // the log gets the stack, the report only the message.
289
+ const message = err instanceof Error ? err.message : String(err);
290
+ this.switchOff(`a fault in the judge: ${message}`, err instanceof Error ? err.stack : undefined);
291
+ return null;
292
+ }
293
+ }
294
+ async decide(incoming, stored) {
295
+ if (this.offReason)
296
+ return null;
297
+ const route = routeOf(incoming.state);
298
+ const candidates = stored
299
+ .map((x, i) => ({ x, i, alike: titleSimilarity(x.title, incoming.title) }))
300
+ .filter(({ x }) => x.status !== "resolved" && routeOf(x.state) === route)
301
+ // The most alike first; of two equally alike, the newer.
302
+ .sort((p, q) => q.alike - p.alike || q.i - p.i);
303
+ if (candidates.length > this.maxCalls) {
304
+ this.counts.capped += 1;
305
+ this.once(`cap\u0000${route}`, `dedup judge: ${candidates.length} open findings on ${route} could be the one just filed there; the judge is asked about the ` +
306
+ `${this.maxCalls} most alike, and the rule, which kept it apart from all of them, decides the rest ` +
307
+ `(at most ${this.maxCalls} calls per filing; logged once per page)`);
308
+ }
309
+ const started = this.now();
310
+ for (const { x } of candidates.slice(0, this.maxCalls)) {
311
+ if (this.offReason)
312
+ break;
313
+ const left = this.filingMs - (this.now() - started);
314
+ if (left <= 0) {
315
+ this.once("time", `dedup judge: the ${durationText(this.filingMs)} a filing may take ran out; the rule decided the pairs left (logged once)`);
316
+ break;
317
+ }
318
+ const d = await decideDuplicate({ route, a: shown(x), b: shown(incoming) }, this.timed(Math.min(this.callMs, left)), () => false);
319
+ this.count(d);
320
+ if (d.by === "model" && d.duplicate && d.judgement)
321
+ return { sameAs: x.id, pSame: d.judgement.pSame };
322
+ }
323
+ return null;
324
+ }
325
+ /** The ask, ended at `ms` and counted. */
326
+ timed(ms) {
327
+ return async (system, tools, kickoff) => {
328
+ const started = this.now();
329
+ this.counts.calls += 1;
330
+ try {
331
+ const turn = await withWatchdog("dedup judge", this.ask(system, tools, kickoff, ms), ms, () => null);
332
+ if (!turn)
333
+ throw new Error(`no answer within ${durationText(ms)}`);
334
+ this.counts.usage = addUsage(this.counts.usage, turn.usage);
335
+ return turn;
336
+ }
337
+ finally {
338
+ this.counts.ms += this.now() - started;
339
+ }
340
+ };
341
+ }
342
+ count(d) {
343
+ if (d.by === "model") {
344
+ this.failuresInARow = 0;
345
+ if (d.duplicate)
346
+ this.counts.same += 1;
347
+ else
348
+ this.counts.different += 1;
349
+ return;
350
+ }
351
+ this.counts.fellBack += 1;
352
+ if (d.failure !== "error") {
353
+ // Unsure, or an answer that contradicted itself: the provider answered, so neither counts towards switching off.
354
+ this.failuresInARow = 0;
355
+ this.once(d.failure ?? "unsure", `dedup judge: ${d.note ?? "no usable answer"} (later ones are counted, not logged)`);
356
+ return;
357
+ }
358
+ this.failuresInARow += 1;
359
+ this.once("error", `dedup judge: ${d.note ?? "a call failed"} (later failures are counted, not logged)`);
360
+ if (this.failuresInARow >= JUDGE_MAX_FAILURES)
361
+ this.switchOff(`${this.failuresInARow} failed calls in a row, the last: ${d.note ?? "no detail"}`);
362
+ }
363
+ switchOff(reason, detail) {
364
+ this.offReason = this.redact(reason);
365
+ this.o.log(this.redact(`dedup judge: switched off after ${reason}; the rule decides every filing for the rest of this run${detail ? `\n${detail}` : ""}`));
366
+ }
367
+ once(kind, line) {
368
+ if (this.logged.has(kind))
369
+ return;
370
+ this.logged.add(kind);
371
+ this.o.log(this.redact(line));
372
+ }
373
+ describe() {
374
+ const t = this.counts;
375
+ if (t.calls === 0 && !this.offReason)
376
+ return null;
377
+ const tokens = t.usage.input + t.usage.output;
378
+ return (`the rule, then the model judge (${this.o.label}) for filings the rule kept apart: ${t.calls} call(s), ` +
379
+ `${t.same} same, ${t.different} different, ${t.fellBack} left to the rule` +
380
+ (t.capped ? `; ${t.capped} filing(s) had more open findings on their page than the ${this.maxCalls} it asks about` : "") +
381
+ (tokens > 0 ? `; ${tokens.toLocaleString("en-US")} tokens` : "") +
382
+ `; ${(t.ms / 1000).toFixed(1)}s` +
383
+ (this.offReason ? `; switched off after ${this.offReason}` : ""));
384
+ }
385
+ }
386
+ /** A finding as the store's judge shows it: what was filed, each field cut to a length the question can carry. */
387
+ function shown(f) {
388
+ const cut = (text, max) => (text.length <= max ? text : `${text.slice(0, max)}…`);
389
+ return { title: cut(f.title, JUDGE_TITLE_CHARS), category: f.category, ...(f.evidence ? { evidence: cut(f.evidence, JUDGE_EVIDENCE_CHARS) } : {}) };
390
+ }
391
+ /** Whether a client answers the judge's sampling requests: sampling with tools, and DEDUP_JUDGE_CAPABILITY. */
392
+ export function clientAnswersJudge(caps) {
393
+ return !!caps?.sampling?.tools && !!caps.experimental?.[DEDUP_JUDGE_CAPABILITY];
394
+ }
395
+ /**
396
+ * How the MCP server dedups for a run: what an attach of the run named, else
397
+ * DEDUP_ENV. A judge asks through the client when the client answers the
398
+ * judge (a `scenescout ci` run, which holds the key, so the server never
399
+ * does), else with a key from the server's environment; with neither it is
400
+ * off, and the note says why. A malformed DEDUP_ENV or DEDUP_PROVIDER_ENV is
401
+ * refused, naming the variable.
402
+ */
403
+ export function planDedup(choice, env, caps) {
404
+ const mode = choice ?? dedupModeFromEnv(env);
405
+ if (mode === "rule")
406
+ return { mode, note: "" };
407
+ if (clientAnswersJudge(caps)) {
408
+ const label = "the CI run's model";
409
+ return {
410
+ mode,
411
+ via: "client",
412
+ label,
413
+ note: `\nFinding dedup: the rule, then ${label}, asked about a filing the rule keeps apart from everything recorded, against the open findings on its page.`,
414
+ };
415
+ }
416
+ const key = judgeKeyConfig(env);
417
+ if (!key.ok)
418
+ return { mode, via: "off", why: key.error, note: `\n⚠ DEDUP JUDGE OFF: ${key.error}. The rule decides duplicates.` };
419
+ const { provider, model, effort } = key.resolved;
420
+ const label = `${provider} ${model}, effort ${effort}`;
421
+ return {
422
+ mode,
423
+ via: "key",
424
+ resolved: key.resolved,
425
+ key: key.key,
426
+ label,
427
+ note: `\nFinding dedup: the rule, then a model judge (${label}), asked about a filing the rule keeps apart from everything recorded, against the open findings on its page. ` +
428
+ `Each pair it is asked about (titles, categories, evidence and the page's path) is sent to ${provider}.`,
429
+ };
430
+ }
431
+ // ── the judge through an MCP client ─────────────────────────────────────────
432
+ /**
433
+ * The client capability, under `experimental`, by which a client says it
434
+ * answers the dedup judge's sampling requests. `scenescout ci` declares it:
435
+ * the server's judge then asks the run's model through the run, and the key
436
+ * never enters the server's process. No other client is sent a sampling
437
+ * request.
438
+ */
439
+ export const DEDUP_JUDGE_CAPABILITY = "scenescout/dedup-judge";
440
+ /**
441
+ * The longest question a client answering the judge accepts. The store's judge
442
+ * cuts titles and evidence (JUDGE_TITLE_CHARS, JUDGE_EVIDENCE_CHARS), so its
443
+ * questions always fit; a longer one is refused, and counted as a failed call.
444
+ */
445
+ export const MAX_JUDGE_KICKOFF_CHARS = 20_000;
446
+ /** The sampling request (MCP sampling/createMessage, with tools) that carries one judge call to the client. */
447
+ export function judgeSamplingParams(system, tools, kickoff) {
448
+ return {
449
+ systemPrompt: system,
450
+ messages: [{ role: "user", content: { type: "text", text: kickoff } }],
451
+ tools: tools.map((t) => ({ name: t.name, description: t.description, inputSchema: { ...t.parameters, type: "object" } })),
452
+ toolChoice: { mode: "required" },
453
+ maxTokens: JUDGE_MAX_OUTPUT_TOKENS,
454
+ includeContext: "none",
455
+ };
456
+ }
457
+ /** An Ask that puts the judge's question to the client. `createMessage` is the server's sampling request. */
458
+ export function samplingAsk(createMessage, callMs = JUDGE_CALL_MS) {
459
+ return async (system, tools, kickoff, limitMs) => turnFromSampling(await createMessage(judgeSamplingParams(system, tools, kickoff), { timeout: Math.min(callMs, limitMs ?? callMs) }));
460
+ }
461
+ /** The client's answer read back as a turn, for parseJudgement. Usage is the client's to count, so the turn carries none. */
462
+ export function turnFromSampling(result) {
463
+ const r = obj(result);
464
+ const content = r ? (Array.isArray(r.content) ? r.content : [r.content]) : [];
465
+ const blocks = content.map(obj).filter((b) => b !== null);
466
+ const calls = blocks
467
+ .filter((b) => b.type === "tool_use")
468
+ .map((b, i) => ({ id: typeof b.id === "string" ? b.id : `call_${i}`, name: typeof b.name === "string" ? b.name : "", input: b.input }));
469
+ const text = blocks
470
+ .filter((b) => b.type === "text")
471
+ .map((b) => String(b.text ?? ""))
472
+ .join("\n")
473
+ .trim();
474
+ const note = calls.length > 0
475
+ ? undefined
476
+ : r?.stopReason === "maxTokens"
477
+ ? "the model's reply was cut at its output limit"
478
+ : text
479
+ ? `the model answered without calling a tool: ${text.slice(0, 200)}`
480
+ : undefined;
481
+ return { text, calls, usage: { ...NO_USAGE }, ...(note ? { note } : {}) };
482
+ }
483
+ /**
484
+ * The client's side: the text of a sampling request shaped as the judge's
485
+ * question, checked for shape and size only. A client answering the judge
486
+ * sends this text on as given, under its own JUDGE_SYSTEM, JUDGE_TOOL and
487
+ * output cap, so the server cannot choose the system prompt, the tools or how
488
+ * much the model may write.
489
+ */
490
+ export function judgeKickoffOf(params) {
491
+ const p = obj(params);
492
+ if (!p)
493
+ return { ok: false, error: "the request has no parameters" };
494
+ const tools = Array.isArray(p.tools) ? p.tools.map(obj) : [];
495
+ if (tools.length !== 1 || tools[0]?.name !== JUDGE_TOOL.name)
496
+ return { ok: false, error: `the request does not offer exactly the ${JUDGE_TOOL.name} tool` };
497
+ const messages = Array.isArray(p.messages) ? p.messages.map(obj) : [];
498
+ if (messages.length !== 1 || messages[0]?.role !== "user")
499
+ return { ok: false, error: "the request is not one user message" };
500
+ const content = messages[0].content;
501
+ const blocks = (Array.isArray(content) ? content : [content]).map(obj);
502
+ const text = blocks.length === 1 && blocks[0]?.type === "text" && typeof blocks[0].text === "string" ? blocks[0].text : "";
503
+ if (!text.trim())
504
+ return { ok: false, error: "the message is not one block of text" };
505
+ if (text.length > MAX_JUDGE_KICKOFF_CHARS)
506
+ return { ok: false, error: `the question is longer than ${MAX_JUDGE_KICKOFF_CHARS} characters` };
507
+ return { ok: true, kickoff: text };
508
+ }
509
+ /** The client's answer: the judge's turn as a sampling result, its tool call as tool_use content. */
510
+ export function samplingResultOf(turn, model) {
511
+ const content = [];
512
+ for (const c of turn.calls) {
513
+ const input = obj(c.input);
514
+ if (input && !c.argsError)
515
+ content.push({ type: "tool_use", id: c.id, name: c.name, input });
516
+ else
517
+ content.push({ type: "text", text: `${c.name || "a tool"} was called with arguments that could not be read: ${c.argsError ?? "not an object"}` });
518
+ }
519
+ const said = [turn.text, turn.note].filter((s) => !!s && s.trim()).join("\n");
520
+ if (said)
521
+ content.push({ type: "text", text: said });
522
+ if (content.length === 0)
523
+ content.push({ type: "text", text: "(no answer)" });
524
+ return { model, role: "assistant", content, stopReason: content.some((b) => b.type === "tool_use") ? "toolUse" : "endTurn" };
525
+ }
206
526
  /**
207
527
  * Accuracy, Brier and ECE of a decider over labelled pairs. `judgements[i]`
208
528
  * answers `labels[i]`; null is a judge that gave no usable answer. The
@@ -80,6 +80,23 @@ export const DESIGN_COLLECT_SCRIPT = `(() => {
80
80
  return false;
81
81
  };
82
82
  const interactiveSel = 'a[href], button, input, select, textarea, [role="button"], [role="link"], [onclick]';
83
+ // The app shell, as the markup declares it. A <header>/<footer> inside a
84
+ // <section> is that section's own heading block, not the page banner (the
85
+ // HTML banner/contentinfo rule), and nothing inside the main content, an
86
+ // article or a dialog is shell however it is marked up.
87
+ const shellSel = 'nav, aside, header, footer, [role="navigation"], [role="banner"], [role="complementary"], [role="contentinfo"]';
88
+ const inShell = (el) => {
89
+ if (el.closest('main, [role="main"], article, [role="article"], dialog, [role="dialog"], [role="alertdialog"]')) return false;
90
+ for (let lm = el.closest(shellSel); lm; lm = lm.parentElement ? lm.parentElement.closest(shellSel) : null) {
91
+ const sectional = (lm.tagName === "HEADER" || lm.tagName === "FOOTER") && !lm.hasAttribute("role");
92
+ if (!sectional || !(lm.parentElement && lm.parentElement.closest("section"))) return true;
93
+ }
94
+ return false;
95
+ };
96
+ const ownFill = (s) => {
97
+ const c = parseColor(s.backgroundColor);
98
+ return (c !== null && c[3] > 0) || (!!s.backgroundImage && s.backgroundImage !== "none");
99
+ };
83
100
  // Saturated (non-gray) color test on a parsed [r,g,b,a].
84
101
  const isSaturated = (p) => p && (Math.max(p[0], p[1], p[2]) - Math.min(p[0], p[1], p[2])) > 40;
85
102
  const hueDeg = (p) => {
@@ -165,6 +182,14 @@ export const DESIGN_COLLECT_SCRIPT = `(() => {
165
182
  fixed: s.position === "fixed" || s.position === "sticky",
166
183
  required: el.hasAttribute("required") || el.getAttribute("aria-required") === "true",
167
184
  submitish: el.matches('button[type="submit"], input[type="submit"]') || /\\b(save|submit|create|send|confirm|apply|continue|next|finish|approve|sign)\\b/i.test(fullText),
185
+ inputType: el.tagName === "INPUT" ? (el.getAttribute("type") || "text").toLowerCase() : "",
186
+ role: (el.getAttribute("role") || "").toLowerCase(),
187
+ filled: ownFill(s),
188
+ inForm: !!el.closest('form, [role="form"]'),
189
+ inRow: !!el.closest('tr, [role="row"]'),
190
+ inSearch: !!el.closest('search, [role="search"]'),
191
+ inBreadcrumb: !!el.closest('[aria-label*="breadcrumb" i], [class*="breadcrumb" i]'),
192
+ shell: inShell(el),
168
193
  sideStripe, gradientText, glass, glow, aiGradient,
169
194
  });
170
195
  }
@@ -235,6 +260,8 @@ const label = (r) => (r.testid ? `[${r.testid}]` : `<${r.tag}> "${r.text.slice(0
235
260
  /** One wording for a small target, page or shell alike: the shell section's, so its prose is unchanged. */
236
261
  const tinyTargetDetail = (r) => `${label(r)} — ${Math.round(r.rect.w)}×${Math.round(r.rect.h)}px tap target`;
237
262
  const clippedTextDetail = (r) => `${label(r)} — text is clipped by its container`;
263
+ /** One wording for body-coloured links, page or shell alike. */
264
+ const indistinctDetail = (bodyColor) => `links with no underline in the body-text colour ${bodyColor}`;
238
265
  /**
239
266
  * Stable identity for one styled element, used to recognise the SAME component
240
267
  * across routes.
@@ -282,6 +309,34 @@ function gridValues(top) {
282
309
  .map((v) => `${v}px`)
283
310
  .join(", ");
284
311
  }
312
+ /** Inputs that are buttons, not fields. */
313
+ const BUTTON_INPUT_TYPES = new Set(["submit", "button", "reset", "image"]);
314
+ /** A button, or a link painted as one. Fields, selects and plain links never compete as actions. */
315
+ function buttonLike(r) {
316
+ if (r.tag === "button" || r.role === "button")
317
+ return true;
318
+ if (r.tag === "input")
319
+ return BUTTON_INPUT_TYPES.has(r.inputType);
320
+ return r.tag === "a" && r.filled;
321
+ }
322
+ /**
323
+ * The fields a user is asked to fill in and submit. Excluded: controls in a
324
+ * table row (a selection checkbox, or a select that edits the row in place),
325
+ * search boxes, and inputs that are buttons. When some fields sit in a <form>,
326
+ * only those count: the rest of the page (filters, toolbars) is not part of
327
+ * what gets submitted. A page with no <form> at all is judged on every field,
328
+ * because many apps build their forms without the element.
329
+ */
330
+ function formFields(records) {
331
+ const candidates = records.filter((r) => r.interactive &&
332
+ (r.tag === "input" || r.tag === "select" || r.tag === "textarea") &&
333
+ !BUTTON_INPUT_TYPES.has(r.inputType) &&
334
+ r.inputType !== "search" &&
335
+ !r.inSearch &&
336
+ !r.inRow);
337
+ const inForm = candidates.filter((r) => r.inForm);
338
+ return inForm.length > 0 ? inForm : candidates;
339
+ }
285
340
  /** Below the WCAG 2.2 target-size minimum; inline links are exempt by that rule. */
286
341
  function tooSmall(r) {
287
342
  return r.tag !== "a" && (r.rect.h < 24 || r.rect.w < 24) && r.rect.h > 0;
@@ -307,11 +362,16 @@ export function analyzeDesign(payload, viewport, chromeKeys = new Set()) {
307
362
  return { report: "DESIGN AUDIT: no visible styled elements found (page empty or not hydrated).", score: null, signatures: [], defects: [] };
308
363
  }
309
364
  const signatures = allRecords.map(styleSignature);
310
- const isChrome = (r) => chromeKeys.has(styleSignature(r));
365
+ // Chrome is what the census has seen on most routes OR what the markup puts
366
+ // in a shell landmark. The census alone needs several audited routes before
367
+ // it knows anything, so the first pages of a run were scored with the whole
368
+ // shell in them and later ones without — a page's score depended on when it
369
+ // was audited. The landmark half is known from the first audit on.
370
+ const isChrome = (r) => r.shell || chromeKeys.has(styleSignature(r));
311
371
  const chromeRecords = allRecords.filter(isChrome);
312
372
  // Everything below scores THIS page. `records` deliberately shadows the full
313
373
  // set so no rule can accidentally reach past the page's own content.
314
- const records = chromeKeys.size > 0 ? allRecords.filter((r) => !isChrome(r)) : allRecords;
374
+ const records = allRecords.filter((r) => !isChrome(r));
315
375
  if (records.length === 0) {
316
376
  return { report: "DESIGN AUDIT: this page is entirely shared layout chrome — nothing page-specific to score.", score: null, signatures, defects: [] };
317
377
  }
@@ -567,7 +627,12 @@ export function analyzeDesign(payload, viewport, chromeKeys = new Set()) {
567
627
  if (r.textLen > 40 && r.tag !== "a" && r.color !== "unknown")
568
628
  bodyColorFreq.set(r.color, (bodyColorFreq.get(r.color) ?? 0) + 1);
569
629
  const dominantBody = [...bodyColorFreq.entries()].sort((a, b) => b[1] - a[1])[0]?.[0];
570
- const indistinct = dominantBody ? records.filter((r) => r.tag === "a" && r.textLen > 0 && !r.underline && r.color === dominantBody) : [];
630
+ const looksLikeBody = (r) => r.tag === "a" && r.textLen > 0 && !r.underline && r.color === dominantBody;
631
+ const indistinct = dominantBody ? records.filter(looksLikeBody) : [];
632
+ // Navigation links are this rule's commonest subject and usually sit in the
633
+ // shell, so the shell's are measured too, against the page's body colour, and
634
+ // reported as the shell's.
635
+ const chromeIndistinct = dominantBody ? chromeRecords.filter(looksLikeBody) : [];
571
636
  if (dominantBody) {
572
637
  if (indistinct.length > 0) {
573
638
  affordances.push(`→ ${indistinct.length} link(s) with no underline AND the same color as body text (e.g. ${label(indistinct[0])}) — invisible as links`);
@@ -619,11 +684,14 @@ export function analyzeDesign(payload, viewport, chromeKeys = new Set()) {
619
684
  // suite never answers.
620
685
  const effort = [];
621
686
  const pageBg = parseRgb(records.find((r) => r.bg && r.bg.startsWith("rgb"))?.bg ?? "") ?? [255, 255, 255, 1];
622
- // A "prominent" action = filled control whose background clearly departs
623
- // from the page background, at a clickable size. That is what the eye lands
624
- // on, so it is the page's implied primary action.
687
+ // A "prominent" action = a button (or a link painted as one) that paints its
688
+ // own background, clearly departing from the page background, at a clickable
689
+ // size. That is what the eye lands on, so it is the page's implied primary
690
+ // action. Fields never qualify — a white text field on a grey page departs
691
+ // from the page background too — nor does a ghost button showing a card's
692
+ // background, nor a breadcrumb, which is a way back rather than an action.
625
693
  const prominent = records.filter((r) => {
626
- if (!r.interactive || r.rect.w < 60 || r.rect.h < 24)
694
+ if (!r.interactive || !buttonLike(r) || !r.filled || r.inBreadcrumb || r.rect.w < 60 || r.rect.h < 24)
627
695
  return false;
628
696
  const bg = parseRgb(r.bg);
629
697
  if (!bg)
@@ -647,7 +715,7 @@ export function analyzeDesign(payload, viewport, chromeKeys = new Set()) {
647
715
  effort.push(`→ the primary action (${label(nearest)}) sits ${Math.round(nearest.rect.y - foldH)}px below the fold — the user must scroll before seeing what this page is for`);
648
716
  }
649
717
  // Form burden: how much is being asked, and how much of it is actually needed.
650
- const fields = records.filter((r) => r.interactive && /input|select|textarea/.test(r.tag));
718
+ const fields = formFields(records);
651
719
  if (fields.length >= 5) {
652
720
  const req = fields.filter((r) => r.required).length;
653
721
  if (req === 0) {
@@ -703,13 +771,14 @@ export function analyzeDesign(payload, viewport, chromeKeys = new Set()) {
703
771
  const chromeSection = [];
704
772
  const chromeDefects = [];
705
773
  if (chromeRecords.length > 0) {
706
- chromeDefects.push(...contrastFailures(chromeRecords).map((detail) => ({ rule: "contrast", detail, chrome: true })), ...chromeRecords.filter((r) => r.interactive && tooSmall(r)).map((r) => ({ rule: "tiny-target", detail: tinyTargetDetail(r), chrome: true })), ...chromeRecords.filter((r) => r.clipped && r.textLen > 0).map((r) => ({ rule: "clipped-text", detail: clippedTextDetail(r), chrome: true })));
707
- const chromeIssues = chromeDefects.map((d) => d.detail);
774
+ chromeDefects.push(...contrastFailures(chromeRecords).map((detail) => ({ rule: "contrast", detail, chrome: true })), ...chromeRecords.filter((r) => r.interactive && tooSmall(r)).map((r) => ({ rule: "tiny-target", detail: tinyTargetDetail(r), chrome: true })), ...chromeRecords.filter((r) => r.clipped && r.textLen > 0).map((r) => ({ rule: "clipped-text", detail: clippedTextDetail(r), chrome: true })), ...(chromeIndistinct.length > 0 ? [{ rule: "indistinct-link", detail: indistinctDetail(dominantBody ?? ""), chrome: true }] : []));
775
+ // A convention (indistinct-link) is a → line, as it is on the page; the rest are measurable defects.
776
+ const chromeIssues = chromeDefects.map((d) => `${d.rule === "indistinct-link" ? "→" : "⚠"} ${d.detail}`);
708
777
  if (chromeIssues.length > 0) {
709
778
  chromeSection.push(`SHARED CHROME (${chromeRecords.length} shell elements, excluded from this page's score and reported here instead):\n` +
710
779
  [...new Set(chromeIssues)]
711
780
  .slice(0, CHROME_ISSUE_CAP)
712
- .map((s) => ` ⚠ ${s}`)
781
+ .map((s) => ` ${s}`)
713
782
  .join("\n") +
714
783
  `\n → these belong to the app shell and recur on every page that renders it. File ONE finding for the shell, not one per page.`);
715
784
  }
@@ -739,7 +808,7 @@ export function analyzeDesign(payload, viewport, chromeKeys = new Set()) {
739
808
  // values on ten pages are one entry on ten routes, and the entry's fingerprint does not move when content does.
740
809
  ...(pad.n > 10 && pad.pct > 20 ? [{ rule: "off-grid-spacing", detail: `paddings off a 4px grid: ${gridValues(pad.top)}` }] : []),
741
810
  ...(mar.n > 10 && mar.pct > 20 ? [{ rule: "off-grid-spacing", detail: `vertical margins off a 4px grid: ${gridValues(mar.top)}` }] : []),
742
- ...(indistinct.length > 0 ? [{ rule: "indistinct-link", detail: `links with no underline in the body-text colour ${dominantBody}` }] : []),
811
+ ...(indistinct.length > 0 ? [{ rule: "indistinct-link", detail: indistinctDetail(dominantBody ?? "") }] : []),
743
812
  ...chromeDefects,
744
813
  ];
745
814
  return { report, score, signatures, defects };
@@ -173,6 +173,23 @@ export function submits(kind, probe) {
173
173
  export function isEmptySubmit(kind, probe) {
174
174
  return probe !== null && submits(kind, probe) && allTextEmpty(probe.fields);
175
175
  }
176
+ /** The words that make a button submit-style, matched as whole words. */
177
+ const SUBMIT_WORDS = new Set(["submit", "send", "save", "create", "apply", "subscribe", "register", "sign", "signin", "signup", "post", "add"]);
178
+ /**
179
+ * Whether a button reads as one that sends something, by its name and test
180
+ * id. Whole words only: a test id is split on its separators and its camel
181
+ * case first, so "sign-in" and "Sign" count but the "sign" inside "assignee"
182
+ * and the "post" inside "postcode" do not.
183
+ */
184
+ export function isSubmitLike(role, name, testid) {
185
+ if (role !== "button")
186
+ return false;
187
+ const words = `${name} ${testid ?? ""}`
188
+ .replace(/([a-z])([A-Z])/g, "$1 $2")
189
+ .toLowerCase()
190
+ .split(/[^a-z]+/);
191
+ return words.some((w) => SUBMIT_WORDS.has(w));
192
+ }
176
193
  /**
177
194
  * Whether a failed page read is the page going away under it (a navigation
178
195
  * or a closed tab), which is expected after a submit and says nothing. Any