scenescout 3.14.0 → 3.15.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +25 -0
- package/README.md +27 -4
- package/dist/check-run.js +22 -11
- package/dist/ci-run.js +113 -17
- package/dist/cli.js +150 -75
- package/dist/commands.js +141 -0
- package/dist/engine/browser.js +546 -133
- package/dist/engine/check.js +77 -30
- package/dist/engine/ci.js +97 -0
- package/dist/engine/claims.js +81 -15
- package/dist/engine/collector.js +307 -65
- package/dist/engine/dedup.js +329 -9
- package/dist/engine/design.js +81 -12
- package/dist/engine/forms.js +17 -0
- package/dist/engine/memory.js +180 -36
- package/dist/engine/oracles.js +126 -1
- package/dist/engine/provider.js +4 -3
- package/dist/engine/refresh.js +270 -22
- package/dist/engine/report.js +8 -1
- package/dist/engine/scripted-login.js +342 -62
- package/dist/first-run.js +577 -0
- package/dist/login-run.js +136 -35
- package/dist/mcp-server.js +75 -16
- package/package.json +3 -2
- package/skills/scenescout/SKILL.md +3 -3
package/dist/engine/dedup.js
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* Measuring the finding-dedup decision against the answer keys, and
|
|
3
|
-
*
|
|
2
|
+
* Measuring the finding-dedup decision against the answer keys, and the
|
|
3
|
+
* model-backed judge the store can ask beside its rule.
|
|
4
4
|
*
|
|
5
5
|
* The store's dedup rule (memory.ts, isDuplicateFinding) decides whether a
|
|
6
6
|
* newly filed finding is one it already records. Its mistakes are costly both
|
|
@@ -12,14 +12,19 @@
|
|
|
12
12
|
*
|
|
13
13
|
* The model judge asks for a verdict (same / different / unsure) and a
|
|
14
14
|
* probability through one tool call, so its answer is structured, not parsed
|
|
15
|
-
* from prose.
|
|
16
|
-
*
|
|
17
|
-
*
|
|
15
|
+
* from prose. Measured on these pairs it beat the rule on both apps
|
|
16
|
+
* (docs/benchmark.md), and every pair it won was a merge the rule missed, so
|
|
17
|
+
* the store asks it only about filings the rule keeps apart (DedupJudge,
|
|
18
|
+
* below): a `scenescout ci` run by default, the MCP server when told to.
|
|
19
|
+
* Nothing here touches the network; the caller's `ask` does.
|
|
18
20
|
*/
|
|
19
21
|
import { createHash } from "node:crypto";
|
|
20
22
|
import { archiveApp, classify, findingText } from "./bench.js";
|
|
21
23
|
import { MIN_FOR_A_VERDICT } from "./calibration.js";
|
|
22
|
-
import {
|
|
24
|
+
import { addUsage, dedupModeFromEnv, judgeKeyConfig, NO_USAGE } from "./ci.js";
|
|
25
|
+
import { withWatchdog } from "./dispatch.js";
|
|
26
|
+
import { findingId, isDuplicateFinding, titleSimilarity } from "./memory.js";
|
|
27
|
+
import { obj } from "./provider.js";
|
|
23
28
|
const pagesOf = (e) => [e.route, ...(e.alsoOn ?? [])];
|
|
24
29
|
const norm = (s) => (s ?? "").toLowerCase().replace(/\s+/g, " ").trim();
|
|
25
30
|
/**
|
|
@@ -178,10 +183,12 @@ export function parseJudgement(turn) {
|
|
|
178
183
|
* Decide one pair. With no judge the rule decides, as the store does. With a
|
|
179
184
|
* judge, the model decides when it answers same or different; when it fails
|
|
180
185
|
* for any reason, or is unsure, the rule decides and the note says why, so a
|
|
181
|
-
* run never silently becomes a run of the rule.
|
|
186
|
+
* run never silently becomes a run of the rule. `ruleVerdict` is the rule's
|
|
187
|
+
* verdict when the caller already has it: the store asks only about pairs its
|
|
188
|
+
* rule keeps apart, and its rule reads more of a finding than a pair carries.
|
|
182
189
|
*/
|
|
183
|
-
export async function decideDuplicate(p, ask) {
|
|
184
|
-
const rule = () => ruleJudgement(p).verdict === "same";
|
|
190
|
+
export async function decideDuplicate(p, ask, ruleVerdict) {
|
|
191
|
+
const rule = ruleVerdict ?? (() => ruleJudgement(p).verdict === "same");
|
|
185
192
|
if (!ask)
|
|
186
193
|
return { duplicate: rule(), by: "rule" };
|
|
187
194
|
let parsed;
|
|
@@ -203,6 +210,319 @@ export async function decideDuplicate(p, ask) {
|
|
|
203
210
|
return { duplicate: rule(), by: "rule", judgement: parsed.judgement, note: "the model judge was unsure; the current rule decided" };
|
|
204
211
|
return { duplicate: parsed.judgement.verdict === "same", by: "model", judgement: parsed.judgement };
|
|
205
212
|
}
|
|
213
|
+
// ── the judge the store asks ────────────────────────────────────────────────
|
|
214
|
+
/** Calls made about one filing at most: its most alike open findings on the page. */
|
|
215
|
+
export const JUDGE_MAX_CALLS = 3;
|
|
216
|
+
/** The longest one judge call may take. */
|
|
217
|
+
export const JUDGE_CALL_MS = 15_000;
|
|
218
|
+
/**
|
|
219
|
+
* The longest the judge may spend on one filing, every call together. Under
|
|
220
|
+
* scout_finding's 60-second watchdog, so a slow provider costs a filing its
|
|
221
|
+
* judge and never the filing.
|
|
222
|
+
*/
|
|
223
|
+
export const JUDGE_FILING_MS = 40_000;
|
|
224
|
+
/** Failed calls in a row after which the judge is switched off for the rest of the run. */
|
|
225
|
+
export const JUDGE_MAX_FAILURES = 3;
|
|
226
|
+
/** The most output one judge call may produce: the answer is one short tool call (24 to 26 tokens when measured). */
|
|
227
|
+
export const JUDGE_MAX_OUTPUT_TOKENS = 2_000;
|
|
228
|
+
/**
|
|
229
|
+
* The most of a title, and of evidence, the store's judge shows the model. A
|
|
230
|
+
* filing with a pasted log for evidence still makes a question of bounded
|
|
231
|
+
* size, which a client answering the judge accepts (MAX_JUDGE_KICKOFF_CHARS).
|
|
232
|
+
*/
|
|
233
|
+
export const JUDGE_TITLE_CHARS = 500;
|
|
234
|
+
export const JUDGE_EVIDENCE_CHARS = 2_000;
|
|
235
|
+
const routeOf = (state) => state.split("#")[0];
|
|
236
|
+
/** A judge time limit as a log line says it: "15s", or "20ms" for a test's. */
|
|
237
|
+
export const durationText = (ms) => (ms < 1000 ? `${ms}ms` : `${Math.round(ms / 1000)}s`);
|
|
238
|
+
/**
|
|
239
|
+
* The model judge as the store asks it (memory.ts DuplicateJudge). It is
|
|
240
|
+
* asked only about a filing the rule keeps apart from everything stored, and
|
|
241
|
+
* only against the open findings on the filing's own page: the most alike
|
|
242
|
+
* title first, at most `maxCalls` of them, within `filingMs`. "Same" merges;
|
|
243
|
+
* anything else (different, unsure, a contradiction, a failed or slow call)
|
|
244
|
+
* leaves the rule's decision, which is to keep the filing apart. So the judge
|
|
245
|
+
* only ever adds merges the rule missed, which is where every pair it won in
|
|
246
|
+
* the measurement was (docs/benchmark.md).
|
|
247
|
+
*
|
|
248
|
+
* Each kind of fall-back is logged once, the cap once per page, and
|
|
249
|
+
* JUDGE_MAX_FAILURES failed calls in a row switch the judge off for the rest
|
|
250
|
+
* of the run. `log` reaches the operator; the caller redacts keys from it.
|
|
251
|
+
*/
|
|
252
|
+
export class DedupJudge {
|
|
253
|
+
ask;
|
|
254
|
+
o;
|
|
255
|
+
counts = { calls: 0, same: 0, different: 0, fellBack: 0, capped: 0, usage: { ...NO_USAGE }, ms: 0 };
|
|
256
|
+
failuresInARow = 0;
|
|
257
|
+
offReason = null;
|
|
258
|
+
logged = new Set();
|
|
259
|
+
maxCalls;
|
|
260
|
+
callMs;
|
|
261
|
+
filingMs;
|
|
262
|
+
now;
|
|
263
|
+
redact;
|
|
264
|
+
constructor(ask, o) {
|
|
265
|
+
this.ask = ask;
|
|
266
|
+
this.o = o;
|
|
267
|
+
this.maxCalls = o.maxCalls ?? JUDGE_MAX_CALLS;
|
|
268
|
+
this.callMs = o.callMs ?? JUDGE_CALL_MS;
|
|
269
|
+
this.filingMs = o.filingMs ?? JUDGE_FILING_MS;
|
|
270
|
+
this.now = o.now ?? Date.now;
|
|
271
|
+
this.redact = o.redact ?? ((text) => text);
|
|
272
|
+
}
|
|
273
|
+
/** What the judge has done this run. */
|
|
274
|
+
get tally() {
|
|
275
|
+
return this.counts;
|
|
276
|
+
}
|
|
277
|
+
/** Why the judge is no longer asked, or null while it is. */
|
|
278
|
+
get off() {
|
|
279
|
+
return this.offReason;
|
|
280
|
+
}
|
|
281
|
+
async judge(incoming, stored) {
|
|
282
|
+
try {
|
|
283
|
+
return await this.decide(incoming, stored);
|
|
284
|
+
}
|
|
285
|
+
catch (err) {
|
|
286
|
+
// decideDuplicate catches whatever a call throws, so this is a fault in this
|
|
287
|
+
// class. It would recur on every filing, so the judge stops here, and says why;
|
|
288
|
+
// the log gets the stack, the report only the message.
|
|
289
|
+
const message = err instanceof Error ? err.message : String(err);
|
|
290
|
+
this.switchOff(`a fault in the judge: ${message}`, err instanceof Error ? err.stack : undefined);
|
|
291
|
+
return null;
|
|
292
|
+
}
|
|
293
|
+
}
|
|
294
|
+
async decide(incoming, stored) {
|
|
295
|
+
if (this.offReason)
|
|
296
|
+
return null;
|
|
297
|
+
const route = routeOf(incoming.state);
|
|
298
|
+
const candidates = stored
|
|
299
|
+
.map((x, i) => ({ x, i, alike: titleSimilarity(x.title, incoming.title) }))
|
|
300
|
+
.filter(({ x }) => x.status !== "resolved" && routeOf(x.state) === route)
|
|
301
|
+
// The most alike first; of two equally alike, the newer.
|
|
302
|
+
.sort((p, q) => q.alike - p.alike || q.i - p.i);
|
|
303
|
+
if (candidates.length > this.maxCalls) {
|
|
304
|
+
this.counts.capped += 1;
|
|
305
|
+
this.once(`cap\u0000${route}`, `dedup judge: ${candidates.length} open findings on ${route} could be the one just filed there; the judge is asked about the ` +
|
|
306
|
+
`${this.maxCalls} most alike, and the rule, which kept it apart from all of them, decides the rest ` +
|
|
307
|
+
`(at most ${this.maxCalls} calls per filing; logged once per page)`);
|
|
308
|
+
}
|
|
309
|
+
const started = this.now();
|
|
310
|
+
for (const { x } of candidates.slice(0, this.maxCalls)) {
|
|
311
|
+
if (this.offReason)
|
|
312
|
+
break;
|
|
313
|
+
const left = this.filingMs - (this.now() - started);
|
|
314
|
+
if (left <= 0) {
|
|
315
|
+
this.once("time", `dedup judge: the ${durationText(this.filingMs)} a filing may take ran out; the rule decided the pairs left (logged once)`);
|
|
316
|
+
break;
|
|
317
|
+
}
|
|
318
|
+
const d = await decideDuplicate({ route, a: shown(x), b: shown(incoming) }, this.timed(Math.min(this.callMs, left)), () => false);
|
|
319
|
+
this.count(d);
|
|
320
|
+
if (d.by === "model" && d.duplicate && d.judgement)
|
|
321
|
+
return { sameAs: x.id, pSame: d.judgement.pSame };
|
|
322
|
+
}
|
|
323
|
+
return null;
|
|
324
|
+
}
|
|
325
|
+
/** The ask, ended at `ms` and counted. */
|
|
326
|
+
timed(ms) {
|
|
327
|
+
return async (system, tools, kickoff) => {
|
|
328
|
+
const started = this.now();
|
|
329
|
+
this.counts.calls += 1;
|
|
330
|
+
try {
|
|
331
|
+
const turn = await withWatchdog("dedup judge", this.ask(system, tools, kickoff, ms), ms, () => null);
|
|
332
|
+
if (!turn)
|
|
333
|
+
throw new Error(`no answer within ${durationText(ms)}`);
|
|
334
|
+
this.counts.usage = addUsage(this.counts.usage, turn.usage);
|
|
335
|
+
return turn;
|
|
336
|
+
}
|
|
337
|
+
finally {
|
|
338
|
+
this.counts.ms += this.now() - started;
|
|
339
|
+
}
|
|
340
|
+
};
|
|
341
|
+
}
|
|
342
|
+
count(d) {
|
|
343
|
+
if (d.by === "model") {
|
|
344
|
+
this.failuresInARow = 0;
|
|
345
|
+
if (d.duplicate)
|
|
346
|
+
this.counts.same += 1;
|
|
347
|
+
else
|
|
348
|
+
this.counts.different += 1;
|
|
349
|
+
return;
|
|
350
|
+
}
|
|
351
|
+
this.counts.fellBack += 1;
|
|
352
|
+
if (d.failure !== "error") {
|
|
353
|
+
// Unsure, or an answer that contradicted itself: the provider answered, so neither counts towards switching off.
|
|
354
|
+
this.failuresInARow = 0;
|
|
355
|
+
this.once(d.failure ?? "unsure", `dedup judge: ${d.note ?? "no usable answer"} (later ones are counted, not logged)`);
|
|
356
|
+
return;
|
|
357
|
+
}
|
|
358
|
+
this.failuresInARow += 1;
|
|
359
|
+
this.once("error", `dedup judge: ${d.note ?? "a call failed"} (later failures are counted, not logged)`);
|
|
360
|
+
if (this.failuresInARow >= JUDGE_MAX_FAILURES)
|
|
361
|
+
this.switchOff(`${this.failuresInARow} failed calls in a row, the last: ${d.note ?? "no detail"}`);
|
|
362
|
+
}
|
|
363
|
+
switchOff(reason, detail) {
|
|
364
|
+
this.offReason = this.redact(reason);
|
|
365
|
+
this.o.log(this.redact(`dedup judge: switched off after ${reason}; the rule decides every filing for the rest of this run${detail ? `\n${detail}` : ""}`));
|
|
366
|
+
}
|
|
367
|
+
once(kind, line) {
|
|
368
|
+
if (this.logged.has(kind))
|
|
369
|
+
return;
|
|
370
|
+
this.logged.add(kind);
|
|
371
|
+
this.o.log(this.redact(line));
|
|
372
|
+
}
|
|
373
|
+
describe() {
|
|
374
|
+
const t = this.counts;
|
|
375
|
+
if (t.calls === 0 && !this.offReason)
|
|
376
|
+
return null;
|
|
377
|
+
const tokens = t.usage.input + t.usage.output;
|
|
378
|
+
return (`the rule, then the model judge (${this.o.label}) for filings the rule kept apart: ${t.calls} call(s), ` +
|
|
379
|
+
`${t.same} same, ${t.different} different, ${t.fellBack} left to the rule` +
|
|
380
|
+
(t.capped ? `; ${t.capped} filing(s) had more open findings on their page than the ${this.maxCalls} it asks about` : "") +
|
|
381
|
+
(tokens > 0 ? `; ${tokens.toLocaleString("en-US")} tokens` : "") +
|
|
382
|
+
`; ${(t.ms / 1000).toFixed(1)}s` +
|
|
383
|
+
(this.offReason ? `; switched off after ${this.offReason}` : ""));
|
|
384
|
+
}
|
|
385
|
+
}
|
|
386
|
+
/** A finding as the store's judge shows it: what was filed, each field cut to a length the question can carry. */
|
|
387
|
+
function shown(f) {
|
|
388
|
+
const cut = (text, max) => (text.length <= max ? text : `${text.slice(0, max)}…`);
|
|
389
|
+
return { title: cut(f.title, JUDGE_TITLE_CHARS), category: f.category, ...(f.evidence ? { evidence: cut(f.evidence, JUDGE_EVIDENCE_CHARS) } : {}) };
|
|
390
|
+
}
|
|
391
|
+
/** Whether a client answers the judge's sampling requests: sampling with tools, and DEDUP_JUDGE_CAPABILITY. */
|
|
392
|
+
export function clientAnswersJudge(caps) {
|
|
393
|
+
return !!caps?.sampling?.tools && !!caps.experimental?.[DEDUP_JUDGE_CAPABILITY];
|
|
394
|
+
}
|
|
395
|
+
/**
|
|
396
|
+
* How the MCP server dedups for a run: what an attach of the run named, else
|
|
397
|
+
* DEDUP_ENV. A judge asks through the client when the client answers the
|
|
398
|
+
* judge (a `scenescout ci` run, which holds the key, so the server never
|
|
399
|
+
* does), else with a key from the server's environment; with neither it is
|
|
400
|
+
* off, and the note says why. A malformed DEDUP_ENV or DEDUP_PROVIDER_ENV is
|
|
401
|
+
* refused, naming the variable.
|
|
402
|
+
*/
|
|
403
|
+
export function planDedup(choice, env, caps) {
|
|
404
|
+
const mode = choice ?? dedupModeFromEnv(env);
|
|
405
|
+
if (mode === "rule")
|
|
406
|
+
return { mode, note: "" };
|
|
407
|
+
if (clientAnswersJudge(caps)) {
|
|
408
|
+
const label = "the CI run's model";
|
|
409
|
+
return {
|
|
410
|
+
mode,
|
|
411
|
+
via: "client",
|
|
412
|
+
label,
|
|
413
|
+
note: `\nFinding dedup: the rule, then ${label}, asked about a filing the rule keeps apart from everything recorded, against the open findings on its page.`,
|
|
414
|
+
};
|
|
415
|
+
}
|
|
416
|
+
const key = judgeKeyConfig(env);
|
|
417
|
+
if (!key.ok)
|
|
418
|
+
return { mode, via: "off", why: key.error, note: `\n⚠ DEDUP JUDGE OFF: ${key.error}. The rule decides duplicates.` };
|
|
419
|
+
const { provider, model, effort } = key.resolved;
|
|
420
|
+
const label = `${provider} ${model}, effort ${effort}`;
|
|
421
|
+
return {
|
|
422
|
+
mode,
|
|
423
|
+
via: "key",
|
|
424
|
+
resolved: key.resolved,
|
|
425
|
+
key: key.key,
|
|
426
|
+
label,
|
|
427
|
+
note: `\nFinding dedup: the rule, then a model judge (${label}), asked about a filing the rule keeps apart from everything recorded, against the open findings on its page. ` +
|
|
428
|
+
`Each pair it is asked about (titles, categories, evidence and the page's path) is sent to ${provider}.`,
|
|
429
|
+
};
|
|
430
|
+
}
|
|
431
|
+
// ── the judge through an MCP client ─────────────────────────────────────────
|
|
432
|
+
/**
|
|
433
|
+
* The client capability, under `experimental`, by which a client says it
|
|
434
|
+
* answers the dedup judge's sampling requests. `scenescout ci` declares it:
|
|
435
|
+
* the server's judge then asks the run's model through the run, and the key
|
|
436
|
+
* never enters the server's process. No other client is sent a sampling
|
|
437
|
+
* request.
|
|
438
|
+
*/
|
|
439
|
+
export const DEDUP_JUDGE_CAPABILITY = "scenescout/dedup-judge";
|
|
440
|
+
/**
|
|
441
|
+
* The longest question a client answering the judge accepts. The store's judge
|
|
442
|
+
* cuts titles and evidence (JUDGE_TITLE_CHARS, JUDGE_EVIDENCE_CHARS), so its
|
|
443
|
+
* questions always fit; a longer one is refused, and counted as a failed call.
|
|
444
|
+
*/
|
|
445
|
+
export const MAX_JUDGE_KICKOFF_CHARS = 20_000;
|
|
446
|
+
/** The sampling request (MCP sampling/createMessage, with tools) that carries one judge call to the client. */
|
|
447
|
+
export function judgeSamplingParams(system, tools, kickoff) {
|
|
448
|
+
return {
|
|
449
|
+
systemPrompt: system,
|
|
450
|
+
messages: [{ role: "user", content: { type: "text", text: kickoff } }],
|
|
451
|
+
tools: tools.map((t) => ({ name: t.name, description: t.description, inputSchema: { ...t.parameters, type: "object" } })),
|
|
452
|
+
toolChoice: { mode: "required" },
|
|
453
|
+
maxTokens: JUDGE_MAX_OUTPUT_TOKENS,
|
|
454
|
+
includeContext: "none",
|
|
455
|
+
};
|
|
456
|
+
}
|
|
457
|
+
/** An Ask that puts the judge's question to the client. `createMessage` is the server's sampling request. */
|
|
458
|
+
export function samplingAsk(createMessage, callMs = JUDGE_CALL_MS) {
|
|
459
|
+
return async (system, tools, kickoff, limitMs) => turnFromSampling(await createMessage(judgeSamplingParams(system, tools, kickoff), { timeout: Math.min(callMs, limitMs ?? callMs) }));
|
|
460
|
+
}
|
|
461
|
+
/** The client's answer read back as a turn, for parseJudgement. Usage is the client's to count, so the turn carries none. */
|
|
462
|
+
export function turnFromSampling(result) {
|
|
463
|
+
const r = obj(result);
|
|
464
|
+
const content = r ? (Array.isArray(r.content) ? r.content : [r.content]) : [];
|
|
465
|
+
const blocks = content.map(obj).filter((b) => b !== null);
|
|
466
|
+
const calls = blocks
|
|
467
|
+
.filter((b) => b.type === "tool_use")
|
|
468
|
+
.map((b, i) => ({ id: typeof b.id === "string" ? b.id : `call_${i}`, name: typeof b.name === "string" ? b.name : "", input: b.input }));
|
|
469
|
+
const text = blocks
|
|
470
|
+
.filter((b) => b.type === "text")
|
|
471
|
+
.map((b) => String(b.text ?? ""))
|
|
472
|
+
.join("\n")
|
|
473
|
+
.trim();
|
|
474
|
+
const note = calls.length > 0
|
|
475
|
+
? undefined
|
|
476
|
+
: r?.stopReason === "maxTokens"
|
|
477
|
+
? "the model's reply was cut at its output limit"
|
|
478
|
+
: text
|
|
479
|
+
? `the model answered without calling a tool: ${text.slice(0, 200)}`
|
|
480
|
+
: undefined;
|
|
481
|
+
return { text, calls, usage: { ...NO_USAGE }, ...(note ? { note } : {}) };
|
|
482
|
+
}
|
|
483
|
+
/**
|
|
484
|
+
* The client's side: the text of a sampling request shaped as the judge's
|
|
485
|
+
* question, checked for shape and size only. A client answering the judge
|
|
486
|
+
* sends this text on as given, under its own JUDGE_SYSTEM, JUDGE_TOOL and
|
|
487
|
+
* output cap, so the server cannot choose the system prompt, the tools or how
|
|
488
|
+
* much the model may write.
|
|
489
|
+
*/
|
|
490
|
+
export function judgeKickoffOf(params) {
|
|
491
|
+
const p = obj(params);
|
|
492
|
+
if (!p)
|
|
493
|
+
return { ok: false, error: "the request has no parameters" };
|
|
494
|
+
const tools = Array.isArray(p.tools) ? p.tools.map(obj) : [];
|
|
495
|
+
if (tools.length !== 1 || tools[0]?.name !== JUDGE_TOOL.name)
|
|
496
|
+
return { ok: false, error: `the request does not offer exactly the ${JUDGE_TOOL.name} tool` };
|
|
497
|
+
const messages = Array.isArray(p.messages) ? p.messages.map(obj) : [];
|
|
498
|
+
if (messages.length !== 1 || messages[0]?.role !== "user")
|
|
499
|
+
return { ok: false, error: "the request is not one user message" };
|
|
500
|
+
const content = messages[0].content;
|
|
501
|
+
const blocks = (Array.isArray(content) ? content : [content]).map(obj);
|
|
502
|
+
const text = blocks.length === 1 && blocks[0]?.type === "text" && typeof blocks[0].text === "string" ? blocks[0].text : "";
|
|
503
|
+
if (!text.trim())
|
|
504
|
+
return { ok: false, error: "the message is not one block of text" };
|
|
505
|
+
if (text.length > MAX_JUDGE_KICKOFF_CHARS)
|
|
506
|
+
return { ok: false, error: `the question is longer than ${MAX_JUDGE_KICKOFF_CHARS} characters` };
|
|
507
|
+
return { ok: true, kickoff: text };
|
|
508
|
+
}
|
|
509
|
+
/** The client's answer: the judge's turn as a sampling result, its tool call as tool_use content. */
|
|
510
|
+
export function samplingResultOf(turn, model) {
|
|
511
|
+
const content = [];
|
|
512
|
+
for (const c of turn.calls) {
|
|
513
|
+
const input = obj(c.input);
|
|
514
|
+
if (input && !c.argsError)
|
|
515
|
+
content.push({ type: "tool_use", id: c.id, name: c.name, input });
|
|
516
|
+
else
|
|
517
|
+
content.push({ type: "text", text: `${c.name || "a tool"} was called with arguments that could not be read: ${c.argsError ?? "not an object"}` });
|
|
518
|
+
}
|
|
519
|
+
const said = [turn.text, turn.note].filter((s) => !!s && s.trim()).join("\n");
|
|
520
|
+
if (said)
|
|
521
|
+
content.push({ type: "text", text: said });
|
|
522
|
+
if (content.length === 0)
|
|
523
|
+
content.push({ type: "text", text: "(no answer)" });
|
|
524
|
+
return { model, role: "assistant", content, stopReason: content.some((b) => b.type === "tool_use") ? "toolUse" : "endTurn" };
|
|
525
|
+
}
|
|
206
526
|
/**
|
|
207
527
|
* Accuracy, Brier and ECE of a decider over labelled pairs. `judgements[i]`
|
|
208
528
|
* answers `labels[i]`; null is a judge that gave no usable answer. The
|
package/dist/engine/design.js
CHANGED
|
@@ -80,6 +80,23 @@ export const DESIGN_COLLECT_SCRIPT = `(() => {
|
|
|
80
80
|
return false;
|
|
81
81
|
};
|
|
82
82
|
const interactiveSel = 'a[href], button, input, select, textarea, [role="button"], [role="link"], [onclick]';
|
|
83
|
+
// The app shell, as the markup declares it. A <header>/<footer> inside a
|
|
84
|
+
// <section> is that section's own heading block, not the page banner (the
|
|
85
|
+
// HTML banner/contentinfo rule), and nothing inside the main content, an
|
|
86
|
+
// article or a dialog is shell however it is marked up.
|
|
87
|
+
const shellSel = 'nav, aside, header, footer, [role="navigation"], [role="banner"], [role="complementary"], [role="contentinfo"]';
|
|
88
|
+
const inShell = (el) => {
|
|
89
|
+
if (el.closest('main, [role="main"], article, [role="article"], dialog, [role="dialog"], [role="alertdialog"]')) return false;
|
|
90
|
+
for (let lm = el.closest(shellSel); lm; lm = lm.parentElement ? lm.parentElement.closest(shellSel) : null) {
|
|
91
|
+
const sectional = (lm.tagName === "HEADER" || lm.tagName === "FOOTER") && !lm.hasAttribute("role");
|
|
92
|
+
if (!sectional || !(lm.parentElement && lm.parentElement.closest("section"))) return true;
|
|
93
|
+
}
|
|
94
|
+
return false;
|
|
95
|
+
};
|
|
96
|
+
const ownFill = (s) => {
|
|
97
|
+
const c = parseColor(s.backgroundColor);
|
|
98
|
+
return (c !== null && c[3] > 0) || (!!s.backgroundImage && s.backgroundImage !== "none");
|
|
99
|
+
};
|
|
83
100
|
// Saturated (non-gray) color test on a parsed [r,g,b,a].
|
|
84
101
|
const isSaturated = (p) => p && (Math.max(p[0], p[1], p[2]) - Math.min(p[0], p[1], p[2])) > 40;
|
|
85
102
|
const hueDeg = (p) => {
|
|
@@ -165,6 +182,14 @@ export const DESIGN_COLLECT_SCRIPT = `(() => {
|
|
|
165
182
|
fixed: s.position === "fixed" || s.position === "sticky",
|
|
166
183
|
required: el.hasAttribute("required") || el.getAttribute("aria-required") === "true",
|
|
167
184
|
submitish: el.matches('button[type="submit"], input[type="submit"]') || /\\b(save|submit|create|send|confirm|apply|continue|next|finish|approve|sign)\\b/i.test(fullText),
|
|
185
|
+
inputType: el.tagName === "INPUT" ? (el.getAttribute("type") || "text").toLowerCase() : "",
|
|
186
|
+
role: (el.getAttribute("role") || "").toLowerCase(),
|
|
187
|
+
filled: ownFill(s),
|
|
188
|
+
inForm: !!el.closest('form, [role="form"]'),
|
|
189
|
+
inRow: !!el.closest('tr, [role="row"]'),
|
|
190
|
+
inSearch: !!el.closest('search, [role="search"]'),
|
|
191
|
+
inBreadcrumb: !!el.closest('[aria-label*="breadcrumb" i], [class*="breadcrumb" i]'),
|
|
192
|
+
shell: inShell(el),
|
|
168
193
|
sideStripe, gradientText, glass, glow, aiGradient,
|
|
169
194
|
});
|
|
170
195
|
}
|
|
@@ -235,6 +260,8 @@ const label = (r) => (r.testid ? `[${r.testid}]` : `<${r.tag}> "${r.text.slice(0
|
|
|
235
260
|
/** One wording for a small target, page or shell alike: the shell section's, so its prose is unchanged. */
|
|
236
261
|
const tinyTargetDetail = (r) => `${label(r)} — ${Math.round(r.rect.w)}×${Math.round(r.rect.h)}px tap target`;
|
|
237
262
|
const clippedTextDetail = (r) => `${label(r)} — text is clipped by its container`;
|
|
263
|
+
/** One wording for body-coloured links, page or shell alike. */
|
|
264
|
+
const indistinctDetail = (bodyColor) => `links with no underline in the body-text colour ${bodyColor}`;
|
|
238
265
|
/**
|
|
239
266
|
* Stable identity for one styled element, used to recognise the SAME component
|
|
240
267
|
* across routes.
|
|
@@ -282,6 +309,34 @@ function gridValues(top) {
|
|
|
282
309
|
.map((v) => `${v}px`)
|
|
283
310
|
.join(", ");
|
|
284
311
|
}
|
|
312
|
+
/** Inputs that are buttons, not fields. */
|
|
313
|
+
const BUTTON_INPUT_TYPES = new Set(["submit", "button", "reset", "image"]);
|
|
314
|
+
/** A button, or a link painted as one. Fields, selects and plain links never compete as actions. */
|
|
315
|
+
function buttonLike(r) {
|
|
316
|
+
if (r.tag === "button" || r.role === "button")
|
|
317
|
+
return true;
|
|
318
|
+
if (r.tag === "input")
|
|
319
|
+
return BUTTON_INPUT_TYPES.has(r.inputType);
|
|
320
|
+
return r.tag === "a" && r.filled;
|
|
321
|
+
}
|
|
322
|
+
/**
|
|
323
|
+
* The fields a user is asked to fill in and submit. Excluded: controls in a
|
|
324
|
+
* table row (a selection checkbox, or a select that edits the row in place),
|
|
325
|
+
* search boxes, and inputs that are buttons. When some fields sit in a <form>,
|
|
326
|
+
* only those count: the rest of the page (filters, toolbars) is not part of
|
|
327
|
+
* what gets submitted. A page with no <form> at all is judged on every field,
|
|
328
|
+
* because many apps build their forms without the element.
|
|
329
|
+
*/
|
|
330
|
+
function formFields(records) {
|
|
331
|
+
const candidates = records.filter((r) => r.interactive &&
|
|
332
|
+
(r.tag === "input" || r.tag === "select" || r.tag === "textarea") &&
|
|
333
|
+
!BUTTON_INPUT_TYPES.has(r.inputType) &&
|
|
334
|
+
r.inputType !== "search" &&
|
|
335
|
+
!r.inSearch &&
|
|
336
|
+
!r.inRow);
|
|
337
|
+
const inForm = candidates.filter((r) => r.inForm);
|
|
338
|
+
return inForm.length > 0 ? inForm : candidates;
|
|
339
|
+
}
|
|
285
340
|
/** Below the WCAG 2.2 target-size minimum; inline links are exempt by that rule. */
|
|
286
341
|
function tooSmall(r) {
|
|
287
342
|
return r.tag !== "a" && (r.rect.h < 24 || r.rect.w < 24) && r.rect.h > 0;
|
|
@@ -307,11 +362,16 @@ export function analyzeDesign(payload, viewport, chromeKeys = new Set()) {
|
|
|
307
362
|
return { report: "DESIGN AUDIT: no visible styled elements found (page empty or not hydrated).", score: null, signatures: [], defects: [] };
|
|
308
363
|
}
|
|
309
364
|
const signatures = allRecords.map(styleSignature);
|
|
310
|
-
|
|
365
|
+
// Chrome is what the census has seen on most routes OR what the markup puts
|
|
366
|
+
// in a shell landmark. The census alone needs several audited routes before
|
|
367
|
+
// it knows anything, so the first pages of a run were scored with the whole
|
|
368
|
+
// shell in them and later ones without — a page's score depended on when it
|
|
369
|
+
// was audited. The landmark half is known from the first audit on.
|
|
370
|
+
const isChrome = (r) => r.shell || chromeKeys.has(styleSignature(r));
|
|
311
371
|
const chromeRecords = allRecords.filter(isChrome);
|
|
312
372
|
// Everything below scores THIS page. `records` deliberately shadows the full
|
|
313
373
|
// set so no rule can accidentally reach past the page's own content.
|
|
314
|
-
const records =
|
|
374
|
+
const records = allRecords.filter((r) => !isChrome(r));
|
|
315
375
|
if (records.length === 0) {
|
|
316
376
|
return { report: "DESIGN AUDIT: this page is entirely shared layout chrome — nothing page-specific to score.", score: null, signatures, defects: [] };
|
|
317
377
|
}
|
|
@@ -567,7 +627,12 @@ export function analyzeDesign(payload, viewport, chromeKeys = new Set()) {
|
|
|
567
627
|
if (r.textLen > 40 && r.tag !== "a" && r.color !== "unknown")
|
|
568
628
|
bodyColorFreq.set(r.color, (bodyColorFreq.get(r.color) ?? 0) + 1);
|
|
569
629
|
const dominantBody = [...bodyColorFreq.entries()].sort((a, b) => b[1] - a[1])[0]?.[0];
|
|
570
|
-
const
|
|
630
|
+
const looksLikeBody = (r) => r.tag === "a" && r.textLen > 0 && !r.underline && r.color === dominantBody;
|
|
631
|
+
const indistinct = dominantBody ? records.filter(looksLikeBody) : [];
|
|
632
|
+
// Navigation links are this rule's commonest subject and usually sit in the
|
|
633
|
+
// shell, so the shell's are measured too, against the page's body colour, and
|
|
634
|
+
// reported as the shell's.
|
|
635
|
+
const chromeIndistinct = dominantBody ? chromeRecords.filter(looksLikeBody) : [];
|
|
571
636
|
if (dominantBody) {
|
|
572
637
|
if (indistinct.length > 0) {
|
|
573
638
|
affordances.push(`→ ${indistinct.length} link(s) with no underline AND the same color as body text (e.g. ${label(indistinct[0])}) — invisible as links`);
|
|
@@ -619,11 +684,14 @@ export function analyzeDesign(payload, viewport, chromeKeys = new Set()) {
|
|
|
619
684
|
// suite never answers.
|
|
620
685
|
const effort = [];
|
|
621
686
|
const pageBg = parseRgb(records.find((r) => r.bg && r.bg.startsWith("rgb"))?.bg ?? "") ?? [255, 255, 255, 1];
|
|
622
|
-
// A "prominent" action =
|
|
623
|
-
// from the page background, at a clickable
|
|
624
|
-
// on, so it is the page's implied primary
|
|
687
|
+
// A "prominent" action = a button (or a link painted as one) that paints its
|
|
688
|
+
// own background, clearly departing from the page background, at a clickable
|
|
689
|
+
// size. That is what the eye lands on, so it is the page's implied primary
|
|
690
|
+
// action. Fields never qualify — a white text field on a grey page departs
|
|
691
|
+
// from the page background too — nor does a ghost button showing a card's
|
|
692
|
+
// background, nor a breadcrumb, which is a way back rather than an action.
|
|
625
693
|
const prominent = records.filter((r) => {
|
|
626
|
-
if (!r.interactive || r.rect.w < 60 || r.rect.h < 24)
|
|
694
|
+
if (!r.interactive || !buttonLike(r) || !r.filled || r.inBreadcrumb || r.rect.w < 60 || r.rect.h < 24)
|
|
627
695
|
return false;
|
|
628
696
|
const bg = parseRgb(r.bg);
|
|
629
697
|
if (!bg)
|
|
@@ -647,7 +715,7 @@ export function analyzeDesign(payload, viewport, chromeKeys = new Set()) {
|
|
|
647
715
|
effort.push(`→ the primary action (${label(nearest)}) sits ${Math.round(nearest.rect.y - foldH)}px below the fold — the user must scroll before seeing what this page is for`);
|
|
648
716
|
}
|
|
649
717
|
// Form burden: how much is being asked, and how much of it is actually needed.
|
|
650
|
-
const fields = records
|
|
718
|
+
const fields = formFields(records);
|
|
651
719
|
if (fields.length >= 5) {
|
|
652
720
|
const req = fields.filter((r) => r.required).length;
|
|
653
721
|
if (req === 0) {
|
|
@@ -703,13 +771,14 @@ export function analyzeDesign(payload, viewport, chromeKeys = new Set()) {
|
|
|
703
771
|
const chromeSection = [];
|
|
704
772
|
const chromeDefects = [];
|
|
705
773
|
if (chromeRecords.length > 0) {
|
|
706
|
-
chromeDefects.push(...contrastFailures(chromeRecords).map((detail) => ({ rule: "contrast", detail, chrome: true })), ...chromeRecords.filter((r) => r.interactive && tooSmall(r)).map((r) => ({ rule: "tiny-target", detail: tinyTargetDetail(r), chrome: true })), ...chromeRecords.filter((r) => r.clipped && r.textLen > 0).map((r) => ({ rule: "clipped-text", detail: clippedTextDetail(r), chrome: true })));
|
|
707
|
-
|
|
774
|
+
chromeDefects.push(...contrastFailures(chromeRecords).map((detail) => ({ rule: "contrast", detail, chrome: true })), ...chromeRecords.filter((r) => r.interactive && tooSmall(r)).map((r) => ({ rule: "tiny-target", detail: tinyTargetDetail(r), chrome: true })), ...chromeRecords.filter((r) => r.clipped && r.textLen > 0).map((r) => ({ rule: "clipped-text", detail: clippedTextDetail(r), chrome: true })), ...(chromeIndistinct.length > 0 ? [{ rule: "indistinct-link", detail: indistinctDetail(dominantBody ?? ""), chrome: true }] : []));
|
|
775
|
+
// A convention (indistinct-link) is a → line, as it is on the page; the rest are measurable defects.
|
|
776
|
+
const chromeIssues = chromeDefects.map((d) => `${d.rule === "indistinct-link" ? "→" : "⚠"} ${d.detail}`);
|
|
708
777
|
if (chromeIssues.length > 0) {
|
|
709
778
|
chromeSection.push(`SHARED CHROME (${chromeRecords.length} shell elements, excluded from this page's score and reported here instead):\n` +
|
|
710
779
|
[...new Set(chromeIssues)]
|
|
711
780
|
.slice(0, CHROME_ISSUE_CAP)
|
|
712
|
-
.map((s) => `
|
|
781
|
+
.map((s) => ` ${s}`)
|
|
713
782
|
.join("\n") +
|
|
714
783
|
`\n → these belong to the app shell and recur on every page that renders it. File ONE finding for the shell, not one per page.`);
|
|
715
784
|
}
|
|
@@ -739,7 +808,7 @@ export function analyzeDesign(payload, viewport, chromeKeys = new Set()) {
|
|
|
739
808
|
// values on ten pages are one entry on ten routes, and the entry's fingerprint does not move when content does.
|
|
740
809
|
...(pad.n > 10 && pad.pct > 20 ? [{ rule: "off-grid-spacing", detail: `paddings off a 4px grid: ${gridValues(pad.top)}` }] : []),
|
|
741
810
|
...(mar.n > 10 && mar.pct > 20 ? [{ rule: "off-grid-spacing", detail: `vertical margins off a 4px grid: ${gridValues(mar.top)}` }] : []),
|
|
742
|
-
...(indistinct.length > 0 ? [{ rule: "indistinct-link", detail:
|
|
811
|
+
...(indistinct.length > 0 ? [{ rule: "indistinct-link", detail: indistinctDetail(dominantBody ?? "") }] : []),
|
|
743
812
|
...chromeDefects,
|
|
744
813
|
];
|
|
745
814
|
return { report, score, signatures, defects };
|
package/dist/engine/forms.js
CHANGED
|
@@ -173,6 +173,23 @@ export function submits(kind, probe) {
|
|
|
173
173
|
export function isEmptySubmit(kind, probe) {
|
|
174
174
|
return probe !== null && submits(kind, probe) && allTextEmpty(probe.fields);
|
|
175
175
|
}
|
|
176
|
+
/** The words that make a button submit-style, matched as whole words. */
|
|
177
|
+
const SUBMIT_WORDS = new Set(["submit", "send", "save", "create", "apply", "subscribe", "register", "sign", "signin", "signup", "post", "add"]);
|
|
178
|
+
/**
|
|
179
|
+
* Whether a button reads as one that sends something, by its name and test
|
|
180
|
+
* id. Whole words only: a test id is split on its separators and its camel
|
|
181
|
+
* case first, so "sign-in" and "Sign" count but the "sign" inside "assignee"
|
|
182
|
+
* and the "post" inside "postcode" do not.
|
|
183
|
+
*/
|
|
184
|
+
export function isSubmitLike(role, name, testid) {
|
|
185
|
+
if (role !== "button")
|
|
186
|
+
return false;
|
|
187
|
+
const words = `${name} ${testid ?? ""}`
|
|
188
|
+
.replace(/([a-z])([A-Z])/g, "$1 $2")
|
|
189
|
+
.toLowerCase()
|
|
190
|
+
.split(/[^a-z]+/);
|
|
191
|
+
return words.some((w) => SUBMIT_WORDS.has(w));
|
|
192
|
+
}
|
|
176
193
|
/**
|
|
177
194
|
* Whether a failed page read is the page going away under it (a navigation
|
|
178
195
|
* or a closed tab), which is expected after a submit and says nothing. Any
|