scenescout 3.14.1 → 3.15.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +19 -0
- package/README.md +26 -3
- package/dist/check-run.js +22 -11
- package/dist/ci-run.js +113 -17
- package/dist/cli.js +119 -6
- package/dist/commands.js +39 -2
- package/dist/engine/browser.js +546 -133
- package/dist/engine/check.js +77 -30
- package/dist/engine/ci.js +97 -0
- package/dist/engine/claims.js +81 -15
- package/dist/engine/collector.js +307 -65
- package/dist/engine/dedup.js +329 -9
- package/dist/engine/design.js +81 -12
- package/dist/engine/forms.js +17 -0
- package/dist/engine/memory.js +180 -36
- package/dist/engine/oracles.js +126 -1
- package/dist/engine/provider.js +4 -3
- package/dist/engine/refresh.js +270 -22
- package/dist/engine/report.js +8 -1
- package/dist/engine/scripted-login.js +342 -62
- package/dist/first-run.js +577 -0
- package/dist/login-run.js +136 -35
- package/dist/mcp-server.js +75 -16
- package/package.json +1 -1
- package/skills/scenescout/SKILL.md +3 -3
package/dist/login-run.js
CHANGED
|
@@ -20,7 +20,7 @@ import { explicitLimits, isTimeoutMessage, LIMIT_NAMES } from "./engine/limits.j
|
|
|
20
20
|
import { redactRoute } from "./engine/check.js";
|
|
21
21
|
import { describeLifetime, readLifetime } from "./engine/expiry.js";
|
|
22
22
|
import { describeSaved, mergeSessionStorage, withSessionStorage, writeProfile } from "./engine/profiles.js";
|
|
23
|
-
import {
|
|
23
|
+
import { afterTyping, chooseFields, describeStep, fieldIdentity, filledNames, nextStep, otpBoxes, quotable, secondsLeft, splitCode, totp as totpCode, TOTP_MIN_SECONDS_LEFT, urlMatches, } from "./engine/scripted-login.js";
|
|
24
24
|
/**
|
|
25
25
|
* Read what the profile keeps from a signed-in context: cookies, localStorage
|
|
26
26
|
* and IndexedDB through Playwright's storage state, and sessionStorage, which
|
|
@@ -250,22 +250,37 @@ function collectFields(selectors) {
|
|
|
250
250
|
}
|
|
251
251
|
return out;
|
|
252
252
|
}
|
|
253
|
-
/**
|
|
253
|
+
/**
|
|
254
|
+
* The error or alert a page shows, if any: the first one on screen with text
|
|
255
|
+
* in it (a hidden, pre-rendered alert says nothing about this run), whole.
|
|
256
|
+
* The caller redacts it before shortening it (quotable).
|
|
257
|
+
*/
|
|
254
258
|
async function alertText(page) {
|
|
255
|
-
|
|
259
|
+
return (page
|
|
256
260
|
.evaluate(() => {
|
|
257
|
-
const el
|
|
258
|
-
|
|
261
|
+
for (const el of Array.from(document.querySelectorAll('[role="alert"], [aria-live="assertive"], .error, .alert'))) {
|
|
262
|
+
const r = el.getBoundingClientRect();
|
|
263
|
+
const st = getComputedStyle(el);
|
|
264
|
+
const text = (el.textContent ?? "").trim();
|
|
265
|
+
if (text && r.width > 0 && r.height > 0 && st.visibility !== "hidden" && st.display !== "none")
|
|
266
|
+
return text;
|
|
267
|
+
}
|
|
268
|
+
return "";
|
|
259
269
|
})
|
|
260
|
-
// The page's message only adds to the
|
|
261
|
-
.catch(() => "");
|
|
262
|
-
return text.slice(0, 200);
|
|
270
|
+
// The page's message only adds to the error being reported; a page navigating away as it is read leaves that error as it stands.
|
|
271
|
+
.catch(() => ""));
|
|
263
272
|
}
|
|
264
273
|
/**
|
|
265
274
|
* Sign in with no one at the keyboard: headless, the credentials from the
|
|
266
275
|
* environment, the form found by engine/scripted-login.ts's rules. Every line
|
|
267
276
|
* it logs and every error it throws goes through the credential redaction
|
|
268
277
|
* first. Saves the profile exactly as the manual login does.
|
|
278
|
+
*
|
|
279
|
+
* Each step fills what the page asks for, then reads the page again until it
|
|
280
|
+
* is clear how to submit (engine/scripted-login.ts's afterTyping): a page that
|
|
281
|
+
* submitted the step itself is not submitted again, a button the page enables
|
|
282
|
+
* only once the form is complete is waited for, and a field the typing
|
|
283
|
+
* revealed is filled first.
|
|
269
284
|
*/
|
|
270
285
|
export async function runScriptedLogin(options, config, redactor, log, now = () => Date.now() / 1000) {
|
|
271
286
|
const redact = (text) => redactor.redact(text);
|
|
@@ -330,7 +345,7 @@ export async function runScriptedLogin(options, config, redactor, log, now = ()
|
|
|
330
345
|
const got = await page.evaluate(collectFields, config.selectors).catch((err) => {
|
|
331
346
|
if (midNavigation(err))
|
|
332
347
|
return null;
|
|
333
|
-
throw err;
|
|
348
|
+
throw fail(err.message);
|
|
334
349
|
});
|
|
335
350
|
if (got === null)
|
|
336
351
|
return null;
|
|
@@ -345,19 +360,49 @@ export async function runScriptedLogin(options, config, redactor, log, now = ()
|
|
|
345
360
|
return `${pageKey()}|${Object.keys(chosen).sort().join(",")}`;
|
|
346
361
|
};
|
|
347
362
|
const stepOpts = (successMatchedNow) => ({
|
|
348
|
-
|
|
363
|
+
hasPassword: config.password !== undefined,
|
|
364
|
+
...(config.code ? { code: config.code.kind } : {}),
|
|
349
365
|
successConfigured: Boolean(config.success.url || config.success.selector),
|
|
350
366
|
successMatched: successMatchedNow,
|
|
351
367
|
});
|
|
368
|
+
/** The code to type now: the fixed one, or the TOTP code, waiting for the next one when this one is about to expire. */
|
|
369
|
+
const currentCode = async (source) => {
|
|
370
|
+
if (source.kind === "fixed")
|
|
371
|
+
return source.code;
|
|
372
|
+
const params = source.params;
|
|
373
|
+
if (secondsLeft(params, now()) < TOTP_MIN_SECONDS_LEFT)
|
|
374
|
+
await page.waitForTimeout(secondsLeft(params, now()) * 1000 + 200);
|
|
375
|
+
const code = totpCode(params, now());
|
|
376
|
+
redactor.add(code);
|
|
377
|
+
return code;
|
|
378
|
+
};
|
|
379
|
+
const at = (f) => page.locator(`[${TAG}="${f.index}"]`);
|
|
380
|
+
/** A failed action as this run's error: the time limit explained, redacted like everything else it throws. */
|
|
381
|
+
const actionError = (err) => {
|
|
382
|
+
const explained = loginTimeout(err, "action", limits.actionMs);
|
|
383
|
+
return fail(explained instanceof Error ? explained.message : String(explained));
|
|
384
|
+
};
|
|
385
|
+
const actionFailed = (err) => Promise.reject(actionError(err));
|
|
386
|
+
/** What the page says, fit to quote. */
|
|
387
|
+
const pageSays = async () => quotable(await alertText(page), redact, 200);
|
|
388
|
+
/** A button's text fit to quote, redacted before it is shortened. */
|
|
389
|
+
const buttonText = (f) => quotable(f.text || f.label || "the submit button", redact, 40);
|
|
390
|
+
/** The run's deadline passed: say what it was waiting for, and what the page says, if anything. */
|
|
391
|
+
const timedOut = async (waitingFor) => {
|
|
392
|
+
const says = await pageSays();
|
|
393
|
+
return fail(`timed out after ${config.timeoutMs / 1000}s waiting for ${waitingFor} (at ${where()})${says ? `. The page says: "${says}"` : ""}`);
|
|
394
|
+
};
|
|
395
|
+
const submitOpts = { passwordless: config.password === undefined };
|
|
352
396
|
let lastWait = "the sign-in form to appear";
|
|
353
397
|
for (;;) {
|
|
354
398
|
if (Date.now() > deadline)
|
|
355
|
-
throw
|
|
399
|
+
throw await timedOut(lastWait);
|
|
356
400
|
const fields = await read();
|
|
357
401
|
if (fields === null) {
|
|
358
402
|
await page.waitForTimeout(POLL_MS);
|
|
359
403
|
continue;
|
|
360
404
|
}
|
|
405
|
+
const keyAtRead = pageKey();
|
|
361
406
|
const chosen = chooseFields(fields);
|
|
362
407
|
const step = nextStep(chosen, progress, stepOpts(await successMatched()));
|
|
363
408
|
if (step.kind === "done") {
|
|
@@ -367,7 +412,7 @@ export async function runScriptedLogin(options, config, redactor, log, now = ()
|
|
|
367
412
|
// a page between a redirect and its first render shows no fields either.
|
|
368
413
|
await page.waitForLoadState("load", { timeout: limits.actionMs }).catch((err) => {
|
|
369
414
|
if (!/timeout/i.test(err.message) && !midNavigation(err))
|
|
370
|
-
throw err;
|
|
415
|
+
throw fail(err.message);
|
|
371
416
|
});
|
|
372
417
|
await page.waitForTimeout(POLL_MS * 2);
|
|
373
418
|
const settled = await read();
|
|
@@ -376,48 +421,104 @@ export async function runScriptedLogin(options, config, redactor, log, now = ()
|
|
|
376
421
|
continue;
|
|
377
422
|
}
|
|
378
423
|
if (step.kind === "refused") {
|
|
379
|
-
const says = await
|
|
424
|
+
const says = await pageSays();
|
|
380
425
|
throw fail(`sign-in refused: ${step.reason}.${says ? ` The page says: "${says}"` : ""}`);
|
|
381
426
|
}
|
|
382
427
|
if (step.kind === "stuck")
|
|
383
428
|
throw fail(`sign-in stuck: ${step.reason}.`);
|
|
384
429
|
if (step.kind === "wait") {
|
|
430
|
+
const credentialSent = progress.submitted.has("password") || progress.submitted.has("otp");
|
|
385
431
|
lastWait =
|
|
386
432
|
progress.submits === 0
|
|
387
433
|
? "the sign-in form to appear (set the field selectors if the form is not found)"
|
|
388
|
-
:
|
|
434
|
+
: !credentialSent
|
|
435
|
+
? "the page to ask for the password or the one-time code (set --password-selector or --otp-selector if the field is there but not found)"
|
|
436
|
+
: "the signed-in page (the success URL or selector, or the sign-in fields to go)";
|
|
389
437
|
await page.waitForTimeout(POLL_MS);
|
|
390
438
|
continue;
|
|
391
439
|
}
|
|
392
440
|
if (progress.submits >= MAX_SUBMITS)
|
|
393
441
|
throw fail(`gave up after ${MAX_SUBMITS} submits without reaching the signed-in page`);
|
|
394
|
-
let
|
|
442
|
+
let boxOpts = {};
|
|
443
|
+
// What this step typed, recorded as submitted only once the step is: a step the page holds back (fill-more) is typed again.
|
|
444
|
+
const typedIds = new Map();
|
|
395
445
|
for (const kind of step.fill) {
|
|
396
446
|
const field = chosen[kind];
|
|
397
|
-
|
|
398
|
-
if (
|
|
399
|
-
const
|
|
400
|
-
if (
|
|
401
|
-
|
|
402
|
-
|
|
403
|
-
|
|
447
|
+
const boxes = kind === "otp" ? otpBoxes(fields, field) : null;
|
|
448
|
+
if (boxes) {
|
|
449
|
+
const split = splitCode(await currentCode(config.code), boxes.length, config.code.kind);
|
|
450
|
+
if (!split.ok)
|
|
451
|
+
throw fail(`sign-in stuck: ${split.error}.`);
|
|
452
|
+
for (const [i, box] of boxes.entries()) {
|
|
453
|
+
try {
|
|
454
|
+
// A box is often enabled only once the one before it is filled.
|
|
455
|
+
await page.waitForFunction(([attr, index]) => {
|
|
456
|
+
const el = document.querySelector(`[${attr}="${index}"]`);
|
|
457
|
+
return el !== null && !el.disabled;
|
|
458
|
+
}, [TAG, String(box.index)], { timeout: limits.actionMs, polling: 50 });
|
|
459
|
+
// Cleared only when it holds something: clearing sends Delete, which some boxes take as a step back to the box before.
|
|
460
|
+
if ((await at(box).inputValue({ timeout: limits.actionMs })) !== "")
|
|
461
|
+
await at(box).fill("", { timeout: limits.actionMs });
|
|
462
|
+
// Typed as keys: a box that moves on to the next by itself, or reads keys rather than its value, still gets its character.
|
|
463
|
+
await at(box).pressSequentially(split.chars[i], { timeout: limits.actionMs });
|
|
464
|
+
}
|
|
465
|
+
catch (err) {
|
|
466
|
+
// The first line only: the call log after it names the character being typed, which no redaction of the whole code catches.
|
|
467
|
+
const explained = loginTimeout(err, "action", limits.actionMs);
|
|
468
|
+
const first = (explained instanceof Error ? explained.message : String(explained)).split("\n")[0];
|
|
469
|
+
throw fail(`could not type into box ${i + 1} of ${boxes.length} of the one-time code: ${first}`);
|
|
470
|
+
}
|
|
471
|
+
}
|
|
472
|
+
boxOpts = { codeBoxes: boxes.length };
|
|
404
473
|
}
|
|
405
|
-
else
|
|
406
|
-
value = kind === "username" ? config.username : config.password;
|
|
407
|
-
|
|
408
|
-
|
|
409
|
-
|
|
410
|
-
.catch((err) => Promise.reject(loginTimeout(err, "action", limits.actionMs)));
|
|
411
|
-
progress.submitted.set(kind, fieldIdentity(field));
|
|
412
|
-
last = field;
|
|
474
|
+
else {
|
|
475
|
+
const value = kind === "otp" ? await currentCode(config.code) : kind === "username" ? config.username : config.password;
|
|
476
|
+
await at(field).fill(value, { timeout: limits.actionMs }).catch(actionFailed);
|
|
477
|
+
}
|
|
478
|
+
typedIds.set(kind, fieldIdentity(field));
|
|
413
479
|
}
|
|
414
|
-
const button = chooseSubmit(fields);
|
|
415
480
|
const before = signature(fields);
|
|
416
|
-
|
|
417
|
-
|
|
418
|
-
|
|
481
|
+
const shownBefore = new Set(Object.keys(chosen));
|
|
482
|
+
const readNow = async () => {
|
|
483
|
+
const now = await read();
|
|
484
|
+
return afterTyping(step.fill, { left: now === null || pageKey() !== keyAtRead, fields: now ?? [] }, shownBefore, submitOpts);
|
|
485
|
+
};
|
|
486
|
+
let next;
|
|
487
|
+
for (;;) {
|
|
488
|
+
await page.waitForTimeout(POLL_MS);
|
|
489
|
+
next = await readNow();
|
|
490
|
+
if (next.kind !== "wait")
|
|
491
|
+
break;
|
|
492
|
+
if (Date.now() > deadline) {
|
|
493
|
+
const control = next.button ? `the "${buttonText(next.button)}" button` : "the field just filled";
|
|
494
|
+
throw await timedOut(`${control} to be enabled, or the page to move on, after filling ${filledNames(step.fill, boxOpts)}`);
|
|
495
|
+
}
|
|
496
|
+
}
|
|
497
|
+
if (next.kind === "fill-more") {
|
|
498
|
+
say(describeStep(step.fill, "more", boxOpts));
|
|
499
|
+
continue;
|
|
500
|
+
}
|
|
501
|
+
let by = "page";
|
|
502
|
+
if (next.kind === "click" || next.kind === "enter") {
|
|
503
|
+
try {
|
|
504
|
+
if (next.kind === "click")
|
|
505
|
+
await at(next.button).click({ timeout: limits.actionMs });
|
|
506
|
+
else
|
|
507
|
+
await at(next.field).press("Enter", { timeout: limits.actionMs });
|
|
508
|
+
}
|
|
509
|
+
catch (err) {
|
|
510
|
+
// The page moved on while the click waited (it took the step itself, or the click went through as it left): the step is
|
|
511
|
+
// done. Otherwise the click's own error is the one to report, also when reading the page to tell fails.
|
|
512
|
+
const moved = await readNow().then((after) => after.kind === "moved", () => false);
|
|
513
|
+
if (!moved)
|
|
514
|
+
throw actionError(err);
|
|
515
|
+
}
|
|
516
|
+
by = next.kind === "click" ? { button: buttonText(next.button) } : "enter";
|
|
517
|
+
}
|
|
518
|
+
for (const [kind, identity] of typedIds)
|
|
519
|
+
progress.submitted.set(kind, identity);
|
|
419
520
|
progress.submits += 1;
|
|
420
|
-
const did = describeStep(step.fill,
|
|
521
|
+
const did = describeStep(step.fill, by, boxOpts);
|
|
421
522
|
say(did);
|
|
422
523
|
lastWait = `the page to move on after that step (${did})`;
|
|
423
524
|
// Wait for the form to move on: another page, or other fields on this one.
|
package/dist/mcp-server.js
CHANGED
|
@@ -3,8 +3,12 @@
|
|
|
3
3
|
* SceneScout MCP server (stdio).
|
|
4
4
|
*
|
|
5
5
|
* Exposes deterministic browser-exploration tools — Playwright actions, state
|
|
6
|
-
* memory, oracles, findings, report — to any MCP client.
|
|
7
|
-
*
|
|
6
|
+
* memory, oracles, findings, report — to any MCP client. The client (e.g.
|
|
7
|
+
* Claude Code on a subscription) is the brain, and no LLM call happens here
|
|
8
|
+
* unless finding dedup is set to ask a model judge (SCENESCOUT_DEDUP, or
|
|
9
|
+
* scout_attach {dedup}): then a filing the dedup rule keeps apart is put to a
|
|
10
|
+
* model, through the client when it is a `scenescout ci` run, else with a key
|
|
11
|
+
* from this process's environment (engine/dedup.ts).
|
|
8
12
|
*
|
|
9
13
|
* The server process is a per-conversation daemon and behaves like one:
|
|
10
14
|
* - Multi-session, genuinely concurrent: named sessions each own a live
|
|
@@ -35,6 +39,9 @@ import { z } from "zod";
|
|
|
35
39
|
import { BrowserEngine } from "./engine/browser.js";
|
|
36
40
|
import { reapOrphanBrowsers } from "./engine/reaper.js";
|
|
37
41
|
import { FINDING_CATEGORIES, isWorthALook, MemoryStore, mergeableCategories, redactSecrets } from "./engine/memory.js";
|
|
42
|
+
import { DEDUP_ENV, DEDUP_MODES, redactKeys, secretValues } from "./engine/ci.js";
|
|
43
|
+
import { DedupJudge, planDedup, samplingAsk } from "./engine/dedup.js";
|
|
44
|
+
import { httpJudgeAsk } from "./ci-run.js";
|
|
38
45
|
import { decodedEntitiesNote, ignoredConventionsNote, LANE_CONVENTION_MAX, LANE_NAME_MAX, LaneLedger, laneCloseGuard, laneReportInstruction, parseLaneReport, summarizeLaneReport, } from "./engine/lane.js";
|
|
39
46
|
import { MAX_UNFILED_NAMED, unfiledDefects } from "./engine/calibration.js";
|
|
40
47
|
import { SessionQueue, withWatchdog } from "./engine/dispatch.js";
|
|
@@ -746,8 +753,15 @@ server.registerTool("scout_attach", {
|
|
|
746
753
|
.max(40)
|
|
747
754
|
.optional()
|
|
748
755
|
.describe("Session name for multi-role runs (e.g. 'admin', 'qa'). Creates/replaces that session's browser and makes it the default. Default: 'default'."),
|
|
756
|
+
dedup: z
|
|
757
|
+
.enum(DEDUP_MODES)
|
|
758
|
+
.optional()
|
|
759
|
+
.describe(`How this run tells a filed finding from one already recorded. Default: the ${DEDUP_ENV} environment variable, else 'rule', the store's rule alone. ` +
|
|
760
|
+
"'judge': the rule first, then, for a filing the rule keeps apart from everything recorded, a model is asked whether it is one of the open findings on its page, and merges it when it says so. " +
|
|
761
|
+
"It needs ANTHROPIC_API_KEY or OPENAI_API_KEY in the server's environment, and sends each pair's titles, categories and evidence, and the page's path, to that provider — ONLY when the user asked for it. " +
|
|
762
|
+
"Applies to every session of the project until the run ends."),
|
|
749
763
|
},
|
|
750
|
-
}, serializedControl(async ({ url, projectPath, storageStatePath, role, mode, headed, browser, viewportWidth, viewportHeight, objective, task, record, trustedEmbeds, paceMs, actionTimeoutMs, navTimeoutMs, session, }) => {
|
|
764
|
+
}, serializedControl(async ({ url, projectPath, storageStatePath, role, mode, headed, browser, viewportWidth, viewportHeight, objective, task, record, trustedEmbeds, paceMs, actionTimeoutMs, navTimeoutMs, session, dedup, }) => {
|
|
751
765
|
try {
|
|
752
766
|
const target = session ?? activeName;
|
|
753
767
|
if (session) {
|
|
@@ -813,6 +827,8 @@ server.registerTool("scout_attach", {
|
|
|
813
827
|
const previous = eng.memory;
|
|
814
828
|
if (previous && previous !== store && ![...engines.values()].some((e) => e !== eng && e.memory === previous))
|
|
815
829
|
previous.endRun();
|
|
830
|
+
// Before the browser starts: a value it cannot use refuses the attach rather than being replaced.
|
|
831
|
+
const dedupNote = configureDedup(store, dedup);
|
|
816
832
|
const viewport = viewportWidth && viewportHeight ? { width: viewportWidth, height: viewportHeight } : undefined;
|
|
817
833
|
const out = await eng.attach({
|
|
818
834
|
url,
|
|
@@ -849,12 +865,61 @@ server.registerTool("scout_attach", {
|
|
|
849
865
|
const recordNote = record && eng.memory?.dir
|
|
850
866
|
? `\n\n📸 RECORDING: a frame of the page after each action, under ${path.join(eng.memory.dir, "recordings", target)}/ (at most ${RECORD_MAX_FRAMES}). scout_report writes them into report.html beside report.md.`
|
|
851
867
|
: "";
|
|
852
|
-
return text(out + conflictNote + recordNote + describePace(eng.pace) + (engines.size > 1 ? `\n${sessionLines()}` : "") + liveLine(), target);
|
|
868
|
+
return text(out + conflictNote + recordNote + dedupNote + describePace(eng.pace) + (engines.size > 1 ? `\n${sessionLines()}` : "") + liveLine(), target);
|
|
853
869
|
}
|
|
854
870
|
catch (err) {
|
|
855
871
|
return errorText(err);
|
|
856
872
|
}
|
|
857
873
|
}));
|
|
874
|
+
/** Keys in this process's environment, and anything shaped like one, taken out of a line before it is shown. */
|
|
875
|
+
const withoutKeys = (text) => redactKeys(text, secretValues(process.env));
|
|
876
|
+
/** A line for the operator, on stderr. */
|
|
877
|
+
function logLine(line) {
|
|
878
|
+
console.error(withoutKeys(`[scenescout] ${line}`));
|
|
879
|
+
}
|
|
880
|
+
/**
|
|
881
|
+
* Set how a project's filings are deduplicated for this run (engine/dedup.ts
|
|
882
|
+
* planDedup): what this attach names, else what an earlier attach of the run
|
|
883
|
+
* named, else DEDUP_ENV. Returns the line the attach reports, or "".
|
|
884
|
+
*/
|
|
885
|
+
function configureDedup(store, asked) {
|
|
886
|
+
if (asked)
|
|
887
|
+
store.dedupChoice = asked;
|
|
888
|
+
const plan = planDedup(store.dedupChoice, process.env, server.server.getClientCapabilities());
|
|
889
|
+
if (plan.mode === "rule") {
|
|
890
|
+
store.dedupJudge = null;
|
|
891
|
+
store.dedupOff = undefined;
|
|
892
|
+
return "";
|
|
893
|
+
}
|
|
894
|
+
// On already for this run: say so only when it has since been switched off.
|
|
895
|
+
if (store.dedupJudge instanceof DedupJudge)
|
|
896
|
+
return store.dedupJudge.off ? `\n⚠ DEDUP JUDGE OFF: switched off earlier in this run after ${store.dedupJudge.off}. The rule decides duplicates.` : "";
|
|
897
|
+
if (plan.via === "off") {
|
|
898
|
+
store.dedupOff = plan.why;
|
|
899
|
+
logLine(`dedup judge off: ${plan.why}; the rule decides duplicates`);
|
|
900
|
+
return plan.note;
|
|
901
|
+
}
|
|
902
|
+
const ask = plan.via === "client" ? samplingAsk((params, options) => server.server.createMessage(params, options)) : httpJudgeAsk(plan.resolved, plan.key);
|
|
903
|
+
store.dedupJudge = new DedupJudge(ask, { label: plan.label, log: logLine, redact: withoutKeys });
|
|
904
|
+
store.dedupOff = undefined;
|
|
905
|
+
return plan.note;
|
|
906
|
+
}
|
|
907
|
+
/** What scout_finding tells the agent about what filing did. */
|
|
908
|
+
function filedText(filed, category) {
|
|
909
|
+
const { finding, isNew, promoted } = filed;
|
|
910
|
+
if (isNew && isWorthALook(finding))
|
|
911
|
+
return `Recorded as worth a look (not a defect in the report's totals): ${finding.title} (id ${finding.id}) — a defect only if your project uses ${finding.convention}`;
|
|
912
|
+
if (isNew)
|
|
913
|
+
return `Finding recorded: [${finding.severity}] ${finding.title} (id ${finding.id})`;
|
|
914
|
+
if (filed.judged)
|
|
915
|
+
return (`Not recorded as new: the dedup judge read it as the same defect as finding ${finding.id} (p_same ${filed.judged.pSame.toFixed(2)}) — [${finding.severity}] ${finding.title}, filed as ${finding.category}, seen in ${finding.runs} runs.` +
|
|
916
|
+
`${promoted ? " That finding was worth a look and is now a defect." : ""} Its title, category, severity and evidence are kept on that finding, so the report shows the merge. If yours is a different defect, file it again with a title that says what differs.`);
|
|
917
|
+
if (promoted)
|
|
918
|
+
return `Merged into finding ${finding.id}, which was worth a look, and promoted to a defect: [${finding.severity}] ${finding.title}. It now counts among the report's findings.`;
|
|
919
|
+
if (finding.regressedAt)
|
|
920
|
+
return `⟳ REOPENED as a REGRESSION: finding ${finding.id} was previously resolved but the evidence reproduces again (seen in ${finding.runs} runs). Worth calling out to the user.`;
|
|
921
|
+
return `Not recorded as new: merged into existing finding ${finding.id} — [${finding.severity}] ${finding.title}${finding.evidence ? ` (evidence: ${finding.evidence.slice(0, 160)})` : " (no evidence)"}, filed as ${finding.category}, seen in ${finding.runs} runs. If yours is a different bug, file it again: under the category that says what is wrong if it is another kind of defect (a finding filed as ${category} merges only with one filed as ${mergeableCategories(category).join(" or ")}), or with evidence naming the request that failed for you (method and path) — two findings are kept apart when both name requests and none is shared.`;
|
|
922
|
+
}
|
|
858
923
|
function sessionLines() {
|
|
859
924
|
const lines = ["Live sessions:"];
|
|
860
925
|
for (const [name, eng] of engines) {
|
|
@@ -911,7 +976,7 @@ server.registerTool("scout_session", {
|
|
|
911
976
|
}
|
|
912
977
|
}));
|
|
913
978
|
server.registerTool("scout_snapshot", {
|
|
914
|
-
description: "Capture the current page state: URL, state fingerprint, interactable elements with refs (e1, e2, …), geometry issues, coverage, and oracle violations since the last action. Re-snapshotting the same route returns a DIFF (refs stay stable). Cheap — prefer this over screenshots.",
|
|
979
|
+
description: "Capture the current page state: URL, state fingerprint, a one-line summary of the main area's heading and text, interactable elements with refs (e1, e2, …) and their state (pressed, selected, checked, expanded, current), what the page announces (alert and status regions, by their text), geometry issues, coverage, and oracle violations since the last action. Re-snapshotting the same route returns a DIFF (refs stay stable). Cheap — prefer this over screenshots.",
|
|
915
980
|
inputSchema: {
|
|
916
981
|
full: z.boolean().default(false).describe("Force a full element list instead of a diff"),
|
|
917
982
|
session: sessionParam,
|
|
@@ -1280,9 +1345,9 @@ server.registerTool("scout_finding", {
|
|
|
1280
1345
|
const eng = engineFor(session);
|
|
1281
1346
|
if (!eng.memory)
|
|
1282
1347
|
throw new Error("Not attached — findings need an active session.");
|
|
1283
|
-
const
|
|
1348
|
+
const filed = await eng.memory.fileFinding({
|
|
1284
1349
|
severity,
|
|
1285
|
-
category
|
|
1350
|
+
category,
|
|
1286
1351
|
title,
|
|
1287
1352
|
detail,
|
|
1288
1353
|
evidence,
|
|
@@ -1291,15 +1356,9 @@ server.registerTool("scout_finding", {
|
|
|
1291
1356
|
state: eng.currentState || "(unknown)",
|
|
1292
1357
|
session: eng.sessionKey,
|
|
1293
1358
|
});
|
|
1294
|
-
|
|
1295
|
-
|
|
1296
|
-
|
|
1297
|
-
: `Finding recorded: [${finding.severity}] ${finding.title} (id ${finding.id})`
|
|
1298
|
-
: promoted
|
|
1299
|
-
? `Merged into finding ${finding.id}, which was worth a look, and promoted to a defect: [${finding.severity}] ${finding.title}. It now counts among the report's findings.`
|
|
1300
|
-
: finding.regressedAt
|
|
1301
|
-
? `⟳ REOPENED as a REGRESSION: finding ${finding.id} was previously resolved but the evidence reproduces again (seen in ${finding.runs} runs). Worth calling out to the user.`
|
|
1302
|
-
: `Not recorded as new: merged into existing finding ${finding.id} — [${finding.severity}] ${finding.title}${finding.evidence ? ` (evidence: ${finding.evidence.slice(0, 160)})` : " (no evidence)"}, filed as ${finding.category}, seen in ${finding.runs} runs. If yours is a different bug, file it again: under the category that says what is wrong if it is another kind of defect (a finding filed as ${category} merges only with one filed as ${mergeableCategories(category).join(" or ")}), or with evidence naming the request that failed for you (method and path) — two findings are kept apart when both name requests and none is shared.`, session);
|
|
1359
|
+
if (filed.judgeError)
|
|
1360
|
+
logLine(`dedup judge: ${filed.judgeError}; the rule decided`);
|
|
1361
|
+
return text(filedText(filed, category), session);
|
|
1303
1362
|
}
|
|
1304
1363
|
catch (err) {
|
|
1305
1364
|
return errorText(err);
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "scenescout",
|
|
3
|
-
"version": "3.
|
|
3
|
+
"version": "3.15.0",
|
|
4
4
|
"description": "SceneScout — exploratory UI testing for AI coding agents. An MCP server that gives any agent (Claude Code, Cursor, VS Code Copilot, Codex, Gemini CLI and others) a structured view of a running web app, always-on oracles, a network-level write policy, memory across runs and a gap-checked report.",
|
|
5
5
|
"license": "MIT",
|
|
6
6
|
"author": "brunoboto96",
|
|
@@ -34,8 +34,8 @@ You are the brain of an exploratory UI tester. The SceneScout MCP server gives y
|
|
|
34
34
|
1. **`scout_crawl` first, always.** One call visits every known route (pass `paths` to sweep a specific subset instead), records coverage, and returns per-route health. This is the whole breadth pass — do not visit routes one-by-one with navigate+snapshot.
|
|
35
35
|
2. **Investigate what the crawl flagged.** For each problem route (violations, dead-ends, auth-redirects): navigate there, `scout_snapshot`, reproduce, then `scout_finding`.
|
|
36
36
|
3. **Run journeys with `scout_run_plan {steps}`.** Mechanical sequences (fill form → submit → check) go in ONE plan call — `steps` is an ordered list of `{action, target, value}` — with `testid=`/`text=`/`label=`/`role=button[name="Save"]` targets — not one LLM turn per click. The plan aborts at the first violation and tells you where; that's your cue to investigate interactively. When the user wants a flow kept working, save the plan that walked it as `.scenescout/flows/<name>.json` — `{"name", "steps"}`, the same steps, starting with a `navigate`, plus `{"action":"expect-text","text"}`, `{"action":"expect-url","pattern"}` or `{"action":"expect-request","request":"GET /api/…","status":200}` where the outcome shows — and `scenescout check` replays it on every pull request. Save only flows that read: a check replays flows under observe's rule unless the project passes `--flow-writes allow`, so a step that sends a write is reported as "could not run" and the check exits 2.
|
|
37
|
-
4. **Snapshot economics:** `scout_snapshot` after landing somewhere new; re-snapshots of the same route return *diffs* with stable refs — "No element changes" costs you almost nothing. `scout_screenshot` ONLY for suspected pixel-native issues (a canvas, a rendering glitch); geometry problems (overlap, off-screen, a covered control) are already in the snapshot as GEOMETRY issues, and images that failed to load are listed under BROKEN IMAGES — file those, quoting the line. `scout_capture {ref}` saves a PNG of one element (plus a `margin` of pixels around it) only when the user asks to see how something looks; it shows, it does not judge. Controls inside embeds (`<iframe>`) are listed with the page's, each marked `⟨in … frame⟩`, and act by ref like any other. A frame of the app's own origin is the app: test it fully. A frame of another site (`cross-origin`) is someone else's system that the user has not authorised you to test: click and type there as a user would, to see the embed render and respond, but never send it hostile input — the engine refuses markup, values over 200 characters, control characters, repeated-click probes and uploads there, and you must not try SQL, template or other injection shapes either — and file what you find there as the embed's behaviour, naming its origin, not as the app's bug. A failing request such a frame sent outside the app says so (`in an embed of …`) and is kept at medium; console errors cannot be told apart by frame and stay the app's; its controls are counted apart and never enter the app's coverage or gap ledger. Its container text is masked, and the writes it sends outside the app are refused in every mode but destructive — unless the user names that origin as a trusted embed (a provider in test mode, say): pass it in `scout_attach {trustedEmbeds: ["https://…"]}` and, in safe-write only, its writes go out. Never add an origin the user did not name. The FRAMES line names frames that were not read; say in your summary that their contents were not explored.
|
|
38
|
-
5. **Native-user behaviours.** `scout_type {ref, textValue}` (or its alias `value`, matching `scout_select` and a plan step) APPENDS when a field already has content (menu clicks often insert @-mention chips or commands into composers — appending preserves them; the result reports what was already there); pass `replace=true` only to deliberately clear, and `pressEnter=true` to submit from the field the way a user would. Before concluding a badge, icon, or "N errors" indicator *does nothing*, `scout_hover` it — tooltips and hover cards are invisible to snapshots and clicks, and hover output includes what appeared. In HEADED mode (`scout_attach {headed:true}`, which the user asks for when they want to watch) the user's physical mouse competes with the synthetic pointer: if a hover reveals nothing and the finding matters, ask the user to move their mouse off the browser window and retry before filing. **Scroll long pages with `scout_scroll`** — the design audit and snapshot measure at the current scroll position, so judge deep sections by scrolling then re-auditing; it refuses to scroll where a real user couldn't and reports SCROLL LOCKED (the leaked modal scroll-lock that silently amputates everything below the fold — snapshots also flag it passively as an OVERLAY line), and scrolling triggers lazy-loaded content whose failures surface as fresh oracle violations. Elements fully clipped inside an overflow-hidden container are flagged UNREACHABLE in GEOMETRY issues — no amount of scrolling reveals them; that's a high-value layout bug, distinct from merely below-the-fold content. **A page can hold SEVERAL independent scroll regions** and plain `scout_scroll` moves the largest one, so a sidebar nav beside a taller main pane never budges: pass `scout_scroll {target:"testid=…"}` to scroll one region. Never report a nav item, tab or list row as missing/truncated until you have scrolled ITS container — content scrolled out of a secondary pane looks exactly like content that was cut off.
|
|
37
|
+
4. **Snapshot economics:** `scout_snapshot` after landing somewhere new; re-snapshots of the same route return *diffs* with stable refs — "No element changes" costs you almost nothing. **Reading a snapshot:** each line is `ref role "name" [flags]`. State flags come first — `pressed`, `selected`, `checked`, `expanded`, `current` — so the active filter, the chosen tab or the open section is in the text, and the diff says when one moves (`now [pressed]`, `no longer [pressed]`); `exercised` means this project already acted on that control, not that it is in any state. What the page announces is listed by its text whether or not it carries a test id: `alert "…"` and `status "…"` lines (role=alert/status, aria-live regions, `<output>`) — read them after every submit before concluding an action gave no feedback; a region that says something new shows in the diff as `~ ref alert "new text" (was …)`. The `main:` line summarises the main area besides its controls (`main: h1 "Orders" · 2 paragraphs · 180 chars of static text`, or `main: EMPTY`): a page whose main area holds only text is not a blank page, and `main: EMPTY` is the line that is. Elements listed only for their test id (wrappers, headings, decorative badges) are not controls: they are never counted as unnamed or as coverage gaps. `scout_screenshot` ONLY for suspected pixel-native issues (a canvas, a rendering glitch); geometry problems (overlap, off-screen, a covered control) are already in the snapshot as GEOMETRY issues, and images that failed to load are listed under BROKEN IMAGES — file those, quoting the line. `scout_capture {ref}` saves a PNG of one element (plus a `margin` of pixels around it) only when the user asks to see how something looks; it shows, it does not judge. Controls inside embeds (`<iframe>`) are listed with the page's, each marked `⟨in … frame⟩`, and act by ref like any other. A frame of the app's own origin is the app: test it fully. A frame of another site (`cross-origin`) is someone else's system that the user has not authorised you to test: click and type there as a user would, to see the embed render and respond, but never send it hostile input — the engine refuses markup, values over 200 characters, control characters, repeated-click probes and uploads there, and you must not try SQL, template or other injection shapes either — and file what you find there as the embed's behaviour, naming its origin, not as the app's bug. A failing request such a frame sent outside the app says so (`in an embed of …`) and is kept at medium; console errors cannot be told apart by frame and stay the app's; its controls are counted apart and never enter the app's coverage or gap ledger. Its container text is masked, and the writes it sends outside the app are refused in every mode but destructive — unless the user names that origin as a trusted embed (a provider in test mode, say): pass it in `scout_attach {trustedEmbeds: ["https://…"]}` and, in safe-write only, its writes go out. Never add an origin the user did not name. The FRAMES line names frames that were not read; say in your summary that their contents were not explored.
|
|
38
|
+
5. **Native-user behaviours.** `scout_type {ref, textValue}` (or its alias `value`, matching `scout_select` and a plan step) APPENDS when a field already has content (menu clicks often insert @-mention chips or commands into composers — appending preserves them; the result reports what was already there); pass `replace=true` only to deliberately clear, and `pressEnter=true` to submit from the field the way a user would. Before concluding a badge, icon, or "N errors" indicator *does nothing*, `scout_hover` it — tooltips and hover cards are invisible to snapshots and clicks, and hover output includes what appeared. In HEADED mode (`scout_attach {headed:true}`, which the user asks for when they want to watch) the user's physical mouse competes with the synthetic pointer: if a hover reveals nothing and the finding matters, ask the user to move their mouse off the browser window and retry before filing. **Scroll long pages with `scout_scroll`** — the design audit and snapshot measure at the current scroll position, so judge deep sections by scrolling then re-auditing; it refuses to scroll where a real user couldn't and reports SCROLL LOCKED (the leaked modal scroll-lock that silently amputates everything below the fold — snapshots also flag it passively as an OVERLAY line), and scrolling triggers lazy-loaded content whose failures surface as fresh oracle violations. Elements fully clipped inside an overflow-hidden container are flagged UNREACHABLE in GEOMETRY issues — no amount of scrolling reveals them; that's a high-value layout bug, distinct from merely below-the-fold content. Controls held outside the visible width of a horizontally scrolling container (a wide table's action column) are one GEOMETRY line per container, `N controls are scrolled out of view inside a horizontally scrolling container …`: reachable by scrolling that container sideways, and worth a look rather than a defect unless the project keeps row actions in view. **A page can hold SEVERAL independent scroll regions** and plain `scout_scroll` moves the largest one, so a sidebar nav beside a taller main pane never budges: pass `scout_scroll {target:"testid=…"}` to scroll one region. Never report a nav item, tab or list row as missing/truncated until you have scrolled ITS container — content scrolled out of a secondary pane looks exactly like content that was cut off.
|
|
39
39
|
6. **The rest of the input vocabulary.** `scout_select` sets a `<select>` option by value or visible label — use it rather than clicking a native dropdown open, which does not render as page DOM. `scout_press` sends a real key to the focused element (`Escape` to dismiss a modal, `Tab` to walk focus order, `Enter` to submit from a field); it is also how the keyboard-only pass at `extensive` is performed, and it vets the focused control first so a destructive action cannot be triggered blind in read-only mode. **`scout_upload {ref}` attaches a file the way a user does** — `ref` is a visible `<input type=file>` (snapshots list these with role `file`; `scout_type` on one redirects here) OR the button/label/dropzone that opens the file chooser (the chooser is intercepted and answered — that is how the hidden input behind a styled "Choose file" control is reached); omit `ref` when the page has exactly one file input, hidden or not (snapshots disclose hidden ones on a FILE INPUTS line). Nothing needs to exist on disk: a small VALID fixture is generated in memory, its kind inferred from the input's `accept` attribute or chosen with `fixture` (`pdf`, `png`, `txt`, `csv`, `json`); `filePath` uploads a real file but must live inside the attached project (fenced, like navigation is fenced to the origin); `name` overrides the filename. The result flags a file that violates `accept` (a mismatch the app then ACCEPTS is a validation finding), warns when the app cleared the input after selection, and says whether a state-changing request fired on selection — if none did, either click the form's submit or read the next snapshot for a client-side rejection. Plans take `{action:"upload", target, value:"pdf"}` steps (`target` required). When the input or its trigger was addressed by `ref`, the gap ledger counts an attached-but-unsent file as filled-never-submitted; the ref-less path has no listed element to mark.
|
|
40
40
|
7. **Say what you are doing: `task` is required before a tool acts.** A session shows two lines to whoever is watching. Its **objective** is the whole remit you were given, set once at `scout_attach {objective}` ("Admin lane: §2 registers, §7 plan gating", "Approve and reject orders as a manager"). Its **task** is what you are doing *right now*, and every tool that changes the app or the page — `scout_navigate`, `scout_back`, `scout_click`, `scout_type`, `scout_select`, `scout_press`, `scout_upload`, `scout_run_plan` — takes it: a few words for the batch in front of you ("Filtering the documents register by status", "Filling the deviation form with invalid dates", "Signing in as QA_Team"). Say what you are DOING, not what you are checking — "§2.4 filtering narrows the set and is reflected in the URL" is the acceptance criteria, which is the result you will judge, not the batch you are running; naming the item is fine ("§2.4: filtering the documents register"). The task STAYS SET until you pass a different one, so a batch costs a few words, not one per call — pass a fresh one whenever you move on. Acting with none standing is refused: the person watching would otherwise see a session clicking through their app with nothing to say why. `scout_journey {action:"start", goal:…}` sets the task too while it runs — use a journey when you are MEASURING a whole user task, the parameter for everything else.
|
|
41
41
|
|
|
@@ -106,7 +106,7 @@ The engine is self-healing (orphaned browsers reaped, wedged calls time out with
|
|
|
106
106
|
- **Always pass `evidence`** to `scout_finding` — a canonical machine signature like `GET /api/reports/dashboard 403` or `widget dashboard-summary-widget shows 0`. It's what deduplicates the same bug across runs when titles get rephrased.
|
|
107
107
|
- File judgment findings too: confusing flows, no-feedback actions, state lost on refresh, permission leaks (low-privilege role reaching admin surface), `missing-testid` (low, where the project's test-id convention is known; worth a look otherwise), unnamed interactables (a11y, low).
|
|
108
108
|
- Respect refusals — never retry or route around a policy refusal; note it and move on.
|
|
109
|
-
- Duplicates are fine; `scout_finding` dedups across runs.
|
|
109
|
+
- Duplicates are fine; `scout_finding` dedups across runs. When it says a model judge merged your filing into another finding, your title, category, severity and evidence are kept on that finding; refile only if it is a different defect, with a title that says what differs. Pass `scout_attach {dedup: "judge"}` only when the user asks for model-judged dedup: it sends finding titles, categories and evidence, and the page's path, to a model provider, and needs a key in the server's environment.
|
|
110
110
|
|
|
111
111
|
## Finishing
|
|
112
112
|
|