scenescout 3.15.0 → 3.17.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +87 -0
- package/README.md +70 -18
- package/dist/browsers.js +28 -0
- package/dist/check-run.js +191 -14
- package/dist/ci-run.js +268 -52
- package/dist/cli.js +107 -47
- package/dist/commands.js +3 -2
- package/dist/engine/baseline.js +377 -0
- package/dist/engine/brief.js +16 -7
- package/dist/engine/browser.js +1147 -286
- package/dist/engine/calibration.js +61 -30
- package/dist/engine/capture.js +164 -0
- package/dist/engine/check.js +244 -42
- package/dist/engine/ci-lanes.js +215 -0
- package/dist/engine/ci.js +136 -18
- package/dist/engine/claims.js +159 -3
- package/dist/engine/collector.js +561 -30
- package/dist/engine/crawl.js +49 -0
- package/dist/engine/design.js +281 -38
- package/dist/engine/export.js +877 -0
- package/dist/engine/fingerprint.js +92 -4
- package/dist/engine/flow.js +18 -6
- package/dist/engine/forms.js +181 -18
- package/dist/engine/journey.js +29 -1
- package/dist/engine/lane.js +13 -3
- package/dist/engine/launch.js +45 -6
- package/dist/engine/limits.js +7 -0
- package/dist/engine/live-page.js +49 -2
- package/dist/engine/live.js +4 -1
- package/dist/engine/memory.js +501 -47
- package/dist/engine/open.js +118 -0
- package/dist/engine/oracles.js +41 -1
- package/dist/engine/plain.js +268 -0
- package/dist/engine/png.js +127 -0
- package/dist/engine/policy.js +379 -9
- package/dist/engine/probes.js +3 -2
- package/dist/engine/profiles.js +45 -9
- package/dist/engine/project-folder.js +191 -0
- package/dist/engine/refresh.js +68 -3
- package/dist/engine/replay.js +63 -10
- package/dist/engine/report.js +241 -40
- package/dist/engine/request.js +317 -23
- package/dist/engine/sarif.js +120 -0
- package/dist/engine/settle.js +67 -0
- package/dist/engine/signed-in.js +256 -0
- package/dist/engine/status-pane-page.js +441 -0
- package/dist/engine/status-pane.js +128 -0
- package/dist/engine/tickets.js +671 -0
- package/dist/engine/unload.js +3 -2
- package/dist/export-run.js +633 -0
- package/dist/first-run.js +5 -0
- package/dist/installer.js +378 -8
- package/dist/intake.js +104 -0
- package/dist/login-run.js +250 -36
- package/dist/mcp-server.js +660 -65
- package/dist/playbook.js +5 -0
- package/dist/prompts.js +106 -0
- package/package.json +8 -5
- package/skills/scenescout/SKILL.md +49 -16
package/dist/mcp-server.js
CHANGED
|
@@ -29,7 +29,10 @@
|
|
|
29
29
|
* a loopback-only live view shows what each one is looking at
|
|
30
30
|
* (`scenescout watch <project>`, engine/live.ts, ADR 7).
|
|
31
31
|
*/
|
|
32
|
+
import { AsyncLocalStorage } from "node:async_hooks";
|
|
33
|
+
import { spawn } from "node:child_process";
|
|
32
34
|
import fs from "node:fs";
|
|
35
|
+
import os from "node:os";
|
|
33
36
|
import path from "node:path";
|
|
34
37
|
import { fileURLToPath } from "node:url";
|
|
35
38
|
import { McpServer } from "@modelcontextprotocol/sdk/server/mcp.js";
|
|
@@ -37,8 +40,11 @@ import { StdioServerTransport } from "@modelcontextprotocol/sdk/server/stdio.js"
|
|
|
37
40
|
import { ErrorCode, GetPromptRequestSchema, ListPromptsRequestSchema, McpError } from "@modelcontextprotocol/sdk/types.js";
|
|
38
41
|
import { z } from "zod";
|
|
39
42
|
import { BrowserEngine } from "./engine/browser.js";
|
|
43
|
+
import { readyBrowser } from "./engine/launch.js";
|
|
44
|
+
import { defaultEngine } from "./browsers.js";
|
|
45
|
+
import { downloadBrowsers, presentBrowsers } from "./installer.js";
|
|
40
46
|
import { reapOrphanBrowsers } from "./engine/reaper.js";
|
|
41
|
-
import { FINDING_CATEGORIES, isWorthALook, MemoryStore, mergeableCategories, redactSecrets } from "./engine/memory.js";
|
|
47
|
+
import { describeMerge, FINDING_CATEGORIES, isWorthALook, MemoryStore, mergeableCategories, redactSecrets } from "./engine/memory.js";
|
|
42
48
|
import { DEDUP_ENV, DEDUP_MODES, redactKeys, secretValues } from "./engine/ci.js";
|
|
43
49
|
import { DedupJudge, planDedup, samplingAsk } from "./engine/dedup.js";
|
|
44
50
|
import { httpJudgeAsk } from "./ci-run.js";
|
|
@@ -47,19 +53,29 @@ import { MAX_UNFILED_NAMED, unfiledDefects } from "./engine/calibration.js";
|
|
|
47
53
|
import { SessionQueue, withWatchdog } from "./engine/dispatch.js";
|
|
48
54
|
import { FIXTURE_KINDS } from "./engine/fixtures.js";
|
|
49
55
|
import { feedForSession, LIVE_ENV, writeStatusFile, LIVE_TOKEN_FILE, liveEngines, liveTokenFileName, pidAlive, statusFileName, LiveServer, StatusBoard, } from "./engine/live.js";
|
|
56
|
+
import { liveViewUrl, MCP_APP_MIME, paneData, paneText, STATUS_PANE_URI, STATUS_POLL_TOOL, STATUS_TOOL, } from "./engine/status-pane.js";
|
|
57
|
+
import { statusPanePage } from "./engine/status-pane-page.js";
|
|
50
58
|
import { formatBriefs, MAX_LANES, planLanes } from "./engine/brief.js";
|
|
59
|
+
import { decideOpen, OPEN_CHOICES, OPEN_ENV, openChoiceFromEnv, openInBrowser } from "./engine/open.js";
|
|
51
60
|
import { DEFAULT_EXPIRY_MARGIN_MINUTES, DEFAULT_RUN_MINUTES, judgeProfileFile } from "./engine/expiry.js";
|
|
52
|
-
import {
|
|
53
|
-
import {
|
|
54
|
-
import {
|
|
61
|
+
import { parseLoginArgs } from "./engine/profiles.js";
|
|
62
|
+
import { LOGIN_WINDOW_MAX_MS, savedLine, startLoginWindow } from "./login-run.js";
|
|
63
|
+
import { LoginWindows, WAIT_SAYS } from "./engine/signed-in.js";
|
|
64
|
+
import { computeGaps, coverageView, formatRouteCoverage, generateReport, replayDocument, reportEvidence } from "./engine/report.js";
|
|
65
|
+
import { DEFAULT_REPORT_AUDIENCE, REPORT_AUDIENCES } from "./engine/plain.js";
|
|
55
66
|
import { describeVerdict, formatWorklist, unknownIds, VERDICTS, verifyWorklist } from "./engine/verify.js";
|
|
67
|
+
import { CRITERION_VERDICTS, findCriterion, formatCriteriaForLanes, formatReading, isTicketFileName, judgeCriterion, MAX_CRITERION_FINDINGS, MAX_REASON, MAX_TICKET_FILE_BYTES, MAX_TICKET_TEXT, MAX_TICKETS, NOT_TESTED_REASONS, parseTickets, TICKET_FILE_EXTENSIONS, } from "./engine/tickets.js";
|
|
56
68
|
import { ACTION_TIMEOUT_ENV, DEFAULT_ACTION_TIMEOUT_MS, DEFAULT_CRAWL_NAV_TIMEOUT_MS, DEFAULT_NAV_TIMEOUT_MS, LIMIT_BOUNDS, NAV_TIMEOUT_ENV, watchdogFor, } from "./engine/limits.js";
|
|
69
|
+
import { MAX_READ_POSTS, READ_POSTS_ENV } from "./engine/policy.js";
|
|
70
|
+
import { chooseProjectFolder, PROJECTS_DIR_ENV, workspaceFromRoots } from "./engine/project-folder.js";
|
|
57
71
|
import { RECORD_MAX_FRAMES, resolveFrame } from "./engine/replay.js";
|
|
58
72
|
import { describePace, normalizePace } from "./engine/settle.js";
|
|
59
73
|
import { needsTask, taskRefusal, TASK_MAX } from "./engine/task.js";
|
|
60
74
|
import { EXPLORE_PROMPT_ARGUMENTS, explorePrompt, loadPlaybook, PLAYBOOK_PROMPT, PLAYBOOK_TOOL, SERVER_INSTRUCTIONS } from "./playbook.js";
|
|
75
|
+
import { livePrompt, loginPrompt, LOGIN_PROMPT, LIVE_PROMPT, LOGIN_PROMPT_ARGUMENTS, LIVE_PROMPT_ARGUMENTS } from "./prompts.js";
|
|
61
76
|
import { formatScan, scanProject } from "./scan.js";
|
|
62
|
-
import { CAPTURE_MARGIN, CAPTURES_DIRNAME, captureFileName, captureResultText, MAX_CAPTURE_MARGIN } from "./engine/capture.js";
|
|
77
|
+
import { CAPTURE_MARGIN, CAPTURES_DIRNAME, captureFileName, captureResultText, describePicture, EVIDENCE_ENV, EVIDENCE_LIMITS, EVIDENCE_MARGIN, EVIDENCE_MODES, evidenceFrame, evidenceSettings, findingPicturePath, MAX_CAPTURE_MARGIN, RECORD_ENV, recordChoice, returnsInline, } from "./engine/capture.js";
|
|
78
|
+
import { decodePng, fitPicture } from "./engine/png.js";
|
|
63
79
|
/** Live sessions: each name owns an independent BrowserEngine (browser + auth). */
|
|
64
80
|
const engines = new Map();
|
|
65
81
|
/**
|
|
@@ -254,8 +270,10 @@ function reportExtras(eng) {
|
|
|
254
270
|
designAudits: eng.memory?.auditsThisRun ?? eng.designAuditCount,
|
|
255
271
|
createdResources: eng.createdResources,
|
|
256
272
|
unvisitedRoutes: unvisited,
|
|
273
|
+
knownRoutes: all,
|
|
257
274
|
mode: eng.mode,
|
|
258
275
|
trustedEmbeds: [...eng.trustedEmbeds],
|
|
276
|
+
readPosts: eng.readPosts.map((e) => e.entry),
|
|
259
277
|
policyAttributed: eng.oracleLog.policyAttributed,
|
|
260
278
|
version: PKG_VERSION,
|
|
261
279
|
attachedSessions: [...engines.keys()],
|
|
@@ -331,9 +349,31 @@ function startLiveServer() {
|
|
|
331
349
|
function liveLine() {
|
|
332
350
|
if (!liveAddress)
|
|
333
351
|
return liveError ? `\nLive view unavailable: ${liveError}` : "";
|
|
334
|
-
return (`\nLive view:
|
|
352
|
+
return (`\nLive view: ${liveViewUrl(liveAddress.port, liveAddress.token)} — give this address to the user so they can watch every session ` +
|
|
335
353
|
`(current tool, page thumbnail, optional live stream). It opens on this machine only and cannot act on the run.`);
|
|
336
354
|
}
|
|
355
|
+
/** What each session's attach decided to open (engine/open.ts). scout_report reads it for the report. */
|
|
356
|
+
const openDecisions = new Map();
|
|
357
|
+
/** The live view is one address for every session, so it is opened once per server. */
|
|
358
|
+
let liveOpened = false;
|
|
359
|
+
/** Whether a result has named the setting that turns opening off; named once. */
|
|
360
|
+
let openSettingNamed = false;
|
|
361
|
+
/** Whether an attach has said why the live view was not opened; said once. */
|
|
362
|
+
let openWhySaid = false;
|
|
363
|
+
/** Open a target in the default browser and say so in a line for the tool result; a failure is said, never thrown. */
|
|
364
|
+
function openForUser(what, target) {
|
|
365
|
+
const outcome = openInBrowser(target, {
|
|
366
|
+
platform: process.platform,
|
|
367
|
+
spawn: (command, args, options) => spawn(command, args, options),
|
|
368
|
+
onError: (why) => logLine(`could not open ${what}: ${why}`),
|
|
369
|
+
});
|
|
370
|
+
if (!outcome.ok)
|
|
371
|
+
return `\nCould not open ${what}: ${outcome.why}.`;
|
|
372
|
+
if (openSettingNamed)
|
|
373
|
+
return `\nOpened ${what} in the default browser.`;
|
|
374
|
+
openSettingNamed = true;
|
|
375
|
+
return `\nOpened ${what} in the default browser (${OPEN_ENV}=none, or scout_attach {open: "none"}, turns this off).`;
|
|
376
|
+
}
|
|
337
377
|
function flushStatus(dir) {
|
|
338
378
|
// Not awaited on a tool call's hot path: status is best-effort observability
|
|
339
379
|
// and must never add blocking filesystem latency there. The writer queues
|
|
@@ -451,11 +491,17 @@ function serializedPerSession(label, fn, timeoutMs = 60_000) {
|
|
|
451
491
|
return sessionQueue.run(session, exec);
|
|
452
492
|
};
|
|
453
493
|
}
|
|
494
|
+
const requestExtra = new AsyncLocalStorage();
|
|
495
|
+
function asRequestExtra(value) {
|
|
496
|
+
return value && typeof value.sendNotification === "function" ? value : undefined;
|
|
497
|
+
}
|
|
454
498
|
/** Control-plane tools (scout_scan/scout_session/scout_close-all) don't target one browser — their own tiny chain keeps them off session queues without racing each other. */
|
|
455
499
|
let controlChain = Promise.resolve();
|
|
456
500
|
function serializedControl(fn) {
|
|
457
501
|
return (...args) => {
|
|
458
|
-
|
|
502
|
+
// The SDK passes the request's extra (progress token, notifications) after the arguments; it is kept for the call, as requestExtra.
|
|
503
|
+
const call = () => requestExtra.run(asRequestExtra(args[1]), () => fn(...args));
|
|
504
|
+
const run = controlChain.then(call, call);
|
|
459
505
|
controlChain = run.catch(() => { });
|
|
460
506
|
return run;
|
|
461
507
|
};
|
|
@@ -497,10 +543,11 @@ server.registerTool(PLAYBOOK_TOOL, {
|
|
|
497
543
|
return errorText(err);
|
|
498
544
|
}
|
|
499
545
|
});
|
|
500
|
-
//
|
|
501
|
-
//
|
|
502
|
-
//
|
|
503
|
-
//
|
|
546
|
+
// Prompts, for clients that list server prompts as commands. Registered on the
|
|
547
|
+
// protocol server directly: the SDK's prompt helper rejects a request that
|
|
548
|
+
// carries no `arguments` object, which is exactly what a client sends when the
|
|
549
|
+
// person typed none. `explore` and `live` accept that; `login` then says the
|
|
550
|
+
// role is missing. None of them takes a password.
|
|
504
551
|
server.server.registerCapabilities({ prompts: {} });
|
|
505
552
|
server.server.setRequestHandler(ListPromptsRequestSchema, () => ({
|
|
506
553
|
prompts: [
|
|
@@ -510,16 +557,36 @@ server.server.setRequestHandler(ListPromptsRequestSchema, () => ({
|
|
|
510
557
|
description: "Start an exploratory test session: loads the SceneScout method and states the target.",
|
|
511
558
|
arguments: EXPLORE_PROMPT_ARGUMENTS,
|
|
512
559
|
},
|
|
560
|
+
{
|
|
561
|
+
name: LIVE_PROMPT,
|
|
562
|
+
title: "Watch the live view",
|
|
563
|
+
description: "Return the loopback live-view URL for the current session. Takes no arguments and no password.",
|
|
564
|
+
arguments: LIVE_PROMPT_ARGUMENTS,
|
|
565
|
+
},
|
|
566
|
+
{
|
|
567
|
+
name: LOGIN_PROMPT,
|
|
568
|
+
title: "Sign in as a role",
|
|
569
|
+
description: "Call scout_login for a role and wait the way that tool waits. The person signs in in the window it opens. Takes the role, and the app URL when you have it. Never a password.",
|
|
570
|
+
arguments: LOGIN_PROMPT_ARGUMENTS,
|
|
571
|
+
},
|
|
513
572
|
],
|
|
514
573
|
}));
|
|
515
574
|
server.server.setRequestHandler(GetPromptRequestSchema, (request) => {
|
|
516
|
-
|
|
517
|
-
throw new McpError(ErrorCode.InvalidParams, `Unknown prompt: ${request.params.name}`);
|
|
575
|
+
const name = request.params.name;
|
|
518
576
|
let message;
|
|
519
577
|
try {
|
|
520
|
-
|
|
578
|
+
if (name === PLAYBOOK_PROMPT)
|
|
579
|
+
message = explorePrompt(loadPlaybook(PACKAGE_ROOT), request.params.arguments);
|
|
580
|
+
else if (name === LIVE_PROMPT)
|
|
581
|
+
message = livePrompt(request.params.arguments);
|
|
582
|
+
else if (name === LOGIN_PROMPT)
|
|
583
|
+
message = loginPrompt(request.params.arguments);
|
|
584
|
+
else
|
|
585
|
+
throw new McpError(ErrorCode.InvalidParams, `Unknown prompt: ${name}`);
|
|
521
586
|
}
|
|
522
587
|
catch (err) {
|
|
588
|
+
if (err instanceof McpError)
|
|
589
|
+
throw err;
|
|
523
590
|
throw new McpError(ErrorCode.InvalidParams, err instanceof Error ? err.message : String(err));
|
|
524
591
|
}
|
|
525
592
|
return { messages: [{ role: "user", content: { type: "text", text: message } }] };
|
|
@@ -568,7 +635,7 @@ server.registerTool("scout_lane_brief", {
|
|
|
568
635
|
runMs: (runMinutes ?? DEFAULT_RUN_MINUTES) * 60_000,
|
|
569
636
|
marginMs: (expiryMarginMinutes ?? DEFAULT_EXPIRY_MARGIN_MINUTES) * 60_000,
|
|
570
637
|
role: eng.auth.role,
|
|
571
|
-
rerun:
|
|
638
|
+
rerun: eng.reloginCommand(eng.auth.role),
|
|
572
639
|
});
|
|
573
640
|
if (verdict.kind === "refuse")
|
|
574
641
|
return errorText(new Error(verdict.message));
|
|
@@ -578,7 +645,11 @@ server.registerTool("scout_lane_brief", {
|
|
|
578
645
|
const all = routes && routes.length > 0 ? routes : eng.allKnownRoutes();
|
|
579
646
|
const briefs = planLanes(all, lanes, { goal, mode: eng.mode, role: eng.role });
|
|
580
647
|
laneLedger.nameBriefed(briefs.map((b) => b.lane), (s) => engines.has(s));
|
|
581
|
-
|
|
648
|
+
// A run given tickets tells every lane which criteria it answers.
|
|
649
|
+
const criteria = eng.memory ? formatCriteriaForLanes(eng.memory.ticketsThisRun().tickets) : "";
|
|
650
|
+
return text(expiryNote +
|
|
651
|
+
formatBriefs(briefs, { goal, mode: eng.mode, role: eng.role, roleProfile: eng.auth.kind === "role" }) +
|
|
652
|
+
(criteria ? `\n${criteria}` : ""), session);
|
|
582
653
|
}
|
|
583
654
|
catch (err) {
|
|
584
655
|
return errorText(err);
|
|
@@ -645,7 +716,7 @@ server.registerTool("scout_lane_report", {
|
|
|
645
716
|
.map((u) => ` · ${u}`)
|
|
646
717
|
.join("\n") +
|
|
647
718
|
(unfiled.length > MAX_UNFILED_NAMED ? `\n … +${unfiled.length - MAX_UNFILED_NAMED} more` : "") +
|
|
648
|
-
`\nFile each with scout_finding (the same evidence), or
|
|
719
|
+
`\nFile each with scout_finding (the same evidence), or have the lane name the finding's id in the decision's "finding", before closing the lane's session. A judged defect that is never filed is not in the report.`
|
|
649
720
|
: "";
|
|
650
721
|
laneLedger.fold(lane, engines.get(lane)?.attached === true);
|
|
651
722
|
// A lane that lost its sign-in and re-attached from its role's profile
|
|
@@ -681,10 +752,14 @@ server.registerTool("scout_scan", {
|
|
|
681
752
|
}
|
|
682
753
|
}));
|
|
683
754
|
server.registerTool("scout_attach", {
|
|
684
|
-
description: "Launch a browser and attach to a running web app. First attach in this conversation and you have read neither the SceneScout skill nor scout_playbook? Call scout_playbook before this. Write policy is enforced at the NETWORK layer: mode='observe' blocks EVERY request that is not a GET (login and token refresh excepted) — choose it for a target that holds real data, where even an ordinary form submission would create a record; mode='read-only' (default) blocks destructive-labeled elements AND all PUT/PATCH/DELETE + destructive POSTs, but lets ordinary form POSTs through; mode='safe-write' allows creating data and permits updates/deletes ONLY on resources this session created (use when the user wants create/edit flows tested); mode='destructive' allows everything — ONLY when the user explicitly confirmed a disposable/seeded environment. Pass `role` to sign in with a login the user saved by `scenescout login <url> --role <name>`, or a Playwright storage-state JSON as storageStatePath. Pass `session` to keep MULTIPLE roles alive at once (one browser each, genuinely concurrent) for collaboration testing — target each directly with every tool's `session` param, or use scout_session to set which one is the default; coverage and findings merge into one project memory.",
|
|
755
|
+
description: "Launch a browser and attach to a running web app. First attach in this conversation and you have read neither the SceneScout skill nor scout_playbook? Call scout_playbook before this. Write policy is enforced at the NETWORK layer: mode='observe' blocks EVERY request that is not a GET (login and token refresh excepted, and POSTs the user named in readPosts) — choose it for a target that holds real data, where even an ordinary form submission would create a record; mode='read-only' (default) blocks destructive-labeled elements AND all PUT/PATCH/DELETE + destructive POSTs, but lets ordinary form POSTs through; mode='safe-write' allows creating data and permits updates/deletes ONLY on resources this session created (use when the user wants create/edit flows tested); mode='destructive' allows everything — ONLY when the user explicitly confirmed a disposable/seeded environment. Pass `role` to sign in with a login the user saved by `scenescout login <url> --role <name>`, or a Playwright storage-state JSON as storageStatePath. Pass `session` to keep MULTIPLE roles alive at once (one browser each, genuinely concurrent) for collaboration testing — target each directly with every tool's `session` param, or use scout_session to set which one is the default; coverage and findings merge into one project memory.",
|
|
685
756
|
inputSchema: {
|
|
686
757
|
url: z.string().describe("Base URL of the running app, e.g. http://localhost:3000"),
|
|
687
|
-
projectPath: z
|
|
758
|
+
projectPath: z
|
|
759
|
+
.string()
|
|
760
|
+
.optional()
|
|
761
|
+
.describe("Absolute path to the project (memory + report live in .scenescout/ here). Pass it whenever you have a project or working folder. " +
|
|
762
|
+
`Omitted: the client's workspace folder, else a folder per tested site under the user's documents folder (Documents/SceneScout/<host>/, or ${PROJECTS_DIR_ENV}), which the result names — tell the user where it is.`),
|
|
688
763
|
storageStatePath: z.string().optional().describe("Optional Playwright storage-state JSON path for authenticated exploration. Not with `role`."),
|
|
689
764
|
role: z
|
|
690
765
|
.string()
|
|
@@ -701,7 +776,7 @@ server.registerTool("scout_attach", {
|
|
|
701
776
|
browser: z
|
|
702
777
|
.enum(["chromium", "firefox", "webkit"])
|
|
703
778
|
.optional()
|
|
704
|
-
.describe("Browser to drive. Default: the SCENESCOUT_BROWSER environment variable, else chromium.
|
|
779
|
+
.describe("Browser to drive. Default: the SCENESCOUT_BROWSER environment variable, else chromium. A build that is not on disk is downloaded on this attach, once, except in CI or with SCENESCOUT_BROWSER_DOWNLOAD=off, where the attach names the command to run. Use firefox or webkit for a cross-browser pass; stay on chromium otherwise."),
|
|
705
780
|
viewportWidth: z.number().int().min(320).max(3840).optional().describe("Viewport width (default 1280); use e.g. 390 for a mobile pass"),
|
|
706
781
|
viewportHeight: z.number().int().min(480).max(2400).optional().describe("Viewport height (default 900)"),
|
|
707
782
|
objective: z
|
|
@@ -729,11 +804,22 @@ server.registerTool("scout_attach", {
|
|
|
729
804
|
.optional()
|
|
730
805
|
.describe('Origins of embedded frames (e.g. "https://pay.example.com") whose writes out of the app may go out — ONLY when the user named them, typically a provider in test mode, and only in safe-write mode. ' +
|
|
731
806
|
"Never add one yourself. Hostile input, repeated-click probes and uploads stay refused in them."),
|
|
807
|
+
readPosts: z
|
|
808
|
+
.array(z.string().max(300))
|
|
809
|
+
.max(MAX_READ_POSTS)
|
|
810
|
+
.optional()
|
|
811
|
+
.describe(`POST endpoints that only read, e.g. ["POST /api/search", "POST https://api.example.com/reports/query"], which observe mode then lets out — ONLY when the user named them. Never add one yourself, even when the gap ledger lists a refused POST: ask the user. ` +
|
|
812
|
+
`Exact paths; * stands for one path segment. Still refused when the path or body looks destructive or the body is a GraphQL mutation. Observe mode only. Default: the ${READ_POSTS_ENV} environment variable, else none.`),
|
|
732
813
|
record: z
|
|
733
814
|
.boolean()
|
|
734
|
-
.
|
|
815
|
+
.optional()
|
|
735
816
|
.describe("Keep a frame of the page after every action, under .scenescout/recordings/, and show it beside that step in report.html. " +
|
|
736
|
-
|
|
817
|
+
`Default: the ${RECORD_ENV} environment variable (on or off), else off: a recording is pictures of the app under test sitting in the project folder. Turn it on for QA work, where the run is evidence and not only a report.`),
|
|
818
|
+
evidence: z
|
|
819
|
+
.enum(EVIDENCE_MODES)
|
|
820
|
+
.optional()
|
|
821
|
+
.describe("What happens to the picture each scout_finding takes of what it is about. 'inline': kept under .scenescout/recordings/, shown in report.html, and returned in the scout_finding result so the conversation shows it. 'file': kept and shown in the report only. 'off': none taken. " +
|
|
822
|
+
`Default: the ${EVIDENCE_ENV} environment variable, else 'inline', or 'file' in a CI job. Pictures are bounded in size and in how many one session returns (see the configuration reference).`),
|
|
737
823
|
actionTimeoutMs: z
|
|
738
824
|
.number()
|
|
739
825
|
.int()
|
|
@@ -760,9 +846,20 @@ server.registerTool("scout_attach", {
|
|
|
760
846
|
"'judge': the rule first, then, for a filing the rule keeps apart from everything recorded, a model is asked whether it is one of the open findings on its page, and merges it when it says so. " +
|
|
761
847
|
"It needs ANTHROPIC_API_KEY or OPENAI_API_KEY in the server's environment, and sends each pair's titles, categories and evidence, and the page's path, to that provider — ONLY when the user asked for it. " +
|
|
762
848
|
"Applies to every session of the project until the run ends."),
|
|
849
|
+
open: z
|
|
850
|
+
.enum(OPEN_CHOICES)
|
|
851
|
+
.optional()
|
|
852
|
+
.describe(`Open the live view (on this attach) and/or report.html (when scout_report writes it) in the user's default browser: 'live', 'report', 'both' or 'none'. ` +
|
|
853
|
+
`Default: the ${OPEN_ENV} environment variable, else 'both' on a local desktop session (headed or headless) and 'none' in CI, over SSH, or with no display. ` +
|
|
854
|
+
"Pass it only when the user asked for something other than the default."),
|
|
763
855
|
},
|
|
764
|
-
}, serializedControl(async ({ url, projectPath, storageStatePath, role, mode, headed, browser, viewportWidth, viewportHeight, objective, task, record, trustedEmbeds, paceMs, actionTimeoutMs, navTimeoutMs, session, dedup, }) => {
|
|
856
|
+
}, serializedControl(async ({ url, projectPath, storageStatePath, role, mode, headed, browser, viewportWidth, viewportHeight, objective, task, record, evidence, trustedEmbeds, readPosts, paceMs, actionTimeoutMs, navTimeoutMs, session, dedup, open, }) => {
|
|
765
857
|
try {
|
|
858
|
+
// Settled first: a refusal leaves the default session as it was.
|
|
859
|
+
const folder = await projectFolderFor(url, projectPath);
|
|
860
|
+
if ("refused" in folder)
|
|
861
|
+
return errorText(new Error(folder.refused));
|
|
862
|
+
projectPath = folder.dir;
|
|
766
863
|
const target = session ?? activeName;
|
|
767
864
|
if (session) {
|
|
768
865
|
activeName = session;
|
|
@@ -824,15 +921,25 @@ server.registerTool("scout_attach", {
|
|
|
824
921
|
// Moving this session to another project leaves its old one; if it was
|
|
825
922
|
// the last session there, that run is over. Re-attaching to the SAME
|
|
826
923
|
// project is the same run, and keeps what the run has learned.
|
|
924
|
+
// Before anything changes: a SCENESCOUT_OPEN the server cannot use refuses the attach.
|
|
925
|
+
const opening = decideOpen(open ?? openChoiceFromEnv(process.env), {
|
|
926
|
+
env: process.env,
|
|
927
|
+
platform: process.platform,
|
|
928
|
+
});
|
|
827
929
|
const previous = eng.memory;
|
|
828
930
|
if (previous && previous !== store && ![...engines.values()].some((e) => e !== eng && e.memory === previous))
|
|
829
931
|
previous.endRun();
|
|
830
932
|
// Before the browser starts: a value it cannot use refuses the attach rather than being replaced.
|
|
831
933
|
const dedupNote = configureDedup(store, dedup);
|
|
934
|
+
const recording = recordChoice(record, process.env);
|
|
935
|
+
const pictures = evidenceSettings(evidence, process.env);
|
|
832
936
|
const viewport = viewportWidth && viewportHeight ? { width: viewportWidth, height: viewportHeight } : undefined;
|
|
937
|
+
// The first attach on a machine without the browser downloads it here, once, rather than failing with a command to run.
|
|
938
|
+
const browserNote = await readyBrowserFor({ engine: browser ?? defaultEngine(process.env), headed: headed ?? false }, requestExtra.getStore());
|
|
833
939
|
const out = await eng.attach({
|
|
834
940
|
url,
|
|
835
941
|
projectDir: projectPath,
|
|
942
|
+
projectChosen: folder.source === "default",
|
|
836
943
|
storageStatePath,
|
|
837
944
|
role,
|
|
838
945
|
mode,
|
|
@@ -846,8 +953,9 @@ server.registerTool("scout_attach", {
|
|
|
846
953
|
objective: objective ?? task,
|
|
847
954
|
paceMs,
|
|
848
955
|
task: objective ? task : undefined,
|
|
849
|
-
record,
|
|
956
|
+
record: recording,
|
|
850
957
|
trustedEmbeds,
|
|
958
|
+
readPosts,
|
|
851
959
|
actionTimeoutMs,
|
|
852
960
|
navTimeoutMs,
|
|
853
961
|
memoryStore: store,
|
|
@@ -859,18 +967,109 @@ server.registerTool("scout_attach", {
|
|
|
859
967
|
await ensureLive(eng.memory.dir);
|
|
860
968
|
writeStatus(target, "idle", "scout_attach");
|
|
861
969
|
}
|
|
970
|
+
// A new attach is a new count of pictures returned.
|
|
971
|
+
evidenceFor.set(eng, { ...pictures, shown: 0 });
|
|
972
|
+
openDecisions.set(target, opening);
|
|
973
|
+
// The address holds the token, and goes only to this machine's opener: the live view's loopback and token rules are unchanged.
|
|
974
|
+
let openNote = "";
|
|
975
|
+
if (opening.live && liveAddress && !liveOpened) {
|
|
976
|
+
liveOpened = true;
|
|
977
|
+
openNote = openForUser("the live view", liveViewUrl(liveAddress.port, liveAddress.token));
|
|
978
|
+
}
|
|
979
|
+
else if (!opening.live && liveAddress && !openWhySaid) {
|
|
980
|
+
// Once per server, so a person wondering why no window appeared is told how to change it.
|
|
981
|
+
openWhySaid = true;
|
|
982
|
+
openNote = `\nNot opened in a browser (${opening.why}); scout_attach {open} or ${OPEN_ENV} changes that.`;
|
|
983
|
+
}
|
|
862
984
|
// Recording writes pictures of the app under test into the project, so
|
|
863
985
|
// a run doing it says where they go rather than leaving the person to
|
|
864
986
|
// find a folder of screenshots later.
|
|
865
|
-
const recordNote =
|
|
987
|
+
const recordNote = recording && eng.memory?.dir
|
|
866
988
|
? `\n\n📸 RECORDING: a frame of the page after each action, under ${path.join(eng.memory.dir, "recordings", target)}/ (at most ${RECORD_MAX_FRAMES}). scout_report writes them into report.html beside report.md.`
|
|
867
989
|
: "";
|
|
868
|
-
|
|
990
|
+
const picturesNote = pictures.mode !== "off" && eng.memory?.dir
|
|
991
|
+
? `\n📷 FINDING PICTURES: ${pictures.mode} (${pictures.source}): each scout_finding keeps a picture of what it names under ${path.join(eng.memory.dir, "recordings")}/ for report.html${pictures.mode === "inline" ? `, and returns the first ${pictures.inlineMax} in its result` : ""}.`
|
|
992
|
+
: "";
|
|
993
|
+
return text(browserNote +
|
|
994
|
+
out +
|
|
995
|
+
(folder.note ? `\n\n${folder.note}` : "") +
|
|
996
|
+
conflictNote +
|
|
997
|
+
recordNote +
|
|
998
|
+
picturesNote +
|
|
999
|
+
dedupNote +
|
|
1000
|
+
describePace(eng.pace) +
|
|
1001
|
+
(engines.size > 1 ? `\n${sessionLines()}` : "") +
|
|
1002
|
+
liveLine() +
|
|
1003
|
+
openNote, target);
|
|
869
1004
|
}
|
|
870
1005
|
catch (err) {
|
|
871
1006
|
return errorText(err);
|
|
872
1007
|
}
|
|
873
1008
|
}));
|
|
1009
|
+
/**
|
|
1010
|
+
* readyBrowser with the real download, returning the line that opens the
|
|
1011
|
+
* attach's answer when it downloaded (empty when it did not). Each line goes
|
|
1012
|
+
* to stderr and, when the client asked for progress, as a progress
|
|
1013
|
+
* notification; the last one is repeated while the download runs, so a client
|
|
1014
|
+
* that extends its timeout on progress keeps waiting.
|
|
1015
|
+
*/
|
|
1016
|
+
async function readyBrowserFor(need, extra) {
|
|
1017
|
+
const token = extra?._meta?.progressToken;
|
|
1018
|
+
let progress = 0;
|
|
1019
|
+
let last = "";
|
|
1020
|
+
const notify = (message) => {
|
|
1021
|
+
if (token === undefined || !extra)
|
|
1022
|
+
return;
|
|
1023
|
+
extra.sendNotification({ method: "notifications/progress", params: { progressToken: token, progress: ++progress, message } }).catch((err) => {
|
|
1024
|
+
logLine(`progress notification failed: ${err instanceof Error ? err.message : String(err)}`);
|
|
1025
|
+
});
|
|
1026
|
+
};
|
|
1027
|
+
const say = (line) => {
|
|
1028
|
+
last = line;
|
|
1029
|
+
logLine(line);
|
|
1030
|
+
notify(line);
|
|
1031
|
+
};
|
|
1032
|
+
const note = await readyBrowser(need, {
|
|
1033
|
+
env: process.env,
|
|
1034
|
+
present: presentBrowsers,
|
|
1035
|
+
say,
|
|
1036
|
+
download: async (targets) => {
|
|
1037
|
+
const heartbeat = setInterval(() => notify(last), 10_000);
|
|
1038
|
+
try {
|
|
1039
|
+
return await downloadBrowsers(targets, "stderr");
|
|
1040
|
+
}
|
|
1041
|
+
finally {
|
|
1042
|
+
clearInterval(heartbeat);
|
|
1043
|
+
}
|
|
1044
|
+
},
|
|
1045
|
+
});
|
|
1046
|
+
return note ? `${note}\n\n` : "";
|
|
1047
|
+
}
|
|
1048
|
+
/**
|
|
1049
|
+
* The folder an attach keeps its files in (engine/project-folder.ts): the
|
|
1050
|
+
* projectPath given, else the client's workspace folder, else a folder per
|
|
1051
|
+
* tested site. Only a client that offers roots is asked for them.
|
|
1052
|
+
*/
|
|
1053
|
+
async function projectFolderFor(url, given) {
|
|
1054
|
+
let workspace = null;
|
|
1055
|
+
if (given === undefined && server.server.getClientCapabilities()?.roots) {
|
|
1056
|
+
try {
|
|
1057
|
+
workspace = workspaceFromRoots((await server.server.listRoots(undefined, { timeout: 3000 })).roots);
|
|
1058
|
+
}
|
|
1059
|
+
catch (err) {
|
|
1060
|
+
logLine(`the client offers a workspace but did not list it (${err.message}); using the default folder`);
|
|
1061
|
+
}
|
|
1062
|
+
}
|
|
1063
|
+
const userDirsFile = path.join(os.homedir(), ".config", "user-dirs.dirs");
|
|
1064
|
+
const userDirs = process.platform === "linux" && fs.existsSync(userDirsFile) ? fs.readFileSync(userDirsFile, "utf8") : undefined;
|
|
1065
|
+
return chooseProjectFolder({
|
|
1066
|
+
given,
|
|
1067
|
+
workspace,
|
|
1068
|
+
url,
|
|
1069
|
+
home: { platform: process.platform, homedir: os.homedir(), env: process.env, userDirs },
|
|
1070
|
+
exists: fs.existsSync,
|
|
1071
|
+
});
|
|
1072
|
+
}
|
|
874
1073
|
/** Keys in this process's environment, and anything shaped like one, taken out of a line before it is shown. */
|
|
875
1074
|
const withoutKeys = (text) => redactKeys(text, secretValues(process.env));
|
|
876
1075
|
/** A line for the operator, on stderr. */
|
|
@@ -904,6 +1103,61 @@ function configureDedup(store, asked) {
|
|
|
904
1103
|
store.dedupOff = undefined;
|
|
905
1104
|
return plan.note;
|
|
906
1105
|
}
|
|
1106
|
+
/** Each session's finding-picture settings from its last attach, and how many pictures its results have carried since. */
|
|
1107
|
+
const evidenceFor = new WeakMap();
|
|
1108
|
+
/**
|
|
1109
|
+
* Take, bound and keep a filed finding's picture (capture.ts decides whether
|
|
1110
|
+
* and of what; png.ts fits it). Returns the line the result adds and, when the
|
|
1111
|
+
* result carries it, the picture. A picture that cannot be taken or kept never
|
|
1112
|
+
* fails the filing, which is already recorded: the line says why there is none.
|
|
1113
|
+
*/
|
|
1114
|
+
async function findingPicture(eng, filed, ref) {
|
|
1115
|
+
const settings = evidenceFor.get(eng);
|
|
1116
|
+
const dir = eng.memory?.dir;
|
|
1117
|
+
if (!settings || !dir)
|
|
1118
|
+
return { line: "" };
|
|
1119
|
+
const { finding, isNew } = filed;
|
|
1120
|
+
const plan = evidenceFrame({
|
|
1121
|
+
mode: settings.mode,
|
|
1122
|
+
ref,
|
|
1123
|
+
pageOpen: eng.alive,
|
|
1124
|
+
isNew,
|
|
1125
|
+
hasPicture: !!finding.picture,
|
|
1126
|
+
regressed: !isNew && !!finding.regressedAt && finding.regressedAt === finding.foundAt,
|
|
1127
|
+
});
|
|
1128
|
+
if (!plan.take)
|
|
1129
|
+
return { line: settings.mode === "off" ? "" : `\nNo picture taken: ${plan.why}.` };
|
|
1130
|
+
try {
|
|
1131
|
+
const shot = await eng.captureEvidence(plan.frame === "element" ? plan.ref : undefined, EVIDENCE_MARGIN);
|
|
1132
|
+
const fitted = fitPicture(decodePng(shot.png), settings.maxPx, settings.maxBytes);
|
|
1133
|
+
if (!fitted)
|
|
1134
|
+
return { line: `\nNo picture kept: none small enough to be readable fits in ${Math.round(settings.maxBytes / 1024)} KB.` };
|
|
1135
|
+
const rel = findingPicturePath(eng.sessionKey, finding.id);
|
|
1136
|
+
const file = path.join(dir, rel);
|
|
1137
|
+
fs.mkdirSync(path.dirname(file), { recursive: true });
|
|
1138
|
+
fs.writeFileSync(file, fitted.png);
|
|
1139
|
+
const picture = {
|
|
1140
|
+
width: fitted.width,
|
|
1141
|
+
height: fitted.height,
|
|
1142
|
+
frame: shot.frame,
|
|
1143
|
+
...(shot.label ? { label: shot.label } : {}),
|
|
1144
|
+
at: new Date().toISOString(),
|
|
1145
|
+
};
|
|
1146
|
+
eng.memory?.setPicture(finding.id, rel, picture);
|
|
1147
|
+
const said = `\n📷 Picture (${describePicture(picture)}${fitted.shrunk ? ", shrunk to fit" : ""}): ${file}${shot.note ? ` — ${shot.note}` : ""}`;
|
|
1148
|
+
if (!returnsInline(settings.mode, settings.shown, settings.inlineMax)) {
|
|
1149
|
+
const capped = settings.mode === "inline"
|
|
1150
|
+
? ` Not shown here: this session has returned its ${settings.inlineMax} (${EVIDENCE_LIMITS.inline.env} sets how many); it is in report.html.`
|
|
1151
|
+
: "";
|
|
1152
|
+
return { line: said + capped };
|
|
1153
|
+
}
|
|
1154
|
+
settings.shown += 1;
|
|
1155
|
+
return { line: said, image: { type: "image", data: fitted.png.toString("base64"), mimeType: "image/png" } };
|
|
1156
|
+
}
|
|
1157
|
+
catch (err) {
|
|
1158
|
+
return { line: `\nNo picture kept: ${err instanceof Error ? err.message : String(err)}` };
|
|
1159
|
+
}
|
|
1160
|
+
}
|
|
907
1161
|
/** What scout_finding tells the agent about what filing did. */
|
|
908
1162
|
function filedText(filed, category) {
|
|
909
1163
|
const { finding, isNew, promoted } = filed;
|
|
@@ -918,7 +1172,7 @@ function filedText(filed, category) {
|
|
|
918
1172
|
return `Merged into finding ${finding.id}, which was worth a look, and promoted to a defect: [${finding.severity}] ${finding.title}. It now counts among the report's findings.`;
|
|
919
1173
|
if (finding.regressedAt)
|
|
920
1174
|
return `⟳ REOPENED as a REGRESSION: finding ${finding.id} was previously resolved but the evidence reproduces again (seen in ${finding.runs} runs). Worth calling out to the user.`;
|
|
921
|
-
return `Not recorded as new: merged into existing finding ${finding.id} — [${finding.severity}] ${finding.title}${finding.evidence ? ` (evidence: ${finding.evidence.slice(0, 160)})` : " (no evidence)"}, filed as ${finding.category}, seen in ${finding.runs} runs. If yours is a different bug, file it again: under the category that says what is wrong if it is another kind of defect (a finding filed as ${category} merges only with one filed as ${mergeableCategories(category).join(" or ")}), or with evidence naming the request that failed for you (method and path) — two findings are kept apart when both name requests and none is shared.`;
|
|
1175
|
+
return `Not recorded as new: merged into existing finding ${finding.id} — [${finding.severity}] ${finding.title}${finding.evidence ? ` (evidence: ${finding.evidence.slice(0, 160)})` : " (no evidence)"}, filed as ${finding.category}, seen in ${finding.runs} run${finding.runs === 1 ? "" : "s"}${filed.merge?.sameRun ? " (already filed this run, so the count did not change)" : ""}.${filed.merge ? describeMerge(filed.merge, finding.convention) : ""} If yours is a different bug, file it again: under the category that says what is wrong if it is another kind of defect (a finding filed as ${category} merges only with one filed as ${mergeableCategories(category).join(" or ")}), or with evidence naming the request that failed for you (method and path) — two findings are kept apart when both name requests and none is shared.`;
|
|
922
1176
|
}
|
|
923
1177
|
function sessionLines() {
|
|
924
1178
|
const lines = ["Live sessions:"];
|
|
@@ -927,6 +1181,114 @@ function sessionLines() {
|
|
|
927
1181
|
}
|
|
928
1182
|
return lines.join("\n");
|
|
929
1183
|
}
|
|
1184
|
+
// Signing in from the conversation: a window the person signs in in, watched
|
|
1185
|
+
// in the background, so a call need not last as long as the person takes and
|
|
1186
|
+
// a second call picks up the same window. One per project and role.
|
|
1187
|
+
const pendingLogins = new LoginWindows(LOGIN_WINDOW_MAX_MS);
|
|
1188
|
+
/** How long one scout_login call waits for the person, by default and at most, in seconds. */
|
|
1189
|
+
const LOGIN_WAIT_DEFAULT_S = 120;
|
|
1190
|
+
const LOGIN_WAIT_MAX_S = 600;
|
|
1191
|
+
/** How often a waiting call tells a client that asked for progress that it is still going, in ms. */
|
|
1192
|
+
const LOGIN_PROGRESS_EVERY_MS = 10_000;
|
|
1193
|
+
server.registerTool("scout_login", {
|
|
1194
|
+
description: "Open a visible browser window for the USER to sign in to the app as a role, and save that sign-in for scout_attach { role }. " +
|
|
1195
|
+
"Use it when an attach is refused because no sign-in is saved for the role, or the saved one has expired. " +
|
|
1196
|
+
'Tell the user first, in plain words: "A browser window is opening. Sign in there as you normally would; it closes by itself once you are in." ' +
|
|
1197
|
+
"The window saves once the user is back on the app with a new session (a round trip through a single sign-on provider is followed, not taken for the end) and closes. " +
|
|
1198
|
+
"Never type credentials into it yourself. Returns once signed in and saved, or after waitSeconds with the window still open: then call scout_login again with the same role to keep waiting. " +
|
|
1199
|
+
"Closing the window saves nothing. Needs a desktop: on a machine with no display, ask the user to run `scenescout login <url> --role <name>` where they can see the window.",
|
|
1200
|
+
inputSchema: {
|
|
1201
|
+
url: z.string().describe("Where to sign in: the app's address or its sign-in page, e.g. http://localhost:3000/login"),
|
|
1202
|
+
role: z.string().max(40).describe("The name to save the sign-in under, e.g. admin; scout_attach { role } signs in with it"),
|
|
1203
|
+
projectPath: z
|
|
1204
|
+
.string()
|
|
1205
|
+
.optional()
|
|
1206
|
+
.describe("Absolute path to the project (the sign-in is saved in .scenescout/auth/ here), as for scout_attach. Omitted: the same folder an attach with no projectPath uses for this site, which the result names."),
|
|
1207
|
+
browser: z
|
|
1208
|
+
.enum(["chromium", "firefox", "webkit"])
|
|
1209
|
+
.optional()
|
|
1210
|
+
.describe("Browser to open. Default: the SCENESCOUT_BROWSER environment variable, else chromium"),
|
|
1211
|
+
successUrl: z
|
|
1212
|
+
.string()
|
|
1213
|
+
.max(500)
|
|
1214
|
+
.optional()
|
|
1215
|
+
.describe("Only when the user says how to tell: signed in once the URL's path contains this, or the URL starts with it (an absolute URL), instead of when a new session appears"),
|
|
1216
|
+
waitSeconds: z
|
|
1217
|
+
.number()
|
|
1218
|
+
.int()
|
|
1219
|
+
.min(1)
|
|
1220
|
+
.max(LOGIN_WAIT_MAX_S)
|
|
1221
|
+
.optional()
|
|
1222
|
+
.describe(`How long this call waits for the user before returning with the window still open (default ${LOGIN_WAIT_DEFAULT_S})`),
|
|
1223
|
+
},
|
|
1224
|
+
}, async (args, extra) => {
|
|
1225
|
+
try {
|
|
1226
|
+
// The folder an attach with no projectPath would use, so the attach after this finds the sign-in.
|
|
1227
|
+
const folder = await projectFolderFor(args.url, args.projectPath);
|
|
1228
|
+
if ("refused" in folder)
|
|
1229
|
+
return errorText(new Error(folder.refused));
|
|
1230
|
+
const projectDir = path.resolve(folder.dir);
|
|
1231
|
+
const where = folder.note ? `\n\n${folder.note}` : "";
|
|
1232
|
+
const parsed = parseLoginArgs([args.url, "--role", args.role, ...(args.browser ? ["--browser", args.browser] : []), ...(args.successUrl ? ["--success-url", args.successUrl] : [])], projectDir);
|
|
1233
|
+
if (!parsed.ok)
|
|
1234
|
+
return errorText(new Error(parsed.error));
|
|
1235
|
+
const options = { ...parsed.options, projectDir };
|
|
1236
|
+
const key = `${options.projectDir}\0${options.role}`;
|
|
1237
|
+
const { window: pending, resumed } = await pendingLogins.get(key, () => startLoginWindow(options));
|
|
1238
|
+
const waitMs = (args.waitSeconds ?? LOGIN_WAIT_DEFAULT_S) * 1000;
|
|
1239
|
+
const progressToken = extra._meta?.progressToken;
|
|
1240
|
+
let ticks = 0;
|
|
1241
|
+
const ticker = progressToken !== undefined
|
|
1242
|
+
? setInterval(() => {
|
|
1243
|
+
ticks += 1;
|
|
1244
|
+
const p = pending.progress();
|
|
1245
|
+
void extra
|
|
1246
|
+
.sendNotification({
|
|
1247
|
+
method: "notifications/progress",
|
|
1248
|
+
params: {
|
|
1249
|
+
progressToken,
|
|
1250
|
+
progress: ticks,
|
|
1251
|
+
message: `Waiting for the sign-in as "${options.role}": ${p.reason === "starting" ? "the window is opening" : WAIT_SAYS[p.reason]}`,
|
|
1252
|
+
},
|
|
1253
|
+
})
|
|
1254
|
+
.catch((err) => console.error(`[scenescout] scout_login progress: ${err instanceof Error ? err.message : String(err)}`));
|
|
1255
|
+
}, LOGIN_PROGRESS_EVERY_MS)
|
|
1256
|
+
: undefined;
|
|
1257
|
+
let timer;
|
|
1258
|
+
let onAbort;
|
|
1259
|
+
const outcome = await Promise.race([
|
|
1260
|
+
pending.done,
|
|
1261
|
+
new Promise((resolve) => {
|
|
1262
|
+
timer = setTimeout(() => resolve("waiting"), waitMs);
|
|
1263
|
+
}),
|
|
1264
|
+
new Promise((resolve) => {
|
|
1265
|
+
onAbort = () => resolve("cancelled");
|
|
1266
|
+
extra.signal.addEventListener("abort", onAbort, { once: true });
|
|
1267
|
+
}),
|
|
1268
|
+
]).finally(() => {
|
|
1269
|
+
clearTimeout(timer);
|
|
1270
|
+
if (ticker)
|
|
1271
|
+
clearInterval(ticker);
|
|
1272
|
+
if (onAbort)
|
|
1273
|
+
extra.signal.removeEventListener("abort", onAbort);
|
|
1274
|
+
});
|
|
1275
|
+
const differs = resumed && pending.url !== options.url ? ` (the window was opened at ${pending.url} by an earlier call; that one is still the one being watched)` : "";
|
|
1276
|
+
if (outcome === "waiting" || outcome === "cancelled") {
|
|
1277
|
+
const p = pending.progress();
|
|
1278
|
+
return text(`Still waiting for the user to sign in as "${options.role}"${differs}: ${p.reason === "starting" ? "the window is opening" : WAIT_SAYS[p.reason]}. ` +
|
|
1279
|
+
`The window stays open for up to ${LOGIN_WINDOW_MAX_MS / 60_000} minutes from when it opened. Call scout_login again with the same role to keep waiting, once the user says they are done or to check.` +
|
|
1280
|
+
where, activeName);
|
|
1281
|
+
}
|
|
1282
|
+
// This call reports the outcome; the next call for the role opens a new window.
|
|
1283
|
+
pendingLogins.reported(key, pending);
|
|
1284
|
+
if (!outcome.ok)
|
|
1285
|
+
return errorText(new Error(`nothing was saved for role "${options.role}": ${outcome.error}`));
|
|
1286
|
+
return text(`${outcome.detected}\n${savedLine(options, outcome.saved)}${where}`, activeName);
|
|
1287
|
+
}
|
|
1288
|
+
catch (err) {
|
|
1289
|
+
return errorText(err);
|
|
1290
|
+
}
|
|
1291
|
+
});
|
|
930
1292
|
server.registerTool("scout_session", {
|
|
931
1293
|
description: "List live sessions, or set which one is the DEFAULT (used by any tool call that omits `session`). Prefer passing `session` directly on each tool call for multi-role work — that's what lets concurrent dispatch happen; scout_session is for sequential convenience (skip repeating `session` on every call) and for checking what's live. Both browsers stay live and authenticated regardless of which is default — re-snapshot a session after a break to see what changed while it was away.",
|
|
932
1294
|
inputSchema: {
|
|
@@ -976,7 +1338,7 @@ server.registerTool("scout_session", {
|
|
|
976
1338
|
}
|
|
977
1339
|
}));
|
|
978
1340
|
server.registerTool("scout_snapshot", {
|
|
979
|
-
description: "Capture the current page state: URL, state fingerprint, a one-line summary of the main area's heading and text, interactable elements with refs (e1, e2, …) and their state (pressed, selected, checked, expanded, current), what the page announces (alert and status regions, by their text), geometry issues, coverage, and oracle violations since the last action. Re-snapshotting
|
|
1341
|
+
description: "Capture the current page state: URL, state fingerprint, a one-line summary of the main area's heading and text, interactable elements with refs (e1, e2, …) and their state (pressed, selected, checked, expanded, current), what the page announces (alert and status regions, by their text), geometry issues, coverage, and oracle violations since the last action. Re-snapshotting a route returns a DIFF against its last snapshot, even after visiting elsewhere, and another tab of the same screen diffs against that screen's last tab; refs stay stable, including across a search or filter that rewrites only the query string. On a dense page, says what past the element cap was cut. Cheap — prefer this over screenshots.",
|
|
980
1342
|
inputSchema: {
|
|
981
1343
|
full: z.boolean().default(false).describe("Force a full element list instead of a diff"),
|
|
982
1344
|
session: sessionParam,
|
|
@@ -990,9 +1352,13 @@ server.registerTool("scout_snapshot", {
|
|
|
990
1352
|
}
|
|
991
1353
|
}));
|
|
992
1354
|
server.registerTool("scout_crawl", {
|
|
993
|
-
description: "Engine-side route sweep in ONE call: visits each path (default: all known routes not yet visited), records states into coverage memory, and returns a per-route health summary (HTTP status, element count, oracle violations, dead-ends, auth-redirects). Navigation-only — safe in read-only mode. Use this FIRST for broad coverage; explore interactively only where it flags problems or where journeys matter.",
|
|
1355
|
+
description: "Engine-side route sweep in ONE call: visits each path (default: all known routes not yet visited), records states into coverage memory, and returns a per-route health summary (HTTP status, element count, what the main area holds, oracle violations, dead-ends, auth-redirects, and ERROR-VIEW or STILL-LOADING for a main area showing only an alert or a loading placeholder). Navigation-only — safe in read-only mode. Use this FIRST for broad coverage; explore interactively only where it flags problems or where journeys matter.",
|
|
994
1356
|
inputSchema: {
|
|
995
|
-
paths: z
|
|
1357
|
+
paths: z
|
|
1358
|
+
.array(z.string())
|
|
1359
|
+
.max(150)
|
|
1360
|
+
.optional()
|
|
1361
|
+
.describe("Paths to visit, e.g. ['/orders','/settings'], or full URLs on the attached origin. A path resolves from the origin's root, whatever page the session attached on. Omit to crawl all unvisited known routes."),
|
|
996
1362
|
session: sessionParam,
|
|
997
1363
|
},
|
|
998
1364
|
}, serializedPerSession("scout_crawl", async ({ paths }, session) => {
|
|
@@ -1004,7 +1370,7 @@ server.registerTool("scout_crawl", {
|
|
|
1004
1370
|
}
|
|
1005
1371
|
}, 600_000));
|
|
1006
1372
|
server.registerTool("scout_run_plan", {
|
|
1007
|
-
description: "Execute up to 20 actions in ONE call — use for mechanical sequences (fill a form, walk a wizard) so each step doesn't cost a round-trip. Targets resolve at execution time by semantic locator: 'testid=…', 'text=…', 'label=…' or 'role=button[name=Save]' (never snapshot refs). The same steps, saved to .scenescout/flows/<name>.json with expect-text / expect-url / expect-request steps added, are replayed by `scenescout check` on every pull request. An `upload` step attaches a file as scout_upload does (target required — the file input or the control that opens its chooser; value = a fixture kind or a project-relative path). The plan ABORTS at the first NEW oracle violation, policy refusal, or failed step, returning a transcript of how far it got; repeats of already-reported violations do not abort (they stay logged for the report).",
|
|
1373
|
+
description: "Execute up to 20 actions in ONE call — use for mechanical sequences (fill a form, walk a wizard) so each step doesn't cost a round-trip. Targets resolve at execution time by semantic locator: 'testid=…', 'text=…', 'label=…' or 'role=button[name=Save]' (never snapshot refs). The same steps, saved to .scenescout/flows/<name>.json with expect-text / expect-url / expect-request steps added, are replayed by `scenescout check` on every pull request. An `upload` step attaches a file as scout_upload does (target required — the file input or the control that opens its chooser; value = a fixture kind or a project-relative path). The plan ABORTS at the first NEW oracle violation, policy refusal, or failed step, returning a transcript of how far it got; repeats of already-reported violations do not abort (they stay logged for the report). For a sweep of independent steps (tabs, filters, pages) pass onViolation \"continue\": a new error status or its console echo is listed on its step's line and the plan goes on; a failed step, a policy refusal or any other violation still stops it. A select step whose value names no option fails at once, listing the options.",
|
|
1008
1374
|
inputSchema: {
|
|
1009
1375
|
steps: z
|
|
1010
1376
|
.array(z.object({
|
|
@@ -1022,30 +1388,39 @@ server.registerTool("scout_run_plan", {
|
|
|
1022
1388
|
}))
|
|
1023
1389
|
.min(1)
|
|
1024
1390
|
.max(20),
|
|
1391
|
+
onViolation: z
|
|
1392
|
+
.enum(["stop", "continue"])
|
|
1393
|
+
.default("stop")
|
|
1394
|
+
.describe("stop (default): end the plan at the first new oracle violation, right for a form flow whose steps depend on each other. continue: list a new http_error or console_error on its step's line and run the next step, for a sweep of independent steps"),
|
|
1025
1395
|
task: taskParam,
|
|
1026
1396
|
objective: legacyObjectiveParam,
|
|
1027
1397
|
session: sessionParam,
|
|
1028
1398
|
},
|
|
1029
|
-
}, serializedPerSession("scout_run_plan", async ({ steps }, session) => {
|
|
1399
|
+
}, serializedPerSession("scout_run_plan", async ({ steps, onViolation }, session) => {
|
|
1030
1400
|
try {
|
|
1031
|
-
return text(await engineFor(session).runPlan(steps), session);
|
|
1401
|
+
return text(await engineFor(session).runPlan(steps, onViolation ?? "stop"), session);
|
|
1032
1402
|
}
|
|
1033
1403
|
catch (err) {
|
|
1034
1404
|
return errorText(err);
|
|
1035
1405
|
}
|
|
1036
1406
|
}, 240_000));
|
|
1407
|
+
const leaveParam = z
|
|
1408
|
+
.boolean()
|
|
1409
|
+
.optional()
|
|
1410
|
+
.describe("How to answer if the page asks to confirm leaving (a beforeunload prompt over unsent input): true leaves and discards that input, false stays. Omitted, observe and read-only stay and other modes leave. The result says when the page asked.");
|
|
1037
1411
|
server.registerTool("scout_click", {
|
|
1038
1412
|
description: "Click an element by its ref from the latest scout_snapshot. Returns the outcome plus any oracle violations triggered. clicks=2 (or 3) probes IMPATIENT-USER behaviour: a rapid multi-click that fires the same state-changing request twice means the control is not guarded against double submission (button stays enabled, endpoint not idempotent) — use it on every important submit/create button once; the result says explicitly whether duplicates fired.",
|
|
1039
1413
|
inputSchema: {
|
|
1040
1414
|
ref: z.string().describe("Element ref, e.g. e12"),
|
|
1041
1415
|
clicks: z.number().int().min(1).max(3).default(1).describe("1 = normal; 2-3 = rapid repeated clicks (double-submit probe)"),
|
|
1416
|
+
leave: leaveParam,
|
|
1042
1417
|
task: taskParam,
|
|
1043
1418
|
objective: legacyObjectiveParam,
|
|
1044
1419
|
session: sessionParam,
|
|
1045
1420
|
},
|
|
1046
|
-
}, serializedPerSession("scout_click", async ({ ref, clicks }, session) => {
|
|
1421
|
+
}, serializedPerSession("scout_click", async ({ ref, clicks, leave }, session) => {
|
|
1047
1422
|
try {
|
|
1048
|
-
return text(await engineFor(session).click(ref, clicks ?? 1), session);
|
|
1423
|
+
return text(await engineFor(session).click(ref, clicks ?? 1, leave), session);
|
|
1049
1424
|
}
|
|
1050
1425
|
catch (err) {
|
|
1051
1426
|
return errorText(err);
|
|
@@ -1121,10 +1496,10 @@ server.registerTool("scout_hover", {
|
|
|
1121
1496
|
}
|
|
1122
1497
|
}));
|
|
1123
1498
|
server.registerTool("scout_select", {
|
|
1124
|
-
description: "Select an option in a <select> by ref.",
|
|
1499
|
+
description: "Select an option in a <select> by ref. The value is matched against the options before anything is picked: an exact value, an exact label, either ignoring case, then a label it starts with. A value matching no option, or several, is refused at once with the options listed.",
|
|
1125
1500
|
inputSchema: {
|
|
1126
1501
|
ref: z.string(),
|
|
1127
|
-
value: z.string().describe("Option value or label"),
|
|
1502
|
+
value: z.string().describe("Option value or label (or the start of a label, when only one option has it)"),
|
|
1128
1503
|
task: taskParam,
|
|
1129
1504
|
objective: legacyObjectiveParam,
|
|
1130
1505
|
session: sessionParam,
|
|
@@ -1138,23 +1513,24 @@ server.registerTool("scout_select", {
|
|
|
1138
1513
|
}
|
|
1139
1514
|
}));
|
|
1140
1515
|
server.registerTool("scout_navigate", {
|
|
1141
|
-
description: "Navigate to a
|
|
1516
|
+
description: "Navigate to a path on the attached origin (e.g. '/orders') or a full URL on it. A path resolves from the origin's root, whatever page the session attached on. Also supports 'back' via scout_back.",
|
|
1142
1517
|
inputSchema: {
|
|
1143
1518
|
target: z.string().describe("Absolute URL or path like /settings"),
|
|
1519
|
+
leave: leaveParam,
|
|
1144
1520
|
task: taskParam,
|
|
1145
1521
|
objective: legacyObjectiveParam,
|
|
1146
1522
|
session: sessionParam,
|
|
1147
1523
|
},
|
|
1148
|
-
}, serializedPerSession("scout_navigate", async ({ target }, session) => {
|
|
1524
|
+
}, serializedPerSession("scout_navigate", async ({ target, leave }, session) => {
|
|
1149
1525
|
try {
|
|
1150
|
-
return text(await engineFor(session).navigate(target), session);
|
|
1526
|
+
return text(await engineFor(session).navigate(target, leave), session);
|
|
1151
1527
|
}
|
|
1152
1528
|
catch (err) {
|
|
1153
1529
|
return errorText(err);
|
|
1154
1530
|
}
|
|
1155
1531
|
}));
|
|
1156
1532
|
server.registerTool("scout_request", {
|
|
1157
|
-
description: "Call the app's own API as this session, with the UI bypassed — the check that turns a hidden or disabled control into a proven refusal. A button that is not shown proves nothing; the same action refused by the server does. The fetch runs IN the page, so it carries the session's cookies and replays the Authorization header the app itself last sent, and it passes through the same interception the write policy is enforced on: in safe-write a mutation on a record this session did not create is refused here exactly as it would be for a click, and that refusal is the engine's safety net, not a finding. Returns the status line, the timing, the headers that decide whether two responses are truly identical (content-type, location, www-authenticate, retry-after, cache-control), and the body. Unlike a shell call, every request is recorded in the run's trail and its signature is what a finding should quote. Paths are fenced to the attached origin: use another session to reach another host.",
|
|
1533
|
+
description: "Call the app's own API as this session, with the UI bypassed — the check that turns a hidden or disabled control into a proven refusal. A button that is not shown proves nothing; the same action refused by the server does. The fetch runs IN the page, so it carries the session's cookies and replays the Authorization header the app itself last sent, and it passes through the same interception the write policy is enforced on: in safe-write a mutation on a record this session did not create is refused here exactly as it would be for a click, and that refusal is the engine's safety net, not a finding. Returns the status line, the timing, the headers that decide whether two responses are truly identical (content-type, location, www-authenticate, retry-after, cache-control), and the body — its first 2000 characters, or the part named by select (one JSON value by path) or offset/limit (a window of characters). Unlike a shell call, every request is recorded in the run's trail and its signature is what a finding should quote. Paths are fenced to the attached origin: use another session to reach another host.",
|
|
1158
1534
|
inputSchema: {
|
|
1159
1535
|
path: z.string().min(1).max(2000).describe("Path on the attached origin, e.g. /api/things/12, or a full URL on that same origin"),
|
|
1160
1536
|
method: z.enum(["GET", "HEAD", "POST", "PUT", "PATCH", "DELETE", "OPTIONS"]).optional().describe("Default GET"),
|
|
@@ -1163,13 +1539,48 @@ server.registerTool("scout_request", {
|
|
|
1163
1539
|
.record(z.string().max(2000))
|
|
1164
1540
|
.optional()
|
|
1165
1541
|
.describe("Extra headers. One given here wins over the app's own, which is how a session tests a different or absent credential."),
|
|
1542
|
+
select: z
|
|
1543
|
+
.string()
|
|
1544
|
+
.max(500)
|
|
1545
|
+
.optional()
|
|
1546
|
+
.describe('Return one value of a JSON response body by its dotted path, e.g. "stats.open" or "items.0.name", pretty-printed and up to 8000 characters. A path that is not there says which keys are.'),
|
|
1547
|
+
offset: z
|
|
1548
|
+
.number()
|
|
1549
|
+
.int()
|
|
1550
|
+
.min(0)
|
|
1551
|
+
.max(1_000_000)
|
|
1552
|
+
.optional()
|
|
1553
|
+
.describe("Return the body (or the selected value) from this character on. The body is cut at 2000 characters by default; the result names the next offset."),
|
|
1554
|
+
limit: z
|
|
1555
|
+
.number()
|
|
1556
|
+
.int()
|
|
1557
|
+
.min(1)
|
|
1558
|
+
.max(8000)
|
|
1559
|
+
.optional()
|
|
1560
|
+
.describe("How many characters to return with offset or select. Default 2000, or 8000 for a select with no offset."),
|
|
1166
1561
|
task: taskParam,
|
|
1167
1562
|
objective: legacyObjectiveParam,
|
|
1168
1563
|
session: sessionParam,
|
|
1169
1564
|
},
|
|
1170
1565
|
}, serializedPerSession("scout_request", async (args, session) => {
|
|
1171
1566
|
try {
|
|
1172
|
-
|
|
1567
|
+
const view = { select: args.select, offset: args.offset, limit: args.limit };
|
|
1568
|
+
return text(await engineFor(session).apiRequest({ method: args.method, path: args.path, body: args.body, headers: args.headers, view }), session);
|
|
1569
|
+
}
|
|
1570
|
+
catch (err) {
|
|
1571
|
+
return errorText(err);
|
|
1572
|
+
}
|
|
1573
|
+
}));
|
|
1574
|
+
server.registerTool("scout_network", {
|
|
1575
|
+
description: "List the data requests (fetch and XHR) the current page made since its document loaded: method, path, status and time, oldest first, each marked with the route it was sent from when a client-side route change moved the page since. Use it when the page and the server seem to disagree — an empty list where the API has data, a stale value after a save — to tell a request that failed, one still pending and one that never ran apart. Read-only: it lists what the browser already saw and sends nothing. Credentials in query strings are redacted. A full page load starts a new list; scout_request's own calls are marked.",
|
|
1576
|
+
inputSchema: {
|
|
1577
|
+
contains: z.string().max(200).optional().describe("Only requests whose path contains this text, e.g. /api/things"),
|
|
1578
|
+
limit: z.number().int().min(1).max(200).optional().describe("How many of the newest to list. Default 40."),
|
|
1579
|
+
session: sessionParam,
|
|
1580
|
+
},
|
|
1581
|
+
}, serializedPerSession("scout_network", async (args, session) => {
|
|
1582
|
+
try {
|
|
1583
|
+
return text(engineFor(session).listPageRequests({ contains: args.contains, limit: args.limit }), session);
|
|
1173
1584
|
}
|
|
1174
1585
|
catch (err) {
|
|
1175
1586
|
return errorText(err);
|
|
@@ -1177,10 +1588,10 @@ server.registerTool("scout_request", {
|
|
|
1177
1588
|
}));
|
|
1178
1589
|
server.registerTool("scout_back", {
|
|
1179
1590
|
description: "Go back in browser history (tests back-button resilience).",
|
|
1180
|
-
inputSchema: { task: taskParam, objective: legacyObjectiveParam, session: sessionParam },
|
|
1181
|
-
}, serializedPerSession("scout_back", async (
|
|
1591
|
+
inputSchema: { leave: leaveParam, task: taskParam, objective: legacyObjectiveParam, session: sessionParam },
|
|
1592
|
+
}, serializedPerSession("scout_back", async ({ leave }, session) => {
|
|
1182
1593
|
try {
|
|
1183
|
-
return text(await engineFor(session).goBack(), session);
|
|
1594
|
+
return text(await engineFor(session).goBack(leave), session);
|
|
1184
1595
|
}
|
|
1185
1596
|
catch (err) {
|
|
1186
1597
|
return errorText(err);
|
|
@@ -1322,7 +1733,7 @@ server.registerTool("scout_capture", {
|
|
|
1322
1733
|
}
|
|
1323
1734
|
}));
|
|
1324
1735
|
server.registerTool("scout_finding", {
|
|
1325
|
-
description: "Record a structured finding (bug, UX issue, or improvement). Deduplicates across runs; automatically captures the recent action trace as the repro. Use for anything worth reporting: crashes, oracle violations you confirmed, dead ends, confusing UX, permission leaks, missing testids — and design-audit improvement opportunities (ux-polish) with their concrete measurements.",
|
|
1736
|
+
description: "Record a structured finding (bug, UX issue, or improvement). Deduplicates across runs; automatically captures the recent action trace as the repro, and a picture of what it is about (the element `ref` names, else the viewport), kept for report.html and returned in this result as an image. Use for anything worth reporting: crashes, oracle violations you confirmed, dead ends, confusing UX, permission leaks, missing testids — and design-audit improvement opportunities (ux-polish) with their concrete measurements.",
|
|
1326
1737
|
inputSchema: {
|
|
1327
1738
|
severity: z.enum(["high", "medium", "low"]),
|
|
1328
1739
|
category: z.enum(FINDING_CATEGORIES).describe("Pick the closest — use 'other' only when nothing fits"),
|
|
@@ -1338,9 +1749,14 @@ server.registerTool("scout_finding", {
|
|
|
1338
1749
|
.max(LANE_CONVENTION_MAX)
|
|
1339
1750
|
.optional()
|
|
1340
1751
|
.describe("Only for a WORTH-A-LOOK finding: the observation is real, and it is a defect only under a convention of this project you cannot see. Name that convention, e.g. 'a 4px spacing scale' or 'test ids on every control'. The report lists it under \"Worth a look\", apart from the defects, and does not count it as one. Not for \"I could not tell\": leave that unfiled or look closer. Omit for a defect."),
|
|
1752
|
+
ref: z
|
|
1753
|
+
.string()
|
|
1754
|
+
.max(20)
|
|
1755
|
+
.optional()
|
|
1756
|
+
.describe("The ref, from the latest scout_snapshot, of the element the finding is about. Its picture (the element plus a margin) is kept with the finding and shown in report.html. Omit it and the picture is the viewport."),
|
|
1341
1757
|
session: sessionParam,
|
|
1342
1758
|
},
|
|
1343
|
-
}, serializedPerSession("scout_finding", async ({ severity, category, title, detail, evidence, convention, }, session) => {
|
|
1759
|
+
}, serializedPerSession("scout_finding", async ({ severity, category, title, detail, evidence, convention, ref, }, session) => {
|
|
1344
1760
|
try {
|
|
1345
1761
|
const eng = engineFor(session);
|
|
1346
1762
|
if (!eng.memory)
|
|
@@ -1358,34 +1774,101 @@ server.registerTool("scout_finding", {
|
|
|
1358
1774
|
});
|
|
1359
1775
|
if (filed.judgeError)
|
|
1360
1776
|
logLine(`dedup judge: ${filed.judgeError}; the rule decided`);
|
|
1361
|
-
|
|
1777
|
+
const picture = await findingPicture(eng, filed, ref);
|
|
1778
|
+
const result = text(filedText(filed, category) + picture.line, session);
|
|
1779
|
+
if (picture.image)
|
|
1780
|
+
result.content.push(picture.image);
|
|
1781
|
+
return result;
|
|
1782
|
+
}
|
|
1783
|
+
catch (err) {
|
|
1784
|
+
return errorText(err);
|
|
1785
|
+
}
|
|
1786
|
+
}));
|
|
1787
|
+
// ---- The run-status pane: an MCP App (SEP-1865, `io.modelcontextprotocol/ui`, spec 2026-01-26). ----
|
|
1788
|
+
// scout_status links the pane through _meta.ui.resourceUri; the pane polls the
|
|
1789
|
+
// app-only tool. Neither goes through a session queue: like the live view, a
|
|
1790
|
+
// watcher must never wait behind the agent's calls, and both only read.
|
|
1791
|
+
function statusPaneData() {
|
|
1792
|
+
const eng = [lastWriter ? engines.get(lastWriter.session) : undefined, ...engines.values()].find((e) => !!e?.memory);
|
|
1793
|
+
const memory = eng?.memory ?? null;
|
|
1794
|
+
let coverage = null;
|
|
1795
|
+
if (eng && memory) {
|
|
1796
|
+
const all = eng.allKnownRoutes();
|
|
1797
|
+
const cov = memory.coverage();
|
|
1798
|
+
coverage = {
|
|
1799
|
+
routesVisited: all.length - eng.unvisitedKnownRoutes().length,
|
|
1800
|
+
routesTotal: all.length,
|
|
1801
|
+
states: cov.states,
|
|
1802
|
+
elementsExercised: cov.elementsExercised,
|
|
1803
|
+
elementsTotal: cov.elementsTotal,
|
|
1804
|
+
};
|
|
1805
|
+
}
|
|
1806
|
+
return paneData({
|
|
1807
|
+
nowMs: Date.now(),
|
|
1808
|
+
version: PKG_VERSION,
|
|
1809
|
+
live: liveAddress,
|
|
1810
|
+
liveError,
|
|
1811
|
+
liveOff: process.env[LIVE_ENV] === "off",
|
|
1812
|
+
sessions: board.list(),
|
|
1813
|
+
findings: memory ? memory.findings : null,
|
|
1814
|
+
runStart: memory?.sessionStart,
|
|
1815
|
+
coverage,
|
|
1816
|
+
});
|
|
1817
|
+
}
|
|
1818
|
+
function statusResult() {
|
|
1819
|
+
try {
|
|
1820
|
+
const data = statusPaneData();
|
|
1821
|
+
return { content: [{ type: "text", text: paneText(data) }], structuredContent: { ...data } };
|
|
1362
1822
|
}
|
|
1363
1823
|
catch (err) {
|
|
1364
1824
|
return errorText(err);
|
|
1365
1825
|
}
|
|
1826
|
+
}
|
|
1827
|
+
server.registerResource("scenescout-status", STATUS_PANE_URI, {
|
|
1828
|
+
title: "SceneScout run status",
|
|
1829
|
+
description: "The run's sessions, open findings by severity, coverage and the live view's address, refreshed every few seconds.",
|
|
1830
|
+
mimeType: MCP_APP_MIME,
|
|
1831
|
+
// No outside origin: the page is one self-contained document, so the host's restrictive default CSP applies.
|
|
1832
|
+
_meta: { ui: { csp: {}, prefersBorder: true } },
|
|
1833
|
+
}, () => ({
|
|
1834
|
+
contents: [{ uri: STATUS_PANE_URI, mimeType: MCP_APP_MIME, text: statusPanePage(PKG_VERSION), _meta: { ui: { csp: {}, prefersBorder: true } } }],
|
|
1366
1835
|
}));
|
|
1836
|
+
server.registerTool(STATUS_TOOL, {
|
|
1837
|
+
title: "Run status",
|
|
1838
|
+
description: "Show the person running you how the run stands: each session's objective and current task, open findings by severity, coverage, and the live view's address. " +
|
|
1839
|
+
"Call it when the user wants to watch or asks how the run is going. A host that renders MCP Apps shows a pane that keeps itself up to date; every other host gets the same as text, and you pass the `Live view:` address on. Takes no input and touches no browser.",
|
|
1840
|
+
// The legacy flat key alongside _meta.ui.resourceUri, as the ext-apps SDK's registerAppTool sets both for hosts that read only the old one.
|
|
1841
|
+
_meta: { ui: { resourceUri: STATUS_PANE_URI }, "ui/resourceUri": STATUS_PANE_URI },
|
|
1842
|
+
}, async () => statusResult());
|
|
1843
|
+
server.registerTool(STATUS_POLL_TOOL, {
|
|
1844
|
+
title: "Run status (for the pane)",
|
|
1845
|
+
description: `Called by the ${STATUS_TOOL} pane every few seconds to refresh itself. Not for the agent: call ${STATUS_TOOL} instead.`,
|
|
1846
|
+
_meta: { ui: { visibility: ["app"] } },
|
|
1847
|
+
}, async () => statusResult());
|
|
1367
1848
|
server.registerTool("scout_coverage", {
|
|
1368
|
-
description: "Show exploration coverage: states visited
|
|
1369
|
-
inputSchema: {
|
|
1370
|
-
|
|
1849
|
+
description: "Show exploration coverage: states visited, which elements remain unexercised, which options of a dropdown used this run no session has chosen yet, and which forms seen this run no session has submitted with every text field blank. Use to decide where to explore next and when the level's budget is satisfied. In a parallel run it shows this session's own work by default — the routes it reached this run and the forms it saw — so one lane is not handed another's gaps; scope:\"project\" shows every session's, each form and route tagged with the sessions that saw it.",
|
|
1850
|
+
inputSchema: {
|
|
1851
|
+
scope: z
|
|
1852
|
+
.enum(["session", "project"])
|
|
1853
|
+
.optional()
|
|
1854
|
+
.describe("'session': only the routes this session reached this run and the forms it saw. 'project': every route in the memory, across runs and sessions. Default: 'session' when other sessions share this project, else 'project'."),
|
|
1855
|
+
session: sessionParam,
|
|
1856
|
+
},
|
|
1857
|
+
}, serializedPerSession("scout_coverage", async (args, session) => {
|
|
1371
1858
|
try {
|
|
1372
1859
|
const eng = engineFor(session);
|
|
1373
|
-
|
|
1860
|
+
const memory = eng.memory;
|
|
1861
|
+
if (!memory)
|
|
1374
1862
|
throw new Error("Not attached.");
|
|
1375
|
-
const
|
|
1863
|
+
const shared = [...engines.values()].some((e) => e !== eng && e.memory === memory);
|
|
1376
1864
|
const unvisited = eng.unvisitedKnownRoutes();
|
|
1377
1865
|
const lines = [
|
|
1378
|
-
...(
|
|
1866
|
+
...(memory.lastSaveError
|
|
1379
1867
|
? [
|
|
1380
|
-
`⚠ MEMORY WRITE FAILING: ${
|
|
1868
|
+
`⚠ MEMORY WRITE FAILING: ${memory.lastSaveError} — coverage/findings since the last successful write are NOT persisted to disk. If this doesn't clear on its own, check the project directory still exists and is writable.`,
|
|
1381
1869
|
]
|
|
1382
1870
|
: []),
|
|
1383
|
-
|
|
1384
|
-
formatRouteCoverage(eng.allKnownRoutes(), unvisited),
|
|
1385
|
-
`Unexercised elements by route:`,
|
|
1386
|
-
...cov.unexercised.slice(0, 25).map((u) => ` ${u.state}: ${u.keys.slice(0, 6).join(", ")}${u.keys.length > 6 ? ` … +${u.keys.length - 6}` : ""}`),
|
|
1387
|
-
...formatUnchosenOptions(eng.memory.unchosenOptions()),
|
|
1388
|
-
...formatNeverSubmittedEmpty(eng.memory.formsNeverSubmittedEmpty()),
|
|
1871
|
+
...coverageView(memory, eng.sessionKey, args.scope ?? (shared ? "session" : "project"), formatRouteCoverage(eng.allKnownRoutes(), unvisited)),
|
|
1389
1872
|
];
|
|
1390
1873
|
return text(lines.join("\n"), session);
|
|
1391
1874
|
}
|
|
@@ -1405,9 +1888,13 @@ server.registerTool("scout_report", {
|
|
|
1405
1888
|
.enum(["index", "full"])
|
|
1406
1889
|
.default("index")
|
|
1407
1890
|
.describe("How much of the history to print. 'index' lists findings from earlier runs, and resolved ones, as a row each: id, severity, age, title. 'full' prints every one in full as before — on one project that was 1.75 MB against 113 KB, nearly half of it findings already fixed. Use 'full' when handing the document to someone who has no access to the memory."),
|
|
1891
|
+
report: z
|
|
1892
|
+
.enum(REPORT_AUDIENCES)
|
|
1893
|
+
.default(DEFAULT_REPORT_AUDIENCE)
|
|
1894
|
+
.describe("Which parts the report carries. 'both' (default): a plain-language section first — a short summary, then each problem with numbered steps, what was expected, what happened, its picture and its impact (blocks users, annoying, cosmetic), each with its technical detail folded beneath — followed by the technical report. 'qa': the plain section alone, for a tester or anyone not technical. 'dev': the technical report alone, as before the plain section existed. It applies to the files this call writes; the live view always shows both."),
|
|
1408
1895
|
session: sessionParam,
|
|
1409
1896
|
},
|
|
1410
|
-
}, serializedPerSession("scout_report", async ({ force, level, history }, session) => {
|
|
1897
|
+
}, serializedPerSession("scout_report", async ({ force, level, history, report }, session) => {
|
|
1411
1898
|
try {
|
|
1412
1899
|
const eng = engineFor(session);
|
|
1413
1900
|
if (!eng.memory)
|
|
@@ -1444,6 +1931,7 @@ server.registerTool("scout_report", {
|
|
|
1444
1931
|
routesTotal: all.length,
|
|
1445
1932
|
designAudits: auditsThisRun,
|
|
1446
1933
|
unvisitedRoutes: unvisited,
|
|
1934
|
+
knownRoutes: all,
|
|
1447
1935
|
mode: eng.mode,
|
|
1448
1936
|
});
|
|
1449
1937
|
if (lvl === "extensive" && gapList.length > 0) {
|
|
@@ -1458,21 +1946,25 @@ server.registerTool("scout_report", {
|
|
|
1458
1946
|
return text(`NOT GENERATED — the '${lvl}' completion contract is unmet:\n\n${gates.join("\n\n")}\n\n` +
|
|
1459
1947
|
`Then call scout_report again. Pass force=true ONLY if the user explicitly capped the budget.`, session);
|
|
1460
1948
|
}
|
|
1461
|
-
const { path: p, summary } = generateReport(eng.memory, eng.oracleLog.all, {
|
|
1949
|
+
const { path: p, summary, html, } = generateReport(eng.memory, eng.oracleLog.all, {
|
|
1462
1950
|
history,
|
|
1951
|
+
report,
|
|
1463
1952
|
routesVisited: all.length - unvisited.length,
|
|
1464
1953
|
routesTotal: all.length,
|
|
1465
1954
|
designAudits: auditsThisRun,
|
|
1466
1955
|
createdResources: eng.createdResources,
|
|
1467
1956
|
unvisitedRoutes: unvisited,
|
|
1957
|
+
knownRoutes: all,
|
|
1468
1958
|
mode: eng.mode,
|
|
1469
1959
|
trustedEmbeds: [...eng.trustedEmbeds],
|
|
1960
|
+
readPosts: eng.readPosts.map((e) => e.entry),
|
|
1470
1961
|
policyAttributed: eng.oracleLog.policyAttributed,
|
|
1471
1962
|
// Which sessions are still open decides whether a quiet one is holding a browser, and how long its trailing idle runs.
|
|
1472
1963
|
attachedSessions: [...engines.keys()],
|
|
1473
1964
|
});
|
|
1474
1965
|
void p;
|
|
1475
|
-
|
|
1966
|
+
const openNote = html && openDecisions.get(session)?.report ? openForUser("the report", html) : "";
|
|
1967
|
+
return text(summary + openNote, session);
|
|
1476
1968
|
}
|
|
1477
1969
|
catch (err) {
|
|
1478
1970
|
return errorText(err);
|
|
@@ -1536,6 +2028,108 @@ server.registerTool("scout_verify", {
|
|
|
1536
2028
|
return errorText(err);
|
|
1537
2029
|
}
|
|
1538
2030
|
}));
|
|
2031
|
+
// Answering the tickets: read their acceptance criteria, then record a verdict on each.
|
|
2032
|
+
server.registerTool("scout_tickets", {
|
|
2033
|
+
description: "Read the tickets or acceptance criteria the person gave this run, pasted (`text`) or from a file (`path`), and keep them so the report answers each criterion: passed, failed or not tested. " +
|
|
2034
|
+
'Recognises Given/When/Then scenarios, checklists, numbered and "AC1:" criteria, and lists under an "Acceptance criteria" heading; several tickets may be given at once. ' +
|
|
2035
|
+
"A ticket with no recognisable criteria is reported as such, never guessed at. Returns each criterion's id (AC1, AC2, …) to judge it by with scout_criterion. Touches no browser.",
|
|
2036
|
+
inputSchema: {
|
|
2037
|
+
text: z.string().max(MAX_TICKET_TEXT).optional().describe("The tickets as pasted. Pass this or `path`."),
|
|
2038
|
+
path: z
|
|
2039
|
+
.string()
|
|
2040
|
+
.optional()
|
|
2041
|
+
.describe(`A ticket file to read (${TICKET_FILE_EXTENSIONS.join(", ")}), absolute or relative to the project folder. Pass this or \`text\`.`),
|
|
2042
|
+
session: sessionParam,
|
|
2043
|
+
},
|
|
2044
|
+
}, serializedPerSession("scout_tickets", async ({ text: pasted, path: file }, session) => {
|
|
2045
|
+
try {
|
|
2046
|
+
const eng = engineFor(session);
|
|
2047
|
+
if (!eng.memory)
|
|
2048
|
+
throw new Error("Not attached — attach first, so the tickets are kept with the project.");
|
|
2049
|
+
if ((pasted === undefined) === (file === undefined))
|
|
2050
|
+
return text("Pass the tickets as `text`, or a file as `path`: one of the two.", session);
|
|
2051
|
+
let body = pasted ?? "";
|
|
2052
|
+
let source = "pasted text";
|
|
2053
|
+
if (file !== undefined) {
|
|
2054
|
+
// The file a link points at is what is read, so its name is what is checked: a "notes.md" link to a key file is refused.
|
|
2055
|
+
const full = fs.realpathSync(path.isAbsolute(file) ? file : path.resolve(path.dirname(eng.memory.dir), file));
|
|
2056
|
+
if (!isTicketFileName(full))
|
|
2057
|
+
return text(`Not read: a ticket file is one of ${TICKET_FILE_EXTENSIONS.join(", ")}. Paste anything else as \`text\`.`, session);
|
|
2058
|
+
const stat = fs.statSync(full);
|
|
2059
|
+
if (!stat.isFile())
|
|
2060
|
+
return text(`Not read: ${full} is not a file.`, session);
|
|
2061
|
+
if (stat.size > MAX_TICKET_FILE_BYTES)
|
|
2062
|
+
return text(`Not read: ${full} is larger than ${MAX_TICKET_FILE_BYTES} bytes.`, session);
|
|
2063
|
+
body = fs.readFileSync(full, "utf8");
|
|
2064
|
+
source = path.basename(full);
|
|
2065
|
+
}
|
|
2066
|
+
if (!body.trim())
|
|
2067
|
+
return text("Nothing to read: the tickets are empty.", session);
|
|
2068
|
+
const parsed = parseTickets(body, source);
|
|
2069
|
+
const kept = eng.memory.addTickets(parsed);
|
|
2070
|
+
// Say what a bound left out, so a long backlog is never answered in part without a word.
|
|
2071
|
+
const cuts = [
|
|
2072
|
+
...(body.length > MAX_TICKET_TEXT ? [`only the first ${MAX_TICKET_TEXT} characters were read`] : []),
|
|
2073
|
+
...(parsed.length >= MAX_TICKETS ? [`at most ${MAX_TICKETS} tickets are read at once`] : []),
|
|
2074
|
+
];
|
|
2075
|
+
return text(formatReading(kept) + (cuts.length ? `\n\n⚠ Not everything was read: ${cuts.join("; ")}. Read the rest in another call.` : ""), session);
|
|
2076
|
+
}
|
|
2077
|
+
catch (err) {
|
|
2078
|
+
return errorText(err);
|
|
2079
|
+
}
|
|
2080
|
+
}));
|
|
2081
|
+
server.registerTool("scout_criterion", {
|
|
2082
|
+
description: "Record whether one acceptance criterion of a ticket read with scout_tickets passed, failed or was not tested, with how sure you are. " +
|
|
2083
|
+
"The link from a criterion to the findings that show it is YOUR judgement, stated with a confidence — never matched on words. " +
|
|
2084
|
+
'A "fail" names the findings that show it (file them with scout_finding first); "not-tested" says why in untestedBecause. Recording the same criterion again from the same session replaces your earlier verdict. Touches no browser.',
|
|
2085
|
+
inputSchema: {
|
|
2086
|
+
ticket: z.string().min(1).describe('The ticket\'s id as scout_tickets gave it, e.g. "PROJ-12" or "T1"'),
|
|
2087
|
+
criterion: z.string().min(1).describe('The criterion\'s id, e.g. "AC2" (or just "2")'),
|
|
2088
|
+
verdict: z.enum(CRITERION_VERDICTS).describe('"pass", "fail" or "not-tested"'),
|
|
2089
|
+
findings: z
|
|
2090
|
+
.array(z.string())
|
|
2091
|
+
.max(MAX_CRITERION_FINDINGS)
|
|
2092
|
+
.optional()
|
|
2093
|
+
.describe("Ids of the findings that show this criterion failing (required for a fail; may be given for a pass, none for not-tested)"),
|
|
2094
|
+
confidence: z.number().min(0).max(1).describe("How sure you are of this verdict and of the findings linked to it, from 0 to 1. State it honestly"),
|
|
2095
|
+
reason: z.string().min(1).max(MAX_REASON).describe("What you saw, or why it could not be tried, in a sentence"),
|
|
2096
|
+
untestedBecause: z
|
|
2097
|
+
.enum(NOT_TESTED_REASONS)
|
|
2098
|
+
.optional()
|
|
2099
|
+
.describe('Only with verdict "not-tested": "no-access" (the role this run used could not reach it), "observe-blocked" (it needs a change sent and this session is in observe mode), "out-of-scope" (it is outside what this run could check, such as an email or another system)'),
|
|
2100
|
+
session: sessionParam,
|
|
2101
|
+
},
|
|
2102
|
+
}, serializedPerSession("scout_criterion", async (args, session) => {
|
|
2103
|
+
try {
|
|
2104
|
+
const eng = engineFor(session);
|
|
2105
|
+
const memory = eng.memory;
|
|
2106
|
+
if (!memory)
|
|
2107
|
+
throw new Error("Not attached.");
|
|
2108
|
+
const judged = judgeCriterion(args, {
|
|
2109
|
+
tickets: memory.tickets,
|
|
2110
|
+
findings: memory.findings,
|
|
2111
|
+
mode: eng.mode,
|
|
2112
|
+
session: eng.sessionKey,
|
|
2113
|
+
at: new Date().toISOString(),
|
|
2114
|
+
});
|
|
2115
|
+
if (!judged.ok)
|
|
2116
|
+
return text(`Not recorded: ${judged.reason}.`, session);
|
|
2117
|
+
memory.addCriterionVerdict(judged.record);
|
|
2118
|
+
const r = judged.record;
|
|
2119
|
+
const ticket = memory.tickets.find((t) => t.id === r.ticket);
|
|
2120
|
+
const criterion = ticket ? findCriterion(ticket, r.criterion) : undefined;
|
|
2121
|
+
const linked = r.findings.map((id) => memory.findings.find((f) => f.id === id)).filter((f) => f !== undefined);
|
|
2122
|
+
const said = r.verdict === "pass" ? "passes" : r.verdict === "fail" ? "fails" : `was not tested (${r.untestedBecause})`;
|
|
2123
|
+
return text([
|
|
2124
|
+
`Recorded: ${r.ticket} ${r.criterion} ${said}, confidence ${r.confidence.toFixed(2)}.`,
|
|
2125
|
+
...(criterion ? [` ${criterion.text}`] : []),
|
|
2126
|
+
...linked.map((f) => ` linked: [${f.severity}] ${f.title} (${f.id})`),
|
|
2127
|
+
].join("\n"), session);
|
|
2128
|
+
}
|
|
2129
|
+
catch (err) {
|
|
2130
|
+
return errorText(err);
|
|
2131
|
+
}
|
|
2132
|
+
}));
|
|
1539
2133
|
server.registerTool("scout_close", {
|
|
1540
2134
|
description: "Close a session's browser (memory persists on disk). Default: the DEFAULT session. Pass session to close a specific one, or all=true to close every live session at the end of a multi-role run. " +
|
|
1541
2135
|
"A session that is a lane of a parallel run (named by scout_lane_brief or scout_lane_report) is not closed until its lane report has been accepted by scout_lane_report, since folding needs the session attached; the refusal names every such lane.",
|
|
@@ -1588,6 +2182,7 @@ server.registerTool("scout_close", {
|
|
|
1588
2182
|
await eng.close();
|
|
1589
2183
|
const saveError = eng.memory?.lastSaveError;
|
|
1590
2184
|
engines.delete(name);
|
|
2185
|
+
openDecisions.delete(name);
|
|
1591
2186
|
// The last session on this project ends its run.
|
|
1592
2187
|
if (eng.memory && ![...engines.values()].some((e) => e.memory === eng.memory))
|
|
1593
2188
|
eng.memory.endRun();
|
|
@@ -1648,7 +2243,7 @@ async function shutdown() {
|
|
|
1648
2243
|
console.error(`[scenescout] could not remove the live view's token file in ${dir}: ${err instanceof Error ? err.message : String(err)}`);
|
|
1649
2244
|
}
|
|
1650
2245
|
}
|
|
1651
|
-
await Promise.allSettled([live?.stop(), ...[...engines.values()].map((e) => e.close())]);
|
|
2246
|
+
await Promise.allSettled([live?.stop(), ...[...engines.values()].map((e) => e.close()), ...pendingLogins.all().map((p) => p.cancel())]);
|
|
1652
2247
|
}
|
|
1653
2248
|
process.on("SIGINT", () => {
|
|
1654
2249
|
void shutdown().finally(() => process.exit(0));
|