scenescout 3.15.0 → 3.17.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +87 -0
- package/README.md +70 -18
- package/dist/browsers.js +28 -0
- package/dist/check-run.js +191 -14
- package/dist/ci-run.js +268 -52
- package/dist/cli.js +107 -47
- package/dist/commands.js +3 -2
- package/dist/engine/baseline.js +377 -0
- package/dist/engine/brief.js +16 -7
- package/dist/engine/browser.js +1147 -286
- package/dist/engine/calibration.js +61 -30
- package/dist/engine/capture.js +164 -0
- package/dist/engine/check.js +244 -42
- package/dist/engine/ci-lanes.js +215 -0
- package/dist/engine/ci.js +136 -18
- package/dist/engine/claims.js +159 -3
- package/dist/engine/collector.js +561 -30
- package/dist/engine/crawl.js +49 -0
- package/dist/engine/design.js +281 -38
- package/dist/engine/export.js +877 -0
- package/dist/engine/fingerprint.js +92 -4
- package/dist/engine/flow.js +18 -6
- package/dist/engine/forms.js +181 -18
- package/dist/engine/journey.js +29 -1
- package/dist/engine/lane.js +13 -3
- package/dist/engine/launch.js +45 -6
- package/dist/engine/limits.js +7 -0
- package/dist/engine/live-page.js +49 -2
- package/dist/engine/live.js +4 -1
- package/dist/engine/memory.js +501 -47
- package/dist/engine/open.js +118 -0
- package/dist/engine/oracles.js +41 -1
- package/dist/engine/plain.js +268 -0
- package/dist/engine/png.js +127 -0
- package/dist/engine/policy.js +379 -9
- package/dist/engine/probes.js +3 -2
- package/dist/engine/profiles.js +45 -9
- package/dist/engine/project-folder.js +191 -0
- package/dist/engine/refresh.js +68 -3
- package/dist/engine/replay.js +63 -10
- package/dist/engine/report.js +241 -40
- package/dist/engine/request.js +317 -23
- package/dist/engine/sarif.js +120 -0
- package/dist/engine/settle.js +67 -0
- package/dist/engine/signed-in.js +256 -0
- package/dist/engine/status-pane-page.js +441 -0
- package/dist/engine/status-pane.js +128 -0
- package/dist/engine/tickets.js +671 -0
- package/dist/engine/unload.js +3 -2
- package/dist/export-run.js +633 -0
- package/dist/first-run.js +5 -0
- package/dist/installer.js +378 -8
- package/dist/intake.js +104 -0
- package/dist/login-run.js +250 -36
- package/dist/mcp-server.js +660 -65
- package/dist/playbook.js +5 -0
- package/dist/prompts.js +106 -0
- package/package.json +8 -5
- package/skills/scenescout/SKILL.md +49 -16
|
@@ -0,0 +1,215 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Lanes for `scenescout ci` (--lanes): the app split between several model
|
|
3
|
+
* loops that explore at once, each in its own browser session and its own
|
|
4
|
+
* part of the app, drawing on the run's one budget, with their findings folded
|
|
5
|
+
* into the one report. This file holds the rules, none of which needs a
|
|
6
|
+
* browser or a network: which routes the planning crawl found, how they are
|
|
7
|
+
* split (brief.ts, the split scout_lane_brief makes), what each lane is told,
|
|
8
|
+
* and how the lanes' endings become the run's. The loop that runs them is
|
|
9
|
+
* src/ci-run.ts; the shared budget is engine/ci.ts's. Why: ADR 20.
|
|
10
|
+
*/
|
|
11
|
+
import { LANE_RULES, planLanes } from "./brief.js";
|
|
12
|
+
import { CI_TOOLS } from "./ci.js";
|
|
13
|
+
/** The session the run attaches first. It plans, holds no lane, and writes the report once every lane is done. */
|
|
14
|
+
export const PLANNER_SESSION = "default";
|
|
15
|
+
/** Crawls the planning step makes at most: each visits the routes the pages of the one before linked to. */
|
|
16
|
+
export const PLAN_CRAWL_ROUNDS = 3;
|
|
17
|
+
/** The most routes a lane's first message lists, and the most crawl lines it is handed. */
|
|
18
|
+
const LIST_MAX = 40;
|
|
19
|
+
/**
|
|
20
|
+
* A lane's tools: an exploring run's, less scout_report. The run writes one
|
|
21
|
+
* report for every lane once all are done; a lane that wrote its own would
|
|
22
|
+
* report the others' work as unfinished.
|
|
23
|
+
*/
|
|
24
|
+
export const LANE_TOOLS = CI_TOOLS.filter((t) => t !== "scout_report");
|
|
25
|
+
// ── what the planning crawl found ───────────────────────────────────────────
|
|
26
|
+
/**
|
|
27
|
+
* A route line of a scout_crawl result: `<path> — <status> · <n> el …`, or
|
|
28
|
+
* `<path> — LOAD FAILED`. The path is what the crawl visited, as the browser
|
|
29
|
+
* opened it.
|
|
30
|
+
*/
|
|
31
|
+
const CRAWL_LINE = /^(\/\S*) — (.+)$/;
|
|
32
|
+
/** The heading of the crawl's problem list, whose entries name a route and say what is wrong on it. */
|
|
33
|
+
const PROBLEMS_HEADING = /^PROBLEM ROUTES \(\d+\):$/;
|
|
34
|
+
/** A problem entry's first line: the route, then ` → …`, `: <why it did not load>`, or nothing (its detail lines follow, indented). */
|
|
35
|
+
const PROBLEM_LINE = /^(\/\S*?)(?: →.*|: .*)?$/;
|
|
36
|
+
/**
|
|
37
|
+
* What a scout_crawl result says about each route it visited: the route's own
|
|
38
|
+
* line, then whatever its problem list says about it. Routes it skipped as
|
|
39
|
+
* off-origin are not the app's, and are left out.
|
|
40
|
+
*/
|
|
41
|
+
export function crawlNotes(text) {
|
|
42
|
+
const notes = new Map();
|
|
43
|
+
const add = (route, line) => {
|
|
44
|
+
const list = notes.get(route);
|
|
45
|
+
if (list)
|
|
46
|
+
list.push(line);
|
|
47
|
+
else
|
|
48
|
+
notes.set(route, [line]);
|
|
49
|
+
};
|
|
50
|
+
let inProblems = false;
|
|
51
|
+
let current = null;
|
|
52
|
+
for (const raw of text.split(/\r?\n/)) {
|
|
53
|
+
const line = raw.trimEnd();
|
|
54
|
+
if (PROBLEMS_HEADING.test(line)) {
|
|
55
|
+
inProblems = true;
|
|
56
|
+
continue;
|
|
57
|
+
}
|
|
58
|
+
if (!inProblems) {
|
|
59
|
+
const m = CRAWL_LINE.exec(line);
|
|
60
|
+
if (m && !/^SKIPPED/.test(m[2]))
|
|
61
|
+
add(m[1], line);
|
|
62
|
+
continue;
|
|
63
|
+
}
|
|
64
|
+
if (line === "") {
|
|
65
|
+
// A blank line ends the entry it follows, not the list.
|
|
66
|
+
current = null;
|
|
67
|
+
continue;
|
|
68
|
+
}
|
|
69
|
+
if (!/^(\/|\s)/.test(line)) {
|
|
70
|
+
// A line that is neither a route nor a detail (the coverage line) ends the list.
|
|
71
|
+
inProblems = false;
|
|
72
|
+
current = null;
|
|
73
|
+
continue;
|
|
74
|
+
}
|
|
75
|
+
if (/^\s/.test(line)) {
|
|
76
|
+
if (current)
|
|
77
|
+
add(current, line);
|
|
78
|
+
continue;
|
|
79
|
+
}
|
|
80
|
+
const p = PROBLEM_LINE.exec(line);
|
|
81
|
+
current = p && notes.has(p[1]) ? p[1] : null;
|
|
82
|
+
if (current)
|
|
83
|
+
add(current, line);
|
|
84
|
+
}
|
|
85
|
+
return notes;
|
|
86
|
+
}
|
|
87
|
+
/** Whether a crawl had nothing left to visit, so another round would find nothing either. */
|
|
88
|
+
export function crawlFoundNothing(text) {
|
|
89
|
+
return /^(?:\[session [^\]]*\]\s*)?(Nothing to crawl|Nothing new to crawl|No routes to crawl yet)/m.test(text);
|
|
90
|
+
}
|
|
91
|
+
/** The path a target URL opens, as a route: the planning crawl never lists the page the run attached on. */
|
|
92
|
+
function routeOf(url) {
|
|
93
|
+
try {
|
|
94
|
+
const u = new URL(url);
|
|
95
|
+
return `${u.pathname}${u.search}`;
|
|
96
|
+
}
|
|
97
|
+
catch {
|
|
98
|
+
return "/";
|
|
99
|
+
}
|
|
100
|
+
}
|
|
101
|
+
/**
|
|
102
|
+
* Session names for the lanes: the lane's own name, unless another lane or the
|
|
103
|
+
* planner already has it. A session name is at most 40 characters.
|
|
104
|
+
*/
|
|
105
|
+
export function laneSessions(names) {
|
|
106
|
+
const taken = new Set([PLANNER_SESSION]);
|
|
107
|
+
return names.map((name, i) => {
|
|
108
|
+
let session = name;
|
|
109
|
+
for (let k = i + 1; taken.has(session); k += 1)
|
|
110
|
+
session = `${name.slice(0, 40 - `-${k}`.length)}-${k}`;
|
|
111
|
+
taken.add(session);
|
|
112
|
+
return session;
|
|
113
|
+
});
|
|
114
|
+
}
|
|
115
|
+
/**
|
|
116
|
+
* Split what the planning crawl found between `count` lanes, by brief.ts: whole
|
|
117
|
+
* modules (a route's first path segment) per lane, balanced by route count.
|
|
118
|
+
* Fewer lanes come back when there are fewer modules. With fewer than two the
|
|
119
|
+
* run has nothing to split, and explores in one loop, saying why.
|
|
120
|
+
*/
|
|
121
|
+
export function planCiLanes(o) {
|
|
122
|
+
const routes = [...new Set([routeOf(o.target), ...o.notes.keys()])];
|
|
123
|
+
if (o.count < 2)
|
|
124
|
+
return { lanes: [], oneLoop: "one lane was asked for" };
|
|
125
|
+
if (o.notes.size === 0 && o.planningFailed)
|
|
126
|
+
return { lanes: [], oneLoop: `${o.planningFailed}, so there was nothing to split` };
|
|
127
|
+
const briefs = planLanes(routes, o.count, { goal: o.focus, mode: o.mode });
|
|
128
|
+
if (briefs.length < 2) {
|
|
129
|
+
const where = briefs[0]?.modules[0];
|
|
130
|
+
return {
|
|
131
|
+
lanes: [],
|
|
132
|
+
oneLoop: `the planning crawl found ${routes.length} route(s), all in ${where ? `one module (${where})` : "no module"}, so there was nothing to split`,
|
|
133
|
+
};
|
|
134
|
+
}
|
|
135
|
+
const sessions = laneSessions(briefs.map((b) => b.lane));
|
|
136
|
+
return {
|
|
137
|
+
lanes: briefs.map((b, i) => ({
|
|
138
|
+
...b,
|
|
139
|
+
session: sessions[i],
|
|
140
|
+
// On the target's origin whatever the route says: a path written `//host/x` must not become another host.
|
|
141
|
+
url: `${new URL(o.target).origin}${b.landing.startsWith("/") ? "" : "/"}${b.landing}`,
|
|
142
|
+
crawl: b.routes.flatMap((r) => o.notes.get(r) ?? []).slice(0, LIST_MAX),
|
|
143
|
+
})),
|
|
144
|
+
};
|
|
145
|
+
}
|
|
146
|
+
// ── what a lane is told ─────────────────────────────────────────────────────
|
|
147
|
+
/**
|
|
148
|
+
* The system prompt every lane is given: the method, then the rules of an
|
|
149
|
+
* unattended run as one lane of several. The same for every lane of a run, so
|
|
150
|
+
* the provider's prompt cache serves them all; what differs goes in the
|
|
151
|
+
* lane's first message (ciLaneKickoff).
|
|
152
|
+
*/
|
|
153
|
+
export function ciLaneSystemPrompt(playbook, o) {
|
|
154
|
+
return (`${playbook}\n\n---\n\n` +
|
|
155
|
+
`# Running in CI, as one lane of several\n\n` +
|
|
156
|
+
`You are running unattended in a CI job, as one of several lanes exploring the same app at once, each in its own browser session and its own part of the app. There is no person to ask: never ask a question or wait for an answer; decide, and say in findings and notes what you assumed.\n\n` +
|
|
157
|
+
`- Your browser is already attached to the target in ${o.mode} mode, as your lane's own session. scout_attach, scout_close, scout_session, scout_playbook and scout_report are not available: the mode, the target and the session are fixed for this run, and the method is the text above. Skip the Setup steps that choose them, and never pass \`session\`.\n` +
|
|
158
|
+
`- The planner has crawled the routes it found, so route coverage may already read as complete. Your first message lists the routes your lane owns, from that crawl, and what it saw on them; the other lanes own the rest, at the same time. A route under your modules that the crawl did not list is yours too.\n` +
|
|
159
|
+
LANE_RULES.map((rule) => `- ${rule}\n`).join("") +
|
|
160
|
+
`- The level is ${o.level}. The run writes one report for every lane once all are done, and its contract counts what every lane did: do your share of it on your routes, the design audits included.\n` +
|
|
161
|
+
`- Tool results are text only; screenshots are not available. Long results are cut: prefer calls that return less, and snapshots only where you act.\n` +
|
|
162
|
+
`- When your routes are done, or your share of the budget is nearly spent, reply with a short summary and no tool call. That ends your lane; the others go on.\n`);
|
|
163
|
+
}
|
|
164
|
+
const n = (x) => x.toLocaleString("en-US");
|
|
165
|
+
/** A lane's first message: which lane it is, what it owns, where its browser is, what the crawl saw there, and its share of the budget. */
|
|
166
|
+
export function ciLaneKickoff(o) {
|
|
167
|
+
const { lane } = o;
|
|
168
|
+
const share = Math.max(1, Math.floor(o.caps.turns / o.laneCount));
|
|
169
|
+
return [
|
|
170
|
+
`Run an exploratory test session following the method, as lane "${lane.session}", one of ${o.laneCount} running at once.`,
|
|
171
|
+
`Target: ${o.url}`,
|
|
172
|
+
`Your lane: ${lane.objective}`,
|
|
173
|
+
`Your routes (${lane.routes.length}): ${lane.routes.slice(0, LIST_MAX).join(", ")}${lane.routes.length > LIST_MAX ? ` … and ${lane.routes.length - LIST_MAX} more` : ""}`,
|
|
174
|
+
`Your browser is on ${routeOf(lane.url)}.`,
|
|
175
|
+
lane.crawl.length > 0 ? `What the planning crawl saw on your routes:\n${lane.crawl.map((l) => ` ${l.trim()}`).join("\n")}` : "",
|
|
176
|
+
`Project directory (for scout_scan): ${o.projectDir}`,
|
|
177
|
+
`Level: ${o.level}`,
|
|
178
|
+
`Write mode: ${o.mode}`,
|
|
179
|
+
o.focus ? `Focus: ${o.focus}` : "",
|
|
180
|
+
`Budget: the run's ${o.caps.turns} model turns, ${n(o.caps.tokens)} tokens and ${Math.round(o.caps.wallMs / 60_000)} minutes are shared by the ${o.laneCount} lanes: plan on about ${share} turns. Several tool calls in one turn cost one turn. Every lane stops at the first cap the run reaches, and the report is written as it stands.`,
|
|
181
|
+
]
|
|
182
|
+
.filter(Boolean)
|
|
183
|
+
.join("\n");
|
|
184
|
+
}
|
|
185
|
+
// ── how the lanes' endings become the run's ─────────────────────────────────
|
|
186
|
+
/** Caps in the order capReached checks them, so a run ended by two names the one checked first. */
|
|
187
|
+
const CAP_ORDER = ["time", "tokens", "turns"];
|
|
188
|
+
/**
|
|
189
|
+
* What ended a run whose exploration was split into lanes. A model API failure
|
|
190
|
+
* in any lane is the run's, since it is what the workflow must fix, and so is
|
|
191
|
+
* a lane that broke after attaching (could-not-start on an attached lane: the
|
|
192
|
+
* run could not finish, exit 2). Otherwise the run ended at a cap if any lane
|
|
193
|
+
* was stopped by one, a lane the time cap left no time to attach included, and
|
|
194
|
+
* the model finished it only when every lane that ran finished by itself. A
|
|
195
|
+
* lane whose browser could not attach does not change that, but is named in
|
|
196
|
+
* the detail; when none could, the run could not start.
|
|
197
|
+
*/
|
|
198
|
+
export function mergeLaneStops(lanes) {
|
|
199
|
+
const failed = lanes.find((l) => l.attached && l.stop === "provider-error");
|
|
200
|
+
if (failed)
|
|
201
|
+
return { stop: "provider-error", stopDetail: `lane ${failed.session}${failed.stopDetail ? `: ${failed.stopDetail}` : ""}` };
|
|
202
|
+
const broke = lanes.find((l) => l.attached && l.stop === "could-not-start");
|
|
203
|
+
if (broke)
|
|
204
|
+
return { stop: "could-not-start", stopDetail: `lane ${broke.session}: ${broke.stopDetail ?? "it stopped for no reason it gave"}` };
|
|
205
|
+
const unattached = lanes.filter((l) => !l.attached && l.stop === "could-not-start").map((l) => l.session);
|
|
206
|
+
if (unattached.length === lanes.length) {
|
|
207
|
+
const first = lanes.find((l) => l.stopDetail);
|
|
208
|
+
return { stop: "could-not-start", stopDetail: `no lane could attach${first ? ` (${first.session}: ${first.stopDetail})` : ""}` };
|
|
209
|
+
}
|
|
210
|
+
const note = unattached.length > 0 ? { stopDetail: `${unattached.length} of ${lanes.length} lanes could not attach: ${unattached.join(", ")}` } : {};
|
|
211
|
+
for (const cap of CAP_ORDER)
|
|
212
|
+
if (lanes.some((l) => l.stop === cap))
|
|
213
|
+
return { stop: cap, ...note };
|
|
214
|
+
return { stop: "done", ...note };
|
|
215
|
+
}
|
package/dist/engine/ci.js
CHANGED
|
@@ -11,9 +11,11 @@
|
|
|
11
11
|
import { createHash } from "node:crypto";
|
|
12
12
|
import path from "node:path";
|
|
13
13
|
import { BROWSER_ENGINES } from "../browsers.js";
|
|
14
|
+
import { MAX_LANES } from "./brief.js";
|
|
14
15
|
import { markdownCell } from "./check.js";
|
|
15
16
|
import { parseLimitFlag } from "./limits.js";
|
|
16
17
|
import { isWorthALook, redactSecrets } from "./memory.js";
|
|
18
|
+
import { SARIF_ANCHOR_FALLBACKS, checkSarifAnchor, sarifLocation } from "./sarif.js";
|
|
17
19
|
// ── options ─────────────────────────────────────────────────────────────────
|
|
18
20
|
export const PROVIDERS = ["anthropic", "openai"];
|
|
19
21
|
/** Where each provider's key is read from. Nothing else is: no flag, no file, no input. */
|
|
@@ -53,6 +55,14 @@ export const DEDUP_ENV = "SCENESCOUT_DEDUP";
|
|
|
53
55
|
export const DEDUP_PROVIDER_ENV = "SCENESCOUT_DEDUP_PROVIDER";
|
|
54
56
|
export const DEFAULT_CAPS = { turns: 40, tokens: 1_500_000, wallMs: 20 * 60_000 };
|
|
55
57
|
const CAP_BOUNDS = { turns: [1, 500], tokens: [1_000, 20_000_000], minutes: [1, 360] };
|
|
58
|
+
/**
|
|
59
|
+
* How many model loops explore at once, each in its own browser session and
|
|
60
|
+
* its own part of the app (engine/ci-lanes.ts). 1 is the single loop. The most
|
|
61
|
+
* is scout_lane_brief's, since the split is the same one. Why the default is
|
|
62
|
+
* what it is: docs/benchmark.md, "Unattended runs".
|
|
63
|
+
*/
|
|
64
|
+
export const DEFAULT_LANES = 1;
|
|
65
|
+
export const MAX_CI_LANES = MAX_LANES;
|
|
56
66
|
/** Every option `scenescout ci` accepts; the ci action's inputs are these names (ci-test holds them equal). */
|
|
57
67
|
export const CI_OPTION_NAMES = [
|
|
58
68
|
"provider",
|
|
@@ -62,6 +72,7 @@ export const CI_OPTION_NAMES = [
|
|
|
62
72
|
"max-turns",
|
|
63
73
|
"max-tokens",
|
|
64
74
|
"max-minutes",
|
|
75
|
+
"lanes",
|
|
65
76
|
"price-in",
|
|
66
77
|
"price-cached-in",
|
|
67
78
|
"price-out",
|
|
@@ -78,6 +89,7 @@ export const CI_OPTION_NAMES = [
|
|
|
78
89
|
"show",
|
|
79
90
|
"compare-url",
|
|
80
91
|
"dedup",
|
|
92
|
+
"sarif-file-anchor",
|
|
81
93
|
];
|
|
82
94
|
/** The longest --show description: it becomes a line of the model's prompt. */
|
|
83
95
|
export const MAX_SHOW = 200;
|
|
@@ -171,9 +183,13 @@ export function parseCiArgs(args, cwd) {
|
|
|
171
183
|
const turns = whole("max-turns", CAP_BOUNDS.turns, DEFAULT_CAPS.turns);
|
|
172
184
|
const tokens = whole("max-tokens", CAP_BOUNDS.tokens, DEFAULT_CAPS.tokens);
|
|
173
185
|
const minutes = whole("max-minutes", CAP_BOUNDS.minutes, DEFAULT_CAPS.wallMs / 60_000);
|
|
174
|
-
|
|
186
|
+
const lanes = whole("lanes", [1, MAX_CI_LANES], DEFAULT_LANES);
|
|
187
|
+
for (const v of [turns, tokens, minutes, lanes])
|
|
175
188
|
if (typeof v === "string")
|
|
176
189
|
return { ok: false, error: v };
|
|
190
|
+
// The lanes share the run's turns rather than getting a cap each: fewer turns than lanes would leave a lane none.
|
|
191
|
+
if (lanes > turns)
|
|
192
|
+
return { ok: false, error: `--lanes ${lanes} needs --max-turns of at least ${lanes}: the lanes share the run's turns, and each needs one` };
|
|
177
193
|
const price = {};
|
|
178
194
|
for (const [flag, field] of [
|
|
179
195
|
["price-in", "input"],
|
|
@@ -218,6 +234,8 @@ export function parseCiArgs(args, cwd) {
|
|
|
218
234
|
return { ok: false, error: `--show is at most ${MAX_SHOW} characters` };
|
|
219
235
|
if (flags.has("show") && !show)
|
|
220
236
|
return { ok: false, error: '--show needs a few words describing the element, e.g. --show "the Save button"' };
|
|
237
|
+
if (show && lanes > 1)
|
|
238
|
+
return { ok: false, error: "--lanes splits an exploration between model loops, and --show explores nothing: give one or the other" };
|
|
221
239
|
let compareUrl;
|
|
222
240
|
if (flags.has("compare-url")) {
|
|
223
241
|
if (!show)
|
|
@@ -244,6 +262,9 @@ export function parseCiArgs(args, cwd) {
|
|
|
244
262
|
const dedup = flags.get("dedup") ?? DEFAULT_CI_DEDUP;
|
|
245
263
|
if (!DEDUP_MODES.includes(dedup))
|
|
246
264
|
return { ok: false, error: `--dedup must be one of ${DEDUP_MODES.join(", ")}` };
|
|
265
|
+
const anchor = flags.has("sarif-file-anchor") ? checkSarifAnchor(flags.get("sarif-file-anchor")) : undefined;
|
|
266
|
+
if (anchor && !anchor.ok)
|
|
267
|
+
return anchor;
|
|
247
268
|
const resolve = (p) => (p.startsWith("/") || /^[A-Za-z]:[\\/]/.test(p) ? p : `${cwd.replace(/[\\/]$/, "")}/${p}`);
|
|
248
269
|
return {
|
|
249
270
|
ok: true,
|
|
@@ -256,6 +277,7 @@ export function parseCiArgs(args, cwd) {
|
|
|
256
277
|
...(effort ? { effort } : {}),
|
|
257
278
|
...(baseUrl ? { baseUrl } : {}),
|
|
258
279
|
caps: { turns: turns, tokens: tokens, wallMs: minutes * 60_000 },
|
|
280
|
+
lanes: lanes,
|
|
259
281
|
...(Object.keys(price).length > 0 ? { price } : {}),
|
|
260
282
|
mode: mode,
|
|
261
283
|
level: level,
|
|
@@ -267,9 +289,23 @@ export function parseCiArgs(args, cwd) {
|
|
|
267
289
|
...(actionTimeout.value !== undefined ? { actionTimeoutMs: actionTimeout.value } : {}),
|
|
268
290
|
...(navTimeout.value !== undefined ? { navTimeoutMs: navTimeout.value } : {}),
|
|
269
291
|
dedup: dedup,
|
|
292
|
+
...(anchor ? { sarifFileAnchor: anchor.value } : {}),
|
|
270
293
|
},
|
|
271
294
|
};
|
|
272
295
|
}
|
|
296
|
+
/**
|
|
297
|
+
* Why scout_attach did not leave a session the run can use, or null when it
|
|
298
|
+
* did: an error result, or a saved sign-in the app no longer accepts (the
|
|
299
|
+
* attach succeeds, and says so on a line of its own).
|
|
300
|
+
*/
|
|
301
|
+
export function attachFailure(r) {
|
|
302
|
+
const authFailed = r.text.split("\n").find((l) => l.startsWith("⚠ AUTH FAILED"));
|
|
303
|
+
if (authFailed)
|
|
304
|
+
return authFailed;
|
|
305
|
+
if (r.isError || /^ERROR:/.test(r.text))
|
|
306
|
+
return r.text.replace(/^ERROR:\s*/, "");
|
|
307
|
+
return null;
|
|
308
|
+
}
|
|
273
309
|
const present = (env, name) => (env[name] ?? "").trim() !== "";
|
|
274
310
|
/**
|
|
275
311
|
* Which provider a run uses: the one whose key is set. With both set the
|
|
@@ -362,6 +398,12 @@ export function childEnv(env) {
|
|
|
362
398
|
out[k] = v;
|
|
363
399
|
// Nobody watches a CI run: the live view would only hold a port open.
|
|
364
400
|
out.SCENESCOUT_LIVE = "off";
|
|
401
|
+
// ...and an unattended run opens nothing in a browser, even on a desktop, unless the user set it.
|
|
402
|
+
if (!out.SCENESCOUT_OPEN?.trim())
|
|
403
|
+
out.SCENESCOUT_OPEN = "none";
|
|
404
|
+
// The loop is text-only (toolResultText), so a finding's picture is kept for the report, never sent back. A choice the job made is its own.
|
|
405
|
+
if (!out.SCENESCOUT_EVIDENCE?.trim())
|
|
406
|
+
out.SCENESCOUT_EVIDENCE = "file";
|
|
365
407
|
return out;
|
|
366
408
|
}
|
|
367
409
|
// ── never printing a key ────────────────────────────────────────────────────
|
|
@@ -392,7 +434,8 @@ export function addUsage(a, b) {
|
|
|
392
434
|
/**
|
|
393
435
|
* Which cap, if any, stops the run before its next model call. Checked
|
|
394
436
|
* between turns: one turn's usage is only known after it, so a run can end
|
|
395
|
-
* up to one turn over the token cap
|
|
437
|
+
* up to one turn over the token cap (one per lane in a run split into lanes:
|
|
438
|
+
* see Budget), and the report says by how much.
|
|
396
439
|
*/
|
|
397
440
|
export function capReached(spend, caps, now) {
|
|
398
441
|
if (now - spend.startedAt >= caps.wallMs)
|
|
@@ -407,6 +450,31 @@ export function capReached(spend, caps, now) {
|
|
|
407
450
|
export function wallLeftMs(spend, caps, now) {
|
|
408
451
|
return Math.max(0, caps.wallMs - (now - spend.startedAt));
|
|
409
452
|
}
|
|
453
|
+
export function newBudget(caps, startedAt) {
|
|
454
|
+
return { caps, startedAt, turns: 0, usage: { ...NO_USAGE }, inFlight: 0 };
|
|
455
|
+
}
|
|
456
|
+
/** What the budget's loops have spent between them. */
|
|
457
|
+
export function budgetSpend(b) {
|
|
458
|
+
return { turns: b.turns, usage: { ...b.usage }, startedAt: b.startedAt };
|
|
459
|
+
}
|
|
460
|
+
/** Take the next turn for one loop, or name the cap that refuses it. */
|
|
461
|
+
export function takeTurn(b, now) {
|
|
462
|
+
const cap = capReached({ turns: b.turns + b.inFlight, usage: b.usage, startedAt: b.startedAt }, b.caps, now);
|
|
463
|
+
if (cap === null)
|
|
464
|
+
b.inFlight += 1;
|
|
465
|
+
return cap;
|
|
466
|
+
}
|
|
467
|
+
/** A taken turn's call has ended: counted with the usage it reported, or given back when it failed. */
|
|
468
|
+
export function settleTurn(b, usage) {
|
|
469
|
+
// A settle with no turn taken would count a call nobody reserved: a bug in the caller, not a state to carry on from.
|
|
470
|
+
if (b.inFlight <= 0)
|
|
471
|
+
throw new Error("settleTurn without a turn taken: every settle must follow a takeTurn that returned null");
|
|
472
|
+
b.inFlight -= 1;
|
|
473
|
+
if (!usage)
|
|
474
|
+
return;
|
|
475
|
+
b.turns += 1;
|
|
476
|
+
b.usage = addUsage(b.usage, usage);
|
|
477
|
+
}
|
|
410
478
|
export const EXIT_CI = { completed: 0, couldNotRun: 2 };
|
|
411
479
|
/**
|
|
412
480
|
* A CI run reports and never gates, so its findings never set the exit code.
|
|
@@ -486,7 +554,8 @@ export function usageLine(spend, model, endedAt, override) {
|
|
|
486
554
|
* closes itself, so the mode and the target cannot change), scout_session,
|
|
487
555
|
* scout_playbook (the method is the system prompt), scout_screenshot (the
|
|
488
556
|
* loop is text-only), scout_resolve (scout_verify records re-tests), and the
|
|
489
|
-
* lane tools (
|
|
557
|
+
* lane tools (a run split into lanes is planned and folded by the run itself,
|
|
558
|
+
* not by a model: engine/ci-lanes.ts, ADR 20).
|
|
490
559
|
*/
|
|
491
560
|
export const CI_TOOLS = [
|
|
492
561
|
"scout_scan",
|
|
@@ -683,12 +752,15 @@ export function ciSummaryMarkdown(r, secrets = []) {
|
|
|
683
752
|
...(r.capture ? [] : [`| Level | ${r.level} — completion contract ${r.contractMet ? "met" : "not met (the report's gap ledger says what is missing)"} |`]),
|
|
684
753
|
`| Mode | ${r.mode} |`,
|
|
685
754
|
`| Model | ${r.provider} ${cell(r.model, secrets)}, effort ${r.effort} |`,
|
|
755
|
+
...(r.lanes ? [`| Lanes | ${lanesCell(r.lanes, secrets)} |`] : []),
|
|
686
756
|
...(r.dedup ? [`| Finding dedup | ${dedupLine(r.dedup)} |`] : []),
|
|
687
757
|
`| Usage | ${usageLine(r.spend, r.model, r.endedAt, r.price)} |`,
|
|
688
758
|
``,
|
|
689
759
|
];
|
|
690
760
|
if (r.capture)
|
|
691
761
|
lines.push(...captureSummaryLines(r.capture, secrets));
|
|
762
|
+
if (r.lanes && r.lanes.sessions.length > 0)
|
|
763
|
+
lines.push(...laneSummaryLines(r.lanes, secrets));
|
|
692
764
|
if (defects.length > 0) {
|
|
693
765
|
lines.push(`| Severity | Category | Finding | Page |`, `|---|---|---|---|`);
|
|
694
766
|
for (const f of defects.slice(0, 50))
|
|
@@ -707,6 +779,22 @@ export function ciSummaryMarkdown(r, secrets = []) {
|
|
|
707
779
|
lines.push(r.capture ? `The pictures are in shots/.` : `The full report, with repro steps and the gap ledger, is report.md.`, ``);
|
|
708
780
|
return lines.join("\n");
|
|
709
781
|
}
|
|
782
|
+
function lanesCell(l, secrets) {
|
|
783
|
+
if (l.oneLoop)
|
|
784
|
+
return `${l.asked} asked; explored in one loop: ${cell(l.oneLoop, secrets)}`;
|
|
785
|
+
const ran = l.sessions.filter((s) => s.attached).length;
|
|
786
|
+
const planned = l.sessions.length;
|
|
787
|
+
return (`${ran} of ${l.asked} asked ran at once, sharing the caps below` +
|
|
788
|
+
(planned < l.asked ? `; the app split into ${planned}` : "") +
|
|
789
|
+
(ran < planned ? `; ${planned - ran} could not attach` : ""));
|
|
790
|
+
}
|
|
791
|
+
function laneSummaryLines(l, secrets) {
|
|
792
|
+
const out = [`| Lane | Owns | Routes | Turns | Tokens | Ended |`, `|---|---|---:|---:|---:|---|`];
|
|
793
|
+
for (const s of l.sessions)
|
|
794
|
+
out.push(`| ${cell(s.session, secrets)} | ${cell(s.modules.join(", "), secrets)} | ${s.routes} | ${s.turns} | ${n(s.usage.input + s.usage.output)} | ${s.stop}${s.stopDetail ? `: ${cell(s.stopDetail, secrets)}` : ""} |`);
|
|
795
|
+
out.push(``);
|
|
796
|
+
return out;
|
|
797
|
+
}
|
|
710
798
|
function captureSummaryLines(c, secrets) {
|
|
711
799
|
const out = [`**Asked to show:** ${cell(c.what, secrets)}`, ``];
|
|
712
800
|
if (c.status !== "captured")
|
|
@@ -781,6 +869,28 @@ export function ciSummaryJson(r, version, secrets = []) {
|
|
|
781
869
|
...(isWorthALook(f) ? { tier: "worth-a-look", convention: clean(f.convention ?? "") } : {}),
|
|
782
870
|
})),
|
|
783
871
|
...(r.capture ? { capture: cleanCapture(r.capture, clean) } : {}),
|
|
872
|
+
...(r.lanes ? { lanes: lanesJson(r.lanes, clean) } : {}),
|
|
873
|
+
};
|
|
874
|
+
}
|
|
875
|
+
/** The lanes as ci.json holds them. Module paths and lane names come from the app's routes: redacted like the rest. */
|
|
876
|
+
function lanesJson(l, clean) {
|
|
877
|
+
return {
|
|
878
|
+
asked: l.asked,
|
|
879
|
+
planned: l.sessions.length,
|
|
880
|
+
ran: l.sessions.filter((s) => s.attached).length,
|
|
881
|
+
...(l.oneLoop ? { oneLoop: clean(l.oneLoop) } : {}),
|
|
882
|
+
sessions: l.sessions.map((s) => ({
|
|
883
|
+
session: clean(s.session),
|
|
884
|
+
modules: s.modules.map(clean),
|
|
885
|
+
routes: s.routes,
|
|
886
|
+
attached: s.attached,
|
|
887
|
+
stop: s.stop,
|
|
888
|
+
...(s.stopDetail ? { detail: clean(s.stopDetail) } : {}),
|
|
889
|
+
turns: s.turns,
|
|
890
|
+
inputTokens: s.usage.input,
|
|
891
|
+
cachedInputTokens: s.usage.cachedInput,
|
|
892
|
+
outputTokens: s.usage.output,
|
|
893
|
+
})),
|
|
784
894
|
};
|
|
785
895
|
}
|
|
786
896
|
/** The capture as ci.json holds it: the words a person or a model wrote, and the URLs, passed through the same redaction as the rest. */
|
|
@@ -802,8 +912,13 @@ function appOrigin(url) {
|
|
|
802
912
|
return url;
|
|
803
913
|
}
|
|
804
914
|
}
|
|
805
|
-
/**
|
|
806
|
-
|
|
915
|
+
/**
|
|
916
|
+
* This run's findings as SARIF 2.1.0: one rule per category, the finding's id
|
|
917
|
+
* as its fingerprint. Code scanning keeps only results located in a repository
|
|
918
|
+
* file, so each points at the anchor (sarif.ts) and carries its page as a
|
|
919
|
+
* logical location, in `properties.route` and in its message.
|
|
920
|
+
*/
|
|
921
|
+
export function ciSarif(r, version, secrets = [], anchor = SARIF_ANCHOR_FALLBACKS[0]) {
|
|
807
922
|
const clean = (s) => redactKeys(redactSecrets(s), secrets);
|
|
808
923
|
const categories = [...new Set(r.findings.map((f) => f.category))].sort();
|
|
809
924
|
return {
|
|
@@ -820,19 +935,22 @@ export function ciSarif(r, version, secrets = []) {
|
|
|
820
935
|
},
|
|
821
936
|
},
|
|
822
937
|
invocations: [{ executionSuccessful: r.stop !== "provider-error" && r.stop !== "could-not-start", properties: { stop: r.stop } }],
|
|
823
|
-
|
|
824
|
-
results: r.findings.map((f) =>
|
|
825
|
-
|
|
826
|
-
|
|
827
|
-
|
|
828
|
-
|
|
829
|
-
|
|
830
|
-
:
|
|
831
|
-
|
|
832
|
-
|
|
833
|
-
|
|
834
|
-
|
|
835
|
-
|
|
938
|
+
properties: { app: clean(appOrigin(r.url)) },
|
|
939
|
+
results: r.findings.map((f) => {
|
|
940
|
+
const route = clean(pathOf(f.url));
|
|
941
|
+
return {
|
|
942
|
+
ruleId: `finding/${f.category}`,
|
|
943
|
+
level: isWorthALook(f) ? "note" : SARIF_LEVEL[f.severity],
|
|
944
|
+
message: {
|
|
945
|
+
text: isWorthALook(f)
|
|
946
|
+
? `Worth a look — ${clean(f.title)} — on ${route}. A defect only if your project uses ${clean(f.convention ?? "a convention")}.`
|
|
947
|
+
: `[${f.severity}] ${clean(f.title)}${f.evidence ? ` — ${clean(f.evidence)}` : ""} — on ${route}`,
|
|
948
|
+
},
|
|
949
|
+
locations: [sarifLocation(anchor, route)],
|
|
950
|
+
partialFingerprints: { "scenescoutFinding/v1": createHash("sha256").update(f.id).digest("hex").slice(0, 32) },
|
|
951
|
+
properties: { route, ...(isWorthALook(f) ? { tier: "worth-a-look", convention: clean(f.convention ?? "") } : {}) },
|
|
952
|
+
};
|
|
953
|
+
}),
|
|
836
954
|
},
|
|
837
955
|
],
|
|
838
956
|
};
|