scenescout 3.11.1 → 3.13.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +31 -0
- package/README.md +29 -3
- package/dist/check-run.js +4 -2
- package/dist/ci-run.js +366 -0
- package/dist/cli.js +69 -0
- package/dist/engine/bench.js +147 -14
- package/dist/engine/browser.js +63 -34
- package/dist/engine/calibration.js +23 -6
- package/dist/engine/check.js +89 -15
- package/dist/engine/ci.js +633 -0
- package/dist/engine/dedup.js +237 -0
- package/dist/engine/design.js +15 -1
- package/dist/engine/fingerprint.js +57 -0
- package/dist/engine/lane.js +57 -3
- package/dist/engine/memory.js +115 -8
- package/dist/engine/policy.js +42 -0
- package/dist/engine/provider.js +193 -0
- package/dist/engine/report.js +37 -2
- package/dist/engine/verify.js +4 -3
- package/dist/mcp-server.js +23 -9
- package/package.json +6 -3
- package/skills/scenescout/SKILL.md +4 -3
|
@@ -0,0 +1,633 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* `scenescout ci`: an exploratory run with no person and no coding agent. A
|
|
3
|
+
* model reached over an HTTP API drives the scout_* tools by the SceneScout
|
|
4
|
+
* method, and the run ends in the ordinary report. This file holds its rules,
|
|
5
|
+
* none of which needs a browser or a network: the options it accepts, which
|
|
6
|
+
* provider a run uses, the caps that end it, what may never be printed, which
|
|
7
|
+
* tools the model is given and how their results are shaped, and the files a
|
|
8
|
+
* run writes. The loop that uses them is src/ci-run.ts; the two APIs' message
|
|
9
|
+
* shapes are engine/provider.ts. Why it reports and never gates: ADR 14.
|
|
10
|
+
*/
|
|
11
|
+
import { createHash } from "node:crypto";
|
|
12
|
+
import path from "node:path";
|
|
13
|
+
import { BROWSER_ENGINES } from "../browsers.js";
|
|
14
|
+
import { markdownCell } from "./check.js";
|
|
15
|
+
import { isWorthALook, redactSecrets } from "./memory.js";
|
|
16
|
+
// ── options ─────────────────────────────────────────────────────────────────
|
|
17
|
+
export const PROVIDERS = ["anthropic", "openai"];
|
|
18
|
+
/** Where each provider's key is read from. Nothing else is: no flag, no file, no input. */
|
|
19
|
+
export const KEY_ENV = { anthropic: "ANTHROPIC_API_KEY", openai: "OPENAI_API_KEY" };
|
|
20
|
+
export const DEFAULT_MODEL = { anthropic: "claude-sonnet-5", openai: "gpt-6-luna" };
|
|
21
|
+
/** The API root each request path is joined to. `--base-url` replaces it, for a compatible endpoint. */
|
|
22
|
+
export const DEFAULT_BASE_URL = { anthropic: "https://api.anthropic.com/v1", openai: "https://api.openai.com/v1" };
|
|
23
|
+
/** Reasoning effort each API accepts. The Messages API has no "none": thinking is lowered, not switched off. */
|
|
24
|
+
export const EFFORTS = {
|
|
25
|
+
anthropic: ["low", "medium", "high", "xhigh", "max"],
|
|
26
|
+
openai: ["none", "low", "medium", "high", "xhigh", "max"],
|
|
27
|
+
};
|
|
28
|
+
export const DEFAULT_EFFORT = "low";
|
|
29
|
+
/**
|
|
30
|
+
* The write policies a CI run may use. `destructive` lets the model send any
|
|
31
|
+
* request, deletes of records it did not create included, so it is taken only
|
|
32
|
+
* with `--allow-destructive` as well: a mode value alone, copied from another
|
|
33
|
+
* workflow or typed by an agent, never enables it. What CI may do is the
|
|
34
|
+
* developer's decision; the default stays read-only (ADR 14).
|
|
35
|
+
*/
|
|
36
|
+
export const CI_MODES = ["observe", "read-only", "safe-write", "destructive"];
|
|
37
|
+
export const CI_LEVELS = ["minimal", "medium", "extensive"];
|
|
38
|
+
export const DEFAULT_CAPS = { turns: 40, tokens: 1_500_000, wallMs: 20 * 60_000 };
|
|
39
|
+
const CAP_BOUNDS = { turns: [1, 500], tokens: [1_000, 20_000_000], minutes: [1, 360] };
|
|
40
|
+
/** Every option `scenescout ci` accepts; the ci action's inputs are these names (ci-test holds them equal). */
|
|
41
|
+
export const CI_OPTION_NAMES = [
|
|
42
|
+
"provider",
|
|
43
|
+
"model",
|
|
44
|
+
"effort",
|
|
45
|
+
"base-url",
|
|
46
|
+
"max-turns",
|
|
47
|
+
"max-tokens",
|
|
48
|
+
"max-minutes",
|
|
49
|
+
"price-in",
|
|
50
|
+
"price-cached-in",
|
|
51
|
+
"price-out",
|
|
52
|
+
"mode",
|
|
53
|
+
"allow-destructive",
|
|
54
|
+
"level",
|
|
55
|
+
"focus",
|
|
56
|
+
"storage-state",
|
|
57
|
+
"browser",
|
|
58
|
+
"project",
|
|
59
|
+
"out",
|
|
60
|
+
];
|
|
61
|
+
const LOOPBACK_HOSTS = new Set(["127.0.0.1", "localhost", "[::1]", "::1"]);
|
|
62
|
+
/** A base URL a key may be sent to: https, or plain http only to this machine. */
|
|
63
|
+
export function checkBaseUrl(raw) {
|
|
64
|
+
let u;
|
|
65
|
+
try {
|
|
66
|
+
u = new URL(raw);
|
|
67
|
+
}
|
|
68
|
+
catch {
|
|
69
|
+
return { ok: false, error: `--base-url is not a URL: ${raw}` };
|
|
70
|
+
}
|
|
71
|
+
if (u.username || u.password)
|
|
72
|
+
return { ok: false, error: "--base-url must carry no credentials: the key is read from the environment" };
|
|
73
|
+
if (u.protocol === "https:")
|
|
74
|
+
return { ok: true, url: u.toString().replace(/\/+$/, "") };
|
|
75
|
+
if (u.protocol === "http:" && LOOPBACK_HOSTS.has(u.hostname))
|
|
76
|
+
return { ok: true, url: u.toString().replace(/\/+$/, "") };
|
|
77
|
+
return { ok: false, error: "--base-url must be https (plain http only to 127.0.0.1 or localhost): the API key is sent to it" };
|
|
78
|
+
}
|
|
79
|
+
export function parseCiArgs(args, cwd) {
|
|
80
|
+
const positional = [];
|
|
81
|
+
const flags = new Map();
|
|
82
|
+
for (let i = 0; i < args.length; i++) {
|
|
83
|
+
const a = args[i];
|
|
84
|
+
if (!a.startsWith("--")) {
|
|
85
|
+
positional.push(a);
|
|
86
|
+
continue;
|
|
87
|
+
}
|
|
88
|
+
const eq = a.indexOf("=");
|
|
89
|
+
const name = eq > 0 ? a.slice(2, eq) : a.slice(2);
|
|
90
|
+
// The one switch: given bare, it takes no value, so the URL after it stays the URL.
|
|
91
|
+
if (name === "allow-destructive" && eq < 0) {
|
|
92
|
+
flags.set(name, "true");
|
|
93
|
+
continue;
|
|
94
|
+
}
|
|
95
|
+
const value = eq > 0 ? a.slice(eq + 1) : args[i + 1];
|
|
96
|
+
if (value === undefined || (eq < 0 && value.startsWith("--")))
|
|
97
|
+
return { ok: false, error: `--${name} needs a value` };
|
|
98
|
+
if (eq < 0)
|
|
99
|
+
i += 1;
|
|
100
|
+
flags.set(name, value);
|
|
101
|
+
}
|
|
102
|
+
const known = new Set(CI_OPTION_NAMES);
|
|
103
|
+
for (const name of flags.keys()) {
|
|
104
|
+
if (name === "api-key" || name === "key")
|
|
105
|
+
return { ok: false, error: `there is no --${name}: the key is read from ANTHROPIC_API_KEY or OPENAI_API_KEY only` };
|
|
106
|
+
if (!known.has(name))
|
|
107
|
+
return { ok: false, error: `unknown option --${name}` };
|
|
108
|
+
}
|
|
109
|
+
if (positional.length !== 1)
|
|
110
|
+
return { ok: false, error: "give exactly one URL to explore, e.g. scenescout ci http://127.0.0.1:3000" };
|
|
111
|
+
let url;
|
|
112
|
+
try {
|
|
113
|
+
url = new URL(positional[0]);
|
|
114
|
+
}
|
|
115
|
+
catch {
|
|
116
|
+
return { ok: false, error: `not a URL: ${positional[0]}` };
|
|
117
|
+
}
|
|
118
|
+
if (url.protocol !== "http:" && url.protocol !== "https:")
|
|
119
|
+
return { ok: false, error: `only http and https URLs can be explored (got ${url.protocol})` };
|
|
120
|
+
if (url.username || url.password)
|
|
121
|
+
return { ok: false, error: "put no credentials in the URL: they would be written into the report. Sign in with --storage-state instead" };
|
|
122
|
+
const provider = flags.get("provider");
|
|
123
|
+
if (provider !== undefined && !PROVIDERS.includes(provider))
|
|
124
|
+
return { ok: false, error: `--provider must be one of ${PROVIDERS.join(", ")}` };
|
|
125
|
+
const effort = flags.get("effort");
|
|
126
|
+
if (effort !== undefined) {
|
|
127
|
+
const allowed = provider ? EFFORTS[provider] : EFFORTS.openai;
|
|
128
|
+
if (!allowed.includes(effort))
|
|
129
|
+
return { ok: false, error: `--effort must be one of ${allowed.join(", ")}${provider ? ` for ${provider}` : ""}` };
|
|
130
|
+
}
|
|
131
|
+
const model = flags.get("model");
|
|
132
|
+
if (model !== undefined && !/^[A-Za-z0-9._:/@-]{1,120}$/.test(model))
|
|
133
|
+
return { ok: false, error: "--model is not a model id" };
|
|
134
|
+
let baseUrl;
|
|
135
|
+
if (flags.has("base-url")) {
|
|
136
|
+
const b = checkBaseUrl(flags.get("base-url"));
|
|
137
|
+
if (!b.ok)
|
|
138
|
+
return b;
|
|
139
|
+
baseUrl = b.url;
|
|
140
|
+
}
|
|
141
|
+
const whole = (name, [lo, hi], fallback) => {
|
|
142
|
+
const raw = flags.get(name);
|
|
143
|
+
if (raw === undefined)
|
|
144
|
+
return fallback;
|
|
145
|
+
const n = Number(raw);
|
|
146
|
+
return Number.isInteger(n) && n >= lo && n <= hi ? n : `--${name} must be a whole number from ${lo} to ${hi}`;
|
|
147
|
+
};
|
|
148
|
+
const turns = whole("max-turns", CAP_BOUNDS.turns, DEFAULT_CAPS.turns);
|
|
149
|
+
const tokens = whole("max-tokens", CAP_BOUNDS.tokens, DEFAULT_CAPS.tokens);
|
|
150
|
+
const minutes = whole("max-minutes", CAP_BOUNDS.minutes, DEFAULT_CAPS.wallMs / 60_000);
|
|
151
|
+
for (const v of [turns, tokens, minutes])
|
|
152
|
+
if (typeof v === "string")
|
|
153
|
+
return { ok: false, error: v };
|
|
154
|
+
const price = {};
|
|
155
|
+
for (const [flag, field] of [
|
|
156
|
+
["price-in", "input"],
|
|
157
|
+
["price-cached-in", "cachedInput"],
|
|
158
|
+
["price-out", "output"],
|
|
159
|
+
]) {
|
|
160
|
+
const raw = flags.get(flag);
|
|
161
|
+
if (raw === undefined)
|
|
162
|
+
continue;
|
|
163
|
+
const v = Number(raw);
|
|
164
|
+
if (raw.trim() === "" || !Number.isFinite(v) || v < 0 || v > 1000)
|
|
165
|
+
return { ok: false, error: `--${flag} must be US dollars per million tokens, from 0 to 1000` };
|
|
166
|
+
price[field] = v;
|
|
167
|
+
}
|
|
168
|
+
const mode = flags.get("mode") ?? "read-only";
|
|
169
|
+
if (!CI_MODES.includes(mode))
|
|
170
|
+
return { ok: false, error: `--mode must be one of ${CI_MODES.join(", ")}` };
|
|
171
|
+
const allow = flags.get("allow-destructive") ?? "false";
|
|
172
|
+
if (allow !== "true" && allow !== "false")
|
|
173
|
+
return { ok: false, error: "--allow-destructive takes no value, or true or false" };
|
|
174
|
+
if (mode === "destructive" && allow !== "true")
|
|
175
|
+
return {
|
|
176
|
+
ok: false,
|
|
177
|
+
error: "--mode destructive needs --allow-destructive as well: it lets the run send any request, deleting records it did not create included",
|
|
178
|
+
};
|
|
179
|
+
const level = flags.get("level") ?? "medium";
|
|
180
|
+
if (!CI_LEVELS.includes(level))
|
|
181
|
+
return { ok: false, error: `--level must be one of ${CI_LEVELS.join(", ")}` };
|
|
182
|
+
const focus = flags.get("focus")?.trim();
|
|
183
|
+
if (focus !== undefined && focus.length > 300)
|
|
184
|
+
return { ok: false, error: "--focus is at most 300 characters" };
|
|
185
|
+
const browser = flags.get("browser");
|
|
186
|
+
if (browser !== undefined && !BROWSER_ENGINES.includes(browser))
|
|
187
|
+
return { ok: false, error: `--browser must be one of ${BROWSER_ENGINES.join(", ")}` };
|
|
188
|
+
const resolve = (p) => (p.startsWith("/") || /^[A-Za-z]:[\\/]/.test(p) ? p : `${cwd.replace(/[\\/]$/, "")}/${p}`);
|
|
189
|
+
return {
|
|
190
|
+
ok: true,
|
|
191
|
+
options: {
|
|
192
|
+
url: url.toString(),
|
|
193
|
+
projectDir: resolve(flags.get("project") ?? cwd),
|
|
194
|
+
...(flags.has("out") ? { outDir: resolve(flags.get("out")) } : {}),
|
|
195
|
+
...(provider ? { provider: provider } : {}),
|
|
196
|
+
...(model ? { model } : {}),
|
|
197
|
+
...(effort ? { effort } : {}),
|
|
198
|
+
...(baseUrl ? { baseUrl } : {}),
|
|
199
|
+
caps: { turns: turns, tokens: tokens, wallMs: minutes * 60_000 },
|
|
200
|
+
...(Object.keys(price).length > 0 ? { price } : {}),
|
|
201
|
+
mode: mode,
|
|
202
|
+
level: level,
|
|
203
|
+
...(focus ? { focus } : {}),
|
|
204
|
+
...(flags.has("storage-state") ? { storageStatePath: resolve(flags.get("storage-state")) } : {}),
|
|
205
|
+
...(browser ? { browser: browser } : {}),
|
|
206
|
+
},
|
|
207
|
+
};
|
|
208
|
+
}
|
|
209
|
+
const present = (env, name) => (env[name] ?? "").trim() !== "";
|
|
210
|
+
/**
|
|
211
|
+
* Which provider a run uses: the one whose key is set. With both set the
|
|
212
|
+
* choice is not guessed: --provider must name it. A --provider whose key is
|
|
213
|
+
* missing is refused before anything starts, naming the variable.
|
|
214
|
+
*/
|
|
215
|
+
export function detectProvider(env, options) {
|
|
216
|
+
const withKey = PROVIDERS.filter((p) => present(env, KEY_ENV[p]));
|
|
217
|
+
let provider;
|
|
218
|
+
if (options.provider) {
|
|
219
|
+
if (!withKey.includes(options.provider))
|
|
220
|
+
return { ok: false, error: `--provider ${options.provider} needs ${KEY_ENV[options.provider]} set in the environment` };
|
|
221
|
+
provider = options.provider;
|
|
222
|
+
}
|
|
223
|
+
else if (withKey.length === 0) {
|
|
224
|
+
return { ok: false, error: `set ${KEY_ENV.anthropic} or ${KEY_ENV.openai} in the environment: a CI run needs a model to drive it` };
|
|
225
|
+
}
|
|
226
|
+
else if (withKey.length > 1) {
|
|
227
|
+
return { ok: false, error: `both ${KEY_ENV.anthropic} and ${KEY_ENV.openai} are set: pass --provider anthropic or --provider openai to choose` };
|
|
228
|
+
}
|
|
229
|
+
else {
|
|
230
|
+
provider = withKey[0];
|
|
231
|
+
}
|
|
232
|
+
const effort = options.effort ?? DEFAULT_EFFORT;
|
|
233
|
+
if (!EFFORTS[provider].includes(effort))
|
|
234
|
+
return { ok: false, error: `--effort must be one of ${EFFORTS[provider].join(", ")} for ${provider}` };
|
|
235
|
+
return { ok: true, resolved: { provider, model: options.model ?? DEFAULT_MODEL[provider], effort, baseUrl: options.baseUrl ?? DEFAULT_BASE_URL[provider] } };
|
|
236
|
+
}
|
|
237
|
+
/**
|
|
238
|
+
* The environment the MCP server and its browser are started with: this
|
|
239
|
+
* process's, without the keys. Nothing on that side needs them, and a key a
|
|
240
|
+
* process never had cannot end up in its logs or its report.
|
|
241
|
+
*/
|
|
242
|
+
export function childEnv(env) {
|
|
243
|
+
const out = {};
|
|
244
|
+
const drop = new Set(Object.values(KEY_ENV));
|
|
245
|
+
for (const [k, v] of Object.entries(env))
|
|
246
|
+
if (v !== undefined && !drop.has(k))
|
|
247
|
+
out[k] = v;
|
|
248
|
+
// Nobody watches a CI run: the live view would only hold a port open.
|
|
249
|
+
out.SCENESCOUT_LIVE = "off";
|
|
250
|
+
return out;
|
|
251
|
+
}
|
|
252
|
+
// ── never printing a key ────────────────────────────────────────────────────
|
|
253
|
+
export const REDACTED_KEY = "[redacted key]";
|
|
254
|
+
/** The key values set in this environment, to be removed from anything printed or written. */
|
|
255
|
+
export function secretValues(env) {
|
|
256
|
+
return Object.values(KEY_ENV)
|
|
257
|
+
.map((name) => (env[name] ?? "").trim())
|
|
258
|
+
.filter((v) => v.length >= 8);
|
|
259
|
+
}
|
|
260
|
+
/** Key-shaped strings, for a key that reached a message from somewhere other than this environment (an echo of another). */
|
|
261
|
+
const KEY_SHAPES = /\bsk-[A-Za-z0-9_-]{16,}/g;
|
|
262
|
+
/**
|
|
263
|
+
* Remove every key value, and anything shaped like a provider key, from a
|
|
264
|
+
* line before it is logged or written. Longest first, so a key that contains
|
|
265
|
+
* another is not left half-printed.
|
|
266
|
+
*/
|
|
267
|
+
export function redactKeys(text, secrets) {
|
|
268
|
+
let out = text;
|
|
269
|
+
for (const s of [...secrets].sort((a, b) => b.length - a.length))
|
|
270
|
+
out = out.split(s).join(REDACTED_KEY);
|
|
271
|
+
return out.replace(KEY_SHAPES, REDACTED_KEY);
|
|
272
|
+
}
|
|
273
|
+
export const NO_USAGE = { input: 0, cachedInput: 0, cacheWrite: 0, output: 0 };
|
|
274
|
+
export function addUsage(a, b) {
|
|
275
|
+
return { input: a.input + b.input, cachedInput: a.cachedInput + b.cachedInput, cacheWrite: a.cacheWrite + b.cacheWrite, output: a.output + b.output };
|
|
276
|
+
}
|
|
277
|
+
/**
|
|
278
|
+
* Which cap, if any, stops the run before its next model call. Checked
|
|
279
|
+
* between turns: one turn's usage is only known after it, so a run can end
|
|
280
|
+
* up to one turn over the token cap, and the report says by how much.
|
|
281
|
+
*/
|
|
282
|
+
export function capReached(spend, caps, now) {
|
|
283
|
+
if (now - spend.startedAt >= caps.wallMs)
|
|
284
|
+
return "time";
|
|
285
|
+
if (spend.usage.input + spend.usage.output >= caps.tokens)
|
|
286
|
+
return "tokens";
|
|
287
|
+
if (spend.turns >= caps.turns)
|
|
288
|
+
return "turns";
|
|
289
|
+
return null;
|
|
290
|
+
}
|
|
291
|
+
/** What is left of the time cap. No model or tool call may run longer. */
|
|
292
|
+
export function wallLeftMs(spend, caps, now) {
|
|
293
|
+
return Math.max(0, caps.wallMs - (now - spend.startedAt));
|
|
294
|
+
}
|
|
295
|
+
export const EXIT_CI = { completed: 0, couldNotRun: 2 };
|
|
296
|
+
/**
|
|
297
|
+
* A CI run reports and never gates, so its findings never set the exit code.
|
|
298
|
+
* 0: the run ran, whether the model finished or a cap ended it, and the report
|
|
299
|
+
* is written. 2: it could not run, or could not finish for a reason the
|
|
300
|
+
* workflow must fix (a key the provider refused, an app that never answered).
|
|
301
|
+
*/
|
|
302
|
+
export function ciExitCode(reason, reportWritten) {
|
|
303
|
+
if (!reportWritten)
|
|
304
|
+
return EXIT_CI.couldNotRun;
|
|
305
|
+
return reason === "provider-error" || reason === "could-not-start" ? EXIT_CI.couldNotRun : EXIT_CI.completed;
|
|
306
|
+
}
|
|
307
|
+
export function describeStop(reason, caps, detail) {
|
|
308
|
+
switch (reason) {
|
|
309
|
+
case "done":
|
|
310
|
+
return "the model finished the run";
|
|
311
|
+
case "turns":
|
|
312
|
+
return `stopped at the turn cap (${caps.turns} model calls)`;
|
|
313
|
+
case "tokens":
|
|
314
|
+
return `stopped at the token cap (${caps.tokens.toLocaleString("en-US")} tokens)`;
|
|
315
|
+
case "time":
|
|
316
|
+
return `stopped at the time cap (${Math.round(caps.wallMs / 60_000)} minutes)`;
|
|
317
|
+
case "provider-error":
|
|
318
|
+
return `stopped: the model's API failed${detail ? ` (${detail})` : ""}`;
|
|
319
|
+
case "could-not-start":
|
|
320
|
+
return `could not start${detail ? `: ${detail}` : ""}`;
|
|
321
|
+
}
|
|
322
|
+
}
|
|
323
|
+
export const PRICES = {
|
|
324
|
+
"gpt-6-luna": { input: 0.1, cachedInput: 0.01, output: 0.5 },
|
|
325
|
+
"gpt-5.6-luna": { input: 0.2, cachedInput: 0.02, output: 1.2 },
|
|
326
|
+
"claude-sonnet-5": { input: 2, cachedInput: 0.2, cacheWrite: 2.5, output: 10 },
|
|
327
|
+
};
|
|
328
|
+
/**
|
|
329
|
+
* The price a run is costed at: the table's entry for the model with any
|
|
330
|
+
* given price put over it. Null when input or output is still unknown, so a
|
|
331
|
+
* model nobody priced is never costed at a guess. Cached input given no price
|
|
332
|
+
* of its own is charged as input, which can only overstate the cost.
|
|
333
|
+
*/
|
|
334
|
+
export function resolvePrice(model, override = {}) {
|
|
335
|
+
const base = PRICES[model] ?? {};
|
|
336
|
+
const merged = { ...base, ...override };
|
|
337
|
+
// A cache-write price belongs to the table's input price; an overridden input price replaces it.
|
|
338
|
+
if (override.input !== undefined)
|
|
339
|
+
delete merged.cacheWrite;
|
|
340
|
+
if (merged.input === undefined || merged.output === undefined)
|
|
341
|
+
return null;
|
|
342
|
+
return {
|
|
343
|
+
input: merged.input,
|
|
344
|
+
cachedInput: merged.cachedInput ?? merged.input,
|
|
345
|
+
output: merged.output,
|
|
346
|
+
...(merged.cacheWrite !== undefined ? { cacheWrite: merged.cacheWrite } : {}),
|
|
347
|
+
};
|
|
348
|
+
}
|
|
349
|
+
/** Estimated cost in dollars, or null when the model's price is not known here or given. */
|
|
350
|
+
export function estimateCost(model, u, override) {
|
|
351
|
+
const p = resolvePrice(model, override);
|
|
352
|
+
if (!p)
|
|
353
|
+
return null;
|
|
354
|
+
const plain = Math.max(0, u.input - u.cachedInput - u.cacheWrite);
|
|
355
|
+
return (plain * p.input + u.cachedInput * p.cachedInput + u.cacheWrite * (p.cacheWrite ?? p.input) + u.output * p.output) / 1_000_000;
|
|
356
|
+
}
|
|
357
|
+
const n = (x) => x.toLocaleString("en-US");
|
|
358
|
+
export function usageLine(spend, model, endedAt, override) {
|
|
359
|
+
const u = spend.usage;
|
|
360
|
+
const secs = Math.round((endedAt - spend.startedAt) / 1000);
|
|
361
|
+
const cost = estimateCost(model, u, override);
|
|
362
|
+
return (`${spend.turns} turn(s), ${n(u.input)} tokens in (${n(u.cachedInput)} cached), ${n(u.output)} out, ` +
|
|
363
|
+
`${Math.floor(secs / 60)}m ${secs % 60}s` +
|
|
364
|
+
(cost === null ? `, cost not estimated (no price known for ${model}; --price-in and --price-out give one)` : `, estimated cost $${cost.toFixed(4)}`));
|
|
365
|
+
}
|
|
366
|
+
// ── the tools the model is given ────────────────────────────────────────────
|
|
367
|
+
/**
|
|
368
|
+
* The scout_* tools a CI model may call. An allowlist, so a tool added to the
|
|
369
|
+
* server later is not handed to an unattended model until someone decides it
|
|
370
|
+
* should be. Left out: scout_attach and scout_close (the run attaches and
|
|
371
|
+
* closes itself, so the mode and the target cannot change), scout_session,
|
|
372
|
+
* scout_playbook (the method is the system prompt), scout_screenshot (the
|
|
373
|
+
* loop is text-only), scout_resolve (scout_verify records re-tests), and the
|
|
374
|
+
* lane tools (one agent, ADR 14).
|
|
375
|
+
*/
|
|
376
|
+
export const CI_TOOLS = [
|
|
377
|
+
"scout_scan",
|
|
378
|
+
"scout_note",
|
|
379
|
+
"scout_crawl",
|
|
380
|
+
"scout_snapshot",
|
|
381
|
+
"scout_navigate",
|
|
382
|
+
"scout_back",
|
|
383
|
+
"scout_click",
|
|
384
|
+
"scout_type",
|
|
385
|
+
"scout_select",
|
|
386
|
+
"scout_press",
|
|
387
|
+
"scout_hover",
|
|
388
|
+
"scout_scroll",
|
|
389
|
+
"scout_upload",
|
|
390
|
+
"scout_run_plan",
|
|
391
|
+
"scout_journey",
|
|
392
|
+
"scout_request",
|
|
393
|
+
"scout_design_audit",
|
|
394
|
+
"scout_coverage",
|
|
395
|
+
"scout_finding",
|
|
396
|
+
"scout_verify",
|
|
397
|
+
"scout_report",
|
|
398
|
+
];
|
|
399
|
+
/**
|
|
400
|
+
* The tools as the model sees them: only the allowed ones, in a fixed order (a
|
|
401
|
+
* changing order would defeat prompt caching), each schema without `session`
|
|
402
|
+
* (one session, always the run's) or `$schema`.
|
|
403
|
+
*/
|
|
404
|
+
export function ciTools(listed) {
|
|
405
|
+
const byName = new Map(listed.map((t) => [t.name, t]));
|
|
406
|
+
return CI_TOOLS.filter((name) => byName.has(name)).map((name) => {
|
|
407
|
+
const t = byName.get(name);
|
|
408
|
+
const schema = { ...(t.inputSchema ?? {}) };
|
|
409
|
+
delete schema.$schema;
|
|
410
|
+
const properties = { ...(schema.properties ?? {}) };
|
|
411
|
+
delete properties.session;
|
|
412
|
+
const required = Array.isArray(schema.required) ? schema.required.filter((r) => r !== "session") : undefined;
|
|
413
|
+
const parameters = { ...schema, type: "object", properties };
|
|
414
|
+
if (required && required.length > 0)
|
|
415
|
+
parameters.required = required;
|
|
416
|
+
else
|
|
417
|
+
delete parameters.required;
|
|
418
|
+
return { name, description: t.description ?? "", parameters };
|
|
419
|
+
});
|
|
420
|
+
}
|
|
421
|
+
/** Arguments the model sent, as the server will get them: an object, never naming a session. */
|
|
422
|
+
export function ciToolArgs(input) {
|
|
423
|
+
if (input === undefined || input === null)
|
|
424
|
+
return { ok: true, args: {} };
|
|
425
|
+
if (typeof input !== "object" || Array.isArray(input))
|
|
426
|
+
return { ok: false, error: "the arguments must be a JSON object" };
|
|
427
|
+
const args = { ...input };
|
|
428
|
+
delete args.session;
|
|
429
|
+
return { ok: true, args };
|
|
430
|
+
}
|
|
431
|
+
/**
|
|
432
|
+
* Arguments a tool call may not carry in a CI run. scout_scan reads a
|
|
433
|
+
* directory's files; the model may scan the run's own project and nothing
|
|
434
|
+
* else, and is given that directory when it names none.
|
|
435
|
+
*/
|
|
436
|
+
export function guardToolArgs(name, args, projectDir) {
|
|
437
|
+
if (name !== "scout_scan")
|
|
438
|
+
return { ok: true, args };
|
|
439
|
+
const given = args.projectPath;
|
|
440
|
+
if (given === undefined)
|
|
441
|
+
return { ok: true, args: { ...args, projectPath: path.resolve(projectDir) } };
|
|
442
|
+
if (typeof given !== "string" || path.resolve(projectDir, given) !== path.resolve(projectDir))
|
|
443
|
+
return { ok: false, error: `scout_scan may scan only this run's project directory, ${projectDir}` };
|
|
444
|
+
return { ok: true, args: { ...args, projectPath: path.resolve(projectDir) } };
|
|
445
|
+
}
|
|
446
|
+
/** The longest tool result handed back to the model, in characters. The report keeps everything; this only bounds the context. */
|
|
447
|
+
export const TOOL_RESULT_MAX_CHARS = 16_000;
|
|
448
|
+
/** One tool result as text: images named rather than sent, and the length bounded. */
|
|
449
|
+
export function toolResultText(content, max = TOOL_RESULT_MAX_CHARS) {
|
|
450
|
+
const parts = (content ?? []).map((c) => (c.type === "text" ? (c.text ?? "") : `[${c.type} omitted: this run is text-only]`));
|
|
451
|
+
const text = parts.join("\n") || "(no output)";
|
|
452
|
+
return text.length <= max ? text : `${text.slice(0, max)}\n… [${n(text.length - max)} more characters cut to keep the context bounded]`;
|
|
453
|
+
}
|
|
454
|
+
// ── what the model is told ──────────────────────────────────────────────────
|
|
455
|
+
export function ciSystemPrompt(playbook, o) {
|
|
456
|
+
return (`${playbook}\n\n---\n\n` +
|
|
457
|
+
`# Running in CI\n\n` +
|
|
458
|
+
`You are running unattended in a CI job. There is no person to ask: never ask a question or wait for an answer; decide, and say in findings and notes what you assumed.\n\n` +
|
|
459
|
+
`- The browser is already attached to the target in ${o.mode} mode, as the default session. scout_attach, scout_close, scout_session and scout_playbook are not available: the mode and the target are fixed for this run, and the method is the text above. Skip the Setup steps that choose them, and never pass \`session\`.\n` +
|
|
460
|
+
`- The level is ${o.level}. When you have met its contract, or your budget is nearly spent, call scout_report {level: "${o.level}"}. Never pass force: if the contract is unmet the run writes the report anyway, with its gap ledger.\n` +
|
|
461
|
+
`- Tool results are text only; screenshots are not available. Long results are cut: prefer calls that return less (a scout_crawl first, snapshots only where you act).\n` +
|
|
462
|
+
`- When you are finished, reply with a short summary and no tool call. That ends the run.\n`);
|
|
463
|
+
}
|
|
464
|
+
export function ciKickoff(o) {
|
|
465
|
+
return [
|
|
466
|
+
`Run an exploratory test session following the method.`,
|
|
467
|
+
`Target: ${o.url}`,
|
|
468
|
+
`Project directory (for scout_scan): ${o.projectDir}`,
|
|
469
|
+
`Level: ${o.level}`,
|
|
470
|
+
`Write mode: ${o.mode}`,
|
|
471
|
+
o.focus ? `Focus: ${o.focus}` : "",
|
|
472
|
+
`Budget: at most ${o.caps.turns} model turns, ${n(o.caps.tokens)} tokens and ${Math.round(o.caps.wallMs / 60_000)} minutes. Several tool calls in one turn cost one turn. The run stops at the first cap reached and writes the report as it stands.`,
|
|
473
|
+
]
|
|
474
|
+
.filter(Boolean)
|
|
475
|
+
.join("\n");
|
|
476
|
+
}
|
|
477
|
+
// ── what a run writes ───────────────────────────────────────────────────────
|
|
478
|
+
/** A finding as memory.json holds it, checked just enough to be reported. */
|
|
479
|
+
export function readFindings(raw) {
|
|
480
|
+
const list = raw && typeof raw === "object" ? raw.findings : undefined;
|
|
481
|
+
if (!Array.isArray(list))
|
|
482
|
+
return [];
|
|
483
|
+
return list.filter((f) => !!f &&
|
|
484
|
+
typeof f === "object" &&
|
|
485
|
+
typeof f.id === "string" &&
|
|
486
|
+
typeof f.title === "string" &&
|
|
487
|
+
["high", "medium", "low"].includes(f.severity) &&
|
|
488
|
+
typeof f.runs === "number");
|
|
489
|
+
}
|
|
490
|
+
/**
|
|
491
|
+
* The findings this run made or saw again: new ids, and ids whose run count
|
|
492
|
+
* went up. The project's memory also holds earlier runs' findings, which the
|
|
493
|
+
* report lists and this run's summary does not claim.
|
|
494
|
+
*/
|
|
495
|
+
export function findingsThisRun(before, after) {
|
|
496
|
+
const runsBefore = new Map(before.map((f) => [f.id, f.runs]));
|
|
497
|
+
return after.filter((f) => f.status !== "resolved" && (!runsBefore.has(f.id) || f.runs > (runsBefore.get(f.id) ?? 0)));
|
|
498
|
+
}
|
|
499
|
+
const SEVERITY_ORDER = { high: 0, medium: 1, low: 2 };
|
|
500
|
+
function pathOf(url) {
|
|
501
|
+
try {
|
|
502
|
+
const u = new URL(url);
|
|
503
|
+
return `${u.pathname}${u.search}`;
|
|
504
|
+
}
|
|
505
|
+
catch {
|
|
506
|
+
return url;
|
|
507
|
+
}
|
|
508
|
+
}
|
|
509
|
+
/** Findings summary lines are page text a model wrote: nothing that looks like a secret, nothing that breaks the table. */
|
|
510
|
+
const cell = (s, secrets) => markdownCell(redactKeys(redactSecrets(s), secrets));
|
|
511
|
+
/** The job summary: how the run ended, what it spent, and this run's findings. The full report is report.md. */
|
|
512
|
+
export function ciSummaryMarkdown(r, secrets = []) {
|
|
513
|
+
const defects = r.findings.filter((f) => !isWorthALook(f)).sort((a, b) => SEVERITY_ORDER[a.severity] - SEVERITY_ORDER[b.severity]);
|
|
514
|
+
const looks = r.findings.filter(isWorthALook);
|
|
515
|
+
const count = (s) => defects.filter((f) => f.severity === s).length;
|
|
516
|
+
const lines = [
|
|
517
|
+
`## SceneScout CI run`,
|
|
518
|
+
``,
|
|
519
|
+
`**${cell(r.url, secrets)}** — ${describeStop(r.stop, r.caps, r.stopDetail && cell(r.stopDetail, secrets))}. This run reports; it does not gate.`,
|
|
520
|
+
``,
|
|
521
|
+
`| | |`,
|
|
522
|
+
`|---|---|`,
|
|
523
|
+
`| Findings this run | ${defects.length} (${count("high")} high, ${count("medium")} medium, ${count("low")} low)${looks.length ? `, ${looks.length} worth a look` : ""} |`,
|
|
524
|
+
`| Level | ${r.level} — completion contract ${r.contractMet ? "met" : "not met (the report's gap ledger says what is missing)"} |`,
|
|
525
|
+
`| Mode | ${r.mode} |`,
|
|
526
|
+
`| Model | ${r.provider} ${cell(r.model, secrets)}, effort ${r.effort} |`,
|
|
527
|
+
`| Usage | ${usageLine(r.spend, r.model, r.endedAt, r.price)} |`,
|
|
528
|
+
``,
|
|
529
|
+
];
|
|
530
|
+
if (defects.length > 0) {
|
|
531
|
+
lines.push(`| Severity | Category | Finding | Page |`, `|---|---|---|---|`);
|
|
532
|
+
for (const f of defects.slice(0, 50))
|
|
533
|
+
lines.push(`| ${f.severity} | ${cell(f.category, secrets)} | ${cell(f.title, secrets)} | ${cell(pathOf(f.url), secrets)} |`);
|
|
534
|
+
if (defects.length > 50)
|
|
535
|
+
lines.push(``, `… and ${defects.length - 50} more in report.md.`);
|
|
536
|
+
lines.push(``);
|
|
537
|
+
}
|
|
538
|
+
if (looks.length > 0) {
|
|
539
|
+
lines.push(`**Worth a look** (a defect only under a convention of the project; not counted):`, ``);
|
|
540
|
+
for (const f of looks.slice(0, 20))
|
|
541
|
+
lines.push(`- ${cell(f.title, secrets)} — a defect only if your project uses ${cell(f.convention ?? "a convention", secrets)}`);
|
|
542
|
+
lines.push(``);
|
|
543
|
+
}
|
|
544
|
+
lines.push(`The full report, with repro steps and the gap ledger, is report.md.`, ``);
|
|
545
|
+
return lines.join("\n");
|
|
546
|
+
}
|
|
547
|
+
export function ciSummaryJson(r, version, secrets = []) {
|
|
548
|
+
const clean = (s) => redactKeys(redactSecrets(s), secrets);
|
|
549
|
+
return {
|
|
550
|
+
tool: "scenescout",
|
|
551
|
+
command: "ci",
|
|
552
|
+
version,
|
|
553
|
+
url: clean(r.url),
|
|
554
|
+
provider: r.provider,
|
|
555
|
+
model: r.model,
|
|
556
|
+
effort: r.effort,
|
|
557
|
+
mode: r.mode,
|
|
558
|
+
level: r.level,
|
|
559
|
+
caps: { turns: r.caps.turns, tokens: r.caps.tokens, minutes: Math.round(r.caps.wallMs / 60_000) },
|
|
560
|
+
stop: { reason: r.stop, ...(r.stopDetail ? { detail: clean(r.stopDetail) } : {}), text: clean(describeStop(r.stop, r.caps, r.stopDetail)) },
|
|
561
|
+
contractMet: r.contractMet,
|
|
562
|
+
usage: {
|
|
563
|
+
turns: r.spend.turns,
|
|
564
|
+
inputTokens: r.spend.usage.input,
|
|
565
|
+
cachedInputTokens: r.spend.usage.cachedInput,
|
|
566
|
+
cacheWriteTokens: r.spend.usage.cacheWrite,
|
|
567
|
+
outputTokens: r.spend.usage.output,
|
|
568
|
+
seconds: Math.round((r.endedAt - r.spend.startedAt) / 1000),
|
|
569
|
+
estimatedCostUsd: estimateCost(r.model, r.spend.usage, r.price),
|
|
570
|
+
},
|
|
571
|
+
counts: {
|
|
572
|
+
high: r.findings.filter((f) => !isWorthALook(f) && f.severity === "high").length,
|
|
573
|
+
medium: r.findings.filter((f) => !isWorthALook(f) && f.severity === "medium").length,
|
|
574
|
+
low: r.findings.filter((f) => !isWorthALook(f) && f.severity === "low").length,
|
|
575
|
+
worthALook: r.findings.filter(isWorthALook).length,
|
|
576
|
+
},
|
|
577
|
+
findings: r.findings.map((f) => ({
|
|
578
|
+
id: f.id,
|
|
579
|
+
severity: f.severity,
|
|
580
|
+
category: f.category,
|
|
581
|
+
title: clean(f.title),
|
|
582
|
+
path: clean(pathOf(f.url)),
|
|
583
|
+
...(f.evidence ? { evidence: clean(f.evidence) } : {}),
|
|
584
|
+
...(isWorthALook(f) ? { tier: "worth-a-look", convention: clean(f.convention ?? "") } : {}),
|
|
585
|
+
})),
|
|
586
|
+
};
|
|
587
|
+
}
|
|
588
|
+
const SARIF_LEVEL = { high: "error", medium: "warning", low: "note" };
|
|
589
|
+
function appOrigin(url) {
|
|
590
|
+
try {
|
|
591
|
+
return new URL(url).origin;
|
|
592
|
+
}
|
|
593
|
+
catch {
|
|
594
|
+
return url;
|
|
595
|
+
}
|
|
596
|
+
}
|
|
597
|
+
/** This run's findings as SARIF 2.1.0: one rule per category, the finding's id as its fingerprint. */
|
|
598
|
+
export function ciSarif(r, version, secrets = []) {
|
|
599
|
+
const clean = (s) => redactKeys(redactSecrets(s), secrets);
|
|
600
|
+
const categories = [...new Set(r.findings.map((f) => f.category))].sort();
|
|
601
|
+
return {
|
|
602
|
+
$schema: "https://json.schemastore.org/sarif-2.1.0.json",
|
|
603
|
+
version: "2.1.0",
|
|
604
|
+
runs: [
|
|
605
|
+
{
|
|
606
|
+
tool: {
|
|
607
|
+
driver: {
|
|
608
|
+
name: "SceneScout",
|
|
609
|
+
version,
|
|
610
|
+
informationUri: "https://github.com/brunoboto96/SceneScout",
|
|
611
|
+
rules: categories.map((c) => ({ id: `finding/${c}`, name: c, shortDescription: { text: `An exploratory finding: ${c}` } })),
|
|
612
|
+
},
|
|
613
|
+
},
|
|
614
|
+
invocations: [{ executionSuccessful: r.stop !== "provider-error" && r.stop !== "could-not-start", properties: { stop: r.stop } }],
|
|
615
|
+
originalUriBaseIds: { APP: { uri: `${appOrigin(r.url)}/` } },
|
|
616
|
+
results: r.findings.map((f) => ({
|
|
617
|
+
ruleId: `finding/${f.category}`,
|
|
618
|
+
level: isWorthALook(f) ? "note" : SARIF_LEVEL[f.severity],
|
|
619
|
+
message: {
|
|
620
|
+
text: isWorthALook(f)
|
|
621
|
+
? `Worth a look — ${clean(f.title)}. A defect only if your project uses ${clean(f.convention ?? "a convention")}.`
|
|
622
|
+
: `[${f.severity}] ${clean(f.title)}${f.evidence ? ` — ${clean(f.evidence)}` : ""}`,
|
|
623
|
+
},
|
|
624
|
+
locations: [{ physicalLocation: { artifactLocation: { uri: clean(pathOf(f.url)).replace(/^\//, ""), uriBaseId: "APP" } } }],
|
|
625
|
+
partialFingerprints: { "scenescoutFinding/v1": createHash("sha256").update(f.id).digest("hex").slice(0, 32) },
|
|
626
|
+
...(isWorthALook(f) ? { properties: { tier: "worth-a-look", convention: clean(f.convention ?? "") } } : {}),
|
|
627
|
+
})),
|
|
628
|
+
},
|
|
629
|
+
],
|
|
630
|
+
};
|
|
631
|
+
}
|
|
632
|
+
/** Where a CI run writes when not told: beside the project's other SceneScout output. */
|
|
633
|
+
export const CI_DIRNAME = "ci";
|