scenescout 3.15.0 → 3.17.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (59) hide show
  1. package/CHANGELOG.md +87 -0
  2. package/README.md +70 -18
  3. package/dist/browsers.js +28 -0
  4. package/dist/check-run.js +191 -14
  5. package/dist/ci-run.js +268 -52
  6. package/dist/cli.js +107 -47
  7. package/dist/commands.js +3 -2
  8. package/dist/engine/baseline.js +377 -0
  9. package/dist/engine/brief.js +16 -7
  10. package/dist/engine/browser.js +1147 -286
  11. package/dist/engine/calibration.js +61 -30
  12. package/dist/engine/capture.js +164 -0
  13. package/dist/engine/check.js +244 -42
  14. package/dist/engine/ci-lanes.js +215 -0
  15. package/dist/engine/ci.js +136 -18
  16. package/dist/engine/claims.js +159 -3
  17. package/dist/engine/collector.js +561 -30
  18. package/dist/engine/crawl.js +49 -0
  19. package/dist/engine/design.js +281 -38
  20. package/dist/engine/export.js +877 -0
  21. package/dist/engine/fingerprint.js +92 -4
  22. package/dist/engine/flow.js +18 -6
  23. package/dist/engine/forms.js +181 -18
  24. package/dist/engine/journey.js +29 -1
  25. package/dist/engine/lane.js +13 -3
  26. package/dist/engine/launch.js +45 -6
  27. package/dist/engine/limits.js +7 -0
  28. package/dist/engine/live-page.js +49 -2
  29. package/dist/engine/live.js +4 -1
  30. package/dist/engine/memory.js +501 -47
  31. package/dist/engine/open.js +118 -0
  32. package/dist/engine/oracles.js +41 -1
  33. package/dist/engine/plain.js +268 -0
  34. package/dist/engine/png.js +127 -0
  35. package/dist/engine/policy.js +379 -9
  36. package/dist/engine/probes.js +3 -2
  37. package/dist/engine/profiles.js +45 -9
  38. package/dist/engine/project-folder.js +191 -0
  39. package/dist/engine/refresh.js +68 -3
  40. package/dist/engine/replay.js +63 -10
  41. package/dist/engine/report.js +241 -40
  42. package/dist/engine/request.js +317 -23
  43. package/dist/engine/sarif.js +120 -0
  44. package/dist/engine/settle.js +67 -0
  45. package/dist/engine/signed-in.js +256 -0
  46. package/dist/engine/status-pane-page.js +441 -0
  47. package/dist/engine/status-pane.js +128 -0
  48. package/dist/engine/tickets.js +671 -0
  49. package/dist/engine/unload.js +3 -2
  50. package/dist/export-run.js +633 -0
  51. package/dist/first-run.js +5 -0
  52. package/dist/installer.js +378 -8
  53. package/dist/intake.js +104 -0
  54. package/dist/login-run.js +250 -36
  55. package/dist/mcp-server.js +660 -65
  56. package/dist/playbook.js +5 -0
  57. package/dist/prompts.js +106 -0
  58. package/package.json +8 -5
  59. package/skills/scenescout/SKILL.md +49 -16
@@ -0,0 +1,215 @@
1
+ /**
2
+ * Lanes for `scenescout ci` (--lanes): the app split between several model
3
+ * loops that explore at once, each in its own browser session and its own
4
+ * part of the app, drawing on the run's one budget, with their findings folded
5
+ * into the one report. This file holds the rules, none of which needs a
6
+ * browser or a network: which routes the planning crawl found, how they are
7
+ * split (brief.ts, the split scout_lane_brief makes), what each lane is told,
8
+ * and how the lanes' endings become the run's. The loop that runs them is
9
+ * src/ci-run.ts; the shared budget is engine/ci.ts's. Why: ADR 20.
10
+ */
11
+ import { LANE_RULES, planLanes } from "./brief.js";
12
+ import { CI_TOOLS } from "./ci.js";
13
+ /** The session the run attaches first. It plans, holds no lane, and writes the report once every lane is done. */
14
+ export const PLANNER_SESSION = "default";
15
+ /** Crawls the planning step makes at most: each visits the routes the pages of the one before linked to. */
16
+ export const PLAN_CRAWL_ROUNDS = 3;
17
+ /** The most routes a lane's first message lists, and the most crawl lines it is handed. */
18
+ const LIST_MAX = 40;
19
+ /**
20
+ * A lane's tools: an exploring run's, less scout_report. The run writes one
21
+ * report for every lane once all are done; a lane that wrote its own would
22
+ * report the others' work as unfinished.
23
+ */
24
+ export const LANE_TOOLS = CI_TOOLS.filter((t) => t !== "scout_report");
25
+ // ── what the planning crawl found ───────────────────────────────────────────
26
+ /**
27
+ * A route line of a scout_crawl result: `<path> — <status> · <n> el …`, or
28
+ * `<path> — LOAD FAILED`. The path is what the crawl visited, as the browser
29
+ * opened it.
30
+ */
31
+ const CRAWL_LINE = /^(\/\S*) — (.+)$/;
32
+ /** The heading of the crawl's problem list, whose entries name a route and say what is wrong on it. */
33
+ const PROBLEMS_HEADING = /^PROBLEM ROUTES \(\d+\):$/;
34
+ /** A problem entry's first line: the route, then ` → …`, `: <why it did not load>`, or nothing (its detail lines follow, indented). */
35
+ const PROBLEM_LINE = /^(\/\S*?)(?: →.*|: .*)?$/;
36
+ /**
37
+ * What a scout_crawl result says about each route it visited: the route's own
38
+ * line, then whatever its problem list says about it. Routes it skipped as
39
+ * off-origin are not the app's, and are left out.
40
+ */
41
+ export function crawlNotes(text) {
42
+ const notes = new Map();
43
+ const add = (route, line) => {
44
+ const list = notes.get(route);
45
+ if (list)
46
+ list.push(line);
47
+ else
48
+ notes.set(route, [line]);
49
+ };
50
+ let inProblems = false;
51
+ let current = null;
52
+ for (const raw of text.split(/\r?\n/)) {
53
+ const line = raw.trimEnd();
54
+ if (PROBLEMS_HEADING.test(line)) {
55
+ inProblems = true;
56
+ continue;
57
+ }
58
+ if (!inProblems) {
59
+ const m = CRAWL_LINE.exec(line);
60
+ if (m && !/^SKIPPED/.test(m[2]))
61
+ add(m[1], line);
62
+ continue;
63
+ }
64
+ if (line === "") {
65
+ // A blank line ends the entry it follows, not the list.
66
+ current = null;
67
+ continue;
68
+ }
69
+ if (!/^(\/|\s)/.test(line)) {
70
+ // A line that is neither a route nor a detail (the coverage line) ends the list.
71
+ inProblems = false;
72
+ current = null;
73
+ continue;
74
+ }
75
+ if (/^\s/.test(line)) {
76
+ if (current)
77
+ add(current, line);
78
+ continue;
79
+ }
80
+ const p = PROBLEM_LINE.exec(line);
81
+ current = p && notes.has(p[1]) ? p[1] : null;
82
+ if (current)
83
+ add(current, line);
84
+ }
85
+ return notes;
86
+ }
87
+ /** Whether a crawl had nothing left to visit, so another round would find nothing either. */
88
+ export function crawlFoundNothing(text) {
89
+ return /^(?:\[session [^\]]*\]\s*)?(Nothing to crawl|Nothing new to crawl|No routes to crawl yet)/m.test(text);
90
+ }
91
+ /** The path a target URL opens, as a route: the planning crawl never lists the page the run attached on. */
92
+ function routeOf(url) {
93
+ try {
94
+ const u = new URL(url);
95
+ return `${u.pathname}${u.search}`;
96
+ }
97
+ catch {
98
+ return "/";
99
+ }
100
+ }
101
+ /**
102
+ * Session names for the lanes: the lane's own name, unless another lane or the
103
+ * planner already has it. A session name is at most 40 characters.
104
+ */
105
+ export function laneSessions(names) {
106
+ const taken = new Set([PLANNER_SESSION]);
107
+ return names.map((name, i) => {
108
+ let session = name;
109
+ for (let k = i + 1; taken.has(session); k += 1)
110
+ session = `${name.slice(0, 40 - `-${k}`.length)}-${k}`;
111
+ taken.add(session);
112
+ return session;
113
+ });
114
+ }
115
+ /**
116
+ * Split what the planning crawl found between `count` lanes, by brief.ts: whole
117
+ * modules (a route's first path segment) per lane, balanced by route count.
118
+ * Fewer lanes come back when there are fewer modules. With fewer than two the
119
+ * run has nothing to split, and explores in one loop, saying why.
120
+ */
121
+ export function planCiLanes(o) {
122
+ const routes = [...new Set([routeOf(o.target), ...o.notes.keys()])];
123
+ if (o.count < 2)
124
+ return { lanes: [], oneLoop: "one lane was asked for" };
125
+ if (o.notes.size === 0 && o.planningFailed)
126
+ return { lanes: [], oneLoop: `${o.planningFailed}, so there was nothing to split` };
127
+ const briefs = planLanes(routes, o.count, { goal: o.focus, mode: o.mode });
128
+ if (briefs.length < 2) {
129
+ const where = briefs[0]?.modules[0];
130
+ return {
131
+ lanes: [],
132
+ oneLoop: `the planning crawl found ${routes.length} route(s), all in ${where ? `one module (${where})` : "no module"}, so there was nothing to split`,
133
+ };
134
+ }
135
+ const sessions = laneSessions(briefs.map((b) => b.lane));
136
+ return {
137
+ lanes: briefs.map((b, i) => ({
138
+ ...b,
139
+ session: sessions[i],
140
+ // On the target's origin whatever the route says: a path written `//host/x` must not become another host.
141
+ url: `${new URL(o.target).origin}${b.landing.startsWith("/") ? "" : "/"}${b.landing}`,
142
+ crawl: b.routes.flatMap((r) => o.notes.get(r) ?? []).slice(0, LIST_MAX),
143
+ })),
144
+ };
145
+ }
146
+ // ── what a lane is told ─────────────────────────────────────────────────────
147
+ /**
148
+ * The system prompt every lane is given: the method, then the rules of an
149
+ * unattended run as one lane of several. The same for every lane of a run, so
150
+ * the provider's prompt cache serves them all; what differs goes in the
151
+ * lane's first message (ciLaneKickoff).
152
+ */
153
+ export function ciLaneSystemPrompt(playbook, o) {
154
+ return (`${playbook}\n\n---\n\n` +
155
+ `# Running in CI, as one lane of several\n\n` +
156
+ `You are running unattended in a CI job, as one of several lanes exploring the same app at once, each in its own browser session and its own part of the app. There is no person to ask: never ask a question or wait for an answer; decide, and say in findings and notes what you assumed.\n\n` +
157
+ `- Your browser is already attached to the target in ${o.mode} mode, as your lane's own session. scout_attach, scout_close, scout_session, scout_playbook and scout_report are not available: the mode, the target and the session are fixed for this run, and the method is the text above. Skip the Setup steps that choose them, and never pass \`session\`.\n` +
158
+ `- The planner has crawled the routes it found, so route coverage may already read as complete. Your first message lists the routes your lane owns, from that crawl, and what it saw on them; the other lanes own the rest, at the same time. A route under your modules that the crawl did not list is yours too.\n` +
159
+ LANE_RULES.map((rule) => `- ${rule}\n`).join("") +
160
+ `- The level is ${o.level}. The run writes one report for every lane once all are done, and its contract counts what every lane did: do your share of it on your routes, the design audits included.\n` +
161
+ `- Tool results are text only; screenshots are not available. Long results are cut: prefer calls that return less, and snapshots only where you act.\n` +
162
+ `- When your routes are done, or your share of the budget is nearly spent, reply with a short summary and no tool call. That ends your lane; the others go on.\n`);
163
+ }
164
+ const n = (x) => x.toLocaleString("en-US");
165
+ /** A lane's first message: which lane it is, what it owns, where its browser is, what the crawl saw there, and its share of the budget. */
166
+ export function ciLaneKickoff(o) {
167
+ const { lane } = o;
168
+ const share = Math.max(1, Math.floor(o.caps.turns / o.laneCount));
169
+ return [
170
+ `Run an exploratory test session following the method, as lane "${lane.session}", one of ${o.laneCount} running at once.`,
171
+ `Target: ${o.url}`,
172
+ `Your lane: ${lane.objective}`,
173
+ `Your routes (${lane.routes.length}): ${lane.routes.slice(0, LIST_MAX).join(", ")}${lane.routes.length > LIST_MAX ? ` … and ${lane.routes.length - LIST_MAX} more` : ""}`,
174
+ `Your browser is on ${routeOf(lane.url)}.`,
175
+ lane.crawl.length > 0 ? `What the planning crawl saw on your routes:\n${lane.crawl.map((l) => ` ${l.trim()}`).join("\n")}` : "",
176
+ `Project directory (for scout_scan): ${o.projectDir}`,
177
+ `Level: ${o.level}`,
178
+ `Write mode: ${o.mode}`,
179
+ o.focus ? `Focus: ${o.focus}` : "",
180
+ `Budget: the run's ${o.caps.turns} model turns, ${n(o.caps.tokens)} tokens and ${Math.round(o.caps.wallMs / 60_000)} minutes are shared by the ${o.laneCount} lanes: plan on about ${share} turns. Several tool calls in one turn cost one turn. Every lane stops at the first cap the run reaches, and the report is written as it stands.`,
181
+ ]
182
+ .filter(Boolean)
183
+ .join("\n");
184
+ }
185
+ // ── how the lanes' endings become the run's ─────────────────────────────────
186
+ /** Caps in the order capReached checks them, so a run ended by two names the one checked first. */
187
+ const CAP_ORDER = ["time", "tokens", "turns"];
188
+ /**
189
+ * What ended a run whose exploration was split into lanes. A model API failure
190
+ * in any lane is the run's, since it is what the workflow must fix, and so is
191
+ * a lane that broke after attaching (could-not-start on an attached lane: the
192
+ * run could not finish, exit 2). Otherwise the run ended at a cap if any lane
193
+ * was stopped by one, a lane the time cap left no time to attach included, and
194
+ * the model finished it only when every lane that ran finished by itself. A
195
+ * lane whose browser could not attach does not change that, but is named in
196
+ * the detail; when none could, the run could not start.
197
+ */
198
+ export function mergeLaneStops(lanes) {
199
+ const failed = lanes.find((l) => l.attached && l.stop === "provider-error");
200
+ if (failed)
201
+ return { stop: "provider-error", stopDetail: `lane ${failed.session}${failed.stopDetail ? `: ${failed.stopDetail}` : ""}` };
202
+ const broke = lanes.find((l) => l.attached && l.stop === "could-not-start");
203
+ if (broke)
204
+ return { stop: "could-not-start", stopDetail: `lane ${broke.session}: ${broke.stopDetail ?? "it stopped for no reason it gave"}` };
205
+ const unattached = lanes.filter((l) => !l.attached && l.stop === "could-not-start").map((l) => l.session);
206
+ if (unattached.length === lanes.length) {
207
+ const first = lanes.find((l) => l.stopDetail);
208
+ return { stop: "could-not-start", stopDetail: `no lane could attach${first ? ` (${first.session}: ${first.stopDetail})` : ""}` };
209
+ }
210
+ const note = unattached.length > 0 ? { stopDetail: `${unattached.length} of ${lanes.length} lanes could not attach: ${unattached.join(", ")}` } : {};
211
+ for (const cap of CAP_ORDER)
212
+ if (lanes.some((l) => l.stop === cap))
213
+ return { stop: cap, ...note };
214
+ return { stop: "done", ...note };
215
+ }
package/dist/engine/ci.js CHANGED
@@ -11,9 +11,11 @@
11
11
  import { createHash } from "node:crypto";
12
12
  import path from "node:path";
13
13
  import { BROWSER_ENGINES } from "../browsers.js";
14
+ import { MAX_LANES } from "./brief.js";
14
15
  import { markdownCell } from "./check.js";
15
16
  import { parseLimitFlag } from "./limits.js";
16
17
  import { isWorthALook, redactSecrets } from "./memory.js";
18
+ import { SARIF_ANCHOR_FALLBACKS, checkSarifAnchor, sarifLocation } from "./sarif.js";
17
19
  // ── options ─────────────────────────────────────────────────────────────────
18
20
  export const PROVIDERS = ["anthropic", "openai"];
19
21
  /** Where each provider's key is read from. Nothing else is: no flag, no file, no input. */
@@ -53,6 +55,14 @@ export const DEDUP_ENV = "SCENESCOUT_DEDUP";
53
55
  export const DEDUP_PROVIDER_ENV = "SCENESCOUT_DEDUP_PROVIDER";
54
56
  export const DEFAULT_CAPS = { turns: 40, tokens: 1_500_000, wallMs: 20 * 60_000 };
55
57
  const CAP_BOUNDS = { turns: [1, 500], tokens: [1_000, 20_000_000], minutes: [1, 360] };
58
+ /**
59
+ * How many model loops explore at once, each in its own browser session and
60
+ * its own part of the app (engine/ci-lanes.ts). 1 is the single loop. The most
61
+ * is scout_lane_brief's, since the split is the same one. Why the default is
62
+ * what it is: docs/benchmark.md, "Unattended runs".
63
+ */
64
+ export const DEFAULT_LANES = 1;
65
+ export const MAX_CI_LANES = MAX_LANES;
56
66
  /** Every option `scenescout ci` accepts; the ci action's inputs are these names (ci-test holds them equal). */
57
67
  export const CI_OPTION_NAMES = [
58
68
  "provider",
@@ -62,6 +72,7 @@ export const CI_OPTION_NAMES = [
62
72
  "max-turns",
63
73
  "max-tokens",
64
74
  "max-minutes",
75
+ "lanes",
65
76
  "price-in",
66
77
  "price-cached-in",
67
78
  "price-out",
@@ -78,6 +89,7 @@ export const CI_OPTION_NAMES = [
78
89
  "show",
79
90
  "compare-url",
80
91
  "dedup",
92
+ "sarif-file-anchor",
81
93
  ];
82
94
  /** The longest --show description: it becomes a line of the model's prompt. */
83
95
  export const MAX_SHOW = 200;
@@ -171,9 +183,13 @@ export function parseCiArgs(args, cwd) {
171
183
  const turns = whole("max-turns", CAP_BOUNDS.turns, DEFAULT_CAPS.turns);
172
184
  const tokens = whole("max-tokens", CAP_BOUNDS.tokens, DEFAULT_CAPS.tokens);
173
185
  const minutes = whole("max-minutes", CAP_BOUNDS.minutes, DEFAULT_CAPS.wallMs / 60_000);
174
- for (const v of [turns, tokens, minutes])
186
+ const lanes = whole("lanes", [1, MAX_CI_LANES], DEFAULT_LANES);
187
+ for (const v of [turns, tokens, minutes, lanes])
175
188
  if (typeof v === "string")
176
189
  return { ok: false, error: v };
190
+ // The lanes share the run's turns rather than getting a cap each: fewer turns than lanes would leave a lane none.
191
+ if (lanes > turns)
192
+ return { ok: false, error: `--lanes ${lanes} needs --max-turns of at least ${lanes}: the lanes share the run's turns, and each needs one` };
177
193
  const price = {};
178
194
  for (const [flag, field] of [
179
195
  ["price-in", "input"],
@@ -218,6 +234,8 @@ export function parseCiArgs(args, cwd) {
218
234
  return { ok: false, error: `--show is at most ${MAX_SHOW} characters` };
219
235
  if (flags.has("show") && !show)
220
236
  return { ok: false, error: '--show needs a few words describing the element, e.g. --show "the Save button"' };
237
+ if (show && lanes > 1)
238
+ return { ok: false, error: "--lanes splits an exploration between model loops, and --show explores nothing: give one or the other" };
221
239
  let compareUrl;
222
240
  if (flags.has("compare-url")) {
223
241
  if (!show)
@@ -244,6 +262,9 @@ export function parseCiArgs(args, cwd) {
244
262
  const dedup = flags.get("dedup") ?? DEFAULT_CI_DEDUP;
245
263
  if (!DEDUP_MODES.includes(dedup))
246
264
  return { ok: false, error: `--dedup must be one of ${DEDUP_MODES.join(", ")}` };
265
+ const anchor = flags.has("sarif-file-anchor") ? checkSarifAnchor(flags.get("sarif-file-anchor")) : undefined;
266
+ if (anchor && !anchor.ok)
267
+ return anchor;
247
268
  const resolve = (p) => (p.startsWith("/") || /^[A-Za-z]:[\\/]/.test(p) ? p : `${cwd.replace(/[\\/]$/, "")}/${p}`);
248
269
  return {
249
270
  ok: true,
@@ -256,6 +277,7 @@ export function parseCiArgs(args, cwd) {
256
277
  ...(effort ? { effort } : {}),
257
278
  ...(baseUrl ? { baseUrl } : {}),
258
279
  caps: { turns: turns, tokens: tokens, wallMs: minutes * 60_000 },
280
+ lanes: lanes,
259
281
  ...(Object.keys(price).length > 0 ? { price } : {}),
260
282
  mode: mode,
261
283
  level: level,
@@ -267,9 +289,23 @@ export function parseCiArgs(args, cwd) {
267
289
  ...(actionTimeout.value !== undefined ? { actionTimeoutMs: actionTimeout.value } : {}),
268
290
  ...(navTimeout.value !== undefined ? { navTimeoutMs: navTimeout.value } : {}),
269
291
  dedup: dedup,
292
+ ...(anchor ? { sarifFileAnchor: anchor.value } : {}),
270
293
  },
271
294
  };
272
295
  }
296
+ /**
297
+ * Why scout_attach did not leave a session the run can use, or null when it
298
+ * did: an error result, or a saved sign-in the app no longer accepts (the
299
+ * attach succeeds, and says so on a line of its own).
300
+ */
301
+ export function attachFailure(r) {
302
+ const authFailed = r.text.split("\n").find((l) => l.startsWith("⚠ AUTH FAILED"));
303
+ if (authFailed)
304
+ return authFailed;
305
+ if (r.isError || /^ERROR:/.test(r.text))
306
+ return r.text.replace(/^ERROR:\s*/, "");
307
+ return null;
308
+ }
273
309
  const present = (env, name) => (env[name] ?? "").trim() !== "";
274
310
  /**
275
311
  * Which provider a run uses: the one whose key is set. With both set the
@@ -362,6 +398,12 @@ export function childEnv(env) {
362
398
  out[k] = v;
363
399
  // Nobody watches a CI run: the live view would only hold a port open.
364
400
  out.SCENESCOUT_LIVE = "off";
401
+ // ...and an unattended run opens nothing in a browser, even on a desktop, unless the user set it.
402
+ if (!out.SCENESCOUT_OPEN?.trim())
403
+ out.SCENESCOUT_OPEN = "none";
404
+ // The loop is text-only (toolResultText), so a finding's picture is kept for the report, never sent back. A choice the job made is its own.
405
+ if (!out.SCENESCOUT_EVIDENCE?.trim())
406
+ out.SCENESCOUT_EVIDENCE = "file";
365
407
  return out;
366
408
  }
367
409
  // ── never printing a key ────────────────────────────────────────────────────
@@ -392,7 +434,8 @@ export function addUsage(a, b) {
392
434
  /**
393
435
  * Which cap, if any, stops the run before its next model call. Checked
394
436
  * between turns: one turn's usage is only known after it, so a run can end
395
- * up to one turn over the token cap, and the report says by how much.
437
+ * up to one turn over the token cap (one per lane in a run split into lanes:
438
+ * see Budget), and the report says by how much.
396
439
  */
397
440
  export function capReached(spend, caps, now) {
398
441
  if (now - spend.startedAt >= caps.wallMs)
@@ -407,6 +450,31 @@ export function capReached(spend, caps, now) {
407
450
  export function wallLeftMs(spend, caps, now) {
408
451
  return Math.max(0, caps.wallMs - (now - spend.startedAt));
409
452
  }
453
+ export function newBudget(caps, startedAt) {
454
+ return { caps, startedAt, turns: 0, usage: { ...NO_USAGE }, inFlight: 0 };
455
+ }
456
+ /** What the budget's loops have spent between them. */
457
+ export function budgetSpend(b) {
458
+ return { turns: b.turns, usage: { ...b.usage }, startedAt: b.startedAt };
459
+ }
460
+ /** Take the next turn for one loop, or name the cap that refuses it. */
461
+ export function takeTurn(b, now) {
462
+ const cap = capReached({ turns: b.turns + b.inFlight, usage: b.usage, startedAt: b.startedAt }, b.caps, now);
463
+ if (cap === null)
464
+ b.inFlight += 1;
465
+ return cap;
466
+ }
467
+ /** A taken turn's call has ended: counted with the usage it reported, or given back when it failed. */
468
+ export function settleTurn(b, usage) {
469
+ // A settle with no turn taken would count a call nobody reserved: a bug in the caller, not a state to carry on from.
470
+ if (b.inFlight <= 0)
471
+ throw new Error("settleTurn without a turn taken: every settle must follow a takeTurn that returned null");
472
+ b.inFlight -= 1;
473
+ if (!usage)
474
+ return;
475
+ b.turns += 1;
476
+ b.usage = addUsage(b.usage, usage);
477
+ }
410
478
  export const EXIT_CI = { completed: 0, couldNotRun: 2 };
411
479
  /**
412
480
  * A CI run reports and never gates, so its findings never set the exit code.
@@ -486,7 +554,8 @@ export function usageLine(spend, model, endedAt, override) {
486
554
  * closes itself, so the mode and the target cannot change), scout_session,
487
555
  * scout_playbook (the method is the system prompt), scout_screenshot (the
488
556
  * loop is text-only), scout_resolve (scout_verify records re-tests), and the
489
- * lane tools (one agent, ADR 14).
557
+ * lane tools (a run split into lanes is planned and folded by the run itself,
558
+ * not by a model: engine/ci-lanes.ts, ADR 20).
490
559
  */
491
560
  export const CI_TOOLS = [
492
561
  "scout_scan",
@@ -683,12 +752,15 @@ export function ciSummaryMarkdown(r, secrets = []) {
683
752
  ...(r.capture ? [] : [`| Level | ${r.level} — completion contract ${r.contractMet ? "met" : "not met (the report's gap ledger says what is missing)"} |`]),
684
753
  `| Mode | ${r.mode} |`,
685
754
  `| Model | ${r.provider} ${cell(r.model, secrets)}, effort ${r.effort} |`,
755
+ ...(r.lanes ? [`| Lanes | ${lanesCell(r.lanes, secrets)} |`] : []),
686
756
  ...(r.dedup ? [`| Finding dedup | ${dedupLine(r.dedup)} |`] : []),
687
757
  `| Usage | ${usageLine(r.spend, r.model, r.endedAt, r.price)} |`,
688
758
  ``,
689
759
  ];
690
760
  if (r.capture)
691
761
  lines.push(...captureSummaryLines(r.capture, secrets));
762
+ if (r.lanes && r.lanes.sessions.length > 0)
763
+ lines.push(...laneSummaryLines(r.lanes, secrets));
692
764
  if (defects.length > 0) {
693
765
  lines.push(`| Severity | Category | Finding | Page |`, `|---|---|---|---|`);
694
766
  for (const f of defects.slice(0, 50))
@@ -707,6 +779,22 @@ export function ciSummaryMarkdown(r, secrets = []) {
707
779
  lines.push(r.capture ? `The pictures are in shots/.` : `The full report, with repro steps and the gap ledger, is report.md.`, ``);
708
780
  return lines.join("\n");
709
781
  }
782
+ function lanesCell(l, secrets) {
783
+ if (l.oneLoop)
784
+ return `${l.asked} asked; explored in one loop: ${cell(l.oneLoop, secrets)}`;
785
+ const ran = l.sessions.filter((s) => s.attached).length;
786
+ const planned = l.sessions.length;
787
+ return (`${ran} of ${l.asked} asked ran at once, sharing the caps below` +
788
+ (planned < l.asked ? `; the app split into ${planned}` : "") +
789
+ (ran < planned ? `; ${planned - ran} could not attach` : ""));
790
+ }
791
+ function laneSummaryLines(l, secrets) {
792
+ const out = [`| Lane | Owns | Routes | Turns | Tokens | Ended |`, `|---|---|---:|---:|---:|---|`];
793
+ for (const s of l.sessions)
794
+ out.push(`| ${cell(s.session, secrets)} | ${cell(s.modules.join(", "), secrets)} | ${s.routes} | ${s.turns} | ${n(s.usage.input + s.usage.output)} | ${s.stop}${s.stopDetail ? `: ${cell(s.stopDetail, secrets)}` : ""} |`);
795
+ out.push(``);
796
+ return out;
797
+ }
710
798
  function captureSummaryLines(c, secrets) {
711
799
  const out = [`**Asked to show:** ${cell(c.what, secrets)}`, ``];
712
800
  if (c.status !== "captured")
@@ -781,6 +869,28 @@ export function ciSummaryJson(r, version, secrets = []) {
781
869
  ...(isWorthALook(f) ? { tier: "worth-a-look", convention: clean(f.convention ?? "") } : {}),
782
870
  })),
783
871
  ...(r.capture ? { capture: cleanCapture(r.capture, clean) } : {}),
872
+ ...(r.lanes ? { lanes: lanesJson(r.lanes, clean) } : {}),
873
+ };
874
+ }
875
+ /** The lanes as ci.json holds them. Module paths and lane names come from the app's routes: redacted like the rest. */
876
+ function lanesJson(l, clean) {
877
+ return {
878
+ asked: l.asked,
879
+ planned: l.sessions.length,
880
+ ran: l.sessions.filter((s) => s.attached).length,
881
+ ...(l.oneLoop ? { oneLoop: clean(l.oneLoop) } : {}),
882
+ sessions: l.sessions.map((s) => ({
883
+ session: clean(s.session),
884
+ modules: s.modules.map(clean),
885
+ routes: s.routes,
886
+ attached: s.attached,
887
+ stop: s.stop,
888
+ ...(s.stopDetail ? { detail: clean(s.stopDetail) } : {}),
889
+ turns: s.turns,
890
+ inputTokens: s.usage.input,
891
+ cachedInputTokens: s.usage.cachedInput,
892
+ outputTokens: s.usage.output,
893
+ })),
784
894
  };
785
895
  }
786
896
  /** The capture as ci.json holds it: the words a person or a model wrote, and the URLs, passed through the same redaction as the rest. */
@@ -802,8 +912,13 @@ function appOrigin(url) {
802
912
  return url;
803
913
  }
804
914
  }
805
- /** This run's findings as SARIF 2.1.0: one rule per category, the finding's id as its fingerprint. */
806
- export function ciSarif(r, version, secrets = []) {
915
+ /**
916
+ * This run's findings as SARIF 2.1.0: one rule per category, the finding's id
917
+ * as its fingerprint. Code scanning keeps only results located in a repository
918
+ * file, so each points at the anchor (sarif.ts) and carries its page as a
919
+ * logical location, in `properties.route` and in its message.
920
+ */
921
+ export function ciSarif(r, version, secrets = [], anchor = SARIF_ANCHOR_FALLBACKS[0]) {
807
922
  const clean = (s) => redactKeys(redactSecrets(s), secrets);
808
923
  const categories = [...new Set(r.findings.map((f) => f.category))].sort();
809
924
  return {
@@ -820,19 +935,22 @@ export function ciSarif(r, version, secrets = []) {
820
935
  },
821
936
  },
822
937
  invocations: [{ executionSuccessful: r.stop !== "provider-error" && r.stop !== "could-not-start", properties: { stop: r.stop } }],
823
- originalUriBaseIds: { APP: { uri: `${appOrigin(r.url)}/` } },
824
- results: r.findings.map((f) => ({
825
- ruleId: `finding/${f.category}`,
826
- level: isWorthALook(f) ? "note" : SARIF_LEVEL[f.severity],
827
- message: {
828
- text: isWorthALook(f)
829
- ? `Worth a look — ${clean(f.title)}. A defect only if your project uses ${clean(f.convention ?? "a convention")}.`
830
- : `[${f.severity}] ${clean(f.title)}${f.evidence ? ` — ${clean(f.evidence)}` : ""}`,
831
- },
832
- locations: [{ physicalLocation: { artifactLocation: { uri: clean(pathOf(f.url)).replace(/^\//, ""), uriBaseId: "APP" } } }],
833
- partialFingerprints: { "scenescoutFinding/v1": createHash("sha256").update(f.id).digest("hex").slice(0, 32) },
834
- ...(isWorthALook(f) ? { properties: { tier: "worth-a-look", convention: clean(f.convention ?? "") } } : {}),
835
- })),
938
+ properties: { app: clean(appOrigin(r.url)) },
939
+ results: r.findings.map((f) => {
940
+ const route = clean(pathOf(f.url));
941
+ return {
942
+ ruleId: `finding/${f.category}`,
943
+ level: isWorthALook(f) ? "note" : SARIF_LEVEL[f.severity],
944
+ message: {
945
+ text: isWorthALook(f)
946
+ ? `Worth a look — ${clean(f.title)} — on ${route}. A defect only if your project uses ${clean(f.convention ?? "a convention")}.`
947
+ : `[${f.severity}] ${clean(f.title)}${f.evidence ? ` — ${clean(f.evidence)}` : ""} — on ${route}`,
948
+ },
949
+ locations: [sarifLocation(anchor, route)],
950
+ partialFingerprints: { "scenescoutFinding/v1": createHash("sha256").update(f.id).digest("hex").slice(0, 32) },
951
+ properties: { route, ...(isWorthALook(f) ? { tier: "worth-a-look", convention: clean(f.convention ?? "") } : {}) },
952
+ };
953
+ }),
836
954
  },
837
955
  ],
838
956
  };