scenescout 3.15.0 → 3.17.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (59) hide show
  1. package/CHANGELOG.md +87 -0
  2. package/README.md +70 -18
  3. package/dist/browsers.js +28 -0
  4. package/dist/check-run.js +191 -14
  5. package/dist/ci-run.js +268 -52
  6. package/dist/cli.js +107 -47
  7. package/dist/commands.js +3 -2
  8. package/dist/engine/baseline.js +377 -0
  9. package/dist/engine/brief.js +16 -7
  10. package/dist/engine/browser.js +1147 -286
  11. package/dist/engine/calibration.js +61 -30
  12. package/dist/engine/capture.js +164 -0
  13. package/dist/engine/check.js +244 -42
  14. package/dist/engine/ci-lanes.js +215 -0
  15. package/dist/engine/ci.js +136 -18
  16. package/dist/engine/claims.js +159 -3
  17. package/dist/engine/collector.js +561 -30
  18. package/dist/engine/crawl.js +49 -0
  19. package/dist/engine/design.js +281 -38
  20. package/dist/engine/export.js +877 -0
  21. package/dist/engine/fingerprint.js +92 -4
  22. package/dist/engine/flow.js +18 -6
  23. package/dist/engine/forms.js +181 -18
  24. package/dist/engine/journey.js +29 -1
  25. package/dist/engine/lane.js +13 -3
  26. package/dist/engine/launch.js +45 -6
  27. package/dist/engine/limits.js +7 -0
  28. package/dist/engine/live-page.js +49 -2
  29. package/dist/engine/live.js +4 -1
  30. package/dist/engine/memory.js +501 -47
  31. package/dist/engine/open.js +118 -0
  32. package/dist/engine/oracles.js +41 -1
  33. package/dist/engine/plain.js +268 -0
  34. package/dist/engine/png.js +127 -0
  35. package/dist/engine/policy.js +379 -9
  36. package/dist/engine/probes.js +3 -2
  37. package/dist/engine/profiles.js +45 -9
  38. package/dist/engine/project-folder.js +191 -0
  39. package/dist/engine/refresh.js +68 -3
  40. package/dist/engine/replay.js +63 -10
  41. package/dist/engine/report.js +241 -40
  42. package/dist/engine/request.js +317 -23
  43. package/dist/engine/sarif.js +120 -0
  44. package/dist/engine/settle.js +67 -0
  45. package/dist/engine/signed-in.js +256 -0
  46. package/dist/engine/status-pane-page.js +441 -0
  47. package/dist/engine/status-pane.js +128 -0
  48. package/dist/engine/tickets.js +671 -0
  49. package/dist/engine/unload.js +3 -2
  50. package/dist/export-run.js +633 -0
  51. package/dist/first-run.js +5 -0
  52. package/dist/installer.js +378 -8
  53. package/dist/intake.js +104 -0
  54. package/dist/login-run.js +250 -36
  55. package/dist/mcp-server.js +660 -65
  56. package/dist/playbook.js +5 -0
  57. package/dist/prompts.js +106 -0
  58. package/package.json +8 -5
  59. package/skills/scenescout/SKILL.md +49 -16
package/dist/ci-run.js CHANGED
@@ -1,8 +1,9 @@
1
1
  /**
2
2
  * Runs `scenescout ci`: starts the SceneScout MCP server as a child process,
3
3
  * attaches it to the app, lets a model drive the scout_* tools until it is
4
- * done or a cap ends the run, then has the report written and writes the CI
5
- * files. The rules (options, caps, redaction, which tools, the files) are in
4
+ * done or a cap ends the run (with --lanes, several conversations with the
5
+ * model at once, one per part of the app, each in its own session), then has
6
+ * the report written and writes the CI files. The rules (options, caps, redaction, which tools, the files) are in
6
7
  * engine/ci.ts and the message shapes in engine/provider.ts; this file only
7
8
  * moves bytes between the model, the server and the disk.
8
9
  *
@@ -20,9 +21,11 @@ import { StdioClientTransport } from "@modelcontextprotocol/sdk/client/stdio.js"
20
21
  import { CreateMessageRequestSchema, ErrorCode, McpError } from "@modelcontextprotocol/sdk/types.js";
21
22
  import { DEDUP_JUDGE_CAPABILITY, durationText, JUDGE_CALL_MS, JUDGE_MAX_OUTPUT_TOKENS, JUDGE_SYSTEM, JUDGE_TOOL, judgeKickoffOf, samplingResultOf, } from "./engine/dedup.js";
22
23
  import { CAPTURE_MARGIN, parseCaptureResult, rebaseUrl, SHOT_FILES, SHOTS_DIRNAME } from "./engine/capture.js";
23
- import { addUsage, wallLeftMs, guardToolArgs, capReached, CAPTURE_TOOLS, childEnv, ciCaptureKickoff, ciCaptureSystemPrompt, CI_DIRNAME, ciExitCode, ciKickoff, ciSarif, ciSummaryJson, ciSummaryMarkdown, ciSystemPrompt, ciToolArgs, ciTools, describeStop, findingsThisRun, NO_USAGE, readFindings, redactKeys, toolResultText, usageLine, } from "./engine/ci.js";
24
+ import { addUsage, attachFailure, budgetSpend, wallLeftMs, guardToolArgs, CAPTURE_TOOLS, childEnv, ciCaptureKickoff, ciCaptureSystemPrompt, CI_DIRNAME, ciExitCode, ciKickoff, ciSarif, ciSummaryJson, ciSummaryMarkdown, ciSystemPrompt, ciToolArgs, ciTools, describeStop, findingsThisRun, newBudget, NO_USAGE, readFindings, redactKeys, settleTurn, takeTurn, toolResultText, usageLine, } from "./engine/ci.js";
25
+ import { ciLaneKickoff, ciLaneSystemPrompt, crawlFoundNothing, crawlNotes, LANE_TOOLS, mergeLaneStops, PLAN_CRAWL_ROUNDS, planCiLanes, PLANNER_SESSION, } from "./engine/ci-lanes.js";
24
26
  import { resolveTimeLimits } from "./engine/limits.js";
25
27
  import { MEMORY_DIRNAME, writeSelfIgnore } from "./engine/memory.js";
28
+ import { sarifFilesFor } from "./engine/sarif.js";
26
29
  import { decodePng, diffImages, encodePng } from "./engine/png.js";
27
30
  import { AnthropicConversation, backoffMs, errorMessage, MalformedReply, OpenAIConversation, retryable, retryAfterMs, } from "./engine/provider.js";
28
31
  import { loadPlaybook } from "./playbook.js";
@@ -253,31 +256,37 @@ function readMemoryFindings(projectDir) {
253
256
  }
254
257
  }
255
258
  /**
256
- * The agent loop. Before each model call the caps are checked; the call runs
257
- * with at most the time that is left; each tool call it asks for is run in
258
- * order, on the one session, and none runs past the time cap: a call reached
259
- * after it is answered as not run, as is any beyond MAX_TOOL_CALLS_PER_TURN.
260
- * A call the loop cannot run (a tool it was not given, arguments that are not
261
- * JSON, a scan of another directory) goes back to the model as an error
262
- * result, so a malformed reply costs a turn, never the run.
259
+ * The agent loop. Before each model call a turn is taken from the budget, or
260
+ * the cap that refuses it ends the loop; the call runs with at most the time
261
+ * that is left; each tool call it asks for is run in order, on the loop's one
262
+ * session, and none runs past the time cap: a call reached after it is
263
+ * answered as not run, as is any beyond MAX_TOOL_CALLS_PER_TURN. A call the
264
+ * loop cannot run (a tool it was not given, arguments that are not JSON, a
265
+ * scan of another directory) goes back to the model as an error result, so a
266
+ * malformed reply costs a turn, never the run.
263
267
  */
264
268
  export async function agentLoop(o) {
265
269
  const now = o.now ?? Date.now;
266
- const spend = o.spend ?? { turns: 0, usage: { ...NO_USAGE }, startedAt: o.startedAt ?? now() };
270
+ const budget = o.budget ?? newBudget(o.caps, o.startedAt ?? now());
271
+ // This loop's own turns and tokens; the budget holds every loop's.
272
+ const spend = { turns: 0, usage: { ...NO_USAGE }, startedAt: budget.startedAt };
273
+ const timeLeft = () => wallLeftMs(spend, budget.caps, now());
267
274
  const allowed = new Set(o.tools.map((t) => t.name));
268
275
  for (;;) {
269
- const cap = capReached(spend, o.caps, now());
276
+ const cap = takeTurn(budget, now());
270
277
  if (cap)
271
278
  return { stop: cap, spend };
272
279
  let turn;
273
280
  try {
274
- turn = await o.client.next(wallLeftMs(spend, o.caps, now()));
281
+ turn = await o.client.next(timeLeft());
275
282
  }
276
283
  catch (err) {
284
+ settleTurn(budget);
277
285
  if (err instanceof OutOfTime)
278
286
  return { stop: "time", spend };
279
287
  return { stop: "provider-error", stopDetail: err instanceof Error ? err.message : String(err), spend };
280
288
  }
289
+ settleTurn(budget, turn.usage);
281
290
  spend.turns += 1;
282
291
  spend.usage = addUsage(spend.usage, turn.usage);
283
292
  if (turn.resume)
@@ -309,15 +318,16 @@ export async function agentLoop(o) {
309
318
  results.push({ id: call.id, isError: true, text: `${call.name} was not run: ${guarded.error}.` });
310
319
  continue;
311
320
  }
312
- const left = wallLeftMs(spend, o.caps, now());
321
+ const left = timeLeft();
313
322
  if (left <= 0) {
314
323
  results.push({ id: call.id, isError: true, text: `${call.name} was not run: the time cap was reached.` });
315
324
  continue;
316
325
  }
326
+ const sent = o.session && o.session.tools.has(call.name) ? { ...guarded.args, session: o.session.name } : guarded.args;
317
327
  try {
318
- const r = await o.host.call(call.name, guarded.args, Math.min(600_000, left));
328
+ const r = await o.host.call(call.name, sent, Math.min(600_000, left));
319
329
  results.push({ id: call.id, isError: r.isError, text: r.text });
320
- o.onResult?.(call.name, guarded.args, r);
330
+ o.onResult?.(call.name, sent, r);
321
331
  }
322
332
  catch (err) {
323
333
  results.push({ id: call.id, isError: true, text: `${call.name} failed: ${err instanceof Error ? err.message : String(err)}` });
@@ -417,10 +427,181 @@ export async function captureShots(o) {
417
427
  return { ...outcome, detail: `the base URL could not be captured: ${err instanceof Error ? err.message : String(err)}` };
418
428
  }
419
429
  }
430
+ /** Whether a listed tool takes `session`: a lane's calls to it are sent to the lane's own. */
431
+ function takesSession(t) {
432
+ const props = t.inputSchema?.properties;
433
+ return !!props && "session" in props;
434
+ }
435
+ const messageOf = (err) => (err instanceof Error ? err.message : String(err));
436
+ /**
437
+ * A run split into lanes (--lanes): plan, then run every lane at once, then
438
+ * fold what they did. The plan is a snapshot and a crawl from the planner's
439
+ * session (no model call: only their time counts), the crawl repeated while it
440
+ * finds routes, and the split brief.ts makes of them. Each lane attaches its
441
+ * own session on the run's target URL, as the planner did (the engine resolves
442
+ * every path against the URL a session attached with), opens its first route,
443
+ * runs the agent loop there with its own conversation, draws its turns from
444
+ * the run's one budget, and is closed when it ends. Their findings are already
445
+ * one: every session files into the project's one memory, whose dedup folds a
446
+ * defect two lanes filed. `outcome` is absent when there was nothing to split,
447
+ * and the run explores in one loop instead.
448
+ */
449
+ async function exploreInLanes(o) {
450
+ const { host, options, log, now, budget } = o;
451
+ const timeLeft = () => wallLeftMs(budgetSpend(budget), options.caps, now());
452
+ // ── plan ──
453
+ // What went wrong while planning, so a plan left with nothing to split says why rather than blaming the app.
454
+ let planningFailed;
455
+ /** One planner call: its text, or undefined after saying why it failed. */
456
+ const plannerCall = async (tool, maxMs, what) => {
457
+ try {
458
+ const r = await host.call(tool, { session: PLANNER_SESSION }, Math.min(maxMs, timeLeft()));
459
+ if (r.isError)
460
+ throw new Error(r.text.replace(/^ERROR:\s*/, ""));
461
+ return r.text;
462
+ }
463
+ catch (err) {
464
+ planningFailed = `the planning ${what} failed: ${messageOf(err).slice(0, 300)}`;
465
+ log(`Lanes: ${planningFailed}.`);
466
+ return undefined;
467
+ }
468
+ };
469
+ // Attaching harvests no links; a snapshot of the page it landed on does, so the first crawl has routes to visit.
470
+ if (timeLeft() > 0)
471
+ await plannerCall("scout_snapshot", 120_000, "snapshot");
472
+ const notes = new Map();
473
+ for (let round = 0; round < PLAN_CRAWL_ROUNDS && timeLeft() > 0; round += 1) {
474
+ const text = await plannerCall("scout_crawl", 600_000, "crawl");
475
+ if (text === undefined)
476
+ break;
477
+ const known = notes.size;
478
+ for (const [route, lines] of crawlNotes(text))
479
+ if (!notes.has(route))
480
+ notes.set(route, lines);
481
+ if (crawlFoundNothing(text) || notes.size === known)
482
+ break;
483
+ }
484
+ const plan = planCiLanes({
485
+ target: options.url,
486
+ notes,
487
+ count: options.lanes,
488
+ focus: options.focus,
489
+ mode: options.mode,
490
+ ...(planningFailed ? { planningFailed } : {}),
491
+ });
492
+ if (plan.oneLoop) {
493
+ log(`Lanes: ${plan.oneLoop}. Exploring in one loop.`);
494
+ return { lanes: { asked: options.lanes, sessions: [], oneLoop: plan.oneLoop } };
495
+ }
496
+ log(`Lanes: ${plan.lanes.length} of ${options.lanes} asked, sharing the caps: ` +
497
+ plan.lanes.map((l) => `${l.session} (${l.modules.join(", ")}; ${l.routes.length} route(s))`).join("; "));
498
+ // ── run ──
499
+ const tools = ciTools(o.listed, LANE_TOOLS);
500
+ const sessionTools = new Set(o.listed.filter(takesSession).map((t) => t.name));
501
+ const system = ciLaneSystemPrompt(loadPlaybook(packageRoot), options);
502
+ const runLane = async (lane) => {
503
+ const say = (line) => log(line.replace(/^(\s*)/, `$1[${lane.session}] `));
504
+ const result = {
505
+ session: lane.session,
506
+ modules: lane.modules,
507
+ routes: lane.routes.length,
508
+ attached: false,
509
+ stop: "could-not-start",
510
+ turns: 0,
511
+ usage: { ...NO_USAGE },
512
+ };
513
+ let attached = false;
514
+ try {
515
+ if (timeLeft() <= 0) {
516
+ say("the time cap was reached before it attached.");
517
+ return { ...result, stop: "time", stopDetail: "the time cap was reached before it attached" };
518
+ }
519
+ const r = await host
520
+ .call("scout_attach", o.attachArgs({ session: lane.session, url: options.url, objective: lane.objective, task: `Starting lane ${lane.session}` }), Math.min(o.attachMs, timeLeft()))
521
+ .catch((err) => ({ text: `ERROR: ${messageOf(err)}`, isError: true }));
522
+ const failed = attachFailure(r);
523
+ if (failed !== null) {
524
+ say(`could not attach: ${failed.slice(0, 300)}`);
525
+ return { ...result, stopDetail: failed.slice(0, 300) };
526
+ }
527
+ attached = true;
528
+ // Its own first route, by its full URL. A landing that does not open costs the lane nothing but the detour: it starts from the target.
529
+ let on = options.url;
530
+ if (lane.url !== options.url && timeLeft() > 0) {
531
+ const opened = await host
532
+ .call("scout_navigate", { session: lane.session, target: lane.url, task: `Opening lane ${lane.session}'s first route` }, Math.min(o.attachMs, timeLeft()))
533
+ .catch((err) => ({ text: `ERROR: ${messageOf(err)}`, isError: true }));
534
+ // With several sessions live the server puts a "[session …]" line first; the verdict is on the line after it.
535
+ const said = opened.text.replace(/^\[session [^\]\n]*\]\n/, "");
536
+ if (opened.isError || /^(ERROR|REFUSED):/.test(said))
537
+ say(`could not open ${lane.landing} (${said.replace(/^ERROR:\s*/, "").slice(0, 200)}); starting from the target instead.`);
538
+ else
539
+ on = lane.url;
540
+ }
541
+ say(`attached on ${new URL(on).pathname}, owning ${lane.modules.join(", ")}.`);
542
+ return await exploreLane(lane, on, say, { ...result, attached: true });
543
+ }
544
+ catch (err) {
545
+ // Not a cap and not the model's API: the lane itself broke. Reported as the lane's, and the run's (mergeLaneStops), never dropped.
546
+ const why = `the lane failed: ${messageOf(err).slice(0, 300)}`;
547
+ say(why);
548
+ return { ...result, attached, stop: "could-not-start", stopDetail: why };
549
+ }
550
+ };
551
+ const exploreLane = async (lane, on, say, result) => {
552
+ const outcome = await agentLoop({
553
+ client: o.makeClient(system, tools, ciLaneKickoff({
554
+ lane: { ...lane, url: on },
555
+ laneCount: plan.lanes.length,
556
+ url: options.url,
557
+ projectDir: options.projectDir,
558
+ mode: options.mode,
559
+ level: options.level,
560
+ focus: options.focus,
561
+ caps: options.caps,
562
+ })),
563
+ host,
564
+ tools,
565
+ caps: options.caps,
566
+ budget,
567
+ log: say,
568
+ now,
569
+ projectDir: options.projectDir,
570
+ session: { name: lane.session, tools: sessionTools },
571
+ });
572
+ say(`ended: ${outcome.stop === "done" ? "the model finished the lane" : describeStop(outcome.stop, options.caps, outcome.stopDetail)}.`);
573
+ // Closed as soon as it is done, so a lane that finished early holds no browser while the others work.
574
+ // Past the time cap nothing more runs here: the run's own close, within FINISH_MS, collects it.
575
+ if (timeLeft() > 0)
576
+ await host
577
+ .call("scout_close", { session: lane.session }, Math.min(30_000, timeLeft()))
578
+ .catch((err) => say(`closing its browser failed: ${messageOf(err)}`));
579
+ return {
580
+ ...result,
581
+ stop: outcome.stop,
582
+ ...(outcome.stopDetail ? { stopDetail: outcome.stopDetail } : {}),
583
+ turns: outcome.spend.turns,
584
+ usage: outcome.spend.usage,
585
+ };
586
+ };
587
+ // ── merge ──
588
+ // Settled, not raced: every lane is waited out before the run moves on. runLane reports its own failures, so none rejects.
589
+ const settled = await Promise.allSettled(plan.lanes.map(runLane));
590
+ const broken = settled.find((s) => s.status === "rejected");
591
+ if (broken)
592
+ throw new Error(`a lane failed outside its own handling: ${messageOf(broken.reason)}`);
593
+ const sessions = settled.map((s) => s.value);
594
+ const merged = mergeLaneStops(sessions);
595
+ return {
596
+ lanes: { asked: options.lanes, sessions },
597
+ outcome: { stop: merged.stop, ...(merged.stopDetail ? { stopDetail: merged.stopDetail } : {}), spend: budgetSpend(budget) },
598
+ };
599
+ }
420
600
  /**
421
601
  * The whole run. `makeClient` is how the model is reached: the HTTP client in
422
- * the CLI, a scripted one in the tests. Everything logged or written passes
423
- * through the key redaction first.
602
+ * the CLI, a scripted one in the tests. It is called once per conversation:
603
+ * once, or once per lane. Everything logged or written passes through the key
604
+ * redaction first.
424
605
  */
425
606
  export async function runCi(options, resolved, deps) {
426
607
  const secrets = deps.secrets ?? [];
@@ -431,18 +612,19 @@ export async function runCi(options, resolved, deps) {
431
612
  const before = readMemoryFindings(options.projectDir);
432
613
  // Pictures an earlier run left in the same output are not this run's: they must never be uploaded as its.
433
614
  fs.rmSync(path.join(outDir, SHOTS_DIRNAME), { recursive: true, force: true });
434
- // One spend for the run: the loop adds its turns, and the dedup judge's calls add their tokens, which the caps count.
435
- const spend = { turns: 0, usage: { ...NO_USAGE }, startedAt };
615
+ // One budget for the run: every loop (one, or each lane) takes its turns from it, and the dedup judge's calls add their tokens, which the caps count.
616
+ const budget = newBudget(options.caps, startedAt);
436
617
  // A run asked to show an element files no findings, so it has nothing to deduplicate.
437
618
  const wantsJudge = options.dedup === "judge" && !options.show;
438
619
  if (wantsJudge && !deps.judge)
439
620
  log("No model was given for the dedup judge; the rule decides duplicates.");
440
621
  const judgeAsk = wantsJudge ? deps.judge : undefined;
441
622
  const judgeCalls = { calls: 0, failed: 0, usage: { ...NO_USAGE }, ms: 0 };
442
- let outcome = { stop: "could-not-start", spend };
623
+ let outcome = { stop: "could-not-start", spend: budgetSpend(budget) };
443
624
  let contractMet = false;
444
625
  let reportWritten = false;
445
626
  let capture;
627
+ let lanes;
446
628
  let host = null;
447
629
  // Set when the exploration ends (or never starts): the report and the close share FINISH_MS from then.
448
630
  let finishBy = 0;
@@ -450,44 +632,67 @@ export async function runCi(options, resolved, deps) {
450
632
  try {
451
633
  // The page-load limit may be longer than the usual attach budget; the attach gets that limit and a minute to launch.
452
634
  const attachMs = Math.max(ATTACH_MS, resolveTimeLimits(options, process.env).navMs + 60_000);
453
- host = await startServer(log, judgeAsk ? judgeHandler({ ask: judgeAsk, model: resolved.model, spend, caps: options.caps, calls: judgeCalls, secrets, now }) : undefined);
454
- const attached = await host.call("scout_attach", {
455
- url: options.url,
635
+ // One judge handler for the run's client: it answers the server's dedup questions and adds their tokens to the run's budget.
636
+ const judge = judgeAsk
637
+ ? judgeHandler({ ask: judgeAsk, model: resolved.model, spend: budget, caps: options.caps, calls: judgeCalls, secrets, now })
638
+ : undefined;
639
+ host = await (deps.startHost ?? startServer)(log, judge);
640
+ // What every session of the run attaches with: the planner's here, and each lane's when the run is split.
641
+ const attachArgs = (a) => ({
642
+ url: a.url,
456
643
  projectPath: options.projectDir,
457
644
  mode: options.mode,
458
- // Named either way, so a SCENESCOUT_DEDUP in the job's environment never decides for the option.
645
+ // Named either way, so a SCENESCOUT_DEDUP in the job's environment never decides for the option. Every session alike: they share one memory.
459
646
  dedup: judgeAsk ? "judge" : "rule",
460
- objective: `CI run: explore at level ${options.level}${options.focus ? `, focusing on ${options.focus}` : ""}`.slice(0, 300),
461
- task: "Starting the CI run",
647
+ objective: a.objective.slice(0, 300),
648
+ task: a.task,
649
+ ...(a.session ? { session: a.session } : {}),
462
650
  ...(options.storageStatePath ? { storageStatePath: options.storageStatePath } : {}),
463
651
  ...(options.browser ? { browser: options.browser } : {}),
464
652
  ...(options.actionTimeoutMs !== undefined ? { actionTimeoutMs: options.actionTimeoutMs } : {}),
465
653
  ...(options.navTimeoutMs !== undefined ? { navTimeoutMs: options.navTimeoutMs } : {}),
466
- }, Math.min(attachMs, options.caps.wallMs));
467
- const authFailed = attached.text.split("\n").find((l) => l.startsWith("⚠ AUTH FAILED"));
468
- if (attached.isError || /^ERROR:/.test(attached.text) || authFailed) {
469
- outcome = { ...outcome, stopDetail: (authFailed ?? attached.text).replace(/^ERROR:\s*/, "").slice(0, 400) };
654
+ });
655
+ const attached = await host.call("scout_attach", attachArgs({
656
+ url: options.url,
657
+ objective: `CI run: explore at level ${options.level}${options.focus ? `, focusing on ${options.focus}` : ""}`,
658
+ task: "Starting the CI run",
659
+ }), Math.min(attachMs, options.caps.wallMs));
660
+ const failed = attachFailure(attached);
661
+ if (failed !== null) {
662
+ outcome = { ...outcome, stopDetail: failed.slice(0, 400) };
470
663
  }
471
664
  else {
472
665
  log(`Attached to ${options.url} in ${options.mode} mode.`);
473
- const tools = ciTools(await host.tools(), options.show ? CAPTURE_TOOLS : undefined);
474
- const system = options.show ? ciCaptureSystemPrompt() : ciSystemPrompt(loadPlaybook(packageRoot), options);
475
- const kickoff = options.show ? ciCaptureKickoff({ url: options.url, show: options.show }) : ciKickoff(options);
666
+ const listed = await host.tools();
667
+ const toolHost = host;
476
668
  let captured = null;
477
- outcome = await agentLoop({
478
- client: deps.makeClient(system, tools, kickoff),
479
- host,
480
- tools,
481
- caps: options.caps,
482
- log,
483
- now,
484
- spend,
485
- projectDir: options.projectDir,
486
- onResult: (name, _args, r) => {
487
- if (name === "scout_capture" && !r.isError)
488
- captured = parseCaptureResult(r.text) ?? captured;
489
- },
490
- });
669
+ const oneLoop = () => {
670
+ const tools = ciTools(listed, options.show ? CAPTURE_TOOLS : undefined);
671
+ const system = options.show ? ciCaptureSystemPrompt() : ciSystemPrompt(loadPlaybook(packageRoot), options);
672
+ const kickoff = options.show ? ciCaptureKickoff({ url: options.url, show: options.show }) : ciKickoff(options);
673
+ return agentLoop({
674
+ client: deps.makeClient(system, tools, kickoff),
675
+ host: toolHost,
676
+ tools,
677
+ caps: options.caps,
678
+ log,
679
+ now,
680
+ budget,
681
+ projectDir: options.projectDir,
682
+ onResult: (name, _args, r) => {
683
+ if (name === "scout_capture" && !r.isError)
684
+ captured = parseCaptureResult(r.text) ?? captured;
685
+ },
686
+ });
687
+ };
688
+ if (options.lanes > 1 && !options.show) {
689
+ const split = await exploreInLanes({ host, listed, options, makeClient: deps.makeClient, log, now, budget, attachArgs, attachMs });
690
+ lanes = split.lanes;
691
+ outcome = split.outcome ?? (await oneLoop());
692
+ }
693
+ else {
694
+ outcome = await oneLoop();
695
+ }
491
696
  log(`Run ended: ${describeStop(outcome.stop, options.caps, outcome.stopDetail)}.`);
492
697
  finishBy = now() + FINISH_MS;
493
698
  if (options.show) {
@@ -496,10 +701,11 @@ export async function runCi(options, resolved, deps) {
496
701
  }
497
702
  else {
498
703
  // The report is written whatever ended the run. A forced report still prints its gaps.
499
- let report = await host.call("scout_report", { level: options.level }, finishLeft());
704
+ // From the planner's session by name: a lane attaching made itself the server's default.
705
+ let report = await host.call("scout_report", { level: options.level, session: PLANNER_SESSION }, finishLeft());
500
706
  contractMet = !report.isError && !/NOT GENERATED/.test(report.text);
501
707
  if (!contractMet && !report.isError)
502
- report = await host.call("scout_report", { level: options.level, force: true }, finishLeft());
708
+ report = await host.call("scout_report", { level: options.level, force: true, session: PLANNER_SESSION }, finishLeft());
503
709
  reportWritten = !report.isError && !/^ERROR:/.test(report.text) && fs.existsSync(path.join(options.projectDir, MEMORY_DIRNAME, "report.md"));
504
710
  if (!reportWritten)
505
711
  log(`The report could not be generated: ${report.text.slice(0, 400)}`);
@@ -536,10 +742,12 @@ export async function runCi(options, resolved, deps) {
536
742
  stop: outcome.stop,
537
743
  ...(outcome.stopDetail ? { stopDetail: outcome.stopDetail } : {}),
538
744
  contractMet,
539
- spend: outcome.spend,
745
+ // The run's, not the last loop's: every lane's turns, and the judge's tokens.
746
+ spend: budgetSpend(budget),
540
747
  endedAt,
541
748
  findings: findingsThisRun(before, readMemoryFindings(options.projectDir)),
542
749
  ...(capture ? { capture } : {}),
750
+ ...(lanes ? { lanes } : {}),
543
751
  ...(options.show
544
752
  ? {}
545
753
  : {
@@ -564,7 +772,15 @@ export async function runCi(options, resolved, deps) {
564
772
  const summary = ciSummaryMarkdown(result, secrets);
565
773
  write("summary.md", summary);
566
774
  write("ci.json", JSON.stringify(ciSummaryJson(result, deps.version, secrets), null, 2) + "\n");
567
- write("ci.sarif", JSON.stringify(ciSarif(result, deps.version, secrets), null, 2) + "\n");
775
+ const { anchor, warning } = sarifFilesFor({
776
+ option: options.sarifFileAnchor,
777
+ env: process.env,
778
+ projectDir: options.projectDir,
779
+ exists: (p) => fs.existsSync(p),
780
+ });
781
+ if (warning)
782
+ log(warning);
783
+ write("ci.sarif", JSON.stringify(ciSarif(result, deps.version, secrets, anchor), null, 2) + "\n");
568
784
  // A capture run's outcome is its pictures and ci.json: it wrote no report by design.
569
785
  if (options.show)
570
786
  reportWritten = true;