scenescout 3.15.0 → 3.17.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +87 -0
- package/README.md +70 -18
- package/dist/browsers.js +28 -0
- package/dist/check-run.js +191 -14
- package/dist/ci-run.js +268 -52
- package/dist/cli.js +107 -47
- package/dist/commands.js +3 -2
- package/dist/engine/baseline.js +377 -0
- package/dist/engine/brief.js +16 -7
- package/dist/engine/browser.js +1147 -286
- package/dist/engine/calibration.js +61 -30
- package/dist/engine/capture.js +164 -0
- package/dist/engine/check.js +244 -42
- package/dist/engine/ci-lanes.js +215 -0
- package/dist/engine/ci.js +136 -18
- package/dist/engine/claims.js +159 -3
- package/dist/engine/collector.js +561 -30
- package/dist/engine/crawl.js +49 -0
- package/dist/engine/design.js +281 -38
- package/dist/engine/export.js +877 -0
- package/dist/engine/fingerprint.js +92 -4
- package/dist/engine/flow.js +18 -6
- package/dist/engine/forms.js +181 -18
- package/dist/engine/journey.js +29 -1
- package/dist/engine/lane.js +13 -3
- package/dist/engine/launch.js +45 -6
- package/dist/engine/limits.js +7 -0
- package/dist/engine/live-page.js +49 -2
- package/dist/engine/live.js +4 -1
- package/dist/engine/memory.js +501 -47
- package/dist/engine/open.js +118 -0
- package/dist/engine/oracles.js +41 -1
- package/dist/engine/plain.js +268 -0
- package/dist/engine/png.js +127 -0
- package/dist/engine/policy.js +379 -9
- package/dist/engine/probes.js +3 -2
- package/dist/engine/profiles.js +45 -9
- package/dist/engine/project-folder.js +191 -0
- package/dist/engine/refresh.js +68 -3
- package/dist/engine/replay.js +63 -10
- package/dist/engine/report.js +241 -40
- package/dist/engine/request.js +317 -23
- package/dist/engine/sarif.js +120 -0
- package/dist/engine/settle.js +67 -0
- package/dist/engine/signed-in.js +256 -0
- package/dist/engine/status-pane-page.js +441 -0
- package/dist/engine/status-pane.js +128 -0
- package/dist/engine/tickets.js +671 -0
- package/dist/engine/unload.js +3 -2
- package/dist/export-run.js +633 -0
- package/dist/first-run.js +5 -0
- package/dist/installer.js +378 -8
- package/dist/intake.js +104 -0
- package/dist/login-run.js +250 -36
- package/dist/mcp-server.js +660 -65
- package/dist/playbook.js +5 -0
- package/dist/prompts.js +106 -0
- package/package.json +8 -5
- package/skills/scenescout/SKILL.md +49 -16
package/dist/ci-run.js
CHANGED
|
@@ -1,8 +1,9 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* Runs `scenescout ci`: starts the SceneScout MCP server as a child process,
|
|
3
3
|
* attaches it to the app, lets a model drive the scout_* tools until it is
|
|
4
|
-
* done or a cap ends the run
|
|
5
|
-
*
|
|
4
|
+
* done or a cap ends the run (with --lanes, several conversations with the
|
|
5
|
+
* model at once, one per part of the app, each in its own session), then has
|
|
6
|
+
* the report written and writes the CI files. The rules (options, caps, redaction, which tools, the files) are in
|
|
6
7
|
* engine/ci.ts and the message shapes in engine/provider.ts; this file only
|
|
7
8
|
* moves bytes between the model, the server and the disk.
|
|
8
9
|
*
|
|
@@ -20,9 +21,11 @@ import { StdioClientTransport } from "@modelcontextprotocol/sdk/client/stdio.js"
|
|
|
20
21
|
import { CreateMessageRequestSchema, ErrorCode, McpError } from "@modelcontextprotocol/sdk/types.js";
|
|
21
22
|
import { DEDUP_JUDGE_CAPABILITY, durationText, JUDGE_CALL_MS, JUDGE_MAX_OUTPUT_TOKENS, JUDGE_SYSTEM, JUDGE_TOOL, judgeKickoffOf, samplingResultOf, } from "./engine/dedup.js";
|
|
22
23
|
import { CAPTURE_MARGIN, parseCaptureResult, rebaseUrl, SHOT_FILES, SHOTS_DIRNAME } from "./engine/capture.js";
|
|
23
|
-
import { addUsage, wallLeftMs, guardToolArgs,
|
|
24
|
+
import { addUsage, attachFailure, budgetSpend, wallLeftMs, guardToolArgs, CAPTURE_TOOLS, childEnv, ciCaptureKickoff, ciCaptureSystemPrompt, CI_DIRNAME, ciExitCode, ciKickoff, ciSarif, ciSummaryJson, ciSummaryMarkdown, ciSystemPrompt, ciToolArgs, ciTools, describeStop, findingsThisRun, newBudget, NO_USAGE, readFindings, redactKeys, settleTurn, takeTurn, toolResultText, usageLine, } from "./engine/ci.js";
|
|
25
|
+
import { ciLaneKickoff, ciLaneSystemPrompt, crawlFoundNothing, crawlNotes, LANE_TOOLS, mergeLaneStops, PLAN_CRAWL_ROUNDS, planCiLanes, PLANNER_SESSION, } from "./engine/ci-lanes.js";
|
|
24
26
|
import { resolveTimeLimits } from "./engine/limits.js";
|
|
25
27
|
import { MEMORY_DIRNAME, writeSelfIgnore } from "./engine/memory.js";
|
|
28
|
+
import { sarifFilesFor } from "./engine/sarif.js";
|
|
26
29
|
import { decodePng, diffImages, encodePng } from "./engine/png.js";
|
|
27
30
|
import { AnthropicConversation, backoffMs, errorMessage, MalformedReply, OpenAIConversation, retryable, retryAfterMs, } from "./engine/provider.js";
|
|
28
31
|
import { loadPlaybook } from "./playbook.js";
|
|
@@ -253,31 +256,37 @@ function readMemoryFindings(projectDir) {
|
|
|
253
256
|
}
|
|
254
257
|
}
|
|
255
258
|
/**
|
|
256
|
-
* The agent loop. Before each model call
|
|
257
|
-
*
|
|
258
|
-
*
|
|
259
|
-
*
|
|
260
|
-
*
|
|
261
|
-
*
|
|
262
|
-
*
|
|
259
|
+
* The agent loop. Before each model call a turn is taken from the budget, or
|
|
260
|
+
* the cap that refuses it ends the loop; the call runs with at most the time
|
|
261
|
+
* that is left; each tool call it asks for is run in order, on the loop's one
|
|
262
|
+
* session, and none runs past the time cap: a call reached after it is
|
|
263
|
+
* answered as not run, as is any beyond MAX_TOOL_CALLS_PER_TURN. A call the
|
|
264
|
+
* loop cannot run (a tool it was not given, arguments that are not JSON, a
|
|
265
|
+
* scan of another directory) goes back to the model as an error result, so a
|
|
266
|
+
* malformed reply costs a turn, never the run.
|
|
263
267
|
*/
|
|
264
268
|
export async function agentLoop(o) {
|
|
265
269
|
const now = o.now ?? Date.now;
|
|
266
|
-
const
|
|
270
|
+
const budget = o.budget ?? newBudget(o.caps, o.startedAt ?? now());
|
|
271
|
+
// This loop's own turns and tokens; the budget holds every loop's.
|
|
272
|
+
const spend = { turns: 0, usage: { ...NO_USAGE }, startedAt: budget.startedAt };
|
|
273
|
+
const timeLeft = () => wallLeftMs(spend, budget.caps, now());
|
|
267
274
|
const allowed = new Set(o.tools.map((t) => t.name));
|
|
268
275
|
for (;;) {
|
|
269
|
-
const cap =
|
|
276
|
+
const cap = takeTurn(budget, now());
|
|
270
277
|
if (cap)
|
|
271
278
|
return { stop: cap, spend };
|
|
272
279
|
let turn;
|
|
273
280
|
try {
|
|
274
|
-
turn = await o.client.next(
|
|
281
|
+
turn = await o.client.next(timeLeft());
|
|
275
282
|
}
|
|
276
283
|
catch (err) {
|
|
284
|
+
settleTurn(budget);
|
|
277
285
|
if (err instanceof OutOfTime)
|
|
278
286
|
return { stop: "time", spend };
|
|
279
287
|
return { stop: "provider-error", stopDetail: err instanceof Error ? err.message : String(err), spend };
|
|
280
288
|
}
|
|
289
|
+
settleTurn(budget, turn.usage);
|
|
281
290
|
spend.turns += 1;
|
|
282
291
|
spend.usage = addUsage(spend.usage, turn.usage);
|
|
283
292
|
if (turn.resume)
|
|
@@ -309,15 +318,16 @@ export async function agentLoop(o) {
|
|
|
309
318
|
results.push({ id: call.id, isError: true, text: `${call.name} was not run: ${guarded.error}.` });
|
|
310
319
|
continue;
|
|
311
320
|
}
|
|
312
|
-
const left =
|
|
321
|
+
const left = timeLeft();
|
|
313
322
|
if (left <= 0) {
|
|
314
323
|
results.push({ id: call.id, isError: true, text: `${call.name} was not run: the time cap was reached.` });
|
|
315
324
|
continue;
|
|
316
325
|
}
|
|
326
|
+
const sent = o.session && o.session.tools.has(call.name) ? { ...guarded.args, session: o.session.name } : guarded.args;
|
|
317
327
|
try {
|
|
318
|
-
const r = await o.host.call(call.name,
|
|
328
|
+
const r = await o.host.call(call.name, sent, Math.min(600_000, left));
|
|
319
329
|
results.push({ id: call.id, isError: r.isError, text: r.text });
|
|
320
|
-
o.onResult?.(call.name,
|
|
330
|
+
o.onResult?.(call.name, sent, r);
|
|
321
331
|
}
|
|
322
332
|
catch (err) {
|
|
323
333
|
results.push({ id: call.id, isError: true, text: `${call.name} failed: ${err instanceof Error ? err.message : String(err)}` });
|
|
@@ -417,10 +427,181 @@ export async function captureShots(o) {
|
|
|
417
427
|
return { ...outcome, detail: `the base URL could not be captured: ${err instanceof Error ? err.message : String(err)}` };
|
|
418
428
|
}
|
|
419
429
|
}
|
|
430
|
+
/** Whether a listed tool takes `session`: a lane's calls to it are sent to the lane's own. */
|
|
431
|
+
function takesSession(t) {
|
|
432
|
+
const props = t.inputSchema?.properties;
|
|
433
|
+
return !!props && "session" in props;
|
|
434
|
+
}
|
|
435
|
+
const messageOf = (err) => (err instanceof Error ? err.message : String(err));
|
|
436
|
+
/**
|
|
437
|
+
* A run split into lanes (--lanes): plan, then run every lane at once, then
|
|
438
|
+
* fold what they did. The plan is a snapshot and a crawl from the planner's
|
|
439
|
+
* session (no model call: only their time counts), the crawl repeated while it
|
|
440
|
+
* finds routes, and the split brief.ts makes of them. Each lane attaches its
|
|
441
|
+
* own session on the run's target URL, as the planner did (the engine resolves
|
|
442
|
+
* every path against the URL a session attached with), opens its first route,
|
|
443
|
+
* runs the agent loop there with its own conversation, draws its turns from
|
|
444
|
+
* the run's one budget, and is closed when it ends. Their findings are already
|
|
445
|
+
* one: every session files into the project's one memory, whose dedup folds a
|
|
446
|
+
* defect two lanes filed. `outcome` is absent when there was nothing to split,
|
|
447
|
+
* and the run explores in one loop instead.
|
|
448
|
+
*/
|
|
449
|
+
async function exploreInLanes(o) {
|
|
450
|
+
const { host, options, log, now, budget } = o;
|
|
451
|
+
const timeLeft = () => wallLeftMs(budgetSpend(budget), options.caps, now());
|
|
452
|
+
// ── plan ──
|
|
453
|
+
// What went wrong while planning, so a plan left with nothing to split says why rather than blaming the app.
|
|
454
|
+
let planningFailed;
|
|
455
|
+
/** One planner call: its text, or undefined after saying why it failed. */
|
|
456
|
+
const plannerCall = async (tool, maxMs, what) => {
|
|
457
|
+
try {
|
|
458
|
+
const r = await host.call(tool, { session: PLANNER_SESSION }, Math.min(maxMs, timeLeft()));
|
|
459
|
+
if (r.isError)
|
|
460
|
+
throw new Error(r.text.replace(/^ERROR:\s*/, ""));
|
|
461
|
+
return r.text;
|
|
462
|
+
}
|
|
463
|
+
catch (err) {
|
|
464
|
+
planningFailed = `the planning ${what} failed: ${messageOf(err).slice(0, 300)}`;
|
|
465
|
+
log(`Lanes: ${planningFailed}.`);
|
|
466
|
+
return undefined;
|
|
467
|
+
}
|
|
468
|
+
};
|
|
469
|
+
// Attaching harvests no links; a snapshot of the page it landed on does, so the first crawl has routes to visit.
|
|
470
|
+
if (timeLeft() > 0)
|
|
471
|
+
await plannerCall("scout_snapshot", 120_000, "snapshot");
|
|
472
|
+
const notes = new Map();
|
|
473
|
+
for (let round = 0; round < PLAN_CRAWL_ROUNDS && timeLeft() > 0; round += 1) {
|
|
474
|
+
const text = await plannerCall("scout_crawl", 600_000, "crawl");
|
|
475
|
+
if (text === undefined)
|
|
476
|
+
break;
|
|
477
|
+
const known = notes.size;
|
|
478
|
+
for (const [route, lines] of crawlNotes(text))
|
|
479
|
+
if (!notes.has(route))
|
|
480
|
+
notes.set(route, lines);
|
|
481
|
+
if (crawlFoundNothing(text) || notes.size === known)
|
|
482
|
+
break;
|
|
483
|
+
}
|
|
484
|
+
const plan = planCiLanes({
|
|
485
|
+
target: options.url,
|
|
486
|
+
notes,
|
|
487
|
+
count: options.lanes,
|
|
488
|
+
focus: options.focus,
|
|
489
|
+
mode: options.mode,
|
|
490
|
+
...(planningFailed ? { planningFailed } : {}),
|
|
491
|
+
});
|
|
492
|
+
if (plan.oneLoop) {
|
|
493
|
+
log(`Lanes: ${plan.oneLoop}. Exploring in one loop.`);
|
|
494
|
+
return { lanes: { asked: options.lanes, sessions: [], oneLoop: plan.oneLoop } };
|
|
495
|
+
}
|
|
496
|
+
log(`Lanes: ${plan.lanes.length} of ${options.lanes} asked, sharing the caps: ` +
|
|
497
|
+
plan.lanes.map((l) => `${l.session} (${l.modules.join(", ")}; ${l.routes.length} route(s))`).join("; "));
|
|
498
|
+
// ── run ──
|
|
499
|
+
const tools = ciTools(o.listed, LANE_TOOLS);
|
|
500
|
+
const sessionTools = new Set(o.listed.filter(takesSession).map((t) => t.name));
|
|
501
|
+
const system = ciLaneSystemPrompt(loadPlaybook(packageRoot), options);
|
|
502
|
+
const runLane = async (lane) => {
|
|
503
|
+
const say = (line) => log(line.replace(/^(\s*)/, `$1[${lane.session}] `));
|
|
504
|
+
const result = {
|
|
505
|
+
session: lane.session,
|
|
506
|
+
modules: lane.modules,
|
|
507
|
+
routes: lane.routes.length,
|
|
508
|
+
attached: false,
|
|
509
|
+
stop: "could-not-start",
|
|
510
|
+
turns: 0,
|
|
511
|
+
usage: { ...NO_USAGE },
|
|
512
|
+
};
|
|
513
|
+
let attached = false;
|
|
514
|
+
try {
|
|
515
|
+
if (timeLeft() <= 0) {
|
|
516
|
+
say("the time cap was reached before it attached.");
|
|
517
|
+
return { ...result, stop: "time", stopDetail: "the time cap was reached before it attached" };
|
|
518
|
+
}
|
|
519
|
+
const r = await host
|
|
520
|
+
.call("scout_attach", o.attachArgs({ session: lane.session, url: options.url, objective: lane.objective, task: `Starting lane ${lane.session}` }), Math.min(o.attachMs, timeLeft()))
|
|
521
|
+
.catch((err) => ({ text: `ERROR: ${messageOf(err)}`, isError: true }));
|
|
522
|
+
const failed = attachFailure(r);
|
|
523
|
+
if (failed !== null) {
|
|
524
|
+
say(`could not attach: ${failed.slice(0, 300)}`);
|
|
525
|
+
return { ...result, stopDetail: failed.slice(0, 300) };
|
|
526
|
+
}
|
|
527
|
+
attached = true;
|
|
528
|
+
// Its own first route, by its full URL. A landing that does not open costs the lane nothing but the detour: it starts from the target.
|
|
529
|
+
let on = options.url;
|
|
530
|
+
if (lane.url !== options.url && timeLeft() > 0) {
|
|
531
|
+
const opened = await host
|
|
532
|
+
.call("scout_navigate", { session: lane.session, target: lane.url, task: `Opening lane ${lane.session}'s first route` }, Math.min(o.attachMs, timeLeft()))
|
|
533
|
+
.catch((err) => ({ text: `ERROR: ${messageOf(err)}`, isError: true }));
|
|
534
|
+
// With several sessions live the server puts a "[session …]" line first; the verdict is on the line after it.
|
|
535
|
+
const said = opened.text.replace(/^\[session [^\]\n]*\]\n/, "");
|
|
536
|
+
if (opened.isError || /^(ERROR|REFUSED):/.test(said))
|
|
537
|
+
say(`could not open ${lane.landing} (${said.replace(/^ERROR:\s*/, "").slice(0, 200)}); starting from the target instead.`);
|
|
538
|
+
else
|
|
539
|
+
on = lane.url;
|
|
540
|
+
}
|
|
541
|
+
say(`attached on ${new URL(on).pathname}, owning ${lane.modules.join(", ")}.`);
|
|
542
|
+
return await exploreLane(lane, on, say, { ...result, attached: true });
|
|
543
|
+
}
|
|
544
|
+
catch (err) {
|
|
545
|
+
// Not a cap and not the model's API: the lane itself broke. Reported as the lane's, and the run's (mergeLaneStops), never dropped.
|
|
546
|
+
const why = `the lane failed: ${messageOf(err).slice(0, 300)}`;
|
|
547
|
+
say(why);
|
|
548
|
+
return { ...result, attached, stop: "could-not-start", stopDetail: why };
|
|
549
|
+
}
|
|
550
|
+
};
|
|
551
|
+
const exploreLane = async (lane, on, say, result) => {
|
|
552
|
+
const outcome = await agentLoop({
|
|
553
|
+
client: o.makeClient(system, tools, ciLaneKickoff({
|
|
554
|
+
lane: { ...lane, url: on },
|
|
555
|
+
laneCount: plan.lanes.length,
|
|
556
|
+
url: options.url,
|
|
557
|
+
projectDir: options.projectDir,
|
|
558
|
+
mode: options.mode,
|
|
559
|
+
level: options.level,
|
|
560
|
+
focus: options.focus,
|
|
561
|
+
caps: options.caps,
|
|
562
|
+
})),
|
|
563
|
+
host,
|
|
564
|
+
tools,
|
|
565
|
+
caps: options.caps,
|
|
566
|
+
budget,
|
|
567
|
+
log: say,
|
|
568
|
+
now,
|
|
569
|
+
projectDir: options.projectDir,
|
|
570
|
+
session: { name: lane.session, tools: sessionTools },
|
|
571
|
+
});
|
|
572
|
+
say(`ended: ${outcome.stop === "done" ? "the model finished the lane" : describeStop(outcome.stop, options.caps, outcome.stopDetail)}.`);
|
|
573
|
+
// Closed as soon as it is done, so a lane that finished early holds no browser while the others work.
|
|
574
|
+
// Past the time cap nothing more runs here: the run's own close, within FINISH_MS, collects it.
|
|
575
|
+
if (timeLeft() > 0)
|
|
576
|
+
await host
|
|
577
|
+
.call("scout_close", { session: lane.session }, Math.min(30_000, timeLeft()))
|
|
578
|
+
.catch((err) => say(`closing its browser failed: ${messageOf(err)}`));
|
|
579
|
+
return {
|
|
580
|
+
...result,
|
|
581
|
+
stop: outcome.stop,
|
|
582
|
+
...(outcome.stopDetail ? { stopDetail: outcome.stopDetail } : {}),
|
|
583
|
+
turns: outcome.spend.turns,
|
|
584
|
+
usage: outcome.spend.usage,
|
|
585
|
+
};
|
|
586
|
+
};
|
|
587
|
+
// ── merge ──
|
|
588
|
+
// Settled, not raced: every lane is waited out before the run moves on. runLane reports its own failures, so none rejects.
|
|
589
|
+
const settled = await Promise.allSettled(plan.lanes.map(runLane));
|
|
590
|
+
const broken = settled.find((s) => s.status === "rejected");
|
|
591
|
+
if (broken)
|
|
592
|
+
throw new Error(`a lane failed outside its own handling: ${messageOf(broken.reason)}`);
|
|
593
|
+
const sessions = settled.map((s) => s.value);
|
|
594
|
+
const merged = mergeLaneStops(sessions);
|
|
595
|
+
return {
|
|
596
|
+
lanes: { asked: options.lanes, sessions },
|
|
597
|
+
outcome: { stop: merged.stop, ...(merged.stopDetail ? { stopDetail: merged.stopDetail } : {}), spend: budgetSpend(budget) },
|
|
598
|
+
};
|
|
599
|
+
}
|
|
420
600
|
/**
|
|
421
601
|
* The whole run. `makeClient` is how the model is reached: the HTTP client in
|
|
422
|
-
* the CLI, a scripted one in the tests.
|
|
423
|
-
* through the key
|
|
602
|
+
* the CLI, a scripted one in the tests. It is called once per conversation:
|
|
603
|
+
* once, or once per lane. Everything logged or written passes through the key
|
|
604
|
+
* redaction first.
|
|
424
605
|
*/
|
|
425
606
|
export async function runCi(options, resolved, deps) {
|
|
426
607
|
const secrets = deps.secrets ?? [];
|
|
@@ -431,18 +612,19 @@ export async function runCi(options, resolved, deps) {
|
|
|
431
612
|
const before = readMemoryFindings(options.projectDir);
|
|
432
613
|
// Pictures an earlier run left in the same output are not this run's: they must never be uploaded as its.
|
|
433
614
|
fs.rmSync(path.join(outDir, SHOTS_DIRNAME), { recursive: true, force: true });
|
|
434
|
-
// One
|
|
435
|
-
const
|
|
615
|
+
// One budget for the run: every loop (one, or each lane) takes its turns from it, and the dedup judge's calls add their tokens, which the caps count.
|
|
616
|
+
const budget = newBudget(options.caps, startedAt);
|
|
436
617
|
// A run asked to show an element files no findings, so it has nothing to deduplicate.
|
|
437
618
|
const wantsJudge = options.dedup === "judge" && !options.show;
|
|
438
619
|
if (wantsJudge && !deps.judge)
|
|
439
620
|
log("No model was given for the dedup judge; the rule decides duplicates.");
|
|
440
621
|
const judgeAsk = wantsJudge ? deps.judge : undefined;
|
|
441
622
|
const judgeCalls = { calls: 0, failed: 0, usage: { ...NO_USAGE }, ms: 0 };
|
|
442
|
-
let outcome = { stop: "could-not-start", spend };
|
|
623
|
+
let outcome = { stop: "could-not-start", spend: budgetSpend(budget) };
|
|
443
624
|
let contractMet = false;
|
|
444
625
|
let reportWritten = false;
|
|
445
626
|
let capture;
|
|
627
|
+
let lanes;
|
|
446
628
|
let host = null;
|
|
447
629
|
// Set when the exploration ends (or never starts): the report and the close share FINISH_MS from then.
|
|
448
630
|
let finishBy = 0;
|
|
@@ -450,44 +632,67 @@ export async function runCi(options, resolved, deps) {
|
|
|
450
632
|
try {
|
|
451
633
|
// The page-load limit may be longer than the usual attach budget; the attach gets that limit and a minute to launch.
|
|
452
634
|
const attachMs = Math.max(ATTACH_MS, resolveTimeLimits(options, process.env).navMs + 60_000);
|
|
453
|
-
|
|
454
|
-
const
|
|
455
|
-
|
|
635
|
+
// One judge handler for the run's client: it answers the server's dedup questions and adds their tokens to the run's budget.
|
|
636
|
+
const judge = judgeAsk
|
|
637
|
+
? judgeHandler({ ask: judgeAsk, model: resolved.model, spend: budget, caps: options.caps, calls: judgeCalls, secrets, now })
|
|
638
|
+
: undefined;
|
|
639
|
+
host = await (deps.startHost ?? startServer)(log, judge);
|
|
640
|
+
// What every session of the run attaches with: the planner's here, and each lane's when the run is split.
|
|
641
|
+
const attachArgs = (a) => ({
|
|
642
|
+
url: a.url,
|
|
456
643
|
projectPath: options.projectDir,
|
|
457
644
|
mode: options.mode,
|
|
458
|
-
// Named either way, so a SCENESCOUT_DEDUP in the job's environment never decides for the option.
|
|
645
|
+
// Named either way, so a SCENESCOUT_DEDUP in the job's environment never decides for the option. Every session alike: they share one memory.
|
|
459
646
|
dedup: judgeAsk ? "judge" : "rule",
|
|
460
|
-
objective:
|
|
461
|
-
task:
|
|
647
|
+
objective: a.objective.slice(0, 300),
|
|
648
|
+
task: a.task,
|
|
649
|
+
...(a.session ? { session: a.session } : {}),
|
|
462
650
|
...(options.storageStatePath ? { storageStatePath: options.storageStatePath } : {}),
|
|
463
651
|
...(options.browser ? { browser: options.browser } : {}),
|
|
464
652
|
...(options.actionTimeoutMs !== undefined ? { actionTimeoutMs: options.actionTimeoutMs } : {}),
|
|
465
653
|
...(options.navTimeoutMs !== undefined ? { navTimeoutMs: options.navTimeoutMs } : {}),
|
|
466
|
-
}
|
|
467
|
-
const
|
|
468
|
-
|
|
469
|
-
|
|
654
|
+
});
|
|
655
|
+
const attached = await host.call("scout_attach", attachArgs({
|
|
656
|
+
url: options.url,
|
|
657
|
+
objective: `CI run: explore at level ${options.level}${options.focus ? `, focusing on ${options.focus}` : ""}`,
|
|
658
|
+
task: "Starting the CI run",
|
|
659
|
+
}), Math.min(attachMs, options.caps.wallMs));
|
|
660
|
+
const failed = attachFailure(attached);
|
|
661
|
+
if (failed !== null) {
|
|
662
|
+
outcome = { ...outcome, stopDetail: failed.slice(0, 400) };
|
|
470
663
|
}
|
|
471
664
|
else {
|
|
472
665
|
log(`Attached to ${options.url} in ${options.mode} mode.`);
|
|
473
|
-
const
|
|
474
|
-
const
|
|
475
|
-
const kickoff = options.show ? ciCaptureKickoff({ url: options.url, show: options.show }) : ciKickoff(options);
|
|
666
|
+
const listed = await host.tools();
|
|
667
|
+
const toolHost = host;
|
|
476
668
|
let captured = null;
|
|
477
|
-
|
|
478
|
-
|
|
479
|
-
|
|
480
|
-
|
|
481
|
-
|
|
482
|
-
|
|
483
|
-
|
|
484
|
-
|
|
485
|
-
|
|
486
|
-
|
|
487
|
-
|
|
488
|
-
|
|
489
|
-
|
|
490
|
-
|
|
669
|
+
const oneLoop = () => {
|
|
670
|
+
const tools = ciTools(listed, options.show ? CAPTURE_TOOLS : undefined);
|
|
671
|
+
const system = options.show ? ciCaptureSystemPrompt() : ciSystemPrompt(loadPlaybook(packageRoot), options);
|
|
672
|
+
const kickoff = options.show ? ciCaptureKickoff({ url: options.url, show: options.show }) : ciKickoff(options);
|
|
673
|
+
return agentLoop({
|
|
674
|
+
client: deps.makeClient(system, tools, kickoff),
|
|
675
|
+
host: toolHost,
|
|
676
|
+
tools,
|
|
677
|
+
caps: options.caps,
|
|
678
|
+
log,
|
|
679
|
+
now,
|
|
680
|
+
budget,
|
|
681
|
+
projectDir: options.projectDir,
|
|
682
|
+
onResult: (name, _args, r) => {
|
|
683
|
+
if (name === "scout_capture" && !r.isError)
|
|
684
|
+
captured = parseCaptureResult(r.text) ?? captured;
|
|
685
|
+
},
|
|
686
|
+
});
|
|
687
|
+
};
|
|
688
|
+
if (options.lanes > 1 && !options.show) {
|
|
689
|
+
const split = await exploreInLanes({ host, listed, options, makeClient: deps.makeClient, log, now, budget, attachArgs, attachMs });
|
|
690
|
+
lanes = split.lanes;
|
|
691
|
+
outcome = split.outcome ?? (await oneLoop());
|
|
692
|
+
}
|
|
693
|
+
else {
|
|
694
|
+
outcome = await oneLoop();
|
|
695
|
+
}
|
|
491
696
|
log(`Run ended: ${describeStop(outcome.stop, options.caps, outcome.stopDetail)}.`);
|
|
492
697
|
finishBy = now() + FINISH_MS;
|
|
493
698
|
if (options.show) {
|
|
@@ -496,10 +701,11 @@ export async function runCi(options, resolved, deps) {
|
|
|
496
701
|
}
|
|
497
702
|
else {
|
|
498
703
|
// The report is written whatever ended the run. A forced report still prints its gaps.
|
|
499
|
-
|
|
704
|
+
// From the planner's session by name: a lane attaching made itself the server's default.
|
|
705
|
+
let report = await host.call("scout_report", { level: options.level, session: PLANNER_SESSION }, finishLeft());
|
|
500
706
|
contractMet = !report.isError && !/NOT GENERATED/.test(report.text);
|
|
501
707
|
if (!contractMet && !report.isError)
|
|
502
|
-
report = await host.call("scout_report", { level: options.level, force: true }, finishLeft());
|
|
708
|
+
report = await host.call("scout_report", { level: options.level, force: true, session: PLANNER_SESSION }, finishLeft());
|
|
503
709
|
reportWritten = !report.isError && !/^ERROR:/.test(report.text) && fs.existsSync(path.join(options.projectDir, MEMORY_DIRNAME, "report.md"));
|
|
504
710
|
if (!reportWritten)
|
|
505
711
|
log(`The report could not be generated: ${report.text.slice(0, 400)}`);
|
|
@@ -536,10 +742,12 @@ export async function runCi(options, resolved, deps) {
|
|
|
536
742
|
stop: outcome.stop,
|
|
537
743
|
...(outcome.stopDetail ? { stopDetail: outcome.stopDetail } : {}),
|
|
538
744
|
contractMet,
|
|
539
|
-
|
|
745
|
+
// The run's, not the last loop's: every lane's turns, and the judge's tokens.
|
|
746
|
+
spend: budgetSpend(budget),
|
|
540
747
|
endedAt,
|
|
541
748
|
findings: findingsThisRun(before, readMemoryFindings(options.projectDir)),
|
|
542
749
|
...(capture ? { capture } : {}),
|
|
750
|
+
...(lanes ? { lanes } : {}),
|
|
543
751
|
...(options.show
|
|
544
752
|
? {}
|
|
545
753
|
: {
|
|
@@ -564,7 +772,15 @@ export async function runCi(options, resolved, deps) {
|
|
|
564
772
|
const summary = ciSummaryMarkdown(result, secrets);
|
|
565
773
|
write("summary.md", summary);
|
|
566
774
|
write("ci.json", JSON.stringify(ciSummaryJson(result, deps.version, secrets), null, 2) + "\n");
|
|
567
|
-
|
|
775
|
+
const { anchor, warning } = sarifFilesFor({
|
|
776
|
+
option: options.sarifFileAnchor,
|
|
777
|
+
env: process.env,
|
|
778
|
+
projectDir: options.projectDir,
|
|
779
|
+
exists: (p) => fs.existsSync(p),
|
|
780
|
+
});
|
|
781
|
+
if (warning)
|
|
782
|
+
log(warning);
|
|
783
|
+
write("ci.sarif", JSON.stringify(ciSarif(result, deps.version, secrets, anchor), null, 2) + "\n");
|
|
568
784
|
// A capture run's outcome is its pictures and ci.json: it wrote no report by design.
|
|
569
785
|
if (options.show)
|
|
570
786
|
reportWritten = true;
|