@mindstudio-ai/remy 0.1.254 → 0.1.256

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.js CHANGED
@@ -2238,6 +2238,9 @@ function findLastSummaryCheckpoint(messages, name) {
2238
2238
  }
2239
2239
  return -1;
2240
2240
  }
2241
+ function portableToolCallId(id) {
2242
+ return id.replace(/[^a-zA-Z0-9_-]/g, "_");
2243
+ }
2241
2244
  function fixOrphanedToolCalls(messages) {
2242
2245
  const toolResultIds = /* @__PURE__ */ new Set();
2243
2246
  for (const msg of messages) {
@@ -2329,14 +2332,30 @@ ${summaryBlock.text}
2329
2332
 
2330
2333
  ${content}` : attachmentHeader;
2331
2334
  }
2332
- return { ...rest, content };
2335
+ return {
2336
+ ...rest,
2337
+ content,
2338
+ ...msg.toolCallId && {
2339
+ toolCallId: portableToolCallId(msg.toolCallId)
2340
+ }
2341
+ };
2333
2342
  }
2334
2343
  if (!Array.isArray(msg.content)) {
2335
2344
  return msg;
2336
2345
  }
2337
2346
  const blocks = msg.content;
2338
2347
  const text = blocks.filter((b) => b.type === "text").map((b) => b.text).join("");
2339
- const toolCalls = blocks.filter((b) => b.type === "tool").map((b) => ({ id: b.id, name: b.name, input: b.input }));
2348
+ const toolBlocks = blocks.filter(
2349
+ (b) => b.type === "tool"
2350
+ );
2351
+ const toolCalls = toolBlocks.map((b) => ({
2352
+ id: portableToolCallId(b.id),
2353
+ name: b.name,
2354
+ input: b.input
2355
+ }));
2356
+ const rewroteToolIds = toolBlocks.some(
2357
+ (b) => portableToolCallId(b.id) !== b.id
2358
+ );
2340
2359
  const cleaned2 = {
2341
2360
  role: msg.role,
2342
2361
  content: text
@@ -2344,7 +2363,7 @@ ${content}` : attachmentHeader;
2344
2363
  if (toolCalls.length > 0) {
2345
2364
  cleaned2.toolCalls = toolCalls;
2346
2365
  }
2347
- if (msg.providerMetadata) {
2366
+ if (msg.providerMetadata && !rewroteToolIds) {
2348
2367
  cleaned2.providerMetadata = msg.providerMetadata;
2349
2368
  }
2350
2369
  if (msg.hidden) {
@@ -3599,6 +3618,7 @@ async function sidecarRequest(endpoint, body = {}, options) {
3599
3618
  throw new Error("Sidecar not available");
3600
3619
  }
3601
3620
  const url = `${baseUrl}${endpoint}`;
3621
+ let data;
3602
3622
  try {
3603
3623
  const res = await fetch(url, {
3604
3624
  method: "POST",
@@ -3610,12 +3630,7 @@ async function sidecarRequest(endpoint, body = {}, options) {
3610
3630
  log5.error("Sidecar error", { endpoint, status: res.status });
3611
3631
  throw new Error(`Sidecar error: ${res.status}`);
3612
3632
  }
3613
- const data = await res.json();
3614
- if (data?.success === false) {
3615
- const code = data.errorCode ? ` [${data.errorCode}]` : "";
3616
- throw new Error(`${data.error || "Unknown error"}${code}`);
3617
- }
3618
- return data;
3633
+ data = await res.json();
3619
3634
  } catch (err) {
3620
3635
  if (err.message.startsWith("Sidecar error")) {
3621
3636
  throw err;
@@ -3623,6 +3638,16 @@ async function sidecarRequest(endpoint, body = {}, options) {
3623
3638
  log5.error("Sidecar connection error", { endpoint, error: err.message });
3624
3639
  throw new Error(`Sidecar connection error: ${err.message}`);
3625
3640
  }
3641
+ if (data?.success === false) {
3642
+ const code = data.errorCode ? ` [${data.errorCode}]` : "";
3643
+ log5.error("Sidecar command failed", {
3644
+ endpoint,
3645
+ error: data.error,
3646
+ errorCode: data.errorCode
3647
+ });
3648
+ throw new Error(`${data.error || "Unknown error"}${code}`);
3649
+ }
3650
+ return data;
3626
3651
  }
3627
3652
  var log5, baseUrl;
3628
3653
  var init_sidecar = __esm({
@@ -3935,7 +3960,9 @@ async function captureAndAnalyzeScreenshot(promptOrOptions) {
3935
3960
  ...height != null ? { height } : {},
3936
3961
  ...format ? { format } : {}
3937
3962
  },
3938
- { timeout: fullPage ? 12e4 : 3e4 }
3963
+ {
3964
+ timeout: fullPage ? FULLPAGE_CAPTURE_TIMEOUT_MS : VIEWPORT_CAPTURE_TIMEOUT_MS
3965
+ }
3939
3966
  );
3940
3967
  url = ssResult?.url || ssResult?.screenshotUrl;
3941
3968
  if (!url) {
@@ -3961,12 +3988,14 @@ async function captureAndAnalyzeScreenshot(promptOrOptions) {
3961
3988
  model
3962
3989
  });
3963
3990
  }
3964
- var SCREENSHOT_ANALYSIS_PROMPT, ANALYSIS_RESPONSE_FORMAT;
3991
+ var VIEWPORT_CAPTURE_TIMEOUT_MS, FULLPAGE_CAPTURE_TIMEOUT_MS, SCREENSHOT_ANALYSIS_PROMPT, ANALYSIS_RESPONSE_FORMAT;
3965
3992
  var init_screenshot = __esm({
3966
3993
  "src/tools/_helpers/screenshot.ts"() {
3967
3994
  "use strict";
3968
3995
  init_sidecar();
3969
3996
  init_analyzeImage();
3997
+ VIEWPORT_CAPTURE_TIMEOUT_MS = 45e3;
3998
+ FULLPAGE_CAPTURE_TIMEOUT_MS = 135e3;
3970
3999
  SCREENSHOT_ANALYSIS_PROMPT = `Describe everything visible on screen from top to bottom \u2014 every element, its position, its size relative to the viewport, its colors, its content. Be comprehensive, thorough, and spatial. After the inventory, note anything that looks visually broken (overlapping elements, clipped text, misaligned components).`;
3971
4000
  ANALYSIS_RESPONSE_FORMAT = `Respond only with your analysis as Markdown and absolutely no other text. Do not use emojis - use unicode if you need symbols.`;
3972
4001
  }
@@ -4673,7 +4702,7 @@ var init_tools2 = __esm({
4673
4702
  "screenshotViewport",
4674
4703
  "setViewport"
4675
4704
  ],
4676
- description: 'snapshot: accessibility tree of the page (waits for network to settle). click: click an element (animated cursor, full event sequence). type: type text into input (one char at a time, works with React/Vue/Svelte). select: select a dropdown option by text. wait: wait for an element to appear (polls 100ms, waits for network). navigate: navigate to a URL within the app (waits for load, subsequent steps run on new page). evaluate: run JS in the page. styles: read computed CSS styles from elements (pass properties array with camelCase names, or omit for defaults). screenshotFullPage: full-page viewport-stitched screenshot (returns CDN url with dimensions). screenshotViewport: screenshot of just the visible viewport \u2014 pass `scrollToSelector` (or `scrollY`) on this step to scroll a section into view and capture it in one atomic step (no separate scroll needed). setViewport: switch the browser between desktop and mobile rendering (pass `mode`: "desktop" or "mobile"). Reloads the page so responsive layouts, media queries, and matchMedia re-evaluate \u2014 use it to QA mobile/responsive views.'
4705
+ description: 'snapshot: accessibility tree of the page (waits for network to settle). click: click an element (animated cursor, full event sequence). type: type text into input (one char at a time, works with React/Vue/Svelte). select: select a dropdown option by text. wait: wait for an element to appear (polls 100ms, waits for network). navigate: navigate to a URL within the app (waits for load, subsequent steps run on new page). evaluate: run JS in the page. styles: read computed CSS styles from elements (pass properties array with camelCase names, or omit for defaults). screenshotFullPage: screenshot of the whole page top-to-bottom (returns a CDN url with dimensions and a written analysis). screenshotViewport: screenshot of just the visible viewport \u2014 pass `scrollToSelector` (or `scrollY`) on this step to scroll a section into view and capture it in one atomic step (no separate scroll needed). setViewport: switch the browser between desktop and mobile rendering (pass `mode`: "desktop" or "mobile"). Reloads the page so responsive layouts, media queries, and matchMedia re-evaluate \u2014 use it to QA mobile/responsive views.'
4677
4706
  },
4678
4707
  ref: {
4679
4708
  type: "string",
@@ -4741,20 +4770,11 @@ var init_tools2 = __esm({
4741
4770
  required: ["steps"]
4742
4771
  }
4743
4772
  },
4744
- {
4745
- clearable: true,
4746
- name: "screenshotFullPage",
4747
- description: "Capture a full-height screenshot of the current page. Returns a CDN URL with full text analysis and description.",
4748
- inputSchema: {
4749
- type: "object",
4750
- properties: {
4751
- path: {
4752
- type: "string",
4753
- description: 'Navigate to this path before capturing (e.g. "/settings"). If omitted, screenshots the current page.'
4754
- }
4755
- }
4756
- }
4757
- },
4773
+ // Captures are `browserCommand` steps only — there is deliberately no
4774
+ // standalone screenshot tool here. Both used to exist for full-page, with
4775
+ // different budgets, different result plumbing, and analysis on only one of
4776
+ // them, so which door you picked changed what you got back.
4777
+ //
4758
4778
  // Read tools so the QA agent can pull full spec detail on demand — the spec
4759
4779
  // context in its prompt is a lightweight index (see prompt.ts) that points
4760
4780
  // here. Routed to the global executeTool in index.ts, mirroring specSync.
@@ -4973,7 +4993,7 @@ async function runBrowserAutomation(task, context, opts) {
4973
4993
  );
4974
4994
  } catch {
4975
4995
  }
4976
- let lastBrowserCommandViewport;
4996
+ let lastCapture = {};
4977
4997
  const result = await runSubAgent({
4978
4998
  system: getBrowserAutomationPrompt(),
4979
4999
  task,
@@ -4995,22 +5015,6 @@ async function runBrowserAutomation(task, context, opts) {
4995
5015
  return `Error setting up browser: ${err.message}`;
4996
5016
  }
4997
5017
  }
4998
- if (name === "screenshotFullPage") {
4999
- try {
5000
- return await captureAndAnalyzeScreenshot({
5001
- path: _input.path,
5002
- fullPage: true,
5003
- onLog,
5004
- model: resolveModel(
5005
- "imageAnalysis",
5006
- context.models,
5007
- context.model
5008
- )
5009
- });
5010
- } catch (err) {
5011
- return `Error taking screenshot: ${err.message}`;
5012
- }
5013
- }
5014
5018
  if (COMMON_READ_TOOL_NAMES.has(name) || name === readSpecTool.definition.name) {
5015
5019
  return executeTool(name, _input, context);
5016
5020
  }
@@ -5032,14 +5036,16 @@ async function runBrowserAutomation(task, context, opts) {
5032
5036
  try {
5033
5037
  const parsed = JSON.parse(result2);
5034
5038
  const screenshotSteps = (parsed.steps || []).filter(
5035
- (s) => s.command === "screenshotViewport" && s.result?.url
5039
+ (s) => CAPTURE_COMMANDS.has(s.command) && s.result?.url
5036
5040
  );
5037
5041
  if (screenshotSteps.length > 0) {
5038
- const lastStep = screenshotSteps[screenshotSteps.length - 1];
5039
- lastBrowserCommandViewport = {
5040
- url: lastStep.result.url,
5041
- styleMap: lastStep.result.styleMap
5042
- };
5042
+ for (const step of screenshotSteps) {
5043
+ const kind = step.command === "screenshotFullPage" ? "fullPage" : "viewport";
5044
+ lastCapture[kind] = {
5045
+ url: step.result.url,
5046
+ styleMap: step.result.styleMap
5047
+ };
5048
+ }
5043
5049
  const visionOverride = {
5044
5050
  model: resolveModel(
5045
5051
  "imageAnalysis",
@@ -5063,13 +5069,12 @@ async function runBrowserAutomation(task, context, opts) {
5063
5069
  );
5064
5070
  try {
5065
5071
  const analyses = JSON.parse(batchResult);
5066
- let ai = 0;
5067
- for (const step of parsed.steps) {
5068
- if (step.command === "screenshotViewport" && step.result?.url && ai < analyses.length) {
5069
- step.result.analysis = analyses[ai]?.output?.analysis || analyses[ai]?.output || "";
5070
- ai++;
5072
+ screenshotSteps.forEach((step, i) => {
5073
+ if (i >= analyses.length) {
5074
+ return;
5071
5075
  }
5072
- }
5076
+ step.result.analysis = analyses[i]?.output?.analysis || analyses[i]?.output || "";
5077
+ });
5073
5078
  } catch {
5074
5079
  log8.debug("Failed to parse batch analysis result", {
5075
5080
  batchResult
@@ -5082,13 +5087,10 @@ async function runBrowserAutomation(task, context, opts) {
5082
5087
  }
5083
5088
  return result2;
5084
5089
  },
5085
- toolRegistry: context.toolRegistry,
5086
- captureArtifacts: ["screenshotFullPage"]
5090
+ toolRegistry: context.toolRegistry
5087
5091
  });
5088
5092
  context.subAgentMessages?.set(context.toolCallId, result.messages);
5089
- const fullPage = result.artifacts?.screenshotFullPage;
5090
- const viewport = lastBrowserCommandViewport;
5091
- const preferred = opts?.capture === "viewport" ? viewport ?? fullPage : fullPage ?? viewport;
5093
+ const preferred = opts?.capture === "viewport" ? lastCapture.viewport ?? lastCapture.fullPage : lastCapture.fullPage ?? lastCapture.viewport;
5092
5094
  return {
5093
5095
  text: result.text,
5094
5096
  ...preferred?.url ? { screenshot: { url: preferred.url, styleMap: preferred.styleMap } } : {}
@@ -5097,7 +5099,7 @@ async function runBrowserAutomation(task, context, opts) {
5097
5099
  release();
5098
5100
  }
5099
5101
  }
5100
- var log8, browserAutomationTool;
5102
+ var log8, CAPTURE_COMMANDS, browserAutomationTool;
5101
5103
  var init_browserAutomation = __esm({
5102
5104
  "src/subagents/browserAutomation/index.ts"() {
5103
5105
  "use strict";
@@ -5114,6 +5116,7 @@ var init_browserAutomation = __esm({
5114
5116
  init_surfaces();
5115
5117
  init_logger();
5116
5118
  log8 = createLogger("browser-automation");
5119
+ CAPTURE_COMMANDS = /* @__PURE__ */ new Set(["screenshotViewport", "screenshotFullPage"]);
5117
5120
  browserAutomationTool = {
5118
5121
  clearable: true,
5119
5122
  definition: {
@@ -5147,7 +5150,55 @@ var init_browserAutomation = __esm({
5147
5150
  });
5148
5151
 
5149
5152
  // src/tools/code/screenshot.ts
5150
- var screenshotTool;
5153
+ async function executeScreenshot(input, onLog, context) {
5154
+ const fullPage = input.fullPage === true;
5155
+ const model = resolveModel("imageAnalysis", context?.models, context?.model);
5156
+ try {
5157
+ if (input.imageUrl) {
5158
+ return await captureAndAnalyzeScreenshot({
5159
+ prompt: input.prompt,
5160
+ imageUrl: input.imageUrl,
5161
+ onLog,
5162
+ model
5163
+ });
5164
+ }
5165
+ if (input.instructions && context) {
5166
+ const shotKind = fullPage ? "full-page" : "viewport";
5167
+ const task = input.path ? `Navigate to "${input.path}", then: ${input.instructions}. After completing these steps, take a ${shotKind} screenshot.` : `${input.instructions}. After completing these steps, take a ${shotKind} screenshot.`;
5168
+ const result = await runBrowserAutomation(task, context, {
5169
+ capture: fullPage ? "fullPage" : "viewport"
5170
+ });
5171
+ if (!result.screenshot) {
5172
+ return result.text;
5173
+ }
5174
+ return await streamScreenshotAnalysis({
5175
+ url: result.screenshot.url,
5176
+ prompt: input.prompt,
5177
+ styleMap: result.screenshot.styleMap,
5178
+ onLog,
5179
+ model
5180
+ });
5181
+ }
5182
+ const release = await acquireBrowserLock();
5183
+ try {
5184
+ return await captureAndAnalyzeScreenshot({
5185
+ prompt: input.prompt,
5186
+ path: input.path,
5187
+ fullPage,
5188
+ width: input.width,
5189
+ height: input.height,
5190
+ format: input.format,
5191
+ onLog,
5192
+ model
5193
+ });
5194
+ } finally {
5195
+ release();
5196
+ }
5197
+ } catch (err) {
5198
+ return `Error taking screenshot: ${err.message}`;
5199
+ }
5200
+ }
5201
+ var screenshotDefinition, screenshotTool;
5151
5202
  var init_screenshot2 = __esm({
5152
5203
  "src/tools/code/screenshot.ts"() {
5153
5204
  "use strict";
@@ -5155,99 +5206,55 @@ var init_screenshot2 = __esm({
5155
5206
  init_browserLock();
5156
5207
  init_browserAutomation();
5157
5208
  init_surfaces();
5158
- screenshotTool = {
5209
+ screenshotDefinition = {
5159
5210
  clearable: true,
5160
- definition: {
5161
- name: "screenshot",
5162
- description: "Capture a screenshot of the app preview and get a description of what's on screen. Choose `fullPage`: `false` captures just the visible viewport (fast \u2014 for a specific section the page is scrolled to), `true` captures the entire page top-to-bottom (slower \u2014 for overall composition or content past the fold). Captures the settled page state \u2014 it cannot catch animations, transitions, or transient state. Optionally provide specific questions about what you're looking for. Use a bulleted list to ask many questions at once. To ask additional questions about a screenshot you have already captured, pass its URL as imageUrl to skip recapture. If the screenshot requires interaction first (logging in, clicking a tab, dismissing a modal, scrolling to a section), use the instructions param to describe the steps. To render a fixed-size image such as an Open Graph share card, set `width` and `height` (e.g. 1200 \xD7 630) and `format: 'png'`: the tool navigates to `path`, clips to exactly those pixel dimensions, and returns the image URL.",
5163
- inputSchema: {
5164
- type: "object",
5165
- properties: {
5166
- fullPage: {
5167
- type: "boolean",
5168
- description: "true = full-height capture of the entire page; false = just the visible viewport. Pick based on whether you need the whole page or a specific section."
5169
- },
5170
- prompt: {
5171
- type: "string",
5172
- description: "Optional question about the screenshot. If omitted, returns a general description of what's visible."
5173
- },
5174
- imageUrl: {
5175
- type: "string",
5176
- description: "URL of an existing screenshot to analyze instead of capturing a new one. Use this for additional questions about a previous screenshot."
5177
- },
5178
- path: {
5179
- type: "string",
5180
- description: 'Navigate to this path before capturing (e.g. "/settings", "/dashboard"). If omitted, screenshots the current page.'
5181
- },
5182
- width: {
5183
- type: "number",
5184
- description: "Exact capture width in pixels. Set together with `height` to render a fixed-size image; clips to exactly this viewport instead of the default preview size."
5185
- },
5186
- height: {
5187
- type: "number",
5188
- description: "Exact capture height in pixels. Set together with `width`."
5189
- },
5190
- format: {
5191
- type: "string",
5192
- enum: ["png", "jpeg"],
5193
- description: "Output image format. Defaults to 'jpeg'. Use 'png' for crisp flat graphics like share cards, where JPEG artifacts show on sharp type and edges."
5194
- },
5195
- instructions: {
5196
- type: "string",
5197
- description: "If the screenshot you need requires interaction first (dismissing a modal, clicking a tab, filling out a form, navigating a flow, scrolling to a section, getting through a login/auth checkpoint), describe the steps to get there. A browser automation agent will follow these instructions, then capture per your `fullPage` choice \u2014 so with `fullPage: false` you can scroll to a section and capture just that viewport. It can bypass auth and get right to where it needs to be if you tell it to authenticate as a test user and give it the path/screen to start its test at. Never describe what names or values to use when applying the instructions - the browser automation agent must use its own values for it to work properly. If a specific auth role is required to access the content, be sure to note that - it can automatically assume it for the purpose of testing. Use only when interaction is required to *reach* the state you want to capture \u2014 log in, dismiss a modal, switch a tab, follow a route, scroll to a section. If your steps are exercising the app's functionality across multiple states (running flows, asserting behavior under interaction, multi-step QA), use `runAutomatedBrowserTest` instead."
5198
- }
5211
+ name: "screenshot",
5212
+ description: "Capture a screenshot of the app preview and get a description of what's on screen. Choose `fullPage`: `false` captures just the visible viewport (fast \u2014 for a specific section the page is scrolled to), `true` captures the entire page top-to-bottom (slower \u2014 for overall composition or content past the fold). Captures the settled page state \u2014 it cannot catch animations, transitions, or transient state. The analysis is not precise about every detail \u2014 for example it cannot reliably identify specific fonts by name, only describe what the letterforms look like. Optionally provide specific questions about what you're looking for. Use a bulleted list to ask many questions at once. To ask additional questions about a screenshot you have already captured, pass its URL as imageUrl to skip recapture. If the screenshot requires interaction first (logging in, clicking a tab, dismissing a modal, scrolling to a section), use the instructions param to describe the steps. To render a fixed-size image such as an Open Graph share card, set `width` and `height` (e.g. 1200 \xD7 630) and `format: 'png'`: the tool navigates to `path`, clips to exactly those pixel dimensions, and returns the image URL.",
5213
+ inputSchema: {
5214
+ type: "object",
5215
+ properties: {
5216
+ fullPage: {
5217
+ type: "boolean",
5218
+ description: "true = full-height capture of the entire page; false = just the visible viewport. Pick based on whether you need the whole page or a specific section."
5199
5219
  },
5200
- required: ["fullPage"]
5201
- }
5202
- },
5203
- async execute(input, context) {
5204
- const fullPage = input.fullPage === true;
5205
- const shotKind = fullPage ? "full-page" : "viewport";
5206
- try {
5207
- if (input.imageUrl) {
5208
- return await captureAndAnalyzeScreenshot({
5209
- prompt: input.prompt,
5210
- imageUrl: input.imageUrl,
5211
- onLog: context?.onLog,
5212
- model: resolveModel("imageAnalysis", context?.models, context?.model)
5213
- });
5214
- }
5215
- if (input.instructions && context) {
5216
- const task = input.path ? `Navigate to "${input.path}", then: ${input.instructions}. After completing these steps, take a ${shotKind} screenshot.` : `${input.instructions}. After completing these steps, take a ${shotKind} screenshot.`;
5217
- const result = await runBrowserAutomation(task, context, {
5218
- capture: fullPage ? "fullPage" : "viewport"
5219
- });
5220
- if (!result.screenshot) {
5221
- return result.text;
5222
- }
5223
- return await streamScreenshotAnalysis({
5224
- url: result.screenshot.url,
5225
- prompt: input.prompt,
5226
- styleMap: result.screenshot.styleMap,
5227
- onLog: context?.onLog,
5228
- model: resolveModel("imageAnalysis", context?.models, context?.model)
5229
- });
5230
- }
5231
- const release = await acquireBrowserLock();
5232
- try {
5233
- return await captureAndAnalyzeScreenshot({
5234
- prompt: input.prompt,
5235
- path: input.path,
5236
- fullPage,
5237
- width: input.width,
5238
- height: input.height,
5239
- format: input.format,
5240
- onLog: context?.onLog,
5241
- model: resolveModel("imageAnalysis", context?.models, context?.model)
5242
- });
5243
- } finally {
5244
- release();
5220
+ prompt: {
5221
+ type: "string",
5222
+ description: "Optional question about the screenshot. If omitted, returns a general description of what's visible."
5223
+ },
5224
+ imageUrl: {
5225
+ type: "string",
5226
+ description: "URL of an existing screenshot to analyze instead of capturing a new one. Use this for additional questions about a previous screenshot."
5227
+ },
5228
+ path: {
5229
+ type: "string",
5230
+ description: 'Navigate to this path before capturing (e.g. "/settings", "/dashboard"). If omitted, screenshots the current page.'
5231
+ },
5232
+ width: {
5233
+ type: "number",
5234
+ description: "Exact capture width in pixels. Set together with `height` to render a fixed-size image; clips to exactly this viewport instead of the default preview size."
5235
+ },
5236
+ height: {
5237
+ type: "number",
5238
+ description: "Exact capture height in pixels. Set together with `width`."
5239
+ },
5240
+ format: {
5241
+ type: "string",
5242
+ enum: ["png", "jpeg"],
5243
+ description: "Output image format. Defaults to 'jpeg'. Use 'png' for crisp flat graphics like share cards, where JPEG artifacts show on sharp type and edges."
5244
+ },
5245
+ instructions: {
5246
+ type: "string",
5247
+ description: "If the screenshot you need requires interaction first (dismissing a modal, clicking a tab, filling out a form, navigating a flow, scrolling to a section, getting through a login/auth checkpoint), describe the steps to get there. A browser automation agent will follow these instructions, then capture per your `fullPage` choice \u2014 so with `fullPage: false` you can scroll to a section and capture just that viewport. It can bypass auth and get right to where it needs to be if you tell it to authenticate as a test user and give it the path/screen to start its test at. Never describe what names or values to use when applying the instructions - the browser automation agent must use its own values for it to work properly. If a specific auth role is required to access the content, be sure to note that - it can automatically assume it for the purpose of testing. Use only when interaction is required to *reach* the state you want to capture \u2014 log in, dismiss a modal, switch a tab, follow a route, scroll to a section. If your steps are exercising the app's functionality across multiple states (running flows, asserting behavior under interaction, multi-step QA), use `runAutomatedBrowserTest` instead."
5245
5248
  }
5246
- } catch (err) {
5247
- return `Error taking screenshot: ${err.message}`;
5248
- }
5249
+ },
5250
+ required: ["fullPage"]
5249
5251
  }
5250
5252
  };
5253
+ screenshotTool = {
5254
+ clearable: true,
5255
+ definition: screenshotDefinition,
5256
+ execute: (input, context) => executeScreenshot(input, context?.onLog, context)
5257
+ };
5251
5258
  }
5252
5259
  });
5253
5260
 
@@ -5479,88 +5486,6 @@ var init_analyzeImage2 = __esm({
5479
5486
  }
5480
5487
  });
5481
5488
 
5482
- // src/subagents/designExpert/tools/screenshot.ts
5483
- var screenshot_exports = {};
5484
- __export(screenshot_exports, {
5485
- definition: () => definition5,
5486
- execute: () => execute5
5487
- });
5488
- async function execute5(input, onLog, context) {
5489
- const fullPage = input.fullPage === true;
5490
- const shotKind = fullPage ? "full-page" : "viewport";
5491
- if (input.instructions && context) {
5492
- try {
5493
- const task = input.path ? `Navigate to "${input.path}", then: ${input.instructions}. After completing these steps, take a ${shotKind} screenshot.` : `${input.instructions}. After completing these steps, take a ${shotKind} screenshot.`;
5494
- const result = await runBrowserAutomation(task, context, {
5495
- capture: fullPage ? "fullPage" : "viewport"
5496
- });
5497
- if (!result.screenshot) {
5498
- return result.text;
5499
- }
5500
- return await streamScreenshotAnalysis({
5501
- url: result.screenshot.url,
5502
- prompt: input.prompt,
5503
- styleMap: result.screenshot.styleMap,
5504
- onLog,
5505
- model: resolveModel("imageAnalysis", context?.models, context?.model)
5506
- });
5507
- } catch (err) {
5508
- return `Error taking interactive screenshot: ${err.message}`;
5509
- }
5510
- }
5511
- const release = await acquireBrowserLock();
5512
- try {
5513
- return await captureAndAnalyzeScreenshot({
5514
- prompt: input.prompt,
5515
- path: input.path,
5516
- fullPage,
5517
- onLog,
5518
- model: resolveModel("imageAnalysis", context?.models, context?.model)
5519
- });
5520
- } catch (err) {
5521
- return `Error taking screenshot: ${err.message}`;
5522
- } finally {
5523
- release();
5524
- }
5525
- }
5526
- var definition5;
5527
- var init_screenshot3 = __esm({
5528
- "src/subagents/designExpert/tools/screenshot.ts"() {
5529
- "use strict";
5530
- init_screenshot();
5531
- init_browserLock();
5532
- init_browserAutomation();
5533
- init_surfaces();
5534
- definition5 = {
5535
- clearable: true,
5536
- name: "screenshot",
5537
- description: "Capture a screenshot of the current app preview and get it back with visual analysis. Choose `fullPage`: `false` captures just the visible viewport (fast \u2014 use it to review a specific section the page is scrolled to), `true` captures the entire page top-to-bottom (slower \u2014 use it to review overall composition or a layout you can't see in one screen). Use to review the current state of the UI being built. Remember, the screenshot analysis is not overly precise - for example, it cannot reliably identify specific fonts by name \u2014 it can only describe what letterforms look like.",
5538
- inputSchema: {
5539
- type: "object",
5540
- properties: {
5541
- fullPage: {
5542
- type: "boolean",
5543
- description: "true = full-height capture of the entire page; false = just the visible viewport. Pick based on whether you need the whole page or a specific section."
5544
- },
5545
- prompt: {
5546
- type: "string",
5547
- description: "Optional specific question about the screenshot. Use a bulleted list to ask many questions at once."
5548
- },
5549
- path: {
5550
- type: "string",
5551
- description: 'Navigate to this path before capturing (e.g. "/settings"). If omitted, screenshots the current page.'
5552
- },
5553
- instructions: {
5554
- type: "string",
5555
- description: "If the screenshot you need requires interaction first (dismissing a modal, clicking a tab, filling out a form, scrolling to a specific section, getting through a login/auth checkpoint), describe the steps to get there. A browser automation agent will follow these instructions, then capture per your `fullPage` choice \u2014 so with `fullPage: false` you can scroll to a section and capture just that viewport. It can bypass auth and get right to where it needs to be if you tell it to authenticate as a test user and give it the path/screen to start at. Never describe what names or values to use when applying the instructions - the browser automation agent must use its own values for it to work properly. If a specific auth role is required to access the content, be sure to note that - it can automatically assume it for the purpose of testing."
5556
- }
5557
- },
5558
- required: ["fullPage"]
5559
- }
5560
- };
5561
- }
5562
- });
5563
-
5564
5489
  // src/subagents/designExpert/tools/images/enhancePrompt.ts
5565
5490
  async function enhanceImagePrompt(params) {
5566
5491
  const {
@@ -5770,10 +5695,10 @@ var init_imageGenerator = __esm({
5770
5695
  // src/subagents/designExpert/tools/images/generateImages.ts
5771
5696
  var generateImages_exports = {};
5772
5697
  __export(generateImages_exports, {
5773
- definition: () => definition6,
5774
- execute: () => execute6
5698
+ definition: () => definition5,
5699
+ execute: () => execute5
5775
5700
  });
5776
- async function execute6(input, onLog, context) {
5701
+ async function execute5(input, onLog, context) {
5777
5702
  return generateImageAssets({
5778
5703
  prompts: input.prompts,
5779
5704
  width: input.width,
@@ -5799,13 +5724,13 @@ async function execute6(input, onLog, context) {
5799
5724
  )
5800
5725
  });
5801
5726
  }
5802
- var definition6;
5727
+ var definition5;
5803
5728
  var init_generateImages = __esm({
5804
5729
  "src/subagents/designExpert/tools/images/generateImages.ts"() {
5805
5730
  "use strict";
5806
5731
  init_imageGenerator();
5807
5732
  init_surfaces();
5808
- definition6 = {
5733
+ definition5 = {
5809
5734
  clearable: false,
5810
5735
  name: "generateImages",
5811
5736
  description: "Generate images. Returns CDN URLs with a quality analysis for each image. Produces high-quality results for everything from photorealistic images and abstract/creative visuals. Pass multiple prompts to generate in parallel. No need to analyze images separately after generating \u2014 the analysis is included.",
@@ -5845,10 +5770,10 @@ var init_generateImages = __esm({
5845
5770
  // src/subagents/designExpert/tools/images/editImages.ts
5846
5771
  var editImages_exports = {};
5847
5772
  __export(editImages_exports, {
5848
- definition: () => definition7,
5849
- execute: () => execute7
5773
+ definition: () => definition6,
5774
+ execute: () => execute6
5850
5775
  });
5851
- async function execute7(input, onLog, context) {
5776
+ async function execute6(input, onLog, context) {
5852
5777
  return generateImageAssets({
5853
5778
  prompts: input.prompts,
5854
5779
  sourceImages: input.sourceImages,
@@ -5874,13 +5799,13 @@ async function execute7(input, onLog, context) {
5874
5799
  )
5875
5800
  });
5876
5801
  }
5877
- var definition7;
5802
+ var definition6;
5878
5803
  var init_editImages = __esm({
5879
5804
  "src/subagents/designExpert/tools/images/editImages.ts"() {
5880
5805
  "use strict";
5881
5806
  init_imageGenerator();
5882
5807
  init_surfaces();
5883
- definition7 = {
5808
+ definition6 = {
5884
5809
  clearable: false,
5885
5810
  name: "editImages",
5886
5811
  description: "Edit or transform existing images. Provide one or more source image URLs as reference and a prompt describing the desired edit. Use for compositing, style transfer, subject transformation, blending multiple references, or incorporating one or more references into something new. Returns CDN URLs with analysis.",
@@ -5995,18 +5920,18 @@ var init_copyEditor = __esm({
5995
5920
  // src/subagents/designExpert/tools/polishCopy.ts
5996
5921
  var polishCopy_exports = {};
5997
5922
  __export(polishCopy_exports, {
5998
- definition: () => definition8,
5999
- execute: () => execute8
5923
+ definition: () => definition7,
5924
+ execute: () => execute7
6000
5925
  });
6001
- async function execute8(input, _onLog, context) {
5926
+ async function execute7(input, _onLog, context) {
6002
5927
  return copyEditorTool.execute(input, context);
6003
5928
  }
6004
- var definition8;
5929
+ var definition7;
6005
5930
  var init_polishCopy = __esm({
6006
5931
  "src/subagents/designExpert/tools/polishCopy.ts"() {
6007
5932
  "use strict";
6008
5933
  init_copyEditor();
6009
- definition8 = {
5934
+ definition7 = {
6010
5935
  clearable: false,
6011
5936
  name: "polishCopy",
6012
5937
  description: "Hand off any user-facing copy you've written \u2014 headlines, captions, labels, body text \u2014 and get back a sharper version: better built for its audience and free of the fingerprints that make writing read as AI. It elevates how the copy communicates without inventing facts or claims you didn't give it. Give it the text plus what it's for (where it appears, the audience).",
@@ -6043,16 +5968,19 @@ var init_tools4 = __esm({
6043
5968
  init_scrapeWebUrl();
6044
5969
  init_analyzeDesign();
6045
5970
  init_analyzeImage2();
6046
- init_screenshot3();
6047
5971
  init_generateImages();
6048
5972
  init_editImages();
6049
5973
  init_polishCopy();
5974
+ init_screenshot2();
6050
5975
  tools = {
6051
5976
  searchGoogle: searchGoogle_exports,
6052
5977
  scrapeWebUrl: scrapeWebUrl_exports,
6053
5978
  analyzeDesign: analyzeDesign_exports,
6054
5979
  analyzeImage: analyzeImage_exports,
6055
- screenshot: screenshot_exports,
5980
+ // Same tool the main agent offers, imported rather than reimplemented — the
5981
+ // two used to be near-identical copies and had already drifted apart. Its core
5982
+ // already takes (input, onLog, context), which is this registry's convention.
5983
+ screenshot: { definition: screenshotDefinition, execute: executeScreenshot },
6056
5984
  generateImages: generateImages_exports,
6057
5985
  editImages: editImages_exports,
6058
5986
  polishCopy: polishCopy_exports
@@ -6299,7 +6227,7 @@ function renderDesignSystemBlock(designSystem) {
6299
6227
  "</design_system>"
6300
6228
  ].join("\n");
6301
6229
  }
6302
- function getDesignExpertPrompt(onboardingState) {
6230
+ function getDesignExpertPrompt(onboardingState, opts) {
6303
6231
  const specContext = loadSpecIndex();
6304
6232
  const indices = getSampleIndices(
6305
6233
  {
@@ -6344,10 +6272,15 @@ ${designSystemBlock}`;
6344
6272
  <project_phase>
6345
6273
  This project is in the "${state}" phase. The codebase is a placeholder scaffold or is being generated for the first time.
6346
6274
  </project_phase>`;
6275
+ }
6276
+ if (opts?.render) {
6277
+ prompt += `
6278
+
6279
+ ${RENDER_TASK_BLOCK}`;
6347
6280
  }
6348
6281
  return prompt;
6349
6282
  }
6350
- var SUBAGENT, RUNTIME_PLACEHOLDERS, PROMPT_TEMPLATE;
6283
+ var SUBAGENT, RUNTIME_PLACEHOLDERS, PROMPT_TEMPLATE, RENDER_TASK_BLOCK;
6351
6284
  var init_prompt2 = __esm({
6352
6285
  "src/subagents/designExpert/prompt.ts"() {
6353
6286
  "use strict";
@@ -6368,6 +6301,11 @@ var init_prompt2 = __esm({
6368
6301
  const k = key.trim();
6369
6302
  return RUNTIME_PLACEHOLDERS.has(k) ? match : readAsset(SUBAGENT, k);
6370
6303
  }).replace(/\n{3,}/g, "\n\n");
6304
+ RENDER_TASK_BLOCK = `<render_task>
6305
+ This task is a render: you are the author of the artifact, not an advisor on it. Everything above describes your usual work \u2014 proposing a design to a developer who then builds it. Here there is no developer downstream. You write the file the brief names, using writeFile (or editFile when it already exists), and that file is the entire deliverable.
6306
+
6307
+ The guidance about specifying layouts in prose, writing implementation notes, and keeping wireframes small doesn't apply here: it exists to hand a design to someone else to build, and you are the one building it. Your reply is a receipt \u2014 one line naming what you wrote.
6308
+ </render_task>`;
6371
6309
  }
6372
6310
  });
6373
6311
 
@@ -6424,17 +6362,19 @@ var init_history = __esm({
6424
6362
  async function runDesignExpert(opts, context) {
6425
6363
  const history = context.conversationMessages ? getSubAgentHistory(context.conversationMessages, "visualDesignExpert") : [];
6426
6364
  return runSubAgent({
6427
- system: getDesignExpertPrompt(context.onboardingState),
6365
+ system: getDesignExpertPrompt(context.onboardingState, {
6366
+ render: opts.render
6367
+ }),
6428
6368
  task: opts.task,
6429
6369
  history: history.length > 0 ? history : void 0,
6430
- tools: opts.enableWrite ? DESIGN_EXPERT_RENDER_TOOLS : DESIGN_EXPERT_TOOLS,
6370
+ tools: opts.render ? DESIGN_EXPERT_RENDER_TOOLS : DESIGN_EXPERT_TOOLS,
6431
6371
  externalTools: /* @__PURE__ */ new Set(),
6432
6372
  executeTool: (name, input, toolCallId, onLog, sams) => {
6433
6373
  const childCtx = toolCallId ? { ...deriveContext(context, toolCallId), subAgentMessages: sams } : { ...context, subAgentMessages: sams };
6434
6374
  if (COMMON_READ_TOOL_NAMES.has(name)) {
6435
6375
  return executeTool(name, input, childCtx);
6436
6376
  }
6437
- if (opts.enableWrite && RENDER_WRITE_TOOL_NAMES.has(name)) {
6377
+ if (opts.render && RENDER_WRITE_TOOL_NAMES.has(name)) {
6438
6378
  return executeTool(name, input, childCtx);
6439
6379
  }
6440
6380
  return executeDesignExpertTool(name, input, childCtx, toolCallId, onLog);
@@ -6460,7 +6400,7 @@ async function runDesignExpert(opts, context) {
6460
6400
  });
6461
6401
  }
6462
6402
  async function runDesignExpertRender(opts, context) {
6463
- return runDesignExpert({ ...opts, enableWrite: true }, context);
6403
+ return runDesignExpert({ ...opts, render: true }, context);
6464
6404
  }
6465
6405
  var DESCRIPTION, RENDER_WRITE_TOOL_NAMES, DESIGN_EXPERT_RENDER_TOOLS, designExpertTool;
6466
6406
  var init_designExpert = __esm({
@@ -6632,14 +6572,14 @@ ${unifiedDiff(filePath, oldContent, "")}`;
6632
6572
  const delivery = exists ? `### Your deliverable
6633
6573
  The pitch deck already exists at \`${filePath}\`. Read it, then update it for the new <pitch_content>, keeping the presentation scaffolding intact \u2014 change only what needs to change.
6634
6574
 
6635
- Reply with a one-line summary of what you changed \u2014 not the HTML.` : `### Your deliverable
6575
+ Then reply with a one-line summary of what you changed.` : `### Your deliverable
6636
6576
  Write the complete pitch deck to \`${filePath}\`. It does not exist yet \u2014 start from this scaffold, keeping its progress bar, chevron navigation, keyboard navigation, and transition mechanics intact:
6637
6577
 
6638
6578
  <pitch_deck_shell>
6639
6579
  ${PITCH_DECK_SHELL}
6640
6580
  </pitch_deck_shell>
6641
6581
 
6642
- Reply with a one-line summary of what you wrote \u2014 not the HTML.`;
6582
+ Then reply with a one-line summary of what you wrote.`;
6643
6583
  const task = `
6644
6584
  <pitch_content>${input.task}</pitch_content>
6645
6585
 
@@ -7097,13 +7037,13 @@ Write the complete Build Overview to \`${OVERVIEW_FILE}\`. The file does not exi
7097
7037
  ${OVERVIEW_SHELL}
7098
7038
  </overview_shell>
7099
7039
 
7100
- Reply with a one-line summary of what you wrote \u2014 not the HTML.`;
7040
+ Then reply with a one-line summary of what you wrote.`;
7101
7041
  }
7102
7042
  function refreshDelivery() {
7103
7043
  return `### Your deliverable
7104
7044
  The Build Overview already exists at \`${OVERVIEW_FILE}\`. Read it, then update it to reflect <overview_copy>, preserving its established skin \u2014 change only what the copy changed.
7105
7045
 
7106
- Reply with a one-line summary of what you changed \u2014 not the HTML.`;
7046
+ Then reply with a one-line summary of what you changed.`;
7107
7047
  }
7108
7048
  var OVERVIEW_FILE, DESIGN_BRIEF, OVERVIEW_SHELL, buildOverviewTool;
7109
7049
  var init_writeBuildOverview = __esm({