@namzu/sdk 45.1.0 → 46.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (118) hide show
  1. package/CHANGELOG.md +111 -0
  2. package/dist/authorization/gate.d.ts +5 -2
  3. package/dist/authorization/gate.d.ts.map +1 -1
  4. package/dist/authorization/gate.js +25 -4
  5. package/dist/authorization/gate.js.map +1 -1
  6. package/dist/authorization/rules.d.ts.map +1 -1
  7. package/dist/authorization/rules.js +19 -0
  8. package/dist/authorization/rules.js.map +1 -1
  9. package/dist/authorization/shell-lexer.d.ts +24 -0
  10. package/dist/authorization/shell-lexer.d.ts.map +1 -1
  11. package/dist/authorization/shell-lexer.js +69 -51
  12. package/dist/authorization/shell-lexer.js.map +1 -1
  13. package/dist/pricing/catalogue.generated.d.ts.map +1 -1
  14. package/dist/pricing/catalogue.generated.js +28 -4
  15. package/dist/pricing/catalogue.generated.js.map +1 -1
  16. package/dist/public-runtime.d.ts +3 -3
  17. package/dist/public-runtime.d.ts.map +1 -1
  18. package/dist/public-runtime.js +4 -2
  19. package/dist/public-runtime.js.map +1 -1
  20. package/dist/public-tools.d.ts +3 -2
  21. package/dist/public-tools.d.ts.map +1 -1
  22. package/dist/public-tools.js +5 -2
  23. package/dist/public-tools.js.map +1 -1
  24. package/dist/public-types.d.ts +3 -3
  25. package/dist/public-types.d.ts.map +1 -1
  26. package/dist/registry/tool/callable.d.ts +22 -0
  27. package/dist/registry/tool/callable.d.ts.map +1 -0
  28. package/dist/registry/tool/callable.js +29 -0
  29. package/dist/registry/tool/callable.js.map +1 -0
  30. package/dist/registry/tool/execute.d.ts.map +1 -1
  31. package/dist/registry/tool/execute.js +8 -1
  32. package/dist/registry/tool/execute.js.map +1 -1
  33. package/dist/runtime/query/executor/tool-call-admission.d.ts +37 -11
  34. package/dist/runtime/query/executor/tool-call-admission.d.ts.map +1 -1
  35. package/dist/runtime/query/executor/tool-call-admission.js +38 -12
  36. package/dist/runtime/query/executor/tool-call-admission.js.map +1 -1
  37. package/dist/runtime/query/executor.d.ts +4 -2
  38. package/dist/runtime/query/executor.d.ts.map +1 -1
  39. package/dist/runtime/query/executor.js +18 -5
  40. package/dist/runtime/query/executor.js.map +1 -1
  41. package/dist/runtime/query/review-policy.d.ts +43 -0
  42. package/dist/runtime/query/review-policy.d.ts.map +1 -1
  43. package/dist/runtime/query/review-policy.js +59 -16
  44. package/dist/runtime/query/review-policy.js.map +1 -1
  45. package/dist/skills/registry.d.ts +6 -0
  46. package/dist/skills/registry.d.ts.map +1 -1
  47. package/dist/skills/registry.js +1 -0
  48. package/dist/skills/registry.js.map +1 -1
  49. package/dist/tools/builtins/computer-use-coordinates.d.ts +65 -0
  50. package/dist/tools/builtins/computer-use-coordinates.d.ts.map +1 -0
  51. package/dist/tools/builtins/computer-use-coordinates.js +123 -0
  52. package/dist/tools/builtins/computer-use-coordinates.js.map +1 -0
  53. package/dist/tools/builtins/computer-use-image.d.ts +77 -0
  54. package/dist/tools/builtins/computer-use-image.d.ts.map +1 -0
  55. package/dist/tools/builtins/computer-use-image.js +223 -0
  56. package/dist/tools/builtins/computer-use-image.js.map +1 -0
  57. package/dist/tools/builtins/computer-use.d.ts +519 -14
  58. package/dist/tools/builtins/computer-use.d.ts.map +1 -1
  59. package/dist/tools/builtins/computer-use.js +1188 -183
  60. package/dist/tools/builtins/computer-use.js.map +1 -1
  61. package/dist/tools/builtins/index.d.ts +2 -1
  62. package/dist/tools/builtins/index.d.ts.map +1 -1
  63. package/dist/tools/builtins/index.js +1 -1
  64. package/dist/tools/builtins/index.js.map +1 -1
  65. package/dist/tools/builtins/skill.d.ts +74 -0
  66. package/dist/tools/builtins/skill.d.ts.map +1 -1
  67. package/dist/tools/builtins/skill.js +214 -174
  68. package/dist/tools/builtins/skill.js.map +1 -1
  69. package/dist/tools/defineTool.d.ts +2 -0
  70. package/dist/tools/defineTool.d.ts.map +1 -1
  71. package/dist/tools/defineTool.js +7 -0
  72. package/dist/tools/defineTool.js.map +1 -1
  73. package/dist/tools/schedules/present.d.ts.map +1 -1
  74. package/dist/tools/schedules/present.js +2 -0
  75. package/dist/tools/schedules/present.js.map +1 -1
  76. package/dist/tools/schedules/schedule-tool.d.ts +6 -5
  77. package/dist/tools/schedules/schedule-tool.d.ts.map +1 -1
  78. package/dist/tools/schedules/schedule-tool.js +121 -30
  79. package/dist/tools/schedules/schedule-tool.js.map +1 -1
  80. package/dist/tools/schedules/types.d.ts +69 -1
  81. package/dist/tools/schedules/types.d.ts.map +1 -1
  82. package/dist/types/authorization/index.d.ts +21 -0
  83. package/dist/types/authorization/index.d.ts.map +1 -1
  84. package/dist/types/authorization/index.js +5 -0
  85. package/dist/types/authorization/index.js.map +1 -1
  86. package/dist/types/computer-use/index.d.ts +176 -0
  87. package/dist/types/computer-use/index.d.ts.map +1 -1
  88. package/dist/types/computer-use/index.js.map +1 -1
  89. package/dist/types/tool/index.d.ts +19 -0
  90. package/dist/types/tool/index.d.ts.map +1 -1
  91. package/dist/types/tool/index.js.map +1 -1
  92. package/package.json +3 -1
  93. package/src/authorization/gate.ts +25 -4
  94. package/src/authorization/rules.ts +18 -0
  95. package/src/authorization/shell-lexer.ts +73 -45
  96. package/src/pricing/catalogue.generated.ts +28 -4
  97. package/src/pricing/rates.source.json +27 -6
  98. package/src/public-runtime.ts +16 -2
  99. package/src/public-tools.ts +19 -1
  100. package/src/public-types.ts +5 -0
  101. package/src/registry/tool/callable.ts +37 -0
  102. package/src/registry/tool/execute.ts +8 -1
  103. package/src/runtime/query/executor/tool-call-admission.ts +60 -16
  104. package/src/runtime/query/executor.ts +19 -4
  105. package/src/runtime/query/review-policy.ts +103 -15
  106. package/src/skills/registry.ts +8 -0
  107. package/src/tools/builtins/computer-use-coordinates.ts +144 -0
  108. package/src/tools/builtins/computer-use-image.ts +278 -0
  109. package/src/tools/builtins/computer-use.ts +1485 -191
  110. package/src/tools/builtins/index.ts +7 -1
  111. package/src/tools/builtins/skill.ts +304 -174
  112. package/src/tools/defineTool.ts +10 -0
  113. package/src/tools/schedules/present.ts +2 -0
  114. package/src/tools/schedules/schedule-tool.ts +139 -33
  115. package/src/tools/schedules/types.ts +69 -1
  116. package/src/types/authorization/index.ts +21 -0
  117. package/src/types/computer-use/index.ts +202 -0
  118. package/src/types/tool/index.ts +19 -0
@@ -1,95 +1,217 @@
1
1
  import { z } from 'zod';
2
+ import { resolveProviderCapabilities } from '../../provider/capabilities.js';
3
+ import { sleep } from '../../utils/backoff.js';
2
4
  import { defineTool } from '../defineTool.js';
5
+ import { neutralizeEnvelopeDelimiter, wrapUntrusted } from '../untrusted-envelope.js';
6
+ import { ScreenshotFrames, assumedDisplay, desktopRectOnImage, pointOnImage, toDisplayPoint, toDisplayRect, toImagePoint, } from './computer-use-coordinates.js';
7
+ import { STANDARD_SCREENSHOT_LIMITS, cropAndFitPng, fitPng, pngSize, } from './computer-use-image.js';
8
+ export { HIGH_RES_SCREENSHOT_LIMITS, STANDARD_SCREENSHOT_LIMITS, screenshotTargetSize, } from './computer-use-image.js';
3
9
  export const COMPUTER_USE_TOOL_NAME = 'computer_use';
10
+ const DEFAULT_SETTLE_MS = 500;
11
+ const DEFAULT_MAX_BATCH_ACTIONS = 20;
12
+ const DEFAULT_MAX_WAIT_MS = 10_000;
13
+ const MAX_LISTED_WINDOWS = 50;
14
+ /** The most of a ui_snapshot the model is shown, in characters (about 4 000 tokens). */
15
+ const MAX_UI_SNAPSHOT_CHARS = 14_000;
16
+ /**
17
+ * Why `provider` cannot drive `computer_use`, or undefined when it can.
18
+ *
19
+ * The tool's only way to show the model the screen is an image in a tool
20
+ * result. A driver that declares `supportsToolResultImages: false` replaces
21
+ * that image with a line of text, so the model acts on a screen it has never
22
+ * seen while every call reports success. Pass the answer to
23
+ * {@link ComputerUseToolOptions.unavailableReason}.
24
+ */
25
+ export function computerUseUnavailableReason(provider) {
26
+ if (resolveProviderCapabilities(provider).supportsToolResultImages)
27
+ return undefined;
28
+ return `The ${provider.id} provider cannot return images in tool results, so the model would never see a screenshot. Use a provider that can (for example Anthropic, Codex or Google) for computer use.`;
29
+ }
4
30
  // ---------------------------------------------------------------------------
5
- // Input schema — discriminated union matching ComputerUseAction
31
+ // Input schema
6
32
  // ---------------------------------------------------------------------------
7
33
  const pointSchema = z.object({
8
34
  x: z.number().int(),
9
35
  y: z.number().int(),
10
36
  });
11
- const mouseButtonSchema = z.enum(['left', 'right', 'middle']);
37
+ const regionSchema = z.object({
38
+ x: z.number().int(),
39
+ y: z.number().int(),
40
+ width: z.number().int().positive(),
41
+ height: z.number().int().positive(),
42
+ });
43
+ // Left when omitted: a model asked to "click the Start button" often leaves
44
+ // the button out, and refusing that cost a whole round trip for nothing.
45
+ const mouseButtonSchema = z.enum(['left', 'right', 'middle']).default('left');
46
+ const screenshotIdSchema = z.string().min(1).optional();
47
+ const cursorPositionSchema = z.object({ type: z.literal('cursor_position') });
48
+ const mouseMoveSchema = z.object({ type: z.literal('mouse_move'), to: pointSchema });
49
+ const mouseClickSchema = z.object({
50
+ type: z.literal('mouse_click'),
51
+ at: pointSchema,
52
+ button: mouseButtonSchema,
53
+ });
54
+ const mouseDragSchema = z.object({
55
+ type: z.literal('mouse_drag'),
56
+ from: pointSchema,
57
+ to: pointSchema,
58
+ button: mouseButtonSchema,
59
+ });
60
+ const scrollSchema = z.object({
61
+ type: z.literal('scroll'),
62
+ at: pointSchema,
63
+ direction: z.enum(['up', 'down', 'left', 'right']),
64
+ amount: z.number().int().positive(),
65
+ });
66
+ const typeTextSchema = z.object({ type: z.literal('type_text'), text: z.string() });
67
+ const keySchema = z.object({ type: z.literal('key'), keys: z.string() });
68
+ const waitSchema = z.object({ type: z.literal('wait'), ms: z.number().int().nonnegative() });
69
+ const listWindowsSchema = z.object({ type: z.literal('list_windows') });
70
+ const focusWindowSchema = z.object({
71
+ type: z.literal('focus_window'),
72
+ window_id: z.string().min(1),
73
+ });
74
+ const UI_ACTIONS = [
75
+ 'invoke',
76
+ 'set_value',
77
+ 'toggle',
78
+ 'select',
79
+ 'expand',
80
+ 'collapse',
81
+ 'focus',
82
+ 'scroll_into_view',
83
+ ];
84
+ const uiSnapshotSchema = z.object({
85
+ type: z.literal('ui_snapshot'),
86
+ window_id: z.string().min(1).optional(),
87
+ });
88
+ const uiActSchema = z.object({
89
+ type: z.literal('ui_act'),
90
+ ref: z.string().min(1),
91
+ action: z.enum(UI_ACTIONS),
92
+ value: z.string().optional(),
93
+ });
94
+ /** What a batch may carry: everything but the image-returning actions and another batch. */
95
+ const batchItemSchema = z.discriminatedUnion('type', [
96
+ cursorPositionSchema,
97
+ mouseMoveSchema,
98
+ mouseClickSchema,
99
+ mouseDragSchema,
100
+ scrollSchema,
101
+ typeTextSchema,
102
+ keySchema,
103
+ waitSchema,
104
+ listWindowsSchema,
105
+ focusWindowSchema,
106
+ uiActSchema,
107
+ ]);
108
+ const withFrame = { screenshot_id: screenshotIdSchema };
12
109
  const actionSchema = z.discriminatedUnion('type', [
13
110
  z.object({ type: z.literal('screenshot') }),
14
- z.object({ type: z.literal('cursor_position') }),
15
- z.object({ type: z.literal('mouse_move'), to: pointSchema }),
16
- z.object({ type: z.literal('mouse_click'), at: pointSchema, button: mouseButtonSchema }),
17
- z.object({
18
- type: z.literal('mouse_drag'),
19
- from: pointSchema,
20
- to: pointSchema,
21
- button: mouseButtonSchema,
22
- }),
111
+ z.object({ type: z.literal('zoom'), region: regionSchema, ...withFrame }),
112
+ cursorPositionSchema.extend(withFrame),
113
+ mouseMoveSchema.extend(withFrame),
114
+ mouseClickSchema.extend(withFrame),
115
+ mouseDragSchema.extend(withFrame),
116
+ scrollSchema.extend(withFrame),
117
+ typeTextSchema.extend(withFrame),
118
+ keySchema.extend(withFrame),
119
+ waitSchema.extend(withFrame),
120
+ listWindowsSchema.extend(withFrame),
121
+ focusWindowSchema.extend(withFrame),
122
+ uiSnapshotSchema,
123
+ uiActSchema.extend(withFrame),
23
124
  z.object({
24
- type: z.literal('scroll'),
25
- at: pointSchema,
26
- direction: z.enum(['up', 'down', 'left', 'right']),
27
- amount: z.number().int().positive(),
125
+ type: z.literal('batch'),
126
+ actions: z.array(batchItemSchema).min(1),
127
+ ...withFrame,
28
128
  }),
29
- z.object({ type: z.literal('type_text'), text: z.string() }),
30
- z.object({ type: z.literal('key'), keys: z.string() }),
31
129
  ]);
32
- /**
33
- * The provider-facing shape is deliberately flat.
34
- *
35
- * The runtime schema above is the authoritative contract: it knows which
36
- * fields each action requires. Rendering that discriminated union produces a
37
- * root `anyOf`, however, and some custom-tool wires reject root combinators
38
- * even when every branch is an object. A model can still see every field and
39
- * every action here; incomplete combinations are rejected by `actionSchema`
40
- * before the host is called, with the recovery hint below.
41
- */
42
- const pointModelInputSchema = {
43
- type: 'object',
44
- properties: {
45
- x: { type: 'integer' },
46
- y: { type: 'integer' },
47
- },
48
- required: ['x', 'y'],
49
- additionalProperties: false,
50
- };
51
- const modelInputSchema = {
52
- type: 'object',
53
- properties: {
54
- type: {
55
- type: 'string',
56
- enum: [
57
- 'screenshot',
58
- 'cursor_position',
59
- 'mouse_move',
60
- 'mouse_click',
61
- 'mouse_drag',
62
- 'scroll',
63
- 'type_text',
64
- 'key',
65
- ],
66
- description: 'Desktop action. screenshot and cursor_position need no other fields; mouse_move needs to; mouse_click needs at and button; mouse_drag needs from, to, and button; scroll needs at, direction, and amount; type_text needs text; key needs keys.',
67
- },
68
- to: pointModelInputSchema,
69
- at: pointModelInputSchema,
70
- from: pointModelInputSchema,
71
- button: { type: 'string', enum: ['left', 'right', 'middle'] },
72
- direction: { type: 'string', enum: ['up', 'down', 'left', 'right'] },
73
- amount: { type: 'integer', description: 'Positive integer scroll distance.' },
74
- text: { type: 'string', description: 'Literal text to type.' },
75
- keys: {
76
- type: 'string',
77
- description: 'Key or key chord to press, for example ENTER or CTRL+R.',
78
- },
79
- },
80
- required: ['type'],
81
- additionalProperties: false,
82
- };
83
- const DESTRUCTIVE_ACTION_TYPES = new Set([
130
+ const HOST_ACTIONS = [
131
+ 'screenshot',
132
+ 'cursor_position',
133
+ 'mouse_move',
84
134
  'mouse_click',
85
135
  'mouse_drag',
136
+ 'scroll',
86
137
  'type_text',
87
138
  'key',
139
+ ];
140
+ const ALL_ACTIONS = [
141
+ 'screenshot',
142
+ 'zoom',
143
+ 'cursor_position',
144
+ 'mouse_move',
145
+ 'mouse_click',
146
+ 'mouse_drag',
88
147
  'scroll',
148
+ 'type_text',
149
+ 'key',
150
+ 'wait',
151
+ 'list_windows',
152
+ 'focus_window',
153
+ 'ui_snapshot',
154
+ 'ui_act',
155
+ 'batch',
156
+ ];
157
+ /** Observations: they change nothing on the desktop. */
158
+ const READ_ONLY_ACTIONS = new Set([
159
+ 'screenshot',
160
+ 'zoom',
161
+ 'cursor_position',
162
+ 'wait',
163
+ 'list_windows',
164
+ 'ui_snapshot',
89
165
  ]);
166
+ const DESTRUCTIVE_ACTIONS = new Set([
167
+ 'mouse_click',
168
+ 'mouse_drag',
169
+ 'type_text',
170
+ 'key',
171
+ 'scroll',
172
+ 'ui_act',
173
+ ]);
174
+ /** Observations that show the model what is on the screen, on their own. */
175
+ const SCREEN_OBSERVATIONS = new Set(['screenshot', 'zoom', 'list_windows', 'ui_snapshot']);
176
+ /**
177
+ * Terminal applications, by the process name a host reports in
178
+ * `WindowInfo.app` (on Windows without `.exe`). Keys are not typed into these.
179
+ */
180
+ const TERMINAL_APPS = /^(WindowsTerminal|OpenConsole|conhost|cmd|powershell|pwsh|wsl|mintty|wezterm(-gui)?|alacritty|Hyper|Tabby|kitty|ConEmu(64)?|putty|Terminal|iTerm2?|gnome-terminal(-server)?|konsole|xterm|xfce4-terminal|tilix|terminator|foot|ghostty|Warp)$/i;
181
+ /** Actions a batch cannot carry: the ones that return an image or a tree, and batch itself. */
182
+ const OUTSIDE_BATCH = new Set(['screenshot', 'zoom', 'ui_snapshot', 'batch']);
183
+ /** A single action's whole result when it only acted: `<label>: done`, then its screenshot. */
184
+ const ACKNOWLEDGEMENT = /^[^\n]*: done(\nScreenshot s\d+ \(\d+x\d+\)\.)?$/;
185
+ function isRecord(value) {
186
+ return typeof value === 'object' && value !== null && !Array.isArray(value);
187
+ }
188
+ function batchActions(input) {
189
+ if (!isRecord(input) || input.type !== 'batch')
190
+ return null;
191
+ return Array.isArray(input.actions) ? input.actions : [];
192
+ }
193
+ /** Read-only when every action is: a batch with one click in it is not. Unknown shapes are not. */
194
+ function isReadOnlyInput(input) {
195
+ const actions = batchActions(input);
196
+ if (actions)
197
+ return (actions.length > 0 &&
198
+ actions.every((item) => isRecord(item) && READ_ONLY_ACTIONS.has(String(item.type))));
199
+ return isRecord(input) && READ_ONLY_ACTIONS.has(String(input.type));
200
+ }
201
+ function isDestructiveInput(input) {
202
+ const actions = batchActions(input);
203
+ if (actions)
204
+ return actions.some((item) => !isRecord(item) || DESTRUCTIVE_ACTIONS.has(String(item.type)));
205
+ return isRecord(input) && DESTRUCTIVE_ACTIONS.has(String(input.type));
206
+ }
207
+ // ---------------------------------------------------------------------------
208
+ // Capabilities
209
+ // ---------------------------------------------------------------------------
90
210
  function requiredCapability(type) {
91
211
  switch (type) {
92
212
  case 'screenshot':
213
+ case 'zoom':
214
+ case 'wait':
93
215
  return 'screenshot';
94
216
  case 'cursor_position':
95
217
  return 'cursorPosition';
@@ -101,13 +223,53 @@ function requiredCapability(type) {
101
223
  case 'type_text':
102
224
  case 'key':
103
225
  return 'keyboard';
226
+ case 'list_windows':
227
+ case 'focus_window':
228
+ return 'windows';
229
+ case 'ui_snapshot':
230
+ case 'ui_act':
231
+ return 'uiTree';
104
232
  default:
105
233
  return null;
106
234
  }
107
235
  }
108
- function buildDescription(host) {
109
- const caps = host.capabilities;
110
- const available = availableActions(caps);
236
+ function hostActionAvailable(caps, action) {
237
+ const required = requiredCapability(action);
238
+ return ((required === null || caps[required] === true) &&
239
+ (caps.supportedActions === undefined || caps.supportedActions.includes(action)) &&
240
+ (action !== 'mouse_click' || caps.mouseClickButtons?.length !== 0) &&
241
+ (action !== 'mouse_drag' || caps.mouseDragButtons?.length !== 0));
242
+ }
243
+ function availableActions(host, caps) {
244
+ const screenshot = hostActionAvailable(caps, 'screenshot');
245
+ const windows = caps.windows === true &&
246
+ typeof host.listWindows === 'function' &&
247
+ typeof host.focusWindow === 'function';
248
+ const uiTree = caps.uiTree === true &&
249
+ typeof host.uiSnapshot === 'function' &&
250
+ typeof host.uiAct === 'function';
251
+ const available = ALL_ACTIONS.filter((action) => {
252
+ switch (action) {
253
+ case 'zoom':
254
+ case 'wait':
255
+ return screenshot;
256
+ case 'list_windows':
257
+ case 'focus_window':
258
+ return windows;
259
+ case 'ui_snapshot':
260
+ case 'ui_act':
261
+ return uiTree;
262
+ case 'batch':
263
+ return false;
264
+ default:
265
+ return hostActionAvailable(caps, action);
266
+ }
267
+ });
268
+ if (available.some((action) => !OUTSIDE_BATCH.has(action)))
269
+ available.push('batch');
270
+ return available;
271
+ }
272
+ function unavailableHostActions(caps) {
111
273
  const unavailable = [];
112
274
  if (!caps.screenshot)
113
275
  unavailable.push('screenshot');
@@ -118,13 +280,21 @@ function buildDescription(host) {
118
280
  if (!caps.keyboard)
119
281
  unavailable.push('keyboard');
120
282
  if (caps.supportedActions) {
121
- for (const action of actionSchema.options.map((option) => option.shape.type.value)) {
122
- if (!available.includes(action))
283
+ for (const action of HOST_ACTIONS) {
284
+ if (!hostActionAvailable(caps, action) && !unavailable.includes(action))
123
285
  unavailable.push(action);
124
286
  }
125
287
  }
288
+ return unavailable;
289
+ }
290
+ // ---------------------------------------------------------------------------
291
+ // Model-facing description and schema
292
+ // ---------------------------------------------------------------------------
293
+ function buildDescription(host, caps, settings) {
294
+ const available = availableActions(host, caps);
295
+ const unavailable = unavailableHostActions(caps);
126
296
  const lines = [
127
- `Controls the user's desktop on a ${caps.displayServer} host. Use to take screenshots and drive mouse/keyboard input for GUI tasks.`,
297
+ `Controls the user's desktop on a ${caps.displayServer} host: screenshots, mouse and keyboard, for GUI tasks.`,
128
298
  `Available actions: ${available.join('; ') || 'none'}.`,
129
299
  ];
130
300
  if (unavailable.length > 0) {
@@ -132,31 +302,79 @@ function buildDescription(host) {
132
302
  ? `Unavailable on this host: ${unavailable.join(', ')} — ${caps.unavailableReason} Do not retry; tell the user.`
133
303
  : `Unavailable on this host: ${unavailable.join(', ')}.`);
134
304
  }
135
- lines.push('Coordinates are in logical pixels from the top-left of the primary display. Call getDisplayGeometry through screenshot output before clicking to confirm bounds.');
305
+ if (!available.includes('screenshot'))
306
+ return finish(lines, caps, available);
307
+ lines.push('Coordinates: every x/y you send is a pixel of the most recent screenshot this tool returned — origin at its top-left, x to the right, y down — never a screen pixel. Each screenshot states its id and size (for example "s3: 1456x819"); stay inside it. The tool maps your coordinates onto the display. To aim at an earlier screenshot, pass its id as screenshot_id.', 'Take a screenshot before your first click.');
308
+ if (settings.screenshotAfterActions)
309
+ lines.push(`After any action that changes something, the tool waits ${settings.settleMs} ms and returns a new screenshot — do not call screenshot again after acting.`);
310
+ lines.push('zoom returns a closer, sharper view of region {x, y, width, height} of the screenshot; your coordinates still refer to the screenshot, not to the zoomed image.', `wait pauses for ms milliseconds (at most ${settings.maxWaitMs} per call, a batch's waits together) and then shows the screen.`);
311
+ if (available.includes('batch'))
312
+ lines.push(`batch: {"type":"batch","actions":[...]} runs up to ${settings.maxBatchActions} actions in order, stops at the first one that fails, and returns one screenshot at the end. Every call costs a model round trip of several seconds, so put the steps you can already predict into one batch (click a field, type, press ENTER) rather than one call each — but end the batch at any step that should bring up a new window, and look before typing into it. A batch cannot contain screenshot${available.includes('ui_snapshot') ? ', zoom or ui_snapshot' : ' or zoom'}.`);
313
+ if (caps.displayServer === 'win32' &&
314
+ available.includes('key') &&
315
+ available.includes('type_text'))
316
+ lines.push('To start a Windows program, a shell command (Start-Process notepad, or cmd.exe /c start calc) is surest when you have a shell tool. The Start menu searches by display names in the system language, and ENTER on a name it does not find opens a web search in the browser.');
317
+ if (available.includes('list_windows'))
318
+ lines.push('list_windows names the open windows with their ids; focus_window brings one to the front.');
319
+ if (available.includes('ui_snapshot'))
320
+ lines.push(`ui_snapshot {window_id} reads a window's controls (its accessibility tree) as text, each control you can act on with a ref such as e12; without window_id it reads the window in front. ui_act {ref, action, value} acts on one control: invoke (press a button, open a menu item), set_value (replace a field's text with value), toggle, select, expand, collapse. Prefer ui_act to clicking pixels when the control is in the tree: it does not depend on coordinates or on which window is in front, and a batch of ui_act steps (for example pressing several buttons) runs in one call. Refs are valid only until the next ui_snapshot; take a new one after the window changes.`);
321
+ lines.push('Typing and keys go to whichever window has focus: confirm on a screenshot that the right window is in front before type_text or key.');
322
+ if (available.includes('list_windows'))
323
+ lines.push('type_text and key are refused while a terminal window is in front — usually the one running this agent, or the user’s own; bring the window you mean to the front first (focus_window, or click it).');
324
+ return finish(lines, caps, available);
325
+ }
326
+ function finish(lines, caps, available) {
136
327
  if (caps.mouseClickButtons && available.includes('mouse_click'))
137
328
  lines.push(`Click buttons: ${caps.mouseClickButtons.join(', ') || 'none'}.`);
138
329
  if (caps.mouseDragButtons && available.includes('mouse_drag'))
139
330
  lines.push(`Drag buttons: ${caps.mouseDragButtons.join(', ') || 'none'}.`);
140
331
  return lines.join(' ');
141
332
  }
142
- function availableActions(caps) {
143
- return actionSchema.options
144
- .map((option) => option.shape.type.value)
145
- .filter((action) => {
146
- const required = requiredCapability(action);
147
- return ((required === null || caps[required] === true) &&
148
- (caps.supportedActions === undefined || caps.supportedActions.includes(action)) &&
149
- (action !== 'mouse_click' || caps.mouseClickButtons?.length !== 0) &&
150
- (action !== 'mouse_drag' || caps.mouseDragButtons?.length !== 0));
151
- });
333
+ /**
334
+ * The provider-facing shape is deliberately flat.
335
+ *
336
+ * The runtime schema above is the authoritative contract: it knows which
337
+ * fields each action requires. Rendering that discriminated union produces a
338
+ * root `anyOf`, however, and some custom-tool wires reject root combinators
339
+ * even when every branch is an object. A model can still see every field and
340
+ * every action here; incomplete combinations are rejected by `actionSchema`
341
+ * before the host is called, with the recovery hint below.
342
+ */
343
+ function pointModelSchema(description) {
344
+ return {
345
+ type: 'object',
346
+ ...(description ? { description } : {}),
347
+ properties: {
348
+ x: { type: 'integer' },
349
+ y: { type: 'integer' },
350
+ },
351
+ required: ['x', 'y'],
352
+ additionalProperties: false,
353
+ };
152
354
  }
153
- function hostModelSchema(caps) {
154
- const schema = structuredClone(modelInputSchema);
155
- const actions = availableActions(caps);
355
+ const ACTION_REQUIREMENTS = {
356
+ screenshot: 'screenshot needs no other fields',
357
+ zoom: 'zoom needs region',
358
+ cursor_position: 'cursor_position needs no other fields',
359
+ mouse_move: 'mouse_move needs to',
360
+ mouse_click: 'mouse_click needs at (button defaults to left)',
361
+ mouse_drag: 'mouse_drag needs from and to (button defaults to left)',
362
+ scroll: 'scroll needs at, direction, and amount',
363
+ type_text: 'type_text needs text',
364
+ key: 'key needs keys',
365
+ wait: 'wait needs ms',
366
+ list_windows: 'list_windows needs no other fields',
367
+ focus_window: 'focus_window needs window_id',
368
+ ui_snapshot: 'ui_snapshot takes an optional window_id',
369
+ ui_act: 'ui_act needs ref and action, and value for set_value',
370
+ batch: 'batch needs actions',
371
+ };
372
+ function hostModelSchema(host, caps, settings) {
156
373
  // An unavailable host remains a diagnostic tool. Avoid invalid empty enums
157
374
  // on provider wires; every execution is still refused before host access.
158
- if (actions.length > 0)
159
- schema.properties.type.enum = actions;
375
+ const offered = availableActions(host, caps);
376
+ const actions = offered.length > 0 ? offered : ALL_ACTIONS.filter((a) => a !== 'batch');
377
+ const items = actions.filter((action) => !OUTSIDE_BATCH.has(action));
160
378
  const buttons = new Set();
161
379
  if (actions.includes('mouse_click'))
162
380
  for (const button of caps.mouseClickButtons ?? ['left', 'right', 'middle'])
@@ -164,10 +382,93 @@ function hostModelSchema(caps) {
164
382
  if (actions.includes('mouse_drag'))
165
383
  for (const button of caps.mouseDragButtons ?? ['left', 'right', 'middle'])
166
384
  buttons.add(button);
167
- if (buttons.size > 0)
168
- schema.properties.button.enum = [...buttons];
169
- return schema;
385
+ const buttonEnum = buttons.size > 0 ? [...buttons] : ['left', 'right', 'middle'];
386
+ const fieldSchemas = (forItems) => {
387
+ const present = forItems ? items : actions;
388
+ const fields = {
389
+ to: pointModelSchema(),
390
+ at: pointModelSchema(),
391
+ from: pointModelSchema(),
392
+ button: { type: 'string', enum: buttonEnum, description: 'Mouse button; left when omitted.' },
393
+ direction: { type: 'string', enum: ['up', 'down', 'left', 'right'] },
394
+ amount: { type: 'integer', description: 'Positive integer scroll distance.' },
395
+ text: { type: 'string', description: 'Literal text to type.' },
396
+ keys: {
397
+ type: 'string',
398
+ description: 'Key or key chord to press, for example ENTER or CTRL+R.',
399
+ },
400
+ ms: {
401
+ type: 'integer',
402
+ description: `Milliseconds to wait, 0 to ${settings.maxWaitMs}.`,
403
+ },
404
+ };
405
+ if (present.includes('focus_window') || (!forItems && present.includes('ui_snapshot')))
406
+ fields.window_id = { type: 'string', description: 'A window id from list_windows.' };
407
+ if (present.includes('ui_act')) {
408
+ fields.ref = {
409
+ type: 'string',
410
+ description: 'A ref from the latest ui_snapshot, such as e12.',
411
+ };
412
+ fields.action = {
413
+ type: 'string',
414
+ enum: [...UI_ACTIONS],
415
+ description: 'What ui_act does to the control.',
416
+ };
417
+ fields.value = { type: 'string', description: 'The text set_value puts in the control.' };
418
+ }
419
+ if (!forItems && present.includes('zoom'))
420
+ fields.region = {
421
+ type: 'object',
422
+ description: 'Area of the screenshot to zoom into, in its pixels.',
423
+ properties: {
424
+ x: { type: 'integer' },
425
+ y: { type: 'integer' },
426
+ width: { type: 'integer' },
427
+ height: { type: 'integer' },
428
+ },
429
+ required: ['x', 'y', 'width', 'height'],
430
+ additionalProperties: false,
431
+ };
432
+ return fields;
433
+ };
434
+ const properties = {
435
+ type: {
436
+ type: 'string',
437
+ enum: actions,
438
+ description: `Desktop action. ${actions.map((action) => ACTION_REQUIREMENTS[action]).join('; ')}.`,
439
+ },
440
+ ...fieldSchemas(false),
441
+ };
442
+ if (actions.includes('batch'))
443
+ properties.actions = {
444
+ type: 'array',
445
+ minItems: 1,
446
+ maxItems: settings.maxBatchActions,
447
+ description: `For type batch: up to ${settings.maxBatchActions} actions run in order (no screenshot, zoom, ui_snapshot or batch inside).`,
448
+ items: {
449
+ type: 'object',
450
+ properties: {
451
+ type: { type: 'string', enum: items },
452
+ ...fieldSchemas(true),
453
+ },
454
+ required: ['type'],
455
+ additionalProperties: false,
456
+ },
457
+ };
458
+ properties.screenshot_id = {
459
+ type: 'string',
460
+ description: 'Optional id of the screenshot your coordinates were read from, such as s3. Defaults to the latest.',
461
+ };
462
+ return {
463
+ type: 'object',
464
+ properties,
465
+ required: ['type'],
466
+ additionalProperties: false,
467
+ };
170
468
  }
469
+ // ---------------------------------------------------------------------------
470
+ // Labels
471
+ // ---------------------------------------------------------------------------
171
472
  function pointLabel(point) {
172
473
  return `(${point.x}, ${point.y})`;
173
474
  }
@@ -176,11 +477,41 @@ function quotedText(value) {
176
477
  const visible = oneLine.length > 64 ? `${oneLine.slice(0, 63)}…` : oneLine;
177
478
  return JSON.stringify(visible);
178
479
  }
179
- /** Human activity text; the raw action union remains the model-facing input. */
180
- function actionLabel(input) {
480
+ function waitLabel(ms) {
481
+ return ms >= 1000 ? `Wait ${Number((ms / 1000).toFixed(1))} s` : `Wait ${ms} ms`;
482
+ }
483
+ /** What a ui_act does, as a verb phrase over the control's description. */
484
+ function uiActLabel(input, describe) {
485
+ const target = describe?.(input.ref) ?? input.ref;
486
+ switch (input.action) {
487
+ case 'invoke':
488
+ return `Press ${target}`;
489
+ case 'set_value':
490
+ return `Set ${target} to ${quotedText(input.value ?? '')}`;
491
+ case 'toggle':
492
+ return `Toggle ${target}`;
493
+ case 'select':
494
+ return `Select ${target}`;
495
+ case 'expand':
496
+ return `Expand ${target}`;
497
+ case 'collapse':
498
+ return `Collapse ${target}`;
499
+ case 'focus':
500
+ return `Focus ${target}`;
501
+ case 'scroll_into_view':
502
+ return `Scroll to ${target}`;
503
+ }
504
+ }
505
+ /**
506
+ * Human activity text for one action; the raw input remains the model-facing
507
+ * record. `describe` names a ui_act ref's control when the tool knows it.
508
+ */
509
+ function itemLabel(input, describe) {
181
510
  switch (input.type) {
182
511
  case 'screenshot':
183
512
  return 'Capture screenshot';
513
+ case 'zoom':
514
+ return `Zoom into ${input.region.width}x${input.region.height} at ${pointLabel(input.region)}`;
184
515
  case 'cursor_position':
185
516
  return 'Read cursor position';
186
517
  case 'mouse_move':
@@ -195,33 +526,49 @@ function actionLabel(input) {
195
526
  return `Type ${quotedText(input.text)}`;
196
527
  case 'key':
197
528
  return `Press ${input.keys}`;
529
+ case 'wait':
530
+ return waitLabel(input.ms);
531
+ case 'list_windows':
532
+ return 'List windows';
533
+ case 'focus_window':
534
+ return `Focus window ${input.window_id}`;
535
+ case 'ui_snapshot':
536
+ return input.window_id
537
+ ? `Read the controls of window ${input.window_id}`
538
+ : 'Read the controls of the front window';
539
+ case 'ui_act':
540
+ return uiActLabel(input, describe);
541
+ case 'batch':
542
+ return batchLabel(input.actions, describe);
198
543
  }
199
544
  }
200
- function resultToToolResult(result) {
201
- switch (result.type) {
202
- case 'screenshot': {
203
- const { data, mimeType, width, height } = result.result;
204
- // `output` used to BE the base64 payload, which meant the model
205
- // received 400 KB–2.7 MB of undecodable characters as text —
206
- // roughly 100k–670k tokens — and could not see the screen at
207
- // all. The image now travels as a content block; `output` keeps
208
- // the short human/transcript-facing description.
209
- return {
210
- success: true,
211
- output: `Screenshot captured (${width}x${height}, ${mimeType}).`,
212
- content: [{ type: 'image', data: data.toString('base64'), mediaType: mimeType }],
213
- data: { mimeType, width, height, encoding: 'base64' },
214
- };
215
- }
216
- case 'cursor_position':
217
- return {
218
- success: true,
219
- output: JSON.stringify(result.point),
220
- data: result.point,
221
- };
222
- case 'ok':
223
- return { success: true, output: 'ok' };
224
- }
545
+ function batchLabel(actions, describe) {
546
+ const parts = actions.map((item) => itemLabel(item, describe));
547
+ const head = `${actions.length} desktop action${actions.length === 1 ? '' : 's'}`;
548
+ const joined = parts.join(' · ');
549
+ return joined.length > 240 ? `${head}: ${joined.slice(0, 239)}…` : `${head}: ${joined}`;
550
+ }
551
+ function resolveSettings(options) {
552
+ const count = (value, fallback, min, name) => {
553
+ if (value === undefined)
554
+ return fallback;
555
+ if (!Number.isSafeInteger(value) || value < min)
556
+ throw new RangeError(`createComputerUseTool: ${name} must be an integer ≥ ${min}`);
557
+ return value;
558
+ };
559
+ const limits = options.screenshotLimits ?? STANDARD_SCREENSHOT_LIMITS;
560
+ if (!Number.isSafeInteger(limits.maxLongEdge) ||
561
+ !Number.isSafeInteger(limits.maxTiles) ||
562
+ limits.maxLongEdge < 28 ||
563
+ limits.maxTiles < 1)
564
+ throw new RangeError('createComputerUseTool: screenshotLimits needs an integer maxLongEdge ≥ 28 and maxTiles ≥ 1');
565
+ return {
566
+ limits,
567
+ settleMs: count(options.settleMs, DEFAULT_SETTLE_MS, 0, 'settleMs'),
568
+ screenshotAfterActions: options.screenshotAfterActions ?? true,
569
+ maxBatchActions: count(options.maxBatchActions, DEFAULT_MAX_BATCH_ACTIONS, 1, 'maxBatchActions'),
570
+ maxWaitMs: count(options.maxWaitMs, DEFAULT_MAX_WAIT_MS, 0, 'maxWaitMs'),
571
+ };
225
572
  }
226
573
  function isOutcomeUnknown(value, action) {
227
574
  if (typeof value !== 'object' || value === null)
@@ -237,30 +584,81 @@ function isOutcomeUnknown(value, action) {
237
584
  typeof candidate.message === 'string' &&
238
585
  candidate.message.length > 0);
239
586
  }
240
- function unknownOutcomeToToolResult(error) {
241
- return {
242
- success: false,
243
- output: '',
244
- error: error.message,
245
- data: {
246
- code: error.code,
247
- action: error.action,
248
- outcome: error.outcome,
249
- retrySafety: error.retrySafety,
250
- timedOut: error.timedOut,
251
- exitCode: error.exitCode,
252
- },
253
- };
587
+ function errorText(error) {
588
+ return error instanceof Error ? error.message : String(error);
589
+ }
590
+ class StepFailure extends Error {
591
+ unknown;
592
+ constructor(message, unknown) {
593
+ super(message);
594
+ this.unknown = unknown;
595
+ }
596
+ }
597
+ /** Whether a call sends the screen to the provider (see `ToolDefinition.capturesScreen`). */
598
+ function capturesScreenInput(input, screenshotAfterActions) {
599
+ const items = batchActions(input) ?? [input];
600
+ return items.some((item) => {
601
+ if (!isRecord(item))
602
+ return true;
603
+ const type = String(item.type);
604
+ if (SCREEN_OBSERVATIONS.has(type))
605
+ return true;
606
+ if (type === 'cursor_position')
607
+ return false;
608
+ // Everything else is followed by a screenshot unless that is switched off.
609
+ return screenshotAfterActions;
610
+ });
611
+ }
612
+ function safeRole(role) {
613
+ return /^[A-Za-z][\w-]{0,39}$/.test(role) ? role : 'Element';
614
+ }
615
+ /** `Button "Beş"`: the control's role and name, one line, cut to fit. */
616
+ function uiElementText(element) {
617
+ const name = element.name.trim();
618
+ return name.length > 0 ? `${safeRole(element.role)} ${quotedText(name)}` : safeRole(element.role);
619
+ }
620
+ /**
621
+ * One line of a ui_snapshot: `[e12] Button "Beş" (disabled) [invoke] @(212, 488)`.
622
+ * The ref and position are the tool's; the rest is the application's, and
623
+ * sits inside the untrusted frame.
624
+ */
625
+ function uiElementLine(element, ref, frame) {
626
+ const parts = [`${ref ? `[${ref}] ` : ''}${uiElementText(element)}`];
627
+ if (element.value !== undefined && element.value !== element.name)
628
+ parts.push(`value=${quotedText(element.value)}`);
629
+ const states = (element.states ?? []).filter((state) => /^[a-z_]{1,24}$/.test(state));
630
+ if (states.length > 0)
631
+ parts.push(`(${states.join(', ')})`);
632
+ if (ref) {
633
+ const actions = (element.actions ?? []).filter((action) => UI_ACTIONS.includes(action));
634
+ if (actions.length > 0)
635
+ parts.push(`[${actions.join(', ')}]`);
636
+ const at = frame && element.bounds ? centreOnImage(frame, element.bounds) : undefined;
637
+ if (at)
638
+ parts.push(`@${pointLabel(at)}`);
639
+ }
640
+ return neutralizeEnvelopeDelimiter(parts.join(' '));
641
+ }
642
+ /** The centre of a virtual-desktop rectangle on a screenshot, or undefined when it is not on its display. */
643
+ function centreOnImage(frame, bounds) {
644
+ const x = Math.floor(bounds.x + bounds.width / 2) - frame.display.x;
645
+ const y = Math.floor(bounds.y + bounds.height / 2) - frame.display.y;
646
+ if (x < 0 || y < 0 || x >= frame.display.width || y >= frame.display.height)
647
+ return undefined;
648
+ return toImagePoint(frame, { x, y });
254
649
  }
255
650
  /**
256
651
  * Factory: given a ComputerUseHost (provided by the consumer — e.g.
257
- * @namzu/computer-use's SubprocessComputerUseHost), returns a ToolDefinition
258
- * that routes the discriminated action to the host and maps results back to
259
- * the SDK's ToolResult shape.
652
+ * @namzu/computer-use's SubprocessComputerUseHost), returns the
653
+ * `computer_use` tool.
260
654
  *
261
- * The tool's description reflects the host's frozen capabilities, and any
262
- * action targeting an unavailable capability is rejected with a clear error
263
- * rather than hanging or failing silently.
655
+ * Every screenshot is fitted to the model's image limits and numbered;
656
+ * every coordinate the model sends is a pixel of one of those screenshots and
657
+ * is mapped onto the host's display. Actions that change something return a
658
+ * fresh screenshot after a short settle delay, and `batch` runs several in
659
+ * one call. The description and schema reflect the host's frozen
660
+ * capabilities, and any action targeting an unavailable capability is
661
+ * rejected before the host is touched.
264
662
  *
265
663
  * @example
266
664
  * ```ts
@@ -272,65 +670,672 @@ function unknownOutcomeToToolResult(error) {
272
670
  * registry.register(createComputerUseTool(host))
273
671
  * ```
274
672
  */
275
- export function createComputerUseTool(host) {
276
- return defineTool({
673
+ export function createComputerUseTool(host, options = {}) {
674
+ const settings = resolveSettings(options);
675
+ const caps = options.unavailableReason
676
+ ? {
677
+ ...host.capabilities,
678
+ screenshot: false,
679
+ mouse: false,
680
+ keyboard: false,
681
+ cursorPosition: false,
682
+ clipboard: false,
683
+ windows: false,
684
+ regionCapture: false,
685
+ uiTree: false,
686
+ unavailableReason: options.unavailableReason,
687
+ }
688
+ : host.capabilities;
689
+ const frames = new ScreenshotFrames();
690
+ const available = new Set(availableActions(host, caps));
691
+ // The controls of the latest ui_snapshot, by the ref the model was shown.
692
+ // Refs count up across snapshots, so a ref from an earlier one is never
693
+ // silently a different control of the latest: it is simply not here.
694
+ let uiRefs = new Map();
695
+ let uiSnapshots = 0;
696
+ let uiRefCount = 0;
697
+ const describeUiRef = (ref) => {
698
+ const entry = uiRefs.get(ref);
699
+ return entry ? `${uiElementText(entry.element)} (${ref})` : undefined;
700
+ };
701
+ const label = (input) => itemLabel(input, describeUiRef);
702
+ const refusal = (type) => {
703
+ const required = requiredCapability(type);
704
+ const why = caps.unavailableReason
705
+ ? ` ${caps.unavailableReason} Do not retry; tell the user.`
706
+ : '';
707
+ if (required !== null && caps[required] !== true)
708
+ return `computer_use: action "${type}" requires capability "${required}" which is not available on this host (displayServer=${caps.displayServer}).${why}`;
709
+ if (!available.has(type))
710
+ return `computer_use: action "${type}" is not supported on this host.${why}`;
711
+ return null;
712
+ };
713
+ /** Capture the display, fit it, and number it. */
714
+ const capture = async () => {
715
+ const result = await host.execute({ type: 'screenshot' });
716
+ if (result.type !== 'screenshot')
717
+ throw new Error(`computer_use: the host answered a screenshot with "${result.type}"`);
718
+ const shot = result.result;
719
+ const image = await fitPng(shot.data, settings.limits);
720
+ const display = shot.display ?? assumedDisplay(image.sourceWidth, image.sourceHeight);
721
+ return { frame: frames.record(image, display), image };
722
+ };
723
+ const frameFor = (id) => {
724
+ if (id !== undefined) {
725
+ const frame = frames.get(id);
726
+ if (!frame)
727
+ throw new StepFailure(`there is no screenshot ${id} (the latest is ${frames.latest()?.id ?? 'none'}); take a screenshot and use its coordinates`);
728
+ return frame;
729
+ }
730
+ const latest = frames.latest();
731
+ if (!latest)
732
+ throw new StepFailure('no screenshot has been taken yet, so there is nothing for coordinates to refer to; take a screenshot first');
733
+ return latest;
734
+ };
735
+ /**
736
+ * Keys go to whichever window has focus, and a model that has not looked
737
+ * does not know which one that is — the terminal running this agent is
738
+ * the usual answer. When the host can show the screen, look first.
739
+ */
740
+ const requireLook = () => {
741
+ if (available.has('screenshot') && !frames.latest())
742
+ throw new StepFailure('no screenshot has been taken yet, so nothing shows which window has focus and would receive the keys; take a screenshot first');
743
+ };
744
+ const mapPoint = (frame, point, field) => {
745
+ if (!pointOnImage(frame, point))
746
+ throw new StepFailure(`${field} ${pointLabel(point)} is outside screenshot ${frame.id}, which is ${frame.imageWidth}x${frame.imageHeight} (x 0–${frame.imageWidth - 1}, y 0–${frame.imageHeight - 1}); coordinates are pixels of the screenshot, not of the screen`);
747
+ return toDisplayPoint(frame, point);
748
+ };
749
+ const runHost = async (action) => {
750
+ try {
751
+ const result = await host.execute(action);
752
+ if (result.type === 'cursor_position') {
753
+ const frame = frames.latest();
754
+ if (!frame)
755
+ return `at display pixel ${pointLabel(result.point)}`;
756
+ return `at ${pointLabel(toImagePoint(frame, result.point))} in ${frame.id}`;
757
+ }
758
+ return 'done';
759
+ }
760
+ catch (error) {
761
+ if (isOutcomeUnknown(error, action.type))
762
+ throw new StepFailure(error.message, error);
763
+ throw new StepFailure(errorText(error));
764
+ }
765
+ };
766
+ const plan = (item, frameId) => {
767
+ const denied = refusal(item.type);
768
+ if (denied)
769
+ throw new StepFailure(denied);
770
+ const mutating = !READ_ONLY_ACTIONS.has(item.type);
771
+ const step = (run) => ({
772
+ item,
773
+ label: label(item),
774
+ mutating,
775
+ run,
776
+ });
777
+ switch (item.type) {
778
+ case 'cursor_position':
779
+ frameFor(frameId);
780
+ return step(() => runHost({ type: 'cursor_position' }));
781
+ case 'mouse_move': {
782
+ const frame = frameFor(frameId);
783
+ const to = mapPoint(frame, item.to, 'to');
784
+ return step(() => runHost({ type: 'mouse_move', to }));
785
+ }
786
+ case 'mouse_click': {
787
+ const buttons = caps.mouseClickButtons;
788
+ if (buttons && !buttons.includes(item.button))
789
+ throw new StepFailure(`computer_use: action "mouse_click" does not support button "${item.button}" on this host.`);
790
+ const frame = frameFor(frameId);
791
+ const at = mapPoint(frame, item.at, 'at');
792
+ return step(() => runHost({ type: 'mouse_click', at, button: item.button }));
793
+ }
794
+ case 'mouse_drag': {
795
+ const buttons = caps.mouseDragButtons;
796
+ if (buttons && !buttons.includes(item.button))
797
+ throw new StepFailure(`computer_use: action "mouse_drag" does not support button "${item.button}" on this host.`);
798
+ const frame = frameFor(frameId);
799
+ const from = mapPoint(frame, item.from, 'from');
800
+ const to = mapPoint(frame, item.to, 'to');
801
+ return step(() => runHost({ type: 'mouse_drag', from, to, button: item.button }));
802
+ }
803
+ case 'scroll': {
804
+ const frame = frameFor(frameId);
805
+ const at = mapPoint(frame, item.at, 'at');
806
+ return step(() => runHost({ type: 'scroll', at, direction: item.direction, amount: item.amount }));
807
+ }
808
+ case 'type_text':
809
+ requireLook();
810
+ return step(() => runHost({ type: 'type_text', text: item.text }));
811
+ case 'key':
812
+ requireLook();
813
+ return step(() => runHost({ type: 'key', keys: item.keys }));
814
+ case 'wait':
815
+ return step(async (signal) => {
816
+ await sleep(item.ms, signal);
817
+ return 'done';
818
+ });
819
+ case 'list_windows':
820
+ return step(() => listWindows());
821
+ case 'focus_window':
822
+ return step(async () => {
823
+ let outcome;
824
+ try {
825
+ outcome = await host.focusWindow(item.window_id);
826
+ }
827
+ catch (error) {
828
+ throw new StepFailure(errorText(error));
829
+ }
830
+ if (!outcome.ok)
831
+ throw new StepFailure(`window ${item.window_id} could not be brought to the front; ${outcome.focusedId ? `window ${outcome.focusedId} is in front` : 'the window in front could not be read'}`);
832
+ return 'done';
833
+ });
834
+ case 'ui_act': {
835
+ const entry = uiRefs.get(item.ref);
836
+ if (!entry)
837
+ throw new StepFailure(uiSnapshots === 0
838
+ ? `there is no ${item.ref}: no ui_snapshot has been taken yet; take one and use its refs`
839
+ : `${item.ref} is not a control of the latest ui_snapshot (u${uiSnapshots}); refs are valid only until the next ui_snapshot, so use the refs it showed`);
840
+ const offered = entry.element.actions;
841
+ if (offered && offered.length > 0 && !offered.includes(item.action))
842
+ throw new StepFailure(`${describeUiRef(item.ref)} offers ${offered.join(', ')}, not ${item.action}`);
843
+ if (item.action === 'set_value' && item.value === undefined)
844
+ throw new StepFailure('set_value needs value, the text to put in the control');
845
+ return step(async () => {
846
+ let outcome;
847
+ try {
848
+ outcome = await host.uiAct(entry.hostRef, item.action, item.value);
849
+ }
850
+ catch (error) {
851
+ throw new StepFailure(errorText(error));
852
+ }
853
+ if (!outcome.ok)
854
+ throw new StepFailure(outcome.detail ?? `${item.action} did not take effect on ${describeUiRef(item.ref)}`);
855
+ return outcome.detail ? `done (${outcome.detail})` : 'done';
856
+ });
857
+ }
858
+ }
859
+ };
860
+ /** Read one window's controls, number the ones the model can act on, and show them. */
861
+ const uiSnapshot = async (input) => {
862
+ let snapshot;
863
+ try {
864
+ snapshot = await host.uiSnapshot(input.window_id);
865
+ }
866
+ catch (error) {
867
+ throw new StepFailure(errorText(error));
868
+ }
869
+ uiSnapshots += 1;
870
+ const id = `u${uiSnapshots}`;
871
+ const refs = new Map();
872
+ const frame = frames.latest();
873
+ const lines = [];
874
+ let chars = 0;
875
+ let shown = 0;
876
+ let total = 0;
877
+ let cut = false;
878
+ const visit = (element, depth) => {
879
+ total += 1;
880
+ const actionable = element.ref.length > 0;
881
+ const children = element.children ?? [];
882
+ // A nameless control nobody can act on says nothing by itself; its
883
+ // children still stand, one level up.
884
+ const silent = !actionable && element.name.trim().length === 0 && element.value === undefined;
885
+ if (!silent && !cut) {
886
+ const ref = actionable ? `e${uiRefCount + 1}` : undefined;
887
+ const line = `${' '.repeat(depth)}${uiElementLine(element, ref, frame)}`;
888
+ if (chars + line.length + 1 > MAX_UI_SNAPSHOT_CHARS) {
889
+ cut = true;
890
+ }
891
+ else {
892
+ lines.push(line);
893
+ chars += line.length + 1;
894
+ shown += 1;
895
+ if (ref) {
896
+ uiRefCount += 1;
897
+ refs.set(ref, { hostRef: element.ref, element });
898
+ }
899
+ }
900
+ }
901
+ for (const child of children)
902
+ visit(child, silent ? depth : depth + 1);
903
+ };
904
+ visit(snapshot.root, 0);
905
+ uiRefs = refs;
906
+ const window = snapshot.windowId && /^[\w.:-]{1,64}$/.test(snapshot.windowId) ? snapshot.windowId : undefined;
907
+ const header = [
908
+ `UI snapshot ${id}${window ? ` of window ${window}` : ''}: ${shown} controls shown, ${refs.size} with a ref you can pass to ui_act. Refs are valid until the next ui_snapshot.`,
909
+ ...(frame
910
+ ? [
911
+ `@(x, y) is a control's centre on screenshot ${frame.id}, for a click when ui_act cannot reach it.`,
912
+ ]
913
+ : []),
914
+ ];
915
+ // The window's title and application, when the tree's own root does not
916
+ // already say them; inside the frame, since an application sets both.
917
+ const about = [
918
+ ...(snapshot.app !== undefined ? [`Application ${quotedText(snapshot.app)}`] : []),
919
+ ...(snapshot.title !== undefined && snapshot.title !== snapshot.root.name
920
+ ? [`Window title ${quotedText(snapshot.title)}`]
921
+ : []),
922
+ ];
923
+ const body = [...about, ...lines].join('\n');
924
+ const footer = cut || snapshot.truncated
925
+ ? [
926
+ cut
927
+ ? `The tree was cut at ${MAX_UI_SNAPSHOT_CHARS} characters after ${shown} of ${total} controls. Read a smaller window, or use a screenshot for the rest.`
928
+ : 'The host stopped reading the tree before it ended; controls further down are missing. Use a screenshot for the rest.',
929
+ ]
930
+ : [];
931
+ const text = [
932
+ ...header,
933
+ wrapUntrusted({
934
+ kind: 'desktop-ui',
935
+ ...(window ? { attributes: { window } } : {}),
936
+ provenance: "The accessibility tree of a window on the user's desktop, as the host read it. Names and values are whatever the application shows, which can include text anyone wrote.",
937
+ }, body),
938
+ ...footer,
939
+ ].join('\n');
940
+ return {
941
+ success: true,
942
+ output: header[0] ?? '',
943
+ content: [{ type: 'text', text }],
944
+ data: {
945
+ uiSnapshot: {
946
+ id,
947
+ ...(window ? { windowId: window } : {}),
948
+ controls: shown,
949
+ refs: refs.size,
950
+ truncated: cut || snapshot.truncated === true,
951
+ },
952
+ },
953
+ };
954
+ };
955
+ /**
956
+ * Keys and text go to the window in front, and a terminal there — the
957
+ * one running this agent, or the user's own — turns a model's typing into
958
+ * a command line: ENTER runs it or sends it. A session had a WIN+R that
959
+ * did not open the Run dialog type "notepad" and ENTER into the user's
960
+ * terminal, and submitted their half-typed message. So with a window list
961
+ * the tool reads what is in front before typing, and refuses a terminal.
962
+ * Without one it cannot tell, and the description's advice stands alone.
963
+ */
964
+ const refuseTerminalInFront = async () => {
965
+ if (!available.has('list_windows'))
966
+ return;
967
+ let windows;
968
+ try {
969
+ windows = await host.listWindows();
970
+ }
971
+ catch (error) {
972
+ throw new StepFailure(`the window in front could not be read, so nothing was typed: ${errorText(error)}`);
973
+ }
974
+ const front = windows.find((window) => window.focused);
975
+ if (front && TERMINAL_APPS.test(front.app))
976
+ throw new StepFailure(`a terminal is in front (${front.app}, window ${front.id}), so nothing was typed: computer_use never sends keys or text to a terminal. Bring the window you mean to the front (focus_window, or click it) and try again; for a command, use a shell tool instead`);
977
+ };
978
+ const listWindows = async () => {
979
+ let windows;
980
+ try {
981
+ windows = await host.listWindows();
982
+ }
983
+ catch (error) {
984
+ throw new StepFailure(errorText(error));
985
+ }
986
+ if (windows.length === 0)
987
+ return 'no windows are open';
988
+ const frame = frames.latest();
989
+ const lines = windows.slice(0, MAX_LISTED_WINDOWS).map((window) => {
990
+ const where = frame
991
+ ? (() => {
992
+ const rect = window.minimized ? null : desktopRectOnImage(frame, window.bounds);
993
+ return rect
994
+ ? `at ${pointLabel(rect)} ${rect.width}x${rect.height} in ${frame.id}`
995
+ : window.minimized
996
+ ? 'minimized'
997
+ : `not on the display of ${frame.id}`;
998
+ })()
999
+ : window.minimized
1000
+ ? 'minimized'
1001
+ : '';
1002
+ return [
1003
+ `- ${window.id}`,
1004
+ JSON.stringify(window.title),
1005
+ `${window.app} (pid ${window.pid})`,
1006
+ ...(window.focused ? ['focused'] : []),
1007
+ ...(where ? [where] : []),
1008
+ ].join(' · ');
1009
+ });
1010
+ const more = windows.length > MAX_LISTED_WINDOWS
1011
+ ? [`… and ${windows.length - MAX_LISTED_WINDOWS} more`]
1012
+ : [];
1013
+ // Titles are whatever an application puts there — a web page's title in
1014
+ // a browser window — so the list is framed as material, not direction.
1015
+ return `${windows.length} window${windows.length === 1 ? '' : 's'}, front to back:\n${wrapUntrusted({
1016
+ kind: 'desktop-windows',
1017
+ provenance: "The open windows on the user's desktop, as the host listed them. Titles are whatever each application shows, which can include text anyone wrote.",
1018
+ }, [...lines, ...more].join('\n'))}`;
1019
+ };
1020
+ const describeFrame = (frame) => {
1021
+ const { display } = frame;
1022
+ const scaled = frame.imageWidth !== display.width || frame.imageHeight !== display.height;
1023
+ return scaled
1024
+ ? `Screenshot ${frame.id}: ${frame.imageWidth}x${frame.imageHeight} pixels, showing the ${display.width}x${display.height} display. Send coordinates in this image's pixels (x 0–${frame.imageWidth - 1}, y 0–${frame.imageHeight - 1}); the tool maps them onto the display.`
1025
+ : `Screenshot ${frame.id}: ${frame.imageWidth}x${frame.imageHeight} pixels, the display at full size. Send coordinates in this image's pixels (x 0–${frame.imageWidth - 1}, y 0–${frame.imageHeight - 1}).`;
1026
+ };
1027
+ const frameData = (frame) => ({
1028
+ id: frame.id,
1029
+ width: frame.imageWidth,
1030
+ height: frame.imageHeight,
1031
+ display: frame.display,
1032
+ mimeType: 'image/png',
1033
+ encoding: 'base64',
1034
+ });
1035
+ const pin = (frame) => [
1036
+ {
1037
+ key: 'computer_use.screenshot',
1038
+ text: `computer_use coordinates are pixels of screenshot ${frame.id} (${frame.imageWidth}x${frame.imageHeight}), origin top-left.`,
1039
+ },
1040
+ ];
1041
+ const imageBlock = (image) => ({
1042
+ type: 'image',
1043
+ data: image.data.toString('base64'),
1044
+ mediaType: 'image/png',
1045
+ });
1046
+ // Said once, beside the first screenshot, where it is read: a model that
1047
+ // has just seen the screen reaches for pixels unless told the host can
1048
+ // name the controls.
1049
+ const firstLookHint = (frame) => frame.id === 's1' && available.has('ui_snapshot')
1050
+ ? [
1051
+ 'This host can also read a window’s controls: list_windows, then ui_snapshot {window_id}, then ui_act by ref (a batch of them for several buttons) — surer than clicking pixels in an ordinary application.',
1052
+ ]
1053
+ : [];
1054
+ const screenshotResult = (shot) => ({
1055
+ success: true,
1056
+ output: `Screenshot ${shot.frame.id} captured (${shot.frame.imageWidth}x${shot.frame.imageHeight} of the ${shot.frame.display.width}x${shot.frame.display.height} display).`,
1057
+ content: [
1058
+ { type: 'text', text: [describeFrame(shot.frame), ...firstLookHint(shot.frame)].join('\n') },
1059
+ imageBlock(shot.image),
1060
+ ],
1061
+ data: { screenshot: frameData(shot.frame) },
1062
+ workingState: pin(shot.frame),
1063
+ });
1064
+ const zoom = async (input) => {
1065
+ const frame = frameFor(input.screenshot_id);
1066
+ const { region } = input;
1067
+ const onImage = {
1068
+ x: Math.max(0, region.x),
1069
+ y: Math.max(0, region.y),
1070
+ width: Math.min(region.x + region.width, frame.imageWidth) - Math.max(0, region.x),
1071
+ height: Math.min(region.y + region.height, frame.imageHeight) - Math.max(0, region.y),
1072
+ };
1073
+ if (onImage.width <= 0 || onImage.height <= 0)
1074
+ throw new StepFailure(`region ${region.width}x${region.height} at ${pointLabel(region)} is outside screenshot ${frame.id} (${frame.imageWidth}x${frame.imageHeight})`);
1075
+ const rect = toDisplayRect(frame, onImage);
1076
+ let fitted;
1077
+ if (caps.regionCapture === true && typeof host.captureRegion === 'function') {
1078
+ const piece = await host.captureRegion(rect);
1079
+ fitted = await fitPng(piece.data, settings.limits);
1080
+ }
1081
+ else {
1082
+ const result = await host.execute({ type: 'screenshot' });
1083
+ if (result.type !== 'screenshot')
1084
+ throw new Error(`computer_use: the host answered a screenshot with "${result.type}"`);
1085
+ const full = result.result;
1086
+ const size = pngSize(full.data);
1087
+ const display = full.display ?? assumedDisplay(size.width, size.height);
1088
+ if (display.width !== frame.display.width || display.height !== frame.display.height)
1089
+ throw new StepFailure(`the display is now ${display.width}x${display.height}, not what ${frame.id} showed; take a new screenshot`);
1090
+ // The capture's own pixels, in case a host captures at another scale than it reports.
1091
+ const sx = size.width / display.width;
1092
+ const sy = size.height / display.height;
1093
+ const left = Math.floor(rect.x * sx);
1094
+ const top = Math.floor(rect.y * sy);
1095
+ fitted = await cropAndFitPng(full.data, {
1096
+ x: left,
1097
+ y: top,
1098
+ width: Math.max(1, Math.min(size.width, Math.ceil((rect.x + rect.width) * sx)) - left),
1099
+ height: Math.max(1, Math.min(size.height, Math.ceil((rect.y + rect.height) * sy)) - top),
1100
+ }, settings.limits);
1101
+ }
1102
+ const detail = fitted.width / onImage.width;
1103
+ const text = `Zoom of ${onImage.width}x${onImage.height} at ${pointLabel(onImage)} in ${frame.id}, shown at ${fitted.width}x${fitted.height} (${detail.toFixed(1)}x the detail of ${frame.id}). Coordinates for actions still refer to ${frame.id} (${frame.imageWidth}x${frame.imageHeight}), not to this image.`;
1104
+ return {
1105
+ success: true,
1106
+ output: `Zoomed into ${onImage.width}x${onImage.height} at ${pointLabel(onImage)} of ${frame.id} (shown at ${fitted.width}x${fitted.height}).`,
1107
+ content: [{ type: 'text', text }, imageBlock(fitted)],
1108
+ data: {
1109
+ zoom: {
1110
+ screenshot: frame.id,
1111
+ region: onImage,
1112
+ width: fitted.width,
1113
+ height: fitted.height,
1114
+ mimeType: 'image/png',
1115
+ encoding: 'base64',
1116
+ },
1117
+ },
1118
+ };
1119
+ };
1120
+ /** Run actions in order, stop at the first failure, then show the screen once. */
1121
+ const runSteps = async (items, frameId, context, single) => {
1122
+ if (items.length > settings.maxBatchActions)
1123
+ return {
1124
+ success: false,
1125
+ output: '',
1126
+ error: `computer_use: a batch carries at most ${settings.maxBatchActions} actions; got ${items.length}. Nothing was run.`,
1127
+ };
1128
+ // Waits share one budget, so a batch stays well inside the tool's
1129
+ // deadline however its waits are split.
1130
+ const waited = items.reduce((total, item) => total + (item.type === 'wait' ? item.ms : 0), 0);
1131
+ if (waited > settings.maxWaitMs)
1132
+ return {
1133
+ success: false,
1134
+ output: '',
1135
+ error: `computer_use: ${single ? 'wait is' : "a batch's waits add up to"} at most ${settings.maxWaitMs} ms; got ${waited}. Nothing was run.`,
1136
+ };
1137
+ // Plan everything before doing anything: a refusal or a coordinate off
1138
+ // the screenshot in step 5 should not arrive after steps 1–4 changed
1139
+ // the desktop.
1140
+ const planned = [];
1141
+ for (const [index, item] of items.entries()) {
1142
+ try {
1143
+ planned.push(plan(item, frameId));
1144
+ }
1145
+ catch (error) {
1146
+ const message = errorText(error);
1147
+ return {
1148
+ success: false,
1149
+ output: '',
1150
+ error: single
1151
+ ? message
1152
+ : `computer_use: action ${index + 1} of ${items.length} (${label(item)}) cannot run: ${message}. Nothing was run.`,
1153
+ };
1154
+ }
1155
+ }
1156
+ const signal = context?.abortSignal;
1157
+ const records = [];
1158
+ let changed = false;
1159
+ let failure;
1160
+ // Whether the window in front was checked since the last step that could
1161
+ // have changed it. Typing does not move focus; everything else may.
1162
+ let frontChecked = false;
1163
+ for (const [index, step] of planned.entries()) {
1164
+ if (signal?.aborted) {
1165
+ failure = { index, error: 'cancelled before it started' };
1166
+ records.push({ label: step.label, status: 'failed', error: failure.error });
1167
+ break;
1168
+ }
1169
+ try {
1170
+ const keyboard = step.item.type === 'type_text' || step.item.type === 'key';
1171
+ if (keyboard && !frontChecked) {
1172
+ await refuseTerminalInFront();
1173
+ frontChecked = true;
1174
+ }
1175
+ if (step.item.type !== 'type_text')
1176
+ frontChecked = false;
1177
+ const note = await step.run(signal);
1178
+ records.push({ label: step.label, status: 'done', note });
1179
+ if (step.mutating)
1180
+ changed = true;
1181
+ }
1182
+ catch (error) {
1183
+ const unknown = error instanceof StepFailure ? error.unknown : undefined;
1184
+ const message = signal?.aborted ? 'cancelled' : errorText(error);
1185
+ failure = { index, error: message, ...(unknown ? { unknown } : {}) };
1186
+ records.push({
1187
+ label: step.label,
1188
+ status: 'failed',
1189
+ error: message,
1190
+ ...(unknown ? { unknown } : {}),
1191
+ });
1192
+ if (unknown)
1193
+ changed = true;
1194
+ break;
1195
+ }
1196
+ }
1197
+ for (const step of planned.slice(records.length))
1198
+ records.push({ label: step.label, status: 'not run' });
1199
+ // The screen after the batch: after anything that may have changed it,
1200
+ // and after a wait, whose whole point is to look again.
1201
+ const looks = settings.screenshotAfterActions &&
1202
+ available.has('screenshot') &&
1203
+ (changed || planned.some((step) => step.item.type === 'wait')) &&
1204
+ !signal?.aborted;
1205
+ let shot;
1206
+ let shotError;
1207
+ if (looks) {
1208
+ try {
1209
+ await sleep(settings.settleMs, signal);
1210
+ shot = await capture();
1211
+ }
1212
+ catch (error) {
1213
+ shotError = signal?.aborted ? 'cancelled' : errorText(error);
1214
+ }
1215
+ }
1216
+ const summary = (record, index) => {
1217
+ const n = single ? '' : `${index + 1}. `;
1218
+ switch (record.status) {
1219
+ case 'done':
1220
+ return `${n}${record.label}: ${record.note}`;
1221
+ case 'failed':
1222
+ return `${n}${record.label}: failed — ${record.error}`;
1223
+ case 'not run':
1224
+ return `${n}${record.label}: not run`;
1225
+ }
1226
+ };
1227
+ const header = single
1228
+ ? []
1229
+ : failure
1230
+ ? [
1231
+ `Batch stopped at action ${failure.index + 1} of ${planned.length}; ${planned.length - failure.index - 1} not run.`,
1232
+ ]
1233
+ : [`Batch: all ${planned.length} actions done.`];
1234
+ const stepLines = records.map(summary);
1235
+ const shotLines = shot
1236
+ ? [describeFrame(shot.frame)]
1237
+ : shotError
1238
+ ? [`The screenshot after acting failed: ${shotError}. Take one before acting again.`]
1239
+ : [];
1240
+ const unknownLines = failure?.unknown ? [failure.unknown.message] : [];
1241
+ const text = [...header, ...stepLines, ...unknownLines, ...shotLines].join('\n');
1242
+ const output = [
1243
+ ...header,
1244
+ ...stepLines,
1245
+ ...(shot
1246
+ ? [`Screenshot ${shot.frame.id} (${shot.frame.imageWidth}x${shot.frame.imageHeight}).`]
1247
+ : []),
1248
+ ].join('\n');
1249
+ const data = {
1250
+ steps: records.map((record) => record.status === 'failed'
1251
+ ? { label: record.label, status: record.status, error: record.error }
1252
+ : { label: record.label, status: record.status }),
1253
+ ...(shot ? { screenshot: frameData(shot.frame) } : {}),
1254
+ ...(failure?.unknown
1255
+ ? {
1256
+ code: failure.unknown.code,
1257
+ action: failure.unknown.action,
1258
+ outcome: failure.unknown.outcome,
1259
+ retrySafety: failure.unknown.retrySafety,
1260
+ timedOut: failure.unknown.timedOut,
1261
+ exitCode: failure.unknown.exitCode,
1262
+ }
1263
+ : {}),
1264
+ };
1265
+ const content = [{ type: 'text', text }];
1266
+ if (shot)
1267
+ content.push(imageBlock(shot.image));
1268
+ const result = {
1269
+ success: failure === undefined,
1270
+ output,
1271
+ content,
1272
+ data,
1273
+ ...(shot ? { workingState: pin(shot.frame) } : {}),
1274
+ };
1275
+ if (failure)
1276
+ result.error = failure.unknown
1277
+ ? failure.unknown.message
1278
+ : single
1279
+ ? `computer_use failed: ${failure.error}`
1280
+ : `Batch stopped at action ${failure.index + 1} of ${planned.length} (${planned[failure.index]?.label}): ${failure.error}`;
1281
+ return result;
1282
+ };
1283
+ // Whether a call sends the screen to the provider: an observation, or an
1284
+ // action that returns a screenshot afterwards. A diagnostic tool that can
1285
+ // reach nothing captures nothing.
1286
+ const observes = available.has('screenshot') || available.has('list_windows') || available.has('ui_snapshot');
1287
+ const capturesScreen = (input) => observes && capturesScreenInput(input, settings.screenshotAfterActions);
1288
+ const tool = defineTool({
277
1289
  name: COMPUTER_USE_TOOL_NAME,
278
- description: buildDescription(host),
1290
+ description: buildDescription(host, caps, settings),
279
1291
  inputSchema: actionSchema,
280
- modelInputSchema: hostModelSchema(host.capabilities),
281
- validationErrorHint: 'Action requirements: mouse_move needs "to"; mouse_click needs "at" and "button"; mouse_drag needs "from", "to", and "button"; scroll needs "at", "direction", and positive "amount"; type_text needs "text"; key needs "keys".',
1292
+ modelInputSchema: hostModelSchema(host, caps, settings),
1293
+ validationErrorHint: `Action requirements: ${ALL_ACTIONS.map((action) => ACTION_REQUIREMENTS[action]).join('; ')}. A batch is {"type":"batch","actions":[...]} with at most ${settings.maxBatchActions} actions and no screenshot, zoom, ui_snapshot or batch inside.`,
282
1294
  category: 'custom',
283
1295
  permissions: [],
284
- readOnly: false,
285
- destructive: (input) => DESTRUCTIVE_ACTION_TYPES.has(input.type),
1296
+ readOnly: (input) => isReadOnlyInput(input),
1297
+ destructive: (input) => isDestructiveInput(input),
1298
+ capturesScreen,
286
1299
  concurrencySafe: false,
287
1300
  presentCall: (input) => ({
288
1301
  kind: 'generic',
289
- label: actionLabel(input),
1302
+ label: label(input),
290
1303
  presentation: 'activity',
291
1304
  }),
292
- presentResult: (_input, result) => result.success && result.output.trim().toLowerCase() === 'ok'
1305
+ // Decided from the result alone: a host may present a finished call
1306
+ // without its input (the CLI passes `{}`). An action that only reports
1307
+ // "done" — and the screenshot that followed it, which is for the model —
1308
+ // adds nothing to the call row; observations, batches and failures do.
1309
+ presentResult: (_input, result) => result.success && ACKNOWLEDGEMENT.test(result.output)
293
1310
  ? { kind: 'generic', label: result.output, visibility: 'hidden' }
294
1311
  : undefined,
295
- async execute(input, _context) {
296
- const required = requiredCapability(input.type);
297
- if (required !== null && host.capabilities[required] !== true) {
298
- return {
299
- success: false,
300
- output: '',
301
- error: `computer_use: action "${input.type}" requires capability "${required}" which is not available on this host (displayServer=${host.capabilities.displayServer}).${host.capabilities.unavailableReason ? ` ${host.capabilities.unavailableReason} Do not retry; tell the user.` : ''}`,
302
- };
303
- }
304
- if (!availableActions(host.capabilities).includes(input.type)) {
305
- return {
306
- success: false,
307
- output: '',
308
- error: `computer_use: action "${input.type}" is not supported on this host.`,
309
- };
310
- }
311
- const buttons = input.type === 'mouse_click'
312
- ? host.capabilities.mouseClickButtons
313
- : input.type === 'mouse_drag'
314
- ? host.capabilities.mouseDragButtons
315
- : undefined;
316
- if (buttons && 'button' in input && !buttons.includes(input.button)) {
317
- return {
318
- success: false,
319
- output: '',
320
- error: `computer_use: action "${input.type}" does not support button "${input.button}" on this host.`,
321
- };
322
- }
1312
+ async execute(input, context) {
1313
+ const denied = refusal(input.type);
1314
+ if (denied)
1315
+ return { success: false, output: '', error: denied };
323
1316
  try {
324
- const result = await host.execute(input);
325
- return resultToToolResult(result);
1317
+ switch (input.type) {
1318
+ case 'screenshot':
1319
+ return screenshotResult(await capture());
1320
+ case 'zoom':
1321
+ return await zoom(input);
1322
+ case 'ui_snapshot':
1323
+ return await uiSnapshot(input);
1324
+ case 'batch':
1325
+ return await runSteps(input.actions, input.screenshot_id, context, false);
1326
+ default: {
1327
+ const { screenshot_id: frameId, ...item } = input;
1328
+ return await runSteps([item], frameId, context, true);
1329
+ }
1330
+ }
326
1331
  }
327
1332
  catch (error) {
328
- if (isOutcomeUnknown(error, input.type)) {
329
- return unknownOutcomeToToolResult(error);
330
- }
1333
+ if (error instanceof StepFailure)
1334
+ return { success: false, output: '', error: `computer_use failed: ${error.message}` };
331
1335
  throw error;
332
1336
  }
333
1337
  },
334
1338
  });
1339
+ return Object.assign(tool, { describeUiRef });
335
1340
  }
336
1341
  //# sourceMappingURL=computer-use.js.map