@namzu/sdk 45.1.0 → 46.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (118) hide show
  1. package/CHANGELOG.md +111 -0
  2. package/dist/authorization/gate.d.ts +5 -2
  3. package/dist/authorization/gate.d.ts.map +1 -1
  4. package/dist/authorization/gate.js +25 -4
  5. package/dist/authorization/gate.js.map +1 -1
  6. package/dist/authorization/rules.d.ts.map +1 -1
  7. package/dist/authorization/rules.js +19 -0
  8. package/dist/authorization/rules.js.map +1 -1
  9. package/dist/authorization/shell-lexer.d.ts +24 -0
  10. package/dist/authorization/shell-lexer.d.ts.map +1 -1
  11. package/dist/authorization/shell-lexer.js +69 -51
  12. package/dist/authorization/shell-lexer.js.map +1 -1
  13. package/dist/pricing/catalogue.generated.d.ts.map +1 -1
  14. package/dist/pricing/catalogue.generated.js +28 -4
  15. package/dist/pricing/catalogue.generated.js.map +1 -1
  16. package/dist/public-runtime.d.ts +3 -3
  17. package/dist/public-runtime.d.ts.map +1 -1
  18. package/dist/public-runtime.js +4 -2
  19. package/dist/public-runtime.js.map +1 -1
  20. package/dist/public-tools.d.ts +3 -2
  21. package/dist/public-tools.d.ts.map +1 -1
  22. package/dist/public-tools.js +5 -2
  23. package/dist/public-tools.js.map +1 -1
  24. package/dist/public-types.d.ts +3 -3
  25. package/dist/public-types.d.ts.map +1 -1
  26. package/dist/registry/tool/callable.d.ts +22 -0
  27. package/dist/registry/tool/callable.d.ts.map +1 -0
  28. package/dist/registry/tool/callable.js +29 -0
  29. package/dist/registry/tool/callable.js.map +1 -0
  30. package/dist/registry/tool/execute.d.ts.map +1 -1
  31. package/dist/registry/tool/execute.js +8 -1
  32. package/dist/registry/tool/execute.js.map +1 -1
  33. package/dist/runtime/query/executor/tool-call-admission.d.ts +37 -11
  34. package/dist/runtime/query/executor/tool-call-admission.d.ts.map +1 -1
  35. package/dist/runtime/query/executor/tool-call-admission.js +38 -12
  36. package/dist/runtime/query/executor/tool-call-admission.js.map +1 -1
  37. package/dist/runtime/query/executor.d.ts +4 -2
  38. package/dist/runtime/query/executor.d.ts.map +1 -1
  39. package/dist/runtime/query/executor.js +18 -5
  40. package/dist/runtime/query/executor.js.map +1 -1
  41. package/dist/runtime/query/review-policy.d.ts +43 -0
  42. package/dist/runtime/query/review-policy.d.ts.map +1 -1
  43. package/dist/runtime/query/review-policy.js +59 -16
  44. package/dist/runtime/query/review-policy.js.map +1 -1
  45. package/dist/skills/registry.d.ts +6 -0
  46. package/dist/skills/registry.d.ts.map +1 -1
  47. package/dist/skills/registry.js +1 -0
  48. package/dist/skills/registry.js.map +1 -1
  49. package/dist/tools/builtins/computer-use-coordinates.d.ts +65 -0
  50. package/dist/tools/builtins/computer-use-coordinates.d.ts.map +1 -0
  51. package/dist/tools/builtins/computer-use-coordinates.js +123 -0
  52. package/dist/tools/builtins/computer-use-coordinates.js.map +1 -0
  53. package/dist/tools/builtins/computer-use-image.d.ts +77 -0
  54. package/dist/tools/builtins/computer-use-image.d.ts.map +1 -0
  55. package/dist/tools/builtins/computer-use-image.js +223 -0
  56. package/dist/tools/builtins/computer-use-image.js.map +1 -0
  57. package/dist/tools/builtins/computer-use.d.ts +519 -14
  58. package/dist/tools/builtins/computer-use.d.ts.map +1 -1
  59. package/dist/tools/builtins/computer-use.js +1188 -183
  60. package/dist/tools/builtins/computer-use.js.map +1 -1
  61. package/dist/tools/builtins/index.d.ts +2 -1
  62. package/dist/tools/builtins/index.d.ts.map +1 -1
  63. package/dist/tools/builtins/index.js +1 -1
  64. package/dist/tools/builtins/index.js.map +1 -1
  65. package/dist/tools/builtins/skill.d.ts +74 -0
  66. package/dist/tools/builtins/skill.d.ts.map +1 -1
  67. package/dist/tools/builtins/skill.js +214 -174
  68. package/dist/tools/builtins/skill.js.map +1 -1
  69. package/dist/tools/defineTool.d.ts +2 -0
  70. package/dist/tools/defineTool.d.ts.map +1 -1
  71. package/dist/tools/defineTool.js +7 -0
  72. package/dist/tools/defineTool.js.map +1 -1
  73. package/dist/tools/schedules/present.d.ts.map +1 -1
  74. package/dist/tools/schedules/present.js +2 -0
  75. package/dist/tools/schedules/present.js.map +1 -1
  76. package/dist/tools/schedules/schedule-tool.d.ts +6 -5
  77. package/dist/tools/schedules/schedule-tool.d.ts.map +1 -1
  78. package/dist/tools/schedules/schedule-tool.js +121 -30
  79. package/dist/tools/schedules/schedule-tool.js.map +1 -1
  80. package/dist/tools/schedules/types.d.ts +69 -1
  81. package/dist/tools/schedules/types.d.ts.map +1 -1
  82. package/dist/types/authorization/index.d.ts +21 -0
  83. package/dist/types/authorization/index.d.ts.map +1 -1
  84. package/dist/types/authorization/index.js +5 -0
  85. package/dist/types/authorization/index.js.map +1 -1
  86. package/dist/types/computer-use/index.d.ts +176 -0
  87. package/dist/types/computer-use/index.d.ts.map +1 -1
  88. package/dist/types/computer-use/index.js.map +1 -1
  89. package/dist/types/tool/index.d.ts +19 -0
  90. package/dist/types/tool/index.d.ts.map +1 -1
  91. package/dist/types/tool/index.js.map +1 -1
  92. package/package.json +3 -1
  93. package/src/authorization/gate.ts +25 -4
  94. package/src/authorization/rules.ts +18 -0
  95. package/src/authorization/shell-lexer.ts +73 -45
  96. package/src/pricing/catalogue.generated.ts +28 -4
  97. package/src/pricing/rates.source.json +27 -6
  98. package/src/public-runtime.ts +16 -2
  99. package/src/public-tools.ts +19 -1
  100. package/src/public-types.ts +5 -0
  101. package/src/registry/tool/callable.ts +37 -0
  102. package/src/registry/tool/execute.ts +8 -1
  103. package/src/runtime/query/executor/tool-call-admission.ts +60 -16
  104. package/src/runtime/query/executor.ts +19 -4
  105. package/src/runtime/query/review-policy.ts +103 -15
  106. package/src/skills/registry.ts +8 -0
  107. package/src/tools/builtins/computer-use-coordinates.ts +144 -0
  108. package/src/tools/builtins/computer-use-image.ts +278 -0
  109. package/src/tools/builtins/computer-use.ts +1485 -191
  110. package/src/tools/builtins/index.ts +7 -1
  111. package/src/tools/builtins/skill.ts +304 -174
  112. package/src/tools/defineTool.ts +10 -0
  113. package/src/tools/schedules/present.ts +2 -0
  114. package/src/tools/schedules/schedule-tool.ts +139 -33
  115. package/src/tools/schedules/types.ts +69 -1
  116. package/src/types/authorization/index.ts +21 -0
  117. package/src/types/computer-use/index.ts +202 -0
  118. package/src/types/tool/index.ts +19 -0
@@ -1,18 +1,116 @@
1
1
  import { z } from 'zod'
2
+ import { resolveProviderCapabilities } from '../../provider/capabilities.js'
2
3
  import type {
3
4
  ComputerUseAction,
4
5
  ComputerUseCapabilities,
5
6
  ComputerUseHost,
6
7
  ComputerUseOutcomeUnknown,
7
- ComputerUseResult,
8
+ Point,
9
+ Rect,
10
+ ScreenshotResult,
11
+ UiActResult,
12
+ UiElement,
13
+ UiElementAction,
14
+ UiSnapshot,
15
+ WindowInfo,
8
16
  } from '../../types/computer-use/index.js'
9
- import type { ToolDefinition, ToolResult } from '../../types/tool/index.js'
17
+ import type { ToolResultBlock } from '../../types/message/index.js'
18
+ import type { LLMProvider } from '../../types/provider/index.js'
19
+ import type { ToolContext, ToolDefinition, ToolResult } from '../../types/tool/index.js'
20
+ import { sleep } from '../../utils/backoff.js'
10
21
  import { defineTool } from '../defineTool.js'
22
+ import { neutralizeEnvelopeDelimiter, wrapUntrusted } from '../untrusted-envelope.js'
23
+ import {
24
+ type ScreenshotFrame,
25
+ ScreenshotFrames,
26
+ assumedDisplay,
27
+ desktopRectOnImage,
28
+ pointOnImage,
29
+ toDisplayPoint,
30
+ toDisplayRect,
31
+ toImagePoint,
32
+ } from './computer-use-coordinates.js'
33
+ import {
34
+ type FittedImage,
35
+ STANDARD_SCREENSHOT_LIMITS,
36
+ type ScreenshotLimits,
37
+ cropAndFitPng,
38
+ fitPng,
39
+ pngSize,
40
+ } from './computer-use-image.js'
41
+
42
+ export {
43
+ HIGH_RES_SCREENSHOT_LIMITS,
44
+ STANDARD_SCREENSHOT_LIMITS,
45
+ screenshotTargetSize,
46
+ } from './computer-use-image.js'
47
+ export type { ImageSize, ScreenshotLimits } from './computer-use-image.js'
11
48
 
12
49
  export const COMPUTER_USE_TOOL_NAME = 'computer_use' as const
13
50
 
51
+ const DEFAULT_SETTLE_MS = 500
52
+ const DEFAULT_MAX_BATCH_ACTIONS = 20
53
+ const DEFAULT_MAX_WAIT_MS = 10_000
54
+ const MAX_LISTED_WINDOWS = 50
55
+ /** The most of a ui_snapshot the model is shown, in characters (about 4 000 tokens). */
56
+ const MAX_UI_SNAPSHOT_CHARS = 14_000
57
+
58
+ /**
59
+ * How `createComputerUseTool` sizes screenshots, paces actions and bounds a
60
+ * batch. Every field is optional.
61
+ */
62
+ export interface ComputerUseToolOptions {
63
+ /**
64
+ * What every screenshot and zoom image is fitted to before the model sees
65
+ * it. Default {@link STANDARD_SCREENSHOT_LIMITS}, which every current vision
66
+ * model takes without a further resize; {@link HIGH_RES_SCREENSHOT_LIMITS}
67
+ * only when every model the session can reach is on that tier.
68
+ */
69
+ readonly screenshotLimits?: ScreenshotLimits
70
+ /**
71
+ * Milliseconds to wait after an action before the screenshot it returns,
72
+ * so a menu has opened or a page has started to paint. Default 500.
73
+ */
74
+ readonly settleMs?: number
75
+ /**
76
+ * Return a fresh screenshot after every call that changed something.
77
+ * Default true. Off, a model has to ask for one — an extra round trip per
78
+ * action, which is what this exists to remove.
79
+ */
80
+ readonly screenshotAfterActions?: boolean
81
+ /** Most actions one `batch` may carry. Default 20. */
82
+ readonly maxBatchActions?: number
83
+ /** Longest single `wait`, in milliseconds. Default 10 000. */
84
+ readonly maxWaitMs?: number
85
+ /**
86
+ * Why computer use cannot work in this session although the host could —
87
+ * typically that the provider cannot put an image in a tool result, so the
88
+ * model would never see a screenshot (see
89
+ * {@link computerUseUnavailableReason}). Set, the tool mounts as a
90
+ * diagnostic: its description says why and every call is refused without
91
+ * touching the host.
92
+ */
93
+ readonly unavailableReason?: string
94
+ }
95
+
96
+ /**
97
+ * Why `provider` cannot drive `computer_use`, or undefined when it can.
98
+ *
99
+ * The tool's only way to show the model the screen is an image in a tool
100
+ * result. A driver that declares `supportsToolResultImages: false` replaces
101
+ * that image with a line of text, so the model acts on a screen it has never
102
+ * seen while every call reports success. Pass the answer to
103
+ * {@link ComputerUseToolOptions.unavailableReason}.
104
+ */
105
+ export function computerUseUnavailableReason(
106
+ provider: Pick<LLMProvider, 'id' | 'capabilities'>,
107
+ ): string | undefined {
108
+ if (resolveProviderCapabilities(provider).supportsToolResultImages) return undefined
109
+ return `The ${provider.id} provider cannot return images in tool results, so the model would never see a screenshot. Use a provider that can (for example Anthropic, Codex or Google) for computer use.`
110
+ }
111
+
14
112
  // ---------------------------------------------------------------------------
15
- // Input schema — discriminated union matching ComputerUseAction
113
+ // Input schema
16
114
  // ---------------------------------------------------------------------------
17
115
 
18
116
  const pointSchema = z.object({
@@ -20,104 +118,218 @@ const pointSchema = z.object({
20
118
  y: z.number().int(),
21
119
  })
22
120
 
23
- const mouseButtonSchema = z.enum(['left', 'right', 'middle'])
121
+ const regionSchema = z.object({
122
+ x: z.number().int(),
123
+ y: z.number().int(),
124
+ width: z.number().int().positive(),
125
+ height: z.number().int().positive(),
126
+ })
127
+
128
+ // Left when omitted: a model asked to "click the Start button" often leaves
129
+ // the button out, and refusing that cost a whole round trip for nothing.
130
+ const mouseButtonSchema = z.enum(['left', 'right', 'middle']).default('left')
131
+ const screenshotIdSchema = z.string().min(1).optional()
132
+
133
+ const cursorPositionSchema = z.object({ type: z.literal('cursor_position') })
134
+ const mouseMoveSchema = z.object({ type: z.literal('mouse_move'), to: pointSchema })
135
+ const mouseClickSchema = z.object({
136
+ type: z.literal('mouse_click'),
137
+ at: pointSchema,
138
+ button: mouseButtonSchema,
139
+ })
140
+ const mouseDragSchema = z.object({
141
+ type: z.literal('mouse_drag'),
142
+ from: pointSchema,
143
+ to: pointSchema,
144
+ button: mouseButtonSchema,
145
+ })
146
+ const scrollSchema = z.object({
147
+ type: z.literal('scroll'),
148
+ at: pointSchema,
149
+ direction: z.enum(['up', 'down', 'left', 'right']),
150
+ amount: z.number().int().positive(),
151
+ })
152
+ const typeTextSchema = z.object({ type: z.literal('type_text'), text: z.string() })
153
+ const keySchema = z.object({ type: z.literal('key'), keys: z.string() })
154
+ const waitSchema = z.object({ type: z.literal('wait'), ms: z.number().int().nonnegative() })
155
+ const listWindowsSchema = z.object({ type: z.literal('list_windows') })
156
+ const focusWindowSchema = z.object({
157
+ type: z.literal('focus_window'),
158
+ window_id: z.string().min(1),
159
+ })
160
+ const UI_ACTIONS = [
161
+ 'invoke',
162
+ 'set_value',
163
+ 'toggle',
164
+ 'select',
165
+ 'expand',
166
+ 'collapse',
167
+ 'focus',
168
+ 'scroll_into_view',
169
+ ] as const satisfies readonly UiElementAction[]
170
+ const uiSnapshotSchema = z.object({
171
+ type: z.literal('ui_snapshot'),
172
+ window_id: z.string().min(1).optional(),
173
+ })
174
+ const uiActSchema = z.object({
175
+ type: z.literal('ui_act'),
176
+ ref: z.string().min(1),
177
+ action: z.enum(UI_ACTIONS),
178
+ value: z.string().optional(),
179
+ })
180
+
181
+ /** What a batch may carry: everything but the image-returning actions and another batch. */
182
+ const batchItemSchema = z.discriminatedUnion('type', [
183
+ cursorPositionSchema,
184
+ mouseMoveSchema,
185
+ mouseClickSchema,
186
+ mouseDragSchema,
187
+ scrollSchema,
188
+ typeTextSchema,
189
+ keySchema,
190
+ waitSchema,
191
+ listWindowsSchema,
192
+ focusWindowSchema,
193
+ uiActSchema,
194
+ ])
195
+
196
+ const withFrame = { screenshot_id: screenshotIdSchema }
24
197
 
25
198
  const actionSchema = z.discriminatedUnion('type', [
26
199
  z.object({ type: z.literal('screenshot') }),
27
- z.object({ type: z.literal('cursor_position') }),
28
- z.object({ type: z.literal('mouse_move'), to: pointSchema }),
29
- z.object({ type: z.literal('mouse_click'), at: pointSchema, button: mouseButtonSchema }),
200
+ z.object({ type: z.literal('zoom'), region: regionSchema, ...withFrame }),
201
+ cursorPositionSchema.extend(withFrame),
202
+ mouseMoveSchema.extend(withFrame),
203
+ mouseClickSchema.extend(withFrame),
204
+ mouseDragSchema.extend(withFrame),
205
+ scrollSchema.extend(withFrame),
206
+ typeTextSchema.extend(withFrame),
207
+ keySchema.extend(withFrame),
208
+ waitSchema.extend(withFrame),
209
+ listWindowsSchema.extend(withFrame),
210
+ focusWindowSchema.extend(withFrame),
211
+ uiSnapshotSchema,
212
+ uiActSchema.extend(withFrame),
30
213
  z.object({
31
- type: z.literal('mouse_drag'),
32
- from: pointSchema,
33
- to: pointSchema,
34
- button: mouseButtonSchema,
214
+ type: z.literal('batch'),
215
+ actions: z.array(batchItemSchema).min(1),
216
+ ...withFrame,
35
217
  }),
36
- z.object({
37
- type: z.literal('scroll'),
38
- at: pointSchema,
39
- direction: z.enum(['up', 'down', 'left', 'right']),
40
- amount: z.number().int().positive(),
41
- }),
42
- z.object({ type: z.literal('type_text'), text: z.string() }),
43
- z.object({ type: z.literal('key'), keys: z.string() }),
44
218
  ])
45
219
 
46
220
  /**
47
- * The provider-facing shape is deliberately flat.
221
+ * The tool's input, inferred from its schema: one action, or
222
+ * `{ type: 'batch', actions: [...] }`.
48
223
  *
49
- * The runtime schema above is the authoritative contract: it knows which
50
- * fields each action requires. Rendering that discriminated union produces a
51
- * root `anyOf`, however, and some custom-tool wires reject root combinators
52
- * even when every branch is an object. A model can still see every field and
53
- * every action here; incomplete combinations are rejected by `actionSchema`
54
- * before the host is called, with the recovery hint below.
55
- */
56
- const pointModelInputSchema = {
57
- type: 'object',
58
- properties: {
59
- x: { type: 'integer' },
60
- y: { type: 'integer' },
61
- },
62
- required: ['x', 'y'],
63
- additionalProperties: false,
64
- } as const
65
-
66
- const modelInputSchema = {
67
- type: 'object',
68
- properties: {
69
- type: {
70
- type: 'string',
71
- enum: [
72
- 'screenshot',
73
- 'cursor_position',
74
- 'mouse_move',
75
- 'mouse_click',
76
- 'mouse_drag',
77
- 'scroll',
78
- 'type_text',
79
- 'key',
80
- ],
81
- description:
82
- 'Desktop action. screenshot and cursor_position need no other fields; mouse_move needs to; mouse_click needs at and button; mouse_drag needs from, to, and button; scroll needs at, direction, and amount; type_text needs text; key needs keys.',
83
- },
84
- to: pointModelInputSchema,
85
- at: pointModelInputSchema,
86
- from: pointModelInputSchema,
87
- button: { type: 'string', enum: ['left', 'right', 'middle'] },
88
- direction: { type: 'string', enum: ['up', 'down', 'left', 'right'] },
89
- amount: { type: 'integer', description: 'Positive integer scroll distance.' },
90
- text: { type: 'string', description: 'Literal text to type.' },
91
- keys: {
92
- type: 'string',
93
- description: 'Key or key chord to press, for example ENTER or CTRL+R.',
94
- },
95
- },
96
- required: ['type'],
97
- additionalProperties: false,
98
- }
99
-
100
- /**
101
- * The tool's input, inferred from its schema.
102
- *
103
- * Exported because `createComputerUseTool` returns a `ToolDefinition<ActionInput>`
224
+ * Exported because `createComputerUseTool` returns a `ToolDefinition<ActionInput>` (a `ComputerUseTool`)
104
225
  * and a consumer typing that variable, or writing a wrapper around it, had no
105
226
  * name for the parameter — the type was module-private while the function
106
227
  * carrying it was public.
107
228
  */
108
229
  export type ActionInput = z.infer<typeof actionSchema>
230
+ type BatchItem = z.infer<typeof batchItemSchema>
231
+ type ToolActionType = ActionInput['type']
232
+
233
+ const HOST_ACTIONS: readonly ComputerUseAction['type'][] = [
234
+ 'screenshot',
235
+ 'cursor_position',
236
+ 'mouse_move',
237
+ 'mouse_click',
238
+ 'mouse_drag',
239
+ 'scroll',
240
+ 'type_text',
241
+ 'key',
242
+ ]
109
243
 
110
- const DESTRUCTIVE_ACTION_TYPES = new Set<ComputerUseAction['type']>([
244
+ const ALL_ACTIONS: readonly ToolActionType[] = [
245
+ 'screenshot',
246
+ 'zoom',
247
+ 'cursor_position',
248
+ 'mouse_move',
249
+ 'mouse_click',
250
+ 'mouse_drag',
251
+ 'scroll',
252
+ 'type_text',
253
+ 'key',
254
+ 'wait',
255
+ 'list_windows',
256
+ 'focus_window',
257
+ 'ui_snapshot',
258
+ 'ui_act',
259
+ 'batch',
260
+ ]
261
+
262
+ /** Observations: they change nothing on the desktop. */
263
+ const READ_ONLY_ACTIONS = new Set<string>([
264
+ 'screenshot',
265
+ 'zoom',
266
+ 'cursor_position',
267
+ 'wait',
268
+ 'list_windows',
269
+ 'ui_snapshot',
270
+ ])
271
+
272
+ const DESTRUCTIVE_ACTIONS = new Set<string>([
111
273
  'mouse_click',
112
274
  'mouse_drag',
113
275
  'type_text',
114
276
  'key',
115
277
  'scroll',
278
+ 'ui_act',
116
279
  ])
117
280
 
118
- function requiredCapability(type: ComputerUseAction['type']): keyof ComputerUseCapabilities | null {
281
+ /** Observations that show the model what is on the screen, on their own. */
282
+ const SCREEN_OBSERVATIONS = new Set<string>(['screenshot', 'zoom', 'list_windows', 'ui_snapshot'])
283
+
284
+ /**
285
+ * Terminal applications, by the process name a host reports in
286
+ * `WindowInfo.app` (on Windows without `.exe`). Keys are not typed into these.
287
+ */
288
+ const TERMINAL_APPS =
289
+ /^(WindowsTerminal|OpenConsole|conhost|cmd|powershell|pwsh|wsl|mintty|wezterm(-gui)?|alacritty|Hyper|Tabby|kitty|ConEmu(64)?|putty|Terminal|iTerm2?|gnome-terminal(-server)?|konsole|xterm|xfce4-terminal|tilix|terminator|foot|ghostty|Warp)$/i
290
+
291
+ /** Actions a batch cannot carry: the ones that return an image or a tree, and batch itself. */
292
+ const OUTSIDE_BATCH = new Set<string>(['screenshot', 'zoom', 'ui_snapshot', 'batch'])
293
+
294
+ /** A single action's whole result when it only acted: `<label>: done`, then its screenshot. */
295
+ const ACKNOWLEDGEMENT = /^[^\n]*: done(\nScreenshot s\d+ \(\d+x\d+\)\.)?$/
296
+
297
+ function isRecord(value: unknown): value is Record<string, unknown> {
298
+ return typeof value === 'object' && value !== null && !Array.isArray(value)
299
+ }
300
+
301
+ function batchActions(input: unknown): readonly unknown[] | null {
302
+ if (!isRecord(input) || input.type !== 'batch') return null
303
+ return Array.isArray(input.actions) ? input.actions : []
304
+ }
305
+
306
+ /** Read-only when every action is: a batch with one click in it is not. Unknown shapes are not. */
307
+ function isReadOnlyInput(input: unknown): boolean {
308
+ const actions = batchActions(input)
309
+ if (actions)
310
+ return (
311
+ actions.length > 0 &&
312
+ actions.every((item) => isRecord(item) && READ_ONLY_ACTIONS.has(String(item.type)))
313
+ )
314
+ return isRecord(input) && READ_ONLY_ACTIONS.has(String(input.type))
315
+ }
316
+
317
+ function isDestructiveInput(input: unknown): boolean {
318
+ const actions = batchActions(input)
319
+ if (actions)
320
+ return actions.some((item) => !isRecord(item) || DESTRUCTIVE_ACTIONS.has(String(item.type)))
321
+ return isRecord(input) && DESTRUCTIVE_ACTIONS.has(String(input.type))
322
+ }
323
+
324
+ // ---------------------------------------------------------------------------
325
+ // Capabilities
326
+ // ---------------------------------------------------------------------------
327
+
328
+ function requiredCapability(type: string): keyof ComputerUseCapabilities | null {
119
329
  switch (type) {
120
330
  case 'screenshot':
331
+ case 'zoom':
332
+ case 'wait':
121
333
  return 'screenshot'
122
334
  case 'cursor_position':
123
335
  return 'cursorPosition'
@@ -129,27 +341,89 @@ function requiredCapability(type: ComputerUseAction['type']): keyof ComputerUseC
129
341
  case 'type_text':
130
342
  case 'key':
131
343
  return 'keyboard'
344
+ case 'list_windows':
345
+ case 'focus_window':
346
+ return 'windows'
347
+ case 'ui_snapshot':
348
+ case 'ui_act':
349
+ return 'uiTree'
132
350
  default:
133
351
  return null
134
352
  }
135
353
  }
136
354
 
137
- function buildDescription(host: ComputerUseHost): string {
138
- const caps = host.capabilities
139
- const available = availableActions(caps)
355
+ function hostActionAvailable(
356
+ caps: ComputerUseCapabilities,
357
+ action: ComputerUseAction['type'],
358
+ ): boolean {
359
+ const required = requiredCapability(action)
360
+ return (
361
+ (required === null || caps[required] === true) &&
362
+ (caps.supportedActions === undefined || caps.supportedActions.includes(action)) &&
363
+ (action !== 'mouse_click' || caps.mouseClickButtons?.length !== 0) &&
364
+ (action !== 'mouse_drag' || caps.mouseDragButtons?.length !== 0)
365
+ )
366
+ }
367
+
368
+ function availableActions(host: ComputerUseHost, caps: ComputerUseCapabilities): ToolActionType[] {
369
+ const screenshot = hostActionAvailable(caps, 'screenshot')
370
+ const windows =
371
+ caps.windows === true &&
372
+ typeof host.listWindows === 'function' &&
373
+ typeof host.focusWindow === 'function'
374
+ const uiTree =
375
+ caps.uiTree === true &&
376
+ typeof host.uiSnapshot === 'function' &&
377
+ typeof host.uiAct === 'function'
378
+ const available = ALL_ACTIONS.filter((action) => {
379
+ switch (action) {
380
+ case 'zoom':
381
+ case 'wait':
382
+ return screenshot
383
+ case 'list_windows':
384
+ case 'focus_window':
385
+ return windows
386
+ case 'ui_snapshot':
387
+ case 'ui_act':
388
+ return uiTree
389
+ case 'batch':
390
+ return false
391
+ default:
392
+ return hostActionAvailable(caps, action)
393
+ }
394
+ })
395
+ if (available.some((action) => !OUTSIDE_BATCH.has(action))) available.push('batch')
396
+ return available
397
+ }
398
+
399
+ function unavailableHostActions(caps: ComputerUseCapabilities): string[] {
140
400
  const unavailable: string[] = []
141
401
  if (!caps.screenshot) unavailable.push('screenshot')
142
402
  if (!caps.cursorPosition) unavailable.push('cursor_position')
143
403
  if (!caps.mouse) unavailable.push('mouse')
144
404
  if (!caps.keyboard) unavailable.push('keyboard')
145
405
  if (caps.supportedActions) {
146
- for (const action of actionSchema.options.map((option) => option.shape.type.value)) {
147
- if (!available.includes(action)) unavailable.push(action)
406
+ for (const action of HOST_ACTIONS) {
407
+ if (!hostActionAvailable(caps, action) && !unavailable.includes(action))
408
+ unavailable.push(action)
148
409
  }
149
410
  }
411
+ return unavailable
412
+ }
413
+
414
+ // ---------------------------------------------------------------------------
415
+ // Model-facing description and schema
416
+ // ---------------------------------------------------------------------------
150
417
 
418
+ function buildDescription(
419
+ host: ComputerUseHost,
420
+ caps: ComputerUseCapabilities,
421
+ settings: ResolvedSettings,
422
+ ): string {
423
+ const available = availableActions(host, caps)
424
+ const unavailable = unavailableHostActions(caps)
151
425
  const lines = [
152
- `Controls the user's desktop on a ${caps.displayServer} host. Use to take screenshots and drive mouse/keyboard input for GUI tasks.`,
426
+ `Controls the user's desktop on a ${caps.displayServer} host: screenshots, mouse and keyboard, for GUI tasks.`,
153
427
  `Available actions: ${available.join('; ') || 'none'}.`,
154
428
  ]
155
429
  if (unavailable.length > 0) {
@@ -159,9 +433,54 @@ function buildDescription(host: ComputerUseHost): string {
159
433
  : `Unavailable on this host: ${unavailable.join(', ')}.`,
160
434
  )
161
435
  }
436
+ if (!available.includes('screenshot')) return finish(lines, caps, available)
162
437
  lines.push(
163
- 'Coordinates are in logical pixels from the top-left of the primary display. Call getDisplayGeometry through screenshot output before clicking to confirm bounds.',
438
+ 'Coordinates: every x/y you send is a pixel of the most recent screenshot this tool returned — origin at its top-left, x to the right, y down — never a screen pixel. Each screenshot states its id and size (for example "s3: 1456x819"); stay inside it. The tool maps your coordinates onto the display. To aim at an earlier screenshot, pass its id as screenshot_id.',
439
+ 'Take a screenshot before your first click.',
164
440
  )
441
+ if (settings.screenshotAfterActions)
442
+ lines.push(
443
+ `After any action that changes something, the tool waits ${settings.settleMs} ms and returns a new screenshot — do not call screenshot again after acting.`,
444
+ )
445
+ lines.push(
446
+ 'zoom returns a closer, sharper view of region {x, y, width, height} of the screenshot; your coordinates still refer to the screenshot, not to the zoomed image.',
447
+ `wait pauses for ms milliseconds (at most ${settings.maxWaitMs} per call, a batch's waits together) and then shows the screen.`,
448
+ )
449
+ if (available.includes('batch'))
450
+ lines.push(
451
+ `batch: {"type":"batch","actions":[...]} runs up to ${settings.maxBatchActions} actions in order, stops at the first one that fails, and returns one screenshot at the end. Every call costs a model round trip of several seconds, so put the steps you can already predict into one batch (click a field, type, press ENTER) rather than one call each — but end the batch at any step that should bring up a new window, and look before typing into it. A batch cannot contain screenshot${available.includes('ui_snapshot') ? ', zoom or ui_snapshot' : ' or zoom'}.`,
452
+ )
453
+ if (
454
+ caps.displayServer === 'win32' &&
455
+ available.includes('key') &&
456
+ available.includes('type_text')
457
+ )
458
+ lines.push(
459
+ 'To start a Windows program, a shell command (Start-Process notepad, or cmd.exe /c start calc) is surest when you have a shell tool. The Start menu searches by display names in the system language, and ENTER on a name it does not find opens a web search in the browser.',
460
+ )
461
+ if (available.includes('list_windows'))
462
+ lines.push(
463
+ 'list_windows names the open windows with their ids; focus_window brings one to the front.',
464
+ )
465
+ if (available.includes('ui_snapshot'))
466
+ lines.push(
467
+ `ui_snapshot {window_id} reads a window's controls (its accessibility tree) as text, each control you can act on with a ref such as e12; without window_id it reads the window in front. ui_act {ref, action, value} acts on one control: invoke (press a button, open a menu item), set_value (replace a field's text with value), toggle, select, expand, collapse. Prefer ui_act to clicking pixels when the control is in the tree: it does not depend on coordinates or on which window is in front, and a batch of ui_act steps (for example pressing several buttons) runs in one call. Refs are valid only until the next ui_snapshot; take a new one after the window changes.`,
468
+ )
469
+ lines.push(
470
+ 'Typing and keys go to whichever window has focus: confirm on a screenshot that the right window is in front before type_text or key.',
471
+ )
472
+ if (available.includes('list_windows'))
473
+ lines.push(
474
+ 'type_text and key are refused while a terminal window is in front — usually the one running this agent, or the user’s own; bring the window you mean to the front first (focus_window, or click it).',
475
+ )
476
+ return finish(lines, caps, available)
477
+ }
478
+
479
+ function finish(
480
+ lines: string[],
481
+ caps: ComputerUseCapabilities,
482
+ available: readonly string[],
483
+ ): string {
165
484
  if (caps.mouseClickButtons && available.includes('mouse_click'))
166
485
  lines.push(`Click buttons: ${caps.mouseClickButtons.join(', ') || 'none'}.`)
167
486
  if (caps.mouseDragButtons && available.includes('mouse_drag'))
@@ -169,35 +488,154 @@ function buildDescription(host: ComputerUseHost): string {
169
488
  return lines.join(' ')
170
489
  }
171
490
 
172
- function availableActions(caps: ComputerUseCapabilities): ComputerUseAction['type'][] {
173
- return actionSchema.options
174
- .map((option) => option.shape.type.value)
175
- .filter((action) => {
176
- const required = requiredCapability(action)
177
- return (
178
- (required === null || caps[required] === true) &&
179
- (caps.supportedActions === undefined || caps.supportedActions.includes(action)) &&
180
- (action !== 'mouse_click' || caps.mouseClickButtons?.length !== 0) &&
181
- (action !== 'mouse_drag' || caps.mouseDragButtons?.length !== 0)
182
- )
183
- })
491
+ /**
492
+ * The provider-facing shape is deliberately flat.
493
+ *
494
+ * The runtime schema above is the authoritative contract: it knows which
495
+ * fields each action requires. Rendering that discriminated union produces a
496
+ * root `anyOf`, however, and some custom-tool wires reject root combinators
497
+ * even when every branch is an object. A model can still see every field and
498
+ * every action here; incomplete combinations are rejected by `actionSchema`
499
+ * before the host is called, with the recovery hint below.
500
+ */
501
+ function pointModelSchema(description?: string): Record<string, unknown> {
502
+ return {
503
+ type: 'object',
504
+ ...(description ? { description } : {}),
505
+ properties: {
506
+ x: { type: 'integer' },
507
+ y: { type: 'integer' },
508
+ },
509
+ required: ['x', 'y'],
510
+ additionalProperties: false,
511
+ }
184
512
  }
185
513
 
186
- function hostModelSchema(caps: ComputerUseCapabilities): Record<string, unknown> {
187
- const schema = structuredClone(modelInputSchema)
188
- const actions = availableActions(caps)
514
+ const ACTION_REQUIREMENTS: Readonly<Record<ToolActionType, string>> = {
515
+ screenshot: 'screenshot needs no other fields',
516
+ zoom: 'zoom needs region',
517
+ cursor_position: 'cursor_position needs no other fields',
518
+ mouse_move: 'mouse_move needs to',
519
+ mouse_click: 'mouse_click needs at (button defaults to left)',
520
+ mouse_drag: 'mouse_drag needs from and to (button defaults to left)',
521
+ scroll: 'scroll needs at, direction, and amount',
522
+ type_text: 'type_text needs text',
523
+ key: 'key needs keys',
524
+ wait: 'wait needs ms',
525
+ list_windows: 'list_windows needs no other fields',
526
+ focus_window: 'focus_window needs window_id',
527
+ ui_snapshot: 'ui_snapshot takes an optional window_id',
528
+ ui_act: 'ui_act needs ref and action, and value for set_value',
529
+ batch: 'batch needs actions',
530
+ }
531
+
532
+ function hostModelSchema(
533
+ host: ComputerUseHost,
534
+ caps: ComputerUseCapabilities,
535
+ settings: ResolvedSettings,
536
+ ): Record<string, unknown> {
189
537
  // An unavailable host remains a diagnostic tool. Avoid invalid empty enums
190
538
  // on provider wires; every execution is still refused before host access.
191
- if (actions.length > 0) schema.properties.type.enum = actions
539
+ const offered = availableActions(host, caps)
540
+ const actions = offered.length > 0 ? offered : ALL_ACTIONS.filter((a) => a !== 'batch')
541
+ const items = actions.filter((action) => !OUTSIDE_BATCH.has(action))
192
542
  const buttons = new Set<string>()
193
543
  if (actions.includes('mouse_click'))
194
544
  for (const button of caps.mouseClickButtons ?? ['left', 'right', 'middle']) buttons.add(button)
195
545
  if (actions.includes('mouse_drag'))
196
546
  for (const button of caps.mouseDragButtons ?? ['left', 'right', 'middle']) buttons.add(button)
197
- if (buttons.size > 0) schema.properties.button.enum = [...buttons]
198
- return schema
547
+ const buttonEnum = buttons.size > 0 ? [...buttons] : ['left', 'right', 'middle']
548
+
549
+ const fieldSchemas = (forItems: boolean): Record<string, unknown> => {
550
+ const present = forItems ? items : actions
551
+ const fields: Record<string, unknown> = {
552
+ to: pointModelSchema(),
553
+ at: pointModelSchema(),
554
+ from: pointModelSchema(),
555
+ button: { type: 'string', enum: buttonEnum, description: 'Mouse button; left when omitted.' },
556
+ direction: { type: 'string', enum: ['up', 'down', 'left', 'right'] },
557
+ amount: { type: 'integer', description: 'Positive integer scroll distance.' },
558
+ text: { type: 'string', description: 'Literal text to type.' },
559
+ keys: {
560
+ type: 'string',
561
+ description: 'Key or key chord to press, for example ENTER or CTRL+R.',
562
+ },
563
+ ms: {
564
+ type: 'integer',
565
+ description: `Milliseconds to wait, 0 to ${settings.maxWaitMs}.`,
566
+ },
567
+ }
568
+ if (present.includes('focus_window') || (!forItems && present.includes('ui_snapshot')))
569
+ fields.window_id = { type: 'string', description: 'A window id from list_windows.' }
570
+ if (present.includes('ui_act')) {
571
+ fields.ref = {
572
+ type: 'string',
573
+ description: 'A ref from the latest ui_snapshot, such as e12.',
574
+ }
575
+ fields.action = {
576
+ type: 'string',
577
+ enum: [...UI_ACTIONS],
578
+ description: 'What ui_act does to the control.',
579
+ }
580
+ fields.value = { type: 'string', description: 'The text set_value puts in the control.' }
581
+ }
582
+ if (!forItems && present.includes('zoom'))
583
+ fields.region = {
584
+ type: 'object',
585
+ description: 'Area of the screenshot to zoom into, in its pixels.',
586
+ properties: {
587
+ x: { type: 'integer' },
588
+ y: { type: 'integer' },
589
+ width: { type: 'integer' },
590
+ height: { type: 'integer' },
591
+ },
592
+ required: ['x', 'y', 'width', 'height'],
593
+ additionalProperties: false,
594
+ }
595
+ return fields
596
+ }
597
+
598
+ const properties: Record<string, unknown> = {
599
+ type: {
600
+ type: 'string',
601
+ enum: actions,
602
+ description: `Desktop action. ${actions.map((action) => ACTION_REQUIREMENTS[action]).join('; ')}.`,
603
+ },
604
+ ...fieldSchemas(false),
605
+ }
606
+ if (actions.includes('batch'))
607
+ properties.actions = {
608
+ type: 'array',
609
+ minItems: 1,
610
+ maxItems: settings.maxBatchActions,
611
+ description: `For type batch: up to ${settings.maxBatchActions} actions run in order (no screenshot, zoom, ui_snapshot or batch inside).`,
612
+ items: {
613
+ type: 'object',
614
+ properties: {
615
+ type: { type: 'string', enum: items },
616
+ ...fieldSchemas(true),
617
+ },
618
+ required: ['type'],
619
+ additionalProperties: false,
620
+ },
621
+ }
622
+ properties.screenshot_id = {
623
+ type: 'string',
624
+ description:
625
+ 'Optional id of the screenshot your coordinates were read from, such as s3. Defaults to the latest.',
626
+ }
627
+ return {
628
+ type: 'object',
629
+ properties,
630
+ required: ['type'],
631
+ additionalProperties: false,
632
+ }
199
633
  }
200
634
 
635
+ // ---------------------------------------------------------------------------
636
+ // Labels
637
+ // ---------------------------------------------------------------------------
638
+
201
639
  function pointLabel(point: { readonly x: number; readonly y: number }): string {
202
640
  return `(${point.x}, ${point.y})`
203
641
  }
@@ -208,11 +646,49 @@ function quotedText(value: string): string {
208
646
  return JSON.stringify(visible)
209
647
  }
210
648
 
211
- /** Human activity text; the raw action union remains the model-facing input. */
212
- function actionLabel(input: ActionInput): string {
649
+ function waitLabel(ms: number): string {
650
+ return ms >= 1000 ? `Wait ${Number((ms / 1000).toFixed(1))} s` : `Wait ${ms} ms`
651
+ }
652
+
653
+ /** What a ui_act does, as a verb phrase over the control's description. */
654
+ function uiActLabel(
655
+ input: { readonly ref: string; readonly action: UiElementAction; readonly value?: string },
656
+ describe?: (ref: string) => string | undefined,
657
+ ): string {
658
+ const target = describe?.(input.ref) ?? input.ref
659
+ switch (input.action) {
660
+ case 'invoke':
661
+ return `Press ${target}`
662
+ case 'set_value':
663
+ return `Set ${target} to ${quotedText(input.value ?? '')}`
664
+ case 'toggle':
665
+ return `Toggle ${target}`
666
+ case 'select':
667
+ return `Select ${target}`
668
+ case 'expand':
669
+ return `Expand ${target}`
670
+ case 'collapse':
671
+ return `Collapse ${target}`
672
+ case 'focus':
673
+ return `Focus ${target}`
674
+ case 'scroll_into_view':
675
+ return `Scroll to ${target}`
676
+ }
677
+ }
678
+
679
+ /**
680
+ * Human activity text for one action; the raw input remains the model-facing
681
+ * record. `describe` names a ui_act ref's control when the tool knows it.
682
+ */
683
+ function itemLabel(
684
+ input: BatchItem | ActionInput,
685
+ describe?: (ref: string) => string | undefined,
686
+ ): string {
213
687
  switch (input.type) {
214
688
  case 'screenshot':
215
689
  return 'Capture screenshot'
690
+ case 'zoom':
691
+ return `Zoom into ${input.region.width}x${input.region.height} at ${pointLabel(input.region)}`
216
692
  case 'cursor_position':
217
693
  return 'Read cursor position'
218
694
  case 'mouse_move':
@@ -227,33 +703,73 @@ function actionLabel(input: ActionInput): string {
227
703
  return `Type ${quotedText(input.text)}`
228
704
  case 'key':
229
705
  return `Press ${input.keys}`
706
+ case 'wait':
707
+ return waitLabel(input.ms)
708
+ case 'list_windows':
709
+ return 'List windows'
710
+ case 'focus_window':
711
+ return `Focus window ${input.window_id}`
712
+ case 'ui_snapshot':
713
+ return input.window_id
714
+ ? `Read the controls of window ${input.window_id}`
715
+ : 'Read the controls of the front window'
716
+ case 'ui_act':
717
+ return uiActLabel(input, describe)
718
+ case 'batch':
719
+ return batchLabel(input.actions, describe)
230
720
  }
231
721
  }
232
722
 
233
- function resultToToolResult(result: ComputerUseResult): ToolResult {
234
- switch (result.type) {
235
- case 'screenshot': {
236
- const { data, mimeType, width, height } = result.result
237
- // `output` used to BE the base64 payload, which meant the model
238
- // received 400 KB–2.7 MB of undecodable characters as text —
239
- // roughly 100k–670k tokens — and could not see the screen at
240
- // all. The image now travels as a content block; `output` keeps
241
- // the short human/transcript-facing description.
242
- return {
243
- success: true,
244
- output: `Screenshot captured (${width}x${height}, ${mimeType}).`,
245
- content: [{ type: 'image', data: data.toString('base64'), mediaType: mimeType }],
246
- data: { mimeType, width, height, encoding: 'base64' },
247
- }
248
- }
249
- case 'cursor_position':
250
- return {
251
- success: true,
252
- output: JSON.stringify(result.point),
253
- data: result.point,
254
- }
255
- case 'ok':
256
- return { success: true, output: 'ok' }
723
+ function batchLabel(
724
+ actions: readonly BatchItem[],
725
+ describe?: (ref: string) => string | undefined,
726
+ ): string {
727
+ const parts = actions.map((item) => itemLabel(item, describe))
728
+ const head = `${actions.length} desktop action${actions.length === 1 ? '' : 's'}`
729
+ const joined = parts.join(' · ')
730
+ return joined.length > 240 ? `${head}: ${joined.slice(0, 239)}…` : `${head}: ${joined}`
731
+ }
732
+
733
+ // ---------------------------------------------------------------------------
734
+ // Execution
735
+ // ---------------------------------------------------------------------------
736
+
737
+ interface ResolvedSettings {
738
+ readonly limits: ScreenshotLimits
739
+ readonly settleMs: number
740
+ readonly screenshotAfterActions: boolean
741
+ readonly maxBatchActions: number
742
+ readonly maxWaitMs: number
743
+ }
744
+
745
+ function resolveSettings(options: ComputerUseToolOptions): ResolvedSettings {
746
+ const count = (value: number | undefined, fallback: number, min: number, name: string) => {
747
+ if (value === undefined) return fallback
748
+ if (!Number.isSafeInteger(value) || value < min)
749
+ throw new RangeError(`createComputerUseTool: ${name} must be an integer ≥ ${min}`)
750
+ return value
751
+ }
752
+ const limits = options.screenshotLimits ?? STANDARD_SCREENSHOT_LIMITS
753
+ if (
754
+ !Number.isSafeInteger(limits.maxLongEdge) ||
755
+ !Number.isSafeInteger(limits.maxTiles) ||
756
+ limits.maxLongEdge < 28 ||
757
+ limits.maxTiles < 1
758
+ )
759
+ throw new RangeError(
760
+ 'createComputerUseTool: screenshotLimits needs an integer maxLongEdge ≥ 28 and maxTiles ≥ 1',
761
+ )
762
+ return {
763
+ limits,
764
+ settleMs: count(options.settleMs, DEFAULT_SETTLE_MS, 0, 'settleMs'),
765
+ screenshotAfterActions: options.screenshotAfterActions ?? true,
766
+ maxBatchActions: count(
767
+ options.maxBatchActions,
768
+ DEFAULT_MAX_BATCH_ACTIONS,
769
+ 1,
770
+ 'maxBatchActions',
771
+ ),
772
+ maxWaitMs: count(options.maxWaitMs, DEFAULT_MAX_WAIT_MS, 0, 'maxWaitMs'),
257
773
  }
258
774
  }
259
775
 
@@ -276,31 +792,133 @@ function isOutcomeUnknown(
276
792
  )
277
793
  }
278
794
 
279
- function unknownOutcomeToToolResult(error: ComputerUseOutcomeUnknown): ToolResult {
280
- return {
281
- success: false,
282
- output: '',
283
- error: error.message,
284
- data: {
285
- code: error.code,
286
- action: error.action,
287
- outcome: error.outcome,
288
- retrySafety: error.retrySafety,
289
- timedOut: error.timedOut,
290
- exitCode: error.exitCode,
291
- },
795
+ function errorText(error: unknown): string {
796
+ return error instanceof Error ? error.message : String(error)
797
+ }
798
+
799
+ /** One planned step: validated, coordinates already mapped, ready to run. */
800
+ interface PlannedStep {
801
+ readonly item: BatchItem
802
+ readonly label: string
803
+ readonly mutating: boolean
804
+ run(signal: AbortSignal | undefined): Promise<string>
805
+ }
806
+
807
+ type StepRecord =
808
+ | { readonly label: string; readonly status: 'done'; readonly note: string }
809
+ | {
810
+ readonly label: string
811
+ readonly status: 'failed'
812
+ readonly error: string
813
+ readonly unknown?: ComputerUseOutcomeUnknown
814
+ }
815
+ | { readonly label: string; readonly status: 'not run' }
816
+
817
+ class StepFailure extends Error {
818
+ constructor(
819
+ message: string,
820
+ readonly unknown?: ComputerUseOutcomeUnknown,
821
+ ) {
822
+ super(message)
292
823
  }
293
824
  }
294
825
 
826
+ interface Capture {
827
+ readonly frame: ScreenshotFrame
828
+ readonly image: FittedImage
829
+ }
830
+
831
+ /** A control of the latest ui_snapshot: the host's ref and what it was. */
832
+ interface UiRef {
833
+ readonly hostRef: string
834
+ readonly element: UiElement
835
+ }
836
+
837
+ /**
838
+ * The `computer_use` tool, with one question a host's review screen can ask
839
+ * of it.
840
+ */
841
+ export interface ComputerUseTool extends ToolDefinition<ActionInput> {
842
+ /**
843
+ * What a `ui_act` ref names in the latest `ui_snapshot` — `Button "Beş"
844
+ * (e30)` — or undefined for a ref it does not hold. For a host that shows
845
+ * a person the call before it runs; the words are the application's, so a
846
+ * host shows them as text and nothing more.
847
+ *
848
+ * @experimental Follows the UI-tree surface; see `ComputerUseCapabilities.uiTree`.
849
+ */
850
+ describeUiRef(ref: string): string | undefined
851
+ }
852
+
853
+ /** Whether a call sends the screen to the provider (see `ToolDefinition.capturesScreen`). */
854
+ function capturesScreenInput(input: unknown, screenshotAfterActions: boolean): boolean {
855
+ const items = batchActions(input) ?? [input]
856
+ return items.some((item) => {
857
+ if (!isRecord(item)) return true
858
+ const type = String(item.type)
859
+ if (SCREEN_OBSERVATIONS.has(type)) return true
860
+ if (type === 'cursor_position') return false
861
+ // Everything else is followed by a screenshot unless that is switched off.
862
+ return screenshotAfterActions
863
+ })
864
+ }
865
+
866
+ function safeRole(role: string): string {
867
+ return /^[A-Za-z][\w-]{0,39}$/.test(role) ? role : 'Element'
868
+ }
869
+
870
+ /** `Button "Beş"`: the control's role and name, one line, cut to fit. */
871
+ function uiElementText(element: UiElement): string {
872
+ const name = element.name.trim()
873
+ return name.length > 0 ? `${safeRole(element.role)} ${quotedText(name)}` : safeRole(element.role)
874
+ }
875
+
876
+ /**
877
+ * One line of a ui_snapshot: `[e12] Button "Beş" (disabled) [invoke] @(212, 488)`.
878
+ * The ref and position are the tool's; the rest is the application's, and
879
+ * sits inside the untrusted frame.
880
+ */
881
+ function uiElementLine(
882
+ element: UiElement,
883
+ ref: string | undefined,
884
+ frame: ScreenshotFrame | undefined,
885
+ ): string {
886
+ const parts = [`${ref ? `[${ref}] ` : ''}${uiElementText(element)}`]
887
+ if (element.value !== undefined && element.value !== element.name)
888
+ parts.push(`value=${quotedText(element.value)}`)
889
+ const states = (element.states ?? []).filter((state) => /^[a-z_]{1,24}$/.test(state))
890
+ if (states.length > 0) parts.push(`(${states.join(', ')})`)
891
+ if (ref) {
892
+ const actions = (element.actions ?? []).filter((action) =>
893
+ (UI_ACTIONS as readonly string[]).includes(action),
894
+ )
895
+ if (actions.length > 0) parts.push(`[${actions.join(', ')}]`)
896
+ const at = frame && element.bounds ? centreOnImage(frame, element.bounds) : undefined
897
+ if (at) parts.push(`@${pointLabel(at)}`)
898
+ }
899
+ return neutralizeEnvelopeDelimiter(parts.join(' '))
900
+ }
901
+
902
+ /** The centre of a virtual-desktop rectangle on a screenshot, or undefined when it is not on its display. */
903
+ function centreOnImage(frame: ScreenshotFrame, bounds: Rect): Point | undefined {
904
+ const x = Math.floor(bounds.x + bounds.width / 2) - frame.display.x
905
+ const y = Math.floor(bounds.y + bounds.height / 2) - frame.display.y
906
+ if (x < 0 || y < 0 || x >= frame.display.width || y >= frame.display.height) return undefined
907
+ return toImagePoint(frame, { x, y })
908
+ }
909
+
295
910
  /**
296
911
  * Factory: given a ComputerUseHost (provided by the consumer — e.g.
297
- * @namzu/computer-use's SubprocessComputerUseHost), returns a ToolDefinition
298
- * that routes the discriminated action to the host and maps results back to
299
- * the SDK's ToolResult shape.
912
+ * @namzu/computer-use's SubprocessComputerUseHost), returns the
913
+ * `computer_use` tool.
300
914
  *
301
- * The tool's description reflects the host's frozen capabilities, and any
302
- * action targeting an unavailable capability is rejected with a clear error
303
- * rather than hanging or failing silently.
915
+ * Every screenshot is fitted to the model's image limits and numbered;
916
+ * every coordinate the model sends is a pixel of one of those screenshots and
917
+ * is mapped onto the host's display. Actions that change something return a
918
+ * fresh screenshot after a short settle delay, and `batch` runs several in
919
+ * one call. The description and schema reflect the host's frozen
920
+ * capabilities, and any action targeting an unavailable capability is
921
+ * rejected before the host is touched.
304
922
  *
305
923
  * @example
306
924
  * ```ts
@@ -312,67 +930,743 @@ function unknownOutcomeToToolResult(error: ComputerUseOutcomeUnknown): ToolResul
312
930
  * registry.register(createComputerUseTool(host))
313
931
  * ```
314
932
  */
315
- export function createComputerUseTool(host: ComputerUseHost): ToolDefinition<ActionInput> {
316
- return defineTool({
933
+ export function createComputerUseTool(
934
+ host: ComputerUseHost,
935
+ options: ComputerUseToolOptions = {},
936
+ ): ComputerUseTool {
937
+ const settings = resolveSettings(options)
938
+ const caps: ComputerUseCapabilities = options.unavailableReason
939
+ ? {
940
+ ...host.capabilities,
941
+ screenshot: false,
942
+ mouse: false,
943
+ keyboard: false,
944
+ cursorPosition: false,
945
+ clipboard: false,
946
+ windows: false,
947
+ regionCapture: false,
948
+ uiTree: false,
949
+ unavailableReason: options.unavailableReason,
950
+ }
951
+ : host.capabilities
952
+ const frames = new ScreenshotFrames()
953
+ const available = new Set<string>(availableActions(host, caps))
954
+
955
+ // The controls of the latest ui_snapshot, by the ref the model was shown.
956
+ // Refs count up across snapshots, so a ref from an earlier one is never
957
+ // silently a different control of the latest: it is simply not here.
958
+ let uiRefs = new Map<string, UiRef>()
959
+ let uiSnapshots = 0
960
+ let uiRefCount = 0
961
+ const describeUiRef = (ref: string): string | undefined => {
962
+ const entry = uiRefs.get(ref)
963
+ return entry ? `${uiElementText(entry.element)} (${ref})` : undefined
964
+ }
965
+ const label = (input: BatchItem | ActionInput): string => itemLabel(input, describeUiRef)
966
+
967
+ const refusal = (type: string): string | null => {
968
+ const required = requiredCapability(type)
969
+ const why = caps.unavailableReason
970
+ ? ` ${caps.unavailableReason} Do not retry; tell the user.`
971
+ : ''
972
+ if (required !== null && caps[required] !== true)
973
+ return `computer_use: action "${type}" requires capability "${required}" which is not available on this host (displayServer=${caps.displayServer}).${why}`
974
+ if (!available.has(type))
975
+ return `computer_use: action "${type}" is not supported on this host.${why}`
976
+ return null
977
+ }
978
+
979
+ /** Capture the display, fit it, and number it. */
980
+ const capture = async (): Promise<Capture> => {
981
+ const result = await host.execute({ type: 'screenshot' })
982
+ if (result.type !== 'screenshot')
983
+ throw new Error(`computer_use: the host answered a screenshot with "${result.type}"`)
984
+ const shot = result.result
985
+ const image = await fitPng(shot.data, settings.limits)
986
+ const display = shot.display ?? assumedDisplay(image.sourceWidth, image.sourceHeight)
987
+ return { frame: frames.record(image, display), image }
988
+ }
989
+
990
+ const frameFor = (id: string | undefined): ScreenshotFrame => {
991
+ if (id !== undefined) {
992
+ const frame = frames.get(id)
993
+ if (!frame)
994
+ throw new StepFailure(
995
+ `there is no screenshot ${id} (the latest is ${frames.latest()?.id ?? 'none'}); take a screenshot and use its coordinates`,
996
+ )
997
+ return frame
998
+ }
999
+ const latest = frames.latest()
1000
+ if (!latest)
1001
+ throw new StepFailure(
1002
+ 'no screenshot has been taken yet, so there is nothing for coordinates to refer to; take a screenshot first',
1003
+ )
1004
+ return latest
1005
+ }
1006
+
1007
+ /**
1008
+ * Keys go to whichever window has focus, and a model that has not looked
1009
+ * does not know which one that is — the terminal running this agent is
1010
+ * the usual answer. When the host can show the screen, look first.
1011
+ */
1012
+ const requireLook = (): void => {
1013
+ if (available.has('screenshot') && !frames.latest())
1014
+ throw new StepFailure(
1015
+ 'no screenshot has been taken yet, so nothing shows which window has focus and would receive the keys; take a screenshot first',
1016
+ )
1017
+ }
1018
+
1019
+ const mapPoint = (frame: ScreenshotFrame, point: Point, field: string): Point => {
1020
+ if (!pointOnImage(frame, point))
1021
+ throw new StepFailure(
1022
+ `${field} ${pointLabel(point)} is outside screenshot ${frame.id}, which is ${frame.imageWidth}x${frame.imageHeight} (x 0–${frame.imageWidth - 1}, y 0–${frame.imageHeight - 1}); coordinates are pixels of the screenshot, not of the screen`,
1023
+ )
1024
+ return toDisplayPoint(frame, point)
1025
+ }
1026
+
1027
+ const runHost = async (action: ComputerUseAction): Promise<string> => {
1028
+ try {
1029
+ const result = await host.execute(action)
1030
+ if (result.type === 'cursor_position') {
1031
+ const frame = frames.latest()
1032
+ if (!frame) return `at display pixel ${pointLabel(result.point)}`
1033
+ return `at ${pointLabel(toImagePoint(frame, result.point))} in ${frame.id}`
1034
+ }
1035
+ return 'done'
1036
+ } catch (error) {
1037
+ if (isOutcomeUnknown(error, action.type)) throw new StepFailure(error.message, error)
1038
+ throw new StepFailure(errorText(error))
1039
+ }
1040
+ }
1041
+
1042
+ const plan = (item: BatchItem, frameId: string | undefined): PlannedStep => {
1043
+ const denied = refusal(item.type)
1044
+ if (denied) throw new StepFailure(denied)
1045
+ const mutating = !READ_ONLY_ACTIONS.has(item.type)
1046
+ const step = (run: PlannedStep['run']): PlannedStep => ({
1047
+ item,
1048
+ label: label(item),
1049
+ mutating,
1050
+ run,
1051
+ })
1052
+ switch (item.type) {
1053
+ case 'cursor_position':
1054
+ frameFor(frameId)
1055
+ return step(() => runHost({ type: 'cursor_position' }))
1056
+ case 'mouse_move': {
1057
+ const frame = frameFor(frameId)
1058
+ const to = mapPoint(frame, item.to, 'to')
1059
+ return step(() => runHost({ type: 'mouse_move', to }))
1060
+ }
1061
+ case 'mouse_click': {
1062
+ const buttons = caps.mouseClickButtons
1063
+ if (buttons && !buttons.includes(item.button))
1064
+ throw new StepFailure(
1065
+ `computer_use: action "mouse_click" does not support button "${item.button}" on this host.`,
1066
+ )
1067
+ const frame = frameFor(frameId)
1068
+ const at = mapPoint(frame, item.at, 'at')
1069
+ return step(() => runHost({ type: 'mouse_click', at, button: item.button }))
1070
+ }
1071
+ case 'mouse_drag': {
1072
+ const buttons = caps.mouseDragButtons
1073
+ if (buttons && !buttons.includes(item.button))
1074
+ throw new StepFailure(
1075
+ `computer_use: action "mouse_drag" does not support button "${item.button}" on this host.`,
1076
+ )
1077
+ const frame = frameFor(frameId)
1078
+ const from = mapPoint(frame, item.from, 'from')
1079
+ const to = mapPoint(frame, item.to, 'to')
1080
+ return step(() => runHost({ type: 'mouse_drag', from, to, button: item.button }))
1081
+ }
1082
+ case 'scroll': {
1083
+ const frame = frameFor(frameId)
1084
+ const at = mapPoint(frame, item.at, 'at')
1085
+ return step(() =>
1086
+ runHost({ type: 'scroll', at, direction: item.direction, amount: item.amount }),
1087
+ )
1088
+ }
1089
+ case 'type_text':
1090
+ requireLook()
1091
+ return step(() => runHost({ type: 'type_text', text: item.text }))
1092
+ case 'key':
1093
+ requireLook()
1094
+ return step(() => runHost({ type: 'key', keys: item.keys }))
1095
+ case 'wait':
1096
+ return step(async (signal) => {
1097
+ await sleep(item.ms, signal)
1098
+ return 'done'
1099
+ })
1100
+ case 'list_windows':
1101
+ return step(() => listWindows())
1102
+ case 'focus_window':
1103
+ return step(async () => {
1104
+ let outcome: Awaited<ReturnType<NonNullable<ComputerUseHost['focusWindow']>>>
1105
+ try {
1106
+ outcome = await (host.focusWindow as NonNullable<ComputerUseHost['focusWindow']>)(
1107
+ item.window_id,
1108
+ )
1109
+ } catch (error) {
1110
+ throw new StepFailure(errorText(error))
1111
+ }
1112
+ if (!outcome.ok)
1113
+ throw new StepFailure(
1114
+ `window ${item.window_id} could not be brought to the front; ${outcome.focusedId ? `window ${outcome.focusedId} is in front` : 'the window in front could not be read'}`,
1115
+ )
1116
+ return 'done'
1117
+ })
1118
+ case 'ui_act': {
1119
+ const entry = uiRefs.get(item.ref)
1120
+ if (!entry)
1121
+ throw new StepFailure(
1122
+ uiSnapshots === 0
1123
+ ? `there is no ${item.ref}: no ui_snapshot has been taken yet; take one and use its refs`
1124
+ : `${item.ref} is not a control of the latest ui_snapshot (u${uiSnapshots}); refs are valid only until the next ui_snapshot, so use the refs it showed`,
1125
+ )
1126
+ const offered = entry.element.actions
1127
+ if (offered && offered.length > 0 && !offered.includes(item.action))
1128
+ throw new StepFailure(
1129
+ `${describeUiRef(item.ref)} offers ${offered.join(', ')}, not ${item.action}`,
1130
+ )
1131
+ if (item.action === 'set_value' && item.value === undefined)
1132
+ throw new StepFailure('set_value needs value, the text to put in the control')
1133
+ return step(async () => {
1134
+ let outcome: UiActResult
1135
+ try {
1136
+ outcome = await (host.uiAct as NonNullable<ComputerUseHost['uiAct']>)(
1137
+ entry.hostRef,
1138
+ item.action,
1139
+ item.value,
1140
+ )
1141
+ } catch (error) {
1142
+ throw new StepFailure(errorText(error))
1143
+ }
1144
+ if (!outcome.ok)
1145
+ throw new StepFailure(
1146
+ outcome.detail ?? `${item.action} did not take effect on ${describeUiRef(item.ref)}`,
1147
+ )
1148
+ return outcome.detail ? `done (${outcome.detail})` : 'done'
1149
+ })
1150
+ }
1151
+ }
1152
+ }
1153
+
1154
+ /** Read one window's controls, number the ones the model can act on, and show them. */
1155
+ const uiSnapshot = async (
1156
+ input: Extract<ActionInput, { type: 'ui_snapshot' }>,
1157
+ ): Promise<ToolResult> => {
1158
+ let snapshot: UiSnapshot
1159
+ try {
1160
+ snapshot = await (host.uiSnapshot as NonNullable<ComputerUseHost['uiSnapshot']>)(
1161
+ input.window_id,
1162
+ )
1163
+ } catch (error) {
1164
+ throw new StepFailure(errorText(error))
1165
+ }
1166
+ uiSnapshots += 1
1167
+ const id = `u${uiSnapshots}`
1168
+ const refs = new Map<string, UiRef>()
1169
+ const frame = frames.latest()
1170
+ const lines: string[] = []
1171
+ let chars = 0
1172
+ let shown = 0
1173
+ let total = 0
1174
+ let cut = false
1175
+ const visit = (element: UiElement, depth: number): void => {
1176
+ total += 1
1177
+ const actionable = element.ref.length > 0
1178
+ const children = element.children ?? []
1179
+ // A nameless control nobody can act on says nothing by itself; its
1180
+ // children still stand, one level up.
1181
+ const silent = !actionable && element.name.trim().length === 0 && element.value === undefined
1182
+ if (!silent && !cut) {
1183
+ const ref = actionable ? `e${uiRefCount + 1}` : undefined
1184
+ const line = `${' '.repeat(depth)}${uiElementLine(element, ref, frame)}`
1185
+ if (chars + line.length + 1 > MAX_UI_SNAPSHOT_CHARS) {
1186
+ cut = true
1187
+ } else {
1188
+ lines.push(line)
1189
+ chars += line.length + 1
1190
+ shown += 1
1191
+ if (ref) {
1192
+ uiRefCount += 1
1193
+ refs.set(ref, { hostRef: element.ref, element })
1194
+ }
1195
+ }
1196
+ }
1197
+ for (const child of children) visit(child, silent ? depth : depth + 1)
1198
+ }
1199
+ visit(snapshot.root, 0)
1200
+ uiRefs = refs
1201
+ const window =
1202
+ snapshot.windowId && /^[\w.:-]{1,64}$/.test(snapshot.windowId) ? snapshot.windowId : undefined
1203
+ const header = [
1204
+ `UI snapshot ${id}${window ? ` of window ${window}` : ''}: ${shown} controls shown, ${refs.size} with a ref you can pass to ui_act. Refs are valid until the next ui_snapshot.`,
1205
+ ...(frame
1206
+ ? [
1207
+ `@(x, y) is a control's centre on screenshot ${frame.id}, for a click when ui_act cannot reach it.`,
1208
+ ]
1209
+ : []),
1210
+ ]
1211
+ // The window's title and application, when the tree's own root does not
1212
+ // already say them; inside the frame, since an application sets both.
1213
+ const about = [
1214
+ ...(snapshot.app !== undefined ? [`Application ${quotedText(snapshot.app)}`] : []),
1215
+ ...(snapshot.title !== undefined && snapshot.title !== snapshot.root.name
1216
+ ? [`Window title ${quotedText(snapshot.title)}`]
1217
+ : []),
1218
+ ]
1219
+ const body = [...about, ...lines].join('\n')
1220
+ const footer =
1221
+ cut || snapshot.truncated
1222
+ ? [
1223
+ cut
1224
+ ? `The tree was cut at ${MAX_UI_SNAPSHOT_CHARS} characters after ${shown} of ${total} controls. Read a smaller window, or use a screenshot for the rest.`
1225
+ : 'The host stopped reading the tree before it ended; controls further down are missing. Use a screenshot for the rest.',
1226
+ ]
1227
+ : []
1228
+ const text = [
1229
+ ...header,
1230
+ wrapUntrusted(
1231
+ {
1232
+ kind: 'desktop-ui',
1233
+ ...(window ? { attributes: { window } } : {}),
1234
+ provenance:
1235
+ "The accessibility tree of a window on the user's desktop, as the host read it. Names and values are whatever the application shows, which can include text anyone wrote.",
1236
+ },
1237
+ body,
1238
+ ),
1239
+ ...footer,
1240
+ ].join('\n')
1241
+ return {
1242
+ success: true,
1243
+ output: header[0] ?? '',
1244
+ content: [{ type: 'text', text }],
1245
+ data: {
1246
+ uiSnapshot: {
1247
+ id,
1248
+ ...(window ? { windowId: window } : {}),
1249
+ controls: shown,
1250
+ refs: refs.size,
1251
+ truncated: cut || snapshot.truncated === true,
1252
+ },
1253
+ },
1254
+ }
1255
+ }
1256
+
1257
+ /**
1258
+ * Keys and text go to the window in front, and a terminal there — the
1259
+ * one running this agent, or the user's own — turns a model's typing into
1260
+ * a command line: ENTER runs it or sends it. A session had a WIN+R that
1261
+ * did not open the Run dialog type "notepad" and ENTER into the user's
1262
+ * terminal, and submitted their half-typed message. So with a window list
1263
+ * the tool reads what is in front before typing, and refuses a terminal.
1264
+ * Without one it cannot tell, and the description's advice stands alone.
1265
+ */
1266
+ const refuseTerminalInFront = async (): Promise<void> => {
1267
+ if (!available.has('list_windows')) return
1268
+ let windows: readonly WindowInfo[]
1269
+ try {
1270
+ windows = await (host.listWindows as NonNullable<ComputerUseHost['listWindows']>)()
1271
+ } catch (error) {
1272
+ throw new StepFailure(
1273
+ `the window in front could not be read, so nothing was typed: ${errorText(error)}`,
1274
+ )
1275
+ }
1276
+ const front = windows.find((window) => window.focused)
1277
+ if (front && TERMINAL_APPS.test(front.app))
1278
+ throw new StepFailure(
1279
+ `a terminal is in front (${front.app}, window ${front.id}), so nothing was typed: computer_use never sends keys or text to a terminal. Bring the window you mean to the front (focus_window, or click it) and try again; for a command, use a shell tool instead`,
1280
+ )
1281
+ }
1282
+
1283
+ const listWindows = async (): Promise<string> => {
1284
+ let windows: readonly WindowInfo[]
1285
+ try {
1286
+ windows = await (host.listWindows as NonNullable<ComputerUseHost['listWindows']>)()
1287
+ } catch (error) {
1288
+ throw new StepFailure(errorText(error))
1289
+ }
1290
+ if (windows.length === 0) return 'no windows are open'
1291
+ const frame = frames.latest()
1292
+ const lines = windows.slice(0, MAX_LISTED_WINDOWS).map((window) => {
1293
+ const where = frame
1294
+ ? (() => {
1295
+ const rect = window.minimized ? null : desktopRectOnImage(frame, window.bounds)
1296
+ return rect
1297
+ ? `at ${pointLabel(rect)} ${rect.width}x${rect.height} in ${frame.id}`
1298
+ : window.minimized
1299
+ ? 'minimized'
1300
+ : `not on the display of ${frame.id}`
1301
+ })()
1302
+ : window.minimized
1303
+ ? 'minimized'
1304
+ : ''
1305
+ return [
1306
+ `- ${window.id}`,
1307
+ JSON.stringify(window.title),
1308
+ `${window.app} (pid ${window.pid})`,
1309
+ ...(window.focused ? ['focused'] : []),
1310
+ ...(where ? [where] : []),
1311
+ ].join(' · ')
1312
+ })
1313
+ const more =
1314
+ windows.length > MAX_LISTED_WINDOWS
1315
+ ? [`… and ${windows.length - MAX_LISTED_WINDOWS} more`]
1316
+ : []
1317
+ // Titles are whatever an application puts there — a web page's title in
1318
+ // a browser window — so the list is framed as material, not direction.
1319
+ return `${windows.length} window${windows.length === 1 ? '' : 's'}, front to back:\n${wrapUntrusted(
1320
+ {
1321
+ kind: 'desktop-windows',
1322
+ provenance:
1323
+ "The open windows on the user's desktop, as the host listed them. Titles are whatever each application shows, which can include text anyone wrote.",
1324
+ },
1325
+ [...lines, ...more].join('\n'),
1326
+ )}`
1327
+ }
1328
+
1329
+ const describeFrame = (frame: ScreenshotFrame): string => {
1330
+ const { display } = frame
1331
+ const scaled = frame.imageWidth !== display.width || frame.imageHeight !== display.height
1332
+ return scaled
1333
+ ? `Screenshot ${frame.id}: ${frame.imageWidth}x${frame.imageHeight} pixels, showing the ${display.width}x${display.height} display. Send coordinates in this image's pixels (x 0–${frame.imageWidth - 1}, y 0–${frame.imageHeight - 1}); the tool maps them onto the display.`
1334
+ : `Screenshot ${frame.id}: ${frame.imageWidth}x${frame.imageHeight} pixels, the display at full size. Send coordinates in this image's pixels (x 0–${frame.imageWidth - 1}, y 0–${frame.imageHeight - 1}).`
1335
+ }
1336
+
1337
+ const frameData = (frame: ScreenshotFrame) => ({
1338
+ id: frame.id,
1339
+ width: frame.imageWidth,
1340
+ height: frame.imageHeight,
1341
+ display: frame.display,
1342
+ mimeType: 'image/png' as const,
1343
+ encoding: 'base64' as const,
1344
+ })
1345
+
1346
+ const pin = (frame: ScreenshotFrame): NonNullable<ToolResult['workingState']> => [
1347
+ {
1348
+ key: 'computer_use.screenshot',
1349
+ text: `computer_use coordinates are pixels of screenshot ${frame.id} (${frame.imageWidth}x${frame.imageHeight}), origin top-left.`,
1350
+ },
1351
+ ]
1352
+
1353
+ const imageBlock = (image: FittedImage): ToolResultBlock => ({
1354
+ type: 'image',
1355
+ data: image.data.toString('base64'),
1356
+ mediaType: 'image/png',
1357
+ })
1358
+
1359
+ // Said once, beside the first screenshot, where it is read: a model that
1360
+ // has just seen the screen reaches for pixels unless told the host can
1361
+ // name the controls.
1362
+ const firstLookHint = (frame: ScreenshotFrame): string[] =>
1363
+ frame.id === 's1' && available.has('ui_snapshot')
1364
+ ? [
1365
+ 'This host can also read a window’s controls: list_windows, then ui_snapshot {window_id}, then ui_act by ref (a batch of them for several buttons) — surer than clicking pixels in an ordinary application.',
1366
+ ]
1367
+ : []
1368
+
1369
+ const screenshotResult = (shot: Capture): ToolResult => ({
1370
+ success: true,
1371
+ output: `Screenshot ${shot.frame.id} captured (${shot.frame.imageWidth}x${shot.frame.imageHeight} of the ${shot.frame.display.width}x${shot.frame.display.height} display).`,
1372
+ content: [
1373
+ { type: 'text', text: [describeFrame(shot.frame), ...firstLookHint(shot.frame)].join('\n') },
1374
+ imageBlock(shot.image),
1375
+ ],
1376
+ data: { screenshot: frameData(shot.frame) },
1377
+ workingState: pin(shot.frame),
1378
+ })
1379
+
1380
+ const zoom = async (input: Extract<ActionInput, { type: 'zoom' }>): Promise<ToolResult> => {
1381
+ const frame = frameFor(input.screenshot_id)
1382
+ const { region } = input
1383
+ const onImage = {
1384
+ x: Math.max(0, region.x),
1385
+ y: Math.max(0, region.y),
1386
+ width: Math.min(region.x + region.width, frame.imageWidth) - Math.max(0, region.x),
1387
+ height: Math.min(region.y + region.height, frame.imageHeight) - Math.max(0, region.y),
1388
+ }
1389
+ if (onImage.width <= 0 || onImage.height <= 0)
1390
+ throw new StepFailure(
1391
+ `region ${region.width}x${region.height} at ${pointLabel(region)} is outside screenshot ${frame.id} (${frame.imageWidth}x${frame.imageHeight})`,
1392
+ )
1393
+ const rect = toDisplayRect(frame, onImage) as Rect
1394
+ let fitted: FittedImage
1395
+ if (caps.regionCapture === true && typeof host.captureRegion === 'function') {
1396
+ const piece = await host.captureRegion(rect)
1397
+ fitted = await fitPng(piece.data, settings.limits)
1398
+ } else {
1399
+ const result = await host.execute({ type: 'screenshot' })
1400
+ if (result.type !== 'screenshot')
1401
+ throw new Error(`computer_use: the host answered a screenshot with "${result.type}"`)
1402
+ const full: ScreenshotResult = result.result
1403
+ const size = pngSize(full.data)
1404
+ const display = full.display ?? assumedDisplay(size.width, size.height)
1405
+ if (display.width !== frame.display.width || display.height !== frame.display.height)
1406
+ throw new StepFailure(
1407
+ `the display is now ${display.width}x${display.height}, not what ${frame.id} showed; take a new screenshot`,
1408
+ )
1409
+ // The capture's own pixels, in case a host captures at another scale than it reports.
1410
+ const sx = size.width / display.width
1411
+ const sy = size.height / display.height
1412
+ const left = Math.floor(rect.x * sx)
1413
+ const top = Math.floor(rect.y * sy)
1414
+ fitted = await cropAndFitPng(
1415
+ full.data,
1416
+ {
1417
+ x: left,
1418
+ y: top,
1419
+ width: Math.max(1, Math.min(size.width, Math.ceil((rect.x + rect.width) * sx)) - left),
1420
+ height: Math.max(1, Math.min(size.height, Math.ceil((rect.y + rect.height) * sy)) - top),
1421
+ },
1422
+ settings.limits,
1423
+ )
1424
+ }
1425
+ const detail = fitted.width / onImage.width
1426
+ const text = `Zoom of ${onImage.width}x${onImage.height} at ${pointLabel(onImage)} in ${frame.id}, shown at ${fitted.width}x${fitted.height} (${detail.toFixed(1)}x the detail of ${frame.id}). Coordinates for actions still refer to ${frame.id} (${frame.imageWidth}x${frame.imageHeight}), not to this image.`
1427
+ return {
1428
+ success: true,
1429
+ output: `Zoomed into ${onImage.width}x${onImage.height} at ${pointLabel(onImage)} of ${frame.id} (shown at ${fitted.width}x${fitted.height}).`,
1430
+ content: [{ type: 'text', text }, imageBlock(fitted)],
1431
+ data: {
1432
+ zoom: {
1433
+ screenshot: frame.id,
1434
+ region: onImage,
1435
+ width: fitted.width,
1436
+ height: fitted.height,
1437
+ mimeType: 'image/png',
1438
+ encoding: 'base64',
1439
+ },
1440
+ },
1441
+ }
1442
+ }
1443
+
1444
+ /** Run actions in order, stop at the first failure, then show the screen once. */
1445
+ const runSteps = async (
1446
+ items: readonly BatchItem[],
1447
+ frameId: string | undefined,
1448
+ context: ToolContext | undefined,
1449
+ single: boolean,
1450
+ ): Promise<ToolResult> => {
1451
+ if (items.length > settings.maxBatchActions)
1452
+ return {
1453
+ success: false,
1454
+ output: '',
1455
+ error: `computer_use: a batch carries at most ${settings.maxBatchActions} actions; got ${items.length}. Nothing was run.`,
1456
+ }
1457
+ // Waits share one budget, so a batch stays well inside the tool's
1458
+ // deadline however its waits are split.
1459
+ const waited = items.reduce((total, item) => total + (item.type === 'wait' ? item.ms : 0), 0)
1460
+ if (waited > settings.maxWaitMs)
1461
+ return {
1462
+ success: false,
1463
+ output: '',
1464
+ error: `computer_use: ${single ? 'wait is' : "a batch's waits add up to"} at most ${settings.maxWaitMs} ms; got ${waited}. Nothing was run.`,
1465
+ }
1466
+ // Plan everything before doing anything: a refusal or a coordinate off
1467
+ // the screenshot in step 5 should not arrive after steps 1–4 changed
1468
+ // the desktop.
1469
+ const planned: PlannedStep[] = []
1470
+ for (const [index, item] of items.entries()) {
1471
+ try {
1472
+ planned.push(plan(item, frameId))
1473
+ } catch (error) {
1474
+ const message = errorText(error)
1475
+ return {
1476
+ success: false,
1477
+ output: '',
1478
+ error: single
1479
+ ? message
1480
+ : `computer_use: action ${index + 1} of ${items.length} (${label(item)}) cannot run: ${message}. Nothing was run.`,
1481
+ }
1482
+ }
1483
+ }
1484
+
1485
+ const signal = context?.abortSignal
1486
+ const records: StepRecord[] = []
1487
+ let changed = false
1488
+ let failure: { index: number; error: string; unknown?: ComputerUseOutcomeUnknown } | undefined
1489
+ // Whether the window in front was checked since the last step that could
1490
+ // have changed it. Typing does not move focus; everything else may.
1491
+ let frontChecked = false
1492
+ for (const [index, step] of planned.entries()) {
1493
+ if (signal?.aborted) {
1494
+ failure = { index, error: 'cancelled before it started' }
1495
+ records.push({ label: step.label, status: 'failed', error: failure.error })
1496
+ break
1497
+ }
1498
+ try {
1499
+ const keyboard = step.item.type === 'type_text' || step.item.type === 'key'
1500
+ if (keyboard && !frontChecked) {
1501
+ await refuseTerminalInFront()
1502
+ frontChecked = true
1503
+ }
1504
+ if (step.item.type !== 'type_text') frontChecked = false
1505
+ const note = await step.run(signal)
1506
+ records.push({ label: step.label, status: 'done', note })
1507
+ if (step.mutating) changed = true
1508
+ } catch (error) {
1509
+ const unknown = error instanceof StepFailure ? error.unknown : undefined
1510
+ const message = signal?.aborted ? 'cancelled' : errorText(error)
1511
+ failure = { index, error: message, ...(unknown ? { unknown } : {}) }
1512
+ records.push({
1513
+ label: step.label,
1514
+ status: 'failed',
1515
+ error: message,
1516
+ ...(unknown ? { unknown } : {}),
1517
+ })
1518
+ if (unknown) changed = true
1519
+ break
1520
+ }
1521
+ }
1522
+ for (const step of planned.slice(records.length))
1523
+ records.push({ label: step.label, status: 'not run' })
1524
+
1525
+ // The screen after the batch: after anything that may have changed it,
1526
+ // and after a wait, whose whole point is to look again.
1527
+ const looks =
1528
+ settings.screenshotAfterActions &&
1529
+ available.has('screenshot') &&
1530
+ (changed || planned.some((step) => step.item.type === 'wait')) &&
1531
+ !signal?.aborted
1532
+ let shot: Capture | undefined
1533
+ let shotError: string | undefined
1534
+ if (looks) {
1535
+ try {
1536
+ await sleep(settings.settleMs, signal)
1537
+ shot = await capture()
1538
+ } catch (error) {
1539
+ shotError = signal?.aborted ? 'cancelled' : errorText(error)
1540
+ }
1541
+ }
1542
+
1543
+ const summary = (record: StepRecord, index: number): string => {
1544
+ const n = single ? '' : `${index + 1}. `
1545
+ switch (record.status) {
1546
+ case 'done':
1547
+ return `${n}${record.label}: ${record.note}`
1548
+ case 'failed':
1549
+ return `${n}${record.label}: failed — ${record.error}`
1550
+ case 'not run':
1551
+ return `${n}${record.label}: not run`
1552
+ }
1553
+ }
1554
+ const header = single
1555
+ ? []
1556
+ : failure
1557
+ ? [
1558
+ `Batch stopped at action ${failure.index + 1} of ${planned.length}; ${planned.length - failure.index - 1} not run.`,
1559
+ ]
1560
+ : [`Batch: all ${planned.length} actions done.`]
1561
+ const stepLines = records.map(summary)
1562
+ const shotLines = shot
1563
+ ? [describeFrame(shot.frame)]
1564
+ : shotError
1565
+ ? [`The screenshot after acting failed: ${shotError}. Take one before acting again.`]
1566
+ : []
1567
+ const unknownLines = failure?.unknown ? [failure.unknown.message] : []
1568
+ const text = [...header, ...stepLines, ...unknownLines, ...shotLines].join('\n')
1569
+ const output = [
1570
+ ...header,
1571
+ ...stepLines,
1572
+ ...(shot
1573
+ ? [`Screenshot ${shot.frame.id} (${shot.frame.imageWidth}x${shot.frame.imageHeight}).`]
1574
+ : []),
1575
+ ].join('\n')
1576
+ const data = {
1577
+ steps: records.map((record) =>
1578
+ record.status === 'failed'
1579
+ ? { label: record.label, status: record.status, error: record.error }
1580
+ : { label: record.label, status: record.status },
1581
+ ),
1582
+ ...(shot ? { screenshot: frameData(shot.frame) } : {}),
1583
+ ...(failure?.unknown
1584
+ ? {
1585
+ code: failure.unknown.code,
1586
+ action: failure.unknown.action,
1587
+ outcome: failure.unknown.outcome,
1588
+ retrySafety: failure.unknown.retrySafety,
1589
+ timedOut: failure.unknown.timedOut,
1590
+ exitCode: failure.unknown.exitCode,
1591
+ }
1592
+ : {}),
1593
+ }
1594
+ const content: ToolResultBlock[] = [{ type: 'text', text }]
1595
+ if (shot) content.push(imageBlock(shot.image))
1596
+ const result: ToolResult = {
1597
+ success: failure === undefined,
1598
+ output,
1599
+ content,
1600
+ data,
1601
+ ...(shot ? { workingState: pin(shot.frame) } : {}),
1602
+ }
1603
+ if (failure)
1604
+ result.error = failure.unknown
1605
+ ? failure.unknown.message
1606
+ : single
1607
+ ? `computer_use failed: ${failure.error}`
1608
+ : `Batch stopped at action ${failure.index + 1} of ${planned.length} (${planned[failure.index]?.label}): ${failure.error}`
1609
+ return result
1610
+ }
1611
+
1612
+ // Whether a call sends the screen to the provider: an observation, or an
1613
+ // action that returns a screenshot afterwards. A diagnostic tool that can
1614
+ // reach nothing captures nothing.
1615
+ const observes =
1616
+ available.has('screenshot') || available.has('list_windows') || available.has('ui_snapshot')
1617
+ const capturesScreen = (input: ActionInput): boolean =>
1618
+ observes && capturesScreenInput(input, settings.screenshotAfterActions)
1619
+
1620
+ const tool = defineTool({
317
1621
  name: COMPUTER_USE_TOOL_NAME,
318
- description: buildDescription(host),
1622
+ description: buildDescription(host, caps, settings),
319
1623
  inputSchema: actionSchema,
320
- modelInputSchema: hostModelSchema(host.capabilities),
321
- validationErrorHint:
322
- 'Action requirements: mouse_move needs "to"; mouse_click needs "at" and "button"; mouse_drag needs "from", "to", and "button"; scroll needs "at", "direction", and positive "amount"; type_text needs "text"; key needs "keys".',
1624
+ modelInputSchema: hostModelSchema(host, caps, settings),
1625
+ validationErrorHint: `Action requirements: ${ALL_ACTIONS.map((action) => ACTION_REQUIREMENTS[action]).join('; ')}. A batch is {"type":"batch","actions":[...]} with at most ${settings.maxBatchActions} actions and no screenshot, zoom, ui_snapshot or batch inside.`,
323
1626
  category: 'custom',
324
1627
  permissions: [],
325
- readOnly: false,
326
- destructive: (input: ActionInput) => DESTRUCTIVE_ACTION_TYPES.has(input.type),
1628
+ readOnly: (input: ActionInput) => isReadOnlyInput(input),
1629
+ destructive: (input: ActionInput) => isDestructiveInput(input),
1630
+ capturesScreen,
327
1631
  concurrencySafe: false,
328
1632
  presentCall: (input) => ({
329
1633
  kind: 'generic',
330
- label: actionLabel(input),
1634
+ label: label(input),
331
1635
  presentation: 'activity',
332
1636
  }),
1637
+ // Decided from the result alone: a host may present a finished call
1638
+ // without its input (the CLI passes `{}`). An action that only reports
1639
+ // "done" — and the screenshot that followed it, which is for the model —
1640
+ // adds nothing to the call row; observations, batches and failures do.
333
1641
  presentResult: (_input, result) =>
334
- result.success && result.output.trim().toLowerCase() === 'ok'
1642
+ result.success && ACKNOWLEDGEMENT.test(result.output)
335
1643
  ? { kind: 'generic', label: result.output, visibility: 'hidden' }
336
1644
  : undefined,
337
1645
 
338
- async execute(input, _context): Promise<ToolResult> {
339
- const required = requiredCapability(input.type)
340
- if (required !== null && host.capabilities[required] !== true) {
341
- return {
342
- success: false,
343
- output: '',
344
- error: `computer_use: action "${input.type}" requires capability "${required}" which is not available on this host (displayServer=${host.capabilities.displayServer}).${host.capabilities.unavailableReason ? ` ${host.capabilities.unavailableReason} Do not retry; tell the user.` : ''}`,
345
- }
346
- }
347
- if (!availableActions(host.capabilities).includes(input.type)) {
348
- return {
349
- success: false,
350
- output: '',
351
- error: `computer_use: action "${input.type}" is not supported on this host.`,
352
- }
353
- }
354
- const buttons =
355
- input.type === 'mouse_click'
356
- ? host.capabilities.mouseClickButtons
357
- : input.type === 'mouse_drag'
358
- ? host.capabilities.mouseDragButtons
359
- : undefined
360
- if (buttons && 'button' in input && !buttons.includes(input.button)) {
361
- return {
362
- success: false,
363
- output: '',
364
- error: `computer_use: action "${input.type}" does not support button "${input.button}" on this host.`,
365
- }
366
- }
1646
+ async execute(input, context): Promise<ToolResult> {
1647
+ const denied = refusal(input.type)
1648
+ if (denied) return { success: false, output: '', error: denied }
367
1649
  try {
368
- const result = await host.execute(input as ComputerUseAction)
369
- return resultToToolResult(result)
370
- } catch (error) {
371
- if (isOutcomeUnknown(error, input.type)) {
372
- return unknownOutcomeToToolResult(error)
1650
+ switch (input.type) {
1651
+ case 'screenshot':
1652
+ return screenshotResult(await capture())
1653
+ case 'zoom':
1654
+ return await zoom(input)
1655
+ case 'ui_snapshot':
1656
+ return await uiSnapshot(input)
1657
+ case 'batch':
1658
+ return await runSteps(input.actions, input.screenshot_id, context, false)
1659
+ default: {
1660
+ const { screenshot_id: frameId, ...item } = input
1661
+ return await runSteps([item as BatchItem], frameId, context, true)
1662
+ }
373
1663
  }
1664
+ } catch (error) {
1665
+ if (error instanceof StepFailure)
1666
+ return { success: false, output: '', error: `computer_use failed: ${error.message}` }
374
1667
  throw error
375
1668
  }
376
1669
  },
377
1670
  })
1671
+ return Object.assign(tool, { describeUiRef })
378
1672
  }