crewly 1.20.40 → 1.20.48

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (80) hide show
  1. package/config/skills/_common/desktop-guards.sh +485 -0
  2. package/config/skills/_common/desktop-guards.test.sh +242 -0
  3. package/config/skills/_common/desktop-perceive.swift +530 -0
  4. package/config/skills/_common/desktop-presence.swift +343 -0
  5. package/config/skills/agent/_common/desktop-guards.sh +4 -0
  6. package/config/skills/agent/computer-use/SKILL.md +88 -0
  7. package/config/skills/agent/computer-use/execute.sh +249 -3
  8. package/config/skills/agent/desktop-app-control/SKILL.md +19 -0
  9. package/config/skills/agent/remote-browser/SKILL.md +19 -0
  10. package/dist/backend/backend/src/controllers/desktop/desktop.controller.d.ts +105 -0
  11. package/dist/backend/backend/src/controllers/desktop/desktop.controller.d.ts.map +1 -0
  12. package/dist/backend/backend/src/controllers/desktop/desktop.controller.js +278 -0
  13. package/dist/backend/backend/src/controllers/desktop/desktop.controller.js.map +1 -0
  14. package/dist/backend/backend/src/controllers/desktop/desktop.routes.d.ts +21 -0
  15. package/dist/backend/backend/src/controllers/desktop/desktop.routes.d.ts.map +1 -0
  16. package/dist/backend/backend/src/controllers/desktop/desktop.routes.js +31 -0
  17. package/dist/backend/backend/src/controllers/desktop/desktop.routes.js.map +1 -0
  18. package/dist/backend/backend/src/routes/api.routes.d.ts.map +1 -1
  19. package/dist/backend/backend/src/routes/api.routes.js +3 -0
  20. package/dist/backend/backend/src/routes/api.routes.js.map +1 -1
  21. package/dist/backend/backend/src/services/cloud/mobile-api-relay.service.d.ts.map +1 -1
  22. package/dist/backend/backend/src/services/cloud/mobile-api-relay.service.js +9 -0
  23. package/dist/backend/backend/src/services/cloud/mobile-api-relay.service.js.map +1 -1
  24. package/dist/backend/backend/src/services/slack/slack-orchestrator-bridge.d.ts +18 -0
  25. package/dist/backend/backend/src/services/slack/slack-orchestrator-bridge.d.ts.map +1 -1
  26. package/dist/backend/backend/src/services/slack/slack-orchestrator-bridge.js +33 -6
  27. package/dist/backend/backend/src/services/slack/slack-orchestrator-bridge.js.map +1 -1
  28. package/dist/backend/backend/src/services/slack/slack-team-channel.service.d.ts +24 -0
  29. package/dist/backend/backend/src/services/slack/slack-team-channel.service.d.ts.map +1 -1
  30. package/dist/backend/backend/src/services/slack/slack-team-channel.service.js +60 -1
  31. package/dist/backend/backend/src/services/slack/slack-team-channel.service.js.map +1 -1
  32. package/dist/backend/backend/src/services/slack/slack.service.d.ts +18 -1
  33. package/dist/backend/backend/src/services/slack/slack.service.d.ts.map +1 -1
  34. package/dist/backend/backend/src/services/slack/slack.service.js +38 -2
  35. package/dist/backend/backend/src/services/slack/slack.service.js.map +1 -1
  36. package/dist/backend/backend/src/types/slack.types.d.ts +10 -0
  37. package/dist/backend/backend/src/types/slack.types.d.ts.map +1 -1
  38. package/dist/backend/backend/src/types/slack.types.js.map +1 -1
  39. package/dist/backend/backend/src/utils/incomplete-turn.utils.d.ts +1 -1
  40. package/dist/backend/backend/src/utils/incomplete-turn.utils.d.ts.map +1 -1
  41. package/dist/backend/backend/src/utils/incomplete-turn.utils.js +4 -0
  42. package/dist/backend/backend/src/utils/incomplete-turn.utils.js.map +1 -1
  43. package/dist/backend/build-info.json +2 -2
  44. package/dist/cli/backend/src/services/slack/slack-orchestrator-bridge.d.ts +18 -0
  45. package/dist/cli/backend/src/services/slack/slack-orchestrator-bridge.d.ts.map +1 -1
  46. package/dist/cli/backend/src/services/slack/slack-orchestrator-bridge.js +33 -6
  47. package/dist/cli/backend/src/services/slack/slack-orchestrator-bridge.js.map +1 -1
  48. package/dist/cli/backend/src/services/slack/slack-team-channel.service.d.ts +24 -0
  49. package/dist/cli/backend/src/services/slack/slack-team-channel.service.d.ts.map +1 -1
  50. package/dist/cli/backend/src/services/slack/slack-team-channel.service.js +60 -1
  51. package/dist/cli/backend/src/services/slack/slack-team-channel.service.js.map +1 -1
  52. package/dist/cli/backend/src/services/slack/slack.service.d.ts +18 -1
  53. package/dist/cli/backend/src/services/slack/slack.service.d.ts.map +1 -1
  54. package/dist/cli/backend/src/services/slack/slack.service.js +38 -2
  55. package/dist/cli/backend/src/services/slack/slack.service.js.map +1 -1
  56. package/dist/cli/backend/src/types/slack.types.d.ts +10 -0
  57. package/dist/cli/backend/src/types/slack.types.d.ts.map +1 -1
  58. package/dist/cli/backend/src/types/slack.types.js.map +1 -1
  59. package/dist/cli/backend/src/utils/incomplete-turn.utils.d.ts +1 -1
  60. package/dist/cli/backend/src/utils/incomplete-turn.utils.d.ts.map +1 -1
  61. package/dist/cli/backend/src/utils/incomplete-turn.utils.js +4 -0
  62. package/dist/cli/backend/src/utils/incomplete-turn.utils.js.map +1 -1
  63. package/package.json +1 -1
  64. package/packages/crewly-agent/src/eval/desktop/desktop-tasks.test.ts +96 -0
  65. package/packages/crewly-agent/src/eval/desktop/desktop-tasks.ts +226 -0
  66. package/packages/crewly-agent/src/runtime/agent-runner.service.ts +20 -1
  67. package/packages/crewly-agent/src/runtime/computer.tool.test.ts +219 -0
  68. package/packages/crewly-agent/src/runtime/computer.tool.ts +405 -0
  69. package/packages/crewly-agent/src/runtime/desktop-checkpoint.test.ts +136 -0
  70. package/packages/crewly-agent/src/runtime/desktop-checkpoint.ts +231 -0
  71. package/packages/crewly-agent/src/runtime/desktop-recovery.test.ts +100 -0
  72. package/packages/crewly-agent/src/runtime/desktop-recovery.ts +195 -0
  73. package/packages/crewly-agent/src/runtime/desktop-task-runtime.test.ts +251 -0
  74. package/packages/crewly-agent/src/runtime/desktop-task-runtime.ts +423 -0
  75. package/packages/crewly-agent/src/runtime/desktop-task.tool.test.ts +218 -0
  76. package/packages/crewly-agent/src/runtime/desktop-task.tool.ts +343 -0
  77. package/packages/crewly-agent/src/runtime/tool-registry.test.ts +17 -0
  78. package/packages/crewly-agent/src/runtime/tool-registry.ts +54 -0
  79. package/packages/crewly-agent/src/runtime/types.ts +10 -1
  80. package/config/skills/agent/vnc-browser/SKILL.md +0 -140
@@ -0,0 +1,405 @@
1
+ /**
2
+ * `computer` tool — desktop control for the in-process runtime.
3
+ *
4
+ * Phase 3 of docs/research/computer-use-capability-assessment.md. The
5
+ * computer-use skill could already drive the desktop, but an in-process agent
6
+ * reached it the long way round: `bash_exec` the script, read the JSON, then
7
+ * `read_file` the screenshot it wrote. Two tool calls per step, and a weak
8
+ * model had to remember a shell invocation and a file path to get one look at
9
+ * the screen.
10
+ *
11
+ * This is one call that returns the result *and* the new screenshot, which is
12
+ * what every published computer-use agent expects.
13
+ *
14
+ * Two deliberate choices:
15
+ *
16
+ * The action names and the coordinate convention match Anthropic's
17
+ * `computer_20250124` tool. Claude-family models have seen that shape in
18
+ * training and need no instruction; for every other model there is a large
19
+ * body of public examples to imitate. Inventing our own names would cost
20
+ * accuracy for nothing. Crewly's element-level actions (`snapshot`,
21
+ * `click_ref`, `fill_ref`, `wait_for`) are added alongside — they have no
22
+ * equivalent in that spec and are the ones a weak model should reach for
23
+ * first, because naming `@e12` cannot miss the way a coordinate can.
24
+ *
25
+ * Nothing here talks to the mouse. Every action shells out to the
26
+ * computer-use skill, so the safety rails — permissions, stop switch, desktop
27
+ * lock, destructive-key and password-field refusals, the audit log — apply
28
+ * exactly once, in one place, to every runtime. A second implementation here
29
+ * would be a second thing to keep in step, and the rails are the part that
30
+ * must not drift.
31
+ *
32
+ * @module runtime/computer.tool
33
+ */
34
+
35
+ import { spawn } from 'child_process';
36
+ import { promises as fs } from 'fs';
37
+ import * as os from 'os';
38
+ import * as path from 'path';
39
+ import { z } from 'zod';
40
+ import type { ToolDefinition } from './types.js';
41
+
42
+ /**
43
+ * Width every screenshot is scaled to before the model sees it.
44
+ *
45
+ * Anthropic's guidance, and the reason is worth keeping in mind: accuracy
46
+ * falls off above roughly this width because the image is downsampled before
47
+ * the model ever sees it, and a model reasoning in the original coordinate
48
+ * space then points at the wrong place. The model works in scaled
49
+ * coordinates; this tool converts them back.
50
+ */
51
+ const TARGET_WIDTH = 1280;
52
+
53
+ /** Actions that move or type, and so return a fresh screenshot afterwards. */
54
+ const MUTATING = new Set([
55
+ 'left_click', 'right_click', 'middle_click', 'double_click', 'triple_click',
56
+ 'left_click_drag', 'mouse_move', 'key', 'type', 'scroll', 'click_ref', 'fill_ref',
57
+ ]);
58
+
59
+ /** How long an action may take before the tool gives up on the skill. */
60
+ const DEFAULT_TIMEOUT_MS = 60_000;
61
+
62
+ /** Injectable IO, for tests. */
63
+ export interface ComputerToolDeps {
64
+ /** Run the skill and return its stdout. */
65
+ runSkill?: (input: Record<string, unknown>) => Promise<string>;
66
+ /** Read a screenshot file as base64. */
67
+ readImage?: (file: string) => Promise<{ data: string; bytes: number }>;
68
+ /** Crewly install directory, holding config/skills. */
69
+ installDir?: string;
70
+ }
71
+
72
+ /** One screen, as the skill reports it. */
73
+ interface DisplayInfo {
74
+ frame: [number, number, number, number];
75
+ scale: number;
76
+ main: boolean;
77
+ }
78
+
79
+ /**
80
+ * Run the computer-use skill with a JSON payload.
81
+ *
82
+ * Failures come back as the skill's own JSON where possible: its refusals
83
+ * (`permission_required`, `screen_locked`, `destructive_blocked`…) already
84
+ * say what to do about them, and rewording them here would only blur that.
85
+ *
86
+ * @param input - The skill's JSON input
87
+ * @param deps - Injected IO
88
+ * @returns Parsed skill output
89
+ */
90
+ async function runSkill(input: Record<string, unknown>, deps: ComputerToolDeps): Promise<Record<string, unknown>> {
91
+ if (deps.runSkill) {
92
+ const raw = await deps.runSkill(input);
93
+ return parseSkillOutput(raw);
94
+ }
95
+
96
+ const installDir = deps.installDir ?? process.env['CREWLY_INSTALL_DIR'] ?? process.cwd();
97
+ const script = path.join(installDir, 'config', 'skills', 'agent', 'computer-use', 'execute.sh');
98
+
99
+ return new Promise((resolve) => {
100
+ const child = spawn('bash', [script, JSON.stringify(input)], {
101
+ env: { ...process.env },
102
+ stdio: ['ignore', 'pipe', 'pipe'],
103
+ });
104
+ let stdout = '';
105
+ let stderr = '';
106
+ const timer = setTimeout(() => {
107
+ child.kill('SIGKILL');
108
+ resolve({
109
+ success: false,
110
+ reason: 'timeout',
111
+ message: `The desktop action did not finish within ${DEFAULT_TIMEOUT_MS / 1000}s.`,
112
+ });
113
+ }, DEFAULT_TIMEOUT_MS);
114
+
115
+ child.stdout.on('data', (chunk) => { stdout += String(chunk); });
116
+ child.stderr.on('data', (chunk) => { stderr += String(chunk); });
117
+ child.on('error', (err) => {
118
+ clearTimeout(timer);
119
+ resolve({ success: false, reason: 'skill_unavailable', message: err.message, script });
120
+ });
121
+ child.on('close', () => {
122
+ clearTimeout(timer);
123
+ resolve(parseSkillOutput(stdout || stderr));
124
+ });
125
+ });
126
+ }
127
+
128
+ /**
129
+ * Parse the skill's stdout.
130
+ *
131
+ * The skill prints one JSON object, but a shell warning can precede it (the
132
+ * shared runner warns when CREWLY_SESSION_NAME is unset), so the last
133
+ * JSON-looking line wins.
134
+ *
135
+ * @param raw - Captured output
136
+ * @returns The parsed object, or a described failure when there is none
137
+ */
138
+ export function parseSkillOutput(raw: string): Record<string, unknown> {
139
+ const lines = (raw ?? '').trim().split('\n').filter((l) => l.trim());
140
+ for (let i = lines.length - 1; i >= 0; i--) {
141
+ const line = lines[i]!.trim();
142
+ if (!line.startsWith('{')) continue;
143
+ try {
144
+ return JSON.parse(line) as Record<string, unknown>;
145
+ } catch {
146
+ // Not the JSON line after all — keep looking backwards.
147
+ }
148
+ }
149
+ // Multi-line pretty-printed JSON (jq's default) is one object across lines.
150
+ const joined = lines.join('\n');
151
+ const start = joined.indexOf('{');
152
+ if (start >= 0) {
153
+ try {
154
+ return JSON.parse(joined.slice(start)) as Record<string, unknown>;
155
+ } catch {
156
+ // Fall through to the described failure.
157
+ }
158
+ }
159
+ return { success: false, reason: 'unparsable', message: raw?.slice(0, 500) || 'The skill produced no output.' };
160
+ }
161
+
162
+ /**
163
+ * The scale between the coordinates the model uses and real screen points.
164
+ *
165
+ * @param displays - What the skill reported
166
+ * @returns Factor to multiply model coordinates by, and the scaled size
167
+ */
168
+ export function scaleFor(displays: DisplayInfo[]): { factor: number; width: number; height: number; screen: [number, number] } {
169
+ const main = displays.find((d) => d.main) ?? displays[0];
170
+ const [, , w, h] = main?.frame ?? [0, 0, TARGET_WIDTH, 800];
171
+ // Never scale up: a small screen is already easier to point at than a large
172
+ // one, and enlarging it would invent precision the model does not have.
173
+ const factor = w > TARGET_WIDTH ? w / TARGET_WIDTH : 1;
174
+ return {
175
+ factor,
176
+ width: Math.round(w / factor),
177
+ height: Math.round(h / factor),
178
+ screen: [w, h],
179
+ };
180
+ }
181
+
182
+ /**
183
+ * Convert a model coordinate into a screen point.
184
+ *
185
+ * @param value - Coordinate in the scaled space the model sees
186
+ * @param factor - From {@link scaleFor}
187
+ * @returns Screen point
188
+ */
189
+ export function toScreen(value: number, factor: number): number {
190
+ return Math.round(value * factor);
191
+ }
192
+
193
+ /**
194
+ * Take a screenshot scaled for the model.
195
+ *
196
+ * @param deps - Injected IO
197
+ * @returns Image payload, or null when the screenshot failed
198
+ */
199
+ async function capture(
200
+ deps: ComputerToolDeps,
201
+ ): Promise<{ data: string; bytes: number; file: string } | null> {
202
+ const file = path.join(os.tmpdir(), `crewly-computer-${process.pid}-${Date.now()}.png`);
203
+ // maxWidth is what makes the screenshot match the coordinate space the
204
+ // model is told to use; without it the two drift and every click is off.
205
+ const shot = await runSkill(
206
+ { action: 'screenshot', output: file, maxWidth: Math.round(TARGET_WIDTH) },
207
+ deps,
208
+ );
209
+ const written = (shot['path'] as string) ?? file;
210
+ try {
211
+ if (deps.readImage) {
212
+ const { data, bytes } = await deps.readImage(written);
213
+ return { data, bytes, file: written };
214
+ }
215
+ const buffer = await fs.readFile(written);
216
+ await fs.unlink(written).catch(() => undefined);
217
+ return { data: buffer.toString('base64'), bytes: buffer.length, file: written };
218
+ } catch {
219
+ return null;
220
+ }
221
+ }
222
+
223
+ /** Arguments the model may send. */
224
+ const computerSchema = z.object({
225
+ action: z.enum([
226
+ // Anthropic computer_20250124 vocabulary.
227
+ 'screenshot', 'left_click', 'right_click', 'middle_click', 'double_click',
228
+ 'triple_click', 'left_click_drag', 'mouse_move', 'key', 'type', 'scroll',
229
+ 'wait', 'cursor_position',
230
+ // Crewly's element-level additions.
231
+ 'snapshot', 'click_ref', 'fill_ref', 'wait_for', 'ocr', 'displays',
232
+ ]).describe('What to do. Prefer snapshot + click_ref/fill_ref over coordinates when the app exposes elements.'),
233
+ coordinate: z.array(z.number()).length(2).optional()
234
+ .describe('[x, y] in the screenshot you were shown, not screen pixels. The tool converts.'),
235
+ start_coordinate: z.array(z.number()).length(2).optional()
236
+ .describe('[x, y] to drag from, for left_click_drag.'),
237
+ text: z.string().optional()
238
+ .describe('Text to type, the key combo for `key` (e.g. "command+s"), or the value for fill_ref.'),
239
+ ref: z.string().optional().describe('Element reference from a snapshot, e.g. "@e12".'),
240
+ app: z.string().optional().describe('Application to snapshot or wait for; defaults to the frontmost.'),
241
+ scroll_direction: z.enum(['up', 'down', 'left', 'right']).optional(),
242
+ scroll_amount: z.number().optional().describe('Scroll clicks; defaults to 3.'),
243
+ duration: z.number().optional().describe('Seconds to wait, for `wait`.'),
244
+ });
245
+
246
+ /**
247
+ * Translate the tool's arguments into the skill's own input.
248
+ *
249
+ * @param args - Validated tool arguments
250
+ * @param factor - Coordinate scale
251
+ * @returns The skill payload, or a refusal when the arguments do not fit
252
+ */
253
+ export function toSkillInput(
254
+ args: z.infer<typeof computerSchema>,
255
+ factor: number,
256
+ ): Record<string, unknown> | { error: string } {
257
+ const point = (pair?: number[]) =>
258
+ pair ? { x: toScreen(pair[0]!, factor), y: toScreen(pair[1]!, factor) } : null;
259
+
260
+ switch (args.action) {
261
+ case 'screenshot':
262
+ return { action: 'screenshot' };
263
+ case 'displays':
264
+ return { action: 'displays' };
265
+ case 'cursor_position':
266
+ // The skill has no cursor read; a screenshot answers the same question
267
+ // and is what the model will ask for next anyway.
268
+ return { action: 'screenshot' };
269
+
270
+ case 'left_click':
271
+ case 'right_click':
272
+ case 'double_click': {
273
+ const p = point(args.coordinate);
274
+ if (!p) return { error: `${args.action} needs a coordinate.` };
275
+ const button = args.action === 'right_click' ? 'right' : args.action === 'double_click' ? 'double' : 'left';
276
+ return { action: 'click', ...p, button };
277
+ }
278
+ case 'middle_click':
279
+ case 'triple_click': {
280
+ // Neither exists in the skill. Saying so beats silently doing something
281
+ // else: a model told "not supported" picks another route, one told
282
+ // "done" builds on a click that never happened.
283
+ return { error: `${args.action} is not supported on this platform. Use left_click, or select the text another way.` };
284
+ }
285
+ case 'mouse_move': {
286
+ const p = point(args.coordinate);
287
+ if (!p) return { error: 'mouse_move needs a coordinate.' };
288
+ return { action: 'move', ...p };
289
+ }
290
+ case 'left_click_drag': {
291
+ const from = point(args.start_coordinate);
292
+ const to = point(args.coordinate);
293
+ if (!from || !to) return { error: 'left_click_drag needs start_coordinate and coordinate.' };
294
+ return { action: 'drag', fromX: from.x, fromY: from.y, toX: to.x, toY: to.y };
295
+ }
296
+ case 'key':
297
+ if (!args.text) return { error: 'key needs `text`, e.g. "command+s".' };
298
+ return { action: 'key', key: args.text };
299
+ case 'type':
300
+ if (!args.text) return { error: 'type needs `text`.' };
301
+ return { action: 'type', text: args.text };
302
+ case 'scroll': {
303
+ const p = point(args.coordinate);
304
+ return {
305
+ action: 'scroll',
306
+ ...(p ?? {}),
307
+ direction: args.scroll_direction ?? 'down',
308
+ amount: args.scroll_amount ?? 3,
309
+ };
310
+ }
311
+
312
+ case 'snapshot':
313
+ return { action: 'snapshot', ...(args.app ? { app: args.app } : {}) };
314
+ case 'click_ref':
315
+ if (!args.ref) return { error: 'click_ref needs `ref`, e.g. "@e12" from a snapshot.' };
316
+ return { action: 'click-ref', ref: args.ref };
317
+ case 'fill_ref':
318
+ if (!args.ref || args.text === undefined) return { error: 'fill_ref needs `ref` and `text`.' };
319
+ return { action: 'fill-ref', ref: args.ref, text: args.text };
320
+ case 'ocr':
321
+ return { action: 'ocr' };
322
+ case 'wait_for':
323
+ if (!args.app && !args.ref && !args.text) {
324
+ return { error: 'wait_for needs one of `app`, `ref` or `text`.' };
325
+ }
326
+ return {
327
+ action: 'wait-for',
328
+ ...(args.app ? { app: args.app } : {}),
329
+ ...(args.ref ? { ref: args.ref } : {}),
330
+ ...(args.text ? { text: args.text } : {}),
331
+ };
332
+ case 'wait':
333
+ return { action: 'wait-for', idle: true, timeoutMs: Math.round((args.duration ?? 1) * 1000) };
334
+ }
335
+ }
336
+
337
+ /**
338
+ * Build the `computer` tool.
339
+ *
340
+ * @param deps - Injected IO for tests
341
+ * @returns The tool definition
342
+ *
343
+ * @example
344
+ * ```ts
345
+ * const tools = { computer: createComputerTool() };
346
+ * ```
347
+ */
348
+ export function createComputerTool(deps: ComputerToolDeps = {}): ToolDefinition {
349
+ return {
350
+ description:
351
+ 'Control this Mac: look at the screen and act on it. ' +
352
+ 'Prefer `snapshot` then `click_ref`/`fill_ref` — naming an element cannot miss the way a coordinate can, ' +
353
+ 'and it keeps working when the window moves. Fall back to coordinates for canvases and custom-drawn UI. ' +
354
+ 'Coordinates are in the screenshot you were shown, not screen pixels. ' +
355
+ 'Destructive key combos, password fields and credential apps are refused, and the owner can stop everything at any time.',
356
+ inputSchema: computerSchema,
357
+ sensitivity: 'destructive',
358
+ execute: async (rawArgs) => {
359
+ const args = rawArgs as z.infer<typeof computerSchema>;
360
+
361
+ // The scale has to come from the live display: the owner may have
362
+ // changed resolution or moved to another screen since the last call.
363
+ const displayResult = await runSkill({ action: 'displays' }, deps);
364
+ const displays = (displayResult['displays'] as DisplayInfo[] | undefined) ?? [];
365
+ const scale = scaleFor(displays);
366
+
367
+ const payload = toSkillInput(args, scale.factor);
368
+ if ('error' in payload) {
369
+ return { success: false, reason: 'bad_arguments', message: payload.error };
370
+ }
371
+
372
+ // A bare screenshot is taken once, by the capture step below. Running
373
+ // the skill's screenshot here as well would shoot the screen twice for
374
+ // one request — slow, and the two images could even differ.
375
+ const result = args.action === 'screenshot'
376
+ ? { success: true }
377
+ : await runSkill(payload, deps);
378
+
379
+ // A refusal is returned as it stands. The rails phrase their own
380
+ // reasons and tell the agent what to do; a screenshot alongside would
381
+ // just be the same screen it could not act on.
382
+ if (result['success'] === false) return result;
383
+
384
+ const out: Record<string, unknown> = {
385
+ ...result,
386
+ action: args.action,
387
+ screen: { width: scale.width, height: scale.height, actual: scale.screen },
388
+ };
389
+
390
+ // One call, one look. Showing the result of an action is what lets a
391
+ // model check its own work instead of assuming the click landed.
392
+ if (MUTATING.has(args.action) || args.action === 'screenshot') {
393
+ const image = await capture(deps);
394
+ if (image) {
395
+ out['type'] = 'image';
396
+ out['mimeType'] = 'image/png';
397
+ out['data'] = image.data;
398
+ out['sizeBytes'] = image.bytes;
399
+ out['note'] = `Screenshot is ${scale.width}×${scale.height}; give coordinates in that space.`;
400
+ }
401
+ }
402
+ return out;
403
+ },
404
+ };
405
+ }
@@ -0,0 +1,136 @@
1
+ /**
2
+ * Tests for checkpoint evaluation.
3
+ *
4
+ * A checkpoint is the only thing that can call a desktop subgoal done, so a
5
+ * checkpoint that passes when it should not is worse than having none: it
6
+ * launders the agent's claim into a verified fact.
7
+ */
8
+
9
+ import { describe, it, expect } from 'vitest';
10
+ import { evaluateCheckpoint, describeCheckpoint, type Checkpoint } from './desktop-checkpoint.js';
11
+
12
+ /** IO over a fake filesystem. */
13
+ function fs(files: Record<string, string>) {
14
+ return {
15
+ readFile: async (p: string) => {
16
+ if (!(p in files)) throw new Error('ENOENT');
17
+ return files[p]!;
18
+ },
19
+ statFile: async (p: string) => {
20
+ if (!(p in files)) throw new Error('ENOENT');
21
+ return { size: files[p]!.length };
22
+ },
23
+ };
24
+ }
25
+
26
+ describe('file checkpoints', () => {
27
+ it('fails an empty file — the signature of a save that never completed', async () => {
28
+ const out = await evaluateCheckpoint({ kind: 'file-exists', path: '/a' }, fs({ '/a': '' }));
29
+ expect(out.passed).toBe(false);
30
+ expect(out.reason).toMatch(/empty/);
31
+ });
32
+
33
+ it('allows an empty file when the task said so', async () => {
34
+ expect((await evaluateCheckpoint({ kind: 'file-exists', path: '/a', allowEmpty: true }, fs({ '/a': '' }))).passed).toBe(true);
35
+ });
36
+
37
+ it('says the file is missing rather than just failing', async () => {
38
+ const out = await evaluateCheckpoint({ kind: 'file-exists', path: '/nope' }, fs({}));
39
+ expect(out.reason).toContain('/nope');
40
+ });
41
+
42
+ it('shows what the file does say when the content is wrong', async () => {
43
+ const out = await evaluateCheckpoint({ kind: 'file-contains', path: '/a', text: 'hello' }, fs({ '/a': 'goodbye' }));
44
+ expect(out.passed).toBe(false);
45
+ expect(out.observed).toBe('goodbye');
46
+ });
47
+
48
+ it('matches a pattern', async () => {
49
+ expect((await evaluateCheckpoint({ kind: 'file-matches', path: '/a', pattern: '^\\d+$' }, fs({ '/a': '42' }))).passed).toBe(true);
50
+ expect((await evaluateCheckpoint({ kind: 'file-matches', path: '/a', pattern: '^\\d+$' }, fs({ '/a': 'x' }))).passed).toBe(false);
51
+ });
52
+
53
+ it('checks the other half of a rename', async () => {
54
+ expect((await evaluateCheckpoint({ kind: 'file-absent', path: '/old' }, fs({}))).passed).toBe(true);
55
+ expect((await evaluateCheckpoint({ kind: 'file-absent', path: '/old' }, fs({ '/old': 'x' }))).passed).toBe(false);
56
+ });
57
+ });
58
+
59
+ describe('screen checkpoints', () => {
60
+ const snapshot = async () => [
61
+ { role: 'AXButton', name: 'Save' },
62
+ { role: 'AXStaticText', name: 'Untitled document' },
63
+ ];
64
+
65
+ it('finds an element by name, and by role when given one', async () => {
66
+ expect((await evaluateCheckpoint({ kind: 'element-present', name: 'Save' }, { snapshot })).passed).toBe(true);
67
+ expect((await evaluateCheckpoint({ kind: 'element-present', name: 'Save', role: 'AXButton' }, { snapshot })).passed).toBe(true);
68
+ expect((await evaluateCheckpoint({ kind: 'element-present', name: 'Save', role: 'AXMenuItem' }, { snapshot })).passed).toBe(false);
69
+ });
70
+
71
+ it('checks a dialog is gone, which is how "closed it" is proved', async () => {
72
+ expect((await evaluateCheckpoint({ kind: 'element-absent', name: 'Save' }, { snapshot })).passed).toBe(false);
73
+ expect((await evaluateCheckpoint({ kind: 'element-absent', name: 'Nothing' }, { snapshot })).passed).toBe(true);
74
+ });
75
+
76
+ it('reads text off the screen for apps with no element tree', async () => {
77
+ const screenText = async () => ['CREWLY EVAL 7734'];
78
+ expect((await evaluateCheckpoint({ kind: 'text-on-screen', text: 'eval 7734' }, { screenText })).passed).toBe(true);
79
+ expect((await evaluateCheckpoint({ kind: 'text-on-screen', text: 'absent' }, { screenText })).passed).toBe(false);
80
+ });
81
+
82
+ it('compares the frontmost app case-insensitively and reports what is actually there', async () => {
83
+ const frontmostApp = async () => 'TextEdit';
84
+ expect((await evaluateCheckpoint({ kind: 'app-frontmost', app: 'textedit' }, { frontmostApp })).passed).toBe(true);
85
+ const out = await evaluateCheckpoint({ kind: 'app-frontmost', app: 'Numbers' }, { frontmostApp });
86
+ expect(out.observed).toBe('TextEdit');
87
+ });
88
+
89
+ it('fails rather than pretends when it cannot see', async () => {
90
+ // No snapshot dependency wired: it must not quietly pass.
91
+ expect((await evaluateCheckpoint({ kind: 'element-present', name: 'Save' }, {})).passed).toBe(false);
92
+ expect((await evaluateCheckpoint({ kind: 'text-on-screen', text: 'x' }, {})).passed).toBe(false);
93
+ expect((await evaluateCheckpoint({ kind: 'app-frontmost', app: 'x' }, {})).passed).toBe(false);
94
+ });
95
+ });
96
+
97
+ describe('shell checkpoints', () => {
98
+ it('passes on exit zero and reports the output on failure', async () => {
99
+ const runShell = async (c: string) =>
100
+ c === 'true' ? { code: 0, stdout: '', stderr: '' } : { code: 1, stdout: '', stderr: 'boom' };
101
+ expect((await evaluateCheckpoint({ kind: 'shell', command: 'true' }, { runShell })).passed).toBe(true);
102
+ const out = await evaluateCheckpoint({ kind: 'shell', command: 'false' }, { runShell });
103
+ expect(out.passed).toBe(false);
104
+ expect(out.observed).toBe('boom');
105
+ });
106
+ });
107
+
108
+ describe('robustness', () => {
109
+ it('turns a thrown check into a failed one, never an exception', async () => {
110
+ const out = await evaluateCheckpoint({ kind: 'file-exists', path: '/a' }, {
111
+ statFile: async () => { throw new Error('disk on fire'); },
112
+ });
113
+ // A broken check must not read as a broken task, but it must not pass.
114
+ expect(out.passed).toBe(false);
115
+ });
116
+ });
117
+
118
+ describe('describeCheckpoint', () => {
119
+ it('reads as a sentence for every kind', () => {
120
+ const all: Checkpoint[] = [
121
+ { kind: 'file-exists', path: '/a' },
122
+ { kind: 'file-contains', path: '/a', text: 'x' },
123
+ { kind: 'file-matches', path: '/a', pattern: 'x' },
124
+ { kind: 'file-absent', path: '/a' },
125
+ { kind: 'app-frontmost', app: 'Finder' },
126
+ { kind: 'element-present', name: 'Save' },
127
+ { kind: 'element-absent', name: 'Save' },
128
+ { kind: 'text-on-screen', text: 'hi' },
129
+ { kind: 'shell', command: 'true' },
130
+ ];
131
+ for (const cp of all) {
132
+ expect(describeCheckpoint(cp).length).toBeGreaterThan(5);
133
+ }
134
+ expect(describeCheckpoint({ kind: 'shell', command: 'x', description: 'the build passes' })).toBe('the build passes');
135
+ });
136
+ });