crewly 1.20.40 → 1.20.47
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/config/skills/_common/desktop-guards.sh +485 -0
- package/config/skills/_common/desktop-guards.test.sh +242 -0
- package/config/skills/_common/desktop-perceive.swift +530 -0
- package/config/skills/_common/desktop-presence.swift +343 -0
- package/config/skills/agent/_common/desktop-guards.sh +4 -0
- package/config/skills/agent/computer-use/SKILL.md +88 -0
- package/config/skills/agent/computer-use/execute.sh +249 -3
- package/config/skills/agent/desktop-app-control/SKILL.md +19 -0
- package/config/skills/agent/remote-browser/SKILL.md +19 -0
- package/dist/backend/backend/src/controllers/desktop/desktop.controller.d.ts +105 -0
- package/dist/backend/backend/src/controllers/desktop/desktop.controller.d.ts.map +1 -0
- package/dist/backend/backend/src/controllers/desktop/desktop.controller.js +278 -0
- package/dist/backend/backend/src/controllers/desktop/desktop.controller.js.map +1 -0
- package/dist/backend/backend/src/controllers/desktop/desktop.routes.d.ts +21 -0
- package/dist/backend/backend/src/controllers/desktop/desktop.routes.d.ts.map +1 -0
- package/dist/backend/backend/src/controllers/desktop/desktop.routes.js +31 -0
- package/dist/backend/backend/src/controllers/desktop/desktop.routes.js.map +1 -0
- package/dist/backend/backend/src/routes/api.routes.d.ts.map +1 -1
- package/dist/backend/backend/src/routes/api.routes.js +3 -0
- package/dist/backend/backend/src/routes/api.routes.js.map +1 -1
- package/dist/backend/backend/src/services/cloud/mobile-api-relay.service.d.ts.map +1 -1
- package/dist/backend/backend/src/services/cloud/mobile-api-relay.service.js +9 -0
- package/dist/backend/backend/src/services/cloud/mobile-api-relay.service.js.map +1 -1
- package/dist/backend/backend/src/services/slack/slack-team-channel.service.d.ts +24 -0
- package/dist/backend/backend/src/services/slack/slack-team-channel.service.d.ts.map +1 -1
- package/dist/backend/backend/src/services/slack/slack-team-channel.service.js +60 -1
- package/dist/backend/backend/src/services/slack/slack-team-channel.service.js.map +1 -1
- package/dist/backend/backend/src/services/slack/slack.service.d.ts +17 -0
- package/dist/backend/backend/src/services/slack/slack.service.d.ts.map +1 -1
- package/dist/backend/backend/src/services/slack/slack.service.js +32 -0
- package/dist/backend/backend/src/services/slack/slack.service.js.map +1 -1
- package/dist/backend/backend/src/types/slack.types.d.ts +10 -0
- package/dist/backend/backend/src/types/slack.types.d.ts.map +1 -1
- package/dist/backend/backend/src/types/slack.types.js.map +1 -1
- package/dist/backend/backend/src/utils/incomplete-turn.utils.d.ts +1 -1
- package/dist/backend/backend/src/utils/incomplete-turn.utils.d.ts.map +1 -1
- package/dist/backend/backend/src/utils/incomplete-turn.utils.js +4 -0
- package/dist/backend/backend/src/utils/incomplete-turn.utils.js.map +1 -1
- package/dist/backend/build-info.json +2 -2
- package/dist/cli/backend/src/services/slack/slack-team-channel.service.d.ts +24 -0
- package/dist/cli/backend/src/services/slack/slack-team-channel.service.d.ts.map +1 -1
- package/dist/cli/backend/src/services/slack/slack-team-channel.service.js +60 -1
- package/dist/cli/backend/src/services/slack/slack-team-channel.service.js.map +1 -1
- package/dist/cli/backend/src/services/slack/slack.service.d.ts +17 -0
- package/dist/cli/backend/src/services/slack/slack.service.d.ts.map +1 -1
- package/dist/cli/backend/src/services/slack/slack.service.js +32 -0
- package/dist/cli/backend/src/services/slack/slack.service.js.map +1 -1
- package/dist/cli/backend/src/types/slack.types.d.ts +10 -0
- package/dist/cli/backend/src/types/slack.types.d.ts.map +1 -1
- package/dist/cli/backend/src/types/slack.types.js.map +1 -1
- package/dist/cli/backend/src/utils/incomplete-turn.utils.d.ts +1 -1
- package/dist/cli/backend/src/utils/incomplete-turn.utils.d.ts.map +1 -1
- package/dist/cli/backend/src/utils/incomplete-turn.utils.js +4 -0
- package/dist/cli/backend/src/utils/incomplete-turn.utils.js.map +1 -1
- package/package.json +1 -1
- package/packages/crewly-agent/src/eval/desktop/desktop-tasks.test.ts +96 -0
- package/packages/crewly-agent/src/eval/desktop/desktop-tasks.ts +226 -0
- package/packages/crewly-agent/src/runtime/agent-runner.service.ts +20 -1
- package/packages/crewly-agent/src/runtime/computer.tool.test.ts +219 -0
- package/packages/crewly-agent/src/runtime/computer.tool.ts +405 -0
- package/packages/crewly-agent/src/runtime/desktop-checkpoint.test.ts +136 -0
- package/packages/crewly-agent/src/runtime/desktop-checkpoint.ts +231 -0
- package/packages/crewly-agent/src/runtime/desktop-recovery.test.ts +100 -0
- package/packages/crewly-agent/src/runtime/desktop-recovery.ts +195 -0
- package/packages/crewly-agent/src/runtime/desktop-task-runtime.test.ts +251 -0
- package/packages/crewly-agent/src/runtime/desktop-task-runtime.ts +423 -0
- package/packages/crewly-agent/src/runtime/desktop-task.tool.test.ts +218 -0
- package/packages/crewly-agent/src/runtime/desktop-task.tool.ts +343 -0
- package/packages/crewly-agent/src/runtime/tool-registry.test.ts +17 -0
- package/packages/crewly-agent/src/runtime/tool-registry.ts +54 -0
- package/packages/crewly-agent/src/runtime/types.ts +10 -1
- package/config/skills/agent/vnc-browser/SKILL.md +0 -140
|
@@ -0,0 +1,405 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* `computer` tool — desktop control for the in-process runtime.
|
|
3
|
+
*
|
|
4
|
+
* Phase 3 of docs/research/computer-use-capability-assessment.md. The
|
|
5
|
+
* computer-use skill could already drive the desktop, but an in-process agent
|
|
6
|
+
* reached it the long way round: `bash_exec` the script, read the JSON, then
|
|
7
|
+
* `read_file` the screenshot it wrote. Two tool calls per step, and a weak
|
|
8
|
+
* model had to remember a shell invocation and a file path to get one look at
|
|
9
|
+
* the screen.
|
|
10
|
+
*
|
|
11
|
+
* This is one call that returns the result *and* the new screenshot, which is
|
|
12
|
+
* what every published computer-use agent expects.
|
|
13
|
+
*
|
|
14
|
+
* Two deliberate choices:
|
|
15
|
+
*
|
|
16
|
+
* The action names and the coordinate convention match Anthropic's
|
|
17
|
+
* `computer_20250124` tool. Claude-family models have seen that shape in
|
|
18
|
+
* training and need no instruction; for every other model there is a large
|
|
19
|
+
* body of public examples to imitate. Inventing our own names would cost
|
|
20
|
+
* accuracy for nothing. Crewly's element-level actions (`snapshot`,
|
|
21
|
+
* `click_ref`, `fill_ref`, `wait_for`) are added alongside — they have no
|
|
22
|
+
* equivalent in that spec and are the ones a weak model should reach for
|
|
23
|
+
* first, because naming `@e12` cannot miss the way a coordinate can.
|
|
24
|
+
*
|
|
25
|
+
* Nothing here talks to the mouse. Every action shells out to the
|
|
26
|
+
* computer-use skill, so the safety rails — permissions, stop switch, desktop
|
|
27
|
+
* lock, destructive-key and password-field refusals, the audit log — apply
|
|
28
|
+
* exactly once, in one place, to every runtime. A second implementation here
|
|
29
|
+
* would be a second thing to keep in step, and the rails are the part that
|
|
30
|
+
* must not drift.
|
|
31
|
+
*
|
|
32
|
+
* @module runtime/computer.tool
|
|
33
|
+
*/
|
|
34
|
+
|
|
35
|
+
import { spawn } from 'child_process';
|
|
36
|
+
import { promises as fs } from 'fs';
|
|
37
|
+
import * as os from 'os';
|
|
38
|
+
import * as path from 'path';
|
|
39
|
+
import { z } from 'zod';
|
|
40
|
+
import type { ToolDefinition } from './types.js';
|
|
41
|
+
|
|
42
|
+
/**
|
|
43
|
+
* Width every screenshot is scaled to before the model sees it.
|
|
44
|
+
*
|
|
45
|
+
* Anthropic's guidance, and the reason is worth keeping in mind: accuracy
|
|
46
|
+
* falls off above roughly this width because the image is downsampled before
|
|
47
|
+
* the model ever sees it, and a model reasoning in the original coordinate
|
|
48
|
+
* space then points at the wrong place. The model works in scaled
|
|
49
|
+
* coordinates; this tool converts them back.
|
|
50
|
+
*/
|
|
51
|
+
const TARGET_WIDTH = 1280;
|
|
52
|
+
|
|
53
|
+
/** Actions that move or type, and so return a fresh screenshot afterwards. */
|
|
54
|
+
const MUTATING = new Set([
|
|
55
|
+
'left_click', 'right_click', 'middle_click', 'double_click', 'triple_click',
|
|
56
|
+
'left_click_drag', 'mouse_move', 'key', 'type', 'scroll', 'click_ref', 'fill_ref',
|
|
57
|
+
]);
|
|
58
|
+
|
|
59
|
+
/** How long an action may take before the tool gives up on the skill. */
|
|
60
|
+
const DEFAULT_TIMEOUT_MS = 60_000;
|
|
61
|
+
|
|
62
|
+
/** Injectable IO, for tests. */
|
|
63
|
+
export interface ComputerToolDeps {
|
|
64
|
+
/** Run the skill and return its stdout. */
|
|
65
|
+
runSkill?: (input: Record<string, unknown>) => Promise<string>;
|
|
66
|
+
/** Read a screenshot file as base64. */
|
|
67
|
+
readImage?: (file: string) => Promise<{ data: string; bytes: number }>;
|
|
68
|
+
/** Crewly install directory, holding config/skills. */
|
|
69
|
+
installDir?: string;
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
/** One screen, as the skill reports it. */
|
|
73
|
+
interface DisplayInfo {
|
|
74
|
+
frame: [number, number, number, number];
|
|
75
|
+
scale: number;
|
|
76
|
+
main: boolean;
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
/**
|
|
80
|
+
* Run the computer-use skill with a JSON payload.
|
|
81
|
+
*
|
|
82
|
+
* Failures come back as the skill's own JSON where possible: its refusals
|
|
83
|
+
* (`permission_required`, `screen_locked`, `destructive_blocked`…) already
|
|
84
|
+
* say what to do about them, and rewording them here would only blur that.
|
|
85
|
+
*
|
|
86
|
+
* @param input - The skill's JSON input
|
|
87
|
+
* @param deps - Injected IO
|
|
88
|
+
* @returns Parsed skill output
|
|
89
|
+
*/
|
|
90
|
+
async function runSkill(input: Record<string, unknown>, deps: ComputerToolDeps): Promise<Record<string, unknown>> {
|
|
91
|
+
if (deps.runSkill) {
|
|
92
|
+
const raw = await deps.runSkill(input);
|
|
93
|
+
return parseSkillOutput(raw);
|
|
94
|
+
}
|
|
95
|
+
|
|
96
|
+
const installDir = deps.installDir ?? process.env['CREWLY_INSTALL_DIR'] ?? process.cwd();
|
|
97
|
+
const script = path.join(installDir, 'config', 'skills', 'agent', 'computer-use', 'execute.sh');
|
|
98
|
+
|
|
99
|
+
return new Promise((resolve) => {
|
|
100
|
+
const child = spawn('bash', [script, JSON.stringify(input)], {
|
|
101
|
+
env: { ...process.env },
|
|
102
|
+
stdio: ['ignore', 'pipe', 'pipe'],
|
|
103
|
+
});
|
|
104
|
+
let stdout = '';
|
|
105
|
+
let stderr = '';
|
|
106
|
+
const timer = setTimeout(() => {
|
|
107
|
+
child.kill('SIGKILL');
|
|
108
|
+
resolve({
|
|
109
|
+
success: false,
|
|
110
|
+
reason: 'timeout',
|
|
111
|
+
message: `The desktop action did not finish within ${DEFAULT_TIMEOUT_MS / 1000}s.`,
|
|
112
|
+
});
|
|
113
|
+
}, DEFAULT_TIMEOUT_MS);
|
|
114
|
+
|
|
115
|
+
child.stdout.on('data', (chunk) => { stdout += String(chunk); });
|
|
116
|
+
child.stderr.on('data', (chunk) => { stderr += String(chunk); });
|
|
117
|
+
child.on('error', (err) => {
|
|
118
|
+
clearTimeout(timer);
|
|
119
|
+
resolve({ success: false, reason: 'skill_unavailable', message: err.message, script });
|
|
120
|
+
});
|
|
121
|
+
child.on('close', () => {
|
|
122
|
+
clearTimeout(timer);
|
|
123
|
+
resolve(parseSkillOutput(stdout || stderr));
|
|
124
|
+
});
|
|
125
|
+
});
|
|
126
|
+
}
|
|
127
|
+
|
|
128
|
+
/**
|
|
129
|
+
* Parse the skill's stdout.
|
|
130
|
+
*
|
|
131
|
+
* The skill prints one JSON object, but a shell warning can precede it (the
|
|
132
|
+
* shared runner warns when CREWLY_SESSION_NAME is unset), so the last
|
|
133
|
+
* JSON-looking line wins.
|
|
134
|
+
*
|
|
135
|
+
* @param raw - Captured output
|
|
136
|
+
* @returns The parsed object, or a described failure when there is none
|
|
137
|
+
*/
|
|
138
|
+
export function parseSkillOutput(raw: string): Record<string, unknown> {
|
|
139
|
+
const lines = (raw ?? '').trim().split('\n').filter((l) => l.trim());
|
|
140
|
+
for (let i = lines.length - 1; i >= 0; i--) {
|
|
141
|
+
const line = lines[i]!.trim();
|
|
142
|
+
if (!line.startsWith('{')) continue;
|
|
143
|
+
try {
|
|
144
|
+
return JSON.parse(line) as Record<string, unknown>;
|
|
145
|
+
} catch {
|
|
146
|
+
// Not the JSON line after all — keep looking backwards.
|
|
147
|
+
}
|
|
148
|
+
}
|
|
149
|
+
// Multi-line pretty-printed JSON (jq's default) is one object across lines.
|
|
150
|
+
const joined = lines.join('\n');
|
|
151
|
+
const start = joined.indexOf('{');
|
|
152
|
+
if (start >= 0) {
|
|
153
|
+
try {
|
|
154
|
+
return JSON.parse(joined.slice(start)) as Record<string, unknown>;
|
|
155
|
+
} catch {
|
|
156
|
+
// Fall through to the described failure.
|
|
157
|
+
}
|
|
158
|
+
}
|
|
159
|
+
return { success: false, reason: 'unparsable', message: raw?.slice(0, 500) || 'The skill produced no output.' };
|
|
160
|
+
}
|
|
161
|
+
|
|
162
|
+
/**
|
|
163
|
+
* The scale between the coordinates the model uses and real screen points.
|
|
164
|
+
*
|
|
165
|
+
* @param displays - What the skill reported
|
|
166
|
+
* @returns Factor to multiply model coordinates by, and the scaled size
|
|
167
|
+
*/
|
|
168
|
+
export function scaleFor(displays: DisplayInfo[]): { factor: number; width: number; height: number; screen: [number, number] } {
|
|
169
|
+
const main = displays.find((d) => d.main) ?? displays[0];
|
|
170
|
+
const [, , w, h] = main?.frame ?? [0, 0, TARGET_WIDTH, 800];
|
|
171
|
+
// Never scale up: a small screen is already easier to point at than a large
|
|
172
|
+
// one, and enlarging it would invent precision the model does not have.
|
|
173
|
+
const factor = w > TARGET_WIDTH ? w / TARGET_WIDTH : 1;
|
|
174
|
+
return {
|
|
175
|
+
factor,
|
|
176
|
+
width: Math.round(w / factor),
|
|
177
|
+
height: Math.round(h / factor),
|
|
178
|
+
screen: [w, h],
|
|
179
|
+
};
|
|
180
|
+
}
|
|
181
|
+
|
|
182
|
+
/**
|
|
183
|
+
* Convert a model coordinate into a screen point.
|
|
184
|
+
*
|
|
185
|
+
* @param value - Coordinate in the scaled space the model sees
|
|
186
|
+
* @param factor - From {@link scaleFor}
|
|
187
|
+
* @returns Screen point
|
|
188
|
+
*/
|
|
189
|
+
export function toScreen(value: number, factor: number): number {
|
|
190
|
+
return Math.round(value * factor);
|
|
191
|
+
}
|
|
192
|
+
|
|
193
|
+
/**
|
|
194
|
+
* Take a screenshot scaled for the model.
|
|
195
|
+
*
|
|
196
|
+
* @param deps - Injected IO
|
|
197
|
+
* @returns Image payload, or null when the screenshot failed
|
|
198
|
+
*/
|
|
199
|
+
async function capture(
|
|
200
|
+
deps: ComputerToolDeps,
|
|
201
|
+
): Promise<{ data: string; bytes: number; file: string } | null> {
|
|
202
|
+
const file = path.join(os.tmpdir(), `crewly-computer-${process.pid}-${Date.now()}.png`);
|
|
203
|
+
// maxWidth is what makes the screenshot match the coordinate space the
|
|
204
|
+
// model is told to use; without it the two drift and every click is off.
|
|
205
|
+
const shot = await runSkill(
|
|
206
|
+
{ action: 'screenshot', output: file, maxWidth: Math.round(TARGET_WIDTH) },
|
|
207
|
+
deps,
|
|
208
|
+
);
|
|
209
|
+
const written = (shot['path'] as string) ?? file;
|
|
210
|
+
try {
|
|
211
|
+
if (deps.readImage) {
|
|
212
|
+
const { data, bytes } = await deps.readImage(written);
|
|
213
|
+
return { data, bytes, file: written };
|
|
214
|
+
}
|
|
215
|
+
const buffer = await fs.readFile(written);
|
|
216
|
+
await fs.unlink(written).catch(() => undefined);
|
|
217
|
+
return { data: buffer.toString('base64'), bytes: buffer.length, file: written };
|
|
218
|
+
} catch {
|
|
219
|
+
return null;
|
|
220
|
+
}
|
|
221
|
+
}
|
|
222
|
+
|
|
223
|
+
/** Arguments the model may send. */
|
|
224
|
+
const computerSchema = z.object({
|
|
225
|
+
action: z.enum([
|
|
226
|
+
// Anthropic computer_20250124 vocabulary.
|
|
227
|
+
'screenshot', 'left_click', 'right_click', 'middle_click', 'double_click',
|
|
228
|
+
'triple_click', 'left_click_drag', 'mouse_move', 'key', 'type', 'scroll',
|
|
229
|
+
'wait', 'cursor_position',
|
|
230
|
+
// Crewly's element-level additions.
|
|
231
|
+
'snapshot', 'click_ref', 'fill_ref', 'wait_for', 'ocr', 'displays',
|
|
232
|
+
]).describe('What to do. Prefer snapshot + click_ref/fill_ref over coordinates when the app exposes elements.'),
|
|
233
|
+
coordinate: z.array(z.number()).length(2).optional()
|
|
234
|
+
.describe('[x, y] in the screenshot you were shown, not screen pixels. The tool converts.'),
|
|
235
|
+
start_coordinate: z.array(z.number()).length(2).optional()
|
|
236
|
+
.describe('[x, y] to drag from, for left_click_drag.'),
|
|
237
|
+
text: z.string().optional()
|
|
238
|
+
.describe('Text to type, the key combo for `key` (e.g. "command+s"), or the value for fill_ref.'),
|
|
239
|
+
ref: z.string().optional().describe('Element reference from a snapshot, e.g. "@e12".'),
|
|
240
|
+
app: z.string().optional().describe('Application to snapshot or wait for; defaults to the frontmost.'),
|
|
241
|
+
scroll_direction: z.enum(['up', 'down', 'left', 'right']).optional(),
|
|
242
|
+
scroll_amount: z.number().optional().describe('Scroll clicks; defaults to 3.'),
|
|
243
|
+
duration: z.number().optional().describe('Seconds to wait, for `wait`.'),
|
|
244
|
+
});
|
|
245
|
+
|
|
246
|
+
/**
|
|
247
|
+
* Translate the tool's arguments into the skill's own input.
|
|
248
|
+
*
|
|
249
|
+
* @param args - Validated tool arguments
|
|
250
|
+
* @param factor - Coordinate scale
|
|
251
|
+
* @returns The skill payload, or a refusal when the arguments do not fit
|
|
252
|
+
*/
|
|
253
|
+
export function toSkillInput(
|
|
254
|
+
args: z.infer<typeof computerSchema>,
|
|
255
|
+
factor: number,
|
|
256
|
+
): Record<string, unknown> | { error: string } {
|
|
257
|
+
const point = (pair?: number[]) =>
|
|
258
|
+
pair ? { x: toScreen(pair[0]!, factor), y: toScreen(pair[1]!, factor) } : null;
|
|
259
|
+
|
|
260
|
+
switch (args.action) {
|
|
261
|
+
case 'screenshot':
|
|
262
|
+
return { action: 'screenshot' };
|
|
263
|
+
case 'displays':
|
|
264
|
+
return { action: 'displays' };
|
|
265
|
+
case 'cursor_position':
|
|
266
|
+
// The skill has no cursor read; a screenshot answers the same question
|
|
267
|
+
// and is what the model will ask for next anyway.
|
|
268
|
+
return { action: 'screenshot' };
|
|
269
|
+
|
|
270
|
+
case 'left_click':
|
|
271
|
+
case 'right_click':
|
|
272
|
+
case 'double_click': {
|
|
273
|
+
const p = point(args.coordinate);
|
|
274
|
+
if (!p) return { error: `${args.action} needs a coordinate.` };
|
|
275
|
+
const button = args.action === 'right_click' ? 'right' : args.action === 'double_click' ? 'double' : 'left';
|
|
276
|
+
return { action: 'click', ...p, button };
|
|
277
|
+
}
|
|
278
|
+
case 'middle_click':
|
|
279
|
+
case 'triple_click': {
|
|
280
|
+
// Neither exists in the skill. Saying so beats silently doing something
|
|
281
|
+
// else: a model told "not supported" picks another route, one told
|
|
282
|
+
// "done" builds on a click that never happened.
|
|
283
|
+
return { error: `${args.action} is not supported on this platform. Use left_click, or select the text another way.` };
|
|
284
|
+
}
|
|
285
|
+
case 'mouse_move': {
|
|
286
|
+
const p = point(args.coordinate);
|
|
287
|
+
if (!p) return { error: 'mouse_move needs a coordinate.' };
|
|
288
|
+
return { action: 'move', ...p };
|
|
289
|
+
}
|
|
290
|
+
case 'left_click_drag': {
|
|
291
|
+
const from = point(args.start_coordinate);
|
|
292
|
+
const to = point(args.coordinate);
|
|
293
|
+
if (!from || !to) return { error: 'left_click_drag needs start_coordinate and coordinate.' };
|
|
294
|
+
return { action: 'drag', fromX: from.x, fromY: from.y, toX: to.x, toY: to.y };
|
|
295
|
+
}
|
|
296
|
+
case 'key':
|
|
297
|
+
if (!args.text) return { error: 'key needs `text`, e.g. "command+s".' };
|
|
298
|
+
return { action: 'key', key: args.text };
|
|
299
|
+
case 'type':
|
|
300
|
+
if (!args.text) return { error: 'type needs `text`.' };
|
|
301
|
+
return { action: 'type', text: args.text };
|
|
302
|
+
case 'scroll': {
|
|
303
|
+
const p = point(args.coordinate);
|
|
304
|
+
return {
|
|
305
|
+
action: 'scroll',
|
|
306
|
+
...(p ?? {}),
|
|
307
|
+
direction: args.scroll_direction ?? 'down',
|
|
308
|
+
amount: args.scroll_amount ?? 3,
|
|
309
|
+
};
|
|
310
|
+
}
|
|
311
|
+
|
|
312
|
+
case 'snapshot':
|
|
313
|
+
return { action: 'snapshot', ...(args.app ? { app: args.app } : {}) };
|
|
314
|
+
case 'click_ref':
|
|
315
|
+
if (!args.ref) return { error: 'click_ref needs `ref`, e.g. "@e12" from a snapshot.' };
|
|
316
|
+
return { action: 'click-ref', ref: args.ref };
|
|
317
|
+
case 'fill_ref':
|
|
318
|
+
if (!args.ref || args.text === undefined) return { error: 'fill_ref needs `ref` and `text`.' };
|
|
319
|
+
return { action: 'fill-ref', ref: args.ref, text: args.text };
|
|
320
|
+
case 'ocr':
|
|
321
|
+
return { action: 'ocr' };
|
|
322
|
+
case 'wait_for':
|
|
323
|
+
if (!args.app && !args.ref && !args.text) {
|
|
324
|
+
return { error: 'wait_for needs one of `app`, `ref` or `text`.' };
|
|
325
|
+
}
|
|
326
|
+
return {
|
|
327
|
+
action: 'wait-for',
|
|
328
|
+
...(args.app ? { app: args.app } : {}),
|
|
329
|
+
...(args.ref ? { ref: args.ref } : {}),
|
|
330
|
+
...(args.text ? { text: args.text } : {}),
|
|
331
|
+
};
|
|
332
|
+
case 'wait':
|
|
333
|
+
return { action: 'wait-for', idle: true, timeoutMs: Math.round((args.duration ?? 1) * 1000) };
|
|
334
|
+
}
|
|
335
|
+
}
|
|
336
|
+
|
|
337
|
+
/**
|
|
338
|
+
* Build the `computer` tool.
|
|
339
|
+
*
|
|
340
|
+
* @param deps - Injected IO for tests
|
|
341
|
+
* @returns The tool definition
|
|
342
|
+
*
|
|
343
|
+
* @example
|
|
344
|
+
* ```ts
|
|
345
|
+
* const tools = { computer: createComputerTool() };
|
|
346
|
+
* ```
|
|
347
|
+
*/
|
|
348
|
+
export function createComputerTool(deps: ComputerToolDeps = {}): ToolDefinition {
|
|
349
|
+
return {
|
|
350
|
+
description:
|
|
351
|
+
'Control this Mac: look at the screen and act on it. ' +
|
|
352
|
+
'Prefer `snapshot` then `click_ref`/`fill_ref` — naming an element cannot miss the way a coordinate can, ' +
|
|
353
|
+
'and it keeps working when the window moves. Fall back to coordinates for canvases and custom-drawn UI. ' +
|
|
354
|
+
'Coordinates are in the screenshot you were shown, not screen pixels. ' +
|
|
355
|
+
'Destructive key combos, password fields and credential apps are refused, and the owner can stop everything at any time.',
|
|
356
|
+
inputSchema: computerSchema,
|
|
357
|
+
sensitivity: 'destructive',
|
|
358
|
+
execute: async (rawArgs) => {
|
|
359
|
+
const args = rawArgs as z.infer<typeof computerSchema>;
|
|
360
|
+
|
|
361
|
+
// The scale has to come from the live display: the owner may have
|
|
362
|
+
// changed resolution or moved to another screen since the last call.
|
|
363
|
+
const displayResult = await runSkill({ action: 'displays' }, deps);
|
|
364
|
+
const displays = (displayResult['displays'] as DisplayInfo[] | undefined) ?? [];
|
|
365
|
+
const scale = scaleFor(displays);
|
|
366
|
+
|
|
367
|
+
const payload = toSkillInput(args, scale.factor);
|
|
368
|
+
if ('error' in payload) {
|
|
369
|
+
return { success: false, reason: 'bad_arguments', message: payload.error };
|
|
370
|
+
}
|
|
371
|
+
|
|
372
|
+
// A bare screenshot is taken once, by the capture step below. Running
|
|
373
|
+
// the skill's screenshot here as well would shoot the screen twice for
|
|
374
|
+
// one request — slow, and the two images could even differ.
|
|
375
|
+
const result = args.action === 'screenshot'
|
|
376
|
+
? { success: true }
|
|
377
|
+
: await runSkill(payload, deps);
|
|
378
|
+
|
|
379
|
+
// A refusal is returned as it stands. The rails phrase their own
|
|
380
|
+
// reasons and tell the agent what to do; a screenshot alongside would
|
|
381
|
+
// just be the same screen it could not act on.
|
|
382
|
+
if (result['success'] === false) return result;
|
|
383
|
+
|
|
384
|
+
const out: Record<string, unknown> = {
|
|
385
|
+
...result,
|
|
386
|
+
action: args.action,
|
|
387
|
+
screen: { width: scale.width, height: scale.height, actual: scale.screen },
|
|
388
|
+
};
|
|
389
|
+
|
|
390
|
+
// One call, one look. Showing the result of an action is what lets a
|
|
391
|
+
// model check its own work instead of assuming the click landed.
|
|
392
|
+
if (MUTATING.has(args.action) || args.action === 'screenshot') {
|
|
393
|
+
const image = await capture(deps);
|
|
394
|
+
if (image) {
|
|
395
|
+
out['type'] = 'image';
|
|
396
|
+
out['mimeType'] = 'image/png';
|
|
397
|
+
out['data'] = image.data;
|
|
398
|
+
out['sizeBytes'] = image.bytes;
|
|
399
|
+
out['note'] = `Screenshot is ${scale.width}×${scale.height}; give coordinates in that space.`;
|
|
400
|
+
}
|
|
401
|
+
}
|
|
402
|
+
return out;
|
|
403
|
+
},
|
|
404
|
+
};
|
|
405
|
+
}
|
|
@@ -0,0 +1,136 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Tests for checkpoint evaluation.
|
|
3
|
+
*
|
|
4
|
+
* A checkpoint is the only thing that can call a desktop subgoal done, so a
|
|
5
|
+
* checkpoint that passes when it should not is worse than having none: it
|
|
6
|
+
* launders the agent's claim into a verified fact.
|
|
7
|
+
*/
|
|
8
|
+
|
|
9
|
+
import { describe, it, expect } from 'vitest';
|
|
10
|
+
import { evaluateCheckpoint, describeCheckpoint, type Checkpoint } from './desktop-checkpoint.js';
|
|
11
|
+
|
|
12
|
+
/** IO over a fake filesystem. */
|
|
13
|
+
function fs(files: Record<string, string>) {
|
|
14
|
+
return {
|
|
15
|
+
readFile: async (p: string) => {
|
|
16
|
+
if (!(p in files)) throw new Error('ENOENT');
|
|
17
|
+
return files[p]!;
|
|
18
|
+
},
|
|
19
|
+
statFile: async (p: string) => {
|
|
20
|
+
if (!(p in files)) throw new Error('ENOENT');
|
|
21
|
+
return { size: files[p]!.length };
|
|
22
|
+
},
|
|
23
|
+
};
|
|
24
|
+
}
|
|
25
|
+
|
|
26
|
+
describe('file checkpoints', () => {
|
|
27
|
+
it('fails an empty file — the signature of a save that never completed', async () => {
|
|
28
|
+
const out = await evaluateCheckpoint({ kind: 'file-exists', path: '/a' }, fs({ '/a': '' }));
|
|
29
|
+
expect(out.passed).toBe(false);
|
|
30
|
+
expect(out.reason).toMatch(/empty/);
|
|
31
|
+
});
|
|
32
|
+
|
|
33
|
+
it('allows an empty file when the task said so', async () => {
|
|
34
|
+
expect((await evaluateCheckpoint({ kind: 'file-exists', path: '/a', allowEmpty: true }, fs({ '/a': '' }))).passed).toBe(true);
|
|
35
|
+
});
|
|
36
|
+
|
|
37
|
+
it('says the file is missing rather than just failing', async () => {
|
|
38
|
+
const out = await evaluateCheckpoint({ kind: 'file-exists', path: '/nope' }, fs({}));
|
|
39
|
+
expect(out.reason).toContain('/nope');
|
|
40
|
+
});
|
|
41
|
+
|
|
42
|
+
it('shows what the file does say when the content is wrong', async () => {
|
|
43
|
+
const out = await evaluateCheckpoint({ kind: 'file-contains', path: '/a', text: 'hello' }, fs({ '/a': 'goodbye' }));
|
|
44
|
+
expect(out.passed).toBe(false);
|
|
45
|
+
expect(out.observed).toBe('goodbye');
|
|
46
|
+
});
|
|
47
|
+
|
|
48
|
+
it('matches a pattern', async () => {
|
|
49
|
+
expect((await evaluateCheckpoint({ kind: 'file-matches', path: '/a', pattern: '^\\d+$' }, fs({ '/a': '42' }))).passed).toBe(true);
|
|
50
|
+
expect((await evaluateCheckpoint({ kind: 'file-matches', path: '/a', pattern: '^\\d+$' }, fs({ '/a': 'x' }))).passed).toBe(false);
|
|
51
|
+
});
|
|
52
|
+
|
|
53
|
+
it('checks the other half of a rename', async () => {
|
|
54
|
+
expect((await evaluateCheckpoint({ kind: 'file-absent', path: '/old' }, fs({}))).passed).toBe(true);
|
|
55
|
+
expect((await evaluateCheckpoint({ kind: 'file-absent', path: '/old' }, fs({ '/old': 'x' }))).passed).toBe(false);
|
|
56
|
+
});
|
|
57
|
+
});
|
|
58
|
+
|
|
59
|
+
describe('screen checkpoints', () => {
|
|
60
|
+
const snapshot = async () => [
|
|
61
|
+
{ role: 'AXButton', name: 'Save' },
|
|
62
|
+
{ role: 'AXStaticText', name: 'Untitled document' },
|
|
63
|
+
];
|
|
64
|
+
|
|
65
|
+
it('finds an element by name, and by role when given one', async () => {
|
|
66
|
+
expect((await evaluateCheckpoint({ kind: 'element-present', name: 'Save' }, { snapshot })).passed).toBe(true);
|
|
67
|
+
expect((await evaluateCheckpoint({ kind: 'element-present', name: 'Save', role: 'AXButton' }, { snapshot })).passed).toBe(true);
|
|
68
|
+
expect((await evaluateCheckpoint({ kind: 'element-present', name: 'Save', role: 'AXMenuItem' }, { snapshot })).passed).toBe(false);
|
|
69
|
+
});
|
|
70
|
+
|
|
71
|
+
it('checks a dialog is gone, which is how "closed it" is proved', async () => {
|
|
72
|
+
expect((await evaluateCheckpoint({ kind: 'element-absent', name: 'Save' }, { snapshot })).passed).toBe(false);
|
|
73
|
+
expect((await evaluateCheckpoint({ kind: 'element-absent', name: 'Nothing' }, { snapshot })).passed).toBe(true);
|
|
74
|
+
});
|
|
75
|
+
|
|
76
|
+
it('reads text off the screen for apps with no element tree', async () => {
|
|
77
|
+
const screenText = async () => ['CREWLY EVAL 7734'];
|
|
78
|
+
expect((await evaluateCheckpoint({ kind: 'text-on-screen', text: 'eval 7734' }, { screenText })).passed).toBe(true);
|
|
79
|
+
expect((await evaluateCheckpoint({ kind: 'text-on-screen', text: 'absent' }, { screenText })).passed).toBe(false);
|
|
80
|
+
});
|
|
81
|
+
|
|
82
|
+
it('compares the frontmost app case-insensitively and reports what is actually there', async () => {
|
|
83
|
+
const frontmostApp = async () => 'TextEdit';
|
|
84
|
+
expect((await evaluateCheckpoint({ kind: 'app-frontmost', app: 'textedit' }, { frontmostApp })).passed).toBe(true);
|
|
85
|
+
const out = await evaluateCheckpoint({ kind: 'app-frontmost', app: 'Numbers' }, { frontmostApp });
|
|
86
|
+
expect(out.observed).toBe('TextEdit');
|
|
87
|
+
});
|
|
88
|
+
|
|
89
|
+
it('fails rather than pretends when it cannot see', async () => {
|
|
90
|
+
// No snapshot dependency wired: it must not quietly pass.
|
|
91
|
+
expect((await evaluateCheckpoint({ kind: 'element-present', name: 'Save' }, {})).passed).toBe(false);
|
|
92
|
+
expect((await evaluateCheckpoint({ kind: 'text-on-screen', text: 'x' }, {})).passed).toBe(false);
|
|
93
|
+
expect((await evaluateCheckpoint({ kind: 'app-frontmost', app: 'x' }, {})).passed).toBe(false);
|
|
94
|
+
});
|
|
95
|
+
});
|
|
96
|
+
|
|
97
|
+
describe('shell checkpoints', () => {
|
|
98
|
+
it('passes on exit zero and reports the output on failure', async () => {
|
|
99
|
+
const runShell = async (c: string) =>
|
|
100
|
+
c === 'true' ? { code: 0, stdout: '', stderr: '' } : { code: 1, stdout: '', stderr: 'boom' };
|
|
101
|
+
expect((await evaluateCheckpoint({ kind: 'shell', command: 'true' }, { runShell })).passed).toBe(true);
|
|
102
|
+
const out = await evaluateCheckpoint({ kind: 'shell', command: 'false' }, { runShell });
|
|
103
|
+
expect(out.passed).toBe(false);
|
|
104
|
+
expect(out.observed).toBe('boom');
|
|
105
|
+
});
|
|
106
|
+
});
|
|
107
|
+
|
|
108
|
+
describe('robustness', () => {
|
|
109
|
+
it('turns a thrown check into a failed one, never an exception', async () => {
|
|
110
|
+
const out = await evaluateCheckpoint({ kind: 'file-exists', path: '/a' }, {
|
|
111
|
+
statFile: async () => { throw new Error('disk on fire'); },
|
|
112
|
+
});
|
|
113
|
+
// A broken check must not read as a broken task, but it must not pass.
|
|
114
|
+
expect(out.passed).toBe(false);
|
|
115
|
+
});
|
|
116
|
+
});
|
|
117
|
+
|
|
118
|
+
describe('describeCheckpoint', () => {
|
|
119
|
+
it('reads as a sentence for every kind', () => {
|
|
120
|
+
const all: Checkpoint[] = [
|
|
121
|
+
{ kind: 'file-exists', path: '/a' },
|
|
122
|
+
{ kind: 'file-contains', path: '/a', text: 'x' },
|
|
123
|
+
{ kind: 'file-matches', path: '/a', pattern: 'x' },
|
|
124
|
+
{ kind: 'file-absent', path: '/a' },
|
|
125
|
+
{ kind: 'app-frontmost', app: 'Finder' },
|
|
126
|
+
{ kind: 'element-present', name: 'Save' },
|
|
127
|
+
{ kind: 'element-absent', name: 'Save' },
|
|
128
|
+
{ kind: 'text-on-screen', text: 'hi' },
|
|
129
|
+
{ kind: 'shell', command: 'true' },
|
|
130
|
+
];
|
|
131
|
+
for (const cp of all) {
|
|
132
|
+
expect(describeCheckpoint(cp).length).toBeGreaterThan(5);
|
|
133
|
+
}
|
|
134
|
+
expect(describeCheckpoint({ kind: 'shell', command: 'x', description: 'the build passes' })).toBe('the build passes');
|
|
135
|
+
});
|
|
136
|
+
});
|