crewly 1.20.40 → 1.20.47
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/config/skills/_common/desktop-guards.sh +485 -0
- package/config/skills/_common/desktop-guards.test.sh +242 -0
- package/config/skills/_common/desktop-perceive.swift +530 -0
- package/config/skills/_common/desktop-presence.swift +343 -0
- package/config/skills/agent/_common/desktop-guards.sh +4 -0
- package/config/skills/agent/computer-use/SKILL.md +88 -0
- package/config/skills/agent/computer-use/execute.sh +249 -3
- package/config/skills/agent/desktop-app-control/SKILL.md +19 -0
- package/config/skills/agent/remote-browser/SKILL.md +19 -0
- package/dist/backend/backend/src/controllers/desktop/desktop.controller.d.ts +105 -0
- package/dist/backend/backend/src/controllers/desktop/desktop.controller.d.ts.map +1 -0
- package/dist/backend/backend/src/controllers/desktop/desktop.controller.js +278 -0
- package/dist/backend/backend/src/controllers/desktop/desktop.controller.js.map +1 -0
- package/dist/backend/backend/src/controllers/desktop/desktop.routes.d.ts +21 -0
- package/dist/backend/backend/src/controllers/desktop/desktop.routes.d.ts.map +1 -0
- package/dist/backend/backend/src/controllers/desktop/desktop.routes.js +31 -0
- package/dist/backend/backend/src/controllers/desktop/desktop.routes.js.map +1 -0
- package/dist/backend/backend/src/routes/api.routes.d.ts.map +1 -1
- package/dist/backend/backend/src/routes/api.routes.js +3 -0
- package/dist/backend/backend/src/routes/api.routes.js.map +1 -1
- package/dist/backend/backend/src/services/cloud/mobile-api-relay.service.d.ts.map +1 -1
- package/dist/backend/backend/src/services/cloud/mobile-api-relay.service.js +9 -0
- package/dist/backend/backend/src/services/cloud/mobile-api-relay.service.js.map +1 -1
- package/dist/backend/backend/src/services/slack/slack-team-channel.service.d.ts +24 -0
- package/dist/backend/backend/src/services/slack/slack-team-channel.service.d.ts.map +1 -1
- package/dist/backend/backend/src/services/slack/slack-team-channel.service.js +60 -1
- package/dist/backend/backend/src/services/slack/slack-team-channel.service.js.map +1 -1
- package/dist/backend/backend/src/services/slack/slack.service.d.ts +17 -0
- package/dist/backend/backend/src/services/slack/slack.service.d.ts.map +1 -1
- package/dist/backend/backend/src/services/slack/slack.service.js +32 -0
- package/dist/backend/backend/src/services/slack/slack.service.js.map +1 -1
- package/dist/backend/backend/src/types/slack.types.d.ts +10 -0
- package/dist/backend/backend/src/types/slack.types.d.ts.map +1 -1
- package/dist/backend/backend/src/types/slack.types.js.map +1 -1
- package/dist/backend/backend/src/utils/incomplete-turn.utils.d.ts +1 -1
- package/dist/backend/backend/src/utils/incomplete-turn.utils.d.ts.map +1 -1
- package/dist/backend/backend/src/utils/incomplete-turn.utils.js +4 -0
- package/dist/backend/backend/src/utils/incomplete-turn.utils.js.map +1 -1
- package/dist/backend/build-info.json +2 -2
- package/dist/cli/backend/src/services/slack/slack-team-channel.service.d.ts +24 -0
- package/dist/cli/backend/src/services/slack/slack-team-channel.service.d.ts.map +1 -1
- package/dist/cli/backend/src/services/slack/slack-team-channel.service.js +60 -1
- package/dist/cli/backend/src/services/slack/slack-team-channel.service.js.map +1 -1
- package/dist/cli/backend/src/services/slack/slack.service.d.ts +17 -0
- package/dist/cli/backend/src/services/slack/slack.service.d.ts.map +1 -1
- package/dist/cli/backend/src/services/slack/slack.service.js +32 -0
- package/dist/cli/backend/src/services/slack/slack.service.js.map +1 -1
- package/dist/cli/backend/src/types/slack.types.d.ts +10 -0
- package/dist/cli/backend/src/types/slack.types.d.ts.map +1 -1
- package/dist/cli/backend/src/types/slack.types.js.map +1 -1
- package/dist/cli/backend/src/utils/incomplete-turn.utils.d.ts +1 -1
- package/dist/cli/backend/src/utils/incomplete-turn.utils.d.ts.map +1 -1
- package/dist/cli/backend/src/utils/incomplete-turn.utils.js +4 -0
- package/dist/cli/backend/src/utils/incomplete-turn.utils.js.map +1 -1
- package/package.json +1 -1
- package/packages/crewly-agent/src/eval/desktop/desktop-tasks.test.ts +96 -0
- package/packages/crewly-agent/src/eval/desktop/desktop-tasks.ts +226 -0
- package/packages/crewly-agent/src/runtime/agent-runner.service.ts +20 -1
- package/packages/crewly-agent/src/runtime/computer.tool.test.ts +219 -0
- package/packages/crewly-agent/src/runtime/computer.tool.ts +405 -0
- package/packages/crewly-agent/src/runtime/desktop-checkpoint.test.ts +136 -0
- package/packages/crewly-agent/src/runtime/desktop-checkpoint.ts +231 -0
- package/packages/crewly-agent/src/runtime/desktop-recovery.test.ts +100 -0
- package/packages/crewly-agent/src/runtime/desktop-recovery.ts +195 -0
- package/packages/crewly-agent/src/runtime/desktop-task-runtime.test.ts +251 -0
- package/packages/crewly-agent/src/runtime/desktop-task-runtime.ts +423 -0
- package/packages/crewly-agent/src/runtime/desktop-task.tool.test.ts +218 -0
- package/packages/crewly-agent/src/runtime/desktop-task.tool.ts +343 -0
- package/packages/crewly-agent/src/runtime/tool-registry.test.ts +17 -0
- package/packages/crewly-agent/src/runtime/tool-registry.ts +54 -0
- package/packages/crewly-agent/src/runtime/types.ts +10 -1
- package/config/skills/agent/vnc-browser/SKILL.md +0 -140
|
@@ -0,0 +1,231 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Checkpoints — how a desktop subgoal is judged done.
|
|
3
|
+
*
|
|
4
|
+
* Phase 4 of docs/research/computer-use-capability-assessment.md. The failure
|
|
5
|
+
* this exists for is the third one in §5.1: the agent says "I saved it" while
|
|
6
|
+
* the save dialog is still open. Its own account of what happened is exactly
|
|
7
|
+
* the thing that cannot be trusted, so a subgoal is complete when something
|
|
8
|
+
* outside the agent says so — a file on disk, an element on screen, a command
|
|
9
|
+
* that exits zero.
|
|
10
|
+
*
|
|
11
|
+
* Checkpoints are deliberately narrow. Each one is a small, decidable
|
|
12
|
+
* question with a yes or no answer and a reason when it is no, because a
|
|
13
|
+
* model that is told "the file has 0 bytes" can act, where one told "task
|
|
14
|
+
* failed" can only guess.
|
|
15
|
+
*
|
|
16
|
+
* @module runtime/desktop-checkpoint
|
|
17
|
+
*/
|
|
18
|
+
|
|
19
|
+
import { spawn } from 'child_process';
|
|
20
|
+
import { promises as fs } from 'fs';
|
|
21
|
+
|
|
22
|
+
/** How long a shell checkpoint may take before it counts as failed. */
|
|
23
|
+
const SHELL_TIMEOUT_MS = 15_000;
|
|
24
|
+
|
|
25
|
+
/**
|
|
26
|
+
* A decidable statement about the world after a subgoal.
|
|
27
|
+
*
|
|
28
|
+
* `shell` is the escape hatch, but the named kinds are preferred: they give a
|
|
29
|
+
* usable reason on failure, where a shell command can only give an exit code.
|
|
30
|
+
*/
|
|
31
|
+
export type Checkpoint =
|
|
32
|
+
/** A file exists and, unless `allowEmpty`, has content. */
|
|
33
|
+
| { kind: 'file-exists'; path: string; allowEmpty?: boolean }
|
|
34
|
+
/** A file exists and contains this text. */
|
|
35
|
+
| { kind: 'file-contains'; path: string; text: string }
|
|
36
|
+
/** A file exists and matches this pattern. */
|
|
37
|
+
| { kind: 'file-matches'; path: string; pattern: string }
|
|
38
|
+
/** A file is gone (a rename's other half, a cleanup). */
|
|
39
|
+
| { kind: 'file-absent'; path: string }
|
|
40
|
+
/** This application is in front. */
|
|
41
|
+
| { kind: 'app-frontmost'; app: string }
|
|
42
|
+
/** An element with this name (and optionally role) is on screen. */
|
|
43
|
+
| { kind: 'element-present'; name: string; role?: string; app?: string }
|
|
44
|
+
/** No element with this name is on screen — a dialog that should be gone. */
|
|
45
|
+
| { kind: 'element-absent'; name: string; app?: string }
|
|
46
|
+
/** This text is readable on screen. */
|
|
47
|
+
| { kind: 'text-on-screen'; text: string }
|
|
48
|
+
/** This command exits zero. */
|
|
49
|
+
| { kind: 'shell'; command: string; description?: string };
|
|
50
|
+
|
|
51
|
+
/** The answer, with enough detail to act on. */
|
|
52
|
+
export interface CheckpointResult {
|
|
53
|
+
passed: boolean;
|
|
54
|
+
/** Why not, phrased for the model to act on. Absent when it passed. */
|
|
55
|
+
reason?: string;
|
|
56
|
+
/** What was actually observed, when that helps more than prose. */
|
|
57
|
+
observed?: string;
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
/** Injectable IO, so the evaluator is testable without a desktop. */
|
|
61
|
+
export interface CheckpointDeps {
|
|
62
|
+
readFile?: (path: string) => Promise<string>;
|
|
63
|
+
statFile?: (path: string) => Promise<{ size: number }>;
|
|
64
|
+
runShell?: (command: string) => Promise<{ code: number; stdout: string; stderr: string }>;
|
|
65
|
+
/** Elements currently on screen, from a desktop snapshot. */
|
|
66
|
+
snapshot?: (app?: string) => Promise<Array<{ role: string; name?: string }>>;
|
|
67
|
+
/** Text currently on screen, from OCR. */
|
|
68
|
+
screenText?: () => Promise<string[]>;
|
|
69
|
+
/** The frontmost application's name. */
|
|
70
|
+
frontmostApp?: () => Promise<string>;
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
/**
|
|
74
|
+
* Run a shell command, capturing its outcome rather than throwing.
|
|
75
|
+
*
|
|
76
|
+
* @param command - The command
|
|
77
|
+
* @returns Exit code and output
|
|
78
|
+
*/
|
|
79
|
+
async function defaultShell(command: string): Promise<{ code: number; stdout: string; stderr: string }> {
|
|
80
|
+
return new Promise((resolve) => {
|
|
81
|
+
const child = spawn('bash', ['-c', command], { stdio: ['ignore', 'pipe', 'pipe'] });
|
|
82
|
+
let stdout = '';
|
|
83
|
+
let stderr = '';
|
|
84
|
+
const timer = setTimeout(() => {
|
|
85
|
+
child.kill('SIGKILL');
|
|
86
|
+
resolve({ code: 124, stdout, stderr: `timed out after ${SHELL_TIMEOUT_MS}ms` });
|
|
87
|
+
}, SHELL_TIMEOUT_MS);
|
|
88
|
+
child.stdout.on('data', (c) => { stdout += String(c); });
|
|
89
|
+
child.stderr.on('data', (c) => { stderr += String(c); });
|
|
90
|
+
child.on('error', (err) => { clearTimeout(timer); resolve({ code: 127, stdout, stderr: err.message }); });
|
|
91
|
+
child.on('close', (code) => { clearTimeout(timer); resolve({ code: code ?? 1, stdout, stderr }); });
|
|
92
|
+
});
|
|
93
|
+
}
|
|
94
|
+
|
|
95
|
+
/**
|
|
96
|
+
* Describe a checkpoint in words, for a plan the owner might read.
|
|
97
|
+
*
|
|
98
|
+
* @param checkpoint - The checkpoint
|
|
99
|
+
* @returns One line
|
|
100
|
+
*
|
|
101
|
+
* @example
|
|
102
|
+
* describeCheckpoint({ kind: 'file-contains', path: '/tmp/a.txt', text: 'ok' })
|
|
103
|
+
* // → '/tmp/a.txt contains "ok"'
|
|
104
|
+
*/
|
|
105
|
+
export function describeCheckpoint(checkpoint: Checkpoint): string {
|
|
106
|
+
switch (checkpoint.kind) {
|
|
107
|
+
case 'file-exists': return `${checkpoint.path} exists${checkpoint.allowEmpty ? '' : ' and is not empty'}`;
|
|
108
|
+
case 'file-contains': return `${checkpoint.path} contains "${checkpoint.text}"`;
|
|
109
|
+
case 'file-matches': return `${checkpoint.path} matches /${checkpoint.pattern}/`;
|
|
110
|
+
case 'file-absent': return `${checkpoint.path} is gone`;
|
|
111
|
+
case 'app-frontmost': return `${checkpoint.app} is in front`;
|
|
112
|
+
case 'element-present': return `"${checkpoint.name}" is on screen`;
|
|
113
|
+
case 'element-absent': return `"${checkpoint.name}" is no longer on screen`;
|
|
114
|
+
case 'text-on-screen': return `"${checkpoint.text}" is readable on screen`;
|
|
115
|
+
case 'shell': return checkpoint.description ?? `\`${checkpoint.command}\` succeeds`;
|
|
116
|
+
}
|
|
117
|
+
}
|
|
118
|
+
|
|
119
|
+
/**
|
|
120
|
+
* Decide whether a checkpoint holds.
|
|
121
|
+
*
|
|
122
|
+
* Never throws: an unreadable file or a crashed command is a failed
|
|
123
|
+
* checkpoint with a reason, not an exception for the caller to interpret.
|
|
124
|
+
*
|
|
125
|
+
* @param checkpoint - What to check
|
|
126
|
+
* @param deps - Injected IO
|
|
127
|
+
* @returns Whether it holds, and why not when it does not
|
|
128
|
+
*
|
|
129
|
+
* @example
|
|
130
|
+
* await evaluateCheckpoint({ kind: 'file-contains', path: '/tmp/a', text: 'hi' })
|
|
131
|
+
* // → { passed: false, reason: '/tmp/a exists but does not contain "hi"', observed: '…' }
|
|
132
|
+
*/
|
|
133
|
+
export async function evaluateCheckpoint(
|
|
134
|
+
checkpoint: Checkpoint,
|
|
135
|
+
deps: CheckpointDeps = {},
|
|
136
|
+
): Promise<CheckpointResult> {
|
|
137
|
+
const readFile = deps.readFile ?? ((p: string) => fs.readFile(p, 'utf8'));
|
|
138
|
+
const statFile = deps.statFile ?? (async (p: string) => ({ size: (await fs.stat(p)).size }));
|
|
139
|
+
const runShell = deps.runShell ?? defaultShell;
|
|
140
|
+
|
|
141
|
+
try {
|
|
142
|
+
switch (checkpoint.kind) {
|
|
143
|
+
case 'file-exists': {
|
|
144
|
+
const stat = await statFile(checkpoint.path).catch(() => null);
|
|
145
|
+
if (!stat) return { passed: false, reason: `${checkpoint.path} does not exist.` };
|
|
146
|
+
if (!checkpoint.allowEmpty && stat.size === 0) {
|
|
147
|
+
// An empty file is the signature of a save that opened a dialog and
|
|
148
|
+
// never completed — worth calling out rather than passing.
|
|
149
|
+
return { passed: false, reason: `${checkpoint.path} exists but is empty — the write did not complete.` };
|
|
150
|
+
}
|
|
151
|
+
return { passed: true };
|
|
152
|
+
}
|
|
153
|
+
|
|
154
|
+
case 'file-absent': {
|
|
155
|
+
const stat = await statFile(checkpoint.path).catch(() => null);
|
|
156
|
+
return stat
|
|
157
|
+
? { passed: false, reason: `${checkpoint.path} is still there.` }
|
|
158
|
+
: { passed: true };
|
|
159
|
+
}
|
|
160
|
+
|
|
161
|
+
case 'file-contains':
|
|
162
|
+
case 'file-matches': {
|
|
163
|
+
const content = await readFile(checkpoint.path).catch(() => null);
|
|
164
|
+
if (content === null) return { passed: false, reason: `${checkpoint.path} does not exist or cannot be read.` };
|
|
165
|
+
const hit = checkpoint.kind === 'file-contains'
|
|
166
|
+
? content.includes(checkpoint.text)
|
|
167
|
+
: new RegExp(checkpoint.pattern).test(content);
|
|
168
|
+
if (hit) return { passed: true };
|
|
169
|
+
const wanted = checkpoint.kind === 'file-contains' ? `"${checkpoint.text}"` : `/${checkpoint.pattern}/`;
|
|
170
|
+
return {
|
|
171
|
+
passed: false,
|
|
172
|
+
reason: `${checkpoint.path} exists but does not match ${wanted}.`,
|
|
173
|
+
observed: content.slice(0, 200),
|
|
174
|
+
};
|
|
175
|
+
}
|
|
176
|
+
|
|
177
|
+
case 'app-frontmost': {
|
|
178
|
+
if (!deps.frontmostApp) return { passed: false, reason: 'Cannot see which application is in front.' };
|
|
179
|
+
const front = await deps.frontmostApp();
|
|
180
|
+
return front.toLowerCase() === checkpoint.app.toLowerCase()
|
|
181
|
+
? { passed: true }
|
|
182
|
+
: { passed: false, reason: `${checkpoint.app} is not in front.`, observed: front };
|
|
183
|
+
}
|
|
184
|
+
|
|
185
|
+
case 'element-present':
|
|
186
|
+
case 'element-absent': {
|
|
187
|
+
if (!deps.snapshot) return { passed: false, reason: 'Cannot read the elements on screen.' };
|
|
188
|
+
const elements = await deps.snapshot(checkpoint.app);
|
|
189
|
+
const wanted = checkpoint.name.toLowerCase();
|
|
190
|
+
const found = elements.find((e) => {
|
|
191
|
+
if (!e.name || !e.name.toLowerCase().includes(wanted)) return false;
|
|
192
|
+
return checkpoint.kind === 'element-present' && checkpoint.role ? e.role === checkpoint.role : true;
|
|
193
|
+
});
|
|
194
|
+
if (checkpoint.kind === 'element-present') {
|
|
195
|
+
return found
|
|
196
|
+
? { passed: true }
|
|
197
|
+
: { passed: false, reason: `Nothing called "${checkpoint.name}" is on screen.` };
|
|
198
|
+
}
|
|
199
|
+
return found
|
|
200
|
+
? { passed: false, reason: `"${checkpoint.name}" is still on screen.`, observed: found.role }
|
|
201
|
+
: { passed: true };
|
|
202
|
+
}
|
|
203
|
+
|
|
204
|
+
case 'text-on-screen': {
|
|
205
|
+
if (!deps.screenText) return { passed: false, reason: 'Cannot read the text on screen.' };
|
|
206
|
+
const lines = await deps.screenText();
|
|
207
|
+
const wanted = checkpoint.text.toLowerCase();
|
|
208
|
+
return lines.some((l) => l.toLowerCase().includes(wanted))
|
|
209
|
+
? { passed: true }
|
|
210
|
+
: { passed: false, reason: `"${checkpoint.text}" is not readable on screen.` };
|
|
211
|
+
}
|
|
212
|
+
|
|
213
|
+
case 'shell': {
|
|
214
|
+
const { code, stdout, stderr } = await runShell(checkpoint.command);
|
|
215
|
+
if (code === 0) return { passed: true };
|
|
216
|
+
return {
|
|
217
|
+
passed: false,
|
|
218
|
+
reason: `${describeCheckpoint(checkpoint)} — exited ${code}.`,
|
|
219
|
+
observed: (stderr || stdout).slice(0, 200),
|
|
220
|
+
};
|
|
221
|
+
}
|
|
222
|
+
}
|
|
223
|
+
} catch (err) {
|
|
224
|
+
// A checkpoint that throws is a checkpoint that did not pass. Turning it
|
|
225
|
+
// into an exception would let a broken check read as a broken task.
|
|
226
|
+
return {
|
|
227
|
+
passed: false,
|
|
228
|
+
reason: `Could not check: ${err instanceof Error ? err.message : String(err)}`,
|
|
229
|
+
};
|
|
230
|
+
}
|
|
231
|
+
}
|
|
@@ -0,0 +1,100 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Tests for surprise detection.
|
|
3
|
+
*
|
|
4
|
+
* The failure being guarded against: a dialog appears at step 15, the agent
|
|
5
|
+
* does not notice, and the next twenty steps act on a world that no longer
|
|
6
|
+
* exists. Each one looks fine on its own.
|
|
7
|
+
*/
|
|
8
|
+
|
|
9
|
+
import { describe, it, expect } from 'vitest';
|
|
10
|
+
import { detectSurprise, shouldRetryAfter, type Scene } from './desktop-recovery.js';
|
|
11
|
+
|
|
12
|
+
describe('detectSurprise', () => {
|
|
13
|
+
it('sees nothing wrong with an ordinary window', () => {
|
|
14
|
+
expect(detectSurprise({ app: 'TextEdit', elements: [{ role: 'AXButton', name: 'Bold' }] })).toBeNull();
|
|
15
|
+
});
|
|
16
|
+
|
|
17
|
+
it('spots a modal and lists the buttons so the agent can choose', () => {
|
|
18
|
+
const out = detectSurprise({
|
|
19
|
+
elements: [
|
|
20
|
+
{ role: 'AXSheet', name: 'Save changes?' },
|
|
21
|
+
{ role: 'AXButton', name: 'Save' },
|
|
22
|
+
{ role: 'AXButton', name: "Don't Save" },
|
|
23
|
+
],
|
|
24
|
+
});
|
|
25
|
+
expect(out?.kind).toBe('modal-dialog');
|
|
26
|
+
expect(out?.instruction).toContain("Don't Save");
|
|
27
|
+
// The most common way to get this wrong is to save when nobody asked.
|
|
28
|
+
expect(out?.instruction).toMatch(/if the task did not ask to save, do not save/i);
|
|
29
|
+
expect(out?.selfRecoverable).toBe(true);
|
|
30
|
+
});
|
|
31
|
+
|
|
32
|
+
it('tells the agent its refs are stale after a dialog', () => {
|
|
33
|
+
const out = detectSurprise({ elements: [{ role: 'AXDialog', name: 'Export' }] });
|
|
34
|
+
expect(out?.instruction).toMatch(/fresh snapshot/);
|
|
35
|
+
});
|
|
36
|
+
|
|
37
|
+
it('treats a permission prompt as the owner\'s decision, not a dialog to answer', () => {
|
|
38
|
+
const out = detectSurprise({
|
|
39
|
+
elements: [
|
|
40
|
+
{ role: 'AXSheet', name: 'Terminal would like to access your Documents folder' },
|
|
41
|
+
{ role: 'AXButton', name: 'Allow' },
|
|
42
|
+
],
|
|
43
|
+
});
|
|
44
|
+
expect(out?.kind).toBe('permission-prompt');
|
|
45
|
+
expect(out?.selfRecoverable).toBe(false);
|
|
46
|
+
});
|
|
47
|
+
|
|
48
|
+
it('stops at a sign-in wall instead of typing credentials', () => {
|
|
49
|
+
for (const name of ['Sign in to continue', 'Enter your password', '请输入验证码']) {
|
|
50
|
+
const out = detectSurprise({ elements: [{ role: 'AXStaticText', name }] });
|
|
51
|
+
expect(out?.kind, name).toBe('login-required');
|
|
52
|
+
expect(out?.selfRecoverable).toBe(false);
|
|
53
|
+
}
|
|
54
|
+
});
|
|
55
|
+
|
|
56
|
+
it('reads a locked screen off the last failure, and calls it unrecoverable', () => {
|
|
57
|
+
const out = detectSurprise({ elements: [], lastFailure: { reason: 'screen_locked' } });
|
|
58
|
+
expect(out?.kind).toBe('screen-locked');
|
|
59
|
+
expect(out?.selfRecoverable).toBe(false);
|
|
60
|
+
});
|
|
61
|
+
|
|
62
|
+
it('warns not to repeat a timed-out action blindly — it may have gone through', () => {
|
|
63
|
+
const out = detectSurprise({ elements: [], lastFailure: { reason: 'timeout' } });
|
|
64
|
+
expect(out?.kind).toBe('app-not-responding');
|
|
65
|
+
expect(out?.instruction).toMatch(/may have gone through/);
|
|
66
|
+
});
|
|
67
|
+
|
|
68
|
+
it('notices focus moving away from the app the plan is about', () => {
|
|
69
|
+
expect(detectSurprise({ app: 'Safari', elements: [] }, 'TextEdit')?.kind).toBe('focus-lost');
|
|
70
|
+
expect(detectSurprise({ app: 'TextEdit', elements: [] }, 'textedit')).toBeNull();
|
|
71
|
+
});
|
|
72
|
+
|
|
73
|
+
it('prefers the specific diagnosis when two apply', () => {
|
|
74
|
+
// A permission prompt is also a modal; the specific instruction is the
|
|
75
|
+
// useful one.
|
|
76
|
+
const out = detectSurprise({
|
|
77
|
+
elements: [
|
|
78
|
+
{ role: 'AXSheet', name: 'Crewly would like to access your Calendar' },
|
|
79
|
+
{ role: 'AXButton', name: 'Allow' },
|
|
80
|
+
],
|
|
81
|
+
});
|
|
82
|
+
expect(out?.kind).toBe('permission-prompt');
|
|
83
|
+
});
|
|
84
|
+
});
|
|
85
|
+
|
|
86
|
+
describe('shouldRetryAfter', () => {
|
|
87
|
+
it('never retries something that needs a person', () => {
|
|
88
|
+
const out = shouldRetryAfter('login-required', 0, false);
|
|
89
|
+
expect(out.retry).toBe(false);
|
|
90
|
+
expect(out.escalation).toMatch(/needs a person/);
|
|
91
|
+
});
|
|
92
|
+
|
|
93
|
+
it('allows three goes at a recoverable surprise, then stops', () => {
|
|
94
|
+
expect(shouldRetryAfter('modal-dialog', 0, true).retry).toBe(true);
|
|
95
|
+
expect(shouldRetryAfter('modal-dialog', 2, true).retry).toBe(true);
|
|
96
|
+
const out = shouldRetryAfter('modal-dialog', 3, true);
|
|
97
|
+
expect(out.retry).toBe(false);
|
|
98
|
+
expect(out.escalation).toMatch(/will not be different/);
|
|
99
|
+
});
|
|
100
|
+
});
|
|
@@ -0,0 +1,195 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Surprises, and what to do about them.
|
|
3
|
+
*
|
|
4
|
+
* Phase 4 of docs/research/computer-use-capability-assessment.md, §5.7. The
|
|
5
|
+
* second failure in §5.1 is the expensive one: at step 15 a dialog appears,
|
|
6
|
+
* the agent does not notice, and the next twenty steps act on a state that no
|
|
7
|
+
* longer exists. Every one of them looks fine in isolation.
|
|
8
|
+
*
|
|
9
|
+
* The surprises are not open-ended. A desktop throws the same handful over
|
|
10
|
+
* and over — a modal sheet, a permission prompt, a beachballing app, focus
|
|
11
|
+
* landing somewhere else, a login wall. Naming them means the agent gets a
|
|
12
|
+
* specific instruction instead of having to work out from a screenshot that
|
|
13
|
+
* something is wrong at all.
|
|
14
|
+
*
|
|
15
|
+
* @module runtime/desktop-recovery
|
|
16
|
+
*/
|
|
17
|
+
|
|
18
|
+
/** What went unexpectedly wrong. */
|
|
19
|
+
export type SurpriseKind =
|
|
20
|
+
/** A sheet or modal is waiting for an answer. */
|
|
21
|
+
| 'modal-dialog'
|
|
22
|
+
/** macOS is asking the user to allow something. */
|
|
23
|
+
| 'permission-prompt'
|
|
24
|
+
/** A sign-in wall. */
|
|
25
|
+
| 'login-required'
|
|
26
|
+
/** The app stopped responding. */
|
|
27
|
+
| 'app-not-responding'
|
|
28
|
+
/** Another application took the front. */
|
|
29
|
+
| 'focus-lost'
|
|
30
|
+
/** The screen was locked mid-task. */
|
|
31
|
+
| 'screen-locked';
|
|
32
|
+
|
|
33
|
+
/** A surprise, and the instruction that goes with it. */
|
|
34
|
+
export interface Surprise {
|
|
35
|
+
kind: SurpriseKind;
|
|
36
|
+
/** What was seen, for the log and for the agent. */
|
|
37
|
+
detail: string;
|
|
38
|
+
/** What the agent should do next, in one instruction. */
|
|
39
|
+
instruction: string;
|
|
40
|
+
/**
|
|
41
|
+
* Whether the agent can deal with this itself. False means a person has to
|
|
42
|
+
* — there is no point spending the budget discovering that.
|
|
43
|
+
*/
|
|
44
|
+
selfRecoverable: boolean;
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
/** An element as a snapshot reports it. */
|
|
48
|
+
export interface SceneElement {
|
|
49
|
+
role: string;
|
|
50
|
+
name?: string;
|
|
51
|
+
subrole?: string;
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
/** Enough of the world to spot a surprise in. */
|
|
55
|
+
export interface Scene {
|
|
56
|
+
app?: string;
|
|
57
|
+
elements: SceneElement[];
|
|
58
|
+
/** Whatever the last action answered, when it failed. */
|
|
59
|
+
lastFailure?: { reason?: string; message?: string };
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
/** Button labels that mean "a modal is waiting". */
|
|
63
|
+
const MODAL_ROLES = new Set(['AXSheet', 'AXDialog']);
|
|
64
|
+
|
|
65
|
+
/** Words that mark a sign-in wall rather than an ordinary form. */
|
|
66
|
+
const LOGIN_WORDS = ['sign in', 'log in', 'login', 'password', 'two-factor', 'verification code', '验证码', '登录'];
|
|
67
|
+
|
|
68
|
+
/** Words macOS uses when it wants the user to allow something. */
|
|
69
|
+
const PERMISSION_WORDS = ['would like to access', 'wants access', 'allow', 'grant access', '访问'];
|
|
70
|
+
|
|
71
|
+
/**
|
|
72
|
+
* Look at a scene and name what is wrong with it, if anything.
|
|
73
|
+
*
|
|
74
|
+
* Order matters: the checks run from most specific to least, because a
|
|
75
|
+
* permission prompt *is* a modal dialog and the specific instruction is the
|
|
76
|
+
* useful one.
|
|
77
|
+
*
|
|
78
|
+
* @param scene - What is on screen, plus the last failure if there was one
|
|
79
|
+
* @param expectedApp - The app the agent believes it is working in
|
|
80
|
+
* @returns The surprise, or null when the scene looks ordinary
|
|
81
|
+
*
|
|
82
|
+
* @example
|
|
83
|
+
* detectSurprise({ app: 'TextEdit', elements: [{ role: 'AXSheet', name: 'Save changes?' }] })
|
|
84
|
+
* // → { kind: 'modal-dialog', instruction: 'Answer the dialog…' }
|
|
85
|
+
*/
|
|
86
|
+
export function detectSurprise(scene: Scene, expectedApp?: string): Surprise | null {
|
|
87
|
+
// A rail already said what was wrong; it outranks anything inferred.
|
|
88
|
+
const failure = scene.lastFailure?.reason;
|
|
89
|
+
if (failure === 'screen_locked') {
|
|
90
|
+
return {
|
|
91
|
+
kind: 'screen-locked',
|
|
92
|
+
detail: 'The screen was locked.',
|
|
93
|
+
instruction: 'Stop and report that the screen is locked. It cannot be unlocked from here.',
|
|
94
|
+
selfRecoverable: false,
|
|
95
|
+
};
|
|
96
|
+
}
|
|
97
|
+
|
|
98
|
+
const named = scene.elements.filter((e) => e.name);
|
|
99
|
+
const textOf = (e: SceneElement) => (e.name ?? '').toLowerCase();
|
|
100
|
+
|
|
101
|
+
// A permission prompt is a modal, so it has to be checked first.
|
|
102
|
+
const permission = named.find((e) => PERMISSION_WORDS.some((w) => textOf(e).includes(w)));
|
|
103
|
+
if (permission && named.some((e) => MODAL_ROLES.has(e.role))) {
|
|
104
|
+
return {
|
|
105
|
+
kind: 'permission-prompt',
|
|
106
|
+
detail: `macOS is asking: "${permission.name}"`,
|
|
107
|
+
instruction:
|
|
108
|
+
'A macOS permission prompt is open. Do not answer it — granting access on the owner\'s behalf is their decision. ' +
|
|
109
|
+
'Stop and tell them what is being asked for.',
|
|
110
|
+
selfRecoverable: false,
|
|
111
|
+
};
|
|
112
|
+
}
|
|
113
|
+
|
|
114
|
+
const login = named.find((e) => LOGIN_WORDS.some((w) => textOf(e).includes(w)));
|
|
115
|
+
if (login) {
|
|
116
|
+
return {
|
|
117
|
+
kind: 'login-required',
|
|
118
|
+
detail: `A sign-in is being asked for: "${login.name}"`,
|
|
119
|
+
instruction:
|
|
120
|
+
'This needs credentials, which desktop control never enters. Stop and ask the owner to sign in, then continue.',
|
|
121
|
+
selfRecoverable: false,
|
|
122
|
+
};
|
|
123
|
+
}
|
|
124
|
+
|
|
125
|
+
const modal = scene.elements.find((e) => MODAL_ROLES.has(e.role));
|
|
126
|
+
if (modal) {
|
|
127
|
+
const buttons = named.filter((e) => e.role === 'AXButton').map((e) => e.name!);
|
|
128
|
+
return {
|
|
129
|
+
kind: 'modal-dialog',
|
|
130
|
+
detail: `A dialog is open${modal.name ? `: "${modal.name}"` : ''}.`,
|
|
131
|
+
instruction:
|
|
132
|
+
`A dialog is waiting for an answer and nothing else will work until it is dealt with. ` +
|
|
133
|
+
(buttons.length
|
|
134
|
+
? `Its buttons are: ${buttons.join(', ')}. Choose the one that matches the task — if the task did not ask to save, do not save. `
|
|
135
|
+
: 'Take a snapshot to see its buttons. ') +
|
|
136
|
+
'Then take a fresh snapshot: the window behind it has probably changed.',
|
|
137
|
+
selfRecoverable: true,
|
|
138
|
+
};
|
|
139
|
+
}
|
|
140
|
+
|
|
141
|
+
if (scene.lastFailure?.reason === 'timeout') {
|
|
142
|
+
return {
|
|
143
|
+
kind: 'app-not-responding',
|
|
144
|
+
detail: 'The last action timed out.',
|
|
145
|
+
instruction:
|
|
146
|
+
'The application is not responding. Wait for the screen to settle, then take a fresh snapshot before trying again. ' +
|
|
147
|
+
'Do not repeat the action blindly — it may have gone through.',
|
|
148
|
+
selfRecoverable: true,
|
|
149
|
+
};
|
|
150
|
+
}
|
|
151
|
+
|
|
152
|
+
if (expectedApp && scene.app && scene.app.toLowerCase() !== expectedApp.toLowerCase()) {
|
|
153
|
+
return {
|
|
154
|
+
kind: 'focus-lost',
|
|
155
|
+
detail: `${scene.app} is in front, not ${expectedApp}.`,
|
|
156
|
+
instruction:
|
|
157
|
+
`Focus moved to ${scene.app}. Bring ${expectedApp} back to the front and take a fresh snapshot — ` +
|
|
158
|
+
'any refs from before are stale.',
|
|
159
|
+
selfRecoverable: true,
|
|
160
|
+
};
|
|
161
|
+
}
|
|
162
|
+
|
|
163
|
+
return null;
|
|
164
|
+
}
|
|
165
|
+
|
|
166
|
+
/**
|
|
167
|
+
* Whether to keep trying after a surprise.
|
|
168
|
+
*
|
|
169
|
+
* Three attempts at the same kind is the limit. A desktop surprise that
|
|
170
|
+
* survives three goes is not one more click away — it is something the agent
|
|
171
|
+
* has misunderstood, and the cheapest next move is to say so.
|
|
172
|
+
*
|
|
173
|
+
* @param kind - What went wrong
|
|
174
|
+
* @param alreadyTried - How many times this kind has been handled in this subgoal
|
|
175
|
+
* @param selfRecoverable - From the surprise
|
|
176
|
+
* @returns Whether to try again, and what to say when not
|
|
177
|
+
*/
|
|
178
|
+
export function shouldRetryAfter(
|
|
179
|
+
kind: SurpriseKind,
|
|
180
|
+
alreadyTried: number,
|
|
181
|
+
selfRecoverable: boolean,
|
|
182
|
+
): { retry: boolean; escalation?: string } {
|
|
183
|
+
if (!selfRecoverable) {
|
|
184
|
+
return { retry: false, escalation: `${kind} needs a person — stop and report it rather than working around it.` };
|
|
185
|
+
}
|
|
186
|
+
if (alreadyTried >= 3) {
|
|
187
|
+
return {
|
|
188
|
+
retry: false,
|
|
189
|
+
escalation:
|
|
190
|
+
`The same problem (${kind}) has come back three times. Stop and report what you tried and what you see — ` +
|
|
191
|
+
'a fourth attempt will not be different.',
|
|
192
|
+
};
|
|
193
|
+
}
|
|
194
|
+
return { retry: true };
|
|
195
|
+
}
|