crewly 1.20.40 → 1.20.47

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (72) hide show
  1. package/config/skills/_common/desktop-guards.sh +485 -0
  2. package/config/skills/_common/desktop-guards.test.sh +242 -0
  3. package/config/skills/_common/desktop-perceive.swift +530 -0
  4. package/config/skills/_common/desktop-presence.swift +343 -0
  5. package/config/skills/agent/_common/desktop-guards.sh +4 -0
  6. package/config/skills/agent/computer-use/SKILL.md +88 -0
  7. package/config/skills/agent/computer-use/execute.sh +249 -3
  8. package/config/skills/agent/desktop-app-control/SKILL.md +19 -0
  9. package/config/skills/agent/remote-browser/SKILL.md +19 -0
  10. package/dist/backend/backend/src/controllers/desktop/desktop.controller.d.ts +105 -0
  11. package/dist/backend/backend/src/controllers/desktop/desktop.controller.d.ts.map +1 -0
  12. package/dist/backend/backend/src/controllers/desktop/desktop.controller.js +278 -0
  13. package/dist/backend/backend/src/controllers/desktop/desktop.controller.js.map +1 -0
  14. package/dist/backend/backend/src/controllers/desktop/desktop.routes.d.ts +21 -0
  15. package/dist/backend/backend/src/controllers/desktop/desktop.routes.d.ts.map +1 -0
  16. package/dist/backend/backend/src/controllers/desktop/desktop.routes.js +31 -0
  17. package/dist/backend/backend/src/controllers/desktop/desktop.routes.js.map +1 -0
  18. package/dist/backend/backend/src/routes/api.routes.d.ts.map +1 -1
  19. package/dist/backend/backend/src/routes/api.routes.js +3 -0
  20. package/dist/backend/backend/src/routes/api.routes.js.map +1 -1
  21. package/dist/backend/backend/src/services/cloud/mobile-api-relay.service.d.ts.map +1 -1
  22. package/dist/backend/backend/src/services/cloud/mobile-api-relay.service.js +9 -0
  23. package/dist/backend/backend/src/services/cloud/mobile-api-relay.service.js.map +1 -1
  24. package/dist/backend/backend/src/services/slack/slack-team-channel.service.d.ts +24 -0
  25. package/dist/backend/backend/src/services/slack/slack-team-channel.service.d.ts.map +1 -1
  26. package/dist/backend/backend/src/services/slack/slack-team-channel.service.js +60 -1
  27. package/dist/backend/backend/src/services/slack/slack-team-channel.service.js.map +1 -1
  28. package/dist/backend/backend/src/services/slack/slack.service.d.ts +17 -0
  29. package/dist/backend/backend/src/services/slack/slack.service.d.ts.map +1 -1
  30. package/dist/backend/backend/src/services/slack/slack.service.js +32 -0
  31. package/dist/backend/backend/src/services/slack/slack.service.js.map +1 -1
  32. package/dist/backend/backend/src/types/slack.types.d.ts +10 -0
  33. package/dist/backend/backend/src/types/slack.types.d.ts.map +1 -1
  34. package/dist/backend/backend/src/types/slack.types.js.map +1 -1
  35. package/dist/backend/backend/src/utils/incomplete-turn.utils.d.ts +1 -1
  36. package/dist/backend/backend/src/utils/incomplete-turn.utils.d.ts.map +1 -1
  37. package/dist/backend/backend/src/utils/incomplete-turn.utils.js +4 -0
  38. package/dist/backend/backend/src/utils/incomplete-turn.utils.js.map +1 -1
  39. package/dist/backend/build-info.json +2 -2
  40. package/dist/cli/backend/src/services/slack/slack-team-channel.service.d.ts +24 -0
  41. package/dist/cli/backend/src/services/slack/slack-team-channel.service.d.ts.map +1 -1
  42. package/dist/cli/backend/src/services/slack/slack-team-channel.service.js +60 -1
  43. package/dist/cli/backend/src/services/slack/slack-team-channel.service.js.map +1 -1
  44. package/dist/cli/backend/src/services/slack/slack.service.d.ts +17 -0
  45. package/dist/cli/backend/src/services/slack/slack.service.d.ts.map +1 -1
  46. package/dist/cli/backend/src/services/slack/slack.service.js +32 -0
  47. package/dist/cli/backend/src/services/slack/slack.service.js.map +1 -1
  48. package/dist/cli/backend/src/types/slack.types.d.ts +10 -0
  49. package/dist/cli/backend/src/types/slack.types.d.ts.map +1 -1
  50. package/dist/cli/backend/src/types/slack.types.js.map +1 -1
  51. package/dist/cli/backend/src/utils/incomplete-turn.utils.d.ts +1 -1
  52. package/dist/cli/backend/src/utils/incomplete-turn.utils.d.ts.map +1 -1
  53. package/dist/cli/backend/src/utils/incomplete-turn.utils.js +4 -0
  54. package/dist/cli/backend/src/utils/incomplete-turn.utils.js.map +1 -1
  55. package/package.json +1 -1
  56. package/packages/crewly-agent/src/eval/desktop/desktop-tasks.test.ts +96 -0
  57. package/packages/crewly-agent/src/eval/desktop/desktop-tasks.ts +226 -0
  58. package/packages/crewly-agent/src/runtime/agent-runner.service.ts +20 -1
  59. package/packages/crewly-agent/src/runtime/computer.tool.test.ts +219 -0
  60. package/packages/crewly-agent/src/runtime/computer.tool.ts +405 -0
  61. package/packages/crewly-agent/src/runtime/desktop-checkpoint.test.ts +136 -0
  62. package/packages/crewly-agent/src/runtime/desktop-checkpoint.ts +231 -0
  63. package/packages/crewly-agent/src/runtime/desktop-recovery.test.ts +100 -0
  64. package/packages/crewly-agent/src/runtime/desktop-recovery.ts +195 -0
  65. package/packages/crewly-agent/src/runtime/desktop-task-runtime.test.ts +251 -0
  66. package/packages/crewly-agent/src/runtime/desktop-task-runtime.ts +423 -0
  67. package/packages/crewly-agent/src/runtime/desktop-task.tool.test.ts +218 -0
  68. package/packages/crewly-agent/src/runtime/desktop-task.tool.ts +343 -0
  69. package/packages/crewly-agent/src/runtime/tool-registry.test.ts +17 -0
  70. package/packages/crewly-agent/src/runtime/tool-registry.ts +54 -0
  71. package/packages/crewly-agent/src/runtime/types.ts +10 -1
  72. package/config/skills/agent/vnc-browser/SKILL.md +0 -140
@@ -0,0 +1,251 @@
1
+ /**
2
+ * Tests for the desktop task runtime.
3
+ *
4
+ * The rule these exist to hold: a subgoal is done when its checkpoint holds,
5
+ * never when the agent says so. Every other behaviour here — budgets,
6
+ * surprise handling, confirmation — is in service of a forty-step task not
7
+ * ending in a confident, false report.
8
+ */
9
+
10
+ import { describe, it, expect, vi } from 'vitest';
11
+ import {
12
+ beginTask, step, chooseRoute, needsConfirmation, awaitConfirmation,
13
+ resolveConfirmation, summarize, activeSubgoal, DEFAULT_TASK_BUDGET,
14
+ type Subgoal, type TaskState,
15
+ } from './desktop-task-runtime.js';
16
+ import type { Scene } from './desktop-recovery.js';
17
+
18
+ /** A plan of two subgoals, each checked against a file. */
19
+ function plan(): Subgoal[] {
20
+ return [
21
+ { id: 's1', goal: 'write the note', checkpoint: { kind: 'file-contains', path: '/tmp/a', text: 'ok' }, maxSteps: 3 },
22
+ { id: 's2', goal: 'rename it', checkpoint: { kind: 'file-exists', path: '/tmp/b' }, maxSteps: 3 },
23
+ ];
24
+ }
25
+
26
+ /** An empty screen — nothing surprising on it. */
27
+ const CALM: Scene = { elements: [] };
28
+
29
+ /** Checkpoint IO that answers from a fake filesystem. */
30
+ function fakeFs(files: Record<string, string>) {
31
+ return {
32
+ readFile: async (p: string) => {
33
+ if (!(p in files)) throw new Error('ENOENT');
34
+ return files[p]!;
35
+ },
36
+ statFile: async (p: string) => {
37
+ if (!(p in files)) throw new Error('ENOENT');
38
+ return { size: files[p]!.length };
39
+ },
40
+ };
41
+ }
42
+
43
+ describe('checkpoints decide, not the agent', () => {
44
+ it('keeps the agent on a subgoal until the world matches, and says what is wrong', async () => {
45
+ const state = beginTask('write and rename', plan());
46
+ const files: Record<string, string> = {};
47
+
48
+ // The agent has "done" the work, but the file is not there.
49
+ const first = await step(state, CALM, fakeFs(files));
50
+ expect(first.kind).toBe('continue');
51
+ expect((first as { hint?: string }).hint).toContain('/tmp/a');
52
+ expect(activeSubgoal(state)!.subgoal.id).toBe('s1');
53
+
54
+ // Now it really is.
55
+ files['/tmp/a'] = 'ok';
56
+ const second = await step(state, CALM, fakeFs(files));
57
+ expect(second.kind).toBe('advance');
58
+ expect((second as { next: Subgoal }).next.id).toBe('s2');
59
+ });
60
+
61
+ it('calls an empty file out rather than passing it — that is the unfinished save', async () => {
62
+ const state = beginTask('save', [
63
+ { id: 's1', goal: 'save the file', checkpoint: { kind: 'file-exists', path: '/tmp/a' } },
64
+ ]);
65
+ const out = await step(state, CALM, fakeFs({ '/tmp/a': '' }));
66
+ expect(out.kind).toBe('continue');
67
+ expect((out as { hint?: string }).hint).toMatch(/empty|did not complete/i);
68
+ });
69
+
70
+ it('reports done only when every checkpoint held', async () => {
71
+ const state = beginTask('write and rename', plan());
72
+ const files = { '/tmp/a': 'ok', '/tmp/b': 'x' };
73
+ expect((await step(state, CALM, fakeFs(files))).kind).toBe('advance');
74
+ expect((await step(state, CALM, fakeFs(files))).kind).toBe('done');
75
+ expect(summarize(state)).toContain('2/2 verified');
76
+ });
77
+
78
+ it('escalates a subgoal that burns its budget, and does not move on', async () => {
79
+ const state = beginTask('write', [
80
+ { id: 's1', goal: 'write the note', checkpoint: { kind: 'file-exists', path: '/tmp/never' }, maxSteps: 2 },
81
+ { id: 's2', goal: 'later work', checkpoint: { kind: 'file-exists', path: '/tmp/b' } },
82
+ ]);
83
+ expect((await step(state, CALM, fakeFs({}))).kind).toBe('continue');
84
+ const out = await step(state, CALM, fakeFs({}));
85
+ expect(out.kind).toBe('escalate');
86
+ // The later subgoal assumed this one worked, so it must not start.
87
+ expect(state.progress[1]!.state).toBe('pending');
88
+ expect((out as { detail: string }).detail).toContain('later subgoals assume');
89
+ });
90
+ });
91
+
92
+ describe('budgets', () => {
93
+ it('stops on the step ceiling and says what got done', async () => {
94
+ const state = beginTask('long task', plan(), { ...DEFAULT_TASK_BUDGET, maxSteps: 1 });
95
+ await step(state, CALM, fakeFs({}));
96
+ const out = await step(state, CALM, fakeFs({}));
97
+ expect(out.kind).toBe('budget-exhausted');
98
+ expect((out as { remaining: string[] }).remaining).toContain('write the note');
99
+ });
100
+
101
+ it('stops on the clock', async () => {
102
+ let clock = 1_000;
103
+ const state = beginTask('slow task', plan(), { ...DEFAULT_TASK_BUDGET, maxDurationMs: 100 }, () => clock);
104
+ clock += 500;
105
+ const out = await step(state, CALM, fakeFs({}), () => clock);
106
+ expect(out.kind).toBe('budget-exhausted');
107
+ expect((out as { reason: string }).reason).toMatch(/longer than/);
108
+ });
109
+ });
110
+
111
+ describe('surprises', () => {
112
+ const dialog: Scene = {
113
+ elements: [
114
+ { role: 'AXSheet', name: 'Save changes?' },
115
+ { role: 'AXButton', name: 'Save' },
116
+ { role: 'AXButton', name: "Don't Save" },
117
+ { role: 'AXButton', name: 'Cancel' },
118
+ ],
119
+ };
120
+
121
+ it('does not test the checkpoint while a dialog is covering the window', async () => {
122
+ const state = beginTask('edit', plan());
123
+ const checkpoint = vi.fn();
124
+ const out = await step(state, dialog, { statFile: checkpoint as never });
125
+ expect(out.kind).toBe('continue');
126
+ // Testing it now would be testing the wrong world.
127
+ expect(checkpoint).not.toHaveBeenCalled();
128
+ expect((out as { hint?: string }).hint).toContain("Don't Save");
129
+ });
130
+
131
+ it('gives up on a surprise that keeps coming back instead of looping', async () => {
132
+ const state = beginTask('edit', plan());
133
+ let out = await step(state, dialog, {});
134
+ out = await step(state, dialog, {});
135
+ out = await step(state, dialog, {});
136
+ out = await step(state, dialog, {});
137
+ expect(out.kind).toBe('escalate');
138
+ expect((out as { detail: string }).detail).toMatch(/three times/);
139
+ });
140
+
141
+ it('never tries to answer a permission prompt on the owner\'s behalf', async () => {
142
+ const state = beginTask('open', plan());
143
+ const prompt: Scene = {
144
+ elements: [
145
+ { role: 'AXSheet', name: 'Terminal would like to access your Documents' },
146
+ { role: 'AXButton', name: 'Allow' },
147
+ ],
148
+ };
149
+ const out = await step(state, prompt, {});
150
+ expect(out.kind).toBe('escalate');
151
+ expect((out as { detail: string }).detail).toMatch(/their decision|owner/i);
152
+ });
153
+
154
+ it('stops at a login wall rather than typing credentials', async () => {
155
+ const state = beginTask('open', plan());
156
+ const out = await step(state, { elements: [{ role: 'AXStaticText', name: 'Sign in to continue' }] }, {});
157
+ expect(out.kind).toBe('escalate');
158
+ expect((out as { reason: string }).reason).toBe('login-required');
159
+ });
160
+
161
+ it('notices focus moving to another app and says the refs are stale', async () => {
162
+ const state = beginTask('work in TextEdit', [
163
+ { id: 's1', goal: 'type', checkpoint: { kind: 'app-frontmost', app: 'TextEdit' } },
164
+ ]);
165
+ const out = await step(state, { app: 'Safari', elements: [] }, {});
166
+ expect(out.kind).toBe('continue');
167
+ expect((out as { hint?: string }).hint).toMatch(/stale/);
168
+ });
169
+ });
170
+
171
+ describe('routing', () => {
172
+ it('prefers a skill over the screen when one exists', () => {
173
+ const out = chooseRoute('send the summary by email', ['gmail-send']);
174
+ expect(out.route).toBe('skill');
175
+ expect(out.candidate).toBe('gmail-send');
176
+ });
177
+
178
+ it('falls back to the screen when the skill is not installed', () => {
179
+ expect(chooseRoute('send the summary by email', []).route).toBe('element');
180
+ });
181
+
182
+ it('uses the browser for a page and pixels for a canvas', () => {
183
+ expect(chooseRoute('open https://example.com and read it').route).toBe('browser');
184
+ expect(chooseRoute('draw a line on the canvas').route).toBe('pixel');
185
+ });
186
+
187
+ it('hands credentials to a person', () => {
188
+ expect(chooseRoute('log in with the password').route).toBe('human');
189
+ });
190
+
191
+ it('defaults to elements, not coordinates', () => {
192
+ const out = chooseRoute('rename the file in Finder');
193
+ expect(out.route).toBe('element');
194
+ expect(out.because).toMatch(/cannot miss/);
195
+ });
196
+ });
197
+
198
+ describe('irreversible actions', () => {
199
+ it('spots the ones that cannot be undone, in either language', () => {
200
+ for (const action of ['click the Send button', 'delete the folder', 'publish the post', '发送邮件', '删除文件']) {
201
+ expect(needsConfirmation(action).required, action).toBe(true);
202
+ }
203
+ });
204
+
205
+ it('does not stop ordinary work', () => {
206
+ // "sender" contains "send" but is not it — the word boundary is the point.
207
+ for (const action of ['open the document', 'read the second row', 'check the sender name']) {
208
+ expect(needsConfirmation(action).required, action).toBe(false);
209
+ }
210
+ });
211
+
212
+ it('holds the task until the owner answers, doing nothing meanwhile', async () => {
213
+ const state = beginTask('send it', plan());
214
+ awaitConfirmation(state, 'click Send on the email');
215
+ const out = await step(state, CALM, fakeFs({ '/tmp/a': 'ok' }));
216
+ // Even though the checkpoint would now pass, nothing moves.
217
+ expect(out.kind).toBe('await-confirmation');
218
+ expect(activeSubgoal(state)!.subgoal.id).toBe('s1');
219
+ });
220
+
221
+ it('carries on when approved', () => {
222
+ const state = beginTask('send it', plan());
223
+ awaitConfirmation(state, 'click Send');
224
+ const out = resolveConfirmation(state, true);
225
+ expect(out.kind).toBe('continue');
226
+ expect(state.awaitingConfirmation).toBeUndefined();
227
+ });
228
+
229
+ it('stops for good when declined, and forbids working around it', () => {
230
+ const state = beginTask('send it', plan());
231
+ awaitConfirmation(state, 'click Send');
232
+ const out = resolveConfirmation(state, false);
233
+ expect(out.kind).toBe('escalate');
234
+ expect((out as { detail: string }).detail).toMatch(/do not look for another way/i);
235
+ expect(state.progress[0]!.state).toBe('skipped');
236
+ });
237
+ });
238
+
239
+ describe('summary', () => {
240
+ it('names a stuck subgoal as stuck rather than rounding up', async () => {
241
+ const state: TaskState = beginTask('two things', plan());
242
+ await step(state, CALM, fakeFs({ '/tmp/a': 'ok' })); // s1 done
243
+ state.progress[1]!.state = 'stuck';
244
+ state.progress[1]!.blockedBy = 'the dialog never closed';
245
+ const text = summarize(state);
246
+ expect(text).toContain('1/2 verified');
247
+ expect(text).toContain('✓ write the note');
248
+ expect(text).toContain('✗ rename it');
249
+ expect(text).toContain('the dialog never closed');
250
+ });
251
+ });
@@ -0,0 +1,423 @@
1
+ /**
2
+ * The desktop task runtime — subgoals, checkpoints, budgets, escalation.
3
+ *
4
+ * Phase 4 of docs/research/computer-use-capability-assessment.md, and the one
5
+ * the document calls the dividing line for unattended work. Phases 1–3 gave
6
+ * an agent safe, precise, one-call access to the desktop. None of that stops
7
+ * a forty-step task from going wrong in the four ways §5.1 lists, because
8
+ * every one of them is about the *shape* of the attempt rather than any
9
+ * single action:
10
+ *
11
+ * - taking the GUI route when a skill would have done it → the router
12
+ * - drifting onto a state that changed at step 15 → checkpoints
13
+ * - reporting success that never happened → checkpoints
14
+ * - doing something irreversible unasked → confirmation
15
+ *
16
+ * A subgoal is not finished when the agent says so. It is finished when its
17
+ * checkpoint holds, and the runtime is what refuses to move on until it does.
18
+ *
19
+ * @module runtime/desktop-task-runtime
20
+ */
21
+
22
+ import { evaluateCheckpoint, describeCheckpoint, type Checkpoint, type CheckpointDeps, type CheckpointResult } from './desktop-checkpoint.js';
23
+ import { detectSurprise, shouldRetryAfter, type Scene, type Surprise } from './desktop-recovery.js';
24
+
25
+ /** One step of the plan. */
26
+ export interface Subgoal {
27
+ id: string;
28
+ /** What to achieve, in the agent's own words. */
29
+ goal: string;
30
+ /** How the runtime will know it happened. */
31
+ checkpoint: Checkpoint;
32
+ /** Steps this subgoal may take before it is declared stuck. */
33
+ maxSteps?: number;
34
+ }
35
+
36
+ /** Where a subgoal stands. */
37
+ export type SubgoalState = 'pending' | 'active' | 'done' | 'stuck' | 'skipped';
38
+
39
+ /** A subgoal plus its progress. */
40
+ export interface SubgoalProgress {
41
+ subgoal: Subgoal;
42
+ state: SubgoalState;
43
+ stepsUsed: number;
44
+ /** Surprises handled while working on it, by kind. */
45
+ recoveries: Record<string, number>;
46
+ /** Why it is stuck, when it is. */
47
+ blockedBy?: string;
48
+ lastCheck?: CheckpointResult;
49
+ }
50
+
51
+ /** Caps for a whole task. */
52
+ export interface TaskBudget {
53
+ /** Total steps across every subgoal. */
54
+ maxSteps: number;
55
+ /** Wall-clock milliseconds. */
56
+ maxDurationMs: number;
57
+ /** Steps any one subgoal may take when it does not say. */
58
+ defaultSubgoalSteps: number;
59
+ }
60
+
61
+ /** Sensible caps. A desktop task that needs more than this has gone wrong. */
62
+ export const DEFAULT_TASK_BUDGET: TaskBudget = {
63
+ maxSteps: 120,
64
+ maxDurationMs: 15 * 60 * 1000,
65
+ defaultSubgoalSteps: 20,
66
+ };
67
+
68
+ /** The layers an action can be taken at, cheapest and most reliable first. */
69
+ export type Route = 'skill' | 'browser' | 'element' | 'pixel' | 'human';
70
+
71
+ /** What the runtime says to do next. */
72
+ export type Directive =
73
+ /** Keep working on this subgoal. */
74
+ | { kind: 'continue'; subgoal: Subgoal; stepsLeft: number; hint?: string }
75
+ /** The checkpoint held; here is the next subgoal. */
76
+ | { kind: 'advance'; completed: Subgoal; next: Subgoal }
77
+ /** Everything's checkpoint held. */
78
+ | { kind: 'done'; summary: string }
79
+ /** Something needs a person. */
80
+ | { kind: 'escalate'; reason: string; detail: string }
81
+ /** An irreversible action is waiting on the owner. */
82
+ | { kind: 'await-confirmation'; action: string; detail: string }
83
+ /** The budget ran out. */
84
+ | { kind: 'budget-exhausted'; reason: string; completed: string[]; remaining: string[] };
85
+
86
+ /**
87
+ * Pick the cheapest layer that can do a thing.
88
+ *
89
+ * The first failure in §5.1 is an agent opening Mail.app to send an email
90
+ * Crewly already has a skill for. Thirty clicks that can each go wrong, in
91
+ * place of one call that cannot. The router exists to make the cheap answer
92
+ * the obvious one.
93
+ *
94
+ * @param goal - The subgoal text
95
+ * @param available - What this install can do (skill ids, connector names)
96
+ * @returns The layer to prefer, and why
97
+ *
98
+ * @example
99
+ * chooseRoute('send the summary by email', ['gmail-send'])
100
+ * // → { route: 'skill', because: 'gmail-send does this without touching the screen' }
101
+ */
102
+ export function chooseRoute(
103
+ goal: string,
104
+ available: string[] = [],
105
+ ): { route: Route; because: string; candidate?: string } {
106
+ const text = goal.toLowerCase();
107
+
108
+ // A named skill beats everything. Matching is on the words a person would
109
+ // use, not on the skill id, because the plan is written in prose.
110
+ const skillHints: Array<{ words: string[]; skill: string }> = [
111
+ { words: ['email', 'mail', '邮件'], skill: 'gmail-send' },
112
+ { words: ['calendar', 'meeting', 'event', '日历'], skill: 'calendar-create' },
113
+ { words: ['drive', 'upload', '上传'], skill: 'drive-upload' },
114
+ { words: ['slack', 'channel', '频道'], skill: 'reply-channel' },
115
+ { words: ['spreadsheet', 'sheet', '表格'], skill: 'sheets-write' },
116
+ { words: ['doc', 'document', '文档'], skill: 'docs-write' },
117
+ ];
118
+ for (const hint of skillHints) {
119
+ if (!hint.words.some((w) => text.includes(w))) continue;
120
+ if (available.includes(hint.skill)) {
121
+ return {
122
+ route: 'skill',
123
+ candidate: hint.skill,
124
+ because: `${hint.skill} does this without touching the screen — no clicks to misfire.`,
125
+ };
126
+ }
127
+ }
128
+
129
+ // No trailing \b on the scheme: ':' and '/' are both non-word characters,
130
+ // so \b between them never matches and every URL fell through to `element`.
131
+ if (/\b(web ?page|website|browser|url)\b|https?:\/\//.test(text)) {
132
+ return { route: 'browser', because: 'A page is DOM, so the browser tools are exact where coordinates are not.' };
133
+ }
134
+
135
+ if (/\b(draw|canvas|game|sketch|paint)\b/.test(text)) {
136
+ return { route: 'pixel', because: 'Nothing here exposes elements, so coordinates are the only option.' };
137
+ }
138
+
139
+ if (/\b(sign in|log ?in|password|2fa|captcha|credential)\b/.test(text)) {
140
+ return { route: 'human', because: 'Credentials are never entered by an agent.' };
141
+ }
142
+
143
+ return {
144
+ route: 'element',
145
+ because: 'Take a snapshot and act on refs — naming an element cannot miss the way a coordinate can.',
146
+ };
147
+ }
148
+
149
+ /** Phrases that mean an action cannot be taken back. */
150
+ const IRREVERSIBLE = [
151
+ 'send', 'submit', 'publish', 'post', 'delete', 'remove', 'erase', 'empty trash',
152
+ 'pay', 'purchase', 'buy', 'transfer', 'confirm order', 'deploy', 'merge',
153
+ '发送', '提交', '发布', '删除', '付款', '购买',
154
+ ];
155
+
156
+ /**
157
+ * Whether an action needs the owner's word before it happens.
158
+ *
159
+ * Deliberately generous: a false positive costs one notification, a false
160
+ * negative sends an email that cannot be unsent. §5.8 makes the confirmation
161
+ * asynchronous precisely so that being generous here is cheap — the owner
162
+ * answers from their phone and the machine is not blocked meanwhile.
163
+ *
164
+ * @param action - What the agent is about to do, in words
165
+ * @returns Whether to ask first, and the phrase that triggered it
166
+ *
167
+ * @example
168
+ * needsConfirmation('click the Send button') // → { required: true, trigger: 'send' }
169
+ */
170
+ export function needsConfirmation(action: string): { required: boolean; trigger?: string } {
171
+ const text = action.toLowerCase();
172
+ for (const phrase of IRREVERSIBLE) {
173
+ // Word boundaries for Latin phrases; CJK has none, so substring is right.
174
+ const pattern = /[a-z]/.test(phrase) ? new RegExp(`\\b${phrase}\\b`) : null;
175
+ if (pattern ? pattern.test(text) : text.includes(phrase)) {
176
+ return { required: true, trigger: phrase };
177
+ }
178
+ }
179
+ return { required: false };
180
+ }
181
+
182
+ /** Everything the runtime remembers about one task. */
183
+ export interface TaskState {
184
+ goal: string;
185
+ progress: SubgoalProgress[];
186
+ budget: TaskBudget;
187
+ startedAt: number;
188
+ stepsUsed: number;
189
+ /** Set while an irreversible action waits on the owner. */
190
+ awaitingConfirmation?: { action: string; since: number };
191
+ }
192
+
193
+ /**
194
+ * Start a task.
195
+ *
196
+ * @param goal - The whole task, in words
197
+ * @param subgoals - The plan
198
+ * @param budget - Caps; defaults are usually right
199
+ * @param now - Clock, injectable for tests
200
+ * @returns Fresh state
201
+ */
202
+ export function beginTask(
203
+ goal: string,
204
+ subgoals: Subgoal[],
205
+ budget: TaskBudget = DEFAULT_TASK_BUDGET,
206
+ now: () => number = Date.now,
207
+ ): TaskState {
208
+ return {
209
+ goal,
210
+ budget,
211
+ startedAt: now(),
212
+ stepsUsed: 0,
213
+ progress: subgoals.map((subgoal, index) => ({
214
+ subgoal,
215
+ state: index === 0 ? 'active' : 'pending',
216
+ stepsUsed: 0,
217
+ recoveries: {},
218
+ })),
219
+ };
220
+ }
221
+
222
+ /** The subgoal being worked on, if any. */
223
+ export function activeSubgoal(state: TaskState): SubgoalProgress | undefined {
224
+ return state.progress.find((p) => p.state === 'active');
225
+ }
226
+
227
+ /**
228
+ * Advance the task by one step's worth of thinking.
229
+ *
230
+ * Called after each action the agent takes. It checks the budget, looks for a
231
+ * surprise, tests the checkpoint, and says what to do next — which is the
232
+ * whole point: the decision to move on is the runtime's, not the agent's.
233
+ *
234
+ * @param state - Mutated in place with progress
235
+ * @param scene - What is on screen now, for surprise detection
236
+ * @param deps - IO for checkpoint evaluation
237
+ * @param now - Clock
238
+ * @returns What the agent should do next
239
+ */
240
+ export async function step(
241
+ state: TaskState,
242
+ scene: Scene,
243
+ deps: CheckpointDeps = {},
244
+ now: () => number = Date.now,
245
+ ): Promise<Directive> {
246
+ const completed = () => state.progress.filter((p) => p.state === 'done').map((p) => p.subgoal.goal);
247
+ const remaining = () => state.progress.filter((p) => p.state !== 'done').map((p) => p.subgoal.goal);
248
+
249
+ if (state.awaitingConfirmation) {
250
+ return {
251
+ kind: 'await-confirmation',
252
+ action: state.awaitingConfirmation.action,
253
+ detail: 'Waiting for the owner to approve. Do nothing else until they answer.',
254
+ };
255
+ }
256
+
257
+ state.stepsUsed += 1;
258
+ const current = activeSubgoal(state);
259
+ if (!current) {
260
+ return { kind: 'done', summary: `All ${state.progress.length} subgoals verified.` };
261
+ }
262
+ current.stepsUsed += 1;
263
+
264
+ // Budget before anything else: an exhausted task should stop, not spend its
265
+ // last step discovering a surprise it has no budget to handle.
266
+ if (state.stepsUsed > state.budget.maxSteps) {
267
+ return {
268
+ kind: 'budget-exhausted',
269
+ reason: `Used ${state.stepsUsed} steps of ${state.budget.maxSteps}.`,
270
+ completed: completed(),
271
+ remaining: remaining(),
272
+ };
273
+ }
274
+ if (now() - state.startedAt > state.budget.maxDurationMs) {
275
+ return {
276
+ kind: 'budget-exhausted',
277
+ reason: `Ran for longer than ${Math.round(state.budget.maxDurationMs / 60000)} minutes.`,
278
+ completed: completed(),
279
+ remaining: remaining(),
280
+ };
281
+ }
282
+
283
+ // A surprise means the screen is not what the plan assumed. Checking the
284
+ // checkpoint now would test the wrong world.
285
+ const surprise = detectSurprise(scene, sceneAppFor(current));
286
+ if (surprise) {
287
+ const kind = surprise.kind;
288
+ current.recoveries[kind] = (current.recoveries[kind] ?? 0) + 1;
289
+ const verdict = shouldRetryAfter(kind, current.recoveries[kind] - 1, surprise.selfRecoverable);
290
+ if (!verdict.retry) {
291
+ current.state = 'stuck';
292
+ current.blockedBy = surprise.detail;
293
+ // The surprise's own instruction is the actionable half ("this is the
294
+ // owner's decision, tell them what is being asked"); dropping it for
295
+ // the generic escalation line would lose exactly the useful part.
296
+ return {
297
+ kind: 'escalate',
298
+ reason: surprise.kind,
299
+ detail: [surprise.detail, surprise.instruction, verdict.escalation].filter(Boolean).join(' '),
300
+ };
301
+ }
302
+ return { kind: 'continue', subgoal: current.subgoal, stepsLeft: stepsLeftFor(current, state), hint: surprise.instruction };
303
+ }
304
+
305
+ // The only thing that completes a subgoal.
306
+ const check = await evaluateCheckpoint(current.subgoal.checkpoint, deps);
307
+ current.lastCheck = check;
308
+
309
+ if (check.passed) {
310
+ current.state = 'done';
311
+ const next = state.progress.find((p) => p.state === 'pending');
312
+ if (!next) {
313
+ return { kind: 'done', summary: `All ${state.progress.length} subgoals verified.` };
314
+ }
315
+ next.state = 'active';
316
+ return { kind: 'advance', completed: current.subgoal, next: next.subgoal };
317
+ }
318
+
319
+ const left = stepsLeftFor(current, state);
320
+ if (left <= 0) {
321
+ current.state = 'stuck';
322
+ current.blockedBy = check.reason ?? 'The checkpoint never held.';
323
+ return {
324
+ kind: 'escalate',
325
+ reason: 'subgoal-stuck',
326
+ detail:
327
+ `"${current.subgoal.goal}" used its whole budget and ${describeCheckpoint(current.subgoal.checkpoint)} is still false. ` +
328
+ `${check.reason ?? ''} Report what you tried rather than continuing — the later subgoals assume this one worked.`.trim(),
329
+ };
330
+ }
331
+
332
+ return {
333
+ kind: 'continue',
334
+ subgoal: current.subgoal,
335
+ stepsLeft: left,
336
+ // The reason is the useful part: "the file is empty" tells the agent the
337
+ // save dialog is still open, where "not done" tells it nothing.
338
+ hint: check.reason,
339
+ };
340
+ }
341
+
342
+ /** Steps this subgoal has left. */
343
+ function stepsLeftFor(progress: SubgoalProgress, state: TaskState): number {
344
+ const cap = progress.subgoal.maxSteps ?? state.budget.defaultSubgoalSteps;
345
+ return cap - progress.stepsUsed;
346
+ }
347
+
348
+ /** The app a subgoal implies, for focus-loss detection. */
349
+ function sceneAppFor(progress: SubgoalProgress): string | undefined {
350
+ const cp = progress.subgoal.checkpoint;
351
+ if (cp.kind === 'app-frontmost') return cp.app;
352
+ if ((cp.kind === 'element-present' || cp.kind === 'element-absent') && cp.app) return cp.app;
353
+ return undefined;
354
+ }
355
+
356
+ /**
357
+ * Hold the task while the owner approves something irreversible.
358
+ *
359
+ * The machine is not blocked — nothing else is running — but the agent is,
360
+ * which is the point: the owner answers from wherever they are and the task
361
+ * picks up where it stopped.
362
+ *
363
+ * @param state - Mutated
364
+ * @param action - What is waiting
365
+ * @param now - Clock
366
+ */
367
+ export function awaitConfirmation(state: TaskState, action: string, now: () => number = Date.now): void {
368
+ state.awaitingConfirmation = { action, since: now() };
369
+ }
370
+
371
+ /**
372
+ * Record the owner's answer.
373
+ *
374
+ * @param state - Mutated
375
+ * @param approved - Their decision
376
+ * @returns What to do next
377
+ */
378
+ export function resolveConfirmation(state: TaskState, approved: boolean): Directive {
379
+ const pending = state.awaitingConfirmation;
380
+ state.awaitingConfirmation = undefined as TaskState['awaitingConfirmation'];
381
+ if (!pending) {
382
+ return { kind: 'escalate', reason: 'no-pending-confirmation', detail: 'Nothing was waiting for approval.' };
383
+ }
384
+ if (approved) {
385
+ const current = activeSubgoal(state);
386
+ return current
387
+ ? { kind: 'continue', subgoal: current.subgoal, stepsLeft: stepsLeftFor(current, state), hint: `Approved: ${pending.action}. Go ahead.` }
388
+ : { kind: 'done', summary: 'Approved, and nothing left to do.' };
389
+ }
390
+ const current = activeSubgoal(state);
391
+ if (current) {
392
+ current.state = 'skipped';
393
+ current.blockedBy = `The owner declined: ${pending.action}`;
394
+ }
395
+ return {
396
+ kind: 'escalate',
397
+ reason: 'declined',
398
+ detail: `The owner declined "${pending.action}". Do not look for another way to do it — stop and report.`,
399
+ };
400
+ }
401
+
402
+ /**
403
+ * A plain-language account of where the task stands.
404
+ *
405
+ * This is what gets reported when the task ends, for whatever reason, so it
406
+ * has to be true rather than encouraging: a stuck subgoal is named as stuck.
407
+ *
408
+ * @param state - The task
409
+ * @returns Lines for the owner
410
+ */
411
+ export function summarize(state: TaskState): string {
412
+ const mark: Record<SubgoalState, string> = {
413
+ done: '✓', stuck: '✗', skipped: '—', active: '→', pending: '·',
414
+ };
415
+ const lines = state.progress.map((p) => {
416
+ const base = `${mark[p.state]} ${p.subgoal.goal}`;
417
+ if (p.state === 'done') return `${base} (verified: ${describeCheckpoint(p.subgoal.checkpoint)})`;
418
+ if (p.blockedBy) return `${base} — ${p.blockedBy}`;
419
+ return base;
420
+ });
421
+ const done = state.progress.filter((p) => p.state === 'done').length;
422
+ return [`${done}/${state.progress.length} verified · ${state.stepsUsed} steps`, ...lines].join('\n');
423
+ }