crewly 1.20.35 → 1.20.47

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (139) hide show
  1. package/config/skills/_common/desktop-guards.sh +485 -0
  2. package/config/skills/_common/desktop-guards.test.sh +242 -0
  3. package/config/skills/_common/desktop-perceive.swift +530 -0
  4. package/config/skills/_common/desktop-presence.swift +343 -0
  5. package/config/skills/_common/lib.sh +6 -0
  6. package/config/skills/agent/_common/desktop-guards.sh +4 -0
  7. package/config/skills/agent/computer-use/SKILL.md +88 -0
  8. package/config/skills/agent/computer-use/execute.sh +249 -3
  9. package/config/skills/agent/core/calendar-create/SKILL.md +10 -0
  10. package/config/skills/agent/core/calendar-create/execute.sh +6 -0
  11. package/config/skills/agent/core/calendar-list/SKILL.md +10 -0
  12. package/config/skills/agent/core/calendar-list/execute.sh +6 -0
  13. package/config/skills/agent/core/docs-read/SKILL.md +10 -0
  14. package/config/skills/agent/core/docs-read/execute.sh +6 -1
  15. package/config/skills/agent/core/docs-write/SKILL.md +10 -0
  16. package/config/skills/agent/core/docs-write/execute.sh +6 -1
  17. package/config/skills/agent/core/drive-read/SKILL.md +10 -0
  18. package/config/skills/agent/core/drive-read/execute.sh +6 -1
  19. package/config/skills/agent/core/drive-search/SKILL.md +10 -0
  20. package/config/skills/agent/core/drive-search/execute.sh +6 -1
  21. package/config/skills/agent/core/drive-upload/SKILL.md +10 -0
  22. package/config/skills/agent/core/drive-upload/execute.sh +6 -1
  23. package/config/skills/agent/core/gmail-read/SKILL.md +10 -0
  24. package/config/skills/agent/core/gmail-read/execute.sh +6 -0
  25. package/config/skills/agent/core/gmail-search/SKILL.md +10 -0
  26. package/config/skills/agent/core/gmail-search/execute.sh +6 -0
  27. package/config/skills/agent/core/gmail-send/SKILL.md +10 -0
  28. package/config/skills/agent/core/gmail-send/execute.sh +6 -0
  29. package/config/skills/agent/core/sheets-read/SKILL.md +10 -0
  30. package/config/skills/agent/core/sheets-read/execute.sh +6 -1
  31. package/config/skills/agent/core/sheets-write/SKILL.md +10 -0
  32. package/config/skills/agent/core/sheets-write/execute.sh +6 -1
  33. package/config/skills/agent/core/slides-create/SKILL.md +10 -0
  34. package/config/skills/agent/core/slides-create/execute.sh +6 -1
  35. package/config/skills/agent/core/slides-read/SKILL.md +10 -0
  36. package/config/skills/agent/core/slides-read/execute.sh +6 -1
  37. package/config/skills/agent/desktop-app-control/SKILL.md +19 -0
  38. package/config/skills/agent/remote-browser/SKILL.md +19 -0
  39. package/config/slack-app-manifest.json +16 -9
  40. package/dist/backend/backend/src/constants.d.ts +18 -4
  41. package/dist/backend/backend/src/constants.d.ts.map +1 -1
  42. package/dist/backend/backend/src/constants.js +16 -4
  43. package/dist/backend/backend/src/constants.js.map +1 -1
  44. package/dist/backend/backend/src/controllers/desktop/desktop.controller.d.ts +105 -0
  45. package/dist/backend/backend/src/controllers/desktop/desktop.controller.d.ts.map +1 -0
  46. package/dist/backend/backend/src/controllers/desktop/desktop.controller.js +278 -0
  47. package/dist/backend/backend/src/controllers/desktop/desktop.controller.js.map +1 -0
  48. package/dist/backend/backend/src/controllers/desktop/desktop.routes.d.ts +21 -0
  49. package/dist/backend/backend/src/controllers/desktop/desktop.routes.d.ts.map +1 -0
  50. package/dist/backend/backend/src/controllers/desktop/desktop.routes.js +31 -0
  51. package/dist/backend/backend/src/controllers/desktop/desktop.routes.js.map +1 -0
  52. package/dist/backend/backend/src/controllers/google/google.controller.d.ts +8 -0
  53. package/dist/backend/backend/src/controllers/google/google.controller.d.ts.map +1 -1
  54. package/dist/backend/backend/src/controllers/google/google.controller.js +137 -37
  55. package/dist/backend/backend/src/controllers/google/google.controller.js.map +1 -1
  56. package/dist/backend/backend/src/controllers/google/google.routes.d.ts +2 -1
  57. package/dist/backend/backend/src/controllers/google/google.routes.d.ts.map +1 -1
  58. package/dist/backend/backend/src/controllers/google/google.routes.js +4 -2
  59. package/dist/backend/backend/src/controllers/google/google.routes.js.map +1 -1
  60. package/dist/backend/backend/src/controllers/slack/slack-error.utils.d.ts +46 -0
  61. package/dist/backend/backend/src/controllers/slack/slack-error.utils.d.ts.map +1 -0
  62. package/dist/backend/backend/src/controllers/slack/slack-error.utils.js +54 -0
  63. package/dist/backend/backend/src/controllers/slack/slack-error.utils.js.map +1 -0
  64. package/dist/backend/backend/src/controllers/slack/slack.controller.d.ts.map +1 -1
  65. package/dist/backend/backend/src/controllers/slack/slack.controller.js +5 -12
  66. package/dist/backend/backend/src/controllers/slack/slack.controller.js.map +1 -1
  67. package/dist/backend/backend/src/routes/api.routes.d.ts.map +1 -1
  68. package/dist/backend/backend/src/routes/api.routes.js +3 -0
  69. package/dist/backend/backend/src/routes/api.routes.js.map +1 -1
  70. package/dist/backend/backend/src/services/cloud/mobile-api-relay.service.d.ts.map +1 -1
  71. package/dist/backend/backend/src/services/cloud/mobile-api-relay.service.js +9 -0
  72. package/dist/backend/backend/src/services/cloud/mobile-api-relay.service.js.map +1 -1
  73. package/dist/backend/backend/src/services/google/google-api.client.d.ts +23 -2
  74. package/dist/backend/backend/src/services/google/google-api.client.d.ts.map +1 -1
  75. package/dist/backend/backend/src/services/google/google-api.client.js +5 -2
  76. package/dist/backend/backend/src/services/google/google-api.client.js.map +1 -1
  77. package/dist/backend/backend/src/services/google/google-workspace-token.service.d.ts +61 -11
  78. package/dist/backend/backend/src/services/google/google-workspace-token.service.d.ts.map +1 -1
  79. package/dist/backend/backend/src/services/google/google-workspace-token.service.js +108 -31
  80. package/dist/backend/backend/src/services/google/google-workspace-token.service.js.map +1 -1
  81. package/dist/backend/backend/src/services/slack/slack-team-channel.service.d.ts +24 -0
  82. package/dist/backend/backend/src/services/slack/slack-team-channel.service.d.ts.map +1 -1
  83. package/dist/backend/backend/src/services/slack/slack-team-channel.service.js +87 -5
  84. package/dist/backend/backend/src/services/slack/slack-team-channel.service.js.map +1 -1
  85. package/dist/backend/backend/src/services/slack/slack.service.d.ts +17 -0
  86. package/dist/backend/backend/src/services/slack/slack.service.d.ts.map +1 -1
  87. package/dist/backend/backend/src/services/slack/slack.service.js +32 -0
  88. package/dist/backend/backend/src/services/slack/slack.service.js.map +1 -1
  89. package/dist/backend/backend/src/types/slack.types.d.ts +10 -0
  90. package/dist/backend/backend/src/types/slack.types.d.ts.map +1 -1
  91. package/dist/backend/backend/src/types/slack.types.js.map +1 -1
  92. package/dist/backend/backend/src/utils/incomplete-turn.utils.d.ts +1 -1
  93. package/dist/backend/backend/src/utils/incomplete-turn.utils.d.ts.map +1 -1
  94. package/dist/backend/backend/src/utils/incomplete-turn.utils.js +4 -0
  95. package/dist/backend/backend/src/utils/incomplete-turn.utils.js.map +1 -1
  96. package/dist/backend/build-info.json +2 -2
  97. package/dist/cli/backend/src/constants.d.ts +18 -4
  98. package/dist/cli/backend/src/constants.d.ts.map +1 -1
  99. package/dist/cli/backend/src/constants.js +16 -4
  100. package/dist/cli/backend/src/constants.js.map +1 -1
  101. package/dist/cli/backend/src/services/slack/slack-team-channel.service.d.ts +24 -0
  102. package/dist/cli/backend/src/services/slack/slack-team-channel.service.d.ts.map +1 -1
  103. package/dist/cli/backend/src/services/slack/slack-team-channel.service.js +87 -5
  104. package/dist/cli/backend/src/services/slack/slack-team-channel.service.js.map +1 -1
  105. package/dist/cli/backend/src/services/slack/slack.service.d.ts +17 -0
  106. package/dist/cli/backend/src/services/slack/slack.service.d.ts.map +1 -1
  107. package/dist/cli/backend/src/services/slack/slack.service.js +32 -0
  108. package/dist/cli/backend/src/services/slack/slack.service.js.map +1 -1
  109. package/dist/cli/backend/src/types/slack.types.d.ts +10 -0
  110. package/dist/cli/backend/src/types/slack.types.d.ts.map +1 -1
  111. package/dist/cli/backend/src/types/slack.types.js.map +1 -1
  112. package/dist/cli/backend/src/utils/incomplete-turn.utils.d.ts +1 -1
  113. package/dist/cli/backend/src/utils/incomplete-turn.utils.d.ts.map +1 -1
  114. package/dist/cli/backend/src/utils/incomplete-turn.utils.js +4 -0
  115. package/dist/cli/backend/src/utils/incomplete-turn.utils.js.map +1 -1
  116. package/frontend/dist/assets/{index-e079a375.js → index-e7785269.js} +267 -267
  117. package/frontend/dist/index.html +1 -1
  118. package/package.json +1 -1
  119. package/packages/crewly-agent/src/eval/desktop/desktop-tasks.test.ts +96 -0
  120. package/packages/crewly-agent/src/eval/desktop/desktop-tasks.ts +226 -0
  121. package/packages/crewly-agent/src/runtime/agent-runner.service.test.ts +10 -1
  122. package/packages/crewly-agent/src/runtime/agent-runner.service.ts +199 -4
  123. package/packages/crewly-agent/src/runtime/computer.tool.test.ts +219 -0
  124. package/packages/crewly-agent/src/runtime/computer.tool.ts +405 -0
  125. package/packages/crewly-agent/src/runtime/desktop-checkpoint.test.ts +136 -0
  126. package/packages/crewly-agent/src/runtime/desktop-checkpoint.ts +231 -0
  127. package/packages/crewly-agent/src/runtime/desktop-recovery.test.ts +100 -0
  128. package/packages/crewly-agent/src/runtime/desktop-recovery.ts +195 -0
  129. package/packages/crewly-agent/src/runtime/desktop-task-runtime.test.ts +251 -0
  130. package/packages/crewly-agent/src/runtime/desktop-task-runtime.ts +423 -0
  131. package/packages/crewly-agent/src/runtime/desktop-task.tool.test.ts +218 -0
  132. package/packages/crewly-agent/src/runtime/desktop-task.tool.ts +343 -0
  133. package/packages/crewly-agent/src/runtime/text-tool-calls.test.ts +144 -0
  134. package/packages/crewly-agent/src/runtime/text-tool-calls.ts +316 -0
  135. package/packages/crewly-agent/src/runtime/text-tool-salvage.test.ts +190 -0
  136. package/packages/crewly-agent/src/runtime/tool-registry.test.ts +17 -0
  137. package/packages/crewly-agent/src/runtime/tool-registry.ts +54 -0
  138. package/packages/crewly-agent/src/runtime/types.ts +16 -1
  139. package/config/skills/agent/vnc-browser/SKILL.md +0 -140
@@ -0,0 +1,136 @@
1
+ /**
2
+ * Tests for checkpoint evaluation.
3
+ *
4
+ * A checkpoint is the only thing that can call a desktop subgoal done, so a
5
+ * checkpoint that passes when it should not is worse than having none: it
6
+ * launders the agent's claim into a verified fact.
7
+ */
8
+
9
+ import { describe, it, expect } from 'vitest';
10
+ import { evaluateCheckpoint, describeCheckpoint, type Checkpoint } from './desktop-checkpoint.js';
11
+
12
+ /** IO over a fake filesystem. */
13
+ function fs(files: Record<string, string>) {
14
+ return {
15
+ readFile: async (p: string) => {
16
+ if (!(p in files)) throw new Error('ENOENT');
17
+ return files[p]!;
18
+ },
19
+ statFile: async (p: string) => {
20
+ if (!(p in files)) throw new Error('ENOENT');
21
+ return { size: files[p]!.length };
22
+ },
23
+ };
24
+ }
25
+
26
+ describe('file checkpoints', () => {
27
+ it('fails an empty file — the signature of a save that never completed', async () => {
28
+ const out = await evaluateCheckpoint({ kind: 'file-exists', path: '/a' }, fs({ '/a': '' }));
29
+ expect(out.passed).toBe(false);
30
+ expect(out.reason).toMatch(/empty/);
31
+ });
32
+
33
+ it('allows an empty file when the task said so', async () => {
34
+ expect((await evaluateCheckpoint({ kind: 'file-exists', path: '/a', allowEmpty: true }, fs({ '/a': '' }))).passed).toBe(true);
35
+ });
36
+
37
+ it('says the file is missing rather than just failing', async () => {
38
+ const out = await evaluateCheckpoint({ kind: 'file-exists', path: '/nope' }, fs({}));
39
+ expect(out.reason).toContain('/nope');
40
+ });
41
+
42
+ it('shows what the file does say when the content is wrong', async () => {
43
+ const out = await evaluateCheckpoint({ kind: 'file-contains', path: '/a', text: 'hello' }, fs({ '/a': 'goodbye' }));
44
+ expect(out.passed).toBe(false);
45
+ expect(out.observed).toBe('goodbye');
46
+ });
47
+
48
+ it('matches a pattern', async () => {
49
+ expect((await evaluateCheckpoint({ kind: 'file-matches', path: '/a', pattern: '^\\d+$' }, fs({ '/a': '42' }))).passed).toBe(true);
50
+ expect((await evaluateCheckpoint({ kind: 'file-matches', path: '/a', pattern: '^\\d+$' }, fs({ '/a': 'x' }))).passed).toBe(false);
51
+ });
52
+
53
+ it('checks the other half of a rename', async () => {
54
+ expect((await evaluateCheckpoint({ kind: 'file-absent', path: '/old' }, fs({}))).passed).toBe(true);
55
+ expect((await evaluateCheckpoint({ kind: 'file-absent', path: '/old' }, fs({ '/old': 'x' }))).passed).toBe(false);
56
+ });
57
+ });
58
+
59
+ describe('screen checkpoints', () => {
60
+ const snapshot = async () => [
61
+ { role: 'AXButton', name: 'Save' },
62
+ { role: 'AXStaticText', name: 'Untitled document' },
63
+ ];
64
+
65
+ it('finds an element by name, and by role when given one', async () => {
66
+ expect((await evaluateCheckpoint({ kind: 'element-present', name: 'Save' }, { snapshot })).passed).toBe(true);
67
+ expect((await evaluateCheckpoint({ kind: 'element-present', name: 'Save', role: 'AXButton' }, { snapshot })).passed).toBe(true);
68
+ expect((await evaluateCheckpoint({ kind: 'element-present', name: 'Save', role: 'AXMenuItem' }, { snapshot })).passed).toBe(false);
69
+ });
70
+
71
+ it('checks a dialog is gone, which is how "closed it" is proved', async () => {
72
+ expect((await evaluateCheckpoint({ kind: 'element-absent', name: 'Save' }, { snapshot })).passed).toBe(false);
73
+ expect((await evaluateCheckpoint({ kind: 'element-absent', name: 'Nothing' }, { snapshot })).passed).toBe(true);
74
+ });
75
+
76
+ it('reads text off the screen for apps with no element tree', async () => {
77
+ const screenText = async () => ['CREWLY EVAL 7734'];
78
+ expect((await evaluateCheckpoint({ kind: 'text-on-screen', text: 'eval 7734' }, { screenText })).passed).toBe(true);
79
+ expect((await evaluateCheckpoint({ kind: 'text-on-screen', text: 'absent' }, { screenText })).passed).toBe(false);
80
+ });
81
+
82
+ it('compares the frontmost app case-insensitively and reports what is actually there', async () => {
83
+ const frontmostApp = async () => 'TextEdit';
84
+ expect((await evaluateCheckpoint({ kind: 'app-frontmost', app: 'textedit' }, { frontmostApp })).passed).toBe(true);
85
+ const out = await evaluateCheckpoint({ kind: 'app-frontmost', app: 'Numbers' }, { frontmostApp });
86
+ expect(out.observed).toBe('TextEdit');
87
+ });
88
+
89
+ it('fails rather than pretends when it cannot see', async () => {
90
+ // No snapshot dependency wired: it must not quietly pass.
91
+ expect((await evaluateCheckpoint({ kind: 'element-present', name: 'Save' }, {})).passed).toBe(false);
92
+ expect((await evaluateCheckpoint({ kind: 'text-on-screen', text: 'x' }, {})).passed).toBe(false);
93
+ expect((await evaluateCheckpoint({ kind: 'app-frontmost', app: 'x' }, {})).passed).toBe(false);
94
+ });
95
+ });
96
+
97
+ describe('shell checkpoints', () => {
98
+ it('passes on exit zero and reports the output on failure', async () => {
99
+ const runShell = async (c: string) =>
100
+ c === 'true' ? { code: 0, stdout: '', stderr: '' } : { code: 1, stdout: '', stderr: 'boom' };
101
+ expect((await evaluateCheckpoint({ kind: 'shell', command: 'true' }, { runShell })).passed).toBe(true);
102
+ const out = await evaluateCheckpoint({ kind: 'shell', command: 'false' }, { runShell });
103
+ expect(out.passed).toBe(false);
104
+ expect(out.observed).toBe('boom');
105
+ });
106
+ });
107
+
108
+ describe('robustness', () => {
109
+ it('turns a thrown check into a failed one, never an exception', async () => {
110
+ const out = await evaluateCheckpoint({ kind: 'file-exists', path: '/a' }, {
111
+ statFile: async () => { throw new Error('disk on fire'); },
112
+ });
113
+ // A broken check must not read as a broken task, but it must not pass.
114
+ expect(out.passed).toBe(false);
115
+ });
116
+ });
117
+
118
+ describe('describeCheckpoint', () => {
119
+ it('reads as a sentence for every kind', () => {
120
+ const all: Checkpoint[] = [
121
+ { kind: 'file-exists', path: '/a' },
122
+ { kind: 'file-contains', path: '/a', text: 'x' },
123
+ { kind: 'file-matches', path: '/a', pattern: 'x' },
124
+ { kind: 'file-absent', path: '/a' },
125
+ { kind: 'app-frontmost', app: 'Finder' },
126
+ { kind: 'element-present', name: 'Save' },
127
+ { kind: 'element-absent', name: 'Save' },
128
+ { kind: 'text-on-screen', text: 'hi' },
129
+ { kind: 'shell', command: 'true' },
130
+ ];
131
+ for (const cp of all) {
132
+ expect(describeCheckpoint(cp).length).toBeGreaterThan(5);
133
+ }
134
+ expect(describeCheckpoint({ kind: 'shell', command: 'x', description: 'the build passes' })).toBe('the build passes');
135
+ });
136
+ });
@@ -0,0 +1,231 @@
1
+ /**
2
+ * Checkpoints — how a desktop subgoal is judged done.
3
+ *
4
+ * Phase 4 of docs/research/computer-use-capability-assessment.md. The failure
5
+ * this exists for is the third one in §5.1: the agent says "I saved it" while
6
+ * the save dialog is still open. Its own account of what happened is exactly
7
+ * the thing that cannot be trusted, so a subgoal is complete when something
8
+ * outside the agent says so — a file on disk, an element on screen, a command
9
+ * that exits zero.
10
+ *
11
+ * Checkpoints are deliberately narrow. Each one is a small, decidable
12
+ * question with a yes or no answer and a reason when it is no, because a
13
+ * model that is told "the file has 0 bytes" can act, where one told "task
14
+ * failed" can only guess.
15
+ *
16
+ * @module runtime/desktop-checkpoint
17
+ */
18
+
19
+ import { spawn } from 'child_process';
20
+ import { promises as fs } from 'fs';
21
+
22
+ /** How long a shell checkpoint may take before it counts as failed. */
23
+ const SHELL_TIMEOUT_MS = 15_000;
24
+
25
+ /**
26
+ * A decidable statement about the world after a subgoal.
27
+ *
28
+ * `shell` is the escape hatch, but the named kinds are preferred: they give a
29
+ * usable reason on failure, where a shell command can only give an exit code.
30
+ */
31
+ export type Checkpoint =
32
+ /** A file exists and, unless `allowEmpty`, has content. */
33
+ | { kind: 'file-exists'; path: string; allowEmpty?: boolean }
34
+ /** A file exists and contains this text. */
35
+ | { kind: 'file-contains'; path: string; text: string }
36
+ /** A file exists and matches this pattern. */
37
+ | { kind: 'file-matches'; path: string; pattern: string }
38
+ /** A file is gone (a rename's other half, a cleanup). */
39
+ | { kind: 'file-absent'; path: string }
40
+ /** This application is in front. */
41
+ | { kind: 'app-frontmost'; app: string }
42
+ /** An element with this name (and optionally role) is on screen. */
43
+ | { kind: 'element-present'; name: string; role?: string; app?: string }
44
+ /** No element with this name is on screen — a dialog that should be gone. */
45
+ | { kind: 'element-absent'; name: string; app?: string }
46
+ /** This text is readable on screen. */
47
+ | { kind: 'text-on-screen'; text: string }
48
+ /** This command exits zero. */
49
+ | { kind: 'shell'; command: string; description?: string };
50
+
51
+ /** The answer, with enough detail to act on. */
52
+ export interface CheckpointResult {
53
+ passed: boolean;
54
+ /** Why not, phrased for the model to act on. Absent when it passed. */
55
+ reason?: string;
56
+ /** What was actually observed, when that helps more than prose. */
57
+ observed?: string;
58
+ }
59
+
60
+ /** Injectable IO, so the evaluator is testable without a desktop. */
61
+ export interface CheckpointDeps {
62
+ readFile?: (path: string) => Promise<string>;
63
+ statFile?: (path: string) => Promise<{ size: number }>;
64
+ runShell?: (command: string) => Promise<{ code: number; stdout: string; stderr: string }>;
65
+ /** Elements currently on screen, from a desktop snapshot. */
66
+ snapshot?: (app?: string) => Promise<Array<{ role: string; name?: string }>>;
67
+ /** Text currently on screen, from OCR. */
68
+ screenText?: () => Promise<string[]>;
69
+ /** The frontmost application's name. */
70
+ frontmostApp?: () => Promise<string>;
71
+ }
72
+
73
+ /**
74
+ * Run a shell command, capturing its outcome rather than throwing.
75
+ *
76
+ * @param command - The command
77
+ * @returns Exit code and output
78
+ */
79
+ async function defaultShell(command: string): Promise<{ code: number; stdout: string; stderr: string }> {
80
+ return new Promise((resolve) => {
81
+ const child = spawn('bash', ['-c', command], { stdio: ['ignore', 'pipe', 'pipe'] });
82
+ let stdout = '';
83
+ let stderr = '';
84
+ const timer = setTimeout(() => {
85
+ child.kill('SIGKILL');
86
+ resolve({ code: 124, stdout, stderr: `timed out after ${SHELL_TIMEOUT_MS}ms` });
87
+ }, SHELL_TIMEOUT_MS);
88
+ child.stdout.on('data', (c) => { stdout += String(c); });
89
+ child.stderr.on('data', (c) => { stderr += String(c); });
90
+ child.on('error', (err) => { clearTimeout(timer); resolve({ code: 127, stdout, stderr: err.message }); });
91
+ child.on('close', (code) => { clearTimeout(timer); resolve({ code: code ?? 1, stdout, stderr }); });
92
+ });
93
+ }
94
+
95
+ /**
96
+ * Describe a checkpoint in words, for a plan the owner might read.
97
+ *
98
+ * @param checkpoint - The checkpoint
99
+ * @returns One line
100
+ *
101
+ * @example
102
+ * describeCheckpoint({ kind: 'file-contains', path: '/tmp/a.txt', text: 'ok' })
103
+ * // → '/tmp/a.txt contains "ok"'
104
+ */
105
+ export function describeCheckpoint(checkpoint: Checkpoint): string {
106
+ switch (checkpoint.kind) {
107
+ case 'file-exists': return `${checkpoint.path} exists${checkpoint.allowEmpty ? '' : ' and is not empty'}`;
108
+ case 'file-contains': return `${checkpoint.path} contains "${checkpoint.text}"`;
109
+ case 'file-matches': return `${checkpoint.path} matches /${checkpoint.pattern}/`;
110
+ case 'file-absent': return `${checkpoint.path} is gone`;
111
+ case 'app-frontmost': return `${checkpoint.app} is in front`;
112
+ case 'element-present': return `"${checkpoint.name}" is on screen`;
113
+ case 'element-absent': return `"${checkpoint.name}" is no longer on screen`;
114
+ case 'text-on-screen': return `"${checkpoint.text}" is readable on screen`;
115
+ case 'shell': return checkpoint.description ?? `\`${checkpoint.command}\` succeeds`;
116
+ }
117
+ }
118
+
119
+ /**
120
+ * Decide whether a checkpoint holds.
121
+ *
122
+ * Never throws: an unreadable file or a crashed command is a failed
123
+ * checkpoint with a reason, not an exception for the caller to interpret.
124
+ *
125
+ * @param checkpoint - What to check
126
+ * @param deps - Injected IO
127
+ * @returns Whether it holds, and why not when it does not
128
+ *
129
+ * @example
130
+ * await evaluateCheckpoint({ kind: 'file-contains', path: '/tmp/a', text: 'hi' })
131
+ * // → { passed: false, reason: '/tmp/a exists but does not contain "hi"', observed: '…' }
132
+ */
133
+ export async function evaluateCheckpoint(
134
+ checkpoint: Checkpoint,
135
+ deps: CheckpointDeps = {},
136
+ ): Promise<CheckpointResult> {
137
+ const readFile = deps.readFile ?? ((p: string) => fs.readFile(p, 'utf8'));
138
+ const statFile = deps.statFile ?? (async (p: string) => ({ size: (await fs.stat(p)).size }));
139
+ const runShell = deps.runShell ?? defaultShell;
140
+
141
+ try {
142
+ switch (checkpoint.kind) {
143
+ case 'file-exists': {
144
+ const stat = await statFile(checkpoint.path).catch(() => null);
145
+ if (!stat) return { passed: false, reason: `${checkpoint.path} does not exist.` };
146
+ if (!checkpoint.allowEmpty && stat.size === 0) {
147
+ // An empty file is the signature of a save that opened a dialog and
148
+ // never completed — worth calling out rather than passing.
149
+ return { passed: false, reason: `${checkpoint.path} exists but is empty — the write did not complete.` };
150
+ }
151
+ return { passed: true };
152
+ }
153
+
154
+ case 'file-absent': {
155
+ const stat = await statFile(checkpoint.path).catch(() => null);
156
+ return stat
157
+ ? { passed: false, reason: `${checkpoint.path} is still there.` }
158
+ : { passed: true };
159
+ }
160
+
161
+ case 'file-contains':
162
+ case 'file-matches': {
163
+ const content = await readFile(checkpoint.path).catch(() => null);
164
+ if (content === null) return { passed: false, reason: `${checkpoint.path} does not exist or cannot be read.` };
165
+ const hit = checkpoint.kind === 'file-contains'
166
+ ? content.includes(checkpoint.text)
167
+ : new RegExp(checkpoint.pattern).test(content);
168
+ if (hit) return { passed: true };
169
+ const wanted = checkpoint.kind === 'file-contains' ? `"${checkpoint.text}"` : `/${checkpoint.pattern}/`;
170
+ return {
171
+ passed: false,
172
+ reason: `${checkpoint.path} exists but does not match ${wanted}.`,
173
+ observed: content.slice(0, 200),
174
+ };
175
+ }
176
+
177
+ case 'app-frontmost': {
178
+ if (!deps.frontmostApp) return { passed: false, reason: 'Cannot see which application is in front.' };
179
+ const front = await deps.frontmostApp();
180
+ return front.toLowerCase() === checkpoint.app.toLowerCase()
181
+ ? { passed: true }
182
+ : { passed: false, reason: `${checkpoint.app} is not in front.`, observed: front };
183
+ }
184
+
185
+ case 'element-present':
186
+ case 'element-absent': {
187
+ if (!deps.snapshot) return { passed: false, reason: 'Cannot read the elements on screen.' };
188
+ const elements = await deps.snapshot(checkpoint.app);
189
+ const wanted = checkpoint.name.toLowerCase();
190
+ const found = elements.find((e) => {
191
+ if (!e.name || !e.name.toLowerCase().includes(wanted)) return false;
192
+ return checkpoint.kind === 'element-present' && checkpoint.role ? e.role === checkpoint.role : true;
193
+ });
194
+ if (checkpoint.kind === 'element-present') {
195
+ return found
196
+ ? { passed: true }
197
+ : { passed: false, reason: `Nothing called "${checkpoint.name}" is on screen.` };
198
+ }
199
+ return found
200
+ ? { passed: false, reason: `"${checkpoint.name}" is still on screen.`, observed: found.role }
201
+ : { passed: true };
202
+ }
203
+
204
+ case 'text-on-screen': {
205
+ if (!deps.screenText) return { passed: false, reason: 'Cannot read the text on screen.' };
206
+ const lines = await deps.screenText();
207
+ const wanted = checkpoint.text.toLowerCase();
208
+ return lines.some((l) => l.toLowerCase().includes(wanted))
209
+ ? { passed: true }
210
+ : { passed: false, reason: `"${checkpoint.text}" is not readable on screen.` };
211
+ }
212
+
213
+ case 'shell': {
214
+ const { code, stdout, stderr } = await runShell(checkpoint.command);
215
+ if (code === 0) return { passed: true };
216
+ return {
217
+ passed: false,
218
+ reason: `${describeCheckpoint(checkpoint)} — exited ${code}.`,
219
+ observed: (stderr || stdout).slice(0, 200),
220
+ };
221
+ }
222
+ }
223
+ } catch (err) {
224
+ // A checkpoint that throws is a checkpoint that did not pass. Turning it
225
+ // into an exception would let a broken check read as a broken task.
226
+ return {
227
+ passed: false,
228
+ reason: `Could not check: ${err instanceof Error ? err.message : String(err)}`,
229
+ };
230
+ }
231
+ }
@@ -0,0 +1,100 @@
1
+ /**
2
+ * Tests for surprise detection.
3
+ *
4
+ * The failure being guarded against: a dialog appears at step 15, the agent
5
+ * does not notice, and the next twenty steps act on a world that no longer
6
+ * exists. Each one looks fine on its own.
7
+ */
8
+
9
+ import { describe, it, expect } from 'vitest';
10
+ import { detectSurprise, shouldRetryAfter, type Scene } from './desktop-recovery.js';
11
+
12
+ describe('detectSurprise', () => {
13
+ it('sees nothing wrong with an ordinary window', () => {
14
+ expect(detectSurprise({ app: 'TextEdit', elements: [{ role: 'AXButton', name: 'Bold' }] })).toBeNull();
15
+ });
16
+
17
+ it('spots a modal and lists the buttons so the agent can choose', () => {
18
+ const out = detectSurprise({
19
+ elements: [
20
+ { role: 'AXSheet', name: 'Save changes?' },
21
+ { role: 'AXButton', name: 'Save' },
22
+ { role: 'AXButton', name: "Don't Save" },
23
+ ],
24
+ });
25
+ expect(out?.kind).toBe('modal-dialog');
26
+ expect(out?.instruction).toContain("Don't Save");
27
+ // The most common way to get this wrong is to save when nobody asked.
28
+ expect(out?.instruction).toMatch(/if the task did not ask to save, do not save/i);
29
+ expect(out?.selfRecoverable).toBe(true);
30
+ });
31
+
32
+ it('tells the agent its refs are stale after a dialog', () => {
33
+ const out = detectSurprise({ elements: [{ role: 'AXDialog', name: 'Export' }] });
34
+ expect(out?.instruction).toMatch(/fresh snapshot/);
35
+ });
36
+
37
+ it('treats a permission prompt as the owner\'s decision, not a dialog to answer', () => {
38
+ const out = detectSurprise({
39
+ elements: [
40
+ { role: 'AXSheet', name: 'Terminal would like to access your Documents folder' },
41
+ { role: 'AXButton', name: 'Allow' },
42
+ ],
43
+ });
44
+ expect(out?.kind).toBe('permission-prompt');
45
+ expect(out?.selfRecoverable).toBe(false);
46
+ });
47
+
48
+ it('stops at a sign-in wall instead of typing credentials', () => {
49
+ for (const name of ['Sign in to continue', 'Enter your password', '请输入验证码']) {
50
+ const out = detectSurprise({ elements: [{ role: 'AXStaticText', name }] });
51
+ expect(out?.kind, name).toBe('login-required');
52
+ expect(out?.selfRecoverable).toBe(false);
53
+ }
54
+ });
55
+
56
+ it('reads a locked screen off the last failure, and calls it unrecoverable', () => {
57
+ const out = detectSurprise({ elements: [], lastFailure: { reason: 'screen_locked' } });
58
+ expect(out?.kind).toBe('screen-locked');
59
+ expect(out?.selfRecoverable).toBe(false);
60
+ });
61
+
62
+ it('warns not to repeat a timed-out action blindly — it may have gone through', () => {
63
+ const out = detectSurprise({ elements: [], lastFailure: { reason: 'timeout' } });
64
+ expect(out?.kind).toBe('app-not-responding');
65
+ expect(out?.instruction).toMatch(/may have gone through/);
66
+ });
67
+
68
+ it('notices focus moving away from the app the plan is about', () => {
69
+ expect(detectSurprise({ app: 'Safari', elements: [] }, 'TextEdit')?.kind).toBe('focus-lost');
70
+ expect(detectSurprise({ app: 'TextEdit', elements: [] }, 'textedit')).toBeNull();
71
+ });
72
+
73
+ it('prefers the specific diagnosis when two apply', () => {
74
+ // A permission prompt is also a modal; the specific instruction is the
75
+ // useful one.
76
+ const out = detectSurprise({
77
+ elements: [
78
+ { role: 'AXSheet', name: 'Crewly would like to access your Calendar' },
79
+ { role: 'AXButton', name: 'Allow' },
80
+ ],
81
+ });
82
+ expect(out?.kind).toBe('permission-prompt');
83
+ });
84
+ });
85
+
86
+ describe('shouldRetryAfter', () => {
87
+ it('never retries something that needs a person', () => {
88
+ const out = shouldRetryAfter('login-required', 0, false);
89
+ expect(out.retry).toBe(false);
90
+ expect(out.escalation).toMatch(/needs a person/);
91
+ });
92
+
93
+ it('allows three goes at a recoverable surprise, then stops', () => {
94
+ expect(shouldRetryAfter('modal-dialog', 0, true).retry).toBe(true);
95
+ expect(shouldRetryAfter('modal-dialog', 2, true).retry).toBe(true);
96
+ const out = shouldRetryAfter('modal-dialog', 3, true);
97
+ expect(out.retry).toBe(false);
98
+ expect(out.escalation).toMatch(/will not be different/);
99
+ });
100
+ });
@@ -0,0 +1,195 @@
1
+ /**
2
+ * Surprises, and what to do about them.
3
+ *
4
+ * Phase 4 of docs/research/computer-use-capability-assessment.md, §5.7. The
5
+ * second failure in §5.1 is the expensive one: at step 15 a dialog appears,
6
+ * the agent does not notice, and the next twenty steps act on a state that no
7
+ * longer exists. Every one of them looks fine in isolation.
8
+ *
9
+ * The surprises are not open-ended. A desktop throws the same handful over
10
+ * and over — a modal sheet, a permission prompt, a beachballing app, focus
11
+ * landing somewhere else, a login wall. Naming them means the agent gets a
12
+ * specific instruction instead of having to work out from a screenshot that
13
+ * something is wrong at all.
14
+ *
15
+ * @module runtime/desktop-recovery
16
+ */
17
+
18
+ /** What went unexpectedly wrong. */
19
+ export type SurpriseKind =
20
+ /** A sheet or modal is waiting for an answer. */
21
+ | 'modal-dialog'
22
+ /** macOS is asking the user to allow something. */
23
+ | 'permission-prompt'
24
+ /** A sign-in wall. */
25
+ | 'login-required'
26
+ /** The app stopped responding. */
27
+ | 'app-not-responding'
28
+ /** Another application took the front. */
29
+ | 'focus-lost'
30
+ /** The screen was locked mid-task. */
31
+ | 'screen-locked';
32
+
33
+ /** A surprise, and the instruction that goes with it. */
34
+ export interface Surprise {
35
+ kind: SurpriseKind;
36
+ /** What was seen, for the log and for the agent. */
37
+ detail: string;
38
+ /** What the agent should do next, in one instruction. */
39
+ instruction: string;
40
+ /**
41
+ * Whether the agent can deal with this itself. False means a person has to
42
+ * — there is no point spending the budget discovering that.
43
+ */
44
+ selfRecoverable: boolean;
45
+ }
46
+
47
+ /** An element as a snapshot reports it. */
48
+ export interface SceneElement {
49
+ role: string;
50
+ name?: string;
51
+ subrole?: string;
52
+ }
53
+
54
+ /** Enough of the world to spot a surprise in. */
55
+ export interface Scene {
56
+ app?: string;
57
+ elements: SceneElement[];
58
+ /** Whatever the last action answered, when it failed. */
59
+ lastFailure?: { reason?: string; message?: string };
60
+ }
61
+
62
+ /** Button labels that mean "a modal is waiting". */
63
+ const MODAL_ROLES = new Set(['AXSheet', 'AXDialog']);
64
+
65
+ /** Words that mark a sign-in wall rather than an ordinary form. */
66
+ const LOGIN_WORDS = ['sign in', 'log in', 'login', 'password', 'two-factor', 'verification code', '验证码', '登录'];
67
+
68
+ /** Words macOS uses when it wants the user to allow something. */
69
+ const PERMISSION_WORDS = ['would like to access', 'wants access', 'allow', 'grant access', '访问'];
70
+
71
+ /**
72
+ * Look at a scene and name what is wrong with it, if anything.
73
+ *
74
+ * Order matters: the checks run from most specific to least, because a
75
+ * permission prompt *is* a modal dialog and the specific instruction is the
76
+ * useful one.
77
+ *
78
+ * @param scene - What is on screen, plus the last failure if there was one
79
+ * @param expectedApp - The app the agent believes it is working in
80
+ * @returns The surprise, or null when the scene looks ordinary
81
+ *
82
+ * @example
83
+ * detectSurprise({ app: 'TextEdit', elements: [{ role: 'AXSheet', name: 'Save changes?' }] })
84
+ * // → { kind: 'modal-dialog', instruction: 'Answer the dialog…' }
85
+ */
86
+ export function detectSurprise(scene: Scene, expectedApp?: string): Surprise | null {
87
+ // A rail already said what was wrong; it outranks anything inferred.
88
+ const failure = scene.lastFailure?.reason;
89
+ if (failure === 'screen_locked') {
90
+ return {
91
+ kind: 'screen-locked',
92
+ detail: 'The screen was locked.',
93
+ instruction: 'Stop and report that the screen is locked. It cannot be unlocked from here.',
94
+ selfRecoverable: false,
95
+ };
96
+ }
97
+
98
+ const named = scene.elements.filter((e) => e.name);
99
+ const textOf = (e: SceneElement) => (e.name ?? '').toLowerCase();
100
+
101
+ // A permission prompt is a modal, so it has to be checked first.
102
+ const permission = named.find((e) => PERMISSION_WORDS.some((w) => textOf(e).includes(w)));
103
+ if (permission && named.some((e) => MODAL_ROLES.has(e.role))) {
104
+ return {
105
+ kind: 'permission-prompt',
106
+ detail: `macOS is asking: "${permission.name}"`,
107
+ instruction:
108
+ 'A macOS permission prompt is open. Do not answer it — granting access on the owner\'s behalf is their decision. ' +
109
+ 'Stop and tell them what is being asked for.',
110
+ selfRecoverable: false,
111
+ };
112
+ }
113
+
114
+ const login = named.find((e) => LOGIN_WORDS.some((w) => textOf(e).includes(w)));
115
+ if (login) {
116
+ return {
117
+ kind: 'login-required',
118
+ detail: `A sign-in is being asked for: "${login.name}"`,
119
+ instruction:
120
+ 'This needs credentials, which desktop control never enters. Stop and ask the owner to sign in, then continue.',
121
+ selfRecoverable: false,
122
+ };
123
+ }
124
+
125
+ const modal = scene.elements.find((e) => MODAL_ROLES.has(e.role));
126
+ if (modal) {
127
+ const buttons = named.filter((e) => e.role === 'AXButton').map((e) => e.name!);
128
+ return {
129
+ kind: 'modal-dialog',
130
+ detail: `A dialog is open${modal.name ? `: "${modal.name}"` : ''}.`,
131
+ instruction:
132
+ `A dialog is waiting for an answer and nothing else will work until it is dealt with. ` +
133
+ (buttons.length
134
+ ? `Its buttons are: ${buttons.join(', ')}. Choose the one that matches the task — if the task did not ask to save, do not save. `
135
+ : 'Take a snapshot to see its buttons. ') +
136
+ 'Then take a fresh snapshot: the window behind it has probably changed.',
137
+ selfRecoverable: true,
138
+ };
139
+ }
140
+
141
+ if (scene.lastFailure?.reason === 'timeout') {
142
+ return {
143
+ kind: 'app-not-responding',
144
+ detail: 'The last action timed out.',
145
+ instruction:
146
+ 'The application is not responding. Wait for the screen to settle, then take a fresh snapshot before trying again. ' +
147
+ 'Do not repeat the action blindly — it may have gone through.',
148
+ selfRecoverable: true,
149
+ };
150
+ }
151
+
152
+ if (expectedApp && scene.app && scene.app.toLowerCase() !== expectedApp.toLowerCase()) {
153
+ return {
154
+ kind: 'focus-lost',
155
+ detail: `${scene.app} is in front, not ${expectedApp}.`,
156
+ instruction:
157
+ `Focus moved to ${scene.app}. Bring ${expectedApp} back to the front and take a fresh snapshot — ` +
158
+ 'any refs from before are stale.',
159
+ selfRecoverable: true,
160
+ };
161
+ }
162
+
163
+ return null;
164
+ }
165
+
166
+ /**
167
+ * Whether to keep trying after a surprise.
168
+ *
169
+ * Three attempts at the same kind is the limit. A desktop surprise that
170
+ * survives three goes is not one more click away — it is something the agent
171
+ * has misunderstood, and the cheapest next move is to say so.
172
+ *
173
+ * @param kind - What went wrong
174
+ * @param alreadyTried - How many times this kind has been handled in this subgoal
175
+ * @param selfRecoverable - From the surprise
176
+ * @returns Whether to try again, and what to say when not
177
+ */
178
+ export function shouldRetryAfter(
179
+ kind: SurpriseKind,
180
+ alreadyTried: number,
181
+ selfRecoverable: boolean,
182
+ ): { retry: boolean; escalation?: string } {
183
+ if (!selfRecoverable) {
184
+ return { retry: false, escalation: `${kind} needs a person — stop and report it rather than working around it.` };
185
+ }
186
+ if (alreadyTried >= 3) {
187
+ return {
188
+ retry: false,
189
+ escalation:
190
+ `The same problem (${kind}) has come back three times. Stop and report what you tried and what you see — ` +
191
+ 'a fourth attempt will not be different.',
192
+ };
193
+ }
194
+ return { retry: true };
195
+ }