crewly 1.20.35 → 1.20.47
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/config/skills/_common/desktop-guards.sh +485 -0
- package/config/skills/_common/desktop-guards.test.sh +242 -0
- package/config/skills/_common/desktop-perceive.swift +530 -0
- package/config/skills/_common/desktop-presence.swift +343 -0
- package/config/skills/_common/lib.sh +6 -0
- package/config/skills/agent/_common/desktop-guards.sh +4 -0
- package/config/skills/agent/computer-use/SKILL.md +88 -0
- package/config/skills/agent/computer-use/execute.sh +249 -3
- package/config/skills/agent/core/calendar-create/SKILL.md +10 -0
- package/config/skills/agent/core/calendar-create/execute.sh +6 -0
- package/config/skills/agent/core/calendar-list/SKILL.md +10 -0
- package/config/skills/agent/core/calendar-list/execute.sh +6 -0
- package/config/skills/agent/core/docs-read/SKILL.md +10 -0
- package/config/skills/agent/core/docs-read/execute.sh +6 -1
- package/config/skills/agent/core/docs-write/SKILL.md +10 -0
- package/config/skills/agent/core/docs-write/execute.sh +6 -1
- package/config/skills/agent/core/drive-read/SKILL.md +10 -0
- package/config/skills/agent/core/drive-read/execute.sh +6 -1
- package/config/skills/agent/core/drive-search/SKILL.md +10 -0
- package/config/skills/agent/core/drive-search/execute.sh +6 -1
- package/config/skills/agent/core/drive-upload/SKILL.md +10 -0
- package/config/skills/agent/core/drive-upload/execute.sh +6 -1
- package/config/skills/agent/core/gmail-read/SKILL.md +10 -0
- package/config/skills/agent/core/gmail-read/execute.sh +6 -0
- package/config/skills/agent/core/gmail-search/SKILL.md +10 -0
- package/config/skills/agent/core/gmail-search/execute.sh +6 -0
- package/config/skills/agent/core/gmail-send/SKILL.md +10 -0
- package/config/skills/agent/core/gmail-send/execute.sh +6 -0
- package/config/skills/agent/core/sheets-read/SKILL.md +10 -0
- package/config/skills/agent/core/sheets-read/execute.sh +6 -1
- package/config/skills/agent/core/sheets-write/SKILL.md +10 -0
- package/config/skills/agent/core/sheets-write/execute.sh +6 -1
- package/config/skills/agent/core/slides-create/SKILL.md +10 -0
- package/config/skills/agent/core/slides-create/execute.sh +6 -1
- package/config/skills/agent/core/slides-read/SKILL.md +10 -0
- package/config/skills/agent/core/slides-read/execute.sh +6 -1
- package/config/skills/agent/desktop-app-control/SKILL.md +19 -0
- package/config/skills/agent/remote-browser/SKILL.md +19 -0
- package/config/slack-app-manifest.json +16 -9
- package/dist/backend/backend/src/constants.d.ts +18 -4
- package/dist/backend/backend/src/constants.d.ts.map +1 -1
- package/dist/backend/backend/src/constants.js +16 -4
- package/dist/backend/backend/src/constants.js.map +1 -1
- package/dist/backend/backend/src/controllers/desktop/desktop.controller.d.ts +105 -0
- package/dist/backend/backend/src/controllers/desktop/desktop.controller.d.ts.map +1 -0
- package/dist/backend/backend/src/controllers/desktop/desktop.controller.js +278 -0
- package/dist/backend/backend/src/controllers/desktop/desktop.controller.js.map +1 -0
- package/dist/backend/backend/src/controllers/desktop/desktop.routes.d.ts +21 -0
- package/dist/backend/backend/src/controllers/desktop/desktop.routes.d.ts.map +1 -0
- package/dist/backend/backend/src/controllers/desktop/desktop.routes.js +31 -0
- package/dist/backend/backend/src/controllers/desktop/desktop.routes.js.map +1 -0
- package/dist/backend/backend/src/controllers/google/google.controller.d.ts +8 -0
- package/dist/backend/backend/src/controllers/google/google.controller.d.ts.map +1 -1
- package/dist/backend/backend/src/controllers/google/google.controller.js +137 -37
- package/dist/backend/backend/src/controllers/google/google.controller.js.map +1 -1
- package/dist/backend/backend/src/controllers/google/google.routes.d.ts +2 -1
- package/dist/backend/backend/src/controllers/google/google.routes.d.ts.map +1 -1
- package/dist/backend/backend/src/controllers/google/google.routes.js +4 -2
- package/dist/backend/backend/src/controllers/google/google.routes.js.map +1 -1
- package/dist/backend/backend/src/controllers/slack/slack-error.utils.d.ts +46 -0
- package/dist/backend/backend/src/controllers/slack/slack-error.utils.d.ts.map +1 -0
- package/dist/backend/backend/src/controllers/slack/slack-error.utils.js +54 -0
- package/dist/backend/backend/src/controllers/slack/slack-error.utils.js.map +1 -0
- package/dist/backend/backend/src/controllers/slack/slack.controller.d.ts.map +1 -1
- package/dist/backend/backend/src/controllers/slack/slack.controller.js +5 -12
- package/dist/backend/backend/src/controllers/slack/slack.controller.js.map +1 -1
- package/dist/backend/backend/src/routes/api.routes.d.ts.map +1 -1
- package/dist/backend/backend/src/routes/api.routes.js +3 -0
- package/dist/backend/backend/src/routes/api.routes.js.map +1 -1
- package/dist/backend/backend/src/services/cloud/mobile-api-relay.service.d.ts.map +1 -1
- package/dist/backend/backend/src/services/cloud/mobile-api-relay.service.js +9 -0
- package/dist/backend/backend/src/services/cloud/mobile-api-relay.service.js.map +1 -1
- package/dist/backend/backend/src/services/google/google-api.client.d.ts +23 -2
- package/dist/backend/backend/src/services/google/google-api.client.d.ts.map +1 -1
- package/dist/backend/backend/src/services/google/google-api.client.js +5 -2
- package/dist/backend/backend/src/services/google/google-api.client.js.map +1 -1
- package/dist/backend/backend/src/services/google/google-workspace-token.service.d.ts +61 -11
- package/dist/backend/backend/src/services/google/google-workspace-token.service.d.ts.map +1 -1
- package/dist/backend/backend/src/services/google/google-workspace-token.service.js +108 -31
- package/dist/backend/backend/src/services/google/google-workspace-token.service.js.map +1 -1
- package/dist/backend/backend/src/services/slack/slack-team-channel.service.d.ts +24 -0
- package/dist/backend/backend/src/services/slack/slack-team-channel.service.d.ts.map +1 -1
- package/dist/backend/backend/src/services/slack/slack-team-channel.service.js +87 -5
- package/dist/backend/backend/src/services/slack/slack-team-channel.service.js.map +1 -1
- package/dist/backend/backend/src/services/slack/slack.service.d.ts +17 -0
- package/dist/backend/backend/src/services/slack/slack.service.d.ts.map +1 -1
- package/dist/backend/backend/src/services/slack/slack.service.js +32 -0
- package/dist/backend/backend/src/services/slack/slack.service.js.map +1 -1
- package/dist/backend/backend/src/types/slack.types.d.ts +10 -0
- package/dist/backend/backend/src/types/slack.types.d.ts.map +1 -1
- package/dist/backend/backend/src/types/slack.types.js.map +1 -1
- package/dist/backend/backend/src/utils/incomplete-turn.utils.d.ts +1 -1
- package/dist/backend/backend/src/utils/incomplete-turn.utils.d.ts.map +1 -1
- package/dist/backend/backend/src/utils/incomplete-turn.utils.js +4 -0
- package/dist/backend/backend/src/utils/incomplete-turn.utils.js.map +1 -1
- package/dist/backend/build-info.json +2 -2
- package/dist/cli/backend/src/constants.d.ts +18 -4
- package/dist/cli/backend/src/constants.d.ts.map +1 -1
- package/dist/cli/backend/src/constants.js +16 -4
- package/dist/cli/backend/src/constants.js.map +1 -1
- package/dist/cli/backend/src/services/slack/slack-team-channel.service.d.ts +24 -0
- package/dist/cli/backend/src/services/slack/slack-team-channel.service.d.ts.map +1 -1
- package/dist/cli/backend/src/services/slack/slack-team-channel.service.js +87 -5
- package/dist/cli/backend/src/services/slack/slack-team-channel.service.js.map +1 -1
- package/dist/cli/backend/src/services/slack/slack.service.d.ts +17 -0
- package/dist/cli/backend/src/services/slack/slack.service.d.ts.map +1 -1
- package/dist/cli/backend/src/services/slack/slack.service.js +32 -0
- package/dist/cli/backend/src/services/slack/slack.service.js.map +1 -1
- package/dist/cli/backend/src/types/slack.types.d.ts +10 -0
- package/dist/cli/backend/src/types/slack.types.d.ts.map +1 -1
- package/dist/cli/backend/src/types/slack.types.js.map +1 -1
- package/dist/cli/backend/src/utils/incomplete-turn.utils.d.ts +1 -1
- package/dist/cli/backend/src/utils/incomplete-turn.utils.d.ts.map +1 -1
- package/dist/cli/backend/src/utils/incomplete-turn.utils.js +4 -0
- package/dist/cli/backend/src/utils/incomplete-turn.utils.js.map +1 -1
- package/frontend/dist/assets/{index-e079a375.js → index-e7785269.js} +267 -267
- package/frontend/dist/index.html +1 -1
- package/package.json +1 -1
- package/packages/crewly-agent/src/eval/desktop/desktop-tasks.test.ts +96 -0
- package/packages/crewly-agent/src/eval/desktop/desktop-tasks.ts +226 -0
- package/packages/crewly-agent/src/runtime/agent-runner.service.test.ts +10 -1
- package/packages/crewly-agent/src/runtime/agent-runner.service.ts +199 -4
- package/packages/crewly-agent/src/runtime/computer.tool.test.ts +219 -0
- package/packages/crewly-agent/src/runtime/computer.tool.ts +405 -0
- package/packages/crewly-agent/src/runtime/desktop-checkpoint.test.ts +136 -0
- package/packages/crewly-agent/src/runtime/desktop-checkpoint.ts +231 -0
- package/packages/crewly-agent/src/runtime/desktop-recovery.test.ts +100 -0
- package/packages/crewly-agent/src/runtime/desktop-recovery.ts +195 -0
- package/packages/crewly-agent/src/runtime/desktop-task-runtime.test.ts +251 -0
- package/packages/crewly-agent/src/runtime/desktop-task-runtime.ts +423 -0
- package/packages/crewly-agent/src/runtime/desktop-task.tool.test.ts +218 -0
- package/packages/crewly-agent/src/runtime/desktop-task.tool.ts +343 -0
- package/packages/crewly-agent/src/runtime/text-tool-calls.test.ts +144 -0
- package/packages/crewly-agent/src/runtime/text-tool-calls.ts +316 -0
- package/packages/crewly-agent/src/runtime/text-tool-salvage.test.ts +190 -0
- package/packages/crewly-agent/src/runtime/tool-registry.test.ts +17 -0
- package/packages/crewly-agent/src/runtime/tool-registry.ts +54 -0
- package/packages/crewly-agent/src/runtime/types.ts +16 -1
- package/config/skills/agent/vnc-browser/SKILL.md +0 -140
|
@@ -0,0 +1,218 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Tests for the `desktop_task` tool.
|
|
3
|
+
*
|
|
4
|
+
* Phase 4 built the machinery; this is what connects it, and the connection
|
|
5
|
+
* is where it can be got wrong. The property under test throughout: an agent
|
|
6
|
+
* cannot mark its own work done, and a turn that ends with unverified work
|
|
7
|
+
* says so.
|
|
8
|
+
*/
|
|
9
|
+
|
|
10
|
+
import { describe, it, expect, beforeEach, vi } from 'vitest';
|
|
11
|
+
import {
|
|
12
|
+
createDesktopTaskTool, resetDesktopTasks, answerDesktopConfirmation, unverifiedDesktopWork,
|
|
13
|
+
} from './desktop-task.tool.js';
|
|
14
|
+
|
|
15
|
+
const SESSION = 'test-agent';
|
|
16
|
+
|
|
17
|
+
/** A perceive stub backed by a scriptable scene. */
|
|
18
|
+
function perceiver(scene: Record<string, unknown> = { success: true, app: 'TextEdit', elements: [] }) {
|
|
19
|
+
return vi.fn(async () => scene);
|
|
20
|
+
}
|
|
21
|
+
|
|
22
|
+
/** A two-subgoal plan checked against shell commands we control. */
|
|
23
|
+
function plan(overrides: Record<string, unknown> = {}) {
|
|
24
|
+
return {
|
|
25
|
+
operation: 'plan' as const,
|
|
26
|
+
goal: 'write and check a file',
|
|
27
|
+
subgoals: [
|
|
28
|
+
{ goal: 'write the note', checkpoint: { kind: 'shell' as const, command: 'test -f /tmp/crewly-tt-a' } },
|
|
29
|
+
{ goal: 'clean up', checkpoint: { kind: 'file-absent' as const, path: '/tmp/crewly-tt-a' } },
|
|
30
|
+
],
|
|
31
|
+
...overrides,
|
|
32
|
+
};
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
describe('planning', () => {
|
|
36
|
+
beforeEach(resetDesktopTasks);
|
|
37
|
+
|
|
38
|
+
it('refuses a plan with no subgoals — a task with no checkpoints proves nothing', async () => {
|
|
39
|
+
const tool = createDesktopTaskTool(SESSION);
|
|
40
|
+
expect(await tool.execute({ operation: 'plan', goal: 'do it' })).toMatchObject({ reason: 'validation' });
|
|
41
|
+
});
|
|
42
|
+
|
|
43
|
+
it('advises a route per subgoal, while it can still change what the agent does', async () => {
|
|
44
|
+
const tool = createDesktopTaskTool(SESSION, { availableSkills: ['gmail-send'] });
|
|
45
|
+
const out = (await tool.execute({
|
|
46
|
+
operation: 'plan',
|
|
47
|
+
goal: 'summarise and send',
|
|
48
|
+
subgoals: [
|
|
49
|
+
{ goal: 'read the report in Preview', checkpoint: { kind: 'app-frontmost', app: 'Preview' } },
|
|
50
|
+
{ goal: 'send it by email', checkpoint: { kind: 'shell', command: 'true' } },
|
|
51
|
+
],
|
|
52
|
+
})) as { routes: Array<Record<string, unknown>> };
|
|
53
|
+
// The point of routing: not opening Mail.app for something a skill does.
|
|
54
|
+
expect(out.routes[1]).toMatchObject({ route: 'skill', use: 'gmail-send' });
|
|
55
|
+
});
|
|
56
|
+
|
|
57
|
+
it('will not step before a plan exists', async () => {
|
|
58
|
+
const tool = createDesktopTaskTool(SESSION);
|
|
59
|
+
expect(await tool.execute({ operation: 'step' })).toMatchObject({ reason: 'no_task' });
|
|
60
|
+
});
|
|
61
|
+
});
|
|
62
|
+
|
|
63
|
+
describe('the agent cannot declare itself done', () => {
|
|
64
|
+
beforeEach(resetDesktopTasks);
|
|
65
|
+
|
|
66
|
+
it('holds the subgoal until the checkpoint holds, and says what is wrong', async () => {
|
|
67
|
+
const tool = createDesktopTaskTool(SESSION, { perceive: perceiver() });
|
|
68
|
+
await tool.execute(plan());
|
|
69
|
+
|
|
70
|
+
const first = (await tool.execute({ operation: 'step' })) as Record<string, unknown>;
|
|
71
|
+
expect(first['directive']).toBe('continue');
|
|
72
|
+
expect(String(first['hint'])).toContain('test -f');
|
|
73
|
+
|
|
74
|
+
// Now make it true.
|
|
75
|
+
const { writeFileSync, unlinkSync } = await import('fs');
|
|
76
|
+
writeFileSync('/tmp/crewly-tt-a', 'x');
|
|
77
|
+
try {
|
|
78
|
+
const second = (await tool.execute({ operation: 'step' })) as Record<string, unknown>;
|
|
79
|
+
expect(second['directive']).toBe('advance');
|
|
80
|
+
expect(second['verified']).toBe('write the note');
|
|
81
|
+
} finally {
|
|
82
|
+
unlinkSync('/tmp/crewly-tt-a');
|
|
83
|
+
}
|
|
84
|
+
});
|
|
85
|
+
|
|
86
|
+
it('reports outstanding work as unverified rather than rounding it up', async () => {
|
|
87
|
+
const tool = createDesktopTaskTool(SESSION, { perceive: perceiver() });
|
|
88
|
+
await tool.execute(plan());
|
|
89
|
+
const out = (await tool.execute({ operation: 'summary' })) as Record<string, unknown>;
|
|
90
|
+
expect(out['complete']).toBe(false);
|
|
91
|
+
expect(out['unverified']).toEqual(['write the note', 'clean up']);
|
|
92
|
+
});
|
|
93
|
+
|
|
94
|
+
it('makes the whole turn incomplete while a subgoal is unverified', async () => {
|
|
95
|
+
const tool = createDesktopTaskTool(SESSION, { perceive: perceiver() });
|
|
96
|
+
await tool.execute(plan());
|
|
97
|
+
// This is what the runner asks before letting a reply go out. A model
|
|
98
|
+
// that stopped because it believed it was finished is exactly the case.
|
|
99
|
+
const outstanding = unverifiedDesktopWork(SESSION);
|
|
100
|
+
expect(outstanding?.goals).toContain('write the note');
|
|
101
|
+
});
|
|
102
|
+
|
|
103
|
+
it('reports nothing outstanding once every checkpoint held', async () => {
|
|
104
|
+
const tool = createDesktopTaskTool(SESSION, { perceive: perceiver() });
|
|
105
|
+
await tool.execute({
|
|
106
|
+
operation: 'plan',
|
|
107
|
+
goal: 'trivial',
|
|
108
|
+
subgoals: [{ goal: 'nothing to do', checkpoint: { kind: 'shell', command: 'true' } }],
|
|
109
|
+
});
|
|
110
|
+
await tool.execute({ operation: 'step' });
|
|
111
|
+
expect(unverifiedDesktopWork(SESSION)).toBeNull();
|
|
112
|
+
});
|
|
113
|
+
|
|
114
|
+
it('knows nothing about a session that never planned anything', () => {
|
|
115
|
+
expect(unverifiedDesktopWork('never-planned')).toBeNull();
|
|
116
|
+
});
|
|
117
|
+
});
|
|
118
|
+
|
|
119
|
+
describe('irreversible actions', () => {
|
|
120
|
+
beforeEach(resetDesktopTasks);
|
|
121
|
+
|
|
122
|
+
it('sends one to the owner and tells the agent to stop, workarounds included', async () => {
|
|
123
|
+
const enqueue = vi.fn(() => ({ id: 'ap-1' }) as never);
|
|
124
|
+
const tool = createDesktopTaskTool(SESSION, { perceive: perceiver(), approvals: { enqueue } });
|
|
125
|
+
await tool.execute(plan());
|
|
126
|
+
|
|
127
|
+
const out = (await tool.execute({ operation: 'confirm', action: 'click Send on the email' })) as Record<string, unknown>;
|
|
128
|
+
expect(out).toMatchObject({ required: true, approvalId: 'ap-1' });
|
|
129
|
+
expect(String(out['message'])).toMatch(/not a workaround/i);
|
|
130
|
+
expect(enqueue).toHaveBeenCalledWith(SESSION, 'desktop_task', 'destructive', expect.objectContaining({
|
|
131
|
+
action: 'click Send on the email',
|
|
132
|
+
}));
|
|
133
|
+
});
|
|
134
|
+
|
|
135
|
+
it('holds every later step until the owner answers', async () => {
|
|
136
|
+
const tool = createDesktopTaskTool(SESSION, { perceive: perceiver() });
|
|
137
|
+
await tool.execute({
|
|
138
|
+
operation: 'plan', goal: 'send it',
|
|
139
|
+
// A checkpoint that would pass immediately, to prove the wait wins.
|
|
140
|
+
subgoals: [{ goal: 'send', checkpoint: { kind: 'shell', command: 'true' } }],
|
|
141
|
+
});
|
|
142
|
+
await tool.execute({ operation: 'confirm', action: 'delete the folder' });
|
|
143
|
+
const out = (await tool.execute({ operation: 'step' })) as Record<string, unknown>;
|
|
144
|
+
expect(out['directive']).toBe('await-confirmation');
|
|
145
|
+
});
|
|
146
|
+
|
|
147
|
+
it('says so plainly when something is reversible, instead of asking anyway', async () => {
|
|
148
|
+
const tool = createDesktopTaskTool(SESSION, { perceive: perceiver() });
|
|
149
|
+
await tool.execute(plan());
|
|
150
|
+
const out = (await tool.execute({ operation: 'confirm', action: 'open the document' })) as Record<string, unknown>;
|
|
151
|
+
// An agent that asks about everything learns nothing about where the line is.
|
|
152
|
+
expect(out).toMatchObject({ required: false });
|
|
153
|
+
});
|
|
154
|
+
|
|
155
|
+
it('carries on when the owner approves', async () => {
|
|
156
|
+
const tool = createDesktopTaskTool(SESSION, { perceive: perceiver() });
|
|
157
|
+
await tool.execute(plan());
|
|
158
|
+
await tool.execute({ operation: 'confirm', action: 'publish the post' });
|
|
159
|
+
const out = answerDesktopConfirmation(SESSION, true);
|
|
160
|
+
expect(out).toMatchObject({ directive: 'continue' });
|
|
161
|
+
});
|
|
162
|
+
|
|
163
|
+
it('stops for good when the owner declines', async () => {
|
|
164
|
+
const tool = createDesktopTaskTool(SESSION, { perceive: perceiver() });
|
|
165
|
+
await tool.execute(plan());
|
|
166
|
+
await tool.execute({ operation: 'confirm', action: 'publish the post' });
|
|
167
|
+
const out = answerDesktopConfirmation(SESSION, false) as Record<string, unknown>;
|
|
168
|
+
expect(out['directive']).toBe('escalate');
|
|
169
|
+
expect(String(out['message'])).toMatch(/do not look for another way/i);
|
|
170
|
+
});
|
|
171
|
+
|
|
172
|
+
it('answers nothing when nothing was waiting', () => {
|
|
173
|
+
expect(answerDesktopConfirmation('idle-session', true)).toBeNull();
|
|
174
|
+
});
|
|
175
|
+
});
|
|
176
|
+
|
|
177
|
+
describe('surprises reach the agent', () => {
|
|
178
|
+
beforeEach(resetDesktopTasks);
|
|
179
|
+
|
|
180
|
+
it('passes the dialog instruction through instead of testing the checkpoint', async () => {
|
|
181
|
+
const perceive = perceiver({
|
|
182
|
+
success: true, app: 'TextEdit',
|
|
183
|
+
elements: [
|
|
184
|
+
{ role: 'AXSheet', name: 'Save changes?' },
|
|
185
|
+
{ role: 'AXButton', name: "Don't Save" },
|
|
186
|
+
],
|
|
187
|
+
});
|
|
188
|
+
const tool = createDesktopTaskTool(SESSION, { perceive });
|
|
189
|
+
await tool.execute(plan());
|
|
190
|
+
const out = (await tool.execute({ operation: 'step' })) as Record<string, unknown>;
|
|
191
|
+
expect(out['directive']).toBe('continue');
|
|
192
|
+
expect(String(out['hint'])).toContain("Don't Save");
|
|
193
|
+
});
|
|
194
|
+
|
|
195
|
+
it('treats a failed snapshot as an empty scene rather than crashing the step', async () => {
|
|
196
|
+
const perceive = vi.fn(async () => ({ success: false, reason: 'screen_locked', message: 'locked' }));
|
|
197
|
+
const tool = createDesktopTaskTool(SESSION, { perceive });
|
|
198
|
+
await tool.execute(plan());
|
|
199
|
+
const out = (await tool.execute({ operation: 'step' })) as Record<string, unknown>;
|
|
200
|
+
// screen_locked is a surprise the runtime knows, and it needs a person.
|
|
201
|
+
expect(out['directive']).toBe('escalate');
|
|
202
|
+
expect(out['reason']).toBe('screen-locked');
|
|
203
|
+
});
|
|
204
|
+
});
|
|
205
|
+
|
|
206
|
+
describe('abandoning', () => {
|
|
207
|
+
beforeEach(resetDesktopTasks);
|
|
208
|
+
|
|
209
|
+
it('drops the task but still hands back what was and was not done', async () => {
|
|
210
|
+
const tool = createDesktopTaskTool(SESSION, { perceive: perceiver() });
|
|
211
|
+
await tool.execute(plan());
|
|
212
|
+
const out = (await tool.execute({ operation: 'abandon', reason: 'the app is not installed' })) as Record<string, unknown>;
|
|
213
|
+
expect(String(out['message'])).toContain('the app is not installed');
|
|
214
|
+
expect(out['summary']).toContain('write the note');
|
|
215
|
+
// And the turn is no longer held open by a task nobody is working on.
|
|
216
|
+
expect(unverifiedDesktopWork(SESSION)).toBeNull();
|
|
217
|
+
});
|
|
218
|
+
});
|
|
@@ -0,0 +1,343 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* `desktop_task` tool — the task runtime, plugged in.
|
|
3
|
+
*
|
|
4
|
+
* Phase 4 built the machinery that decides when a desktop subgoal is really
|
|
5
|
+
* done; this is what connects it to an agent. Without it the runtime was a
|
|
6
|
+
* tested library nothing called, which is the least useful state for a piece
|
|
7
|
+
* of safety machinery to be in.
|
|
8
|
+
*
|
|
9
|
+
* Three things make it hard to bypass rather than merely available:
|
|
10
|
+
*
|
|
11
|
+
* - `step` is the only way to advance. An agent cannot mark a subgoal done;
|
|
12
|
+
* it asks, and the checkpoint answers.
|
|
13
|
+
* - `summary` reports unverified subgoals as unverified. A turn that ends
|
|
14
|
+
* with work outstanding says so in the same words the owner would use.
|
|
15
|
+
* - An irreversible action goes to the approval queue that already exists,
|
|
16
|
+
* so the owner answers it wherever they already answer approvals.
|
|
17
|
+
*
|
|
18
|
+
* The tool holds one task per agent session. A desktop has one mouse, so an
|
|
19
|
+
* agent running two desktop tasks at once is a bug rather than a use case.
|
|
20
|
+
*
|
|
21
|
+
* @module runtime/desktop-task.tool
|
|
22
|
+
*/
|
|
23
|
+
|
|
24
|
+
import { z } from 'zod';
|
|
25
|
+
import type { ToolDefinition } from './types.js';
|
|
26
|
+
import type { ApprovalQueueService } from './approval-queue.service.js';
|
|
27
|
+
import {
|
|
28
|
+
beginTask, step as runtimeStep, summarize, activeSubgoal, chooseRoute,
|
|
29
|
+
needsConfirmation, awaitConfirmation, resolveConfirmation,
|
|
30
|
+
DEFAULT_TASK_BUDGET, type Subgoal, type TaskState, type Directive,
|
|
31
|
+
} from './desktop-task-runtime.js';
|
|
32
|
+
import type { Checkpoint, CheckpointDeps } from './desktop-checkpoint.js';
|
|
33
|
+
import type { Scene, SceneElement } from './desktop-recovery.js';
|
|
34
|
+
|
|
35
|
+
/** What the tool needs from the outside world. */
|
|
36
|
+
export interface DesktopTaskDeps {
|
|
37
|
+
/** Run a computer-use action; used to read the scene and test checkpoints. */
|
|
38
|
+
perceive?: (input: Record<string, unknown>) => Promise<Record<string, unknown>>;
|
|
39
|
+
/** Where an irreversible action goes for the owner to answer. */
|
|
40
|
+
approvals?: Pick<ApprovalQueueService, 'enqueue'>;
|
|
41
|
+
/** Skills this install has, so the router can prefer one over the screen. */
|
|
42
|
+
availableSkills?: string[];
|
|
43
|
+
now?: () => number;
|
|
44
|
+
}
|
|
45
|
+
|
|
46
|
+
/** One task per session: a desktop has one mouse. */
|
|
47
|
+
const tasks = new Map<string, TaskState>();
|
|
48
|
+
|
|
49
|
+
/** Approval ids waiting on the owner, by session. */
|
|
50
|
+
const pendingApprovals = new Map<string, string>();
|
|
51
|
+
|
|
52
|
+
/** Drop a session's task (tests, and session teardown). */
|
|
53
|
+
export function resetDesktopTasks(): void {
|
|
54
|
+
tasks.clear();
|
|
55
|
+
pendingApprovals.clear();
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
/** The checkpoint shapes an agent may declare. Mirrors {@link Checkpoint}. */
|
|
59
|
+
const checkpointSchema = z.union([
|
|
60
|
+
z.object({ kind: z.literal('file-exists'), path: z.string(), allowEmpty: z.boolean().optional() }),
|
|
61
|
+
z.object({ kind: z.literal('file-contains'), path: z.string(), text: z.string() }),
|
|
62
|
+
z.object({ kind: z.literal('file-matches'), path: z.string(), pattern: z.string() }),
|
|
63
|
+
z.object({ kind: z.literal('file-absent'), path: z.string() }),
|
|
64
|
+
z.object({ kind: z.literal('app-frontmost'), app: z.string() }),
|
|
65
|
+
z.object({ kind: z.literal('element-present'), name: z.string(), role: z.string().optional(), app: z.string().optional() }),
|
|
66
|
+
z.object({ kind: z.literal('element-absent'), name: z.string(), app: z.string().optional() }),
|
|
67
|
+
z.object({ kind: z.literal('text-on-screen'), text: z.string() }),
|
|
68
|
+
z.object({ kind: z.literal('shell'), command: z.string(), description: z.string().optional() }),
|
|
69
|
+
]);
|
|
70
|
+
|
|
71
|
+
const schema = z.object({
|
|
72
|
+
operation: z.enum(['plan', 'step', 'confirm', 'summary', 'abandon'])
|
|
73
|
+
.describe('plan a task, take a step, ask the owner about something irreversible, report, or give up.'),
|
|
74
|
+
goal: z.string().optional().describe('The whole task, for `plan`.'),
|
|
75
|
+
subgoals: z.array(z.object({
|
|
76
|
+
goal: z.string().describe('What to achieve, in your own words.'),
|
|
77
|
+
checkpoint: checkpointSchema.describe('How the runtime will know it happened — not your say-so.'),
|
|
78
|
+
maxSteps: z.number().optional(),
|
|
79
|
+
})).optional().describe('The plan, for `plan`. Three to six subgoals is usually right.'),
|
|
80
|
+
action: z.string().optional().describe('For `confirm`: the irreversible thing you are about to do.'),
|
|
81
|
+
reason: z.string().optional().describe('For `abandon`: why you are stopping.'),
|
|
82
|
+
});
|
|
83
|
+
|
|
84
|
+
/**
|
|
85
|
+
* Turn a directive into something a model can act on without re-reading the
|
|
86
|
+
* whole plan every step.
|
|
87
|
+
*
|
|
88
|
+
* @param directive - From the runtime
|
|
89
|
+
* @param state - Current task
|
|
90
|
+
* @returns The tool result
|
|
91
|
+
*/
|
|
92
|
+
function renderDirective(directive: Directive, state: TaskState): Record<string, unknown> {
|
|
93
|
+
const base = { success: true, directive: directive.kind };
|
|
94
|
+
switch (directive.kind) {
|
|
95
|
+
case 'continue':
|
|
96
|
+
return {
|
|
97
|
+
...base,
|
|
98
|
+
subgoal: directive.subgoal.goal,
|
|
99
|
+
stepsLeft: directive.stepsLeft,
|
|
100
|
+
// The hint is the reason the checkpoint did not hold, or what the
|
|
101
|
+
// surprise was. It is the actionable part; the goal is just context.
|
|
102
|
+
...(directive.hint ? { hint: directive.hint } : {}),
|
|
103
|
+
};
|
|
104
|
+
case 'advance':
|
|
105
|
+
return { ...base, verified: directive.completed.goal, next: directive.next.goal };
|
|
106
|
+
case 'done':
|
|
107
|
+
return { ...base, summary: summarize(state), message: directive.summary };
|
|
108
|
+
case 'escalate':
|
|
109
|
+
return {
|
|
110
|
+
...base, success: false, reason: directive.reason, message: directive.detail,
|
|
111
|
+
summary: summarize(state),
|
|
112
|
+
};
|
|
113
|
+
case 'await-confirmation':
|
|
114
|
+
return { ...base, waitingFor: directive.action, message: directive.detail };
|
|
115
|
+
case 'budget-exhausted':
|
|
116
|
+
return {
|
|
117
|
+
...base, success: false, reason: 'budget_exhausted', message: directive.reason,
|
|
118
|
+
completed: directive.completed, remaining: directive.remaining, summary: summarize(state),
|
|
119
|
+
};
|
|
120
|
+
}
|
|
121
|
+
}
|
|
122
|
+
|
|
123
|
+
/**
|
|
124
|
+
* Read the screen into the shape the surprise detector wants.
|
|
125
|
+
*
|
|
126
|
+
* A failed or missing snapshot yields an empty scene rather than throwing:
|
|
127
|
+
* not being able to look is itself worth continuing past, and the checkpoint
|
|
128
|
+
* will fail honestly a moment later.
|
|
129
|
+
*
|
|
130
|
+
* @param deps - Injected IO
|
|
131
|
+
* @returns The scene
|
|
132
|
+
*/
|
|
133
|
+
async function readScene(deps: DesktopTaskDeps): Promise<Scene> {
|
|
134
|
+
if (!deps.perceive) return { elements: [] };
|
|
135
|
+
const snap = await deps.perceive({ action: 'snapshot' }).catch(() => null);
|
|
136
|
+
if (!snap || snap['success'] === false) {
|
|
137
|
+
return {
|
|
138
|
+
elements: [],
|
|
139
|
+
...(snap ? { lastFailure: { reason: String(snap['reason'] ?? ''), message: String(snap['message'] ?? '') } } : {}),
|
|
140
|
+
};
|
|
141
|
+
}
|
|
142
|
+
const elements = (snap['elements'] as Array<Record<string, unknown>> | undefined) ?? [];
|
|
143
|
+
return {
|
|
144
|
+
...(snap['app'] ? { app: String(snap['app']) } : {}),
|
|
145
|
+
elements: elements.map((e): SceneElement => ({
|
|
146
|
+
role: String(e['role'] ?? ''),
|
|
147
|
+
...(e['name'] ? { name: String(e['name']) } : {}),
|
|
148
|
+
...(e['subrole'] ? { subrole: String(e['subrole']) } : {}),
|
|
149
|
+
})),
|
|
150
|
+
};
|
|
151
|
+
}
|
|
152
|
+
|
|
153
|
+
/** Checkpoint IO backed by the desktop, for the screen-reading kinds. */
|
|
154
|
+
function checkpointDeps(deps: DesktopTaskDeps): CheckpointDeps {
|
|
155
|
+
if (!deps.perceive) return {};
|
|
156
|
+
return {
|
|
157
|
+
snapshot: async (app?: string) => {
|
|
158
|
+
const snap = await deps.perceive!({ action: 'snapshot', ...(app ? { app } : {}) });
|
|
159
|
+
const elements = (snap['elements'] as Array<Record<string, unknown>> | undefined) ?? [];
|
|
160
|
+
return elements.map((e) => ({
|
|
161
|
+
role: String(e['role'] ?? ''),
|
|
162
|
+
...(e['name'] ? { name: String(e['name']) } : {}),
|
|
163
|
+
}));
|
|
164
|
+
},
|
|
165
|
+
screenText: async () => {
|
|
166
|
+
const out = await deps.perceive!({ action: 'ocr' });
|
|
167
|
+
const items = (out['items'] as Array<Record<string, unknown>> | undefined) ?? [];
|
|
168
|
+
return items.map((i) => String(i['text'] ?? ''));
|
|
169
|
+
},
|
|
170
|
+
frontmostApp: async () => {
|
|
171
|
+
const out = await deps.perceive!({ action: 'snapshot' });
|
|
172
|
+
return String(out['app'] ?? '');
|
|
173
|
+
},
|
|
174
|
+
};
|
|
175
|
+
}
|
|
176
|
+
|
|
177
|
+
/**
|
|
178
|
+
* Build the `desktop_task` tool.
|
|
179
|
+
*
|
|
180
|
+
* @param sessionName - Whose task this is
|
|
181
|
+
* @param deps - Injected IO
|
|
182
|
+
* @returns The tool definition
|
|
183
|
+
*/
|
|
184
|
+
export function createDesktopTaskTool(sessionName: string, deps: DesktopTaskDeps = {}): ToolDefinition {
|
|
185
|
+
const now = deps.now ?? Date.now;
|
|
186
|
+
|
|
187
|
+
return {
|
|
188
|
+
description:
|
|
189
|
+
'Run a multi-step desktop task so it cannot end in a false "done". ' +
|
|
190
|
+
'Call `plan` once with subgoals and a checkpoint for each — a checkpoint is something outside you that ' +
|
|
191
|
+
'proves the subgoal happened (a file with the right contents, a dialog that is gone, an app in front). ' +
|
|
192
|
+
'Then use the `computer` tool to work, and call `step` after each action: it tells you whether the ' +
|
|
193
|
+
'checkpoint held, what changed unexpectedly, and when to move on. ' +
|
|
194
|
+
'Call `confirm` before anything irreversible (sending, deleting, publishing, paying) — the owner answers. ' +
|
|
195
|
+
'You cannot mark a subgoal done yourself; that is the point.',
|
|
196
|
+
inputSchema: schema,
|
|
197
|
+
sensitivity: 'sensitive',
|
|
198
|
+
execute: async (rawArgs) => {
|
|
199
|
+
const args = rawArgs as z.infer<typeof schema>;
|
|
200
|
+
const existing = tasks.get(sessionName);
|
|
201
|
+
|
|
202
|
+
switch (args.operation) {
|
|
203
|
+
case 'plan': {
|
|
204
|
+
if (!args.goal || !args.subgoals?.length) {
|
|
205
|
+
return { success: false, reason: 'validation', message: 'plan needs `goal` and at least one subgoal.' };
|
|
206
|
+
}
|
|
207
|
+
const subgoals: Subgoal[] = args.subgoals.map((s, i) => ({
|
|
208
|
+
id: `s${i + 1}`,
|
|
209
|
+
goal: s.goal,
|
|
210
|
+
checkpoint: s.checkpoint as Checkpoint,
|
|
211
|
+
...(s.maxSteps ? { maxSteps: s.maxSteps } : {}),
|
|
212
|
+
}));
|
|
213
|
+
const state = beginTask(args.goal, subgoals, DEFAULT_TASK_BUDGET, now);
|
|
214
|
+
tasks.set(sessionName, state);
|
|
215
|
+
|
|
216
|
+
// The route is advice given once, up front, where it can still
|
|
217
|
+
// change what the agent does — after it has opened an app it is
|
|
218
|
+
// too late to say a skill would have been better.
|
|
219
|
+
const routes = subgoals.map((s) => {
|
|
220
|
+
const r = chooseRoute(s.goal, deps.availableSkills ?? []);
|
|
221
|
+
return { subgoal: s.goal, route: r.route, because: r.because, ...(r.candidate ? { use: r.candidate } : {}) };
|
|
222
|
+
});
|
|
223
|
+
return {
|
|
224
|
+
success: true,
|
|
225
|
+
planned: subgoals.length,
|
|
226
|
+
first: subgoals[0]!.goal,
|
|
227
|
+
routes,
|
|
228
|
+
budget: { steps: state.budget.maxSteps, minutes: Math.round(state.budget.maxDurationMs / 60000) },
|
|
229
|
+
message: 'Work on the first subgoal, then call step. You cannot skip a checkpoint.',
|
|
230
|
+
};
|
|
231
|
+
}
|
|
232
|
+
|
|
233
|
+
case 'step': {
|
|
234
|
+
if (!existing) {
|
|
235
|
+
return { success: false, reason: 'no_task', message: 'No task planned. Call `plan` first.' };
|
|
236
|
+
}
|
|
237
|
+
const scene = await readScene(deps);
|
|
238
|
+
const directive = await runtimeStep(existing, scene, checkpointDeps(deps), now);
|
|
239
|
+
if (directive.kind === 'done' || directive.kind === 'escalate' || directive.kind === 'budget-exhausted') {
|
|
240
|
+
// The task is over either way; keep the state for `summary` but
|
|
241
|
+
// stop it being stepped again.
|
|
242
|
+
tasks.set(sessionName, existing);
|
|
243
|
+
}
|
|
244
|
+
return renderDirective(directive, existing);
|
|
245
|
+
}
|
|
246
|
+
|
|
247
|
+
case 'confirm': {
|
|
248
|
+
if (!existing) {
|
|
249
|
+
return { success: false, reason: 'no_task', message: 'No task planned. Call `plan` first.' };
|
|
250
|
+
}
|
|
251
|
+
if (!args.action) {
|
|
252
|
+
return { success: false, reason: 'validation', message: 'confirm needs `action` — what you are about to do.' };
|
|
253
|
+
}
|
|
254
|
+
const verdict = needsConfirmation(args.action);
|
|
255
|
+
if (!verdict.required) {
|
|
256
|
+
// Saying so beats silently approving: an agent that asks about
|
|
257
|
+
// everything learns nothing, and one told "no need" learns where
|
|
258
|
+
// the line is.
|
|
259
|
+
return {
|
|
260
|
+
success: true, required: false,
|
|
261
|
+
message: `"${args.action}" is reversible — go ahead without asking.`,
|
|
262
|
+
};
|
|
263
|
+
}
|
|
264
|
+
awaitConfirmation(existing, args.action, now);
|
|
265
|
+
const approval = deps.approvals?.enqueue(sessionName, 'desktop_task', 'destructive', {
|
|
266
|
+
action: args.action,
|
|
267
|
+
goal: existing.goal,
|
|
268
|
+
subgoal: activeSubgoal(existing)?.subgoal.goal ?? '',
|
|
269
|
+
});
|
|
270
|
+
if (approval) pendingApprovals.set(sessionName, approval.id);
|
|
271
|
+
return {
|
|
272
|
+
success: true,
|
|
273
|
+
required: true,
|
|
274
|
+
trigger: verdict.trigger,
|
|
275
|
+
...(approval ? { approvalId: approval.id } : {}),
|
|
276
|
+
message:
|
|
277
|
+
`Waiting for the owner to approve "${args.action}". Do nothing else until they answer — ` +
|
|
278
|
+
'not a workaround, not a different route to the same effect.',
|
|
279
|
+
};
|
|
280
|
+
}
|
|
281
|
+
|
|
282
|
+
case 'summary': {
|
|
283
|
+
if (!existing) return { success: true, message: 'No task in progress.' };
|
|
284
|
+
const outstanding = existing.progress.filter((p) => p.state !== 'done');
|
|
285
|
+
return {
|
|
286
|
+
success: true,
|
|
287
|
+
complete: outstanding.length === 0,
|
|
288
|
+
// Reported as unverified, not as "probably fine". This is the
|
|
289
|
+
// sentence that stops a turn ending in a confident false report.
|
|
290
|
+
unverified: outstanding.map((p) => p.subgoal.goal),
|
|
291
|
+
summary: summarize(existing),
|
|
292
|
+
};
|
|
293
|
+
}
|
|
294
|
+
|
|
295
|
+
case 'abandon': {
|
|
296
|
+
tasks.delete(sessionName);
|
|
297
|
+
pendingApprovals.delete(sessionName);
|
|
298
|
+
return {
|
|
299
|
+
success: true,
|
|
300
|
+
message: `Task dropped${args.reason ? `: ${args.reason}` : ''}. Say what was and was not done.`,
|
|
301
|
+
...(existing ? { summary: summarize(existing) } : {}),
|
|
302
|
+
};
|
|
303
|
+
}
|
|
304
|
+
}
|
|
305
|
+
},
|
|
306
|
+
};
|
|
307
|
+
}
|
|
308
|
+
|
|
309
|
+
/**
|
|
310
|
+
* Record the owner's answer to a pending confirmation.
|
|
311
|
+
*
|
|
312
|
+
* Called by whatever surfaces approvals (the REST endpoint, Slack), not by
|
|
313
|
+
* the agent — which is the whole point of the confirmation.
|
|
314
|
+
*
|
|
315
|
+
* @param sessionName - Whose task
|
|
316
|
+
* @param approved - The owner's decision
|
|
317
|
+
* @returns What the agent will be told next, or null when nothing was waiting
|
|
318
|
+
*/
|
|
319
|
+
export function answerDesktopConfirmation(sessionName: string, approved: boolean): Record<string, unknown> | null {
|
|
320
|
+
const state = tasks.get(sessionName);
|
|
321
|
+
if (!state?.awaitingConfirmation) return null;
|
|
322
|
+
const directive = resolveConfirmation(state, approved);
|
|
323
|
+
pendingApprovals.delete(sessionName);
|
|
324
|
+
return renderDirective(directive, state);
|
|
325
|
+
}
|
|
326
|
+
|
|
327
|
+
/**
|
|
328
|
+
* Whether a session has desktop work that never passed its checkpoint.
|
|
329
|
+
*
|
|
330
|
+
* The runner asks this at the end of a turn: a task with unverified subgoals
|
|
331
|
+
* makes the turn incomplete, which is reported to the owner in the same
|
|
332
|
+
* machinery that already reports a truncated or interrupted turn.
|
|
333
|
+
*
|
|
334
|
+
* @param sessionName - Whose task
|
|
335
|
+
* @returns The unfinished subgoals, or null when there is nothing outstanding
|
|
336
|
+
*/
|
|
337
|
+
export function unverifiedDesktopWork(sessionName: string): { goals: string[]; summary: string } | null {
|
|
338
|
+
const state = tasks.get(sessionName);
|
|
339
|
+
if (!state) return null;
|
|
340
|
+
const outstanding = state.progress.filter((p) => p.state !== 'done' && p.state !== 'skipped');
|
|
341
|
+
if (outstanding.length === 0) return null;
|
|
342
|
+
return { goals: outstanding.map((p) => p.subgoal.goal), summary: summarize(state) };
|
|
343
|
+
}
|