@vellumai/assistant 0.12.2-dev.202609182214.bd8f92f → 0.12.2-dev.202609190043.ee5ba34

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (73) hide show
  1. package/docs/desktop-browser-cli.md +20 -1
  2. package/node_modules/@vellumai/gateway-client/src/gateway-ipc-contracts.ts +20 -0
  3. package/openapi.yaml +8 -0
  4. package/package.json +1 -1
  5. package/src/__tests__/checker.test.ts +11 -3
  6. package/src/__tests__/computer-use-skill-manifest-regression.test.ts +0 -25
  7. package/src/__tests__/computer-use-tools.test.ts +1 -233
  8. package/src/__tests__/conversation-tool-setup-attribution.test.ts +23 -0
  9. package/src/__tests__/interrupt-on-send.test.ts +57 -43
  10. package/src/__tests__/interrupt-turn-note.test.ts +5 -6
  11. package/src/__tests__/migration-export-to-gcs.test.ts +142 -0
  12. package/src/__tests__/plugin-api-provider.test.ts +5 -0
  13. package/src/__tests__/settings-routes.test.ts +50 -1
  14. package/src/activation/progress-store.test.ts +12 -7
  15. package/src/browser/virtual-desktop-target.ts +2 -8
  16. package/src/config/bundled-skills/computer-use/SKILL.md +1 -1
  17. package/src/config/bundled-skills/computer-use/TOOLS.json +89 -1
  18. package/src/daemon/conversation-interrupt-repair.ts +4 -43
  19. package/src/daemon/conversation-interrupt.ts +7 -11
  20. package/src/daemon/conversation-messaging.ts +6 -4
  21. package/src/daemon/conversation-surfaces.ts +27 -1
  22. package/src/daemon/conversation-tool-setup.ts +7 -2
  23. package/src/daemon/conversation.ts +13 -8
  24. package/src/daemon/host-cu-proxy.ts +30 -19
  25. package/src/daemon/virtual-desktop-context.ts +21 -0
  26. package/src/desktop/__tests__/fake-desktop.ts +1 -0
  27. package/src/desktop/desktop-accessibility-script.ts +88 -0
  28. package/src/desktop/desktop-accessibility.test.ts +160 -0
  29. package/src/desktop/desktop-accessibility.ts +169 -0
  30. package/src/desktop/desktop-automation-lease.test.ts +71 -0
  31. package/src/desktop/desktop-automation-lease.ts +40 -12
  32. package/src/desktop/desktop-browser-endpoint.ts +1 -0
  33. package/src/desktop/desktop-computer-use-routing.test.ts +330 -0
  34. package/src/desktop/desktop-computer-use.test.ts +450 -0
  35. package/src/desktop/desktop-computer-use.ts +503 -0
  36. package/src/desktop/desktop-dependencies.ts +10 -2
  37. package/src/desktop/desktop-session-manager.test.ts +63 -18
  38. package/src/desktop/desktop-session-manager.ts +68 -11
  39. package/src/desktop/virtual-desktop-feature.ts +52 -2
  40. package/src/live-voice/__tests__/live-voice-vad.test.ts +15 -9
  41. package/src/persistence/conversation-crud.ts +4 -4
  42. package/src/plugin-api/effective-context-window.ts +33 -0
  43. package/src/plugin-api/index.ts +9 -2
  44. package/src/plugins/defaults/memory/v3/__tests__/pool-log-store.test.ts +165 -27
  45. package/src/plugins/defaults/memory/v3/__tests__/pool-select.test.ts +511 -30
  46. package/src/plugins/defaults/memory/v3/__tests__/shadow-plugin.test.ts +46 -27
  47. package/src/plugins/defaults/memory/v3/orchestrate.ts +75 -37
  48. package/src/plugins/defaults/memory/v3/pool-log-store.ts +65 -79
  49. package/src/plugins/defaults/memory/v3/pool-select.test.ts +14 -5
  50. package/src/plugins/defaults/memory/v3/pool-select.ts +447 -65
  51. package/src/runtime/AGENTS.md +1 -1
  52. package/src/runtime/migrations/__tests__/vbundle-import-parity.test.ts +67 -1
  53. package/src/runtime/migrations/vbundle-builder.ts +24 -1
  54. package/src/runtime/migrations/vbundle-import-analyzer.ts +13 -4
  55. package/src/runtime/migrations/vbundle-import-policy.ts +10 -0
  56. package/src/runtime/migrations/vbundle-importer.ts +4 -1
  57. package/src/runtime/migrations/vbundle-streaming-importer.ts +4 -1
  58. package/src/runtime/migrations/vbundle-validator.ts +1 -0
  59. package/src/runtime/routes/migration-routes.ts +73 -2
  60. package/src/runtime/routes/settings-routes.ts +10 -4
  61. package/src/tools/client-os.ts +15 -0
  62. package/src/tools/computer-use/skill-proxy-bridge.ts +1 -1
  63. package/src/tools/computer-use/target.ts +42 -0
  64. package/src/tools/execution-target.ts +14 -16
  65. package/src/tools/permission-checker.ts +3 -3
  66. package/src/tools/policy-context.ts +3 -1
  67. package/src/tools/skills/load.ts +2 -1
  68. package/src/tools/skills/skill-tool-factory.ts +11 -0
  69. package/src/tools/tool-approval-handler.ts +5 -1
  70. package/src/tools/types.ts +7 -1
  71. package/src/util/abort-reasons.ts +18 -26
  72. package/src/__tests__/computer-use-window-schema.test.ts +0 -24
  73. package/src/tools/computer-use/definitions.ts +0 -607
@@ -24,7 +24,7 @@ The desktop client bypasses personal-browser discovery, extension reconnect wait
24
24
 
25
25
  Tab IDs are ephemeral numeric aliases for this managed browser's CDP target IDs. They are never personal extension tab IDs. Initial attachment selects an existing HTTP(S) page or opens a blank tab. Use `tabs list` and `tabs select --tab-id <id>` to choose explicitly. Navigation, tab selection, document replacement and release clear the desktop snapshot map. Personal-browser snapshots use a separate namespace.
26
26
 
27
- CDP mouse events use page viewport CSS coordinates. Before dispatching them, the client animates a purple arrow overlay to the same point. The overlay is excluded from accessibility and hit testing. It appears in the existing desktop stream and is removed on release. It does not move the OS pointer, appear in browser toolbar UI or visualize every programmatic DOM operation. Native dialogs and other applications are outside the browser CLI; users can interact with them directly in the expanded desktop. The picture-in-picture preview is view-only. Direct user interaction does not pause automation. For logins, the assistant uses saved credentials first and securely collects missing credentials with `assistant credentials prompt`, then fills the login form itself. Every CAPTCHA and bot-detection challenge, including sliders and press-and-hold checks, is handed to the user before the assistant attempts it. When human interaction is needed, the assistant calls `ask_question` with `desktopHelp: { message, doneLabel, skipLabel }` in the user's language. The tool releases held browser input and reserves the desktop before publishing a question with `presentation: "virtual_desktop"`. The live viewer mounts only in the attended Electron window, with browser focus as a fallback on older shells and the web, so conversations release it when focus changes. A busy viewer offers Reconnect if the previous connection has not finished closing. The compact card lives in the scrolling message transcript, with expandable instructions and a preview capped at 384 pixels wide. It hosts the shared live viewer, with Step In opening its full desktop modal and Done or Skip resolving the existing question response. The assistant retains an exclusive human-help reservation while waiting. Done resumes the same conversation's automation lease for its fresh snapshot; Skip, closure, timeout and cancellation release it. The browser idle timer does not expire a pending human-help reservation. The live preview connects after Show live preview or Step In, and reuses an already-open picture-in-picture viewer. Other devices receiving the card do not automatically open a stream. When the request resolves, its interactive viewer closes or returns to the read-only PiP session that was open before the request. Skip does not imply the obstacle was resolved. Older clients show the same request as a normal Done/Skip question, with model-localized labels retained in history. Channels without dynamic UI return guidance to continue in the app rather than waiting on an unusable card.
27
+ CDP mouse events use page viewport CSS coordinates. Before dispatching them, the client animates a purple arrow overlay to the same point. The overlay is excluded from accessibility and hit testing. It appears in the existing desktop stream and is removed on release. It does not move the OS pointer, appear in browser toolbar UI or visualize every programmatic DOM operation. Native dialogs and other applications use the computer-use skill or direct interaction in the expanded desktop. The picture-in-picture preview is view-only. Direct user interaction does not pause automation. For logins, the assistant uses saved credentials first and securely collects missing credentials with `assistant credentials prompt`, then fills the login form itself. Every CAPTCHA and bot-detection challenge, including sliders and press-and-hold checks, is handed to the user before the assistant attempts it. When human interaction is needed, the assistant calls `ask_question` with `desktopHelp: { message, doneLabel, skipLabel }` in the user's language. The tool releases held browser input and reserves the desktop before publishing a question with `presentation: "virtual_desktop"`. The live viewer mounts only in the attended Electron window, with browser focus as a fallback on older shells and the web, so conversations release it when focus changes. A busy viewer offers Reconnect if the previous connection has not finished closing. The compact card lives in the scrolling message transcript, with expandable instructions and a preview capped at 384 pixels wide. It hosts the shared live viewer, with Step In opening its full desktop modal and Done or Skip resolving the existing question response. The assistant retains an exclusive human-help reservation while waiting. Done resumes the same conversation's automation lease for its fresh snapshot; Skip, closure, timeout and cancellation release it. The browser idle timer does not expire a pending human-help reservation. The live preview connects after Show live preview or Step In, and reuses an already-open picture-in-picture viewer. Other devices receiving the card do not automatically open a stream. When the request resolves, its interactive viewer closes or returns to the read-only PiP session that was open before the request. Skip does not imply the obstacle was resolved. Older clients show the same request as a normal Done/Skip question, with model-localized labels retained in history. Channels without dynamic UI return guidance to continue in the app rather than waiting on an unusable card.
28
28
 
29
29
  The client records key and mouse presses before dispatch. On release it opens a fresh bounded cleanup connection, attaches to the same live targets, releases uncertain held input and removes overlays. A failed cleanup preserves state and the automation slot for retry. Turn-triggered cancellation retries cleanup automatically until it succeeds or ownership changes. Dispatched actions are never automatically retried. Closed targets need no input cleanup. Browser-process loss disposes the client, and later requests discover the replacement process.
30
30
 
@@ -35,3 +35,22 @@ Validation: focused client tests exercise shared snapshot/click behavior, namesp
35
35
  The desktop header icon pulses in the assistant's avatar accent color while a browser automation lease is active, including between browser commands. It shares the progress indicator's accent and neutral fallback. Reduced-motion clients show a solid accent. Setup status exposes the optional `automationActive` field; `desktop_activity_changed` events refresh it on acquisition and cancellation or release. Installation progress retains its `assistant:self:desktop` sync invalidations. The indicator reads status without starting installation, and reconnects refetch the current lease state.
36
36
 
37
37
  Desktop streaming checks the current in-memory gateway feature flags during connection startup. Once connected, frame and drain callbacks do not check feature flags or load workspace configuration. Turning the flag off prevents new connections; an existing stream continues until it closes or the desktop stops.
38
+
39
+ ## Computer use on the same desktop
40
+
41
+ The existing `computer-use` skill defaults to this desktop in platform-hosted
42
+ web conversations. Native desktop clients, including the Mac app connected to a
43
+ platform-hosted assistant, keep the connected-computer default and require
44
+ `target: "assistant-desktop"` to select the virtual desktop explicitly.
45
+ Non-platform assistants keep their connected-computer behavior.
46
+
47
+ Explicit `target: "connected-computer"` or `target_client_id` selects a connected
48
+ computer on every client. The selected target never falls back to another
49
+ computer when unavailable. Virtual desktop observations return full-screen
50
+ screenshots and mouse actions use screen coordinates.
51
+
52
+ Observe first and pass the returned `observation_id` with each native action.
53
+ Browser commands, user handoff, errors, and interruption require a fresh
54
+ observation before returning to native input. Browser commands and computer use
55
+ share session ownership, cancellation, and the user-help reservation. Use
56
+ browser commands for page-level operations and computer use for desktop UI.
@@ -43,6 +43,26 @@ export type GatewayLogsTailRouteParams = z.infer<
43
43
  typeof GatewayLogsTailRouteParamsSchema
44
44
  >;
45
45
 
46
+ /**
47
+ * `gateway_debug_export`: a consistent copy of the gateway database plus its
48
+ * recent log files, as one gzipped tar, for a debug bundle. No params.
49
+ */
50
+ export const GatewayDebugExportIpcParamsSchema = z
51
+ .object({})
52
+ .strict()
53
+ .default({});
54
+
55
+ export const GatewayDebugExportIpcResponseSchema = z.object({
56
+ ok: z.literal(true),
57
+ /** `tar.gz` bytes, base64. Contains gateway.sqlite and gateway-logs/. */
58
+ archive_base64: z.string(),
59
+ size_bytes: z.number().int().nonnegative(),
60
+ });
61
+
62
+ export type GatewayDebugExportIpcResponse = z.infer<
63
+ typeof GatewayDebugExportIpcResponseSchema
64
+ >;
65
+
46
66
  export const GatewayLogsTailIpcResponseSchema = z.object({
47
67
  lines: z.array(z.record(z.string(), z.unknown())),
48
68
  truncated: z.boolean(),
package/openapi.yaml CHANGED
@@ -23549,6 +23549,14 @@ paths:
23549
23549
  description:
23550
23550
  description: Human-readable export description.
23551
23551
  type: string
23552
+ profile:
23553
+ description:
23554
+ "Export profile. 'migration' (default) builds a teleport bundle. 'debug' builds a bundle for Vellum staff
23555
+ to open on a debug clone: no credentials, plus the gateway's database and logs under gateway/."
23556
+ type: string
23557
+ enum:
23558
+ - migration
23559
+ - debug
23552
23560
  required:
23553
23561
  - upload_url
23554
23562
  responses:
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@vellumai/assistant",
3
- "version": "0.12.2-dev.202609182214.bd8f92f",
3
+ "version": "0.12.2-dev.202609190043.ee5ba34",
4
4
  "license": "MIT",
5
5
  "type": "module",
6
6
  "exports": {
@@ -140,9 +140,17 @@ registerSkillTools("app-builder", [mockBundledSkillTool]);
140
140
 
141
141
  // Register CU tools so check() can look them up in the tool registry
142
142
  // instead of falling through to Medium (unknown tool).
143
- import { allComputerUseTools } from "../tools/computer-use/definitions.js";
144
- for (const tool of allComputerUseTools) {
145
- registerTool(tool);
143
+ import { COMPUTER_USE_TOOL_NAMES } from "./test-support/computer-use-skill-harness.js";
144
+ for (const name of COMPUTER_USE_TOOL_NAMES) {
145
+ registerTool({
146
+ name,
147
+ description: name,
148
+ category: "computer-use",
149
+ defaultRiskLevel: RiskLevel.Low,
150
+ executionTarget: "host",
151
+ input_schema: { type: "object", properties: {} },
152
+ execute: async () => ({ content: "ok", isError: false }),
153
+ });
146
154
  }
147
155
 
148
156
  function writeSkill(
@@ -3,7 +3,6 @@ import { resolve } from "node:path";
3
3
  import { afterAll, describe, expect, test } from "bun:test";
4
4
 
5
5
  import { RiskLevel } from "../permissions/types.js";
6
- import { allComputerUseTools } from "../tools/computer-use/definitions.js";
7
6
  import {
8
7
  __resetRegistryForTesting,
9
8
  getTool,
@@ -74,30 +73,6 @@ describe("computer-use skill manifest regression", () => {
74
73
  }
75
74
  });
76
75
 
77
- test("manifest descriptions match core definitions", async () => {
78
- await initializeTools();
79
-
80
- for (const cuTool of allComputerUseTools) {
81
- const manifestTool = manifest.tools.find(
82
- (t: { name: string }) => t.name === cuTool.name,
83
- );
84
- expect(manifestTool).toBeDefined();
85
- expect(manifestTool.description).toBe(cuTool.description);
86
- }
87
- });
88
-
89
- test("manifest input_schema matches core definitions", async () => {
90
- await initializeTools();
91
-
92
- for (const cuTool of allComputerUseTools) {
93
- const manifestTool = manifest.tools.find(
94
- (t: { name: string }) => t.name === cuTool.name,
95
- );
96
- expect(manifestTool).toBeDefined();
97
- expect(manifestTool.input_schema).toEqual(cuTool.input_schema);
98
- }
99
- });
100
-
101
76
  test("CU action tools are not registered as core tools after initializeTools()", async () => {
102
77
  await initializeTools();
103
78
 
@@ -1,246 +1,14 @@
1
1
  import { describe, expect, test } from "bun:test";
2
2
 
3
- import {
4
- allComputerUseTools,
5
- computerUseClickTool,
6
- computerUseDoneTool,
7
- computerUseDragTool,
8
- computerUseKeyTool,
9
- computerUseObserveTool,
10
- computerUseOpenAppTool,
11
- computerUseRespondTool,
12
- computerUseRunAppleScriptTool,
13
- computerUseScrollTool,
14
- computerUseTypeTextTool,
15
- computerUseWaitTool,
16
- } from "../tools/computer-use/definitions.js";
17
3
  import { forwardComputerUseProxyTool } from "../tools/computer-use/skill-proxy-bridge.js";
18
4
  import type { ToolContext } from "../tools/types.js";
19
5
 
20
- interface JsonSchema {
21
- type?: string;
22
- required?: string[];
23
- properties?: Record<string, unknown>;
24
- }
25
-
26
- /** Cast a tool definition's input_schema to a usable JSON Schema shape. */
27
- function schema(tool: { input_schema?: object }): JsonSchema {
28
- return tool.input_schema as JsonSchema;
29
- }
30
-
31
6
  const ctx: ToolContext = {
32
7
  workingDir: "/tmp",
33
- conversationId: "test-conversation",
8
+ conversationId: "conv-123",
34
9
  trustClass: "guardian",
35
10
  };
36
11
 
37
- // ── Tool definitions ────────────────────────────────────────────────
38
-
39
- describe("computer-use tool definitions", () => {
40
- test("allComputerUseTools contains 12 tools", () => {
41
- expect(allComputerUseTools.length).toBe(12);
42
- });
43
-
44
- test("all tools belong to computer-use category", () => {
45
- for (const tool of allComputerUseTools) {
46
- expect(tool.category).toBe("computer-use");
47
- }
48
- });
49
-
50
- test("all tools have unique names", () => {
51
- const names = allComputerUseTools.map((t) => t.name);
52
- expect(new Set(names).size).toBe(names.length);
53
- });
54
-
55
- test("all tools have descriptions", () => {
56
- for (const tool of allComputerUseTools) {
57
- expect(tool.description!.length).toBeGreaterThan(0);
58
- }
59
- });
60
- });
61
-
62
- // ── observe ─────────────────────────────────────────────────────────
63
-
64
- describe("computer_use_observe", () => {
65
- test("supports target_client_id", () => {
66
- const props = schema(computerUseObserveTool).properties as Record<
67
- string,
68
- { type: string }
69
- >;
70
- expect(props.target_client_id.type).toBe("string");
71
- });
72
- });
73
-
74
- // ── Unified click tool ──────────────────────────────────────────────
75
-
76
- describe("computer_use_click (unified)", () => {
77
- test("has correct name", () => {
78
- expect(computerUseClickTool.name).toBe("computer_use_click");
79
- });
80
-
81
- test("schema requires reasoning", () => {
82
- expect(schema(computerUseClickTool).required).toContain("reasoning");
83
- });
84
-
85
- test("schema supports click_type enum", () => {
86
- const props = schema(computerUseClickTool).properties as Record<
87
- string,
88
- { type: string; enum?: string[] }
89
- >;
90
- expect(props.click_type.type).toBe("string");
91
- expect(props.click_type.enum).toEqual(["single", "double", "right"]);
92
- });
93
-
94
- test("schema supports element_id and coordinates", () => {
95
- const props = schema(computerUseClickTool).properties as Record<
96
- string,
97
- { type: string }
98
- >;
99
- expect(props.element_id.type).toBe("integer");
100
- expect(props.x.type).toBe("integer");
101
- expect(props.y.type).toBe("integer");
102
- });
103
-
104
- test("execute returns isError when no proxy resolver is configured", async () => {
105
- const result = await computerUseClickTool.execute({}, ctx);
106
- expect(result.isError).toBe(true);
107
- expect(result.content).toContain(
108
- "The Vellum desktop app is needed to view or control your screen",
109
- );
110
- expect(result.content).toContain("https://www.vellum.ai/downloads");
111
- });
112
- });
113
-
114
- // ── type_text ───────────────────────────────────────────────────────
115
-
116
- describe("computer_use_type_text", () => {
117
- test("requires text and reasoning", () => {
118
- expect(schema(computerUseTypeTextTool).required).toContain("text");
119
- expect(schema(computerUseTypeTextTool).required).toContain("reasoning");
120
- });
121
-
122
- test("execute returns isError when no proxy resolver is configured", async () => {
123
- const result = await computerUseTypeTextTool.execute({}, ctx);
124
- expect(result.isError).toBe(true);
125
- expect(result.content).toContain(
126
- "The Vellum desktop app is needed to view or control your screen",
127
- );
128
- expect(result.content).toContain("https://www.vellum.ai/downloads");
129
- });
130
- });
131
-
132
- // ── key ─────────────────────────────────────────────────────────────
133
-
134
- describe("computer_use_key", () => {
135
- test("requires key and reasoning", () => {
136
- expect(schema(computerUseKeyTool).required).toContain("key");
137
- expect(schema(computerUseKeyTool).required).toContain("reasoning");
138
- });
139
-
140
- test("execute returns isError when no proxy resolver is configured", async () => {
141
- const result = await computerUseKeyTool.execute({}, ctx);
142
- expect(result.isError).toBe(true);
143
- expect(result.content).toContain(
144
- "The Vellum desktop app is needed to view or control your screen",
145
- );
146
- expect(result.content).toContain("https://www.vellum.ai/downloads");
147
- });
148
- });
149
-
150
- // ── scroll ──────────────────────────────────────────────────────────
151
-
152
- describe("computer_use_scroll", () => {
153
- test("requires direction, amount, and reasoning", () => {
154
- expect(schema(computerUseScrollTool).required).toContain("direction");
155
- expect(schema(computerUseScrollTool).required).toContain("amount");
156
- expect(schema(computerUseScrollTool).required).toContain("reasoning");
157
- });
158
-
159
- test("direction enum includes up, down, left, right", () => {
160
- const props = schema(computerUseScrollTool).properties as Record<
161
- string,
162
- { enum?: string[] }
163
- >;
164
- expect(props.direction.enum).toEqual(["up", "down", "left", "right"]);
165
- });
166
- });
167
-
168
- // ── drag ────────────────────────────────────────────────────────────
169
-
170
- describe("computer_use_drag", () => {
171
- test("supports source and destination coordinates", () => {
172
- const props = schema(computerUseDragTool).properties as Record<
173
- string,
174
- { type: string }
175
- >;
176
- expect(props.element_id.type).toBe("integer");
177
- expect(props.to_element_id.type).toBe("integer");
178
- expect(props.x.type).toBe("integer");
179
- expect(props.y.type).toBe("integer");
180
- expect(props.to_x.type).toBe("integer");
181
- expect(props.to_y.type).toBe("integer");
182
- });
183
-
184
- test("requires reasoning only", () => {
185
- expect(schema(computerUseDragTool).required).toEqual(["reasoning"]);
186
- });
187
- });
188
-
189
- // ── wait ────────────────────────────────────────────────────────────
190
-
191
- describe("computer_use_wait", () => {
192
- test("requires duration_ms and reasoning", () => {
193
- expect(schema(computerUseWaitTool).required).toContain("duration_ms");
194
- expect(schema(computerUseWaitTool).required).toContain("reasoning");
195
- });
196
- });
197
-
198
- // ── open_app ────────────────────────────────────────────────────────
199
-
200
- describe("computer_use_open_app", () => {
201
- test("requires app_name and reasoning", () => {
202
- expect(schema(computerUseOpenAppTool).required).toContain("app_name");
203
- expect(schema(computerUseOpenAppTool).required).toContain("reasoning");
204
- });
205
- });
206
-
207
- // ── run_applescript ─────────────────────────────────────────────────
208
-
209
- describe("computer_use_run_applescript", () => {
210
- test("requires script and reasoning", () => {
211
- expect(schema(computerUseRunAppleScriptTool).required).toContain("script");
212
- expect(schema(computerUseRunAppleScriptTool).required).toContain(
213
- "reasoning",
214
- );
215
- });
216
-
217
- test("description warns against do shell script", () => {
218
- expect(computerUseRunAppleScriptTool.description).toContain(
219
- "do shell script",
220
- );
221
- expect(computerUseRunAppleScriptTool.description).toContain("blocked");
222
- });
223
- });
224
-
225
- // ── done ────────────────────────────────────────────────────────────
226
-
227
- describe("computer_use_done", () => {
228
- test("requires summary", () => {
229
- expect(schema(computerUseDoneTool).required).toContain("summary");
230
- });
231
- });
232
-
233
- // ── respond ─────────────────────────────────────────────────────────
234
-
235
- describe("computer_use_respond", () => {
236
- test("requires answer and reasoning", () => {
237
- expect(schema(computerUseRespondTool).required).toContain("answer");
238
- expect(schema(computerUseRespondTool).required).toContain("reasoning");
239
- });
240
- });
241
-
242
- // ── skill-proxy-bridge ──────────────────────────────────────────────
243
-
244
12
  describe("forwardComputerUseProxyTool", () => {
245
13
  test("returns error when no proxy resolver available", async () => {
246
14
  const result = await forwardComputerUseProxyTool(
@@ -10,6 +10,7 @@
10
10
  import { beforeEach, describe, expect, mock, spyOn, test } from "bun:test";
11
11
 
12
12
  import type { Conversation } from "../daemon/conversation.js";
13
+ import { shouldUseVirtualDesktop } from "../desktop/virtual-desktop-feature.js";
13
14
  import type { PermissionPrompter } from "../permissions/prompter.js";
14
15
  import type { SecretPrompter } from "../permissions/secret-prompter.js";
15
16
  import type { ToolExecutor } from "../tools/executor.js";
@@ -307,6 +308,28 @@ describe("createToolExecutor attribution threading", () => {
307
308
  });
308
309
  });
309
310
 
311
+ test("desktop routing honors the pinned interface when the live client changes", async () => {
312
+ for (const transportInterface of ["web", "macos"] as const) {
313
+ const { executor, calls } = makeCapturingExecutor();
314
+ await makeToolFn(
315
+ executor,
316
+ makeCtx({
317
+ transportInterface: transportInterface === "web" ? "macos" : "web",
318
+ toolContextPin: { hasNoClient: false, transportInterface },
319
+ getTurnActorPrincipalId: () => "user-123",
320
+ currentTurnTrustContext: {
321
+ sourceChannel: "vellum",
322
+ trustClass: "guardian",
323
+ },
324
+ }),
325
+ )("computer_use_observe", {});
326
+
327
+ expect(shouldUseVirtualDesktop(calls[0].context, true)).toBe(
328
+ transportInterface === "web",
329
+ );
330
+ }
331
+ });
332
+
310
333
  describe("createToolExecutor isInteractive threading", () => {
311
334
  test("uses the resolved turn-level interactivity over live client state", async () => {
312
335
  // A scheduled/background turn (currentTurnIsNonInteractive=true) must read
@@ -21,6 +21,7 @@ import {
21
21
  abortedToolResultText,
22
22
  CANCELLED_TOOL_RESULT_TEXT,
23
23
  createAbortReason,
24
+ INTERRUPTED_TURN_NOTE_TEXT,
24
25
  PREEMPTED_TOOL_RESULT_TEXT,
25
26
  } from "../util/abort-reasons.js";
26
27
 
@@ -378,8 +379,8 @@ describe("interruptRunningTurn", () => {
378
379
 
379
380
  test("arms the interrupted-turn note when the abort caught no tool call", async () => {
380
381
  // The abort landed during the provider call, so there is no `tool_use` to
381
- // answer and nothing in the history says the turn was cut off. The note
382
- // rides on the interrupting user message instead.
382
+ // answer. The note on the interrupting user message is the only thing that
383
+ // tells the model what happened.
383
384
  const turn = registerBusyTurn({
384
385
  messages: [
385
386
  { role: "user", content: [{ type: "text", text: "research this" }] },
@@ -393,9 +394,9 @@ describe("interruptRunningTurn", () => {
393
394
  expect(persisted).toEqual([]);
394
395
  });
395
396
 
396
- test("leaves the note disarmed when the repair answered a tool call", async () => {
397
- // The preempted `tool_result` already tells the model a message cut it
398
- // off, so repeating that in a note on the message is noise.
397
+ test("arms the note alongside the fact-only result of a stopped tool call", async () => {
398
+ // The two texts do different jobs: the `tool_result` states what happened
399
+ // to that one call, the note says what to do about the interruption.
399
400
  const turn = registerBusyTurn({
400
401
  messages: [assistantWithToolUse("tool-1")],
401
402
  });
@@ -404,13 +405,36 @@ describe("interruptRunningTurn", () => {
404
405
 
405
406
  expect(persisted).toHaveLength(1);
406
407
  expect(persisted[0].content).toContain(PREEMPTED_TOOL_RESULT_TEXT);
407
- expect(turn.conversation.pendingInterruptNote).toBe(false);
408
+ expect(turn.conversation.pendingInterruptNote).toBe(true);
409
+ });
410
+
411
+ test("arms exactly one note however many parallel calls were stopped", async () => {
412
+ // Every stopped call gets its own fact-only result; the behavior
413
+ // instruction is read once, off the interrupting message.
414
+ const turn = registerBusyTurn({
415
+ messages: [assistantWithToolUse("tool-1", "tool-2", "tool-3")],
416
+ });
417
+
418
+ await interruptRunningTurn(turn.conversation, { origin: "test" });
419
+
420
+ const results = JSON.parse(persisted[0].content) as Array<{
421
+ tool_use_id: string;
422
+ content: string;
423
+ }>;
424
+ expect(results.map((block) => block.tool_use_id)).toEqual([
425
+ "tool-1",
426
+ "tool-2",
427
+ "tool-3",
428
+ ]);
429
+ for (const block of results) {
430
+ expect(block.content).toBe(PREEMPTED_TOOL_RESULT_TEXT);
431
+ }
432
+ expect(turn.conversation.pendingInterruptNote).toBe(true);
408
433
  });
409
434
 
410
- test("leaves the note disarmed when the loop wrote its own preempted result", async () => {
435
+ test("arms the note when the loop wrote its own preempted result", async () => {
411
436
  // The agent loop's abort handler unwound cleanly and answered the batch
412
- // itself, so the repair finds nothing to do and the notice is already
413
- // there.
437
+ // itself, so the repair finds nothing to do. The note is owed all the same.
414
438
  const turn = registerBusyTurn({
415
439
  messages: [
416
440
  assistantWithToolUse("tool-1"),
@@ -421,12 +445,10 @@ describe("interruptRunningTurn", () => {
421
445
  await interruptRunningTurn(turn.conversation, { origin: "test" });
422
446
 
423
447
  expect(persisted).toEqual([]);
424
- expect(turn.conversation.pendingInterruptNote).toBe(false);
448
+ expect(turn.conversation.pendingInterruptNote).toBe(true);
425
449
  });
426
450
 
427
451
  test("arms the note when the cut-off tool call was an ordinary cancel", async () => {
428
- // A tail answered with the plain cancel wording is a stop the user made
429
- // earlier, not this interrupt's notice, so the note is still owed.
430
452
  const turn = registerBusyTurn({
431
453
  messages: [
432
454
  assistantWithToolUse("tool-1"),
@@ -633,6 +655,9 @@ describe("interruptRunningTurn", () => {
633
655
  // Armed instead, so the drain that runs the queued message repairs it.
634
656
  expect(turn.conversation.pendingInterruptRepair).toBe(true);
635
657
  expect(turn.activityEvents).toEqual([]);
658
+ // The message joins the queue rather than replacing the turn, so it is not
659
+ // the message that interrupted anything and carries no note.
660
+ expect(turn.conversation.pendingInterruptNote).toBe(false);
636
661
  });
637
662
 
638
663
  test("a second interrupt stops the turn the first one started", async () => {
@@ -932,7 +957,7 @@ describe("repairInterruptedToolUseBlocks", () => {
932
957
  } as unknown as Conversation;
933
958
  }
934
959
 
935
- test("writes one result per abandoned call, worded for a preemption", async () => {
960
+ test("writes one fact-only result per abandoned call", async () => {
936
961
  const messages: Message[] = [assistantWithToolUse("tool-1", "tool-2")];
937
962
 
938
963
  await repairInterruptedToolUseBlocks(fakeConversation(messages), {
@@ -961,31 +986,6 @@ describe("repairInterruptedToolUseBlocks", () => {
961
986
  );
962
987
  });
963
988
 
964
- test("reports the preempted result it wrote", async () => {
965
- const messages: Message[] = [assistantWithToolUse("tool-1")];
966
-
967
- const result = await repairInterruptedToolUseBlocks(
968
- fakeConversation(messages),
969
- { force: true },
970
- );
971
-
972
- expect(result.preemptedToolResultOnTail).toBe(true);
973
- });
974
-
975
- test("reports no preempted result on a history that ends on plain text", async () => {
976
- const messages: Message[] = [
977
- { role: "user", content: [{ type: "text", text: "hi" }] },
978
- { role: "assistant", content: [{ type: "text", text: "hello" }] },
979
- ];
980
-
981
- const result = await repairInterruptedToolUseBlocks(
982
- fakeConversation(messages),
983
- { force: true },
984
- );
985
-
986
- expect(result.preemptedToolResultOnTail).toBe(false);
987
- });
988
-
989
989
  test("adds nothing when the loop already wrote its own results", async () => {
990
990
  // The agent loop's abort handler synthesizes results of its own when it
991
991
  // unwinds cleanly, so exactly one result per call exists either way.
@@ -994,14 +994,12 @@ describe("repairInterruptedToolUseBlocks", () => {
994
994
  toolResult("tool-1", PREEMPTED_TOOL_RESULT_TEXT),
995
995
  ];
996
996
 
997
- const result = await repairInterruptedToolUseBlocks(
998
- fakeConversation(messages),
999
- { force: true },
1000
- );
997
+ await repairInterruptedToolUseBlocks(fakeConversation(messages), {
998
+ force: true,
999
+ });
1001
1000
 
1002
1001
  expect(messages).toHaveLength(2);
1003
1002
  expect(persisted).toEqual([]);
1004
- expect(result.preemptedToolResultOnTail).toBe(true);
1005
1003
  });
1006
1004
 
1007
1005
  test("does nothing on a history that ends on an ordinary assistant reply", async () => {
@@ -1083,3 +1081,19 @@ describe("abortedToolResultText", () => {
1083
1081
  );
1084
1082
  });
1085
1083
  });
1084
+
1085
+ describe("the interrupt texts", () => {
1086
+ test("the tool result states the fact and nothing more", () => {
1087
+ // Repeated once per stopped call in a parallel batch, so an instruction
1088
+ // here is read as many times as the batch was wide.
1089
+ expect(PREEMPTED_TOOL_RESULT_TEXT).toBe(
1090
+ "Stopped early because the user sent a new message. This call may have completed anyway; check before repeating it.",
1091
+ );
1092
+ });
1093
+
1094
+ test("the note carries the behavior and leads with the reply", () => {
1095
+ expect(INTERRUPTED_TURN_NOTE_TEXT).toBe(
1096
+ "<interrupted_turn>The user sent this while you were still working, so that work stopped. Reply right away, in a line or two, before any more thinking or tool calls, so the user knows you heard them. Then pick the earlier work back up only if it is still wanted. Never apologize for or mention the interruption.</interrupted_turn>",
1097
+ );
1098
+ });
1099
+ });
@@ -1,10 +1,9 @@
1
1
  /**
2
- * An interrupt whose abort landed during the provider call leaves no
3
- * `tool_use` to answer, so nothing in the history tells the model its turn was
4
- * cut off. The interrupting user message carries the notice instead, on its
5
- * LLM-facing content only: the persisted row stays exactly what the user
6
- * typed, so no client renders it, and `interruptedPriorTurn` in the row's
7
- * metadata is what rebuilds the identical block on every later load.
2
+ * The message that interrupted a turn carries the note saying what to do about
3
+ * the interruption, on its LLM-facing content only: the persisted row stays
4
+ * exactly what the user typed, so no client renders it, and
5
+ * `interruptedPriorTurn` in the row's metadata is what rebuilds the identical
6
+ * block on every later load.
8
7
  */
9
8
  import { beforeEach, describe, expect, test } from "bun:test";
10
9