@vellumai/assistant 0.12.2-dev.202609182314.c515b72 → 0.12.2-dev.202609190043.ee5ba34

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (64) hide show
  1. package/docs/desktop-browser-cli.md +20 -1
  2. package/node_modules/@vellumai/gateway-client/src/gateway-ipc-contracts.ts +20 -0
  3. package/openapi.yaml +8 -0
  4. package/package.json +1 -1
  5. package/src/__tests__/checker.test.ts +11 -3
  6. package/src/__tests__/computer-use-skill-manifest-regression.test.ts +0 -25
  7. package/src/__tests__/computer-use-tools.test.ts +1 -233
  8. package/src/__tests__/conversation-tool-setup-attribution.test.ts +23 -0
  9. package/src/__tests__/migration-export-to-gcs.test.ts +142 -0
  10. package/src/__tests__/plugin-api-provider.test.ts +5 -0
  11. package/src/__tests__/settings-routes.test.ts +50 -1
  12. package/src/activation/progress-store.test.ts +12 -7
  13. package/src/browser/virtual-desktop-target.ts +2 -8
  14. package/src/config/bundled-skills/computer-use/SKILL.md +1 -1
  15. package/src/config/bundled-skills/computer-use/TOOLS.json +89 -1
  16. package/src/daemon/conversation-surfaces.ts +27 -1
  17. package/src/daemon/conversation-tool-setup.ts +7 -2
  18. package/src/daemon/host-cu-proxy.ts +30 -19
  19. package/src/daemon/virtual-desktop-context.ts +21 -0
  20. package/src/desktop/__tests__/fake-desktop.ts +1 -0
  21. package/src/desktop/desktop-accessibility-script.ts +88 -0
  22. package/src/desktop/desktop-accessibility.test.ts +160 -0
  23. package/src/desktop/desktop-accessibility.ts +169 -0
  24. package/src/desktop/desktop-automation-lease.test.ts +71 -0
  25. package/src/desktop/desktop-automation-lease.ts +40 -12
  26. package/src/desktop/desktop-browser-endpoint.ts +1 -0
  27. package/src/desktop/desktop-computer-use-routing.test.ts +330 -0
  28. package/src/desktop/desktop-computer-use.test.ts +450 -0
  29. package/src/desktop/desktop-computer-use.ts +503 -0
  30. package/src/desktop/desktop-dependencies.ts +10 -2
  31. package/src/desktop/desktop-session-manager.test.ts +63 -18
  32. package/src/desktop/desktop-session-manager.ts +68 -11
  33. package/src/desktop/virtual-desktop-feature.ts +52 -2
  34. package/src/live-voice/__tests__/live-voice-vad.test.ts +15 -9
  35. package/src/plugin-api/effective-context-window.ts +33 -0
  36. package/src/plugin-api/index.ts +9 -2
  37. package/src/plugins/defaults/memory/v3/__tests__/pool-log-store.test.ts +165 -27
  38. package/src/plugins/defaults/memory/v3/__tests__/pool-select.test.ts +511 -30
  39. package/src/plugins/defaults/memory/v3/__tests__/shadow-plugin.test.ts +46 -27
  40. package/src/plugins/defaults/memory/v3/orchestrate.ts +75 -37
  41. package/src/plugins/defaults/memory/v3/pool-log-store.ts +65 -79
  42. package/src/plugins/defaults/memory/v3/pool-select.test.ts +14 -5
  43. package/src/plugins/defaults/memory/v3/pool-select.ts +447 -65
  44. package/src/runtime/migrations/__tests__/vbundle-import-parity.test.ts +67 -1
  45. package/src/runtime/migrations/vbundle-builder.ts +24 -1
  46. package/src/runtime/migrations/vbundle-import-analyzer.ts +13 -4
  47. package/src/runtime/migrations/vbundle-import-policy.ts +10 -0
  48. package/src/runtime/migrations/vbundle-importer.ts +4 -1
  49. package/src/runtime/migrations/vbundle-streaming-importer.ts +4 -1
  50. package/src/runtime/migrations/vbundle-validator.ts +1 -0
  51. package/src/runtime/routes/migration-routes.ts +73 -2
  52. package/src/runtime/routes/settings-routes.ts +10 -4
  53. package/src/tools/client-os.ts +15 -0
  54. package/src/tools/computer-use/skill-proxy-bridge.ts +1 -1
  55. package/src/tools/computer-use/target.ts +42 -0
  56. package/src/tools/execution-target.ts +14 -16
  57. package/src/tools/permission-checker.ts +3 -3
  58. package/src/tools/policy-context.ts +3 -1
  59. package/src/tools/skills/load.ts +2 -1
  60. package/src/tools/skills/skill-tool-factory.ts +11 -0
  61. package/src/tools/tool-approval-handler.ts +5 -1
  62. package/src/tools/types.ts +7 -1
  63. package/src/__tests__/computer-use-window-schema.test.ts +0 -24
  64. package/src/tools/computer-use/definitions.ts +0 -607
@@ -24,7 +24,7 @@ The desktop client bypasses personal-browser discovery, extension reconnect wait
24
24
 
25
25
  Tab IDs are ephemeral numeric aliases for this managed browser's CDP target IDs. They are never personal extension tab IDs. Initial attachment selects an existing HTTP(S) page or opens a blank tab. Use `tabs list` and `tabs select --tab-id <id>` to choose explicitly. Navigation, tab selection, document replacement and release clear the desktop snapshot map. Personal-browser snapshots use a separate namespace.
26
26
 
27
- CDP mouse events use page viewport CSS coordinates. Before dispatching them, the client animates a purple arrow overlay to the same point. The overlay is excluded from accessibility and hit testing. It appears in the existing desktop stream and is removed on release. It does not move the OS pointer, appear in browser toolbar UI or visualize every programmatic DOM operation. Native dialogs and other applications are outside the browser CLI; users can interact with them directly in the expanded desktop. The picture-in-picture preview is view-only. Direct user interaction does not pause automation. For logins, the assistant uses saved credentials first and securely collects missing credentials with `assistant credentials prompt`, then fills the login form itself. Every CAPTCHA and bot-detection challenge, including sliders and press-and-hold checks, is handed to the user before the assistant attempts it. When human interaction is needed, the assistant calls `ask_question` with `desktopHelp: { message, doneLabel, skipLabel }` in the user's language. The tool releases held browser input and reserves the desktop before publishing a question with `presentation: "virtual_desktop"`. The live viewer mounts only in the attended Electron window, with browser focus as a fallback on older shells and the web, so conversations release it when focus changes. A busy viewer offers Reconnect if the previous connection has not finished closing. The compact card lives in the scrolling message transcript, with expandable instructions and a preview capped at 384 pixels wide. It hosts the shared live viewer, with Step In opening its full desktop modal and Done or Skip resolving the existing question response. The assistant retains an exclusive human-help reservation while waiting. Done resumes the same conversation's automation lease for its fresh snapshot; Skip, closure, timeout and cancellation release it. The browser idle timer does not expire a pending human-help reservation. The live preview connects after Show live preview or Step In, and reuses an already-open picture-in-picture viewer. Other devices receiving the card do not automatically open a stream. When the request resolves, its interactive viewer closes or returns to the read-only PiP session that was open before the request. Skip does not imply the obstacle was resolved. Older clients show the same request as a normal Done/Skip question, with model-localized labels retained in history. Channels without dynamic UI return guidance to continue in the app rather than waiting on an unusable card.
27
+ CDP mouse events use page viewport CSS coordinates. Before dispatching them, the client animates a purple arrow overlay to the same point. The overlay is excluded from accessibility and hit testing. It appears in the existing desktop stream and is removed on release. It does not move the OS pointer, appear in browser toolbar UI or visualize every programmatic DOM operation. Native dialogs and other applications use the computer-use skill or direct interaction in the expanded desktop. The picture-in-picture preview is view-only. Direct user interaction does not pause automation. For logins, the assistant uses saved credentials first and securely collects missing credentials with `assistant credentials prompt`, then fills the login form itself. Every CAPTCHA and bot-detection challenge, including sliders and press-and-hold checks, is handed to the user before the assistant attempts it. When human interaction is needed, the assistant calls `ask_question` with `desktopHelp: { message, doneLabel, skipLabel }` in the user's language. The tool releases held browser input and reserves the desktop before publishing a question with `presentation: "virtual_desktop"`. The live viewer mounts only in the attended Electron window, with browser focus as a fallback on older shells and the web, so conversations release it when focus changes. A busy viewer offers Reconnect if the previous connection has not finished closing. The compact card lives in the scrolling message transcript, with expandable instructions and a preview capped at 384 pixels wide. It hosts the shared live viewer, with Step In opening its full desktop modal and Done or Skip resolving the existing question response. The assistant retains an exclusive human-help reservation while waiting. Done resumes the same conversation's automation lease for its fresh snapshot; Skip, closure, timeout and cancellation release it. The browser idle timer does not expire a pending human-help reservation. The live preview connects after Show live preview or Step In, and reuses an already-open picture-in-picture viewer. Other devices receiving the card do not automatically open a stream. When the request resolves, its interactive viewer closes or returns to the read-only PiP session that was open before the request. Skip does not imply the obstacle was resolved. Older clients show the same request as a normal Done/Skip question, with model-localized labels retained in history. Channels without dynamic UI return guidance to continue in the app rather than waiting on an unusable card.
28
28
 
29
29
  The client records key and mouse presses before dispatch. On release it opens a fresh bounded cleanup connection, attaches to the same live targets, releases uncertain held input and removes overlays. A failed cleanup preserves state and the automation slot for retry. Turn-triggered cancellation retries cleanup automatically until it succeeds or ownership changes. Dispatched actions are never automatically retried. Closed targets need no input cleanup. Browser-process loss disposes the client, and later requests discover the replacement process.
30
30
 
@@ -35,3 +35,22 @@ Validation: focused client tests exercise shared snapshot/click behavior, namesp
35
35
  The desktop header icon pulses in the assistant's avatar accent color while a browser automation lease is active, including between browser commands. It shares the progress indicator's accent and neutral fallback. Reduced-motion clients show a solid accent. Setup status exposes the optional `automationActive` field; `desktop_activity_changed` events refresh it on acquisition and cancellation or release. Installation progress retains its `assistant:self:desktop` sync invalidations. The indicator reads status without starting installation, and reconnects refetch the current lease state.
36
36
 
37
37
  Desktop streaming checks the current in-memory gateway feature flags during connection startup. Once connected, frame and drain callbacks do not check feature flags or load workspace configuration. Turning the flag off prevents new connections; an existing stream continues until it closes or the desktop stops.
38
+
39
+ ## Computer use on the same desktop
40
+
41
+ The existing `computer-use` skill defaults to this desktop in platform-hosted
42
+ web conversations. Native desktop clients, including the Mac app connected to a
43
+ platform-hosted assistant, keep the connected-computer default and require
44
+ `target: "assistant-desktop"` to select the virtual desktop explicitly.
45
+ Non-platform assistants keep their connected-computer behavior.
46
+
47
+ Explicit `target: "connected-computer"` or `target_client_id` selects a connected
48
+ computer on every client. The selected target never falls back to another
49
+ computer when unavailable. Virtual desktop observations return full-screen
50
+ screenshots and mouse actions use screen coordinates.
51
+
52
+ Observe first and pass the returned `observation_id` with each native action.
53
+ Browser commands, user handoff, errors, and interruption require a fresh
54
+ observation before returning to native input. Browser commands and computer use
55
+ share session ownership, cancellation, and the user-help reservation. Use
56
+ browser commands for page-level operations and computer use for desktop UI.
@@ -43,6 +43,26 @@ export type GatewayLogsTailRouteParams = z.infer<
43
43
  typeof GatewayLogsTailRouteParamsSchema
44
44
  >;
45
45
 
46
+ /**
47
+ * `gateway_debug_export`: a consistent copy of the gateway database plus its
48
+ * recent log files, as one gzipped tar, for a debug bundle. No params.
49
+ */
50
+ export const GatewayDebugExportIpcParamsSchema = z
51
+ .object({})
52
+ .strict()
53
+ .default({});
54
+
55
+ export const GatewayDebugExportIpcResponseSchema = z.object({
56
+ ok: z.literal(true),
57
+ /** `tar.gz` bytes, base64. Contains gateway.sqlite and gateway-logs/. */
58
+ archive_base64: z.string(),
59
+ size_bytes: z.number().int().nonnegative(),
60
+ });
61
+
62
+ export type GatewayDebugExportIpcResponse = z.infer<
63
+ typeof GatewayDebugExportIpcResponseSchema
64
+ >;
65
+
46
66
  export const GatewayLogsTailIpcResponseSchema = z.object({
47
67
  lines: z.array(z.record(z.string(), z.unknown())),
48
68
  truncated: z.boolean(),
package/openapi.yaml CHANGED
@@ -23549,6 +23549,14 @@ paths:
23549
23549
  description:
23550
23550
  description: Human-readable export description.
23551
23551
  type: string
23552
+ profile:
23553
+ description:
23554
+ "Export profile. 'migration' (default) builds a teleport bundle. 'debug' builds a bundle for Vellum staff
23555
+ to open on a debug clone: no credentials, plus the gateway's database and logs under gateway/."
23556
+ type: string
23557
+ enum:
23558
+ - migration
23559
+ - debug
23552
23560
  required:
23553
23561
  - upload_url
23554
23562
  responses:
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@vellumai/assistant",
3
- "version": "0.12.2-dev.202609182314.c515b72",
3
+ "version": "0.12.2-dev.202609190043.ee5ba34",
4
4
  "license": "MIT",
5
5
  "type": "module",
6
6
  "exports": {
@@ -140,9 +140,17 @@ registerSkillTools("app-builder", [mockBundledSkillTool]);
140
140
 
141
141
  // Register CU tools so check() can look them up in the tool registry
142
142
  // instead of falling through to Medium (unknown tool).
143
- import { allComputerUseTools } from "../tools/computer-use/definitions.js";
144
- for (const tool of allComputerUseTools) {
145
- registerTool(tool);
143
+ import { COMPUTER_USE_TOOL_NAMES } from "./test-support/computer-use-skill-harness.js";
144
+ for (const name of COMPUTER_USE_TOOL_NAMES) {
145
+ registerTool({
146
+ name,
147
+ description: name,
148
+ category: "computer-use",
149
+ defaultRiskLevel: RiskLevel.Low,
150
+ executionTarget: "host",
151
+ input_schema: { type: "object", properties: {} },
152
+ execute: async () => ({ content: "ok", isError: false }),
153
+ });
146
154
  }
147
155
 
148
156
  function writeSkill(
@@ -3,7 +3,6 @@ import { resolve } from "node:path";
3
3
  import { afterAll, describe, expect, test } from "bun:test";
4
4
 
5
5
  import { RiskLevel } from "../permissions/types.js";
6
- import { allComputerUseTools } from "../tools/computer-use/definitions.js";
7
6
  import {
8
7
  __resetRegistryForTesting,
9
8
  getTool,
@@ -74,30 +73,6 @@ describe("computer-use skill manifest regression", () => {
74
73
  }
75
74
  });
76
75
 
77
- test("manifest descriptions match core definitions", async () => {
78
- await initializeTools();
79
-
80
- for (const cuTool of allComputerUseTools) {
81
- const manifestTool = manifest.tools.find(
82
- (t: { name: string }) => t.name === cuTool.name,
83
- );
84
- expect(manifestTool).toBeDefined();
85
- expect(manifestTool.description).toBe(cuTool.description);
86
- }
87
- });
88
-
89
- test("manifest input_schema matches core definitions", async () => {
90
- await initializeTools();
91
-
92
- for (const cuTool of allComputerUseTools) {
93
- const manifestTool = manifest.tools.find(
94
- (t: { name: string }) => t.name === cuTool.name,
95
- );
96
- expect(manifestTool).toBeDefined();
97
- expect(manifestTool.input_schema).toEqual(cuTool.input_schema);
98
- }
99
- });
100
-
101
76
  test("CU action tools are not registered as core tools after initializeTools()", async () => {
102
77
  await initializeTools();
103
78
 
@@ -1,246 +1,14 @@
1
1
  import { describe, expect, test } from "bun:test";
2
2
 
3
- import {
4
- allComputerUseTools,
5
- computerUseClickTool,
6
- computerUseDoneTool,
7
- computerUseDragTool,
8
- computerUseKeyTool,
9
- computerUseObserveTool,
10
- computerUseOpenAppTool,
11
- computerUseRespondTool,
12
- computerUseRunAppleScriptTool,
13
- computerUseScrollTool,
14
- computerUseTypeTextTool,
15
- computerUseWaitTool,
16
- } from "../tools/computer-use/definitions.js";
17
3
  import { forwardComputerUseProxyTool } from "../tools/computer-use/skill-proxy-bridge.js";
18
4
  import type { ToolContext } from "../tools/types.js";
19
5
 
20
- interface JsonSchema {
21
- type?: string;
22
- required?: string[];
23
- properties?: Record<string, unknown>;
24
- }
25
-
26
- /** Cast a tool definition's input_schema to a usable JSON Schema shape. */
27
- function schema(tool: { input_schema?: object }): JsonSchema {
28
- return tool.input_schema as JsonSchema;
29
- }
30
-
31
6
  const ctx: ToolContext = {
32
7
  workingDir: "/tmp",
33
- conversationId: "test-conversation",
8
+ conversationId: "conv-123",
34
9
  trustClass: "guardian",
35
10
  };
36
11
 
37
- // ── Tool definitions ────────────────────────────────────────────────
38
-
39
- describe("computer-use tool definitions", () => {
40
- test("allComputerUseTools contains 12 tools", () => {
41
- expect(allComputerUseTools.length).toBe(12);
42
- });
43
-
44
- test("all tools belong to computer-use category", () => {
45
- for (const tool of allComputerUseTools) {
46
- expect(tool.category).toBe("computer-use");
47
- }
48
- });
49
-
50
- test("all tools have unique names", () => {
51
- const names = allComputerUseTools.map((t) => t.name);
52
- expect(new Set(names).size).toBe(names.length);
53
- });
54
-
55
- test("all tools have descriptions", () => {
56
- for (const tool of allComputerUseTools) {
57
- expect(tool.description!.length).toBeGreaterThan(0);
58
- }
59
- });
60
- });
61
-
62
- // ── observe ─────────────────────────────────────────────────────────
63
-
64
- describe("computer_use_observe", () => {
65
- test("supports target_client_id", () => {
66
- const props = schema(computerUseObserveTool).properties as Record<
67
- string,
68
- { type: string }
69
- >;
70
- expect(props.target_client_id.type).toBe("string");
71
- });
72
- });
73
-
74
- // ── Unified click tool ──────────────────────────────────────────────
75
-
76
- describe("computer_use_click (unified)", () => {
77
- test("has correct name", () => {
78
- expect(computerUseClickTool.name).toBe("computer_use_click");
79
- });
80
-
81
- test("schema requires reasoning", () => {
82
- expect(schema(computerUseClickTool).required).toContain("reasoning");
83
- });
84
-
85
- test("schema supports click_type enum", () => {
86
- const props = schema(computerUseClickTool).properties as Record<
87
- string,
88
- { type: string; enum?: string[] }
89
- >;
90
- expect(props.click_type.type).toBe("string");
91
- expect(props.click_type.enum).toEqual(["single", "double", "right"]);
92
- });
93
-
94
- test("schema supports element_id and coordinates", () => {
95
- const props = schema(computerUseClickTool).properties as Record<
96
- string,
97
- { type: string }
98
- >;
99
- expect(props.element_id.type).toBe("integer");
100
- expect(props.x.type).toBe("integer");
101
- expect(props.y.type).toBe("integer");
102
- });
103
-
104
- test("execute returns isError when no proxy resolver is configured", async () => {
105
- const result = await computerUseClickTool.execute({}, ctx);
106
- expect(result.isError).toBe(true);
107
- expect(result.content).toContain(
108
- "The Vellum desktop app is needed to view or control your screen",
109
- );
110
- expect(result.content).toContain("https://www.vellum.ai/downloads");
111
- });
112
- });
113
-
114
- // ── type_text ───────────────────────────────────────────────────────
115
-
116
- describe("computer_use_type_text", () => {
117
- test("requires text and reasoning", () => {
118
- expect(schema(computerUseTypeTextTool).required).toContain("text");
119
- expect(schema(computerUseTypeTextTool).required).toContain("reasoning");
120
- });
121
-
122
- test("execute returns isError when no proxy resolver is configured", async () => {
123
- const result = await computerUseTypeTextTool.execute({}, ctx);
124
- expect(result.isError).toBe(true);
125
- expect(result.content).toContain(
126
- "The Vellum desktop app is needed to view or control your screen",
127
- );
128
- expect(result.content).toContain("https://www.vellum.ai/downloads");
129
- });
130
- });
131
-
132
- // ── key ─────────────────────────────────────────────────────────────
133
-
134
- describe("computer_use_key", () => {
135
- test("requires key and reasoning", () => {
136
- expect(schema(computerUseKeyTool).required).toContain("key");
137
- expect(schema(computerUseKeyTool).required).toContain("reasoning");
138
- });
139
-
140
- test("execute returns isError when no proxy resolver is configured", async () => {
141
- const result = await computerUseKeyTool.execute({}, ctx);
142
- expect(result.isError).toBe(true);
143
- expect(result.content).toContain(
144
- "The Vellum desktop app is needed to view or control your screen",
145
- );
146
- expect(result.content).toContain("https://www.vellum.ai/downloads");
147
- });
148
- });
149
-
150
- // ── scroll ──────────────────────────────────────────────────────────
151
-
152
- describe("computer_use_scroll", () => {
153
- test("requires direction, amount, and reasoning", () => {
154
- expect(schema(computerUseScrollTool).required).toContain("direction");
155
- expect(schema(computerUseScrollTool).required).toContain("amount");
156
- expect(schema(computerUseScrollTool).required).toContain("reasoning");
157
- });
158
-
159
- test("direction enum includes up, down, left, right", () => {
160
- const props = schema(computerUseScrollTool).properties as Record<
161
- string,
162
- { enum?: string[] }
163
- >;
164
- expect(props.direction.enum).toEqual(["up", "down", "left", "right"]);
165
- });
166
- });
167
-
168
- // ── drag ────────────────────────────────────────────────────────────
169
-
170
- describe("computer_use_drag", () => {
171
- test("supports source and destination coordinates", () => {
172
- const props = schema(computerUseDragTool).properties as Record<
173
- string,
174
- { type: string }
175
- >;
176
- expect(props.element_id.type).toBe("integer");
177
- expect(props.to_element_id.type).toBe("integer");
178
- expect(props.x.type).toBe("integer");
179
- expect(props.y.type).toBe("integer");
180
- expect(props.to_x.type).toBe("integer");
181
- expect(props.to_y.type).toBe("integer");
182
- });
183
-
184
- test("requires reasoning only", () => {
185
- expect(schema(computerUseDragTool).required).toEqual(["reasoning"]);
186
- });
187
- });
188
-
189
- // ── wait ────────────────────────────────────────────────────────────
190
-
191
- describe("computer_use_wait", () => {
192
- test("requires duration_ms and reasoning", () => {
193
- expect(schema(computerUseWaitTool).required).toContain("duration_ms");
194
- expect(schema(computerUseWaitTool).required).toContain("reasoning");
195
- });
196
- });
197
-
198
- // ── open_app ────────────────────────────────────────────────────────
199
-
200
- describe("computer_use_open_app", () => {
201
- test("requires app_name and reasoning", () => {
202
- expect(schema(computerUseOpenAppTool).required).toContain("app_name");
203
- expect(schema(computerUseOpenAppTool).required).toContain("reasoning");
204
- });
205
- });
206
-
207
- // ── run_applescript ─────────────────────────────────────────────────
208
-
209
- describe("computer_use_run_applescript", () => {
210
- test("requires script and reasoning", () => {
211
- expect(schema(computerUseRunAppleScriptTool).required).toContain("script");
212
- expect(schema(computerUseRunAppleScriptTool).required).toContain(
213
- "reasoning",
214
- );
215
- });
216
-
217
- test("description warns against do shell script", () => {
218
- expect(computerUseRunAppleScriptTool.description).toContain(
219
- "do shell script",
220
- );
221
- expect(computerUseRunAppleScriptTool.description).toContain("blocked");
222
- });
223
- });
224
-
225
- // ── done ────────────────────────────────────────────────────────────
226
-
227
- describe("computer_use_done", () => {
228
- test("requires summary", () => {
229
- expect(schema(computerUseDoneTool).required).toContain("summary");
230
- });
231
- });
232
-
233
- // ── respond ─────────────────────────────────────────────────────────
234
-
235
- describe("computer_use_respond", () => {
236
- test("requires answer and reasoning", () => {
237
- expect(schema(computerUseRespondTool).required).toContain("answer");
238
- expect(schema(computerUseRespondTool).required).toContain("reasoning");
239
- });
240
- });
241
-
242
- // ── skill-proxy-bridge ──────────────────────────────────────────────
243
-
244
12
  describe("forwardComputerUseProxyTool", () => {
245
13
  test("returns error when no proxy resolver available", async () => {
246
14
  const result = await forwardComputerUseProxyTool(
@@ -10,6 +10,7 @@
10
10
  import { beforeEach, describe, expect, mock, spyOn, test } from "bun:test";
11
11
 
12
12
  import type { Conversation } from "../daemon/conversation.js";
13
+ import { shouldUseVirtualDesktop } from "../desktop/virtual-desktop-feature.js";
13
14
  import type { PermissionPrompter } from "../permissions/prompter.js";
14
15
  import type { SecretPrompter } from "../permissions/secret-prompter.js";
15
16
  import type { ToolExecutor } from "../tools/executor.js";
@@ -307,6 +308,28 @@ describe("createToolExecutor attribution threading", () => {
307
308
  });
308
309
  });
309
310
 
311
+ test("desktop routing honors the pinned interface when the live client changes", async () => {
312
+ for (const transportInterface of ["web", "macos"] as const) {
313
+ const { executor, calls } = makeCapturingExecutor();
314
+ await makeToolFn(
315
+ executor,
316
+ makeCtx({
317
+ transportInterface: transportInterface === "web" ? "macos" : "web",
318
+ toolContextPin: { hasNoClient: false, transportInterface },
319
+ getTurnActorPrincipalId: () => "user-123",
320
+ currentTurnTrustContext: {
321
+ sourceChannel: "vellum",
322
+ trustClass: "guardian",
323
+ },
324
+ }),
325
+ )("computer_use_observe", {});
326
+
327
+ expect(shouldUseVirtualDesktop(calls[0].context, true)).toBe(
328
+ transportInterface === "web",
329
+ );
330
+ }
331
+ });
332
+
310
333
  describe("createToolExecutor isInteractive threading", () => {
311
334
  test("uses the resolved turn-level interactivity over live client state", async () => {
312
335
  // A scheduled/background turn (currentTurnIsNonInteractive=true) must read
@@ -37,6 +37,23 @@ const testDbDir = join(testDir, "data", "db");
37
37
  const testDbPath = join(testDbDir, "assistant.db");
38
38
  const testConfigPath = join(testDir, "config.json");
39
39
 
40
+ // The gateway's debug export arrives over IPC as a base64 tar.gz. Stand in
41
+ // for the socket so the debug profile can be exercised without a gateway.
42
+ const FAKE_GATEWAY_ARCHIVE = Buffer.from("not really a tar.gz");
43
+ const gatewayDebugExportMock = mock(async () => ({
44
+ ok: true as const,
45
+ archive_base64: FAKE_GATEWAY_ARCHIVE.toString("base64"),
46
+ size_bytes: FAKE_GATEWAY_ARCHIVE.length,
47
+ }));
48
+ mock.module("../ipc/gateway-client.js", () => ({
49
+ ipcCallPersistent: (method: string) => {
50
+ if (method !== "gateway_debug_export") {
51
+ throw new Error(`unexpected IPC method ${method}`);
52
+ }
53
+ return gatewayDebugExportMock();
54
+ },
55
+ }));
56
+
40
57
  mock.module("../permissions/trust-store.js", () => ({
41
58
  getAllRules: () => [],
42
59
  isStarterBundleAccepted: () => false,
@@ -328,6 +345,131 @@ describe("handleMigrationExportToGcs — happy path", () => {
328
345
  }, 15_000);
329
346
  });
330
347
 
348
+ describe("handleMigrationExportToGcs — debug profile", () => {
349
+ test("adds the gateway's archive under gateway/ and marks the manifest", async () => {
350
+ gatewayDebugExportMock.mockClear();
351
+ let capturedBody: Buffer | undefined;
352
+ const fixture = await startFixtureServer(async (req, res) => {
353
+ capturedBody = await collectBody(req);
354
+ res.writeHead(200);
355
+ res.end();
356
+ });
357
+
358
+ try {
359
+ const req = new Request("http://localhost/v1/migrations/export-to-gcs", {
360
+ method: "POST",
361
+ headers: { "Content-Type": "application/json" },
362
+ body: JSON.stringify({
363
+ upload_url: makeFakeSignedUploadUrl(fixture.port),
364
+ profile: "debug",
365
+ }),
366
+ });
367
+ const res = await callHandler(
368
+ handleMigrationExportToGcs,
369
+ req,
370
+ undefined,
371
+ 202,
372
+ );
373
+ const body = (await res.json()) as AcceptedResponse;
374
+ const terminal = await waitForJobTerminal(body.job_id);
375
+ expect(terminal.status).toBe("complete");
376
+
377
+ expect(gatewayDebugExportMock).toHaveBeenCalledTimes(1);
378
+ const validation = validateVBundle(new Uint8Array(capturedBody!));
379
+ expect(validation.is_valid).toBe(true);
380
+ const manifest = validation.manifest!;
381
+ expect(manifest.export_options.include_gateway).toBe(true);
382
+ const gatewayEntry = manifest.contents.find(
383
+ (f) => f.path === "gateway/export.tar.gz",
384
+ );
385
+ expect(gatewayEntry?.size_bytes).toBe(FAKE_GATEWAY_ARCHIVE.length);
386
+ // Staff never receive credentials. The store mock reports itself
387
+ // unreachable, which would force `secrets_redacted: false` had the
388
+ // handler tried to collect them.
389
+ expect(manifest.secrets_redacted).toBe(true);
390
+ expect(
391
+ manifest.contents.some((f) => f.path.startsWith("credentials/")),
392
+ ).toBe(false);
393
+ } finally {
394
+ await fixture.close();
395
+ }
396
+ }, 15_000);
397
+
398
+ test("the default profile never asks the gateway", async () => {
399
+ gatewayDebugExportMock.mockClear();
400
+ let capturedBody: Buffer | undefined;
401
+ const fixture = await startFixtureServer(async (req, res) => {
402
+ capturedBody = await collectBody(req);
403
+ res.writeHead(200);
404
+ res.end();
405
+ });
406
+
407
+ try {
408
+ const req = new Request("http://localhost/v1/migrations/export-to-gcs", {
409
+ method: "POST",
410
+ headers: { "Content-Type": "application/json" },
411
+ body: JSON.stringify({
412
+ upload_url: makeFakeSignedUploadUrl(fixture.port),
413
+ }),
414
+ });
415
+ const res = await callHandler(
416
+ handleMigrationExportToGcs,
417
+ req,
418
+ undefined,
419
+ 202,
420
+ );
421
+ const body = (await res.json()) as AcceptedResponse;
422
+ const terminal = await waitForJobTerminal(body.job_id);
423
+ expect(terminal.status).toBe("complete");
424
+
425
+ expect(gatewayDebugExportMock).not.toHaveBeenCalled();
426
+ const manifest = validateVBundle(new Uint8Array(capturedBody!)).manifest!;
427
+ expect(manifest.export_options.include_gateway).toBeUndefined();
428
+ expect(manifest.contents.some((f) => f.path.startsWith("gateway/"))).toBe(
429
+ false,
430
+ );
431
+ } finally {
432
+ await fixture.close();
433
+ }
434
+ }, 15_000);
435
+
436
+ test("a debug bundle is not sent at all if the gateway cannot export", async () => {
437
+ gatewayDebugExportMock.mockImplementationOnce(async () => {
438
+ throw new Error("gateway socket missing");
439
+ });
440
+ let putCount = 0;
441
+ const fixture = await startFixtureServer(async (_req, res) => {
442
+ putCount += 1;
443
+ res.writeHead(200);
444
+ res.end();
445
+ });
446
+
447
+ try {
448
+ const req = new Request("http://localhost/v1/migrations/export-to-gcs", {
449
+ method: "POST",
450
+ headers: { "Content-Type": "application/json" },
451
+ body: JSON.stringify({
452
+ upload_url: makeFakeSignedUploadUrl(fixture.port),
453
+ profile: "debug",
454
+ }),
455
+ });
456
+ const res = await callHandler(
457
+ handleMigrationExportToGcs,
458
+ req,
459
+ undefined,
460
+ 202,
461
+ );
462
+ const body = (await res.json()) as { job_id: string };
463
+ const terminal = await waitForJobTerminal(body.job_id);
464
+ expect(terminal.status).toBe("failed");
465
+ expect(terminal.error?.code).toBe("gateway_debug_export_failed");
466
+ expect(putCount).toBe(0);
467
+ } finally {
468
+ await fixture.close();
469
+ }
470
+ }, 15_000);
471
+ });
472
+
331
473
  describe("handleMigrationExportToGcs — concurrency", () => {
332
474
  test("second export while first is running returns 409 export_in_progress", async () => {
333
475
  // First fixture: deliberately slow so the first job stays pending long
@@ -21,4 +21,9 @@ describe("plugin-api provider access", () => {
21
21
  // registry), not a disjoint, uninitialized module copy.
22
22
  expect(PLUGIN_API_EXPORTS).toContain("getConfiguredProvider");
23
23
  });
24
+
25
+ test("getEffectiveContextWindow is exported and shim-rebound", () => {
26
+ expect(typeof pluginApi.getEffectiveContextWindow).toBe("function");
27
+ expect(PLUGIN_API_EXPORTS).toContain("getEffectiveContextWindow");
28
+ });
24
29
  });