@vellumai/assistant 0.12.2-dev.202609182314.c515b72 → 0.12.2-dev.202609190043.ee5ba34
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/docs/desktop-browser-cli.md +20 -1
- package/node_modules/@vellumai/gateway-client/src/gateway-ipc-contracts.ts +20 -0
- package/openapi.yaml +8 -0
- package/package.json +1 -1
- package/src/__tests__/checker.test.ts +11 -3
- package/src/__tests__/computer-use-skill-manifest-regression.test.ts +0 -25
- package/src/__tests__/computer-use-tools.test.ts +1 -233
- package/src/__tests__/conversation-tool-setup-attribution.test.ts +23 -0
- package/src/__tests__/migration-export-to-gcs.test.ts +142 -0
- package/src/__tests__/plugin-api-provider.test.ts +5 -0
- package/src/__tests__/settings-routes.test.ts +50 -1
- package/src/activation/progress-store.test.ts +12 -7
- package/src/browser/virtual-desktop-target.ts +2 -8
- package/src/config/bundled-skills/computer-use/SKILL.md +1 -1
- package/src/config/bundled-skills/computer-use/TOOLS.json +89 -1
- package/src/daemon/conversation-surfaces.ts +27 -1
- package/src/daemon/conversation-tool-setup.ts +7 -2
- package/src/daemon/host-cu-proxy.ts +30 -19
- package/src/daemon/virtual-desktop-context.ts +21 -0
- package/src/desktop/__tests__/fake-desktop.ts +1 -0
- package/src/desktop/desktop-accessibility-script.ts +88 -0
- package/src/desktop/desktop-accessibility.test.ts +160 -0
- package/src/desktop/desktop-accessibility.ts +169 -0
- package/src/desktop/desktop-automation-lease.test.ts +71 -0
- package/src/desktop/desktop-automation-lease.ts +40 -12
- package/src/desktop/desktop-browser-endpoint.ts +1 -0
- package/src/desktop/desktop-computer-use-routing.test.ts +330 -0
- package/src/desktop/desktop-computer-use.test.ts +450 -0
- package/src/desktop/desktop-computer-use.ts +503 -0
- package/src/desktop/desktop-dependencies.ts +10 -2
- package/src/desktop/desktop-session-manager.test.ts +63 -18
- package/src/desktop/desktop-session-manager.ts +68 -11
- package/src/desktop/virtual-desktop-feature.ts +52 -2
- package/src/live-voice/__tests__/live-voice-vad.test.ts +15 -9
- package/src/plugin-api/effective-context-window.ts +33 -0
- package/src/plugin-api/index.ts +9 -2
- package/src/plugins/defaults/memory/v3/__tests__/pool-log-store.test.ts +165 -27
- package/src/plugins/defaults/memory/v3/__tests__/pool-select.test.ts +511 -30
- package/src/plugins/defaults/memory/v3/__tests__/shadow-plugin.test.ts +46 -27
- package/src/plugins/defaults/memory/v3/orchestrate.ts +75 -37
- package/src/plugins/defaults/memory/v3/pool-log-store.ts +65 -79
- package/src/plugins/defaults/memory/v3/pool-select.test.ts +14 -5
- package/src/plugins/defaults/memory/v3/pool-select.ts +447 -65
- package/src/runtime/migrations/__tests__/vbundle-import-parity.test.ts +67 -1
- package/src/runtime/migrations/vbundle-builder.ts +24 -1
- package/src/runtime/migrations/vbundle-import-analyzer.ts +13 -4
- package/src/runtime/migrations/vbundle-import-policy.ts +10 -0
- package/src/runtime/migrations/vbundle-importer.ts +4 -1
- package/src/runtime/migrations/vbundle-streaming-importer.ts +4 -1
- package/src/runtime/migrations/vbundle-validator.ts +1 -0
- package/src/runtime/routes/migration-routes.ts +73 -2
- package/src/runtime/routes/settings-routes.ts +10 -4
- package/src/tools/client-os.ts +15 -0
- package/src/tools/computer-use/skill-proxy-bridge.ts +1 -1
- package/src/tools/computer-use/target.ts +42 -0
- package/src/tools/execution-target.ts +14 -16
- package/src/tools/permission-checker.ts +3 -3
- package/src/tools/policy-context.ts +3 -1
- package/src/tools/skills/load.ts +2 -1
- package/src/tools/skills/skill-tool-factory.ts +11 -0
- package/src/tools/tool-approval-handler.ts +5 -1
- package/src/tools/types.ts +7 -1
- package/src/__tests__/computer-use-window-schema.test.ts +0 -24
- package/src/tools/computer-use/definitions.ts +0 -607
|
@@ -24,7 +24,7 @@ The desktop client bypasses personal-browser discovery, extension reconnect wait
|
|
|
24
24
|
|
|
25
25
|
Tab IDs are ephemeral numeric aliases for this managed browser's CDP target IDs. They are never personal extension tab IDs. Initial attachment selects an existing HTTP(S) page or opens a blank tab. Use `tabs list` and `tabs select --tab-id <id>` to choose explicitly. Navigation, tab selection, document replacement and release clear the desktop snapshot map. Personal-browser snapshots use a separate namespace.
|
|
26
26
|
|
|
27
|
-
CDP mouse events use page viewport CSS coordinates. Before dispatching them, the client animates a purple arrow overlay to the same point. The overlay is excluded from accessibility and hit testing. It appears in the existing desktop stream and is removed on release. It does not move the OS pointer, appear in browser toolbar UI or visualize every programmatic DOM operation. Native dialogs and other applications
|
|
27
|
+
CDP mouse events use page viewport CSS coordinates. Before dispatching them, the client animates a purple arrow overlay to the same point. The overlay is excluded from accessibility and hit testing. It appears in the existing desktop stream and is removed on release. It does not move the OS pointer, appear in browser toolbar UI or visualize every programmatic DOM operation. Native dialogs and other applications use the computer-use skill or direct interaction in the expanded desktop. The picture-in-picture preview is view-only. Direct user interaction does not pause automation. For logins, the assistant uses saved credentials first and securely collects missing credentials with `assistant credentials prompt`, then fills the login form itself. Every CAPTCHA and bot-detection challenge, including sliders and press-and-hold checks, is handed to the user before the assistant attempts it. When human interaction is needed, the assistant calls `ask_question` with `desktopHelp: { message, doneLabel, skipLabel }` in the user's language. The tool releases held browser input and reserves the desktop before publishing a question with `presentation: "virtual_desktop"`. The live viewer mounts only in the attended Electron window, with browser focus as a fallback on older shells and the web, so conversations release it when focus changes. A busy viewer offers Reconnect if the previous connection has not finished closing. The compact card lives in the scrolling message transcript, with expandable instructions and a preview capped at 384 pixels wide. It hosts the shared live viewer, with Step In opening its full desktop modal and Done or Skip resolving the existing question response. The assistant retains an exclusive human-help reservation while waiting. Done resumes the same conversation's automation lease for its fresh snapshot; Skip, closure, timeout and cancellation release it. The browser idle timer does not expire a pending human-help reservation. The live preview connects after Show live preview or Step In, and reuses an already-open picture-in-picture viewer. Other devices receiving the card do not automatically open a stream. When the request resolves, its interactive viewer closes or returns to the read-only PiP session that was open before the request. Skip does not imply the obstacle was resolved. Older clients show the same request as a normal Done/Skip question, with model-localized labels retained in history. Channels without dynamic UI return guidance to continue in the app rather than waiting on an unusable card.
|
|
28
28
|
|
|
29
29
|
The client records key and mouse presses before dispatch. On release it opens a fresh bounded cleanup connection, attaches to the same live targets, releases uncertain held input and removes overlays. A failed cleanup preserves state and the automation slot for retry. Turn-triggered cancellation retries cleanup automatically until it succeeds or ownership changes. Dispatched actions are never automatically retried. Closed targets need no input cleanup. Browser-process loss disposes the client, and later requests discover the replacement process.
|
|
30
30
|
|
|
@@ -35,3 +35,22 @@ Validation: focused client tests exercise shared snapshot/click behavior, namesp
|
|
|
35
35
|
The desktop header icon pulses in the assistant's avatar accent color while a browser automation lease is active, including between browser commands. It shares the progress indicator's accent and neutral fallback. Reduced-motion clients show a solid accent. Setup status exposes the optional `automationActive` field; `desktop_activity_changed` events refresh it on acquisition and cancellation or release. Installation progress retains its `assistant:self:desktop` sync invalidations. The indicator reads status without starting installation, and reconnects refetch the current lease state.
|
|
36
36
|
|
|
37
37
|
Desktop streaming checks the current in-memory gateway feature flags during connection startup. Once connected, frame and drain callbacks do not check feature flags or load workspace configuration. Turning the flag off prevents new connections; an existing stream continues until it closes or the desktop stops.
|
|
38
|
+
|
|
39
|
+
## Computer use on the same desktop
|
|
40
|
+
|
|
41
|
+
The existing `computer-use` skill defaults to this desktop in platform-hosted
|
|
42
|
+
web conversations. Native desktop clients, including the Mac app connected to a
|
|
43
|
+
platform-hosted assistant, keep the connected-computer default and require
|
|
44
|
+
`target: "assistant-desktop"` to select the virtual desktop explicitly.
|
|
45
|
+
Non-platform assistants keep their connected-computer behavior.
|
|
46
|
+
|
|
47
|
+
Explicit `target: "connected-computer"` or `target_client_id` selects a connected
|
|
48
|
+
computer on every client. The selected target never falls back to another
|
|
49
|
+
computer when unavailable. Virtual desktop observations return full-screen
|
|
50
|
+
screenshots and mouse actions use screen coordinates.
|
|
51
|
+
|
|
52
|
+
Observe first and pass the returned `observation_id` with each native action.
|
|
53
|
+
Browser commands, user handoff, errors, and interruption require a fresh
|
|
54
|
+
observation before returning to native input. Browser commands and computer use
|
|
55
|
+
share session ownership, cancellation, and the user-help reservation. Use
|
|
56
|
+
browser commands for page-level operations and computer use for desktop UI.
|
|
@@ -43,6 +43,26 @@ export type GatewayLogsTailRouteParams = z.infer<
|
|
|
43
43
|
typeof GatewayLogsTailRouteParamsSchema
|
|
44
44
|
>;
|
|
45
45
|
|
|
46
|
+
/**
|
|
47
|
+
* `gateway_debug_export`: a consistent copy of the gateway database plus its
|
|
48
|
+
* recent log files, as one gzipped tar, for a debug bundle. No params.
|
|
49
|
+
*/
|
|
50
|
+
export const GatewayDebugExportIpcParamsSchema = z
|
|
51
|
+
.object({})
|
|
52
|
+
.strict()
|
|
53
|
+
.default({});
|
|
54
|
+
|
|
55
|
+
export const GatewayDebugExportIpcResponseSchema = z.object({
|
|
56
|
+
ok: z.literal(true),
|
|
57
|
+
/** `tar.gz` bytes, base64. Contains gateway.sqlite and gateway-logs/. */
|
|
58
|
+
archive_base64: z.string(),
|
|
59
|
+
size_bytes: z.number().int().nonnegative(),
|
|
60
|
+
});
|
|
61
|
+
|
|
62
|
+
export type GatewayDebugExportIpcResponse = z.infer<
|
|
63
|
+
typeof GatewayDebugExportIpcResponseSchema
|
|
64
|
+
>;
|
|
65
|
+
|
|
46
66
|
export const GatewayLogsTailIpcResponseSchema = z.object({
|
|
47
67
|
lines: z.array(z.record(z.string(), z.unknown())),
|
|
48
68
|
truncated: z.boolean(),
|
package/openapi.yaml
CHANGED
|
@@ -23549,6 +23549,14 @@ paths:
|
|
|
23549
23549
|
description:
|
|
23550
23550
|
description: Human-readable export description.
|
|
23551
23551
|
type: string
|
|
23552
|
+
profile:
|
|
23553
|
+
description:
|
|
23554
|
+
"Export profile. 'migration' (default) builds a teleport bundle. 'debug' builds a bundle for Vellum staff
|
|
23555
|
+
to open on a debug clone: no credentials, plus the gateway's database and logs under gateway/."
|
|
23556
|
+
type: string
|
|
23557
|
+
enum:
|
|
23558
|
+
- migration
|
|
23559
|
+
- debug
|
|
23552
23560
|
required:
|
|
23553
23561
|
- upload_url
|
|
23554
23562
|
responses:
|
package/package.json
CHANGED
|
@@ -140,9 +140,17 @@ registerSkillTools("app-builder", [mockBundledSkillTool]);
|
|
|
140
140
|
|
|
141
141
|
// Register CU tools so check() can look them up in the tool registry
|
|
142
142
|
// instead of falling through to Medium (unknown tool).
|
|
143
|
-
import {
|
|
144
|
-
for (const
|
|
145
|
-
registerTool(
|
|
143
|
+
import { COMPUTER_USE_TOOL_NAMES } from "./test-support/computer-use-skill-harness.js";
|
|
144
|
+
for (const name of COMPUTER_USE_TOOL_NAMES) {
|
|
145
|
+
registerTool({
|
|
146
|
+
name,
|
|
147
|
+
description: name,
|
|
148
|
+
category: "computer-use",
|
|
149
|
+
defaultRiskLevel: RiskLevel.Low,
|
|
150
|
+
executionTarget: "host",
|
|
151
|
+
input_schema: { type: "object", properties: {} },
|
|
152
|
+
execute: async () => ({ content: "ok", isError: false }),
|
|
153
|
+
});
|
|
146
154
|
}
|
|
147
155
|
|
|
148
156
|
function writeSkill(
|
|
@@ -3,7 +3,6 @@ import { resolve } from "node:path";
|
|
|
3
3
|
import { afterAll, describe, expect, test } from "bun:test";
|
|
4
4
|
|
|
5
5
|
import { RiskLevel } from "../permissions/types.js";
|
|
6
|
-
import { allComputerUseTools } from "../tools/computer-use/definitions.js";
|
|
7
6
|
import {
|
|
8
7
|
__resetRegistryForTesting,
|
|
9
8
|
getTool,
|
|
@@ -74,30 +73,6 @@ describe("computer-use skill manifest regression", () => {
|
|
|
74
73
|
}
|
|
75
74
|
});
|
|
76
75
|
|
|
77
|
-
test("manifest descriptions match core definitions", async () => {
|
|
78
|
-
await initializeTools();
|
|
79
|
-
|
|
80
|
-
for (const cuTool of allComputerUseTools) {
|
|
81
|
-
const manifestTool = manifest.tools.find(
|
|
82
|
-
(t: { name: string }) => t.name === cuTool.name,
|
|
83
|
-
);
|
|
84
|
-
expect(manifestTool).toBeDefined();
|
|
85
|
-
expect(manifestTool.description).toBe(cuTool.description);
|
|
86
|
-
}
|
|
87
|
-
});
|
|
88
|
-
|
|
89
|
-
test("manifest input_schema matches core definitions", async () => {
|
|
90
|
-
await initializeTools();
|
|
91
|
-
|
|
92
|
-
for (const cuTool of allComputerUseTools) {
|
|
93
|
-
const manifestTool = manifest.tools.find(
|
|
94
|
-
(t: { name: string }) => t.name === cuTool.name,
|
|
95
|
-
);
|
|
96
|
-
expect(manifestTool).toBeDefined();
|
|
97
|
-
expect(manifestTool.input_schema).toEqual(cuTool.input_schema);
|
|
98
|
-
}
|
|
99
|
-
});
|
|
100
|
-
|
|
101
76
|
test("CU action tools are not registered as core tools after initializeTools()", async () => {
|
|
102
77
|
await initializeTools();
|
|
103
78
|
|
|
@@ -1,246 +1,14 @@
|
|
|
1
1
|
import { describe, expect, test } from "bun:test";
|
|
2
2
|
|
|
3
|
-
import {
|
|
4
|
-
allComputerUseTools,
|
|
5
|
-
computerUseClickTool,
|
|
6
|
-
computerUseDoneTool,
|
|
7
|
-
computerUseDragTool,
|
|
8
|
-
computerUseKeyTool,
|
|
9
|
-
computerUseObserveTool,
|
|
10
|
-
computerUseOpenAppTool,
|
|
11
|
-
computerUseRespondTool,
|
|
12
|
-
computerUseRunAppleScriptTool,
|
|
13
|
-
computerUseScrollTool,
|
|
14
|
-
computerUseTypeTextTool,
|
|
15
|
-
computerUseWaitTool,
|
|
16
|
-
} from "../tools/computer-use/definitions.js";
|
|
17
3
|
import { forwardComputerUseProxyTool } from "../tools/computer-use/skill-proxy-bridge.js";
|
|
18
4
|
import type { ToolContext } from "../tools/types.js";
|
|
19
5
|
|
|
20
|
-
interface JsonSchema {
|
|
21
|
-
type?: string;
|
|
22
|
-
required?: string[];
|
|
23
|
-
properties?: Record<string, unknown>;
|
|
24
|
-
}
|
|
25
|
-
|
|
26
|
-
/** Cast a tool definition's input_schema to a usable JSON Schema shape. */
|
|
27
|
-
function schema(tool: { input_schema?: object }): JsonSchema {
|
|
28
|
-
return tool.input_schema as JsonSchema;
|
|
29
|
-
}
|
|
30
|
-
|
|
31
6
|
const ctx: ToolContext = {
|
|
32
7
|
workingDir: "/tmp",
|
|
33
|
-
conversationId: "
|
|
8
|
+
conversationId: "conv-123",
|
|
34
9
|
trustClass: "guardian",
|
|
35
10
|
};
|
|
36
11
|
|
|
37
|
-
// ── Tool definitions ────────────────────────────────────────────────
|
|
38
|
-
|
|
39
|
-
describe("computer-use tool definitions", () => {
|
|
40
|
-
test("allComputerUseTools contains 12 tools", () => {
|
|
41
|
-
expect(allComputerUseTools.length).toBe(12);
|
|
42
|
-
});
|
|
43
|
-
|
|
44
|
-
test("all tools belong to computer-use category", () => {
|
|
45
|
-
for (const tool of allComputerUseTools) {
|
|
46
|
-
expect(tool.category).toBe("computer-use");
|
|
47
|
-
}
|
|
48
|
-
});
|
|
49
|
-
|
|
50
|
-
test("all tools have unique names", () => {
|
|
51
|
-
const names = allComputerUseTools.map((t) => t.name);
|
|
52
|
-
expect(new Set(names).size).toBe(names.length);
|
|
53
|
-
});
|
|
54
|
-
|
|
55
|
-
test("all tools have descriptions", () => {
|
|
56
|
-
for (const tool of allComputerUseTools) {
|
|
57
|
-
expect(tool.description!.length).toBeGreaterThan(0);
|
|
58
|
-
}
|
|
59
|
-
});
|
|
60
|
-
});
|
|
61
|
-
|
|
62
|
-
// ── observe ─────────────────────────────────────────────────────────
|
|
63
|
-
|
|
64
|
-
describe("computer_use_observe", () => {
|
|
65
|
-
test("supports target_client_id", () => {
|
|
66
|
-
const props = schema(computerUseObserveTool).properties as Record<
|
|
67
|
-
string,
|
|
68
|
-
{ type: string }
|
|
69
|
-
>;
|
|
70
|
-
expect(props.target_client_id.type).toBe("string");
|
|
71
|
-
});
|
|
72
|
-
});
|
|
73
|
-
|
|
74
|
-
// ── Unified click tool ──────────────────────────────────────────────
|
|
75
|
-
|
|
76
|
-
describe("computer_use_click (unified)", () => {
|
|
77
|
-
test("has correct name", () => {
|
|
78
|
-
expect(computerUseClickTool.name).toBe("computer_use_click");
|
|
79
|
-
});
|
|
80
|
-
|
|
81
|
-
test("schema requires reasoning", () => {
|
|
82
|
-
expect(schema(computerUseClickTool).required).toContain("reasoning");
|
|
83
|
-
});
|
|
84
|
-
|
|
85
|
-
test("schema supports click_type enum", () => {
|
|
86
|
-
const props = schema(computerUseClickTool).properties as Record<
|
|
87
|
-
string,
|
|
88
|
-
{ type: string; enum?: string[] }
|
|
89
|
-
>;
|
|
90
|
-
expect(props.click_type.type).toBe("string");
|
|
91
|
-
expect(props.click_type.enum).toEqual(["single", "double", "right"]);
|
|
92
|
-
});
|
|
93
|
-
|
|
94
|
-
test("schema supports element_id and coordinates", () => {
|
|
95
|
-
const props = schema(computerUseClickTool).properties as Record<
|
|
96
|
-
string,
|
|
97
|
-
{ type: string }
|
|
98
|
-
>;
|
|
99
|
-
expect(props.element_id.type).toBe("integer");
|
|
100
|
-
expect(props.x.type).toBe("integer");
|
|
101
|
-
expect(props.y.type).toBe("integer");
|
|
102
|
-
});
|
|
103
|
-
|
|
104
|
-
test("execute returns isError when no proxy resolver is configured", async () => {
|
|
105
|
-
const result = await computerUseClickTool.execute({}, ctx);
|
|
106
|
-
expect(result.isError).toBe(true);
|
|
107
|
-
expect(result.content).toContain(
|
|
108
|
-
"The Vellum desktop app is needed to view or control your screen",
|
|
109
|
-
);
|
|
110
|
-
expect(result.content).toContain("https://www.vellum.ai/downloads");
|
|
111
|
-
});
|
|
112
|
-
});
|
|
113
|
-
|
|
114
|
-
// ── type_text ───────────────────────────────────────────────────────
|
|
115
|
-
|
|
116
|
-
describe("computer_use_type_text", () => {
|
|
117
|
-
test("requires text and reasoning", () => {
|
|
118
|
-
expect(schema(computerUseTypeTextTool).required).toContain("text");
|
|
119
|
-
expect(schema(computerUseTypeTextTool).required).toContain("reasoning");
|
|
120
|
-
});
|
|
121
|
-
|
|
122
|
-
test("execute returns isError when no proxy resolver is configured", async () => {
|
|
123
|
-
const result = await computerUseTypeTextTool.execute({}, ctx);
|
|
124
|
-
expect(result.isError).toBe(true);
|
|
125
|
-
expect(result.content).toContain(
|
|
126
|
-
"The Vellum desktop app is needed to view or control your screen",
|
|
127
|
-
);
|
|
128
|
-
expect(result.content).toContain("https://www.vellum.ai/downloads");
|
|
129
|
-
});
|
|
130
|
-
});
|
|
131
|
-
|
|
132
|
-
// ── key ─────────────────────────────────────────────────────────────
|
|
133
|
-
|
|
134
|
-
describe("computer_use_key", () => {
|
|
135
|
-
test("requires key and reasoning", () => {
|
|
136
|
-
expect(schema(computerUseKeyTool).required).toContain("key");
|
|
137
|
-
expect(schema(computerUseKeyTool).required).toContain("reasoning");
|
|
138
|
-
});
|
|
139
|
-
|
|
140
|
-
test("execute returns isError when no proxy resolver is configured", async () => {
|
|
141
|
-
const result = await computerUseKeyTool.execute({}, ctx);
|
|
142
|
-
expect(result.isError).toBe(true);
|
|
143
|
-
expect(result.content).toContain(
|
|
144
|
-
"The Vellum desktop app is needed to view or control your screen",
|
|
145
|
-
);
|
|
146
|
-
expect(result.content).toContain("https://www.vellum.ai/downloads");
|
|
147
|
-
});
|
|
148
|
-
});
|
|
149
|
-
|
|
150
|
-
// ── scroll ──────────────────────────────────────────────────────────
|
|
151
|
-
|
|
152
|
-
describe("computer_use_scroll", () => {
|
|
153
|
-
test("requires direction, amount, and reasoning", () => {
|
|
154
|
-
expect(schema(computerUseScrollTool).required).toContain("direction");
|
|
155
|
-
expect(schema(computerUseScrollTool).required).toContain("amount");
|
|
156
|
-
expect(schema(computerUseScrollTool).required).toContain("reasoning");
|
|
157
|
-
});
|
|
158
|
-
|
|
159
|
-
test("direction enum includes up, down, left, right", () => {
|
|
160
|
-
const props = schema(computerUseScrollTool).properties as Record<
|
|
161
|
-
string,
|
|
162
|
-
{ enum?: string[] }
|
|
163
|
-
>;
|
|
164
|
-
expect(props.direction.enum).toEqual(["up", "down", "left", "right"]);
|
|
165
|
-
});
|
|
166
|
-
});
|
|
167
|
-
|
|
168
|
-
// ── drag ────────────────────────────────────────────────────────────
|
|
169
|
-
|
|
170
|
-
describe("computer_use_drag", () => {
|
|
171
|
-
test("supports source and destination coordinates", () => {
|
|
172
|
-
const props = schema(computerUseDragTool).properties as Record<
|
|
173
|
-
string,
|
|
174
|
-
{ type: string }
|
|
175
|
-
>;
|
|
176
|
-
expect(props.element_id.type).toBe("integer");
|
|
177
|
-
expect(props.to_element_id.type).toBe("integer");
|
|
178
|
-
expect(props.x.type).toBe("integer");
|
|
179
|
-
expect(props.y.type).toBe("integer");
|
|
180
|
-
expect(props.to_x.type).toBe("integer");
|
|
181
|
-
expect(props.to_y.type).toBe("integer");
|
|
182
|
-
});
|
|
183
|
-
|
|
184
|
-
test("requires reasoning only", () => {
|
|
185
|
-
expect(schema(computerUseDragTool).required).toEqual(["reasoning"]);
|
|
186
|
-
});
|
|
187
|
-
});
|
|
188
|
-
|
|
189
|
-
// ── wait ────────────────────────────────────────────────────────────
|
|
190
|
-
|
|
191
|
-
describe("computer_use_wait", () => {
|
|
192
|
-
test("requires duration_ms and reasoning", () => {
|
|
193
|
-
expect(schema(computerUseWaitTool).required).toContain("duration_ms");
|
|
194
|
-
expect(schema(computerUseWaitTool).required).toContain("reasoning");
|
|
195
|
-
});
|
|
196
|
-
});
|
|
197
|
-
|
|
198
|
-
// ── open_app ────────────────────────────────────────────────────────
|
|
199
|
-
|
|
200
|
-
describe("computer_use_open_app", () => {
|
|
201
|
-
test("requires app_name and reasoning", () => {
|
|
202
|
-
expect(schema(computerUseOpenAppTool).required).toContain("app_name");
|
|
203
|
-
expect(schema(computerUseOpenAppTool).required).toContain("reasoning");
|
|
204
|
-
});
|
|
205
|
-
});
|
|
206
|
-
|
|
207
|
-
// ── run_applescript ─────────────────────────────────────────────────
|
|
208
|
-
|
|
209
|
-
describe("computer_use_run_applescript", () => {
|
|
210
|
-
test("requires script and reasoning", () => {
|
|
211
|
-
expect(schema(computerUseRunAppleScriptTool).required).toContain("script");
|
|
212
|
-
expect(schema(computerUseRunAppleScriptTool).required).toContain(
|
|
213
|
-
"reasoning",
|
|
214
|
-
);
|
|
215
|
-
});
|
|
216
|
-
|
|
217
|
-
test("description warns against do shell script", () => {
|
|
218
|
-
expect(computerUseRunAppleScriptTool.description).toContain(
|
|
219
|
-
"do shell script",
|
|
220
|
-
);
|
|
221
|
-
expect(computerUseRunAppleScriptTool.description).toContain("blocked");
|
|
222
|
-
});
|
|
223
|
-
});
|
|
224
|
-
|
|
225
|
-
// ── done ────────────────────────────────────────────────────────────
|
|
226
|
-
|
|
227
|
-
describe("computer_use_done", () => {
|
|
228
|
-
test("requires summary", () => {
|
|
229
|
-
expect(schema(computerUseDoneTool).required).toContain("summary");
|
|
230
|
-
});
|
|
231
|
-
});
|
|
232
|
-
|
|
233
|
-
// ── respond ─────────────────────────────────────────────────────────
|
|
234
|
-
|
|
235
|
-
describe("computer_use_respond", () => {
|
|
236
|
-
test("requires answer and reasoning", () => {
|
|
237
|
-
expect(schema(computerUseRespondTool).required).toContain("answer");
|
|
238
|
-
expect(schema(computerUseRespondTool).required).toContain("reasoning");
|
|
239
|
-
});
|
|
240
|
-
});
|
|
241
|
-
|
|
242
|
-
// ── skill-proxy-bridge ──────────────────────────────────────────────
|
|
243
|
-
|
|
244
12
|
describe("forwardComputerUseProxyTool", () => {
|
|
245
13
|
test("returns error when no proxy resolver available", async () => {
|
|
246
14
|
const result = await forwardComputerUseProxyTool(
|
|
@@ -10,6 +10,7 @@
|
|
|
10
10
|
import { beforeEach, describe, expect, mock, spyOn, test } from "bun:test";
|
|
11
11
|
|
|
12
12
|
import type { Conversation } from "../daemon/conversation.js";
|
|
13
|
+
import { shouldUseVirtualDesktop } from "../desktop/virtual-desktop-feature.js";
|
|
13
14
|
import type { PermissionPrompter } from "../permissions/prompter.js";
|
|
14
15
|
import type { SecretPrompter } from "../permissions/secret-prompter.js";
|
|
15
16
|
import type { ToolExecutor } from "../tools/executor.js";
|
|
@@ -307,6 +308,28 @@ describe("createToolExecutor attribution threading", () => {
|
|
|
307
308
|
});
|
|
308
309
|
});
|
|
309
310
|
|
|
311
|
+
test("desktop routing honors the pinned interface when the live client changes", async () => {
|
|
312
|
+
for (const transportInterface of ["web", "macos"] as const) {
|
|
313
|
+
const { executor, calls } = makeCapturingExecutor();
|
|
314
|
+
await makeToolFn(
|
|
315
|
+
executor,
|
|
316
|
+
makeCtx({
|
|
317
|
+
transportInterface: transportInterface === "web" ? "macos" : "web",
|
|
318
|
+
toolContextPin: { hasNoClient: false, transportInterface },
|
|
319
|
+
getTurnActorPrincipalId: () => "user-123",
|
|
320
|
+
currentTurnTrustContext: {
|
|
321
|
+
sourceChannel: "vellum",
|
|
322
|
+
trustClass: "guardian",
|
|
323
|
+
},
|
|
324
|
+
}),
|
|
325
|
+
)("computer_use_observe", {});
|
|
326
|
+
|
|
327
|
+
expect(shouldUseVirtualDesktop(calls[0].context, true)).toBe(
|
|
328
|
+
transportInterface === "web",
|
|
329
|
+
);
|
|
330
|
+
}
|
|
331
|
+
});
|
|
332
|
+
|
|
310
333
|
describe("createToolExecutor isInteractive threading", () => {
|
|
311
334
|
test("uses the resolved turn-level interactivity over live client state", async () => {
|
|
312
335
|
// A scheduled/background turn (currentTurnIsNonInteractive=true) must read
|
|
@@ -37,6 +37,23 @@ const testDbDir = join(testDir, "data", "db");
|
|
|
37
37
|
const testDbPath = join(testDbDir, "assistant.db");
|
|
38
38
|
const testConfigPath = join(testDir, "config.json");
|
|
39
39
|
|
|
40
|
+
// The gateway's debug export arrives over IPC as a base64 tar.gz. Stand in
|
|
41
|
+
// for the socket so the debug profile can be exercised without a gateway.
|
|
42
|
+
const FAKE_GATEWAY_ARCHIVE = Buffer.from("not really a tar.gz");
|
|
43
|
+
const gatewayDebugExportMock = mock(async () => ({
|
|
44
|
+
ok: true as const,
|
|
45
|
+
archive_base64: FAKE_GATEWAY_ARCHIVE.toString("base64"),
|
|
46
|
+
size_bytes: FAKE_GATEWAY_ARCHIVE.length,
|
|
47
|
+
}));
|
|
48
|
+
mock.module("../ipc/gateway-client.js", () => ({
|
|
49
|
+
ipcCallPersistent: (method: string) => {
|
|
50
|
+
if (method !== "gateway_debug_export") {
|
|
51
|
+
throw new Error(`unexpected IPC method ${method}`);
|
|
52
|
+
}
|
|
53
|
+
return gatewayDebugExportMock();
|
|
54
|
+
},
|
|
55
|
+
}));
|
|
56
|
+
|
|
40
57
|
mock.module("../permissions/trust-store.js", () => ({
|
|
41
58
|
getAllRules: () => [],
|
|
42
59
|
isStarterBundleAccepted: () => false,
|
|
@@ -328,6 +345,131 @@ describe("handleMigrationExportToGcs — happy path", () => {
|
|
|
328
345
|
}, 15_000);
|
|
329
346
|
});
|
|
330
347
|
|
|
348
|
+
describe("handleMigrationExportToGcs — debug profile", () => {
|
|
349
|
+
test("adds the gateway's archive under gateway/ and marks the manifest", async () => {
|
|
350
|
+
gatewayDebugExportMock.mockClear();
|
|
351
|
+
let capturedBody: Buffer | undefined;
|
|
352
|
+
const fixture = await startFixtureServer(async (req, res) => {
|
|
353
|
+
capturedBody = await collectBody(req);
|
|
354
|
+
res.writeHead(200);
|
|
355
|
+
res.end();
|
|
356
|
+
});
|
|
357
|
+
|
|
358
|
+
try {
|
|
359
|
+
const req = new Request("http://localhost/v1/migrations/export-to-gcs", {
|
|
360
|
+
method: "POST",
|
|
361
|
+
headers: { "Content-Type": "application/json" },
|
|
362
|
+
body: JSON.stringify({
|
|
363
|
+
upload_url: makeFakeSignedUploadUrl(fixture.port),
|
|
364
|
+
profile: "debug",
|
|
365
|
+
}),
|
|
366
|
+
});
|
|
367
|
+
const res = await callHandler(
|
|
368
|
+
handleMigrationExportToGcs,
|
|
369
|
+
req,
|
|
370
|
+
undefined,
|
|
371
|
+
202,
|
|
372
|
+
);
|
|
373
|
+
const body = (await res.json()) as AcceptedResponse;
|
|
374
|
+
const terminal = await waitForJobTerminal(body.job_id);
|
|
375
|
+
expect(terminal.status).toBe("complete");
|
|
376
|
+
|
|
377
|
+
expect(gatewayDebugExportMock).toHaveBeenCalledTimes(1);
|
|
378
|
+
const validation = validateVBundle(new Uint8Array(capturedBody!));
|
|
379
|
+
expect(validation.is_valid).toBe(true);
|
|
380
|
+
const manifest = validation.manifest!;
|
|
381
|
+
expect(manifest.export_options.include_gateway).toBe(true);
|
|
382
|
+
const gatewayEntry = manifest.contents.find(
|
|
383
|
+
(f) => f.path === "gateway/export.tar.gz",
|
|
384
|
+
);
|
|
385
|
+
expect(gatewayEntry?.size_bytes).toBe(FAKE_GATEWAY_ARCHIVE.length);
|
|
386
|
+
// Staff never receive credentials. The store mock reports itself
|
|
387
|
+
// unreachable, which would force `secrets_redacted: false` had the
|
|
388
|
+
// handler tried to collect them.
|
|
389
|
+
expect(manifest.secrets_redacted).toBe(true);
|
|
390
|
+
expect(
|
|
391
|
+
manifest.contents.some((f) => f.path.startsWith("credentials/")),
|
|
392
|
+
).toBe(false);
|
|
393
|
+
} finally {
|
|
394
|
+
await fixture.close();
|
|
395
|
+
}
|
|
396
|
+
}, 15_000);
|
|
397
|
+
|
|
398
|
+
test("the default profile never asks the gateway", async () => {
|
|
399
|
+
gatewayDebugExportMock.mockClear();
|
|
400
|
+
let capturedBody: Buffer | undefined;
|
|
401
|
+
const fixture = await startFixtureServer(async (req, res) => {
|
|
402
|
+
capturedBody = await collectBody(req);
|
|
403
|
+
res.writeHead(200);
|
|
404
|
+
res.end();
|
|
405
|
+
});
|
|
406
|
+
|
|
407
|
+
try {
|
|
408
|
+
const req = new Request("http://localhost/v1/migrations/export-to-gcs", {
|
|
409
|
+
method: "POST",
|
|
410
|
+
headers: { "Content-Type": "application/json" },
|
|
411
|
+
body: JSON.stringify({
|
|
412
|
+
upload_url: makeFakeSignedUploadUrl(fixture.port),
|
|
413
|
+
}),
|
|
414
|
+
});
|
|
415
|
+
const res = await callHandler(
|
|
416
|
+
handleMigrationExportToGcs,
|
|
417
|
+
req,
|
|
418
|
+
undefined,
|
|
419
|
+
202,
|
|
420
|
+
);
|
|
421
|
+
const body = (await res.json()) as AcceptedResponse;
|
|
422
|
+
const terminal = await waitForJobTerminal(body.job_id);
|
|
423
|
+
expect(terminal.status).toBe("complete");
|
|
424
|
+
|
|
425
|
+
expect(gatewayDebugExportMock).not.toHaveBeenCalled();
|
|
426
|
+
const manifest = validateVBundle(new Uint8Array(capturedBody!)).manifest!;
|
|
427
|
+
expect(manifest.export_options.include_gateway).toBeUndefined();
|
|
428
|
+
expect(manifest.contents.some((f) => f.path.startsWith("gateway/"))).toBe(
|
|
429
|
+
false,
|
|
430
|
+
);
|
|
431
|
+
} finally {
|
|
432
|
+
await fixture.close();
|
|
433
|
+
}
|
|
434
|
+
}, 15_000);
|
|
435
|
+
|
|
436
|
+
test("a debug bundle is not sent at all if the gateway cannot export", async () => {
|
|
437
|
+
gatewayDebugExportMock.mockImplementationOnce(async () => {
|
|
438
|
+
throw new Error("gateway socket missing");
|
|
439
|
+
});
|
|
440
|
+
let putCount = 0;
|
|
441
|
+
const fixture = await startFixtureServer(async (_req, res) => {
|
|
442
|
+
putCount += 1;
|
|
443
|
+
res.writeHead(200);
|
|
444
|
+
res.end();
|
|
445
|
+
});
|
|
446
|
+
|
|
447
|
+
try {
|
|
448
|
+
const req = new Request("http://localhost/v1/migrations/export-to-gcs", {
|
|
449
|
+
method: "POST",
|
|
450
|
+
headers: { "Content-Type": "application/json" },
|
|
451
|
+
body: JSON.stringify({
|
|
452
|
+
upload_url: makeFakeSignedUploadUrl(fixture.port),
|
|
453
|
+
profile: "debug",
|
|
454
|
+
}),
|
|
455
|
+
});
|
|
456
|
+
const res = await callHandler(
|
|
457
|
+
handleMigrationExportToGcs,
|
|
458
|
+
req,
|
|
459
|
+
undefined,
|
|
460
|
+
202,
|
|
461
|
+
);
|
|
462
|
+
const body = (await res.json()) as { job_id: string };
|
|
463
|
+
const terminal = await waitForJobTerminal(body.job_id);
|
|
464
|
+
expect(terminal.status).toBe("failed");
|
|
465
|
+
expect(terminal.error?.code).toBe("gateway_debug_export_failed");
|
|
466
|
+
expect(putCount).toBe(0);
|
|
467
|
+
} finally {
|
|
468
|
+
await fixture.close();
|
|
469
|
+
}
|
|
470
|
+
}, 15_000);
|
|
471
|
+
});
|
|
472
|
+
|
|
331
473
|
describe("handleMigrationExportToGcs — concurrency", () => {
|
|
332
474
|
test("second export while first is running returns 409 export_in_progress", async () => {
|
|
333
475
|
// First fixture: deliberately slow so the first job stays pending long
|
|
@@ -21,4 +21,9 @@ describe("plugin-api provider access", () => {
|
|
|
21
21
|
// registry), not a disjoint, uninitialized module copy.
|
|
22
22
|
expect(PLUGIN_API_EXPORTS).toContain("getConfiguredProvider");
|
|
23
23
|
});
|
|
24
|
+
|
|
25
|
+
test("getEffectiveContextWindow is exported and shim-rebound", () => {
|
|
26
|
+
expect(typeof pluginApi.getEffectiveContextWindow).toBe("function");
|
|
27
|
+
expect(PLUGIN_API_EXPORTS).toContain("getEffectiveContextWindow");
|
|
28
|
+
});
|
|
24
29
|
});
|