@vellumai/assistant 0.12.2-dev.202609182214.bd8f92f → 0.12.2-dev.202609190043.ee5ba34
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/docs/desktop-browser-cli.md +20 -1
- package/node_modules/@vellumai/gateway-client/src/gateway-ipc-contracts.ts +20 -0
- package/openapi.yaml +8 -0
- package/package.json +1 -1
- package/src/__tests__/checker.test.ts +11 -3
- package/src/__tests__/computer-use-skill-manifest-regression.test.ts +0 -25
- package/src/__tests__/computer-use-tools.test.ts +1 -233
- package/src/__tests__/conversation-tool-setup-attribution.test.ts +23 -0
- package/src/__tests__/interrupt-on-send.test.ts +57 -43
- package/src/__tests__/interrupt-turn-note.test.ts +5 -6
- package/src/__tests__/migration-export-to-gcs.test.ts +142 -0
- package/src/__tests__/plugin-api-provider.test.ts +5 -0
- package/src/__tests__/settings-routes.test.ts +50 -1
- package/src/activation/progress-store.test.ts +12 -7
- package/src/browser/virtual-desktop-target.ts +2 -8
- package/src/config/bundled-skills/computer-use/SKILL.md +1 -1
- package/src/config/bundled-skills/computer-use/TOOLS.json +89 -1
- package/src/daemon/conversation-interrupt-repair.ts +4 -43
- package/src/daemon/conversation-interrupt.ts +7 -11
- package/src/daemon/conversation-messaging.ts +6 -4
- package/src/daemon/conversation-surfaces.ts +27 -1
- package/src/daemon/conversation-tool-setup.ts +7 -2
- package/src/daemon/conversation.ts +13 -8
- package/src/daemon/host-cu-proxy.ts +30 -19
- package/src/daemon/virtual-desktop-context.ts +21 -0
- package/src/desktop/__tests__/fake-desktop.ts +1 -0
- package/src/desktop/desktop-accessibility-script.ts +88 -0
- package/src/desktop/desktop-accessibility.test.ts +160 -0
- package/src/desktop/desktop-accessibility.ts +169 -0
- package/src/desktop/desktop-automation-lease.test.ts +71 -0
- package/src/desktop/desktop-automation-lease.ts +40 -12
- package/src/desktop/desktop-browser-endpoint.ts +1 -0
- package/src/desktop/desktop-computer-use-routing.test.ts +330 -0
- package/src/desktop/desktop-computer-use.test.ts +450 -0
- package/src/desktop/desktop-computer-use.ts +503 -0
- package/src/desktop/desktop-dependencies.ts +10 -2
- package/src/desktop/desktop-session-manager.test.ts +63 -18
- package/src/desktop/desktop-session-manager.ts +68 -11
- package/src/desktop/virtual-desktop-feature.ts +52 -2
- package/src/live-voice/__tests__/live-voice-vad.test.ts +15 -9
- package/src/persistence/conversation-crud.ts +4 -4
- package/src/plugin-api/effective-context-window.ts +33 -0
- package/src/plugin-api/index.ts +9 -2
- package/src/plugins/defaults/memory/v3/__tests__/pool-log-store.test.ts +165 -27
- package/src/plugins/defaults/memory/v3/__tests__/pool-select.test.ts +511 -30
- package/src/plugins/defaults/memory/v3/__tests__/shadow-plugin.test.ts +46 -27
- package/src/plugins/defaults/memory/v3/orchestrate.ts +75 -37
- package/src/plugins/defaults/memory/v3/pool-log-store.ts +65 -79
- package/src/plugins/defaults/memory/v3/pool-select.test.ts +14 -5
- package/src/plugins/defaults/memory/v3/pool-select.ts +447 -65
- package/src/runtime/AGENTS.md +1 -1
- package/src/runtime/migrations/__tests__/vbundle-import-parity.test.ts +67 -1
- package/src/runtime/migrations/vbundle-builder.ts +24 -1
- package/src/runtime/migrations/vbundle-import-analyzer.ts +13 -4
- package/src/runtime/migrations/vbundle-import-policy.ts +10 -0
- package/src/runtime/migrations/vbundle-importer.ts +4 -1
- package/src/runtime/migrations/vbundle-streaming-importer.ts +4 -1
- package/src/runtime/migrations/vbundle-validator.ts +1 -0
- package/src/runtime/routes/migration-routes.ts +73 -2
- package/src/runtime/routes/settings-routes.ts +10 -4
- package/src/tools/client-os.ts +15 -0
- package/src/tools/computer-use/skill-proxy-bridge.ts +1 -1
- package/src/tools/computer-use/target.ts +42 -0
- package/src/tools/execution-target.ts +14 -16
- package/src/tools/permission-checker.ts +3 -3
- package/src/tools/policy-context.ts +3 -1
- package/src/tools/skills/load.ts +2 -1
- package/src/tools/skills/skill-tool-factory.ts +11 -0
- package/src/tools/tool-approval-handler.ts +5 -1
- package/src/tools/types.ts +7 -1
- package/src/util/abort-reasons.ts +18 -26
- package/src/__tests__/computer-use-window-schema.test.ts +0 -24
- package/src/tools/computer-use/definitions.ts +0 -607
|
@@ -24,7 +24,7 @@ The desktop client bypasses personal-browser discovery, extension reconnect wait
|
|
|
24
24
|
|
|
25
25
|
Tab IDs are ephemeral numeric aliases for this managed browser's CDP target IDs. They are never personal extension tab IDs. Initial attachment selects an existing HTTP(S) page or opens a blank tab. Use `tabs list` and `tabs select --tab-id <id>` to choose explicitly. Navigation, tab selection, document replacement and release clear the desktop snapshot map. Personal-browser snapshots use a separate namespace.
|
|
26
26
|
|
|
27
|
-
CDP mouse events use page viewport CSS coordinates. Before dispatching them, the client animates a purple arrow overlay to the same point. The overlay is excluded from accessibility and hit testing. It appears in the existing desktop stream and is removed on release. It does not move the OS pointer, appear in browser toolbar UI or visualize every programmatic DOM operation. Native dialogs and other applications
|
|
27
|
+
CDP mouse events use page viewport CSS coordinates. Before dispatching them, the client animates a purple arrow overlay to the same point. The overlay is excluded from accessibility and hit testing. It appears in the existing desktop stream and is removed on release. It does not move the OS pointer, appear in browser toolbar UI or visualize every programmatic DOM operation. Native dialogs and other applications use the computer-use skill or direct interaction in the expanded desktop. The picture-in-picture preview is view-only. Direct user interaction does not pause automation. For logins, the assistant uses saved credentials first and securely collects missing credentials with `assistant credentials prompt`, then fills the login form itself. Every CAPTCHA and bot-detection challenge, including sliders and press-and-hold checks, is handed to the user before the assistant attempts it. When human interaction is needed, the assistant calls `ask_question` with `desktopHelp: { message, doneLabel, skipLabel }` in the user's language. The tool releases held browser input and reserves the desktop before publishing a question with `presentation: "virtual_desktop"`. The live viewer mounts only in the attended Electron window, with browser focus as a fallback on older shells and the web, so conversations release it when focus changes. A busy viewer offers Reconnect if the previous connection has not finished closing. The compact card lives in the scrolling message transcript, with expandable instructions and a preview capped at 384 pixels wide. It hosts the shared live viewer, with Step In opening its full desktop modal and Done or Skip resolving the existing question response. The assistant retains an exclusive human-help reservation while waiting. Done resumes the same conversation's automation lease for its fresh snapshot; Skip, closure, timeout and cancellation release it. The browser idle timer does not expire a pending human-help reservation. The live preview connects after Show live preview or Step In, and reuses an already-open picture-in-picture viewer. Other devices receiving the card do not automatically open a stream. When the request resolves, its interactive viewer closes or returns to the read-only PiP session that was open before the request. Skip does not imply the obstacle was resolved. Older clients show the same request as a normal Done/Skip question, with model-localized labels retained in history. Channels without dynamic UI return guidance to continue in the app rather than waiting on an unusable card.
|
|
28
28
|
|
|
29
29
|
The client records key and mouse presses before dispatch. On release it opens a fresh bounded cleanup connection, attaches to the same live targets, releases uncertain held input and removes overlays. A failed cleanup preserves state and the automation slot for retry. Turn-triggered cancellation retries cleanup automatically until it succeeds or ownership changes. Dispatched actions are never automatically retried. Closed targets need no input cleanup. Browser-process loss disposes the client, and later requests discover the replacement process.
|
|
30
30
|
|
|
@@ -35,3 +35,22 @@ Validation: focused client tests exercise shared snapshot/click behavior, namesp
|
|
|
35
35
|
The desktop header icon pulses in the assistant's avatar accent color while a browser automation lease is active, including between browser commands. It shares the progress indicator's accent and neutral fallback. Reduced-motion clients show a solid accent. Setup status exposes the optional `automationActive` field; `desktop_activity_changed` events refresh it on acquisition and cancellation or release. Installation progress retains its `assistant:self:desktop` sync invalidations. The indicator reads status without starting installation, and reconnects refetch the current lease state.
|
|
36
36
|
|
|
37
37
|
Desktop streaming checks the current in-memory gateway feature flags during connection startup. Once connected, frame and drain callbacks do not check feature flags or load workspace configuration. Turning the flag off prevents new connections; an existing stream continues until it closes or the desktop stops.
|
|
38
|
+
|
|
39
|
+
## Computer use on the same desktop
|
|
40
|
+
|
|
41
|
+
The existing `computer-use` skill defaults to this desktop in platform-hosted
|
|
42
|
+
web conversations. Native desktop clients, including the Mac app connected to a
|
|
43
|
+
platform-hosted assistant, keep the connected-computer default and require
|
|
44
|
+
`target: "assistant-desktop"` to select the virtual desktop explicitly.
|
|
45
|
+
Non-platform assistants keep their connected-computer behavior.
|
|
46
|
+
|
|
47
|
+
Explicit `target: "connected-computer"` or `target_client_id` selects a connected
|
|
48
|
+
computer on every client. The selected target never falls back to another
|
|
49
|
+
computer when unavailable. Virtual desktop observations return full-screen
|
|
50
|
+
screenshots and mouse actions use screen coordinates.
|
|
51
|
+
|
|
52
|
+
Observe first and pass the returned `observation_id` with each native action.
|
|
53
|
+
Browser commands, user handoff, errors, and interruption require a fresh
|
|
54
|
+
observation before returning to native input. Browser commands and computer use
|
|
55
|
+
share session ownership, cancellation, and the user-help reservation. Use
|
|
56
|
+
browser commands for page-level operations and computer use for desktop UI.
|
|
@@ -43,6 +43,26 @@ export type GatewayLogsTailRouteParams = z.infer<
|
|
|
43
43
|
typeof GatewayLogsTailRouteParamsSchema
|
|
44
44
|
>;
|
|
45
45
|
|
|
46
|
+
/**
|
|
47
|
+
* `gateway_debug_export`: a consistent copy of the gateway database plus its
|
|
48
|
+
* recent log files, as one gzipped tar, for a debug bundle. No params.
|
|
49
|
+
*/
|
|
50
|
+
export const GatewayDebugExportIpcParamsSchema = z
|
|
51
|
+
.object({})
|
|
52
|
+
.strict()
|
|
53
|
+
.default({});
|
|
54
|
+
|
|
55
|
+
export const GatewayDebugExportIpcResponseSchema = z.object({
|
|
56
|
+
ok: z.literal(true),
|
|
57
|
+
/** `tar.gz` bytes, base64. Contains gateway.sqlite and gateway-logs/. */
|
|
58
|
+
archive_base64: z.string(),
|
|
59
|
+
size_bytes: z.number().int().nonnegative(),
|
|
60
|
+
});
|
|
61
|
+
|
|
62
|
+
export type GatewayDebugExportIpcResponse = z.infer<
|
|
63
|
+
typeof GatewayDebugExportIpcResponseSchema
|
|
64
|
+
>;
|
|
65
|
+
|
|
46
66
|
export const GatewayLogsTailIpcResponseSchema = z.object({
|
|
47
67
|
lines: z.array(z.record(z.string(), z.unknown())),
|
|
48
68
|
truncated: z.boolean(),
|
package/openapi.yaml
CHANGED
|
@@ -23549,6 +23549,14 @@ paths:
|
|
|
23549
23549
|
description:
|
|
23550
23550
|
description: Human-readable export description.
|
|
23551
23551
|
type: string
|
|
23552
|
+
profile:
|
|
23553
|
+
description:
|
|
23554
|
+
"Export profile. 'migration' (default) builds a teleport bundle. 'debug' builds a bundle for Vellum staff
|
|
23555
|
+
to open on a debug clone: no credentials, plus the gateway's database and logs under gateway/."
|
|
23556
|
+
type: string
|
|
23557
|
+
enum:
|
|
23558
|
+
- migration
|
|
23559
|
+
- debug
|
|
23552
23560
|
required:
|
|
23553
23561
|
- upload_url
|
|
23554
23562
|
responses:
|
package/package.json
CHANGED
|
@@ -140,9 +140,17 @@ registerSkillTools("app-builder", [mockBundledSkillTool]);
|
|
|
140
140
|
|
|
141
141
|
// Register CU tools so check() can look them up in the tool registry
|
|
142
142
|
// instead of falling through to Medium (unknown tool).
|
|
143
|
-
import {
|
|
144
|
-
for (const
|
|
145
|
-
registerTool(
|
|
143
|
+
import { COMPUTER_USE_TOOL_NAMES } from "./test-support/computer-use-skill-harness.js";
|
|
144
|
+
for (const name of COMPUTER_USE_TOOL_NAMES) {
|
|
145
|
+
registerTool({
|
|
146
|
+
name,
|
|
147
|
+
description: name,
|
|
148
|
+
category: "computer-use",
|
|
149
|
+
defaultRiskLevel: RiskLevel.Low,
|
|
150
|
+
executionTarget: "host",
|
|
151
|
+
input_schema: { type: "object", properties: {} },
|
|
152
|
+
execute: async () => ({ content: "ok", isError: false }),
|
|
153
|
+
});
|
|
146
154
|
}
|
|
147
155
|
|
|
148
156
|
function writeSkill(
|
|
@@ -3,7 +3,6 @@ import { resolve } from "node:path";
|
|
|
3
3
|
import { afterAll, describe, expect, test } from "bun:test";
|
|
4
4
|
|
|
5
5
|
import { RiskLevel } from "../permissions/types.js";
|
|
6
|
-
import { allComputerUseTools } from "../tools/computer-use/definitions.js";
|
|
7
6
|
import {
|
|
8
7
|
__resetRegistryForTesting,
|
|
9
8
|
getTool,
|
|
@@ -74,30 +73,6 @@ describe("computer-use skill manifest regression", () => {
|
|
|
74
73
|
}
|
|
75
74
|
});
|
|
76
75
|
|
|
77
|
-
test("manifest descriptions match core definitions", async () => {
|
|
78
|
-
await initializeTools();
|
|
79
|
-
|
|
80
|
-
for (const cuTool of allComputerUseTools) {
|
|
81
|
-
const manifestTool = manifest.tools.find(
|
|
82
|
-
(t: { name: string }) => t.name === cuTool.name,
|
|
83
|
-
);
|
|
84
|
-
expect(manifestTool).toBeDefined();
|
|
85
|
-
expect(manifestTool.description).toBe(cuTool.description);
|
|
86
|
-
}
|
|
87
|
-
});
|
|
88
|
-
|
|
89
|
-
test("manifest input_schema matches core definitions", async () => {
|
|
90
|
-
await initializeTools();
|
|
91
|
-
|
|
92
|
-
for (const cuTool of allComputerUseTools) {
|
|
93
|
-
const manifestTool = manifest.tools.find(
|
|
94
|
-
(t: { name: string }) => t.name === cuTool.name,
|
|
95
|
-
);
|
|
96
|
-
expect(manifestTool).toBeDefined();
|
|
97
|
-
expect(manifestTool.input_schema).toEqual(cuTool.input_schema);
|
|
98
|
-
}
|
|
99
|
-
});
|
|
100
|
-
|
|
101
76
|
test("CU action tools are not registered as core tools after initializeTools()", async () => {
|
|
102
77
|
await initializeTools();
|
|
103
78
|
|
|
@@ -1,246 +1,14 @@
|
|
|
1
1
|
import { describe, expect, test } from "bun:test";
|
|
2
2
|
|
|
3
|
-
import {
|
|
4
|
-
allComputerUseTools,
|
|
5
|
-
computerUseClickTool,
|
|
6
|
-
computerUseDoneTool,
|
|
7
|
-
computerUseDragTool,
|
|
8
|
-
computerUseKeyTool,
|
|
9
|
-
computerUseObserveTool,
|
|
10
|
-
computerUseOpenAppTool,
|
|
11
|
-
computerUseRespondTool,
|
|
12
|
-
computerUseRunAppleScriptTool,
|
|
13
|
-
computerUseScrollTool,
|
|
14
|
-
computerUseTypeTextTool,
|
|
15
|
-
computerUseWaitTool,
|
|
16
|
-
} from "../tools/computer-use/definitions.js";
|
|
17
3
|
import { forwardComputerUseProxyTool } from "../tools/computer-use/skill-proxy-bridge.js";
|
|
18
4
|
import type { ToolContext } from "../tools/types.js";
|
|
19
5
|
|
|
20
|
-
interface JsonSchema {
|
|
21
|
-
type?: string;
|
|
22
|
-
required?: string[];
|
|
23
|
-
properties?: Record<string, unknown>;
|
|
24
|
-
}
|
|
25
|
-
|
|
26
|
-
/** Cast a tool definition's input_schema to a usable JSON Schema shape. */
|
|
27
|
-
function schema(tool: { input_schema?: object }): JsonSchema {
|
|
28
|
-
return tool.input_schema as JsonSchema;
|
|
29
|
-
}
|
|
30
|
-
|
|
31
6
|
const ctx: ToolContext = {
|
|
32
7
|
workingDir: "/tmp",
|
|
33
|
-
conversationId: "
|
|
8
|
+
conversationId: "conv-123",
|
|
34
9
|
trustClass: "guardian",
|
|
35
10
|
};
|
|
36
11
|
|
|
37
|
-
// ── Tool definitions ────────────────────────────────────────────────
|
|
38
|
-
|
|
39
|
-
describe("computer-use tool definitions", () => {
|
|
40
|
-
test("allComputerUseTools contains 12 tools", () => {
|
|
41
|
-
expect(allComputerUseTools.length).toBe(12);
|
|
42
|
-
});
|
|
43
|
-
|
|
44
|
-
test("all tools belong to computer-use category", () => {
|
|
45
|
-
for (const tool of allComputerUseTools) {
|
|
46
|
-
expect(tool.category).toBe("computer-use");
|
|
47
|
-
}
|
|
48
|
-
});
|
|
49
|
-
|
|
50
|
-
test("all tools have unique names", () => {
|
|
51
|
-
const names = allComputerUseTools.map((t) => t.name);
|
|
52
|
-
expect(new Set(names).size).toBe(names.length);
|
|
53
|
-
});
|
|
54
|
-
|
|
55
|
-
test("all tools have descriptions", () => {
|
|
56
|
-
for (const tool of allComputerUseTools) {
|
|
57
|
-
expect(tool.description!.length).toBeGreaterThan(0);
|
|
58
|
-
}
|
|
59
|
-
});
|
|
60
|
-
});
|
|
61
|
-
|
|
62
|
-
// ── observe ─────────────────────────────────────────────────────────
|
|
63
|
-
|
|
64
|
-
describe("computer_use_observe", () => {
|
|
65
|
-
test("supports target_client_id", () => {
|
|
66
|
-
const props = schema(computerUseObserveTool).properties as Record<
|
|
67
|
-
string,
|
|
68
|
-
{ type: string }
|
|
69
|
-
>;
|
|
70
|
-
expect(props.target_client_id.type).toBe("string");
|
|
71
|
-
});
|
|
72
|
-
});
|
|
73
|
-
|
|
74
|
-
// ── Unified click tool ──────────────────────────────────────────────
|
|
75
|
-
|
|
76
|
-
describe("computer_use_click (unified)", () => {
|
|
77
|
-
test("has correct name", () => {
|
|
78
|
-
expect(computerUseClickTool.name).toBe("computer_use_click");
|
|
79
|
-
});
|
|
80
|
-
|
|
81
|
-
test("schema requires reasoning", () => {
|
|
82
|
-
expect(schema(computerUseClickTool).required).toContain("reasoning");
|
|
83
|
-
});
|
|
84
|
-
|
|
85
|
-
test("schema supports click_type enum", () => {
|
|
86
|
-
const props = schema(computerUseClickTool).properties as Record<
|
|
87
|
-
string,
|
|
88
|
-
{ type: string; enum?: string[] }
|
|
89
|
-
>;
|
|
90
|
-
expect(props.click_type.type).toBe("string");
|
|
91
|
-
expect(props.click_type.enum).toEqual(["single", "double", "right"]);
|
|
92
|
-
});
|
|
93
|
-
|
|
94
|
-
test("schema supports element_id and coordinates", () => {
|
|
95
|
-
const props = schema(computerUseClickTool).properties as Record<
|
|
96
|
-
string,
|
|
97
|
-
{ type: string }
|
|
98
|
-
>;
|
|
99
|
-
expect(props.element_id.type).toBe("integer");
|
|
100
|
-
expect(props.x.type).toBe("integer");
|
|
101
|
-
expect(props.y.type).toBe("integer");
|
|
102
|
-
});
|
|
103
|
-
|
|
104
|
-
test("execute returns isError when no proxy resolver is configured", async () => {
|
|
105
|
-
const result = await computerUseClickTool.execute({}, ctx);
|
|
106
|
-
expect(result.isError).toBe(true);
|
|
107
|
-
expect(result.content).toContain(
|
|
108
|
-
"The Vellum desktop app is needed to view or control your screen",
|
|
109
|
-
);
|
|
110
|
-
expect(result.content).toContain("https://www.vellum.ai/downloads");
|
|
111
|
-
});
|
|
112
|
-
});
|
|
113
|
-
|
|
114
|
-
// ── type_text ───────────────────────────────────────────────────────
|
|
115
|
-
|
|
116
|
-
describe("computer_use_type_text", () => {
|
|
117
|
-
test("requires text and reasoning", () => {
|
|
118
|
-
expect(schema(computerUseTypeTextTool).required).toContain("text");
|
|
119
|
-
expect(schema(computerUseTypeTextTool).required).toContain("reasoning");
|
|
120
|
-
});
|
|
121
|
-
|
|
122
|
-
test("execute returns isError when no proxy resolver is configured", async () => {
|
|
123
|
-
const result = await computerUseTypeTextTool.execute({}, ctx);
|
|
124
|
-
expect(result.isError).toBe(true);
|
|
125
|
-
expect(result.content).toContain(
|
|
126
|
-
"The Vellum desktop app is needed to view or control your screen",
|
|
127
|
-
);
|
|
128
|
-
expect(result.content).toContain("https://www.vellum.ai/downloads");
|
|
129
|
-
});
|
|
130
|
-
});
|
|
131
|
-
|
|
132
|
-
// ── key ─────────────────────────────────────────────────────────────
|
|
133
|
-
|
|
134
|
-
describe("computer_use_key", () => {
|
|
135
|
-
test("requires key and reasoning", () => {
|
|
136
|
-
expect(schema(computerUseKeyTool).required).toContain("key");
|
|
137
|
-
expect(schema(computerUseKeyTool).required).toContain("reasoning");
|
|
138
|
-
});
|
|
139
|
-
|
|
140
|
-
test("execute returns isError when no proxy resolver is configured", async () => {
|
|
141
|
-
const result = await computerUseKeyTool.execute({}, ctx);
|
|
142
|
-
expect(result.isError).toBe(true);
|
|
143
|
-
expect(result.content).toContain(
|
|
144
|
-
"The Vellum desktop app is needed to view or control your screen",
|
|
145
|
-
);
|
|
146
|
-
expect(result.content).toContain("https://www.vellum.ai/downloads");
|
|
147
|
-
});
|
|
148
|
-
});
|
|
149
|
-
|
|
150
|
-
// ── scroll ──────────────────────────────────────────────────────────
|
|
151
|
-
|
|
152
|
-
describe("computer_use_scroll", () => {
|
|
153
|
-
test("requires direction, amount, and reasoning", () => {
|
|
154
|
-
expect(schema(computerUseScrollTool).required).toContain("direction");
|
|
155
|
-
expect(schema(computerUseScrollTool).required).toContain("amount");
|
|
156
|
-
expect(schema(computerUseScrollTool).required).toContain("reasoning");
|
|
157
|
-
});
|
|
158
|
-
|
|
159
|
-
test("direction enum includes up, down, left, right", () => {
|
|
160
|
-
const props = schema(computerUseScrollTool).properties as Record<
|
|
161
|
-
string,
|
|
162
|
-
{ enum?: string[] }
|
|
163
|
-
>;
|
|
164
|
-
expect(props.direction.enum).toEqual(["up", "down", "left", "right"]);
|
|
165
|
-
});
|
|
166
|
-
});
|
|
167
|
-
|
|
168
|
-
// ── drag ────────────────────────────────────────────────────────────
|
|
169
|
-
|
|
170
|
-
describe("computer_use_drag", () => {
|
|
171
|
-
test("supports source and destination coordinates", () => {
|
|
172
|
-
const props = schema(computerUseDragTool).properties as Record<
|
|
173
|
-
string,
|
|
174
|
-
{ type: string }
|
|
175
|
-
>;
|
|
176
|
-
expect(props.element_id.type).toBe("integer");
|
|
177
|
-
expect(props.to_element_id.type).toBe("integer");
|
|
178
|
-
expect(props.x.type).toBe("integer");
|
|
179
|
-
expect(props.y.type).toBe("integer");
|
|
180
|
-
expect(props.to_x.type).toBe("integer");
|
|
181
|
-
expect(props.to_y.type).toBe("integer");
|
|
182
|
-
});
|
|
183
|
-
|
|
184
|
-
test("requires reasoning only", () => {
|
|
185
|
-
expect(schema(computerUseDragTool).required).toEqual(["reasoning"]);
|
|
186
|
-
});
|
|
187
|
-
});
|
|
188
|
-
|
|
189
|
-
// ── wait ────────────────────────────────────────────────────────────
|
|
190
|
-
|
|
191
|
-
describe("computer_use_wait", () => {
|
|
192
|
-
test("requires duration_ms and reasoning", () => {
|
|
193
|
-
expect(schema(computerUseWaitTool).required).toContain("duration_ms");
|
|
194
|
-
expect(schema(computerUseWaitTool).required).toContain("reasoning");
|
|
195
|
-
});
|
|
196
|
-
});
|
|
197
|
-
|
|
198
|
-
// ── open_app ────────────────────────────────────────────────────────
|
|
199
|
-
|
|
200
|
-
describe("computer_use_open_app", () => {
|
|
201
|
-
test("requires app_name and reasoning", () => {
|
|
202
|
-
expect(schema(computerUseOpenAppTool).required).toContain("app_name");
|
|
203
|
-
expect(schema(computerUseOpenAppTool).required).toContain("reasoning");
|
|
204
|
-
});
|
|
205
|
-
});
|
|
206
|
-
|
|
207
|
-
// ── run_applescript ─────────────────────────────────────────────────
|
|
208
|
-
|
|
209
|
-
describe("computer_use_run_applescript", () => {
|
|
210
|
-
test("requires script and reasoning", () => {
|
|
211
|
-
expect(schema(computerUseRunAppleScriptTool).required).toContain("script");
|
|
212
|
-
expect(schema(computerUseRunAppleScriptTool).required).toContain(
|
|
213
|
-
"reasoning",
|
|
214
|
-
);
|
|
215
|
-
});
|
|
216
|
-
|
|
217
|
-
test("description warns against do shell script", () => {
|
|
218
|
-
expect(computerUseRunAppleScriptTool.description).toContain(
|
|
219
|
-
"do shell script",
|
|
220
|
-
);
|
|
221
|
-
expect(computerUseRunAppleScriptTool.description).toContain("blocked");
|
|
222
|
-
});
|
|
223
|
-
});
|
|
224
|
-
|
|
225
|
-
// ── done ────────────────────────────────────────────────────────────
|
|
226
|
-
|
|
227
|
-
describe("computer_use_done", () => {
|
|
228
|
-
test("requires summary", () => {
|
|
229
|
-
expect(schema(computerUseDoneTool).required).toContain("summary");
|
|
230
|
-
});
|
|
231
|
-
});
|
|
232
|
-
|
|
233
|
-
// ── respond ─────────────────────────────────────────────────────────
|
|
234
|
-
|
|
235
|
-
describe("computer_use_respond", () => {
|
|
236
|
-
test("requires answer and reasoning", () => {
|
|
237
|
-
expect(schema(computerUseRespondTool).required).toContain("answer");
|
|
238
|
-
expect(schema(computerUseRespondTool).required).toContain("reasoning");
|
|
239
|
-
});
|
|
240
|
-
});
|
|
241
|
-
|
|
242
|
-
// ── skill-proxy-bridge ──────────────────────────────────────────────
|
|
243
|
-
|
|
244
12
|
describe("forwardComputerUseProxyTool", () => {
|
|
245
13
|
test("returns error when no proxy resolver available", async () => {
|
|
246
14
|
const result = await forwardComputerUseProxyTool(
|
|
@@ -10,6 +10,7 @@
|
|
|
10
10
|
import { beforeEach, describe, expect, mock, spyOn, test } from "bun:test";
|
|
11
11
|
|
|
12
12
|
import type { Conversation } from "../daemon/conversation.js";
|
|
13
|
+
import { shouldUseVirtualDesktop } from "../desktop/virtual-desktop-feature.js";
|
|
13
14
|
import type { PermissionPrompter } from "../permissions/prompter.js";
|
|
14
15
|
import type { SecretPrompter } from "../permissions/secret-prompter.js";
|
|
15
16
|
import type { ToolExecutor } from "../tools/executor.js";
|
|
@@ -307,6 +308,28 @@ describe("createToolExecutor attribution threading", () => {
|
|
|
307
308
|
});
|
|
308
309
|
});
|
|
309
310
|
|
|
311
|
+
test("desktop routing honors the pinned interface when the live client changes", async () => {
|
|
312
|
+
for (const transportInterface of ["web", "macos"] as const) {
|
|
313
|
+
const { executor, calls } = makeCapturingExecutor();
|
|
314
|
+
await makeToolFn(
|
|
315
|
+
executor,
|
|
316
|
+
makeCtx({
|
|
317
|
+
transportInterface: transportInterface === "web" ? "macos" : "web",
|
|
318
|
+
toolContextPin: { hasNoClient: false, transportInterface },
|
|
319
|
+
getTurnActorPrincipalId: () => "user-123",
|
|
320
|
+
currentTurnTrustContext: {
|
|
321
|
+
sourceChannel: "vellum",
|
|
322
|
+
trustClass: "guardian",
|
|
323
|
+
},
|
|
324
|
+
}),
|
|
325
|
+
)("computer_use_observe", {});
|
|
326
|
+
|
|
327
|
+
expect(shouldUseVirtualDesktop(calls[0].context, true)).toBe(
|
|
328
|
+
transportInterface === "web",
|
|
329
|
+
);
|
|
330
|
+
}
|
|
331
|
+
});
|
|
332
|
+
|
|
310
333
|
describe("createToolExecutor isInteractive threading", () => {
|
|
311
334
|
test("uses the resolved turn-level interactivity over live client state", async () => {
|
|
312
335
|
// A scheduled/background turn (currentTurnIsNonInteractive=true) must read
|
|
@@ -21,6 +21,7 @@ import {
|
|
|
21
21
|
abortedToolResultText,
|
|
22
22
|
CANCELLED_TOOL_RESULT_TEXT,
|
|
23
23
|
createAbortReason,
|
|
24
|
+
INTERRUPTED_TURN_NOTE_TEXT,
|
|
24
25
|
PREEMPTED_TOOL_RESULT_TEXT,
|
|
25
26
|
} from "../util/abort-reasons.js";
|
|
26
27
|
|
|
@@ -378,8 +379,8 @@ describe("interruptRunningTurn", () => {
|
|
|
378
379
|
|
|
379
380
|
test("arms the interrupted-turn note when the abort caught no tool call", async () => {
|
|
380
381
|
// The abort landed during the provider call, so there is no `tool_use` to
|
|
381
|
-
// answer
|
|
382
|
-
//
|
|
382
|
+
// answer. The note on the interrupting user message is the only thing that
|
|
383
|
+
// tells the model what happened.
|
|
383
384
|
const turn = registerBusyTurn({
|
|
384
385
|
messages: [
|
|
385
386
|
{ role: "user", content: [{ type: "text", text: "research this" }] },
|
|
@@ -393,9 +394,9 @@ describe("interruptRunningTurn", () => {
|
|
|
393
394
|
expect(persisted).toEqual([]);
|
|
394
395
|
});
|
|
395
396
|
|
|
396
|
-
test("
|
|
397
|
-
// The
|
|
398
|
-
//
|
|
397
|
+
test("arms the note alongside the fact-only result of a stopped tool call", async () => {
|
|
398
|
+
// The two texts do different jobs: the `tool_result` states what happened
|
|
399
|
+
// to that one call, the note says what to do about the interruption.
|
|
399
400
|
const turn = registerBusyTurn({
|
|
400
401
|
messages: [assistantWithToolUse("tool-1")],
|
|
401
402
|
});
|
|
@@ -404,13 +405,36 @@ describe("interruptRunningTurn", () => {
|
|
|
404
405
|
|
|
405
406
|
expect(persisted).toHaveLength(1);
|
|
406
407
|
expect(persisted[0].content).toContain(PREEMPTED_TOOL_RESULT_TEXT);
|
|
407
|
-
expect(turn.conversation.pendingInterruptNote).toBe(
|
|
408
|
+
expect(turn.conversation.pendingInterruptNote).toBe(true);
|
|
409
|
+
});
|
|
410
|
+
|
|
411
|
+
test("arms exactly one note however many parallel calls were stopped", async () => {
|
|
412
|
+
// Every stopped call gets its own fact-only result; the behavior
|
|
413
|
+
// instruction is read once, off the interrupting message.
|
|
414
|
+
const turn = registerBusyTurn({
|
|
415
|
+
messages: [assistantWithToolUse("tool-1", "tool-2", "tool-3")],
|
|
416
|
+
});
|
|
417
|
+
|
|
418
|
+
await interruptRunningTurn(turn.conversation, { origin: "test" });
|
|
419
|
+
|
|
420
|
+
const results = JSON.parse(persisted[0].content) as Array<{
|
|
421
|
+
tool_use_id: string;
|
|
422
|
+
content: string;
|
|
423
|
+
}>;
|
|
424
|
+
expect(results.map((block) => block.tool_use_id)).toEqual([
|
|
425
|
+
"tool-1",
|
|
426
|
+
"tool-2",
|
|
427
|
+
"tool-3",
|
|
428
|
+
]);
|
|
429
|
+
for (const block of results) {
|
|
430
|
+
expect(block.content).toBe(PREEMPTED_TOOL_RESULT_TEXT);
|
|
431
|
+
}
|
|
432
|
+
expect(turn.conversation.pendingInterruptNote).toBe(true);
|
|
408
433
|
});
|
|
409
434
|
|
|
410
|
-
test("
|
|
435
|
+
test("arms the note when the loop wrote its own preempted result", async () => {
|
|
411
436
|
// The agent loop's abort handler unwound cleanly and answered the batch
|
|
412
|
-
// itself, so the repair finds nothing to do
|
|
413
|
-
// there.
|
|
437
|
+
// itself, so the repair finds nothing to do. The note is owed all the same.
|
|
414
438
|
const turn = registerBusyTurn({
|
|
415
439
|
messages: [
|
|
416
440
|
assistantWithToolUse("tool-1"),
|
|
@@ -421,12 +445,10 @@ describe("interruptRunningTurn", () => {
|
|
|
421
445
|
await interruptRunningTurn(turn.conversation, { origin: "test" });
|
|
422
446
|
|
|
423
447
|
expect(persisted).toEqual([]);
|
|
424
|
-
expect(turn.conversation.pendingInterruptNote).toBe(
|
|
448
|
+
expect(turn.conversation.pendingInterruptNote).toBe(true);
|
|
425
449
|
});
|
|
426
450
|
|
|
427
451
|
test("arms the note when the cut-off tool call was an ordinary cancel", async () => {
|
|
428
|
-
// A tail answered with the plain cancel wording is a stop the user made
|
|
429
|
-
// earlier, not this interrupt's notice, so the note is still owed.
|
|
430
452
|
const turn = registerBusyTurn({
|
|
431
453
|
messages: [
|
|
432
454
|
assistantWithToolUse("tool-1"),
|
|
@@ -633,6 +655,9 @@ describe("interruptRunningTurn", () => {
|
|
|
633
655
|
// Armed instead, so the drain that runs the queued message repairs it.
|
|
634
656
|
expect(turn.conversation.pendingInterruptRepair).toBe(true);
|
|
635
657
|
expect(turn.activityEvents).toEqual([]);
|
|
658
|
+
// The message joins the queue rather than replacing the turn, so it is not
|
|
659
|
+
// the message that interrupted anything and carries no note.
|
|
660
|
+
expect(turn.conversation.pendingInterruptNote).toBe(false);
|
|
636
661
|
});
|
|
637
662
|
|
|
638
663
|
test("a second interrupt stops the turn the first one started", async () => {
|
|
@@ -932,7 +957,7 @@ describe("repairInterruptedToolUseBlocks", () => {
|
|
|
932
957
|
} as unknown as Conversation;
|
|
933
958
|
}
|
|
934
959
|
|
|
935
|
-
test("writes one result per abandoned call
|
|
960
|
+
test("writes one fact-only result per abandoned call", async () => {
|
|
936
961
|
const messages: Message[] = [assistantWithToolUse("tool-1", "tool-2")];
|
|
937
962
|
|
|
938
963
|
await repairInterruptedToolUseBlocks(fakeConversation(messages), {
|
|
@@ -961,31 +986,6 @@ describe("repairInterruptedToolUseBlocks", () => {
|
|
|
961
986
|
);
|
|
962
987
|
});
|
|
963
988
|
|
|
964
|
-
test("reports the preempted result it wrote", async () => {
|
|
965
|
-
const messages: Message[] = [assistantWithToolUse("tool-1")];
|
|
966
|
-
|
|
967
|
-
const result = await repairInterruptedToolUseBlocks(
|
|
968
|
-
fakeConversation(messages),
|
|
969
|
-
{ force: true },
|
|
970
|
-
);
|
|
971
|
-
|
|
972
|
-
expect(result.preemptedToolResultOnTail).toBe(true);
|
|
973
|
-
});
|
|
974
|
-
|
|
975
|
-
test("reports no preempted result on a history that ends on plain text", async () => {
|
|
976
|
-
const messages: Message[] = [
|
|
977
|
-
{ role: "user", content: [{ type: "text", text: "hi" }] },
|
|
978
|
-
{ role: "assistant", content: [{ type: "text", text: "hello" }] },
|
|
979
|
-
];
|
|
980
|
-
|
|
981
|
-
const result = await repairInterruptedToolUseBlocks(
|
|
982
|
-
fakeConversation(messages),
|
|
983
|
-
{ force: true },
|
|
984
|
-
);
|
|
985
|
-
|
|
986
|
-
expect(result.preemptedToolResultOnTail).toBe(false);
|
|
987
|
-
});
|
|
988
|
-
|
|
989
989
|
test("adds nothing when the loop already wrote its own results", async () => {
|
|
990
990
|
// The agent loop's abort handler synthesizes results of its own when it
|
|
991
991
|
// unwinds cleanly, so exactly one result per call exists either way.
|
|
@@ -994,14 +994,12 @@ describe("repairInterruptedToolUseBlocks", () => {
|
|
|
994
994
|
toolResult("tool-1", PREEMPTED_TOOL_RESULT_TEXT),
|
|
995
995
|
];
|
|
996
996
|
|
|
997
|
-
|
|
998
|
-
|
|
999
|
-
|
|
1000
|
-
);
|
|
997
|
+
await repairInterruptedToolUseBlocks(fakeConversation(messages), {
|
|
998
|
+
force: true,
|
|
999
|
+
});
|
|
1001
1000
|
|
|
1002
1001
|
expect(messages).toHaveLength(2);
|
|
1003
1002
|
expect(persisted).toEqual([]);
|
|
1004
|
-
expect(result.preemptedToolResultOnTail).toBe(true);
|
|
1005
1003
|
});
|
|
1006
1004
|
|
|
1007
1005
|
test("does nothing on a history that ends on an ordinary assistant reply", async () => {
|
|
@@ -1083,3 +1081,19 @@ describe("abortedToolResultText", () => {
|
|
|
1083
1081
|
);
|
|
1084
1082
|
});
|
|
1085
1083
|
});
|
|
1084
|
+
|
|
1085
|
+
describe("the interrupt texts", () => {
|
|
1086
|
+
test("the tool result states the fact and nothing more", () => {
|
|
1087
|
+
// Repeated once per stopped call in a parallel batch, so an instruction
|
|
1088
|
+
// here is read as many times as the batch was wide.
|
|
1089
|
+
expect(PREEMPTED_TOOL_RESULT_TEXT).toBe(
|
|
1090
|
+
"Stopped early because the user sent a new message. This call may have completed anyway; check before repeating it.",
|
|
1091
|
+
);
|
|
1092
|
+
});
|
|
1093
|
+
|
|
1094
|
+
test("the note carries the behavior and leads with the reply", () => {
|
|
1095
|
+
expect(INTERRUPTED_TURN_NOTE_TEXT).toBe(
|
|
1096
|
+
"<interrupted_turn>The user sent this while you were still working, so that work stopped. Reply right away, in a line or two, before any more thinking or tool calls, so the user knows you heard them. Then pick the earlier work back up only if it is still wanted. Never apologize for or mention the interruption.</interrupted_turn>",
|
|
1097
|
+
);
|
|
1098
|
+
});
|
|
1099
|
+
});
|
|
@@ -1,10 +1,9 @@
|
|
|
1
1
|
/**
|
|
2
|
-
*
|
|
3
|
-
*
|
|
4
|
-
*
|
|
5
|
-
*
|
|
6
|
-
*
|
|
7
|
-
* metadata is what rebuilds the identical block on every later load.
|
|
2
|
+
* The message that interrupted a turn carries the note saying what to do about
|
|
3
|
+
* the interruption, on its LLM-facing content only: the persisted row stays
|
|
4
|
+
* exactly what the user typed, so no client renders it, and
|
|
5
|
+
* `interruptedPriorTurn` in the row's metadata is what rebuilds the identical
|
|
6
|
+
* block on every later load.
|
|
8
7
|
*/
|
|
9
8
|
import { beforeEach, describe, expect, test } from "bun:test";
|
|
10
9
|
|