@allternit/computer-driver 0.0.0-stage → 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/v2.d.ts ADDED
@@ -0,0 +1,279 @@
1
+ import { type AllternitComputers, type Approval, type ToolsetResult } from "./client.ts";
2
+ import type { ComputerV2ActInput, ComputerV2ReadUiInput, ComputerV2RequestHumanInput, ComputerV2RunBatchInput, ComputerV2RunParallelInput, ComputerV2RunSkillInput, ComputerV2RunSubtaskInput, ComputerV2SkillsInput, ComputerV2StructuredMemberName, ComputerV2UseCredentialInput, ComputerV2VerifyInput } from "./v2-generated.ts";
3
+ export type { ComputerV2ActInput, ComputerV2MemberInputs, ComputerV2ReadUiInput, ComputerV2RequestHumanInput, ComputerV2RunBatchInput, ComputerV2RunParallelInput, ComputerV2RunSkillInput, ComputerV2RunSubtaskInput, ComputerV2SkillsInput, ComputerV2StructuredMemberName, ComputerV2UseCredentialInput, ComputerV2VerifyInput, } from "./v2-generated.ts";
4
+ export { COMPUTER_V2_MEMBER_NAMES, COMPUTER_V2_STRUCTURED_MEMBER_NAMES } from "./v2-generated.ts";
5
+ /** Marker the Allternit wire tooling uses to recognise the v2 function tool. */
6
+ export declare const V2_TOOL_MARKER = "[allternit.computer.v2]";
7
+ /** The function-tool name the structured members are exposed under. */
8
+ export declare const COMPUTER_V2_TOOL_NAME: "computer_v2";
9
+ /** A v2 member answered is_error:true, or returned no JSON where JSON was expected. */
10
+ export declare class ComputerV2Error extends Error {
11
+ readonly member: string;
12
+ readonly result: ToolsetResult;
13
+ constructor(member: string, result: ToolsetResult, message?: string);
14
+ }
15
+ /** One element of a read_ui element map. Tree ids start with "e"; vision marks with "v". */
16
+ export interface UiElement {
17
+ id: string;
18
+ role: string;
19
+ name?: string;
20
+ value?: unknown;
21
+ /** [x, y, w, h] in screen points. */
22
+ bbox?: [number, number, number, number];
23
+ enabled?: boolean;
24
+ focused?: boolean;
25
+ actions?: string[];
26
+ parent?: string;
27
+ /** Tree path (role:ordinal chain); only with read_ui paths:true. */
28
+ path?: string;
29
+ /** 8x8 pixel hash; only with read_ui crops:true. */
30
+ crop?: string;
31
+ source?: "vision";
32
+ /** Set-of-marks number for vision elements. */
33
+ mark?: number;
34
+ }
35
+ /** The grounder's answer when read_ui got a `target`. */
36
+ export interface GroundedResult {
37
+ status: string;
38
+ id?: string;
39
+ point?: [number, number];
40
+ confidence?: number;
41
+ error?: string;
42
+ [key: string]: unknown;
43
+ }
44
+ /** read_ui result: one window's element map (or the diff since `since`). */
45
+ export interface ReadUiResult {
46
+ window?: {
47
+ pid?: number;
48
+ window_id?: number;
49
+ app?: string;
50
+ title?: string;
51
+ };
52
+ version: number;
53
+ engine?: string | null;
54
+ cached?: boolean;
55
+ live?: Record<string, unknown>;
56
+ /** Which source answered: ax, vision or hybrid. */
57
+ source?: string;
58
+ source_reason?: string;
59
+ degraded?: boolean;
60
+ degraded_reason?: string;
61
+ /** Present when `since` was answered with a diff. */
62
+ diff?: {
63
+ added?: UiElement[];
64
+ changed?: UiElement[];
65
+ removed?: Array<Record<string, unknown>>;
66
+ [key: string]: unknown;
67
+ };
68
+ /** True when the `since` version had aged out and the whole map was re-sent. */
69
+ reset?: boolean;
70
+ elements?: UiElement[];
71
+ total?: number;
72
+ truncated?: boolean;
73
+ marks?: Array<Record<string, unknown>>;
74
+ vision_version?: number;
75
+ grounded?: GroundedResult;
76
+ ms?: number;
77
+ [key: string]: unknown;
78
+ }
79
+ /** act result. `stale_version` carries the fresh map under `map`: re-plan and retry. */
80
+ export interface ActResult {
81
+ status: "done" | "changed" | "stale" | "stale_version" | string;
82
+ version: number;
83
+ engine?: string | null;
84
+ map?: ReadUiResult;
85
+ changes?: unknown;
86
+ settled?: boolean;
87
+ ms?: number;
88
+ [key: string]: unknown;
89
+ }
90
+ export interface RunBatchStepResult {
91
+ status: "ok" | "failed" | string;
92
+ code?: string;
93
+ error?: string;
94
+ ms: number;
95
+ [key: string]: unknown;
96
+ }
97
+ /** run_batch result: one row per step; a failed check stops the batch at `failed_at`. */
98
+ export interface RunBatchResult {
99
+ ok: boolean;
100
+ steps: RunBatchStepResult[];
101
+ version: number;
102
+ engine?: string | null;
103
+ failed_at?: number;
104
+ changes?: unknown;
105
+ cross_check?: unknown;
106
+ ms?: number;
107
+ [key: string]: unknown;
108
+ }
109
+ export type VerifyCheck = ComputerV2VerifyInput["checks"][number];
110
+ export interface VerifyCheckResult {
111
+ check: VerifyCheck;
112
+ ok: boolean;
113
+ detail: string;
114
+ }
115
+ /** verify result: `ok` is true only when every check held. */
116
+ export interface VerifyResult {
117
+ ok: boolean;
118
+ results: VerifyCheckResult[];
119
+ ms?: number;
120
+ [key: string]: unknown;
121
+ }
122
+ /** run_subtask / run_skill end statuses (safety verdicts included). */
123
+ export type SubtaskStatus = "done" | "escalated" | "failed" | "needs_confirmation" | "paused" | "denied" | "use_api" | string;
124
+ export interface SubtaskCache {
125
+ status: "hit" | "healed" | "recorded" | "miss" | "diverged" | string;
126
+ replayed_steps: number;
127
+ healed_steps: Array<Record<string, unknown>>;
128
+ stored: boolean;
129
+ key?: string;
130
+ skill?: string;
131
+ reason?: string;
132
+ [key: string]: unknown;
133
+ }
134
+ /**
135
+ * run_subtask / run_skill result. Anything but `done` carries `next` (what the
136
+ * server suggests) and usually `screen`; `needs_confirmation`/`paused`/`denied`
137
+ * carry `held_step`; `use_api` carries `api` (the MCP/API call to make yourself).
138
+ */
139
+ export interface SubtaskResult {
140
+ status: SubtaskStatus;
141
+ goal: string;
142
+ decisions: number;
143
+ actions: number;
144
+ elapsed_ms: number;
145
+ decision_ms: number;
146
+ action_ms: number;
147
+ oracle: {
148
+ tokens: number;
149
+ cost_usd: number;
150
+ };
151
+ inputs_typed: number;
152
+ success: unknown;
153
+ steps: Array<Record<string, unknown>>;
154
+ cache: SubtaskCache;
155
+ reason?: string;
156
+ held_step?: Record<string, unknown>;
157
+ api?: {
158
+ tool?: string;
159
+ description?: string;
160
+ [key: string]: unknown;
161
+ };
162
+ screen?: unknown;
163
+ next?: string;
164
+ [key: string]: unknown;
165
+ }
166
+ /** One run_parallel result row: a run_subtask result plus its slot and computer. */
167
+ export interface ParallelSubtaskResult extends SubtaskResult {
168
+ subtask?: number;
169
+ computer?: string;
170
+ }
171
+ /**
172
+ * run_parallel result. Plain mode: `results` in order, `status` done/partial/
173
+ * failed. Best-of-N mode (best_of + computers): `rollouts` with narratives and
174
+ * `chosen` naming the winning rollout, `status` done or escalated.
175
+ */
176
+ export interface ParallelResult {
177
+ status: "done" | "partial" | "failed" | "escalated" | string;
178
+ done?: number;
179
+ subtasks: number;
180
+ max_parallel?: number;
181
+ best_of?: number;
182
+ elapsed_ms: number;
183
+ cost_usd: number;
184
+ results?: Array<ParallelSubtaskResult | {
185
+ status: "skipped";
186
+ reason: string;
187
+ subtask?: number;
188
+ computer?: string;
189
+ }>;
190
+ chosen?: {
191
+ rollout?: string;
192
+ computer?: string;
193
+ why?: unknown;
194
+ } | null;
195
+ reason?: unknown;
196
+ judge_decisions?: number;
197
+ rollouts?: Array<{
198
+ rollout: string;
199
+ computer: string;
200
+ status: string;
201
+ narrative: string;
202
+ result: SubtaskResult;
203
+ }>;
204
+ next?: string;
205
+ [key: string]: unknown;
206
+ }
207
+ /** One saved skill (a recording taught with run_subtask save_as). */
208
+ export interface SkillInfo {
209
+ name: string;
210
+ goal: string;
211
+ app?: string | null;
212
+ inputs: string[];
213
+ steps: number;
214
+ runs: number;
215
+ healed: number;
216
+ updated_at?: string;
217
+ last_used_at?: string | null;
218
+ [key: string]: unknown;
219
+ }
220
+ /** skills result: the saved skills this person can run with run_skill. */
221
+ export interface SkillsResult {
222
+ skills: SkillInfo[];
223
+ forgot?: string;
224
+ [key: string]: unknown;
225
+ }
226
+ export interface V2CallOptions {
227
+ client: AllternitComputers;
228
+ computerId: string;
229
+ /** Called when the server holds the call (409 approval_required). true → approve + resend with the grant; false/absent → the held result resolves. */
230
+ onApproval?: (approval: Approval) => boolean | Promise<boolean>;
231
+ }
232
+ /** Run one structured v2 member through POST /v1/computers/{id}/toolset, answering an approval hold via `onApproval`. */
233
+ export declare function runComputerV2Member(o: V2CallOptions, member: ComputerV2StructuredMemberName | string, input: unknown): Promise<ToolsetResult>;
234
+ /**
235
+ * Typed methods for the ten structured v2 members of one computer. Construct
236
+ * once per computer and pass it to your loop, or call the members directly.
237
+ */
238
+ export declare class ComputerV2Driver {
239
+ #private;
240
+ constructor(o: V2CallOptions);
241
+ /** Read the target window or app's UI as a structured element tree (no screenshot). */
242
+ read_ui(input?: ComputerV2ReadUiInput): Promise<ReadUiResult>;
243
+ /** Act on an element id from read_ui. */
244
+ act(input: ComputerV2ActInput): Promise<ActResult>;
245
+ /** Run ordered steps in one window in a single call; a failed check stops the batch. */
246
+ run_batch(input: ComputerV2RunBatchInput): Promise<RunBatchResult>;
247
+ /** Check bounded conditions on the current UI without acting. */
248
+ verify(input: ComputerV2VerifyInput): Promise<VerifyResult>;
249
+ /** Pause and hand the computer to a person; resolves when they signal done or the timeout elapses. */
250
+ request_human(input?: ComputerV2RequestHumanInput): Promise<string>;
251
+ /** Type a vault credential into the focused field; the value never enters the model context. */
252
+ use_credential(input: ComputerV2UseCredentialInput): Promise<string>;
253
+ /** Hand a bounded UI subtask to the fast decision loop. */
254
+ run_subtask(input: ComputerV2RunSubtaskInput): Promise<SubtaskResult>;
255
+ /** Run independent bounded subtasks on separate computers at the same time. */
256
+ run_parallel(input: ComputerV2RunParallelInput): Promise<ParallelResult>;
257
+ /** Run a saved skill (a recording taught with run_subtask save_as) by name. */
258
+ run_skill(input: ComputerV2RunSkillInput): Promise<SubtaskResult>;
259
+ /** List the saved skills this person can run (pass forget to delete one first). */
260
+ skills(input?: ComputerV2SkillsInput): Promise<SkillsResult>;
261
+ }
262
+ /** The `computer_v2` function tool, provider-neutral. Adapt `schema` to your SDK's spelling (Anthropic `input_schema`, OpenAI `parameters`, Gemini `parameters`). */
263
+ export interface ComputerV2ToolDefinition {
264
+ name: typeof COMPUTER_V2_TOOL_NAME;
265
+ description: string;
266
+ /** JSON Schema for the tool's arguments: `action` plus every member's fields (all optional besides action). */
267
+ schema: Record<string, unknown>;
268
+ }
269
+ /** Description gizzi-equivalent for the `computer_v2` tool, built from the contract's member descriptions. */
270
+ export declare function computerV2ToolDescription(): string;
271
+ /**
272
+ * The combined JSON Schema for the `computer_v2` tool: `action` (one of the ten
273
+ * structured members) plus the union of every member's input fields, each
274
+ * optional — a call maps 1:1 onto `{ member: action, input: <the rest> }`, and
275
+ * the server validates against the member's own contract schema.
276
+ */
277
+ export declare function computerV2Parameters(): Record<string, unknown>;
278
+ /** The provider-neutral `computer_v2` function tool definition. */
279
+ export declare function computerV2Tool(): ComputerV2ToolDefinition;
package/dist/v2.js ADDED
@@ -0,0 +1,154 @@
1
+ // Contract v2 (`allternit.computer.v2`): the ten driver-backed structured
2
+ // members — read_ui, act, run_batch, verify, request_human, use_credential,
3
+ // run_subtask, run_parallel, run_skill, skills — as typed client methods, plus
4
+ // the `computer_v2` function tool every adapter exposes next to the pixel tool.
5
+ //
6
+ // Input types and member metadata come from src/v2-generated.ts (emitted by
7
+ // contracts/computer-toolset/generate.mjs from allternit-computer-v2.json), so
8
+ // they cannot drift from the server contract. Result shapes are the executor's
9
+ // JSON answers (allternit-api computer_v2.rs / computer_subtask.rs /
10
+ // computer_parallel.rs and the driver docs tools/allternit-driver.mdx).
11
+ import { resultText, } from "./client.js";
12
+ import { COMPUTER_V2_STRUCTURED_MEMBER_NAMES, COMPUTER_V2_STRUCTURED_MEMBERS, COMPUTER_V2_TOOL_SCHEMAS, } from "./v2-generated.js";
13
+ export { COMPUTER_V2_MEMBER_NAMES, COMPUTER_V2_STRUCTURED_MEMBER_NAMES } from "./v2-generated.js";
14
+ /** Marker the Allternit wire tooling uses to recognise the v2 function tool. */
15
+ export const V2_TOOL_MARKER = "[allternit.computer.v2]";
16
+ /** The function-tool name the structured members are exposed under. */
17
+ export const COMPUTER_V2_TOOL_NAME = "computer_v2";
18
+ // --------------------------------------------------------------------- errors
19
+ /** A v2 member answered is_error:true, or returned no JSON where JSON was expected. */
20
+ export class ComputerV2Error extends Error {
21
+ member;
22
+ result;
23
+ constructor(member, result, message) {
24
+ super(message ?? (resultText(result) || `The ${member} call failed.`));
25
+ this.name = "ComputerV2Error";
26
+ this.member = member;
27
+ this.result = result;
28
+ }
29
+ }
30
+ /** Run one structured v2 member through POST /v1/computers/{id}/toolset, answering an approval hold via `onApproval`. */
31
+ export function runComputerV2Member(o, member, input) {
32
+ return o.client.toolsetWithApproval(o.computerId, { toolset: "computer", member, input: (input ?? {}) }, o.onApproval);
33
+ }
34
+ /**
35
+ * Typed methods for the ten structured v2 members of one computer. Construct
36
+ * once per computer and pass it to your loop, or call the members directly.
37
+ */
38
+ export class ComputerV2Driver {
39
+ #o;
40
+ constructor(o) {
41
+ this.#o = o;
42
+ }
43
+ /** Read the target window or app's UI as a structured element tree (no screenshot). */
44
+ read_ui(input = {}) {
45
+ return this.#json("read_ui", input);
46
+ }
47
+ /** Act on an element id from read_ui. */
48
+ act(input) {
49
+ return this.#json("act", input);
50
+ }
51
+ /** Run ordered steps in one window in a single call; a failed check stops the batch. */
52
+ run_batch(input) {
53
+ return this.#json("run_batch", input);
54
+ }
55
+ /** Check bounded conditions on the current UI without acting. */
56
+ verify(input) {
57
+ return this.#json("verify", input);
58
+ }
59
+ /** Pause and hand the computer to a person; resolves when they signal done or the timeout elapses. */
60
+ request_human(input = {}) {
61
+ return this.#text("request_human", input);
62
+ }
63
+ /** Type a vault credential into the focused field; the value never enters the model context. */
64
+ use_credential(input) {
65
+ return this.#text("use_credential", input);
66
+ }
67
+ /** Hand a bounded UI subtask to the fast decision loop. */
68
+ run_subtask(input) {
69
+ return this.#json("run_subtask", input);
70
+ }
71
+ /** Run independent bounded subtasks on separate computers at the same time. */
72
+ run_parallel(input) {
73
+ return this.#json("run_parallel", input);
74
+ }
75
+ /** Run a saved skill (a recording taught with run_subtask save_as) by name. */
76
+ run_skill(input) {
77
+ return this.#json("run_skill", input);
78
+ }
79
+ /** List the saved skills this person can run (pass forget to delete one first). */
80
+ skills(input = {}) {
81
+ return this.#json("skills", input);
82
+ }
83
+ async #text(member, input) {
84
+ const res = await runComputerV2Member(this.#o, member, input);
85
+ if (res.is_error)
86
+ throw new ComputerV2Error(member, res);
87
+ return resultText(res);
88
+ }
89
+ async #json(member, input) {
90
+ const res = await runComputerV2Member(this.#o, member, input);
91
+ if (res.is_error)
92
+ throw new ComputerV2Error(member, res);
93
+ const text = resultText(res);
94
+ try {
95
+ return JSON.parse(text);
96
+ }
97
+ catch {
98
+ throw new ComputerV2Error(member, res, `The ${member} call returned no JSON.`);
99
+ }
100
+ }
101
+ }
102
+ // ---------------------------------------------------------- the computer_v2 tool
103
+ // Steering every model gets for the structured members. The canonical copy
104
+ // gizzi sends is cmd/gizzi-code/src/runtime/tools/computer-toolset/adapter.ts
105
+ // (PREFER_STRUCTURED / PREFER_SUBTASK / toolDescriptionV2) — keep the wording
106
+ // in sync when either side changes.
107
+ const PREFER_STRUCTURED = [
108
+ "Prefer read_ui + act/run_batch over screenshot + pixel-by-pixel loops: the element tree is faster, cheaper and stable across resizes.",
109
+ "When the tree is empty (canvas/game/remote desktop), read_ui falls back to vision by itself: elements with source \"vision\" and a mark number act like any other id; pass target (e.g. 'the Export button') to ground one element. Reach for screenshots only when that still isn't enough.",
110
+ ].join(" ");
111
+ const PREFER_SUBTASK = [
112
+ "Plan, then delegate: give each bounded UI step sequence (fill a form, search and pick, toggle settings) to run_subtask with the goal, the literal inputs it may type and success checks, instead of choosing every click yourself.",
113
+ "A typical task is one run_subtask call plus your final answer. Keep your own calls for planning, for judgment the subtask hands back (status escalated: continue from the screen it returns), and for steps outside the UI.",
114
+ "API over GUI: when one of your MCP tools or an API does the goal, call it instead of driving the screen; when unsure, pass the candidates as run_subtask api_options (status use_api names the one to call).",
115
+ "Independent subtasks on separate computers go in one run_parallel call. For a high-value subtask, best_of N with N sandbox computers runs N rollouts and a judge picks one by their step narratives (never on the person's own machine).",
116
+ ].join(" ");
117
+ /** Description gizzi-equivalent for the `computer_v2` tool, built from the contract's member descriptions. */
118
+ export function computerV2ToolDescription() {
119
+ const lines = COMPUTER_V2_STRUCTURED_MEMBERS.map((m) => `- ${m.name}: ${m.description}`);
120
+ return [
121
+ `${V2_TOOL_MARKER} Read and drive this computer's UI as a structured element tree (no screenshots needed). Set \`action\` to one of the actions below and pass that action's fields.`,
122
+ PREFER_SUBTASK,
123
+ PREFER_STRUCTURED,
124
+ "Element ids come from read_ui and are stable until the UI changes; pass the version back to act/run_batch to catch a moved UI.",
125
+ "run_batch runs many steps in one call and stops at the first failed check, then returns one fresh read — batch aggressively.",
126
+ "use_credential types a vault secret or TOTP code into the focused field; the value is never shown to you.",
127
+ "request_human pauses for a person (CAPTCHA, 2FA, judgment calls) and resumes the session when they signal done.",
128
+ "",
129
+ ...lines,
130
+ ].join("\n");
131
+ }
132
+ const STRUCTURED_SCHEMAS = COMPUTER_V2_TOOL_SCHEMAS;
133
+ /**
134
+ * The combined JSON Schema for the `computer_v2` tool: `action` (one of the ten
135
+ * structured members) plus the union of every member's input fields, each
136
+ * optional — a call maps 1:1 onto `{ member: action, input: <the rest> }`, and
137
+ * the server validates against the member's own contract schema.
138
+ */
139
+ export function computerV2Parameters() {
140
+ const properties = {
141
+ action: { type: "string", enum: [...COMPUTER_V2_STRUCTURED_MEMBER_NAMES], description: "Which computer_v2 action to run." },
142
+ };
143
+ for (const name of COMPUTER_V2_STRUCTURED_MEMBER_NAMES) {
144
+ for (const [k, v] of Object.entries(STRUCTURED_SCHEMAS[name]?.properties ?? {})) {
145
+ if (!properties[k])
146
+ properties[k] = v;
147
+ }
148
+ }
149
+ return { type: "object", properties, required: ["action"], additionalProperties: false };
150
+ }
151
+ /** The provider-neutral `computer_v2` function tool definition. */
152
+ export function computerV2Tool() {
153
+ return { name: COMPUTER_V2_TOOL_NAME, description: computerV2ToolDescription(), schema: computerV2Parameters() };
154
+ }
package/package.json CHANGED
@@ -1,6 +1,60 @@
1
1
  {
2
2
  "name": "@allternit/computer-driver",
3
- "version": "0.0.0-stage",
4
- "stub": true,
5
- "description": "Temporary package placeholder for staged publishing"
6
- }
3
+ "version": "0.2.0",
4
+ "description": "Drop-in drivers for Allternit hosted computers: a plain client, contract-v2 structured UI driving (read_ui, act, run_batch, verify, run_subtask, …) and Anthropic (computer/browser toolset), OpenAI computer-use and Gemini computer_use adapters.",
5
+ "license": "Apache-2.0",
6
+ "type": "module",
7
+ "main": "./dist/index.js",
8
+ "types": "./dist/index.d.ts",
9
+ "exports": {
10
+ ".": {
11
+ "types": "./dist/index.d.ts",
12
+ "import": "./dist/index.js"
13
+ },
14
+ "./anthropic": {
15
+ "types": "./dist/anthropic.d.ts",
16
+ "import": "./dist/anthropic.js"
17
+ },
18
+ "./openai": {
19
+ "types": "./dist/openai.d.ts",
20
+ "import": "./dist/openai.js"
21
+ },
22
+ "./gemini": {
23
+ "types": "./dist/gemini.d.ts",
24
+ "import": "./dist/gemini.js"
25
+ }
26
+ },
27
+ "files": [
28
+ "dist",
29
+ "README.md",
30
+ "CHANGELOG.md"
31
+ ],
32
+ "engines": {
33
+ "node": ">=18"
34
+ },
35
+ "scripts": {
36
+ "build": "tsc -p tsconfig.json",
37
+ "typecheck": "tsc -p tsconfig.json --noEmit"
38
+ },
39
+ "peerDependencies": {
40
+ "@anthropic-ai/sdk": ">=0.132.0"
41
+ },
42
+ "peerDependenciesMeta": {
43
+ "@anthropic-ai/sdk": {
44
+ "optional": true
45
+ }
46
+ },
47
+ "devDependencies": {
48
+ "@anthropic-ai/sdk": "0.132.0",
49
+ "typescript": "^5.9.0"
50
+ },
51
+ "repository": {
52
+ "type": "git",
53
+ "url": "git+https://github.com/Allternit/allternit-platform.git",
54
+ "directory": "sdk/computer-driver"
55
+ },
56
+ "homepage": "https://docs.allternit.com/api/platform/sdks/typescript",
57
+ "publishConfig": {
58
+ "access": "public"
59
+ }
60
+ }