openmausbot 0.1.78 → 0.1.80

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (56) hide show
  1. package/dist/assets/index-Cu8BkIfo.js +305 -0
  2. package/dist/assets/{index-BqDJqf2C.js → index-D6RZKWks.js} +1 -1
  3. package/dist/assets/index-tfpeJAIG.css +1 -0
  4. package/dist/index.html +2 -2
  5. package/dist-server/container-mcp.js +11 -5
  6. package/dist-server/drivers/agents-proxy.js +21 -6
  7. package/dist-server/enterprise/server/index.js +0 -1
  8. package/dist-server/index.js +15942 -6859
  9. package/dist-server/local-computer.js +0 -1
  10. package/dist-server/mcp-server.js +8 -1
  11. package/dist-server/openmausbot.js +9303 -786
  12. package/dist-server/pair-cli.js +9303 -786
  13. package/dist-server/proxy-paths.js +0 -1
  14. package/dist-server/server/box.js +98 -41
  15. package/dist-server/server/browser-engine.js +1 -1
  16. package/dist-server/server/browser-runtime.js +11 -3
  17. package/dist-server/server/browser-tool-shape.js +72 -0
  18. package/dist-server/server/claude-accounts.js +1 -0
  19. package/dist-server/server/cli-setup.js +1 -1
  20. package/dist-server/server/composio.js +18 -0
  21. package/dist-server/server/config.js +16 -4
  22. package/dist-server/server/container-computer.js +0 -12
  23. package/dist-server/server/drivers/acp/core.js +4 -19
  24. package/dist-server/server/drivers/agents-proxy.js +27 -2
  25. package/dist-server/server/drivers/boxagent.js +48 -4
  26. package/dist-server/server/drivers/chat-mcp-tools.js +368 -0
  27. package/dist-server/server/drivers/chat-tool-approval.js +42 -0
  28. package/dist-server/server/drivers/claude.js +86 -38
  29. package/dist-server/server/drivers/codex.js +225 -75
  30. package/dist-server/server/drivers/grok.js +4 -0
  31. package/dist-server/server/drivers/minimax.js +11 -3
  32. package/dist-server/server/drivers/openai-chat-protocol.js +92 -0
  33. package/dist-server/server/drivers/openai-chat.js +332 -88
  34. package/dist-server/server/drivers/openai-compat.js +4 -0
  35. package/dist-server/server/drivers/pi.js +1 -9
  36. package/dist-server/server/harness/registry.js +2 -2
  37. package/dist-server/server/index.js +267 -50
  38. package/dist-server/server/managed-desktop.js +205 -0
  39. package/dist-server/server/model-context-window.js +20 -0
  40. package/dist-server/server/proxy-paths.js +0 -1
  41. package/dist-server/server/redact.js +1 -0
  42. package/dist-server/server/room-handoffs.js +57 -9
  43. package/dist-server/server/store.js +13 -1
  44. package/dist-server/server/tts/chatterbox.js +83 -0
  45. package/dist-server/server/tts/index.js +45 -12
  46. package/dist-server/server/workspace.js +12 -2
  47. package/dist-server/shared/ask-question.js +28 -0
  48. package/dist-server/shared/computer-contention.js +17 -0
  49. package/dist-server/vps-container-mcp.js +11 -5
  50. package/enterprise/server/index.js +0 -1
  51. package/package.json +1 -1
  52. package/dist/assets/index-DjwAroIZ.js +0 -305
  53. package/dist/assets/index-Pbb6Ao0s.css +0 -1
  54. package/dist-server/computer-proxy.js +0 -1196
  55. package/dist-server/server/computer-proxy.js +0 -1070
  56. package/dist-server/server/remote-computer.js +0 -170
@@ -1,1070 +0,0 @@
1
- // computer-proxy — a minimal MCP stdio server the claude CLI spawns
2
- // (agentcal's permission-proxy pattern, dedicated entry file so there is
3
- // no argv-dispatch fork-bomb hazard). It gives the agent its bot's cloud
4
- // computer (box.ascii.dev) as CUA-grade tools.
5
- //
6
- // Transport: every action goes through the box's REST run-command
7
- // endpoint (no inbound port on the box, no tunnel), so a round trip is
8
- // expensive (~TLS + shell spawn). The whole design is therefore built
9
- // around ONE round trip per step:
10
- //
11
- // act + settle + capture + base64 all run in a single shell command,
12
- // and the resulting frame rides back in the SAME tool result as an MCP
13
- // image block ("act and observe"). The agent never needs a follow-up
14
- // screenshot call, which halves the model inferences per UI step —
15
- // the same shape Anthropic's own computer-use loop uses.
16
- //
17
- // Other latency rules that live here:
18
- // - JPEG, not PNG (5-10x fewer bytes, identical vision tokens), and
19
- // the downscale only runs when the display is wider than the model's
20
- // coordinate space.
21
- // - Coordinate scaling happens box-side in shell arithmetic, so there
22
- // is no separate "what size is the display" round trip per turn.
23
- // - Frames come back inline in stdout when small enough; the files API
24
- // is only a fallback (one extra hop) for big ones.
25
- // - computer_batch runs a whole mechanical sequence (click, type, tab,
26
- // type, Enter) in one round trip with one frame at the end.
27
- //
28
- // stdout is the MCP channel — never console.log here.
29
- import { normalizeBrowserUrl, normalizeCrop, ObservationCoordinator, parseBrowserTargets, safeBrowserUrl, } from "./computer-observation.js";
30
- import { CONTROL_REFUSAL, createControlClient } from "./control-client.js";
31
- import { ensureRemoteCuaCommand, isolatedRemoteCommand, MAX_REMOTE_COMMAND_LENGTH, REMOTE_CUA_EXECUTABLE, REMOTE_CUA_SESSION, REMOTE_CUA_SOCKET, REMOTE_CUA_VERSION, semanticBrowserCommand, } from "./remote-computer.js";
32
- const BOX_API = process.env.OGB_BOX_API ?? "https://ascii.dev/api/box/v1";
33
- const boxId = process.env.OGB_BOX_ID ?? "";
34
- const token = process.env.OGB_BOX_TOKEN ?? "";
35
- // Who-is-driving: while the person holds control in the app, every tool
36
- // below is refused (not queued — a queued click lands after they've moved
37
- // on). Even screenshot: the person may be typing a credential, and the
38
- // safest screen for the model to see is the one AFTER the hand-back.
39
- /** Poll cadence while waiting for a hand-back, and the patience ceiling.
40
- * Env-tunable so the contract test doesn't spend wall-clock on it. */
41
- const CONTROL_POLL_MS = Math.max(Number(process.env.OMB_CONTROL_POLL_MS) || 1_500, 25);
42
- const CONTROL_WAIT_MS = Math.max(Number(process.env.OMB_CONTROL_WAIT_MS) || 600_000, 100);
43
- // The cache must never outlive the poll cadence, or a hand-back would be
44
- // seen a stale cache-window late.
45
- const control = createControlClient({ cacheMs: Math.min(750, CONTROL_POLL_MS) });
46
- /** The coordinate space the model sees: frames are downscaled to this
47
- * width, and clicks are scaled back up to the real display box-side. */
48
- const SHOT_WIDTH = 1280;
49
- const JPEG_QUALITY = 75;
50
- const SHOT_PATH = "/tmp/ogb-shot.jpg";
51
- /** How long the desktop gets to repaint before the fused capture. */
52
- const SETTLE_MS = 350;
53
- /** Gap between batched actions so focus changes land before typing. */
54
- const ACTION_GAP_MS = 120;
55
- const CHROME_PROFILE = "$HOME/.openmausbot/chrome-profile";
56
- const CHROME_DEBUG_FLAGS = `--user-data-dir="${CHROME_PROFILE}" --password-store=basic --disable-session-crashed-bubble --no-first-run --remote-debugging-address=127.0.0.1 --remote-debugging-port=9222`;
57
- // Keep one durable browser identity regardless of which Chromium binary an
58
- // image supplies. Existing profiles are merged without overwriting files and
59
- // moved aside as backups before the conventional paths become symlinks.
60
- const CHROME_PROFILE_SETUP = [
61
- `profile="${CHROME_PROFILE}"`,
62
- 'mkdir -p "$profile" "$HOME/.config"',
63
- 'chmod 700 "$profile"',
64
- 'for browser_dir in "$HOME/.config/google-chrome" "$HOME/.config/chromium"; do',
65
- ' if [ -e "$browser_dir" ] && [ ! -L "$browser_dir" ]; then',
66
- ' if [ -d "$browser_dir" ] && ! cp -a -n "$browser_dir"/. "$profile"/; then',
67
- ' echo "failed to copy browser profile: $browser_dir" >&2',
68
- " exit 1",
69
- " fi",
70
- ' mv "$browser_dir" "$browser_dir.pre-openmausbot-$(date +%s)-$$"',
71
- " fi",
72
- ' if [ -L "$browser_dir" ]; then rm -f "$browser_dir"; fi',
73
- ' ln -s "$profile" "$browser_dir"',
74
- "done",
75
- ].join("\n");
76
- /** Frames larger than this come back over the files API instead of
77
- * inline stdout (keeps us clear of the command endpoint's stdout cap). */
78
- const INLINE_MAX_BYTES = 400_000;
79
- /** Boxes archive themselves when idle (billing pauses, the disk survives),
80
- * which can happen mid-conversation — after that every command comes back
81
- * 409 machine_not_running. Wake it and carry on rather than handing the
82
- * agent a cryptic failure it can only guess at. */
83
- async function resumeBox() {
84
- const auth = { authorization: `Bearer ${token}`, "content-type": "application/json" };
85
- await fetch(`${BOX_API}/boxes/${boxId}/resume`, { method: "POST", headers: auth }).catch(() => null);
86
- const deadline = Date.now() + 90_000;
87
- while (Date.now() < deadline) {
88
- await new Promise((r) => setTimeout(r, 2000));
89
- const res = await fetch(`${BOX_API}/boxes/${boxId}`, { headers: auth }).catch(() => null);
90
- const body = await res?.json().catch(() => null);
91
- const state = body?.box?.state;
92
- if (state && ["idle", "ready", "running"].includes(state))
93
- return true;
94
- if (state === "error")
95
- return false;
96
- }
97
- return false;
98
- }
99
- async function runOnBox(command, timeoutMs = 60_000, allowWake = true) {
100
- // Old boxes may predate noEnv:true. Run every agent-issued command with an
101
- // explicit desktop-only environment so provider/account credentials cannot
102
- // leak through `computer_exec` or a child GUI process.
103
- const isolatedCommand = isolatedRemoteCommand(command);
104
- const res = await fetch(`${BOX_API}/boxes/${boxId}/commands`, {
105
- method: "POST",
106
- headers: { authorization: `Bearer ${token}`, "content-type": "application/json" },
107
- body: JSON.stringify({ command: isolatedCommand }),
108
- signal: AbortSignal.timeout(timeoutMs),
109
- });
110
- const body = await res.json().catch(() => null);
111
- if (res.status === 409 && allowWake) {
112
- const code = body?.code ?? body?.error?.code ?? "";
113
- if (/machine_not_running|box_starting|not_running|starting/i.test(String(code))) {
114
- const woke = await resumeBox();
115
- if (woke)
116
- return runOnBox(command, timeoutMs, false);
117
- return { ok: false, exitCode: null, stdout: "", stderr: "the computer is asleep and did not wake in time" };
118
- }
119
- }
120
- return {
121
- ok: res.ok && body?.exitCode === 0,
122
- exitCode: body?.exitCode ?? null,
123
- stdout: body?.stdout ?? "",
124
- stderr: body?.stderr ?? String(body?.message ?? (res.ok ? "" : `HTTP ${res.status}`)),
125
- };
126
- }
127
- const observations = new ObservationCoordinator();
128
- function metricsText() {
129
- return JSON.stringify(observations.metrics);
130
- }
131
- async function browserTargets(countObservation = true) {
132
- // DevTools stays loopback-only inside the box. Only redacted fields are
133
- // ever formatted into tool output; comparisonUrl remains internal.
134
- const out = await runOnBox("curl -sf --max-time 2 http://127.0.0.1:9222/json/list", 5_000);
135
- const targets = out.ok ? parseBrowserTargets(out.stdout) : [];
136
- if (countObservation && targets.length)
137
- observations.noteStructuredObservation();
138
- return targets;
139
- }
140
- async function waitForNavigation(value, attempts = 3) {
141
- const expected = normalizeBrowserUrl(value);
142
- if (!expected) {
143
- observations.noteVerification(false);
144
- return { ok: false, targets: [] };
145
- }
146
- let targets = [];
147
- for (let attempt = 0; attempt < attempts; attempt += 1) {
148
- if (attempt > 0) {
149
- observations.noteRetry();
150
- await new Promise((resolve) => setTimeout(resolve, 1_000));
151
- }
152
- targets = await browserTargets(false);
153
- if (targets.some((target) => target.comparisonUrl === expected)) {
154
- observations.noteVerification(true);
155
- return { ok: true, targets };
156
- }
157
- }
158
- observations.noteVerification(false);
159
- return { ok: false, targets };
160
- }
161
- const ENV = 'export DISPLAY=${DISPLAY:-:0}';
162
- const CUA_ENV = "CUA_DRIVER_INSTALL_CHANNEL=python_package CUA_DRIVER_RS_TELEMETRY_ENABLED=0";
163
- /** Resolve the real display size into $W/$H for box-side click scaling. */
164
- const GEOMETRY = [
165
- "g=$(xdotool getdisplaygeometry 2>/dev/null)",
166
- 'W=${g%% *}',
167
- 'H=${g##* }',
168
- `case "$W" in ''|*[!0-9]*) W=${SHOT_WIDTH}; H=0;; esac`,
169
- ].join("; ");
170
- /** Shell that turns a screenshot-space coordinate into a display one.
171
- * The capture only downscales when the display is WIDER than the model's
172
- * space, so scaling must be conditional on exactly the same test — on a
173
- * 1024-wide desktop the frame is native size and a blind /1280 would put
174
- * every click at 80% of where the model aimed. */
175
- function scaled(varName, value) {
176
- const v = Math.round(value);
177
- return `if [ "$W" -gt ${SHOT_WIDTH} ] 2>/dev/null; then ${varName}=$(( ${v} * W / ${SHOT_WIDTH} )); else ${varName}=${v}; fi`;
178
- }
179
- /** Prefer the official driver but keep the proven X11 command as a degraded
180
- * path while a first install is finishing or if the daemon needs repair. */
181
- function cuaOrX11(tool, argumentsShell, fallback) {
182
- return [
183
- `if [ -x ${REMOTE_CUA_EXECUTABLE} ] && ${REMOTE_CUA_EXECUTABLE} status --socket ${REMOTE_CUA_SOCKET} >/dev/null 2>&1;`,
184
- `then if CUA_OUT=$(env ${CUA_ENV} ${REMOTE_CUA_EXECUTABLE} call ${tool} ${argumentsShell} --socket ${REMOTE_CUA_SOCKET} 2>/tmp/ogb-cua-call.error);`,
185
- `then echo "BACKEND CUA"; echo "CUA_RESULT $(printf %s "$CUA_OUT" | base64 -w0 2>/dev/null || printf %s "$CUA_OUT" | base64 | tr -d '\\n')"`,
186
- `else ${fallback}; X11_RC=$?; echo "BACKEND X11"; [ "$X11_RC" -eq 0 ]; fi`,
187
- `else ${fallback}; X11_RC=$?; echo "BACKEND X11"; [ "$X11_RC" -eq 0 ]; fi`,
188
- ].join(" ");
189
- }
190
- /** act → settle → capture → canonical hash → optional crop → inline bytes.
191
- * The hash is taken before cropping, so change detection always describes
192
- * the full screen. A requested crop fails closed when conversion fails. */
193
- function captureBlock(settleMs = SETTLE_MS, crop = null) {
194
- const downscale = crop
195
- ? `if [ "$W" -gt ${SHOT_WIDTH} ] 2>/dev/null; then if ! command -v convert >/dev/null 2>&1 || ! convert "$f" -thumbnail ${SHOT_WIDTH}x -quality ${JPEG_QUALITY} "$f" 2>/dev/null; then echo CROP_FAILED; exit 0; fi; fi`
196
- : `if [ "$W" -gt ${SHOT_WIDTH} ] 2>/dev/null && command -v convert >/dev/null 2>&1; then convert "$f" -thumbnail ${SHOT_WIDTH}x -quality ${JPEG_QUALITY} "$f" 2>/dev/null || true; fi`;
197
- const cropSteps = crop
198
- ? [
199
- `if ! command -v convert >/dev/null 2>&1 || ! convert "$f" -crop ${crop.width}x${crop.height}+${crop.x}+${crop.y} +repage "$f" 2>/dev/null; then echo CROP_FAILED; exit 0; fi`,
200
- `if [ ! -s "$f" ]; then echo CROP_FAILED; exit 0; fi`,
201
- ]
202
- : [];
203
- return [
204
- settleMs > 0 ? `sleep ${(settleMs / 1000).toFixed(2)}` : "true",
205
- `f=${SHOT_PATH}`,
206
- 'raw=/tmp/ogb-shot.png',
207
- `rm -f "$f" 2>/dev/null || true`,
208
- `rm -f "$raw" 2>/dev/null || true`,
209
- `if [ -x ${REMOTE_CUA_EXECUTABLE} ] && ${REMOTE_CUA_EXECUTABLE} status --socket ${REMOTE_CUA_SOCKET} >/dev/null 2>&1 && env ${CUA_ENV} ${REMOTE_CUA_EXECUTABLE} call get_desktop_state ${shellQuote(JSON.stringify({ scope: "desktop", session: REMOTE_CUA_SESSION }))} --socket ${REMOTE_CUA_SOCKET} --screenshot-out-file "$raw" >/dev/null 2>&1 && command -v convert >/dev/null 2>&1 && convert "$raw" -quality ${JPEG_QUALITY} "$f" 2>/dev/null; then echo "CAPTURE CUA"; else scrot -o -q ${JPEG_QUALITY} "$f" 2>/dev/null || import -window root -quality ${JPEG_QUALITY} "$f" 2>/dev/null || ffmpeg -y -f x11grab -i "$DISPLAY" -frames:v 1 -q:v 6 "$f" >/dev/null 2>&1; echo "CAPTURE X11"; fi`,
210
- // only re-encode when the display is bigger than the model's space —
211
- // ImageMagick startup is the most expensive step in the old pipeline
212
- downscale,
213
- `if [ ! -s "$f" ]; then echo SHOT_FAILED; exit 0; fi`,
214
- 'echo "GEOM $W $H"',
215
- 'echo "HASH $(md5sum "$f" 2>/dev/null | cut -d\' \' -f1)"',
216
- ...cropSteps,
217
- 's=$(stat -c%s "$f" 2>/dev/null || echo 0)',
218
- // SIZE is what makes the inline path safe: the frame is only trusted
219
- // when the bytes we decoded match the bytes the box says it wrote
220
- 'echo "SIZE $s"',
221
- `if [ "$s" -gt 0 ] && [ "$s" -le ${INLINE_MAX_BYTES} ]; then echo "B64 $(base64 -w0 "$f" 2>/dev/null || base64 "$f" | tr -d '\\n')"; fi`,
222
- ].join("; ");
223
- }
224
- /** A frame is only trusted when the bytes are a WHOLE image. Checking the
225
- * magic number alone is not enough: the box's command stdout has been
226
- * observed truncating a payload, and a truncated JPEG still starts with a
227
- * valid header — it just renders as a grey half-frame for the model. So
228
- * every frame must also end with its terminator, and (when the box told
229
- * us how many bytes it wrote) match that length exactly. */
230
- function wholeImage(bytes, expectedBytes) {
231
- if (bytes.length < 512)
232
- return false;
233
- if (expectedBytes && bytes.length !== expectedBytes)
234
- return false;
235
- const jpeg = bytes[0] === 0xff && bytes[1] === 0xd8;
236
- const png = bytes[0] === 0x89 && bytes[1] === 0x50 && bytes[2] === 0x4e && bytes[3] === 0x47;
237
- if (jpeg) {
238
- // EOI marker, allowing for trailing padding some encoders append
239
- const tail = bytes.subarray(Math.max(0, bytes.length - 32));
240
- return tail.includes(Buffer.from([0xff, 0xd9]));
241
- }
242
- if (png) {
243
- const tail = bytes.subarray(Math.max(0, bytes.length - 12));
244
- return tail.includes(Buffer.from("IEND", "ascii"));
245
- }
246
- return false;
247
- }
248
- /** Big frames (and any inline read that came back malformed) are fetched
249
- * over HTTP: raw artifact bytes first, the files API's base64-in-JSON
250
- * envelope second. Both are validated — an error page served with a 200
251
- * must fall through, not reach the model as an "image". */
252
- async function fetchFrame(expectedBytes) {
253
- const auth = { authorization: `Bearer ${token}` };
254
- try {
255
- const res = await fetch(`${BOX_API}/boxes/${boxId}/artifacts?path=${encodeURIComponent(SHOT_PATH)}`, { headers: auth, signal: AbortSignal.timeout(30_000) });
256
- if (res.ok) {
257
- const bytes = Buffer.from(await res.arrayBuffer());
258
- if (wholeImage(bytes, expectedBytes))
259
- return bytes.toString("base64");
260
- }
261
- }
262
- catch {
263
- /* fall through to the files API */
264
- }
265
- try {
266
- const res = await fetch(`${BOX_API}/boxes/${boxId}/files?path=${encodeURIComponent(SHOT_PATH)}&encoding=base64`, { headers: auth, signal: AbortSignal.timeout(30_000) });
267
- const body = await res.json().catch(() => null);
268
- const content = body?.content;
269
- if (!res.ok || typeof content !== "string" || !content)
270
- return null;
271
- return wholeImage(Buffer.from(content, "base64"), expectedBytes) ? content : null;
272
- }
273
- catch {
274
- return null;
275
- }
276
- }
277
- let inlineWorks = true; // flipped off for the proxy's life on first garbage
278
- let lastDisplayGeometry = null;
279
- let semanticBrowserUrl = null;
280
- let semanticBrowserRefs = new Set();
281
- function geometryFrom(stdout) {
282
- const match = stdout.match(/^GEOM\s+(\d+)\s+(\d+)$/m);
283
- if (!match)
284
- return null;
285
- const width = Number(match[1]);
286
- const height = Number(match[2]);
287
- return width > 0 && height > 0 ? { width, height } : null;
288
- }
289
- function automationSummary(stdout) {
290
- if (/^BACKEND CUA$/m.test(stdout)) {
291
- const encoded = stdout.match(/^CUA_RESULT\s+([^\s]+)$/m)?.[1];
292
- if (!encoded)
293
- return `Cua Driver ${REMOTE_CUA_VERSION}`;
294
- try {
295
- const result = JSON.parse(Buffer.from(encoded, "base64").toString("utf8"));
296
- const details = [result.effect, result.route, result.escalation]
297
- .filter((value) => typeof value === "string" && Boolean(value))
298
- .slice(0, 3);
299
- return [`Cua Driver ${REMOTE_CUA_VERSION}`, ...details].join(" · ");
300
- }
301
- catch {
302
- return `Cua Driver ${REMOTE_CUA_VERSION}`;
303
- }
304
- }
305
- return /^BACKEND X11$/m.test(stdout) ? "X11 fallback" : "automation backend unavailable";
306
- }
307
- async function observationBounds() {
308
- let geometry = lastDisplayGeometry;
309
- if (!geometry) {
310
- const out = await runOnBox([ENV, GEOMETRY, 'echo "GEOM $W $H"'].join("; "), 15_000);
311
- geometry = geometryFrom(out.stdout);
312
- if (geometry)
313
- lastDisplayGeometry = geometry;
314
- }
315
- if (!geometry)
316
- return null;
317
- const scale = geometry.width > SHOT_WIDTH ? SHOT_WIDTH / geometry.width : 1;
318
- return {
319
- width: Math.round(geometry.width * scale),
320
- height: Math.round(geometry.height * scale),
321
- };
322
- }
323
- async function frameFrom(out) {
324
- if (/SHOT_FAILED|CROP_FAILED/.test(out.stdout))
325
- return null;
326
- let hash = null;
327
- let geometry = null;
328
- let inline = "";
329
- let size = 0;
330
- for (const line of out.stdout.split("\n")) {
331
- if (line.startsWith("HASH "))
332
- hash = line.slice(5).trim() || null;
333
- else if (line.startsWith("SIZE "))
334
- size = Number(line.slice(5).trim()) || 0;
335
- else if (line.startsWith("GEOM ")) {
336
- const [w, h] = line.slice(5).trim().split(/\s+/).map(Number);
337
- if (Number.isFinite(w) && w > 0)
338
- geometry = { width: w, height: Number.isFinite(h) ? h : 0 };
339
- }
340
- else if (line.startsWith("B64 "))
341
- inline = line.slice(4).trim();
342
- }
343
- if (geometry?.height)
344
- lastDisplayGeometry = geometry;
345
- if (inline && inlineWorks) {
346
- const bytes = Buffer.from(inline, "base64");
347
- if (wholeImage(bytes, size || undefined))
348
- return { data: inline, mime: "image/jpeg", hash, geometry };
349
- // stdout mangled it (the failure this channel is known for) — never
350
- // hand a partial frame to the model; fetch it and stop trusting stdout
351
- inlineWorks = false;
352
- }
353
- const fetched = await fetchFrame(size || undefined);
354
- if (!fetched)
355
- return null;
356
- return { data: fetched, mime: "image/jpeg", hash, geometry };
357
- }
358
- const send = (obj) => {
359
- process.stdout.write(JSON.stringify(obj) + "\n");
360
- };
361
- const text = (id, t, isError = false) => send({ jsonrpc: "2.0", id, result: { content: [{ type: "text", text: t }], isError: isError || undefined } });
362
- /** An action result: the text plus the frame the action produced. When
363
- * the pixels are byte-identical to the frame the model just saw, the
364
- * image is dropped — it already has it, and it costs ~1.2k tokens. */
365
- function observed(id, note, frame, crop = null, followsAction = true, isError = false) {
366
- if (!frame) {
367
- return text(id, `${note}\n(couldn't capture the screen — call screenshot to retry)`, isError);
368
- }
369
- const observation = observations.observeFrame(frame.hash ?? (crop ? null : frame.data), crop);
370
- if (!observation.changed) {
371
- // deliberately does NOT suggest repeating the action: the action may
372
- // well have landed, and re-clicking a button that already submitted
373
- // is the expensive kind of wrong
374
- const guidance = followsAction
375
- ? " Don't repeat the action — it may already have succeeded. If you expected a change, call screenshot again after it has had time to render."
376
- : " No new image is attached.";
377
- return text(id, `${note}\n(the screen is identical to the frame you already have.${guidance})`, isError);
378
- }
379
- send({
380
- jsonrpc: "2.0",
381
- id,
382
- result: {
383
- content: [
384
- { type: "text", text: note },
385
- { type: "image", data: frame.data, mimeType: frame.mime },
386
- ],
387
- isError: isError || undefined,
388
- },
389
- });
390
- }
391
- const OBSERVE_PROPS = {
392
- observe: {
393
- type: "boolean",
394
- description: "default true — return a fresh screenshot with the result. Set false only when chaining mechanical steps you don't need to see.",
395
- },
396
- settle_ms: { type: "number", description: "wait before the screenshot, default 350, max 3000" },
397
- };
398
- const TOOLS = [
399
- {
400
- name: "screenshot",
401
- description: "See the bot's cloud computer screen when visual state is needed. First prefer browser_state for Chrome title/URL checks. The frame is captured fresh; byte-identical pixels are not resent.",
402
- inputSchema: {
403
- type: "object",
404
- properties: {
405
- region: {
406
- type: "object",
407
- description: "Optional crop in the coordinates of the last screenshot.",
408
- properties: {
409
- x: { type: "number" },
410
- y: { type: "number" },
411
- width: { type: "number" },
412
- height: { type: "number" },
413
- },
414
- required: ["x", "y", "width", "height"],
415
- },
416
- },
417
- },
418
- },
419
- {
420
- name: "browser_state",
421
- description: "Read structured Chrome page titles and safe URLs. Credentials, query strings, and fragments are removed before output.",
422
- inputSchema: { type: "object", properties: {} },
423
- },
424
- {
425
- name: "browser_snapshot",
426
- description: "Read Chrome's semantic accessibility tree and return fresh element refs. Prefer this over screenshots for links, buttons, and form fields.",
427
- inputSchema: { type: "object", properties: {} },
428
- },
429
- {
430
- name: "browser_click",
431
- description: "Click one element ref from the most recent browser_snapshot and return the resulting screen.",
432
- inputSchema: {
433
- type: "object",
434
- properties: { ref: { type: "string" }, ...OBSERVE_PROPS },
435
- required: ["ref"],
436
- },
437
- },
438
- {
439
- name: "browser_fill",
440
- description: "Replace the text in one field ref from the most recent browser_snapshot and return the resulting screen.",
441
- inputSchema: {
442
- type: "object",
443
- properties: { ref: { type: "string" }, text: { type: "string" }, ...OBSERVE_PROPS },
444
- required: ["ref", "text"],
445
- },
446
- },
447
- {
448
- name: "wait_for_navigation",
449
- description: "Verify that Chrome reached one exact http(s) URL, including its query and fragment, with at most three bounded checks.",
450
- inputSchema: {
451
- type: "object",
452
- properties: { url: { type: "string" } },
453
- required: ["url"],
454
- },
455
- },
456
- {
457
- name: "observation_metrics",
458
- description: "Return this turn's observation, action, retry, and verification counters.",
459
- inputSchema: { type: "object", properties: {} },
460
- },
461
- {
462
- name: "computer_status",
463
- description: "Report whether the cloud computer is using Cua Driver or the degraded X11 fallback.",
464
- inputSchema: { type: "object", properties: {} },
465
- },
466
- {
467
- name: "computer_request_help",
468
- description: "Ask the person to take over this computer (a login, a CAPTCHA, anything you should not do alone) and wait until they hand control back. Also call it with no reason when an action was refused because a person is already driving. You cannot take control yourself — this only asks.",
469
- inputSchema: {
470
- type: "object",
471
- properties: {
472
- reason: {
473
- type: "string",
474
- description: "one short sentence the person will read — what you need their hands for",
475
- },
476
- },
477
- },
478
- },
479
- {
480
- name: "click",
481
- description: "Click on the computer's screen and return the resulting screen. Use pixel coordinates exactly as they appear in the last frame you were given — any scaling to the real display is handled for you.",
482
- inputSchema: {
483
- type: "object",
484
- properties: {
485
- x: { type: "number" },
486
- y: { type: "number" },
487
- button: { type: "string", enum: ["left", "right"], description: "default left" },
488
- double: { type: "boolean", description: "double-click" },
489
- },
490
- required: ["x", "y"],
491
- },
492
- },
493
- {
494
- name: "type_text",
495
- description: "Type text at the current focus and return the resulting screen.",
496
- inputSchema: {
497
- type: "object",
498
- properties: { text: { type: "string" }, ...OBSERVE_PROPS },
499
- required: ["text"],
500
- },
501
- },
502
- {
503
- name: "press_key",
504
- description: 'Press a key or chord and return the resulting screen. xdotool syntax: "Return", "Tab", "ctrl+c", "alt+F4", "ctrl+shift+t".',
505
- inputSchema: {
506
- type: "object",
507
- properties: { keys: { type: "string" }, ...OBSERVE_PROPS },
508
- required: ["keys"],
509
- },
510
- },
511
- {
512
- name: "scroll",
513
- description: "Scroll the screen up or down by N clicks and return the resulting screen.",
514
- inputSchema: {
515
- type: "object",
516
- properties: {
517
- direction: { type: "string", enum: ["up", "down"] },
518
- clicks: { type: "number", description: "default 3" },
519
- ...OBSERVE_PROPS,
520
- },
521
- required: ["direction"],
522
- },
523
- },
524
- {
525
- name: "computer_batch",
526
- description: "Run several UI actions in ONE go and return the screen at the end — much faster than separate calls (one round trip, one screenshot). Use it for mechanical sequences you can predict without looking in between, e.g. click a field, type, Tab, type, press Return. Stop the batch before anything whose outcome you need to see first.",
527
- inputSchema: {
528
- type: "object",
529
- properties: {
530
- actions: {
531
- type: "array",
532
- description: "in order; each is {action: click|type_text|press_key|scroll|wait, ...its params}",
533
- items: {
534
- type: "object",
535
- properties: {
536
- action: { type: "string", enum: ["click", "type_text", "press_key", "scroll", "wait"] },
537
- x: { type: "number" },
538
- y: { type: "number" },
539
- button: { type: "string", enum: ["left", "right"] },
540
- double: { type: "boolean" },
541
- text: { type: "string" },
542
- keys: { type: "string" },
543
- direction: { type: "string", enum: ["up", "down"] },
544
- clicks: { type: "number" },
545
- ms: { type: "number", description: "wait: milliseconds, max 5000" },
546
- },
547
- required: ["action"],
548
- },
549
- },
550
- ...OBSERVE_PROPS,
551
- },
552
- required: ["actions"],
553
- },
554
- },
555
- {
556
- name: "computer_exec",
557
- description: "Run a shell command on the bot's cloud computer (Linux, passwordless sudo, X11 desktop). Returns stdout/stderr/exit code — and, unlike the UI tools, no screenshot unless you ask for one.",
558
- inputSchema: {
559
- type: "object",
560
- properties: {
561
- command: { type: "string", maxLength: MAX_REMOTE_COMMAND_LENGTH },
562
- observe: {
563
- type: "boolean",
564
- description: "default false — set true to also return a screenshot (e.g. after launching a GUI app)",
565
- },
566
- },
567
- required: ["command"],
568
- },
569
- },
570
- {
571
- name: "wait_for",
572
- description: "Wait on the bot's cloud computer until a condition holds, then return in ONE call instead of repeatedly polling with computer_exec or screenshots. Give only the fields needed by the chosen condition.",
573
- inputSchema: {
574
- type: "object",
575
- properties: {
576
- condition: {
577
- type: "string",
578
- enum: ["http_ready", "tcp_ready", "output_matches", "file_exists"],
579
- description: "http_ready = a URL returns a successful response; tcp_ready = a local port accepts; output_matches = a bounded, read-only command's output matches a pattern; file_exists = a path appears",
580
- },
581
- url: { type: "string", description: "http_ready only: the URL to poll, e.g. http://localhost:3000/health" },
582
- port: { type: "integer", description: "tcp_ready only: the local TCP port, e.g. 5432" },
583
- command: {
584
- type: "string",
585
- description: "output_matches only: a quick, read-only shell command whose combined output is checked each poll",
586
- },
587
- pattern: { type: "string", description: "output_matches only: extended regex the output must match, e.g. ready|listening" },
588
- path: { type: "string", description: "file_exists only: absolute path on the computer" },
589
- timeout_seconds: { type: "integer", description: "give up after this many seconds; default 60, max 240" },
590
- ...OBSERVE_PROPS,
591
- },
592
- required: ["condition"],
593
- },
594
- },
595
- {
596
- name: "open_url",
597
- description: "Open a URL in the computer's own Chrome, verify the exact destination when DevTools is available, and return the resulting screen.",
598
- inputSchema: {
599
- type: "object",
600
- properties: { url: { type: "string" }, ...OBSERVE_PROPS },
601
- required: ["url"],
602
- },
603
- },
604
- ];
605
- const shellQuote = (s) => `'${s.replace(/'/g, "'\\''")}'`;
606
- const settleOf = (args) => Math.min(Math.max(Number(args?.settle_ms) || SETTLE_MS, 0), 3000);
607
- const wantsFrame = (args) => args?.observe !== false;
608
- /** One action → the shell that performs it (scaling clicks box-side). */
609
- function actionShell(a) {
610
- const kind = String(a?.action ?? "");
611
- if (kind === "click") {
612
- const x = Math.round(Number(a.x));
613
- const y = Math.round(Number(a.y));
614
- if (!Number.isFinite(x) || !Number.isFinite(y))
615
- return { error: "click needs numeric x,y" };
616
- const btn = a.button === "right" ? 3 : 1;
617
- const rep = a.double ? "--repeat 2 --delay 60 " : "";
618
- const button = a.button === "right" ? "right" : "left";
619
- const count = a.double ? 2 : 1;
620
- const fallback = `xdotool mousemove $CX $CY click ${rep}${btn}`;
621
- const args = `$(printf '{"x":%s,"y":%s,"button":"${button}","count":${count},"scope":"desktop","session":"${REMOTE_CUA_SESSION}"}' "$CX" "$CY")`;
622
- return `${scaled("CX", x)}; ${scaled("CY", y)}; CUA_ARGS=${args}; ${cuaOrX11("click", '"$CUA_ARGS"', fallback)}`;
623
- }
624
- if (kind === "type_text") {
625
- const t = String(a.text ?? "");
626
- if (!t)
627
- return { error: "type_text needs text" };
628
- const cuaArgs = shellQuote(JSON.stringify({ text: t, scope: "desktop", session: REMOTE_CUA_SESSION }));
629
- return cuaOrX11("type_text", cuaArgs, `xdotool type --clearmodifiers --delay 8 -- ${shellQuote(t)}`);
630
- }
631
- if (kind === "press_key") {
632
- const keys = String(a.keys ?? "").replace(/[^\w+]/g, "");
633
- if (!keys)
634
- return { error: "press_key needs keys" };
635
- const parts = keys.split("+").filter(Boolean);
636
- const tool = parts.length > 1 ? "hotkey" : "press_key";
637
- const cuaArgs = shellQuote(JSON.stringify(parts.length > 1
638
- ? { keys: parts, scope: "desktop", session: REMOTE_CUA_SESSION }
639
- : { key: parts[0]?.toLowerCase(), scope: "desktop", session: REMOTE_CUA_SESSION }));
640
- return cuaOrX11(tool, cuaArgs, `xdotool key ${keys}`);
641
- }
642
- if (kind === "scroll") {
643
- const clicks = Math.min(Math.max(Math.round(Number(a.clicks) || 3), 1), 20);
644
- const btn = a.direction === "up" ? 4 : 5;
645
- const direction = a.direction === "up" ? "up" : "down";
646
- const fallback = `xdotool click --repeat ${clicks} ${btn}`;
647
- const args = `$(printf '{"x":%s,"y":%s,"direction":"${direction}","amount":${clicks},"by":"line","scope":"desktop","session":"${REMOTE_CUA_SESSION}"}' "$((W / 2))" "$((H / 2))")`;
648
- return `CUA_ARGS=${args}; ${cuaOrX11("scroll", '"$CUA_ARGS"', fallback)}`;
649
- }
650
- if (kind === "wait") {
651
- const ms = Math.min(Math.max(Number(a.ms) || 500, 0), 5000);
652
- return `sleep ${(ms / 1000).toFixed(2)}`;
653
- }
654
- return { error: `unknown action ${kind || "(missing)"}` };
655
- }
656
- /** The whole point: one round trip carries geometry, the actions, the
657
- * settle, the capture and the frame bytes. */
658
- async function actAndObserve(id, actions, note, args, timeoutMs = 60_000) {
659
- const parts = [];
660
- for (const a of actions) {
661
- const shell = actionShell(a);
662
- if (typeof shell !== "string")
663
- return text(id, shell.error, true);
664
- // X11 needs a beat between steps — a click that focuses a field and
665
- // an immediate type will drop leading characters
666
- if (parts.length)
667
- parts.push(`sleep ${(ACTION_GAP_MS / 1000).toFixed(2)}`);
668
- parts.push(shell);
669
- }
670
- observations.noteAction(actions.filter((action) => action?.action !== "wait").length);
671
- const observe = wantsFrame(args);
672
- // The actions run in a guarded group so a failing xdotool is REPORTED
673
- // rather than silently swallowed by the capture that follows it — but
674
- // the capture still runs, so the model always gets to see the state it
675
- // ended up in. Joining with ";" alone made a failed action look
676
- // identical to one that did nothing.
677
- const guarded = `if { ${parts.join("; ")}; }; then ACT=ok; else ACT=failed; fi`;
678
- const command = [
679
- ENV,
680
- GEOMETRY,
681
- ensureRemoteCuaCommand(),
682
- guarded,
683
- observe ? captureBlock(settleOf(args)) : "true",
684
- 'echo "ACT $ACT"',
685
- ].join("; ");
686
- const out = await runOnBox(command, timeoutMs);
687
- const acted = /^ACT ok$/m.test(out.stdout);
688
- if (!acted && !out.stdout.includes("GEOM")) {
689
- return text(id, `${note.replace(/^./, (c) => c.toLowerCase())} failed: ${out.stderr.slice(0, 200) || `exit ${out.exitCode}`}`, true);
690
- }
691
- const backend = automationSummary(out.stdout);
692
- const full = acted
693
- ? `${note}\n(${backend})`
694
- : `${note}\n(the action reported an error: ${out.stderr.slice(0, 160) || "no detail"}; ${backend})`;
695
- if (!observe)
696
- return text(id, full, !acted);
697
- return observed(id, full, await frameFrom(out));
698
- }
699
- async function semanticActAndObserve(id, action, ref, value, args) {
700
- if (!semanticBrowserUrl || !semanticBrowserRefs.has(ref)) {
701
- return text(id, "that browser ref is stale or unknown — take a new browser_snapshot", true);
702
- }
703
- const observe = wantsFrame(args);
704
- const semantic = semanticBrowserCommand(action, {
705
- ref,
706
- ...(action === "fill" ? { text: value ?? "" } : {}),
707
- url: semanticBrowserUrl,
708
- });
709
- const guarded = `if ${semantic}; then SEM=ok; else SEM=failed; fi`;
710
- const command = [
711
- ENV,
712
- GEOMETRY,
713
- guarded,
714
- ensureRemoteCuaCommand(),
715
- observe ? captureBlock(settleOf(args)) : "true",
716
- 'echo "SEM $SEM"',
717
- ].join("; ");
718
- observations.noteAction();
719
- const out = await runOnBox(command, action === "fill" ? 120_000 : 60_000);
720
- const acted = /^SEM ok$/m.test(out.stdout);
721
- // DOM mutations can invalidate backend node IDs; force a fresh snapshot
722
- // after every semantic action instead of risking a click on an old target.
723
- semanticBrowserRefs.clear();
724
- const note = acted
725
- ? action === "fill"
726
- ? `filled ${ref} with ${value?.length ?? 0} chars (trusted Chrome DevTools input)`
727
- : `clicked ${ref} (trusted Chrome DevTools input)`
728
- : `${action} ${ref} failed: ${out.stderr.slice(0, 200) || "the page changed; take a new browser_snapshot"}`;
729
- if (!observe)
730
- return text(id, note, !acted);
731
- return observed(id, note, await frameFrom(out));
732
- }
733
- /** The only tools a bot may use while the person is driving: asking to be
734
- * told when they finish, and the two that read nothing from the screen. */
735
- const OPEN_WHILE_DRIVEN = new Set(["computer_request_help", "computer_status", "observation_metrics"]);
736
- async function call(id, name, args) {
737
- if (!OPEN_WHILE_DRIVEN.has(name)) {
738
- const state = await control.state(true);
739
- if (state.held)
740
- return text(id, state.blockedReason ?? CONTROL_REFUSAL, true);
741
- }
742
- if (name === "computer_request_help") {
743
- if (!control.configured) {
744
- return text(id, "nobody can be paged for this computer right now — carry on carefully", true);
745
- }
746
- const initial = await control.state(true);
747
- if (initial.held && initial.blockedReason)
748
- return text(id, initial.blockedReason, true);
749
- // If the person is already driving, don't clobber whatever plea they
750
- // are reading — just wait for the hand-back.
751
- const requestId = initial.held ? null : await control.requestHelp(String(args?.reason ?? ""));
752
- if (!initial.held && requestId === null) {
753
- return text(id, "The person could not be paged for this computer right now. Carry on carefully or tell them in chat.", true);
754
- }
755
- let sawHold = initial.held;
756
- const deadline = Date.now() + CONTROL_WAIT_MS;
757
- while (Date.now() < deadline) {
758
- await new Promise((resolve) => setTimeout(resolve, CONTROL_POLL_MS));
759
- const state = await control.state(true);
760
- if (state.held && state.blockedReason) {
761
- if (requestId)
762
- await control.expireHelp(requestId);
763
- return text(id, state.blockedReason, true);
764
- }
765
- if (state.held)
766
- sawHold = true;
767
- if (!state.held && !state.helpOpen) {
768
- return text(id, sawHold
769
- ? "The person has finished driving and handed control back. The screen may have changed while they drove — take a fresh screenshot before your next action."
770
- : "The person saw your request and dismissed it without taking control. Carry on yourself.");
771
- }
772
- }
773
- if (requestId)
774
- await control.expireHelp(requestId);
775
- return text(id, "Nobody took control within the wait window. Carry on carefully, or ask again if you are truly stuck.", true);
776
- }
777
- if (name === "screenshot") {
778
- let crop = null;
779
- if (args.region !== undefined) {
780
- const bounds = await observationBounds();
781
- if (!bounds)
782
- return text(id, "crop unavailable: could not determine the screenshot dimensions", true);
783
- crop = normalizeCrop(args.region, bounds.width, bounds.height);
784
- if (!crop) {
785
- return text(id, `region must be at least 32×32 and stay within the ${bounds.width}×${bounds.height} screenshot`, true);
786
- }
787
- }
788
- const out = await runOnBox([ENV, GEOMETRY, ensureRemoteCuaCommand(), captureBlock(0, crop)].join("; "), 60_000);
789
- if (/CROP_FAILED/.test(out.stdout)) {
790
- return text(id, `crop failed: ${out.stderr.slice(0, 200) || "ImageMagick could not create the requested region"}`, true);
791
- }
792
- const frame = await frameFrom(out);
793
- if (!frame) {
794
- return text(id, `screenshot failed: ${out.stderr.slice(0, 200) || "capture produced no frame"}`, true);
795
- }
796
- return observed(id, crop ? "cropped screen captured" : "screen captured", frame, crop, false);
797
- }
798
- if (name === "browser_state") {
799
- const targets = await browserTargets();
800
- return text(id, targets.length
801
- ? `Structured browser state:\n${targets.map((target) => `- ${target.title || "Untitled"}: ${target.url}`).join("\n")}`
802
- : "Structured browser state unavailable. Use screenshot only if visual state is necessary.");
803
- }
804
- if (name === "browser_snapshot") {
805
- const out = await runOnBox(semanticBrowserCommand("snapshot", {}), 20_000);
806
- if (!out.ok) {
807
- semanticBrowserUrl = null;
808
- semanticBrowserRefs.clear();
809
- return text(id, "Semantic browser state is unavailable. Open Chrome with open_url, or use screenshot.", true);
810
- }
811
- try {
812
- const snapshot = JSON.parse(out.stdout);
813
- if (!Array.isArray(snapshot.elements) || typeof snapshot.url !== "string")
814
- throw new Error("invalid snapshot");
815
- semanticBrowserUrl = snapshot.url;
816
- semanticBrowserRefs = new Set(snapshot.elements.map((element) => element.ref));
817
- observations.noteStructuredObservation();
818
- const publicUrl = safeBrowserUrl(snapshot.url) ?? "URL unavailable";
819
- const lines = snapshot.elements.map((element) => `- [${element.ref}] ${element.role}${element.disabled ? " disabled" : ""}: ${element.name.replace(/\s+/g, " ").slice(0, 180)}`);
820
- return text(id, `Semantic browser snapshot — ${snapshot.title || "Untitled"}: ${publicUrl}\n${lines.join("\n") || "No interactive elements found."}`);
821
- }
822
- catch {
823
- semanticBrowserUrl = null;
824
- semanticBrowserRefs.clear();
825
- return text(id, "Chrome returned an invalid semantic snapshot; use screenshot.", true);
826
- }
827
- }
828
- if (name === "browser_click") {
829
- const ref = String(args.ref ?? "");
830
- return semanticActAndObserve(id, "click", ref, undefined, args);
831
- }
832
- if (name === "browser_fill") {
833
- const ref = String(args.ref ?? "");
834
- return semanticActAndObserve(id, "fill", ref, String(args.text ?? ""), args);
835
- }
836
- if (name === "wait_for_navigation") {
837
- const url = String(args.url ?? "");
838
- const publicUrl = safeBrowserUrl(url);
839
- if (!normalizeBrowserUrl(url) || !publicUrl) {
840
- observations.noteVerification(false);
841
- return text(id, "wait_for_navigation needs a valid http(s) URL", true);
842
- }
843
- const result = await waitForNavigation(url);
844
- return text(id, result.ok
845
- ? `navigation verified: ${publicUrl}`
846
- : `navigation not verified after 3 checks. Current structured state: ${result.targets.map((target) => target.url).join(", ") || "unavailable"}. Use screenshot only if needed.`, !result.ok);
847
- }
848
- if (name === "observation_metrics")
849
- return text(id, metricsText());
850
- if (name === "computer_status") {
851
- const command = [
852
- ENV,
853
- ensureRemoteCuaCommand(),
854
- `if [ -x ${REMOTE_CUA_EXECUTABLE} ] && ${REMOTE_CUA_EXECUTABLE} status --socket ${REMOTE_CUA_SOCKET} >/dev/null 2>&1; then`,
855
- ` echo "CUA $(${REMOTE_CUA_EXECUTABLE} --version)"`,
856
- ` env ${CUA_ENV} ${REMOTE_CUA_EXECUTABLE} call health_report '{}' --socket ${REMOTE_CUA_SOCKET} 2>/dev/null || true`,
857
- "else echo 'X11 fallback'; fi",
858
- ].join("\n");
859
- const out = await runOnBox(command, 20_000);
860
- if (!/^CUA /m.test(out.stdout)) {
861
- return text(id, "Cloud computer automation: X11 fallback (Cua Driver is still installing or needs repair).", true);
862
- }
863
- const overall = out.stdout.match(/"overall"\s*:\s*"(ok|degraded|failed)"/)?.[1] ?? "unknown";
864
- return text(id, `Cloud computer automation: Cua Driver ${REMOTE_CUA_VERSION} (${overall}).`);
865
- }
866
- if (name === "click") {
867
- const x = Math.round(Number(args.x));
868
- const y = Math.round(Number(args.y));
869
- if (!Number.isFinite(x) || !Number.isFinite(y))
870
- return text(id, "click needs numeric x,y", true);
871
- const what = `${args.double ? "double-clicked" : args.button === "right" ? "right-clicked" : "clicked"} ${x},${y}`;
872
- return actAndObserve(id, [{ ...args, action: "click" }], what, args);
873
- }
874
- if (name === "type_text") {
875
- const t = String(args.text ?? "");
876
- if (!t)
877
- return text(id, "nothing to type", true);
878
- return actAndObserve(id, [{ action: "type_text", text: t }], `typed ${t.length} chars`, args, 120_000);
879
- }
880
- if (name === "press_key") {
881
- const keys = String(args.keys ?? "").replace(/[^\w+]/g, "");
882
- if (!keys)
883
- return text(id, "press_key needs keys", true);
884
- return actAndObserve(id, [{ action: "press_key", keys }], `pressed ${keys}`, args);
885
- }
886
- if (name === "scroll") {
887
- const clicks = Math.min(Math.max(Math.round(Number(args.clicks) || 3), 1), 20);
888
- const direction = args.direction === "up" ? "up" : "down";
889
- return actAndObserve(id, [{ action: "scroll", direction, clicks }], `scrolled ${direction} ${clicks}`, args);
890
- }
891
- if (name === "computer_batch") {
892
- const actions = Array.isArray(args.actions) ? args.actions.slice(0, 24) : [];
893
- if (!actions.length)
894
- return text(id, "computer_batch needs a non-empty actions array", true);
895
- const summary = actions
896
- .map((a) => a.action === "click"
897
- ? `click ${Math.round(Number(a.x))},${Math.round(Number(a.y))}`
898
- : a.action === "type_text"
899
- ? `type ${String(a.text ?? "").length} chars`
900
- : a.action === "press_key"
901
- ? `key ${a.keys}`
902
- : a.action === "scroll"
903
- ? `scroll ${a.direction ?? "down"}`
904
- : `wait ${Math.min(Number(a.ms) || 500, 5000)}ms`)
905
- .join(" → ");
906
- return actAndObserve(id, actions, `ran ${actions.length} actions: ${summary}`, args, 180_000);
907
- }
908
- if (name === "computer_exec") {
909
- const command = String(args.command ?? "");
910
- if (command.length > MAX_REMOTE_COMMAND_LENGTH) {
911
- return text(id, `command is too long (maximum ${MAX_REMOTE_COMMAND_LENGTH} characters)`, true);
912
- }
913
- observations.noteAction();
914
- const out = await runOnBox(command, 120_000);
915
- const note = `exit ${out.exitCode}\n${out.stdout.slice(-6000)}${out.stderr ? `\n[stderr]\n${out.stderr.slice(-2000)}` : ""}`;
916
- if (args.observe !== true)
917
- return text(id, note);
918
- const shot = await runOnBox([ENV, GEOMETRY, ensureRemoteCuaCommand(), captureBlock()].join("; "), 60_000);
919
- return observed(id, note, await frameFrom(shot));
920
- }
921
- if (name === "wait_for") {
922
- const condition = String(args.condition ?? "").trim().toLowerCase();
923
- const timeout = Math.min(Math.max(Math.trunc(Number(args.timeout_seconds) || 60), 1), 240);
924
- let check = "";
925
- let label = "";
926
- if (condition === "http_ready") {
927
- const url = String(args.url ?? "").trim();
928
- if (!/^https?:\/\//i.test(url)) {
929
- return text(id, 'http_ready needs "url", e.g. {"condition":"http_ready","url":"http://localhost:3000/health"}.', true);
930
- }
931
- check = `curl -fsS -o /dev/null --connect-timeout 1 --max-time 1 ${shellQuote(url)}`;
932
- label = url;
933
- }
934
- else if (condition === "tcp_ready") {
935
- const portNumber = Math.trunc(Number(args.port));
936
- if (!Number.isInteger(portNumber) || portNumber < 1 || portNumber > 65535) {
937
- return text(id, 'tcp_ready needs "port" 1-65535, e.g. {"condition":"tcp_ready","port":5432}.', true);
938
- }
939
- check = `timeout --kill-after=1s 1s bash -c ${shellQuote(`echo > /dev/tcp/127.0.0.1/${portNumber}`)} 2>/dev/null`;
940
- label = `port ${portNumber}`;
941
- }
942
- else if (condition === "output_matches") {
943
- const probe = String(args.command ?? "").slice(0, 2000);
944
- const pattern = String(args.pattern ?? "").slice(0, 500);
945
- if (!probe.trim() || !pattern.trim()) {
946
- return text(id, 'output_matches needs "command" and "pattern", e.g. {"condition":"output_matches","command":"tail -1 /tmp/build.log","pattern":"done|failed"}.', true);
947
- }
948
- // A probe is rerun until it matches, so bound every individual run.
949
- // This prevents a hung command from outliving the advertised wait.
950
- check = `timeout --kill-after=1s 1s bash -c ${shellQuote(probe)} 2>&1 | grep -Eq -- ${shellQuote(pattern)}`;
951
- label = `/${pattern}/ in ${probe.slice(0, 80)}`;
952
- }
953
- else if (condition === "file_exists") {
954
- const target = String(args.path ?? "").trim();
955
- if (!target.startsWith("/")) {
956
- return text(id, 'file_exists needs an absolute "path", e.g. {"condition":"file_exists","path":"/tmp/render.done"}.', true);
957
- }
958
- check = `[ -e ${shellQuote(target)} ]`;
959
- label = target;
960
- }
961
- else {
962
- return text(id, 'wait_for needs "condition": http_ready, tcp_ready, output_matches, or file_exists.', true);
963
- }
964
- const loop = [
965
- "MET=no",
966
- "START=$SECONDS",
967
- `END=$((SECONDS+${timeout}))`,
968
- `while [ "$SECONDS" -lt "$END" ]; do if ${check}; then MET=yes; break; fi; REMAIN=$((END-SECONDS)); [ "$REMAIN" -le 0 ] && break; [ "$REMAIN" -lt 2 ] && sleep "$REMAIN" || sleep 2; done`,
969
- 'echo "WAIT_RESULT $MET ELAPSED $((SECONDS-START))"',
970
- ].join("; ");
971
- observations.noteAction();
972
- // Keep this condition-only. A person can take control during a long wait;
973
- // appending a screenshot to the same remote shell would bypass the fresh
974
- // control check and could capture credentials they typed. A follow-up
975
- // screenshot is a separate tool call and therefore re-checks the lease.
976
- const out = await runOnBox(loop, (timeout + 15) * 1000);
977
- const marker = out.stdout.match(/^WAIT_RESULT (yes|no) ELAPSED (\d+)$/m);
978
- if (!out.ok || !marker) {
979
- const detail = out.stderr.slice(0, 300) || `exit ${out.exitCode ?? "unknown"}`;
980
- return text(id, `wait_for could not check ${label}: ${detail}`, true);
981
- }
982
- const met = marker[1] === "yes";
983
- const elapsed = marker[2];
984
- const note = met
985
- ? `condition met: ${label} (~${elapsed}s)`
986
- : `timed out after ${timeout}s waiting for ${label} — inspect with computer_exec (logs, process list) before waiting again.`;
987
- return text(id, note, !met);
988
- }
989
- if (name === "open_url") {
990
- const url = String(args.url ?? "");
991
- const normalized = normalizeBrowserUrl(url);
992
- const publicUrl = safeBrowserUrl(url);
993
- if (!normalized || !publicUrl)
994
- return text(id, "only valid http(s) URLs", true);
995
- const q = shellQuote(normalized);
996
- const observe = wantsFrame(args);
997
- // launch, then poll for a browser window instead of a blind sleep —
998
- // a fast page returns in a fraction of the old fixed 3s
999
- const command = [
1000
- ENV,
1001
- GEOMETRY,
1002
- CHROME_PROFILE_SETUP,
1003
- `(google-chrome ${CHROME_DEBUG_FLAGS} ${q} || chromium ${CHROME_DEBUG_FLAGS} ${q} || chromium-browser ${CHROME_DEBUG_FLAGS} ${q} || xdg-open ${q}) >/dev/null 2>&1 &`,
1004
- 'for i in 1 2 3 4 5 6 7 8 9 10 11 12; do xdotool search --onlyvisible --class "chrom" >/dev/null 2>&1 && break; sleep 0.25; done',
1005
- ensureRemoteCuaCommand(),
1006
- observe ? captureBlock(600) : "true",
1007
- ].join("; ");
1008
- observations.noteAction();
1009
- const out = await runOnBox(command, 60_000);
1010
- const verification = await waitForNavigation(normalized, 1);
1011
- const current = verification.targets.map((target) => target.url).join(", ") || "unavailable";
1012
- const note = verification.ok
1013
- ? `opened and navigation verified: ${publicUrl}`
1014
- : `opened ${publicUrl}, but the exact destination was not verified. Current structured state: ${current}`;
1015
- if (!observe)
1016
- return text(id, note);
1017
- return observed(id, note, await frameFrom(out));
1018
- }
1019
- return text(id, `unknown tool ${name}`, true);
1020
- }
1021
- async function handle(msg) {
1022
- if (msg.method === "initialize") {
1023
- return send({
1024
- jsonrpc: "2.0",
1025
- id: msg.id,
1026
- result: {
1027
- protocolVersion: msg.params?.protocolVersion ?? "2024-11-05",
1028
- capabilities: { tools: {} },
1029
- serverInfo: { name: "openmausbot-computer", version: "3" },
1030
- },
1031
- });
1032
- }
1033
- if (msg.method === "tools/list")
1034
- return send({ jsonrpc: "2.0", id: msg.id, result: { tools: TOOLS } });
1035
- if (msg.method === "tools/call") {
1036
- try {
1037
- return await call(msg.id, msg.params?.name, msg.params?.arguments ?? {});
1038
- }
1039
- catch (e) {
1040
- const error = e instanceof Error ? e : new Error(String(e));
1041
- const timedOut = error.name === "TimeoutError" || /timed?\s*out|timeout/i.test(error.message);
1042
- return text(msg.id, timedOut
1043
- ? "computer tool timed out. The action may or may not have completed; take a screenshot to inspect the current state before retrying it."
1044
- : `computer tool failed: ${error.message}`, true);
1045
- }
1046
- }
1047
- if (String(msg.method ?? "").startsWith("notifications/"))
1048
- return;
1049
- if (msg.id != null) {
1050
- send({ jsonrpc: "2.0", id: msg.id, error: { code: -32601, message: `method not found: ${msg.method}` } });
1051
- }
1052
- }
1053
- let buf = "";
1054
- process.stdin.on("data", (chunk) => {
1055
- buf += chunk;
1056
- let nl;
1057
- while ((nl = buf.indexOf("\n")) !== -1) {
1058
- const line = buf.slice(0, nl);
1059
- buf = buf.slice(nl + 1);
1060
- if (!line.trim())
1061
- continue;
1062
- try {
1063
- void handle(JSON.parse(line));
1064
- }
1065
- catch {
1066
- /* ignore malformed lines */
1067
- }
1068
- }
1069
- });
1070
- process.stdin.on("end", () => process.exit(0));