@agent-compose/sdk 0.8.4 → 0.8.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (44) hide show
  1. package/dist/agent/agent-context.d.ts +9 -1
  2. package/dist/agent/agent-loop.d.ts +10 -1
  3. package/dist/client.d.ts +171 -33
  4. package/dist/directives.d.ts +14 -0
  5. package/dist/generated/verb-synopsis.d.ts +34 -0
  6. package/dist/index.d.ts +6 -4
  7. package/dist/index.js +1024 -39
  8. package/dist/runtimes/_cli-agent.d.ts +106 -0
  9. package/dist/runtimes/claude-code.d.ts +31 -1
  10. package/dist/runtimes/openai-desktop.d.ts +50 -0
  11. package/dist/runtimes/openai-desktop.js +1048 -57
  12. package/dist/runtimes/openai-desktop.test.d.ts +20 -0
  13. package/dist/runtimes/tool-pulse.test.d.ts +17 -0
  14. package/dist/sandbox/devbox.d.ts +5 -5
  15. package/dist/sandbox/registry.d.ts +12 -0
  16. package/dist/sandbox/sizes.d.ts +11 -5
  17. package/dist/sandbox.d.ts +1 -1
  18. package/dist/step-invocation/types.d.ts +1 -1
  19. package/dist/types/api-conversations.d.ts +85 -12
  20. package/dist/types/api-factory.d.ts +111 -1
  21. package/dist/types/conversation-stream.d.ts +22 -1
  22. package/dist/types/protocol.d.ts +118 -1
  23. package/dist/types/runtime.d.ts +71 -0
  24. package/package.json +1 -1
  25. package/src/agent/agent-context.ts +43 -9
  26. package/src/agent/agent-loop.ts +11 -5
  27. package/src/agent/desktop-open.ts +13 -1
  28. package/src/client.ts +256 -38
  29. package/src/directives.ts +21 -1
  30. package/src/generated/verb-synopsis.ts +544 -0
  31. package/src/index.ts +17 -3
  32. package/src/runtimes/_cli-agent.ts +313 -22
  33. package/src/runtimes/claude-code.ts +249 -12
  34. package/src/runtimes/openai-desktop.ts +82 -19
  35. package/src/sandbox/devbox.ts +5 -5
  36. package/src/sandbox/providers/e2b.ts +60 -16
  37. package/src/sandbox/registry.ts +19 -1
  38. package/src/sandbox/sizes.ts +11 -5
  39. package/src/sandbox.ts +1 -0
  40. package/src/types/api-conversations.ts +65 -13
  41. package/src/types/api-factory.ts +121 -1
  42. package/src/types/conversation-stream.ts +24 -1
  43. package/src/types/protocol.ts +113 -1
  44. package/src/types/runtime.ts +63 -0
@@ -34,7 +34,10 @@
34
34
  * token-metering gateway's Anthropic passthrough (ADR-0039).
35
35
  */
36
36
 
37
- import type { AgentMessage, AgentMessageTaskNotification } from "../index.js";
37
+ import type {
38
+ AgentMessage, AgentMessageCompaction, AgentMessageTaskNotification, AgentMessageTaskProgress,
39
+ WorkflowProgressEntry,
40
+ } from "../index.js";
38
41
  import { createCliAgentRuntime, shellQuote, type CliAgentSpec, type CliReasoningEffort } from "./_cli-agent.js";
39
42
  import { formatError } from "../utils/errors.js";
40
43
 
@@ -171,6 +174,37 @@ export function parseTaskNotifications(
171
174
  return out;
172
175
  }
173
176
 
177
+ /** Clamp for a steer delivered into a child's thread — same ceiling as a
178
+ * notification report: plenty for any real addendum, bounded against a
179
+ * runaway blob. */
180
+ const SUBAGENT_USER_MESSAGE_MAX = 20_000;
181
+
182
+ /** One user-role TEXT blob from the stream, mapped with sidechain awareness.
183
+ * Task-notification blocks parse into structure wherever they appear (their
184
+ * arrival side is resume-dependent — see the transport notes above). What
185
+ * remains is then split by attribution: TOP-LEVEL text (no parent id) stays
186
+ * unmapped exactly as before — prompt echoes and system reminders are not
187
+ * agent output. SIDECHAIN text (parent id present) is a message landing in
188
+ * a child subagent's thread — a delivered SendMessage steer — and forwards
189
+ * as `subagent_user_message`, except harness plumbing (`<system-reminder>`
190
+ * wrappers the CLI injects into child threads), which no renderer should
191
+ * see. Exported for tests. */
192
+ export function sidechainAwareUserText(
193
+ text: string, parentToolUseId: string | undefined, timestamp: string,
194
+ ): AgentMessage[] {
195
+ const notifications = parseTaskNotifications(text, timestamp);
196
+ if (notifications.length > 0) return notifications;
197
+ if (!parentToolUseId) return [];
198
+ const trimmed = text.trim();
199
+ if (trimmed.length === 0 || trimmed.startsWith("<system-reminder>")) return [];
200
+ return [{
201
+ type: "subagent_user_message",
202
+ text: clip(trimmed, SUBAGENT_USER_MESSAGE_MAX),
203
+ parentToolUseId,
204
+ timestamp,
205
+ }];
206
+ }
207
+
174
208
  /** One `system`/`task_notification` stream-json event mapped onto the same
175
209
  * structured message the XML parse produces, or null when the event names
176
210
  * no task id. Pure and tolerant over untrusted harness JSON: absent fields
@@ -206,6 +240,175 @@ export function parseSystemTaskNotification(
206
240
  };
207
241
  }
208
242
 
243
+ // ── Task progress (live background-workflow evidence) ───────────────────────
244
+ //
245
+ // While a harness Workflow (the CLI's in-harness dynamic-workflow tool)
246
+ // runs in the background, stream-json emits `system`/`task_progress`
247
+ // events carrying the workflow's CUMULATIVE `workflow_progress` array —
248
+ // verified live on 2.1.241 (`-p --output-format stream-json`): the array
249
+ // re-sends on every agent state change immediately and on a ~10s heartbeat
250
+ // otherwise, each `workflow_agent` entry updated in place (label,
251
+ // phase, model, state, tokens, toolCalls, durationMs). This is the ONLY
252
+ // per-agent evidence the wire ever carries — the completion notification
253
+ // reports aggregates only — so dropping it leaves workflow cards blind.
254
+ // Preview text (promptPreview / resultPreview) is internal plumbing and is
255
+ // not forwarded, same rule as task notifications.
256
+
257
+ const PROGRESS_PHASES_MAX = 32;
258
+ const PROGRESS_AGENTS_MAX = 128;
259
+ const PROGRESS_LABEL_MAX = 200;
260
+ const PROGRESS_ERROR_MAX = 500;
261
+
262
+ const AGENT_STATES: ReadonlySet<string> = new Set(["start", "progress", "done", "error"]);
263
+
264
+ /** One `system`/`task_progress` event mapped onto the structured progress
265
+ * message, or null only when it names no task. Workflow tasks carry the
266
+ * cumulative `workflow_progress` entries; a plain background Agent-task
267
+ * heartbeat (no workflow entries — verified live on 2.1.241, ~10s cadence
268
+ * while the task runs) forwards with `workflow: []` so the platform can
269
+ * DECLARE the running task as session background work (the delegation-
270
+ * survives-the-human contract: parks, user interrupts, and supersedes all
271
+ * read the declared-children registry). Completion evidence still rides
272
+ * `task_notification`. Pure and tolerant over untrusted harness JSON:
273
+ * malformed entries are skipped, everything is clamped (phases keep the
274
+ * FIRST 32 — they are seeded up front; agents keep the LAST 128 — late
275
+ * spawns matter more than a long-settled prefix). Exported for tests. */
276
+ export function parseSystemTaskProgress(
277
+ p: Record<string, unknown>, timestamp: string,
278
+ ): AgentMessageTaskProgress | null {
279
+ const taskId = typeof p.task_id === "string" && p.task_id.trim().length > 0 ? p.task_id.trim() : null;
280
+ if (!taskId) return null;
281
+ const raw = Array.isArray(p.workflow_progress) ? p.workflow_progress : [];
282
+ const nonneg = (v: unknown): number | undefined =>
283
+ typeof v === "number" && Number.isFinite(v) && v >= 0 ? Math.floor(v) : undefined;
284
+ const str = (v: unknown, max: number): string | undefined =>
285
+ typeof v === "string" && v.length > 0 ? clip(v, max) : undefined;
286
+ const phases: WorkflowProgressEntry[] = [];
287
+ const agents: WorkflowProgressEntry[] = [];
288
+ for (const e of raw) {
289
+ if (typeof e !== "object" || e === null) continue;
290
+ const o = e as Record<string, unknown>;
291
+ const index = nonneg(o.index);
292
+ if (index === undefined) continue;
293
+ if (o.type === "workflow_phase") {
294
+ const title = str(o.title, PROGRESS_LABEL_MAX);
295
+ if (title && phases.length < PROGRESS_PHASES_MAX) phases.push({ kind: "phase", index, title });
296
+ continue;
297
+ }
298
+ if (o.type !== "workflow_agent") continue;
299
+ const label = str(o.label, PROGRESS_LABEL_MAX);
300
+ const state = typeof o.state === "string" && AGENT_STATES.has(o.state) ? o.state as "start" | "progress" | "done" | "error" : null;
301
+ if (!label || !state) continue;
302
+ agents.push({
303
+ kind: "agent", index, label, state,
304
+ ...(nonneg(o.phaseIndex) !== undefined ? { phaseIndex: nonneg(o.phaseIndex) } : {}),
305
+ ...(str(o.phaseTitle, PROGRESS_LABEL_MAX) ? { phaseTitle: str(o.phaseTitle, PROGRESS_LABEL_MAX) } : {}),
306
+ ...(str(o.model, 100) ? { model: str(o.model, 100) } : {}),
307
+ ...(str(o.agentId, 128) ? { agentId: str(o.agentId, 128) } : {}),
308
+ ...(nonneg(o.tokens) !== undefined ? { tokens: nonneg(o.tokens) } : {}),
309
+ ...(nonneg(o.toolCalls) !== undefined ? { toolCalls: nonneg(o.toolCalls) } : {}),
310
+ ...(nonneg(o.durationMs) !== undefined ? { durationMs: nonneg(o.durationMs) } : {}),
311
+ ...(nonneg(o.startedAt) !== undefined ? { startedAt: nonneg(o.startedAt) } : {}),
312
+ ...(state === "error" && str(o.error, PROGRESS_ERROR_MAX) ? { error: str(o.error, PROGRESS_ERROR_MAX) } : {}),
313
+ ...(o.cached === true ? { cached: true as const } : {}),
314
+ ...(o.skipped === true ? { skipped: true as const } : {}),
315
+ });
316
+ }
317
+ // Empty = a plain Agent-task heartbeat, forwarded as liveness evidence
318
+ // (see the header) — the workflow delta folder ignores it (no entries),
319
+ // and only the background-work declare consumes it.
320
+ const workflow = [...phases, ...agents.slice(-PROGRESS_AGENTS_MAX)];
321
+ const toolUseId = typeof p.tool_use_id === "string" && p.tool_use_id.length > 0 ? p.tool_use_id : null;
322
+ const u = (typeof p.usage === "object" && p.usage !== null ? p.usage : {}) as Record<string, unknown>;
323
+ const tokens = nonneg(u.total_tokens);
324
+ const toolUses = nonneg(u.tool_uses);
325
+ const durationMs = nonneg(u.duration_ms);
326
+ return {
327
+ type: "task_progress",
328
+ taskId: clip(taskId, 128),
329
+ ...(toolUseId ? { toolUseId: clip(toolUseId, 128) } : {}),
330
+ ...(tokens !== undefined || toolUses !== undefined || durationMs !== undefined
331
+ ? { usage: {
332
+ ...(tokens !== undefined ? { tokens } : {}),
333
+ ...(toolUses !== undefined ? { toolUses } : {}),
334
+ ...(durationMs !== undefined ? { durationMs } : {}),
335
+ } }
336
+ : {}),
337
+ workflow,
338
+ timestamp,
339
+ };
340
+ }
341
+
342
+ // ── Context compaction (the harness summarizing its own conversation) ───────
343
+ //
344
+ // Captured live against 2.1.212 (the baked E2B pin) and 2.1.241 — identical
345
+ // wire shapes on both, `/compact` and forced-auto alike:
346
+ //
347
+ // {"type":"system","subtype":"status","status":"compacting"}
348
+ // {"type":"system","subtype":"status","status":null,
349
+ // "compact_result":"success"|"failed"[,"compact_error":"…"]}
350
+ // {"type":"system","subtype":"init",…} (fresh init, success only)
351
+ // {"type":"system","subtype":"compact_boundary","compact_metadata":
352
+ // {"trigger":"auto"|"manual","pre_tokens":N,"post_tokens":N,
353
+ // "cumulative_dropped_tokens":N,"duration_ms":N,"preserved_segment":…}}
354
+ //
355
+ // Followed by the continuation summary as a SYNTHETIC user message (string
356
+ // content, `isSynthetic: true`) — deliberately NOT forwarded: it quotes
357
+ // conversation content verbatim (its "All user messages" section replays
358
+ // platform notices word for word), which is exactly how the 2026-08-23
359
+ // incident re-labeled a platform system notice as a "recalled memory".
360
+ // The boundary does not replay on later resumes (verified live).
361
+ //
362
+ // A compaction spans 13-21s in the small captures and MINUTES at real
363
+ // context sizes — forwarding the lifecycle is what lets downstream render
364
+ // the silence as work instead of death (the 94%-auto-compact dead-air
365
+ // incident).
366
+
367
+ const COMPACT_ERROR_MAX = 500;
368
+
369
+ /** One `system` event of the compaction family mapped onto the structured
370
+ * lifecycle message, or null when the event is not compaction-shaped (a
371
+ * plain `status` event with neither the compacting status nor a
372
+ * compact_result stays unmapped). Pure and tolerant over untrusted harness
373
+ * JSON. Exported for tests. */
374
+ export function parseSystemCompaction(
375
+ p: Record<string, unknown>, timestamp: string,
376
+ ): AgentMessageCompaction | null {
377
+ const nonneg = (v: unknown): number | undefined =>
378
+ typeof v === "number" && Number.isFinite(v) && v >= 0 ? Math.floor(v) : undefined;
379
+ if (p.subtype === "status") {
380
+ if (p.status === "compacting") return { type: "compaction", phase: "start", timestamp };
381
+ if (typeof p.compact_result === "string") {
382
+ const failed = p.compact_result !== "success";
383
+ const error = typeof p.compact_error === "string" && p.compact_error.length > 0
384
+ ? clip(p.compact_error, COMPACT_ERROR_MAX) : undefined;
385
+ return {
386
+ type: "compaction", phase: "settled",
387
+ result: failed ? "failed" : "success",
388
+ ...(failed && error ? { error } : {}),
389
+ timestamp,
390
+ };
391
+ }
392
+ return null;
393
+ }
394
+ if (p.subtype !== "compact_boundary") return null;
395
+ const meta = (typeof p.compact_metadata === "object" && p.compact_metadata !== null
396
+ ? p.compact_metadata : {}) as Record<string, unknown>;
397
+ const preTokens = nonneg(meta.pre_tokens);
398
+ const postTokens = nonneg(meta.post_tokens);
399
+ const droppedTokens = nonneg(meta.cumulative_dropped_tokens);
400
+ const durationMs = nonneg(meta.duration_ms);
401
+ return {
402
+ type: "compaction", phase: "boundary",
403
+ trigger: meta.trigger === "manual" ? "manual" : "auto",
404
+ ...(preTokens !== undefined ? { preTokens } : {}),
405
+ ...(postTokens !== undefined ? { postTokens } : {}),
406
+ ...(droppedTokens !== undefined ? { droppedTokens } : {}),
407
+ ...(durationMs !== undefined ? { durationMs } : {}),
408
+ timestamp,
409
+ };
410
+ }
411
+
209
412
  /** Claude Code's real reasoning knob is its own `--effort <level>` flag
210
413
  * (low|medium|high|xhigh|max — verified against `claude -p --help`). The
211
414
  * CliReasoningEffort union IS the CLI's vocabulary, so the level rides the
@@ -247,6 +450,14 @@ export const claudeCodeSpec: CliAgentSpec = {
247
450
  streamInput: {
248
451
  promptLine: (prompt) => JSON.stringify({ type: "user", message: { role: "user", content: prompt } }),
249
452
  messageLine: (text) => JSON.stringify({ type: "user", message: { role: "user", content: text } }),
453
+ // The ESC equivalent (verified live against claude 2.1.236): the CLI's
454
+ // control layer processes this OUT OF BAND — a 120s foreground Bash call
455
+ // aborted 6s in, the control_response answered instantly, the run ended
456
+ // ~100ms later with `result: error_during_execution`, and `--resume` on
457
+ // the same session id kept the full turn context.
458
+ interruptLine: (requestId) => JSON.stringify({
459
+ type: "control_request", request_id: requestId, request: { subtype: "interrupt" },
460
+ }),
250
461
  },
251
462
  buildCommand: ({ promptPath, sessionId, model, cwd, effort, streamInput }) => {
252
463
  const flags = [
@@ -285,9 +496,10 @@ export const claudeCodeSpec: CliAgentSpec = {
285
496
  // tool_use id; null at top level). Forwarded on tool_use/tool_result so
286
497
  // renderers can nest child activity under the spawning call instead of
287
498
  // flattening it into the parent transcript unattributed.
288
- const parent = typeof p.parent_tool_use_id === "string" && p.parent_tool_use_id.length > 0
289
- ? { parentToolUseId: p.parent_tool_use_id }
290
- : {};
499
+ const parentId = typeof p.parent_tool_use_id === "string" && p.parent_tool_use_id.length > 0
500
+ ? p.parent_tool_use_id
501
+ : undefined;
502
+ const parent = parentId ? { parentToolUseId: parentId } : {};
291
503
  switch (p.type) {
292
504
  // Assistant API message: content blocks → text / thinking / tool_use.
293
505
  case "assistant": {
@@ -321,12 +533,25 @@ export const claudeCodeSpec: CliAgentSpec = {
321
533
  // User API message: the CLI echoes tool results back as user content,
322
534
  // and injects `<task-notification>` blocks (background-task completion
323
535
  // evidence) as user TEXT — parsed into structure, never dropped and
324
- // never forwarded raw. Other user text (the echo of the prompt, system
325
- // reminders) stays unmapped: it is not agent output.
536
+ // never forwarded raw. TOP-LEVEL user text (the echo of the prompt,
537
+ // system reminders) stays unmapped: it is not agent output. SIDECHAIN
538
+ // user text (parent_tool_use_id present) is a message landing in a
539
+ // CHILD's thread — the delivered form of a SendMessage steer to a
540
+ // running subagent ("queued for delivery at its next tool round") —
541
+ // and forwards as `subagent_user_message` so the steer renders inside
542
+ // the child's mini-session instead of vanishing (task #97; before
543
+ // this, a queued steer was visible only as the parent's opaque
544
+ // SendMessage tool call).
326
545
  case "user": {
327
546
  const message = p.message as { content?: ClaudeContentBlock[] | string } | undefined;
547
+ // HARNESS-synthesized user messages (`isSynthetic` — e.g. a
548
+ // post-compaction continuation summary, including a CHILD's own)
549
+ // are never a delivered steer: suppress the steer arm, keep the
550
+ // notification parse. Structural marker only, same doctrine as the
551
+ // `<synthetic>` model stamp above.
552
+ const steerParent = p.isSynthetic === true ? undefined : parentId;
328
553
  if (typeof message?.content === "string") {
329
- return parseTaskNotifications(message.content, ts);
554
+ return sidechainAwareUserText(message.content, steerParent, ts);
330
555
  }
331
556
  const blocks = Array.isArray(message?.content) ? message.content : [];
332
557
  return blocks.flatMap((b): AgentMessage[] => {
@@ -337,7 +562,7 @@ export const claudeCodeSpec: CliAgentSpec = {
337
562
  }];
338
563
  }
339
564
  if (b.type === "text" && typeof b.text === "string") {
340
- return parseTaskNotifications(b.text, ts);
565
+ return sidechainAwareUserText(b.text, steerParent, ts);
341
566
  }
342
567
  return [];
343
568
  });
@@ -410,11 +635,23 @@ export const claudeCodeSpec: CliAgentSpec = {
410
635
  // System events: init carries the session id (extractSessionId), and
411
636
  // the background-task lane rides here too — `task_notification` is
412
637
  // the completion evidence stream-json actually emits (see the section
413
- // header above), so it maps to the structured message BOTH turn
414
- // engines persist. Everything else under system (task_started,
415
- // task_updated, background_tasks_changed, thinking_tokens) is
416
- // lifecycle noise here.
638
+ // header above), and `task_progress` the LIVE background-task feed
639
+ // (parseSystemTaskProgress): per-agent workflow entries for Workflow
640
+ // tasks, `workflow: []` heartbeats for plain background Agent tasks —
641
+ // both forward so the platform can declare the running task as
642
+ // session background work. Compaction rides here as well (the status
643
+ // pair + compact_boundary — parseSystemCompaction). Everything else
644
+ // under system (task_started, task_updated, background_tasks_changed,
645
+ // thinking_tokens) is lifecycle noise here.
417
646
  case "system": {
647
+ if (p.subtype === "task_progress") {
648
+ const progress = parseSystemTaskProgress(p, ts);
649
+ return progress ? [progress] : [];
650
+ }
651
+ if (p.subtype === "status" || p.subtype === "compact_boundary") {
652
+ const compaction = parseSystemCompaction(p, ts);
653
+ return compaction ? [compaction] : [];
654
+ }
418
655
  if (p.subtype !== "task_notification") return [];
419
656
  const notification = parseSystemTaskNotification(p, ts);
420
657
  return notification ? [notification] : [];
@@ -3,6 +3,17 @@
3
3
  *
4
4
  * Works against DesktopSandboxProvider — no E2B SDK dependency here.
5
5
  * The action→screenshot loop runs until the model emits text, then yields it.
6
+ *
7
+ * COORDINATE SPACES (incident note). Every screenshot is downscaled to the
8
+ * model's DISPLAY geometry (1024×720) before it is sent, so the model emits
9
+ * coordinates in the SCALED screenshot's space — never the real desktop's.
10
+ * An earlier revision executed those coordinates against the real desktop
11
+ * using hardcoded real-geometry constants, so on any sandbox whose actual
12
+ * geometry differed, every click landed in the wrong place. The provider has
13
+ * no geometry call, so the real geometry is read from each screenshot buffer
14
+ * itself (sharp metadata), and the resulting per-capture `CaptureScale` maps
15
+ * the model's coordinates back onto the exact frame it saw. The old constants
16
+ * survive ONLY as a fallback for the metadata-missing case.
6
17
  */
7
18
 
8
19
  import OpenAI from "openai";
@@ -12,8 +23,11 @@ import { defineRuntime, formatError } from "../index.js";
12
23
 
13
24
  const DISPLAY_WIDTH = 1024;
14
25
  const DISPLAY_HEIGHT = 720;
15
- const DESKTOP_WIDTH = 1280;
16
- const DESKTOP_HEIGHT = 800;
26
+ /** FALLBACK real-desktop geometry — consulted ONLY when sharp cannot read a
27
+ * screenshot's dimensions. The truth is derived per capture from the raw
28
+ * screenshot buffer itself (see `captureScaledScreenshot`). */
29
+ const FALLBACK_DESKTOP_WIDTH = 1280;
30
+ const FALLBACK_DESKTOP_HEIGHT = 800;
17
31
  const MAX_TURNS = 40;
18
32
 
19
33
  export interface OpenAIDesktopRunnerOptions extends RuntimeOptions {
@@ -27,8 +41,9 @@ export interface OpenAIDesktopRunnerOptions extends RuntimeOptions {
27
41
  // so the exchange is described here instead of with SDK types.
28
42
 
29
43
  /** One model-emitted desktop action. Coordinates arrive in DISPLAY (model)
30
- * space and are rescaled to the real desktop before dispatch. */
31
- interface ComputerAction {
44
+ * space and are rescaled through the CURRENT capture's `CaptureScale`
45
+ * before dispatch. Exported so tests can type their fixtures. */
46
+ export interface ComputerAction {
32
47
  type: string;
33
48
  coordinate?: [number, number];
34
49
  startCoordinate?: [number, number];
@@ -99,12 +114,16 @@ export class OpenAIDesktopRunner implements ModelExecutionContract {
99
114
 
100
115
  yield { type: "init", sessionId: "e2b-desktop", timestamp: ts() };
101
116
 
102
- const screenshot = await captureScaledScreenshot(sandbox);
117
+ const first = await captureScaledScreenshot(sandbox);
118
+ // The scale of the LATEST frame the model has seen — its next batch of
119
+ // coordinates is in that frame's space, so every fresh capture below
120
+ // replaces this before the frame is sent back.
121
+ let scale = first.scale;
103
122
 
104
123
  const messages: InputMessage[] = [{
105
124
  role: "user",
106
125
  content: [
107
- { type: "input_image", image_url: `data:image/png;base64,${screenshot}` },
126
+ { type: "input_image", image_url: `data:image/png;base64,${first.b64}` },
108
127
  { type: "input_text", text: opts.prompt },
109
128
  ],
110
129
  }];
@@ -137,7 +156,7 @@ export class OpenAIDesktopRunner implements ModelExecutionContract {
137
156
 
138
157
  for (const call of computerCalls) {
139
158
  console.log(`${label} → ${call.action.type}`);
140
- await executeAction(sandbox, call.action).catch(
159
+ await executeAction(sandbox, call.action, scale).catch(
141
160
  (err: unknown) => console.warn(`${label} Action failed (non-fatal): ${formatError(err)}`),
142
161
  );
143
162
  }
@@ -145,10 +164,11 @@ export class OpenAIDesktopRunner implements ModelExecutionContract {
145
164
  if (computerCalls.length === 0) break;
146
165
 
147
166
  const next = await captureScaledScreenshot(sandbox);
167
+ scale = next.scale;
148
168
  messages.push({ role: "user", content: computerCalls.map((call): InputPart => ({
149
169
  type: "computer_call_output",
150
170
  call_id: call.call_id,
151
- output: { type: "input_image", image_url: `data:image/png;base64,${next}` },
171
+ output: { type: "input_image", image_url: `data:image/png;base64,${next.b64}` },
152
172
  })) });
153
173
  }
154
174
 
@@ -160,21 +180,58 @@ export class OpenAIDesktopRunner implements ModelExecutionContract {
160
180
  // Helpers — work against DesktopSandboxProvider interface
161
181
  // ---------------------------------------------------------------------------
162
182
 
163
- async function captureScaledScreenshot(sandbox: DesktopSandboxProvider): Promise<string> {
183
+ /** Real-desktop pixels per model-DISPLAY pixel, derived from ONE capture.
184
+ * Coordinates the model emits against that capture are multiplied by these
185
+ * factors (and rounded) before touching the provider. */
186
+ export interface CaptureScale {
187
+ scaleX: number;
188
+ scaleY: number;
189
+ }
190
+
191
+ /** One captured frame: the model-facing image plus the scale that maps the
192
+ * model's coordinates back onto this frame's real desktop. */
193
+ export interface ScaledCapture {
194
+ /** Base64 PNG resized to DISPLAY_WIDTH×DISPLAY_HEIGHT (fit "fill"). */
195
+ b64: string;
196
+ scale: CaptureScale;
197
+ }
198
+
199
+ /** Derive a capture's scale from its real pixel dimensions. The constants
200
+ * are a FALLBACK only — used when sharp cannot read the frame's metadata;
201
+ * a real dimension always wins. */
202
+ export function scaleFromDimensions(width: number | undefined, height: number | undefined): CaptureScale {
203
+ return {
204
+ scaleX: (width ?? FALLBACK_DESKTOP_WIDTH) / DISPLAY_WIDTH,
205
+ scaleY: (height ?? FALLBACK_DESKTOP_HEIGHT) / DISPLAY_HEIGHT,
206
+ };
207
+ }
208
+
209
+ /** Capture one frame: read the REAL geometry from the raw screenshot's own
210
+ * metadata, downscale to the model's DISPLAY geometry, and return both the
211
+ * image and the scale that maps model coordinates back onto this frame. */
212
+ export async function captureScaledScreenshot(sandbox: DesktopSandboxProvider): Promise<ScaledCapture> {
164
213
  const raw = await sandbox.screenshot();
165
- const scaled = await sharp(raw)
214
+ const image = sharp(raw);
215
+ const { width, height } = await image.metadata();
216
+ const scaled = await image
166
217
  .resize(DISPLAY_WIDTH, DISPLAY_HEIGHT, { kernel: "lanczos3", fit: "fill" })
167
218
  .png()
168
219
  .toBuffer();
169
- return scaled.toString("base64");
220
+ return { b64: scaled.toString("base64"), scale: scaleFromDimensions(width, height) };
170
221
  }
171
222
 
172
- function scaleX(x: number): number { return Math.round(x * DESKTOP_WIDTH / DISPLAY_WIDTH); }
173
- function scaleY(y: number): number { return Math.round(y * DESKTOP_HEIGHT / DISPLAY_HEIGHT); }
174
-
175
- async function executeAction(sandbox: DesktopSandboxProvider, action: ComputerAction): Promise<void> {
223
+ /** Dispatch one model action against the provider, mapping every coordinate
224
+ * through the CURRENT capture's scale. Non-coordinate payloads — scroll
225
+ * deltas (ticks), typed text, key chords — pass through untouched. */
226
+ export async function executeAction(
227
+ sandbox: DesktopSandboxProvider,
228
+ action: ComputerAction,
229
+ scale: CaptureScale,
230
+ ): Promise<void> {
231
+ const mapX = (x: number) => Math.round(x * scale.scaleX);
232
+ const mapY = (y: number) => Math.round(y * scale.scaleY);
176
233
  const [x, y] = action.coordinate
177
- ? [scaleX(action.coordinate[0]), scaleY(action.coordinate[1])]
234
+ ? [mapX(action.coordinate[0]), mapY(action.coordinate[1])]
178
235
  : [0, 0];
179
236
 
180
237
  switch (action.type) {
@@ -185,12 +242,18 @@ async function executeAction(sandbox: DesktopSandboxProvider, action: ComputerAc
185
242
  case "move": await sandbox.moveMouse(x, y); break;
186
243
  case "type": await sandbox.write(action.text ?? ""); break;
187
244
  case "key": await sandbox.press(action.key ?? ""); break;
188
- case "scroll": await sandbox.scroll(action.direction === "up" ? "up" : "down", action.ticks ?? 3); break;
245
+ case "scroll":
246
+ // The scroll POSITION is a coordinate — mapped (the provider has no
247
+ // positional scroll, so position lands via moveMouse). The scroll
248
+ // DELTA (ticks) is not a coordinate — untouched.
249
+ if (action.coordinate) await sandbox.moveMouse(x, y);
250
+ await sandbox.scroll(action.direction === "up" ? "up" : "down", action.ticks ?? 3);
251
+ break;
189
252
  case "drag":
190
253
  if (action.startCoordinate && action.endCoordinate) {
191
254
  await sandbox.drag(
192
- [scaleX(action.startCoordinate[0]), scaleY(action.startCoordinate[1])],
193
- [scaleX(action.endCoordinate[0]), scaleY(action.endCoordinate[1])],
255
+ [mapX(action.startCoordinate[0]), mapY(action.startCoordinate[1])],
256
+ [mapX(action.endCoordinate[0]), mapY(action.endCoordinate[1])],
194
257
  );
195
258
  }
196
259
  break;
@@ -5,13 +5,13 @@
5
5
  * This module is the SINGLE SOURCE for the machine's template identity: the
6
6
  * recipe lives in `infra/e2b-template/devbox.ts`, `infra/e2b-template/build.ts`
7
7
  * bakes it under the alias below, and the server imports the alias from here to
8
- * provision a machine (exactly how `e2bBaseTemplate` reaches the E2B provider —
9
- * a stable ALIAS, never a snapshot id, so the same string resolves in whichever
10
- * E2B account a deployment uses).
8
+ * provision a machine (exactly how `e2bAgentEnvTemplate` reaches the E2B
9
+ * provider — a stable ALIAS, never a snapshot id, so the same string resolves
10
+ * in whichever E2B account a deployment uses).
11
11
  *
12
12
  * Interactive only. Workflow RUNS never execute on a devbox (ADR-0038 §2.6):
13
- * runs boot the clean per-size `agent-compose-base-<size>` / `agent-env-<size>`
14
- * templates so reproducibility never inherits hand-configured drift.
13
+ * runs boot the clean per-size `agent-env-<size>` template (the same image
14
+ * sessions boot) so reproducibility never inherits hand-configured drift.
15
15
  */
16
16
 
17
17
  /** Stable E2B template ALIAS for the per-member desktop machine. Baked by
@@ -5,7 +5,7 @@
5
5
  * registry entry.
6
6
  */
7
7
 
8
- import { Sandbox } from "e2b";
8
+ import { Sandbox, Template } from "e2b";
9
9
  import type { CommandHandle } from "e2b";
10
10
  import type { Sandbox as Desktop } from "@e2b/desktop";
11
11
  import pRetry from "p-retry";
@@ -13,7 +13,7 @@ import type {
13
13
  SandboxProvider, SandboxCommandRunOptions, SandboxBackgroundProcess, SandboxPtyHandle,
14
14
  } from "../../types/sandbox.js";
15
15
  import { toE2bNetwork } from "../network-policy.js";
16
- import { DEFAULT_SANDBOX_SIZE, e2bBaseTemplate, isE2bSupportedSize } from "../sizes.js";
16
+ import { DEFAULT_SANDBOX_SIZE, e2bAgentEnvTemplate, isE2bSupportedSize } from "../sizes.js";
17
17
  import { AGENT_COMPOSE_TAG } from "../provider-def.js";
18
18
  import type { OwnedSandbox, SandboxProviderDef } from "../provider-def.js";
19
19
 
@@ -297,18 +297,24 @@ export const e2bProviderDef: SandboxProviderDef = {
297
297
  // E2B has no create-time resource knob (specs are baked into the
298
298
  // template/snapshot), so honouring `size` on E2B = picking a PRE-SIZED
299
299
  // template, not passing the field through. We resolve the boot template
300
- // from `size` below (`e2bBaseTemplate(size)`) when the caller gave no
300
+ // from `size` below (`e2bAgentEnvTemplate(size)`) when the caller gave no
301
301
  // explicit template/bootFrom; the field itself is never forwarded to E2B.
302
302
  create: async ({ template, timeoutMs, networkPolicy, size, ...rest }) => {
303
303
  // `template` is an E2B template id/alias, a snapshot id (a valid create
304
304
  // source that persists beyond its origin sandbox — bootFrom parity), or
305
- // absent. When absent we pick the SIZE-MATCHED platform base alias
306
- // (`agent-compose-base-<size>`) so `resources.size` gives the same
307
- // machine spec on E2B as on Vercel — the cross-provider parity this whole
308
- // change exists for. Only when no size is resolvable at all do we fall
309
- // back to E2B_DEFAULT_TEMPLATE if set (a prebuilt base — e.g. one with the claude CLI
310
- // + chromium baked in and more RAM than the stock 482MB base), the
311
- // per-deployment analogue of Vercel's node24; else E2B's stock base.
305
+ // absent. When absent we pick the SIZE-MATCHED platform agent-env alias
306
+ // (`agent-env-<size>`) — the SAME image every cloud session boots
307
+ // (server/src/sandbox/persistent.ts resolves the identical
308
+ // `e2bAgentEnvTemplate(size)`), so a template-less workflow run gets the
309
+ // full session toolchain (claude runtime, dev toolbelt, agentc, fsgw
310
+ // client) instead of a thinner image. One seam, both lanes: the
311
+ // session/run divergence that caused fleet-wide `exit 127`s (missing
312
+ // harness CLIs, fixed by on-demand install in v0.10.73) is structurally
313
+ // gone — there is no separate run-lane template to drift. The per-size
314
+ // aliasing also keeps `resources.size` giving the same machine spec on
315
+ // E2B as on Vercel (cross-provider parity). Only when no size is
316
+ // resolvable at all do we fall back to E2B_DEFAULT_TEMPLATE if set (the
317
+ // per-deployment analogue of Vercel's node24); else E2B's stock base.
312
318
  // Self-provisioning runtimes (claude/codex/amp via bootFrom:"reuse") install
313
319
  // their CLI on the base and cache it in the captured snapshot, so a
314
320
  // template-less first run boots, installs, snapshots. Clamp the lifetime to
@@ -346,18 +352,19 @@ export const e2bProviderDef: SandboxProviderDef = {
346
352
  };
347
353
  // Boot template resolution, in priority order:
348
354
  // 1. explicit `template`/bootFrom (a pinned snapshot or alias),
349
- // 2. else the SIZE-MATCHED base alias `agent-compose-base-<size>`
350
- // (`size` resolved to DEFAULT_SANDBOX_SIZE when unset) — this is the
351
- // cross-provider parity path,
355
+ // 2. else the SIZE-MATCHED agent-env alias `agent-env-<size>` — the
356
+ // session-identical image (`size` resolved to DEFAULT_SANDBOX_SIZE
357
+ // when unset). NEVER the legacy `agent-compose-base-<size>` — that
358
+ // family remains a valid EXPLICIT bootFrom target only,
352
359
  // 3. else E2B_DEFAULT_TEMPLATE as an ultimate per-deployment fallback,
353
360
  // 4. else E2B's stock base.
354
- // `32vcpu-64gb` has no base-<size> template (Pro caps ~8 vCPU); the
361
+ // `32vcpu-64gb` has no per-size template (Pro caps ~8 vCPU); the
355
362
  // register + invoke guards reject it before a run reaches here, so we
356
- // never synthesize a non-existent `agent-compose-base-32vcpu-64gb` alias.
363
+ // never synthesize a non-existent `agent-env-32vcpu-64gb` alias.
357
364
  const resolvedSize = size ?? DEFAULT_SANDBOX_SIZE;
358
365
  const tmpl =
359
366
  template ??
360
- (isE2bSupportedSize(resolvedSize) ? e2bBaseTemplate(resolvedSize) : undefined) ??
367
+ (isE2bSupportedSize(resolvedSize) ? e2bAgentEnvTemplate(resolvedSize) : undefined) ??
361
368
  process.env.E2B_DEFAULT_TEMPLATE;
362
369
  return makeE2bSandboxProvider(
363
370
  await (tmpl ? Sandbox.create(tmpl, sandboxOpts) : Sandbox.create(sandboxOpts)),
@@ -383,4 +390,41 @@ export const e2bProviderDef: SandboxProviderDef = {
383
390
  // E2B snapshots are team-scoped; delete by id. Best-effort like Vercel's.
384
391
  await Sandbox.deleteSnapshot(snapshotId, { apiKey: env.E2B_API_KEY });
385
392
  },
393
+ snapshotExists: async (snapshotId, env) => {
394
+ // Existence probe for a CREATE SOURCE (the `snapshotResolves` contract:
395
+ // `true` = resolves, `false` = the provider definitively says it does not
396
+ // exist, anything indeterminate PROPAGATES — never reported as missing).
397
+ //
398
+ // e2b 2.30.5 has no GET-snapshot-by-id, so this composes the two
399
+ // documented lookups, cheapest first:
400
+ // 1. `Template.exists` — ONE round-trip to the template-existence
401
+ // endpoint (`GET /templates/aliases/{alias}`), with DOCUMENTED
402
+ // not-found semantics: 404 → false, 403 → exists but owned by
403
+ // another team → true (the SDK's own mapping). Snapshots are
404
+ // templates provider-side, and this probe also resolves the
405
+ // `agent-env-*` template ALIASES that E2B-pinned default templates
406
+ // register as their bootFrom — boot-time platform validation
407
+ // (validatePlatformSnapshots) probes those through this same seam,
408
+ // so a snapshots-only lookup would falsely alert "unresolvable" on
409
+ // every alias.
410
+ // 2. A paged scan of the team's snapshot list (`Sandbox.listSnapshots`)
411
+ // — authoritative for `createSnapshot` artifacts whatever their id
412
+ // shape, reached only when the template probe answered not-found.
413
+ // Bounded by the team's snapshot count (session rings are GC'd to a
414
+ // fixed retain depth), and this is rare-path code: machine-loss
415
+ // recovery and boot validation, never a hot loop.
416
+ // `false` therefore means BOTH documented lookups answered not-found.
417
+ // (`Sandbox.listSnapshots({ sandboxId })` — source-filtered — was
418
+ // rejected as the primary: callers hold only the snapshot id, and a
419
+ // filtered miss would still need the full scan before "absent" is
420
+ // honest.) Transport/auth faults from either call throw — indeterminate.
421
+ const opts = { apiKey: env.E2B_API_KEY };
422
+ if (await Template.exists(snapshotId, opts)) return true;
423
+ const paginator = Sandbox.listSnapshots(opts);
424
+ while (paginator.hasNext) {
425
+ const items = await pRetry(() => paginator.nextItems(), { retries: 3, minTimeout: 500, factor: 2 });
426
+ if (items.some((s) => s.snapshotId === snapshotId || s.names.includes(snapshotId))) return true;
427
+ }
428
+ return false;
429
+ },
386
430
  };