@agent-compose/sdk 0.8.3 → 0.8.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (44) hide show
  1. package/dist/agent/agent-context.d.ts +9 -1
  2. package/dist/agent/agent-loop.d.ts +14 -1
  3. package/dist/client.d.ts +171 -33
  4. package/dist/directives.d.ts +14 -0
  5. package/dist/generated/verb-synopsis.d.ts +34 -0
  6. package/dist/index.d.ts +7 -5
  7. package/dist/index.js +1043 -41
  8. package/dist/runtimes/_cli-agent.d.ts +118 -0
  9. package/dist/runtimes/claude-code.d.ts +31 -1
  10. package/dist/runtimes/openai-desktop.d.ts +50 -0
  11. package/dist/runtimes/openai-desktop.js +1065 -59
  12. package/dist/runtimes/openai-desktop.test.d.ts +20 -0
  13. package/dist/runtimes/tool-pulse.test.d.ts +17 -0
  14. package/dist/sandbox/devbox.d.ts +5 -5
  15. package/dist/sandbox/registry.d.ts +12 -0
  16. package/dist/sandbox/sizes.d.ts +11 -5
  17. package/dist/sandbox.d.ts +2 -2
  18. package/dist/step-invocation/types.d.ts +1 -1
  19. package/dist/types/api-conversations.d.ts +85 -12
  20. package/dist/types/api-factory.d.ts +111 -1
  21. package/dist/types/conversation-stream.d.ts +22 -1
  22. package/dist/types/protocol.d.ts +130 -1
  23. package/dist/types/runtime.d.ts +71 -0
  24. package/package.json +1 -1
  25. package/src/agent/agent-context.ts +43 -9
  26. package/src/agent/agent-loop.ts +14 -3
  27. package/src/agent/desktop-open.ts +13 -1
  28. package/src/client.ts +256 -38
  29. package/src/directives.ts +21 -1
  30. package/src/generated/verb-synopsis.ts +544 -0
  31. package/src/index.ts +20 -5
  32. package/src/runtimes/_cli-agent.ts +333 -22
  33. package/src/runtimes/claude-code.ts +260 -14
  34. package/src/runtimes/openai-desktop.ts +82 -19
  35. package/src/sandbox/devbox.ts +5 -5
  36. package/src/sandbox/providers/e2b.ts +89 -17
  37. package/src/sandbox/registry.ts +19 -1
  38. package/src/sandbox/sizes.ts +11 -5
  39. package/src/sandbox.ts +2 -1
  40. package/src/types/api-conversations.ts +65 -13
  41. package/src/types/api-factory.ts +121 -1
  42. package/src/types/conversation-stream.ts +24 -1
  43. package/src/types/protocol.ts +127 -1
  44. package/src/types/runtime.ts +63 -0
@@ -34,7 +34,10 @@
34
34
  * token-metering gateway's Anthropic passthrough (ADR-0039).
35
35
  */
36
36
 
37
- import type { AgentMessage, AgentMessageTaskNotification } from "../index.js";
37
+ import type {
38
+ AgentMessage, AgentMessageCompaction, AgentMessageTaskNotification, AgentMessageTaskProgress,
39
+ WorkflowProgressEntry,
40
+ } from "../index.js";
38
41
  import { createCliAgentRuntime, shellQuote, type CliAgentSpec, type CliReasoningEffort } from "./_cli-agent.js";
39
42
  import { formatError } from "../utils/errors.js";
40
43
 
@@ -171,6 +174,37 @@ export function parseTaskNotifications(
171
174
  return out;
172
175
  }
173
176
 
177
+ /** Clamp for a steer delivered into a child's thread — same ceiling as a
178
+ * notification report: plenty for any real addendum, bounded against a
179
+ * runaway blob. */
180
+ const SUBAGENT_USER_MESSAGE_MAX = 20_000;
181
+
182
+ /** One user-role TEXT blob from the stream, mapped with sidechain awareness.
183
+ * Task-notification blocks parse into structure wherever they appear (their
184
+ * arrival side is resume-dependent — see the transport notes above). What
185
+ * remains is then split by attribution: TOP-LEVEL text (no parent id) stays
186
+ * unmapped exactly as before — prompt echoes and system reminders are not
187
+ * agent output. SIDECHAIN text (parent id present) is a message landing in
188
+ * a child subagent's thread — a delivered SendMessage steer — and forwards
189
+ * as `subagent_user_message`, except harness plumbing (`<system-reminder>`
190
+ * wrappers the CLI injects into child threads), which no renderer should
191
+ * see. Exported for tests. */
192
+ export function sidechainAwareUserText(
193
+ text: string, parentToolUseId: string | undefined, timestamp: string,
194
+ ): AgentMessage[] {
195
+ const notifications = parseTaskNotifications(text, timestamp);
196
+ if (notifications.length > 0) return notifications;
197
+ if (!parentToolUseId) return [];
198
+ const trimmed = text.trim();
199
+ if (trimmed.length === 0 || trimmed.startsWith("<system-reminder>")) return [];
200
+ return [{
201
+ type: "subagent_user_message",
202
+ text: clip(trimmed, SUBAGENT_USER_MESSAGE_MAX),
203
+ parentToolUseId,
204
+ timestamp,
205
+ }];
206
+ }
207
+
174
208
  /** One `system`/`task_notification` stream-json event mapped onto the same
175
209
  * structured message the XML parse produces, or null when the event names
176
210
  * no task id. Pure and tolerant over untrusted harness JSON: absent fields
@@ -206,6 +240,175 @@ export function parseSystemTaskNotification(
206
240
  };
207
241
  }
208
242
 
243
+ // ── Task progress (live background-workflow evidence) ───────────────────────
244
+ //
245
+ // While a harness Workflow (the CLI's in-harness dynamic-workflow tool)
246
+ // runs in the background, stream-json emits `system`/`task_progress`
247
+ // events carrying the workflow's CUMULATIVE `workflow_progress` array —
248
+ // verified live on 2.1.241 (`-p --output-format stream-json`): the array
249
+ // re-sends on every agent state change immediately and on a ~10s heartbeat
250
+ // otherwise, each `workflow_agent` entry updated in place (label,
251
+ // phase, model, state, tokens, toolCalls, durationMs). This is the ONLY
252
+ // per-agent evidence the wire ever carries — the completion notification
253
+ // reports aggregates only — so dropping it leaves workflow cards blind.
254
+ // Preview text (promptPreview / resultPreview) is internal plumbing and is
255
+ // not forwarded, same rule as task notifications.
256
+
257
+ const PROGRESS_PHASES_MAX = 32;
258
+ const PROGRESS_AGENTS_MAX = 128;
259
+ const PROGRESS_LABEL_MAX = 200;
260
+ const PROGRESS_ERROR_MAX = 500;
261
+
262
+ const AGENT_STATES: ReadonlySet<string> = new Set(["start", "progress", "done", "error"]);
263
+
264
+ /** One `system`/`task_progress` event mapped onto the structured progress
265
+ * message, or null only when it names no task. Workflow tasks carry the
266
+ * cumulative `workflow_progress` entries; a plain background Agent-task
267
+ * heartbeat (no workflow entries — verified live on 2.1.241, ~10s cadence
268
+ * while the task runs) forwards with `workflow: []` so the platform can
269
+ * DECLARE the running task as session background work (the delegation-
270
+ * survives-the-human contract: parks, user interrupts, and supersedes all
271
+ * read the declared-children registry). Completion evidence still rides
272
+ * `task_notification`. Pure and tolerant over untrusted harness JSON:
273
+ * malformed entries are skipped, everything is clamped (phases keep the
274
+ * FIRST 32 — they are seeded up front; agents keep the LAST 128 — late
275
+ * spawns matter more than a long-settled prefix). Exported for tests. */
276
+ export function parseSystemTaskProgress(
277
+ p: Record<string, unknown>, timestamp: string,
278
+ ): AgentMessageTaskProgress | null {
279
+ const taskId = typeof p.task_id === "string" && p.task_id.trim().length > 0 ? p.task_id.trim() : null;
280
+ if (!taskId) return null;
281
+ const raw = Array.isArray(p.workflow_progress) ? p.workflow_progress : [];
282
+ const nonneg = (v: unknown): number | undefined =>
283
+ typeof v === "number" && Number.isFinite(v) && v >= 0 ? Math.floor(v) : undefined;
284
+ const str = (v: unknown, max: number): string | undefined =>
285
+ typeof v === "string" && v.length > 0 ? clip(v, max) : undefined;
286
+ const phases: WorkflowProgressEntry[] = [];
287
+ const agents: WorkflowProgressEntry[] = [];
288
+ for (const e of raw) {
289
+ if (typeof e !== "object" || e === null) continue;
290
+ const o = e as Record<string, unknown>;
291
+ const index = nonneg(o.index);
292
+ if (index === undefined) continue;
293
+ if (o.type === "workflow_phase") {
294
+ const title = str(o.title, PROGRESS_LABEL_MAX);
295
+ if (title && phases.length < PROGRESS_PHASES_MAX) phases.push({ kind: "phase", index, title });
296
+ continue;
297
+ }
298
+ if (o.type !== "workflow_agent") continue;
299
+ const label = str(o.label, PROGRESS_LABEL_MAX);
300
+ const state = typeof o.state === "string" && AGENT_STATES.has(o.state) ? o.state as "start" | "progress" | "done" | "error" : null;
301
+ if (!label || !state) continue;
302
+ agents.push({
303
+ kind: "agent", index, label, state,
304
+ ...(nonneg(o.phaseIndex) !== undefined ? { phaseIndex: nonneg(o.phaseIndex) } : {}),
305
+ ...(str(o.phaseTitle, PROGRESS_LABEL_MAX) ? { phaseTitle: str(o.phaseTitle, PROGRESS_LABEL_MAX) } : {}),
306
+ ...(str(o.model, 100) ? { model: str(o.model, 100) } : {}),
307
+ ...(str(o.agentId, 128) ? { agentId: str(o.agentId, 128) } : {}),
308
+ ...(nonneg(o.tokens) !== undefined ? { tokens: nonneg(o.tokens) } : {}),
309
+ ...(nonneg(o.toolCalls) !== undefined ? { toolCalls: nonneg(o.toolCalls) } : {}),
310
+ ...(nonneg(o.durationMs) !== undefined ? { durationMs: nonneg(o.durationMs) } : {}),
311
+ ...(nonneg(o.startedAt) !== undefined ? { startedAt: nonneg(o.startedAt) } : {}),
312
+ ...(state === "error" && str(o.error, PROGRESS_ERROR_MAX) ? { error: str(o.error, PROGRESS_ERROR_MAX) } : {}),
313
+ ...(o.cached === true ? { cached: true as const } : {}),
314
+ ...(o.skipped === true ? { skipped: true as const } : {}),
315
+ });
316
+ }
317
+ // Empty = a plain Agent-task heartbeat, forwarded as liveness evidence
318
+ // (see the header) — the workflow delta folder ignores it (no entries),
319
+ // and only the background-work declare consumes it.
320
+ const workflow = [...phases, ...agents.slice(-PROGRESS_AGENTS_MAX)];
321
+ const toolUseId = typeof p.tool_use_id === "string" && p.tool_use_id.length > 0 ? p.tool_use_id : null;
322
+ const u = (typeof p.usage === "object" && p.usage !== null ? p.usage : {}) as Record<string, unknown>;
323
+ const tokens = nonneg(u.total_tokens);
324
+ const toolUses = nonneg(u.tool_uses);
325
+ const durationMs = nonneg(u.duration_ms);
326
+ return {
327
+ type: "task_progress",
328
+ taskId: clip(taskId, 128),
329
+ ...(toolUseId ? { toolUseId: clip(toolUseId, 128) } : {}),
330
+ ...(tokens !== undefined || toolUses !== undefined || durationMs !== undefined
331
+ ? { usage: {
332
+ ...(tokens !== undefined ? { tokens } : {}),
333
+ ...(toolUses !== undefined ? { toolUses } : {}),
334
+ ...(durationMs !== undefined ? { durationMs } : {}),
335
+ } }
336
+ : {}),
337
+ workflow,
338
+ timestamp,
339
+ };
340
+ }
341
+
342
+ // ── Context compaction (the harness summarizing its own conversation) ───────
343
+ //
344
+ // Captured live against 2.1.212 (the baked E2B pin) and 2.1.241 — identical
345
+ // wire shapes on both, `/compact` and forced-auto alike:
346
+ //
347
+ // {"type":"system","subtype":"status","status":"compacting"}
348
+ // {"type":"system","subtype":"status","status":null,
349
+ // "compact_result":"success"|"failed"[,"compact_error":"…"]}
350
+ // {"type":"system","subtype":"init",…} (fresh init, success only)
351
+ // {"type":"system","subtype":"compact_boundary","compact_metadata":
352
+ // {"trigger":"auto"|"manual","pre_tokens":N,"post_tokens":N,
353
+ // "cumulative_dropped_tokens":N,"duration_ms":N,"preserved_segment":…}}
354
+ //
355
+ // Followed by the continuation summary as a SYNTHETIC user message (string
356
+ // content, `isSynthetic: true`) — deliberately NOT forwarded: it quotes
357
+ // conversation content verbatim (its "All user messages" section replays
358
+ // platform notices word for word), which is exactly how the 2026-08-23
359
+ // incident re-labeled a platform system notice as a "recalled memory".
360
+ // The boundary does not replay on later resumes (verified live).
361
+ //
362
+ // A compaction spans 13-21s in the small captures and MINUTES at real
363
+ // context sizes — forwarding the lifecycle is what lets downstream render
364
+ // the silence as work instead of death (the 94%-auto-compact dead-air
365
+ // incident).
366
+
367
+ const COMPACT_ERROR_MAX = 500;
368
+
369
+ /** One `system` event of the compaction family mapped onto the structured
370
+ * lifecycle message, or null when the event is not compaction-shaped (a
371
+ * plain `status` event with neither the compacting status nor a
372
+ * compact_result stays unmapped). Pure and tolerant over untrusted harness
373
+ * JSON. Exported for tests. */
374
+ export function parseSystemCompaction(
375
+ p: Record<string, unknown>, timestamp: string,
376
+ ): AgentMessageCompaction | null {
377
+ const nonneg = (v: unknown): number | undefined =>
378
+ typeof v === "number" && Number.isFinite(v) && v >= 0 ? Math.floor(v) : undefined;
379
+ if (p.subtype === "status") {
380
+ if (p.status === "compacting") return { type: "compaction", phase: "start", timestamp };
381
+ if (typeof p.compact_result === "string") {
382
+ const failed = p.compact_result !== "success";
383
+ const error = typeof p.compact_error === "string" && p.compact_error.length > 0
384
+ ? clip(p.compact_error, COMPACT_ERROR_MAX) : undefined;
385
+ return {
386
+ type: "compaction", phase: "settled",
387
+ result: failed ? "failed" : "success",
388
+ ...(failed && error ? { error } : {}),
389
+ timestamp,
390
+ };
391
+ }
392
+ return null;
393
+ }
394
+ if (p.subtype !== "compact_boundary") return null;
395
+ const meta = (typeof p.compact_metadata === "object" && p.compact_metadata !== null
396
+ ? p.compact_metadata : {}) as Record<string, unknown>;
397
+ const preTokens = nonneg(meta.pre_tokens);
398
+ const postTokens = nonneg(meta.post_tokens);
399
+ const droppedTokens = nonneg(meta.cumulative_dropped_tokens);
400
+ const durationMs = nonneg(meta.duration_ms);
401
+ return {
402
+ type: "compaction", phase: "boundary",
403
+ trigger: meta.trigger === "manual" ? "manual" : "auto",
404
+ ...(preTokens !== undefined ? { preTokens } : {}),
405
+ ...(postTokens !== undefined ? { postTokens } : {}),
406
+ ...(droppedTokens !== undefined ? { droppedTokens } : {}),
407
+ ...(durationMs !== undefined ? { durationMs } : {}),
408
+ timestamp,
409
+ };
410
+ }
411
+
209
412
  /** Claude Code's real reasoning knob is its own `--effort <level>` flag
210
413
  * (low|medium|high|xhigh|max — verified against `claude -p --help`). The
211
414
  * CliReasoningEffort union IS the CLI's vocabulary, so the level rides the
@@ -247,6 +450,14 @@ export const claudeCodeSpec: CliAgentSpec = {
247
450
  streamInput: {
248
451
  promptLine: (prompt) => JSON.stringify({ type: "user", message: { role: "user", content: prompt } }),
249
452
  messageLine: (text) => JSON.stringify({ type: "user", message: { role: "user", content: text } }),
453
+ // The ESC equivalent (verified live against claude 2.1.236): the CLI's
454
+ // control layer processes this OUT OF BAND — a 120s foreground Bash call
455
+ // aborted 6s in, the control_response answered instantly, the run ended
456
+ // ~100ms later with `result: error_during_execution`, and `--resume` on
457
+ // the same session id kept the full turn context.
458
+ interruptLine: (requestId) => JSON.stringify({
459
+ type: "control_request", request_id: requestId, request: { subtype: "interrupt" },
460
+ }),
250
461
  },
251
462
  buildCommand: ({ promptPath, sessionId, model, cwd, effort, streamInput }) => {
252
463
  const flags = [
@@ -285,17 +496,27 @@ export const claudeCodeSpec: CliAgentSpec = {
285
496
  // tool_use id; null at top level). Forwarded on tool_use/tool_result so
286
497
  // renderers can nest child activity under the spawning call instead of
287
498
  // flattening it into the parent transcript unattributed.
288
- const parent = typeof p.parent_tool_use_id === "string" && p.parent_tool_use_id.length > 0
289
- ? { parentToolUseId: p.parent_tool_use_id }
290
- : {};
499
+ const parentId = typeof p.parent_tool_use_id === "string" && p.parent_tool_use_id.length > 0
500
+ ? p.parent_tool_use_id
501
+ : undefined;
502
+ const parent = parentId ? { parentToolUseId: parentId } : {};
291
503
  switch (p.type) {
292
504
  // Assistant API message: content blocks → text / thinking / tool_use.
293
505
  case "assistant": {
294
- const message = p.message as { content?: ClaudeContentBlock[] } | undefined;
506
+ const message = p.message as { content?: ClaudeContentBlock[]; model?: unknown } | undefined;
507
+ // HARNESS-synthesized assistant messages: the CLI stamps
508
+ // `message.model: "<synthetic>"` on content it composed itself —
509
+ // slash-command stdout, model/skills advisories — never the model
510
+ // speaking (verified live on claude 2.1.212 and 2.1.236: /model and
511
+ // /context output both arrive this way). Their text forwards as
512
+ // `harness_notice` so downstream renders it as a quiet system note
513
+ // instead of agent prose (the 2026-08-22 plumbing-as-content leak).
514
+ // Structural marker only — no content sniffing.
515
+ const synthetic = message?.model === "<synthetic>";
295
516
  const blocks = Array.isArray(message?.content) ? message.content : [];
296
517
  return blocks.flatMap((b): AgentMessage[] => {
297
518
  if (b.type === "text" && typeof b.text === "string" && b.text.length > 0) {
298
- return [{ type: "text", text: b.text, timestamp: ts }];
519
+ return [{ type: synthetic ? "harness_notice" : "text", text: b.text, timestamp: ts }];
299
520
  }
300
521
  if (b.type === "thinking" && typeof b.thinking === "string" && b.thinking.length > 0) {
301
522
  return [{ type: "thinking", text: b.thinking, timestamp: ts }];
@@ -312,12 +533,25 @@ export const claudeCodeSpec: CliAgentSpec = {
312
533
  // User API message: the CLI echoes tool results back as user content,
313
534
  // and injects `<task-notification>` blocks (background-task completion
314
535
  // evidence) as user TEXT — parsed into structure, never dropped and
315
- // never forwarded raw. Other user text (the echo of the prompt, system
316
- // reminders) stays unmapped: it is not agent output.
536
+ // never forwarded raw. TOP-LEVEL user text (the echo of the prompt,
537
+ // system reminders) stays unmapped: it is not agent output. SIDECHAIN
538
+ // user text (parent_tool_use_id present) is a message landing in a
539
+ // CHILD's thread — the delivered form of a SendMessage steer to a
540
+ // running subagent ("queued for delivery at its next tool round") —
541
+ // and forwards as `subagent_user_message` so the steer renders inside
542
+ // the child's mini-session instead of vanishing (task #97; before
543
+ // this, a queued steer was visible only as the parent's opaque
544
+ // SendMessage tool call).
317
545
  case "user": {
318
546
  const message = p.message as { content?: ClaudeContentBlock[] | string } | undefined;
547
+ // HARNESS-synthesized user messages (`isSynthetic` — e.g. a
548
+ // post-compaction continuation summary, including a CHILD's own)
549
+ // are never a delivered steer: suppress the steer arm, keep the
550
+ // notification parse. Structural marker only, same doctrine as the
551
+ // `<synthetic>` model stamp above.
552
+ const steerParent = p.isSynthetic === true ? undefined : parentId;
319
553
  if (typeof message?.content === "string") {
320
- return parseTaskNotifications(message.content, ts);
554
+ return sidechainAwareUserText(message.content, steerParent, ts);
321
555
  }
322
556
  const blocks = Array.isArray(message?.content) ? message.content : [];
323
557
  return blocks.flatMap((b): AgentMessage[] => {
@@ -328,7 +562,7 @@ export const claudeCodeSpec: CliAgentSpec = {
328
562
  }];
329
563
  }
330
564
  if (b.type === "text" && typeof b.text === "string") {
331
- return parseTaskNotifications(b.text, ts);
565
+ return sidechainAwareUserText(b.text, steerParent, ts);
332
566
  }
333
567
  return [];
334
568
  });
@@ -401,11 +635,23 @@ export const claudeCodeSpec: CliAgentSpec = {
401
635
  // System events: init carries the session id (extractSessionId), and
402
636
  // the background-task lane rides here too — `task_notification` is
403
637
  // the completion evidence stream-json actually emits (see the section
404
- // header above), so it maps to the structured message BOTH turn
405
- // engines persist. Everything else under system (task_started,
406
- // task_updated, background_tasks_changed, thinking_tokens) is
407
- // lifecycle noise here.
638
+ // header above), and `task_progress` the LIVE background-task feed
639
+ // (parseSystemTaskProgress): per-agent workflow entries for Workflow
640
+ // tasks, `workflow: []` heartbeats for plain background Agent tasks —
641
+ // both forward so the platform can declare the running task as
642
+ // session background work. Compaction rides here as well (the status
643
+ // pair + compact_boundary — parseSystemCompaction). Everything else
644
+ // under system (task_started, task_updated, background_tasks_changed,
645
+ // thinking_tokens) is lifecycle noise here.
408
646
  case "system": {
647
+ if (p.subtype === "task_progress") {
648
+ const progress = parseSystemTaskProgress(p, ts);
649
+ return progress ? [progress] : [];
650
+ }
651
+ if (p.subtype === "status" || p.subtype === "compact_boundary") {
652
+ const compaction = parseSystemCompaction(p, ts);
653
+ return compaction ? [compaction] : [];
654
+ }
409
655
  if (p.subtype !== "task_notification") return [];
410
656
  const notification = parseSystemTaskNotification(p, ts);
411
657
  return notification ? [notification] : [];
@@ -3,6 +3,17 @@
3
3
  *
4
4
  * Works against DesktopSandboxProvider — no E2B SDK dependency here.
5
5
  * The action→screenshot loop runs until the model emits text, then yields it.
6
+ *
7
+ * COORDINATE SPACES (incident note). Every screenshot is downscaled to the
8
+ * model's DISPLAY geometry (1024×720) before it is sent, so the model emits
9
+ * coordinates in the SCALED screenshot's space — never the real desktop's.
10
+ * An earlier revision executed those coordinates against the real desktop
11
+ * using hardcoded real-geometry constants, so on any sandbox whose actual
12
+ * geometry differed, every click landed in the wrong place. The provider has
13
+ * no geometry call, so the real geometry is read from each screenshot buffer
14
+ * itself (sharp metadata), and the resulting per-capture `CaptureScale` maps
15
+ * the model's coordinates back onto the exact frame it saw. The old constants
16
+ * survive ONLY as a fallback for the metadata-missing case.
6
17
  */
7
18
 
8
19
  import OpenAI from "openai";
@@ -12,8 +23,11 @@ import { defineRuntime, formatError } from "../index.js";
12
23
 
13
24
  const DISPLAY_WIDTH = 1024;
14
25
  const DISPLAY_HEIGHT = 720;
15
- const DESKTOP_WIDTH = 1280;
16
- const DESKTOP_HEIGHT = 800;
26
+ /** FALLBACK real-desktop geometry — consulted ONLY when sharp cannot read a
27
+ * screenshot's dimensions. The truth is derived per capture from the raw
28
+ * screenshot buffer itself (see `captureScaledScreenshot`). */
29
+ const FALLBACK_DESKTOP_WIDTH = 1280;
30
+ const FALLBACK_DESKTOP_HEIGHT = 800;
17
31
  const MAX_TURNS = 40;
18
32
 
19
33
  export interface OpenAIDesktopRunnerOptions extends RuntimeOptions {
@@ -27,8 +41,9 @@ export interface OpenAIDesktopRunnerOptions extends RuntimeOptions {
27
41
  // so the exchange is described here instead of with SDK types.
28
42
 
29
43
  /** One model-emitted desktop action. Coordinates arrive in DISPLAY (model)
30
- * space and are rescaled to the real desktop before dispatch. */
31
- interface ComputerAction {
44
+ * space and are rescaled through the CURRENT capture's `CaptureScale`
45
+ * before dispatch. Exported so tests can type their fixtures. */
46
+ export interface ComputerAction {
32
47
  type: string;
33
48
  coordinate?: [number, number];
34
49
  startCoordinate?: [number, number];
@@ -99,12 +114,16 @@ export class OpenAIDesktopRunner implements ModelExecutionContract {
99
114
 
100
115
  yield { type: "init", sessionId: "e2b-desktop", timestamp: ts() };
101
116
 
102
- const screenshot = await captureScaledScreenshot(sandbox);
117
+ const first = await captureScaledScreenshot(sandbox);
118
+ // The scale of the LATEST frame the model has seen — its next batch of
119
+ // coordinates is in that frame's space, so every fresh capture below
120
+ // replaces this before the frame is sent back.
121
+ let scale = first.scale;
103
122
 
104
123
  const messages: InputMessage[] = [{
105
124
  role: "user",
106
125
  content: [
107
- { type: "input_image", image_url: `data:image/png;base64,${screenshot}` },
126
+ { type: "input_image", image_url: `data:image/png;base64,${first.b64}` },
108
127
  { type: "input_text", text: opts.prompt },
109
128
  ],
110
129
  }];
@@ -137,7 +156,7 @@ export class OpenAIDesktopRunner implements ModelExecutionContract {
137
156
 
138
157
  for (const call of computerCalls) {
139
158
  console.log(`${label} → ${call.action.type}`);
140
- await executeAction(sandbox, call.action).catch(
159
+ await executeAction(sandbox, call.action, scale).catch(
141
160
  (err: unknown) => console.warn(`${label} Action failed (non-fatal): ${formatError(err)}`),
142
161
  );
143
162
  }
@@ -145,10 +164,11 @@ export class OpenAIDesktopRunner implements ModelExecutionContract {
145
164
  if (computerCalls.length === 0) break;
146
165
 
147
166
  const next = await captureScaledScreenshot(sandbox);
167
+ scale = next.scale;
148
168
  messages.push({ role: "user", content: computerCalls.map((call): InputPart => ({
149
169
  type: "computer_call_output",
150
170
  call_id: call.call_id,
151
- output: { type: "input_image", image_url: `data:image/png;base64,${next}` },
171
+ output: { type: "input_image", image_url: `data:image/png;base64,${next.b64}` },
152
172
  })) });
153
173
  }
154
174
 
@@ -160,21 +180,58 @@ export class OpenAIDesktopRunner implements ModelExecutionContract {
160
180
  // Helpers — work against DesktopSandboxProvider interface
161
181
  // ---------------------------------------------------------------------------
162
182
 
163
- async function captureScaledScreenshot(sandbox: DesktopSandboxProvider): Promise<string> {
183
+ /** Real-desktop pixels per model-DISPLAY pixel, derived from ONE capture.
184
+ * Coordinates the model emits against that capture are multiplied by these
185
+ * factors (and rounded) before touching the provider. */
186
+ export interface CaptureScale {
187
+ scaleX: number;
188
+ scaleY: number;
189
+ }
190
+
191
+ /** One captured frame: the model-facing image plus the scale that maps the
192
+ * model's coordinates back onto this frame's real desktop. */
193
+ export interface ScaledCapture {
194
+ /** Base64 PNG resized to DISPLAY_WIDTH×DISPLAY_HEIGHT (fit "fill"). */
195
+ b64: string;
196
+ scale: CaptureScale;
197
+ }
198
+
199
+ /** Derive a capture's scale from its real pixel dimensions. The constants
200
+ * are a FALLBACK only — used when sharp cannot read the frame's metadata;
201
+ * a real dimension always wins. */
202
+ export function scaleFromDimensions(width: number | undefined, height: number | undefined): CaptureScale {
203
+ return {
204
+ scaleX: (width ?? FALLBACK_DESKTOP_WIDTH) / DISPLAY_WIDTH,
205
+ scaleY: (height ?? FALLBACK_DESKTOP_HEIGHT) / DISPLAY_HEIGHT,
206
+ };
207
+ }
208
+
209
+ /** Capture one frame: read the REAL geometry from the raw screenshot's own
210
+ * metadata, downscale to the model's DISPLAY geometry, and return both the
211
+ * image and the scale that maps model coordinates back onto this frame. */
212
+ export async function captureScaledScreenshot(sandbox: DesktopSandboxProvider): Promise<ScaledCapture> {
164
213
  const raw = await sandbox.screenshot();
165
- const scaled = await sharp(raw)
214
+ const image = sharp(raw);
215
+ const { width, height } = await image.metadata();
216
+ const scaled = await image
166
217
  .resize(DISPLAY_WIDTH, DISPLAY_HEIGHT, { kernel: "lanczos3", fit: "fill" })
167
218
  .png()
168
219
  .toBuffer();
169
- return scaled.toString("base64");
220
+ return { b64: scaled.toString("base64"), scale: scaleFromDimensions(width, height) };
170
221
  }
171
222
 
172
- function scaleX(x: number): number { return Math.round(x * DESKTOP_WIDTH / DISPLAY_WIDTH); }
173
- function scaleY(y: number): number { return Math.round(y * DESKTOP_HEIGHT / DISPLAY_HEIGHT); }
174
-
175
- async function executeAction(sandbox: DesktopSandboxProvider, action: ComputerAction): Promise<void> {
223
+ /** Dispatch one model action against the provider, mapping every coordinate
224
+ * through the CURRENT capture's scale. Non-coordinate payloads — scroll
225
+ * deltas (ticks), typed text, key chords — pass through untouched. */
226
+ export async function executeAction(
227
+ sandbox: DesktopSandboxProvider,
228
+ action: ComputerAction,
229
+ scale: CaptureScale,
230
+ ): Promise<void> {
231
+ const mapX = (x: number) => Math.round(x * scale.scaleX);
232
+ const mapY = (y: number) => Math.round(y * scale.scaleY);
176
233
  const [x, y] = action.coordinate
177
- ? [scaleX(action.coordinate[0]), scaleY(action.coordinate[1])]
234
+ ? [mapX(action.coordinate[0]), mapY(action.coordinate[1])]
178
235
  : [0, 0];
179
236
 
180
237
  switch (action.type) {
@@ -185,12 +242,18 @@ async function executeAction(sandbox: DesktopSandboxProvider, action: ComputerAc
185
242
  case "move": await sandbox.moveMouse(x, y); break;
186
243
  case "type": await sandbox.write(action.text ?? ""); break;
187
244
  case "key": await sandbox.press(action.key ?? ""); break;
188
- case "scroll": await sandbox.scroll(action.direction === "up" ? "up" : "down", action.ticks ?? 3); break;
245
+ case "scroll":
246
+ // The scroll POSITION is a coordinate — mapped (the provider has no
247
+ // positional scroll, so position lands via moveMouse). The scroll
248
+ // DELTA (ticks) is not a coordinate — untouched.
249
+ if (action.coordinate) await sandbox.moveMouse(x, y);
250
+ await sandbox.scroll(action.direction === "up" ? "up" : "down", action.ticks ?? 3);
251
+ break;
189
252
  case "drag":
190
253
  if (action.startCoordinate && action.endCoordinate) {
191
254
  await sandbox.drag(
192
- [scaleX(action.startCoordinate[0]), scaleY(action.startCoordinate[1])],
193
- [scaleX(action.endCoordinate[0]), scaleY(action.endCoordinate[1])],
255
+ [mapX(action.startCoordinate[0]), mapY(action.startCoordinate[1])],
256
+ [mapX(action.endCoordinate[0]), mapY(action.endCoordinate[1])],
194
257
  );
195
258
  }
196
259
  break;
@@ -5,13 +5,13 @@
5
5
  * This module is the SINGLE SOURCE for the machine's template identity: the
6
6
  * recipe lives in `infra/e2b-template/devbox.ts`, `infra/e2b-template/build.ts`
7
7
  * bakes it under the alias below, and the server imports the alias from here to
8
- * provision a machine (exactly how `e2bBaseTemplate` reaches the E2B provider —
9
- * a stable ALIAS, never a snapshot id, so the same string resolves in whichever
10
- * E2B account a deployment uses).
8
+ * provision a machine (exactly how `e2bAgentEnvTemplate` reaches the E2B
9
+ * provider — a stable ALIAS, never a snapshot id, so the same string resolves
10
+ * in whichever E2B account a deployment uses).
11
11
  *
12
12
  * Interactive only. Workflow RUNS never execute on a devbox (ADR-0038 §2.6):
13
- * runs boot the clean per-size `agent-compose-base-<size>` / `agent-env-<size>`
14
- * templates so reproducibility never inherits hand-configured drift.
13
+ * runs boot the clean per-size `agent-env-<size>` template (the same image
14
+ * sessions boot) so reproducibility never inherits hand-configured drift.
15
15
  */
16
16
 
17
17
  /** Stable E2B template ALIAS for the per-member desktop machine. Baked by