@agent-compose/sdk 0.8.4 → 0.8.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (106) hide show
  1. package/README.md +213 -189
  2. package/dist/agent/agent-context.d.ts +9 -1
  3. package/dist/agent/agent-loop.d.ts +14 -6
  4. package/dist/agent/perf-sampler.d.ts +27 -2
  5. package/dist/agent/run-agent.d.ts +1 -1
  6. package/dist/client.d.ts +250 -59
  7. package/dist/directives.d.ts +14 -0
  8. package/dist/display.d.ts +7 -0
  9. package/dist/errors.d.ts +1 -1
  10. package/dist/generated/agentc-commands.d.ts +34 -0
  11. package/dist/index.d.ts +13 -11
  12. package/dist/index.js +1692 -194
  13. package/dist/request-context/request-context.d.ts +1 -1
  14. package/dist/runtimes/_cli-agent.d.ts +278 -58
  15. package/dist/runtimes/claude-code.d.ts +90 -1
  16. package/dist/runtimes/claude.d.ts +1 -1
  17. package/dist/runtimes/codex.d.ts +94 -6
  18. package/dist/runtimes/codex.mid-turn-hook.test.d.ts +10 -0
  19. package/dist/runtimes/openai-desktop.d.ts +50 -0
  20. package/dist/runtimes/openai-desktop.js +1689 -211
  21. package/dist/runtimes/openai-desktop.test.d.ts +20 -0
  22. package/dist/runtimes/opencode.d.ts +48 -11
  23. package/dist/runtimes/opencode.test.d.ts +14 -0
  24. package/dist/runtimes/tool-pulse.test.d.ts +17 -0
  25. package/dist/sandbox/baked-clis.d.ts +75 -0
  26. package/dist/sandbox/devbox.d.ts +5 -5
  27. package/dist/sandbox/exec-stream.d.ts +1 -2
  28. package/dist/sandbox/network-policy.d.ts +23 -5
  29. package/dist/sandbox/registry.d.ts +12 -0
  30. package/dist/sandbox/sizes.d.ts +11 -5
  31. package/dist/sandbox.d.ts +5 -3
  32. package/dist/step-invocation/protocol.d.ts +3 -4
  33. package/dist/step-invocation/server.d.ts +2 -2
  34. package/dist/step-invocation/types.d.ts +2 -2
  35. package/dist/types/api-conversations.d.ts +513 -27
  36. package/dist/types/api-factory.d.ts +183 -3
  37. package/dist/types/api-projects.d.ts +480 -0
  38. package/dist/types/api-runs.d.ts +8 -0
  39. package/dist/types/api-scopes.d.ts +32 -3
  40. package/dist/types/conversation-stream.d.ts +27 -1
  41. package/dist/types/execution-context.d.ts +1 -1
  42. package/dist/types/protocol.d.ts +182 -2
  43. package/dist/types/runtime.d.ts +80 -2
  44. package/dist/types/workflow-metadata.d.ts +2 -4
  45. package/dist/types/workflow-plan.d.ts +1 -3
  46. package/dist/utils/bundler.d.ts +23 -0
  47. package/dist/workflow-steps/observability.d.ts +2 -3
  48. package/dist/workflow-steps/runner.d.ts +5 -8
  49. package/dist/workflow-steps/types.d.ts +8 -10
  50. package/dist/workflow-steps/workflow.d.ts +2 -1
  51. package/dist/workflows/engine.d.ts +3 -5
  52. package/dist/workflows/invoke-child.d.ts +2 -2
  53. package/package.json +2 -2
  54. package/src/agent/agent-context.ts +193 -116
  55. package/src/agent/agent-loop.ts +16 -9
  56. package/src/agent/desktop-open.ts +13 -1
  57. package/src/agent/perf-sampler.ts +54 -3
  58. package/src/agent/run-agent.ts +1 -1
  59. package/src/client.ts +418 -80
  60. package/src/directives.ts +21 -1
  61. package/src/display.ts +12 -0
  62. package/src/errors.ts +1 -0
  63. package/src/generated/agentc-commands.ts +571 -0
  64. package/src/index.ts +65 -18
  65. package/src/pause/pause-core.ts +2 -1
  66. package/src/request-context/request-context.ts +1 -1
  67. package/src/runtimes/_cli-agent.ts +607 -132
  68. package/src/runtimes/claude-code.ts +427 -20
  69. package/src/runtimes/claude.ts +1 -1
  70. package/src/runtimes/codex.ts +188 -19
  71. package/src/runtimes/openai-desktop.ts +82 -19
  72. package/src/runtimes/opencode.ts +195 -26
  73. package/src/sandbox/baked-clis.ts +86 -0
  74. package/src/sandbox/devbox.ts +5 -5
  75. package/src/sandbox/exec-stream.ts +1 -2
  76. package/src/sandbox/network-policy.ts +51 -7
  77. package/src/sandbox/providers/e2b.ts +63 -19
  78. package/src/sandbox/providers/vercel.ts +6 -6
  79. package/src/sandbox/registry.ts +19 -1
  80. package/src/sandbox/sizes.ts +11 -5
  81. package/src/sandbox.ts +9 -2
  82. package/src/step-invocation/invoker.ts +2 -6
  83. package/src/step-invocation/protocol.ts +3 -4
  84. package/src/step-invocation/server.ts +2 -2
  85. package/src/types/api-conversations.ts +424 -29
  86. package/src/types/api-factory.ts +189 -3
  87. package/src/types/api-projects.ts +443 -0
  88. package/src/types/api-runs.ts +5 -0
  89. package/src/types/api-scopes.ts +32 -3
  90. package/src/types/conversation-stream.ts +29 -1
  91. package/src/types/execution-context.ts +1 -1
  92. package/src/types/protocol.ts +180 -2
  93. package/src/types/runtime.ts +71 -2
  94. package/src/types/sandbox-environment.ts +1 -2
  95. package/src/types/workflow-metadata.ts +2 -4
  96. package/src/types/workflow-plan.ts +1 -3
  97. package/src/utils/bundler.ts +88 -19
  98. package/src/workflow-steps/observability.ts +2 -3
  99. package/src/workflow-steps/runner.ts +5 -8
  100. package/src/workflow-steps/types.ts +8 -10
  101. package/src/workflow-steps/workflow.ts +2 -1
  102. package/src/workflows/engine.ts +3 -5
  103. package/src/workflows/invoke-child.ts +2 -2
  104. package/dist/pause/__tests__/errors.test.d.ts +0 -1
  105. package/dist/pause/__tests__/wrappers.test.d.ts +0 -1
  106. package/dist/step-invocation/__tests__/protocol.test.d.ts +0 -1
@@ -5,17 +5,28 @@
5
5
  * only stream-parse what it prints.
6
6
  *
7
7
  * Auth: set `OPENAI_API_KEY` (or `CODEX_API_KEY`) in the sandbox env via a
8
- * workflow secret. The runtime installs the `codex` CLI (`@openai/codex`) on
9
- * demand — no image baking needed; pair with `snapshots: { bootFrom: "reuse" }`
10
- * to install once and boot from the captured snapshot on every run after.
8
+ * workflow secret. The E2B session image bakes the pinned `codex` CLI
9
+ * (`@openai/codex@CODEX_CLI_VERSION`); on any other machine the runtime
10
+ * installs that same version on demand (pair with
11
+ * `snapshots: { bootFrom: "reuse" }` to install once and boot from the
12
+ * captured snapshot on every run after).
11
13
  *
12
14
  * Verified against codex-cli 0.124.0: `codex exec --json` + resume-by-thread,
13
- * with the command_execution / reasoning / agent_message item shapes below.
15
+ * with the command_execution / reasoning / agent_message item shapes below;
16
+ * the plan's `todo_list` item against codex-rs rust-v0.153.4 and
17
+ * rust-v0.159.2 (see codexPlanMessages); the command_execution /
18
+ * agent_message / turn.completed shapes re-verified live against 0.160.0
19
+ * (the pin), whose plan-tool sources are byte-identical to rust-v0.159.2.
14
20
  */
15
21
 
16
22
  import type { AgentMessage } from "../index.js";
17
- import { createCliAgentRuntime, shellQuote, type CliAgentSpec, type CliReasoningEffort } from "./_cli-agent.js";
23
+ import type { AgentMessagePlan } from "../types/protocol.js";
24
+ import {
25
+ MID_TURN_DELIVERED_ENV, MID_TURN_INBOX_ENV, createCliAgentRuntime, shellQuote,
26
+ type CliAgentSpec, type CliReasoningEffort,
27
+ } from "./_cli-agent.js";
18
28
  import { formatError } from "../utils/errors.js";
29
+ import { CODEX_CLI_VERSION } from "../sandbox/baked-clis.js";
19
30
 
20
31
  function now(): string { return new Date().toISOString(); }
21
32
 
@@ -24,6 +35,97 @@ function now(): string { return new Date().toISOString(); }
24
35
  * overrides the bundled `@openai/codex` the adapter drives underneath. */
25
36
  const CODEX_ACP_ADAPTER = "@agentclientprotocol/codex-acp@0.1.0";
26
37
 
38
+ /** The Bash PreToolUse hook rtk installs for Codex (`rtk init -g --codex`
39
+ * writes exactly this entry — matcher `Bash`, command `rtk hook codex` —
40
+ * into $CODEX_HOME/hooks.json). `rtk hook codex` exists since rtk 0.50.0;
41
+ * the image bakes RTK_VERSION (sandbox/baked-clis.ts), which this payload
42
+ * was run against. Codex's shell tool matches the `Bash` matcher (codex-rs
43
+ * rust-v0.160.0, the pinned CODEX_CLI_VERSION; identical at
44
+ * rust-v0.159.2), and the hook runs under `$SHELL -lc` with the PreToolUse
45
+ * JSON on stdin. `rtk hook codex` answers `permissionDecision: "allow"` +
46
+ * `updatedInput` rewriting `git status` to `rtk git status` when it has a
47
+ * filter for the command; Codex applies the replacement before its own
48
+ * approval and sandbox checks. For a command it cannot compress (an
49
+ * unknown tool, a pipe into one, substitutions, heredocs, redirects, a
50
+ * `RTK_DISABLED=1` prefix, an unknown permission mode) it prints nothing
51
+ * and exits 0, and Codex runs the original unchanged.
52
+ *
53
+ * The wrapper fails OPEN both ways, like RTK_BASH_HOOK_COMMAND
54
+ * (claude-code.ts). `command -v` covers an image without rtk (the devbox,
55
+ * a bare Vercel VM): silent, exit 0, so Codex runs the command instead of
56
+ * failing every shell call's hook with 127. The trailing `exit 0` covers
57
+ * an rtk that cannot answer: Codex treats a PreToolUse hook's exit 2 with
58
+ * stderr as a BLOCK of the tool call (codex-rs hooks/src/events/
59
+ * pre_tool_use.rs at rust-v0.160.0 — the model reads "Command blocked by
60
+ * PreToolUse hook: <stderr>"), and clap exits 2 with its usage on stderr
61
+ * for a subcommand it does not know. That was 2026-10-03: the image's rtk
62
+ * was a cache-served 0.45.0 with no `hook codex`, and this hook — `exec`
63
+ * handing rtk's exit code to Codex — blocked every shell command of every
64
+ * codex session. rtk's hooks never exit non-zero on purpose (they fail
65
+ * open with no stdout), so a non-zero exit is always a broken rtk, and the
66
+ * compressor must never cost the worker its shell: the exit code is
67
+ * dropped, and the smoke gate's codex-rtk-hook-rewrite check is what
68
+ * proves the rewrite itself. */
69
+ export const RTK_CODEX_HOOK_COMMAND =
70
+ "command -v rtk >/dev/null 2>&1 && rtk hook codex; exit 0";
71
+
72
+ /** The awk program of the mid-turn hook below: the inbox lines not yet fed
73
+ * (JSON string literals, one per line — `codexSpec.midTurnInput.messageLine`)
74
+ * become ONE PostToolUse hook output. Each literal's quotes are stripped and
75
+ * the bodies are joined with an escaped blank line; the bodies are already
76
+ * JSON-escaped, so the result is one valid JSON string. Nothing is printed
77
+ * when no line is due, and codex ignores an empty stdout. */
78
+ const CODEX_MID_TURN_HOOK_AWK = String.raw`NR > fed && NR <= upto { s = substr($0, 2, length($0) - 2); out = (out == "" ? s : out "\\n\\n" s) } END { if (out != "") printf "{\"hookSpecificOutput\":{\"hookEventName\":\"PostToolUse\",\"additionalContext\":\"%s\"}}\n", out }`;
79
+
80
+ /** The PostToolUse command hook that carries a mid-turn message into a
81
+ * RUNNING codex turn (the 2026-10-02 relayed-steer incident: three of the
82
+ * owner's instructions waited 48-51 minutes for a codex build to end).
83
+ * Codex runs it, under `$SHELL -lc` with the event JSON on stdin and the
84
+ * codex process's own environment, after every tool call it completes
85
+ * (codex-rs core/src/tools/registry.rs → hook_runtime.rs at rust-v0.160.0,
86
+ * the pin). The launch wrapper exported the turn's inbox and delivered-
87
+ * counter paths into that environment (sdk _cli-agent.ts
88
+ * toolHookInboxFragment); the hook forwards the inbox lines past the
89
+ * counter as the hook's `additionalContext`, which codex records as
90
+ * developer context in the live turn's history before the model's next
91
+ * request (hook_runtime.rs record_additional_contexts), then advances the
92
+ * counter — the same ack the server's inject lane trusts for the stdin
93
+ * lane. Not a platform turn (no inbox in the environment): silent, exit 0.
94
+ * Codex's default spill threshold for a hook's additional context is 2,500
95
+ * tokens (hooks/src/output_spill.rs); a longer message reaches the model
96
+ * as a preview plus a file pointer, which a steer never is. */
97
+ export const CODEX_MID_TURN_HOOK_COMMAND =
98
+ // Drain the event JSON first, so codex's stdin write never meets a closed pipe.
99
+ "cat >/dev/null; "
100
+ + `[ -n "\${${MID_TURN_INBOX_ENV}:-}" ] && [ -f "$${MID_TURN_INBOX_ENV}" ] || exit 0; `
101
+ + `ac_fed=$(cat "$${MID_TURN_DELIVERED_ENV}" 2>/dev/null); ac_fed=\${ac_fed:-0}; `
102
+ + `ac_lines=$(wc -l < "$${MID_TURN_INBOX_ENV}" 2>/dev/null); ac_lines=\${ac_lines:-0}; `
103
+ + `[ "$ac_lines" -gt "$ac_fed" ] || exit 0; `
104
+ + `awk -v fed="$ac_fed" -v upto="$ac_lines" '${CODEX_MID_TURN_HOOK_AWK}' "$${MID_TURN_INBOX_ENV}" `
105
+ + `&& echo "$ac_lines" > "$${MID_TURN_DELIVERED_ENV}"; exit 0`;
106
+
107
+ /** $CODEX_HOME/hooks.json for every platform Codex session (the server
108
+ * writes it at boot, session-runtime-config.ts). Codex only RUNS a
109
+ * user-level hook it has persisted trust for — the TUI's review prompt
110
+ * has no headless counterpart, and `--dangerously-bypass-hook-trust` would
111
+ * run a cloned repository's `.codex/hooks.json` unreviewed too — so the
112
+ * server also writes the trust record for each of these exact hooks into
113
+ * the config.toml it generates (sandbox/codex-hooks.ts). Verified live
114
+ * against codex 0.159.2 and 0.160.0 for the rtk hook: with the record,
115
+ * `codex exec` rewrote `git status` through rtk with no bypass flag;
116
+ * without it, the hook was skipped.
117
+ *
118
+ * The PostToolUse group carries NO matcher: codex runs a matcher-less hook
119
+ * after every tool it completes (hooks/src/events/common.rs
120
+ * matches_matcher: an absent matcher is a match), so a mid-turn message
121
+ * lands at the next tool step whatever the tool was. */
122
+ export const CODEX_PLATFORM_HOOKS = {
123
+ hooks: {
124
+ PreToolUse: [{ matcher: "Bash", hooks: [{ type: "command", command: RTK_CODEX_HOOK_COMMAND }] }],
125
+ PostToolUse: [{ hooks: [{ type: "command", command: CODEX_MID_TURN_HOOK_COMMAND }] }],
126
+ },
127
+ } as const;
128
+
27
129
  /** Known-noise codex ADVISORY lines. codex emits these as `item.completed`
28
130
  * error items on the `--json` stream (exec maps every `Warning` notification
29
131
  * to an error item — verified against codex-cli 0.147.0), so without a filter
@@ -101,6 +203,41 @@ export function codexWriterLockPreflight(
101
203
  + `flock -n "$ac_lk" rm -f -- "$ac_lk" 2>/dev/null || true; fi; `;
102
204
  }
103
205
 
206
+ /**
207
+ * codex's PLAN (its update_plan tool) on the `--json` stream: one
208
+ * `todo_list` item per turn, `item.started` on the plan's first update,
209
+ * `item.updated` (same id, the whole list) on every later one, and
210
+ * `item.completed` at the turn's end. That is codex-rs exec's
211
+ * event_processor_with_jsonl_output.rs, identical in rust-v0.153.4 and
212
+ * rust-v0.159.2, where it is the only `item.updated` codex emits; the item
213
+ * `{ id, type: "todo_list", items: [{ text, completed }] }` is the shape every
214
+ * todo_list item our machines have persisted carries. Each event maps to ONE
215
+ * whole-plan `plan` message, so the transcript's checklist, and the worker's
216
+ * status line, follow the plan while the turn runs (before this, the list
217
+ * surfaced once, at the turn's end, as a raw `todo_list` tool card).
218
+ *
219
+ * A step carries only `completed`: codex's `pending` and `in_progress` both
220
+ * serialize as false. Its plan tool allows at most one step in progress and
221
+ * a plan runs in order, so the first step not completed is the one under
222
+ * way; the rest are pending. The stream names no priority, so none is set.
223
+ */
224
+ function codexPlanMessages(item: Record<string, unknown>, timestamp: string): AgentMessage[] {
225
+ const steps = Array.isArray(item.items) ? item.items : [];
226
+ let underWay = false;
227
+ const entries: AgentMessagePlan["entries"] = [];
228
+ for (const step of steps) {
229
+ if (typeof step !== "object" || step === null) continue;
230
+ const text = (step as { text?: unknown }).text;
231
+ if (typeof text !== "string" || text.trim().length === 0) continue;
232
+ let status: AgentMessagePlan["entries"][number]["status"];
233
+ if ((step as { completed?: unknown }).completed === true) status = "completed";
234
+ else if (!underWay) { status = "in_progress"; underWay = true; }
235
+ else status = "pending";
236
+ entries.push({ content: text.trim(), status });
237
+ }
238
+ return entries.length > 0 ? [{ type: "plan", entries, timestamp }] : [];
239
+ }
240
+
104
241
  /** Exported for the fixture-based parity tests (ADR-0020): the legacy JSONL
105
242
  * `mapEvent` is the golden the ACP normaliser is asserted equal to. Not part
106
243
  * of the public runtime surface — `createCodexRuntime` stays the entry point. */
@@ -113,6 +250,22 @@ export const codexSpec: CliAgentSpec = {
113
250
  // every resume of the same thread, so supersede teardown must KILL it —
114
251
  // never detach it alive (server runner-kill.ts consumes this flag).
115
252
  exclusiveSessionWriter: true,
253
+ // Mid-turn input (the 2026-10-02 relayed-steer incident). codex reads
254
+ // stdin ONCE, as the prompt: `codex exec -` submits one UserTurn and never
255
+ // reads stdin again (codex-rs exec/src/lib.rs read_prompt_from_stdin and
256
+ // its single InitialOperation::UserTurn at rust-v0.160.0; the app-server's
257
+ // `turn/steer` is a different transport, not `codex exec`). So a running
258
+ // turn takes a message through the PostToolUse hook above: codex runs it
259
+ // after every tool call it completes, with the codex process's own
260
+ // environment; the hook prints the unfed inbox lines as `additionalContext`
261
+ // and codex records them as developer context in the live turn's history
262
+ // before the model's next request. One JSON string literal per inbox line,
263
+ // so the hook splices bodies without decoding. Codex builds the PostToolUse
264
+ // payload only for a tool call it counts as successful (tools/registry.rs),
265
+ // so a message lands at the next successful tool step. Source-verified at
266
+ // rust-v0.160.0 (the pinned CODEX_CLI_VERSION); the end-to-end run against
267
+ // a live codex is the acceptance check still owed.
268
+ midTurnInput: { transport: "tool-hook", messageLine: (text) => JSON.stringify(text) },
116
269
  // ACP-mode invocation (ADR-0020 increment 1). `codex` has no native `--acp`
117
270
  // flag; the adapter (a Rust binary shipped via npm) speaks ACP and drives a
118
271
  // compatible bundled `@openai/codex`. The runner attempts this first and
@@ -125,9 +278,11 @@ export const codexSpec: CliAgentSpec = {
125
278
  // (e.g. a sandbox-baked one) instead of the adapter's bundled copy.
126
279
  ...(process.env.CODEX_PATH ? { env: { CODEX_PATH: process.env.CODEX_PATH } } : {}),
127
280
  },
128
- // Global npm install; symlink onto PATH only if the global bin dir isn't
129
- // already there (so a non-login `sh -c` can find it).
130
- install: 'sudo npm install -g @openai/codex && (command -v codex >/dev/null 2>&1 || sudo ln -sf "$(npm prefix -g)/bin/codex" /usr/local/bin/codex)',
281
+ // The E2B session image bakes this exact version (sandbox/baked-clis.ts),
282
+ // so this runs only on a machine whose image predates the bake: a global
283
+ // npm install of the SAME pin, symlinked onto PATH only if the global bin
284
+ // dir isn't already there (so a non-login `sh -c` can find it).
285
+ install: `sudo npm install -g @openai/codex@${CODEX_CLI_VERSION} && (command -v codex >/dev/null 2>&1 || sudo ln -sf "$(npm prefix -g)/bin/codex" /usr/local/bin/codex)`,
131
286
  // Codex reads the prompt from stdin when invoked as `codex exec ... -`.
132
287
  promptPayload: (prompt) => prompt,
133
288
  buildCommand: ({ promptPath, sessionId, model, cwd, effort }) => {
@@ -140,13 +295,13 @@ export const codexSpec: CliAgentSpec = {
140
295
  // for running in environments that are externally sandboxed".
141
296
  "--dangerously-bypass-approvals-and-sandbox",
142
297
  ...(model ? ["-m", shellQuote(model)] : []),
143
- // codex's own reasoning knob — a config override, valid values
144
- // none|minimal|low|medium|high|xhigh (codex docs; xhigh is the
145
- // codex-max-tier deep-reasoning level). There is NO "max" in codex's
146
- // vocabulary: the server rejects it for codex sessions
147
- // (sessionEffortLockError), and this clamp to xhigh is the type-level
148
- // backstop for a caller that bypasses that gate. The value comes from
149
- // the closed CliReasoningEffort set, so it is shell-safe unquoted.
298
+ // codex's own reasoning knob — a config override. Every codex model
299
+ // takes low|medium|high|xhigh; only the GPT-6 and GPT-5.6 presets add
300
+ // max (codex 0.153+), so the platform offers codex up to xhigh: the
301
+ // server rejects "max" for codex sessions (sessionEffortLockError), and
302
+ // this clamp to xhigh is the type-level backstop for a caller that
303
+ // bypasses that gate. The value comes from the closed
304
+ // CliReasoningEffort set, so it is shell-safe unquoted.
150
305
  ...(effort ? ["-c", `model_reasoning_effort=${effort === "max" ? "xhigh" : effort}`] : []),
151
306
  ].join(" ");
152
307
  // Working directory via a shell `cd` — the idiom EVERY other runtime uses
@@ -172,11 +327,19 @@ export const codexSpec: CliAgentSpec = {
172
327
  mapEvent: (p): AgentMessage[] => {
173
328
  const ts = now();
174
329
  switch (p.type) {
330
+ case "item.updated": {
331
+ // Only the plan updates in place (codexPlanMessages); any other
332
+ // item surfaces on its started / completed events below.
333
+ const item = p.item as Record<string, unknown> | undefined;
334
+ return item?.type === "todo_list" ? codexPlanMessages(item, ts) : [];
335
+ }
175
336
  case "item.started":
176
337
  case "item.completed": {
177
338
  const item = p.item as Record<string, unknown> | undefined;
178
339
  if (!item) return [];
179
340
  const itype = String(item.type ?? "");
341
+ // The plan, whole, on its first update and at the turn's end.
342
+ if (itype === "todo_list") return codexPlanMessages(item, ts);
180
343
  // Text + reasoning land on completion (started carries no final text).
181
344
  if (itype === "agent_message") {
182
345
  return p.type === "item.completed" ? [{ type: "text", text: String(item.text ?? ""), timestamp: ts }] : [];
@@ -209,12 +372,17 @@ export const codexSpec: CliAgentSpec = {
209
372
  }
210
373
  return [{ type: "error", text, timestamp: ts }];
211
374
  }
212
- // file_change / mcp_tool_call / web_search / todo: surface once, on completion.
375
+ // file_change / mcp_tool_call / web_search: surface once, on completion.
213
376
  if (p.type === "item.completed") {
214
377
  return [{ type: "tool_use", toolName: itype || "item", toolInput: item, toolUseId: String(item.id ?? ""), timestamp: ts }];
215
378
  }
216
379
  return [];
217
380
  }
381
+ // The turn's token classes as codex reports them (codex-rs
382
+ // exec_events.rs `Usage`): input, cached input, cache-write input,
383
+ // output and reasoning output. Every class is kept — cache writes land
384
+ // in the creation class, reasoning rides beside output (it is counted
385
+ // inside output_tokens, so the four totals stay additive).
218
386
  case "turn.completed": {
219
387
  const u = p.usage as Record<string, number> | undefined;
220
388
  if (!u) return [];
@@ -223,9 +391,10 @@ export const codexSpec: CliAgentSpec = {
223
391
  inputTokens: u.input_tokens ?? 0,
224
392
  outputTokens: u.output_tokens ?? 0,
225
393
  cacheReadTokens: u.cached_input_tokens ?? 0,
226
- cacheCreationTokens: 0,
394
+ cacheCreationTokens: u.cache_write_input_tokens ?? 0,
227
395
  durationMs: 0,
228
396
  numTurns: 1,
397
+ ...(typeof u.reasoning_output_tokens === "number" ? { reasoningOutputTokens: u.reasoning_output_tokens } : {}),
229
398
  timestamp: ts,
230
399
  }];
231
400
  }
@@ -238,8 +407,8 @@ export const codexSpec: CliAgentSpec = {
238
407
  },
239
408
  };
240
409
 
241
- /** The effort levels codex actually has (`model_reasoning_effort`):
242
- * low|medium|high|xhigh — no "max" (that level is Claude Code's alone). */
410
+ /** The effort levels this runtime passes to codex (`model_reasoning_effort`):
411
+ * low|medium|high|xhigh, the levels every codex model takes. */
243
412
  export type CodexReasoningEffort = Exclude<CliReasoningEffort, "max">;
244
413
 
245
414
  export interface CodexRuntimeConfig {
@@ -3,6 +3,17 @@
3
3
  *
4
4
  * Works against DesktopSandboxProvider — no E2B SDK dependency here.
5
5
  * The action→screenshot loop runs until the model emits text, then yields it.
6
+ *
7
+ * COORDINATE SPACES (incident note). Every screenshot is downscaled to the
8
+ * model's DISPLAY geometry (1024×720) before it is sent, so the model emits
9
+ * coordinates in the SCALED screenshot's space — never the real desktop's.
10
+ * An earlier revision executed those coordinates against the real desktop
11
+ * using hardcoded real-geometry constants, so on any sandbox whose actual
12
+ * geometry differed, every click landed in the wrong place. The provider has
13
+ * no geometry call, so the real geometry is read from each screenshot buffer
14
+ * itself (sharp metadata), and the resulting per-capture `CaptureScale` maps
15
+ * the model's coordinates back onto the exact frame it saw. The old constants
16
+ * survive ONLY as a fallback for the metadata-missing case.
6
17
  */
7
18
 
8
19
  import OpenAI from "openai";
@@ -12,8 +23,11 @@ import { defineRuntime, formatError } from "../index.js";
12
23
 
13
24
  const DISPLAY_WIDTH = 1024;
14
25
  const DISPLAY_HEIGHT = 720;
15
- const DESKTOP_WIDTH = 1280;
16
- const DESKTOP_HEIGHT = 800;
26
+ /** FALLBACK real-desktop geometry — consulted ONLY when sharp cannot read a
27
+ * screenshot's dimensions. The truth is derived per capture from the raw
28
+ * screenshot buffer itself (see `captureScaledScreenshot`). */
29
+ const FALLBACK_DESKTOP_WIDTH = 1280;
30
+ const FALLBACK_DESKTOP_HEIGHT = 800;
17
31
  const MAX_TURNS = 40;
18
32
 
19
33
  export interface OpenAIDesktopRunnerOptions extends RuntimeOptions {
@@ -27,8 +41,9 @@ export interface OpenAIDesktopRunnerOptions extends RuntimeOptions {
27
41
  // so the exchange is described here instead of with SDK types.
28
42
 
29
43
  /** One model-emitted desktop action. Coordinates arrive in DISPLAY (model)
30
- * space and are rescaled to the real desktop before dispatch. */
31
- interface ComputerAction {
44
+ * space and are rescaled through the CURRENT capture's `CaptureScale`
45
+ * before dispatch. Exported so tests can type their fixtures. */
46
+ export interface ComputerAction {
32
47
  type: string;
33
48
  coordinate?: [number, number];
34
49
  startCoordinate?: [number, number];
@@ -99,12 +114,16 @@ export class OpenAIDesktopRunner implements ModelExecutionContract {
99
114
 
100
115
  yield { type: "init", sessionId: "e2b-desktop", timestamp: ts() };
101
116
 
102
- const screenshot = await captureScaledScreenshot(sandbox);
117
+ const first = await captureScaledScreenshot(sandbox);
118
+ // The scale of the LATEST frame the model has seen — its next batch of
119
+ // coordinates is in that frame's space, so every fresh capture below
120
+ // replaces this before the frame is sent back.
121
+ let scale = first.scale;
103
122
 
104
123
  const messages: InputMessage[] = [{
105
124
  role: "user",
106
125
  content: [
107
- { type: "input_image", image_url: `data:image/png;base64,${screenshot}` },
126
+ { type: "input_image", image_url: `data:image/png;base64,${first.b64}` },
108
127
  { type: "input_text", text: opts.prompt },
109
128
  ],
110
129
  }];
@@ -137,7 +156,7 @@ export class OpenAIDesktopRunner implements ModelExecutionContract {
137
156
 
138
157
  for (const call of computerCalls) {
139
158
  console.log(`${label} → ${call.action.type}`);
140
- await executeAction(sandbox, call.action).catch(
159
+ await executeAction(sandbox, call.action, scale).catch(
141
160
  (err: unknown) => console.warn(`${label} Action failed (non-fatal): ${formatError(err)}`),
142
161
  );
143
162
  }
@@ -145,10 +164,11 @@ export class OpenAIDesktopRunner implements ModelExecutionContract {
145
164
  if (computerCalls.length === 0) break;
146
165
 
147
166
  const next = await captureScaledScreenshot(sandbox);
167
+ scale = next.scale;
148
168
  messages.push({ role: "user", content: computerCalls.map((call): InputPart => ({
149
169
  type: "computer_call_output",
150
170
  call_id: call.call_id,
151
- output: { type: "input_image", image_url: `data:image/png;base64,${next}` },
171
+ output: { type: "input_image", image_url: `data:image/png;base64,${next.b64}` },
152
172
  })) });
153
173
  }
154
174
 
@@ -160,21 +180,58 @@ export class OpenAIDesktopRunner implements ModelExecutionContract {
160
180
  // Helpers — work against DesktopSandboxProvider interface
161
181
  // ---------------------------------------------------------------------------
162
182
 
163
- async function captureScaledScreenshot(sandbox: DesktopSandboxProvider): Promise<string> {
183
+ /** Real-desktop pixels per model-DISPLAY pixel, derived from ONE capture.
184
+ * Coordinates the model emits against that capture are multiplied by these
185
+ * factors (and rounded) before touching the provider. */
186
+ export interface CaptureScale {
187
+ scaleX: number;
188
+ scaleY: number;
189
+ }
190
+
191
+ /** One captured frame: the model-facing image plus the scale that maps the
192
+ * model's coordinates back onto this frame's real desktop. */
193
+ export interface ScaledCapture {
194
+ /** Base64 PNG resized to DISPLAY_WIDTH×DISPLAY_HEIGHT (fit "fill"). */
195
+ b64: string;
196
+ scale: CaptureScale;
197
+ }
198
+
199
+ /** Derive a capture's scale from its real pixel dimensions. The constants
200
+ * are a FALLBACK only — used when sharp cannot read the frame's metadata;
201
+ * a real dimension always wins. */
202
+ export function scaleFromDimensions(width: number | undefined, height: number | undefined): CaptureScale {
203
+ return {
204
+ scaleX: (width ?? FALLBACK_DESKTOP_WIDTH) / DISPLAY_WIDTH,
205
+ scaleY: (height ?? FALLBACK_DESKTOP_HEIGHT) / DISPLAY_HEIGHT,
206
+ };
207
+ }
208
+
209
+ /** Capture one frame: read the REAL geometry from the raw screenshot's own
210
+ * metadata, downscale to the model's DISPLAY geometry, and return both the
211
+ * image and the scale that maps model coordinates back onto this frame. */
212
+ export async function captureScaledScreenshot(sandbox: DesktopSandboxProvider): Promise<ScaledCapture> {
164
213
  const raw = await sandbox.screenshot();
165
- const scaled = await sharp(raw)
214
+ const image = sharp(raw);
215
+ const { width, height } = await image.metadata();
216
+ const scaled = await image
166
217
  .resize(DISPLAY_WIDTH, DISPLAY_HEIGHT, { kernel: "lanczos3", fit: "fill" })
167
218
  .png()
168
219
  .toBuffer();
169
- return scaled.toString("base64");
220
+ return { b64: scaled.toString("base64"), scale: scaleFromDimensions(width, height) };
170
221
  }
171
222
 
172
- function scaleX(x: number): number { return Math.round(x * DESKTOP_WIDTH / DISPLAY_WIDTH); }
173
- function scaleY(y: number): number { return Math.round(y * DESKTOP_HEIGHT / DISPLAY_HEIGHT); }
174
-
175
- async function executeAction(sandbox: DesktopSandboxProvider, action: ComputerAction): Promise<void> {
223
+ /** Dispatch one model action against the provider, mapping every coordinate
224
+ * through the CURRENT capture's scale. Non-coordinate payloads — scroll
225
+ * deltas (ticks), typed text, key chords — pass through untouched. */
226
+ export async function executeAction(
227
+ sandbox: DesktopSandboxProvider,
228
+ action: ComputerAction,
229
+ scale: CaptureScale,
230
+ ): Promise<void> {
231
+ const mapX = (x: number) => Math.round(x * scale.scaleX);
232
+ const mapY = (y: number) => Math.round(y * scale.scaleY);
176
233
  const [x, y] = action.coordinate
177
- ? [scaleX(action.coordinate[0]), scaleY(action.coordinate[1])]
234
+ ? [mapX(action.coordinate[0]), mapY(action.coordinate[1])]
178
235
  : [0, 0];
179
236
 
180
237
  switch (action.type) {
@@ -185,12 +242,18 @@ async function executeAction(sandbox: DesktopSandboxProvider, action: ComputerAc
185
242
  case "move": await sandbox.moveMouse(x, y); break;
186
243
  case "type": await sandbox.write(action.text ?? ""); break;
187
244
  case "key": await sandbox.press(action.key ?? ""); break;
188
- case "scroll": await sandbox.scroll(action.direction === "up" ? "up" : "down", action.ticks ?? 3); break;
245
+ case "scroll":
246
+ // The scroll POSITION is a coordinate — mapped (the provider has no
247
+ // positional scroll, so position lands via moveMouse). The scroll
248
+ // DELTA (ticks) is not a coordinate — untouched.
249
+ if (action.coordinate) await sandbox.moveMouse(x, y);
250
+ await sandbox.scroll(action.direction === "up" ? "up" : "down", action.ticks ?? 3);
251
+ break;
189
252
  case "drag":
190
253
  if (action.startCoordinate && action.endCoordinate) {
191
254
  await sandbox.drag(
192
- [scaleX(action.startCoordinate[0]), scaleY(action.startCoordinate[1])],
193
- [scaleX(action.endCoordinate[0]), scaleY(action.endCoordinate[1])],
255
+ [mapX(action.startCoordinate[0]), mapY(action.startCoordinate[1])],
256
+ [mapX(action.endCoordinate[0]), mapY(action.endCoordinate[1])],
194
257
  );
195
258
  }
196
259
  break;