@agent-compose/sdk 0.8.4 → 0.8.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +213 -189
- package/dist/agent/agent-context.d.ts +9 -1
- package/dist/agent/agent-loop.d.ts +14 -6
- package/dist/agent/perf-sampler.d.ts +27 -2
- package/dist/agent/run-agent.d.ts +1 -1
- package/dist/client.d.ts +250 -59
- package/dist/directives.d.ts +14 -0
- package/dist/display.d.ts +7 -0
- package/dist/errors.d.ts +1 -1
- package/dist/generated/agentc-commands.d.ts +34 -0
- package/dist/index.d.ts +13 -11
- package/dist/index.js +1692 -194
- package/dist/request-context/request-context.d.ts +1 -1
- package/dist/runtimes/_cli-agent.d.ts +278 -58
- package/dist/runtimes/claude-code.d.ts +90 -1
- package/dist/runtimes/claude.d.ts +1 -1
- package/dist/runtimes/codex.d.ts +94 -6
- package/dist/runtimes/codex.mid-turn-hook.test.d.ts +10 -0
- package/dist/runtimes/openai-desktop.d.ts +50 -0
- package/dist/runtimes/openai-desktop.js +1689 -211
- package/dist/runtimes/openai-desktop.test.d.ts +20 -0
- package/dist/runtimes/opencode.d.ts +48 -11
- package/dist/runtimes/opencode.test.d.ts +14 -0
- package/dist/runtimes/tool-pulse.test.d.ts +17 -0
- package/dist/sandbox/baked-clis.d.ts +75 -0
- package/dist/sandbox/devbox.d.ts +5 -5
- package/dist/sandbox/exec-stream.d.ts +1 -2
- package/dist/sandbox/network-policy.d.ts +23 -5
- package/dist/sandbox/registry.d.ts +12 -0
- package/dist/sandbox/sizes.d.ts +11 -5
- package/dist/sandbox.d.ts +5 -3
- package/dist/step-invocation/protocol.d.ts +3 -4
- package/dist/step-invocation/server.d.ts +2 -2
- package/dist/step-invocation/types.d.ts +2 -2
- package/dist/types/api-conversations.d.ts +513 -27
- package/dist/types/api-factory.d.ts +183 -3
- package/dist/types/api-projects.d.ts +480 -0
- package/dist/types/api-runs.d.ts +8 -0
- package/dist/types/api-scopes.d.ts +32 -3
- package/dist/types/conversation-stream.d.ts +27 -1
- package/dist/types/execution-context.d.ts +1 -1
- package/dist/types/protocol.d.ts +182 -2
- package/dist/types/runtime.d.ts +80 -2
- package/dist/types/workflow-metadata.d.ts +2 -4
- package/dist/types/workflow-plan.d.ts +1 -3
- package/dist/utils/bundler.d.ts +23 -0
- package/dist/workflow-steps/observability.d.ts +2 -3
- package/dist/workflow-steps/runner.d.ts +5 -8
- package/dist/workflow-steps/types.d.ts +8 -10
- package/dist/workflow-steps/workflow.d.ts +2 -1
- package/dist/workflows/engine.d.ts +3 -5
- package/dist/workflows/invoke-child.d.ts +2 -2
- package/package.json +2 -2
- package/src/agent/agent-context.ts +193 -116
- package/src/agent/agent-loop.ts +16 -9
- package/src/agent/desktop-open.ts +13 -1
- package/src/agent/perf-sampler.ts +54 -3
- package/src/agent/run-agent.ts +1 -1
- package/src/client.ts +418 -80
- package/src/directives.ts +21 -1
- package/src/display.ts +12 -0
- package/src/errors.ts +1 -0
- package/src/generated/agentc-commands.ts +571 -0
- package/src/index.ts +65 -18
- package/src/pause/pause-core.ts +2 -1
- package/src/request-context/request-context.ts +1 -1
- package/src/runtimes/_cli-agent.ts +607 -132
- package/src/runtimes/claude-code.ts +427 -20
- package/src/runtimes/claude.ts +1 -1
- package/src/runtimes/codex.ts +188 -19
- package/src/runtimes/openai-desktop.ts +82 -19
- package/src/runtimes/opencode.ts +195 -26
- package/src/sandbox/baked-clis.ts +86 -0
- package/src/sandbox/devbox.ts +5 -5
- package/src/sandbox/exec-stream.ts +1 -2
- package/src/sandbox/network-policy.ts +51 -7
- package/src/sandbox/providers/e2b.ts +63 -19
- package/src/sandbox/providers/vercel.ts +6 -6
- package/src/sandbox/registry.ts +19 -1
- package/src/sandbox/sizes.ts +11 -5
- package/src/sandbox.ts +9 -2
- package/src/step-invocation/invoker.ts +2 -6
- package/src/step-invocation/protocol.ts +3 -4
- package/src/step-invocation/server.ts +2 -2
- package/src/types/api-conversations.ts +424 -29
- package/src/types/api-factory.ts +189 -3
- package/src/types/api-projects.ts +443 -0
- package/src/types/api-runs.ts +5 -0
- package/src/types/api-scopes.ts +32 -3
- package/src/types/conversation-stream.ts +29 -1
- package/src/types/execution-context.ts +1 -1
- package/src/types/protocol.ts +180 -2
- package/src/types/runtime.ts +71 -2
- package/src/types/sandbox-environment.ts +1 -2
- package/src/types/workflow-metadata.ts +2 -4
- package/src/types/workflow-plan.ts +1 -3
- package/src/utils/bundler.ts +88 -19
- package/src/workflow-steps/observability.ts +2 -3
- package/src/workflow-steps/runner.ts +5 -8
- package/src/workflow-steps/types.ts +8 -10
- package/src/workflow-steps/workflow.ts +2 -1
- package/src/workflows/engine.ts +3 -5
- package/src/workflows/invoke-child.ts +2 -2
- package/dist/pause/__tests__/errors.test.d.ts +0 -1
- package/dist/pause/__tests__/wrappers.test.d.ts +0 -1
- package/dist/step-invocation/__tests__/protocol.test.d.ts +0 -1
package/src/runtimes/codex.ts
CHANGED
|
@@ -5,17 +5,28 @@
|
|
|
5
5
|
* only stream-parse what it prints.
|
|
6
6
|
*
|
|
7
7
|
* Auth: set `OPENAI_API_KEY` (or `CODEX_API_KEY`) in the sandbox env via a
|
|
8
|
-
* workflow secret. The
|
|
9
|
-
*
|
|
10
|
-
*
|
|
8
|
+
* workflow secret. The E2B session image bakes the pinned `codex` CLI
|
|
9
|
+
* (`@openai/codex@CODEX_CLI_VERSION`); on any other machine the runtime
|
|
10
|
+
* installs that same version on demand (pair with
|
|
11
|
+
* `snapshots: { bootFrom: "reuse" }` to install once and boot from the
|
|
12
|
+
* captured snapshot on every run after).
|
|
11
13
|
*
|
|
12
14
|
* Verified against codex-cli 0.124.0: `codex exec --json` + resume-by-thread,
|
|
13
|
-
* with the command_execution / reasoning / agent_message item shapes below
|
|
15
|
+
* with the command_execution / reasoning / agent_message item shapes below;
|
|
16
|
+
* the plan's `todo_list` item against codex-rs rust-v0.153.4 and
|
|
17
|
+
* rust-v0.159.2 (see codexPlanMessages); the command_execution /
|
|
18
|
+
* agent_message / turn.completed shapes re-verified live against 0.160.0
|
|
19
|
+
* (the pin), whose plan-tool sources are byte-identical to rust-v0.159.2.
|
|
14
20
|
*/
|
|
15
21
|
|
|
16
22
|
import type { AgentMessage } from "../index.js";
|
|
17
|
-
import
|
|
23
|
+
import type { AgentMessagePlan } from "../types/protocol.js";
|
|
24
|
+
import {
|
|
25
|
+
MID_TURN_DELIVERED_ENV, MID_TURN_INBOX_ENV, createCliAgentRuntime, shellQuote,
|
|
26
|
+
type CliAgentSpec, type CliReasoningEffort,
|
|
27
|
+
} from "./_cli-agent.js";
|
|
18
28
|
import { formatError } from "../utils/errors.js";
|
|
29
|
+
import { CODEX_CLI_VERSION } from "../sandbox/baked-clis.js";
|
|
19
30
|
|
|
20
31
|
function now(): string { return new Date().toISOString(); }
|
|
21
32
|
|
|
@@ -24,6 +35,97 @@ function now(): string { return new Date().toISOString(); }
|
|
|
24
35
|
* overrides the bundled `@openai/codex` the adapter drives underneath. */
|
|
25
36
|
const CODEX_ACP_ADAPTER = "@agentclientprotocol/codex-acp@0.1.0";
|
|
26
37
|
|
|
38
|
+
/** The Bash PreToolUse hook rtk installs for Codex (`rtk init -g --codex`
|
|
39
|
+
* writes exactly this entry — matcher `Bash`, command `rtk hook codex` —
|
|
40
|
+
* into $CODEX_HOME/hooks.json). `rtk hook codex` exists since rtk 0.50.0;
|
|
41
|
+
* the image bakes RTK_VERSION (sandbox/baked-clis.ts), which this payload
|
|
42
|
+
* was run against. Codex's shell tool matches the `Bash` matcher (codex-rs
|
|
43
|
+
* rust-v0.160.0, the pinned CODEX_CLI_VERSION; identical at
|
|
44
|
+
* rust-v0.159.2), and the hook runs under `$SHELL -lc` with the PreToolUse
|
|
45
|
+
* JSON on stdin. `rtk hook codex` answers `permissionDecision: "allow"` +
|
|
46
|
+
* `updatedInput` rewriting `git status` to `rtk git status` when it has a
|
|
47
|
+
* filter for the command; Codex applies the replacement before its own
|
|
48
|
+
* approval and sandbox checks. For a command it cannot compress (an
|
|
49
|
+
* unknown tool, a pipe into one, substitutions, heredocs, redirects, a
|
|
50
|
+
* `RTK_DISABLED=1` prefix, an unknown permission mode) it prints nothing
|
|
51
|
+
* and exits 0, and Codex runs the original unchanged.
|
|
52
|
+
*
|
|
53
|
+
* The wrapper fails OPEN both ways, like RTK_BASH_HOOK_COMMAND
|
|
54
|
+
* (claude-code.ts). `command -v` covers an image without rtk (the devbox,
|
|
55
|
+
* a bare Vercel VM): silent, exit 0, so Codex runs the command instead of
|
|
56
|
+
* failing every shell call's hook with 127. The trailing `exit 0` covers
|
|
57
|
+
* an rtk that cannot answer: Codex treats a PreToolUse hook's exit 2 with
|
|
58
|
+
* stderr as a BLOCK of the tool call (codex-rs hooks/src/events/
|
|
59
|
+
* pre_tool_use.rs at rust-v0.160.0 — the model reads "Command blocked by
|
|
60
|
+
* PreToolUse hook: <stderr>"), and clap exits 2 with its usage on stderr
|
|
61
|
+
* for a subcommand it does not know. That was 2026-10-03: the image's rtk
|
|
62
|
+
* was a cache-served 0.45.0 with no `hook codex`, and this hook — `exec`
|
|
63
|
+
* handing rtk's exit code to Codex — blocked every shell command of every
|
|
64
|
+
* codex session. rtk's hooks never exit non-zero on purpose (they fail
|
|
65
|
+
* open with no stdout), so a non-zero exit is always a broken rtk, and the
|
|
66
|
+
* compressor must never cost the worker its shell: the exit code is
|
|
67
|
+
* dropped, and the smoke gate's codex-rtk-hook-rewrite check is what
|
|
68
|
+
* proves the rewrite itself. */
|
|
69
|
+
export const RTK_CODEX_HOOK_COMMAND =
|
|
70
|
+
"command -v rtk >/dev/null 2>&1 && rtk hook codex; exit 0";
|
|
71
|
+
|
|
72
|
+
/** The awk program of the mid-turn hook below: the inbox lines not yet fed
|
|
73
|
+
* (JSON string literals, one per line — `codexSpec.midTurnInput.messageLine`)
|
|
74
|
+
* become ONE PostToolUse hook output. Each literal's quotes are stripped and
|
|
75
|
+
* the bodies are joined with an escaped blank line; the bodies are already
|
|
76
|
+
* JSON-escaped, so the result is one valid JSON string. Nothing is printed
|
|
77
|
+
* when no line is due, and codex ignores an empty stdout. */
|
|
78
|
+
const CODEX_MID_TURN_HOOK_AWK = String.raw`NR > fed && NR <= upto { s = substr($0, 2, length($0) - 2); out = (out == "" ? s : out "\\n\\n" s) } END { if (out != "") printf "{\"hookSpecificOutput\":{\"hookEventName\":\"PostToolUse\",\"additionalContext\":\"%s\"}}\n", out }`;
|
|
79
|
+
|
|
80
|
+
/** The PostToolUse command hook that carries a mid-turn message into a
|
|
81
|
+
* RUNNING codex turn (the 2026-10-02 relayed-steer incident: three of the
|
|
82
|
+
* owner's instructions waited 48-51 minutes for a codex build to end).
|
|
83
|
+
* Codex runs it, under `$SHELL -lc` with the event JSON on stdin and the
|
|
84
|
+
* codex process's own environment, after every tool call it completes
|
|
85
|
+
* (codex-rs core/src/tools/registry.rs → hook_runtime.rs at rust-v0.160.0,
|
|
86
|
+
* the pin). The launch wrapper exported the turn's inbox and delivered-
|
|
87
|
+
* counter paths into that environment (sdk _cli-agent.ts
|
|
88
|
+
* toolHookInboxFragment); the hook forwards the inbox lines past the
|
|
89
|
+
* counter as the hook's `additionalContext`, which codex records as
|
|
90
|
+
* developer context in the live turn's history before the model's next
|
|
91
|
+
* request (hook_runtime.rs record_additional_contexts), then advances the
|
|
92
|
+
* counter — the same ack the server's inject lane trusts for the stdin
|
|
93
|
+
* lane. Not a platform turn (no inbox in the environment): silent, exit 0.
|
|
94
|
+
* Codex's default spill threshold for a hook's additional context is 2,500
|
|
95
|
+
* tokens (hooks/src/output_spill.rs); a longer message reaches the model
|
|
96
|
+
* as a preview plus a file pointer, which a steer never is. */
|
|
97
|
+
export const CODEX_MID_TURN_HOOK_COMMAND =
|
|
98
|
+
// Drain the event JSON first, so codex's stdin write never meets a closed pipe.
|
|
99
|
+
"cat >/dev/null; "
|
|
100
|
+
+ `[ -n "\${${MID_TURN_INBOX_ENV}:-}" ] && [ -f "$${MID_TURN_INBOX_ENV}" ] || exit 0; `
|
|
101
|
+
+ `ac_fed=$(cat "$${MID_TURN_DELIVERED_ENV}" 2>/dev/null); ac_fed=\${ac_fed:-0}; `
|
|
102
|
+
+ `ac_lines=$(wc -l < "$${MID_TURN_INBOX_ENV}" 2>/dev/null); ac_lines=\${ac_lines:-0}; `
|
|
103
|
+
+ `[ "$ac_lines" -gt "$ac_fed" ] || exit 0; `
|
|
104
|
+
+ `awk -v fed="$ac_fed" -v upto="$ac_lines" '${CODEX_MID_TURN_HOOK_AWK}' "$${MID_TURN_INBOX_ENV}" `
|
|
105
|
+
+ `&& echo "$ac_lines" > "$${MID_TURN_DELIVERED_ENV}"; exit 0`;
|
|
106
|
+
|
|
107
|
+
/** $CODEX_HOME/hooks.json for every platform Codex session (the server
|
|
108
|
+
* writes it at boot, session-runtime-config.ts). Codex only RUNS a
|
|
109
|
+
* user-level hook it has persisted trust for — the TUI's review prompt
|
|
110
|
+
* has no headless counterpart, and `--dangerously-bypass-hook-trust` would
|
|
111
|
+
* run a cloned repository's `.codex/hooks.json` unreviewed too — so the
|
|
112
|
+
* server also writes the trust record for each of these exact hooks into
|
|
113
|
+
* the config.toml it generates (sandbox/codex-hooks.ts). Verified live
|
|
114
|
+
* against codex 0.159.2 and 0.160.0 for the rtk hook: with the record,
|
|
115
|
+
* `codex exec` rewrote `git status` through rtk with no bypass flag;
|
|
116
|
+
* without it, the hook was skipped.
|
|
117
|
+
*
|
|
118
|
+
* The PostToolUse group carries NO matcher: codex runs a matcher-less hook
|
|
119
|
+
* after every tool it completes (hooks/src/events/common.rs
|
|
120
|
+
* matches_matcher: an absent matcher is a match), so a mid-turn message
|
|
121
|
+
* lands at the next tool step whatever the tool was. */
|
|
122
|
+
export const CODEX_PLATFORM_HOOKS = {
|
|
123
|
+
hooks: {
|
|
124
|
+
PreToolUse: [{ matcher: "Bash", hooks: [{ type: "command", command: RTK_CODEX_HOOK_COMMAND }] }],
|
|
125
|
+
PostToolUse: [{ hooks: [{ type: "command", command: CODEX_MID_TURN_HOOK_COMMAND }] }],
|
|
126
|
+
},
|
|
127
|
+
} as const;
|
|
128
|
+
|
|
27
129
|
/** Known-noise codex ADVISORY lines. codex emits these as `item.completed`
|
|
28
130
|
* error items on the `--json` stream (exec maps every `Warning` notification
|
|
29
131
|
* to an error item — verified against codex-cli 0.147.0), so without a filter
|
|
@@ -101,6 +203,41 @@ export function codexWriterLockPreflight(
|
|
|
101
203
|
+ `flock -n "$ac_lk" rm -f -- "$ac_lk" 2>/dev/null || true; fi; `;
|
|
102
204
|
}
|
|
103
205
|
|
|
206
|
+
/**
|
|
207
|
+
* codex's PLAN (its update_plan tool) on the `--json` stream: one
|
|
208
|
+
* `todo_list` item per turn, `item.started` on the plan's first update,
|
|
209
|
+
* `item.updated` (same id, the whole list) on every later one, and
|
|
210
|
+
* `item.completed` at the turn's end. That is codex-rs exec's
|
|
211
|
+
* event_processor_with_jsonl_output.rs, identical in rust-v0.153.4 and
|
|
212
|
+
* rust-v0.159.2, where it is the only `item.updated` codex emits; the item
|
|
213
|
+
* `{ id, type: "todo_list", items: [{ text, completed }] }` is the shape every
|
|
214
|
+
* todo_list item our machines have persisted carries. Each event maps to ONE
|
|
215
|
+
* whole-plan `plan` message, so the transcript's checklist, and the worker's
|
|
216
|
+
* status line, follow the plan while the turn runs (before this, the list
|
|
217
|
+
* surfaced once, at the turn's end, as a raw `todo_list` tool card).
|
|
218
|
+
*
|
|
219
|
+
* A step carries only `completed`: codex's `pending` and `in_progress` both
|
|
220
|
+
* serialize as false. Its plan tool allows at most one step in progress and
|
|
221
|
+
* a plan runs in order, so the first step not completed is the one under
|
|
222
|
+
* way; the rest are pending. The stream names no priority, so none is set.
|
|
223
|
+
*/
|
|
224
|
+
function codexPlanMessages(item: Record<string, unknown>, timestamp: string): AgentMessage[] {
|
|
225
|
+
const steps = Array.isArray(item.items) ? item.items : [];
|
|
226
|
+
let underWay = false;
|
|
227
|
+
const entries: AgentMessagePlan["entries"] = [];
|
|
228
|
+
for (const step of steps) {
|
|
229
|
+
if (typeof step !== "object" || step === null) continue;
|
|
230
|
+
const text = (step as { text?: unknown }).text;
|
|
231
|
+
if (typeof text !== "string" || text.trim().length === 0) continue;
|
|
232
|
+
let status: AgentMessagePlan["entries"][number]["status"];
|
|
233
|
+
if ((step as { completed?: unknown }).completed === true) status = "completed";
|
|
234
|
+
else if (!underWay) { status = "in_progress"; underWay = true; }
|
|
235
|
+
else status = "pending";
|
|
236
|
+
entries.push({ content: text.trim(), status });
|
|
237
|
+
}
|
|
238
|
+
return entries.length > 0 ? [{ type: "plan", entries, timestamp }] : [];
|
|
239
|
+
}
|
|
240
|
+
|
|
104
241
|
/** Exported for the fixture-based parity tests (ADR-0020): the legacy JSONL
|
|
105
242
|
* `mapEvent` is the golden the ACP normaliser is asserted equal to. Not part
|
|
106
243
|
* of the public runtime surface — `createCodexRuntime` stays the entry point. */
|
|
@@ -113,6 +250,22 @@ export const codexSpec: CliAgentSpec = {
|
|
|
113
250
|
// every resume of the same thread, so supersede teardown must KILL it —
|
|
114
251
|
// never detach it alive (server runner-kill.ts consumes this flag).
|
|
115
252
|
exclusiveSessionWriter: true,
|
|
253
|
+
// Mid-turn input (the 2026-10-02 relayed-steer incident). codex reads
|
|
254
|
+
// stdin ONCE, as the prompt: `codex exec -` submits one UserTurn and never
|
|
255
|
+
// reads stdin again (codex-rs exec/src/lib.rs read_prompt_from_stdin and
|
|
256
|
+
// its single InitialOperation::UserTurn at rust-v0.160.0; the app-server's
|
|
257
|
+
// `turn/steer` is a different transport, not `codex exec`). So a running
|
|
258
|
+
// turn takes a message through the PostToolUse hook above: codex runs it
|
|
259
|
+
// after every tool call it completes, with the codex process's own
|
|
260
|
+
// environment; the hook prints the unfed inbox lines as `additionalContext`
|
|
261
|
+
// and codex records them as developer context in the live turn's history
|
|
262
|
+
// before the model's next request. One JSON string literal per inbox line,
|
|
263
|
+
// so the hook splices bodies without decoding. Codex builds the PostToolUse
|
|
264
|
+
// payload only for a tool call it counts as successful (tools/registry.rs),
|
|
265
|
+
// so a message lands at the next successful tool step. Source-verified at
|
|
266
|
+
// rust-v0.160.0 (the pinned CODEX_CLI_VERSION); the end-to-end run against
|
|
267
|
+
// a live codex is the acceptance check still owed.
|
|
268
|
+
midTurnInput: { transport: "tool-hook", messageLine: (text) => JSON.stringify(text) },
|
|
116
269
|
// ACP-mode invocation (ADR-0020 increment 1). `codex` has no native `--acp`
|
|
117
270
|
// flag; the adapter (a Rust binary shipped via npm) speaks ACP and drives a
|
|
118
271
|
// compatible bundled `@openai/codex`. The runner attempts this first and
|
|
@@ -125,9 +278,11 @@ export const codexSpec: CliAgentSpec = {
|
|
|
125
278
|
// (e.g. a sandbox-baked one) instead of the adapter's bundled copy.
|
|
126
279
|
...(process.env.CODEX_PATH ? { env: { CODEX_PATH: process.env.CODEX_PATH } } : {}),
|
|
127
280
|
},
|
|
128
|
-
//
|
|
129
|
-
//
|
|
130
|
-
|
|
281
|
+
// The E2B session image bakes this exact version (sandbox/baked-clis.ts),
|
|
282
|
+
// so this runs only on a machine whose image predates the bake: a global
|
|
283
|
+
// npm install of the SAME pin, symlinked onto PATH only if the global bin
|
|
284
|
+
// dir isn't already there (so a non-login `sh -c` can find it).
|
|
285
|
+
install: `sudo npm install -g @openai/codex@${CODEX_CLI_VERSION} && (command -v codex >/dev/null 2>&1 || sudo ln -sf "$(npm prefix -g)/bin/codex" /usr/local/bin/codex)`,
|
|
131
286
|
// Codex reads the prompt from stdin when invoked as `codex exec ... -`.
|
|
132
287
|
promptPayload: (prompt) => prompt,
|
|
133
288
|
buildCommand: ({ promptPath, sessionId, model, cwd, effort }) => {
|
|
@@ -140,13 +295,13 @@ export const codexSpec: CliAgentSpec = {
|
|
|
140
295
|
// for running in environments that are externally sandboxed".
|
|
141
296
|
"--dangerously-bypass-approvals-and-sandbox",
|
|
142
297
|
...(model ? ["-m", shellQuote(model)] : []),
|
|
143
|
-
// codex's own reasoning knob — a config override
|
|
144
|
-
//
|
|
145
|
-
//
|
|
146
|
-
//
|
|
147
|
-
//
|
|
148
|
-
//
|
|
149
|
-
//
|
|
298
|
+
// codex's own reasoning knob — a config override. Every codex model
|
|
299
|
+
// takes low|medium|high|xhigh; only the GPT-6 and GPT-5.6 presets add
|
|
300
|
+
// max (codex 0.153+), so the platform offers codex up to xhigh: the
|
|
301
|
+
// server rejects "max" for codex sessions (sessionEffortLockError), and
|
|
302
|
+
// this clamp to xhigh is the type-level backstop for a caller that
|
|
303
|
+
// bypasses that gate. The value comes from the closed
|
|
304
|
+
// CliReasoningEffort set, so it is shell-safe unquoted.
|
|
150
305
|
...(effort ? ["-c", `model_reasoning_effort=${effort === "max" ? "xhigh" : effort}`] : []),
|
|
151
306
|
].join(" ");
|
|
152
307
|
// Working directory via a shell `cd` — the idiom EVERY other runtime uses
|
|
@@ -172,11 +327,19 @@ export const codexSpec: CliAgentSpec = {
|
|
|
172
327
|
mapEvent: (p): AgentMessage[] => {
|
|
173
328
|
const ts = now();
|
|
174
329
|
switch (p.type) {
|
|
330
|
+
case "item.updated": {
|
|
331
|
+
// Only the plan updates in place (codexPlanMessages); any other
|
|
332
|
+
// item surfaces on its started / completed events below.
|
|
333
|
+
const item = p.item as Record<string, unknown> | undefined;
|
|
334
|
+
return item?.type === "todo_list" ? codexPlanMessages(item, ts) : [];
|
|
335
|
+
}
|
|
175
336
|
case "item.started":
|
|
176
337
|
case "item.completed": {
|
|
177
338
|
const item = p.item as Record<string, unknown> | undefined;
|
|
178
339
|
if (!item) return [];
|
|
179
340
|
const itype = String(item.type ?? "");
|
|
341
|
+
// The plan, whole, on its first update and at the turn's end.
|
|
342
|
+
if (itype === "todo_list") return codexPlanMessages(item, ts);
|
|
180
343
|
// Text + reasoning land on completion (started carries no final text).
|
|
181
344
|
if (itype === "agent_message") {
|
|
182
345
|
return p.type === "item.completed" ? [{ type: "text", text: String(item.text ?? ""), timestamp: ts }] : [];
|
|
@@ -209,12 +372,17 @@ export const codexSpec: CliAgentSpec = {
|
|
|
209
372
|
}
|
|
210
373
|
return [{ type: "error", text, timestamp: ts }];
|
|
211
374
|
}
|
|
212
|
-
// file_change / mcp_tool_call / web_search
|
|
375
|
+
// file_change / mcp_tool_call / web_search: surface once, on completion.
|
|
213
376
|
if (p.type === "item.completed") {
|
|
214
377
|
return [{ type: "tool_use", toolName: itype || "item", toolInput: item, toolUseId: String(item.id ?? ""), timestamp: ts }];
|
|
215
378
|
}
|
|
216
379
|
return [];
|
|
217
380
|
}
|
|
381
|
+
// The turn's token classes as codex reports them (codex-rs
|
|
382
|
+
// exec_events.rs `Usage`): input, cached input, cache-write input,
|
|
383
|
+
// output and reasoning output. Every class is kept — cache writes land
|
|
384
|
+
// in the creation class, reasoning rides beside output (it is counted
|
|
385
|
+
// inside output_tokens, so the four totals stay additive).
|
|
218
386
|
case "turn.completed": {
|
|
219
387
|
const u = p.usage as Record<string, number> | undefined;
|
|
220
388
|
if (!u) return [];
|
|
@@ -223,9 +391,10 @@ export const codexSpec: CliAgentSpec = {
|
|
|
223
391
|
inputTokens: u.input_tokens ?? 0,
|
|
224
392
|
outputTokens: u.output_tokens ?? 0,
|
|
225
393
|
cacheReadTokens: u.cached_input_tokens ?? 0,
|
|
226
|
-
cacheCreationTokens: 0,
|
|
394
|
+
cacheCreationTokens: u.cache_write_input_tokens ?? 0,
|
|
227
395
|
durationMs: 0,
|
|
228
396
|
numTurns: 1,
|
|
397
|
+
...(typeof u.reasoning_output_tokens === "number" ? { reasoningOutputTokens: u.reasoning_output_tokens } : {}),
|
|
229
398
|
timestamp: ts,
|
|
230
399
|
}];
|
|
231
400
|
}
|
|
@@ -238,8 +407,8 @@ export const codexSpec: CliAgentSpec = {
|
|
|
238
407
|
},
|
|
239
408
|
};
|
|
240
409
|
|
|
241
|
-
/** The effort levels
|
|
242
|
-
* low|medium|high|xhigh
|
|
410
|
+
/** The effort levels this runtime passes to codex (`model_reasoning_effort`):
|
|
411
|
+
* low|medium|high|xhigh, the levels every codex model takes. */
|
|
243
412
|
export type CodexReasoningEffort = Exclude<CliReasoningEffort, "max">;
|
|
244
413
|
|
|
245
414
|
export interface CodexRuntimeConfig {
|
|
@@ -3,6 +3,17 @@
|
|
|
3
3
|
*
|
|
4
4
|
* Works against DesktopSandboxProvider — no E2B SDK dependency here.
|
|
5
5
|
* The action→screenshot loop runs until the model emits text, then yields it.
|
|
6
|
+
*
|
|
7
|
+
* COORDINATE SPACES (incident note). Every screenshot is downscaled to the
|
|
8
|
+
* model's DISPLAY geometry (1024×720) before it is sent, so the model emits
|
|
9
|
+
* coordinates in the SCALED screenshot's space — never the real desktop's.
|
|
10
|
+
* An earlier revision executed those coordinates against the real desktop
|
|
11
|
+
* using hardcoded real-geometry constants, so on any sandbox whose actual
|
|
12
|
+
* geometry differed, every click landed in the wrong place. The provider has
|
|
13
|
+
* no geometry call, so the real geometry is read from each screenshot buffer
|
|
14
|
+
* itself (sharp metadata), and the resulting per-capture `CaptureScale` maps
|
|
15
|
+
* the model's coordinates back onto the exact frame it saw. The old constants
|
|
16
|
+
* survive ONLY as a fallback for the metadata-missing case.
|
|
6
17
|
*/
|
|
7
18
|
|
|
8
19
|
import OpenAI from "openai";
|
|
@@ -12,8 +23,11 @@ import { defineRuntime, formatError } from "../index.js";
|
|
|
12
23
|
|
|
13
24
|
const DISPLAY_WIDTH = 1024;
|
|
14
25
|
const DISPLAY_HEIGHT = 720;
|
|
15
|
-
|
|
16
|
-
|
|
26
|
+
/** FALLBACK real-desktop geometry — consulted ONLY when sharp cannot read a
|
|
27
|
+
* screenshot's dimensions. The truth is derived per capture from the raw
|
|
28
|
+
* screenshot buffer itself (see `captureScaledScreenshot`). */
|
|
29
|
+
const FALLBACK_DESKTOP_WIDTH = 1280;
|
|
30
|
+
const FALLBACK_DESKTOP_HEIGHT = 800;
|
|
17
31
|
const MAX_TURNS = 40;
|
|
18
32
|
|
|
19
33
|
export interface OpenAIDesktopRunnerOptions extends RuntimeOptions {
|
|
@@ -27,8 +41,9 @@ export interface OpenAIDesktopRunnerOptions extends RuntimeOptions {
|
|
|
27
41
|
// so the exchange is described here instead of with SDK types.
|
|
28
42
|
|
|
29
43
|
/** One model-emitted desktop action. Coordinates arrive in DISPLAY (model)
|
|
30
|
-
* space and are rescaled
|
|
31
|
-
|
|
44
|
+
* space and are rescaled through the CURRENT capture's `CaptureScale`
|
|
45
|
+
* before dispatch. Exported so tests can type their fixtures. */
|
|
46
|
+
export interface ComputerAction {
|
|
32
47
|
type: string;
|
|
33
48
|
coordinate?: [number, number];
|
|
34
49
|
startCoordinate?: [number, number];
|
|
@@ -99,12 +114,16 @@ export class OpenAIDesktopRunner implements ModelExecutionContract {
|
|
|
99
114
|
|
|
100
115
|
yield { type: "init", sessionId: "e2b-desktop", timestamp: ts() };
|
|
101
116
|
|
|
102
|
-
const
|
|
117
|
+
const first = await captureScaledScreenshot(sandbox);
|
|
118
|
+
// The scale of the LATEST frame the model has seen — its next batch of
|
|
119
|
+
// coordinates is in that frame's space, so every fresh capture below
|
|
120
|
+
// replaces this before the frame is sent back.
|
|
121
|
+
let scale = first.scale;
|
|
103
122
|
|
|
104
123
|
const messages: InputMessage[] = [{
|
|
105
124
|
role: "user",
|
|
106
125
|
content: [
|
|
107
|
-
{ type: "input_image", image_url: `data:image/png;base64,${
|
|
126
|
+
{ type: "input_image", image_url: `data:image/png;base64,${first.b64}` },
|
|
108
127
|
{ type: "input_text", text: opts.prompt },
|
|
109
128
|
],
|
|
110
129
|
}];
|
|
@@ -137,7 +156,7 @@ export class OpenAIDesktopRunner implements ModelExecutionContract {
|
|
|
137
156
|
|
|
138
157
|
for (const call of computerCalls) {
|
|
139
158
|
console.log(`${label} → ${call.action.type}`);
|
|
140
|
-
await executeAction(sandbox, call.action).catch(
|
|
159
|
+
await executeAction(sandbox, call.action, scale).catch(
|
|
141
160
|
(err: unknown) => console.warn(`${label} Action failed (non-fatal): ${formatError(err)}`),
|
|
142
161
|
);
|
|
143
162
|
}
|
|
@@ -145,10 +164,11 @@ export class OpenAIDesktopRunner implements ModelExecutionContract {
|
|
|
145
164
|
if (computerCalls.length === 0) break;
|
|
146
165
|
|
|
147
166
|
const next = await captureScaledScreenshot(sandbox);
|
|
167
|
+
scale = next.scale;
|
|
148
168
|
messages.push({ role: "user", content: computerCalls.map((call): InputPart => ({
|
|
149
169
|
type: "computer_call_output",
|
|
150
170
|
call_id: call.call_id,
|
|
151
|
-
output: { type: "input_image", image_url: `data:image/png;base64,${next}` },
|
|
171
|
+
output: { type: "input_image", image_url: `data:image/png;base64,${next.b64}` },
|
|
152
172
|
})) });
|
|
153
173
|
}
|
|
154
174
|
|
|
@@ -160,21 +180,58 @@ export class OpenAIDesktopRunner implements ModelExecutionContract {
|
|
|
160
180
|
// Helpers — work against DesktopSandboxProvider interface
|
|
161
181
|
// ---------------------------------------------------------------------------
|
|
162
182
|
|
|
163
|
-
|
|
183
|
+
/** Real-desktop pixels per model-DISPLAY pixel, derived from ONE capture.
|
|
184
|
+
* Coordinates the model emits against that capture are multiplied by these
|
|
185
|
+
* factors (and rounded) before touching the provider. */
|
|
186
|
+
export interface CaptureScale {
|
|
187
|
+
scaleX: number;
|
|
188
|
+
scaleY: number;
|
|
189
|
+
}
|
|
190
|
+
|
|
191
|
+
/** One captured frame: the model-facing image plus the scale that maps the
|
|
192
|
+
* model's coordinates back onto this frame's real desktop. */
|
|
193
|
+
export interface ScaledCapture {
|
|
194
|
+
/** Base64 PNG resized to DISPLAY_WIDTH×DISPLAY_HEIGHT (fit "fill"). */
|
|
195
|
+
b64: string;
|
|
196
|
+
scale: CaptureScale;
|
|
197
|
+
}
|
|
198
|
+
|
|
199
|
+
/** Derive a capture's scale from its real pixel dimensions. The constants
|
|
200
|
+
* are a FALLBACK only — used when sharp cannot read the frame's metadata;
|
|
201
|
+
* a real dimension always wins. */
|
|
202
|
+
export function scaleFromDimensions(width: number | undefined, height: number | undefined): CaptureScale {
|
|
203
|
+
return {
|
|
204
|
+
scaleX: (width ?? FALLBACK_DESKTOP_WIDTH) / DISPLAY_WIDTH,
|
|
205
|
+
scaleY: (height ?? FALLBACK_DESKTOP_HEIGHT) / DISPLAY_HEIGHT,
|
|
206
|
+
};
|
|
207
|
+
}
|
|
208
|
+
|
|
209
|
+
/** Capture one frame: read the REAL geometry from the raw screenshot's own
|
|
210
|
+
* metadata, downscale to the model's DISPLAY geometry, and return both the
|
|
211
|
+
* image and the scale that maps model coordinates back onto this frame. */
|
|
212
|
+
export async function captureScaledScreenshot(sandbox: DesktopSandboxProvider): Promise<ScaledCapture> {
|
|
164
213
|
const raw = await sandbox.screenshot();
|
|
165
|
-
const
|
|
214
|
+
const image = sharp(raw);
|
|
215
|
+
const { width, height } = await image.metadata();
|
|
216
|
+
const scaled = await image
|
|
166
217
|
.resize(DISPLAY_WIDTH, DISPLAY_HEIGHT, { kernel: "lanczos3", fit: "fill" })
|
|
167
218
|
.png()
|
|
168
219
|
.toBuffer();
|
|
169
|
-
return scaled.toString("base64");
|
|
220
|
+
return { b64: scaled.toString("base64"), scale: scaleFromDimensions(width, height) };
|
|
170
221
|
}
|
|
171
222
|
|
|
172
|
-
|
|
173
|
-
|
|
174
|
-
|
|
175
|
-
async function executeAction(
|
|
223
|
+
/** Dispatch one model action against the provider, mapping every coordinate
|
|
224
|
+
* through the CURRENT capture's scale. Non-coordinate payloads — scroll
|
|
225
|
+
* deltas (ticks), typed text, key chords — pass through untouched. */
|
|
226
|
+
export async function executeAction(
|
|
227
|
+
sandbox: DesktopSandboxProvider,
|
|
228
|
+
action: ComputerAction,
|
|
229
|
+
scale: CaptureScale,
|
|
230
|
+
): Promise<void> {
|
|
231
|
+
const mapX = (x: number) => Math.round(x * scale.scaleX);
|
|
232
|
+
const mapY = (y: number) => Math.round(y * scale.scaleY);
|
|
176
233
|
const [x, y] = action.coordinate
|
|
177
|
-
? [
|
|
234
|
+
? [mapX(action.coordinate[0]), mapY(action.coordinate[1])]
|
|
178
235
|
: [0, 0];
|
|
179
236
|
|
|
180
237
|
switch (action.type) {
|
|
@@ -185,12 +242,18 @@ async function executeAction(sandbox: DesktopSandboxProvider, action: ComputerAc
|
|
|
185
242
|
case "move": await sandbox.moveMouse(x, y); break;
|
|
186
243
|
case "type": await sandbox.write(action.text ?? ""); break;
|
|
187
244
|
case "key": await sandbox.press(action.key ?? ""); break;
|
|
188
|
-
case "scroll":
|
|
245
|
+
case "scroll":
|
|
246
|
+
// The scroll POSITION is a coordinate — mapped (the provider has no
|
|
247
|
+
// positional scroll, so position lands via moveMouse). The scroll
|
|
248
|
+
// DELTA (ticks) is not a coordinate — untouched.
|
|
249
|
+
if (action.coordinate) await sandbox.moveMouse(x, y);
|
|
250
|
+
await sandbox.scroll(action.direction === "up" ? "up" : "down", action.ticks ?? 3);
|
|
251
|
+
break;
|
|
189
252
|
case "drag":
|
|
190
253
|
if (action.startCoordinate && action.endCoordinate) {
|
|
191
254
|
await sandbox.drag(
|
|
192
|
-
[
|
|
193
|
-
[
|
|
255
|
+
[mapX(action.startCoordinate[0]), mapY(action.startCoordinate[1])],
|
|
256
|
+
[mapX(action.endCoordinate[0]), mapY(action.endCoordinate[1])],
|
|
194
257
|
);
|
|
195
258
|
}
|
|
196
259
|
break;
|