@agent-compose/sdk 0.6.0 → 0.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (126) hide show
  1. package/README.md +66 -39
  2. package/dist/agent/__tests__/runtime-json-schema.test.d.ts +10 -0
  3. package/dist/agent/agent-context.d.ts +21 -1
  4. package/dist/agent/agent-loop.d.ts +24 -1
  5. package/dist/client.d.ts +338 -534
  6. package/dist/directives.d.ts +112 -0
  7. package/dist/display.d.ts +242 -0
  8. package/dist/errors.d.ts +24 -1
  9. package/dist/index.d.ts +34 -13
  10. package/dist/index.js +2984 -861
  11. package/dist/pause/wrappers.d.ts +31 -9
  12. package/dist/processors/ask-human.d.ts +30 -0
  13. package/dist/processors/ask-human.test.d.ts +1 -0
  14. package/dist/processors/index.d.ts +1 -0
  15. package/dist/runtimes/_acp-client.d.ts +46 -1
  16. package/dist/runtimes/_cli-agent.d.ts +58 -4
  17. package/dist/runtimes/_jsonl-guard.d.ts +103 -0
  18. package/dist/runtimes/amp.d.ts +2 -2
  19. package/dist/runtimes/claude-code.d.ts +59 -0
  20. package/dist/runtimes/claude-code.test.d.ts +14 -0
  21. package/dist/runtimes/claude.d.ts +16 -0
  22. package/dist/runtimes/claude.test.d.ts +8 -0
  23. package/dist/runtimes/codex.d.ts +9 -3
  24. package/dist/runtimes/cursor.d.ts +9 -0
  25. package/dist/runtimes/droid.d.ts +9 -0
  26. package/dist/runtimes/jsonl-guard.test.d.ts +19 -0
  27. package/dist/runtimes/openai-desktop.js +2922 -861
  28. package/dist/runtimes/opencode.d.ts +25 -0
  29. package/dist/runtimes/vercel.js +22 -1
  30. package/dist/sandbox/devbox.d.ts +42 -0
  31. package/dist/sandbox/exec-stream.d.ts +14 -0
  32. package/dist/sandbox/network-policy.d.ts +100 -0
  33. package/dist/sandbox/provider-def.d.ts +79 -0
  34. package/dist/sandbox/providers/desktop.d.ts +10 -0
  35. package/dist/sandbox/providers/e2b.d.ts +17 -0
  36. package/dist/sandbox/providers/local.d.ts +11 -0
  37. package/dist/sandbox/providers/vercel.d.ts +18 -0
  38. package/dist/sandbox/registry.d.ts +45 -0
  39. package/dist/sandbox/sizes.d.ts +68 -0
  40. package/dist/sandbox.d.ts +24 -299
  41. package/dist/step-invocation/__tests__/foreground-recovery.test.d.ts +1 -0
  42. package/dist/step-invocation/invoker.d.ts +24 -1
  43. package/dist/step-invocation/protocol.d.ts +13 -0
  44. package/dist/types/api-compliance.d.ts +71 -0
  45. package/dist/types/api-conversations.d.ts +492 -0
  46. package/dist/types/api-factory.d.ts +309 -0
  47. package/dist/types/api-projects.d.ts +131 -0
  48. package/dist/types/api-runs.d.ts +377 -0
  49. package/dist/types/api-scopes.d.ts +102 -0
  50. package/dist/types/conversation-stream.d.ts +191 -0
  51. package/dist/types/execution-context.d.ts +12 -2
  52. package/dist/types/protocol.d.ts +30 -1
  53. package/dist/types/sandbox-environment.d.ts +8 -5
  54. package/dist/types/sandbox.d.ts +79 -0
  55. package/dist/types/workflow-metadata.d.ts +33 -8
  56. package/dist/types/workflow-plan.d.ts +10 -0
  57. package/dist/types/workflow.d.ts +18 -193
  58. package/dist/utils/bundler.d.ts +12 -1
  59. package/dist/utils/errors.d.ts +9 -1
  60. package/dist/workflow-steps/index.d.ts +1 -1
  61. package/dist/workflow-steps/observability.d.ts +8 -1
  62. package/dist/workflow-steps/runner.d.ts +3 -3
  63. package/dist/workflow-steps/step.d.ts +15 -1
  64. package/dist/workflow-steps/types.d.ts +19 -5
  65. package/dist/workflow-steps/workflow.d.ts +22 -1
  66. package/dist/workflows/engine.d.ts +3 -2
  67. package/dist/workflows/invoke-child.d.ts +2 -2
  68. package/package.json +1 -1
  69. package/src/agent/agent-context.ts +206 -16
  70. package/src/agent/agent-loop.ts +40 -4
  71. package/src/agent/run-agent.ts +9 -1
  72. package/src/client.ts +909 -621
  73. package/src/directives.ts +184 -0
  74. package/src/display.ts +788 -0
  75. package/src/errors.ts +39 -0
  76. package/src/index.ts +117 -10
  77. package/src/pause/wrappers.ts +44 -9
  78. package/src/processors/ask-human.ts +136 -0
  79. package/src/processors/index.ts +5 -0
  80. package/src/runtimes/_acp-client.ts +72 -3
  81. package/src/runtimes/_cli-agent.ts +171 -38
  82. package/src/runtimes/_jsonl-guard.ts +219 -0
  83. package/src/runtimes/claude-code.ts +246 -0
  84. package/src/runtimes/claude.ts +32 -2
  85. package/src/runtimes/codex.ts +55 -3
  86. package/src/runtimes/cursor.ts +59 -0
  87. package/src/runtimes/droid.ts +63 -0
  88. package/src/runtimes/openai-desktop.ts +59 -14
  89. package/src/runtimes/opencode.ts +61 -0
  90. package/src/sandbox/devbox.ts +48 -0
  91. package/src/sandbox/exec-stream.ts +48 -0
  92. package/src/sandbox/network-policy.ts +181 -0
  93. package/src/sandbox/provider-def.ts +94 -0
  94. package/src/sandbox/providers/desktop.ts +57 -0
  95. package/src/sandbox/providers/e2b.ts +354 -0
  96. package/src/sandbox/providers/local.ts +106 -0
  97. package/src/sandbox/providers/vercel.ts +331 -0
  98. package/src/sandbox/registry.ts +198 -0
  99. package/src/sandbox/sizes.ts +95 -0
  100. package/src/sandbox.ts +59 -1263
  101. package/src/step-invocation/invoker.ts +319 -34
  102. package/src/step-invocation/protocol.ts +19 -0
  103. package/src/types/api-compliance.ts +79 -0
  104. package/src/types/api-conversations.ts +522 -0
  105. package/src/types/api-factory.ts +336 -0
  106. package/src/types/api-projects.ts +140 -0
  107. package/src/types/api-runs.ts +412 -0
  108. package/src/types/api-scopes.ts +102 -0
  109. package/src/types/conversation-stream.ts +231 -0
  110. package/src/types/execution-context.ts +10 -2
  111. package/src/types/protocol.ts +33 -0
  112. package/src/types/sandbox-environment.ts +28 -9
  113. package/src/types/sandbox.ts +78 -0
  114. package/src/types/workflow-metadata.ts +35 -8
  115. package/src/types/workflow-plan.ts +11 -0
  116. package/src/types/workflow.ts +25 -280
  117. package/src/utils/bundler.ts +32 -5
  118. package/src/utils/errors.ts +34 -2
  119. package/src/workflow-steps/index.ts +1 -0
  120. package/src/workflow-steps/observability.ts +19 -8
  121. package/src/workflow-steps/runner.ts +4 -4
  122. package/src/workflow-steps/step.ts +49 -1
  123. package/src/workflow-steps/types.ts +20 -5
  124. package/src/workflow-steps/workflow.ts +22 -1
  125. package/src/workflows/engine.ts +3 -2
  126. package/src/workflows/invoke-child.ts +2 -2
@@ -0,0 +1,246 @@
1
+ /**
2
+ * Claude Code CLI runtime — drives the `claude` coding CLI headless INSIDE
3
+ * the sandbox (`claude -p --output-format stream-json`) and maps its JSONL
4
+ * event stream onto the AgentMessage contract.
5
+ *
6
+ * This is a different mechanism from `claudeRuntime` (claude.ts): that one
7
+ * drives the Anthropic Agent SDK in the CALLING process, which makes it
8
+ * unusable wherever the caller is the server (a cloud session's turns) —
9
+ * "never execute user-provided source in the server process". This runtime
10
+ * is a CLI-agent spec like codex/opencode: the agent loop runs entirely
11
+ * in-sandbox and we only stream-parse the events it prints, so the
12
+ * server→sandbox cloud-executor path can host it (CLOUD_SESSION_RUNTIMES).
13
+ *
14
+ * Transports, mirroring codexSpec exactly:
15
+ * - ACP: the Zed adapter (`npx @zed-industries/claude-code-acp`) — the SAME
16
+ * pinned adapter the local bridge daemon launches (cli/src/bridge/acp.ts
17
+ * imports the pin from here). Engages only where the provider exposes
18
+ * `commands.spawnDuplex` (the in-VM/local view); the server→sandbox
19
+ * E2B/Vercel view has no duplex, so cloud-session turns always drive the
20
+ * JSONL path below. Cold start: the adapter is npx-fetched on first
21
+ * launch (~10-30s once per sandbox lifetime; node/npm are in the
22
+ * agent-env image).
23
+ * - JSONL: `claude -p --output-format stream-json --verbose`, resume via
24
+ * `--resume <session_id>`. The `claude` binary is BAKED into the
25
+ * `agent-env-<size>` E2B image at /usr/local/bin/claude
26
+ * (infra/e2b-template/build.ts), so there is no install cold-start on the
27
+ * platform path; the `install` below is the bare-sandbox fallback.
28
+ *
29
+ * Auth (in-sandbox env, in the CLI's own precedence order):
30
+ * - `CLAUDE_CODE_OAUTH_TOKEN` — a user subscription (BYOS setup-token /
31
+ * OAuth grant); rides the user's plan directly against Anthropic.
32
+ * - `ANTHROPIC_API_KEY` (+ `ANTHROPIC_BASE_URL`) — an API key; for cloud
33
+ * sessions this is the gateway-minted session virtual key pointed at the
34
+ * token-metering gateway's Anthropic passthrough (ADR-0039).
35
+ */
36
+
37
+ import type { AgentMessage } from "../index.js";
38
+ import { createCliAgentRuntime, shellQuote, type CliAgentSpec, type CliReasoningEffort } from "./_cli-agent.js";
39
+ import { formatError } from "../utils/errors.js";
40
+
41
+ function now(): string { return new Date().toISOString(); }
42
+
43
+ /** Pinned Claude Code ACP adapter (Zed's npm shim — `claude` has no native
44
+ * ACP mode). Pinned EXACT, not a range: the adapter's README warns of
45
+ * breaking changes; bump deliberately and re-verify a live turn. The local
46
+ * bridge daemon (cli/src/bridge/acp.ts) imports this same pin so both
47
+ * executors launch the identical adapter. Verified against 0.16.2. */
48
+ export const CLAUDE_CODE_ACP_ADAPTER = "@zed-industries/claude-code-acp@0.16.2";
49
+
50
+ /** One content block inside a `claude` stream-json assistant/user message. */
51
+ interface ClaudeContentBlock {
52
+ type?: string;
53
+ text?: string;
54
+ thinking?: string;
55
+ id?: string;
56
+ name?: string;
57
+ input?: Record<string, unknown>;
58
+ tool_use_id?: string;
59
+ content?: unknown;
60
+ is_error?: boolean;
61
+ }
62
+
63
+ /** Flatten a tool_result's content (string, or an array of text blocks) into
64
+ * the plain string the AgentMessage contract carries. */
65
+ function toolResultText(content: unknown): string {
66
+ if (typeof content === "string") return content;
67
+ if (Array.isArray(content)) {
68
+ return content
69
+ .map((b: ClaudeContentBlock) => (typeof b?.text === "string" ? b.text : ""))
70
+ .join("");
71
+ }
72
+ return content == null ? "" : JSON.stringify(content) ?? "";
73
+ }
74
+
75
+ /** Extended-thinking token budget per effort level — Claude Code's real knob
76
+ * is the `MAX_THINKING_TOKENS` env var (its documented settings env), set
77
+ * per invocation below. Values mirror the platform turn loop's
78
+ * `EFFORT_BUDGET_TOKENS` (server model-client) so "high" means the same
79
+ * thing in a channel and in a claude-code session. */
80
+ export const CLAUDE_CODE_THINKING_TOKENS: Record<CliReasoningEffort, number> =
81
+ { low: 2_048, medium: 8_192, high: 16_384 };
82
+
83
+ export const claudeCodeSpec: CliAgentSpec = {
84
+ kind: "claude-code",
85
+ authEnv: "ANTHROPIC_API_KEY",
86
+ bin: "claude",
87
+ // ACP-mode invocation — the same npx adapter shape as codexSpec (and the
88
+ // same launch the bridge daemon uses for local claude-code sessions).
89
+ acp: { command: "npx", args: ["--yes", CLAUDE_CODE_ACP_ADAPTER] },
90
+ // Bare-sandbox fallback only: the agent-env image bakes /usr/local/bin/claude
91
+ // (same recipe — infra/e2b-template/build.ts), so the probe short-circuits
92
+ // on the platform path. Anthropic's installer drops a versioned binary under
93
+ // $HOME with a ~/.local/bin/claude launcher; resolve the symlink and copy
94
+ // the self-contained binary to a world-executable system path.
95
+ install:
96
+ 'curl -fsSL https://claude.ai/install.sh | bash && ' +
97
+ 'REAL=$(readlink -f "$HOME/.local/bin/claude") && test -f "$REAL" && ' +
98
+ 'sudo cp "$REAL" /usr/local/bin/claude && sudo chmod 0755 /usr/local/bin/claude',
99
+ // Claude reads the prompt from stdin in -p mode.
100
+ promptPayload: (prompt) => prompt,
101
+ buildCommand: ({ promptPath, sessionId, model, cwd, effort }) => {
102
+ const flags = [
103
+ "-p",
104
+ // stream-json is the JSONL event stream; -p requires --verbose with it.
105
+ "--output-format stream-json",
106
+ "--verbose",
107
+ // Raw API stream events ride along as `stream_event` lines — the
108
+ // text_delta source for progressive rendering (mapEvent below). The
109
+ // complete `assistant` message events still arrive; deltas are
110
+ // additive and live-only.
111
+ "--include-partial-messages",
112
+ // agent-compose already runs the CLI inside an isolated sandbox VM, so
113
+ // permission prompts must not block a headless turn. IS_SANDBOX=1 is
114
+ // claude's own attestation env for exactly this: it lifts the flag's
115
+ // root-user refusal (the E2B agent-env user is root).
116
+ "--dangerously-skip-permissions",
117
+ ...(model ? [`--model ${shellQuote(model)}`] : []),
118
+ ...(sessionId ? [`--resume ${shellQuote(sessionId)}`] : []),
119
+ ].join(" ");
120
+ // Reasoning effort rides as the CLI's own thinking-budget env, same
121
+ // per-invocation idiom as IS_SANDBOX (never persisted into settings).
122
+ const thinking = effort ? `MAX_THINKING_TOKENS=${CLAUDE_CODE_THINKING_TOKENS[effort]} ` : "";
123
+ return `${cwd ? `cd ${shellQuote(cwd)} && ` : ""}${thinking}IS_SANDBOX=1 claude ${flags} < ${shellQuote(promptPath)}`;
124
+ },
125
+ // Every stream-json event carries the session id; the init event is first.
126
+ extractSessionId: (p) => (typeof p.session_id === "string" ? p.session_id : undefined),
127
+ mapEvent: (p): AgentMessage[] => {
128
+ const ts = now();
129
+ switch (p.type) {
130
+ // Assistant API message: content blocks → text / thinking / tool_use.
131
+ case "assistant": {
132
+ const message = p.message as { content?: ClaudeContentBlock[] } | undefined;
133
+ const blocks = Array.isArray(message?.content) ? message.content : [];
134
+ return blocks.flatMap((b): AgentMessage[] => {
135
+ if (b.type === "text" && typeof b.text === "string" && b.text.length > 0) {
136
+ return [{ type: "text", text: b.text, timestamp: ts }];
137
+ }
138
+ if (b.type === "thinking" && typeof b.thinking === "string" && b.thinking.length > 0) {
139
+ return [{ type: "thinking", text: b.thinking, timestamp: ts }];
140
+ }
141
+ if (b.type === "tool_use") {
142
+ return [{
143
+ type: "tool_use", toolName: String(b.name ?? "tool"),
144
+ toolInput: b.input ?? {}, toolUseId: String(b.id ?? ""), timestamp: ts,
145
+ }];
146
+ }
147
+ return [];
148
+ });
149
+ }
150
+ // User API message: the CLI echoes tool results back as user content.
151
+ case "user": {
152
+ const message = p.message as { content?: ClaudeContentBlock[] } | undefined;
153
+ const blocks = Array.isArray(message?.content) ? message.content : [];
154
+ return blocks.flatMap((b): AgentMessage[] =>
155
+ b.type === "tool_result"
156
+ ? [{
157
+ type: "tool_result", toolUseId: String(b.tool_use_id ?? ""),
158
+ output: toolResultText(b.content), isError: b.is_error === true, timestamp: ts,
159
+ }]
160
+ : []);
161
+ }
162
+ // Raw API stream event (--include-partial-messages): text deltas of
163
+ // the in-progress block map to the live-only `text_delta` kind so a
164
+ // streaming consumer (the cloud executor's partial frames) can render
165
+ // progressively. Everything else under stream_event (message_start,
166
+ // content_block_start/stop, thinking/input_json deltas) is noise here —
167
+ // the complete `assistant` event above remains the durable source.
168
+ case "stream_event": {
169
+ const ev = p.event as {
170
+ type?: string;
171
+ delta?: { type?: string; text?: string };
172
+ message?: { usage?: Record<string, number> };
173
+ usage?: Record<string, number>;
174
+ } | undefined;
175
+ if (
176
+ ev?.type === "content_block_delta" &&
177
+ ev.delta?.type === "text_delta" &&
178
+ typeof ev.delta.text === "string" &&
179
+ ev.delta.text.length > 0
180
+ ) {
181
+ return [{ type: "text_delta", text: ev.delta.text, timestamp: ts }];
182
+ }
183
+ // Live-only incremental usage (the text_delta of token counts):
184
+ // message_start carries the model call's input-side finals;
185
+ // message_delta carries the call's CUMULATIVE output so far. The
186
+ // terminal `result` usage above stays the durable turn total.
187
+ if (ev?.type === "message_start" && ev.message?.usage) {
188
+ const u = ev.message.usage;
189
+ return [{
190
+ type: "usage_delta", boundary: "call_start",
191
+ inputTokens: u.input_tokens ?? 0,
192
+ outputTokens: u.output_tokens ?? 0,
193
+ cacheReadTokens: u.cache_read_input_tokens ?? 0,
194
+ cacheCreationTokens: u.cache_creation_input_tokens ?? 0,
195
+ timestamp: ts,
196
+ }];
197
+ }
198
+ if (ev?.type === "message_delta" && ev.usage && typeof ev.usage.output_tokens === "number") {
199
+ return [{
200
+ type: "usage_delta", boundary: "call_delta",
201
+ inputTokens: 0, outputTokens: ev.usage.output_tokens,
202
+ cacheReadTokens: 0, cacheCreationTokens: 0,
203
+ timestamp: ts,
204
+ }];
205
+ }
206
+ return [];
207
+ }
208
+ // Terminal result: usage on success (the text already streamed via the
209
+ // assistant events); an honest string error on failure.
210
+ case "result": {
211
+ if (p.is_error === true) {
212
+ return [{ type: "error", text: formatError(p.result ?? p.subtype ?? p), timestamp: ts }];
213
+ }
214
+ const u = p.usage as Record<string, number> | undefined;
215
+ if (!u) return [];
216
+ return [{
217
+ type: "usage",
218
+ inputTokens: u.input_tokens ?? 0,
219
+ outputTokens: u.output_tokens ?? 0,
220
+ cacheReadTokens: u.cache_read_input_tokens ?? 0,
221
+ cacheCreationTokens: u.cache_creation_input_tokens ?? 0,
222
+ durationMs: typeof p.duration_ms === "number" ? p.duration_ms : 0,
223
+ numTurns: typeof p.num_turns === "number" ? p.num_turns : 1,
224
+ timestamp: ts,
225
+ }];
226
+ }
227
+ // system/init carries the session id (extractSessionId); nothing to map.
228
+ default:
229
+ return [];
230
+ }
231
+ },
232
+ };
233
+
234
+ export interface ClaudeCodeRuntimeConfig {
235
+ /** Claude model id (`--model`). Omit to use the CLI's configured default. */
236
+ model?: string;
237
+ /** Extended-thinking effort (`MAX_THINKING_TOKENS`). Omit for the CLI's
238
+ * default behaviour (no forced budget). */
239
+ effort?: CliReasoningEffort;
240
+ }
241
+
242
+ export function createClaudeCodeRuntime(config: ClaudeCodeRuntimeConfig = {}) {
243
+ return createCliAgentRuntime(claudeCodeSpec, config.model, config.effort);
244
+ }
245
+
246
+ export default createClaudeCodeRuntime();
@@ -131,6 +131,30 @@ function resolveThinking(model: string, thinking: ThinkingConfig | undefined): T
131
131
  return thinking;
132
132
  }
133
133
 
134
+ /** Bounded tail of the Claude Code process's stderr. The Agent SDK's
135
+ * process-exit error carries ONLY the exit code ("Claude Code process
136
+ * exited with code 1") — the CLI's actual complaint goes to stderr, which
137
+ * query() surfaces exclusively through the `stderr` callback. Without
138
+ * capturing it, a startup failure (bad flag, root refusal, dead config)
139
+ * reaches the run card as an uninformative exit code. Same 2000-char cap
140
+ * as the CLI-agent runtimes' exit-code tail (_cli-agent.ts). */
141
+ export const STDERR_TAIL_CAP = 2_000;
142
+
143
+ export class StderrTail {
144
+ private buf = "";
145
+ append(data: string): void {
146
+ this.buf = (this.buf + data).slice(-STDERR_TAIL_CAP);
147
+ }
148
+ /** The captured tail, trimmed — empty string when nothing was written. */
149
+ get text(): string {
150
+ return this.buf.trim();
151
+ }
152
+ /** Suffix an error message with the tail (no-op when stderr was silent). */
153
+ decorate(message: string): string {
154
+ return this.text ? `${message}\nstderr tail:\n${this.text}` : message;
155
+ }
156
+ }
157
+
134
158
  /** Canonical install path for Claude Code inside a Vercel sandbox.
135
159
  * Our `agent-env` setup copies the native binary here as a real file
136
160
  * (not a symlink) so it survives snapshotting. The Anthropic installer
@@ -273,6 +297,8 @@ export class ClaudeRunner implements ModelExecutionContract {
273
297
  .filter((s): s is string => Boolean(s && s.trim()))
274
298
  .join("\n\n");
275
299
 
300
+ const stderrTail = new StderrTail();
301
+
276
302
  try {
277
303
  let emittedAssistantText = false;
278
304
  for await (const message of query({
@@ -287,6 +313,7 @@ export class ClaudeRunner implements ModelExecutionContract {
287
313
  effort: this.config.effort,
288
314
  cwd: this.options.cwd,
289
315
  abortController: queryAbort,
316
+ stderr: (data: string) => stderrTail.append(data),
290
317
  env: { ...process.env, ...(this.config.env ?? {}) },
291
318
  pathToClaudeCodeExecutable: this.config.pathToClaudeCodeExecutable
292
319
  ?? process.env.CLAUDE_CODE_EXECUTABLE
@@ -312,7 +339,7 @@ export class ClaudeRunner implements ModelExecutionContract {
312
339
  // arrived in the stream.
313
340
  if (raw.type === "result" && msg.type === "text" && emittedAssistantText) continue;
314
341
  if (msg.type === "text" && raw.type === "assistant") emittedAssistantText = true;
315
- yield msg;
342
+ yield msg.type === "error" ? { ...msg, text: stderrTail.decorate(msg.text) } : msg;
316
343
  }
317
344
  // End-of-iteration signal: SDK emits a `result` message after the
318
345
  // assistant's turn completes. Close the queue so the AsyncIterable
@@ -333,7 +360,10 @@ export class ClaudeRunner implements ModelExecutionContract {
333
360
  if (/maximum number of turns/i.test(text)) {
334
361
  yield { type: "done", sessionId: opts.sessionId ?? "", timestamp: now() };
335
362
  } else {
336
- yield { type: "error", text, timestamp: now() };
363
+ // The process-exit throw ("Claude Code process exited with code N")
364
+ // carries no detail — the CLI's complaint went to stderr. Attach the
365
+ // captured tail so the run card says WHY, not just the exit code.
366
+ yield { type: "error", text: stderrTail.decorate(text), timestamp: now() };
337
367
  }
338
368
  } finally {
339
369
  // Belt-and-braces — close on error/abort too so the dangling
@@ -14,7 +14,7 @@
14
14
  */
15
15
 
16
16
  import type { AgentMessage } from "../index.js";
17
- import { createCliAgentRuntime, shellQuote, type CliAgentSpec } from "./_cli-agent.js";
17
+ import { createCliAgentRuntime, shellQuote, type CliAgentSpec, type CliReasoningEffort } from "./_cli-agent.js";
18
18
  import { formatError } from "../utils/errors.js";
19
19
 
20
20
  function now(): string { return new Date().toISOString(); }
@@ -24,6 +24,35 @@ function now(): string { return new Date().toISOString(); }
24
24
  * overrides the bundled `@openai/codex` the adapter drives underneath. */
25
25
  const CODEX_ACP_ADAPTER = "@agentclientprotocol/codex-acp@0.1.0";
26
26
 
27
+ /** Known-noise codex ADVISORY lines. codex emits these as `item.completed`
28
+ * error items on the `--json` stream (exec maps every `Warning` notification
29
+ * to an error item — verified against codex-cli 0.147.0), so without a filter
30
+ * they render as conversation content. They are startup/turn advisories, not
31
+ * turn failures: route them to the runner's logs and keep them out of the
32
+ * transcript.
33
+ *
34
+ * Deliberately a NARROW allowlist of known warning classes — never a blanket
35
+ * stderr/error swallow; real errors must keep surfacing. Current classes:
36
+ *
37
+ * - model-metadata fallback: fires whenever the model id is one codex's
38
+ * bundled catalog cannot resolve — every gateway-prefixed session id
39
+ * (`openrouter/openai/gpt-…`) trips it, because the catalog's namespaced
40
+ * -suffix matcher strips at most ONE namespace segment. Benign on the
41
+ * gateway lane: the fallback profile (272k context, plain chat wire) is
42
+ * the correct one for a chat-completions gateway, and the only codex
43
+ * config that could silence it at source (`model_catalog_json`) demands a
44
+ * full authoritative catalog with per-model instruction templates and
45
+ * hard-fails codex at startup on any shape drift. */
46
+ const CODEX_ADVISORY_PATTERNS: readonly RegExp[] = [
47
+ /Model metadata for .+ not found\. Defaulting to fallback metadata/,
48
+ ];
49
+
50
+ /** True when an error-item message is a known codex advisory (log-level
51
+ * noise), not a real error. Exported for tests. */
52
+ export function isCodexAdvisoryNoise(text: string): boolean {
53
+ return CODEX_ADVISORY_PATTERNS.some((re) => re.test(text));
54
+ }
55
+
27
56
  /** Exported for the fixture-based parity tests (ADR-0020): the legacy JSONL
28
57
  * `mapEvent` is the golden the ACP normaliser is asserted equal to. Not part
29
58
  * of the public runtime surface — `createCodexRuntime` stays the entry point. */
@@ -48,7 +77,7 @@ export const codexSpec: CliAgentSpec = {
48
77
  install: 'sudo npm install -g @openai/codex && (command -v codex >/dev/null 2>&1 || sudo ln -sf "$(npm prefix -g)/bin/codex" /usr/local/bin/codex)',
49
78
  // Codex reads the prompt from stdin when invoked as `codex exec ... -`.
50
79
  promptPayload: (prompt) => prompt,
51
- buildCommand: ({ promptPath, sessionId, model, cwd }) => {
80
+ buildCommand: ({ promptPath, sessionId, model, cwd, effort }) => {
52
81
  const flags = [
53
82
  "--json",
54
83
  "--skip-git-repo-check",
@@ -58,6 +87,10 @@ export const codexSpec: CliAgentSpec = {
58
87
  // for running in environments that are externally sandboxed".
59
88
  "--dangerously-bypass-approvals-and-sandbox",
60
89
  ...(model ? ["-m", shellQuote(model)] : []),
90
+ // codex's own reasoning knob — a config override, valid values
91
+ // low|medium|high (plus "minimal", unused here). The value comes from
92
+ // the closed CliReasoningEffort set, so it is shell-safe unquoted.
93
+ ...(effort ? ["-c", `model_reasoning_effort=${effort}`] : []),
61
94
  ...(cwd ? ["-C", shellQuote(cwd)] : []),
62
95
  ].join(" ");
63
96
  // Fresh turn: `codex exec <flags> - < prompt`. Continue a thread:
@@ -93,6 +126,22 @@ export const codexSpec: CliAgentSpec = {
93
126
  const failed = item.status === "failed" || (typeof item.exit_code === "number" && item.exit_code !== 0);
94
127
  return [{ type: "tool_result", toolUseId: id, output: String(item.aggregated_output ?? item.output ?? ""), isError: failed, timestamp: ts }];
95
128
  }
129
+ // Error ITEMS (`{"item":{"type":"error","message":…}}` — e.g. codex's
130
+ // "Falling back…" notices) are errors, not tool calls: the generic
131
+ // branch below used to surface them as a tool named "error" with the
132
+ // raw item as input. Normalise to the error AgentMessage with a
133
+ // STRING text, extracting `.message`.
134
+ if (itype === "error") {
135
+ if (p.type !== "item.completed") return [];
136
+ const text = formatError(item.message ?? item);
137
+ // Known advisory classes go to the runner's logs, never the
138
+ // transcript (see CODEX_ADVISORY_PATTERNS).
139
+ if (isCodexAdvisoryNoise(text)) {
140
+ console.warn(`[codex] advisory (kept out of transcript): ${text}`);
141
+ return [];
142
+ }
143
+ return [{ type: "error", text, timestamp: ts }];
144
+ }
96
145
  // file_change / mcp_tool_call / web_search / todo: surface once, on completion.
97
146
  if (p.type === "item.completed") {
98
147
  return [{ type: "tool_use", toolName: itype || "item", toolInput: item, toolUseId: String(item.id ?? ""), timestamp: ts }];
@@ -125,10 +174,13 @@ export const codexSpec: CliAgentSpec = {
125
174
  export interface CodexRuntimeConfig {
126
175
  /** Codex model id (`-m`). Omit to use the codex CLI's configured default. */
127
176
  model?: string;
177
+ /** Reasoning effort (`-c model_reasoning_effort=<level>`). Omit to use the
178
+ * codex CLI's configured default. */
179
+ effort?: CliReasoningEffort;
128
180
  }
129
181
 
130
182
  export function createCodexRuntime(config: CodexRuntimeConfig = {}) {
131
- return createCliAgentRuntime(codexSpec, config.model);
183
+ return createCliAgentRuntime(codexSpec, config.model, config.effort);
132
184
  }
133
185
 
134
186
  export default createCodexRuntime();
@@ -0,0 +1,59 @@
1
+ /**
2
+ * Cursor CLI runtime — drives Cursor's `cursor-agent` inside the sandbox.
3
+ *
4
+ * ACP-native: `cursor-agent acp` is a protocolVersion-1 ACP server (verified
5
+ * live on E2B 2026-06-30), so the runner delegates the wire protocol to
6
+ * AcpClientPeer; the JSONL members below are the version-mismatch fallback.
7
+ *
8
+ * Auth: CURSOR_API_KEY — Cursor's OWN platform key, NOT OpenRouter. In ACP mode
9
+ * `cursor-agent acp` takes no model flag, so it runs the account's default model;
10
+ * the `--model` ids (auto, gpt-5.3-codex, composer-2.5,
11
+ * claude-opus-4-8-thinking-high; full list via `cursor-agent --list-models`)
12
+ * only apply to the JSONL `-p` fallback. cursor brings HARNESS diversity (a
13
+ * different agent scaffold) to a cross-functional / review panel.
14
+ *
15
+ * Verified live on E2B (2026-06-30): `curl https://cursor.com/install` →
16
+ * ~/.local/bin/cursor-agent (v2026.06.29); `cursor-agent acp` answered the ACP
17
+ * `initialize` with protocolVersion 1; CURSOR_API_KEY authenticated.
18
+ */
19
+ import type { AgentMessage } from "../index.js";
20
+ import { createCliAgentRuntime, shellQuote, type CliAgentSpec } from "./_cli-agent.js";
21
+
22
+ function now(): string { return new Date().toISOString(); }
23
+
24
+ export const cursorSpec: CliAgentSpec = {
25
+ kind: "cursor",
26
+ authEnv: "CURSOR_API_KEY",
27
+ bin: "cursor-agent",
28
+ // ACP runs the account default; "auto" is Cursor's own auto-routing label.
29
+ defaultModel: "auto",
30
+ // `cursor-agent acp` is a protocolVersion-1 ACP server — the runner delegates
31
+ // the whole wire protocol to AcpClientPeer. It takes no model flag, so the
32
+ // session runs Cursor's account-default model.
33
+ acp: { command: "cursor-agent", args: ["acp"] },
34
+ // Self-install on first use; symlink onto PATH for a non-login `sh -c`.
35
+ install: 'curl https://cursor.com/install -fsS | bash && (command -v cursor-agent >/dev/null 2>&1 || sudo ln -sf "$HOME/.local/bin/cursor-agent" /usr/local/bin/cursor-agent)',
36
+ // ── JSONL fallback (only if the ACP handshake negotiates a non-1 version;
37
+ // cursor is v1, so vestigial). `-p` needs `--force` to clear the
38
+ // workspace-trust gate non-interactively.
39
+ promptPayload: (prompt) => prompt,
40
+ buildCommand: ({ promptPath, model, cwd }) =>
41
+ `${cwd ? `cd ${shellQuote(cwd)} && ` : ""}cursor-agent -p --force ${model ? `--model ${shellQuote(model)} ` : ""}--output-format text "$(cat ${shellQuote(promptPath)})"`,
42
+ extractSessionId: () => undefined,
43
+ mapEvent: (p): AgentMessage[] => {
44
+ const ts = now();
45
+ const text = typeof p.text === "string" ? p.text : typeof p.content === "string" ? p.content : "";
46
+ return text ? [{ type: "text", text, timestamp: ts }] : [];
47
+ },
48
+ };
49
+
50
+ export interface CursorRuntimeConfig {
51
+ /** Cursor model id; ACP mode ignores it (account default) — applies to the `-p` fallback. */
52
+ model?: string;
53
+ }
54
+
55
+ export function createCursorRuntime(config: CursorRuntimeConfig = {}) {
56
+ return createCliAgentRuntime(cursorSpec, config.model ?? cursorSpec.defaultModel);
57
+ }
58
+
59
+ export default createCursorRuntime();
@@ -0,0 +1,63 @@
1
+ /**
2
+ * Factory `droid` runtime — drives `droid exec` headless inside the sandbox,
3
+ * JSONL via `--output-format json`.
4
+ *
5
+ * Driven PURELY via OpenRouter BYOK — NO Factory login (verified live on E2B
6
+ * 2026-06-30): a `~/.factory/settings.json` `customModels` entry points at
7
+ * OpenRouter, and the model id is `custom:<displayName>-<index>`. The workflow
8
+ * provisions settings.json (see the dynamic-task route step); this spec just
9
+ * builds the exec command. Auth env is OPENROUTER_API_KEY (the BYOK inference
10
+ * key shared with codex + opencode).
11
+ *
12
+ * NOT ACP: `droid exec` is JSONL. (Its stream-jsonrpc mode ignores --model and
13
+ * sets model/autonomy via JSON-RPC; the plain `--output-format json` mode honours
14
+ * --model, which is what we use.) So it always drives the JSONL path. droid adds
15
+ * Factory's agent HARNESS to a cross-functional / review panel.
16
+ *
17
+ * Verified live on E2B (2026-06-30): install via app.factory.ai/cli;
18
+ * `droid exec --auto low --model custom:GLM-5.2-OR-0 --output-format json`
19
+ * returned `{"type":"result","result":"…","session_id":"…","usage":{…}}` driven
20
+ * through OpenRouter with no Factory account.
21
+ */
22
+ import type { AgentMessage } from "../index.js";
23
+ import { createCliAgentRuntime, shellQuote, type CliAgentSpec } from "./_cli-agent.js";
24
+
25
+ function now(): string { return new Date().toISOString(); }
26
+
27
+ export const droidSpec: CliAgentSpec = {
28
+ kind: "droid",
29
+ // OpenRouter is the inference gateway (BYOK custom model). No FACTORY_API_KEY.
30
+ authEnv: "OPENROUTER_API_KEY",
31
+ bin: "droid",
32
+ // Matches the first customModels entry the workflow writes to settings.json.
33
+ defaultModel: "custom:GLM-5.2-OR-0",
34
+ install: 'curl -fsSL https://app.factory.ai/cli | sh && (command -v droid >/dev/null 2>&1 || sudo ln -sf "$HOME/.local/bin/droid" /usr/local/bin/droid)',
35
+ // ── JSONL path (always — droid exec is not ACP). `--auto medium` lets the
36
+ // agent create/edit files + run commands; `-f` reads the prompt from a file.
37
+ promptPayload: (prompt) => prompt,
38
+ buildCommand: ({ promptPath, model, cwd }) =>
39
+ `${cwd ? `cd ${shellQuote(cwd)} && ` : ""}droid exec --auto medium ${model ? `--model ${shellQuote(model)} ` : ""}--output-format json -f ${shellQuote(promptPath)}`,
40
+ extractSessionId: (p) => (typeof p.session_id === "string" ? p.session_id : undefined),
41
+ mapEvent: (p): AgentMessage[] => {
42
+ const ts = now();
43
+ // `--output-format json` emits a single terminal result object.
44
+ const text =
45
+ p.type === "result" && typeof p.result === "string" ? p.result
46
+ : typeof p.text === "string" ? p.text
47
+ : typeof p.content === "string" ? p.content
48
+ : "";
49
+ return text ? [{ type: "text", text, timestamp: ts }] : [];
50
+ },
51
+ // No `acp` → JSONL-only spec.
52
+ };
53
+
54
+ export interface DroidRuntimeConfig {
55
+ /** `custom:<displayName>-<index>` matching the provisioned settings.json. */
56
+ model?: string;
57
+ }
58
+
59
+ export function createDroidRuntime(config: DroidRuntimeConfig = {}) {
60
+ return createCliAgentRuntime(droidSpec, config.model ?? droidSpec.defaultModel);
61
+ }
62
+
63
+ export default createDroidRuntime();
@@ -20,6 +20,52 @@ export interface OpenAIDesktopRunnerOptions extends RuntimeOptions {
20
20
  openaiKey: string;
21
21
  }
22
22
 
23
+ // ── computer-use-preview wire types ─────────────────────────────────────────
24
+ // The narrow slice of the Responses API this loop exchanges, typed locally:
25
+ // the OpenAI SDK's published types pin computer-use actions to the stable
26
+ // x/y shape, while this preview runner speaks the coordinate-tuple wire —
27
+ // so the exchange is described here instead of with SDK types.
28
+
29
+ /** One model-emitted desktop action. Coordinates arrive in DISPLAY (model)
30
+ * space and are rescaled to the real desktop before dispatch. */
31
+ interface ComputerAction {
32
+ type: string;
33
+ coordinate?: [number, number];
34
+ startCoordinate?: [number, number];
35
+ endCoordinate?: [number, number];
36
+ text?: string;
37
+ key?: string;
38
+ direction?: string;
39
+ ticks?: number;
40
+ }
41
+
42
+ /** Response output blocks this loop consumes — anything else is ignored. */
43
+ type ResponseBlock =
44
+ | { type: "computer_call"; call_id: string; action: ComputerAction }
45
+ | { type: "text"; text: string };
46
+
47
+ /** User-turn content parts this loop sends. */
48
+ type InputPart =
49
+ | { type: "input_image"; image_url: string }
50
+ | { type: "input_text"; text: string }
51
+ | { type: "computer_call_output"; call_id: string; output: { type: "input_image"; image_url: string } };
52
+
53
+ interface InputMessage {
54
+ role: "user";
55
+ content: InputPart[];
56
+ }
57
+
58
+ /** The one `responses.create` call shape this runner makes. */
59
+ interface ComputerUseResponsesApi {
60
+ create(params: {
61
+ model: string;
62
+ tools: Array<{ type: "computer_use_preview"; name: string; display_width: number; display_height: number; environment: "linux" }>;
63
+ input: InputMessage[];
64
+ truncation: "auto";
65
+ previous_response_id?: string;
66
+ }): Promise<{ id: string; output?: ResponseBlock[] }>;
67
+ }
68
+
23
69
  export class OpenAIDesktopRunner implements ModelExecutionContract {
24
70
  readonly kind = "openai-desktop";
25
71
  readonly model = "computer-use-preview";
@@ -55,8 +101,7 @@ export class OpenAIDesktopRunner implements ModelExecutionContract {
55
101
 
56
102
  const screenshot = await captureScaledScreenshot(sandbox);
57
103
 
58
- // eslint-disable-next-line @typescript-eslint/no-explicit-any
59
- const messages: any[] = [{
104
+ const messages: InputMessage[] = [{
60
105
  role: "user",
61
106
  content: [
62
107
  { type: "input_image", image_url: `data:image/png;base64,${screenshot}` },
@@ -64,11 +109,14 @@ export class OpenAIDesktopRunner implements ModelExecutionContract {
64
109
  ],
65
110
  }];
66
111
 
112
+ // The SDK's typed `responses.create` rejects this preview exchange (see
113
+ // the wire-types block above), so cast the resource once to the local
114
+ // narrow interface.
115
+ const responses = this.openai.responses as unknown as ComputerUseResponsesApi;
67
116
  let previousResponseId: string | undefined;
68
117
 
69
118
  for (let turn = 0; turn < MAX_TURNS; turn++) {
70
- // eslint-disable-next-line @typescript-eslint/no-explicit-any
71
- const response = await (this.openai as any).responses.create({
119
+ const response = await responses.create({
72
120
  model: "computer-use-preview",
73
121
  tools: [{ type: "computer_use_preview", name: "computer", display_width: DISPLAY_WIDTH, display_height: DISPLAY_HEIGHT, environment: "linux" }],
74
122
  input: messages,
@@ -77,14 +125,13 @@ export class OpenAIDesktopRunner implements ModelExecutionContract {
77
125
  });
78
126
  previousResponseId = response.id;
79
127
 
80
- // eslint-disable-next-line @typescript-eslint/no-explicit-any
81
- const computerCalls: any[] = response.output?.filter((b: any) => b.type === "computer_call") ?? [];
82
- // eslint-disable-next-line @typescript-eslint/no-explicit-any
83
- const textBlocks: any[] = response.output?.filter((b: any) => b.type === "text") ?? [];
128
+ const computerCalls = response.output?.filter(
129
+ (b): b is Extract<ResponseBlock, { type: "computer_call" }> => b.type === "computer_call") ?? [];
130
+ const textBlocks = response.output?.filter(
131
+ (b): b is Extract<ResponseBlock, { type: "text" }> => b.type === "text") ?? [];
84
132
 
85
133
  if (textBlocks.length > 0) {
86
- // eslint-disable-next-line @typescript-eslint/no-explicit-any
87
- yield { type: "text", text: textBlocks.map((b: any) => b.text).join("\n"), timestamp: ts() };
134
+ yield { type: "text", text: textBlocks.map((b) => b.text).join("\n"), timestamp: ts() };
88
135
  break;
89
136
  }
90
137
 
@@ -98,8 +145,7 @@ export class OpenAIDesktopRunner implements ModelExecutionContract {
98
145
  if (computerCalls.length === 0) break;
99
146
 
100
147
  const next = await captureScaledScreenshot(sandbox);
101
- // eslint-disable-next-line @typescript-eslint/no-explicit-any
102
- messages.push({ role: "user", content: computerCalls.map((call: any) => ({
148
+ messages.push({ role: "user", content: computerCalls.map((call): InputPart => ({
103
149
  type: "computer_call_output",
104
150
  call_id: call.call_id,
105
151
  output: { type: "input_image", image_url: `data:image/png;base64,${next}` },
@@ -126,8 +172,7 @@ async function captureScaledScreenshot(sandbox: DesktopSandboxProvider): Promise
126
172
  function scaleX(x: number): number { return Math.round(x * DESKTOP_WIDTH / DISPLAY_WIDTH); }
127
173
  function scaleY(y: number): number { return Math.round(y * DESKTOP_HEIGHT / DISPLAY_HEIGHT); }
128
174
 
129
- // eslint-disable-next-line @typescript-eslint/no-explicit-any
130
- async function executeAction(sandbox: DesktopSandboxProvider, action: Record<string, any>): Promise<void> {
175
+ async function executeAction(sandbox: DesktopSandboxProvider, action: ComputerAction): Promise<void> {
131
176
  const [x, y] = action.coordinate
132
177
  ? [scaleX(action.coordinate[0]), scaleY(action.coordinate[1])]
133
178
  : [0, 0];