@agent-compose/sdk 0.5.1 → 0.5.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (96) hide show
  1. package/dist/active-step.d.ts +60 -0
  2. package/dist/agent/agent-loop-steer.test.d.ts +1 -0
  3. package/dist/agent/agent-loop.d.ts +46 -0
  4. package/dist/agent/async-queue.d.ts +29 -0
  5. package/dist/agent/protocol.d.ts +9 -1
  6. package/dist/agent/resolve-agent-id.test.d.ts +1 -0
  7. package/dist/agent/run-agent.d.ts +16 -3
  8. package/dist/agent/steer-control.d.ts +57 -0
  9. package/dist/agent/steer-control.test.d.ts +1 -0
  10. package/dist/client.d.ts +161 -0
  11. package/dist/index.d.ts +16 -6
  12. package/dist/index.js +1586 -158
  13. package/dist/pause/__tests__/agent-loop-checkpoint.test.d.ts +1 -0
  14. package/dist/pause/__tests__/checkpoint.test.d.ts +1 -0
  15. package/dist/pause/__tests__/errors.test.d.ts +1 -0
  16. package/dist/pause/__tests__/manager.test.d.ts +1 -0
  17. package/dist/pause/__tests__/pause-core.test.d.ts +1 -0
  18. package/dist/pause/__tests__/state-dir.test.d.ts +1 -0
  19. package/dist/pause/__tests__/wrappers.test.d.ts +1 -0
  20. package/dist/pause/checkpoint.d.ts +28 -0
  21. package/dist/pause/errors.d.ts +52 -0
  22. package/dist/pause/manager.d.ts +63 -0
  23. package/dist/pause/pause-core.d.ts +101 -0
  24. package/dist/pause/state-dir.d.ts +80 -0
  25. package/dist/pause/wrappers.d.ts +41 -0
  26. package/dist/request-context/request-context.d.ts +12 -0
  27. package/dist/runtimes/_cli-agent.d.ts +72 -0
  28. package/dist/runtimes/amp.d.ts +22 -0
  29. package/dist/runtimes/claude.d.ts +6 -0
  30. package/dist/runtimes/codex.d.ts +20 -0
  31. package/dist/runtimes/openai-desktop.d.ts +2 -0
  32. package/dist/runtimes/openai-desktop.js +1578 -157
  33. package/dist/runtimes/vercel.d.ts +53 -1
  34. package/dist/runtimes/vercel.js +60 -8
  35. package/dist/runtimes/vercel.test.d.ts +1 -0
  36. package/dist/sse.d.ts +2 -3
  37. package/dist/step-invocation/index.d.ts +2 -2
  38. package/dist/step-invocation/invoker.d.ts +3 -0
  39. package/dist/step-invocation/protocol.d.ts +12 -0
  40. package/dist/step-invocation/server.d.ts +1 -0
  41. package/dist/step-invocation/types.d.ts +40 -5
  42. package/dist/types/events.d.ts +9 -0
  43. package/dist/types/execution-context.d.ts +25 -0
  44. package/dist/types/protocol.d.ts +8 -0
  45. package/dist/types/runtime.d.ts +55 -0
  46. package/dist/types/sandbox.d.ts +4 -4
  47. package/dist/utils/schemas.d.ts +2 -0
  48. package/dist/workflow-steps/__tests__/pause-wiring.test.d.ts +1 -0
  49. package/dist/workflow-steps/index.d.ts +2 -0
  50. package/dist/workflow-steps/observability.d.ts +43 -11
  51. package/dist/workflow-steps/run-callback.d.ts +39 -0
  52. package/dist/workflow-steps/runner.d.ts +8 -0
  53. package/package.json +1 -1
  54. package/src/active-step.ts +124 -0
  55. package/src/agent/agent-loop.ts +253 -19
  56. package/src/agent/async-queue.ts +61 -0
  57. package/src/agent/protocol.ts +12 -2
  58. package/src/agent/run-agent.ts +184 -8
  59. package/src/agent/steer-control.ts +125 -0
  60. package/src/client.ts +277 -0
  61. package/src/index.ts +38 -4
  62. package/src/pause/checkpoint.ts +44 -0
  63. package/src/pause/errors.ts +70 -0
  64. package/src/pause/manager.ts +177 -0
  65. package/src/pause/pause-core.ts +267 -0
  66. package/src/pause/state-dir.ts +262 -0
  67. package/src/pause/wrappers.ts +79 -0
  68. package/src/request-context/request-context.ts +17 -2
  69. package/src/runtimes/_cli-agent.ts +161 -0
  70. package/src/runtimes/amp.ts +94 -0
  71. package/src/runtimes/claude.ts +101 -6
  72. package/src/runtimes/codex.ts +109 -0
  73. package/src/runtimes/openai-desktop.ts +11 -0
  74. package/src/runtimes/vercel.ts +78 -2
  75. package/src/sandbox.ts +39 -20
  76. package/src/sse.ts +8 -6
  77. package/src/step-invocation/index.ts +2 -1
  78. package/src/step-invocation/invoker.ts +107 -29
  79. package/src/step-invocation/protocol.ts +16 -0
  80. package/src/step-invocation/server.ts +45 -12
  81. package/src/step-invocation/types.ts +43 -7
  82. package/src/tools/coding.ts +16 -5
  83. package/src/types/events.ts +9 -0
  84. package/src/types/execution-context.ts +25 -0
  85. package/src/types/protocol.ts +8 -0
  86. package/src/types/runtime.ts +52 -0
  87. package/src/types/sandbox.ts +8 -4
  88. package/src/types/workflow.ts +6 -1
  89. package/src/utils/bundler.ts +8 -3
  90. package/src/utils/schemas.ts +2 -0
  91. package/src/workflow-steps/index.ts +3 -0
  92. package/src/workflow-steps/observability.ts +84 -13
  93. package/src/workflow-steps/run-callback.ts +72 -0
  94. package/src/workflow-steps/runner.ts +70 -8
  95. package/dist/utils/discovery.d.ts +0 -2
  96. package/src/utils/discovery.ts +0 -4
@@ -1,9 +1,10 @@
1
1
  /** Claude Agent SDK runtime — replaces the old Claude CLI subprocess runtime. */
2
2
 
3
- import { query, type HookCallback, type PreToolUseHookInput, type ThinkingConfig } from "@anthropic-ai/claude-agent-sdk";
3
+ import { query, type HookCallback, type PreToolUseHookInput, type SDKUserMessage, type ThinkingConfig } from "@anthropic-ai/claude-agent-sdk";
4
4
  import type { AgentMessage, ModelExecutionContract, RuntimeOptions, SandboxProvider, ToolCallGateResult } from "../index.js";
5
5
  import { defineRuntime } from "../types/runtime.js";
6
6
  import { DEFAULT_CLAUDE_MODEL } from "../agent/agent-loop.js";
7
+ import { AsyncQueue } from "../agent/async-queue.js";
7
8
  import { runProcessorChain } from "../processors/runner.js";
8
9
  import type { ProcessorContext, ToolCall } from "../processors/processor.js";
9
10
  import { RequestContext } from "../request-context/request-context.js";
@@ -26,6 +27,12 @@ function translateMessage(message: Record<string, unknown>): AgentMessage[] {
26
27
  const b = block as Record<string, unknown>;
27
28
  if (b.type === "text") msgs.push({ type: "text", text: String(b.text ?? ""), timestamp: ts });
28
29
  if (b.type === "thinking") msgs.push({ type: "thinking", text: String(b.thinking ?? ""), timestamp: ts });
30
+ // `redacted_thinking` is the encrypted-thinking variant emitted by
31
+ // reasoning models that don't expose chain-of-thought (the content
32
+ // is an opaque encrypted blob, not human-readable text). Surface
33
+ // it as a placeholder so the Agent tab's thinking badge isn't
34
+ // visually empty for those models.
35
+ if (b.type === "redacted_thinking") msgs.push({ type: "thinking", text: "(redacted)", timestamp: ts });
29
36
  if (b.type === "tool_use") msgs.push({ type: "tool_use", toolName: String(b.name ?? ""), toolInput: (b.input ?? {}) as Record<string, unknown>, toolUseId: String(b.id ?? ""), timestamp: ts });
30
37
  }
31
38
  return msgs;
@@ -82,24 +89,63 @@ export interface ClaudeRuntimeConfig {
82
89
  effort?: "low" | "medium" | "high" | "xhigh" | "max";
83
90
  }
84
91
 
92
+ /** Canonical install path for Claude Code inside a Vercel sandbox.
93
+ * Our `agent-env` setup copies the native binary here as a real file
94
+ * (not a symlink) so it survives snapshotting. The Anthropic installer
95
+ * drops a symlink at `~/.local/bin/claude` pointing into user-home,
96
+ * which Vercel sandbox snapshots don't preserve reliably — pinning
97
+ * `/usr/local/bin/claude` instead bypasses both PATH-priority issues
98
+ * (`~/.local/bin` comes earlier than `/usr/local/bin`) and the
99
+ * broken-symlink-after-restore problem. Callers can override via
100
+ * `pathToClaudeCodeExecutable` or `CLAUDE_CODE_EXECUTABLE`. */
101
+ const DEFAULT_CLAUDE_PATH = "/usr/local/bin/claude";
102
+
85
103
  export class ClaudeRunner implements ModelExecutionContract {
86
104
  supportsToolCallProcessor = true;
105
+ readonly kind = "claude";
106
+
107
+ // ADR-0006 pause-resume hooks intentionally omitted.
108
+ //
109
+ // The Claude Agent SDK keeps the conversation SERVER-SIDE at
110
+ // Anthropic, addressed by `session_id`. Each `sendMessage` call
111
+ // passes `resume: sessionId` and Anthropic rehydrates the prior
112
+ // transcript; `ClaudeRunner` holds no instance state across calls.
113
+ //
114
+ // `sessionId` itself lives in the agent loop (see
115
+ // agent-loop.ts :: `lastSessionId`) and is part of the loop's own
116
+ // state file — restored automatically on pause-resume. So the loop
117
+ // gets everything it needs without any per-runtime blob; defining
118
+ // `captureCheckpoint`/`restoreCheckpoint` here would just be empty.
119
+ //
120
+ // Caveat: Anthropic's server-side session retention is bounded by
121
+ // their TTL (not ours). For workflow pauses inside that window
122
+ // (default workflowRunTimeout is 7 days) the resume succeeds; a
123
+ // 30-day pause may get session-not-found from `resume:` and would
124
+ // need a fallback to client-managed messages. Documented limitation.
87
125
 
88
126
  constructor(
89
- // Claude Agent SDK executes in this process. In dispatched workflows this
90
- // process is already the runner sandbox, so there is no remote provider hop.
91
127
  _sandbox: SandboxProvider,
92
128
  private readonly options: RuntimeOptions = {},
93
129
  private readonly config: ClaudeRuntimeConfig = {},
94
130
  ) {}
95
131
 
132
+ get model(): string {
133
+ return this.config.model ?? this.options.model ?? DEFAULT_CLAUDE_MODEL;
134
+ }
135
+
96
136
  async gateToolCall(call: ToolCall, ctx: ProcessorContext): Promise<ToolCallGateResult> {
97
137
  const verdict = await runProcessorChain(this.options.processors ?? [], p => p.processToolCall, call, ctx);
98
138
  if (verdict.kind === "continue") return { kind: "allow", call: verdict.value };
99
139
  return verdict;
100
140
  }
101
141
 
102
- async *sendMessage(opts: { prompt: string; sessionId?: string; iteration?: number; signal?: AbortSignal }): AsyncGenerator<AgentMessage> {
142
+ async *sendMessage(opts: {
143
+ prompt: string;
144
+ sessionId?: string;
145
+ iteration?: number;
146
+ signal?: AbortSignal;
147
+ inboxStream?: AsyncIterable<{ text: string; senderName?: string | null }>;
148
+ }): AsyncGenerator<AgentMessage> {
103
149
  const requestContext = this.options.requestContext ?? RequestContext.fromReserved({
104
150
  teamId: "", runId: "", workflowId: "",
105
151
  factoryId: null, apiKeyScopes: [], parentRunId: null,
@@ -124,10 +170,48 @@ export class ClaudeRunner implements ModelExecutionContract {
124
170
  return { continue: false, systemMessage: gate.reason };
125
171
  };
126
172
 
173
+ // Streaming-input mode (Claude Agent SDK):
174
+ // When the caller passes `inboxStream`, we build a pushable
175
+ // AsyncIterable<SDKUserMessage> queue. The initial prompt is
176
+ // pushed as the first user turn; inbox messages arriving while
177
+ // the assistant is mid-turn are pushed as additional user turns
178
+ // and the SDK delivers them at the next safe boundary.
179
+ //
180
+ // Without an inboxStream we keep the simple string-prompt path —
181
+ // no queue, same behaviour as before. This keeps non-interactive
182
+ // workflows (memory extractor, etc.) on the original code path.
183
+ //
184
+ // The queue is closed when the SDK's `result` message indicates
185
+ // the iteration's assistant turns are done; without close() the
186
+ // iterable would block forever waiting for more user input.
187
+ const inboxQueue = opts.inboxStream ? new AsyncQueue<SDKUserMessage>() : null;
188
+ if (inboxQueue) {
189
+ inboxQueue.push({
190
+ type: "user",
191
+ message: { role: "user", content: opts.prompt },
192
+ parent_tool_use_id: null,
193
+ });
194
+ // Drain the inbox iterable into the queue. Runs in parallel
195
+ // with the assistant; pushes user turns as they arrive.
196
+ void (async () => {
197
+ try {
198
+ for await (const ev of opts.inboxStream!) {
199
+ const tag = ev.senderName ? `[message from ${ev.senderName}]\n` : "[user message]\n";
200
+ inboxQueue.push({
201
+ type: "user",
202
+ message: { role: "user", content: tag + ev.text },
203
+ parent_tool_use_id: null,
204
+ });
205
+ }
206
+ } catch { /* iterable ended or aborted */ }
207
+ })();
208
+ }
209
+ const promptForQuery: string | AsyncIterable<SDKUserMessage> = inboxQueue ?? opts.prompt;
210
+
127
211
  try {
128
212
  let emittedAssistantText = false;
129
213
  for await (const message of query({
130
- prompt: opts.prompt,
214
+ prompt: promptForQuery,
131
215
  options: {
132
216
  tools: this.options.allowedTools,
133
217
  allowedTools: this.options.allowedTools,
@@ -138,7 +222,9 @@ export class ClaudeRunner implements ModelExecutionContract {
138
222
  effort: this.config.effort,
139
223
  cwd: this.options.cwd,
140
224
  env: { ...process.env, ...(this.config.env ?? {}) },
141
- pathToClaudeCodeExecutable: this.config.pathToClaudeCodeExecutable ?? process.env.CLAUDE_CODE_EXECUTABLE ?? "claude",
225
+ pathToClaudeCodeExecutable: this.config.pathToClaudeCodeExecutable
226
+ ?? process.env.CLAUDE_CODE_EXECUTABLE
227
+ ?? DEFAULT_CLAUDE_PATH,
142
228
  ...(this.config.claudeMdContent ? { systemPrompt: { type: "preset" as const, preset: "claude_code" as const, append: this.config.claudeMdContent } } : {}),
143
229
  resume: opts.sessionId,
144
230
  mcpServers: this.config.mcpServers,
@@ -158,9 +244,18 @@ export class ClaudeRunner implements ModelExecutionContract {
158
244
  if (msg.type === "text" && raw.type === "assistant") emittedAssistantText = true;
159
245
  yield msg;
160
246
  }
247
+ // End-of-iteration signal: SDK emits a `result` message after the
248
+ // assistant's turn completes. Close the queue so the AsyncIterable
249
+ // drains and query() returns. Without this, the iterable would
250
+ // block forever waiting for the next user turn that never comes.
251
+ if (raw.type === "result" && inboxQueue) inboxQueue.close();
161
252
  }
162
253
  } catch (err) {
163
254
  yield { type: "error", text: formatError(err), timestamp: now() };
255
+ } finally {
256
+ // Belt-and-braces — close on error/abort too so the dangling
257
+ // iterable doesn't leak the inboxStream consumer.
258
+ inboxQueue?.close();
164
259
  }
165
260
  }
166
261
  }
@@ -0,0 +1,109 @@
1
+ /**
2
+ * Codex CLI runtime — drives OpenAI's `codex exec --json` agentic CLI inside
3
+ * the sandbox and maps its JSONL event stream onto the AgentMessage contract.
4
+ * Built on the shared CLI-agent base; Codex brings its own loop + tools, so we
5
+ * only stream-parse what it prints.
6
+ *
7
+ * Auth: set `CODEX_API_KEY` (or `OPENAI_API_KEY`) in the sandbox env via a
8
+ * workflow secret. Requires the `codex` CLI installed in the sandbox image.
9
+ *
10
+ * ⚠️ NOT verified against a live `codex` run — the resume flag and the exact
11
+ * item shapes (command_execution / reasoning fields) are mapped from the docs
12
+ * (developers.openai.com/codex/noninteractive). Verify before production use.
13
+ */
14
+
15
+ import type { AgentMessage } from "../index.js";
16
+ import { createCliAgentRuntime, shellQuote, type CliAgentSpec } from "./_cli-agent.js";
17
+ import { formatError } from "../utils/errors.js";
18
+
19
+ function now(): string { return new Date().toISOString(); }
20
+
21
+ const codexSpec: CliAgentSpec = {
22
+ kind: "codex",
23
+ authEnv: "CODEX_API_KEY",
24
+ // Codex reads the prompt from stdin when invoked as `codex exec ... -`.
25
+ promptPayload: (prompt) => prompt,
26
+ buildCommand: ({ promptPath, sessionId, model, cwd }) => {
27
+ const flags = [
28
+ "--json",
29
+ "--skip-git-repo-check",
30
+ // agent-compose already runs us inside an isolated sandbox VM, so codex
31
+ // must not try to nest its own seccomp/landlock sandbox or block on
32
+ // approvals (non-interactive). codex docs: this flag is "intended solely
33
+ // for running in environments that are externally sandboxed".
34
+ "--dangerously-bypass-approvals-and-sandbox",
35
+ ...(model ? ["-m", shellQuote(model)] : []),
36
+ ...(cwd ? ["-C", shellQuote(cwd)] : []),
37
+ ].join(" ");
38
+ // Fresh turn: `codex exec <flags> - < prompt`. Continue a thread:
39
+ // `codex exec resume <id> <flags> - < prompt`. (`-` = read prompt from stdin.)
40
+ const exec = sessionId
41
+ ? `codex exec resume ${shellQuote(sessionId)} ${flags}`
42
+ : `codex exec ${flags}`;
43
+ return `${exec} - < ${shellQuote(promptPath)}`;
44
+ },
45
+ extractSessionId: (p) =>
46
+ p.type === "thread.started" && typeof p.thread_id === "string" ? p.thread_id : undefined,
47
+ mapEvent: (p): AgentMessage[] => {
48
+ const ts = now();
49
+ switch (p.type) {
50
+ case "item.started":
51
+ case "item.completed": {
52
+ const item = p.item as Record<string, unknown> | undefined;
53
+ if (!item) return [];
54
+ const itype = String(item.type ?? "");
55
+ // Text + reasoning land on completion (started carries no final text).
56
+ if (itype === "agent_message") {
57
+ return p.type === "item.completed" ? [{ type: "text", text: String(item.text ?? ""), timestamp: ts }] : [];
58
+ }
59
+ if (itype === "reasoning") {
60
+ return p.type === "item.completed" ? [{ type: "thinking", text: String(item.text ?? ""), timestamp: ts }] : [];
61
+ }
62
+ // Command execution: started → tool_use, completed → tool_result.
63
+ if (itype === "command_execution") {
64
+ const id = String(item.id ?? "");
65
+ if (p.type === "item.started") {
66
+ return [{ type: "tool_use", toolName: "shell", toolInput: { command: String(item.command ?? "") }, toolUseId: id, timestamp: ts }];
67
+ }
68
+ const failed = item.status === "failed" || (typeof item.exit_code === "number" && item.exit_code !== 0);
69
+ return [{ type: "tool_result", toolUseId: id, output: String(item.aggregated_output ?? item.output ?? ""), isError: failed, timestamp: ts }];
70
+ }
71
+ // file_change / mcp_tool_call / web_search / todo: surface once, on completion.
72
+ if (p.type === "item.completed") {
73
+ return [{ type: "tool_use", toolName: itype || "item", toolInput: item, toolUseId: String(item.id ?? ""), timestamp: ts }];
74
+ }
75
+ return [];
76
+ }
77
+ case "turn.completed": {
78
+ const u = p.usage as Record<string, number> | undefined;
79
+ if (!u) return [];
80
+ return [{
81
+ type: "usage",
82
+ inputTokens: u.input_tokens ?? 0,
83
+ outputTokens: u.output_tokens ?? 0,
84
+ cacheReadTokens: u.cached_input_tokens ?? 0,
85
+ cacheCreationTokens: 0,
86
+ durationMs: 0,
87
+ numTurns: 1,
88
+ timestamp: ts,
89
+ }];
90
+ }
91
+ case "turn.failed":
92
+ case "error":
93
+ return [{ type: "error", text: formatError(p.error ?? p.message ?? p), timestamp: ts }];
94
+ default:
95
+ return [];
96
+ }
97
+ },
98
+ };
99
+
100
+ export interface CodexRuntimeConfig {
101
+ /** Codex model id (`-m`). Omit to use the codex CLI's configured default. */
102
+ model?: string;
103
+ }
104
+
105
+ export function createCodexRuntime(config: CodexRuntimeConfig = {}) {
106
+ return createCliAgentRuntime(codexSpec, config.model);
107
+ }
108
+
109
+ export default createCodexRuntime();
@@ -21,9 +21,20 @@ export interface OpenAIDesktopRunnerOptions extends RuntimeOptions {
21
21
  }
22
22
 
23
23
  export class OpenAIDesktopRunner implements ModelExecutionContract {
24
+ readonly kind = "openai-desktop";
25
+ readonly model = "computer-use-preview";
24
26
  private openai: OpenAI;
25
27
  private label: string;
26
28
 
29
+ // ADR-0006 pause-resume hooks intentionally omitted.
30
+ //
31
+ // The computer-use-preview model treats every `sendMessage` as a
32
+ // fresh conversation seeded with the current desktop screenshot —
33
+ // the `messages` array and `previousResponseId` are scoped to ONE
34
+ // call, never carried across. The desktop screenshot IS the state,
35
+ // and the sandbox snapshot already captures that on the filesystem.
36
+ // No runtime-private blob needed.
37
+
27
38
  constructor(
28
39
  private sandbox: DesktopSandboxProvider,
29
40
  opts: OpenAIDesktopRunnerOptions,
@@ -3,7 +3,7 @@
3
3
  * owned coding tools over SandboxProvider.
4
4
  */
5
5
 
6
- import { streamText, stepCountIs, tool, type LanguageModel } from "ai";
6
+ import { streamText, stepCountIs, tool, gateway, type LanguageModel } from "ai";
7
7
  import type { AgentMessage, ModelExecutionContract, RuntimeOptions, SandboxProvider, ToolCallGateResult } from "../index.js";
8
8
  import { defineRuntime } from "../types/runtime.js";
9
9
  import { codingTools, type CodingTool } from "../tools/index.js";
@@ -15,12 +15,27 @@ import { formatError } from "../utils/errors.js";
15
15
  type AiToolSet = Record<string, ReturnType<typeof tool<Record<string, unknown>, string>>>;
16
16
 
17
17
  export interface VercelRuntimeConfig {
18
- /** Vercel AI SDK language model (e.g. openai("gpt-5"), anthropic("claude-sonnet-4-5")). */
18
+ /** The model to drive — this is the only thing that varies per provider;
19
+ * there is no per-provider runtime. `LanguageModel` accepts every provider:
20
+ * - a gateway model-id string routed via the Vercel AI Gateway (set
21
+ * `AI_GATEWAY_API_KEY`; no provider package needed), e.g. "openai/gpt-5",
22
+ * "google/gemini-2.5-pro", "xai/grok-4", "deepseek/deepseek-chat",
23
+ * "mistral/mistral-large-latest", "anthropic/claude-sonnet-4-5";
24
+ * - or a `LanguageModel` object from a provider package (add the dep + set
25
+ * its API-key env), e.g. `openai("gpt-5")`, `google("gemini-2.5-pro")`. */
19
26
  model: LanguageModel;
20
27
  /** Optional system prompt prepended to every model call. */
21
28
  system?: string;
22
29
  /** Override/extend the default coding tools. Defaults: Read, Write, Edit, Bash. */
23
30
  tools?: readonly CodingTool[];
31
+ /** Short runtime self-id surfaced on `agent.spawned` so the dashboard can
32
+ * show a per-agent runtime icon (e.g. "openai", "gemini"). Provider presets
33
+ * set this; bare `createVercelRuntime` callers can leave it unset. */
34
+ kind?: string;
35
+ /** Display model id surfaced on `agent.spawned` (the Agent tab labels which
36
+ * model each agent ran). `model` above is the AI SDK LanguageModel object;
37
+ * this is its human-readable id string. */
38
+ modelId?: string;
24
39
  }
25
40
 
26
41
  function now(): string { return new Date().toISOString(); }
@@ -71,6 +86,11 @@ function toAgentMessages(part: Record<string, unknown>): AgentMessage[] {
71
86
 
72
87
  export class VercelRunner implements ModelExecutionContract {
73
88
  supportsToolCallProcessor = true;
89
+ /** Surfaced on `agent.spawned` for the dashboard's per-agent runtime icon +
90
+ * model label. Set from the (provider preset's) config; undefined for a
91
+ * bare `createVercelRuntime` that didn't label itself. */
92
+ readonly kind?: string;
93
+ readonly model?: string;
74
94
  private readonly tools: readonly CodingTool[];
75
95
  private readonly messages: unknown[] = [];
76
96
 
@@ -80,6 +100,8 @@ export class VercelRunner implements ModelExecutionContract {
80
100
  private readonly config: VercelRuntimeConfig,
81
101
  ) {
82
102
  this.tools = config.tools ?? codingTools;
103
+ this.kind = config.kind;
104
+ this.model = config.modelId;
83
105
  }
84
106
 
85
107
  async gateToolCall(call: ToolCall, ctx: ProcessorContext): Promise<ToolCallGateResult> {
@@ -88,6 +110,32 @@ export class VercelRunner implements ModelExecutionContract {
88
110
  return verdict;
89
111
  }
90
112
 
113
+ /** ADR-0006 pause-resume hooks. The Vercel AI SDK holds the running
114
+ * conversation client-side in `this.messages` — every `streamText`
115
+ * call passes the full array as `messages` and reconstructs it from
116
+ * `response.messages` after completion. A pause-induced subprocess
117
+ * exit loses the array; the new subprocess constructs a fresh
118
+ * `VercelRunner` with `this.messages = []` and the next sendMessage
119
+ * would see only the current iteration's prompt — conversation
120
+ * history broken. The agent loop calls these at every iteration
121
+ * boundary so the messages array round-trips through the sandbox
122
+ * snapshot. */
123
+ captureCheckpoint(): unknown {
124
+ return { messages: [...this.messages] };
125
+ }
126
+
127
+ restoreCheckpoint(blob: unknown): void {
128
+ if (!blob || typeof blob !== "object") {
129
+ throw new Error("Vercel runtime checkpoint is invalid: expected an object with messages[]");
130
+ }
131
+ const incoming = (blob as { messages?: unknown }).messages;
132
+ if (!Array.isArray(incoming)) {
133
+ throw new Error("Vercel runtime checkpoint is invalid: expected messages[]");
134
+ }
135
+ this.messages.length = 0;
136
+ this.messages.push(...incoming);
137
+ }
138
+
91
139
  private buildTools(iteration: number, signal?: AbortSignal): AiToolSet {
92
140
  const set: AiToolSet = {};
93
141
  for (const t of this.tools) {
@@ -178,3 +226,31 @@ export function createVercelRuntime(config: VercelRuntimeConfig) {
178
226
  create: (sandbox, opts) => new VercelRunner(sandbox, opts, config),
179
227
  });
180
228
  }
229
+
230
+ /** One model offered by the Vercel AI Gateway. Pass `id` straight to
231
+ * `createVercelRuntime({ model: id })`. */
232
+ export interface VercelRuntimeModel {
233
+ /** Gateway model id, e.g. "openai/gpt-5". Usable directly as the runtime model. */
234
+ id: string;
235
+ /** Human-readable display name. */
236
+ name: string;
237
+ }
238
+
239
+ /**
240
+ * List the models the Vercel runtime accepts as a gateway model-id string — the
241
+ * LIVE Vercel AI Gateway catalog, so it never goes stale. This is the canonical
242
+ * answer to "what models can I pass to `createVercelRuntime`?" for the string
243
+ * form (`createVercelRuntime({ model: "openai/gpt-5" })`).
244
+ *
245
+ * Requires `AI_GATEWAY_API_KEY`. The other form — a `LanguageModel` object from
246
+ * an `@ai-sdk/<provider>` package — supports whatever that provider package
247
+ * does (see its docs); there's no single cross-form list because the runtime is
248
+ * model-agnostic. Browse the catalog in a UI at https://vercel.com/ai-gateway/models.
249
+ */
250
+ export async function listVercelRuntimeModels(): Promise<VercelRuntimeModel[]> {
251
+ const { models } = await gateway.getAvailableModels();
252
+ return models
253
+ .filter((m) => m.modelType == null || m.modelType === "language")
254
+ .map((m) => ({ id: m.id, name: m.name }))
255
+ .sort((a, b) => a.id.localeCompare(b.id));
256
+ }
package/src/sandbox.ts CHANGED
@@ -118,9 +118,28 @@ interface SandboxProviderDef {
118
118
  // ── E2B helpers ───────────────────────────────────────────────────────────────
119
119
 
120
120
  export function makeSandboxProvider(sb: Sandbox | Desktop): SandboxProvider {
121
+ const commands = sb.commands as { run: (cmd: string, opts?: unknown) => Promise<{ exitCode: number; stdout: string; stderr?: string }> };
121
122
  return {
122
123
  sandboxId: sb.sandboxId,
123
- commands: sb.commands,
124
+ commands: {
125
+ async run(cmd, opts) {
126
+ let stderr = "";
127
+ const result = await commands.run(cmd, {
128
+ ...opts,
129
+ onStderr: (chunk: string) => {
130
+ stderr += chunk;
131
+ opts?.onStderr?.(chunk);
132
+ },
133
+ });
134
+ return {
135
+ exitCode: result.exitCode,
136
+ stdout: result.stdout,
137
+ stderr: typeof (result as { stderr?: unknown }).stderr === "string"
138
+ ? (result as { stderr: string }).stderr
139
+ : stderr,
140
+ };
141
+ },
142
+ },
124
143
  files: {
125
144
  async write(path, content) { await sb.files.write(path, content); },
126
145
  },
@@ -161,7 +180,7 @@ export async function parseSseExecStream(
161
180
  body: ReadableStream<Uint8Array>,
162
181
  opts?: ParseSseExecStreamOptions,
163
182
  ): Promise<SandboxCommandResult> {
164
- let stdout = "", exitCode = 0, exited = false;
183
+ let stdout = "", stderr = "", exitCode = 0, exited = false;
165
184
  const reader = body.getReader(), decoder = new TextDecoder();
166
185
  let buf = "";
167
186
  while (true) {
@@ -173,9 +192,10 @@ export async function parseSseExecStream(
173
192
  for (const line of lines) {
174
193
  if (!line.startsWith("data: ")) continue;
175
194
  let event: SseEvent;
176
- try { event = JSON.parse(line.slice(6)) as SseEvent; } catch { continue; }
195
+ try { event = JSON.parse(line.slice(6)) as SseEvent; }
196
+ catch (err) { throw new Error(`Malformed sandbox exec event: ${err instanceof Error ? err.message : String(err)}`); }
177
197
  if (event.type === "stdout") { stdout += event.data ?? ""; opts?.onStdout?.(event.data ?? ""); }
178
- else if (event.type === "stderr") { opts?.onStderr?.(event.data ?? ""); }
198
+ else if (event.type === "stderr") { stderr += event.data ?? ""; opts?.onStderr?.(event.data ?? ""); }
179
199
  else if (event.type === "exit") { exitCode = event.exitCode ?? 0; exited = true; }
180
200
  else if (event.type === "error") { throw new Error(`Sandbox exec error: ${event.data}`); }
181
201
  }
@@ -183,7 +203,8 @@ export async function parseSseExecStream(
183
203
  // QUIC tunnels can delay the stream-end signal even after all data has arrived.
184
204
  if (exited) break;
185
205
  }
186
- return { exitCode, stdout };
206
+ if (!exited) throw new Error("Sandbox exec stream ended without an exit event");
207
+ return { exitCode, stdout, stderr };
187
208
  }
188
209
 
189
210
  // ── Vercel helpers ────────────────────────────────────────────────────────────
@@ -200,31 +221,28 @@ function makeVercelSandboxProvider(sb: any, globalEnvs?: Record<string, string>)
200
221
  sandboxId: sb.sandboxId,
201
222
  commands: {
202
223
  async run(cmd, opts) {
203
- // Vercel's runCommand takes a native `sudo` flag — pass it on the `sh`
204
- // invocation so the whole shell (and every command it spawns) runs as
205
- // root, matching `sb.runCommand({ cmd, args, sudo: true })` semantics.
206
- const sudo = opts?.sudo === true;
207
- if (opts?.background) {
208
- // eslint-disable-next-line @typescript-eslint/no-explicit-any
209
- void (sb.runCommand({ cmd: "sh", args: ["-c", cmd], cwd: opts.cwd, env: mergeEnvs(opts.envs), detached: true, ...(sudo ? { sudo: true } : {}) }) as Promise<any>)
210
- .catch((err: unknown) => console.error(`[sandbox] background command failed: ${err instanceof Error ? err.message : String(err)}`));
211
- return { exitCode: 0, stdout: "" };
212
- }
213
224
  const signal = opts?.timeoutMs ? AbortSignal.timeout(opts.timeoutMs) : undefined;
214
225
  let stdout = "";
226
+ let stderr = "";
227
+ // `sudo: true` is Vercel's native root flag — applied to the `sh`
228
+ // invocation so the whole shell (and its children) runs as root.
215
229
  // eslint-disable-next-line @typescript-eslint/no-explicit-any
216
- const handle: any = await sb.runCommand({ cmd: "sh", args: ["-c", cmd], cwd: opts?.cwd, env: mergeEnvs(opts?.envs), detached: true, signal, ...(sudo ? { sudo: true } : {}) });
230
+ const handle: any = await sb.runCommand({ cmd: "sh", args: ["-c", cmd], cwd: opts?.cwd, env: mergeEnvs(opts?.envs), detached: true, signal, ...(opts?.sudo ? { sudo: true } : {}) });
217
231
  // Reconnect to the already-running command on transient stream failures (e.g. BrotliDecompressionError).
232
+ // `h.logs()` replays from the start on reconnect, so reset accumulators per
233
+ // attempt to avoid double-counting. The streaming callbacks may still fire
234
+ // for duplicate chunks during retries — an acceptable tradeoff for resilience.
218
235
  await pRetry(async (attempt) => {
219
236
  // eslint-disable-next-line @typescript-eslint/no-explicit-any
220
237
  const h: any = attempt === 1 ? handle : await sb.getCommand(handle.cmdId);
238
+ if (attempt > 1) { stdout = ""; stderr = ""; }
221
239
  for await (const log of h.logs()) {
222
240
  if (log.stream === "stdout") { stdout += log.data; opts?.onStdout?.(log.data); }
223
- else { opts?.onStderr?.(log.data); }
241
+ else { stderr += log.data; opts?.onStderr?.(log.data); }
224
242
  }
225
243
  }, { retries: 3, minTimeout: 1_000, factor: 2 });
226
244
  const finished = await handle.wait();
227
- return { exitCode: finished.exitCode, stdout };
245
+ return { exitCode: finished.exitCode, stdout, stderr };
228
246
  },
229
247
  },
230
248
  files: {
@@ -277,12 +295,13 @@ export function makeLocalSandboxProvider(): SandboxProvider {
277
295
  stdio: ["ignore", "pipe", "pipe"],
278
296
  });
279
297
  let stdout = "";
298
+ let stderr = "";
280
299
  proc.stdout?.setEncoding("utf8");
281
300
  proc.stderr?.setEncoding("utf8");
282
301
  proc.stdout?.on("data", (chunk: string) => { stdout += chunk; opts?.onStdout?.(chunk); });
283
- proc.stderr?.on("data", (chunk: string) => { opts?.onStderr?.(chunk); });
302
+ proc.stderr?.on("data", (chunk: string) => { stderr += chunk; opts?.onStderr?.(chunk); });
284
303
  proc.on("error", reject);
285
- proc.on("close", (code) => resolve({ exitCode: code ?? 0, stdout }));
304
+ proc.on("close", (code) => resolve({ exitCode: code ?? 0, stdout, stderr }));
286
305
  });
287
306
  },
288
307
  },
package/src/sse.ts CHANGED
@@ -7,9 +7,8 @@
7
7
  * - `event`: event name (from `event:` line; `""` if absent)
8
8
  * - `data`: parsed JSON payload (the SDK's stream events are always JSON)
9
9
  *
10
- * Malformed payloads are silently skipped — same behaviour as the previous
11
- * CLI-side parser. Caller drives the loop via `for await (...)` and is
12
- * responsible for breaking on terminal events.
10
+ * Malformed payloads throw. A dropped terminal event is worse than a loud
11
+ * protocol error for log consumers.
13
12
  *
14
13
  * Output type stays loose (`Record<string, unknown>`) on purpose: typed
15
14
  * unions like `RunEvent` aren't structurally narrowable from
@@ -42,10 +41,13 @@ export async function* parseSseStream(
42
41
  if (line.startsWith("event:")) { event = line.slice(6).trim(); continue; }
43
42
  if (line.startsWith("data:")) { dataLines.push(line.slice(5).trim()); continue; }
44
43
  if (line === "" && dataLines.length > 0) {
44
+ let data: Record<string, unknown>;
45
45
  try {
46
- const data = JSON.parse(dataLines.join("\n")) as Record<string, unknown>;
47
- yield { id: seq, event, data };
48
- } catch { /* malformed — skip */ }
46
+ data = JSON.parse(dataLines.join("\n")) as Record<string, unknown>;
47
+ } catch (err) {
48
+ throw new Error(`Malformed SSE JSON payload for event "${event || "message"}": ${err instanceof Error ? err.message : String(err)}`);
49
+ }
50
+ yield { id: seq, event, data };
49
51
  event = ""; seq = 0; dataLines = [];
50
52
  }
51
53
  }
@@ -20,6 +20,7 @@
20
20
 
21
21
  export {
22
22
  STEP_RESULT_PREFIX,
23
+ STEP_PAUSE_PREFIX,
23
24
  STEP_ENV,
24
25
  RUNNER_BUNDLE_PATH,
25
26
  RUNNER_COMMAND,
@@ -30,4 +31,4 @@ export { invokeStep, parseStepResult, buildStepEnvs } from "./invoker.js";
30
31
  export { serveStep } from "./server.js";
31
32
  export type { StepHandler, ServeStepRequest, StepHandlerResult } from "./server.js";
32
33
  export { StepExecutionError } from "./types.js";
33
- export type { StepRequest, StepResult, StepInvocationError } from "./types.js";
34
+ export type { StepRequest, StepResult, StepInvocationError, StepPauseRequest } from "./types.js";