@agent-compose/sdk 0.6.0 → 0.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (126) hide show
  1. package/README.md +66 -39
  2. package/dist/agent/__tests__/runtime-json-schema.test.d.ts +10 -0
  3. package/dist/agent/agent-context.d.ts +21 -1
  4. package/dist/agent/agent-loop.d.ts +24 -1
  5. package/dist/client.d.ts +338 -534
  6. package/dist/directives.d.ts +112 -0
  7. package/dist/display.d.ts +242 -0
  8. package/dist/errors.d.ts +24 -1
  9. package/dist/index.d.ts +34 -13
  10. package/dist/index.js +2984 -861
  11. package/dist/pause/wrappers.d.ts +31 -9
  12. package/dist/processors/ask-human.d.ts +30 -0
  13. package/dist/processors/ask-human.test.d.ts +1 -0
  14. package/dist/processors/index.d.ts +1 -0
  15. package/dist/runtimes/_acp-client.d.ts +46 -1
  16. package/dist/runtimes/_cli-agent.d.ts +58 -4
  17. package/dist/runtimes/_jsonl-guard.d.ts +103 -0
  18. package/dist/runtimes/amp.d.ts +2 -2
  19. package/dist/runtimes/claude-code.d.ts +59 -0
  20. package/dist/runtimes/claude-code.test.d.ts +14 -0
  21. package/dist/runtimes/claude.d.ts +16 -0
  22. package/dist/runtimes/claude.test.d.ts +8 -0
  23. package/dist/runtimes/codex.d.ts +9 -3
  24. package/dist/runtimes/cursor.d.ts +9 -0
  25. package/dist/runtimes/droid.d.ts +9 -0
  26. package/dist/runtimes/jsonl-guard.test.d.ts +19 -0
  27. package/dist/runtimes/openai-desktop.js +2922 -861
  28. package/dist/runtimes/opencode.d.ts +25 -0
  29. package/dist/runtimes/vercel.js +22 -1
  30. package/dist/sandbox/devbox.d.ts +42 -0
  31. package/dist/sandbox/exec-stream.d.ts +14 -0
  32. package/dist/sandbox/network-policy.d.ts +100 -0
  33. package/dist/sandbox/provider-def.d.ts +79 -0
  34. package/dist/sandbox/providers/desktop.d.ts +10 -0
  35. package/dist/sandbox/providers/e2b.d.ts +17 -0
  36. package/dist/sandbox/providers/local.d.ts +11 -0
  37. package/dist/sandbox/providers/vercel.d.ts +18 -0
  38. package/dist/sandbox/registry.d.ts +45 -0
  39. package/dist/sandbox/sizes.d.ts +68 -0
  40. package/dist/sandbox.d.ts +24 -299
  41. package/dist/step-invocation/__tests__/foreground-recovery.test.d.ts +1 -0
  42. package/dist/step-invocation/invoker.d.ts +24 -1
  43. package/dist/step-invocation/protocol.d.ts +13 -0
  44. package/dist/types/api-compliance.d.ts +71 -0
  45. package/dist/types/api-conversations.d.ts +492 -0
  46. package/dist/types/api-factory.d.ts +309 -0
  47. package/dist/types/api-projects.d.ts +131 -0
  48. package/dist/types/api-runs.d.ts +377 -0
  49. package/dist/types/api-scopes.d.ts +102 -0
  50. package/dist/types/conversation-stream.d.ts +191 -0
  51. package/dist/types/execution-context.d.ts +12 -2
  52. package/dist/types/protocol.d.ts +30 -1
  53. package/dist/types/sandbox-environment.d.ts +8 -5
  54. package/dist/types/sandbox.d.ts +79 -0
  55. package/dist/types/workflow-metadata.d.ts +33 -8
  56. package/dist/types/workflow-plan.d.ts +10 -0
  57. package/dist/types/workflow.d.ts +18 -193
  58. package/dist/utils/bundler.d.ts +12 -1
  59. package/dist/utils/errors.d.ts +9 -1
  60. package/dist/workflow-steps/index.d.ts +1 -1
  61. package/dist/workflow-steps/observability.d.ts +8 -1
  62. package/dist/workflow-steps/runner.d.ts +3 -3
  63. package/dist/workflow-steps/step.d.ts +15 -1
  64. package/dist/workflow-steps/types.d.ts +19 -5
  65. package/dist/workflow-steps/workflow.d.ts +22 -1
  66. package/dist/workflows/engine.d.ts +3 -2
  67. package/dist/workflows/invoke-child.d.ts +2 -2
  68. package/package.json +1 -1
  69. package/src/agent/agent-context.ts +206 -16
  70. package/src/agent/agent-loop.ts +40 -4
  71. package/src/agent/run-agent.ts +9 -1
  72. package/src/client.ts +909 -621
  73. package/src/directives.ts +184 -0
  74. package/src/display.ts +788 -0
  75. package/src/errors.ts +39 -0
  76. package/src/index.ts +117 -10
  77. package/src/pause/wrappers.ts +44 -9
  78. package/src/processors/ask-human.ts +136 -0
  79. package/src/processors/index.ts +5 -0
  80. package/src/runtimes/_acp-client.ts +72 -3
  81. package/src/runtimes/_cli-agent.ts +171 -38
  82. package/src/runtimes/_jsonl-guard.ts +219 -0
  83. package/src/runtimes/claude-code.ts +246 -0
  84. package/src/runtimes/claude.ts +32 -2
  85. package/src/runtimes/codex.ts +55 -3
  86. package/src/runtimes/cursor.ts +59 -0
  87. package/src/runtimes/droid.ts +63 -0
  88. package/src/runtimes/openai-desktop.ts +59 -14
  89. package/src/runtimes/opencode.ts +61 -0
  90. package/src/sandbox/devbox.ts +48 -0
  91. package/src/sandbox/exec-stream.ts +48 -0
  92. package/src/sandbox/network-policy.ts +181 -0
  93. package/src/sandbox/provider-def.ts +94 -0
  94. package/src/sandbox/providers/desktop.ts +57 -0
  95. package/src/sandbox/providers/e2b.ts +354 -0
  96. package/src/sandbox/providers/local.ts +106 -0
  97. package/src/sandbox/providers/vercel.ts +331 -0
  98. package/src/sandbox/registry.ts +198 -0
  99. package/src/sandbox/sizes.ts +95 -0
  100. package/src/sandbox.ts +59 -1263
  101. package/src/step-invocation/invoker.ts +319 -34
  102. package/src/step-invocation/protocol.ts +19 -0
  103. package/src/types/api-compliance.ts +79 -0
  104. package/src/types/api-conversations.ts +522 -0
  105. package/src/types/api-factory.ts +336 -0
  106. package/src/types/api-projects.ts +140 -0
  107. package/src/types/api-runs.ts +412 -0
  108. package/src/types/api-scopes.ts +102 -0
  109. package/src/types/conversation-stream.ts +231 -0
  110. package/src/types/execution-context.ts +10 -2
  111. package/src/types/protocol.ts +33 -0
  112. package/src/types/sandbox-environment.ts +28 -9
  113. package/src/types/sandbox.ts +78 -0
  114. package/src/types/workflow-metadata.ts +35 -8
  115. package/src/types/workflow-plan.ts +11 -0
  116. package/src/types/workflow.ts +25 -280
  117. package/src/utils/bundler.ts +32 -5
  118. package/src/utils/errors.ts +34 -2
  119. package/src/workflow-steps/index.ts +1 -0
  120. package/src/workflow-steps/observability.ts +19 -8
  121. package/src/workflow-steps/runner.ts +4 -4
  122. package/src/workflow-steps/step.ts +49 -1
  123. package/src/workflow-steps/types.ts +20 -5
  124. package/src/workflow-steps/workflow.ts +22 -1
  125. package/src/workflows/engine.ts +3 -2
  126. package/src/workflows/invoke-child.ts +2 -2
@@ -0,0 +1,331 @@
1
+ /**
2
+ * Vercel sandbox provider (@vercel/sandbox 2.x) — tag building, the provider
3
+ * wrapper, and the "vercel" registry entry.
4
+ *
5
+ * CRITICAL: every VALUE import of @vercel/sandbox in this file is dynamic
6
+ * (`await import(...)`) so the package stays out of runner.bundle.js — the
7
+ * runner never calls these paths. Top-level imports are type-only (erased).
8
+ */
9
+
10
+ import pRetry from "p-retry";
11
+ import { SandboxUnavailableError } from "../../sandbox-errors.js";
12
+ import type { SandboxProvider } from "../../types/sandbox.js";
13
+ import { toVercelNetworkPolicy } from "../network-policy.js";
14
+ import { SANDBOX_VCPUS, DEFAULT_SANDBOX_SIZE } from "../sizes.js";
15
+ import { AGENT_COMPOSE_TAG } from "../provider-def.js";
16
+ import type { OwnedSandbox, SandboxProviderDef } from "../provider-def.js";
17
+ // Type-only — erased at compile time, so @vercel/sandbox stays out of
18
+ // runner.bundle.js (the value imports below are dynamic for the same reason).
19
+ import type { Sandbox as VercelSandbox, Command as VercelCommand } from "@vercel/sandbox";
20
+
21
+ /** Upper bound for how old a live Vercel sandbox can be: Pro/Enterprise plan
22
+ * cap (5h) + 1h slack for clock skew and `extendTimeout()` calls. */
23
+ const VERCEL_VM_LIFETIME_WINDOW_MS = 6 * 60 * 60 * 1000;
24
+
25
+ /** Vercel caps sandbox tags at 5. Build the tag set with the fleet `executor`
26
+ * tag always present and always winning over caller metadata — `listOwned` /
27
+ * orphan reconciliation scope on it, so a clobbered or dropped executor tag
28
+ * makes the sandbox invisible to cleanup. When metadata overflows the cap,
29
+ * keep executor + the lexicographically-first 4 metadata keys, so what gets
30
+ * dropped is deterministic rather than dependent on object insertion order. */
31
+ export const VERCEL_MAX_TAGS = 5;
32
+ export function buildVercelTags(metadata: Record<string, string>): Record<string, string> {
33
+ const tags: Record<string, string> = { ...metadata, executor: AGENT_COMPOSE_TAG };
34
+ const extras = Object.keys(tags).filter((k) => k !== "executor").sort();
35
+ for (const key of extras.slice(VERCEL_MAX_TAGS - 1)) delete tags[key];
36
+ return tags;
37
+ }
38
+
39
+ function makeVercelSandboxProvider(sb: VercelSandbox, globalEnvs?: Record<string, string>): SandboxProvider {
40
+ // Vercel's Sandbox.create({ env }) does NOT flow to runCommand subprocesses — they start fresh shells.
41
+ // Capture the sandbox-level envs and merge them into every runCommand call so that env vars like
42
+ // ANTHROPIC_API_KEY (placeholder for network policy injection) are actually visible to subprocesses.
43
+ const mergeEnvs = (cmdEnvs?: Record<string, string>) =>
44
+ globalEnvs ? { ...globalEnvs, ...cmdEnvs } : cmdEnvs;
45
+
46
+ // Wrap a sandbox-side failure as `SandboxUnavailableError`. `retryable` is
47
+ // decided purely by *where* the error was caught, not by inspecting its
48
+ // shape: true before the runner launched any user code (file write / command
49
+ // launch / reconnect — safe to re-provision and retry), false once the
50
+ // runner streamed (user code may have run). The original message is
51
+ // preserved for server-side diagnostics.
52
+ const asUnavailable = (err: unknown, retryable: boolean): never => {
53
+ throw new SandboxUnavailableError(err instanceof Error ? err.message : String(err), { retryable, sandboxId: sb.name });
54
+ };
55
+
56
+ return {
57
+ // @vercel/sandbox 2.x is name-keyed: `name` is the stable identifier
58
+ // (`Sandbox.get({ name })`); the v1 `sandboxId` getter is gone. Our
59
+ // provider-facing field keeps its name — it's "the provider-native id".
60
+ sandboxId: sb.name,
61
+ commands: {
62
+ async run(cmd, opts) {
63
+ const signal = opts?.timeoutMs ? AbortSignal.timeout(opts.timeoutMs) : undefined;
64
+ let stdout = "";
65
+ let stderr = "";
66
+ // `sudo: true` is Vercel's native root flag — applied to the `sh`
67
+ // invocation so the whole shell (and its children) runs as root.
68
+ // A launch failure means the command never started — nothing
69
+ // user-side ran, so any error here is safe to re-provision + retry.
70
+ const handle: VercelCommand = await sb.runCommand({ cmd: "sh", args: ["-c", cmd], cwd: opts?.cwd, env: mergeEnvs(opts?.envs), detached: true, signal, ...(opts?.sudo ? { sudo: true } : {}) })
71
+ .catch((err: unknown) => asUnavailable(err, true));
72
+ // Stream + wait, BOTH inside the same reconnect loop. Reconnect to the
73
+ // already-running command on transient stream failures (e.g.
74
+ // BrotliDecompressionError). `h.logs()` replays from the start on
75
+ // reconnect, so dedupe by byte offset: `stdout` / `stderr` hold
76
+ // everything already delivered to the caller — skip the replayed
77
+ // prefix and append/emit only the genuinely-new suffix. The caller's
78
+ // sinks must see each byte EXACTLY once: the step-recovery log
79
+ // backfill slices the tee'd log file at the delivered-chars offset,
80
+ // so a replayed re-emit would inflate that offset past the file
81
+ // length and silently drop the undelivered tail.
82
+ //
83
+ // `wait()` MUST live inside this same retry, not after it. Vercel's
84
+ // log stream can close independently of the command actually
85
+ // exiting — observed live as every long-running Vercel step in the
86
+ // fleet dying at ~20-26 minutes with a generic "operation timed
87
+ // out" on `wait()`, never on `logs()`. A `wait()` fault is the same
88
+ // transport-fault class as a `logs()` fault: the command is
89
+ // `detached: true` and therefore still running on a lone `wait()`
90
+ // hiccup, so it must reconnect and retry too instead of failing the
91
+ // whole (possibly multi-hour) step non-retryably on one bad call.
92
+ const collect = async () => {
93
+ const finished = await pRetry(async (attempt) => {
94
+ const h = attempt === 1 ? handle : await sb.getCommand(handle.cmdId);
95
+ // Chars of this attempt's (replayed-from-start) stream seen so far,
96
+ // per stream. New bytes are whatever extends past the delivered
97
+ // accumulator; replay chunk boundaries need not match the original's.
98
+ let seenOut = 0;
99
+ let seenErr = 0;
100
+ const fresh = (data: string, seen: number, delivered: number): string => {
101
+ const skip = Math.min(Math.max(delivered - seen, 0), data.length);
102
+ return skip > 0 ? data.slice(skip) : data;
103
+ };
104
+ for await (const log of h.logs()) {
105
+ if (log.stream === "stdout") {
106
+ const add = fresh(log.data, seenOut, stdout.length);
107
+ seenOut += log.data.length;
108
+ if (add) { stdout += add; opts?.onStdout?.(add); }
109
+ } else {
110
+ const add = fresh(log.data, seenErr, stderr.length);
111
+ seenErr += log.data.length;
112
+ if (add) { stderr += add; opts?.onStderr?.(add); }
113
+ }
114
+ }
115
+ return await h.wait();
116
+ }, { retries: 3, minTimeout: 1_000, factor: 2 });
117
+ return { exitCode: finished.exitCode, stdout, stderr };
118
+ };
119
+ // `timeoutMs` MUST be authoritative. The AbortSignal alone is not: a
120
+ // command that RUNS but emits nothing (e.g. a wedged `archil checkout`
121
+ // on a blocked data plane) leaves `logs()`/`wait()` pending and the
122
+ // abort never interrupts the await — the call hangs for the activity's
123
+ // whole multi-hour ceiling. Race a hard client-side deadline so a hung
124
+ // command fails fast and the caller's degrade/retry policy takes over.
125
+ let timer: ReturnType<typeof setTimeout> | undefined;
126
+ const deadline = opts?.timeoutMs
127
+ ? new Promise<never>((_, reject) => {
128
+ timer = setTimeout(
129
+ () => reject(new Error(`command timed out after ${opts.timeoutMs}ms: ${cmd.slice(0, 200)}`)),
130
+ opts.timeoutMs);
131
+ })
132
+ : undefined;
133
+ try {
134
+ return await (deadline ? Promise.race([collect(), deadline]) : collect());
135
+ } catch (err) {
136
+ // A timeout (hard deadline or AbortSignal) is a COMMAND overrunning
137
+ // its budget, not the sandbox dying — the VM is alive. Surface it as
138
+ // a plain error so the caller's own retry/deadline policy decides
139
+ // (classifying it as terminal sandbox-unavailable failed whole runs
140
+ // over a single slow mount probe, observed live).
141
+ if (err instanceof Error && err.message.startsWith("command timed out after")) throw err;
142
+ if (signal?.aborted) throw new Error(`command timed out after ${opts?.timeoutMs}ms: ${cmd.slice(0, 200)}`);
143
+ return asUnavailable(err, false);
144
+ } finally {
145
+ if (timer) clearTimeout(timer);
146
+ }
147
+ },
148
+ },
149
+ files: {
150
+ async write(path: string, content: string) {
151
+ // File writes happen before the runner launches (step input / pause
152
+ // resolution files), so a failure here means nothing ran — retryable.
153
+ await sb.writeFiles([{ path, content }]).catch((err: unknown) => asUnavailable(err, true));
154
+ },
155
+ // Read over Vercel's HTTP file API — a DIFFERENT transport from the
156
+ // runCommand log stream, so a readback works after a stream/wait fault
157
+ // (the property the foreground recovery's log backfill relies on).
158
+ // No `asUnavailable` wrap: `read` is consumed only by the best-effort
159
+ // recovery backfill — classifying a missing log file as sandbox death
160
+ // would be wrong.
161
+ async read(path: string): Promise<string> {
162
+ const buf: Buffer | null = await sb.readFileToBuffer({ path });
163
+ if (buf === null) throw new Error(`file not found: ${path}`);
164
+ return buf.toString("utf8");
165
+ },
166
+ },
167
+ // Propagate errors — `killAllRunSandboxes` relies on kill failures being
168
+ // observable so it can leave `sandbox_id` set for `findOrphanedSandboxes`
169
+ // to retry on next boot. Swallowing here makes the orphan retry loop blind.
170
+ //
171
+ // kill must DESTROY: in @vercel/sandbox 2.x stop() only halts the VM — the
172
+ // name-keyed sandbox record persists (reserving the name; commands against
173
+ // a stale handle can transparently resume a stopped VM). delete() is the
174
+ // destroying call ("after deletion the instance becomes inert"), so stop
175
+ // then delete.
176
+ async kill() { await sb.stop(); await sb.delete(); },
177
+ // Vercel native snapshot — used by sandbox-environments to capture the
178
+ // configured VM after the customer's `setup()` completes. `expiration: 0`
179
+ // is Vercel's "never expires" value; env snapshots are long-lived by
180
+ // design (they ARE the env) so we always pass it.
181
+ async snapshot() {
182
+ const res = await sb.snapshot({ expiration: 0 });
183
+ // `sizeBytes` is typed non-optional but populated by the API — keep the
184
+ // runtime guard so a missing field degrades to "unknown size", not NaN.
185
+ const sizeBytes = typeof res.sizeBytes === "number" ? res.sizeBytes : undefined;
186
+ return {
187
+ snapshotId: res.snapshotId,
188
+ ...(sizeBytes !== undefined ? { sizeBytes } : {}),
189
+ };
190
+ },
191
+ // Push a freshly-resolved egress policy onto the live sandbox —
192
+ // @vercel/sandbox 2.x `update({ networkPolicy })`. Lets the server
193
+ // re-resolve the run policy (re-minting connector access tokens)
194
+ // before each step instead of living with the policy baked at create.
195
+ // E2B implements the same seam via `sb.updateNetwork(toE2bNetwork(...))`.
196
+ async updateNetworkPolicy(policy) {
197
+ await sb.update({ networkPolicy: toVercelNetworkPolicy(policy) });
198
+ },
199
+ };
200
+ }
201
+
202
+ async function listOwned(env: Record<string, string>): Promise<OwnedSandbox[]> {
203
+ const { Sandbox: VercelSandbox } = await import("@vercel/sandbox");
204
+ const creds = { token: env.VERCEL_ACCESS_TOKEN, teamId: env.VERCEL_TEAM_ID, projectId: env.VERCEL_PROJECT_ID };
205
+ // Scoped to our fleet by the `executor` tag (stamped at create) and
206
+ // bounded to `VERCEL_VM_LIFETIME_WINDOW_MS` — Pro/Enterprise caps VM
207
+ // lifetime at 5h, +1h slack for clock skew and `extendTimeout()`. The
208
+ // sort is pinned to createdAt-desc explicitly — the early-stop below is
209
+ // only correct under that order, and the server's default sort is not a
210
+ // contract we can rely on — so we stop at the first item older than the
211
+ // window instead of paging through thousands of historical terminal
212
+ // sandboxes (10k+ in v0.6.28). Truncating silently would hide orphans,
213
+ // so we throw at the page cap instead — 200 pages × the API's 50-item
214
+ // limit preserves the previous 10k scan bound (v1 paged 200 × 50). Each
215
+ // page is wrapped in pRetry so a transient 5xx/429/network blip doesn't
216
+ // cost a reconcile tick.
217
+ //
218
+ // 2.x cutover caveat: sandboxes created by the pre-2.x code carry no
219
+ // tags, so for up to VERCEL_VM_LIFETIME_WINDOW_MS after a deploy they
220
+ // are invisible here (and their persisted v1 sandbox ids don't resolve
221
+ // via `Sandbox.get({ name })`). Self-limiting — the provider's 5h
222
+ // lifetime cap expires them — but drain in-flight Vercel runs before
223
+ // deploying if losing their sandbox state matters.
224
+ const since = Date.now() - VERCEL_VM_LIFETIME_WINDOW_MS;
225
+ const out: OwnedSandbox[] = [];
226
+ let cursor: string | undefined;
227
+ for (let page = 0; page < 200; page++) {
228
+ const result = await pRetry(
229
+ // limit is capped at 50 by the v2 list API — 51+ answers 400
230
+ // (probed live 2026-06-10; not documented in the SDK types).
231
+ () => VercelSandbox.list({ ...creds, limit: 50, sortBy: "createdAt", sortOrder: "desc", tags: { executor: AGENT_COMPOSE_TAG }, ...(cursor !== undefined ? { cursor } : {}) }),
232
+ { retries: 3, minTimeout: 500, factor: 2 },
233
+ );
234
+ for (const sb of result.sandboxes) {
235
+ if (sb.createdAt < since) return out;
236
+ if (sb.status === "running") {
237
+ out.push({ sandboxId: sb.name, createdAt: new Date(sb.createdAt), metadata: sb.tags ?? {} });
238
+ }
239
+ }
240
+ if (result.pagination.next === null) return out;
241
+ cursor = result.pagination.next;
242
+ }
243
+ throw new Error("Vercel listSandboxes exceeded 200 pages (10k sandboxes) within the lifetime window — refuse to silently truncate");
244
+ }
245
+
246
+ export const vercelProviderDef: SandboxProviderDef = {
247
+ requiredEnv: {
248
+ VERCEL_ACCESS_TOKEN: "Vercel access token — vercel.com/account/tokens",
249
+ VERCEL_TEAM_ID: "Vercel team ID — vercel.com/account/settings",
250
+ VERCEL_PROJECT_ID: "Vercel project ID — vercel.com/[team]/[project]/settings",
251
+ },
252
+ create: async (opts, env) => {
253
+ // Dynamic import keeps @vercel/sandbox out of runner.bundle.js (runner never calls this path)
254
+ const { Sandbox: VercelSandbox } = await import("@vercel/sandbox");
255
+ const creds = { token: env.VERCEL_ACCESS_TOKEN, teamId: env.VERCEL_TEAM_ID, projectId: env.VERCEL_PROJECT_ID };
256
+ // `template` is a snapshot id (runtime is baked in); absent → fresh node24.
257
+ const np = opts.networkPolicy ? { networkPolicy: toVercelNetworkPolicy(opts.networkPolicy) } : {};
258
+ // Caller metadata rides as Vercel tags — `listOwned` filters on the
259
+ // `executor` key, mirroring E2B's metadata-tag scoping. buildVercelTags
260
+ // guarantees executor wins over metadata and survives the 5-tag cap.
261
+ const tags = buildVercelTags(opts.metadata);
262
+ // No explicit template/bootFrom → VERCEL_DEFAULT_SNAPSHOT if set (the
263
+ // platform agent-env base: claude, archil, rtk, bun, agentc CLI, the
264
+ // SDK in /workspace/node_modules, /ac:* skills, AGENTS.md — built by
265
+ // .agentc/environments/agent-env.ts), else raw node24. The E2B
266
+ // analogue is E2B_DEFAULT_TEMPLATE (see providers/e2b.ts). Because
267
+ // `saveLatest`/`reuse` captures snapshot the whole filesystem, everything
268
+ // a run bakes on top of this base chains from it — the base tools are in
269
+ // every derived snapshot for free.
270
+ const tmpl = opts.template ?? process.env.VERCEL_DEFAULT_SNAPSHOT;
271
+ // Machine size → vCPUs (RAM auto-follows at 2048 MB/vCPU). Always sent
272
+ // explicitly so the spec is deterministic and self-documenting rather
273
+ // than riding Vercel's implicit default; "small" maps to that default
274
+ // anyway, so existing runs are unchanged.
275
+ const resources = { vcpus: SANDBOX_VCPUS[opts.size ?? DEFAULT_SANDBOX_SIZE] };
276
+ // `persistent: false` — @vercel/sandbox 2.x creates persistent-by-default
277
+ // sandboxes: stop() auto-snapshots (and keeps billing storage), commands
278
+ // transparently resume a stopped VM, and kill no longer destroys. Our
279
+ // sandboxes are single-run and lifecycle-managed by the engine (explicit
280
+ // snapshot() / kill()), so opt out in BOTH branches.
281
+ const sb = await VercelSandbox.create(tmpl
282
+ ? { source: { type: "snapshot" as const, snapshotId: tmpl }, timeout: opts.timeoutMs, env: opts.envs, tags, persistent: false, resources, ...np, ...creds }
283
+ : { runtime: "node24" as const, timeout: opts.timeoutMs, env: opts.envs, tags, persistent: false, resources, ...np, ...creds },
284
+ );
285
+ // Pass envs as globalEnvs so they're injected into every runCommand subprocess.
286
+ // (Vercel's Sandbox.create env parameter does not flow to runCommand subprocesses.)
287
+ return makeVercelSandboxProvider(sb, opts.envs);
288
+ },
289
+ reconnect: async (sandboxId) => {
290
+ const { Sandbox: VercelSandbox } = await import("@vercel/sandbox");
291
+ const creds = {
292
+ token: process.env.VERCEL_ACCESS_TOKEN ?? "",
293
+ teamId: process.env.VERCEL_TEAM_ID ?? "",
294
+ projectId: process.env.VERCEL_PROJECT_ID ?? "",
295
+ };
296
+ // Reconnecting happens before any user code runs, so any failure here
297
+ // (sandbox already stopping, API blip) is a retryable sandbox-unavailable
298
+ // — the workflow re-provisions a fresh one rather than failing the run.
299
+ // 2.x is name-keyed: the id we persisted IS the sandbox name.
300
+ const sb = await VercelSandbox.get({ name: sandboxId, ...creds }).catch((err: unknown) => {
301
+ throw new SandboxUnavailableError(err instanceof Error ? err.message : String(err), { retryable: true, sandboxId });
302
+ });
303
+ return makeVercelSandboxProvider(sb);
304
+ },
305
+ getActiveCount: async (env) => (await listOwned(env)).length,
306
+ listOwned,
307
+ deleteSnapshot: async (snapshotId, env) => {
308
+ const { Snapshot } = await import("@vercel/sandbox");
309
+ const creds = { token: env.VERCEL_ACCESS_TOKEN, teamId: env.VERCEL_TEAM_ID, projectId: env.VERCEL_PROJECT_ID };
310
+ const snap = await Snapshot.get({ snapshotId, ...creds });
311
+ await snap.delete();
312
+ },
313
+ snapshotExists: async (snapshotId, env) => {
314
+ const { Snapshot } = await import("@vercel/sandbox");
315
+ const creds = { token: env.VERCEL_ACCESS_TOKEN, teamId: env.VERCEL_TEAM_ID, projectId: env.VERCEL_PROJECT_ID };
316
+ // Metadata-only lookup (no sandbox provisioned). A clean fetch ⇒ resolves.
317
+ // A 404 / not-found ⇒ definitively absent (false). Anything else (auth,
318
+ // 5xx, network) is INDETERMINATE and must propagate — never reported as
319
+ // "missing", which would falsely alert on a transient blip.
320
+ try {
321
+ await Snapshot.get({ snapshotId, ...creds });
322
+ return true;
323
+ } catch (err) {
324
+ const message = err instanceof Error ? err.message : String(err ?? "");
325
+ const status = (err as { status?: number; statusCode?: number })?.status
326
+ ?? (err as { statusCode?: number })?.statusCode;
327
+ if (status === 404 || /\b404\b|not[\s_-]?found|no such snapshot/i.test(message)) return false;
328
+ throw err;
329
+ }
330
+ },
331
+ };
@@ -0,0 +1,198 @@
1
+ /**
2
+ * Sandbox provider registry — creates and manages isolated execution
3
+ * environments across every registered provider, and owns the ONE sandbox
4
+ * retry policy. Providers: "vercel" (Vercel Sandbox), "e2b" (E2B),
5
+ * "e2b-desktop" (E2B Desktop). Each provider validates its required env vars
6
+ * at creation time.
7
+ */
8
+
9
+ import { SandboxNotFoundError, RateLimitError } from "e2b";
10
+ import pRetry from "p-retry";
11
+ import type { FailedAttemptError } from "p-retry";
12
+ import type { SandboxProvider } from "../types/sandbox.js";
13
+ import type { SandboxCreateOpts, SandboxProviderDef, OwnedSandbox } from "./provider-def.js";
14
+ import { vercelProviderDef } from "./providers/vercel.js";
15
+ import { e2bProviderDef } from "./providers/e2b.js";
16
+ import { e2bDesktopProviderDef } from "./providers/desktop.js";
17
+
18
+ export type SandboxProviderName = "vercel" | "e2b" | "e2b-desktop";
19
+
20
+ export const SANDBOX_PROVIDERS: Record<string, SandboxProviderDef> = {
21
+ "vercel": vercelProviderDef,
22
+ "e2b": e2bProviderDef,
23
+ "e2b-desktop": e2bDesktopProviderDef,
24
+ };
25
+
26
+ /** Is this a FLEETING sandbox-provider race worth retrying — the sandbox being
27
+ * paused / reclaimed / stopped under load, an upstream 5xx/429, or a dropped
28
+ * connection — vs a PERMANENT failure that must fail fast?
29
+ *
30
+ * E2B raises TYPED errors, so prefer `instanceof` (upgrade-stable — survives any
31
+ * message rewording): a reconnect/snapshot that hits a sandbox mid-pause throws
32
+ * `SandboxNotFoundError`, rate limits throw `RateLimitError`. (Permanent
33
+ * template/snapshot-not-found only occurs on CREATE, which is no longer retried.)
34
+ * The message regex is the FALLBACK for the untyped Vercel transport + raw socket
35
+ * errors — and deliberately omits bare "timeout"/"network": E2B's permanent
36
+ * plan-cap 400 ("timeout … exceeds the maximum sandbox lifetime") contains
37
+ * "timeout" and must fail fast, not retry 4×. */
38
+ function isTransientSandboxError(error: unknown): boolean {
39
+ if (error instanceof SandboxNotFoundError || error instanceof RateLimitError) return true;
40
+ const message = error instanceof Error ? error.message : String(error ?? "");
41
+ // The 5xx match is anchored to a status-code context ("status 503",
42
+ // "502 Bad Gateway", …) — a bare \b5\d\d\b would classify ANY standalone
43
+ // 500-599 number in a message (durations, row counts) as retryable.
44
+ return /paus|stopp|resum|transition|status(?:\s+code)?[:\s]+5\d\d\b|\b5\d\d\s+(?:internal|bad gateway|service|gateway)|\b429\b|ECONNRESET|ECONNREFUSED|ETIMEDOUT|socket hang up|fetch failed/i
45
+ .test(message);
46
+ }
47
+
48
+ /** The ONE place the sandbox retry policy lives. Provisioning, reconnecting, and
49
+ * snapshotting all race with the sandbox being paused/reclaimed; this runs the call
50
+ * through p-retry's generic backoff, retrying ONLY the transient race (and failing
51
+ * fast otherwise). reconnectSandbox and provider.snapshot() all go through it, so
52
+ * no provider re-implements the pattern or can forget it. `shouldRetry` is
53
+ * overridable for call sites where the default classification is wrong (see
54
+ * reconnectSandbox's SandboxNotFoundError handling). */
55
+ function withSandboxRetry<T>(
56
+ fn: () => Promise<T>,
57
+ shouldRetry: (error: FailedAttemptError) => boolean = isTransientSandboxError,
58
+ ): Promise<T> {
59
+ return pRetry(fn, { retries: 4, minTimeout: 400, factor: 2, shouldRetry });
60
+ }
61
+
62
+ /** Add snapshot-retry to a freshly-obtained provider. create/reconnect are retried at
63
+ * the registry call (`createSandbox`/`reconnectSandbox`); snapshot is a method, so it
64
+ * gets the same `withSandboxRetry` here. `commands.run` is deliberately NOT wrapped —
65
+ * retrying user-code execution could double-apply side effects, so step execution
66
+ * relies on the engine's pre-step sandbox recovery instead. No-op without a snapshot. */
67
+ function withSnapshotRetry(p: SandboxProvider): SandboxProvider {
68
+ if (!p.snapshot) return p;
69
+ const snapshot = p.snapshot.bind(p);
70
+ return { ...p, snapshot: () => withSandboxRetry(snapshot) };
71
+ }
72
+
73
+ /** Provision a sandbox for the named provider. */
74
+ export async function createSandbox(provider: SandboxProviderName, opts: SandboxCreateOpts): Promise<SandboxProvider> {
75
+ const def = SANDBOX_PROVIDERS[provider];
76
+ if (!def) throw new Error(`Unknown sandbox provider: "${provider}". Known: ${Object.keys(SANDBOX_PROVIDERS).join(", ")}`);
77
+ const missing = Object.entries(def.requiredEnv)
78
+ .filter(([key]) => !process.env[key])
79
+ .map(([key, desc]) => ` ${key} — ${desc}`);
80
+ if (missing.length > 0) throw new Error(`Sandbox provider "${provider}" requires env vars:\n${missing.join("\n")}`);
81
+ const env = Object.fromEntries(Object.keys(def.requiredEnv).map((k) => [k, process.env[k]!]));
82
+ // Provision once — do NOT wrap create in withSandboxRetry: a create whose RESPONSE
83
+ // is lost (client-side timeout after the provider already provisioned) would, on
84
+ // retry, mint a SECOND billed sandbox and orphan the first. Provision failures are
85
+ // re-attempted at the workflow layer. The result is still wrapped so snapshot()
86
+ // retries the transient pause/reclaim race (which is where the race actually lives).
87
+ return withSnapshotRetry(await def.create(opts, env));
88
+ }
89
+
90
+ /** Reconnect to an existing sandbox by provider + provider-native sandbox ID. */
91
+ export async function reconnectSandbox(provider: SandboxProviderName, sandboxId: string): Promise<SandboxProvider> {
92
+ const def = SANDBOX_PROVIDERS[provider];
93
+ if (!def?.reconnect) throw new Error(`Provider "${provider}" does not support reconnect`);
94
+ // Retry the transient reconnect/resume race, then add snapshot-retry to the
95
+ // result. On RECONNECT, SandboxNotFoundError usually means the sandbox is
96
+ // genuinely gone (killed / lifetime-expired) — permanent — but it can also
97
+ // be the transient mid-pause race, so it gets exactly ONE retry instead of
98
+ // burning the full 4-attempt backoff before surfacing.
99
+ return withSnapshotRetry(await withSandboxRetry(
100
+ () => def.reconnect!(sandboxId),
101
+ (error) => error instanceof SandboxNotFoundError ? error.attemptNumber <= 1 : isTransientSandboxError(error),
102
+ ));
103
+ }
104
+
105
+ /** Delete a snapshot by id on the named provider. No live sandbox needed. */
106
+ export async function deleteSandboxSnapshot(provider: SandboxProviderName, snapshotId: string): Promise<void> {
107
+ const def = SANDBOX_PROVIDERS[provider];
108
+ if (!def?.deleteSnapshot) throw new Error(`Provider "${provider}" does not support deleteSnapshot`);
109
+ const missing = Object.entries(def.requiredEnv)
110
+ .filter(([k]) => !process.env[k])
111
+ .map(([k, desc]) => ` ${k} — ${desc}`);
112
+ if (missing.length > 0) throw new Error(`Sandbox provider "${provider}" requires env vars:\n${missing.join("\n")}`);
113
+ return def.deleteSnapshot(snapshotId, Object.fromEntries(Object.keys(def.requiredEnv).map(k => [k, process.env[k]!])));
114
+ }
115
+
116
+ /**
117
+ * Does `snapshotId` resolve on the named provider for this account? A cheap
118
+ * metadata lookup — never provisions a sandbox. `true` = resolves, `false` =
119
+ * provider says it does not exist. Transport/credential errors propagate (an
120
+ * indeterminate result is not "missing"). Used at server boot to validate the
121
+ * platform base snapshots the default templates boot from.
122
+ */
123
+ export async function snapshotResolves(provider: SandboxProviderName, snapshotId: string): Promise<boolean> {
124
+ const def = SANDBOX_PROVIDERS[provider];
125
+ if (!def?.snapshotExists) throw new Error(`Provider "${provider}" does not support snapshotExists`);
126
+ const missing = Object.entries(def.requiredEnv)
127
+ .filter(([k]) => !process.env[k])
128
+ .map(([k, desc]) => ` ${k} — ${desc}`);
129
+ if (missing.length > 0) throw new Error(`Sandbox provider "${provider}" requires env vars:\n${missing.join("\n")}`);
130
+ return def.snapshotExists(snapshotId, Object.fromEntries(Object.keys(def.requiredEnv).map(k => [k, process.env[k]!])));
131
+ }
132
+
133
+ /**
134
+ * Current active-sandbox count per configured provider, for quota gauges.
135
+ * Skips providers whose required env isn't set or which don't implement
136
+ * `getActiveCount`. Errors are surfaced per-provider so one flaky provider
137
+ * doesn't silence the rest.
138
+ */
139
+ export type SandboxQuotaResult = Partial<Record<SandboxProviderName, number | Error>>;
140
+
141
+ export async function getSandboxQuotas(): Promise<SandboxQuotaResult> {
142
+ const out: SandboxQuotaResult = {};
143
+ await Promise.all(
144
+ (Object.entries(SANDBOX_PROVIDERS) as Array<[SandboxProviderName, SandboxProviderDef]>).map(async ([name, def]) => {
145
+ if (!def.getActiveCount) return;
146
+ if (Object.keys(def.requiredEnv).some(k => !process.env[k])) return;
147
+ const env = Object.fromEntries(Object.keys(def.requiredEnv).map(k => [k, process.env[k]!]));
148
+ try { out[name] = await def.getActiveCount(env); }
149
+ catch (err) { out[name] = err instanceof Error ? err : new Error(String(err)); }
150
+ }),
151
+ );
152
+ return out;
153
+ }
154
+
155
+ /**
156
+ * List every alive sandbox owned by this fleet, across all configured
157
+ * providers — the orphan-reconciler SoT. Providers without `listOwned`
158
+ * or missing env are skipped. Errors propagate per-provider so one flaky
159
+ * provider doesn't silence the rest.
160
+ */
161
+ export type OwnedSandboxResult = Partial<Record<SandboxProviderName, OwnedSandbox[] | Error>>;
162
+
163
+ export async function listOwnedSandboxes(): Promise<OwnedSandboxResult> {
164
+ const out: OwnedSandboxResult = {};
165
+ await Promise.all(
166
+ (Object.entries(SANDBOX_PROVIDERS) as Array<[SandboxProviderName, SandboxProviderDef]>).map(async ([name, def]) => {
167
+ if (!def.listOwned) return;
168
+ if (Object.keys(def.requiredEnv).some(k => !process.env[k])) return;
169
+ const env = Object.fromEntries(Object.keys(def.requiredEnv).map(k => [k, process.env[k]!]));
170
+ try { out[name] = await def.listOwned(env); }
171
+ catch (err) { out[name] = err instanceof Error ? err : new Error(String(err)); }
172
+ }),
173
+ );
174
+ return out;
175
+ }
176
+
177
+ /** Kill a sandbox by provider + native ID. Used by the orphan reconciler. */
178
+ export async function killSandboxById(provider: SandboxProviderName, sandboxId: string): Promise<void> {
179
+ const def = SANDBOX_PROVIDERS[provider];
180
+ if (!def?.reconnect) throw new Error(`Provider "${provider}" does not support reconnect (required for kill-by-id)`);
181
+ const sb = await def.reconnect(sandboxId);
182
+ await sb.kill();
183
+ }
184
+
185
+ /** Kill all sandboxes across all registered providers. Call at startup to clean up after crashes. */
186
+ export async function killAllSandboxes(
187
+ onError?: (provider: SandboxProviderName, err: unknown) => void,
188
+ ): Promise<void> {
189
+ await Promise.allSettled(
190
+ (Object.entries(SANDBOX_PROVIDERS) as Array<[SandboxProviderName, SandboxProviderDef]>)
191
+ .filter(([, def]) => def.killAll)
192
+ .map(([name, def]) => {
193
+ const env = Object.fromEntries(Object.keys(def.requiredEnv).map(k => [k, process.env[k] ?? ""]));
194
+ if (Object.keys(def.requiredEnv).some(k => !process.env[k])) return Promise.resolve();
195
+ return def.killAll!(env).catch(err => onError?.(name, err));
196
+ }),
197
+ );
198
+ }
@@ -0,0 +1,95 @@
1
+ /**
2
+ * Sandbox machine sizes + the E2B template aliases derived from them.
3
+ *
4
+ * A coarse hardware knob that maps to provider machine specs at create time:
5
+ * Vercel honours it natively via `resources.vcpus`; E2B sizing is baked into
6
+ * the template, so on E2B a size resolves to a pre-built per-size template.
7
+ */
8
+
9
+ /** Sandbox hardware SKU. Named for the actual machine spec (vCPU + RAM) rather
10
+ * than abstract t-shirt sizes. Memory is always 2048 MB per vCPU:
11
+ * 2vcpu-4gb = 2 vCPU / 4 GiB (Vercel's own default machine)
12
+ * 4vcpu-8gb = 4 vCPU / 8 GiB
13
+ * 8vcpu-16gb = 8 vCPU / 16 GiB (per-sandbox ceiling on STANDARD accounts —
14
+ * probed live: 16 & 32 vCPU 400 on dev)
15
+ * 32vcpu-64gb = 32 vCPU / 64 GiB (ENTERPRISE ONLY — standard accounts reject >8 vCPU) */
16
+ export type SandboxSize = "2vcpu-4gb" | "4vcpu-8gb" | "8vcpu-16gb" | "32vcpu-64gb";
17
+
18
+ /** SKU → Vercel vCPU count (RAM follows at 2048 MB/vCPU). */
19
+ export const SANDBOX_VCPUS: Record<SandboxSize, number> = {
20
+ "2vcpu-4gb": 2,
21
+ "4vcpu-8gb": 4,
22
+ "8vcpu-16gb": 8,
23
+ "32vcpu-64gb": 32,
24
+ };
25
+
26
+ /** SDK fallback size when neither the caller nor the deployment specifies one.
27
+ * Deliberately conservative — the OPERATIONAL default is the server's
28
+ * `SANDBOX_DEFAULT_SIZE` env var (now also `2vcpu-4gb`). Keeping the default
29
+ * small matters because Vercel rate-limits creation by vCPUs-per-window
30
+ * (`api-sandboxes-vcpus-creation`); a large default 429s bursty/simultaneous
31
+ * creates. Workloads that need more RAM/CPU declare `resources.size` on the
32
+ * workflow rather than inflating the default for everyone. */
33
+ export const DEFAULT_SANDBOX_SIZE: SandboxSize = "2vcpu-4gb";
34
+
35
+ /** The E2B sizes we pre-build a template for. E2B sizing is template-baked
36
+ * (no per-create cpu/mem knob), so honouring `resources.size` on E2B means
37
+ * ONE pre-built template per size. `32vcpu-64gb` is absent (E2B has no
38
+ * >8-vCPU equivalent). `8vcpu-16gb` is also absent: it needs 16 GiB RAM, but
39
+ * the E2B account caps memory at 8 GiB (`Template.build` 400s with
40
+ * "Memory can't be higher than 8192 MiB"). Add it back here (and rebuild the
41
+ * templates) only once the account's memory limit is raised. The register/
42
+ * invoke guards reject an unsupported E2B size before it can reach here. */
43
+ export const E2B_TEMPLATE_SIZES: readonly SandboxSize[] = [
44
+ "2vcpu-4gb",
45
+ "4vcpu-8gb",
46
+ ];
47
+
48
+ /** Is `size` one E2B can be built/booted at? `32vcpu-64gb` (no >8-vCPU E2B
49
+ * equivalent) and `8vcpu-16gb` (exceeds the account's 8 GiB memory cap) are
50
+ * not — the guards lean on this so the "no E2B equivalent" decision lives in
51
+ * exactly one place. */
52
+ export function isE2bSupportedSize(size: SandboxSize): boolean {
53
+ return E2B_TEMPLATE_SIZES.includes(size);
54
+ }
55
+
56
+ /** Machine spec for a SandboxSize, in the shape `Template.build` wants. RAM is
57
+ * always 2048 MB/vCPU, matching the size name + Vercel parity
58
+ * (`SANDBOX_VCPUS` × 2048). Used by `infra/e2b-template/build.ts` to stamp the
59
+ * per-size base + agent-env templates. */
60
+ export function e2bMachineSpec(size: SandboxSize): { cpuCount: number; memoryMB: number } {
61
+ const cpuCount = SANDBOX_VCPUS[size];
62
+ return { cpuCount, memoryMB: cpuCount * 2048 };
63
+ }
64
+
65
+ /** Stable E2B template ALIAS for the platform base at a given size
66
+ * (`agent-compose-base-<size>`). Aliases — not snapshot ids — so the refs are
67
+ * multi-account-clean: the same string resolves in any E2B account that built
68
+ * the templates. Built by `infra/e2b-template/build.ts`; the E2B provider
69
+ * boots this when a run on E2B declares no explicit `bootFrom`/template. */
70
+ export function e2bBaseTemplate(size: SandboxSize): string {
71
+ return `agent-compose-base-${size}`;
72
+ }
73
+
74
+ /** Stable E2B template ALIAS for the agent runtime (base + claude binary) at a
75
+ * given size (`agent-env-<size>`). The agent default templates boot from this
76
+ * via `bootFrom: { snapshotId: e2bAgentEnvTemplate(size) }`. Multi-account-clean
77
+ * for the same reason as `e2bBaseTemplate`. */
78
+ export function e2bAgentEnvTemplate(size: SandboxSize): string {
79
+ return `agent-env-${size}`;
80
+ }
81
+
82
+ /** Is `id` one of the platform-managed E2B template aliases — a
83
+ * `agent-compose-base-<size>` or `agent-env-<size>` for a supported size?
84
+ *
85
+ * These are stable, platform-built, multi-account-clean strings (NOT tenant
86
+ * captures), so any workflow may boot from them: the same alias resolves to
87
+ * the same platform base in every account, and there is no tenant data behind
88
+ * it to leak. The dispatch snapshot gate uses this to vouch a platform E2B
89
+ * boot alias without it having to be a team-owned capture or a published
90
+ * template's bootFrom. */
91
+ export function isPlatformE2bTemplateAlias(id: string): boolean {
92
+ return E2B_TEMPLATE_SIZES.some(
93
+ (size) => id === e2bBaseTemplate(size) || id === e2bAgentEnvTemplate(size),
94
+ );
95
+ }