@agent-compose/sdk 0.5.6 → 0.5.8

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (44) hide show
  1. package/README.md +4 -4
  2. package/dist/agent/__tests__/run-agent-liveness.test.d.ts +17 -0
  3. package/dist/agent/agent-context.d.ts +67 -0
  4. package/dist/agent/agent-loop-contract.test.d.ts +1 -0
  5. package/dist/agent/agent-loop.d.ts +2 -1
  6. package/dist/client.d.ts +129 -24
  7. package/dist/index.d.ts +9 -4
  8. package/dist/index.js +553 -89
  9. package/dist/pause/wrappers.d.ts +7 -11
  10. package/dist/runtimes/claude.d.ts +9 -1
  11. package/dist/runtimes/openai-desktop.js +548 -89
  12. package/dist/sandbox-errors.d.ts +49 -0
  13. package/dist/sandbox.d.ts +92 -13
  14. package/dist/step-invocation/protocol.d.ts +6 -0
  15. package/dist/step-invocation/types.d.ts +1 -1
  16. package/dist/types/execution-context.d.ts +1 -3
  17. package/dist/types/sandbox-environment.d.ts +1 -10
  18. package/dist/types/sandbox.d.ts +27 -3
  19. package/dist/types/workflow-metadata.d.ts +81 -13
  20. package/dist/types/workflow.d.ts +45 -10
  21. package/dist/utils/bundler.d.ts +40 -9
  22. package/dist/workflow-steps/workflow.d.ts +4 -3
  23. package/package.json +2 -2
  24. package/src/agent/agent-context.ts +212 -0
  25. package/src/agent/agent-loop.ts +78 -10
  26. package/src/agent/run-agent.ts +37 -1
  27. package/src/client.ts +232 -26
  28. package/src/index.ts +9 -4
  29. package/src/pause/wrappers.ts +7 -21
  30. package/src/runtimes/claude.ts +66 -8
  31. package/src/sandbox-errors.ts +53 -0
  32. package/src/sandbox.ts +438 -61
  33. package/src/step-invocation/invoker.ts +66 -8
  34. package/src/step-invocation/protocol.ts +9 -0
  35. package/src/step-invocation/server.ts +27 -4
  36. package/src/step-invocation/types.ts +1 -1
  37. package/src/types/execution-context.ts +1 -3
  38. package/src/types/sandbox-environment.ts +1 -11
  39. package/src/types/sandbox.ts +28 -3
  40. package/src/types/workflow-metadata.ts +91 -16
  41. package/src/types/workflow.ts +45 -12
  42. package/src/utils/bundler.ts +46 -13
  43. package/src/workflow-steps/workflow.ts +4 -3
  44. package/src/workflows/invoke-child.ts +7 -1
package/src/sandbox.ts CHANGED
@@ -8,10 +8,14 @@
8
8
  import { promises as fs } from "node:fs";
9
9
  import { dirname } from "node:path";
10
10
  import { spawn } from "node:child_process";
11
- import { Sandbox } from "e2b";
11
+ import { Sandbox, SandboxNotFoundError, RateLimitError } from "e2b";
12
12
  import { Sandbox as Desktop } from "@e2b/desktop";
13
13
  import pRetry from "p-retry";
14
+ import type { FailedAttemptError } from "p-retry";
15
+ import { SandboxUnavailableError } from "./sandbox-errors.js";
14
16
  import type { SandboxProvider, DesktopSandboxProvider, SandboxCommandResult } from "./types/sandbox.js";
17
+ import type { ConnectorRequestRules } from "./types/workflow-metadata.js";
18
+ import type { NetworkPolicy as VercelNetworkPolicy, NetworkPolicyRule as VercelNetworkPolicyRule } from "@vercel/sandbox";
15
19
 
16
20
  export type { SandboxProvider, DesktopSandboxProvider, SandboxCommandRunOptions, SandboxCommandResult } from "./types/sandbox.js";
17
21
 
@@ -34,10 +38,13 @@ export const AGENT_COMPOSE_TAG = process.env.AGENT_COMPOSE_TAG ??
34
38
  const VERCEL_VM_LIFETIME_WINDOW_MS = 6 * 60 * 60 * 1000;
35
39
 
36
40
  /**
37
- * Vercel-compatible network policy for outbound HTTPS requests.
38
- * When a sandbox makes a request matching a domain in `allow`, the Vercel
39
- * firewall injects the specified headers before forwarding — credentials never
40
- * exist inside the VM. E2B ignores this field (future self-hosted mapping TBD).
41
+ * Network policy for outbound HTTPS requests — ONE shape for every provider;
42
+ * only the enforcement point differs. When a sandbox makes a request matching
43
+ * a domain in `allow`, the egress layer injects the specified headers before
44
+ * forwarding — credentials never exist inside the VM. Vercel's firewall
45
+ * enforces this natively (`requestRules` translate to native `match` rules);
46
+ * E2B enforces it via the embedded iron-proxy, which pulls the identical
47
+ * resolved policy from the server.
41
48
  */
42
49
  export interface SandboxNetworkHeaderTransform {
43
50
  headers?: Record<string, string>;
@@ -45,6 +52,10 @@ export interface SandboxNetworkHeaderTransform {
45
52
 
46
53
  export interface SandboxNetworkAllowRule {
47
54
  transform?: SandboxNetworkHeaderTransform[];
55
+ /** Tier-2 request gate (method/path) — the transform (and, on iron-proxy,
56
+ * the request itself) only applies when the request matches. Present on
57
+ * connector-auth rules; see `ConnectorRequestRules`. */
58
+ requestRules?: ConnectorRequestRules;
48
59
  }
49
60
 
50
61
  export interface SandboxNetworkSubnetPolicy {
@@ -60,20 +71,59 @@ export type SandboxNetworkPolicy =
60
71
  subnets?: SandboxNetworkSubnetPolicy;
61
72
  };
62
73
 
74
+ /** Sandbox machine size. A coarse small/medium/large knob that maps to
75
+ * provider machine specs at create time. Vercel honours it natively via
76
+ * `resources.vcpus` (2048 MB RAM per vCPU). E2B sizing is baked into the
77
+ * template, so E2B ignores this field. Default: "small". */
78
+ /** Sandbox hardware SKU. Named for the actual machine spec (vCPU + RAM) rather
79
+ * than abstract t-shirt sizes. Memory is always 2048 MB per vCPU:
80
+ * 2vcpu-4gb = 2 vCPU / 4 GiB (Vercel's own default machine)
81
+ * 4vcpu-8gb = 4 vCPU / 8 GiB
82
+ * 8vcpu-16gb = 8 vCPU / 16 GiB (per-sandbox ceiling on STANDARD accounts —
83
+ * probed live: 16 & 32 vCPU 400 on dev)
84
+ * 32vcpu-64gb = 32 vCPU / 64 GiB (ENTERPRISE ONLY — standard accounts reject >8 vCPU) */
85
+ export type SandboxSize = "2vcpu-4gb" | "4vcpu-8gb" | "8vcpu-16gb" | "32vcpu-64gb";
86
+
87
+ /** SKU → Vercel vCPU count (RAM follows at 2048 MB/vCPU). */
88
+ export const SANDBOX_VCPUS: Record<SandboxSize, number> = {
89
+ "2vcpu-4gb": 2,
90
+ "4vcpu-8gb": 4,
91
+ "8vcpu-16gb": 8,
92
+ "32vcpu-64gb": 32,
93
+ };
94
+
95
+ /** SDK fallback size when neither the caller nor the deployment specifies one.
96
+ * Deliberately conservative — the OPERATIONAL default is chosen per-environment
97
+ * by the server via the `SANDBOX_DEFAULT_SIZE` env var (prod = Enterprise →
98
+ * `32vcpu-64gb`; dev = `8vcpu-16gb`, since the dev account caps at 8 vCPU).
99
+ * TODO(sandbox-size): the prod default is temporarily `32vcpu-64gb` for a
100
+ * memory-hungry workload — lower it back when no longer needed. */
101
+ export const DEFAULT_SANDBOX_SIZE: SandboxSize = "2vcpu-4gb";
102
+
63
103
  export interface SandboxCreateOpts {
64
104
  envs: Record<string, string>;
65
105
  metadata: Record<string, string>;
66
106
  timeoutMs: number;
67
- /** Provider-specific template/snapshot identifier. E2B: template ID; Vercel: snapshot ID. */
107
+ /** Provider-specific template/snapshot identifier. E2B: template id or
108
+ * snapshot id (omit → E2B's default base). Vercel: snapshot id (omit → node24). */
68
109
  template?: string;
69
- /** Outbound request policy. Vercel only — E2B silently ignores. */
110
+ /** Machine size. Vercel maps it to `resources.vcpus`; E2B ignores it
111
+ * (size is template-defined). Omit → `DEFAULT_SANDBOX_SIZE`. */
112
+ size?: SandboxSize;
113
+ /** Outbound request policy with header transforms — ONE shape for every
114
+ * provider; only the enforcement point differs. Vercel's firewall
115
+ * enforces + injects natively from this value. E2B enforces it via an
116
+ * EMBEDDED iron-proxy inside the VM (loopback DNS + iptables, started
117
+ * at boot), which fetches the identical resolved policy from the
118
+ * server's per-run egress-policy endpoint — so this field is not passed
119
+ * to E2B's API. See server/src/sandbox/iron-proxy.ts. */
70
120
  networkPolicy?: SandboxNetworkPolicy;
71
121
  }
72
122
 
73
123
  /**
74
124
  * What the provider reports as "currently alive" — the single input to orphan
75
125
  * reconciliation. `metadata` is best-effort: E2B populates it from sandbox
76
- * labels, Vercel leaves it empty (the API has no metadata field). Callers
126
+ * labels, Vercel from sandbox tags (max 5, stamped at create). Callers
77
127
  * that need runId correlation cross-reference `sandboxId` against their own
78
128
  * state (server: `workflow_runs.sandbox_id` + `run_agent_sandboxes.provider_sandbox_id`).
79
129
  */
@@ -92,18 +142,17 @@ interface SandboxProviderDef {
92
142
  * Returns the number of sandboxes currently running against this provider.
93
143
  * Used for quota observability — account-wide, not per-instance.
94
144
  *
95
- * E2B scopes by our AGENT_COMPOSE_TAG metadata so each machine only reports
96
- * its own sandboxes; DD aggregates across instances with `sum by {provider}`.
97
- * Vercel has no user metadata support, so it counts every running sandbox in
98
- * the configured project (assumed dedicated to agent-compose).
145
+ * Both providers scope by our AGENT_COMPOSE_TAG (E2B metadata labels,
146
+ * Vercel tags) so each fleet only reports its own sandboxes; DD aggregates
147
+ * across instances with `sum by {provider}`.
99
148
  */
100
149
  getActiveCount?: (env: Record<string, string>) => Promise<number>;
101
150
  /**
102
151
  * Provider's view of what's currently alive for our account/fleet.
103
152
  * The reconciliation SoT — a sandbox missing from this list IS dead,
104
153
  * regardless of what our DB says. Implemented by listing the provider's
105
- * running sandboxes; E2B filters by metadata tag, Vercel lists the whole
106
- * project (it has no metadata search). 200-row cap applies to Vercel.
154
+ * running sandboxes; E2B filters by metadata tag, Vercel by the `executor`
155
+ * sandbox tag.
107
156
  */
108
157
  listOwned?: (env: Record<string, string>) => Promise<OwnedSandbox[]>;
109
158
  /**
@@ -147,6 +196,80 @@ export function makeSandboxProvider(sb: Sandbox | Desktop): SandboxProvider {
147
196
  };
148
197
  }
149
198
 
199
+ /** E2B caps a sandbox's lifetime per plan (Hobby 1h, Pro 24h) and 400s when the
200
+ * create timeout exceeds it — and our `AC_SANDBOX_DEADLINE` (4.5h) tops Hobby.
201
+ * Clamp the E2B create timeout to this max (raise on Pro via `E2B_MAX_SANDBOX_MS`).
202
+ * Read at call time so the workflow bundle stays env-free. */
203
+ function e2bMaxSandboxMs(): number {
204
+ return Number(process.env.E2B_MAX_SANDBOX_MS) || 60 * 60 * 1000;
205
+ }
206
+
207
+ /** Is this a FLEETING sandbox-provider race worth retrying — the sandbox being
208
+ * paused / reclaimed / stopped under load, an upstream 5xx/429, or a dropped
209
+ * connection — vs a PERMANENT failure that must fail fast?
210
+ *
211
+ * E2B raises TYPED errors, so prefer `instanceof` (upgrade-stable — survives any
212
+ * message rewording): a reconnect/snapshot that hits a sandbox mid-pause throws
213
+ * `SandboxNotFoundError`, rate limits throw `RateLimitError`. (Permanent
214
+ * template/snapshot-not-found only occurs on CREATE, which is no longer retried.)
215
+ * The message regex is the FALLBACK for the untyped Vercel transport + raw socket
216
+ * errors — and deliberately omits bare "timeout"/"network": E2B's permanent
217
+ * plan-cap 400 ("timeout … exceeds the maximum sandbox lifetime") contains
218
+ * "timeout" and must fail fast, not retry 4×. */
219
+ function isTransientSandboxError(error: unknown): boolean {
220
+ if (error instanceof SandboxNotFoundError || error instanceof RateLimitError) return true;
221
+ const message = error instanceof Error ? error.message : String(error ?? "");
222
+ // The 5xx match is anchored to a status-code context ("status 503",
223
+ // "502 Bad Gateway", …) — a bare \b5\d\d\b would classify ANY standalone
224
+ // 500-599 number in a message (durations, row counts) as retryable.
225
+ return /paus|stopp|resum|transition|status(?:\s+code)?[:\s]+5\d\d\b|\b5\d\d\s+(?:internal|bad gateway|service|gateway)|\b429\b|ECONNRESET|ECONNREFUSED|ETIMEDOUT|socket hang up|fetch failed/i
226
+ .test(message);
227
+ }
228
+
229
+ /** The ONE place the sandbox retry policy lives. Provisioning, reconnecting, and
230
+ * snapshotting all race with the sandbox being paused/reclaimed; this runs the call
231
+ * through p-retry's generic backoff, retrying ONLY the transient race (and failing
232
+ * fast otherwise). reconnectSandbox and provider.snapshot() all go through it, so
233
+ * no provider re-implements the pattern or can forget it. `shouldRetry` is
234
+ * overridable for call sites where the default classification is wrong (see
235
+ * reconnectSandbox's SandboxNotFoundError handling). */
236
+ function withSandboxRetry<T>(
237
+ fn: () => Promise<T>,
238
+ shouldRetry: (error: FailedAttemptError) => boolean = isTransientSandboxError,
239
+ ): Promise<T> {
240
+ return pRetry(fn, { retries: 4, minTimeout: 400, factor: 2, shouldRetry });
241
+ }
242
+
243
+ /** Add snapshot-retry to a freshly-obtained provider. create/reconnect are retried at
244
+ * the registry call (`createSandbox`/`reconnectSandbox`); snapshot is a method, so it
245
+ * gets the same `withSandboxRetry` here. `commands.run` is deliberately NOT wrapped —
246
+ * retrying user-code execution could double-apply side effects, so step execution
247
+ * relies on the engine's pre-step sandbox recovery instead. No-op without a snapshot. */
248
+ function withSnapshotRetry(p: SandboxProvider): SandboxProvider {
249
+ if (!p.snapshot) return p;
250
+ const snapshot = p.snapshot.bind(p);
251
+ return { ...p, snapshot: () => withSandboxRetry(snapshot) };
252
+ }
253
+
254
+ /** E2B base provider + live-filesystem snapshot. E2B's `createSnapshot()` captures
255
+ * the running sandbox as a persistent snapshot whose id is usable as a
256
+ * `Sandbox.create()` source and outlives the origin sandbox — so E2B reaches
257
+ * snapshot / `bootFrom` parity with Vercel, and snapshot-backed `ctx.pause` works
258
+ * on E2B. (Desktop intentionally omits this — it is registry-only, not selectable.) */
259
+ function makeE2bSandboxProvider(sb: Sandbox): SandboxProvider {
260
+ return {
261
+ ...makeSandboxProvider(sb),
262
+ async snapshot() {
263
+ // Raw capture — the transient pause/reclaim race is retried centrally:
264
+ // createSandbox/reconnectSandbox wrap every provider's snapshot() in
265
+ // withSandboxRetry (via withSnapshotRetry). create itself is deliberately
266
+ // NOT retried — see createSandbox.
267
+ const { snapshotId } = await sb.createSnapshot();
268
+ return { snapshotId };
269
+ },
270
+ };
271
+ }
272
+
150
273
  export function makeDesktopSandboxProvider(sb: Desktop): DesktopSandboxProvider {
151
274
  return {
152
275
  ...makeSandboxProvider(sb),
@@ -209,6 +332,88 @@ export async function parseSseExecStream(
209
332
 
210
333
  // ── Vercel helpers ────────────────────────────────────────────────────────────
211
334
 
335
+ /** Matches a path containing a dot-dot segment — literal (`/../`) or
336
+ * percent-encoded (`%2e%2e`, `%2f` boundaries). `DOT_SEGMENT_PATH_RE2` is
337
+ * the RE2 form handed to Vercel's matcher: wrapped in `.*` so it behaves
338
+ * identically under search and full-match semantics (RE2 has no lookaround,
339
+ * so the guard is a positive match on the traversal pattern, used as a
340
+ * credential-stripping first rule). `DOT_SEGMENT_RE` is the local JS twin
341
+ * used to validate DECLARED prefixes. */
342
+ export const DOT_SEGMENT_PATH_RE2 = ".*(?:^|/|%2[fF])(?:\\.|%2[eE]){2}(?:/|%2[fF]|$).*";
343
+ const DOT_SEGMENT_RE = /(?:^|\/|%2f)(?:\.|%2e){2}(?:\/|%2f|$)/i;
344
+
345
+ /** A Tier-2 path gate decides where a brokered credential may ride, so a
346
+ * malformed declared prefix is a config bug that must fail closed, not a
347
+ * string we forward verbatim to a third-party matcher. */
348
+ function assertSafePathPrefix(domain: string, prefix: string): void {
349
+ if (!prefix.startsWith("/") || DOT_SEGMENT_RE.test(prefix)) {
350
+ throw new Error(
351
+ `Invalid Tier-2 pathPrefix ${JSON.stringify(prefix)} for "${domain}": prefixes must start with "/" and contain no dot-segments`,
352
+ );
353
+ }
354
+ }
355
+
356
+ /**
357
+ * Translate our policy shape into Vercel's native `NetworkPolicy`.
358
+ *
359
+ * The shapes are identical except for Tier-2: our `requestRules`
360
+ * (`{ methods, pathPrefixes }`) become Vercel `match` rules
361
+ * (`{ method, path: { startsWith } }`) — one rule per path prefix, since a
362
+ * Vercel matcher carries a single path.
363
+ *
364
+ * Dot-segment defense: prefix matching is raw — `/repos/acme/../../user`
365
+ * starts with `/repos/acme/` but the origin normalizes it to `/user`, riding
366
+ * the credential outside its declared scope. We don't get to assume the
367
+ * enforcer RFC-3986-normalizes before matching, so every domain with a path
368
+ * gate gets a transform-free FIRST rule matching dot-segment paths
369
+ * (first-match-wins ⇒ such requests go out credential-free), and declared
370
+ * prefixes themselves are validated (must start with "/", no dot-segments).
371
+ *
372
+ * Accepted divergence from iron-proxy (E2B): off-gate requests to an allowed
373
+ * domain pass WITHOUT the transform (Vercel terminates TLS only for
374
+ * transform-bearing domains and applies first-match-wins) — iron-proxy
375
+ * REFUSES them instead. Reachability differs; the property that matters is
376
+ * preserved on both substrates: the brokered credential rides only declared
377
+ * method/path combinations. Pinned by vercel-network-policy.test.ts.
378
+ */
379
+ export function toVercelNetworkPolicy(policy: SandboxNetworkPolicy): VercelNetworkPolicy {
380
+ if (typeof policy === "string" || !policy.allow || Array.isArray(policy.allow)) {
381
+ return policy as VercelNetworkPolicy;
382
+ }
383
+ const allow: Record<string, VercelNetworkPolicyRule[]> = {};
384
+ for (const [domain, rules] of Object.entries(policy.allow)) {
385
+ const translated = rules.flatMap((rule): VercelNetworkPolicyRule[] => {
386
+ const { requestRules, ...rest } = rule;
387
+ const methods = requestRules?.methods ?? [];
388
+ const prefixes = requestRules?.pathPrefixes ?? [];
389
+ if (methods.length === 0 && prefixes.length === 0) return [rest];
390
+ const methodMatch = methods.length > 0 ? { method: methods } : {};
391
+ if (prefixes.length === 0) return [{ ...rest, match: methodMatch }];
392
+ for (const prefix of prefixes) assertSafePathPrefix(domain, prefix);
393
+ return prefixes.map((prefix) => ({ ...rest, match: { ...methodMatch, path: { startsWith: prefix } } }));
394
+ });
395
+ const hasPathGate = rules.some((rule) => (rule.requestRules?.pathPrefixes?.length ?? 0) > 0);
396
+ allow[domain] = hasPathGate
397
+ ? [{ match: { path: { regex: DOT_SEGMENT_PATH_RE2 } } }, ...translated]
398
+ : translated;
399
+ }
400
+ return { ...policy, allow };
401
+ }
402
+
403
+ /** Vercel caps sandbox tags at 5. Build the tag set with the fleet `executor`
404
+ * tag always present and always winning over caller metadata — `listOwned` /
405
+ * orphan reconciliation scope on it, so a clobbered or dropped executor tag
406
+ * makes the sandbox invisible to cleanup. When metadata overflows the cap,
407
+ * keep executor + the lexicographically-first 4 metadata keys, so what gets
408
+ * dropped is deterministic rather than dependent on object insertion order. */
409
+ export const VERCEL_MAX_TAGS = 5;
410
+ export function buildVercelTags(metadata: Record<string, string>): Record<string, string> {
411
+ const tags: Record<string, string> = { ...metadata, executor: AGENT_COMPOSE_TAG };
412
+ const extras = Object.keys(tags).filter((k) => k !== "executor").sort();
413
+ for (const key of extras.slice(VERCEL_MAX_TAGS - 1)) delete tags[key];
414
+ return tags;
415
+ }
416
+
212
417
  // eslint-disable-next-line @typescript-eslint/no-explicit-any
213
418
  function makeVercelSandboxProvider(sb: any, globalEnvs?: Record<string, string>): SandboxProvider {
214
419
  // Vercel's Sandbox.create({ env }) does NOT flow to runCommand subprocesses — they start fresh shells.
@@ -217,8 +422,21 @@ function makeVercelSandboxProvider(sb: any, globalEnvs?: Record<string, string>)
217
422
  const mergeEnvs = (cmdEnvs?: Record<string, string>) =>
218
423
  globalEnvs ? { ...globalEnvs, ...cmdEnvs } : cmdEnvs;
219
424
 
425
+ // Wrap a sandbox-side failure as `SandboxUnavailableError`. `retryable` is
426
+ // decided purely by *where* the error was caught, not by inspecting its
427
+ // shape: true before the runner launched any user code (file write / command
428
+ // launch / reconnect — safe to re-provision and retry), false once the
429
+ // runner streamed (user code may have run). The original message is
430
+ // preserved for server-side diagnostics.
431
+ const asUnavailable = (err: unknown, retryable: boolean): never => {
432
+ throw new SandboxUnavailableError(err instanceof Error ? err.message : String(err), { retryable, sandboxId: sb.name });
433
+ };
434
+
220
435
  return {
221
- sandboxId: sb.sandboxId,
436
+ // @vercel/sandbox 2.x is name-keyed: `name` is the stable identifier
437
+ // (`Sandbox.get({ name })`); the v1 `sandboxId` getter is gone. Our
438
+ // provider-facing field keeps its name — it's "the provider-native id".
439
+ sandboxId: sb.name,
222
440
  commands: {
223
441
  async run(cmd, opts) {
224
442
  const signal = opts?.timeoutMs ? AbortSignal.timeout(opts.timeoutMs) : undefined;
@@ -226,34 +444,75 @@ function makeVercelSandboxProvider(sb: any, globalEnvs?: Record<string, string>)
226
444
  let stderr = "";
227
445
  // `sudo: true` is Vercel's native root flag — applied to the `sh`
228
446
  // invocation so the whole shell (and its children) runs as root.
447
+ // A launch failure means the command never started — nothing
448
+ // user-side ran, so any error here is safe to re-provision + retry.
229
449
  // eslint-disable-next-line @typescript-eslint/no-explicit-any
230
- const handle: any = await sb.runCommand({ cmd: "sh", args: ["-c", cmd], cwd: opts?.cwd, env: mergeEnvs(opts?.envs), detached: true, signal, ...(opts?.sudo ? { sudo: true } : {}) });
231
- // Reconnect to the already-running command on transient stream failures (e.g. BrotliDecompressionError).
232
- // `h.logs()` replays from the start on reconnect, so reset accumulators per
233
- // attempt to avoid double-counting. The streaming callbacks may still fire
234
- // for duplicate chunks during retries — an acceptable tradeoff for resilience.
235
- await pRetry(async (attempt) => {
236
- // eslint-disable-next-line @typescript-eslint/no-explicit-any
237
- const h: any = attempt === 1 ? handle : await sb.getCommand(handle.cmdId);
238
- if (attempt > 1) { stdout = ""; stderr = ""; }
239
- for await (const log of h.logs()) {
240
- if (log.stream === "stdout") { stdout += log.data; opts?.onStdout?.(log.data); }
241
- else { stderr += log.data; opts?.onStderr?.(log.data); }
242
- }
243
- }, { retries: 3, minTimeout: 1_000, factor: 2 });
244
- const finished = await handle.wait();
245
- return { exitCode: finished.exitCode, stdout, stderr };
450
+ const handle: any = await sb.runCommand({ cmd: "sh", args: ["-c", cmd], cwd: opts?.cwd, env: mergeEnvs(opts?.envs), detached: true, signal, ...(opts?.sudo ? { sudo: true } : {}) })
451
+ .catch((err: unknown) => asUnavailable(err, true));
452
+ // Stream + wait. Reconnect to the already-running command on transient
453
+ // stream failures (e.g. BrotliDecompressionError). `h.logs()` replays
454
+ // from the start on reconnect, so reset accumulators per attempt to
455
+ // avoid double-counting.
456
+ const collect = async () => {
457
+ await pRetry(async (attempt) => {
458
+ // eslint-disable-next-line @typescript-eslint/no-explicit-any
459
+ const h: any = attempt === 1 ? handle : await sb.getCommand(handle.cmdId);
460
+ if (attempt > 1) { stdout = ""; stderr = ""; }
461
+ for await (const log of h.logs()) {
462
+ if (log.stream === "stdout") { stdout += log.data; opts?.onStdout?.(log.data); }
463
+ else { stderr += log.data; opts?.onStderr?.(log.data); }
464
+ }
465
+ }, { retries: 3, minTimeout: 1_000, factor: 2 });
466
+ const finished = await handle.wait();
467
+ return { exitCode: finished.exitCode, stdout, stderr };
468
+ };
469
+ // `timeoutMs` MUST be authoritative. The AbortSignal alone is not: a
470
+ // command that RUNS but emits nothing (e.g. a wedged `archil checkout`
471
+ // on a blocked data plane) leaves `logs()`/`wait()` pending and the
472
+ // abort never interrupts the await — the call hangs for the activity's
473
+ // whole multi-hour ceiling. Race a hard client-side deadline so a hung
474
+ // command fails fast and the caller's degrade/retry policy takes over.
475
+ let timer: ReturnType<typeof setTimeout> | undefined;
476
+ const deadline = opts?.timeoutMs
477
+ ? new Promise<never>((_, reject) => {
478
+ timer = setTimeout(
479
+ () => reject(new Error(`command timed out after ${opts.timeoutMs}ms: ${cmd.slice(0, 200)}`)),
480
+ opts.timeoutMs);
481
+ })
482
+ : undefined;
483
+ try {
484
+ return await (deadline ? Promise.race([collect(), deadline]) : collect());
485
+ } catch (err) {
486
+ // A timeout (hard deadline or AbortSignal) is a COMMAND overrunning
487
+ // its budget, not the sandbox dying — the VM is alive. Surface it as
488
+ // a plain error so the caller's own retry/deadline policy decides
489
+ // (classifying it as terminal sandbox-unavailable failed whole runs
490
+ // over a single slow mount probe, observed live).
491
+ if (err instanceof Error && err.message.startsWith("command timed out after")) throw err;
492
+ if (signal?.aborted) throw new Error(`command timed out after ${opts?.timeoutMs}ms: ${cmd.slice(0, 200)}`);
493
+ return asUnavailable(err, false);
494
+ } finally {
495
+ if (timer) clearTimeout(timer);
496
+ }
246
497
  },
247
498
  },
248
499
  files: {
249
500
  async write(path: string, content: string) {
250
- await sb.writeFiles([{ path, content }]);
501
+ // File writes happen before the runner launches (step input / pause
502
+ // resolution files), so a failure here means nothing ran — retryable.
503
+ await sb.writeFiles([{ path, content }]).catch((err: unknown) => asUnavailable(err, true));
251
504
  },
252
505
  },
253
506
  // Propagate errors — `killAllRunSandboxes` relies on kill failures being
254
507
  // observable so it can leave `sandbox_id` set for `findOrphanedSandboxes`
255
508
  // to retry on next boot. Swallowing here makes the orphan retry loop blind.
256
- async kill() { await sb.stop(); },
509
+ //
510
+ // kill must DESTROY: in @vercel/sandbox 2.x stop() only halts the VM — the
511
+ // name-keyed sandbox record persists (reserving the name; commands against
512
+ // a stale handle can transparently resume a stopped VM). delete() is the
513
+ // destroying call ("after deletion the instance becomes inert"), so stop
514
+ // then delete.
515
+ async kill() { await sb.stop(); await sb.delete(); },
257
516
  // Vercel native snapshot — used by sandbox-environments to capture the
258
517
  // configured VM after the customer's `setup()` completes. `expiration: 0`
259
518
  // is Vercel's "never expires" value; env snapshots are long-lived by
@@ -267,6 +526,15 @@ function makeVercelSandboxProvider(sb: any, globalEnvs?: Record<string, string>)
267
526
  ...(sizeBytes !== undefined ? { sizeBytes } : {}),
268
527
  };
269
528
  },
529
+ // Push a freshly-resolved egress policy onto the live sandbox —
530
+ // @vercel/sandbox 2.x `update({ networkPolicy })`. Lets the server
531
+ // re-resolve the run policy (re-minting connector access tokens)
532
+ // before each step instead of living with the policy baked at create.
533
+ // E2B leaves this undefined: its enforcement (iron-proxy) lives inside
534
+ // the VM and is configured at boot.
535
+ async updateNetworkPolicy(policy) {
536
+ await sb.update({ networkPolicy: toVercelNetworkPolicy(policy) });
537
+ },
270
538
  };
271
539
  }
272
540
 
@@ -329,10 +597,33 @@ const SANDBOX_PROVIDERS: Record<string, SandboxProviderDef> = {
329
597
  const { Sandbox: VercelSandbox } = await import("@vercel/sandbox");
330
598
  const creds = { token: env.VERCEL_ACCESS_TOKEN, teamId: env.VERCEL_TEAM_ID, projectId: env.VERCEL_PROJECT_ID };
331
599
  // `template` is a snapshot id (runtime is baked in); absent → fresh node24.
332
- const np = opts.networkPolicy;
333
- const sb = await VercelSandbox.create(opts.template
334
- ? { source: { type: "snapshot" as const, snapshotId: opts.template }, timeout: opts.timeoutMs, env: opts.envs, ...(np ? { networkPolicy: np } : {}), ...creds }
335
- : { runtime: "node24" as const, timeout: opts.timeoutMs, env: opts.envs, ...(np ? { networkPolicy: np } : {}), ...creds },
600
+ const np = opts.networkPolicy ? { networkPolicy: toVercelNetworkPolicy(opts.networkPolicy) } : {};
601
+ // Caller metadata rides as Vercel tags — `listOwned` filters on the
602
+ // `executor` key, mirroring E2B's metadata-tag scoping. buildVercelTags
603
+ // guarantees executor wins over metadata and survives the 5-tag cap.
604
+ const tags = buildVercelTags(opts.metadata);
605
+ // No explicit template/bootFrom → VERCEL_DEFAULT_SNAPSHOT if set (the
606
+ // platform agent-env base: claude, archil, rtk, bun, agentc CLI, the
607
+ // SDK in /workspace/node_modules, /ac:* skills, AGENTS.md — built by
608
+ // .agentc/environments/agent-env.ts), else raw node24. The E2B
609
+ // analogue is E2B_DEFAULT_TEMPLATE below. Because `saveLatest`/
610
+ // `reuse` captures snapshot the whole filesystem, everything a run
611
+ // bakes on top of this base chains from it — the base tools are in
612
+ // every derived snapshot for free.
613
+ const tmpl = opts.template ?? process.env.VERCEL_DEFAULT_SNAPSHOT;
614
+ // Machine size → vCPUs (RAM auto-follows at 2048 MB/vCPU). Always sent
615
+ // explicitly so the spec is deterministic and self-documenting rather
616
+ // than riding Vercel's implicit default; "small" maps to that default
617
+ // anyway, so existing runs are unchanged.
618
+ const resources = { vcpus: SANDBOX_VCPUS[opts.size ?? DEFAULT_SANDBOX_SIZE] };
619
+ // `persistent: false` — @vercel/sandbox 2.x creates persistent-by-default
620
+ // sandboxes: stop() auto-snapshots (and keeps billing storage), commands
621
+ // transparently resume a stopped VM, and kill no longer destroys. Our
622
+ // sandboxes are single-run and lifecycle-managed by the engine (explicit
623
+ // snapshot() / kill()), so opt out in BOTH branches.
624
+ const sb = await VercelSandbox.create(tmpl
625
+ ? { source: { type: "snapshot" as const, snapshotId: tmpl }, timeout: opts.timeoutMs, env: opts.envs, tags, persistent: false, resources, ...np, ...creds }
626
+ : { runtime: "node24" as const, timeout: opts.timeoutMs, env: opts.envs, tags, persistent: false, resources, ...np, ...creds },
336
627
  );
337
628
  // Pass envs as globalEnvs so they're injected into every runCommand subprocess.
338
629
  // (Vercel's Sandbox.create env parameter does not flow to runCommand subprocesses.)
@@ -345,33 +636,58 @@ const SANDBOX_PROVIDERS: Record<string, SandboxProviderDef> = {
345
636
  teamId: process.env.VERCEL_TEAM_ID ?? "",
346
637
  projectId: process.env.VERCEL_PROJECT_ID ?? "",
347
638
  };
348
- return makeVercelSandboxProvider(await VercelSandbox.get({ sandboxId, ...creds }));
639
+ // Reconnecting happens before any user code runs, so any failure here
640
+ // (sandbox already stopping, API blip) is a retryable sandbox-unavailable
641
+ // — the workflow re-provisions a fresh one rather than failing the run.
642
+ // 2.x is name-keyed: the id we persisted IS the sandbox name.
643
+ const sb = await VercelSandbox.get({ name: sandboxId, ...creds }).catch((err: unknown) => {
644
+ throw new SandboxUnavailableError(err instanceof Error ? err.message : String(err), { retryable: true, sandboxId });
645
+ });
646
+ return makeVercelSandboxProvider(sb);
349
647
  },
350
648
  getActiveCount: async (env) => (await SANDBOX_PROVIDERS.vercel.listOwned!(env)).length,
351
649
  listOwned: async (env) => {
352
650
  const { Sandbox: VercelSandbox } = await import("@vercel/sandbox");
353
651
  const creds = { token: env.VERCEL_ACCESS_TOKEN, teamId: env.VERCEL_TEAM_ID, projectId: env.VERCEL_PROJECT_ID };
354
- // Bounds the scan to `VERCEL_VM_LIFETIME_WINDOW_MS` — Pro/Enterprise caps
355
- // VM lifetime at 5h, +1h slack for clock skew and `extendTimeout()`.
356
- // Without this, projects with thousands of historical terminal sandboxes
357
- // page forever (10k+ in v0.6.28). Truncating silently would hide orphans,
358
- // so we throw at the 50-page cap instead. Each page is wrapped in pRetry
359
- // so a transient 5xx/429/network blip doesn't cost a reconcile tick.
652
+ // Scoped to our fleet by the `executor` tag (stamped at create) and
653
+ // bounded to `VERCEL_VM_LIFETIME_WINDOW_MS` — Pro/Enterprise caps VM
654
+ // lifetime at 5h, +1h slack for clock skew and `extendTimeout()`. The
655
+ // sort is pinned to createdAt-desc explicitly — the early-stop below is
656
+ // only correct under that order, and the server's default sort is not a
657
+ // contract we can rely on — so we stop at the first item older than the
658
+ // window instead of paging through thousands of historical terminal
659
+ // sandboxes (10k+ in v0.6.28). Truncating silently would hide orphans,
660
+ // so we throw at the page cap instead — 200 pages × the API's 50-item
661
+ // limit preserves the previous 10k scan bound (v1 paged 200 × 50). Each
662
+ // page is wrapped in pRetry so a transient 5xx/429/network blip doesn't
663
+ // cost a reconcile tick.
664
+ //
665
+ // 2.x cutover caveat: sandboxes created by the pre-2.x code carry no
666
+ // tags, so for up to VERCEL_VM_LIFETIME_WINDOW_MS after a deploy they
667
+ // are invisible here (and their persisted v1 sandbox ids don't resolve
668
+ // via `Sandbox.get({ name })`). Self-limiting — the provider's 5h
669
+ // lifetime cap expires them — but drain in-flight Vercel runs before
670
+ // deploying if losing their sandbox state matters.
360
671
  const since = Date.now() - VERCEL_VM_LIFETIME_WINDOW_MS;
361
672
  const out: OwnedSandbox[] = [];
362
- let until: number | undefined;
363
- for (let page = 0; page < 50; page++) {
364
- const { json } = await pRetry(
365
- () => VercelSandbox.list({ ...creds, limit: 200, since, ...(until !== undefined ? { until } : {}) }),
673
+ let cursor: string | undefined;
674
+ for (let page = 0; page < 200; page++) {
675
+ const result = await pRetry(
676
+ // limit is capped at 50 by the v2 list API — 51+ answers 400
677
+ // (probed live 2026-06-10; not documented in the SDK types).
678
+ () => VercelSandbox.list({ ...creds, limit: 50, sortBy: "createdAt", sortOrder: "desc", tags: { executor: AGENT_COMPOSE_TAG }, ...(cursor !== undefined ? { cursor } : {}) }),
366
679
  { retries: 3, minTimeout: 500, factor: 2 },
367
680
  );
368
- for (const sb of json.sandboxes) if (sb.status === "running") {
369
- out.push({ sandboxId: sb.id, createdAt: new Date(sb.createdAt), metadata: {} });
681
+ for (const sb of result.sandboxes) {
682
+ if (sb.createdAt < since) return out;
683
+ if (sb.status === "running") {
684
+ out.push({ sandboxId: sb.name, createdAt: new Date(sb.createdAt), metadata: sb.tags ?? {} });
685
+ }
370
686
  }
371
- if (json.pagination.next === null) return out;
372
- until = json.pagination.next;
687
+ if (result.pagination.next === null) return out;
688
+ cursor = result.pagination.next;
373
689
  }
374
- throw new Error("Vercel listSandboxes exceeded 50 pages within the lifetime window — refuse to silently truncate");
690
+ throw new Error("Vercel listSandboxes exceeded 200 pages (10k sandboxes) within the lifetime window — refuse to silently truncate");
375
691
  },
376
692
  deleteSnapshot: async (snapshotId, env) => {
377
693
  const { Snapshot } = await import("@vercel/sandbox");
@@ -382,11 +698,45 @@ const SANDBOX_PROVIDERS: Record<string, SandboxProviderDef> = {
382
698
  },
383
699
  "e2b": {
384
700
  requiredEnv: { E2B_API_KEY: "E2B API key — e2b.dev/dashboard" },
385
- create: async ({ template, ...sandboxOpts }) => {
386
- if (!template) throw new Error("E2B provider requires an explicit `template` (Dockerfile-based — no default base image)");
387
- return makeSandboxProvider(await Sandbox.create(template, sandboxOpts));
701
+ // `size` dropped here: E2B has no create-time resource knob (specs are
702
+ // baked into the template/snapshot), so sizing on E2B = a pre-sized
703
+ // template, not this field. Honoured only on Vercel.
704
+ create: async ({ template, timeoutMs, networkPolicy: _np, size: _size, ...rest }) => {
705
+ // `template` is an E2B template id, a snapshot id (a valid create source
706
+ // that persists beyond its origin sandbox — bootFrom parity), or absent →
707
+ // E2B's default base (Debian + node + npm/python/git), mirroring Vercel's
708
+ // node24 default. With no explicit template/bootFrom, fall back to
709
+ // E2B_DEFAULT_TEMPLATE if set (a prebuilt base — e.g. one with the claude CLI
710
+ // + chromium baked in and more RAM than the stock 482MB base), the
711
+ // per-deployment analogue of Vercel's node24; else E2B's stock base.
712
+ // Self-provisioning runtimes (claude/codex/amp via bootFrom:"reuse") install
713
+ // their CLI on the base and cache it in the captured snapshot, so a
714
+ // template-less first run boots, installs, snapshots. Clamp the lifetime to
715
+ // E2B's per-plan cap (see e2bMaxSandboxMs) so a 4.5h AC_SANDBOX_DEADLINE
716
+ // doesn't 400 on smaller plans.
717
+ //
718
+ // `networkPolicy` is deliberately NOT passed to E2B's API. E2B has no
719
+ // header-injecting firewall; enforcement happens INSIDE the VM via
720
+ // the embedded iron-proxy started at sandbox boot, which pulls the
721
+ // identical resolved policy from the server. One policy shape, two
722
+ // enforcement points. See server/src/sandbox/iron-proxy.ts.
723
+ const maxMs = e2bMaxSandboxMs();
724
+ if (timeoutMs > maxMs) {
725
+ // The clamp turns E2B's explicit create-time 400 into a silent mid-run
726
+ // sandbox death at the cap — say so up front instead of letting a run
727
+ // budgeted past the cap discover it as a terminal infra failure.
728
+ console.warn(
729
+ `[sandbox] e2b create timeout clamped: requested ${timeoutMs}ms exceeds the plan cap ${maxMs}ms — ` +
730
+ `the sandbox dies at the cap, not the requested deadline. Raise E2B_MAX_SANDBOX_MS on plans that allow more.`,
731
+ );
732
+ }
733
+ const sandboxOpts = { ...rest, timeoutMs: Math.min(timeoutMs, maxMs) };
734
+ const tmpl = template ?? process.env.E2B_DEFAULT_TEMPLATE;
735
+ return makeE2bSandboxProvider(
736
+ await (tmpl ? Sandbox.create(tmpl, sandboxOpts) : Sandbox.create(sandboxOpts)),
737
+ );
388
738
  },
389
- reconnect: async (sandboxId) => makeSandboxProvider(
739
+ reconnect: async (sandboxId) => makeE2bSandboxProvider(
390
740
  await Sandbox.connect(sandboxId, { apiKey: process.env.E2B_API_KEY ?? "", timeoutMs: 60 * 60 * 1000 }),
391
741
  ),
392
742
  killAll: async () => {
@@ -413,12 +763,25 @@ const SANDBOX_PROVIDERS: Record<string, SandboxProviderDef> = {
413
763
  }
414
764
  return out;
415
765
  },
766
+ deleteSnapshot: async (snapshotId, env) => {
767
+ // E2B snapshots are team-scoped; delete by id. Best-effort like Vercel's.
768
+ await Sandbox.deleteSnapshot(snapshotId, { apiKey: env.E2B_API_KEY });
769
+ },
416
770
  },
417
771
  "e2b-desktop": {
418
772
  requiredEnv: { E2B_API_KEY: "E2B API key — e2b.dev/dashboard" },
419
- create: async ({ template, ...sandboxOpts }) => {
773
+ // Mirror "e2b": `networkPolicy` is deliberately NOT passed to the SDK —
774
+ // its transforms carry live connector tokens, and a credential-bearing
775
+ // object must never cross into a third-party library's option bag.
776
+ // Lifetime is clamped to E2B's per-plan cap the same way.
777
+ // `size` dropped here: E2B has no create-time resource knob (specs are
778
+ // baked into the template/snapshot), so sizing on E2B = a pre-sized
779
+ // template, not this field. Honoured only on Vercel.
780
+ create: async ({ template, timeoutMs, networkPolicy: _np, size: _size, ...rest }) => {
420
781
  if (!template) throw new Error("E2B Desktop provider requires an explicit `template` (Dockerfile-based — no default base image)");
421
- return makeDesktopSandboxProvider(await Desktop.create(template, sandboxOpts));
782
+ return makeDesktopSandboxProvider(
783
+ await Desktop.create(template, { ...rest, timeoutMs: Math.min(timeoutMs, e2bMaxSandboxMs()) }),
784
+ );
422
785
  },
423
786
  },
424
787
  };
@@ -431,14 +794,28 @@ export async function createSandbox(provider: SandboxProviderName, opts: Sandbox
431
794
  .filter(([key]) => !process.env[key])
432
795
  .map(([key, desc]) => ` ${key} — ${desc}`);
433
796
  if (missing.length > 0) throw new Error(`Sandbox provider "${provider}" requires env vars:\n${missing.join("\n")}`);
434
- return def.create(opts, Object.fromEntries(Object.keys(def.requiredEnv).map(k => [k, process.env[k]!])));
797
+ const env = Object.fromEntries(Object.keys(def.requiredEnv).map((k) => [k, process.env[k]!]));
798
+ // Provision once — do NOT wrap create in withSandboxRetry: a create whose RESPONSE
799
+ // is lost (client-side timeout after the provider already provisioned) would, on
800
+ // retry, mint a SECOND billed sandbox and orphan the first. Provision failures are
801
+ // re-attempted at the workflow layer. The result is still wrapped so snapshot()
802
+ // retries the transient pause/reclaim race (which is where the race actually lives).
803
+ return withSnapshotRetry(await def.create(opts, env));
435
804
  }
436
805
 
437
806
  /** Reconnect to an existing sandbox by provider + provider-native sandbox ID. */
438
807
  export async function reconnectSandbox(provider: SandboxProviderName, sandboxId: string): Promise<SandboxProvider> {
439
808
  const def = SANDBOX_PROVIDERS[provider];
440
809
  if (!def?.reconnect) throw new Error(`Provider "${provider}" does not support reconnect`);
441
- return def.reconnect(sandboxId);
810
+ // Retry the transient reconnect/resume race, then add snapshot-retry to the
811
+ // result. On RECONNECT, SandboxNotFoundError usually means the sandbox is
812
+ // genuinely gone (killed / lifetime-expired) — permanent — but it can also
813
+ // be the transient mid-pause race, so it gets exactly ONE retry instead of
814
+ // burning the full 4-attempt backoff before surfacing.
815
+ return withSnapshotRetry(await withSandboxRetry(
816
+ () => def.reconnect!(sandboxId),
817
+ (error) => error instanceof SandboxNotFoundError ? error.attemptNumber <= 1 : isTransientSandboxError(error),
818
+ ));
442
819
  }
443
820
 
444
821
  /** Delete a snapshot by id on the named provider. No live sandbox needed. */