@agent-compose/sdk 0.5.6 → 0.5.7

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/sandbox.ts CHANGED
@@ -8,10 +8,14 @@
8
8
  import { promises as fs } from "node:fs";
9
9
  import { dirname } from "node:path";
10
10
  import { spawn } from "node:child_process";
11
- import { Sandbox } from "e2b";
11
+ import { Sandbox, SandboxNotFoundError, RateLimitError } from "e2b";
12
12
  import { Sandbox as Desktop } from "@e2b/desktop";
13
13
  import pRetry from "p-retry";
14
+ import type { FailedAttemptError } from "p-retry";
15
+ import { SandboxUnavailableError } from "./sandbox-errors.js";
14
16
  import type { SandboxProvider, DesktopSandboxProvider, SandboxCommandResult } from "./types/sandbox.js";
17
+ import type { ConnectorRequestRules } from "./types/workflow-metadata.js";
18
+ import type { NetworkPolicy as VercelNetworkPolicy, NetworkPolicyRule as VercelNetworkPolicyRule } from "@vercel/sandbox";
15
19
 
16
20
  export type { SandboxProvider, DesktopSandboxProvider, SandboxCommandRunOptions, SandboxCommandResult } from "./types/sandbox.js";
17
21
 
@@ -34,10 +38,13 @@ export const AGENT_COMPOSE_TAG = process.env.AGENT_COMPOSE_TAG ??
34
38
  const VERCEL_VM_LIFETIME_WINDOW_MS = 6 * 60 * 60 * 1000;
35
39
 
36
40
  /**
37
- * Vercel-compatible network policy for outbound HTTPS requests.
38
- * When a sandbox makes a request matching a domain in `allow`, the Vercel
39
- * firewall injects the specified headers before forwarding — credentials never
40
- * exist inside the VM. E2B ignores this field (future self-hosted mapping TBD).
41
+ * Network policy for outbound HTTPS requests — ONE shape for every provider;
42
+ * only the enforcement point differs. When a sandbox makes a request matching
43
+ * a domain in `allow`, the egress layer injects the specified headers before
44
+ * forwarding — credentials never exist inside the VM. Vercel's firewall
45
+ * enforces this natively (`requestRules` translate to native `match` rules);
46
+ * E2B enforces it via the embedded iron-proxy, which pulls the identical
47
+ * resolved policy from the server.
41
48
  */
42
49
  export interface SandboxNetworkHeaderTransform {
43
50
  headers?: Record<string, string>;
@@ -45,6 +52,10 @@ export interface SandboxNetworkHeaderTransform {
45
52
 
46
53
  export interface SandboxNetworkAllowRule {
47
54
  transform?: SandboxNetworkHeaderTransform[];
55
+ /** Tier-2 request gate (method/path) — the transform (and, on iron-proxy,
56
+ * the request itself) only applies when the request matches. Present on
57
+ * connector-auth rules; see `ConnectorRequestRules`. */
58
+ requestRules?: ConnectorRequestRules;
48
59
  }
49
60
 
50
61
  export interface SandboxNetworkSubnetPolicy {
@@ -64,16 +75,23 @@ export interface SandboxCreateOpts {
64
75
  envs: Record<string, string>;
65
76
  metadata: Record<string, string>;
66
77
  timeoutMs: number;
67
- /** Provider-specific template/snapshot identifier. E2B: template ID; Vercel: snapshot ID. */
78
+ /** Provider-specific template/snapshot identifier. E2B: template id or
79
+ * snapshot id (omit → E2B's default base). Vercel: snapshot id (omit → node24). */
68
80
  template?: string;
69
- /** Outbound request policy. Vercel only — E2B silently ignores. */
81
+ /** Outbound request policy with header transforms — ONE shape for every
82
+ * provider; only the enforcement point differs. Vercel's firewall
83
+ * enforces + injects natively from this value. E2B enforces it via an
84
+ * EMBEDDED iron-proxy inside the VM (loopback DNS + iptables, started
85
+ * at boot), which fetches the identical resolved policy from the
86
+ * server's per-run egress-policy endpoint — so this field is not passed
87
+ * to E2B's API. See server/src/sandbox/iron-proxy.ts. */
70
88
  networkPolicy?: SandboxNetworkPolicy;
71
89
  }
72
90
 
73
91
  /**
74
92
  * What the provider reports as "currently alive" — the single input to orphan
75
93
  * reconciliation. `metadata` is best-effort: E2B populates it from sandbox
76
- * labels, Vercel leaves it empty (the API has no metadata field). Callers
94
+ * labels, Vercel from sandbox tags (max 5, stamped at create). Callers
77
95
  * that need runId correlation cross-reference `sandboxId` against their own
78
96
  * state (server: `workflow_runs.sandbox_id` + `run_agent_sandboxes.provider_sandbox_id`).
79
97
  */
@@ -92,18 +110,17 @@ interface SandboxProviderDef {
92
110
  * Returns the number of sandboxes currently running against this provider.
93
111
  * Used for quota observability — account-wide, not per-instance.
94
112
  *
95
- * E2B scopes by our AGENT_COMPOSE_TAG metadata so each machine only reports
96
- * its own sandboxes; DD aggregates across instances with `sum by {provider}`.
97
- * Vercel has no user metadata support, so it counts every running sandbox in
98
- * the configured project (assumed dedicated to agent-compose).
113
+ * Both providers scope by our AGENT_COMPOSE_TAG (E2B metadata labels,
114
+ * Vercel tags) so each fleet only reports its own sandboxes; DD aggregates
115
+ * across instances with `sum by {provider}`.
99
116
  */
100
117
  getActiveCount?: (env: Record<string, string>) => Promise<number>;
101
118
  /**
102
119
  * Provider's view of what's currently alive for our account/fleet.
103
120
  * The reconciliation SoT — a sandbox missing from this list IS dead,
104
121
  * regardless of what our DB says. Implemented by listing the provider's
105
- * running sandboxes; E2B filters by metadata tag, Vercel lists the whole
106
- * project (it has no metadata search). 200-row cap applies to Vercel.
122
+ * running sandboxes; E2B filters by metadata tag, Vercel by the `executor`
123
+ * sandbox tag.
107
124
  */
108
125
  listOwned?: (env: Record<string, string>) => Promise<OwnedSandbox[]>;
109
126
  /**
@@ -147,6 +164,80 @@ export function makeSandboxProvider(sb: Sandbox | Desktop): SandboxProvider {
147
164
  };
148
165
  }
149
166
 
167
+ /** E2B caps a sandbox's lifetime per plan (Hobby 1h, Pro 24h) and 400s when the
168
+ * create timeout exceeds it — and our `AC_SANDBOX_DEADLINE` (4.5h) tops Hobby.
169
+ * Clamp the E2B create timeout to this max (raise on Pro via `E2B_MAX_SANDBOX_MS`).
170
+ * Read at call time so the workflow bundle stays env-free. */
171
+ function e2bMaxSandboxMs(): number {
172
+ return Number(process.env.E2B_MAX_SANDBOX_MS) || 60 * 60 * 1000;
173
+ }
174
+
175
+ /** Is this a FLEETING sandbox-provider race worth retrying — the sandbox being
176
+ * paused / reclaimed / stopped under load, an upstream 5xx/429, or a dropped
177
+ * connection — vs a PERMANENT failure that must fail fast?
178
+ *
179
+ * E2B raises TYPED errors, so prefer `instanceof` (upgrade-stable — survives any
180
+ * message rewording): a reconnect/snapshot that hits a sandbox mid-pause throws
181
+ * `SandboxNotFoundError`, rate limits throw `RateLimitError`. (Permanent
182
+ * template/snapshot-not-found only occurs on CREATE, which is no longer retried.)
183
+ * The message regex is the FALLBACK for the untyped Vercel transport + raw socket
184
+ * errors — and deliberately omits bare "timeout"/"network": E2B's permanent
185
+ * plan-cap 400 ("timeout … exceeds the maximum sandbox lifetime") contains
186
+ * "timeout" and must fail fast, not retry 4×. */
187
+ function isTransientSandboxError(error: unknown): boolean {
188
+ if (error instanceof SandboxNotFoundError || error instanceof RateLimitError) return true;
189
+ const message = error instanceof Error ? error.message : String(error ?? "");
190
+ // The 5xx match is anchored to a status-code context ("status 503",
191
+ // "502 Bad Gateway", …) — a bare \b5\d\d\b would classify ANY standalone
192
+ // 500-599 number in a message (durations, row counts) as retryable.
193
+ return /paus|stopp|resum|transition|status(?:\s+code)?[:\s]+5\d\d\b|\b5\d\d\s+(?:internal|bad gateway|service|gateway)|\b429\b|ECONNRESET|ECONNREFUSED|ETIMEDOUT|socket hang up|fetch failed/i
194
+ .test(message);
195
+ }
196
+
197
+ /** The ONE place the sandbox retry policy lives. Provisioning, reconnecting, and
198
+ * snapshotting all race with the sandbox being paused/reclaimed; this runs the call
199
+ * through p-retry's generic backoff, retrying ONLY the transient race (and failing
200
+ * fast otherwise). reconnectSandbox and provider.snapshot() all go through it, so
201
+ * no provider re-implements the pattern or can forget it. `shouldRetry` is
202
+ * overridable for call sites where the default classification is wrong (see
203
+ * reconnectSandbox's SandboxNotFoundError handling). */
204
+ function withSandboxRetry<T>(
205
+ fn: () => Promise<T>,
206
+ shouldRetry: (error: FailedAttemptError) => boolean = isTransientSandboxError,
207
+ ): Promise<T> {
208
+ return pRetry(fn, { retries: 4, minTimeout: 400, factor: 2, shouldRetry });
209
+ }
210
+
211
+ /** Add snapshot-retry to a freshly-obtained provider. create/reconnect are retried at
212
+ * the registry call (`createSandbox`/`reconnectSandbox`); snapshot is a method, so it
213
+ * gets the same `withSandboxRetry` here. `commands.run` is deliberately NOT wrapped —
214
+ * retrying user-code execution could double-apply side effects, so step execution
215
+ * relies on the engine's pre-step sandbox recovery instead. No-op without a snapshot. */
216
+ function withSnapshotRetry(p: SandboxProvider): SandboxProvider {
217
+ if (!p.snapshot) return p;
218
+ const snapshot = p.snapshot.bind(p);
219
+ return { ...p, snapshot: () => withSandboxRetry(snapshot) };
220
+ }
221
+
222
+ /** E2B base provider + live-filesystem snapshot. E2B's `createSnapshot()` captures
223
+ * the running sandbox as a persistent snapshot whose id is usable as a
224
+ * `Sandbox.create()` source and outlives the origin sandbox — so E2B reaches
225
+ * snapshot / `bootFrom` parity with Vercel, and snapshot-backed `ctx.pause` works
226
+ * on E2B. (Desktop intentionally omits this — it is registry-only, not selectable.) */
227
+ function makeE2bSandboxProvider(sb: Sandbox): SandboxProvider {
228
+ return {
229
+ ...makeSandboxProvider(sb),
230
+ async snapshot() {
231
+ // Raw capture — the transient pause/reclaim race is retried centrally:
232
+ // createSandbox/reconnectSandbox wrap every provider's snapshot() in
233
+ // withSandboxRetry (via withSnapshotRetry). create itself is deliberately
234
+ // NOT retried — see createSandbox.
235
+ const { snapshotId } = await sb.createSnapshot();
236
+ return { snapshotId };
237
+ },
238
+ };
239
+ }
240
+
150
241
  export function makeDesktopSandboxProvider(sb: Desktop): DesktopSandboxProvider {
151
242
  return {
152
243
  ...makeSandboxProvider(sb),
@@ -209,6 +300,88 @@ export async function parseSseExecStream(
209
300
 
210
301
  // ── Vercel helpers ────────────────────────────────────────────────────────────
211
302
 
303
+ /** Matches a path containing a dot-dot segment — literal (`/../`) or
304
+ * percent-encoded (`%2e%2e`, `%2f` boundaries). `DOT_SEGMENT_PATH_RE2` is
305
+ * the RE2 form handed to Vercel's matcher: wrapped in `.*` so it behaves
306
+ * identically under search and full-match semantics (RE2 has no lookaround,
307
+ * so the guard is a positive match on the traversal pattern, used as a
308
+ * credential-stripping first rule). `DOT_SEGMENT_RE` is the local JS twin
309
+ * used to validate DECLARED prefixes. */
310
+ export const DOT_SEGMENT_PATH_RE2 = ".*(?:^|/|%2[fF])(?:\\.|%2[eE]){2}(?:/|%2[fF]|$).*";
311
+ const DOT_SEGMENT_RE = /(?:^|\/|%2f)(?:\.|%2e){2}(?:\/|%2f|$)/i;
312
+
313
+ /** A Tier-2 path gate decides where a brokered credential may ride, so a
314
+ * malformed declared prefix is a config bug that must fail closed, not a
315
+ * string we forward verbatim to a third-party matcher. */
316
+ function assertSafePathPrefix(domain: string, prefix: string): void {
317
+ if (!prefix.startsWith("/") || DOT_SEGMENT_RE.test(prefix)) {
318
+ throw new Error(
319
+ `Invalid Tier-2 pathPrefix ${JSON.stringify(prefix)} for "${domain}": prefixes must start with "/" and contain no dot-segments`,
320
+ );
321
+ }
322
+ }
323
+
324
+ /**
325
+ * Translate our policy shape into Vercel's native `NetworkPolicy`.
326
+ *
327
+ * The shapes are identical except for Tier-2: our `requestRules`
328
+ * (`{ methods, pathPrefixes }`) become Vercel `match` rules
329
+ * (`{ method, path: { startsWith } }`) — one rule per path prefix, since a
330
+ * Vercel matcher carries a single path.
331
+ *
332
+ * Dot-segment defense: prefix matching is raw — `/repos/acme/../../user`
333
+ * starts with `/repos/acme/` but the origin normalizes it to `/user`, riding
334
+ * the credential outside its declared scope. We don't get to assume the
335
+ * enforcer RFC-3986-normalizes before matching, so every domain with a path
336
+ * gate gets a transform-free FIRST rule matching dot-segment paths
337
+ * (first-match-wins ⇒ such requests go out credential-free), and declared
338
+ * prefixes themselves are validated (must start with "/", no dot-segments).
339
+ *
340
+ * Accepted divergence from iron-proxy (E2B): off-gate requests to an allowed
341
+ * domain pass WITHOUT the transform (Vercel terminates TLS only for
342
+ * transform-bearing domains and applies first-match-wins) — iron-proxy
343
+ * REFUSES them instead. Reachability differs; the property that matters is
344
+ * preserved on both substrates: the brokered credential rides only declared
345
+ * method/path combinations. Pinned by vercel-network-policy.test.ts.
346
+ */
347
+ export function toVercelNetworkPolicy(policy: SandboxNetworkPolicy): VercelNetworkPolicy {
348
+ if (typeof policy === "string" || !policy.allow || Array.isArray(policy.allow)) {
349
+ return policy as VercelNetworkPolicy;
350
+ }
351
+ const allow: Record<string, VercelNetworkPolicyRule[]> = {};
352
+ for (const [domain, rules] of Object.entries(policy.allow)) {
353
+ const translated = rules.flatMap((rule): VercelNetworkPolicyRule[] => {
354
+ const { requestRules, ...rest } = rule;
355
+ const methods = requestRules?.methods ?? [];
356
+ const prefixes = requestRules?.pathPrefixes ?? [];
357
+ if (methods.length === 0 && prefixes.length === 0) return [rest];
358
+ const methodMatch = methods.length > 0 ? { method: methods } : {};
359
+ if (prefixes.length === 0) return [{ ...rest, match: methodMatch }];
360
+ for (const prefix of prefixes) assertSafePathPrefix(domain, prefix);
361
+ return prefixes.map((prefix) => ({ ...rest, match: { ...methodMatch, path: { startsWith: prefix } } }));
362
+ });
363
+ const hasPathGate = rules.some((rule) => (rule.requestRules?.pathPrefixes?.length ?? 0) > 0);
364
+ allow[domain] = hasPathGate
365
+ ? [{ match: { path: { regex: DOT_SEGMENT_PATH_RE2 } } }, ...translated]
366
+ : translated;
367
+ }
368
+ return { ...policy, allow };
369
+ }
370
+
371
+ /** Vercel caps sandbox tags at 5. Build the tag set with the fleet `executor`
372
+ * tag always present and always winning over caller metadata — `listOwned` /
373
+ * orphan reconciliation scope on it, so a clobbered or dropped executor tag
374
+ * makes the sandbox invisible to cleanup. When metadata overflows the cap,
375
+ * keep executor + the lexicographically-first 4 metadata keys, so what gets
376
+ * dropped is deterministic rather than dependent on object insertion order. */
377
+ export const VERCEL_MAX_TAGS = 5;
378
+ export function buildVercelTags(metadata: Record<string, string>): Record<string, string> {
379
+ const tags: Record<string, string> = { ...metadata, executor: AGENT_COMPOSE_TAG };
380
+ const extras = Object.keys(tags).filter((k) => k !== "executor").sort();
381
+ for (const key of extras.slice(VERCEL_MAX_TAGS - 1)) delete tags[key];
382
+ return tags;
383
+ }
384
+
212
385
  // eslint-disable-next-line @typescript-eslint/no-explicit-any
213
386
  function makeVercelSandboxProvider(sb: any, globalEnvs?: Record<string, string>): SandboxProvider {
214
387
  // Vercel's Sandbox.create({ env }) does NOT flow to runCommand subprocesses — they start fresh shells.
@@ -217,8 +390,21 @@ function makeVercelSandboxProvider(sb: any, globalEnvs?: Record<string, string>)
217
390
  const mergeEnvs = (cmdEnvs?: Record<string, string>) =>
218
391
  globalEnvs ? { ...globalEnvs, ...cmdEnvs } : cmdEnvs;
219
392
 
393
+ // Wrap a sandbox-side failure as `SandboxUnavailableError`. `retryable` is
394
+ // decided purely by *where* the error was caught, not by inspecting its
395
+ // shape: true before the runner launched any user code (file write / command
396
+ // launch / reconnect — safe to re-provision and retry), false once the
397
+ // runner streamed (user code may have run). The original message is
398
+ // preserved for server-side diagnostics.
399
+ const asUnavailable = (err: unknown, retryable: boolean): never => {
400
+ throw new SandboxUnavailableError(err instanceof Error ? err.message : String(err), { retryable, sandboxId: sb.name });
401
+ };
402
+
220
403
  return {
221
- sandboxId: sb.sandboxId,
404
+ // @vercel/sandbox 2.x is name-keyed: `name` is the stable identifier
405
+ // (`Sandbox.get({ name })`); the v1 `sandboxId` getter is gone. Our
406
+ // provider-facing field keeps its name — it's "the provider-native id".
407
+ sandboxId: sb.name,
222
408
  commands: {
223
409
  async run(cmd, opts) {
224
410
  const signal = opts?.timeoutMs ? AbortSignal.timeout(opts.timeoutMs) : undefined;
@@ -226,34 +412,63 @@ function makeVercelSandboxProvider(sb: any, globalEnvs?: Record<string, string>)
226
412
  let stderr = "";
227
413
  // `sudo: true` is Vercel's native root flag — applied to the `sh`
228
414
  // invocation so the whole shell (and its children) runs as root.
415
+ // A launch failure means the command never started — nothing
416
+ // user-side ran, so any error here is safe to re-provision + retry.
229
417
  // eslint-disable-next-line @typescript-eslint/no-explicit-any
230
- const handle: any = await sb.runCommand({ cmd: "sh", args: ["-c", cmd], cwd: opts?.cwd, env: mergeEnvs(opts?.envs), detached: true, signal, ...(opts?.sudo ? { sudo: true } : {}) });
418
+ const handle: any = await sb.runCommand({ cmd: "sh", args: ["-c", cmd], cwd: opts?.cwd, env: mergeEnvs(opts?.envs), detached: true, signal, ...(opts?.sudo ? { sudo: true } : {}) })
419
+ .catch((err: unknown) => asUnavailable(err, true));
231
420
  // Reconnect to the already-running command on transient stream failures (e.g. BrotliDecompressionError).
232
421
  // `h.logs()` replays from the start on reconnect, so reset accumulators per
233
422
  // attempt to avoid double-counting. The streaming callbacks may still fire
234
423
  // for duplicate chunks during retries — an acceptable tradeoff for resilience.
235
- await pRetry(async (attempt) => {
236
- // eslint-disable-next-line @typescript-eslint/no-explicit-any
237
- const h: any = attempt === 1 ? handle : await sb.getCommand(handle.cmdId);
238
- if (attempt > 1) { stdout = ""; stderr = ""; }
239
- for await (const log of h.logs()) {
240
- if (log.stream === "stdout") { stdout += log.data; opts?.onStdout?.(log.data); }
241
- else { stderr += log.data; opts?.onStderr?.(log.data); }
424
+ //
425
+ // A failure here (mid-stream / on `wait`) is NOT safe to retry: the
426
+ // runner already launched, so user code may have produced side
427
+ // effects. Surface it as a terminal `SandboxUnavailableError` so it's
428
+ // classified honestly but not auto-replayed.
429
+ try {
430
+ await pRetry(async (attempt) => {
431
+ // eslint-disable-next-line @typescript-eslint/no-explicit-any
432
+ const h: any = attempt === 1 ? handle : await sb.getCommand(handle.cmdId);
433
+ if (attempt > 1) { stdout = ""; stderr = ""; }
434
+ for await (const log of h.logs()) {
435
+ if (log.stream === "stdout") { stdout += log.data; opts?.onStdout?.(log.data); }
436
+ else { stderr += log.data; opts?.onStderr?.(log.data); }
437
+ }
438
+ }, { retries: 3, minTimeout: 1_000, factor: 2 });
439
+ const finished = await handle.wait();
440
+ return { exitCode: finished.exitCode, stdout, stderr };
441
+ } catch (err) {
442
+ // A caller-requested timeout (opts.timeoutMs → AbortSignal) is a
443
+ // COMMAND timing out, not the sandbox dying — the VM is alive and
444
+ // answering; one command overran its budget. Classifying it as
445
+ // terminal sandbox-unavailable failed whole runs over a single
446
+ // slow mount probe (observed live). Surface it as a plain error
447
+ // so the caller's own retry/deadline policy decides.
448
+ if (signal?.aborted) {
449
+ throw new Error(`command timed out after ${opts?.timeoutMs}ms: ${cmd.slice(0, 200)}`);
242
450
  }
243
- }, { retries: 3, minTimeout: 1_000, factor: 2 });
244
- const finished = await handle.wait();
245
- return { exitCode: finished.exitCode, stdout, stderr };
451
+ return asUnavailable(err, false);
452
+ }
246
453
  },
247
454
  },
248
455
  files: {
249
456
  async write(path: string, content: string) {
250
- await sb.writeFiles([{ path, content }]);
457
+ // File writes happen before the runner launches (step input / pause
458
+ // resolution files), so a failure here means nothing ran — retryable.
459
+ await sb.writeFiles([{ path, content }]).catch((err: unknown) => asUnavailable(err, true));
251
460
  },
252
461
  },
253
462
  // Propagate errors — `killAllRunSandboxes` relies on kill failures being
254
463
  // observable so it can leave `sandbox_id` set for `findOrphanedSandboxes`
255
464
  // to retry on next boot. Swallowing here makes the orphan retry loop blind.
256
- async kill() { await sb.stop(); },
465
+ //
466
+ // kill must DESTROY: in @vercel/sandbox 2.x stop() only halts the VM — the
467
+ // name-keyed sandbox record persists (reserving the name; commands against
468
+ // a stale handle can transparently resume a stopped VM). delete() is the
469
+ // destroying call ("after deletion the instance becomes inert"), so stop
470
+ // then delete.
471
+ async kill() { await sb.stop(); await sb.delete(); },
257
472
  // Vercel native snapshot — used by sandbox-environments to capture the
258
473
  // configured VM after the customer's `setup()` completes. `expiration: 0`
259
474
  // is Vercel's "never expires" value; env snapshots are long-lived by
@@ -267,6 +482,15 @@ function makeVercelSandboxProvider(sb: any, globalEnvs?: Record<string, string>)
267
482
  ...(sizeBytes !== undefined ? { sizeBytes } : {}),
268
483
  };
269
484
  },
485
+ // Push a freshly-resolved egress policy onto the live sandbox —
486
+ // @vercel/sandbox 2.x `update({ networkPolicy })`. Lets the server
487
+ // re-resolve the run policy (re-minting connector access tokens)
488
+ // before each step instead of living with the policy baked at create.
489
+ // E2B leaves this undefined: its enforcement (iron-proxy) lives inside
490
+ // the VM and is configured at boot.
491
+ async updateNetworkPolicy(policy) {
492
+ await sb.update({ networkPolicy: toVercelNetworkPolicy(policy) });
493
+ },
270
494
  };
271
495
  }
272
496
 
@@ -329,10 +553,28 @@ const SANDBOX_PROVIDERS: Record<string, SandboxProviderDef> = {
329
553
  const { Sandbox: VercelSandbox } = await import("@vercel/sandbox");
330
554
  const creds = { token: env.VERCEL_ACCESS_TOKEN, teamId: env.VERCEL_TEAM_ID, projectId: env.VERCEL_PROJECT_ID };
331
555
  // `template` is a snapshot id (runtime is baked in); absent → fresh node24.
332
- const np = opts.networkPolicy;
333
- const sb = await VercelSandbox.create(opts.template
334
- ? { source: { type: "snapshot" as const, snapshotId: opts.template }, timeout: opts.timeoutMs, env: opts.envs, ...(np ? { networkPolicy: np } : {}), ...creds }
335
- : { runtime: "node24" as const, timeout: opts.timeoutMs, env: opts.envs, ...(np ? { networkPolicy: np } : {}), ...creds },
556
+ const np = opts.networkPolicy ? { networkPolicy: toVercelNetworkPolicy(opts.networkPolicy) } : {};
557
+ // Caller metadata rides as Vercel tags — `listOwned` filters on the
558
+ // `executor` key, mirroring E2B's metadata-tag scoping. buildVercelTags
559
+ // guarantees executor wins over metadata and survives the 5-tag cap.
560
+ const tags = buildVercelTags(opts.metadata);
561
+ // No explicit template/bootFrom → VERCEL_DEFAULT_SNAPSHOT if set (the
562
+ // platform agent-env base: claude, archil, rtk, bun, agentc CLI, the
563
+ // SDK in /workspace/node_modules, /ac:* skills, AGENTS.md — built by
564
+ // .agentc/environments/agent-env.ts), else raw node24. The E2B
565
+ // analogue is E2B_DEFAULT_TEMPLATE below. Because `saveLatest`/
566
+ // `reuse` captures snapshot the whole filesystem, everything a run
567
+ // bakes on top of this base chains from it — the base tools are in
568
+ // every derived snapshot for free.
569
+ const tmpl = opts.template ?? process.env.VERCEL_DEFAULT_SNAPSHOT;
570
+ // `persistent: false` — @vercel/sandbox 2.x creates persistent-by-default
571
+ // sandboxes: stop() auto-snapshots (and keeps billing storage), commands
572
+ // transparently resume a stopped VM, and kill no longer destroys. Our
573
+ // sandboxes are single-run and lifecycle-managed by the engine (explicit
574
+ // snapshot() / kill()), so opt out in BOTH branches.
575
+ const sb = await VercelSandbox.create(tmpl
576
+ ? { source: { type: "snapshot" as const, snapshotId: tmpl }, timeout: opts.timeoutMs, env: opts.envs, tags, persistent: false, ...np, ...creds }
577
+ : { runtime: "node24" as const, timeout: opts.timeoutMs, env: opts.envs, tags, persistent: false, ...np, ...creds },
336
578
  );
337
579
  // Pass envs as globalEnvs so they're injected into every runCommand subprocess.
338
580
  // (Vercel's Sandbox.create env parameter does not flow to runCommand subprocesses.)
@@ -345,33 +587,58 @@ const SANDBOX_PROVIDERS: Record<string, SandboxProviderDef> = {
345
587
  teamId: process.env.VERCEL_TEAM_ID ?? "",
346
588
  projectId: process.env.VERCEL_PROJECT_ID ?? "",
347
589
  };
348
- return makeVercelSandboxProvider(await VercelSandbox.get({ sandboxId, ...creds }));
590
+ // Reconnecting happens before any user code runs, so any failure here
591
+ // (sandbox already stopping, API blip) is a retryable sandbox-unavailable
592
+ // — the workflow re-provisions a fresh one rather than failing the run.
593
+ // 2.x is name-keyed: the id we persisted IS the sandbox name.
594
+ const sb = await VercelSandbox.get({ name: sandboxId, ...creds }).catch((err: unknown) => {
595
+ throw new SandboxUnavailableError(err instanceof Error ? err.message : String(err), { retryable: true, sandboxId });
596
+ });
597
+ return makeVercelSandboxProvider(sb);
349
598
  },
350
599
  getActiveCount: async (env) => (await SANDBOX_PROVIDERS.vercel.listOwned!(env)).length,
351
600
  listOwned: async (env) => {
352
601
  const { Sandbox: VercelSandbox } = await import("@vercel/sandbox");
353
602
  const creds = { token: env.VERCEL_ACCESS_TOKEN, teamId: env.VERCEL_TEAM_ID, projectId: env.VERCEL_PROJECT_ID };
354
- // Bounds the scan to `VERCEL_VM_LIFETIME_WINDOW_MS` — Pro/Enterprise caps
355
- // VM lifetime at 5h, +1h slack for clock skew and `extendTimeout()`.
356
- // Without this, projects with thousands of historical terminal sandboxes
357
- // page forever (10k+ in v0.6.28). Truncating silently would hide orphans,
358
- // so we throw at the 50-page cap instead. Each page is wrapped in pRetry
359
- // so a transient 5xx/429/network blip doesn't cost a reconcile tick.
603
+ // Scoped to our fleet by the `executor` tag (stamped at create) and
604
+ // bounded to `VERCEL_VM_LIFETIME_WINDOW_MS` — Pro/Enterprise caps VM
605
+ // lifetime at 5h, +1h slack for clock skew and `extendTimeout()`. The
606
+ // sort is pinned to createdAt-desc explicitly — the early-stop below is
607
+ // only correct under that order, and the server's default sort is not a
608
+ // contract we can rely on — so we stop at the first item older than the
609
+ // window instead of paging through thousands of historical terminal
610
+ // sandboxes (10k+ in v0.6.28). Truncating silently would hide orphans,
611
+ // so we throw at the page cap instead — 200 pages × the API's 50-item
612
+ // limit preserves the previous 10k scan bound (v1 paged 200 × 50). Each
613
+ // page is wrapped in pRetry so a transient 5xx/429/network blip doesn't
614
+ // cost a reconcile tick.
615
+ //
616
+ // 2.x cutover caveat: sandboxes created by the pre-2.x code carry no
617
+ // tags, so for up to VERCEL_VM_LIFETIME_WINDOW_MS after a deploy they
618
+ // are invisible here (and their persisted v1 sandbox ids don't resolve
619
+ // via `Sandbox.get({ name })`). Self-limiting — the provider's 5h
620
+ // lifetime cap expires them — but drain in-flight Vercel runs before
621
+ // deploying if losing their sandbox state matters.
360
622
  const since = Date.now() - VERCEL_VM_LIFETIME_WINDOW_MS;
361
623
  const out: OwnedSandbox[] = [];
362
- let until: number | undefined;
363
- for (let page = 0; page < 50; page++) {
364
- const { json } = await pRetry(
365
- () => VercelSandbox.list({ ...creds, limit: 200, since, ...(until !== undefined ? { until } : {}) }),
624
+ let cursor: string | undefined;
625
+ for (let page = 0; page < 200; page++) {
626
+ const result = await pRetry(
627
+ // limit is capped at 50 by the v2 list API — 51+ answers 400
628
+ // (probed live 2026-06-10; not documented in the SDK types).
629
+ () => VercelSandbox.list({ ...creds, limit: 50, sortBy: "createdAt", sortOrder: "desc", tags: { executor: AGENT_COMPOSE_TAG }, ...(cursor !== undefined ? { cursor } : {}) }),
366
630
  { retries: 3, minTimeout: 500, factor: 2 },
367
631
  );
368
- for (const sb of json.sandboxes) if (sb.status === "running") {
369
- out.push({ sandboxId: sb.id, createdAt: new Date(sb.createdAt), metadata: {} });
632
+ for (const sb of result.sandboxes) {
633
+ if (sb.createdAt < since) return out;
634
+ if (sb.status === "running") {
635
+ out.push({ sandboxId: sb.name, createdAt: new Date(sb.createdAt), metadata: sb.tags ?? {} });
636
+ }
370
637
  }
371
- if (json.pagination.next === null) return out;
372
- until = json.pagination.next;
638
+ if (result.pagination.next === null) return out;
639
+ cursor = result.pagination.next;
373
640
  }
374
- throw new Error("Vercel listSandboxes exceeded 50 pages within the lifetime window — refuse to silently truncate");
641
+ throw new Error("Vercel listSandboxes exceeded 200 pages (10k sandboxes) within the lifetime window — refuse to silently truncate");
375
642
  },
376
643
  deleteSnapshot: async (snapshotId, env) => {
377
644
  const { Snapshot } = await import("@vercel/sandbox");
@@ -382,11 +649,42 @@ const SANDBOX_PROVIDERS: Record<string, SandboxProviderDef> = {
382
649
  },
383
650
  "e2b": {
384
651
  requiredEnv: { E2B_API_KEY: "E2B API key — e2b.dev/dashboard" },
385
- create: async ({ template, ...sandboxOpts }) => {
386
- if (!template) throw new Error("E2B provider requires an explicit `template` (Dockerfile-based — no default base image)");
387
- return makeSandboxProvider(await Sandbox.create(template, sandboxOpts));
652
+ create: async ({ template, timeoutMs, networkPolicy: _np, ...rest }) => {
653
+ // `template` is an E2B template id, a snapshot id (a valid create source
654
+ // that persists beyond its origin sandbox — bootFrom parity), or absent →
655
+ // E2B's default base (Debian + node + npm/python/git), mirroring Vercel's
656
+ // node24 default. With no explicit template/bootFrom, fall back to
657
+ // E2B_DEFAULT_TEMPLATE if set (a prebuilt base — e.g. one with the claude CLI
658
+ // + chromium baked in and more RAM than the stock 482MB base), the
659
+ // per-deployment analogue of Vercel's node24; else E2B's stock base.
660
+ // Self-provisioning runtimes (claude/codex/amp via bootFrom:"reuse") install
661
+ // their CLI on the base and cache it in the captured snapshot, so a
662
+ // template-less first run boots, installs, snapshots. Clamp the lifetime to
663
+ // E2B's per-plan cap (see e2bMaxSandboxMs) so a 4.5h AC_SANDBOX_DEADLINE
664
+ // doesn't 400 on smaller plans.
665
+ //
666
+ // `networkPolicy` is deliberately NOT passed to E2B's API. E2B has no
667
+ // header-injecting firewall; enforcement happens INSIDE the VM via
668
+ // the embedded iron-proxy started at sandbox boot, which pulls the
669
+ // identical resolved policy from the server. One policy shape, two
670
+ // enforcement points. See server/src/sandbox/iron-proxy.ts.
671
+ const maxMs = e2bMaxSandboxMs();
672
+ if (timeoutMs > maxMs) {
673
+ // The clamp turns E2B's explicit create-time 400 into a silent mid-run
674
+ // sandbox death at the cap — say so up front instead of letting a run
675
+ // budgeted past the cap discover it as a terminal infra failure.
676
+ console.warn(
677
+ `[sandbox] e2b create timeout clamped: requested ${timeoutMs}ms exceeds the plan cap ${maxMs}ms — ` +
678
+ `the sandbox dies at the cap, not the requested deadline. Raise E2B_MAX_SANDBOX_MS on plans that allow more.`,
679
+ );
680
+ }
681
+ const sandboxOpts = { ...rest, timeoutMs: Math.min(timeoutMs, maxMs) };
682
+ const tmpl = template ?? process.env.E2B_DEFAULT_TEMPLATE;
683
+ return makeE2bSandboxProvider(
684
+ await (tmpl ? Sandbox.create(tmpl, sandboxOpts) : Sandbox.create(sandboxOpts)),
685
+ );
388
686
  },
389
- reconnect: async (sandboxId) => makeSandboxProvider(
687
+ reconnect: async (sandboxId) => makeE2bSandboxProvider(
390
688
  await Sandbox.connect(sandboxId, { apiKey: process.env.E2B_API_KEY ?? "", timeoutMs: 60 * 60 * 1000 }),
391
689
  ),
392
690
  killAll: async () => {
@@ -413,12 +711,22 @@ const SANDBOX_PROVIDERS: Record<string, SandboxProviderDef> = {
413
711
  }
414
712
  return out;
415
713
  },
714
+ deleteSnapshot: async (snapshotId, env) => {
715
+ // E2B snapshots are team-scoped; delete by id. Best-effort like Vercel's.
716
+ await Sandbox.deleteSnapshot(snapshotId, { apiKey: env.E2B_API_KEY });
717
+ },
416
718
  },
417
719
  "e2b-desktop": {
418
720
  requiredEnv: { E2B_API_KEY: "E2B API key — e2b.dev/dashboard" },
419
- create: async ({ template, ...sandboxOpts }) => {
721
+ // Mirror "e2b": `networkPolicy` is deliberately NOT passed to the SDK —
722
+ // its transforms carry live connector tokens, and a credential-bearing
723
+ // object must never cross into a third-party library's option bag.
724
+ // Lifetime is clamped to E2B's per-plan cap the same way.
725
+ create: async ({ template, timeoutMs, networkPolicy: _np, ...rest }) => {
420
726
  if (!template) throw new Error("E2B Desktop provider requires an explicit `template` (Dockerfile-based — no default base image)");
421
- return makeDesktopSandboxProvider(await Desktop.create(template, sandboxOpts));
727
+ return makeDesktopSandboxProvider(
728
+ await Desktop.create(template, { ...rest, timeoutMs: Math.min(timeoutMs, e2bMaxSandboxMs()) }),
729
+ );
422
730
  },
423
731
  },
424
732
  };
@@ -431,14 +739,28 @@ export async function createSandbox(provider: SandboxProviderName, opts: Sandbox
431
739
  .filter(([key]) => !process.env[key])
432
740
  .map(([key, desc]) => ` ${key} — ${desc}`);
433
741
  if (missing.length > 0) throw new Error(`Sandbox provider "${provider}" requires env vars:\n${missing.join("\n")}`);
434
- return def.create(opts, Object.fromEntries(Object.keys(def.requiredEnv).map(k => [k, process.env[k]!])));
742
+ const env = Object.fromEntries(Object.keys(def.requiredEnv).map((k) => [k, process.env[k]!]));
743
+ // Provision once — do NOT wrap create in withSandboxRetry: a create whose RESPONSE
744
+ // is lost (client-side timeout after the provider already provisioned) would, on
745
+ // retry, mint a SECOND billed sandbox and orphan the first. Provision failures are
746
+ // re-attempted at the workflow layer. The result is still wrapped so snapshot()
747
+ // retries the transient pause/reclaim race (which is where the race actually lives).
748
+ return withSnapshotRetry(await def.create(opts, env));
435
749
  }
436
750
 
437
751
  /** Reconnect to an existing sandbox by provider + provider-native sandbox ID. */
438
752
  export async function reconnectSandbox(provider: SandboxProviderName, sandboxId: string): Promise<SandboxProvider> {
439
753
  const def = SANDBOX_PROVIDERS[provider];
440
754
  if (!def?.reconnect) throw new Error(`Provider "${provider}" does not support reconnect`);
441
- return def.reconnect(sandboxId);
755
+ // Retry the transient reconnect/resume race, then add snapshot-retry to the
756
+ // result. On RECONNECT, SandboxNotFoundError usually means the sandbox is
757
+ // genuinely gone (killed / lifetime-expired) — permanent — but it can also
758
+ // be the transient mid-pause race, so it gets exactly ONE retry instead of
759
+ // burning the full 4-attempt backoff before surfacing.
760
+ return withSnapshotRetry(await withSandboxRetry(
761
+ () => def.reconnect!(sandboxId),
762
+ (error) => error instanceof SandboxNotFoundError ? error.attemptNumber <= 1 : isTransientSandboxError(error),
763
+ ));
442
764
  }
443
765
 
444
766
  /** Delete a snapshot by id on the named provider. No live sandbox needed. */