@agent-compose/sdk 0.8.1 → 0.8.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (48) hide show
  1. package/dist/agent/__tests__/perf-sampler.test.d.ts +10 -0
  2. package/dist/agent/agent-context.d.ts +1 -1
  3. package/dist/agent/agent-loop.d.ts +5 -1
  4. package/dist/agent/desktop-open.d.ts +184 -0
  5. package/dist/agent/perf-sampler.d.ts +99 -0
  6. package/dist/agent/services-manifest.d.ts +88 -0
  7. package/dist/agent/services-restore.d.ts +58 -0
  8. package/dist/client.d.ts +189 -15
  9. package/dist/display.d.ts +17 -0
  10. package/dist/index.d.ts +14 -5
  11. package/dist/index.js +1625 -120
  12. package/dist/runtimes/_cli-agent.d.ts +372 -2
  13. package/dist/runtimes/claude-code.d.ts +12 -0
  14. package/dist/runtimes/codex.buildcommand.test.d.ts +9 -0
  15. package/dist/runtimes/codex.d.ts +8 -0
  16. package/dist/runtimes/openai-desktop.js +1555 -120
  17. package/dist/runtimes/session-env.test.d.ts +14 -0
  18. package/dist/sandbox/sizes.d.ts +120 -30
  19. package/dist/sandbox.d.ts +1 -1
  20. package/dist/types/api-conversations.d.ts +476 -1
  21. package/dist/types/api-factory.d.ts +164 -7
  22. package/dist/types/api-runs.d.ts +23 -1
  23. package/dist/types/protocol.d.ts +32 -1
  24. package/dist/types/runtime.d.ts +120 -0
  25. package/dist/types/workflow-metadata.d.ts +6 -5
  26. package/package.json +1 -1
  27. package/src/agent/agent-context.ts +128 -28
  28. package/src/agent/agent-loop.ts +10 -3
  29. package/src/agent/desktop-open.ts +418 -0
  30. package/src/agent/perf-sampler.ts +202 -0
  31. package/src/agent/services-manifest.ts +356 -0
  32. package/src/agent/services-restore.ts +195 -0
  33. package/src/client.ts +384 -32
  34. package/src/display.ts +44 -1
  35. package/src/index.ts +74 -7
  36. package/src/runtimes/_cli-agent.ts +1160 -67
  37. package/src/runtimes/claude-code.ts +187 -12
  38. package/src/runtimes/codex.ts +65 -2
  39. package/src/sandbox/providers/e2b.ts +8 -4
  40. package/src/sandbox/providers/local.ts +16 -4
  41. package/src/sandbox/sizes.ts +127 -44
  42. package/src/sandbox.ts +8 -0
  43. package/src/types/api-conversations.ts +461 -2
  44. package/src/types/api-factory.ts +165 -7
  45. package/src/types/api-runs.ts +25 -1
  46. package/src/types/protocol.ts +30 -1
  47. package/src/types/runtime.ts +122 -0
  48. package/src/types/workflow-metadata.ts +6 -5
@@ -34,7 +34,7 @@
34
34
  * token-metering gateway's Anthropic passthrough (ADR-0039).
35
35
  */
36
36
 
37
- import type { AgentMessage } from "../index.js";
37
+ import type { AgentMessage, AgentMessageTaskNotification } from "../index.js";
38
38
  import { createCliAgentRuntime, shellQuote, type CliAgentSpec, type CliReasoningEffort } from "./_cli-agent.js";
39
39
  import { formatError } from "../utils/errors.js";
40
40
 
@@ -72,6 +72,140 @@ function toolResultText(content: unknown): string {
72
72
  return content == null ? "" : JSON.stringify(content) ?? "";
73
73
  }
74
74
 
75
+ // ── Task notifications (background-task completion evidence) ────────────────
76
+ //
77
+ // When a background task stops — an async Agent spawn finishing, a
78
+ // background command exiting — claude-code's completion evidence reaches
79
+ // the stream in TWO shapes, and BOTH are mapped here (either one dropped
80
+ // leaves transcripts believing async agents run forever — the incident
81
+ // where spawn cards never left "launched"):
82
+ //
83
+ // - `system` events with `subtype: "task_notification"` — the ONLY shape
84
+ // `-p --output-format stream-json` actually emits, verified live on
85
+ // 2.1.212 (the baked E2B version) and 2.1.236, both when the task
86
+ // finishes MID-turn and on the idle wake (where it precedes a fresh
87
+ // system/init and a result stamped `origin.kind: "task-notification"`).
88
+ // Flat JSON: task_id / tool_use_id / status / summary, plus
89
+ // `usage.{total_tokens,tool_uses,duration_ms}` for agent tasks.
90
+ // - `<task-notification>` XML blocks as USER-role text — the form the
91
+ // harness injects into the model's own conversation (and the shape a
92
+ // resumed turn can surface as a user event). Kept as the second arm so
93
+ // neither transport ever depends on which side of a resume the
94
+ // notification lands.
95
+ //
96
+ // Parsed into structure; the internal plumbing either shape carries
97
+ // (output-file paths, resume hints in <note>/<diagnostics>) is deliberately
98
+ // not forwarded — no renderer should ever see it.
99
+
100
+ const TASK_NOTIFICATION_RE = /<task-notification>([\s\S]*?)<\/task-notification>/g;
101
+
102
+ const NOTIFICATION_SUMMARY_MAX = 500;
103
+ const NOTIFICATION_REPORT_MAX = 20_000;
104
+
105
+ /** First `<tag>…</tag>` inside a notification body, or null. */
106
+ function innerTag(body: string, tag: string): string | null {
107
+ const m = new RegExp(`<${tag}>([\\s\\S]*?)</${tag}>`).exec(body);
108
+ const text = m?.[1]?.trim() ?? "";
109
+ return text.length > 0 ? text : null;
110
+ }
111
+
112
+ /** The harness entity-escapes text it embeds into notification XML —
113
+ * reverse the closed five so the report reads as the agent wrote it. */
114
+ function unescapeEntities(s: string): string {
115
+ return s
116
+ .replace(/&lt;/g, "<")
117
+ .replace(/&gt;/g, ">")
118
+ .replace(/&quot;/g, '"')
119
+ .replace(/&#39;/g, "'")
120
+ .replace(/&amp;/g, "&");
121
+ }
122
+
123
+ const clip = (s: string, max: number): string =>
124
+ (s.length > max ? `${s.slice(0, max - 1)}…` : s);
125
+
126
+ function intTag(body: string, tag: string): number | undefined {
127
+ const raw = innerTag(body, tag);
128
+ if (!raw) return undefined;
129
+ const n = Number.parseInt(raw, 10);
130
+ return Number.isFinite(n) && n >= 0 ? n : undefined;
131
+ }
132
+
133
+ /** Parse every `<task-notification>` block out of one user-role text blob.
134
+ * Pure and tolerant over untrusted harness text: a block without a task id
135
+ * is skipped, absent fields stay absent, everything is clamped. Exported
136
+ * for tests. */
137
+ export function parseTaskNotifications(
138
+ text: string, timestamp: string,
139
+ ): AgentMessageTaskNotification[] {
140
+ if (!text.includes("<task-notification>")) return [];
141
+ const out: AgentMessageTaskNotification[] = [];
142
+ TASK_NOTIFICATION_RE.lastIndex = 0;
143
+ for (let m = TASK_NOTIFICATION_RE.exec(text); m !== null; m = TASK_NOTIFICATION_RE.exec(text)) {
144
+ const body = m[1] ?? "";
145
+ const taskId = innerTag(body, "task-id");
146
+ if (!taskId) continue;
147
+ const toolUseId = innerTag(body, "tool-use-id");
148
+ const summary = innerTag(body, "summary");
149
+ const report = innerTag(body, "result");
150
+ const usageBody = /<usage>([\s\S]*?)<\/usage>/.exec(body)?.[1] ?? "";
151
+ const tokens = intTag(usageBody, "subagent_tokens");
152
+ const toolUses = intTag(usageBody, "tool_uses");
153
+ const durationMs = intTag(usageBody, "duration_ms");
154
+ out.push({
155
+ type: "task_notification",
156
+ taskId: clip(taskId, 128),
157
+ ...(toolUseId ? { toolUseId: clip(toolUseId, 128) } : {}),
158
+ status: innerTag(body, "status") ?? "finished",
159
+ summary: summary ? clip(unescapeEntities(summary.replace(/\s+/g, " ")), NOTIFICATION_SUMMARY_MAX) : "",
160
+ ...(report ? { report: clip(unescapeEntities(report), NOTIFICATION_REPORT_MAX) } : {}),
161
+ ...(tokens !== undefined || toolUses !== undefined || durationMs !== undefined
162
+ ? { usage: {
163
+ ...(tokens !== undefined ? { tokens } : {}),
164
+ ...(toolUses !== undefined ? { toolUses } : {}),
165
+ ...(durationMs !== undefined ? { durationMs } : {}),
166
+ } }
167
+ : {}),
168
+ timestamp,
169
+ });
170
+ }
171
+ return out;
172
+ }
173
+
174
+ /** One `system`/`task_notification` stream-json event mapped onto the same
175
+ * structured message the XML parse produces, or null when the event names
176
+ * no task id. Pure and tolerant over untrusted harness JSON: absent fields
177
+ * stay absent, everything is clamped; `output_file` (internal plumbing) is
178
+ * deliberately not forwarded. Exported for tests. */
179
+ export function parseSystemTaskNotification(
180
+ p: Record<string, unknown>, timestamp: string,
181
+ ): AgentMessageTaskNotification | null {
182
+ const taskId = typeof p.task_id === "string" && p.task_id.trim().length > 0 ? p.task_id.trim() : null;
183
+ if (!taskId) return null;
184
+ const toolUseId = typeof p.tool_use_id === "string" && p.tool_use_id.length > 0 ? p.tool_use_id : null;
185
+ const summary = typeof p.summary === "string" ? p.summary : "";
186
+ const nonneg = (v: unknown): number | undefined =>
187
+ typeof v === "number" && Number.isFinite(v) && v >= 0 ? Math.floor(v) : undefined;
188
+ const u = (typeof p.usage === "object" && p.usage !== null ? p.usage : {}) as Record<string, unknown>;
189
+ const tokens = nonneg(u.total_tokens);
190
+ const toolUses = nonneg(u.tool_uses);
191
+ const durationMs = nonneg(u.duration_ms);
192
+ return {
193
+ type: "task_notification",
194
+ taskId: clip(taskId, 128),
195
+ ...(toolUseId ? { toolUseId: clip(toolUseId, 128) } : {}),
196
+ status: typeof p.status === "string" && p.status.length > 0 ? p.status : "finished",
197
+ summary: clip(summary.replace(/\s+/g, " ").trim(), NOTIFICATION_SUMMARY_MAX),
198
+ ...(tokens !== undefined || toolUses !== undefined || durationMs !== undefined
199
+ ? { usage: {
200
+ ...(tokens !== undefined ? { tokens } : {}),
201
+ ...(toolUses !== undefined ? { toolUses } : {}),
202
+ ...(durationMs !== undefined ? { durationMs } : {}),
203
+ } }
204
+ : {}),
205
+ timestamp,
206
+ };
207
+ }
208
+
75
209
  /** Claude Code's real reasoning knob is its own `--effort <level>` flag
76
210
  * (low|medium|high|xhigh|max — verified against `claude -p --help`). The
77
211
  * CliReasoningEffort union IS the CLI's vocabulary, so the level rides the
@@ -100,12 +234,30 @@ export const claudeCodeSpec: CliAgentSpec = {
100
234
  'sudo cp "$REAL" /usr/local/bin/claude && sudo chmod 0755 /usr/local/bin/claude',
101
235
  // Claude reads the prompt from stdin in -p mode.
102
236
  promptPayload: (prompt) => prompt,
103
- buildCommand: ({ promptPath, sessionId, model, cwd, effort }) => {
237
+ // Mid-turn stream input (steering): with `--input-format stream-json`,
238
+ // stdin carries JSONL user messages, and one that arrives WHILE a turn
239
+ // runs is folded into the running turn at the next tool boundary — the
240
+ // interactive UI's queued-user-input behaviour, verified live against
241
+ // claude 2.1.233 (a message injected during a 20s Bash call shaped the
242
+ // same turn's final reply, and the tool ran undisturbed). Slash-command
243
+ // expansion and `--resume` both work unchanged in this mode (verified on
244
+ // the same build). A message that lands after the result would start a
245
+ // NEW turn in-process, which is why the transport's feeder stops at the
246
+ // result line instead of forwarding past it.
247
+ streamInput: {
248
+ promptLine: (prompt) => JSON.stringify({ type: "user", message: { role: "user", content: prompt } }),
249
+ messageLine: (text) => JSON.stringify({ type: "user", message: { role: "user", content: text } }),
250
+ },
251
+ buildCommand: ({ promptPath, sessionId, model, cwd, effort, streamInput }) => {
104
252
  const flags = [
105
253
  "-p",
106
254
  // stream-json is the JSONL event stream; -p requires --verbose with it.
107
255
  "--output-format stream-json",
108
256
  "--verbose",
257
+ // Stream-input turns feed stdin as JSONL user messages from the
258
+ // transport's FIFO (mid-turn injection); plain turns keep the raw
259
+ // prompt file.
260
+ ...(streamInput ? ["--input-format stream-json"] : []),
109
261
  // Raw API stream events ride along as `stream_event` lines — the
110
262
  // text_delta source for progressive rendering (mapEvent below). The
111
263
  // complete `assistant` message events still arrive; deltas are
@@ -157,17 +309,29 @@ export const claudeCodeSpec: CliAgentSpec = {
157
309
  return [];
158
310
  });
159
311
  }
160
- // User API message: the CLI echoes tool results back as user content.
312
+ // User API message: the CLI echoes tool results back as user content,
313
+ // and injects `<task-notification>` blocks (background-task completion
314
+ // evidence) as user TEXT — parsed into structure, never dropped and
315
+ // never forwarded raw. Other user text (the echo of the prompt, system
316
+ // reminders) stays unmapped: it is not agent output.
161
317
  case "user": {
162
- const message = p.message as { content?: ClaudeContentBlock[] } | undefined;
318
+ const message = p.message as { content?: ClaudeContentBlock[] | string } | undefined;
319
+ if (typeof message?.content === "string") {
320
+ return parseTaskNotifications(message.content, ts);
321
+ }
163
322
  const blocks = Array.isArray(message?.content) ? message.content : [];
164
- return blocks.flatMap((b): AgentMessage[] =>
165
- b.type === "tool_result"
166
- ? [{
167
- type: "tool_result", toolUseId: String(b.tool_use_id ?? ""),
168
- output: toolResultText(b.content), isError: b.is_error === true, ...parent, timestamp: ts,
169
- }]
170
- : []);
323
+ return blocks.flatMap((b): AgentMessage[] => {
324
+ if (b.type === "tool_result") {
325
+ return [{
326
+ type: "tool_result", toolUseId: String(b.tool_use_id ?? ""),
327
+ output: toolResultText(b.content), isError: b.is_error === true, ...parent, timestamp: ts,
328
+ }];
329
+ }
330
+ if (b.type === "text" && typeof b.text === "string") {
331
+ return parseTaskNotifications(b.text, ts);
332
+ }
333
+ return [];
334
+ });
171
335
  }
172
336
  // Raw API stream event (--include-partial-messages): text deltas of
173
337
  // the in-progress block map to the live-only `text_delta` kind so a
@@ -234,7 +398,18 @@ export const claudeCodeSpec: CliAgentSpec = {
234
398
  timestamp: ts,
235
399
  }];
236
400
  }
237
- // system/init carries the session id (extractSessionId); nothing to map.
401
+ // System events: init carries the session id (extractSessionId), and
402
+ // the background-task lane rides here too — `task_notification` is
403
+ // the completion evidence stream-json actually emits (see the section
404
+ // header above), so it maps to the structured message BOTH turn
405
+ // engines persist. Everything else under system (task_started,
406
+ // task_updated, background_tasks_changed, thinking_tokens) is
407
+ // lifecycle noise here.
408
+ case "system": {
409
+ if (p.subtype !== "task_notification") return [];
410
+ const notification = parseSystemTaskNotification(p, ts);
411
+ return notification ? [notification] : [];
412
+ }
238
413
  default:
239
414
  return [];
240
415
  }
@@ -53,6 +53,54 @@ export function isCodexAdvisoryNoise(text: string): boolean {
53
53
  return CODEX_ADVISORY_PATTERNS.some((re) => re.test(text));
54
54
  }
55
55
 
56
+ // ── Thread-store writer-lock preflight (2026-08-18 prod incident) ────────────
57
+ // codex guards each persisted thread with an OS advisory flock on
58
+ // `$CODEX_HOME/thread-writer-locks/<thread-id>.lock` (codex-rs
59
+ // thread-store/src/local/writer_lock.rs). Two consequences drive this
60
+ // preflight's shape:
61
+ // - the flock is held exactly as long as the holding PROCESS lives — the
62
+ // kernel releases it on any death (SIGKILL, VM crash), so a lock that
63
+ // still blocks is held by a LIVE process, never a stale file;
64
+ // - codex sweeps unheld (stale) lock FILES itself on store init.
65
+ // The incident: a canceled turn's codex survived its fire-and-forget SIGTERM
66
+ // long enough for the user's next turn to launch `codex exec resume`, which
67
+ // died with `thread-store conflict: thread … already has an active writer`
68
+ // (code -32600). So before a RESUME the launch script:
69
+ // 1. waits (bounded) for a dying previous writer to release the flock —
70
+ // the canceled predecessor was already TERM'd, it just needs a moment;
71
+ // 2. clears the lock file ONLY under a successfully acquired `flock -n`
72
+ // (proof the holder is dead) — belt-and-suspenders on top of codex's
73
+ // own sweep, and the guard against any future file-existence check;
74
+ // 3. NEVER touches a lock whose holder is alive: removing a live-held
75
+ // lock file would let a second writer flock a fresh inode (two writers
76
+ // on one rollout), so a still-held lock after the wait is left for
77
+ // codex to surface as the honest conflict it is.
78
+ // No `flock(1)` on the guest (or no lock file) → the preflight is a no-op.
79
+
80
+ /** Bounded wait for a dying previous writer: attempts × sleep = 5s, matching
81
+ * the teardown's SIGTERM grace (sdk _cli-agent.ts REAP_TERM_WAIT_ATTEMPTS). */
82
+ export const CODEX_LOCK_WAIT_ATTEMPTS = 20;
83
+ export const CODEX_LOCK_WAIT_SECONDS = "0.25";
84
+
85
+ /** Thread ids are uuid-shaped; anything else skips the preflight entirely so
86
+ * hostile config can never become shell injection via the lock path. */
87
+ const CODEX_THREAD_ID_SAFE = /^[A-Za-z0-9_-]{1,64}$/;
88
+
89
+ /** The sh fragment prepended to a resume launch. `waitAttempts` is a test
90
+ * seam (the live-holder test must not sleep 5s); production callers take
91
+ * the default. Exported for tests. */
92
+ export function codexWriterLockPreflight(
93
+ threadId: string,
94
+ waitAttempts: number = CODEX_LOCK_WAIT_ATTEMPTS,
95
+ ): string {
96
+ if (!CODEX_THREAD_ID_SAFE.test(threadId)) return "";
97
+ return `ac_lk="\${CODEX_HOME:-$HOME/.codex}/thread-writer-locks/${threadId}.lock"; `
98
+ + `if [ -e "$ac_lk" ] && command -v flock >/dev/null 2>&1; then ac_lki=0; `
99
+ + `while ! flock -n "$ac_lk" true 2>/dev/null && [ "$ac_lki" -lt ${waitAttempts} ]; do `
100
+ + `sleep ${CODEX_LOCK_WAIT_SECONDS}; ac_lki=$((ac_lki+1)); done; `
101
+ + `flock -n "$ac_lk" rm -f -- "$ac_lk" 2>/dev/null || true; fi; `;
102
+ }
103
+
56
104
  /** Exported for the fixture-based parity tests (ADR-0020): the legacy JSONL
57
105
  * `mapEvent` is the golden the ACP normaliser is asserted equal to. Not part
58
106
  * of the public runtime surface — `createCodexRuntime` stays the entry point. */
@@ -60,6 +108,11 @@ export const codexSpec: CliAgentSpec = {
60
108
  kind: "codex",
61
109
  authEnv: "CODEX_API_KEY",
62
110
  bin: "codex",
111
+ // codex holds a per-thread advisory flock for its writer's whole lifetime
112
+ // (see the writer-lock preflight above): a live predecessor process blocks
113
+ // every resume of the same thread, so supersede teardown must KILL it —
114
+ // never detach it alive (server runner-kill.ts consumes this flag).
115
+ exclusiveSessionWriter: true,
63
116
  // ACP-mode invocation (ADR-0020 increment 1). `codex` has no native `--acp`
64
117
  // flag; the adapter (a Rust binary shipped via npm) speaks ACP and drives a
65
118
  // compatible bundled `@openai/codex`. The runner attempts this first and
@@ -95,14 +148,24 @@ export const codexSpec: CliAgentSpec = {
95
148
  // backstop for a caller that bypasses that gate. The value comes from
96
149
  // the closed CliReasoningEffort set, so it is shell-safe unquoted.
97
150
  ...(effort ? ["-c", `model_reasoning_effort=${effort === "max" ? "xhigh" : effort}`] : []),
98
- ...(cwd ? ["-C", shellQuote(cwd)] : []),
99
151
  ].join(" ");
152
+ // Working directory via a shell `cd` — the idiom EVERY other runtime uses
153
+ // (claude-code/cursor/droid/opencode) — NOT codex's `-C` flag. `-C` is
154
+ // accepted by `codex exec` but REJECTED by `codex exec resume`
155
+ // ("unexpected argument '-C'"), so a fresh turn worked and every follow-up
156
+ // died. `cd` sets the cwd identically for both paths; promptPath is
157
+ // absolute (/tmp/…), so the `< prompt` redirect survives the cd.
158
+ const cd = cwd ? `cd ${shellQuote(cwd)} && ` : "";
159
+ // Resume-only launch preflight: self-heal a released-or-dying writer
160
+ // lock before `codex exec resume` (see codexWriterLockPreflight). A
161
+ // fresh turn mints a new thread id — no lock can exist for it yet.
162
+ const preflight = sessionId ? codexWriterLockPreflight(sessionId) : "";
100
163
  // Fresh turn: `codex exec <flags> - < prompt`. Continue a thread:
101
164
  // `codex exec resume <id> <flags> - < prompt`. (`-` = read prompt from stdin.)
102
165
  const exec = sessionId
103
166
  ? `codex exec resume ${shellQuote(sessionId)} ${flags}`
104
167
  : `codex exec ${flags}`;
105
- return `${exec} - < ${shellQuote(promptPath)}`;
168
+ return `${preflight}${cd}${exec} - < ${shellQuote(promptPath)}`;
106
169
  },
107
170
  extractSessionId: (p) =>
108
171
  p.type === "thread.started" && typeof p.thread_id === "string" ? p.thread_id : undefined,
@@ -237,11 +237,15 @@ function makeE2bSandboxProvider(sb: Sandbox): SandboxProvider {
237
237
  return { resumeHandle: sb.sandboxId };
238
238
  },
239
239
  // The provider kill-clock seam: e2b's `setTimeout` REPLACES the deadline
240
- // (extend or reduce) relative to now. The server's session lifecycle
241
- // keeps this behind its own suspend clock so the deferred pause always
242
- // wins the race against the create-time timeout.
240
+ // (extend or reduce) relative to now, in MILLISECONDS. The server's
241
+ // session lifecycle uses it two ways: a FAR-HORIZON push while the
242
+ // session is active (the kill-clock is an orphan backstop, never
243
+ // load-bearing during a live turn) and a short re-arm when the deferred
244
+ // pause is about to park the VM. Clamped to the plan cap exactly like
245
+ // the create timeout — an over-cap push would 400 and leave the OLD
246
+ // (possibly short) deadline standing.
243
247
  async extendLifetime(ms) {
244
- await sb.setTimeout(ms);
248
+ await sb.setTimeout(Math.min(ms, e2bMaxSandboxMs()));
245
249
  },
246
250
  // Push a freshly-resolved egress policy onto the live sandbox via E2B's
247
251
  // native `updateNetwork` — the E2B analogue of Vercel's `update({
@@ -5,6 +5,16 @@
5
5
  import { promises as fs } from "node:fs";
6
6
  import { dirname } from "node:path";
7
7
  import { spawn } from "node:child_process";
8
+ /** The two listener registrations we need, declared LOCALLY.
9
+ * `ChildProcess` inherits `.on` from EventEmitter, but when a tree ends up
10
+ * with more than one @types/node the inheritance link breaks and `.on`
11
+ * disappears — which resolution you get depends on install layout, so this
12
+ * typechecked locally and failed CI three times. Referencing no node types
13
+ * at all is the only version-proof shape. */
14
+ type ProcListeners = {
15
+ on(event: "error", cb: (err: Error) => void): unknown;
16
+ on(event: "close", cb: (code: number | null) => void): unknown;
17
+ };
8
18
  import { Readable, Writable } from "node:stream";
9
19
  import type { SandboxProvider } from "../../types/sandbox.js";
10
20
 
@@ -36,8 +46,9 @@ export function makeLocalSandboxProvider(): SandboxProvider {
36
46
  proc.stderr?.setEncoding("utf8");
37
47
  proc.stdout?.on("data", (chunk: string) => { stdout += chunk; opts?.onStdout?.(chunk); });
38
48
  proc.stderr?.on("data", (chunk: string) => { stderr += chunk; opts?.onStderr?.(chunk); });
39
- proc.on("error", reject);
40
- proc.on("close", (code) => resolve({ exitCode: code ?? 0, stdout, stderr }));
49
+ const ev = proc as unknown as ProcListeners;
50
+ ev.on("error", reject);
51
+ ev.on("close", (code: number | null) => resolve({ exitCode: code ?? 0, stdout, stderr }));
41
52
  });
42
53
  },
43
54
  // Duplex spawn — the in-VM `child_process` pipe the ACP client needs.
@@ -67,8 +78,9 @@ export function makeLocalSandboxProvider(): SandboxProvider {
67
78
  proc.stderr.on("data", (chunk: string) => { stderr += chunk; });
68
79
 
69
80
  const exited = new Promise<{ exitCode: number; stderr: string }>((resolve, reject) => {
70
- proc.on("error", reject);
71
- proc.on("close", (code) => resolve({ exitCode: code ?? 0, stderr }));
81
+ const ev = proc as unknown as ProcListeners;
82
+ ev.on("error", reject);
83
+ ev.on("close", (code: number | null) => resolve({ exitCode: code ?? 0, stderr }));
72
84
  });
73
85
 
74
86
  return {
@@ -1,27 +1,68 @@
1
1
  /**
2
- * Sandbox machine sizes + the E2B template aliases derived from them.
2
+ * Sandbox machine sizes — THE single source of the size vocabulary.
3
3
  *
4
- * A coarse hardware knob that maps to provider machine specs at create time:
5
- * Vercel honours it natively via `resources.vcpus`; E2B sizing is baked into
6
- * the template, so on E2B a size resolves to a pre-built per-size template.
4
+ * Everything that names a size (the server's `SANDBOX_DEFAULT_SIZE` env enum,
5
+ * the register/invoke zod schemas, the run + session row types, the CLI's
6
+ * `--size` flag, the dashboard pickers via `GET /v1/sandbox-sizes`) derives
7
+ * from `SANDBOX_SIZES` / `SandboxSize` here. Adding a size is a ONE-LINE edit
8
+ * to `SANDBOX_MACHINES`: the type, the enums, the E2B build matrix and the
9
+ * pickers all follow. Do NOT re-declare the union inline anywhere.
10
+ *
11
+ * A size is a coarse hardware knob that maps to provider machine specs:
12
+ * Vercel honours it natively via `resources.vcpus`; E2B sizing is BAKED INTO
13
+ * THE TEMPLATE (e2b 2.30.5 has no create-time cpu/mem knob — `NewSandbox`
14
+ * carries only `templateID`), so on E2B a size resolves to a pre-built
15
+ * per-size template.
7
16
  */
8
17
 
9
- /** Sandbox hardware SKU. Named for the actual machine spec (vCPU + RAM) rather
10
- * than abstract t-shirt sizes. Memory is always 2048 MB per vCPU:
11
- * 2vcpu-4gb = 2 vCPU / 4 GiB (Vercel's own default machine)
12
- * 4vcpu-8gb = 4 vCPU / 8 GiB
13
- * 8vcpu-16gb = 8 vCPU / 16 GiB (per-sandbox ceiling on STANDARD accounts —
14
- * probed live: 16 & 32 vCPU 400 on dev)
15
- * 32vcpu-64gb = 32 vCPU / 64 GiB (ENTERPRISE ONLY — standard accounts reject >8 vCPU) */
16
- export type SandboxSize = "2vcpu-4gb" | "4vcpu-8gb" | "8vcpu-16gb" | "32vcpu-64gb";
17
-
18
- /** SKU → Vercel vCPU count (RAM follows at 2048 MB/vCPU). */
19
- export const SANDBOX_VCPUS: Record<SandboxSize, number> = {
20
- "2vcpu-4gb": 2,
21
- "4vcpu-8gb": 4,
22
- "8vcpu-16gb": 8,
23
- "32vcpu-64gb": 32,
24
- };
18
+ /** Every sandbox hardware SKU, with its real machine spec. Named for the
19
+ * machine (vCPU + RAM) rather than abstract t-shirt sizes, so a size can
20
+ * never quietly mean something different than it says.
21
+ *
22
+ * RAM is 2048 MB/vCPU everywhere EXCEPT `8vcpu-8gb`, which exists because
23
+ * E2B caps a sandbox at 8 vCPU / 8192 MB (e2b.dev/docs/billing: Hobby and
24
+ * Pro both "8 vCPU / 8 GB", raised only by arrangement): 8 vCPU at the 2 GB
25
+ * rule would need 16 GiB and cannot be built. `8vcpu-8gb` is the CPU ceiling
26
+ * at the memory ceiling — the only way to get 8 cores on E2B today, and a
27
+ * spec already proven bakeable by the devbox (`E2B_DEVBOX_SPEC`).
28
+ *
29
+ * Adding an entry here automatically: widens `SandboxSize`, widens every
30
+ * derived enum, and — if it fits under the E2B caps — adds it to
31
+ * `E2B_TEMPLATE_SIZES`, which is what `infra/e2b-template/build.ts` and the
32
+ * `sandbox-images` CI job loop over. Two templates get baked per size, so
33
+ * the matrix is not free; see that workflow's header. */
34
+ export const SANDBOX_MACHINES = {
35
+ /** 1 vCPU / 2 GiB — the cheap floor. Plenty for a terminal session or a
36
+ * shell-shaped agent; tight for a big `bun install` or a browser. */
37
+ "1vcpu-2gb": { vcpus: 1, memoryMB: 2048 },
38
+ /** 2 vCPU / 4 GiB — the default (Vercel's own default machine too). */
39
+ "2vcpu-4gb": { vcpus: 2, memoryMB: 4096 },
40
+ /** 4 vCPU / 8 GiB — comfortable for builds and multi-tool agent turns. */
41
+ "4vcpu-8gb": { vcpus: 4, memoryMB: 8192 },
42
+ /** 8 vCPU / 8 GiB — E2B's per-sandbox CEILING (cores maxed at the memory
43
+ * cap). NOT expressible on Vercel, whose RAM follows vCPUs at 2 GB each. */
44
+ "8vcpu-8gb": { vcpus: 8, memoryMB: 8192 },
45
+ /** 8 vCPU / 16 GiB — Vercel only; exceeds E2B's 8 GiB memory cap. */
46
+ "8vcpu-16gb": { vcpus: 8, memoryMB: 16384 },
47
+ /** 32 vCPU / 64 GiB — Vercel Enterprise only; far past every E2B cap. */
48
+ "32vcpu-64gb": { vcpus: 32, memoryMB: 65536 },
49
+ } as const satisfies Record<string, { vcpus: number; memoryMB: number }>;
50
+
51
+ /** Sandbox hardware SKU. Derived from `SANDBOX_MACHINES` — never re-spelled
52
+ * as an inline union. */
53
+ export type SandboxSize = keyof typeof SANDBOX_MACHINES;
54
+
55
+ /** The vocabulary as an ordered, smallest-first array — the shape zod
56
+ * (`z.enum`), the CLI's `--size` validation, and the wire catalogue want.
57
+ * Ordering is the pickers' display order, so keep it ascending. */
58
+ export const SANDBOX_SIZES = Object.keys(SANDBOX_MACHINES) as readonly SandboxSize[] as
59
+ readonly [SandboxSize, ...SandboxSize[]];
60
+
61
+ /** SKU → Vercel vCPU count (Vercel's RAM follows automatically at 2048
62
+ * MB/vCPU — which is why `isVercelSupportedSize` exists). */
63
+ export const SANDBOX_VCPUS: Record<SandboxSize, number> = Object.fromEntries(
64
+ SANDBOX_SIZES.map((s) => [s, SANDBOX_MACHINES[s].vcpus]),
65
+ ) as Record<SandboxSize, number>;
25
66
 
26
67
  /** SDK fallback size when neither the caller nor the deployment specifies one.
27
68
  * Deliberately conservative — the OPERATIONAL default is the server's
@@ -29,37 +70,79 @@ export const SANDBOX_VCPUS: Record<SandboxSize, number> = {
29
70
  * small matters because Vercel rate-limits creation by vCPUs-per-window
30
71
  * (`api-sandboxes-vcpus-creation`); a large default 429s bursty/simultaneous
31
72
  * creates. Workloads that need more RAM/CPU declare `resources.size` on the
32
- * workflow rather than inflating the default for everyone. */
73
+ * workflow rather than inflating the default for everyone.
74
+ *
75
+ * NOT `1vcpu-2gb`: the floor is an opt-IN for cheap sessions, not a quiet
76
+ * downgrade of every existing run's machine. */
33
77
  export const DEFAULT_SANDBOX_SIZE: SandboxSize = "2vcpu-4gb";
34
78
 
35
- /** The E2B sizes we pre-build a template for. E2B sizing is template-baked
36
- * (no per-create cpu/mem knob), so honouring `resources.size` on E2B means
37
- * ONE pre-built template per size. `32vcpu-64gb` is absent (E2B has no
38
- * >8-vCPU equivalent). `8vcpu-16gb` is also absent: it needs 16 GiB RAM, but
39
- * the E2B account caps memory at 8 GiB (`Template.build` 400s with
40
- * "Memory can't be higher than 8192 MiB"). Add it back here (and rebuild the
41
- * templates) only once the account's memory limit is raised. The register/
42
- * invoke guards reject an unsupported E2B size before it can reach here. */
43
- export const E2B_TEMPLATE_SIZES: readonly SandboxSize[] = [
44
- "2vcpu-4gb",
45
- "4vcpu-8gb",
46
- ];
47
-
48
- /** Is `size` one E2B can be built/booted at? `32vcpu-64gb` (no >8-vCPU E2B
49
- * equivalent) and `8vcpu-16gb` (exceeds the account's 8 GiB memory cap) are
50
- * not — the guards lean on this so the "no E2B equivalent" decision lives in
51
- * exactly one place. */
79
+ /** SESSION default — deliberately one size up from the run default
80
+ * (2026-08-13): a session's sandbox carries the full desktop toolbelt
81
+ * (VS Code + Chromium + dockerd) plus the KasmVNC encoder at the 60fps
82
+ * cap, and that stack swap-thrashes on 4 GiB while the encoder starves on
83
+ * 2 shared vCPUs. Workflow runs keep DEFAULT_SANDBOX_SIZE — no desktop,
84
+ * no toolbelt weight. Sessions bill active time only (parked = storage),
85
+ * so the delta applies to active hours, not the fleet. */
86
+ export const SESSION_DEFAULT_SANDBOX_SIZE: SandboxSize = "4vcpu-8gb";
87
+
88
+ /** E2B's per-sandbox ceiling on the plans we run (e2b.dev/docs/billing —
89
+ * Hobby: "8 vCPU / 8 GB"; Pro: the same, "8+" only by arrangement with
90
+ * support). Recorded live too: `Template.build` 400s with "Memory can't be
91
+ * higher than 8192 MiB" past the memory cap.
92
+ *
93
+ * These two numbers are the ONLY knob for which sizes get an E2B template —
94
+ * raise them after E2B raises the account limit and the build matrix (and
95
+ * therefore the session picker) widens on its own. */
96
+ export const E2B_MAX_VCPUS = 8;
97
+ export const E2B_MAX_MEMORY_MB = 8192;
98
+
99
+ /** The E2B sizes we pre-build a template for — DERIVED from the caps, not
100
+ * hand-listed, so a new `SANDBOX_MACHINES` entry can never be offered
101
+ * without a template or omitted despite fitting. E2B sizing is
102
+ * template-baked (no per-create cpu/mem knob), so honouring `resources.size`
103
+ * on E2B means ONE pre-built template per size; `infra/e2b-template/build.ts`
104
+ * loops exactly this list. The register / invoke / session-spawn / resize
105
+ * guards all reject an unsupported E2B size before it can reach a create. */
106
+ export const E2B_TEMPLATE_SIZES: readonly SandboxSize[] = SANDBOX_SIZES.filter(
107
+ (s) => SANDBOX_MACHINES[s].vcpus <= E2B_MAX_VCPUS
108
+ && SANDBOX_MACHINES[s].memoryMB <= E2B_MAX_MEMORY_MB,
109
+ );
110
+
111
+ /** Is `size` one E2B can be built/booted at? False for the sizes past E2B's
112
+ * 8 vCPU / 8 GiB ceiling (`8vcpu-16gb`, `32vcpu-64gb`) — those run on Vercel.
113
+ * The "no E2B equivalent" decision lives in exactly one place: the caps. */
52
114
  export function isE2bSupportedSize(size: SandboxSize): boolean {
53
115
  return E2B_TEMPLATE_SIZES.includes(size);
54
116
  }
55
117
 
56
- /** Machine spec for a SandboxSize, in the shape `Template.build` wants. RAM is
57
- * always 2048 MB/vCPU, matching the size name + Vercel parity
58
- * (`SANDBOX_VCPUS` × 2048). Used by `infra/e2b-template/build.ts` to stamp the
59
- * per-size base + agent-env templates. */
118
+ /** Vercel's fixed memory-per-vCPU ratio. Vercel takes `resources.vcpus` and
119
+ * allocates RAM itself at this rate — there is no independent memory knob. */
120
+ export const VERCEL_MEMORY_MB_PER_VCPU = 2048;
121
+
122
+ /** Is `size` expressible on Vercel? Only when its RAM matches what Vercel
123
+ * would allocate for that vCPU count — otherwise asking for it would hand
124
+ * the caller a machine that does not match the name (`8vcpu-8gb` would come
125
+ * back with 16 GiB). Vercel's own ceiling (32 vCPU, Enterprise) is a plan
126
+ * matter, not a shape matter, so it is not encoded here. */
127
+ export function isVercelSupportedSize(size: SandboxSize): boolean {
128
+ const m = SANDBOX_MACHINES[size];
129
+ return m.memoryMB === m.vcpus * VERCEL_MEMORY_MB_PER_VCPU;
130
+ }
131
+
132
+ /** Machine spec for a SandboxSize, in the shape `Template.build` wants. Used
133
+ * by `infra/e2b-template/build.ts` to stamp the per-size base + agent-env
134
+ * templates. Reads the explicit table rather than deriving RAM from vCPUs —
135
+ * `8vcpu-8gb` is deliberately off the 2048 MB/vCPU line. */
60
136
  export function e2bMachineSpec(size: SandboxSize): { cpuCount: number; memoryMB: number } {
61
- const cpuCount = SANDBOX_VCPUS[size];
62
- return { cpuCount, memoryMB: cpuCount * 2048 };
137
+ const m = SANDBOX_MACHINES[size];
138
+ return { cpuCount: m.vcpus, memoryMB: m.memoryMB };
139
+ }
140
+
141
+ /** Human label for a size — "2 vCPU · 4 GB". The wire catalogue carries it so
142
+ * the dashboard never has to parse the id back into numbers. */
143
+ export function sandboxSizeLabel(size: SandboxSize): string {
144
+ const m = SANDBOX_MACHINES[size];
145
+ return `${m.vcpus} vCPU · ${Math.round(m.memoryMB / 1024)} GB`;
63
146
  }
64
147
 
65
148
  /** Stable E2B template ALIAS for the platform base at a given size
package/src/sandbox.ts CHANGED
@@ -23,10 +23,18 @@ export { DOT_SEGMENT_PATH_RE2, toVercelNetworkPolicy, toE2bNetwork } from "./san
23
23
 
24
24
  export type { SandboxSize } from "./sandbox/sizes.js";
25
25
  export {
26
+ SANDBOX_SIZES,
27
+ SANDBOX_MACHINES,
26
28
  SANDBOX_VCPUS,
27
29
  DEFAULT_SANDBOX_SIZE,
30
+ SESSION_DEFAULT_SANDBOX_SIZE,
28
31
  E2B_TEMPLATE_SIZES,
32
+ E2B_MAX_VCPUS,
33
+ E2B_MAX_MEMORY_MB,
34
+ VERCEL_MEMORY_MB_PER_VCPU,
29
35
  isE2bSupportedSize,
36
+ isVercelSupportedSize,
37
+ sandboxSizeLabel,
30
38
  e2bMachineSpec,
31
39
  e2bBaseTemplate,
32
40
  e2bAgentEnvTemplate,