@agent-compose/sdk 0.8.1 → 0.8.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/agent/__tests__/perf-sampler.test.d.ts +10 -0
- package/dist/agent/agent-context.d.ts +1 -1
- package/dist/agent/agent-loop.d.ts +5 -1
- package/dist/agent/desktop-open.d.ts +184 -0
- package/dist/agent/perf-sampler.d.ts +99 -0
- package/dist/agent/services-manifest.d.ts +88 -0
- package/dist/agent/services-restore.d.ts +58 -0
- package/dist/client.d.ts +189 -15
- package/dist/display.d.ts +17 -0
- package/dist/index.d.ts +14 -5
- package/dist/index.js +1625 -120
- package/dist/runtimes/_cli-agent.d.ts +372 -2
- package/dist/runtimes/claude-code.d.ts +12 -0
- package/dist/runtimes/codex.buildcommand.test.d.ts +9 -0
- package/dist/runtimes/codex.d.ts +8 -0
- package/dist/runtimes/openai-desktop.js +1555 -120
- package/dist/runtimes/session-env.test.d.ts +14 -0
- package/dist/sandbox/sizes.d.ts +120 -30
- package/dist/sandbox.d.ts +1 -1
- package/dist/types/api-conversations.d.ts +476 -1
- package/dist/types/api-factory.d.ts +164 -7
- package/dist/types/api-runs.d.ts +23 -1
- package/dist/types/protocol.d.ts +32 -1
- package/dist/types/runtime.d.ts +120 -0
- package/dist/types/workflow-metadata.d.ts +6 -5
- package/package.json +1 -1
- package/src/agent/agent-context.ts +128 -28
- package/src/agent/agent-loop.ts +10 -3
- package/src/agent/desktop-open.ts +418 -0
- package/src/agent/perf-sampler.ts +202 -0
- package/src/agent/services-manifest.ts +356 -0
- package/src/agent/services-restore.ts +195 -0
- package/src/client.ts +384 -32
- package/src/display.ts +44 -1
- package/src/index.ts +74 -7
- package/src/runtimes/_cli-agent.ts +1160 -67
- package/src/runtimes/claude-code.ts +187 -12
- package/src/runtimes/codex.ts +65 -2
- package/src/sandbox/providers/e2b.ts +8 -4
- package/src/sandbox/providers/local.ts +16 -4
- package/src/sandbox/sizes.ts +127 -44
- package/src/sandbox.ts +8 -0
- package/src/types/api-conversations.ts +461 -2
- package/src/types/api-factory.ts +165 -7
- package/src/types/api-runs.ts +25 -1
- package/src/types/protocol.ts +30 -1
- package/src/types/runtime.ts +122 -0
- package/src/types/workflow-metadata.ts +6 -5
|
@@ -34,7 +34,7 @@
|
|
|
34
34
|
* token-metering gateway's Anthropic passthrough (ADR-0039).
|
|
35
35
|
*/
|
|
36
36
|
|
|
37
|
-
import type { AgentMessage } from "../index.js";
|
|
37
|
+
import type { AgentMessage, AgentMessageTaskNotification } from "../index.js";
|
|
38
38
|
import { createCliAgentRuntime, shellQuote, type CliAgentSpec, type CliReasoningEffort } from "./_cli-agent.js";
|
|
39
39
|
import { formatError } from "../utils/errors.js";
|
|
40
40
|
|
|
@@ -72,6 +72,140 @@ function toolResultText(content: unknown): string {
|
|
|
72
72
|
return content == null ? "" : JSON.stringify(content) ?? "";
|
|
73
73
|
}
|
|
74
74
|
|
|
75
|
+
// ── Task notifications (background-task completion evidence) ────────────────
|
|
76
|
+
//
|
|
77
|
+
// When a background task stops — an async Agent spawn finishing, a
|
|
78
|
+
// background command exiting — claude-code's completion evidence reaches
|
|
79
|
+
// the stream in TWO shapes, and BOTH are mapped here (either one dropped
|
|
80
|
+
// leaves transcripts believing async agents run forever — the incident
|
|
81
|
+
// where spawn cards never left "launched"):
|
|
82
|
+
//
|
|
83
|
+
// - `system` events with `subtype: "task_notification"` — the ONLY shape
|
|
84
|
+
// `-p --output-format stream-json` actually emits, verified live on
|
|
85
|
+
// 2.1.212 (the baked E2B version) and 2.1.236, both when the task
|
|
86
|
+
// finishes MID-turn and on the idle wake (where it precedes a fresh
|
|
87
|
+
// system/init and a result stamped `origin.kind: "task-notification"`).
|
|
88
|
+
// Flat JSON: task_id / tool_use_id / status / summary, plus
|
|
89
|
+
// `usage.{total_tokens,tool_uses,duration_ms}` for agent tasks.
|
|
90
|
+
// - `<task-notification>` XML blocks as USER-role text — the form the
|
|
91
|
+
// harness injects into the model's own conversation (and the shape a
|
|
92
|
+
// resumed turn can surface as a user event). Kept as the second arm so
|
|
93
|
+
// neither transport ever depends on which side of a resume the
|
|
94
|
+
// notification lands.
|
|
95
|
+
//
|
|
96
|
+
// Parsed into structure; the internal plumbing either shape carries
|
|
97
|
+
// (output-file paths, resume hints in <note>/<diagnostics>) is deliberately
|
|
98
|
+
// not forwarded — no renderer should ever see it.
|
|
99
|
+
|
|
100
|
+
const TASK_NOTIFICATION_RE = /<task-notification>([\s\S]*?)<\/task-notification>/g;
|
|
101
|
+
|
|
102
|
+
const NOTIFICATION_SUMMARY_MAX = 500;
|
|
103
|
+
const NOTIFICATION_REPORT_MAX = 20_000;
|
|
104
|
+
|
|
105
|
+
/** First `<tag>…</tag>` inside a notification body, or null. */
|
|
106
|
+
function innerTag(body: string, tag: string): string | null {
|
|
107
|
+
const m = new RegExp(`<${tag}>([\\s\\S]*?)</${tag}>`).exec(body);
|
|
108
|
+
const text = m?.[1]?.trim() ?? "";
|
|
109
|
+
return text.length > 0 ? text : null;
|
|
110
|
+
}
|
|
111
|
+
|
|
112
|
+
/** The harness entity-escapes text it embeds into notification XML —
|
|
113
|
+
* reverse the closed five so the report reads as the agent wrote it. */
|
|
114
|
+
function unescapeEntities(s: string): string {
|
|
115
|
+
return s
|
|
116
|
+
.replace(/</g, "<")
|
|
117
|
+
.replace(/>/g, ">")
|
|
118
|
+
.replace(/"/g, '"')
|
|
119
|
+
.replace(/'/g, "'")
|
|
120
|
+
.replace(/&/g, "&");
|
|
121
|
+
}
|
|
122
|
+
|
|
123
|
+
const clip = (s: string, max: number): string =>
|
|
124
|
+
(s.length > max ? `${s.slice(0, max - 1)}…` : s);
|
|
125
|
+
|
|
126
|
+
function intTag(body: string, tag: string): number | undefined {
|
|
127
|
+
const raw = innerTag(body, tag);
|
|
128
|
+
if (!raw) return undefined;
|
|
129
|
+
const n = Number.parseInt(raw, 10);
|
|
130
|
+
return Number.isFinite(n) && n >= 0 ? n : undefined;
|
|
131
|
+
}
|
|
132
|
+
|
|
133
|
+
/** Parse every `<task-notification>` block out of one user-role text blob.
|
|
134
|
+
* Pure and tolerant over untrusted harness text: a block without a task id
|
|
135
|
+
* is skipped, absent fields stay absent, everything is clamped. Exported
|
|
136
|
+
* for tests. */
|
|
137
|
+
export function parseTaskNotifications(
|
|
138
|
+
text: string, timestamp: string,
|
|
139
|
+
): AgentMessageTaskNotification[] {
|
|
140
|
+
if (!text.includes("<task-notification>")) return [];
|
|
141
|
+
const out: AgentMessageTaskNotification[] = [];
|
|
142
|
+
TASK_NOTIFICATION_RE.lastIndex = 0;
|
|
143
|
+
for (let m = TASK_NOTIFICATION_RE.exec(text); m !== null; m = TASK_NOTIFICATION_RE.exec(text)) {
|
|
144
|
+
const body = m[1] ?? "";
|
|
145
|
+
const taskId = innerTag(body, "task-id");
|
|
146
|
+
if (!taskId) continue;
|
|
147
|
+
const toolUseId = innerTag(body, "tool-use-id");
|
|
148
|
+
const summary = innerTag(body, "summary");
|
|
149
|
+
const report = innerTag(body, "result");
|
|
150
|
+
const usageBody = /<usage>([\s\S]*?)<\/usage>/.exec(body)?.[1] ?? "";
|
|
151
|
+
const tokens = intTag(usageBody, "subagent_tokens");
|
|
152
|
+
const toolUses = intTag(usageBody, "tool_uses");
|
|
153
|
+
const durationMs = intTag(usageBody, "duration_ms");
|
|
154
|
+
out.push({
|
|
155
|
+
type: "task_notification",
|
|
156
|
+
taskId: clip(taskId, 128),
|
|
157
|
+
...(toolUseId ? { toolUseId: clip(toolUseId, 128) } : {}),
|
|
158
|
+
status: innerTag(body, "status") ?? "finished",
|
|
159
|
+
summary: summary ? clip(unescapeEntities(summary.replace(/\s+/g, " ")), NOTIFICATION_SUMMARY_MAX) : "",
|
|
160
|
+
...(report ? { report: clip(unescapeEntities(report), NOTIFICATION_REPORT_MAX) } : {}),
|
|
161
|
+
...(tokens !== undefined || toolUses !== undefined || durationMs !== undefined
|
|
162
|
+
? { usage: {
|
|
163
|
+
...(tokens !== undefined ? { tokens } : {}),
|
|
164
|
+
...(toolUses !== undefined ? { toolUses } : {}),
|
|
165
|
+
...(durationMs !== undefined ? { durationMs } : {}),
|
|
166
|
+
} }
|
|
167
|
+
: {}),
|
|
168
|
+
timestamp,
|
|
169
|
+
});
|
|
170
|
+
}
|
|
171
|
+
return out;
|
|
172
|
+
}
|
|
173
|
+
|
|
174
|
+
/** One `system`/`task_notification` stream-json event mapped onto the same
|
|
175
|
+
* structured message the XML parse produces, or null when the event names
|
|
176
|
+
* no task id. Pure and tolerant over untrusted harness JSON: absent fields
|
|
177
|
+
* stay absent, everything is clamped; `output_file` (internal plumbing) is
|
|
178
|
+
* deliberately not forwarded. Exported for tests. */
|
|
179
|
+
export function parseSystemTaskNotification(
|
|
180
|
+
p: Record<string, unknown>, timestamp: string,
|
|
181
|
+
): AgentMessageTaskNotification | null {
|
|
182
|
+
const taskId = typeof p.task_id === "string" && p.task_id.trim().length > 0 ? p.task_id.trim() : null;
|
|
183
|
+
if (!taskId) return null;
|
|
184
|
+
const toolUseId = typeof p.tool_use_id === "string" && p.tool_use_id.length > 0 ? p.tool_use_id : null;
|
|
185
|
+
const summary = typeof p.summary === "string" ? p.summary : "";
|
|
186
|
+
const nonneg = (v: unknown): number | undefined =>
|
|
187
|
+
typeof v === "number" && Number.isFinite(v) && v >= 0 ? Math.floor(v) : undefined;
|
|
188
|
+
const u = (typeof p.usage === "object" && p.usage !== null ? p.usage : {}) as Record<string, unknown>;
|
|
189
|
+
const tokens = nonneg(u.total_tokens);
|
|
190
|
+
const toolUses = nonneg(u.tool_uses);
|
|
191
|
+
const durationMs = nonneg(u.duration_ms);
|
|
192
|
+
return {
|
|
193
|
+
type: "task_notification",
|
|
194
|
+
taskId: clip(taskId, 128),
|
|
195
|
+
...(toolUseId ? { toolUseId: clip(toolUseId, 128) } : {}),
|
|
196
|
+
status: typeof p.status === "string" && p.status.length > 0 ? p.status : "finished",
|
|
197
|
+
summary: clip(summary.replace(/\s+/g, " ").trim(), NOTIFICATION_SUMMARY_MAX),
|
|
198
|
+
...(tokens !== undefined || toolUses !== undefined || durationMs !== undefined
|
|
199
|
+
? { usage: {
|
|
200
|
+
...(tokens !== undefined ? { tokens } : {}),
|
|
201
|
+
...(toolUses !== undefined ? { toolUses } : {}),
|
|
202
|
+
...(durationMs !== undefined ? { durationMs } : {}),
|
|
203
|
+
} }
|
|
204
|
+
: {}),
|
|
205
|
+
timestamp,
|
|
206
|
+
};
|
|
207
|
+
}
|
|
208
|
+
|
|
75
209
|
/** Claude Code's real reasoning knob is its own `--effort <level>` flag
|
|
76
210
|
* (low|medium|high|xhigh|max — verified against `claude -p --help`). The
|
|
77
211
|
* CliReasoningEffort union IS the CLI's vocabulary, so the level rides the
|
|
@@ -100,12 +234,30 @@ export const claudeCodeSpec: CliAgentSpec = {
|
|
|
100
234
|
'sudo cp "$REAL" /usr/local/bin/claude && sudo chmod 0755 /usr/local/bin/claude',
|
|
101
235
|
// Claude reads the prompt from stdin in -p mode.
|
|
102
236
|
promptPayload: (prompt) => prompt,
|
|
103
|
-
|
|
237
|
+
// Mid-turn stream input (steering): with `--input-format stream-json`,
|
|
238
|
+
// stdin carries JSONL user messages, and one that arrives WHILE a turn
|
|
239
|
+
// runs is folded into the running turn at the next tool boundary — the
|
|
240
|
+
// interactive UI's queued-user-input behaviour, verified live against
|
|
241
|
+
// claude 2.1.233 (a message injected during a 20s Bash call shaped the
|
|
242
|
+
// same turn's final reply, and the tool ran undisturbed). Slash-command
|
|
243
|
+
// expansion and `--resume` both work unchanged in this mode (verified on
|
|
244
|
+
// the same build). A message that lands after the result would start a
|
|
245
|
+
// NEW turn in-process, which is why the transport's feeder stops at the
|
|
246
|
+
// result line instead of forwarding past it.
|
|
247
|
+
streamInput: {
|
|
248
|
+
promptLine: (prompt) => JSON.stringify({ type: "user", message: { role: "user", content: prompt } }),
|
|
249
|
+
messageLine: (text) => JSON.stringify({ type: "user", message: { role: "user", content: text } }),
|
|
250
|
+
},
|
|
251
|
+
buildCommand: ({ promptPath, sessionId, model, cwd, effort, streamInput }) => {
|
|
104
252
|
const flags = [
|
|
105
253
|
"-p",
|
|
106
254
|
// stream-json is the JSONL event stream; -p requires --verbose with it.
|
|
107
255
|
"--output-format stream-json",
|
|
108
256
|
"--verbose",
|
|
257
|
+
// Stream-input turns feed stdin as JSONL user messages from the
|
|
258
|
+
// transport's FIFO (mid-turn injection); plain turns keep the raw
|
|
259
|
+
// prompt file.
|
|
260
|
+
...(streamInput ? ["--input-format stream-json"] : []),
|
|
109
261
|
// Raw API stream events ride along as `stream_event` lines — the
|
|
110
262
|
// text_delta source for progressive rendering (mapEvent below). The
|
|
111
263
|
// complete `assistant` message events still arrive; deltas are
|
|
@@ -157,17 +309,29 @@ export const claudeCodeSpec: CliAgentSpec = {
|
|
|
157
309
|
return [];
|
|
158
310
|
});
|
|
159
311
|
}
|
|
160
|
-
// User API message: the CLI echoes tool results back as user content
|
|
312
|
+
// User API message: the CLI echoes tool results back as user content,
|
|
313
|
+
// and injects `<task-notification>` blocks (background-task completion
|
|
314
|
+
// evidence) as user TEXT — parsed into structure, never dropped and
|
|
315
|
+
// never forwarded raw. Other user text (the echo of the prompt, system
|
|
316
|
+
// reminders) stays unmapped: it is not agent output.
|
|
161
317
|
case "user": {
|
|
162
|
-
const message = p.message as { content?: ClaudeContentBlock[] } | undefined;
|
|
318
|
+
const message = p.message as { content?: ClaudeContentBlock[] | string } | undefined;
|
|
319
|
+
if (typeof message?.content === "string") {
|
|
320
|
+
return parseTaskNotifications(message.content, ts);
|
|
321
|
+
}
|
|
163
322
|
const blocks = Array.isArray(message?.content) ? message.content : [];
|
|
164
|
-
return blocks.flatMap((b): AgentMessage[] =>
|
|
165
|
-
b.type === "tool_result"
|
|
166
|
-
|
|
167
|
-
|
|
168
|
-
|
|
169
|
-
|
|
170
|
-
|
|
323
|
+
return blocks.flatMap((b): AgentMessage[] => {
|
|
324
|
+
if (b.type === "tool_result") {
|
|
325
|
+
return [{
|
|
326
|
+
type: "tool_result", toolUseId: String(b.tool_use_id ?? ""),
|
|
327
|
+
output: toolResultText(b.content), isError: b.is_error === true, ...parent, timestamp: ts,
|
|
328
|
+
}];
|
|
329
|
+
}
|
|
330
|
+
if (b.type === "text" && typeof b.text === "string") {
|
|
331
|
+
return parseTaskNotifications(b.text, ts);
|
|
332
|
+
}
|
|
333
|
+
return [];
|
|
334
|
+
});
|
|
171
335
|
}
|
|
172
336
|
// Raw API stream event (--include-partial-messages): text deltas of
|
|
173
337
|
// the in-progress block map to the live-only `text_delta` kind so a
|
|
@@ -234,7 +398,18 @@ export const claudeCodeSpec: CliAgentSpec = {
|
|
|
234
398
|
timestamp: ts,
|
|
235
399
|
}];
|
|
236
400
|
}
|
|
237
|
-
//
|
|
401
|
+
// System events: init carries the session id (extractSessionId), and
|
|
402
|
+
// the background-task lane rides here too — `task_notification` is
|
|
403
|
+
// the completion evidence stream-json actually emits (see the section
|
|
404
|
+
// header above), so it maps to the structured message BOTH turn
|
|
405
|
+
// engines persist. Everything else under system (task_started,
|
|
406
|
+
// task_updated, background_tasks_changed, thinking_tokens) is
|
|
407
|
+
// lifecycle noise here.
|
|
408
|
+
case "system": {
|
|
409
|
+
if (p.subtype !== "task_notification") return [];
|
|
410
|
+
const notification = parseSystemTaskNotification(p, ts);
|
|
411
|
+
return notification ? [notification] : [];
|
|
412
|
+
}
|
|
238
413
|
default:
|
|
239
414
|
return [];
|
|
240
415
|
}
|
package/src/runtimes/codex.ts
CHANGED
|
@@ -53,6 +53,54 @@ export function isCodexAdvisoryNoise(text: string): boolean {
|
|
|
53
53
|
return CODEX_ADVISORY_PATTERNS.some((re) => re.test(text));
|
|
54
54
|
}
|
|
55
55
|
|
|
56
|
+
// ── Thread-store writer-lock preflight (2026-08-18 prod incident) ────────────
|
|
57
|
+
// codex guards each persisted thread with an OS advisory flock on
|
|
58
|
+
// `$CODEX_HOME/thread-writer-locks/<thread-id>.lock` (codex-rs
|
|
59
|
+
// thread-store/src/local/writer_lock.rs). Two consequences drive this
|
|
60
|
+
// preflight's shape:
|
|
61
|
+
// - the flock is held exactly as long as the holding PROCESS lives — the
|
|
62
|
+
// kernel releases it on any death (SIGKILL, VM crash), so a lock that
|
|
63
|
+
// still blocks is held by a LIVE process, never a stale file;
|
|
64
|
+
// - codex sweeps unheld (stale) lock FILES itself on store init.
|
|
65
|
+
// The incident: a canceled turn's codex survived its fire-and-forget SIGTERM
|
|
66
|
+
// long enough for the user's next turn to launch `codex exec resume`, which
|
|
67
|
+
// died with `thread-store conflict: thread … already has an active writer`
|
|
68
|
+
// (code -32600). So before a RESUME the launch script:
|
|
69
|
+
// 1. waits (bounded) for a dying previous writer to release the flock —
|
|
70
|
+
// the canceled predecessor was already TERM'd, it just needs a moment;
|
|
71
|
+
// 2. clears the lock file ONLY under a successfully acquired `flock -n`
|
|
72
|
+
// (proof the holder is dead) — belt-and-suspenders on top of codex's
|
|
73
|
+
// own sweep, and the guard against any future file-existence check;
|
|
74
|
+
// 3. NEVER touches a lock whose holder is alive: removing a live-held
|
|
75
|
+
// lock file would let a second writer flock a fresh inode (two writers
|
|
76
|
+
// on one rollout), so a still-held lock after the wait is left for
|
|
77
|
+
// codex to surface as the honest conflict it is.
|
|
78
|
+
// No `flock(1)` on the guest (or no lock file) → the preflight is a no-op.
|
|
79
|
+
|
|
80
|
+
/** Bounded wait for a dying previous writer: attempts × sleep = 5s, matching
|
|
81
|
+
* the teardown's SIGTERM grace (sdk _cli-agent.ts REAP_TERM_WAIT_ATTEMPTS). */
|
|
82
|
+
export const CODEX_LOCK_WAIT_ATTEMPTS = 20;
|
|
83
|
+
export const CODEX_LOCK_WAIT_SECONDS = "0.25";
|
|
84
|
+
|
|
85
|
+
/** Thread ids are uuid-shaped; anything else skips the preflight entirely so
|
|
86
|
+
* hostile config can never become shell injection via the lock path. */
|
|
87
|
+
const CODEX_THREAD_ID_SAFE = /^[A-Za-z0-9_-]{1,64}$/;
|
|
88
|
+
|
|
89
|
+
/** The sh fragment prepended to a resume launch. `waitAttempts` is a test
|
|
90
|
+
* seam (the live-holder test must not sleep 5s); production callers take
|
|
91
|
+
* the default. Exported for tests. */
|
|
92
|
+
export function codexWriterLockPreflight(
|
|
93
|
+
threadId: string,
|
|
94
|
+
waitAttempts: number = CODEX_LOCK_WAIT_ATTEMPTS,
|
|
95
|
+
): string {
|
|
96
|
+
if (!CODEX_THREAD_ID_SAFE.test(threadId)) return "";
|
|
97
|
+
return `ac_lk="\${CODEX_HOME:-$HOME/.codex}/thread-writer-locks/${threadId}.lock"; `
|
|
98
|
+
+ `if [ -e "$ac_lk" ] && command -v flock >/dev/null 2>&1; then ac_lki=0; `
|
|
99
|
+
+ `while ! flock -n "$ac_lk" true 2>/dev/null && [ "$ac_lki" -lt ${waitAttempts} ]; do `
|
|
100
|
+
+ `sleep ${CODEX_LOCK_WAIT_SECONDS}; ac_lki=$((ac_lki+1)); done; `
|
|
101
|
+
+ `flock -n "$ac_lk" rm -f -- "$ac_lk" 2>/dev/null || true; fi; `;
|
|
102
|
+
}
|
|
103
|
+
|
|
56
104
|
/** Exported for the fixture-based parity tests (ADR-0020): the legacy JSONL
|
|
57
105
|
* `mapEvent` is the golden the ACP normaliser is asserted equal to. Not part
|
|
58
106
|
* of the public runtime surface — `createCodexRuntime` stays the entry point. */
|
|
@@ -60,6 +108,11 @@ export const codexSpec: CliAgentSpec = {
|
|
|
60
108
|
kind: "codex",
|
|
61
109
|
authEnv: "CODEX_API_KEY",
|
|
62
110
|
bin: "codex",
|
|
111
|
+
// codex holds a per-thread advisory flock for its writer's whole lifetime
|
|
112
|
+
// (see the writer-lock preflight above): a live predecessor process blocks
|
|
113
|
+
// every resume of the same thread, so supersede teardown must KILL it —
|
|
114
|
+
// never detach it alive (server runner-kill.ts consumes this flag).
|
|
115
|
+
exclusiveSessionWriter: true,
|
|
63
116
|
// ACP-mode invocation (ADR-0020 increment 1). `codex` has no native `--acp`
|
|
64
117
|
// flag; the adapter (a Rust binary shipped via npm) speaks ACP and drives a
|
|
65
118
|
// compatible bundled `@openai/codex`. The runner attempts this first and
|
|
@@ -95,14 +148,24 @@ export const codexSpec: CliAgentSpec = {
|
|
|
95
148
|
// backstop for a caller that bypasses that gate. The value comes from
|
|
96
149
|
// the closed CliReasoningEffort set, so it is shell-safe unquoted.
|
|
97
150
|
...(effort ? ["-c", `model_reasoning_effort=${effort === "max" ? "xhigh" : effort}`] : []),
|
|
98
|
-
...(cwd ? ["-C", shellQuote(cwd)] : []),
|
|
99
151
|
].join(" ");
|
|
152
|
+
// Working directory via a shell `cd` — the idiom EVERY other runtime uses
|
|
153
|
+
// (claude-code/cursor/droid/opencode) — NOT codex's `-C` flag. `-C` is
|
|
154
|
+
// accepted by `codex exec` but REJECTED by `codex exec resume`
|
|
155
|
+
// ("unexpected argument '-C'"), so a fresh turn worked and every follow-up
|
|
156
|
+
// died. `cd` sets the cwd identically for both paths; promptPath is
|
|
157
|
+
// absolute (/tmp/…), so the `< prompt` redirect survives the cd.
|
|
158
|
+
const cd = cwd ? `cd ${shellQuote(cwd)} && ` : "";
|
|
159
|
+
// Resume-only launch preflight: self-heal a released-or-dying writer
|
|
160
|
+
// lock before `codex exec resume` (see codexWriterLockPreflight). A
|
|
161
|
+
// fresh turn mints a new thread id — no lock can exist for it yet.
|
|
162
|
+
const preflight = sessionId ? codexWriterLockPreflight(sessionId) : "";
|
|
100
163
|
// Fresh turn: `codex exec <flags> - < prompt`. Continue a thread:
|
|
101
164
|
// `codex exec resume <id> <flags> - < prompt`. (`-` = read prompt from stdin.)
|
|
102
165
|
const exec = sessionId
|
|
103
166
|
? `codex exec resume ${shellQuote(sessionId)} ${flags}`
|
|
104
167
|
: `codex exec ${flags}`;
|
|
105
|
-
return `${exec} - < ${shellQuote(promptPath)}`;
|
|
168
|
+
return `${preflight}${cd}${exec} - < ${shellQuote(promptPath)}`;
|
|
106
169
|
},
|
|
107
170
|
extractSessionId: (p) =>
|
|
108
171
|
p.type === "thread.started" && typeof p.thread_id === "string" ? p.thread_id : undefined,
|
|
@@ -237,11 +237,15 @@ function makeE2bSandboxProvider(sb: Sandbox): SandboxProvider {
|
|
|
237
237
|
return { resumeHandle: sb.sandboxId };
|
|
238
238
|
},
|
|
239
239
|
// The provider kill-clock seam: e2b's `setTimeout` REPLACES the deadline
|
|
240
|
-
// (extend or reduce) relative to now. The server's
|
|
241
|
-
//
|
|
242
|
-
//
|
|
240
|
+
// (extend or reduce) relative to now, in MILLISECONDS. The server's
|
|
241
|
+
// session lifecycle uses it two ways: a FAR-HORIZON push while the
|
|
242
|
+
// session is active (the kill-clock is an orphan backstop, never
|
|
243
|
+
// load-bearing during a live turn) and a short re-arm when the deferred
|
|
244
|
+
// pause is about to park the VM. Clamped to the plan cap exactly like
|
|
245
|
+
// the create timeout — an over-cap push would 400 and leave the OLD
|
|
246
|
+
// (possibly short) deadline standing.
|
|
243
247
|
async extendLifetime(ms) {
|
|
244
|
-
await sb.setTimeout(ms);
|
|
248
|
+
await sb.setTimeout(Math.min(ms, e2bMaxSandboxMs()));
|
|
245
249
|
},
|
|
246
250
|
// Push a freshly-resolved egress policy onto the live sandbox via E2B's
|
|
247
251
|
// native `updateNetwork` — the E2B analogue of Vercel's `update({
|
|
@@ -5,6 +5,16 @@
|
|
|
5
5
|
import { promises as fs } from "node:fs";
|
|
6
6
|
import { dirname } from "node:path";
|
|
7
7
|
import { spawn } from "node:child_process";
|
|
8
|
+
/** The two listener registrations we need, declared LOCALLY.
|
|
9
|
+
* `ChildProcess` inherits `.on` from EventEmitter, but when a tree ends up
|
|
10
|
+
* with more than one @types/node the inheritance link breaks and `.on`
|
|
11
|
+
* disappears — which resolution you get depends on install layout, so this
|
|
12
|
+
* typechecked locally and failed CI three times. Referencing no node types
|
|
13
|
+
* at all is the only version-proof shape. */
|
|
14
|
+
type ProcListeners = {
|
|
15
|
+
on(event: "error", cb: (err: Error) => void): unknown;
|
|
16
|
+
on(event: "close", cb: (code: number | null) => void): unknown;
|
|
17
|
+
};
|
|
8
18
|
import { Readable, Writable } from "node:stream";
|
|
9
19
|
import type { SandboxProvider } from "../../types/sandbox.js";
|
|
10
20
|
|
|
@@ -36,8 +46,9 @@ export function makeLocalSandboxProvider(): SandboxProvider {
|
|
|
36
46
|
proc.stderr?.setEncoding("utf8");
|
|
37
47
|
proc.stdout?.on("data", (chunk: string) => { stdout += chunk; opts?.onStdout?.(chunk); });
|
|
38
48
|
proc.stderr?.on("data", (chunk: string) => { stderr += chunk; opts?.onStderr?.(chunk); });
|
|
39
|
-
proc
|
|
40
|
-
|
|
49
|
+
const ev = proc as unknown as ProcListeners;
|
|
50
|
+
ev.on("error", reject);
|
|
51
|
+
ev.on("close", (code: number | null) => resolve({ exitCode: code ?? 0, stdout, stderr }));
|
|
41
52
|
});
|
|
42
53
|
},
|
|
43
54
|
// Duplex spawn — the in-VM `child_process` pipe the ACP client needs.
|
|
@@ -67,8 +78,9 @@ export function makeLocalSandboxProvider(): SandboxProvider {
|
|
|
67
78
|
proc.stderr.on("data", (chunk: string) => { stderr += chunk; });
|
|
68
79
|
|
|
69
80
|
const exited = new Promise<{ exitCode: number; stderr: string }>((resolve, reject) => {
|
|
70
|
-
proc
|
|
71
|
-
|
|
81
|
+
const ev = proc as unknown as ProcListeners;
|
|
82
|
+
ev.on("error", reject);
|
|
83
|
+
ev.on("close", (code: number | null) => resolve({ exitCode: code ?? 0, stderr }));
|
|
72
84
|
});
|
|
73
85
|
|
|
74
86
|
return {
|
package/src/sandbox/sizes.ts
CHANGED
|
@@ -1,27 +1,68 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* Sandbox machine sizes
|
|
2
|
+
* Sandbox machine sizes — THE single source of the size vocabulary.
|
|
3
3
|
*
|
|
4
|
-
*
|
|
5
|
-
*
|
|
6
|
-
*
|
|
4
|
+
* Everything that names a size (the server's `SANDBOX_DEFAULT_SIZE` env enum,
|
|
5
|
+
* the register/invoke zod schemas, the run + session row types, the CLI's
|
|
6
|
+
* `--size` flag, the dashboard pickers via `GET /v1/sandbox-sizes`) derives
|
|
7
|
+
* from `SANDBOX_SIZES` / `SandboxSize` here. Adding a size is a ONE-LINE edit
|
|
8
|
+
* to `SANDBOX_MACHINES`: the type, the enums, the E2B build matrix and the
|
|
9
|
+
* pickers all follow. Do NOT re-declare the union inline anywhere.
|
|
10
|
+
*
|
|
11
|
+
* A size is a coarse hardware knob that maps to provider machine specs:
|
|
12
|
+
* Vercel honours it natively via `resources.vcpus`; E2B sizing is BAKED INTO
|
|
13
|
+
* THE TEMPLATE (e2b 2.30.5 has no create-time cpu/mem knob — `NewSandbox`
|
|
14
|
+
* carries only `templateID`), so on E2B a size resolves to a pre-built
|
|
15
|
+
* per-size template.
|
|
7
16
|
*/
|
|
8
17
|
|
|
9
|
-
/**
|
|
10
|
-
* than abstract t-shirt sizes
|
|
11
|
-
*
|
|
12
|
-
*
|
|
13
|
-
*
|
|
14
|
-
*
|
|
15
|
-
*
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
|
|
18
|
+
/** Every sandbox hardware SKU, with its real machine spec. Named for the
|
|
19
|
+
* machine (vCPU + RAM) rather than abstract t-shirt sizes, so a size can
|
|
20
|
+
* never quietly mean something different than it says.
|
|
21
|
+
*
|
|
22
|
+
* RAM is 2048 MB/vCPU everywhere EXCEPT `8vcpu-8gb`, which exists because
|
|
23
|
+
* E2B caps a sandbox at 8 vCPU / 8192 MB (e2b.dev/docs/billing: Hobby and
|
|
24
|
+
* Pro both "8 vCPU / 8 GB", raised only by arrangement): 8 vCPU at the 2 GB
|
|
25
|
+
* rule would need 16 GiB and cannot be built. `8vcpu-8gb` is the CPU ceiling
|
|
26
|
+
* at the memory ceiling — the only way to get 8 cores on E2B today, and a
|
|
27
|
+
* spec already proven bakeable by the devbox (`E2B_DEVBOX_SPEC`).
|
|
28
|
+
*
|
|
29
|
+
* Adding an entry here automatically: widens `SandboxSize`, widens every
|
|
30
|
+
* derived enum, and — if it fits under the E2B caps — adds it to
|
|
31
|
+
* `E2B_TEMPLATE_SIZES`, which is what `infra/e2b-template/build.ts` and the
|
|
32
|
+
* `sandbox-images` CI job loop over. Two templates get baked per size, so
|
|
33
|
+
* the matrix is not free; see that workflow's header. */
|
|
34
|
+
export const SANDBOX_MACHINES = {
|
|
35
|
+
/** 1 vCPU / 2 GiB — the cheap floor. Plenty for a terminal session or a
|
|
36
|
+
* shell-shaped agent; tight for a big `bun install` or a browser. */
|
|
37
|
+
"1vcpu-2gb": { vcpus: 1, memoryMB: 2048 },
|
|
38
|
+
/** 2 vCPU / 4 GiB — the default (Vercel's own default machine too). */
|
|
39
|
+
"2vcpu-4gb": { vcpus: 2, memoryMB: 4096 },
|
|
40
|
+
/** 4 vCPU / 8 GiB — comfortable for builds and multi-tool agent turns. */
|
|
41
|
+
"4vcpu-8gb": { vcpus: 4, memoryMB: 8192 },
|
|
42
|
+
/** 8 vCPU / 8 GiB — E2B's per-sandbox CEILING (cores maxed at the memory
|
|
43
|
+
* cap). NOT expressible on Vercel, whose RAM follows vCPUs at 2 GB each. */
|
|
44
|
+
"8vcpu-8gb": { vcpus: 8, memoryMB: 8192 },
|
|
45
|
+
/** 8 vCPU / 16 GiB — Vercel only; exceeds E2B's 8 GiB memory cap. */
|
|
46
|
+
"8vcpu-16gb": { vcpus: 8, memoryMB: 16384 },
|
|
47
|
+
/** 32 vCPU / 64 GiB — Vercel Enterprise only; far past every E2B cap. */
|
|
48
|
+
"32vcpu-64gb": { vcpus: 32, memoryMB: 65536 },
|
|
49
|
+
} as const satisfies Record<string, { vcpus: number; memoryMB: number }>;
|
|
50
|
+
|
|
51
|
+
/** Sandbox hardware SKU. Derived from `SANDBOX_MACHINES` — never re-spelled
|
|
52
|
+
* as an inline union. */
|
|
53
|
+
export type SandboxSize = keyof typeof SANDBOX_MACHINES;
|
|
54
|
+
|
|
55
|
+
/** The vocabulary as an ordered, smallest-first array — the shape zod
|
|
56
|
+
* (`z.enum`), the CLI's `--size` validation, and the wire catalogue want.
|
|
57
|
+
* Ordering is the pickers' display order, so keep it ascending. */
|
|
58
|
+
export const SANDBOX_SIZES = Object.keys(SANDBOX_MACHINES) as readonly SandboxSize[] as
|
|
59
|
+
readonly [SandboxSize, ...SandboxSize[]];
|
|
60
|
+
|
|
61
|
+
/** SKU → Vercel vCPU count (Vercel's RAM follows automatically at 2048
|
|
62
|
+
* MB/vCPU — which is why `isVercelSupportedSize` exists). */
|
|
63
|
+
export const SANDBOX_VCPUS: Record<SandboxSize, number> = Object.fromEntries(
|
|
64
|
+
SANDBOX_SIZES.map((s) => [s, SANDBOX_MACHINES[s].vcpus]),
|
|
65
|
+
) as Record<SandboxSize, number>;
|
|
25
66
|
|
|
26
67
|
/** SDK fallback size when neither the caller nor the deployment specifies one.
|
|
27
68
|
* Deliberately conservative — the OPERATIONAL default is the server's
|
|
@@ -29,37 +70,79 @@ export const SANDBOX_VCPUS: Record<SandboxSize, number> = {
|
|
|
29
70
|
* small matters because Vercel rate-limits creation by vCPUs-per-window
|
|
30
71
|
* (`api-sandboxes-vcpus-creation`); a large default 429s bursty/simultaneous
|
|
31
72
|
* creates. Workloads that need more RAM/CPU declare `resources.size` on the
|
|
32
|
-
* workflow rather than inflating the default for everyone.
|
|
73
|
+
* workflow rather than inflating the default for everyone.
|
|
74
|
+
*
|
|
75
|
+
* NOT `1vcpu-2gb`: the floor is an opt-IN for cheap sessions, not a quiet
|
|
76
|
+
* downgrade of every existing run's machine. */
|
|
33
77
|
export const DEFAULT_SANDBOX_SIZE: SandboxSize = "2vcpu-4gb";
|
|
34
78
|
|
|
35
|
-
/**
|
|
36
|
-
* (
|
|
37
|
-
*
|
|
38
|
-
*
|
|
39
|
-
*
|
|
40
|
-
*
|
|
41
|
-
*
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
"
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
*
|
|
50
|
-
*
|
|
51
|
-
*
|
|
79
|
+
/** SESSION default — deliberately one size up from the run default
|
|
80
|
+
* (2026-08-13): a session's sandbox carries the full desktop toolbelt
|
|
81
|
+
* (VS Code + Chromium + dockerd) plus the KasmVNC encoder at the 60fps
|
|
82
|
+
* cap, and that stack swap-thrashes on 4 GiB while the encoder starves on
|
|
83
|
+
* 2 shared vCPUs. Workflow runs keep DEFAULT_SANDBOX_SIZE — no desktop,
|
|
84
|
+
* no toolbelt weight. Sessions bill active time only (parked = storage),
|
|
85
|
+
* so the delta applies to active hours, not the fleet. */
|
|
86
|
+
export const SESSION_DEFAULT_SANDBOX_SIZE: SandboxSize = "4vcpu-8gb";
|
|
87
|
+
|
|
88
|
+
/** E2B's per-sandbox ceiling on the plans we run (e2b.dev/docs/billing —
|
|
89
|
+
* Hobby: "8 vCPU / 8 GB"; Pro: the same, "8+" only by arrangement with
|
|
90
|
+
* support). Recorded live too: `Template.build` 400s with "Memory can't be
|
|
91
|
+
* higher than 8192 MiB" past the memory cap.
|
|
92
|
+
*
|
|
93
|
+
* These two numbers are the ONLY knob for which sizes get an E2B template —
|
|
94
|
+
* raise them after E2B raises the account limit and the build matrix (and
|
|
95
|
+
* therefore the session picker) widens on its own. */
|
|
96
|
+
export const E2B_MAX_VCPUS = 8;
|
|
97
|
+
export const E2B_MAX_MEMORY_MB = 8192;
|
|
98
|
+
|
|
99
|
+
/** The E2B sizes we pre-build a template for — DERIVED from the caps, not
|
|
100
|
+
* hand-listed, so a new `SANDBOX_MACHINES` entry can never be offered
|
|
101
|
+
* without a template or omitted despite fitting. E2B sizing is
|
|
102
|
+
* template-baked (no per-create cpu/mem knob), so honouring `resources.size`
|
|
103
|
+
* on E2B means ONE pre-built template per size; `infra/e2b-template/build.ts`
|
|
104
|
+
* loops exactly this list. The register / invoke / session-spawn / resize
|
|
105
|
+
* guards all reject an unsupported E2B size before it can reach a create. */
|
|
106
|
+
export const E2B_TEMPLATE_SIZES: readonly SandboxSize[] = SANDBOX_SIZES.filter(
|
|
107
|
+
(s) => SANDBOX_MACHINES[s].vcpus <= E2B_MAX_VCPUS
|
|
108
|
+
&& SANDBOX_MACHINES[s].memoryMB <= E2B_MAX_MEMORY_MB,
|
|
109
|
+
);
|
|
110
|
+
|
|
111
|
+
/** Is `size` one E2B can be built/booted at? False for the sizes past E2B's
|
|
112
|
+
* 8 vCPU / 8 GiB ceiling (`8vcpu-16gb`, `32vcpu-64gb`) — those run on Vercel.
|
|
113
|
+
* The "no E2B equivalent" decision lives in exactly one place: the caps. */
|
|
52
114
|
export function isE2bSupportedSize(size: SandboxSize): boolean {
|
|
53
115
|
return E2B_TEMPLATE_SIZES.includes(size);
|
|
54
116
|
}
|
|
55
117
|
|
|
56
|
-
/**
|
|
57
|
-
*
|
|
58
|
-
|
|
59
|
-
|
|
118
|
+
/** Vercel's fixed memory-per-vCPU ratio. Vercel takes `resources.vcpus` and
|
|
119
|
+
* allocates RAM itself at this rate — there is no independent memory knob. */
|
|
120
|
+
export const VERCEL_MEMORY_MB_PER_VCPU = 2048;
|
|
121
|
+
|
|
122
|
+
/** Is `size` expressible on Vercel? Only when its RAM matches what Vercel
|
|
123
|
+
* would allocate for that vCPU count — otherwise asking for it would hand
|
|
124
|
+
* the caller a machine that does not match the name (`8vcpu-8gb` would come
|
|
125
|
+
* back with 16 GiB). Vercel's own ceiling (32 vCPU, Enterprise) is a plan
|
|
126
|
+
* matter, not a shape matter, so it is not encoded here. */
|
|
127
|
+
export function isVercelSupportedSize(size: SandboxSize): boolean {
|
|
128
|
+
const m = SANDBOX_MACHINES[size];
|
|
129
|
+
return m.memoryMB === m.vcpus * VERCEL_MEMORY_MB_PER_VCPU;
|
|
130
|
+
}
|
|
131
|
+
|
|
132
|
+
/** Machine spec for a SandboxSize, in the shape `Template.build` wants. Used
|
|
133
|
+
* by `infra/e2b-template/build.ts` to stamp the per-size base + agent-env
|
|
134
|
+
* templates. Reads the explicit table rather than deriving RAM from vCPUs —
|
|
135
|
+
* `8vcpu-8gb` is deliberately off the 2048 MB/vCPU line. */
|
|
60
136
|
export function e2bMachineSpec(size: SandboxSize): { cpuCount: number; memoryMB: number } {
|
|
61
|
-
const
|
|
62
|
-
return { cpuCount, memoryMB:
|
|
137
|
+
const m = SANDBOX_MACHINES[size];
|
|
138
|
+
return { cpuCount: m.vcpus, memoryMB: m.memoryMB };
|
|
139
|
+
}
|
|
140
|
+
|
|
141
|
+
/** Human label for a size — "2 vCPU · 4 GB". The wire catalogue carries it so
|
|
142
|
+
* the dashboard never has to parse the id back into numbers. */
|
|
143
|
+
export function sandboxSizeLabel(size: SandboxSize): string {
|
|
144
|
+
const m = SANDBOX_MACHINES[size];
|
|
145
|
+
return `${m.vcpus} vCPU · ${Math.round(m.memoryMB / 1024)} GB`;
|
|
63
146
|
}
|
|
64
147
|
|
|
65
148
|
/** Stable E2B template ALIAS for the platform base at a given size
|
package/src/sandbox.ts
CHANGED
|
@@ -23,10 +23,18 @@ export { DOT_SEGMENT_PATH_RE2, toVercelNetworkPolicy, toE2bNetwork } from "./san
|
|
|
23
23
|
|
|
24
24
|
export type { SandboxSize } from "./sandbox/sizes.js";
|
|
25
25
|
export {
|
|
26
|
+
SANDBOX_SIZES,
|
|
27
|
+
SANDBOX_MACHINES,
|
|
26
28
|
SANDBOX_VCPUS,
|
|
27
29
|
DEFAULT_SANDBOX_SIZE,
|
|
30
|
+
SESSION_DEFAULT_SANDBOX_SIZE,
|
|
28
31
|
E2B_TEMPLATE_SIZES,
|
|
32
|
+
E2B_MAX_VCPUS,
|
|
33
|
+
E2B_MAX_MEMORY_MB,
|
|
34
|
+
VERCEL_MEMORY_MB_PER_VCPU,
|
|
29
35
|
isE2bSupportedSize,
|
|
36
|
+
isVercelSupportedSize,
|
|
37
|
+
sandboxSizeLabel,
|
|
30
38
|
e2bMachineSpec,
|
|
31
39
|
e2bBaseTemplate,
|
|
32
40
|
e2bAgentEnvTemplate,
|