@agent-compose/sdk 0.2.1 → 0.2.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,305 @@
1
+ /**
2
+ * Claude CLI runtime — drives agents via the Claude CLI inside a sandbox.
3
+ *
4
+ * Works with any SandboxProvider (E2B, Vercel, etc.) — the provider is
5
+ * selected via SANDBOX_PROVIDER env var, not the runtime definition.
6
+ *
7
+ * Usage:
8
+ * import { createClaudeRuntime } from "@agent-compose/sdk";
9
+ * export default createClaudeRuntime({ claudeMdContent: "..." }); // optional global instructions
10
+ */
11
+
12
+ import type { SandboxProvider, AgentMessage, ModelExecutionContract, RuntimeOptions } from "../index.js";
13
+ import { defineRuntime } from "../types/runtime.js";
14
+ import { DEFAULT_CLAUDE_MODEL } from "../agent/agent-loop.js";
15
+ import { formatError } from "../utils/errors.js";
16
+
17
+ const NO_TIMEOUT = 0; // claude exits on its own via --max-turns
18
+
19
+
20
+ function translateEvent(event: Record<string, unknown>): AgentMessage[] {
21
+ const msgs: AgentMessage[] = [];
22
+ const ts = new Date().toISOString();
23
+
24
+ if (event.type === "system" && event.subtype === "init") {
25
+ msgs.push({ type: "init", sessionId: String(event.session_id ?? ""), timestamp: ts });
26
+ return msgs;
27
+ }
28
+
29
+ if (event.type === "assistant") {
30
+ const message = event.message as { content?: unknown[] } | undefined;
31
+ for (const block of message?.content ?? []) {
32
+ const b = block as Record<string, unknown>;
33
+ if (b.type === "text") msgs.push({ type: "text", text: String(b.text ?? ""), timestamp: ts });
34
+ if (b.type === "thinking") msgs.push({ type: "thinking", text: String(b.thinking ?? ""), timestamp: ts });
35
+ if (b.type === "tool_use") msgs.push({ type: "tool_use", toolName: String(b.name ?? ""), toolInput: (b.input ?? {}) as Record<string, unknown>, toolUseId: String(b.id ?? ""), timestamp: ts });
36
+ }
37
+ return msgs;
38
+ }
39
+
40
+ if (event.type === "user") {
41
+ const message = event.message as { content?: unknown[] } | undefined;
42
+ for (const block of message?.content ?? []) {
43
+ const b = block as Record<string, unknown>;
44
+ if (b.type === "tool_result") {
45
+ const raw = b.content;
46
+ msgs.push({ type: "tool_result", toolUseId: String(b.tool_use_id ?? ""), output: typeof raw === "string" ? raw : JSON.stringify(raw ?? ""), isError: Boolean(b.is_error), timestamp: ts });
47
+ }
48
+ }
49
+ return msgs;
50
+ }
51
+
52
+ if (event.type === "result") {
53
+ const usage = event.usage as Record<string, number> | undefined;
54
+ if (usage) {
55
+ msgs.push({ type: "usage", inputTokens: usage.input_tokens ?? 0, outputTokens: usage.output_tokens ?? 0, cacheReadTokens: usage.cache_read_input_tokens ?? 0, cacheCreationTokens: usage.cache_creation_input_tokens ?? 0, durationMs: Number(event.duration_ms ?? 0), numTurns: Number(event.num_turns ?? 0), timestamp: ts });
56
+ }
57
+ if (event.is_error || event.subtype === "error_during_execution") {
58
+ const raw = event.error ?? event.result ?? event.message;
59
+ msgs.push({ type: "error", text: raw == null ? "Unknown error" : typeof raw === "object" ? JSON.stringify(raw) : String(raw), timestamp: ts });
60
+ } else {
61
+ msgs.push({ type: "done", sessionId: String(event.session_id ?? ""), timestamp: ts });
62
+ }
63
+ return msgs;
64
+ }
65
+
66
+ return msgs;
67
+ }
68
+
69
+ export class ClaudeRunner implements ModelExecutionContract {
70
+ private claudeMdWritten = false;
71
+ private homeDir: string | null = null;
72
+ private label: string;
73
+
74
+ constructor(
75
+ private sandbox: SandboxProvider,
76
+ private options: RuntimeOptions = {},
77
+ private claudeMdContent: string = "",
78
+ private configEnv: Record<string, string> = {},
79
+ private configModel?: string,
80
+ private configMcpServers?: Record<string, { command: string; args?: string[]; env?: Record<string, string> }>,
81
+ ) {
82
+ this.label = options.label ?? "[Agent]";
83
+ }
84
+
85
+ private async getHomeDir(): Promise<string> {
86
+ if (!this.homeDir) {
87
+ const { stdout } = await this.sandbox.commands.run("echo $HOME", { timeoutMs: 30_000 });
88
+ this.homeDir = stdout.trim() || "/home/user";
89
+ }
90
+ return this.homeDir;
91
+ }
92
+
93
+ private async deployGlobalConfig(): Promise<void> {
94
+ if (this.claudeMdWritten) return;
95
+ const home = await this.getHomeDir();
96
+
97
+ await this.sandbox.commands.run(`mkdir -p "${home}/.claude" && find "${home}/.claude" -name "*.json" -not -name "CLAUDE.md" -delete 2>/dev/null || true`, { timeoutMs: 5000 });
98
+ await this.sandbox.files.write(`${home}/.claude/CLAUDE.md`, this.claudeMdContent);
99
+ // Write env overrides to settings.json so the CLI picks them up at startup.
100
+ // This is how the CLI reads ANTHROPIC_BASE_URL, ANTHROPIC_AUTH_TOKEN, etc.
101
+ if (Object.keys(this.configEnv).length) {
102
+ const settings = JSON.stringify({ env: this.configEnv });
103
+ const b64 = Buffer.from(settings).toString("base64");
104
+ await this.sandbox.commands.run(`echo "${b64}" | base64 -d > "${home}/.claude/settings.json"`, { timeoutMs: 5_000 });
105
+ }
106
+ // MCP servers: config-level (from createClaudeRuntime) as base, opts-level (from agent definition) overrides.
107
+ const mcpServers = this.configMcpServers;
108
+ if (mcpServers && Object.keys(mcpServers).length) {
109
+ // MCP servers live in ~/.claude.json per Claude Code docs (not ~/.claude/settings.json).
110
+ // Use a shell command rather than files.write — Vercel's writeFiles API defaults paths to
111
+ // /vercel/sandbox, so absolute paths in the home directory may not resolve correctly.
112
+ const b64 = Buffer.from(JSON.stringify({ mcpServers })).toString("base64");
113
+ await this.sandbox.commands.run(`echo "${b64}" | base64 -d > "${home}/.claude.json"`, { timeoutMs: 5_000 });
114
+ process.stderr.write(`${this.label}[mcp] wrote ${Object.keys(mcpServers).join(",")} to ~/.claude.json\n`);
115
+ }
116
+ // Configure git in one shot: disable credential prompts, clear credential helper,
117
+ // and set a placeholder Authorization header for github.com HTTPS requests.
118
+ // The sandbox provider's network policy firewall replaces this header with the
119
+ // real token before forwarding — the token is never readable inside the VM.
120
+ // (api.github.com is the REST API, not git; git only uses github.com URLs.)
121
+ const placeholderB64 = Buffer.from("x-access-token:placeholder").toString("base64");
122
+ await this.sandbox.commands.run(
123
+ [
124
+ `git config --global core.askPass ""`,
125
+ `git config --global credential.interactive false`,
126
+ `git config --global credential.helper ""`,
127
+ `git config --global "http.https://github.com/.extraHeader" "Authorization: Basic ${placeholderB64}"`,
128
+ ].join(" && "),
129
+ { timeoutMs: 5_000 },
130
+ ).catch((e: unknown) => console.warn(`${this.label}[init] git config failed (non-fatal): ${formatError(e)}`));
131
+
132
+ try {
133
+ await this.sandbox.commands.run("rtk init --global --hook-only --auto-patch", { timeoutMs: 10_000 });
134
+ } catch {
135
+ console.warn("[Transport] rtk init failed (non-fatal) — token compression disabled");
136
+ }
137
+ this.claudeMdWritten = true;
138
+ }
139
+
140
+ /** Stream one CLI invocation, yielding AgentMessages. Resolves when the process exits. */
141
+ private async *_runCli(cmd: string, label: string, signal?: AbortSignal): AsyncGenerator<AgentMessage> {
142
+ type QueueItem = AgentMessage | { sentinel: "error"; text: string } | null;
143
+ const queue: QueueItem[] = [];
144
+ let notify: (() => void) | null = null;
145
+
146
+ function enqueue(items: AgentMessage[]): void { queue.push(...items); notify?.(); notify = null; }
147
+
148
+ let lineBuffer = "";
149
+ let stderrBuffer = "";
150
+
151
+ const maxTurns = this.options.maxTurns ?? 40;
152
+
153
+ function handleStdout(data: string): void {
154
+ lineBuffer += data;
155
+ const lines = lineBuffer.split("\n");
156
+ lineBuffer = lines.pop() ?? "";
157
+ for (const line of lines) {
158
+ const trimmed = line.trim();
159
+ if (!trimmed) continue;
160
+ try {
161
+ const event = JSON.parse(trimmed) as Record<string, unknown>;
162
+ if (event.type === "assistant") {
163
+ const content = (event.message as { content?: unknown[] })?.content ?? [];
164
+ for (const block of content) {
165
+ const b = block as Record<string, unknown>;
166
+ if (b.type === "text") process.stdout.write(".");
167
+ if (b.type === "tool_use") process.stderr.write(`\n${label} ${b.name}(${JSON.stringify(b.input).slice(0, 80)})`);
168
+ }
169
+ } else if (event.type === "result") {
170
+ process.stderr.write(`\n${label}[result:${event.num_turns}/${maxTurns}/${event.subtype}]\n`);
171
+ } else if (event.type === "system") {
172
+ const ev = event as Record<string,unknown>;
173
+ const mcpSrvs = (ev.mcp_servers as {name:string;status:string}[] | undefined) ?? [];
174
+ const mcpStr = mcpSrvs.length ? ` mcp=[${mcpSrvs.map(s=>`${s.name}:${s.status}`).join(",")}]` : " mcp=[]";
175
+ process.stderr.write(`\n${label}[init:${ev.session_id}]${mcpStr}\n`);
176
+ }
177
+ enqueue(translateEvent(event));
178
+ } catch { /* non-JSON line */ }
179
+ }
180
+ }
181
+
182
+ process.stderr.write(`${label}[cwd:${this.options.cwd ?? "(none)"}]\n`);
183
+ const runPromise = this.sandbox.commands.run(cmd, {
184
+ cwd: this.options.cwd,
185
+ timeoutMs: NO_TIMEOUT,
186
+ onStdout: handleStdout,
187
+ onStderr: (data: string) => { stderrBuffer += data; process.stderr.write(`${label}[stderr] ${data.trimEnd()}\n`); },
188
+ });
189
+
190
+ runPromise
191
+ .catch((err: unknown) => {
192
+ const e = err as Record<string, unknown>;
193
+ if (e && typeof e === "object" && "exitCode" in e) process.stderr.write(`${label}[transport] command failed — exitCode=${e.exitCode} error=${JSON.stringify(e.error)} stdout_tail=${String(e.stdout ?? "").slice(-200)} stderr_tail=${String(e.stderr ?? "").slice(-200)}\n`);
194
+ queue.push({ sentinel: "error", text: stderrBuffer.trim() ? `Agent process failed: ${formatError(err)}\n\n${stderrBuffer.trim()}` : `Agent process failed: ${formatError(err)}` });
195
+ })
196
+ .finally(() => {
197
+ if (lineBuffer.trim()) { try { enqueue(translateEvent(JSON.parse(lineBuffer.trim()) as Record<string, unknown>)); } catch { /* ignore */ } }
198
+ queue.push(null);
199
+ notify?.(); notify = null;
200
+ });
201
+
202
+ while (true) {
203
+ while (queue.length > 0) {
204
+ const item = queue.shift()!;
205
+ if (item === null) return;
206
+ if ("sentinel" in item) { yield { type: "error", text: item.text, timestamp: new Date().toISOString() }; return; }
207
+ if (signal?.aborted) return;
208
+ const msg = item as AgentMessage;
209
+ yield (msg.type === "error" && stderrBuffer.trim() && !msg.text.includes(stderrBuffer.trim()))
210
+ ? { ...msg, text: `${msg.text}\n\n${stderrBuffer.trim()}` }
211
+ : msg;
212
+ }
213
+ await new Promise<void>(r => { notify = r; });
214
+ }
215
+ }
216
+
217
+ async *sendMessage(opts: { prompt: string; sessionId?: string; signal?: AbortSignal }): AsyncGenerator<AgentMessage> {
218
+ await this.deployGlobalConfig();
219
+
220
+ if (opts.sessionId) {
221
+ await this.sandbox.commands.run(`pkill -x claude 2>/dev/null || true`, { timeoutMs: 10_000 });
222
+ }
223
+
224
+ const maxTurns = this.options.maxTurns ?? 40;
225
+ const model = this.configModel ?? this.options.model ?? DEFAULT_CLAUDE_MODEL;
226
+ const label = this.label;
227
+
228
+ const promptFile = `/tmp/agent-prompt-${Date.now()}.txt`;
229
+ await this.sandbox.files.write(promptFile, opts.prompt);
230
+
231
+ // Don't restrict tools when MCP servers are configured — --allowedTools also
232
+ // blocks MCP tools, which would prevent the agent from using them.
233
+ const hasMcp = this.configMcpServers && Object.keys(this.configMcpServers).length > 0;
234
+ const allowedToolsFlag = (!hasMcp && this.options.allowedTools?.length)
235
+ ? ["--allowedTools", this.options.allowedTools.join(",")]
236
+ : [];
237
+
238
+ const buildCmd = (sessionId?: string) => [
239
+ "claude", "--dangerously-skip-permissions", "--verbose",
240
+ "--output-format", "stream-json",
241
+ "--max-turns", String(maxTurns),
242
+ "--model", model,
243
+ ...allowedToolsFlag,
244
+ ...(sessionId ? ["--resume", sessionId] : []),
245
+ "-p", `"$(cat ${promptFile})"`,
246
+ ].join(" ");
247
+
248
+ const MAX_529_RETRIES = 5;
249
+ const BACKOFF_MS = [60_000, 120_000, 240_000, 480_000, 600_000];
250
+ let resumeSessionId = opts.sessionId;
251
+
252
+ for (let attempt = 0; attempt <= MAX_529_RETRIES; attempt++) {
253
+ let got529 = false;
254
+ let newSession = "";
255
+
256
+ for await (const msg of this._runCli(buildCmd(resumeSessionId), label, opts.signal)) {
257
+ if (msg.type === "init") newSession = msg.sessionId;
258
+ // Intercept 529 — retry instead of propagating
259
+ if (msg.type === "error" && msg.text.includes("529")) { got529 = true; break; }
260
+ yield msg;
261
+ if (msg.type === "done" || msg.type === "error") return;
262
+ }
263
+
264
+ if (!got529) return;
265
+ if (attempt === MAX_529_RETRIES) {
266
+ yield { type: "error", text: "Agent process failed: Repeated 529 Overloaded errors — all retries exhausted", timestamp: new Date().toISOString() };
267
+ return;
268
+ }
269
+
270
+ if (newSession) resumeSessionId = newSession;
271
+ const wait = BACKOFF_MS[attempt];
272
+ process.stderr.write(`\n${label}[529] Overloaded — resuming in ${wait / 1000}s (attempt ${attempt + 1}/${MAX_529_RETRIES})${resumeSessionId ? " with session resume" : ""}\n`);
273
+ await new Promise(r => setTimeout(r, wait));
274
+ }
275
+ }
276
+ }
277
+
278
+ export interface ClaudeRuntimeConfig {
279
+ /** Global instructions written to ~/.claude/CLAUDE.md in every sandbox. */
280
+ claudeMdContent?: string;
281
+ /**
282
+ * Environment variables set before each CLI invocation.
283
+ * Use to configure API provider routing (e.g. OpenRouter):
284
+ * env: { ANTHROPIC_BASE_URL: "https://openrouter.ai/api", ANTHROPIC_AUTH_TOKEN: "$OPENROUTER_API_KEY", ANTHROPIC_API_KEY: "" }
285
+ */
286
+ env?: Record<string, string>;
287
+ /** Model to use (e.g. "claude-opus-4-6", "anthropic/claude-sonnet-4-6"). */
288
+ model?: string;
289
+ /** MCP servers to configure in the sandbox. Written to ~/.claude.json before the first iteration. */
290
+ mcpServers?: Record<string, { command: string; args?: string[]; env?: Record<string, string> }>;
291
+ }
292
+
293
+ /**
294
+ * Create a Claude CLI runtime.
295
+ *
296
+ * Encapsulates everything about how the Claude CLI runs in a sandbox:
297
+ * API provider routing, MCP server configuration, and global instructions.
298
+ */
299
+ export function createClaudeRuntime(config: ClaudeRuntimeConfig = {}) {
300
+ return defineRuntime({
301
+ create: (sandbox, opts) => new ClaudeRunner(sandbox, opts, config.claudeMdContent ?? "", config.env ?? {}, config.model, config.mcpServers),
302
+ });
303
+ }
304
+
305
+ export default createClaudeRuntime();
@@ -0,0 +1,151 @@
1
+ /**
2
+ * OpenAI Desktop Runner — ModelExecutionContract implementation using computer-use-preview.
3
+ *
4
+ * Works against DesktopSandboxProvider — no E2B SDK dependency here.
5
+ * The action→screenshot loop runs until the model emits text, then yields it.
6
+ */
7
+
8
+ import OpenAI from "openai";
9
+ import sharp from "sharp";
10
+ import type { DesktopSandboxProvider, AgentMessage, ModelExecutionContract, RuntimeOptions } from "../index.js";
11
+ import { defineRuntime, formatError } from "../index.js";
12
+
13
+ const DISPLAY_WIDTH = 1024;
14
+ const DISPLAY_HEIGHT = 720;
15
+ const DESKTOP_WIDTH = 1280;
16
+ const DESKTOP_HEIGHT = 800;
17
+ const MAX_TURNS = 40;
18
+
19
+ export interface OpenAIDesktopRunnerOptions extends RuntimeOptions {
20
+ openaiKey: string;
21
+ }
22
+
23
+ export class OpenAIDesktopRunner implements ModelExecutionContract {
24
+ private openai: OpenAI;
25
+ private label: string;
26
+
27
+ constructor(
28
+ private sandbox: DesktopSandboxProvider,
29
+ opts: OpenAIDesktopRunnerOptions,
30
+ ) {
31
+ this.openai = new OpenAI({ apiKey: opts.openaiKey });
32
+ this.label = opts.label ?? `[${sandbox.sandboxId.slice(-8)}]`;
33
+ }
34
+
35
+ async *sendMessage(opts: {
36
+ prompt: string;
37
+ sessionId?: string;
38
+ }): AsyncGenerator<AgentMessage> {
39
+ const { sandbox } = this;
40
+ const ts = () => new Date().toISOString();
41
+ const label = this.label;
42
+
43
+ yield { type: "init", sessionId: "e2b-desktop", timestamp: ts() };
44
+
45
+ const screenshot = await captureScaledScreenshot(sandbox);
46
+
47
+ // eslint-disable-next-line @typescript-eslint/no-explicit-any
48
+ const messages: any[] = [{
49
+ role: "user",
50
+ content: [
51
+ { type: "input_image", image_url: `data:image/png;base64,${screenshot}` },
52
+ { type: "input_text", text: opts.prompt },
53
+ ],
54
+ }];
55
+
56
+ let previousResponseId: string | undefined;
57
+
58
+ for (let turn = 0; turn < MAX_TURNS; turn++) {
59
+ // eslint-disable-next-line @typescript-eslint/no-explicit-any
60
+ const response = await (this.openai as any).responses.create({
61
+ model: "computer-use-preview",
62
+ tools: [{ type: "computer_use_preview", name: "computer", display_width: DISPLAY_WIDTH, display_height: DISPLAY_HEIGHT, environment: "linux" }],
63
+ input: messages,
64
+ truncation: "auto",
65
+ ...(previousResponseId ? { previous_response_id: previousResponseId } : {}),
66
+ });
67
+ previousResponseId = response.id;
68
+
69
+ // eslint-disable-next-line @typescript-eslint/no-explicit-any
70
+ const computerCalls: any[] = response.output?.filter((b: any) => b.type === "computer_call") ?? [];
71
+ // eslint-disable-next-line @typescript-eslint/no-explicit-any
72
+ const textBlocks: any[] = response.output?.filter((b: any) => b.type === "text") ?? [];
73
+
74
+ if (textBlocks.length > 0) {
75
+ // eslint-disable-next-line @typescript-eslint/no-explicit-any
76
+ yield { type: "text", text: textBlocks.map((b: any) => b.text).join("\n"), timestamp: ts() };
77
+ break;
78
+ }
79
+
80
+ for (const call of computerCalls) {
81
+ console.log(`${label} → ${call.action.type}`);
82
+ await executeAction(sandbox, call.action).catch(
83
+ (err: unknown) => console.warn(`${label} Action failed (non-fatal): ${formatError(err)}`),
84
+ );
85
+ }
86
+
87
+ if (computerCalls.length === 0) break;
88
+
89
+ const next = await captureScaledScreenshot(sandbox);
90
+ // eslint-disable-next-line @typescript-eslint/no-explicit-any
91
+ messages.push({ role: "user", content: computerCalls.map((call: any) => ({
92
+ type: "computer_call_output",
93
+ call_id: call.call_id,
94
+ output: { type: "input_image", image_url: `data:image/png;base64,${next}` },
95
+ })) });
96
+ }
97
+
98
+ yield { type: "done", sessionId: "e2b-desktop", timestamp: ts() };
99
+ }
100
+ }
101
+
102
+ // ---------------------------------------------------------------------------
103
+ // Helpers — work against DesktopSandboxProvider interface
104
+ // ---------------------------------------------------------------------------
105
+
106
+ async function captureScaledScreenshot(sandbox: DesktopSandboxProvider): Promise<string> {
107
+ const raw = await sandbox.screenshot();
108
+ const scaled = await sharp(raw)
109
+ .resize(DISPLAY_WIDTH, DISPLAY_HEIGHT, { kernel: "lanczos3", fit: "fill" })
110
+ .png()
111
+ .toBuffer();
112
+ return scaled.toString("base64");
113
+ }
114
+
115
+ function scaleX(x: number): number { return Math.round(x * DESKTOP_WIDTH / DISPLAY_WIDTH); }
116
+ function scaleY(y: number): number { return Math.round(y * DESKTOP_HEIGHT / DISPLAY_HEIGHT); }
117
+
118
+ // eslint-disable-next-line @typescript-eslint/no-explicit-any
119
+ async function executeAction(sandbox: DesktopSandboxProvider, action: Record<string, any>): Promise<void> {
120
+ const [x, y] = action.coordinate
121
+ ? [scaleX(action.coordinate[0]), scaleY(action.coordinate[1])]
122
+ : [0, 0];
123
+
124
+ switch (action.type) {
125
+ case "click": await sandbox.leftClick(x, y); break;
126
+ case "double_click": await sandbox.doubleClick(x, y); break;
127
+ case "right_click": await sandbox.rightClick(x, y); break;
128
+ case "middle_click": await sandbox.middleClick(x, y); break;
129
+ case "move": await sandbox.moveMouse(x, y); break;
130
+ case "type": await sandbox.write(action.text ?? ""); break;
131
+ case "key": await sandbox.press(action.key ?? ""); break;
132
+ case "scroll": await sandbox.scroll(action.direction === "up" ? "up" : "down", action.ticks ?? 3); break;
133
+ case "drag":
134
+ if (action.startCoordinate && action.endCoordinate) {
135
+ await sandbox.drag(
136
+ [scaleX(action.startCoordinate[0]), scaleY(action.startCoordinate[1])],
137
+ [scaleX(action.endCoordinate[0]), scaleY(action.endCoordinate[1])],
138
+ );
139
+ }
140
+ break;
141
+ }
142
+ await new Promise(r => setTimeout(r, 200));
143
+ }
144
+
145
+ export default defineRuntime<DesktopSandboxProvider>({
146
+ create: (sandbox, opts) => {
147
+ const openaiKey = process.env.OPENAI_API_KEY;
148
+ if (!openaiKey) throw new Error("OPENAI_API_KEY environment variable is required for the openai-desktop runtime");
149
+ return new OpenAIDesktopRunner(sandbox, { ...opts, openaiKey });
150
+ },
151
+ });