@agent-compose/sdk 0.5.9 → 0.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (49) hide show
  1. package/dist/agent/agent-context.d.ts +1 -1
  2. package/dist/agent/agent-loop.d.ts +0 -8
  3. package/dist/agent/pause-client.d.ts +50 -0
  4. package/dist/index.d.ts +15 -6
  5. package/dist/index.js +537 -124
  6. package/dist/processors/ask-human.d.ts +30 -0
  7. package/dist/processors/ask-human.test.d.ts +1 -0
  8. package/dist/processors/gate-pause.d.ts +46 -0
  9. package/dist/processors/gate-pause.test.d.ts +1 -0
  10. package/dist/processors/index.d.ts +3 -0
  11. package/dist/runtimes/_cli-agent.d.ts +9 -0
  12. package/dist/runtimes/cursor.d.ts +9 -0
  13. package/dist/runtimes/droid.d.ts +9 -0
  14. package/dist/runtimes/openai-desktop.js +522 -122
  15. package/dist/runtimes/opencode.d.ts +25 -0
  16. package/dist/runtimes/vercel.js +11 -1
  17. package/dist/step-invocation/__tests__/background-invoker.test.d.ts +1 -0
  18. package/dist/step-invocation/index.d.ts +2 -1
  19. package/dist/step-invocation/invoker.d.ts +49 -0
  20. package/dist/step-invocation/protocol.d.ts +8 -0
  21. package/dist/types/runtime.d.ts +7 -0
  22. package/dist/types/sandbox.d.ts +54 -0
  23. package/dist/types/workflow.d.ts +12 -0
  24. package/dist/utils/errors.d.ts +9 -1
  25. package/package.json +1 -1
  26. package/src/agent/agent-context.ts +23 -16
  27. package/src/agent/agent-loop.ts +13 -33
  28. package/src/agent/pause-client.ts +108 -0
  29. package/src/agent/run-agent.ts +16 -7
  30. package/src/index.ts +27 -4
  31. package/src/processors/ask-human.ts +136 -0
  32. package/src/processors/gate-pause.ts +94 -0
  33. package/src/processors/index.ts +11 -0
  34. package/src/runtimes/_cli-agent.ts +13 -5
  35. package/src/runtimes/claude.ts +10 -6
  36. package/src/runtimes/cursor.ts +59 -0
  37. package/src/runtimes/droid.ts +63 -0
  38. package/src/runtimes/opencode.ts +61 -0
  39. package/src/sandbox.ts +78 -3
  40. package/src/step-invocation/index.ts +2 -1
  41. package/src/step-invocation/invoker.ts +359 -86
  42. package/src/step-invocation/protocol.ts +11 -0
  43. package/src/types/runtime.ts +7 -0
  44. package/src/types/sandbox.ts +53 -0
  45. package/src/types/workflow.ts +12 -0
  46. package/src/utils/errors.ts +19 -2
  47. package/dist/agent/local-pause-request.d.ts +0 -49
  48. package/src/agent/local-pause-request.ts +0 -90
  49. /package/dist/agent/{local-pause-request.test.d.ts → pause-client.test.d.ts} +0 -0
@@ -0,0 +1,136 @@
1
+ /**
2
+ * Ask-human processor (ADR-0028).
3
+ *
4
+ * Makes "ask a human" a FIRST-CLASS agent affordance: when the agent calls the
5
+ * built-in `AskUserQuestion` tool, this short-circuits it into a SERVER pause
6
+ * (`requestPauseAndAwait`) — the run freezes (compute stops), the question +
7
+ * options land on the human's pause feed, and the human's answer comes back as
8
+ * the tool result. No `agentc pause` CLI for the model to remember, and no
9
+ * dependency on prompt discipline: the moment the agent asks, the run pauses.
10
+ *
11
+ * Loud by construction — the danger this fixes is a pause that SILENTLY doesn't
12
+ * happen (a stale in-sandbox CLI, a non-E2B substrate, an auth error) letting
13
+ * the agent proceed as if it had an answer:
14
+ * - pause cannot be created (server reject) → `Verdict.abort` ENDS the agent
15
+ * loop with a WorkflowError. The run fails loud; it never guesses an answer.
16
+ * - pause expires / is cancelled → the tool result says NO answer came and to
17
+ * not assume one.
18
+ *
19
+ * Lives in the shared `gateToolCall` chain, so one implementation covers every
20
+ * runtime (the Claude Agent SDK `PreToolUse` hook and the ACP permission path).
21
+ * No run credential in the env (local / non-sandbox) → no-op: `AskUserQuestion`
22
+ * passes through untouched so a dev invocation isn't hard-failed.
23
+ */
24
+
25
+ import type { Processor, ProcessorContext, ToolCall } from "./processor.js";
26
+ import { Verdict } from "./processor.js";
27
+ import { requestPauseAndAwait } from "../agent/pause-client.js";
28
+ import type { GatePauseConnection } from "./gate-pause.js";
29
+
30
+ /** The Claude built-in tool an agent uses to ask the user a question. */
31
+ export const ASK_USER_QUESTION_TOOL = "AskUserQuestion";
32
+
33
+ /** One question in an `AskUserQuestion` call (only the fields we read). */
34
+ interface AskQuestion { question?: unknown; header?: unknown; options?: unknown }
35
+
36
+ /** Pull the human-facing question + its option labels out of an
37
+ * `AskUserQuestion` tool input. We pause on the FIRST question (the common
38
+ * case); any others are folded into the reason so nothing is lost. */
39
+ function parseAsk(input: Record<string, unknown>): { reason: string; options: Array<{ id: string; label: string }> } {
40
+ const questions = Array.isArray(input.questions) ? (input.questions as AskQuestion[]) : [];
41
+ const first = questions[0] ?? {};
42
+ const head = typeof first.question === "string" && first.question.trim()
43
+ ? first.question.trim()
44
+ : "The agent needs your input to continue.";
45
+ const extra = questions.length > 1
46
+ ? ` (+${questions.length - 1} more question${questions.length > 2 ? "s" : ""})`
47
+ : "";
48
+ const options = Array.isArray(first.options)
49
+ ? (first.options as Array<{ label?: unknown }>)
50
+ .map((o) => String(o?.label ?? "").trim())
51
+ .filter(Boolean)
52
+ .map((label) => ({ id: label, label }))
53
+ : [];
54
+ return { reason: head + extra, options };
55
+ }
56
+
57
+ /** Unwrap the dashboard's `{ decision }` resume payload to the raw answer text. */
58
+ function answerText(decision: unknown): string {
59
+ const raw = decision !== null && typeof decision === "object" && "decision" in decision
60
+ ? (decision as { decision: unknown }).decision
61
+ : decision;
62
+ return typeof raw === "string" ? raw.trim() : raw == null ? "" : JSON.stringify(raw);
63
+ }
64
+
65
+ /** Recognise the agent shelling out to `agentc pause` and extract the same
66
+ * {reason, options} we'd get from AskUserQuestion. We INTERCEPT it here — in the
67
+ * tool gate, BEFORE the command runs — so the pause takes the DESIGNED path (the
68
+ * answer returns as the tool result and the agent loop continues) instead of the
69
+ * command actually blocking inside the sandbox shell, which froze the runner
70
+ * mid-tool-exec and never resumed the loop. */
71
+ function parseAgentcPause(command: string): { reason: string; options: Array<{ id: string; label: string }> } | null {
72
+ if (!/(^|\s|&&|;|\|)\s*agentc\s+pause(\s|$)/.test(command)) return null;
73
+ const r = command.match(/--reason(?:=|\s+)(?:"([^"]*)"|'([^']*)'|(\S+))/);
74
+ const reason = (r?.[1] ?? r?.[2] ?? r?.[3] ?? "The agent needs your input to continue.").trim();
75
+ const options = [...command.matchAll(/--option(?:=|\s+)(?:"([^"]*)"|'([^']*)'|(\S+))/g)]
76
+ .map((m) => (m[1] ?? m[2] ?? m[3] ?? "").trim())
77
+ .filter(Boolean)
78
+ .map((label) => ({ id: label, label }));
79
+ return { reason, options };
80
+ }
81
+
82
+ /** Pull the ask (reason + options) from either the `AskUserQuestion` tool OR a
83
+ * `Bash` call running `agentc pause`. Null for anything else. */
84
+ function extractAsk(call: ToolCall): { reason: string; options: Array<{ id: string; label: string }> } | null {
85
+ if (call.toolName === ASK_USER_QUESTION_TOOL) return parseAsk(call.toolInput);
86
+ if (call.toolName === "Bash") {
87
+ const cmd = (call.toolInput as { command?: unknown })?.command;
88
+ return typeof cmd === "string" ? parseAgentcPause(cmd) : null;
89
+ }
90
+ return null;
91
+ }
92
+
93
+ export function createAskHumanProcessor(opts: { connection?: GatePauseConnection } = {}): Processor {
94
+ return {
95
+ name: "ask-human",
96
+ async processToolCall(call: ToolCall, ctx: ProcessorContext) {
97
+ // Fires for AskUserQuestion OR a `Bash` call running `agentc pause` — both
98
+ // ask a human and must take the SAME processor path so the answer returns
99
+ // as the tool result and the agent loop continues.
100
+ const ask = extractAsk(call);
101
+ if (!ask) return Verdict.continue(call);
102
+
103
+ const conn = opts.connection ?? {
104
+ baseUrl: process.env.AGENT_COMPOSE_URL ?? "",
105
+ token: process.env.AGENT_COMPOSE_RUN_TOKEN ?? "",
106
+ runId: process.env.RUN_ID ?? "",
107
+ };
108
+ // No run credential (local / non-sandbox) — can't pause; let the tool
109
+ // through rather than hard-failing a dev invocation.
110
+ if (!conn.baseUrl || !conn.token || !conn.runId) return Verdict.continue(call);
111
+
112
+ let decision;
113
+ try {
114
+ decision = await requestPauseAndAwait({
115
+ baseUrl: conn.baseUrl, token: conn.token, runId: conn.runId,
116
+ reason: ask.reason,
117
+ ...(ask.options.length ? { options: ask.options } : {}),
118
+ action: { tool: call.toolName, input: call.toolInput },
119
+ signal: ctx.abortSignal,
120
+ });
121
+ } catch (err) {
122
+ // The pause could NOT be honored (server reject, non-E2B substrate, auth).
123
+ // Abort the loop — never let the agent proceed as if it had an answer.
124
+ const msg = err instanceof Error ? err.message : String(err);
125
+ return Verdict.abort(`Could not ask the human — the run could not be paused (${msg}). Stopping rather than guessing an answer.`);
126
+ }
127
+
128
+ if (decision.status === "resolved") {
129
+ const answer = answerText(decision.decision);
130
+ return Verdict.deny(`The human answered: ${answer || "(no text returned)"}. Continue using this answer.`);
131
+ }
132
+ // Expired / cancelled — no answer. Do NOT let the agent assume one.
133
+ return Verdict.deny(`No answer came back (${decision.status}). Do NOT assume an answer — ask again, or stop and report exactly what you need from a human.`);
134
+ },
135
+ };
136
+ }
@@ -0,0 +1,94 @@
1
+ /**
2
+ * Gate-pause processor (ADR-0028).
3
+ *
4
+ * A pre-tool gate that, when its `policy` flags a tool call as needing human
5
+ * approval, pauses the run as a SERVER operation (`requestPauseAndAwait`) and
6
+ * blocks until a human answers — then allows or denies the tool.
7
+ *
8
+ * It pauses through the server pause API (`requestPauseAndAwait`), so the pause
9
+ * is a server operation no `try/catch` can swallow. It does NOT use today's
10
+ * `ctx.pause`, which still throws `PauseSignal` — exactly the swallowable path
11
+ * ADR-0028 supersedes for agents. (Routing `ctx.pause` itself through this same
12
+ * server pause is the natural unification — ADR-0028 open question — at which
13
+ * point "the pause API" and "ctx.pause" become one thing.) Because it lives in
14
+ * the shared `gateToolCall` chain, one implementation covers every runtime — the
15
+ * Claude Agent SDK `PreToolUse` hook and the ACP `session/request_permission`
16
+ * path both route through it.
17
+ *
18
+ * Opt-in: the workflow/agent supplies the `policy` (which may be a heuristic or
19
+ * an async classifier agent). The connection defaults to the run credential in
20
+ * the sandbox env; absent it (local / non-sandbox), the gate is a no-op and
21
+ * tool calls pass through untouched.
22
+ */
23
+
24
+ import type { Processor, ProcessorContext, ToolCall } from "./processor.js";
25
+ import { Verdict } from "./processor.js";
26
+ import { requestPauseAndAwait } from "../agent/pause-client.js";
27
+
28
+ export interface GatePauseApproval {
29
+ /** Human-readable question shown on the approval UI. */
30
+ reason: string;
31
+ /** Decision options the human picks from. Defaults to Approve / Deny. */
32
+ options?: Array<{ id: string; label: string }>;
33
+ }
34
+
35
+ /** Decide whether a tool call needs human approval. Return `false` to let it
36
+ * through untouched, or an approval request to pause the run until a human
37
+ * answers. May be async (e.g. a small classifier agent). */
38
+ export type GatePausePolicy = (
39
+ call: ToolCall,
40
+ ctx: ProcessorContext,
41
+ ) => (false | GatePauseApproval) | Promise<false | GatePauseApproval>;
42
+
43
+ export interface GatePauseConnection { baseUrl: string; token: string; runId: string }
44
+
45
+ const DENY_ANSWERS = new Set(["deny", "no", "reject", "decline", "block"]);
46
+
47
+ /** Unwrap the dashboard's `{ decision }` resume payload to the raw answer text. */
48
+ function answerText(decision: unknown): string {
49
+ const raw = decision !== null && typeof decision === "object" && "decision" in decision
50
+ ? (decision as { decision: unknown }).decision
51
+ : decision;
52
+ return typeof raw === "string" ? raw.trim() : raw == null ? "" : JSON.stringify(raw);
53
+ }
54
+
55
+ export function createGatePauseProcessor(opts: {
56
+ policy: GatePausePolicy;
57
+ /** Override the server connection (defaults to the sandbox run-credential env). */
58
+ connection?: GatePauseConnection;
59
+ }): Processor {
60
+ return {
61
+ name: "gate-pause",
62
+ async processToolCall(call: ToolCall, ctx: ProcessorContext) {
63
+ const approval = await opts.policy(call, ctx);
64
+ if (!approval) return Verdict.continue(call);
65
+
66
+ const conn = opts.connection ?? {
67
+ baseUrl: process.env.AGENT_COMPOSE_URL ?? "",
68
+ token: process.env.AGENT_COMPOSE_RUN_TOKEN ?? "",
69
+ runId: process.env.RUN_ID ?? "",
70
+ };
71
+ // No run credential (local / non-sandbox) — can't pause; let it through
72
+ // rather than hard-failing a dev invocation.
73
+ if (!conn.baseUrl || !conn.token || !conn.runId) return Verdict.continue(call);
74
+
75
+ const decision = await requestPauseAndAwait({
76
+ baseUrl: conn.baseUrl, token: conn.token, runId: conn.runId,
77
+ reason: approval.reason,
78
+ action: { tool: call.toolName, input: call.toolInput },
79
+ options: approval.options ?? [{ id: "approve", label: "Approve" }, { id: "deny", label: "Deny" }],
80
+ signal: ctx.abortSignal,
81
+ });
82
+
83
+ const answer = answerText(decision.decision);
84
+ // Approve unless the human explicitly denied (or no answer came back on
85
+ // expiry/cancel). A free-form answer that isn't a deny word lets the tool
86
+ // run, with the human's guidance available to the model on the next turn.
87
+ if (decision.status === "resolved" && !DENY_ANSWERS.has(answer.toLowerCase())) {
88
+ return Verdict.continue(call);
89
+ }
90
+ const why = decision.status === "resolved" ? `denied: ${answer}` : `${decision.status} with no approval`;
91
+ return Verdict.deny(`Human ${why}. Tool "${call.toolName}" was not run — adjust course or ask again.`);
92
+ },
93
+ };
94
+ }
@@ -14,3 +14,14 @@ export {
14
14
  requireScope,
15
15
  redactPattern,
16
16
  } from "./builtins.js";
17
+
18
+ // ADR-0028 — server-driven human-approval gate. Pauses the run through the
19
+ // server pause API and blocks until a human answers; covers every runtime
20
+ // through the shared gateToolCall chain.
21
+ export { createGatePauseProcessor } from "./gate-pause.js";
22
+ export type { GatePausePolicy, GatePauseApproval, GatePauseConnection } from "./gate-pause.js";
23
+
24
+ // ADR-0028 — first-class "ask a human": maps the agent's `AskUserQuestion` tool
25
+ // to a server pause and returns the human's answer as the tool result. Added by
26
+ // default to every agent (no-op unless the tool is granted + called).
27
+ export { createAskHumanProcessor, ASK_USER_QUESTION_TOOL } from "./ask-human.js";
@@ -56,6 +56,16 @@ const ACP_FALLBACK = Symbol("acp-fallback");
56
56
  * ops tuning; defaults sane. */
57
57
  export const ACP_HANDSHAKE_TIMEOUT_MS = Number(process.env.AC_ACP_HANDSHAKE_TIMEOUT_MS) || 10_000;
58
58
 
59
+ /** Idle deadline for a PROMPT TURN (distinct from the handshake gate above). A
60
+ * turn is killed only if it goes fully SILENT for this long — the deadline is
61
+ * re-armed on every streamed message, so a long, *streaming* turn never trips
62
+ * it. This must be generous: a reasoning model (GLM, gpt-5-codex) can think for
63
+ * tens of seconds between tool calls with no wire activity, which is NOT a hang.
64
+ * The 10s handshake timeout was far too tight here and killed live GLM turns
65
+ * mid-report. Only a genuinely wedged CLI (the Gemini-style hang) should trip
66
+ * this. Overridable via the env for ops tuning. */
67
+ export const ACP_TURN_IDLE_TIMEOUT_MS = Number(process.env.AC_ACP_TURN_IDLE_TIMEOUT_MS) || 120_000;
68
+
59
69
  /** Readiness gate for the LIVE ACP attempt. The duplex-stdin transport in
60
70
  * `spawnAcpProcess` is now real (`commands.spawnDuplex` on the local provider),
61
71
  * so the agent's `initialize` request bytes are delivered and the handshake can
@@ -448,7 +458,7 @@ export class CliAgentRunner implements ModelExecutionContract {
448
458
  let watchdog: ReturnType<typeof setTimeout> | undefined;
449
459
  const armWatchdog = () => {
450
460
  if (watchdog) clearTimeout(watchdog);
451
- watchdog = setTimeout(() => { stalled = true; peer.cancel(); }, ACP_HANDSHAKE_TIMEOUT_MS);
461
+ watchdog = setTimeout(() => { stalled = true; peer.cancel(); }, ACP_TURN_IDLE_TIMEOUT_MS);
452
462
  };
453
463
  try {
454
464
  armWatchdog();
@@ -466,7 +476,7 @@ export class CliAgentRunner implements ModelExecutionContract {
466
476
  // and surface the stall as an error.
467
477
  await this.teardownAcp(proc, peer);
468
478
  this.acpSession = undefined;
469
- yield { type: "error", text: `${this.spec.kind} prompt turn stalled (no activity for ${ACP_HANDSHAKE_TIMEOUT_MS}ms)`, timestamp: now() };
479
+ yield { type: "error", text: `${this.spec.kind} prompt turn stalled (no activity for ${ACP_TURN_IDLE_TIMEOUT_MS}ms)`, timestamp: now() };
470
480
  return;
471
481
  }
472
482
 
@@ -502,9 +512,7 @@ export class CliAgentRunner implements ModelExecutionContract {
502
512
  }
503
513
 
504
514
  const cmd = `${acp.command} ${acp.args.map(shellQuote).join(" ")}`.trim();
505
- // Expose this agent's id to the adapter (and the tool subprocesses it
506
- // spawns) so `agentc pause` scopes its state-dir marker to this agent.
507
- const agentEnv = { ...acp.env, ...(this.options.agentId ? { AGENT_COMPOSE_AGENT_ID: this.options.agentId } : {}) };
515
+ const agentEnv = { ...acp.env };
508
516
  const proc = spawnDuplex.call(this.sandbox.commands, cmd, {
509
517
  ...(this.options.cwd ? { cwd: this.options.cwd } : {}),
510
518
  ...(Object.keys(agentEnv).length > 0 ? { envs: agentEnv } : {}),
@@ -265,6 +265,14 @@ export class ClaudeRunner implements ModelExecutionContract {
265
265
  else opts.signal.addEventListener("abort", onLoopAbort, { once: true });
266
266
  }
267
267
 
268
+ // System-prompt append: the platform manual (agentManual, threaded from
269
+ // agent()) + any caller-supplied claudeMdContent. The Agent SDK does NOT
270
+ // auto-load CLAUDE.md from cwd (no settingSources), so without this the
271
+ // agent never reliably sees the manual — including how to pause.
272
+ const systemPromptAppend = [this.config.claudeMdContent, this.options.agentManual]
273
+ .filter((s): s is string => Boolean(s && s.trim()))
274
+ .join("\n\n");
275
+
268
276
  try {
269
277
  let emittedAssistantText = false;
270
278
  for await (const message of query({
@@ -279,15 +287,11 @@ export class ClaudeRunner implements ModelExecutionContract {
279
287
  effort: this.config.effort,
280
288
  cwd: this.options.cwd,
281
289
  abortController: queryAbort,
282
- // Expose this agent's id to its tool subprocesses so `agentc pause`
283
- // can scope its state-dir pause-request marker to the right agent
284
- // (concurrent agent() calls share one sandbox). The loop's agentId is
285
- // authoritative, so it wins over a stale env / config value.
286
- env: { ...process.env, ...(this.config.env ?? {}), ...(this.options.agentId ? { AGENT_COMPOSE_AGENT_ID: this.options.agentId } : {}) },
290
+ env: { ...process.env, ...(this.config.env ?? {}) },
287
291
  pathToClaudeCodeExecutable: this.config.pathToClaudeCodeExecutable
288
292
  ?? process.env.CLAUDE_CODE_EXECUTABLE
289
293
  ?? DEFAULT_CLAUDE_PATH,
290
- ...(this.config.claudeMdContent ? { systemPrompt: { type: "preset" as const, preset: "claude_code" as const, append: this.config.claudeMdContent } } : {}),
294
+ ...(systemPromptAppend ? { systemPrompt: { type: "preset" as const, preset: "claude_code" as const, append: systemPromptAppend } } : {}),
291
295
  // Turn skills ON (and auto-add the `Skill` tool). The agent-env
292
296
  // bakes the `/ac:*` skills; without this the Agent SDK leaves
293
297
  // them un-enabled and the agent can't invoke them.
@@ -0,0 +1,59 @@
1
+ /**
2
+ * Cursor CLI runtime — drives Cursor's `cursor-agent` inside the sandbox.
3
+ *
4
+ * ACP-native: `cursor-agent acp` is a protocolVersion-1 ACP server (verified
5
+ * live on E2B 2026-06-30), so the runner delegates the wire protocol to
6
+ * AcpClientPeer; the JSONL members below are the version-mismatch fallback.
7
+ *
8
+ * Auth: CURSOR_API_KEY — Cursor's OWN platform key, NOT OpenRouter. In ACP mode
9
+ * `cursor-agent acp` takes no model flag, so it runs the account's default model;
10
+ * the `--model` ids (auto, gpt-5.3-codex, composer-2.5,
11
+ * claude-opus-4-8-thinking-high; full list via `cursor-agent --list-models`)
12
+ * only apply to the JSONL `-p` fallback. cursor brings HARNESS diversity (a
13
+ * different agent scaffold) to a cross-functional / review panel.
14
+ *
15
+ * Verified live on E2B (2026-06-30): `curl https://cursor.com/install` →
16
+ * ~/.local/bin/cursor-agent (v2026.06.29); `cursor-agent acp` answered the ACP
17
+ * `initialize` with protocolVersion 1; CURSOR_API_KEY authenticated.
18
+ */
19
+ import type { AgentMessage } from "../index.js";
20
+ import { createCliAgentRuntime, shellQuote, type CliAgentSpec } from "./_cli-agent.js";
21
+
22
+ function now(): string { return new Date().toISOString(); }
23
+
24
+ export const cursorSpec: CliAgentSpec = {
25
+ kind: "cursor",
26
+ authEnv: "CURSOR_API_KEY",
27
+ bin: "cursor-agent",
28
+ // ACP runs the account default; "auto" is Cursor's own auto-routing label.
29
+ defaultModel: "auto",
30
+ // `cursor-agent acp` is a protocolVersion-1 ACP server — the runner delegates
31
+ // the whole wire protocol to AcpClientPeer. It takes no model flag, so the
32
+ // session runs Cursor's account-default model.
33
+ acp: { command: "cursor-agent", args: ["acp"] },
34
+ // Self-install on first use; symlink onto PATH for a non-login `sh -c`.
35
+ install: 'curl https://cursor.com/install -fsS | bash && (command -v cursor-agent >/dev/null 2>&1 || sudo ln -sf "$HOME/.local/bin/cursor-agent" /usr/local/bin/cursor-agent)',
36
+ // ── JSONL fallback (only if the ACP handshake negotiates a non-1 version;
37
+ // cursor is v1, so vestigial). `-p` needs `--force` to clear the
38
+ // workspace-trust gate non-interactively.
39
+ promptPayload: (prompt) => prompt,
40
+ buildCommand: ({ promptPath, model, cwd }) =>
41
+ `${cwd ? `cd ${shellQuote(cwd)} && ` : ""}cursor-agent -p --force ${model ? `--model ${shellQuote(model)} ` : ""}--output-format text "$(cat ${shellQuote(promptPath)})"`,
42
+ extractSessionId: () => undefined,
43
+ mapEvent: (p): AgentMessage[] => {
44
+ const ts = now();
45
+ const text = typeof p.text === "string" ? p.text : typeof p.content === "string" ? p.content : "";
46
+ return text ? [{ type: "text", text, timestamp: ts }] : [];
47
+ },
48
+ };
49
+
50
+ export interface CursorRuntimeConfig {
51
+ /** Cursor model id; ACP mode ignores it (account default) — applies to the `-p` fallback. */
52
+ model?: string;
53
+ }
54
+
55
+ export function createCursorRuntime(config: CursorRuntimeConfig = {}) {
56
+ return createCliAgentRuntime(cursorSpec, config.model ?? cursorSpec.defaultModel);
57
+ }
58
+
59
+ export default createCursorRuntime();
@@ -0,0 +1,63 @@
1
+ /**
2
+ * Factory `droid` runtime — drives `droid exec` headless inside the sandbox,
3
+ * JSONL via `--output-format json`.
4
+ *
5
+ * Driven PURELY via OpenRouter BYOK — NO Factory login (verified live on E2B
6
+ * 2026-06-30): a `~/.factory/settings.json` `customModels` entry points at
7
+ * OpenRouter, and the model id is `custom:<displayName>-<index>`. The workflow
8
+ * provisions settings.json (see the dynamic-task route step); this spec just
9
+ * builds the exec command. Auth env is OPENROUTER_API_KEY (the BYOK inference
10
+ * key shared with codex + opencode).
11
+ *
12
+ * NOT ACP: `droid exec` is JSONL. (Its stream-jsonrpc mode ignores --model and
13
+ * sets model/autonomy via JSON-RPC; the plain `--output-format json` mode honours
14
+ * --model, which is what we use.) So it always drives the JSONL path. droid adds
15
+ * Factory's agent HARNESS to a cross-functional / review panel.
16
+ *
17
+ * Verified live on E2B (2026-06-30): install via app.factory.ai/cli;
18
+ * `droid exec --auto low --model custom:GLM-5.2-OR-0 --output-format json`
19
+ * returned `{"type":"result","result":"…","session_id":"…","usage":{…}}` driven
20
+ * through OpenRouter with no Factory account.
21
+ */
22
+ import type { AgentMessage } from "../index.js";
23
+ import { createCliAgentRuntime, shellQuote, type CliAgentSpec } from "./_cli-agent.js";
24
+
25
+ function now(): string { return new Date().toISOString(); }
26
+
27
+ export const droidSpec: CliAgentSpec = {
28
+ kind: "droid",
29
+ // OpenRouter is the inference gateway (BYOK custom model). No FACTORY_API_KEY.
30
+ authEnv: "OPENROUTER_API_KEY",
31
+ bin: "droid",
32
+ // Matches the first customModels entry the workflow writes to settings.json.
33
+ defaultModel: "custom:GLM-5.2-OR-0",
34
+ install: 'curl -fsSL https://app.factory.ai/cli | sh && (command -v droid >/dev/null 2>&1 || sudo ln -sf "$HOME/.local/bin/droid" /usr/local/bin/droid)',
35
+ // ── JSONL path (always — droid exec is not ACP). `--auto medium` lets the
36
+ // agent create/edit files + run commands; `-f` reads the prompt from a file.
37
+ promptPayload: (prompt) => prompt,
38
+ buildCommand: ({ promptPath, model, cwd }) =>
39
+ `${cwd ? `cd ${shellQuote(cwd)} && ` : ""}droid exec --auto medium ${model ? `--model ${shellQuote(model)} ` : ""}--output-format json -f ${shellQuote(promptPath)}`,
40
+ extractSessionId: (p) => (typeof p.session_id === "string" ? p.session_id : undefined),
41
+ mapEvent: (p): AgentMessage[] => {
42
+ const ts = now();
43
+ // `--output-format json` emits a single terminal result object.
44
+ const text =
45
+ p.type === "result" && typeof p.result === "string" ? p.result
46
+ : typeof p.text === "string" ? p.text
47
+ : typeof p.content === "string" ? p.content
48
+ : "";
49
+ return text ? [{ type: "text", text, timestamp: ts }] : [];
50
+ },
51
+ // No `acp` → JSONL-only spec.
52
+ };
53
+
54
+ export interface DroidRuntimeConfig {
55
+ /** `custom:<displayName>-<index>` matching the provisioned settings.json. */
56
+ model?: string;
57
+ }
58
+
59
+ export function createDroidRuntime(config: DroidRuntimeConfig = {}) {
60
+ return createCliAgentRuntime(droidSpec, config.model ?? droidSpec.defaultModel);
61
+ }
62
+
63
+ export default createDroidRuntime();
@@ -0,0 +1,61 @@
1
+ /**
2
+ * OpenCode CLI runtime — drives sst's `opencode` agentic CLI inside the sandbox.
3
+ * OpenCode speaks ACP natively (`opencode acp`, protocolVersion 1 — verified
4
+ * live on E2B), so the runner drives it over ACP; the JSONL members below are
5
+ * only the version-mismatch fallback (vestigial for a v1 agent).
6
+ *
7
+ * Auth + model via OpenRouter: set OPENROUTER_API_KEY (a factory/workflow
8
+ * secret) and use a model id like `openrouter/z-ai/glm-5.2`. The runtime
9
+ * installs the `opencode-ai` CLI on demand; pair with
10
+ * `snapshots: { bootFrom: "reuse" }` to install once and boot from the capture.
11
+ *
12
+ * Verified live on E2B (2026-06-30): `npm i -g opencode-ai` (v1.17.12);
13
+ * `opencode run --model openrouter/z-ai/glm-5.2` drove a GLM-5.2 turn through
14
+ * OpenRouter; `opencode acp` answered the ACP `initialize` handshake with
15
+ * protocolVersion 1.
16
+ */
17
+
18
+ import type { AgentMessage } from "../index.js";
19
+ import { createCliAgentRuntime, shellQuote, type CliAgentSpec } from "./_cli-agent.js";
20
+
21
+ function now(): string { return new Date().toISOString(); }
22
+
23
+ export const opencodeSpec: CliAgentSpec = {
24
+ kind: "opencode",
25
+ // OpenRouter is the gateway: opencode reads OPENROUTER_API_KEY from the env
26
+ // and serves any `openrouter/<provider>/<model>` id (e.g. z-ai/glm-5.2).
27
+ authEnv: "OPENROUTER_API_KEY",
28
+ bin: "opencode",
29
+ defaultModel: "openrouter/z-ai/glm-5.2",
30
+ // ACP-mode invocation — `opencode acp` is a protocolVersion-1 ACP server, so
31
+ // the runner delegates the whole wire protocol to AcpClientPeer. The model is
32
+ // resolved from opencode's config / the `--model` it was started with; the
33
+ // run provisioning writes the OpenRouter default so ACP turns use GLM-5.2.
34
+ acp: { command: "opencode", args: ["acp"] },
35
+ // Global npm install; symlink onto PATH only if the global bin dir isn't
36
+ // already there (so a non-login `sh -c` can find it).
37
+ install: 'sudo npm install -g opencode-ai && (command -v opencode >/dev/null 2>&1 || sudo ln -sf "$(npm prefix -g)/bin/opencode" /usr/local/bin/opencode)',
38
+ // ── JSONL fallback (only reached if the ACP handshake negotiates a non-1
39
+ // version; opencode is v1, so this is vestigial). `opencode run` prints
40
+ // human-formatted text, so we capture the prompt round-trip as one message.
41
+ promptPayload: (prompt) => prompt,
42
+ buildCommand: ({ promptPath, model, cwd }) =>
43
+ `${cwd ? `cd ${shellQuote(cwd)} && ` : ""}opencode run ${model ? `--model ${shellQuote(model)} ` : ""}"$(cat ${shellQuote(promptPath)})"`,
44
+ extractSessionId: () => undefined,
45
+ mapEvent: (p): AgentMessage[] => {
46
+ const ts = now();
47
+ const text = typeof p.text === "string" ? p.text : typeof p.content === "string" ? p.content : "";
48
+ return text ? [{ type: "text", text, timestamp: ts }] : [];
49
+ },
50
+ };
51
+
52
+ export interface OpencodeRuntimeConfig {
53
+ /** OpenRouter-prefixed model id (default `openrouter/z-ai/glm-5.2`). */
54
+ model?: string;
55
+ }
56
+
57
+ export function createOpencodeRuntime(config: OpencodeRuntimeConfig = {}) {
58
+ return createCliAgentRuntime(opencodeSpec, config.model ?? opencodeSpec.defaultModel);
59
+ }
60
+
61
+ export default createOpencodeRuntime();
package/src/sandbox.ts CHANGED
@@ -10,12 +10,12 @@ import { dirname } from "node:path";
10
10
  import { spawn } from "node:child_process";
11
11
  import { Readable, Writable } from "node:stream";
12
12
  import { Sandbox, SandboxNotFoundError, RateLimitError } from "e2b";
13
- import type { SandboxNetworkOpts as E2bNetworkOpts, SandboxNetworkRule as E2bNetworkRule } from "e2b";
13
+ import type { SandboxNetworkOpts as E2bNetworkOpts, SandboxNetworkRule as E2bNetworkRule, CommandHandle } from "e2b";
14
14
  import { Sandbox as Desktop } from "@e2b/desktop";
15
15
  import pRetry from "p-retry";
16
16
  import type { FailedAttemptError } from "p-retry";
17
17
  import { SandboxUnavailableError } from "./sandbox-errors.js";
18
- import type { SandboxProvider, DesktopSandboxProvider, SandboxCommandResult } from "./types/sandbox.js";
18
+ import type { SandboxProvider, DesktopSandboxProvider, SandboxCommandResult, SandboxCommandRunOptions, SandboxBackgroundProcess } from "./types/sandbox.js";
19
19
  import type { ConnectorRequestRules } from "./types/workflow-metadata.js";
20
20
  import type { NetworkPolicy as VercelNetworkPolicy, NetworkPolicyRule as VercelNetworkPolicyRule } from "@vercel/sandbox";
21
21
 
@@ -301,6 +301,15 @@ export function makeSandboxProvider(sb: Sandbox | Desktop): SandboxProvider {
301
301
  async write(path, content) {
302
302
  await (sb.files.write as (p: string, d: string) => Promise<unknown>)(path, content);
303
303
  },
304
+ // Read over the envd HTTP API (`Sandbox.files.read` → `GET /files`), a
305
+ // DIFFERENT transport from `commands` (the connect-web gRPC stream). A large
306
+ // readback over HTTP is decoded by `fetch`'s native `Content-Encoding`
307
+ // handling, so it is immune to the connect-web "received unsupported
308
+ // compressed output" failure that can abort `commands.run` output on a big
309
+ // frame — the property `launchStep` relies on to recover full logs.
310
+ async read(path) {
311
+ return await (sb.files.read as (p: string) => Promise<string>)(path);
312
+ },
304
313
  },
305
314
  // e2b 2.30 `kill()` returns Promise<boolean>; our provider contract is
306
315
  // Promise<void>, so discard the result.
@@ -363,14 +372,69 @@ function withSnapshotRetry(p: SandboxProvider): SandboxProvider {
363
372
  return { ...p, snapshot: () => withSandboxRetry(snapshot) };
364
373
  }
365
374
 
375
+ /** Wrap an E2B `CommandHandle` as a provider-agnostic background process.
376
+ * `wait()` is normalised NOT to throw on a non-zero exit (mirroring the
377
+ * `commands.run` contract in `makeSandboxProvider`) so callers branch on
378
+ * `exitCode` instead of catching. */
379
+ function wrapE2bBackgroundProcess(handle: CommandHandle): SandboxBackgroundProcess {
380
+ return {
381
+ pid: handle.pid,
382
+ async wait() {
383
+ try {
384
+ const r = await handle.wait();
385
+ return { exitCode: r.exitCode ?? 0, stdout: r.stdout, stderr: r.stderr };
386
+ } catch (e) {
387
+ const ce = e as { exitCode?: unknown; stdout?: unknown; stderr?: unknown };
388
+ if (typeof ce.exitCode === "number") {
389
+ return {
390
+ exitCode: ce.exitCode,
391
+ stdout: typeof ce.stdout === "string" ? ce.stdout : "",
392
+ stderr: typeof ce.stderr === "string" ? ce.stderr : "",
393
+ };
394
+ }
395
+ throw e;
396
+ }
397
+ },
398
+ async kill() { await handle.kill(); },
399
+ };
400
+ }
401
+
366
402
  /** E2B base provider + live-filesystem snapshot. E2B's `createSnapshot()` captures
367
403
  * the running sandbox as a persistent snapshot whose id is usable as a
368
404
  * `Sandbox.create()` source and outlives the origin sandbox — so E2B reaches
369
405
  * snapshot / `bootFrom` parity with Vercel, and snapshot-backed `ctx.pause` works
370
406
  * on E2B. (Desktop intentionally omits this — it is registry-only, not selectable.) */
371
407
  function makeE2bSandboxProvider(sb: Sandbox): SandboxProvider {
408
+ const base = makeSandboxProvider(sb);
372
409
  return {
373
- ...makeSandboxProvider(sb),
410
+ ...base,
411
+ commands: {
412
+ ...base.commands,
413
+ // ADR-0028: background launch + reconnect-by-pid. The step runner runs as
414
+ // a background command so a server-driven pause can freeze it mid-turn
415
+ // (pauseProcess) and the resume activity can re-attach by pid and await
416
+ // its exit — continuing the SAME process, no re-run. E2B-only; Vercel's
417
+ // provider omits these and pauses via snapshot + re-run.
418
+ async runBackground(cmd, opts) {
419
+ const { sudo, onStdout, onStderr, ...rest } = (opts ?? {}) as SandboxCommandRunOptions;
420
+ const handle = await sb.commands.run(cmd, {
421
+ ...rest,
422
+ background: true,
423
+ ...(sudo ? { user: "root" } : {}),
424
+ ...(onStdout ? { onStdout } : {}),
425
+ ...(onStderr ? { onStderr } : {}),
426
+ });
427
+ return wrapE2bBackgroundProcess(handle);
428
+ },
429
+ async connectProcess(pid, opts) {
430
+ const handle = await sb.commands.connect(pid, {
431
+ ...(opts?.timeoutMs !== undefined ? { timeoutMs: opts.timeoutMs } : {}),
432
+ ...(opts?.onStdout ? { onStdout: opts.onStdout } : {}),
433
+ ...(opts?.onStderr ? { onStderr: opts.onStderr } : {}),
434
+ });
435
+ return wrapE2bBackgroundProcess(handle);
436
+ },
437
+ },
374
438
  async snapshot() {
375
439
  // Raw capture — the transient pause/reclaim race is retried centrally:
376
440
  // createSandbox/reconnectSandbox wrap every provider's snapshot() in
@@ -379,6 +443,14 @@ function makeE2bSandboxProvider(sb: Sandbox): SandboxProvider {
379
443
  const { snapshotId } = await sb.createSnapshot();
380
444
  return { snapshotId };
381
445
  },
446
+ // ADR-0027: native VM-suspend. Freezes the live process in place (zero
447
+ // compute) and returns the sandbox id as the resume handle — resume is
448
+ // `reconnectSandbox(...)` (Sandbox.connect), which auto-resumes a paused VM.
449
+ // Distinct from snapshot(): no FS image, no kill, no re-run from the top.
450
+ async pauseProcess() {
451
+ await sb.pause();
452
+ return { resumeHandle: sb.sandboxId };
453
+ },
382
454
  // Push a freshly-resolved egress policy onto the live sandbox via E2B's
383
455
  // native `updateNetwork` — the E2B analogue of Vercel's `update({
384
456
  // networkPolicy })`. Lets the server re-resolve the run policy (re-minting
@@ -814,6 +886,9 @@ export function makeLocalSandboxProvider(): SandboxProvider {
814
886
  await fs.mkdir(dirname(path), { recursive: true });
815
887
  await fs.writeFile(path, content);
816
888
  },
889
+ async read(path) {
890
+ return await fs.readFile(path, "utf8");
891
+ },
817
892
  },
818
893
  async kill() { /* caller IS the sandbox — killing it is the server's job */ },
819
894
  };
@@ -27,7 +27,8 @@ export {
27
27
  stepInputPath,
28
28
  requestContextPath,
29
29
  } from "./protocol.js";
30
- export { invokeStep, parseStepResult, buildStepEnvs } from "./invoker.js";
30
+ export { invokeStep, launchStep, reconnectStep, parseStepResult, buildStepEnvs } from "./invoker.js";
31
+ export type { RunningStep, InvokeStepOptions } from "./invoker.js";
31
32
  export { serveStep } from "./server.js";
32
33
  export type { StepHandler, ServeStepRequest, StepHandlerResult } from "./server.js";
33
34
  export { StepExecutionError } from "./types.js";