@agent-compose/sdk 0.6.0 → 0.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -71,29 +71,36 @@ Names like \`note.created\` / \`brief.posted\` surface in the Workbench;
71
71
 
72
72
  ## Writing workflow / agent code — the SDK
73
73
 
74
- \`@agent-compose/sdk\` is installed in \`/workspace\` — import it from any script
75
- you write there:
74
+ \`@agent-compose/sdk\` is installed in \`/workspace\`. **To author a workflow,
75
+ ALWAYS run \`/ac:generate-workflow\`** (and \`/ac:generate-agent\` for an agent
76
+ step) instead of writing source from memory — the skill scaffolds the correct,
77
+ current shape. Then \`agentc register <file.ts>\` (or \`/ac:register\`).
76
78
 
77
- import { defineWorkflow, agent, AgentComposeClient } from "@agent-compose/sdk";
79
+ The skill writes **step-form** (a builder of discrete, durable \`.step()\`s).
80
+ Never write the legacy run-form (\`defineWorkflow({ run(ctx, sandbox) { … } })\`):
81
+ it is one opaque step, so any failure or resume re-runs the whole body — and
82
+ **pause does not work in run-form**.
78
83
 
79
- Use \`/ac:generate-workflow\` / \`/ac:generate-agent\` to scaffold, then
80
- \`agentc register <file.ts>\` (or \`/ac:register\`).
84
+ ## Pausing to ask the human
81
85
 
82
- ## Pausing to ask the human — \`agentc pause\`
83
-
84
- When you can't or shouldn't proceed without a human, run \`agentc pause\`, then
85
- **END YOUR TURN**:
86
+ To ask a human and get an answer back, use the **\`AskUserQuestion\`** tool if
87
+ you have it; otherwise run **\`agentc pause\`**:
86
88
 
87
89
  agentc pause --reason "Notion returned 401 — connect Notion to continue" \\
88
90
  --option retry --option skip
89
91
 
90
- \`agentc pause\` does NOT block and does NOT print the answer. It records your
91
- question and returns immediately. The moment you end your turn, the run pauses
92
- (your sandbox is snapshotted and compute stops while the human decides) and the
93
- human's answer is delivered to you as your **next message** — you pick up
94
- exactly where you left off, with the answer in hand. So: ask, end your turn,
95
- and wait. Do NOT keep working, do NOT call more tools, and do NOT mark the task
96
- complete after pausing.
92
+ **Both BLOCK and hand you the answer inline.** While you wait, the run is
93
+ suspended — your sandbox is frozen and compute stops, so a pause is free while
94
+ the human decides. When they answer, the call RETURNS with their decision: the
95
+ \`AskUserQuestion\` tool result, or \`agentc pause\`'s output
96
+ (\`▶ Resumed. The human answered: …\`), carries it.
97
+
98
+ **Then USE that answer to finish your work — do NOT end your turn.** This is NOT
99
+ fire-and-forget, and the answer does NOT arrive in a later message: it comes
100
+ back right where you called it, on the SAME turn. The shape is: ask → the call
101
+ blocks → it returns the human's answer → you act on it and produce your result.
102
+ Never end your turn before the call returns, never guess an answer, and never
103
+ proceed without one.
97
104
 
98
105
  Reach for it the moment you hit — or foresee — any of these:
99
106
  - **A wall only a human can clear:** a 401/403, a missing credential, an
@@ -164,9 +164,16 @@ export async function agentLoop<TResponse = unknown>(opts: AgentLoopOpts<TRespon
164
164
  const logLabel = opts.label ?? "[Agent Loop]";
165
165
  const startedAt = Date.now();
166
166
  // No budget ⇒ no turn cap: the harness runtime (Claude Code) decides when it's
167
- // done. A numeric budget is an explicit caller choice, not a default we impose.
167
+ // done in one unbounded session. A numeric budget is an explicit caller choice.
168
+ // Default to a SMALL allowance (not 1) so the loop can RE-PROMPT for the closing
169
+ // <status>/<response> when a session ends EARLY — e.g. a human pause (agentc
170
+ // pause / AskUserQuestion) freezes the VM mid-tool and the resumed session can
171
+ // drop its final turn after doing the work. The re-prompt (PROTOCOL_SUFFIX)
172
+ // only fires when an iteration produced no exit_signal, so a clean run still
173
+ // settles in ONE iteration — these extra iterations are a recovery path, not
174
+ // the norm.
168
175
  const turnsPerIteration = opts.turnsPerIteration;
169
- const maxIterations = opts.maxIterations ?? (turnsPerIteration === undefined ? 1 : 8);
176
+ const maxIterations = opts.maxIterations ?? (turnsPerIteration === undefined ? 3 : 8);
170
177
  // A responseSchema is a CONTRACT, not a hope: when the agent's <response> fails
171
178
  // validation, the loop re-prompts with the exact errors until it conforms —
172
179
  // without consuming the caller's iteration budget. The backstop below only
@@ -14,6 +14,7 @@ import { z } from "zod";
14
14
  import { randomUUID } from "node:crypto";
15
15
  import { getActiveStep, nextAgentCallInActiveStep } from "../active-step.js";
16
16
  import { corePause, type PauseRequest } from "../pause/pause-core.js";
17
+ import { createAskHumanProcessor } from "../processors/ask-human.js";
17
18
  import { agentLoop } from "./agent-loop.js";
18
19
  import { consumeSteerPending, runControlPoller } from "./steer-control.js";
19
20
  import type { AgentLifecycleEvent, AgentLoopResult } from "./agent-loop.js";
@@ -330,6 +331,13 @@ export async function agent<T = unknown>(opts: AgentOpts<T>): Promise<AgentLoopR
330
331
  // (the boundary self-pause trigger + the prompt instruction above). Human
331
332
  // steering works regardless — see the ungated control poller.
332
333
  const mode = opts.mode ?? "auto";
334
+ // Ask-a-human is TOOLS-driven, not a separate mode: the ask-human processor is
335
+ // a chain default that maps the `AskUserQuestion` tool to a server pause (the
336
+ // run freezes, the human answers, the answer returns as the tool result). It's
337
+ // a no-op for any agent that doesn't have / call `AskUserQuestion`, so granting
338
+ // that tool to an agent IS its "can ask a human" switch — withhold it and the
339
+ // agent simply can't (it reports blockers up instead).
340
+ const processors = [createAskHumanProcessor(), ...(opts.processors ?? [])];
333
341
 
334
342
  try {
335
343
  return await agentLoop({
@@ -361,7 +369,7 @@ export async function agent<T = unknown>(opts: AgentOpts<T>): Promise<AgentLoopR
361
369
  ...(opts.events ? { onAgentLifecycleEvent: (event: AgentLifecycleEvent) => { void opts.events?.emit(event); } } : {}),
362
370
  ...(opts.onAgentEvent ? { onAgentEvent: opts.onAgentEvent } : {}),
363
371
  ...(opts.onIteration ? { onIteration: opts.onIteration } : {}),
364
- ...(opts.processors?.length ? { processors: opts.processors } : {}),
372
+ ...(processors.length ? { processors } : {}),
365
373
  requestContext: opts.requestContext ?? RequestContext.fromReserved({
366
374
  teamId: "", runId: "", workflowId: "",
367
375
  factoryId: null, apiKeyScopes: [], parentRunId: null,
package/src/index.ts CHANGED
@@ -77,6 +77,8 @@ export {
77
77
  requireScope,
78
78
  redactPattern,
79
79
  createGatePauseProcessor,
80
+ createAskHumanProcessor,
81
+ ASK_USER_QUESTION_TOOL,
80
82
  } from "./processors/index.js";
81
83
  export type {
82
84
  Processor,
@@ -174,6 +176,17 @@ export { default as codexRuntime } from "./runtimes/codex.js";
174
176
  export { createAmpRuntime } from "./runtimes/amp.js";
175
177
  export type { AmpRuntimeConfig } from "./runtimes/amp.js";
176
178
  export { default as ampRuntime } from "./runtimes/amp.js";
179
+ export { createOpencodeRuntime, opencodeSpec } from "./runtimes/opencode.js";
180
+ export type { OpencodeRuntimeConfig } from "./runtimes/opencode.js";
181
+ export { default as opencodeRuntime } from "./runtimes/opencode.js";
182
+
183
+ export { createCursorRuntime, cursorSpec } from "./runtimes/cursor.js";
184
+ export type { CursorRuntimeConfig } from "./runtimes/cursor.js";
185
+ export { default as cursorRuntime } from "./runtimes/cursor.js";
186
+
187
+ export { createDroidRuntime, droidSpec } from "./runtimes/droid.js";
188
+ export type { DroidRuntimeConfig } from "./runtimes/droid.js";
189
+ export { default as droidRuntime } from "./runtimes/droid.js";
177
190
 
178
191
  // Built-in coding tools for Vercel AI SDK runtime.
179
192
  export { bashTool, codingTools, editTool, readTool, writeTool } from "./tools/index.js";
@@ -0,0 +1,136 @@
1
+ /**
2
+ * Ask-human processor (ADR-0028).
3
+ *
4
+ * Makes "ask a human" a FIRST-CLASS agent affordance: when the agent calls the
5
+ * built-in `AskUserQuestion` tool, this short-circuits it into a SERVER pause
6
+ * (`requestPauseAndAwait`) — the run freezes (compute stops), the question +
7
+ * options land on the human's pause feed, and the human's answer comes back as
8
+ * the tool result. No `agentc pause` CLI for the model to remember, and no
9
+ * dependency on prompt discipline: the moment the agent asks, the run pauses.
10
+ *
11
+ * Loud by construction — the danger this fixes is a pause that SILENTLY doesn't
12
+ * happen (a stale in-sandbox CLI, a non-E2B substrate, an auth error) letting
13
+ * the agent proceed as if it had an answer:
14
+ * - pause cannot be created (server reject) → `Verdict.abort` ENDS the agent
15
+ * loop with a WorkflowError. The run fails loud; it never guesses an answer.
16
+ * - pause expires / is cancelled → the tool result says NO answer came and to
17
+ * not assume one.
18
+ *
19
+ * Lives in the shared `gateToolCall` chain, so one implementation covers every
20
+ * runtime (the Claude Agent SDK `PreToolUse` hook and the ACP permission path).
21
+ * No run credential in the env (local / non-sandbox) → no-op: `AskUserQuestion`
22
+ * passes through untouched so a dev invocation isn't hard-failed.
23
+ */
24
+
25
+ import type { Processor, ProcessorContext, ToolCall } from "./processor.js";
26
+ import { Verdict } from "./processor.js";
27
+ import { requestPauseAndAwait } from "../agent/pause-client.js";
28
+ import type { GatePauseConnection } from "./gate-pause.js";
29
+
30
+ /** The Claude built-in tool an agent uses to ask the user a question. */
31
+ export const ASK_USER_QUESTION_TOOL = "AskUserQuestion";
32
+
33
+ /** One question in an `AskUserQuestion` call (only the fields we read). */
34
+ interface AskQuestion { question?: unknown; header?: unknown; options?: unknown }
35
+
36
+ /** Pull the human-facing question + its option labels out of an
37
+ * `AskUserQuestion` tool input. We pause on the FIRST question (the common
38
+ * case); any others are folded into the reason so nothing is lost. */
39
+ function parseAsk(input: Record<string, unknown>): { reason: string; options: Array<{ id: string; label: string }> } {
40
+ const questions = Array.isArray(input.questions) ? (input.questions as AskQuestion[]) : [];
41
+ const first = questions[0] ?? {};
42
+ const head = typeof first.question === "string" && first.question.trim()
43
+ ? first.question.trim()
44
+ : "The agent needs your input to continue.";
45
+ const extra = questions.length > 1
46
+ ? ` (+${questions.length - 1} more question${questions.length > 2 ? "s" : ""})`
47
+ : "";
48
+ const options = Array.isArray(first.options)
49
+ ? (first.options as Array<{ label?: unknown }>)
50
+ .map((o) => String(o?.label ?? "").trim())
51
+ .filter(Boolean)
52
+ .map((label) => ({ id: label, label }))
53
+ : [];
54
+ return { reason: head + extra, options };
55
+ }
56
+
57
+ /** Unwrap the dashboard's `{ decision }` resume payload to the raw answer text. */
58
+ function answerText(decision: unknown): string {
59
+ const raw = decision !== null && typeof decision === "object" && "decision" in decision
60
+ ? (decision as { decision: unknown }).decision
61
+ : decision;
62
+ return typeof raw === "string" ? raw.trim() : raw == null ? "" : JSON.stringify(raw);
63
+ }
64
+
65
+ /** Recognise the agent shelling out to `agentc pause` and extract the same
66
+ * {reason, options} we'd get from AskUserQuestion. We INTERCEPT it here — in the
67
+ * tool gate, BEFORE the command runs — so the pause takes the DESIGNED path (the
68
+ * answer returns as the tool result and the agent loop continues) instead of the
69
+ * command actually blocking inside the sandbox shell, which froze the runner
70
+ * mid-tool-exec and never resumed the loop. */
71
+ function parseAgentcPause(command: string): { reason: string; options: Array<{ id: string; label: string }> } | null {
72
+ if (!/(^|\s|&&|;|\|)\s*agentc\s+pause(\s|$)/.test(command)) return null;
73
+ const r = command.match(/--reason(?:=|\s+)(?:"([^"]*)"|'([^']*)'|(\S+))/);
74
+ const reason = (r?.[1] ?? r?.[2] ?? r?.[3] ?? "The agent needs your input to continue.").trim();
75
+ const options = [...command.matchAll(/--option(?:=|\s+)(?:"([^"]*)"|'([^']*)'|(\S+))/g)]
76
+ .map((m) => (m[1] ?? m[2] ?? m[3] ?? "").trim())
77
+ .filter(Boolean)
78
+ .map((label) => ({ id: label, label }));
79
+ return { reason, options };
80
+ }
81
+
82
+ /** Pull the ask (reason + options) from either the `AskUserQuestion` tool OR a
83
+ * `Bash` call running `agentc pause`. Null for anything else. */
84
+ function extractAsk(call: ToolCall): { reason: string; options: Array<{ id: string; label: string }> } | null {
85
+ if (call.toolName === ASK_USER_QUESTION_TOOL) return parseAsk(call.toolInput);
86
+ if (call.toolName === "Bash") {
87
+ const cmd = (call.toolInput as { command?: unknown })?.command;
88
+ return typeof cmd === "string" ? parseAgentcPause(cmd) : null;
89
+ }
90
+ return null;
91
+ }
92
+
93
+ export function createAskHumanProcessor(opts: { connection?: GatePauseConnection } = {}): Processor {
94
+ return {
95
+ name: "ask-human",
96
+ async processToolCall(call: ToolCall, ctx: ProcessorContext) {
97
+ // Fires for AskUserQuestion OR a `Bash` call running `agentc pause` — both
98
+ // ask a human and must take the SAME processor path so the answer returns
99
+ // as the tool result and the agent loop continues.
100
+ const ask = extractAsk(call);
101
+ if (!ask) return Verdict.continue(call);
102
+
103
+ const conn = opts.connection ?? {
104
+ baseUrl: process.env.AGENT_COMPOSE_URL ?? "",
105
+ token: process.env.AGENT_COMPOSE_RUN_TOKEN ?? "",
106
+ runId: process.env.RUN_ID ?? "",
107
+ };
108
+ // No run credential (local / non-sandbox) — can't pause; let the tool
109
+ // through rather than hard-failing a dev invocation.
110
+ if (!conn.baseUrl || !conn.token || !conn.runId) return Verdict.continue(call);
111
+
112
+ let decision;
113
+ try {
114
+ decision = await requestPauseAndAwait({
115
+ baseUrl: conn.baseUrl, token: conn.token, runId: conn.runId,
116
+ reason: ask.reason,
117
+ ...(ask.options.length ? { options: ask.options } : {}),
118
+ action: { tool: call.toolName, input: call.toolInput },
119
+ signal: ctx.abortSignal,
120
+ });
121
+ } catch (err) {
122
+ // The pause could NOT be honored (server reject, non-E2B substrate, auth).
123
+ // Abort the loop — never let the agent proceed as if it had an answer.
124
+ const msg = err instanceof Error ? err.message : String(err);
125
+ return Verdict.abort(`Could not ask the human — the run could not be paused (${msg}). Stopping rather than guessing an answer.`);
126
+ }
127
+
128
+ if (decision.status === "resolved") {
129
+ const answer = answerText(decision.decision);
130
+ return Verdict.deny(`The human answered: ${answer || "(no text returned)"}. Continue using this answer.`);
131
+ }
132
+ // Expired / cancelled — no answer. Do NOT let the agent assume one.
133
+ return Verdict.deny(`No answer came back (${decision.status}). Do NOT assume an answer — ask again, or stop and report exactly what you need from a human.`);
134
+ },
135
+ };
136
+ }
@@ -20,3 +20,8 @@ export {
20
20
  // through the shared gateToolCall chain.
21
21
  export { createGatePauseProcessor } from "./gate-pause.js";
22
22
  export type { GatePausePolicy, GatePauseApproval, GatePauseConnection } from "./gate-pause.js";
23
+
24
+ // ADR-0028 — first-class "ask a human": maps the agent's `AskUserQuestion` tool
25
+ // to a server pause and returns the human's answer as the tool result. Added by
26
+ // default to every agent (no-op unless the tool is granted + called).
27
+ export { createAskHumanProcessor, ASK_USER_QUESTION_TOOL } from "./ask-human.js";
@@ -56,6 +56,16 @@ const ACP_FALLBACK = Symbol("acp-fallback");
56
56
  * ops tuning; defaults sane. */
57
57
  export const ACP_HANDSHAKE_TIMEOUT_MS = Number(process.env.AC_ACP_HANDSHAKE_TIMEOUT_MS) || 10_000;
58
58
 
59
+ /** Idle deadline for a PROMPT TURN (distinct from the handshake gate above). A
60
+ * turn is killed only if it goes fully SILENT for this long — the deadline is
61
+ * re-armed on every streamed message, so a long, *streaming* turn never trips
62
+ * it. This must be generous: a reasoning model (GLM, gpt-5-codex) can think for
63
+ * tens of seconds between tool calls with no wire activity, which is NOT a hang.
64
+ * The 10s handshake timeout was far too tight here and killed live GLM turns
65
+ * mid-report. Only a genuinely wedged CLI (the Gemini-style hang) should trip
66
+ * this. Overridable via the env for ops tuning. */
67
+ export const ACP_TURN_IDLE_TIMEOUT_MS = Number(process.env.AC_ACP_TURN_IDLE_TIMEOUT_MS) || 120_000;
68
+
59
69
  /** Readiness gate for the LIVE ACP attempt. The duplex-stdin transport in
60
70
  * `spawnAcpProcess` is now real (`commands.spawnDuplex` on the local provider),
61
71
  * so the agent's `initialize` request bytes are delivered and the handshake can
@@ -448,7 +458,7 @@ export class CliAgentRunner implements ModelExecutionContract {
448
458
  let watchdog: ReturnType<typeof setTimeout> | undefined;
449
459
  const armWatchdog = () => {
450
460
  if (watchdog) clearTimeout(watchdog);
451
- watchdog = setTimeout(() => { stalled = true; peer.cancel(); }, ACP_HANDSHAKE_TIMEOUT_MS);
461
+ watchdog = setTimeout(() => { stalled = true; peer.cancel(); }, ACP_TURN_IDLE_TIMEOUT_MS);
452
462
  };
453
463
  try {
454
464
  armWatchdog();
@@ -466,7 +476,7 @@ export class CliAgentRunner implements ModelExecutionContract {
466
476
  // and surface the stall as an error.
467
477
  await this.teardownAcp(proc, peer);
468
478
  this.acpSession = undefined;
469
- yield { type: "error", text: `${this.spec.kind} prompt turn stalled (no activity for ${ACP_HANDSHAKE_TIMEOUT_MS}ms)`, timestamp: now() };
479
+ yield { type: "error", text: `${this.spec.kind} prompt turn stalled (no activity for ${ACP_TURN_IDLE_TIMEOUT_MS}ms)`, timestamp: now() };
470
480
  return;
471
481
  }
472
482
 
@@ -0,0 +1,59 @@
1
+ /**
2
+ * Cursor CLI runtime — drives Cursor's `cursor-agent` inside the sandbox.
3
+ *
4
+ * ACP-native: `cursor-agent acp` is a protocolVersion-1 ACP server (verified
5
+ * live on E2B 2026-06-30), so the runner delegates the wire protocol to
6
+ * AcpClientPeer; the JSONL members below are the version-mismatch fallback.
7
+ *
8
+ * Auth: CURSOR_API_KEY — Cursor's OWN platform key, NOT OpenRouter. In ACP mode
9
+ * `cursor-agent acp` takes no model flag, so it runs the account's default model;
10
+ * the `--model` ids (auto, gpt-5.3-codex, composer-2.5,
11
+ * claude-opus-4-8-thinking-high; full list via `cursor-agent --list-models`)
12
+ * only apply to the JSONL `-p` fallback. cursor brings HARNESS diversity (a
13
+ * different agent scaffold) to a cross-functional / review panel.
14
+ *
15
+ * Verified live on E2B (2026-06-30): `curl https://cursor.com/install` →
16
+ * ~/.local/bin/cursor-agent (v2026.06.29); `cursor-agent acp` answered the ACP
17
+ * `initialize` with protocolVersion 1; CURSOR_API_KEY authenticated.
18
+ */
19
+ import type { AgentMessage } from "../index.js";
20
+ import { createCliAgentRuntime, shellQuote, type CliAgentSpec } from "./_cli-agent.js";
21
+
22
+ function now(): string { return new Date().toISOString(); }
23
+
24
+ export const cursorSpec: CliAgentSpec = {
25
+ kind: "cursor",
26
+ authEnv: "CURSOR_API_KEY",
27
+ bin: "cursor-agent",
28
+ // ACP runs the account default; "auto" is Cursor's own auto-routing label.
29
+ defaultModel: "auto",
30
+ // `cursor-agent acp` is a protocolVersion-1 ACP server — the runner delegates
31
+ // the whole wire protocol to AcpClientPeer. It takes no model flag, so the
32
+ // session runs Cursor's account-default model.
33
+ acp: { command: "cursor-agent", args: ["acp"] },
34
+ // Self-install on first use; symlink onto PATH for a non-login `sh -c`.
35
+ install: 'curl https://cursor.com/install -fsS | bash && (command -v cursor-agent >/dev/null 2>&1 || sudo ln -sf "$HOME/.local/bin/cursor-agent" /usr/local/bin/cursor-agent)',
36
+ // ── JSONL fallback (only if the ACP handshake negotiates a non-1 version;
37
+ // cursor is v1, so vestigial). `-p` needs `--force` to clear the
38
+ // workspace-trust gate non-interactively.
39
+ promptPayload: (prompt) => prompt,
40
+ buildCommand: ({ promptPath, model, cwd }) =>
41
+ `${cwd ? `cd ${shellQuote(cwd)} && ` : ""}cursor-agent -p --force ${model ? `--model ${shellQuote(model)} ` : ""}--output-format text "$(cat ${shellQuote(promptPath)})"`,
42
+ extractSessionId: () => undefined,
43
+ mapEvent: (p): AgentMessage[] => {
44
+ const ts = now();
45
+ const text = typeof p.text === "string" ? p.text : typeof p.content === "string" ? p.content : "";
46
+ return text ? [{ type: "text", text, timestamp: ts }] : [];
47
+ },
48
+ };
49
+
50
+ export interface CursorRuntimeConfig {
51
+ /** Cursor model id; ACP mode ignores it (account default) — applies to the `-p` fallback. */
52
+ model?: string;
53
+ }
54
+
55
+ export function createCursorRuntime(config: CursorRuntimeConfig = {}) {
56
+ return createCliAgentRuntime(cursorSpec, config.model ?? cursorSpec.defaultModel);
57
+ }
58
+
59
+ export default createCursorRuntime();
@@ -0,0 +1,63 @@
1
+ /**
2
+ * Factory `droid` runtime — drives `droid exec` headless inside the sandbox,
3
+ * JSONL via `--output-format json`.
4
+ *
5
+ * Driven PURELY via OpenRouter BYOK — NO Factory login (verified live on E2B
6
+ * 2026-06-30): a `~/.factory/settings.json` `customModels` entry points at
7
+ * OpenRouter, and the model id is `custom:<displayName>-<index>`. The workflow
8
+ * provisions settings.json (see the dynamic-task route step); this spec just
9
+ * builds the exec command. Auth env is OPENROUTER_API_KEY (the BYOK inference
10
+ * key shared with codex + opencode).
11
+ *
12
+ * NOT ACP: `droid exec` is JSONL. (Its stream-jsonrpc mode ignores --model and
13
+ * sets model/autonomy via JSON-RPC; the plain `--output-format json` mode honours
14
+ * --model, which is what we use.) So it always drives the JSONL path. droid adds
15
+ * Factory's agent HARNESS to a cross-functional / review panel.
16
+ *
17
+ * Verified live on E2B (2026-06-30): install via app.factory.ai/cli;
18
+ * `droid exec --auto low --model custom:GLM-5.2-OR-0 --output-format json`
19
+ * returned `{"type":"result","result":"…","session_id":"…","usage":{…}}` driven
20
+ * through OpenRouter with no Factory account.
21
+ */
22
+ import type { AgentMessage } from "../index.js";
23
+ import { createCliAgentRuntime, shellQuote, type CliAgentSpec } from "./_cli-agent.js";
24
+
25
+ function now(): string { return new Date().toISOString(); }
26
+
27
+ export const droidSpec: CliAgentSpec = {
28
+ kind: "droid",
29
+ // OpenRouter is the inference gateway (BYOK custom model). No FACTORY_API_KEY.
30
+ authEnv: "OPENROUTER_API_KEY",
31
+ bin: "droid",
32
+ // Matches the first customModels entry the workflow writes to settings.json.
33
+ defaultModel: "custom:GLM-5.2-OR-0",
34
+ install: 'curl -fsSL https://app.factory.ai/cli | sh && (command -v droid >/dev/null 2>&1 || sudo ln -sf "$HOME/.local/bin/droid" /usr/local/bin/droid)',
35
+ // ── JSONL path (always — droid exec is not ACP). `--auto medium` lets the
36
+ // agent create/edit files + run commands; `-f` reads the prompt from a file.
37
+ promptPayload: (prompt) => prompt,
38
+ buildCommand: ({ promptPath, model, cwd }) =>
39
+ `${cwd ? `cd ${shellQuote(cwd)} && ` : ""}droid exec --auto medium ${model ? `--model ${shellQuote(model)} ` : ""}--output-format json -f ${shellQuote(promptPath)}`,
40
+ extractSessionId: (p) => (typeof p.session_id === "string" ? p.session_id : undefined),
41
+ mapEvent: (p): AgentMessage[] => {
42
+ const ts = now();
43
+ // `--output-format json` emits a single terminal result object.
44
+ const text =
45
+ p.type === "result" && typeof p.result === "string" ? p.result
46
+ : typeof p.text === "string" ? p.text
47
+ : typeof p.content === "string" ? p.content
48
+ : "";
49
+ return text ? [{ type: "text", text, timestamp: ts }] : [];
50
+ },
51
+ // No `acp` → JSONL-only spec.
52
+ };
53
+
54
+ export interface DroidRuntimeConfig {
55
+ /** `custom:<displayName>-<index>` matching the provisioned settings.json. */
56
+ model?: string;
57
+ }
58
+
59
+ export function createDroidRuntime(config: DroidRuntimeConfig = {}) {
60
+ return createCliAgentRuntime(droidSpec, config.model ?? droidSpec.defaultModel);
61
+ }
62
+
63
+ export default createDroidRuntime();
@@ -0,0 +1,61 @@
1
+ /**
2
+ * OpenCode CLI runtime — drives sst's `opencode` agentic CLI inside the sandbox.
3
+ * OpenCode speaks ACP natively (`opencode acp`, protocolVersion 1 — verified
4
+ * live on E2B), so the runner drives it over ACP; the JSONL members below are
5
+ * only the version-mismatch fallback (vestigial for a v1 agent).
6
+ *
7
+ * Auth + model via OpenRouter: set OPENROUTER_API_KEY (a factory/workflow
8
+ * secret) and use a model id like `openrouter/z-ai/glm-5.2`. The runtime
9
+ * installs the `opencode-ai` CLI on demand; pair with
10
+ * `snapshots: { bootFrom: "reuse" }` to install once and boot from the capture.
11
+ *
12
+ * Verified live on E2B (2026-06-30): `npm i -g opencode-ai` (v1.17.12);
13
+ * `opencode run --model openrouter/z-ai/glm-5.2` drove a GLM-5.2 turn through
14
+ * OpenRouter; `opencode acp` answered the ACP `initialize` handshake with
15
+ * protocolVersion 1.
16
+ */
17
+
18
+ import type { AgentMessage } from "../index.js";
19
+ import { createCliAgentRuntime, shellQuote, type CliAgentSpec } from "./_cli-agent.js";
20
+
21
+ function now(): string { return new Date().toISOString(); }
22
+
23
+ export const opencodeSpec: CliAgentSpec = {
24
+ kind: "opencode",
25
+ // OpenRouter is the gateway: opencode reads OPENROUTER_API_KEY from the env
26
+ // and serves any `openrouter/<provider>/<model>` id (e.g. z-ai/glm-5.2).
27
+ authEnv: "OPENROUTER_API_KEY",
28
+ bin: "opencode",
29
+ defaultModel: "openrouter/z-ai/glm-5.2",
30
+ // ACP-mode invocation — `opencode acp` is a protocolVersion-1 ACP server, so
31
+ // the runner delegates the whole wire protocol to AcpClientPeer. The model is
32
+ // resolved from opencode's config / the `--model` it was started with; the
33
+ // run provisioning writes the OpenRouter default so ACP turns use GLM-5.2.
34
+ acp: { command: "opencode", args: ["acp"] },
35
+ // Global npm install; symlink onto PATH only if the global bin dir isn't
36
+ // already there (so a non-login `sh -c` can find it).
37
+ install: 'sudo npm install -g opencode-ai && (command -v opencode >/dev/null 2>&1 || sudo ln -sf "$(npm prefix -g)/bin/opencode" /usr/local/bin/opencode)',
38
+ // ── JSONL fallback (only reached if the ACP handshake negotiates a non-1
39
+ // version; opencode is v1, so this is vestigial). `opencode run` prints
40
+ // human-formatted text, so we capture the prompt round-trip as one message.
41
+ promptPayload: (prompt) => prompt,
42
+ buildCommand: ({ promptPath, model, cwd }) =>
43
+ `${cwd ? `cd ${shellQuote(cwd)} && ` : ""}opencode run ${model ? `--model ${shellQuote(model)} ` : ""}"$(cat ${shellQuote(promptPath)})"`,
44
+ extractSessionId: () => undefined,
45
+ mapEvent: (p): AgentMessage[] => {
46
+ const ts = now();
47
+ const text = typeof p.text === "string" ? p.text : typeof p.content === "string" ? p.content : "";
48
+ return text ? [{ type: "text", text, timestamp: ts }] : [];
49
+ },
50
+ };
51
+
52
+ export interface OpencodeRuntimeConfig {
53
+ /** OpenRouter-prefixed model id (default `openrouter/z-ai/glm-5.2`). */
54
+ model?: string;
55
+ }
56
+
57
+ export function createOpencodeRuntime(config: OpencodeRuntimeConfig = {}) {
58
+ return createCliAgentRuntime(opencodeSpec, config.model ?? opencodeSpec.defaultModel);
59
+ }
60
+
61
+ export default createOpencodeRuntime();
package/src/sandbox.ts CHANGED
@@ -301,6 +301,15 @@ export function makeSandboxProvider(sb: Sandbox | Desktop): SandboxProvider {
301
301
  async write(path, content) {
302
302
  await (sb.files.write as (p: string, d: string) => Promise<unknown>)(path, content);
303
303
  },
304
+ // Read over the envd HTTP API (`Sandbox.files.read` → `GET /files`), a
305
+ // DIFFERENT transport from `commands` (the connect-web gRPC stream). A large
306
+ // readback over HTTP is decoded by `fetch`'s native `Content-Encoding`
307
+ // handling, so it is immune to the connect-web "received unsupported
308
+ // compressed output" failure that can abort `commands.run` output on a big
309
+ // frame — the property `launchStep` relies on to recover full logs.
310
+ async read(path) {
311
+ return await (sb.files.read as (p: string) => Promise<string>)(path);
312
+ },
304
313
  },
305
314
  // e2b 2.30 `kill()` returns Promise<boolean>; our provider contract is
306
315
  // Promise<void>, so discard the result.
@@ -877,6 +886,9 @@ export function makeLocalSandboxProvider(): SandboxProvider {
877
886
  await fs.mkdir(dirname(path), { recursive: true });
878
887
  await fs.writeFile(path, content);
879
888
  },
889
+ async read(path) {
890
+ return await fs.readFile(path, "utf8");
891
+ },
880
892
  },
881
893
  async kill() { /* caller IS the sandbox — killing it is the server's job */ },
882
894
  };