@agent-compose/sdk 0.5.9 → 0.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (49) hide show
  1. package/dist/agent/agent-context.d.ts +1 -1
  2. package/dist/agent/agent-loop.d.ts +0 -8
  3. package/dist/agent/pause-client.d.ts +50 -0
  4. package/dist/index.d.ts +15 -6
  5. package/dist/index.js +537 -124
  6. package/dist/processors/ask-human.d.ts +30 -0
  7. package/dist/processors/ask-human.test.d.ts +1 -0
  8. package/dist/processors/gate-pause.d.ts +46 -0
  9. package/dist/processors/gate-pause.test.d.ts +1 -0
  10. package/dist/processors/index.d.ts +3 -0
  11. package/dist/runtimes/_cli-agent.d.ts +9 -0
  12. package/dist/runtimes/cursor.d.ts +9 -0
  13. package/dist/runtimes/droid.d.ts +9 -0
  14. package/dist/runtimes/openai-desktop.js +522 -122
  15. package/dist/runtimes/opencode.d.ts +25 -0
  16. package/dist/runtimes/vercel.js +11 -1
  17. package/dist/step-invocation/__tests__/background-invoker.test.d.ts +1 -0
  18. package/dist/step-invocation/index.d.ts +2 -1
  19. package/dist/step-invocation/invoker.d.ts +49 -0
  20. package/dist/step-invocation/protocol.d.ts +8 -0
  21. package/dist/types/runtime.d.ts +7 -0
  22. package/dist/types/sandbox.d.ts +54 -0
  23. package/dist/types/workflow.d.ts +12 -0
  24. package/dist/utils/errors.d.ts +9 -1
  25. package/package.json +1 -1
  26. package/src/agent/agent-context.ts +23 -16
  27. package/src/agent/agent-loop.ts +13 -33
  28. package/src/agent/pause-client.ts +108 -0
  29. package/src/agent/run-agent.ts +16 -7
  30. package/src/index.ts +27 -4
  31. package/src/processors/ask-human.ts +136 -0
  32. package/src/processors/gate-pause.ts +94 -0
  33. package/src/processors/index.ts +11 -0
  34. package/src/runtimes/_cli-agent.ts +13 -5
  35. package/src/runtimes/claude.ts +10 -6
  36. package/src/runtimes/cursor.ts +59 -0
  37. package/src/runtimes/droid.ts +63 -0
  38. package/src/runtimes/opencode.ts +61 -0
  39. package/src/sandbox.ts +78 -3
  40. package/src/step-invocation/index.ts +2 -1
  41. package/src/step-invocation/invoker.ts +359 -86
  42. package/src/step-invocation/protocol.ts +11 -0
  43. package/src/types/runtime.ts +7 -0
  44. package/src/types/sandbox.ts +53 -0
  45. package/src/types/workflow.ts +12 -0
  46. package/src/utils/errors.ts +19 -2
  47. package/dist/agent/local-pause-request.d.ts +0 -49
  48. package/src/agent/local-pause-request.ts +0 -90
  49. /package/dist/agent/{local-pause-request.test.d.ts → pause-client.test.d.ts} +0 -0
@@ -0,0 +1,25 @@
1
+ /**
2
+ * OpenCode CLI runtime — drives sst's `opencode` agentic CLI inside the sandbox.
3
+ * OpenCode speaks ACP natively (`opencode acp`, protocolVersion 1 — verified
4
+ * live on E2B), so the runner drives it over ACP; the JSONL members below are
5
+ * only the version-mismatch fallback (vestigial for a v1 agent).
6
+ *
7
+ * Auth + model via OpenRouter: set OPENROUTER_API_KEY (a factory/workflow
8
+ * secret) and use a model id like `openrouter/z-ai/glm-5.2`. The runtime
9
+ * installs the `opencode-ai` CLI on demand; pair with
10
+ * `snapshots: { bootFrom: "reuse" }` to install once and boot from the capture.
11
+ *
12
+ * Verified live on E2B (2026-06-30): `npm i -g opencode-ai` (v1.17.12);
13
+ * `opencode run --model openrouter/z-ai/glm-5.2` drove a GLM-5.2 turn through
14
+ * OpenRouter; `opencode acp` answered the ACP `initialize` handshake with
15
+ * protocolVersion 1.
16
+ */
17
+ import { type CliAgentSpec } from "./_cli-agent.js";
18
+ export declare const opencodeSpec: CliAgentSpec;
19
+ export interface OpencodeRuntimeConfig {
20
+ /** OpenRouter-prefixed model id (default `openrouter/z-ai/glm-5.2`). */
21
+ model?: string;
22
+ }
23
+ export declare function createOpencodeRuntime(config?: OpencodeRuntimeConfig): import("../index.js").AgentRuntime<import("../index.js").SandboxProvider>;
24
+ declare const _default: import("../index.js").AgentRuntime<import("../index.js").SandboxProvider>;
25
+ export default _default;
@@ -717,7 +717,17 @@ async function corePause(req, coord, kind = "custom") {
717
717
 
718
718
  // src/utils/errors.ts
719
719
  function formatError(err) {
720
- return err instanceof Error ? err.message : String(err);
720
+ if (!(err instanceof Error))
721
+ return String(err);
722
+ const parts = [err.message];
723
+ const seen = new Set([err]);
724
+ let cause = err.cause;
725
+ while (cause != null && !seen.has(cause)) {
726
+ seen.add(cause);
727
+ parts.push(cause instanceof Error ? cause.message : String(cause));
728
+ cause = cause instanceof Error ? cause.cause : undefined;
729
+ }
730
+ return parts.join(": ");
721
731
  }
722
732
 
723
733
  // src/runtimes/vercel.ts
@@ -18,7 +18,8 @@
18
18
  * to test in isolation.
19
19
  */
20
20
  export { STEP_RESULT_PREFIX, STEP_PAUSE_PREFIX, STEP_ENV, RUNNER_BUNDLE_PATH, RUNNER_COMMAND, stepInputPath, requestContextPath, } from "./protocol.js";
21
- export { invokeStep, parseStepResult, buildStepEnvs } from "./invoker.js";
21
+ export { invokeStep, launchStep, reconnectStep, parseStepResult, buildStepEnvs } from "./invoker.js";
22
+ export type { RunningStep, InvokeStepOptions } from "./invoker.js";
22
23
  export { serveStep } from "./server.js";
23
24
  export type { StepHandler, ServeStepRequest, StepHandlerResult } from "./server.js";
24
25
  export { StepExecutionError } from "./types.js";
@@ -64,5 +64,54 @@ export interface InvokeStepOptions {
64
64
  * the caller's job; the activity batches and inserts at step completion. */
65
65
  onStdout?: (line: string) => void;
66
66
  onStderr?: (line: string) => void;
67
+ /** Called when the LIVE output stream fails mid-run (e.g. E2B's connect-web
68
+ * transport throws `received unsupported compressed output` on a large
69
+ * compressed frame) and the invoker falls back to recovering the result +
70
+ * full logs from the durable token-keyed files. The run is NOT failed — this
71
+ * is the hook to emit a structured alert so the degradation is visible/paged.
72
+ * Best-effort: keep it cheap and non-throwing. */
73
+ onStreamDegraded?: (info: {
74
+ error: unknown;
75
+ runnerPid: number;
76
+ resultToken: string;
77
+ }) => void;
78
+ }
79
+ /** A step running as a background command (ADR-0028). The activity races its
80
+ * `wait()` against a server pause request; on a pause it freezes the VM
81
+ * (`pauseProcess`) and persists `runnerPid` + `resultToken` so the resume
82
+ * activity can `reconnectStep(...)` to the SAME process — no re-run. */
83
+ export interface RunningStep<TOutput = unknown> {
84
+ /** OS pid of the background runner inside the VM — the reconnect handle. */
85
+ runnerPid: number;
86
+ /** Per-invocation token keying the durable result file the runner writes. */
87
+ resultToken: string;
88
+ /** Await the runner's exit and classify its output into a `StepResult`. */
89
+ wait(): Promise<StepResult<TOutput>>;
67
90
  }
68
91
  export declare function invokeStep<TOutput = unknown>(sandbox: SandboxProvider, request: StepRequest, opts?: InvokeStepOptions): Promise<StepResult<TOutput>>;
92
+ /**
93
+ * Launch the step runner as a BACKGROUND command (ADR-0028) and return a handle
94
+ * the activity drives: it races `wait()` against a server pause request and, on
95
+ * a pause, freezes the VM (`pauseProcess`) and persists `runnerPid` +
96
+ * `resultToken` so `reconnectStep` can continue the SAME process — no re-run.
97
+ * Requires a provider with `commands.runBackground` (E2B); the foreground
98
+ * `invokeStep` is the path for providers without it (Vercel).
99
+ */
100
+ export declare function launchStep<TOutput = unknown>(sandbox: SandboxProvider, request: StepRequest, opts?: InvokeStepOptions): Promise<RunningStep<TOutput>>;
101
+ /**
102
+ * Resume a previously-paused background runner (ADR-0028) and return a
103
+ * `RunningStep` handle — uniform with `launchStep` so the activity can race
104
+ * `wait()` against a fresh pause request (a resumed step can pause again). After
105
+ * the workflow reconnects the suspended VM (`Sandbox.connect` auto-resumes it),
106
+ * this re-attaches to the still-running runner by `runnerPid`; `wait()` awaits
107
+ * its exit (event-driven — no polling) and classifies the output (the durable
108
+ * result file is authoritative on this path). If the runner already exited
109
+ * during resume, the re-attach fails and `wait()` classifies from the file.
110
+ */
111
+ export declare function reconnectStep<TOutput = unknown>(sandbox: SandboxProvider, resume: {
112
+ runnerPid: number;
113
+ resultToken: string;
114
+ stepIndex: number;
115
+ }, opts?: Pick<InvokeStepOptions, "onStdout" | "onStderr"> & {
116
+ signal?: AbortSignal;
117
+ }): Promise<RunningStep<TOutput>>;
@@ -60,3 +60,11 @@ export declare function requestContextPath(stepIndex: number): string;
60
60
  * drop the tail of a heavy stdout stream — the invoker falls back to
61
61
  * reading this file when no sentinel is found on stdout. */
62
62
  export declare function stepResultFilePath(token: string): string;
63
+ /** Sandbox-side path where the runner's stdout is tee'd as a durable LOG file,
64
+ * keyed by the per-invocation token. The live output rides E2B's connect-web
65
+ * command stream, which THROWS on a compressed large frame (gRPC-web cannot
66
+ * decode message compression) and kills the feed mid-run. This file, read back
67
+ * over the envd HTTP file transport (compression-immune), lets the invoker
68
+ * recover the FULL logs after such a fault instead of losing the tail — the
69
+ * log-side analogue of `stepResultFilePath` for the result. */
70
+ export declare function stepLogFilePath(token: string): string;
@@ -29,6 +29,13 @@ export interface RuntimeOptions {
29
29
  /** Agent id and label for processor context / adapter logs. */
30
30
  agentId?: string;
31
31
  iteration?: number;
32
+ /** The platform manual (file conventions, connectors & access, how to pause).
33
+ * `agent()` builds it per-run (`buildAgentContextDoc`) and threads it here so
34
+ * a runtime that supports a system-prompt append (the claude runtime) injects
35
+ * it directly — instead of relying on the agent to `cat` the on-disk
36
+ * AGENTS.md/CLAUDE.md, which the Agent SDK doesn't auto-load and which can
37
+ * fail to write on a read-only/degraded working dir. */
38
+ agentManual?: string;
32
39
  /** The run's pause boundary, threaded from the agent loop so a runtime-driven
33
40
  * pre-tool gate (e.g. the ACP `session/request_permission` path through
34
41
  * `gateToolCall`) can raise a human-approval `ctx.pause`. The runtime binds it
@@ -21,6 +21,29 @@ export interface SandboxCommandResult {
21
21
  stdout: string;
22
22
  stderr: string;
23
23
  }
24
+ /** A long-running command launched in the background (ADR-0028). Unlike
25
+ * `commands.run` (which awaits completion on one connection), a background
26
+ * command keeps running inside the VM independent of the launching
27
+ * connection: it survives `pauseProcess()`/resume and is re-attachable by
28
+ * `pid` after a fresh `Sandbox.connect`. This is what lets the platform
29
+ * freeze an agent mid-turn for a human-in-the-loop pause and continue the
30
+ * SAME process on resume — no re-run.
31
+ *
32
+ * Implemented ONLY by process-resume-capable providers (E2B); the presence
33
+ * of `commands.runBackground` IS the capability flag, paired with
34
+ * `pauseProcess`. Providers without it leave both undefined and pause via
35
+ * `snapshot()` + re-run instead. */
36
+ export interface SandboxBackgroundProcess {
37
+ /** OS pid inside the VM — the durable handle used to reconnect after a
38
+ * pause/resume cycle via `commands.connectProcess(pid)`. */
39
+ pid: number;
40
+ /** Resolve when the process exits, with its buffered result. Live output
41
+ * streams to the `onStdout`/`onStderr` passed at launch / connect time.
42
+ * Does NOT throw on a non-zero exit — the result carries `exitCode`. */
43
+ wait(): Promise<SandboxCommandResult>;
44
+ /** Force-terminate the process. */
45
+ kill(): Promise<void>;
46
+ }
24
47
  /** A spawned long-lived command with a writable stdin and readable stdout,
25
48
  * exposed as byte web-streams. Unlike `commands.run` (which buffers to
26
49
  * completion and exposes stdout only via an `onStdout` callback), a duplex
@@ -78,9 +101,28 @@ export interface SandboxProvider {
78
101
  * view). The vercel/e2b providers (server→sandbox) leave it undefined; an
79
102
  * ACP caller that finds it absent falls back to the JSONL transport. */
80
103
  spawnDuplex?(cmd: string, opts?: SandboxSpawnDuplexOptions): SandboxDuplexProcess;
104
+ /** Launch a command in the background and return immediately with a
105
+ * reconnectable handle (ADR-0028). The process survives the launching
106
+ * connection dropping AND a `pauseProcess()`/resume cycle. OPTIONAL —
107
+ * only process-resume providers (E2B) implement it; its presence (paired
108
+ * with `pauseProcess`) is the native-pause capability flag. */
109
+ runBackground?(cmd: string, opts?: SandboxCommandRunOptions): Promise<SandboxBackgroundProcess>;
110
+ /** Re-attach to a background command by `pid` after a fresh
111
+ * `Sandbox.connect` (the resume half of `runBackground`). OPTIONAL,
112
+ * E2B-only. Throws if no process with that pid is running. */
113
+ connectProcess?(pid: number, opts?: Pick<SandboxCommandRunOptions, "onStdout" | "onStderr" | "timeoutMs">): Promise<SandboxBackgroundProcess>;
81
114
  };
82
115
  files: {
83
116
  write(path: string, content: string): Promise<void>;
117
+ /** Read a file's text content over the provider's FILE transport. On E2B this
118
+ * is the envd HTTP API (`Sandbox.files.read`), a DIFFERENT transport from
119
+ * `commands` — so a large readback is immune to the connect-web gRPC
120
+ * message-compression that can abort `commands.run` output on a big frame
121
+ * ("received unsupported compressed output"). This is what lets `launchStep`
122
+ * recover the full logs + result after a live-stream fault. OPTIONAL —
123
+ * implemented where durable file-readback is needed (E2B, local); providers
124
+ * that never drive the recovery path (Vercel — foreground only) may omit it. */
125
+ read?(path: string): Promise<string>;
84
126
  };
85
127
  kill(): Promise<void>;
86
128
  /** Capture the running sandbox's state as a reusable snapshot. Vercel and E2B
@@ -95,6 +137,18 @@ export interface SandboxProvider {
95
137
  snapshotId: string;
96
138
  sizeBytes?: number;
97
139
  }>;
140
+ /** Suspend the live VM in place and return a handle to resume it (ADR-0027).
141
+ * Present ONLY on process-resume-capable providers (E2B via `sandbox.pause()`,
142
+ * returning the sandbox id; resume is `Sandbox.connect(handle)`, which
143
+ * auto-resumes the paused VM). Unlike `snapshot()` — which captures an FS
144
+ * image, kills the origin, and re-runs the step from a fresh sandbox — a
145
+ * process-resume pause FREEZES the live process (zero compute) and continues
146
+ * it exactly where it blocked. The presence of this method IS the capability
147
+ * flag: providers without native VM-suspend leave it undefined and fall back
148
+ * to `snapshot()` + re-run. */
149
+ pauseProcess?(): Promise<{
150
+ resumeHandle: string;
151
+ }>;
98
152
  /** Replace the live sandbox's egress policy in place — so the server can
99
153
  * push a freshly resolved policy (with re-minted connector access tokens)
100
154
  * before each step instead of relying on the policy baked at create.
@@ -90,6 +90,12 @@ export interface WorkflowDefinition<TOutput = unknown, TInput extends Record<str
90
90
  /** Same as `input`, for the workflow's return value. Captured into
91
91
  * `outputSchema` metadata and rendered in the IO panel. */
92
92
  output?: z.ZodType<TOutput>;
93
+ /**
94
+ * @deprecated Legacy run-form. Prefer step-form — the
95
+ * `.step(defineStep(...))` builder — for per-step durability/replay and
96
+ * working pause. A run-form body compiles to one opaque step
97
+ * (`compileRunForm`), so any failure/resume re-runs the whole body.
98
+ */
93
99
  run: WorkflowFn<TOutput, TInput>;
94
100
  /**
95
101
  * All snapshot config — boot source plus capture mode.
@@ -210,6 +216,12 @@ export interface WorkflowDefinition<TOutput = unknown, TInput extends Record<str
210
216
  * stores the step plan; runner subprocesses execute one step at a time
211
217
  * via the StepInvocation seam.
212
218
  */
219
+ /**
220
+ * @deprecated Run-form is legacy. Use the step-form overload —
221
+ * `defineWorkflow({ id, input, output }).step(defineStep(...)).build()` — for
222
+ * durable, replayable steps and working pause. Run-form compiles to a single
223
+ * opaque step (`compileRunForm`); there is no per-step replay.
224
+ */
213
225
  export declare function defineWorkflow<TOutput = unknown, TInput extends Record<string, unknown> = Record<string, unknown>>(def: WorkflowDefinition<TOutput, TInput>): Workflow<TInput, TOutput>;
214
226
  export declare function defineWorkflow<TInput, TOutput>(def: StepWorkflowDefinition<TInput, TOutput>): import("../workflow-steps/workflow.js").WorkflowBuilder<TInput, TInput>;
215
227
  /** Observability-only lifecycle hooks — passed to the workflow engine, not workflow authors. */
@@ -1,2 +1,10 @@
1
- /** Convert any thrown value to a string message. */
1
+ /** Convert any thrown value to a string message, INCLUDING its `.cause` chain.
2
+ *
3
+ * Many wrapped errors carry the real reason on `.cause` and only a generic
4
+ * summary on `.message` — the Temporal SDK's "Failed to start Workflow" is the
5
+ * canonical example (its `.cause` is the actual gRPC rejection, e.g. "search
6
+ * attribute X is not defined"). Returning `.message` alone swallowed that, so
7
+ * failures surfaced as opaque one-liners. Walk the chain and join the messages
8
+ * so the root cause is always visible. Cycle-guarded against self-referential
9
+ * `cause` links. */
2
10
  export declare function formatError(err: unknown): string;
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@agent-compose/sdk",
3
- "version": "0.5.9",
3
+ "version": "0.7.0",
4
4
  "description": "Client library for agent-compose — define agents, runtimes, and workflows, and invoke them against an agent-compose server.",
5
5
  "license": "MIT",
6
6
  "repository": {
@@ -71,29 +71,36 @@ Names like \`note.created\` / \`brief.posted\` surface in the Workbench;
71
71
 
72
72
  ## Writing workflow / agent code — the SDK
73
73
 
74
- \`@agent-compose/sdk\` is installed in \`/workspace\` — import it from any script
75
- you write there:
74
+ \`@agent-compose/sdk\` is installed in \`/workspace\`. **To author a workflow,
75
+ ALWAYS run \`/ac:generate-workflow\`** (and \`/ac:generate-agent\` for an agent
76
+ step) instead of writing source from memory — the skill scaffolds the correct,
77
+ current shape. Then \`agentc register <file.ts>\` (or \`/ac:register\`).
76
78
 
77
- import { defineWorkflow, agent, AgentComposeClient } from "@agent-compose/sdk";
79
+ The skill writes **step-form** (a builder of discrete, durable \`.step()\`s).
80
+ Never write the legacy run-form (\`defineWorkflow({ run(ctx, sandbox) { … } })\`):
81
+ it is one opaque step, so any failure or resume re-runs the whole body — and
82
+ **pause does not work in run-form**.
78
83
 
79
- Use \`/ac:generate-workflow\` / \`/ac:generate-agent\` to scaffold, then
80
- \`agentc register <file.ts>\` (or \`/ac:register\`).
84
+ ## Pausing to ask the human
81
85
 
82
- ## Pausing to ask the human — \`agentc pause\`
83
-
84
- When you can't or shouldn't proceed without a human, run \`agentc pause\`, then
85
- **END YOUR TURN**:
86
+ To ask a human and get an answer back, use the **\`AskUserQuestion\`** tool if
87
+ you have it; otherwise run **\`agentc pause\`**:
86
88
 
87
89
  agentc pause --reason "Notion returned 401 — connect Notion to continue" \\
88
90
  --option retry --option skip
89
91
 
90
- \`agentc pause\` does NOT block and does NOT print the answer. It records your
91
- question and returns immediately. The moment you end your turn, the run pauses
92
- (your sandbox is snapshotted and compute stops while the human decides) and the
93
- human's answer is delivered to you as your **next message** — you pick up
94
- exactly where you left off, with the answer in hand. So: ask, end your turn,
95
- and wait. Do NOT keep working, do NOT call more tools, and do NOT mark the task
96
- complete after pausing.
92
+ **Both BLOCK and hand you the answer inline.** While you wait, the run is
93
+ suspended — your sandbox is frozen and compute stops, so a pause is free while
94
+ the human decides. When they answer, the call RETURNS with their decision: the
95
+ \`AskUserQuestion\` tool result, or \`agentc pause\`'s output
96
+ (\`▶ Resumed. The human answered: …\`), carries it.
97
+
98
+ **Then USE that answer to finish your work — do NOT end your turn.** This is NOT
99
+ fire-and-forget, and the answer does NOT arrive in a later message: it comes
100
+ back right where you called it, on the SAME turn. The shape is: ask → the call
101
+ blocks → it returns the human's answer → you act on it and produce your result.
102
+ Never end your turn before the call returns, never guess an answer, and never
103
+ proceed without one.
97
104
 
98
105
  Reach for it the moment you hit — or foresee — any of these:
99
106
  - **A wall only a human can clear:** a 401/403, a missing credential, an
@@ -15,7 +15,6 @@ import { RequestContext } from "../request-context/request-context.js";
15
15
  import { PauseManager } from "../pause/manager.js";
16
16
  import { PauseSignal, isPauseSignal } from "../pause/pause-core.js";
17
17
  import { SteerDecisionSchema, type SteerDecision, type SteerPayload } from "./steer-control.js";
18
- import { type LocalPauseRequest } from "./local-pause-request.js";
19
18
 
20
19
  export const DEFAULT_CLAUDE_MODEL = "claude-fable-5";
21
20
 
@@ -154,13 +153,6 @@ export interface AgentLoopOpts<TResponse = unknown> {
154
153
  * The boundary consumes it once per check, and only when no steer is
155
154
  * already staged. */
156
155
  consumeSteerPending?: () => SteerPayload | null;
157
- /** Take-once read of a local `agentc pause` request this agent dropped in the
158
- * state dir during its turn (the CLI writes it; see local-pause-request.ts).
159
- * Returns the request (reason + offered options) or null. The loop turns it
160
- * into a staged self-pause the next boundary takes — the same snapshot-release
161
- * path as `needs_input`, but triggered by an explicit `agentc pause` call so
162
- * it is honored regardless of `mode`. */
163
- consumeLocalPauseRequest?: () => LocalPauseRequest | null;
164
156
  /** PR 7: `auto` (autonomous, default) or `hitl` (can be steer-paused). The
165
157
  * boundary is inert unless `hitl`. */
166
158
  mode?: "auto" | "hitl";
@@ -172,9 +164,16 @@ export async function agentLoop<TResponse = unknown>(opts: AgentLoopOpts<TRespon
172
164
  const logLabel = opts.label ?? "[Agent Loop]";
173
165
  const startedAt = Date.now();
174
166
  // No budget ⇒ no turn cap: the harness runtime (Claude Code) decides when it's
175
- // done. A numeric budget is an explicit caller choice, not a default we impose.
167
+ // done in one unbounded session. A numeric budget is an explicit caller choice.
168
+ // Default to a SMALL allowance (not 1) so the loop can RE-PROMPT for the closing
169
+ // <status>/<response> when a session ends EARLY — e.g. a human pause (agentc
170
+ // pause / AskUserQuestion) freezes the VM mid-tool and the resumed session can
171
+ // drop its final turn after doing the work. The re-prompt (PROTOCOL_SUFFIX)
172
+ // only fires when an iteration produced no exit_signal, so a clean run still
173
+ // settles in ONE iteration — these extra iterations are a recovery path, not
174
+ // the norm.
176
175
  const turnsPerIteration = opts.turnsPerIteration;
177
- const maxIterations = opts.maxIterations ?? (turnsPerIteration === undefined ? 1 : 8);
176
+ const maxIterations = opts.maxIterations ?? (turnsPerIteration === undefined ? 3 : 8);
178
177
  // A responseSchema is a CONTRACT, not a hope: when the agent's <response> fails
179
178
  // validation, the loop re-prompts with the exact errors until it conforms —
180
179
  // without consuming the caller's iteration budget. The backstop below only
@@ -462,29 +461,10 @@ export async function agentLoop<TResponse = unknown>(opts: AgentLoopOpts<TRespon
462
461
 
463
462
  let status = parseAgentStatus(responseText);
464
463
 
465
- // `agentc pause` — the agent shelled out to the CLI during this turn, which
466
- // dropped a durable pause-request marker in the state dir. Honor it like a
467
- // self-pause: stage the pause the NEXT boundary takes, carrying the agent's
468
- // question as the reason and any offered choices as the pause payload (the
469
- // dashboard renders them as buttons). Checked BEFORE the settle / needs_input
470
- // paths and independent of the <status> block — the common case is an agent
471
- // that called the tool and ended its turn with no status at all, which would
472
- // otherwise fall through to the empty-output re-prompt below. UNGATED by
473
- // `mode`: an explicit `agentc pause` is a deliberate ask, not the `needs_input`
474
- // heuristic that only `hitl` agents may trigger.
475
- if (opts.pause && pendingSteerPause === null) {
476
- const localPause = opts.consumeLocalPauseRequest?.() ?? null;
477
- if (localPause) {
478
- pendingSteerPause = {
479
- reason: localPause.reason,
480
- correlationKey: null,
481
- ...(localPause.options && localPause.options.length > 0 ? { payload: { options: localPause.options } } : {}),
482
- at: Date.now(),
483
- };
484
- blockerStreak = null; // an explicit ask is not a stuck loop
485
- continue; // pause fires at the next boundary
486
- }
487
- }
464
+ // ADR-0028 — `agentc pause` is no longer a marker the loop consumes here. It
465
+ // is a server operation: the CLI blocks on the pause API and the run is
466
+ // frozen in place by the step activity, with no loop involvement. The steer
467
+ // (human-driven) and `needs_input` (self-pause) paths below are unaffected.
488
468
 
489
469
  // Inline safeParse (instead of letting parseAgentResponse validate)
490
470
  // so a schema failure surfaces via lastResponseValidationError on
@@ -0,0 +1,108 @@
1
+ /**
2
+ * In-sandbox pause client (ADR-0028). The single way an agent pauses its OWN
3
+ * run: `agentc pause` and the gate-pause processor both call
4
+ * `requestPauseAndAwait`, which drives the server pause API and blocks until a
5
+ * human resolves it.
6
+ *
7
+ * Create-then-poll, NOT a held connection:
8
+ * 1. POST /pauses creates the pending row. The server wakes the step
9
+ * activity, which freezes the live VM in place (E2B native suspend). The
10
+ * poll loop below is part of that frozen process — between polls it costs
11
+ * ZERO compute while the run is parked.
12
+ * 2. GET /pauses/:id long-polls (each request a bounded ~5s server-side
13
+ * LISTEN race) until the row is terminal, then returns the human's answer.
14
+ *
15
+ * There is no marker file and no exception threaded through workflow code — the
16
+ * pause is a server operation, so nothing a `try/catch` can swallow.
17
+ */
18
+
19
+ export interface PauseDecision {
20
+ status: "resolved" | "expired" | "cancelled";
21
+ /** The human's answer (present on `resolved`). Caller-defined shape; the
22
+ * dashboard sends `{ decision: string }`. Null on expiry/cancel. */
23
+ decision: unknown;
24
+ }
25
+
26
+ export interface RequestPauseOptions {
27
+ /** `AGENT_COMPOSE_URL` — the server base URL injected into the sandbox. */
28
+ baseUrl: string;
29
+ /** `AGENT_COMPOSE_RUN_TOKEN` — the run credential the agent already carries. */
30
+ token: string;
31
+ /** `RUN_ID`. */
32
+ runId: string;
33
+ /** Human-readable reason shown on the pause feed / approval UI. */
34
+ reason: string;
35
+ /** Optional pause TTL; the workflow auto-expires the pause after this. */
36
+ ttlMs?: number;
37
+ /** Optional decision options the human picks from (e.g. approve / deny). */
38
+ options?: Array<{ id: string; label: string }>;
39
+ /** The action under review (e.g. the tool call), surfaced to the human. */
40
+ action?: Record<string, unknown>;
41
+ /** Optional second-key resume route. */
42
+ correlationKey?: string;
43
+ /** Abort the wait (the agent loop's signal). */
44
+ signal?: AbortSignal;
45
+ }
46
+
47
+ /** Inject a custom fetch (tests). Defaults to global fetch. */
48
+ type FetchFn = typeof fetch;
49
+
50
+ const isAbort = (signal: AbortSignal | undefined) => Boolean(signal?.aborted);
51
+
52
+ export async function requestPauseAndAwait(
53
+ opts: RequestPauseOptions,
54
+ fetchImpl: FetchFn = fetch,
55
+ ): Promise<PauseDecision> {
56
+ const base = opts.baseUrl.replace(/\/+$/, "");
57
+ const headers = {
58
+ authorization: `Bearer ${opts.token}`,
59
+ "content-type": "application/json",
60
+ accept: "application/json",
61
+ } as const;
62
+
63
+ // 1. Create the pending pause. The server wakes the step activity → the live
64
+ // VM suspends in place. 202 → { pauseId }.
65
+ const createRes = await fetchImpl(`${base}/api/v1/runs/${opts.runId}/pauses`, {
66
+ method: "POST",
67
+ headers,
68
+ body: JSON.stringify({
69
+ reason: opts.reason,
70
+ ...(opts.ttlMs !== undefined ? { ttlMs: opts.ttlMs } : {}),
71
+ ...(opts.options ? { options: opts.options } : {}),
72
+ ...(opts.action ? { action: opts.action } : {}),
73
+ ...(opts.correlationKey ? { correlationKey: opts.correlationKey } : {}),
74
+ }),
75
+ ...(opts.signal ? { signal: opts.signal } : {}),
76
+ });
77
+ if (!createRes.ok) {
78
+ const text = await createRes.text().catch(() => "");
79
+ throw new Error(`pause create failed (${createRes.status}): ${text.slice(0, 300)}`);
80
+ }
81
+ const { pauseId } = await createRes.json() as { pauseId: string };
82
+
83
+ // 2. Poll until terminal. Each GET is a bounded server long-poll; the loop
84
+ // reissues. A transient error backs off and retries — the durable row is
85
+ // the source of truth, so a dropped poll never loses the decision.
86
+ const pollUrl = `${base}/api/v1/runs/${opts.runId}/pauses/${pauseId}`;
87
+ let backoffMs = 0;
88
+ for (;;) {
89
+ if (isAbort(opts.signal)) throw new Error("pause wait aborted");
90
+ if (backoffMs > 0) await new Promise((r) => setTimeout(r, backoffMs));
91
+ let res: Awaited<ReturnType<FetchFn>>;
92
+ try {
93
+ res = await fetchImpl(pollUrl, { method: "GET", headers, ...(opts.signal ? { signal: opts.signal } : {}) });
94
+ } catch (err) {
95
+ if (isAbort(opts.signal)) throw err;
96
+ backoffMs = Math.min(8_000, (backoffMs || 500) * 2);
97
+ continue;
98
+ }
99
+ if (!res.ok) {
100
+ backoffMs = Math.min(8_000, (backoffMs || 500) * 2);
101
+ continue;
102
+ }
103
+ backoffMs = 0;
104
+ const body = await res.json() as { status: string; resumePayload?: unknown };
105
+ if (body.status === "pending") continue;
106
+ return { status: body.status as PauseDecision["status"], decision: body.resumePayload ?? null };
107
+ }
108
+ }
@@ -14,15 +14,15 @@ import { z } from "zod";
14
14
  import { randomUUID } from "node:crypto";
15
15
  import { getActiveStep, nextAgentCallInActiveStep } from "../active-step.js";
16
16
  import { corePause, type PauseRequest } from "../pause/pause-core.js";
17
+ import { createAskHumanProcessor } from "../processors/ask-human.js";
17
18
  import { agentLoop } from "./agent-loop.js";
18
19
  import { consumeSteerPending, runControlPoller } from "./steer-control.js";
19
- import { consumeLocalPauseRequest } from "./local-pause-request.js";
20
20
  import type { AgentLifecycleEvent, AgentLoopResult } from "./agent-loop.js";
21
21
  import { AsyncQueue } from "./async-queue.js";
22
22
  import type { AgentMessage, AgentStatus } from "../types/protocol.js";
23
23
  import type { AgentRuntime, RuntimeOptions } from "../types/runtime.js";
24
24
  import type { SandboxProvider } from "../types/sandbox.js";
25
- import { writeAgentContext } from "./agent-context.js";
25
+ import { writeAgentContext, buildAgentContextDoc } from "./agent-context.js";
26
26
  import type { AgentBudget } from "../types/workflow.js";
27
27
  import type { Processor } from "../processors/processor.js";
28
28
  import { RequestContext } from "../request-context/request-context.js";
@@ -331,10 +331,22 @@ export async function agent<T = unknown>(opts: AgentOpts<T>): Promise<AgentLoopR
331
331
  // (the boundary self-pause trigger + the prompt instruction above). Human
332
332
  // steering works regardless — see the ungated control poller.
333
333
  const mode = opts.mode ?? "auto";
334
+ // Ask-a-human is TOOLS-driven, not a separate mode: the ask-human processor is
335
+ // a chain default that maps the `AskUserQuestion` tool to a server pause (the
336
+ // run freezes, the human answers, the answer returns as the tool result). It's
337
+ // a no-op for any agent that doesn't have / call `AskUserQuestion`, so granting
338
+ // that tool to an agent IS its "can ask a human" switch — withhold it and the
339
+ // agent simply can't (it reports blockers up instead).
340
+ const processors = [createAskHumanProcessor(), ...(opts.processors ?? [])];
334
341
 
335
342
  try {
336
343
  return await agentLoop({
337
- runtime: (runtimeOpts: RuntimeOptions) => opts.runtime.create(opts.sandbox, runtimeOpts),
344
+ // Inject the platform manual into every runtime create() so a runtime that
345
+ // supports a system-prompt append (claude) carries it IN CONTEXT — not
346
+ // dependent on the agent choosing to `cat` AGENTS.md (which the SDK doesn't
347
+ // auto-load) or on writeAgentContext succeeding (it EACCES's on a read-only
348
+ // /workspace). buildAgentContextDoc is the same content writeAgentContext writes.
349
+ runtime: (runtimeOpts: RuntimeOptions) => opts.runtime.create(opts.sandbox, { ...runtimeOpts, agentManual: buildAgentContextDoc(process.env) }),
338
350
  agentId,
339
351
  ...(opts.label !== undefined ? { label: opts.label } : {}),
340
352
  ...(opts.budget?.turnsPerIteration !== undefined ? { turnsPerIteration: opts.budget.turnsPerIteration } : {}),
@@ -357,7 +369,7 @@ export async function agent<T = unknown>(opts: AgentOpts<T>): Promise<AgentLoopR
357
369
  ...(opts.events ? { onAgentLifecycleEvent: (event: AgentLifecycleEvent) => { void opts.events?.emit(event); } } : {}),
358
370
  ...(opts.onAgentEvent ? { onAgentEvent: opts.onAgentEvent } : {}),
359
371
  ...(opts.onIteration ? { onIteration: opts.onIteration } : {}),
360
- ...(opts.processors?.length ? { processors: opts.processors } : {}),
372
+ ...(processors.length ? { processors } : {}),
361
373
  requestContext: opts.requestContext ?? RequestContext.fromReserved({
362
374
  teamId: "", runId: "", workflowId: "",
363
375
  factoryId: null, apiKeyScopes: [], parentRunId: null,
@@ -366,9 +378,6 @@ export async function agent<T = unknown>(opts: AgentOpts<T>): Promise<AgentLoopR
366
378
  // PR 7 steer-pause wiring.
367
379
  mode,
368
380
  consumeSteerPending: () => consumeSteerPending(agentId),
369
- // `agentc pause` self-pause: the loop reads the marker this agent's CLI
370
- // dropped in the state dir and stages a snapshot-release pause from it.
371
- consumeLocalPauseRequest: () => consumeLocalPauseRequest(agentId),
372
381
  ...(steerPause ? { pause: steerPause } : {}),
373
382
  });
374
383
  } finally {
package/src/index.ts CHANGED
@@ -76,12 +76,18 @@ export {
76
76
  humanApproval,
77
77
  requireScope,
78
78
  redactPattern,
79
+ createGatePauseProcessor,
80
+ createAskHumanProcessor,
81
+ ASK_USER_QUESTION_TOOL,
79
82
  } from "./processors/index.js";
80
83
  export type {
81
84
  Processor,
82
85
  ProcessorContext,
83
86
  ProcessorVerdict,
84
87
  ToolCall,
88
+ GatePausePolicy,
89
+ GatePauseApproval,
90
+ GatePauseConnection,
85
91
  } from "./processors/index.js";
86
92
 
87
93
  // Protocol types (agent-loop input/output shapes)
@@ -170,6 +176,17 @@ export { default as codexRuntime } from "./runtimes/codex.js";
170
176
  export { createAmpRuntime } from "./runtimes/amp.js";
171
177
  export type { AmpRuntimeConfig } from "./runtimes/amp.js";
172
178
  export { default as ampRuntime } from "./runtimes/amp.js";
179
+ export { createOpencodeRuntime, opencodeSpec } from "./runtimes/opencode.js";
180
+ export type { OpencodeRuntimeConfig } from "./runtimes/opencode.js";
181
+ export { default as opencodeRuntime } from "./runtimes/opencode.js";
182
+
183
+ export { createCursorRuntime, cursorSpec } from "./runtimes/cursor.js";
184
+ export type { CursorRuntimeConfig } from "./runtimes/cursor.js";
185
+ export { default as cursorRuntime } from "./runtimes/cursor.js";
186
+
187
+ export { createDroidRuntime, droidSpec } from "./runtimes/droid.js";
188
+ export type { DroidRuntimeConfig } from "./runtimes/droid.js";
189
+ export { default as droidRuntime } from "./runtimes/droid.js";
173
190
 
174
191
  // Built-in coding tools for Vercel AI SDK runtime.
175
192
  export { bashTool, codingTools, editTool, readTool, writeTool } from "./tools/index.js";
@@ -232,6 +249,8 @@ export type {
232
249
  // `invokeStep` (server) and `serveStep` (runner).
233
250
  export {
234
251
  invokeStep,
252
+ launchStep,
253
+ reconnectStep,
235
254
  serveStep,
236
255
  parseStepResult,
237
256
  buildStepEnvs,
@@ -256,10 +275,12 @@ export {
256
275
  export type { PauseErrorCode } from "./pause/errors.js";
257
276
  export type { PauseRequest } from "./pause/pause-core.js";
258
277
  export type { WaitForEventRequest } from "./pause/wrappers.js";
259
- // `agentc pause` writes a local pause-request marker the agent loop consumes
260
- // (the snapshot-release bridge); the CLI imports the writer.
261
- export { writeLocalPauseRequest, consumeLocalPauseRequest } from "./agent/local-pause-request.js";
262
- export type { LocalPauseRequest, PauseOption } from "./agent/local-pause-request.js";
278
+ // ADR-0028 — the in-sandbox pause client. `agentc pause` and the gate-pause
279
+ // processor both pause their OWN run through the server pause API and block
280
+ // until a human answers (create-then-poll; the VM suspends in place while
281
+ // parked). No marker, no exception threaded through workflow code.
282
+ export { requestPauseAndAwait } from "./agent/pause-client.js";
283
+ export type { PauseDecision, RequestPauseOptions } from "./agent/pause-client.js";
263
284
  export type {
264
285
  StepRequest,
265
286
  StepResult,
@@ -268,6 +289,8 @@ export type {
268
289
  StepHandler,
269
290
  StepHandlerResult,
270
291
  ServeStepRequest,
292
+ RunningStep,
293
+ InvokeStepOptions,
271
294
  } from "./step-invocation/index.js";
272
295
 
273
296
  // Agent loop — for workflows that embed an LLM agent in their run() body.