@agent-compose/sdk 0.5.9 → 0.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/agent/agent-context.d.ts +1 -1
- package/dist/agent/agent-loop.d.ts +0 -8
- package/dist/agent/pause-client.d.ts +50 -0
- package/dist/index.d.ts +15 -6
- package/dist/index.js +537 -124
- package/dist/processors/ask-human.d.ts +30 -0
- package/dist/processors/ask-human.test.d.ts +1 -0
- package/dist/processors/gate-pause.d.ts +46 -0
- package/dist/processors/gate-pause.test.d.ts +1 -0
- package/dist/processors/index.d.ts +3 -0
- package/dist/runtimes/_cli-agent.d.ts +9 -0
- package/dist/runtimes/cursor.d.ts +9 -0
- package/dist/runtimes/droid.d.ts +9 -0
- package/dist/runtimes/openai-desktop.js +522 -122
- package/dist/runtimes/opencode.d.ts +25 -0
- package/dist/runtimes/vercel.js +11 -1
- package/dist/step-invocation/__tests__/background-invoker.test.d.ts +1 -0
- package/dist/step-invocation/index.d.ts +2 -1
- package/dist/step-invocation/invoker.d.ts +49 -0
- package/dist/step-invocation/protocol.d.ts +8 -0
- package/dist/types/runtime.d.ts +7 -0
- package/dist/types/sandbox.d.ts +54 -0
- package/dist/types/workflow.d.ts +12 -0
- package/dist/utils/errors.d.ts +9 -1
- package/package.json +1 -1
- package/src/agent/agent-context.ts +23 -16
- package/src/agent/agent-loop.ts +13 -33
- package/src/agent/pause-client.ts +108 -0
- package/src/agent/run-agent.ts +16 -7
- package/src/index.ts +27 -4
- package/src/processors/ask-human.ts +136 -0
- package/src/processors/gate-pause.ts +94 -0
- package/src/processors/index.ts +11 -0
- package/src/runtimes/_cli-agent.ts +13 -5
- package/src/runtimes/claude.ts +10 -6
- package/src/runtimes/cursor.ts +59 -0
- package/src/runtimes/droid.ts +63 -0
- package/src/runtimes/opencode.ts +61 -0
- package/src/sandbox.ts +78 -3
- package/src/step-invocation/index.ts +2 -1
- package/src/step-invocation/invoker.ts +359 -86
- package/src/step-invocation/protocol.ts +11 -0
- package/src/types/runtime.ts +7 -0
- package/src/types/sandbox.ts +53 -0
- package/src/types/workflow.ts +12 -0
- package/src/utils/errors.ts +19 -2
- package/dist/agent/local-pause-request.d.ts +0 -49
- package/src/agent/local-pause-request.ts +0 -90
- /package/dist/agent/{local-pause-request.test.d.ts → pause-client.test.d.ts} +0 -0
|
@@ -0,0 +1,25 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* OpenCode CLI runtime — drives sst's `opencode` agentic CLI inside the sandbox.
|
|
3
|
+
* OpenCode speaks ACP natively (`opencode acp`, protocolVersion 1 — verified
|
|
4
|
+
* live on E2B), so the runner drives it over ACP; the JSONL members below are
|
|
5
|
+
* only the version-mismatch fallback (vestigial for a v1 agent).
|
|
6
|
+
*
|
|
7
|
+
* Auth + model via OpenRouter: set OPENROUTER_API_KEY (a factory/workflow
|
|
8
|
+
* secret) and use a model id like `openrouter/z-ai/glm-5.2`. The runtime
|
|
9
|
+
* installs the `opencode-ai` CLI on demand; pair with
|
|
10
|
+
* `snapshots: { bootFrom: "reuse" }` to install once and boot from the capture.
|
|
11
|
+
*
|
|
12
|
+
* Verified live on E2B (2026-06-30): `npm i -g opencode-ai` (v1.17.12);
|
|
13
|
+
* `opencode run --model openrouter/z-ai/glm-5.2` drove a GLM-5.2 turn through
|
|
14
|
+
* OpenRouter; `opencode acp` answered the ACP `initialize` handshake with
|
|
15
|
+
* protocolVersion 1.
|
|
16
|
+
*/
|
|
17
|
+
import { type CliAgentSpec } from "./_cli-agent.js";
|
|
18
|
+
export declare const opencodeSpec: CliAgentSpec;
|
|
19
|
+
export interface OpencodeRuntimeConfig {
|
|
20
|
+
/** OpenRouter-prefixed model id (default `openrouter/z-ai/glm-5.2`). */
|
|
21
|
+
model?: string;
|
|
22
|
+
}
|
|
23
|
+
export declare function createOpencodeRuntime(config?: OpencodeRuntimeConfig): import("../index.js").AgentRuntime<import("../index.js").SandboxProvider>;
|
|
24
|
+
declare const _default: import("../index.js").AgentRuntime<import("../index.js").SandboxProvider>;
|
|
25
|
+
export default _default;
|
package/dist/runtimes/vercel.js
CHANGED
|
@@ -717,7 +717,17 @@ async function corePause(req, coord, kind = "custom") {
|
|
|
717
717
|
|
|
718
718
|
// src/utils/errors.ts
|
|
719
719
|
function formatError(err) {
|
|
720
|
-
|
|
720
|
+
if (!(err instanceof Error))
|
|
721
|
+
return String(err);
|
|
722
|
+
const parts = [err.message];
|
|
723
|
+
const seen = new Set([err]);
|
|
724
|
+
let cause = err.cause;
|
|
725
|
+
while (cause != null && !seen.has(cause)) {
|
|
726
|
+
seen.add(cause);
|
|
727
|
+
parts.push(cause instanceof Error ? cause.message : String(cause));
|
|
728
|
+
cause = cause instanceof Error ? cause.cause : undefined;
|
|
729
|
+
}
|
|
730
|
+
return parts.join(": ");
|
|
721
731
|
}
|
|
722
732
|
|
|
723
733
|
// src/runtimes/vercel.ts
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
export {};
|
|
@@ -18,7 +18,8 @@
|
|
|
18
18
|
* to test in isolation.
|
|
19
19
|
*/
|
|
20
20
|
export { STEP_RESULT_PREFIX, STEP_PAUSE_PREFIX, STEP_ENV, RUNNER_BUNDLE_PATH, RUNNER_COMMAND, stepInputPath, requestContextPath, } from "./protocol.js";
|
|
21
|
-
export { invokeStep, parseStepResult, buildStepEnvs } from "./invoker.js";
|
|
21
|
+
export { invokeStep, launchStep, reconnectStep, parseStepResult, buildStepEnvs } from "./invoker.js";
|
|
22
|
+
export type { RunningStep, InvokeStepOptions } from "./invoker.js";
|
|
22
23
|
export { serveStep } from "./server.js";
|
|
23
24
|
export type { StepHandler, ServeStepRequest, StepHandlerResult } from "./server.js";
|
|
24
25
|
export { StepExecutionError } from "./types.js";
|
|
@@ -64,5 +64,54 @@ export interface InvokeStepOptions {
|
|
|
64
64
|
* the caller's job; the activity batches and inserts at step completion. */
|
|
65
65
|
onStdout?: (line: string) => void;
|
|
66
66
|
onStderr?: (line: string) => void;
|
|
67
|
+
/** Called when the LIVE output stream fails mid-run (e.g. E2B's connect-web
|
|
68
|
+
* transport throws `received unsupported compressed output` on a large
|
|
69
|
+
* compressed frame) and the invoker falls back to recovering the result +
|
|
70
|
+
* full logs from the durable token-keyed files. The run is NOT failed — this
|
|
71
|
+
* is the hook to emit a structured alert so the degradation is visible/paged.
|
|
72
|
+
* Best-effort: keep it cheap and non-throwing. */
|
|
73
|
+
onStreamDegraded?: (info: {
|
|
74
|
+
error: unknown;
|
|
75
|
+
runnerPid: number;
|
|
76
|
+
resultToken: string;
|
|
77
|
+
}) => void;
|
|
78
|
+
}
|
|
79
|
+
/** A step running as a background command (ADR-0028). The activity races its
|
|
80
|
+
* `wait()` against a server pause request; on a pause it freezes the VM
|
|
81
|
+
* (`pauseProcess`) and persists `runnerPid` + `resultToken` so the resume
|
|
82
|
+
* activity can `reconnectStep(...)` to the SAME process — no re-run. */
|
|
83
|
+
export interface RunningStep<TOutput = unknown> {
|
|
84
|
+
/** OS pid of the background runner inside the VM — the reconnect handle. */
|
|
85
|
+
runnerPid: number;
|
|
86
|
+
/** Per-invocation token keying the durable result file the runner writes. */
|
|
87
|
+
resultToken: string;
|
|
88
|
+
/** Await the runner's exit and classify its output into a `StepResult`. */
|
|
89
|
+
wait(): Promise<StepResult<TOutput>>;
|
|
67
90
|
}
|
|
68
91
|
export declare function invokeStep<TOutput = unknown>(sandbox: SandboxProvider, request: StepRequest, opts?: InvokeStepOptions): Promise<StepResult<TOutput>>;
|
|
92
|
+
/**
|
|
93
|
+
* Launch the step runner as a BACKGROUND command (ADR-0028) and return a handle
|
|
94
|
+
* the activity drives: it races `wait()` against a server pause request and, on
|
|
95
|
+
* a pause, freezes the VM (`pauseProcess`) and persists `runnerPid` +
|
|
96
|
+
* `resultToken` so `reconnectStep` can continue the SAME process — no re-run.
|
|
97
|
+
* Requires a provider with `commands.runBackground` (E2B); the foreground
|
|
98
|
+
* `invokeStep` is the path for providers without it (Vercel).
|
|
99
|
+
*/
|
|
100
|
+
export declare function launchStep<TOutput = unknown>(sandbox: SandboxProvider, request: StepRequest, opts?: InvokeStepOptions): Promise<RunningStep<TOutput>>;
|
|
101
|
+
/**
|
|
102
|
+
* Resume a previously-paused background runner (ADR-0028) and return a
|
|
103
|
+
* `RunningStep` handle — uniform with `launchStep` so the activity can race
|
|
104
|
+
* `wait()` against a fresh pause request (a resumed step can pause again). After
|
|
105
|
+
* the workflow reconnects the suspended VM (`Sandbox.connect` auto-resumes it),
|
|
106
|
+
* this re-attaches to the still-running runner by `runnerPid`; `wait()` awaits
|
|
107
|
+
* its exit (event-driven — no polling) and classifies the output (the durable
|
|
108
|
+
* result file is authoritative on this path). If the runner already exited
|
|
109
|
+
* during resume, the re-attach fails and `wait()` classifies from the file.
|
|
110
|
+
*/
|
|
111
|
+
export declare function reconnectStep<TOutput = unknown>(sandbox: SandboxProvider, resume: {
|
|
112
|
+
runnerPid: number;
|
|
113
|
+
resultToken: string;
|
|
114
|
+
stepIndex: number;
|
|
115
|
+
}, opts?: Pick<InvokeStepOptions, "onStdout" | "onStderr"> & {
|
|
116
|
+
signal?: AbortSignal;
|
|
117
|
+
}): Promise<RunningStep<TOutput>>;
|
|
@@ -60,3 +60,11 @@ export declare function requestContextPath(stepIndex: number): string;
|
|
|
60
60
|
* drop the tail of a heavy stdout stream — the invoker falls back to
|
|
61
61
|
* reading this file when no sentinel is found on stdout. */
|
|
62
62
|
export declare function stepResultFilePath(token: string): string;
|
|
63
|
+
/** Sandbox-side path where the runner's stdout is tee'd as a durable LOG file,
|
|
64
|
+
* keyed by the per-invocation token. The live output rides E2B's connect-web
|
|
65
|
+
* command stream, which THROWS on a compressed large frame (gRPC-web cannot
|
|
66
|
+
* decode message compression) and kills the feed mid-run. This file, read back
|
|
67
|
+
* over the envd HTTP file transport (compression-immune), lets the invoker
|
|
68
|
+
* recover the FULL logs after such a fault instead of losing the tail — the
|
|
69
|
+
* log-side analogue of `stepResultFilePath` for the result. */
|
|
70
|
+
export declare function stepLogFilePath(token: string): string;
|
package/dist/types/runtime.d.ts
CHANGED
|
@@ -29,6 +29,13 @@ export interface RuntimeOptions {
|
|
|
29
29
|
/** Agent id and label for processor context / adapter logs. */
|
|
30
30
|
agentId?: string;
|
|
31
31
|
iteration?: number;
|
|
32
|
+
/** The platform manual (file conventions, connectors & access, how to pause).
|
|
33
|
+
* `agent()` builds it per-run (`buildAgentContextDoc`) and threads it here so
|
|
34
|
+
* a runtime that supports a system-prompt append (the claude runtime) injects
|
|
35
|
+
* it directly — instead of relying on the agent to `cat` the on-disk
|
|
36
|
+
* AGENTS.md/CLAUDE.md, which the Agent SDK doesn't auto-load and which can
|
|
37
|
+
* fail to write on a read-only/degraded working dir. */
|
|
38
|
+
agentManual?: string;
|
|
32
39
|
/** The run's pause boundary, threaded from the agent loop so a runtime-driven
|
|
33
40
|
* pre-tool gate (e.g. the ACP `session/request_permission` path through
|
|
34
41
|
* `gateToolCall`) can raise a human-approval `ctx.pause`. The runtime binds it
|
package/dist/types/sandbox.d.ts
CHANGED
|
@@ -21,6 +21,29 @@ export interface SandboxCommandResult {
|
|
|
21
21
|
stdout: string;
|
|
22
22
|
stderr: string;
|
|
23
23
|
}
|
|
24
|
+
/** A long-running command launched in the background (ADR-0028). Unlike
|
|
25
|
+
* `commands.run` (which awaits completion on one connection), a background
|
|
26
|
+
* command keeps running inside the VM independent of the launching
|
|
27
|
+
* connection: it survives `pauseProcess()`/resume and is re-attachable by
|
|
28
|
+
* `pid` after a fresh `Sandbox.connect`. This is what lets the platform
|
|
29
|
+
* freeze an agent mid-turn for a human-in-the-loop pause and continue the
|
|
30
|
+
* SAME process on resume — no re-run.
|
|
31
|
+
*
|
|
32
|
+
* Implemented ONLY by process-resume-capable providers (E2B); the presence
|
|
33
|
+
* of `commands.runBackground` IS the capability flag, paired with
|
|
34
|
+
* `pauseProcess`. Providers without it leave both undefined and pause via
|
|
35
|
+
* `snapshot()` + re-run instead. */
|
|
36
|
+
export interface SandboxBackgroundProcess {
|
|
37
|
+
/** OS pid inside the VM — the durable handle used to reconnect after a
|
|
38
|
+
* pause/resume cycle via `commands.connectProcess(pid)`. */
|
|
39
|
+
pid: number;
|
|
40
|
+
/** Resolve when the process exits, with its buffered result. Live output
|
|
41
|
+
* streams to the `onStdout`/`onStderr` passed at launch / connect time.
|
|
42
|
+
* Does NOT throw on a non-zero exit — the result carries `exitCode`. */
|
|
43
|
+
wait(): Promise<SandboxCommandResult>;
|
|
44
|
+
/** Force-terminate the process. */
|
|
45
|
+
kill(): Promise<void>;
|
|
46
|
+
}
|
|
24
47
|
/** A spawned long-lived command with a writable stdin and readable stdout,
|
|
25
48
|
* exposed as byte web-streams. Unlike `commands.run` (which buffers to
|
|
26
49
|
* completion and exposes stdout only via an `onStdout` callback), a duplex
|
|
@@ -78,9 +101,28 @@ export interface SandboxProvider {
|
|
|
78
101
|
* view). The vercel/e2b providers (server→sandbox) leave it undefined; an
|
|
79
102
|
* ACP caller that finds it absent falls back to the JSONL transport. */
|
|
80
103
|
spawnDuplex?(cmd: string, opts?: SandboxSpawnDuplexOptions): SandboxDuplexProcess;
|
|
104
|
+
/** Launch a command in the background and return immediately with a
|
|
105
|
+
* reconnectable handle (ADR-0028). The process survives the launching
|
|
106
|
+
* connection dropping AND a `pauseProcess()`/resume cycle. OPTIONAL —
|
|
107
|
+
* only process-resume providers (E2B) implement it; its presence (paired
|
|
108
|
+
* with `pauseProcess`) is the native-pause capability flag. */
|
|
109
|
+
runBackground?(cmd: string, opts?: SandboxCommandRunOptions): Promise<SandboxBackgroundProcess>;
|
|
110
|
+
/** Re-attach to a background command by `pid` after a fresh
|
|
111
|
+
* `Sandbox.connect` (the resume half of `runBackground`). OPTIONAL,
|
|
112
|
+
* E2B-only. Throws if no process with that pid is running. */
|
|
113
|
+
connectProcess?(pid: number, opts?: Pick<SandboxCommandRunOptions, "onStdout" | "onStderr" | "timeoutMs">): Promise<SandboxBackgroundProcess>;
|
|
81
114
|
};
|
|
82
115
|
files: {
|
|
83
116
|
write(path: string, content: string): Promise<void>;
|
|
117
|
+
/** Read a file's text content over the provider's FILE transport. On E2B this
|
|
118
|
+
* is the envd HTTP API (`Sandbox.files.read`), a DIFFERENT transport from
|
|
119
|
+
* `commands` — so a large readback is immune to the connect-web gRPC
|
|
120
|
+
* message-compression that can abort `commands.run` output on a big frame
|
|
121
|
+
* ("received unsupported compressed output"). This is what lets `launchStep`
|
|
122
|
+
* recover the full logs + result after a live-stream fault. OPTIONAL —
|
|
123
|
+
* implemented where durable file-readback is needed (E2B, local); providers
|
|
124
|
+
* that never drive the recovery path (Vercel — foreground only) may omit it. */
|
|
125
|
+
read?(path: string): Promise<string>;
|
|
84
126
|
};
|
|
85
127
|
kill(): Promise<void>;
|
|
86
128
|
/** Capture the running sandbox's state as a reusable snapshot. Vercel and E2B
|
|
@@ -95,6 +137,18 @@ export interface SandboxProvider {
|
|
|
95
137
|
snapshotId: string;
|
|
96
138
|
sizeBytes?: number;
|
|
97
139
|
}>;
|
|
140
|
+
/** Suspend the live VM in place and return a handle to resume it (ADR-0027).
|
|
141
|
+
* Present ONLY on process-resume-capable providers (E2B via `sandbox.pause()`,
|
|
142
|
+
* returning the sandbox id; resume is `Sandbox.connect(handle)`, which
|
|
143
|
+
* auto-resumes the paused VM). Unlike `snapshot()` — which captures an FS
|
|
144
|
+
* image, kills the origin, and re-runs the step from a fresh sandbox — a
|
|
145
|
+
* process-resume pause FREEZES the live process (zero compute) and continues
|
|
146
|
+
* it exactly where it blocked. The presence of this method IS the capability
|
|
147
|
+
* flag: providers without native VM-suspend leave it undefined and fall back
|
|
148
|
+
* to `snapshot()` + re-run. */
|
|
149
|
+
pauseProcess?(): Promise<{
|
|
150
|
+
resumeHandle: string;
|
|
151
|
+
}>;
|
|
98
152
|
/** Replace the live sandbox's egress policy in place — so the server can
|
|
99
153
|
* push a freshly resolved policy (with re-minted connector access tokens)
|
|
100
154
|
* before each step instead of relying on the policy baked at create.
|
package/dist/types/workflow.d.ts
CHANGED
|
@@ -90,6 +90,12 @@ export interface WorkflowDefinition<TOutput = unknown, TInput extends Record<str
|
|
|
90
90
|
/** Same as `input`, for the workflow's return value. Captured into
|
|
91
91
|
* `outputSchema` metadata and rendered in the IO panel. */
|
|
92
92
|
output?: z.ZodType<TOutput>;
|
|
93
|
+
/**
|
|
94
|
+
* @deprecated Legacy run-form. Prefer step-form — the
|
|
95
|
+
* `.step(defineStep(...))` builder — for per-step durability/replay and
|
|
96
|
+
* working pause. A run-form body compiles to one opaque step
|
|
97
|
+
* (`compileRunForm`), so any failure/resume re-runs the whole body.
|
|
98
|
+
*/
|
|
93
99
|
run: WorkflowFn<TOutput, TInput>;
|
|
94
100
|
/**
|
|
95
101
|
* All snapshot config — boot source plus capture mode.
|
|
@@ -210,6 +216,12 @@ export interface WorkflowDefinition<TOutput = unknown, TInput extends Record<str
|
|
|
210
216
|
* stores the step plan; runner subprocesses execute one step at a time
|
|
211
217
|
* via the StepInvocation seam.
|
|
212
218
|
*/
|
|
219
|
+
/**
|
|
220
|
+
* @deprecated Run-form is legacy. Use the step-form overload —
|
|
221
|
+
* `defineWorkflow({ id, input, output }).step(defineStep(...)).build()` — for
|
|
222
|
+
* durable, replayable steps and working pause. Run-form compiles to a single
|
|
223
|
+
* opaque step (`compileRunForm`); there is no per-step replay.
|
|
224
|
+
*/
|
|
213
225
|
export declare function defineWorkflow<TOutput = unknown, TInput extends Record<string, unknown> = Record<string, unknown>>(def: WorkflowDefinition<TOutput, TInput>): Workflow<TInput, TOutput>;
|
|
214
226
|
export declare function defineWorkflow<TInput, TOutput>(def: StepWorkflowDefinition<TInput, TOutput>): import("../workflow-steps/workflow.js").WorkflowBuilder<TInput, TInput>;
|
|
215
227
|
/** Observability-only lifecycle hooks — passed to the workflow engine, not workflow authors. */
|
package/dist/utils/errors.d.ts
CHANGED
|
@@ -1,2 +1,10 @@
|
|
|
1
|
-
/** Convert any thrown value to a string message.
|
|
1
|
+
/** Convert any thrown value to a string message, INCLUDING its `.cause` chain.
|
|
2
|
+
*
|
|
3
|
+
* Many wrapped errors carry the real reason on `.cause` and only a generic
|
|
4
|
+
* summary on `.message` — the Temporal SDK's "Failed to start Workflow" is the
|
|
5
|
+
* canonical example (its `.cause` is the actual gRPC rejection, e.g. "search
|
|
6
|
+
* attribute X is not defined"). Returning `.message` alone swallowed that, so
|
|
7
|
+
* failures surfaced as opaque one-liners. Walk the chain and join the messages
|
|
8
|
+
* so the root cause is always visible. Cycle-guarded against self-referential
|
|
9
|
+
* `cause` links. */
|
|
2
10
|
export declare function formatError(err: unknown): string;
|
package/package.json
CHANGED
|
@@ -71,29 +71,36 @@ Names like \`note.created\` / \`brief.posted\` surface in the Workbench;
|
|
|
71
71
|
|
|
72
72
|
## Writing workflow / agent code — the SDK
|
|
73
73
|
|
|
74
|
-
\`@agent-compose/sdk\` is installed in \`/workspace
|
|
75
|
-
|
|
74
|
+
\`@agent-compose/sdk\` is installed in \`/workspace\`. **To author a workflow,
|
|
75
|
+
ALWAYS run \`/ac:generate-workflow\`** (and \`/ac:generate-agent\` for an agent
|
|
76
|
+
step) instead of writing source from memory — the skill scaffolds the correct,
|
|
77
|
+
current shape. Then \`agentc register <file.ts>\` (or \`/ac:register\`).
|
|
76
78
|
|
|
77
|
-
|
|
79
|
+
The skill writes **step-form** (a builder of discrete, durable \`.step()\`s).
|
|
80
|
+
Never write the legacy run-form (\`defineWorkflow({ run(ctx, sandbox) { … } })\`):
|
|
81
|
+
it is one opaque step, so any failure or resume re-runs the whole body — and
|
|
82
|
+
**pause does not work in run-form**.
|
|
78
83
|
|
|
79
|
-
|
|
80
|
-
\`agentc register <file.ts>\` (or \`/ac:register\`).
|
|
84
|
+
## Pausing to ask the human
|
|
81
85
|
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
When you can't or shouldn't proceed without a human, run \`agentc pause\`, then
|
|
85
|
-
**END YOUR TURN**:
|
|
86
|
+
To ask a human and get an answer back, use the **\`AskUserQuestion\`** tool if
|
|
87
|
+
you have it; otherwise run **\`agentc pause\`**:
|
|
86
88
|
|
|
87
89
|
agentc pause --reason "Notion returned 401 — connect Notion to continue" \\
|
|
88
90
|
--option retry --option skip
|
|
89
91
|
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
92
|
+
**Both BLOCK and hand you the answer inline.** While you wait, the run is
|
|
93
|
+
suspended — your sandbox is frozen and compute stops, so a pause is free while
|
|
94
|
+
the human decides. When they answer, the call RETURNS with their decision: the
|
|
95
|
+
\`AskUserQuestion\` tool result, or \`agentc pause\`'s output
|
|
96
|
+
(\`▶ Resumed. The human answered: …\`), carries it.
|
|
97
|
+
|
|
98
|
+
**Then USE that answer to finish your work — do NOT end your turn.** This is NOT
|
|
99
|
+
fire-and-forget, and the answer does NOT arrive in a later message: it comes
|
|
100
|
+
back right where you called it, on the SAME turn. The shape is: ask → the call
|
|
101
|
+
blocks → it returns the human's answer → you act on it and produce your result.
|
|
102
|
+
Never end your turn before the call returns, never guess an answer, and never
|
|
103
|
+
proceed without one.
|
|
97
104
|
|
|
98
105
|
Reach for it the moment you hit — or foresee — any of these:
|
|
99
106
|
- **A wall only a human can clear:** a 401/403, a missing credential, an
|
package/src/agent/agent-loop.ts
CHANGED
|
@@ -15,7 +15,6 @@ import { RequestContext } from "../request-context/request-context.js";
|
|
|
15
15
|
import { PauseManager } from "../pause/manager.js";
|
|
16
16
|
import { PauseSignal, isPauseSignal } from "../pause/pause-core.js";
|
|
17
17
|
import { SteerDecisionSchema, type SteerDecision, type SteerPayload } from "./steer-control.js";
|
|
18
|
-
import { type LocalPauseRequest } from "./local-pause-request.js";
|
|
19
18
|
|
|
20
19
|
export const DEFAULT_CLAUDE_MODEL = "claude-fable-5";
|
|
21
20
|
|
|
@@ -154,13 +153,6 @@ export interface AgentLoopOpts<TResponse = unknown> {
|
|
|
154
153
|
* The boundary consumes it once per check, and only when no steer is
|
|
155
154
|
* already staged. */
|
|
156
155
|
consumeSteerPending?: () => SteerPayload | null;
|
|
157
|
-
/** Take-once read of a local `agentc pause` request this agent dropped in the
|
|
158
|
-
* state dir during its turn (the CLI writes it; see local-pause-request.ts).
|
|
159
|
-
* Returns the request (reason + offered options) or null. The loop turns it
|
|
160
|
-
* into a staged self-pause the next boundary takes — the same snapshot-release
|
|
161
|
-
* path as `needs_input`, but triggered by an explicit `agentc pause` call so
|
|
162
|
-
* it is honored regardless of `mode`. */
|
|
163
|
-
consumeLocalPauseRequest?: () => LocalPauseRequest | null;
|
|
164
156
|
/** PR 7: `auto` (autonomous, default) or `hitl` (can be steer-paused). The
|
|
165
157
|
* boundary is inert unless `hitl`. */
|
|
166
158
|
mode?: "auto" | "hitl";
|
|
@@ -172,9 +164,16 @@ export async function agentLoop<TResponse = unknown>(opts: AgentLoopOpts<TRespon
|
|
|
172
164
|
const logLabel = opts.label ?? "[Agent Loop]";
|
|
173
165
|
const startedAt = Date.now();
|
|
174
166
|
// No budget ⇒ no turn cap: the harness runtime (Claude Code) decides when it's
|
|
175
|
-
// done. A numeric budget is an explicit caller choice
|
|
167
|
+
// done in one unbounded session. A numeric budget is an explicit caller choice.
|
|
168
|
+
// Default to a SMALL allowance (not 1) so the loop can RE-PROMPT for the closing
|
|
169
|
+
// <status>/<response> when a session ends EARLY — e.g. a human pause (agentc
|
|
170
|
+
// pause / AskUserQuestion) freezes the VM mid-tool and the resumed session can
|
|
171
|
+
// drop its final turn after doing the work. The re-prompt (PROTOCOL_SUFFIX)
|
|
172
|
+
// only fires when an iteration produced no exit_signal, so a clean run still
|
|
173
|
+
// settles in ONE iteration — these extra iterations are a recovery path, not
|
|
174
|
+
// the norm.
|
|
176
175
|
const turnsPerIteration = opts.turnsPerIteration;
|
|
177
|
-
const maxIterations = opts.maxIterations ?? (turnsPerIteration === undefined ?
|
|
176
|
+
const maxIterations = opts.maxIterations ?? (turnsPerIteration === undefined ? 3 : 8);
|
|
178
177
|
// A responseSchema is a CONTRACT, not a hope: when the agent's <response> fails
|
|
179
178
|
// validation, the loop re-prompts with the exact errors until it conforms —
|
|
180
179
|
// without consuming the caller's iteration budget. The backstop below only
|
|
@@ -462,29 +461,10 @@ export async function agentLoop<TResponse = unknown>(opts: AgentLoopOpts<TRespon
|
|
|
462
461
|
|
|
463
462
|
let status = parseAgentStatus(responseText);
|
|
464
463
|
|
|
465
|
-
// `agentc pause`
|
|
466
|
-
//
|
|
467
|
-
//
|
|
468
|
-
//
|
|
469
|
-
// dashboard renders them as buttons). Checked BEFORE the settle / needs_input
|
|
470
|
-
// paths and independent of the <status> block — the common case is an agent
|
|
471
|
-
// that called the tool and ended its turn with no status at all, which would
|
|
472
|
-
// otherwise fall through to the empty-output re-prompt below. UNGATED by
|
|
473
|
-
// `mode`: an explicit `agentc pause` is a deliberate ask, not the `needs_input`
|
|
474
|
-
// heuristic that only `hitl` agents may trigger.
|
|
475
|
-
if (opts.pause && pendingSteerPause === null) {
|
|
476
|
-
const localPause = opts.consumeLocalPauseRequest?.() ?? null;
|
|
477
|
-
if (localPause) {
|
|
478
|
-
pendingSteerPause = {
|
|
479
|
-
reason: localPause.reason,
|
|
480
|
-
correlationKey: null,
|
|
481
|
-
...(localPause.options && localPause.options.length > 0 ? { payload: { options: localPause.options } } : {}),
|
|
482
|
-
at: Date.now(),
|
|
483
|
-
};
|
|
484
|
-
blockerStreak = null; // an explicit ask is not a stuck loop
|
|
485
|
-
continue; // pause fires at the next boundary
|
|
486
|
-
}
|
|
487
|
-
}
|
|
464
|
+
// ADR-0028 — `agentc pause` is no longer a marker the loop consumes here. It
|
|
465
|
+
// is a server operation: the CLI blocks on the pause API and the run is
|
|
466
|
+
// frozen in place by the step activity, with no loop involvement. The steer
|
|
467
|
+
// (human-driven) and `needs_input` (self-pause) paths below are unaffected.
|
|
488
468
|
|
|
489
469
|
// Inline safeParse (instead of letting parseAgentResponse validate)
|
|
490
470
|
// so a schema failure surfaces via lastResponseValidationError on
|
|
@@ -0,0 +1,108 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* In-sandbox pause client (ADR-0028). The single way an agent pauses its OWN
|
|
3
|
+
* run: `agentc pause` and the gate-pause processor both call
|
|
4
|
+
* `requestPauseAndAwait`, which drives the server pause API and blocks until a
|
|
5
|
+
* human resolves it.
|
|
6
|
+
*
|
|
7
|
+
* Create-then-poll, NOT a held connection:
|
|
8
|
+
* 1. POST /pauses creates the pending row. The server wakes the step
|
|
9
|
+
* activity, which freezes the live VM in place (E2B native suspend). The
|
|
10
|
+
* poll loop below is part of that frozen process — between polls it costs
|
|
11
|
+
* ZERO compute while the run is parked.
|
|
12
|
+
* 2. GET /pauses/:id long-polls (each request a bounded ~5s server-side
|
|
13
|
+
* LISTEN race) until the row is terminal, then returns the human's answer.
|
|
14
|
+
*
|
|
15
|
+
* There is no marker file and no exception threaded through workflow code — the
|
|
16
|
+
* pause is a server operation, so nothing a `try/catch` can swallow.
|
|
17
|
+
*/
|
|
18
|
+
|
|
19
|
+
export interface PauseDecision {
|
|
20
|
+
status: "resolved" | "expired" | "cancelled";
|
|
21
|
+
/** The human's answer (present on `resolved`). Caller-defined shape; the
|
|
22
|
+
* dashboard sends `{ decision: string }`. Null on expiry/cancel. */
|
|
23
|
+
decision: unknown;
|
|
24
|
+
}
|
|
25
|
+
|
|
26
|
+
export interface RequestPauseOptions {
|
|
27
|
+
/** `AGENT_COMPOSE_URL` — the server base URL injected into the sandbox. */
|
|
28
|
+
baseUrl: string;
|
|
29
|
+
/** `AGENT_COMPOSE_RUN_TOKEN` — the run credential the agent already carries. */
|
|
30
|
+
token: string;
|
|
31
|
+
/** `RUN_ID`. */
|
|
32
|
+
runId: string;
|
|
33
|
+
/** Human-readable reason shown on the pause feed / approval UI. */
|
|
34
|
+
reason: string;
|
|
35
|
+
/** Optional pause TTL; the workflow auto-expires the pause after this. */
|
|
36
|
+
ttlMs?: number;
|
|
37
|
+
/** Optional decision options the human picks from (e.g. approve / deny). */
|
|
38
|
+
options?: Array<{ id: string; label: string }>;
|
|
39
|
+
/** The action under review (e.g. the tool call), surfaced to the human. */
|
|
40
|
+
action?: Record<string, unknown>;
|
|
41
|
+
/** Optional second-key resume route. */
|
|
42
|
+
correlationKey?: string;
|
|
43
|
+
/** Abort the wait (the agent loop's signal). */
|
|
44
|
+
signal?: AbortSignal;
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
/** Inject a custom fetch (tests). Defaults to global fetch. */
|
|
48
|
+
type FetchFn = typeof fetch;
|
|
49
|
+
|
|
50
|
+
const isAbort = (signal: AbortSignal | undefined) => Boolean(signal?.aborted);
|
|
51
|
+
|
|
52
|
+
export async function requestPauseAndAwait(
|
|
53
|
+
opts: RequestPauseOptions,
|
|
54
|
+
fetchImpl: FetchFn = fetch,
|
|
55
|
+
): Promise<PauseDecision> {
|
|
56
|
+
const base = opts.baseUrl.replace(/\/+$/, "");
|
|
57
|
+
const headers = {
|
|
58
|
+
authorization: `Bearer ${opts.token}`,
|
|
59
|
+
"content-type": "application/json",
|
|
60
|
+
accept: "application/json",
|
|
61
|
+
} as const;
|
|
62
|
+
|
|
63
|
+
// 1. Create the pending pause. The server wakes the step activity → the live
|
|
64
|
+
// VM suspends in place. 202 → { pauseId }.
|
|
65
|
+
const createRes = await fetchImpl(`${base}/api/v1/runs/${opts.runId}/pauses`, {
|
|
66
|
+
method: "POST",
|
|
67
|
+
headers,
|
|
68
|
+
body: JSON.stringify({
|
|
69
|
+
reason: opts.reason,
|
|
70
|
+
...(opts.ttlMs !== undefined ? { ttlMs: opts.ttlMs } : {}),
|
|
71
|
+
...(opts.options ? { options: opts.options } : {}),
|
|
72
|
+
...(opts.action ? { action: opts.action } : {}),
|
|
73
|
+
...(opts.correlationKey ? { correlationKey: opts.correlationKey } : {}),
|
|
74
|
+
}),
|
|
75
|
+
...(opts.signal ? { signal: opts.signal } : {}),
|
|
76
|
+
});
|
|
77
|
+
if (!createRes.ok) {
|
|
78
|
+
const text = await createRes.text().catch(() => "");
|
|
79
|
+
throw new Error(`pause create failed (${createRes.status}): ${text.slice(0, 300)}`);
|
|
80
|
+
}
|
|
81
|
+
const { pauseId } = await createRes.json() as { pauseId: string };
|
|
82
|
+
|
|
83
|
+
// 2. Poll until terminal. Each GET is a bounded server long-poll; the loop
|
|
84
|
+
// reissues. A transient error backs off and retries — the durable row is
|
|
85
|
+
// the source of truth, so a dropped poll never loses the decision.
|
|
86
|
+
const pollUrl = `${base}/api/v1/runs/${opts.runId}/pauses/${pauseId}`;
|
|
87
|
+
let backoffMs = 0;
|
|
88
|
+
for (;;) {
|
|
89
|
+
if (isAbort(opts.signal)) throw new Error("pause wait aborted");
|
|
90
|
+
if (backoffMs > 0) await new Promise((r) => setTimeout(r, backoffMs));
|
|
91
|
+
let res: Awaited<ReturnType<FetchFn>>;
|
|
92
|
+
try {
|
|
93
|
+
res = await fetchImpl(pollUrl, { method: "GET", headers, ...(opts.signal ? { signal: opts.signal } : {}) });
|
|
94
|
+
} catch (err) {
|
|
95
|
+
if (isAbort(opts.signal)) throw err;
|
|
96
|
+
backoffMs = Math.min(8_000, (backoffMs || 500) * 2);
|
|
97
|
+
continue;
|
|
98
|
+
}
|
|
99
|
+
if (!res.ok) {
|
|
100
|
+
backoffMs = Math.min(8_000, (backoffMs || 500) * 2);
|
|
101
|
+
continue;
|
|
102
|
+
}
|
|
103
|
+
backoffMs = 0;
|
|
104
|
+
const body = await res.json() as { status: string; resumePayload?: unknown };
|
|
105
|
+
if (body.status === "pending") continue;
|
|
106
|
+
return { status: body.status as PauseDecision["status"], decision: body.resumePayload ?? null };
|
|
107
|
+
}
|
|
108
|
+
}
|
package/src/agent/run-agent.ts
CHANGED
|
@@ -14,15 +14,15 @@ import { z } from "zod";
|
|
|
14
14
|
import { randomUUID } from "node:crypto";
|
|
15
15
|
import { getActiveStep, nextAgentCallInActiveStep } from "../active-step.js";
|
|
16
16
|
import { corePause, type PauseRequest } from "../pause/pause-core.js";
|
|
17
|
+
import { createAskHumanProcessor } from "../processors/ask-human.js";
|
|
17
18
|
import { agentLoop } from "./agent-loop.js";
|
|
18
19
|
import { consumeSteerPending, runControlPoller } from "./steer-control.js";
|
|
19
|
-
import { consumeLocalPauseRequest } from "./local-pause-request.js";
|
|
20
20
|
import type { AgentLifecycleEvent, AgentLoopResult } from "./agent-loop.js";
|
|
21
21
|
import { AsyncQueue } from "./async-queue.js";
|
|
22
22
|
import type { AgentMessage, AgentStatus } from "../types/protocol.js";
|
|
23
23
|
import type { AgentRuntime, RuntimeOptions } from "../types/runtime.js";
|
|
24
24
|
import type { SandboxProvider } from "../types/sandbox.js";
|
|
25
|
-
import { writeAgentContext } from "./agent-context.js";
|
|
25
|
+
import { writeAgentContext, buildAgentContextDoc } from "./agent-context.js";
|
|
26
26
|
import type { AgentBudget } from "../types/workflow.js";
|
|
27
27
|
import type { Processor } from "../processors/processor.js";
|
|
28
28
|
import { RequestContext } from "../request-context/request-context.js";
|
|
@@ -331,10 +331,22 @@ export async function agent<T = unknown>(opts: AgentOpts<T>): Promise<AgentLoopR
|
|
|
331
331
|
// (the boundary self-pause trigger + the prompt instruction above). Human
|
|
332
332
|
// steering works regardless — see the ungated control poller.
|
|
333
333
|
const mode = opts.mode ?? "auto";
|
|
334
|
+
// Ask-a-human is TOOLS-driven, not a separate mode: the ask-human processor is
|
|
335
|
+
// a chain default that maps the `AskUserQuestion` tool to a server pause (the
|
|
336
|
+
// run freezes, the human answers, the answer returns as the tool result). It's
|
|
337
|
+
// a no-op for any agent that doesn't have / call `AskUserQuestion`, so granting
|
|
338
|
+
// that tool to an agent IS its "can ask a human" switch — withhold it and the
|
|
339
|
+
// agent simply can't (it reports blockers up instead).
|
|
340
|
+
const processors = [createAskHumanProcessor(), ...(opts.processors ?? [])];
|
|
334
341
|
|
|
335
342
|
try {
|
|
336
343
|
return await agentLoop({
|
|
337
|
-
|
|
344
|
+
// Inject the platform manual into every runtime create() so a runtime that
|
|
345
|
+
// supports a system-prompt append (claude) carries it IN CONTEXT — not
|
|
346
|
+
// dependent on the agent choosing to `cat` AGENTS.md (which the SDK doesn't
|
|
347
|
+
// auto-load) or on writeAgentContext succeeding (it EACCES's on a read-only
|
|
348
|
+
// /workspace). buildAgentContextDoc is the same content writeAgentContext writes.
|
|
349
|
+
runtime: (runtimeOpts: RuntimeOptions) => opts.runtime.create(opts.sandbox, { ...runtimeOpts, agentManual: buildAgentContextDoc(process.env) }),
|
|
338
350
|
agentId,
|
|
339
351
|
...(opts.label !== undefined ? { label: opts.label } : {}),
|
|
340
352
|
...(opts.budget?.turnsPerIteration !== undefined ? { turnsPerIteration: opts.budget.turnsPerIteration } : {}),
|
|
@@ -357,7 +369,7 @@ export async function agent<T = unknown>(opts: AgentOpts<T>): Promise<AgentLoopR
|
|
|
357
369
|
...(opts.events ? { onAgentLifecycleEvent: (event: AgentLifecycleEvent) => { void opts.events?.emit(event); } } : {}),
|
|
358
370
|
...(opts.onAgentEvent ? { onAgentEvent: opts.onAgentEvent } : {}),
|
|
359
371
|
...(opts.onIteration ? { onIteration: opts.onIteration } : {}),
|
|
360
|
-
...(
|
|
372
|
+
...(processors.length ? { processors } : {}),
|
|
361
373
|
requestContext: opts.requestContext ?? RequestContext.fromReserved({
|
|
362
374
|
teamId: "", runId: "", workflowId: "",
|
|
363
375
|
factoryId: null, apiKeyScopes: [], parentRunId: null,
|
|
@@ -366,9 +378,6 @@ export async function agent<T = unknown>(opts: AgentOpts<T>): Promise<AgentLoopR
|
|
|
366
378
|
// PR 7 steer-pause wiring.
|
|
367
379
|
mode,
|
|
368
380
|
consumeSteerPending: () => consumeSteerPending(agentId),
|
|
369
|
-
// `agentc pause` self-pause: the loop reads the marker this agent's CLI
|
|
370
|
-
// dropped in the state dir and stages a snapshot-release pause from it.
|
|
371
|
-
consumeLocalPauseRequest: () => consumeLocalPauseRequest(agentId),
|
|
372
381
|
...(steerPause ? { pause: steerPause } : {}),
|
|
373
382
|
});
|
|
374
383
|
} finally {
|
package/src/index.ts
CHANGED
|
@@ -76,12 +76,18 @@ export {
|
|
|
76
76
|
humanApproval,
|
|
77
77
|
requireScope,
|
|
78
78
|
redactPattern,
|
|
79
|
+
createGatePauseProcessor,
|
|
80
|
+
createAskHumanProcessor,
|
|
81
|
+
ASK_USER_QUESTION_TOOL,
|
|
79
82
|
} from "./processors/index.js";
|
|
80
83
|
export type {
|
|
81
84
|
Processor,
|
|
82
85
|
ProcessorContext,
|
|
83
86
|
ProcessorVerdict,
|
|
84
87
|
ToolCall,
|
|
88
|
+
GatePausePolicy,
|
|
89
|
+
GatePauseApproval,
|
|
90
|
+
GatePauseConnection,
|
|
85
91
|
} from "./processors/index.js";
|
|
86
92
|
|
|
87
93
|
// Protocol types (agent-loop input/output shapes)
|
|
@@ -170,6 +176,17 @@ export { default as codexRuntime } from "./runtimes/codex.js";
|
|
|
170
176
|
export { createAmpRuntime } from "./runtimes/amp.js";
|
|
171
177
|
export type { AmpRuntimeConfig } from "./runtimes/amp.js";
|
|
172
178
|
export { default as ampRuntime } from "./runtimes/amp.js";
|
|
179
|
+
export { createOpencodeRuntime, opencodeSpec } from "./runtimes/opencode.js";
|
|
180
|
+
export type { OpencodeRuntimeConfig } from "./runtimes/opencode.js";
|
|
181
|
+
export { default as opencodeRuntime } from "./runtimes/opencode.js";
|
|
182
|
+
|
|
183
|
+
export { createCursorRuntime, cursorSpec } from "./runtimes/cursor.js";
|
|
184
|
+
export type { CursorRuntimeConfig } from "./runtimes/cursor.js";
|
|
185
|
+
export { default as cursorRuntime } from "./runtimes/cursor.js";
|
|
186
|
+
|
|
187
|
+
export { createDroidRuntime, droidSpec } from "./runtimes/droid.js";
|
|
188
|
+
export type { DroidRuntimeConfig } from "./runtimes/droid.js";
|
|
189
|
+
export { default as droidRuntime } from "./runtimes/droid.js";
|
|
173
190
|
|
|
174
191
|
// Built-in coding tools for Vercel AI SDK runtime.
|
|
175
192
|
export { bashTool, codingTools, editTool, readTool, writeTool } from "./tools/index.js";
|
|
@@ -232,6 +249,8 @@ export type {
|
|
|
232
249
|
// `invokeStep` (server) and `serveStep` (runner).
|
|
233
250
|
export {
|
|
234
251
|
invokeStep,
|
|
252
|
+
launchStep,
|
|
253
|
+
reconnectStep,
|
|
235
254
|
serveStep,
|
|
236
255
|
parseStepResult,
|
|
237
256
|
buildStepEnvs,
|
|
@@ -256,10 +275,12 @@ export {
|
|
|
256
275
|
export type { PauseErrorCode } from "./pause/errors.js";
|
|
257
276
|
export type { PauseRequest } from "./pause/pause-core.js";
|
|
258
277
|
export type { WaitForEventRequest } from "./pause/wrappers.js";
|
|
259
|
-
//
|
|
260
|
-
//
|
|
261
|
-
|
|
262
|
-
|
|
278
|
+
// ADR-0028 — the in-sandbox pause client. `agentc pause` and the gate-pause
|
|
279
|
+
// processor both pause their OWN run through the server pause API and block
|
|
280
|
+
// until a human answers (create-then-poll; the VM suspends in place while
|
|
281
|
+
// parked). No marker, no exception threaded through workflow code.
|
|
282
|
+
export { requestPauseAndAwait } from "./agent/pause-client.js";
|
|
283
|
+
export type { PauseDecision, RequestPauseOptions } from "./agent/pause-client.js";
|
|
263
284
|
export type {
|
|
264
285
|
StepRequest,
|
|
265
286
|
StepResult,
|
|
@@ -268,6 +289,8 @@ export type {
|
|
|
268
289
|
StepHandler,
|
|
269
290
|
StepHandlerResult,
|
|
270
291
|
ServeStepRequest,
|
|
292
|
+
RunningStep,
|
|
293
|
+
InvokeStepOptions,
|
|
271
294
|
} from "./step-invocation/index.js";
|
|
272
295
|
|
|
273
296
|
// Agent loop — for workflows that embed an LLM agent in their run() body.
|