@agent-compose/sdk 0.5.8 → 0.5.9

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (70) hide show
  1. package/dist/agent/agent-context.d.ts +1 -1
  2. package/dist/agent/agent-loop.d.ts +22 -12
  3. package/dist/agent/local-pause-request.d.ts +49 -0
  4. package/dist/agent/local-pause-request.test.d.ts +1 -0
  5. package/dist/agent/steer-control.d.ts +22 -6
  6. package/dist/client.d.ts +12 -1
  7. package/dist/index.d.ts +4 -2
  8. package/dist/index.js +1275 -527
  9. package/dist/pause/checkpoint.d.ts +27 -10
  10. package/dist/pause/manager.d.ts +1 -0
  11. package/dist/pause/pause-core.d.ts +23 -0
  12. package/dist/pause/state-dir.d.ts +1 -1
  13. package/dist/processors/builtins.d.ts +20 -1
  14. package/dist/processors/index.d.ts +1 -1
  15. package/dist/processors/processor.d.ts +13 -0
  16. package/dist/runtimes/_acp-client.d.ts +140 -0
  17. package/dist/runtimes/_cli-agent.d.ts +155 -3
  18. package/dist/runtimes/amp.d.ts +2 -2
  19. package/dist/runtimes/cli-agent-acp-live.test.d.ts +30 -0
  20. package/dist/runtimes/cli-agent.test.d.ts +22 -6
  21. package/dist/runtimes/codex.d.ts +7 -2
  22. package/dist/runtimes/openai-desktop.js +1263 -527
  23. package/dist/runtimes/vercel.js +389 -2
  24. package/dist/sandbox.d.ts +113 -19
  25. package/dist/types/__tests__/environment-build-flag.test.d.ts +1 -0
  26. package/dist/types/__tests__/workflow-metadata-provider.test.d.ts +1 -0
  27. package/dist/types/execution-context.d.ts +0 -8
  28. package/dist/types/protocol.d.ts +32 -1
  29. package/dist/types/runtime.d.ts +7 -0
  30. package/dist/types/sandbox-environment.d.ts +6 -1
  31. package/dist/types/sandbox.d.ts +41 -6
  32. package/dist/types/workflow-metadata.d.ts +40 -10
  33. package/dist/types/workflow.d.ts +22 -6
  34. package/dist/utils/bundler.d.ts +5 -1
  35. package/dist/workflow-steps/observability.d.ts +28 -2
  36. package/dist/workflow-steps/types.d.ts +11 -7
  37. package/dist/workflow-steps/workflow.d.ts +3 -2
  38. package/package.json +3 -2
  39. package/src/agent/agent-context.ts +14 -6
  40. package/src/agent/agent-loop.ts +59 -10
  41. package/src/agent/local-pause-request.ts +90 -0
  42. package/src/agent/run-agent.ts +6 -2
  43. package/src/agent/steer-control.ts +21 -7
  44. package/src/client.ts +35 -1
  45. package/src/index.ts +10 -2
  46. package/src/pause/checkpoint.ts +33 -14
  47. package/src/pause/manager.ts +2 -2
  48. package/src/pause/pause-core.ts +35 -0
  49. package/src/pause/state-dir.ts +2 -2
  50. package/src/processors/builtins.ts +44 -1
  51. package/src/processors/index.ts +1 -0
  52. package/src/processors/processor.ts +13 -0
  53. package/src/runtimes/_acp-client.ts +516 -0
  54. package/src/runtimes/_cli-agent.ts +418 -3
  55. package/src/runtimes/claude.ts +27 -3
  56. package/src/runtimes/codex.ts +21 -1
  57. package/src/runtimes/vercel.ts +4 -1
  58. package/src/sandbox.ts +361 -54
  59. package/src/types/execution-context.ts +0 -8
  60. package/src/types/protocol.ts +27 -1
  61. package/src/types/runtime.ts +7 -0
  62. package/src/types/sandbox-environment.ts +12 -1
  63. package/src/types/sandbox.ts +40 -6
  64. package/src/types/workflow-metadata.ts +42 -10
  65. package/src/types/workflow.ts +22 -7
  66. package/src/utils/bundler.ts +6 -1
  67. package/src/workflow-steps/observability.ts +51 -5
  68. package/src/workflow-steps/runner.ts +9 -5
  69. package/src/workflow-steps/types.ts +11 -7
  70. package/src/workflow-steps/workflow.ts +3 -2
@@ -4,6 +4,7 @@
4
4
  import type { SandboxProvider } from "./sandbox.js";
5
5
  import type { AgentMessage } from "./protocol.js";
6
6
  import type { Processor, ProcessorContext, ToolCall } from "../processors/processor.js";
7
+ import type { BoundaryPauseFn } from "../pause/pause-core.js";
7
8
  import type { RequestContext } from "../request-context/request-context.js";
8
9
  /** Configuration for a single MCP server. */
9
10
  export interface McpServerConfig {
@@ -28,6 +29,12 @@ export interface RuntimeOptions {
28
29
  /** Agent id and label for processor context / adapter logs. */
29
30
  agentId?: string;
30
31
  iteration?: number;
32
+ /** The run's pause boundary, threaded from the agent loop so a runtime-driven
33
+ * pre-tool gate (e.g. the ACP `session/request_permission` path through
34
+ * `gateToolCall`) can raise a human-approval `ctx.pause`. The runtime binds it
35
+ * to the current `{ agentId, iteration }` when it builds a ProcessorContext.
36
+ * Absent ⇒ a processor pause throws (no boundary; see `boundProcessorPause`). */
37
+ pause?: BoundaryPauseFn;
31
38
  /** Optional JSON schema for runtimes with native structured-output support. */
32
39
  outputFormat?: {
33
40
  type: "json_schema";
@@ -35,7 +35,7 @@
35
35
  */
36
36
  import type { SandboxProvider } from "./sandbox.js";
37
37
  import type { Workflow } from "../workflow-steps/types.js";
38
- import type { SnapshotConfig } from "./workflow-metadata.js";
38
+ import type { SandboxResources, SnapshotConfig } from "./workflow-metadata.js";
39
39
  export interface SandboxEnvironmentDefinition {
40
40
  name: string;
41
41
  description?: string;
@@ -45,6 +45,11 @@ export interface SandboxEnvironmentDefinition {
45
45
  * useful — an env with no snapshot can't be referenced as a
46
46
  * `bootFrom` on another workflow). */
47
47
  snapshots?: SnapshotConfig;
48
+ /** Which substrate to build this environment on (and any sizing).
49
+ * An env image is provider-specific — a snapshot captured on E2B
50
+ * can't boot on Vercel and vice-versa — so building the E2B base/
51
+ * agent-env requires `resources: { provider: "e2b" }`. */
52
+ resources?: SandboxResources;
48
53
  }
49
54
  /** Sugar over `defineWorkflow` for setup-only workflows that exist to
50
55
  * capture a snapshot. The workflow takes no meaningful input and returns
@@ -21,6 +21,36 @@ export interface SandboxCommandResult {
21
21
  stdout: string;
22
22
  stderr: string;
23
23
  }
24
+ /** A spawned long-lived command with a writable stdin and readable stdout,
25
+ * exposed as byte web-streams. Unlike `commands.run` (which buffers to
26
+ * completion and exposes stdout only via an `onStdout` callback), a duplex
27
+ * handle keeps the process alive and lets the caller WRITE to its stdin —
28
+ * the half `commands.run` cannot provide. It is the transport the ACP client
29
+ * (`AcpClientPeer` over `ndJsonStream`) needs: the agent CLI reads JSON-RPC
30
+ * request frames on stdin and answers on stdout.
31
+ *
32
+ * Only `makeLocalSandboxProvider` implements it. The runner runs IN the
33
+ * sandbox VM and spawns CLIs via the local provider (a plain `child_process`
34
+ * pipe), so a duplex stdin works identically on Vercel/E2B/local. The
35
+ * vercel/e2b providers are the SERVER→sandbox view and never spawn the in-VM
36
+ * CLI, so they leave `spawnDuplex` undefined and callers fall back cleanly. */
37
+ export interface SandboxDuplexProcess {
38
+ /** Subprocess stdin. JSON-RPC request frames are written here. */
39
+ stdin: WritableStream<Uint8Array>;
40
+ /** Subprocess stdout. JSON-RPC response/notification frames arrive here. */
41
+ stdout: ReadableStream<Uint8Array>;
42
+ /** Resolves when the subprocess exits, carrying the captured stderr tail. */
43
+ exited: Promise<{
44
+ exitCode: number;
45
+ stderr: string;
46
+ }>;
47
+ /** Force-terminate the subprocess. */
48
+ kill(): void;
49
+ }
50
+ export interface SandboxSpawnDuplexOptions {
51
+ cwd?: string;
52
+ envs?: Record<string, string>;
53
+ }
24
54
  /**
25
55
  * A sandbox provider implements the RAW provider operations only. It does NOT
26
56
  * implement transient-failure retry/backoff: reconnecting and snapshotting both
@@ -43,6 +73,11 @@ export interface SandboxProvider {
43
73
  cwd?: string;
44
74
  commands: {
45
75
  run(cmd: string, opts?: SandboxCommandRunOptions): Promise<SandboxCommandResult>;
76
+ /** Spawn a long-lived command with a real duplex stdin/stdout. OPTIONAL —
77
+ * implemented only by `makeLocalSandboxProvider` (the in-VM `child_process`
78
+ * view). The vercel/e2b providers (server→sandbox) leave it undefined; an
79
+ * ACP caller that finds it absent falls back to the JSONL transport. */
80
+ spawnDuplex?(cmd: string, opts?: SandboxSpawnDuplexOptions): SandboxDuplexProcess;
46
81
  };
47
82
  files: {
48
83
  write(path: string, content: string): Promise<void>;
@@ -60,12 +95,12 @@ export interface SandboxProvider {
60
95
  snapshotId: string;
61
96
  sizeBytes?: number;
62
97
  }>;
63
- /** Replace the live sandbox's egress policy in place. Vercel implements
64
- * it via `sandbox.update({ networkPolicy })` (2.x) so the server can
65
- * push a freshly resolved policy — with re-minted connector access
66
- * tokens — before each step instead of relying on the policy baked at
67
- * create. Providers whose enforcement lives inside the VM (E2B
68
- * iron-proxy) leave it undefined. */
98
+ /** Replace the live sandbox's egress policy in place — so the server can
99
+ * push a freshly resolved policy (with re-minted connector access tokens)
100
+ * before each step instead of relying on the policy baked at create.
101
+ * Vercel implements it via `sandbox.update({ networkPolicy })` (2.x); E2B
102
+ * via its native `sandbox.updateNetwork(...)`. Providers without a live
103
+ * network-update primitive leave it undefined. */
69
104
  updateNetworkPolicy?(policy: SandboxNetworkPolicy): Promise<void>;
70
105
  }
71
106
  /** Stateless provider-level snapshot deletion — no live sandbox needed,
@@ -8,8 +8,13 @@
8
8
  * Keep this module free of value imports from `types/workflow.ts` or
9
9
  * `workflow-steps/workflow.ts` — it is the cycle-break point.
10
10
  */
11
- import type { SandboxNetworkPolicy, SandboxSize } from "../sandbox.js";
11
+ import type { SandboxNetworkPolicy, SandboxSize, SandboxProviderName } from "../sandbox.js";
12
12
  import type { Processor } from "../processors/processor.js";
13
+ /** The sandbox providers selectable per-workflow. A narrowing of
14
+ * `SandboxProviderName` to the two substrates that execute step workloads —
15
+ * `e2b-desktop` (registry-only, never a workflow's runtime) is deliberately
16
+ * excluded so a workflow can only ask for a substrate that actually runs steps. */
17
+ export type WorkflowSandboxProvider = Extract<SandboxProviderName, "vercel" | "e2b">;
13
18
  /**
14
19
  * Workflow-level metadata read by the server at registration. Lives on
15
20
  * every `Workflow` as `workflow.metadata`, regardless of which form of
@@ -75,11 +80,14 @@ export interface IOSchema {
75
80
  /** Alias kept for backwards source-compatibility with the original
76
81
  * output-only release. New code should prefer `IOSchema`. */
77
82
  export type OutputSchema = IOSchema;
78
- /** Per-connector HTTP request matcher (Tier-2 capability narrowing). The
79
- * iron-proxy / firewall consults these when deciding whether to attach the
80
- * brokered Authorization header to an outbound request. A request to the
81
- * connector host whose method is not in `methods`, or whose path matches no
82
- * entry in `pathPrefixes`, is refused (403) and the token is WITHHELD. */
83
+ /** Per-connector HTTP request matcher (Tier-2 capability narrowing). Vercel's
84
+ * firewall consults these when deciding whether to attach the brokered
85
+ * Authorization header to an outbound request: a request to the connector
86
+ * host whose method is not in `methods`, or whose path matches no entry in
87
+ * `pathPrefixes`, goes out WITHOUT the token. NOT enforced on E2B — its
88
+ * native rules carry only a header transform, no method/path matcher, so the
89
+ * token rides every request to an allowed connector host there (the host
90
+ * allowlist still confines which hosts are reachable). See `toE2bNetwork`. */
83
91
  export interface ConnectorRequestRules {
84
92
  /** Allowed HTTP methods (upper-case). Omit = any method (subject to
85
93
  * `access`). */
@@ -138,10 +146,19 @@ export interface InvokePolicy {
138
146
  * today; kept as its own object so finer controls (disk, gpu, …) can be
139
147
  * added later without reshaping `WorkflowMetadata`. */
140
148
  export interface SandboxResources {
141
- /** Machine size — `small | medium | large`. Maps to provider specs at
142
- * create time (Vercel: 2 / 4 / 8 vCPU, 2048 MB RAM per vCPU). Omit →
143
- * `"small"`. E2B sizing is template-defined and ignores this. */
149
+ /** Machine hardware SKU — one of the `SandboxSize` vCPU strings
150
+ * (`2vcpu-4gb` | `4vcpu-8gb` | `8vcpu-16gb` | `32vcpu-64gb`). Maps to
151
+ * provider specs at create time (Vercel: 2 / 4 / 8 / 32 vCPU, 2048 MB RAM
152
+ * per vCPU). Omit → the smallest SKU. E2B sizing is template-defined and
153
+ * ignores this. */
144
154
  size?: SandboxSize;
155
+ /** Sandbox provider this workflow's runs execute on — `"vercel"` or
156
+ * `"e2b"`. Optional and additive: omit and the run resolves to the
157
+ * platform default (`vercel`). Set explicitly to pin a workflow to a
158
+ * substrate regardless of the deployment default. Snapshot formats are
159
+ * provider-specific, so the resolved provider is stamped on the run row
160
+ * at first dispatch and reused across pause/resume/replay. */
161
+ provider?: WorkflowSandboxProvider;
145
162
  }
146
163
  export interface WorkflowMetadata {
147
164
  /** One-line, human-readable description of what the workflow does.
@@ -163,7 +180,8 @@ export interface WorkflowMetadata {
163
180
  outputSchema?: IOSchema;
164
181
  /** All snapshot config — boot source + capture mode. */
165
182
  snapshots?: SnapshotConfig;
166
- /** Sandbox machine resources (size). Optional; omit → small. */
183
+ /** Sandbox machine resources — size + provider. Optional; omit → smallest
184
+ * SKU on the platform-default provider (`vercel`). */
167
185
  resources?: SandboxResources;
168
186
  processors?: readonly Processor[];
169
187
  /** Connector requirements — providers whose APIs this workflow calls.
@@ -176,6 +194,18 @@ export interface WorkflowMetadata {
176
194
  /** Tier-1 invoke ACL — who may dispatch this connector-brokering workflow.
177
195
  * See `InvokePolicy`. */
178
196
  invokePolicy?: InvokePolicy;
197
+ /** Set by `defineSandboxEnvironment` to mark this workflow as an ENVIRONMENT
198
+ * BUILD — a setup-only workflow whose job is to leave its VM configured and
199
+ * snapshot it (base-env / agent-env). Environment builds build a platform
200
+ * IMAGE and never use the shared factory drive, so the server SKIPS mounting
201
+ * /factory for them: a live Archil mount baked into the captured snapshot
202
+ * fails the NEXT boot's re-mount ("an older Archil process is still running
203
+ * for this mountpoint"), degrading /factory for every workflow booting from
204
+ * that snapshot. Absent on ordinary workflows — which mount /factory exactly
205
+ * as before. Optional + additive: an ABSENT flag contributes nothing to the
206
+ * canonical metadata hash (frozen-metadata rule), so existing workflows are
207
+ * not forced to re-register. */
208
+ environmentBuild?: boolean;
179
209
  }
180
210
  /**
181
211
  * Pull the server-readable declarations off a source object (run-form
@@ -41,9 +41,16 @@ export interface WorkflowCtx<TInput extends Record<string, unknown> = Record<str
41
41
  /** Persist key-value metadata on the run record (e.g. prUrl, planUrl). */
42
42
  setMetadata: (data: Record<string, unknown>) => Promise<void>;
43
43
  /**
44
- * Wrap a named step for observability. Emits step_started / step_completed /
45
- * step_failed lifecycle events with duration. Use for long phases you want
44
+ * Durable named step (ADR-0012). Runs `fn` once and memoises its result to
45
+ * the sandbox state-dir; on a pause-resume re-entry the recorded value is
46
+ * returned and `fn` is NOT re-run (a duration-0 "restored" sub-step). Also
47
+ * emits substep_started / substep_completed / substep_failed lifecycle
48
+ * events with duration — use for long phases you want both durable and
46
49
  * visible on the run's timeline (setup, external API calls, submit).
50
+ *
51
+ * Names must be unique within a step body (they key the memoise file).
52
+ * A body that pauses is never memoised — the resume re-runs it. For
53
+ * cross-process side effects (DB writes, emails) use `invokeChild`.
47
54
  */
48
55
  step<T>(name: string, fn: () => Promise<T>): Promise<T>;
49
56
  /** Pass to `agent({ events: ctx.agentEvents })` to stream agent lifecycle events. */
@@ -101,10 +108,12 @@ export interface WorkflowDefinition<TOutput = unknown, TInput extends Record<str
101
108
  */
102
109
  snapshots?: SnapshotConfig;
103
110
  /**
104
- * Sandbox machine size — `small` (default) | `medium` | `large`. Maps to
105
- * provider machine specs at create (Vercel: 2 / 4 / 8 vCPU, 2048 MB RAM
106
- * per vCPU). Optional; omit for `small`. E2B sizing is template-defined
107
- * and ignores this. Per-invocation `invoke({ size })` overrides it.
111
+ * Sandbox resources — machine SKU (`size`, a `SandboxSize` vCPU string:
112
+ * `2vcpu-4gb` | `4vcpu-8gb` | `8vcpu-16gb` | `32vcpu-64gb`) and `provider`
113
+ * (`vercel` | `e2b`). `size` maps to provider machine specs at create
114
+ * (Vercel: vCPUs, 2048 MB RAM per vCPU); E2B sizing is template-defined and
115
+ * ignores it. Optional; omit → smallest SKU on the default provider.
116
+ * Per-invocation `invoke({ size })` overrides the size.
108
117
  */
109
118
  resources?: SandboxResources;
110
119
  /**
@@ -178,6 +187,13 @@ export interface WorkflowDefinition<TOutput = unknown, TInput extends Record<str
178
187
  * invokePolicy: { users: "owner", workflows: ["nightly-orchestrator"] }
179
188
  */
180
189
  invokePolicy?: InvokePolicy;
190
+ /**
191
+ * Internal — set by `defineSandboxEnvironment`, not by workflow authors.
192
+ * Marks the workflow as an environment build (base-env / agent-env) so the
193
+ * server skips mounting the shared factory drive for its runs (#13). See
194
+ * `WorkflowMetadata.environmentBuild`.
195
+ */
196
+ environmentBuild?: boolean;
181
197
  }
182
198
  /**
183
199
  * Declare a workflow. Two forms; both return a `Workflow` whose
@@ -84,7 +84,8 @@ export interface BundledWorkflow {
84
84
  /** Snapshot config from the workflow definition — `bootFrom` (where to
85
85
  * restore at run start), `save`, `retain`. */
86
86
  snapshots?: SnapshotConfig;
87
- /** Sandbox machine size declared via `defineWorkflow({ resources: { size } })`. */
87
+ /** Sandbox resources declared via `defineWorkflow({ resources })` — machine
88
+ * SKU (`size`) and `provider` (`vercel` | `e2b`). */
88
89
  resources?: SandboxResources;
89
90
  workflowPlan: WorkflowPlan;
90
91
  /** Compact JSON-Schema-shaped description of the workflow's input
@@ -100,6 +101,9 @@ export interface BundledWorkflow {
100
101
  /** Tier-1 invoke ACL — who may dispatch this connector-brokering
101
102
  * workflow (`defineWorkflow({ invokePolicy })`). */
102
103
  invokePolicy?: InvokePolicy;
104
+ /** Set by `defineSandboxEnvironment` — marks an environment build so the
105
+ * server skips the /factory mount for its runs (#13). */
106
+ environmentBuild?: boolean;
103
107
  }
104
108
  /**
105
109
  * Parse the bundled source and assert the default export is a CallExpression
@@ -31,13 +31,19 @@
31
31
  import type { AgentLifecycleEvent } from "../agent/agent-loop.js";
32
32
  import type { AgentEventSink } from "../types/workflow.js";
33
33
  import type { LiveAgentEventEmitter } from "./run-callback.js";
34
+ import type { ScopedMemoize } from "../pause/checkpoint.js";
34
35
  /** One named sub-step (from `ctx.step("name", async () => ...)`).
35
- * Becomes a `workflow_substep_*` lifecycle event on the run timeline. */
36
+ * Becomes a `workflow_substep_*` lifecycle event on the run timeline.
37
+ *
38
+ * `restored` is the durable-step resume case (ADR-0012): the body did NOT
39
+ * re-run — its memoised result was read from the state-dir checkpoint a
40
+ * prior subprocess wrote. Always `durationMs: 0` (no work happened this
41
+ * pass) and never carries `error`. */
36
42
  export interface SubStepEvent {
37
43
  name: string;
38
44
  startedAt: number;
39
45
  durationMs: number;
40
- status: "completed" | "failed";
46
+ status: "completed" | "failed" | "restored";
41
47
  /** Present when status="failed" — the user error's message. */
42
48
  error?: string;
43
49
  }
@@ -64,14 +70,34 @@ export declare class StepObservabilityCollector {
64
70
  private events;
65
71
  private subSteps;
66
72
  private readonly liveEmitter?;
73
+ /** State-dir memoise bound to this step's `step<idx>` scope. Present in
74
+ * the sandbox runner (durable `ctx.step`); absent in-process so unit
75
+ * tests record observability without touching disk. */
76
+ private readonly memoize?;
77
+ /** Sub-step names already used in THIS step body. A durable `ctx.step`
78
+ * memoises by name, so a reused name would alias two distinct bodies
79
+ * onto one checkpoint file — guarded by throwing on reuse (ADR-0012). */
80
+ private readonly usedStepNames;
67
81
  /** Seqs of events the server confirmed via the live route. */
68
82
  private readonly ackedSeqs;
69
83
  /** Promises for in-flight live emits — awaited at snapshot time. */
70
84
  private readonly inFlight;
71
85
  constructor(opts?: {
72
86
  liveEmitter?: LiveAgentEventEmitter;
87
+ memoize?: ScopedMemoize;
73
88
  });
74
89
  readonly setMetadata: (data: Record<string, unknown>) => Promise<void>;
90
+ /**
91
+ * Durable named sub-step (ADR-0012). The body's result is memoised to the
92
+ * state-dir checkpoint keyed `step<idx>.<name>`; on a pause-resume re-entry
93
+ * the value is read from disk and `fn` is NOT re-run (recorded as a
94
+ * duration-0 `"restored"` sub-step). A body that PAUSES (PauseSignal)
95
+ * propagates BEFORE any memoise write — partial work is never stored, so
96
+ * the resume re-runs it.
97
+ *
98
+ * When no memoise store is wired (in-process tests), the body always runs
99
+ * and is recorded as a normal timed `"completed"` / `"failed"` sub-step.
100
+ */
75
101
  readonly step: <T>(name: string, fn: () => Promise<T>) => Promise<T>;
76
102
  readonly agentEvents: AgentEventSink;
77
103
  /** Wait for in-flight live emits to settle (or timeout) so the
@@ -36,14 +36,18 @@ export interface StepContext<TInput = unknown> extends BaseExecutionContext {
36
36
  */
37
37
  setMetadata(data: Record<string, unknown>): Promise<void>;
38
38
  /**
39
- * Wrap a named sub-step for observability. Emits
40
- * `workflow_substep_started` / `workflow_substep_completed` /
41
- * `workflow_substep_failed` lifecycle events on the run timeline with
42
- * the measured `durationMs`. Use for long sub-phases inside one step
43
- * (setup, external API call, submit).
39
+ * Durable named sub-step (ADR-0012). Runs `fn` once and memoises its
40
+ * result to the state-dir checkpoint keyed `step<idx>.<name>`; on a
41
+ * pause-resume re-entry the value is read from disk and `fn` is NOT
42
+ * re-run. Emits `workflow_substep_started` /
43
+ * `workflow_substep_completed` / `workflow_substep_failed` lifecycle
44
+ * events on the run timeline with the measured `durationMs` — a restored
45
+ * sub-step reports `durationMs: 0` and status `"restored"`.
44
46
  *
45
- * The block runs even if observability flushing fails — instrumentation
46
- * never breaks the workflow.
47
+ * Names must be unique within one step body — they key the memoise file,
48
+ * so a reused name throws. A body that pauses (PauseSignal) is never
49
+ * memoised: the resume re-enters and re-runs it. For cross-process side
50
+ * effects (DB writes, emails) use `invokeChild`, not `step`.
47
51
  */
48
52
  step<T>(name: string, fn: () => Promise<T>): Promise<T>;
49
53
  /**
@@ -46,8 +46,9 @@ export interface StepWorkflowDefinition<TInput, TOutput> {
46
46
  networkPolicy?: SandboxNetworkPolicy;
47
47
  placeholders?: Record<string, string>;
48
48
  snapshots?: SnapshotConfig;
49
- /** Sandbox machine size — small (default) | medium | large. Vercel maps
50
- * it to vCPUs; E2B sizing is template-defined. Omit → small. */
49
+ /** Sandbox resources — machine SKU (`size`, a `SandboxSize` vCPU string)
50
+ * and `provider` (`vercel` | `e2b`). Vercel maps `size` to vCPUs; E2B
51
+ * sizing is template-defined. Omit → smallest SKU on the default provider. */
51
52
  resources?: SandboxResources;
52
53
  processors?: readonly Processor[];
53
54
  }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@agent-compose/sdk",
3
- "version": "0.5.8",
3
+ "version": "0.5.9",
4
4
  "description": "Client library for agent-compose — define agents, runtimes, and workflows, and invoke them against an agent-compose server.",
5
5
  "license": "MIT",
6
6
  "repository": {
@@ -60,13 +60,14 @@
60
60
  "vitest": "^3.0.0"
61
61
  },
62
62
  "dependencies": {
63
+ "@agentclientprotocol/sdk": "0.29.0",
63
64
  "@anthropic-ai/claude-agent-sdk": "^0.2.129",
64
65
  "@babel/parser": "^7.29.3",
65
66
  "@babel/types": "^7.29.0",
66
67
  "@e2b/desktop": "^1.1.2",
67
68
  "@vercel/sandbox": "^2.2.0",
68
69
  "ai": "^6.0.175",
69
- "e2b": "^2.3.0",
70
+ "e2b": "2.30.5",
70
71
  "ofetch": "^1.5.1",
71
72
  "openai": "^6.33.0",
72
73
  "p-retry": "^6.2.0",
@@ -81,11 +81,19 @@ Use \`/ac:generate-workflow\` / \`/ac:generate-agent\` to scaffold, then
81
81
 
82
82
  ## Pausing to ask the human — \`agentc pause\`
83
83
 
84
- When you can't or shouldn't proceed without a human, run \`agentc pause\`. It
85
- blocks until they answer on the dashboard, then prints their answer to stdout:
84
+ When you can't or shouldn't proceed without a human, run \`agentc pause\`, then
85
+ **END YOUR TURN**:
86
86
 
87
- ANSWER=$(agentc pause --reason "Notion returned 401 — connect Notion to continue" \\
88
- --option retry --option skip)
87
+ agentc pause --reason "Notion returned 401 — connect Notion to continue" \\
88
+ --option retry --option skip
89
+
90
+ \`agentc pause\` does NOT block and does NOT print the answer. It records your
91
+ question and returns immediately. The moment you end your turn, the run pauses
92
+ (your sandbox is snapshotted and compute stops while the human decides) and the
93
+ human's answer is delivered to you as your **next message** — you pick up
94
+ exactly where you left off, with the answer in hand. So: ask, end your turn,
95
+ and wait. Do NOT keep working, do NOT call more tools, and do NOT mark the task
96
+ complete after pausing.
89
97
 
90
98
  Reach for it the moment you hit — or foresee — any of these:
91
99
  - **A wall only a human can clear:** a 401/403, a missing credential, an
@@ -99,8 +107,8 @@ Reach for it the moment you hit — or foresee — any of these:
99
107
  several valid paths, a conflict with existing state, missing input only they have.
100
108
 
101
109
  You compose the \`--reason\` (the ask) yourself; pass \`--option\` choices when
102
- there are clear ones, omit them for a free-form answer. Read the printed answer
103
- and act on it. Each agent pauses independently — pausing doesn't stop the others.
110
+ there are clear ones, omit them for a free-form answer. Each agent pauses
111
+ independently — pausing doesn't stop the others.
104
112
 
105
113
  ## Credentials
106
114
 
@@ -9,11 +9,13 @@ import { AgentStatusSchema, parseAgentResponse } from "./protocol.js";
9
9
  import type { AgentStatus, AgentMessage } from "./protocol.js";
10
10
  import { randomUUID } from "node:crypto";
11
11
  import type { Processor, ProcessorContext } from "../processors/processor.js";
12
+ import { boundProcessorPause, type BoundaryPauseFn } from "../pause/pause-core.js";
12
13
  import { runProcessorChain } from "../processors/runner.js";
13
14
  import { RequestContext } from "../request-context/request-context.js";
14
15
  import { PauseManager } from "../pause/manager.js";
15
16
  import { PauseSignal, isPauseSignal } from "../pause/pause-core.js";
16
17
  import { SteerDecisionSchema, type SteerDecision, type SteerPayload } from "./steer-control.js";
18
+ import { type LocalPauseRequest } from "./local-pause-request.js";
17
19
 
18
20
  export const DEFAULT_CLAUDE_MODEL = "claude-fable-5";
19
21
 
@@ -55,7 +57,8 @@ export type AgentMessageSummary =
55
57
  | { type: "tool_result"; toolUseId: string; output: string; isError: boolean }
56
58
  | { type: "usage"; inputTokens: number; outputTokens: number; cacheReadTokens: number; cacheCreationTokens: number; durationMs: number; numTurns: number; model?: string }
57
59
  | { type: "done"; sessionId: string }
58
- | { type: "error"; text: string };
60
+ | { type: "error"; text: string }
61
+ | { type: "plan"; entries: { content: string; priority: "high" | "medium" | "low"; status: "pending" | "in_progress" | "completed" }[] };
59
62
 
60
63
  function truncate(value: string): string {
61
64
  return value.length > MESSAGE_PREVIEW_CHARS ? `${value.slice(0, MESSAGE_PREVIEW_CHARS)}…` : value;
@@ -83,6 +86,11 @@ export function summarizeAgentMessage(msg: AgentMessage): AgentMessageSummary {
83
86
  };
84
87
  case "done": return { type: "done", sessionId: msg.sessionId };
85
88
  case "error": return { type: "error", text: truncate(msg.text) };
89
+ // ACP `plan` (WS-C / ADR-0020 Q2). Pure observability — forwarded to
90
+ // onAgentEvent / agent.message; it is NOT an AgentStatus and never feeds
91
+ // self-pause. Pass the entries straight through (the dashboard renders the
92
+ // structured plan; no preview truncation needed — entries are short).
93
+ case "plan": return { type: "plan", entries: msg.entries };
86
94
  }
87
95
  }
88
96
 
@@ -135,19 +143,24 @@ export interface AgentLoopOpts<TResponse = unknown> {
135
143
  * streaming input (ignored otherwise — handled at the runtime).
136
144
  */
137
145
  inbox?: import("./async-queue.js").AsyncQueue<{ text: string; senderName?: string | null }>;
138
- /** PR 7 steer-pause: the boundary calls this to pause the workflow for a
139
- * human steer. agent() builds it (a corePause closed over runId/stepIndex);
140
- * the loop supplies the agentScope so the pauseId is stable across resume.
141
- * Absent ⇒ no steer pause (local tests, non-sandbox callers). */
142
- pause?: <T = unknown>(
143
- req: { reason: string; correlationKey?: string; schema?: z.ZodType<T> },
144
- agentScope: { agentId: string; iteration: number },
145
- ) => Promise<T>;
146
+ /** The run's pause boundary. agent() builds it (a corePause closed over
147
+ * runId/stepIndex); the loop supplies the agentScope so the pauseId is
148
+ * stable across resume. Drives both PR-7 steer-pause AND a processor's
149
+ * `ctx.pause` (human-approval gate). Absent ⇒ no pause (local tests,
150
+ * non-sandbox callers). */
151
+ pause?: BoundaryPauseFn;
146
152
  /** PR 7: take-once read of this agent's pending steer (set by the control
147
153
  * poller). Returns the steer's payload (reason / correlationKey) or null.
148
154
  * The boundary consumes it once per check, and only when no steer is
149
155
  * already staged. */
150
156
  consumeSteerPending?: () => SteerPayload | null;
157
+ /** Take-once read of a local `agentc pause` request this agent dropped in the
158
+ * state dir during its turn (the CLI writes it; see local-pause-request.ts).
159
+ * Returns the request (reason + offered options) or null. The loop turns it
160
+ * into a staged self-pause the next boundary takes — the same snapshot-release
161
+ * path as `needs_input`, but triggered by an explicit `agentc pause` call so
162
+ * it is honored regardless of `mode`. */
163
+ consumeLocalPauseRequest?: () => LocalPauseRequest | null;
151
164
  /** PR 7: `auto` (autonomous, default) or `hitl` (can be steer-paused). The
152
165
  * boundary is inert unless `hitl`. */
153
166
  mode?: "auto" | "hitl";
@@ -180,6 +193,7 @@ export async function agentLoop<TResponse = unknown>(opts: AgentLoopOpts<TRespon
180
193
  retryCount: 0,
181
194
  agentId,
182
195
  iteration,
196
+ pause: boundProcessorPause(opts.pause, { agentId, iteration }),
183
197
  });
184
198
 
185
199
  if (!opts.runtime) throw new Error("agentLoop: opts.runtime is required");
@@ -191,6 +205,10 @@ export async function agentLoop<TResponse = unknown>(opts: AgentLoopOpts<TRespon
191
205
  processors,
192
206
  requestContext,
193
207
  agentId,
208
+ // Thread the pause boundary so a runtime-driven pre-tool gate (e.g. the ACP
209
+ // `session/request_permission` path through CliAgentRunner.gateToolCall) can
210
+ // raise a human-approval `ctx.pause`, not just on the loop's own hooks.
211
+ ...(opts.pause ? { pause: opts.pause } : {}),
194
212
  ...(opts.responseSchema ? { outputFormat: { type: "json_schema" as const, schema: z.toJSONSchema(opts.responseSchema) as Record<string, unknown> } } : {}),
195
213
  });
196
214
 
@@ -222,7 +240,7 @@ export async function agentLoop<TResponse = unknown>(opts: AgentLoopOpts<TRespon
222
240
  let resumed = false;
223
241
  // PR 7: a pending human steer-pause intent. Persisted in loop state so it
224
242
  // survives a resume and the boundary re-issues the pause on re-entry.
225
- let pendingSteerPause: { reason: string; correlationKey: string | null; at: number } | null = null;
243
+ let pendingSteerPause: { reason: string; correlationKey: string | null; at: number; payload?: Record<string, unknown> } | null = null;
226
244
 
227
245
  const pauseManager = new PauseManager(agentId);
228
246
  const restore = await pauseManager.restoreAgentLoop<TResponse>(client);
@@ -342,6 +360,7 @@ export async function agentLoop<TResponse = unknown>(opts: AgentLoopOpts<TRespon
342
360
  {
343
361
  reason: pendingSteerPause.reason,
344
362
  ...(pendingSteerPause.correlationKey !== null ? { correlationKey: pendingSteerPause.correlationKey } : {}),
363
+ ...(pendingSteerPause.payload !== undefined ? { payload: pendingSteerPause.payload } : {}),
345
364
  schema: SteerDecisionSchema,
346
365
  },
347
366
  { agentId, iteration },
@@ -403,6 +422,11 @@ export async function agentLoop<TResponse = unknown>(opts: AgentLoopOpts<TRespon
403
422
  // (otherwise it restarts the task in a fresh session and repeats side effects).
404
423
  sessionId: lastSessionId ?? undefined,
405
424
  iteration: iteration + 1,
425
+ // Thread the loop's abort signal so a runtime that owns a cancellable
426
+ // transport (the ACP path's `session/cancel`, ADR-0020) actually unwinds
427
+ // when the loop aborts — input/output processor `abort`, or any external
428
+ // cancel. Without this the cancel/session-cancel wiring is dead.
429
+ signal: loopAbort.signal,
406
430
  ...(opts.inbox ? { inboxStream: opts.inbox } : {}),
407
431
  })) {
408
432
  // processOutput chain — deny drops the message from accumulation;
@@ -437,6 +461,31 @@ export async function agentLoop<TResponse = unknown>(opts: AgentLoopOpts<TRespon
437
461
  lastResponseText = responseText;
438
462
 
439
463
  let status = parseAgentStatus(responseText);
464
+
465
+ // `agentc pause` — the agent shelled out to the CLI during this turn, which
466
+ // dropped a durable pause-request marker in the state dir. Honor it like a
467
+ // self-pause: stage the pause the NEXT boundary takes, carrying the agent's
468
+ // question as the reason and any offered choices as the pause payload (the
469
+ // dashboard renders them as buttons). Checked BEFORE the settle / needs_input
470
+ // paths and independent of the <status> block — the common case is an agent
471
+ // that called the tool and ended its turn with no status at all, which would
472
+ // otherwise fall through to the empty-output re-prompt below. UNGATED by
473
+ // `mode`: an explicit `agentc pause` is a deliberate ask, not the `needs_input`
474
+ // heuristic that only `hitl` agents may trigger.
475
+ if (opts.pause && pendingSteerPause === null) {
476
+ const localPause = opts.consumeLocalPauseRequest?.() ?? null;
477
+ if (localPause) {
478
+ pendingSteerPause = {
479
+ reason: localPause.reason,
480
+ correlationKey: null,
481
+ ...(localPause.options && localPause.options.length > 0 ? { payload: { options: localPause.options } } : {}),
482
+ at: Date.now(),
483
+ };
484
+ blockerStreak = null; // an explicit ask is not a stuck loop
485
+ continue; // pause fires at the next boundary
486
+ }
487
+ }
488
+
440
489
  // Inline safeParse (instead of letting parseAgentResponse validate)
441
490
  // so a schema failure surfaces via lastResponseValidationError on
442
491
  // the next iteration — the model needs that feedback to fix its