@agent-compose/sdk 0.5.8 → 0.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (78) hide show
  1. package/dist/agent/agent-context.d.ts +1 -1
  2. package/dist/agent/agent-loop.d.ts +14 -12
  3. package/dist/agent/pause-client.d.ts +50 -0
  4. package/dist/agent/pause-client.test.d.ts +1 -0
  5. package/dist/agent/steer-control.d.ts +22 -6
  6. package/dist/client.d.ts +12 -1
  7. package/dist/index.d.ts +7 -5
  8. package/dist/index.js +2379 -1463
  9. package/dist/pause/checkpoint.d.ts +27 -10
  10. package/dist/pause/manager.d.ts +1 -0
  11. package/dist/pause/pause-core.d.ts +23 -0
  12. package/dist/pause/state-dir.d.ts +1 -1
  13. package/dist/processors/builtins.d.ts +20 -1
  14. package/dist/processors/gate-pause.d.ts +46 -0
  15. package/dist/processors/gate-pause.test.d.ts +1 -0
  16. package/dist/processors/index.d.ts +3 -1
  17. package/dist/processors/processor.d.ts +13 -0
  18. package/dist/runtimes/_acp-client.d.ts +140 -0
  19. package/dist/runtimes/_cli-agent.d.ts +155 -3
  20. package/dist/runtimes/amp.d.ts +2 -2
  21. package/dist/runtimes/cli-agent-acp-live.test.d.ts +30 -0
  22. package/dist/runtimes/cli-agent.test.d.ts +22 -6
  23. package/dist/runtimes/codex.d.ts +7 -2
  24. package/dist/runtimes/openai-desktop.js +2365 -1463
  25. package/dist/runtimes/vercel.js +389 -2
  26. package/dist/sandbox.d.ts +113 -19
  27. package/dist/step-invocation/__tests__/background-invoker.test.d.ts +1 -0
  28. package/dist/step-invocation/index.d.ts +2 -1
  29. package/dist/step-invocation/invoker.d.ts +36 -0
  30. package/dist/types/__tests__/environment-build-flag.test.d.ts +1 -0
  31. package/dist/types/__tests__/workflow-metadata-provider.test.d.ts +1 -0
  32. package/dist/types/execution-context.d.ts +0 -8
  33. package/dist/types/protocol.d.ts +32 -1
  34. package/dist/types/runtime.d.ts +14 -0
  35. package/dist/types/sandbox-environment.d.ts +6 -1
  36. package/dist/types/sandbox.d.ts +86 -6
  37. package/dist/types/workflow-metadata.d.ts +40 -10
  38. package/dist/types/workflow.d.ts +22 -6
  39. package/dist/utils/bundler.d.ts +5 -1
  40. package/dist/workflow-steps/observability.d.ts +28 -2
  41. package/dist/workflow-steps/types.d.ts +11 -7
  42. package/dist/workflow-steps/workflow.d.ts +3 -2
  43. package/package.json +3 -2
  44. package/src/agent/agent-context.ts +14 -6
  45. package/src/agent/agent-loop.ts +32 -10
  46. package/src/agent/pause-client.ts +108 -0
  47. package/src/agent/run-agent.ts +9 -4
  48. package/src/agent/steer-control.ts +21 -7
  49. package/src/client.ts +35 -1
  50. package/src/index.ts +20 -2
  51. package/src/pause/checkpoint.ts +33 -14
  52. package/src/pause/manager.ts +2 -2
  53. package/src/pause/pause-core.ts +35 -0
  54. package/src/pause/state-dir.ts +2 -2
  55. package/src/processors/builtins.ts +44 -1
  56. package/src/processors/gate-pause.ts +94 -0
  57. package/src/processors/index.ts +7 -0
  58. package/src/processors/processor.ts +13 -0
  59. package/src/runtimes/_acp-client.ts +516 -0
  60. package/src/runtimes/_cli-agent.ts +416 -3
  61. package/src/runtimes/claude.ts +31 -3
  62. package/src/runtimes/codex.ts +21 -1
  63. package/src/runtimes/vercel.ts +4 -1
  64. package/src/sandbox.ts +426 -56
  65. package/src/step-invocation/index.ts +2 -1
  66. package/src/step-invocation/invoker.ts +195 -84
  67. package/src/types/execution-context.ts +0 -8
  68. package/src/types/protocol.ts +27 -1
  69. package/src/types/runtime.ts +14 -0
  70. package/src/types/sandbox-environment.ts +12 -1
  71. package/src/types/sandbox.ts +84 -6
  72. package/src/types/workflow-metadata.ts +42 -10
  73. package/src/types/workflow.ts +22 -7
  74. package/src/utils/bundler.ts +6 -1
  75. package/src/workflow-steps/observability.ts +51 -5
  76. package/src/workflow-steps/runner.ts +9 -5
  77. package/src/workflow-steps/types.ts +11 -7
  78. package/src/workflow-steps/workflow.ts +3 -2
@@ -27,6 +27,23 @@ export interface AgentMessageToolResult extends AgentMessageBase {
27
27
  toolUseId: string;
28
28
  output: string;
29
29
  isError: boolean;
30
+ /** Structured file diffs produced by the tool call, carried alongside the
31
+ * rendered `output` so a richer renderer can use the structure without
32
+ * re-plumbing the normaliser. Sourced from ACP `tool_call`/`tool_call_update`
33
+ * `diff` content blocks (WS-C / ADR-0020 Q3). Optional and additive:
34
+ * existing producers (claude / vercel / legacy JSONL) omit it. `oldText` is
35
+ * null for a newly created file. */
36
+ diffs?: {
37
+ path: string;
38
+ oldText: string | null;
39
+ newText: string;
40
+ }[];
41
+ /** File locations touched by the tool call (ACP `locations` field), enabling
42
+ * "follow-along" UI. Optional and additive (WS-C / ADR-0020 Q3). */
43
+ locations?: {
44
+ path: string;
45
+ line?: number;
46
+ }[];
30
47
  }
31
48
  export interface AgentMessageDone extends AgentMessageBase {
32
49
  type: "done";
@@ -45,7 +62,21 @@ export interface AgentMessageUsage extends AgentMessageBase {
45
62
  durationMs: number;
46
63
  numTurns: number;
47
64
  }
48
- export type AgentMessage = AgentMessageInit | AgentMessageText | AgentMessageThinking | AgentMessageToolUse | AgentMessageToolResult | AgentMessageDone | AgentMessageError | AgentMessageUsage;
65
+ /** Structured execution plan emitted by an agent (ACP `plan` session update,
66
+ * WS-C / ADR-0020 Q2). Each `plan` notification REPLACES the whole plan — the
67
+ * normaliser emits one `AgentMessagePlan` per notification carrying the entire
68
+ * `entries` array, and downstream treats the latest as authoritative. This is
69
+ * NOT an `AgentStatus` and does not feed self-pause; it is observability only.
70
+ * Additive 9th kind: existing producers never emit it. */
71
+ export interface AgentMessagePlan extends AgentMessageBase {
72
+ type: "plan";
73
+ entries: {
74
+ content: string;
75
+ priority: "high" | "medium" | "low";
76
+ status: "pending" | "in_progress" | "completed";
77
+ }[];
78
+ }
79
+ export type AgentMessage = AgentMessageInit | AgentMessageText | AgentMessageThinking | AgentMessageToolUse | AgentMessageToolResult | AgentMessageDone | AgentMessageError | AgentMessageUsage | AgentMessagePlan;
49
80
  /** Status block the agent emits to signal iteration completion or blockers. */
50
81
  export interface AgentStatus {
51
82
  summary: string;
@@ -4,6 +4,7 @@
4
4
  import type { SandboxProvider } from "./sandbox.js";
5
5
  import type { AgentMessage } from "./protocol.js";
6
6
  import type { Processor, ProcessorContext, ToolCall } from "../processors/processor.js";
7
+ import type { BoundaryPauseFn } from "../pause/pause-core.js";
7
8
  import type { RequestContext } from "../request-context/request-context.js";
8
9
  /** Configuration for a single MCP server. */
9
10
  export interface McpServerConfig {
@@ -28,6 +29,19 @@ export interface RuntimeOptions {
28
29
  /** Agent id and label for processor context / adapter logs. */
29
30
  agentId?: string;
30
31
  iteration?: number;
32
+ /** The platform manual (file conventions, connectors & access, how to pause).
33
+ * `agent()` builds it per-run (`buildAgentContextDoc`) and threads it here so
34
+ * a runtime that supports a system-prompt append (the claude runtime) injects
35
+ * it directly — instead of relying on the agent to `cat` the on-disk
36
+ * AGENTS.md/CLAUDE.md, which the Agent SDK doesn't auto-load and which can
37
+ * fail to write on a read-only/degraded working dir. */
38
+ agentManual?: string;
39
+ /** The run's pause boundary, threaded from the agent loop so a runtime-driven
40
+ * pre-tool gate (e.g. the ACP `session/request_permission` path through
41
+ * `gateToolCall`) can raise a human-approval `ctx.pause`. The runtime binds it
42
+ * to the current `{ agentId, iteration }` when it builds a ProcessorContext.
43
+ * Absent ⇒ a processor pause throws (no boundary; see `boundProcessorPause`). */
44
+ pause?: BoundaryPauseFn;
31
45
  /** Optional JSON schema for runtimes with native structured-output support. */
32
46
  outputFormat?: {
33
47
  type: "json_schema";
@@ -35,7 +35,7 @@
35
35
  */
36
36
  import type { SandboxProvider } from "./sandbox.js";
37
37
  import type { Workflow } from "../workflow-steps/types.js";
38
- import type { SnapshotConfig } from "./workflow-metadata.js";
38
+ import type { SandboxResources, SnapshotConfig } from "./workflow-metadata.js";
39
39
  export interface SandboxEnvironmentDefinition {
40
40
  name: string;
41
41
  description?: string;
@@ -45,6 +45,11 @@ export interface SandboxEnvironmentDefinition {
45
45
  * useful — an env with no snapshot can't be referenced as a
46
46
  * `bootFrom` on another workflow). */
47
47
  snapshots?: SnapshotConfig;
48
+ /** Which substrate to build this environment on (and any sizing).
49
+ * An env image is provider-specific — a snapshot captured on E2B
50
+ * can't boot on Vercel and vice-versa — so building the E2B base/
51
+ * agent-env requires `resources: { provider: "e2b" }`. */
52
+ resources?: SandboxResources;
48
53
  }
49
54
  /** Sugar over `defineWorkflow` for setup-only workflows that exist to
50
55
  * capture a snapshot. The workflow takes no meaningful input and returns
@@ -21,6 +21,59 @@ export interface SandboxCommandResult {
21
21
  stdout: string;
22
22
  stderr: string;
23
23
  }
24
+ /** A long-running command launched in the background (ADR-0028). Unlike
25
+ * `commands.run` (which awaits completion on one connection), a background
26
+ * command keeps running inside the VM independent of the launching
27
+ * connection: it survives `pauseProcess()`/resume and is re-attachable by
28
+ * `pid` after a fresh `Sandbox.connect`. This is what lets the platform
29
+ * freeze an agent mid-turn for a human-in-the-loop pause and continue the
30
+ * SAME process on resume — no re-run.
31
+ *
32
+ * Implemented ONLY by process-resume-capable providers (E2B); the presence
33
+ * of `commands.runBackground` IS the capability flag, paired with
34
+ * `pauseProcess`. Providers without it leave both undefined and pause via
35
+ * `snapshot()` + re-run instead. */
36
+ export interface SandboxBackgroundProcess {
37
+ /** OS pid inside the VM — the durable handle used to reconnect after a
38
+ * pause/resume cycle via `commands.connectProcess(pid)`. */
39
+ pid: number;
40
+ /** Resolve when the process exits, with its buffered result. Live output
41
+ * streams to the `onStdout`/`onStderr` passed at launch / connect time.
42
+ * Does NOT throw on a non-zero exit — the result carries `exitCode`. */
43
+ wait(): Promise<SandboxCommandResult>;
44
+ /** Force-terminate the process. */
45
+ kill(): Promise<void>;
46
+ }
47
+ /** A spawned long-lived command with a writable stdin and readable stdout,
48
+ * exposed as byte web-streams. Unlike `commands.run` (which buffers to
49
+ * completion and exposes stdout only via an `onStdout` callback), a duplex
50
+ * handle keeps the process alive and lets the caller WRITE to its stdin —
51
+ * the half `commands.run` cannot provide. It is the transport the ACP client
52
+ * (`AcpClientPeer` over `ndJsonStream`) needs: the agent CLI reads JSON-RPC
53
+ * request frames on stdin and answers on stdout.
54
+ *
55
+ * Only `makeLocalSandboxProvider` implements it. The runner runs IN the
56
+ * sandbox VM and spawns CLIs via the local provider (a plain `child_process`
57
+ * pipe), so a duplex stdin works identically on Vercel/E2B/local. The
58
+ * vercel/e2b providers are the SERVER→sandbox view and never spawn the in-VM
59
+ * CLI, so they leave `spawnDuplex` undefined and callers fall back cleanly. */
60
+ export interface SandboxDuplexProcess {
61
+ /** Subprocess stdin. JSON-RPC request frames are written here. */
62
+ stdin: WritableStream<Uint8Array>;
63
+ /** Subprocess stdout. JSON-RPC response/notification frames arrive here. */
64
+ stdout: ReadableStream<Uint8Array>;
65
+ /** Resolves when the subprocess exits, carrying the captured stderr tail. */
66
+ exited: Promise<{
67
+ exitCode: number;
68
+ stderr: string;
69
+ }>;
70
+ /** Force-terminate the subprocess. */
71
+ kill(): void;
72
+ }
73
+ export interface SandboxSpawnDuplexOptions {
74
+ cwd?: string;
75
+ envs?: Record<string, string>;
76
+ }
24
77
  /**
25
78
  * A sandbox provider implements the RAW provider operations only. It does NOT
26
79
  * implement transient-failure retry/backoff: reconnecting and snapshotting both
@@ -43,6 +96,21 @@ export interface SandboxProvider {
43
96
  cwd?: string;
44
97
  commands: {
45
98
  run(cmd: string, opts?: SandboxCommandRunOptions): Promise<SandboxCommandResult>;
99
+ /** Spawn a long-lived command with a real duplex stdin/stdout. OPTIONAL —
100
+ * implemented only by `makeLocalSandboxProvider` (the in-VM `child_process`
101
+ * view). The vercel/e2b providers (server→sandbox) leave it undefined; an
102
+ * ACP caller that finds it absent falls back to the JSONL transport. */
103
+ spawnDuplex?(cmd: string, opts?: SandboxSpawnDuplexOptions): SandboxDuplexProcess;
104
+ /** Launch a command in the background and return immediately with a
105
+ * reconnectable handle (ADR-0028). The process survives the launching
106
+ * connection dropping AND a `pauseProcess()`/resume cycle. OPTIONAL —
107
+ * only process-resume providers (E2B) implement it; its presence (paired
108
+ * with `pauseProcess`) is the native-pause capability flag. */
109
+ runBackground?(cmd: string, opts?: SandboxCommandRunOptions): Promise<SandboxBackgroundProcess>;
110
+ /** Re-attach to a background command by `pid` after a fresh
111
+ * `Sandbox.connect` (the resume half of `runBackground`). OPTIONAL,
112
+ * E2B-only. Throws if no process with that pid is running. */
113
+ connectProcess?(pid: number, opts?: Pick<SandboxCommandRunOptions, "onStdout" | "onStderr" | "timeoutMs">): Promise<SandboxBackgroundProcess>;
46
114
  };
47
115
  files: {
48
116
  write(path: string, content: string): Promise<void>;
@@ -60,12 +128,24 @@ export interface SandboxProvider {
60
128
  snapshotId: string;
61
129
  sizeBytes?: number;
62
130
  }>;
63
- /** Replace the live sandbox's egress policy in place. Vercel implements
64
- * it via `sandbox.update({ networkPolicy })` (2.x) so the server can
65
- * push a freshly resolved policy — with re-minted connector access
66
- * tokens — before each step instead of relying on the policy baked at
67
- * create. Providers whose enforcement lives inside the VM (E2B
68
- * iron-proxy) leave it undefined. */
131
+ /** Suspend the live VM in place and return a handle to resume it (ADR-0027).
132
+ * Present ONLY on process-resume-capable providers (E2B via `sandbox.pause()`,
133
+ * returning the sandbox id; resume is `Sandbox.connect(handle)`, which
134
+ * auto-resumes the paused VM). Unlike `snapshot()` — which captures an FS
135
+ * image, kills the origin, and re-runs the step from a fresh sandbox — a
136
+ * process-resume pause FREEZES the live process (zero compute) and continues
137
+ * it exactly where it blocked. The presence of this method IS the capability
138
+ * flag: providers without native VM-suspend leave it undefined and fall back
139
+ * to `snapshot()` + re-run. */
140
+ pauseProcess?(): Promise<{
141
+ resumeHandle: string;
142
+ }>;
143
+ /** Replace the live sandbox's egress policy in place — so the server can
144
+ * push a freshly resolved policy (with re-minted connector access tokens)
145
+ * before each step instead of relying on the policy baked at create.
146
+ * Vercel implements it via `sandbox.update({ networkPolicy })` (2.x); E2B
147
+ * via its native `sandbox.updateNetwork(...)`. Providers without a live
148
+ * network-update primitive leave it undefined. */
69
149
  updateNetworkPolicy?(policy: SandboxNetworkPolicy): Promise<void>;
70
150
  }
71
151
  /** Stateless provider-level snapshot deletion — no live sandbox needed,
@@ -8,8 +8,13 @@
8
8
  * Keep this module free of value imports from `types/workflow.ts` or
9
9
  * `workflow-steps/workflow.ts` — it is the cycle-break point.
10
10
  */
11
- import type { SandboxNetworkPolicy, SandboxSize } from "../sandbox.js";
11
+ import type { SandboxNetworkPolicy, SandboxSize, SandboxProviderName } from "../sandbox.js";
12
12
  import type { Processor } from "../processors/processor.js";
13
+ /** The sandbox providers selectable per-workflow. A narrowing of
14
+ * `SandboxProviderName` to the two substrates that execute step workloads —
15
+ * `e2b-desktop` (registry-only, never a workflow's runtime) is deliberately
16
+ * excluded so a workflow can only ask for a substrate that actually runs steps. */
17
+ export type WorkflowSandboxProvider = Extract<SandboxProviderName, "vercel" | "e2b">;
13
18
  /**
14
19
  * Workflow-level metadata read by the server at registration. Lives on
15
20
  * every `Workflow` as `workflow.metadata`, regardless of which form of
@@ -75,11 +80,14 @@ export interface IOSchema {
75
80
  /** Alias kept for backwards source-compatibility with the original
76
81
  * output-only release. New code should prefer `IOSchema`. */
77
82
  export type OutputSchema = IOSchema;
78
- /** Per-connector HTTP request matcher (Tier-2 capability narrowing). The
79
- * iron-proxy / firewall consults these when deciding whether to attach the
80
- * brokered Authorization header to an outbound request. A request to the
81
- * connector host whose method is not in `methods`, or whose path matches no
82
- * entry in `pathPrefixes`, is refused (403) and the token is WITHHELD. */
83
+ /** Per-connector HTTP request matcher (Tier-2 capability narrowing). Vercel's
84
+ * firewall consults these when deciding whether to attach the brokered
85
+ * Authorization header to an outbound request: a request to the connector
86
+ * host whose method is not in `methods`, or whose path matches no entry in
87
+ * `pathPrefixes`, goes out WITHOUT the token. NOT enforced on E2B — its
88
+ * native rules carry only a header transform, no method/path matcher, so the
89
+ * token rides every request to an allowed connector host there (the host
90
+ * allowlist still confines which hosts are reachable). See `toE2bNetwork`. */
83
91
  export interface ConnectorRequestRules {
84
92
  /** Allowed HTTP methods (upper-case). Omit = any method (subject to
85
93
  * `access`). */
@@ -138,10 +146,19 @@ export interface InvokePolicy {
138
146
  * today; kept as its own object so finer controls (disk, gpu, …) can be
139
147
  * added later without reshaping `WorkflowMetadata`. */
140
148
  export interface SandboxResources {
141
- /** Machine size — `small | medium | large`. Maps to provider specs at
142
- * create time (Vercel: 2 / 4 / 8 vCPU, 2048 MB RAM per vCPU). Omit →
143
- * `"small"`. E2B sizing is template-defined and ignores this. */
149
+ /** Machine hardware SKU — one of the `SandboxSize` vCPU strings
150
+ * (`2vcpu-4gb` | `4vcpu-8gb` | `8vcpu-16gb` | `32vcpu-64gb`). Maps to
151
+ * provider specs at create time (Vercel: 2 / 4 / 8 / 32 vCPU, 2048 MB RAM
152
+ * per vCPU). Omit → the smallest SKU. E2B sizing is template-defined and
153
+ * ignores this. */
144
154
  size?: SandboxSize;
155
+ /** Sandbox provider this workflow's runs execute on — `"vercel"` or
156
+ * `"e2b"`. Optional and additive: omit and the run resolves to the
157
+ * platform default (`vercel`). Set explicitly to pin a workflow to a
158
+ * substrate regardless of the deployment default. Snapshot formats are
159
+ * provider-specific, so the resolved provider is stamped on the run row
160
+ * at first dispatch and reused across pause/resume/replay. */
161
+ provider?: WorkflowSandboxProvider;
145
162
  }
146
163
  export interface WorkflowMetadata {
147
164
  /** One-line, human-readable description of what the workflow does.
@@ -163,7 +180,8 @@ export interface WorkflowMetadata {
163
180
  outputSchema?: IOSchema;
164
181
  /** All snapshot config — boot source + capture mode. */
165
182
  snapshots?: SnapshotConfig;
166
- /** Sandbox machine resources (size). Optional; omit → small. */
183
+ /** Sandbox machine resources — size + provider. Optional; omit → smallest
184
+ * SKU on the platform-default provider (`vercel`). */
167
185
  resources?: SandboxResources;
168
186
  processors?: readonly Processor[];
169
187
  /** Connector requirements — providers whose APIs this workflow calls.
@@ -176,6 +194,18 @@ export interface WorkflowMetadata {
176
194
  /** Tier-1 invoke ACL — who may dispatch this connector-brokering workflow.
177
195
  * See `InvokePolicy`. */
178
196
  invokePolicy?: InvokePolicy;
197
+ /** Set by `defineSandboxEnvironment` to mark this workflow as an ENVIRONMENT
198
+ * BUILD — a setup-only workflow whose job is to leave its VM configured and
199
+ * snapshot it (base-env / agent-env). Environment builds build a platform
200
+ * IMAGE and never use the shared factory drive, so the server SKIPS mounting
201
+ * /factory for them: a live Archil mount baked into the captured snapshot
202
+ * fails the NEXT boot's re-mount ("an older Archil process is still running
203
+ * for this mountpoint"), degrading /factory for every workflow booting from
204
+ * that snapshot. Absent on ordinary workflows — which mount /factory exactly
205
+ * as before. Optional + additive: an ABSENT flag contributes nothing to the
206
+ * canonical metadata hash (frozen-metadata rule), so existing workflows are
207
+ * not forced to re-register. */
208
+ environmentBuild?: boolean;
179
209
  }
180
210
  /**
181
211
  * Pull the server-readable declarations off a source object (run-form
@@ -41,9 +41,16 @@ export interface WorkflowCtx<TInput extends Record<string, unknown> = Record<str
41
41
  /** Persist key-value metadata on the run record (e.g. prUrl, planUrl). */
42
42
  setMetadata: (data: Record<string, unknown>) => Promise<void>;
43
43
  /**
44
- * Wrap a named step for observability. Emits step_started / step_completed /
45
- * step_failed lifecycle events with duration. Use for long phases you want
44
+ * Durable named step (ADR-0012). Runs `fn` once and memoises its result to
45
+ * the sandbox state-dir; on a pause-resume re-entry the recorded value is
46
+ * returned and `fn` is NOT re-run (a duration-0 "restored" sub-step). Also
47
+ * emits substep_started / substep_completed / substep_failed lifecycle
48
+ * events with duration — use for long phases you want both durable and
46
49
  * visible on the run's timeline (setup, external API calls, submit).
50
+ *
51
+ * Names must be unique within a step body (they key the memoise file).
52
+ * A body that pauses is never memoised — the resume re-runs it. For
53
+ * cross-process side effects (DB writes, emails) use `invokeChild`.
47
54
  */
48
55
  step<T>(name: string, fn: () => Promise<T>): Promise<T>;
49
56
  /** Pass to `agent({ events: ctx.agentEvents })` to stream agent lifecycle events. */
@@ -101,10 +108,12 @@ export interface WorkflowDefinition<TOutput = unknown, TInput extends Record<str
101
108
  */
102
109
  snapshots?: SnapshotConfig;
103
110
  /**
104
- * Sandbox machine size — `small` (default) | `medium` | `large`. Maps to
105
- * provider machine specs at create (Vercel: 2 / 4 / 8 vCPU, 2048 MB RAM
106
- * per vCPU). Optional; omit for `small`. E2B sizing is template-defined
107
- * and ignores this. Per-invocation `invoke({ size })` overrides it.
111
+ * Sandbox resources — machine SKU (`size`, a `SandboxSize` vCPU string:
112
+ * `2vcpu-4gb` | `4vcpu-8gb` | `8vcpu-16gb` | `32vcpu-64gb`) and `provider`
113
+ * (`vercel` | `e2b`). `size` maps to provider machine specs at create
114
+ * (Vercel: vCPUs, 2048 MB RAM per vCPU); E2B sizing is template-defined and
115
+ * ignores it. Optional; omit → smallest SKU on the default provider.
116
+ * Per-invocation `invoke({ size })` overrides the size.
108
117
  */
109
118
  resources?: SandboxResources;
110
119
  /**
@@ -178,6 +187,13 @@ export interface WorkflowDefinition<TOutput = unknown, TInput extends Record<str
178
187
  * invokePolicy: { users: "owner", workflows: ["nightly-orchestrator"] }
179
188
  */
180
189
  invokePolicy?: InvokePolicy;
190
+ /**
191
+ * Internal — set by `defineSandboxEnvironment`, not by workflow authors.
192
+ * Marks the workflow as an environment build (base-env / agent-env) so the
193
+ * server skips mounting the shared factory drive for its runs (#13). See
194
+ * `WorkflowMetadata.environmentBuild`.
195
+ */
196
+ environmentBuild?: boolean;
181
197
  }
182
198
  /**
183
199
  * Declare a workflow. Two forms; both return a `Workflow` whose
@@ -84,7 +84,8 @@ export interface BundledWorkflow {
84
84
  /** Snapshot config from the workflow definition — `bootFrom` (where to
85
85
  * restore at run start), `save`, `retain`. */
86
86
  snapshots?: SnapshotConfig;
87
- /** Sandbox machine size declared via `defineWorkflow({ resources: { size } })`. */
87
+ /** Sandbox resources declared via `defineWorkflow({ resources })` — machine
88
+ * SKU (`size`) and `provider` (`vercel` | `e2b`). */
88
89
  resources?: SandboxResources;
89
90
  workflowPlan: WorkflowPlan;
90
91
  /** Compact JSON-Schema-shaped description of the workflow's input
@@ -100,6 +101,9 @@ export interface BundledWorkflow {
100
101
  /** Tier-1 invoke ACL — who may dispatch this connector-brokering
101
102
  * workflow (`defineWorkflow({ invokePolicy })`). */
102
103
  invokePolicy?: InvokePolicy;
104
+ /** Set by `defineSandboxEnvironment` — marks an environment build so the
105
+ * server skips the /factory mount for its runs (#13). */
106
+ environmentBuild?: boolean;
103
107
  }
104
108
  /**
105
109
  * Parse the bundled source and assert the default export is a CallExpression
@@ -31,13 +31,19 @@
31
31
  import type { AgentLifecycleEvent } from "../agent/agent-loop.js";
32
32
  import type { AgentEventSink } from "../types/workflow.js";
33
33
  import type { LiveAgentEventEmitter } from "./run-callback.js";
34
+ import type { ScopedMemoize } from "../pause/checkpoint.js";
34
35
  /** One named sub-step (from `ctx.step("name", async () => ...)`).
35
- * Becomes a `workflow_substep_*` lifecycle event on the run timeline. */
36
+ * Becomes a `workflow_substep_*` lifecycle event on the run timeline.
37
+ *
38
+ * `restored` is the durable-step resume case (ADR-0012): the body did NOT
39
+ * re-run — its memoised result was read from the state-dir checkpoint a
40
+ * prior subprocess wrote. Always `durationMs: 0` (no work happened this
41
+ * pass) and never carries `error`. */
36
42
  export interface SubStepEvent {
37
43
  name: string;
38
44
  startedAt: number;
39
45
  durationMs: number;
40
- status: "completed" | "failed";
46
+ status: "completed" | "failed" | "restored";
41
47
  /** Present when status="failed" — the user error's message. */
42
48
  error?: string;
43
49
  }
@@ -64,14 +70,34 @@ export declare class StepObservabilityCollector {
64
70
  private events;
65
71
  private subSteps;
66
72
  private readonly liveEmitter?;
73
+ /** State-dir memoise bound to this step's `step<idx>` scope. Present in
74
+ * the sandbox runner (durable `ctx.step`); absent in-process so unit
75
+ * tests record observability without touching disk. */
76
+ private readonly memoize?;
77
+ /** Sub-step names already used in THIS step body. A durable `ctx.step`
78
+ * memoises by name, so a reused name would alias two distinct bodies
79
+ * onto one checkpoint file — guarded by throwing on reuse (ADR-0012). */
80
+ private readonly usedStepNames;
67
81
  /** Seqs of events the server confirmed via the live route. */
68
82
  private readonly ackedSeqs;
69
83
  /** Promises for in-flight live emits — awaited at snapshot time. */
70
84
  private readonly inFlight;
71
85
  constructor(opts?: {
72
86
  liveEmitter?: LiveAgentEventEmitter;
87
+ memoize?: ScopedMemoize;
73
88
  });
74
89
  readonly setMetadata: (data: Record<string, unknown>) => Promise<void>;
90
+ /**
91
+ * Durable named sub-step (ADR-0012). The body's result is memoised to the
92
+ * state-dir checkpoint keyed `step<idx>.<name>`; on a pause-resume re-entry
93
+ * the value is read from disk and `fn` is NOT re-run (recorded as a
94
+ * duration-0 `"restored"` sub-step). A body that PAUSES (PauseSignal)
95
+ * propagates BEFORE any memoise write — partial work is never stored, so
96
+ * the resume re-runs it.
97
+ *
98
+ * When no memoise store is wired (in-process tests), the body always runs
99
+ * and is recorded as a normal timed `"completed"` / `"failed"` sub-step.
100
+ */
75
101
  readonly step: <T>(name: string, fn: () => Promise<T>) => Promise<T>;
76
102
  readonly agentEvents: AgentEventSink;
77
103
  /** Wait for in-flight live emits to settle (or timeout) so the
@@ -36,14 +36,18 @@ export interface StepContext<TInput = unknown> extends BaseExecutionContext {
36
36
  */
37
37
  setMetadata(data: Record<string, unknown>): Promise<void>;
38
38
  /**
39
- * Wrap a named sub-step for observability. Emits
40
- * `workflow_substep_started` / `workflow_substep_completed` /
41
- * `workflow_substep_failed` lifecycle events on the run timeline with
42
- * the measured `durationMs`. Use for long sub-phases inside one step
43
- * (setup, external API call, submit).
39
+ * Durable named sub-step (ADR-0012). Runs `fn` once and memoises its
40
+ * result to the state-dir checkpoint keyed `step<idx>.<name>`; on a
41
+ * pause-resume re-entry the value is read from disk and `fn` is NOT
42
+ * re-run. Emits `workflow_substep_started` /
43
+ * `workflow_substep_completed` / `workflow_substep_failed` lifecycle
44
+ * events on the run timeline with the measured `durationMs` — a restored
45
+ * sub-step reports `durationMs: 0` and status `"restored"`.
44
46
  *
45
- * The block runs even if observability flushing fails — instrumentation
46
- * never breaks the workflow.
47
+ * Names must be unique within one step body — they key the memoise file,
48
+ * so a reused name throws. A body that pauses (PauseSignal) is never
49
+ * memoised: the resume re-enters and re-runs it. For cross-process side
50
+ * effects (DB writes, emails) use `invokeChild`, not `step`.
47
51
  */
48
52
  step<T>(name: string, fn: () => Promise<T>): Promise<T>;
49
53
  /**
@@ -46,8 +46,9 @@ export interface StepWorkflowDefinition<TInput, TOutput> {
46
46
  networkPolicy?: SandboxNetworkPolicy;
47
47
  placeholders?: Record<string, string>;
48
48
  snapshots?: SnapshotConfig;
49
- /** Sandbox machine size — small (default) | medium | large. Vercel maps
50
- * it to vCPUs; E2B sizing is template-defined. Omit → small. */
49
+ /** Sandbox resources — machine SKU (`size`, a `SandboxSize` vCPU string)
50
+ * and `provider` (`vercel` | `e2b`). Vercel maps `size` to vCPUs; E2B
51
+ * sizing is template-defined. Omit → smallest SKU on the default provider. */
51
52
  resources?: SandboxResources;
52
53
  processors?: readonly Processor[];
53
54
  }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@agent-compose/sdk",
3
- "version": "0.5.8",
3
+ "version": "0.6.0",
4
4
  "description": "Client library for agent-compose — define agents, runtimes, and workflows, and invoke them against an agent-compose server.",
5
5
  "license": "MIT",
6
6
  "repository": {
@@ -60,13 +60,14 @@
60
60
  "vitest": "^3.0.0"
61
61
  },
62
62
  "dependencies": {
63
+ "@agentclientprotocol/sdk": "0.29.0",
63
64
  "@anthropic-ai/claude-agent-sdk": "^0.2.129",
64
65
  "@babel/parser": "^7.29.3",
65
66
  "@babel/types": "^7.29.0",
66
67
  "@e2b/desktop": "^1.1.2",
67
68
  "@vercel/sandbox": "^2.2.0",
68
69
  "ai": "^6.0.175",
69
- "e2b": "^2.3.0",
70
+ "e2b": "2.30.5",
70
71
  "ofetch": "^1.5.1",
71
72
  "openai": "^6.33.0",
72
73
  "p-retry": "^6.2.0",
@@ -81,11 +81,19 @@ Use \`/ac:generate-workflow\` / \`/ac:generate-agent\` to scaffold, then
81
81
 
82
82
  ## Pausing to ask the human — \`agentc pause\`
83
83
 
84
- When you can't or shouldn't proceed without a human, run \`agentc pause\`. It
85
- blocks until they answer on the dashboard, then prints their answer to stdout:
84
+ When you can't or shouldn't proceed without a human, run \`agentc pause\`, then
85
+ **END YOUR TURN**:
86
86
 
87
- ANSWER=$(agentc pause --reason "Notion returned 401 — connect Notion to continue" \\
88
- --option retry --option skip)
87
+ agentc pause --reason "Notion returned 401 — connect Notion to continue" \\
88
+ --option retry --option skip
89
+
90
+ \`agentc pause\` does NOT block and does NOT print the answer. It records your
91
+ question and returns immediately. The moment you end your turn, the run pauses
92
+ (your sandbox is snapshotted and compute stops while the human decides) and the
93
+ human's answer is delivered to you as your **next message** — you pick up
94
+ exactly where you left off, with the answer in hand. So: ask, end your turn,
95
+ and wait. Do NOT keep working, do NOT call more tools, and do NOT mark the task
96
+ complete after pausing.
89
97
 
90
98
  Reach for it the moment you hit — or foresee — any of these:
91
99
  - **A wall only a human can clear:** a 401/403, a missing credential, an
@@ -99,8 +107,8 @@ Reach for it the moment you hit — or foresee — any of these:
99
107
  several valid paths, a conflict with existing state, missing input only they have.
100
108
 
101
109
  You compose the \`--reason\` (the ask) yourself; pass \`--option\` choices when
102
- there are clear ones, omit them for a free-form answer. Read the printed answer
103
- and act on it. Each agent pauses independently — pausing doesn't stop the others.
110
+ there are clear ones, omit them for a free-form answer. Each agent pauses
111
+ independently — pausing doesn't stop the others.
104
112
 
105
113
  ## Credentials
106
114
 
@@ -9,6 +9,7 @@ import { AgentStatusSchema, parseAgentResponse } from "./protocol.js";
9
9
  import type { AgentStatus, AgentMessage } from "./protocol.js";
10
10
  import { randomUUID } from "node:crypto";
11
11
  import type { Processor, ProcessorContext } from "../processors/processor.js";
12
+ import { boundProcessorPause, type BoundaryPauseFn } from "../pause/pause-core.js";
12
13
  import { runProcessorChain } from "../processors/runner.js";
13
14
  import { RequestContext } from "../request-context/request-context.js";
14
15
  import { PauseManager } from "../pause/manager.js";
@@ -55,7 +56,8 @@ export type AgentMessageSummary =
55
56
  | { type: "tool_result"; toolUseId: string; output: string; isError: boolean }
56
57
  | { type: "usage"; inputTokens: number; outputTokens: number; cacheReadTokens: number; cacheCreationTokens: number; durationMs: number; numTurns: number; model?: string }
57
58
  | { type: "done"; sessionId: string }
58
- | { type: "error"; text: string };
59
+ | { type: "error"; text: string }
60
+ | { type: "plan"; entries: { content: string; priority: "high" | "medium" | "low"; status: "pending" | "in_progress" | "completed" }[] };
59
61
 
60
62
  function truncate(value: string): string {
61
63
  return value.length > MESSAGE_PREVIEW_CHARS ? `${value.slice(0, MESSAGE_PREVIEW_CHARS)}…` : value;
@@ -83,6 +85,11 @@ export function summarizeAgentMessage(msg: AgentMessage): AgentMessageSummary {
83
85
  };
84
86
  case "done": return { type: "done", sessionId: msg.sessionId };
85
87
  case "error": return { type: "error", text: truncate(msg.text) };
88
+ // ACP `plan` (WS-C / ADR-0020 Q2). Pure observability — forwarded to
89
+ // onAgentEvent / agent.message; it is NOT an AgentStatus and never feeds
90
+ // self-pause. Pass the entries straight through (the dashboard renders the
91
+ // structured plan; no preview truncation needed — entries are short).
92
+ case "plan": return { type: "plan", entries: msg.entries };
86
93
  }
87
94
  }
88
95
 
@@ -135,14 +142,12 @@ export interface AgentLoopOpts<TResponse = unknown> {
135
142
  * streaming input (ignored otherwise — handled at the runtime).
136
143
  */
137
144
  inbox?: import("./async-queue.js").AsyncQueue<{ text: string; senderName?: string | null }>;
138
- /** PR 7 steer-pause: the boundary calls this to pause the workflow for a
139
- * human steer. agent() builds it (a corePause closed over runId/stepIndex);
140
- * the loop supplies the agentScope so the pauseId is stable across resume.
141
- * Absent ⇒ no steer pause (local tests, non-sandbox callers). */
142
- pause?: <T = unknown>(
143
- req: { reason: string; correlationKey?: string; schema?: z.ZodType<T> },
144
- agentScope: { agentId: string; iteration: number },
145
- ) => Promise<T>;
145
+ /** The run's pause boundary. agent() builds it (a corePause closed over
146
+ * runId/stepIndex); the loop supplies the agentScope so the pauseId is
147
+ * stable across resume. Drives both PR-7 steer-pause AND a processor's
148
+ * `ctx.pause` (human-approval gate). Absent ⇒ no pause (local tests,
149
+ * non-sandbox callers). */
150
+ pause?: BoundaryPauseFn;
146
151
  /** PR 7: take-once read of this agent's pending steer (set by the control
147
152
  * poller). Returns the steer's payload (reason / correlationKey) or null.
148
153
  * The boundary consumes it once per check, and only when no steer is
@@ -180,6 +185,7 @@ export async function agentLoop<TResponse = unknown>(opts: AgentLoopOpts<TRespon
180
185
  retryCount: 0,
181
186
  agentId,
182
187
  iteration,
188
+ pause: boundProcessorPause(opts.pause, { agentId, iteration }),
183
189
  });
184
190
 
185
191
  if (!opts.runtime) throw new Error("agentLoop: opts.runtime is required");
@@ -191,6 +197,10 @@ export async function agentLoop<TResponse = unknown>(opts: AgentLoopOpts<TRespon
191
197
  processors,
192
198
  requestContext,
193
199
  agentId,
200
+ // Thread the pause boundary so a runtime-driven pre-tool gate (e.g. the ACP
201
+ // `session/request_permission` path through CliAgentRunner.gateToolCall) can
202
+ // raise a human-approval `ctx.pause`, not just on the loop's own hooks.
203
+ ...(opts.pause ? { pause: opts.pause } : {}),
194
204
  ...(opts.responseSchema ? { outputFormat: { type: "json_schema" as const, schema: z.toJSONSchema(opts.responseSchema) as Record<string, unknown> } } : {}),
195
205
  });
196
206
 
@@ -222,7 +232,7 @@ export async function agentLoop<TResponse = unknown>(opts: AgentLoopOpts<TRespon
222
232
  let resumed = false;
223
233
  // PR 7: a pending human steer-pause intent. Persisted in loop state so it
224
234
  // survives a resume and the boundary re-issues the pause on re-entry.
225
- let pendingSteerPause: { reason: string; correlationKey: string | null; at: number } | null = null;
235
+ let pendingSteerPause: { reason: string; correlationKey: string | null; at: number; payload?: Record<string, unknown> } | null = null;
226
236
 
227
237
  const pauseManager = new PauseManager(agentId);
228
238
  const restore = await pauseManager.restoreAgentLoop<TResponse>(client);
@@ -342,6 +352,7 @@ export async function agentLoop<TResponse = unknown>(opts: AgentLoopOpts<TRespon
342
352
  {
343
353
  reason: pendingSteerPause.reason,
344
354
  ...(pendingSteerPause.correlationKey !== null ? { correlationKey: pendingSteerPause.correlationKey } : {}),
355
+ ...(pendingSteerPause.payload !== undefined ? { payload: pendingSteerPause.payload } : {}),
345
356
  schema: SteerDecisionSchema,
346
357
  },
347
358
  { agentId, iteration },
@@ -403,6 +414,11 @@ export async function agentLoop<TResponse = unknown>(opts: AgentLoopOpts<TRespon
403
414
  // (otherwise it restarts the task in a fresh session and repeats side effects).
404
415
  sessionId: lastSessionId ?? undefined,
405
416
  iteration: iteration + 1,
417
+ // Thread the loop's abort signal so a runtime that owns a cancellable
418
+ // transport (the ACP path's `session/cancel`, ADR-0020) actually unwinds
419
+ // when the loop aborts — input/output processor `abort`, or any external
420
+ // cancel. Without this the cancel/session-cancel wiring is dead.
421
+ signal: loopAbort.signal,
406
422
  ...(opts.inbox ? { inboxStream: opts.inbox } : {}),
407
423
  })) {
408
424
  // processOutput chain — deny drops the message from accumulation;
@@ -437,6 +453,12 @@ export async function agentLoop<TResponse = unknown>(opts: AgentLoopOpts<TRespon
437
453
  lastResponseText = responseText;
438
454
 
439
455
  let status = parseAgentStatus(responseText);
456
+
457
+ // ADR-0028 — `agentc pause` is no longer a marker the loop consumes here. It
458
+ // is a server operation: the CLI blocks on the pause API and the run is
459
+ // frozen in place by the step activity, with no loop involvement. The steer
460
+ // (human-driven) and `needs_input` (self-pause) paths below are unaffected.
461
+
440
462
  // Inline safeParse (instead of letting parseAgentResponse validate)
441
463
  // so a schema failure surfaces via lastResponseValidationError on
442
464
  // the next iteration — the model needs that feedback to fix its