@agent-compose/sdk 0.8.5 → 0.8.7

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (100) hide show
  1. package/README.md +213 -189
  2. package/dist/agent/agent-context.d.ts +3 -3
  3. package/dist/agent/agent-loop.d.ts +6 -5
  4. package/dist/agent/perf-sampler.d.ts +27 -2
  5. package/dist/agent/run-agent.d.ts +1 -1
  6. package/dist/client.d.ts +119 -54
  7. package/dist/directives.d.ts +3 -3
  8. package/dist/display.d.ts +7 -0
  9. package/dist/errors.d.ts +1 -1
  10. package/dist/generated/agentc-commands.d.ts +34 -0
  11. package/dist/index.d.ts +12 -12
  12. package/dist/index.js +771 -204
  13. package/dist/request-context/request-context.d.ts +1 -1
  14. package/dist/runtimes/_cli-agent.d.ts +185 -68
  15. package/dist/runtimes/_reported-model.d.ts +16 -0
  16. package/dist/runtimes/claude-code.d.ts +60 -1
  17. package/dist/runtimes/claude.d.ts +1 -1
  18. package/dist/runtimes/codex.d.ts +94 -6
  19. package/dist/runtimes/codex.mid-turn-hook.test.d.ts +10 -0
  20. package/dist/runtimes/model-report.test.d.ts +14 -0
  21. package/dist/runtimes/openai-desktop.js +741 -200
  22. package/dist/runtimes/opencode.d.ts +48 -11
  23. package/dist/runtimes/opencode.test.d.ts +14 -0
  24. package/dist/sandbox/baked-clis.d.ts +75 -0
  25. package/dist/sandbox/exec-stream.d.ts +1 -2
  26. package/dist/sandbox/network-policy.d.ts +23 -5
  27. package/dist/sandbox.d.ts +4 -2
  28. package/dist/step-invocation/protocol.d.ts +3 -4
  29. package/dist/step-invocation/server.d.ts +2 -2
  30. package/dist/step-invocation/types.d.ts +1 -1
  31. package/dist/types/api-conversations.d.ts +442 -29
  32. package/dist/types/api-factory.d.ts +99 -10
  33. package/dist/types/api-projects.d.ts +521 -0
  34. package/dist/types/api-runs.d.ts +83 -0
  35. package/dist/types/api-scopes.d.ts +32 -3
  36. package/dist/types/conversation-stream.d.ts +5 -0
  37. package/dist/types/execution-context.d.ts +1 -1
  38. package/dist/types/protocol.d.ts +86 -2
  39. package/dist/types/runtime.d.ts +9 -2
  40. package/dist/types/workflow-metadata.d.ts +2 -4
  41. package/dist/types/workflow-plan.d.ts +1 -3
  42. package/dist/utils/bundler.d.ts +23 -0
  43. package/dist/workflow-steps/observability.d.ts +2 -3
  44. package/dist/workflow-steps/runner.d.ts +5 -8
  45. package/dist/workflow-steps/types.d.ts +8 -10
  46. package/dist/workflow-steps/workflow.d.ts +2 -1
  47. package/dist/workflows/engine.d.ts +3 -5
  48. package/dist/workflows/invoke-child.d.ts +2 -2
  49. package/package.json +2 -2
  50. package/src/agent/agent-context.ts +168 -125
  51. package/src/agent/agent-loop.ts +7 -6
  52. package/src/agent/perf-sampler.ts +54 -3
  53. package/src/agent/run-agent.ts +1 -1
  54. package/src/client.ts +226 -71
  55. package/src/directives.ts +3 -3
  56. package/src/display.ts +12 -0
  57. package/src/errors.ts +1 -0
  58. package/src/generated/agentc-commands.ts +571 -0
  59. package/src/index.ts +57 -21
  60. package/src/pause/pause-core.ts +2 -1
  61. package/src/request-context/request-context.ts +1 -1
  62. package/src/runtimes/_cli-agent.ts +318 -122
  63. package/src/runtimes/_reported-model.ts +24 -0
  64. package/src/runtimes/claude-code.ts +195 -12
  65. package/src/runtimes/claude.ts +9 -2
  66. package/src/runtimes/codex.ts +188 -19
  67. package/src/runtimes/opencode.ts +195 -26
  68. package/src/sandbox/baked-clis.ts +86 -0
  69. package/src/sandbox/exec-stream.ts +1 -2
  70. package/src/sandbox/network-policy.ts +51 -7
  71. package/src/sandbox/providers/e2b.ts +3 -3
  72. package/src/sandbox/providers/vercel.ts +6 -6
  73. package/src/sandbox.ts +8 -2
  74. package/src/step-invocation/invoker.ts +2 -6
  75. package/src/step-invocation/protocol.ts +3 -4
  76. package/src/step-invocation/server.ts +2 -2
  77. package/src/types/api-conversations.ts +366 -23
  78. package/src/types/api-factory.ts +95 -10
  79. package/src/types/api-projects.ts +477 -0
  80. package/src/types/api-runs.ts +73 -0
  81. package/src/types/api-scopes.ts +32 -3
  82. package/src/types/conversation-stream.ts +5 -0
  83. package/src/types/execution-context.ts +1 -1
  84. package/src/types/protocol.ts +91 -2
  85. package/src/types/runtime.ts +8 -2
  86. package/src/types/sandbox-environment.ts +1 -2
  87. package/src/types/workflow-metadata.ts +2 -4
  88. package/src/types/workflow-plan.ts +1 -3
  89. package/src/utils/bundler.ts +88 -19
  90. package/src/workflow-steps/observability.ts +2 -3
  91. package/src/workflow-steps/runner.ts +5 -8
  92. package/src/workflow-steps/types.ts +8 -10
  93. package/src/workflow-steps/workflow.ts +2 -1
  94. package/src/workflows/engine.ts +3 -5
  95. package/src/workflows/invoke-child.ts +2 -2
  96. package/dist/generated/verb-synopsis.d.ts +0 -34
  97. package/dist/pause/__tests__/errors.test.d.ts +0 -1
  98. package/dist/pause/__tests__/wrappers.test.d.ts +0 -1
  99. package/dist/step-invocation/__tests__/protocol.test.d.ts +0 -1
  100. package/src/generated/verb-synopsis.ts +0 -544
package/README.md CHANGED
@@ -1,35 +1,34 @@
1
1
  # @agent-compose/sdk
2
2
 
3
- TypeScript SDK for [agent-compose](https://github.com/Layr-Labs/agent-compose). Use it to:
3
+ TypeScript SDK for [agent-compose](https://github.com/Chris-Moller/agent-compose). Use it to:
4
4
 
5
- - **Author workflows** that run agentic LLM loops inside isolated sandboxes
6
- - **Define runtimes** that wrap a coding-CLI tool (Claude Code, OpenAI Desktop, …) into a sandbox-portable agent loop
7
- - **Register, invoke, observe, and cancel** workflows via the HTTP API (`AgentComposeClient`)
5
+ - **Author workflows** that run agent loops inside isolated sandboxes
6
+ - **Use or define runtimes** that drive a coding agent (Claude Code, Codex, Amp, OpenCode, Cursor, Droid, …) as an agent loop
7
+ - **Register, invoke, observe, and cancel** workflows over the HTTP API (`AgentComposeClient`)
8
8
  - **Manage factories, secrets, API keys, and snapshots** programmatically
9
9
 
10
- The hierarchy: a **team** owns one or more **factories** (project containers); each factory owns workflow templates, secrets, and runs. Workflows are versioned per `(factory, name, version)`. New code that doesn't care about factories transparently lands in `default` — every team has one.
10
+ The hierarchy: a **team** (shown as a space in the dashboard) owns one or more **factories**; each factory owns workflow templates, secrets, runs and a drive. Workflows are versioned per `(factory, name, version)`. Every team has a `default` factory, and every call that takes a `factorySlug` falls back to it.
11
11
 
12
12
  ---
13
13
 
14
14
  ## Installation
15
15
 
16
16
  ```bash
17
- npm install @agent-compose/sdk
18
- # peer dep:
19
- npm install zod
17
+ npm install @agent-compose/sdk zod # zod ^4 is a peer dependency
20
18
  ```
21
19
 
20
+ `AgentComposeClient` runs on Node ≥ 20 or Bun. Bundling a workflow for
21
+ registration (`bundleWorkflow`, `agentc register`) needs Bun.
22
+
22
23
  ---
23
24
 
24
25
  ## Authoring a workflow
25
26
 
26
27
  A workflow is a typed chain of discrete steps:
27
28
  `defineWorkflow({ id, input, output }).step(defineStep(...)).build()`.
28
- Every step is a durable replay checkpoint — the engine records each
29
+ Every step is a durable replay checkpoint: the engine records each
29
30
  step's validated output, so a retry or resume picks up after the last
30
- completed step instead of re-running the whole workflow. Pausing for
31
- human input (`ctx.pause`, `ctx.waitForEvent`) only works in this
32
- step-form.
31
+ completed step instead of re-running the whole workflow.
33
32
 
34
33
  ```typescript
35
34
  // my-workflow.ts
@@ -46,14 +45,14 @@ const review = defineStep({
46
45
  output: Output,
47
46
  run: async (ctx) => {
48
47
  const result = await agent({
49
- sandbox: ctx.sandbox, // the run's own VM
48
+ sandbox: ctx.sandbox!, // the run's own VM
50
49
  runtime: claudeRuntime,
51
50
  prompt: `${PROMPT}\n\nRepository: ${ctx.input.repo}`,
52
51
  tools: ["Bash", "Read", "Edit", "Write", "Grep", "Glob"],
53
52
  budget: { turnsPerIteration: 40, maxIterations: 8 },
54
53
  });
55
- await ctx.setMetadata({ summary: result.status?.summary });
56
- return { ok: result.status?.completed ?? false, summary: result.status?.summary };
54
+ await ctx.setMetadata({ summary: result.lastStatus?.summary });
55
+ return { ok: result.lastStatus?.exit_signal ?? false, summary: result.lastStatus?.summary };
57
56
  },
58
57
  });
59
58
 
@@ -61,9 +60,11 @@ export default defineWorkflow({
61
60
  id: "my-workflow",
62
61
  input: Input,
63
62
  output: Output,
64
- // Optional: outbound network rules that the runner sandbox will enforce
65
- // (Vercel only — E2B ignores). Use `$VAR` placeholders for secrets that
66
- // get resolved from the per-workflow secret store at dispatch time.
63
+ // Optional: outbound network rules the runner sandbox enforces. Both
64
+ // providers enforce the host list and inject the headers; only Vercel
65
+ // also enforces per-host method/path rules. `$VAR` placeholders resolve
66
+ // from the workflow's secrets at dispatch, and the value never enters
67
+ // the sandbox.
67
68
  networkPolicy: {
68
69
  allow: {
69
70
  "*": [],
@@ -75,16 +76,16 @@ export default defineWorkflow({
75
76
  .build();
76
77
  ```
77
78
 
78
- The builder chain enforces at compile time that each step's `input`
79
+ The builder chain checks at compile time that each step's `input`
79
80
  schema matches the previous step's `output`, and `.build()` verifies
80
81
  the final step's `output` is the same Zod instance as the workflow's
81
- declared `output`. `networkPolicy` / `placeholders` / `snapshots` on
82
- the definition become metadata the bundler picks up at registration
83
- time.
82
+ declared `output`. `networkPolicy`, `placeholders`, `snapshots`,
83
+ `resources` and the other definition fields become metadata the bundler
84
+ picks up at registration.
84
85
 
85
- > **Legacy run-form.** `defineWorkflow({ run(ctx, sandbox) { … } })` is
86
- > `@deprecated` — it compiles to a single opaque step, so there is no
87
- > per-step replay and pause doesn't work. Don't author new ones.
86
+ The run-form (`defineWorkflow({ run(ctx, sandbox) { … } })`) has been
87
+ removed: `defineWorkflow` throws when it sees a `run` key. Step-form is
88
+ the only shape, and pausing only works there.
88
89
 
89
90
  ### What a step can do with `ctx`
90
91
 
@@ -92,129 +93,145 @@ Each step's `run(ctx)` receives a `StepContext`:
92
93
 
93
94
  ```ts
94
95
  interface StepContext<TInput> {
95
- input: TInput; // validated against the step's `input` schema
96
- run: { id: string };
97
- sandbox?: SandboxProvider; // the run's VM — pass to `agent({ sandbox: ctx.sandbox })`
98
- setMetadata: (data: Record<string, unknown>) => Promise<void>;
99
- step<T>(name: string, fn: () => Promise<T>): Promise<T>; // durable named sub-step
100
- pause<T>(req: PauseRequest<T>): Promise<T>; // wait for a human, durably
96
+ input: TInput; // validated against the step's `input` schema
97
+ stepName: string;
98
+ run: { id: string };
99
+ sandbox?: SandboxProvider; // the run's VM: pass to `agent({ sandbox: ctx.sandbox })`
100
+ abortSignal: AbortSignal; // fires when the run is canceled
101
+ requestContext: RequestContext; // tenant identity, for processors
102
+ agentEvents: AgentEventSink; // `agent({ events: ctx.agentEvents })`
103
+ setMetadata(data: Record<string, unknown>): Promise<void>;
104
+ step<T>(name: string, fn: () => Promise<T>): Promise<T>; // durable named sub-step
105
+ pause<T>(req: PauseRequest<T>): Promise<T>; // wait for a human, durably
101
106
  sleep(durationMs: number): Promise<void>;
102
107
  waitForEvent<T>(req: WaitForEventRequest<T>): Promise<T>;
103
- invokeChild(name, input?, opts?): Promise<RunStatus>; // run another workflow
108
+ requestDecision(req: DecisionRequest): Promise<{ decision: string }>;
109
+ invokeChild(name, input?, opts?): Promise<RunStatus>; // run another workflow
104
110
  }
105
111
  ```
106
112
 
107
113
  `ctx.step("phase-name", () => …)` wraps a phase *within* a step for the
108
- run timeline — it memoises the result (a pause-resume re-entry restores
114
+ run timeline. It memoises the result (a pause-resume re-entry restores
109
115
  it instead of re-running) and emits `workflow_substep_started` /
110
116
  `workflow_substep_completed` / `workflow_substep_failed` lifecycle
111
- events with duration. Use it for setup, external API calls, or anything
112
- you want visible on the dashboard's run detail page.
117
+ events with the duration. Use it for setup, external API calls, or
118
+ anything you want visible on the run page.
113
119
 
114
120
  ### The `agent` loop
115
121
 
116
122
  ```ts
117
123
  agent({
118
- sandbox, // ctx.sandbox — the run's VM
119
- runtime, // claudeRuntime, or your own via createClaudeRuntime / defineRuntime
120
- prompt, // raw markdown — `--- frontmatter ---` is auto-stripped
121
- tools?, // model tool allowlist (defaults inside agentLoop)
122
- budget?, // { turnsPerIteration, maxIterations }
123
- workingDir?, // every shell command runs here
124
- responseSchema?, // zod — when set, the loop demands a `<response>` block on exit
125
- onAgentEvent?, // per-message hook (e.g. wire to telemetry)
126
- onIteration?, // per-iteration hook with parsed `<status>` block
124
+ sandbox, // ctx.sandbox, the run's VM
125
+ runtime, // claudeRuntime, codexRuntime, …, or your own via defineRuntime
126
+ prompt, // raw markdown; a `--- frontmatter ---` block is stripped
127
+ tools?, // model tool allowlist (defaults inside agentLoop; [] disables tools)
128
+ budget?, // { turnsPerIteration, maxIterations }; omitted: no turn cap, up to 3 iterations
129
+ workingDir?, // every shell command runs here (default: the run's dir on the drive)
130
+ responseSchema?, // zod; when set, the loop demands a `<response>` block on exit
131
+ mode?, // "auto" (default) or "hitl" (a human can steer it mid-run)
132
+ events?, // lifecycle sink, e.g. ctx.agentEvents
133
+ onAgentEvent?, // per-message hook
134
+ onIteration?, // per-iteration hook with the parsed `<status>` block
135
+ processors?, // pre/post pipeline around the loop
127
136
  })
128
- // → AgentLoopResult { status?, response? (when responseSchema set), iterations, … }
137
+ // → AgentLoopResult { agentId, label, sessionId, lastStatus, iterations, response? }
129
138
  ```
130
139
 
131
- The protocol is simple: the model emits XML-tagged blocks (`<status>` /
132
- `<response>`) the loop parses. See `sdk/src/agent/protocol-suffix.md` for
133
- the full instructions appended to every prompt.
140
+ The model emits XML-tagged blocks (`<status>`, `<response>`) that the loop
141
+ parses. `lastStatus` is the last `<status>` block: `{ summary, completed,
142
+ blockers, exit_signal }`. See `sdk/src/agent/protocol-suffix.md` for the
143
+ instructions appended to every prompt.
134
144
 
135
145
  ---
136
146
 
137
- ## Defining a runtime
147
+ ## Runtimes
138
148
 
139
- A "runtime" wraps an agent's underlying execution model — usually a coding
140
- CLI like Claude Code or OpenAI Desktop — so `agent` can drive it. The SDK
141
- ships built-ins; you only need a custom one for an exotic provider.
149
+ A runtime wraps an agent's execution model (usually a coding CLI) so
150
+ `agent` can drive it. The SDK ships built-ins; you only need a custom one
151
+ for a provider it does not cover.
142
152
 
143
153
  ### Built-in runtimes
144
154
 
155
+ Each comes as a ready default and a `create…` factory that takes config:
156
+
157
+ | Default | Factory | Drives |
158
+ |---|---|---|
159
+ | `claudeRuntime` | `createClaudeRuntime` | Claude through the Claude Agent SDK, from the runner process |
160
+ | `claudeCodeRuntime` | `createClaudeCodeRuntime` | the Claude Code CLI |
161
+ | `codexRuntime` | `createCodexRuntime` | the Codex CLI |
162
+ | `ampRuntime` | `createAmpRuntime` | Amp |
163
+ | `opencodeRuntime` | `createOpencodeRuntime` | OpenCode |
164
+ | `cursorRuntime` | `createCursorRuntime` | Cursor's `cursor-agent` CLI |
165
+ | `droidRuntime` | `createDroidRuntime` | Factory's `droid` CLI |
166
+ | | `createVercelRuntime` | a model through the Vercel AI SDK |
167
+
168
+ `openAIDesktopRuntime` is not in the package root: it pulls in `sharp` for
169
+ screenshots, whose native binding cannot be cross-compiled. Import it from
170
+ its subpath when you want it:
171
+
145
172
  ```ts
146
- import {
147
- createClaudeRuntime, // factory, takes config
148
- claudeRuntime, // pre-built default (DEFAULT_CLAUDE_MODEL)
149
- ClaudeRunner, // class, if you need to override
150
- } from "@agent-compose/sdk";
151
-
152
- // openAIDesktopRuntime is NOT in the package root (it pulls in `sharp` for
153
- // screenshot capture; the native binding can't be cross-compiled). Import
154
- // directly when you actually want the desktop runtime:
155
- import openAIDesktopRuntime from "@agent-compose/sdk/runtimes/openai-desktop.js";
173
+ import openAIDesktopRuntime from "@agent-compose/sdk/runtimes/openai-desktop";
156
174
  ```
157
175
 
158
176
  ### Custom runtime
159
177
 
160
178
  ```ts
161
- import { defineRuntime, type AgentRuntime } from "@agent-compose/sdk";
162
-
163
- const myRuntime: AgentRuntime = defineRuntime({
164
- create: (sandbox, opts) => {
165
- // Return a ModelExecutionContract — see sdk/src/types/runtime.ts
166
- return {
167
- sendMessage({ prompt, sessionId, signal }) {
168
- // Async generator that yields AgentMessage chunks the loop parses.
169
- return /* … */;
170
- },
171
- };
172
- },
179
+ import { defineRuntime, type AgentMessage } from "@agent-compose/sdk";
180
+
181
+ export default defineRuntime({
182
+ create: (sandbox, opts) => ({
183
+ // Return a ModelExecutionContract; see sdk/src/types/runtime.ts
184
+ async *sendMessage({ prompt, sessionId, signal }): AsyncGenerator<AgentMessage> {
185
+ // Run the model against `sandbox` and yield AgentMessage chunks the loop parses.
186
+ },
187
+ }),
173
188
  });
174
189
  ```
175
190
 
176
- `AgentRuntime` is a tagged record with `create(sandbox, RuntimeOptions) →
177
- ModelExecutionContract`. There is **no `provider` field** on it — the
178
- runtime is bound to the workflow at author time (you pass it to `agent`),
179
- not selected by the server.
191
+ `AgentRuntime` is `{ create(sandbox, RuntimeOptions) → ModelExecutionContract }`.
192
+ There is **no `provider` field** on it: the runtime is bound to the
193
+ workflow at author time (you pass it to `agent`), not selected by the
194
+ server.
180
195
 
181
196
  ---
182
197
 
183
198
  ## Registering a workflow
184
199
 
185
- The `agentc` CLI handles the bundling-and-registration step for you:
200
+ The `agentc` CLI bundles and registers in one step:
186
201
 
187
202
  ```bash
188
203
  agentc register my-workflow.ts -n my-workflow
189
204
  ```
190
205
 
191
- Under the hood that calls `bundleWorkflow(workflowPath)` (resolves imports,
192
- inlines runtime sources via dynamic-require traversal) and `POST
193
- /api/v1/factories/<slug>/templates` with the bundled source. If you need
194
- to drive registration from your own build pipeline, you can do the same
195
- thing via the SDK directly:
206
+ Under the hood that calls `bundleWorkflow(workflowPath)`, which bundles the
207
+ source with `Bun.build`, checks that the default export is a
208
+ `defineWorkflow(...)` workflow, and returns the source with a manifest the
209
+ server verifies against the source hash. Then it posts to
210
+ `/api/v1/factories/<slug>/templates`. To register from your own build
211
+ pipeline (under Bun):
196
212
 
197
213
  ```ts
198
214
  import { AgentComposeClient, bundleWorkflow } from "@agent-compose/sdk";
199
215
 
200
- const client = new AgentComposeClient(
201
- "https://your-server.example.com",
202
- process.env.AGENT_COMPOSE_API_KEY!,
203
- );
216
+ const client = new AgentComposeClient({
217
+ baseUrl: "https://your-server.example.com", // default: AGENT_COMPOSE_URL, else https://api.agentcompose.ai
218
+ apiKey: process.env.AGENT_COMPOSE_API_KEY!, // default: AGENT_COMPOSE_API_KEY
219
+ });
204
220
 
205
221
  const bundled = await bundleWorkflow("./my-workflow.ts");
206
222
  await client.register({
207
- name: "my-workflow",
208
- source: bundled.source,
209
- runtimes: bundled.runtimes, // [{ name, source }] — embedded so the runner has them locally
210
- schedule: "*/30 * * * *", // optional cron
211
- factorySlug: "default", // optional — defaults to "default"
212
- // snapshots, networkPolicy, placeholders — all optional
223
+ name: "my-workflow",
224
+ source: bundled.source,
225
+ manifest: bundled.manifest, // required
226
+ workflowPlan: bundled.workflowPlan,
227
+ schedule: "*/30 * * * *", // optional cron (UTC)
228
+ factorySlug: "default", // optional, defaults to "default"
229
+ // pass bundled.networkPolicy, placeholders, snapshots, resources, … to keep
230
+ // what the workflow declares
213
231
  });
214
232
  ```
215
233
 
216
- `register()` requires the caller's API key to carry the `admin` scope
217
- (or full team-access for legacy keys without scopes).
234
+ `register()` needs the `manage` scope on the caller's key.
218
235
 
219
236
  ---
220
237
 
@@ -223,37 +240,36 @@ await client.register({
223
240
  Two flavours:
224
241
 
225
242
  ```ts
226
- // Fire-and-forget — returns the run id immediately.
243
+ // Fire-and-forget: returns the run id at once.
227
244
  const { id } = await client.invoke("my-workflow", {
228
245
  repo: "owner/repo",
229
246
  });
230
247
 
231
- // Block until the run settles (default 30min timeout, 1s poll).
248
+ // Block until the run settles (default 30 min timeout, 1 s poll).
232
249
  const status = await client.invokeAndWait("my-workflow", { repo: "owner/repo" }, {
233
250
  timeoutMs: 5 * 60_000,
234
251
  pollIntervalMs: 2000,
235
252
  });
236
253
  console.log(status.status); // "success" | "failed" | "abandoned" | "canceled"
237
- console.log(status.output); // workflow's return value
254
+ console.log(status.output); // the workflow's return value
238
255
  ```
239
256
 
240
257
  `output` is the final step's return value, validated against the
241
258
  workflow's declared `output` schema. `setMetadata()` writes to a
242
- separate `metadata` field — useful for "side-channel" facts (PR url, plan
243
- url) without polluting the structured return.
259
+ separate `metadata` field, for side-channel facts (a PR URL, a plan URL)
260
+ that don't belong in the structured return.
244
261
 
245
- `invoke` and `invokeAndWait` both accept `{ factorySlug, snapshots,
246
- networkPolicy, placeholders, parentRunId }` as the
247
- third argument. Per-invocation `snapshots` merges field-by-field with
262
+ `invoke` and `invokeAndWait` take `{ factorySlug, snapshots,
263
+ networkPolicy, placeholders, size, parentRunId, idempotencyKey }` as the
264
+ third argument. Per-invocation `snapshots` merges field by field with
248
265
  the registered default. `factorySlug` defaults to `"default"`.
249
266
 
250
267
  ### Auto parent/child tracing
251
268
 
252
- The SDK detects `process.env.RUN_ID` (set by the runner sandbox on every
253
- dispatch) and automatically threads it as `parentRunId` on subsequent
254
- `invoke()` calls. Workflows that fan out to other workflows get a
255
- parent/child tree in the dashboard for free. Pass `parentRunId: null`
256
- to opt out.
269
+ The SDK reads `process.env.RUN_ID` (set by the runner sandbox on every
270
+ dispatch) and threads it as `parentRunId` on `invoke()` calls. Workflows
271
+ that fan out to other workflows get a parent/child tree in the dashboard
272
+ for free. Pass `parentRunId: null` to opt out.
257
273
 
258
274
  ### Cancelling a run
259
275
 
@@ -261,36 +277,36 @@ to opt out.
261
277
  await client.cancelRun(runId);
262
278
  ```
263
279
 
264
- Idempotent — cancelling an already-terminal run returns the current state
265
- without throwing. The server stamps the run as `canceled`, kills any live
266
- sandboxes, and emits a `run_canceled` event on the stream.
280
+ Idempotent: cancelling a run that already ended returns its current state
281
+ without throwing. The server marks the run `canceled`, emits a
282
+ `run_canceled` event on the stream and kills its sandbox.
267
283
 
268
- ### Streaming live logs
284
+ ### Streaming live events
269
285
 
270
286
  `streamRunLogs` returns an async generator of `RunEvent`s in real time,
271
- re-attaching via SSE under the hood. Pass `lastEventId` (the highest
272
- `seq` you've already processed) to resume after a reconnect.
287
+ over SSE. Pass `lastEventId` (the highest `seq` you have processed) to
288
+ resume after a reconnect.
273
289
 
274
290
  ```ts
275
291
  for await (const ev of client.streamRunLogs(runId, { lastEventId: 0 })) {
276
- console.log(ev.event, ev.seq, ev.data);
292
+ console.log(ev.event, ev.seq);
277
293
  if (ev.event === "run_complete" || ev.event === "run_failed" || ev.event === "run_canceled") {
278
294
  break;
279
295
  }
280
296
  }
281
297
  ```
282
298
 
283
- `AbortSignal` works too — pass `{ signal }` and call `controller.abort()`
284
- to tear the stream down from the caller side.
299
+ `AbortSignal` works too: pass `{ signal }` and call `controller.abort()`
300
+ to close the stream from the caller side.
285
301
 
286
302
  ---
287
303
 
288
304
  ## Factories
289
305
 
290
- Factories are project containers within a team. Each factory has its own
291
- workflow templates, secrets, runs, and (optionally) scoped API keys. New
292
- projects don't need to think about them — `default` is auto-created per
293
- team and is what the SDK falls back to when `factorySlug` is omitted.
306
+ Factories are containers within a team. Each has its own workflow
307
+ templates, secrets, runs and drive, and API keys can be limited to one.
308
+ New projects don't need to think about them: `default` exists in every
309
+ team and is what the SDK uses when `factorySlug` is omitted.
294
310
 
295
311
  ```ts
296
312
  // CRUD on factories
@@ -300,61 +316,67 @@ const f = await client.getFactory("ci-bots");
300
316
  await client.updateFactory("ci-bots", { name: "Continuous-Integration Bots" });
301
317
  await client.deleteFactory("ci-bots");
302
318
 
303
- // Templates list — flat across factories, or scoped to one
304
- const all = await client.listTemplates();
319
+ // Templates: across factories, or scoped to one
320
+ const all = await client.listTemplates();
305
321
  const scoped = await client.listTemplates({ factorySlug: "ci-bots" });
306
322
 
307
- // Register / invoke / secret operations all accept factorySlug
308
- await client.register({ name: "scrape", source, factorySlug: "ci-bots", … });
323
+ // Register, invoke and secret calls all accept factorySlug
309
324
  await client.invoke("scrape", { url: "…" }, { factorySlug: "ci-bots" });
310
325
  await client.setSecret("scrape", "GH_TOKEN", "ghp_…", { factorySlug: "ci-bots" });
311
326
  ```
312
327
 
313
- CLI equivalents: `agentc factory list | create | get | update | delete`,
314
- plus `--factory <slug>` on every other command.
328
+ CLI equivalents: `agentc factory list | create | use | delete | show`,
329
+ plus `--factory <slug>` on most other commands.
315
330
 
316
331
  ---
317
332
 
318
- ## Per-workflow secrets
333
+ ## Secrets
319
334
 
320
- Secrets live in GCP Secret Manager, one row per `(factory, workflow, key)`.
321
- They're injected as env vars into the runner sandbox at dispatch time,
322
- never persisted in the VM. Values are write-only — the API only returns
323
- metadata (key, timestamps).
335
+ Secret values live in GCP Secret Manager and are write-only: the API only
336
+ returns metadata (key, timestamps). A run sees its factory's secrets plus
337
+ its workflow's own, the workflow's winning on a clash. A secret the
338
+ workflow's `networkPolicy` references as `$VAR` is brokered: the platform
339
+ adds it to matching requests at the network layer and the sandbox env holds
340
+ only a placeholder, or nothing. Any other secret is a plain env var in the
341
+ runner sandbox.
324
342
 
325
343
  ```ts
344
+ // Per-workflow
326
345
  await client.setSecret("my-workflow", "ANTHROPIC_API_KEY", process.env.ANTHROPIC_API_KEY!);
327
346
  const list = await client.listSecrets("my-workflow"); // [{ key, createdAt, updatedAt }]
328
347
  await client.deleteSecret("my-workflow", "STALE_KEY");
329
348
 
330
- // Scope to a non-default factory:
331
- await client.setSecret("scrape", "GH_TOKEN", "ghp_…", { factorySlug: "ci-bots" });
349
+ // Factory-wide (every workflow in the factory inherits them)
350
+ await client.setFactorySecret("SLACK_WEBHOOK_URL", "https://hooks.slack.com/…", { factorySlug: "ci-bots" });
332
351
  ```
333
352
 
334
- Mutations require `admin` scope.
353
+ Factory secrets need the `manage` scope. Changing a workflow's secrets
354
+ needs `invoke` plus write access to that workflow.
335
355
 
336
356
  ---
337
357
 
338
358
  ## API keys
339
359
 
340
- Mint and list scoped keys programmatically (requires an `admin`-scoped
341
- caller key). New keys are returned **once**, in the same response as the
342
- metadata — copy the `ac_…` value immediately.
360
+ Mint and list keys programmatically (the caller's key needs `admin`). A new
361
+ key is returned **once**, in the same response as its metadata: copy the
362
+ `ac_…` value at once.
343
363
 
344
364
  ```ts
345
365
  const created = await client.createApiKey({
346
366
  name: "ci-dispatcher",
347
- scopes: ["read", "invoke"],
367
+ scopes: ["read", "invoke"], // omitted = ["read"]
348
368
  expiresAt: new Date(Date.now() + 30 * 86_400_000).toISOString(), // 30 days
349
- // factorySlug: "ci-bots" // optional — scopes the key to a single factory
369
+ // factorySlug: "ci-bots" // optional: limit the key to one factory
350
370
  });
351
- console.log(created.key); // "ac_…" — the only time you'll see this
371
+ console.log(created.key); // "ac_…", the only time you will see it
352
372
 
353
- const all = await client.listApiKeys();
373
+ const page = await client.listApiKeys();
354
374
  ```
355
375
 
356
- CLI equivalent: `agentc keys create <name> --scopes read,invoke
357
- --expires-in 30d`.
376
+ Scopes: `read`, `invoke` (dispatch runs), `manage` (operate a factory:
377
+ templates, secrets, repos) and `admin` (members, roles, keys). A key can
378
+ only mint scopes its holder has. CLI equivalent:
379
+ `agentc keys create <name> --scopes read,invoke --expires-in 30d`.
358
380
 
359
381
  ---
360
382
 
@@ -365,7 +387,7 @@ const usage = await client.getUsage(
365
387
  new Date(Date.now() - 30 * 86_400_000),
366
388
  new Date(),
367
389
  );
368
- // usage.rows: [{ day, runs, sandbox_seconds, … }]
390
+ // usage.data: [{ eventType, unit, total, tags }]
369
391
  ```
370
392
 
371
393
  CLI equivalent: `agentc usage`.
@@ -374,101 +396,103 @@ CLI equivalent: `agentc usage`.
374
396
 
375
397
  ## Snapshots (replay-friendly sandboxes)
376
398
 
377
- Long-running workflows can capture the runner sandbox as a Vercel snapshot
378
- on success (`snapshots: { saveLatest: true }`). Other workflows reference
379
- that snapshot via `snapshots.bootFrom` to boot into the same prepared VM
380
- (deps installed, repo cloned, etc.) instead of repeating setup.
399
+ A workflow can capture its runner sandbox as a snapshot on success
400
+ (`snapshots: { saveLatest: true }`), on Vercel or E2B. Other runs boot from
401
+ it with `snapshots.bootFrom` into the same prepared VM (deps installed, repo
402
+ cloned) instead of repeating the setup.
381
403
 
382
404
  ```ts
383
405
  // Capture per-invocation:
384
406
  await client.invoke("my-workflow", input, { snapshots: { saveLatest: true } });
385
407
 
386
- // Boot from a captured snapshot — pick the id from `agentc snapshot
387
- // list` or the dashboard snapshots page:
408
+ // Boot from a captured snapshot (take the id from `agentc snapshot list`
409
+ // or the dashboard):
388
410
  await client.invoke("my-workflow", input, {
389
411
  snapshots: { bootFrom: { snapshotId: "snap_…" } },
390
412
  });
391
413
 
392
- // Set a default at registration time:
414
+ // Set a default in the definition:
393
415
  defineWorkflow({ id, input, output, snapshots: { saveLatest: true } }).step(…).build();
394
416
 
395
- // Retain every step's snapshot (not just the latest):
417
+ // Keep every step's snapshot (not just the latest):
396
418
  defineWorkflow({ id, input, output, snapshots: { saveLatest: true, retainSteps: true } }).step(…).build();
397
419
 
398
- // Browse / clean up:
399
- const page = await client.listSnapshotsPage({ workflow: "my-workflow", limit: 50 });
420
+ // Browse and clean up:
421
+ const page = await client.listSnapshotsPage({ workflow: "my-workflow", limit: 50 });
400
422
  const snaps = page.data;
401
423
  await client.deleteRunSnapshot(snaps[0].runId, snaps[0].snapshotId);
402
424
  ```
403
425
 
404
- CLI equivalents: `agentc snapshot list` / `agentc snapshot delete <run-id> <snapshot-id>`.
426
+ CLI equivalents: `agentc snapshot list`, `agentc snapshot show <run-id>`,
427
+ `agentc snapshot delete <run-id> [snapshot-id]`.
405
428
 
406
- ### Per-invocation overrides merge field-by-field
429
+ ### Per-invocation overrides merge field by field
407
430
 
408
- The `snapshots` object on `invoke()` is **merged** with the registered template's `snapshots` config — you can override `bootFrom` alone without losing `saveLatest`, or vice versa.
431
+ The `snapshots` object on `invoke()` is **merged** with the registered template's `snapshots` config: you can override `bootFrom` alone without losing `saveLatest`, or the other way round.
409
432
 
410
433
  ---
411
434
 
412
435
  ## Authentication
413
436
 
414
- The SDK accepts a Bearer API key (`ac_…`). Mint one from the dashboard:
415
- sign in at `<server-url>/login`, then **Settings → API Keys → Create key**.
416
-
417
- Default scopes (`read + invoke`) are right for a CI / dispatch caller. Tick
418
- `admin` only if this key needs to register templates, mint other keys, or
419
- manage secrets.
437
+ The SDK sends a Bearer API key (`ac_…`). Mint one in the dashboard: sign
438
+ in, open the space's settings, **Credentials → API keys**. A new key gets
439
+ `read` by default; tick `invoke` to dispatch runs, `manage` to register
440
+ workflows or set secrets, and `admin` only to mint keys or manage members.
420
441
 
421
442
  ```ts
422
- const client = new AgentComposeClient(
423
- process.env.AGENT_COMPOSE_URL!,
424
- process.env.AGENT_COMPOSE_API_KEY!,
425
- );
443
+ const client = new AgentComposeClient({
444
+ baseUrl: process.env.AGENT_COMPOSE_URL,
445
+ apiKey: process.env.AGENT_COMPOSE_API_KEY,
446
+ });
426
447
  ```
427
448
 
428
- The dashboard itself uses the cookie-bound session path; the SDK is for
429
- programmatic / server-to-server callers.
449
+ Both options fall back to those env vars (and the URL to
450
+ `https://api.agentcompose.ai`), so `new AgentComposeClient()` works inside a
451
+ run sandbox. The dashboard itself uses a session cookie; the SDK is for
452
+ programmatic and server-to-server callers.
430
453
 
431
454
  ---
432
455
 
433
- ## Public exports — quick reference
456
+ ## Public exports: quick reference
434
457
 
435
458
  | Export | What |
436
459
  |---|---|
437
- | `defineWorkflow` | Typed step-workflow builder — `defineWorkflow({ id, input, output }).step(...).build()` |
460
+ | `defineWorkflow` | Typed step-workflow builder: `defineWorkflow({ id, input, output }).step(...).build()` |
438
461
  | `defineStep` | Declare one typed workflow step (`{ name, input, output, run }`) |
439
462
  | `defineRuntime` | Wrap an agent execution provider as an `AgentRuntime` |
440
- | `defineSandboxEnvironment` | Sugar for declaring a workflow whose primary purpose is to build a snapshot for others to boot from |
441
- | `agent` / `agentLoop` | Embed an LLM loop inside a workflow |
463
+ | `defineSandboxEnvironment` | Declare a workflow whose purpose is to build a snapshot for others to boot from |
464
+ | `agent` / `agentLoop` | Embed an agent loop inside a workflow |
442
465
  | `runWorkflow` | Local engine for running a workflow in-process (test harness) |
443
- | `bundleWorkflow` | Resolve + inline a workflow's runtime sources for registration |
444
- | `claudeRuntime` / `createClaudeRuntime` / `ClaudeRunner` | Built-in Claude Code runtime + factory |
445
- | `AgentComposeClient` | HTTP client — register, invoke, cancel, stream logs, factories, snapshots, secrets, API keys, usage |
466
+ | `bundleWorkflow` | Bundle and validate a workflow for registration (Bun) |
467
+ | `claudeRuntime`, `codexRuntime`, … and their `create…` factories | Built-in runtimes (table above) |
468
+ | `AgentComposeClient` | HTTP client: register, invoke, cancel, stream, factories, snapshots, secrets, API keys, usage, conversations, sessions, projects |
446
469
  | `AgentComposeError` | Thrown by every non-2xx HTTP response |
447
470
  | `parseAgentStatus` / `parseAgentResponse` / `AgentStatusSchema` / `AgentMessageSchema` | Protocol parsers |
448
- | `parseSseStream` | Generic SSE chunk decoder (used by `streamRunLogs`) |
449
- | `createSandbox` / `reconnectSandbox` / `killAllSandboxes` / `killSandboxById` / `getSandboxQuotas` / `listOwnedSandboxes` / `deleteSandboxSnapshot` | Sandbox-provider helpers (Vercel + E2B) |
471
+ | `parseSseStream` | Generic SSE decoder (used by `streamRunLogs`) |
472
+ | `createSandbox` / `reconnectSandbox` / `killAllSandboxes` / `killSandboxById` / `getSandboxQuotas` / `listOwnedSandboxes` / `deleteSandboxSnapshot` | Sandbox-provider helpers (Vercel and E2B) |
450
473
 
451
- Type exports: `Workflow`, `Step`, `StepContext`, `WorkflowBuilder`,
474
+ Type exports include `Workflow`, `Step`, `StepContext`, `WorkflowBuilder`,
452
475
  `WorkflowHooks`, `AgentBudget`, `AgentRuntime`, `RuntimeOptions`,
453
476
  `ModelExecutionContract`, `McpServerConfig`, `AgentMessage` (and its
454
477
  variants), `AgentStatus`, `RunStatus`, `RegisterResult`, `RunEvent`,
455
478
  `FactoryRow`, `SnapshotListEntry`, `ApiKey`, `ApiKeyCreated`,
456
479
  `UsageRollupRow`, `UsageResponse`, `CancelRunResponse`, `AgentLoopResult`,
457
480
  `AgentOpts`, `SandboxProvider`, `DesktopSandboxProvider`,
458
- `SandboxNetworkPolicy`, `SandboxCreateOpts`, `OwnedSandbox`,
481
+ `SandboxNetworkPolicy`, `SandboxCreateOpts`, `OwnedSandbox` and
459
482
  `BundledWorkflow`.
460
483
 
461
- For the canonical signatures, follow your IDE's go-to-definition into
462
- `@agent-compose/sdk` — `sdk/src/index.ts` is the public surface and the
463
- files it re-exports from carry full inline docstrings.
484
+ For the exact signatures, follow your IDE's go-to-definition into
485
+ `@agent-compose/sdk`: `sdk/src/index.ts` is the public surface, and the
486
+ files it re-exports from carry full docstrings.
464
487
 
465
488
  ---
466
489
 
467
490
  ## Errors
468
491
 
469
- All non-2xx HTTP responses throw `AgentComposeError(status, message)`. The
470
- `message` is the server's `{ error: string }` body when present, falling
471
- back to the HTTP status text:
492
+ Every non-2xx HTTP response throws `AgentComposeError`. It carries `status`
493
+ and a `message` taken from the server's `{ error: string }` body when
494
+ present (else the HTTP status text), plus `code` and `capability` on an
495
+ authorization denial and `retryAfterMs` when the server sent `Retry-After`:
472
496
 
473
497
  ```ts
474
498
  import { AgentComposeError } from "@agent-compose/sdk";
@@ -482,4 +506,4 @@ try {
482
506
  }
483
507
  ```
484
508
 
485
- `invokeAndWait` throws `AgentComposeError(504, …)` on timeout for symmetry.
509
+ `invokeAndWait` throws `AgentComposeError(504, …)` on timeout.