@agent-compose/sdk 0.1.0 → 0.2.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +276 -94
- package/dist/client.d.ts +178 -21
- package/dist/index.d.ts +2 -1
- package/dist/index.js +146 -21
- package/dist/runtimes/openai-desktop.js +145 -21
- package/dist/sse.d.ts +25 -0
- package/dist/types/events.d.ts +5 -0
- package/dist/types/sandbox-environment.d.ts +13 -12
- package/dist/types/workflow.d.ts +13 -11
- package/dist/utils/bundler.d.ts +6 -5
- package/package.json +1 -1
package/README.md
CHANGED
|
@@ -1,6 +1,10 @@
|
|
|
1
1
|
# @agent-compose/sdk
|
|
2
2
|
|
|
3
|
-
|
|
3
|
+
TypeScript SDK for agent-compose. Use it to:
|
|
4
|
+
|
|
5
|
+
- **Author workflows** that run agentic LLM loops inside isolated sandboxes
|
|
6
|
+
- **Define runtimes** that wrap a coding-CLI tool (Claude Code, OpenAI Desktop, …) into a sandbox-portable agent loop
|
|
7
|
+
- **Register, invoke, and observe** workflows via the HTTP API (`AgentComposeClient`)
|
|
4
8
|
|
|
5
9
|
---
|
|
6
10
|
|
|
@@ -14,139 +18,317 @@ npm install zod
|
|
|
14
18
|
|
|
15
19
|
---
|
|
16
20
|
|
|
17
|
-
##
|
|
21
|
+
## Authoring a workflow
|
|
22
|
+
|
|
23
|
+
A workflow is `async (ctx, sandbox) => T`. Two positional args:
|
|
24
|
+
|
|
25
|
+
- **`ctx`** carries the run identity (`run.id`), the caller's `input`, plus
|
|
26
|
+
observability helpers (`setMetadata`, `step`).
|
|
27
|
+
- **`sandbox`** is a capability the engine constructs once for the run —
|
|
28
|
+
pass it to `runAgent({ sandbox, ... })` and to any helper that takes a
|
|
29
|
+
`SandboxProvider` (file writers, git utilities, command runners).
|
|
18
30
|
|
|
19
31
|
```typescript
|
|
20
|
-
// my-
|
|
21
|
-
import {
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
|
|
32
|
+
// my-workflow.ts
|
|
33
|
+
import { defineWorkflow, runAgent, claudeRuntime } from "@agent-compose/sdk";
|
|
34
|
+
import PROMPT from "./prompt.md" with { type: "text" };
|
|
35
|
+
|
|
36
|
+
export default defineWorkflow({
|
|
37
|
+
async run(ctx, sandbox) {
|
|
38
|
+
const repo = (ctx.input?.repo as string | undefined) ?? "owner/repo";
|
|
39
|
+
|
|
40
|
+
const result = await runAgent({
|
|
41
|
+
sandbox,
|
|
42
|
+
runtime: claudeRuntime,
|
|
43
|
+
prompt: PROMPT,
|
|
44
|
+
promptVars: { REPO: repo },
|
|
45
|
+
tools: ["Bash", "Read", "Edit", "Write", "Grep", "Glob"],
|
|
46
|
+
budget: { turnsPerIteration: 40, maxIterations: 8 },
|
|
47
|
+
});
|
|
48
|
+
|
|
49
|
+
await ctx.setMetadata({ summary: result.status?.summary });
|
|
50
|
+
return { ok: result.status?.completed ?? false };
|
|
51
|
+
},
|
|
52
|
+
// Optional: outbound network rules that the runner sandbox will enforce
|
|
53
|
+
// (Vercel only — E2B ignores). Use `$VAR` placeholders for secrets that
|
|
54
|
+
// get resolved from the per-workflow secret store at dispatch time.
|
|
55
|
+
networkPolicy: {
|
|
56
|
+
allow: {
|
|
57
|
+
"*": [],
|
|
58
|
+
"api.anthropic.com": [{ transform: [{ headers: { "x-api-key": "$ANTHROPIC_API_KEY" } }] }],
|
|
59
|
+
},
|
|
60
|
+
},
|
|
28
61
|
});
|
|
29
62
|
```
|
|
30
63
|
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
64
|
+
`defineWorkflow` is a thin sugar — it returns the bare `run` function with
|
|
65
|
+
`networkPolicy` / `placeholders` / `snapshot` / `saveSnapshot` attached as
|
|
66
|
+
metadata that the bundler picks up at registration time. A plain
|
|
67
|
+
`export default async (ctx, sandbox) => {...}` is also valid; you just lose
|
|
68
|
+
the metadata channel.
|
|
69
|
+
|
|
70
|
+
### What the workflow can do with `ctx`
|
|
71
|
+
|
|
72
|
+
```ts
|
|
73
|
+
interface WorkflowCtx {
|
|
74
|
+
run: { id: string };
|
|
75
|
+
input?: Record<string, unknown>;
|
|
76
|
+
setMetadata: (data: Record<string, unknown>) => Promise<void>;
|
|
77
|
+
step<T>(name: string, fn: () => Promise<T>): Promise<T>;
|
|
78
|
+
}
|
|
79
|
+
```
|
|
34
80
|
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
81
|
+
`step("phase-name", () => …)` wraps a phase for the run timeline — emits
|
|
82
|
+
`step_started` / `step_completed` / `step_failed` lifecycle events with
|
|
83
|
+
duration. Use it for setup, external API calls, or anything you want
|
|
84
|
+
visible on the dashboard's run detail page.
|
|
85
|
+
|
|
86
|
+
### The `runAgent` loop
|
|
87
|
+
|
|
88
|
+
```ts
|
|
89
|
+
runAgent({
|
|
90
|
+
sandbox, // the workflow's sandbox arg
|
|
91
|
+
runtime, // claudeRuntime, or your own via createClaudeRuntime / defineRuntime
|
|
92
|
+
prompt, // raw markdown — `--- frontmatter ---` is auto-stripped
|
|
93
|
+
promptVars?, // {{VAR}} substitutions; WORKING_DIR + DIFF_BASE auto-populate
|
|
94
|
+
tools?, // model tool allowlist (defaults inside agentLoop)
|
|
95
|
+
budget?, // { turnsPerIteration, maxIterations }
|
|
96
|
+
workingDir?, // every shell command runs here
|
|
97
|
+
responseSchema?, // zod — when set, the loop demands a `<response>` block on exit
|
|
98
|
+
onAgentEvent?, // per-message hook (e.g. wire to telemetry)
|
|
99
|
+
onIteration?, // per-iteration hook with parsed `<status>` block
|
|
100
|
+
})
|
|
101
|
+
// → AgentLoopResult { status?, response? (when responseSchema set), iterations, … }
|
|
45
102
|
```
|
|
46
103
|
|
|
104
|
+
The protocol is simple: the model emits XML-tagged blocks (`<status>` /
|
|
105
|
+
`<response>`) the loop parses. See `sdk/src/agent/protocol-suffix.md` for
|
|
106
|
+
the full instructions appended to every prompt.
|
|
107
|
+
|
|
47
108
|
---
|
|
48
109
|
|
|
49
|
-
##
|
|
110
|
+
## Defining a runtime
|
|
50
111
|
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
112
|
+
A "runtime" wraps an agent's underlying execution model — usually a coding
|
|
113
|
+
CLI like Claude Code or OpenAI Desktop — so `runAgent` can drive it. The SDK
|
|
114
|
+
ships built-ins; you only need a custom one for an exotic provider.
|
|
54
115
|
|
|
55
|
-
|
|
56
|
-
const planner = await ctx.spawnAgent("planner", {
|
|
57
|
-
data: { task: ctx.input.task },
|
|
58
|
-
});
|
|
116
|
+
### Built-in runtimes
|
|
59
117
|
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
|
|
118
|
+
```ts
|
|
119
|
+
import {
|
|
120
|
+
createClaudeRuntime, // factory, takes config
|
|
121
|
+
claudeRuntime, // pre-built default (DEFAULT_CLAUDE_MODEL)
|
|
122
|
+
ClaudeRunner, // class, if you need to override
|
|
123
|
+
} from "@agent-compose/sdk";
|
|
64
124
|
|
|
65
|
-
|
|
66
|
-
|
|
125
|
+
// openAIDesktopRuntime is NOT in the package root (it pulls in `sharp` for
|
|
126
|
+
// screenshot capture; the native binding can't be cross-compiled). Import
|
|
127
|
+
// directly when you actually want the desktop runtime:
|
|
128
|
+
import openAIDesktopRuntime from "@agent-compose/sdk/runtimes/openai-desktop.js";
|
|
129
|
+
```
|
|
67
130
|
|
|
68
|
-
|
|
131
|
+
### Custom runtime
|
|
132
|
+
|
|
133
|
+
```ts
|
|
134
|
+
import { defineRuntime, type AgentRuntime } from "@agent-compose/sdk";
|
|
135
|
+
|
|
136
|
+
const myRuntime: AgentRuntime = defineRuntime({
|
|
137
|
+
create: (sandbox, opts) => {
|
|
138
|
+
// Return a ModelExecutionContract — see sdk/src/types/runtime.ts
|
|
139
|
+
return {
|
|
140
|
+
sendMessage({ prompt, sessionId, signal }) {
|
|
141
|
+
// Async generator that yields AgentMessage chunks the loop parses.
|
|
142
|
+
return /* … */;
|
|
143
|
+
},
|
|
144
|
+
};
|
|
145
|
+
},
|
|
146
|
+
});
|
|
69
147
|
```
|
|
70
148
|
|
|
149
|
+
`AgentRuntime` is a tagged record with `create(sandbox, RuntimeOptions) →
|
|
150
|
+
ModelExecutionContract`. There is **no `provider` field** on it — the
|
|
151
|
+
runtime is bound to the workflow at author time (you pass it to `runAgent`),
|
|
152
|
+
not selected by the server.
|
|
153
|
+
|
|
71
154
|
---
|
|
72
155
|
|
|
73
|
-
##
|
|
156
|
+
## Registering a workflow
|
|
74
157
|
|
|
75
|
-
`
|
|
158
|
+
The `agentc` CLI handles the bundling-and-registration step for you:
|
|
76
159
|
|
|
77
|
-
```
|
|
78
|
-
|
|
160
|
+
```bash
|
|
161
|
+
agentc register my-workflow.ts -n my-workflow
|
|
162
|
+
```
|
|
163
|
+
|
|
164
|
+
Under the hood that calls `bundleWorkflow(workflowPath)` (resolves imports,
|
|
165
|
+
inlines runtime sources via dynamic-require traversal) and `POST /api/v1/templates`
|
|
166
|
+
with the bundled source. If you need to drive registration from your own
|
|
167
|
+
build pipeline, you can do the same thing via the SDK directly:
|
|
168
|
+
|
|
169
|
+
```ts
|
|
170
|
+
import { AgentComposeClient, bundleWorkflow } from "@agent-compose/sdk";
|
|
79
171
|
|
|
80
172
|
const client = new AgentComposeClient(
|
|
81
|
-
"
|
|
82
|
-
process.env.
|
|
173
|
+
"https://your-server.example.com",
|
|
174
|
+
process.env.AGENT_COMPOSE_API_KEY!,
|
|
83
175
|
);
|
|
84
176
|
|
|
85
|
-
|
|
177
|
+
const bundled = await bundleWorkflow("./my-workflow.ts");
|
|
86
178
|
await client.register({
|
|
87
|
-
name:
|
|
88
|
-
|
|
89
|
-
//
|
|
179
|
+
name: "my-workflow",
|
|
180
|
+
source: bundled.source,
|
|
181
|
+
runtimes: bundled.runtimes, // [{ name, source }] — embedded so the runner has them locally
|
|
182
|
+
schedule: "*/30 * * * *", // optional cron
|
|
183
|
+
// snapshot, saveSnapshot, networkPolicy, placeholders — all optional
|
|
90
184
|
});
|
|
91
185
|
```
|
|
92
186
|
|
|
93
|
-
|
|
94
|
-
-
|
|
95
|
-
- `./planner/index.ts`
|
|
96
|
-
- `./agents/planner.ts`
|
|
97
|
-
- `./agents/planner/index.ts`
|
|
98
|
-
|
|
99
|
-
For runtime `"claude"`, it looks for:
|
|
100
|
-
- `./claude.ts`
|
|
101
|
-
- `./runtimes/claude.ts`
|
|
187
|
+
`register()` requires the caller's API key to carry the `admin` scope
|
|
188
|
+
(or full team-access for legacy keys without scopes).
|
|
102
189
|
|
|
103
190
|
---
|
|
104
191
|
|
|
105
|
-
##
|
|
192
|
+
## Invoking a workflow
|
|
106
193
|
|
|
107
|
-
|
|
108
|
-
|
|
194
|
+
Two flavours:
|
|
195
|
+
|
|
196
|
+
```ts
|
|
197
|
+
// Fire-and-forget — returns the run id immediately.
|
|
109
198
|
const { id } = await client.invoke("my-workflow", {
|
|
110
|
-
|
|
199
|
+
repo: "owner/repo",
|
|
111
200
|
});
|
|
112
201
|
|
|
113
|
-
//
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
|
|
118
|
-
|
|
202
|
+
// Block until the run settles (default 30min timeout, 1s poll).
|
|
203
|
+
const status = await client.invokeAndWait("my-workflow", { repo: "owner/repo" }, {
|
|
204
|
+
timeoutMs: 5 * 60_000,
|
|
205
|
+
pollIntervalMs: 2000,
|
|
206
|
+
});
|
|
207
|
+
console.log(status.status); // "success" | "failed" | "abandoned"
|
|
208
|
+
console.log(status.output); // workflow's return value
|
|
209
|
+
```
|
|
210
|
+
|
|
211
|
+
`output` is the workflow's `run()` return value (whatever `defineWorkflow({
|
|
212
|
+
async run() { return … } })` resolves to). `setMetadata()` writes to a
|
|
213
|
+
separate `metadata` field — useful for "side-channel" facts (PR url, plan
|
|
214
|
+
url) without polluting the structured return.
|
|
215
|
+
|
|
216
|
+
### Auto parent/child tracing
|
|
217
|
+
|
|
218
|
+
The SDK detects `process.env.RUN_ID` (set by the runner sandbox on every
|
|
219
|
+
dispatch) and automatically threads it as `parentRunId` on subsequent
|
|
220
|
+
`invoke()` calls. Workflows that fan out to other workflows get a
|
|
221
|
+
parent/child tree in the dashboard for free. Pass `parentRunId: null`
|
|
222
|
+
to opt out.
|
|
223
|
+
|
|
224
|
+
---
|
|
225
|
+
|
|
226
|
+
## Per-workflow secrets
|
|
227
|
+
|
|
228
|
+
Secrets are stored in GCP Secret Manager, one row per `(team, workflow,
|
|
229
|
+
key)`. They're injected as env vars into the runner sandbox at dispatch
|
|
230
|
+
time, never persisted in the VM. Values are write-only — the API only
|
|
231
|
+
returns metadata (key, timestamps).
|
|
119
232
|
|
|
120
|
-
|
|
233
|
+
```ts
|
|
234
|
+
await client.setSecret("my-workflow", "ANTHROPIC_API_KEY", process.env.ANTHROPIC_API_KEY!);
|
|
235
|
+
const list = await client.listSecrets("my-workflow"); // [{ key, createdAt, updatedAt }]
|
|
236
|
+
await client.deleteSecret("my-workflow", "STALE_KEY");
|
|
121
237
|
```
|
|
122
238
|
|
|
239
|
+
Mutations require `admin` scope.
|
|
240
|
+
|
|
123
241
|
---
|
|
124
242
|
|
|
125
|
-
##
|
|
243
|
+
## Snapshots (replay-friendly sandboxes)
|
|
126
244
|
|
|
127
|
-
|
|
128
|
-
|
|
129
|
-
|
|
130
|
-
|
|
131
|
-
|
|
132
|
-
|
|
133
|
-
|
|
134
|
-
|
|
135
|
-
|
|
136
|
-
//
|
|
137
|
-
|
|
138
|
-
|
|
139
|
-
|
|
140
|
-
|
|
141
|
-
|
|
142
|
-
|
|
143
|
-
|
|
144
|
-
|
|
145
|
-
|
|
146
|
-
|
|
147
|
-
|
|
148
|
-
|
|
149
|
-
|
|
150
|
-
|
|
151
|
-
|
|
245
|
+
Long-running workflows can capture the runner sandbox as a Vercel snapshot
|
|
246
|
+
on success (`saveSnapshot: true`). Other workflows reference that snapshot
|
|
247
|
+
via the `snapshot` field to boot into the same prepared VM (deps installed,
|
|
248
|
+
repo cloned, etc.) instead of repeating setup.
|
|
249
|
+
|
|
250
|
+
```ts
|
|
251
|
+
// Capture per-invocation:
|
|
252
|
+
await client.invoke("my-workflow", input, { saveSnapshot: true });
|
|
253
|
+
|
|
254
|
+
// Boot from a different snapshot per invocation (per-invocation override):
|
|
255
|
+
await client.invoke("my-workflow", input, { snapshot: "<run-id-or-name>" });
|
|
256
|
+
|
|
257
|
+
// Or set a default at registration time:
|
|
258
|
+
defineWorkflow({ run, saveSnapshot: true });
|
|
259
|
+
|
|
260
|
+
// Browse / clean up:
|
|
261
|
+
const snaps = await client.listSnapshots({ workflow: "my-workflow", limit: 50 });
|
|
262
|
+
await client.deleteSnapshot(snaps[0].runId);
|
|
263
|
+
```
|
|
264
|
+
|
|
265
|
+
CLI equivalents: `agentc snapshot list` / `agentc snapshot delete <run-id>`.
|
|
266
|
+
|
|
267
|
+
---
|
|
268
|
+
|
|
269
|
+
## Authentication
|
|
270
|
+
|
|
271
|
+
The SDK accepts a Bearer API key (`ac_…`). Mint one from the dashboard:
|
|
272
|
+
sign in at `<server-url>/login`, then **Settings → API Keys → Create key**.
|
|
273
|
+
|
|
274
|
+
Default scopes (`read + invoke`) are right for a CI / dispatch caller. Tick
|
|
275
|
+
`admin` only if this key needs to register templates, mint other keys, or
|
|
276
|
+
manage secrets.
|
|
277
|
+
|
|
278
|
+
```ts
|
|
279
|
+
const client = new AgentComposeClient(
|
|
280
|
+
process.env.AGENT_COMPOSE_URL!,
|
|
281
|
+
process.env.AGENT_COMPOSE_API_KEY!,
|
|
282
|
+
);
|
|
152
283
|
```
|
|
284
|
+
|
|
285
|
+
The dashboard itself uses the cookie-bound session path; the SDK is for
|
|
286
|
+
programmatic / server-to-server callers.
|
|
287
|
+
|
|
288
|
+
---
|
|
289
|
+
|
|
290
|
+
## Public exports — quick reference
|
|
291
|
+
|
|
292
|
+
| Export | What |
|
|
293
|
+
|---|---|
|
|
294
|
+
| `defineWorkflow` | Attach metadata to a workflow `run` function |
|
|
295
|
+
| `defineRuntime` | Wrap an agent execution provider as an `AgentRuntime` |
|
|
296
|
+
| `defineSandboxEnvironment` | Sugar for declaring a workflow whose primary purpose is to build a snapshot for others to boot from |
|
|
297
|
+
| `runAgent` / `agentLoop` | Embed an LLM loop inside a workflow |
|
|
298
|
+
| `claudeRuntime` / `createClaudeRuntime` / `ClaudeRunner` | Built-in Claude Code runtime + factory |
|
|
299
|
+
| `AgentComposeClient` | HTTP client (register, invoke, status, snapshots, secrets) |
|
|
300
|
+
| `runWorkflow` | Local engine for running a workflow in-process (test harness) |
|
|
301
|
+
| `bundleWorkflow` | Resolve + inline a workflow's runtime sources for registration |
|
|
302
|
+
| `parseAgentStatus` / `parseAgentResponse` / `AgentStatusSchema` / `AgentMessageSchema` | Protocol parsers |
|
|
303
|
+
|
|
304
|
+
Type exports: `WorkflowFn`, `WorkflowCtx`, `WorkflowDefinition`, `AgentBudget`,
|
|
305
|
+
`AgentRuntime`, `RuntimeOptions`, `ModelExecutionContract`,
|
|
306
|
+
`AgentMessage` (and its variants), `AgentStatus`, `RunStatus`,
|
|
307
|
+
`RegisterResult`, `RunEvent`, `AgentLoopResult`, `RunAgentOpts`,
|
|
308
|
+
`SandboxProvider`, `SandboxNetworkPolicy`, `BundledWorkflow`.
|
|
309
|
+
|
|
310
|
+
For the canonical signatures, follow your IDE's go-to-definition into
|
|
311
|
+
`@agent-compose/sdk` — `sdk/src/index.ts` is the public surface and the
|
|
312
|
+
files it re-exports from carry full inline docstrings.
|
|
313
|
+
|
|
314
|
+
---
|
|
315
|
+
|
|
316
|
+
## Errors
|
|
317
|
+
|
|
318
|
+
All non-2xx HTTP responses throw `AgentComposeError(status, message)`. The
|
|
319
|
+
`message` is the server's `{ error: string }` body when present, falling
|
|
320
|
+
back to the HTTP status text:
|
|
321
|
+
|
|
322
|
+
```ts
|
|
323
|
+
import { AgentComposeError } from "@agent-compose/sdk";
|
|
324
|
+
|
|
325
|
+
try {
|
|
326
|
+
await client.invoke("missing-workflow");
|
|
327
|
+
} catch (err) {
|
|
328
|
+
if (err instanceof AgentComposeError && err.status === 404) {
|
|
329
|
+
// template not registered
|
|
330
|
+
}
|
|
331
|
+
}
|
|
332
|
+
```
|
|
333
|
+
|
|
334
|
+
`invokeAndWait` throws `AgentComposeError(504, …)` on timeout for symmetry.
|