@agent-compose/sdk 0.2.2 → 0.2.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +145 -33
- package/dist/agent/agent-loop.d.ts +83 -5
- package/dist/agent/run-agent.d.ts +34 -9
- package/dist/client.d.ts +247 -99
- package/dist/index.d.ts +26 -11
- package/dist/index.js +1968 -746
- package/dist/processors/builtins.d.ts +35 -0
- package/dist/processors/index.d.ts +4 -0
- package/dist/processors/processor.d.ts +91 -0
- package/dist/processors/processor.test.d.ts +1 -0
- package/dist/processors/runner.d.ts +19 -0
- package/dist/request-context/index.d.ts +2 -0
- package/dist/request-context/request-context.d.ts +159 -0
- package/dist/request-context/request-context.test.d.ts +1 -0
- package/dist/runtimes/claude.d.ts +27 -50
- package/dist/runtimes/openai-desktop.js +1919 -742
- package/dist/runtimes/vercel.d.ts +34 -0
- package/dist/runtimes/vercel.js +474 -0
- package/dist/sandbox.d.ts +29 -25
- package/dist/step-invocation/__tests__/invoker.test.d.ts +1 -0
- package/dist/step-invocation/__tests__/protocol.test.d.ts +1 -0
- package/dist/step-invocation/__tests__/server.test.d.ts +1 -0
- package/dist/step-invocation/index.d.ts +25 -0
- package/dist/step-invocation/invoker.d.ts +65 -0
- package/dist/step-invocation/protocol.d.ts +44 -0
- package/dist/step-invocation/server.d.ts +63 -0
- package/dist/step-invocation/types.d.ts +72 -0
- package/dist/tools/coding.d.ts +49 -0
- package/dist/tools/coding.test.d.ts +1 -0
- package/dist/tools/index.d.ts +2 -0
- package/dist/types/events.d.ts +36 -0
- package/dist/types/execution-context.d.ts +22 -0
- package/dist/types/runtime.d.ts +32 -0
- package/dist/types/sandbox-environment.d.ts +5 -2
- package/dist/types/sandbox.d.ts +14 -12
- package/dist/types/workflow-metadata.d.ts +51 -0
- package/dist/types/workflow-plan.d.ts +19 -0
- package/dist/types/workflow.d.ts +57 -17
- package/dist/utils/bundler.d.ts +62 -3
- package/dist/workflow-steps/__tests__/observability.test.d.ts +1 -0
- package/dist/workflow-steps/index.d.ts +10 -0
- package/dist/workflow-steps/observability.d.ts +58 -0
- package/dist/workflow-steps/runner.d.ts +96 -0
- package/dist/workflow-steps/step.d.ts +25 -0
- package/dist/workflow-steps/types.d.ts +135 -0
- package/dist/workflow-steps/workflow-steps.test.d.ts +1 -0
- package/dist/workflow-steps/workflow.d.ts +50 -0
- package/dist/workflows/engine.d.ts +27 -13
- package/dist/workflows/invoke-child.d.ts +10 -0
- package/package.json +25 -15
- package/src/agent/agent-loop.ts +197 -26
- package/src/agent/run-agent.ts +40 -15
- package/src/client.ts +326 -76
- package/src/index.ts +124 -10
- package/src/processors/builtins.ts +72 -0
- package/src/processors/index.ts +15 -0
- package/src/processors/processor.ts +103 -0
- package/src/processors/runner.ts +42 -0
- package/src/request-context/index.ts +17 -0
- package/src/request-context/request-context.ts +302 -0
- package/src/runtimes/claude.ts +123 -254
- package/src/runtimes/vercel.ts +180 -0
- package/src/sandbox.ts +53 -21
- package/src/step-invocation/index.ts +33 -0
- package/src/step-invocation/invoker.ts +204 -0
- package/src/step-invocation/protocol.ts +57 -0
- package/src/step-invocation/server.ts +184 -0
- package/src/step-invocation/types.ts +70 -0
- package/src/tools/coding.ts +126 -0
- package/src/tools/index.ts +8 -0
- package/src/types/events.ts +40 -0
- package/src/types/execution-context.ts +30 -0
- package/src/types/runtime.ts +24 -0
- package/src/types/sandbox-environment.ts +7 -5
- package/src/types/sandbox.ts +16 -12
- package/src/types/workflow-metadata.ts +84 -0
- package/src/types/workflow-plan.ts +24 -0
- package/src/types/workflow.ts +139 -25
- package/src/utils/bundler.ts +213 -19
- package/src/utils/source-loader.ts +2 -2
- package/src/workflow-steps/index.ts +30 -0
- package/src/workflow-steps/observability.ts +103 -0
- package/src/workflow-steps/runner.ts +244 -0
- package/src/workflow-steps/step.ts +38 -0
- package/src/workflow-steps/types.ts +134 -0
- package/src/workflow-steps/workflow.ts +95 -0
- package/src/workflows/engine.ts +69 -40
- package/src/workflows/invoke-child.ts +29 -0
package/README.md
CHANGED
|
@@ -1,10 +1,13 @@
|
|
|
1
1
|
# @agent-compose/sdk
|
|
2
2
|
|
|
3
|
-
TypeScript SDK for agent-compose. Use it to:
|
|
3
|
+
TypeScript SDK for [agent-compose](https://github.com/Layr-Labs/agent-compose). Use it to:
|
|
4
4
|
|
|
5
5
|
- **Author workflows** that run agentic LLM loops inside isolated sandboxes
|
|
6
6
|
- **Define runtimes** that wrap a coding-CLI tool (Claude Code, OpenAI Desktop, …) into a sandbox-portable agent loop
|
|
7
|
-
- **Register, invoke, and
|
|
7
|
+
- **Register, invoke, observe, and cancel** workflows via the HTTP API (`AgentComposeClient`)
|
|
8
|
+
- **Manage factories, secrets, API keys, and snapshots** programmatically
|
|
9
|
+
|
|
10
|
+
The hierarchy: a **team** owns one or more **factories** (project containers); each factory owns workflow templates, secrets, and runs. Workflows are versioned per `(factory, name, version)`. New code that doesn't care about factories transparently lands in `default` — every team has one.
|
|
8
11
|
|
|
9
12
|
---
|
|
10
13
|
|
|
@@ -25,23 +28,22 @@ A workflow is `async (ctx, sandbox) => T`. Two positional args:
|
|
|
25
28
|
- **`ctx`** carries the run identity (`run.id`), the caller's `input`, plus
|
|
26
29
|
observability helpers (`setMetadata`, `step`).
|
|
27
30
|
- **`sandbox`** is a capability the engine constructs once for the run —
|
|
28
|
-
pass it to `
|
|
31
|
+
pass it to `agent({ sandbox, ... })` and to any helper that takes a
|
|
29
32
|
`SandboxProvider` (file writers, git utilities, command runners).
|
|
30
33
|
|
|
31
34
|
```typescript
|
|
32
35
|
// my-workflow.ts
|
|
33
|
-
import { defineWorkflow,
|
|
36
|
+
import { defineWorkflow, agent, claudeRuntime } from "@agent-compose/sdk";
|
|
34
37
|
import PROMPT from "./prompt.md" with { type: "text" };
|
|
35
38
|
|
|
36
39
|
export default defineWorkflow({
|
|
37
40
|
async run(ctx, sandbox) {
|
|
38
41
|
const repo = (ctx.input?.repo as string | undefined) ?? "owner/repo";
|
|
39
42
|
|
|
40
|
-
const result = await
|
|
43
|
+
const result = await agent({
|
|
41
44
|
sandbox,
|
|
42
45
|
runtime: claudeRuntime,
|
|
43
|
-
prompt: PROMPT
|
|
44
|
-
promptVars: { REPO: repo },
|
|
46
|
+
prompt: `${PROMPT}\n\nRepository: ${repo}`,
|
|
45
47
|
tools: ["Bash", "Read", "Edit", "Write", "Grep", "Glob"],
|
|
46
48
|
budget: { turnsPerIteration: 40, maxIterations: 8 },
|
|
47
49
|
});
|
|
@@ -83,14 +85,13 @@ interface WorkflowCtx {
|
|
|
83
85
|
duration. Use it for setup, external API calls, or anything you want
|
|
84
86
|
visible on the dashboard's run detail page.
|
|
85
87
|
|
|
86
|
-
### The `
|
|
88
|
+
### The `agent` loop
|
|
87
89
|
|
|
88
90
|
```ts
|
|
89
|
-
|
|
91
|
+
agent({
|
|
90
92
|
sandbox, // the workflow's sandbox arg
|
|
91
93
|
runtime, // claudeRuntime, or your own via createClaudeRuntime / defineRuntime
|
|
92
94
|
prompt, // raw markdown — `--- frontmatter ---` is auto-stripped
|
|
93
|
-
promptVars?, // {{VAR}} substitutions; WORKING_DIR + DIFF_BASE auto-populate
|
|
94
95
|
tools?, // model tool allowlist (defaults inside agentLoop)
|
|
95
96
|
budget?, // { turnsPerIteration, maxIterations }
|
|
96
97
|
workingDir?, // every shell command runs here
|
|
@@ -110,7 +111,7 @@ the full instructions appended to every prompt.
|
|
|
110
111
|
## Defining a runtime
|
|
111
112
|
|
|
112
113
|
A "runtime" wraps an agent's underlying execution model — usually a coding
|
|
113
|
-
CLI like Claude Code or OpenAI Desktop — so `
|
|
114
|
+
CLI like Claude Code or OpenAI Desktop — so `agent` can drive it. The SDK
|
|
114
115
|
ships built-ins; you only need a custom one for an exotic provider.
|
|
115
116
|
|
|
116
117
|
### Built-in runtimes
|
|
@@ -148,7 +149,7 @@ const myRuntime: AgentRuntime = defineRuntime({
|
|
|
148
149
|
|
|
149
150
|
`AgentRuntime` is a tagged record with `create(sandbox, RuntimeOptions) →
|
|
150
151
|
ModelExecutionContract`. There is **no `provider` field** on it — the
|
|
151
|
-
runtime is bound to the workflow at author time (you pass it to `
|
|
152
|
+
runtime is bound to the workflow at author time (you pass it to `agent`),
|
|
152
153
|
not selected by the server.
|
|
153
154
|
|
|
154
155
|
---
|
|
@@ -162,9 +163,10 @@ agentc register my-workflow.ts -n my-workflow
|
|
|
162
163
|
```
|
|
163
164
|
|
|
164
165
|
Under the hood that calls `bundleWorkflow(workflowPath)` (resolves imports,
|
|
165
|
-
inlines runtime sources via dynamic-require traversal) and `POST
|
|
166
|
-
with the bundled source. If you need
|
|
167
|
-
build pipeline, you can do the same
|
|
166
|
+
inlines runtime sources via dynamic-require traversal) and `POST
|
|
167
|
+
/api/v1/factories/<slug>/templates` with the bundled source. If you need
|
|
168
|
+
to drive registration from your own build pipeline, you can do the same
|
|
169
|
+
thing via the SDK directly:
|
|
168
170
|
|
|
169
171
|
```ts
|
|
170
172
|
import { AgentComposeClient, bundleWorkflow } from "@agent-compose/sdk";
|
|
@@ -176,10 +178,11 @@ const client = new AgentComposeClient(
|
|
|
176
178
|
|
|
177
179
|
const bundled = await bundleWorkflow("./my-workflow.ts");
|
|
178
180
|
await client.register({
|
|
179
|
-
name:
|
|
180
|
-
source:
|
|
181
|
-
runtimes:
|
|
182
|
-
schedule:
|
|
181
|
+
name: "my-workflow",
|
|
182
|
+
source: bundled.source,
|
|
183
|
+
runtimes: bundled.runtimes, // [{ name, source }] — embedded so the runner has them locally
|
|
184
|
+
schedule: "*/30 * * * *", // optional cron
|
|
185
|
+
factorySlug: "default", // optional — defaults to "default"
|
|
183
186
|
// snapshot, saveSnapshot, networkPolicy, placeholders — all optional
|
|
184
187
|
});
|
|
185
188
|
```
|
|
@@ -204,7 +207,7 @@ const status = await client.invokeAndWait("my-workflow", { repo: "owner/repo" },
|
|
|
204
207
|
timeoutMs: 5 * 60_000,
|
|
205
208
|
pollIntervalMs: 2000,
|
|
206
209
|
});
|
|
207
|
-
console.log(status.status); // "success" | "failed" | "abandoned"
|
|
210
|
+
console.log(status.status); // "success" | "failed" | "abandoned" | "canceled"
|
|
208
211
|
console.log(status.output); // workflow's return value
|
|
209
212
|
```
|
|
210
213
|
|
|
@@ -213,6 +216,10 @@ async run() { return … } })` resolves to). `setMetadata()` writes to a
|
|
|
213
216
|
separate `metadata` field — useful for "side-channel" facts (PR url, plan
|
|
214
217
|
url) without polluting the structured return.
|
|
215
218
|
|
|
219
|
+
`invoke` and `invokeAndWait` both accept `{ factorySlug, snapshot,
|
|
220
|
+
saveSnapshot, parentRunId }` as the third argument. `factorySlug` defaults
|
|
221
|
+
to `"default"`.
|
|
222
|
+
|
|
216
223
|
### Auto parent/child tracing
|
|
217
224
|
|
|
218
225
|
The SDK detects `process.env.RUN_ID` (set by the runner sandbox on every
|
|
@@ -221,25 +228,123 @@ dispatch) and automatically threads it as `parentRunId` on subsequent
|
|
|
221
228
|
parent/child tree in the dashboard for free. Pass `parentRunId: null`
|
|
222
229
|
to opt out.
|
|
223
230
|
|
|
231
|
+
### Cancelling a run
|
|
232
|
+
|
|
233
|
+
```ts
|
|
234
|
+
await client.cancelRun(runId);
|
|
235
|
+
```
|
|
236
|
+
|
|
237
|
+
Idempotent — cancelling an already-terminal run returns the current state
|
|
238
|
+
without throwing. The server stamps the run as `canceled`, kills any live
|
|
239
|
+
sandboxes, and emits a `run_canceled` event on the stream.
|
|
240
|
+
|
|
241
|
+
### Streaming live logs
|
|
242
|
+
|
|
243
|
+
`streamRunLogs` returns an async generator of `RunEvent`s in real time,
|
|
244
|
+
re-attaching via SSE under the hood. Pass `lastEventId` (the highest
|
|
245
|
+
`seq` you've already processed) to resume after a reconnect.
|
|
246
|
+
|
|
247
|
+
```ts
|
|
248
|
+
for await (const ev of client.streamRunLogs(runId, { lastEventId: 0 })) {
|
|
249
|
+
console.log(ev.event, ev.seq, ev.data);
|
|
250
|
+
if (ev.event === "run_complete" || ev.event === "run_failed" || ev.event === "run_canceled") {
|
|
251
|
+
break;
|
|
252
|
+
}
|
|
253
|
+
}
|
|
254
|
+
```
|
|
255
|
+
|
|
256
|
+
`AbortSignal` works too — pass `{ signal }` and call `controller.abort()`
|
|
257
|
+
to tear the stream down from the caller side.
|
|
258
|
+
|
|
259
|
+
---
|
|
260
|
+
|
|
261
|
+
## Factories
|
|
262
|
+
|
|
263
|
+
Factories are project containers within a team. Each factory has its own
|
|
264
|
+
workflow templates, secrets, runs, and (optionally) scoped API keys. New
|
|
265
|
+
projects don't need to think about them — `default` is auto-created per
|
|
266
|
+
team and is what the SDK falls back to when `factorySlug` is omitted.
|
|
267
|
+
|
|
268
|
+
```ts
|
|
269
|
+
// CRUD on factories
|
|
270
|
+
await client.createFactory({ slug: "ci-bots", name: "CI Bots", description: "…" });
|
|
271
|
+
const factories = await client.listFactories();
|
|
272
|
+
const f = await client.getFactory("ci-bots");
|
|
273
|
+
await client.updateFactory("ci-bots", { name: "Continuous-Integration Bots" });
|
|
274
|
+
await client.deleteFactory("ci-bots");
|
|
275
|
+
|
|
276
|
+
// Templates list — flat across factories, or scoped to one
|
|
277
|
+
const all = await client.listTemplates();
|
|
278
|
+
const scoped = await client.listTemplates({ factorySlug: "ci-bots" });
|
|
279
|
+
|
|
280
|
+
// Register / invoke / secret operations all accept factorySlug
|
|
281
|
+
await client.register({ name: "scrape", source, factorySlug: "ci-bots", … });
|
|
282
|
+
await client.invoke("scrape", { url: "…" }, { factorySlug: "ci-bots" });
|
|
283
|
+
await client.setSecret("scrape", "GH_TOKEN", "ghp_…", { factorySlug: "ci-bots" });
|
|
284
|
+
```
|
|
285
|
+
|
|
286
|
+
CLI equivalents: `agentc factory list | create | get | update | delete`,
|
|
287
|
+
plus `--factory <slug>` on every other command.
|
|
288
|
+
|
|
224
289
|
---
|
|
225
290
|
|
|
226
291
|
## Per-workflow secrets
|
|
227
292
|
|
|
228
|
-
Secrets
|
|
229
|
-
|
|
230
|
-
|
|
231
|
-
|
|
293
|
+
Secrets live in GCP Secret Manager, one row per `(factory, workflow, key)`.
|
|
294
|
+
They're injected as env vars into the runner sandbox at dispatch time,
|
|
295
|
+
never persisted in the VM. Values are write-only — the API only returns
|
|
296
|
+
metadata (key, timestamps).
|
|
232
297
|
|
|
233
298
|
```ts
|
|
234
299
|
await client.setSecret("my-workflow", "ANTHROPIC_API_KEY", process.env.ANTHROPIC_API_KEY!);
|
|
235
300
|
const list = await client.listSecrets("my-workflow"); // [{ key, createdAt, updatedAt }]
|
|
236
301
|
await client.deleteSecret("my-workflow", "STALE_KEY");
|
|
302
|
+
|
|
303
|
+
// Scope to a non-default factory:
|
|
304
|
+
await client.setSecret("scrape", "GH_TOKEN", "ghp_…", { factorySlug: "ci-bots" });
|
|
237
305
|
```
|
|
238
306
|
|
|
239
307
|
Mutations require `admin` scope.
|
|
240
308
|
|
|
241
309
|
---
|
|
242
310
|
|
|
311
|
+
## API keys
|
|
312
|
+
|
|
313
|
+
Mint and list scoped keys programmatically (requires an `admin`-scoped
|
|
314
|
+
caller key). New keys are returned **once**, in the same response as the
|
|
315
|
+
metadata — copy the `ac_…` value immediately.
|
|
316
|
+
|
|
317
|
+
```ts
|
|
318
|
+
const created = await client.createApiKey({
|
|
319
|
+
name: "ci-dispatcher",
|
|
320
|
+
scopes: ["read", "invoke"],
|
|
321
|
+
expiresAt: new Date(Date.now() + 30 * 86_400_000).toISOString(), // 30 days
|
|
322
|
+
// factorySlug: "ci-bots" // optional — scopes the key to a single factory
|
|
323
|
+
});
|
|
324
|
+
console.log(created.key); // "ac_…" — the only time you'll see this
|
|
325
|
+
|
|
326
|
+
const all = await client.listApiKeys();
|
|
327
|
+
```
|
|
328
|
+
|
|
329
|
+
CLI equivalent: `agentc keys create <name> --scopes read,invoke
|
|
330
|
+
--expires-in 30d`.
|
|
331
|
+
|
|
332
|
+
---
|
|
333
|
+
|
|
334
|
+
## Usage
|
|
335
|
+
|
|
336
|
+
```ts
|
|
337
|
+
const usage = await client.getUsage(
|
|
338
|
+
new Date(Date.now() - 30 * 86_400_000),
|
|
339
|
+
new Date(),
|
|
340
|
+
);
|
|
341
|
+
// usage.rows: [{ day, runs, sandbox_seconds, … }]
|
|
342
|
+
```
|
|
343
|
+
|
|
344
|
+
CLI equivalent: `agentc usage`.
|
|
345
|
+
|
|
346
|
+
---
|
|
347
|
+
|
|
243
348
|
## Snapshots (replay-friendly sandboxes)
|
|
244
349
|
|
|
245
350
|
Long-running workflows can capture the runner sandbox as a Vercel snapshot
|
|
@@ -294,18 +399,25 @@ programmatic / server-to-server callers.
|
|
|
294
399
|
| `defineWorkflow` | Attach metadata to a workflow `run` function |
|
|
295
400
|
| `defineRuntime` | Wrap an agent execution provider as an `AgentRuntime` |
|
|
296
401
|
| `defineSandboxEnvironment` | Sugar for declaring a workflow whose primary purpose is to build a snapshot for others to boot from |
|
|
297
|
-
| `
|
|
298
|
-
| `claudeRuntime` / `createClaudeRuntime` / `ClaudeRunner` | Built-in Claude Code runtime + factory |
|
|
299
|
-
| `AgentComposeClient` | HTTP client (register, invoke, status, snapshots, secrets) |
|
|
402
|
+
| `agent` / `agentLoop` | Embed an LLM loop inside a workflow |
|
|
300
403
|
| `runWorkflow` | Local engine for running a workflow in-process (test harness) |
|
|
301
404
|
| `bundleWorkflow` | Resolve + inline a workflow's runtime sources for registration |
|
|
405
|
+
| `claudeRuntime` / `createClaudeRuntime` / `ClaudeRunner` | Built-in Claude Code runtime + factory |
|
|
406
|
+
| `AgentComposeClient` | HTTP client — register, invoke, cancel, stream logs, factories, snapshots, secrets, API keys, usage |
|
|
407
|
+
| `AgentComposeError` | Thrown by every non-2xx HTTP response |
|
|
302
408
|
| `parseAgentStatus` / `parseAgentResponse` / `AgentStatusSchema` / `AgentMessageSchema` | Protocol parsers |
|
|
303
|
-
|
|
304
|
-
|
|
305
|
-
|
|
306
|
-
|
|
307
|
-
`
|
|
308
|
-
`
|
|
409
|
+
| `parseSseStream` | Generic SSE chunk decoder (used by `streamRunLogs`) |
|
|
410
|
+
| `createSandbox` / `reconnectSandbox` / `killAllSandboxes` / `killSandboxById` / `getSandboxQuotas` / `listOwnedSandboxes` / `deleteSandboxSnapshot` | Sandbox-provider helpers (Vercel + E2B) |
|
|
411
|
+
|
|
412
|
+
Type exports: `WorkflowFn`, `WorkflowCtx`, `WorkflowDefinition`,
|
|
413
|
+
`WorkflowHooks`, `AgentBudget`, `AgentRuntime`, `RuntimeOptions`,
|
|
414
|
+
`ModelExecutionContract`, `McpServerConfig`, `AgentMessage` (and its
|
|
415
|
+
variants), `AgentStatus`, `RunStatus`, `RegisterResult`, `RunEvent`,
|
|
416
|
+
`FactoryRow`, `SnapshotListEntry`, `ApiKey`, `ApiKeyCreated`,
|
|
417
|
+
`UsageRollupRow`, `UsageResponse`, `CancelRunResponse`, `AgentLoopResult`,
|
|
418
|
+
`AgentOpts`, `SandboxProvider`, `DesktopSandboxProvider`,
|
|
419
|
+
`SandboxNetworkPolicy`, `SandboxCreateOpts`, `OwnedSandbox`,
|
|
420
|
+
`BundledWorkflow`.
|
|
309
421
|
|
|
310
422
|
For the canonical signatures, follow your IDE's go-to-definition into
|
|
311
423
|
`@agent-compose/sdk` — `sdk/src/index.ts` is the public surface and the
|
|
@@ -5,23 +5,101 @@
|
|
|
5
5
|
import type { RuntimeOptions, ModelExecutionContract } from "../index.js";
|
|
6
6
|
import { z } from "zod";
|
|
7
7
|
import type { AgentStatus, AgentMessage } from "./protocol.js";
|
|
8
|
+
import type { Processor } from "../processors/processor.js";
|
|
9
|
+
import { RequestContext } from "../request-context/request-context.js";
|
|
8
10
|
export declare const DEFAULT_CLAUDE_MODEL = "claude-opus-4-7";
|
|
9
11
|
export declare function parseAgentStatus(text: string): AgentStatus | null;
|
|
10
|
-
export interface AgentLoopResult {
|
|
12
|
+
export interface AgentLoopResult<TResponse = unknown> {
|
|
13
|
+
agentId: string;
|
|
14
|
+
label: string;
|
|
11
15
|
sessionId: string;
|
|
12
16
|
lastStatus: AgentStatus | null;
|
|
13
17
|
iterations: number;
|
|
14
|
-
response?:
|
|
18
|
+
response?: TResponse;
|
|
15
19
|
}
|
|
16
|
-
export
|
|
20
|
+
export type AgentMessageSummary = {
|
|
21
|
+
type: "init";
|
|
22
|
+
sessionId: string;
|
|
23
|
+
} | {
|
|
24
|
+
type: "text";
|
|
25
|
+
text: string;
|
|
26
|
+
} | {
|
|
27
|
+
type: "thinking";
|
|
28
|
+
text: string;
|
|
29
|
+
} | {
|
|
30
|
+
type: "tool_use";
|
|
31
|
+
toolName: string;
|
|
32
|
+
toolUseId: string;
|
|
33
|
+
toolInputPreview: string;
|
|
34
|
+
} | {
|
|
35
|
+
type: "tool_result";
|
|
36
|
+
toolUseId: string;
|
|
37
|
+
output: string;
|
|
38
|
+
isError: boolean;
|
|
39
|
+
} | {
|
|
40
|
+
type: "usage";
|
|
41
|
+
inputTokens: number;
|
|
42
|
+
outputTokens: number;
|
|
43
|
+
cacheReadTokens: number;
|
|
44
|
+
cacheCreationTokens: number;
|
|
45
|
+
durationMs: number;
|
|
46
|
+
numTurns: number;
|
|
47
|
+
} | {
|
|
48
|
+
type: "done";
|
|
49
|
+
sessionId: string;
|
|
50
|
+
} | {
|
|
51
|
+
type: "error";
|
|
52
|
+
text: string;
|
|
53
|
+
};
|
|
54
|
+
export declare function summarizeAgentMessage(msg: AgentMessage): AgentMessageSummary;
|
|
55
|
+
export type AgentLifecycleEvent = {
|
|
56
|
+
event: "agent.spawned";
|
|
57
|
+
at: number;
|
|
58
|
+
agentId: string;
|
|
59
|
+
label: string;
|
|
60
|
+
allowedTools?: string[];
|
|
61
|
+
} | {
|
|
62
|
+
event: "agent.message";
|
|
63
|
+
at: number;
|
|
64
|
+
agentId: string;
|
|
65
|
+
label: string;
|
|
66
|
+
iteration: number;
|
|
67
|
+
message: AgentMessageSummary;
|
|
68
|
+
} | {
|
|
69
|
+
event: "agent.iteration";
|
|
70
|
+
at: number;
|
|
71
|
+
agentId: string;
|
|
72
|
+
label: string;
|
|
73
|
+
iteration: number;
|
|
74
|
+
status: AgentStatus | null;
|
|
75
|
+
} | {
|
|
76
|
+
event: "agent.settled";
|
|
77
|
+
at: number;
|
|
78
|
+
agentId: string;
|
|
79
|
+
label: string;
|
|
80
|
+
outcome: "success" | "failed";
|
|
81
|
+
iterations: number;
|
|
82
|
+
durationMs: number;
|
|
83
|
+
failureReason?: string;
|
|
84
|
+
};
|
|
85
|
+
export interface AgentLoopOpts<TResponse = unknown> {
|
|
86
|
+
agentId?: string;
|
|
17
87
|
label?: string;
|
|
88
|
+
onAgentLifecycleEvent?: (event: AgentLifecycleEvent) => void;
|
|
18
89
|
onIteration?: (iteration: number, status: AgentStatus | null) => void;
|
|
19
90
|
turnsPerIteration?: number;
|
|
20
91
|
maxIterations?: number;
|
|
21
92
|
buildPrompt: (lastStatus: AgentStatus | null, iteration: number) => string;
|
|
22
93
|
onAgentEvent?: (iteration: number, msg: AgentMessage) => void;
|
|
23
94
|
allowedTools?: string[];
|
|
24
|
-
responseSchema?: z.ZodType<
|
|
95
|
+
responseSchema?: z.ZodType<TResponse>;
|
|
25
96
|
runtime?: (opts: RuntimeOptions) => ModelExecutionContract;
|
|
26
97
|
cwd?: string;
|
|
27
|
-
|
|
98
|
+
/** Processor chain — see sdk/src/processors. Runs sequentially around
|
|
99
|
+
* prompts and emitted messages. Tool-call gating is a no-op until a
|
|
100
|
+
* runtime that exposes a pre-tool-use seam is wired (candidate #1). */
|
|
101
|
+
processors?: readonly Processor[];
|
|
102
|
+
/** Per-run typed bag — propagated to every processor as `ctx.requestContext`. */
|
|
103
|
+
requestContext?: RequestContext;
|
|
104
|
+
}
|
|
105
|
+
export declare function agentLoop<TResponse = unknown>(opts: AgentLoopOpts<TResponse>): Promise<AgentLoopResult<TResponse>>;
|
|
@@ -1,22 +1,25 @@
|
|
|
1
1
|
/**
|
|
2
|
-
*
|
|
2
|
+
* agent — canonical entry point for embedding an LLM agent inside a
|
|
3
3
|
* workflow. The workflow's `run()` body calls it; the loop executes
|
|
4
4
|
* against the runner's own VM.
|
|
5
5
|
*
|
|
6
6
|
* Glue packaged so workflows don't duplicate it:
|
|
7
|
-
* - Inject `{{VAR}}` placeholders into the prompt template.
|
|
8
7
|
* - Strip the `--- frontmatter ---` header authors use for IDE hints.
|
|
9
8
|
* - Append PROTOCOL_SUFFIX (status/response format instructions).
|
|
10
9
|
* - Append a response-format appendix when `responseSchema` is set.
|
|
11
10
|
* - Delegate to `agentLoop`.
|
|
12
11
|
*/
|
|
13
12
|
import { z } from "zod";
|
|
14
|
-
import type { AgentLoopResult } from "./agent-loop.js";
|
|
13
|
+
import type { AgentLifecycleEvent, AgentLoopResult } from "./agent-loop.js";
|
|
15
14
|
import type { AgentMessage, AgentStatus } from "../types/protocol.js";
|
|
16
15
|
import type { AgentRuntime } from "../types/runtime.js";
|
|
17
16
|
import type { SandboxProvider } from "../types/sandbox.js";
|
|
18
17
|
import type { AgentBudget } from "../types/workflow.js";
|
|
19
|
-
|
|
18
|
+
import type { Processor } from "../processors/processor.js";
|
|
19
|
+
import { RequestContext } from "../request-context/request-context.js";
|
|
20
|
+
export interface AgentOpts<T = unknown> {
|
|
21
|
+
/** Stable system identity for this agent loop. Generated when omitted. */
|
|
22
|
+
agentId?: string;
|
|
20
23
|
/** Sandbox the runtime executes commands against. Inside a workflow,
|
|
21
24
|
* always pass `ctx.sandbox` — a pre-constructed local provider for
|
|
22
25
|
* the runner's own VM. Exposed as a parameter so tests and non-workflow
|
|
@@ -24,12 +27,9 @@ export interface RunAgentOpts<T = unknown> {
|
|
|
24
27
|
sandbox: SandboxProvider;
|
|
25
28
|
/** Runtime definition from `createClaudeRuntime({...})` (or custom). */
|
|
26
29
|
runtime: AgentRuntime;
|
|
27
|
-
/** Prompt
|
|
30
|
+
/** Prompt text. Authors can include YAML-style `--- frontmatter ---`
|
|
28
31
|
* at the top for IDE hints; it's stripped before the model sees it. */
|
|
29
32
|
prompt: string;
|
|
30
|
-
/** Substitution map for `{{VAR}}` placeholders in the prompt. `WORKING_DIR`
|
|
31
|
-
* and `DIFF_BASE` auto-populate from `opts.workingDir` unless overridden. */
|
|
32
|
-
promptVars?: Record<string, string>;
|
|
33
33
|
/** `cwd` forwarded to the runtime — every shell command runs here. */
|
|
34
34
|
workingDir?: string;
|
|
35
35
|
/** Tools the model may use. Defaults to a safe kitchen-sink set inside
|
|
@@ -43,16 +43,41 @@ export interface RunAgentOpts<T = unknown> {
|
|
|
43
43
|
responseSchema?: z.ZodType<T>;
|
|
44
44
|
/** Label prefix for runtime stderr ("[sbid][agent]" by default). */
|
|
45
45
|
label?: string;
|
|
46
|
+
/** Lifecycle event sink from workflow ctx. Emits agent.spawned / iteration / settled. */
|
|
47
|
+
events?: {
|
|
48
|
+
emit: (event: AgentLifecycleEvent) => void | Promise<void>;
|
|
49
|
+
};
|
|
46
50
|
/** Per-message event callback — wire this to your workflow's event
|
|
47
51
|
* telemetry if you want per-tool-call observability. */
|
|
48
52
|
onAgentEvent?: (iteration: number, msg: AgentMessage) => void;
|
|
49
53
|
/** Per-iteration status callback — fires after each model turn with the
|
|
50
54
|
* parsed `<status>` block (or null if the model didn't emit one). */
|
|
51
55
|
onIteration?: (iteration: number, status: AgentStatus | null) => void;
|
|
56
|
+
/**
|
|
57
|
+
* Processors run as a typed pre/post pipeline around the agent loop:
|
|
58
|
+
* - processInput before each iteration's prompt is sent
|
|
59
|
+
* - processOutput on each emitted message
|
|
60
|
+
* - processToolCall before each tool call (runtime-driven; dormant
|
|
61
|
+
* until a runtime that supports pre-tool gating is wired in)
|
|
62
|
+
*
|
|
63
|
+
* Common pattern with workflow-level defaults:
|
|
64
|
+
* agent({ processors: [...ctx.processors, mySpecific], ... })
|
|
65
|
+
*/
|
|
66
|
+
processors?: readonly Processor[];
|
|
67
|
+
/**
|
|
68
|
+
* Per-run request context. Passed through to processors as
|
|
69
|
+
* `ctx.requestContext`; carries tenant identity (teamId, factoryId, scopes)
|
|
70
|
+
* and the freeform user namespace.
|
|
71
|
+
*
|
|
72
|
+
* Inside a workflow, pass `ctx.requestContext`. Tests/non-workflow callers
|
|
73
|
+
* can omit; a degenerate context is synthesised so processors that don't
|
|
74
|
+
* read identity (e.g. redactPattern) still work.
|
|
75
|
+
*/
|
|
76
|
+
requestContext?: RequestContext;
|
|
52
77
|
}
|
|
53
78
|
/**
|
|
54
79
|
* Run an agent loop inside a workflow. Returns the loop's final
|
|
55
80
|
* `AgentLoopResult`, including `response` when a `responseSchema` was
|
|
56
81
|
* supplied and the model validated against it.
|
|
57
82
|
*/
|
|
58
|
-
export declare function
|
|
83
|
+
export declare function agent<T = unknown>(opts: AgentOpts<T>): Promise<AgentLoopResult<T>>;
|