@agent-compose/sdk 0.2.2 → 0.2.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (88) hide show
  1. package/README.md +145 -33
  2. package/dist/agent/agent-loop.d.ts +83 -5
  3. package/dist/agent/run-agent.d.ts +34 -9
  4. package/dist/client.d.ts +247 -99
  5. package/dist/index.d.ts +26 -11
  6. package/dist/index.js +1968 -746
  7. package/dist/processors/builtins.d.ts +35 -0
  8. package/dist/processors/index.d.ts +4 -0
  9. package/dist/processors/processor.d.ts +91 -0
  10. package/dist/processors/processor.test.d.ts +1 -0
  11. package/dist/processors/runner.d.ts +19 -0
  12. package/dist/request-context/index.d.ts +2 -0
  13. package/dist/request-context/request-context.d.ts +159 -0
  14. package/dist/request-context/request-context.test.d.ts +1 -0
  15. package/dist/runtimes/claude.d.ts +27 -50
  16. package/dist/runtimes/openai-desktop.js +1919 -742
  17. package/dist/runtimes/vercel.d.ts +34 -0
  18. package/dist/runtimes/vercel.js +474 -0
  19. package/dist/sandbox.d.ts +29 -25
  20. package/dist/step-invocation/__tests__/invoker.test.d.ts +1 -0
  21. package/dist/step-invocation/__tests__/protocol.test.d.ts +1 -0
  22. package/dist/step-invocation/__tests__/server.test.d.ts +1 -0
  23. package/dist/step-invocation/index.d.ts +25 -0
  24. package/dist/step-invocation/invoker.d.ts +65 -0
  25. package/dist/step-invocation/protocol.d.ts +44 -0
  26. package/dist/step-invocation/server.d.ts +63 -0
  27. package/dist/step-invocation/types.d.ts +72 -0
  28. package/dist/tools/coding.d.ts +49 -0
  29. package/dist/tools/coding.test.d.ts +1 -0
  30. package/dist/tools/index.d.ts +2 -0
  31. package/dist/types/events.d.ts +36 -0
  32. package/dist/types/execution-context.d.ts +22 -0
  33. package/dist/types/runtime.d.ts +32 -0
  34. package/dist/types/sandbox-environment.d.ts +5 -2
  35. package/dist/types/sandbox.d.ts +14 -12
  36. package/dist/types/workflow-metadata.d.ts +51 -0
  37. package/dist/types/workflow-plan.d.ts +19 -0
  38. package/dist/types/workflow.d.ts +57 -17
  39. package/dist/utils/bundler.d.ts +62 -3
  40. package/dist/workflow-steps/__tests__/observability.test.d.ts +1 -0
  41. package/dist/workflow-steps/index.d.ts +10 -0
  42. package/dist/workflow-steps/observability.d.ts +58 -0
  43. package/dist/workflow-steps/runner.d.ts +96 -0
  44. package/dist/workflow-steps/step.d.ts +25 -0
  45. package/dist/workflow-steps/types.d.ts +135 -0
  46. package/dist/workflow-steps/workflow-steps.test.d.ts +1 -0
  47. package/dist/workflow-steps/workflow.d.ts +50 -0
  48. package/dist/workflows/engine.d.ts +27 -13
  49. package/dist/workflows/invoke-child.d.ts +10 -0
  50. package/package.json +25 -15
  51. package/src/agent/agent-loop.ts +197 -26
  52. package/src/agent/run-agent.ts +40 -15
  53. package/src/client.ts +326 -76
  54. package/src/index.ts +124 -10
  55. package/src/processors/builtins.ts +72 -0
  56. package/src/processors/index.ts +15 -0
  57. package/src/processors/processor.ts +103 -0
  58. package/src/processors/runner.ts +42 -0
  59. package/src/request-context/index.ts +17 -0
  60. package/src/request-context/request-context.ts +302 -0
  61. package/src/runtimes/claude.ts +123 -254
  62. package/src/runtimes/vercel.ts +180 -0
  63. package/src/sandbox.ts +53 -21
  64. package/src/step-invocation/index.ts +33 -0
  65. package/src/step-invocation/invoker.ts +204 -0
  66. package/src/step-invocation/protocol.ts +57 -0
  67. package/src/step-invocation/server.ts +184 -0
  68. package/src/step-invocation/types.ts +70 -0
  69. package/src/tools/coding.ts +126 -0
  70. package/src/tools/index.ts +8 -0
  71. package/src/types/events.ts +40 -0
  72. package/src/types/execution-context.ts +30 -0
  73. package/src/types/runtime.ts +24 -0
  74. package/src/types/sandbox-environment.ts +7 -5
  75. package/src/types/sandbox.ts +16 -12
  76. package/src/types/workflow-metadata.ts +84 -0
  77. package/src/types/workflow-plan.ts +24 -0
  78. package/src/types/workflow.ts +139 -25
  79. package/src/utils/bundler.ts +213 -19
  80. package/src/utils/source-loader.ts +2 -2
  81. package/src/workflow-steps/index.ts +30 -0
  82. package/src/workflow-steps/observability.ts +103 -0
  83. package/src/workflow-steps/runner.ts +244 -0
  84. package/src/workflow-steps/step.ts +38 -0
  85. package/src/workflow-steps/types.ts +134 -0
  86. package/src/workflow-steps/workflow.ts +95 -0
  87. package/src/workflows/engine.ts +69 -40
  88. package/src/workflows/invoke-child.ts +29 -0
package/README.md CHANGED
@@ -1,10 +1,13 @@
1
1
  # @agent-compose/sdk
2
2
 
3
- TypeScript SDK for agent-compose. Use it to:
3
+ TypeScript SDK for [agent-compose](https://github.com/Layr-Labs/agent-compose). Use it to:
4
4
 
5
5
  - **Author workflows** that run agentic LLM loops inside isolated sandboxes
6
6
  - **Define runtimes** that wrap a coding-CLI tool (Claude Code, OpenAI Desktop, …) into a sandbox-portable agent loop
7
- - **Register, invoke, and observe** workflows via the HTTP API (`AgentComposeClient`)
7
+ - **Register, invoke, observe, and cancel** workflows via the HTTP API (`AgentComposeClient`)
8
+ - **Manage factories, secrets, API keys, and snapshots** programmatically
9
+
10
+ The hierarchy: a **team** owns one or more **factories** (project containers); each factory owns workflow templates, secrets, and runs. Workflows are versioned per `(factory, name, version)`. New code that doesn't care about factories transparently lands in `default` — every team has one.
8
11
 
9
12
  ---
10
13
 
@@ -25,23 +28,22 @@ A workflow is `async (ctx, sandbox) => T`. Two positional args:
25
28
  - **`ctx`** carries the run identity (`run.id`), the caller's `input`, plus
26
29
  observability helpers (`setMetadata`, `step`).
27
30
  - **`sandbox`** is a capability the engine constructs once for the run —
28
- pass it to `runAgent({ sandbox, ... })` and to any helper that takes a
31
+ pass it to `agent({ sandbox, ... })` and to any helper that takes a
29
32
  `SandboxProvider` (file writers, git utilities, command runners).
30
33
 
31
34
  ```typescript
32
35
  // my-workflow.ts
33
- import { defineWorkflow, runAgent, claudeRuntime } from "@agent-compose/sdk";
36
+ import { defineWorkflow, agent, claudeRuntime } from "@agent-compose/sdk";
34
37
  import PROMPT from "./prompt.md" with { type: "text" };
35
38
 
36
39
  export default defineWorkflow({
37
40
  async run(ctx, sandbox) {
38
41
  const repo = (ctx.input?.repo as string | undefined) ?? "owner/repo";
39
42
 
40
- const result = await runAgent({
43
+ const result = await agent({
41
44
  sandbox,
42
45
  runtime: claudeRuntime,
43
- prompt: PROMPT,
44
- promptVars: { REPO: repo },
46
+ prompt: `${PROMPT}\n\nRepository: ${repo}`,
45
47
  tools: ["Bash", "Read", "Edit", "Write", "Grep", "Glob"],
46
48
  budget: { turnsPerIteration: 40, maxIterations: 8 },
47
49
  });
@@ -83,14 +85,13 @@ interface WorkflowCtx {
83
85
  duration. Use it for setup, external API calls, or anything you want
84
86
  visible on the dashboard's run detail page.
85
87
 
86
- ### The `runAgent` loop
88
+ ### The `agent` loop
87
89
 
88
90
  ```ts
89
- runAgent({
91
+ agent({
90
92
  sandbox, // the workflow's sandbox arg
91
93
  runtime, // claudeRuntime, or your own via createClaudeRuntime / defineRuntime
92
94
  prompt, // raw markdown — `--- frontmatter ---` is auto-stripped
93
- promptVars?, // {{VAR}} substitutions; WORKING_DIR + DIFF_BASE auto-populate
94
95
  tools?, // model tool allowlist (defaults inside agentLoop)
95
96
  budget?, // { turnsPerIteration, maxIterations }
96
97
  workingDir?, // every shell command runs here
@@ -110,7 +111,7 @@ the full instructions appended to every prompt.
110
111
  ## Defining a runtime
111
112
 
112
113
  A "runtime" wraps an agent's underlying execution model — usually a coding
113
- CLI like Claude Code or OpenAI Desktop — so `runAgent` can drive it. The SDK
114
+ CLI like Claude Code or OpenAI Desktop — so `agent` can drive it. The SDK
114
115
  ships built-ins; you only need a custom one for an exotic provider.
115
116
 
116
117
  ### Built-in runtimes
@@ -148,7 +149,7 @@ const myRuntime: AgentRuntime = defineRuntime({
148
149
 
149
150
  `AgentRuntime` is a tagged record with `create(sandbox, RuntimeOptions) →
150
151
  ModelExecutionContract`. There is **no `provider` field** on it — the
151
- runtime is bound to the workflow at author time (you pass it to `runAgent`),
152
+ runtime is bound to the workflow at author time (you pass it to `agent`),
152
153
  not selected by the server.
153
154
 
154
155
  ---
@@ -162,9 +163,10 @@ agentc register my-workflow.ts -n my-workflow
162
163
  ```
163
164
 
164
165
  Under the hood that calls `bundleWorkflow(workflowPath)` (resolves imports,
165
- inlines runtime sources via dynamic-require traversal) and `POST /api/v1/templates`
166
- with the bundled source. If you need to drive registration from your own
167
- build pipeline, you can do the same thing via the SDK directly:
166
+ inlines runtime sources via dynamic-require traversal) and `POST
167
+ /api/v1/factories/<slug>/templates` with the bundled source. If you need
168
+ to drive registration from your own build pipeline, you can do the same
169
+ thing via the SDK directly:
168
170
 
169
171
  ```ts
170
172
  import { AgentComposeClient, bundleWorkflow } from "@agent-compose/sdk";
@@ -176,10 +178,11 @@ const client = new AgentComposeClient(
176
178
 
177
179
  const bundled = await bundleWorkflow("./my-workflow.ts");
178
180
  await client.register({
179
- name: "my-workflow",
180
- source: bundled.source,
181
- runtimes: bundled.runtimes, // [{ name, source }] — embedded so the runner has them locally
182
- schedule: "*/30 * * * *", // optional cron
181
+ name: "my-workflow",
182
+ source: bundled.source,
183
+ runtimes: bundled.runtimes, // [{ name, source }] — embedded so the runner has them locally
184
+ schedule: "*/30 * * * *", // optional cron
185
+ factorySlug: "default", // optional — defaults to "default"
183
186
  // snapshot, saveSnapshot, networkPolicy, placeholders — all optional
184
187
  });
185
188
  ```
@@ -204,7 +207,7 @@ const status = await client.invokeAndWait("my-workflow", { repo: "owner/repo" },
204
207
  timeoutMs: 5 * 60_000,
205
208
  pollIntervalMs: 2000,
206
209
  });
207
- console.log(status.status); // "success" | "failed" | "abandoned"
210
+ console.log(status.status); // "success" | "failed" | "abandoned" | "canceled"
208
211
  console.log(status.output); // workflow's return value
209
212
  ```
210
213
 
@@ -213,6 +216,10 @@ async run() { return … } })` resolves to). `setMetadata()` writes to a
213
216
  separate `metadata` field — useful for "side-channel" facts (PR url, plan
214
217
  url) without polluting the structured return.
215
218
 
219
+ `invoke` and `invokeAndWait` both accept `{ factorySlug, snapshot,
220
+ saveSnapshot, parentRunId }` as the third argument. `factorySlug` defaults
221
+ to `"default"`.
222
+
216
223
  ### Auto parent/child tracing
217
224
 
218
225
  The SDK detects `process.env.RUN_ID` (set by the runner sandbox on every
@@ -221,25 +228,123 @@ dispatch) and automatically threads it as `parentRunId` on subsequent
221
228
  parent/child tree in the dashboard for free. Pass `parentRunId: null`
222
229
  to opt out.
223
230
 
231
+ ### Cancelling a run
232
+
233
+ ```ts
234
+ await client.cancelRun(runId);
235
+ ```
236
+
237
+ Idempotent — cancelling an already-terminal run returns the current state
238
+ without throwing. The server stamps the run as `canceled`, kills any live
239
+ sandboxes, and emits a `run_canceled` event on the stream.
240
+
241
+ ### Streaming live logs
242
+
243
+ `streamRunLogs` returns an async generator of `RunEvent`s in real time,
244
+ re-attaching via SSE under the hood. Pass `lastEventId` (the highest
245
+ `seq` you've already processed) to resume after a reconnect.
246
+
247
+ ```ts
248
+ for await (const ev of client.streamRunLogs(runId, { lastEventId: 0 })) {
249
+ console.log(ev.event, ev.seq, ev.data);
250
+ if (ev.event === "run_complete" || ev.event === "run_failed" || ev.event === "run_canceled") {
251
+ break;
252
+ }
253
+ }
254
+ ```
255
+
256
+ `AbortSignal` works too — pass `{ signal }` and call `controller.abort()`
257
+ to tear the stream down from the caller side.
258
+
259
+ ---
260
+
261
+ ## Factories
262
+
263
+ Factories are project containers within a team. Each factory has its own
264
+ workflow templates, secrets, runs, and (optionally) scoped API keys. New
265
+ projects don't need to think about them — `default` is auto-created per
266
+ team and is what the SDK falls back to when `factorySlug` is omitted.
267
+
268
+ ```ts
269
+ // CRUD on factories
270
+ await client.createFactory({ slug: "ci-bots", name: "CI Bots", description: "…" });
271
+ const factories = await client.listFactories();
272
+ const f = await client.getFactory("ci-bots");
273
+ await client.updateFactory("ci-bots", { name: "Continuous-Integration Bots" });
274
+ await client.deleteFactory("ci-bots");
275
+
276
+ // Templates list — flat across factories, or scoped to one
277
+ const all = await client.listTemplates();
278
+ const scoped = await client.listTemplates({ factorySlug: "ci-bots" });
279
+
280
+ // Register / invoke / secret operations all accept factorySlug
281
+ await client.register({ name: "scrape", source, factorySlug: "ci-bots", … });
282
+ await client.invoke("scrape", { url: "…" }, { factorySlug: "ci-bots" });
283
+ await client.setSecret("scrape", "GH_TOKEN", "ghp_…", { factorySlug: "ci-bots" });
284
+ ```
285
+
286
+ CLI equivalents: `agentc factory list | create | get | update | delete`,
287
+ plus `--factory <slug>` on every other command.
288
+
224
289
  ---
225
290
 
226
291
  ## Per-workflow secrets
227
292
 
228
- Secrets are stored in GCP Secret Manager, one row per `(team, workflow,
229
- key)`. They're injected as env vars into the runner sandbox at dispatch
230
- time, never persisted in the VM. Values are write-only — the API only
231
- returns metadata (key, timestamps).
293
+ Secrets live in GCP Secret Manager, one row per `(factory, workflow, key)`.
294
+ They're injected as env vars into the runner sandbox at dispatch time,
295
+ never persisted in the VM. Values are write-only — the API only returns
296
+ metadata (key, timestamps).
232
297
 
233
298
  ```ts
234
299
  await client.setSecret("my-workflow", "ANTHROPIC_API_KEY", process.env.ANTHROPIC_API_KEY!);
235
300
  const list = await client.listSecrets("my-workflow"); // [{ key, createdAt, updatedAt }]
236
301
  await client.deleteSecret("my-workflow", "STALE_KEY");
302
+
303
+ // Scope to a non-default factory:
304
+ await client.setSecret("scrape", "GH_TOKEN", "ghp_…", { factorySlug: "ci-bots" });
237
305
  ```
238
306
 
239
307
  Mutations require `admin` scope.
240
308
 
241
309
  ---
242
310
 
311
+ ## API keys
312
+
313
+ Mint and list scoped keys programmatically (requires an `admin`-scoped
314
+ caller key). New keys are returned **once**, in the same response as the
315
+ metadata — copy the `ac_…` value immediately.
316
+
317
+ ```ts
318
+ const created = await client.createApiKey({
319
+ name: "ci-dispatcher",
320
+ scopes: ["read", "invoke"],
321
+ expiresAt: new Date(Date.now() + 30 * 86_400_000).toISOString(), // 30 days
322
+ // factorySlug: "ci-bots" // optional — scopes the key to a single factory
323
+ });
324
+ console.log(created.key); // "ac_…" — the only time you'll see this
325
+
326
+ const all = await client.listApiKeys();
327
+ ```
328
+
329
+ CLI equivalent: `agentc keys create <name> --scopes read,invoke
330
+ --expires-in 30d`.
331
+
332
+ ---
333
+
334
+ ## Usage
335
+
336
+ ```ts
337
+ const usage = await client.getUsage(
338
+ new Date(Date.now() - 30 * 86_400_000),
339
+ new Date(),
340
+ );
341
+ // usage.rows: [{ day, runs, sandbox_seconds, … }]
342
+ ```
343
+
344
+ CLI equivalent: `agentc usage`.
345
+
346
+ ---
347
+
243
348
  ## Snapshots (replay-friendly sandboxes)
244
349
 
245
350
  Long-running workflows can capture the runner sandbox as a Vercel snapshot
@@ -294,18 +399,25 @@ programmatic / server-to-server callers.
294
399
  | `defineWorkflow` | Attach metadata to a workflow `run` function |
295
400
  | `defineRuntime` | Wrap an agent execution provider as an `AgentRuntime` |
296
401
  | `defineSandboxEnvironment` | Sugar for declaring a workflow whose primary purpose is to build a snapshot for others to boot from |
297
- | `runAgent` / `agentLoop` | Embed an LLM loop inside a workflow |
298
- | `claudeRuntime` / `createClaudeRuntime` / `ClaudeRunner` | Built-in Claude Code runtime + factory |
299
- | `AgentComposeClient` | HTTP client (register, invoke, status, snapshots, secrets) |
402
+ | `agent` / `agentLoop` | Embed an LLM loop inside a workflow |
300
403
  | `runWorkflow` | Local engine for running a workflow in-process (test harness) |
301
404
  | `bundleWorkflow` | Resolve + inline a workflow's runtime sources for registration |
405
+ | `claudeRuntime` / `createClaudeRuntime` / `ClaudeRunner` | Built-in Claude Code runtime + factory |
406
+ | `AgentComposeClient` | HTTP client — register, invoke, cancel, stream logs, factories, snapshots, secrets, API keys, usage |
407
+ | `AgentComposeError` | Thrown by every non-2xx HTTP response |
302
408
  | `parseAgentStatus` / `parseAgentResponse` / `AgentStatusSchema` / `AgentMessageSchema` | Protocol parsers |
303
-
304
- Type exports: `WorkflowFn`, `WorkflowCtx`, `WorkflowDefinition`, `AgentBudget`,
305
- `AgentRuntime`, `RuntimeOptions`, `ModelExecutionContract`,
306
- `AgentMessage` (and its variants), `AgentStatus`, `RunStatus`,
307
- `RegisterResult`, `RunEvent`, `AgentLoopResult`, `RunAgentOpts`,
308
- `SandboxProvider`, `SandboxNetworkPolicy`, `BundledWorkflow`.
409
+ | `parseSseStream` | Generic SSE chunk decoder (used by `streamRunLogs`) |
410
+ | `createSandbox` / `reconnectSandbox` / `killAllSandboxes` / `killSandboxById` / `getSandboxQuotas` / `listOwnedSandboxes` / `deleteSandboxSnapshot` | Sandbox-provider helpers (Vercel + E2B) |
411
+
412
+ Type exports: `WorkflowFn`, `WorkflowCtx`, `WorkflowDefinition`,
413
+ `WorkflowHooks`, `AgentBudget`, `AgentRuntime`, `RuntimeOptions`,
414
+ `ModelExecutionContract`, `McpServerConfig`, `AgentMessage` (and its
415
+ variants), `AgentStatus`, `RunStatus`, `RegisterResult`, `RunEvent`,
416
+ `FactoryRow`, `SnapshotListEntry`, `ApiKey`, `ApiKeyCreated`,
417
+ `UsageRollupRow`, `UsageResponse`, `CancelRunResponse`, `AgentLoopResult`,
418
+ `AgentOpts`, `SandboxProvider`, `DesktopSandboxProvider`,
419
+ `SandboxNetworkPolicy`, `SandboxCreateOpts`, `OwnedSandbox`,
420
+ `BundledWorkflow`.
309
421
 
310
422
  For the canonical signatures, follow your IDE's go-to-definition into
311
423
  `@agent-compose/sdk` — `sdk/src/index.ts` is the public surface and the
@@ -5,23 +5,101 @@
5
5
  import type { RuntimeOptions, ModelExecutionContract } from "../index.js";
6
6
  import { z } from "zod";
7
7
  import type { AgentStatus, AgentMessage } from "./protocol.js";
8
+ import type { Processor } from "../processors/processor.js";
9
+ import { RequestContext } from "../request-context/request-context.js";
8
10
  export declare const DEFAULT_CLAUDE_MODEL = "claude-opus-4-7";
9
11
  export declare function parseAgentStatus(text: string): AgentStatus | null;
10
- export interface AgentLoopResult {
12
+ export interface AgentLoopResult<TResponse = unknown> {
13
+ agentId: string;
14
+ label: string;
11
15
  sessionId: string;
12
16
  lastStatus: AgentStatus | null;
13
17
  iterations: number;
14
- response?: unknown;
18
+ response?: TResponse;
15
19
  }
16
- export declare function agentLoop(opts: {
20
+ export type AgentMessageSummary = {
21
+ type: "init";
22
+ sessionId: string;
23
+ } | {
24
+ type: "text";
25
+ text: string;
26
+ } | {
27
+ type: "thinking";
28
+ text: string;
29
+ } | {
30
+ type: "tool_use";
31
+ toolName: string;
32
+ toolUseId: string;
33
+ toolInputPreview: string;
34
+ } | {
35
+ type: "tool_result";
36
+ toolUseId: string;
37
+ output: string;
38
+ isError: boolean;
39
+ } | {
40
+ type: "usage";
41
+ inputTokens: number;
42
+ outputTokens: number;
43
+ cacheReadTokens: number;
44
+ cacheCreationTokens: number;
45
+ durationMs: number;
46
+ numTurns: number;
47
+ } | {
48
+ type: "done";
49
+ sessionId: string;
50
+ } | {
51
+ type: "error";
52
+ text: string;
53
+ };
54
+ export declare function summarizeAgentMessage(msg: AgentMessage): AgentMessageSummary;
55
+ export type AgentLifecycleEvent = {
56
+ event: "agent.spawned";
57
+ at: number;
58
+ agentId: string;
59
+ label: string;
60
+ allowedTools?: string[];
61
+ } | {
62
+ event: "agent.message";
63
+ at: number;
64
+ agentId: string;
65
+ label: string;
66
+ iteration: number;
67
+ message: AgentMessageSummary;
68
+ } | {
69
+ event: "agent.iteration";
70
+ at: number;
71
+ agentId: string;
72
+ label: string;
73
+ iteration: number;
74
+ status: AgentStatus | null;
75
+ } | {
76
+ event: "agent.settled";
77
+ at: number;
78
+ agentId: string;
79
+ label: string;
80
+ outcome: "success" | "failed";
81
+ iterations: number;
82
+ durationMs: number;
83
+ failureReason?: string;
84
+ };
85
+ export interface AgentLoopOpts<TResponse = unknown> {
86
+ agentId?: string;
17
87
  label?: string;
88
+ onAgentLifecycleEvent?: (event: AgentLifecycleEvent) => void;
18
89
  onIteration?: (iteration: number, status: AgentStatus | null) => void;
19
90
  turnsPerIteration?: number;
20
91
  maxIterations?: number;
21
92
  buildPrompt: (lastStatus: AgentStatus | null, iteration: number) => string;
22
93
  onAgentEvent?: (iteration: number, msg: AgentMessage) => void;
23
94
  allowedTools?: string[];
24
- responseSchema?: z.ZodType<unknown>;
95
+ responseSchema?: z.ZodType<TResponse>;
25
96
  runtime?: (opts: RuntimeOptions) => ModelExecutionContract;
26
97
  cwd?: string;
27
- }): Promise<AgentLoopResult>;
98
+ /** Processor chain — see sdk/src/processors. Runs sequentially around
99
+ * prompts and emitted messages. Tool-call gating is a no-op until a
100
+ * runtime that exposes a pre-tool-use seam is wired (candidate #1). */
101
+ processors?: readonly Processor[];
102
+ /** Per-run typed bag — propagated to every processor as `ctx.requestContext`. */
103
+ requestContext?: RequestContext;
104
+ }
105
+ export declare function agentLoop<TResponse = unknown>(opts: AgentLoopOpts<TResponse>): Promise<AgentLoopResult<TResponse>>;
@@ -1,22 +1,25 @@
1
1
  /**
2
- * runAgent — canonical entry point for embedding an LLM agent inside a
2
+ * agent — canonical entry point for embedding an LLM agent inside a
3
3
  * workflow. The workflow's `run()` body calls it; the loop executes
4
4
  * against the runner's own VM.
5
5
  *
6
6
  * Glue packaged so workflows don't duplicate it:
7
- * - Inject `{{VAR}}` placeholders into the prompt template.
8
7
  * - Strip the `--- frontmatter ---` header authors use for IDE hints.
9
8
  * - Append PROTOCOL_SUFFIX (status/response format instructions).
10
9
  * - Append a response-format appendix when `responseSchema` is set.
11
10
  * - Delegate to `agentLoop`.
12
11
  */
13
12
  import { z } from "zod";
14
- import type { AgentLoopResult } from "./agent-loop.js";
13
+ import type { AgentLifecycleEvent, AgentLoopResult } from "./agent-loop.js";
15
14
  import type { AgentMessage, AgentStatus } from "../types/protocol.js";
16
15
  import type { AgentRuntime } from "../types/runtime.js";
17
16
  import type { SandboxProvider } from "../types/sandbox.js";
18
17
  import type { AgentBudget } from "../types/workflow.js";
19
- export interface RunAgentOpts<T = unknown> {
18
+ import type { Processor } from "../processors/processor.js";
19
+ import { RequestContext } from "../request-context/request-context.js";
20
+ export interface AgentOpts<T = unknown> {
21
+ /** Stable system identity for this agent loop. Generated when omitted. */
22
+ agentId?: string;
20
23
  /** Sandbox the runtime executes commands against. Inside a workflow,
21
24
  * always pass `ctx.sandbox` — a pre-constructed local provider for
22
25
  * the runner's own VM. Exposed as a parameter so tests and non-workflow
@@ -24,12 +27,9 @@ export interface RunAgentOpts<T = unknown> {
24
27
  sandbox: SandboxProvider;
25
28
  /** Runtime definition from `createClaudeRuntime({...})` (or custom). */
26
29
  runtime: AgentRuntime;
27
- /** Prompt template. Authors can include YAML-style `--- frontmatter ---`
30
+ /** Prompt text. Authors can include YAML-style `--- frontmatter ---`
28
31
  * at the top for IDE hints; it's stripped before the model sees it. */
29
32
  prompt: string;
30
- /** Substitution map for `{{VAR}}` placeholders in the prompt. `WORKING_DIR`
31
- * and `DIFF_BASE` auto-populate from `opts.workingDir` unless overridden. */
32
- promptVars?: Record<string, string>;
33
33
  /** `cwd` forwarded to the runtime — every shell command runs here. */
34
34
  workingDir?: string;
35
35
  /** Tools the model may use. Defaults to a safe kitchen-sink set inside
@@ -43,16 +43,41 @@ export interface RunAgentOpts<T = unknown> {
43
43
  responseSchema?: z.ZodType<T>;
44
44
  /** Label prefix for runtime stderr ("[sbid][agent]" by default). */
45
45
  label?: string;
46
+ /** Lifecycle event sink from workflow ctx. Emits agent.spawned / iteration / settled. */
47
+ events?: {
48
+ emit: (event: AgentLifecycleEvent) => void | Promise<void>;
49
+ };
46
50
  /** Per-message event callback — wire this to your workflow's event
47
51
  * telemetry if you want per-tool-call observability. */
48
52
  onAgentEvent?: (iteration: number, msg: AgentMessage) => void;
49
53
  /** Per-iteration status callback — fires after each model turn with the
50
54
  * parsed `<status>` block (or null if the model didn't emit one). */
51
55
  onIteration?: (iteration: number, status: AgentStatus | null) => void;
56
+ /**
57
+ * Processors run as a typed pre/post pipeline around the agent loop:
58
+ * - processInput before each iteration's prompt is sent
59
+ * - processOutput on each emitted message
60
+ * - processToolCall before each tool call (runtime-driven; dormant
61
+ * until a runtime that supports pre-tool gating is wired in)
62
+ *
63
+ * Common pattern with workflow-level defaults:
64
+ * agent({ processors: [...ctx.processors, mySpecific], ... })
65
+ */
66
+ processors?: readonly Processor[];
67
+ /**
68
+ * Per-run request context. Passed through to processors as
69
+ * `ctx.requestContext`; carries tenant identity (teamId, factoryId, scopes)
70
+ * and the freeform user namespace.
71
+ *
72
+ * Inside a workflow, pass `ctx.requestContext`. Tests/non-workflow callers
73
+ * can omit; a degenerate context is synthesised so processors that don't
74
+ * read identity (e.g. redactPattern) still work.
75
+ */
76
+ requestContext?: RequestContext;
52
77
  }
53
78
  /**
54
79
  * Run an agent loop inside a workflow. Returns the loop's final
55
80
  * `AgentLoopResult`, including `response` when a `responseSchema` was
56
81
  * supplied and the model validated against it.
57
82
  */
58
- export declare function runAgent<T = unknown>(opts: RunAgentOpts<T>): Promise<AgentLoopResult>;
83
+ export declare function agent<T = unknown>(opts: AgentOpts<T>): Promise<AgentLoopResult<T>>;