@agent-compose/sdk 0.2.0 → 0.2.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +22 -136
- package/dist/client.d.ts +5 -0
- package/dist/index.js +6 -2
- package/dist/runtimes/openai-desktop.js +6 -2
- package/package.json +10 -2
- package/src/agent/agent-loop.ts +131 -0
- package/src/agent/protocol-suffix.md +57 -0
- package/src/agent/protocol.ts +22 -0
- package/src/agent/run-agent.ts +141 -0
- package/src/client.ts +446 -0
- package/src/env.d.ts +5 -0
- package/src/errors.ts +10 -0
- package/src/index.ts +111 -0
- package/src/runtimes/claude.ts +305 -0
- package/src/runtimes/openai-desktop.ts +151 -0
- package/src/sandbox.ts +458 -0
- package/src/sse.ts +56 -0
- package/src/types/events.ts +51 -0
- package/src/types/protocol.ts +74 -0
- package/src/types/runtime.ts +59 -0
- package/src/types/sandbox-environment.ts +64 -0
- package/src/types/sandbox.ts +51 -0
- package/src/types/workflow.ts +128 -0
- package/src/utils/bundler.ts +81 -0
- package/src/utils/discovery.ts +4 -0
- package/src/utils/errors.ts +4 -0
- package/src/utils/schemas.ts +10 -0
- package/src/utils/source-loader.ts +16 -0
- package/src/workflows/engine.ts +110 -0
package/README.md
CHANGED
|
@@ -1,13 +1,10 @@
|
|
|
1
1
|
# @agent-compose/sdk
|
|
2
2
|
|
|
3
|
-
TypeScript SDK for
|
|
3
|
+
TypeScript SDK for agent-compose. Use it to:
|
|
4
4
|
|
|
5
5
|
- **Author workflows** that run agentic LLM loops inside isolated sandboxes
|
|
6
6
|
- **Define runtimes** that wrap a coding-CLI tool (Claude Code, OpenAI Desktop, …) into a sandbox-portable agent loop
|
|
7
|
-
- **Register, invoke,
|
|
8
|
-
- **Manage factories, secrets, API keys, and snapshots** programmatically
|
|
9
|
-
|
|
10
|
-
The hierarchy: a **team** owns one or more **factories** (project containers); each factory owns workflow templates, secrets, and runs. Workflows are versioned per `(factory, name, version)`. New code that doesn't care about factories transparently lands in `default` — every team has one.
|
|
7
|
+
- **Register, invoke, and observe** workflows via the HTTP API (`AgentComposeClient`)
|
|
11
8
|
|
|
12
9
|
---
|
|
13
10
|
|
|
@@ -165,10 +162,9 @@ agentc register my-workflow.ts -n my-workflow
|
|
|
165
162
|
```
|
|
166
163
|
|
|
167
164
|
Under the hood that calls `bundleWorkflow(workflowPath)` (resolves imports,
|
|
168
|
-
inlines runtime sources via dynamic-require traversal) and `POST
|
|
169
|
-
|
|
170
|
-
|
|
171
|
-
thing via the SDK directly:
|
|
165
|
+
inlines runtime sources via dynamic-require traversal) and `POST /api/v1/templates`
|
|
166
|
+
with the bundled source. If you need to drive registration from your own
|
|
167
|
+
build pipeline, you can do the same thing via the SDK directly:
|
|
172
168
|
|
|
173
169
|
```ts
|
|
174
170
|
import { AgentComposeClient, bundleWorkflow } from "@agent-compose/sdk";
|
|
@@ -180,11 +176,10 @@ const client = new AgentComposeClient(
|
|
|
180
176
|
|
|
181
177
|
const bundled = await bundleWorkflow("./my-workflow.ts");
|
|
182
178
|
await client.register({
|
|
183
|
-
name:
|
|
184
|
-
source:
|
|
185
|
-
runtimes:
|
|
186
|
-
schedule:
|
|
187
|
-
factorySlug: "default", // optional — defaults to "default"
|
|
179
|
+
name: "my-workflow",
|
|
180
|
+
source: bundled.source,
|
|
181
|
+
runtimes: bundled.runtimes, // [{ name, source }] — embedded so the runner has them locally
|
|
182
|
+
schedule: "*/30 * * * *", // optional cron
|
|
188
183
|
// snapshot, saveSnapshot, networkPolicy, placeholders — all optional
|
|
189
184
|
});
|
|
190
185
|
```
|
|
@@ -209,7 +204,7 @@ const status = await client.invokeAndWait("my-workflow", { repo: "owner/repo" },
|
|
|
209
204
|
timeoutMs: 5 * 60_000,
|
|
210
205
|
pollIntervalMs: 2000,
|
|
211
206
|
});
|
|
212
|
-
console.log(status.status); // "success" | "failed" | "abandoned"
|
|
207
|
+
console.log(status.status); // "success" | "failed" | "abandoned"
|
|
213
208
|
console.log(status.output); // workflow's return value
|
|
214
209
|
```
|
|
215
210
|
|
|
@@ -218,10 +213,6 @@ async run() { return … } })` resolves to). `setMetadata()` writes to a
|
|
|
218
213
|
separate `metadata` field — useful for "side-channel" facts (PR url, plan
|
|
219
214
|
url) without polluting the structured return.
|
|
220
215
|
|
|
221
|
-
`invoke` and `invokeAndWait` both accept `{ factorySlug, snapshot,
|
|
222
|
-
saveSnapshot, parentRunId }` as the third argument. `factorySlug` defaults
|
|
223
|
-
to `"default"`.
|
|
224
|
-
|
|
225
216
|
### Auto parent/child tracing
|
|
226
217
|
|
|
227
218
|
The SDK detects `process.env.RUN_ID` (set by the runner sandbox on every
|
|
@@ -230,123 +221,25 @@ dispatch) and automatically threads it as `parentRunId` on subsequent
|
|
|
230
221
|
parent/child tree in the dashboard for free. Pass `parentRunId: null`
|
|
231
222
|
to opt out.
|
|
232
223
|
|
|
233
|
-
### Cancelling a run
|
|
234
|
-
|
|
235
|
-
```ts
|
|
236
|
-
await client.cancelRun(runId);
|
|
237
|
-
```
|
|
238
|
-
|
|
239
|
-
Idempotent — cancelling an already-terminal run returns the current state
|
|
240
|
-
without throwing. The server stamps the run as `canceled`, kills any live
|
|
241
|
-
sandboxes, and emits a `run_canceled` event on the stream.
|
|
242
|
-
|
|
243
|
-
### Streaming live logs
|
|
244
|
-
|
|
245
|
-
`streamRunLogs` returns an async generator of `RunEvent`s in real time,
|
|
246
|
-
re-attaching via SSE under the hood. Pass `lastEventId` (the highest
|
|
247
|
-
`seq` you've already processed) to resume after a reconnect.
|
|
248
|
-
|
|
249
|
-
```ts
|
|
250
|
-
for await (const ev of client.streamRunLogs(runId, { lastEventId: 0 })) {
|
|
251
|
-
console.log(ev.event, ev.seq, ev.data);
|
|
252
|
-
if (ev.event === "run_complete" || ev.event === "run_failed" || ev.event === "run_canceled") {
|
|
253
|
-
break;
|
|
254
|
-
}
|
|
255
|
-
}
|
|
256
|
-
```
|
|
257
|
-
|
|
258
|
-
`AbortSignal` works too — pass `{ signal }` and call `controller.abort()`
|
|
259
|
-
to tear the stream down from the caller side.
|
|
260
|
-
|
|
261
|
-
---
|
|
262
|
-
|
|
263
|
-
## Factories
|
|
264
|
-
|
|
265
|
-
Factories are project containers within a team. Each factory has its own
|
|
266
|
-
workflow templates, secrets, runs, and (optionally) scoped API keys. New
|
|
267
|
-
projects don't need to think about them — `default` is auto-created per
|
|
268
|
-
team and is what the SDK falls back to when `factorySlug` is omitted.
|
|
269
|
-
|
|
270
|
-
```ts
|
|
271
|
-
// CRUD on factories
|
|
272
|
-
await client.createFactory({ slug: "ci-bots", name: "CI Bots", description: "…" });
|
|
273
|
-
const factories = await client.listFactories();
|
|
274
|
-
const f = await client.getFactory("ci-bots");
|
|
275
|
-
await client.updateFactory("ci-bots", { name: "Continuous-Integration Bots" });
|
|
276
|
-
await client.deleteFactory("ci-bots");
|
|
277
|
-
|
|
278
|
-
// Templates list — flat across factories, or scoped to one
|
|
279
|
-
const all = await client.listTemplates();
|
|
280
|
-
const scoped = await client.listTemplates({ factorySlug: "ci-bots" });
|
|
281
|
-
|
|
282
|
-
// Register / invoke / secret operations all accept factorySlug
|
|
283
|
-
await client.register({ name: "scrape", source, factorySlug: "ci-bots", … });
|
|
284
|
-
await client.invoke("scrape", { url: "…" }, { factorySlug: "ci-bots" });
|
|
285
|
-
await client.setSecret("scrape", "GH_TOKEN", "ghp_…", { factorySlug: "ci-bots" });
|
|
286
|
-
```
|
|
287
|
-
|
|
288
|
-
CLI equivalents: `agentc factory list | create | get | update | delete`,
|
|
289
|
-
plus `--factory <slug>` on every other command.
|
|
290
|
-
|
|
291
224
|
---
|
|
292
225
|
|
|
293
226
|
## Per-workflow secrets
|
|
294
227
|
|
|
295
|
-
Secrets
|
|
296
|
-
They're injected as env vars into the runner sandbox at dispatch
|
|
297
|
-
never persisted in the VM. Values are write-only — the API only
|
|
298
|
-
metadata (key, timestamps).
|
|
228
|
+
Secrets are stored in GCP Secret Manager, one row per `(team, workflow,
|
|
229
|
+
key)`. They're injected as env vars into the runner sandbox at dispatch
|
|
230
|
+
time, never persisted in the VM. Values are write-only — the API only
|
|
231
|
+
returns metadata (key, timestamps).
|
|
299
232
|
|
|
300
233
|
```ts
|
|
301
234
|
await client.setSecret("my-workflow", "ANTHROPIC_API_KEY", process.env.ANTHROPIC_API_KEY!);
|
|
302
235
|
const list = await client.listSecrets("my-workflow"); // [{ key, createdAt, updatedAt }]
|
|
303
236
|
await client.deleteSecret("my-workflow", "STALE_KEY");
|
|
304
|
-
|
|
305
|
-
// Scope to a non-default factory:
|
|
306
|
-
await client.setSecret("scrape", "GH_TOKEN", "ghp_…", { factorySlug: "ci-bots" });
|
|
307
237
|
```
|
|
308
238
|
|
|
309
239
|
Mutations require `admin` scope.
|
|
310
240
|
|
|
311
241
|
---
|
|
312
242
|
|
|
313
|
-
## API keys
|
|
314
|
-
|
|
315
|
-
Mint and list scoped keys programmatically (requires an `admin`-scoped
|
|
316
|
-
caller key). New keys are returned **once**, in the same response as the
|
|
317
|
-
metadata — copy the `ac_…` value immediately.
|
|
318
|
-
|
|
319
|
-
```ts
|
|
320
|
-
const created = await client.createApiKey({
|
|
321
|
-
name: "ci-dispatcher",
|
|
322
|
-
scopes: ["read", "invoke"],
|
|
323
|
-
expiresAt: new Date(Date.now() + 30 * 86_400_000).toISOString(), // 30 days
|
|
324
|
-
// factorySlug: "ci-bots" // optional — scopes the key to a single factory
|
|
325
|
-
});
|
|
326
|
-
console.log(created.key); // "ac_…" — the only time you'll see this
|
|
327
|
-
|
|
328
|
-
const all = await client.listApiKeys();
|
|
329
|
-
```
|
|
330
|
-
|
|
331
|
-
CLI equivalent: `agentc keys create <name> --scopes read,invoke
|
|
332
|
-
--expires-in 30d`.
|
|
333
|
-
|
|
334
|
-
---
|
|
335
|
-
|
|
336
|
-
## Usage
|
|
337
|
-
|
|
338
|
-
```ts
|
|
339
|
-
const usage = await client.getUsage(
|
|
340
|
-
new Date(Date.now() - 30 * 86_400_000),
|
|
341
|
-
new Date(),
|
|
342
|
-
);
|
|
343
|
-
// usage.rows: [{ day, runs, sandbox_seconds, … }]
|
|
344
|
-
```
|
|
345
|
-
|
|
346
|
-
CLI equivalent: `agentc usage`.
|
|
347
|
-
|
|
348
|
-
---
|
|
349
|
-
|
|
350
243
|
## Snapshots (replay-friendly sandboxes)
|
|
351
244
|
|
|
352
245
|
Long-running workflows can capture the runner sandbox as a Vercel snapshot
|
|
@@ -402,24 +295,17 @@ programmatic / server-to-server callers.
|
|
|
402
295
|
| `defineRuntime` | Wrap an agent execution provider as an `AgentRuntime` |
|
|
403
296
|
| `defineSandboxEnvironment` | Sugar for declaring a workflow whose primary purpose is to build a snapshot for others to boot from |
|
|
404
297
|
| `runAgent` / `agentLoop` | Embed an LLM loop inside a workflow |
|
|
298
|
+
| `claudeRuntime` / `createClaudeRuntime` / `ClaudeRunner` | Built-in Claude Code runtime + factory |
|
|
299
|
+
| `AgentComposeClient` | HTTP client (register, invoke, status, snapshots, secrets) |
|
|
405
300
|
| `runWorkflow` | Local engine for running a workflow in-process (test harness) |
|
|
406
301
|
| `bundleWorkflow` | Resolve + inline a workflow's runtime sources for registration |
|
|
407
|
-
| `claudeRuntime` / `createClaudeRuntime` / `ClaudeRunner` | Built-in Claude Code runtime + factory |
|
|
408
|
-
| `AgentComposeClient` | HTTP client — register, invoke, cancel, stream logs, factories, snapshots, secrets, API keys, usage |
|
|
409
|
-
| `AgentComposeError` | Thrown by every non-2xx HTTP response |
|
|
410
302
|
| `parseAgentStatus` / `parseAgentResponse` / `AgentStatusSchema` / `AgentMessageSchema` | Protocol parsers |
|
|
411
|
-
|
|
412
|
-
|
|
413
|
-
|
|
414
|
-
|
|
415
|
-
`
|
|
416
|
-
`
|
|
417
|
-
variants), `AgentStatus`, `RunStatus`, `RegisterResult`, `RunEvent`,
|
|
418
|
-
`FactoryRow`, `SnapshotListEntry`, `ApiKey`, `ApiKeyCreated`,
|
|
419
|
-
`UsageRollupRow`, `UsageResponse`, `CancelRunResponse`, `AgentLoopResult`,
|
|
420
|
-
`RunAgentOpts`, `SandboxProvider`, `DesktopSandboxProvider`,
|
|
421
|
-
`SandboxNetworkPolicy`, `SandboxCreateOpts`, `OwnedSandbox`,
|
|
422
|
-
`BundledWorkflow`.
|
|
303
|
+
|
|
304
|
+
Type exports: `WorkflowFn`, `WorkflowCtx`, `WorkflowDefinition`, `AgentBudget`,
|
|
305
|
+
`AgentRuntime`, `RuntimeOptions`, `ModelExecutionContract`,
|
|
306
|
+
`AgentMessage` (and its variants), `AgentStatus`, `RunStatus`,
|
|
307
|
+
`RegisterResult`, `RunEvent`, `AgentLoopResult`, `RunAgentOpts`,
|
|
308
|
+
`SandboxProvider`, `SandboxNetworkPolicy`, `BundledWorkflow`.
|
|
423
309
|
|
|
424
310
|
For the canonical signatures, follow your IDE's go-to-definition into
|
|
425
311
|
`@agent-compose/sdk` — `sdk/src/index.ts` is the public surface and the
|
package/dist/client.d.ts
CHANGED
|
@@ -11,6 +11,7 @@
|
|
|
11
11
|
* register`) or build sources yourself and pass them directly.
|
|
12
12
|
*/
|
|
13
13
|
import type { RunEvent } from "./types/events.js";
|
|
14
|
+
import type { SandboxNetworkPolicy } from "./sandbox.js";
|
|
14
15
|
export interface RegisterResult {
|
|
15
16
|
id: string;
|
|
16
17
|
name: string;
|
|
@@ -136,6 +137,8 @@ export declare class AgentComposeClient {
|
|
|
136
137
|
saveSnapshot?: boolean;
|
|
137
138
|
parentRunId?: string | null;
|
|
138
139
|
factorySlug?: string;
|
|
140
|
+
networkPolicy?: SandboxNetworkPolicy;
|
|
141
|
+
placeholders?: Record<string, string>;
|
|
139
142
|
}): Promise<{
|
|
140
143
|
id: string;
|
|
141
144
|
}>;
|
|
@@ -151,6 +154,8 @@ export declare class AgentComposeClient {
|
|
|
151
154
|
saveSnapshot?: boolean;
|
|
152
155
|
parentRunId?: string | null;
|
|
153
156
|
factorySlug?: string;
|
|
157
|
+
networkPolicy?: SandboxNetworkPolicy;
|
|
158
|
+
placeholders?: Record<string, string>;
|
|
154
159
|
timeoutMs?: number;
|
|
155
160
|
pollIntervalMs?: number;
|
|
156
161
|
}): Promise<RunStatus>;
|
package/dist/index.js
CHANGED
|
@@ -455,7 +455,9 @@ class AgentComposeClient {
|
|
|
455
455
|
input,
|
|
456
456
|
...opts?.snapshot !== undefined ? { snapshot: opts.snapshot } : {},
|
|
457
457
|
...opts?.saveSnapshot !== undefined ? { saveSnapshot: opts.saveSnapshot } : {},
|
|
458
|
-
...parentRunId ? { parentRunId } : {}
|
|
458
|
+
...parentRunId ? { parentRunId } : {},
|
|
459
|
+
...opts?.networkPolicy !== undefined ? { networkPolicy: opts.networkPolicy } : {},
|
|
460
|
+
...opts?.placeholders !== undefined ? { placeholders: opts.placeholders } : {}
|
|
459
461
|
}
|
|
460
462
|
});
|
|
461
463
|
}
|
|
@@ -466,7 +468,9 @@ class AgentComposeClient {
|
|
|
466
468
|
...opts?.snapshot !== undefined ? { snapshot: opts.snapshot } : {},
|
|
467
469
|
...opts?.saveSnapshot !== undefined ? { saveSnapshot: opts.saveSnapshot } : {},
|
|
468
470
|
...opts?.parentRunId !== undefined ? { parentRunId: opts.parentRunId } : {},
|
|
469
|
-
...opts?.factorySlug !== undefined ? { factorySlug: opts.factorySlug } : {}
|
|
471
|
+
...opts?.factorySlug !== undefined ? { factorySlug: opts.factorySlug } : {},
|
|
472
|
+
...opts?.networkPolicy !== undefined ? { networkPolicy: opts.networkPolicy } : {},
|
|
473
|
+
...opts?.placeholders !== undefined ? { placeholders: opts.placeholders } : {}
|
|
470
474
|
});
|
|
471
475
|
const deadline = Date.now() + timeoutMs;
|
|
472
476
|
while (Date.now() < deadline) {
|
|
@@ -455,7 +455,9 @@ class AgentComposeClient {
|
|
|
455
455
|
input,
|
|
456
456
|
...opts?.snapshot !== undefined ? { snapshot: opts.snapshot } : {},
|
|
457
457
|
...opts?.saveSnapshot !== undefined ? { saveSnapshot: opts.saveSnapshot } : {},
|
|
458
|
-
...parentRunId ? { parentRunId } : {}
|
|
458
|
+
...parentRunId ? { parentRunId } : {},
|
|
459
|
+
...opts?.networkPolicy !== undefined ? { networkPolicy: opts.networkPolicy } : {},
|
|
460
|
+
...opts?.placeholders !== undefined ? { placeholders: opts.placeholders } : {}
|
|
459
461
|
}
|
|
460
462
|
});
|
|
461
463
|
}
|
|
@@ -466,7 +468,9 @@ class AgentComposeClient {
|
|
|
466
468
|
...opts?.snapshot !== undefined ? { snapshot: opts.snapshot } : {},
|
|
467
469
|
...opts?.saveSnapshot !== undefined ? { saveSnapshot: opts.saveSnapshot } : {},
|
|
468
470
|
...opts?.parentRunId !== undefined ? { parentRunId: opts.parentRunId } : {},
|
|
469
|
-
...opts?.factorySlug !== undefined ? { factorySlug: opts.factorySlug } : {}
|
|
471
|
+
...opts?.factorySlug !== undefined ? { factorySlug: opts.factorySlug } : {},
|
|
472
|
+
...opts?.networkPolicy !== undefined ? { networkPolicy: opts.networkPolicy } : {},
|
|
473
|
+
...opts?.placeholders !== undefined ? { placeholders: opts.placeholders } : {}
|
|
470
474
|
});
|
|
471
475
|
const deadline = Date.now() + timeoutMs;
|
|
472
476
|
while (Date.now() < deadline) {
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@agent-compose/sdk",
|
|
3
|
-
"version": "0.2.
|
|
3
|
+
"version": "0.2.2",
|
|
4
4
|
"description": "Client library for agent-compose — define agents, runtimes, and workflows, and invoke them against an agent-compose server.",
|
|
5
5
|
"license": "MIT",
|
|
6
6
|
"repository": {
|
|
@@ -24,7 +24,15 @@
|
|
|
24
24
|
"default": "./dist/runtimes/openai-desktop.js"
|
|
25
25
|
}
|
|
26
26
|
},
|
|
27
|
-
"files": [
|
|
27
|
+
"files": [
|
|
28
|
+
"dist",
|
|
29
|
+
"src/**/*.ts",
|
|
30
|
+
"src/**/*.md",
|
|
31
|
+
"!src/**/__tests__/**",
|
|
32
|
+
"!src/**/*.test.ts",
|
|
33
|
+
"README.md",
|
|
34
|
+
"LICENSE"
|
|
35
|
+
],
|
|
28
36
|
"engines": {
|
|
29
37
|
"node": ">=20"
|
|
30
38
|
},
|
|
@@ -0,0 +1,131 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Agent loop — runner-agnostic iteration driver operating through ModelExecutionContract.
|
|
3
|
+
* Each iteration: build prompt → sendMessage() → parse <status> → done / continue / circuit-break.
|
|
4
|
+
*/
|
|
5
|
+
|
|
6
|
+
import type { RuntimeOptions, ModelExecutionContract } from "../index.js";
|
|
7
|
+
import { z } from "zod";
|
|
8
|
+
import { AgentStatusSchema, parseAgentResponse } from "./protocol.js";
|
|
9
|
+
import type { AgentStatus, AgentMessage } from "./protocol.js";
|
|
10
|
+
|
|
11
|
+
export const DEFAULT_CLAUDE_MODEL = "claude-opus-4-7";
|
|
12
|
+
|
|
13
|
+
const SAME_BLOCKER_ITERATIONS = 3;
|
|
14
|
+
const STALL_ITERATIONS = 3;
|
|
15
|
+
|
|
16
|
+
export function parseAgentStatus(text: string): AgentStatus | null {
|
|
17
|
+
const match = text.match(/<status>([\s\S]*?)<\/status>/);
|
|
18
|
+
if (!match) return null;
|
|
19
|
+
try {
|
|
20
|
+
const result = AgentStatusSchema.safeParse(JSON.parse(match[1].trim()));
|
|
21
|
+
return result.success ? result.data : null;
|
|
22
|
+
} catch { return null; }
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
const DEFAULT_ALLOWED_TOOLS = ["Read", "Write", "Edit", "Bash", "Glob", "Grep", "WebFetch"];
|
|
26
|
+
|
|
27
|
+
export interface AgentLoopResult {
|
|
28
|
+
sessionId: string;
|
|
29
|
+
lastStatus: AgentStatus | null;
|
|
30
|
+
iterations: number;
|
|
31
|
+
response?: unknown;
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
export async function agentLoop(opts: {
|
|
35
|
+
label?: string;
|
|
36
|
+
onIteration?: (iteration: number, status: AgentStatus | null) => void;
|
|
37
|
+
turnsPerIteration?: number;
|
|
38
|
+
maxIterations?: number;
|
|
39
|
+
buildPrompt: (lastStatus: AgentStatus | null, iteration: number) => string;
|
|
40
|
+
onAgentEvent?: (iteration: number, msg: AgentMessage) => void;
|
|
41
|
+
allowedTools?: string[];
|
|
42
|
+
responseSchema?: z.ZodType<unknown>;
|
|
43
|
+
runtime?: (opts: RuntimeOptions) => ModelExecutionContract;
|
|
44
|
+
cwd?: string;
|
|
45
|
+
}): Promise<AgentLoopResult> {
|
|
46
|
+
const label = opts.label ?? "[Agent Loop]";
|
|
47
|
+
const turnsPerIteration = opts.turnsPerIteration ?? 40;
|
|
48
|
+
const maxIterations = opts.maxIterations ?? 8;
|
|
49
|
+
|
|
50
|
+
if (!opts.runtime) throw new Error("agentLoop: opts.runtime is required");
|
|
51
|
+
const client = opts.runtime({
|
|
52
|
+
maxTurns: turnsPerIteration,
|
|
53
|
+
allowedTools: opts.allowedTools ?? DEFAULT_ALLOWED_TOOLS,
|
|
54
|
+
label,
|
|
55
|
+
cwd: opts.cwd,
|
|
56
|
+
});
|
|
57
|
+
|
|
58
|
+
let lastSessionId = "";
|
|
59
|
+
let lastStatus: AgentStatus | null = null;
|
|
60
|
+
let iterationsWithoutStatus = 0;
|
|
61
|
+
let blockerStreak: { key: string; count: number } | null = null;
|
|
62
|
+
|
|
63
|
+
for (let iteration = 0; iteration < maxIterations; iteration++) {
|
|
64
|
+
const prompt = opts.buildPrompt(lastStatus, iteration);
|
|
65
|
+
process.stderr.write(`${label} iteration ${iteration + 1}/${maxIterations} · ${turnsPerIteration} turns\n`);
|
|
66
|
+
|
|
67
|
+
let responseText = "";
|
|
68
|
+
for await (const msg of client.sendMessage({ prompt, sessionId: iteration > 0 ? lastSessionId : undefined })) {
|
|
69
|
+
opts.onAgentEvent?.(iteration, msg);
|
|
70
|
+
if (msg.type === "init") lastSessionId = msg.sessionId;
|
|
71
|
+
if (msg.type === "text") responseText += msg.text;
|
|
72
|
+
if (msg.type === "error") throw new Error(`Agent error: ${msg.text}`);
|
|
73
|
+
}
|
|
74
|
+
process.stdout.write("\n");
|
|
75
|
+
|
|
76
|
+
let status = parseAgentStatus(responseText);
|
|
77
|
+
process.stderr.write(`\n${label} iteration ${iteration + 1} status: exit_signal=${status?.exit_signal ?? "(no status)"} blockers=${JSON.stringify(status?.blockers ?? [])}\n`);
|
|
78
|
+
lastStatus = status ?? lastStatus;
|
|
79
|
+
opts.onIteration?.(iteration + 1, status);
|
|
80
|
+
|
|
81
|
+
if (status?.exit_signal && (status.blockers?.length ?? 0) === 0) {
|
|
82
|
+
let response: unknown = status;
|
|
83
|
+
if (opts.responseSchema) {
|
|
84
|
+
const raw = parseAgentResponse(responseText);
|
|
85
|
+
if (raw === null) {
|
|
86
|
+
process.stderr.write(`\n${label} NO <response> BLOCK — response tail: ${responseText.slice(-400)}\n`);
|
|
87
|
+
status = { ...status!, exit_signal: false, blockers: ["No <response> block found — emit a <response> block with the required JSON fields before setting exit_signal: true"] };
|
|
88
|
+
opts.onIteration?.(iteration + 1, status);
|
|
89
|
+
continue;
|
|
90
|
+
}
|
|
91
|
+
const parsed = opts.responseSchema.safeParse({ ...status, ...(raw as object) });
|
|
92
|
+
if (!parsed.success) {
|
|
93
|
+
process.stderr.write(`\n${label} <response> SCHEMA FAILED: ${parsed.error.message}\nraw: ${JSON.stringify(raw).slice(0, 400)}\n`);
|
|
94
|
+
status = { ...status!, exit_signal: false, blockers: [`<response> schema validation failed: ${parsed.error.message}`] };
|
|
95
|
+
opts.onIteration?.(iteration + 1, status);
|
|
96
|
+
continue;
|
|
97
|
+
}
|
|
98
|
+
response = parsed.data;
|
|
99
|
+
}
|
|
100
|
+
process.stderr.write(`${label} done after ${iteration + 1}/${maxIterations} iterations\n`);
|
|
101
|
+
return { sessionId: lastSessionId, lastStatus: status, iterations: iteration + 1, response };
|
|
102
|
+
}
|
|
103
|
+
|
|
104
|
+
if (!status) {
|
|
105
|
+
if (++iterationsWithoutStatus >= STALL_ITERATIONS)
|
|
106
|
+
throw new Error(`${label} stalled: no <status> block after ${iterationsWithoutStatus} iterations`);
|
|
107
|
+
} else {
|
|
108
|
+
iterationsWithoutStatus = 0;
|
|
109
|
+
}
|
|
110
|
+
|
|
111
|
+
if (status?.blockers?.length) {
|
|
112
|
+
const key = status.blockers.join("|");
|
|
113
|
+
if (blockerStreak !== null && blockerStreak.key === key) {
|
|
114
|
+
if (++blockerStreak.count >= SAME_BLOCKER_ITERATIONS)
|
|
115
|
+
throw new Error(`${label} circuit break: same blocker repeated ${blockerStreak.count}x — "${status.blockers[0]}"`);
|
|
116
|
+
} else {
|
|
117
|
+
blockerStreak = { key, count: 1 };
|
|
118
|
+
}
|
|
119
|
+
} else {
|
|
120
|
+
blockerStreak = null;
|
|
121
|
+
}
|
|
122
|
+
|
|
123
|
+
if (iteration + 1 < maxIterations)
|
|
124
|
+
process.stderr.write(`${label} continuing to iteration ${iteration + 2}/${maxIterations}\n`);
|
|
125
|
+
}
|
|
126
|
+
|
|
127
|
+
if (opts.responseSchema)
|
|
128
|
+
throw new Error(`${label} did not produce a valid <response> after ${maxIterations} iterations`);
|
|
129
|
+
process.stderr.write(`${label} exhausted ${maxIterations} iterations, proceeding with available work\n`);
|
|
130
|
+
return { sessionId: lastSessionId, lastStatus, iterations: maxIterations };
|
|
131
|
+
}
|
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
## Status Signal
|
|
2
|
+
|
|
3
|
+
When you have finished your work or are blocked, emit a `<status>` block at the end of your response:
|
|
4
|
+
|
|
5
|
+
```json
|
|
6
|
+
<status>
|
|
7
|
+
{
|
|
8
|
+
"summary": "one sentence describing what was done or what is blocking",
|
|
9
|
+
"completed": ["each acceptance criterion that is now fully met"],
|
|
10
|
+
"blockers": [],
|
|
11
|
+
"changed_files": ["relative/path/to/file"],
|
|
12
|
+
"tests_run": true,
|
|
13
|
+
"exit_signal": true
|
|
14
|
+
}
|
|
15
|
+
</status>
|
|
16
|
+
```
|
|
17
|
+
|
|
18
|
+
**Field semantics:**
|
|
19
|
+
- `summary`: one sentence — what was accomplished or what is blocking
|
|
20
|
+
- `completed`: acceptance criteria items that are fully done — be specific
|
|
21
|
+
- `blockers`: non-empty when `exit_signal: false` — describe the exact obstacle
|
|
22
|
+
- `changed_files`: relative paths of files you created or modified
|
|
23
|
+
- `tests_run`: `true` if you ran any test suite (pass or fail); `false` if no tests exist or you skipped them
|
|
24
|
+
- `exit_signal: true` — set when ALL acceptance criteria are met and no blockers remain
|
|
25
|
+
- `exit_signal: false` — set when blocked or unfinished; `blockers` must be non-empty
|
|
26
|
+
|
|
27
|
+
**If you are still actively working** and have not reached a natural stopping point, do NOT emit a `<status>` block — just keep working.
|
|
28
|
+
|
|
29
|
+
**Example (done):**
|
|
30
|
+
|
|
31
|
+
```json
|
|
32
|
+
<status>
|
|
33
|
+
{
|
|
34
|
+
"summary": "Added input validation middleware to /api/tasks with tests",
|
|
35
|
+
"completed": ["POST /api/tasks validates required fields", "Returns 400 with details on invalid input", "Unit tests passing"],
|
|
36
|
+
"blockers": [],
|
|
37
|
+
"changed_files": ["src/middleware/validate.ts", "src/routes/tasks.ts", "tests/validate.test.ts"],
|
|
38
|
+
"tests_run": true,
|
|
39
|
+
"exit_signal": true
|
|
40
|
+
}
|
|
41
|
+
</status>
|
|
42
|
+
```
|
|
43
|
+
|
|
44
|
+
**Example (blocked):**
|
|
45
|
+
|
|
46
|
+
```json
|
|
47
|
+
<status>
|
|
48
|
+
{
|
|
49
|
+
"summary": "Implemented middleware but tests are failing due to module resolution",
|
|
50
|
+
"completed": ["Middleware created and wired into route"],
|
|
51
|
+
"blockers": ["Tests fail: cannot resolve import './validate' — module resolution config unclear"],
|
|
52
|
+
"changed_files": ["src/middleware/validate.ts"],
|
|
53
|
+
"tests_run": true,
|
|
54
|
+
"exit_signal": false
|
|
55
|
+
}
|
|
56
|
+
</status>
|
|
57
|
+
```
|
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
import { z } from "zod";
|
|
2
|
+
import { AgentStatusSchema } from "../utils/schemas.js";
|
|
3
|
+
import type { AgentMessage } from "../types/protocol.js";
|
|
4
|
+
|
|
5
|
+
export type {
|
|
6
|
+
AgentMessage, AgentMessageInit, AgentMessageText, AgentMessageThinking,
|
|
7
|
+
AgentMessageToolUse, AgentMessageToolResult, AgentMessageDone,
|
|
8
|
+
AgentMessageError, AgentMessageUsage, AgentStatus,
|
|
9
|
+
} from "../types/protocol.js";
|
|
10
|
+
|
|
11
|
+
export { AgentStatusSchema };
|
|
12
|
+
|
|
13
|
+
export const AgentMessageSchema = z.object({
|
|
14
|
+
type: z.enum(["init", "text", "thinking", "tool_use", "tool_result", "done", "error", "usage"]),
|
|
15
|
+
timestamp: z.string(),
|
|
16
|
+
}).passthrough() as unknown as z.ZodType<AgentMessage>;
|
|
17
|
+
|
|
18
|
+
export function parseAgentResponse(text: string): unknown {
|
|
19
|
+
const match = text.match(/<response>([\s\S]*?)<\/response>/);
|
|
20
|
+
if (!match) return null;
|
|
21
|
+
try { return JSON.parse(match[1].trim()); } catch { return null; }
|
|
22
|
+
}
|