@agent-compose/sdk 0.6.0 → 0.8.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +66 -39
- package/dist/agent/__tests__/runtime-json-schema.test.d.ts +10 -0
- package/dist/agent/agent-context.d.ts +21 -1
- package/dist/agent/agent-loop.d.ts +24 -1
- package/dist/client.d.ts +338 -534
- package/dist/directives.d.ts +112 -0
- package/dist/display.d.ts +242 -0
- package/dist/errors.d.ts +24 -1
- package/dist/index.d.ts +34 -13
- package/dist/index.js +2984 -861
- package/dist/pause/wrappers.d.ts +31 -9
- package/dist/processors/ask-human.d.ts +30 -0
- package/dist/processors/ask-human.test.d.ts +1 -0
- package/dist/processors/index.d.ts +1 -0
- package/dist/runtimes/_acp-client.d.ts +46 -1
- package/dist/runtimes/_cli-agent.d.ts +58 -4
- package/dist/runtimes/_jsonl-guard.d.ts +103 -0
- package/dist/runtimes/amp.d.ts +2 -2
- package/dist/runtimes/claude-code.d.ts +59 -0
- package/dist/runtimes/claude-code.test.d.ts +14 -0
- package/dist/runtimes/claude.d.ts +16 -0
- package/dist/runtimes/claude.test.d.ts +8 -0
- package/dist/runtimes/codex.d.ts +9 -3
- package/dist/runtimes/cursor.d.ts +9 -0
- package/dist/runtimes/droid.d.ts +9 -0
- package/dist/runtimes/jsonl-guard.test.d.ts +19 -0
- package/dist/runtimes/openai-desktop.js +2922 -861
- package/dist/runtimes/opencode.d.ts +25 -0
- package/dist/runtimes/vercel.js +22 -1
- package/dist/sandbox/devbox.d.ts +42 -0
- package/dist/sandbox/exec-stream.d.ts +14 -0
- package/dist/sandbox/network-policy.d.ts +100 -0
- package/dist/sandbox/provider-def.d.ts +79 -0
- package/dist/sandbox/providers/desktop.d.ts +10 -0
- package/dist/sandbox/providers/e2b.d.ts +17 -0
- package/dist/sandbox/providers/local.d.ts +11 -0
- package/dist/sandbox/providers/vercel.d.ts +18 -0
- package/dist/sandbox/registry.d.ts +45 -0
- package/dist/sandbox/sizes.d.ts +68 -0
- package/dist/sandbox.d.ts +24 -299
- package/dist/step-invocation/__tests__/foreground-recovery.test.d.ts +1 -0
- package/dist/step-invocation/invoker.d.ts +24 -1
- package/dist/step-invocation/protocol.d.ts +13 -0
- package/dist/types/api-compliance.d.ts +71 -0
- package/dist/types/api-conversations.d.ts +492 -0
- package/dist/types/api-factory.d.ts +309 -0
- package/dist/types/api-projects.d.ts +131 -0
- package/dist/types/api-runs.d.ts +377 -0
- package/dist/types/api-scopes.d.ts +102 -0
- package/dist/types/conversation-stream.d.ts +191 -0
- package/dist/types/execution-context.d.ts +12 -2
- package/dist/types/protocol.d.ts +30 -1
- package/dist/types/sandbox-environment.d.ts +8 -5
- package/dist/types/sandbox.d.ts +79 -0
- package/dist/types/workflow-metadata.d.ts +33 -8
- package/dist/types/workflow-plan.d.ts +10 -0
- package/dist/types/workflow.d.ts +18 -193
- package/dist/utils/bundler.d.ts +12 -1
- package/dist/utils/errors.d.ts +9 -1
- package/dist/workflow-steps/index.d.ts +1 -1
- package/dist/workflow-steps/observability.d.ts +8 -1
- package/dist/workflow-steps/runner.d.ts +3 -3
- package/dist/workflow-steps/step.d.ts +15 -1
- package/dist/workflow-steps/types.d.ts +19 -5
- package/dist/workflow-steps/workflow.d.ts +22 -1
- package/dist/workflows/engine.d.ts +3 -2
- package/dist/workflows/invoke-child.d.ts +2 -2
- package/package.json +1 -1
- package/src/agent/agent-context.ts +206 -16
- package/src/agent/agent-loop.ts +40 -4
- package/src/agent/run-agent.ts +9 -1
- package/src/client.ts +909 -621
- package/src/directives.ts +184 -0
- package/src/display.ts +788 -0
- package/src/errors.ts +39 -0
- package/src/index.ts +117 -10
- package/src/pause/wrappers.ts +44 -9
- package/src/processors/ask-human.ts +136 -0
- package/src/processors/index.ts +5 -0
- package/src/runtimes/_acp-client.ts +72 -3
- package/src/runtimes/_cli-agent.ts +171 -38
- package/src/runtimes/_jsonl-guard.ts +219 -0
- package/src/runtimes/claude-code.ts +246 -0
- package/src/runtimes/claude.ts +32 -2
- package/src/runtimes/codex.ts +55 -3
- package/src/runtimes/cursor.ts +59 -0
- package/src/runtimes/droid.ts +63 -0
- package/src/runtimes/openai-desktop.ts +59 -14
- package/src/runtimes/opencode.ts +61 -0
- package/src/sandbox/devbox.ts +48 -0
- package/src/sandbox/exec-stream.ts +48 -0
- package/src/sandbox/network-policy.ts +181 -0
- package/src/sandbox/provider-def.ts +94 -0
- package/src/sandbox/providers/desktop.ts +57 -0
- package/src/sandbox/providers/e2b.ts +354 -0
- package/src/sandbox/providers/local.ts +106 -0
- package/src/sandbox/providers/vercel.ts +331 -0
- package/src/sandbox/registry.ts +198 -0
- package/src/sandbox/sizes.ts +95 -0
- package/src/sandbox.ts +59 -1263
- package/src/step-invocation/invoker.ts +319 -34
- package/src/step-invocation/protocol.ts +19 -0
- package/src/types/api-compliance.ts +79 -0
- package/src/types/api-conversations.ts +522 -0
- package/src/types/api-factory.ts +336 -0
- package/src/types/api-projects.ts +140 -0
- package/src/types/api-runs.ts +412 -0
- package/src/types/api-scopes.ts +102 -0
- package/src/types/conversation-stream.ts +231 -0
- package/src/types/execution-context.ts +10 -2
- package/src/types/protocol.ts +33 -0
- package/src/types/sandbox-environment.ts +28 -9
- package/src/types/sandbox.ts +78 -0
- package/src/types/workflow-metadata.ts +35 -8
- package/src/types/workflow-plan.ts +11 -0
- package/src/types/workflow.ts +25 -280
- package/src/utils/bundler.ts +32 -5
- package/src/utils/errors.ts +34 -2
- package/src/workflow-steps/index.ts +1 -0
- package/src/workflow-steps/observability.ts +19 -8
- package/src/workflow-steps/runner.ts +4 -4
- package/src/workflow-steps/step.ts +49 -1
- package/src/workflow-steps/types.ts +20 -5
- package/src/workflow-steps/workflow.ts +22 -1
- package/src/workflows/engine.ts +3 -2
- package/src/workflows/invoke-child.ts +2 -2
package/README.md
CHANGED
|
@@ -23,34 +23,44 @@ npm install zod
|
|
|
23
23
|
|
|
24
24
|
## Authoring a workflow
|
|
25
25
|
|
|
26
|
-
A workflow is
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
26
|
+
A workflow is a typed chain of discrete steps:
|
|
27
|
+
`defineWorkflow({ id, input, output }).step(defineStep(...)).build()`.
|
|
28
|
+
Every step is a durable replay checkpoint — the engine records each
|
|
29
|
+
step's validated output, so a retry or resume picks up after the last
|
|
30
|
+
completed step instead of re-running the whole workflow. Pausing for
|
|
31
|
+
human input (`ctx.pause`, `ctx.waitForEvent`) only works in this
|
|
32
|
+
step-form.
|
|
33
33
|
|
|
34
34
|
```typescript
|
|
35
35
|
// my-workflow.ts
|
|
36
|
-
import { defineWorkflow, agent, claudeRuntime } from "@agent-compose/sdk";
|
|
36
|
+
import { defineWorkflow, defineStep, agent, claudeRuntime } from "@agent-compose/sdk";
|
|
37
|
+
import { z } from "zod";
|
|
37
38
|
import PROMPT from "./prompt.md" with { type: "text" };
|
|
38
39
|
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
const repo = (ctx.input?.repo as string | undefined) ?? "owner/repo";
|
|
40
|
+
const Input = z.object({ repo: z.string() });
|
|
41
|
+
const Output = z.object({ ok: z.boolean(), summary: z.string().optional() });
|
|
42
42
|
|
|
43
|
+
const review = defineStep({
|
|
44
|
+
name: "review",
|
|
45
|
+
input: Input,
|
|
46
|
+
output: Output,
|
|
47
|
+
run: async (ctx) => {
|
|
43
48
|
const result = await agent({
|
|
44
|
-
sandbox,
|
|
45
|
-
runtime:
|
|
46
|
-
prompt:
|
|
47
|
-
tools:
|
|
48
|
-
budget:
|
|
49
|
+
sandbox: ctx.sandbox, // the run's own VM
|
|
50
|
+
runtime: claudeRuntime,
|
|
51
|
+
prompt: `${PROMPT}\n\nRepository: ${ctx.input.repo}`,
|
|
52
|
+
tools: ["Bash", "Read", "Edit", "Write", "Grep", "Glob"],
|
|
53
|
+
budget: { turnsPerIteration: 40, maxIterations: 8 },
|
|
49
54
|
});
|
|
50
|
-
|
|
51
55
|
await ctx.setMetadata({ summary: result.status?.summary });
|
|
52
|
-
return { ok: result.status?.completed ?? false };
|
|
56
|
+
return { ok: result.status?.completed ?? false, summary: result.status?.summary };
|
|
53
57
|
},
|
|
58
|
+
});
|
|
59
|
+
|
|
60
|
+
export default defineWorkflow({
|
|
61
|
+
id: "my-workflow",
|
|
62
|
+
input: Input,
|
|
63
|
+
output: Output,
|
|
54
64
|
// Optional: outbound network rules that the runner sandbox will enforce
|
|
55
65
|
// (Vercel only — E2B ignores). Use `$VAR` placeholders for secrets that
|
|
56
66
|
// get resolved from the per-workflow secret store at dispatch time.
|
|
@@ -60,36 +70,52 @@ export default defineWorkflow({
|
|
|
60
70
|
"api.anthropic.com": [{ transform: [{ headers: { "x-api-key": "$ANTHROPIC_API_KEY" } }] }],
|
|
61
71
|
},
|
|
62
72
|
},
|
|
63
|
-
})
|
|
73
|
+
})
|
|
74
|
+
.step(review)
|
|
75
|
+
.build();
|
|
64
76
|
```
|
|
65
77
|
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
|
|
78
|
+
The builder chain enforces at compile time that each step's `input`
|
|
79
|
+
schema matches the previous step's `output`, and `.build()` verifies
|
|
80
|
+
the final step's `output` is the same Zod instance as the workflow's
|
|
81
|
+
declared `output`. `networkPolicy` / `placeholders` / `snapshots` on
|
|
82
|
+
the definition become metadata the bundler picks up at registration
|
|
83
|
+
time.
|
|
84
|
+
|
|
85
|
+
> **Legacy run-form.** `defineWorkflow({ run(ctx, sandbox) { … } })` is
|
|
86
|
+
> `@deprecated` — it compiles to a single opaque step, so there is no
|
|
87
|
+
> per-step replay and pause doesn't work. Don't author new ones.
|
|
88
|
+
|
|
89
|
+
### What a step can do with `ctx`
|
|
71
90
|
|
|
72
|
-
|
|
91
|
+
Each step's `run(ctx)` receives a `StepContext`:
|
|
73
92
|
|
|
74
93
|
```ts
|
|
75
|
-
interface
|
|
94
|
+
interface StepContext<TInput> {
|
|
95
|
+
input: TInput; // validated against the step's `input` schema
|
|
76
96
|
run: { id: string };
|
|
77
|
-
|
|
97
|
+
sandbox?: SandboxProvider; // the run's VM — pass to `agent({ sandbox: ctx.sandbox })`
|
|
78
98
|
setMetadata: (data: Record<string, unknown>) => Promise<void>;
|
|
79
|
-
step<T>(name: string, fn: () => Promise<T>): Promise<T>;
|
|
99
|
+
step<T>(name: string, fn: () => Promise<T>): Promise<T>; // durable named sub-step
|
|
100
|
+
pause<T>(req: PauseRequest<T>): Promise<T>; // wait for a human, durably
|
|
101
|
+
sleep(durationMs: number): Promise<void>;
|
|
102
|
+
waitForEvent<T>(req: WaitForEventRequest<T>): Promise<T>;
|
|
103
|
+
invokeChild(name, input?, opts?): Promise<RunStatus>; // run another workflow
|
|
80
104
|
}
|
|
81
105
|
```
|
|
82
106
|
|
|
83
|
-
`step("phase-name", () => …)` wraps a phase
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
107
|
+
`ctx.step("phase-name", () => …)` wraps a phase *within* a step for the
|
|
108
|
+
run timeline — it memoises the result (a pause-resume re-entry restores
|
|
109
|
+
it instead of re-running) and emits `workflow_substep_started` /
|
|
110
|
+
`workflow_substep_completed` / `workflow_substep_failed` lifecycle
|
|
111
|
+
events with duration. Use it for setup, external API calls, or anything
|
|
112
|
+
you want visible on the dashboard's run detail page.
|
|
87
113
|
|
|
88
114
|
### The `agent` loop
|
|
89
115
|
|
|
90
116
|
```ts
|
|
91
117
|
agent({
|
|
92
|
-
sandbox, // the
|
|
118
|
+
sandbox, // ctx.sandbox — the run's VM
|
|
93
119
|
runtime, // claudeRuntime, or your own via createClaudeRuntime / defineRuntime
|
|
94
120
|
prompt, // raw markdown — `--- frontmatter ---` is auto-stripped
|
|
95
121
|
tools?, // model tool allowlist (defaults inside agentLoop)
|
|
@@ -211,8 +237,8 @@ console.log(status.status); // "success" | "failed" | "abandoned" | "canceled"
|
|
|
211
237
|
console.log(status.output); // workflow's return value
|
|
212
238
|
```
|
|
213
239
|
|
|
214
|
-
`output` is the
|
|
215
|
-
|
|
240
|
+
`output` is the final step's return value, validated against the
|
|
241
|
+
workflow's declared `output` schema. `setMetadata()` writes to a
|
|
216
242
|
separate `metadata` field — useful for "side-channel" facts (PR url, plan
|
|
217
243
|
url) without polluting the structured return.
|
|
218
244
|
|
|
@@ -364,10 +390,10 @@ await client.invoke("my-workflow", input, {
|
|
|
364
390
|
});
|
|
365
391
|
|
|
366
392
|
// Set a default at registration time:
|
|
367
|
-
defineWorkflow({
|
|
393
|
+
defineWorkflow({ id, input, output, snapshots: { saveLatest: true } }).step(…).build();
|
|
368
394
|
|
|
369
395
|
// Retain every step's snapshot (not just the latest):
|
|
370
|
-
defineWorkflow({
|
|
396
|
+
defineWorkflow({ id, input, output, snapshots: { saveLatest: true, retainSteps: true } }).step(…).build();
|
|
371
397
|
|
|
372
398
|
// Browse / clean up:
|
|
373
399
|
const page = await client.listSnapshotsPage({ workflow: "my-workflow", limit: 50 });
|
|
@@ -408,7 +434,8 @@ programmatic / server-to-server callers.
|
|
|
408
434
|
|
|
409
435
|
| Export | What |
|
|
410
436
|
|---|---|
|
|
411
|
-
| `defineWorkflow` |
|
|
437
|
+
| `defineWorkflow` | Typed step-workflow builder — `defineWorkflow({ id, input, output }).step(...).build()` |
|
|
438
|
+
| `defineStep` | Declare one typed workflow step (`{ name, input, output, run }`) |
|
|
412
439
|
| `defineRuntime` | Wrap an agent execution provider as an `AgentRuntime` |
|
|
413
440
|
| `defineSandboxEnvironment` | Sugar for declaring a workflow whose primary purpose is to build a snapshot for others to boot from |
|
|
414
441
|
| `agent` / `agentLoop` | Embed an LLM loop inside a workflow |
|
|
@@ -421,7 +448,7 @@ programmatic / server-to-server callers.
|
|
|
421
448
|
| `parseSseStream` | Generic SSE chunk decoder (used by `streamRunLogs`) |
|
|
422
449
|
| `createSandbox` / `reconnectSandbox` / `killAllSandboxes` / `killSandboxById` / `getSandboxQuotas` / `listOwnedSandboxes` / `deleteSandboxSnapshot` | Sandbox-provider helpers (Vercel + E2B) |
|
|
423
450
|
|
|
424
|
-
Type exports: `
|
|
451
|
+
Type exports: `Workflow`, `Step`, `StepContext`, `WorkflowBuilder`,
|
|
425
452
|
`WorkflowHooks`, `AgentBudget`, `AgentRuntime`, `RuntimeOptions`,
|
|
426
453
|
`ModelExecutionContract`, `McpServerConfig`, `AgentMessage` (and its
|
|
427
454
|
variants), `AgentStatus`, `RunStatus`, `RegisterResult`, `RunEvent`,
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* responseSchema → runtime outputFormat must NOT carry zod v4's
|
|
3
|
+
* `$schema: …/draft/2020-12/schema` header. Claude Code validates
|
|
4
|
+
* `--json-schema` with a validator that has no 2020-12 meta-schema
|
|
5
|
+
* registered, so a schema carrying the header kills the CLI at startup
|
|
6
|
+
* ("Claude Code process exited with code 1" before any API call) — the
|
|
7
|
+
* incident that took down every dynamic-task run after the 2026-07-17
|
|
8
|
+
* agent-env rebake baked claude 2.1.212.
|
|
9
|
+
*/
|
|
10
|
+
export {};
|
|
@@ -19,7 +19,27 @@ import type { SandboxProvider } from "../types/sandbox.js";
|
|
|
19
19
|
* that credentials are network-injected (never in the env). The live
|
|
20
20
|
* "Connectors & access" section is appended per-run by `buildAgentContextDoc`.
|
|
21
21
|
*/
|
|
22
|
-
export declare const AGENT_COMPOSE_MANUAL = "# Working inside an Agent Compose sandbox\n\nYou are an agent running in a per-run sandbox on the Agent Compose platform.\nUse the **`agentc` CLI** and the **`@agent-compose/sdk`** for everything below \u2014\ndo NOT hand-roll raw HTTP/curl calls against the platform API. The CLI is on\nyour PATH and already authenticated from the environment\n(`AGENT_COMPOSE_URL` / `AGENT_COMPOSE_API_KEY` / `AGENT_COMPOSE_FACTORY` are\ninjected for this run), so commands just work \u2014 no login, no keys to manage.\n\nThe `/ac:*` skills are installed as Claude Code slash commands (`/ac:invoke`,\n`/ac:events`, `/ac:logs`, `/ac:register`, \u2026) \u2014 reach for them too.\n\n## Files \u2014 your outputs persist by default\n\nYour working directory defaults to **`$AGENT_COMPOSE_RUN_DIR`** \u2014 a per-run\ndirectory on the shared factory drive\n(`$AGENT_COMPOSE_FACTORY_DIR/<workflow>/<version>/<run-id>/`) the platform\ncreates and attributes to this run. **Files you write here persist by\ndefault** \u2014 they show up in the dashboard's Files tab and the run's Artifacts\ncard, with no API calls to save them. The dir already exists and is writable.\n\nNeed throwaway scratch \u2014 heavy build output, package caches, temp files?\n`cd /tmp` (or any path outside `/factory`): anything off the factory drive is\nephemeral and discarded when the sandbox ends. In short: **stay in your working\ndir to keep something, `cd` out to throw it away.**\n\nThe whole shared drive is POSIX-mounted at `/factory`; the dashboard-visible\nroot is `$AGENT_COMPOSE_FACTORY_DIR` (`/factory/files`). Earlier versions and\nruns live in sibling dirs under\n`$AGENT_COMPOSE_FACTORY_DIR/$AGENT_COMPOSE_WORKFLOW/` \u2014 read them for prior\ncontext. Other workflows' dirs are present but not your concern.\n\n## Events \u2014 the factory timeline\n\nRecord something on the run/factory timeline (the dashboard renders these)\nwith the CLI \u2014 your run id is `$RUN_ID`:\n\n agentc events send \"$RUN_ID\" <name> --summary \"<one line>\" [--body '<json>']\n\nNames like `note.created` / `brief.posted` surface in the Workbench;\n`agentc events list` reads them back. `/ac:events` is the skill equivalent.\n\n## Runs\n\n agentc list # registered workflows (/ac:list)\n agentc logs \"$RUN_ID\" # a run's logs (/ac:logs)\n agentc invoke <workflow> -i '<json>' # dispatch a workflow (/ac:invoke)\n\n## Writing workflow / agent code \u2014 the SDK\n\n`@agent-compose/sdk` is installed in `/workspace` \u2014 import it from any script\nyou write there:\n\n import { defineWorkflow, agent, AgentComposeClient } from \"@agent-compose/sdk\";\n\nUse `/ac:generate-workflow` / `/ac:generate-agent` to scaffold, then\n`agentc register <file.ts>` (or `/ac:register`).\n\n## Pausing to ask the human \u2014 `agentc pause`\n\nWhen you can't or shouldn't proceed without a human, run `agentc pause`, then\n**END YOUR TURN**:\n\n agentc pause --reason \"Notion returned 401 \u2014 connect Notion to continue\" \\\n --option retry --option skip\n\n`agentc pause` does NOT block and does NOT print the answer. It records your\nquestion and returns immediately. The moment you end your turn, the run pauses\n(your sandbox is snapshotted and compute stops while the human decides) and the\nhuman's answer is delivered to you as your **next message** \u2014 you pick up\nexactly where you left off, with the answer in hand. So: ask, end your turn,\nand wait. Do NOT keep working, do NOT call more tools, and do NOT mark the task\ncomplete after pausing.\n\nReach for it the moment you hit \u2014 or foresee \u2014 any of these:\n- **A wall only a human can clear:** a 401/403, a missing credential, an\n unconnected provider, a host the network refuses. Do NOT retry blindly or try\n to work around it \u2014 pause and say what needs enabling.\n- **A durable or outward-facing action that needs sign-off:** registering a\n workflow, deploying, sending email/messages, deleting or overwriting shared\n data, spending money. Prepare everything, then pause for approval BEFORE you\n commit it.\n- **A judgment call only the human can settle:** an under-specified request,\n several valid paths, a conflict with existing state, missing input only they have.\n\nYou compose the `--reason` (the ask) yourself; pass `--option` choices when\nthere are clear ones, omit them for a free-form answer. Each agent pauses\nindependently \u2014 pausing doesn't stop the others.\n\n## Credentials\n\nConnector credentials (Google, GitHub, \u2026) are NEVER in your environment.\nThey're injected at the network layer when you call an allowed host \u2014 make the\nrequest **without** an Authorization header and the platform adds it. Don't try\nto read or exfiltrate tokens; they aren't here. The \"Connectors & access\"\nsection below (when present) lists exactly which providers this run can reach.\n\n## Tools in this environment\n\n- `agentc` \u2014 Agent Compose CLI (your primary interface; authed from env)\n- `@agent-compose/sdk` \u2014 installed in /workspace for writing workflows\n- `/ac:*` Claude Code skills \u2014 slash commands for the above\n- `archil` (factory drive), `rtk`, `bun`\n- A world-writable `/workspace` working directory";
|
|
22
|
+
export declare const AGENT_COMPOSE_MANUAL = "# Working inside an Agent Compose sandbox\n\nYou are an agent running in a per-run sandbox on the Agent Compose platform.\nUse the **`agentc` CLI** and the **`@agent-compose/sdk`** for everything below \u2014\ndo NOT hand-roll raw HTTP/curl calls against the platform API. The CLI is on\nyour PATH and already authenticated from the environment\n(`AGENT_COMPOSE_URL` / `AGENT_COMPOSE_API_KEY` / `AGENT_COMPOSE_FACTORY` are\ninjected for this run), so commands just work \u2014 no login, no keys to manage.\n\nThe `/ac:*` skills are installed as Claude Code slash commands (`/ac:invoke`,\n`/ac:events`, `/ac:logs`, `/ac:register`, \u2026) \u2014 reach for them too.\n\n## Files \u2014 your outputs persist by default\n\nYour working directory defaults to **`$AGENT_COMPOSE_RUN_DIR`** \u2014 a per-run\ndirectory on the shared factory drive\n(`$AGENT_COMPOSE_FACTORY_DIR/<workflow>/<version>/<run-id>/`) the platform\ncreates and attributes to this run. **Files you write here persist by\ndefault** \u2014 they show up in the dashboard's Files tab and the run's Artifacts\ncard, with no API calls to save them. The dir already exists and is writable.\n\nNeed throwaway scratch \u2014 heavy build output, package caches, temp files?\n`cd /tmp` (or any path outside `/factory`): anything off the factory drive is\nephemeral and discarded when the sandbox ends. In short: **stay in your working\ndir to keep something, `cd` out to throw it away.**\n\nThe whole shared drive is POSIX-mounted at `/factory`; the dashboard-visible\nroot is `$AGENT_COMPOSE_FACTORY_DIR` (`/factory/files`). Earlier versions and\nruns live in sibling dirs under\n`$AGENT_COMPOSE_FACTORY_DIR/$AGENT_COMPOSE_WORKFLOW/` \u2014 read them for prior\ncontext. Other workflows' dirs are present but not your concern.\n\n## Events \u2014 the factory timeline\n\nRecord something on the run/factory timeline (the dashboard renders these)\nwith the CLI \u2014 your run id is `$RUN_ID`:\n\n agentc events send \"$RUN_ID\" <name> --summary \"<one line>\" [--body '<json>']\n\nNames like `note.created` / `brief.posted` surface in the Workbench;\n`agentc events list` reads them back. `/ac:events` is the skill equivalent.\n\n## Runs\n\n agentc list # registered workflows (/ac:list)\n agentc logs \"$RUN_ID\" # a run's logs (/ac:logs)\n agentc invoke <workflow> -i '<json>' # dispatch a workflow (/ac:invoke)\n\n## Writing workflow / agent code \u2014 the SDK\n\n`@agent-compose/sdk` is installed in `/workspace`. **To author a workflow,\nALWAYS run `/ac:generate-workflow`** (and `/ac:generate-agent` for an agent\nstep) instead of writing source from memory \u2014 the skill scaffolds the correct,\ncurrent shape. Then `agentc register <file.ts>` (or `/ac:register`).\n\nThe skill writes **step-form** (a builder of discrete, durable `.step()`s).\nThe legacy run-form (`defineWorkflow({ run(ctx, sandbox) { \u2026 } })`) has been\nREMOVED from the SDK \u2014 registering one fails with an error. Step-form is the\nonly shape: durable per-step replay, and pause only works there.\n\n## Pausing to ask the human\n\nTo ask a human and get an answer back, use the **`AskUserQuestion`** tool if\nyou have it; otherwise run **`agentc pause`**:\n\n agentc pause --reason \"Notion returned 401 \u2014 connect Notion to continue\" \\\n --option retry --option skip\n\n**Both BLOCK and hand you the answer inline.** While you wait, the run is\nsuspended \u2014 your sandbox is frozen and compute stops, so a pause is free while\nthe human decides. When they answer, the call RETURNS with their decision: the\n`AskUserQuestion` tool result, or `agentc pause`'s output\n(`\u25B6 Resumed. The human answered: \u2026`), carries it.\n\n**Then USE that answer to finish your work \u2014 do NOT end your turn.** This is NOT\nfire-and-forget, and the answer does NOT arrive in a later message: it comes\nback right where you called it, on the SAME turn. The shape is: ask \u2192 the call\nblocks \u2192 it returns the human's answer \u2192 you act on it and produce your result.\nNever end your turn before the call returns, never guess an answer, and never\nproceed without one.\n\nReach for it the moment you hit \u2014 or foresee \u2014 any of these:\n- **A wall only a human can clear:** a 401/403, a missing credential, an\n unconnected provider, a host the network refuses. Do NOT retry blindly or try\n to work around it \u2014 pause and say what needs enabling.\n- **A durable or outward-facing action that needs sign-off:** registering a\n workflow, deploying, sending email/messages, deleting or overwriting shared\n data, spending money. Prepare everything, then pause for approval BEFORE you\n commit it.\n- **A judgment call only the human can settle:** an under-specified request,\n several valid paths, a conflict with existing state, missing input only they have.\n\nYou compose the `--reason` (the ask) yourself; pass `--option` choices when\nthere are clear ones, omit them for a free-form answer. Each agent pauses\nindependently \u2014 pausing doesn't stop the others.\n\n## Credentials\n\nConnector credentials (Google, GitHub, \u2026) are NEVER in your environment.\nThey're injected at the network layer when you call an allowed host \u2014 make the\nrequest **without** an Authorization header and the platform adds it. Don't try\nto read or exfiltrate tokens; they aren't here. The \"Connectors & access\"\nsection below (when present) lists exactly which providers this run can reach.\n\n## Computer Use \u2014 you have a real desktop, and it is already running\n\n**This machine has a graphical desktop.** Every session machine does \u2014 terminal\nsessions included \u2014 and the platform brings it UP AT BOOT, before your first\nturn: an X server on `DISPLAY=:0`, the openbox window manager, wallpaper and a\npanel. You do not start it, you do not wait for a human to open it, and you do\nnot need a viewer. Go straight to driving it.\n\n(The one exception, and it is rare: an image built without the GUI stack has no\ndisplay at all, and `DISPLAY=:0 xdotool getdisplaygeometry` errors outright.\nThat single case is the only one where this section does not apply \u2014 a\nscreenshot showing only wallpaper is NOT it, and neither is an app that failed\nto start.)\n\n**This is how you SEE anything.** Any question of the form \"does it render?\",\n\"is the page actually working?\", \"did the markers show up?\", \"what does it look\nlike?\" is answered by opening it on this desktop and screenshotting it \u2014 not by\nreasoning about the code, and not by a headless render (which proves the process\nstarts, not that the thing draws). Verify visually before you report visually.\n\n- **Input** \u2014 `xdotool` against `DISPLAY=:0`: `DISPLAY=:0 xdotool mousemove <x> <y>`,\n `DISPLAY=:0 xdotool click 1` (1=left, 3=right), `DISPLAY=:0 xdotool type 'text'`,\n `DISPLAY=:0 xdotool key Return` (also `ctrl+c`, `Tab`, `super`, \u2026).\n- **Screenshots** \u2014 `scrot` (or ImageMagick's `import`):\n `DISPLAY=:0 scrot /tmp/screen.png`, then READ the PNG to see the screen,\n before and after you act. A screenshot is your only eyes here.\n- **Apps + windows** \u2014 a plain X session. Launch in the background:\n `DISPLAY=:0 <app> &`. Two things that trip agents up, both normal:\n - a GUI app needs a **beat to map its window** \u2014 screenshot, and if you see\n only wallpaper, wait a couple of seconds and screenshot again before\n concluding anything;\n - **Chromium needs `--no-sandbox`** in this environment (nested sandbox).\n The whole recipe for looking at a local page:\n `DISPLAY=:0 chromium --no-sandbox --disable-gpu --start-maximized <url> &`\n then `sleep 5`, then `DISPLAY=:0 scrot /tmp/screen.png` and read it.\n If a window still never appears, read the app's own log (`/tmp/*.log`) \u2014 the\n desktop is not the thing that failed. Do NOT abandon it for a headless\n screenshot: headless cannot tell you what the human will see.\n- **A human can watch** \u2014 the session header carries a **Desktop** button in the\n dashboard, and what a teammate sees there is exactly this display. The desktop\n runs whether or not anyone is looking; never wait for a viewer.\n\nNothing here changes the credentials rule above: tokens are injected at the\nnetwork layer, never present on the desktop or in any file you can read \u2014 so\nthere is nothing to type, paste, or screenshot a credential from.\n\n## Recording a demo \u2014 the desktop, captured to a video the human can play\n\n\"Record a demo of you using X\" is a normal ask, and this machine does it:\nstart a screen recording, drive the app with `xdotool` exactly as in Computer\nUse, stop the recording, and report the file. (For a LIVE view no recording is\nneeded \u2014 the session header's **Desktop** button already streams this display\nto any teammate watching; a recording is the durable, replayable artifact.\nBoth modes exist; say so when it matters.)\n\n**ffmpeg is NOT pre-installed** \u2014 install it first, once per machine:\n\n sudo apt-get update -q && sudo apt-get install -y -q ffmpeg\n\n(drop `sudo` if you are already root). Then the whole recipe:\n\n DISPLAY=:0 ffmpeg -f x11grab \\\n -video_size \"$(DISPLAY=:0 xdotool getdisplaygeometry | tr ' ' x)\" \\\n -framerate 10 -i :0 -c:v libvpx -b:v 1M -deadline realtime -cpu-used 8 \\\n demo.webm &\n FFMPEG_PID=$!\n # ... drive the app with xdotool, screenshotting as you go ...\n kill -INT \"$FFMPEG_PID\" && wait \"$FFMPEG_PID\"\n\nThe gotchas, each one earned:\n- **Stop with SIGINT (`kill -INT`), never SIGKILL** \u2014 ffmpeg finalizes the\n file on SIGINT; a hard kill truncates the encode mid-write.\n- **Record WebM (matroska-family), not MP4** \u2014 mp4 writes its moov atom at the\n END, so a killed or crashed encode leaves an UNPLAYABLE file; webm stays\n playable up to the last written frame and plays natively in the browser.\n MP4's only edge is compatibility with some external players \u2014 transcode\n afterwards if you truly need it, never record straight to it.\n- **`-video_size` must match the real screen** \u2014 x11grab does not default to\n it; read the geometry from `xdotool getdisplaygeometry` as above.\n- **10\u201312 fps is right for a screen demo** \u2014 small files, legible UI motion;\n this is not video production.\n- **Write to the drive, not /tmp** \u2014 the recording must land in your working\n directory to persist and show up in Files; a file in /tmp dies with the\n sandbox.\n- When you stop, **TELL the human the exact drive path** of the video \u2014 a\n recording they cannot find might as well not exist.\n\n## Previews \u2014 register every server you serve (cloud sessions)\n\nIn a cloud session, a dev server listening on a port becomes a hosted,\nmember-gated URL the human can open \u2014 but ONLY if you register it:\n\n agentc preview open <port> [--name <label>] [--path </landing>]\n # hosted URL + an \"Open preview\" card\n agentc preview list # the registry \u2014 what is live right now\n agentc preview close <port> # take one down\n\n(`agentc preview announce` is the same verb as `open` \u2014 announce what you\nserve.) `--name` is the human-readable label; `--path` is where the app\nshould open (e.g. `/dashboard`) \u2014 the card and every chip land the human\nthere instead of a bare `/`.\n\nRegister EVERY server you start for a human, the moment it is listening, and\ntell them the URL the command printed. The registry is the only discoverable\nrecord of what this machine serves: an unregistered server keeps running, but\nnobody \u2014 not the human, not the assistant \u2014 can find its URL, and when the\nsandbox recycles it is gone without a trace. Never guess or hand out a raw\nport; the hosted URL from `agentc preview open` is the only address that\nworks outside this machine. (Outside a cloud session the command errors\nhonestly \u2014 there is no session sandbox to expose.)\n\nWhat registration buys you: the human sees each registered preview as a card\nin the conversation and a row in the session's Previews menu \u2014 MANY at once,\none per port \u2014 and the assistant resolves \"open the preview\" from this same\nregistry (its `list_previews` read), so what you register is exactly what\ngets opened. On deployments with subdomain previews the hosted URL is a real\norigin of its own \u2014 absolute asset paths and client-side routing work, the\nwhole app is navigable \u2014 so serve normally and let the platform address it;\nnever rewrite your app to a path prefix.\n\n## Tools in this environment\n\n- `agentc` \u2014 Agent Compose CLI (your primary interface; authed from env)\n- `@agent-compose/sdk` \u2014 installed in /workspace for writing workflows\n- `/ac:*` Claude Code skills \u2014 slash commands for the above\n- `archil` (factory drive), `rtk`, `bun`\n- `xdotool` / `scrot` \u2014 drive + screenshot the desktop (if this machine has one; see Computer Use)\n- A world-writable `/workspace` working directory";
|
|
23
|
+
/** Parameters for the `agentc session add` education brief (ADR-0055 §8). */
|
|
24
|
+
export interface AddedSessionBriefParams {
|
|
25
|
+
conversationId: string;
|
|
26
|
+
serverUrl: string;
|
|
27
|
+
dashboardUrl: string | null;
|
|
28
|
+
}
|
|
29
|
+
/**
|
|
30
|
+
* The education brief `agentc session add` writes into a connected LOCAL
|
|
31
|
+
* session's CLAUDE.md/AGENTS.md — the same content family as
|
|
32
|
+
* `AGENT_COMPOSE_MANUAL`, rendered for the local context (one source,
|
|
33
|
+
* rendered per context). The manual above must stay BYTE-IDENTICAL (the
|
|
34
|
+
* server and base-env bake static copies of it); this function renders a
|
|
35
|
+
* sibling document and never touches it. The local context differs from the
|
|
36
|
+
* sandbox in exactly the ways stated here: auth rides the bridge credential
|
|
37
|
+
* fallback instead of injected run env; no factory drive is mounted — the
|
|
38
|
+
* files on this machine belong to the human; and the bound conversation is
|
|
39
|
+
* MIRROR-ONLY (ADR-0055 §8.6) — teammates read along but can never message
|
|
40
|
+
* the session through it, so the brief must not promise an inbound channel.
|
|
41
|
+
*/
|
|
42
|
+
export declare function buildAddedSessionBrief(p: AddedSessionBriefParams): string;
|
|
23
43
|
/**
|
|
24
44
|
* One connector this run can reach, as the agent should see it. Strictly
|
|
25
45
|
* NON-SECRET — hosts, methods, paths, identity only. The access token is
|
|
@@ -11,6 +11,20 @@ import { RequestContext } from "../request-context/request-context.js";
|
|
|
11
11
|
import { type SteerPayload } from "./steer-control.js";
|
|
12
12
|
export declare const DEFAULT_CLAUDE_MODEL = "claude-fable-5";
|
|
13
13
|
export declare function parseAgentStatus(text: string): AgentStatus | null;
|
|
14
|
+
/**
|
|
15
|
+
* A `responseSchema` rendered for the runtime's structured-output surface.
|
|
16
|
+
*
|
|
17
|
+
* zod v4's `toJSONSchema` stamps `$schema: "…/draft/2020-12/schema"` on the
|
|
18
|
+
* result. Claude Code validates `--json-schema` with a validator that has no
|
|
19
|
+
* 2020-12 meta-schema registered, so any schema CARRYING that header is
|
|
20
|
+
* rejected at CLI startup — the process exits 1 before its first API call
|
|
21
|
+
* and every agent with a responseSchema dies on spawn (observed live on
|
|
22
|
+
* claude 2.1.212: `--json-schema is not a valid JSON Schema: no schema with
|
|
23
|
+
* key or ref "https://json-schema.org/draft/2020-12/schema"`). The header is
|
|
24
|
+
* pure metadata — drop it; the schema body is draft-07-compatible for every
|
|
25
|
+
* shape zod emits from our workflow schemas.
|
|
26
|
+
*/
|
|
27
|
+
export declare function runtimeJsonSchema(schema: z.ZodType<unknown>): Record<string, unknown>;
|
|
14
28
|
export interface AgentLoopResult<TResponse = unknown> {
|
|
15
29
|
agentId: string;
|
|
16
30
|
label: string;
|
|
@@ -62,7 +76,15 @@ export type AgentMessageSummary = {
|
|
|
62
76
|
status: "pending" | "in_progress" | "completed";
|
|
63
77
|
}[];
|
|
64
78
|
};
|
|
65
|
-
|
|
79
|
+
/** Everything but the live-only streaming chunk: `text_delta` never becomes
|
|
80
|
+
* an agent.message event (the terminating `text` carries the whole block) —
|
|
81
|
+
* the loop filters it before summarizing. */
|
|
82
|
+
type DurableAgentMessage = Exclude<AgentMessage, {
|
|
83
|
+
type: "text_delta";
|
|
84
|
+
} | {
|
|
85
|
+
type: "usage_delta";
|
|
86
|
+
}>;
|
|
87
|
+
export declare function summarizeAgentMessage(msg: DurableAgentMessage): AgentMessageSummary;
|
|
66
88
|
export type AgentLifecycleEvent = {
|
|
67
89
|
event: "agent.spawned";
|
|
68
90
|
at: number;
|
|
@@ -152,3 +174,4 @@ export interface AgentLoopOpts<TResponse = unknown> {
|
|
|
152
174
|
mode?: "auto" | "hitl";
|
|
153
175
|
}
|
|
154
176
|
export declare function agentLoop<TResponse = unknown>(opts: AgentLoopOpts<TResponse>): Promise<AgentLoopResult<TResponse>>;
|
|
177
|
+
export {};
|