@agent-compose/sdk 0.6.0 → 0.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (126) hide show
  1. package/README.md +66 -39
  2. package/dist/agent/__tests__/runtime-json-schema.test.d.ts +10 -0
  3. package/dist/agent/agent-context.d.ts +21 -1
  4. package/dist/agent/agent-loop.d.ts +24 -1
  5. package/dist/client.d.ts +338 -534
  6. package/dist/directives.d.ts +112 -0
  7. package/dist/display.d.ts +242 -0
  8. package/dist/errors.d.ts +24 -1
  9. package/dist/index.d.ts +34 -13
  10. package/dist/index.js +2984 -861
  11. package/dist/pause/wrappers.d.ts +31 -9
  12. package/dist/processors/ask-human.d.ts +30 -0
  13. package/dist/processors/ask-human.test.d.ts +1 -0
  14. package/dist/processors/index.d.ts +1 -0
  15. package/dist/runtimes/_acp-client.d.ts +46 -1
  16. package/dist/runtimes/_cli-agent.d.ts +58 -4
  17. package/dist/runtimes/_jsonl-guard.d.ts +103 -0
  18. package/dist/runtimes/amp.d.ts +2 -2
  19. package/dist/runtimes/claude-code.d.ts +59 -0
  20. package/dist/runtimes/claude-code.test.d.ts +14 -0
  21. package/dist/runtimes/claude.d.ts +16 -0
  22. package/dist/runtimes/claude.test.d.ts +8 -0
  23. package/dist/runtimes/codex.d.ts +9 -3
  24. package/dist/runtimes/cursor.d.ts +9 -0
  25. package/dist/runtimes/droid.d.ts +9 -0
  26. package/dist/runtimes/jsonl-guard.test.d.ts +19 -0
  27. package/dist/runtimes/openai-desktop.js +2922 -861
  28. package/dist/runtimes/opencode.d.ts +25 -0
  29. package/dist/runtimes/vercel.js +22 -1
  30. package/dist/sandbox/devbox.d.ts +42 -0
  31. package/dist/sandbox/exec-stream.d.ts +14 -0
  32. package/dist/sandbox/network-policy.d.ts +100 -0
  33. package/dist/sandbox/provider-def.d.ts +79 -0
  34. package/dist/sandbox/providers/desktop.d.ts +10 -0
  35. package/dist/sandbox/providers/e2b.d.ts +17 -0
  36. package/dist/sandbox/providers/local.d.ts +11 -0
  37. package/dist/sandbox/providers/vercel.d.ts +18 -0
  38. package/dist/sandbox/registry.d.ts +45 -0
  39. package/dist/sandbox/sizes.d.ts +68 -0
  40. package/dist/sandbox.d.ts +24 -299
  41. package/dist/step-invocation/__tests__/foreground-recovery.test.d.ts +1 -0
  42. package/dist/step-invocation/invoker.d.ts +24 -1
  43. package/dist/step-invocation/protocol.d.ts +13 -0
  44. package/dist/types/api-compliance.d.ts +71 -0
  45. package/dist/types/api-conversations.d.ts +492 -0
  46. package/dist/types/api-factory.d.ts +309 -0
  47. package/dist/types/api-projects.d.ts +131 -0
  48. package/dist/types/api-runs.d.ts +377 -0
  49. package/dist/types/api-scopes.d.ts +102 -0
  50. package/dist/types/conversation-stream.d.ts +191 -0
  51. package/dist/types/execution-context.d.ts +12 -2
  52. package/dist/types/protocol.d.ts +30 -1
  53. package/dist/types/sandbox-environment.d.ts +8 -5
  54. package/dist/types/sandbox.d.ts +79 -0
  55. package/dist/types/workflow-metadata.d.ts +33 -8
  56. package/dist/types/workflow-plan.d.ts +10 -0
  57. package/dist/types/workflow.d.ts +18 -193
  58. package/dist/utils/bundler.d.ts +12 -1
  59. package/dist/utils/errors.d.ts +9 -1
  60. package/dist/workflow-steps/index.d.ts +1 -1
  61. package/dist/workflow-steps/observability.d.ts +8 -1
  62. package/dist/workflow-steps/runner.d.ts +3 -3
  63. package/dist/workflow-steps/step.d.ts +15 -1
  64. package/dist/workflow-steps/types.d.ts +19 -5
  65. package/dist/workflow-steps/workflow.d.ts +22 -1
  66. package/dist/workflows/engine.d.ts +3 -2
  67. package/dist/workflows/invoke-child.d.ts +2 -2
  68. package/package.json +1 -1
  69. package/src/agent/agent-context.ts +206 -16
  70. package/src/agent/agent-loop.ts +40 -4
  71. package/src/agent/run-agent.ts +9 -1
  72. package/src/client.ts +909 -621
  73. package/src/directives.ts +184 -0
  74. package/src/display.ts +788 -0
  75. package/src/errors.ts +39 -0
  76. package/src/index.ts +117 -10
  77. package/src/pause/wrappers.ts +44 -9
  78. package/src/processors/ask-human.ts +136 -0
  79. package/src/processors/index.ts +5 -0
  80. package/src/runtimes/_acp-client.ts +72 -3
  81. package/src/runtimes/_cli-agent.ts +171 -38
  82. package/src/runtimes/_jsonl-guard.ts +219 -0
  83. package/src/runtimes/claude-code.ts +246 -0
  84. package/src/runtimes/claude.ts +32 -2
  85. package/src/runtimes/codex.ts +55 -3
  86. package/src/runtimes/cursor.ts +59 -0
  87. package/src/runtimes/droid.ts +63 -0
  88. package/src/runtimes/openai-desktop.ts +59 -14
  89. package/src/runtimes/opencode.ts +61 -0
  90. package/src/sandbox/devbox.ts +48 -0
  91. package/src/sandbox/exec-stream.ts +48 -0
  92. package/src/sandbox/network-policy.ts +181 -0
  93. package/src/sandbox/provider-def.ts +94 -0
  94. package/src/sandbox/providers/desktop.ts +57 -0
  95. package/src/sandbox/providers/e2b.ts +354 -0
  96. package/src/sandbox/providers/local.ts +106 -0
  97. package/src/sandbox/providers/vercel.ts +331 -0
  98. package/src/sandbox/registry.ts +198 -0
  99. package/src/sandbox/sizes.ts +95 -0
  100. package/src/sandbox.ts +59 -1263
  101. package/src/step-invocation/invoker.ts +319 -34
  102. package/src/step-invocation/protocol.ts +19 -0
  103. package/src/types/api-compliance.ts +79 -0
  104. package/src/types/api-conversations.ts +522 -0
  105. package/src/types/api-factory.ts +336 -0
  106. package/src/types/api-projects.ts +140 -0
  107. package/src/types/api-runs.ts +412 -0
  108. package/src/types/api-scopes.ts +102 -0
  109. package/src/types/conversation-stream.ts +231 -0
  110. package/src/types/execution-context.ts +10 -2
  111. package/src/types/protocol.ts +33 -0
  112. package/src/types/sandbox-environment.ts +28 -9
  113. package/src/types/sandbox.ts +78 -0
  114. package/src/types/workflow-metadata.ts +35 -8
  115. package/src/types/workflow-plan.ts +11 -0
  116. package/src/types/workflow.ts +25 -280
  117. package/src/utils/bundler.ts +32 -5
  118. package/src/utils/errors.ts +34 -2
  119. package/src/workflow-steps/index.ts +1 -0
  120. package/src/workflow-steps/observability.ts +19 -8
  121. package/src/workflow-steps/runner.ts +4 -4
  122. package/src/workflow-steps/step.ts +49 -1
  123. package/src/workflow-steps/types.ts +20 -5
  124. package/src/workflow-steps/workflow.ts +22 -1
  125. package/src/workflows/engine.ts +3 -2
  126. package/src/workflows/invoke-child.ts +2 -2
package/README.md CHANGED
@@ -23,34 +23,44 @@ npm install zod
23
23
 
24
24
  ## Authoring a workflow
25
25
 
26
- A workflow is `async (ctx, sandbox) => T`. Two positional args:
27
-
28
- - **`ctx`** carries the run identity (`run.id`), the caller's `input`, plus
29
- observability helpers (`setMetadata`, `step`).
30
- - **`sandbox`** is a capability the engine constructs once for the run —
31
- pass it to `agent({ sandbox, ... })` and to any helper that takes a
32
- `SandboxProvider` (file writers, git utilities, command runners).
26
+ A workflow is a typed chain of discrete steps:
27
+ `defineWorkflow({ id, input, output }).step(defineStep(...)).build()`.
28
+ Every step is a durable replay checkpoint — the engine records each
29
+ step's validated output, so a retry or resume picks up after the last
30
+ completed step instead of re-running the whole workflow. Pausing for
31
+ human input (`ctx.pause`, `ctx.waitForEvent`) only works in this
32
+ step-form.
33
33
 
34
34
  ```typescript
35
35
  // my-workflow.ts
36
- import { defineWorkflow, agent, claudeRuntime } from "@agent-compose/sdk";
36
+ import { defineWorkflow, defineStep, agent, claudeRuntime } from "@agent-compose/sdk";
37
+ import { z } from "zod";
37
38
  import PROMPT from "./prompt.md" with { type: "text" };
38
39
 
39
- export default defineWorkflow({
40
- async run(ctx, sandbox) {
41
- const repo = (ctx.input?.repo as string | undefined) ?? "owner/repo";
40
+ const Input = z.object({ repo: z.string() });
41
+ const Output = z.object({ ok: z.boolean(), summary: z.string().optional() });
42
42
 
43
+ const review = defineStep({
44
+ name: "review",
45
+ input: Input,
46
+ output: Output,
47
+ run: async (ctx) => {
43
48
  const result = await agent({
44
- sandbox,
45
- runtime: claudeRuntime,
46
- prompt: `${PROMPT}\n\nRepository: ${repo}`,
47
- tools: ["Bash", "Read", "Edit", "Write", "Grep", "Glob"],
48
- budget: { turnsPerIteration: 40, maxIterations: 8 },
49
+ sandbox: ctx.sandbox, // the run's own VM
50
+ runtime: claudeRuntime,
51
+ prompt: `${PROMPT}\n\nRepository: ${ctx.input.repo}`,
52
+ tools: ["Bash", "Read", "Edit", "Write", "Grep", "Glob"],
53
+ budget: { turnsPerIteration: 40, maxIterations: 8 },
49
54
  });
50
-
51
55
  await ctx.setMetadata({ summary: result.status?.summary });
52
- return { ok: result.status?.completed ?? false };
56
+ return { ok: result.status?.completed ?? false, summary: result.status?.summary };
53
57
  },
58
+ });
59
+
60
+ export default defineWorkflow({
61
+ id: "my-workflow",
62
+ input: Input,
63
+ output: Output,
54
64
  // Optional: outbound network rules that the runner sandbox will enforce
55
65
  // (Vercel only — E2B ignores). Use `$VAR` placeholders for secrets that
56
66
  // get resolved from the per-workflow secret store at dispatch time.
@@ -60,36 +70,52 @@ export default defineWorkflow({
60
70
  "api.anthropic.com": [{ transform: [{ headers: { "x-api-key": "$ANTHROPIC_API_KEY" } }] }],
61
71
  },
62
72
  },
63
- });
73
+ })
74
+ .step(review)
75
+ .build();
64
76
  ```
65
77
 
66
- `defineWorkflow` is a thin sugar — it returns the bare `run` function with
67
- `networkPolicy` / `placeholders` / `snapshots`
68
- attached as metadata that the bundler picks up at registration time. A
69
- plain `export default async (ctx, sandbox) => {...}` is also valid; you
70
- just lose the metadata channel.
78
+ The builder chain enforces at compile time that each step's `input`
79
+ schema matches the previous step's `output`, and `.build()` verifies
80
+ the final step's `output` is the same Zod instance as the workflow's
81
+ declared `output`. `networkPolicy` / `placeholders` / `snapshots` on
82
+ the definition become metadata the bundler picks up at registration
83
+ time.
84
+
85
+ > **Legacy run-form.** `defineWorkflow({ run(ctx, sandbox) { … } })` is
86
+ > `@deprecated` — it compiles to a single opaque step, so there is no
87
+ > per-step replay and pause doesn't work. Don't author new ones.
88
+
89
+ ### What a step can do with `ctx`
71
90
 
72
- ### What the workflow can do with `ctx`
91
+ Each step's `run(ctx)` receives a `StepContext`:
73
92
 
74
93
  ```ts
75
- interface WorkflowCtx {
94
+ interface StepContext<TInput> {
95
+ input: TInput; // validated against the step's `input` schema
76
96
  run: { id: string };
77
- input?: Record<string, unknown>;
97
+ sandbox?: SandboxProvider; // the run's VM — pass to `agent({ sandbox: ctx.sandbox })`
78
98
  setMetadata: (data: Record<string, unknown>) => Promise<void>;
79
- step<T>(name: string, fn: () => Promise<T>): Promise<T>;
99
+ step<T>(name: string, fn: () => Promise<T>): Promise<T>; // durable named sub-step
100
+ pause<T>(req: PauseRequest<T>): Promise<T>; // wait for a human, durably
101
+ sleep(durationMs: number): Promise<void>;
102
+ waitForEvent<T>(req: WaitForEventRequest<T>): Promise<T>;
103
+ invokeChild(name, input?, opts?): Promise<RunStatus>; // run another workflow
80
104
  }
81
105
  ```
82
106
 
83
- `step("phase-name", () => …)` wraps a phase for the run timeline — emits
84
- `step_started` / `step_completed` / `step_failed` lifecycle events with
85
- duration. Use it for setup, external API calls, or anything you want
86
- visible on the dashboard's run detail page.
107
+ `ctx.step("phase-name", () => …)` wraps a phase *within* a step for the
108
+ run timeline — it memoises the result (a pause-resume re-entry restores
109
+ it instead of re-running) and emits `workflow_substep_started` /
110
+ `workflow_substep_completed` / `workflow_substep_failed` lifecycle
111
+ events with duration. Use it for setup, external API calls, or anything
112
+ you want visible on the dashboard's run detail page.
87
113
 
88
114
  ### The `agent` loop
89
115
 
90
116
  ```ts
91
117
  agent({
92
- sandbox, // the workflow's sandbox arg
118
+ sandbox, // ctx.sandbox — the run's VM
93
119
  runtime, // claudeRuntime, or your own via createClaudeRuntime / defineRuntime
94
120
  prompt, // raw markdown — `--- frontmatter ---` is auto-stripped
95
121
  tools?, // model tool allowlist (defaults inside agentLoop)
@@ -211,8 +237,8 @@ console.log(status.status); // "success" | "failed" | "abandoned" | "canceled"
211
237
  console.log(status.output); // workflow's return value
212
238
  ```
213
239
 
214
- `output` is the workflow's `run()` return value (whatever `defineWorkflow({
215
- async run() { return … } })` resolves to). `setMetadata()` writes to a
240
+ `output` is the final step's return value, validated against the
241
+ workflow's declared `output` schema. `setMetadata()` writes to a
216
242
  separate `metadata` field — useful for "side-channel" facts (PR url, plan
217
243
  url) without polluting the structured return.
218
244
 
@@ -364,10 +390,10 @@ await client.invoke("my-workflow", input, {
364
390
  });
365
391
 
366
392
  // Set a default at registration time:
367
- defineWorkflow({ run, snapshots: { saveLatest: true } });
393
+ defineWorkflow({ id, input, output, snapshots: { saveLatest: true } }).step(…).build();
368
394
 
369
395
  // Retain every step's snapshot (not just the latest):
370
- defineWorkflow({ run, snapshots: { saveLatest: true, retainSteps: true } });
396
+ defineWorkflow({ id, input, output, snapshots: { saveLatest: true, retainSteps: true } }).step(…).build();
371
397
 
372
398
  // Browse / clean up:
373
399
  const page = await client.listSnapshotsPage({ workflow: "my-workflow", limit: 50 });
@@ -408,7 +434,8 @@ programmatic / server-to-server callers.
408
434
 
409
435
  | Export | What |
410
436
  |---|---|
411
- | `defineWorkflow` | Attach metadata to a workflow `run` function |
437
+ | `defineWorkflow` | Typed step-workflow builder — `defineWorkflow({ id, input, output }).step(...).build()` |
438
+ | `defineStep` | Declare one typed workflow step (`{ name, input, output, run }`) |
412
439
  | `defineRuntime` | Wrap an agent execution provider as an `AgentRuntime` |
413
440
  | `defineSandboxEnvironment` | Sugar for declaring a workflow whose primary purpose is to build a snapshot for others to boot from |
414
441
  | `agent` / `agentLoop` | Embed an LLM loop inside a workflow |
@@ -421,7 +448,7 @@ programmatic / server-to-server callers.
421
448
  | `parseSseStream` | Generic SSE chunk decoder (used by `streamRunLogs`) |
422
449
  | `createSandbox` / `reconnectSandbox` / `killAllSandboxes` / `killSandboxById` / `getSandboxQuotas` / `listOwnedSandboxes` / `deleteSandboxSnapshot` | Sandbox-provider helpers (Vercel + E2B) |
423
450
 
424
- Type exports: `WorkflowFn`, `WorkflowCtx`, `WorkflowDefinition`,
451
+ Type exports: `Workflow`, `Step`, `StepContext`, `WorkflowBuilder`,
425
452
  `WorkflowHooks`, `AgentBudget`, `AgentRuntime`, `RuntimeOptions`,
426
453
  `ModelExecutionContract`, `McpServerConfig`, `AgentMessage` (and its
427
454
  variants), `AgentStatus`, `RunStatus`, `RegisterResult`, `RunEvent`,
@@ -0,0 +1,10 @@
1
+ /**
2
+ * responseSchema → runtime outputFormat must NOT carry zod v4's
3
+ * `$schema: …/draft/2020-12/schema` header. Claude Code validates
4
+ * `--json-schema` with a validator that has no 2020-12 meta-schema
5
+ * registered, so a schema carrying the header kills the CLI at startup
6
+ * ("Claude Code process exited with code 1" before any API call) — the
7
+ * incident that took down every dynamic-task run after the 2026-07-17
8
+ * agent-env rebake baked claude 2.1.212.
9
+ */
10
+ export {};
@@ -19,7 +19,27 @@ import type { SandboxProvider } from "../types/sandbox.js";
19
19
  * that credentials are network-injected (never in the env). The live
20
20
  * "Connectors & access" section is appended per-run by `buildAgentContextDoc`.
21
21
  */
22
- export declare const AGENT_COMPOSE_MANUAL = "# Working inside an Agent Compose sandbox\n\nYou are an agent running in a per-run sandbox on the Agent Compose platform.\nUse the **`agentc` CLI** and the **`@agent-compose/sdk`** for everything below \u2014\ndo NOT hand-roll raw HTTP/curl calls against the platform API. The CLI is on\nyour PATH and already authenticated from the environment\n(`AGENT_COMPOSE_URL` / `AGENT_COMPOSE_API_KEY` / `AGENT_COMPOSE_FACTORY` are\ninjected for this run), so commands just work \u2014 no login, no keys to manage.\n\nThe `/ac:*` skills are installed as Claude Code slash commands (`/ac:invoke`,\n`/ac:events`, `/ac:logs`, `/ac:register`, \u2026) \u2014 reach for them too.\n\n## Files \u2014 your outputs persist by default\n\nYour working directory defaults to **`$AGENT_COMPOSE_RUN_DIR`** \u2014 a per-run\ndirectory on the shared factory drive\n(`$AGENT_COMPOSE_FACTORY_DIR/<workflow>/<version>/<run-id>/`) the platform\ncreates and attributes to this run. **Files you write here persist by\ndefault** \u2014 they show up in the dashboard's Files tab and the run's Artifacts\ncard, with no API calls to save them. The dir already exists and is writable.\n\nNeed throwaway scratch \u2014 heavy build output, package caches, temp files?\n`cd /tmp` (or any path outside `/factory`): anything off the factory drive is\nephemeral and discarded when the sandbox ends. In short: **stay in your working\ndir to keep something, `cd` out to throw it away.**\n\nThe whole shared drive is POSIX-mounted at `/factory`; the dashboard-visible\nroot is `$AGENT_COMPOSE_FACTORY_DIR` (`/factory/files`). Earlier versions and\nruns live in sibling dirs under\n`$AGENT_COMPOSE_FACTORY_DIR/$AGENT_COMPOSE_WORKFLOW/` \u2014 read them for prior\ncontext. Other workflows' dirs are present but not your concern.\n\n## Events \u2014 the factory timeline\n\nRecord something on the run/factory timeline (the dashboard renders these)\nwith the CLI \u2014 your run id is `$RUN_ID`:\n\n agentc events send \"$RUN_ID\" <name> --summary \"<one line>\" [--body '<json>']\n\nNames like `note.created` / `brief.posted` surface in the Workbench;\n`agentc events list` reads them back. `/ac:events` is the skill equivalent.\n\n## Runs\n\n agentc list # registered workflows (/ac:list)\n agentc logs \"$RUN_ID\" # a run's logs (/ac:logs)\n agentc invoke <workflow> -i '<json>' # dispatch a workflow (/ac:invoke)\n\n## Writing workflow / agent code \u2014 the SDK\n\n`@agent-compose/sdk` is installed in `/workspace` \u2014 import it from any script\nyou write there:\n\n import { defineWorkflow, agent, AgentComposeClient } from \"@agent-compose/sdk\";\n\nUse `/ac:generate-workflow` / `/ac:generate-agent` to scaffold, then\n`agentc register <file.ts>` (or `/ac:register`).\n\n## Pausing to ask the human \u2014 `agentc pause`\n\nWhen you can't or shouldn't proceed without a human, run `agentc pause`, then\n**END YOUR TURN**:\n\n agentc pause --reason \"Notion returned 401 \u2014 connect Notion to continue\" \\\n --option retry --option skip\n\n`agentc pause` does NOT block and does NOT print the answer. It records your\nquestion and returns immediately. The moment you end your turn, the run pauses\n(your sandbox is snapshotted and compute stops while the human decides) and the\nhuman's answer is delivered to you as your **next message** \u2014 you pick up\nexactly where you left off, with the answer in hand. So: ask, end your turn,\nand wait. Do NOT keep working, do NOT call more tools, and do NOT mark the task\ncomplete after pausing.\n\nReach for it the moment you hit \u2014 or foresee \u2014 any of these:\n- **A wall only a human can clear:** a 401/403, a missing credential, an\n unconnected provider, a host the network refuses. Do NOT retry blindly or try\n to work around it \u2014 pause and say what needs enabling.\n- **A durable or outward-facing action that needs sign-off:** registering a\n workflow, deploying, sending email/messages, deleting or overwriting shared\n data, spending money. Prepare everything, then pause for approval BEFORE you\n commit it.\n- **A judgment call only the human can settle:** an under-specified request,\n several valid paths, a conflict with existing state, missing input only they have.\n\nYou compose the `--reason` (the ask) yourself; pass `--option` choices when\nthere are clear ones, omit them for a free-form answer. Each agent pauses\nindependently \u2014 pausing doesn't stop the others.\n\n## Credentials\n\nConnector credentials (Google, GitHub, \u2026) are NEVER in your environment.\nThey're injected at the network layer when you call an allowed host \u2014 make the\nrequest **without** an Authorization header and the platform adds it. Don't try\nto read or exfiltrate tokens; they aren't here. The \"Connectors & access\"\nsection below (when present) lists exactly which providers this run can reach.\n\n## Tools in this environment\n\n- `agentc` \u2014 Agent Compose CLI (your primary interface; authed from env)\n- `@agent-compose/sdk` \u2014 installed in /workspace for writing workflows\n- `/ac:*` Claude Code skills \u2014 slash commands for the above\n- `archil` (factory drive), `rtk`, `bun`\n- A world-writable `/workspace` working directory";
22
+ export declare const AGENT_COMPOSE_MANUAL = "# Working inside an Agent Compose sandbox\n\nYou are an agent running in a per-run sandbox on the Agent Compose platform.\nUse the **`agentc` CLI** and the **`@agent-compose/sdk`** for everything below \u2014\ndo NOT hand-roll raw HTTP/curl calls against the platform API. The CLI is on\nyour PATH and already authenticated from the environment\n(`AGENT_COMPOSE_URL` / `AGENT_COMPOSE_API_KEY` / `AGENT_COMPOSE_FACTORY` are\ninjected for this run), so commands just work \u2014 no login, no keys to manage.\n\nThe `/ac:*` skills are installed as Claude Code slash commands (`/ac:invoke`,\n`/ac:events`, `/ac:logs`, `/ac:register`, \u2026) \u2014 reach for them too.\n\n## Files \u2014 your outputs persist by default\n\nYour working directory defaults to **`$AGENT_COMPOSE_RUN_DIR`** \u2014 a per-run\ndirectory on the shared factory drive\n(`$AGENT_COMPOSE_FACTORY_DIR/<workflow>/<version>/<run-id>/`) the platform\ncreates and attributes to this run. **Files you write here persist by\ndefault** \u2014 they show up in the dashboard's Files tab and the run's Artifacts\ncard, with no API calls to save them. The dir already exists and is writable.\n\nNeed throwaway scratch \u2014 heavy build output, package caches, temp files?\n`cd /tmp` (or any path outside `/factory`): anything off the factory drive is\nephemeral and discarded when the sandbox ends. In short: **stay in your working\ndir to keep something, `cd` out to throw it away.**\n\nThe whole shared drive is POSIX-mounted at `/factory`; the dashboard-visible\nroot is `$AGENT_COMPOSE_FACTORY_DIR` (`/factory/files`). Earlier versions and\nruns live in sibling dirs under\n`$AGENT_COMPOSE_FACTORY_DIR/$AGENT_COMPOSE_WORKFLOW/` \u2014 read them for prior\ncontext. Other workflows' dirs are present but not your concern.\n\n## Events \u2014 the factory timeline\n\nRecord something on the run/factory timeline (the dashboard renders these)\nwith the CLI \u2014 your run id is `$RUN_ID`:\n\n agentc events send \"$RUN_ID\" <name> --summary \"<one line>\" [--body '<json>']\n\nNames like `note.created` / `brief.posted` surface in the Workbench;\n`agentc events list` reads them back. `/ac:events` is the skill equivalent.\n\n## Runs\n\n agentc list # registered workflows (/ac:list)\n agentc logs \"$RUN_ID\" # a run's logs (/ac:logs)\n agentc invoke <workflow> -i '<json>' # dispatch a workflow (/ac:invoke)\n\n## Writing workflow / agent code \u2014 the SDK\n\n`@agent-compose/sdk` is installed in `/workspace`. **To author a workflow,\nALWAYS run `/ac:generate-workflow`** (and `/ac:generate-agent` for an agent\nstep) instead of writing source from memory \u2014 the skill scaffolds the correct,\ncurrent shape. Then `agentc register <file.ts>` (or `/ac:register`).\n\nThe skill writes **step-form** (a builder of discrete, durable `.step()`s).\nThe legacy run-form (`defineWorkflow({ run(ctx, sandbox) { \u2026 } })`) has been\nREMOVED from the SDK \u2014 registering one fails with an error. Step-form is the\nonly shape: durable per-step replay, and pause only works there.\n\n## Pausing to ask the human\n\nTo ask a human and get an answer back, use the **`AskUserQuestion`** tool if\nyou have it; otherwise run **`agentc pause`**:\n\n agentc pause --reason \"Notion returned 401 \u2014 connect Notion to continue\" \\\n --option retry --option skip\n\n**Both BLOCK and hand you the answer inline.** While you wait, the run is\nsuspended \u2014 your sandbox is frozen and compute stops, so a pause is free while\nthe human decides. When they answer, the call RETURNS with their decision: the\n`AskUserQuestion` tool result, or `agentc pause`'s output\n(`\u25B6 Resumed. The human answered: \u2026`), carries it.\n\n**Then USE that answer to finish your work \u2014 do NOT end your turn.** This is NOT\nfire-and-forget, and the answer does NOT arrive in a later message: it comes\nback right where you called it, on the SAME turn. The shape is: ask \u2192 the call\nblocks \u2192 it returns the human's answer \u2192 you act on it and produce your result.\nNever end your turn before the call returns, never guess an answer, and never\nproceed without one.\n\nReach for it the moment you hit \u2014 or foresee \u2014 any of these:\n- **A wall only a human can clear:** a 401/403, a missing credential, an\n unconnected provider, a host the network refuses. Do NOT retry blindly or try\n to work around it \u2014 pause and say what needs enabling.\n- **A durable or outward-facing action that needs sign-off:** registering a\n workflow, deploying, sending email/messages, deleting or overwriting shared\n data, spending money. Prepare everything, then pause for approval BEFORE you\n commit it.\n- **A judgment call only the human can settle:** an under-specified request,\n several valid paths, a conflict with existing state, missing input only they have.\n\nYou compose the `--reason` (the ask) yourself; pass `--option` choices when\nthere are clear ones, omit them for a free-form answer. Each agent pauses\nindependently \u2014 pausing doesn't stop the others.\n\n## Credentials\n\nConnector credentials (Google, GitHub, \u2026) are NEVER in your environment.\nThey're injected at the network layer when you call an allowed host \u2014 make the\nrequest **without** an Authorization header and the platform adds it. Don't try\nto read or exfiltrate tokens; they aren't here. The \"Connectors & access\"\nsection below (when present) lists exactly which providers this run can reach.\n\n## Computer Use \u2014 you have a real desktop, and it is already running\n\n**This machine has a graphical desktop.** Every session machine does \u2014 terminal\nsessions included \u2014 and the platform brings it UP AT BOOT, before your first\nturn: an X server on `DISPLAY=:0`, the openbox window manager, wallpaper and a\npanel. You do not start it, you do not wait for a human to open it, and you do\nnot need a viewer. Go straight to driving it.\n\n(The one exception, and it is rare: an image built without the GUI stack has no\ndisplay at all, and `DISPLAY=:0 xdotool getdisplaygeometry` errors outright.\nThat single case is the only one where this section does not apply \u2014 a\nscreenshot showing only wallpaper is NOT it, and neither is an app that failed\nto start.)\n\n**This is how you SEE anything.** Any question of the form \"does it render?\",\n\"is the page actually working?\", \"did the markers show up?\", \"what does it look\nlike?\" is answered by opening it on this desktop and screenshotting it \u2014 not by\nreasoning about the code, and not by a headless render (which proves the process\nstarts, not that the thing draws). Verify visually before you report visually.\n\n- **Input** \u2014 `xdotool` against `DISPLAY=:0`: `DISPLAY=:0 xdotool mousemove <x> <y>`,\n `DISPLAY=:0 xdotool click 1` (1=left, 3=right), `DISPLAY=:0 xdotool type 'text'`,\n `DISPLAY=:0 xdotool key Return` (also `ctrl+c`, `Tab`, `super`, \u2026).\n- **Screenshots** \u2014 `scrot` (or ImageMagick's `import`):\n `DISPLAY=:0 scrot /tmp/screen.png`, then READ the PNG to see the screen,\n before and after you act. A screenshot is your only eyes here.\n- **Apps + windows** \u2014 a plain X session. Launch in the background:\n `DISPLAY=:0 <app> &`. Two things that trip agents up, both normal:\n - a GUI app needs a **beat to map its window** \u2014 screenshot, and if you see\n only wallpaper, wait a couple of seconds and screenshot again before\n concluding anything;\n - **Chromium needs `--no-sandbox`** in this environment (nested sandbox).\n The whole recipe for looking at a local page:\n `DISPLAY=:0 chromium --no-sandbox --disable-gpu --start-maximized <url> &`\n then `sleep 5`, then `DISPLAY=:0 scrot /tmp/screen.png` and read it.\n If a window still never appears, read the app's own log (`/tmp/*.log`) \u2014 the\n desktop is not the thing that failed. Do NOT abandon it for a headless\n screenshot: headless cannot tell you what the human will see.\n- **A human can watch** \u2014 the session header carries a **Desktop** button in the\n dashboard, and what a teammate sees there is exactly this display. The desktop\n runs whether or not anyone is looking; never wait for a viewer.\n\nNothing here changes the credentials rule above: tokens are injected at the\nnetwork layer, never present on the desktop or in any file you can read \u2014 so\nthere is nothing to type, paste, or screenshot a credential from.\n\n## Recording a demo \u2014 the desktop, captured to a video the human can play\n\n\"Record a demo of you using X\" is a normal ask, and this machine does it:\nstart a screen recording, drive the app with `xdotool` exactly as in Computer\nUse, stop the recording, and report the file. (For a LIVE view no recording is\nneeded \u2014 the session header's **Desktop** button already streams this display\nto any teammate watching; a recording is the durable, replayable artifact.\nBoth modes exist; say so when it matters.)\n\n**ffmpeg is NOT pre-installed** \u2014 install it first, once per machine:\n\n sudo apt-get update -q && sudo apt-get install -y -q ffmpeg\n\n(drop `sudo` if you are already root). Then the whole recipe:\n\n DISPLAY=:0 ffmpeg -f x11grab \\\n -video_size \"$(DISPLAY=:0 xdotool getdisplaygeometry | tr ' ' x)\" \\\n -framerate 10 -i :0 -c:v libvpx -b:v 1M -deadline realtime -cpu-used 8 \\\n demo.webm &\n FFMPEG_PID=$!\n # ... drive the app with xdotool, screenshotting as you go ...\n kill -INT \"$FFMPEG_PID\" && wait \"$FFMPEG_PID\"\n\nThe gotchas, each one earned:\n- **Stop with SIGINT (`kill -INT`), never SIGKILL** \u2014 ffmpeg finalizes the\n file on SIGINT; a hard kill truncates the encode mid-write.\n- **Record WebM (matroska-family), not MP4** \u2014 mp4 writes its moov atom at the\n END, so a killed or crashed encode leaves an UNPLAYABLE file; webm stays\n playable up to the last written frame and plays natively in the browser.\n MP4's only edge is compatibility with some external players \u2014 transcode\n afterwards if you truly need it, never record straight to it.\n- **`-video_size` must match the real screen** \u2014 x11grab does not default to\n it; read the geometry from `xdotool getdisplaygeometry` as above.\n- **10\u201312 fps is right for a screen demo** \u2014 small files, legible UI motion;\n this is not video production.\n- **Write to the drive, not /tmp** \u2014 the recording must land in your working\n directory to persist and show up in Files; a file in /tmp dies with the\n sandbox.\n- When you stop, **TELL the human the exact drive path** of the video \u2014 a\n recording they cannot find might as well not exist.\n\n## Previews \u2014 register every server you serve (cloud sessions)\n\nIn a cloud session, a dev server listening on a port becomes a hosted,\nmember-gated URL the human can open \u2014 but ONLY if you register it:\n\n agentc preview open <port> [--name <label>] [--path </landing>]\n # hosted URL + an \"Open preview\" card\n agentc preview list # the registry \u2014 what is live right now\n agentc preview close <port> # take one down\n\n(`agentc preview announce` is the same verb as `open` \u2014 announce what you\nserve.) `--name` is the human-readable label; `--path` is where the app\nshould open (e.g. `/dashboard`) \u2014 the card and every chip land the human\nthere instead of a bare `/`.\n\nRegister EVERY server you start for a human, the moment it is listening, and\ntell them the URL the command printed. The registry is the only discoverable\nrecord of what this machine serves: an unregistered server keeps running, but\nnobody \u2014 not the human, not the assistant \u2014 can find its URL, and when the\nsandbox recycles it is gone without a trace. Never guess or hand out a raw\nport; the hosted URL from `agentc preview open` is the only address that\nworks outside this machine. (Outside a cloud session the command errors\nhonestly \u2014 there is no session sandbox to expose.)\n\nWhat registration buys you: the human sees each registered preview as a card\nin the conversation and a row in the session's Previews menu \u2014 MANY at once,\none per port \u2014 and the assistant resolves \"open the preview\" from this same\nregistry (its `list_previews` read), so what you register is exactly what\ngets opened. On deployments with subdomain previews the hosted URL is a real\norigin of its own \u2014 absolute asset paths and client-side routing work, the\nwhole app is navigable \u2014 so serve normally and let the platform address it;\nnever rewrite your app to a path prefix.\n\n## Tools in this environment\n\n- `agentc` \u2014 Agent Compose CLI (your primary interface; authed from env)\n- `@agent-compose/sdk` \u2014 installed in /workspace for writing workflows\n- `/ac:*` Claude Code skills \u2014 slash commands for the above\n- `archil` (factory drive), `rtk`, `bun`\n- `xdotool` / `scrot` \u2014 drive + screenshot the desktop (if this machine has one; see Computer Use)\n- A world-writable `/workspace` working directory";
23
+ /** Parameters for the `agentc session add` education brief (ADR-0055 §8). */
24
+ export interface AddedSessionBriefParams {
25
+ conversationId: string;
26
+ serverUrl: string;
27
+ dashboardUrl: string | null;
28
+ }
29
+ /**
30
+ * The education brief `agentc session add` writes into a connected LOCAL
31
+ * session's CLAUDE.md/AGENTS.md — the same content family as
32
+ * `AGENT_COMPOSE_MANUAL`, rendered for the local context (one source,
33
+ * rendered per context). The manual above must stay BYTE-IDENTICAL (the
34
+ * server and base-env bake static copies of it); this function renders a
35
+ * sibling document and never touches it. The local context differs from the
36
+ * sandbox in exactly the ways stated here: auth rides the bridge credential
37
+ * fallback instead of injected run env; no factory drive is mounted — the
38
+ * files on this machine belong to the human; and the bound conversation is
39
+ * MIRROR-ONLY (ADR-0055 §8.6) — teammates read along but can never message
40
+ * the session through it, so the brief must not promise an inbound channel.
41
+ */
42
+ export declare function buildAddedSessionBrief(p: AddedSessionBriefParams): string;
23
43
  /**
24
44
  * One connector this run can reach, as the agent should see it. Strictly
25
45
  * NON-SECRET — hosts, methods, paths, identity only. The access token is
@@ -11,6 +11,20 @@ import { RequestContext } from "../request-context/request-context.js";
11
11
  import { type SteerPayload } from "./steer-control.js";
12
12
  export declare const DEFAULT_CLAUDE_MODEL = "claude-fable-5";
13
13
  export declare function parseAgentStatus(text: string): AgentStatus | null;
14
+ /**
15
+ * A `responseSchema` rendered for the runtime's structured-output surface.
16
+ *
17
+ * zod v4's `toJSONSchema` stamps `$schema: "…/draft/2020-12/schema"` on the
18
+ * result. Claude Code validates `--json-schema` with a validator that has no
19
+ * 2020-12 meta-schema registered, so any schema CARRYING that header is
20
+ * rejected at CLI startup — the process exits 1 before its first API call
21
+ * and every agent with a responseSchema dies on spawn (observed live on
22
+ * claude 2.1.212: `--json-schema is not a valid JSON Schema: no schema with
23
+ * key or ref "https://json-schema.org/draft/2020-12/schema"`). The header is
24
+ * pure metadata — drop it; the schema body is draft-07-compatible for every
25
+ * shape zod emits from our workflow schemas.
26
+ */
27
+ export declare function runtimeJsonSchema(schema: z.ZodType<unknown>): Record<string, unknown>;
14
28
  export interface AgentLoopResult<TResponse = unknown> {
15
29
  agentId: string;
16
30
  label: string;
@@ -62,7 +76,15 @@ export type AgentMessageSummary = {
62
76
  status: "pending" | "in_progress" | "completed";
63
77
  }[];
64
78
  };
65
- export declare function summarizeAgentMessage(msg: AgentMessage): AgentMessageSummary;
79
+ /** Everything but the live-only streaming chunk: `text_delta` never becomes
80
+ * an agent.message event (the terminating `text` carries the whole block) —
81
+ * the loop filters it before summarizing. */
82
+ type DurableAgentMessage = Exclude<AgentMessage, {
83
+ type: "text_delta";
84
+ } | {
85
+ type: "usage_delta";
86
+ }>;
87
+ export declare function summarizeAgentMessage(msg: DurableAgentMessage): AgentMessageSummary;
66
88
  export type AgentLifecycleEvent = {
67
89
  event: "agent.spawned";
68
90
  at: number;
@@ -152,3 +174,4 @@ export interface AgentLoopOpts<TResponse = unknown> {
152
174
  mode?: "auto" | "hitl";
153
175
  }
154
176
  export declare function agentLoop<TResponse = unknown>(opts: AgentLoopOpts<TResponse>): Promise<AgentLoopResult<TResponse>>;
177
+ export {};